@flame0510/project-aether 1.7.0 → 1.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,5 @@
1
+ import SystemPageClient from './PageClient';
2
+
3
+ export default function SystemPage() {
4
+ return <SystemPageClient />;
5
+ }
package/daemon.js CHANGED
@@ -1,7 +1,8 @@
1
1
  #!/usr/bin/env node
2
2
  /**
3
3
  * Rev4a Daemon — SQLite event logger
4
- * Polls OpenClaw sessions every 30s, writes events to SQLite.
4
+ * Polls OpenClaw sessions every 30s, writes events to SQLite, and samples the host
5
+ * machine (CPU, RAM, swap, storage) every 30s on a timer of its own.
5
6
  */
6
7
 
7
8
  'use strict';
@@ -12,7 +13,13 @@ const { spawnSync, execSync } = require('child_process');
12
13
  const Database = require('better-sqlite3');
13
14
  const path = require('path');
14
15
 
15
- const DB_PATH = process.env.REV4A_DB || path.join(os.homedir(), '.config', 'rev4a', 'data', 'events.db');
16
+ // Same rule as lib/rev4a-paths.ts: REV4A_DB, else <data dir>/data/events.db, where the data
17
+ // dir is REV4A_DATA_DIR or ~/.config/rev4a — the API reads the files the daemon writes.
18
+ const DATA_DIR = process.env.REV4A_DATA_DIR
19
+ || (process.platform === 'win32'
20
+ ? path.join(process.env.APPDATA || path.join(os.homedir(), 'AppData', 'Roaming'), 'rev4a')
21
+ : path.join(os.homedir(), '.config', 'rev4a'));
22
+ const DB_PATH = process.env.REV4A_DB || path.join(DATA_DIR, 'data', 'events.db');
16
23
  const POLL_INTERVAL_MS = 30_000;
17
24
 
18
25
  // Cost rates per 1M tokens (separate in/out pricing)
@@ -71,20 +78,6 @@ process.on('uncaughtException', (err) => {
71
78
  // Do NOT exit — the watchdog reads this from the healthcheck
72
79
  });
73
80
 
74
- db.exec(`
75
- CREATE TABLE IF NOT EXISTS system_metrics (
76
- id INTEGER PRIMARY KEY AUTOINCREMENT,
77
- ts INTEGER NOT NULL,
78
- cpu_percent REAL,
79
- ram_used_mb INTEGER,
80
- ram_total_mb INTEGER,
81
- disk_used_gb REAL,
82
- disk_total_gb REAL,
83
- load_avg_1m REAL
84
- );
85
- CREATE INDEX IF NOT EXISTS idx_metrics_ts ON system_metrics(ts);
86
- `);
87
-
88
81
  db.exec(`
89
82
  CREATE TABLE IF NOT EXISTS sessions (
90
83
  session_id TEXT PRIMARY KEY,
@@ -176,210 +169,283 @@ const knownSessions = new Map(); // session_id -> { status, tokens_in, tokens_ou
176
169
  let pollCount = 0;
177
170
 
178
171
  // ── System Metrics ───────────────────────────────────────────────────────────
172
+ //
173
+ // Machine-wide CPU, RAM, swap and storage of the host Rev4a runs on, sampled on their
174
+ // own timer (METRICS_INTERVAL_MS) — independent of the OpenClaw session poll, which has
175
+ // no source on a host without the `openclaw` CLI.
179
176
 
180
- const insertMetric = db.prepare(`
181
- INSERT INTO system_metrics (ts, cpu_percent, ram_used_mb, ram_total_mb, disk_used_gb, disk_total_gb, load_avg_1m)
182
- VALUES (@ts, @cpu_percent, @ram_used_mb, @ram_total_mb, @disk_used_gb, @disk_total_gb, @load_avg_1m)
183
- `);
184
- const pruneMetrics = db.prepare(`DELETE FROM system_metrics WHERE ts < ?`);
185
- const getLastTwoMetrics = db.prepare(`SELECT cpu_percent, ram_used_mb, ram_total_mb FROM system_metrics ORDER BY ts DESC LIMIT 2`);
177
+ /**
178
+ * Machine metrics live in their own database next to events.db, so the event log stays
179
+ * sessions and events only and the history can be dropped without touching them. The
180
+ * daemon is its only writer; GET /api/metrics reads it (lib/metrics-db.ts, same path).
181
+ * Anomalies are still events, in events.db, where the live feed reads them.
182
+ *
183
+ * Opened in a try: a metrics.db that cannot be opened (corrupt, unwritable, disk full)
184
+ * turns sampling off with a log line and leaves the session poll running.
185
+ */
186
+ const METRICS_DB_PATH = path.join(path.dirname(DB_PATH), 'metrics.db');
187
+ const metrics = openMetricsStore();
186
188
 
187
- // Anomaly cooldown: track last anomaly alert timestamps
188
- const lastAnomalyAlert = { cpu: 0, ram: 0 };
189
- const ANOMALY_COOLDOWN_MS = 5 * 60 * 1000; // 5 minutes
189
+ function openMetricsStore() {
190
+ try {
191
+ const mdb = new Database(METRICS_DB_PATH);
192
+ mdb.pragma('journal_mode = WAL');
193
+ mdb.pragma('synchronous = NORMAL');
194
+ mdb.exec(`
195
+ CREATE TABLE IF NOT EXISTS system_metrics (
196
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
197
+ ts INTEGER NOT NULL,
198
+ cpu_percent REAL,
199
+ ram_used_mb INTEGER,
200
+ ram_total_mb INTEGER,
201
+ swap_used_mb INTEGER,
202
+ swap_total_mb INTEGER,
203
+ load_avg_1m REAL,
204
+ load_avg_5m REAL,
205
+ load_avg_15m REAL
206
+ );
207
+ CREATE INDEX IF NOT EXISTS idx_metrics_ts ON system_metrics(ts);
190
208
 
191
- function getCpuPercent() {
192
- // Container-accurate CPU: prefer cgroup v2 usage delta over host cumulative ticks.
193
- const nowWallUsec = Number(process.hrtime.bigint() / 1000n);
194
- const cgroupUsageUsec = readCgroupUsageUsec();
195
-
196
- if (cgroupUsageUsec !== null) {
197
- if (lastCpuSample && lastCpuSample.cgroupUsageUsec !== null) {
198
- const deltaUsageUsec = cgroupUsageUsec - lastCpuSample.cgroupUsageUsec;
199
- const deltaWallUsec = nowWallUsec - lastCpuSample.wallUsec;
200
- if (deltaUsageUsec >= 0 && deltaWallUsec > 0) {
201
- const limitCores = getCpuLimitCores();
202
- const raw = (deltaUsageUsec / deltaWallUsec) / limitCores * 100;
203
- const clamped = Math.max(0, Math.min(100, raw));
204
- lastCpuSample = { wallUsec: nowWallUsec, cgroupUsageUsec, host: null };
205
- return Number(clamped.toFixed(2));
206
- }
207
- }
208
- lastCpuSample = { wallUsec: nowWallUsec, cgroupUsageUsec, host: null };
209
- return 0;
209
+ CREATE TABLE IF NOT EXISTS system_disks (
210
+ ts INTEGER NOT NULL,
211
+ mount TEXT NOT NULL,
212
+ roles TEXT,
213
+ used_mb INTEGER,
214
+ avail_mb INTEGER,
215
+ total_mb INTEGER
216
+ );
217
+ CREATE INDEX IF NOT EXISTS idx_disks_ts ON system_disks(ts);
218
+ CREATE INDEX IF NOT EXISTS idx_disks_mount_ts ON system_disks(mount, ts);
219
+ `);
220
+ return {
221
+ db: mdb,
222
+ insertMetric: mdb.prepare(`
223
+ INSERT INTO system_metrics (ts, cpu_percent, ram_used_mb, ram_total_mb, swap_used_mb, swap_total_mb,
224
+ load_avg_1m, load_avg_5m, load_avg_15m)
225
+ VALUES (@ts, @cpu_percent, @ram_used_mb, @ram_total_mb, @swap_used_mb, @swap_total_mb,
226
+ @load_avg_1m, @load_avg_5m, @load_avg_15m)
227
+ `),
228
+ insertDisk: mdb.prepare(`
229
+ INSERT INTO system_disks (ts, mount, roles, used_mb, avail_mb, total_mb)
230
+ VALUES (@ts, @mount, @roles, @used_mb, @avail_mb, @total_mb)
231
+ `),
232
+ pruneMetrics: mdb.prepare('DELETE FROM system_metrics WHERE ts < ?'),
233
+ pruneDisks: mdb.prepare('DELETE FROM system_disks WHERE ts < ?'),
234
+ getLastTwoMetrics: mdb.prepare('SELECT cpu_percent, ram_used_mb, ram_total_mb FROM system_metrics ORDER BY ts DESC LIMIT 2'),
235
+ };
236
+ } catch (err) {
237
+ log(`[METRICS] disabled: cannot open ${METRICS_DB_PATH}: ${err.message}`);
238
+ return null;
210
239
  }
240
+ }
211
241
 
212
- const host = readHostCpuSample();
213
- if (lastCpuSample && lastCpuSample.host) {
214
- const deltaTotal = host.total - lastCpuSample.host.total;
215
- const deltaIdle = host.idle - lastCpuSample.host.idle;
216
- if (deltaTotal > 0) {
217
- const busy = Math.max(0, deltaTotal - deltaIdle);
218
- const raw = (busy / deltaTotal) * 100;
219
- const clamped = Math.max(0, Math.min(100, raw));
220
- lastCpuSample = { wallUsec: nowWallUsec, cgroupUsageUsec: null, host };
221
- return Number(clamped.toFixed(2));
222
- }
242
+ const METRICS_INTERVAL_MS = 30_000;
243
+ const METRICS_RETENTION_S = 30 * 24 * 3600;
244
+ /** How long to wait before asking Docker for its root directory again after a failure. */
245
+ const DOCKER_ROOT_RETRY_MS = 10 * 60 * 1000;
246
+
247
+ // Anomaly cooldown: last alert per metric (cpu, ram); disks use the crossing state below
248
+ const lastAnomalyAlert = {};
249
+ const ANOMALY_COOLDOWN_MS = 5 * 60 * 1000; // 5 minutes
250
+ const DISK_ANOMALY_PERCENT = 90; // at or above, like the System page's critical level
251
+ /** Filesystems above the threshold: a disk stays full, so it is reported once per crossing (and once after a restart — in memory only). */
252
+ const diskAboveThreshold = new Set();
253
+
254
+ /** Host CPU ticks summed over every core (`/proc/stat` on Linux): the whole machine. */
255
+ function readHostCpuSample() {
256
+ let idle = 0;
257
+ let total = 0;
258
+ for (const cpu of os.cpus()) {
259
+ idle += cpu.times.idle;
260
+ total += cpu.times.user + cpu.times.nice + cpu.times.sys + cpu.times.irq + cpu.times.idle;
223
261
  }
224
- lastCpuSample = { wallUsec: nowWallUsec, cgroupUsageUsec: null, host };
225
- return 0;
262
+ return { idle, total };
226
263
  }
227
264
 
228
- const CGROUP_CPU_STAT = '/sys/fs/cgroup/cpu.stat';
229
- const CGROUP_CPU_MAX = '/sys/fs/cgroup/cpu.max';
230
- const CGROUP_CPUSET_EFFECTIVE = '/sys/fs/cgroup/cpuset.cpus.effective';
231
- let lastCpuSample = null;
265
+ let lastCpuSample = readHostCpuSample();
232
266
 
233
- function readText(filePath) {
234
- try {
235
- return fs.readFileSync(filePath, 'utf8').trim();
236
- } catch {
237
- return null;
238
- }
267
+ /** Busy share of all cores since the previous sample, 0-100. */
268
+ function getCpuPercent() {
269
+ const sample = readHostCpuSample();
270
+ const deltaTotal = sample.total - lastCpuSample.total;
271
+ const deltaIdle = sample.idle - lastCpuSample.idle;
272
+ lastCpuSample = sample;
273
+ if (deltaTotal <= 0) return 0;
274
+ const busy = Math.max(0, deltaTotal - deltaIdle);
275
+ return Number(Math.max(0, Math.min(100, (busy / deltaTotal) * 100)).toFixed(2));
239
276
  }
240
277
 
241
- function parseCpuListCount(value) {
242
- if (!value) return 0;
243
- let count = 0;
244
- for (const token of value.split(',')) {
245
- const part = token.trim();
246
- if (!part) continue;
247
- if (part.includes('-')) {
248
- const [startRaw, endRaw] = part.split('-');
249
- const start = Number(startRaw);
250
- const end = Number(endRaw);
251
- if (Number.isFinite(start) && Number.isFinite(end) && end >= start) {
252
- count += (end - start + 1);
253
- }
254
- } else {
255
- const cpu = Number(part);
256
- if (Number.isFinite(cpu)) count += 1;
257
- }
258
- }
259
- return count;
278
+ /**
279
+ * macOS (development): used = active + wired + compressed pages, close to Activity
280
+ * Monitor's "Memory Used" — os.freemem() there counts only free pages, so the cache
281
+ * would read as used and every Mac would sit near 100 %. Swap from `vm.swapusage`.
282
+ */
283
+ function getDarwinMemory() {
284
+ const vm = spawnSync('vm_stat', { encoding: 'utf8', timeout: 3000 });
285
+ if (vm.status !== 0) return null;
286
+ const pageSize = Number(/page size of (\d+) bytes/.exec(vm.stdout)?.[1]);
287
+ const pages = (name) => Number(new RegExp(`${name}:\\s+(\\d+)`).exec(vm.stdout)?.[1] ?? NaN);
288
+ const used = (pages('Pages active') + pages('Pages wired down') + pages('Pages occupied by compressor')) * pageSize;
289
+ if (!Number.isFinite(used)) return null;
290
+ const swap = spawnSync('sysctl', ['-n', 'vm.swapusage'], { encoding: 'utf8', timeout: 3000 });
291
+ const swapMb = (label) => Number(new RegExp(`${label} = ([\\d.]+)M`).exec(swap.stdout || '')?.[1] ?? NaN);
292
+ const swapTotal = swapMb('total');
293
+ return {
294
+ ram_used_mb: Math.round(used / 1024 / 1024),
295
+ ram_total_mb: Math.round(os.totalmem() / 1024 / 1024),
296
+ swap_used_mb: Number.isFinite(swapTotal) ? Math.round(swapMb('used')) : null,
297
+ swap_total_mb: Number.isFinite(swapTotal) ? Math.round(swapTotal) : null,
298
+ };
260
299
  }
261
300
 
262
- function getCpuLimitCores() {
263
- const cpuMax = readText(CGROUP_CPU_MAX);
264
- if (cpuMax) {
265
- const [quotaRaw, periodRaw] = cpuMax.split(/\s+/);
266
- if (quotaRaw && quotaRaw !== 'max') {
267
- const quota = Number(quotaRaw);
268
- const period = Number(periodRaw);
269
- if (Number.isFinite(quota) && Number.isFinite(period) && quota > 0 && period > 0) {
270
- return Math.max(quota / period, 0.001);
271
- }
272
- }
301
+ /**
302
+ * RAM and swap in MB. On Linux from /proc/meminfo: used = MemTotal − MemAvailable, the
303
+ * figure `free` reports (page cache the kernel can reclaim is not "used"). On macOS see
304
+ * getDarwinMemory(); anywhere else os.totalmem()/os.freemem(), and no swap.
305
+ */
306
+ function getMemory() {
307
+ const mb = (kb) => Math.round(kb / 1024);
308
+ if (process.platform === 'darwin') {
309
+ const mac = getDarwinMemory();
310
+ if (mac) return mac;
273
311
  }
274
-
275
- const cpusetCount = parseCpuListCount(readText(CGROUP_CPUSET_EFFECTIVE));
276
- if (cpusetCount > 0) return cpusetCount;
277
-
278
- return os.cpus().length || 1;
312
+ try {
313
+ const info = {};
314
+ for (const line of fs.readFileSync('/proc/meminfo', 'utf8').split('\n')) {
315
+ const m = /^(\w+):\s+(\d+)/.exec(line);
316
+ if (m) info[m[1]] = Number(m[2]);
317
+ }
318
+ if (info.MemTotal && info.MemAvailable !== undefined) {
319
+ return {
320
+ ram_used_mb: mb(info.MemTotal - info.MemAvailable),
321
+ ram_total_mb: mb(info.MemTotal),
322
+ swap_used_mb: mb((info.SwapTotal || 0) - (info.SwapFree || 0)),
323
+ swap_total_mb: mb(info.SwapTotal || 0),
324
+ };
325
+ }
326
+ } catch { /* not Linux */ }
327
+ const total = os.totalmem();
328
+ return {
329
+ ram_used_mb: Math.round((total - os.freemem()) / 1024 / 1024),
330
+ ram_total_mb: Math.round(total / 1024 / 1024),
331
+ swap_used_mb: null,
332
+ swap_total_mb: null,
333
+ };
279
334
  }
280
335
 
281
- function readCgroupUsageUsec() {
282
- const stat = readText(CGROUP_CPU_STAT);
283
- if (!stat) return null;
284
- const line = stat.split('\n').find((x) => x.startsWith('usage_usec '));
285
- if (!line) return null;
286
- const usage = Number(line.split(/\s+/)[1]);
287
- if (!Number.isFinite(usage)) return null;
288
- return usage;
336
+ let dockerRootDir = null;
337
+ let dockerRootCheckedAt = 0;
338
+
339
+ /** Docker's data root (where images and agent volumes live), asked once and cached. */
340
+ function getDockerRootDir() {
341
+ if (dockerRootDir || Date.now() - dockerRootCheckedAt < DOCKER_ROOT_RETRY_MS) return dockerRootDir;
342
+ dockerRootCheckedAt = Date.now();
343
+ const r = spawnSync('docker', ['info', '--format', '{{.DockerRootDir}}'], { encoding: 'utf8', timeout: 5000 });
344
+ const dir = r.status === 0 ? String(r.stdout).trim() : '';
345
+ if (dir.startsWith('/')) dockerRootDir = dir;
346
+ return dockerRootDir;
289
347
  }
290
348
 
291
- function readHostCpuSample() {
292
- const cpus = os.cpus();
293
- let idle = 0;
294
- let total = 0;
295
- for (const cpu of cpus) {
296
- idle += cpu.times.idle;
297
- total += cpu.times.user + cpu.times.nice + cpu.times.sys + cpu.times.irq + cpu.times.idle;
349
+ /**
350
+ * The filesystems worth watching: `/`, Docker's root and Rev4a's data directory, each
351
+ * filesystem once (paths on the same device are merged, their roles listed together).
352
+ * A path that cannot be read — Docker Desktop's root lives inside its VM — is skipped.
353
+ * Sizes follow `df`: used = blocks − free, avail = what a normal user can still write.
354
+ */
355
+ function getDisks() {
356
+ const candidates = [
357
+ ['/', 'root'],
358
+ [getDockerRootDir(), 'docker'],
359
+ [path.dirname(DB_PATH), 'data'],
360
+ ];
361
+ const byDevice = new Map();
362
+ for (const [mount, role] of candidates) {
363
+ if (!mount) continue;
364
+ try {
365
+ const dev = fs.statSync(mount).dev;
366
+ const known = byDevice.get(dev);
367
+ if (known) { known.roles.push(role); continue; }
368
+ const st = fs.statfsSync(mount);
369
+ const mb = (blocks) => Math.round((blocks * st.bsize) / 1024 / 1024);
370
+ byDevice.set(dev, {
371
+ mount,
372
+ roles: [role],
373
+ used_mb: mb(st.blocks - st.bfree),
374
+ avail_mb: mb(st.bavail),
375
+ total_mb: mb(st.blocks),
376
+ });
377
+ } catch { /* not readable here */ }
298
378
  }
299
- return { idle, total };
379
+ return [...byDevice.values()];
300
380
  }
301
381
 
302
- // The filesystem holding Rev4a's own data (the events database's directory), in GB.
303
- // statfs works the same on Linux and macOS; the previous `df -BG /data` read a mount
304
- // that existed only on one old server and used GNU-only flags, so it reported 0/0.
305
- function getDiskStats() {
306
- try {
307
- const st = fs.statfsSync(path.dirname(DB_PATH));
308
- const gb = (blocks) => Math.round((blocks * st.bsize) / 1024 ** 3);
309
- return { used: gb(st.blocks - st.bfree), total: gb(st.blocks) };
310
- } catch (e) {
311
- return { used: 0, total: 0 };
382
+ /** Share of a filesystem in use the way `df` computes it: used / (used + avail). */
383
+ function diskPercent(d) {
384
+ const denom = d.used_mb + d.avail_mb;
385
+ return denom > 0 ? Math.round((d.used_mb / denom) * 100) : 0;
386
+ }
387
+
388
+ /** Write a system_anomaly event. `cooldownKey` limits it to one per ANOMALY_COOLDOWN_MS; null writes it now. */
389
+ function recordAnomaly(cooldownKey, metric, values, threshold, message) {
390
+ const nowMs = Date.now();
391
+ if (cooldownKey) {
392
+ if (nowMs - (lastAnomalyAlert[cooldownKey] || 0) <= ANOMALY_COOLDOWN_MS) return;
393
+ lastAnomalyAlert[cooldownKey] = nowMs;
312
394
  }
395
+ insertEvent.run({
396
+ ts: nowMs,
397
+ session_id: 'system',
398
+ type: 'system_anomaly',
399
+ data: JSON.stringify({ metric, values, threshold, message }),
400
+ });
401
+ log(`[ANOMALY] ${message}`);
313
402
  }
314
403
 
315
404
  function collectSystemMetrics() {
316
405
  const now = Math.floor(Date.now() / 1000);
317
406
  const cpu_percent = getCpuPercent();
318
- const ram_total = os.totalmem();
319
- const ram_free = os.freemem();
320
- const ram_used_mb = Math.round((ram_total - ram_free) / 1024 / 1024);
321
- const ram_total_mb = Math.round(ram_total / 1024 / 1024);
322
- const disk = getDiskStats();
323
- const load_avg_1m = parseFloat(os.loadavg()[0].toFixed(2));
324
-
325
- insertMetric.run({
326
- ts: now,
327
- cpu_percent,
328
- ram_used_mb,
329
- ram_total_mb,
330
- disk_used_gb: disk.used,
331
- disk_total_gb: disk.total,
332
- load_avg_1m,
333
- });
407
+ const mem = getMemory();
408
+ const disks = getDisks();
409
+ const [load1, load5, load15] = os.loadavg().map((v) => Number(v.toFixed(2)));
334
410
 
335
- // Prune old data (keep 30 days)
336
- pruneMetrics.run(now - 30 * 24 * 3600);
337
-
338
- // Anomaly detection — check last 2 consecutive samples
339
- const recent = getLastTwoMetrics.all();
411
+ metrics.db.transaction(() => {
412
+ metrics.insertMetric.run({
413
+ ts: now,
414
+ cpu_percent,
415
+ ...mem,
416
+ load_avg_1m: load1,
417
+ load_avg_5m: load5,
418
+ load_avg_15m: load15,
419
+ });
420
+ for (const d of disks) metrics.insertDisk.run({ ts: now, ...d, roles: d.roles.join(',') });
421
+ metrics.pruneMetrics.run(now - METRICS_RETENTION_S);
422
+ metrics.pruneDisks.run(now - METRICS_RETENTION_S);
423
+ })();
424
+
425
+ // Anomalies: CPU > 85 % and RAM > 90 % on two consecutive samples (with a cooldown); a
426
+ // filesystem once when it reaches DISK_ANOMALY_PERCENT, again only after it drops back
427
+ // under it — disk usage does not spike and fall back, a cooldown would repeat forever.
428
+ const recent = metrics.getLastTwoMetrics.all();
340
429
  if (recent.length === 2) {
341
- const nowMs = Date.now();
342
- // CPU >85% for 2 consecutive samples
343
430
  if (recent[0].cpu_percent > 85 && recent[1].cpu_percent > 85) {
344
- if (nowMs - lastAnomalyAlert.cpu > ANOMALY_COOLDOWN_MS) {
345
- lastAnomalyAlert.cpu = nowMs;
346
- insertEvent.run({
347
- ts: Date.now(),
348
- session_id: 'system',
349
- type: 'system_anomaly',
350
- data: JSON.stringify({
351
- metric: 'cpu',
352
- values: [recent[1].cpu_percent, recent[0].cpu_percent],
353
- threshold: 85,
354
- message: `CPU high: ${recent[0].cpu_percent}% for 2 consecutive samples`,
355
- }),
356
- });
357
- log(`[ANOMALY] CPU high: ${recent[0].cpu_percent}%`);
358
- }
431
+ recordAnomaly('cpu', 'cpu', [recent[1].cpu_percent, recent[0].cpu_percent], 85,
432
+ `CPU high: ${recent[0].cpu_percent}% for 2 consecutive samples`);
359
433
  }
360
- // RAM >90% for 2 consecutive samples
361
- const ram0pct = recent[0].ram_total_mb > 0 ? Math.round(recent[0].ram_used_mb / recent[0].ram_total_mb * 100) : 0;
362
- const ram1pct = recent[1].ram_total_mb > 0 ? Math.round(recent[1].ram_used_mb / recent[1].ram_total_mb * 100) : 0;
363
- if (ram0pct > 90 && ram1pct > 90) {
364
- if (nowMs - lastAnomalyAlert.ram > ANOMALY_COOLDOWN_MS) {
365
- lastAnomalyAlert.ram = nowMs;
366
- insertEvent.run({
367
- ts: Date.now(),
368
- session_id: 'system',
369
- type: 'system_anomaly',
370
- data: JSON.stringify({
371
- metric: 'ram',
372
- values: [ram1pct, ram0pct],
373
- threshold: 90,
374
- message: `RAM high: ${ram0pct}% for 2 consecutive samples`,
375
- }),
376
- });
377
- log(`[ANOMALY] RAM high: ${ram0pct}%`);
378
- }
434
+ const pct = (r) => (r.ram_total_mb > 0 ? Math.round((r.ram_used_mb / r.ram_total_mb) * 100) : 0);
435
+ if (pct(recent[0]) > 90 && pct(recent[1]) > 90) {
436
+ recordAnomaly('ram', 'ram', [pct(recent[1]), pct(recent[0])], 90,
437
+ `RAM high: ${pct(recent[0])}% for 2 consecutive samples`);
438
+ }
439
+ }
440
+ for (const d of disks) {
441
+ const p = diskPercent(d);
442
+ if (p < DISK_ANOMALY_PERCENT) {
443
+ diskAboveThreshold.delete(d.mount);
444
+ } else if (!diskAboveThreshold.has(d.mount)) {
445
+ diskAboveThreshold.add(d.mount);
446
+ recordAnomaly(null, 'disk', [p], DISK_ANOMALY_PERCENT, `Disk high: ${d.mount} ${p}% used`);
379
447
  }
380
448
  }
381
-
382
- log(`[METRICS] CPU:${cpu_percent}% RAM:${ram_used_mb}/${ram_total_mb}MB Disk:${disk.used}/${disk.total}GB Load:${load_avg_1m}`);
383
449
  }
384
450
 
385
451
  // ── Helpers ─────────────────────────────────────────────────────────────────
@@ -694,9 +760,6 @@ function pollSessions() {
694
760
  log(`[TIMEOUT] ${session_id} missing for ${Math.round(missingFor / 60000)} min`);
695
761
  }
696
762
 
697
- // Collect system metrics after each poll
698
- try { collectSystemMetrics(); } catch (e) { log(`[METRICS ERROR] ${e.message}`); }
699
-
700
763
  // Force names and parents declared via lineage — overrides any previous label, including the generic placeholder older data carries
701
764
  const updateLabel = db.prepare('UPDATE sessions SET label = ?, parent_id = ? WHERE session_id = ?');
702
765
  const applyLineage = db.transaction(() => {
@@ -746,17 +809,35 @@ log(`Poll interval: ${POLL_INTERVAL_MS / 1000}s`);
746
809
  pollSessions();
747
810
  const timer = setInterval(pollSessions, POLL_INTERVAL_MS);
748
811
 
812
+ // Machine metrics: the first sample a few seconds after start (the CPU figure needs a
813
+ // delta from the baseline taken at load), then every interval. Off when metrics.db failed.
814
+ function sampleMetrics() {
815
+ try { collectSystemMetrics(); } catch (e) { log(`[METRICS ERROR] ${e.message}`); }
816
+ }
817
+ const FIRST_SAMPLE_DELAY_MS = 5000;
818
+ let metricsTimer = null;
819
+ const firstSample = metrics
820
+ ? setTimeout(() => { sampleMetrics(); metricsTimer = setInterval(sampleMetrics, METRICS_INTERVAL_MS); }, FIRST_SAMPLE_DELAY_MS)
821
+ : null;
822
+ if (metrics) log(`Machine metrics every ${METRICS_INTERVAL_MS / 1000}s into ${METRICS_DB_PATH}`);
823
+
749
824
  // Graceful shutdown
750
825
  process.on('SIGTERM', () => {
751
826
  log('SIGTERM received, shutting down...');
752
827
  clearInterval(timer);
828
+ clearTimeout(firstSample);
829
+ clearInterval(metricsTimer);
753
830
  db.close();
831
+ metrics?.db.close();
754
832
  process.exit(0);
755
833
  });
756
834
 
757
835
  process.on('SIGINT', () => {
758
836
  log('SIGINT received, shutting down...');
759
837
  clearInterval(timer);
838
+ clearTimeout(firstSample);
839
+ clearInterval(metricsTimer);
760
840
  db.close();
841
+ metrics?.db.close();
761
842
  process.exit(0);
762
843
  });