@flame0510/project-aether 1.6.2 → 1.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +6 -2
  3. package/agent-templates/atlas/HEARTBEAT.md +1 -1
  4. package/app/agents/ImageDownloadBanner.tsx +171 -37
  5. package/app/api/agents/download-image/route.ts +29 -5
  6. package/app/api/agents/image-status/route.ts +17 -2
  7. package/app/api/assistant/route.ts +21 -5
  8. package/app/api/auth/login/route.ts +2 -2
  9. package/app/api/metrics/route.ts +126 -23
  10. package/app/api/setup/agent-image/route.ts +6 -4
  11. package/app/api/system-health/route.ts +25 -21
  12. package/app/components/DashboardToolbar.tsx +2 -2
  13. package/app/components/LineageGraphPage.tsx +4 -4
  14. package/app/components/SessionDrawer.tsx +8 -8
  15. package/app/components/Sidebar.tsx +10 -0
  16. package/app/components/Skeleton.tsx +4 -1
  17. package/app/components/SystemCockpit.tsx +83 -3
  18. package/app/components/ui/Meter.tsx +34 -0
  19. package/app/components/ui/TimeSeriesChart.tsx +226 -0
  20. package/app/components/ui/index.ts +2 -0
  21. package/app/globals.css +42 -1
  22. package/app/setup/PageClient.tsx +1 -1
  23. package/app/system/PageClient.tsx +263 -0
  24. package/app/system/SystemSkeleton.tsx +115 -0
  25. package/app/system/loading.tsx +13 -0
  26. package/app/system/page.tsx +5 -0
  27. package/bin/postinstall.js +5 -1
  28. package/daemon.js +274 -214
  29. package/docs/ARCHITECTURE.md +65 -34
  30. package/docs/CONTAINER-TERMINAL.md +17 -8
  31. package/docs/DESIGN-SYSTEM.md +22 -12
  32. package/docs/FRONTEND-ARCHITECTURE.md +8 -4
  33. package/docs/REV4A.md +37 -86
  34. package/docs/dev/API-REFERENCE.md +102 -19
  35. package/docs/dev/DATABASE.md +79 -28
  36. package/docs/dev/SESSION-MAINTENANCE-PLAN.md +6 -6
  37. package/docs/rag/DATA-FRESHNESS.md +31 -15
  38. package/docs/rag/GLOSSARY.md +8 -5
  39. package/docs/rag/REV4A-OVERVIEW.md +14 -7
  40. package/docs/rag/WHAT-I-CAN-ANSWER.md +4 -3
  41. package/lib/agent-images.ts +43 -14
  42. package/lib/buildAgentImage.ts +142 -6
  43. package/lib/db-bootstrap.mjs +0 -11
  44. package/lib/metrics-db.ts +48 -0
  45. package/lib/patterns/sessionPresentation.ts +4 -2
  46. package/lib/rev4a-auth.d.ts +1 -0
  47. package/lib/rev4a-auth.js +18 -2
  48. package/next.config.mjs +9 -1
  49. package/package.json +2 -2
  50. package/scripts/backup.sh +48 -54
  51. package/scripts/check-language.mjs +21 -3
  52. package/scripts/restore.sh +77 -59
package/daemon.js CHANGED
@@ -1,7 +1,8 @@
1
1
  #!/usr/bin/env node
2
2
  /**
3
3
  * Rev4a Daemon — SQLite event logger
4
- * Polls OpenClaw sessions every 30s, writes events to SQLite.
4
+ * Polls OpenClaw sessions every 30s, writes events to SQLite, and samples the host
5
+ * machine (CPU, RAM, swap, storage) every 30s on a timer of its own.
5
6
  */
6
7
 
7
8
  'use strict';
@@ -12,9 +13,14 @@ const { spawnSync, execSync } = require('child_process');
12
13
  const Database = require('better-sqlite3');
13
14
  const path = require('path');
14
15
 
15
- const DB_PATH = process.env.REV4A_DB || path.join(os.homedir(), '.config', 'rev4a', 'data', 'events.db');
16
+ // Same rule as lib/rev4a-paths.ts: REV4A_DB, else <data dir>/data/events.db, where the data
17
+ // dir is REV4A_DATA_DIR or ~/.config/rev4a — the API reads the files the daemon writes.
18
+ const DATA_DIR = process.env.REV4A_DATA_DIR
19
+ || (process.platform === 'win32'
20
+ ? path.join(process.env.APPDATA || path.join(os.homedir(), 'AppData', 'Roaming'), 'rev4a')
21
+ : path.join(os.homedir(), '.config', 'rev4a'));
22
+ const DB_PATH = process.env.REV4A_DB || path.join(DATA_DIR, 'data', 'events.db');
16
23
  const POLL_INTERVAL_MS = 30_000;
17
- const POLL_INTERVAL_ACTIVE_MS = 15_000; // 15s when working sessions detected
18
24
 
19
25
  // Cost rates per 1M tokens (separate in/out pricing)
20
26
  const MODEL_PRICING = {
@@ -72,20 +78,6 @@ process.on('uncaughtException', (err) => {
72
78
  // Do NOT exit — the watchdog reads this from the healthcheck
73
79
  });
74
80
 
75
- db.exec(`
76
- CREATE TABLE IF NOT EXISTS system_metrics (
77
- id INTEGER PRIMARY KEY AUTOINCREMENT,
78
- ts INTEGER NOT NULL,
79
- cpu_percent REAL,
80
- ram_used_mb INTEGER,
81
- ram_total_mb INTEGER,
82
- disk_used_gb REAL,
83
- disk_total_gb REAL,
84
- load_avg_1m REAL
85
- );
86
- CREATE INDEX IF NOT EXISTS idx_metrics_ts ON system_metrics(ts);
87
- `);
88
-
89
81
  db.exec(`
90
82
  CREATE TABLE IF NOT EXISTS sessions (
91
83
  session_id TEXT PRIMARY KEY,
@@ -177,215 +169,283 @@ const knownSessions = new Map(); // session_id -> { status, tokens_in, tokens_ou
177
169
  let pollCount = 0;
178
170
 
179
171
  // ── System Metrics ───────────────────────────────────────────────────────────
172
+ //
173
+ // Machine-wide CPU, RAM, swap and storage of the host Rev4a runs on, sampled on their
174
+ // own timer (METRICS_INTERVAL_MS) — independent of the OpenClaw session poll, which has
175
+ // no source on a host without the `openclaw` CLI.
180
176
 
181
- const insertMetric = db.prepare(`
182
- INSERT INTO system_metrics (ts, cpu_percent, ram_used_mb, ram_total_mb, disk_used_gb, disk_total_gb, load_avg_1m)
183
- VALUES (@ts, @cpu_percent, @ram_used_mb, @ram_total_mb, @disk_used_gb, @disk_total_gb, @load_avg_1m)
184
- `);
185
- const pruneMetrics = db.prepare(`DELETE FROM system_metrics WHERE ts < ?`);
186
- const getLastTwoMetrics = db.prepare(`SELECT cpu_percent, ram_used_mb, ram_total_mb FROM system_metrics ORDER BY ts DESC LIMIT 2`);
177
+ /**
178
+ * Machine metrics live in their own database next to events.db, so the event log stays
179
+ * sessions and events only and the history can be dropped without touching them. The
180
+ * daemon is its only writer; GET /api/metrics reads it (lib/metrics-db.ts, same path).
181
+ * Anomalies are still events, in events.db, where the live feed reads them.
182
+ *
183
+ * Opened in a try: a metrics.db that cannot be opened (corrupt, unwritable, disk full)
184
+ * turns sampling off with a log line and leaves the session poll running.
185
+ */
186
+ const METRICS_DB_PATH = path.join(path.dirname(DB_PATH), 'metrics.db');
187
+ const metrics = openMetricsStore();
187
188
 
188
- // Anomaly cooldown: track last anomaly alert timestamps
189
- const lastAnomalyAlert = { cpu: 0, ram: 0 };
190
- const ANOMALY_COOLDOWN_MS = 5 * 60 * 1000; // 5 minutes
189
+ function openMetricsStore() {
190
+ try {
191
+ const mdb = new Database(METRICS_DB_PATH);
192
+ mdb.pragma('journal_mode = WAL');
193
+ mdb.pragma('synchronous = NORMAL');
194
+ mdb.exec(`
195
+ CREATE TABLE IF NOT EXISTS system_metrics (
196
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
197
+ ts INTEGER NOT NULL,
198
+ cpu_percent REAL,
199
+ ram_used_mb INTEGER,
200
+ ram_total_mb INTEGER,
201
+ swap_used_mb INTEGER,
202
+ swap_total_mb INTEGER,
203
+ load_avg_1m REAL,
204
+ load_avg_5m REAL,
205
+ load_avg_15m REAL
206
+ );
207
+ CREATE INDEX IF NOT EXISTS idx_metrics_ts ON system_metrics(ts);
191
208
 
192
- function getCpuPercent() {
193
- // Container-accurate CPU: prefer cgroup v2 usage delta over host cumulative ticks.
194
- const nowWallUsec = Number(process.hrtime.bigint() / 1000n);
195
- const cgroupUsageUsec = readCgroupUsageUsec();
196
-
197
- if (cgroupUsageUsec !== null) {
198
- if (lastCpuSample && lastCpuSample.cgroupUsageUsec !== null) {
199
- const deltaUsageUsec = cgroupUsageUsec - lastCpuSample.cgroupUsageUsec;
200
- const deltaWallUsec = nowWallUsec - lastCpuSample.wallUsec;
201
- if (deltaUsageUsec >= 0 && deltaWallUsec > 0) {
202
- const limitCores = getCpuLimitCores();
203
- const raw = (deltaUsageUsec / deltaWallUsec) / limitCores * 100;
204
- const clamped = Math.max(0, Math.min(100, raw));
205
- lastCpuSample = { wallUsec: nowWallUsec, cgroupUsageUsec, host: null };
206
- return Number(clamped.toFixed(2));
207
- }
208
- }
209
- lastCpuSample = { wallUsec: nowWallUsec, cgroupUsageUsec, host: null };
210
- return 0;
209
+ CREATE TABLE IF NOT EXISTS system_disks (
210
+ ts INTEGER NOT NULL,
211
+ mount TEXT NOT NULL,
212
+ roles TEXT,
213
+ used_mb INTEGER,
214
+ avail_mb INTEGER,
215
+ total_mb INTEGER
216
+ );
217
+ CREATE INDEX IF NOT EXISTS idx_disks_ts ON system_disks(ts);
218
+ CREATE INDEX IF NOT EXISTS idx_disks_mount_ts ON system_disks(mount, ts);
219
+ `);
220
+ return {
221
+ db: mdb,
222
+ insertMetric: mdb.prepare(`
223
+ INSERT INTO system_metrics (ts, cpu_percent, ram_used_mb, ram_total_mb, swap_used_mb, swap_total_mb,
224
+ load_avg_1m, load_avg_5m, load_avg_15m)
225
+ VALUES (@ts, @cpu_percent, @ram_used_mb, @ram_total_mb, @swap_used_mb, @swap_total_mb,
226
+ @load_avg_1m, @load_avg_5m, @load_avg_15m)
227
+ `),
228
+ insertDisk: mdb.prepare(`
229
+ INSERT INTO system_disks (ts, mount, roles, used_mb, avail_mb, total_mb)
230
+ VALUES (@ts, @mount, @roles, @used_mb, @avail_mb, @total_mb)
231
+ `),
232
+ pruneMetrics: mdb.prepare('DELETE FROM system_metrics WHERE ts < ?'),
233
+ pruneDisks: mdb.prepare('DELETE FROM system_disks WHERE ts < ?'),
234
+ getLastTwoMetrics: mdb.prepare('SELECT cpu_percent, ram_used_mb, ram_total_mb FROM system_metrics ORDER BY ts DESC LIMIT 2'),
235
+ };
236
+ } catch (err) {
237
+ log(`[METRICS] disabled: cannot open ${METRICS_DB_PATH}: ${err.message}`);
238
+ return null;
211
239
  }
240
+ }
212
241
 
213
- const host = readHostCpuSample();
214
- if (lastCpuSample && lastCpuSample.host) {
215
- const deltaTotal = host.total - lastCpuSample.host.total;
216
- const deltaIdle = host.idle - lastCpuSample.host.idle;
217
- if (deltaTotal > 0) {
218
- const busy = Math.max(0, deltaTotal - deltaIdle);
219
- const raw = (busy / deltaTotal) * 100;
220
- const clamped = Math.max(0, Math.min(100, raw));
221
- lastCpuSample = { wallUsec: nowWallUsec, cgroupUsageUsec: null, host };
222
- return Number(clamped.toFixed(2));
223
- }
242
+ const METRICS_INTERVAL_MS = 30_000;
243
+ const METRICS_RETENTION_S = 30 * 24 * 3600;
244
+ /** How long to wait before asking Docker for its root directory again after a failure. */
245
+ const DOCKER_ROOT_RETRY_MS = 10 * 60 * 1000;
246
+
247
+ // Anomaly cooldown: last alert per metric (cpu, ram); disks use the crossing state below
248
+ const lastAnomalyAlert = {};
249
+ const ANOMALY_COOLDOWN_MS = 5 * 60 * 1000; // 5 minutes
250
+ const DISK_ANOMALY_PERCENT = 90; // at or above, like the System page's critical level
251
+ /** Filesystems above the threshold: a disk stays full, so it is reported once per crossing (and once after a restart — in memory only). */
252
+ const diskAboveThreshold = new Set();
253
+
254
+ /** Host CPU ticks summed over every core (`/proc/stat` on Linux): the whole machine. */
255
+ function readHostCpuSample() {
256
+ let idle = 0;
257
+ let total = 0;
258
+ for (const cpu of os.cpus()) {
259
+ idle += cpu.times.idle;
260
+ total += cpu.times.user + cpu.times.nice + cpu.times.sys + cpu.times.irq + cpu.times.idle;
224
261
  }
225
- lastCpuSample = { wallUsec: nowWallUsec, cgroupUsageUsec: null, host };
226
- return 0;
262
+ return { idle, total };
227
263
  }
228
264
 
229
- const CGROUP_CPU_STAT = '/sys/fs/cgroup/cpu.stat';
230
- const CGROUP_CPU_MAX = '/sys/fs/cgroup/cpu.max';
231
- const CGROUP_CPUSET_EFFECTIVE = '/sys/fs/cgroup/cpuset.cpus.effective';
232
- let lastCpuSample = null;
265
+ let lastCpuSample = readHostCpuSample();
233
266
 
234
- function readText(filePath) {
235
- try {
236
- return fs.readFileSync(filePath, 'utf8').trim();
237
- } catch {
238
- return null;
239
- }
267
+ /** Busy share of all cores since the previous sample, 0-100. */
268
+ function getCpuPercent() {
269
+ const sample = readHostCpuSample();
270
+ const deltaTotal = sample.total - lastCpuSample.total;
271
+ const deltaIdle = sample.idle - lastCpuSample.idle;
272
+ lastCpuSample = sample;
273
+ if (deltaTotal <= 0) return 0;
274
+ const busy = Math.max(0, deltaTotal - deltaIdle);
275
+ return Number(Math.max(0, Math.min(100, (busy / deltaTotal) * 100)).toFixed(2));
240
276
  }
241
277
 
242
- function parseCpuListCount(value) {
243
- if (!value) return 0;
244
- let count = 0;
245
- for (const token of value.split(',')) {
246
- const part = token.trim();
247
- if (!part) continue;
248
- if (part.includes('-')) {
249
- const [startRaw, endRaw] = part.split('-');
250
- const start = Number(startRaw);
251
- const end = Number(endRaw);
252
- if (Number.isFinite(start) && Number.isFinite(end) && end >= start) {
253
- count += (end - start + 1);
254
- }
255
- } else {
256
- const cpu = Number(part);
257
- if (Number.isFinite(cpu)) count += 1;
258
- }
259
- }
260
- return count;
278
+ /**
279
+ * macOS (development): used = active + wired + compressed pages, close to Activity
280
+ * Monitor's "Memory Used" — os.freemem() there counts only free pages, so the cache
281
+ * would read as used and every Mac would sit near 100 %. Swap from `vm.swapusage`.
282
+ */
283
+ function getDarwinMemory() {
284
+ const vm = spawnSync('vm_stat', { encoding: 'utf8', timeout: 3000 });
285
+ if (vm.status !== 0) return null;
286
+ const pageSize = Number(/page size of (\d+) bytes/.exec(vm.stdout)?.[1]);
287
+ const pages = (name) => Number(new RegExp(`${name}:\\s+(\\d+)`).exec(vm.stdout)?.[1] ?? NaN);
288
+ const used = (pages('Pages active') + pages('Pages wired down') + pages('Pages occupied by compressor')) * pageSize;
289
+ if (!Number.isFinite(used)) return null;
290
+ const swap = spawnSync('sysctl', ['-n', 'vm.swapusage'], { encoding: 'utf8', timeout: 3000 });
291
+ const swapMb = (label) => Number(new RegExp(`${label} = ([\\d.]+)M`).exec(swap.stdout || '')?.[1] ?? NaN);
292
+ const swapTotal = swapMb('total');
293
+ return {
294
+ ram_used_mb: Math.round(used / 1024 / 1024),
295
+ ram_total_mb: Math.round(os.totalmem() / 1024 / 1024),
296
+ swap_used_mb: Number.isFinite(swapTotal) ? Math.round(swapMb('used')) : null,
297
+ swap_total_mb: Number.isFinite(swapTotal) ? Math.round(swapTotal) : null,
298
+ };
261
299
  }
262
300
 
263
- function getCpuLimitCores() {
264
- const cpuMax = readText(CGROUP_CPU_MAX);
265
- if (cpuMax) {
266
- const [quotaRaw, periodRaw] = cpuMax.split(/\s+/);
267
- if (quotaRaw && quotaRaw !== 'max') {
268
- const quota = Number(quotaRaw);
269
- const period = Number(periodRaw);
270
- if (Number.isFinite(quota) && Number.isFinite(period) && quota > 0 && period > 0) {
271
- return Math.max(quota / period, 0.001);
272
- }
273
- }
301
+ /**
302
+ * RAM and swap in MB. On Linux from /proc/meminfo: used = MemTotal − MemAvailable, the
303
+ * figure `free` reports (page cache the kernel can reclaim is not "used"). On macOS see
304
+ * getDarwinMemory(); anywhere else os.totalmem()/os.freemem(), and no swap.
305
+ */
306
+ function getMemory() {
307
+ const mb = (kb) => Math.round(kb / 1024);
308
+ if (process.platform === 'darwin') {
309
+ const mac = getDarwinMemory();
310
+ if (mac) return mac;
274
311
  }
275
-
276
- const cpusetCount = parseCpuListCount(readText(CGROUP_CPUSET_EFFECTIVE));
277
- if (cpusetCount > 0) return cpusetCount;
278
-
279
- return os.cpus().length || 1;
312
+ try {
313
+ const info = {};
314
+ for (const line of fs.readFileSync('/proc/meminfo', 'utf8').split('\n')) {
315
+ const m = /^(\w+):\s+(\d+)/.exec(line);
316
+ if (m) info[m[1]] = Number(m[2]);
317
+ }
318
+ if (info.MemTotal && info.MemAvailable !== undefined) {
319
+ return {
320
+ ram_used_mb: mb(info.MemTotal - info.MemAvailable),
321
+ ram_total_mb: mb(info.MemTotal),
322
+ swap_used_mb: mb((info.SwapTotal || 0) - (info.SwapFree || 0)),
323
+ swap_total_mb: mb(info.SwapTotal || 0),
324
+ };
325
+ }
326
+ } catch { /* not Linux */ }
327
+ const total = os.totalmem();
328
+ return {
329
+ ram_used_mb: Math.round((total - os.freemem()) / 1024 / 1024),
330
+ ram_total_mb: Math.round(total / 1024 / 1024),
331
+ swap_used_mb: null,
332
+ swap_total_mb: null,
333
+ };
280
334
  }
281
335
 
282
- function readCgroupUsageUsec() {
283
- const stat = readText(CGROUP_CPU_STAT);
284
- if (!stat) return null;
285
- const line = stat.split('\n').find((x) => x.startsWith('usage_usec '));
286
- if (!line) return null;
287
- const usage = Number(line.split(/\s+/)[1]);
288
- if (!Number.isFinite(usage)) return null;
289
- return usage;
336
+ let dockerRootDir = null;
337
+ let dockerRootCheckedAt = 0;
338
+
339
+ /** Docker's data root (where images and agent volumes live), asked once and cached. */
340
+ function getDockerRootDir() {
341
+ if (dockerRootDir || Date.now() - dockerRootCheckedAt < DOCKER_ROOT_RETRY_MS) return dockerRootDir;
342
+ dockerRootCheckedAt = Date.now();
343
+ const r = spawnSync('docker', ['info', '--format', '{{.DockerRootDir}}'], { encoding: 'utf8', timeout: 5000 });
344
+ const dir = r.status === 0 ? String(r.stdout).trim() : '';
345
+ if (dir.startsWith('/')) dockerRootDir = dir;
346
+ return dockerRootDir;
290
347
  }
291
348
 
292
- function readHostCpuSample() {
293
- const cpus = os.cpus();
294
- let idle = 0;
295
- let total = 0;
296
- for (const cpu of cpus) {
297
- idle += cpu.times.idle;
298
- total += cpu.times.user + cpu.times.nice + cpu.times.sys + cpu.times.irq + cpu.times.idle;
349
+ /**
350
+ * The filesystems worth watching: `/`, Docker's root and Rev4a's data directory, each
351
+ * filesystem once (paths on the same device are merged, their roles listed together).
352
+ * A path that cannot be read — Docker Desktop's root lives inside its VM — is skipped.
353
+ * Sizes follow `df`: used = blocks − free, avail = what a normal user can still write.
354
+ */
355
+ function getDisks() {
356
+ const candidates = [
357
+ ['/', 'root'],
358
+ [getDockerRootDir(), 'docker'],
359
+ [path.dirname(DB_PATH), 'data'],
360
+ ];
361
+ const byDevice = new Map();
362
+ for (const [mount, role] of candidates) {
363
+ if (!mount) continue;
364
+ try {
365
+ const dev = fs.statSync(mount).dev;
366
+ const known = byDevice.get(dev);
367
+ if (known) { known.roles.push(role); continue; }
368
+ const st = fs.statfsSync(mount);
369
+ const mb = (blocks) => Math.round((blocks * st.bsize) / 1024 / 1024);
370
+ byDevice.set(dev, {
371
+ mount,
372
+ roles: [role],
373
+ used_mb: mb(st.blocks - st.bfree),
374
+ avail_mb: mb(st.bavail),
375
+ total_mb: mb(st.blocks),
376
+ });
377
+ } catch { /* not readable here */ }
299
378
  }
300
- return { idle, total };
379
+ return [...byDevice.values()];
301
380
  }
302
381
 
303
- function getDiskStats() {
304
- try {
305
- const result = spawnSync('df', ['-BG', '/data', '--output=used,size'], {
306
- encoding: 'utf8', timeout: 5000, killSignal: 'SIGKILL',
307
- });
308
- if (result.status !== 0 || !result.stdout) return { used: 0, total: 0 };
309
- const lines = result.stdout.trim().split('\n');
310
- // lines[0] = header, lines[1] = data
311
- if (lines.length < 2) return { used: 0, total: 0 };
312
- const parts = lines[1].trim().split(/\s+/);
313
- const used = parseFloat(parts[0]) || 0; // already in GB (BG flag strips G)
314
- const total = parseFloat(parts[1]) || 0;
315
- return { used, total };
316
- } catch (e) {
317
- return { used: 0, total: 0 };
382
+ /** Share of a filesystem in use the way `df` computes it: used / (used + avail). */
383
+ function diskPercent(d) {
384
+ const denom = d.used_mb + d.avail_mb;
385
+ return denom > 0 ? Math.round((d.used_mb / denom) * 100) : 0;
386
+ }
387
+
388
+ /** Write a system_anomaly event. `cooldownKey` limits it to one per ANOMALY_COOLDOWN_MS; null writes it now. */
389
+ function recordAnomaly(cooldownKey, metric, values, threshold, message) {
390
+ const nowMs = Date.now();
391
+ if (cooldownKey) {
392
+ if (nowMs - (lastAnomalyAlert[cooldownKey] || 0) <= ANOMALY_COOLDOWN_MS) return;
393
+ lastAnomalyAlert[cooldownKey] = nowMs;
318
394
  }
395
+ insertEvent.run({
396
+ ts: nowMs,
397
+ session_id: 'system',
398
+ type: 'system_anomaly',
399
+ data: JSON.stringify({ metric, values, threshold, message }),
400
+ });
401
+ log(`[ANOMALY] ${message}`);
319
402
  }
320
403
 
321
404
  function collectSystemMetrics() {
322
405
  const now = Math.floor(Date.now() / 1000);
323
406
  const cpu_percent = getCpuPercent();
324
- const ram_total = os.totalmem();
325
- const ram_free = os.freemem();
326
- const ram_used_mb = Math.round((ram_total - ram_free) / 1024 / 1024);
327
- const ram_total_mb = Math.round(ram_total / 1024 / 1024);
328
- const disk = getDiskStats();
329
- const load_avg_1m = parseFloat(os.loadavg()[0].toFixed(2));
330
-
331
- insertMetric.run({
332
- ts: now,
333
- cpu_percent,
334
- ram_used_mb,
335
- ram_total_mb,
336
- disk_used_gb: disk.used,
337
- disk_total_gb: disk.total,
338
- load_avg_1m,
339
- });
407
+ const mem = getMemory();
408
+ const disks = getDisks();
409
+ const [load1, load5, load15] = os.loadavg().map((v) => Number(v.toFixed(2)));
340
410
 
341
- // Prune old data (keep 30 days)
342
- pruneMetrics.run(now - 30 * 24 * 3600);
343
-
344
- // Anomaly detection — check last 2 consecutive samples
345
- const recent = getLastTwoMetrics.all();
411
+ metrics.db.transaction(() => {
412
+ metrics.insertMetric.run({
413
+ ts: now,
414
+ cpu_percent,
415
+ ...mem,
416
+ load_avg_1m: load1,
417
+ load_avg_5m: load5,
418
+ load_avg_15m: load15,
419
+ });
420
+ for (const d of disks) metrics.insertDisk.run({ ts: now, ...d, roles: d.roles.join(',') });
421
+ metrics.pruneMetrics.run(now - METRICS_RETENTION_S);
422
+ metrics.pruneDisks.run(now - METRICS_RETENTION_S);
423
+ })();
424
+
425
+ // Anomalies: CPU > 85 % and RAM > 90 % on two consecutive samples (with a cooldown); a
426
+ // filesystem once when it reaches DISK_ANOMALY_PERCENT, again only after it drops back
427
+ // under it — disk usage does not spike and fall back, a cooldown would repeat forever.
428
+ const recent = metrics.getLastTwoMetrics.all();
346
429
  if (recent.length === 2) {
347
- const nowMs = Date.now();
348
- // CPU >85% for 2 consecutive samples
349
430
  if (recent[0].cpu_percent > 85 && recent[1].cpu_percent > 85) {
350
- if (nowMs - lastAnomalyAlert.cpu > ANOMALY_COOLDOWN_MS) {
351
- lastAnomalyAlert.cpu = nowMs;
352
- insertEvent.run({
353
- ts: Date.now(),
354
- session_id: 'system',
355
- type: 'system_anomaly',
356
- data: JSON.stringify({
357
- metric: 'cpu',
358
- values: [recent[1].cpu_percent, recent[0].cpu_percent],
359
- threshold: 85,
360
- message: `CPU alta: ${recent[0].cpu_percent}% per 2 campioni consecutivi`,
361
- }),
362
- });
363
- log(`[ANOMALY] CPU alta: ${recent[0].cpu_percent}%`);
364
- }
431
+ recordAnomaly('cpu', 'cpu', [recent[1].cpu_percent, recent[0].cpu_percent], 85,
432
+ `CPU high: ${recent[0].cpu_percent}% for 2 consecutive samples`);
365
433
  }
366
- // RAM >90% for 2 consecutive samples
367
- const ram0pct = recent[0].ram_total_mb > 0 ? Math.round(recent[0].ram_used_mb / recent[0].ram_total_mb * 100) : 0;
368
- const ram1pct = recent[1].ram_total_mb > 0 ? Math.round(recent[1].ram_used_mb / recent[1].ram_total_mb * 100) : 0;
369
- if (ram0pct > 90 && ram1pct > 90) {
370
- if (nowMs - lastAnomalyAlert.ram > ANOMALY_COOLDOWN_MS) {
371
- lastAnomalyAlert.ram = nowMs;
372
- insertEvent.run({
373
- ts: Date.now(),
374
- session_id: 'system',
375
- type: 'system_anomaly',
376
- data: JSON.stringify({
377
- metric: 'ram',
378
- values: [ram1pct, ram0pct],
379
- threshold: 90,
380
- message: `RAM alta: ${ram0pct}% per 2 campioni consecutivi`,
381
- }),
382
- });
383
- log(`[ANOMALY] RAM alta: ${ram0pct}%`);
384
- }
434
+ const pct = (r) => (r.ram_total_mb > 0 ? Math.round((r.ram_used_mb / r.ram_total_mb) * 100) : 0);
435
+ if (pct(recent[0]) > 90 && pct(recent[1]) > 90) {
436
+ recordAnomaly('ram', 'ram', [pct(recent[1]), pct(recent[0])], 90,
437
+ `RAM high: ${pct(recent[0])}% for 2 consecutive samples`);
438
+ }
439
+ }
440
+ for (const d of disks) {
441
+ const p = diskPercent(d);
442
+ if (p < DISK_ANOMALY_PERCENT) {
443
+ diskAboveThreshold.delete(d.mount);
444
+ } else if (!diskAboveThreshold.has(d.mount)) {
445
+ diskAboveThreshold.add(d.mount);
446
+ recordAnomaly(null, 'disk', [p], DISK_ANOMALY_PERCENT, `Disk high: ${d.mount} ${p}% used`);
385
447
  }
386
448
  }
387
-
388
- log(`[METRICS] CPU:${cpu_percent}% RAM:${ram_used_mb}/${ram_total_mb}MB Disk:${disk.used}/${disk.total}GB Load:${load_avg_1m}`);
389
449
  }
390
450
 
391
451
  // ── Helpers ─────────────────────────────────────────────────────────────────
@@ -540,7 +600,7 @@ function pollSessions() {
540
600
  const session_id = s.key;
541
601
 
542
602
  // Skip Telegram channel/group sessions (multi-user).
543
- // Keep Telegram direct sessions (agent:ops:telegram:argus:direct:...) as root nodes
603
+ // Keep Telegram direct sessions (agent:<id>:telegram:<account>:direct:...) as root nodes
544
604
  // since they are the parent of all sub-agents spawned via Telegram.
545
605
  if (session_id.includes(':telegram:') && !session_id.includes(':direct:')) continue;
546
606
  const { label, parent_id: inferredParent } = parseSessionKey(s.key);
@@ -698,27 +758,9 @@ function pollSessions() {
698
758
  }),
699
759
  });
700
760
  log(`[TIMEOUT] ${session_id} missing for ${Math.round(missingFor / 60000)} min`);
701
-
702
- // Notify Michele via openclaw message (only if openclaw is available)
703
- try {
704
- if (!openclawMissing) {
705
- execSync('which openclaw', { stdio: 'ignore', timeout: 3000 });
706
- }
707
- spawnSync('openclaw', [
708
- 'message', 'send',
709
- '--account', 'ops',
710
- '--target', '297086793',
711
- '--text', `⚠️ Rev4a: agent timeout\n\`${session_id.slice(-36)}\`\nMissing for ${Math.round(missingFor / 60000)} min without completing.`,
712
- ], { encoding: 'utf8', timeout: 10_000, killSignal: 'SIGKILL' });
713
- } catch (e) {
714
- log(`[TIMEOUT] Telegram notification failed: ${e.message}`);
715
- }
716
761
  }
717
762
 
718
- // Collect system metrics after each poll
719
- try { collectSystemMetrics(); } catch (e) { log(`[METRICS ERROR] ${e.message}`); }
720
-
721
- // Force names and parents declared via lineage — overrides any previous label (including "Sub-agente")
763
+ // Force names and parents declared via lineage — overrides any previous label, including the generic placeholder older data carries
722
764
  const updateLabel = db.prepare('UPDATE sessions SET label = ?, parent_id = ? WHERE session_id = ?');
723
765
  const applyLineage = db.transaction(() => {
724
766
  for (const [child_id, agent_name] of Object.entries(declaredNames)) {
@@ -741,7 +783,7 @@ function pollSessions() {
741
783
  log(`Retention cleanup: ${r1.changes} cron sessions, ${r2.changes} cron events deleted`);
742
784
  }
743
785
 
744
- // Prune knownSessions — delete completed sessions da >30 giorni
786
+ // Prune knownSessions — delete completed sessions older than 30 days
745
787
  const cutoff = Date.now() - (30 * 24 * 60 * 60 * 1000);
746
788
  for (const [id, snap] of knownSessions) {
747
789
  if (snap.status === 'completed' && snap.updatedAt && snap.updatedAt < cutoff) {
@@ -767,17 +809,35 @@ log(`Poll interval: ${POLL_INTERVAL_MS / 1000}s`);
767
809
  pollSessions();
768
810
  const timer = setInterval(pollSessions, POLL_INTERVAL_MS);
769
811
 
812
+ // Machine metrics: the first sample a few seconds after start (the CPU figure needs a
813
+ // delta from the baseline taken at load), then every interval. Off when metrics.db failed.
814
+ function sampleMetrics() {
815
+ try { collectSystemMetrics(); } catch (e) { log(`[METRICS ERROR] ${e.message}`); }
816
+ }
817
+ const FIRST_SAMPLE_DELAY_MS = 5000;
818
+ let metricsTimer = null;
819
+ const firstSample = metrics
820
+ ? setTimeout(() => { sampleMetrics(); metricsTimer = setInterval(sampleMetrics, METRICS_INTERVAL_MS); }, FIRST_SAMPLE_DELAY_MS)
821
+ : null;
822
+ if (metrics) log(`Machine metrics every ${METRICS_INTERVAL_MS / 1000}s into ${METRICS_DB_PATH}`);
823
+
770
824
  // Graceful shutdown
771
825
  process.on('SIGTERM', () => {
772
826
  log('SIGTERM received, shutting down...');
773
827
  clearInterval(timer);
828
+ clearTimeout(firstSample);
829
+ clearInterval(metricsTimer);
774
830
  db.close();
831
+ metrics?.db.close();
775
832
  process.exit(0);
776
833
  });
777
834
 
778
835
  process.on('SIGINT', () => {
779
836
  log('SIGINT received, shutting down...');
780
837
  clearInterval(timer);
838
+ clearTimeout(firstSample);
839
+ clearInterval(metricsTimer);
781
840
  db.close();
841
+ metrics?.db.close();
782
842
  process.exit(0);
783
843
  });