@flame0510/project-aether 1.7.0 → 1.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -2
- package/app/agents/PageClient.tsx +3 -0
- package/app/agents/create/loading.tsx +51 -0
- package/app/agents/loading.tsx +2 -11
- package/app/api/assistant/route.ts +9 -2
- package/app/api/metrics/route.ts +126 -23
- package/app/api/system-health/route.ts +24 -20
- package/app/components/Sidebar.tsx +10 -0
- package/app/components/Skeleton.tsx +49 -1
- package/app/components/SystemCockpit.tsx +82 -2
- package/app/components/ui/Meter.tsx +34 -0
- package/app/components/ui/TimeSeriesChart.tsx +226 -0
- package/app/components/ui/index.ts +2 -0
- package/app/config/loading.tsx +4 -9
- package/app/containers/ContainersClient.tsx +4 -2
- package/app/containers/loading.tsx +2 -18
- package/app/containers/terminal/[id]/TerminalClient.tsx +230 -276
- package/app/containers/terminal/[id]/loading.tsx +20 -0
- package/app/crons/loading.tsx +4 -9
- package/app/gateway/loading.tsx +3 -7
- package/app/globals.css +51 -0
- package/app/lineage/loading.tsx +4 -9
- package/app/memory/loading.tsx +4 -9
- package/app/plugins/loading.tsx +4 -9
- package/app/skills/loading.tsx +4 -9
- package/app/system/PageClient.tsx +263 -0
- package/app/system/SystemSkeleton.tsx +115 -0
- package/app/system/loading.tsx +13 -0
- package/app/system/page.tsx +5 -0
- package/app/tools/loading.tsx +4 -9
- package/bin/rev4a.js +8 -4
- package/daemon.js +271 -190
- package/docs/ARCHITECTURE.md +49 -46
- package/docs/CONTAINER-TERMINAL.md +169 -261
- package/docs/DESIGN-SYSTEM.md +1 -0
- package/docs/FRONTEND-ARCHITECTURE.md +7 -2
- package/docs/REV4A.md +5 -3
- package/docs/dev/API-REFERENCE.md +61 -16
- package/docs/dev/DATABASE.md +65 -25
- package/docs/rag/DATA-FRESHNESS.md +13 -8
- package/docs/rag/GLOSSARY.md +7 -4
- package/docs/rag/REV4A-OVERVIEW.md +5 -2
- package/docs/rag/WHAT-I-CAN-ANSWER.md +1 -0
- package/lib/db-bootstrap.mjs +0 -11
- package/lib/metrics-db.ts +48 -0
- package/package.json +5 -1
- package/scripts/backup.sh +3 -3
- package/server.js +183 -0
- package/terminal-ws-server.js +329 -96
package/daemon.js
CHANGED
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
/**
|
|
3
3
|
* Rev4a Daemon — SQLite event logger
|
|
4
|
-
* Polls OpenClaw sessions every 30s, writes events to SQLite
|
|
4
|
+
* Polls OpenClaw sessions every 30s, writes events to SQLite, and samples the host
|
|
5
|
+
* machine (CPU, RAM, swap, storage) every 30s on a timer of its own.
|
|
5
6
|
*/
|
|
6
7
|
|
|
7
8
|
'use strict';
|
|
@@ -12,7 +13,13 @@ const { spawnSync, execSync } = require('child_process');
|
|
|
12
13
|
const Database = require('better-sqlite3');
|
|
13
14
|
const path = require('path');
|
|
14
15
|
|
|
15
|
-
|
|
16
|
+
// Same rule as lib/rev4a-paths.ts: REV4A_DB, else <data dir>/data/events.db, where the data
|
|
17
|
+
// dir is REV4A_DATA_DIR or ~/.config/rev4a — the API reads the files the daemon writes.
|
|
18
|
+
const DATA_DIR = process.env.REV4A_DATA_DIR
|
|
19
|
+
|| (process.platform === 'win32'
|
|
20
|
+
? path.join(process.env.APPDATA || path.join(os.homedir(), 'AppData', 'Roaming'), 'rev4a')
|
|
21
|
+
: path.join(os.homedir(), '.config', 'rev4a'));
|
|
22
|
+
const DB_PATH = process.env.REV4A_DB || path.join(DATA_DIR, 'data', 'events.db');
|
|
16
23
|
const POLL_INTERVAL_MS = 30_000;
|
|
17
24
|
|
|
18
25
|
// Cost rates per 1M tokens (separate in/out pricing)
|
|
@@ -71,20 +78,6 @@ process.on('uncaughtException', (err) => {
|
|
|
71
78
|
// Do NOT exit — the watchdog reads this from the healthcheck
|
|
72
79
|
});
|
|
73
80
|
|
|
74
|
-
db.exec(`
|
|
75
|
-
CREATE TABLE IF NOT EXISTS system_metrics (
|
|
76
|
-
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
77
|
-
ts INTEGER NOT NULL,
|
|
78
|
-
cpu_percent REAL,
|
|
79
|
-
ram_used_mb INTEGER,
|
|
80
|
-
ram_total_mb INTEGER,
|
|
81
|
-
disk_used_gb REAL,
|
|
82
|
-
disk_total_gb REAL,
|
|
83
|
-
load_avg_1m REAL
|
|
84
|
-
);
|
|
85
|
-
CREATE INDEX IF NOT EXISTS idx_metrics_ts ON system_metrics(ts);
|
|
86
|
-
`);
|
|
87
|
-
|
|
88
81
|
db.exec(`
|
|
89
82
|
CREATE TABLE IF NOT EXISTS sessions (
|
|
90
83
|
session_id TEXT PRIMARY KEY,
|
|
@@ -176,210 +169,283 @@ const knownSessions = new Map(); // session_id -> { status, tokens_in, tokens_ou
|
|
|
176
169
|
let pollCount = 0;
|
|
177
170
|
|
|
178
171
|
// ── System Metrics ───────────────────────────────────────────────────────────
|
|
172
|
+
//
|
|
173
|
+
// Machine-wide CPU, RAM, swap and storage of the host Rev4a runs on, sampled on their
|
|
174
|
+
// own timer (METRICS_INTERVAL_MS) — independent of the OpenClaw session poll, which has
|
|
175
|
+
// no source on a host without the `openclaw` CLI.
|
|
179
176
|
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
177
|
+
/**
|
|
178
|
+
* Machine metrics live in their own database next to events.db, so the event log stays
|
|
179
|
+
* sessions and events only and the history can be dropped without touching them. The
|
|
180
|
+
* daemon is its only writer; GET /api/metrics reads it (lib/metrics-db.ts, same path).
|
|
181
|
+
* Anomalies are still events, in events.db, where the live feed reads them.
|
|
182
|
+
*
|
|
183
|
+
* Opened in a try: a metrics.db that cannot be opened (corrupt, unwritable, disk full)
|
|
184
|
+
* turns sampling off with a log line and leaves the session poll running.
|
|
185
|
+
*/
|
|
186
|
+
const METRICS_DB_PATH = path.join(path.dirname(DB_PATH), 'metrics.db');
|
|
187
|
+
const metrics = openMetricsStore();
|
|
186
188
|
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
const
|
|
189
|
+
function openMetricsStore() {
|
|
190
|
+
try {
|
|
191
|
+
const mdb = new Database(METRICS_DB_PATH);
|
|
192
|
+
mdb.pragma('journal_mode = WAL');
|
|
193
|
+
mdb.pragma('synchronous = NORMAL');
|
|
194
|
+
mdb.exec(`
|
|
195
|
+
CREATE TABLE IF NOT EXISTS system_metrics (
|
|
196
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
197
|
+
ts INTEGER NOT NULL,
|
|
198
|
+
cpu_percent REAL,
|
|
199
|
+
ram_used_mb INTEGER,
|
|
200
|
+
ram_total_mb INTEGER,
|
|
201
|
+
swap_used_mb INTEGER,
|
|
202
|
+
swap_total_mb INTEGER,
|
|
203
|
+
load_avg_1m REAL,
|
|
204
|
+
load_avg_5m REAL,
|
|
205
|
+
load_avg_15m REAL
|
|
206
|
+
);
|
|
207
|
+
CREATE INDEX IF NOT EXISTS idx_metrics_ts ON system_metrics(ts);
|
|
190
208
|
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
209
|
+
CREATE TABLE IF NOT EXISTS system_disks (
|
|
210
|
+
ts INTEGER NOT NULL,
|
|
211
|
+
mount TEXT NOT NULL,
|
|
212
|
+
roles TEXT,
|
|
213
|
+
used_mb INTEGER,
|
|
214
|
+
avail_mb INTEGER,
|
|
215
|
+
total_mb INTEGER
|
|
216
|
+
);
|
|
217
|
+
CREATE INDEX IF NOT EXISTS idx_disks_ts ON system_disks(ts);
|
|
218
|
+
CREATE INDEX IF NOT EXISTS idx_disks_mount_ts ON system_disks(mount, ts);
|
|
219
|
+
`);
|
|
220
|
+
return {
|
|
221
|
+
db: mdb,
|
|
222
|
+
insertMetric: mdb.prepare(`
|
|
223
|
+
INSERT INTO system_metrics (ts, cpu_percent, ram_used_mb, ram_total_mb, swap_used_mb, swap_total_mb,
|
|
224
|
+
load_avg_1m, load_avg_5m, load_avg_15m)
|
|
225
|
+
VALUES (@ts, @cpu_percent, @ram_used_mb, @ram_total_mb, @swap_used_mb, @swap_total_mb,
|
|
226
|
+
@load_avg_1m, @load_avg_5m, @load_avg_15m)
|
|
227
|
+
`),
|
|
228
|
+
insertDisk: mdb.prepare(`
|
|
229
|
+
INSERT INTO system_disks (ts, mount, roles, used_mb, avail_mb, total_mb)
|
|
230
|
+
VALUES (@ts, @mount, @roles, @used_mb, @avail_mb, @total_mb)
|
|
231
|
+
`),
|
|
232
|
+
pruneMetrics: mdb.prepare('DELETE FROM system_metrics WHERE ts < ?'),
|
|
233
|
+
pruneDisks: mdb.prepare('DELETE FROM system_disks WHERE ts < ?'),
|
|
234
|
+
getLastTwoMetrics: mdb.prepare('SELECT cpu_percent, ram_used_mb, ram_total_mb FROM system_metrics ORDER BY ts DESC LIMIT 2'),
|
|
235
|
+
};
|
|
236
|
+
} catch (err) {
|
|
237
|
+
log(`[METRICS] disabled: cannot open ${METRICS_DB_PATH}: ${err.message}`);
|
|
238
|
+
return null;
|
|
210
239
|
}
|
|
240
|
+
}
|
|
211
241
|
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
242
|
+
const METRICS_INTERVAL_MS = 30_000;
|
|
243
|
+
const METRICS_RETENTION_S = 30 * 24 * 3600;
|
|
244
|
+
/** How long to wait before asking Docker for its root directory again after a failure. */
|
|
245
|
+
const DOCKER_ROOT_RETRY_MS = 10 * 60 * 1000;
|
|
246
|
+
|
|
247
|
+
// Anomaly cooldown: last alert per metric (cpu, ram); disks use the crossing state below
|
|
248
|
+
const lastAnomalyAlert = {};
|
|
249
|
+
const ANOMALY_COOLDOWN_MS = 5 * 60 * 1000; // 5 minutes
|
|
250
|
+
const DISK_ANOMALY_PERCENT = 90; // at or above, like the System page's critical level
|
|
251
|
+
/** Filesystems above the threshold: a disk stays full, so it is reported once per crossing (and once after a restart — in memory only). */
|
|
252
|
+
const diskAboveThreshold = new Set();
|
|
253
|
+
|
|
254
|
+
/** Host CPU ticks summed over every core (`/proc/stat` on Linux): the whole machine. */
|
|
255
|
+
function readHostCpuSample() {
|
|
256
|
+
let idle = 0;
|
|
257
|
+
let total = 0;
|
|
258
|
+
for (const cpu of os.cpus()) {
|
|
259
|
+
idle += cpu.times.idle;
|
|
260
|
+
total += cpu.times.user + cpu.times.nice + cpu.times.sys + cpu.times.irq + cpu.times.idle;
|
|
223
261
|
}
|
|
224
|
-
|
|
225
|
-
return 0;
|
|
262
|
+
return { idle, total };
|
|
226
263
|
}
|
|
227
264
|
|
|
228
|
-
|
|
229
|
-
const CGROUP_CPU_MAX = '/sys/fs/cgroup/cpu.max';
|
|
230
|
-
const CGROUP_CPUSET_EFFECTIVE = '/sys/fs/cgroup/cpuset.cpus.effective';
|
|
231
|
-
let lastCpuSample = null;
|
|
265
|
+
let lastCpuSample = readHostCpuSample();
|
|
232
266
|
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
267
|
+
/** Busy share of all cores since the previous sample, 0-100. */
|
|
268
|
+
function getCpuPercent() {
|
|
269
|
+
const sample = readHostCpuSample();
|
|
270
|
+
const deltaTotal = sample.total - lastCpuSample.total;
|
|
271
|
+
const deltaIdle = sample.idle - lastCpuSample.idle;
|
|
272
|
+
lastCpuSample = sample;
|
|
273
|
+
if (deltaTotal <= 0) return 0;
|
|
274
|
+
const busy = Math.max(0, deltaTotal - deltaIdle);
|
|
275
|
+
return Number(Math.max(0, Math.min(100, (busy / deltaTotal) * 100)).toFixed(2));
|
|
239
276
|
}
|
|
240
277
|
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
278
|
+
/**
|
|
279
|
+
* macOS (development): used = active + wired + compressed pages, close to Activity
|
|
280
|
+
* Monitor's "Memory Used" — os.freemem() there counts only free pages, so the cache
|
|
281
|
+
* would read as used and every Mac would sit near 100 %. Swap from `vm.swapusage`.
|
|
282
|
+
*/
|
|
283
|
+
function getDarwinMemory() {
|
|
284
|
+
const vm = spawnSync('vm_stat', { encoding: 'utf8', timeout: 3000 });
|
|
285
|
+
if (vm.status !== 0) return null;
|
|
286
|
+
const pageSize = Number(/page size of (\d+) bytes/.exec(vm.stdout)?.[1]);
|
|
287
|
+
const pages = (name) => Number(new RegExp(`${name}:\\s+(\\d+)`).exec(vm.stdout)?.[1] ?? NaN);
|
|
288
|
+
const used = (pages('Pages active') + pages('Pages wired down') + pages('Pages occupied by compressor')) * pageSize;
|
|
289
|
+
if (!Number.isFinite(used)) return null;
|
|
290
|
+
const swap = spawnSync('sysctl', ['-n', 'vm.swapusage'], { encoding: 'utf8', timeout: 3000 });
|
|
291
|
+
const swapMb = (label) => Number(new RegExp(`${label} = ([\\d.]+)M`).exec(swap.stdout || '')?.[1] ?? NaN);
|
|
292
|
+
const swapTotal = swapMb('total');
|
|
293
|
+
return {
|
|
294
|
+
ram_used_mb: Math.round(used / 1024 / 1024),
|
|
295
|
+
ram_total_mb: Math.round(os.totalmem() / 1024 / 1024),
|
|
296
|
+
swap_used_mb: Number.isFinite(swapTotal) ? Math.round(swapMb('used')) : null,
|
|
297
|
+
swap_total_mb: Number.isFinite(swapTotal) ? Math.round(swapTotal) : null,
|
|
298
|
+
};
|
|
260
299
|
}
|
|
261
300
|
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
}
|
|
301
|
+
/**
|
|
302
|
+
* RAM and swap in MB. On Linux from /proc/meminfo: used = MemTotal − MemAvailable, the
|
|
303
|
+
* figure `free` reports (page cache the kernel can reclaim is not "used"). On macOS see
|
|
304
|
+
* getDarwinMemory(); anywhere else os.totalmem()/os.freemem(), and no swap.
|
|
305
|
+
*/
|
|
306
|
+
function getMemory() {
|
|
307
|
+
const mb = (kb) => Math.round(kb / 1024);
|
|
308
|
+
if (process.platform === 'darwin') {
|
|
309
|
+
const mac = getDarwinMemory();
|
|
310
|
+
if (mac) return mac;
|
|
273
311
|
}
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
312
|
+
try {
|
|
313
|
+
const info = {};
|
|
314
|
+
for (const line of fs.readFileSync('/proc/meminfo', 'utf8').split('\n')) {
|
|
315
|
+
const m = /^(\w+):\s+(\d+)/.exec(line);
|
|
316
|
+
if (m) info[m[1]] = Number(m[2]);
|
|
317
|
+
}
|
|
318
|
+
if (info.MemTotal && info.MemAvailable !== undefined) {
|
|
319
|
+
return {
|
|
320
|
+
ram_used_mb: mb(info.MemTotal - info.MemAvailable),
|
|
321
|
+
ram_total_mb: mb(info.MemTotal),
|
|
322
|
+
swap_used_mb: mb((info.SwapTotal || 0) - (info.SwapFree || 0)),
|
|
323
|
+
swap_total_mb: mb(info.SwapTotal || 0),
|
|
324
|
+
};
|
|
325
|
+
}
|
|
326
|
+
} catch { /* not Linux */ }
|
|
327
|
+
const total = os.totalmem();
|
|
328
|
+
return {
|
|
329
|
+
ram_used_mb: Math.round((total - os.freemem()) / 1024 / 1024),
|
|
330
|
+
ram_total_mb: Math.round(total / 1024 / 1024),
|
|
331
|
+
swap_used_mb: null,
|
|
332
|
+
swap_total_mb: null,
|
|
333
|
+
};
|
|
279
334
|
}
|
|
280
335
|
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
336
|
+
let dockerRootDir = null;
|
|
337
|
+
let dockerRootCheckedAt = 0;
|
|
338
|
+
|
|
339
|
+
/** Docker's data root (where images and agent volumes live), asked once and cached. */
|
|
340
|
+
function getDockerRootDir() {
|
|
341
|
+
if (dockerRootDir || Date.now() - dockerRootCheckedAt < DOCKER_ROOT_RETRY_MS) return dockerRootDir;
|
|
342
|
+
dockerRootCheckedAt = Date.now();
|
|
343
|
+
const r = spawnSync('docker', ['info', '--format', '{{.DockerRootDir}}'], { encoding: 'utf8', timeout: 5000 });
|
|
344
|
+
const dir = r.status === 0 ? String(r.stdout).trim() : '';
|
|
345
|
+
if (dir.startsWith('/')) dockerRootDir = dir;
|
|
346
|
+
return dockerRootDir;
|
|
289
347
|
}
|
|
290
348
|
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
349
|
+
/**
|
|
350
|
+
* The filesystems worth watching: `/`, Docker's root and Rev4a's data directory, each
|
|
351
|
+
* filesystem once (paths on the same device are merged, their roles listed together).
|
|
352
|
+
* A path that cannot be read — Docker Desktop's root lives inside its VM — is skipped.
|
|
353
|
+
* Sizes follow `df`: used = blocks − free, avail = what a normal user can still write.
|
|
354
|
+
*/
|
|
355
|
+
function getDisks() {
|
|
356
|
+
const candidates = [
|
|
357
|
+
['/', 'root'],
|
|
358
|
+
[getDockerRootDir(), 'docker'],
|
|
359
|
+
[path.dirname(DB_PATH), 'data'],
|
|
360
|
+
];
|
|
361
|
+
const byDevice = new Map();
|
|
362
|
+
for (const [mount, role] of candidates) {
|
|
363
|
+
if (!mount) continue;
|
|
364
|
+
try {
|
|
365
|
+
const dev = fs.statSync(mount).dev;
|
|
366
|
+
const known = byDevice.get(dev);
|
|
367
|
+
if (known) { known.roles.push(role); continue; }
|
|
368
|
+
const st = fs.statfsSync(mount);
|
|
369
|
+
const mb = (blocks) => Math.round((blocks * st.bsize) / 1024 / 1024);
|
|
370
|
+
byDevice.set(dev, {
|
|
371
|
+
mount,
|
|
372
|
+
roles: [role],
|
|
373
|
+
used_mb: mb(st.blocks - st.bfree),
|
|
374
|
+
avail_mb: mb(st.bavail),
|
|
375
|
+
total_mb: mb(st.blocks),
|
|
376
|
+
});
|
|
377
|
+
} catch { /* not readable here */ }
|
|
298
378
|
}
|
|
299
|
-
return
|
|
379
|
+
return [...byDevice.values()];
|
|
300
380
|
}
|
|
301
381
|
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
382
|
+
/** Share of a filesystem in use the way `df` computes it: used / (used + avail). */
|
|
383
|
+
function diskPercent(d) {
|
|
384
|
+
const denom = d.used_mb + d.avail_mb;
|
|
385
|
+
return denom > 0 ? Math.round((d.used_mb / denom) * 100) : 0;
|
|
386
|
+
}
|
|
387
|
+
|
|
388
|
+
/** Write a system_anomaly event. `cooldownKey` limits it to one per ANOMALY_COOLDOWN_MS; null writes it now. */
|
|
389
|
+
function recordAnomaly(cooldownKey, metric, values, threshold, message) {
|
|
390
|
+
const nowMs = Date.now();
|
|
391
|
+
if (cooldownKey) {
|
|
392
|
+
if (nowMs - (lastAnomalyAlert[cooldownKey] || 0) <= ANOMALY_COOLDOWN_MS) return;
|
|
393
|
+
lastAnomalyAlert[cooldownKey] = nowMs;
|
|
312
394
|
}
|
|
395
|
+
insertEvent.run({
|
|
396
|
+
ts: nowMs,
|
|
397
|
+
session_id: 'system',
|
|
398
|
+
type: 'system_anomaly',
|
|
399
|
+
data: JSON.stringify({ metric, values, threshold, message }),
|
|
400
|
+
});
|
|
401
|
+
log(`[ANOMALY] ${message}`);
|
|
313
402
|
}
|
|
314
403
|
|
|
315
404
|
function collectSystemMetrics() {
|
|
316
405
|
const now = Math.floor(Date.now() / 1000);
|
|
317
406
|
const cpu_percent = getCpuPercent();
|
|
318
|
-
const
|
|
319
|
-
const
|
|
320
|
-
const
|
|
321
|
-
const ram_total_mb = Math.round(ram_total / 1024 / 1024);
|
|
322
|
-
const disk = getDiskStats();
|
|
323
|
-
const load_avg_1m = parseFloat(os.loadavg()[0].toFixed(2));
|
|
324
|
-
|
|
325
|
-
insertMetric.run({
|
|
326
|
-
ts: now,
|
|
327
|
-
cpu_percent,
|
|
328
|
-
ram_used_mb,
|
|
329
|
-
ram_total_mb,
|
|
330
|
-
disk_used_gb: disk.used,
|
|
331
|
-
disk_total_gb: disk.total,
|
|
332
|
-
load_avg_1m,
|
|
333
|
-
});
|
|
407
|
+
const mem = getMemory();
|
|
408
|
+
const disks = getDisks();
|
|
409
|
+
const [load1, load5, load15] = os.loadavg().map((v) => Number(v.toFixed(2)));
|
|
334
410
|
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
411
|
+
metrics.db.transaction(() => {
|
|
412
|
+
metrics.insertMetric.run({
|
|
413
|
+
ts: now,
|
|
414
|
+
cpu_percent,
|
|
415
|
+
...mem,
|
|
416
|
+
load_avg_1m: load1,
|
|
417
|
+
load_avg_5m: load5,
|
|
418
|
+
load_avg_15m: load15,
|
|
419
|
+
});
|
|
420
|
+
for (const d of disks) metrics.insertDisk.run({ ts: now, ...d, roles: d.roles.join(',') });
|
|
421
|
+
metrics.pruneMetrics.run(now - METRICS_RETENTION_S);
|
|
422
|
+
metrics.pruneDisks.run(now - METRICS_RETENTION_S);
|
|
423
|
+
})();
|
|
424
|
+
|
|
425
|
+
// Anomalies: CPU > 85 % and RAM > 90 % on two consecutive samples (with a cooldown); a
|
|
426
|
+
// filesystem once when it reaches DISK_ANOMALY_PERCENT, again only after it drops back
|
|
427
|
+
// under it — disk usage does not spike and fall back, a cooldown would repeat forever.
|
|
428
|
+
const recent = metrics.getLastTwoMetrics.all();
|
|
340
429
|
if (recent.length === 2) {
|
|
341
|
-
const nowMs = Date.now();
|
|
342
|
-
// CPU >85% for 2 consecutive samples
|
|
343
430
|
if (recent[0].cpu_percent > 85 && recent[1].cpu_percent > 85) {
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
insertEvent.run({
|
|
347
|
-
ts: Date.now(),
|
|
348
|
-
session_id: 'system',
|
|
349
|
-
type: 'system_anomaly',
|
|
350
|
-
data: JSON.stringify({
|
|
351
|
-
metric: 'cpu',
|
|
352
|
-
values: [recent[1].cpu_percent, recent[0].cpu_percent],
|
|
353
|
-
threshold: 85,
|
|
354
|
-
message: `CPU high: ${recent[0].cpu_percent}% for 2 consecutive samples`,
|
|
355
|
-
}),
|
|
356
|
-
});
|
|
357
|
-
log(`[ANOMALY] CPU high: ${recent[0].cpu_percent}%`);
|
|
358
|
-
}
|
|
431
|
+
recordAnomaly('cpu', 'cpu', [recent[1].cpu_percent, recent[0].cpu_percent], 85,
|
|
432
|
+
`CPU high: ${recent[0].cpu_percent}% for 2 consecutive samples`);
|
|
359
433
|
}
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
threshold: 90,
|
|
374
|
-
message: `RAM high: ${ram0pct}% for 2 consecutive samples`,
|
|
375
|
-
}),
|
|
376
|
-
});
|
|
377
|
-
log(`[ANOMALY] RAM high: ${ram0pct}%`);
|
|
378
|
-
}
|
|
434
|
+
const pct = (r) => (r.ram_total_mb > 0 ? Math.round((r.ram_used_mb / r.ram_total_mb) * 100) : 0);
|
|
435
|
+
if (pct(recent[0]) > 90 && pct(recent[1]) > 90) {
|
|
436
|
+
recordAnomaly('ram', 'ram', [pct(recent[1]), pct(recent[0])], 90,
|
|
437
|
+
`RAM high: ${pct(recent[0])}% for 2 consecutive samples`);
|
|
438
|
+
}
|
|
439
|
+
}
|
|
440
|
+
for (const d of disks) {
|
|
441
|
+
const p = diskPercent(d);
|
|
442
|
+
if (p < DISK_ANOMALY_PERCENT) {
|
|
443
|
+
diskAboveThreshold.delete(d.mount);
|
|
444
|
+
} else if (!diskAboveThreshold.has(d.mount)) {
|
|
445
|
+
diskAboveThreshold.add(d.mount);
|
|
446
|
+
recordAnomaly(null, 'disk', [p], DISK_ANOMALY_PERCENT, `Disk high: ${d.mount} ${p}% used`);
|
|
379
447
|
}
|
|
380
448
|
}
|
|
381
|
-
|
|
382
|
-
log(`[METRICS] CPU:${cpu_percent}% RAM:${ram_used_mb}/${ram_total_mb}MB Disk:${disk.used}/${disk.total}GB Load:${load_avg_1m}`);
|
|
383
449
|
}
|
|
384
450
|
|
|
385
451
|
// ── Helpers ─────────────────────────────────────────────────────────────────
|
|
@@ -694,9 +760,6 @@ function pollSessions() {
|
|
|
694
760
|
log(`[TIMEOUT] ${session_id} missing for ${Math.round(missingFor / 60000)} min`);
|
|
695
761
|
}
|
|
696
762
|
|
|
697
|
-
// Collect system metrics after each poll
|
|
698
|
-
try { collectSystemMetrics(); } catch (e) { log(`[METRICS ERROR] ${e.message}`); }
|
|
699
|
-
|
|
700
763
|
// Force names and parents declared via lineage — overrides any previous label, including the generic placeholder older data carries
|
|
701
764
|
const updateLabel = db.prepare('UPDATE sessions SET label = ?, parent_id = ? WHERE session_id = ?');
|
|
702
765
|
const applyLineage = db.transaction(() => {
|
|
@@ -746,17 +809,35 @@ log(`Poll interval: ${POLL_INTERVAL_MS / 1000}s`);
|
|
|
746
809
|
pollSessions();
|
|
747
810
|
const timer = setInterval(pollSessions, POLL_INTERVAL_MS);
|
|
748
811
|
|
|
812
|
+
// Machine metrics: the first sample a few seconds after start (the CPU figure needs a
|
|
813
|
+
// delta from the baseline taken at load), then every interval. Off when metrics.db failed.
|
|
814
|
+
function sampleMetrics() {
|
|
815
|
+
try { collectSystemMetrics(); } catch (e) { log(`[METRICS ERROR] ${e.message}`); }
|
|
816
|
+
}
|
|
817
|
+
const FIRST_SAMPLE_DELAY_MS = 5000;
|
|
818
|
+
let metricsTimer = null;
|
|
819
|
+
const firstSample = metrics
|
|
820
|
+
? setTimeout(() => { sampleMetrics(); metricsTimer = setInterval(sampleMetrics, METRICS_INTERVAL_MS); }, FIRST_SAMPLE_DELAY_MS)
|
|
821
|
+
: null;
|
|
822
|
+
if (metrics) log(`Machine metrics every ${METRICS_INTERVAL_MS / 1000}s into ${METRICS_DB_PATH}`);
|
|
823
|
+
|
|
749
824
|
// Graceful shutdown
|
|
750
825
|
process.on('SIGTERM', () => {
|
|
751
826
|
log('SIGTERM received, shutting down...');
|
|
752
827
|
clearInterval(timer);
|
|
828
|
+
clearTimeout(firstSample);
|
|
829
|
+
clearInterval(metricsTimer);
|
|
753
830
|
db.close();
|
|
831
|
+
metrics?.db.close();
|
|
754
832
|
process.exit(0);
|
|
755
833
|
});
|
|
756
834
|
|
|
757
835
|
process.on('SIGINT', () => {
|
|
758
836
|
log('SIGINT received, shutting down...');
|
|
759
837
|
clearInterval(timer);
|
|
838
|
+
clearTimeout(firstSample);
|
|
839
|
+
clearInterval(metricsTimer);
|
|
760
840
|
db.close();
|
|
841
|
+
metrics?.db.close();
|
|
761
842
|
process.exit(0);
|
|
762
843
|
});
|