@flame0510/project-aether 1.6.2 → 1.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +6 -2
- package/agent-templates/atlas/HEARTBEAT.md +1 -1
- package/app/agents/ImageDownloadBanner.tsx +171 -37
- package/app/api/agents/download-image/route.ts +29 -5
- package/app/api/agents/image-status/route.ts +17 -2
- package/app/api/assistant/route.ts +21 -5
- package/app/api/auth/login/route.ts +2 -2
- package/app/api/metrics/route.ts +126 -23
- package/app/api/setup/agent-image/route.ts +6 -4
- package/app/api/system-health/route.ts +25 -21
- package/app/components/DashboardToolbar.tsx +2 -2
- package/app/components/LineageGraphPage.tsx +4 -4
- package/app/components/SessionDrawer.tsx +8 -8
- package/app/components/Sidebar.tsx +10 -0
- package/app/components/Skeleton.tsx +4 -1
- package/app/components/SystemCockpit.tsx +83 -3
- package/app/components/ui/Meter.tsx +34 -0
- package/app/components/ui/TimeSeriesChart.tsx +226 -0
- package/app/components/ui/index.ts +2 -0
- package/app/globals.css +42 -1
- package/app/setup/PageClient.tsx +1 -1
- package/app/system/PageClient.tsx +263 -0
- package/app/system/SystemSkeleton.tsx +115 -0
- package/app/system/loading.tsx +13 -0
- package/app/system/page.tsx +5 -0
- package/bin/postinstall.js +5 -1
- package/daemon.js +274 -214
- package/docs/ARCHITECTURE.md +65 -34
- package/docs/CONTAINER-TERMINAL.md +17 -8
- package/docs/DESIGN-SYSTEM.md +22 -12
- package/docs/FRONTEND-ARCHITECTURE.md +8 -4
- package/docs/REV4A.md +37 -86
- package/docs/dev/API-REFERENCE.md +102 -19
- package/docs/dev/DATABASE.md +79 -28
- package/docs/dev/SESSION-MAINTENANCE-PLAN.md +6 -6
- package/docs/rag/DATA-FRESHNESS.md +31 -15
- package/docs/rag/GLOSSARY.md +8 -5
- package/docs/rag/REV4A-OVERVIEW.md +14 -7
- package/docs/rag/WHAT-I-CAN-ANSWER.md +4 -3
- package/lib/agent-images.ts +43 -14
- package/lib/buildAgentImage.ts +142 -6
- package/lib/db-bootstrap.mjs +0 -11
- package/lib/metrics-db.ts +48 -0
- package/lib/patterns/sessionPresentation.ts +4 -2
- package/lib/rev4a-auth.d.ts +1 -0
- package/lib/rev4a-auth.js +18 -2
- package/next.config.mjs +9 -1
- package/package.json +2 -2
- package/scripts/backup.sh +48 -54
- package/scripts/check-language.mjs +21 -3
- package/scripts/restore.sh +77 -59
package/daemon.js
CHANGED
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
/**
|
|
3
3
|
* Rev4a Daemon — SQLite event logger
|
|
4
|
-
* Polls OpenClaw sessions every 30s, writes events to SQLite
|
|
4
|
+
* Polls OpenClaw sessions every 30s, writes events to SQLite, and samples the host
|
|
5
|
+
* machine (CPU, RAM, swap, storage) every 30s on a timer of its own.
|
|
5
6
|
*/
|
|
6
7
|
|
|
7
8
|
'use strict';
|
|
@@ -12,9 +13,14 @@ const { spawnSync, execSync } = require('child_process');
|
|
|
12
13
|
const Database = require('better-sqlite3');
|
|
13
14
|
const path = require('path');
|
|
14
15
|
|
|
15
|
-
|
|
16
|
+
// Same rule as lib/rev4a-paths.ts: REV4A_DB, else <data dir>/data/events.db, where the data
|
|
17
|
+
// dir is REV4A_DATA_DIR or ~/.config/rev4a — the API reads the files the daemon writes.
|
|
18
|
+
const DATA_DIR = process.env.REV4A_DATA_DIR
|
|
19
|
+
|| (process.platform === 'win32'
|
|
20
|
+
? path.join(process.env.APPDATA || path.join(os.homedir(), 'AppData', 'Roaming'), 'rev4a')
|
|
21
|
+
: path.join(os.homedir(), '.config', 'rev4a'));
|
|
22
|
+
const DB_PATH = process.env.REV4A_DB || path.join(DATA_DIR, 'data', 'events.db');
|
|
16
23
|
const POLL_INTERVAL_MS = 30_000;
|
|
17
|
-
const POLL_INTERVAL_ACTIVE_MS = 15_000; // 15s when working sessions detected
|
|
18
24
|
|
|
19
25
|
// Cost rates per 1M tokens (separate in/out pricing)
|
|
20
26
|
const MODEL_PRICING = {
|
|
@@ -72,20 +78,6 @@ process.on('uncaughtException', (err) => {
|
|
|
72
78
|
// Do NOT exit — the watchdog reads this from the healthcheck
|
|
73
79
|
});
|
|
74
80
|
|
|
75
|
-
db.exec(`
|
|
76
|
-
CREATE TABLE IF NOT EXISTS system_metrics (
|
|
77
|
-
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
78
|
-
ts INTEGER NOT NULL,
|
|
79
|
-
cpu_percent REAL,
|
|
80
|
-
ram_used_mb INTEGER,
|
|
81
|
-
ram_total_mb INTEGER,
|
|
82
|
-
disk_used_gb REAL,
|
|
83
|
-
disk_total_gb REAL,
|
|
84
|
-
load_avg_1m REAL
|
|
85
|
-
);
|
|
86
|
-
CREATE INDEX IF NOT EXISTS idx_metrics_ts ON system_metrics(ts);
|
|
87
|
-
`);
|
|
88
|
-
|
|
89
81
|
db.exec(`
|
|
90
82
|
CREATE TABLE IF NOT EXISTS sessions (
|
|
91
83
|
session_id TEXT PRIMARY KEY,
|
|
@@ -177,215 +169,283 @@ const knownSessions = new Map(); // session_id -> { status, tokens_in, tokens_ou
|
|
|
177
169
|
let pollCount = 0;
|
|
178
170
|
|
|
179
171
|
// ── System Metrics ───────────────────────────────────────────────────────────
|
|
172
|
+
//
|
|
173
|
+
// Machine-wide CPU, RAM, swap and storage of the host Rev4a runs on, sampled on their
|
|
174
|
+
// own timer (METRICS_INTERVAL_MS) — independent of the OpenClaw session poll, which has
|
|
175
|
+
// no source on a host without the `openclaw` CLI.
|
|
180
176
|
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
177
|
+
/**
|
|
178
|
+
* Machine metrics live in their own database next to events.db, so the event log stays
|
|
179
|
+
* sessions and events only and the history can be dropped without touching them. The
|
|
180
|
+
* daemon is its only writer; GET /api/metrics reads it (lib/metrics-db.ts, same path).
|
|
181
|
+
* Anomalies are still events, in events.db, where the live feed reads them.
|
|
182
|
+
*
|
|
183
|
+
* Opened in a try: a metrics.db that cannot be opened (corrupt, unwritable, disk full)
|
|
184
|
+
* turns sampling off with a log line and leaves the session poll running.
|
|
185
|
+
*/
|
|
186
|
+
const METRICS_DB_PATH = path.join(path.dirname(DB_PATH), 'metrics.db');
|
|
187
|
+
const metrics = openMetricsStore();
|
|
187
188
|
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
const
|
|
189
|
+
function openMetricsStore() {
|
|
190
|
+
try {
|
|
191
|
+
const mdb = new Database(METRICS_DB_PATH);
|
|
192
|
+
mdb.pragma('journal_mode = WAL');
|
|
193
|
+
mdb.pragma('synchronous = NORMAL');
|
|
194
|
+
mdb.exec(`
|
|
195
|
+
CREATE TABLE IF NOT EXISTS system_metrics (
|
|
196
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
197
|
+
ts INTEGER NOT NULL,
|
|
198
|
+
cpu_percent REAL,
|
|
199
|
+
ram_used_mb INTEGER,
|
|
200
|
+
ram_total_mb INTEGER,
|
|
201
|
+
swap_used_mb INTEGER,
|
|
202
|
+
swap_total_mb INTEGER,
|
|
203
|
+
load_avg_1m REAL,
|
|
204
|
+
load_avg_5m REAL,
|
|
205
|
+
load_avg_15m REAL
|
|
206
|
+
);
|
|
207
|
+
CREATE INDEX IF NOT EXISTS idx_metrics_ts ON system_metrics(ts);
|
|
191
208
|
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
209
|
+
CREATE TABLE IF NOT EXISTS system_disks (
|
|
210
|
+
ts INTEGER NOT NULL,
|
|
211
|
+
mount TEXT NOT NULL,
|
|
212
|
+
roles TEXT,
|
|
213
|
+
used_mb INTEGER,
|
|
214
|
+
avail_mb INTEGER,
|
|
215
|
+
total_mb INTEGER
|
|
216
|
+
);
|
|
217
|
+
CREATE INDEX IF NOT EXISTS idx_disks_ts ON system_disks(ts);
|
|
218
|
+
CREATE INDEX IF NOT EXISTS idx_disks_mount_ts ON system_disks(mount, ts);
|
|
219
|
+
`);
|
|
220
|
+
return {
|
|
221
|
+
db: mdb,
|
|
222
|
+
insertMetric: mdb.prepare(`
|
|
223
|
+
INSERT INTO system_metrics (ts, cpu_percent, ram_used_mb, ram_total_mb, swap_used_mb, swap_total_mb,
|
|
224
|
+
load_avg_1m, load_avg_5m, load_avg_15m)
|
|
225
|
+
VALUES (@ts, @cpu_percent, @ram_used_mb, @ram_total_mb, @swap_used_mb, @swap_total_mb,
|
|
226
|
+
@load_avg_1m, @load_avg_5m, @load_avg_15m)
|
|
227
|
+
`),
|
|
228
|
+
insertDisk: mdb.prepare(`
|
|
229
|
+
INSERT INTO system_disks (ts, mount, roles, used_mb, avail_mb, total_mb)
|
|
230
|
+
VALUES (@ts, @mount, @roles, @used_mb, @avail_mb, @total_mb)
|
|
231
|
+
`),
|
|
232
|
+
pruneMetrics: mdb.prepare('DELETE FROM system_metrics WHERE ts < ?'),
|
|
233
|
+
pruneDisks: mdb.prepare('DELETE FROM system_disks WHERE ts < ?'),
|
|
234
|
+
getLastTwoMetrics: mdb.prepare('SELECT cpu_percent, ram_used_mb, ram_total_mb FROM system_metrics ORDER BY ts DESC LIMIT 2'),
|
|
235
|
+
};
|
|
236
|
+
} catch (err) {
|
|
237
|
+
log(`[METRICS] disabled: cannot open ${METRICS_DB_PATH}: ${err.message}`);
|
|
238
|
+
return null;
|
|
211
239
|
}
|
|
240
|
+
}
|
|
212
241
|
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
242
|
+
const METRICS_INTERVAL_MS = 30_000;
|
|
243
|
+
const METRICS_RETENTION_S = 30 * 24 * 3600;
|
|
244
|
+
/** How long to wait before asking Docker for its root directory again after a failure. */
|
|
245
|
+
const DOCKER_ROOT_RETRY_MS = 10 * 60 * 1000;
|
|
246
|
+
|
|
247
|
+
// Anomaly cooldown: last alert per metric (cpu, ram); disks use the crossing state below
|
|
248
|
+
const lastAnomalyAlert = {};
|
|
249
|
+
const ANOMALY_COOLDOWN_MS = 5 * 60 * 1000; // 5 minutes
|
|
250
|
+
const DISK_ANOMALY_PERCENT = 90; // at or above, like the System page's critical level
|
|
251
|
+
/** Filesystems above the threshold: a disk stays full, so it is reported once per crossing (and once after a restart — in memory only). */
|
|
252
|
+
const diskAboveThreshold = new Set();
|
|
253
|
+
|
|
254
|
+
/** Host CPU ticks summed over every core (`/proc/stat` on Linux): the whole machine. */
|
|
255
|
+
function readHostCpuSample() {
|
|
256
|
+
let idle = 0;
|
|
257
|
+
let total = 0;
|
|
258
|
+
for (const cpu of os.cpus()) {
|
|
259
|
+
idle += cpu.times.idle;
|
|
260
|
+
total += cpu.times.user + cpu.times.nice + cpu.times.sys + cpu.times.irq + cpu.times.idle;
|
|
224
261
|
}
|
|
225
|
-
|
|
226
|
-
return 0;
|
|
262
|
+
return { idle, total };
|
|
227
263
|
}
|
|
228
264
|
|
|
229
|
-
|
|
230
|
-
const CGROUP_CPU_MAX = '/sys/fs/cgroup/cpu.max';
|
|
231
|
-
const CGROUP_CPUSET_EFFECTIVE = '/sys/fs/cgroup/cpuset.cpus.effective';
|
|
232
|
-
let lastCpuSample = null;
|
|
265
|
+
let lastCpuSample = readHostCpuSample();
|
|
233
266
|
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
267
|
+
/** Busy share of all cores since the previous sample, 0-100. */
|
|
268
|
+
function getCpuPercent() {
|
|
269
|
+
const sample = readHostCpuSample();
|
|
270
|
+
const deltaTotal = sample.total - lastCpuSample.total;
|
|
271
|
+
const deltaIdle = sample.idle - lastCpuSample.idle;
|
|
272
|
+
lastCpuSample = sample;
|
|
273
|
+
if (deltaTotal <= 0) return 0;
|
|
274
|
+
const busy = Math.max(0, deltaTotal - deltaIdle);
|
|
275
|
+
return Number(Math.max(0, Math.min(100, (busy / deltaTotal) * 100)).toFixed(2));
|
|
240
276
|
}
|
|
241
277
|
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
278
|
+
/**
|
|
279
|
+
* macOS (development): used = active + wired + compressed pages, close to Activity
|
|
280
|
+
* Monitor's "Memory Used" — os.freemem() there counts only free pages, so the cache
|
|
281
|
+
* would read as used and every Mac would sit near 100 %. Swap from `vm.swapusage`.
|
|
282
|
+
*/
|
|
283
|
+
function getDarwinMemory() {
|
|
284
|
+
const vm = spawnSync('vm_stat', { encoding: 'utf8', timeout: 3000 });
|
|
285
|
+
if (vm.status !== 0) return null;
|
|
286
|
+
const pageSize = Number(/page size of (\d+) bytes/.exec(vm.stdout)?.[1]);
|
|
287
|
+
const pages = (name) => Number(new RegExp(`${name}:\\s+(\\d+)`).exec(vm.stdout)?.[1] ?? NaN);
|
|
288
|
+
const used = (pages('Pages active') + pages('Pages wired down') + pages('Pages occupied by compressor')) * pageSize;
|
|
289
|
+
if (!Number.isFinite(used)) return null;
|
|
290
|
+
const swap = spawnSync('sysctl', ['-n', 'vm.swapusage'], { encoding: 'utf8', timeout: 3000 });
|
|
291
|
+
const swapMb = (label) => Number(new RegExp(`${label} = ([\\d.]+)M`).exec(swap.stdout || '')?.[1] ?? NaN);
|
|
292
|
+
const swapTotal = swapMb('total');
|
|
293
|
+
return {
|
|
294
|
+
ram_used_mb: Math.round(used / 1024 / 1024),
|
|
295
|
+
ram_total_mb: Math.round(os.totalmem() / 1024 / 1024),
|
|
296
|
+
swap_used_mb: Number.isFinite(swapTotal) ? Math.round(swapMb('used')) : null,
|
|
297
|
+
swap_total_mb: Number.isFinite(swapTotal) ? Math.round(swapTotal) : null,
|
|
298
|
+
};
|
|
261
299
|
}
|
|
262
300
|
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
}
|
|
301
|
+
/**
|
|
302
|
+
* RAM and swap in MB. On Linux from /proc/meminfo: used = MemTotal − MemAvailable, the
|
|
303
|
+
* figure `free` reports (page cache the kernel can reclaim is not "used"). On macOS see
|
|
304
|
+
* getDarwinMemory(); anywhere else os.totalmem()/os.freemem(), and no swap.
|
|
305
|
+
*/
|
|
306
|
+
function getMemory() {
|
|
307
|
+
const mb = (kb) => Math.round(kb / 1024);
|
|
308
|
+
if (process.platform === 'darwin') {
|
|
309
|
+
const mac = getDarwinMemory();
|
|
310
|
+
if (mac) return mac;
|
|
274
311
|
}
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
312
|
+
try {
|
|
313
|
+
const info = {};
|
|
314
|
+
for (const line of fs.readFileSync('/proc/meminfo', 'utf8').split('\n')) {
|
|
315
|
+
const m = /^(\w+):\s+(\d+)/.exec(line);
|
|
316
|
+
if (m) info[m[1]] = Number(m[2]);
|
|
317
|
+
}
|
|
318
|
+
if (info.MemTotal && info.MemAvailable !== undefined) {
|
|
319
|
+
return {
|
|
320
|
+
ram_used_mb: mb(info.MemTotal - info.MemAvailable),
|
|
321
|
+
ram_total_mb: mb(info.MemTotal),
|
|
322
|
+
swap_used_mb: mb((info.SwapTotal || 0) - (info.SwapFree || 0)),
|
|
323
|
+
swap_total_mb: mb(info.SwapTotal || 0),
|
|
324
|
+
};
|
|
325
|
+
}
|
|
326
|
+
} catch { /* not Linux */ }
|
|
327
|
+
const total = os.totalmem();
|
|
328
|
+
return {
|
|
329
|
+
ram_used_mb: Math.round((total - os.freemem()) / 1024 / 1024),
|
|
330
|
+
ram_total_mb: Math.round(total / 1024 / 1024),
|
|
331
|
+
swap_used_mb: null,
|
|
332
|
+
swap_total_mb: null,
|
|
333
|
+
};
|
|
280
334
|
}
|
|
281
335
|
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
336
|
+
let dockerRootDir = null;
|
|
337
|
+
let dockerRootCheckedAt = 0;
|
|
338
|
+
|
|
339
|
+
/** Docker's data root (where images and agent volumes live), asked once and cached. */
|
|
340
|
+
function getDockerRootDir() {
|
|
341
|
+
if (dockerRootDir || Date.now() - dockerRootCheckedAt < DOCKER_ROOT_RETRY_MS) return dockerRootDir;
|
|
342
|
+
dockerRootCheckedAt = Date.now();
|
|
343
|
+
const r = spawnSync('docker', ['info', '--format', '{{.DockerRootDir}}'], { encoding: 'utf8', timeout: 5000 });
|
|
344
|
+
const dir = r.status === 0 ? String(r.stdout).trim() : '';
|
|
345
|
+
if (dir.startsWith('/')) dockerRootDir = dir;
|
|
346
|
+
return dockerRootDir;
|
|
290
347
|
}
|
|
291
348
|
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
349
|
+
/**
|
|
350
|
+
* The filesystems worth watching: `/`, Docker's root and Rev4a's data directory, each
|
|
351
|
+
* filesystem once (paths on the same device are merged, their roles listed together).
|
|
352
|
+
* A path that cannot be read — Docker Desktop's root lives inside its VM — is skipped.
|
|
353
|
+
* Sizes follow `df`: used = blocks − free, avail = what a normal user can still write.
|
|
354
|
+
*/
|
|
355
|
+
function getDisks() {
|
|
356
|
+
const candidates = [
|
|
357
|
+
['/', 'root'],
|
|
358
|
+
[getDockerRootDir(), 'docker'],
|
|
359
|
+
[path.dirname(DB_PATH), 'data'],
|
|
360
|
+
];
|
|
361
|
+
const byDevice = new Map();
|
|
362
|
+
for (const [mount, role] of candidates) {
|
|
363
|
+
if (!mount) continue;
|
|
364
|
+
try {
|
|
365
|
+
const dev = fs.statSync(mount).dev;
|
|
366
|
+
const known = byDevice.get(dev);
|
|
367
|
+
if (known) { known.roles.push(role); continue; }
|
|
368
|
+
const st = fs.statfsSync(mount);
|
|
369
|
+
const mb = (blocks) => Math.round((blocks * st.bsize) / 1024 / 1024);
|
|
370
|
+
byDevice.set(dev, {
|
|
371
|
+
mount,
|
|
372
|
+
roles: [role],
|
|
373
|
+
used_mb: mb(st.blocks - st.bfree),
|
|
374
|
+
avail_mb: mb(st.bavail),
|
|
375
|
+
total_mb: mb(st.blocks),
|
|
376
|
+
});
|
|
377
|
+
} catch { /* not readable here */ }
|
|
299
378
|
}
|
|
300
|
-
return
|
|
379
|
+
return [...byDevice.values()];
|
|
301
380
|
}
|
|
302
381
|
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
return { used, total };
|
|
316
|
-
} catch (e) {
|
|
317
|
-
return { used: 0, total: 0 };
|
|
382
|
+
/** Share of a filesystem in use the way `df` computes it: used / (used + avail). */
|
|
383
|
+
function diskPercent(d) {
|
|
384
|
+
const denom = d.used_mb + d.avail_mb;
|
|
385
|
+
return denom > 0 ? Math.round((d.used_mb / denom) * 100) : 0;
|
|
386
|
+
}
|
|
387
|
+
|
|
388
|
+
/** Write a system_anomaly event. `cooldownKey` limits it to one per ANOMALY_COOLDOWN_MS; null writes it now. */
|
|
389
|
+
function recordAnomaly(cooldownKey, metric, values, threshold, message) {
|
|
390
|
+
const nowMs = Date.now();
|
|
391
|
+
if (cooldownKey) {
|
|
392
|
+
if (nowMs - (lastAnomalyAlert[cooldownKey] || 0) <= ANOMALY_COOLDOWN_MS) return;
|
|
393
|
+
lastAnomalyAlert[cooldownKey] = nowMs;
|
|
318
394
|
}
|
|
395
|
+
insertEvent.run({
|
|
396
|
+
ts: nowMs,
|
|
397
|
+
session_id: 'system',
|
|
398
|
+
type: 'system_anomaly',
|
|
399
|
+
data: JSON.stringify({ metric, values, threshold, message }),
|
|
400
|
+
});
|
|
401
|
+
log(`[ANOMALY] ${message}`);
|
|
319
402
|
}
|
|
320
403
|
|
|
321
404
|
function collectSystemMetrics() {
|
|
322
405
|
const now = Math.floor(Date.now() / 1000);
|
|
323
406
|
const cpu_percent = getCpuPercent();
|
|
324
|
-
const
|
|
325
|
-
const
|
|
326
|
-
const
|
|
327
|
-
const ram_total_mb = Math.round(ram_total / 1024 / 1024);
|
|
328
|
-
const disk = getDiskStats();
|
|
329
|
-
const load_avg_1m = parseFloat(os.loadavg()[0].toFixed(2));
|
|
330
|
-
|
|
331
|
-
insertMetric.run({
|
|
332
|
-
ts: now,
|
|
333
|
-
cpu_percent,
|
|
334
|
-
ram_used_mb,
|
|
335
|
-
ram_total_mb,
|
|
336
|
-
disk_used_gb: disk.used,
|
|
337
|
-
disk_total_gb: disk.total,
|
|
338
|
-
load_avg_1m,
|
|
339
|
-
});
|
|
407
|
+
const mem = getMemory();
|
|
408
|
+
const disks = getDisks();
|
|
409
|
+
const [load1, load5, load15] = os.loadavg().map((v) => Number(v.toFixed(2)));
|
|
340
410
|
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
411
|
+
metrics.db.transaction(() => {
|
|
412
|
+
metrics.insertMetric.run({
|
|
413
|
+
ts: now,
|
|
414
|
+
cpu_percent,
|
|
415
|
+
...mem,
|
|
416
|
+
load_avg_1m: load1,
|
|
417
|
+
load_avg_5m: load5,
|
|
418
|
+
load_avg_15m: load15,
|
|
419
|
+
});
|
|
420
|
+
for (const d of disks) metrics.insertDisk.run({ ts: now, ...d, roles: d.roles.join(',') });
|
|
421
|
+
metrics.pruneMetrics.run(now - METRICS_RETENTION_S);
|
|
422
|
+
metrics.pruneDisks.run(now - METRICS_RETENTION_S);
|
|
423
|
+
})();
|
|
424
|
+
|
|
425
|
+
// Anomalies: CPU > 85 % and RAM > 90 % on two consecutive samples (with a cooldown); a
|
|
426
|
+
// filesystem once when it reaches DISK_ANOMALY_PERCENT, again only after it drops back
|
|
427
|
+
// under it — disk usage does not spike and fall back, a cooldown would repeat forever.
|
|
428
|
+
const recent = metrics.getLastTwoMetrics.all();
|
|
346
429
|
if (recent.length === 2) {
|
|
347
|
-
const nowMs = Date.now();
|
|
348
|
-
// CPU >85% for 2 consecutive samples
|
|
349
430
|
if (recent[0].cpu_percent > 85 && recent[1].cpu_percent > 85) {
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
insertEvent.run({
|
|
353
|
-
ts: Date.now(),
|
|
354
|
-
session_id: 'system',
|
|
355
|
-
type: 'system_anomaly',
|
|
356
|
-
data: JSON.stringify({
|
|
357
|
-
metric: 'cpu',
|
|
358
|
-
values: [recent[1].cpu_percent, recent[0].cpu_percent],
|
|
359
|
-
threshold: 85,
|
|
360
|
-
message: `CPU alta: ${recent[0].cpu_percent}% per 2 campioni consecutivi`,
|
|
361
|
-
}),
|
|
362
|
-
});
|
|
363
|
-
log(`[ANOMALY] CPU alta: ${recent[0].cpu_percent}%`);
|
|
364
|
-
}
|
|
431
|
+
recordAnomaly('cpu', 'cpu', [recent[1].cpu_percent, recent[0].cpu_percent], 85,
|
|
432
|
+
`CPU high: ${recent[0].cpu_percent}% for 2 consecutive samples`);
|
|
365
433
|
}
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
threshold: 90,
|
|
380
|
-
message: `RAM alta: ${ram0pct}% per 2 campioni consecutivi`,
|
|
381
|
-
}),
|
|
382
|
-
});
|
|
383
|
-
log(`[ANOMALY] RAM alta: ${ram0pct}%`);
|
|
384
|
-
}
|
|
434
|
+
const pct = (r) => (r.ram_total_mb > 0 ? Math.round((r.ram_used_mb / r.ram_total_mb) * 100) : 0);
|
|
435
|
+
if (pct(recent[0]) > 90 && pct(recent[1]) > 90) {
|
|
436
|
+
recordAnomaly('ram', 'ram', [pct(recent[1]), pct(recent[0])], 90,
|
|
437
|
+
`RAM high: ${pct(recent[0])}% for 2 consecutive samples`);
|
|
438
|
+
}
|
|
439
|
+
}
|
|
440
|
+
for (const d of disks) {
|
|
441
|
+
const p = diskPercent(d);
|
|
442
|
+
if (p < DISK_ANOMALY_PERCENT) {
|
|
443
|
+
diskAboveThreshold.delete(d.mount);
|
|
444
|
+
} else if (!diskAboveThreshold.has(d.mount)) {
|
|
445
|
+
diskAboveThreshold.add(d.mount);
|
|
446
|
+
recordAnomaly(null, 'disk', [p], DISK_ANOMALY_PERCENT, `Disk high: ${d.mount} ${p}% used`);
|
|
385
447
|
}
|
|
386
448
|
}
|
|
387
|
-
|
|
388
|
-
log(`[METRICS] CPU:${cpu_percent}% RAM:${ram_used_mb}/${ram_total_mb}MB Disk:${disk.used}/${disk.total}GB Load:${load_avg_1m}`);
|
|
389
449
|
}
|
|
390
450
|
|
|
391
451
|
// ── Helpers ─────────────────────────────────────────────────────────────────
|
|
@@ -540,7 +600,7 @@ function pollSessions() {
|
|
|
540
600
|
const session_id = s.key;
|
|
541
601
|
|
|
542
602
|
// Skip Telegram channel/group sessions (multi-user).
|
|
543
|
-
// Keep Telegram direct sessions (agent
|
|
603
|
+
// Keep Telegram direct sessions (agent:<id>:telegram:<account>:direct:...) as root nodes
|
|
544
604
|
// since they are the parent of all sub-agents spawned via Telegram.
|
|
545
605
|
if (session_id.includes(':telegram:') && !session_id.includes(':direct:')) continue;
|
|
546
606
|
const { label, parent_id: inferredParent } = parseSessionKey(s.key);
|
|
@@ -698,27 +758,9 @@ function pollSessions() {
|
|
|
698
758
|
}),
|
|
699
759
|
});
|
|
700
760
|
log(`[TIMEOUT] ${session_id} missing for ${Math.round(missingFor / 60000)} min`);
|
|
701
|
-
|
|
702
|
-
// Notify Michele via openclaw message (only if openclaw is available)
|
|
703
|
-
try {
|
|
704
|
-
if (!openclawMissing) {
|
|
705
|
-
execSync('which openclaw', { stdio: 'ignore', timeout: 3000 });
|
|
706
|
-
}
|
|
707
|
-
spawnSync('openclaw', [
|
|
708
|
-
'message', 'send',
|
|
709
|
-
'--account', 'ops',
|
|
710
|
-
'--target', '297086793',
|
|
711
|
-
'--text', `⚠️ Rev4a: agent timeout\n\`${session_id.slice(-36)}\`\nMissing for ${Math.round(missingFor / 60000)} min without completing.`,
|
|
712
|
-
], { encoding: 'utf8', timeout: 10_000, killSignal: 'SIGKILL' });
|
|
713
|
-
} catch (e) {
|
|
714
|
-
log(`[TIMEOUT] Telegram notification failed: ${e.message}`);
|
|
715
|
-
}
|
|
716
761
|
}
|
|
717
762
|
|
|
718
|
-
//
|
|
719
|
-
try { collectSystemMetrics(); } catch (e) { log(`[METRICS ERROR] ${e.message}`); }
|
|
720
|
-
|
|
721
|
-
// Force names and parents declared via lineage — overrides any previous label (including "Sub-agente")
|
|
763
|
+
// Force names and parents declared via lineage — overrides any previous label, including the generic placeholder older data carries
|
|
722
764
|
const updateLabel = db.prepare('UPDATE sessions SET label = ?, parent_id = ? WHERE session_id = ?');
|
|
723
765
|
const applyLineage = db.transaction(() => {
|
|
724
766
|
for (const [child_id, agent_name] of Object.entries(declaredNames)) {
|
|
@@ -741,7 +783,7 @@ function pollSessions() {
|
|
|
741
783
|
log(`Retention cleanup: ${r1.changes} cron sessions, ${r2.changes} cron events deleted`);
|
|
742
784
|
}
|
|
743
785
|
|
|
744
|
-
// Prune knownSessions — delete completed sessions
|
|
786
|
+
// Prune knownSessions — delete completed sessions older than 30 days
|
|
745
787
|
const cutoff = Date.now() - (30 * 24 * 60 * 60 * 1000);
|
|
746
788
|
for (const [id, snap] of knownSessions) {
|
|
747
789
|
if (snap.status === 'completed' && snap.updatedAt && snap.updatedAt < cutoff) {
|
|
@@ -767,17 +809,35 @@ log(`Poll interval: ${POLL_INTERVAL_MS / 1000}s`);
|
|
|
767
809
|
pollSessions();
|
|
768
810
|
const timer = setInterval(pollSessions, POLL_INTERVAL_MS);
|
|
769
811
|
|
|
812
|
+
// Machine metrics: the first sample a few seconds after start (the CPU figure needs a
|
|
813
|
+
// delta from the baseline taken at load), then every interval. Off when metrics.db failed.
|
|
814
|
+
function sampleMetrics() {
|
|
815
|
+
try { collectSystemMetrics(); } catch (e) { log(`[METRICS ERROR] ${e.message}`); }
|
|
816
|
+
}
|
|
817
|
+
const FIRST_SAMPLE_DELAY_MS = 5000;
|
|
818
|
+
let metricsTimer = null;
|
|
819
|
+
const firstSample = metrics
|
|
820
|
+
? setTimeout(() => { sampleMetrics(); metricsTimer = setInterval(sampleMetrics, METRICS_INTERVAL_MS); }, FIRST_SAMPLE_DELAY_MS)
|
|
821
|
+
: null;
|
|
822
|
+
if (metrics) log(`Machine metrics every ${METRICS_INTERVAL_MS / 1000}s into ${METRICS_DB_PATH}`);
|
|
823
|
+
|
|
770
824
|
// Graceful shutdown
|
|
771
825
|
process.on('SIGTERM', () => {
|
|
772
826
|
log('SIGTERM received, shutting down...');
|
|
773
827
|
clearInterval(timer);
|
|
828
|
+
clearTimeout(firstSample);
|
|
829
|
+
clearInterval(metricsTimer);
|
|
774
830
|
db.close();
|
|
831
|
+
metrics?.db.close();
|
|
775
832
|
process.exit(0);
|
|
776
833
|
});
|
|
777
834
|
|
|
778
835
|
process.on('SIGINT', () => {
|
|
779
836
|
log('SIGINT received, shutting down...');
|
|
780
837
|
clearInterval(timer);
|
|
838
|
+
clearTimeout(firstSample);
|
|
839
|
+
clearInterval(metricsTimer);
|
|
781
840
|
db.close();
|
|
841
|
+
metrics?.db.close();
|
|
782
842
|
process.exit(0);
|
|
783
843
|
});
|