@flame0510/project-aether 1.10.0 → 1.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/README.md +1 -1
  2. package/app/agents/CostSection.tsx +2 -8
  3. package/app/agents/PageClient.tsx +7 -0
  4. package/app/agents/PanelRow.tsx +9 -0
  5. package/app/agents/ResourceSection.tsx +105 -0
  6. package/app/api/assistant/route.ts +2 -2
  7. package/app/api/containers/route.ts +18 -2
  8. package/app/api/costs/agent/route.ts +2 -3
  9. package/app/api/metrics/alerts/route.ts +47 -0
  10. package/app/api/metrics/containers/route.ts +54 -0
  11. package/app/api/metrics/route.ts +4 -6
  12. package/app/components/ui/Accordion.tsx +45 -0
  13. package/app/components/ui/TimeSeriesChart.tsx +20 -2
  14. package/app/components/ui/index.ts +1 -0
  15. package/app/containers/ContainersClient.tsx +48 -6
  16. package/app/globals.css +36 -0
  17. package/app/system/AgentCharts.tsx +138 -0
  18. package/app/system/AgentsSection.tsx +281 -0
  19. package/app/system/PageClient.tsx +7 -7
  20. package/app/system/RecentAlerts.tsx +72 -0
  21. package/app/system/SystemSkeleton.tsx +53 -1
  22. package/app/system/loading.tsx +5 -1
  23. package/daemon.js +195 -4
  24. package/docs/ARCHITECTURE.md +36 -5
  25. package/docs/FRONTEND-ARCHITECTURE.md +12 -3
  26. package/docs/REV4A.md +4 -4
  27. package/docs/dev/API-REFERENCE.md +91 -17
  28. package/docs/dev/DATABASE.md +58 -2
  29. package/docs/rag/DATA-FRESHNESS.md +16 -1
  30. package/docs/rag/GLOSSARY.md +2 -2
  31. package/docs/rag/REV4A-OVERVIEW.md +7 -3
  32. package/docs/rag/WHAT-I-CAN-ANSWER.md +4 -1
  33. package/lib/container-metrics.ts +340 -0
  34. package/lib/docker-socket-path.js +133 -0
  35. package/lib/docker-socket.ts +10 -99
  36. package/lib/docker-stats.js +284 -0
  37. package/lib/metrics-db.ts +15 -6
  38. package/lib/utils/format.ts +29 -3
  39. package/package.json +1 -1
  40. package/scripts/test-docker-stats.mjs +270 -0
package/daemon.js CHANGED
@@ -2,8 +2,9 @@
2
2
  /**
3
3
  * Rev4a Daemon — SQLite event logger
4
4
  * Polls OpenClaw sessions every 30s, writes events to SQLite, samples the host machine
5
- * (CPU, RAM, swap, storage) every 30s, and every 5 min reads what each running agent
6
- * spent, as its OpenClaw priced it, into costs.db — each on a timer of its own.
5
+ * (CPU, RAM, swap, storage) every 30s, samples what each container consumes every 60s
6
+ * (metrics.db), and every 5 min reads what each running agent spent, as its OpenClaw priced
7
+ * it, into costs.db — each on a timer of its own.
7
8
  */
8
9
 
9
10
  'use strict';
@@ -13,6 +14,9 @@ const fs = require('fs');
13
14
  const { spawnSync, execSync, execFile } = require('child_process');
14
15
  const Database = require('better-sqlite3');
15
16
  const path = require('path');
17
+ // Shared with the dashboard (lib/docker-socket.ts): one copy of where the Docker socket is.
18
+ const { dockerRequestJson } = require('./lib/docker-socket-path.js');
19
+ const dockerStats = require('./lib/docker-stats.js');
16
20
 
17
21
  // Same rule as lib/rev4a-paths.ts: REV4A_DB, else <data dir>/data/events.db, where the data
18
22
  // dir is REV4A_DATA_DIR or ~/.config/rev4a — the API reads the files the daemon writes.
@@ -184,6 +188,57 @@ CREATE TABLE IF NOT EXISTS system_disks (
184
188
  );
185
189
  CREATE INDEX IF NOT EXISTS idx_disks_ts ON system_disks(ts);
186
190
  CREATE INDEX IF NOT EXISTS idx_disks_mount_ts ON system_disks(mount, ts);
191
+
192
+ -- What each running container consumes, one row per container per sample (60 s). CPU is in
193
+ -- cores (CPU-seconds per second) and every rate in bytes per second, both null for the first
194
+ -- sample after a start and after a counter reset. Memory is the working set. A stopped
195
+ -- container has no rows: a gap, not zeros.
196
+ CREATE TABLE IF NOT EXISTS container_metrics (
197
+ ts INTEGER NOT NULL,
198
+ container TEXT NOT NULL,
199
+ cpu_cores REAL,
200
+ mem_mb INTEGER,
201
+ pids INTEGER,
202
+ net_rx_bps INTEGER,
203
+ net_tx_bps INTEGER,
204
+ blk_read_bps INTEGER,
205
+ blk_write_bps INTEGER
206
+ );
207
+ CREATE INDEX IF NOT EXISTS idx_container_metrics_ts ON container_metrics(ts);
208
+ CREATE INDEX IF NOT EXISTS idx_container_metrics_c_ts ON container_metrics(container, ts);
209
+
210
+ -- Who each container is: the agent's name (AGENT_NAME), whether it is an agent, and its named
211
+ -- volumes (the link to docker_storage).
212
+ CREATE TABLE IF NOT EXISTS container_sources (
213
+ container TEXT PRIMARY KEY,
214
+ agent_id TEXT,
215
+ name TEXT,
216
+ is_agent INTEGER NOT NULL DEFAULT 0,
217
+ volumes TEXT,
218
+ last_seen INTEGER NOT NULL
219
+ );
220
+
221
+ -- Docker's own cores and memory: the denominators for a container's share (on Docker Desktop
222
+ -- that is its VM, not the Mac).
223
+ CREATE TABLE IF NOT EXISTS docker_host (
224
+ id INTEGER PRIMARY KEY CHECK (id = 1),
225
+ ncpu INTEGER,
226
+ mem_total_mb INTEGER,
227
+ updated_at INTEGER
228
+ );
229
+
230
+ -- Disk Docker holds, read from /system/df every 10 min: kind 'volume' (name = the volume;
231
+ -- reclaimable_mb = its size when no container uses it), 'layer' (a container's writable layer),
232
+ -- 'images' and 'build_cache' (totals, with what could be freed).
233
+ CREATE TABLE IF NOT EXISTS docker_storage (
234
+ ts INTEGER NOT NULL,
235
+ kind TEXT NOT NULL,
236
+ name TEXT NOT NULL DEFAULT '',
237
+ size_mb INTEGER,
238
+ reclaimable_mb INTEGER
239
+ );
240
+ CREATE INDEX IF NOT EXISTS idx_docker_storage_ts ON docker_storage(ts);
241
+ CREATE INDEX IF NOT EXISTS idx_docker_storage_kind_ts ON docker_storage(kind, name, ts);
187
242
  `);
188
243
  return {
189
244
  db: mdb,
@@ -200,6 +255,26 @@ CREATE INDEX IF NOT EXISTS idx_disks_mount_ts ON system_disks(mount, ts);
200
255
  pruneMetrics: mdb.prepare('DELETE FROM system_metrics WHERE ts < ?'),
201
256
  pruneDisks: mdb.prepare('DELETE FROM system_disks WHERE ts < ?'),
202
257
  getLastTwoMetrics: mdb.prepare('SELECT cpu_percent, ram_used_mb, ram_total_mb FROM system_metrics ORDER BY ts DESC LIMIT 2'),
258
+ insertContainerMetric: mdb.prepare(`
259
+ INSERT INTO container_metrics (ts, container, cpu_cores, mem_mb, pids, net_rx_bps, net_tx_bps, blk_read_bps, blk_write_bps)
260
+ VALUES (@ts, @container, @cpu_cores, @mem_mb, @pids, @net_rx_bps, @net_tx_bps, @blk_read_bps, @blk_write_bps)
261
+ `),
262
+ upsertContainerSource: mdb.prepare(`
263
+ INSERT INTO container_sources (container, agent_id, name, is_agent, volumes, last_seen)
264
+ VALUES (@container, @agent_id, @name, @is_agent, @volumes, @last_seen)
265
+ ON CONFLICT(container) DO UPDATE SET agent_id = excluded.agent_id, name = excluded.name,
266
+ is_agent = excluded.is_agent, volumes = excluded.volumes, last_seen = excluded.last_seen
267
+ `),
268
+ upsertDockerHost: mdb.prepare(`
269
+ INSERT INTO docker_host (id, ncpu, mem_total_mb, updated_at) VALUES (1, @ncpu, @mem_total_mb, @updated_at)
270
+ ON CONFLICT(id) DO UPDATE SET ncpu = excluded.ncpu, mem_total_mb = excluded.mem_total_mb, updated_at = excluded.updated_at
271
+ `),
272
+ insertDockerStorage: mdb.prepare(`
273
+ INSERT INTO docker_storage (ts, kind, name, size_mb, reclaimable_mb) VALUES (@ts, @kind, @name, @size_mb, @reclaimable_mb)
274
+ `),
275
+ pruneContainerMetrics: mdb.prepare('DELETE FROM container_metrics WHERE ts < ?'),
276
+ pruneContainerSources: mdb.prepare('DELETE FROM container_sources WHERE last_seen < ?'),
277
+ pruneDockerStorage: mdb.prepare('DELETE FROM docker_storage WHERE ts < ?'),
203
278
  };
204
279
  } catch (err) {
205
280
  log(`[METRICS] disabled: cannot open ${METRICS_DB_PATH}: ${err.message}`);
@@ -416,6 +491,90 @@ function collectSystemMetrics() {
416
491
  }
417
492
  }
418
493
 
494
+ // ── Container consumption ────────────────────────────────────────────────────
495
+ //
496
+ // What each running container uses — CPU, memory, processes, network, block I/O — and the
497
+ // disk Docker holds for them, so the System page can say *who* is using the machine. Read from
498
+ // the Docker Engine API (lib/docker-stats.js); written to metrics.db, next to the machine's
499
+ // own numbers. Every running container, agents and the rest alike: the page tells them apart.
500
+
501
+ const CONTAINERS_INTERVAL_MS = 60_000;
502
+ /** The first pass only primes the CPU counters; the second, this long after, is the first with a CPU figure. */
503
+ const CONTAINERS_FIRST_DELAY_MS = 6_000;
504
+ const CONTAINERS_SECOND_DELAY_MS = 15_000;
505
+ const STORAGE_INTERVAL_MS = 10 * 60_000;
506
+ const STORAGE_FIRST_DELAY_MS = 25_000;
507
+
508
+ // A rate over a gap longer than three intervals (a laptop that slept, skipped passes) is no rate.
509
+ const containerSampler = dockerStats.createContainerSampler({
510
+ request: dockerRequestJson,
511
+ maxWindowS: (3 * CONTAINERS_INTERVAL_MS) / 1000,
512
+ });
513
+ let containersRunning = false;
514
+ let dockerContainersDown = false;
515
+
516
+ async function collectContainerMetrics() {
517
+ if (!metrics || containersRunning) return;
518
+ containersRunning = true;
519
+ try {
520
+ let sample;
521
+ try {
522
+ sample = await containerSampler.sample();
523
+ dockerContainersDown = false;
524
+ } catch (e) {
525
+ // Docker down or not yet up: said once, not every minute; it recovers by itself.
526
+ if (!dockerContainersDown) log(`[CONTAINERS] Docker unreachable: ${e.message}`);
527
+ dockerContainersDown = true;
528
+ return;
529
+ }
530
+ metrics.db.transaction(() => {
531
+ for (const row of sample.rows) metrics.insertContainerMetric.run({ ts: sample.ts, ...row });
532
+ for (const source of sample.sources) metrics.upsertContainerSource.run({ ...source, last_seen: sample.ts });
533
+ if (sample.host) metrics.upsertDockerHost.run({ ncpu: sample.host.ncpu, mem_total_mb: sample.host.memTotalMb, updated_at: sample.ts });
534
+ metrics.pruneContainerMetrics.run(sample.ts - METRICS_RETENTION_S);
535
+ metrics.pruneContainerSources.run(sample.ts - METRICS_RETENTION_S);
536
+ })();
537
+ } finally {
538
+ containersRunning = false;
539
+ }
540
+ }
541
+
542
+ let storageRunning = false;
543
+ let dockerStorageFailing = false;
544
+
545
+ async function collectDockerStorage() {
546
+ if (!metrics || storageRunning) return;
547
+ storageRunning = true;
548
+ try {
549
+ let df;
550
+ try {
551
+ df = dockerStats.parseStorage(await dockerRequestJson('GET', '/system/df', dockerStats.DF_TIMEOUT_MS));
552
+ if (!df) throw new Error('not a /system/df answer');
553
+ dockerStorageFailing = false;
554
+ } catch (e) {
555
+ // The last reading stays in the table: a failed one records nothing.
556
+ if (!dockerStorageFailing) log(`[CONTAINERS] Docker storage unreadable: ${e.message}`);
557
+ dockerStorageFailing = true;
558
+ return;
559
+ }
560
+ const ts = Math.floor(Date.now() / 1000);
561
+ const mb = (bytes) => (bytes === null ? null : Math.round(bytes / dockerStats.MB));
562
+ metrics.db.transaction(() => {
563
+ for (const v of df.volumes) {
564
+ metrics.insertDockerStorage.run({ ts, kind: 'volume', name: v.name, size_mb: mb(v.sizeBytes), reclaimable_mb: v.unused ? mb(v.sizeBytes) : 0 });
565
+ }
566
+ for (const l of df.layers) {
567
+ metrics.insertDockerStorage.run({ ts, kind: 'layer', name: l.container, size_mb: mb(l.sizeBytes), reclaimable_mb: null });
568
+ }
569
+ metrics.insertDockerStorage.run({ ts, kind: 'images', name: '', size_mb: mb(df.images.sizeBytes), reclaimable_mb: mb(df.images.reclaimableBytes) });
570
+ metrics.insertDockerStorage.run({ ts, kind: 'build_cache', name: '', size_mb: mb(df.buildCache.sizeBytes), reclaimable_mb: mb(df.buildCache.reclaimableBytes) });
571
+ metrics.pruneDockerStorage.run(ts - METRICS_RETENTION_S);
572
+ })();
573
+ } finally {
574
+ storageRunning = false;
575
+ }
576
+ }
577
+
419
578
  // ── Agent Costs ──────────────────────────────────────────────────────────────
420
579
  //
421
580
  // What each agent spent, as its own OpenClaw computed it. Rev4a writes every model's
@@ -620,8 +779,8 @@ async function gatewayCall(container, method, params) {
620
779
  return JSON.parse(out.slice(start));
621
780
  }
622
781
 
623
- /** A Docker container name, as `docker ps` prints it; anything else is never exec'd. */
624
- const CONTAINER_NAME_RE = /^[A-Za-z0-9][A-Za-z0-9_.-]{0,127}$/;
782
+ /** A Docker container name, as `docker ps` prints it; anything else is never exec'd (one rule: lib/docker-stats.js). */
783
+ const CONTAINER_NAME_RE = dockerStats.CONTAINER_NAME_RE;
625
784
  /**
626
785
  * Each field quoted and escaped by Docker itself (`json`, `printf "%q"`), so a display
627
786
  * name holding a tab or a newline stays one field of one line. Only AGENT_NAME is read
@@ -1218,6 +1377,32 @@ const firstSample = metrics
1218
1377
  : null;
1219
1378
  if (metrics) log(`Machine metrics every ${METRICS_INTERVAL_MS / 1000}s into ${METRICS_DB_PATH}`);
1220
1379
 
1380
+ // Container consumption: a priming pass, a short wait for the first CPU figure, then every
1381
+ // interval; the disk Docker holds on a slower timer. Off when metrics.db failed.
1382
+ function sampleContainers() {
1383
+ collectContainerMetrics().catch((e) => log(`[CONTAINERS ERROR] ${e.message}`));
1384
+ }
1385
+ function sampleStorage() {
1386
+ collectDockerStorage().catch((e) => log(`[CONTAINERS ERROR] ${e.message}`));
1387
+ }
1388
+ let containersTimer = null;
1389
+ let containersPasses = 0;
1390
+ function scheduleContainers(delay) {
1391
+ containersTimer = setTimeout(() => {
1392
+ sampleContainers();
1393
+ containersPasses++;
1394
+ scheduleContainers(containersPasses === 1 ? CONTAINERS_SECOND_DELAY_MS : CONTAINERS_INTERVAL_MS);
1395
+ }, delay);
1396
+ }
1397
+ let storageTimer = null;
1398
+ const firstStorage = metrics
1399
+ ? setTimeout(() => { sampleStorage(); storageTimer = setInterval(sampleStorage, STORAGE_INTERVAL_MS); }, STORAGE_FIRST_DELAY_MS)
1400
+ : null;
1401
+ if (metrics) {
1402
+ scheduleContainers(CONTAINERS_FIRST_DELAY_MS);
1403
+ log(`Container consumption every ${CONTAINERS_INTERVAL_MS / 1000}s, Docker storage every ${STORAGE_INTERVAL_MS / 60000} min, into ${METRICS_DB_PATH}`);
1404
+ }
1405
+
1221
1406
  // Agent costs: first pass shortly after start (agents may still be booting), then every
1222
1407
  // interval. Off when costs.db failed.
1223
1408
  function sampleCosts() {
@@ -1235,6 +1420,9 @@ process.on('SIGTERM', () => {
1235
1420
  clearInterval(timer);
1236
1421
  clearTimeout(firstSample);
1237
1422
  clearInterval(metricsTimer);
1423
+ clearTimeout(containersTimer);
1424
+ clearTimeout(firstStorage);
1425
+ clearInterval(storageTimer);
1238
1426
  clearTimeout(firstCosts);
1239
1427
  clearInterval(costsTimer);
1240
1428
  db.close();
@@ -1248,6 +1436,9 @@ process.on('SIGINT', () => {
1248
1436
  clearInterval(timer);
1249
1437
  clearTimeout(firstSample);
1250
1438
  clearInterval(metricsTimer);
1439
+ clearTimeout(containersTimer);
1440
+ clearTimeout(firstStorage);
1441
+ clearInterval(storageTimer);
1251
1442
  clearTimeout(firstCosts);
1252
1443
  clearInterval(costsTimer);
1253
1444
  db.close();
@@ -1,7 +1,7 @@
1
1
  # Rev4a Architecture — Design & Vision
2
2
 
3
3
  > **Status:** Active — `main` branch
4
- > **Last updated:** 2026-09-27
4
+ > **Last updated:** 2026-10-01
5
5
  > **Goal:** Transform Rev4a from a monitoring dashboard into a central orchestrator for a distributed multi-container agency.
6
6
 
7
7
  ---
@@ -72,7 +72,7 @@
72
72
 
73
73
  **Credentials vault:** The `/credentials` page stores third-party service tokens in `credentials.db` and installs them into containers on explicit sync. Provider API keys live separately in `provider-keys.json`, managed from the Gateway UI.
74
74
 
75
- **Container management UI:** The `/containers` page lists all Docker containers, running or stopped, with status, image, IP and ports, and opens a web terminal into any running one. Fully functional. (Machine CPU, RAM and disk are on `/system`.)
75
+ **Container management UI:** The `/containers` page lists all Docker containers, running or stopped, with status, image, IP and ports, and opens a web terminal into any running one. Fully functional. Each running container's CPU, memory and processes are on its row, with a link to its charts on `/system`, which also shows the machine's CPU, RAM and disk.
76
76
 
77
77
  **Workspace API:** Lazy-loaded file tree explorer with real-time reads (no caching), supports both host and container workspaces via `docker exec`.
78
78
 
@@ -578,10 +578,12 @@ The Rev4a daemon (`daemon.js`) is a standalone Node.js process that bridges the
578
578
  3. Emit `spawn` / `complete` / `fail` / `spawn_timeout` events into the `events` table
579
579
  4. Sample the host machine (CPU, RAM, swap, storage) every 30 s into `metrics.db`, on a
580
580
  timer of its own
581
- 5. Detect anomalies (CPU > 85%, RAM > 90%, a disk > 90%) and record them as events
582
- 6. Read what each running agent spent, as its own OpenClaw priced it, every 5 min into
581
+ 5. Sample what each container consumes every 60 s, and the disk Docker holds every 10 min,
582
+ into `metrics.db` (see [Container Consumption](#container-consumption))
583
+ 6. Detect anomalies (CPU > 85%, RAM > 90%, a disk at 90%) and record them as events
584
+ 7. Read what each running agent spent, as its own OpenClaw priced it, every 5 min into
583
585
  `costs.db`, and each vendor's balance or usage every hour (see [Agent Costs](#agent-costs))
584
- 7. Manage DB lifecycle (WAL mode; events.db checkpointed after each poll cycle)
586
+ 8. Manage DB lifecycle (WAL mode; events.db checkpointed after each poll cycle)
585
587
 
586
588
  ### Poll Interval
587
589
 
@@ -629,6 +631,35 @@ taken at load), then one every 30 s. Samples older than 30 days are pruned at ev
629
631
  it through `lib/metrics-db.ts` (read-only, `null` when there is nothing to read);
630
632
  `GET /api/metrics` serves the System page (`/system`) and the dashboard's Machine card.
631
633
 
634
+ ### Container Consumption
635
+
636
+ The System page says *that* the machine is busy; this says *who*. On a timer of its own the
637
+ daemon reads every running container from the Docker Engine API — `GET /containers/{id}/stats?stream=false&one-shot=true`,
638
+ 10–20 ms a container, where `docker stats --no-stream` waits a second each and returns formatted
639
+ strings — and writes `container_metrics` into `metrics.db` (schema in `docs/dev/DATABASE.md`):
640
+
641
+ - **CPU** from the cumulative CPU time (ns): the difference between two samples over the
642
+ elapsed time, in cores; the first sample after a start only primes the counters. A counter
643
+ that goes backwards (the container restarted) gives no figure for that sample, never a spike.
644
+ - **Memory** is the working set — usage minus the inactive page cache (`inactive_file` on
645
+ cgroup v2, `total_inactive_file` on v1), what `docker stats` shows.
646
+ - **Processes, network and block I/O** as gauges and per-second rates. A container that stops
647
+ leaves a gap; one that is gone takes its counters with it.
648
+ - **Names:** an agent's `AGENT_NAME` (read once per container id from its inspect), else the
649
+ container's name. Every running container is sampled — agents, and the rest as "container".
650
+ - **Docker's disk** (`/system/df`, every 10 min): each named volume — which gives each agent's
651
+ storage, through the volume names its mounts list —, each writable layer, the images and the
652
+ build cache with what no container/build uses. Sizes and "unused" are readings, not advice.
653
+ - The shares need Docker's own cores and memory (`/info`, re-read every 10 min): on Docker
654
+ Desktop those are its VM's (8 GB of a 16 GB Mac), on the VPS the machine's.
655
+
656
+ The socket path rules (env, `/var/run`, Docker Desktop's `~/.docker/run`, `docker context`) live
657
+ once, in `lib/docker-socket-path.js`, required by both the daemon and `lib/docker-socket.ts`. A
658
+ Docker that is down is logged once and recovered from by itself; a failed disk reading leaves the
659
+ previous one in place. `lib/container-metrics.ts` reads it all for `GET /api/metrics/containers`
660
+ (the System page's *Agents and containers* section, the agent panel's RESOURCES, the Containers
661
+ page). Anomalies per container (processes, sustained memory) are not raised yet.
662
+
632
663
  ### Agent Costs
633
664
 
634
665
  Rev4a computes no cost. Every model in the `rev4a` provider block synced into the agents
@@ -1,6 +1,6 @@
1
1
  # Rev4a Frontend Architecture
2
2
 
3
- > **Last updated:** 2026-09-27
3
+ > **Last updated:** 2026-10-01
4
4
 
5
5
  ## Layering
6
6
 
@@ -39,7 +39,7 @@ All shared UI primitives live in `app/components/ui/` and are exported from `app
39
39
  | `Metric` | `Metric.tsx` | Metric card with title, value, subtitle, tone, and a `size` (`md` default, `sm` for a grid inside a dialog: smaller value, tighter spacing). |
40
40
  | `StatusCard` | `StatusCard.tsx` | Health/status report card. |
41
41
  | `Meter` / `meterLevel` | `Meter.tsx` | Horizontal 0-100 % fill whose colour carries the level (`ok` accent, `warning` yellow, `critical` red); the level is also written out beside it, never colour alone. `meterLevel(percent, warning, critical)` picks the level. ARIA `role="meter"`. |
42
- | `TimeSeriesChart` | `TimeSeriesChart.tsx` | One series over time in plain SVG measured to its container (text never scaled): a 2px accent line for the value, a 10 % wash up to the bucket's peak, a hairline grid at 0 / 50 / 100 %, the latest value labelled at the line's end. Points further apart than 1.5 buckets are not joined (a gap means no samples). Hover or keyboard focus (← → Esc) shows a crosshair snapped to the nearest point with value and peak; a **Table** disclosure holds the same data. `stale` dims the previous render while a new range loads. `yMax` / `unit` scale and label a percentage by default; `formatValue` replaces the label format everywhere (axis, end label, tooltip, table) — the Costs page passes dollars. The left and right margins grow with the longest axis label and the end label, so a longer format is never cut. Used by the System and Costs pages. |
42
+ | `TimeSeriesChart` | `TimeSeriesChart.tsx` | One series over time in plain SVG measured to its container (text never scaled): a 2px accent line for the value, a 10 % wash up to the bucket's peak, a hairline grid at 0 / 50 / 100 %, the latest value labelled at the line's end. Points further apart than 1.5 buckets are not joined (a gap means no samples). Hover or keyboard focus (← → Esc) shows a crosshair snapped to the nearest point with value and peak; a **Table** disclosure holds the same data. `stale` dims the previous render while a new range loads. `yMax` (a number, or `'auto'`: a rounded ceiling just above the highest value or peak — for a series that is a small share of its natural scale, like one agent's CPU) / `unit` scale and label a percentage by default; `formatValue` replaces the label format everywhere (axis, end label, tooltip, table) — the Costs page passes dollars. The left and right margins grow with the longest axis label and the end label, so a longer format is never cut. Used by the System and Costs pages. |
43
43
  | `Surface` | `Surface.tsx` | Shared panel/card surface, variant prop. |
44
44
  | `Page` / `PageHeader` | `Page.tsx` | Full-page layout shell. |
45
45
  | `Icons` | `Icons.tsx` | SVG icons (`EyeIcon`, `EyeOffIcon`), 16/20px shared. |
@@ -47,9 +47,12 @@ All shared UI primitives live in `app/components/ui/` and are exported from `app
47
47
  | `UpdateSection` | `app/agents/UpdateSection.tsx` | OPENCLAW VERSION section of the agent detail panel: the version the agent runs, **Update to <version>** when a newer supported version is downloaded (confirm modal, disabled while the agent is stopped), the running update's steps with backup progress, the outcome, and **Roll back to <version>** after an update. Polls `/api/agents/[id]/update` every 2 s while an update or rollback runs; one action at a time (keyed busy state). |
48
48
  | `BackupSection` | inline in `app/agents/PageClient.tsx` | BACKUP section of the agent detail panel, on the cold backup and the restore. **Backup Now** starts `POST /api/agents/[id]/cold-backup`; while the job runs a banner shows the file and its live percent with **Cancel** (`DELETE /cold-backup`). **Restore** (after a confirm) starts `POST /restore` and a banner shows `Restoring <file>…` (no percent: the extract is a single `tar xzf`, and there is no Cancel). Both sections poll their `GET` every 2 s while running, and on mount pick up a job that is already running — a backup lives in a Docker helper, a restore in `agent_restores`, so reloading the page or navigating away never loses them nor allows a second one (the server answers 409 anyway). Delete per row; all actions disabled while one runs; keyed busy state `{ kind, file }` so only the row in action shows the spinner. On the agent list, an activity Badge (fed by `/api/agents/activity-summary`, polled at 2 s only while something runs, otherwise riding the 15 s list poll) reads `BACKUP nn%`, `RESTORING`, `RECREATING`, `EDITING` or `UPDATING`. |
49
49
  | `RecreateSection` | inline in `app/agents/PageClient.tsx` | RECREATE section of the agent detail panel. **Recreate Container** starts `POST /api/agents/[id]/recreate` (202) after a confirm; a banner then shows the phase — *Backing up … nn%* while the cold backup runs, *Recreating container…* while the container is rebuilt and the gateway starts. The section polls `GET /recreate` every 2 s, and on mount picks up a recreate that is already running, so a reload or navigation never loses it; it refetches the agent once the job reports `done`. |
50
- | System page | `app/system/PageClient.tsx` | `/system`: CPU, memory and storage of the host, from `GET /api/metrics`. Three cards (CPU with cores and load; memory with available and swap; one `Meter` per filesystem with its roles) — thresholds 85/95 % for CPU and memory, 80/90 % for storage — then a range switch (`Tabs`: 1h · 24h · 7d · 30d) scoping the charts below it: CPU and memory average with peak, one chart per filesystem. Polls every 30 s with an `AbortController` ref (a range change aborts the previous fetch). Times in the configured Rev4a timezone (`useRev4aTimezone`). Before the first answer the cards and charts are skeletons shaped like them (`app/system/SystemSkeleton.tsx`, also the route's `loading.tsx`), so nothing jumps; the hostname line is cut with an ellipsis and never widens the page. Says *No samples yet* before the daemon's first sample and *Not collecting* when the latest sample is older than three intervals. Two columns of charts from lg (992px) up, one below. |
50
+ | `Accordion` | `Accordion.tsx` | A row that opens: the header is a real `<button type="button">` (`aria-expanded`, `aria-controls` while open) carrying a `title`, `summary` figures and an optional muted `detail` line, so the answer to "what is this now" is on screen closed; the body holds what takes room and is **mounted only while open**, so a chart inside fetches nothing while closed. Controlled (`open` / `onToggle`: the parent can deep-link or fetch on it), optional `id` for a link to land on. Square, flat, tokens only. On mobile the summary wraps under the title. Used by the System page's agents. |
51
+ | System page | `app/system/PageClient.tsx` | `/system`: CPU, memory and storage of the host, from `GET /api/metrics`. Three cards (CPU with cores and load; memory with available and swap; one `Meter` per filesystem with its roles) — thresholds 85/95 % for CPU and memory, 80/90 % for storage — then a range switch (`Tabs`: 1h · 24h · 7d · 30d) scoping the charts below it: CPU and memory average with peak, one chart per filesystem. Polls every 30 s with an `AbortController` ref (a range change aborts the previous fetch). Times in the configured Rev4a timezone (`useRev4aTimezone`). Before the first answer the cards and charts are skeletons shaped like them (`app/system/SystemSkeleton.tsx`, also the route's `loading.tsx`), so nothing jumps; the hostname line is cut with an ellipsis and never widens the page. Says *No samples yet* before the daemon's first sample and *Not collecting* when the latest sample is older than three intervals. Two columns of charts from lg (992px) up, one below; an odd last chart spans both columns instead of leaving a hole. Under them, **Agents and containers** (`app/system/AgentsSection.tsx`, heading in the accent eyebrow with a rule above — `system-section-heading`): *Docker storage* (images, volumes, writable layers, build cache, each with what no container uses — wording neutral on purpose: an unused image may be a rollback target, an unused volume the cold backups), then **one `Accordion` per container**, sorted by CPU now (`GET /api/metrics/containers`, polled every 30 s with an `AbortController` ref). Closed: name, an *agent* / *container* pill, *stopped* when it is — or *no data* when Docker still lists it but its stats call fails — and CPU (share of Docker's cores), memory, storage, PIDs, with average · peak over the range tabs, cores, network and memory share on a muted line. Open (`AgentCharts.tsx`): the same three charts as the machine — CPU (share), Memory, Storage — for that container, fetched when it opens (`?container=`), polled every 30 s, aborted on close or range change; each with `yMax="auto"` (an agent is 0.01–2 % of Docker's cores) and sizes in MB/GB, on `subtle` surfaces inside the accordion. A last row, *Everything else*, is the machine minus the containers. `/system?agent=<container>` opens one and scrolls to it. Then **Recent alerts** (`RecentAlerts.tsx`, `GET /api/metrics/alerts`): the machine's latest 10 anomalies with time (configured timezone), metric pill and message. The section has its own skeleton (`SystemAgentsSkeleton`, also in the route's `loading.tsx`). |
51
52
  | Costs page | `app/costs/PageClient.tsx` | `/costs`: what each agent spent, as its own OpenClaw priced it, from `GET /api/costs/usage`. A range switch (`Tabs`: Today · 7 days · 30 days, UTC days) scopes everything below it: three cards (spent with tokens and days; the cost split into input / output / cache read / cache write; the tokens with the share of input served from cache), the daily spend (`TimeSeriesChart` with dollars, not on Today), then *By agent* and *By model* side by side — each row a name and figure over a share `Meter`, the pattern of the System storage card — *Against the bill* (per vendor, what it billed, from balance or usage readings, next to what the agents priced over the same readings, top-ups apart), and the most expensive sessions. Warnings above: *No figures yet* before the daemon's first pass, *Not current* for agents still being read whose last read failed or is older than three intervals (a stopped or deleted agent keeps its figures unflagged), and *Calls without a price* by model (a model with no price is also marked in *By model*, so a $0.00 is never read as free). The header shows the DeepSeek band now (peak / off-peak, until when, UTC) when a DeepSeek model is on offer. Polls every 60 s with an `AbortController` ref. Skeletons shaped like the content (`app/costs/CostsSkeleton.tsx`, also the route's `loading.tsx`). Costs under a cent keep four decimals. |
52
53
  | Agent costs | `app/agents/CostSection.tsx` | The COSTS section of the agent panel, after MODEL: today (UTC), last 7 and 30 days with tokens, the top model, calls without a price, when the agent was last read (a note when not recently: a stopped agent keeps its last figures), and a link to `/costs`. From `GET /api/costs/agent`, every 60 s with an `AbortController` ref. |
54
+ | Agent resources | `app/agents/ResourceSection.tsx` | The RESOURCES section of the agent panel, after COSTS: CPU and memory now and over the last 24 h, processes, network, storage (volume and layer), from `GET /api/metrics/containers`, every 60 s with an `AbortController` ref, reset on agent switch; *Stopped* when the agent is not running, *Not collecting* when the newest sample is older than three intervals (the "now" figures then stand down), a refresh-failure line while the last answer stays, and a link to its charts on `/system?agent=<container>` — the history lives in one place. Rows are the shared `PanelRow` (`app/agents/PanelRow.tsx`, also used by COSTS). |
55
+ | Containers page | `app/containers/ContainersClient.tsx` | `/containers`: one card per Docker container; a running one the daemon has sampled adds a line — CPU share, memory, PIDs — and *Charts →* to its accordion on `/system`. The figures are a bonus: a daemon that has not sampled yet, or a failed request, leaves the list as it was. |
53
56
  | Machine card | `app/components/SystemCockpit.tsx` | On the dashboard, next to Active sessions: CPU, RAM and the fullest disk now, each figure coloured only when past its threshold (same thresholds as the System page), the issues named in words, and a link to `/system`. It says *No samples yet* before the first sample, *Not collecting* when the newest is stale, and *Metrics unavailable* when a refresh fails (keeping the last figures). A `Surface`, not a `Metric`: a danger `Metric` paints its whole value red. |
54
57
  | Container terminal | `app/containers/terminal/[id]/TerminalClient.tsx` | Full-screen shell into a container: xterm.js 6 (`@xterm/xterm` + `addon-fit` refitted by a `ResizeObserver`, `addon-web-links`), loaded client-side only, palette from the design tokens. Keystrokes are sent raw (`onData`, `onBinary`), the size on connect and whenever the columns or rows change. A bar with `ui-kicker`, the container name, a status `Badge` (Connecting / Connected / Reconnecting / Closed) and `Button`s **Reconnect** / **Close**; styles are the `terminal-page` classes. Session id per container in `sessionStorage` (a UUID from `getRandomValues`: `randomUUID` is missing over plain HTTP), so a reload resumes the shell; after a drop it reconnects with `resume=1` and is told when the shell is gone; retries with backoff (six failed opens, about 40 s, then a message) and stops on the server's final close codes with the reason shown. **Close** while reconnecting still ends the waiting shell. Protocol and sessions: `docs/CONTAINER-TERMINAL.md`. |
55
58
  | `ImageDownloadBanner` | `app/agents/ImageDownloadBanner.tsx` | Agent image banner on the Agents page. Polls `/api/agents/image-status` every 2 s; offers **Download Image** when no supported version is downloaded, **Download <version>** when the registry publishes a newer one, and while a download runs shows the percent of layers finished and the latest line of docker's output (the bar is indeterminate until a percent can be computed). **Cancel** (`DELETE /api/agents/download-image`, own loading state) appears once `image-status` reports the download running; before that the button reads *Starting…* and is disabled. Every outcome — ready and cancelled for 3 s, a failure until the next action — comes from the server's `lastResult`, so one that ended while the page was closed or reloading is still reported if it is less than 30 s old (aged with `serverTime`). Each outcome is announced once per page load, although the Agents page mounts the banner in three places (mobile list, mobile detail, desktop). Downloading changes no agent. |
@@ -67,6 +70,12 @@ All shared UI primitives live in `app/components/ui/` and are exported from `app
67
70
  | `Badge` | `Badge.tsx` | Inline status tag with tone variants (success, danger, warning, neutral). Used for channel chips, pairing labels, error/success messages. |
68
71
  | `Skeleton` + shape helpers | `app/components/Skeleton.tsx` | Shimmer placeholders. `Skeleton` is the primitive; the rest mirror one specific layout each: `TreeSkeleton`, `CodeSkeleton`, `CronJobsSkeleton`, `CronRunsSkeleton`, `CredentialCardsSkeleton`, `AgentSyncRowsSkeleton`, `SkeletonLines`, `SkeletonMetric`, `CardRowSkeleton`. A route's `loading.tsx` sits in `RouteLoadingShell` (full width up to `maxWidth`, centred): the app shell's content area is a column flexbox, and a `margin: 0 auto` child without `width: 100%` shrinks to its content — every loader did, and its rows came out as wide as its 120 px title (measured: 120 px instead of 1040). `ListPageSkeleton` (a title over rows) is the list pages' loader. Create Agent (`app/agents/create/loading.tsx`) and the container terminal (`app/containers/terminal/[id]/loading.tsx`) have their own, shaped like the page, instead of inheriting the Agents / Containers rows; the Agents page prefetches `/agents/create` (its **+ New** uses `router.push`, which prefetches nothing) and the Containers **Terminal** is a `Link`, so both open without a full reload. A shape helper must match the real markup it stands in for — same row structure, same paddings, same element count where the count is known — so nothing reflows when data replaces it. Inside text (a `<p>`, a `<span>`) pass `as="span"`: a `<div>` there is invalid HTML, and the parser splits the paragraph in the server-rendered page. The System page keeps its own shapes in `app/system/SystemSkeleton.tsx`, each shimmer inside the class of the text it replaces so the line boxes are the real ones — measured to 0 px of movement on desktop and mobile. |
69
72
 
73
+ The System page's container rows distinguish a recent Docker listing from fresh resource
74
+ statistics. A row seen in Docker without figures says *no data*. When the newest statistics
75
+ sample ages past three intervals, current CPU, memory and processes disappear; an older row
76
+ says *not current*. The page advances its clock at each poll and recomputes age from
77
+ `sampled_at`, so repeated failed refreshes cannot leave a previous answer labelled as current.
78
+
70
79
  ### Rules
71
80
 
72
81
  1. **Prefer existing primitives over new ad-hoc markup.** Every new component starts from the shared catalog.
package/docs/REV4A.md CHANGED
@@ -1,6 +1,6 @@
1
1
  # Rev4a — VPS Dashboard
2
2
 
3
- > **Last updated:** 2026-09-27
3
+ > **Last updated:** 2026-10-01
4
4
 
5
5
  A Next.js 16 dashboard for monitoring and managing the OpenClaw ecosystem.
6
6
 
@@ -101,9 +101,9 @@ When environment variables are saved via the Config page (Save & Restart):
101
101
  |---|---|
102
102
  | `/` | System overview — health, sessions, the machine's CPU / RAM / disk (links to `/system`), what the agents spent today (as on `/costs`), live feed |
103
103
  | `/agents` | Running agents (Docker containers with `AGENT_ID`), gateway token management, agent creation wizard, channel manager (Telegram pairing) |
104
- | `/containers` | All Docker containers on the host |
104
+ | `/containers` | All Docker containers on the host, each running one with its CPU, memory and processes now (from the daemon's container samples) and a link to its charts |
105
105
  | `/costs` | What each agent spent, as its own OpenClaw priced it with the rates Rev4a syncs: today / 7 / 30 UTC days, by day, by agent, by model, the most expensive sessions, the calls that could not be priced, the DeepSeek band now, and a check against what DeepSeek and OpenRouter actually billed (read by the daemon every 5 min into `costs.db`; each agent's own figures are also in its panel, COSTS) |
106
- | `/system` | The host machine: CPU, memory and swap, storage per filesystem — current values and 1h / 24h / 7d / 30d history (from `metrics.db`, sampled by the daemon every 30 s) |
106
+ | `/system` | The host machine: CPU, memory and swap, storage per filesystem — current values and 1h / 24h / 7d / 30d history (from `metrics.db`, sampled by the daemon every 30 s); **Agents and containers**: what Docker holds on disk, and one row per container — CPU, memory, storage, processes — that opens to its own CPU / Memory / Storage charts (sampled every 60 s; a recently seen row without current statistics says "no data", and stale current figures are hidden; each agent's numbers are also in its panel, RESOURCES); and the machine's recent alerts |
107
107
  | `/wizard` | First-run setup wizard (Welcome → Providers → Agent → Ready) |
108
108
  | `/workspace` | File explorer with tree view + editor — VPS host or container workspaces |
109
109
  | `/lineage` | Agent lineage / orchestration tree |
@@ -308,7 +308,7 @@ Store third-party service credentials and install them into agent containers.
308
308
  ### Container management (`/containers`)
309
309
  View all running Docker containers with:
310
310
  - Name, image, status, ports, uptime
311
- - Resource usage (CPU, memory)
311
+ - Resource usage now (CPU share, memory, processes) for each running container, with *Charts →* to its history on `/system`
312
312
  - Quick links to agent control UIs
313
313
 
314
314
  ### Agent management (`/agents`)
@@ -1,6 +1,6 @@
1
1
  # Rev4a API Reference
2
2
 
3
- > **Last updated:** 2026-09-27
3
+ > **Last updated:** 2026-10-01
4
4
 
5
5
  All routes are under `/api/`. Authentication is required on every endpoint
6
6
  unless otherwise noted.
@@ -2045,6 +2045,81 @@ average and peak: 60 s for `1h`, 720 s for `24h`, 5040 s for `7d`, 21 600 s for
2045
2045
  means the daemon is not sampling; the page says so.
2046
2046
  - `host` is read live by the API process, which runs on the same host.
2047
2047
 
2048
+ ### `GET /api/metrics/containers?range=1h|24h|7d|30d`
2049
+ What every container consumes — now, and as an average and peak over the range — plus what is
2050
+ not a container and the disk Docker holds. From `metrics.db` (the daemon samples every 60 s,
2051
+ see `docs/dev/DATABASE.md`). Default range `1h`.
2052
+
2053
+ **Auth:** browser cookie or Bearer
2054
+
2055
+ **Response:**
2056
+ ```json
2057
+ {
2058
+ "available": true, "range": "1h", "interval_s": 60,
2059
+ "host": { "ncpu": 10, "mem_total_mb": 7837 },
2060
+ "sampled_at": 1790789760, "age_s": 10,
2061
+ "containers": [
2062
+ { "container": "agent_9253eee3", "name": "test", "agent_id": "agent_9253eee3", "is_agent": true,
2063
+ "running": true, "last_seen": 1790789760,
2064
+ "now": { "cpu_cores": 0.022, "cpu_percent": 0.22, "mem_mb": 668, "mem_percent": 8.5, "pids": 12,
2065
+ "net_rx_bps": 0, "net_tx_bps": 0, "blk_read_bps": 0, "blk_write_bps": 0 },
2066
+ "range": { "cpu_avg_percent": 1.02, "cpu_max_percent": 2.17, "mem_avg_mb": 679, "mem_max_mb": 701 },
2067
+ "storage": { "volume_mb": 1544, "layer_mb": 10, "total_mb": 1554, "ts": 1790789645 } }
2068
+ ],
2069
+ "rest": { "cpu_percent": 42.26, "mem_mb": 11864 },
2070
+ "docker_storage": { "ts": 1790789645, "images_mb": 4734, "images_reclaimable_mb": 3518,
2071
+ "build_cache_mb": 1597, "build_cache_reclaimable_mb": 1597,
2072
+ "volumes_mb": 10120, "unused_volumes_mb": 7337, "layers_mb": 30,
2073
+ "unused_volumes": [ { "name": "rev4a-backups", "size_mb": 7337 } ] }
2074
+ }
2075
+ ```
2076
+
2077
+ - `cpu_percent` is a share of Docker's cores (`host.ncpu`), `mem_percent` of Docker's memory —
2078
+ on Docker Desktop that is its VM, on the VPS the machine. `cpu_cores` is the raw figure.
2079
+ - `now` is `null` for a container that is not running (no sample within 3 intervals of the
2080
+ newest sample — running is judged against the newest sample, not the wall clock, so a
2081
+ stopped daemon does not make every container look stopped);
2082
+ `running: false` ones stay listed while they were seen in the range. A rate that is null in
2083
+ the newest sample (the first after a start) is replaced by the previous reading.
2084
+ - The list is sorted running first, then by CPU now. `storage` is the named volumes plus the
2085
+ writable layer, from the last Docker disk reading (every 10 min).
2086
+ - `rest` is the machine's newest sample minus the running containers (the containers' CPU
2087
+ re-scaled to the machine's cores first, so on Docker Desktop the two shares still compare):
2088
+ the host, Rev4a and other software (`null` without a machine sample or Docker's figures).
2089
+ - `docker_storage.*_reclaimable_mb` and `unused_volumes` (the 5 largest): what **no container
2090
+ uses** — images of older agent versions you may still roll back to count as such; a volume
2091
+ like the cold backups' is listed, not advised for removal. `volumes_mb` sums all volumes;
2092
+ `unused_volumes_mb` is the total of what no container uses, `unused_volumes` only names the
2093
+ five largest.
2094
+ - `available: false` until the daemon has created the tables; with the tables present and no
2095
+ rows yet it answers `available: true` with an empty list.
2096
+
2097
+ **With `&container=<name>`** — that container's history, bucketed like `GET /api/metrics`:
2098
+ ```json
2099
+ { "available": true, "range": "1h", "interval_s": 60, "host": { "ncpu": 10, "mem_total_mb": 7837 },
2100
+ "container": "agent_9253eee3", "name": "test", "bucket_s": 120,
2101
+ "history": [ { "ts": 1790789640, "cpu_avg": 1.41, "cpu_max": 2.17, "mem_avg": 685, "mem_max": 701 } ],
2102
+ "storage_bucket_s": 1200, "storage_history": [ { "ts": 1790788800, "total_mb": 1554 } ] }
2103
+ ```
2104
+ `cpu_*` are shares of Docker's cores, `mem_*` MB. A container that was not running has no
2105
+ points for that stretch (a gap). A name that matches the rules but was never sampled answers
2106
+ `available: true` with an empty history.
2107
+
2108
+ **Errors:** `400` for another `range` or an invalid container name. A `metrics.db` that
2109
+ cannot be opened (missing, corrupt) answers `available: false`, not an error.
2110
+
2111
+ ### `GET /api/metrics/alerts?limit=10`
2112
+ The machine's latest `system_anomaly` events (CPU above 85 % or memory above 90 % on two
2113
+ samples in a row, a disk at 90 %), newest first, from `events.db`. `limit` 1–50, default 10.
2114
+
2115
+ **Auth:** browser cookie or Bearer
2116
+
2117
+ **Response:**
2118
+ ```json
2119
+ { "alerts": [ { "ts": 1790789625355, "metric": "disk", "message": "Disk high: / 94% used", "threshold": 90, "values": [94] } ] }
2120
+ ```
2121
+ `ts` is in **milliseconds** (the `events` table's unit); `metric` is `cpu`, `ram`, `disk` or `null`.
2122
+
2048
2123
  ---
2049
2124
 
2050
2125
  ## System Health
@@ -2302,27 +2377,26 @@ Full key management documentation at [PROVIDERS.md](PROVIDERS.md).
2302
2377
  ## Containers
2303
2378
 
2304
2379
  ### `GET /api/containers`
2305
- List all running Docker containers with resource usage.
2380
+ Every Docker container, running or stopped, with its identity and network details. What a
2381
+ running container consumes lives in `GET /api/metrics/containers`.
2306
2382
 
2307
- **Auth:** browser cookie
2383
+ **Auth:** browser cookie or Bearer
2308
2384
 
2309
- **Response:**
2385
+ **Response:** a bare array —
2310
2386
  ```json
2311
- {
2312
- "containers": [
2313
- {
2314
- "name": "openclaw-atlas",
2315
- "image": "openclaw-agent-base:2026.9.3",
2316
- "status": "running",
2317
- "ports": ["0.0.0.0:3731->3000/tcp"],
2318
- "created": "2026-06-20T10:00:00Z",
2319
- "cpu_percent": 2.1,
2320
- "mem_usage_mb": 340
2321
- }
2322
- ]
2323
- }
2387
+ [
2388
+ { "id": "a1b2c3d4e5f6", "name": "openclaw-atlas", "image": "openclaw-agent-base:2026.9.3",
2389
+ "status": "Up 3 hours", "state": "running", "ports": "0.0.0.0:3731->3000/tcp",
2390
+ "ip": "172.18.0.2", "agentId": "agent_atlas", "displayName": "Atlas", "created": null }
2391
+ ]
2324
2392
  ```
2325
2393
 
2394
+ - `name` is the daemon's `primaryName` (the name with one leading slash) — the same key the
2395
+ metrics store and the `/system?agent=` link use.
2396
+ - `displayName` is the agent's `AGENT_NAME`, read from the inspect. If an inspect fails, the
2397
+ row still comes from Docker's list (`ip` and `displayName` then `null`): a container is
2398
+ never dropped, and never renamed, because one call failed.
2399
+
2326
2400
  The container list is the only container route. There is no logs endpoint and no
2327
2401
  action endpoint: logs are read through the terminal WebSocket, and lifecycle
2328
2402
  operations on agent containers go through `/api/agents/[id]/lifecycle`.
@@ -1,9 +1,9 @@
1
1
  # Rev4a — Database Reference
2
2
 
3
- > **Last updated:** 2026-09-27
3
+ > **Last updated:** 2026-10-01
4
4
 
5
5
  Rev4a keeps four SQLite files in WAL mode, by default all in the data directory's `data/`:
6
- `events.db` (sessions and events, below), `metrics.db` (machine metrics, see
6
+ `events.db` (sessions and events, below), `metrics.db` (machine and container metrics, see
7
7
  [Metrics Database](#metrics-database-metricsdb)), `costs.db` (what the agents spent, see
8
8
  [Costs Database](#costs-database-costsdb)) and `credentials.db` (see
9
9
  [Credentials Database](#credentials-database-credentialsdb)).
@@ -318,6 +318,62 @@ has dropped back under, or once after each daemon start: the state is kept in me
318
318
  Installs from before this database may still have a `system_metrics` table in
319
319
  `events.db`: nothing reads or writes it any more, and it can be dropped.
320
320
 
321
+ ### Container consumption (also in `metrics.db`)
322
+
323
+ What each running container uses, and the disk Docker holds, so the System page can say
324
+ *who* is using the machine. The daemon writes it from the Docker Engine API
325
+ (`lib/docker-stats.js`; where the socket is comes from `lib/docker-socket-path.js`, shared
326
+ with the dashboard's `lib/docker-socket.ts`); `lib/container-metrics.ts` reads it for
327
+ `GET /api/metrics/containers`, the agent panel and the Containers page. The tables are
328
+ created with `CREATE TABLE IF NOT EXISTS` when the daemon starts, so an update adds them;
329
+ until then the API answers `available: false`.
330
+
331
+ **Sampling:** containers every 60 s, on a timer of its own — a priming pass 6 s after the
332
+ daemon starts (it only records the gauges: a CPU figure needs two readings), the first
333
+ complete one 15 s later — and Docker's disk (`/system/df`) every 10 min, the first reading
334
+ 25 s after start. **Retention:** 30 days, pruned at every sample. A stopped container has no
335
+ rows: a gap, not zeros. Size, estimated: 60 s × 10 containers × 30 days ≈ 430 000 rows ≈ 35 MB
336
+ (to be measured on the VPS; rollups to 5 min may be added if it grows).
337
+
338
+ ```sql
339
+ -- One row per running container per sample. CPU in cores (CPU-seconds per second), rates in
340
+ -- bytes per second; each is NULL on the first sample after a start and when its counter went
341
+ -- backwards (the container restarted) — never a negative rate or a spike. Memory is the
342
+ -- working set: usage − inactive page cache, what `docker stats` shows.
343
+ CREATE TABLE container_metrics (
344
+ ts INTEGER NOT NULL, -- Unix seconds
345
+ container TEXT NOT NULL,
346
+ cpu_cores REAL,
347
+ mem_mb INTEGER,
348
+ pids INTEGER,
349
+ net_rx_bps INTEGER, net_tx_bps INTEGER,
350
+ blk_read_bps INTEGER, blk_write_bps INTEGER
351
+ );
352
+ CREATE INDEX idx_container_metrics_ts ON container_metrics(ts);
353
+ CREATE INDEX idx_container_metrics_c_ts ON container_metrics(container, ts);
354
+
355
+ -- Who each container is: an agent's name (AGENT_NAME), whether it is an agent, and its named
356
+ -- volumes (the link to docker_storage). Forgotten after the retention.
357
+ CREATE TABLE container_sources (
358
+ container TEXT PRIMARY KEY,
359
+ agent_id TEXT, name TEXT, is_agent INTEGER NOT NULL DEFAULT 0,
360
+ volumes TEXT, -- comma-separated volume names
361
+ last_seen INTEGER NOT NULL
362
+ );
363
+
364
+ -- Docker's own cores and memory: the denominators for a container's share (on Docker Desktop
365
+ -- that is its VM — 8 GB of a 16 GB Mac — not the machine). One row.
366
+ CREATE TABLE docker_host (id INTEGER PRIMARY KEY CHECK (id = 1), ncpu INTEGER, mem_total_mb INTEGER, updated_at INTEGER);
367
+
368
+ -- Disk Docker holds, one set of rows per reading: kind 'volume' (name = the volume;
369
+ -- reclaimable_mb = its size when no container uses it), 'layer' (name = a container, its
370
+ -- writable layer), 'images' and 'build_cache' (totals; reclaimable_mb = what no container /
371
+ -- no build uses). A size Docker could not compute is NULL.
372
+ CREATE TABLE docker_storage (ts INTEGER NOT NULL, kind TEXT NOT NULL, name TEXT NOT NULL DEFAULT '', size_mb INTEGER, reclaimable_mb INTEGER);
373
+ CREATE INDEX idx_docker_storage_ts ON docker_storage(ts);
374
+ CREATE INDEX idx_docker_storage_kind_ts ON docker_storage(kind, name, ts);
375
+ ```
376
+
321
377
  ## Costs Database (`costs.db`)
322
378
 
323
379
  What each agent spent, **as its own OpenClaw priced it** — Rev4a computes no cost here.