@flame0510/project-aether 1.3.0 → 1.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. package/README.md +1 -0
  2. package/agent-templates/README.md +42 -22
  3. package/agent-templates/base-image/Dockerfile +42 -33
  4. package/agent-templates/base-image/entrypoint.sh +67 -12
  5. package/app/agents/BrowserAccessSection.tsx +510 -0
  6. package/app/agents/ChannelManager.tsx +19 -11
  7. package/app/agents/ImageDownloadBanner.tsx +53 -19
  8. package/app/agents/ModelSection.tsx +4 -1
  9. package/app/agents/PageClient.tsx +629 -167
  10. package/app/agents/UpdateSection.tsx +300 -0
  11. package/app/agents/create/PageClient.tsx +11 -49
  12. package/app/agents/create/page.tsx +14 -28
  13. package/app/api/agents/[id]/backup/route.ts +26 -69
  14. package/app/api/agents/[id]/channels/pairing/route.ts +3 -3
  15. package/app/api/agents/[id]/channels/telegram/route.ts +2 -2
  16. package/app/api/agents/[id]/cold-backup/route.ts +56 -0
  17. package/app/api/agents/[id]/devices/route.ts +126 -0
  18. package/app/api/agents/[id]/invite-link/route.ts +53 -0
  19. package/app/api/agents/[id]/lifecycle/route.ts +3 -0
  20. package/app/api/agents/[id]/open-control-ui/route.ts +58 -0
  21. package/app/api/agents/[id]/recreate/route.ts +33 -163
  22. package/app/api/agents/[id]/restart/route.ts +5 -0
  23. package/app/api/agents/[id]/restore/route.ts +40 -70
  24. package/app/api/agents/[id]/route.ts +38 -150
  25. package/app/api/agents/[id]/update/rollback/route.ts +30 -0
  26. package/app/api/agents/[id]/update/route.ts +50 -0
  27. package/app/api/agents/activity-summary/route.ts +67 -0
  28. package/app/api/agents/create/route.ts +38 -92
  29. package/app/api/agents/devices-summary/route.ts +37 -0
  30. package/app/api/agents/download-image/route.ts +16 -9
  31. package/app/api/agents/image-status/route.ts +31 -111
  32. package/app/api/agents/route.ts +25 -49
  33. package/app/api/agents/token/route.ts +33 -10
  34. package/app/api/assistant/route.ts +2 -2
  35. package/app/api/gateway/agent/route.ts +14 -0
  36. package/app/api/gateway/provider/balance/route.ts +5 -2
  37. package/app/api/gateway/sync.ts +97 -14
  38. package/app/api/setup/agent-image/route.ts +14 -42
  39. package/app/api/version/route.ts +2 -1
  40. package/app/components/DashboardToolbar.tsx +1 -1
  41. package/app/gateway/PageClient.tsx +27 -32
  42. package/bin/rev4a.js +43 -41
  43. package/daemon.js +6 -6
  44. package/docs/ARCHITECTURE.md +107 -9
  45. package/docs/FRONTEND-ARCHITECTURE.md +8 -1
  46. package/docs/REV4A.md +54 -17
  47. package/docs/dev/API-REFERENCE.md +573 -105
  48. package/docs/dev/DATABASE.md +96 -0
  49. package/docs/dev/GATEWAY.md +21 -6
  50. package/docs/rag/DATA-FRESHNESS.md +6 -4
  51. package/docs/rag/GLOSSARY.md +12 -3
  52. package/docs/rag/REV4A-OVERVIEW.md +18 -5
  53. package/docs/rag/WHAT-I-CAN-ANSWER.md +6 -2
  54. package/instrumentation.ts +43 -0
  55. package/lib/agent-busy.ts +21 -0
  56. package/lib/agent-devices.ts +361 -0
  57. package/lib/agent-edit-state.ts +108 -0
  58. package/lib/agent-edit.ts +149 -0
  59. package/lib/agent-images.ts +375 -0
  60. package/lib/agent-ports-server.ts +78 -0
  61. package/lib/agent-ports.ts +91 -0
  62. package/lib/agent-recreate-state.ts +108 -0
  63. package/lib/agent-recreate.ts +359 -0
  64. package/lib/agent-restore-state.ts +107 -0
  65. package/lib/agent-restore.ts +141 -0
  66. package/lib/agent-setup.ts +66 -17
  67. package/lib/agent-update-state.ts +122 -0
  68. package/lib/agent-update.ts +456 -0
  69. package/lib/agent-versions.json +14 -0
  70. package/lib/agent-versions.ts +80 -0
  71. package/lib/buildAgentImage.ts +88 -290
  72. package/lib/channelManager.ts +149 -102
  73. package/lib/cold-backup.ts +354 -0
  74. package/lib/credentials/delivery.ts +3 -3
  75. package/lib/db-bootstrap.mjs +76 -0
  76. package/lib/docker-utils.ts +3 -3
  77. package/lib/provider-balance.ts +33 -12
  78. package/package.json +1 -1
@@ -0,0 +1,359 @@
1
+ /**
2
+ * Recreating an agent container on a given image while keeping everything else: the
3
+ * persistent volume, the AGENT_*, MODEL_*, OPENCLAW_* and TZ environment, the AGENT_ID
4
+ * and routing labels, the published ports and the network.
5
+ *
6
+ * Shared by the Update action (another version), the recreate action (same version,
7
+ * `startAgentRecreate`) and the edit route (PATCH, synchronous, no backup). The caller
8
+ * picks the image and takes care of backups.
9
+ *
10
+ * Arguments go to `docker` as an argv array, and environment values reach it through
11
+ * its own environment (`-e KEY`), so the gateway token never sits in the process table.
12
+ */
13
+ import { dockerFetch } from '@/lib/docker-socket';
14
+ import { runDocker, resolveRecreateImage } from '@/lib/agent-images';
15
+ import { hostPortsFromArgs } from '@/lib/agent-ports';
16
+ import { getHostPortHolders, describeHolder, type HostPortHolder } from '@/lib/agent-ports-server';
17
+ import { getMountFlags, applyRuntimeConfig } from '@/lib/agent-setup';
18
+ import { isValidAgentId } from '@/lib/container';
19
+ import { BACKUP_VOLUME, coldBackupStatus, isColdBackupRunning, startColdBackup, waitForColdBackup } from '@/lib/cold-backup';
20
+ import { waitForGatewayReady } from '@/lib/agent-readiness';
21
+ import { patchRev4aProvider } from '@/app/api/gateway/sync';
22
+ import { activeRecreateAgentIds, insertRecreate, isRecreateActive, latestRecreate, markInterruptedRecreates, updateRecreateRow, type AgentRecreateRow } from '@/lib/agent-recreate-state';
23
+ import { agentBusyReason } from '@/lib/agent-busy';
24
+
25
+ const KEEP_ENV_PREFIXES = ['AGENT_', 'MODEL_', 'OPENCLAW_', 'TZ='];
26
+ const KEEP_LABEL_PREFIXES = ['AGENT_ID', 'traefik.', 'description', 'maintainer'];
27
+
28
+ /** How long the recreated gateway may take to report ready. */
29
+ const RECREATE_READY_TIMEOUT_MS = 5 * 60 * 1000;
30
+
31
+ /** A cold backup of a 13 GB volume takes ~8 minutes; anything past this is stuck. */
32
+ const COLD_BACKUP_WAIT_TIMEOUT_MS = 30 * 60 * 1000;
33
+
34
+ export interface AgentContainerInspect {
35
+ Id?: string;
36
+ Name?: string;
37
+ Image?: string;
38
+ Config?: { Image?: string; Env?: string[]; Labels?: Record<string, string> };
39
+ State?: { Running?: boolean; Status?: string };
40
+ HostConfig?: { NetworkMode?: string; PortBindings?: Record<string, { HostPort?: string }[] | null> };
41
+ NetworkSettings?: { Networks?: Record<string, unknown>; Ports?: Record<string, unknown> };
42
+ }
43
+
44
+ /** The container carrying `AGENT_ID=<agentId>`, inspected; null when there is none. */
45
+ export async function inspectAgentContainer(agentId: string): Promise<AgentContainerInspect | null> {
46
+ const filters = encodeURIComponent(JSON.stringify({ label: [`AGENT_ID=${agentId}`] }));
47
+ const list = await dockerFetch<{ Id?: string }[]>('GET', `/containers/json?all=true&filters=${filters}`);
48
+ const id = Array.isArray(list) ? list[0]?.Id : undefined;
49
+ if (!id) return null;
50
+ const info = await dockerFetch<AgentContainerInspect & { Id?: string }>('GET', `/containers/${id}/json`);
51
+ return typeof info?.Id === 'string' ? info : null;
52
+ }
53
+
54
+ /**
55
+ * Start the container and confirm it really came back. A container can report
56
+ * `running` while its network endpoint failed to attach — the state a failed port bind
57
+ * leaves behind — which would otherwise pass as a successful start.
58
+ */
59
+ export async function startAndVerifyContainer(agentId: string, name: string): Promise<{ ok: true } | { ok: false; error: string }> {
60
+ try {
61
+ await runDocker(['start', name], { timeoutMs: 60_000 });
62
+ } catch (e) {
63
+ return { ok: false, error: shortDockerError(e) };
64
+ }
65
+ const after = await inspectAgentContainer(agentId).catch(() => null);
66
+ if (!after?.State?.Running) return { ok: false, error: 'the container is not running' };
67
+ if (Object.keys(after.NetworkSettings?.Networks ?? {}).length === 0) {
68
+ return { ok: false, error: 'the container came up without a network (one of its ports is taken by another container)' };
69
+ }
70
+ return { ok: true };
71
+ }
72
+
73
+ /** A docker failure carries a whole stack trace; the first line is the useful part. */
74
+ function shortDockerError(e: unknown): string {
75
+ const message = (e as Error)?.message ?? String(e);
76
+ return message.split('\n').map((l) => l.trim()).filter(Boolean)[0] ?? message;
77
+ }
78
+
79
+ /**
80
+ * Remove the container and run a new one on `image` with the same parameters. The new
81
+ * container is named after the agent id, as the create route names it. `opts.env`
82
+ * overrides single environment values (the edit route renames the agent this way),
83
+ * `opts.portArgs` replaces the port bindings carried over from the inspect.
84
+ */
85
+ export async function recreateAgentContainer(
86
+ agentId: string,
87
+ container: AgentContainerInspect,
88
+ image: string,
89
+ opts: { env?: Record<string, string>; portArgs?: string[] } = {},
90
+ ): Promise<void> {
91
+ // A running container carries its attached networks; a container whose network
92
+ // endpoint was lost (or one inspected while detached) reports none, so the mode it
93
+ // was created with is the fallback. 'default' is docker's alias for the default bridge.
94
+ const network = Object.keys(container.NetworkSettings?.Networks ?? {})[0]
95
+ ?? (container.HostConfig?.NetworkMode && container.HostConfig.NetworkMode !== 'default'
96
+ ? container.HostConfig.NetworkMode
97
+ : container.HostConfig?.NetworkMode === 'default' ? 'bridge' : undefined);
98
+ if (!network) throw new Error('Cannot determine the network: the container has none attached');
99
+
100
+ const env: Record<string, string> = {};
101
+ for (const entry of container.Config?.Env ?? []) {
102
+ if (!KEEP_ENV_PREFIXES.some((p) => entry.startsWith(p))) continue;
103
+ const eq = entry.indexOf('=');
104
+ if (eq > 0) env[entry.slice(0, eq)] = entry.slice(eq + 1);
105
+ }
106
+ Object.assign(env, opts.env ?? {});
107
+
108
+ const labelArgs = Object.entries(container.Config?.Labels ?? {})
109
+ .filter(([k]) => KEEP_LABEL_PREFIXES.some((p) => k === p || k.startsWith(p)))
110
+ .flatMap(([k, v]) => ['-l', `${k}=${v}`]);
111
+
112
+ const portArgs = opts.portArgs ?? Object.entries(container.HostConfig?.PortBindings ?? {}).flatMap(([containerPort, bindings]) => {
113
+ const hostPort = bindings?.[0]?.HostPort;
114
+ if (!hostPort) return [];
115
+ const proto = containerPort.includes('/udp') ? '/udp' : '';
116
+ return ['-p', `${hostPort}:${containerPort.replace(/\/.*$/, '')}${proto}`];
117
+ });
118
+
119
+ // Refuse on a port clash before anything is removed: the old container is the
120
+ // fallback the caller restarts, so a recreate refused here leaves the agent up. A
121
+ // clash cannot be recovered from by retrying, so it must not reach `docker run`.
122
+ const wanted = hostPortsFromArgs(portArgs);
123
+ if (wanted.length && container.Id) {
124
+ const held = await getHostPortHolders({ excludeContainerId: container.Id }).catch(() => new Map<number, HostPortHolder>());
125
+ const conflicts = wanted.filter((p) => held.has(p));
126
+ if (conflicts.length) {
127
+ const holders = [...new Set(conflicts.map((p) => describeHolder(held.get(p)!)))];
128
+ const many = conflicts.length > 1;
129
+ throw new Error(
130
+ `${many ? 'Ports' : 'Port'} ${conflicts.join(', ')} ${many ? 'are' : 'is'} already published by ${holders.join(', ')}. `
131
+ + 'Change this agent\'s port range in the Edit panel, then retry: nothing was changed.',
132
+ );
133
+ }
134
+ }
135
+
136
+ const oldName = (container.Name ?? '').replace(/^\//, '') || agentId;
137
+ await runDocker(['rm', '-f', oldName], { timeoutMs: 60_000 });
138
+
139
+ const runArgs = (img: string): string[] => [
140
+ 'run', '-d',
141
+ '--name', agentId,
142
+ '--network', network,
143
+ '--restart', 'unless-stopped',
144
+ '--add-host', 'host.docker.internal:host-gateway',
145
+ '-v', `agent-${agentId}-data:/root`,
146
+ ...getMountFlags(),
147
+ ...labelArgs,
148
+ ...portArgs,
149
+ img,
150
+ ];
151
+
152
+ try {
153
+ await runDocker(runArgs(image), { timeoutMs: 120_000, env });
154
+ } catch (first) {
155
+ // A failed `docker run` here would leave the agent down (the old container is
156
+ // already gone), so try once more. The usual cause is the image tag disappearing
157
+ // between the caller resolving it and this run (retention untags in the
158
+ // background) or a transient daemon error: re-resolving from the inspected
159
+ // container re-tags the image id when the tag is gone. A second failure propagates.
160
+ console.warn(`[agent:recreate] docker run failed for ${agentId}, retrying:`, (first as Error).message);
161
+ await runDocker(['rm', '-f', agentId], { timeoutMs: 60_000 }).catch(() => {});
162
+ const retryImage = await resolveRecreateImage(container).catch(() => image);
163
+ try {
164
+ await runDocker(runArgs(retryImage), { timeoutMs: 120_000, env });
165
+ } catch (second) {
166
+ // `docker run` creates the container before its network step, so a failed start
167
+ // leaves a container behind — one that reports `running` with no network and
168
+ // would be picked up as if it were the agent. Remove it, so the state the caller
169
+ // reports matches reality.
170
+ await runDocker(['rm', '-f', agentId], { timeoutMs: 60_000 }).catch(() => {});
171
+ throw second;
172
+ }
173
+ }
174
+ }
175
+
176
+ // ── The recreate action ──────────────────────────────────────────────────────
177
+
178
+ export class RecreateRefusedError extends Error {}
179
+
180
+ /** Recreate jobs running in this process, so a second start for the same agent is refused. */
181
+ const running = new Set<string>();
182
+
183
+ /**
184
+ * Start a recreate: a cold backup (the agent stops for it), then the container is
185
+ * rebuilt on the image it already runs, and the gateway is waited for. Returns once
186
+ * the job has started; the steps continue in the background and land in
187
+ * `agent_recreates`, so a reload or a Rev4a restart never loses track of it.
188
+ */
189
+ export async function startAgentRecreate(agentId: string): Promise<number> {
190
+ if (!isValidAgentId(agentId)) throw new RecreateRefusedError('Invalid agent id');
191
+ if (running.has(agentId)) throw new RecreateRefusedError('A recreate of this agent is already running');
192
+ // Claimed synchronously, before the first await: two requests arriving together must
193
+ // not both get past the guards (the second would later clear the first one's guard).
194
+ running.add(agentId);
195
+ try {
196
+ // One source for every long operation: an update, recreate, restore, edit or backup.
197
+ const busy = await agentBusyReason(agentId);
198
+ if (busy) throw new RecreateRefusedError(busy);
199
+
200
+ const container = await inspectAgentContainer(agentId);
201
+ if (!container) throw new RecreateRefusedError(`No container found with AGENT_ID '${agentId}'`);
202
+ // Refuse before any state is written when the image cannot be resolved.
203
+ const image = await resolveRecreateImage(container);
204
+
205
+ const id = insertRecreate(agentId, image);
206
+ void runRecreate(id, agentId, image).finally(() => running.delete(agentId));
207
+ return id;
208
+ } catch (e) {
209
+ running.delete(agentId);
210
+ throw e;
211
+ }
212
+ }
213
+
214
+ async function runRecreate(id: number, agentId: string, image: string): Promise<void> {
215
+ let stoppedByBackup = false;
216
+ try {
217
+ const { file } = await startColdBackup(agentId, { kind: 'prerecreate', leaveStopped: true });
218
+ stoppedByBackup = true;
219
+ updateRecreateRow(id, { backup_file: file });
220
+ const backup = await waitForColdBackup(agentId, { timeoutMs: COLD_BACKUP_WAIT_TIMEOUT_MS });
221
+ if (backup.status !== 'succeeded') throw new Error(`The pre-recreate backup failed: ${backup.error ?? 'unknown error'}`);
222
+
223
+ updateRecreateRow(id, { status: 'recreating' });
224
+ const container = await inspectAgentContainer(agentId);
225
+ if (!container) throw new Error('The agent container disappeared during the recreate');
226
+ await recreateAgentContainer(agentId, container, image);
227
+ stoppedByBackup = false;
228
+
229
+ if (!(await waitForGatewayReady(agentId, RECREATE_READY_TIMEOUT_MS))) {
230
+ throw new Error(`The gateway did not finish starting within ${RECREATE_READY_TIMEOUT_MS / 60_000} minutes`);
231
+ }
232
+ await applyRuntimeConfig(agentId);
233
+ try {
234
+ patchRev4aProvider(agentId);
235
+ } catch (e) {
236
+ console.error('[agent:recreate] provider patch failed:', (e as Error).message);
237
+ }
238
+
239
+ updateRecreateRow(id, { status: 'done' });
240
+
241
+ // Retention: keep the newest two prerecreate archives (lib/cold-backup.ts). Best-effort.
242
+ prunePrerecreateBackups(agentId)
243
+ .then((removed) => { if (removed.length) console.log(`[agent:recreate] removed old prerecreate backups: ${removed.join(', ')}`); })
244
+ .catch((err: unknown) => console.warn('[agent:recreate] backup cleanup failed:', (err as Error).message));
245
+ } catch (e) {
246
+ // A failure while the backup had stopped the old container and the recreate had
247
+ // not run yet: start the old container again, so a failed recreate never leaves
248
+ // the agent down. Nothing is left to start only when the rebuild itself failed
249
+ // after removing the container — the retry inside recreateAgentContainer already
250
+ // tried once, so the row says so and a new recreate is needed.
251
+ if (stoppedByBackup) {
252
+ const container = await inspectAgentContainer(agentId).catch(() => null);
253
+ const name = container?.Name?.replace(/^\//, '');
254
+ if (name && !container?.State?.Running) {
255
+ await runDocker(['start', name], { timeoutMs: 60_000 }).catch(() => {});
256
+ } else if (!container) {
257
+ console.error(`[agent:recreate] ${agentId}: the container is gone after the failed rebuild; run Recreate again`);
258
+ }
259
+ }
260
+ updateRecreateRow(id, { status: 'failed', error: shortError(e) });
261
+ }
262
+ }
263
+
264
+ /** The useful part of an error for the panel: a `docker` failure carries a whole stack trace. */
265
+ function shortError(e: unknown): string {
266
+ const message = (e as Error)?.message ?? String(e);
267
+ const lines = message.split('\n').map((l) => l.trim()).filter(Boolean);
268
+ return lines[0] ?? message;
269
+ }
270
+
271
+ export interface AgentRecreateView {
272
+ id: number;
273
+ agentId: string;
274
+ status: AgentRecreateRow['status'];
275
+ image: string | null;
276
+ backupFile: string | null;
277
+ /** Cold backup progress while `backing_up`. */
278
+ backupPercent: number | null;
279
+ error: string | null;
280
+ startedAtMs: number;
281
+ finishedAtMs: number | null;
282
+ }
283
+
284
+ /** The latest recreate of an agent, with live backup progress. */
285
+ export async function agentRecreateView(agentId: string): Promise<AgentRecreateView | null> {
286
+ const row = latestRecreate(agentId);
287
+ if (!row) return null;
288
+ const backupPercent = row.status === 'backing_up' ? (await coldBackupStatus(agentId).catch(() => null))?.percent ?? null : null;
289
+ return {
290
+ id: row.id,
291
+ agentId: row.agent_id,
292
+ status: row.status,
293
+ image: row.image,
294
+ backupFile: row.backup_file,
295
+ backupPercent,
296
+ error: row.error,
297
+ startedAtMs: row.started_at,
298
+ finishedAtMs: row.finished_at,
299
+ };
300
+ }
301
+
302
+ /**
303
+ * At startup, no recreate job can be running: rows still active were cut off by the
304
+ * restart. They become `interrupted`; an agent the backup had stopped is started
305
+ * again, best-effort — but never while its backup helper is still archiving the
306
+ * volume, or the archive would be inconsistent. Startup is not held for that: each
307
+ * agent is finished in the background.
308
+ */
309
+ export async function recoverInterruptedRecreates(): Promise<number> {
310
+ const ids = activeRecreateAgentIds();
311
+ if (ids.size === 0) return 0;
312
+ const interrupted = markInterruptedRecreates();
313
+ for (const agentId of ids) void finishInterruptedRecreate(agentId);
314
+ return interrupted;
315
+ }
316
+
317
+ async function finishInterruptedRecreate(agentId: string): Promise<void> {
318
+ // A helper that outlived the restart keeps archiving; wait for it before touching
319
+ // the agent. `waitForColdBackup` resolves immediately with the last job when none runs.
320
+ const running = await isColdBackupRunning(agentId).catch(() => false);
321
+ if (running) await waitForColdBackup(agentId).catch(() => null);
322
+ const container = await inspectAgentContainer(agentId).catch(() => null);
323
+ const name = container?.Name?.replace(/^\//, '');
324
+ if (!name || container?.State?.Running) return;
325
+ const started = await startAndVerifyContainer(agentId, name);
326
+ if (!started.ok) console.warn(`[agent:recreate] ${agentId} did not come back after an interrupted recreate: ${started.error}`);
327
+ }
328
+
329
+ // ── Prerecreate retention ────────────────────────────────────────────────────
330
+
331
+ /** How many prerecreate archives to keep per agent once a recreate has committed. */
332
+ const KEEP_PRERECREATE = 2;
333
+
334
+ /**
335
+ * Delete the agent's older prerecreate archives, keeping the newest
336
+ * `KEEP_PRERECREATE`. Preupdate backups are not touched: they are the Update
337
+ * rollback point. Returns the file names removed.
338
+ */
339
+ export async function prunePrerecreateBackups(agentId: string): Promise<string[]> {
340
+ const pattern = `agent-${agentId}-prerecreate-*.tar.gz`;
341
+ const stale: string[] = [];
342
+ await runDocker(
343
+ // $PATTERN stays unquoted so the shell expands the wildcard; agent ids are
344
+ // validated, so the pattern carries no shell metacharacters beyond the glob.
345
+ ['run', '--rm', '-v', `${BACKUP_VOLUME}:/backup`, 'alpine', 'sh', '-c',
346
+ 'cd /backup && ls -1t $PATTERN 2>/dev/null | tail -n +3'],
347
+ { timeoutMs: 60_000, env: { PATTERN: pattern }, onLine: (line) => {
348
+ const name = line.trim();
349
+ if (name.endsWith('.tar.gz')) stale.push(name);
350
+ } },
351
+ ).catch(() => {});
352
+ for (const file of stale) {
353
+ await runDocker(
354
+ ['run', '--rm', '-v', `${BACKUP_VOLUME}:/backup`, 'alpine', 'sh', '-c', 'rm -f "/backup/$FILE"'],
355
+ { timeoutMs: 60_000, env: { FILE: file } },
356
+ ).catch(() => {});
357
+ }
358
+ return stale;
359
+ }
@@ -0,0 +1,107 @@
1
+ /**
2
+ * Persisted state of agent restores: the `agent_restores` table (lib/db-bootstrap.mjs).
3
+ *
4
+ * Kept apart from lib/agent-restore.ts so busy checks (lib/agent-busy.ts) can ask which
5
+ * agents are being restored without importing the restore action.
6
+ */
7
+ import { openDb } from '@/lib/db';
8
+
9
+ export type RestoreStatus = 'restoring' | 'done' | 'failed' | 'interrupted';
10
+
11
+ /** Statuses during which the agent must not be touched by anything else. */
12
+ export const ACTIVE_RESTORE_STATUSES: readonly RestoreStatus[] = ['restoring'];
13
+
14
+ export interface AgentRestoreRow {
15
+ id: number;
16
+ agent_id: string;
17
+ status: RestoreStatus;
18
+ file: string | null;
19
+ error: string | null;
20
+ started_at: number;
21
+ updated_at: number;
22
+ finished_at: number | null;
23
+ }
24
+
25
+ const FINAL: readonly RestoreStatus[] = ['done', 'failed', 'interrupted'];
26
+
27
+ export function insertRestore(agentId: string, file: string): number {
28
+ const db = openDb(false);
29
+ try {
30
+ const now = Date.now();
31
+ const info = db.prepare(
32
+ `INSERT INTO agent_restores (agent_id, status, file, started_at, updated_at)
33
+ VALUES (?, 'restoring', ?, ?, ?)`,
34
+ ).run(agentId, file, now, now);
35
+ return Number(info.lastInsertRowid);
36
+ } finally {
37
+ db.close();
38
+ }
39
+ }
40
+
41
+ type Writable = Partial<Pick<AgentRestoreRow, 'status' | 'error'>>;
42
+
43
+ export function updateRestoreRow(id: number, fields: Writable): void {
44
+ const entries = Object.entries(fields).filter(([, v]) => v !== undefined);
45
+ const now = Date.now();
46
+ const sets = [...entries.map(([k]) => `${k} = ?`), 'updated_at = ?'];
47
+ const values: unknown[] = [...entries.map(([, v]) => v), now];
48
+ if (fields.status && FINAL.includes(fields.status)) {
49
+ sets.push('finished_at = ?');
50
+ values.push(now);
51
+ } else if (fields.status) {
52
+ sets.push('finished_at = NULL');
53
+ }
54
+ const db = openDb(false);
55
+ try {
56
+ db.prepare(`UPDATE agent_restores SET ${sets.join(', ')} WHERE id = ?`).run(...values, id);
57
+ } finally {
58
+ db.close();
59
+ }
60
+ }
61
+
62
+ export function latestRestore(agentId: string): AgentRestoreRow | null {
63
+ const db = openDb(true);
64
+ try {
65
+ return (db.prepare('SELECT * FROM agent_restores WHERE agent_id = ? ORDER BY id DESC LIMIT 1').get(agentId) as AgentRestoreRow | undefined) ?? null;
66
+ } finally {
67
+ db.close();
68
+ }
69
+ }
70
+
71
+ export function isRestoreActive(agentId: string): boolean {
72
+ const row = latestRestore(agentId);
73
+ return !!row && ACTIVE_RESTORE_STATUSES.includes(row.status);
74
+ }
75
+
76
+ /** AGENT_IDs with a restore in an active status. */
77
+ export function activeRestoreAgentIds(): Set<string> {
78
+ const db = openDb(true);
79
+ try {
80
+ const placeholders = ACTIVE_RESTORE_STATUSES.map(() => '?').join(', ');
81
+ const rows = db.prepare(`SELECT DISTINCT agent_id FROM agent_restores WHERE status IN (${placeholders})`).all(...ACTIVE_RESTORE_STATUSES) as { agent_id: string }[];
82
+ return new Set(rows.map((r) => r.agent_id));
83
+ } finally {
84
+ db.close();
85
+ }
86
+ }
87
+
88
+ /**
89
+ * At startup no restore job can be running: a row still active was cut off by the
90
+ * restart. It becomes `interrupted`, keeping the archive it was restoring.
91
+ */
92
+ export function markInterruptedRestores(): number {
93
+ const db = openDb(false);
94
+ try {
95
+ const placeholders = ACTIVE_RESTORE_STATUSES.map(() => '?').join(', ');
96
+ const now = Date.now();
97
+ const info = db.prepare(
98
+ `UPDATE agent_restores
99
+ SET error = COALESCE(error, 'Rev4a restarted while this step was running: ' || status),
100
+ status = 'interrupted', updated_at = ?, finished_at = ?
101
+ WHERE status IN (${placeholders})`,
102
+ ).run(now, now, ...ACTIVE_RESTORE_STATUSES);
103
+ return info.changes;
104
+ } finally {
105
+ db.close();
106
+ }
107
+ }
@@ -0,0 +1,141 @@
1
+ /**
2
+ * Restoring an agent's volume from a backup archive.
3
+ *
4
+ * The volume is cleared and the archive extracted into it — minutes for a large
5
+ * workspace — then the container is started again even when the extract failed. The
6
+ * job is recorded in `agent_restores` and runs in the background, so a page reload or
7
+ * a Rev4a restart never loses track of it and a second restore cannot start on top of
8
+ * the first (lib/agent-busy.ts answers 409 while one runs).
9
+ */
10
+ import { isValidAgentId } from '@/lib/container';
11
+ import { runDocker } from '@/lib/agent-images';
12
+ import { BACKUP_VOLUME } from '@/lib/cold-backup';
13
+ import { inspectAgentContainer, startAndVerifyContainer } from '@/lib/agent-recreate';
14
+ import { activeRestoreAgentIds, insertRestore, isRestoreActive, latestRestore, markInterruptedRestores, updateRestoreRow, type AgentRestoreRow } from '@/lib/agent-restore-state';
15
+ import { agentBusyReason } from '@/lib/agent-busy';
16
+
17
+ /** A 13 GB workspace takes minutes to decompress; 30 minutes leaves room and still ends. */
18
+ const EXTRACT_TIMEOUT_MS = 30 * 60 * 1000;
19
+
20
+ export class RestoreRefusedError extends Error {}
21
+
22
+ /** Archive names are `<agent>-<something>.tar.gz`; nothing that could carry shell syntax. */
23
+ function isValidBackupFile(agentId: string, file: string): boolean {
24
+ return file.startsWith(`agent-${agentId}-`) && /^[A-Za-z0-9._-]+\.tar\.gz$/.test(file);
25
+ }
26
+
27
+ /** Restore jobs running in this process, so a second start for the same agent is refused. */
28
+ const running = new Set<string>();
29
+
30
+ export async function startAgentRestore(agentId: string, file: string): Promise<number> {
31
+ if (!isValidAgentId(agentId)) throw new RestoreRefusedError('Invalid agent id');
32
+ if (!isValidBackupFile(agentId, file)) throw new RestoreRefusedError('Invalid backup file name');
33
+ if (running.has(agentId) || isRestoreActive(agentId)) throw new RestoreRefusedError('A restore of this agent is already running');
34
+ // Claimed synchronously, before the first await, so two requests cannot both pass.
35
+ running.add(agentId);
36
+ try {
37
+ // One source for every long operation: an update, recreate, restore, edit or backup.
38
+ const busy = await agentBusyReason(agentId);
39
+ if (busy) throw new RestoreRefusedError(busy);
40
+
41
+ // The archive must exist before anything is stopped or cleared. It used to be
42
+ // shape-checked only, so a file that had been pruned or deleted made the job wipe
43
+ // the volume and then fail, leaving the agent on an empty volume.
44
+ await runDocker(
45
+ ['run', '--rm', '-v', `${BACKUP_VOLUME}:/backup`, 'alpine', 'sh', '-c', 'test -f "/backup/$FILE"'],
46
+ { timeoutMs: 30_000, env: { FILE: file } },
47
+ ).catch(() => {
48
+ throw new RestoreRefusedError(`Backup file '${file}' not found`);
49
+ });
50
+
51
+ const id = insertRestore(agentId, file);
52
+ void runRestore(id, agentId, file).finally(() => running.delete(agentId));
53
+ return id;
54
+ } catch (e) {
55
+ running.delete(agentId);
56
+ throw e;
57
+ }
58
+ }
59
+
60
+ async function runRestore(id: number, agentId: string, file: string): Promise<void> {
61
+ const container = await inspectAgentContainer(agentId).catch(() => null);
62
+ const name = container?.Name?.replace(/^\//, '');
63
+ let failure: string | null = null;
64
+ try {
65
+ if (name && container?.State?.Running) {
66
+ await runDocker(['stop', '-t', '30', name], { timeoutMs: 45_000 }).catch(() => {});
67
+ }
68
+
69
+ // Clear the volume, then extract. The file name reaches the container through its
70
+ // environment and the whole command is one argv element, so it is never shell syntax.
71
+ await runDocker(
72
+ [
73
+ 'run', '--rm',
74
+ '-v', `agent-${agentId}-data:/target`,
75
+ '-v', `${BACKUP_VOLUME}:/backup`,
76
+ 'alpine', 'sh', '-c',
77
+ 'test -f "/backup/$FILE" || exit 3; rm -rf /target/* /target/.[!.]* /target/..?* 2>/dev/null; tar xzf "/backup/$FILE" -C /target',
78
+ ],
79
+ { timeoutMs: EXTRACT_TIMEOUT_MS, env: { FILE: file } },
80
+ );
81
+ } catch (e) {
82
+ failure = (e as Error)?.message ?? String(e);
83
+ } finally {
84
+ // The container is started again even when the extract failed, so the agent never
85
+ // stays down. The row is only `done` when it really came back: a container can
86
+ // report running while its network never attached, which is not a usable agent.
87
+ if (name) {
88
+ const started = await startAndVerifyContainer(agentId, name);
89
+ if (!started.ok) {
90
+ const note = `the container did not come back: ${started.error}`;
91
+ failure = failure ? `${failure}; ${note}` : `the volume was restored, but ${note}`;
92
+ }
93
+ }
94
+ if (failure) updateRestoreRow(id, { status: 'failed', error: failure });
95
+ else updateRestoreRow(id, { status: 'done' });
96
+ }
97
+ }
98
+
99
+ export interface AgentRestoreView {
100
+ id: number;
101
+ agentId: string;
102
+ status: AgentRestoreRow['status'];
103
+ file: string | null;
104
+ error: string | null;
105
+ startedAtMs: number;
106
+ finishedAtMs: number | null;
107
+ }
108
+
109
+ /** The latest restore of an agent, so the panel can resume after a reload. */
110
+ export function agentRestoreView(agentId: string): AgentRestoreView | null {
111
+ const row = latestRestore(agentId);
112
+ if (!row) return null;
113
+ return {
114
+ id: row.id,
115
+ agentId: row.agent_id,
116
+ status: row.status,
117
+ file: row.file,
118
+ error: row.error,
119
+ startedAtMs: row.started_at,
120
+ finishedAtMs: row.finished_at,
121
+ };
122
+ }
123
+
124
+ /**
125
+ * At startup no restore job can be running: rows still active were cut off by the
126
+ * restart. They become `interrupted`; a container left stopped is started again,
127
+ * best-effort, so nothing stays down.
128
+ */
129
+ export async function recoverInterruptedRestores(): Promise<number> {
130
+ const ids = activeRestoreAgentIds();
131
+ if (ids.size === 0) return 0;
132
+ const interrupted = markInterruptedRestores();
133
+ for (const agentId of ids) {
134
+ const container = await inspectAgentContainer(agentId).catch(() => null);
135
+ const name = container?.Name?.replace(/^\//, '');
136
+ if (!name || container?.State?.Running) continue;
137
+ const started = await startAndVerifyContainer(agentId, name);
138
+ if (!started.ok) console.warn(`[agent:restore] ${agentId} did not come back after an interrupted restore: ${started.error}`);
139
+ }
140
+ return interrupted;
141
+ }