@volter/twin-fly 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,539 @@
1
+ // The REAL execution plane: `FlyContainerRuntime` implemented over the LOCAL Docker daemon via
2
+ // the `docker` CLI. This is what makes the fly twin a REAL-PLANE twin (the supabase doctrine at
3
+ // the compute layer): the control plane is twinned, but a created/started machine genuinely runs
4
+ // `config.image` as a local container — config.env (plus Fly's documented FLY_* runtime
5
+ // environment) becomes the container env, every service's internal_port is published on an
6
+ // ephemeral LOOPBACK port, config.mounts become named docker volumes, stop is `docker stop`,
7
+ // start RECREATES the container from the image (Fly resets a stopped Machine's rootfs — only
8
+ // mounted volumes survive), destroy is `docker rm -f`, exec is `docker exec`.
9
+ //
10
+ // This module makes NO vendor network calls (D4 is about the vendor; the Docker daemon is local
11
+ // infrastructure, the same class as supabase's real Postgres) and is NEVER exercised by
12
+ // capability verifies or the gate: the handler defaults to the pure-ledger VIRTUAL runtime, and
13
+ // this file's only proof is fly-docker.integration.test.ts, which self-skips LOUDLY when
14
+ // `docker info` fails. `dockerAvailable()` probes the daemon (not just the binary), the
15
+ // cookbook/saas-ai-supabase/run.ts pattern.
16
+ //
17
+ // Naming: every artifact is namespaced `fly-twin-…` so a crashed run can be swept with
18
+ // `docker ps -aq --filter label=dev.volter.fly-twin | xargs docker rm -f`.
19
+ import { spawn, spawnSync } from 'node:child_process';
20
+ import { createHash } from 'node:crypto';
21
+ import { existsSync, mkdirSync, readdirSync, statfsSync } from 'node:fs';
22
+ import { homedir } from 'node:os';
23
+ import { join, resolve } from 'node:path';
24
+ import { withFileLock } from '@volter/world-core';
25
+ import type { FlyContainerRuntime, FlyContainerSpec, FlyExecResult } from './fly-machines.ts';
26
+
27
+ export const FLY_DOCKER_LABEL = 'dev.volter.fly-twin';
28
+ const FLY_WORLD_OWNER_LABEL = 'dev.volter.world-owner';
29
+ const FLY_MEMORY_LABEL = 'dev.volter.memory-mib';
30
+ const FLY_CPU_LABEL = 'dev.volter.cpus';
31
+ const MIB = 1024 * 1024;
32
+
33
+ /** Writable storage a machine needs for itself before its own memory-sized footprint counts: an
34
+ * image pull plus a writable layer. A property of the substrate, not a knob. */
35
+ const IMAGE_FOOTPRINT_MIB = 2048;
36
+
37
+ /** MiB of writable storage admission keeps FREE beyond the machine's own need — the DEFAULT of a
38
+ * configured mechanism value, not a constant. A World whose host disk is smaller than the
39
+ * reserve configures it down (`world-fly serve --storage-reserve-mib N`,
40
+ * `FlyDockerRuntimeOptions.storageReserveMiB`); the runtime never lowers its own floor. */
41
+ export const DEFAULT_STORAGE_RESERVE_MIB = 2048;
42
+
43
+ /** Options of the REAL execution plane. Configuration reaches this runtime the way the plane
44
+ * itself does — resolved at the HOST BOUNDARY (cli.ts via fly-runtime-choice.ts) and handed in,
45
+ * never read from the process here. */
46
+ export interface FlyDockerRuntimeOptions {
47
+ /** MiB kept free beyond a machine's own writable-storage need (default 2048). */
48
+ storageReserveMiB?: number;
49
+ }
50
+
51
+ function docker(args: string[], opts: { timeoutMs?: number } = {}): { status: number; stdout: string; stderr: string } {
52
+ const r = spawnSync('docker', args, { encoding: 'utf8', timeout: opts.timeoutMs ?? 120_000 });
53
+ if (r.error) throw new Error(`docker ${args[0]}: ${r.error.message}`);
54
+ return { status: r.status ?? 1, stdout: r.stdout ?? '', stderr: r.stderr ?? '' };
55
+ }
56
+
57
+ /**
58
+ * The writable-storage floor for ONE machine: its own need (never below an image pull plus a
59
+ * writable layer) plus the reserve the host keeps free. Pure arithmetic, so the admission rule is
60
+ * checkable without a daemon.
61
+ */
62
+ export function writableStorageFloorMiB(
63
+ memoryMiB: number,
64
+ reserveMiB: number = DEFAULT_STORAGE_RESERVE_MIB,
65
+ ): { requiredMiB: number; reserveMiB: number; floorMiB: number } {
66
+ const requiredMiB = Math.max(IMAGE_FOOTPRINT_MIB, memoryMiB);
67
+ return { requiredMiB, reserveMiB, floorMiB: requiredMiB + reserveMiB };
68
+ }
69
+
70
+ /** What the writable-storage probe measured, and the SUBSTRATE it measured it on (named in the
71
+ * capacity log so a refusal is diagnosable; the caller's error stays in World vocabulary). */
72
+ export type WritableStorage = { availableMiB: number; substrate: string };
73
+
74
+ /**
75
+ * The host facts the writable-storage probe reads. Injected — the same seam doctrine the
76
+ * execution plane itself uses — so every branch (host-visible docker root, Colima, OrbStack,
77
+ * Docker Desktop) is provable OFFLINE against fakes, with no daemon anywhere.
78
+ */
79
+ export interface StorageProbeHost {
80
+ readonly platform: string;
81
+ readonly home: string;
82
+ /** Does this HOST path exist? */
83
+ exists(path: string): boolean;
84
+ /** Free MiB on the host volume holding `path`; undefined when it cannot be read. */
85
+ freeMiB(path: string): number | undefined;
86
+ /** Entries of a host directory; [] when unreadable. */
87
+ entries(path: string): string[];
88
+ docker(args: string[]): { status: number; stdout: string; stderr: string };
89
+ run(command: string, args: string[], timeoutMs: number): { status: number | null; stdout: string; stderr: string };
90
+ warn(line: string): void;
91
+ }
92
+
93
+ const HOST_STORAGE_PROBE: StorageProbeHost = {
94
+ platform: process.platform,
95
+ home: homedir(),
96
+ exists: (path) => existsSync(path),
97
+ freeMiB(path) {
98
+ try {
99
+ const fs = statfsSync(path);
100
+ return Math.floor((Number(fs.bavail) * Number(fs.bsize)) / MIB);
101
+ } catch {
102
+ return undefined;
103
+ }
104
+ },
105
+ entries(path) {
106
+ try {
107
+ return readdirSync(path);
108
+ } catch {
109
+ return [];
110
+ }
111
+ },
112
+ docker: (args) => docker(args),
113
+ run(command, args, timeoutMs) {
114
+ const r = spawnSync(command, args, { encoding: 'utf8', timeout: timeoutMs });
115
+ return { status: r.status, stdout: r.stdout ?? '', stderr: r.stderr ?? '' };
116
+ },
117
+ warn: (line) => { process.stderr.write(line); },
118
+ };
119
+
120
+ /**
121
+ * How much writable storage the local execution substrate can still take, measured INSIDE that
122
+ * substrate when the docker root is not host-visible. Honest by construction: every branch either
123
+ * MEASURES a real filesystem or returns undefined (admission then refuses) — capacity is never
124
+ * assumed, defaulted or fabricated.
125
+ */
126
+ export function probeWritableStorage(rootDir: string, host: StorageProbeHost = HOST_STORAGE_PROBE): WritableStorage | undefined {
127
+ if (rootDir && host.exists(rootDir)) {
128
+ const availableMiB = host.freeMiB(rootDir);
129
+ if (availableMiB === undefined) {
130
+ host.warn(`[capacity] writable-storage verification failed: docker root "${rootDir}" could not be measured\n`);
131
+ return undefined;
132
+ }
133
+ return { availableMiB, substrate: `docker root "${rootDir}"` };
134
+ }
135
+ // The daemon keeps its data in a hidden Linux machine. Detect WHICH one from the active
136
+ // endpoint (and, for the substrates that also install the classic socket path, from the daemon's
137
+ // own identity), then measure that machine's storage; this detail never crosses the service log.
138
+ const context = host.docker(['context', 'inspect', '--format', '{{json .Endpoints.docker.Host}}']);
139
+ let endpoint = '';
140
+ try { endpoint = context.status === 0 ? JSON.parse(context.stdout.trim() || '""') as string : ''; } catch {
141
+ host.warn('[capacity] writable-storage verification failed: the active docker context endpoint could not be read\n');
142
+ return undefined;
143
+ }
144
+ // `~/.colima/<profile>/docker.sock` is colima's own documented DOCKER_HOST; a colima install
145
+ // that predates the dotted directory uses `.../colima/<profile>/docker.sock`. Both are the same
146
+ // substrate, so the leading dot is optional here — anchored to a path segment so no unrelated
147
+ // directory ending in "colima" can claim the branch.
148
+ const colima = /(?:^|\/)\.?colima\/([^/]+)\/docker\.sock$/u.exec(endpoint);
149
+ if (colima) return colimaStorage(colima[1]!, rootDir, host);
150
+ // OrbStack is asked about BEFORE Docker Desktop: both keep a growable image under the user's
151
+ // Library and both may be installed at once, so the ACTIVE daemon — not whichever directory
152
+ // happens to exist — decides which image is the one being written to.
153
+ if (host.platform === 'darwin' && isOrbStack(endpoint, host)) return orbstackStorage(host);
154
+ // Docker Desktop keeps its Linux machine's disk as a growable image under the user's
155
+ // Library container; free space on the volume HOLDING that image is the binding
156
+ // constraint on how much more the daemon can write, and it is host-visible — no
157
+ // container run, no network. (The VM's internal ceiling is typically far larger.)
158
+ const desktopData = join(host.home, 'Library', 'Containers', 'com.docker.docker', 'Data');
159
+ if (host.platform === 'darwin' && host.exists(desktopData)) {
160
+ const availableMiB = host.freeMiB(desktopData);
161
+ if (availableMiB === undefined) {
162
+ host.warn(`[capacity] writable-storage verification failed: the Docker Desktop data volume "${desktopData}" could not be measured\n`);
163
+ return undefined;
164
+ }
165
+ return { availableMiB, substrate: `Docker Desktop data volume "${desktopData}"` };
166
+ }
167
+ host.warn(`[capacity] writable-storage verification failed: docker root "${rootDir}" is not host-visible and endpoint "${endpoint}" is neither a Colima profile, OrbStack, nor Docker Desktop on macOS\n`);
168
+ return undefined;
169
+ }
170
+
171
+ /** Is the ACTIVE daemon OrbStack? Its own socket path is the cheap tell; OrbStack also takes over
172
+ * the classic `/var/run/docker.sock`, so an endpoint that says nothing is settled by asking the
173
+ * daemon what it is — `docker info` reports the operating system literally as "OrbStack". */
174
+ function isOrbStack(endpoint: string, host: StorageProbeHost): boolean {
175
+ if (/\/\.orbstack\/run\/docker\.sock$/u.test(endpoint)) return true;
176
+ const info = host.docker(['info', '--format', '{{.OperatingSystem}}']);
177
+ return info.status === 0 && /^orbstack$/iu.test(info.stdout.trim());
178
+ }
179
+
180
+ /** OrbStack keeps the WHOLE Linux machine — images, containers, volumes — in one SPARSE
181
+ * `data.img.raw` inside its group container ("it only takes as much space as you use, and
182
+ * automatically shrinks when you delete data", per the file's own README), with no disk ceiling
183
+ * configured by default. So free space on the host volume HOLDING that image is the binding
184
+ * constraint on how much more the daemon can write, and it is host-visible — no container run,
185
+ * no ssh, no network. Same shape as the Docker Desktop branch. The group container carries
186
+ * Apple's team prefix, so it is FOUND rather than hard-coded; when it cannot be found the probe
187
+ * refuses instead of guessing a volume. */
188
+ function orbstackStorage(host: StorageProbeHost): WritableStorage | undefined {
189
+ const groupContainers = join(host.home, 'Library', 'Group Containers');
190
+ const container = host.entries(groupContainers).filter((entry) => entry.endsWith('.dev.orbstack')).sort()[0];
191
+ const dataDir = container === undefined ? undefined : join(groupContainers, container, 'data');
192
+ if (dataDir === undefined || !host.exists(dataDir)) {
193
+ host.warn(`[capacity] writable-storage verification failed: the active daemon is OrbStack but its data image was not found under "${groupContainers}"\n`);
194
+ return undefined;
195
+ }
196
+ const availableMiB = host.freeMiB(dataDir);
197
+ if (availableMiB === undefined) {
198
+ host.warn(`[capacity] writable-storage verification failed: the OrbStack data volume "${dataDir}" could not be measured\n`);
199
+ return undefined;
200
+ }
201
+ return { availableMiB, substrate: `OrbStack data volume "${dataDir}"` };
202
+ }
203
+
204
+ /** Colima keeps the daemon in a hidden Linux machine reachable over its own ssh config; query
205
+ * that filesystem internally. */
206
+ function colimaStorage(profile: string, rootDir: string, host: StorageProbeHost): WritableStorage | undefined {
207
+ const config = host.run('colima', ['-p', profile, 'ssh-config'], 5_000);
208
+ if (config.status !== 0) {
209
+ host.warn(`[capacity] writable-storage verification failed: colima ssh-config for profile "${profile}" exited ${String(config.status)}\n`);
210
+ return undefined;
211
+ }
212
+ const field = (name: string): string | undefined => {
213
+ const value = new RegExp(`^\\s*${name}\\s+(.+?)\\s*$`, 'mu').exec(config.stdout)?.[1];
214
+ return value?.replace(/^"|"$/gu, '');
215
+ };
216
+ const identity = field('IdentityFile');
217
+ const user = field('User');
218
+ const hostname = field('Hostname');
219
+ const port = field('Port');
220
+ const controlPath = field('ControlPath');
221
+ if (!identity || !user || !hostname || !port) {
222
+ host.warn(`[capacity] writable-storage verification failed: colima ssh-config for profile "${profile}" is missing connection fields\n`);
223
+ return undefined;
224
+ }
225
+ const status = host.run('ssh', [
226
+ '-o', 'StrictHostKeyChecking=no', '-o', 'UserKnownHostsFile=/dev/null', '-o', 'BatchMode=yes',
227
+ '-o', 'IdentitiesOnly=yes', ...(controlPath ? ['-o', 'ControlMaster=auto', '-o', `ControlPath=${controlPath}`, '-o', 'ControlPersist=yes'] : []),
228
+ '-i', identity, '-p', port, `${user}@${hostname}`, 'df', '-Pk', rootDir || '/',
229
+ ], 5_000);
230
+ if (status.status !== 0) {
231
+ host.warn(`[capacity] internal writable-storage inspection failed: ${(status.stderr || '').trim()}\n`);
232
+ return undefined;
233
+ }
234
+ const line = status.stdout.trim().split(/\r?\n/u).at(-1) ?? '';
235
+ const fields = line.trim().split(/\s+/u);
236
+ const availableKiB = Number(fields[3]);
237
+ if (!Number.isFinite(availableKiB)) {
238
+ host.warn(`[capacity] writable-storage verification failed: colima profile "${profile}" reported an unreadable df line "${line}"\n`);
239
+ return undefined;
240
+ }
241
+ return { availableMiB: Math.floor(availableKiB / 1024), substrate: `colima profile "${profile}"` };
242
+ }
243
+
244
+ /** Is a Docker DAEMON actually reachable (not just the binary present)? */
245
+ export function dockerAvailable(): boolean {
246
+ try {
247
+ const r = spawnSync('docker', ['info'], { stdio: 'ignore', timeout: 15_000 });
248
+ return r.status === 0;
249
+ } catch {
250
+ return false;
251
+ }
252
+ }
253
+
254
+ export function flyContainerName(machineId: string): string {
255
+ return `fly-twin-${machineId}`;
256
+ }
257
+ export function flyVolumeName(volumeId: string): string {
258
+ return `fly-twin-${volumeId}`;
259
+ }
260
+
261
+ export class FlyDockerRuntime implements FlyContainerRuntime {
262
+ readonly kind = 'docker';
263
+ readonly #ownerId: string;
264
+ readonly #logFollowers = new Map<string, ReturnType<typeof spawn>>();
265
+ readonly #machineIds = new Map<string, string>();
266
+ readonly #capacityLock: string;
267
+ readonly #storageReserveMiB: number;
268
+
269
+ constructor(root = process.cwd(), options: FlyDockerRuntimeOptions = {}) {
270
+ // A malformed reserve FAILS here rather than silently reverting to the default: a World that
271
+ // asked for a different admission floor and got the old one would be told nothing.
272
+ const reserve = options.storageReserveMiB;
273
+ if (reserve !== undefined && (!Number.isInteger(reserve) || reserve < 0)) {
274
+ throw new Error(`machine provider: storage reserve must be a non-negative whole number of MiB (got ${JSON.stringify(reserve)})`);
275
+ }
276
+ this.#storageReserveMiB = reserve ?? DEFAULT_STORAGE_RESERVE_MIB;
277
+ // The service data-root is stable for the lifetime of one named World and
278
+ // opaque outside this implementation. Hashing it gives every backing
279
+ // resource an exact lifecycle owner without exposing host paths.
280
+ this.#ownerId = createHash('sha256').update(resolve(root)).digest('hex').slice(0, 32);
281
+ // Subordinate compute from Worlds rooted in different repositories still shares one machine,
282
+ // so admission uses one user-machine lock rather than a per-checkout lock.
283
+ this.#capacityLock = join(homedir(), '.volter', 'world-resource-claims', 'subordinate-compute.lock');
284
+ mkdirSync(join(homedir(), '.volter', 'world-resource-claims'), { recursive: true, mode: 0o700 });
285
+ }
286
+
287
+ async run(spec: FlyContainerSpec): Promise<{ containerRef: string; publishedPorts?: Array<{ internal: number; host: number }> }> {
288
+ const containerRef = withFileLock(this.#capacityLock, () => {
289
+ this.#admit(spec);
290
+ const name = flyContainerName(spec.machineId);
291
+ const args = [
292
+ 'run', '-d', '--name', name,
293
+ '--memory', `${spec.memoryMiB}m`,
294
+ '--memory-swap', `${spec.memoryMiB}m`,
295
+ '--cpus', String(spec.cpus),
296
+ '--label', `${FLY_DOCKER_LABEL}=${spec.appName}`,
297
+ '--label', `${FLY_WORLD_OWNER_LABEL}=${this.#ownerId}`,
298
+ '--label', `${FLY_MEMORY_LABEL}=${spec.memoryMiB}`,
299
+ '--label', `${FLY_CPU_LABEL}=${spec.cpus}`,
300
+ ];
301
+ for (const [k, v] of Object.entries(spec.env)) args.push('-e', `${k}=${v}`);
302
+ // Publish every internal port on an EPHEMERAL loopback port (fixed host ports would collide
303
+ // across machines); the assigned ports are read back below and ledgered as twin_local_ports.
304
+ for (const p of spec.ports) args.push('-p', `127.0.0.1::${p.internal}`);
305
+ for (const m of spec.mounts) args.push('-v', `${flyVolumeName(m.volumeId)}:${m.path}`);
306
+ if (spec.entrypoint && spec.entrypoint.length > 0) args.push('--entrypoint', spec.entrypoint[0]!);
307
+ args.push(spec.image);
308
+ if (spec.entrypoint && spec.entrypoint.length > 1) args.push(...spec.entrypoint.slice(1));
309
+ if (spec.cmd && spec.cmd.length > 0) args.push(...spec.cmd);
310
+ const r = docker(args);
311
+ if (r.status !== 0) {
312
+ const detail = r.stderr.trim() || `launch command exited ${r.status}`;
313
+ process.stderr.write(`[machine ${spec.machineId}] internal launch failure: ${detail}\n`);
314
+ const reason = /no space left on device|enospc/iu.test(detail) ? 'insufficient writable storage'
315
+ : /out of memory|cannot allocate memory|killed/iu.test(detail) ? 'insufficient memory'
316
+ : /cannot connect|not running|daemon/iu.test(detail) ? 'local execution capacity is unavailable'
317
+ : 'local execution failed';
318
+ throw new Error(`${this.#context(spec)}: ${reason}. Log: ${this.#logPath()}`);
319
+ }
320
+ return r.stdout.trim();
321
+ });
322
+ this.#machineIds.set(containerRef, spec.machineId);
323
+ this.#followMachineLogs(containerRef, spec.machineId);
324
+ const publishedPorts: Array<{ internal: number; host: number }> = [];
325
+ for (const p of spec.ports) {
326
+ const port = docker(['port', containerRef, String(p.internal)]);
327
+ // "127.0.0.1:55007" (possibly several lines for v4/v6) — take the first v4 mapping.
328
+ const match = port.stdout.split('\n').map((l) => l.trim()).find((l) => l.startsWith('127.0.0.1:'));
329
+ if (match) publishedPorts.push({ internal: p.internal, host: Number(match.split(':').pop()) });
330
+ }
331
+ return { containerRef, publishedPorts };
332
+ }
333
+
334
+ /** Admit and reserve by the same limits that `run` enforces. The lock around this method and
335
+ * resource creation prevents two concurrent machine requests from both seeing the same final
336
+ * capacity. Storage is measured inside the local execution substrate when it is discoverable;
337
+ * callers still receive only the declared World/service/resource vocabulary. */
338
+ #admit(spec: FlyContainerSpec): void {
339
+ const info = docker(['info', '--format', '{{.MemTotal}} {{.NCPU}} {{.DockerRootDir}}']);
340
+ if (info.status !== 0) {
341
+ process.stderr.write(`[machine ${spec.machineId}] capacity inspection failed: ${info.stderr.trim()}\n`);
342
+ throw new Error(`${this.#context(spec)}: local execution capacity is unavailable. Log: ${this.#logPath()}`);
343
+ }
344
+ const [memoryRaw, cpuRaw, rootDir = ''] = info.stdout.trim().split(/\s+/u);
345
+ const totalMemoryMiB = Math.floor(Number(memoryRaw) / MIB);
346
+ const totalCpus = Number(cpuRaw);
347
+ if (!Number.isFinite(totalMemoryMiB) || totalMemoryMiB <= 0 || !Number.isFinite(totalCpus) || totalCpus <= 0) {
348
+ process.stderr.write(`[machine ${spec.machineId}] invalid capacity report: ${info.stdout.trim()}\n`);
349
+ throw new Error(`${this.#context(spec)}: local execution capacity could not be verified. Log: ${this.#logPath()}`);
350
+ }
351
+ const refs = docker(['ps', '-aq', '--filter', `label=${FLY_DOCKER_LABEL}`]);
352
+ if (refs.status !== 0) {
353
+ process.stderr.write(`[machine ${spec.machineId}] reservation inspection failed: ${refs.stderr.trim()}\n`);
354
+ throw new Error(`${this.#context(spec)}: local execution capacity could not be verified. Log: ${this.#logPath()}`);
355
+ }
356
+ let reservedMemoryMiB = 0;
357
+ let reservedCpus = 0;
358
+ for (const ref of refs.stdout.split(/\s+/u).filter(Boolean)) {
359
+ const reservation = docker(['inspect', '--format',
360
+ `{{index .Config.Labels "${FLY_MEMORY_LABEL}"}}|{{index .Config.Labels "${FLY_CPU_LABEL}"}}|{{.HostConfig.Memory}}|{{.HostConfig.NanoCpus}}`, ref]);
361
+ if (reservation.status !== 0) {
362
+ process.stderr.write(`[machine ${spec.machineId}] existing reservation inspection failed for ${ref.slice(0, 12)}: ${reservation.stderr.trim()}\n`);
363
+ throw new Error(`${this.#context(spec)}: resource admission refused because existing local execution capacity could not be verified. Log: ${this.#logPath()}`);
364
+ }
365
+ const [memoryLabel, cpuLabel, enforcedMemoryBytes, enforcedNanoCpus] = reservation.stdout.trim().split('|');
366
+ const memoryMiB = Number(memoryLabel) || Math.ceil(Number(enforcedMemoryBytes) / MIB);
367
+ const cpus = Number(cpuLabel) || (Number(enforcedNanoCpus) / 1_000_000_000);
368
+ if (!Number.isFinite(memoryMiB) || memoryMiB <= 0 || !Number.isFinite(cpus) || cpus <= 0) {
369
+ process.stderr.write(`[machine ${spec.machineId}] existing execution resource ${ref.slice(0, 12)} has no enforceable memory/CPU reservation\n`);
370
+ throw new Error(`${this.#context(spec)}: resource admission refused because an existing local execution resource has undeclared capacity. Stop its owning World, then retry. Log: ${this.#logPath()}`);
371
+ }
372
+ reservedMemoryMiB += memoryMiB;
373
+ reservedCpus += cpus;
374
+ }
375
+ const memorySafetyMiB = Math.max(256, Math.min(1024, Math.floor(totalMemoryMiB * 0.1)));
376
+ const availableMemoryMiB = Math.max(0, totalMemoryMiB - memorySafetyMiB - reservedMemoryMiB);
377
+ const availableCpus = Math.max(0, totalCpus - reservedCpus);
378
+ if (spec.memoryMiB > availableMemoryMiB || spec.cpus > availableCpus) {
379
+ const details = [
380
+ ...(spec.memoryMiB > availableMemoryMiB ? [`memory requires ${spec.memoryMiB} MiB, ${availableMemoryMiB} MiB available`] : []),
381
+ ...(spec.cpus > availableCpus ? [`CPU requires ${spec.cpus}, ${availableCpus} available`] : []),
382
+ ];
383
+ throw new Error(`${this.#context(spec)}: resource admission refused (${details.join('; ')}). Log: ${this.#logPath()}`);
384
+ }
385
+
386
+ const storage = probeWritableStorage(rootDir);
387
+ const { requiredMiB, reserveMiB, floorMiB } = writableStorageFloorMiB(spec.memoryMiB, this.#storageReserveMiB);
388
+ if (storage === undefined) {
389
+ throw new Error(`${this.#context(spec)}: resource admission refused because writable-storage capacity could not be verified. Log: ${this.#logPath()}`);
390
+ }
391
+ if (floorMiB > storage.availableMiB) {
392
+ // The substrate and the measured number belong in the capacity log (a refusal has to be
393
+ // diagnosable); the caller's error stays in World/service/resource vocabulary.
394
+ process.stderr.write(`[capacity] writable-storage refused: ${storage.substrate} reports ${storage.availableMiB} MiB available, ${floorMiB} MiB needed (${requiredMiB} MiB machine + ${reserveMiB} MiB reserve)\n`);
395
+ throw new Error(`${this.#context(spec)}: resource admission refused (writable storage requires ${requiredMiB} MiB plus ${reserveMiB} MiB safety, ${storage.availableMiB} MiB available). Log: ${this.#logPath()}`);
396
+ }
397
+ }
398
+
399
+ #context(spec: FlyContainerSpec): string {
400
+ const world = process.env.VOLTER_WORLD_NAME ?? 'standalone';
401
+ const service = process.env.VOLTER_WORLD_SERVICE_ID ?? 'machine-provider';
402
+ return `World "${world}": service "${service}": machine "${spec.machineId}"`;
403
+ }
404
+
405
+ #logPath(): string {
406
+ return process.env.VOLTER_WORLD_SERVICE_LOG ?? '(service stderr)';
407
+ }
408
+
409
+ async stop(containerRef: string, opts: { signal?: string; timeoutSeconds?: number } = {}): Promise<void> {
410
+ const args = ['stop'];
411
+ if (opts.signal) args.push('--signal', opts.signal);
412
+ if (opts.timeoutSeconds !== undefined) args.push('--time', String(opts.timeoutSeconds));
413
+ args.push(containerRef);
414
+ const r = docker(args);
415
+ if (r.status !== 0) throw new Error(r.stderr.trim() || `docker stop exited ${r.status}`);
416
+ }
417
+
418
+ /**
419
+ * Start a STOPPED machine — by RECREATING the container, never `docker start`.
420
+ *
421
+ * Real Fly resets the root filesystem on stop→start: "Stopped Machines that are restarted are
422
+ * completely reset to their original state so that they start clean on the next run"
423
+ * (fly.io/docs/machines/api/machines-resource). `docker start` resumes the old writable layer,
424
+ * which would make this twin MORE FORGIVING than the vendor at exactly the point where the
425
+ * difference bites: an app that writes state outside a volume would pass local rehearsal and
426
+ * lose that state on the first real stop/start cycle.
427
+ *
428
+ * So the old container is removed and a fresh one is run from `spec.image` with the same
429
+ * env/ports/mounts wiring. `docker rm -v` removes only ANONYMOUS volumes, so the named
430
+ * `fly-twin-<volume id>` volumes (this twin's mapping of Fly volumes) carry over — mounted
431
+ * data survives, everything else starts clean, exactly like the vendor. The new container gets
432
+ * a new handle and new ephemeral loopback ports; both are returned and re-ledgered.
433
+ */
434
+ async start(containerRef: string, spec: FlyContainerSpec): Promise<{ containerRef: string; publishedPorts?: Array<{ internal: number; host: number }> }> {
435
+ // A ref that is already gone is not a failure: the fresh run below IS the reset.
436
+ try { await this.remove(containerRef); } catch { /* already reaped */ }
437
+ return this.run(spec);
438
+ }
439
+
440
+ async remove(containerRef: string): Promise<void> {
441
+ const r = docker(['rm', '-f', '-v', containerRef]);
442
+ if (r.status !== 0) throw new Error(r.stderr.trim() || `docker rm exited ${r.status}`);
443
+ this.#logFollowers.get(containerRef)?.kill('SIGTERM');
444
+ this.#logFollowers.delete(containerRef);
445
+ this.#machineIds.delete(containerRef);
446
+ }
447
+
448
+ async pause(containerRef: string): Promise<void> {
449
+ const r = docker(['pause', containerRef]);
450
+ if (r.status !== 0) throw new Error(r.stderr.trim() || `docker pause exited ${r.status}`);
451
+ }
452
+
453
+ async unpause(containerRef: string): Promise<void> {
454
+ const r = docker(['unpause', containerRef]);
455
+ if (r.status !== 0) throw new Error(r.stderr.trim() || `docker unpause exited ${r.status}`);
456
+ }
457
+
458
+ async signal(containerRef: string, signal: string): Promise<void> {
459
+ const r = docker(['kill', '--signal', signal, containerRef]);
460
+ if (r.status !== 0) throw new Error(r.stderr.trim() || `docker kill exited ${r.status}`);
461
+ }
462
+
463
+ async exec(containerRef: string, cmd: string[], opts: { timeoutSeconds?: number } = {}): Promise<FlyExecResult> {
464
+ const r = docker(['exec', containerRef, ...cmd], { timeoutMs: (opts.timeoutSeconds ?? 60) * 1000 });
465
+ return { exitCode: r.status, stdout: r.stdout, stderr: r.stderr };
466
+ }
467
+
468
+ async inspect(containerRef: string): Promise<{ running: boolean; exitCode?: number; oomKilled?: boolean }> {
469
+ const r = docker(['inspect', '--format', '{{.State.Running}} {{.State.ExitCode}} {{.State.OOMKilled}}', containerRef]);
470
+ if (r.status !== 0) return { running: false }; // container gone entirely
471
+ const [running, exitCode, oomKilled] = r.stdout.trim().split(/\s+/);
472
+ return {
473
+ running: running === 'true',
474
+ ...(exitCode !== undefined ? { exitCode: Number(exitCode) } : {}),
475
+ ...(oomKilled !== undefined ? { oomKilled: oomKilled === 'true' } : {}),
476
+ };
477
+ }
478
+
479
+ async createVolume(volumeId: string): Promise<void> {
480
+ const r = docker([
481
+ 'volume', 'create',
482
+ '--label', FLY_DOCKER_LABEL,
483
+ '--label', `${FLY_WORLD_OWNER_LABEL}=${this.#ownerId}`,
484
+ flyVolumeName(volumeId),
485
+ ]);
486
+ if (r.status !== 0) throw new Error(r.stderr.trim() || `docker volume create exited ${r.status}`);
487
+ }
488
+
489
+ async removeVolume(volumeId: string): Promise<void> {
490
+ const r = docker(['volume', 'rm', '-f', flyVolumeName(volumeId)]);
491
+ if (r.status !== 0) throw new Error(r.stderr.trim() || `docker volume rm exited ${r.status}`);
492
+ }
493
+
494
+ async cleanup(): Promise<void> {
495
+ const selector = `label=${FLY_WORLD_OWNER_LABEL}=${this.#ownerId}`;
496
+ const containers = docker(['ps', '-aq', '--filter', selector]);
497
+ if (containers.status !== 0) throw new Error(containers.stderr.trim() || 'could not enumerate owned compute resources');
498
+ const containerRefs = containers.stdout.split(/\s+/u).filter(Boolean);
499
+ if (containerRefs.length > 0) {
500
+ const removed = docker(['rm', '-f', '-v', ...containerRefs]);
501
+ if (removed.status !== 0) throw new Error(removed.stderr.trim() || 'could not reclaim owned compute resources');
502
+ }
503
+
504
+ const volumes = docker(['volume', 'ls', '-q', '--filter', selector]);
505
+ if (volumes.status !== 0) throw new Error(volumes.stderr.trim() || 'could not enumerate owned persistent resources');
506
+ const volumeRefs = volumes.stdout.split(/\s+/u).filter(Boolean);
507
+ if (volumeRefs.length > 0) {
508
+ const removed = docker(['volume', 'rm', '-f', ...volumeRefs]);
509
+ if (removed.status !== 0) throw new Error(removed.stderr.trim() || 'could not reclaim owned persistent resources');
510
+ }
511
+ for (const follower of this.#logFollowers.values()) follower.kill('SIGTERM');
512
+ this.#logFollowers.clear();
513
+ this.#machineIds.clear();
514
+ }
515
+
516
+ /** Forward each owned Machine's output into the provider service's stdout/stderr. The World
517
+ * already retains that declared service log, so diagnostics remain at the World boundary and
518
+ * never require callers to identify or inspect this runtime's backing substrate. */
519
+ #followMachineLogs(containerRef: string, machineId: string): void {
520
+ if (this.#logFollowers.has(containerRef)) return;
521
+ const follower = spawn('docker', ['logs', '--follow', '--timestamps', containerRef], {
522
+ stdio: ['ignore', 'pipe', 'pipe'],
523
+ });
524
+ this.#logFollowers.set(containerRef, follower);
525
+ const write = (stream: NodeJS.WriteStream, chunk: Buffer): void => {
526
+ const prefix = `[machine ${machineId}] `;
527
+ const text = chunk.toString('utf8').replace(/\n(?=.)/gu, `\n${prefix}`);
528
+ stream.write(`${prefix}${text}`);
529
+ };
530
+ follower.stdout?.on('data', (chunk: Buffer) => write(process.stdout, chunk));
531
+ follower.stderr?.on('data', (chunk: Buffer) => write(process.stderr, chunk));
532
+ follower.once('error', (error) => {
533
+ process.stderr.write(`[machine ${machineId}] log stream unavailable: ${error.message}\n`);
534
+ });
535
+ follower.once('close', () => {
536
+ if (this.#logFollowers.get(containerRef) === follower) this.#logFollowers.delete(containerRef);
537
+ });
538
+ }
539
+ }