@volter/twin-fly 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,509 @@
1
+ // The REAL execution plane: `FlyContainerRuntime` implemented over the LOCAL Docker daemon via
2
+ // the `docker` CLI. This is what makes the fly twin a REAL-PLANE twin (the supabase doctrine at
3
+ // the compute layer): the control plane is twinned, but a created/started machine genuinely runs
4
+ // `config.image` as a local container — config.env (plus Fly's documented FLY_* runtime
5
+ // environment) becomes the container env, every service's internal_port is published on an
6
+ // ephemeral LOOPBACK port, config.mounts become named docker volumes, stop is `docker stop`,
7
+ // start RECREATES the container from the image (Fly resets a stopped Machine's rootfs — only
8
+ // mounted volumes survive), destroy is `docker rm -f`, exec is `docker exec`.
9
+ //
10
+ // This module makes NO vendor network calls (D4 is about the vendor; the Docker daemon is local
11
+ // infrastructure, the same class as supabase's real Postgres) and is NEVER exercised by
12
+ // capability verifies or the gate: the handler defaults to the pure-ledger VIRTUAL runtime, and
13
+ // this file's only proof is fly-docker.integration.test.ts, which self-skips LOUDLY when
14
+ // `docker info` fails. `dockerAvailable()` probes the daemon (not just the binary), the
15
+ // cookbook/saas-ai-supabase/run.ts pattern.
16
+ //
17
+ // Naming: every artifact is namespaced `fly-twin-…` so a crashed run can be swept with
18
+ // `docker ps -aq --filter label=dev.volter.fly-twin | xargs docker rm -f`.
19
+ import { spawn, spawnSync } from 'node:child_process';
20
+ import { createHash } from 'node:crypto';
21
+ import { existsSync, mkdirSync, readdirSync, statfsSync } from 'node:fs';
22
+ import { homedir } from 'node:os';
23
+ import { join, resolve } from 'node:path';
24
+ import { withFileLock } from '@volter/world-core';
25
+ export const FLY_DOCKER_LABEL = 'dev.volter.fly-twin';
26
+ const FLY_WORLD_OWNER_LABEL = 'dev.volter.world-owner';
27
+ const FLY_MEMORY_LABEL = 'dev.volter.memory-mib';
28
+ const FLY_CPU_LABEL = 'dev.volter.cpus';
29
+ const MIB = 1024 * 1024;
30
+ /** Writable storage a machine needs for itself before its own memory-sized footprint counts: an
31
+ * image pull plus a writable layer. A property of the substrate, not a knob. */
32
+ const IMAGE_FOOTPRINT_MIB = 2048;
33
+ /** MiB of writable storage admission keeps FREE beyond the machine's own need — the DEFAULT of a
34
+ * configured mechanism value, not a constant. A World whose host disk is smaller than the
35
+ * reserve configures it down (`world-fly serve --storage-reserve-mib N`,
36
+ * `FlyDockerRuntimeOptions.storageReserveMiB`); the runtime never lowers its own floor. */
37
+ export const DEFAULT_STORAGE_RESERVE_MIB = 2048;
38
+ function docker(args, opts = {}) {
39
+ const r = spawnSync('docker', args, { encoding: 'utf8', timeout: opts.timeoutMs ?? 120_000 });
40
+ if (r.error)
41
+ throw new Error(`docker ${args[0]}: ${r.error.message}`);
42
+ return { status: r.status ?? 1, stdout: r.stdout ?? '', stderr: r.stderr ?? '' };
43
+ }
44
+ /**
45
+ * The writable-storage floor for ONE machine: its own need (never below an image pull plus a
46
+ * writable layer) plus the reserve the host keeps free. Pure arithmetic, so the admission rule is
47
+ * checkable without a daemon.
48
+ */
49
+ export function writableStorageFloorMiB(memoryMiB, reserveMiB = DEFAULT_STORAGE_RESERVE_MIB) {
50
+ const requiredMiB = Math.max(IMAGE_FOOTPRINT_MIB, memoryMiB);
51
+ return { requiredMiB, reserveMiB, floorMiB: requiredMiB + reserveMiB };
52
+ }
53
+ const HOST_STORAGE_PROBE = {
54
+ platform: process.platform,
55
+ home: homedir(),
56
+ exists: (path) => existsSync(path),
57
+ freeMiB(path) {
58
+ try {
59
+ const fs = statfsSync(path);
60
+ return Math.floor((Number(fs.bavail) * Number(fs.bsize)) / MIB);
61
+ }
62
+ catch {
63
+ return undefined;
64
+ }
65
+ },
66
+ entries(path) {
67
+ try {
68
+ return readdirSync(path);
69
+ }
70
+ catch {
71
+ return [];
72
+ }
73
+ },
74
+ docker: (args) => docker(args),
75
+ run(command, args, timeoutMs) {
76
+ const r = spawnSync(command, args, { encoding: 'utf8', timeout: timeoutMs });
77
+ return { status: r.status, stdout: r.stdout ?? '', stderr: r.stderr ?? '' };
78
+ },
79
+ warn: (line) => { process.stderr.write(line); },
80
+ };
81
+ /**
82
+ * How much writable storage the local execution substrate can still take, measured INSIDE that
83
+ * substrate when the docker root is not host-visible. Honest by construction: every branch either
84
+ * MEASURES a real filesystem or returns undefined (admission then refuses) — capacity is never
85
+ * assumed, defaulted or fabricated.
86
+ */
87
+ export function probeWritableStorage(rootDir, host = HOST_STORAGE_PROBE) {
88
+ if (rootDir && host.exists(rootDir)) {
89
+ const availableMiB = host.freeMiB(rootDir);
90
+ if (availableMiB === undefined) {
91
+ host.warn(`[capacity] writable-storage verification failed: docker root "${rootDir}" could not be measured\n`);
92
+ return undefined;
93
+ }
94
+ return { availableMiB, substrate: `docker root "${rootDir}"` };
95
+ }
96
+ // The daemon keeps its data in a hidden Linux machine. Detect WHICH one from the active
97
+ // endpoint (and, for the substrates that also install the classic socket path, from the daemon's
98
+ // own identity), then measure that machine's storage; this detail never crosses the service log.
99
+ const context = host.docker(['context', 'inspect', '--format', '{{json .Endpoints.docker.Host}}']);
100
+ let endpoint = '';
101
+ try {
102
+ endpoint = context.status === 0 ? JSON.parse(context.stdout.trim() || '""') : '';
103
+ }
104
+ catch {
105
+ host.warn('[capacity] writable-storage verification failed: the active docker context endpoint could not be read\n');
106
+ return undefined;
107
+ }
108
+ // `~/.colima/<profile>/docker.sock` is colima's own documented DOCKER_HOST; a colima install
109
+ // that predates the dotted directory uses `.../colima/<profile>/docker.sock`. Both are the same
110
+ // substrate, so the leading dot is optional here — anchored to a path segment so no unrelated
111
+ // directory ending in "colima" can claim the branch.
112
+ const colima = /(?:^|\/)\.?colima\/([^/]+)\/docker\.sock$/u.exec(endpoint);
113
+ if (colima)
114
+ return colimaStorage(colima[1], rootDir, host);
115
+ // OrbStack is asked about BEFORE Docker Desktop: both keep a growable image under the user's
116
+ // Library and both may be installed at once, so the ACTIVE daemon — not whichever directory
117
+ // happens to exist — decides which image is the one being written to.
118
+ if (host.platform === 'darwin' && isOrbStack(endpoint, host))
119
+ return orbstackStorage(host);
120
+ // Docker Desktop keeps its Linux machine's disk as a growable image under the user's
121
+ // Library container; free space on the volume HOLDING that image is the binding
122
+ // constraint on how much more the daemon can write, and it is host-visible — no
123
+ // container run, no network. (The VM's internal ceiling is typically far larger.)
124
+ const desktopData = join(host.home, 'Library', 'Containers', 'com.docker.docker', 'Data');
125
+ if (host.platform === 'darwin' && host.exists(desktopData)) {
126
+ const availableMiB = host.freeMiB(desktopData);
127
+ if (availableMiB === undefined) {
128
+ host.warn(`[capacity] writable-storage verification failed: the Docker Desktop data volume "${desktopData}" could not be measured\n`);
129
+ return undefined;
130
+ }
131
+ return { availableMiB, substrate: `Docker Desktop data volume "${desktopData}"` };
132
+ }
133
+ host.warn(`[capacity] writable-storage verification failed: docker root "${rootDir}" is not host-visible and endpoint "${endpoint}" is neither a Colima profile, OrbStack, nor Docker Desktop on macOS\n`);
134
+ return undefined;
135
+ }
136
+ /** Is the ACTIVE daemon OrbStack? Its own socket path is the cheap tell; OrbStack also takes over
137
+ * the classic `/var/run/docker.sock`, so an endpoint that says nothing is settled by asking the
138
+ * daemon what it is — `docker info` reports the operating system literally as "OrbStack". */
139
+ function isOrbStack(endpoint, host) {
140
+ if (/\/\.orbstack\/run\/docker\.sock$/u.test(endpoint))
141
+ return true;
142
+ const info = host.docker(['info', '--format', '{{.OperatingSystem}}']);
143
+ return info.status === 0 && /^orbstack$/iu.test(info.stdout.trim());
144
+ }
145
+ /** OrbStack keeps the WHOLE Linux machine — images, containers, volumes — in one SPARSE
146
+ * `data.img.raw` inside its group container ("it only takes as much space as you use, and
147
+ * automatically shrinks when you delete data", per the file's own README), with no disk ceiling
148
+ * configured by default. So free space on the host volume HOLDING that image is the binding
149
+ * constraint on how much more the daemon can write, and it is host-visible — no container run,
150
+ * no ssh, no network. Same shape as the Docker Desktop branch. The group container carries
151
+ * Apple's team prefix, so it is FOUND rather than hard-coded; when it cannot be found the probe
152
+ * refuses instead of guessing a volume. */
153
+ function orbstackStorage(host) {
154
+ const groupContainers = join(host.home, 'Library', 'Group Containers');
155
+ const container = host.entries(groupContainers).filter((entry) => entry.endsWith('.dev.orbstack')).sort()[0];
156
+ const dataDir = container === undefined ? undefined : join(groupContainers, container, 'data');
157
+ if (dataDir === undefined || !host.exists(dataDir)) {
158
+ host.warn(`[capacity] writable-storage verification failed: the active daemon is OrbStack but its data image was not found under "${groupContainers}"\n`);
159
+ return undefined;
160
+ }
161
+ const availableMiB = host.freeMiB(dataDir);
162
+ if (availableMiB === undefined) {
163
+ host.warn(`[capacity] writable-storage verification failed: the OrbStack data volume "${dataDir}" could not be measured\n`);
164
+ return undefined;
165
+ }
166
+ return { availableMiB, substrate: `OrbStack data volume "${dataDir}"` };
167
+ }
168
+ /** Colima keeps the daemon in a hidden Linux machine reachable over its own ssh config; query
169
+ * that filesystem internally. */
170
+ function colimaStorage(profile, rootDir, host) {
171
+ const config = host.run('colima', ['-p', profile, 'ssh-config'], 5_000);
172
+ if (config.status !== 0) {
173
+ host.warn(`[capacity] writable-storage verification failed: colima ssh-config for profile "${profile}" exited ${String(config.status)}\n`);
174
+ return undefined;
175
+ }
176
+ const field = (name) => {
177
+ const value = new RegExp(`^\\s*${name}\\s+(.+?)\\s*$`, 'mu').exec(config.stdout)?.[1];
178
+ return value?.replace(/^"|"$/gu, '');
179
+ };
180
+ const identity = field('IdentityFile');
181
+ const user = field('User');
182
+ const hostname = field('Hostname');
183
+ const port = field('Port');
184
+ const controlPath = field('ControlPath');
185
+ if (!identity || !user || !hostname || !port) {
186
+ host.warn(`[capacity] writable-storage verification failed: colima ssh-config for profile "${profile}" is missing connection fields\n`);
187
+ return undefined;
188
+ }
189
+ const status = host.run('ssh', [
190
+ '-o', 'StrictHostKeyChecking=no', '-o', 'UserKnownHostsFile=/dev/null', '-o', 'BatchMode=yes',
191
+ '-o', 'IdentitiesOnly=yes', ...(controlPath ? ['-o', 'ControlMaster=auto', '-o', `ControlPath=${controlPath}`, '-o', 'ControlPersist=yes'] : []),
192
+ '-i', identity, '-p', port, `${user}@${hostname}`, 'df', '-Pk', rootDir || '/',
193
+ ], 5_000);
194
+ if (status.status !== 0) {
195
+ host.warn(`[capacity] internal writable-storage inspection failed: ${(status.stderr || '').trim()}\n`);
196
+ return undefined;
197
+ }
198
+ const line = status.stdout.trim().split(/\r?\n/u).at(-1) ?? '';
199
+ const fields = line.trim().split(/\s+/u);
200
+ const availableKiB = Number(fields[3]);
201
+ if (!Number.isFinite(availableKiB)) {
202
+ host.warn(`[capacity] writable-storage verification failed: colima profile "${profile}" reported an unreadable df line "${line}"\n`);
203
+ return undefined;
204
+ }
205
+ return { availableMiB: Math.floor(availableKiB / 1024), substrate: `colima profile "${profile}"` };
206
+ }
207
+ /** Is a Docker DAEMON actually reachable (not just the binary present)? */
208
+ export function dockerAvailable() {
209
+ try {
210
+ const r = spawnSync('docker', ['info'], { stdio: 'ignore', timeout: 15_000 });
211
+ return r.status === 0;
212
+ }
213
+ catch {
214
+ return false;
215
+ }
216
+ }
217
+ export function flyContainerName(machineId) {
218
+ return `fly-twin-${machineId}`;
219
+ }
220
+ export function flyVolumeName(volumeId) {
221
+ return `fly-twin-${volumeId}`;
222
+ }
223
+ export class FlyDockerRuntime {
224
+ kind = 'docker';
225
+ #ownerId;
226
+ #logFollowers = new Map();
227
+ #machineIds = new Map();
228
+ #capacityLock;
229
+ #storageReserveMiB;
230
+ constructor(root = process.cwd(), options = {}) {
231
+ // A malformed reserve FAILS here rather than silently reverting to the default: a World that
232
+ // asked for a different admission floor and got the old one would be told nothing.
233
+ const reserve = options.storageReserveMiB;
234
+ if (reserve !== undefined && (!Number.isInteger(reserve) || reserve < 0)) {
235
+ throw new Error(`machine provider: storage reserve must be a non-negative whole number of MiB (got ${JSON.stringify(reserve)})`);
236
+ }
237
+ this.#storageReserveMiB = reserve ?? DEFAULT_STORAGE_RESERVE_MIB;
238
+ // The service data-root is stable for the lifetime of one named World and
239
+ // opaque outside this implementation. Hashing it gives every backing
240
+ // resource an exact lifecycle owner without exposing host paths.
241
+ this.#ownerId = createHash('sha256').update(resolve(root)).digest('hex').slice(0, 32);
242
+ // Subordinate compute from Worlds rooted in different repositories still shares one machine,
243
+ // so admission uses one user-machine lock rather than a per-checkout lock.
244
+ this.#capacityLock = join(homedir(), '.volter', 'world-resource-claims', 'subordinate-compute.lock');
245
+ mkdirSync(join(homedir(), '.volter', 'world-resource-claims'), { recursive: true, mode: 0o700 });
246
+ }
247
+ async run(spec) {
248
+ const containerRef = withFileLock(this.#capacityLock, () => {
249
+ this.#admit(spec);
250
+ const name = flyContainerName(spec.machineId);
251
+ const args = [
252
+ 'run', '-d', '--name', name,
253
+ '--memory', `${spec.memoryMiB}m`,
254
+ '--memory-swap', `${spec.memoryMiB}m`,
255
+ '--cpus', String(spec.cpus),
256
+ '--label', `${FLY_DOCKER_LABEL}=${spec.appName}`,
257
+ '--label', `${FLY_WORLD_OWNER_LABEL}=${this.#ownerId}`,
258
+ '--label', `${FLY_MEMORY_LABEL}=${spec.memoryMiB}`,
259
+ '--label', `${FLY_CPU_LABEL}=${spec.cpus}`,
260
+ ];
261
+ for (const [k, v] of Object.entries(spec.env))
262
+ args.push('-e', `${k}=${v}`);
263
+ // Publish every internal port on an EPHEMERAL loopback port (fixed host ports would collide
264
+ // across machines); the assigned ports are read back below and ledgered as twin_local_ports.
265
+ for (const p of spec.ports)
266
+ args.push('-p', `127.0.0.1::${p.internal}`);
267
+ for (const m of spec.mounts)
268
+ args.push('-v', `${flyVolumeName(m.volumeId)}:${m.path}`);
269
+ if (spec.entrypoint && spec.entrypoint.length > 0)
270
+ args.push('--entrypoint', spec.entrypoint[0]);
271
+ args.push(spec.image);
272
+ if (spec.entrypoint && spec.entrypoint.length > 1)
273
+ args.push(...spec.entrypoint.slice(1));
274
+ if (spec.cmd && spec.cmd.length > 0)
275
+ args.push(...spec.cmd);
276
+ const r = docker(args);
277
+ if (r.status !== 0) {
278
+ const detail = r.stderr.trim() || `launch command exited ${r.status}`;
279
+ process.stderr.write(`[machine ${spec.machineId}] internal launch failure: ${detail}\n`);
280
+ const reason = /no space left on device|enospc/iu.test(detail) ? 'insufficient writable storage'
281
+ : /out of memory|cannot allocate memory|killed/iu.test(detail) ? 'insufficient memory'
282
+ : /cannot connect|not running|daemon/iu.test(detail) ? 'local execution capacity is unavailable'
283
+ : 'local execution failed';
284
+ throw new Error(`${this.#context(spec)}: ${reason}. Log: ${this.#logPath()}`);
285
+ }
286
+ return r.stdout.trim();
287
+ });
288
+ this.#machineIds.set(containerRef, spec.machineId);
289
+ this.#followMachineLogs(containerRef, spec.machineId);
290
+ const publishedPorts = [];
291
+ for (const p of spec.ports) {
292
+ const port = docker(['port', containerRef, String(p.internal)]);
293
+ // "127.0.0.1:55007" (possibly several lines for v4/v6) — take the first v4 mapping.
294
+ const match = port.stdout.split('\n').map((l) => l.trim()).find((l) => l.startsWith('127.0.0.1:'));
295
+ if (match)
296
+ publishedPorts.push({ internal: p.internal, host: Number(match.split(':').pop()) });
297
+ }
298
+ return { containerRef, publishedPorts };
299
+ }
300
+ /** Admit and reserve by the same limits that `run` enforces. The lock around this method and
301
+ * resource creation prevents two concurrent machine requests from both seeing the same final
302
+ * capacity. Storage is measured inside the local execution substrate when it is discoverable;
303
+ * callers still receive only the declared World/service/resource vocabulary. */
304
+ #admit(spec) {
305
+ const info = docker(['info', '--format', '{{.MemTotal}} {{.NCPU}} {{.DockerRootDir}}']);
306
+ if (info.status !== 0) {
307
+ process.stderr.write(`[machine ${spec.machineId}] capacity inspection failed: ${info.stderr.trim()}\n`);
308
+ throw new Error(`${this.#context(spec)}: local execution capacity is unavailable. Log: ${this.#logPath()}`);
309
+ }
310
+ const [memoryRaw, cpuRaw, rootDir = ''] = info.stdout.trim().split(/\s+/u);
311
+ const totalMemoryMiB = Math.floor(Number(memoryRaw) / MIB);
312
+ const totalCpus = Number(cpuRaw);
313
+ if (!Number.isFinite(totalMemoryMiB) || totalMemoryMiB <= 0 || !Number.isFinite(totalCpus) || totalCpus <= 0) {
314
+ process.stderr.write(`[machine ${spec.machineId}] invalid capacity report: ${info.stdout.trim()}\n`);
315
+ throw new Error(`${this.#context(spec)}: local execution capacity could not be verified. Log: ${this.#logPath()}`);
316
+ }
317
+ const refs = docker(['ps', '-aq', '--filter', `label=${FLY_DOCKER_LABEL}`]);
318
+ if (refs.status !== 0) {
319
+ process.stderr.write(`[machine ${spec.machineId}] reservation inspection failed: ${refs.stderr.trim()}\n`);
320
+ throw new Error(`${this.#context(spec)}: local execution capacity could not be verified. Log: ${this.#logPath()}`);
321
+ }
322
+ let reservedMemoryMiB = 0;
323
+ let reservedCpus = 0;
324
+ for (const ref of refs.stdout.split(/\s+/u).filter(Boolean)) {
325
+ const reservation = docker(['inspect', '--format',
326
+ `{{index .Config.Labels "${FLY_MEMORY_LABEL}"}}|{{index .Config.Labels "${FLY_CPU_LABEL}"}}|{{.HostConfig.Memory}}|{{.HostConfig.NanoCpus}}`, ref]);
327
+ if (reservation.status !== 0) {
328
+ process.stderr.write(`[machine ${spec.machineId}] existing reservation inspection failed for ${ref.slice(0, 12)}: ${reservation.stderr.trim()}\n`);
329
+ throw new Error(`${this.#context(spec)}: resource admission refused because existing local execution capacity could not be verified. Log: ${this.#logPath()}`);
330
+ }
331
+ const [memoryLabel, cpuLabel, enforcedMemoryBytes, enforcedNanoCpus] = reservation.stdout.trim().split('|');
332
+ const memoryMiB = Number(memoryLabel) || Math.ceil(Number(enforcedMemoryBytes) / MIB);
333
+ const cpus = Number(cpuLabel) || (Number(enforcedNanoCpus) / 1_000_000_000);
334
+ if (!Number.isFinite(memoryMiB) || memoryMiB <= 0 || !Number.isFinite(cpus) || cpus <= 0) {
335
+ process.stderr.write(`[machine ${spec.machineId}] existing execution resource ${ref.slice(0, 12)} has no enforceable memory/CPU reservation\n`);
336
+ throw new Error(`${this.#context(spec)}: resource admission refused because an existing local execution resource has undeclared capacity. Stop its owning World, then retry. Log: ${this.#logPath()}`);
337
+ }
338
+ reservedMemoryMiB += memoryMiB;
339
+ reservedCpus += cpus;
340
+ }
341
+ const memorySafetyMiB = Math.max(256, Math.min(1024, Math.floor(totalMemoryMiB * 0.1)));
342
+ const availableMemoryMiB = Math.max(0, totalMemoryMiB - memorySafetyMiB - reservedMemoryMiB);
343
+ const availableCpus = Math.max(0, totalCpus - reservedCpus);
344
+ if (spec.memoryMiB > availableMemoryMiB || spec.cpus > availableCpus) {
345
+ const details = [
346
+ ...(spec.memoryMiB > availableMemoryMiB ? [`memory requires ${spec.memoryMiB} MiB, ${availableMemoryMiB} MiB available`] : []),
347
+ ...(spec.cpus > availableCpus ? [`CPU requires ${spec.cpus}, ${availableCpus} available`] : []),
348
+ ];
349
+ throw new Error(`${this.#context(spec)}: resource admission refused (${details.join('; ')}). Log: ${this.#logPath()}`);
350
+ }
351
+ const storage = probeWritableStorage(rootDir);
352
+ const { requiredMiB, reserveMiB, floorMiB } = writableStorageFloorMiB(spec.memoryMiB, this.#storageReserveMiB);
353
+ if (storage === undefined) {
354
+ throw new Error(`${this.#context(spec)}: resource admission refused because writable-storage capacity could not be verified. Log: ${this.#logPath()}`);
355
+ }
356
+ if (floorMiB > storage.availableMiB) {
357
+ // The substrate and the measured number belong in the capacity log (a refusal has to be
358
+ // diagnosable); the caller's error stays in World/service/resource vocabulary.
359
+ process.stderr.write(`[capacity] writable-storage refused: ${storage.substrate} reports ${storage.availableMiB} MiB available, ${floorMiB} MiB needed (${requiredMiB} MiB machine + ${reserveMiB} MiB reserve)\n`);
360
+ throw new Error(`${this.#context(spec)}: resource admission refused (writable storage requires ${requiredMiB} MiB plus ${reserveMiB} MiB safety, ${storage.availableMiB} MiB available). Log: ${this.#logPath()}`);
361
+ }
362
+ }
363
+ #context(spec) {
364
+ const world = process.env.VOLTER_WORLD_NAME ?? 'standalone';
365
+ const service = process.env.VOLTER_WORLD_SERVICE_ID ?? 'machine-provider';
366
+ return `World "${world}": service "${service}": machine "${spec.machineId}"`;
367
+ }
368
+ #logPath() {
369
+ return process.env.VOLTER_WORLD_SERVICE_LOG ?? '(service stderr)';
370
+ }
371
+ async stop(containerRef, opts = {}) {
372
+ const args = ['stop'];
373
+ if (opts.signal)
374
+ args.push('--signal', opts.signal);
375
+ if (opts.timeoutSeconds !== undefined)
376
+ args.push('--time', String(opts.timeoutSeconds));
377
+ args.push(containerRef);
378
+ const r = docker(args);
379
+ if (r.status !== 0)
380
+ throw new Error(r.stderr.trim() || `docker stop exited ${r.status}`);
381
+ }
382
+ /**
383
+ * Start a STOPPED machine — by RECREATING the container, never `docker start`.
384
+ *
385
+ * Real Fly resets the root filesystem on stop→start: "Stopped Machines that are restarted are
386
+ * completely reset to their original state so that they start clean on the next run"
387
+ * (fly.io/docs/machines/api/machines-resource). `docker start` resumes the old writable layer,
388
+ * which would make this twin MORE FORGIVING than the vendor at exactly the point where the
389
+ * difference bites: an app that writes state outside a volume would pass local rehearsal and
390
+ * lose that state on the first real stop/start cycle.
391
+ *
392
+ * So the old container is removed and a fresh one is run from `spec.image` with the same
393
+ * env/ports/mounts wiring. `docker rm -v` removes only ANONYMOUS volumes, so the named
394
+ * `fly-twin-<volume id>` volumes (this twin's mapping of Fly volumes) carry over — mounted
395
+ * data survives, everything else starts clean, exactly like the vendor. The new container gets
396
+ * a new handle and new ephemeral loopback ports; both are returned and re-ledgered.
397
+ */
398
+ async start(containerRef, spec) {
399
+ // A ref that is already gone is not a failure: the fresh run below IS the reset.
400
+ try {
401
+ await this.remove(containerRef);
402
+ }
403
+ catch { /* already reaped */ }
404
+ return this.run(spec);
405
+ }
406
+ async remove(containerRef) {
407
+ const r = docker(['rm', '-f', '-v', containerRef]);
408
+ if (r.status !== 0)
409
+ throw new Error(r.stderr.trim() || `docker rm exited ${r.status}`);
410
+ this.#logFollowers.get(containerRef)?.kill('SIGTERM');
411
+ this.#logFollowers.delete(containerRef);
412
+ this.#machineIds.delete(containerRef);
413
+ }
414
+ async pause(containerRef) {
415
+ const r = docker(['pause', containerRef]);
416
+ if (r.status !== 0)
417
+ throw new Error(r.stderr.trim() || `docker pause exited ${r.status}`);
418
+ }
419
+ async unpause(containerRef) {
420
+ const r = docker(['unpause', containerRef]);
421
+ if (r.status !== 0)
422
+ throw new Error(r.stderr.trim() || `docker unpause exited ${r.status}`);
423
+ }
424
+ async signal(containerRef, signal) {
425
+ const r = docker(['kill', '--signal', signal, containerRef]);
426
+ if (r.status !== 0)
427
+ throw new Error(r.stderr.trim() || `docker kill exited ${r.status}`);
428
+ }
429
+ async exec(containerRef, cmd, opts = {}) {
430
+ const r = docker(['exec', containerRef, ...cmd], { timeoutMs: (opts.timeoutSeconds ?? 60) * 1000 });
431
+ return { exitCode: r.status, stdout: r.stdout, stderr: r.stderr };
432
+ }
433
+ async inspect(containerRef) {
434
+ const r = docker(['inspect', '--format', '{{.State.Running}} {{.State.ExitCode}} {{.State.OOMKilled}}', containerRef]);
435
+ if (r.status !== 0)
436
+ return { running: false }; // container gone entirely
437
+ const [running, exitCode, oomKilled] = r.stdout.trim().split(/\s+/);
438
+ return {
439
+ running: running === 'true',
440
+ ...(exitCode !== undefined ? { exitCode: Number(exitCode) } : {}),
441
+ ...(oomKilled !== undefined ? { oomKilled: oomKilled === 'true' } : {}),
442
+ };
443
+ }
444
+ async createVolume(volumeId) {
445
+ const r = docker([
446
+ 'volume', 'create',
447
+ '--label', FLY_DOCKER_LABEL,
448
+ '--label', `${FLY_WORLD_OWNER_LABEL}=${this.#ownerId}`,
449
+ flyVolumeName(volumeId),
450
+ ]);
451
+ if (r.status !== 0)
452
+ throw new Error(r.stderr.trim() || `docker volume create exited ${r.status}`);
453
+ }
454
+ async removeVolume(volumeId) {
455
+ const r = docker(['volume', 'rm', '-f', flyVolumeName(volumeId)]);
456
+ if (r.status !== 0)
457
+ throw new Error(r.stderr.trim() || `docker volume rm exited ${r.status}`);
458
+ }
459
+ async cleanup() {
460
+ const selector = `label=${FLY_WORLD_OWNER_LABEL}=${this.#ownerId}`;
461
+ const containers = docker(['ps', '-aq', '--filter', selector]);
462
+ if (containers.status !== 0)
463
+ throw new Error(containers.stderr.trim() || 'could not enumerate owned compute resources');
464
+ const containerRefs = containers.stdout.split(/\s+/u).filter(Boolean);
465
+ if (containerRefs.length > 0) {
466
+ const removed = docker(['rm', '-f', '-v', ...containerRefs]);
467
+ if (removed.status !== 0)
468
+ throw new Error(removed.stderr.trim() || 'could not reclaim owned compute resources');
469
+ }
470
+ const volumes = docker(['volume', 'ls', '-q', '--filter', selector]);
471
+ if (volumes.status !== 0)
472
+ throw new Error(volumes.stderr.trim() || 'could not enumerate owned persistent resources');
473
+ const volumeRefs = volumes.stdout.split(/\s+/u).filter(Boolean);
474
+ if (volumeRefs.length > 0) {
475
+ const removed = docker(['volume', 'rm', '-f', ...volumeRefs]);
476
+ if (removed.status !== 0)
477
+ throw new Error(removed.stderr.trim() || 'could not reclaim owned persistent resources');
478
+ }
479
+ for (const follower of this.#logFollowers.values())
480
+ follower.kill('SIGTERM');
481
+ this.#logFollowers.clear();
482
+ this.#machineIds.clear();
483
+ }
484
+ /** Forward each owned Machine's output into the provider service's stdout/stderr. The World
485
+ * already retains that declared service log, so diagnostics remain at the World boundary and
486
+ * never require callers to identify or inspect this runtime's backing substrate. */
487
+ #followMachineLogs(containerRef, machineId) {
488
+ if (this.#logFollowers.has(containerRef))
489
+ return;
490
+ const follower = spawn('docker', ['logs', '--follow', '--timestamps', containerRef], {
491
+ stdio: ['ignore', 'pipe', 'pipe'],
492
+ });
493
+ this.#logFollowers.set(containerRef, follower);
494
+ const write = (stream, chunk) => {
495
+ const prefix = `[machine ${machineId}] `;
496
+ const text = chunk.toString('utf8').replace(/\n(?=.)/gu, `\n${prefix}`);
497
+ stream.write(`${prefix}${text}`);
498
+ };
499
+ follower.stdout?.on('data', (chunk) => write(process.stdout, chunk));
500
+ follower.stderr?.on('data', (chunk) => write(process.stderr, chunk));
501
+ follower.once('error', (error) => {
502
+ process.stderr.write(`[machine ${machineId}] log stream unavailable: ${error.message}\n`);
503
+ });
504
+ follower.once('close', () => {
505
+ if (this.#logFollowers.get(containerRef) === follower)
506
+ this.#logFollowers.delete(containerRef);
507
+ });
508
+ }
509
+ }
@@ -0,0 +1,152 @@
1
+ export declare const SERVICE = "fly";
2
+ /** One port the execution plane should expose on loopback. `internal` is the service's
3
+ * internal_port from the machine config; the runtime picks (and reports) the host port. */
4
+ export type FlyPortSpec = {
5
+ internal: number;
6
+ };
7
+ /** What the control plane asks the execution plane to run. Derived from fly.MachineConfig. */
8
+ export type FlyContainerSpec = {
9
+ machineId: string;
10
+ appName: string;
11
+ machineName: string;
12
+ /** The docker image to run (fly.MachineConfig.image). */
13
+ image: string;
14
+ /** Guest limits from fly.MachineConfig.guest. A real execution runtime MUST enforce them. */
15
+ memoryMiB: number;
16
+ cpus: number;
17
+ /** config.env merged UNDER Fly's documented FLY_* runtime environment (flyRuntimeEnv). */
18
+ env: Record<string, string>;
19
+ /** init.cmd / init.entrypoint (fly.MachineInit). */
20
+ cmd?: string[];
21
+ entrypoint?: string[];
22
+ /** Every service's internal_port, deduped. */
23
+ ports: FlyPortSpec[];
24
+ /** config.mounts resolved to volume ids: mount a named local volume at `path`. */
25
+ mounts: Array<{
26
+ volumeId: string;
27
+ path: string;
28
+ }>;
29
+ };
30
+ export type FlyExecResult = {
31
+ exitCode: number;
32
+ stdout: string;
33
+ stderr: string;
34
+ };
35
+ /**
36
+ * The execution plane, injected. `fly-docker.ts` is the real one (docker CLI over the local
37
+ * daemon); `VIRTUAL_FLY_RUNTIME` below is the offline default (pure ledger, nothing runs);
38
+ * tests inject recording/failing fakes. Optional members are capabilities a runtime may lack —
39
+ * the handler degrades honestly (exec without a runtime that can exec → Fly-shaped 400, never a
40
+ * fabricated success).
41
+ */
42
+ export interface FlyContainerRuntime {
43
+ /** 'virtual' | 'docker' | a test fake's own tag. Recorded on the machine row (twin-only). */
44
+ readonly kind: string;
45
+ /** Launch the container. Returns the runtime's handle + any published loopback ports. */
46
+ run(spec: FlyContainerSpec): Promise<{
47
+ containerRef: string;
48
+ publishedPorts?: Array<{
49
+ internal: number;
50
+ host: number;
51
+ }>;
52
+ }>;
53
+ /** Stop the container (SIGTERM by default; `signal`/`timeoutSeconds` from Fly's StopRequest). */
54
+ stop(containerRef: string, opts?: {
55
+ signal?: string;
56
+ timeoutSeconds?: number;
57
+ }): Promise<void>;
58
+ /** Start a STOPPED container. Fly RESETS a stopped Machine's root filesystem on start —
59
+ * "Stopped Machines that are restarted are completely reset to their original state so that
60
+ * they start clean on the next run" (fly.io/docs/machines/api/machines-resource) — so a
61
+ * faithful execution plane RECREATES the container from `spec` instead of resuming the old
62
+ * writable layer; only mounted volumes survive. It therefore returns a NEW handle and NEW
63
+ * published ports, exactly like `run`. Absent → the handler removes and re-runs the spec,
64
+ * which is the same semantics by another route. */
65
+ start?(containerRef: string, spec: FlyContainerSpec): Promise<{
66
+ containerRef: string;
67
+ publishedPorts?: Array<{
68
+ internal: number;
69
+ host: number;
70
+ }>;
71
+ }>;
72
+ /** Remove the container (docker rm -f). */
73
+ remove(containerRef: string): Promise<void>;
74
+ /** Suspend/resume (docker pause/unpause). Optional; absent → suspend degrades to stop. */
75
+ pause?(containerRef: string): Promise<void>;
76
+ unpause?(containerRef: string): Promise<void>;
77
+ /** Deliver a signal without a state transition (docker kill --signal). */
78
+ signal?(containerRef: string, signal: string): Promise<void>;
79
+ /** Run a command inside the machine (docker exec). Absent → Fly-shaped 400 (see fly-twin.ts). */
80
+ exec?(containerRef: string, cmd: string[], opts?: {
81
+ timeoutSeconds?: number;
82
+ }): Promise<FlyExecResult>;
83
+ /** Is the container still running? Absent → the control-plane state is taken as-is. */
84
+ inspect?(containerRef: string): Promise<{
85
+ running: boolean;
86
+ exitCode?: number;
87
+ oomKilled?: boolean;
88
+ }>;
89
+ /** Materialize/destroy a named local volume (docker volume create/rm). Optional. */
90
+ createVolume?(volumeId: string): Promise<void>;
91
+ removeVolume?(volumeId: string): Promise<void>;
92
+ /** Reclaim every backing resource owned by this runtime instance. */
93
+ cleanup?(): Promise<void>;
94
+ }
95
+ /**
96
+ * The offline default: a pure-ledger execution plane. Nothing runs, nothing listens; every
97
+ * lifecycle call succeeds and `inspect` reports the container as still running (so control-plane
98
+ * state is authoritative). This is what makes the ENTIRE control plane — states, events, wait,
99
+ * leases — verifiable with no Docker daemon anywhere near the gate.
100
+ */
101
+ export declare const VIRTUAL_FLY_RUNTIME: FlyContainerRuntime;
102
+ export declare function flyRuntimeEnv(m: {
103
+ id: string;
104
+ app_name: string;
105
+ region: string;
106
+ instance_id: string;
107
+ image: string;
108
+ memory_mb: number;
109
+ private_ip: string;
110
+ }): Record<string, string>;
111
+ export declare const FLY_MACHINE_STATES: readonly ["created", "starting", "started", "stopping", "stopped", "suspending", "suspended", "destroying", "destroyed", "replacing", "failed"];
112
+ export type FlyMachineState = typeof FLY_MACHINE_STATES[number];
113
+ export declare function nowIso(occurredAt?: string): string;
114
+ export declare function nowEpochMs(occurredAt?: string): number;
115
+ /** The live rows of a type (protocol 2: a subject's id is the vendor's — an app's name, a machine's id, a volume's id —
116
+ * with no type prefix; a delete is the kernel's tombstone `deleted: true`, cleared by the subject's next write). */
117
+ export declare function rows(type: string, root?: string): Array<Record<string, unknown>>;
118
+ /** Every row of a type the tree has ever held, tombstoned ones included — the ordinal an id seed takes. */
119
+ export declare function countRows(type: string, root?: string): number;
120
+ export declare function getRow(type: string, id: string, root?: string): Record<string, unknown> | undefined;
121
+ /**
122
+ * The one kernel write choke point. Every write folds a per-subject `_rev` ordinal into `fields`
123
+ * (read-modify-write off the current projection): the kernel dedupes actions by content +
124
+ * millisecond timestamp, so without `rev` a machine returning to a PREVIOUS value under a pinned
125
+ * clock (start→stop→start in one verify) would silently land as `replayed` — reply and stored
126
+ * state disagreeing (the upstash lesson, ADDING_A_TWIN §5).
127
+ */
128
+ export declare function write(type: string, id: string, fields: Record<string, unknown>, op: string, root: string | undefined, occurredAt: string | undefined): Promise<Record<string, unknown>>;
129
+ export declare function newMachineId(seed: string): string;
130
+ export declare function newInstanceId(occurredAt: string | undefined, seed: string): string;
131
+ export declare function newVolumeId(seed: string): string;
132
+ export declare function newLeaseNonce(seed: string): string;
133
+ /** An app's internal numeric id — the same seeded source, read as a 24-bit integer. */
134
+ export declare function newAppNumericId(seed: string): number;
135
+ export declare function newMachineName(seed: string): string;
136
+ /** 6PN address: deterministic per machine id (fdaa:… shape, fly-go's "internal 6PN address").
137
+ * Ledgered only — nothing routes it locally; published loopback ports are the local
138
+ * reachability story. */
139
+ export declare function machinePrivateIp(machineId: string): string;
140
+ export type FlyMachineEvent = {
141
+ id: string;
142
+ type: string;
143
+ status: string;
144
+ source: 'flyd' | 'user';
145
+ timestamp: number;
146
+ /** fly-go MachineEvent.request — used for the exit payload ({exit_event:{exit_code}}), which is
147
+ * what flyctl reads to say WHY a machine died. Populated by the runtime-exit fold. */
148
+ request?: Record<string, unknown>;
149
+ };
150
+ export declare function machineEvent(type: string, status: string, source: 'flyd' | 'user', occurredAt: string | undefined, seq: number, request?: Record<string, unknown>): FlyMachineEvent;
151
+ export declare function imageRefOf(image: string): Record<string, unknown>;
152
+ export declare function containerSpecOf(m: Record<string, unknown>): FlyContainerSpec;