@namzu/sandbox 14.0.0 → 16.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. package/CHANGELOG.md +924 -0
  2. package/README.md +369 -14
  3. package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
  4. package/dist/backends/aci-standby-pool/index.js +13 -1
  5. package/dist/backends/aci-standby-pool/index.js.map +1 -1
  6. package/dist/backends/docker/index.d.ts +169 -6
  7. package/dist/backends/docker/index.d.ts.map +1 -1
  8. package/dist/backends/docker/index.js +499 -85
  9. package/dist/backends/docker/index.js.map +1 -1
  10. package/dist/backends/firecracker/index.d.ts.map +1 -1
  11. package/dist/backends/firecracker/index.js +12 -2
  12. package/dist/backends/firecracker/index.js.map +1 -1
  13. package/dist/backends/firecracker/protocol.d.ts +459 -8
  14. package/dist/backends/firecracker/protocol.d.ts.map +1 -1
  15. package/dist/backends/firecracker/protocol.js +136 -0
  16. package/dist/backends/firecracker/protocol.js.map +1 -1
  17. package/dist/backends/firecracker/transport.d.ts +539 -6
  18. package/dist/backends/firecracker/transport.d.ts.map +1 -1
  19. package/dist/backends/firecracker/transport.js +1171 -24
  20. package/dist/backends/firecracker/transport.js.map +1 -1
  21. package/dist/backends/kubernetes/egress-policy.d.ts +1181 -13
  22. package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -1
  23. package/dist/backends/kubernetes/egress-policy.js +2350 -31
  24. package/dist/backends/kubernetes/egress-policy.js.map +1 -1
  25. package/dist/backends/kubernetes/identity.d.ts +193 -0
  26. package/dist/backends/kubernetes/identity.d.ts.map +1 -0
  27. package/dist/backends/kubernetes/identity.js +147 -0
  28. package/dist/backends/kubernetes/identity.js.map +1 -0
  29. package/dist/backends/kubernetes/index.d.ts +678 -33
  30. package/dist/backends/kubernetes/index.d.ts.map +1 -1
  31. package/dist/backends/kubernetes/index.js +1180 -95
  32. package/dist/backends/kubernetes/index.js.map +1 -1
  33. package/dist/backends/kubernetes/ingress-policy.d.ts +375 -0
  34. package/dist/backends/kubernetes/ingress-policy.d.ts.map +1 -0
  35. package/dist/backends/kubernetes/ingress-policy.js +1050 -0
  36. package/dist/backends/kubernetes/ingress-policy.js.map +1 -0
  37. package/dist/backends/kubernetes/k8s-client.d.ts +213 -4
  38. package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -1
  39. package/dist/backends/kubernetes/k8s-client.js +359 -52
  40. package/dist/backends/kubernetes/k8s-client.js.map +1 -1
  41. package/dist/backends/kubernetes/lease.d.ts +40 -14
  42. package/dist/backends/kubernetes/lease.d.ts.map +1 -1
  43. package/dist/backends/kubernetes/lease.js +68 -18
  44. package/dist/backends/kubernetes/lease.js.map +1 -1
  45. package/dist/backends/kubernetes/objects.d.ts +423 -3
  46. package/dist/backends/kubernetes/objects.d.ts.map +1 -1
  47. package/dist/backends/kubernetes/objects.js +364 -2
  48. package/dist/backends/kubernetes/objects.js.map +1 -1
  49. package/dist/backends/kubernetes/per-sandbox-policy.d.ts +219 -0
  50. package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -0
  51. package/dist/backends/kubernetes/per-sandbox-policy.js +375 -0
  52. package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -0
  53. package/dist/backends/kubernetes/rbac.d.ts +153 -0
  54. package/dist/backends/kubernetes/rbac.d.ts.map +1 -0
  55. package/dist/backends/kubernetes/rbac.js +177 -0
  56. package/dist/backends/kubernetes/rbac.js.map +1 -0
  57. package/dist/backends/kubernetes/sandbox.d.ts +81 -14
  58. package/dist/backends/kubernetes/sandbox.d.ts.map +1 -1
  59. package/dist/backends/kubernetes/sandbox.js +149 -15
  60. package/dist/backends/kubernetes/sandbox.js.map +1 -1
  61. package/dist/backends/kubernetes/transport.d.ts +935 -9
  62. package/dist/backends/kubernetes/transport.d.ts.map +1 -1
  63. package/dist/backends/kubernetes/transport.js +1958 -62
  64. package/dist/backends/kubernetes/transport.js.map +1 -1
  65. package/dist/backends/kubernetes/workspace.d.ts +1149 -18
  66. package/dist/backends/kubernetes/workspace.d.ts.map +1 -1
  67. package/dist/backends/kubernetes/workspace.js +2825 -186
  68. package/dist/backends/kubernetes/workspace.js.map +1 -1
  69. package/dist/backends/remote-execution-controller.d.ts +14 -0
  70. package/dist/backends/remote-execution-controller.d.ts.map +1 -1
  71. package/dist/backends/remote-execution-controller.js.map +1 -1
  72. package/dist/index.d.ts +294 -18
  73. package/dist/index.d.ts.map +1 -1
  74. package/dist/index.js +280 -10
  75. package/dist/index.js.map +1 -1
  76. package/dist/testing/sandbox-conformance.d.ts +39 -5
  77. package/dist/testing/sandbox-conformance.d.ts.map +1 -1
  78. package/dist/testing/sandbox-conformance.js +436 -5
  79. package/dist/testing/sandbox-conformance.js.map +1 -1
  80. package/package.json +3 -3
  81. package/src/backends/aci-standby-pool/index.ts +16 -1
  82. package/src/backends/docker/index.ts +617 -100
  83. package/src/backends/firecracker/index.ts +14 -2
  84. package/src/backends/firecracker/protocol.ts +514 -6
  85. package/src/backends/firecracker/transport.ts +1492 -40
  86. package/src/backends/kubernetes/egress-policy.ts +3334 -55
  87. package/src/backends/kubernetes/identity.ts +261 -0
  88. package/src/backends/kubernetes/index.ts +1785 -127
  89. package/src/backends/kubernetes/ingress-policy.ts +1344 -0
  90. package/src/backends/kubernetes/k8s-client.ts +444 -54
  91. package/src/backends/kubernetes/lease.ts +75 -19
  92. package/src/backends/kubernetes/objects.ts +626 -6
  93. package/src/backends/kubernetes/per-sandbox-policy.ts +497 -0
  94. package/src/backends/kubernetes/rbac.ts +192 -0
  95. package/src/backends/kubernetes/sandbox.ts +218 -20
  96. package/src/backends/kubernetes/transport.ts +2733 -124
  97. package/src/backends/kubernetes/workspace.ts +4476 -222
  98. package/src/backends/remote-execution-controller.ts +14 -0
  99. package/src/index.ts +668 -19
  100. package/src/testing/sandbox-conformance.ts +540 -5
@@ -37,6 +37,13 @@ const WORKER_PORT_INSIDE_CONTAINER = 2024;
37
37
  * `create()` call.
38
38
  */
39
39
  export function buildDockerBackend(config) {
40
+ // Refused here rather than at the first spawn, so a config that cannot be
41
+ // rendered — a CPU limit that cannot mean anything, or writable paths
42
+ // beside a writable root filesystem — surfaces during host wiring instead
43
+ // of as a container that failed to come up. The same checks run where the
44
+ // argv is built, because that is the only place a caller cannot skip them.
45
+ assertCpuLimitIsRenderable(config.cpuLimit);
46
+ assertRootfsOptionsAreCoherent(config);
40
47
  const readiness = resolveReadinessOptions('docker', config.readyTimeoutMs, config.readyPollIntervalMs, {
41
48
  timeoutMs: DEFAULT_READY_TIMEOUT_MS,
42
49
  pollIntervalMs: DEFAULT_READY_POLL_MS,
@@ -181,9 +188,11 @@ export function egressProxyOptions(config, policy) {
181
188
  * are the defaults every container runtime hardening guide starts with,
182
189
  * and none of them were present.
183
190
  *
184
- * `--cap-drop=ALL` is deliberately not softened by a re-add list: a
185
- * workload that genuinely needs a capability should say so through
186
- * `extraRunArgs` and be visible in review.
191
+ * `--cap-drop=ALL` is deliberately not softened by a re-add list, and there is
192
+ * no config field that could soften it either: a workload that genuinely needs
193
+ * a capability needs a change to this file, where the diff says which
194
+ * capability and why. A re-add list on the config would grant it to every
195
+ * sandbox the host spawns, quietly, which is how a baseline stops being one.
187
196
  *
188
197
  * **It carries a second, independent load, and this is the one that would
189
198
  * survive being forgotten.** An egress policy of `deny-all` is enforced by
@@ -208,10 +217,471 @@ export function egressProxyOptions(config, policy) {
208
217
  *
209
218
  * Recorded here because the first justification above would survive
210
219
  * softening this flag and the second would not.
220
+ *
221
+ * `--ipc private` closes a door that is not the one its name suggests, and the
222
+ * difference is worth being exact about. Moby runs `private`, `shareable` and
223
+ * `none` through the SAME branch (`daemon/oci_linux.go`, `WithNamespaces`), so
224
+ * `shareable` already gives every container an IPC namespace of its own: a
225
+ * daemon whose `default-ipc-mode` is `shareable` does not merge anybody's
226
+ * namespaces, and what an unset flag buys on such a daemon is not a shared one
227
+ * either. What separates the modes is reachability. Docker's run reference
228
+ * defines `shareable` as "Own private IPC namespace, with a possibility to
229
+ * share it with other containers", and that possibility is `--ipc
230
+ * container:<name>`, which joins another container's IPC namespace — and what
231
+ * that join needs from the target is a shared-memory directory to enter:
232
+ * `daemon.getIPCContainer` resolves the target by name and is gated on its
233
+ * `ShmPath`, which a container created `private` does not have. So on a
234
+ * daemon defaulting to `shareable` this container's namespace, and the System V
235
+ * shared memory, semaphores and message queues namespaced with it, are joinable
236
+ * by anything else on that host which knows the container's name; `--ipc
237
+ * private` removes that reachability. Creating the joining container still takes
238
+ * access to the same daemon, so this is not a boundary against an unprivileged
239
+ * attacker, and it is not claimed as one here. What it buys is that the answer
240
+ * is in THIS argv rather than in the host's `daemon.json`, which is the only
241
+ * place the daemon's default is written down. `--ipc none` is deliberately not
242
+ * used: it takes `/dev/shm` away, and chromium — which the reference image
243
+ * ships for browser automation — uses it for every renderer process.
244
+ *
245
+ * `--read-only` makes the image itself not a place the workload can write. It
246
+ * is rendered by {@link renderHardeningArgs} rather than listed below, because
247
+ * it is the one flag here a host can turn off (`readOnlyRootfs: false`), and a
248
+ * flag in this array would keep being applied after the field said it was not —
249
+ * a control accepted and not applied, which is the failure this file refuses
250
+ * everywhere else. The layout's own RW binds (`outputs`, `scratch`) are
251
+ * separate mounts and are unaffected; what stays writable inside the
252
+ * container's filesystem is named, path by path and with the reason, in
253
+ * {@link renderWritableRootfsArgs}. A host whose image needs a path that list
254
+ * does not name adds it through `writableRootfsPaths`.
255
+ *
256
+ * Three controls from the published container-hardening guidance are
257
+ * deliberately absent, and the reason is here rather than implied:
258
+ *
259
+ * - **A seccomp profile.** Docker already applies its built-in profile to
260
+ * every container unless something passes `seccomp=unconfined`, and nothing
261
+ * in this backend does — so the tier is filtered, and what is missing is a
262
+ * profile TIGHTER than docker's default. Shipping one means shipping a
263
+ * hand-written file whose deny list has to be correct for whatever image
264
+ * the host names, and this repository cannot test it against the reference
265
+ * image's own toolchain (chromium, LibreOffice, the numpy/scipy/duckdb
266
+ * stack). A profile that blocks a syscall one of those needs breaks the
267
+ * sandbox at a point no test here would catch, which is worse than the gap
268
+ * it closes. A host that needs a tighter profile sets `seccomp-profile` in
269
+ * the daemon's `daemon.json`, where it applies to this container and every
270
+ * other one; `--security-opt seccomp=<file>` is the per-container form, and
271
+ * it is not offered as a config field because a path in a config field is a
272
+ * file the daemon reads from the HOST, which is a different machine from
273
+ * the one this backend runs on whenever it drives a remote daemon.
274
+ * - **`--userns-remap`.** It is not a `docker run` flag at all: it is a
275
+ * daemon property (`userns-remap` in `daemon.json`, or `dockerd
276
+ * --userns-remap=`), and per container the CLI only chooses between the
277
+ * namespaces the daemon already made (`--userns=host|private`). Whether a
278
+ * remapped namespace exists is therefore settled before this argv is read,
279
+ * and a flag here could not settle it — which is the whole reason the
280
+ * control is absent rather than configurable: this backend has nothing to
281
+ * say about a mapping that belongs to the machine the daemon runs on.
282
+ * Enabling it on the host is a real upgrade to this tier (uid 0 inside maps
283
+ * to an unprivileged uid outside) and costs this backend nothing; the README
284
+ * says so.
285
+ * - **`--user`.** Supported, and unset by default on purpose — see the
286
+ * `runAsUser` field, which is where a host that knows its image sets it.
211
287
  */
212
- const HARDENING_ARGS = ['--cap-drop=ALL', '--security-opt=no-new-privileges'];
288
+ const HARDENING_ARGS = [
289
+ '--cap-drop=ALL',
290
+ '--security-opt=no-new-privileges',
291
+ '--ipc',
292
+ 'private',
293
+ ];
213
294
  /** Name the container reaches the host-side egress proxy by. */
214
295
  const PROXY_HOST_ALIAS = 'namzu-egress';
296
+ /**
297
+ * Mount options for every scratch mount this backend creates.
298
+ *
299
+ * `exec` is the load-bearing one and the reason this is a named constant
300
+ * rather than a literal at the call site. Docker does NOT default a `--tmpfs`
301
+ * mount to a usable scratch directory: `withMounts` in moby's
302
+ * `daemon/oci_linux.go` starts every user tmpfs from
303
+ * `["noexec", "nosuid", "nodev", <propagation>]` and appends whatever the
304
+ * caller passed, so `--tmpfs /tmp` on its own is **noexec**. A workload that
305
+ * compiles a program into `/tmp` and runs it — `gcc -o /tmp/a.out … &&
306
+ * /tmp/a.out`, or a python `ctypes.CDLL` of a library it just built there —
307
+ * would meet `Permission denied` on an executable file, an error that reads
308
+ * as a broken sandbox rather than as a mount option. Scratch here is as
309
+ * executable as it was before this backend mounted a tmpfs over it.
310
+ *
311
+ * `nosuid` and `nodev` are kept from docker's defaults: the tmpfs is the one
312
+ * place inside the container a workload can write an arbitrary file to, and
313
+ * neither a setuid binary nor a device node there has any use that is worth
314
+ * the escalation path — with `--cap-drop=ALL` no device node could be created
315
+ * there anyway.
316
+ *
317
+ * `mode=1777` is stated rather than inherited from the kernel's tmpfs default
318
+ * (which is the same value): the mounts have to be writable by whichever uid
319
+ * the image runs as, and the backend does not know that uid. A sticky,
320
+ * world-writable scratch directory is what `/tmp` is, and `--read-only` here
321
+ * is about the image, not about the uid.
322
+ */
323
+ const TMPFS_MOUNT_OPTIONS = 'nosuid,nodev,exec,mode=1777';
324
+ /**
325
+ * Paths the reference image needs writable under `--read-only`, as `--tmpfs`.
326
+ *
327
+ * `--read-only` says the image is not the workload's disk. It does not say
328
+ * nothing may be written, and the difference is the sandbox: the layout's own
329
+ * RW binds (`outputs`, `scratch`) are separate mounts and are unaffected, but
330
+ * the image's toolchain writes inside the container's own filesystem, and a
331
+ * `--read-only` that stops it is worse than the gap it closes. Read off
332
+ * `worker/Dockerfile`, whose whole purpose is producing DOCX/XLSX/PPTX/PDF
333
+ * deliverables:
334
+ *
335
+ * - `/tmp` — `TMPDIR` for python's `tempfile`, for LibreOffice's extraction
336
+ * and for pip's wheel builds, and the conventional place to build and run
337
+ * something disposable. Every scratch mount takes
338
+ * {@link TMPFS_MOUNT_OPTIONS}, which is where the `exec` docker would not
339
+ * have given us is argued for.
340
+ * - `/var/tmp` — the second location the temp-file conventions fall back to,
341
+ * for a temp file that is meant to outlive an interrupted run.
342
+ * - `/home/namzu` — the image's `HOME` (`useradd --create-home namzu`, uid
343
+ * 1001; docker sets `HOME` from the image's passwd entry). LibreOffice
344
+ * refuses a headless conversion without a writable user profile
345
+ * (`~/.config/libreoffice`), matplotlib builds a font cache in
346
+ * `~/.cache/matplotlib`, fontconfig keeps a user cache, npm's cache is
347
+ * `~/.npm`, and `pip install --user` needs `~/.local`.
348
+ * - `/workspace` — the image's `WORKDIR`, chowned to `namzu` on purpose
349
+ * (`chown -R namzu:namzu /workspace`). Leaving it out would make the
350
+ * Dockerfile's own guarantee false.
351
+ *
352
+ * These four are the REFERENCE image's needs, not a claim about anyone else's.
353
+ * A host that points `image` at its own build names what that image needs in
354
+ * `writableRootfsPaths`, which is why the field exists at all: the backend
355
+ * cannot read an image's writable set, and the alternative to asking is
356
+ * guessing. A path a root-running image wants (its `HOME` is `/root`) is a
357
+ * `writableRootfsPaths` entry for exactly that reason — `/root` is not in this
358
+ * list, because the shipped image does not run as root and a tmpfs nobody
359
+ * writes to is a claim that something does.
360
+ *
361
+ * A path the LAYOUT already mounts is skipped rather than mounted twice:
362
+ * docker refuses two mounts at one destination (`Duplicate mount point`), and
363
+ * the bind the host asked for is the one that must win. A path the HOST names
364
+ * that the layout also mounts is refused instead of skipped, because there the
365
+ * two requests contradict each other and nothing should choose between them
366
+ * silently.
367
+ *
368
+ * No `size=` is set. The kernel caps a tmpfs at half the host's RAM, and tmpfs
369
+ * pages are accounted to the container's memory cgroup, so a run that sets
370
+ * `--memory` already bounds scratch with the limit the host chose — while any
371
+ * number picked here would fail a workload that writes a bigger temp file than
372
+ * we guessed, with `ENOSPC` rather than a diagnosis.
373
+ *
374
+ * **The other half of that trade, said out loud because a host will meet it.**
375
+ * Scratch now lives in RAM instead of on the container's writable layer, so a
376
+ * temp file larger than half the host's RAM — or larger than `--memory`, which
377
+ * is the tighter of the two whenever the host set one — fails with `ENOSPC` or
378
+ * is OOM-killed, where writing it to disk used to succeed. That is the cost of
379
+ * not leaving the root filesystem writable, and it is not a bug to be reported.
380
+ * The remedy that keeps the baseline is the layout's own `scratch`, which is a
381
+ * bind to a host directory and therefore still disk-backed: a host with room on
382
+ * disk gives the layout one there and points `TMPDIR` at its container path
383
+ * through the per-call `env` option, so the spill lands on that disk instead of
384
+ * on a tmpfs. `readOnlyRootfs: false` is the other way, and the one to reach for
385
+ * second: it puts scratch back on the container's writable layer and gives up
386
+ * the rest of the baseline with it.
387
+ */
388
+ const DEFAULT_WRITABLE_ROOTFS_PATHS = [
389
+ '/tmp',
390
+ '/var/tmp',
391
+ '/workspace',
392
+ '/home/namzu',
393
+ ];
394
+ /**
395
+ * The spelling docker compares a container path by.
396
+ *
397
+ * Docker cleans a mount destination before it uses it, so `/tmp/`, `//tmp` and
398
+ * `/tmp/.` are one directory to it and to the kernel. The check below is an
399
+ * exact-string comparison, so without this a layout that spelled one of its
400
+ * mounts any of those ways would not match the tmpfs default at the same
401
+ * directory: the argv would carry both a `--tmpfs /tmp:...` and a bind at
402
+ * `/tmp/`, and moby would clean the two destinations into one and refuse the
403
+ * container at spawn with `Duplicate mount point: /tmp` — the failure the check
404
+ * exists to prevent, on the one path no test in this repository can reach.
405
+ * `resolveLayout` does not normalise these (it fills in defaults and compares
406
+ * spellings as written), so the cleaning has to happen here, where the
407
+ * comparison does.
408
+ */
409
+ function cleanContainerPath(path) {
410
+ const kept = [];
411
+ for (const segment of path.split('/')) {
412
+ // Empty segments are `//`, `.` is the directory itself; `..` cancels the
413
+ // segment before it, which is what the kernel does with it too.
414
+ if (segment === '' || segment === '.')
415
+ continue;
416
+ if (segment === '..')
417
+ kept.pop();
418
+ else
419
+ kept.push(segment);
420
+ }
421
+ return `/${kept.join('/')}`;
422
+ }
423
+ /**
424
+ * Every destination the layout mounts something at, in the spelling docker
425
+ * itself compares them by.
426
+ *
427
+ * The collision check below is exact-string, so a layout path spelled `/tmp/`
428
+ * would slip past it and docker would then refuse the container with
429
+ * `Duplicate mount point` — a failure at spawn, on the one path that cannot be
430
+ * tested without a daemon. Cleaning each path to the spelling moby reduces it
431
+ * to is what makes the check cover every way of writing the same directory;
432
+ * see {@link cleanContainerPath}.
433
+ */
434
+ function mountedContainerPaths(layout) {
435
+ return [
436
+ layout.outputs.containerPath,
437
+ layout.uploads?.containerPath,
438
+ layout.scratch?.containerPath,
439
+ layout.toolResults?.containerPath,
440
+ layout.transcripts?.containerPath,
441
+ ...(layout.skills?.map((skill) => skill.containerPath) ?? []),
442
+ ]
443
+ .filter((path) => Boolean(path))
444
+ .map(cleanContainerPath);
445
+ }
446
+ /**
447
+ * Refuse rootfs options that cannot both be honoured.
448
+ *
449
+ * `writableRootfsPaths` beside `readOnlyRootfs: false` is a contradiction: with
450
+ * a writable root filesystem every path is already writable, so the tmpfs
451
+ * mounts would either be dropped (a control accepted and not applied) or take a
452
+ * directory off the image for no reason. Refusing is the honest answer, and it
453
+ * is the same one the sibling backends give a per-sandbox control they cannot
454
+ * express.
455
+ *
456
+ * Called at construction and again where the argv is built, so a config that
457
+ * reaches `create()` by some path other than `buildDockerBackend` is refused
458
+ * too.
459
+ */
460
+ export function assertRootfsOptionsAreCoherent(config) {
461
+ const paths = config.writableRootfsPaths;
462
+ if (config.readOnlyRootfs !== false || paths === undefined || paths.length === 0)
463
+ return;
464
+ throw new Error('writableRootfsPaths was set on a docker backend configured with readOnlyRootfs: false. With a writable root filesystem every path inside the container is already writable, so these --tmpfs mounts would add nothing and take the named directories off the image. Refusing rather than accepting a control that cannot be applied: drop the paths, or drop readOnlyRootfs: false and let the read-only baseline stand.');
465
+ }
466
+ /**
467
+ * Refuse a `--cpus` value that cannot mean what it says.
468
+ *
469
+ * This covers non-finite and non-positive values and does NOT claim to cover
470
+ * every bound the daemon would refuse. The difference is worth stating, because
471
+ * the two classes fail in different places and only one of them is decidable
472
+ * here. A negative, `NaN` or `Infinity` renders into the argv as text the
473
+ * daemon either rejects or turns into a bound nobody asked for, and `0` is the
474
+ * opposite of a bound (`NanoCPUs` of zero is how a container says "no CPU
475
+ * limit"), so a host that wrote one of those hears about it during wiring
476
+ * rather than as a container that never came up.
477
+ *
478
+ * The upper bound is not ours to check. Moby's `verifyPlatformContainerResources`
479
+ * refuses `NanoCPUs` above the DAEMON host's CPU count (`"range of CPUs is from
480
+ * 0.01 to N.00, as there are only N CPUs available"`), and the same function
481
+ * deliberately sets no floor of its own on Linux, leaving that to the kernel.
482
+ * Neither number is knowable from here: the `docker` binary this backend drives
483
+ * can be pointed at a daemon on another machine (`DOCKER_HOST`), and even
484
+ * locally `os.cpus().length` is this machine's view rather than the daemon's
485
+ * own `runtime.NumCPU()`. Refusing on a guess at it would break a host whose
486
+ * daemon has more cores than the process driving it, which is a worse failure
487
+ * than the one it would catch — those arrive from the daemon with its own
488
+ * message, at spawn, where every other daemon-side refusal arrives too.
489
+ */
490
+ export function assertCpuLimitIsRenderable(cpuLimit) {
491
+ if (cpuLimit === undefined)
492
+ return;
493
+ if (!Number.isFinite(cpuLimit) || cpuLimit <= 0) {
494
+ throw new Error(`cpuLimit must be a finite number greater than 0 (docker's --cpus takes a decimal, e.g. 1.5); got ${String(cpuLimit)}. Refusing rather than rendering an argv whose value means something other than what was written.`);
495
+ }
496
+ }
497
+ /**
498
+ * `--tmpfs` flags for the paths that stay writable under `--read-only`.
499
+ *
500
+ * See {@link DEFAULT_WRITABLE_ROOTFS_PATHS} for the paths themselves and why
501
+ * each is there. Returns nothing when the read-only root filesystem is off, and
502
+ * the two ways a host names paths that cannot be mounted — a contradiction with
503
+ * `readOnlyRootfs: false`, or a path the layout already mounts — are refusals
504
+ * rather than a silently shorter list.
505
+ */
506
+ export function renderWritableRootfsArgs(config) {
507
+ assertRootfsOptionsAreCoherent(config);
508
+ if (config.readOnlyRootfs === false)
509
+ return [];
510
+ const mounted = new Set(mountedContainerPaths(config.layout));
511
+ const requested = config.writableRootfsPaths ?? [];
512
+ for (const path of requested) {
513
+ // Every segment non-empty and none of them `.` or `..`: an absolute path
514
+ // with at least one component. Anything else is refused because each
515
+ // rejected shape is a directory this file's exact-string checks could
516
+ // hold two spellings of — `/tmp/`, `//tmp` and `/tmp/.` are all `/tmp` to
517
+ // the kernel, so a default that mounted `/tmp` and a host entry that
518
+ // mounted `/tmp/` would each pass the duplicate-mount check and then be
519
+ // refused by docker at spawn, on the one path no test here can reach.
520
+ // `/` itself is refused as well, and for its own reason: it would make
521
+ // the whole read-only root filesystem writable again.
522
+ const segments = path.split('/');
523
+ const wellFormed = path.startsWith('/') &&
524
+ segments.length > 1 &&
525
+ segments.slice(1).every((segment) => segment !== '' && segment !== '.' && segment !== '..');
526
+ if (!wellFormed) {
527
+ throw new Error(`writableRootfsPaths entry ${JSON.stringify(path)} is not a normalised absolute path inside the container. Docker requires an absolute mount path with no empty, '.' or '..' segment and no trailing slash, and '/' would make the whole filesystem writable again rather than adding a scratch directory.`);
528
+ }
529
+ if (mounted.has(path)) {
530
+ throw new Error(`writableRootfsPaths names ${path}, which this layout already mounts. Docker refuses two mounts at one destination ("Duplicate mount point"), so which one won would be decided by argument order rather than by anyone's intent. Drop the entry, or change the layout's own mount to the mode you want.`);
531
+ }
532
+ }
533
+ // The set collapses a host entry that repeats a default, which would
534
+ // otherwise emit the same destination twice and be refused by docker.
535
+ const paths = [
536
+ ...new Set([
537
+ ...DEFAULT_WRITABLE_ROOTFS_PATHS.filter((path) => !mounted.has(path)),
538
+ ...requested,
539
+ ]),
540
+ ];
541
+ return paths.flatMap((path) => ['--tmpfs', `${path}:${TMPFS_MOUNT_OPTIONS}`]);
542
+ }
543
+ /**
544
+ * The confinement preamble for one container, in argv order.
545
+ *
546
+ * A function rather than a bare constant because `--read-only` is switchable
547
+ * and the flags that follow it describe what stays writable while it is on:
548
+ * `readOnlyRootfs: false` removes both the flag and the mounts. That is the only
549
+ * thing it removes. Everything in {@link HARDENING_ARGS} is applied
550
+ * unconditionally and no field can turn one of those off, so the argv for
551
+ * `readOnlyRootfs: false` is the argv this backend produced before any of this
552
+ * existed PLUS `--ipc private` — those two flags are the whole previous argv,
553
+ * and `--ipc private` is now unconditional. `--ipc` is not folded under this
554
+ * switch, because the field names the root filesystem: a host that turned the
555
+ * read-only rootfs off would be turning IPC isolation off as well, silently,
556
+ * for a reason the name of the field does not say. A switch has to mean one
557
+ * thing.
558
+ */
559
+ export function renderHardeningArgs(config) {
560
+ return [
561
+ ...HARDENING_ARGS,
562
+ ...(config.readOnlyRootfs === false ? [] : ['--read-only']),
563
+ ...renderWritableRootfsArgs(config),
564
+ ];
565
+ }
566
+ /**
567
+ * The complete `docker run` argv, as a value.
568
+ *
569
+ * Extracted for the same reason {@link resolveNetwork} and
570
+ * {@link egressProxyOptions} were: everything downstream of it needs a running
571
+ * Docker daemon, so a confinement flag that never reached the argv — or one
572
+ * that reached it in an order that cancels another — could only be caught by an
573
+ * operator noticing its effect missing in production. Spawning a fake `docker`
574
+ * and reading back what it was handed proves what the fake was told and nothing
575
+ * about the container the daemon would build. Here the whole baseline is one
576
+ * array, and an edit that drops a flag fails a test rather than a deployment.
577
+ *
578
+ * Order matters in exactly two places, and both are asserted by the test that
579
+ * pins this: the image is the last argument, because everything after it is a
580
+ * command for the container rather than a flag for docker; and every flag that
581
+ * takes a value is pushed as two argv entries rather than one string, so no
582
+ * value is ever re-split by anything downstream.
583
+ */
584
+ export function buildDockerRunArgs(input) {
585
+ const { config, options, containerName, network, hostReachability, egressProxyPort } = input;
586
+ const layout = config.layout;
587
+ assertRootfsOptionsAreCoherent(config);
588
+ assertCpuLimitIsRenderable(config.cpuLimit);
589
+ const args = [
590
+ 'run',
591
+ '--detach',
592
+ '--rm',
593
+ '--name',
594
+ containerName,
595
+ '--network',
596
+ network,
597
+ ...renderHardeningArgs(config),
598
+ ];
599
+ if (config.runAsUser) {
600
+ args.push('--user', config.runAsUser);
601
+ }
602
+ // `--label key=value` flags. Validate first — an empty key or
603
+ // a key containing `=` would silently produce a malformed
604
+ // label that downstream `docker ps --filter label=…` queries
605
+ // could not match reliably. Throw before the spawn so misuse
606
+ // surfaces during construction, not as a mysterious "container
607
+ // has no labels" later.
608
+ if (config.labels) {
609
+ for (const [key, value] of Object.entries(config.labels)) {
610
+ if (!key || key.includes('=')) {
611
+ throw new Error(`docker label key ${JSON.stringify(key)} is invalid (empty or contains '=')`);
612
+ }
613
+ args.push('--label', `${key}=${value}`);
614
+ }
615
+ }
616
+ args.push(...renderLayoutMountArgs(layout));
617
+ // Forward only the workspace root so the worker's lexical
618
+ // resolver agrees with the bind target. The full layout used
619
+ // to ride along as `NAMZU_SANDBOX_LAYOUT`, but the worker
620
+ // never branched on it; the manifest's only consumer was a
621
+ // log line. A skill loader that needs the manifest will
622
+ // write it to a bind path the worker reads at startup —
623
+ // avoids env-size limits, keeps the wire shape minimal.
624
+ if (egressProxyPort !== undefined) {
625
+ // `host-gateway` is docker's own portable name for the host from
626
+ // inside a container; hard-coding a bridge address would break on
627
+ // every platform whose bridge is numbered differently. The proxy
628
+ // itself binds loopback, so this alias is the only way in.
629
+ args.push('--add-host', `${PROXY_HOST_ALIAS}:host-gateway`);
630
+ const proxyUrl = `http://${PROXY_HOST_ALIAS}:${egressProxyPort}`;
631
+ // Both spellings: tooling is split between them, and a workload
632
+ // that reads only the one that is missing bypasses the boundary
633
+ // entirely — which would look exactly like the policy working.
634
+ for (const key of ['HTTP_PROXY', 'http_proxy', 'HTTPS_PROXY', 'https_proxy']) {
635
+ args.push('--env', `${key}=${proxyUrl}`);
636
+ }
637
+ // Loopback must not be proxied, or the worker cannot talk to
638
+ // itself.
639
+ args.push('--env', 'NO_PROXY=localhost,127.0.0.1');
640
+ args.push('--env', 'no_proxy=localhost,127.0.0.1');
641
+ }
642
+ // `outputs` is required by validation, so its containerPath is always
643
+ // available — the worker uses it as its workspace root.
644
+ args.push('--env', `NAMZU_SANDBOX_WORKSPACE=${layout.outputs.containerPath}`);
645
+ args.push('--env', `NAMZU_SANDBOX_READ_ROOTS=${renderLayoutReadRootsEnv(layout)}`);
646
+ args.push('--env', `NAMZU_SANDBOX_WRITE_ROOTS=${renderLayoutWriteRootsEnv(layout)}`);
647
+ // Only publish a host port when the consumer is going to reach
648
+ // the worker through the docker host's loopback (CLI / direct
649
+ // dev). For `container-network` reachability we leave the port
650
+ // unpublished — sibling containers reach the worker by its DNS
651
+ // name on the shared bridge, no host port required.
652
+ //
653
+ // Let Docker pick the host port instead of pre-reserving one
654
+ // in this process. The reservePort()-then-publish-fixed-port
655
+ // pattern had a TOCTOU window: the OS could hand the port to
656
+ // another process between our `server.close()` and Docker's
657
+ // `bind()`. Letting Docker pick (`--publish-all`) and reading
658
+ // the mapping back via `docker inspect` removes the race.
659
+ if (hostReachability === 'host-port') {
660
+ args.push('--publish', `127.0.0.1::${WORKER_PORT_INSIDE_CONTAINER}`);
661
+ }
662
+ if (config.runtime) {
663
+ args.push('--runtime', config.runtime);
664
+ }
665
+ // The three bounds the host can set, together and in one order, so a
666
+ // reader of a `docker inspect` sees them side by side. `--memory` and
667
+ // `--pids-limit` keep their existing treatment (a non-positive or absent
668
+ // value means "not set"); `--cpus` refuses a value that would mean
669
+ // something else, which is why it is the one with a check in front of it.
670
+ if (options.memoryLimitMb && options.memoryLimitMb > 0) {
671
+ args.push('--memory', `${options.memoryLimitMb}m`);
672
+ }
673
+ if (options.maxProcesses && options.maxProcesses > 0) {
674
+ args.push('--pids-limit', String(options.maxProcesses));
675
+ }
676
+ if (config.cpuLimit !== undefined) {
677
+ args.push('--cpus', String(config.cpuLimit));
678
+ }
679
+ for (const [key, value] of Object.entries(options.env ?? {})) {
680
+ args.push('--env', `${key}=${value}`);
681
+ }
682
+ args.push(config.image);
683
+ return args;
684
+ }
215
685
  async function spawnDockerSandbox(config, options, readiness) {
216
686
  options.signal?.throwIfAborted();
217
687
  const resolvedLayout = config.layout;
@@ -251,7 +721,6 @@ async function spawnDockerSandbox(config, options, readiness) {
251
721
  await egressProxy?.close().catch(() => undefined);
252
722
  throw err;
253
723
  }
254
- const runtime = config.runtime;
255
724
  const containerName = `namzu-sandbox-${id}`;
256
725
  // All bind sources come from the consumer-supplied layout. The
257
726
  // backend never allocates host directories and never removes them
@@ -287,87 +756,14 @@ async function spawnDockerSandbox(config, options, readiness) {
287
756
  // always available — the worker uses it as its workspace root.
288
757
  const rootDir = resolvedLayout.outputs.containerPath;
289
758
  try {
290
- // Let Docker pick the host port instead of pre-reserving one
291
- // in this process. The reservePort()-then-publish-fixed-port
292
- // pattern had a TOCTOU window: the OS could hand the port to
293
- // another process between our `server.close()` and Docker's
294
- // `bind()`. Letting Docker pick (`--publish-all`) and reading
295
- // the mapping back via `docker inspect` removes the race.
296
- const args = [
297
- 'run',
298
- '--detach',
299
- '--rm',
300
- '--name',
759
+ const args = buildDockerRunArgs({
760
+ config,
761
+ options,
301
762
  containerName,
302
- '--network',
303
763
  network,
304
- ...HARDENING_ARGS,
305
- ...(config.runAsUser ? ['--user', config.runAsUser] : []),
306
- ];
307
- // `--label key=value` flags. Validate first — an empty key or
308
- // a key containing `=` would silently produce a malformed
309
- // label that downstream `docker ps --filter label=…` queries
310
- // could not match reliably. Throw before the spawn so misuse
311
- // surfaces during construction, not as a mysterious "container
312
- // has no labels" later.
313
- if (config.labels) {
314
- for (const [key, value] of Object.entries(config.labels)) {
315
- if (!key || key.includes('=')) {
316
- throw new Error(`docker label key ${JSON.stringify(key)} is invalid (empty or contains '=')`);
317
- }
318
- args.push('--label', `${key}=${value}`);
319
- }
320
- }
321
- args.push(...renderLayoutMountArgs(resolvedLayout));
322
- // Forward only the workspace root so the worker's lexical
323
- // resolver agrees with the bind target. The full layout used
324
- // to ride along as `NAMZU_SANDBOX_LAYOUT`, but the worker
325
- // never branched on it; the manifest's only consumer was a
326
- // log line. A skill loader that needs the manifest will
327
- // write it to a bind path the worker reads at startup —
328
- // avoids env-size limits, keeps the wire shape minimal.
329
- if (egressProxy) {
330
- // `host-gateway` is docker's own portable name for the host from
331
- // inside a container; hard-coding a bridge address would break on
332
- // every platform whose bridge is numbered differently. The proxy
333
- // itself binds loopback, so this alias is the only way in.
334
- args.push('--add-host', `${PROXY_HOST_ALIAS}:host-gateway`);
335
- const proxyUrl = `http://${PROXY_HOST_ALIAS}:${egressProxy.port}`;
336
- // Both spellings: tooling is split between them, and a workload
337
- // that reads only the one that is missing bypasses the boundary
338
- // entirely — which would look exactly like the policy working.
339
- for (const key of ['HTTP_PROXY', 'http_proxy', 'HTTPS_PROXY', 'https_proxy']) {
340
- args.push('--env', `${key}=${proxyUrl}`);
341
- }
342
- // Loopback must not be proxied, or the worker cannot talk to
343
- // itself.
344
- args.push('--env', 'NO_PROXY=localhost,127.0.0.1');
345
- args.push('--env', 'no_proxy=localhost,127.0.0.1');
346
- }
347
- args.push('--env', `NAMZU_SANDBOX_WORKSPACE=${rootDir}`);
348
- args.push('--env', `NAMZU_SANDBOX_READ_ROOTS=${renderLayoutReadRootsEnv(resolvedLayout)}`);
349
- args.push('--env', `NAMZU_SANDBOX_WRITE_ROOTS=${renderLayoutWriteRootsEnv(resolvedLayout)}`);
350
- // Only publish a host port when the consumer is going to reach
351
- // the worker through the docker host's loopback (CLI / direct
352
- // dev). For `container-network` reachability we leave the port
353
- // unpublished — sibling containers reach the worker by its DNS
354
- // name on the shared bridge, no host port required.
355
- if (hostReachability === 'host-port') {
356
- args.push('--publish', `127.0.0.1::${WORKER_PORT_INSIDE_CONTAINER}`);
357
- }
358
- if (runtime) {
359
- args.push('--runtime', runtime);
360
- }
361
- if (options.memoryLimitMb && options.memoryLimitMb > 0) {
362
- args.push('--memory', `${options.memoryLimitMb}m`);
363
- }
364
- if (options.maxProcesses && options.maxProcesses > 0) {
365
- args.push('--pids-limit', String(options.maxProcesses));
366
- }
367
- for (const [key, value] of Object.entries(options.env ?? {})) {
368
- args.push('--env', `${key}=${value}`);
369
- }
370
- args.push(config.image);
764
+ hostReachability,
765
+ ...(egressProxy ? { egressProxyPort: egressProxy.port } : {}),
766
+ });
371
767
  await runOnce(docker, args, options.signal);
372
768
  if (hostReachability === 'host-port') {
373
769
  hostPort = await readMappedPort(docker, containerName, options.signal);
@@ -528,12 +924,30 @@ async function spawnDockerSandbox(config, options, readiness) {
528
924
  throw new Error(`write-file failed: HTTP ${res.status} ${await res.text()}`);
529
925
  }
530
926
  },
531
- async readFile(path) {
927
+ /**
928
+ * The worker's `/read-file` has no range, so a caller asking for one
929
+ * is REFUSED rather than handed the whole file.
930
+ *
931
+ * {@link Sandbox.readFile} draws that line: a backend that takes
932
+ * `offset`/`length` and answers with everything has given a wrong
933
+ * answer, not a degraded one, and the caller stops looking. Declaring
934
+ * the one-parameter form would not close it — through the `Sandbox`
935
+ * type a caller can still pass options — so the refusal is explicit.
936
+ *
937
+ * `options.signal` IS honoured: the contract says it aborts the read,
938
+ * and one HTTP request is the whole read here, so it is handed to
939
+ * `fetch`.
940
+ */
941
+ async readFile(path, options) {
532
942
  assertActive();
943
+ if (options?.offset !== undefined || options?.length !== undefined) {
944
+ throw new Error('readFile: the docker worker serves whole files only, so offset/length cannot be honoured. Read the file whole, or use a backend that streams.');
945
+ }
533
946
  const res = await fetch(`${baseUrl}/read-file`, {
534
947
  method: 'POST',
535
948
  headers: { 'content-type': 'application/json' },
536
949
  body: JSON.stringify({ path, encoding: 'base64' }),
950
+ signal: options?.signal,
537
951
  });
538
952
  if (!res.ok) {
539
953
  throw new Error(`read-file failed: HTTP ${res.status} ${await res.text()}`);