@namzu/sandbox 15.0.0 → 16.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +86 -0
- package/README.md +59 -0
- package/dist/backends/docker/index.d.ts +169 -6
- package/dist/backends/docker/index.d.ts.map +1 -1
- package/dist/backends/docker/index.js +480 -84
- package/dist/backends/docker/index.js.map +1 -1
- package/dist/backends/kubernetes/egress-policy.d.ts +193 -102
- package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -1
- package/dist/backends/kubernetes/egress-policy.js +321 -146
- package/dist/backends/kubernetes/egress-policy.js.map +1 -1
- package/dist/backends/kubernetes/per-sandbox-policy.d.ts +6 -6
- package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -1
- package/dist/backends/kubernetes/per-sandbox-policy.js +21 -53
- package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -1
- package/dist/index.d.ts +63 -5
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +33 -5
- package/dist/index.js.map +1 -1
- package/package.json +1 -1
- package/src/backends/docker/index.ts +595 -99
- package/src/backends/kubernetes/egress-policy.ts +455 -187
- package/src/backends/kubernetes/per-sandbox-policy.ts +21 -66
- package/src/index.ts +73 -5
|
@@ -97,14 +97,70 @@ export interface DockerBackendInternalConfig {
|
|
|
97
97
|
/**
|
|
98
98
|
* `--user` value for the container, e.g. `'1000:1000'` or `'nobody'`.
|
|
99
99
|
*
|
|
100
|
-
* Left unset by default because
|
|
101
|
-
*
|
|
102
|
-
*
|
|
103
|
-
* non-root
|
|
104
|
-
*
|
|
100
|
+
* Left unset by default because `--user` does not ADD a non-root user, it
|
|
101
|
+
* OVERRIDES the image's own choice of one. The reference image ends with
|
|
102
|
+
* `USER namzu` (uid 1001, its `/workspace` chowned to match), so this
|
|
103
|
+
* backend's default is already non-root for the image it ships — a
|
|
104
|
+
* hard-coded uid here would replace that with a guess, and the guess is
|
|
105
|
+
* wrong for any image whose files are owned by someone else, which
|
|
106
|
+
* surfaces as `EACCES` on a path the workload was told it could write.
|
|
107
|
+
* Set it when the image does not declare a user of its own, or when the
|
|
108
|
+
* host wants a different one than it declares.
|
|
105
109
|
*/
|
|
106
110
|
readonly runAsUser?: string
|
|
107
111
|
|
|
112
|
+
/**
|
|
113
|
+
* CPU cores the container may use, rendered as `--cpus`. Unset by default.
|
|
114
|
+
*
|
|
115
|
+
* `--memory` and `--pids-limit` bound what a workload can take from the
|
|
116
|
+
* host, and CPU had no equivalent at all — no default, no knob — which
|
|
117
|
+
* reads as an oversight rather than a decision. It stays unset for the
|
|
118
|
+
* same reason neither of those two has a numeric default: the right value
|
|
119
|
+
* is a property of the host's machine and of what the workload is for, and
|
|
120
|
+
* any number this backend picked would silently throttle a run that
|
|
121
|
+
* finishes inside its timeout today. A host that wants the bound says what
|
|
122
|
+
* it is; the value is a decimal (`--cpus 1.5` is one and a half cores'
|
|
123
|
+
* worth of time, not a rounding).
|
|
124
|
+
*
|
|
125
|
+
* It lives on this config rather than beside `memoryLimitMb` on the
|
|
126
|
+
* per-call options because the documented deployment constructs one
|
|
127
|
+
* provider per task, so construction time IS per-task — and a control
|
|
128
|
+
* added to the tier-agnostic per-call shape would have to be refused by
|
|
129
|
+
* the ACI and kubernetes backends, which cannot apply a per-sandbox CPU
|
|
130
|
+
* limit any more than they can apply the memory and process ones.
|
|
131
|
+
*/
|
|
132
|
+
readonly cpuLimit?: number
|
|
133
|
+
|
|
134
|
+
/**
|
|
135
|
+
* Mount the container's root filesystem read-only. Default `true`.
|
|
136
|
+
*
|
|
137
|
+
* See {@link HARDENING_ARGS} for why the default is on and
|
|
138
|
+
* {@link renderWritableRootfsArgs} for the paths that stay writable while
|
|
139
|
+
* it is. Set it to `false` to make every path inside the container
|
|
140
|
+
* writable again, which is what a host whose image writes somewhere the
|
|
141
|
+
* writable set cannot describe needs, and which is why the switch exists
|
|
142
|
+
* instead of an unwritten rule that the baseline is absolute. It turns off
|
|
143
|
+
* that one control and nothing else: `--cap-drop=ALL`,
|
|
144
|
+
* `--security-opt=no-new-privileges` and `--ipc private` are applied
|
|
145
|
+
* whatever this says. It is a config field rather than an argument so that
|
|
146
|
+
* turning it off is a line somebody wrote on purpose, and not the default
|
|
147
|
+
* anyone gets by not looking.
|
|
148
|
+
*/
|
|
149
|
+
readonly readOnlyRootfs?: boolean
|
|
150
|
+
|
|
151
|
+
/**
|
|
152
|
+
* Extra paths to keep writable under `--read-only`, each mounted `--tmpfs`.
|
|
153
|
+
*
|
|
154
|
+
* The default set ({@link DEFAULT_WRITABLE_ROOTFS_PATHS}) is the reference
|
|
155
|
+
* image's needs, read off its Dockerfile; this is how a host that points
|
|
156
|
+
* `image` somewhere else says what ITS image needs, because the backend
|
|
157
|
+
* cannot read that out of an image and guessing is what these paths would
|
|
158
|
+
* otherwise be. A path the layout already mounts is refused rather than
|
|
159
|
+
* mounted twice (`Duplicate mount point`), and setting this at all beside
|
|
160
|
+
* `readOnlyRootfs: false` is refused as a contradiction.
|
|
161
|
+
*/
|
|
162
|
+
readonly writableRootfsPaths?: readonly string[]
|
|
163
|
+
|
|
108
164
|
/**
|
|
109
165
|
* Credentials the egress proxy stamps on, per host.
|
|
110
166
|
*
|
|
@@ -188,6 +244,13 @@ const WORKER_PORT_INSIDE_CONTAINER = 2024
|
|
|
188
244
|
* `create()` call.
|
|
189
245
|
*/
|
|
190
246
|
export function buildDockerBackend(config: DockerBackendInternalConfig): SandboxBackend {
|
|
247
|
+
// Refused here rather than at the first spawn, so a config that cannot be
|
|
248
|
+
// rendered — a CPU limit that cannot mean anything, or writable paths
|
|
249
|
+
// beside a writable root filesystem — surfaces during host wiring instead
|
|
250
|
+
// of as a container that failed to come up. The same checks run where the
|
|
251
|
+
// argv is built, because that is the only place a caller cannot skip them.
|
|
252
|
+
assertCpuLimitIsRenderable(config.cpuLimit)
|
|
253
|
+
assertRootfsOptionsAreCoherent(config)
|
|
191
254
|
const readiness = resolveReadinessOptions(
|
|
192
255
|
'docker',
|
|
193
256
|
config.readyTimeoutMs,
|
|
@@ -363,9 +426,11 @@ export function egressProxyOptions(
|
|
|
363
426
|
* are the defaults every container runtime hardening guide starts with,
|
|
364
427
|
* and none of them were present.
|
|
365
428
|
*
|
|
366
|
-
* `--cap-drop=ALL` is deliberately not softened by a re-add list
|
|
367
|
-
*
|
|
368
|
-
*
|
|
429
|
+
* `--cap-drop=ALL` is deliberately not softened by a re-add list, and there is
|
|
430
|
+
* no config field that could soften it either: a workload that genuinely needs
|
|
431
|
+
* a capability needs a change to this file, where the diff says which
|
|
432
|
+
* capability and why. A re-add list on the config would grant it to every
|
|
433
|
+
* sandbox the host spawns, quietly, which is how a baseline stops being one.
|
|
369
434
|
*
|
|
370
435
|
* **It carries a second, independent load, and this is the one that would
|
|
371
436
|
* survive being forgotten.** An egress policy of `deny-all` is enforced by
|
|
@@ -390,12 +455,527 @@ export function egressProxyOptions(
|
|
|
390
455
|
*
|
|
391
456
|
* Recorded here because the first justification above would survive
|
|
392
457
|
* softening this flag and the second would not.
|
|
458
|
+
*
|
|
459
|
+
* `--ipc private` closes a door that is not the one its name suggests, and the
|
|
460
|
+
* difference is worth being exact about. Moby runs `private`, `shareable` and
|
|
461
|
+
* `none` through the SAME branch (`daemon/oci_linux.go`, `WithNamespaces`), so
|
|
462
|
+
* `shareable` already gives every container an IPC namespace of its own: a
|
|
463
|
+
* daemon whose `default-ipc-mode` is `shareable` does not merge anybody's
|
|
464
|
+
* namespaces, and what an unset flag buys on such a daemon is not a shared one
|
|
465
|
+
* either. What separates the modes is reachability. Docker's run reference
|
|
466
|
+
* defines `shareable` as "Own private IPC namespace, with a possibility to
|
|
467
|
+
* share it with other containers", and that possibility is `--ipc
|
|
468
|
+
* container:<name>`, which joins another container's IPC namespace — and what
|
|
469
|
+
* that join needs from the target is a shared-memory directory to enter:
|
|
470
|
+
* `daemon.getIPCContainer` resolves the target by name and is gated on its
|
|
471
|
+
* `ShmPath`, which a container created `private` does not have. So on a
|
|
472
|
+
* daemon defaulting to `shareable` this container's namespace, and the System V
|
|
473
|
+
* shared memory, semaphores and message queues namespaced with it, are joinable
|
|
474
|
+
* by anything else on that host which knows the container's name; `--ipc
|
|
475
|
+
* private` removes that reachability. Creating the joining container still takes
|
|
476
|
+
* access to the same daemon, so this is not a boundary against an unprivileged
|
|
477
|
+
* attacker, and it is not claimed as one here. What it buys is that the answer
|
|
478
|
+
* is in THIS argv rather than in the host's `daemon.json`, which is the only
|
|
479
|
+
* place the daemon's default is written down. `--ipc none` is deliberately not
|
|
480
|
+
* used: it takes `/dev/shm` away, and chromium — which the reference image
|
|
481
|
+
* ships for browser automation — uses it for every renderer process.
|
|
482
|
+
*
|
|
483
|
+
* `--read-only` makes the image itself not a place the workload can write. It
|
|
484
|
+
* is rendered by {@link renderHardeningArgs} rather than listed below, because
|
|
485
|
+
* it is the one flag here a host can turn off (`readOnlyRootfs: false`), and a
|
|
486
|
+
* flag in this array would keep being applied after the field said it was not —
|
|
487
|
+
* a control accepted and not applied, which is the failure this file refuses
|
|
488
|
+
* everywhere else. The layout's own RW binds (`outputs`, `scratch`) are
|
|
489
|
+
* separate mounts and are unaffected; what stays writable inside the
|
|
490
|
+
* container's filesystem is named, path by path and with the reason, in
|
|
491
|
+
* {@link renderWritableRootfsArgs}. A host whose image needs a path that list
|
|
492
|
+
* does not name adds it through `writableRootfsPaths`.
|
|
493
|
+
*
|
|
494
|
+
* Three controls from the published container-hardening guidance are
|
|
495
|
+
* deliberately absent, and the reason is here rather than implied:
|
|
496
|
+
*
|
|
497
|
+
* - **A seccomp profile.** Docker already applies its built-in profile to
|
|
498
|
+
* every container unless something passes `seccomp=unconfined`, and nothing
|
|
499
|
+
* in this backend does — so the tier is filtered, and what is missing is a
|
|
500
|
+
* profile TIGHTER than docker's default. Shipping one means shipping a
|
|
501
|
+
* hand-written file whose deny list has to be correct for whatever image
|
|
502
|
+
* the host names, and this repository cannot test it against the reference
|
|
503
|
+
* image's own toolchain (chromium, LibreOffice, the numpy/scipy/duckdb
|
|
504
|
+
* stack). A profile that blocks a syscall one of those needs breaks the
|
|
505
|
+
* sandbox at a point no test here would catch, which is worse than the gap
|
|
506
|
+
* it closes. A host that needs a tighter profile sets `seccomp-profile` in
|
|
507
|
+
* the daemon's `daemon.json`, where it applies to this container and every
|
|
508
|
+
* other one; `--security-opt seccomp=<file>` is the per-container form, and
|
|
509
|
+
* it is not offered as a config field because a path in a config field is a
|
|
510
|
+
* file the daemon reads from the HOST, which is a different machine from
|
|
511
|
+
* the one this backend runs on whenever it drives a remote daemon.
|
|
512
|
+
* - **`--userns-remap`.** It is not a `docker run` flag at all: it is a
|
|
513
|
+
* daemon property (`userns-remap` in `daemon.json`, or `dockerd
|
|
514
|
+
* --userns-remap=`), and per container the CLI only chooses between the
|
|
515
|
+
* namespaces the daemon already made (`--userns=host|private`). Whether a
|
|
516
|
+
* remapped namespace exists is therefore settled before this argv is read,
|
|
517
|
+
* and a flag here could not settle it — which is the whole reason the
|
|
518
|
+
* control is absent rather than configurable: this backend has nothing to
|
|
519
|
+
* say about a mapping that belongs to the machine the daemon runs on.
|
|
520
|
+
* Enabling it on the host is a real upgrade to this tier (uid 0 inside maps
|
|
521
|
+
* to an unprivileged uid outside) and costs this backend nothing; the README
|
|
522
|
+
* says so.
|
|
523
|
+
* - **`--user`.** Supported, and unset by default on purpose — see the
|
|
524
|
+
* `runAsUser` field, which is where a host that knows its image sets it.
|
|
393
525
|
*/
|
|
394
|
-
const HARDENING_ARGS: readonly string[] = [
|
|
526
|
+
const HARDENING_ARGS: readonly string[] = [
|
|
527
|
+
'--cap-drop=ALL',
|
|
528
|
+
'--security-opt=no-new-privileges',
|
|
529
|
+
'--ipc',
|
|
530
|
+
'private',
|
|
531
|
+
]
|
|
395
532
|
|
|
396
533
|
/** Name the container reaches the host-side egress proxy by. */
|
|
397
534
|
const PROXY_HOST_ALIAS = 'namzu-egress'
|
|
398
535
|
|
|
536
|
+
/**
|
|
537
|
+
* Mount options for every scratch mount this backend creates.
|
|
538
|
+
*
|
|
539
|
+
* `exec` is the load-bearing one and the reason this is a named constant
|
|
540
|
+
* rather than a literal at the call site. Docker does NOT default a `--tmpfs`
|
|
541
|
+
* mount to a usable scratch directory: `withMounts` in moby's
|
|
542
|
+
* `daemon/oci_linux.go` starts every user tmpfs from
|
|
543
|
+
* `["noexec", "nosuid", "nodev", <propagation>]` and appends whatever the
|
|
544
|
+
* caller passed, so `--tmpfs /tmp` on its own is **noexec**. A workload that
|
|
545
|
+
* compiles a program into `/tmp` and runs it — `gcc -o /tmp/a.out … &&
|
|
546
|
+
* /tmp/a.out`, or a python `ctypes.CDLL` of a library it just built there —
|
|
547
|
+
* would meet `Permission denied` on an executable file, an error that reads
|
|
548
|
+
* as a broken sandbox rather than as a mount option. Scratch here is as
|
|
549
|
+
* executable as it was before this backend mounted a tmpfs over it.
|
|
550
|
+
*
|
|
551
|
+
* `nosuid` and `nodev` are kept from docker's defaults: the tmpfs is the one
|
|
552
|
+
* place inside the container a workload can write an arbitrary file to, and
|
|
553
|
+
* neither a setuid binary nor a device node there has any use that is worth
|
|
554
|
+
* the escalation path — with `--cap-drop=ALL` no device node could be created
|
|
555
|
+
* there anyway.
|
|
556
|
+
*
|
|
557
|
+
* `mode=1777` is stated rather than inherited from the kernel's tmpfs default
|
|
558
|
+
* (which is the same value): the mounts have to be writable by whichever uid
|
|
559
|
+
* the image runs as, and the backend does not know that uid. A sticky,
|
|
560
|
+
* world-writable scratch directory is what `/tmp` is, and `--read-only` here
|
|
561
|
+
* is about the image, not about the uid.
|
|
562
|
+
*/
|
|
563
|
+
const TMPFS_MOUNT_OPTIONS = 'nosuid,nodev,exec,mode=1777'
|
|
564
|
+
|
|
565
|
+
/**
|
|
566
|
+
* Paths the reference image needs writable under `--read-only`, as `--tmpfs`.
|
|
567
|
+
*
|
|
568
|
+
* `--read-only` says the image is not the workload's disk. It does not say
|
|
569
|
+
* nothing may be written, and the difference is the sandbox: the layout's own
|
|
570
|
+
* RW binds (`outputs`, `scratch`) are separate mounts and are unaffected, but
|
|
571
|
+
* the image's toolchain writes inside the container's own filesystem, and a
|
|
572
|
+
* `--read-only` that stops it is worse than the gap it closes. Read off
|
|
573
|
+
* `worker/Dockerfile`, whose whole purpose is producing DOCX/XLSX/PPTX/PDF
|
|
574
|
+
* deliverables:
|
|
575
|
+
*
|
|
576
|
+
* - `/tmp` — `TMPDIR` for python's `tempfile`, for LibreOffice's extraction
|
|
577
|
+
* and for pip's wheel builds, and the conventional place to build and run
|
|
578
|
+
* something disposable. Every scratch mount takes
|
|
579
|
+
* {@link TMPFS_MOUNT_OPTIONS}, which is where the `exec` docker would not
|
|
580
|
+
* have given us is argued for.
|
|
581
|
+
* - `/var/tmp` — the second location the temp-file conventions fall back to,
|
|
582
|
+
* for a temp file that is meant to outlive an interrupted run.
|
|
583
|
+
* - `/home/namzu` — the image's `HOME` (`useradd --create-home namzu`, uid
|
|
584
|
+
* 1001; docker sets `HOME` from the image's passwd entry). LibreOffice
|
|
585
|
+
* refuses a headless conversion without a writable user profile
|
|
586
|
+
* (`~/.config/libreoffice`), matplotlib builds a font cache in
|
|
587
|
+
* `~/.cache/matplotlib`, fontconfig keeps a user cache, npm's cache is
|
|
588
|
+
* `~/.npm`, and `pip install --user` needs `~/.local`.
|
|
589
|
+
* - `/workspace` — the image's `WORKDIR`, chowned to `namzu` on purpose
|
|
590
|
+
* (`chown -R namzu:namzu /workspace`). Leaving it out would make the
|
|
591
|
+
* Dockerfile's own guarantee false.
|
|
592
|
+
*
|
|
593
|
+
* These four are the REFERENCE image's needs, not a claim about anyone else's.
|
|
594
|
+
* A host that points `image` at its own build names what that image needs in
|
|
595
|
+
* `writableRootfsPaths`, which is why the field exists at all: the backend
|
|
596
|
+
* cannot read an image's writable set, and the alternative to asking is
|
|
597
|
+
* guessing. A path a root-running image wants (its `HOME` is `/root`) is a
|
|
598
|
+
* `writableRootfsPaths` entry for exactly that reason — `/root` is not in this
|
|
599
|
+
* list, because the shipped image does not run as root and a tmpfs nobody
|
|
600
|
+
* writes to is a claim that something does.
|
|
601
|
+
*
|
|
602
|
+
* A path the LAYOUT already mounts is skipped rather than mounted twice:
|
|
603
|
+
* docker refuses two mounts at one destination (`Duplicate mount point`), and
|
|
604
|
+
* the bind the host asked for is the one that must win. A path the HOST names
|
|
605
|
+
* that the layout also mounts is refused instead of skipped, because there the
|
|
606
|
+
* two requests contradict each other and nothing should choose between them
|
|
607
|
+
* silently.
|
|
608
|
+
*
|
|
609
|
+
* No `size=` is set. The kernel caps a tmpfs at half the host's RAM, and tmpfs
|
|
610
|
+
* pages are accounted to the container's memory cgroup, so a run that sets
|
|
611
|
+
* `--memory` already bounds scratch with the limit the host chose — while any
|
|
612
|
+
* number picked here would fail a workload that writes a bigger temp file than
|
|
613
|
+
* we guessed, with `ENOSPC` rather than a diagnosis.
|
|
614
|
+
*
|
|
615
|
+
* **The other half of that trade, said out loud because a host will meet it.**
|
|
616
|
+
* Scratch now lives in RAM instead of on the container's writable layer, so a
|
|
617
|
+
* temp file larger than half the host's RAM — or larger than `--memory`, which
|
|
618
|
+
* is the tighter of the two whenever the host set one — fails with `ENOSPC` or
|
|
619
|
+
* is OOM-killed, where writing it to disk used to succeed. That is the cost of
|
|
620
|
+
* not leaving the root filesystem writable, and it is not a bug to be reported.
|
|
621
|
+
* The remedy that keeps the baseline is the layout's own `scratch`, which is a
|
|
622
|
+
* bind to a host directory and therefore still disk-backed: a host with room on
|
|
623
|
+
* disk gives the layout one there and points `TMPDIR` at its container path
|
|
624
|
+
* through the per-call `env` option, so the spill lands on that disk instead of
|
|
625
|
+
* on a tmpfs. `readOnlyRootfs: false` is the other way, and the one to reach for
|
|
626
|
+
* second: it puts scratch back on the container's writable layer and gives up
|
|
627
|
+
* the rest of the baseline with it.
|
|
628
|
+
*/
|
|
629
|
+
const DEFAULT_WRITABLE_ROOTFS_PATHS: readonly string[] = [
|
|
630
|
+
'/tmp',
|
|
631
|
+
'/var/tmp',
|
|
632
|
+
'/workspace',
|
|
633
|
+
'/home/namzu',
|
|
634
|
+
]
|
|
635
|
+
|
|
636
|
+
/** The backend config the hardening flags are rendered from. */
|
|
637
|
+
export type DockerHardeningConfig = Pick<
|
|
638
|
+
DockerBackendInternalConfig,
|
|
639
|
+
'cpuLimit' | 'layout' | 'readOnlyRootfs' | 'writableRootfsPaths'
|
|
640
|
+
>
|
|
641
|
+
|
|
642
|
+
/**
|
|
643
|
+
* The spelling docker compares a container path by.
|
|
644
|
+
*
|
|
645
|
+
* Docker cleans a mount destination before it uses it, so `/tmp/`, `//tmp` and
|
|
646
|
+
* `/tmp/.` are one directory to it and to the kernel. The check below is an
|
|
647
|
+
* exact-string comparison, so without this a layout that spelled one of its
|
|
648
|
+
* mounts any of those ways would not match the tmpfs default at the same
|
|
649
|
+
* directory: the argv would carry both a `--tmpfs /tmp:...` and a bind at
|
|
650
|
+
* `/tmp/`, and moby would clean the two destinations into one and refuse the
|
|
651
|
+
* container at spawn with `Duplicate mount point: /tmp` — the failure the check
|
|
652
|
+
* exists to prevent, on the one path no test in this repository can reach.
|
|
653
|
+
* `resolveLayout` does not normalise these (it fills in defaults and compares
|
|
654
|
+
* spellings as written), so the cleaning has to happen here, where the
|
|
655
|
+
* comparison does.
|
|
656
|
+
*/
|
|
657
|
+
function cleanContainerPath(path: string): string {
|
|
658
|
+
const kept: string[] = []
|
|
659
|
+
for (const segment of path.split('/')) {
|
|
660
|
+
// Empty segments are `//`, `.` is the directory itself; `..` cancels the
|
|
661
|
+
// segment before it, which is what the kernel does with it too.
|
|
662
|
+
if (segment === '' || segment === '.') continue
|
|
663
|
+
if (segment === '..') kept.pop()
|
|
664
|
+
else kept.push(segment)
|
|
665
|
+
}
|
|
666
|
+
return `/${kept.join('/')}`
|
|
667
|
+
}
|
|
668
|
+
|
|
669
|
+
/**
|
|
670
|
+
* Every destination the layout mounts something at, in the spelling docker
|
|
671
|
+
* itself compares them by.
|
|
672
|
+
*
|
|
673
|
+
* The collision check below is exact-string, so a layout path spelled `/tmp/`
|
|
674
|
+
* would slip past it and docker would then refuse the container with
|
|
675
|
+
* `Duplicate mount point` — a failure at spawn, on the one path that cannot be
|
|
676
|
+
* tested without a daemon. Cleaning each path to the spelling moby reduces it
|
|
677
|
+
* to is what makes the check cover every way of writing the same directory;
|
|
678
|
+
* see {@link cleanContainerPath}.
|
|
679
|
+
*/
|
|
680
|
+
function mountedContainerPaths(layout: ResolvedContainerSandboxLayout): string[] {
|
|
681
|
+
return [
|
|
682
|
+
layout.outputs.containerPath,
|
|
683
|
+
layout.uploads?.containerPath,
|
|
684
|
+
layout.scratch?.containerPath,
|
|
685
|
+
layout.toolResults?.containerPath,
|
|
686
|
+
layout.transcripts?.containerPath,
|
|
687
|
+
...(layout.skills?.map((skill) => skill.containerPath) ?? []),
|
|
688
|
+
]
|
|
689
|
+
.filter((path): path is string => Boolean(path))
|
|
690
|
+
.map(cleanContainerPath)
|
|
691
|
+
}
|
|
692
|
+
|
|
693
|
+
/**
|
|
694
|
+
* Refuse rootfs options that cannot both be honoured.
|
|
695
|
+
*
|
|
696
|
+
* `writableRootfsPaths` beside `readOnlyRootfs: false` is a contradiction: with
|
|
697
|
+
* a writable root filesystem every path is already writable, so the tmpfs
|
|
698
|
+
* mounts would either be dropped (a control accepted and not applied) or take a
|
|
699
|
+
* directory off the image for no reason. Refusing is the honest answer, and it
|
|
700
|
+
* is the same one the sibling backends give a per-sandbox control they cannot
|
|
701
|
+
* express.
|
|
702
|
+
*
|
|
703
|
+
* Called at construction and again where the argv is built, so a config that
|
|
704
|
+
* reaches `create()` by some path other than `buildDockerBackend` is refused
|
|
705
|
+
* too.
|
|
706
|
+
*/
|
|
707
|
+
export function assertRootfsOptionsAreCoherent(config: DockerHardeningConfig): void {
|
|
708
|
+
const paths = config.writableRootfsPaths
|
|
709
|
+
if (config.readOnlyRootfs !== false || paths === undefined || paths.length === 0) return
|
|
710
|
+
throw new Error(
|
|
711
|
+
'writableRootfsPaths was set on a docker backend configured with readOnlyRootfs: false. With a writable root filesystem every path inside the container is already writable, so these --tmpfs mounts would add nothing and take the named directories off the image. Refusing rather than accepting a control that cannot be applied: drop the paths, or drop readOnlyRootfs: false and let the read-only baseline stand.',
|
|
712
|
+
)
|
|
713
|
+
}
|
|
714
|
+
|
|
715
|
+
/**
|
|
716
|
+
* Refuse a `--cpus` value that cannot mean what it says.
|
|
717
|
+
*
|
|
718
|
+
* This covers non-finite and non-positive values and does NOT claim to cover
|
|
719
|
+
* every bound the daemon would refuse. The difference is worth stating, because
|
|
720
|
+
* the two classes fail in different places and only one of them is decidable
|
|
721
|
+
* here. A negative, `NaN` or `Infinity` renders into the argv as text the
|
|
722
|
+
* daemon either rejects or turns into a bound nobody asked for, and `0` is the
|
|
723
|
+
* opposite of a bound (`NanoCPUs` of zero is how a container says "no CPU
|
|
724
|
+
* limit"), so a host that wrote one of those hears about it during wiring
|
|
725
|
+
* rather than as a container that never came up.
|
|
726
|
+
*
|
|
727
|
+
* The upper bound is not ours to check. Moby's `verifyPlatformContainerResources`
|
|
728
|
+
* refuses `NanoCPUs` above the DAEMON host's CPU count (`"range of CPUs is from
|
|
729
|
+
* 0.01 to N.00, as there are only N CPUs available"`), and the same function
|
|
730
|
+
* deliberately sets no floor of its own on Linux, leaving that to the kernel.
|
|
731
|
+
* Neither number is knowable from here: the `docker` binary this backend drives
|
|
732
|
+
* can be pointed at a daemon on another machine (`DOCKER_HOST`), and even
|
|
733
|
+
* locally `os.cpus().length` is this machine's view rather than the daemon's
|
|
734
|
+
* own `runtime.NumCPU()`. Refusing on a guess at it would break a host whose
|
|
735
|
+
* daemon has more cores than the process driving it, which is a worse failure
|
|
736
|
+
* than the one it would catch — those arrive from the daemon with its own
|
|
737
|
+
* message, at spawn, where every other daemon-side refusal arrives too.
|
|
738
|
+
*/
|
|
739
|
+
export function assertCpuLimitIsRenderable(cpuLimit: number | undefined): void {
|
|
740
|
+
if (cpuLimit === undefined) return
|
|
741
|
+
if (!Number.isFinite(cpuLimit) || cpuLimit <= 0) {
|
|
742
|
+
throw new Error(
|
|
743
|
+
`cpuLimit must be a finite number greater than 0 (docker's --cpus takes a decimal, e.g. 1.5); got ${String(cpuLimit)}. Refusing rather than rendering an argv whose value means something other than what was written.`,
|
|
744
|
+
)
|
|
745
|
+
}
|
|
746
|
+
}
|
|
747
|
+
|
|
748
|
+
/**
|
|
749
|
+
* `--tmpfs` flags for the paths that stay writable under `--read-only`.
|
|
750
|
+
*
|
|
751
|
+
* See {@link DEFAULT_WRITABLE_ROOTFS_PATHS} for the paths themselves and why
|
|
752
|
+
* each is there. Returns nothing when the read-only root filesystem is off, and
|
|
753
|
+
* the two ways a host names paths that cannot be mounted — a contradiction with
|
|
754
|
+
* `readOnlyRootfs: false`, or a path the layout already mounts — are refusals
|
|
755
|
+
* rather than a silently shorter list.
|
|
756
|
+
*/
|
|
757
|
+
export function renderWritableRootfsArgs(config: DockerHardeningConfig): string[] {
|
|
758
|
+
assertRootfsOptionsAreCoherent(config)
|
|
759
|
+
if (config.readOnlyRootfs === false) return []
|
|
760
|
+
|
|
761
|
+
const mounted = new Set(mountedContainerPaths(config.layout))
|
|
762
|
+
const requested = config.writableRootfsPaths ?? []
|
|
763
|
+
for (const path of requested) {
|
|
764
|
+
// Every segment non-empty and none of them `.` or `..`: an absolute path
|
|
765
|
+
// with at least one component. Anything else is refused because each
|
|
766
|
+
// rejected shape is a directory this file's exact-string checks could
|
|
767
|
+
// hold two spellings of — `/tmp/`, `//tmp` and `/tmp/.` are all `/tmp` to
|
|
768
|
+
// the kernel, so a default that mounted `/tmp` and a host entry that
|
|
769
|
+
// mounted `/tmp/` would each pass the duplicate-mount check and then be
|
|
770
|
+
// refused by docker at spawn, on the one path no test here can reach.
|
|
771
|
+
// `/` itself is refused as well, and for its own reason: it would make
|
|
772
|
+
// the whole read-only root filesystem writable again.
|
|
773
|
+
const segments = path.split('/')
|
|
774
|
+
const wellFormed =
|
|
775
|
+
path.startsWith('/') &&
|
|
776
|
+
segments.length > 1 &&
|
|
777
|
+
segments.slice(1).every((segment) => segment !== '' && segment !== '.' && segment !== '..')
|
|
778
|
+
if (!wellFormed) {
|
|
779
|
+
throw new Error(
|
|
780
|
+
`writableRootfsPaths entry ${JSON.stringify(path)} is not a normalised absolute path inside the container. Docker requires an absolute mount path with no empty, '.' or '..' segment and no trailing slash, and '/' would make the whole filesystem writable again rather than adding a scratch directory.`,
|
|
781
|
+
)
|
|
782
|
+
}
|
|
783
|
+
if (mounted.has(path)) {
|
|
784
|
+
throw new Error(
|
|
785
|
+
`writableRootfsPaths names ${path}, which this layout already mounts. Docker refuses two mounts at one destination ("Duplicate mount point"), so which one won would be decided by argument order rather than by anyone's intent. Drop the entry, or change the layout's own mount to the mode you want.`,
|
|
786
|
+
)
|
|
787
|
+
}
|
|
788
|
+
}
|
|
789
|
+
|
|
790
|
+
// The set collapses a host entry that repeats a default, which would
|
|
791
|
+
// otherwise emit the same destination twice and be refused by docker.
|
|
792
|
+
const paths = [
|
|
793
|
+
...new Set([
|
|
794
|
+
...DEFAULT_WRITABLE_ROOTFS_PATHS.filter((path) => !mounted.has(path)),
|
|
795
|
+
...requested,
|
|
796
|
+
]),
|
|
797
|
+
]
|
|
798
|
+
return paths.flatMap((path) => ['--tmpfs', `${path}:${TMPFS_MOUNT_OPTIONS}`])
|
|
799
|
+
}
|
|
800
|
+
|
|
801
|
+
/**
|
|
802
|
+
* The confinement preamble for one container, in argv order.
|
|
803
|
+
*
|
|
804
|
+
* A function rather than a bare constant because `--read-only` is switchable
|
|
805
|
+
* and the flags that follow it describe what stays writable while it is on:
|
|
806
|
+
* `readOnlyRootfs: false` removes both the flag and the mounts. That is the only
|
|
807
|
+
* thing it removes. Everything in {@link HARDENING_ARGS} is applied
|
|
808
|
+
* unconditionally and no field can turn one of those off, so the argv for
|
|
809
|
+
* `readOnlyRootfs: false` is the argv this backend produced before any of this
|
|
810
|
+
* existed PLUS `--ipc private` — those two flags are the whole previous argv,
|
|
811
|
+
* and `--ipc private` is now unconditional. `--ipc` is not folded under this
|
|
812
|
+
* switch, because the field names the root filesystem: a host that turned the
|
|
813
|
+
* read-only rootfs off would be turning IPC isolation off as well, silently,
|
|
814
|
+
* for a reason the name of the field does not say. A switch has to mean one
|
|
815
|
+
* thing.
|
|
816
|
+
*/
|
|
817
|
+
export function renderHardeningArgs(config: DockerHardeningConfig): string[] {
|
|
818
|
+
return [
|
|
819
|
+
...HARDENING_ARGS,
|
|
820
|
+
...(config.readOnlyRootfs === false ? [] : ['--read-only']),
|
|
821
|
+
...renderWritableRootfsArgs(config),
|
|
822
|
+
]
|
|
823
|
+
}
|
|
824
|
+
|
|
825
|
+
/**
|
|
826
|
+
* Everything {@link buildDockerRunArgs} renders, as a value.
|
|
827
|
+
*
|
|
828
|
+
* The pieces that come from the daemon or from the host are inputs rather than
|
|
829
|
+
* lookups: which network the container attaches to, and whether an egress proxy
|
|
830
|
+
* is listening and on which port. Both are already resolved by the caller, and
|
|
831
|
+
* reading them here would put a daemon call back inside the function whose
|
|
832
|
+
* whole point is that it needs none.
|
|
833
|
+
*/
|
|
834
|
+
export interface DockerRunArgvInput {
|
|
835
|
+
readonly config: DockerBackendInternalConfig
|
|
836
|
+
readonly options: SandboxBackendOptions
|
|
837
|
+
readonly containerName: string
|
|
838
|
+
readonly network: string
|
|
839
|
+
readonly hostReachability: 'host-port' | 'container-network'
|
|
840
|
+
/**
|
|
841
|
+
* Port the host-side egress proxy listens on, when one is running. Absent
|
|
842
|
+
* means no proxy, and no proxy environment is passed in — which is not the
|
|
843
|
+
* same fact as a proxy that was configured and is unreachable.
|
|
844
|
+
*/
|
|
845
|
+
readonly egressProxyPort?: number
|
|
846
|
+
}
|
|
847
|
+
|
|
848
|
+
/**
|
|
849
|
+
* The complete `docker run` argv, as a value.
|
|
850
|
+
*
|
|
851
|
+
* Extracted for the same reason {@link resolveNetwork} and
|
|
852
|
+
* {@link egressProxyOptions} were: everything downstream of it needs a running
|
|
853
|
+
* Docker daemon, so a confinement flag that never reached the argv — or one
|
|
854
|
+
* that reached it in an order that cancels another — could only be caught by an
|
|
855
|
+
* operator noticing its effect missing in production. Spawning a fake `docker`
|
|
856
|
+
* and reading back what it was handed proves what the fake was told and nothing
|
|
857
|
+
* about the container the daemon would build. Here the whole baseline is one
|
|
858
|
+
* array, and an edit that drops a flag fails a test rather than a deployment.
|
|
859
|
+
*
|
|
860
|
+
* Order matters in exactly two places, and both are asserted by the test that
|
|
861
|
+
* pins this: the image is the last argument, because everything after it is a
|
|
862
|
+
* command for the container rather than a flag for docker; and every flag that
|
|
863
|
+
* takes a value is pushed as two argv entries rather than one string, so no
|
|
864
|
+
* value is ever re-split by anything downstream.
|
|
865
|
+
*/
|
|
866
|
+
export function buildDockerRunArgs(input: DockerRunArgvInput): string[] {
|
|
867
|
+
const { config, options, containerName, network, hostReachability, egressProxyPort } = input
|
|
868
|
+
const layout = config.layout
|
|
869
|
+
assertRootfsOptionsAreCoherent(config)
|
|
870
|
+
assertCpuLimitIsRenderable(config.cpuLimit)
|
|
871
|
+
|
|
872
|
+
const args: string[] = [
|
|
873
|
+
'run',
|
|
874
|
+
'--detach',
|
|
875
|
+
'--rm',
|
|
876
|
+
'--name',
|
|
877
|
+
containerName,
|
|
878
|
+
'--network',
|
|
879
|
+
network,
|
|
880
|
+
...renderHardeningArgs(config),
|
|
881
|
+
]
|
|
882
|
+
if (config.runAsUser) {
|
|
883
|
+
args.push('--user', config.runAsUser)
|
|
884
|
+
}
|
|
885
|
+
|
|
886
|
+
// `--label key=value` flags. Validate first — an empty key or
|
|
887
|
+
// a key containing `=` would silently produce a malformed
|
|
888
|
+
// label that downstream `docker ps --filter label=…` queries
|
|
889
|
+
// could not match reliably. Throw before the spawn so misuse
|
|
890
|
+
// surfaces during construction, not as a mysterious "container
|
|
891
|
+
// has no labels" later.
|
|
892
|
+
if (config.labels) {
|
|
893
|
+
for (const [key, value] of Object.entries(config.labels)) {
|
|
894
|
+
if (!key || key.includes('=')) {
|
|
895
|
+
throw new Error(
|
|
896
|
+
`docker label key ${JSON.stringify(key)} is invalid (empty or contains '=')`,
|
|
897
|
+
)
|
|
898
|
+
}
|
|
899
|
+
args.push('--label', `${key}=${value}`)
|
|
900
|
+
}
|
|
901
|
+
}
|
|
902
|
+
|
|
903
|
+
args.push(...renderLayoutMountArgs(layout))
|
|
904
|
+
// Forward only the workspace root so the worker's lexical
|
|
905
|
+
// resolver agrees with the bind target. The full layout used
|
|
906
|
+
// to ride along as `NAMZU_SANDBOX_LAYOUT`, but the worker
|
|
907
|
+
// never branched on it; the manifest's only consumer was a
|
|
908
|
+
// log line. A skill loader that needs the manifest will
|
|
909
|
+
// write it to a bind path the worker reads at startup —
|
|
910
|
+
// avoids env-size limits, keeps the wire shape minimal.
|
|
911
|
+
if (egressProxyPort !== undefined) {
|
|
912
|
+
// `host-gateway` is docker's own portable name for the host from
|
|
913
|
+
// inside a container; hard-coding a bridge address would break on
|
|
914
|
+
// every platform whose bridge is numbered differently. The proxy
|
|
915
|
+
// itself binds loopback, so this alias is the only way in.
|
|
916
|
+
args.push('--add-host', `${PROXY_HOST_ALIAS}:host-gateway`)
|
|
917
|
+
const proxyUrl = `http://${PROXY_HOST_ALIAS}:${egressProxyPort}`
|
|
918
|
+
// Both spellings: tooling is split between them, and a workload
|
|
919
|
+
// that reads only the one that is missing bypasses the boundary
|
|
920
|
+
// entirely — which would look exactly like the policy working.
|
|
921
|
+
for (const key of ['HTTP_PROXY', 'http_proxy', 'HTTPS_PROXY', 'https_proxy']) {
|
|
922
|
+
args.push('--env', `${key}=${proxyUrl}`)
|
|
923
|
+
}
|
|
924
|
+
// Loopback must not be proxied, or the worker cannot talk to
|
|
925
|
+
// itself.
|
|
926
|
+
args.push('--env', 'NO_PROXY=localhost,127.0.0.1')
|
|
927
|
+
args.push('--env', 'no_proxy=localhost,127.0.0.1')
|
|
928
|
+
}
|
|
929
|
+
|
|
930
|
+
// `outputs` is required by validation, so its containerPath is always
|
|
931
|
+
// available — the worker uses it as its workspace root.
|
|
932
|
+
args.push('--env', `NAMZU_SANDBOX_WORKSPACE=${layout.outputs.containerPath}`)
|
|
933
|
+
args.push('--env', `NAMZU_SANDBOX_READ_ROOTS=${renderLayoutReadRootsEnv(layout)}`)
|
|
934
|
+
args.push('--env', `NAMZU_SANDBOX_WRITE_ROOTS=${renderLayoutWriteRootsEnv(layout)}`)
|
|
935
|
+
|
|
936
|
+
// Only publish a host port when the consumer is going to reach
|
|
937
|
+
// the worker through the docker host's loopback (CLI / direct
|
|
938
|
+
// dev). For `container-network` reachability we leave the port
|
|
939
|
+
// unpublished — sibling containers reach the worker by its DNS
|
|
940
|
+
// name on the shared bridge, no host port required.
|
|
941
|
+
//
|
|
942
|
+
// Let Docker pick the host port instead of pre-reserving one
|
|
943
|
+
// in this process. The reservePort()-then-publish-fixed-port
|
|
944
|
+
// pattern had a TOCTOU window: the OS could hand the port to
|
|
945
|
+
// another process between our `server.close()` and Docker's
|
|
946
|
+
// `bind()`. Letting Docker pick (`--publish-all`) and reading
|
|
947
|
+
// the mapping back via `docker inspect` removes the race.
|
|
948
|
+
if (hostReachability === 'host-port') {
|
|
949
|
+
args.push('--publish', `127.0.0.1::${WORKER_PORT_INSIDE_CONTAINER}`)
|
|
950
|
+
}
|
|
951
|
+
|
|
952
|
+
if (config.runtime) {
|
|
953
|
+
args.push('--runtime', config.runtime)
|
|
954
|
+
}
|
|
955
|
+
|
|
956
|
+
// The three bounds the host can set, together and in one order, so a
|
|
957
|
+
// reader of a `docker inspect` sees them side by side. `--memory` and
|
|
958
|
+
// `--pids-limit` keep their existing treatment (a non-positive or absent
|
|
959
|
+
// value means "not set"); `--cpus` refuses a value that would mean
|
|
960
|
+
// something else, which is why it is the one with a check in front of it.
|
|
961
|
+
if (options.memoryLimitMb && options.memoryLimitMb > 0) {
|
|
962
|
+
args.push('--memory', `${options.memoryLimitMb}m`)
|
|
963
|
+
}
|
|
964
|
+
if (options.maxProcesses && options.maxProcesses > 0) {
|
|
965
|
+
args.push('--pids-limit', String(options.maxProcesses))
|
|
966
|
+
}
|
|
967
|
+
if (config.cpuLimit !== undefined) {
|
|
968
|
+
args.push('--cpus', String(config.cpuLimit))
|
|
969
|
+
}
|
|
970
|
+
|
|
971
|
+
for (const [key, value] of Object.entries(options.env ?? {})) {
|
|
972
|
+
args.push('--env', `${key}=${value}`)
|
|
973
|
+
}
|
|
974
|
+
|
|
975
|
+
args.push(config.image)
|
|
976
|
+
return args
|
|
977
|
+
}
|
|
978
|
+
|
|
399
979
|
async function spawnDockerSandbox(
|
|
400
980
|
config: DockerBackendInternalConfig,
|
|
401
981
|
options: SandboxBackendOptions,
|
|
@@ -448,7 +1028,6 @@ async function spawnDockerSandbox(
|
|
|
448
1028
|
await egressProxy?.close().catch(() => undefined)
|
|
449
1029
|
throw err
|
|
450
1030
|
}
|
|
451
|
-
const runtime = config.runtime
|
|
452
1031
|
const containerName = `namzu-sandbox-${id}`
|
|
453
1032
|
|
|
454
1033
|
// All bind sources come from the consumer-supplied layout. The
|
|
@@ -487,97 +1066,14 @@ async function spawnDockerSandbox(
|
|
|
487
1066
|
const rootDir = resolvedLayout.outputs.containerPath
|
|
488
1067
|
|
|
489
1068
|
try {
|
|
490
|
-
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
// another process between our `server.close()` and Docker's
|
|
494
|
-
// `bind()`. Letting Docker pick (`--publish-all`) and reading
|
|
495
|
-
// the mapping back via `docker inspect` removes the race.
|
|
496
|
-
const args: string[] = [
|
|
497
|
-
'run',
|
|
498
|
-
'--detach',
|
|
499
|
-
'--rm',
|
|
500
|
-
'--name',
|
|
1069
|
+
const args = buildDockerRunArgs({
|
|
1070
|
+
config,
|
|
1071
|
+
options,
|
|
501
1072
|
containerName,
|
|
502
|
-
'--network',
|
|
503
1073
|
network,
|
|
504
|
-
|
|
505
|
-
...(
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
// `--label key=value` flags. Validate first — an empty key or
|
|
509
|
-
// a key containing `=` would silently produce a malformed
|
|
510
|
-
// label that downstream `docker ps --filter label=…` queries
|
|
511
|
-
// could not match reliably. Throw before the spawn so misuse
|
|
512
|
-
// surfaces during construction, not as a mysterious "container
|
|
513
|
-
// has no labels" later.
|
|
514
|
-
if (config.labels) {
|
|
515
|
-
for (const [key, value] of Object.entries(config.labels)) {
|
|
516
|
-
if (!key || key.includes('=')) {
|
|
517
|
-
throw new Error(
|
|
518
|
-
`docker label key ${JSON.stringify(key)} is invalid (empty or contains '=')`,
|
|
519
|
-
)
|
|
520
|
-
}
|
|
521
|
-
args.push('--label', `${key}=${value}`)
|
|
522
|
-
}
|
|
523
|
-
}
|
|
524
|
-
|
|
525
|
-
args.push(...renderLayoutMountArgs(resolvedLayout))
|
|
526
|
-
// Forward only the workspace root so the worker's lexical
|
|
527
|
-
// resolver agrees with the bind target. The full layout used
|
|
528
|
-
// to ride along as `NAMZU_SANDBOX_LAYOUT`, but the worker
|
|
529
|
-
// never branched on it; the manifest's only consumer was a
|
|
530
|
-
// log line. A skill loader that needs the manifest will
|
|
531
|
-
// write it to a bind path the worker reads at startup —
|
|
532
|
-
// avoids env-size limits, keeps the wire shape minimal.
|
|
533
|
-
if (egressProxy) {
|
|
534
|
-
// `host-gateway` is docker's own portable name for the host from
|
|
535
|
-
// inside a container; hard-coding a bridge address would break on
|
|
536
|
-
// every platform whose bridge is numbered differently. The proxy
|
|
537
|
-
// itself binds loopback, so this alias is the only way in.
|
|
538
|
-
args.push('--add-host', `${PROXY_HOST_ALIAS}:host-gateway`)
|
|
539
|
-
const proxyUrl = `http://${PROXY_HOST_ALIAS}:${egressProxy.port}`
|
|
540
|
-
// Both spellings: tooling is split between them, and a workload
|
|
541
|
-
// that reads only the one that is missing bypasses the boundary
|
|
542
|
-
// entirely — which would look exactly like the policy working.
|
|
543
|
-
for (const key of ['HTTP_PROXY', 'http_proxy', 'HTTPS_PROXY', 'https_proxy']) {
|
|
544
|
-
args.push('--env', `${key}=${proxyUrl}`)
|
|
545
|
-
}
|
|
546
|
-
// Loopback must not be proxied, or the worker cannot talk to
|
|
547
|
-
// itself.
|
|
548
|
-
args.push('--env', 'NO_PROXY=localhost,127.0.0.1')
|
|
549
|
-
args.push('--env', 'no_proxy=localhost,127.0.0.1')
|
|
550
|
-
}
|
|
551
|
-
|
|
552
|
-
args.push('--env', `NAMZU_SANDBOX_WORKSPACE=${rootDir}`)
|
|
553
|
-
args.push('--env', `NAMZU_SANDBOX_READ_ROOTS=${renderLayoutReadRootsEnv(resolvedLayout)}`)
|
|
554
|
-
args.push('--env', `NAMZU_SANDBOX_WRITE_ROOTS=${renderLayoutWriteRootsEnv(resolvedLayout)}`)
|
|
555
|
-
|
|
556
|
-
// Only publish a host port when the consumer is going to reach
|
|
557
|
-
// the worker through the docker host's loopback (CLI / direct
|
|
558
|
-
// dev). For `container-network` reachability we leave the port
|
|
559
|
-
// unpublished — sibling containers reach the worker by its DNS
|
|
560
|
-
// name on the shared bridge, no host port required.
|
|
561
|
-
if (hostReachability === 'host-port') {
|
|
562
|
-
args.push('--publish', `127.0.0.1::${WORKER_PORT_INSIDE_CONTAINER}`)
|
|
563
|
-
}
|
|
564
|
-
|
|
565
|
-
if (runtime) {
|
|
566
|
-
args.push('--runtime', runtime)
|
|
567
|
-
}
|
|
568
|
-
|
|
569
|
-
if (options.memoryLimitMb && options.memoryLimitMb > 0) {
|
|
570
|
-
args.push('--memory', `${options.memoryLimitMb}m`)
|
|
571
|
-
}
|
|
572
|
-
if (options.maxProcesses && options.maxProcesses > 0) {
|
|
573
|
-
args.push('--pids-limit', String(options.maxProcesses))
|
|
574
|
-
}
|
|
575
|
-
|
|
576
|
-
for (const [key, value] of Object.entries(options.env ?? {})) {
|
|
577
|
-
args.push('--env', `${key}=${value}`)
|
|
578
|
-
}
|
|
579
|
-
|
|
580
|
-
args.push(config.image)
|
|
1074
|
+
hostReachability,
|
|
1075
|
+
...(egressProxy ? { egressProxyPort: egressProxy.port } : {}),
|
|
1076
|
+
})
|
|
581
1077
|
|
|
582
1078
|
await runOnce(docker, args, options.signal)
|
|
583
1079
|
if (hostReachability === 'host-port') {
|