@namzu/sandbox 15.0.0 → 17.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/CHANGELOG.md +324 -0
  2. package/README.md +223 -0
  3. package/dist/backends/aci-standby-pool/index.d.ts +22 -4
  4. package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
  5. package/dist/backends/aci-standby-pool/index.js +31 -7
  6. package/dist/backends/aci-standby-pool/index.js.map +1 -1
  7. package/dist/backends/docker/index.d.ts +408 -26
  8. package/dist/backends/docker/index.d.ts.map +1 -1
  9. package/dist/backends/docker/index.js +1173 -168
  10. package/dist/backends/docker/index.js.map +1 -1
  11. package/dist/backends/firecracker/transport.d.ts +156 -1
  12. package/dist/backends/firecracker/transport.d.ts.map +1 -1
  13. package/dist/backends/firecracker/transport.js +223 -29
  14. package/dist/backends/firecracker/transport.js.map +1 -1
  15. package/dist/backends/http-worker-client.d.ts +64 -2
  16. package/dist/backends/http-worker-client.d.ts.map +1 -1
  17. package/dist/backends/http-worker-client.js +78 -7
  18. package/dist/backends/http-worker-client.js.map +1 -1
  19. package/dist/backends/kubernetes/egress-policy.d.ts +193 -102
  20. package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -1
  21. package/dist/backends/kubernetes/egress-policy.js +321 -146
  22. package/dist/backends/kubernetes/egress-policy.js.map +1 -1
  23. package/dist/backends/kubernetes/per-sandbox-policy.d.ts +6 -6
  24. package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -1
  25. package/dist/backends/kubernetes/per-sandbox-policy.js +21 -53
  26. package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -1
  27. package/dist/backends/kubernetes/transport.d.ts +7 -0
  28. package/dist/backends/kubernetes/transport.d.ts.map +1 -1
  29. package/dist/backends/kubernetes/transport.js.map +1 -1
  30. package/dist/egress/proxy.d.ts +47 -2
  31. package/dist/egress/proxy.d.ts.map +1 -1
  32. package/dist/egress/proxy.js +31 -7
  33. package/dist/egress/proxy.js.map +1 -1
  34. package/dist/index.d.ts +130 -6
  35. package/dist/index.d.ts.map +1 -1
  36. package/dist/index.js +55 -5
  37. package/dist/index.js.map +1 -1
  38. package/package.json +4 -4
  39. package/src/backends/aci-standby-pool/index.ts +37 -7
  40. package/src/backends/docker/index.ts +1475 -196
  41. package/src/backends/firecracker/transport.ts +387 -36
  42. package/src/backends/http-worker-client.ts +89 -5
  43. package/src/backends/kubernetes/egress-policy.ts +455 -187
  44. package/src/backends/kubernetes/per-sandbox-policy.ts +21 -66
  45. package/src/backends/kubernetes/transport.ts +7 -0
  46. package/src/egress/proxy.ts +65 -8
  47. package/src/index.ts +162 -5
@@ -14,17 +14,37 @@
14
14
  * Trust model:
15
15
  * - Container is the trust boundary; everything inside is treated
16
16
  * as untrusted code.
17
- * - Worker only listens on loopback inside its own netns; the
18
- * host adapter reaches it via Docker's port-forward.
19
- * - Outbound network from the worker is restricted by host-side
20
- * firewall config (see {@link DockerBackendConfig.network}) plus
21
- * the egress proxy when one is configured (P3.2).
17
+ * - Every call to the worker's control API carries the per-instance
18
+ * `NAMZU_SANDBOX_TOKEN` this backend mints at create time and
19
+ * injects into the container's environment; a worker the host did
20
+ * not create must be provisioned with its own. The worker requires
21
+ * it on every route but `/healthz`, and refuses to start at all if
22
+ * it has none and is bound to anything routable.
23
+ * - Outbound network from the worker is restricted by the network
24
+ * it is attached to (see {@link DockerBackendConfig.network}) and,
25
+ * for a host allowlist, by an egress proxy running as a sibling
26
+ * container on that same network — the only way its traffic reaches
27
+ * the internet, because the network is `--internal` and has no route
28
+ * off it. The proxy is a container on that subnet like any other, so
29
+ * it is not the sandbox's only reachable destination. The proxy
30
+ * environment the sandbox is given directs traffic; the topology is
31
+ * what confines it. See {@link assertNetworkCarriesThePolicy} and
32
+ * #398.
33
+ *
34
+ * The credential above is what a previous version of this docblock
35
+ * claimed network placement alone provided. It said the worker "only
36
+ * listens on loopback inside its own netns", which is false — the worker
37
+ * binds every interface by default and has to, because a published
38
+ * container port forwards to the container's interface address rather
39
+ * than to its loopback, so a loopback-bound worker is unreachable through
40
+ * the port this backend publishes. The boundary was the network the
41
+ * container is attached to, and wanted a credential behind it.
22
42
  */
23
43
  import { spawn } from 'node:child_process';
44
+ import { randomBytes } from 'node:crypto';
24
45
  import { SANDBOX_DEFAULT_OUTPUTS_PATH, SANDBOX_DEFAULT_SCRATCH_PATH, SANDBOX_DEFAULT_SKILLS_PARENT, SANDBOX_DEFAULT_TOOL_RESULTS_PATH, SANDBOX_DEFAULT_TRANSCRIPTS_PATH, SANDBOX_DEFAULT_UPLOADS_PATH, generateSandboxId, walkFilesViaExec, withHint, } from '@namzu/sdk';
25
- import { EgressProxy } from '../../egress/index.js';
26
46
  import { ContainerSandboxLayoutValidationError, } from '../../index.js';
27
- import { HttpWorkerClient } from '../http-worker-client.js';
47
+ import { HttpWorkerClient, WORKER_UNAUTHORIZED_HINT, workerAuthorization, } from '../http-worker-client.js';
28
48
  import { OperationDeadline, OperationDeadlineExpired, probeHttpHealth, resolveReadinessOptions, runFailureCleanup, } from '../readiness.js';
29
49
  import { RemoteCancellationUnknownError, RemoteProtocolError, } from '../remote-execution-controller.js';
30
50
  const DEFAULT_DOCKER_BINARY = 'docker';
@@ -37,6 +57,13 @@ const WORKER_PORT_INSIDE_CONTAINER = 2024;
37
57
  * `create()` call.
38
58
  */
39
59
  export function buildDockerBackend(config) {
60
+ // Refused here rather than at the first spawn, so a config that cannot be
61
+ // rendered — a CPU limit that cannot mean anything, or writable paths
62
+ // beside a writable root filesystem — surfaces during host wiring instead
63
+ // of as a container that failed to come up. The same checks run where the
64
+ // argv is built, because that is the only place a caller cannot skip them.
65
+ assertCpuLimitIsRenderable(config.cpuLimit);
66
+ assertRootfsOptionsAreCoherent(config);
40
67
  const readiness = resolveReadinessOptions('docker', config.readyTimeoutMs, config.readyPollIntervalMs, {
41
68
  timeoutMs: DEFAULT_READY_TIMEOUT_MS,
42
69
  pollIntervalMs: DEFAULT_READY_POLL_MS,
@@ -65,6 +92,14 @@ export function buildDockerBackend(config) {
65
92
  * just the way out. It now keeps the configured network, and
66
93
  * {@link assertNetworkCarriesThePolicy} is what makes that network a
67
94
  * boundary.
95
+ *
96
+ * Every policy that returns the configured network depends on that same
97
+ * assertion, which is why the two live side by side: this function says which
98
+ * network the container joins, and that one refuses the network if it cannot
99
+ * do what the policy asks of it (#398). An allowlist used to answer the
100
+ * configured network and get a proxy environment variable pointed at a
101
+ * host-side listener — enforcement by convention, which an uncooperative
102
+ * process simply ignores.
68
103
  */
69
104
  export function resolveNetwork(configured, egress, hasProxy = false) {
70
105
  if (!egress)
@@ -81,7 +116,7 @@ export function resolveNetwork(configured, egress, hasProxy = false) {
81
116
  // reporting that it had been restricted.
82
117
  if (hasProxy)
83
118
  return configured;
84
- throw new Error(`The docker sandbox backend cannot enforce an egress policy of kind '${egress.kind}' without an egress proxy: it has nothing to filter hosts through. Construct the provider with one, or use 'deny-all' / 'allow-all'. Refusing rather than silently granting full network access.`);
119
+ throw new Error(`The docker sandbox backend cannot enforce an egress policy of kind '${egress.kind}' without an egress proxy: it has nothing to filter hosts through. Name the image it should run the proxy as (egressProxyImage — see packages/sandbox/egress-proxy/Dockerfile), or use 'deny-all' / 'allow-all'. Refusing rather than silently granting full network access.`);
85
120
  }
86
121
  }
87
122
  /**
@@ -118,9 +153,9 @@ export function isInternalNetwork(inspectedInternalFlag) {
118
153
  /**
119
154
  * Refuse a container whose network cannot do what was asked of it.
120
155
  *
121
- * Two requirements meet on the same object here, and both were previously
122
- * unstated — which is how the backend came to ship a default configuration
123
- * that could not create a sandbox at all:
156
+ * Three requirements meet on the same object here, and the first two were
157
+ * previously unstated — which is how the backend came to ship a default
158
+ * configuration that could not create a sandbox at all:
124
159
  *
125
160
  * - **A published host port needs a route out.** Docker binds the port by
126
161
  * NAT to the container's address, so a container with no address gets no
@@ -136,12 +171,31 @@ export function isInternalNetwork(inspectedInternalFlag) {
136
171
  * nothing. `deny-all` pointed at the default bridge would be full egress
137
172
  * under a policy object claiming none — the "accepted and silently
138
173
  * ignored" failure the rest of this file exists to refuse.
174
+ * - **A host allowlist needs one too, for the same reason and with one
175
+ * addition.** Until #398 the allowlist tier answered the configured
176
+ * network and pointed the sandbox at a proxy running on the host's
177
+ * loopback, and the only thing making traffic go through it was
178
+ * `HTTP_PROXY`. That is a request, not a boundary: anything inside the
179
+ * container that opens a socket directly reaches the network with the
180
+ * allowlist unconsulted, and untrusted code is the caller least likely to
181
+ * honour a convention. With the proxy as a sibling container on an
182
+ * `--internal` network, the only way the sandbox's traffic reaches the
183
+ * internet is that container, and it cannot change that because
184
+ * `--cap-drop=ALL` took `NET_ADMIN` away (see {@link HARDENING_ARGS}). The
185
+ * environment variables stay and now DIRECT traffic rather than permit it.
186
+ * On a network that is not internal there is no such route to remove, and
187
+ * the policy is back to being advisory — which is what this refuses.
139
188
  *
140
- * They are exact opposites, so `deny-all` over a published host port is
141
- * impossible rather than merely unsupported: no arrangement of docker
189
+ * The first two are exact opposites, so `deny-all` over a published host port
190
+ * is impossible rather than merely unsupported: no arrangement of docker
142
191
  * networking both denies all egress and lets the host reach the worker over
143
- * TCP. Closing that needs the control channel moved off TCP — see #398 —
144
- * and is not a flag this function could accept.
192
+ * TCP. Closing that needs the control channel moved off TCP — see #398 — and
193
+ * is not a flag this function could accept. The same opposition now reaches
194
+ * an allowlist policy, which is new: an allowlist on an internal network must
195
+ * be reached by container name, so a host that used a published port with a
196
+ * host allowlist has to move that consumer onto the internal network. That is
197
+ * a consequence of the boundary existing at all and not a gap in this
198
+ * function; the refusal below names the mode to move to.
145
199
  */
146
200
  export function assertNetworkCarriesThePolicy(network, reachability, egress, inspectedInternalFlag) {
147
201
  const internal = isInternalNetwork(inspectedInternalFlag);
@@ -151,26 +205,154 @@ export function assertNetworkCarriesThePolicy(network, reachability, egress, ins
151
205
  if (egress?.kind === 'deny-all' && !internal) {
152
206
  throw new Error(`The docker sandbox backend was asked for an egress policy of 'deny-all' on network '${network}', but that network is not internal, so the container can still reach the world. Create it with 'docker network create --internal ${network}' — an internal bridge denies egress in the kernel, rather than through an environment variable a workload may decline to read, while sibling containers still reach the worker by name. Refusing rather than reporting a boundary that is not there.`);
153
207
  }
208
+ if (needsEgressProxy(egress) && !internal) {
209
+ throw new Error(`The docker sandbox backend was asked for an egress policy of kind '${egress?.kind}' on network '${network}', but that network is not internal, so nothing stops the container from reaching the world directly and the allowlist would only be a proxy environment variable a workload may decline to read. Create the network with 'docker network create --internal ${network}': the sandbox then reaches the internet only through the egress proxy container, which is the boundary — and set hostReachability: 'container-network' to reach the worker by name on it, since a published host port needs a route out this network does not have. Refusing rather than reporting a boundary that is not there.`);
210
+ }
154
211
  }
155
212
  /**
156
- * The options the boundary is built from, as a value.
213
+ * Build the proxy container's configuration.
157
214
  *
158
- * Extracted for the same reason {@link resolveNetwork} is: everything
159
- * downstream of here needs a running Docker daemon, so a policy that never
160
- * reached the proxy could only be caught by an operator noticing their
161
- * traffic denied in production. A knob a host sets and the boundary never
162
- * receives is the failure this shape exists to make testable.
215
+ * `allowedHosts` is already resolved here rather than passed as a policy, and
216
+ * that is the one behavioural difference this change carries into the
217
+ * boundary: the in-process proxy called the host's resolver per request, and
218
+ * the container cannot. A `resolver` policy that rotates is honoured at
219
+ * `create()` and again at every `setNetworkPolicy()` — the container has no
220
+ * channel back to the host's resolver, and a channel the container CAN reach
221
+ * is one the sandbox can reach too, which would let the sandbox ask for its
222
+ * own allowlist to be widened. See `docs/sdk/sandbox-egress.md`.
163
223
  */
164
- export function egressProxyOptions(config, policy) {
224
+ export function egressProxyContainerConfig(config, allowedHosts, port) {
165
225
  return {
166
- // Re-resolved per request rather than captured once, so a `resolver`
167
- // policy that rotates is honoured and `setNetworkPolicy` can swap it
168
- // on a live sandbox.
169
- allowedHosts: () => resolveAllowedHosts(policy),
226
+ port,
227
+ allowedHosts,
170
228
  credentials: config.brokeredCredentials ?? [],
171
229
  ...(config.allowInwardFor ? { allowInwardFor: config.allowInwardFor } : {}),
230
+ selfNames: [PROXY_HOST_ALIAS],
172
231
  };
173
232
  }
233
+ /**
234
+ * `--label key=value` flags, validated before they reach the daemon.
235
+ *
236
+ * An empty key or one containing `=` would silently produce a malformed label
237
+ * that a downstream `docker ps --filter label=…` could not match, so misuse
238
+ * surfaces during construction rather than as a container that mysteriously
239
+ * has no labels.
240
+ */
241
+ function renderLabelArgs(labels) {
242
+ const args = [];
243
+ if (!labels)
244
+ return args;
245
+ for (const [key, value] of Object.entries(labels)) {
246
+ if (!key || key.includes('=')) {
247
+ throw new Error(`docker label key ${JSON.stringify(key)} is invalid (empty or contains '=')`);
248
+ }
249
+ args.push('--label', `${key}=${value}`);
250
+ }
251
+ return args;
252
+ }
253
+ /**
254
+ * Name the proxy container reads its configuration from.
255
+ *
256
+ * The NAME travels in the argv and the VALUE travels in the `docker` CLI
257
+ * child's environment, which is why this is one constant used by both ends of
258
+ * that pair rather than a literal written twice.
259
+ */
260
+ const EGRESS_PROXY_CONFIG_ENV = 'NAMZU_EGRESS_PROXY_CONFIG';
261
+ /**
262
+ * The `docker run` argv for the egress proxy container, as a value.
263
+ *
264
+ * Extracted for the reason every other argv in this file is: nothing
265
+ * downstream of it can run without a docker daemon, so a flag that never
266
+ * reached it — or a `--network` that named the wrong side of a dual-homed
267
+ * container — could only be caught by an operator in production, where the
268
+ * symptom is a sandbox that reaches nothing.
269
+ *
270
+ * The two networks are the whole topology and they are two calls:
271
+ * `docker run --network <upstream>` gives the container a default route, and
272
+ * {@link renderEgressProxyAttachArgs}'s `docker network connect` adds the
273
+ * internal network afterwards. The order is load-bearing. Attached to the
274
+ * internal network FIRST it would come up with no default route and no way to
275
+ * acquire one, and the proxy would be a boundary in front of nothing.
276
+ *
277
+ * The hardening baseline is the sandbox's, minus the parts that describe a
278
+ * filesystem. `--cap-drop=ALL`, `--no-new-privileges` and `--ipc private` are
279
+ * applied exactly as {@link HARDENING_ARGS} defines them, because this
280
+ * container sits between untrusted code and the internet and is the last one in
281
+ * the deployment that should be holding a capability. `--read-only` is the
282
+ * sandbox's too, with one tmpfs: the proxy writes nothing, and the tmpfs is
283
+ * there so that a Node process which one day wants a temp file fails at a
284
+ * filesystem boundary rather than at a mysterious `EROFS`.
285
+ *
286
+ * The image is last, so nothing after it is read as a flag — the same rule the
287
+ * sandbox argv follows, and here it also means the image's own `CMD` starts the
288
+ * proxy rather than this argv naming an entrypoint.
289
+ */
290
+ export function renderEgressProxyRunArgs(input) {
291
+ const { config, containerName, upstreamNetwork, internalNetwork } = input;
292
+ if (!config.egressProxyImage) {
293
+ throw new Error('renderEgressProxyRunArgs was called without config.egressProxyImage; the docker backend cannot start an egress proxy container it has no image for. resolveNetwork refuses an allowlist policy in this state, so reaching here means the refusal was bypassed.');
294
+ }
295
+ if (upstreamNetwork === 'none') {
296
+ throw new Error(`egressProxyUpstreamNetwork is 'none', which would give the proxy container no interface and no route: it would come up unable to reach anything, and the only way the sandbox's traffic reaches the internet would be a boundary that cannot reach it itself. Name a bridge, or drop the field to take the 'bridge' default.`);
297
+ }
298
+ if (upstreamNetwork === internalNetwork) {
299
+ throw new Error(`egressProxyUpstreamNetwork names '${upstreamNetwork}', which is the same network the sandbox is on — the internal one. A container whose primary network is internal comes up with no default route and never acquires one, so the proxy would start with no way to reach the internet: the boundary would be a container that can only talk to the sandbox. Name a different network for egressProxyUpstreamNetwork, or drop the field to take the 'bridge' default.`);
300
+ }
301
+ return [
302
+ 'run',
303
+ '--detach',
304
+ '--rm',
305
+ '--name',
306
+ containerName,
307
+ // The name the sandbox dials, and the name the proxy's own loop guard
308
+ // has to recognise as itself. Set as the container's hostname so the
309
+ // entrypoint can read it back with `os.hostname()` instead of holding a
310
+ // second copy of the constant.
311
+ '--hostname',
312
+ PROXY_HOST_ALIAS,
313
+ '--network',
314
+ upstreamNetwork,
315
+ ...HARDENING_ARGS,
316
+ '--read-only',
317
+ '--tmpfs',
318
+ `/tmp:${TMPFS_MOUNT_OPTIONS}`,
319
+ ...renderLabelArgs(config.labels),
320
+ // The NAME only: docker takes the value from the environment of the
321
+ // `docker` CLI process it names, which the caller sets (see
322
+ // `startEgressProxyContainer`). The value is the whole policy —
323
+ // including brokered credential values — and an argv is a worse place
324
+ // for it than an environment on every platform that has a process
325
+ // table: `/proc/<pid>/cmdline` is world-readable on Linux, so putting
326
+ // it here published it to every local user on the docker host for as
327
+ // long as the client ran, where the child's own environment is readable
328
+ // only by the user that owns the process. It is still readable by
329
+ // anything with access to the daemon, through `docker inspect`, and
330
+ // `docs/sdk/sandbox-egress.md` says so.
331
+ '--env',
332
+ EGRESS_PROXY_CONFIG_ENV,
333
+ config.egressProxyImage,
334
+ ];
335
+ }
336
+ /**
337
+ * The `docker network connect` argv that makes the proxy dual-homed.
338
+ *
339
+ * `--alias` is what puts `namzu-egress` in the internal network's DNS, which
340
+ * is the name {@link buildDockerRunArgs} hands the sandbox in its proxy
341
+ * environment. The alias rather than the container name because the alias is
342
+ * the constant: a container named `namzu-egress-<sandbox-id>` would otherwise
343
+ * make the sandbox's `HTTP_PROXY` value depend on a generated id, and the two
344
+ * would have to be kept in step by hand.
345
+ */
346
+ export function renderEgressProxyAttachArgs(input) {
347
+ return [
348
+ 'network',
349
+ 'connect',
350
+ '--alias',
351
+ PROXY_HOST_ALIAS,
352
+ input.internalNetwork,
353
+ input.containerName,
354
+ ];
355
+ }
174
356
  /**
175
357
  * Confinement flags applied to every container.
176
358
  *
@@ -181,9 +363,11 @@ export function egressProxyOptions(config, policy) {
181
363
  * are the defaults every container runtime hardening guide starts with,
182
364
  * and none of them were present.
183
365
  *
184
- * `--cap-drop=ALL` is deliberately not softened by a re-add list: a
185
- * workload that genuinely needs a capability should say so through
186
- * `extraRunArgs` and be visible in review.
366
+ * `--cap-drop=ALL` is deliberately not softened by a re-add list, and there is
367
+ * no config field that could soften it either: a workload that genuinely needs
368
+ * a capability needs a change to this file, where the diff says which
369
+ * capability and why. A re-add list on the config would grant it to every
370
+ * sandbox the host spawns, quietly, which is how a baseline stops being one.
187
371
  *
188
372
  * **It carries a second, independent load, and this is the one that would
189
373
  * survive being forgotten.** An egress policy of `deny-all` is enforced by
@@ -208,51 +392,564 @@ export function egressProxyOptions(config, policy) {
208
392
  *
209
393
  * Recorded here because the first justification above would survive
210
394
  * softening this flag and the second would not.
395
+ *
396
+ * `--ipc private` closes a door that is not the one its name suggests, and the
397
+ * difference is worth being exact about. Moby runs `private`, `shareable` and
398
+ * `none` through the SAME branch (`daemon/oci_linux.go`, `WithNamespaces`), so
399
+ * `shareable` already gives every container an IPC namespace of its own: a
400
+ * daemon whose `default-ipc-mode` is `shareable` does not merge anybody's
401
+ * namespaces, and what an unset flag buys on such a daemon is not a shared one
402
+ * either. What separates the modes is reachability. Docker's run reference
403
+ * defines `shareable` as "Own private IPC namespace, with a possibility to
404
+ * share it with other containers", and that possibility is `--ipc
405
+ * container:<name>`, which joins another container's IPC namespace — and what
406
+ * that join needs from the target is a shared-memory directory to enter:
407
+ * `daemon.getIPCContainer` resolves the target by name and is gated on its
408
+ * `ShmPath`, which a container created `private` does not have. So on a
409
+ * daemon defaulting to `shareable` this container's namespace, and the System V
410
+ * shared memory, semaphores and message queues namespaced with it, are joinable
411
+ * by anything else on that host which knows the container's name; `--ipc
412
+ * private` removes that reachability. Creating the joining container still takes
413
+ * access to the same daemon, so this is not a boundary against an unprivileged
414
+ * attacker, and it is not claimed as one here. What it buys is that the answer
415
+ * is in THIS argv rather than in the host's `daemon.json`, which is the only
416
+ * place the daemon's default is written down. `--ipc none` is deliberately not
417
+ * used: it takes `/dev/shm` away, and chromium — which the reference image
418
+ * ships for browser automation — uses it for every renderer process.
419
+ *
420
+ * `--read-only` makes the image itself not a place the workload can write. It
421
+ * is rendered by {@link renderHardeningArgs} rather than listed below, because
422
+ * it is the one flag here a host can turn off (`readOnlyRootfs: false`), and a
423
+ * flag in this array would keep being applied after the field said it was not —
424
+ * a control accepted and not applied, which is the failure this file refuses
425
+ * everywhere else. The layout's own RW binds (`outputs`, `scratch`) are
426
+ * separate mounts and are unaffected; what stays writable inside the
427
+ * container's filesystem is named, path by path and with the reason, in
428
+ * {@link renderWritableRootfsArgs}. A host whose image needs a path that list
429
+ * does not name adds it through `writableRootfsPaths`.
430
+ *
431
+ * Three controls from the published container-hardening guidance are
432
+ * deliberately absent, and the reason is here rather than implied:
433
+ *
434
+ * - **A seccomp profile.** Docker already applies its built-in profile to
435
+ * every container unless something passes `seccomp=unconfined`, and nothing
436
+ * in this backend does — so the tier is filtered, and what is missing is a
437
+ * profile TIGHTER than docker's default. Shipping one means shipping a
438
+ * hand-written file whose deny list has to be correct for whatever image
439
+ * the host names, and this repository cannot test it against the reference
440
+ * image's own toolchain (chromium, LibreOffice, the numpy/scipy/duckdb
441
+ * stack). A profile that blocks a syscall one of those needs breaks the
442
+ * sandbox at a point no test here would catch, which is worse than the gap
443
+ * it closes. A host that needs a tighter profile sets `seccomp-profile` in
444
+ * the daemon's `daemon.json`, where it applies to this container and every
445
+ * other one; `--security-opt seccomp=<file>` is the per-container form, and
446
+ * it is not offered as a config field because a path in a config field is a
447
+ * file the daemon reads from the HOST, which is a different machine from
448
+ * the one this backend runs on whenever it drives a remote daemon.
449
+ * - **`--userns-remap`.** It is not a `docker run` flag at all: it is a
450
+ * daemon property (`userns-remap` in `daemon.json`, or `dockerd
451
+ * --userns-remap=`), and per container the CLI only chooses between the
452
+ * namespaces the daemon already made (`--userns=host|private`). Whether a
453
+ * remapped namespace exists is therefore settled before this argv is read,
454
+ * and a flag here could not settle it — which is the whole reason the
455
+ * control is absent rather than configurable: this backend has nothing to
456
+ * say about a mapping that belongs to the machine the daemon runs on.
457
+ * Enabling it on the host is a real upgrade to this tier (uid 0 inside maps
458
+ * to an unprivileged uid outside) and costs this backend nothing; the README
459
+ * says so.
460
+ * - **`--user`.** Supported, and unset by default on purpose — see the
461
+ * `runAsUser` field, which is where a host that knows its image sets it.
462
+ */
463
+ const HARDENING_ARGS = [
464
+ '--cap-drop=ALL',
465
+ '--security-opt=no-new-privileges',
466
+ '--ipc',
467
+ 'private',
468
+ ];
469
+ /**
470
+ * Name the sandbox reaches the egress proxy by.
471
+ *
472
+ * This used to be a `--add-host namzu-egress:host-gateway` entry, because the
473
+ * proxy ran on the host's loopback and `host-gateway` is docker's portable
474
+ * name for the host from inside a container. It is now the proxy's own
475
+ * container name and network alias on the internal network, which needs no
476
+ * alias file at all: docker's embedded DNS resolves a container's aliases for
477
+ * every container on the same user-defined network. The constant survives the
478
+ * mechanism because what the sandbox is told has not changed — only what makes
479
+ * the name resolve, and whether anything else can be reached.
211
480
  */
212
- const HARDENING_ARGS = ['--cap-drop=ALL', '--security-opt=no-new-privileges'];
213
- /** Name the container reaches the host-side egress proxy by. */
214
481
  const PROXY_HOST_ALIAS = 'namzu-egress';
482
+ /**
483
+ * Port the egress proxy listens on inside its own container.
484
+ *
485
+ * A fixed port rather than one the host reads back, because there is no host
486
+ * port involved: the sandbox dials the proxy container directly on the network
487
+ * they share. The old arrangement had to read a published port back out of
488
+ * `docker inspect`, with the race and the failure mode that came with it.
489
+ */
490
+ const EGRESS_PROXY_PORT_INSIDE_CONTAINER = 2025;
491
+ /** Name of the sibling container the egress proxy runs in, for one sandbox. */
492
+ function egressProxyContainerName(sandboxId) {
493
+ return `${PROXY_HOST_ALIAS}-${sandboxId}`;
494
+ }
495
+ /**
496
+ * Mount options for every scratch mount this backend creates.
497
+ *
498
+ * `exec` is the load-bearing one and the reason this is a named constant
499
+ * rather than a literal at the call site. Docker does NOT default a `--tmpfs`
500
+ * mount to a usable scratch directory: `withMounts` in moby's
501
+ * `daemon/oci_linux.go` starts every user tmpfs from
502
+ * `["noexec", "nosuid", "nodev", <propagation>]` and appends whatever the
503
+ * caller passed, so `--tmpfs /tmp` on its own is **noexec**. A workload that
504
+ * compiles a program into `/tmp` and runs it — `gcc -o /tmp/a.out … &&
505
+ * /tmp/a.out`, or a python `ctypes.CDLL` of a library it just built there —
506
+ * would meet `Permission denied` on an executable file, an error that reads
507
+ * as a broken sandbox rather than as a mount option. Scratch here is as
508
+ * executable as it was before this backend mounted a tmpfs over it.
509
+ *
510
+ * `nosuid` and `nodev` are kept from docker's defaults: the tmpfs is the one
511
+ * place inside the container a workload can write an arbitrary file to, and
512
+ * neither a setuid binary nor a device node there has any use that is worth
513
+ * the escalation path — with `--cap-drop=ALL` no device node could be created
514
+ * there anyway.
515
+ *
516
+ * `mode=1777` is stated rather than inherited from the kernel's tmpfs default
517
+ * (which is the same value): the mounts have to be writable by whichever uid
518
+ * the image runs as, and the backend does not know that uid. A sticky,
519
+ * world-writable scratch directory is what `/tmp` is, and `--read-only` here
520
+ * is about the image, not about the uid.
521
+ */
522
+ const TMPFS_MOUNT_OPTIONS = 'nosuid,nodev,exec,mode=1777';
523
+ /**
524
+ * Paths the reference image needs writable under `--read-only`, as `--tmpfs`.
525
+ *
526
+ * `--read-only` says the image is not the workload's disk. It does not say
527
+ * nothing may be written, and the difference is the sandbox: the layout's own
528
+ * RW binds (`outputs`, `scratch`) are separate mounts and are unaffected, but
529
+ * the image's toolchain writes inside the container's own filesystem, and a
530
+ * `--read-only` that stops it is worse than the gap it closes. Read off
531
+ * `worker/Dockerfile`, whose whole purpose is producing DOCX/XLSX/PPTX/PDF
532
+ * deliverables:
533
+ *
534
+ * - `/tmp` — `TMPDIR` for python's `tempfile`, for LibreOffice's extraction
535
+ * and for pip's wheel builds, and the conventional place to build and run
536
+ * something disposable. Every scratch mount takes
537
+ * {@link TMPFS_MOUNT_OPTIONS}, which is where the `exec` docker would not
538
+ * have given us is argued for.
539
+ * - `/var/tmp` — the second location the temp-file conventions fall back to,
540
+ * for a temp file that is meant to outlive an interrupted run.
541
+ * - `/home/namzu` — the image's `HOME` (`useradd --create-home namzu`, uid
542
+ * 1001; docker sets `HOME` from the image's passwd entry). LibreOffice
543
+ * refuses a headless conversion without a writable user profile
544
+ * (`~/.config/libreoffice`), matplotlib builds a font cache in
545
+ * `~/.cache/matplotlib`, fontconfig keeps a user cache, npm's cache is
546
+ * `~/.npm`, and `pip install --user` needs `~/.local`.
547
+ * - `/workspace` — the image's `WORKDIR`, chowned to `namzu` on purpose
548
+ * (`chown -R namzu:namzu /workspace`). Leaving it out would make the
549
+ * Dockerfile's own guarantee false.
550
+ *
551
+ * These four are the REFERENCE image's needs, not a claim about anyone else's.
552
+ * A host that points `image` at its own build names what that image needs in
553
+ * `writableRootfsPaths`, which is why the field exists at all: the backend
554
+ * cannot read an image's writable set, and the alternative to asking is
555
+ * guessing. A path a root-running image wants (its `HOME` is `/root`) is a
556
+ * `writableRootfsPaths` entry for exactly that reason — `/root` is not in this
557
+ * list, because the shipped image does not run as root and a tmpfs nobody
558
+ * writes to is a claim that something does.
559
+ *
560
+ * A path the LAYOUT already mounts is skipped rather than mounted twice:
561
+ * docker refuses two mounts at one destination (`Duplicate mount point`), and
562
+ * the bind the host asked for is the one that must win. A path the HOST names
563
+ * that the layout also mounts is refused instead of skipped, because there the
564
+ * two requests contradict each other and nothing should choose between them
565
+ * silently.
566
+ *
567
+ * No `size=` is set. The kernel caps a tmpfs at half the host's RAM, and tmpfs
568
+ * pages are accounted to the container's memory cgroup, so a run that sets
569
+ * `--memory` already bounds scratch with the limit the host chose — while any
570
+ * number picked here would fail a workload that writes a bigger temp file than
571
+ * we guessed, with `ENOSPC` rather than a diagnosis.
572
+ *
573
+ * **The other half of that trade, said out loud because a host will meet it.**
574
+ * Scratch now lives in RAM instead of on the container's writable layer, so a
575
+ * temp file larger than half the host's RAM — or larger than `--memory`, which
576
+ * is the tighter of the two whenever the host set one — fails with `ENOSPC` or
577
+ * is OOM-killed, where writing it to disk used to succeed. That is the cost of
578
+ * not leaving the root filesystem writable, and it is not a bug to be reported.
579
+ * The remedy that keeps the baseline is the layout's own `scratch`, which is a
580
+ * bind to a host directory and therefore still disk-backed: a host with room on
581
+ * disk gives the layout one there and points `TMPDIR` at its container path
582
+ * through the per-call `env` option, so the spill lands on that disk instead of
583
+ * on a tmpfs. `readOnlyRootfs: false` is the other way, and the one to reach for
584
+ * second: it puts scratch back on the container's writable layer and gives up
585
+ * the rest of the baseline with it.
586
+ */
587
+ const DEFAULT_WRITABLE_ROOTFS_PATHS = [
588
+ '/tmp',
589
+ '/var/tmp',
590
+ '/workspace',
591
+ '/home/namzu',
592
+ ];
593
+ /**
594
+ * The spelling docker compares a container path by.
595
+ *
596
+ * Docker cleans a mount destination before it uses it, so `/tmp/`, `//tmp` and
597
+ * `/tmp/.` are one directory to it and to the kernel. The check below is an
598
+ * exact-string comparison, so without this a layout that spelled one of its
599
+ * mounts any of those ways would not match the tmpfs default at the same
600
+ * directory: the argv would carry both a `--tmpfs /tmp:...` and a bind at
601
+ * `/tmp/`, and moby would clean the two destinations into one and refuse the
602
+ * container at spawn with `Duplicate mount point: /tmp` — the failure the check
603
+ * exists to prevent, on the one path no test in this repository can reach.
604
+ * `resolveLayout` does not normalise these (it fills in defaults and compares
605
+ * spellings as written), so the cleaning has to happen here, where the
606
+ * comparison does.
607
+ */
608
+ function cleanContainerPath(path) {
609
+ const kept = [];
610
+ for (const segment of path.split('/')) {
611
+ // Empty segments are `//`, `.` is the directory itself; `..` cancels the
612
+ // segment before it, which is what the kernel does with it too.
613
+ if (segment === '' || segment === '.')
614
+ continue;
615
+ if (segment === '..')
616
+ kept.pop();
617
+ else
618
+ kept.push(segment);
619
+ }
620
+ return `/${kept.join('/')}`;
621
+ }
622
+ /**
623
+ * Every destination the layout mounts something at, in the spelling docker
624
+ * itself compares them by.
625
+ *
626
+ * The collision check below is exact-string, so a layout path spelled `/tmp/`
627
+ * would slip past it and docker would then refuse the container with
628
+ * `Duplicate mount point` — a failure at spawn, on the one path that cannot be
629
+ * tested without a daemon. Cleaning each path to the spelling moby reduces it
630
+ * to is what makes the check cover every way of writing the same directory;
631
+ * see {@link cleanContainerPath}.
632
+ */
633
+ function mountedContainerPaths(layout) {
634
+ return [
635
+ layout.outputs.containerPath,
636
+ layout.uploads?.containerPath,
637
+ layout.scratch?.containerPath,
638
+ layout.toolResults?.containerPath,
639
+ layout.transcripts?.containerPath,
640
+ ...(layout.skills?.map((skill) => skill.containerPath) ?? []),
641
+ ]
642
+ .filter((path) => Boolean(path))
643
+ .map(cleanContainerPath);
644
+ }
645
+ /**
646
+ * Refuse rootfs options that cannot both be honoured.
647
+ *
648
+ * `writableRootfsPaths` beside `readOnlyRootfs: false` is a contradiction: with
649
+ * a writable root filesystem every path is already writable, so the tmpfs
650
+ * mounts would either be dropped (a control accepted and not applied) or take a
651
+ * directory off the image for no reason. Refusing is the honest answer, and it
652
+ * is the same one the sibling backends give a per-sandbox control they cannot
653
+ * express.
654
+ *
655
+ * Called at construction and again where the argv is built, so a config that
656
+ * reaches `create()` by some path other than `buildDockerBackend` is refused
657
+ * too.
658
+ */
659
+ export function assertRootfsOptionsAreCoherent(config) {
660
+ const paths = config.writableRootfsPaths;
661
+ if (config.readOnlyRootfs !== false || paths === undefined || paths.length === 0)
662
+ return;
663
+ throw new Error('writableRootfsPaths was set on a docker backend configured with readOnlyRootfs: false. With a writable root filesystem every path inside the container is already writable, so these --tmpfs mounts would add nothing and take the named directories off the image. Refusing rather than accepting a control that cannot be applied: drop the paths, or drop readOnlyRootfs: false and let the read-only baseline stand.');
664
+ }
665
+ /**
666
+ * Refuse a `--cpus` value that cannot mean what it says.
667
+ *
668
+ * This covers non-finite and non-positive values and does NOT claim to cover
669
+ * every bound the daemon would refuse. The difference is worth stating, because
670
+ * the two classes fail in different places and only one of them is decidable
671
+ * here. A negative, `NaN` or `Infinity` renders into the argv as text the
672
+ * daemon either rejects or turns into a bound nobody asked for, and `0` is the
673
+ * opposite of a bound (`NanoCPUs` of zero is how a container says "no CPU
674
+ * limit"), so a host that wrote one of those hears about it during wiring
675
+ * rather than as a container that never came up.
676
+ *
677
+ * The upper bound is not ours to check. Moby's `verifyPlatformContainerResources`
678
+ * refuses `NanoCPUs` above the DAEMON host's CPU count (`"range of CPUs is from
679
+ * 0.01 to N.00, as there are only N CPUs available"`), and the same function
680
+ * deliberately sets no floor of its own on Linux, leaving that to the kernel.
681
+ * Neither number is knowable from here: the `docker` binary this backend drives
682
+ * can be pointed at a daemon on another machine (`DOCKER_HOST`), and even
683
+ * locally `os.cpus().length` is this machine's view rather than the daemon's
684
+ * own `runtime.NumCPU()`. Refusing on a guess at it would break a host whose
685
+ * daemon has more cores than the process driving it, which is a worse failure
686
+ * than the one it would catch — those arrive from the daemon with its own
687
+ * message, at spawn, where every other daemon-side refusal arrives too.
688
+ */
689
+ export function assertCpuLimitIsRenderable(cpuLimit) {
690
+ if (cpuLimit === undefined)
691
+ return;
692
+ if (!Number.isFinite(cpuLimit) || cpuLimit <= 0) {
693
+ throw new Error(`cpuLimit must be a finite number greater than 0 (docker's --cpus takes a decimal, e.g. 1.5); got ${String(cpuLimit)}. Refusing rather than rendering an argv whose value means something other than what was written.`);
694
+ }
695
+ }
696
+ /**
697
+ * `--tmpfs` flags for the paths that stay writable under `--read-only`.
698
+ *
699
+ * See {@link DEFAULT_WRITABLE_ROOTFS_PATHS} for the paths themselves and why
700
+ * each is there. Returns nothing when the read-only root filesystem is off, and
701
+ * the two ways a host names paths that cannot be mounted — a contradiction with
702
+ * `readOnlyRootfs: false`, or a path the layout already mounts — are refusals
703
+ * rather than a silently shorter list.
704
+ */
705
+ export function renderWritableRootfsArgs(config) {
706
+ assertRootfsOptionsAreCoherent(config);
707
+ if (config.readOnlyRootfs === false)
708
+ return [];
709
+ const mounted = new Set(mountedContainerPaths(config.layout));
710
+ const requested = config.writableRootfsPaths ?? [];
711
+ for (const path of requested) {
712
+ // Every segment non-empty and none of them `.` or `..`: an absolute path
713
+ // with at least one component. Anything else is refused because each
714
+ // rejected shape is a directory this file's exact-string checks could
715
+ // hold two spellings of — `/tmp/`, `//tmp` and `/tmp/.` are all `/tmp` to
716
+ // the kernel, so a default that mounted `/tmp` and a host entry that
717
+ // mounted `/tmp/` would each pass the duplicate-mount check and then be
718
+ // refused by docker at spawn, on the one path no test here can reach.
719
+ // `/` itself is refused as well, and for its own reason: it would make
720
+ // the whole read-only root filesystem writable again.
721
+ const segments = path.split('/');
722
+ const wellFormed = path.startsWith('/') &&
723
+ segments.length > 1 &&
724
+ segments.slice(1).every((segment) => segment !== '' && segment !== '.' && segment !== '..');
725
+ if (!wellFormed) {
726
+ throw new Error(`writableRootfsPaths entry ${JSON.stringify(path)} is not a normalised absolute path inside the container. Docker requires an absolute mount path with no empty, '.' or '..' segment and no trailing slash, and '/' would make the whole filesystem writable again rather than adding a scratch directory.`);
727
+ }
728
+ if (mounted.has(path)) {
729
+ throw new Error(`writableRootfsPaths names ${path}, which this layout already mounts. Docker refuses two mounts at one destination ("Duplicate mount point"), so which one won would be decided by argument order rather than by anyone's intent. Drop the entry, or change the layout's own mount to the mode you want.`);
730
+ }
731
+ }
732
+ // The set collapses a host entry that repeats a default, which would
733
+ // otherwise emit the same destination twice and be refused by docker.
734
+ const paths = [
735
+ ...new Set([
736
+ ...DEFAULT_WRITABLE_ROOTFS_PATHS.filter((path) => !mounted.has(path)),
737
+ ...requested,
738
+ ]),
739
+ ];
740
+ return paths.flatMap((path) => ['--tmpfs', `${path}:${TMPFS_MOUNT_OPTIONS}`]);
741
+ }
742
+ /**
743
+ * The confinement preamble for one container, in argv order.
744
+ *
745
+ * A function rather than a bare constant because `--read-only` is switchable
746
+ * and the flags that follow it describe what stays writable while it is on:
747
+ * `readOnlyRootfs: false` removes both the flag and the mounts. That is the only
748
+ * thing it removes. Everything in {@link HARDENING_ARGS} is applied
749
+ * unconditionally and no field can turn one of those off, so the argv for
750
+ * `readOnlyRootfs: false` is the argv this backend produced before any of this
751
+ * existed PLUS `--ipc private` — those two flags are the whole previous argv,
752
+ * and `--ipc private` is now unconditional. `--ipc` is not folded under this
753
+ * switch, because the field names the root filesystem: a host that turned the
754
+ * read-only rootfs off would be turning IPC isolation off as well, silently,
755
+ * for a reason the name of the field does not say. A switch has to mean one
756
+ * thing.
757
+ */
758
+ export function renderHardeningArgs(config) {
759
+ return [
760
+ ...HARDENING_ARGS,
761
+ ...(config.readOnlyRootfs === false ? [] : ['--read-only']),
762
+ ...renderWritableRootfsArgs(config),
763
+ ];
764
+ }
765
+ /**
766
+ * The complete `docker run` argv, as a value.
767
+ *
768
+ * Extracted for the same reason {@link resolveNetwork} and
769
+ * {@link egressProxyContainerConfig} were: everything downstream of it needs a running
770
+ * Docker daemon, so a confinement flag that never reached the argv — or one
771
+ * that reached it in an order that cancels another — could only be caught by an
772
+ * operator noticing its effect missing in production. Spawning a fake `docker`
773
+ * and reading back what it was handed proves what the fake was told and nothing
774
+ * about the container the daemon would build. Here the whole baseline is one
775
+ * array, and an edit that drops a flag fails a test rather than a deployment.
776
+ *
777
+ * Order matters in exactly two places, and both are asserted by the test that
778
+ * pins this: the image is the last argument, because everything after it is a
779
+ * command for the container rather than a flag for docker; and every flag that
780
+ * takes a value is pushed as two argv entries rather than one string, so no
781
+ * value is ever re-split by anything downstream.
782
+ */
783
+ export function buildDockerRunArgs(input) {
784
+ const { config, options, containerName, network, hostReachability, egressProxyPort } = input;
785
+ const layout = config.layout;
786
+ assertRootfsOptionsAreCoherent(config);
787
+ assertCpuLimitIsRenderable(config.cpuLimit);
788
+ const args = [
789
+ 'run',
790
+ '--detach',
791
+ '--rm',
792
+ '--name',
793
+ containerName,
794
+ '--network',
795
+ network,
796
+ ...renderHardeningArgs(config),
797
+ ];
798
+ if (config.runAsUser) {
799
+ args.push('--user', config.runAsUser);
800
+ }
801
+ args.push(...renderLabelArgs(config.labels));
802
+ args.push(...renderLayoutMountArgs(layout));
803
+ // Forward only the workspace root so the worker's lexical
804
+ // resolver agrees with the bind target. The full layout used
805
+ // to ride along as `NAMZU_SANDBOX_LAYOUT`, but the worker
806
+ // never branched on it; the manifest's only consumer was a
807
+ // log line. A skill loader that needs the manifest will
808
+ // write it to a bind path the worker reads at startup —
809
+ // avoids env-size limits, keeps the wire shape minimal.
810
+ if (egressProxyPort !== undefined) {
811
+ // No `--add-host`, and no `host-gateway`. The proxy is a container on
812
+ // this container's own internal network and `PROXY_HOST_ALIAS` is its
813
+ // network alias there (see {@link renderEgressProxyAttachArgs}), which
814
+ // docker's embedded DNS resolves with no alias file involved. The alias
815
+ // entry this replaced pointed at the host's loopback, where the proxy
816
+ // used to run — and a name resolving to a host that is not on this
817
+ // network is exactly the route this change removes.
818
+ const proxyUrl = `http://${PROXY_HOST_ALIAS}:${egressProxyPort}`;
819
+ // Both spellings: tooling is split between them, and a workload that
820
+ // reads only the one that is missing loses the redirection.
821
+ //
822
+ // **These are convenience, not the boundary, and that is the point of
823
+ // the arrangement.** A tool that honours them sends its requests to the
824
+ // proxy; a tool that ignores them — a Go or Rust binary that does not
825
+ // read proxy env, `curl --noproxy '*'`, a raw socket — has nowhere to
826
+ // send anything. This container is on an `--internal` network with no
827
+ // default route, and the only host on it is the proxy, so the
828
+ // uncooperative path fails with `Network unreachable` rather than
829
+ // bypassing the allowlist. What makes that true is
830
+ // {@link assertNetworkCarriesThePolicy} refusing a network that is not
831
+ // internal, not this block of environment.
832
+ for (const key of ['HTTP_PROXY', 'http_proxy', 'HTTPS_PROXY', 'https_proxy']) {
833
+ args.push('--env', `${key}=${proxyUrl}`);
834
+ }
835
+ // Loopback must not be proxied, or the worker cannot talk to
836
+ // itself.
837
+ args.push('--env', 'NO_PROXY=localhost,127.0.0.1');
838
+ args.push('--env', 'no_proxy=localhost,127.0.0.1');
839
+ }
840
+ // `outputs` is required by validation, so its containerPath is always
841
+ // available — the worker uses it as its workspace root.
842
+ args.push('--env', `NAMZU_SANDBOX_WORKSPACE=${layout.outputs.containerPath}`);
843
+ args.push('--env', `NAMZU_SANDBOX_READ_ROOTS=${renderLayoutReadRootsEnv(layout)}`);
844
+ args.push('--env', `NAMZU_SANDBOX_WRITE_ROOTS=${renderLayoutWriteRootsEnv(layout)}`);
845
+ // Only publish a host port when the consumer is going to reach
846
+ // the worker through the docker host's loopback (CLI / direct
847
+ // dev). For `container-network` reachability we leave the port
848
+ // unpublished — sibling containers reach the worker by its DNS
849
+ // name on the shared bridge, no host port required.
850
+ //
851
+ // Let Docker pick the host port instead of pre-reserving one
852
+ // in this process. The reservePort()-then-publish-fixed-port
853
+ // pattern had a TOCTOU window: the OS could hand the port to
854
+ // another process between our `server.close()` and Docker's
855
+ // `bind()`. Letting Docker pick (`--publish-all`) and reading
856
+ // the mapping back via `docker inspect` removes the race.
857
+ if (hostReachability === 'host-port') {
858
+ args.push('--publish', `127.0.0.1::${WORKER_PORT_INSIDE_CONTAINER}`);
859
+ }
860
+ if (config.runtime) {
861
+ args.push('--runtime', config.runtime);
862
+ }
863
+ // The three bounds the host can set, together and in one order, so a
864
+ // reader of a `docker inspect` sees them side by side. `--memory` and
865
+ // `--pids-limit` keep their existing treatment (a non-positive or absent
866
+ // value means "not set"); `--cpus` refuses a value that would mean
867
+ // something else, which is why it is the one with a check in front of it.
868
+ if (options.memoryLimitMb && options.memoryLimitMb > 0) {
869
+ args.push('--memory', `${options.memoryLimitMb}m`);
870
+ }
871
+ if (options.maxProcesses && options.maxProcesses > 0) {
872
+ args.push('--pids-limit', String(options.maxProcesses));
873
+ }
874
+ if (config.cpuLimit !== undefined) {
875
+ args.push('--cpus', String(config.cpuLimit));
876
+ }
877
+ for (const [key, value] of Object.entries(options.env ?? {})) {
878
+ args.push('--env', `${key}=${value}`);
879
+ }
880
+ // The worker's credential, rendered VALUELESS and last among the `--env`
881
+ // flags.
882
+ //
883
+ // Valueless, because `docker run --env NAME` reads the value out of the
884
+ // docker CLI's own environment — which `runOnce` is handed — and an argv
885
+ // is the wrong place for a secret: `ps` shows it to every user on the
886
+ // host for as long as the CLI lives, and a non-zero run puts the whole
887
+ // argv into the error this backend throws. The CLI environment is
888
+ // readable only by the same user and root, and the message is redacted
889
+ // besides.
890
+ //
891
+ // Last, because docker applies repeated `--env` flags in order and the
892
+ // last one wins: a host that separately sets `NAMZU_SANDBOX_TOKEN` in
893
+ // `options.env` — a copied example, an inherited environment — must not
894
+ // be able to displace the value its own client is sending, which would
895
+ // produce a container that rejects every call and reads as a broken
896
+ // worker rather than as a duplicated setting.
897
+ args.push('--env', 'NAMZU_SANDBOX_TOKEN');
898
+ args.push(config.image);
899
+ return args;
900
+ }
215
901
  async function spawnDockerSandbox(config, options, readiness) {
216
902
  options.signal?.throwIfAborted();
217
903
  const resolvedLayout = config.layout;
218
904
  const id = generateSandboxId();
219
905
  const docker = config.dockerBinary ?? DEFAULT_DOCKER_BINARY;
220
- // The boundary a host allowlist is actually enforced at. Started before
221
- // the container so its address can be handed in as proxy environment,
222
- // and torn down with the sandbox — a proxy holding real credentials
223
- // must not outlive the thing it was filtering for.
224
- let egressProxy;
225
- if (needsEgressProxy(options.egress) && options.egress) {
226
- const policy = options.egress;
227
- try {
228
- egressProxy = await new EgressProxy(egressProxyOptions(config, policy)).listen();
229
- options.signal?.throwIfAborted();
230
- }
231
- catch (error) {
232
- await egressProxy?.close().catch(() => undefined);
233
- throw error;
234
- }
235
- }
906
+ // The worker's per-instance credential, minted HERE — per `create()`, not
907
+ // per process and never per image. Three properties are the point, and
908
+ // each one rules out a cheaper shape:
909
+ //
910
+ // - Per instance. A token baked into the image is shared by every
911
+ // container ever built from it and readable by anything that can pull
912
+ // it, which is a worse artifact than a documented absence: it looks
913
+ // like a credential while separating nobody.
914
+ // - Not in an argv, and not in any message. It rides in the docker CLI
915
+ // child's environment, which `runOnce` is handed, and the CLI resolves
916
+ // it there for the valueless `--env NAMZU_SANDBOX_TOKEN`; a non-zero
917
+ // `docker run` renders its argv with every `--env` value redacted. So
918
+ // `ps` on the host does not show it and the error this backend throws
919
+ // does not carry it — neither of which was true when the value was
920
+ // rendered into the argv.
921
+ // - Where it IS visible, said plainly: the container's own config, so
922
+ // `docker inspect <name>` shows it for the container's life, to anyone
923
+ // who can already talk to the daemon — the same authority that can
924
+ // `docker exec` into the sandbox. And the worker's own `/proc` inside
925
+ // the container, to a workload that shares its uid. Both are why it is
926
+ // per-instance and dies with the container rather than being shared or
927
+ // long-lived.
928
+ // - Dead with the container. Nothing revokes it, because the only process
929
+ // that would accept it is removed with the sandbox, and the container's
930
+ // config that still holds it is removed with it.
931
+ //
932
+ // 32 bytes rather than a uuid: this is a secret, not an identifier, and
933
+ // base64url keeps it one argv-free environment value on every platform.
934
+ const workerToken = randomBytes(32).toString('base64url');
236
935
  const hostReachability = config.hostReachability ?? 'host-port';
237
- const network = resolveNetwork(config.network ?? 'none', options.egress, egressProxy !== undefined);
936
+ // Whether a proxy CONTAINER will run, which is what `resolveNetwork` turns
937
+ // on: an allowlist policy needs the image to run one as, and without it
938
+ // there is no boundary to hand traffic to. The refusal lives in
939
+ // `resolveNetwork` so that every caller of it gets the same answer, and it
940
+ // fires before anything is started.
941
+ const proxyPlanned = needsEgressProxy(options.egress) && config.egressProxyImage !== undefined;
942
+ const network = resolveNetwork(config.network ?? 'none', options.egress, proxyPlanned);
238
943
  // Whether this network can carry the reachability mode and the policy is
239
944
  // a fact about the network, so it is checked against the daemon rather
240
- // than inferred from its name. Before the container starts on purpose: a
945
+ // than inferred from its name. Before anything starts on purpose: a
241
946
  // refusal here is a wiring mistake and must not arrive dressed as a
242
947
  // container that failed to come up, which is exactly how it used to
243
948
  // arrive.
244
- try {
245
- assertNetworkCarriesThePolicy(network, hostReachability, options.egress, await inspectNetworkInternalFlag(docker, network, options.signal));
246
- }
247
- catch (err) {
248
- // The allowlist kinds start a proxy above, and this is outside the
249
- // try/catch that owns teardown — so without this the refusal would
250
- // leave a listening server on loopback stamping real credentials.
251
- await egressProxy?.close().catch(() => undefined);
252
- throw err;
253
- }
254
- const runtime = config.runtime;
949
+ assertNetworkCarriesThePolicy(network, hostReachability, options.egress, await inspectNetworkInternalFlag(docker, network, options.signal));
255
950
  const containerName = `namzu-sandbox-${id}`;
951
+ /** The proxy container's name, once it is being started. */
952
+ let egressProxyContainer;
256
953
  // All bind sources come from the consumer-supplied layout. The
257
954
  // backend never allocates host directories and never removes them
258
955
  // — that pre-existing single-mount mkdtemp path was the source of
@@ -266,20 +963,31 @@ async function spawnDockerSandbox(config, options, readiness) {
266
963
  // reconciliation; an external daemon that commits after this delete still
267
964
  // needs its ordinary label/name reaper.
268
965
  const removeContainer = runOnceQuiet(docker, ['rm', '-f', containerName], signal);
269
- // The proxy starts BEFORE the container and its only other close is
270
- // in `destroy()`, which a create that never returned can never
271
- // reach. So every failure between the two — a daemon that is down, a
272
- // port that could not be read, a worker that missed its readiness
273
- // deadline, a label the validator rejected — left a listening server
274
- // on loopback stamping real credential headers, plus a retained
275
- // event-loop handle, and a retry loop left one per attempt. That is
276
- // exactly the invariant this file states where the proxy is started:
277
- // it must not outlive the thing it was filtering for.
278
- // Start both teardown arms before awaiting either. A stuck runtime must
279
- // not prevent the proxy from releasing its credential-bearing listener.
280
- const closeProxy = egressProxy?.close().catch(() => undefined) ?? Promise.resolve();
281
- egressProxy = undefined;
282
- await Promise.all([removeContainer, closeProxy]);
966
+ // The proxy container starts BEFORE the sandbox and its only other
967
+ // teardown is in `destroy()`, which a create that never returned can
968
+ // never reach. So every failure between the two — a daemon that is
969
+ // down, a port that could not be read, a worker that missed its
970
+ // readiness deadline, an abort — would leave a container holding real
971
+ // credentials and a live route to the internet, and a retry loop left
972
+ // one per attempt. That is the invariant this file states where the
973
+ // proxy is started: it must not outlive the thing it was filtering
974
+ // for. Start both arms before awaiting either: a stuck runtime must
975
+ // not keep the credential-bearing one alive.
976
+ //
977
+ // The proxy arm is the reconciler rather than a `runOnceQuiet` on this
978
+ // signal, and the difference is the grace this function's caller runs
979
+ // under: `runFailureCleanup` spends ONE SECOND on both arms together,
980
+ // so a daemon that is slow to answer rather than down would have its
981
+ // `docker rm -f` killed at that boundary — the credential-bearing
982
+ // container outliving the sandbox by exactly the failure this exists to
983
+ // prevent. The reconciler carries its own deadline, is not reachable by
984
+ // any abort, and is not waited on by the caller's grace either: it
985
+ // finishes whether or not this function is still listening.
986
+ const removeProxy = egressProxyContainer === undefined
987
+ ? Promise.resolve()
988
+ : removeEgressProxyContainer(docker, egressProxyContainer);
989
+ egressProxyContainer = undefined;
990
+ await Promise.all([removeContainer, removeProxy]);
283
991
  }
284
992
  let hostPort;
285
993
  let baseUrl;
@@ -287,88 +995,51 @@ async function spawnDockerSandbox(config, options, readiness) {
287
995
  // always available — the worker uses it as its workspace root.
288
996
  const rootDir = resolvedLayout.outputs.containerPath;
289
997
  try {
290
- // Let Docker pick the host port instead of pre-reserving one
291
- // in this process. The reservePort()-then-publish-fixed-port
292
- // pattern had a TOCTOU window: the OS could hand the port to
293
- // another process between our `server.close()` and Docker's
294
- // `bind()`. Letting Docker pick (`--publish-all`) and reading
295
- // the mapping back via `docker inspect` removes the race.
296
- const args = [
297
- 'run',
298
- '--detach',
299
- '--rm',
300
- '--name',
998
+ // The boundary a host allowlist is actually enforced at, as a sibling
999
+ // container on this container's network (#398). Started before the
1000
+ // sandbox so its alias is in the network's DNS by the time the sandbox's
1001
+ // proxy environment resolves it, and removed with the sandbox — a proxy
1002
+ // holding real credentials must not outlive the thing it was filtering
1003
+ // for, and on this topology a leftover one is also a live route to the
1004
+ // internet for anything that can reach that network.
1005
+ //
1006
+ // Inside this `try` on purpose, which the first cut of this change got
1007
+ // wrong. This block has three suspension points — the resolver, the
1008
+ // proxy's `docker run`, and an explicit `throwIfAborted()` — and an
1009
+ // abort at any of them leaves a container holding real credentials with
1010
+ // nothing that will ever remove it: `destroy()` is unreachable, because
1011
+ // `create()` never returned a handle. Outside the `try` the abort
1012
+ // rethrew past `cleanupOnFailure`; inside it, every path that does not
1013
+ // return a `Sandbox` goes through that cleanup, which removes the proxy
1014
+ // by name on a deadline of its own. The abort that arrives here has to
1015
+ // be raised by a call inside the block — the sandbox's own `docker run`
1016
+ // does that itself — so an abort before this point reaches the same
1017
+ // place by the same route.
1018
+ if (proxyPlanned && options.egress) {
1019
+ egressProxyContainer = egressProxyContainerName(id);
1020
+ // Resolved once, here, rather than per request — see
1021
+ // `egressProxyContainerConfig` for what that costs and why it is paid.
1022
+ const allowedHosts = await resolveAllowedHosts(options.egress);
1023
+ options.signal?.throwIfAborted();
1024
+ await startEgressProxyContainer({
1025
+ docker,
1026
+ config,
1027
+ containerName: egressProxyContainer,
1028
+ internalNetwork: network,
1029
+ allowedHosts,
1030
+ signal: options.signal,
1031
+ });
1032
+ options.signal?.throwIfAborted();
1033
+ }
1034
+ const args = buildDockerRunArgs({
1035
+ config,
1036
+ options,
301
1037
  containerName,
302
- '--network',
303
1038
  network,
304
- ...HARDENING_ARGS,
305
- ...(config.runAsUser ? ['--user', config.runAsUser] : []),
306
- ];
307
- // `--label key=value` flags. Validate first — an empty key or
308
- // a key containing `=` would silently produce a malformed
309
- // label that downstream `docker ps --filter label=…` queries
310
- // could not match reliably. Throw before the spawn so misuse
311
- // surfaces during construction, not as a mysterious "container
312
- // has no labels" later.
313
- if (config.labels) {
314
- for (const [key, value] of Object.entries(config.labels)) {
315
- if (!key || key.includes('=')) {
316
- throw new Error(`docker label key ${JSON.stringify(key)} is invalid (empty or contains '=')`);
317
- }
318
- args.push('--label', `${key}=${value}`);
319
- }
320
- }
321
- args.push(...renderLayoutMountArgs(resolvedLayout));
322
- // Forward only the workspace root so the worker's lexical
323
- // resolver agrees with the bind target. The full layout used
324
- // to ride along as `NAMZU_SANDBOX_LAYOUT`, but the worker
325
- // never branched on it; the manifest's only consumer was a
326
- // log line. A skill loader that needs the manifest will
327
- // write it to a bind path the worker reads at startup —
328
- // avoids env-size limits, keeps the wire shape minimal.
329
- if (egressProxy) {
330
- // `host-gateway` is docker's own portable name for the host from
331
- // inside a container; hard-coding a bridge address would break on
332
- // every platform whose bridge is numbered differently. The proxy
333
- // itself binds loopback, so this alias is the only way in.
334
- args.push('--add-host', `${PROXY_HOST_ALIAS}:host-gateway`);
335
- const proxyUrl = `http://${PROXY_HOST_ALIAS}:${egressProxy.port}`;
336
- // Both spellings: tooling is split between them, and a workload
337
- // that reads only the one that is missing bypasses the boundary
338
- // entirely — which would look exactly like the policy working.
339
- for (const key of ['HTTP_PROXY', 'http_proxy', 'HTTPS_PROXY', 'https_proxy']) {
340
- args.push('--env', `${key}=${proxyUrl}`);
341
- }
342
- // Loopback must not be proxied, or the worker cannot talk to
343
- // itself.
344
- args.push('--env', 'NO_PROXY=localhost,127.0.0.1');
345
- args.push('--env', 'no_proxy=localhost,127.0.0.1');
346
- }
347
- args.push('--env', `NAMZU_SANDBOX_WORKSPACE=${rootDir}`);
348
- args.push('--env', `NAMZU_SANDBOX_READ_ROOTS=${renderLayoutReadRootsEnv(resolvedLayout)}`);
349
- args.push('--env', `NAMZU_SANDBOX_WRITE_ROOTS=${renderLayoutWriteRootsEnv(resolvedLayout)}`);
350
- // Only publish a host port when the consumer is going to reach
351
- // the worker through the docker host's loopback (CLI / direct
352
- // dev). For `container-network` reachability we leave the port
353
- // unpublished — sibling containers reach the worker by its DNS
354
- // name on the shared bridge, no host port required.
355
- if (hostReachability === 'host-port') {
356
- args.push('--publish', `127.0.0.1::${WORKER_PORT_INSIDE_CONTAINER}`);
357
- }
358
- if (runtime) {
359
- args.push('--runtime', runtime);
360
- }
361
- if (options.memoryLimitMb && options.memoryLimitMb > 0) {
362
- args.push('--memory', `${options.memoryLimitMb}m`);
363
- }
364
- if (options.maxProcesses && options.maxProcesses > 0) {
365
- args.push('--pids-limit', String(options.maxProcesses));
366
- }
367
- for (const [key, value] of Object.entries(options.env ?? {})) {
368
- args.push('--env', `${key}=${value}`);
369
- }
370
- args.push(config.image);
371
- await runOnce(docker, args, options.signal);
1039
+ hostReachability,
1040
+ ...(egressProxyContainer ? { egressProxyPort: EGRESS_PROXY_PORT_INSIDE_CONTAINER } : {}),
1041
+ });
1042
+ await runOnce(docker, args, options.signal, { NAMZU_SANDBOX_TOKEN: workerToken });
372
1043
  if (hostReachability === 'host-port') {
373
1044
  hostPort = await readMappedPort(docker, containerName, options.signal);
374
1045
  baseUrl = `http://127.0.0.1:${hostPort}`;
@@ -391,7 +1062,7 @@ async function spawnDockerSandbox(config, options, readiness) {
391
1062
  let retirementPromise;
392
1063
  let teardownPromise;
393
1064
  let teardownComplete = false;
394
- const workerClient = new HttpWorkerClient(baseUrl);
1065
+ const workerClient = new HttpWorkerClient(baseUrl, workerToken);
395
1066
  const assertActive = () => {
396
1067
  if (lifecycle !== 'active') {
397
1068
  throw new Error(`Sandbox ${id} is ${lifecycle}; no new worker operation can be admitted`);
@@ -412,11 +1083,18 @@ async function spawnDockerSandbox(config, options, readiness) {
412
1083
  teardownError = error;
413
1084
  }
414
1085
  finally {
415
- try {
416
- await egressProxy?.close();
417
- }
418
- catch (error) {
419
- teardownError ??= error;
1086
+ // The proxy goes with the sandbox, for the reason it is
1087
+ // started before it: a container holding real credentials must
1088
+ // not outlive the thing it was filtering for, and on this
1089
+ // topology leaving it running would also leave a live route to
1090
+ // the internet on a network the sandbox was confined to.
1091
+ if (egressProxyContainer !== undefined) {
1092
+ try {
1093
+ await runOnce(docker, ['rm', '-f', egressProxyContainer], signal);
1094
+ }
1095
+ catch (error) {
1096
+ teardownError ??= error;
1097
+ }
420
1098
  }
421
1099
  }
422
1100
  if (teardownError !== undefined)
@@ -495,10 +1173,77 @@ async function spawnDockerSandbox(config, options, readiness) {
495
1173
  // here and doing nothing would leave the caller believing the
496
1174
  // sandbox had been confined when it had not. Same rule the
497
1175
  // egress-kind refusal above follows.
498
- if (!egressProxy) {
499
- throw withHint(new Error('This sandbox cannot change its network policy: it was created without an egress proxy, so its network was fixed at creation and there is nothing to narrow. Refusing rather than accepting a policy that would not be applied.'), 'Construct the provider with an egress proxy to make the policy mutable, or create a second sandbox under the narrower policy.');
1176
+ if (egressProxyContainer === undefined) {
1177
+ throw withHint(new Error('This sandbox cannot change its network policy: it was created without an egress proxy, so its network was fixed at creation and there is nothing to narrow. Refusing rather than accepting a policy that would not be applied.'), 'Construct the provider with an egress proxy (and an egressProxyImage to run it as) to make the policy mutable, or create a second sandbox under the narrower policy.');
1178
+ }
1179
+ // The policy lives in the proxy container's environment, which is
1180
+ // written when the container starts and cannot be rewritten from
1181
+ // outside. So a live policy change REPLACES the container: the
1182
+ // allowlist the caller asked for is the allowlist the next
1183
+ // connection meets, and the connections open at that instant fail
1184
+ // closed. That is a real difference from the in-process proxy this
1185
+ // used to be — it swapped the allowlist in place, with no window in
1186
+ // which the proxy was absent — and it is the honest trade for a
1187
+ // boundary the sandbox cannot route around. See
1188
+ // `docs/sdk/sandbox-egress.md`.
1189
+ //
1190
+ // The name is read out of the closure once, here, rather than at
1191
+ // each use below: `cleanupOnFailure` clears it, and a swap that
1192
+ // raced a failing create would otherwise call `assertActive()` and
1193
+ // then name `undefined` in the removal.
1194
+ const proxyContainer = egressProxyContainer;
1195
+ try {
1196
+ await restartEgressProxyContainer({
1197
+ docker,
1198
+ config,
1199
+ containerName: proxyContainer,
1200
+ internalNetwork: network,
1201
+ allowedHosts: policy.allowedHosts,
1202
+ });
1203
+ // A teardown that landed while the replacement was starting
1204
+ // leaves a container nothing else will ever remove: `destroy()`
1205
+ // is running or has run, `egressProxyContainer` is only removed
1206
+ // from `teardownSandbox` and `cleanupOnFailure`, and the
1207
+ // `assertActive()` above ran BEFORE the first `await` — before
1208
+ // the swap suspended inside `restartEgressProxyContainer`, which
1209
+ // is exactly when a teardown lands. So the swap fails here
1210
+ // rather than reporting a policy change on a sandbox that no
1211
+ // longer exists.
1212
+ assertActive();
1213
+ }
1214
+ finally {
1215
+ // ... and the replacement is removed even so, because the check
1216
+ // above cannot be where the guarantee lives. It sits AFTER the
1217
+ // container comes into existence, and that ordering is what
1218
+ // makes it airtight rather than merely narrower than the check
1219
+ // at entry. A generation counter read before the `docker run`
1220
+ // has the opposite shape: it can only refuse a start it already
1221
+ // knows about, and a teardown that begins between that refusal
1222
+ // being evaluated and the daemon committing the container is in
1223
+ // no check's view — the container exists and nothing has looked
1224
+ // since. Here the two orderings partition the space instead. A
1225
+ // teardown that began before this point has already set
1226
+ // `lifecycle`, synchronously (`teardownSandbox` and `retire()`
1227
+ // both do, before their first `await`), so this removal runs. A
1228
+ // teardown that begins after it issues its own `rm -f` for this
1229
+ // same name — `teardownSandbox` reads `egressProxyContainer`,
1230
+ // which is this container — against a container that, at this
1231
+ // point, exists. Whichever of the two runs second finds the
1232
+ // container and removes it, and both are idempotent.
1233
+ //
1234
+ // A signal-less remover, and not `runOnce` on some signal, for
1235
+ // the reason `removeEgressProxyContainer` exists: the teardown
1236
+ // this is racing may be one whose own signal was already
1237
+ // aborted, and it removes nothing at all in that case (the whole
1238
+ // container set is left, which is what `Sandbox.destroy`
1239
+ // promises to settle promptly over). That is the caller's
1240
+ // contract and not something to defeat — but a proxy container
1241
+ // holding brokered credentials and a live route to the internet
1242
+ // is not something to leave with it either.
1243
+ if (lifecycle !== 'active') {
1244
+ await removeEgressProxyContainer(docker, proxyContainer);
1245
+ }
500
1246
  }
501
- egressProxy.setAllowedHosts(async () => policy.allowedHosts);
502
1247
  },
503
1248
  async writeFile(path, content) {
504
1249
  assertActive();
@@ -507,7 +1252,10 @@ async function spawnDockerSandbox(config, options, readiness) {
507
1252
  try {
508
1253
  res = await fetch(`${baseUrl}/write-file`, {
509
1254
  method: 'POST',
510
- headers: { 'content-type': 'application/json' },
1255
+ headers: {
1256
+ 'content-type': 'application/json',
1257
+ ...workerAuthorization(workerToken),
1258
+ },
511
1259
  body: JSON.stringify({
512
1260
  path,
513
1261
  content: buf.toString('base64'),
@@ -524,6 +1272,9 @@ async function spawnDockerSandbox(config, options, readiness) {
524
1272
  : 'unknown';
525
1273
  throw new Error(`namzu-sandbox /write-file fetch failed (baseUrl=${baseUrl}, path=${path}): ${err instanceof Error ? err.message : String(err)} — cause: ${causeMsg}`, { cause: err });
526
1274
  }
1275
+ if (res.status === 401) {
1276
+ throw withHint(new Error(`write-file failed: HTTP 401 ${await res.text()}`), WORKER_UNAUTHORIZED_HINT);
1277
+ }
527
1278
  if (!res.ok) {
528
1279
  throw new Error(`write-file failed: HTTP ${res.status} ${await res.text()}`);
529
1280
  }
@@ -549,10 +1300,16 @@ async function spawnDockerSandbox(config, options, readiness) {
549
1300
  }
550
1301
  const res = await fetch(`${baseUrl}/read-file`, {
551
1302
  method: 'POST',
552
- headers: { 'content-type': 'application/json' },
1303
+ headers: {
1304
+ 'content-type': 'application/json',
1305
+ ...workerAuthorization(workerToken),
1306
+ },
553
1307
  body: JSON.stringify({ path, encoding: 'base64' }),
554
1308
  signal: options?.signal,
555
1309
  });
1310
+ if (res.status === 401) {
1311
+ throw withHint(new Error(`read-file failed: HTTP 401 ${await res.text()}`), WORKER_UNAUTHORIZED_HINT);
1312
+ }
556
1313
  if (!res.ok) {
557
1314
  throw new Error(`read-file failed: HTTP ${res.status} ${await res.text()}`);
558
1315
  }
@@ -596,6 +1353,164 @@ async function spawnDockerSandbox(config, options, readiness) {
596
1353
  },
597
1354
  };
598
1355
  }
1356
+ /** The proxy's upstream network when the host named none. See the field. */
1357
+ const DEFAULT_EGRESS_PROXY_UPSTREAM_NETWORK = 'bridge';
1358
+ /**
1359
+ * How long a reconciliation remove of the proxy container may take.
1360
+ *
1361
+ * The same order as `retire()`'s own deadline, and for the same reason: this
1362
+ * bounds a call to a daemon that may never answer. It is a deadline of its own
1363
+ * rather than the caller's, because the caller is usually an aborted operation
1364
+ * — see {@link removeEgressProxyContainer}.
1365
+ */
1366
+ const EGRESS_PROXY_REMOVE_TIMEOUT_MS = 5_000;
1367
+ /**
1368
+ * Remove the proxy container, on a path that is already failing.
1369
+ *
1370
+ * Three properties, each of which the first cut of this change got wrong, and
1371
+ * each of which is why this is a function rather than a `runOnceQuiet` call at
1372
+ * three sites:
1373
+ *
1374
+ * - **It never takes the caller's signal.** A reconciliation remove handed an
1375
+ * already-aborted signal does not reach the daemon at all: `runOnce` and
1376
+ * `runOnceQuiet` install an abort listener that kills the child the moment
1377
+ * they see one, so `docker rm -f` dies before it is spawned and the
1378
+ * container survives — in exactly the case this exists to handle, an
1379
+ * operation that was aborted midway through starting it.
1380
+ * - **It is bounded anyway.** A fresh deadline of its own, so a daemon that
1381
+ * never answers cannot hang the failure path that is cleaning up after it.
1382
+ * - **It does not throw.** Whatever went wrong to bring the caller here is the
1383
+ * thing the caller needs to hear; a failure to remove is reported by the
1384
+ * container's own name in `docker ps` and by the label reaper this file
1385
+ * documents, not by replacing that error with a worse one.
1386
+ */
1387
+ async function removeEgressProxyContainer(docker, containerName) {
1388
+ const deadline = new OperationDeadline(EGRESS_PROXY_REMOVE_TIMEOUT_MS, `egress proxy container ${containerName} removal`);
1389
+ try {
1390
+ await deadline.run((signal) => runOnceQuiet(docker, ['rm', '-f', containerName], signal));
1391
+ }
1392
+ catch {
1393
+ // Deliberately swallowed. See the third property above.
1394
+ }
1395
+ }
1396
+ /**
1397
+ * Start the proxy container: run it, join it to the internal network, prove it
1398
+ * is up.
1399
+ *
1400
+ * Three steps and not one, in this order, for reasons that are all about the
1401
+ * topology rather than about docker's ergonomics. `docker run --network
1402
+ * <upstream>` gives the proxy its default route — the leg that reaches the
1403
+ * internet. `docker network connect --alias` adds the internal leg the sandbox
1404
+ * dials it on, and it has to be second: a container created on an internal
1405
+ * network first comes up with no default route and never gets one, which is a
1406
+ * proxy that can reach nothing. The readiness check is third because the first
1407
+ * two are `docker run` exiting 0, and `docker run --detach` exits 0 for a
1408
+ * container whose entrypoint is about to fail — an image without the compiled
1409
+ * module in it, or a config the entrypoint refused. Those failures arrive as a
1410
+ * sandbox whose every outbound request fails, which reads as the policy
1411
+ * working.
1412
+ *
1413
+ * A failure after the container exists removes it before rethrowing. Leaving
1414
+ * it would leave a container holding real credentials and a live route to the
1415
+ * internet with no sandbox it belongs to. This is not the only remover on that
1416
+ * path: the block that starts this container sits INSIDE the `try` that owns
1417
+ * `cleanupOnFailure`, and the name it is started under is in
1418
+ * `egressProxyContainer` from before the call, so the caller's cleanup removes
1419
+ * the same name again. The first cut of this change had the block outside that
1420
+ * `try` and this sentence said the caller's own cleanup could not see the
1421
+ * container; the two were wrong together, and the leak was real.
1422
+ */
1423
+ async function startEgressProxyContainer(input) {
1424
+ const { docker, config, containerName, internalNetwork, allowedHosts, signal } = input;
1425
+ const argvInput = {
1426
+ config,
1427
+ containerName,
1428
+ upstreamNetwork: config.egressProxyUpstreamNetwork ?? DEFAULT_EGRESS_PROXY_UPSTREAM_NETWORK,
1429
+ internalNetwork,
1430
+ };
1431
+ // The policy, on its way to the container's environment by way of the
1432
+ // `docker` CLI's own. See `renderEgressProxyRunArgs` for why it does not
1433
+ // travel in the argv.
1434
+ const proxyEnvironment = {
1435
+ [EGRESS_PROXY_CONFIG_ENV]: JSON.stringify(egressProxyContainerConfig(config, allowedHosts, EGRESS_PROXY_PORT_INSIDE_CONTAINER)),
1436
+ };
1437
+ try {
1438
+ await runOnce(docker, renderEgressProxyRunArgs(argvInput), signal, proxyEnvironment);
1439
+ }
1440
+ catch (error) {
1441
+ // The container may exist even though this call did not return
1442
+ // successfully: `docker run --detach` is a CLI process talking to a
1443
+ // daemon, and the daemon can commit the container while the client is
1444
+ // killed, times out, or fails to report the id back. Remove by name
1445
+ // before rethrowing, which is what the docblock above promised in the
1446
+ // first cut of this change and did not do.
1447
+ await removeEgressProxyContainer(docker, containerName);
1448
+ // An abort is not a failure to start, and the caller that aborted is
1449
+ // entitled to hear its own reason back — the same rule every other
1450
+ // catch in this file follows. Without this an acquisition timeout would
1451
+ // arrive dressed as an image-install problem, which is the diagnosis
1452
+ // shape this file refuses everywhere else.
1453
+ signal?.throwIfAborted();
1454
+ throw withHint(new Error(`Could not start the egress proxy container '${containerName}' from image '${config.egressProxyImage}': ${error instanceof Error ? error.message : String(error)}`), 'Build that image with `docker build -f packages/sandbox/egress-proxy/Dockerfile -t <tag> packages/sandbox` (after `pnpm --filter @namzu/sandbox build`), and name the tag in egressProxyImage. The sandbox is deliberately not started when the only way its traffic reaches the internet is missing.');
1455
+ }
1456
+ try {
1457
+ await runOnce(docker, renderEgressProxyAttachArgs(argvInput), signal);
1458
+ await assertEgressProxyContainerIsRunning(docker, containerName, signal);
1459
+ }
1460
+ catch (error) {
1461
+ await removeEgressProxyContainer(docker, containerName);
1462
+ throw error;
1463
+ }
1464
+ }
1465
+ /**
1466
+ * Replace the proxy container, for `setNetworkPolicy`.
1467
+ *
1468
+ * The policy the container enforces is in its environment, which is written
1469
+ * when it starts and is not writable from outside, so a live policy change is
1470
+ * a new container. The window between the two is one in which the sandbox has
1471
+ * no route out at all — the old proxy is gone and the new one is not up yet —
1472
+ * which fails CLOSED. That is the property worth naming: the alternative
1473
+ * ordering (start the second, then remove the first) has no such window, but
1474
+ * two containers cannot hold the same name or the same network alias, so it
1475
+ * would need a second name the sandbox's `HTTP_PROXY` does not know.
1476
+ */
1477
+ async function restartEgressProxyContainer(input) {
1478
+ await removeEgressProxyContainer(input.docker, input.containerName);
1479
+ try {
1480
+ await startEgressProxyContainer(input);
1481
+ }
1482
+ catch (error) {
1483
+ // The hint names both states rather than assuming the safe one. The
1484
+ // removal above swallows its own failure, so "the old container is
1485
+ // gone" is not something this function knows: if the daemon never
1486
+ // answered the remove, the previous container is still up and still
1487
+ // enforcing the policy the caller just replaced — which would be a
1488
+ // worse thing to misreport than a sandbox with no route out.
1489
+ throw withHint(error instanceof Error ? error : new Error(String(error)), "Check `docker ps` for this sandbox's proxy container before relying on the new policy: the replacement did not come up, so this sandbox either has no route out at all (every request fails closed) or still has the previous container enforcing the policy you just replaced. Retry, or destroy the sandbox and create it under the policy you want.");
1490
+ }
1491
+ }
1492
+ /**
1493
+ * Whether the daemon says the proxy container is still up.
1494
+ *
1495
+ * Read the same way {@link inspectNetworkInternalFlag} reads the network's
1496
+ * `Internal` flag, and for the same reason: an unreadable answer is not
1497
+ * evidence. A container that has already exited is gone (`--rm` removed it),
1498
+ * so the failure arrives as a failed `docker inspect` rather than as a
1499
+ * `false`, and both have to land on the same refusal.
1500
+ */
1501
+ async function assertEgressProxyContainerIsRunning(docker, containerName, signal) {
1502
+ let running = '';
1503
+ try {
1504
+ running = await runOnce(docker, ['inspect', '--format', '{{.State.Running}}', containerName], signal);
1505
+ }
1506
+ catch {
1507
+ signal?.throwIfAborted();
1508
+ running = '';
1509
+ }
1510
+ if (running.trim() === 'true')
1511
+ return;
1512
+ throw withHint(new Error(`The egress proxy container '${containerName}' is not running after being started, so this sandbox has no boundary to reach and no other route out.`), 'Almost always the image: it either lacks the compiled module (build the package before the image — `pnpm --filter @namzu/sandbox build`) or the entrypoint refused its configuration. `docker logs <container>` has the line the entrypoint wrote; the container is started with --rm, so an already-exited one is gone and its output with it.');
1513
+ }
599
1514
  /**
600
1515
  * Ask Docker which host port it bound to the worker port. Used
601
1516
  * instead of the pre-reserve-then-publish pattern (which had a
@@ -690,10 +1605,100 @@ async function waitForWorkerReady(baseUrl, timeoutMs, pollMs, signal) {
690
1605
  // which is what a reader needs.
691
1606
  throw withHint(new Error(`namzu-sandbox worker did not become ready within ${timeoutMs}ms: ${lastError instanceof Error ? lastError.message : String(lastError)}`), 'Check that the container runtime is running and that the sandbox worker image is built and reachable. A cold image pull can also exceed this window — raise the readiness timeout before assuming the worker is broken.');
692
1607
  }
693
- function runOnce(binary, args, signal) {
1608
+ /**
1609
+ * The argv as it may appear in an error message: the KEYS of every env
1610
+ * entry, with the values replaced.
1611
+ *
1612
+ * A rendered argv is the last place a secret should survive. A non-zero
1613
+ * `docker run` is a routine outcome — a missing image, a name conflict, a
1614
+ * daemon hiccup, ENOSPC — and its message goes wherever the sandbox
1615
+ * package's errors go: a log line, a telemetry batch, a CI transcript, a
1616
+ * pasted bug report. The env flags carry the worker's credential and every
1617
+ * value the host put in `options.env` (an API key, a broker token), and
1618
+ * none of them are needed to explain an exit code. The keys are kept
1619
+ * because they are what distinguishes "the image could not be pulled" from
1620
+ * "the environment was rejected".
1621
+ *
1622
+ * EVERY SPELLING docker accepts for that flag, not the one this backend
1623
+ * happens to emit today. `-e` IS `--env`, separated or `=`-attached, and
1624
+ * this function used to compare each element to the literal `'--env'`: the
1625
+ * long separated form this builder writes was redacted and `-e K=V`,
1626
+ * `--env=K=V` and `-e=K=V` were printed in full. A future caller writing
1627
+ * any of the three would have put a credential in a log line behind a
1628
+ * docblock that promised it would not. The covered forms are `--env K=V`,
1629
+ * `--env=K=V`, `-e K=V`, `-e=K=V` and the attached short form `-eK=V`; the
1630
+ * one shape it does not read is a value attached to an `-e` bundled into a
1631
+ * group of other short flags (`-iteK=V`), which no caller here writes and
1632
+ * which no rule short of matching `-e` anywhere inside an option could
1633
+ * catch. A valueless entry in any form (`--env K`) is passed through: it
1634
+ * resolves from the CLI's own environment and carries no value to redact.
1635
+ *
1636
+ * That redacts the workspace paths the layout is rendered from as well,
1637
+ * which are not secrets. They are also not what an exit code is about, and
1638
+ * a rule with exceptions is a rule that leaks the first time someone's
1639
+ * credential does not look like one.
1640
+ */
1641
+ export function redactDockerArgv(args) {
1642
+ const rendered = [...args];
1643
+ /** `K=V` → `K=<redacted>`, keeping the key; a valueless entry is left alone. */
1644
+ const redactEntry = (entry) => {
1645
+ const separator = entry.indexOf('=');
1646
+ return separator > 0 ? `${entry.slice(0, separator)}=<redacted>` : entry;
1647
+ };
1648
+ for (let index = 0; index < rendered.length; index += 1) {
1649
+ const arg = rendered[index];
1650
+ if (arg === '--env' || arg === '-e') {
1651
+ const entry = rendered[index + 1];
1652
+ if (entry !== undefined)
1653
+ rendered[index + 1] = redactEntry(entry);
1654
+ index += 1;
1655
+ continue;
1656
+ }
1657
+ // Attached, where the option's value is the rest of the same element:
1658
+ // the `=` forms, and the short form with no separator (`-eK=V`).
1659
+ const prefix = arg.startsWith('--env=')
1660
+ ? '--env='
1661
+ : arg.startsWith('-e=')
1662
+ ? '-e='
1663
+ : arg.startsWith('-e') && arg.length > 2
1664
+ ? '-e'
1665
+ : undefined;
1666
+ if (prefix !== undefined)
1667
+ rendered[index] = prefix + redactEntry(arg.slice(prefix.length));
1668
+ }
1669
+ return rendered;
1670
+ }
1671
+ /**
1672
+ * Run a docker subcommand to completion.
1673
+ *
1674
+ * `extraEnv` is added to the child's environment rather than the parent's, and
1675
+ * it is how a value reaches a container without entering the argv this file
1676
+ * builds. Two callers use it, for the same reason:
1677
+ *
1678
+ * - The sandbox's own `docker run`, which hands over the worker's
1679
+ * per-instance credential for the valueless `--env NAMZU_SANDBOX_TOKEN` its
1680
+ * argv carries (minted in `spawnDockerSandbox`).
1681
+ * - The egress proxy's `docker run`, which names its configuration variable
1682
+ * in the argv and hands the value over here. See
1683
+ * `renderEgressProxyRunArgs`.
1684
+ *
1685
+ * Anything a process is told through the environment is visible to `ps`'s
1686
+ * neighbour, `/proc/<pid>/environ`, only to the user that owns it — while an
1687
+ * argv is world-readable on Linux.
1688
+ */
1689
+ function runOnce(binary, args, signal, extraEnv) {
694
1690
  return new Promise((resolve, reject) => {
695
1691
  signal?.throwIfAborted();
696
- const child = spawn(binary, args, { stdio: ['ignore', 'pipe', 'pipe'] });
1692
+ const child = spawn(binary, args, {
1693
+ stdio: ['ignore', 'pipe', 'pipe'],
1694
+ // The one channel a value can ride in without entering the argv
1695
+ // this process builds: `ps` shows an argv to every user on the
1696
+ // host, and `/proc/<pid>/environ` is readable only by the same
1697
+ // user and root. `docker run --env NAME` (no `=`) reads the value
1698
+ // out of the CLI's own environment, which is why the credential
1699
+ // is passed this way and rendered valueless in the argv.
1700
+ ...(extraEnv ? { env: { ...process.env, ...extraEnv } } : {}),
1701
+ });
697
1702
  let stdout = '';
698
1703
  let stderr = '';
699
1704
  let settled = false;
@@ -725,7 +1730,7 @@ function runOnce(binary, args, signal) {
725
1730
  if (code === 0)
726
1731
  finish(undefined, stdout.trim());
727
1732
  else
728
- finish(new Error(`${binary} ${args.join(' ')} exited ${code}: ${stderr.trim()}`));
1733
+ finish(new Error(`${binary} ${redactDockerArgv(args).join(' ')} exited ${code}: ${stderr.trim()}`));
729
1734
  });
730
1735
  if (signal?.aborted)
731
1736
  abort();