@namzu/sandbox 16.0.0 → 17.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/CHANGELOG.md +238 -0
  2. package/README.md +164 -0
  3. package/dist/backends/aci-standby-pool/index.d.ts +22 -4
  4. package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
  5. package/dist/backends/aci-standby-pool/index.js +31 -7
  6. package/dist/backends/aci-standby-pool/index.js.map +1 -1
  7. package/dist/backends/docker/index.d.ts +243 -24
  8. package/dist/backends/docker/index.d.ts.map +1 -1
  9. package/dist/backends/docker/index.js +717 -108
  10. package/dist/backends/docker/index.js.map +1 -1
  11. package/dist/backends/firecracker/transport.d.ts +156 -1
  12. package/dist/backends/firecracker/transport.d.ts.map +1 -1
  13. package/dist/backends/firecracker/transport.js +223 -29
  14. package/dist/backends/firecracker/transport.js.map +1 -1
  15. package/dist/backends/http-worker-client.d.ts +64 -2
  16. package/dist/backends/http-worker-client.d.ts.map +1 -1
  17. package/dist/backends/http-worker-client.js +78 -7
  18. package/dist/backends/http-worker-client.js.map +1 -1
  19. package/dist/backends/kubernetes/transport.d.ts +7 -0
  20. package/dist/backends/kubernetes/transport.d.ts.map +1 -1
  21. package/dist/backends/kubernetes/transport.js.map +1 -1
  22. package/dist/egress/proxy.d.ts +47 -2
  23. package/dist/egress/proxy.d.ts.map +1 -1
  24. package/dist/egress/proxy.js +31 -7
  25. package/dist/egress/proxy.js.map +1 -1
  26. package/dist/index.d.ts +67 -1
  27. package/dist/index.d.ts.map +1 -1
  28. package/dist/index.js +22 -0
  29. package/dist/index.js.map +1 -1
  30. package/package.json +4 -4
  31. package/src/backends/aci-standby-pool/index.ts +37 -7
  32. package/src/backends/docker/index.ts +909 -126
  33. package/src/backends/firecracker/transport.ts +387 -36
  34. package/src/backends/http-worker-client.ts +89 -5
  35. package/src/backends/kubernetes/transport.ts +7 -0
  36. package/src/egress/proxy.ts +65 -8
  37. package/src/index.ts +89 -0
@@ -14,14 +14,35 @@
14
14
  * Trust model:
15
15
  * - Container is the trust boundary; everything inside is treated
16
16
  * as untrusted code.
17
- * - Worker only listens on loopback inside its own netns; the
18
- * host adapter reaches it via Docker's port-forward.
19
- * - Outbound network from the worker is restricted by host-side
20
- * firewall config (see {@link DockerBackendConfig.network}) plus
21
- * the egress proxy when one is configured (P3.2).
17
+ * - Every call to the worker's control API carries the per-instance
18
+ * `NAMZU_SANDBOX_TOKEN` this backend mints at create time and
19
+ * injects into the container's environment; a worker the host did
20
+ * not create must be provisioned with its own. The worker requires
21
+ * it on every route but `/healthz`, and refuses to start at all if
22
+ * it has none and is bound to anything routable.
23
+ * - Outbound network from the worker is restricted by the network
24
+ * it is attached to (see {@link DockerBackendConfig.network}) and,
25
+ * for a host allowlist, by an egress proxy running as a sibling
26
+ * container on that same network — the only way its traffic reaches
27
+ * the internet, because the network is `--internal` and has no route
28
+ * off it. The proxy is a container on that subnet like any other, so
29
+ * it is not the sandbox's only reachable destination. The proxy
30
+ * environment the sandbox is given directs traffic; the topology is
31
+ * what confines it. See {@link assertNetworkCarriesThePolicy} and
32
+ * #398.
33
+ *
34
+ * The credential above is what a previous version of this docblock
35
+ * claimed network placement alone provided. It said the worker "only
36
+ * listens on loopback inside its own netns", which is false — the worker
37
+ * binds every interface by default and has to, because a published
38
+ * container port forwards to the container's interface address rather
39
+ * than to its loopback, so a loopback-bound worker is unreachable through
40
+ * the port this backend publishes. The boundary was the network the
41
+ * container is attached to, and wanted a credential behind it.
22
42
  */
23
43
 
24
44
  import { spawn } from 'node:child_process'
45
+ import { randomBytes } from 'node:crypto'
25
46
 
26
47
  import {
27
48
  type ContainerSandboxLayout,
@@ -46,12 +67,7 @@ import {
46
67
  walkFilesViaExec,
47
68
  withHint,
48
69
  } from '@namzu/sdk'
49
- import { EgressProxy } from '../../egress/index.js'
50
- import type {
51
- BrokeredCredential,
52
- EgressProxyOptions,
53
- RunningEgressProxy,
54
- } from '../../egress/index.js'
70
+ import type { BrokeredCredential } from '../../egress/index.js'
55
71
 
56
72
  import {
57
73
  ContainerSandboxLayoutValidationError,
@@ -59,7 +75,11 @@ import {
59
75
  type SandboxBackend,
60
76
  type SandboxBackendOptions,
61
77
  } from '../../index.js'
62
- import { HttpWorkerClient } from '../http-worker-client.js'
78
+ import {
79
+ HttpWorkerClient,
80
+ WORKER_UNAUTHORIZED_HINT,
81
+ workerAuthorization,
82
+ } from '../http-worker-client.js'
63
83
  import {
64
84
  OperationDeadline,
65
85
  OperationDeadlineExpired,
@@ -189,6 +209,55 @@ export interface DockerBackendInternalConfig {
189
209
  */
190
210
  readonly allowInwardFor?: readonly string[]
191
211
 
212
+ /**
213
+ * Image the egress proxy runs as, when the policy needs one.
214
+ *
215
+ * A host allowlist (`static` or `resolver`) is enforced by a proxy, and
216
+ * this backend no longer runs that proxy in its own process: it runs it as
217
+ * a sibling container, dual-homed between the sandbox's `--internal`
218
+ * network (where the sandbox can reach it, and nothing else) and an
219
+ * ordinary bridge (where it reaches the internet). See
220
+ * {@link renderEgressProxyRunArgs} for the argv and
221
+ * `egress-proxy/Dockerfile` for the image to build.
222
+ *
223
+ * It is a second image rather than a reuse of `image` on purpose, and the
224
+ * reason is deployment-shaped rather than hygienic: the sandbox image is a
225
+ * host-supplied string that this backend cannot read, so there is no way to
226
+ * know whether the image named there has the proxy module in it, and the
227
+ * bind-mount alternative breaks exactly on the remote-daemon deployment
228
+ * (`hostReachability: 'container-network'`) where the SDK's own filesystem
229
+ * is not the daemon's.
230
+ *
231
+ * Unset, an allowlist policy is REFUSED at `create()` rather than
232
+ * downgraded — the same rule the rest of this file applies to a policy it
233
+ * cannot enforce. `deny-all` and `allow-all` need no image.
234
+ */
235
+ readonly egressProxyImage?: string
236
+
237
+ /**
238
+ * Network the egress proxy joins for its route to the internet. Default
239
+ * `'bridge'`, docker's own default bridge.
240
+ *
241
+ * This is the proxy's second leg, and it exists because the internal
242
+ * network the sandbox sits on has no route out by design. `'bridge'` is
243
+ * the default because it always exists and always has NAT, which keeps a
244
+ * host's first allowlist policy working without a second network to
245
+ * create.
246
+ *
247
+ * **The default is also the one wart of this topology, and a host that
248
+ * shares a docker daemon should know it.** The proxy listens on every
249
+ * interface inside its own container, so any other container attached to
250
+ * the same network reaches it too — and this proxy enforces its allowlist
251
+ * for whoever asks and stamps brokered credentials on the requests it
252
+ * forwards. On `'bridge'` that means every container on the host's default
253
+ * bridge. Name a dedicated network here (one only this deployment's
254
+ * containers join) to decide who that is. There is deliberately no switch
255
+ * that narrows the listener instead: the container has no way to know
256
+ * which of its own interfaces is which, and a listener bound to the wrong
257
+ * one is a sandbox with no route out at all.
258
+ */
259
+ readonly egressProxyUpstreamNetwork?: string
260
+
192
261
  readonly network?: 'none' | 'bridge' | string
193
262
  readonly readyPollIntervalMs?: number
194
263
  readonly readyTimeoutMs?: number
@@ -229,6 +298,12 @@ export interface DockerBackendInternalConfig {
229
298
  * monitoring filters) via `docker ps --filter label=…`. Keys
230
299
  * containing `=` or empty names throw at spawn time — the docker
231
300
  * CLI accepts them but the resulting label split is ambiguous.
301
+ *
302
+ * They are applied to the egress proxy's container too, when one runs.
303
+ * That is deliberate in both directions: a reaper that collects a
304
+ * sandbox's containers should collect the proxy with them, and a proxy
305
+ * left running by a sandbox that died is a container holding real
306
+ * credentials with a route to the internet.
232
307
  */
233
308
  readonly labels?: Readonly<Record<string, string>>
234
309
  }
@@ -285,6 +360,14 @@ export function buildDockerBackend(config: DockerBackendInternalConfig): Sandbox
285
360
  * just the way out. It now keeps the configured network, and
286
361
  * {@link assertNetworkCarriesThePolicy} is what makes that network a
287
362
  * boundary.
363
+ *
364
+ * Every policy that returns the configured network depends on that same
365
+ * assertion, which is why the two live side by side: this function says which
366
+ * network the container joins, and that one refuses the network if it cannot
367
+ * do what the policy asks of it (#398). An allowlist used to answer the
368
+ * configured network and get a proxy environment variable pointed at a
369
+ * host-side listener — enforcement by convention, which an uncooperative
370
+ * process simply ignores.
288
371
  */
289
372
  export function resolveNetwork(
290
373
  configured: string,
@@ -305,7 +388,7 @@ export function resolveNetwork(
305
388
  // reporting that it had been restricted.
306
389
  if (hasProxy) return configured
307
390
  throw new Error(
308
- `The docker sandbox backend cannot enforce an egress policy of kind '${egress.kind}' without an egress proxy: it has nothing to filter hosts through. Construct the provider with one, or use 'deny-all' / 'allow-all'. Refusing rather than silently granting full network access.`,
391
+ `The docker sandbox backend cannot enforce an egress policy of kind '${egress.kind}' without an egress proxy: it has nothing to filter hosts through. Name the image it should run the proxy as (egressProxyImage — see packages/sandbox/egress-proxy/Dockerfile), or use 'deny-all' / 'allow-all'. Refusing rather than silently granting full network access.`,
309
392
  )
310
393
  }
311
394
  }
@@ -347,9 +430,9 @@ export function isInternalNetwork(inspectedInternalFlag: string): boolean {
347
430
  /**
348
431
  * Refuse a container whose network cannot do what was asked of it.
349
432
  *
350
- * Two requirements meet on the same object here, and both were previously
351
- * unstated — which is how the backend came to ship a default configuration
352
- * that could not create a sandbox at all:
433
+ * Three requirements meet on the same object here, and the first two were
434
+ * previously unstated — which is how the backend came to ship a default
435
+ * configuration that could not create a sandbox at all:
353
436
  *
354
437
  * - **A published host port needs a route out.** Docker binds the port by
355
438
  * NAT to the container's address, so a container with no address gets no
@@ -365,12 +448,31 @@ export function isInternalNetwork(inspectedInternalFlag: string): boolean {
365
448
  * nothing. `deny-all` pointed at the default bridge would be full egress
366
449
  * under a policy object claiming none — the "accepted and silently
367
450
  * ignored" failure the rest of this file exists to refuse.
451
+ * - **A host allowlist needs one too, for the same reason and with one
452
+ * addition.** Until #398 the allowlist tier answered the configured
453
+ * network and pointed the sandbox at a proxy running on the host's
454
+ * loopback, and the only thing making traffic go through it was
455
+ * `HTTP_PROXY`. That is a request, not a boundary: anything inside the
456
+ * container that opens a socket directly reaches the network with the
457
+ * allowlist unconsulted, and untrusted code is the caller least likely to
458
+ * honour a convention. With the proxy as a sibling container on an
459
+ * `--internal` network, the only way the sandbox's traffic reaches the
460
+ * internet is that container, and it cannot change that because
461
+ * `--cap-drop=ALL` took `NET_ADMIN` away (see {@link HARDENING_ARGS}). The
462
+ * environment variables stay and now DIRECT traffic rather than permit it.
463
+ * On a network that is not internal there is no such route to remove, and
464
+ * the policy is back to being advisory — which is what this refuses.
368
465
  *
369
- * They are exact opposites, so `deny-all` over a published host port is
370
- * impossible rather than merely unsupported: no arrangement of docker
466
+ * The first two are exact opposites, so `deny-all` over a published host port
467
+ * is impossible rather than merely unsupported: no arrangement of docker
371
468
  * networking both denies all egress and lets the host reach the worker over
372
- * TCP. Closing that needs the control channel moved off TCP — see #398 —
373
- * and is not a flag this function could accept.
469
+ * TCP. Closing that needs the control channel moved off TCP — see #398 — and
470
+ * is not a flag this function could accept. The same opposition now reaches
471
+ * an allowlist policy, which is new: an allowlist on an internal network must
472
+ * be reached by container name, so a host that used a published port with a
473
+ * host allowlist has to move that consumer onto the internal network. That is
474
+ * a consequence of the boundary existing at all and not a gap in this
475
+ * function; the refusal below names the mode to move to.
374
476
  */
375
477
  export function assertNetworkCarriesThePolicy(
376
478
  network: string,
@@ -391,29 +493,210 @@ export function assertNetworkCarriesThePolicy(
391
493
  `The docker sandbox backend was asked for an egress policy of 'deny-all' on network '${network}', but that network is not internal, so the container can still reach the world. Create it with 'docker network create --internal ${network}' — an internal bridge denies egress in the kernel, rather than through an environment variable a workload may decline to read, while sibling containers still reach the worker by name. Refusing rather than reporting a boundary that is not there.`,
392
494
  )
393
495
  }
496
+
497
+ if (needsEgressProxy(egress) && !internal) {
498
+ throw new Error(
499
+ `The docker sandbox backend was asked for an egress policy of kind '${egress?.kind}' on network '${network}', but that network is not internal, so nothing stops the container from reaching the world directly and the allowlist would only be a proxy environment variable a workload may decline to read. Create the network with 'docker network create --internal ${network}': the sandbox then reaches the internet only through the egress proxy container, which is the boundary — and set hostReachability: 'container-network' to reach the worker by name on it, since a published host port needs a route out this network does not have. Refusing rather than reporting a boundary that is not there.`,
500
+ )
501
+ }
502
+ }
503
+
504
+ /**
505
+ * What the proxy container is told, as a value.
506
+ *
507
+ * The boundary's whole configuration. It is a value for the reason
508
+ * {@link resolveNetwork} is one: everything downstream of here needs a running
509
+ * Docker daemon, so a credential that failed to reach the boundary could only
510
+ * be caught by an operator noticing their requests arrive unauthenticated in
511
+ * production. A knob a host sets and the boundary never receives is the
512
+ * failure this shape exists to make testable.
513
+ *
514
+ * `parseProxyConfig` in `egress-proxy/server.mjs` refuses every shape in here
515
+ * it cannot read — and this function's own output is fed to that parser by
516
+ * `__tests__/hardening.test.ts`, so the two ends are pinned against each other
517
+ * rather than each against its own idea of the shape. A field renamed on one
518
+ * side alone fails a test rather than starting a boundary that enforces
519
+ * something nobody wrote.
520
+ */
521
+ export interface EgressProxyContainerConfig {
522
+ readonly port: number
523
+ readonly allowedHosts: readonly string[]
524
+ readonly credentials: readonly BrokeredCredential[]
525
+ readonly allowInwardFor?: readonly string[]
526
+ /**
527
+ * Names that denote the proxy itself, for its loop guard. The container's
528
+ * own hostname is added to this by the entrypoint; this field carries the
529
+ * network alias, which is the name the sandbox actually dials.
530
+ */
531
+ readonly selfNames?: readonly string[]
394
532
  }
395
533
 
396
534
  /**
397
- * The options the boundary is built from, as a value.
535
+ * Build the proxy container's configuration.
398
536
  *
399
- * Extracted for the same reason {@link resolveNetwork} is: everything
400
- * downstream of here needs a running Docker daemon, so a policy that never
401
- * reached the proxy could only be caught by an operator noticing their
402
- * traffic denied in production. A knob a host sets and the boundary never
403
- * receives is the failure this shape exists to make testable.
537
+ * `allowedHosts` is already resolved here rather than passed as a policy, and
538
+ * that is the one behavioural difference this change carries into the
539
+ * boundary: the in-process proxy called the host's resolver per request, and
540
+ * the container cannot. A `resolver` policy that rotates is honoured at
541
+ * `create()` and again at every `setNetworkPolicy()` — the container has no
542
+ * channel back to the host's resolver, and a channel the container CAN reach
543
+ * is one the sandbox can reach too, which would let the sandbox ask for its
544
+ * own allowlist to be widened. See `docs/sdk/sandbox-egress.md`.
404
545
  */
405
- export function egressProxyOptions(
546
+ export function egressProxyContainerConfig(
406
547
  config: Pick<DockerBackendInternalConfig, 'brokeredCredentials' | 'allowInwardFor'>,
407
- policy: EgressPolicy,
408
- ): EgressProxyOptions {
548
+ allowedHosts: readonly string[],
549
+ port: number,
550
+ ): EgressProxyContainerConfig {
409
551
  return {
410
- // Re-resolved per request rather than captured once, so a `resolver`
411
- // policy that rotates is honoured and `setNetworkPolicy` can swap it
412
- // on a live sandbox.
413
- allowedHosts: () => resolveAllowedHosts(policy),
552
+ port,
553
+ allowedHosts,
414
554
  credentials: config.brokeredCredentials ?? [],
415
555
  ...(config.allowInwardFor ? { allowInwardFor: config.allowInwardFor } : {}),
556
+ selfNames: [PROXY_HOST_ALIAS],
557
+ }
558
+ }
559
+
560
+ /**
561
+ * `--label key=value` flags, validated before they reach the daemon.
562
+ *
563
+ * An empty key or one containing `=` would silently produce a malformed label
564
+ * that a downstream `docker ps --filter label=…` could not match, so misuse
565
+ * surfaces during construction rather than as a container that mysteriously
566
+ * has no labels.
567
+ */
568
+ function renderLabelArgs(labels: Readonly<Record<string, string>> | undefined): string[] {
569
+ const args: string[] = []
570
+ if (!labels) return args
571
+ for (const [key, value] of Object.entries(labels)) {
572
+ if (!key || key.includes('=')) {
573
+ throw new Error(`docker label key ${JSON.stringify(key)} is invalid (empty or contains '=')`)
574
+ }
575
+ args.push('--label', `${key}=${value}`)
416
576
  }
577
+ return args
578
+ }
579
+
580
+ /**
581
+ * Name the proxy container reads its configuration from.
582
+ *
583
+ * The NAME travels in the argv and the VALUE travels in the `docker` CLI
584
+ * child's environment, which is why this is one constant used by both ends of
585
+ * that pair rather than a literal written twice.
586
+ */
587
+ const EGRESS_PROXY_CONFIG_ENV = 'NAMZU_EGRESS_PROXY_CONFIG'
588
+
589
+ /** Everything {@link renderEgressProxyRunArgs} renders, as a value. */
590
+ export interface EgressProxyArgvInput {
591
+ readonly config: DockerBackendInternalConfig
592
+ readonly containerName: string
593
+ /** Network the proxy reaches the internet through. See `egressProxyUpstreamNetwork`. */
594
+ readonly upstreamNetwork: string
595
+ /** The internal network the sandbox is on, which the proxy is also joined to. */
596
+ readonly internalNetwork: string
597
+ }
598
+
599
+ /**
600
+ * The `docker run` argv for the egress proxy container, as a value.
601
+ *
602
+ * Extracted for the reason every other argv in this file is: nothing
603
+ * downstream of it can run without a docker daemon, so a flag that never
604
+ * reached it — or a `--network` that named the wrong side of a dual-homed
605
+ * container — could only be caught by an operator in production, where the
606
+ * symptom is a sandbox that reaches nothing.
607
+ *
608
+ * The two networks are the whole topology and they are two calls:
609
+ * `docker run --network <upstream>` gives the container a default route, and
610
+ * {@link renderEgressProxyAttachArgs}'s `docker network connect` adds the
611
+ * internal network afterwards. The order is load-bearing. Attached to the
612
+ * internal network FIRST it would come up with no default route and no way to
613
+ * acquire one, and the proxy would be a boundary in front of nothing.
614
+ *
615
+ * The hardening baseline is the sandbox's, minus the parts that describe a
616
+ * filesystem. `--cap-drop=ALL`, `--no-new-privileges` and `--ipc private` are
617
+ * applied exactly as {@link HARDENING_ARGS} defines them, because this
618
+ * container sits between untrusted code and the internet and is the last one in
619
+ * the deployment that should be holding a capability. `--read-only` is the
620
+ * sandbox's too, with one tmpfs: the proxy writes nothing, and the tmpfs is
621
+ * there so that a Node process which one day wants a temp file fails at a
622
+ * filesystem boundary rather than at a mysterious `EROFS`.
623
+ *
624
+ * The image is last, so nothing after it is read as a flag — the same rule the
625
+ * sandbox argv follows, and here it also means the image's own `CMD` starts the
626
+ * proxy rather than this argv naming an entrypoint.
627
+ */
628
+ export function renderEgressProxyRunArgs(input: EgressProxyArgvInput): string[] {
629
+ const { config, containerName, upstreamNetwork, internalNetwork } = input
630
+ if (!config.egressProxyImage) {
631
+ throw new Error(
632
+ 'renderEgressProxyRunArgs was called without config.egressProxyImage; the docker backend cannot start an egress proxy container it has no image for. resolveNetwork refuses an allowlist policy in this state, so reaching here means the refusal was bypassed.',
633
+ )
634
+ }
635
+ if (upstreamNetwork === 'none') {
636
+ throw new Error(
637
+ `egressProxyUpstreamNetwork is 'none', which would give the proxy container no interface and no route: it would come up unable to reach anything, and the only way the sandbox's traffic reaches the internet would be a boundary that cannot reach it itself. Name a bridge, or drop the field to take the 'bridge' default.`,
638
+ )
639
+ }
640
+ if (upstreamNetwork === internalNetwork) {
641
+ throw new Error(
642
+ `egressProxyUpstreamNetwork names '${upstreamNetwork}', which is the same network the sandbox is on — the internal one. A container whose primary network is internal comes up with no default route and never acquires one, so the proxy would start with no way to reach the internet: the boundary would be a container that can only talk to the sandbox. Name a different network for egressProxyUpstreamNetwork, or drop the field to take the 'bridge' default.`,
643
+ )
644
+ }
645
+ return [
646
+ 'run',
647
+ '--detach',
648
+ '--rm',
649
+ '--name',
650
+ containerName,
651
+ // The name the sandbox dials, and the name the proxy's own loop guard
652
+ // has to recognise as itself. Set as the container's hostname so the
653
+ // entrypoint can read it back with `os.hostname()` instead of holding a
654
+ // second copy of the constant.
655
+ '--hostname',
656
+ PROXY_HOST_ALIAS,
657
+ '--network',
658
+ upstreamNetwork,
659
+ ...HARDENING_ARGS,
660
+ '--read-only',
661
+ '--tmpfs',
662
+ `/tmp:${TMPFS_MOUNT_OPTIONS}`,
663
+ ...renderLabelArgs(config.labels),
664
+ // The NAME only: docker takes the value from the environment of the
665
+ // `docker` CLI process it names, which the caller sets (see
666
+ // `startEgressProxyContainer`). The value is the whole policy —
667
+ // including brokered credential values — and an argv is a worse place
668
+ // for it than an environment on every platform that has a process
669
+ // table: `/proc/<pid>/cmdline` is world-readable on Linux, so putting
670
+ // it here published it to every local user on the docker host for as
671
+ // long as the client ran, where the child's own environment is readable
672
+ // only by the user that owns the process. It is still readable by
673
+ // anything with access to the daemon, through `docker inspect`, and
674
+ // `docs/sdk/sandbox-egress.md` says so.
675
+ '--env',
676
+ EGRESS_PROXY_CONFIG_ENV,
677
+ config.egressProxyImage,
678
+ ]
679
+ }
680
+
681
+ /**
682
+ * The `docker network connect` argv that makes the proxy dual-homed.
683
+ *
684
+ * `--alias` is what puts `namzu-egress` in the internal network's DNS, which
685
+ * is the name {@link buildDockerRunArgs} hands the sandbox in its proxy
686
+ * environment. The alias rather than the container name because the alias is
687
+ * the constant: a container named `namzu-egress-<sandbox-id>` would otherwise
688
+ * make the sandbox's `HTTP_PROXY` value depend on a generated id, and the two
689
+ * would have to be kept in step by hand.
690
+ */
691
+ export function renderEgressProxyAttachArgs(input: EgressProxyArgvInput): string[] {
692
+ return [
693
+ 'network',
694
+ 'connect',
695
+ '--alias',
696
+ PROXY_HOST_ALIAS,
697
+ input.internalNetwork,
698
+ input.containerName,
699
+ ]
417
700
  }
418
701
 
419
702
  /**
@@ -530,9 +813,35 @@ const HARDENING_ARGS: readonly string[] = [
530
813
  'private',
531
814
  ]
532
815
 
533
- /** Name the container reaches the host-side egress proxy by. */
816
+ /**
817
+ * Name the sandbox reaches the egress proxy by.
818
+ *
819
+ * This used to be a `--add-host namzu-egress:host-gateway` entry, because the
820
+ * proxy ran on the host's loopback and `host-gateway` is docker's portable
821
+ * name for the host from inside a container. It is now the proxy's own
822
+ * container name and network alias on the internal network, which needs no
823
+ * alias file at all: docker's embedded DNS resolves a container's aliases for
824
+ * every container on the same user-defined network. The constant survives the
825
+ * mechanism because what the sandbox is told has not changed — only what makes
826
+ * the name resolve, and whether anything else can be reached.
827
+ */
534
828
  const PROXY_HOST_ALIAS = 'namzu-egress'
535
829
 
830
+ /**
831
+ * Port the egress proxy listens on inside its own container.
832
+ *
833
+ * A fixed port rather than one the host reads back, because there is no host
834
+ * port involved: the sandbox dials the proxy container directly on the network
835
+ * they share. The old arrangement had to read a published port back out of
836
+ * `docker inspect`, with the race and the failure mode that came with it.
837
+ */
838
+ const EGRESS_PROXY_PORT_INSIDE_CONTAINER = 2025
839
+
840
+ /** Name of the sibling container the egress proxy runs in, for one sandbox. */
841
+ function egressProxyContainerName(sandboxId: string): string {
842
+ return `${PROXY_HOST_ALIAS}-${sandboxId}`
843
+ }
844
+
536
845
  /**
537
846
  * Mount options for every scratch mount this backend creates.
538
847
  *
@@ -838,9 +1147,12 @@ export interface DockerRunArgvInput {
838
1147
  readonly network: string
839
1148
  readonly hostReachability: 'host-port' | 'container-network'
840
1149
  /**
841
- * Port the host-side egress proxy listens on, when one is running. Absent
842
- * means no proxy, and no proxy environment is passed in — which is not the
843
- * same fact as a proxy that was configured and is unreachable.
1150
+ * Port the egress proxy listens on, when one is running. Absent means no
1151
+ * proxy, and no proxy environment is passed in — which is not the same
1152
+ * fact as a proxy that was configured and is unreachable.
1153
+ *
1154
+ * It is the proxy CONTAINER's port, on the internal network the two
1155
+ * containers share; nothing is published on the host any more.
844
1156
  */
845
1157
  readonly egressProxyPort?: number
846
1158
  }
@@ -849,7 +1161,7 @@ export interface DockerRunArgvInput {
849
1161
  * The complete `docker run` argv, as a value.
850
1162
  *
851
1163
  * Extracted for the same reason {@link resolveNetwork} and
852
- * {@link egressProxyOptions} were: everything downstream of it needs a running
1164
+ * {@link egressProxyContainerConfig} were: everything downstream of it needs a running
853
1165
  * Docker daemon, so a confinement flag that never reached the argv — or one
854
1166
  * that reached it in an order that cancels another — could only be caught by an
855
1167
  * operator noticing its effect missing in production. Spawning a fake `docker`
@@ -883,22 +1195,7 @@ export function buildDockerRunArgs(input: DockerRunArgvInput): string[] {
883
1195
  args.push('--user', config.runAsUser)
884
1196
  }
885
1197
 
886
- // `--label key=value` flags. Validate first — an empty key or
887
- // a key containing `=` would silently produce a malformed
888
- // label that downstream `docker ps --filter label=…` queries
889
- // could not match reliably. Throw before the spawn so misuse
890
- // surfaces during construction, not as a mysterious "container
891
- // has no labels" later.
892
- if (config.labels) {
893
- for (const [key, value] of Object.entries(config.labels)) {
894
- if (!key || key.includes('=')) {
895
- throw new Error(
896
- `docker label key ${JSON.stringify(key)} is invalid (empty or contains '=')`,
897
- )
898
- }
899
- args.push('--label', `${key}=${value}`)
900
- }
901
- }
1198
+ args.push(...renderLabelArgs(config.labels))
902
1199
 
903
1200
  args.push(...renderLayoutMountArgs(layout))
904
1201
  // Forward only the workspace root so the worker's lexical
@@ -909,15 +1206,27 @@ export function buildDockerRunArgs(input: DockerRunArgvInput): string[] {
909
1206
  // write it to a bind path the worker reads at startup —
910
1207
  // avoids env-size limits, keeps the wire shape minimal.
911
1208
  if (egressProxyPort !== undefined) {
912
- // `host-gateway` is docker's own portable name for the host from
913
- // inside a container; hard-coding a bridge address would break on
914
- // every platform whose bridge is numbered differently. The proxy
915
- // itself binds loopback, so this alias is the only way in.
916
- args.push('--add-host', `${PROXY_HOST_ALIAS}:host-gateway`)
1209
+ // No `--add-host`, and no `host-gateway`. The proxy is a container on
1210
+ // this container's own internal network and `PROXY_HOST_ALIAS` is its
1211
+ // network alias there (see {@link renderEgressProxyAttachArgs}), which
1212
+ // docker's embedded DNS resolves with no alias file involved. The alias
1213
+ // entry this replaced pointed at the host's loopback, where the proxy
1214
+ // used to run — and a name resolving to a host that is not on this
1215
+ // network is exactly the route this change removes.
917
1216
  const proxyUrl = `http://${PROXY_HOST_ALIAS}:${egressProxyPort}`
918
- // Both spellings: tooling is split between them, and a workload
919
- // that reads only the one that is missing bypasses the boundary
920
- // entirely — which would look exactly like the policy working.
1217
+ // Both spellings: tooling is split between them, and a workload that
1218
+ // reads only the one that is missing loses the redirection.
1219
+ //
1220
+ // **These are convenience, not the boundary, and that is the point of
1221
+ // the arrangement.** A tool that honours them sends its requests to the
1222
+ // proxy; a tool that ignores them — a Go or Rust binary that does not
1223
+ // read proxy env, `curl --noproxy '*'`, a raw socket — has nowhere to
1224
+ // send anything. This container is on an `--internal` network with no
1225
+ // default route, and the only host on it is the proxy, so the
1226
+ // uncooperative path fails with `Network unreachable` rather than
1227
+ // bypassing the allowlist. What makes that true is
1228
+ // {@link assertNetworkCarriesThePolicy} refusing a network that is not
1229
+ // internal, not this block of environment.
921
1230
  for (const key of ['HTTP_PROXY', 'http_proxy', 'HTTPS_PROXY', 'https_proxy']) {
922
1231
  args.push('--env', `${key}=${proxyUrl}`)
923
1232
  }
@@ -972,6 +1281,25 @@ export function buildDockerRunArgs(input: DockerRunArgvInput): string[] {
972
1281
  args.push('--env', `${key}=${value}`)
973
1282
  }
974
1283
 
1284
+ // The worker's credential, rendered VALUELESS and last among the `--env`
1285
+ // flags.
1286
+ //
1287
+ // Valueless, because `docker run --env NAME` reads the value out of the
1288
+ // docker CLI's own environment — which `runOnce` is handed — and an argv
1289
+ // is the wrong place for a secret: `ps` shows it to every user on the
1290
+ // host for as long as the CLI lives, and a non-zero run puts the whole
1291
+ // argv into the error this backend throws. The CLI environment is
1292
+ // readable only by the same user and root, and the message is redacted
1293
+ // besides.
1294
+ //
1295
+ // Last, because docker applies repeated `--env` flags in order and the
1296
+ // last one wins: a host that separately sets `NAMZU_SANDBOX_TOKEN` in
1297
+ // `options.env` — a copied example, an inherited environment — must not
1298
+ // be able to displace the value its own client is sending, which would
1299
+ // produce a container that rejects every call and reads as a broken
1300
+ // worker rather than as a duplicated setting.
1301
+ args.push('--env', 'NAMZU_SANDBOX_TOKEN')
1302
+
975
1303
  args.push(config.image)
976
1304
  return args
977
1305
  }
@@ -986,50 +1314,61 @@ async function spawnDockerSandbox(
986
1314
  const id = generateSandboxId()
987
1315
  const docker = config.dockerBinary ?? DEFAULT_DOCKER_BINARY
988
1316
 
989
- // The boundary a host allowlist is actually enforced at. Started before
990
- // the container so its address can be handed in as proxy environment,
991
- // and torn down with the sandbox — a proxy holding real credentials
992
- // must not outlive the thing it was filtering for.
993
- let egressProxy: RunningEgressProxy | undefined
994
- if (needsEgressProxy(options.egress) && options.egress) {
995
- const policy = options.egress
996
- try {
997
- egressProxy = await new EgressProxy(egressProxyOptions(config, policy)).listen()
998
- options.signal?.throwIfAborted()
999
- } catch (error) {
1000
- await egressProxy?.close().catch(() => undefined)
1001
- throw error
1002
- }
1003
- }
1317
+ // The worker's per-instance credential, minted HERE — per `create()`, not
1318
+ // per process and never per image. Three properties are the point, and
1319
+ // each one rules out a cheaper shape:
1320
+ //
1321
+ // - Per instance. A token baked into the image is shared by every
1322
+ // container ever built from it and readable by anything that can pull
1323
+ // it, which is a worse artifact than a documented absence: it looks
1324
+ // like a credential while separating nobody.
1325
+ // - Not in an argv, and not in any message. It rides in the docker CLI
1326
+ // child's environment, which `runOnce` is handed, and the CLI resolves
1327
+ // it there for the valueless `--env NAMZU_SANDBOX_TOKEN`; a non-zero
1328
+ // `docker run` renders its argv with every `--env` value redacted. So
1329
+ // `ps` on the host does not show it and the error this backend throws
1330
+ // does not carry it — neither of which was true when the value was
1331
+ // rendered into the argv.
1332
+ // - Where it IS visible, said plainly: the container's own config, so
1333
+ // `docker inspect <name>` shows it for the container's life, to anyone
1334
+ // who can already talk to the daemon — the same authority that can
1335
+ // `docker exec` into the sandbox. And the worker's own `/proc` inside
1336
+ // the container, to a workload that shares its uid. Both are why it is
1337
+ // per-instance and dies with the container rather than being shared or
1338
+ // long-lived.
1339
+ // - Dead with the container. Nothing revokes it, because the only process
1340
+ // that would accept it is removed with the sandbox, and the container's
1341
+ // config that still holds it is removed with it.
1342
+ //
1343
+ // 32 bytes rather than a uuid: this is a secret, not an identifier, and
1344
+ // base64url keeps it one argv-free environment value on every platform.
1345
+ const workerToken = randomBytes(32).toString('base64url')
1004
1346
 
1005
1347
  const hostReachability = config.hostReachability ?? 'host-port'
1006
- const network = resolveNetwork(
1007
- config.network ?? 'none',
1008
- options.egress,
1009
- egressProxy !== undefined,
1010
- )
1348
+ // Whether a proxy CONTAINER will run, which is what `resolveNetwork` turns
1349
+ // on: an allowlist policy needs the image to run one as, and without it
1350
+ // there is no boundary to hand traffic to. The refusal lives in
1351
+ // `resolveNetwork` so that every caller of it gets the same answer, and it
1352
+ // fires before anything is started.
1353
+ const proxyPlanned = needsEgressProxy(options.egress) && config.egressProxyImage !== undefined
1354
+ const network = resolveNetwork(config.network ?? 'none', options.egress, proxyPlanned)
1011
1355
  // Whether this network can carry the reachability mode and the policy is
1012
1356
  // a fact about the network, so it is checked against the daemon rather
1013
- // than inferred from its name. Before the container starts on purpose: a
1357
+ // than inferred from its name. Before anything starts on purpose: a
1014
1358
  // refusal here is a wiring mistake and must not arrive dressed as a
1015
1359
  // container that failed to come up, which is exactly how it used to
1016
1360
  // arrive.
1017
- try {
1018
- assertNetworkCarriesThePolicy(
1019
- network,
1020
- hostReachability,
1021
- options.egress,
1022
- await inspectNetworkInternalFlag(docker, network, options.signal),
1023
- )
1024
- } catch (err) {
1025
- // The allowlist kinds start a proxy above, and this is outside the
1026
- // try/catch that owns teardown — so without this the refusal would
1027
- // leave a listening server on loopback stamping real credentials.
1028
- await egressProxy?.close().catch(() => undefined)
1029
- throw err
1030
- }
1361
+ assertNetworkCarriesThePolicy(
1362
+ network,
1363
+ hostReachability,
1364
+ options.egress,
1365
+ await inspectNetworkInternalFlag(docker, network, options.signal),
1366
+ )
1031
1367
  const containerName = `namzu-sandbox-${id}`
1032
1368
 
1369
+ /** The proxy container's name, once it is being started. */
1370
+ let egressProxyContainer: string | undefined
1371
+
1033
1372
  // All bind sources come from the consumer-supplied layout. The
1034
1373
  // backend never allocates host directories and never removes them
1035
1374
  // — that pre-existing single-mount mkdtemp path was the source of
@@ -1043,20 +1382,32 @@ async function spawnDockerSandbox(
1043
1382
  // reconciliation; an external daemon that commits after this delete still
1044
1383
  // needs its ordinary label/name reaper.
1045
1384
  const removeContainer = runOnceQuiet(docker, ['rm', '-f', containerName], signal)
1046
- // The proxy starts BEFORE the container and its only other close is
1047
- // in `destroy()`, which a create that never returned can never
1048
- // reach. So every failure between the two — a daemon that is down, a
1049
- // port that could not be read, a worker that missed its readiness
1050
- // deadline, a label the validator rejected — left a listening server
1051
- // on loopback stamping real credential headers, plus a retained
1052
- // event-loop handle, and a retry loop left one per attempt. That is
1053
- // exactly the invariant this file states where the proxy is started:
1054
- // it must not outlive the thing it was filtering for.
1055
- // Start both teardown arms before awaiting either. A stuck runtime must
1056
- // not prevent the proxy from releasing its credential-bearing listener.
1057
- const closeProxy = egressProxy?.close().catch(() => undefined) ?? Promise.resolve()
1058
- egressProxy = undefined
1059
- await Promise.all([removeContainer, closeProxy])
1385
+ // The proxy container starts BEFORE the sandbox and its only other
1386
+ // teardown is in `destroy()`, which a create that never returned can
1387
+ // never reach. So every failure between the two — a daemon that is
1388
+ // down, a port that could not be read, a worker that missed its
1389
+ // readiness deadline, an abort — would leave a container holding real
1390
+ // credentials and a live route to the internet, and a retry loop left
1391
+ // one per attempt. That is the invariant this file states where the
1392
+ // proxy is started: it must not outlive the thing it was filtering
1393
+ // for. Start both arms before awaiting either: a stuck runtime must
1394
+ // not keep the credential-bearing one alive.
1395
+ //
1396
+ // The proxy arm is the reconciler rather than a `runOnceQuiet` on this
1397
+ // signal, and the difference is the grace this function's caller runs
1398
+ // under: `runFailureCleanup` spends ONE SECOND on both arms together,
1399
+ // so a daemon that is slow to answer rather than down would have its
1400
+ // `docker rm -f` killed at that boundary — the credential-bearing
1401
+ // container outliving the sandbox by exactly the failure this exists to
1402
+ // prevent. The reconciler carries its own deadline, is not reachable by
1403
+ // any abort, and is not waited on by the caller's grace either: it
1404
+ // finishes whether or not this function is still listening.
1405
+ const removeProxy =
1406
+ egressProxyContainer === undefined
1407
+ ? Promise.resolve()
1408
+ : removeEgressProxyContainer(docker, egressProxyContainer)
1409
+ egressProxyContainer = undefined
1410
+ await Promise.all([removeContainer, removeProxy])
1060
1411
  }
1061
1412
 
1062
1413
  let hostPort: number
@@ -1066,16 +1417,53 @@ async function spawnDockerSandbox(
1066
1417
  const rootDir = resolvedLayout.outputs.containerPath
1067
1418
 
1068
1419
  try {
1420
+ // The boundary a host allowlist is actually enforced at, as a sibling
1421
+ // container on this container's network (#398). Started before the
1422
+ // sandbox so its alias is in the network's DNS by the time the sandbox's
1423
+ // proxy environment resolves it, and removed with the sandbox — a proxy
1424
+ // holding real credentials must not outlive the thing it was filtering
1425
+ // for, and on this topology a leftover one is also a live route to the
1426
+ // internet for anything that can reach that network.
1427
+ //
1428
+ // Inside this `try` on purpose, which the first cut of this change got
1429
+ // wrong. This block has three suspension points — the resolver, the
1430
+ // proxy's `docker run`, and an explicit `throwIfAborted()` — and an
1431
+ // abort at any of them leaves a container holding real credentials with
1432
+ // nothing that will ever remove it: `destroy()` is unreachable, because
1433
+ // `create()` never returned a handle. Outside the `try` the abort
1434
+ // rethrew past `cleanupOnFailure`; inside it, every path that does not
1435
+ // return a `Sandbox` goes through that cleanup, which removes the proxy
1436
+ // by name on a deadline of its own. The abort that arrives here has to
1437
+ // be raised by a call inside the block — the sandbox's own `docker run`
1438
+ // does that itself — so an abort before this point reaches the same
1439
+ // place by the same route.
1440
+ if (proxyPlanned && options.egress) {
1441
+ egressProxyContainer = egressProxyContainerName(id)
1442
+ // Resolved once, here, rather than per request — see
1443
+ // `egressProxyContainerConfig` for what that costs and why it is paid.
1444
+ const allowedHosts = await resolveAllowedHosts(options.egress)
1445
+ options.signal?.throwIfAborted()
1446
+ await startEgressProxyContainer({
1447
+ docker,
1448
+ config,
1449
+ containerName: egressProxyContainer,
1450
+ internalNetwork: network,
1451
+ allowedHosts,
1452
+ signal: options.signal,
1453
+ })
1454
+ options.signal?.throwIfAborted()
1455
+ }
1456
+
1069
1457
  const args = buildDockerRunArgs({
1070
1458
  config,
1071
1459
  options,
1072
1460
  containerName,
1073
1461
  network,
1074
1462
  hostReachability,
1075
- ...(egressProxy ? { egressProxyPort: egressProxy.port } : {}),
1463
+ ...(egressProxyContainer ? { egressProxyPort: EGRESS_PROXY_PORT_INSIDE_CONTAINER } : {}),
1076
1464
  })
1077
1465
 
1078
- await runOnce(docker, args, options.signal)
1466
+ await runOnce(docker, args, options.signal, { NAMZU_SANDBOX_TOKEN: workerToken })
1079
1467
  if (hostReachability === 'host-port') {
1080
1468
  hostPort = await readMappedPort(docker, containerName, options.signal)
1081
1469
  baseUrl = `http://127.0.0.1:${hostPort}`
@@ -1108,7 +1496,7 @@ async function spawnDockerSandbox(
1108
1496
  let retirementPromise: Promise<{ readonly accepted: boolean; readonly error?: Error }> | undefined
1109
1497
  let teardownPromise: Promise<void> | undefined
1110
1498
  let teardownComplete = false
1111
- const workerClient = new HttpWorkerClient(baseUrl)
1499
+ const workerClient = new HttpWorkerClient(baseUrl, workerToken)
1112
1500
  const assertActive = (): void => {
1113
1501
  if (lifecycle !== 'active') {
1114
1502
  throw new Error(`Sandbox ${id} is ${lifecycle}; no new worker operation can be admitted`)
@@ -1125,10 +1513,17 @@ async function spawnDockerSandbox(
1125
1513
  } catch (error) {
1126
1514
  teardownError = error
1127
1515
  } finally {
1128
- try {
1129
- await egressProxy?.close()
1130
- } catch (error) {
1131
- teardownError ??= error
1516
+ // The proxy goes with the sandbox, for the reason it is
1517
+ // started before it: a container holding real credentials must
1518
+ // not outlive the thing it was filtering for, and on this
1519
+ // topology leaving it running would also leave a live route to
1520
+ // the internet on a network the sandbox was confined to.
1521
+ if (egressProxyContainer !== undefined) {
1522
+ try {
1523
+ await runOnce(docker, ['rm', '-f', egressProxyContainer], signal)
1524
+ } catch (error) {
1525
+ teardownError ??= error
1526
+ }
1132
1527
  }
1133
1528
  }
1134
1529
  if (teardownError !== undefined) throw teardownError
@@ -1212,15 +1607,81 @@ async function spawnDockerSandbox(
1212
1607
  // here and doing nothing would leave the caller believing the
1213
1608
  // sandbox had been confined when it had not. Same rule the
1214
1609
  // egress-kind refusal above follows.
1215
- if (!egressProxy) {
1610
+ if (egressProxyContainer === undefined) {
1216
1611
  throw withHint(
1217
1612
  new Error(
1218
1613
  'This sandbox cannot change its network policy: it was created without an egress proxy, so its network was fixed at creation and there is nothing to narrow. Refusing rather than accepting a policy that would not be applied.',
1219
1614
  ),
1220
- 'Construct the provider with an egress proxy to make the policy mutable, or create a second sandbox under the narrower policy.',
1615
+ 'Construct the provider with an egress proxy (and an egressProxyImage to run it as) to make the policy mutable, or create a second sandbox under the narrower policy.',
1221
1616
  )
1222
1617
  }
1223
- egressProxy.setAllowedHosts(async () => policy.allowedHosts)
1618
+ // The policy lives in the proxy container's environment, which is
1619
+ // written when the container starts and cannot be rewritten from
1620
+ // outside. So a live policy change REPLACES the container: the
1621
+ // allowlist the caller asked for is the allowlist the next
1622
+ // connection meets, and the connections open at that instant fail
1623
+ // closed. That is a real difference from the in-process proxy this
1624
+ // used to be — it swapped the allowlist in place, with no window in
1625
+ // which the proxy was absent — and it is the honest trade for a
1626
+ // boundary the sandbox cannot route around. See
1627
+ // `docs/sdk/sandbox-egress.md`.
1628
+ //
1629
+ // The name is read out of the closure once, here, rather than at
1630
+ // each use below: `cleanupOnFailure` clears it, and a swap that
1631
+ // raced a failing create would otherwise call `assertActive()` and
1632
+ // then name `undefined` in the removal.
1633
+ const proxyContainer = egressProxyContainer
1634
+ try {
1635
+ await restartEgressProxyContainer({
1636
+ docker,
1637
+ config,
1638
+ containerName: proxyContainer,
1639
+ internalNetwork: network,
1640
+ allowedHosts: policy.allowedHosts,
1641
+ })
1642
+ // A teardown that landed while the replacement was starting
1643
+ // leaves a container nothing else will ever remove: `destroy()`
1644
+ // is running or has run, `egressProxyContainer` is only removed
1645
+ // from `teardownSandbox` and `cleanupOnFailure`, and the
1646
+ // `assertActive()` above ran BEFORE the first `await` — before
1647
+ // the swap suspended inside `restartEgressProxyContainer`, which
1648
+ // is exactly when a teardown lands. So the swap fails here
1649
+ // rather than reporting a policy change on a sandbox that no
1650
+ // longer exists.
1651
+ assertActive()
1652
+ } finally {
1653
+ // ... and the replacement is removed even so, because the check
1654
+ // above cannot be where the guarantee lives. It sits AFTER the
1655
+ // container comes into existence, and that ordering is what
1656
+ // makes it airtight rather than merely narrower than the check
1657
+ // at entry. A generation counter read before the `docker run`
1658
+ // has the opposite shape: it can only refuse a start it already
1659
+ // knows about, and a teardown that begins between that refusal
1660
+ // being evaluated and the daemon committing the container is in
1661
+ // no check's view — the container exists and nothing has looked
1662
+ // since. Here the two orderings partition the space instead. A
1663
+ // teardown that began before this point has already set
1664
+ // `lifecycle`, synchronously (`teardownSandbox` and `retire()`
1665
+ // both do, before their first `await`), so this removal runs. A
1666
+ // teardown that begins after it issues its own `rm -f` for this
1667
+ // same name — `teardownSandbox` reads `egressProxyContainer`,
1668
+ // which is this container — against a container that, at this
1669
+ // point, exists. Whichever of the two runs second finds the
1670
+ // container and removes it, and both are idempotent.
1671
+ //
1672
+ // A signal-less remover, and not `runOnce` on some signal, for
1673
+ // the reason `removeEgressProxyContainer` exists: the teardown
1674
+ // this is racing may be one whose own signal was already
1675
+ // aborted, and it removes nothing at all in that case (the whole
1676
+ // container set is left, which is what `Sandbox.destroy`
1677
+ // promises to settle promptly over). That is the caller's
1678
+ // contract and not something to defeat — but a proxy container
1679
+ // holding brokered credentials and a live route to the internet
1680
+ // is not something to leave with it either.
1681
+ if (lifecycle !== 'active') {
1682
+ await removeEgressProxyContainer(docker, proxyContainer)
1683
+ }
1684
+ }
1224
1685
  },
1225
1686
 
1226
1687
  async writeFile(path: string, content: string | Buffer): Promise<void> {
@@ -1230,7 +1691,10 @@ async function spawnDockerSandbox(
1230
1691
  try {
1231
1692
  res = await fetch(`${baseUrl}/write-file`, {
1232
1693
  method: 'POST',
1233
- headers: { 'content-type': 'application/json' },
1694
+ headers: {
1695
+ 'content-type': 'application/json',
1696
+ ...workerAuthorization(workerToken),
1697
+ },
1234
1698
  body: JSON.stringify({
1235
1699
  path,
1236
1700
  content: buf.toString('base64'),
@@ -1250,6 +1714,12 @@ async function spawnDockerSandbox(
1250
1714
  { cause: err },
1251
1715
  )
1252
1716
  }
1717
+ if (res.status === 401) {
1718
+ throw withHint(
1719
+ new Error(`write-file failed: HTTP 401 ${await res.text()}`),
1720
+ WORKER_UNAUTHORIZED_HINT,
1721
+ )
1722
+ }
1253
1723
  if (!res.ok) {
1254
1724
  throw new Error(`write-file failed: HTTP ${res.status} ${await res.text()}`)
1255
1725
  }
@@ -1278,10 +1748,19 @@ async function spawnDockerSandbox(
1278
1748
  }
1279
1749
  const res = await fetch(`${baseUrl}/read-file`, {
1280
1750
  method: 'POST',
1281
- headers: { 'content-type': 'application/json' },
1751
+ headers: {
1752
+ 'content-type': 'application/json',
1753
+ ...workerAuthorization(workerToken),
1754
+ },
1282
1755
  body: JSON.stringify({ path, encoding: 'base64' }),
1283
1756
  signal: options?.signal,
1284
1757
  })
1758
+ if (res.status === 401) {
1759
+ throw withHint(
1760
+ new Error(`read-file failed: HTTP 401 ${await res.text()}`),
1761
+ WORKER_UNAUTHORIZED_HINT,
1762
+ )
1763
+ }
1285
1764
  if (!res.ok) {
1286
1765
  throw new Error(`read-file failed: HTTP ${res.status} ${await res.text()}`)
1287
1766
  }
@@ -1336,6 +1815,211 @@ async function spawnDockerSandbox(
1336
1815
  }
1337
1816
  }
1338
1817
 
1818
+ /** The proxy's upstream network when the host named none. See the field. */
1819
+ const DEFAULT_EGRESS_PROXY_UPSTREAM_NETWORK = 'bridge'
1820
+
1821
+ /**
1822
+ * How long a reconciliation remove of the proxy container may take.
1823
+ *
1824
+ * The same order as `retire()`'s own deadline, and for the same reason: this
1825
+ * bounds a call to a daemon that may never answer. It is a deadline of its own
1826
+ * rather than the caller's, because the caller is usually an aborted operation
1827
+ * — see {@link removeEgressProxyContainer}.
1828
+ */
1829
+ const EGRESS_PROXY_REMOVE_TIMEOUT_MS = 5_000
1830
+
1831
+ /**
1832
+ * Remove the proxy container, on a path that is already failing.
1833
+ *
1834
+ * Three properties, each of which the first cut of this change got wrong, and
1835
+ * each of which is why this is a function rather than a `runOnceQuiet` call at
1836
+ * three sites:
1837
+ *
1838
+ * - **It never takes the caller's signal.** A reconciliation remove handed an
1839
+ * already-aborted signal does not reach the daemon at all: `runOnce` and
1840
+ * `runOnceQuiet` install an abort listener that kills the child the moment
1841
+ * they see one, so `docker rm -f` dies before it is spawned and the
1842
+ * container survives — in exactly the case this exists to handle, an
1843
+ * operation that was aborted midway through starting it.
1844
+ * - **It is bounded anyway.** A fresh deadline of its own, so a daemon that
1845
+ * never answers cannot hang the failure path that is cleaning up after it.
1846
+ * - **It does not throw.** Whatever went wrong to bring the caller here is the
1847
+ * thing the caller needs to hear; a failure to remove is reported by the
1848
+ * container's own name in `docker ps` and by the label reaper this file
1849
+ * documents, not by replacing that error with a worse one.
1850
+ */
1851
+ async function removeEgressProxyContainer(docker: string, containerName: string): Promise<void> {
1852
+ const deadline = new OperationDeadline(
1853
+ EGRESS_PROXY_REMOVE_TIMEOUT_MS,
1854
+ `egress proxy container ${containerName} removal`,
1855
+ )
1856
+ try {
1857
+ await deadline.run((signal) => runOnceQuiet(docker, ['rm', '-f', containerName], signal))
1858
+ } catch {
1859
+ // Deliberately swallowed. See the third property above.
1860
+ }
1861
+ }
1862
+
1863
+ interface EgressProxyContainerInput {
1864
+ readonly docker: string
1865
+ readonly config: DockerBackendInternalConfig
1866
+ readonly containerName: string
1867
+ readonly internalNetwork: string
1868
+ /**
1869
+ * The hosts the proxy will permit, already resolved. Resolved by the
1870
+ * caller rather than passed as a policy because the two callers resolve
1871
+ * differently and the difference is the point: `create()` resolves a
1872
+ * policy once through {@link resolveAllowedHosts}, and `setNetworkPolicy`
1873
+ * is handed a list by definition (`SandboxNetworkPolicy`).
1874
+ */
1875
+ readonly allowedHosts: readonly string[]
1876
+ readonly signal?: AbortSignal
1877
+ }
1878
+
1879
+ /**
1880
+ * Start the proxy container: run it, join it to the internal network, prove it
1881
+ * is up.
1882
+ *
1883
+ * Three steps and not one, in this order, for reasons that are all about the
1884
+ * topology rather than about docker's ergonomics. `docker run --network
1885
+ * <upstream>` gives the proxy its default route — the leg that reaches the
1886
+ * internet. `docker network connect --alias` adds the internal leg the sandbox
1887
+ * dials it on, and it has to be second: a container created on an internal
1888
+ * network first comes up with no default route and never gets one, which is a
1889
+ * proxy that can reach nothing. The readiness check is third because the first
1890
+ * two are `docker run` exiting 0, and `docker run --detach` exits 0 for a
1891
+ * container whose entrypoint is about to fail — an image without the compiled
1892
+ * module in it, or a config the entrypoint refused. Those failures arrive as a
1893
+ * sandbox whose every outbound request fails, which reads as the policy
1894
+ * working.
1895
+ *
1896
+ * A failure after the container exists removes it before rethrowing. Leaving
1897
+ * it would leave a container holding real credentials and a live route to the
1898
+ * internet with no sandbox it belongs to. This is not the only remover on that
1899
+ * path: the block that starts this container sits INSIDE the `try` that owns
1900
+ * `cleanupOnFailure`, and the name it is started under is in
1901
+ * `egressProxyContainer` from before the call, so the caller's cleanup removes
1902
+ * the same name again. The first cut of this change had the block outside that
1903
+ * `try` and this sentence said the caller's own cleanup could not see the
1904
+ * container; the two were wrong together, and the leak was real.
1905
+ */
1906
+ async function startEgressProxyContainer(input: EgressProxyContainerInput): Promise<void> {
1907
+ const { docker, config, containerName, internalNetwork, allowedHosts, signal } = input
1908
+
1909
+ const argvInput: EgressProxyArgvInput = {
1910
+ config,
1911
+ containerName,
1912
+ upstreamNetwork: config.egressProxyUpstreamNetwork ?? DEFAULT_EGRESS_PROXY_UPSTREAM_NETWORK,
1913
+ internalNetwork,
1914
+ }
1915
+ // The policy, on its way to the container's environment by way of the
1916
+ // `docker` CLI's own. See `renderEgressProxyRunArgs` for why it does not
1917
+ // travel in the argv.
1918
+ const proxyEnvironment = {
1919
+ [EGRESS_PROXY_CONFIG_ENV]: JSON.stringify(
1920
+ egressProxyContainerConfig(config, allowedHosts, EGRESS_PROXY_PORT_INSIDE_CONTAINER),
1921
+ ),
1922
+ }
1923
+
1924
+ try {
1925
+ await runOnce(docker, renderEgressProxyRunArgs(argvInput), signal, proxyEnvironment)
1926
+ } catch (error) {
1927
+ // The container may exist even though this call did not return
1928
+ // successfully: `docker run --detach` is a CLI process talking to a
1929
+ // daemon, and the daemon can commit the container while the client is
1930
+ // killed, times out, or fails to report the id back. Remove by name
1931
+ // before rethrowing, which is what the docblock above promised in the
1932
+ // first cut of this change and did not do.
1933
+ await removeEgressProxyContainer(docker, containerName)
1934
+ // An abort is not a failure to start, and the caller that aborted is
1935
+ // entitled to hear its own reason back — the same rule every other
1936
+ // catch in this file follows. Without this an acquisition timeout would
1937
+ // arrive dressed as an image-install problem, which is the diagnosis
1938
+ // shape this file refuses everywhere else.
1939
+ signal?.throwIfAborted()
1940
+ throw withHint(
1941
+ new Error(
1942
+ `Could not start the egress proxy container '${containerName}' from image '${config.egressProxyImage}': ${error instanceof Error ? error.message : String(error)}`,
1943
+ ),
1944
+ 'Build that image with `docker build -f packages/sandbox/egress-proxy/Dockerfile -t <tag> packages/sandbox` (after `pnpm --filter @namzu/sandbox build`), and name the tag in egressProxyImage. The sandbox is deliberately not started when the only way its traffic reaches the internet is missing.',
1945
+ )
1946
+ }
1947
+
1948
+ try {
1949
+ await runOnce(docker, renderEgressProxyAttachArgs(argvInput), signal)
1950
+ await assertEgressProxyContainerIsRunning(docker, containerName, signal)
1951
+ } catch (error) {
1952
+ await removeEgressProxyContainer(docker, containerName)
1953
+ throw error
1954
+ }
1955
+ }
1956
+
1957
+ /**
1958
+ * Replace the proxy container, for `setNetworkPolicy`.
1959
+ *
1960
+ * The policy the container enforces is in its environment, which is written
1961
+ * when it starts and is not writable from outside, so a live policy change is
1962
+ * a new container. The window between the two is one in which the sandbox has
1963
+ * no route out at all — the old proxy is gone and the new one is not up yet —
1964
+ * which fails CLOSED. That is the property worth naming: the alternative
1965
+ * ordering (start the second, then remove the first) has no such window, but
1966
+ * two containers cannot hold the same name or the same network alias, so it
1967
+ * would need a second name the sandbox's `HTTP_PROXY` does not know.
1968
+ */
1969
+ async function restartEgressProxyContainer(
1970
+ input: Omit<EgressProxyContainerInput, 'signal'>,
1971
+ ): Promise<void> {
1972
+ await removeEgressProxyContainer(input.docker, input.containerName)
1973
+ try {
1974
+ await startEgressProxyContainer(input)
1975
+ } catch (error) {
1976
+ // The hint names both states rather than assuming the safe one. The
1977
+ // removal above swallows its own failure, so "the old container is
1978
+ // gone" is not something this function knows: if the daemon never
1979
+ // answered the remove, the previous container is still up and still
1980
+ // enforcing the policy the caller just replaced — which would be a
1981
+ // worse thing to misreport than a sandbox with no route out.
1982
+ throw withHint(
1983
+ error instanceof Error ? error : new Error(String(error)),
1984
+ "Check `docker ps` for this sandbox's proxy container before relying on the new policy: the replacement did not come up, so this sandbox either has no route out at all (every request fails closed) or still has the previous container enforcing the policy you just replaced. Retry, or destroy the sandbox and create it under the policy you want.",
1985
+ )
1986
+ }
1987
+ }
1988
+
1989
+ /**
1990
+ * Whether the daemon says the proxy container is still up.
1991
+ *
1992
+ * Read the same way {@link inspectNetworkInternalFlag} reads the network's
1993
+ * `Internal` flag, and for the same reason: an unreadable answer is not
1994
+ * evidence. A container that has already exited is gone (`--rm` removed it),
1995
+ * so the failure arrives as a failed `docker inspect` rather than as a
1996
+ * `false`, and both have to land on the same refusal.
1997
+ */
1998
+ async function assertEgressProxyContainerIsRunning(
1999
+ docker: string,
2000
+ containerName: string,
2001
+ signal?: AbortSignal,
2002
+ ): Promise<void> {
2003
+ let running = ''
2004
+ try {
2005
+ running = await runOnce(
2006
+ docker,
2007
+ ['inspect', '--format', '{{.State.Running}}', containerName],
2008
+ signal,
2009
+ )
2010
+ } catch {
2011
+ signal?.throwIfAborted()
2012
+ running = ''
2013
+ }
2014
+ if (running.trim() === 'true') return
2015
+ throw withHint(
2016
+ new Error(
2017
+ `The egress proxy container '${containerName}' is not running after being started, so this sandbox has no boundary to reach and no other route out.`,
2018
+ ),
2019
+ 'Almost always the image: it either lacks the compiled module (build the package before the image — `pnpm --filter @namzu/sandbox build`) or the entrypoint refused its configuration. `docker logs <container>` has the line the entrypoint wrote; the container is started with --rm, so an already-exited one is gone and its output with it.',
2020
+ )
2021
+ }
2022
+
1339
2023
  /**
1340
2024
  * Ask Docker which host port it bound to the worker port. Used
1341
2025
  * instead of the pre-reserve-then-publish pattern (which had a
@@ -1455,10 +2139,104 @@ async function waitForWorkerReady(
1455
2139
  )
1456
2140
  }
1457
2141
 
1458
- function runOnce(binary: string, args: string[], signal?: AbortSignal): Promise<string> {
2142
+ /**
2143
+ * The argv as it may appear in an error message: the KEYS of every env
2144
+ * entry, with the values replaced.
2145
+ *
2146
+ * A rendered argv is the last place a secret should survive. A non-zero
2147
+ * `docker run` is a routine outcome — a missing image, a name conflict, a
2148
+ * daemon hiccup, ENOSPC — and its message goes wherever the sandbox
2149
+ * package's errors go: a log line, a telemetry batch, a CI transcript, a
2150
+ * pasted bug report. The env flags carry the worker's credential and every
2151
+ * value the host put in `options.env` (an API key, a broker token), and
2152
+ * none of them are needed to explain an exit code. The keys are kept
2153
+ * because they are what distinguishes "the image could not be pulled" from
2154
+ * "the environment was rejected".
2155
+ *
2156
+ * EVERY SPELLING docker accepts for that flag, not the one this backend
2157
+ * happens to emit today. `-e` IS `--env`, separated or `=`-attached, and
2158
+ * this function used to compare each element to the literal `'--env'`: the
2159
+ * long separated form this builder writes was redacted and `-e K=V`,
2160
+ * `--env=K=V` and `-e=K=V` were printed in full. A future caller writing
2161
+ * any of the three would have put a credential in a log line behind a
2162
+ * docblock that promised it would not. The covered forms are `--env K=V`,
2163
+ * `--env=K=V`, `-e K=V`, `-e=K=V` and the attached short form `-eK=V`; the
2164
+ * one shape it does not read is a value attached to an `-e` bundled into a
2165
+ * group of other short flags (`-iteK=V`), which no caller here writes and
2166
+ * which no rule short of matching `-e` anywhere inside an option could
2167
+ * catch. A valueless entry in any form (`--env K`) is passed through: it
2168
+ * resolves from the CLI's own environment and carries no value to redact.
2169
+ *
2170
+ * That redacts the workspace paths the layout is rendered from as well,
2171
+ * which are not secrets. They are also not what an exit code is about, and
2172
+ * a rule with exceptions is a rule that leaks the first time someone's
2173
+ * credential does not look like one.
2174
+ */
2175
+ export function redactDockerArgv(args: readonly string[]): string[] {
2176
+ const rendered = [...args]
2177
+ /** `K=V` → `K=<redacted>`, keeping the key; a valueless entry is left alone. */
2178
+ const redactEntry = (entry: string): string => {
2179
+ const separator = entry.indexOf('=')
2180
+ return separator > 0 ? `${entry.slice(0, separator)}=<redacted>` : entry
2181
+ }
2182
+ for (let index = 0; index < rendered.length; index += 1) {
2183
+ const arg = rendered[index] as string
2184
+ if (arg === '--env' || arg === '-e') {
2185
+ const entry = rendered[index + 1]
2186
+ if (entry !== undefined) rendered[index + 1] = redactEntry(entry)
2187
+ index += 1
2188
+ continue
2189
+ }
2190
+ // Attached, where the option's value is the rest of the same element:
2191
+ // the `=` forms, and the short form with no separator (`-eK=V`).
2192
+ const prefix = arg.startsWith('--env=')
2193
+ ? '--env='
2194
+ : arg.startsWith('-e=')
2195
+ ? '-e='
2196
+ : arg.startsWith('-e') && arg.length > 2
2197
+ ? '-e'
2198
+ : undefined
2199
+ if (prefix !== undefined) rendered[index] = prefix + redactEntry(arg.slice(prefix.length))
2200
+ }
2201
+ return rendered
2202
+ }
2203
+
2204
+ /**
2205
+ * Run a docker subcommand to completion.
2206
+ *
2207
+ * `extraEnv` is added to the child's environment rather than the parent's, and
2208
+ * it is how a value reaches a container without entering the argv this file
2209
+ * builds. Two callers use it, for the same reason:
2210
+ *
2211
+ * - The sandbox's own `docker run`, which hands over the worker's
2212
+ * per-instance credential for the valueless `--env NAMZU_SANDBOX_TOKEN` its
2213
+ * argv carries (minted in `spawnDockerSandbox`).
2214
+ * - The egress proxy's `docker run`, which names its configuration variable
2215
+ * in the argv and hands the value over here. See
2216
+ * `renderEgressProxyRunArgs`.
2217
+ *
2218
+ * Anything a process is told through the environment is visible to `ps`'s
2219
+ * neighbour, `/proc/<pid>/environ`, only to the user that owns it — while an
2220
+ * argv is world-readable on Linux.
2221
+ */
2222
+ function runOnce(
2223
+ binary: string,
2224
+ args: string[],
2225
+ signal?: AbortSignal,
2226
+ extraEnv?: Readonly<Record<string, string>>,
2227
+ ): Promise<string> {
1459
2228
  return new Promise((resolve, reject) => {
1460
2229
  signal?.throwIfAborted()
1461
- const child = spawn(binary, args, { stdio: ['ignore', 'pipe', 'pipe'] })
2230
+ const child = spawn(binary, args, {
2231
+ stdio: ['ignore', 'pipe', 'pipe'],
2232
+ // The one channel a value can ride in without entering the argv
2233
+ // this process builds: `ps` shows an argv to every user on the
2234
+ // host, and `/proc/<pid>/environ` is readable only by the same
2235
+ // user and root. `docker run --env NAME` (no `=`) reads the value
2236
+ // out of the CLI's own environment, which is why the credential
2237
+ // is passed this way and rendered valueless in the argv.
2238
+ ...(extraEnv ? { env: { ...process.env, ...extraEnv } } : {}),
2239
+ })
1462
2240
  let stdout = ''
1463
2241
  let stderr = ''
1464
2242
  let settled = false
@@ -1485,7 +2263,12 @@ function runOnce(binary: string, args: string[], signal?: AbortSignal): Promise<
1485
2263
  child.on('error', (error) => finish(error))
1486
2264
  child.on('close', (code) => {
1487
2265
  if (code === 0) finish(undefined, stdout.trim())
1488
- else finish(new Error(`${binary} ${args.join(' ')} exited ${code}: ${stderr.trim()}`))
2266
+ else
2267
+ finish(
2268
+ new Error(
2269
+ `${binary} ${redactDockerArgv(args).join(' ')} exited ${code}: ${stderr.trim()}`,
2270
+ ),
2271
+ )
1489
2272
  })
1490
2273
  if (signal?.aborted) abort()
1491
2274
  else signal?.addEventListener('abort', abort, { once: true })