@namzu/sandbox 15.0.0 → 17.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +324 -0
- package/README.md +223 -0
- package/dist/backends/aci-standby-pool/index.d.ts +22 -4
- package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
- package/dist/backends/aci-standby-pool/index.js +31 -7
- package/dist/backends/aci-standby-pool/index.js.map +1 -1
- package/dist/backends/docker/index.d.ts +408 -26
- package/dist/backends/docker/index.d.ts.map +1 -1
- package/dist/backends/docker/index.js +1173 -168
- package/dist/backends/docker/index.js.map +1 -1
- package/dist/backends/firecracker/transport.d.ts +156 -1
- package/dist/backends/firecracker/transport.d.ts.map +1 -1
- package/dist/backends/firecracker/transport.js +223 -29
- package/dist/backends/firecracker/transport.js.map +1 -1
- package/dist/backends/http-worker-client.d.ts +64 -2
- package/dist/backends/http-worker-client.d.ts.map +1 -1
- package/dist/backends/http-worker-client.js +78 -7
- package/dist/backends/http-worker-client.js.map +1 -1
- package/dist/backends/kubernetes/egress-policy.d.ts +193 -102
- package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -1
- package/dist/backends/kubernetes/egress-policy.js +321 -146
- package/dist/backends/kubernetes/egress-policy.js.map +1 -1
- package/dist/backends/kubernetes/per-sandbox-policy.d.ts +6 -6
- package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -1
- package/dist/backends/kubernetes/per-sandbox-policy.js +21 -53
- package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -1
- package/dist/backends/kubernetes/transport.d.ts +7 -0
- package/dist/backends/kubernetes/transport.d.ts.map +1 -1
- package/dist/backends/kubernetes/transport.js.map +1 -1
- package/dist/egress/proxy.d.ts +47 -2
- package/dist/egress/proxy.d.ts.map +1 -1
- package/dist/egress/proxy.js +31 -7
- package/dist/egress/proxy.js.map +1 -1
- package/dist/index.d.ts +130 -6
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +55 -5
- package/dist/index.js.map +1 -1
- package/package.json +4 -4
- package/src/backends/aci-standby-pool/index.ts +37 -7
- package/src/backends/docker/index.ts +1475 -196
- package/src/backends/firecracker/transport.ts +387 -36
- package/src/backends/http-worker-client.ts +89 -5
- package/src/backends/kubernetes/egress-policy.ts +455 -187
- package/src/backends/kubernetes/per-sandbox-policy.ts +21 -66
- package/src/backends/kubernetes/transport.ts +7 -0
- package/src/egress/proxy.ts +65 -8
- package/src/index.ts +162 -5
|
@@ -14,14 +14,35 @@
|
|
|
14
14
|
* Trust model:
|
|
15
15
|
* - Container is the trust boundary; everything inside is treated
|
|
16
16
|
* as untrusted code.
|
|
17
|
-
* -
|
|
18
|
-
*
|
|
19
|
-
*
|
|
20
|
-
*
|
|
21
|
-
*
|
|
17
|
+
* - Every call to the worker's control API carries the per-instance
|
|
18
|
+
* `NAMZU_SANDBOX_TOKEN` this backend mints at create time and
|
|
19
|
+
* injects into the container's environment; a worker the host did
|
|
20
|
+
* not create must be provisioned with its own. The worker requires
|
|
21
|
+
* it on every route but `/healthz`, and refuses to start at all if
|
|
22
|
+
* it has none and is bound to anything routable.
|
|
23
|
+
* - Outbound network from the worker is restricted by the network
|
|
24
|
+
* it is attached to (see {@link DockerBackendConfig.network}) and,
|
|
25
|
+
* for a host allowlist, by an egress proxy running as a sibling
|
|
26
|
+
* container on that same network — the only way its traffic reaches
|
|
27
|
+
* the internet, because the network is `--internal` and has no route
|
|
28
|
+
* off it. The proxy is a container on that subnet like any other, so
|
|
29
|
+
* it is not the sandbox's only reachable destination. The proxy
|
|
30
|
+
* environment the sandbox is given directs traffic; the topology is
|
|
31
|
+
* what confines it. See {@link assertNetworkCarriesThePolicy} and
|
|
32
|
+
* #398.
|
|
33
|
+
*
|
|
34
|
+
* The credential above is what a previous version of this docblock
|
|
35
|
+
* claimed network placement alone provided. It said the worker "only
|
|
36
|
+
* listens on loopback inside its own netns", which is false — the worker
|
|
37
|
+
* binds every interface by default and has to, because a published
|
|
38
|
+
* container port forwards to the container's interface address rather
|
|
39
|
+
* than to its loopback, so a loopback-bound worker is unreachable through
|
|
40
|
+
* the port this backend publishes. The boundary was the network the
|
|
41
|
+
* container is attached to, and wanted a credential behind it.
|
|
22
42
|
*/
|
|
23
43
|
|
|
24
44
|
import { spawn } from 'node:child_process'
|
|
45
|
+
import { randomBytes } from 'node:crypto'
|
|
25
46
|
|
|
26
47
|
import {
|
|
27
48
|
type ContainerSandboxLayout,
|
|
@@ -46,12 +67,7 @@ import {
|
|
|
46
67
|
walkFilesViaExec,
|
|
47
68
|
withHint,
|
|
48
69
|
} from '@namzu/sdk'
|
|
49
|
-
import {
|
|
50
|
-
import type {
|
|
51
|
-
BrokeredCredential,
|
|
52
|
-
EgressProxyOptions,
|
|
53
|
-
RunningEgressProxy,
|
|
54
|
-
} from '../../egress/index.js'
|
|
70
|
+
import type { BrokeredCredential } from '../../egress/index.js'
|
|
55
71
|
|
|
56
72
|
import {
|
|
57
73
|
ContainerSandboxLayoutValidationError,
|
|
@@ -59,7 +75,11 @@ import {
|
|
|
59
75
|
type SandboxBackend,
|
|
60
76
|
type SandboxBackendOptions,
|
|
61
77
|
} from '../../index.js'
|
|
62
|
-
import {
|
|
78
|
+
import {
|
|
79
|
+
HttpWorkerClient,
|
|
80
|
+
WORKER_UNAUTHORIZED_HINT,
|
|
81
|
+
workerAuthorization,
|
|
82
|
+
} from '../http-worker-client.js'
|
|
63
83
|
import {
|
|
64
84
|
OperationDeadline,
|
|
65
85
|
OperationDeadlineExpired,
|
|
@@ -97,14 +117,70 @@ export interface DockerBackendInternalConfig {
|
|
|
97
117
|
/**
|
|
98
118
|
* `--user` value for the container, e.g. `'1000:1000'` or `'nobody'`.
|
|
99
119
|
*
|
|
100
|
-
* Left unset by default because
|
|
101
|
-
*
|
|
102
|
-
*
|
|
103
|
-
* non-root
|
|
104
|
-
*
|
|
120
|
+
* Left unset by default because `--user` does not ADD a non-root user, it
|
|
121
|
+
* OVERRIDES the image's own choice of one. The reference image ends with
|
|
122
|
+
* `USER namzu` (uid 1001, its `/workspace` chowned to match), so this
|
|
123
|
+
* backend's default is already non-root for the image it ships — a
|
|
124
|
+
* hard-coded uid here would replace that with a guess, and the guess is
|
|
125
|
+
* wrong for any image whose files are owned by someone else, which
|
|
126
|
+
* surfaces as `EACCES` on a path the workload was told it could write.
|
|
127
|
+
* Set it when the image does not declare a user of its own, or when the
|
|
128
|
+
* host wants a different one than it declares.
|
|
105
129
|
*/
|
|
106
130
|
readonly runAsUser?: string
|
|
107
131
|
|
|
132
|
+
/**
|
|
133
|
+
* CPU cores the container may use, rendered as `--cpus`. Unset by default.
|
|
134
|
+
*
|
|
135
|
+
* `--memory` and `--pids-limit` bound what a workload can take from the
|
|
136
|
+
* host, and CPU had no equivalent at all — no default, no knob — which
|
|
137
|
+
* reads as an oversight rather than a decision. It stays unset for the
|
|
138
|
+
* same reason neither of those two has a numeric default: the right value
|
|
139
|
+
* is a property of the host's machine and of what the workload is for, and
|
|
140
|
+
* any number this backend picked would silently throttle a run that
|
|
141
|
+
* finishes inside its timeout today. A host that wants the bound says what
|
|
142
|
+
* it is; the value is a decimal (`--cpus 1.5` is one and a half cores'
|
|
143
|
+
* worth of time, not a rounding).
|
|
144
|
+
*
|
|
145
|
+
* It lives on this config rather than beside `memoryLimitMb` on the
|
|
146
|
+
* per-call options because the documented deployment constructs one
|
|
147
|
+
* provider per task, so construction time IS per-task — and a control
|
|
148
|
+
* added to the tier-agnostic per-call shape would have to be refused by
|
|
149
|
+
* the ACI and kubernetes backends, which cannot apply a per-sandbox CPU
|
|
150
|
+
* limit any more than they can apply the memory and process ones.
|
|
151
|
+
*/
|
|
152
|
+
readonly cpuLimit?: number
|
|
153
|
+
|
|
154
|
+
/**
|
|
155
|
+
* Mount the container's root filesystem read-only. Default `true`.
|
|
156
|
+
*
|
|
157
|
+
* See {@link HARDENING_ARGS} for why the default is on and
|
|
158
|
+
* {@link renderWritableRootfsArgs} for the paths that stay writable while
|
|
159
|
+
* it is. Set it to `false` to make every path inside the container
|
|
160
|
+
* writable again, which is what a host whose image writes somewhere the
|
|
161
|
+
* writable set cannot describe needs, and which is why the switch exists
|
|
162
|
+
* instead of an unwritten rule that the baseline is absolute. It turns off
|
|
163
|
+
* that one control and nothing else: `--cap-drop=ALL`,
|
|
164
|
+
* `--security-opt=no-new-privileges` and `--ipc private` are applied
|
|
165
|
+
* whatever this says. It is a config field rather than an argument so that
|
|
166
|
+
* turning it off is a line somebody wrote on purpose, and not the default
|
|
167
|
+
* anyone gets by not looking.
|
|
168
|
+
*/
|
|
169
|
+
readonly readOnlyRootfs?: boolean
|
|
170
|
+
|
|
171
|
+
/**
|
|
172
|
+
* Extra paths to keep writable under `--read-only`, each mounted `--tmpfs`.
|
|
173
|
+
*
|
|
174
|
+
* The default set ({@link DEFAULT_WRITABLE_ROOTFS_PATHS}) is the reference
|
|
175
|
+
* image's needs, read off its Dockerfile; this is how a host that points
|
|
176
|
+
* `image` somewhere else says what ITS image needs, because the backend
|
|
177
|
+
* cannot read that out of an image and guessing is what these paths would
|
|
178
|
+
* otherwise be. A path the layout already mounts is refused rather than
|
|
179
|
+
* mounted twice (`Duplicate mount point`), and setting this at all beside
|
|
180
|
+
* `readOnlyRootfs: false` is refused as a contradiction.
|
|
181
|
+
*/
|
|
182
|
+
readonly writableRootfsPaths?: readonly string[]
|
|
183
|
+
|
|
108
184
|
/**
|
|
109
185
|
* Credentials the egress proxy stamps on, per host.
|
|
110
186
|
*
|
|
@@ -133,6 +209,55 @@ export interface DockerBackendInternalConfig {
|
|
|
133
209
|
*/
|
|
134
210
|
readonly allowInwardFor?: readonly string[]
|
|
135
211
|
|
|
212
|
+
/**
|
|
213
|
+
* Image the egress proxy runs as, when the policy needs one.
|
|
214
|
+
*
|
|
215
|
+
* A host allowlist (`static` or `resolver`) is enforced by a proxy, and
|
|
216
|
+
* this backend no longer runs that proxy in its own process: it runs it as
|
|
217
|
+
* a sibling container, dual-homed between the sandbox's `--internal`
|
|
218
|
+
* network (where the sandbox can reach it, and nothing else) and an
|
|
219
|
+
* ordinary bridge (where it reaches the internet). See
|
|
220
|
+
* {@link renderEgressProxyRunArgs} for the argv and
|
|
221
|
+
* `egress-proxy/Dockerfile` for the image to build.
|
|
222
|
+
*
|
|
223
|
+
* It is a second image rather than a reuse of `image` on purpose, and the
|
|
224
|
+
* reason is deployment-shaped rather than hygienic: the sandbox image is a
|
|
225
|
+
* host-supplied string that this backend cannot read, so there is no way to
|
|
226
|
+
* know whether the image named there has the proxy module in it, and the
|
|
227
|
+
* bind-mount alternative breaks exactly on the remote-daemon deployment
|
|
228
|
+
* (`hostReachability: 'container-network'`) where the SDK's own filesystem
|
|
229
|
+
* is not the daemon's.
|
|
230
|
+
*
|
|
231
|
+
* Unset, an allowlist policy is REFUSED at `create()` rather than
|
|
232
|
+
* downgraded — the same rule the rest of this file applies to a policy it
|
|
233
|
+
* cannot enforce. `deny-all` and `allow-all` need no image.
|
|
234
|
+
*/
|
|
235
|
+
readonly egressProxyImage?: string
|
|
236
|
+
|
|
237
|
+
/**
|
|
238
|
+
* Network the egress proxy joins for its route to the internet. Default
|
|
239
|
+
* `'bridge'`, docker's own default bridge.
|
|
240
|
+
*
|
|
241
|
+
* This is the proxy's second leg, and it exists because the internal
|
|
242
|
+
* network the sandbox sits on has no route out by design. `'bridge'` is
|
|
243
|
+
* the default because it always exists and always has NAT, which keeps a
|
|
244
|
+
* host's first allowlist policy working without a second network to
|
|
245
|
+
* create.
|
|
246
|
+
*
|
|
247
|
+
* **The default is also the one wart of this topology, and a host that
|
|
248
|
+
* shares a docker daemon should know it.** The proxy listens on every
|
|
249
|
+
* interface inside its own container, so any other container attached to
|
|
250
|
+
* the same network reaches it too — and this proxy enforces its allowlist
|
|
251
|
+
* for whoever asks and stamps brokered credentials on the requests it
|
|
252
|
+
* forwards. On `'bridge'` that means every container on the host's default
|
|
253
|
+
* bridge. Name a dedicated network here (one only this deployment's
|
|
254
|
+
* containers join) to decide who that is. There is deliberately no switch
|
|
255
|
+
* that narrows the listener instead: the container has no way to know
|
|
256
|
+
* which of its own interfaces is which, and a listener bound to the wrong
|
|
257
|
+
* one is a sandbox with no route out at all.
|
|
258
|
+
*/
|
|
259
|
+
readonly egressProxyUpstreamNetwork?: string
|
|
260
|
+
|
|
136
261
|
readonly network?: 'none' | 'bridge' | string
|
|
137
262
|
readonly readyPollIntervalMs?: number
|
|
138
263
|
readonly readyTimeoutMs?: number
|
|
@@ -173,6 +298,12 @@ export interface DockerBackendInternalConfig {
|
|
|
173
298
|
* monitoring filters) via `docker ps --filter label=…`. Keys
|
|
174
299
|
* containing `=` or empty names throw at spawn time — the docker
|
|
175
300
|
* CLI accepts them but the resulting label split is ambiguous.
|
|
301
|
+
*
|
|
302
|
+
* They are applied to the egress proxy's container too, when one runs.
|
|
303
|
+
* That is deliberate in both directions: a reaper that collects a
|
|
304
|
+
* sandbox's containers should collect the proxy with them, and a proxy
|
|
305
|
+
* left running by a sandbox that died is a container holding real
|
|
306
|
+
* credentials with a route to the internet.
|
|
176
307
|
*/
|
|
177
308
|
readonly labels?: Readonly<Record<string, string>>
|
|
178
309
|
}
|
|
@@ -188,6 +319,13 @@ const WORKER_PORT_INSIDE_CONTAINER = 2024
|
|
|
188
319
|
* `create()` call.
|
|
189
320
|
*/
|
|
190
321
|
export function buildDockerBackend(config: DockerBackendInternalConfig): SandboxBackend {
|
|
322
|
+
// Refused here rather than at the first spawn, so a config that cannot be
|
|
323
|
+
// rendered — a CPU limit that cannot mean anything, or writable paths
|
|
324
|
+
// beside a writable root filesystem — surfaces during host wiring instead
|
|
325
|
+
// of as a container that failed to come up. The same checks run where the
|
|
326
|
+
// argv is built, because that is the only place a caller cannot skip them.
|
|
327
|
+
assertCpuLimitIsRenderable(config.cpuLimit)
|
|
328
|
+
assertRootfsOptionsAreCoherent(config)
|
|
191
329
|
const readiness = resolveReadinessOptions(
|
|
192
330
|
'docker',
|
|
193
331
|
config.readyTimeoutMs,
|
|
@@ -222,6 +360,14 @@ export function buildDockerBackend(config: DockerBackendInternalConfig): Sandbox
|
|
|
222
360
|
* just the way out. It now keeps the configured network, and
|
|
223
361
|
* {@link assertNetworkCarriesThePolicy} is what makes that network a
|
|
224
362
|
* boundary.
|
|
363
|
+
*
|
|
364
|
+
* Every policy that returns the configured network depends on that same
|
|
365
|
+
* assertion, which is why the two live side by side: this function says which
|
|
366
|
+
* network the container joins, and that one refuses the network if it cannot
|
|
367
|
+
* do what the policy asks of it (#398). An allowlist used to answer the
|
|
368
|
+
* configured network and get a proxy environment variable pointed at a
|
|
369
|
+
* host-side listener — enforcement by convention, which an uncooperative
|
|
370
|
+
* process simply ignores.
|
|
225
371
|
*/
|
|
226
372
|
export function resolveNetwork(
|
|
227
373
|
configured: string,
|
|
@@ -242,7 +388,7 @@ export function resolveNetwork(
|
|
|
242
388
|
// reporting that it had been restricted.
|
|
243
389
|
if (hasProxy) return configured
|
|
244
390
|
throw new Error(
|
|
245
|
-
`The docker sandbox backend cannot enforce an egress policy of kind '${egress.kind}' without an egress proxy: it has nothing to filter hosts through.
|
|
391
|
+
`The docker sandbox backend cannot enforce an egress policy of kind '${egress.kind}' without an egress proxy: it has nothing to filter hosts through. Name the image it should run the proxy as (egressProxyImage — see packages/sandbox/egress-proxy/Dockerfile), or use 'deny-all' / 'allow-all'. Refusing rather than silently granting full network access.`,
|
|
246
392
|
)
|
|
247
393
|
}
|
|
248
394
|
}
|
|
@@ -284,9 +430,9 @@ export function isInternalNetwork(inspectedInternalFlag: string): boolean {
|
|
|
284
430
|
/**
|
|
285
431
|
* Refuse a container whose network cannot do what was asked of it.
|
|
286
432
|
*
|
|
287
|
-
*
|
|
288
|
-
* unstated — which is how the backend came to ship a default
|
|
289
|
-
* that could not create a sandbox at all:
|
|
433
|
+
* Three requirements meet on the same object here, and the first two were
|
|
434
|
+
* previously unstated — which is how the backend came to ship a default
|
|
435
|
+
* configuration that could not create a sandbox at all:
|
|
290
436
|
*
|
|
291
437
|
* - **A published host port needs a route out.** Docker binds the port by
|
|
292
438
|
* NAT to the container's address, so a container with no address gets no
|
|
@@ -302,12 +448,31 @@ export function isInternalNetwork(inspectedInternalFlag: string): boolean {
|
|
|
302
448
|
* nothing. `deny-all` pointed at the default bridge would be full egress
|
|
303
449
|
* under a policy object claiming none — the "accepted and silently
|
|
304
450
|
* ignored" failure the rest of this file exists to refuse.
|
|
451
|
+
* - **A host allowlist needs one too, for the same reason and with one
|
|
452
|
+
* addition.** Until #398 the allowlist tier answered the configured
|
|
453
|
+
* network and pointed the sandbox at a proxy running on the host's
|
|
454
|
+
* loopback, and the only thing making traffic go through it was
|
|
455
|
+
* `HTTP_PROXY`. That is a request, not a boundary: anything inside the
|
|
456
|
+
* container that opens a socket directly reaches the network with the
|
|
457
|
+
* allowlist unconsulted, and untrusted code is the caller least likely to
|
|
458
|
+
* honour a convention. With the proxy as a sibling container on an
|
|
459
|
+
* `--internal` network, the only way the sandbox's traffic reaches the
|
|
460
|
+
* internet is that container, and it cannot change that because
|
|
461
|
+
* `--cap-drop=ALL` took `NET_ADMIN` away (see {@link HARDENING_ARGS}). The
|
|
462
|
+
* environment variables stay and now DIRECT traffic rather than permit it.
|
|
463
|
+
* On a network that is not internal there is no such route to remove, and
|
|
464
|
+
* the policy is back to being advisory — which is what this refuses.
|
|
305
465
|
*
|
|
306
|
-
*
|
|
307
|
-
* impossible rather than merely unsupported: no arrangement of docker
|
|
466
|
+
* The first two are exact opposites, so `deny-all` over a published host port
|
|
467
|
+
* is impossible rather than merely unsupported: no arrangement of docker
|
|
308
468
|
* networking both denies all egress and lets the host reach the worker over
|
|
309
|
-
* TCP. Closing that needs the control channel moved off TCP — see #398 —
|
|
310
|
-
*
|
|
469
|
+
* TCP. Closing that needs the control channel moved off TCP — see #398 — and
|
|
470
|
+
* is not a flag this function could accept. The same opposition now reaches
|
|
471
|
+
* an allowlist policy, which is new: an allowlist on an internal network must
|
|
472
|
+
* be reached by container name, so a host that used a published port with a
|
|
473
|
+
* host allowlist has to move that consumer onto the internal network. That is
|
|
474
|
+
* a consequence of the boundary existing at all and not a gap in this
|
|
475
|
+
* function; the refusal below names the mode to move to.
|
|
311
476
|
*/
|
|
312
477
|
export function assertNetworkCarriesThePolicy(
|
|
313
478
|
network: string,
|
|
@@ -328,31 +493,212 @@ export function assertNetworkCarriesThePolicy(
|
|
|
328
493
|
`The docker sandbox backend was asked for an egress policy of 'deny-all' on network '${network}', but that network is not internal, so the container can still reach the world. Create it with 'docker network create --internal ${network}' — an internal bridge denies egress in the kernel, rather than through an environment variable a workload may decline to read, while sibling containers still reach the worker by name. Refusing rather than reporting a boundary that is not there.`,
|
|
329
494
|
)
|
|
330
495
|
}
|
|
496
|
+
|
|
497
|
+
if (needsEgressProxy(egress) && !internal) {
|
|
498
|
+
throw new Error(
|
|
499
|
+
`The docker sandbox backend was asked for an egress policy of kind '${egress?.kind}' on network '${network}', but that network is not internal, so nothing stops the container from reaching the world directly and the allowlist would only be a proxy environment variable a workload may decline to read. Create the network with 'docker network create --internal ${network}': the sandbox then reaches the internet only through the egress proxy container, which is the boundary — and set hostReachability: 'container-network' to reach the worker by name on it, since a published host port needs a route out this network does not have. Refusing rather than reporting a boundary that is not there.`,
|
|
500
|
+
)
|
|
501
|
+
}
|
|
331
502
|
}
|
|
332
503
|
|
|
333
504
|
/**
|
|
334
|
-
*
|
|
505
|
+
* What the proxy container is told, as a value.
|
|
506
|
+
*
|
|
507
|
+
* The boundary's whole configuration. It is a value for the reason
|
|
508
|
+
* {@link resolveNetwork} is one: everything downstream of here needs a running
|
|
509
|
+
* Docker daemon, so a credential that failed to reach the boundary could only
|
|
510
|
+
* be caught by an operator noticing their requests arrive unauthenticated in
|
|
511
|
+
* production. A knob a host sets and the boundary never receives is the
|
|
512
|
+
* failure this shape exists to make testable.
|
|
335
513
|
*
|
|
336
|
-
*
|
|
337
|
-
*
|
|
338
|
-
*
|
|
339
|
-
*
|
|
340
|
-
*
|
|
514
|
+
* `parseProxyConfig` in `egress-proxy/server.mjs` refuses every shape in here
|
|
515
|
+
* it cannot read — and this function's own output is fed to that parser by
|
|
516
|
+
* `__tests__/hardening.test.ts`, so the two ends are pinned against each other
|
|
517
|
+
* rather than each against its own idea of the shape. A field renamed on one
|
|
518
|
+
* side alone fails a test rather than starting a boundary that enforces
|
|
519
|
+
* something nobody wrote.
|
|
341
520
|
*/
|
|
342
|
-
export
|
|
521
|
+
export interface EgressProxyContainerConfig {
|
|
522
|
+
readonly port: number
|
|
523
|
+
readonly allowedHosts: readonly string[]
|
|
524
|
+
readonly credentials: readonly BrokeredCredential[]
|
|
525
|
+
readonly allowInwardFor?: readonly string[]
|
|
526
|
+
/**
|
|
527
|
+
* Names that denote the proxy itself, for its loop guard. The container's
|
|
528
|
+
* own hostname is added to this by the entrypoint; this field carries the
|
|
529
|
+
* network alias, which is the name the sandbox actually dials.
|
|
530
|
+
*/
|
|
531
|
+
readonly selfNames?: readonly string[]
|
|
532
|
+
}
|
|
533
|
+
|
|
534
|
+
/**
|
|
535
|
+
* Build the proxy container's configuration.
|
|
536
|
+
*
|
|
537
|
+
* `allowedHosts` is already resolved here rather than passed as a policy, and
|
|
538
|
+
* that is the one behavioural difference this change carries into the
|
|
539
|
+
* boundary: the in-process proxy called the host's resolver per request, and
|
|
540
|
+
* the container cannot. A `resolver` policy that rotates is honoured at
|
|
541
|
+
* `create()` and again at every `setNetworkPolicy()` — the container has no
|
|
542
|
+
* channel back to the host's resolver, and a channel the container CAN reach
|
|
543
|
+
* is one the sandbox can reach too, which would let the sandbox ask for its
|
|
544
|
+
* own allowlist to be widened. See `docs/sdk/sandbox-egress.md`.
|
|
545
|
+
*/
|
|
546
|
+
export function egressProxyContainerConfig(
|
|
343
547
|
config: Pick<DockerBackendInternalConfig, 'brokeredCredentials' | 'allowInwardFor'>,
|
|
344
|
-
|
|
345
|
-
|
|
548
|
+
allowedHosts: readonly string[],
|
|
549
|
+
port: number,
|
|
550
|
+
): EgressProxyContainerConfig {
|
|
346
551
|
return {
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
// on a live sandbox.
|
|
350
|
-
allowedHosts: () => resolveAllowedHosts(policy),
|
|
552
|
+
port,
|
|
553
|
+
allowedHosts,
|
|
351
554
|
credentials: config.brokeredCredentials ?? [],
|
|
352
555
|
...(config.allowInwardFor ? { allowInwardFor: config.allowInwardFor } : {}),
|
|
556
|
+
selfNames: [PROXY_HOST_ALIAS],
|
|
353
557
|
}
|
|
354
558
|
}
|
|
355
559
|
|
|
560
|
+
/**
|
|
561
|
+
* `--label key=value` flags, validated before they reach the daemon.
|
|
562
|
+
*
|
|
563
|
+
* An empty key or one containing `=` would silently produce a malformed label
|
|
564
|
+
* that a downstream `docker ps --filter label=…` could not match, so misuse
|
|
565
|
+
* surfaces during construction rather than as a container that mysteriously
|
|
566
|
+
* has no labels.
|
|
567
|
+
*/
|
|
568
|
+
function renderLabelArgs(labels: Readonly<Record<string, string>> | undefined): string[] {
|
|
569
|
+
const args: string[] = []
|
|
570
|
+
if (!labels) return args
|
|
571
|
+
for (const [key, value] of Object.entries(labels)) {
|
|
572
|
+
if (!key || key.includes('=')) {
|
|
573
|
+
throw new Error(`docker label key ${JSON.stringify(key)} is invalid (empty or contains '=')`)
|
|
574
|
+
}
|
|
575
|
+
args.push('--label', `${key}=${value}`)
|
|
576
|
+
}
|
|
577
|
+
return args
|
|
578
|
+
}
|
|
579
|
+
|
|
580
|
+
/**
|
|
581
|
+
* Name the proxy container reads its configuration from.
|
|
582
|
+
*
|
|
583
|
+
* The NAME travels in the argv and the VALUE travels in the `docker` CLI
|
|
584
|
+
* child's environment, which is why this is one constant used by both ends of
|
|
585
|
+
* that pair rather than a literal written twice.
|
|
586
|
+
*/
|
|
587
|
+
const EGRESS_PROXY_CONFIG_ENV = 'NAMZU_EGRESS_PROXY_CONFIG'
|
|
588
|
+
|
|
589
|
+
/** Everything {@link renderEgressProxyRunArgs} renders, as a value. */
|
|
590
|
+
export interface EgressProxyArgvInput {
|
|
591
|
+
readonly config: DockerBackendInternalConfig
|
|
592
|
+
readonly containerName: string
|
|
593
|
+
/** Network the proxy reaches the internet through. See `egressProxyUpstreamNetwork`. */
|
|
594
|
+
readonly upstreamNetwork: string
|
|
595
|
+
/** The internal network the sandbox is on, which the proxy is also joined to. */
|
|
596
|
+
readonly internalNetwork: string
|
|
597
|
+
}
|
|
598
|
+
|
|
599
|
+
/**
|
|
600
|
+
* The `docker run` argv for the egress proxy container, as a value.
|
|
601
|
+
*
|
|
602
|
+
* Extracted for the reason every other argv in this file is: nothing
|
|
603
|
+
* downstream of it can run without a docker daemon, so a flag that never
|
|
604
|
+
* reached it — or a `--network` that named the wrong side of a dual-homed
|
|
605
|
+
* container — could only be caught by an operator in production, where the
|
|
606
|
+
* symptom is a sandbox that reaches nothing.
|
|
607
|
+
*
|
|
608
|
+
* The two networks are the whole topology and they are two calls:
|
|
609
|
+
* `docker run --network <upstream>` gives the container a default route, and
|
|
610
|
+
* {@link renderEgressProxyAttachArgs}'s `docker network connect` adds the
|
|
611
|
+
* internal network afterwards. The order is load-bearing. Attached to the
|
|
612
|
+
* internal network FIRST it would come up with no default route and no way to
|
|
613
|
+
* acquire one, and the proxy would be a boundary in front of nothing.
|
|
614
|
+
*
|
|
615
|
+
* The hardening baseline is the sandbox's, minus the parts that describe a
|
|
616
|
+
* filesystem. `--cap-drop=ALL`, `--no-new-privileges` and `--ipc private` are
|
|
617
|
+
* applied exactly as {@link HARDENING_ARGS} defines them, because this
|
|
618
|
+
* container sits between untrusted code and the internet and is the last one in
|
|
619
|
+
* the deployment that should be holding a capability. `--read-only` is the
|
|
620
|
+
* sandbox's too, with one tmpfs: the proxy writes nothing, and the tmpfs is
|
|
621
|
+
* there so that a Node process which one day wants a temp file fails at a
|
|
622
|
+
* filesystem boundary rather than at a mysterious `EROFS`.
|
|
623
|
+
*
|
|
624
|
+
* The image is last, so nothing after it is read as a flag — the same rule the
|
|
625
|
+
* sandbox argv follows, and here it also means the image's own `CMD` starts the
|
|
626
|
+
* proxy rather than this argv naming an entrypoint.
|
|
627
|
+
*/
|
|
628
|
+
export function renderEgressProxyRunArgs(input: EgressProxyArgvInput): string[] {
|
|
629
|
+
const { config, containerName, upstreamNetwork, internalNetwork } = input
|
|
630
|
+
if (!config.egressProxyImage) {
|
|
631
|
+
throw new Error(
|
|
632
|
+
'renderEgressProxyRunArgs was called without config.egressProxyImage; the docker backend cannot start an egress proxy container it has no image for. resolveNetwork refuses an allowlist policy in this state, so reaching here means the refusal was bypassed.',
|
|
633
|
+
)
|
|
634
|
+
}
|
|
635
|
+
if (upstreamNetwork === 'none') {
|
|
636
|
+
throw new Error(
|
|
637
|
+
`egressProxyUpstreamNetwork is 'none', which would give the proxy container no interface and no route: it would come up unable to reach anything, and the only way the sandbox's traffic reaches the internet would be a boundary that cannot reach it itself. Name a bridge, or drop the field to take the 'bridge' default.`,
|
|
638
|
+
)
|
|
639
|
+
}
|
|
640
|
+
if (upstreamNetwork === internalNetwork) {
|
|
641
|
+
throw new Error(
|
|
642
|
+
`egressProxyUpstreamNetwork names '${upstreamNetwork}', which is the same network the sandbox is on — the internal one. A container whose primary network is internal comes up with no default route and never acquires one, so the proxy would start with no way to reach the internet: the boundary would be a container that can only talk to the sandbox. Name a different network for egressProxyUpstreamNetwork, or drop the field to take the 'bridge' default.`,
|
|
643
|
+
)
|
|
644
|
+
}
|
|
645
|
+
return [
|
|
646
|
+
'run',
|
|
647
|
+
'--detach',
|
|
648
|
+
'--rm',
|
|
649
|
+
'--name',
|
|
650
|
+
containerName,
|
|
651
|
+
// The name the sandbox dials, and the name the proxy's own loop guard
|
|
652
|
+
// has to recognise as itself. Set as the container's hostname so the
|
|
653
|
+
// entrypoint can read it back with `os.hostname()` instead of holding a
|
|
654
|
+
// second copy of the constant.
|
|
655
|
+
'--hostname',
|
|
656
|
+
PROXY_HOST_ALIAS,
|
|
657
|
+
'--network',
|
|
658
|
+
upstreamNetwork,
|
|
659
|
+
...HARDENING_ARGS,
|
|
660
|
+
'--read-only',
|
|
661
|
+
'--tmpfs',
|
|
662
|
+
`/tmp:${TMPFS_MOUNT_OPTIONS}`,
|
|
663
|
+
...renderLabelArgs(config.labels),
|
|
664
|
+
// The NAME only: docker takes the value from the environment of the
|
|
665
|
+
// `docker` CLI process it names, which the caller sets (see
|
|
666
|
+
// `startEgressProxyContainer`). The value is the whole policy —
|
|
667
|
+
// including brokered credential values — and an argv is a worse place
|
|
668
|
+
// for it than an environment on every platform that has a process
|
|
669
|
+
// table: `/proc/<pid>/cmdline` is world-readable on Linux, so putting
|
|
670
|
+
// it here published it to every local user on the docker host for as
|
|
671
|
+
// long as the client ran, where the child's own environment is readable
|
|
672
|
+
// only by the user that owns the process. It is still readable by
|
|
673
|
+
// anything with access to the daemon, through `docker inspect`, and
|
|
674
|
+
// `docs/sdk/sandbox-egress.md` says so.
|
|
675
|
+
'--env',
|
|
676
|
+
EGRESS_PROXY_CONFIG_ENV,
|
|
677
|
+
config.egressProxyImage,
|
|
678
|
+
]
|
|
679
|
+
}
|
|
680
|
+
|
|
681
|
+
/**
|
|
682
|
+
* The `docker network connect` argv that makes the proxy dual-homed.
|
|
683
|
+
*
|
|
684
|
+
* `--alias` is what puts `namzu-egress` in the internal network's DNS, which
|
|
685
|
+
* is the name {@link buildDockerRunArgs} hands the sandbox in its proxy
|
|
686
|
+
* environment. The alias rather than the container name because the alias is
|
|
687
|
+
* the constant: a container named `namzu-egress-<sandbox-id>` would otherwise
|
|
688
|
+
* make the sandbox's `HTTP_PROXY` value depend on a generated id, and the two
|
|
689
|
+
* would have to be kept in step by hand.
|
|
690
|
+
*/
|
|
691
|
+
export function renderEgressProxyAttachArgs(input: EgressProxyArgvInput): string[] {
|
|
692
|
+
return [
|
|
693
|
+
'network',
|
|
694
|
+
'connect',
|
|
695
|
+
'--alias',
|
|
696
|
+
PROXY_HOST_ALIAS,
|
|
697
|
+
input.internalNetwork,
|
|
698
|
+
input.containerName,
|
|
699
|
+
]
|
|
700
|
+
}
|
|
701
|
+
|
|
356
702
|
/**
|
|
357
703
|
* Confinement flags applied to every container.
|
|
358
704
|
*
|
|
@@ -363,9 +709,11 @@ export function egressProxyOptions(
|
|
|
363
709
|
* are the defaults every container runtime hardening guide starts with,
|
|
364
710
|
* and none of them were present.
|
|
365
711
|
*
|
|
366
|
-
* `--cap-drop=ALL` is deliberately not softened by a re-add list
|
|
367
|
-
*
|
|
368
|
-
*
|
|
712
|
+
* `--cap-drop=ALL` is deliberately not softened by a re-add list, and there is
|
|
713
|
+
* no config field that could soften it either: a workload that genuinely needs
|
|
714
|
+
* a capability needs a change to this file, where the diff says which
|
|
715
|
+
* capability and why. A re-add list on the config would grant it to every
|
|
716
|
+
* sandbox the host spawns, quietly, which is how a baseline stops being one.
|
|
369
717
|
*
|
|
370
718
|
* **It carries a second, independent load, and this is the one that would
|
|
371
719
|
* survive being forgotten.** An egress policy of `deny-all` is enforced by
|
|
@@ -390,12 +738,572 @@ export function egressProxyOptions(
|
|
|
390
738
|
*
|
|
391
739
|
* Recorded here because the first justification above would survive
|
|
392
740
|
* softening this flag and the second would not.
|
|
741
|
+
*
|
|
742
|
+
* `--ipc private` closes a door that is not the one its name suggests, and the
|
|
743
|
+
* difference is worth being exact about. Moby runs `private`, `shareable` and
|
|
744
|
+
* `none` through the SAME branch (`daemon/oci_linux.go`, `WithNamespaces`), so
|
|
745
|
+
* `shareable` already gives every container an IPC namespace of its own: a
|
|
746
|
+
* daemon whose `default-ipc-mode` is `shareable` does not merge anybody's
|
|
747
|
+
* namespaces, and what an unset flag buys on such a daemon is not a shared one
|
|
748
|
+
* either. What separates the modes is reachability. Docker's run reference
|
|
749
|
+
* defines `shareable` as "Own private IPC namespace, with a possibility to
|
|
750
|
+
* share it with other containers", and that possibility is `--ipc
|
|
751
|
+
* container:<name>`, which joins another container's IPC namespace — and what
|
|
752
|
+
* that join needs from the target is a shared-memory directory to enter:
|
|
753
|
+
* `daemon.getIPCContainer` resolves the target by name and is gated on its
|
|
754
|
+
* `ShmPath`, which a container created `private` does not have. So on a
|
|
755
|
+
* daemon defaulting to `shareable` this container's namespace, and the System V
|
|
756
|
+
* shared memory, semaphores and message queues namespaced with it, are joinable
|
|
757
|
+
* by anything else on that host which knows the container's name; `--ipc
|
|
758
|
+
* private` removes that reachability. Creating the joining container still takes
|
|
759
|
+
* access to the same daemon, so this is not a boundary against an unprivileged
|
|
760
|
+
* attacker, and it is not claimed as one here. What it buys is that the answer
|
|
761
|
+
* is in THIS argv rather than in the host's `daemon.json`, which is the only
|
|
762
|
+
* place the daemon's default is written down. `--ipc none` is deliberately not
|
|
763
|
+
* used: it takes `/dev/shm` away, and chromium — which the reference image
|
|
764
|
+
* ships for browser automation — uses it for every renderer process.
|
|
765
|
+
*
|
|
766
|
+
* `--read-only` makes the image itself not a place the workload can write. It
|
|
767
|
+
* is rendered by {@link renderHardeningArgs} rather than listed below, because
|
|
768
|
+
* it is the one flag here a host can turn off (`readOnlyRootfs: false`), and a
|
|
769
|
+
* flag in this array would keep being applied after the field said it was not —
|
|
770
|
+
* a control accepted and not applied, which is the failure this file refuses
|
|
771
|
+
* everywhere else. The layout's own RW binds (`outputs`, `scratch`) are
|
|
772
|
+
* separate mounts and are unaffected; what stays writable inside the
|
|
773
|
+
* container's filesystem is named, path by path and with the reason, in
|
|
774
|
+
* {@link renderWritableRootfsArgs}. A host whose image needs a path that list
|
|
775
|
+
* does not name adds it through `writableRootfsPaths`.
|
|
776
|
+
*
|
|
777
|
+
* Three controls from the published container-hardening guidance are
|
|
778
|
+
* deliberately absent, and the reason is here rather than implied:
|
|
779
|
+
*
|
|
780
|
+
* - **A seccomp profile.** Docker already applies its built-in profile to
|
|
781
|
+
* every container unless something passes `seccomp=unconfined`, and nothing
|
|
782
|
+
* in this backend does — so the tier is filtered, and what is missing is a
|
|
783
|
+
* profile TIGHTER than docker's default. Shipping one means shipping a
|
|
784
|
+
* hand-written file whose deny list has to be correct for whatever image
|
|
785
|
+
* the host names, and this repository cannot test it against the reference
|
|
786
|
+
* image's own toolchain (chromium, LibreOffice, the numpy/scipy/duckdb
|
|
787
|
+
* stack). A profile that blocks a syscall one of those needs breaks the
|
|
788
|
+
* sandbox at a point no test here would catch, which is worse than the gap
|
|
789
|
+
* it closes. A host that needs a tighter profile sets `seccomp-profile` in
|
|
790
|
+
* the daemon's `daemon.json`, where it applies to this container and every
|
|
791
|
+
* other one; `--security-opt seccomp=<file>` is the per-container form, and
|
|
792
|
+
* it is not offered as a config field because a path in a config field is a
|
|
793
|
+
* file the daemon reads from the HOST, which is a different machine from
|
|
794
|
+
* the one this backend runs on whenever it drives a remote daemon.
|
|
795
|
+
* - **`--userns-remap`.** It is not a `docker run` flag at all: it is a
|
|
796
|
+
* daemon property (`userns-remap` in `daemon.json`, or `dockerd
|
|
797
|
+
* --userns-remap=`), and per container the CLI only chooses between the
|
|
798
|
+
* namespaces the daemon already made (`--userns=host|private`). Whether a
|
|
799
|
+
* remapped namespace exists is therefore settled before this argv is read,
|
|
800
|
+
* and a flag here could not settle it — which is the whole reason the
|
|
801
|
+
* control is absent rather than configurable: this backend has nothing to
|
|
802
|
+
* say about a mapping that belongs to the machine the daemon runs on.
|
|
803
|
+
* Enabling it on the host is a real upgrade to this tier (uid 0 inside maps
|
|
804
|
+
* to an unprivileged uid outside) and costs this backend nothing; the README
|
|
805
|
+
* says so.
|
|
806
|
+
* - **`--user`.** Supported, and unset by default on purpose — see the
|
|
807
|
+
* `runAsUser` field, which is where a host that knows its image sets it.
|
|
393
808
|
*/
|
|
394
|
-
const HARDENING_ARGS: readonly string[] = [
|
|
809
|
+
const HARDENING_ARGS: readonly string[] = [
|
|
810
|
+
'--cap-drop=ALL',
|
|
811
|
+
'--security-opt=no-new-privileges',
|
|
812
|
+
'--ipc',
|
|
813
|
+
'private',
|
|
814
|
+
]
|
|
395
815
|
|
|
396
|
-
/**
|
|
816
|
+
/**
|
|
817
|
+
* Name the sandbox reaches the egress proxy by.
|
|
818
|
+
*
|
|
819
|
+
* This used to be a `--add-host namzu-egress:host-gateway` entry, because the
|
|
820
|
+
* proxy ran on the host's loopback and `host-gateway` is docker's portable
|
|
821
|
+
* name for the host from inside a container. It is now the proxy's own
|
|
822
|
+
* container name and network alias on the internal network, which needs no
|
|
823
|
+
* alias file at all: docker's embedded DNS resolves a container's aliases for
|
|
824
|
+
* every container on the same user-defined network. The constant survives the
|
|
825
|
+
* mechanism because what the sandbox is told has not changed — only what makes
|
|
826
|
+
* the name resolve, and whether anything else can be reached.
|
|
827
|
+
*/
|
|
397
828
|
const PROXY_HOST_ALIAS = 'namzu-egress'
|
|
398
829
|
|
|
830
|
+
/**
|
|
831
|
+
* Port the egress proxy listens on inside its own container.
|
|
832
|
+
*
|
|
833
|
+
* A fixed port rather than one the host reads back, because there is no host
|
|
834
|
+
* port involved: the sandbox dials the proxy container directly on the network
|
|
835
|
+
* they share. The old arrangement had to read a published port back out of
|
|
836
|
+
* `docker inspect`, with the race and the failure mode that came with it.
|
|
837
|
+
*/
|
|
838
|
+
const EGRESS_PROXY_PORT_INSIDE_CONTAINER = 2025
|
|
839
|
+
|
|
840
|
+
/** Name of the sibling container the egress proxy runs in, for one sandbox. */
|
|
841
|
+
function egressProxyContainerName(sandboxId: string): string {
|
|
842
|
+
return `${PROXY_HOST_ALIAS}-${sandboxId}`
|
|
843
|
+
}
|
|
844
|
+
|
|
845
|
+
/**
|
|
846
|
+
* Mount options for every scratch mount this backend creates.
|
|
847
|
+
*
|
|
848
|
+
* `exec` is the load-bearing one and the reason this is a named constant
|
|
849
|
+
* rather than a literal at the call site. Docker does NOT default a `--tmpfs`
|
|
850
|
+
* mount to a usable scratch directory: `withMounts` in moby's
|
|
851
|
+
* `daemon/oci_linux.go` starts every user tmpfs from
|
|
852
|
+
* `["noexec", "nosuid", "nodev", <propagation>]` and appends whatever the
|
|
853
|
+
* caller passed, so `--tmpfs /tmp` on its own is **noexec**. A workload that
|
|
854
|
+
* compiles a program into `/tmp` and runs it — `gcc -o /tmp/a.out … &&
|
|
855
|
+
* /tmp/a.out`, or a python `ctypes.CDLL` of a library it just built there —
|
|
856
|
+
* would meet `Permission denied` on an executable file, an error that reads
|
|
857
|
+
* as a broken sandbox rather than as a mount option. Scratch here is as
|
|
858
|
+
* executable as it was before this backend mounted a tmpfs over it.
|
|
859
|
+
*
|
|
860
|
+
* `nosuid` and `nodev` are kept from docker's defaults: the tmpfs is the one
|
|
861
|
+
* place inside the container a workload can write an arbitrary file to, and
|
|
862
|
+
* neither a setuid binary nor a device node there has any use that is worth
|
|
863
|
+
* the escalation path — with `--cap-drop=ALL` no device node could be created
|
|
864
|
+
* there anyway.
|
|
865
|
+
*
|
|
866
|
+
* `mode=1777` is stated rather than inherited from the kernel's tmpfs default
|
|
867
|
+
* (which is the same value): the mounts have to be writable by whichever uid
|
|
868
|
+
* the image runs as, and the backend does not know that uid. A sticky,
|
|
869
|
+
* world-writable scratch directory is what `/tmp` is, and `--read-only` here
|
|
870
|
+
* is about the image, not about the uid.
|
|
871
|
+
*/
|
|
872
|
+
const TMPFS_MOUNT_OPTIONS = 'nosuid,nodev,exec,mode=1777'
|
|
873
|
+
|
|
874
|
+
/**
|
|
875
|
+
* Paths the reference image needs writable under `--read-only`, as `--tmpfs`.
|
|
876
|
+
*
|
|
877
|
+
* `--read-only` says the image is not the workload's disk. It does not say
|
|
878
|
+
* nothing may be written, and the difference is the sandbox: the layout's own
|
|
879
|
+
* RW binds (`outputs`, `scratch`) are separate mounts and are unaffected, but
|
|
880
|
+
* the image's toolchain writes inside the container's own filesystem, and a
|
|
881
|
+
* `--read-only` that stops it is worse than the gap it closes. Read off
|
|
882
|
+
* `worker/Dockerfile`, whose whole purpose is producing DOCX/XLSX/PPTX/PDF
|
|
883
|
+
* deliverables:
|
|
884
|
+
*
|
|
885
|
+
* - `/tmp` — `TMPDIR` for python's `tempfile`, for LibreOffice's extraction
|
|
886
|
+
* and for pip's wheel builds, and the conventional place to build and run
|
|
887
|
+
* something disposable. Every scratch mount takes
|
|
888
|
+
* {@link TMPFS_MOUNT_OPTIONS}, which is where the `exec` docker would not
|
|
889
|
+
* have given us is argued for.
|
|
890
|
+
* - `/var/tmp` — the second location the temp-file conventions fall back to,
|
|
891
|
+
* for a temp file that is meant to outlive an interrupted run.
|
|
892
|
+
* - `/home/namzu` — the image's `HOME` (`useradd --create-home namzu`, uid
|
|
893
|
+
* 1001; docker sets `HOME` from the image's passwd entry). LibreOffice
|
|
894
|
+
* refuses a headless conversion without a writable user profile
|
|
895
|
+
* (`~/.config/libreoffice`), matplotlib builds a font cache in
|
|
896
|
+
* `~/.cache/matplotlib`, fontconfig keeps a user cache, npm's cache is
|
|
897
|
+
* `~/.npm`, and `pip install --user` needs `~/.local`.
|
|
898
|
+
* - `/workspace` — the image's `WORKDIR`, chowned to `namzu` on purpose
|
|
899
|
+
* (`chown -R namzu:namzu /workspace`). Leaving it out would make the
|
|
900
|
+
* Dockerfile's own guarantee false.
|
|
901
|
+
*
|
|
902
|
+
* These four are the REFERENCE image's needs, not a claim about anyone else's.
|
|
903
|
+
* A host that points `image` at its own build names what that image needs in
|
|
904
|
+
* `writableRootfsPaths`, which is why the field exists at all: the backend
|
|
905
|
+
* cannot read an image's writable set, and the alternative to asking is
|
|
906
|
+
* guessing. A path a root-running image wants (its `HOME` is `/root`) is a
|
|
907
|
+
* `writableRootfsPaths` entry for exactly that reason — `/root` is not in this
|
|
908
|
+
* list, because the shipped image does not run as root and a tmpfs nobody
|
|
909
|
+
* writes to is a claim that something does.
|
|
910
|
+
*
|
|
911
|
+
* A path the LAYOUT already mounts is skipped rather than mounted twice:
|
|
912
|
+
* docker refuses two mounts at one destination (`Duplicate mount point`), and
|
|
913
|
+
* the bind the host asked for is the one that must win. A path the HOST names
|
|
914
|
+
* that the layout also mounts is refused instead of skipped, because there the
|
|
915
|
+
* two requests contradict each other and nothing should choose between them
|
|
916
|
+
* silently.
|
|
917
|
+
*
|
|
918
|
+
* No `size=` is set. The kernel caps a tmpfs at half the host's RAM, and tmpfs
|
|
919
|
+
* pages are accounted to the container's memory cgroup, so a run that sets
|
|
920
|
+
* `--memory` already bounds scratch with the limit the host chose — while any
|
|
921
|
+
* number picked here would fail a workload that writes a bigger temp file than
|
|
922
|
+
* we guessed, with `ENOSPC` rather than a diagnosis.
|
|
923
|
+
*
|
|
924
|
+
* **The other half of that trade, said out loud because a host will meet it.**
|
|
925
|
+
* Scratch now lives in RAM instead of on the container's writable layer, so a
|
|
926
|
+
* temp file larger than half the host's RAM — or larger than `--memory`, which
|
|
927
|
+
* is the tighter of the two whenever the host set one — fails with `ENOSPC` or
|
|
928
|
+
* is OOM-killed, where writing it to disk used to succeed. That is the cost of
|
|
929
|
+
* not leaving the root filesystem writable, and it is not a bug to be reported.
|
|
930
|
+
* The remedy that keeps the baseline is the layout's own `scratch`, which is a
|
|
931
|
+
* bind to a host directory and therefore still disk-backed: a host with room on
|
|
932
|
+
* disk gives the layout one there and points `TMPDIR` at its container path
|
|
933
|
+
* through the per-call `env` option, so the spill lands on that disk instead of
|
|
934
|
+
* on a tmpfs. `readOnlyRootfs: false` is the other way, and the one to reach for
|
|
935
|
+
* second: it puts scratch back on the container's writable layer and gives up
|
|
936
|
+
* the rest of the baseline with it.
|
|
937
|
+
*/
|
|
938
|
+
const DEFAULT_WRITABLE_ROOTFS_PATHS: readonly string[] = [
|
|
939
|
+
'/tmp',
|
|
940
|
+
'/var/tmp',
|
|
941
|
+
'/workspace',
|
|
942
|
+
'/home/namzu',
|
|
943
|
+
]
|
|
944
|
+
|
|
945
|
+
/** The backend config the hardening flags are rendered from. */
|
|
946
|
+
export type DockerHardeningConfig = Pick<
|
|
947
|
+
DockerBackendInternalConfig,
|
|
948
|
+
'cpuLimit' | 'layout' | 'readOnlyRootfs' | 'writableRootfsPaths'
|
|
949
|
+
>
|
|
950
|
+
|
|
951
|
+
/**
|
|
952
|
+
* The spelling docker compares a container path by.
|
|
953
|
+
*
|
|
954
|
+
* Docker cleans a mount destination before it uses it, so `/tmp/`, `//tmp` and
|
|
955
|
+
* `/tmp/.` are one directory to it and to the kernel. The check below is an
|
|
956
|
+
* exact-string comparison, so without this a layout that spelled one of its
|
|
957
|
+
* mounts any of those ways would not match the tmpfs default at the same
|
|
958
|
+
* directory: the argv would carry both a `--tmpfs /tmp:...` and a bind at
|
|
959
|
+
* `/tmp/`, and moby would clean the two destinations into one and refuse the
|
|
960
|
+
* container at spawn with `Duplicate mount point: /tmp` — the failure the check
|
|
961
|
+
* exists to prevent, on the one path no test in this repository can reach.
|
|
962
|
+
* `resolveLayout` does not normalise these (it fills in defaults and compares
|
|
963
|
+
* spellings as written), so the cleaning has to happen here, where the
|
|
964
|
+
* comparison does.
|
|
965
|
+
*/
|
|
966
|
+
function cleanContainerPath(path: string): string {
|
|
967
|
+
const kept: string[] = []
|
|
968
|
+
for (const segment of path.split('/')) {
|
|
969
|
+
// Empty segments are `//`, `.` is the directory itself; `..` cancels the
|
|
970
|
+
// segment before it, which is what the kernel does with it too.
|
|
971
|
+
if (segment === '' || segment === '.') continue
|
|
972
|
+
if (segment === '..') kept.pop()
|
|
973
|
+
else kept.push(segment)
|
|
974
|
+
}
|
|
975
|
+
return `/${kept.join('/')}`
|
|
976
|
+
}
|
|
977
|
+
|
|
978
|
+
/**
|
|
979
|
+
* Every destination the layout mounts something at, in the spelling docker
|
|
980
|
+
* itself compares them by.
|
|
981
|
+
*
|
|
982
|
+
* The collision check below is exact-string, so a layout path spelled `/tmp/`
|
|
983
|
+
* would slip past it and docker would then refuse the container with
|
|
984
|
+
* `Duplicate mount point` — a failure at spawn, on the one path that cannot be
|
|
985
|
+
* tested without a daemon. Cleaning each path to the spelling moby reduces it
|
|
986
|
+
* to is what makes the check cover every way of writing the same directory;
|
|
987
|
+
* see {@link cleanContainerPath}.
|
|
988
|
+
*/
|
|
989
|
+
function mountedContainerPaths(layout: ResolvedContainerSandboxLayout): string[] {
|
|
990
|
+
return [
|
|
991
|
+
layout.outputs.containerPath,
|
|
992
|
+
layout.uploads?.containerPath,
|
|
993
|
+
layout.scratch?.containerPath,
|
|
994
|
+
layout.toolResults?.containerPath,
|
|
995
|
+
layout.transcripts?.containerPath,
|
|
996
|
+
...(layout.skills?.map((skill) => skill.containerPath) ?? []),
|
|
997
|
+
]
|
|
998
|
+
.filter((path): path is string => Boolean(path))
|
|
999
|
+
.map(cleanContainerPath)
|
|
1000
|
+
}
|
|
1001
|
+
|
|
1002
|
+
/**
|
|
1003
|
+
* Refuse rootfs options that cannot both be honoured.
|
|
1004
|
+
*
|
|
1005
|
+
* `writableRootfsPaths` beside `readOnlyRootfs: false` is a contradiction: with
|
|
1006
|
+
* a writable root filesystem every path is already writable, so the tmpfs
|
|
1007
|
+
* mounts would either be dropped (a control accepted and not applied) or take a
|
|
1008
|
+
* directory off the image for no reason. Refusing is the honest answer, and it
|
|
1009
|
+
* is the same one the sibling backends give a per-sandbox control they cannot
|
|
1010
|
+
* express.
|
|
1011
|
+
*
|
|
1012
|
+
* Called at construction and again where the argv is built, so a config that
|
|
1013
|
+
* reaches `create()` by some path other than `buildDockerBackend` is refused
|
|
1014
|
+
* too.
|
|
1015
|
+
*/
|
|
1016
|
+
export function assertRootfsOptionsAreCoherent(config: DockerHardeningConfig): void {
|
|
1017
|
+
const paths = config.writableRootfsPaths
|
|
1018
|
+
if (config.readOnlyRootfs !== false || paths === undefined || paths.length === 0) return
|
|
1019
|
+
throw new Error(
|
|
1020
|
+
'writableRootfsPaths was set on a docker backend configured with readOnlyRootfs: false. With a writable root filesystem every path inside the container is already writable, so these --tmpfs mounts would add nothing and take the named directories off the image. Refusing rather than accepting a control that cannot be applied: drop the paths, or drop readOnlyRootfs: false and let the read-only baseline stand.',
|
|
1021
|
+
)
|
|
1022
|
+
}
|
|
1023
|
+
|
|
1024
|
+
/**
|
|
1025
|
+
* Refuse a `--cpus` value that cannot mean what it says.
|
|
1026
|
+
*
|
|
1027
|
+
* This covers non-finite and non-positive values and does NOT claim to cover
|
|
1028
|
+
* every bound the daemon would refuse. The difference is worth stating, because
|
|
1029
|
+
* the two classes fail in different places and only one of them is decidable
|
|
1030
|
+
* here. A negative, `NaN` or `Infinity` renders into the argv as text the
|
|
1031
|
+
* daemon either rejects or turns into a bound nobody asked for, and `0` is the
|
|
1032
|
+
* opposite of a bound (`NanoCPUs` of zero is how a container says "no CPU
|
|
1033
|
+
* limit"), so a host that wrote one of those hears about it during wiring
|
|
1034
|
+
* rather than as a container that never came up.
|
|
1035
|
+
*
|
|
1036
|
+
* The upper bound is not ours to check. Moby's `verifyPlatformContainerResources`
|
|
1037
|
+
* refuses `NanoCPUs` above the DAEMON host's CPU count (`"range of CPUs is from
|
|
1038
|
+
* 0.01 to N.00, as there are only N CPUs available"`), and the same function
|
|
1039
|
+
* deliberately sets no floor of its own on Linux, leaving that to the kernel.
|
|
1040
|
+
* Neither number is knowable from here: the `docker` binary this backend drives
|
|
1041
|
+
* can be pointed at a daemon on another machine (`DOCKER_HOST`), and even
|
|
1042
|
+
* locally `os.cpus().length` is this machine's view rather than the daemon's
|
|
1043
|
+
* own `runtime.NumCPU()`. Refusing on a guess at it would break a host whose
|
|
1044
|
+
* daemon has more cores than the process driving it, which is a worse failure
|
|
1045
|
+
* than the one it would catch — those arrive from the daemon with its own
|
|
1046
|
+
* message, at spawn, where every other daemon-side refusal arrives too.
|
|
1047
|
+
*/
|
|
1048
|
+
export function assertCpuLimitIsRenderable(cpuLimit: number | undefined): void {
|
|
1049
|
+
if (cpuLimit === undefined) return
|
|
1050
|
+
if (!Number.isFinite(cpuLimit) || cpuLimit <= 0) {
|
|
1051
|
+
throw new Error(
|
|
1052
|
+
`cpuLimit must be a finite number greater than 0 (docker's --cpus takes a decimal, e.g. 1.5); got ${String(cpuLimit)}. Refusing rather than rendering an argv whose value means something other than what was written.`,
|
|
1053
|
+
)
|
|
1054
|
+
}
|
|
1055
|
+
}
|
|
1056
|
+
|
|
1057
|
+
/**
|
|
1058
|
+
* `--tmpfs` flags for the paths that stay writable under `--read-only`.
|
|
1059
|
+
*
|
|
1060
|
+
* See {@link DEFAULT_WRITABLE_ROOTFS_PATHS} for the paths themselves and why
|
|
1061
|
+
* each is there. Returns nothing when the read-only root filesystem is off, and
|
|
1062
|
+
* the two ways a host names paths that cannot be mounted — a contradiction with
|
|
1063
|
+
* `readOnlyRootfs: false`, or a path the layout already mounts — are refusals
|
|
1064
|
+
* rather than a silently shorter list.
|
|
1065
|
+
*/
|
|
1066
|
+
export function renderWritableRootfsArgs(config: DockerHardeningConfig): string[] {
|
|
1067
|
+
assertRootfsOptionsAreCoherent(config)
|
|
1068
|
+
if (config.readOnlyRootfs === false) return []
|
|
1069
|
+
|
|
1070
|
+
const mounted = new Set(mountedContainerPaths(config.layout))
|
|
1071
|
+
const requested = config.writableRootfsPaths ?? []
|
|
1072
|
+
for (const path of requested) {
|
|
1073
|
+
// Every segment non-empty and none of them `.` or `..`: an absolute path
|
|
1074
|
+
// with at least one component. Anything else is refused because each
|
|
1075
|
+
// rejected shape is a directory this file's exact-string checks could
|
|
1076
|
+
// hold two spellings of — `/tmp/`, `//tmp` and `/tmp/.` are all `/tmp` to
|
|
1077
|
+
// the kernel, so a default that mounted `/tmp` and a host entry that
|
|
1078
|
+
// mounted `/tmp/` would each pass the duplicate-mount check and then be
|
|
1079
|
+
// refused by docker at spawn, on the one path no test here can reach.
|
|
1080
|
+
// `/` itself is refused as well, and for its own reason: it would make
|
|
1081
|
+
// the whole read-only root filesystem writable again.
|
|
1082
|
+
const segments = path.split('/')
|
|
1083
|
+
const wellFormed =
|
|
1084
|
+
path.startsWith('/') &&
|
|
1085
|
+
segments.length > 1 &&
|
|
1086
|
+
segments.slice(1).every((segment) => segment !== '' && segment !== '.' && segment !== '..')
|
|
1087
|
+
if (!wellFormed) {
|
|
1088
|
+
throw new Error(
|
|
1089
|
+
`writableRootfsPaths entry ${JSON.stringify(path)} is not a normalised absolute path inside the container. Docker requires an absolute mount path with no empty, '.' or '..' segment and no trailing slash, and '/' would make the whole filesystem writable again rather than adding a scratch directory.`,
|
|
1090
|
+
)
|
|
1091
|
+
}
|
|
1092
|
+
if (mounted.has(path)) {
|
|
1093
|
+
throw new Error(
|
|
1094
|
+
`writableRootfsPaths names ${path}, which this layout already mounts. Docker refuses two mounts at one destination ("Duplicate mount point"), so which one won would be decided by argument order rather than by anyone's intent. Drop the entry, or change the layout's own mount to the mode you want.`,
|
|
1095
|
+
)
|
|
1096
|
+
}
|
|
1097
|
+
}
|
|
1098
|
+
|
|
1099
|
+
// The set collapses a host entry that repeats a default, which would
|
|
1100
|
+
// otherwise emit the same destination twice and be refused by docker.
|
|
1101
|
+
const paths = [
|
|
1102
|
+
...new Set([
|
|
1103
|
+
...DEFAULT_WRITABLE_ROOTFS_PATHS.filter((path) => !mounted.has(path)),
|
|
1104
|
+
...requested,
|
|
1105
|
+
]),
|
|
1106
|
+
]
|
|
1107
|
+
return paths.flatMap((path) => ['--tmpfs', `${path}:${TMPFS_MOUNT_OPTIONS}`])
|
|
1108
|
+
}
|
|
1109
|
+
|
|
1110
|
+
/**
|
|
1111
|
+
* The confinement preamble for one container, in argv order.
|
|
1112
|
+
*
|
|
1113
|
+
* A function rather than a bare constant because `--read-only` is switchable
|
|
1114
|
+
* and the flags that follow it describe what stays writable while it is on:
|
|
1115
|
+
* `readOnlyRootfs: false` removes both the flag and the mounts. That is the only
|
|
1116
|
+
* thing it removes. Everything in {@link HARDENING_ARGS} is applied
|
|
1117
|
+
* unconditionally and no field can turn one of those off, so the argv for
|
|
1118
|
+
* `readOnlyRootfs: false` is the argv this backend produced before any of this
|
|
1119
|
+
* existed PLUS `--ipc private` — those two flags are the whole previous argv,
|
|
1120
|
+
* and `--ipc private` is now unconditional. `--ipc` is not folded under this
|
|
1121
|
+
* switch, because the field names the root filesystem: a host that turned the
|
|
1122
|
+
* read-only rootfs off would be turning IPC isolation off as well, silently,
|
|
1123
|
+
* for a reason the name of the field does not say. A switch has to mean one
|
|
1124
|
+
* thing.
|
|
1125
|
+
*/
|
|
1126
|
+
export function renderHardeningArgs(config: DockerHardeningConfig): string[] {
|
|
1127
|
+
return [
|
|
1128
|
+
...HARDENING_ARGS,
|
|
1129
|
+
...(config.readOnlyRootfs === false ? [] : ['--read-only']),
|
|
1130
|
+
...renderWritableRootfsArgs(config),
|
|
1131
|
+
]
|
|
1132
|
+
}
|
|
1133
|
+
|
|
1134
|
+
/**
|
|
1135
|
+
* Everything {@link buildDockerRunArgs} renders, as a value.
|
|
1136
|
+
*
|
|
1137
|
+
* The pieces that come from the daemon or from the host are inputs rather than
|
|
1138
|
+
* lookups: which network the container attaches to, and whether an egress proxy
|
|
1139
|
+
* is listening and on which port. Both are already resolved by the caller, and
|
|
1140
|
+
* reading them here would put a daemon call back inside the function whose
|
|
1141
|
+
* whole point is that it needs none.
|
|
1142
|
+
*/
|
|
1143
|
+
export interface DockerRunArgvInput {
|
|
1144
|
+
readonly config: DockerBackendInternalConfig
|
|
1145
|
+
readonly options: SandboxBackendOptions
|
|
1146
|
+
readonly containerName: string
|
|
1147
|
+
readonly network: string
|
|
1148
|
+
readonly hostReachability: 'host-port' | 'container-network'
|
|
1149
|
+
/**
|
|
1150
|
+
* Port the egress proxy listens on, when one is running. Absent means no
|
|
1151
|
+
* proxy, and no proxy environment is passed in — which is not the same
|
|
1152
|
+
* fact as a proxy that was configured and is unreachable.
|
|
1153
|
+
*
|
|
1154
|
+
* It is the proxy CONTAINER's port, on the internal network the two
|
|
1155
|
+
* containers share; nothing is published on the host any more.
|
|
1156
|
+
*/
|
|
1157
|
+
readonly egressProxyPort?: number
|
|
1158
|
+
}
|
|
1159
|
+
|
|
1160
|
+
/**
|
|
1161
|
+
* The complete `docker run` argv, as a value.
|
|
1162
|
+
*
|
|
1163
|
+
* Extracted for the same reason {@link resolveNetwork} and
|
|
1164
|
+
* {@link egressProxyContainerConfig} were: everything downstream of it needs a running
|
|
1165
|
+
* Docker daemon, so a confinement flag that never reached the argv — or one
|
|
1166
|
+
* that reached it in an order that cancels another — could only be caught by an
|
|
1167
|
+
* operator noticing its effect missing in production. Spawning a fake `docker`
|
|
1168
|
+
* and reading back what it was handed proves what the fake was told and nothing
|
|
1169
|
+
* about the container the daemon would build. Here the whole baseline is one
|
|
1170
|
+
* array, and an edit that drops a flag fails a test rather than a deployment.
|
|
1171
|
+
*
|
|
1172
|
+
* Order matters in exactly two places, and both are asserted by the test that
|
|
1173
|
+
* pins this: the image is the last argument, because everything after it is a
|
|
1174
|
+
* command for the container rather than a flag for docker; and every flag that
|
|
1175
|
+
* takes a value is pushed as two argv entries rather than one string, so no
|
|
1176
|
+
* value is ever re-split by anything downstream.
|
|
1177
|
+
*/
|
|
1178
|
+
export function buildDockerRunArgs(input: DockerRunArgvInput): string[] {
|
|
1179
|
+
const { config, options, containerName, network, hostReachability, egressProxyPort } = input
|
|
1180
|
+
const layout = config.layout
|
|
1181
|
+
assertRootfsOptionsAreCoherent(config)
|
|
1182
|
+
assertCpuLimitIsRenderable(config.cpuLimit)
|
|
1183
|
+
|
|
1184
|
+
const args: string[] = [
|
|
1185
|
+
'run',
|
|
1186
|
+
'--detach',
|
|
1187
|
+
'--rm',
|
|
1188
|
+
'--name',
|
|
1189
|
+
containerName,
|
|
1190
|
+
'--network',
|
|
1191
|
+
network,
|
|
1192
|
+
...renderHardeningArgs(config),
|
|
1193
|
+
]
|
|
1194
|
+
if (config.runAsUser) {
|
|
1195
|
+
args.push('--user', config.runAsUser)
|
|
1196
|
+
}
|
|
1197
|
+
|
|
1198
|
+
args.push(...renderLabelArgs(config.labels))
|
|
1199
|
+
|
|
1200
|
+
args.push(...renderLayoutMountArgs(layout))
|
|
1201
|
+
// Forward only the workspace root so the worker's lexical
|
|
1202
|
+
// resolver agrees with the bind target. The full layout used
|
|
1203
|
+
// to ride along as `NAMZU_SANDBOX_LAYOUT`, but the worker
|
|
1204
|
+
// never branched on it; the manifest's only consumer was a
|
|
1205
|
+
// log line. A skill loader that needs the manifest will
|
|
1206
|
+
// write it to a bind path the worker reads at startup —
|
|
1207
|
+
// avoids env-size limits, keeps the wire shape minimal.
|
|
1208
|
+
if (egressProxyPort !== undefined) {
|
|
1209
|
+
// No `--add-host`, and no `host-gateway`. The proxy is a container on
|
|
1210
|
+
// this container's own internal network and `PROXY_HOST_ALIAS` is its
|
|
1211
|
+
// network alias there (see {@link renderEgressProxyAttachArgs}), which
|
|
1212
|
+
// docker's embedded DNS resolves with no alias file involved. The alias
|
|
1213
|
+
// entry this replaced pointed at the host's loopback, where the proxy
|
|
1214
|
+
// used to run — and a name resolving to a host that is not on this
|
|
1215
|
+
// network is exactly the route this change removes.
|
|
1216
|
+
const proxyUrl = `http://${PROXY_HOST_ALIAS}:${egressProxyPort}`
|
|
1217
|
+
// Both spellings: tooling is split between them, and a workload that
|
|
1218
|
+
// reads only the one that is missing loses the redirection.
|
|
1219
|
+
//
|
|
1220
|
+
// **These are convenience, not the boundary, and that is the point of
|
|
1221
|
+
// the arrangement.** A tool that honours them sends its requests to the
|
|
1222
|
+
// proxy; a tool that ignores them — a Go or Rust binary that does not
|
|
1223
|
+
// read proxy env, `curl --noproxy '*'`, a raw socket — has nowhere to
|
|
1224
|
+
// send anything. This container is on an `--internal` network with no
|
|
1225
|
+
// default route, and the only host on it is the proxy, so the
|
|
1226
|
+
// uncooperative path fails with `Network unreachable` rather than
|
|
1227
|
+
// bypassing the allowlist. What makes that true is
|
|
1228
|
+
// {@link assertNetworkCarriesThePolicy} refusing a network that is not
|
|
1229
|
+
// internal, not this block of environment.
|
|
1230
|
+
for (const key of ['HTTP_PROXY', 'http_proxy', 'HTTPS_PROXY', 'https_proxy']) {
|
|
1231
|
+
args.push('--env', `${key}=${proxyUrl}`)
|
|
1232
|
+
}
|
|
1233
|
+
// Loopback must not be proxied, or the worker cannot talk to
|
|
1234
|
+
// itself.
|
|
1235
|
+
args.push('--env', 'NO_PROXY=localhost,127.0.0.1')
|
|
1236
|
+
args.push('--env', 'no_proxy=localhost,127.0.0.1')
|
|
1237
|
+
}
|
|
1238
|
+
|
|
1239
|
+
// `outputs` is required by validation, so its containerPath is always
|
|
1240
|
+
// available — the worker uses it as its workspace root.
|
|
1241
|
+
args.push('--env', `NAMZU_SANDBOX_WORKSPACE=${layout.outputs.containerPath}`)
|
|
1242
|
+
args.push('--env', `NAMZU_SANDBOX_READ_ROOTS=${renderLayoutReadRootsEnv(layout)}`)
|
|
1243
|
+
args.push('--env', `NAMZU_SANDBOX_WRITE_ROOTS=${renderLayoutWriteRootsEnv(layout)}`)
|
|
1244
|
+
|
|
1245
|
+
// Only publish a host port when the consumer is going to reach
|
|
1246
|
+
// the worker through the docker host's loopback (CLI / direct
|
|
1247
|
+
// dev). For `container-network` reachability we leave the port
|
|
1248
|
+
// unpublished — sibling containers reach the worker by its DNS
|
|
1249
|
+
// name on the shared bridge, no host port required.
|
|
1250
|
+
//
|
|
1251
|
+
// Let Docker pick the host port instead of pre-reserving one
|
|
1252
|
+
// in this process. The reservePort()-then-publish-fixed-port
|
|
1253
|
+
// pattern had a TOCTOU window: the OS could hand the port to
|
|
1254
|
+
// another process between our `server.close()` and Docker's
|
|
1255
|
+
// `bind()`. Letting Docker pick (`--publish-all`) and reading
|
|
1256
|
+
// the mapping back via `docker inspect` removes the race.
|
|
1257
|
+
if (hostReachability === 'host-port') {
|
|
1258
|
+
args.push('--publish', `127.0.0.1::${WORKER_PORT_INSIDE_CONTAINER}`)
|
|
1259
|
+
}
|
|
1260
|
+
|
|
1261
|
+
if (config.runtime) {
|
|
1262
|
+
args.push('--runtime', config.runtime)
|
|
1263
|
+
}
|
|
1264
|
+
|
|
1265
|
+
// The three bounds the host can set, together and in one order, so a
|
|
1266
|
+
// reader of a `docker inspect` sees them side by side. `--memory` and
|
|
1267
|
+
// `--pids-limit` keep their existing treatment (a non-positive or absent
|
|
1268
|
+
// value means "not set"); `--cpus` refuses a value that would mean
|
|
1269
|
+
// something else, which is why it is the one with a check in front of it.
|
|
1270
|
+
if (options.memoryLimitMb && options.memoryLimitMb > 0) {
|
|
1271
|
+
args.push('--memory', `${options.memoryLimitMb}m`)
|
|
1272
|
+
}
|
|
1273
|
+
if (options.maxProcesses && options.maxProcesses > 0) {
|
|
1274
|
+
args.push('--pids-limit', String(options.maxProcesses))
|
|
1275
|
+
}
|
|
1276
|
+
if (config.cpuLimit !== undefined) {
|
|
1277
|
+
args.push('--cpus', String(config.cpuLimit))
|
|
1278
|
+
}
|
|
1279
|
+
|
|
1280
|
+
for (const [key, value] of Object.entries(options.env ?? {})) {
|
|
1281
|
+
args.push('--env', `${key}=${value}`)
|
|
1282
|
+
}
|
|
1283
|
+
|
|
1284
|
+
// The worker's credential, rendered VALUELESS and last among the `--env`
|
|
1285
|
+
// flags.
|
|
1286
|
+
//
|
|
1287
|
+
// Valueless, because `docker run --env NAME` reads the value out of the
|
|
1288
|
+
// docker CLI's own environment — which `runOnce` is handed — and an argv
|
|
1289
|
+
// is the wrong place for a secret: `ps` shows it to every user on the
|
|
1290
|
+
// host for as long as the CLI lives, and a non-zero run puts the whole
|
|
1291
|
+
// argv into the error this backend throws. The CLI environment is
|
|
1292
|
+
// readable only by the same user and root, and the message is redacted
|
|
1293
|
+
// besides.
|
|
1294
|
+
//
|
|
1295
|
+
// Last, because docker applies repeated `--env` flags in order and the
|
|
1296
|
+
// last one wins: a host that separately sets `NAMZU_SANDBOX_TOKEN` in
|
|
1297
|
+
// `options.env` — a copied example, an inherited environment — must not
|
|
1298
|
+
// be able to displace the value its own client is sending, which would
|
|
1299
|
+
// produce a container that rejects every call and reads as a broken
|
|
1300
|
+
// worker rather than as a duplicated setting.
|
|
1301
|
+
args.push('--env', 'NAMZU_SANDBOX_TOKEN')
|
|
1302
|
+
|
|
1303
|
+
args.push(config.image)
|
|
1304
|
+
return args
|
|
1305
|
+
}
|
|
1306
|
+
|
|
399
1307
|
async function spawnDockerSandbox(
|
|
400
1308
|
config: DockerBackendInternalConfig,
|
|
401
1309
|
options: SandboxBackendOptions,
|
|
@@ -406,51 +1314,61 @@ async function spawnDockerSandbox(
|
|
|
406
1314
|
const id = generateSandboxId()
|
|
407
1315
|
const docker = config.dockerBinary ?? DEFAULT_DOCKER_BINARY
|
|
408
1316
|
|
|
409
|
-
// The
|
|
410
|
-
//
|
|
411
|
-
//
|
|
412
|
-
//
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
1317
|
+
// The worker's per-instance credential, minted HERE — per `create()`, not
|
|
1318
|
+
// per process and never per image. Three properties are the point, and
|
|
1319
|
+
// each one rules out a cheaper shape:
|
|
1320
|
+
//
|
|
1321
|
+
// - Per instance. A token baked into the image is shared by every
|
|
1322
|
+
// container ever built from it and readable by anything that can pull
|
|
1323
|
+
// it, which is a worse artifact than a documented absence: it looks
|
|
1324
|
+
// like a credential while separating nobody.
|
|
1325
|
+
// - Not in an argv, and not in any message. It rides in the docker CLI
|
|
1326
|
+
// child's environment, which `runOnce` is handed, and the CLI resolves
|
|
1327
|
+
// it there for the valueless `--env NAMZU_SANDBOX_TOKEN`; a non-zero
|
|
1328
|
+
// `docker run` renders its argv with every `--env` value redacted. So
|
|
1329
|
+
// `ps` on the host does not show it and the error this backend throws
|
|
1330
|
+
// does not carry it — neither of which was true when the value was
|
|
1331
|
+
// rendered into the argv.
|
|
1332
|
+
// - Where it IS visible, said plainly: the container's own config, so
|
|
1333
|
+
// `docker inspect <name>` shows it for the container's life, to anyone
|
|
1334
|
+
// who can already talk to the daemon — the same authority that can
|
|
1335
|
+
// `docker exec` into the sandbox. And the worker's own `/proc` inside
|
|
1336
|
+
// the container, to a workload that shares its uid. Both are why it is
|
|
1337
|
+
// per-instance and dies with the container rather than being shared or
|
|
1338
|
+
// long-lived.
|
|
1339
|
+
// - Dead with the container. Nothing revokes it, because the only process
|
|
1340
|
+
// that would accept it is removed with the sandbox, and the container's
|
|
1341
|
+
// config that still holds it is removed with it.
|
|
1342
|
+
//
|
|
1343
|
+
// 32 bytes rather than a uuid: this is a secret, not an identifier, and
|
|
1344
|
+
// base64url keeps it one argv-free environment value on every platform.
|
|
1345
|
+
const workerToken = randomBytes(32).toString('base64url')
|
|
424
1346
|
|
|
425
1347
|
const hostReachability = config.hostReachability ?? 'host-port'
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
1348
|
+
// Whether a proxy CONTAINER will run, which is what `resolveNetwork` turns
|
|
1349
|
+
// on: an allowlist policy needs the image to run one as, and without it
|
|
1350
|
+
// there is no boundary to hand traffic to. The refusal lives in
|
|
1351
|
+
// `resolveNetwork` so that every caller of it gets the same answer, and it
|
|
1352
|
+
// fires before anything is started.
|
|
1353
|
+
const proxyPlanned = needsEgressProxy(options.egress) && config.egressProxyImage !== undefined
|
|
1354
|
+
const network = resolveNetwork(config.network ?? 'none', options.egress, proxyPlanned)
|
|
431
1355
|
// Whether this network can carry the reachability mode and the policy is
|
|
432
1356
|
// a fact about the network, so it is checked against the daemon rather
|
|
433
|
-
// than inferred from its name. Before
|
|
1357
|
+
// than inferred from its name. Before anything starts on purpose: a
|
|
434
1358
|
// refusal here is a wiring mistake and must not arrive dressed as a
|
|
435
1359
|
// container that failed to come up, which is exactly how it used to
|
|
436
1360
|
// arrive.
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
)
|
|
444
|
-
} catch (err) {
|
|
445
|
-
// The allowlist kinds start a proxy above, and this is outside the
|
|
446
|
-
// try/catch that owns teardown — so without this the refusal would
|
|
447
|
-
// leave a listening server on loopback stamping real credentials.
|
|
448
|
-
await egressProxy?.close().catch(() => undefined)
|
|
449
|
-
throw err
|
|
450
|
-
}
|
|
451
|
-
const runtime = config.runtime
|
|
1361
|
+
assertNetworkCarriesThePolicy(
|
|
1362
|
+
network,
|
|
1363
|
+
hostReachability,
|
|
1364
|
+
options.egress,
|
|
1365
|
+
await inspectNetworkInternalFlag(docker, network, options.signal),
|
|
1366
|
+
)
|
|
452
1367
|
const containerName = `namzu-sandbox-${id}`
|
|
453
1368
|
|
|
1369
|
+
/** The proxy container's name, once it is being started. */
|
|
1370
|
+
let egressProxyContainer: string | undefined
|
|
1371
|
+
|
|
454
1372
|
// All bind sources come from the consumer-supplied layout. The
|
|
455
1373
|
// backend never allocates host directories and never removes them
|
|
456
1374
|
// — that pre-existing single-mount mkdtemp path was the source of
|
|
@@ -464,20 +1382,32 @@ async function spawnDockerSandbox(
|
|
|
464
1382
|
// reconciliation; an external daemon that commits after this delete still
|
|
465
1383
|
// needs its ordinary label/name reaper.
|
|
466
1384
|
const removeContainer = runOnceQuiet(docker, ['rm', '-f', containerName], signal)
|
|
467
|
-
// The proxy starts BEFORE the
|
|
468
|
-
// in `destroy()`, which a create that never returned can
|
|
469
|
-
// reach. So every failure between the two — a daemon that is
|
|
470
|
-
// port that could not be read, a worker that missed its
|
|
471
|
-
// deadline,
|
|
472
|
-
//
|
|
473
|
-
//
|
|
474
|
-
//
|
|
475
|
-
//
|
|
476
|
-
//
|
|
477
|
-
//
|
|
478
|
-
|
|
479
|
-
|
|
480
|
-
|
|
1385
|
+
// The proxy container starts BEFORE the sandbox and its only other
|
|
1386
|
+
// teardown is in `destroy()`, which a create that never returned can
|
|
1387
|
+
// never reach. So every failure between the two — a daemon that is
|
|
1388
|
+
// down, a port that could not be read, a worker that missed its
|
|
1389
|
+
// readiness deadline, an abort — would leave a container holding real
|
|
1390
|
+
// credentials and a live route to the internet, and a retry loop left
|
|
1391
|
+
// one per attempt. That is the invariant this file states where the
|
|
1392
|
+
// proxy is started: it must not outlive the thing it was filtering
|
|
1393
|
+
// for. Start both arms before awaiting either: a stuck runtime must
|
|
1394
|
+
// not keep the credential-bearing one alive.
|
|
1395
|
+
//
|
|
1396
|
+
// The proxy arm is the reconciler rather than a `runOnceQuiet` on this
|
|
1397
|
+
// signal, and the difference is the grace this function's caller runs
|
|
1398
|
+
// under: `runFailureCleanup` spends ONE SECOND on both arms together,
|
|
1399
|
+
// so a daemon that is slow to answer rather than down would have its
|
|
1400
|
+
// `docker rm -f` killed at that boundary — the credential-bearing
|
|
1401
|
+
// container outliving the sandbox by exactly the failure this exists to
|
|
1402
|
+
// prevent. The reconciler carries its own deadline, is not reachable by
|
|
1403
|
+
// any abort, and is not waited on by the caller's grace either: it
|
|
1404
|
+
// finishes whether or not this function is still listening.
|
|
1405
|
+
const removeProxy =
|
|
1406
|
+
egressProxyContainer === undefined
|
|
1407
|
+
? Promise.resolve()
|
|
1408
|
+
: removeEgressProxyContainer(docker, egressProxyContainer)
|
|
1409
|
+
egressProxyContainer = undefined
|
|
1410
|
+
await Promise.all([removeContainer, removeProxy])
|
|
481
1411
|
}
|
|
482
1412
|
|
|
483
1413
|
let hostPort: number
|
|
@@ -487,99 +1417,53 @@ async function spawnDockerSandbox(
|
|
|
487
1417
|
const rootDir = resolvedLayout.outputs.containerPath
|
|
488
1418
|
|
|
489
1419
|
try {
|
|
490
|
-
//
|
|
491
|
-
//
|
|
492
|
-
//
|
|
493
|
-
//
|
|
494
|
-
//
|
|
495
|
-
//
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
//
|
|
509
|
-
//
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
|
|
524
|
-
|
|
525
|
-
args.push(...renderLayoutMountArgs(resolvedLayout))
|
|
526
|
-
// Forward only the workspace root so the worker's lexical
|
|
527
|
-
// resolver agrees with the bind target. The full layout used
|
|
528
|
-
// to ride along as `NAMZU_SANDBOX_LAYOUT`, but the worker
|
|
529
|
-
// never branched on it; the manifest's only consumer was a
|
|
530
|
-
// log line. A skill loader that needs the manifest will
|
|
531
|
-
// write it to a bind path the worker reads at startup —
|
|
532
|
-
// avoids env-size limits, keeps the wire shape minimal.
|
|
533
|
-
if (egressProxy) {
|
|
534
|
-
// `host-gateway` is docker's own portable name for the host from
|
|
535
|
-
// inside a container; hard-coding a bridge address would break on
|
|
536
|
-
// every platform whose bridge is numbered differently. The proxy
|
|
537
|
-
// itself binds loopback, so this alias is the only way in.
|
|
538
|
-
args.push('--add-host', `${PROXY_HOST_ALIAS}:host-gateway`)
|
|
539
|
-
const proxyUrl = `http://${PROXY_HOST_ALIAS}:${egressProxy.port}`
|
|
540
|
-
// Both spellings: tooling is split between them, and a workload
|
|
541
|
-
// that reads only the one that is missing bypasses the boundary
|
|
542
|
-
// entirely — which would look exactly like the policy working.
|
|
543
|
-
for (const key of ['HTTP_PROXY', 'http_proxy', 'HTTPS_PROXY', 'https_proxy']) {
|
|
544
|
-
args.push('--env', `${key}=${proxyUrl}`)
|
|
545
|
-
}
|
|
546
|
-
// Loopback must not be proxied, or the worker cannot talk to
|
|
547
|
-
// itself.
|
|
548
|
-
args.push('--env', 'NO_PROXY=localhost,127.0.0.1')
|
|
549
|
-
args.push('--env', 'no_proxy=localhost,127.0.0.1')
|
|
550
|
-
}
|
|
551
|
-
|
|
552
|
-
args.push('--env', `NAMZU_SANDBOX_WORKSPACE=${rootDir}`)
|
|
553
|
-
args.push('--env', `NAMZU_SANDBOX_READ_ROOTS=${renderLayoutReadRootsEnv(resolvedLayout)}`)
|
|
554
|
-
args.push('--env', `NAMZU_SANDBOX_WRITE_ROOTS=${renderLayoutWriteRootsEnv(resolvedLayout)}`)
|
|
555
|
-
|
|
556
|
-
// Only publish a host port when the consumer is going to reach
|
|
557
|
-
// the worker through the docker host's loopback (CLI / direct
|
|
558
|
-
// dev). For `container-network` reachability we leave the port
|
|
559
|
-
// unpublished — sibling containers reach the worker by its DNS
|
|
560
|
-
// name on the shared bridge, no host port required.
|
|
561
|
-
if (hostReachability === 'host-port') {
|
|
562
|
-
args.push('--publish', `127.0.0.1::${WORKER_PORT_INSIDE_CONTAINER}`)
|
|
563
|
-
}
|
|
564
|
-
|
|
565
|
-
if (runtime) {
|
|
566
|
-
args.push('--runtime', runtime)
|
|
567
|
-
}
|
|
568
|
-
|
|
569
|
-
if (options.memoryLimitMb && options.memoryLimitMb > 0) {
|
|
570
|
-
args.push('--memory', `${options.memoryLimitMb}m`)
|
|
571
|
-
}
|
|
572
|
-
if (options.maxProcesses && options.maxProcesses > 0) {
|
|
573
|
-
args.push('--pids-limit', String(options.maxProcesses))
|
|
574
|
-
}
|
|
575
|
-
|
|
576
|
-
for (const [key, value] of Object.entries(options.env ?? {})) {
|
|
577
|
-
args.push('--env', `${key}=${value}`)
|
|
1420
|
+
// The boundary a host allowlist is actually enforced at, as a sibling
|
|
1421
|
+
// container on this container's network (#398). Started before the
|
|
1422
|
+
// sandbox so its alias is in the network's DNS by the time the sandbox's
|
|
1423
|
+
// proxy environment resolves it, and removed with the sandbox — a proxy
|
|
1424
|
+
// holding real credentials must not outlive the thing it was filtering
|
|
1425
|
+
// for, and on this topology a leftover one is also a live route to the
|
|
1426
|
+
// internet for anything that can reach that network.
|
|
1427
|
+
//
|
|
1428
|
+
// Inside this `try` on purpose, which the first cut of this change got
|
|
1429
|
+
// wrong. This block has three suspension points — the resolver, the
|
|
1430
|
+
// proxy's `docker run`, and an explicit `throwIfAborted()` — and an
|
|
1431
|
+
// abort at any of them leaves a container holding real credentials with
|
|
1432
|
+
// nothing that will ever remove it: `destroy()` is unreachable, because
|
|
1433
|
+
// `create()` never returned a handle. Outside the `try` the abort
|
|
1434
|
+
// rethrew past `cleanupOnFailure`; inside it, every path that does not
|
|
1435
|
+
// return a `Sandbox` goes through that cleanup, which removes the proxy
|
|
1436
|
+
// by name on a deadline of its own. The abort that arrives here has to
|
|
1437
|
+
// be raised by a call inside the block — the sandbox's own `docker run`
|
|
1438
|
+
// does that itself — so an abort before this point reaches the same
|
|
1439
|
+
// place by the same route.
|
|
1440
|
+
if (proxyPlanned && options.egress) {
|
|
1441
|
+
egressProxyContainer = egressProxyContainerName(id)
|
|
1442
|
+
// Resolved once, here, rather than per request — see
|
|
1443
|
+
// `egressProxyContainerConfig` for what that costs and why it is paid.
|
|
1444
|
+
const allowedHosts = await resolveAllowedHosts(options.egress)
|
|
1445
|
+
options.signal?.throwIfAborted()
|
|
1446
|
+
await startEgressProxyContainer({
|
|
1447
|
+
docker,
|
|
1448
|
+
config,
|
|
1449
|
+
containerName: egressProxyContainer,
|
|
1450
|
+
internalNetwork: network,
|
|
1451
|
+
allowedHosts,
|
|
1452
|
+
signal: options.signal,
|
|
1453
|
+
})
|
|
1454
|
+
options.signal?.throwIfAborted()
|
|
578
1455
|
}
|
|
579
1456
|
|
|
580
|
-
args
|
|
1457
|
+
const args = buildDockerRunArgs({
|
|
1458
|
+
config,
|
|
1459
|
+
options,
|
|
1460
|
+
containerName,
|
|
1461
|
+
network,
|
|
1462
|
+
hostReachability,
|
|
1463
|
+
...(egressProxyContainer ? { egressProxyPort: EGRESS_PROXY_PORT_INSIDE_CONTAINER } : {}),
|
|
1464
|
+
})
|
|
581
1465
|
|
|
582
|
-
await runOnce(docker, args, options.signal)
|
|
1466
|
+
await runOnce(docker, args, options.signal, { NAMZU_SANDBOX_TOKEN: workerToken })
|
|
583
1467
|
if (hostReachability === 'host-port') {
|
|
584
1468
|
hostPort = await readMappedPort(docker, containerName, options.signal)
|
|
585
1469
|
baseUrl = `http://127.0.0.1:${hostPort}`
|
|
@@ -612,7 +1496,7 @@ async function spawnDockerSandbox(
|
|
|
612
1496
|
let retirementPromise: Promise<{ readonly accepted: boolean; readonly error?: Error }> | undefined
|
|
613
1497
|
let teardownPromise: Promise<void> | undefined
|
|
614
1498
|
let teardownComplete = false
|
|
615
|
-
const workerClient = new HttpWorkerClient(baseUrl)
|
|
1499
|
+
const workerClient = new HttpWorkerClient(baseUrl, workerToken)
|
|
616
1500
|
const assertActive = (): void => {
|
|
617
1501
|
if (lifecycle !== 'active') {
|
|
618
1502
|
throw new Error(`Sandbox ${id} is ${lifecycle}; no new worker operation can be admitted`)
|
|
@@ -629,10 +1513,17 @@ async function spawnDockerSandbox(
|
|
|
629
1513
|
} catch (error) {
|
|
630
1514
|
teardownError = error
|
|
631
1515
|
} finally {
|
|
632
|
-
|
|
633
|
-
|
|
634
|
-
|
|
635
|
-
|
|
1516
|
+
// The proxy goes with the sandbox, for the reason it is
|
|
1517
|
+
// started before it: a container holding real credentials must
|
|
1518
|
+
// not outlive the thing it was filtering for, and on this
|
|
1519
|
+
// topology leaving it running would also leave a live route to
|
|
1520
|
+
// the internet on a network the sandbox was confined to.
|
|
1521
|
+
if (egressProxyContainer !== undefined) {
|
|
1522
|
+
try {
|
|
1523
|
+
await runOnce(docker, ['rm', '-f', egressProxyContainer], signal)
|
|
1524
|
+
} catch (error) {
|
|
1525
|
+
teardownError ??= error
|
|
1526
|
+
}
|
|
636
1527
|
}
|
|
637
1528
|
}
|
|
638
1529
|
if (teardownError !== undefined) throw teardownError
|
|
@@ -716,15 +1607,81 @@ async function spawnDockerSandbox(
|
|
|
716
1607
|
// here and doing nothing would leave the caller believing the
|
|
717
1608
|
// sandbox had been confined when it had not. Same rule the
|
|
718
1609
|
// egress-kind refusal above follows.
|
|
719
|
-
if (
|
|
1610
|
+
if (egressProxyContainer === undefined) {
|
|
720
1611
|
throw withHint(
|
|
721
1612
|
new Error(
|
|
722
1613
|
'This sandbox cannot change its network policy: it was created without an egress proxy, so its network was fixed at creation and there is nothing to narrow. Refusing rather than accepting a policy that would not be applied.',
|
|
723
1614
|
),
|
|
724
|
-
'Construct the provider with an egress proxy to make the policy mutable, or create a second sandbox under the narrower policy.',
|
|
1615
|
+
'Construct the provider with an egress proxy (and an egressProxyImage to run it as) to make the policy mutable, or create a second sandbox under the narrower policy.',
|
|
725
1616
|
)
|
|
726
1617
|
}
|
|
727
|
-
|
|
1618
|
+
// The policy lives in the proxy container's environment, which is
|
|
1619
|
+
// written when the container starts and cannot be rewritten from
|
|
1620
|
+
// outside. So a live policy change REPLACES the container: the
|
|
1621
|
+
// allowlist the caller asked for is the allowlist the next
|
|
1622
|
+
// connection meets, and the connections open at that instant fail
|
|
1623
|
+
// closed. That is a real difference from the in-process proxy this
|
|
1624
|
+
// used to be — it swapped the allowlist in place, with no window in
|
|
1625
|
+
// which the proxy was absent — and it is the honest trade for a
|
|
1626
|
+
// boundary the sandbox cannot route around. See
|
|
1627
|
+
// `docs/sdk/sandbox-egress.md`.
|
|
1628
|
+
//
|
|
1629
|
+
// The name is read out of the closure once, here, rather than at
|
|
1630
|
+
// each use below: `cleanupOnFailure` clears it, and a swap that
|
|
1631
|
+
// raced a failing create would otherwise call `assertActive()` and
|
|
1632
|
+
// then name `undefined` in the removal.
|
|
1633
|
+
const proxyContainer = egressProxyContainer
|
|
1634
|
+
try {
|
|
1635
|
+
await restartEgressProxyContainer({
|
|
1636
|
+
docker,
|
|
1637
|
+
config,
|
|
1638
|
+
containerName: proxyContainer,
|
|
1639
|
+
internalNetwork: network,
|
|
1640
|
+
allowedHosts: policy.allowedHosts,
|
|
1641
|
+
})
|
|
1642
|
+
// A teardown that landed while the replacement was starting
|
|
1643
|
+
// leaves a container nothing else will ever remove: `destroy()`
|
|
1644
|
+
// is running or has run, `egressProxyContainer` is only removed
|
|
1645
|
+
// from `teardownSandbox` and `cleanupOnFailure`, and the
|
|
1646
|
+
// `assertActive()` above ran BEFORE the first `await` — before
|
|
1647
|
+
// the swap suspended inside `restartEgressProxyContainer`, which
|
|
1648
|
+
// is exactly when a teardown lands. So the swap fails here
|
|
1649
|
+
// rather than reporting a policy change on a sandbox that no
|
|
1650
|
+
// longer exists.
|
|
1651
|
+
assertActive()
|
|
1652
|
+
} finally {
|
|
1653
|
+
// ... and the replacement is removed even so, because the check
|
|
1654
|
+
// above cannot be where the guarantee lives. It sits AFTER the
|
|
1655
|
+
// container comes into existence, and that ordering is what
|
|
1656
|
+
// makes it airtight rather than merely narrower than the check
|
|
1657
|
+
// at entry. A generation counter read before the `docker run`
|
|
1658
|
+
// has the opposite shape: it can only refuse a start it already
|
|
1659
|
+
// knows about, and a teardown that begins between that refusal
|
|
1660
|
+
// being evaluated and the daemon committing the container is in
|
|
1661
|
+
// no check's view — the container exists and nothing has looked
|
|
1662
|
+
// since. Here the two orderings partition the space instead. A
|
|
1663
|
+
// teardown that began before this point has already set
|
|
1664
|
+
// `lifecycle`, synchronously (`teardownSandbox` and `retire()`
|
|
1665
|
+
// both do, before their first `await`), so this removal runs. A
|
|
1666
|
+
// teardown that begins after it issues its own `rm -f` for this
|
|
1667
|
+
// same name — `teardownSandbox` reads `egressProxyContainer`,
|
|
1668
|
+
// which is this container — against a container that, at this
|
|
1669
|
+
// point, exists. Whichever of the two runs second finds the
|
|
1670
|
+
// container and removes it, and both are idempotent.
|
|
1671
|
+
//
|
|
1672
|
+
// A signal-less remover, and not `runOnce` on some signal, for
|
|
1673
|
+
// the reason `removeEgressProxyContainer` exists: the teardown
|
|
1674
|
+
// this is racing may be one whose own signal was already
|
|
1675
|
+
// aborted, and it removes nothing at all in that case (the whole
|
|
1676
|
+
// container set is left, which is what `Sandbox.destroy`
|
|
1677
|
+
// promises to settle promptly over). That is the caller's
|
|
1678
|
+
// contract and not something to defeat — but a proxy container
|
|
1679
|
+
// holding brokered credentials and a live route to the internet
|
|
1680
|
+
// is not something to leave with it either.
|
|
1681
|
+
if (lifecycle !== 'active') {
|
|
1682
|
+
await removeEgressProxyContainer(docker, proxyContainer)
|
|
1683
|
+
}
|
|
1684
|
+
}
|
|
728
1685
|
},
|
|
729
1686
|
|
|
730
1687
|
async writeFile(path: string, content: string | Buffer): Promise<void> {
|
|
@@ -734,7 +1691,10 @@ async function spawnDockerSandbox(
|
|
|
734
1691
|
try {
|
|
735
1692
|
res = await fetch(`${baseUrl}/write-file`, {
|
|
736
1693
|
method: 'POST',
|
|
737
|
-
headers: {
|
|
1694
|
+
headers: {
|
|
1695
|
+
'content-type': 'application/json',
|
|
1696
|
+
...workerAuthorization(workerToken),
|
|
1697
|
+
},
|
|
738
1698
|
body: JSON.stringify({
|
|
739
1699
|
path,
|
|
740
1700
|
content: buf.toString('base64'),
|
|
@@ -754,6 +1714,12 @@ async function spawnDockerSandbox(
|
|
|
754
1714
|
{ cause: err },
|
|
755
1715
|
)
|
|
756
1716
|
}
|
|
1717
|
+
if (res.status === 401) {
|
|
1718
|
+
throw withHint(
|
|
1719
|
+
new Error(`write-file failed: HTTP 401 ${await res.text()}`),
|
|
1720
|
+
WORKER_UNAUTHORIZED_HINT,
|
|
1721
|
+
)
|
|
1722
|
+
}
|
|
757
1723
|
if (!res.ok) {
|
|
758
1724
|
throw new Error(`write-file failed: HTTP ${res.status} ${await res.text()}`)
|
|
759
1725
|
}
|
|
@@ -782,10 +1748,19 @@ async function spawnDockerSandbox(
|
|
|
782
1748
|
}
|
|
783
1749
|
const res = await fetch(`${baseUrl}/read-file`, {
|
|
784
1750
|
method: 'POST',
|
|
785
|
-
headers: {
|
|
1751
|
+
headers: {
|
|
1752
|
+
'content-type': 'application/json',
|
|
1753
|
+
...workerAuthorization(workerToken),
|
|
1754
|
+
},
|
|
786
1755
|
body: JSON.stringify({ path, encoding: 'base64' }),
|
|
787
1756
|
signal: options?.signal,
|
|
788
1757
|
})
|
|
1758
|
+
if (res.status === 401) {
|
|
1759
|
+
throw withHint(
|
|
1760
|
+
new Error(`read-file failed: HTTP 401 ${await res.text()}`),
|
|
1761
|
+
WORKER_UNAUTHORIZED_HINT,
|
|
1762
|
+
)
|
|
1763
|
+
}
|
|
789
1764
|
if (!res.ok) {
|
|
790
1765
|
throw new Error(`read-file failed: HTTP ${res.status} ${await res.text()}`)
|
|
791
1766
|
}
|
|
@@ -840,6 +1815,211 @@ async function spawnDockerSandbox(
|
|
|
840
1815
|
}
|
|
841
1816
|
}
|
|
842
1817
|
|
|
1818
|
+
/** The proxy's upstream network when the host named none. See the field. */
|
|
1819
|
+
const DEFAULT_EGRESS_PROXY_UPSTREAM_NETWORK = 'bridge'
|
|
1820
|
+
|
|
1821
|
+
/**
|
|
1822
|
+
* How long a reconciliation remove of the proxy container may take.
|
|
1823
|
+
*
|
|
1824
|
+
* The same order as `retire()`'s own deadline, and for the same reason: this
|
|
1825
|
+
* bounds a call to a daemon that may never answer. It is a deadline of its own
|
|
1826
|
+
* rather than the caller's, because the caller is usually an aborted operation
|
|
1827
|
+
* — see {@link removeEgressProxyContainer}.
|
|
1828
|
+
*/
|
|
1829
|
+
const EGRESS_PROXY_REMOVE_TIMEOUT_MS = 5_000
|
|
1830
|
+
|
|
1831
|
+
/**
|
|
1832
|
+
* Remove the proxy container, on a path that is already failing.
|
|
1833
|
+
*
|
|
1834
|
+
* Three properties, each of which the first cut of this change got wrong, and
|
|
1835
|
+
* each of which is why this is a function rather than a `runOnceQuiet` call at
|
|
1836
|
+
* three sites:
|
|
1837
|
+
*
|
|
1838
|
+
* - **It never takes the caller's signal.** A reconciliation remove handed an
|
|
1839
|
+
* already-aborted signal does not reach the daemon at all: `runOnce` and
|
|
1840
|
+
* `runOnceQuiet` install an abort listener that kills the child the moment
|
|
1841
|
+
* they see one, so `docker rm -f` dies before it is spawned and the
|
|
1842
|
+
* container survives — in exactly the case this exists to handle, an
|
|
1843
|
+
* operation that was aborted midway through starting it.
|
|
1844
|
+
* - **It is bounded anyway.** A fresh deadline of its own, so a daemon that
|
|
1845
|
+
* never answers cannot hang the failure path that is cleaning up after it.
|
|
1846
|
+
* - **It does not throw.** Whatever went wrong to bring the caller here is the
|
|
1847
|
+
* thing the caller needs to hear; a failure to remove is reported by the
|
|
1848
|
+
* container's own name in `docker ps` and by the label reaper this file
|
|
1849
|
+
* documents, not by replacing that error with a worse one.
|
|
1850
|
+
*/
|
|
1851
|
+
async function removeEgressProxyContainer(docker: string, containerName: string): Promise<void> {
|
|
1852
|
+
const deadline = new OperationDeadline(
|
|
1853
|
+
EGRESS_PROXY_REMOVE_TIMEOUT_MS,
|
|
1854
|
+
`egress proxy container ${containerName} removal`,
|
|
1855
|
+
)
|
|
1856
|
+
try {
|
|
1857
|
+
await deadline.run((signal) => runOnceQuiet(docker, ['rm', '-f', containerName], signal))
|
|
1858
|
+
} catch {
|
|
1859
|
+
// Deliberately swallowed. See the third property above.
|
|
1860
|
+
}
|
|
1861
|
+
}
|
|
1862
|
+
|
|
1863
|
+
interface EgressProxyContainerInput {
|
|
1864
|
+
readonly docker: string
|
|
1865
|
+
readonly config: DockerBackendInternalConfig
|
|
1866
|
+
readonly containerName: string
|
|
1867
|
+
readonly internalNetwork: string
|
|
1868
|
+
/**
|
|
1869
|
+
* The hosts the proxy will permit, already resolved. Resolved by the
|
|
1870
|
+
* caller rather than passed as a policy because the two callers resolve
|
|
1871
|
+
* differently and the difference is the point: `create()` resolves a
|
|
1872
|
+
* policy once through {@link resolveAllowedHosts}, and `setNetworkPolicy`
|
|
1873
|
+
* is handed a list by definition (`SandboxNetworkPolicy`).
|
|
1874
|
+
*/
|
|
1875
|
+
readonly allowedHosts: readonly string[]
|
|
1876
|
+
readonly signal?: AbortSignal
|
|
1877
|
+
}
|
|
1878
|
+
|
|
1879
|
+
/**
|
|
1880
|
+
* Start the proxy container: run it, join it to the internal network, prove it
|
|
1881
|
+
* is up.
|
|
1882
|
+
*
|
|
1883
|
+
* Three steps and not one, in this order, for reasons that are all about the
|
|
1884
|
+
* topology rather than about docker's ergonomics. `docker run --network
|
|
1885
|
+
* <upstream>` gives the proxy its default route — the leg that reaches the
|
|
1886
|
+
* internet. `docker network connect --alias` adds the internal leg the sandbox
|
|
1887
|
+
* dials it on, and it has to be second: a container created on an internal
|
|
1888
|
+
* network first comes up with no default route and never gets one, which is a
|
|
1889
|
+
* proxy that can reach nothing. The readiness check is third because the first
|
|
1890
|
+
* two are `docker run` exiting 0, and `docker run --detach` exits 0 for a
|
|
1891
|
+
* container whose entrypoint is about to fail — an image without the compiled
|
|
1892
|
+
* module in it, or a config the entrypoint refused. Those failures arrive as a
|
|
1893
|
+
* sandbox whose every outbound request fails, which reads as the policy
|
|
1894
|
+
* working.
|
|
1895
|
+
*
|
|
1896
|
+
* A failure after the container exists removes it before rethrowing. Leaving
|
|
1897
|
+
* it would leave a container holding real credentials and a live route to the
|
|
1898
|
+
* internet with no sandbox it belongs to. This is not the only remover on that
|
|
1899
|
+
* path: the block that starts this container sits INSIDE the `try` that owns
|
|
1900
|
+
* `cleanupOnFailure`, and the name it is started under is in
|
|
1901
|
+
* `egressProxyContainer` from before the call, so the caller's cleanup removes
|
|
1902
|
+
* the same name again. The first cut of this change had the block outside that
|
|
1903
|
+
* `try` and this sentence said the caller's own cleanup could not see the
|
|
1904
|
+
* container; the two were wrong together, and the leak was real.
|
|
1905
|
+
*/
|
|
1906
|
+
async function startEgressProxyContainer(input: EgressProxyContainerInput): Promise<void> {
|
|
1907
|
+
const { docker, config, containerName, internalNetwork, allowedHosts, signal } = input
|
|
1908
|
+
|
|
1909
|
+
const argvInput: EgressProxyArgvInput = {
|
|
1910
|
+
config,
|
|
1911
|
+
containerName,
|
|
1912
|
+
upstreamNetwork: config.egressProxyUpstreamNetwork ?? DEFAULT_EGRESS_PROXY_UPSTREAM_NETWORK,
|
|
1913
|
+
internalNetwork,
|
|
1914
|
+
}
|
|
1915
|
+
// The policy, on its way to the container's environment by way of the
|
|
1916
|
+
// `docker` CLI's own. See `renderEgressProxyRunArgs` for why it does not
|
|
1917
|
+
// travel in the argv.
|
|
1918
|
+
const proxyEnvironment = {
|
|
1919
|
+
[EGRESS_PROXY_CONFIG_ENV]: JSON.stringify(
|
|
1920
|
+
egressProxyContainerConfig(config, allowedHosts, EGRESS_PROXY_PORT_INSIDE_CONTAINER),
|
|
1921
|
+
),
|
|
1922
|
+
}
|
|
1923
|
+
|
|
1924
|
+
try {
|
|
1925
|
+
await runOnce(docker, renderEgressProxyRunArgs(argvInput), signal, proxyEnvironment)
|
|
1926
|
+
} catch (error) {
|
|
1927
|
+
// The container may exist even though this call did not return
|
|
1928
|
+
// successfully: `docker run --detach` is a CLI process talking to a
|
|
1929
|
+
// daemon, and the daemon can commit the container while the client is
|
|
1930
|
+
// killed, times out, or fails to report the id back. Remove by name
|
|
1931
|
+
// before rethrowing, which is what the docblock above promised in the
|
|
1932
|
+
// first cut of this change and did not do.
|
|
1933
|
+
await removeEgressProxyContainer(docker, containerName)
|
|
1934
|
+
// An abort is not a failure to start, and the caller that aborted is
|
|
1935
|
+
// entitled to hear its own reason back — the same rule every other
|
|
1936
|
+
// catch in this file follows. Without this an acquisition timeout would
|
|
1937
|
+
// arrive dressed as an image-install problem, which is the diagnosis
|
|
1938
|
+
// shape this file refuses everywhere else.
|
|
1939
|
+
signal?.throwIfAborted()
|
|
1940
|
+
throw withHint(
|
|
1941
|
+
new Error(
|
|
1942
|
+
`Could not start the egress proxy container '${containerName}' from image '${config.egressProxyImage}': ${error instanceof Error ? error.message : String(error)}`,
|
|
1943
|
+
),
|
|
1944
|
+
'Build that image with `docker build -f packages/sandbox/egress-proxy/Dockerfile -t <tag> packages/sandbox` (after `pnpm --filter @namzu/sandbox build`), and name the tag in egressProxyImage. The sandbox is deliberately not started when the only way its traffic reaches the internet is missing.',
|
|
1945
|
+
)
|
|
1946
|
+
}
|
|
1947
|
+
|
|
1948
|
+
try {
|
|
1949
|
+
await runOnce(docker, renderEgressProxyAttachArgs(argvInput), signal)
|
|
1950
|
+
await assertEgressProxyContainerIsRunning(docker, containerName, signal)
|
|
1951
|
+
} catch (error) {
|
|
1952
|
+
await removeEgressProxyContainer(docker, containerName)
|
|
1953
|
+
throw error
|
|
1954
|
+
}
|
|
1955
|
+
}
|
|
1956
|
+
|
|
1957
|
+
/**
|
|
1958
|
+
* Replace the proxy container, for `setNetworkPolicy`.
|
|
1959
|
+
*
|
|
1960
|
+
* The policy the container enforces is in its environment, which is written
|
|
1961
|
+
* when it starts and is not writable from outside, so a live policy change is
|
|
1962
|
+
* a new container. The window between the two is one in which the sandbox has
|
|
1963
|
+
* no route out at all — the old proxy is gone and the new one is not up yet —
|
|
1964
|
+
* which fails CLOSED. That is the property worth naming: the alternative
|
|
1965
|
+
* ordering (start the second, then remove the first) has no such window, but
|
|
1966
|
+
* two containers cannot hold the same name or the same network alias, so it
|
|
1967
|
+
* would need a second name the sandbox's `HTTP_PROXY` does not know.
|
|
1968
|
+
*/
|
|
1969
|
+
async function restartEgressProxyContainer(
|
|
1970
|
+
input: Omit<EgressProxyContainerInput, 'signal'>,
|
|
1971
|
+
): Promise<void> {
|
|
1972
|
+
await removeEgressProxyContainer(input.docker, input.containerName)
|
|
1973
|
+
try {
|
|
1974
|
+
await startEgressProxyContainer(input)
|
|
1975
|
+
} catch (error) {
|
|
1976
|
+
// The hint names both states rather than assuming the safe one. The
|
|
1977
|
+
// removal above swallows its own failure, so "the old container is
|
|
1978
|
+
// gone" is not something this function knows: if the daemon never
|
|
1979
|
+
// answered the remove, the previous container is still up and still
|
|
1980
|
+
// enforcing the policy the caller just replaced — which would be a
|
|
1981
|
+
// worse thing to misreport than a sandbox with no route out.
|
|
1982
|
+
throw withHint(
|
|
1983
|
+
error instanceof Error ? error : new Error(String(error)),
|
|
1984
|
+
"Check `docker ps` for this sandbox's proxy container before relying on the new policy: the replacement did not come up, so this sandbox either has no route out at all (every request fails closed) or still has the previous container enforcing the policy you just replaced. Retry, or destroy the sandbox and create it under the policy you want.",
|
|
1985
|
+
)
|
|
1986
|
+
}
|
|
1987
|
+
}
|
|
1988
|
+
|
|
1989
|
+
/**
|
|
1990
|
+
* Whether the daemon says the proxy container is still up.
|
|
1991
|
+
*
|
|
1992
|
+
* Read the same way {@link inspectNetworkInternalFlag} reads the network's
|
|
1993
|
+
* `Internal` flag, and for the same reason: an unreadable answer is not
|
|
1994
|
+
* evidence. A container that has already exited is gone (`--rm` removed it),
|
|
1995
|
+
* so the failure arrives as a failed `docker inspect` rather than as a
|
|
1996
|
+
* `false`, and both have to land on the same refusal.
|
|
1997
|
+
*/
|
|
1998
|
+
async function assertEgressProxyContainerIsRunning(
|
|
1999
|
+
docker: string,
|
|
2000
|
+
containerName: string,
|
|
2001
|
+
signal?: AbortSignal,
|
|
2002
|
+
): Promise<void> {
|
|
2003
|
+
let running = ''
|
|
2004
|
+
try {
|
|
2005
|
+
running = await runOnce(
|
|
2006
|
+
docker,
|
|
2007
|
+
['inspect', '--format', '{{.State.Running}}', containerName],
|
|
2008
|
+
signal,
|
|
2009
|
+
)
|
|
2010
|
+
} catch {
|
|
2011
|
+
signal?.throwIfAborted()
|
|
2012
|
+
running = ''
|
|
2013
|
+
}
|
|
2014
|
+
if (running.trim() === 'true') return
|
|
2015
|
+
throw withHint(
|
|
2016
|
+
new Error(
|
|
2017
|
+
`The egress proxy container '${containerName}' is not running after being started, so this sandbox has no boundary to reach and no other route out.`,
|
|
2018
|
+
),
|
|
2019
|
+
'Almost always the image: it either lacks the compiled module (build the package before the image — `pnpm --filter @namzu/sandbox build`) or the entrypoint refused its configuration. `docker logs <container>` has the line the entrypoint wrote; the container is started with --rm, so an already-exited one is gone and its output with it.',
|
|
2020
|
+
)
|
|
2021
|
+
}
|
|
2022
|
+
|
|
843
2023
|
/**
|
|
844
2024
|
* Ask Docker which host port it bound to the worker port. Used
|
|
845
2025
|
* instead of the pre-reserve-then-publish pattern (which had a
|
|
@@ -959,10 +2139,104 @@ async function waitForWorkerReady(
|
|
|
959
2139
|
)
|
|
960
2140
|
}
|
|
961
2141
|
|
|
962
|
-
|
|
2142
|
+
/**
|
|
2143
|
+
* The argv as it may appear in an error message: the KEYS of every env
|
|
2144
|
+
* entry, with the values replaced.
|
|
2145
|
+
*
|
|
2146
|
+
* A rendered argv is the last place a secret should survive. A non-zero
|
|
2147
|
+
* `docker run` is a routine outcome — a missing image, a name conflict, a
|
|
2148
|
+
* daemon hiccup, ENOSPC — and its message goes wherever the sandbox
|
|
2149
|
+
* package's errors go: a log line, a telemetry batch, a CI transcript, a
|
|
2150
|
+
* pasted bug report. The env flags carry the worker's credential and every
|
|
2151
|
+
* value the host put in `options.env` (an API key, a broker token), and
|
|
2152
|
+
* none of them are needed to explain an exit code. The keys are kept
|
|
2153
|
+
* because they are what distinguishes "the image could not be pulled" from
|
|
2154
|
+
* "the environment was rejected".
|
|
2155
|
+
*
|
|
2156
|
+
* EVERY SPELLING docker accepts for that flag, not the one this backend
|
|
2157
|
+
* happens to emit today. `-e` IS `--env`, separated or `=`-attached, and
|
|
2158
|
+
* this function used to compare each element to the literal `'--env'`: the
|
|
2159
|
+
* long separated form this builder writes was redacted and `-e K=V`,
|
|
2160
|
+
* `--env=K=V` and `-e=K=V` were printed in full. A future caller writing
|
|
2161
|
+
* any of the three would have put a credential in a log line behind a
|
|
2162
|
+
* docblock that promised it would not. The covered forms are `--env K=V`,
|
|
2163
|
+
* `--env=K=V`, `-e K=V`, `-e=K=V` and the attached short form `-eK=V`; the
|
|
2164
|
+
* one shape it does not read is a value attached to an `-e` bundled into a
|
|
2165
|
+
* group of other short flags (`-iteK=V`), which no caller here writes and
|
|
2166
|
+
* which no rule short of matching `-e` anywhere inside an option could
|
|
2167
|
+
* catch. A valueless entry in any form (`--env K`) is passed through: it
|
|
2168
|
+
* resolves from the CLI's own environment and carries no value to redact.
|
|
2169
|
+
*
|
|
2170
|
+
* That redacts the workspace paths the layout is rendered from as well,
|
|
2171
|
+
* which are not secrets. They are also not what an exit code is about, and
|
|
2172
|
+
* a rule with exceptions is a rule that leaks the first time someone's
|
|
2173
|
+
* credential does not look like one.
|
|
2174
|
+
*/
|
|
2175
|
+
export function redactDockerArgv(args: readonly string[]): string[] {
|
|
2176
|
+
const rendered = [...args]
|
|
2177
|
+
/** `K=V` → `K=<redacted>`, keeping the key; a valueless entry is left alone. */
|
|
2178
|
+
const redactEntry = (entry: string): string => {
|
|
2179
|
+
const separator = entry.indexOf('=')
|
|
2180
|
+
return separator > 0 ? `${entry.slice(0, separator)}=<redacted>` : entry
|
|
2181
|
+
}
|
|
2182
|
+
for (let index = 0; index < rendered.length; index += 1) {
|
|
2183
|
+
const arg = rendered[index] as string
|
|
2184
|
+
if (arg === '--env' || arg === '-e') {
|
|
2185
|
+
const entry = rendered[index + 1]
|
|
2186
|
+
if (entry !== undefined) rendered[index + 1] = redactEntry(entry)
|
|
2187
|
+
index += 1
|
|
2188
|
+
continue
|
|
2189
|
+
}
|
|
2190
|
+
// Attached, where the option's value is the rest of the same element:
|
|
2191
|
+
// the `=` forms, and the short form with no separator (`-eK=V`).
|
|
2192
|
+
const prefix = arg.startsWith('--env=')
|
|
2193
|
+
? '--env='
|
|
2194
|
+
: arg.startsWith('-e=')
|
|
2195
|
+
? '-e='
|
|
2196
|
+
: arg.startsWith('-e') && arg.length > 2
|
|
2197
|
+
? '-e'
|
|
2198
|
+
: undefined
|
|
2199
|
+
if (prefix !== undefined) rendered[index] = prefix + redactEntry(arg.slice(prefix.length))
|
|
2200
|
+
}
|
|
2201
|
+
return rendered
|
|
2202
|
+
}
|
|
2203
|
+
|
|
2204
|
+
/**
|
|
2205
|
+
* Run a docker subcommand to completion.
|
|
2206
|
+
*
|
|
2207
|
+
* `extraEnv` is added to the child's environment rather than the parent's, and
|
|
2208
|
+
* it is how a value reaches a container without entering the argv this file
|
|
2209
|
+
* builds. Two callers use it, for the same reason:
|
|
2210
|
+
*
|
|
2211
|
+
* - The sandbox's own `docker run`, which hands over the worker's
|
|
2212
|
+
* per-instance credential for the valueless `--env NAMZU_SANDBOX_TOKEN` its
|
|
2213
|
+
* argv carries (minted in `spawnDockerSandbox`).
|
|
2214
|
+
* - The egress proxy's `docker run`, which names its configuration variable
|
|
2215
|
+
* in the argv and hands the value over here. See
|
|
2216
|
+
* `renderEgressProxyRunArgs`.
|
|
2217
|
+
*
|
|
2218
|
+
* Anything a process is told through the environment is visible to `ps`'s
|
|
2219
|
+
* neighbour, `/proc/<pid>/environ`, only to the user that owns it — while an
|
|
2220
|
+
* argv is world-readable on Linux.
|
|
2221
|
+
*/
|
|
2222
|
+
function runOnce(
|
|
2223
|
+
binary: string,
|
|
2224
|
+
args: string[],
|
|
2225
|
+
signal?: AbortSignal,
|
|
2226
|
+
extraEnv?: Readonly<Record<string, string>>,
|
|
2227
|
+
): Promise<string> {
|
|
963
2228
|
return new Promise((resolve, reject) => {
|
|
964
2229
|
signal?.throwIfAborted()
|
|
965
|
-
const child = spawn(binary, args, {
|
|
2230
|
+
const child = spawn(binary, args, {
|
|
2231
|
+
stdio: ['ignore', 'pipe', 'pipe'],
|
|
2232
|
+
// The one channel a value can ride in without entering the argv
|
|
2233
|
+
// this process builds: `ps` shows an argv to every user on the
|
|
2234
|
+
// host, and `/proc/<pid>/environ` is readable only by the same
|
|
2235
|
+
// user and root. `docker run --env NAME` (no `=`) reads the value
|
|
2236
|
+
// out of the CLI's own environment, which is why the credential
|
|
2237
|
+
// is passed this way and rendered valueless in the argv.
|
|
2238
|
+
...(extraEnv ? { env: { ...process.env, ...extraEnv } } : {}),
|
|
2239
|
+
})
|
|
966
2240
|
let stdout = ''
|
|
967
2241
|
let stderr = ''
|
|
968
2242
|
let settled = false
|
|
@@ -989,7 +2263,12 @@ function runOnce(binary: string, args: string[], signal?: AbortSignal): Promise<
|
|
|
989
2263
|
child.on('error', (error) => finish(error))
|
|
990
2264
|
child.on('close', (code) => {
|
|
991
2265
|
if (code === 0) finish(undefined, stdout.trim())
|
|
992
|
-
else
|
|
2266
|
+
else
|
|
2267
|
+
finish(
|
|
2268
|
+
new Error(
|
|
2269
|
+
`${binary} ${redactDockerArgv(args).join(' ')} exited ${code}: ${stderr.trim()}`,
|
|
2270
|
+
),
|
|
2271
|
+
)
|
|
993
2272
|
})
|
|
994
2273
|
if (signal?.aborted) abort()
|
|
995
2274
|
else signal?.addEventListener('abort', abort, { once: true })
|