@namzu/sandbox 16.0.0 → 17.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +331 -0
- package/README.md +164 -0
- package/dist/backends/aci-standby-pool/index.d.ts +22 -4
- package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
- package/dist/backends/aci-standby-pool/index.js +31 -7
- package/dist/backends/aci-standby-pool/index.js.map +1 -1
- package/dist/backends/docker/index.d.ts +243 -24
- package/dist/backends/docker/index.d.ts.map +1 -1
- package/dist/backends/docker/index.js +717 -108
- package/dist/backends/docker/index.js.map +1 -1
- package/dist/backends/firecracker/transport.d.ts +156 -1
- package/dist/backends/firecracker/transport.d.ts.map +1 -1
- package/dist/backends/firecracker/transport.js +223 -29
- package/dist/backends/firecracker/transport.js.map +1 -1
- package/dist/backends/http-worker-client.d.ts +64 -2
- package/dist/backends/http-worker-client.d.ts.map +1 -1
- package/dist/backends/http-worker-client.js +78 -7
- package/dist/backends/http-worker-client.js.map +1 -1
- package/dist/backends/kubernetes/egress-policy.d.ts +4 -1
- package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -1
- package/dist/backends/kubernetes/egress-policy.js +54 -3
- package/dist/backends/kubernetes/egress-policy.js.map +1 -1
- package/dist/backends/kubernetes/transport.d.ts +7 -0
- package/dist/backends/kubernetes/transport.d.ts.map +1 -1
- package/dist/backends/kubernetes/transport.js.map +1 -1
- package/dist/egress/proxy.d.ts +47 -2
- package/dist/egress/proxy.d.ts.map +1 -1
- package/dist/egress/proxy.js +31 -7
- package/dist/egress/proxy.js.map +1 -1
- package/dist/index.d.ts +67 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +22 -0
- package/dist/index.js.map +1 -1
- package/package.json +4 -4
- package/src/backends/aci-standby-pool/index.ts +37 -7
- package/src/backends/docker/index.ts +909 -126
- package/src/backends/firecracker/transport.ts +387 -36
- package/src/backends/http-worker-client.ts +89 -5
- package/src/backends/kubernetes/egress-policy.ts +56 -3
- package/src/backends/kubernetes/transport.ts +7 -0
- package/src/egress/proxy.ts +65 -8
- package/src/index.ts +89 -0
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,336 @@
|
|
|
1
1
|
# @namzu/sandbox
|
|
2
2
|
|
|
3
|
+
## 17.0.1
|
|
4
|
+
|
|
5
|
+
### Patch Changes
|
|
6
|
+
|
|
7
|
+
- 27a1665: A kubernetes deployment with `egress.policy: { kind: 'no-network' }` can create
|
|
8
|
+
sandboxes again. It could not before: every `create()` was refused with
|
|
9
|
+
`KubernetesEgressPolicyMismatchError` — a host that maps that to a reason such
|
|
10
|
+
as `sandbox-egress-policy-unverified` fails the run closed — and no policy an
|
|
11
|
+
operator could apply would have cleared it.
|
|
12
|
+
|
|
13
|
+
The named-object check compared `spec.egress` to the translation with a
|
|
14
|
+
deep-equal, and a core `NetworkPolicy` never stores an empty rule list:
|
|
15
|
+
`egress` is `omitempty` on the wire struct, so an object applied with
|
|
16
|
+
`egress: []` reads back with no `egress` key at all, and a merge patch cannot
|
|
17
|
+
put one back.
|
|
18
|
+
`no-network`'s whole translation IS that empty list, so `undefined !== []` made
|
|
19
|
+
the check unsatisfiable by any object a cluster can store. `verifyEgressPolicyApplied`
|
|
20
|
+
now reads an absent `egress` as the empty list it is, in that one comparison.
|
|
21
|
+
They are one policy and not merely one shape, because the check has already
|
|
22
|
+
required `policyTypes` to include `'Egress'` on a core policy, and that alone
|
|
23
|
+
denies all egress.
|
|
24
|
+
|
|
25
|
+
Nothing else is loosened. A live object whose rule array is non-empty still
|
|
26
|
+
fails a translation that intended none, and a live object with no `egress` at
|
|
27
|
+
all still fails a translation that intended rules — an absent list means
|
|
28
|
+
deny-all, which is not what config asked for. The comparison still exists to
|
|
29
|
+
refuse an object enforcing anything other than what was intended.
|
|
30
|
+
|
|
31
|
+
`patch`, and the reason it is not a `major`: this is not a change to the public
|
|
32
|
+
surface. No export, field, option or default moved, and the accepted set grows
|
|
33
|
+
by exactly the object the API server stores for the intent the deployment had
|
|
34
|
+
already declared. The refusal that disappears is one no consumer could have
|
|
35
|
+
been relying on — `no-network` is configured in order to create sandboxes, and
|
|
36
|
+
the only thing the refusal ever did was prevent that — so no upgrade action is
|
|
37
|
+
required and no deployment that passes today starts failing.
|
|
38
|
+
|
|
39
|
+
There was also no setting that avoided the refusal, so nothing to unset:
|
|
40
|
+
`egress.verify: 'named-object-only'` opts out of the UNION check and nothing
|
|
41
|
+
else — `egressUnionVerificationEnabled` gates `verifyUnion`, while the
|
|
42
|
+
named-object check runs on every create path regardless — so a `no-network`
|
|
43
|
+
host was refused under either setting. The ways out were to configure no
|
|
44
|
+
`egress` at all, or to fall back to `deny-all`, which allows the cluster
|
|
45
|
+
resolver on port 53 and is therefore not the same boundary.
|
|
46
|
+
|
|
47
|
+
Verified against a live `kind` cluster (v1.37), not only against a fake API
|
|
48
|
+
server: the `no-network` manifest applied, the object read back with this
|
|
49
|
+
backend's own client (a core `NetworkPolicy` stored with no `egress` key, and
|
|
50
|
+
`kubectl patch --type=merge -p '{"spec":{"egress":[]}}'` does not restore one),
|
|
51
|
+
and the real `verifyEgressPolicyApplied` run on what came off the cluster —
|
|
52
|
+
refused before this change, verified after, with a live object carrying a rule
|
|
53
|
+
and a live object carrying none both still refused.
|
|
54
|
+
|
|
55
|
+
- 9092a37: Nothing a consumer installs changes. The one edited file is
|
|
56
|
+
`k8s/__tests__/entrypoint.test.ts`, which `package.json#files` excludes from the
|
|
57
|
+
tarball; the harness only, and `entrypoint.sh` itself is deliberately untouched —
|
|
58
|
+
runtime behaviour, exports, types and defaults are the same.
|
|
59
|
+
|
|
60
|
+
`entrypoint.test.ts` has been intermittently red on `main`, twice in a release
|
|
61
|
+
run:
|
|
62
|
+
|
|
63
|
+
```
|
|
64
|
+
FAIL entrypoint.test.ts > entrypoint.sh prestop: the flush a stopping pod gets
|
|
65
|
+
> returns rather than hanging when the init does not go
|
|
66
|
+
AssertionError: expected 6 to be greater than or equal to 900
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
The mechanism is a race in the harness, not in the image. `entrypoint.sh` reads
|
|
70
|
+
`/proc/<pid>/comm` before it signals anything and exits 0 immediately when the
|
|
71
|
+
name is not the init it expects — a fail-closed refusal the image must keep, and
|
|
72
|
+
the reason `entrypoint.sh` is not what changed here. `spawnStandIn` echoed `$!`
|
|
73
|
+
for a child that had not `exec`'d yet, so inside that window the hook read `sh`
|
|
74
|
+
where the symlinked `tini` was intended, refused, and skipped its whole wait. The
|
|
75
|
+
case then measured the refusal — a few milliseconds of elapsed time and a handful
|
|
76
|
+
of ticks — instead of the flush it is about, and failed against a constant that
|
|
77
|
+
looks nothing like the number it got.
|
|
78
|
+
|
|
79
|
+
The fix is a bounded, loud-failing readiness poll, `awaitStandInExec`, routed
|
|
80
|
+
through `spawnStandIn` so every stand-in call site gets it: cases now wait for
|
|
81
|
+
the process to become the binary it was started as before they hand its pid to
|
|
82
|
+
the hook. It waits for the SPECIFIC name rather than for `sh` to disappear,
|
|
83
|
+
because the name is what the hook turns on and one call site deliberately wants a
|
|
84
|
+
plain `sleep` to stay a `sleep`. A stand-in that died during the wait says so
|
|
85
|
+
instead of waiting its bound out.
|
|
86
|
+
|
|
87
|
+
It also refuses up front an expected name longer than the kernel can report: the
|
|
88
|
+
kernel keeps at most 15 characters in `/proc/<pid>/comm`, so a longer name could
|
|
89
|
+
never match and the wait would fail on its deadline rather than on the truth.
|
|
90
|
+
Latent today — every stand-in this suite starts is named well inside the limit —
|
|
91
|
+
and the guard exists so a future one is a one-line failure that names the limit.
|
|
92
|
+
|
|
93
|
+
Verified: the race reproduced deterministically before the fix, 20/20 runs green
|
|
94
|
+
under load after it, and removing the wait restores the failure byte-identically.
|
|
95
|
+
|
|
96
|
+
## 17.0.0
|
|
97
|
+
|
|
98
|
+
### Major Changes
|
|
99
|
+
|
|
100
|
+
- 1722448: **A `container:docker` sandbox with an egress allowlist now needs an internal
|
|
101
|
+
network, a proxy image, and `container-network` reachability. Take this version
|
|
102
|
+
if you use one; otherwise nothing you install changes shape.**
|
|
103
|
+
|
|
104
|
+
The allowlist tier used to be enforced by `HTTP_PROXY` and nothing else. The
|
|
105
|
+
proxy ran in the process that created the sandbox, on the host's loopback, and
|
|
106
|
+
the sandbox kept ordinary bridge networking with full outbound reachability —
|
|
107
|
+
`--add-host namzu-egress:host-gateway` was the only thing pointing traffic at
|
|
108
|
+
it. Anything inside the container that opened a socket directly reached the
|
|
109
|
+
network with the allowlist unconsulted. It is now a sibling container on an
|
|
110
|
+
`--internal` network the sandbox is also on, which has no route off it: the
|
|
111
|
+
sandbox's traffic reaches the internet only through that container, and
|
|
112
|
+
everything else a host attaches to that network is a container on the sandbox's
|
|
113
|
+
own subnet.
|
|
114
|
+
|
|
115
|
+
**What a host using `EgressPolicy` of `static` or `resolver` must now do**, all
|
|
116
|
+
three refused at `create()` rather than downgraded:
|
|
117
|
+
|
|
118
|
+
1. `network` must be an `--internal` network: `docker network create
|
|
119
|
+
--internal <name>`. The network's `Internal` flag is read back from the
|
|
120
|
+
daemon; the name is never trusted.
|
|
121
|
+
2. `hostReachability` must be `'container-network'`. A published host port
|
|
122
|
+
needs a route out and an internal network has none — the refusal names the
|
|
123
|
+
mode to move to. **A host-side consumer (the CLI, direct dev) that reached
|
|
124
|
+
the worker on `127.0.0.1:<port>` under an allowlist policy must move to a
|
|
125
|
+
consumer on the internal network.**
|
|
126
|
+
3. `egressProxyImage` must name the proxy image:
|
|
127
|
+
|
|
128
|
+
```bash
|
|
129
|
+
pnpm --filter @namzu/sandbox build
|
|
130
|
+
docker build -f packages/sandbox/egress-proxy/Dockerfile -t <tag> packages/sandbox
|
|
131
|
+
```
|
|
132
|
+
|
|
133
|
+
It is a second image because the sandbox image is a string this backend
|
|
134
|
+
cannot read, and the bind-mount alternative breaks on the remote-daemon
|
|
135
|
+
deployment `container-network` exists for.
|
|
136
|
+
|
|
137
|
+
`HTTP_PROXY` and friends are still set and now direct traffic rather than
|
|
138
|
+
permit it: a tool that honours them goes through the boundary, one that ignores
|
|
139
|
+
them fails `Network unreachable` instead of bypassing the policy.
|
|
140
|
+
|
|
141
|
+
**Four behaviour changes to know about, all of them consequences of the
|
|
142
|
+
boundary existing:**
|
|
143
|
+
|
|
144
|
+
- A `resolver` policy is resolved at `create()` and at each
|
|
145
|
+
`setNetworkPolicy()`, not per request. A resolver that rotates between those
|
|
146
|
+
moments is not picked up until the next one.
|
|
147
|
+
- `setNetworkPolicy()` replaces the proxy container rather than swapping the
|
|
148
|
+
allowlist in place. The window between the two has no proxy in it and fails
|
|
149
|
+
**closed** — no request is permitted that the new policy would refuse.
|
|
150
|
+
- `brokeredCredentials` are passed to the proxy container's environment. They
|
|
151
|
+
still never enter the sandbox; they are now readable by anything with access
|
|
152
|
+
to the docker daemon, which should be treated as credential access. (On the
|
|
153
|
+
way there the value goes through the `docker` CLI child's ENVIRONMENT, not its
|
|
154
|
+
argv — an argv is world-readable in `/proc/<pid>/cmdline` on Linux.) Note also
|
|
155
|
+
that `createSandboxProvider` has never forwarded `brokeredCredentials` at all;
|
|
156
|
+
that pre-existing gap is unchanged.
|
|
157
|
+
- `egressProxyUpstreamNetwork` (default `bridge`) is where the proxy reaches
|
|
158
|
+
the internet. Anything else attached to that network can reach the proxy and
|
|
159
|
+
use it, so name a dedicated one on a shared daemon. `'none'`, and the internal
|
|
160
|
+
network itself, are refused: either leaves the proxy with no route out.
|
|
161
|
+
- Everything else on the SANDBOX's internal network is reachable from the
|
|
162
|
+
sandbox — a second sandbox, a second sandbox's proxy. That is a property of
|
|
163
|
+
the network rather than of this tier; give each sandbox its own internal
|
|
164
|
+
network when they should not see one another.
|
|
165
|
+
|
|
166
|
+
`deny-all` and `allow-all` are unaffected beyond the internal network
|
|
167
|
+
`deny-all` already required. `EgressProxy` gains two optional options
|
|
168
|
+
(`bindHost`, `selfNames`) and the container entrypoint sets them; the class is
|
|
169
|
+
otherwise unchanged.
|
|
170
|
+
|
|
171
|
+
- f1fd331: The container backend's worker now authenticates its caller, and a worker that
|
|
172
|
+
cannot refuses to start on a routable bind. This is a `major` because the wire
|
|
173
|
+
shape of a public surface changed: every route but `GET /healthz` now requires
|
|
174
|
+
`Authorization: Bearer <NAMZU_SANDBOX_TOKEN>`, a missing or wrong token is
|
|
175
|
+
`401 {"error":"unauthorized"}`, and a worker with no token will not listen on
|
|
176
|
+
anything but loopback. A token that is set but cannot be answered — empty,
|
|
177
|
+
padded, or carrying a character an HTTP header cannot hold — now refuses to
|
|
178
|
+
start rather than leaving a worker that looks authenticated and refuses its own
|
|
179
|
+
host.
|
|
180
|
+
|
|
181
|
+
Nothing in the package's TypeScript API changed — no export was removed,
|
|
182
|
+
renamed or narrowed, and `HttpWorkerClient` and `execViaHttpWorker` only gained
|
|
183
|
+
an optional token parameter. What changed is the contract between this backend
|
|
184
|
+
and the worker process it runs, and that contract is deployed separately: the
|
|
185
|
+
worker lives in `packages/sandbox/worker/server.js`, which the tarball does not
|
|
186
|
+
ship, so the image is built from whatever checkout the operator has.
|
|
187
|
+
|
|
188
|
+
**What breaks, and for whom.**
|
|
189
|
+
|
|
190
|
+
- **A worker image rebuilt from this release, paired with a host that predates
|
|
191
|
+
it.** The old host injects no token, so the new worker refuses to start on
|
|
192
|
+
the default `0.0.0.0` bind and every `create()` fails at readiness with the
|
|
193
|
+
container already exited. The refusal names the variable. Bring the host up
|
|
194
|
+
in the same step, or set the escape below.
|
|
195
|
+
- **Anything that talks to the worker without going through this backend.** The
|
|
196
|
+
route contract is now authenticated; a probe, a health-check script that
|
|
197
|
+
calls `/execute`, or a hand-rolled client must present the token.
|
|
198
|
+
- **The standby-pool backend.** A pooled worker built from this release will
|
|
199
|
+
refuse to boot on its routable default bind, because this backend has no way
|
|
200
|
+
to hand it a credential. The claim API admits exactly one property override,
|
|
201
|
+
and it is not `env` — it is a config map, which on Linux reaches the container
|
|
202
|
+
as a file mount under `/mnt/configmap/<containername>/<key>`, not as an
|
|
203
|
+
environment variable, while the worker reads its token from `process.env` at
|
|
204
|
+
startup. The platform does not validate config-map values either, and its own
|
|
205
|
+
guidance is that a value affecting application security belongs in an
|
|
206
|
+
environment variable. So the per-instance credential for this backend is NOT
|
|
207
|
+
implemented here, and the change that would close the gap — a per-claim value
|
|
208
|
+
in the config map, read by the worker — is written down in
|
|
209
|
+
`docs/sdk/container-sandbox-worker.md` rather than half-built. A token on the
|
|
210
|
+
shared profile does not fix it either: the worker would boot and then `401`
|
|
211
|
+
every call the backend makes, because this backend's client is constructed
|
|
212
|
+
without a token and it has no field to carry one.
|
|
213
|
+
|
|
214
|
+
**A host and a worker from the same release need no configuration.** The
|
|
215
|
+
container backend mints 32 random bytes per `create()`, hands the value to the
|
|
216
|
+
docker CLI in its own environment (resolved by a valueless
|
|
217
|
+
`--env NAMZU_SANDBOX_TOKEN`, so it is in no argv), and sends it as a bearer
|
|
218
|
+
header on every call it makes. Upgrading both together is the whole migration.
|
|
219
|
+
The token is per instance, never written to disk, and dies with the container.
|
|
220
|
+
It is readable where a peer in the same container or a caller who can already
|
|
221
|
+
talk to the docker daemon could read it — `docker inspect` shows it on the
|
|
222
|
+
container config for the container's life — which is why it is per-instance
|
|
223
|
+
rather than shared, and why it is a defence in depth behind network placement
|
|
224
|
+
rather than a replacement for it.
|
|
225
|
+
|
|
226
|
+
**The standby pool: place it first, decide about the credential second.** If you
|
|
227
|
+
run that backend, the control that actually carries the exposure is the network,
|
|
228
|
+
and it is the one to get right before anything else — `aci-standby-pool` already
|
|
229
|
+
refuses to claim a container group without `subnetId` (unless you set
|
|
230
|
+
`allowPublicAddress: true` and mean it), so the group sits on a private address
|
|
231
|
+
and that network is what stands in front of the worker. The refusal is unchanged
|
|
232
|
+
by this release; what has changed is that it is now the first half of the answer
|
|
233
|
+
rather than the whole of it. The second half is the escape below, which is what
|
|
234
|
+
makes a pooled worker start at all today: put
|
|
235
|
+
`NAMZU_SANDBOX_ALLOW_UNAUTHENTICATED=1` on the container group profile, with the
|
|
236
|
+
group on a private address. It is needed there only because this backend cannot
|
|
237
|
+
present a credential, and it turns a worker that cannot boot into one that
|
|
238
|
+
serves inside that address. Read the standby-pool bullet under "What breaks"
|
|
239
|
+
before taking that step, and note that a token on the shared profile is not an
|
|
240
|
+
alternative to it.
|
|
241
|
+
|
|
242
|
+
**For a deployment the container backend creates**, the escape is a migration
|
|
243
|
+
tool rather than a destination: it is what keeps you running while the host and
|
|
244
|
+
the worker image are brought up in the same step.
|
|
245
|
+
|
|
246
|
+
**To keep the old behaviour on purpose**, set
|
|
247
|
+
`NAMZU_SANDBOX_ALLOW_UNAUTHENTICATED=1` in the worker's environment — the
|
|
248
|
+
container backend's `options.env` for a sandbox it creates, or the standby
|
|
249
|
+
pool's container group profile for a pooled one. That gives the credential up
|
|
250
|
+
rather than deferring it: with it set and no token, every route but `/healthz`
|
|
251
|
+
is open to whoever can route to the container, which is exactly what the
|
|
252
|
+
default used to be. On the pool, the address comes first and this second; on
|
|
253
|
+
every other backend, both the host and the image should be upgraded in one step
|
|
254
|
+
instead. Only `1`, `true`, `yes` and `on` count, in any case and with no
|
|
255
|
+
surrounding whitespace; everything else — `= 0`, `= false`, `= " yes "` — means
|
|
256
|
+
off, so the flag cannot turn itself on.
|
|
257
|
+
|
|
258
|
+
Also unchanged, and worth repeating from the README: the transport is plain
|
|
259
|
+
HTTP, so this is a bearer token over the wire and it is defence in depth behind
|
|
260
|
+
network placement, not a replacement for it. See
|
|
261
|
+
`docs/sdk/container-sandbox-worker.md`.
|
|
262
|
+
|
|
263
|
+
### Minor Changes
|
|
264
|
+
|
|
265
|
+
- 01fbc9b: Three additive declarations, no default changed and nothing removed:
|
|
266
|
+
|
|
267
|
+
- `MicroVMBackendConfig.onExecTiming` — the owned Firecracker tier's provider
|
|
268
|
+
config gains an optional per-exec timing hook.
|
|
269
|
+
- `FirecrackerTransportTiming` — exported from the package entry point, the
|
|
270
|
+
shape that hook is called with.
|
|
271
|
+
- `VsockTransportOptions.onExecTiming` — the same hook at the transport level,
|
|
272
|
+
for a host that builds a `VsockAgentTransport` itself.
|
|
273
|
+
|
|
274
|
+
A host that sets none of them sends, receives and waits for exactly what it did
|
|
275
|
+
before: the timing accumulator is created only when the hook is set, and the
|
|
276
|
+
`undefined` checks that skip creating it skip every clock read that would have
|
|
277
|
+
filled it.
|
|
278
|
+
|
|
279
|
+
The hook reports one `exec()`'s wall clock as named phases — `dialMs`,
|
|
280
|
+
`reserveMs`, `executeMs` and `drainMs`, the four the Kubernetes tier's
|
|
281
|
+
`onTiming` already reports, plus `firstFrameMs`, `terminatorMs` and
|
|
282
|
+
`peerCloseMs` for the intervals inside the execute round trip. The three
|
|
283
|
+
sub-phases are absent, rather than zero, when their phase was never reached; the
|
|
284
|
+
four base phases are always present and use `0` for "this never happened" (a
|
|
285
|
+
call that failed at the dial reports `executeMs: 0`). The payload is durations
|
|
286
|
+
only: never the agent token, a command, its arguments, or any output.
|
|
287
|
+
|
|
288
|
+
One behaviour change worth naming, because it is the reason the hook is useful:
|
|
289
|
+
the transport's execution adapter and its `RemoteExecutionController` are now
|
|
290
|
+
built per call instead of once per transport. That controller holds no per-call
|
|
291
|
+
state, so nothing observable changes — but an adapter built once had nowhere
|
|
292
|
+
per-call to accumulate into, and a single accumulator on the transport would
|
|
293
|
+
have two concurrent `exec()` calls writing each other's phases. The Kubernetes
|
|
294
|
+
tier's own transport has been arranged this way since it was written.
|
|
295
|
+
|
|
296
|
+
`POST_RESPONSE_CLOSE_TIMEOUT_MS` (1 s) is unchanged and its behaviour is
|
|
297
|
+
unchanged; it is now documented as what it always was — a reject-only guard
|
|
298
|
+
that fails a socket whose peer never closes, never a wait a successful call
|
|
299
|
+
pays. If your host puts a relay between this process and the guest, `peerCloseMs`
|
|
300
|
+
is the number that tells you whether that relay holds the FIN: a hold **under**
|
|
301
|
+
a second is reported there on a call that resolved, and a hold **at or past**
|
|
302
|
+
a second fails the call with `vsock transport: exec peer did not close after
|
|
303
|
+
terminator`, reporting no `peerCloseMs` at all — the close in that case is the
|
|
304
|
+
one this transport causes itself when the guard fires, and its own constant is
|
|
305
|
+
not published as an elapsed time. So a resolved exec's fixed cost cannot be
|
|
306
|
+
hiding in that window: whichever way the window goes, it is either reported or
|
|
307
|
+
it is a rejection, and never a silent second.
|
|
308
|
+
|
|
309
|
+
### Patch Changes
|
|
310
|
+
|
|
311
|
+
- 656598a: Nothing a consumer installs changes. The one edited file is
|
|
312
|
+
`src/backends/docker/__tests__/leaf-permissions.smoke.test.ts`, which
|
|
313
|
+
`package.json#files` excludes from the tarball; runtime behaviour, exports, types
|
|
314
|
+
and defaults are untouched.
|
|
315
|
+
|
|
316
|
+
The `Sandbox smoke` workflow has been red on `main` since the docker hardening
|
|
317
|
+
landed (`481fb8ff`, `--read-only` on by default). That case asserts that uid 1001
|
|
318
|
+
cannot `mkdir` into the unbound `/mnt/user-data`, and it pinned the refusal to
|
|
319
|
+
one spelling — `permission denied`. With a read-only rootfs the kernel answers
|
|
320
|
+
`read-only file system` instead, because EROFS is consulted before the DAC check
|
|
321
|
+
that would have produced EACCES. The property the case exists for never changed
|
|
322
|
+
(a bound writable leaf would let the `mkdir` succeed, and the `--rc` assertion
|
|
323
|
+
still catches that); only the kernel's wording did.
|
|
324
|
+
|
|
325
|
+
The assertion now accepts either refusal and says why both are legitimate, so the
|
|
326
|
+
next change to which check fires first is a one-line read rather than a day of
|
|
327
|
+
red. `dash` and Docker are both absent from the machine this was written on, so
|
|
328
|
+
the fix is verified by the workflow that runs this file, not locally.
|
|
329
|
+
|
|
330
|
+
- Updated dependencies [4e8cf5c]
|
|
331
|
+
- Updated dependencies [19abb2c]
|
|
332
|
+
- @namzu/sdk@42.0.0
|
|
333
|
+
|
|
3
334
|
## 16.0.0
|
|
4
335
|
|
|
5
336
|
### Major Changes
|
package/README.md
CHANGED
|
@@ -122,6 +122,59 @@ tier gets uid 0 mapped to an unprivileged uid outside.
|
|
|
122
122
|
overrides the image's own choice and the reference image already ends with
|
|
123
123
|
`USER namzu`.
|
|
124
124
|
|
|
125
|
+
## Container egress boundary
|
|
126
|
+
|
|
127
|
+
An egress policy of `static` or `resolver` — a host allowlist — is enforced by
|
|
128
|
+
the egress proxy running as a **sibling container**, not by the process that
|
|
129
|
+
created the sandbox. The proxy is dual-homed: `docker run` puts it on an
|
|
130
|
+
ordinary network so it has a route to the internet, and `docker network
|
|
131
|
+
connect --alias namzu-egress` adds the `--internal` network the sandbox is on.
|
|
132
|
+
The sandbox joins that internal network alone and has no route off it — a route
|
|
133
|
+
it cannot put back, because `--cap-drop=ALL` removed `NET_ADMIN`.
|
|
134
|
+
`--add-host namzu-egress:host-gateway` and the loopback proxy that needed it are
|
|
135
|
+
gone.
|
|
136
|
+
|
|
137
|
+
"No route off it" is the accurate half of that, and it is worth being exact
|
|
138
|
+
about the other: the internal network is a subnet, so the sandbox can also
|
|
139
|
+
reach whatever else a host attaches there — a second sandbox, a second
|
|
140
|
+
sandbox's proxy. Give each sandbox its own internal network when they should
|
|
141
|
+
not see one another; see
|
|
142
|
+
[docs/sdk/sandbox-egress.md](../../docs/sdk/sandbox-egress.md).
|
|
143
|
+
|
|
144
|
+
`HTTP_PROXY`, `http_proxy`, `HTTPS_PROXY`, `https_proxy` and `NO_PROXY` are
|
|
145
|
+
still set on the sandbox, and what they are has changed. They no longer permit
|
|
146
|
+
traffic, they direct it: a tool that honours them sends its request through the
|
|
147
|
+
boundary, and a tool that ignores them — a binary that does not read proxy
|
|
148
|
+
environment, `curl --noproxy '*'`, a raw socket — has nowhere to send anything
|
|
149
|
+
and fails with `Network unreachable`. Before this, that second tool reached the
|
|
150
|
+
network with the allowlist unconsulted.
|
|
151
|
+
|
|
152
|
+
Three things a host must supply, each refused at `create()` rather than
|
|
153
|
+
downgraded: an `--internal` network in `network` (`docker network create
|
|
154
|
+
--internal <name>`), `hostReachability: 'container-network'` (a published host
|
|
155
|
+
port needs a route out that this network does not have), and
|
|
156
|
+
`egressProxyImage` — the proxy image, built from
|
|
157
|
+
`packages/sandbox/egress-proxy/Dockerfile` exactly as the sandbox image is
|
|
158
|
+
built from `packages/sandbox/worker/Dockerfile`. Nothing in this repository
|
|
159
|
+
pushes an image; both are built by hand and named by tag in the config.
|
|
160
|
+
|
|
161
|
+
The proxy container carries the same baseline the sandbox does
|
|
162
|
+
(`--cap-drop=ALL`, `--security-opt=no-new-privileges`, `--ipc private`,
|
|
163
|
+
`--read-only`), because it is the process standing between untrusted code and
|
|
164
|
+
the internet. `deny-all` and `allow-all` need none of this beyond the internal
|
|
165
|
+
network `deny-all` already required.
|
|
166
|
+
|
|
167
|
+
What the boundary does not cover — domain fronting inside a `CONNECT` tunnel, a
|
|
168
|
+
`resolver` policy resolved at `create()` and `setNetworkPolicy()` rather than
|
|
169
|
+
per request, `setNetworkPolicy()` replacing the proxy container rather than
|
|
170
|
+
swapping a list in place, brokered credentials now readable by anything with
|
|
171
|
+
daemon access, the proxy's listener being reachable by whatever else shares its
|
|
172
|
+
upstream network, and everything else the sandbox shares its own internal
|
|
173
|
+
network with — is stated in
|
|
174
|
+
[docs/sdk/sandbox-egress.md](../../docs/sdk/sandbox-egress.md), along with the
|
|
175
|
+
argv-level evidence this change is verified by and the fact that no test here
|
|
176
|
+
starts a container.
|
|
177
|
+
|
|
125
178
|
## Protocol readiness and cancellation
|
|
126
179
|
|
|
127
180
|
Pass `SandboxExecOptions.signal` to stop a command on any shipped backend. A
|
|
@@ -581,6 +634,117 @@ sees the agent's own settings — `NAMZU_SANDBOX_WORKSPACE` among them. A
|
|
|
581
634
|
workload that needs a value in its terminal passes it in `env` on the
|
|
582
635
|
`openTerminal` call, which still wins over everything else, `TERM` included.
|
|
583
636
|
|
|
637
|
+
## The container worker's control API, and its token
|
|
638
|
+
|
|
639
|
+
The container backend runs `worker/server.js` — a different process from the
|
|
640
|
+
guest agent above, with a different wire: plain HTTP on a port Docker forwards,
|
|
641
|
+
not a framed stream over a socket. It serves `GET /healthz`,
|
|
642
|
+
`POST /execute`, `POST /executions/reserve`, `POST /cancel`, `POST /read-file`
|
|
643
|
+
and `POST /write-file`, and until recently it authenticated nothing: any peer
|
|
644
|
+
that could route to the container could run a command or read and write a file
|
|
645
|
+
inside it. What made that defensible was the network the container is attached
|
|
646
|
+
to, which is a property of a deployment rather than of the worker, and is
|
|
647
|
+
absent wherever a control plane can put the container on a public address.
|
|
648
|
+
|
|
649
|
+
Every route but `GET /healthz` now requires
|
|
650
|
+
`Authorization: Bearer <NAMZU_SANDBOX_TOKEN>`. A missing, empty or different
|
|
651
|
+
token is answered `401` with `{"error":"unauthorized"}` and nothing else — no
|
|
652
|
+
expected/got, no length, no hint about which routes exist — and the request
|
|
653
|
+
never reaches a handler, so a refused `/write-file` writes nothing. The
|
|
654
|
+
comparison is a fixed-width SHA-256 digest compare, so neither the value nor
|
|
655
|
+
its length is learnable by probing, and the gate runs before every dispatch,
|
|
656
|
+
including the 404, so an unauthenticated caller cannot tell a real route from a
|
|
657
|
+
missing one, or a wrong token from a missing one. `/healthz` requires no token
|
|
658
|
+
and never echoes one, because it is what the host polls before it has any other
|
|
659
|
+
business with the worker, and it answers with liveness and the protocol version
|
|
660
|
+
only. That exemption is an exact match on the whole URL, so `POST /healthz`,
|
|
661
|
+
`GET /healthz?x=1` and `GET /healthz/` are gated like anything else. It also
|
|
662
|
+
leaves exactly one bit readable without a credential, deliberately: the
|
|
663
|
+
retiring-worker check answers before the `/healthz` dispatch, so a worker that
|
|
664
|
+
has poisoned itself answers `503 {"error":"worker_retiring"}` where a healthy
|
|
665
|
+
one answers `200`. That is the drain signal, and the readiness probe — which
|
|
666
|
+
has no credential to present — is who reads it; on every route that does
|
|
667
|
+
anything, a caller without the token gets the same `401` from a retiring worker
|
|
668
|
+
and a serving one.
|
|
669
|
+
|
|
670
|
+
**The token is minted per instance, by whoever starts the container.** The
|
|
671
|
+
container backend generates 32 random bytes at `create()` time and hands the
|
|
672
|
+
value to the docker CLI in ITS environment, resolved by the valueless
|
|
673
|
+
`--env NAMZU_SANDBOX_TOKEN`. It shares a destination with three variables that
|
|
674
|
+
already exist — the workspace path and the read/write roots land in the
|
|
675
|
+
container's environment through the same `--env` mechanism — and nothing else:
|
|
676
|
+
those three are rendered in the argv as `--env K=V`, values and all, while this
|
|
677
|
+
one is valueless, which is docker's form for "take the value from the CLI's own
|
|
678
|
+
environment". So it is in no argv, `ps` on the host does not show it, and a
|
|
679
|
+
failed `docker run` renders its argv with every `--env` value redacted, in all
|
|
680
|
+
four spellings of the flag, so the error a host logs does not carry it either.
|
|
681
|
+
What it IS visible in, said plainly: the container's own config, so
|
|
682
|
+
`docker inspect <name>` shows it for the container's life to anyone who can
|
|
683
|
+
already talk to the daemon, and the worker's `/proc` inside the container to a
|
|
684
|
+
workload sharing its uid. That is why it is per-instance and why it dies with
|
|
685
|
+
the container — an image-level secret would be shared by every container ever
|
|
686
|
+
built from it and readable by anything that can pull it, and a profile-level one
|
|
687
|
+
would be shared by every instance claimed from a pool. The variable carries the
|
|
688
|
+
`NAMZU_SANDBOX_` prefix, which is what keeps it out of every command the sandbox
|
|
689
|
+
runs: the worker strips that prefix from the environment it hands to a spawned
|
|
690
|
+
command, so a sandboxed task cannot read the credential out of its own
|
|
691
|
+
environment.
|
|
692
|
+
|
|
693
|
+
**A worker with no token refuses to start on a routable address.** The bind
|
|
694
|
+
default is `0.0.0.0` and stays that way: a published container port forwards to
|
|
695
|
+
the container's interface address rather than to its loopback, so narrowing the
|
|
696
|
+
default disables the container backend instead of hardening it. The credential
|
|
697
|
+
is what makes that default defensible, so its absence fails closed:
|
|
698
|
+
|
|
699
|
+
| What the worker was given | What it does |
|
|
700
|
+
|---|---|
|
|
701
|
+
| `NAMZU_SANDBOX_TOKEN` set | Requires it on every route but `/healthz`, whatever the bind address |
|
|
702
|
+
| No token, bound to loopback (`127.0.0.1`, `::1`, `localhost`) | Starts, unauthenticated. Nothing outside the container's own network namespace can open that socket, and refusing here would break the host-beside-docker dev case while closing nothing |
|
|
703
|
+
| No token, bound anywhere else — including the `0.0.0.0` default | Refuses to start, exit 1, naming the variable and every way out |
|
|
704
|
+
| `NAMZU_SANDBOX_TOKEN` set but **empty**, or with leading/trailing whitespace | Refuses to start in every mode. An empty value is the shape an injected secret takes when the injection resolved to nothing; a padded one is read from a trimmed header and so can never be presented, leaving a worker that looks authenticated and refuses its own host for the container's life |
|
|
705
|
+
| `NAMZU_SANDBOX_TOKEN` set to something **no HTTP header can carry** — a code point above U+00FF, or a C0 control other than HTAB | Refuses to start in every mode. A header value is one byte per character, so the client's own `fetch` throws before the request leaves the host for the first, and the HTTP parser on this side drops the connection for the second. Same "boots authenticated, refuses its host" state, caught at boot instead of at the first call |
|
|
706
|
+
| No token, routable, and `NAMZU_SANDBOX_ALLOW_UNAUTHENTICATED=1` | Starts, unauthenticated, on purpose |
|
|
707
|
+
|
|
708
|
+
**The escape, and what it gives up.**
|
|
709
|
+
`NAMZU_SANDBOX_ALLOW_UNAUTHENTICATED=1` is the explicit, named way to keep an
|
|
710
|
+
existing deployment running unauthenticated. It gives up the credential
|
|
711
|
+
entirely rather than deferring it: every route but `/healthz` is then open to
|
|
712
|
+
whoever can route to the container, which is exactly the pre-token behaviour
|
|
713
|
+
and exactly what the credential was added to remove. Only `1`, `true`, `yes` and
|
|
714
|
+
`on`, in any case and with no surrounding whitespace, turn it on: `= 0` and
|
|
715
|
+
`= false` mean off, and so does everything else, because a flag whose
|
|
716
|
+
off-spelling turns it on is a trap — and so is a security escape that accepts
|
|
717
|
+
the shape a value takes when an editor adds a space to it. A configured token
|
|
718
|
+
always wins over the escape, since the escape can only mean "serve without a
|
|
719
|
+
credential", never "ignore the one I was given".
|
|
720
|
+
|
|
721
|
+
**A worker this host did not create must be provisioned with the token by
|
|
722
|
+
whoever does.** There is no channel back: the worker is handed its credential
|
|
723
|
+
at startup and never publishes it, so a warm pool, a shared container-group
|
|
724
|
+
profile or a container someone else started cannot be authenticated against by
|
|
725
|
+
a host that has no way to learn what it holds. The standby-pool backend cannot
|
|
726
|
+
send a per-instance token: the one property override its claim API admits is a
|
|
727
|
+
config map, and a config map arrives as a file mount under `/mnt/configmap`,
|
|
728
|
+
not as an environment variable — while this worker reads its token from
|
|
729
|
+
`process.env` at startup and never looks at the filesystem for one. That is a
|
|
730
|
+
channel that exists and a credential that still does not go in it, for two
|
|
731
|
+
reasons rather than none: the worker would not read it, and Microsoft's own
|
|
732
|
+
guidance is that config map values are not validated by the runtime and are not
|
|
733
|
+
where a value affecting application security belongs. A token on the shared
|
|
734
|
+
profile would be one credential for every instance claimed from the pool, and
|
|
735
|
+
this backend has no field to present one anyway — so what runs there today is a
|
|
736
|
+
private address (`subnetId`) plus the worker-side escape. The full answer, and
|
|
737
|
+
the change that would close the gap, are in
|
|
738
|
+
`docs/sdk/container-sandbox-worker.md`.
|
|
739
|
+
|
|
740
|
+
**The transport is not confidential.** `Authorization: Bearer` over plain HTTP
|
|
741
|
+
is replayable by anything on the path, so this is defence in depth behind
|
|
742
|
+
network placement, not a replacement for it. On the container backend the
|
|
743
|
+
worker is reached over Docker's port-forward on host loopback, or by DNS name
|
|
744
|
+
on a private bridge; it does not make a public-address deployment safe, which is
|
|
745
|
+
why the standby-pool backend still refuses to claim one without `subnetId`. The
|
|
746
|
+
egress proxy is unrelated to any of this and checks nothing inbound.
|
|
747
|
+
|
|
584
748
|
## Firecracker workspace channels
|
|
585
749
|
|
|
586
750
|
The Firecracker backend exposes two optional same-sandbox channels. Call
|
|
@@ -132,10 +132,28 @@ export declare function assertEnforceable(options: SandboxBackendOptions): void;
|
|
|
132
132
|
* so nothing looks wrong. A caller who never heard of `subnetId` gets a
|
|
133
133
|
* working sandbox on the internet and no signal at all.
|
|
134
134
|
*
|
|
135
|
-
* What is on that address matters: `worker/server.js`
|
|
136
|
-
*
|
|
137
|
-
*
|
|
138
|
-
*
|
|
135
|
+
* What is on that address matters: `worker/server.js` binds every interface
|
|
136
|
+
* and, inside a private network, that was the boundary doing the work — the
|
|
137
|
+
* worker authenticated nobody. With a public address there is no boundary
|
|
138
|
+
* left, and the worker's `/execute` is reachable by anyone.
|
|
139
|
+
*
|
|
140
|
+
* The worker now requires a per-instance `Authorization: Bearer` token on
|
|
141
|
+
* every route but `/healthz`, and REFUSES TO START on a routable bind when
|
|
142
|
+
* it has none. This backend cannot supply one, and the reason this comment
|
|
143
|
+
* used to give for that was wrong: it said the claim API's single admitted
|
|
144
|
+
* override — a config map — is no channel into the container group. It is
|
|
145
|
+
* one, on Linux a file mount under `/mnt/configmap/<containername>/<key>`
|
|
146
|
+
* carrying caller-supplied per-claim values. The credential is declined
|
|
147
|
+
* anyway, for two reasons that survive their source: the worker reads its
|
|
148
|
+
* token from `process.env` at startup and never looks for a file, so a
|
|
149
|
+
* mounted value is not read; and Microsoft's own guidance is that config
|
|
150
|
+
* map values are not validated by the runtime and that values affecting
|
|
151
|
+
* application security "should be made available to the container using
|
|
152
|
+
* environment variables". The full answer, and the change that would close
|
|
153
|
+
* the gap, is in `docs/sdk/container-sandbox-worker.md`. What that means
|
|
154
|
+
* for an operator is written down there too, rather than left to be
|
|
155
|
+
* discovered at startup; this refusal is unchanged, and now stands in
|
|
156
|
+
* front of a worker that would refuse the routable case itself.
|
|
139
157
|
*
|
|
140
158
|
* Defaulting to refusal rather than to a warning, because a warning on a
|
|
141
159
|
* path that otherwise succeeds is read once and never again.
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../../../src/backends/aci-standby-pool/index.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAwCG;AAEH,OAAO,KAAK,EAEX,8BAA8B,EAU9B,MAAM,YAAY,CAAA;AAGnB,OAAO,KAAK,EAAE,cAAc,EAAE,qBAAqB,EAAE,MAAM,gBAAgB,CAAA;AAc3E;;;;;GAKG;AACH,MAAM,MAAM,gBAAgB,GAAG,MAAM,OAAO,CAAC,MAAM,CAAC,CAAA;AAEpD,MAAM,WAAW,mCAAmC;IACnD,QAAQ,CAAC,cAAc,EAAE,MAAM,CAAA;IAC/B,QAAQ,CAAC,aAAa,EAAE,MAAM,CAAA;IAC9B,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAA;IACzB;;;;OAIG;IACH,QAAQ,CAAC,qBAAqB,EAAE,MAAM,CAAA;IACtC;;;OAGG;IACH,QAAQ,CAAC,+BAA+B,EAAE,MAAM,CAAA;IAChD;;;OAGG;IACH,QAAQ,CAAC,6BAA6B,CAAC,EAAE,MAAM,CAAA;IAC/C;;;OAGG;IACH,QAAQ,CAAC,MAAM,EAAE,8BAA8B,CAAA;IAC/C;;OAEG;IACH,QAAQ,CAAC,WAAW,EAAE,gBAAgB,CAAA;IACtC;;;;;;OAMG;IACH,QAAQ,CAAC,QAAQ,CAAC,EAAE,MAAM,CAAA;IAC1B;;;;;;;;;;;;;OAaG;IACH,QAAQ,CAAC,kBAAkB,CAAC,EAAE,OAAO,CAAA;IACrC,QAAQ,CAAC,mBAAmB,CAAC,EAAE,MAAM,CAAA;IACrC,QAAQ,CAAC,cAAc,CAAC,EAAE,MAAM,CAAA;IAChC;;OAEG;IACH,QAAQ,CAAC,UAAU,CAAC,EAAE,MAAM,CAAA;IAC5B,QAAQ,CAAC,aAAa,CAAC,EAAE,MAAM,CAAA;IAC/B;;;;;;OAMG;IACH,QAAQ,CAAC,mBAAmB,CAAC,EAAE,MAAM,CAAA;CACrC;AASD;;;;GAIG;AACH,wBAAgB,0BAA0B,CACzC,MAAM,EAAE,mCAAmC,GACzC,cAAc,CAiBhB;AAoMD,wBAAgB,iBAAiB,CAAC,OAAO,EAAE,qBAAqB,GAAG,IAAI,CActE;AAED
|
|
1
|
+
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../../../src/backends/aci-standby-pool/index.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAwCG;AAEH,OAAO,KAAK,EAEX,8BAA8B,EAU9B,MAAM,YAAY,CAAA;AAGnB,OAAO,KAAK,EAAE,cAAc,EAAE,qBAAqB,EAAE,MAAM,gBAAgB,CAAA;AAc3E;;;;;GAKG;AACH,MAAM,MAAM,gBAAgB,GAAG,MAAM,OAAO,CAAC,MAAM,CAAC,CAAA;AAEpD,MAAM,WAAW,mCAAmC;IACnD,QAAQ,CAAC,cAAc,EAAE,MAAM,CAAA;IAC/B,QAAQ,CAAC,aAAa,EAAE,MAAM,CAAA;IAC9B,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAA;IACzB;;;;OAIG;IACH,QAAQ,CAAC,qBAAqB,EAAE,MAAM,CAAA;IACtC;;;OAGG;IACH,QAAQ,CAAC,+BAA+B,EAAE,MAAM,CAAA;IAChD;;;OAGG;IACH,QAAQ,CAAC,6BAA6B,CAAC,EAAE,MAAM,CAAA;IAC/C;;;OAGG;IACH,QAAQ,CAAC,MAAM,EAAE,8BAA8B,CAAA;IAC/C;;OAEG;IACH,QAAQ,CAAC,WAAW,EAAE,gBAAgB,CAAA;IACtC;;;;;;OAMG;IACH,QAAQ,CAAC,QAAQ,CAAC,EAAE,MAAM,CAAA;IAC1B;;;;;;;;;;;;;OAaG;IACH,QAAQ,CAAC,kBAAkB,CAAC,EAAE,OAAO,CAAA;IACrC,QAAQ,CAAC,mBAAmB,CAAC,EAAE,MAAM,CAAA;IACrC,QAAQ,CAAC,cAAc,CAAC,EAAE,MAAM,CAAA;IAChC;;OAEG;IACH,QAAQ,CAAC,UAAU,CAAC,EAAE,MAAM,CAAA;IAC5B,QAAQ,CAAC,aAAa,CAAC,EAAE,MAAM,CAAA;IAC/B;;;;;;OAMG;IACH,QAAQ,CAAC,mBAAmB,CAAC,EAAE,MAAM,CAAA;CACrC;AASD;;;;GAIG;AACH,wBAAgB,0BAA0B,CACzC,MAAM,EAAE,mCAAmC,GACzC,cAAc,CAiBhB;AAoMD,wBAAgB,iBAAiB,CAAC,OAAO,EAAE,qBAAqB,GAAG,IAAI,CActE;AAED;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAkCG;AACH,wBAAgB,0BAA0B,CAAC,MAAM,EAAE;IAClD,QAAQ,CAAC,EAAE,MAAM,CAAA;IACjB,kBAAkB,CAAC,EAAE,OAAO,CAAA;CAC5B,GAAG,IAAI,CAOP"}
|
|
@@ -39,8 +39,8 @@
|
|
|
39
39
|
* a pool-side knob, not a backend knob — the backend never
|
|
40
40
|
* chooses; it just PUTs against whichever pool the caller named.
|
|
41
41
|
*/
|
|
42
|
-
import { generateSandboxId, walkFilesViaExec } from '@namzu/sdk';
|
|
43
|
-
import { HttpWorkerClient } from '../http-worker-client.js';
|
|
42
|
+
import { generateSandboxId, walkFilesViaExec, withHint } from '@namzu/sdk';
|
|
43
|
+
import { HttpWorkerClient, STANDBY_POOL_UNAUTHORIZED_HINT } from '../http-worker-client.js';
|
|
44
44
|
import { OperationDeadline, OperationDeadlineExpired, probeHttpHealth, resolveReadinessOptions, runFailureCleanup, } from '../readiness.js';
|
|
45
45
|
import { RemoteCancellationUnknownError, RemoteProtocolError, } from '../remote-execution-controller.js';
|
|
46
46
|
const DEFAULT_READY_POLL_MS = 500;
|
|
@@ -226,10 +226,28 @@ export function assertEnforceable(options) {
|
|
|
226
226
|
* so nothing looks wrong. A caller who never heard of `subnetId` gets a
|
|
227
227
|
* working sandbox on the internet and no signal at all.
|
|
228
228
|
*
|
|
229
|
-
* What is on that address matters: `worker/server.js`
|
|
230
|
-
*
|
|
231
|
-
*
|
|
232
|
-
*
|
|
229
|
+
* What is on that address matters: `worker/server.js` binds every interface
|
|
230
|
+
* and, inside a private network, that was the boundary doing the work — the
|
|
231
|
+
* worker authenticated nobody. With a public address there is no boundary
|
|
232
|
+
* left, and the worker's `/execute` is reachable by anyone.
|
|
233
|
+
*
|
|
234
|
+
* The worker now requires a per-instance `Authorization: Bearer` token on
|
|
235
|
+
* every route but `/healthz`, and REFUSES TO START on a routable bind when
|
|
236
|
+
* it has none. This backend cannot supply one, and the reason this comment
|
|
237
|
+
* used to give for that was wrong: it said the claim API's single admitted
|
|
238
|
+
* override — a config map — is no channel into the container group. It is
|
|
239
|
+
* one, on Linux a file mount under `/mnt/configmap/<containername>/<key>`
|
|
240
|
+
* carrying caller-supplied per-claim values. The credential is declined
|
|
241
|
+
* anyway, for two reasons that survive their source: the worker reads its
|
|
242
|
+
* token from `process.env` at startup and never looks for a file, so a
|
|
243
|
+
* mounted value is not read; and Microsoft's own guidance is that config
|
|
244
|
+
* map values are not validated by the runtime and that values affecting
|
|
245
|
+
* application security "should be made available to the container using
|
|
246
|
+
* environment variables". The full answer, and the change that would close
|
|
247
|
+
* the gap, is in `docs/sdk/container-sandbox-worker.md`. What that means
|
|
248
|
+
* for an operator is written down there too, rather than left to be
|
|
249
|
+
* discovered at startup; this refusal is unchanged, and now stands in
|
|
250
|
+
* front of a worker that would refuse the routable case itself.
|
|
233
251
|
*
|
|
234
252
|
* Defaulting to refusal rather than to a warning, because a warning on a
|
|
235
253
|
* path that otherwise succeeds is read once and never again.
|
|
@@ -239,7 +257,7 @@ export function assertNotPubliclyAddressed(config) {
|
|
|
239
257
|
return;
|
|
240
258
|
if (config.allowPublicAddress)
|
|
241
259
|
return;
|
|
242
|
-
throw new Error('The standby-pool sandbox backend will not claim a container group without `subnetId`: with no subnet the platform assigns a public address,
|
|
260
|
+
throw new Error('The standby-pool sandbox backend will not claim a container group without `subnetId`: with no subnet the platform assigns a public address, so the group is reachable by anything that can route to it, and this backend has no credential to put in front of that. A worker with no token refuses to start on a routable bind at all, and a worker that has one is only as strong as its transport — plain HTTP, where a bearer token is replayable by anything on the path. Supply `subnetId` to inject the group into a private network, or set `allowPublicAddress: true` if this is a benchmark and you mean it.');
|
|
243
261
|
}
|
|
244
262
|
async function spawnAciSandbox(config, options, readiness) {
|
|
245
263
|
options.signal?.throwIfAborted();
|
|
@@ -407,6 +425,9 @@ async function spawnAciSandbox(config, options, readiness) {
|
|
|
407
425
|
encoding: 'base64',
|
|
408
426
|
}),
|
|
409
427
|
});
|
|
428
|
+
if (res.status === 401) {
|
|
429
|
+
throw withHint(new Error(`write-file failed: HTTP 401 ${await res.text()}`), STANDBY_POOL_UNAUTHORIZED_HINT);
|
|
430
|
+
}
|
|
410
431
|
if (!res.ok) {
|
|
411
432
|
throw new Error(`write-file failed: HTTP ${res.status} ${await res.text()}`);
|
|
412
433
|
}
|
|
@@ -430,6 +451,9 @@ async function spawnAciSandbox(config, options, readiness) {
|
|
|
430
451
|
body: JSON.stringify({ path, encoding: 'base64' }),
|
|
431
452
|
signal: options?.signal,
|
|
432
453
|
});
|
|
454
|
+
if (res.status === 401) {
|
|
455
|
+
throw withHint(new Error(`read-file failed: HTTP 401 ${await res.text()}`), STANDBY_POOL_UNAUTHORIZED_HINT);
|
|
456
|
+
}
|
|
433
457
|
if (!res.ok) {
|
|
434
458
|
throw new Error(`read-file failed: HTTP ${res.status} ${await res.text()}`);
|
|
435
459
|
}
|