@namzu/sandbox 1.1.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +234 -0
- package/README.md +205 -105
- package/dist/backends/aci-standby-pool/__tests__/unenforceable-controls.test.d.ts +2 -0
- package/dist/backends/aci-standby-pool/__tests__/unenforceable-controls.test.d.ts.map +1 -0
- package/dist/backends/aci-standby-pool/__tests__/unenforceable-controls.test.js +61 -0
- package/dist/backends/aci-standby-pool/__tests__/unenforceable-controls.test.js.map +1 -0
- package/dist/backends/aci-standby-pool/index.d.ts +2 -1
- package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
- package/dist/backends/aci-standby-pool/index.js +36 -1
- package/dist/backends/aci-standby-pool/index.js.map +1 -1
- package/dist/backends/docker/__tests__/hardening.test.d.ts +2 -0
- package/dist/backends/docker/__tests__/hardening.test.d.ts.map +1 -0
- package/dist/backends/docker/__tests__/hardening.test.js +32 -0
- package/dist/backends/docker/__tests__/hardening.test.js.map +1 -0
- package/dist/backends/docker/__tests__/leaf-permissions.smoke.test.d.ts +1 -1
- package/dist/backends/docker/__tests__/leaf-permissions.smoke.test.js +1 -1
- package/dist/backends/docker/index.d.ts +47 -3
- package/dist/backends/docker/index.d.ts.map +1 -1
- package/dist/backends/docker/index.js +138 -5
- package/dist/backends/docker/index.js.map +1 -1
- package/dist/backends/firecracker/__tests__/agent-timeout-clamp.test.d.ts +16 -0
- package/dist/backends/firecracker/__tests__/agent-timeout-clamp.test.d.ts.map +1 -0
- package/dist/backends/firecracker/__tests__/agent-timeout-clamp.test.js +37 -0
- package/dist/backends/firecracker/__tests__/agent-timeout-clamp.test.js.map +1 -0
- package/dist/backends/firecracker/__tests__/backend.test.js +11 -3
- package/dist/backends/firecracker/__tests__/backend.test.js.map +1 -1
- package/dist/backends/firecracker/__tests__/control-plane-mtls.test.js +10 -2
- package/dist/backends/firecracker/__tests__/control-plane-mtls.test.js.map +1 -1
- package/dist/backends/firecracker/__tests__/egress-policy.test.d.ts +2 -0
- package/dist/backends/firecracker/__tests__/egress-policy.test.d.ts.map +1 -0
- package/dist/backends/firecracker/__tests__/egress-policy.test.js +67 -0
- package/dist/backends/firecracker/__tests__/egress-policy.test.js.map +1 -0
- package/dist/backends/firecracker/__tests__/fixtures/ipc-path.d.ts +21 -0
- package/dist/backends/firecracker/__tests__/fixtures/ipc-path.d.ts.map +1 -0
- package/dist/backends/firecracker/__tests__/fixtures/ipc-path.js +30 -0
- package/dist/backends/firecracker/__tests__/fixtures/ipc-path.js.map +1 -0
- package/dist/backends/firecracker/__tests__/protocol.test.js +7 -17
- package/dist/backends/firecracker/__tests__/protocol.test.js.map +1 -1
- package/dist/backends/firecracker/__tests__/transport.test.js +11 -3
- package/dist/backends/firecracker/__tests__/transport.test.js.map +1 -1
- package/dist/backends/firecracker/index.d.ts +20 -1
- package/dist/backends/firecracker/index.d.ts.map +1 -1
- package/dist/backends/firecracker/index.js +60 -15
- package/dist/backends/firecracker/index.js.map +1 -1
- package/dist/egress/__tests__/allowlist.test.d.ts +2 -0
- package/dist/egress/__tests__/allowlist.test.d.ts.map +1 -0
- package/dist/egress/__tests__/allowlist.test.js +85 -0
- package/dist/egress/__tests__/allowlist.test.js.map +1 -0
- package/dist/egress/__tests__/proxy.test.d.ts +2 -0
- package/dist/egress/__tests__/proxy.test.d.ts.map +1 -0
- package/dist/egress/__tests__/proxy.test.js +177 -0
- package/dist/egress/__tests__/proxy.test.js.map +1 -0
- package/dist/egress/allowlist.d.ts +40 -0
- package/dist/egress/allowlist.d.ts.map +1 -0
- package/dist/egress/allowlist.js +81 -0
- package/dist/egress/allowlist.js.map +1 -0
- package/dist/egress/index.d.ts +4 -0
- package/dist/egress/index.d.ts.map +1 -0
- package/dist/egress/index.js +3 -0
- package/dist/egress/index.js.map +1 -0
- package/dist/egress/proxy.d.ts +90 -0
- package/dist/egress/proxy.d.ts.map +1 -0
- package/dist/egress/proxy.js +194 -0
- package/dist/egress/proxy.js.map +1 -0
- package/dist/index.d.ts +101 -190
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +62 -80
- package/dist/index.js.map +1 -1
- package/dist/index.test.js +18 -39
- package/dist/index.test.js.map +1 -1
- package/package.json +5 -4
- package/src/backends/aci-standby-pool/__tests__/unenforceable-controls.test.ts +69 -0
- package/src/backends/aci-standby-pool/index.ts +42 -1
- package/src/backends/docker/__tests__/hardening.test.ts +43 -0
- package/src/backends/docker/__tests__/leaf-permissions.smoke.test.ts +1 -1
- package/src/backends/docker/index.ts +204 -12
- package/src/backends/firecracker/__tests__/agent-timeout-clamp.test.ts +48 -0
- package/src/backends/firecracker/__tests__/backend.test.ts +76 -65
- package/src/backends/firecracker/__tests__/control-plane-mtls.test.ts +10 -2
- package/src/backends/firecracker/__tests__/egress-policy.test.ts +91 -0
- package/src/backends/firecracker/__tests__/fixtures/ipc-path.ts +31 -0
- package/src/backends/firecracker/__tests__/protocol.test.ts +8 -23
- package/src/backends/firecracker/__tests__/transport.test.ts +11 -3
- package/src/backends/firecracker/index.ts +66 -13
- package/src/egress/__tests__/allowlist.test.ts +103 -0
- package/src/egress/__tests__/proxy.test.ts +212 -0
- package/src/egress/allowlist.ts +82 -0
- package/src/egress/index.ts +7 -0
- package/src/egress/proxy.ts +294 -0
- package/src/index.test.ts +19 -41
- package/src/index.ts +170 -259
|
@@ -40,10 +40,14 @@ import {
|
|
|
40
40
|
type SandboxFileEntry,
|
|
41
41
|
type SandboxId,
|
|
42
42
|
type SandboxStatus,
|
|
43
|
+
withHint,
|
|
43
44
|
} from '@namzu/sdk'
|
|
45
|
+
import { EgressProxy } from '../../egress/index.js'
|
|
46
|
+
import type { BrokeredCredential, RunningEgressProxy } from '../../egress/index.js'
|
|
44
47
|
|
|
45
48
|
import {
|
|
46
49
|
ContainerSandboxLayoutValidationError,
|
|
50
|
+
type EgressPolicy,
|
|
47
51
|
type SandboxBackend,
|
|
48
52
|
type SandboxBackendOptions,
|
|
49
53
|
} from '../../index.js'
|
|
@@ -69,6 +73,30 @@ export interface DockerBackendInternalConfig {
|
|
|
69
73
|
*/
|
|
70
74
|
readonly layout: ResolvedContainerSandboxLayout
|
|
71
75
|
readonly dockerBinary?: string
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* `--user` value for the container, e.g. `'1000:1000'` or `'nobody'`.
|
|
79
|
+
*
|
|
80
|
+
* Left unset by default because the correct uid depends on the image's
|
|
81
|
+
* own filesystem ownership, and forcing one would break every image
|
|
82
|
+
* that expects root at startup. Set it whenever the image supports a
|
|
83
|
+
* non-root user — a container running as root is one bind-mount
|
|
84
|
+
* misconfiguration away from writing the host.
|
|
85
|
+
*/
|
|
86
|
+
readonly runAsUser?: string
|
|
87
|
+
|
|
88
|
+
/**
|
|
89
|
+
* Credentials the egress proxy stamps on, per host.
|
|
90
|
+
*
|
|
91
|
+
* The point is that the real value never enters the sandbox. Any token
|
|
92
|
+
* the agent needs to reach an allowed host used to have to be in the
|
|
93
|
+
* container's environment — readable by the untrusted code it is meant
|
|
94
|
+
* to be isolated from, via `/proc/self/environ` or via a prompt
|
|
95
|
+
* injection that exfiltrates it over the very egress the policy
|
|
96
|
+
* permits. Here it is held host-side and applied at the boundary.
|
|
97
|
+
*/
|
|
98
|
+
readonly brokeredCredentials?: readonly BrokeredCredential[]
|
|
99
|
+
|
|
72
100
|
readonly network?: 'none' | 'bridge' | string
|
|
73
101
|
readonly readyPollIntervalMs?: number
|
|
74
102
|
readonly readyTimeoutMs?: number
|
|
@@ -77,8 +105,8 @@ export interface DockerBackendInternalConfig {
|
|
|
77
105
|
* (vanilla Docker namespaces, what Docker Desktop ships). Linux
|
|
78
106
|
* production deployments that have registered gVisor on the host
|
|
79
107
|
* daemon can pass `runsc` to upgrade to a userspace-kernel trust
|
|
80
|
-
* boundary —
|
|
81
|
-
*
|
|
108
|
+
* boundary — the usual primitive for running untrusted code at
|
|
109
|
+
* scale. Hosts can also pass a custom runtime name registered in
|
|
82
110
|
* `daemon.json`. macOS Docker Desktop has no `runsc` runtime, so
|
|
83
111
|
* the default `runc` is the only option there; that's documented
|
|
84
112
|
* as the local-dev tier in the package README.
|
|
@@ -133,6 +161,81 @@ export function buildDockerBackend(config: DockerBackendInternalConfig): Sandbox
|
|
|
133
161
|
}
|
|
134
162
|
}
|
|
135
163
|
|
|
164
|
+
/**
|
|
165
|
+
* Reconcile the configured docker network with the caller's egress policy.
|
|
166
|
+
*
|
|
167
|
+
* The policy used to be accepted and silently ignored, which is worse than
|
|
168
|
+
* not supporting it: a host that set `deny-all` believed the container had
|
|
169
|
+
* no network and it had the configured one. Docker can enforce `deny-all`
|
|
170
|
+
* natively (`--network none`); it cannot enforce a host allowlist without
|
|
171
|
+
* a proxy this backend does not have, so those policies are REFUSED rather
|
|
172
|
+
* than quietly downgraded to "allow everything".
|
|
173
|
+
*/
|
|
174
|
+
export function resolveNetwork(
|
|
175
|
+
configured: string,
|
|
176
|
+
egress: EgressPolicy | undefined,
|
|
177
|
+
hasProxy = false,
|
|
178
|
+
): string {
|
|
179
|
+
if (!egress) return configured
|
|
180
|
+
|
|
181
|
+
switch (egress.kind) {
|
|
182
|
+
case 'deny-all':
|
|
183
|
+
return 'none'
|
|
184
|
+
case 'allow-all':
|
|
185
|
+
return configured
|
|
186
|
+
default:
|
|
187
|
+
// A host allowlist needs something to filter through. With the
|
|
188
|
+
// egress proxy the container keeps its network and every request
|
|
189
|
+
// crosses that boundary; without one there is nothing to enforce
|
|
190
|
+
// with, and accepting the policy would grant everything while
|
|
191
|
+
// reporting that it had been restricted.
|
|
192
|
+
if (hasProxy) return configured
|
|
193
|
+
throw new Error(
|
|
194
|
+
`The docker sandbox backend cannot enforce an egress policy of kind '${egress.kind}' without an egress proxy: it has nothing to filter hosts through. Construct the provider with one, or use 'deny-all' / 'allow-all'. Refusing rather than silently granting full network access.`,
|
|
195
|
+
)
|
|
196
|
+
}
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
/**
|
|
200
|
+
* Hosts an allowlist policy permits.
|
|
201
|
+
*
|
|
202
|
+
* Only the two filtering kinds reach here. `deny-all` is enforced by the
|
|
203
|
+
* container runtime itself and `allow-all` needs no boundary, so routing
|
|
204
|
+
* either through an allowlist would answer a question nobody asked — and
|
|
205
|
+
* for `allow-all` it would answer "nothing", denying everything.
|
|
206
|
+
*/
|
|
207
|
+
export async function resolveAllowedHosts(egress: EgressPolicy): Promise<readonly string[]> {
|
|
208
|
+
if (egress.kind === 'static') return egress.allowedHosts
|
|
209
|
+
if (egress.kind === 'resolver') return await egress.resolve()
|
|
210
|
+
throw new Error(
|
|
211
|
+
`Egress policy of kind "${egress.kind}" does not describe a host allowlist and must not be routed through the proxy.`,
|
|
212
|
+
)
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
/** Whether a policy needs a boundary before it can be enforced at all. */
|
|
216
|
+
export function needsEgressProxy(egress: EgressPolicy | undefined): boolean {
|
|
217
|
+
return egress?.kind === 'static' || egress?.kind === 'resolver'
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
/**
|
|
221
|
+
* Confinement flags applied to every container.
|
|
222
|
+
*
|
|
223
|
+
* A sandbox whose containers run as root with the full default capability
|
|
224
|
+
* set is not confining much: `CAP_DAC_OVERRIDE` alone walks past the
|
|
225
|
+
* read-only bind mounts the layout sets up, and without
|
|
226
|
+
* `no-new-privileges` a setuid binary inside the image re-escalates. These
|
|
227
|
+
* are the defaults every container runtime hardening guide starts with,
|
|
228
|
+
* and none of them were present.
|
|
229
|
+
*
|
|
230
|
+
* `--cap-drop=ALL` is deliberately not softened by a re-add list: a
|
|
231
|
+
* workload that genuinely needs a capability should say so through
|
|
232
|
+
* `extraRunArgs` and be visible in review.
|
|
233
|
+
*/
|
|
234
|
+
const HARDENING_ARGS: readonly string[] = ['--cap-drop=ALL', '--security-opt=no-new-privileges']
|
|
235
|
+
|
|
236
|
+
/** Name the container reaches the host-side egress proxy by. */
|
|
237
|
+
const PROXY_HOST_ALIAS = 'namzu-egress'
|
|
238
|
+
|
|
136
239
|
async function spawnDockerSandbox(
|
|
137
240
|
config: DockerBackendInternalConfig,
|
|
138
241
|
options: SandboxBackendOptions,
|
|
@@ -140,7 +243,28 @@ async function spawnDockerSandbox(
|
|
|
140
243
|
const resolvedLayout = config.layout
|
|
141
244
|
const id = generateSandboxId()
|
|
142
245
|
const docker = config.dockerBinary ?? DEFAULT_DOCKER_BINARY
|
|
143
|
-
|
|
246
|
+
|
|
247
|
+
// The boundary a host allowlist is actually enforced at. Started before
|
|
248
|
+
// the container so its address can be handed in as proxy environment,
|
|
249
|
+
// and torn down with the sandbox — a proxy holding real credentials
|
|
250
|
+
// must not outlive the thing it was filtering for.
|
|
251
|
+
let egressProxy: RunningEgressProxy | undefined
|
|
252
|
+
if (needsEgressProxy(options.egress) && options.egress) {
|
|
253
|
+
const policy = options.egress
|
|
254
|
+
egressProxy = await new EgressProxy({
|
|
255
|
+
// Re-resolved per request rather than captured once, so a
|
|
256
|
+
// `resolver` policy that rotates is honoured and
|
|
257
|
+
// `setNetworkPolicy` can swap it on a live sandbox.
|
|
258
|
+
allowedHosts: () => resolveAllowedHosts(policy),
|
|
259
|
+
credentials: config.brokeredCredentials ?? [],
|
|
260
|
+
}).listen()
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
const network = resolveNetwork(
|
|
264
|
+
config.network ?? 'none',
|
|
265
|
+
options.egress,
|
|
266
|
+
egressProxy !== undefined,
|
|
267
|
+
)
|
|
144
268
|
const runtime = config.runtime
|
|
145
269
|
const hostReachability = config.hostReachability ?? 'host-port'
|
|
146
270
|
const containerName = `namzu-sandbox-${id}`
|
|
@@ -157,6 +281,17 @@ async function spawnDockerSandbox(
|
|
|
157
281
|
if (containerStarted) {
|
|
158
282
|
await runOnceQuiet(docker, ['rm', '-f', containerName])
|
|
159
283
|
}
|
|
284
|
+
// The proxy starts BEFORE the container and its only other close is
|
|
285
|
+
// in `destroy()`, which a create that never returned can never
|
|
286
|
+
// reach. So every failure between the two — a daemon that is down, a
|
|
287
|
+
// port that could not be read, a worker that missed its readiness
|
|
288
|
+
// deadline, a label the validator rejected — left a listening server
|
|
289
|
+
// on loopback stamping real credential headers, plus a retained
|
|
290
|
+
// event-loop handle, and a retry loop left one per attempt. That is
|
|
291
|
+
// exactly the invariant this file states where the proxy is started:
|
|
292
|
+
// it must not outlive the thing it was filtering for.
|
|
293
|
+
await egressProxy?.close().catch(() => undefined)
|
|
294
|
+
egressProxy = undefined
|
|
160
295
|
}
|
|
161
296
|
|
|
162
297
|
let hostPort: number
|
|
@@ -180,6 +315,8 @@ async function spawnDockerSandbox(
|
|
|
180
315
|
containerName,
|
|
181
316
|
'--network',
|
|
182
317
|
network,
|
|
318
|
+
...HARDENING_ARGS,
|
|
319
|
+
...(config.runAsUser ? ['--user', config.runAsUser] : []),
|
|
183
320
|
]
|
|
184
321
|
|
|
185
322
|
// `--label key=value` flags. Validate first — an empty key or
|
|
@@ -207,6 +344,25 @@ async function spawnDockerSandbox(
|
|
|
207
344
|
// log line. A skill loader that needs the manifest will
|
|
208
345
|
// write it to a bind path the worker reads at startup —
|
|
209
346
|
// avoids env-size limits, keeps the wire shape minimal.
|
|
347
|
+
if (egressProxy) {
|
|
348
|
+
// `host-gateway` is docker's own portable name for the host from
|
|
349
|
+
// inside a container; hard-coding a bridge address would break on
|
|
350
|
+
// every platform whose bridge is numbered differently. The proxy
|
|
351
|
+
// itself binds loopback, so this alias is the only way in.
|
|
352
|
+
args.push('--add-host', `${PROXY_HOST_ALIAS}:host-gateway`)
|
|
353
|
+
const proxyUrl = `http://${PROXY_HOST_ALIAS}:${egressProxy.port}`
|
|
354
|
+
// Both spellings: tooling is split between them, and a workload
|
|
355
|
+
// that reads only the one that is missing bypasses the boundary
|
|
356
|
+
// entirely — which would look exactly like the policy working.
|
|
357
|
+
for (const key of ['HTTP_PROXY', 'http_proxy', 'HTTPS_PROXY', 'https_proxy']) {
|
|
358
|
+
args.push('--env', `${key}=${proxyUrl}`)
|
|
359
|
+
}
|
|
360
|
+
// Loopback must not be proxied, or the worker cannot talk to
|
|
361
|
+
// itself.
|
|
362
|
+
args.push('--env', 'NO_PROXY=localhost,127.0.0.1')
|
|
363
|
+
args.push('--env', 'no_proxy=localhost,127.0.0.1')
|
|
364
|
+
}
|
|
365
|
+
|
|
210
366
|
args.push('--env', `NAMZU_SANDBOX_WORKSPACE=${rootDir}`)
|
|
211
367
|
args.push('--env', `NAMZU_SANDBOX_READ_ROOTS=${renderLayoutReadRootsEnv(resolvedLayout)}`)
|
|
212
368
|
args.push('--env', `NAMZU_SANDBOX_WRITE_ROOTS=${renderLayoutWriteRootsEnv(resolvedLayout)}`)
|
|
@@ -287,6 +443,24 @@ async function spawnDockerSandbox(
|
|
|
287
443
|
}
|
|
288
444
|
},
|
|
289
445
|
|
|
446
|
+
async setNetworkPolicy(policy): Promise<void> {
|
|
447
|
+
// Enforceable only through the egress proxy. Without one the
|
|
448
|
+
// container's network was fixed at creation — `--network none`
|
|
449
|
+
// or not — and there is nothing to narrow: accepting the policy
|
|
450
|
+
// here and doing nothing would leave the caller believing the
|
|
451
|
+
// sandbox had been confined when it had not. Same rule the
|
|
452
|
+
// egress-kind refusal above follows.
|
|
453
|
+
if (!egressProxy) {
|
|
454
|
+
throw withHint(
|
|
455
|
+
new Error(
|
|
456
|
+
'This sandbox cannot change its network policy: it was created without an egress proxy, so its network was fixed at creation and there is nothing to narrow. Refusing rather than accepting a policy that would not be applied.',
|
|
457
|
+
),
|
|
458
|
+
'Construct the provider with an egress proxy to make the policy mutable, or create a second sandbox under the narrower policy.',
|
|
459
|
+
)
|
|
460
|
+
}
|
|
461
|
+
egressProxy.setAllowedHosts(async () => policy.allowedHosts)
|
|
462
|
+
},
|
|
463
|
+
|
|
290
464
|
async writeFile(path: string, content: string | Buffer): Promise<void> {
|
|
291
465
|
const buf = Buffer.isBuffer(content) ? content : Buffer.from(content, 'utf8')
|
|
292
466
|
let res: Response
|
|
@@ -341,6 +515,11 @@ async function spawnDockerSandbox(
|
|
|
341
515
|
async destroy(): Promise<void> {
|
|
342
516
|
status = 'destroyed'
|
|
343
517
|
await runOnceQuiet(docker, ['rm', '-f', containerName])
|
|
518
|
+
// The proxy holds real credentials and a live allowlist. Leaving
|
|
519
|
+
// it listening after the sandbox it was filtering for is gone
|
|
520
|
+
// means a loopback port that still stamps a token onto anything
|
|
521
|
+
// that asks — outliving the only thing that justified it.
|
|
522
|
+
await egressProxy?.close()
|
|
344
523
|
// Backend never allocates host paths — every bind source
|
|
345
524
|
// comes from the consumer-supplied layout. Container
|
|
346
525
|
// teardown is sufficient; the consumer's own lifecycle
|
|
@@ -366,8 +545,11 @@ async function readMappedPort(docker: string, containerName: string): Promise<nu
|
|
|
366
545
|
])
|
|
367
546
|
const port = Number(inspectOutput.trim())
|
|
368
547
|
if (!Number.isInteger(port) || port <= 0 || port > 65535) {
|
|
369
|
-
throw
|
|
370
|
-
|
|
548
|
+
throw withHint(
|
|
549
|
+
new Error(
|
|
550
|
+
`docker inspect returned no usable host port mapping for ${containerName}: '${inspectOutput}'`,
|
|
551
|
+
),
|
|
552
|
+
'The container started but its worker port was never published. Usually the container exited immediately — check its logs — or the host had no free port to bind.',
|
|
371
553
|
)
|
|
372
554
|
}
|
|
373
555
|
return port
|
|
@@ -405,9 +587,12 @@ async function execViaWorker(
|
|
|
405
587
|
: cause
|
|
406
588
|
? String(cause)
|
|
407
589
|
: 'unknown'
|
|
408
|
-
throw
|
|
409
|
-
|
|
410
|
-
|
|
590
|
+
throw withHint(
|
|
591
|
+
new Error(
|
|
592
|
+
`namzu-sandbox /execute fetch failed (baseUrl=${baseUrl}): ${err instanceof Error ? err.message : String(err)} — cause: ${causeMsg}`,
|
|
593
|
+
{ cause: err },
|
|
594
|
+
),
|
|
595
|
+
'The container was reachable when it started, so it has most likely exited or been killed since — an out-of-memory kill under `memoryLimitMb` is the common cause. Check the container logs and its exit code.',
|
|
411
596
|
)
|
|
412
597
|
}
|
|
413
598
|
if (!res.ok || !res.body) {
|
|
@@ -541,10 +726,17 @@ async function waitForWorkerReady(
|
|
|
541
726
|
}
|
|
542
727
|
await new Promise((resolve) => setTimeout(resolve, pollMs))
|
|
543
728
|
}
|
|
544
|
-
throw
|
|
545
|
-
|
|
546
|
-
|
|
547
|
-
|
|
729
|
+
// A hint attached at the throw site, where the cause is actually known.
|
|
730
|
+
// The container runtime's own message says a request failed; it cannot
|
|
731
|
+
// say that the image may not be built or the daemon may not be running,
|
|
732
|
+
// which is what a reader needs.
|
|
733
|
+
throw withHint(
|
|
734
|
+
new Error(
|
|
735
|
+
`namzu-sandbox worker did not become ready within ${timeoutMs}ms: ${
|
|
736
|
+
lastError instanceof Error ? lastError.message : String(lastError)
|
|
737
|
+
}`,
|
|
738
|
+
),
|
|
739
|
+
'Check that the container runtime is running and that the sandbox worker image is built and reachable. A cold image pull can also exceed this window — raise the readiness timeout before assuming the worker is broken.',
|
|
548
740
|
)
|
|
549
741
|
}
|
|
550
742
|
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `resolveTimeoutMs` is the guest agent's only guard against a
|
|
3
|
+
* caller-requested `timeoutMs` that has no schema ceiling upstream (it
|
|
4
|
+
* traces back to the bash tool's model-authored `timeout` argument — see
|
|
5
|
+
* `packages/sdk/src/tools/builtins/bash.ts`). A request above the cap must
|
|
6
|
+
* be refused (the sibling `worker/server.js` answers `invalid_timeout` on
|
|
7
|
+
* response), not silently shortened — a caller that asked for more and got
|
|
8
|
+
* less without being told would believe its process was protected for the
|
|
9
|
+
* duration it actually asked for.
|
|
10
|
+
*
|
|
11
|
+
* Pure-function test, no socket/process involved, so it runs on every
|
|
12
|
+
* platform (unlike `backend.test.ts` / `transport.test.ts`, which spawn
|
|
13
|
+
* `/bin/sh` and skip on Windows).
|
|
14
|
+
*/
|
|
15
|
+
|
|
16
|
+
import { createRequire } from 'node:module'
|
|
17
|
+
|
|
18
|
+
import { describe, expect, it } from 'vitest'
|
|
19
|
+
|
|
20
|
+
const require_ = createRequire(import.meta.url)
|
|
21
|
+
|
|
22
|
+
interface AgentModule {
|
|
23
|
+
resolveTimeoutMs(rawTimeoutMs: unknown): number
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
const agent = require_('../../../../agent/agent.cjs') as AgentModule
|
|
27
|
+
|
|
28
|
+
describe('agent.cjs resolveTimeoutMs', () => {
|
|
29
|
+
it('passes through a caller value within the cap', () => {
|
|
30
|
+
expect(agent.resolveTimeoutMs(10_000)).toBe(10_000)
|
|
31
|
+
})
|
|
32
|
+
|
|
33
|
+
it('falls back to the default for an omitted value', () => {
|
|
34
|
+
expect(agent.resolveTimeoutMs(undefined)).toBe(5 * 60 * 1000)
|
|
35
|
+
})
|
|
36
|
+
|
|
37
|
+
it('refuses a request far above the cap instead of honoring or silently shortening it', () => {
|
|
38
|
+
expect(() => agent.resolveTimeoutMs(Number.MAX_SAFE_INTEGER)).toThrow(/timeoutMs/)
|
|
39
|
+
expect(() => agent.resolveTimeoutMs(10 ** 12)).toThrow(/timeoutMs/)
|
|
40
|
+
})
|
|
41
|
+
|
|
42
|
+
it('refuses a non-finite or non-positive value', () => {
|
|
43
|
+
expect(() => agent.resolveTimeoutMs(0)).toThrow(/timeoutMs/)
|
|
44
|
+
expect(() => agent.resolveTimeoutMs(-1)).toThrow(/timeoutMs/)
|
|
45
|
+
expect(() => agent.resolveTimeoutMs(Number.NaN)).toThrow(/timeoutMs/)
|
|
46
|
+
expect(() => agent.resolveTimeoutMs(Number.POSITIVE_INFINITY)).toThrow(/timeoutMs/)
|
|
47
|
+
})
|
|
48
|
+
})
|
|
@@ -20,6 +20,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
|
|
|
20
20
|
|
|
21
21
|
import { buildFirecrackerBackend, normalizeHandle } from '../index.js'
|
|
22
22
|
import type { WireSandboxAgentHandle } from '../transport.js'
|
|
23
|
+
import { localIpcPath } from './fixtures/ipc-path.js'
|
|
23
24
|
import {
|
|
24
25
|
CA_CRT,
|
|
25
26
|
CLIENT_CRT,
|
|
@@ -27,6 +28,13 @@ import {
|
|
|
27
28
|
type RelayHandle,
|
|
28
29
|
startMtlsRelay,
|
|
29
30
|
} from './fixtures/mtls-pki.js'
|
|
31
|
+
// The loopback agent under test is the guest-side agent for a Linux microVM:
|
|
32
|
+
// it spawns `/bin/sh` to run commands, and that binary does not exist on
|
|
33
|
+
// Windows. These cases assert behavior the platform cannot produce, so they
|
|
34
|
+
// skip there rather than leaving the suite permanently red for Windows
|
|
35
|
+
// contributors. The socket-address fixture IS platform-correct, so the
|
|
36
|
+
// transport is still exercised wherever it can be.
|
|
37
|
+
const IS_WINDOWS = process.platform === 'win32'
|
|
30
38
|
|
|
31
39
|
const require_ = createRequire(import.meta.url)
|
|
32
40
|
|
|
@@ -44,7 +52,7 @@ let realPath: string | undefined
|
|
|
44
52
|
|
|
45
53
|
beforeEach(() => {
|
|
46
54
|
workDir = mkdtempSync(join(tmpdir(), 'fc-backend-test-'))
|
|
47
|
-
sockPath =
|
|
55
|
+
sockPath = localIpcPath(workDir)
|
|
48
56
|
realPath = process.env.PATH
|
|
49
57
|
process.env.NAMZU_SANDBOX_WORKSPACE = workDir
|
|
50
58
|
delete require_.cache[require_.resolve('../../../../agent/agent.cjs')]
|
|
@@ -101,7 +109,7 @@ function stubOrchestrator(handle: WireSandboxAgentHandle): {
|
|
|
101
109
|
return { calls }
|
|
102
110
|
}
|
|
103
111
|
|
|
104
|
-
describe('buildFirecrackerBackend (loopback agent)', () => {
|
|
112
|
+
describe.skipIf(IS_WINDOWS)('buildFirecrackerBackend (loopback agent)', () => {
|
|
105
113
|
it('creates a Sandbox handle and round-trips exec/write/read/listFiles/destroy', async () => {
|
|
106
114
|
server = await startAgent()
|
|
107
115
|
const { calls } = stubOrchestrator({ kind: 'unix', path: sockPath })
|
|
@@ -350,69 +358,72 @@ describe('normalizeHandle (mtls cert injection)', () => {
|
|
|
350
358
|
})
|
|
351
359
|
})
|
|
352
360
|
|
|
353
|
-
describe(
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
361
|
+
describe.skipIf(IS_WINDOWS)(
|
|
362
|
+
'buildFirecrackerBackend (network mode over a loopback mTLS relay)',
|
|
363
|
+
() => {
|
|
364
|
+
it('injects the client cert + round-trips exec/file-IO through the relay', async () => {
|
|
365
|
+
// The agent on a unix socket; a loopback mTLS relay in front of it.
|
|
366
|
+
server = await startAgent()
|
|
367
|
+
relay = await startMtlsRelay(sockPath)
|
|
368
|
+
// The orchestrator returns a WIRE mtls handle (NO cert material) whose
|
|
369
|
+
// host:port point at the relay; the relay resolves the sandboxId.
|
|
370
|
+
const { calls } = stubOrchestrator({
|
|
371
|
+
kind: 'mtls',
|
|
372
|
+
host: '127.0.0.1',
|
|
373
|
+
port: relay.port,
|
|
374
|
+
sandboxId: 'sb-net',
|
|
375
|
+
})
|
|
376
|
+
|
|
377
|
+
const backend = buildFirecrackerBackend({
|
|
378
|
+
orchestratorEndpoint: 'https://orchestrator.test/',
|
|
379
|
+
getToken: async () => 'tok',
|
|
380
|
+
readyTimeoutMs: 5_000,
|
|
381
|
+
readyPollIntervalMs: 50,
|
|
382
|
+
// The CONSUMER-injected client material — never returned by the
|
|
383
|
+
// orchestrator. Merged onto the handle's `tls` block.
|
|
384
|
+
mtls: {
|
|
385
|
+
ca: CA_CRT,
|
|
386
|
+
cert: CLIENT_CRT,
|
|
387
|
+
key: CLIENT_KEY,
|
|
388
|
+
servername: 'sandbox.fc.internal',
|
|
389
|
+
},
|
|
390
|
+
})
|
|
391
|
+
|
|
392
|
+
const sandbox = await backend.create({ workingDirectory: workDir })
|
|
393
|
+
expect(sandbox.status).toBe('ready')
|
|
394
|
+
|
|
395
|
+
// The dialer wrote the `SANDBOX <id>` routing preamble (the relay's key).
|
|
396
|
+
expect(await relay.preamble()).toBe('sb-net')
|
|
397
|
+
|
|
398
|
+
// Full exec + file-IO round-trip over the mTLS tunnel.
|
|
399
|
+
const r = await sandbox.exec('/bin/sh', ['-c', 'echo hello-mtls'])
|
|
400
|
+
expect(r.stdout).toContain('hello-mtls')
|
|
401
|
+
await sandbox.writeFile('out/r.txt', 'via-relay')
|
|
402
|
+
expect((await sandbox.readFile('out/r.txt')).toString('utf8')).toBe('via-relay')
|
|
403
|
+
|
|
404
|
+
await sandbox.destroy()
|
|
405
|
+
expect(calls.some((c) => c.method === 'DELETE')).toBe(true)
|
|
365
406
|
})
|
|
366
407
|
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
expect(await relay.preamble()).toBe('sb-net')
|
|
387
|
-
|
|
388
|
-
// Full exec + file-IO round-trip over the mTLS tunnel.
|
|
389
|
-
const r = await sandbox.exec('/bin/sh', ['-c', 'echo hello-mtls'])
|
|
390
|
-
expect(r.stdout).toContain('hello-mtls')
|
|
391
|
-
await sandbox.writeFile('out/r.txt', 'via-relay')
|
|
392
|
-
expect((await sandbox.readFile('out/r.txt')).toString('utf8')).toBe('via-relay')
|
|
393
|
-
|
|
394
|
-
await sandbox.destroy()
|
|
395
|
-
expect(calls.some((c) => c.method === 'DELETE')).toBe(true)
|
|
396
|
-
})
|
|
397
|
-
|
|
398
|
-
it('rejects a create when the orchestrator returns mtls but no cert material is injected', async () => {
|
|
399
|
-
server = await startAgent()
|
|
400
|
-
relay = await startMtlsRelay(sockPath)
|
|
401
|
-
stubOrchestrator({
|
|
402
|
-
kind: 'mtls',
|
|
403
|
-
host: '127.0.0.1',
|
|
404
|
-
port: relay.port,
|
|
405
|
-
sandboxId: 'sb-net',
|
|
406
|
-
})
|
|
407
|
-
const backend = buildFirecrackerBackend({
|
|
408
|
-
orchestratorEndpoint: 'https://orchestrator.test/',
|
|
409
|
-
getToken: async () => 'tok',
|
|
410
|
-
readyTimeoutMs: 2_000,
|
|
411
|
-
readyPollIntervalMs: 50,
|
|
412
|
-
// No `mtls` material → normalizeHandle must throw at create time.
|
|
408
|
+
it('rejects a create when the orchestrator returns mtls but no cert material is injected', async () => {
|
|
409
|
+
server = await startAgent()
|
|
410
|
+
relay = await startMtlsRelay(sockPath)
|
|
411
|
+
stubOrchestrator({
|
|
412
|
+
kind: 'mtls',
|
|
413
|
+
host: '127.0.0.1',
|
|
414
|
+
port: relay.port,
|
|
415
|
+
sandboxId: 'sb-net',
|
|
416
|
+
})
|
|
417
|
+
const backend = buildFirecrackerBackend({
|
|
418
|
+
orchestratorEndpoint: 'https://orchestrator.test/',
|
|
419
|
+
getToken: async () => 'tok',
|
|
420
|
+
readyTimeoutMs: 2_000,
|
|
421
|
+
readyPollIntervalMs: 50,
|
|
422
|
+
// No `mtls` material → normalizeHandle must throw at create time.
|
|
423
|
+
})
|
|
424
|
+
await expect(backend.create({ workingDirectory: workDir })).rejects.toThrow(
|
|
425
|
+
/no client cert material was injected/,
|
|
426
|
+
)
|
|
413
427
|
})
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
)
|
|
417
|
-
})
|
|
418
|
-
})
|
|
428
|
+
},
|
|
429
|
+
)
|
|
@@ -30,6 +30,7 @@ import { join } from 'node:path'
|
|
|
30
30
|
import { afterEach, beforeEach, describe, expect, it } from 'vitest'
|
|
31
31
|
|
|
32
32
|
import { buildFirecrackerBackend } from '../index.js'
|
|
33
|
+
import { localIpcPath } from './fixtures/ipc-path.js'
|
|
33
34
|
import {
|
|
34
35
|
CA_CRT,
|
|
35
36
|
CLIENT_CRT,
|
|
@@ -39,6 +40,13 @@ import {
|
|
|
39
40
|
SERVER_CRT,
|
|
40
41
|
SERVER_KEY,
|
|
41
42
|
} from './fixtures/mtls-pki.js'
|
|
43
|
+
// The loopback agent under test is the guest-side agent for a Linux microVM:
|
|
44
|
+
// it spawns `/bin/sh` to run commands, and that binary does not exist on
|
|
45
|
+
// Windows. These cases assert behavior the platform cannot produce, so they
|
|
46
|
+
// skip there rather than leaving the suite permanently red for Windows
|
|
47
|
+
// contributors. The socket-address fixture IS platform-correct, so the
|
|
48
|
+
// transport is still exercised wherever it can be.
|
|
49
|
+
const IS_WINDOWS = process.platform === 'win32'
|
|
42
50
|
|
|
43
51
|
const require_ = createRequire(import.meta.url)
|
|
44
52
|
|
|
@@ -54,7 +62,7 @@ let agent: AgentModule
|
|
|
54
62
|
|
|
55
63
|
beforeEach(() => {
|
|
56
64
|
workDir = mkdtempSync(join(tmpdir(), 'fc-cp-mtls-'))
|
|
57
|
-
sockPath =
|
|
65
|
+
sockPath = localIpcPath(workDir)
|
|
58
66
|
process.env.NAMZU_SANDBOX_WORKSPACE = workDir
|
|
59
67
|
delete require_.cache[require_.resolve('../../../../agent/agent.cjs')]
|
|
60
68
|
agent = require_('../../../../agent/agent.cjs') as AgentModule
|
|
@@ -143,7 +151,7 @@ function startMtlsOrchestrator(handlePath: string): Promise<{
|
|
|
143
151
|
})
|
|
144
152
|
}
|
|
145
153
|
|
|
146
|
-
describe('buildFirecrackerBackend (control-plane mTLS dial)', () => {
|
|
154
|
+
describe.skipIf(IS_WINDOWS)('buildFirecrackerBackend (control-plane mTLS dial)', () => {
|
|
147
155
|
it('create POST succeeds over mTLS and round-trips an exec', async () => {
|
|
148
156
|
agentServer = await startAgent()
|
|
149
157
|
const orch = await startMtlsOrchestrator(sockPath)
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
import { describe, expect, it, vi } from 'vitest'
|
|
2
|
+
|
|
3
|
+
import type { SandboxBackendOptions } from '../../../index.js'
|
|
4
|
+
import { resolveEgressAllowlist } from '../index.js'
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* Omitting the allowlist means "no allowlist to apply", which the
|
|
8
|
+
* orchestrator reads as unrestricted. That makes omission the encoding for
|
|
9
|
+
* `allow-all` and only for `allow-all`.
|
|
10
|
+
*
|
|
11
|
+
* `resolver` used to be omitted too — so two opposite intentions shared one
|
|
12
|
+
* encoding, the tenant-scoped allowlist the variant exists for was silently
|
|
13
|
+
* absent, and the callback that would have produced it was never called
|
|
14
|
+
* anywhere in the repo. Whichever way the orchestrator reads an omitted
|
|
15
|
+
* field, one of the two variants was always mis-enforced, and the one that
|
|
16
|
+
* failed OPEN was the one whose entire purpose is restriction.
|
|
17
|
+
*/
|
|
18
|
+
|
|
19
|
+
const opts = (egress?: SandboxBackendOptions['egress']): SandboxBackendOptions => ({
|
|
20
|
+
workingDirectory: '/w',
|
|
21
|
+
...(egress ? { egress } : {}),
|
|
22
|
+
})
|
|
23
|
+
|
|
24
|
+
describe('each policy gets its own encoding', () => {
|
|
25
|
+
it('allow-all omits the field', async () => {
|
|
26
|
+
expect(await resolveEgressAllowlist(opts({ kind: 'allow-all' }))).toBeUndefined()
|
|
27
|
+
})
|
|
28
|
+
|
|
29
|
+
it('deny-all sends an explicitly empty list, not an absent one', async () => {
|
|
30
|
+
expect(await resolveEgressAllowlist(opts({ kind: 'deny-all' }))).toEqual([])
|
|
31
|
+
})
|
|
32
|
+
|
|
33
|
+
it('static forwards the hosts as given', async () => {
|
|
34
|
+
expect(
|
|
35
|
+
await resolveEgressAllowlist(opts({ kind: 'static', allowedHosts: ['a.example'] })),
|
|
36
|
+
).toEqual(['a.example'])
|
|
37
|
+
})
|
|
38
|
+
|
|
39
|
+
it('resolver calls the callback and forwards what it returned', async () => {
|
|
40
|
+
const resolve = vi.fn(async () => ['tenant-a.example', 'tenant-b.example'])
|
|
41
|
+
expect(await resolveEgressAllowlist(opts({ kind: 'resolver', resolve }))).toEqual([
|
|
42
|
+
'tenant-a.example',
|
|
43
|
+
'tenant-b.example',
|
|
44
|
+
])
|
|
45
|
+
// The callback IS the feature. It was declared and never invoked.
|
|
46
|
+
expect(resolve).toHaveBeenCalledOnce()
|
|
47
|
+
})
|
|
48
|
+
|
|
49
|
+
it('never encodes resolver the same way as allow-all', async () => {
|
|
50
|
+
const restrictive = await resolveEgressAllowlist(
|
|
51
|
+
opts({ kind: 'resolver', resolve: async () => ['only-this.example'] }),
|
|
52
|
+
)
|
|
53
|
+
const unrestricted = await resolveEgressAllowlist(opts({ kind: 'allow-all' }))
|
|
54
|
+
expect(restrictive).not.toEqual(unrestricted)
|
|
55
|
+
})
|
|
56
|
+
|
|
57
|
+
it('sends an empty resolver result as a real deny-all', async () => {
|
|
58
|
+
// An empty allowlist is a decision, not an absence. Collapsing it
|
|
59
|
+
// back to omission would turn "this tenant may reach nothing" into
|
|
60
|
+
// "this tenant may reach everything".
|
|
61
|
+
expect(
|
|
62
|
+
await resolveEgressAllowlist(opts({ kind: 'resolver', resolve: async () => [] })),
|
|
63
|
+
).toEqual([])
|
|
64
|
+
})
|
|
65
|
+
|
|
66
|
+
it('omits the field when the host set no policy at all', async () => {
|
|
67
|
+
expect(await resolveEgressAllowlist(opts())).toBeUndefined()
|
|
68
|
+
})
|
|
69
|
+
})
|
|
70
|
+
|
|
71
|
+
describe('an unknown policy', () => {
|
|
72
|
+
it('refuses rather than defaulting to unrestricted', async () => {
|
|
73
|
+
await expect(resolveEgressAllowlist(opts({ kind: 'something-new' } as never))).rejects.toThrow(
|
|
74
|
+
/Refusing rather than defaulting to unrestricted/,
|
|
75
|
+
)
|
|
76
|
+
})
|
|
77
|
+
|
|
78
|
+
it('propagates a failing resolver instead of falling back to open', async () => {
|
|
79
|
+
// A resolver that cannot answer must not become "allow everything".
|
|
80
|
+
await expect(
|
|
81
|
+
resolveEgressAllowlist(
|
|
82
|
+
opts({
|
|
83
|
+
kind: 'resolver',
|
|
84
|
+
resolve: async () => {
|
|
85
|
+
throw new Error('directory unreachable')
|
|
86
|
+
},
|
|
87
|
+
}),
|
|
88
|
+
),
|
|
89
|
+
).rejects.toThrow('directory unreachable')
|
|
90
|
+
})
|
|
91
|
+
})
|