@namzu/sandbox 15.0.0 → 17.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +324 -0
- package/README.md +223 -0
- package/dist/backends/aci-standby-pool/index.d.ts +22 -4
- package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
- package/dist/backends/aci-standby-pool/index.js +31 -7
- package/dist/backends/aci-standby-pool/index.js.map +1 -1
- package/dist/backends/docker/index.d.ts +408 -26
- package/dist/backends/docker/index.d.ts.map +1 -1
- package/dist/backends/docker/index.js +1173 -168
- package/dist/backends/docker/index.js.map +1 -1
- package/dist/backends/firecracker/transport.d.ts +156 -1
- package/dist/backends/firecracker/transport.d.ts.map +1 -1
- package/dist/backends/firecracker/transport.js +223 -29
- package/dist/backends/firecracker/transport.js.map +1 -1
- package/dist/backends/http-worker-client.d.ts +64 -2
- package/dist/backends/http-worker-client.d.ts.map +1 -1
- package/dist/backends/http-worker-client.js +78 -7
- package/dist/backends/http-worker-client.js.map +1 -1
- package/dist/backends/kubernetes/egress-policy.d.ts +193 -102
- package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -1
- package/dist/backends/kubernetes/egress-policy.js +321 -146
- package/dist/backends/kubernetes/egress-policy.js.map +1 -1
- package/dist/backends/kubernetes/per-sandbox-policy.d.ts +6 -6
- package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -1
- package/dist/backends/kubernetes/per-sandbox-policy.js +21 -53
- package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -1
- package/dist/backends/kubernetes/transport.d.ts +7 -0
- package/dist/backends/kubernetes/transport.d.ts.map +1 -1
- package/dist/backends/kubernetes/transport.js.map +1 -1
- package/dist/egress/proxy.d.ts +47 -2
- package/dist/egress/proxy.d.ts.map +1 -1
- package/dist/egress/proxy.js +31 -7
- package/dist/egress/proxy.js.map +1 -1
- package/dist/index.d.ts +130 -6
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +55 -5
- package/dist/index.js.map +1 -1
- package/package.json +4 -4
- package/src/backends/aci-standby-pool/index.ts +37 -7
- package/src/backends/docker/index.ts +1475 -196
- package/src/backends/firecracker/transport.ts +387 -36
- package/src/backends/http-worker-client.ts +89 -5
- package/src/backends/kubernetes/egress-policy.ts +455 -187
- package/src/backends/kubernetes/per-sandbox-policy.ts +21 -66
- package/src/backends/kubernetes/transport.ts +7 -0
- package/src/egress/proxy.ts +65 -8
- package/src/index.ts +162 -5
|
@@ -87,6 +87,7 @@ import {
|
|
|
87
87
|
type KubernetesTranslatedEgressPolicy,
|
|
88
88
|
PER_SANDBOX_NARROWING_REFUSAL,
|
|
89
89
|
assertHostsFitNarrowing,
|
|
90
|
+
assertUsableHost,
|
|
90
91
|
buildCiliumEgressManifest,
|
|
91
92
|
perSandboxEgressLabelKey,
|
|
92
93
|
verifyEgressPolicyApplied,
|
|
@@ -214,18 +215,15 @@ export class KubernetesAdmissionFenceMissingError extends Error {
|
|
|
214
215
|
* Raised for an `allowedHosts` entry this backend will not translate —
|
|
215
216
|
* re-exported rather than defined here.
|
|
216
217
|
*
|
|
217
|
-
* It lives in `egress-policy.ts` beside {@link assertHostsFitNarrowing}
|
|
218
|
-
* because the config-level
|
|
219
|
-
* class for the same entry
|
|
220
|
-
* here would make the two modules import each other. The
|
|
221
|
-
* re-exported so every existing import path — including the
|
|
222
|
-
* keeps resolving.
|
|
218
|
+
* It lives in `egress-policy.ts` beside {@link assertHostsFitNarrowing} and
|
|
219
|
+
* {@link assertUsableHost}, because the config-level translation raises the
|
|
220
|
+
* same class for the same entry and this module imports both checks from
|
|
221
|
+
* there: defining it here would make the two modules import each other. The
|
|
222
|
+
* identifier is re-exported so every existing import path — including the
|
|
223
|
+
* package index — keeps resolving.
|
|
223
224
|
*/
|
|
224
225
|
export { KubernetesNetworkPolicyHostError }
|
|
225
226
|
|
|
226
|
-
/** A DNS name, lowercase, no scheme, no port, no wildcard. */
|
|
227
|
-
const DNS_NAME = /^[a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*$/
|
|
228
|
-
|
|
229
227
|
/**
|
|
230
228
|
* One `allowedHosts` entry, validated and canonicalised to the bytes a
|
|
231
229
|
* `CiliumNetworkPolicy` should carry.
|
|
@@ -235,6 +233,13 @@ const DNS_NAME = /^[a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?
|
|
|
235
233
|
* `matchName` is compared against what the DNS proxy saw — lowercase. The
|
|
236
234
|
* docker backend accepts either case, and a list that worked there and threw
|
|
237
235
|
* here would be a portability trap with no boundary behind it.
|
|
236
|
+
*
|
|
237
|
+
* The grammar itself is {@link assertUsableHost}'s, in `egress-policy.ts`
|
|
238
|
+
* beside the error it throws, because it is no longer only this writer's: the
|
|
239
|
+
* shared translation calls it too, so a config-level allowlist is refused the
|
|
240
|
+
* same entries this path refuses. What is HERE is the part that is this
|
|
241
|
+
* writer's alone — the canonicalisation, applied before the fence is read and
|
|
242
|
+
* before anything is queued behind a previous call.
|
|
238
243
|
*/
|
|
239
244
|
function normalizeHost(entry: string): string {
|
|
240
245
|
if (typeof entry !== 'string') {
|
|
@@ -245,55 +250,6 @@ function normalizeHost(entry: string): string {
|
|
|
245
250
|
return canonical
|
|
246
251
|
}
|
|
247
252
|
|
|
248
|
-
function assertUsableHost(entry: string): void {
|
|
249
|
-
if (typeof entry !== 'string' || entry === '') {
|
|
250
|
-
throw new KubernetesNetworkPolicyHostError(String(entry), 'is empty')
|
|
251
|
-
}
|
|
252
|
-
const bare = entry.startsWith('.') ? entry.slice(1) : entry
|
|
253
|
-
if (bare === '') {
|
|
254
|
-
throw new KubernetesNetworkPolicyHostError(entry, 'names no domain after its leading dot')
|
|
255
|
-
}
|
|
256
|
-
if (entry.includes('*')) {
|
|
257
|
-
throw new KubernetesNetworkPolicyHostError(
|
|
258
|
-
entry,
|
|
259
|
-
"contains a glob; a domain and its subdomains are written with a leading dot ('.example.com'), which becomes matchName plus matchPattern",
|
|
260
|
-
)
|
|
261
|
-
}
|
|
262
|
-
if (!DNS_NAME.test(bare)) {
|
|
263
|
-
throw new KubernetesNetworkPolicyHostError(
|
|
264
|
-
entry,
|
|
265
|
-
'is not a DNS name (a scheme, a path, a port suffix and an IP address all land here; letter case is canonicalised before this check, so it is never the cause)',
|
|
266
|
-
)
|
|
267
|
-
}
|
|
268
|
-
if (bare.length > 253) {
|
|
269
|
-
throw new KubernetesNetworkPolicyHostError(entry, 'is longer than a DNS name may be')
|
|
270
|
-
}
|
|
271
|
-
// A leading-dot entry becomes `matchPattern: '*.<domain>'`, and a
|
|
272
|
-
// single-label domain there is a whole public suffix — `.com`, `.org`.
|
|
273
|
-
// That is not an allowlist entry, and the shipped admission fence refuses
|
|
274
|
-
// the pattern it would produce, so refusing it HERE is what turns an
|
|
275
|
-
// opaque 403 from the API server into an error naming the entry.
|
|
276
|
-
if (entry.startsWith('.') && !bare.includes('.')) {
|
|
277
|
-
throw new KubernetesNetworkPolicyHostError(
|
|
278
|
-
entry,
|
|
279
|
-
"names a whole top-level domain ('.com' means every name under it); a domain entry needs at least two labels, as in '.example.com', and the shipped admission policy refuses the '*.com' pattern this would emit",
|
|
280
|
-
)
|
|
281
|
-
}
|
|
282
|
-
// An IPv4 literal passes the grammar above — every label is digits, and
|
|
283
|
-
// digits are legal in a DNS label. It is still not a hostname: a DNS
|
|
284
|
-
// top-level label is never all-numeric, and Cilium's `matchName` is
|
|
285
|
-
// compared against names the DNS proxy SAW, which an address never is. A
|
|
286
|
-
// policy carrying one is admitted and matches nothing, which reads from
|
|
287
|
-
// outside exactly like a policy that is working.
|
|
288
|
-
const lastLabel = bare.slice(bare.lastIndexOf('.') + 1)
|
|
289
|
-
if (/^[0-9]+$/.test(lastLabel)) {
|
|
290
|
-
throw new KubernetesNetworkPolicyHostError(
|
|
291
|
-
entry,
|
|
292
|
-
'ends in an all-numeric label, so it is an address rather than a hostname; toFQDNs matches names a DNS lookup returned, and an address is never one of them (use config.egress.policy for address-based egress)',
|
|
293
|
-
)
|
|
294
|
-
}
|
|
295
|
-
}
|
|
296
|
-
|
|
297
253
|
/**
|
|
298
254
|
* Raised when the object an acquire created reported no `metadata.uid`.
|
|
299
255
|
*
|
|
@@ -434,7 +390,7 @@ export function buildPerSandboxPolicySetter(
|
|
|
434
390
|
policyKind: 'static',
|
|
435
391
|
...(perSandbox.narrowing !== undefined ? { narrowing: perSandbox.narrowing } : {}),
|
|
436
392
|
ownerReferences: [perSandboxPolicyOwnerReference(owner)],
|
|
437
|
-
|
|
393
|
+
refusalContext: PER_SANDBOX_NARROWING_REFUSAL,
|
|
438
394
|
})
|
|
439
395
|
|
|
440
396
|
const write = async (translated: KubernetesTranslatedEgressPolicy): Promise<void> => {
|
|
@@ -497,13 +453,12 @@ export function buildPerSandboxPolicySetter(
|
|
|
497
453
|
perSandbox.narrowing,
|
|
498
454
|
// The per-sandbox context, from `egress-policy.ts` rather than
|
|
499
455
|
// rebuilt here: it is the SAME value `buildCiliumEgressManifest`
|
|
500
|
-
// refuses by on this path (`
|
|
501
|
-
//
|
|
502
|
-
//
|
|
503
|
-
//
|
|
504
|
-
// config
|
|
505
|
-
//
|
|
506
|
-
// and its subdomains.
|
|
456
|
+
// refuses by on this path (`refusalContext`), so the earlier check
|
|
457
|
+
// and the translation's own cannot name different fields — and the
|
|
458
|
+
// sentence it carries is the true one here, where the entry really
|
|
459
|
+
// does become a name plus a `*.domain` pattern. Sending this caller
|
|
460
|
+
// to `config.egress.ciliumNarrowing` would name a field a
|
|
461
|
+
// `perSandbox` host never set.
|
|
507
462
|
PER_SANDBOX_NARROWING_REFUSAL,
|
|
508
463
|
)
|
|
509
464
|
// A previous call's failure belongs to that caller; this one still runs,
|
|
@@ -530,6 +530,13 @@ export interface KubernetesTransportOptions
|
|
|
530
530
|
* Fires once per completed `exec()` call (success or failure) with
|
|
531
531
|
* the four phase durations above. The payload is exactly those four
|
|
532
532
|
* numbers — never the token, never a command, argv, or output.
|
|
533
|
+
*
|
|
534
|
+
* THIS is the timing hook for this tier. The base options also carry
|
|
535
|
+
* `VsockTransportOptions.onExecTiming`, which is inherited here and is
|
|
536
|
+
* INERT: it belongs to the firecracker tier, whose `exec()` reports into
|
|
537
|
+
* it, while this transport drives the shared wire through
|
|
538
|
+
* `executeStreamed` and never calls that `exec()`. Setting it on a
|
|
539
|
+
* Kubernetes transport starts nothing.
|
|
533
540
|
*/
|
|
534
541
|
readonly onTiming?: (timing: KubernetesTransportTiming) => void
|
|
535
542
|
/**
|
package/src/egress/proxy.ts
CHANGED
|
@@ -71,13 +71,46 @@ export interface EgressProxyOptions {
|
|
|
71
71
|
* reach — which is the hole this screen exists to close.
|
|
72
72
|
*/
|
|
73
73
|
readonly allowInwardFor?: readonly string[]
|
|
74
|
+
/**
|
|
75
|
+
* Address this proxy listens on. Default `127.0.0.1`.
|
|
76
|
+
*
|
|
77
|
+
* The default is loopback because a proxy holding real credentials and
|
|
78
|
+
* bound to every interface is reachable by anything on the network, which
|
|
79
|
+
* is the opposite of what it exists for. There is exactly one deployment
|
|
80
|
+
* where that reasoning flips, and it is the whole reason this is
|
|
81
|
+
* configurable: the docker tier runs the proxy as a container of its own
|
|
82
|
+
* (`egress-proxy/server.mjs`), the sandbox is attached to an `--internal`
|
|
83
|
+
* network with no route out, and this container is the only way its traffic
|
|
84
|
+
* reaches the internet. There, `0.0.0.0` inside the proxy container is not
|
|
85
|
+
* "every interface on the host" — the container IS the boundary — and what
|
|
86
|
+
* else the sandbox can reach on that network is whatever other containers a
|
|
87
|
+
* host attached to it, which `docs/sdk/sandbox-egress.md` states.
|
|
88
|
+
*/
|
|
89
|
+
readonly bindHost?: string
|
|
90
|
+
/**
|
|
91
|
+
* Names that denote THIS proxy, for the loop guard in `listen`.
|
|
92
|
+
*
|
|
93
|
+
* `isSelf` refuses a request whose target is the proxy itself, because
|
|
94
|
+
* forwarding it makes the proxy call itself until the process runs out of
|
|
95
|
+
* sockets — a failure that arrives as a hang rather than as an error. On
|
|
96
|
+
* loopback the target is one of `127.0.0.1`, `localhost`, `::1` and the
|
|
97
|
+
* check is closed by construction. Bound to another address it is not:
|
|
98
|
+
* anything that names this proxy by a name it answers to — its container
|
|
99
|
+
* name or network alias, say — would walk straight into that loop. So
|
|
100
|
+
* every such name is listed here by whoever knows it.
|
|
101
|
+
*/
|
|
102
|
+
readonly selfNames?: readonly string[]
|
|
74
103
|
/** Injected in tests. Defaults to the platform resolver. */
|
|
75
104
|
readonly resolveAddresses?: ScreeningLookupOptions['resolve']
|
|
76
105
|
}
|
|
77
106
|
|
|
78
107
|
export interface RunningEgressProxy {
|
|
79
108
|
readonly port: number
|
|
80
|
-
/**
|
|
109
|
+
/**
|
|
110
|
+
* `http://<the address this proxy bound>:<port>` — what a sandbox sets
|
|
111
|
+
* `HTTP_PROXY` to. Loopback by default; the address is whatever
|
|
112
|
+
* {@link EgressProxyOptions.bindHost} said.
|
|
113
|
+
*/
|
|
81
114
|
readonly url: string
|
|
82
115
|
/** Swap the allowlist on a live proxy. See `setNetworkPolicy`. */
|
|
83
116
|
setAllowedHosts(resolve: () => Promise<readonly string[]>): void
|
|
@@ -86,6 +119,12 @@ export interface RunningEgressProxy {
|
|
|
86
119
|
|
|
87
120
|
const DENIED_STATUS = 403
|
|
88
121
|
|
|
122
|
+
/** The default bind address. See {@link EgressProxyOptions.bindHost}. */
|
|
123
|
+
const LOOPBACK_BIND_HOST = '127.0.0.1'
|
|
124
|
+
|
|
125
|
+
/** Names that mean "this proxy" on loopback, whatever it bound to. */
|
|
126
|
+
const LOOPBACK_SELF_NAMES: readonly string[] = ['127.0.0.1', 'localhost', '::1']
|
|
127
|
+
|
|
89
128
|
export class EgressProxy {
|
|
90
129
|
private resolveAllowed: () => Promise<readonly string[]>
|
|
91
130
|
private readonly credentials: readonly BrokeredCredential[]
|
|
@@ -101,6 +140,8 @@ export class EgressProxy {
|
|
|
101
140
|
*/
|
|
102
141
|
private readonly lookup: ReturnType<typeof createScreeningLookup>
|
|
103
142
|
private readonly inwardAllowed: readonly string[]
|
|
143
|
+
private readonly bindHost: string
|
|
144
|
+
private readonly selfNames: readonly string[]
|
|
104
145
|
|
|
105
146
|
constructor(options: EgressProxyOptions) {
|
|
106
147
|
this.resolveAllowed = options.allowedHosts
|
|
@@ -108,6 +149,8 @@ export class EgressProxy {
|
|
|
108
149
|
this.upgradeToHttps = options.upgradeToHttps ?? true
|
|
109
150
|
this.onDenied = options.onDenied
|
|
110
151
|
this.inwardAllowed = options.allowInwardFor ?? []
|
|
152
|
+
this.bindHost = options.bindHost ?? LOOPBACK_BIND_HOST
|
|
153
|
+
this.selfNames = options.selfNames ?? []
|
|
111
154
|
this.lookup = createScreeningLookup(
|
|
112
155
|
{
|
|
113
156
|
...(options.allowInwardFor ? { allowInwardFor: options.allowInwardFor } : {}),
|
|
@@ -131,10 +174,12 @@ export class EgressProxy {
|
|
|
131
174
|
})
|
|
132
175
|
await new Promise<void>((resolve, reject) => {
|
|
133
176
|
server.once('error', reject)
|
|
134
|
-
// Loopback
|
|
135
|
-
// every interface is reachable by anything on the network, which
|
|
136
|
-
//
|
|
137
|
-
|
|
177
|
+
// Loopback by default. A proxy that holds real credentials and binds
|
|
178
|
+
// every interface is reachable by anything on the network, which is
|
|
179
|
+
// the opposite of what it exists for — see
|
|
180
|
+
// {@link EgressProxyOptions.bindHost} for the one deployment where
|
|
181
|
+
// the container itself is the boundary and this is not that.
|
|
182
|
+
server.listen(port, this.bindHost, () => {
|
|
138
183
|
server.off('error', reject)
|
|
139
184
|
resolve()
|
|
140
185
|
})
|
|
@@ -153,7 +198,7 @@ export class EgressProxy {
|
|
|
153
198
|
|
|
154
199
|
return {
|
|
155
200
|
port: boundPort,
|
|
156
|
-
url: `http
|
|
201
|
+
url: `http://${this.bindHost}:${boundPort}`,
|
|
157
202
|
setAllowedHosts: (resolve) => {
|
|
158
203
|
this.resolveAllowed = resolve
|
|
159
204
|
},
|
|
@@ -363,10 +408,22 @@ export class EgressProxy {
|
|
|
363
408
|
})
|
|
364
409
|
}
|
|
365
410
|
|
|
366
|
-
/**
|
|
411
|
+
/**
|
|
412
|
+
* Whether a target names this proxy. See the loop guard in `listen`.
|
|
413
|
+
*
|
|
414
|
+
* The loopback spellings are always this proxy — a client inside the same
|
|
415
|
+
* network namespace reaches it at one of them whatever it bound to, and
|
|
416
|
+
* the bound address when that address is a literal is one more. Anything
|
|
417
|
+
* else has to be named by the caller, through
|
|
418
|
+
* {@link EgressProxyOptions.selfNames}: this process cannot know which
|
|
419
|
+
* names resolve to it on the network it is attached to, and a guard that
|
|
420
|
+
* guessed would either miss the loop or refuse a legitimate target.
|
|
421
|
+
*/
|
|
367
422
|
private isSelf(host: string, port?: number): boolean {
|
|
368
423
|
if (port === undefined || port !== this.selfPort) return false
|
|
369
|
-
|
|
424
|
+
if (host === this.bindHost) return true
|
|
425
|
+
if (LOOPBACK_SELF_NAMES.includes(host)) return true
|
|
426
|
+
return this.selfNames.includes(host)
|
|
370
427
|
}
|
|
371
428
|
|
|
372
429
|
private credentialFor(host: string): BrokeredCredential | undefined {
|
package/src/index.ts
CHANGED
|
@@ -9,11 +9,29 @@
|
|
|
9
9
|
*
|
|
10
10
|
* Two tiers, each a trust boundary:
|
|
11
11
|
*
|
|
12
|
-
* • `container` — one OCI container per task,
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
12
|
+
* • `container` — one OCI container per task, with the container itself
|
|
13
|
+
* as the boundary: kernel namespaces, or a userspace-kernel runtime
|
|
14
|
+
* where one is installed. The same path on a laptop and on a Linux
|
|
15
|
+
* replica anywhere. The tier for trusted prompts and contained
|
|
16
|
+
* workloads. What confines the workload INSIDE the container is
|
|
17
|
+
* applied by each backend and documented where it is applied, not
|
|
18
|
+
* promised here: `container:docker` drops every capability, sets
|
|
19
|
+
* no-new-privileges, gives the container an IPC namespace nothing else can
|
|
20
|
+
* join and mounts its root filesystem read-only over a named writable set
|
|
21
|
+
* (see `backends/docker/index.ts`), while the ACI standby pool takes
|
|
22
|
+
* its controls from the container-group profile its pool was built
|
|
23
|
+
* from. This line used to claim "seccomp on, tmpfs workdir, no
|
|
24
|
+
* network unless asked" on both their behalf, and none of the three
|
|
25
|
+
* is a property this package establishes: nothing here passes a
|
|
26
|
+
* seccomp flag, so what filters a container's syscalls is the daemon's
|
|
27
|
+
* own profile rather than a default this package sets; the directory
|
|
28
|
+
* the agent works in is the layout's `outputs` bind mount — a host
|
|
29
|
+
* directory the run is collected from — rather than a tmpfs that would
|
|
30
|
+
* lose it when the container exits; and the network a container is
|
|
31
|
+
* attached to is a fact about
|
|
32
|
+
* the backend's own configuration — a daemon network whose internals
|
|
33
|
+
* the egress policy is checked against, or the container group's
|
|
34
|
+
* subnet or public address.
|
|
17
35
|
*
|
|
18
36
|
* • `microvm` — one hardware-virtualized guest per task. The boundary
|
|
19
37
|
* to reach for when the prompt itself is adversarial, at the cost
|
|
@@ -44,6 +62,7 @@ import type {
|
|
|
44
62
|
import { buildAciStandbyPoolBackend } from './backends/aci-standby-pool/index.js'
|
|
45
63
|
import { buildDockerBackend, resolveLayout } from './backends/docker/index.js'
|
|
46
64
|
import { buildFirecrackerBackend } from './backends/firecracker/index.js'
|
|
65
|
+
import type { FirecrackerTransportTiming } from './backends/firecracker/transport.js'
|
|
47
66
|
import type { KubernetesEgressConfig } from './backends/kubernetes/egress-policy.js'
|
|
48
67
|
import {
|
|
49
68
|
type KubernetesAgentAddressMode,
|
|
@@ -108,6 +127,7 @@ export {
|
|
|
108
127
|
AgentWriteFileTooLargeError,
|
|
109
128
|
DEFAULT_MAX_WRITE_FILE_BYTES,
|
|
110
129
|
FIRECRACKER_AGENT_PROTOCOL_VERSION,
|
|
130
|
+
type FirecrackerTransportTiming,
|
|
111
131
|
GUEST_FRAME_LIMIT_BYTES,
|
|
112
132
|
type SandboxAgentHandle,
|
|
113
133
|
TCP_PREAUTH_FRAME_LIMIT_BYTES,
|
|
@@ -641,8 +661,51 @@ export interface ContainerBackendConfig {
|
|
|
641
661
|
*
|
|
642
662
|
* Egress from the sandbox is governed separately by `EgressPolicy`,
|
|
643
663
|
* which is checked against this network rather than trusted.
|
|
664
|
+
*
|
|
665
|
+
* An egress policy of `static` or `resolver` — a host allowlist —
|
|
666
|
+
* additionally REQUIRES this network to be `--internal`, and requires
|
|
667
|
+
* `hostReachability: 'container-network'`. The allowlist is enforced by
|
|
668
|
+
* the egress proxy running as a sibling container on this network: the
|
|
669
|
+
* sandbox is attached to this network alone, and an internal network has
|
|
670
|
+
* no route off it, so the only way the sandbox's traffic reaches the
|
|
671
|
+
* internet is through that container. It is a container on a subnet like
|
|
672
|
+
* any other there, so what else a host attaches to this network — a second
|
|
673
|
+
* sandbox, and that sandbox's proxy — is reachable from this one too; on a
|
|
674
|
+
* network with a route out the allowlist is a proxy environment variable a
|
|
675
|
+
* workload may decline to read, and `create()` refuses rather than
|
|
676
|
+
* reporting a boundary that is not there.
|
|
644
677
|
*/
|
|
645
678
|
readonly network?: 'none' | 'bridge' | string
|
|
679
|
+
/**
|
|
680
|
+
* Image the egress proxy runs as, when the policy needs one.
|
|
681
|
+
*
|
|
682
|
+
* Required for a `static` or `resolver` policy, refused without it. The
|
|
683
|
+
* image is `packages/sandbox/egress-proxy/Dockerfile`:
|
|
684
|
+
*
|
|
685
|
+
* pnpm --filter @namzu/sandbox build
|
|
686
|
+
* docker build -f packages/sandbox/egress-proxy/Dockerfile \
|
|
687
|
+
* -t <tag> packages/sandbox
|
|
688
|
+
*
|
|
689
|
+
* It is a second image rather than the sandbox's own because the
|
|
690
|
+
* sandbox image is a string this backend cannot read: there is no way to
|
|
691
|
+
* know whether it contains the proxy module, and the bind-mount
|
|
692
|
+
* alternative fails on exactly the remote-daemon deployment
|
|
693
|
+
* `hostReachability: 'container-network'` exists for. `deny-all` and
|
|
694
|
+
* `allow-all` need no image.
|
|
695
|
+
*/
|
|
696
|
+
readonly egressProxyImage?: string
|
|
697
|
+
/**
|
|
698
|
+
* Network the egress proxy joins for its route to the internet. Default
|
|
699
|
+
* `'bridge'`, docker's own default bridge.
|
|
700
|
+
*
|
|
701
|
+
* The proxy is dual-homed: it sits on the internal network the sandbox
|
|
702
|
+
* is on, and on this one, which is how it reaches the world. Name a
|
|
703
|
+
* dedicated network when the daemon is shared, because anything else
|
|
704
|
+
* attached to this one can reach the proxy — and this proxy enforces its
|
|
705
|
+
* allowlist for whoever asks and stamps brokered credentials on what it
|
|
706
|
+
* forwards.
|
|
707
|
+
*/
|
|
708
|
+
readonly egressProxyUpstreamNetwork?: string
|
|
646
709
|
/**
|
|
647
710
|
* Maximum time spent waiting for the container worker's `/healthz`
|
|
648
711
|
* readiness probe. Must be a positive integer within Node's timer range.
|
|
@@ -685,6 +748,46 @@ export interface ContainerBackendConfig {
|
|
|
685
748
|
* collisions with Docker / orchestrator labels.
|
|
686
749
|
*/
|
|
687
750
|
readonly labels?: Readonly<Record<string, string>>
|
|
751
|
+
/**
|
|
752
|
+
* CPU cores the container may use, rendered as `--cpus`. Unset by
|
|
753
|
+
* default, like `memoryLimitMb` and `maxProcesses`, and for the same
|
|
754
|
+
* reason: the value that is right is a property of the host's machine
|
|
755
|
+
* and of the workload, and a number chosen here would silently throttle
|
|
756
|
+
* runs that finish inside their timeout today.
|
|
757
|
+
*
|
|
758
|
+
* It is set at provider construction rather than per `create()` call,
|
|
759
|
+
* because the documented deployment builds one provider per task — and
|
|
760
|
+
* because the ACI and kubernetes backends cannot apply a per-sandbox CPU
|
|
761
|
+
* limit, so a per-call field would be a control they would have to
|
|
762
|
+
* refuse. See `backends/docker/index.ts` for what it renders.
|
|
763
|
+
*/
|
|
764
|
+
readonly cpuLimit?: number
|
|
765
|
+
/**
|
|
766
|
+
* Mount the container's root filesystem read-only. Default `true`.
|
|
767
|
+
*
|
|
768
|
+
* On by default with the paths that stay writable named in
|
|
769
|
+
* `backends/docker/index.ts` (`writableRootfsPaths` extends them). Set it
|
|
770
|
+
* to `false` to make the whole container filesystem writable again, which a
|
|
771
|
+
* host whose image writes somewhere the writable set cannot describe needs,
|
|
772
|
+
* and which is why the switch exists rather than the baseline being
|
|
773
|
+
* unconditional. It gives up that one control: the capability drop,
|
|
774
|
+
* `no-new-privileges` and `--ipc private` are applied to every container
|
|
775
|
+
* whatever this says, so it is not a way back to the previous argv.
|
|
776
|
+
*/
|
|
777
|
+
readonly readOnlyRootfs?: boolean
|
|
778
|
+
/**
|
|
779
|
+
* Extra paths to keep writable under `--read-only`, each mounted
|
|
780
|
+
* `--tmpfs`.
|
|
781
|
+
*
|
|
782
|
+
* The default set is the reference image's needs, read off its Dockerfile.
|
|
783
|
+
* A host that points `image` at its own build says what that image needs
|
|
784
|
+
* here, because the backend cannot read an image's writable set and the
|
|
785
|
+
* alternative to asking is guessing. Setting this beside
|
|
786
|
+
* `readOnlyRootfs: false` is refused: with a writable root filesystem the
|
|
787
|
+
* mounts would add nothing, and accepting a control that is not applied is
|
|
788
|
+
* the failure this package refuses everywhere else.
|
|
789
|
+
*/
|
|
790
|
+
readonly writableRootfsPaths?: readonly string[]
|
|
688
791
|
}
|
|
689
792
|
|
|
690
793
|
/**
|
|
@@ -738,6 +841,28 @@ export type MicroVMBackendConfig = {
|
|
|
738
841
|
readonly readyTimeoutMs?: number
|
|
739
842
|
/** Delay between guest-agent health probes. Default 250ms. */
|
|
740
843
|
readonly readyPollIntervalMs?: number
|
|
844
|
+
/**
|
|
845
|
+
* Fires once per completed `exec()` on this provider's sandboxes with
|
|
846
|
+
* that call's wall-time breakdown — the reserve round trip, the execute
|
|
847
|
+
* round trip, the dials inside them, the first reply frame, the
|
|
848
|
+
* terminator frame and the peer's own close. See
|
|
849
|
+
* {@link FirecrackerTransportTiming} for what each number is and is not,
|
|
850
|
+
* and {@link VsockTransportOptions.onExecTiming} for why the field is
|
|
851
|
+
* named this rather than the kubernetes tier's `onTiming`.
|
|
852
|
+
*
|
|
853
|
+
* Opt-in and additive, with no default: a host that sets nothing sends,
|
|
854
|
+
* receives and waits for exactly what it did before. The intended use is
|
|
855
|
+
* attribution — a relay or a guest that adds a fixed cost to every call
|
|
856
|
+
* moves a named phase, and one that adds none leaves the numbers at the
|
|
857
|
+
* cost of a command's own runtime.
|
|
858
|
+
*
|
|
859
|
+
* It is an OBSERVER, not a control: it cannot change a command's result,
|
|
860
|
+
* it is called after the call has settled, and a listener that throws is
|
|
861
|
+
* that listener's problem. The payload is durations only — never the
|
|
862
|
+
* agent token, a command, its arguments, or any output — so a host may
|
|
863
|
+
* log it without leaking what the sandbox ran.
|
|
864
|
+
*/
|
|
865
|
+
readonly onExecTiming?: (timing: FirecrackerTransportTiming) => void
|
|
741
866
|
/**
|
|
742
867
|
* NETWORK-mode mTLS client material (ses_051 P4 client-proxy
|
|
743
868
|
* bridge). When present, the orchestrator returns an `mtls` agent
|
|
@@ -1259,7 +1384,18 @@ function pickBackend(config: SandboxProviderConfig): SandboxBackend {
|
|
|
1259
1384
|
: {}),
|
|
1260
1385
|
...(backend.network !== undefined ? { network: backend.network } : {}),
|
|
1261
1386
|
...(backend.allowInwardFor !== undefined ? { allowInwardFor: backend.allowInwardFor } : {}),
|
|
1387
|
+
...(backend.egressProxyImage !== undefined
|
|
1388
|
+
? { egressProxyImage: backend.egressProxyImage }
|
|
1389
|
+
: {}),
|
|
1390
|
+
...(backend.egressProxyUpstreamNetwork !== undefined
|
|
1391
|
+
? { egressProxyUpstreamNetwork: backend.egressProxyUpstreamNetwork }
|
|
1392
|
+
: {}),
|
|
1262
1393
|
...(backend.labels !== undefined ? { labels: backend.labels } : {}),
|
|
1394
|
+
...(backend.cpuLimit !== undefined ? { cpuLimit: backend.cpuLimit } : {}),
|
|
1395
|
+
...(backend.readOnlyRootfs !== undefined ? { readOnlyRootfs: backend.readOnlyRootfs } : {}),
|
|
1396
|
+
...(backend.writableRootfsPaths !== undefined
|
|
1397
|
+
? { writableRootfsPaths: backend.writableRootfsPaths }
|
|
1398
|
+
: {}),
|
|
1263
1399
|
})
|
|
1264
1400
|
}
|
|
1265
1401
|
if (backend.tier === 'container' && backend.runtime === 'runsc') {
|
|
@@ -1279,7 +1415,18 @@ function pickBackend(config: SandboxProviderConfig): SandboxBackend {
|
|
|
1279
1415
|
: {}),
|
|
1280
1416
|
...(backend.network !== undefined ? { network: backend.network } : {}),
|
|
1281
1417
|
...(backend.allowInwardFor !== undefined ? { allowInwardFor: backend.allowInwardFor } : {}),
|
|
1418
|
+
...(backend.egressProxyImage !== undefined
|
|
1419
|
+
? { egressProxyImage: backend.egressProxyImage }
|
|
1420
|
+
: {}),
|
|
1421
|
+
...(backend.egressProxyUpstreamNetwork !== undefined
|
|
1422
|
+
? { egressProxyUpstreamNetwork: backend.egressProxyUpstreamNetwork }
|
|
1423
|
+
: {}),
|
|
1282
1424
|
...(backend.labels !== undefined ? { labels: backend.labels } : {}),
|
|
1425
|
+
...(backend.cpuLimit !== undefined ? { cpuLimit: backend.cpuLimit } : {}),
|
|
1426
|
+
...(backend.readOnlyRootfs !== undefined ? { readOnlyRootfs: backend.readOnlyRootfs } : {}),
|
|
1427
|
+
...(backend.writableRootfsPaths !== undefined
|
|
1428
|
+
? { writableRootfsPaths: backend.writableRootfsPaths }
|
|
1429
|
+
: {}),
|
|
1283
1430
|
})
|
|
1284
1431
|
}
|
|
1285
1432
|
// `microvm:self-hosted` targeting the OWNED Azure Firecracker
|
|
@@ -1304,6 +1451,16 @@ function pickBackend(config: SandboxProviderConfig): SandboxBackend {
|
|
|
1304
1451
|
...(backend.readyPollIntervalMs !== undefined
|
|
1305
1452
|
? { readyPollIntervalMs: backend.readyPollIntervalMs }
|
|
1306
1453
|
: {}),
|
|
1454
|
+
// Forwarded into the transport's own options rather than onto the
|
|
1455
|
+
// backend's config: the backend reads nothing here, and the phase
|
|
1456
|
+
// numbers are the transport's to report — see
|
|
1457
|
+
// `VsockTransportOptions.onExecTiming`, which is where a host that
|
|
1458
|
+
// builds a `VsockAgentTransport` directly sets it too. Absent ⇒
|
|
1459
|
+
// this spread adds no key at all and the transport is built with
|
|
1460
|
+
// the options it was always built with.
|
|
1461
|
+
...(backend.onExecTiming !== undefined
|
|
1462
|
+
? { transport: { onExecTiming: backend.onExecTiming } }
|
|
1463
|
+
: {}),
|
|
1307
1464
|
...(backend.mtls !== undefined ? { mtls: backend.mtls } : {}),
|
|
1308
1465
|
...(backend.controlPlaneMtls !== undefined
|
|
1309
1466
|
? { controlPlaneMtls: backend.controlPlaneMtls }
|