@namzu/sandbox 13.0.0 → 14.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +309 -0
- package/README.md +151 -0
- package/dist/backends/firecracker/protocol.d.ts +22 -0
- package/dist/backends/firecracker/protocol.d.ts.map +1 -1
- package/dist/backends/firecracker/protocol.js.map +1 -1
- package/dist/backends/firecracker/transport.d.ts +104 -9
- package/dist/backends/firecracker/transport.d.ts.map +1 -1
- package/dist/backends/firecracker/transport.js +139 -13
- package/dist/backends/firecracker/transport.js.map +1 -1
- package/dist/backends/kubernetes/egress-policy.d.ts +219 -0
- package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -0
- package/dist/backends/kubernetes/egress-policy.js +314 -0
- package/dist/backends/kubernetes/egress-policy.js.map +1 -0
- package/dist/backends/kubernetes/index.d.ts +374 -0
- package/dist/backends/kubernetes/index.d.ts.map +1 -0
- package/dist/backends/kubernetes/index.js +671 -0
- package/dist/backends/kubernetes/index.js.map +1 -0
- package/dist/backends/kubernetes/k8s-client.d.ts +125 -0
- package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -0
- package/dist/backends/kubernetes/k8s-client.js +246 -0
- package/dist/backends/kubernetes/k8s-client.js.map +1 -0
- package/dist/backends/kubernetes/lease.d.ts +119 -0
- package/dist/backends/kubernetes/lease.d.ts.map +1 -0
- package/dist/backends/kubernetes/lease.js +151 -0
- package/dist/backends/kubernetes/lease.js.map +1 -0
- package/dist/backends/kubernetes/objects.d.ts +282 -0
- package/dist/backends/kubernetes/objects.d.ts.map +1 -0
- package/dist/backends/kubernetes/objects.js +156 -0
- package/dist/backends/kubernetes/objects.js.map +1 -0
- package/dist/backends/kubernetes/privilege-probe.d.ts +136 -0
- package/dist/backends/kubernetes/privilege-probe.d.ts.map +1 -0
- package/dist/backends/kubernetes/privilege-probe.js +185 -0
- package/dist/backends/kubernetes/privilege-probe.js.map +1 -0
- package/dist/backends/kubernetes/sandbox.d.ts +123 -0
- package/dist/backends/kubernetes/sandbox.d.ts.map +1 -0
- package/dist/backends/kubernetes/sandbox.js +299 -0
- package/dist/backends/kubernetes/sandbox.js.map +1 -0
- package/dist/backends/kubernetes/transport.d.ts +122 -0
- package/dist/backends/kubernetes/transport.d.ts.map +1 -0
- package/dist/backends/kubernetes/transport.js +197 -0
- package/dist/backends/kubernetes/transport.js.map +1 -0
- package/dist/backends/kubernetes/workspace.d.ts +381 -0
- package/dist/backends/kubernetes/workspace.d.ts.map +1 -0
- package/dist/backends/kubernetes/workspace.js +1064 -0
- package/dist/backends/kubernetes/workspace.js.map +1 -0
- package/dist/index.d.ts +132 -2
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +102 -34
- package/dist/index.js.map +1 -1
- package/dist/testing/sandbox-conformance.d.ts +193 -0
- package/dist/testing/sandbox-conformance.d.ts.map +1 -0
- package/dist/testing/sandbox-conformance.js +465 -0
- package/dist/testing/sandbox-conformance.js.map +1 -0
- package/package.json +5 -4
- package/src/backends/firecracker/protocol.ts +27 -0
- package/src/backends/firecracker/transport.ts +199 -28
- package/src/backends/kubernetes/egress-policy.ts +437 -0
- package/src/backends/kubernetes/index.ts +1012 -0
- package/src/backends/kubernetes/k8s-client.ts +352 -0
- package/src/backends/kubernetes/lease.ts +198 -0
- package/src/backends/kubernetes/objects.ts +363 -0
- package/src/backends/kubernetes/privilege-probe.ts +261 -0
- package/src/backends/kubernetes/sandbox.ts +395 -0
- package/src/backends/kubernetes/transport.ts +286 -0
- package/src/backends/kubernetes/workspace.ts +1386 -0
- package/src/index.ts +257 -35
- package/src/testing/sandbox-conformance.ts +667 -0
|
@@ -0,0 +1,1012 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Kubernetes / agent-sandbox backend — acquire, readiness and teardown.
|
|
3
|
+
*
|
|
4
|
+
* Sibling of `firecracker/` and `aci-standby-pool/`: same
|
|
5
|
+
* {@link SandboxBackend} surface, same "the host supplies the credential
|
|
6
|
+
* callback and this package carries no cloud SDK" boundary, a different
|
|
7
|
+
* control plane. Here the control plane is the Kubernetes API server and the
|
|
8
|
+
* warm pool is a `SandboxWarmPool` reconciled by the agent-sandbox controller
|
|
9
|
+
* (kubernetes-sigs/agent-sandbox v1.0.2).
|
|
10
|
+
*
|
|
11
|
+
* Registered as `microvm` because {@link SandboxTier} names the strength of
|
|
12
|
+
* the boundary, not the orchestrator that starts it: a pod scheduled onto a
|
|
13
|
+
* Kata RuntimeClass runs in a hardware-virtualized guest. The tier also keeps
|
|
14
|
+
* the backend out of the container tier's mandatory
|
|
15
|
+
* `ContainerSandboxLayout`, which a remote-copy backend has no use for —
|
|
16
|
+
* Firecracker took the same exemption.
|
|
17
|
+
*
|
|
18
|
+
* ## Two acquire paths, and why they create different kinds
|
|
19
|
+
*
|
|
20
|
+
* - `warmPoolName` set → POST a `SandboxClaim` at the named pool. The
|
|
21
|
+
* controller binds an already-running Sandbox out of the pool, which is
|
|
22
|
+
* what makes a sub-second acquire possible at all.
|
|
23
|
+
* - `warmPoolName` unset → POST a `Sandbox` directly. This is necessitated
|
|
24
|
+
* rather than chosen: `SandboxClaimSpec.warmPoolRef` is a REQUIRED field,
|
|
25
|
+
* so a pool-less claim does not exist in the API.
|
|
26
|
+
*
|
|
27
|
+
* The claim this backend POSTs is PRISTINE: `warmPoolRef` and a lifecycle
|
|
28
|
+
* bound, nothing else. `spec.env` and `spec.volumeClaimTemplates` are
|
|
29
|
+
* available on the claim and are never set, because a claim carrying either
|
|
30
|
+
* is forced to cold-start instead of adopting a pool sandbox — the single
|
|
31
|
+
* most expensive mistake available on this path, and a silent one, since such
|
|
32
|
+
* a claim still works and only the latency shows it. Per-sandbox controls
|
|
33
|
+
* that would need those fields are refused by {@link assertEnforceable}
|
|
34
|
+
* rather than accepted and dropped.
|
|
35
|
+
*
|
|
36
|
+
* ## The bound sandbox is not named after the claim
|
|
37
|
+
*
|
|
38
|
+
* A pool sandbox keeps the generated name the pool gave it when the claim
|
|
39
|
+
* adopts it. The backend therefore reads the bound identity back out of
|
|
40
|
+
* `status.sandbox` and never derives it from the claim's own name. A test
|
|
41
|
+
* covers exactly that asymmetry.
|
|
42
|
+
*
|
|
43
|
+
* ## Credential
|
|
44
|
+
*
|
|
45
|
+
* The per-instance agent bind token is the backing pod's own
|
|
46
|
+
* `metadata.uid`: the host learns it from the API server after readiness, the
|
|
47
|
+
* guest learns it through the downward API (`NAMZU_AGENT_BIND_TOKEN`), and
|
|
48
|
+
* nothing has to be minted, stored or injected at claim time. One GET, no
|
|
49
|
+
* claim mutation, so the warm path stays pristine. A resumed pod is a new pod
|
|
50
|
+
* with a new uid, which is correct: it is a new instance.
|
|
51
|
+
*
|
|
52
|
+
* ## The lease
|
|
53
|
+
*
|
|
54
|
+
* The `shutdownTime` acquire stamps is the leak guard AND, unrenewed, a
|
|
55
|
+
* deadline on the run. The Sandbox handle owns a renewal loop that PATCHes
|
|
56
|
+
* it forward every half-TTL for as long as the handle is alive, and
|
|
57
|
+
* `destroy()` stops it — so a live run keeps its pod and a dead host still
|
|
58
|
+
* costs the cluster exactly one expiry. See `lease.ts`.
|
|
59
|
+
*
|
|
60
|
+
* ## The privilege probe
|
|
61
|
+
*
|
|
62
|
+
* `create()` does not resolve until the guest has reported — and this
|
|
63
|
+
* backend has checked — that it is deprivileged: all four capability masks
|
|
64
|
+
* zero, `NoNewPrivs: 1`, read out of `/proc/self/status` over the agent's
|
|
65
|
+
* `execute` op, on a clock of its own so a guest that goes quiet is refused
|
|
66
|
+
* rather than waited on. A refusal destroys the instance and rejects, so no
|
|
67
|
+
* handle to an under-hardened sandbox escapes. There is no off switch. See
|
|
68
|
+
* `privilege-probe.ts`.
|
|
69
|
+
*
|
|
70
|
+
* ## Not watch
|
|
71
|
+
*
|
|
72
|
+
* Readiness is polled against the shared {@link OperationDeadline}, exactly as
|
|
73
|
+
* ACI polls `provisioningState`. A watch would buy nothing on a path whose
|
|
74
|
+
* whole budget is under a second, and would cost resourceVersion tracking,
|
|
75
|
+
* bookmarks, 410-relist and reconnect backoff.
|
|
76
|
+
*
|
|
77
|
+
* ## Egress
|
|
78
|
+
*
|
|
79
|
+
* `config.egress` is optional and, when set, translated and VERIFIED — never
|
|
80
|
+
* created — by `egress-policy.ts`. Verification happens once, lazily, on the
|
|
81
|
+
* first `create()`, so `buildKubernetesBackend` itself still contacts
|
|
82
|
+
* nothing. Every Sandbox this file creates directly (`buildSandboxBody`)
|
|
83
|
+
* carries {@link sandboxTemplateLabel} on its podTemplate specifically so
|
|
84
|
+
* that translated policy's `podSelector` has something stable to match —
|
|
85
|
+
* see `objects.ts`'s doc comment on that label for why agent-sandbox's own
|
|
86
|
+
* controller-owned label does not cover this path.
|
|
87
|
+
*/
|
|
88
|
+
|
|
89
|
+
import type { Sandbox } from '@namzu/sdk'
|
|
90
|
+
import { generateSandboxId } from '@namzu/sdk'
|
|
91
|
+
|
|
92
|
+
import type { SandboxBackend, SandboxBackendOptions } from '../../index.js'
|
|
93
|
+
import {
|
|
94
|
+
OperationDeadline,
|
|
95
|
+
OperationDeadlineExpired,
|
|
96
|
+
resolveReadinessOptions,
|
|
97
|
+
runFailureCleanup,
|
|
98
|
+
} from '../readiness.js'
|
|
99
|
+
import {
|
|
100
|
+
type KubernetesEgressConfig,
|
|
101
|
+
assertEgressPolicyIsEnforceable,
|
|
102
|
+
defaultEgressPolicyName,
|
|
103
|
+
translateEgressPolicy,
|
|
104
|
+
verifyEgressPolicyApplied,
|
|
105
|
+
} from './egress-policy.js'
|
|
106
|
+
import {
|
|
107
|
+
type KubernetesAccess,
|
|
108
|
+
KubernetesAlreadyGoneError,
|
|
109
|
+
type KubernetesClient,
|
|
110
|
+
createKubernetesClient,
|
|
111
|
+
} from './k8s-client.js'
|
|
112
|
+
import {
|
|
113
|
+
type PodListResource,
|
|
114
|
+
type PodResource,
|
|
115
|
+
READY_CONDITION,
|
|
116
|
+
SANDBOX_API_GROUP,
|
|
117
|
+
SANDBOX_API_VERSION,
|
|
118
|
+
SANDBOX_EXTENSIONS_API_GROUP,
|
|
119
|
+
type SandboxClaimResource,
|
|
120
|
+
type SandboxPodTemplate,
|
|
121
|
+
type SandboxResource,
|
|
122
|
+
type SandboxTemplateResource,
|
|
123
|
+
type SandboxVolumeClaimTemplate,
|
|
124
|
+
claimCollectionPath,
|
|
125
|
+
claimPath,
|
|
126
|
+
isConditionTrue,
|
|
127
|
+
isPodLive,
|
|
128
|
+
podListPath,
|
|
129
|
+
podPath,
|
|
130
|
+
sandboxCollectionPath,
|
|
131
|
+
sandboxPath,
|
|
132
|
+
sandboxTemplateLabel,
|
|
133
|
+
sandboxTemplatePath,
|
|
134
|
+
} from './objects.js'
|
|
135
|
+
import { privilegeProbeTimedOut, runPrivilegeProbe } from './privilege-probe.js'
|
|
136
|
+
import { buildKubernetesSandbox } from './sandbox.js'
|
|
137
|
+
import { KubernetesAgentTransport } from './transport.js'
|
|
138
|
+
|
|
139
|
+
export type { KubernetesEgressConfig, KubernetesEgressEngine } from './egress-policy.js'
|
|
140
|
+
|
|
141
|
+
/**
|
|
142
|
+
* How the backend reaches the API server. Two sources, neither needing a YAML
|
|
143
|
+
* parser — see `k8s-client.ts` for the reasoning.
|
|
144
|
+
*/
|
|
145
|
+
export type KubernetesClusterAccess =
|
|
146
|
+
| { readonly inCluster: true }
|
|
147
|
+
| {
|
|
148
|
+
readonly inCluster?: false
|
|
149
|
+
readonly server: string
|
|
150
|
+
readonly ca?: string | Buffer
|
|
151
|
+
readonly getToken: () => Promise<string>
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
export interface KubernetesBackendInternalConfig {
|
|
155
|
+
readonly access: KubernetesClusterAccess
|
|
156
|
+
/** Namespace the claims, sandboxes and pods live in. */
|
|
157
|
+
readonly namespace: string
|
|
158
|
+
/**
|
|
159
|
+
* SandboxTemplate whose `podTemplate` a POOL-LESS create copies into the
|
|
160
|
+
* Sandbox it posts. The warm path never reads it — the pool's own
|
|
161
|
+
* `sandboxTemplateRef` decides there — but an operator reading this config
|
|
162
|
+
* still learns which template these sandboxes are built from.
|
|
163
|
+
*/
|
|
164
|
+
readonly sandboxTemplateName: string
|
|
165
|
+
/** Named `SandboxWarmPool`. Absent → every create is a direct Sandbox. */
|
|
166
|
+
readonly warmPoolName?: string
|
|
167
|
+
/** TCP port the guest agent listens on. Default {@link DEFAULT_AGENT_PORT}. */
|
|
168
|
+
readonly agentPort?: number
|
|
169
|
+
readonly readyPollIntervalMs?: number
|
|
170
|
+
readonly readyTimeoutMs?: number
|
|
171
|
+
/**
|
|
172
|
+
* Lifetime bound written into every created object, and the amount each
|
|
173
|
+
* lease renewal pushes the expiry forward. Default 1 hour.
|
|
174
|
+
*/
|
|
175
|
+
readonly claimTtlSeconds?: number
|
|
176
|
+
/**
|
|
177
|
+
* Every lease-renewal failure that is not "the object is already gone".
|
|
178
|
+
* `@namzu/sandbox` owns no logger and reads none from module scope, so a
|
|
179
|
+
* diagnostic it cannot print is handed to the host that can. Renewal
|
|
180
|
+
* retries on the next tick either way; nothing here changes behaviour.
|
|
181
|
+
*/
|
|
182
|
+
readonly onLeaseRenewalError?: (error: unknown) => void
|
|
183
|
+
/**
|
|
184
|
+
* RuntimeClass for a POOL-LESS create. Refused together with
|
|
185
|
+
* `warmPoolName`: a pooled sandbox's runtime class is fixed by the pool's
|
|
186
|
+
* SandboxTemplate and cannot be chosen per claim.
|
|
187
|
+
*/
|
|
188
|
+
readonly runtimeClassName?: string
|
|
189
|
+
/**
|
|
190
|
+
* Egress policy this backend's `NetworkPolicy` (or `CiliumNetworkPolicy`,
|
|
191
|
+
* under `engine: 'cilium'`) is expected to carry. Unset means this backend
|
|
192
|
+
* neither computes nor verifies one — the cluster's default posture (the
|
|
193
|
+
* SandboxTemplate's own managed NetworkPolicy) is all that applies. See
|
|
194
|
+
* `egress-policy.ts`.
|
|
195
|
+
*/
|
|
196
|
+
readonly egress?: KubernetesEgressConfig
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
/**
|
|
200
|
+
* The address the guest agent answers on, plus the token to present.
|
|
201
|
+
*
|
|
202
|
+
* Structurally the `tcp` arm of the transport's `SandboxAgentHandle`. It is
|
|
203
|
+
* declared here rather than imported so acquire does not depend on the
|
|
204
|
+
* transport landing first; the two are asserted equal where they meet.
|
|
205
|
+
*/
|
|
206
|
+
export interface KubernetesAgentAddress {
|
|
207
|
+
readonly kind: 'tcp'
|
|
208
|
+
readonly host: string
|
|
209
|
+
readonly port: number
|
|
210
|
+
readonly token: string
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
/** What the controller bound, read back off the object's own status. */
|
|
214
|
+
export interface KubernetesSandboxBinding {
|
|
215
|
+
/** The Sandbox's own name — NOT the claim's. */
|
|
216
|
+
readonly name: string
|
|
217
|
+
readonly podIPs?: readonly string[]
|
|
218
|
+
readonly serviceFQDN?: string
|
|
219
|
+
/** `Sandbox.status.selector`, when the path that read it had it for free. */
|
|
220
|
+
readonly podSelector?: string
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
/** One acquired sandbox: what it is, where it answers, how to give it back. */
|
|
224
|
+
export interface KubernetesAcquisition {
|
|
225
|
+
readonly binding: KubernetesSandboxBinding
|
|
226
|
+
readonly agent: KubernetesAgentAddress
|
|
227
|
+
/** API path of the object THIS backend created — the claim, or the Sandbox. */
|
|
228
|
+
readonly ownedPath: string
|
|
229
|
+
/** The TTL acquire stamped, which every renewal re-stamps. */
|
|
230
|
+
readonly ttlSeconds: number
|
|
231
|
+
/** DELETE that object. An already-gone object counts as released. */
|
|
232
|
+
release(signal?: AbortSignal): Promise<void>
|
|
233
|
+
/**
|
|
234
|
+
* Merge-PATCH the created object's expiry forward to `shutdownTime`.
|
|
235
|
+
*
|
|
236
|
+
* On the acquisition rather than in `lease.ts` because only this path
|
|
237
|
+
* knows WHICH object it created and therefore where the field lives: a
|
|
238
|
+
* `SandboxClaim` carries it at `spec.lifecycle.shutdownTime`, a directly
|
|
239
|
+
* created `Sandbox` at `spec.shutdownTime` (v1beta1 as served keeps it at
|
|
240
|
+
* the top of `spec`). A merge patch of the nested object leaves
|
|
241
|
+
* `shutdownPolicy` alone.
|
|
242
|
+
*/
|
|
243
|
+
renew(shutdownTime: string, signal?: AbortSignal): Promise<void>
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
/**
|
|
247
|
+
* The same number the Firecracker guest agent listens on over vsock
|
|
248
|
+
* (`DEFAULT_AGENT_VSOCK_PORT`), so one agent has one port across both tiers
|
|
249
|
+
* and a manifest, a NetworkPolicy and a transport can all name it from
|
|
250
|
+
* memory. Unprivileged, and the sandbox pod is not sharing it with anything.
|
|
251
|
+
*/
|
|
252
|
+
export const DEFAULT_AGENT_PORT = 1024
|
|
253
|
+
|
|
254
|
+
/**
|
|
255
|
+
* Deliberately far below ACI's 500 ms. A pool bind lands in ~120 ms on a warm
|
|
256
|
+
* cluster, so a half-second poll would spend most of the sub-second acquire
|
|
257
|
+
* budget asleep; 50 ms costs a handful of cheap GETs and gives the measurement
|
|
258
|
+
* somewhere to land.
|
|
259
|
+
*/
|
|
260
|
+
const DEFAULT_READY_POLL_MS = 50
|
|
261
|
+
const DEFAULT_READY_TIMEOUT_MS = 60_000
|
|
262
|
+
const DEFAULT_CLAIM_TTL_SECONDS = 3_600
|
|
263
|
+
|
|
264
|
+
/**
|
|
265
|
+
* The ceiling on the privilege probe's own clock — see
|
|
266
|
+
* {@link resolveProbeTimeoutMs} for where the rest of the number comes from.
|
|
267
|
+
*
|
|
268
|
+
* The probe is one `cat` of a pseudo-file over an already-established path,
|
|
269
|
+
* so half a minute would already be generous and a quarter of one is plenty.
|
|
270
|
+
* The number matters because the alternative is not "a bit longer": with no
|
|
271
|
+
* clock of its own the probe falls back on the execution controller's generic
|
|
272
|
+
* defaults — a five-minute execution observation, then a cancel-confirm and a
|
|
273
|
+
* drain — so a guest that accepts the TCP connection and then stops answering
|
|
274
|
+
* would keep a 60 s `create()` pending for over six minutes.
|
|
275
|
+
*/
|
|
276
|
+
const PRIVILEGE_PROBE_TIMEOUT_CAP_MS = 15_000
|
|
277
|
+
|
|
278
|
+
/**
|
|
279
|
+
* How long the privilege probe may take, given the caller's readiness budget.
|
|
280
|
+
*
|
|
281
|
+
* `readyTimeoutMs` bounds the CONTROL plane and has usually expired by the
|
|
282
|
+
* time the probe starts, so the probe cannot share it — but it is still the
|
|
283
|
+
* number the caller chose to describe how long an acquire may take, so the
|
|
284
|
+
* probe is allowed exactly that much again and no more, capped. A caller who
|
|
285
|
+
* asked for a 500 ms acquire gets a 500 ms probe; one who asked for two
|
|
286
|
+
* minutes of cold start still gets {@link PRIVILEGE_PROBE_TIMEOUT_CAP_MS}.
|
|
287
|
+
* Deliberately not a separate config key: a knob whose only correct value is
|
|
288
|
+
* "long enough for one `cat`" is a knob that only ever gets set wrong.
|
|
289
|
+
*/
|
|
290
|
+
export function resolveProbeTimeoutMs(readyTimeoutMs: number): number {
|
|
291
|
+
return Math.min(readyTimeoutMs, PRIVILEGE_PROBE_TIMEOUT_CAP_MS)
|
|
292
|
+
}
|
|
293
|
+
|
|
294
|
+
/**
|
|
295
|
+
* The readiness bounds every path in this backend polls against — acquire,
|
|
296
|
+
* and `workspace.ts`'s create/suspend/resume. One function so the two cannot
|
|
297
|
+
* drift apart on defaults.
|
|
298
|
+
*/
|
|
299
|
+
export function resolveKubernetesReadiness(config: {
|
|
300
|
+
readonly readyTimeoutMs?: number
|
|
301
|
+
readonly readyPollIntervalMs?: number
|
|
302
|
+
}): { readonly timeoutMs: number; readonly pollIntervalMs: number } {
|
|
303
|
+
return resolveReadinessOptions('kubernetes', config.readyTimeoutMs, config.readyPollIntervalMs, {
|
|
304
|
+
timeoutMs: DEFAULT_READY_TIMEOUT_MS,
|
|
305
|
+
pollIntervalMs: DEFAULT_READY_POLL_MS,
|
|
306
|
+
})
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
/**
|
|
310
|
+
* Per-sandbox controls this backend cannot apply, and therefore refuses.
|
|
311
|
+
*
|
|
312
|
+
* `env` is the load-bearing one. A SandboxClaim CAN carry `spec.env`, so this
|
|
313
|
+
* looks at first like a control that fits — but a claim that sets it is forced
|
|
314
|
+
* to cold-start instead of adopting a pool sandbox, which turns a 120 ms
|
|
315
|
+
* acquire into a full pod start. Accepting env here would buy a caller a
|
|
316
|
+
* feature and silently take the warm pool away, and the only symptom would be
|
|
317
|
+
* latency. The limits belong on the SandboxTemplate the pool is built from.
|
|
318
|
+
*
|
|
319
|
+
* `egress` is refused because this backend applies egress at the network
|
|
320
|
+
* layer, on the template's NetworkPolicy, which cannot be rewritten per
|
|
321
|
+
* running sandbox. The container backend's habit of emitting proxy
|
|
322
|
+
* environment variables as a substitute is not repeated here: a policy
|
|
323
|
+
* accepted and quietly not enforced is worse than one that is refused.
|
|
324
|
+
*/
|
|
325
|
+
const UNSUPPORTED_PER_SANDBOX_CONTROLS = [
|
|
326
|
+
['egress', 'network egress policy'],
|
|
327
|
+
['memoryLimitMb', 'memory limit'],
|
|
328
|
+
['maxProcesses', 'process limit'],
|
|
329
|
+
['env', 'environment variables'],
|
|
330
|
+
] as const
|
|
331
|
+
|
|
332
|
+
export function assertEnforceable(options: SandboxBackendOptions): void {
|
|
333
|
+
const unenforceable = UNSUPPORTED_PER_SANDBOX_CONTROLS.filter(([key]) => {
|
|
334
|
+
const value = options[key]
|
|
335
|
+
return value !== undefined && (key !== 'env' || Object.keys(value).length > 0)
|
|
336
|
+
})
|
|
337
|
+
if (unenforceable.length === 0) return
|
|
338
|
+
|
|
339
|
+
throw new Error(
|
|
340
|
+
`The kubernetes sandbox backend cannot enforce per-sandbox ${unenforceable
|
|
341
|
+
.map(([, label]) => label)
|
|
342
|
+
.join(
|
|
343
|
+
', ',
|
|
344
|
+
)}: a SandboxClaim carrying env or volumes is forced to cold-start instead of adopting a warm pool sandbox, and egress is a NetworkPolicy on the pool's SandboxTemplate rather than a per-sandbox setting. Set them on the SandboxTemplate the SandboxWarmPool is built from, or use a backend that applies them per sandbox. Refusing rather than accepting a control that would be silently dropped.`,
|
|
345
|
+
)
|
|
346
|
+
}
|
|
347
|
+
|
|
348
|
+
/**
|
|
349
|
+
* Refuse a runtime class the pool path cannot honour.
|
|
350
|
+
*
|
|
351
|
+
* A pooled sandbox is already running by the time a claim reaches it, under
|
|
352
|
+
* whatever RuntimeClass its SandboxTemplate named. `runtimeClassName` in this
|
|
353
|
+
* config would therefore be read, accepted and ignored — and the thing it
|
|
354
|
+
* selects is the VM boundary, which is the last control to lose quietly.
|
|
355
|
+
*/
|
|
356
|
+
export function assertRuntimeClassIsApplicable(config: {
|
|
357
|
+
warmPoolName?: string
|
|
358
|
+
runtimeClassName?: string
|
|
359
|
+
}): void {
|
|
360
|
+
if (config.runtimeClassName === undefined || config.warmPoolName === undefined) return
|
|
361
|
+
throw new Error(
|
|
362
|
+
`The kubernetes sandbox backend cannot apply runtimeClassName ${JSON.stringify(config.runtimeClassName)} to sandboxes claimed from warm pool ${JSON.stringify(config.warmPoolName)}: a pooled sandbox is already running under the RuntimeClass its SandboxTemplate named, and a claim cannot change it. Set runtimeClassName on that SandboxTemplate's podTemplate, or drop warmPoolName to have this backend create each Sandbox itself.`,
|
|
363
|
+
)
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
/**
|
|
367
|
+
* Build a {@link SandboxBackend} against a cluster running the agent-sandbox
|
|
368
|
+
* controller. Construction is synchronous and contacts nothing: readiness
|
|
369
|
+
* bounds and the config refusals are validated here so a misconfiguration
|
|
370
|
+
* surfaces during host wiring rather than mid-run, and the first API call
|
|
371
|
+
* happens on the first `create()`.
|
|
372
|
+
*/
|
|
373
|
+
export function buildKubernetesBackend(config: KubernetesBackendInternalConfig): SandboxBackend {
|
|
374
|
+
const readiness = resolveKubernetesReadiness(config)
|
|
375
|
+
assertRuntimeClassIsApplicable(config)
|
|
376
|
+
// A hostname allowlist with no FQDN-capable engine declared is a
|
|
377
|
+
// configuration error, not a runtime one — it can be decided from
|
|
378
|
+
// `config.egress.policy.kind` alone, with no API call, so it is refused
|
|
379
|
+
// here, synchronously, the same moment the two checks above are.
|
|
380
|
+
if (config.egress) {
|
|
381
|
+
assertEgressPolicyIsEnforceable(config.egress.policy, config.egress.engine ?? 'core')
|
|
382
|
+
}
|
|
383
|
+
const client = createKubernetesClient(clientAccess(config))
|
|
384
|
+
// Verify-not-trust runs once, lazily, on the first `create()` — never here,
|
|
385
|
+
// because `buildKubernetesBackend` is documented to contact nothing. A
|
|
386
|
+
// failed attempt is not cached: a transient API error should not wedge
|
|
387
|
+
// every later create() behind the same stale rejection forever.
|
|
388
|
+
let egressVerification: Promise<void> | undefined
|
|
389
|
+
return {
|
|
390
|
+
tier: 'microvm',
|
|
391
|
+
name: 'kubernetes',
|
|
392
|
+
async create(options: SandboxBackendOptions): Promise<Sandbox> {
|
|
393
|
+
if (config.egress) {
|
|
394
|
+
egressVerification ??= verifyEgressPolicyConfigured(
|
|
395
|
+
client,
|
|
396
|
+
config.namespace,
|
|
397
|
+
config.sandboxTemplateName,
|
|
398
|
+
config.egress,
|
|
399
|
+
options.signal,
|
|
400
|
+
).catch((err: unknown) => {
|
|
401
|
+
egressVerification = undefined
|
|
402
|
+
throw err
|
|
403
|
+
})
|
|
404
|
+
await egressVerification
|
|
405
|
+
}
|
|
406
|
+
const acquisition = await acquireKubernetesSandbox(client, config, options, readiness)
|
|
407
|
+
return await admitProbedSandbox(
|
|
408
|
+
acquisition,
|
|
409
|
+
config,
|
|
410
|
+
options,
|
|
411
|
+
resolveProbeTimeoutMs(readiness.timeoutMs),
|
|
412
|
+
)
|
|
413
|
+
},
|
|
414
|
+
}
|
|
415
|
+
}
|
|
416
|
+
|
|
417
|
+
/**
|
|
418
|
+
* Translate `egress.policy` and confirm an operator applied a matching
|
|
419
|
+
* object — the whole verify-not-trust step, isolated so `create()` above
|
|
420
|
+
* stays about ONE thing (memoize-once-per-backend) rather than two.
|
|
421
|
+
*
|
|
422
|
+
* Exported because `workspace.ts` runs the identical step: a workspace does
|
|
423
|
+
* not go through `buildKubernetesBackend`, and a config `egress` honoured on
|
|
424
|
+
* one entry point and ignored on the other would be a silent downgrade of the
|
|
425
|
+
* boundary this backend calls primary. `sandboxTemplateName` is the template
|
|
426
|
+
* the caller is actually building from — it decides both the default policy
|
|
427
|
+
* name and the pod label the policy's selector has to match, and a workspace
|
|
428
|
+
* may be built from a different template than the task path's.
|
|
429
|
+
*/
|
|
430
|
+
export async function verifyEgressPolicyConfigured(
|
|
431
|
+
client: KubernetesClient,
|
|
432
|
+
namespace: string,
|
|
433
|
+
sandboxTemplateName: string,
|
|
434
|
+
egress: KubernetesEgressConfig,
|
|
435
|
+
signal?: AbortSignal,
|
|
436
|
+
): Promise<void> {
|
|
437
|
+
const engine = egress.engine ?? 'core'
|
|
438
|
+
const translated = await translateEgressPolicy(egress.policy, engine, {
|
|
439
|
+
namespace,
|
|
440
|
+
name: egress.networkPolicyName ?? defaultEgressPolicyName(sandboxTemplateName),
|
|
441
|
+
sandboxTemplateName,
|
|
442
|
+
})
|
|
443
|
+
await verifyEgressPolicyApplied(client, translated, signal)
|
|
444
|
+
}
|
|
445
|
+
|
|
446
|
+
/** Config → the client's own access shape. Shared with `workspace.ts`. */
|
|
447
|
+
export function clientAccess(config: KubernetesBackendInternalConfig): KubernetesAccess {
|
|
448
|
+
const access = config.access
|
|
449
|
+
if (access.inCluster === true) return { inCluster: true }
|
|
450
|
+
return {
|
|
451
|
+
server: access.server,
|
|
452
|
+
namespace: config.namespace,
|
|
453
|
+
getToken: access.getToken,
|
|
454
|
+
...(access.ca !== undefined ? { ca: access.ca } : {}),
|
|
455
|
+
}
|
|
456
|
+
}
|
|
457
|
+
|
|
458
|
+
/**
|
|
459
|
+
* Claim or create, wait for Ready, read the bound identity back, resolve the
|
|
460
|
+
* address and learn the pod's uid — or leave nothing behind trying.
|
|
461
|
+
*
|
|
462
|
+
* Exported because the sandbox surface is built on top of this record rather
|
|
463
|
+
* than beside it: one acquire path, one cleanup path, whatever ends up
|
|
464
|
+
* wrapping them.
|
|
465
|
+
*/
|
|
466
|
+
export async function acquireKubernetesSandbox(
|
|
467
|
+
client: KubernetesClient,
|
|
468
|
+
config: KubernetesBackendInternalConfig,
|
|
469
|
+
options: SandboxBackendOptions,
|
|
470
|
+
readiness: { readonly timeoutMs: number; readonly pollIntervalMs: number },
|
|
471
|
+
): Promise<KubernetesAcquisition> {
|
|
472
|
+
options.signal?.throwIfAborted()
|
|
473
|
+
assertEnforceable(options)
|
|
474
|
+
assertRuntimeClassIsApplicable(config)
|
|
475
|
+
|
|
476
|
+
const namespace = config.namespace
|
|
477
|
+
const ttlSeconds = config.claimTtlSeconds ?? DEFAULT_CLAIM_TTL_SECONDS
|
|
478
|
+
const shutdownTime = new Date(Date.now() + ttlSeconds * 1_000).toISOString()
|
|
479
|
+
// Client-owned name, as on ACI: it lets failure cleanup DELETE the object
|
|
480
|
+
// even when the create response never arrived. `generateSandboxId` returns
|
|
481
|
+
// a lowercase UUID, which is already a legal DNS-1123 name suffix.
|
|
482
|
+
const objectName = `namzu-task-${generateSandboxId()}`
|
|
483
|
+
const ownedPath =
|
|
484
|
+
config.warmPoolName !== undefined
|
|
485
|
+
? claimPath(namespace, objectName)
|
|
486
|
+
: sandboxPath(namespace, objectName)
|
|
487
|
+
|
|
488
|
+
const release = async (signal?: AbortSignal): Promise<void> => {
|
|
489
|
+
try {
|
|
490
|
+
await client.request('DELETE', ownedPath, undefined, signal)
|
|
491
|
+
} catch (err) {
|
|
492
|
+
// The object is gone, which is the state DELETE was asking for.
|
|
493
|
+
if (!(err instanceof KubernetesAlreadyGoneError)) throw err
|
|
494
|
+
}
|
|
495
|
+
}
|
|
496
|
+
|
|
497
|
+
// Where the expiry lives differs by KIND, and only this function knows
|
|
498
|
+
// which kind it created: a claim keeps it under `spec.lifecycle`, a
|
|
499
|
+
// directly created Sandbox at the top of `spec` (v1beta1 as served has
|
|
500
|
+
// not moved it under `lifecycle` yet). Merge-patch semantics (RFC 7386,
|
|
501
|
+
// the only content type this client's PATCH sends) merge the nested
|
|
502
|
+
// object, so `shutdownPolicy: Delete` survives every renewal.
|
|
503
|
+
// The parameter is deliberately NOT named `shutdownTime`: the stamp above
|
|
504
|
+
// is the one the create body carries, and no renewal ever re-sends it.
|
|
505
|
+
const renew = async (nextShutdownTime: string, signal?: AbortSignal): Promise<void> => {
|
|
506
|
+
const patch =
|
|
507
|
+
config.warmPoolName !== undefined
|
|
508
|
+
? { spec: { lifecycle: { shutdownTime: nextShutdownTime } } }
|
|
509
|
+
: { spec: { shutdownTime: nextShutdownTime } }
|
|
510
|
+
await client.request('PATCH', ownedPath, patch, signal)
|
|
511
|
+
}
|
|
512
|
+
|
|
513
|
+
// One clock over the whole path — the create POST included, so a hung API
|
|
514
|
+
// server cannot leave `create()` pending past the caller's timeout.
|
|
515
|
+
const deadline = new OperationDeadline(
|
|
516
|
+
readiness.timeoutMs,
|
|
517
|
+
'kubernetes readiness',
|
|
518
|
+
options.signal,
|
|
519
|
+
)
|
|
520
|
+
|
|
521
|
+
// The pool-less path reads its pod template BEFORE anything is created, so
|
|
522
|
+
// a missing or malformed template fails with nothing to clean up — hence
|
|
523
|
+
// this sits outside the cleanup block below. It is read per create rather
|
|
524
|
+
// than cached: an operator editing the template expects the next sandbox to
|
|
525
|
+
// use it, and this path is not the sub-second one.
|
|
526
|
+
const createPath =
|
|
527
|
+
config.warmPoolName !== undefined
|
|
528
|
+
? claimCollectionPath(namespace)
|
|
529
|
+
: sandboxCollectionPath(namespace)
|
|
530
|
+
const createBody =
|
|
531
|
+
config.warmPoolName !== undefined
|
|
532
|
+
? buildClaimBody(namespace, objectName, config.warmPoolName, shutdownTime)
|
|
533
|
+
: buildSandboxBody({
|
|
534
|
+
namespace,
|
|
535
|
+
name: objectName,
|
|
536
|
+
template: await deadline.run((signal) =>
|
|
537
|
+
readSandboxTemplate(client, namespace, config.sandboxTemplateName, signal),
|
|
538
|
+
),
|
|
539
|
+
sandboxTemplateName: config.sandboxTemplateName,
|
|
540
|
+
shutdownTime,
|
|
541
|
+
...(config.runtimeClassName !== undefined
|
|
542
|
+
? { runtimeClassName: config.runtimeClassName }
|
|
543
|
+
: {}),
|
|
544
|
+
})
|
|
545
|
+
|
|
546
|
+
try {
|
|
547
|
+
// Inside the cleanup block: a POST that fails client-side may still have
|
|
548
|
+
// committed, so the only safe assumption is that the object exists.
|
|
549
|
+
await deadline.run((signal) => client.request('POST', createPath, createBody, signal))
|
|
550
|
+
const binding =
|
|
551
|
+
config.warmPoolName !== undefined
|
|
552
|
+
? await pollForBinding(
|
|
553
|
+
async (signal) =>
|
|
554
|
+
bindingFromClaim(
|
|
555
|
+
await client.request<SandboxClaimResource>(
|
|
556
|
+
'GET',
|
|
557
|
+
claimPath(namespace, objectName),
|
|
558
|
+
undefined,
|
|
559
|
+
signal,
|
|
560
|
+
),
|
|
561
|
+
objectName,
|
|
562
|
+
),
|
|
563
|
+
deadline,
|
|
564
|
+
readiness,
|
|
565
|
+
`claim ${objectName}`,
|
|
566
|
+
)
|
|
567
|
+
: await pollForBinding(
|
|
568
|
+
async (signal) =>
|
|
569
|
+
bindingFromSandbox(
|
|
570
|
+
await client.request<SandboxResource>(
|
|
571
|
+
'GET',
|
|
572
|
+
sandboxPath(namespace, objectName),
|
|
573
|
+
undefined,
|
|
574
|
+
signal,
|
|
575
|
+
),
|
|
576
|
+
),
|
|
577
|
+
deadline,
|
|
578
|
+
readiness,
|
|
579
|
+
`sandbox ${objectName}`,
|
|
580
|
+
)
|
|
581
|
+
|
|
582
|
+
const token = await deadline.run((signal) =>
|
|
583
|
+
readPodBindToken(client, namespace, binding, signal),
|
|
584
|
+
)
|
|
585
|
+
return {
|
|
586
|
+
binding,
|
|
587
|
+
agent: resolveAgentAddress(binding, config.agentPort ?? DEFAULT_AGENT_PORT, token),
|
|
588
|
+
ownedPath,
|
|
589
|
+
ttlSeconds,
|
|
590
|
+
release,
|
|
591
|
+
renew,
|
|
592
|
+
}
|
|
593
|
+
} catch (err) {
|
|
594
|
+
// One cleanup for every way out of the block above, on its own short
|
|
595
|
+
// budget: the readiness clock has already expired in the common case,
|
|
596
|
+
// so spending it again would either skip cleanup or leave `create()`
|
|
597
|
+
// pending without a bound. An object that is already gone is success.
|
|
598
|
+
await runFailureCleanup(async (signal) => {
|
|
599
|
+
await release(signal)
|
|
600
|
+
})
|
|
601
|
+
throw err
|
|
602
|
+
}
|
|
603
|
+
}
|
|
604
|
+
|
|
605
|
+
/**
|
|
606
|
+
* The claim body, in full. Everything absent from it is absent on purpose:
|
|
607
|
+
* no `env`, no `volumeClaimTemplates`, no `additionalPodMetadata`. Either of
|
|
608
|
+
* the first two forces a cold start upstream and takes the warm pool away.
|
|
609
|
+
*
|
|
610
|
+
* `shutdownTime` + `shutdownPolicy: 'Delete'` is the leak guard: it bounds the
|
|
611
|
+
* object by the wall clock whatever the host does, so a host that dies
|
|
612
|
+
* mid-acquire costs the cluster one TTL rather than one leaked sandbox
|
|
613
|
+
* forever. `ttlSecondsAfterFinished` deliberately does NOT appear — its timer
|
|
614
|
+
* starts from the Finished condition, which a crashed host never reaches.
|
|
615
|
+
*/
|
|
616
|
+
function buildClaimBody(
|
|
617
|
+
namespace: string,
|
|
618
|
+
name: string,
|
|
619
|
+
warmPoolName: string,
|
|
620
|
+
shutdownTime: string,
|
|
621
|
+
): Record<string, unknown> {
|
|
622
|
+
return {
|
|
623
|
+
apiVersion: `${SANDBOX_EXTENSIONS_API_GROUP}/${SANDBOX_API_VERSION}`,
|
|
624
|
+
kind: 'SandboxClaim',
|
|
625
|
+
metadata: { name, namespace },
|
|
626
|
+
spec: {
|
|
627
|
+
warmPoolRef: { name: warmPoolName },
|
|
628
|
+
lifecycle: { shutdownTime, shutdownPolicy: 'Delete' },
|
|
629
|
+
},
|
|
630
|
+
}
|
|
631
|
+
}
|
|
632
|
+
|
|
633
|
+
/**
|
|
634
|
+
* What a directly created Sandbox copies out of a `SandboxTemplate`, and the
|
|
635
|
+
* two things it decides for itself.
|
|
636
|
+
*/
|
|
637
|
+
export interface SandboxBodyOptions {
|
|
638
|
+
readonly namespace: string
|
|
639
|
+
readonly name: string
|
|
640
|
+
readonly template: SandboxTemplateCopy
|
|
641
|
+
/** The template the copy came from — the value of {@link sandboxTemplateLabel}. */
|
|
642
|
+
readonly sandboxTemplateName: string
|
|
643
|
+
readonly runtimeClassName?: string
|
|
644
|
+
/**
|
|
645
|
+
* RFC 3339 expiry, paired with `shutdownPolicy: Delete`. ABSENT means the
|
|
646
|
+
* object carries no expiry at all and nothing reaps it on the wall clock:
|
|
647
|
+
* that is the persistent workspace (`workspace.ts`), which is explicitly
|
|
648
|
+
* managed and must survive a host that stops renewing. Every task sandbox
|
|
649
|
+
* sets it, because an unbounded task sandbox is a leak.
|
|
650
|
+
*/
|
|
651
|
+
readonly shutdownTime?: string
|
|
652
|
+
}
|
|
653
|
+
|
|
654
|
+
/**
|
|
655
|
+
* The pool-less body. `Sandbox.spec` has no `templateRef` — only a
|
|
656
|
+
* SandboxWarmPool consumes a SandboxTemplate — so the template's podTemplate
|
|
657
|
+
* is copied in here by the client.
|
|
658
|
+
*
|
|
659
|
+
* `service: true` is forced rather than inherited: a Sandbox without a Service
|
|
660
|
+
* has no `status.serviceFQDN`, and then the only address left is a pod IP that
|
|
661
|
+
* changes on every resume.
|
|
662
|
+
*
|
|
663
|
+
* `volumeClaimTemplates` is copied VERBATIM when the template declares any.
|
|
664
|
+
* Dropping it would be silent: the Sandbox would come up healthy with no disk,
|
|
665
|
+
* the container's `volumeDevices`/`volumeMounts` entry would fail to resolve
|
|
666
|
+
* (or, worse, resolve to an empty emptyDir on some paths), and the only
|
|
667
|
+
* symptom of a workspace that lost its disk would be that yesterday's files
|
|
668
|
+
* are gone. The controller wires the mount by the entry's own NAME,
|
|
669
|
+
* StatefulSet style, so the copy needs no matching `volumes:` entry and this
|
|
670
|
+
* function adds none.
|
|
671
|
+
*
|
|
672
|
+
* The podTemplate's metadata gains {@link sandboxTemplateLabel}: this Sandbox
|
|
673
|
+
* is created DIRECTLY, never adopted out of a pool, so it never gets
|
|
674
|
+
* agent-sandbox's own controller-owned
|
|
675
|
+
* `agents.x-k8s.io/sandbox-template-ref-hash` label (that is written only on
|
|
676
|
+
* bind). Without a label of its own a direct Sandbox's pod would carry
|
|
677
|
+
* nothing `egress-policy.ts`'s translated `NetworkPolicy` could select it
|
|
678
|
+
* by. Existing labels on the copied template are preserved — this ADDS to
|
|
679
|
+
* them rather than replacing the object outright — but this backend's own
|
|
680
|
+
* key always wins if the template happened to set it too, since this is the
|
|
681
|
+
* label the translated policy is built to match.
|
|
682
|
+
*/
|
|
683
|
+
export function buildSandboxBody(options: SandboxBodyOptions): Record<string, unknown> {
|
|
684
|
+
const podTemplate = options.template.podTemplate
|
|
685
|
+
const spec =
|
|
686
|
+
options.runtimeClassName !== undefined
|
|
687
|
+
? { ...podTemplate.spec, runtimeClassName: options.runtimeClassName }
|
|
688
|
+
: { ...podTemplate.spec }
|
|
689
|
+
const metadata = {
|
|
690
|
+
...podTemplate.metadata,
|
|
691
|
+
labels: {
|
|
692
|
+
...podTemplate.metadata?.labels,
|
|
693
|
+
...sandboxTemplateLabel(options.sandboxTemplateName),
|
|
694
|
+
},
|
|
695
|
+
}
|
|
696
|
+
return {
|
|
697
|
+
apiVersion: `${SANDBOX_API_GROUP}/${SANDBOX_API_VERSION}`,
|
|
698
|
+
kind: 'Sandbox',
|
|
699
|
+
metadata: { name: options.name, namespace: options.namespace },
|
|
700
|
+
spec: {
|
|
701
|
+
operatingMode: 'Running',
|
|
702
|
+
service: true,
|
|
703
|
+
...(options.shutdownTime !== undefined
|
|
704
|
+
? { shutdownTime: options.shutdownTime, shutdownPolicy: 'Delete' }
|
|
705
|
+
: {}),
|
|
706
|
+
...(options.template.volumeClaimTemplates !== undefined
|
|
707
|
+
? { volumeClaimTemplates: options.template.volumeClaimTemplates }
|
|
708
|
+
: {}),
|
|
709
|
+
podTemplate: { ...podTemplate, metadata, spec },
|
|
710
|
+
},
|
|
711
|
+
}
|
|
712
|
+
}
|
|
713
|
+
|
|
714
|
+
/** The two halves of a `SandboxTemplate` a directly created Sandbox copies. */
|
|
715
|
+
export interface SandboxTemplateCopy {
|
|
716
|
+
readonly podTemplate: SandboxPodTemplate
|
|
717
|
+
/** Absent when the template declares no disk, which is every task template. */
|
|
718
|
+
readonly volumeClaimTemplates?: readonly SandboxVolumeClaimTemplate[]
|
|
719
|
+
}
|
|
720
|
+
|
|
721
|
+
export async function readSandboxTemplate(
|
|
722
|
+
client: KubernetesClient,
|
|
723
|
+
namespace: string,
|
|
724
|
+
sandboxTemplateName: string,
|
|
725
|
+
signal?: AbortSignal,
|
|
726
|
+
): Promise<SandboxTemplateCopy> {
|
|
727
|
+
const template = await client.request<SandboxTemplateResource>(
|
|
728
|
+
'GET',
|
|
729
|
+
sandboxTemplatePath(namespace, sandboxTemplateName),
|
|
730
|
+
undefined,
|
|
731
|
+
signal,
|
|
732
|
+
)
|
|
733
|
+
const podTemplate = template?.spec?.podTemplate
|
|
734
|
+
if (!podTemplate || typeof podTemplate.spec !== 'object' || podTemplate.spec === null) {
|
|
735
|
+
throw new Error(
|
|
736
|
+
`kubernetes: SandboxTemplate ${sandboxTemplateName} in namespace ${namespace} carries no spec.podTemplate.spec, so there is nothing to create a pool-less Sandbox from. Sandbox.spec has no templateRef — the podTemplate has to be copied in.`,
|
|
737
|
+
)
|
|
738
|
+
}
|
|
739
|
+
const volumeClaimTemplates = template?.spec?.volumeClaimTemplates
|
|
740
|
+
return {
|
|
741
|
+
podTemplate,
|
|
742
|
+
...(volumeClaimTemplates !== undefined ? { volumeClaimTemplates } : {}),
|
|
743
|
+
}
|
|
744
|
+
}
|
|
745
|
+
|
|
746
|
+
/**
|
|
747
|
+
* Where the agent answers.
|
|
748
|
+
*
|
|
749
|
+
* Its own function, and the Service FQDN wins over a pod IP, because the
|
|
750
|
+
* address outlives the pod: a suspended-then-resumed workspace comes back as a
|
|
751
|
+
* new pod with a new IP behind the same name, and the transport re-resolves
|
|
752
|
+
* the name on every dial. A literal IP baked into a long-lived handle is the
|
|
753
|
+
* bug that would produce.
|
|
754
|
+
*/
|
|
755
|
+
export function resolveAgentAddress(
|
|
756
|
+
binding: KubernetesSandboxBinding,
|
|
757
|
+
agentPort: number,
|
|
758
|
+
token: string,
|
|
759
|
+
): KubernetesAgentAddress {
|
|
760
|
+
const host = binding.serviceFQDN ?? binding.podIPs?.[0]
|
|
761
|
+
if (host === undefined || host === '') {
|
|
762
|
+
throw new Error(
|
|
763
|
+
`kubernetes: sandbox ${binding.name} reported Ready with neither a serviceFQDN nor a pod IP, so its agent has no address to dial. Set 'service: true' on the SandboxTemplate the pool is built from.`,
|
|
764
|
+
)
|
|
765
|
+
}
|
|
766
|
+
return { kind: 'tcp', host, port: agentPort, token }
|
|
767
|
+
}
|
|
768
|
+
|
|
769
|
+
function bindingFromClaim(
|
|
770
|
+
claim: SandboxClaimResource | undefined,
|
|
771
|
+
claimName: string,
|
|
772
|
+
): KubernetesSandboxBinding | undefined {
|
|
773
|
+
if (!isConditionTrue(claim?.status?.conditions, READY_CONDITION)) return undefined
|
|
774
|
+
const bound = claim?.status?.sandbox
|
|
775
|
+
// Ready with no bound name is the controller contradicting itself; polling
|
|
776
|
+
// on would just burn the deadline waiting for a field that is finished.
|
|
777
|
+
if (!bound?.name) {
|
|
778
|
+
throw new Error(
|
|
779
|
+
`kubernetes: SandboxClaim ${claimName} reported Ready but named no sandbox in status.sandbox.name, so there is nothing to address.`,
|
|
780
|
+
)
|
|
781
|
+
}
|
|
782
|
+
return {
|
|
783
|
+
name: bound.name,
|
|
784
|
+
...(bound.podIPs !== undefined ? { podIPs: bound.podIPs } : {}),
|
|
785
|
+
...(bound.serviceFQDN !== undefined ? { serviceFQDN: bound.serviceFQDN } : {}),
|
|
786
|
+
}
|
|
787
|
+
}
|
|
788
|
+
|
|
789
|
+
/** Ready-or-not-yet, read off a `Sandbox`'s own status. Shared with `workspace.ts`. */
|
|
790
|
+
export function bindingFromSandbox(
|
|
791
|
+
sandbox: SandboxResource | undefined,
|
|
792
|
+
): KubernetesSandboxBinding | undefined {
|
|
793
|
+
if (!isConditionTrue(sandbox?.status?.conditions, READY_CONDITION)) return undefined
|
|
794
|
+
const name = sandbox?.metadata?.name
|
|
795
|
+
if (!name) {
|
|
796
|
+
throw new Error('kubernetes: Sandbox reported Ready with no metadata.name')
|
|
797
|
+
}
|
|
798
|
+
const status = sandbox?.status
|
|
799
|
+
return {
|
|
800
|
+
name,
|
|
801
|
+
...(status?.podIPs !== undefined ? { podIPs: status.podIPs } : {}),
|
|
802
|
+
...(status?.serviceFQDN !== undefined ? { serviceFQDN: status.serviceFQDN } : {}),
|
|
803
|
+
...(status?.selector !== undefined ? { podSelector: status.selector } : {}),
|
|
804
|
+
}
|
|
805
|
+
}
|
|
806
|
+
|
|
807
|
+
/**
|
|
808
|
+
* Poll until `read` reports a binding. `read` returns `undefined` for "not
|
|
809
|
+
* yet" and throws for a failure worth surfacing; the deadline owns every wait,
|
|
810
|
+
* including the sleep between attempts, so an expired clock cannot be extended
|
|
811
|
+
* by one more round trip. Shaped after ACI's `pollForRunningIp`.
|
|
812
|
+
*/
|
|
813
|
+
export async function pollForBinding(
|
|
814
|
+
read: (signal: AbortSignal) => Promise<KubernetesSandboxBinding | undefined>,
|
|
815
|
+
deadline: OperationDeadline,
|
|
816
|
+
readiness: { readonly timeoutMs: number; readonly pollIntervalMs: number },
|
|
817
|
+
label: string,
|
|
818
|
+
): Promise<KubernetesSandboxBinding> {
|
|
819
|
+
while (deadline.remainingMs() > 0) {
|
|
820
|
+
try {
|
|
821
|
+
const binding = await deadline.run(read)
|
|
822
|
+
if (binding) return binding
|
|
823
|
+
} catch (err) {
|
|
824
|
+
if (err instanceof OperationDeadlineExpired) break
|
|
825
|
+
throw err
|
|
826
|
+
}
|
|
827
|
+
try {
|
|
828
|
+
await deadline.delay(readiness.pollIntervalMs)
|
|
829
|
+
} catch (err) {
|
|
830
|
+
if (err instanceof OperationDeadlineExpired) break
|
|
831
|
+
throw err
|
|
832
|
+
}
|
|
833
|
+
}
|
|
834
|
+
throw new Error(`kubernetes: ${label} never became Ready (${readiness.timeoutMs}ms)`)
|
|
835
|
+
}
|
|
836
|
+
|
|
837
|
+
/**
|
|
838
|
+
* The per-instance bind token: the backing pod's `metadata.uid`.
|
|
839
|
+
*
|
|
840
|
+
* The pod is named after its Sandbox in agent-sandbox v1.0.2 — verified
|
|
841
|
+
* against a running cluster — but that is an observation, not a documented
|
|
842
|
+
* guarantee, and `Sandbox.status` exposes no pod name to fall back on. So the
|
|
843
|
+
* fast path is one GET by that name, and the only cost of the name convention
|
|
844
|
+
* changing upstream is a second round trip through `status.selector`, which is
|
|
845
|
+
* exactly what the controller publishes the selector for.
|
|
846
|
+
*/
|
|
847
|
+
export async function readPodBindToken(
|
|
848
|
+
client: KubernetesClient,
|
|
849
|
+
namespace: string,
|
|
850
|
+
binding: KubernetesSandboxBinding,
|
|
851
|
+
signal?: AbortSignal,
|
|
852
|
+
): Promise<string> {
|
|
853
|
+
try {
|
|
854
|
+
const pod = await client.request<PodResource>(
|
|
855
|
+
'GET',
|
|
856
|
+
podPath(namespace, binding.name),
|
|
857
|
+
undefined,
|
|
858
|
+
signal,
|
|
859
|
+
)
|
|
860
|
+
const uid = pod?.metadata?.uid
|
|
861
|
+
// `isPodLive` matters on the RESUME path in `workspace.ts`: a resumed
|
|
862
|
+
// pod keeps its name, so for as long as the outgoing one is
|
|
863
|
+
// terminating this GET can answer with the pod that is leaving and a
|
|
864
|
+
// uid the new agent will refuse. On the acquire path nothing is
|
|
865
|
+
// terminating and the filter never fires.
|
|
866
|
+
if (uid && isPodLive(pod)) return uid
|
|
867
|
+
} catch (err) {
|
|
868
|
+
if (!(err instanceof KubernetesAlreadyGoneError)) throw err
|
|
869
|
+
}
|
|
870
|
+
|
|
871
|
+
const selector =
|
|
872
|
+
binding.podSelector ?? (await readSandboxSelector(client, namespace, binding, signal))
|
|
873
|
+
if (selector !== undefined && selector !== '') {
|
|
874
|
+
const list = await client.request<PodListResource>(
|
|
875
|
+
'GET',
|
|
876
|
+
podListPath(namespace, selector),
|
|
877
|
+
undefined,
|
|
878
|
+
signal,
|
|
879
|
+
)
|
|
880
|
+
for (const pod of list?.items ?? []) {
|
|
881
|
+
const uid = pod.metadata?.uid
|
|
882
|
+
if (uid && isPodLive(pod)) return uid
|
|
883
|
+
}
|
|
884
|
+
}
|
|
885
|
+
throw new Error(
|
|
886
|
+
`kubernetes: could not read a pod uid for sandbox ${binding.name} in namespace ${namespace} — no live pod of that name, and its status.selector matched no live pod either (a pod carrying a deletionTimestamp, or in phase Succeeded/Failed, is never bound to). The pod uid is the agent's bind token, so the sandbox is refused rather than returned unauthenticated.`,
|
|
887
|
+
)
|
|
888
|
+
}
|
|
889
|
+
|
|
890
|
+
async function readSandboxSelector(
|
|
891
|
+
client: KubernetesClient,
|
|
892
|
+
namespace: string,
|
|
893
|
+
binding: KubernetesSandboxBinding,
|
|
894
|
+
signal?: AbortSignal,
|
|
895
|
+
): Promise<string | undefined> {
|
|
896
|
+
try {
|
|
897
|
+
const sandbox = await client.request<SandboxResource>(
|
|
898
|
+
'GET',
|
|
899
|
+
sandboxPath(namespace, binding.name),
|
|
900
|
+
undefined,
|
|
901
|
+
signal,
|
|
902
|
+
)
|
|
903
|
+
return sandbox?.status?.selector
|
|
904
|
+
} catch (err) {
|
|
905
|
+
if (err instanceof KubernetesAlreadyGoneError) return undefined
|
|
906
|
+
throw err
|
|
907
|
+
}
|
|
908
|
+
}
|
|
909
|
+
|
|
910
|
+
/**
|
|
911
|
+
* Build the Sandbox, prove it is deprivileged, and only then hand it back.
|
|
912
|
+
*
|
|
913
|
+
* The probe runs BEFORE `create()` resolves, so a caller never holds a
|
|
914
|
+
* reference to an under-hardened sandbox — a probe that refuses destroys the
|
|
915
|
+
* instance on a bounded cleanup budget and rethrows, exactly as a readiness
|
|
916
|
+
* failure does. There is no configuration that skips it: the whole value of
|
|
917
|
+
* checking the deprivileging on every acquire rather than once by hand is
|
|
918
|
+
* that it cannot be forgotten, and an off switch is a way to forget it.
|
|
919
|
+
*
|
|
920
|
+
* The probe goes through the Sandbox's own `exec`, not the raw transport, so
|
|
921
|
+
* it traverses the same reserve/admit/stream/confirm path every later call
|
|
922
|
+
* will. A sandbox that cannot answer the probe is one a caller could not use
|
|
923
|
+
* either.
|
|
924
|
+
*
|
|
925
|
+
* ## And it runs on a clock
|
|
926
|
+
*
|
|
927
|
+
* This is the first thing on the acquire path that talks to the GUEST, and
|
|
928
|
+
* the readiness deadline that bounded everything before it has already
|
|
929
|
+
* expired. Left unbounded the probe would inherit the execution controller's
|
|
930
|
+
* generic defaults instead — a five-minute observation, then a cancel-confirm
|
|
931
|
+
* and a drain — so a pod whose agent has wedged (out of memory, an event loop
|
|
932
|
+
* the workload blocked) would keep `create()` pending for minutes past the
|
|
933
|
+
* caller's `readyTimeoutMs` with nothing reported. It gets its own deadline,
|
|
934
|
+
* whose expiry takes the same cleanup-and-reject path every other refusal
|
|
935
|
+
* does, in words that name the hang rather than blame a missing `cat`.
|
|
936
|
+
*/
|
|
937
|
+
async function admitProbedSandbox(
|
|
938
|
+
acquisition: KubernetesAcquisition,
|
|
939
|
+
config: KubernetesBackendInternalConfig,
|
|
940
|
+
options: SandboxBackendOptions,
|
|
941
|
+
probeTimeoutMs: number,
|
|
942
|
+
): Promise<Sandbox> {
|
|
943
|
+
const sandbox = buildKubernetesSandbox({
|
|
944
|
+
name: acquisition.binding.name,
|
|
945
|
+
rootDir: options.workingDirectory,
|
|
946
|
+
transport: new KubernetesAgentTransport(acquisition.agent),
|
|
947
|
+
release: acquisition.release,
|
|
948
|
+
renew: acquisition.renew,
|
|
949
|
+
ttlSeconds: acquisition.ttlSeconds,
|
|
950
|
+
...(config.onLeaseRenewalError !== undefined
|
|
951
|
+
? { onRenewalError: config.onLeaseRenewalError }
|
|
952
|
+
: {}),
|
|
953
|
+
})
|
|
954
|
+
try {
|
|
955
|
+
await probeSandboxPrivileges(sandbox, acquisition.binding.name, probeTimeoutMs, options.signal)
|
|
956
|
+
} catch (err) {
|
|
957
|
+
await runFailureCleanup(async (signal) => {
|
|
958
|
+
await sandbox.destroy({ signal })
|
|
959
|
+
})
|
|
960
|
+
throw err
|
|
961
|
+
}
|
|
962
|
+
return sandbox
|
|
963
|
+
}
|
|
964
|
+
|
|
965
|
+
/**
|
|
966
|
+
* Run the probe against a built Sandbox, on its own clock, and throw if it
|
|
967
|
+
* refuses. Cleanup is the CALLER's, and the two callers want opposite things:
|
|
968
|
+
* a task acquire destroys the instance, while `workspace.ts` suspends it,
|
|
969
|
+
* because deleting a workspace deletes its disk and a probe refusal is not a
|
|
970
|
+
* reason to lose a caller's files.
|
|
971
|
+
*/
|
|
972
|
+
export async function probeSandboxPrivileges(
|
|
973
|
+
sandbox: Sandbox,
|
|
974
|
+
sandboxName: string,
|
|
975
|
+
probeTimeoutMs: number,
|
|
976
|
+
signal?: AbortSignal,
|
|
977
|
+
): Promise<void> {
|
|
978
|
+
// Labelled with the sandbox, so an expiry read off a log line says which
|
|
979
|
+
// acquire stopped answering — and so the catch below can tell THIS
|
|
980
|
+
// deadline from any other that might surface through the same exec.
|
|
981
|
+
const probeLabel = `kubernetes privilege probe ${sandboxName}`
|
|
982
|
+
try {
|
|
983
|
+
signal?.throwIfAborted()
|
|
984
|
+
// The deadline's signal covers both ways this should stop early: it
|
|
985
|
+
// aborts on expiry, and it aborts with the caller's own reason when
|
|
986
|
+
// `signal` does. Handing it to `exec` is what releases the guest-side
|
|
987
|
+
// execution rather than merely abandoning the wait.
|
|
988
|
+
await new OperationDeadline(probeTimeoutMs, probeLabel, signal).run(
|
|
989
|
+
async (execSignal) =>
|
|
990
|
+
await runPrivilegeProbe(
|
|
991
|
+
async (command, args) => await sandbox.exec(command, args, { signal: execSignal }),
|
|
992
|
+
sandboxName,
|
|
993
|
+
),
|
|
994
|
+
)
|
|
995
|
+
// An abort that lands WHILE the probe is in flight must not leave a
|
|
996
|
+
// live sandbox behind: the probe itself may well have finished, and
|
|
997
|
+
// the caller who cancelled is about to stop holding the reference
|
|
998
|
+
// that could destroy it. Same cleanup, one branch later.
|
|
999
|
+
signal?.throwIfAborted()
|
|
1000
|
+
} catch (err) {
|
|
1001
|
+
// A caller who cancelled mid-probe gets THEIR reason, not the probe's
|
|
1002
|
+
// account of a command that was cancelled out from under it.
|
|
1003
|
+
signal?.throwIfAborted()
|
|
1004
|
+
// A probe that ran out of time is a probe that could not run, and is
|
|
1005
|
+
// refused in those words: `OperationDeadlineExpired` on its own would
|
|
1006
|
+
// leave a reader guessing which half of the acquire went quiet.
|
|
1007
|
+
if (err instanceof OperationDeadlineExpired && err.label === probeLabel) {
|
|
1008
|
+
throw privilegeProbeTimedOut(sandboxName, probeTimeoutMs, err)
|
|
1009
|
+
}
|
|
1010
|
+
throw err
|
|
1011
|
+
}
|
|
1012
|
+
}
|