@namzu/sandbox 13.0.0 → 15.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +1147 -0
- package/README.md +447 -0
- package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
- package/dist/backends/aci-standby-pool/index.js +13 -1
- package/dist/backends/aci-standby-pool/index.js.map +1 -1
- package/dist/backends/docker/index.d.ts.map +1 -1
- package/dist/backends/docker/index.js +19 -1
- package/dist/backends/docker/index.js.map +1 -1
- package/dist/backends/firecracker/index.d.ts.map +1 -1
- package/dist/backends/firecracker/index.js +12 -2
- package/dist/backends/firecracker/index.js.map +1 -1
- package/dist/backends/firecracker/protocol.d.ts +481 -8
- package/dist/backends/firecracker/protocol.d.ts.map +1 -1
- package/dist/backends/firecracker/protocol.js +136 -0
- package/dist/backends/firecracker/protocol.js.map +1 -1
- package/dist/backends/firecracker/transport.d.ts +642 -14
- package/dist/backends/firecracker/transport.d.ts.map +1 -1
- package/dist/backends/firecracker/transport.js +1307 -34
- package/dist/backends/firecracker/transport.js.map +1 -1
- package/dist/backends/kubernetes/egress-policy.d.ts +1296 -0
- package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -0
- package/dist/backends/kubernetes/egress-policy.js +2458 -0
- package/dist/backends/kubernetes/egress-policy.js.map +1 -0
- package/dist/backends/kubernetes/identity.d.ts +193 -0
- package/dist/backends/kubernetes/identity.d.ts.map +1 -0
- package/dist/backends/kubernetes/identity.js +147 -0
- package/dist/backends/kubernetes/identity.js.map +1 -0
- package/dist/backends/kubernetes/index.d.ts +1019 -0
- package/dist/backends/kubernetes/index.d.ts.map +1 -0
- package/dist/backends/kubernetes/index.js +1756 -0
- package/dist/backends/kubernetes/index.js.map +1 -0
- package/dist/backends/kubernetes/ingress-policy.d.ts +375 -0
- package/dist/backends/kubernetes/ingress-policy.d.ts.map +1 -0
- package/dist/backends/kubernetes/ingress-policy.js +1050 -0
- package/dist/backends/kubernetes/ingress-policy.js.map +1 -0
- package/dist/backends/kubernetes/k8s-client.d.ts +334 -0
- package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -0
- package/dist/backends/kubernetes/k8s-client.js +553 -0
- package/dist/backends/kubernetes/k8s-client.js.map +1 -0
- package/dist/backends/kubernetes/lease.d.ts +145 -0
- package/dist/backends/kubernetes/lease.d.ts.map +1 -0
- package/dist/backends/kubernetes/lease.js +201 -0
- package/dist/backends/kubernetes/lease.js.map +1 -0
- package/dist/backends/kubernetes/objects.d.ts +702 -0
- package/dist/backends/kubernetes/objects.d.ts.map +1 -0
- package/dist/backends/kubernetes/objects.js +518 -0
- package/dist/backends/kubernetes/objects.js.map +1 -0
- package/dist/backends/kubernetes/per-sandbox-policy.d.ts +219 -0
- package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -0
- package/dist/backends/kubernetes/per-sandbox-policy.js +407 -0
- package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -0
- package/dist/backends/kubernetes/privilege-probe.d.ts +136 -0
- package/dist/backends/kubernetes/privilege-probe.d.ts.map +1 -0
- package/dist/backends/kubernetes/privilege-probe.js +185 -0
- package/dist/backends/kubernetes/privilege-probe.js.map +1 -0
- package/dist/backends/kubernetes/rbac.d.ts +153 -0
- package/dist/backends/kubernetes/rbac.d.ts.map +1 -0
- package/dist/backends/kubernetes/rbac.js +177 -0
- package/dist/backends/kubernetes/rbac.js.map +1 -0
- package/dist/backends/kubernetes/sandbox.d.ts +190 -0
- package/dist/backends/kubernetes/sandbox.d.ts.map +1 -0
- package/dist/backends/kubernetes/sandbox.js +433 -0
- package/dist/backends/kubernetes/sandbox.js.map +1 -0
- package/dist/backends/kubernetes/transport.d.ts +1048 -0
- package/dist/backends/kubernetes/transport.d.ts.map +1 -0
- package/dist/backends/kubernetes/transport.js +2093 -0
- package/dist/backends/kubernetes/transport.js.map +1 -0
- package/dist/backends/kubernetes/workspace.d.ts +1512 -0
- package/dist/backends/kubernetes/workspace.d.ts.map +1 -0
- package/dist/backends/kubernetes/workspace.js +3703 -0
- package/dist/backends/kubernetes/workspace.js.map +1 -0
- package/dist/backends/remote-execution-controller.d.ts +14 -0
- package/dist/backends/remote-execution-controller.d.ts.map +1 -1
- package/dist/backends/remote-execution-controller.js.map +1 -1
- package/dist/index.d.ts +350 -2
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +344 -34
- package/dist/index.js.map +1 -1
- package/dist/testing/sandbox-conformance.d.ts +227 -0
- package/dist/testing/sandbox-conformance.d.ts.map +1 -0
- package/dist/testing/sandbox-conformance.js +896 -0
- package/dist/testing/sandbox-conformance.js.map +1 -0
- package/package.json +5 -4
- package/src/backends/aci-standby-pool/index.ts +16 -1
- package/src/backends/docker/index.ts +22 -1
- package/src/backends/firecracker/index.ts +14 -2
- package/src/backends/firecracker/protocol.ts +541 -6
- package/src/backends/firecracker/transport.ts +1687 -64
- package/src/backends/kubernetes/egress-policy.ts +3448 -0
- package/src/backends/kubernetes/identity.ts +261 -0
- package/src/backends/kubernetes/index.ts +2670 -0
- package/src/backends/kubernetes/ingress-policy.ts +1344 -0
- package/src/backends/kubernetes/k8s-client.ts +742 -0
- package/src/backends/kubernetes/lease.ts +254 -0
- package/src/backends/kubernetes/objects.ts +983 -0
- package/src/backends/kubernetes/per-sandbox-policy.ts +542 -0
- package/src/backends/kubernetes/privilege-probe.ts +261 -0
- package/src/backends/kubernetes/rbac.ts +192 -0
- package/src/backends/kubernetes/sandbox.ts +593 -0
- package/src/backends/kubernetes/transport.ts +2895 -0
- package/src/backends/kubernetes/workspace.ts +5640 -0
- package/src/backends/remote-execution-controller.ts +14 -0
- package/src/index.ts +838 -35
- package/src/testing/sandbox-conformance.ts +1202 -0
|
@@ -0,0 +1,2670 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Kubernetes / agent-sandbox backend — acquire, readiness and teardown.
|
|
3
|
+
*
|
|
4
|
+
* Sibling of `firecracker/` and `aci-standby-pool/`: same
|
|
5
|
+
* {@link SandboxBackend} surface, same "the host supplies the credential
|
|
6
|
+
* callback and this package carries no cloud SDK" boundary, a different
|
|
7
|
+
* control plane. Here the control plane is the Kubernetes API server and the
|
|
8
|
+
* warm pool is a `SandboxWarmPool` reconciled by the agent-sandbox controller
|
|
9
|
+
* (kubernetes-sigs/agent-sandbox v1.0.2).
|
|
10
|
+
*
|
|
11
|
+
* Registered as `microvm` because {@link SandboxTier} names the strength of
|
|
12
|
+
* the boundary, not the orchestrator that starts it: a pod scheduled onto a
|
|
13
|
+
* Kata RuntimeClass runs in a hardware-virtualized guest. The tier also keeps
|
|
14
|
+
* the backend out of the container tier's mandatory
|
|
15
|
+
* `ContainerSandboxLayout`, which a remote-copy backend has no use for —
|
|
16
|
+
* Firecracker took the same exemption.
|
|
17
|
+
*
|
|
18
|
+
* ## Two acquire paths, and why they create different kinds
|
|
19
|
+
*
|
|
20
|
+
* - `warmPoolName` set → POST a `SandboxClaim` at the named pool. The
|
|
21
|
+
* controller binds an already-running Sandbox out of the pool, which is
|
|
22
|
+
* what makes a sub-second acquire possible at all.
|
|
23
|
+
* - `warmPoolName` unset → POST a `Sandbox` directly. This is necessitated
|
|
24
|
+
* rather than chosen: `SandboxClaimSpec.warmPoolRef` is a REQUIRED field,
|
|
25
|
+
* so a pool-less claim does not exist in the API.
|
|
26
|
+
*
|
|
27
|
+
* The claim this backend POSTs is PRISTINE: `warmPoolRef` and a lifecycle
|
|
28
|
+
* bound, nothing else. `spec.env` and `spec.volumeClaimTemplates` are
|
|
29
|
+
* available on the claim and are never set, because a claim carrying either
|
|
30
|
+
* is forced to cold-start instead of adopting a pool sandbox — the single
|
|
31
|
+
* most expensive mistake available on this path, and a silent one, since such
|
|
32
|
+
* a claim still works and only the latency shows it. Per-sandbox controls
|
|
33
|
+
* that would need those fields are refused by {@link assertEnforceable}
|
|
34
|
+
* rather than accepted and dropped.
|
|
35
|
+
*
|
|
36
|
+
* ## The bound sandbox is not named after the claim
|
|
37
|
+
*
|
|
38
|
+
* A pool sandbox keeps the generated name the pool gave it when the claim
|
|
39
|
+
* adopts it. The backend therefore reads the bound identity back out of
|
|
40
|
+
* `status.sandbox` and never derives it from the claim's own name. A test
|
|
41
|
+
* covers exactly that asymmetry.
|
|
42
|
+
*
|
|
43
|
+
* ## Credential
|
|
44
|
+
*
|
|
45
|
+
* The per-instance agent bind token is the backing pod's own
|
|
46
|
+
* `metadata.uid`: the host learns it from the API server after readiness, the
|
|
47
|
+
* guest learns it through the downward API (`NAMZU_AGENT_BIND_TOKEN`), and
|
|
48
|
+
* nothing has to be minted, stored or injected at claim time. One GET, no
|
|
49
|
+
* claim mutation, so the warm path stays pristine. A resumed pod is a new pod
|
|
50
|
+
* with a new uid, which is correct: it is a new instance.
|
|
51
|
+
*
|
|
52
|
+
* ## The lease
|
|
53
|
+
*
|
|
54
|
+
* The `shutdownTime` acquire stamps is the leak guard AND, unrenewed, a
|
|
55
|
+
* deadline on the run. The Sandbox handle owns a renewal loop that PATCHes
|
|
56
|
+
* it forward every half-TTL for as long as the handle is alive, and
|
|
57
|
+
* `destroy()` stops it — so a live run keeps its pod and a dead host still
|
|
58
|
+
* costs the cluster exactly one expiry. See `lease.ts`.
|
|
59
|
+
*
|
|
60
|
+
* ## The privilege probe
|
|
61
|
+
*
|
|
62
|
+
* `create()` does not resolve until the guest has reported — and this
|
|
63
|
+
* backend has checked — that it is deprivileged: all four capability masks
|
|
64
|
+
* zero, `NoNewPrivs: 1`, read out of `/proc/self/status` over the agent's
|
|
65
|
+
* `execute` op, on a clock of its own so a guest that goes quiet is refused
|
|
66
|
+
* rather than waited on. A refusal destroys the instance and rejects, so no
|
|
67
|
+
* handle to an under-hardened sandbox escapes. There is no off switch. See
|
|
68
|
+
* `privilege-probe.ts`.
|
|
69
|
+
*
|
|
70
|
+
* ## Not watch
|
|
71
|
+
*
|
|
72
|
+
* Readiness is polled against the shared {@link OperationDeadline}, exactly as
|
|
73
|
+
* ACI polls `provisioningState`. A watch would buy nothing on a path whose
|
|
74
|
+
* whole budget is under a second, and would cost resourceVersion tracking,
|
|
75
|
+
* bookmarks, 410-relist and reconnect backoff.
|
|
76
|
+
*
|
|
77
|
+
* ## Egress
|
|
78
|
+
*
|
|
79
|
+
* `config.egress` is optional and, when set, translated and VERIFIED — never
|
|
80
|
+
* created — by `egress-policy.ts`, in two steps. The NAMED object is GETted
|
|
81
|
+
* and compared to the translation exactly, once, lazily, on the first
|
|
82
|
+
* `create()`, so `buildKubernetesBackend` itself still contacts nothing.
|
|
83
|
+
* Then, because the API server UNIONS every policy selecting a pod, every
|
|
84
|
+
* `NetworkPolicy` in the namespace (and, under `engine: 'cilium'`, every
|
|
85
|
+
* `CiliumNetworkPolicy`) is enumerated against the pod's real labels and the
|
|
86
|
+
* create is refused when any of them lets out more than the translation does
|
|
87
|
+
* — a second policy widens egress however exactly the named one matches, and
|
|
88
|
+
* a `SandboxTemplate`'s own `networkPolicy` block becomes exactly such a
|
|
89
|
+
* policy. `egress.verify: 'named-object-only'` is the opt-out and restores
|
|
90
|
+
* the first step alone. Every Sandbox this file creates directly
|
|
91
|
+
* (`buildSandboxBody`) carries {@link sandboxTemplateLabel} on its
|
|
92
|
+
* podTemplate specifically so that translated policy's `podSelector` has
|
|
93
|
+
* something stable to match — see `objects.ts`'s doc comment on that label
|
|
94
|
+
* for why agent-sandbox's own controller-owned label does not cover this
|
|
95
|
+
* path.
|
|
96
|
+
*
|
|
97
|
+
* ## Ingress
|
|
98
|
+
*
|
|
99
|
+
* `config.ingress` is the opposite default: verification is ON unless a
|
|
100
|
+
* deployment says `'unverified'`. Before the POST for a direct Sandbox, and
|
|
101
|
+
* after the bind for a claimed one, `ingress-policy.ts` lists the namespace's
|
|
102
|
+
* policies and refuses unless one of them actually closes the agent port on
|
|
103
|
+
* the labels this pod carries. Nothing checked that before, while three
|
|
104
|
+
* pieces of shipped text said it was covered — see that module's own doc
|
|
105
|
+
* comment for what was measured.
|
|
106
|
+
*/
|
|
107
|
+
|
|
108
|
+
import type { Sandbox, SandboxNetworkPolicy } from '@namzu/sdk'
|
|
109
|
+
import { generateSandboxId } from '@namzu/sdk'
|
|
110
|
+
|
|
111
|
+
import type { SandboxBackend, SandboxBackendOptions } from '../../index.js'
|
|
112
|
+
import {
|
|
113
|
+
OperationDeadline,
|
|
114
|
+
OperationDeadlineExpired,
|
|
115
|
+
resolveReadinessOptions,
|
|
116
|
+
runFailureCleanup,
|
|
117
|
+
} from '../readiness.js'
|
|
118
|
+
import {
|
|
119
|
+
type EgressProfileLabel,
|
|
120
|
+
type KubernetesEgressConfig,
|
|
121
|
+
KubernetesPodLabelNotObservedError,
|
|
122
|
+
KubernetesPodLabelsRejectedError,
|
|
123
|
+
type KubernetesTranslatedEgressPolicy,
|
|
124
|
+
assertEgressPolicyIsEnforceable,
|
|
125
|
+
assertEgressProfileIsUsable,
|
|
126
|
+
assertPerSandboxEgressIsUsable,
|
|
127
|
+
composeAdditionalPodLabels,
|
|
128
|
+
defaultEgressPolicyName,
|
|
129
|
+
egressProfileLabel,
|
|
130
|
+
egressUnionVerificationEnabled,
|
|
131
|
+
perSandboxEgressLabelKey,
|
|
132
|
+
translateEgressPolicy,
|
|
133
|
+
verifyEgressPolicyApplied,
|
|
134
|
+
verifyEgressPolicyUnion,
|
|
135
|
+
} from './egress-policy.js'
|
|
136
|
+
import {
|
|
137
|
+
type KubernetesIngressConfig,
|
|
138
|
+
ingressVerificationEnabled,
|
|
139
|
+
resolveIngressEngine,
|
|
140
|
+
verifyIngressPolicyApplied,
|
|
141
|
+
} from './ingress-policy.js'
|
|
142
|
+
import {
|
|
143
|
+
type KubernetesAccess,
|
|
144
|
+
KubernetesAlreadyGoneError,
|
|
145
|
+
KubernetesApiError,
|
|
146
|
+
KubernetesApiTimeoutError,
|
|
147
|
+
type KubernetesClient,
|
|
148
|
+
type KubernetesClientOptions,
|
|
149
|
+
KubernetesCredentialError,
|
|
150
|
+
createKubernetesClient,
|
|
151
|
+
} from './k8s-client.js'
|
|
152
|
+
import {
|
|
153
|
+
type KubernetesCondition,
|
|
154
|
+
type PodListResource,
|
|
155
|
+
type PodResource,
|
|
156
|
+
READY_CONDITION,
|
|
157
|
+
SANDBOX_API_GROUP,
|
|
158
|
+
SANDBOX_API_VERSION,
|
|
159
|
+
SANDBOX_EXTENSIONS_API_GROUP,
|
|
160
|
+
type SandboxClaimListResource,
|
|
161
|
+
type SandboxClaimResource,
|
|
162
|
+
type SandboxPodTemplate,
|
|
163
|
+
type SandboxResource,
|
|
164
|
+
type SandboxTemplateResource,
|
|
165
|
+
type SandboxVolumeClaimTemplate,
|
|
166
|
+
type SandboxWarmPoolResource,
|
|
167
|
+
claimCollectionPath,
|
|
168
|
+
claimListPath,
|
|
169
|
+
claimPath,
|
|
170
|
+
isConditionTrue,
|
|
171
|
+
isPodLive,
|
|
172
|
+
podCollectionPath,
|
|
173
|
+
podListPath,
|
|
174
|
+
podPath,
|
|
175
|
+
readPodIP,
|
|
176
|
+
sandboxCollectionPath,
|
|
177
|
+
sandboxPath,
|
|
178
|
+
sandboxTemplateLabel,
|
|
179
|
+
sandboxTemplatePath,
|
|
180
|
+
warmPoolPath,
|
|
181
|
+
} from './objects.js'
|
|
182
|
+
import {
|
|
183
|
+
KubernetesOwnerUidMissingError,
|
|
184
|
+
type PerSandboxPolicyOwner,
|
|
185
|
+
buildAdmissionFence,
|
|
186
|
+
buildPerSandboxPolicySetter,
|
|
187
|
+
} from './per-sandbox-policy.js'
|
|
188
|
+
import { privilegeProbeTimedOut, runPrivilegeProbe } from './privilege-probe.js'
|
|
189
|
+
import { buildKubernetesSandbox } from './sandbox.js'
|
|
190
|
+
import { KubernetesAgentTransport } from './transport.js'
|
|
191
|
+
|
|
192
|
+
export type {
|
|
193
|
+
KubernetesEgressConfig,
|
|
194
|
+
KubernetesEgressEngine,
|
|
195
|
+
KubernetesEgressPolicy,
|
|
196
|
+
KubernetesEgressVerification,
|
|
197
|
+
KubernetesOnlyEgressPolicy,
|
|
198
|
+
KubernetesPerSandboxEgressConfig,
|
|
199
|
+
} from './egress-policy.js'
|
|
200
|
+
export type { KubernetesIngressConfig, KubernetesIngressEngine } from './ingress-policy.js'
|
|
201
|
+
|
|
202
|
+
/**
|
|
203
|
+
* How the backend reaches the API server. Two sources, neither needing a YAML
|
|
204
|
+
* parser — see `k8s-client.ts` for the reasoning.
|
|
205
|
+
*/
|
|
206
|
+
export type KubernetesClusterAccess =
|
|
207
|
+
| { readonly inCluster: true }
|
|
208
|
+
| {
|
|
209
|
+
readonly inCluster?: false
|
|
210
|
+
readonly server: string
|
|
211
|
+
readonly ca?: string | Buffer
|
|
212
|
+
readonly getToken: () => Promise<string>
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
export interface KubernetesBackendInternalConfig {
|
|
216
|
+
readonly access: KubernetesClusterAccess
|
|
217
|
+
/** Namespace the claims, sandboxes and pods live in. */
|
|
218
|
+
readonly namespace: string
|
|
219
|
+
/**
|
|
220
|
+
* SandboxTemplate whose `podTemplate` a POOL-LESS create copies into the
|
|
221
|
+
* Sandbox it posts. The warm path never reads it — the pool's own
|
|
222
|
+
* `sandboxTemplateRef` decides there — but an operator reading this config
|
|
223
|
+
* still learns which template these sandboxes are built from.
|
|
224
|
+
*/
|
|
225
|
+
readonly sandboxTemplateName: string
|
|
226
|
+
/** Named `SandboxWarmPool`. Absent → every create is a direct Sandbox. */
|
|
227
|
+
readonly warmPoolName?: string
|
|
228
|
+
/** TCP port the guest agent listens on. Default {@link DEFAULT_AGENT_PORT}. */
|
|
229
|
+
readonly agentPort?: number
|
|
230
|
+
/**
|
|
231
|
+
* Which of a sandbox's two addresses the transport dials. Default
|
|
232
|
+
* `'service'` — see {@link KubernetesAgentAddressMode}, which is where
|
|
233
|
+
* the choice is explained, because it is a fact about where the HOST
|
|
234
|
+
* runs rather than about the cluster.
|
|
235
|
+
*/
|
|
236
|
+
readonly agentAddress?: KubernetesAgentAddressMode
|
|
237
|
+
readonly readyPollIntervalMs?: number
|
|
238
|
+
readonly readyTimeoutMs?: number
|
|
239
|
+
/**
|
|
240
|
+
* Lifetime bound written into every created object, and the amount each
|
|
241
|
+
* lease renewal pushes the expiry forward. Default 1 hour.
|
|
242
|
+
*/
|
|
243
|
+
readonly claimTtlSeconds?: number
|
|
244
|
+
/**
|
|
245
|
+
* Every lease-renewal failure that is not "the object is already gone".
|
|
246
|
+
* `@namzu/sandbox` owns no logger and reads none from module scope, so a
|
|
247
|
+
* diagnostic it cannot print is handed to the host that can. Renewal
|
|
248
|
+
* retries on the next tick either way; nothing here changes behaviour.
|
|
249
|
+
*/
|
|
250
|
+
readonly onLeaseRenewalError?: (error: unknown) => void
|
|
251
|
+
/**
|
|
252
|
+
* RuntimeClass for a POOL-LESS create. Refused together with
|
|
253
|
+
* `warmPoolName`: a pooled sandbox's runtime class is fixed by the pool's
|
|
254
|
+
* SandboxTemplate and cannot be chosen per claim.
|
|
255
|
+
*/
|
|
256
|
+
readonly runtimeClassName?: string
|
|
257
|
+
/**
|
|
258
|
+
* Egress policy this backend's `NetworkPolicy` (or `CiliumNetworkPolicy`,
|
|
259
|
+
* under `engine: 'cilium'`) is expected to carry. Unset means this backend
|
|
260
|
+
* neither computes nor verifies one, and OUTBOUND traffic is then whatever
|
|
261
|
+
* the cluster's own policies happen to allow.
|
|
262
|
+
*
|
|
263
|
+
* Deliberately not described as "covered by the SandboxTemplate's managed
|
|
264
|
+
* NetworkPolicy", which is what this comment used to claim: that policy
|
|
265
|
+
* selects `agents.x-k8s.io/sandbox-template-ref-hash`, a label the
|
|
266
|
+
* controller writes only onto a Sandbox adopted out of a `SandboxWarmPool`
|
|
267
|
+
* and never onto one this backend POSTs — so for every pool-less sandbox
|
|
268
|
+
* and every workspace it selects nothing at all. See `ingress-policy.ts`.
|
|
269
|
+
*/
|
|
270
|
+
readonly egress?: KubernetesEgressConfig
|
|
271
|
+
/**
|
|
272
|
+
* Whether the agent port's INGRESS boundary is verified before a sandbox
|
|
273
|
+
* is created, and against which policy resources.
|
|
274
|
+
*
|
|
275
|
+
* Unset means VERIFY — the one field in this config whose absent value is
|
|
276
|
+
* the strict one, because the deployment that needs the check is the one
|
|
277
|
+
* that would never have set it. `'unverified'` reads no policy and issues
|
|
278
|
+
* no request, for a deployment whose boundary lives somewhere a namespaced
|
|
279
|
+
* Role cannot see. See `ingress-policy.ts`.
|
|
280
|
+
*/
|
|
281
|
+
readonly ingress?: KubernetesIngressConfig
|
|
282
|
+
/**
|
|
283
|
+
* Bound on every single Kubernetes API request this backend sends. See
|
|
284
|
+
* {@link KubernetesClientOptions.requestTimeoutMs} in `k8s-client.ts`,
|
|
285
|
+
* which owns the default (30 s), the floor (1 s) and the reason there is
|
|
286
|
+
* no value that disables it.
|
|
287
|
+
*/
|
|
288
|
+
readonly apiRequestTimeoutMs?: number
|
|
289
|
+
/**
|
|
290
|
+
* Interval of the negotiated liveness heartbeat on `openTerminal` and
|
|
291
|
+
* `openTcpConnection` streams. Default
|
|
292
|
+
* {@link DEFAULT_STREAM_HEARTBEAT_MS}; `0` turns it off and restores the
|
|
293
|
+
* pre-heartbeat behaviour exactly. See {@link resolveStreamHeartbeatMs}.
|
|
294
|
+
*/
|
|
295
|
+
readonly streamHeartbeatMs?: number
|
|
296
|
+
/**
|
|
297
|
+
* Extra labels written onto every `SandboxClaim` this backend POSTs —
|
|
298
|
+
* `metadata.labels`, and nowhere else. Never merged into
|
|
299
|
+
* `additionalPodMetadata`: those are POD labels a running Sandbox and its
|
|
300
|
+
* `NetworkPolicy` selectors read, and a host's own recovery bookkeeping
|
|
301
|
+
* has no business changing what a pod is selected by. Absent means no
|
|
302
|
+
* labels beyond what the controller itself writes, and every claim body
|
|
303
|
+
* this backend sends is byte-for-byte what it always was.
|
|
304
|
+
*
|
|
305
|
+
* The intended use is a host-instance identity — e.g.
|
|
306
|
+
* `{ 'sandbox.namzu.ai/host-instance': hostId }` — so a restarted host
|
|
307
|
+
* can find and {@link releaseKubernetesTaskSandboxes} its predecessor's
|
|
308
|
+
* claims well before `claimTtlSeconds` would reap them on its own.
|
|
309
|
+
*/
|
|
310
|
+
readonly claimLabels?: Record<string, string>
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
/**
|
|
314
|
+
* Default {@link KubernetesBackendInternalConfig.streamHeartbeatMs} — 15 s,
|
|
315
|
+
* so a stream whose peer vanished without a FIN or an RST is given up on
|
|
316
|
+
* within 45 s rather than never.
|
|
317
|
+
*
|
|
318
|
+
* The Kubernetes backend opts IN here; `VsockTransportOptions.heartbeatMs`
|
|
319
|
+
* stays undefined by default, because that transport is shared with the
|
|
320
|
+
* Firecracker tier and a default there would force-close an existing
|
|
321
|
+
* consumer's quiet-but-alive terminal.
|
|
322
|
+
*/
|
|
323
|
+
export const DEFAULT_STREAM_HEARTBEAT_MS = 15_000
|
|
324
|
+
|
|
325
|
+
/**
|
|
326
|
+
* Validate the configured stream-heartbeat interval, or supply the default.
|
|
327
|
+
*
|
|
328
|
+
* `0` IS accepted here, unlike `apiRequestTimeoutMs`: the heartbeat is a new
|
|
329
|
+
* capability that a deployment behind a middlebox with its own idea about
|
|
330
|
+
* unexpected frames may want off, and turning it off restores exactly the
|
|
331
|
+
* behaviour every release before this one had. An unanswered API request has
|
|
332
|
+
* no such prior behaviour worth restoring.
|
|
333
|
+
*/
|
|
334
|
+
export function resolveStreamHeartbeatMs(value: number | undefined): number {
|
|
335
|
+
if (value === undefined) return DEFAULT_STREAM_HEARTBEAT_MS
|
|
336
|
+
if (!Number.isSafeInteger(value) || value < 0) {
|
|
337
|
+
throw new Error(
|
|
338
|
+
`kubernetes: streamHeartbeatMs must be a non-negative integer, got ${JSON.stringify(
|
|
339
|
+
value,
|
|
340
|
+
)}. Use 0 to send no heartbeats at all, which is how every release before this one behaved; the default is ${DEFAULT_STREAM_HEARTBEAT_MS}ms.`,
|
|
341
|
+
)
|
|
342
|
+
}
|
|
343
|
+
return value
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
/**
|
|
347
|
+
* The client options every `createKubernetesClient` call in this backend is
|
|
348
|
+
* built with. One function so the five call sites — the provider, and each
|
|
349
|
+
* of the workspace verbs — cannot drift apart on which bounds they honour.
|
|
350
|
+
*/
|
|
351
|
+
export function clientOptions(config: KubernetesBackendInternalConfig): KubernetesClientOptions {
|
|
352
|
+
return { requestTimeoutMs: config.apiRequestTimeoutMs }
|
|
353
|
+
}
|
|
354
|
+
|
|
355
|
+
/**
|
|
356
|
+
* Which address a sandbox's agent is dialed at. A property of where the HOST
|
|
357
|
+
* runs, not of the cluster.
|
|
358
|
+
*
|
|
359
|
+
* - `'service'` (default) — the Sandbox's `status.serviceFQDN`,
|
|
360
|
+
* `<name>.<namespace>.svc.cluster.local`. It outlives the pod: a resumed
|
|
361
|
+
* workspace comes back behind the same name, and every dial re-resolves
|
|
362
|
+
* it. The catch is that only the cluster's own DNS answers it, so this is
|
|
363
|
+
* correct exactly when the host itself runs inside the cluster.
|
|
364
|
+
* - `'pod-ip'` — the bound pod's IP, read from the same `GET` that reads
|
|
365
|
+
* its uid, so the address and the bind token are always one pod's. For a
|
|
366
|
+
* host OUTSIDE the cluster on a routable pod network (a peered VNet, a
|
|
367
|
+
* node-local operator, a CI runner with a route): it needs no cluster
|
|
368
|
+
* resolver at all. The cost is that a pod IP dies with its pod, which is
|
|
369
|
+
* why a resume re-reads it and a connect failure re-reads it once.
|
|
370
|
+
*
|
|
371
|
+
* The cluster still decides whether either address is REACHABLE: `'pod-ip'`
|
|
372
|
+
* additionally needs a NetworkPolicy that admits the host's own address
|
|
373
|
+
* range on the agent port. Neither mode changes the bind token, the
|
|
374
|
+
* privilege probe or egress verification.
|
|
375
|
+
*/
|
|
376
|
+
export type KubernetesAgentAddressMode = 'service' | 'pod-ip'
|
|
377
|
+
|
|
378
|
+
/**
|
|
379
|
+
* The address the guest agent answers on, plus the token to present.
|
|
380
|
+
*
|
|
381
|
+
* Structurally the `tcp` arm of the transport's `SandboxAgentHandle`. It is
|
|
382
|
+
* declared here rather than imported so acquire does not depend on the
|
|
383
|
+
* transport landing first; the two are asserted equal where they meet.
|
|
384
|
+
*/
|
|
385
|
+
export interface KubernetesAgentAddress {
|
|
386
|
+
readonly kind: 'tcp'
|
|
387
|
+
readonly host: string
|
|
388
|
+
readonly port: number
|
|
389
|
+
readonly token: string
|
|
390
|
+
}
|
|
391
|
+
|
|
392
|
+
/** What the controller bound, read back off the object's own status. */
|
|
393
|
+
export interface KubernetesSandboxBinding {
|
|
394
|
+
/** The Sandbox's own name — NOT the claim's. */
|
|
395
|
+
readonly name: string
|
|
396
|
+
readonly podIPs?: readonly string[]
|
|
397
|
+
readonly serviceFQDN?: string
|
|
398
|
+
/** `Sandbox.status.selector`, when the path that read it had it for free. */
|
|
399
|
+
readonly podSelector?: string
|
|
400
|
+
}
|
|
401
|
+
|
|
402
|
+
/** One acquired sandbox: what it is, where it answers, how to give it back. */
|
|
403
|
+
export interface KubernetesAcquisition {
|
|
404
|
+
readonly binding: KubernetesSandboxBinding
|
|
405
|
+
readonly agent: KubernetesAgentAddress
|
|
406
|
+
/**
|
|
407
|
+
* Re-read the live pod and recompute {@link agent} from it. Present only
|
|
408
|
+
* under `agentAddress: 'pod-ip'`, where the address is a literal that
|
|
409
|
+
* dies with its pod; a Service FQDN needs no such thing, and handing one
|
|
410
|
+
* over anyway would give the default mode a re-read it never had.
|
|
411
|
+
*/
|
|
412
|
+
readonly refreshAgent?: (signal?: AbortSignal) => Promise<KubernetesAgentAddress>
|
|
413
|
+
/** API path of the object THIS backend created — the claim, or the Sandbox. */
|
|
414
|
+
readonly ownedPath: string
|
|
415
|
+
/**
|
|
416
|
+
* The object this backend created, identified well enough to be named in
|
|
417
|
+
* another object's `ownerReferences`: its kind, its name and the uid the
|
|
418
|
+
* API server assigned it.
|
|
419
|
+
*
|
|
420
|
+
* Present only when `config.egress.perSandbox` is configured, because it
|
|
421
|
+
* is read from the create reply and the readiness polls and nothing else
|
|
422
|
+
* needs it — a deployment that never writes a per-sandbox policy should
|
|
423
|
+
* not start carrying a field whose absence would otherwise be a bug.
|
|
424
|
+
*/
|
|
425
|
+
readonly owner?: PerSandboxPolicyOwner
|
|
426
|
+
/**
|
|
427
|
+
* The VALUE of the per-sandbox selector label on this sandbox's pod —
|
|
428
|
+
* the created object's name, confirmed on the bound pod before the
|
|
429
|
+
* sandbox was admitted. Present under the same condition as
|
|
430
|
+
* {@link owner}.
|
|
431
|
+
*/
|
|
432
|
+
readonly perSandboxLabelValue?: string
|
|
433
|
+
/** The TTL acquire stamped, which every renewal re-stamps. */
|
|
434
|
+
readonly ttlSeconds: number
|
|
435
|
+
/** DELETE that object. An already-gone object counts as released. */
|
|
436
|
+
release(signal?: AbortSignal): Promise<void>
|
|
437
|
+
/**
|
|
438
|
+
* Merge-PATCH the created object's expiry forward to `shutdownTime`.
|
|
439
|
+
*
|
|
440
|
+
* On the acquisition rather than in `lease.ts` because only this path
|
|
441
|
+
* knows WHICH object it created and therefore where the field lives: a
|
|
442
|
+
* `SandboxClaim` carries it at `spec.lifecycle.shutdownTime`, a directly
|
|
443
|
+
* created `Sandbox` at `spec.shutdownTime` (v1beta1 as served keeps it at
|
|
444
|
+
* the top of `spec`). A merge patch of the nested object leaves
|
|
445
|
+
* `shutdownPolicy` alone.
|
|
446
|
+
*/
|
|
447
|
+
renew(shutdownTime: string, signal?: AbortSignal): Promise<void>
|
|
448
|
+
}
|
|
449
|
+
|
|
450
|
+
/**
|
|
451
|
+
* The same number the Firecracker guest agent listens on over vsock
|
|
452
|
+
* (`DEFAULT_AGENT_VSOCK_PORT`), so one agent has one port across both tiers
|
|
453
|
+
* and a manifest, a NetworkPolicy and a transport can all name it from
|
|
454
|
+
* memory. Unprivileged, and the sandbox pod is not sharing it with anything.
|
|
455
|
+
*/
|
|
456
|
+
export const DEFAULT_AGENT_PORT = 1024
|
|
457
|
+
|
|
458
|
+
/**
|
|
459
|
+
* Deliberately far below ACI's 500 ms. A pool bind lands in ~120 ms on a warm
|
|
460
|
+
* cluster, so a half-second poll would spend most of the sub-second acquire
|
|
461
|
+
* budget asleep; 50 ms costs a handful of cheap GETs and gives the measurement
|
|
462
|
+
* somewhere to land.
|
|
463
|
+
*/
|
|
464
|
+
const DEFAULT_READY_POLL_MS = 50
|
|
465
|
+
const DEFAULT_READY_TIMEOUT_MS = 60_000
|
|
466
|
+
const DEFAULT_CLAIM_TTL_SECONDS = 3_600
|
|
467
|
+
|
|
468
|
+
/**
|
|
469
|
+
* The ceiling on the privilege probe's own clock — see
|
|
470
|
+
* {@link resolveProbeTimeoutMs} for where the rest of the number comes from.
|
|
471
|
+
*
|
|
472
|
+
* The probe is one `cat` of a pseudo-file over an already-established path,
|
|
473
|
+
* so half a minute would already be generous and a quarter of one is plenty.
|
|
474
|
+
* The number matters because the alternative is not "a bit longer": with no
|
|
475
|
+
* clock of its own the probe falls back on the execution controller's generic
|
|
476
|
+
* defaults — a five-minute execution observation, then a cancel-confirm and a
|
|
477
|
+
* drain — so a guest that accepts the TCP connection and then stops answering
|
|
478
|
+
* would keep a 60 s `create()` pending for over six minutes.
|
|
479
|
+
*/
|
|
480
|
+
const PRIVILEGE_PROBE_TIMEOUT_CAP_MS = 15_000
|
|
481
|
+
|
|
482
|
+
/**
|
|
483
|
+
* How long the privilege probe may take, given the caller's readiness budget.
|
|
484
|
+
*
|
|
485
|
+
* `readyTimeoutMs` bounds the CONTROL plane and has usually expired by the
|
|
486
|
+
* time the probe starts, so the probe cannot share it — but it is still the
|
|
487
|
+
* number the caller chose to describe how long an acquire may take, so the
|
|
488
|
+
* probe is allowed exactly that much again and no more, capped. A caller who
|
|
489
|
+
* asked for a 500 ms acquire gets a 500 ms probe; one who asked for two
|
|
490
|
+
* minutes of cold start still gets {@link PRIVILEGE_PROBE_TIMEOUT_CAP_MS}.
|
|
491
|
+
* Deliberately not a separate config key: a knob whose only correct value is
|
|
492
|
+
* "long enough for one `cat`" is a knob that only ever gets set wrong.
|
|
493
|
+
*/
|
|
494
|
+
export function resolveProbeTimeoutMs(readyTimeoutMs: number): number {
|
|
495
|
+
return Math.min(readyTimeoutMs, PRIVILEGE_PROBE_TIMEOUT_CAP_MS)
|
|
496
|
+
}
|
|
497
|
+
|
|
498
|
+
/**
|
|
499
|
+
* The readiness bounds every path in this backend polls against — acquire,
|
|
500
|
+
* and `workspace.ts`'s create/suspend/resume. One function so the two cannot
|
|
501
|
+
* drift apart on defaults.
|
|
502
|
+
*/
|
|
503
|
+
export function resolveKubernetesReadiness(config: {
|
|
504
|
+
readonly readyTimeoutMs?: number
|
|
505
|
+
readonly readyPollIntervalMs?: number
|
|
506
|
+
}): { readonly timeoutMs: number; readonly pollIntervalMs: number } {
|
|
507
|
+
return resolveReadinessOptions('kubernetes', config.readyTimeoutMs, config.readyPollIntervalMs, {
|
|
508
|
+
timeoutMs: DEFAULT_READY_TIMEOUT_MS,
|
|
509
|
+
pollIntervalMs: DEFAULT_READY_POLL_MS,
|
|
510
|
+
})
|
|
511
|
+
}
|
|
512
|
+
|
|
513
|
+
/**
|
|
514
|
+
* Per-sandbox controls this backend cannot apply, and therefore refuses.
|
|
515
|
+
*
|
|
516
|
+
* `env` is the load-bearing one. A SandboxClaim CAN carry `spec.env`, so this
|
|
517
|
+
* looks at first like a control that fits — but a claim that sets it is forced
|
|
518
|
+
* to cold-start instead of adopting a pool sandbox, which turns a 120 ms
|
|
519
|
+
* acquire into a full pod start. Accepting env here would buy a caller a
|
|
520
|
+
* feature and silently take the warm pool away, and the only symptom would be
|
|
521
|
+
* latency. The limits belong on the SandboxTemplate the pool is built from.
|
|
522
|
+
*
|
|
523
|
+
* `egress` is refused because this backend applies egress at the network
|
|
524
|
+
* layer, on the template's NetworkPolicy, which cannot be rewritten per
|
|
525
|
+
* running sandbox. The container backend's habit of emitting proxy
|
|
526
|
+
* environment variables as a substitute is not repeated here: a policy
|
|
527
|
+
* accepted and quietly not enforced is worse than one that is refused.
|
|
528
|
+
*/
|
|
529
|
+
const UNSUPPORTED_PER_SANDBOX_CONTROLS = [
|
|
530
|
+
['egress', 'network egress policy'],
|
|
531
|
+
['memoryLimitMb', 'memory limit'],
|
|
532
|
+
['maxProcesses', 'process limit'],
|
|
533
|
+
['env', 'environment variables'],
|
|
534
|
+
] as const
|
|
535
|
+
|
|
536
|
+
export function assertEnforceable(options: SandboxBackendOptions): void {
|
|
537
|
+
const unenforceable = UNSUPPORTED_PER_SANDBOX_CONTROLS.filter(([key]) => {
|
|
538
|
+
const value = options[key]
|
|
539
|
+
return value !== undefined && (key !== 'env' || Object.keys(value).length > 0)
|
|
540
|
+
})
|
|
541
|
+
if (unenforceable.length === 0) return
|
|
542
|
+
|
|
543
|
+
throw new Error(
|
|
544
|
+
`The kubernetes sandbox backend cannot enforce per-sandbox ${unenforceable
|
|
545
|
+
.map(([, label]) => label)
|
|
546
|
+
.join(
|
|
547
|
+
', ',
|
|
548
|
+
)}: a SandboxClaim carrying env or volumes is forced to cold-start instead of adopting a warm pool sandbox, and egress is a NetworkPolicy on the pool's SandboxTemplate rather than a per-sandbox setting. Set them on the SandboxTemplate the SandboxWarmPool is built from, or use a backend that applies them per sandbox. Refusing rather than accepting a control that would be silently dropped.`,
|
|
549
|
+
)
|
|
550
|
+
}
|
|
551
|
+
|
|
552
|
+
/**
|
|
553
|
+
* Refuse a runtime class the pool path cannot honour.
|
|
554
|
+
*
|
|
555
|
+
* A pooled sandbox is already running by the time a claim reaches it, under
|
|
556
|
+
* whatever RuntimeClass its SandboxTemplate named. `runtimeClassName` in this
|
|
557
|
+
* config would therefore be read, accepted and ignored — and the thing it
|
|
558
|
+
* selects is the VM boundary, which is the last control to lose quietly.
|
|
559
|
+
*/
|
|
560
|
+
export function assertRuntimeClassIsApplicable(config: {
|
|
561
|
+
warmPoolName?: string
|
|
562
|
+
runtimeClassName?: string
|
|
563
|
+
}): void {
|
|
564
|
+
if (config.runtimeClassName === undefined || config.warmPoolName === undefined) return
|
|
565
|
+
throw new Error(
|
|
566
|
+
`The kubernetes sandbox backend cannot apply runtimeClassName ${JSON.stringify(config.runtimeClassName)} to sandboxes claimed from warm pool ${JSON.stringify(config.warmPoolName)}: a pooled sandbox is already running under the RuntimeClass its SandboxTemplate named, and a claim cannot change it. Set runtimeClassName on that SandboxTemplate's podTemplate, or drop warmPoolName to have this backend create each Sandbox itself.`,
|
|
567
|
+
)
|
|
568
|
+
}
|
|
569
|
+
|
|
570
|
+
/**
|
|
571
|
+
* Build a {@link SandboxBackend} against a cluster running the agent-sandbox
|
|
572
|
+
* controller. Construction is synchronous and contacts nothing: readiness
|
|
573
|
+
* bounds and the config refusals are validated here so a misconfiguration
|
|
574
|
+
* surfaces during host wiring rather than mid-run, and the first API call
|
|
575
|
+
* happens on the first `create()`.
|
|
576
|
+
*/
|
|
577
|
+
export function buildKubernetesBackend(config: KubernetesBackendInternalConfig): SandboxBackend {
|
|
578
|
+
const readiness = resolveKubernetesReadiness(config)
|
|
579
|
+
assertRuntimeClassIsApplicable(config)
|
|
580
|
+
// A hostname allowlist with no FQDN-capable engine declared is a
|
|
581
|
+
// configuration error, not a runtime one — it can be decided from
|
|
582
|
+
// `config.egress.policy.kind` alone, with no API call, so it is refused
|
|
583
|
+
// here, synchronously, the same moment the two checks above are.
|
|
584
|
+
if (config.egress) {
|
|
585
|
+
assertEgressPolicyIsEnforceable(
|
|
586
|
+
config.egress.policy,
|
|
587
|
+
config.egress.engine ?? 'core',
|
|
588
|
+
config.egress.ciliumNarrowing,
|
|
589
|
+
)
|
|
590
|
+
}
|
|
591
|
+
// And the same for an egress PROFILE that could never be written as a
|
|
592
|
+
// label — a value the API server would reject leaves either a claim
|
|
593
|
+
// nothing binds or a policy nobody can apply, and both are decidable
|
|
594
|
+
// from config alone.
|
|
595
|
+
assertEgressProfileIsUsable(config.egress)
|
|
596
|
+
// And the same for per-sandbox egress: an engine that cannot express a
|
|
597
|
+
// hostname, an unnamed admission fence or a selector key the API server
|
|
598
|
+
// would refuse are all decidable from config alone, and a host that
|
|
599
|
+
// learns any of them from its first `setNetworkPolicy` call learns it an
|
|
600
|
+
// hour into a run that cannot be redone.
|
|
601
|
+
assertPerSandboxEgressIsUsable(config.egress)
|
|
602
|
+
// Resolved here as well as at each session, so a configuration this
|
|
603
|
+
// backend will never honour is refused while `buildKubernetesBackend` is
|
|
604
|
+
// still on the stack rather than on someone's first `create()`.
|
|
605
|
+
resolveStreamHeartbeatMs(config.streamHeartbeatMs)
|
|
606
|
+
const client = createKubernetesClient(clientAccess(config), clientOptions(config))
|
|
607
|
+
// Verify-not-trust runs once, lazily, on the first `create()` — never here,
|
|
608
|
+
// because `buildKubernetesBackend` is documented to contact nothing. A
|
|
609
|
+
// failed attempt is not cached: a transient API error should not wedge
|
|
610
|
+
// every later create() behind the same stale rejection forever. The
|
|
611
|
+
// boundary object holds both halves and both memos; it lives as long as
|
|
612
|
+
// this backend does, which is what makes the named-object check
|
|
613
|
+
// once-per-backend rather than once-per-create.
|
|
614
|
+
const egressBoundary = buildEgressBoundary(client, config, config.sandboxTemplateName)
|
|
615
|
+
// Ingress is cached PER LABEL SET rather than once per backend, because
|
|
616
|
+
// unlike the egress NAMED-object check it is a question about one pod: a
|
|
617
|
+
// pooled sandbox's labels come off the pool's template and a pool-less
|
|
618
|
+
// one's off this config, and a single memo would answer for a pod it never
|
|
619
|
+
// examined. Same failure handling as the egress memo — a failed attempt is
|
|
620
|
+
// dropped, so a transient API error does not wedge every later create()
|
|
621
|
+
// behind it. The egress UNION check is keyed the same way, for the same
|
|
622
|
+
// reason, and additionally expires: see {@link EGRESS_UNION_CACHE_TTL_MS}.
|
|
623
|
+
const ingressVerifier = buildIngressVerifier(client, config, new Map())
|
|
624
|
+
// One fence per backend, so its memo is shared by every sandbox this
|
|
625
|
+
// backend hands out rather than re-proved per handle. `undefined` when
|
|
626
|
+
// per-sandbox egress is not configured, which is what makes
|
|
627
|
+
// `setNetworkPolicy` absent from the handle — presence follows
|
|
628
|
+
// CONFIGURATION and never a runtime probe, so a caller's capability
|
|
629
|
+
// detection cannot depend on when it asked.
|
|
630
|
+
const perSandbox = config.egress?.perSandbox
|
|
631
|
+
const fence = perSandbox === undefined ? undefined : buildAdmissionFence(client, perSandbox)
|
|
632
|
+
return {
|
|
633
|
+
tier: 'microvm',
|
|
634
|
+
name: 'kubernetes',
|
|
635
|
+
async create(options: SandboxBackendOptions): Promise<Sandbox> {
|
|
636
|
+
await egressBoundary?.verifyNamedObject(options.signal)
|
|
637
|
+
const acquisition = await acquireKubernetesSandbox(
|
|
638
|
+
client,
|
|
639
|
+
config,
|
|
640
|
+
options,
|
|
641
|
+
readiness,
|
|
642
|
+
ingressVerifier,
|
|
643
|
+
egressBoundary,
|
|
644
|
+
)
|
|
645
|
+
const egress = config.egress
|
|
646
|
+
const setNetworkPolicy =
|
|
647
|
+
fence !== undefined &&
|
|
648
|
+
egress?.perSandbox !== undefined &&
|
|
649
|
+
acquisition.owner !== undefined &&
|
|
650
|
+
acquisition.perSandboxLabelValue !== undefined
|
|
651
|
+
? buildPerSandboxPolicySetter({
|
|
652
|
+
client,
|
|
653
|
+
fence,
|
|
654
|
+
namespace: config.namespace,
|
|
655
|
+
egress: { ...egress, perSandbox: egress.perSandbox },
|
|
656
|
+
owner: acquisition.owner,
|
|
657
|
+
selectorValue: acquisition.perSandboxLabelValue,
|
|
658
|
+
})
|
|
659
|
+
: undefined
|
|
660
|
+
return await admitProbedSandbox(
|
|
661
|
+
acquisition,
|
|
662
|
+
config,
|
|
663
|
+
options,
|
|
664
|
+
resolveProbeTimeoutMs(readiness.timeoutMs),
|
|
665
|
+
setNetworkPolicy,
|
|
666
|
+
)
|
|
667
|
+
},
|
|
668
|
+
}
|
|
669
|
+
}
|
|
670
|
+
|
|
671
|
+
/**
|
|
672
|
+
* The two egress checks a create path runs, with their memos.
|
|
673
|
+
*
|
|
674
|
+
* `undefined` when `config.egress` is unset — the whole boundary is one
|
|
675
|
+
* absent object rather than a flag every call site re-reads, the same shape
|
|
676
|
+
* {@link IngressVerifier} uses for its own opt-out.
|
|
677
|
+
*
|
|
678
|
+
* Exported because `workspace.ts` runs the identical steps: a workspace does
|
|
679
|
+
* not go through `buildKubernetesBackend`, and a config `egress` honoured on
|
|
680
|
+
* one entry point and ignored on the other would be a silent downgrade of the
|
|
681
|
+
* boundary this backend calls primary. It builds its own boundary per create,
|
|
682
|
+
* which is what makes its checks per-call rather than memoized — creating a
|
|
683
|
+
* workspace is a rare, explicit act with nothing to amortise, and a policy
|
|
684
|
+
* deleted since the last call must be noticed.
|
|
685
|
+
*/
|
|
686
|
+
export interface KubernetesEgressBoundary {
|
|
687
|
+
/**
|
|
688
|
+
* Translate `egress.policy` and confirm an operator applied a matching
|
|
689
|
+
* object — the original verify-not-trust step, unchanged, including its
|
|
690
|
+
* exact-match comparison and its once-per-boundary memo.
|
|
691
|
+
*/
|
|
692
|
+
verifyNamedObject(signal?: AbortSignal): Promise<void>
|
|
693
|
+
/**
|
|
694
|
+
* Enumerate every policy selecting THIS pod and refuse when any of them
|
|
695
|
+
* allows egress the translation does not. A no-op under
|
|
696
|
+
* `egress.verify: 'named-object-only'`.
|
|
697
|
+
*/
|
|
698
|
+
verifyUnion(
|
|
699
|
+
podLabels: Readonly<Record<string, string>>,
|
|
700
|
+
subject: string,
|
|
701
|
+
signal?: AbortSignal,
|
|
702
|
+
): Promise<void>
|
|
703
|
+
}
|
|
704
|
+
|
|
705
|
+
/**
|
|
706
|
+
* How long a union pass is trusted for one label set.
|
|
707
|
+
*
|
|
708
|
+
* Five minutes rather than the backend's lifetime, which is what the
|
|
709
|
+
* named-object check alone used to get: an operator who applies a widening
|
|
710
|
+
* policy at 10:00 should not have it go unnoticed until the host restarts.
|
|
711
|
+
* It is a cache, not a watch — `k8s-client.ts`'s "no watch, no informers, no
|
|
712
|
+
* resourceVersion tracking" invariant is untouched, because the only thing
|
|
713
|
+
* kept across calls is "this exact label set passed at this time".
|
|
714
|
+
*/
|
|
715
|
+
export const EGRESS_UNION_CACHE_TTL_MS = 5 * 60 * 1_000
|
|
716
|
+
|
|
717
|
+
/** One cached pass, and when it was taken. */
|
|
718
|
+
interface CachedPass {
|
|
719
|
+
readonly at: number
|
|
720
|
+
readonly pending: Promise<void>
|
|
721
|
+
}
|
|
722
|
+
|
|
723
|
+
/**
|
|
724
|
+
* Build the egress boundary this config asks for, or nothing at all.
|
|
725
|
+
*
|
|
726
|
+
* `sandboxTemplateName` is the template the caller is actually building from
|
|
727
|
+
* — it decides both the default policy name and the pod label the policy's
|
|
728
|
+
* selector has to match, and a workspace may be built from a different
|
|
729
|
+
* template than the task path's.
|
|
730
|
+
*
|
|
731
|
+
* `now` is injected only so the TTL above can be tested without waiting five
|
|
732
|
+
* minutes; nothing else passes it.
|
|
733
|
+
*/
|
|
734
|
+
export function buildEgressBoundary(
|
|
735
|
+
client: KubernetesClient,
|
|
736
|
+
config: KubernetesBackendInternalConfig,
|
|
737
|
+
sandboxTemplateName: string,
|
|
738
|
+
now: () => number = Date.now,
|
|
739
|
+
): KubernetesEgressBoundary | undefined {
|
|
740
|
+
const egress = config.egress
|
|
741
|
+
if (egress === undefined) return undefined
|
|
742
|
+
const engine = egress.engine ?? 'core'
|
|
743
|
+
// One resolution of the profile, shared by the policy NAME and the policy
|
|
744
|
+
// SELECTOR: under a profile the default name gains the profile segment
|
|
745
|
+
// (one template under two profiles is two policy objects) and the
|
|
746
|
+
// selector gains the label, and the two must not be able to disagree.
|
|
747
|
+
const profile = egressProfileLabel(egress)
|
|
748
|
+
const target = {
|
|
749
|
+
namespace: config.namespace,
|
|
750
|
+
name: egress.networkPolicyName ?? defaultEgressPolicyName(sandboxTemplateName, profile?.value),
|
|
751
|
+
sandboxTemplateName,
|
|
752
|
+
...(profile !== undefined ? { profile } : {}),
|
|
753
|
+
}
|
|
754
|
+
// Translated ONCE per boundary, not once per check: a `resolver` policy's
|
|
755
|
+
// `resolve()` is the host's own closure and may cost a network call, and
|
|
756
|
+
// running the two checks against two independently resolved allowlists
|
|
757
|
+
// would compare each against a different translation.
|
|
758
|
+
let translation: Promise<KubernetesTranslatedEgressPolicy> | undefined
|
|
759
|
+
const translate = (): Promise<KubernetesTranslatedEgressPolicy> => {
|
|
760
|
+
translation ??= translateEgressPolicy(
|
|
761
|
+
egress.policy,
|
|
762
|
+
engine,
|
|
763
|
+
target,
|
|
764
|
+
egress.ciliumNarrowing,
|
|
765
|
+
).catch((err: unknown) => {
|
|
766
|
+
translation = undefined
|
|
767
|
+
throw err
|
|
768
|
+
})
|
|
769
|
+
return translation
|
|
770
|
+
}
|
|
771
|
+
let namedObject: Promise<void> | undefined
|
|
772
|
+
const passes = new Map<string, CachedPass>()
|
|
773
|
+
// The per-sandbox selector label carries a once-ever value, so it is
|
|
774
|
+
// excluded from the memo key — see {@link policyCacheKey}. Resolved once
|
|
775
|
+
// here rather than per check, because the resolver validates as it
|
|
776
|
+
// resolves and a per-check throw would surface from a cache lookup.
|
|
777
|
+
const perSandboxKey = perSandboxEgressLabelKey(egress)
|
|
778
|
+
return {
|
|
779
|
+
async verifyNamedObject(signal) {
|
|
780
|
+
namedObject ??= (async () => {
|
|
781
|
+
await verifyEgressPolicyApplied(client, await translate(), signal)
|
|
782
|
+
})().catch((err: unknown) => {
|
|
783
|
+
namedObject = undefined
|
|
784
|
+
throw err
|
|
785
|
+
})
|
|
786
|
+
await namedObject
|
|
787
|
+
},
|
|
788
|
+
async verifyUnion(podLabels, subject, signal) {
|
|
789
|
+
if (!egressUnionVerificationEnabled(egress)) return
|
|
790
|
+
const translated = await translate()
|
|
791
|
+
const key = policyCacheKey(podLabels, perSandboxKey)
|
|
792
|
+
const cached = passes.get(key)
|
|
793
|
+
if (cached !== undefined && now() - cached.at < EGRESS_UNION_CACHE_TTL_MS) {
|
|
794
|
+
await cached.pending
|
|
795
|
+
return
|
|
796
|
+
}
|
|
797
|
+
const pending = verifyEgressPolicyUnion(
|
|
798
|
+
client,
|
|
799
|
+
translated,
|
|
800
|
+
{ namespace: config.namespace, podLabels, engine, subject },
|
|
801
|
+
signal,
|
|
802
|
+
).catch((err: unknown) => {
|
|
803
|
+
// A failed attempt is never cached — same rule the named-object
|
|
804
|
+
// memo has always had.
|
|
805
|
+
passes.delete(key)
|
|
806
|
+
throw err
|
|
807
|
+
})
|
|
808
|
+
passes.set(key, { at: now(), pending })
|
|
809
|
+
await pending
|
|
810
|
+
},
|
|
811
|
+
}
|
|
812
|
+
}
|
|
813
|
+
|
|
814
|
+
/**
|
|
815
|
+
* What a create path calls to prove the agent port is closed before it hands
|
|
816
|
+
* a sandbox back. `undefined` when `config.ingress` is `'unverified'`, so the
|
|
817
|
+
* opt-out is one absent function rather than a flag every call site re-reads.
|
|
818
|
+
*/
|
|
819
|
+
export type IngressVerifier = (
|
|
820
|
+
podLabels: Readonly<Record<string, string>>,
|
|
821
|
+
subject: string,
|
|
822
|
+
signal?: AbortSignal,
|
|
823
|
+
) => Promise<void>
|
|
824
|
+
|
|
825
|
+
/**
|
|
826
|
+
* Canonical key for one label set — order-independent, so two spellings of
|
|
827
|
+
* the same pod share a memo.
|
|
828
|
+
*
|
|
829
|
+
* `excludeKey` drops the PER-SANDBOX egress label, whose value is unique per
|
|
830
|
+
* acquire. Both memos exist to amortise a namespace-wide policy enumeration
|
|
831
|
+
* across every sandbox a backend produces, and a key that carried a
|
|
832
|
+
* once-ever value would give every acquire a miss and leave an entry behind
|
|
833
|
+
* that nothing ever looks up again — a full enumeration per sandbox, and a
|
|
834
|
+
* Map that grows for the host's whole life.
|
|
835
|
+
*
|
|
836
|
+
* Dropping it is sound at the moment these checks run: the only policy that
|
|
837
|
+
* could select a pod BY that key and value is that sandbox's own, whose name
|
|
838
|
+
* is generated in the same call and which does not exist yet. Every other
|
|
839
|
+
* policy selecting the pod — the operator's baseline, a per-profile one,
|
|
840
|
+
* anything hand-written — selects on the labels that remain, so two pods
|
|
841
|
+
* differing only in this label are the same question. The label itself is
|
|
842
|
+
* still PRESENT in the label set each check is run against; only the memo's
|
|
843
|
+
* key ignores it.
|
|
844
|
+
*/
|
|
845
|
+
function policyCacheKey(podLabels: Readonly<Record<string, string>>, excludeKey?: string): string {
|
|
846
|
+
return JSON.stringify(
|
|
847
|
+
Object.entries(podLabels)
|
|
848
|
+
.filter(([k]) => k !== excludeKey)
|
|
849
|
+
.sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0)),
|
|
850
|
+
)
|
|
851
|
+
}
|
|
852
|
+
|
|
853
|
+
/**
|
|
854
|
+
* Build the ingress check this config asks for, or nothing at all.
|
|
855
|
+
*
|
|
856
|
+
* `cache` is the provider path's per-label-set memo; a workspace passes none,
|
|
857
|
+
* and re-checks on every call for the same reason its egress check does —
|
|
858
|
+
* creating a workspace is a rare, explicit act with nothing to amortise, and
|
|
859
|
+
* a policy deleted since the last call must be noticed.
|
|
860
|
+
*/
|
|
861
|
+
export function buildIngressVerifier(
|
|
862
|
+
client: KubernetesClient,
|
|
863
|
+
config: KubernetesBackendInternalConfig,
|
|
864
|
+
cache?: Map<string, Promise<void>>,
|
|
865
|
+
): IngressVerifier | undefined {
|
|
866
|
+
if (!ingressVerificationEnabled(config.ingress)) return undefined
|
|
867
|
+
const engine = resolveIngressEngine(config.ingress, config.egress?.engine)
|
|
868
|
+
const agentPort = config.agentPort ?? DEFAULT_AGENT_PORT
|
|
869
|
+
// Same exclusion, same reason, as the egress union memo above: a
|
|
870
|
+
// once-ever label value in the key would make this memo a per-acquire
|
|
871
|
+
// miss and an unbounded Map. See {@link policyCacheKey}.
|
|
872
|
+
const perSandboxKey = perSandboxEgressLabelKey(config.egress)
|
|
873
|
+
return async (podLabels, subject, signal) => {
|
|
874
|
+
const target = { namespace: config.namespace, podLabels, agentPort, engine, subject }
|
|
875
|
+
if (cache === undefined) {
|
|
876
|
+
await verifyIngressPolicyApplied(client, target, signal)
|
|
877
|
+
return
|
|
878
|
+
}
|
|
879
|
+
const key = policyCacheKey(podLabels, perSandboxKey)
|
|
880
|
+
let pending = cache.get(key)
|
|
881
|
+
if (pending === undefined) {
|
|
882
|
+
pending = verifyIngressPolicyApplied(client, target, signal).catch((err: unknown) => {
|
|
883
|
+
cache.delete(key)
|
|
884
|
+
throw err
|
|
885
|
+
})
|
|
886
|
+
cache.set(key, pending)
|
|
887
|
+
}
|
|
888
|
+
await pending
|
|
889
|
+
}
|
|
890
|
+
}
|
|
891
|
+
|
|
892
|
+
/** Config → the client's own access shape. Shared with `workspace.ts`. */
|
|
893
|
+
export function clientAccess(config: KubernetesBackendInternalConfig): KubernetesAccess {
|
|
894
|
+
const access = config.access
|
|
895
|
+
if (access.inCluster === true) return { inCluster: true }
|
|
896
|
+
return {
|
|
897
|
+
server: access.server,
|
|
898
|
+
namespace: config.namespace,
|
|
899
|
+
getToken: access.getToken,
|
|
900
|
+
...(access.ca !== undefined ? { ca: access.ca } : {}),
|
|
901
|
+
}
|
|
902
|
+
}
|
|
903
|
+
|
|
904
|
+
/**
|
|
905
|
+
* Why an acquire was refused, in the terms an operator acts on rather than
|
|
906
|
+
* the terms the failure happened to arrive in.
|
|
907
|
+
*
|
|
908
|
+
* - `'api-unreachable'` — the API server could not be reached, or kept
|
|
909
|
+
* answering with a status that means "not now": a connect failure, a 429,
|
|
910
|
+
* a 5xx. Retried inside the readiness budget before it ever reaches a
|
|
911
|
+
* caller, so seeing it means the whole budget was spent failing.
|
|
912
|
+
* - `'api-timeout'` — requests were accepted and never answered, until
|
|
913
|
+
* `apiRequestTimeoutMs` gave up on them. Also retried first.
|
|
914
|
+
* - `'forbidden'` — 401 or 403. The host's ServiceAccount cannot do this;
|
|
915
|
+
* no amount of waiting changes that. See the RBAC section of
|
|
916
|
+
* `docs/sdk/kubernetes-sandbox.md`.
|
|
917
|
+
* - `'claim-rejected'` — the controller REFUSED the claim, and said why.
|
|
918
|
+
* {@link KubernetesAcquireError.controllerReason} carries its own word for
|
|
919
|
+
* it. This is the one that used to burn the entire readiness budget before
|
|
920
|
+
* failing.
|
|
921
|
+
* - `'capacity'` — the pod exists and cannot be placed: the scheduler
|
|
922
|
+
* reports `PodScheduled=False` with reason `Unschedulable`. The cluster is
|
|
923
|
+
* full, or nothing matches the template's placement rules.
|
|
924
|
+
* - `'image-pull'` — the pod was placed and its container cannot start
|
|
925
|
+
* because the image will not pull. Permanent until an operator fixes the
|
|
926
|
+
* reference or the pull credential.
|
|
927
|
+
* - `'not-ready'` — none of the above: the readiness budget expired with the
|
|
928
|
+
* cluster reporting nothing wrong. A slow cold start, a webhook, an
|
|
929
|
+
* admission controller, a CNI that never attached the pod.
|
|
930
|
+
*/
|
|
931
|
+
export type KubernetesAcquireFailureReason =
|
|
932
|
+
| 'api-unreachable'
|
|
933
|
+
| 'api-timeout'
|
|
934
|
+
| 'forbidden'
|
|
935
|
+
| 'claim-rejected'
|
|
936
|
+
| 'capacity'
|
|
937
|
+
| 'image-pull'
|
|
938
|
+
| 'not-ready'
|
|
939
|
+
|
|
940
|
+
/**
|
|
941
|
+
* An acquire that was refused, carrying WHY in a field rather than in prose.
|
|
942
|
+
*
|
|
943
|
+
* Before this class a burst past node capacity and an API outage were the
|
|
944
|
+
* same plain `Error`, and a host could only tell them apart by matching
|
|
945
|
+
* message text that any release is free to reword. `reason` is the diagnosis,
|
|
946
|
+
* `retryable` is the advice that follows from it, and `cause` is the original
|
|
947
|
+
* failure — unmodified, so a host that already catches
|
|
948
|
+
* {@link ReadinessPollTimeout}, {@link KubernetesApiTimeoutError} or
|
|
949
|
+
* `KubernetesCredentialError` finds it there.
|
|
950
|
+
*
|
|
951
|
+
* `retryable` is about THIS acquire being worth attempting again, not about
|
|
952
|
+
* anything having been retried. Transient API failures are already retried
|
|
953
|
+
* inside the readiness budget, so a `retryable: true` that reaches a caller
|
|
954
|
+
* means the whole budget was spent on them.
|
|
955
|
+
*
|
|
956
|
+
* Not every acquire failure becomes one of these, and that is deliberate: a
|
|
957
|
+
* refusal this class cannot honestly diagnose — a malformed template, a 400
|
|
958
|
+
* from an admission webhook, a controller that reported Ready and named no
|
|
959
|
+
* sandbox — travels out as itself rather than being filed under whichever of
|
|
960
|
+
* the seven reasons is least wrong. `KubernetesApiError` carries the status
|
|
961
|
+
* for those.
|
|
962
|
+
*/
|
|
963
|
+
export class KubernetesAcquireError extends Error {
|
|
964
|
+
override readonly name = 'KubernetesAcquireError'
|
|
965
|
+
readonly reason: KubernetesAcquireFailureReason
|
|
966
|
+
/** Whether attempting the same acquire again could plausibly succeed. */
|
|
967
|
+
readonly retryable: boolean
|
|
968
|
+
/**
|
|
969
|
+
* `status.conditions[Ready].reason`, verbatim, when the controller
|
|
970
|
+
* refused the claim — `WarmPoolNotFound`, `TemplateNotFound`,
|
|
971
|
+
* `InvalidMetadata`, `EnvVarsInjectionRejected`. Present only for
|
|
972
|
+
* `'claim-rejected'`.
|
|
973
|
+
*/
|
|
974
|
+
readonly controllerReason?: string
|
|
975
|
+
/** The controller's own message for the same condition. */
|
|
976
|
+
readonly controllerMessage?: string
|
|
977
|
+
|
|
978
|
+
constructor(details: {
|
|
979
|
+
readonly reason: KubernetesAcquireFailureReason
|
|
980
|
+
readonly retryable: boolean
|
|
981
|
+
readonly message: string
|
|
982
|
+
readonly controllerReason?: string
|
|
983
|
+
readonly controllerMessage?: string
|
|
984
|
+
readonly cause?: unknown
|
|
985
|
+
}) {
|
|
986
|
+
super(details.message, details.cause !== undefined ? { cause: details.cause } : undefined)
|
|
987
|
+
this.reason = details.reason
|
|
988
|
+
this.retryable = details.retryable
|
|
989
|
+
if (details.controllerReason !== undefined) this.controllerReason = details.controllerReason
|
|
990
|
+
if (details.controllerMessage !== undefined) this.controllerMessage = details.controllerMessage
|
|
991
|
+
}
|
|
992
|
+
}
|
|
993
|
+
|
|
994
|
+
/**
|
|
995
|
+
* The `status.conditions[Ready].reason` values that mean the controller has
|
|
996
|
+
* DECIDED, so waiting is pointless.
|
|
997
|
+
*
|
|
998
|
+
* ## Where these strings came from
|
|
999
|
+
*
|
|
1000
|
+
* Not from the issue that asked for this, and not from upstream source: this
|
|
1001
|
+
* repo vendors none of agent-sandbox's Go, so a literal copied out of a
|
|
1002
|
+
* changelog is a literal nobody here can check. Each of the four was produced
|
|
1003
|
+
* against the deployed controller (kind v1.37.0, agent-sandbox v1.0.2,
|
|
1004
|
+
* 2026-09-17) by making the claim it describes and reading the condition
|
|
1005
|
+
* back:
|
|
1006
|
+
*
|
|
1007
|
+
* | reason | how it was produced | the controller's message |
|
|
1008
|
+
* |---|---|---|
|
|
1009
|
+
* | `WarmPoolNotFound` | claim at a pool that does not exist | `SandboxWarmPool "…" not found` |
|
|
1010
|
+
* | `TemplateNotFound` | claim at a pool whose template does not exist | `SandboxTemplate "…" not found` |
|
|
1011
|
+
* | `InvalidMetadata` | claim with an `additionalPodMetadata` label outside the allowed domains | `invalid additionalPodMetadata: …` |
|
|
1012
|
+
* | `EnvVarsInjectionRejected` | claim with `spec.env` against a template that forbids injection | `environment variable injection rejected: …` |
|
|
1013
|
+
*
|
|
1014
|
+
* The transient reasons seen on the SAME cluster, which must NOT be in this
|
|
1015
|
+
* set, were `DependenciesNotReady` (pod exists, still Pending) and
|
|
1016
|
+
* `DependenciesReady` (the Ready=True reason).
|
|
1017
|
+
*
|
|
1018
|
+
* ## Why an unknown reason is not terminal
|
|
1019
|
+
*
|
|
1020
|
+
* A wrong literal here fails in one of two ways, and only one of them is
|
|
1021
|
+
* recoverable. Too few entries: a rejected claim waits out the readiness
|
|
1022
|
+
* budget, which is exactly the behaviour every release before this one had.
|
|
1023
|
+
* Too many: an acquire that would have succeeded is refused on a guess. So
|
|
1024
|
+
* the set is a closed list of measured strings and everything else falls
|
|
1025
|
+
* through to the deadline.
|
|
1026
|
+
*/
|
|
1027
|
+
// Frozen because it is exported from the package root: an array handed to
|
|
1028
|
+
// every consumer is one a cast can push onto, and an entry added there would
|
|
1029
|
+
// change fail-fast for the whole process. The `readonly string[]` annotation
|
|
1030
|
+
// is deliberate rather than `as const` — `includes` on a literal tuple only
|
|
1031
|
+
// accepts the literals, and the whole point is to ask it about a reason no
|
|
1032
|
+
// one here has seen.
|
|
1033
|
+
export const TERMINAL_CLAIM_REASONS: readonly string[] = Object.freeze([
|
|
1034
|
+
'WarmPoolNotFound',
|
|
1035
|
+
'TemplateNotFound',
|
|
1036
|
+
'InvalidMetadata',
|
|
1037
|
+
'EnvVarsInjectionRejected',
|
|
1038
|
+
])
|
|
1039
|
+
|
|
1040
|
+
/**
|
|
1041
|
+
* The `Ready` condition a claim reports, whatever its status — the one
|
|
1042
|
+
* {@link isConditionTrue} deliberately cannot return, because it answers a
|
|
1043
|
+
* boolean question and this one needs the reason.
|
|
1044
|
+
*/
|
|
1045
|
+
function readyCondition(
|
|
1046
|
+
conditions: readonly KubernetesCondition[] | undefined,
|
|
1047
|
+
): KubernetesCondition | undefined {
|
|
1048
|
+
return conditions?.find((c) => c.type === READY_CONDITION)
|
|
1049
|
+
}
|
|
1050
|
+
|
|
1051
|
+
/**
|
|
1052
|
+
* What this backend asked the controller to put on the claim's pod, carried
|
|
1053
|
+
* into the rejection so an `InvalidMetadata` refusal can say what was sent
|
|
1054
|
+
* and which knob changes it.
|
|
1055
|
+
*
|
|
1056
|
+
* Context, never a second decision: whether a claim is refused at all is
|
|
1057
|
+
* {@link TERMINAL_CLAIM_REASONS}' answer and nobody else's, so an empty map
|
|
1058
|
+
* changes nothing about the class thrown, the reason on it, or when it is
|
|
1059
|
+
* raised.
|
|
1060
|
+
*/
|
|
1061
|
+
export interface ClaimPodMetadataContext {
|
|
1062
|
+
readonly namespace: string
|
|
1063
|
+
/** `spec.additionalPodMetadata.labels`, exactly as sent. */
|
|
1064
|
+
readonly requestedPodLabels: Readonly<Record<string, string>>
|
|
1065
|
+
/** The egress profile among those labels, when one is configured. */
|
|
1066
|
+
readonly profile?: EgressProfileLabel
|
|
1067
|
+
}
|
|
1068
|
+
|
|
1069
|
+
/**
|
|
1070
|
+
* A claim the controller has refused, or `undefined` for one it is still
|
|
1071
|
+
* working on.
|
|
1072
|
+
*
|
|
1073
|
+
* `status: 'False'` alone is not a refusal — it is also what a claim looks
|
|
1074
|
+
* like for the whole of a cold start — so the REASON decides, against
|
|
1075
|
+
* {@link TERMINAL_CLAIM_REASONS}.
|
|
1076
|
+
*
|
|
1077
|
+
* ONE class comes out of here whatever the reason, and that is deliberate:
|
|
1078
|
+
* `InvalidMetadata` is how the controller refuses a pod label whose domain is
|
|
1079
|
+
* not on its allowlist — the profile's, today — and a host's `catch` must not
|
|
1080
|
+
* have to be written differently depending on whether a profile happens to be
|
|
1081
|
+
* configured. `metadata` only decides what rides along as the `cause`: a
|
|
1082
|
+
* {@link KubernetesPodLabelsRejectedError} naming the map that was sent
|
|
1083
|
+
* and the `config.egress.profileLabelKey` that moves it, which the
|
|
1084
|
+
* controller's own message cannot know about.
|
|
1085
|
+
*/
|
|
1086
|
+
export function classifyClaimRejection(
|
|
1087
|
+
claim: SandboxClaimResource | undefined,
|
|
1088
|
+
claimName: string,
|
|
1089
|
+
metadata?: ClaimPodMetadataContext,
|
|
1090
|
+
): KubernetesAcquireError | undefined {
|
|
1091
|
+
const condition = readyCondition(claim?.status?.conditions)
|
|
1092
|
+
if (condition === undefined || condition.status !== 'False') return undefined
|
|
1093
|
+
const reason = condition.reason
|
|
1094
|
+
if (reason === undefined || !TERMINAL_CLAIM_REASONS.includes(reason)) return undefined
|
|
1095
|
+
// Only for the reason the pod metadata can actually cause, and only when
|
|
1096
|
+
// this backend sent any: a `WarmPoolNotFound` carrying a labels-and-
|
|
1097
|
+
// allowlist explanation would send an operator after the wrong thing.
|
|
1098
|
+
const cause =
|
|
1099
|
+
reason === 'InvalidMetadata' &&
|
|
1100
|
+
metadata !== undefined &&
|
|
1101
|
+
Object.keys(metadata.requestedPodLabels).length > 0
|
|
1102
|
+
? new KubernetesPodLabelsRejectedError(
|
|
1103
|
+
metadata.requestedPodLabels,
|
|
1104
|
+
claimName,
|
|
1105
|
+
metadata.namespace,
|
|
1106
|
+
reason,
|
|
1107
|
+
condition.message ?? '(the controller reported no message)',
|
|
1108
|
+
metadata.profile,
|
|
1109
|
+
)
|
|
1110
|
+
: undefined
|
|
1111
|
+
return new KubernetesAcquireError({
|
|
1112
|
+
reason: 'claim-rejected',
|
|
1113
|
+
retryable: false,
|
|
1114
|
+
controllerReason: reason,
|
|
1115
|
+
...(condition.message !== undefined ? { controllerMessage: condition.message } : {}),
|
|
1116
|
+
...(cause !== undefined ? { cause } : {}),
|
|
1117
|
+
message: `kubernetes: the agent-sandbox controller refused SandboxClaim ${claimName} with reason ${reason}${
|
|
1118
|
+
condition.message !== undefined ? `: ${condition.message}` : ''
|
|
1119
|
+
}. That is a decision, not a delay, so the readiness budget was not waited out.${
|
|
1120
|
+
cause !== undefined ? ` ${cause.message}` : ''
|
|
1121
|
+
}`,
|
|
1122
|
+
})
|
|
1123
|
+
}
|
|
1124
|
+
|
|
1125
|
+
/**
|
|
1126
|
+
* How long to wait before repeating a failed readiness read, or `undefined`
|
|
1127
|
+
* when the failure is not worth repeating.
|
|
1128
|
+
*
|
|
1129
|
+
* Retryable: a connect failure (the socket, not the answer), a request the
|
|
1130
|
+
* `apiRequestTimeoutMs` bound gave up on, a 429 (the API server's own
|
|
1131
|
+
* priority-and-fairness queue shedding load) and any 5xx. Not retryable: 401
|
|
1132
|
+
* and 403, which are a decision; 404 and 410, which are an answer; 409, which
|
|
1133
|
+
* a caller resolves by re-reading; and everything this backend threw itself.
|
|
1134
|
+
*
|
|
1135
|
+
* The wait is the poll's own cadence unless the server named one — then its
|
|
1136
|
+
* `Retry-After`, because the server knows when its queue drains and this code
|
|
1137
|
+
* does not. Nothing here consults a clock: the caller's deadline owns the
|
|
1138
|
+
* sleep, so a long `Retry-After` spends the readiness budget rather than
|
|
1139
|
+
* extending it.
|
|
1140
|
+
*/
|
|
1141
|
+
/**
|
|
1142
|
+
* The largest delay a timer can hold — `2^31 - 1` ms, Node's own ceiling.
|
|
1143
|
+
* Above it `setTimeout` warns and fires immediately, which is the opposite of
|
|
1144
|
+
* what a long `Retry-After` asked for.
|
|
1145
|
+
*/
|
|
1146
|
+
const MAX_RETRY_DELAY_MS = 2_147_483_647
|
|
1147
|
+
|
|
1148
|
+
export function retryDelayForApiFailure(err: unknown, pollIntervalMs: number): number | undefined {
|
|
1149
|
+
if (err instanceof KubernetesApiTimeoutError) return pollIntervalMs
|
|
1150
|
+
if (!(err instanceof KubernetesApiError)) return undefined
|
|
1151
|
+
if (err.transport === 'connect') return pollIntervalMs
|
|
1152
|
+
const status = err.status
|
|
1153
|
+
if (status === undefined) return undefined
|
|
1154
|
+
// `Retry-After` may only ever SLOW the poll down, and only within what a
|
|
1155
|
+
// timer can express. A header of `0`, or one naming a moment already past,
|
|
1156
|
+
// would otherwise turn the retry into a hot loop against a server that is
|
|
1157
|
+
// already shedding load — the caller's own cadence is the rate this loop
|
|
1158
|
+
// runs at when nothing is wrong. And a header naming a moment years away
|
|
1159
|
+
// overflows `setTimeout`, which then fires at once rather than never,
|
|
1160
|
+
// producing the same hot loop from the opposite direction. The readiness
|
|
1161
|
+
// deadline ends the wait either way; the clamp only stops the wait from
|
|
1162
|
+
// silently becoming no wait at all.
|
|
1163
|
+
if (status === 429 || status >= 500) {
|
|
1164
|
+
const asked = err.retryAfterMs ?? pollIntervalMs
|
|
1165
|
+
return Math.min(Math.max(asked, pollIntervalMs), MAX_RETRY_DELAY_MS)
|
|
1166
|
+
}
|
|
1167
|
+
return undefined
|
|
1168
|
+
}
|
|
1169
|
+
|
|
1170
|
+
/**
|
|
1171
|
+
* How long the one diagnostic pod read after a failed acquire may take.
|
|
1172
|
+
*
|
|
1173
|
+
* Same shape and the same argument as `runFailureCleanup`'s grace: the
|
|
1174
|
+
* readiness clock has already expired, so this cannot share it, and a
|
|
1175
|
+
* diagnosis that could hang would keep `create()` pending past the budget the
|
|
1176
|
+
* caller chose — for a nicer error message. One second, and a diagnosis that
|
|
1177
|
+
* does not arrive is simply not made.
|
|
1178
|
+
*/
|
|
1179
|
+
const ACQUIRE_DIAGNOSIS_GRACE_MS = 1_000
|
|
1180
|
+
|
|
1181
|
+
/**
|
|
1182
|
+
* The image-pull `status.containerStatuses[].state.waiting.reason` values the
|
|
1183
|
+
* kubelet reports. Measured on kind v1.37.0 (2026-09-17): a container whose
|
|
1184
|
+
* image does not exist waits as `ErrImagePull` for the first attempts and
|
|
1185
|
+
* settles into `ImagePullBackOff`. The other two are the kubelet's names for
|
|
1186
|
+
* a pull that resolved and then failed, and for a registry that cannot be
|
|
1187
|
+
* reached at all.
|
|
1188
|
+
*/
|
|
1189
|
+
const IMAGE_PULL_WAITING_REASONS: readonly string[] = [
|
|
1190
|
+
'ErrImagePull',
|
|
1191
|
+
'ImagePullBackOff',
|
|
1192
|
+
'ImageInspectError',
|
|
1193
|
+
'RegistryUnavailable',
|
|
1194
|
+
]
|
|
1195
|
+
|
|
1196
|
+
/**
|
|
1197
|
+
* Ask the pod why it is not ready, once, after the budget has already gone.
|
|
1198
|
+
*
|
|
1199
|
+
* Nothing on the healthy path calls this and nothing waits on it: it runs
|
|
1200
|
+
* exactly when an acquire has already failed, and its whole output is a
|
|
1201
|
+
* better {@link KubernetesAcquireFailureReason} than `'not-ready'`. A read
|
|
1202
|
+
* that fails, a pod that is not there and a pod with nothing to say all
|
|
1203
|
+
* produce `undefined`, which leaves the reason where it was.
|
|
1204
|
+
*
|
|
1205
|
+
* It must run BEFORE the cleanup DELETE, because the pod goes away with the
|
|
1206
|
+
* object it belongs to.
|
|
1207
|
+
*
|
|
1208
|
+
* It takes the CALLER's signal where `runFailureCleanup` deliberately does
|
|
1209
|
+
* not: cleanup must finish or the cluster keeps the object, while a diagnosis
|
|
1210
|
+
* is only a better sentence for an error a caller who aborted will never
|
|
1211
|
+
* read.
|
|
1212
|
+
*/
|
|
1213
|
+
async function diagnoseUnreadyPod(
|
|
1214
|
+
client: KubernetesClient,
|
|
1215
|
+
namespace: string,
|
|
1216
|
+
podName: string,
|
|
1217
|
+
callerSignal: AbortSignal | undefined,
|
|
1218
|
+
): Promise<'capacity' | 'image-pull' | undefined> {
|
|
1219
|
+
let pod: PodResource | undefined
|
|
1220
|
+
try {
|
|
1221
|
+
const deadline = new OperationDeadline(
|
|
1222
|
+
ACQUIRE_DIAGNOSIS_GRACE_MS,
|
|
1223
|
+
'kubernetes acquire diagnosis',
|
|
1224
|
+
callerSignal,
|
|
1225
|
+
)
|
|
1226
|
+
pod = await deadline.run((signal) =>
|
|
1227
|
+
client.request<PodResource>('GET', podPath(namespace, podName), undefined, signal),
|
|
1228
|
+
)
|
|
1229
|
+
} catch {
|
|
1230
|
+
// The acquire failure is the primary one and keeps its reason. A
|
|
1231
|
+
// diagnosis that cannot be made is not a second failure to report.
|
|
1232
|
+
return undefined
|
|
1233
|
+
}
|
|
1234
|
+
const scheduled = pod?.status?.conditions?.find((c) => c.type === 'PodScheduled')
|
|
1235
|
+
if (scheduled?.status === 'False' && scheduled.reason === 'Unschedulable') return 'capacity'
|
|
1236
|
+
for (const container of pod?.status?.containerStatuses ?? []) {
|
|
1237
|
+
const reason = container.state?.waiting?.reason
|
|
1238
|
+
if (reason !== undefined && IMAGE_PULL_WAITING_REASONS.includes(reason)) return 'image-pull'
|
|
1239
|
+
}
|
|
1240
|
+
return undefined
|
|
1241
|
+
}
|
|
1242
|
+
|
|
1243
|
+
/**
|
|
1244
|
+
* The refusal a caller sees, given the failure that actually happened and
|
|
1245
|
+
* whatever the pod had to say about it.
|
|
1246
|
+
*
|
|
1247
|
+
* Returns `undefined` for a failure none of the seven reasons describes —
|
|
1248
|
+
* see {@link KubernetesAcquireError} for why that is a deliberate hole rather
|
|
1249
|
+
* than a missing case. A {@link KubernetesAcquireError} that arrived from
|
|
1250
|
+
* deeper in (the claim rejection) is returned unchanged: it is already the
|
|
1251
|
+
* diagnosis.
|
|
1252
|
+
*/
|
|
1253
|
+
export function classifyAcquireFailure(
|
|
1254
|
+
err: unknown,
|
|
1255
|
+
podDiagnosis: 'capacity' | 'image-pull' | undefined,
|
|
1256
|
+
): KubernetesAcquireError | undefined {
|
|
1257
|
+
if (err instanceof KubernetesAcquireError) return err
|
|
1258
|
+
if (err instanceof KubernetesApiTimeoutError) {
|
|
1259
|
+
return new KubernetesAcquireError({
|
|
1260
|
+
reason: 'api-timeout',
|
|
1261
|
+
retryable: true,
|
|
1262
|
+
message: `kubernetes: the acquire was refused because the API server did not answer in time — ${err.message}`,
|
|
1263
|
+
cause: err,
|
|
1264
|
+
})
|
|
1265
|
+
}
|
|
1266
|
+
if (err instanceof KubernetesCredentialError) {
|
|
1267
|
+
return new KubernetesAcquireError({
|
|
1268
|
+
reason: 'forbidden',
|
|
1269
|
+
retryable: false,
|
|
1270
|
+
message: `kubernetes: the acquire was refused because the API server rejected this host's credential — ${err.message}. Check the host ServiceAccount's Role against the RBAC section of the Kubernetes sandbox documentation.`,
|
|
1271
|
+
cause: err,
|
|
1272
|
+
})
|
|
1273
|
+
}
|
|
1274
|
+
if (err instanceof KubernetesApiError) {
|
|
1275
|
+
// The same predicate the poll retries on, so "worth trying again" has
|
|
1276
|
+
// one definition and a caller cannot be told a failure is retryable
|
|
1277
|
+
// that the poll would have declined to retry. The interval is
|
|
1278
|
+
// irrelevant here — only whether an answer comes back at all.
|
|
1279
|
+
if (retryDelayForApiFailure(err, 1) === undefined) return undefined
|
|
1280
|
+
return new KubernetesAcquireError({
|
|
1281
|
+
reason: 'api-unreachable',
|
|
1282
|
+
retryable: true,
|
|
1283
|
+
message: `kubernetes: the acquire was refused because the API server could not serve it — ${err.message}`,
|
|
1284
|
+
cause: err,
|
|
1285
|
+
})
|
|
1286
|
+
}
|
|
1287
|
+
if (err instanceof ReadinessPollTimeout) {
|
|
1288
|
+
// ORDER MATTERS, and this is the order: what the cluster SAID beats
|
|
1289
|
+
// what the failures suggest. A pod diagnosis is a condition the API
|
|
1290
|
+
// server published about this pod, read after the budget had already
|
|
1291
|
+
// gone; `err.cause` is at best the failure the poll was still meeting
|
|
1292
|
+
// at that moment. On a saturated cluster both are present at once — a
|
|
1293
|
+
// pod nothing can schedule AND an API server shedding load — and
|
|
1294
|
+
// reporting `api-unreachable` there would hide the very reason this
|
|
1295
|
+
// function exists to produce, and would turn `image-pull`'s
|
|
1296
|
+
// `retryable: false` into a `true` that has a host retrying forever
|
|
1297
|
+
// against an image reference that will never resolve. A diagnosis also
|
|
1298
|
+
// cannot be stale in the way a cause can: it only exists because the
|
|
1299
|
+
// API server answered one more read, moments ago.
|
|
1300
|
+
if (podDiagnosis === 'capacity') {
|
|
1301
|
+
return new KubernetesAcquireError({
|
|
1302
|
+
reason: 'capacity',
|
|
1303
|
+
retryable: true,
|
|
1304
|
+
message: `kubernetes: the acquire was refused because its pod could not be scheduled — the cluster reports PodScheduled=False/Unschedulable. ${err.message}`,
|
|
1305
|
+
cause: err,
|
|
1306
|
+
})
|
|
1307
|
+
}
|
|
1308
|
+
if (podDiagnosis === 'image-pull') {
|
|
1309
|
+
return new KubernetesAcquireError({
|
|
1310
|
+
reason: 'image-pull',
|
|
1311
|
+
retryable: false,
|
|
1312
|
+
message: `kubernetes: the acquire was refused because its pod's container image will not pull. ${err.message}`,
|
|
1313
|
+
cause: err,
|
|
1314
|
+
})
|
|
1315
|
+
}
|
|
1316
|
+
// Nothing measured, so the failures the poll kept meeting decide: a
|
|
1317
|
+
// poll that spent its budget retrying API failures did not fail
|
|
1318
|
+
// because the sandbox was slow; it failed because the control plane
|
|
1319
|
+
// was. `pollForBinding` carries the failure it was STILL meeting onto
|
|
1320
|
+
// the timeout, and the reason follows it rather than the timeout.
|
|
1321
|
+
const underlying = classifyAcquireFailure(err.cause, undefined)
|
|
1322
|
+
if (underlying !== undefined) {
|
|
1323
|
+
return new KubernetesAcquireError({
|
|
1324
|
+
reason: underlying.reason,
|
|
1325
|
+
retryable: underlying.retryable,
|
|
1326
|
+
message: underlying.message,
|
|
1327
|
+
cause: err,
|
|
1328
|
+
})
|
|
1329
|
+
}
|
|
1330
|
+
return new KubernetesAcquireError({
|
|
1331
|
+
reason: 'not-ready',
|
|
1332
|
+
retryable: true,
|
|
1333
|
+
message: err.message,
|
|
1334
|
+
cause: err,
|
|
1335
|
+
})
|
|
1336
|
+
}
|
|
1337
|
+
if (err instanceof OperationDeadlineExpired) {
|
|
1338
|
+
return new KubernetesAcquireError({
|
|
1339
|
+
reason: 'not-ready',
|
|
1340
|
+
retryable: true,
|
|
1341
|
+
message: `kubernetes: the acquire ran out of readiness budget — ${err.message}`,
|
|
1342
|
+
cause: err,
|
|
1343
|
+
})
|
|
1344
|
+
}
|
|
1345
|
+
return undefined
|
|
1346
|
+
}
|
|
1347
|
+
|
|
1348
|
+
/**
|
|
1349
|
+
* Claim or create, wait for Ready, read the bound identity back, resolve the
|
|
1350
|
+
* address and learn the pod's uid — or leave nothing behind trying.
|
|
1351
|
+
*
|
|
1352
|
+
* Exported because the sandbox surface is built on top of this record rather
|
|
1353
|
+
* than beside it: one acquire path, one cleanup path, whatever ends up
|
|
1354
|
+
* wrapping them.
|
|
1355
|
+
*
|
|
1356
|
+
* ## What it refuses with
|
|
1357
|
+
*
|
|
1358
|
+
* Every refusal this function can diagnose arrives as a
|
|
1359
|
+
* {@link KubernetesAcquireError} naming one of seven reasons, with the
|
|
1360
|
+
* original failure as its `cause`. Three things stay outside that:
|
|
1361
|
+
* configuration refused before anything is created
|
|
1362
|
+
* ({@link assertEnforceable}, {@link assertRuntimeClassIsApplicable}), a
|
|
1363
|
+
* caller's own abort, and a failure none of the seven reasons honestly
|
|
1364
|
+
* describes — see {@link KubernetesAcquireError} for why the last one is a
|
|
1365
|
+
* hole on purpose.
|
|
1366
|
+
*
|
|
1367
|
+
* ## What it retries, and what it will not
|
|
1368
|
+
*
|
|
1369
|
+
* A readiness GET that fails transiently — a connect failure, a request the
|
|
1370
|
+
* API bound gave up on, a 429, a 5xx — is repeated INSIDE the readiness
|
|
1371
|
+
* deadline, honouring `Retry-After`. One clock, so a retry spends the budget
|
|
1372
|
+
* rather than extending it, and a `create()` cannot outlive the timeout its
|
|
1373
|
+
* caller chose. The create POST is never retried: it is not idempotent, and a
|
|
1374
|
+
* POST whose answer never arrived may already have committed — which is why
|
|
1375
|
+
* cleanup deletes the client-owned name whatever happened.
|
|
1376
|
+
*/
|
|
1377
|
+
export async function acquireKubernetesSandbox(
|
|
1378
|
+
client: KubernetesClient,
|
|
1379
|
+
config: KubernetesBackendInternalConfig,
|
|
1380
|
+
options: SandboxBackendOptions,
|
|
1381
|
+
readiness: { readonly timeoutMs: number; readonly pollIntervalMs: number },
|
|
1382
|
+
verifyIngress: IngressVerifier | undefined = buildIngressVerifier(client, config),
|
|
1383
|
+
egressBoundary: KubernetesEgressBoundary | undefined = buildEgressBoundary(
|
|
1384
|
+
client,
|
|
1385
|
+
config,
|
|
1386
|
+
config.sandboxTemplateName,
|
|
1387
|
+
),
|
|
1388
|
+
): Promise<KubernetesAcquisition> {
|
|
1389
|
+
options.signal?.throwIfAborted()
|
|
1390
|
+
assertEnforceable(options)
|
|
1391
|
+
assertRuntimeClassIsApplicable(config)
|
|
1392
|
+
|
|
1393
|
+
const namespace = config.namespace
|
|
1394
|
+
const ttlSeconds = config.claimTtlSeconds ?? DEFAULT_CLAIM_TTL_SECONDS
|
|
1395
|
+
const shutdownTime = new Date(Date.now() + ttlSeconds * 1_000).toISOString()
|
|
1396
|
+
// Client-owned name, as on ACI: it lets failure cleanup DELETE the object
|
|
1397
|
+
// even when the create response never arrived. `generateSandboxId` returns
|
|
1398
|
+
// a lowercase UUID, which is already a legal DNS-1123 name suffix.
|
|
1399
|
+
const objectName = `namzu-task-${generateSandboxId()}`
|
|
1400
|
+
const ownedPath =
|
|
1401
|
+
config.warmPoolName !== undefined
|
|
1402
|
+
? claimPath(namespace, objectName)
|
|
1403
|
+
: sandboxPath(namespace, objectName)
|
|
1404
|
+
|
|
1405
|
+
const release = async (signal?: AbortSignal): Promise<void> => {
|
|
1406
|
+
try {
|
|
1407
|
+
await client.request('DELETE', ownedPath, undefined, signal)
|
|
1408
|
+
} catch (err) {
|
|
1409
|
+
// The object is gone, which is the state DELETE was asking for.
|
|
1410
|
+
if (!(err instanceof KubernetesAlreadyGoneError)) throw err
|
|
1411
|
+
}
|
|
1412
|
+
}
|
|
1413
|
+
|
|
1414
|
+
// Where the expiry lives differs by KIND, and only this function knows
|
|
1415
|
+
// which kind it created: a claim keeps it under `spec.lifecycle`, a
|
|
1416
|
+
// directly created Sandbox at the top of `spec` (v1beta1 as served has
|
|
1417
|
+
// not moved it under `lifecycle` yet). Merge-patch semantics (RFC 7386,
|
|
1418
|
+
// the only content type this client's PATCH sends) merge the nested
|
|
1419
|
+
// object, so `shutdownPolicy: Delete` survives every renewal.
|
|
1420
|
+
// The parameter is deliberately NOT named `shutdownTime`: the stamp above
|
|
1421
|
+
// is the one the create body carries, and no renewal ever re-sends it.
|
|
1422
|
+
const renew = async (nextShutdownTime: string, signal?: AbortSignal): Promise<void> => {
|
|
1423
|
+
const patch =
|
|
1424
|
+
config.warmPoolName !== undefined
|
|
1425
|
+
? { spec: { lifecycle: { shutdownTime: nextShutdownTime } } }
|
|
1426
|
+
: { spec: { shutdownTime: nextShutdownTime } }
|
|
1427
|
+
await client.request('PATCH', ownedPath, patch, signal)
|
|
1428
|
+
}
|
|
1429
|
+
|
|
1430
|
+
// One clock over the whole path — the create POST included, so a hung API
|
|
1431
|
+
// server cannot leave `create()` pending past the caller's timeout.
|
|
1432
|
+
const deadline = new OperationDeadline(
|
|
1433
|
+
readiness.timeoutMs,
|
|
1434
|
+
'kubernetes readiness',
|
|
1435
|
+
options.signal,
|
|
1436
|
+
)
|
|
1437
|
+
|
|
1438
|
+
// The pool-less path reads its pod template BEFORE anything is created, so
|
|
1439
|
+
// a missing or malformed template fails with nothing to clean up — hence
|
|
1440
|
+
// this sits outside the cleanup block below. It is read per create rather
|
|
1441
|
+
// than cached: an operator editing the template expects the next sandbox to
|
|
1442
|
+
// use it, and this path is not the sub-second one.
|
|
1443
|
+
const createPath =
|
|
1444
|
+
config.warmPoolName !== undefined
|
|
1445
|
+
? claimCollectionPath(namespace)
|
|
1446
|
+
: sandboxCollectionPath(namespace)
|
|
1447
|
+
// Composed ONCE, here, and handed to whichever body builder runs below:
|
|
1448
|
+
// the claim's `additionalPodMetadata.labels` and a direct Sandbox's pod
|
|
1449
|
+
// template metadata are the same map, and the translated policy's selector
|
|
1450
|
+
// is built from the same resolution. See `composeAdditionalPodLabels`.
|
|
1451
|
+
// Empty — the only case before a profile is configured — means every body
|
|
1452
|
+
// below is byte for byte what it was.
|
|
1453
|
+
// The per-sandbox selector label rides in the SAME map, through the same
|
|
1454
|
+
// composer, as a second key rather than a second construction — the whole
|
|
1455
|
+
// reason `composeAdditionalPodLabels` takes an `extra`. Its value is the
|
|
1456
|
+
// name of the object this acquire is about to create, which is unique per
|
|
1457
|
+
// acquire and known BEFORE the POST: that is what lets the label travel as
|
|
1458
|
+
// claim-time pod metadata (warm-safe, no cold start) instead of as a patch
|
|
1459
|
+
// to a running pod this backend has no verb for.
|
|
1460
|
+
const perSandboxLabelKey = perSandboxEgressLabelKey(config.egress)
|
|
1461
|
+
const podLabels = composeAdditionalPodLabels(
|
|
1462
|
+
config.egress,
|
|
1463
|
+
perSandboxLabelKey !== undefined ? { [perSandboxLabelKey]: objectName } : undefined,
|
|
1464
|
+
)
|
|
1465
|
+
const profile = egressProfileLabel(config.egress)
|
|
1466
|
+
// Read off the create reply, and off the readiness polls if that reply
|
|
1467
|
+
// carried no object — it is the uid a per-sandbox policy's
|
|
1468
|
+
// `ownerReferences` names and the suffix of its name, so the cluster can
|
|
1469
|
+
// garbage-collect the policy with the object this backend owns.
|
|
1470
|
+
let ownerUid: string | undefined
|
|
1471
|
+
const buildCreateBody = async (): Promise<Record<string, unknown>> => {
|
|
1472
|
+
if (config.warmPoolName !== undefined) {
|
|
1473
|
+
return buildClaimBody({
|
|
1474
|
+
namespace,
|
|
1475
|
+
name: objectName,
|
|
1476
|
+
warmPoolName: config.warmPoolName,
|
|
1477
|
+
shutdownTime,
|
|
1478
|
+
...(config.claimLabels !== undefined ? { labels: config.claimLabels } : {}),
|
|
1479
|
+
podLabels,
|
|
1480
|
+
})
|
|
1481
|
+
}
|
|
1482
|
+
const template = await deadline.run((signal) =>
|
|
1483
|
+
readSandboxTemplate(client, namespace, config.sandboxTemplateName, signal),
|
|
1484
|
+
)
|
|
1485
|
+
// A direct Sandbox's pod labels are decided HERE, by the body below,
|
|
1486
|
+
// so both network boundaries are checked against the real labels before
|
|
1487
|
+
// anything is created — a refusal leaves no Sandbox and no PVC behind
|
|
1488
|
+
// rather than one of each to clean up. This is EARLIER than the plan
|
|
1489
|
+
// for the egress union check asked for (it said "after binding", for
|
|
1490
|
+
// the bound pod's labels); a pool-less Sandbox's labels are knowable
|
|
1491
|
+
// before the POST, and refusing with nothing created is strictly
|
|
1492
|
+
// better than refusing with an object to clean up.
|
|
1493
|
+
const directPodLabels = sandboxPodLabels(template, config.sandboxTemplateName, podLabels)
|
|
1494
|
+
const directSubject = `to create Sandbox ${objectName} in namespace ${namespace}`
|
|
1495
|
+
if (verifyIngress !== undefined) {
|
|
1496
|
+
await deadline.run((signal) => verifyIngress(directPodLabels, directSubject, signal))
|
|
1497
|
+
}
|
|
1498
|
+
if (egressBoundary !== undefined) {
|
|
1499
|
+
await deadline.run((signal) =>
|
|
1500
|
+
egressBoundary.verifyUnion(directPodLabels, directSubject, signal),
|
|
1501
|
+
)
|
|
1502
|
+
}
|
|
1503
|
+
return buildSandboxBody({
|
|
1504
|
+
namespace,
|
|
1505
|
+
name: objectName,
|
|
1506
|
+
template,
|
|
1507
|
+
sandboxTemplateName: config.sandboxTemplateName,
|
|
1508
|
+
shutdownTime,
|
|
1509
|
+
podLabels,
|
|
1510
|
+
...(config.runtimeClassName !== undefined
|
|
1511
|
+
? { runtimeClassName: config.runtimeClassName }
|
|
1512
|
+
: {}),
|
|
1513
|
+
})
|
|
1514
|
+
}
|
|
1515
|
+
|
|
1516
|
+
let createBody: Record<string, unknown>
|
|
1517
|
+
try {
|
|
1518
|
+
createBody = await buildCreateBody()
|
|
1519
|
+
} catch (err) {
|
|
1520
|
+
// Nothing exists yet, so there is nothing to clean up and no pod to
|
|
1521
|
+
// ask — but a 403 on the template read is still a `forbidden` acquire,
|
|
1522
|
+
// and a caller should not have to tell that apart by where it
|
|
1523
|
+
// happened.
|
|
1524
|
+
throw classifyAcquireFailure(err, undefined) ?? err
|
|
1525
|
+
}
|
|
1526
|
+
|
|
1527
|
+
// The pod the diagnosis below asks, when there is one. A directly created
|
|
1528
|
+
// Sandbox is backed by a pod of its own name; a CLAIM's pod is not knowable
|
|
1529
|
+
// until the controller has named a sandbox in `status.sandbox`, which it
|
|
1530
|
+
// does before Ready on a cold start and never on a rejected claim.
|
|
1531
|
+
let diagnosablePodName: string | undefined =
|
|
1532
|
+
config.warmPoolName !== undefined ? undefined : objectName
|
|
1533
|
+
// One definition of "worth trying again", shared by the poll that retries
|
|
1534
|
+
// and the classification that reports — see {@link retryDelayForApiFailure}.
|
|
1535
|
+
const pollBehaviour: ReadinessPollBehaviour = {
|
|
1536
|
+
retryDelayFor: (err) => retryDelayForApiFailure(err, readiness.pollIntervalMs),
|
|
1537
|
+
}
|
|
1538
|
+
|
|
1539
|
+
try {
|
|
1540
|
+
// Inside the cleanup block: a POST that fails client-side may still have
|
|
1541
|
+
// committed, so the only safe assumption is that the object exists.
|
|
1542
|
+
const created = await deadline.run((signal) =>
|
|
1543
|
+
client.request<{ readonly metadata?: { readonly uid?: string } }>(
|
|
1544
|
+
'POST',
|
|
1545
|
+
createPath,
|
|
1546
|
+
createBody,
|
|
1547
|
+
signal,
|
|
1548
|
+
),
|
|
1549
|
+
)
|
|
1550
|
+
ownerUid ??= created?.metadata?.uid
|
|
1551
|
+
const binding =
|
|
1552
|
+
config.warmPoolName !== undefined
|
|
1553
|
+
? await pollForBinding(
|
|
1554
|
+
async (signal) => {
|
|
1555
|
+
const claim = await client.request<SandboxClaimResource>(
|
|
1556
|
+
'GET',
|
|
1557
|
+
claimPath(namespace, objectName),
|
|
1558
|
+
undefined,
|
|
1559
|
+
signal,
|
|
1560
|
+
)
|
|
1561
|
+
diagnosablePodName = claim?.status?.sandbox?.name ?? diagnosablePodName
|
|
1562
|
+
ownerUid ??= claim?.metadata?.uid
|
|
1563
|
+
// The pod labels this backend asked the controller for
|
|
1564
|
+
// travel into the read, not because the fail-fast needs
|
|
1565
|
+
// them — `InvalidMetadata` is already one of
|
|
1566
|
+
// `TERMINAL_CLAIM_REASONS`, and that is the ONE
|
|
1567
|
+
// taxonomy a refused claim is reported under — but so
|
|
1568
|
+
// the refusal can carry what was actually sent and how
|
|
1569
|
+
// to change it. See {@link ClaimPodMetadataContext}.
|
|
1570
|
+
return bindingFromClaim(claim, objectName, {
|
|
1571
|
+
namespace,
|
|
1572
|
+
requestedPodLabels: podLabels,
|
|
1573
|
+
...(profile !== undefined ? { profile } : {}),
|
|
1574
|
+
})
|
|
1575
|
+
},
|
|
1576
|
+
deadline,
|
|
1577
|
+
readiness,
|
|
1578
|
+
`claim ${objectName}`,
|
|
1579
|
+
pollBehaviour,
|
|
1580
|
+
)
|
|
1581
|
+
: await pollForBinding(
|
|
1582
|
+
async (signal) => {
|
|
1583
|
+
const sandbox = await client.request<SandboxResource>(
|
|
1584
|
+
'GET',
|
|
1585
|
+
sandboxPath(namespace, objectName),
|
|
1586
|
+
undefined,
|
|
1587
|
+
signal,
|
|
1588
|
+
)
|
|
1589
|
+
ownerUid ??= sandbox?.metadata?.uid
|
|
1590
|
+
return bindingFromSandbox(sandbox)
|
|
1591
|
+
},
|
|
1592
|
+
deadline,
|
|
1593
|
+
readiness,
|
|
1594
|
+
`sandbox ${objectName}`,
|
|
1595
|
+
pollBehaviour,
|
|
1596
|
+
)
|
|
1597
|
+
|
|
1598
|
+
const agentPort = config.agentPort ?? DEFAULT_AGENT_PORT
|
|
1599
|
+
const mode = config.agentAddress ?? 'service'
|
|
1600
|
+
// One read, two facts: the bind token and — under `'pod-ip'` — the
|
|
1601
|
+
// address, off the same pod. See {@link readAddressedPod}, which under
|
|
1602
|
+
// a configured profile also waits for the label the controller
|
|
1603
|
+
// patches onto the bound pod before anything is admitted.
|
|
1604
|
+
const pod = await readAddressedPod(
|
|
1605
|
+
client,
|
|
1606
|
+
namespace,
|
|
1607
|
+
binding,
|
|
1608
|
+
deadline,
|
|
1609
|
+
readiness,
|
|
1610
|
+
mode,
|
|
1611
|
+
podLabels,
|
|
1612
|
+
)
|
|
1613
|
+
// The one refusal this capability must not skip: an unlabelled pod
|
|
1614
|
+
// handed back runs under whatever policy DOES select it while the host
|
|
1615
|
+
// believes it is on a narrower profile. This throws INSIDE the try
|
|
1616
|
+
// block, so the cleanup below releases the claim (or deletes the
|
|
1617
|
+
// Sandbox) exactly as a failed privilege probe does — and the probe
|
|
1618
|
+
// itself, which runs in `admitProbedSandbox` after this function
|
|
1619
|
+
// returns, is therefore never reached with the label unobserved.
|
|
1620
|
+
assertRequestedPodLabelsObserved(
|
|
1621
|
+
pod,
|
|
1622
|
+
podLabels,
|
|
1623
|
+
`Sandbox ${binding.name} in namespace ${namespace}`,
|
|
1624
|
+
)
|
|
1625
|
+
// A CLAIMED sandbox's pod was built from the pool's own template, so
|
|
1626
|
+
// its labels are not knowable until the controller has bound one.
|
|
1627
|
+
// Checking here rather than not at all is the trade: a refusal
|
|
1628
|
+
// releases the claim through the cleanup below, which is the same
|
|
1629
|
+
// path a failed privilege probe takes.
|
|
1630
|
+
if (config.warmPoolName !== undefined) {
|
|
1631
|
+
const boundSubject = `the pod bound to Sandbox ${binding.name} in namespace ${namespace}`
|
|
1632
|
+
if (verifyIngress !== undefined) {
|
|
1633
|
+
await deadline.run((signal) => verifyIngress(pod.labels ?? {}, boundSubject, signal))
|
|
1634
|
+
}
|
|
1635
|
+
// The egress union check asks the same question of the same labels
|
|
1636
|
+
// — which policies select THIS pod — so it runs at the same two
|
|
1637
|
+
// points, and a refusal here releases the claim through the cleanup
|
|
1638
|
+
// below, exactly as a failed privilege probe does.
|
|
1639
|
+
if (egressBoundary !== undefined) {
|
|
1640
|
+
await deadline.run((signal) =>
|
|
1641
|
+
egressBoundary.verifyUnion(pod.labels ?? {}, boundSubject, signal),
|
|
1642
|
+
)
|
|
1643
|
+
}
|
|
1644
|
+
}
|
|
1645
|
+
// Only when the capability is configured, and only after the pod has
|
|
1646
|
+
// been confirmed to carry the selector label: a policy written for a
|
|
1647
|
+
// pod that never got the label would select nothing while the caller
|
|
1648
|
+
// was told its egress had been narrowed.
|
|
1649
|
+
const owner =
|
|
1650
|
+
perSandboxLabelKey === undefined
|
|
1651
|
+
? undefined
|
|
1652
|
+
: {
|
|
1653
|
+
kind: (config.warmPoolName !== undefined ? 'SandboxClaim' : 'Sandbox') as
|
|
1654
|
+
| 'SandboxClaim'
|
|
1655
|
+
| 'Sandbox',
|
|
1656
|
+
name: objectName,
|
|
1657
|
+
uid: assertOwnerUid(ownerUid, objectName, namespace),
|
|
1658
|
+
}
|
|
1659
|
+
return {
|
|
1660
|
+
binding,
|
|
1661
|
+
agent: resolveAgentAddress(binding, agentPort, pod.uid, {
|
|
1662
|
+
mode,
|
|
1663
|
+
...(pod.podIP !== undefined ? { podIP: pod.podIP } : {}),
|
|
1664
|
+
}),
|
|
1665
|
+
// Only the literal-address mode gets one — see the field.
|
|
1666
|
+
...(mode === 'pod-ip'
|
|
1667
|
+
? { refreshAgent: buildAgentAddressRefresh(client, namespace, binding, agentPort, mode) }
|
|
1668
|
+
: {}),
|
|
1669
|
+
ownedPath,
|
|
1670
|
+
...(owner !== undefined ? { owner, perSandboxLabelValue: objectName } : {}),
|
|
1671
|
+
ttlSeconds,
|
|
1672
|
+
release,
|
|
1673
|
+
renew,
|
|
1674
|
+
}
|
|
1675
|
+
} catch (err) {
|
|
1676
|
+
// Asked BEFORE cleanup, because the pod goes away with the object, and
|
|
1677
|
+
// only for a timeout — every other failure already knows what it was.
|
|
1678
|
+
const podDiagnosis =
|
|
1679
|
+
err instanceof ReadinessPollTimeout && diagnosablePodName !== undefined
|
|
1680
|
+
? await diagnoseUnreadyPod(client, namespace, diagnosablePodName, options.signal)
|
|
1681
|
+
: undefined
|
|
1682
|
+
// One cleanup for every way out of the block above, on its own short
|
|
1683
|
+
// budget: the readiness clock has already expired in the common case,
|
|
1684
|
+
// so spending it again would either skip cleanup or leave `create()`
|
|
1685
|
+
// pending without a bound. An object that is already gone is success.
|
|
1686
|
+
await runFailureCleanup(async (signal) => {
|
|
1687
|
+
await release(signal)
|
|
1688
|
+
})
|
|
1689
|
+
throw classifyAcquireFailure(err, podDiagnosis) ?? err
|
|
1690
|
+
}
|
|
1691
|
+
}
|
|
1692
|
+
|
|
1693
|
+
/**
|
|
1694
|
+
* The claim body, in full. Everything absent from it is absent on purpose: no
|
|
1695
|
+
* `env` and no `volumeClaimTemplates`, because either forces a cold start
|
|
1696
|
+
* upstream and takes the warm pool away. `additionalPodMetadata` is the one
|
|
1697
|
+
* piece of claim-time metadata that does NOT — see `podLabels` below.
|
|
1698
|
+
*
|
|
1699
|
+
* `shutdownTime` + `shutdownPolicy: 'Delete'` is the leak guard: it bounds the
|
|
1700
|
+
* object by the wall clock whatever the host does, so a host that dies
|
|
1701
|
+
* mid-acquire costs the cluster one TTL rather than one leaked sandbox
|
|
1702
|
+
* forever. `ttlSecondsAfterFinished` deliberately does NOT appear — its timer
|
|
1703
|
+
* starts from the Finished condition, which a crashed host never reaches.
|
|
1704
|
+
*
|
|
1705
|
+
* `labels` (from {@link KubernetesBackendInternalConfig.claimLabels}) is the
|
|
1706
|
+
* host's own bookkeeping and goes ONLY onto `metadata.labels` — never into
|
|
1707
|
+
* `additionalPodMetadata`, because those are POD labels that change what
|
|
1708
|
+
* selectors match a running sandbox, and a host's crash-recovery identity has
|
|
1709
|
+
* no business doing that. Absent or empty, the body is exactly what it was
|
|
1710
|
+
* before `claimLabels` existed.
|
|
1711
|
+
*
|
|
1712
|
+
* `podLabels` is the other map, on the other object: whatever
|
|
1713
|
+
* `composeAdditionalPodLabels` produced — the egress profile today — which
|
|
1714
|
+
* the controller merges onto the pod it binds. It is the one piece of
|
|
1715
|
+
* claim-time metadata that does NOT cost a cold start, which is why `env` and
|
|
1716
|
+
* `volumeClaimTemplates` are still absent from this body and this is not.
|
|
1717
|
+
* Empty, `additionalPodMetadata` does not appear at all and the body is byte
|
|
1718
|
+
* for byte what it always was.
|
|
1719
|
+
*/
|
|
1720
|
+
interface ClaimBodyOptions {
|
|
1721
|
+
readonly namespace: string
|
|
1722
|
+
readonly name: string
|
|
1723
|
+
readonly warmPoolName: string
|
|
1724
|
+
readonly shutdownTime: string
|
|
1725
|
+
/** `metadata.labels` on the CLAIM. See above. */
|
|
1726
|
+
readonly labels?: Record<string, string>
|
|
1727
|
+
/** `spec.additionalPodMetadata.labels` — the POD's. See above. */
|
|
1728
|
+
readonly podLabels?: Readonly<Record<string, string>>
|
|
1729
|
+
}
|
|
1730
|
+
|
|
1731
|
+
function buildClaimBody(options: ClaimBodyOptions): Record<string, unknown> {
|
|
1732
|
+
const { labels, podLabels } = options
|
|
1733
|
+
return {
|
|
1734
|
+
apiVersion: `${SANDBOX_EXTENSIONS_API_GROUP}/${SANDBOX_API_VERSION}`,
|
|
1735
|
+
kind: 'SandboxClaim',
|
|
1736
|
+
metadata: {
|
|
1737
|
+
name: options.name,
|
|
1738
|
+
namespace: options.namespace,
|
|
1739
|
+
...(labels !== undefined && Object.keys(labels).length > 0 ? { labels } : {}),
|
|
1740
|
+
},
|
|
1741
|
+
spec: {
|
|
1742
|
+
warmPoolRef: { name: options.warmPoolName },
|
|
1743
|
+
...(podLabels !== undefined && Object.keys(podLabels).length > 0
|
|
1744
|
+
? { additionalPodMetadata: { labels: podLabels } }
|
|
1745
|
+
: {}),
|
|
1746
|
+
lifecycle: { shutdownTime: options.shutdownTime, shutdownPolicy: 'Delete' },
|
|
1747
|
+
},
|
|
1748
|
+
}
|
|
1749
|
+
}
|
|
1750
|
+
|
|
1751
|
+
/**
|
|
1752
|
+
* Refuse a bound pod that does not carry EVERY label this backend asked the
|
|
1753
|
+
* controller to put on it — the egress profile today, and whatever else
|
|
1754
|
+
* `composeAdditionalPodLabels` later contributes to the same map.
|
|
1755
|
+
*
|
|
1756
|
+
* The same set {@link readAddressedPod} waits for, deliberately: a wait that
|
|
1757
|
+
* covered more than the refusal would burn the readiness budget on a label
|
|
1758
|
+
* nothing then checked, and a refusal that covered more than the wait would
|
|
1759
|
+
* refuse a pod that had simply not been patched yet. An empty map (the
|
|
1760
|
+
* unprofiled path) checks nothing and returns.
|
|
1761
|
+
*
|
|
1762
|
+
* Called after {@link readAddressedPod} has already waited on the readiness
|
|
1763
|
+
* deadline, so reaching a missing label here means it never arrived, not that
|
|
1764
|
+
* it had not arrived yet.
|
|
1765
|
+
*/
|
|
1766
|
+
function assertRequestedPodLabelsObserved(
|
|
1767
|
+
pod: KubernetesBoundPod,
|
|
1768
|
+
requested: Readonly<Record<string, string>>,
|
|
1769
|
+
subject: string,
|
|
1770
|
+
): void {
|
|
1771
|
+
for (const [key, value] of Object.entries(requested)) {
|
|
1772
|
+
if (pod.labels?.[key] === value) continue
|
|
1773
|
+
throw new KubernetesPodLabelNotObservedError({ key, value }, subject, pod.labels ?? {})
|
|
1774
|
+
}
|
|
1775
|
+
}
|
|
1776
|
+
|
|
1777
|
+
/**
|
|
1778
|
+
* The uid of the object this acquire created, or a refusal naming what was
|
|
1779
|
+
* missing.
|
|
1780
|
+
*
|
|
1781
|
+
* Only reached when `config.egress.perSandbox` is configured, and only after
|
|
1782
|
+
* a create reply and every readiness poll have been read without one — the
|
|
1783
|
+
* API server assigns `metadata.uid` on admission and returns the object it
|
|
1784
|
+
* created, so an object with no uid is an API server that answered something
|
|
1785
|
+
* other than what it was asked for. Refusing is right: without the uid there
|
|
1786
|
+
* is no owner reference, and a per-sandbox policy with no owner is one the
|
|
1787
|
+
* cluster never collects.
|
|
1788
|
+
*/
|
|
1789
|
+
function assertOwnerUid(uid: string | undefined, objectName: string, namespace: string): string {
|
|
1790
|
+
if (uid !== undefined && uid !== '') return uid
|
|
1791
|
+
throw new KubernetesOwnerUidMissingError(objectName, namespace)
|
|
1792
|
+
}
|
|
1793
|
+
|
|
1794
|
+
/**
|
|
1795
|
+
* Options for {@link releaseKubernetesTaskSandboxes}.
|
|
1796
|
+
*/
|
|
1797
|
+
export interface KubernetesReleaseTaskSandboxesOptions {
|
|
1798
|
+
/**
|
|
1799
|
+
* Required, and refused if empty — see {@link releaseKubernetesTaskSandboxes}.
|
|
1800
|
+
* The same selector syntax a `kubectl get --selector` takes, e.g.
|
|
1801
|
+
* `sandbox.namzu.ai/host-instance=host-a`.
|
|
1802
|
+
*/
|
|
1803
|
+
readonly labelSelector: string
|
|
1804
|
+
readonly signal?: AbortSignal
|
|
1805
|
+
}
|
|
1806
|
+
|
|
1807
|
+
/**
|
|
1808
|
+
* Recover a crashed host's claims: LIST every `SandboxClaim` carrying
|
|
1809
|
+
* `labelSelector`, `DELETE` each, and report what was removed.
|
|
1810
|
+
*
|
|
1811
|
+
* This deletes CLAIMS only. The controller's own garbage collection —
|
|
1812
|
+
* ownerReferences from claim to the Sandbox it bound, and from Sandbox to
|
|
1813
|
+
* Pod and Service — takes the rest down behind it; nothing here reads or
|
|
1814
|
+
* touches a Sandbox or a Pod directly. A claim already gone (raced by the
|
|
1815
|
+
* controller's own TTL reaper, or a second release call) counts as removed
|
|
1816
|
+
* rather than a failure, the same convention every other DELETE in this
|
|
1817
|
+
* backend follows.
|
|
1818
|
+
*
|
|
1819
|
+
* `labelSelector` is REQUIRED and refused, synchronously, before a single
|
|
1820
|
+
* request goes out, if it is absent or empty: a release that could fall back
|
|
1821
|
+
* to matching every claim (or every claim of the template) would delete a
|
|
1822
|
+
* live fleet's work the first time a caller passed one by mistake. There is
|
|
1823
|
+
* no default selector for exactly this reason.
|
|
1824
|
+
*/
|
|
1825
|
+
export async function releaseKubernetesTaskSandboxes(
|
|
1826
|
+
config: KubernetesBackendInternalConfig,
|
|
1827
|
+
options: KubernetesReleaseTaskSandboxesOptions,
|
|
1828
|
+
): Promise<{ readonly deleted: number; readonly names: readonly string[] }> {
|
|
1829
|
+
if (options.labelSelector === '') {
|
|
1830
|
+
throw new Error(
|
|
1831
|
+
'kubernetes: releaseKubernetesTaskSandboxes requires a non-empty labelSelector — a release with no selector would delete every SandboxClaim in the namespace, including ones a live host still owns. Pass the selector that names only the claims you mean to recover.',
|
|
1832
|
+
)
|
|
1833
|
+
}
|
|
1834
|
+
options.signal?.throwIfAborted()
|
|
1835
|
+
const namespace = config.namespace
|
|
1836
|
+
const client = createKubernetesClient(clientAccess(config), clientOptions(config))
|
|
1837
|
+
const list = await client.request<SandboxClaimListResource>(
|
|
1838
|
+
'GET',
|
|
1839
|
+
claimListPath(namespace, options.labelSelector),
|
|
1840
|
+
undefined,
|
|
1841
|
+
options.signal,
|
|
1842
|
+
)
|
|
1843
|
+
const names = (list?.items ?? [])
|
|
1844
|
+
.map((claim) => claim.metadata?.name)
|
|
1845
|
+
.filter((name): name is string => typeof name === 'string' && name !== '')
|
|
1846
|
+
// Concurrent, not one at a time: a crash-recovery release can carry a
|
|
1847
|
+
// whole host's worth of claims, and nothing here needs the ordering a
|
|
1848
|
+
// sequential loop would impose — each DELETE is independent and
|
|
1849
|
+
// idempotent (an already-gone claim is tolerated below). A non-tolerated
|
|
1850
|
+
// failure still rejects the whole call, exactly as a sequential loop
|
|
1851
|
+
// would have on its first such failure.
|
|
1852
|
+
await Promise.all(
|
|
1853
|
+
names.map(async (name) => {
|
|
1854
|
+
try {
|
|
1855
|
+
await client.request('DELETE', claimPath(namespace, name), undefined, options.signal)
|
|
1856
|
+
} catch (err) {
|
|
1857
|
+
if (!(err instanceof KubernetesAlreadyGoneError)) throw err
|
|
1858
|
+
}
|
|
1859
|
+
}),
|
|
1860
|
+
)
|
|
1861
|
+
return { deleted: names.length, names }
|
|
1862
|
+
}
|
|
1863
|
+
|
|
1864
|
+
/** Result of {@link readKubernetesTaskCapacity}. */
|
|
1865
|
+
export interface KubernetesTaskCapacity {
|
|
1866
|
+
/** `config.warmPoolName`'s own replica counts, straight off its `status`/`spec`. */
|
|
1867
|
+
readonly warmPool: {
|
|
1868
|
+
/** `status.readyReplicas`. `0` when the field is absent (a brand-new or empty pool). */
|
|
1869
|
+
readonly ready: number
|
|
1870
|
+
/** `spec.replicas`. `0` when the field is absent. */
|
|
1871
|
+
readonly desired: number
|
|
1872
|
+
}
|
|
1873
|
+
/** Every `SandboxClaim` in the namespace bound to `config.warmPoolName`, whatever its state. */
|
|
1874
|
+
readonly activeClaims: number
|
|
1875
|
+
/** Every Pod in the namespace currently in phase `Pending`. */
|
|
1876
|
+
readonly pendingPods: number
|
|
1877
|
+
}
|
|
1878
|
+
|
|
1879
|
+
/** Options for {@link readKubernetesTaskCapacity}. */
|
|
1880
|
+
export interface KubernetesReadTaskCapacityOptions {
|
|
1881
|
+
readonly signal?: AbortSignal
|
|
1882
|
+
}
|
|
1883
|
+
|
|
1884
|
+
/**
|
|
1885
|
+
* Read task-pool headroom before admitting more work: three GETs, no writes.
|
|
1886
|
+
*
|
|
1887
|
+
* `config.warmPoolName` is required — this reads the exact object a claim's
|
|
1888
|
+
* `warmPoolRef` names, so a pool-less backend (every create is a direct
|
|
1889
|
+
* Sandbox) has no pool to report on. `activeClaims` is every claim in the
|
|
1890
|
+
* namespace whose `spec.warmPoolRef.name` matches this pool, counted rather
|
|
1891
|
+
* than trusted from a label, because a claim's `warmPoolRef` is the one
|
|
1892
|
+
* field the API itself guarantees. `pendingPods` is every Pod in the
|
|
1893
|
+
* namespace still in phase `Pending` — a coarse but honest signal of
|
|
1894
|
+
* in-flight scale-up the ready-replica count alone does not carry, on
|
|
1895
|
+
* either the warm or the pool-less path.
|
|
1896
|
+
*/
|
|
1897
|
+
export async function readKubernetesTaskCapacity(
|
|
1898
|
+
config: KubernetesBackendInternalConfig,
|
|
1899
|
+
options?: KubernetesReadTaskCapacityOptions,
|
|
1900
|
+
): Promise<KubernetesTaskCapacity> {
|
|
1901
|
+
options?.signal?.throwIfAborted()
|
|
1902
|
+
if (config.warmPoolName === undefined) {
|
|
1903
|
+
throw new Error(
|
|
1904
|
+
'kubernetes: readKubernetesTaskCapacity requires config.warmPoolName — there is no SandboxWarmPool to report on for a backend that creates every sandbox directly.',
|
|
1905
|
+
)
|
|
1906
|
+
}
|
|
1907
|
+
const namespace = config.namespace
|
|
1908
|
+
const warmPoolName = config.warmPoolName
|
|
1909
|
+
const client = createKubernetesClient(clientAccess(config), clientOptions(config))
|
|
1910
|
+
const [pool, claims, pods] = await Promise.all([
|
|
1911
|
+
client.request<SandboxWarmPoolResource>(
|
|
1912
|
+
'GET',
|
|
1913
|
+
warmPoolPath(namespace, warmPoolName),
|
|
1914
|
+
undefined,
|
|
1915
|
+
options?.signal,
|
|
1916
|
+
),
|
|
1917
|
+
client.request<SandboxClaimListResource>(
|
|
1918
|
+
'GET',
|
|
1919
|
+
claimCollectionPath(namespace),
|
|
1920
|
+
undefined,
|
|
1921
|
+
options?.signal,
|
|
1922
|
+
),
|
|
1923
|
+
client.request<PodListResource>(
|
|
1924
|
+
'GET',
|
|
1925
|
+
podCollectionPath(namespace),
|
|
1926
|
+
undefined,
|
|
1927
|
+
options?.signal,
|
|
1928
|
+
),
|
|
1929
|
+
])
|
|
1930
|
+
const activeClaims = (claims?.items ?? []).filter(
|
|
1931
|
+
(claim) => claim.spec?.warmPoolRef?.name === warmPoolName,
|
|
1932
|
+
).length
|
|
1933
|
+
const pendingPods = (pods?.items ?? []).filter((pod) => pod.status?.phase === 'Pending').length
|
|
1934
|
+
return {
|
|
1935
|
+
warmPool: {
|
|
1936
|
+
ready: pool?.status?.readyReplicas ?? 0,
|
|
1937
|
+
desired: pool?.spec?.replicas ?? 0,
|
|
1938
|
+
},
|
|
1939
|
+
activeClaims,
|
|
1940
|
+
pendingPods,
|
|
1941
|
+
}
|
|
1942
|
+
}
|
|
1943
|
+
|
|
1944
|
+
/**
|
|
1945
|
+
* What a directly created Sandbox copies out of a `SandboxTemplate`, and the
|
|
1946
|
+
* two things it decides for itself.
|
|
1947
|
+
*/
|
|
1948
|
+
export interface SandboxBodyOptions {
|
|
1949
|
+
readonly namespace: string
|
|
1950
|
+
readonly name: string
|
|
1951
|
+
readonly template: SandboxTemplateCopy
|
|
1952
|
+
/** The template the copy came from — the value of {@link sandboxTemplateLabel}. */
|
|
1953
|
+
readonly sandboxTemplateName: string
|
|
1954
|
+
readonly runtimeClassName?: string
|
|
1955
|
+
/**
|
|
1956
|
+
* RFC 3339 expiry, paired with `shutdownPolicy: Delete`. ABSENT means the
|
|
1957
|
+
* object carries no expiry at all and nothing reaps it on the wall clock:
|
|
1958
|
+
* that is the persistent workspace (`workspace.ts`), which is explicitly
|
|
1959
|
+
* managed and must survive a host that stops renewing. Every task sandbox
|
|
1960
|
+
* sets it, because an unbounded task sandbox is a leak.
|
|
1961
|
+
*/
|
|
1962
|
+
readonly shutdownTime?: string
|
|
1963
|
+
/**
|
|
1964
|
+
* Annotations to stamp on the Sandbox's OWN metadata at creation.
|
|
1965
|
+
*
|
|
1966
|
+
* One caller, and everything it writes is a fact the object has to carry
|
|
1967
|
+
* from the moment it exists rather than from its first patch: the holder
|
|
1968
|
+
* epoch of a workspace created under one, so there is no window in which
|
|
1969
|
+
* it stands unfenced, and the revision of the pod template it was built
|
|
1970
|
+
* from, so there is no window in which it claims none. Both are
|
|
1971
|
+
* `workspace.ts`'s — see `HOLDER_EPOCH_ANNOTATION_KEY` and
|
|
1972
|
+
* `POD_TEMPLATE_HASH_ANNOTATION_KEY`.
|
|
1973
|
+
*
|
|
1974
|
+
* Absent, the body is byte for byte what it always was, which is what
|
|
1975
|
+
* keeps every task sandbox's create unchanged — a task sandbox is
|
|
1976
|
+
* ephemeral, so it has no revision to drift from and nothing to fence.
|
|
1977
|
+
*/
|
|
1978
|
+
readonly annotations?: Readonly<Record<string, string>>
|
|
1979
|
+
/**
|
|
1980
|
+
* Extra labels stamped onto the POD template's metadata, beside the
|
|
1981
|
+
* template label this body always adds.
|
|
1982
|
+
*
|
|
1983
|
+
* The same map `buildClaimBody` puts on a claim's
|
|
1984
|
+
* `additionalPodMetadata.labels`, from the same
|
|
1985
|
+
* `composeAdditionalPodLabels` call — a direct Sandbox has no controller
|
|
1986
|
+
* to merge them for it, so the create body stamps them itself and the two
|
|
1987
|
+
* paths produce one set of pod labels. Absent or empty, the pod template
|
|
1988
|
+
* is byte for byte what it was.
|
|
1989
|
+
*/
|
|
1990
|
+
readonly podLabels?: Readonly<Record<string, string>>
|
|
1991
|
+
}
|
|
1992
|
+
|
|
1993
|
+
/**
|
|
1994
|
+
* The pool-less body. `Sandbox.spec` has no `templateRef` — only a
|
|
1995
|
+
* SandboxWarmPool consumes a SandboxTemplate — so the template's podTemplate
|
|
1996
|
+
* is copied in here by the client.
|
|
1997
|
+
*
|
|
1998
|
+
* `service: true` is forced rather than inherited: a Sandbox without a Service
|
|
1999
|
+
* has no `status.serviceFQDN`, and then the only address left is a pod IP that
|
|
2000
|
+
* changes on every resume.
|
|
2001
|
+
*
|
|
2002
|
+
* `volumeClaimTemplates` is copied VERBATIM when the template declares any.
|
|
2003
|
+
* Dropping it would be silent: the Sandbox would come up healthy with no disk,
|
|
2004
|
+
* the container's `volumeDevices`/`volumeMounts` entry would fail to resolve
|
|
2005
|
+
* (or, worse, resolve to an empty emptyDir on some paths), and the only
|
|
2006
|
+
* symptom of a workspace that lost its disk would be that yesterday's files
|
|
2007
|
+
* are gone. The controller wires the mount by the entry's own NAME,
|
|
2008
|
+
* StatefulSet style, so the copy needs no matching `volumes:` entry and this
|
|
2009
|
+
* function adds none.
|
|
2010
|
+
*
|
|
2011
|
+
* The podTemplate's metadata gains {@link sandboxTemplateLabel}: this Sandbox
|
|
2012
|
+
* is created DIRECTLY, never adopted out of a pool, so it never gets
|
|
2013
|
+
* agent-sandbox's own controller-owned
|
|
2014
|
+
* `agents.x-k8s.io/sandbox-template-ref-hash` label (that is written only on
|
|
2015
|
+
* bind). Without a label of its own a direct Sandbox's pod would carry
|
|
2016
|
+
* nothing `egress-policy.ts`'s translated `NetworkPolicy` could select it
|
|
2017
|
+
* by. Existing labels on the copied template are preserved — this ADDS to
|
|
2018
|
+
* them rather than replacing the object outright — but this backend's own
|
|
2019
|
+
* key always wins if the template happened to set it too, since this is the
|
|
2020
|
+
* label the translated policy is built to match.
|
|
2021
|
+
*/
|
|
2022
|
+
export function buildSandboxBody(options: SandboxBodyOptions): Record<string, unknown> {
|
|
2023
|
+
return {
|
|
2024
|
+
apiVersion: `${SANDBOX_API_GROUP}/${SANDBOX_API_VERSION}`,
|
|
2025
|
+
kind: 'Sandbox',
|
|
2026
|
+
metadata: {
|
|
2027
|
+
name: options.name,
|
|
2028
|
+
namespace: options.namespace,
|
|
2029
|
+
...(options.annotations !== undefined ? { annotations: options.annotations } : {}),
|
|
2030
|
+
},
|
|
2031
|
+
spec: {
|
|
2032
|
+
operatingMode: 'Running',
|
|
2033
|
+
service: true,
|
|
2034
|
+
...(options.shutdownTime !== undefined
|
|
2035
|
+
? { shutdownTime: options.shutdownTime, shutdownPolicy: 'Delete' }
|
|
2036
|
+
: {}),
|
|
2037
|
+
...(options.template.volumeClaimTemplates !== undefined
|
|
2038
|
+
? { volumeClaimTemplates: options.template.volumeClaimTemplates }
|
|
2039
|
+
: {}),
|
|
2040
|
+
podTemplate: sandboxPodTemplate(
|
|
2041
|
+
options.template,
|
|
2042
|
+
options.sandboxTemplateName,
|
|
2043
|
+
options.runtimeClassName,
|
|
2044
|
+
options.podLabels,
|
|
2045
|
+
),
|
|
2046
|
+
},
|
|
2047
|
+
}
|
|
2048
|
+
}
|
|
2049
|
+
|
|
2050
|
+
/**
|
|
2051
|
+
* The `spec.podTemplate` a directly created Sandbox carries: the template's,
|
|
2052
|
+
* with this backend's overlays — the template label and whatever
|
|
2053
|
+
* {@link composeAdditionalPodLabels} produced ({@link sandboxPodLabels}), and
|
|
2054
|
+
* the configured `runtimeClassName`.
|
|
2055
|
+
*
|
|
2056
|
+
* Its own function because it is now built twice: once into the create POST
|
|
2057
|
+
* by {@link buildSandboxBody}, and once into the JSON Patch that refreshes a
|
|
2058
|
+
* standing workspace's pod template (`workspace.ts`). Two expressions of the
|
|
2059
|
+
* same overlay would drift, and the one that drifted would report a workspace
|
|
2060
|
+
* as off-template forever — the hash under
|
|
2061
|
+
* `sandbox.namzu.ai/pod-template-hash` is taken over exactly this object, so
|
|
2062
|
+
* a second spelling is a second revision.
|
|
2063
|
+
*
|
|
2064
|
+
* `podLabels` is on this signature rather than only on the create body for
|
|
2065
|
+
* exactly that reason. A refresh rewrites `/spec/podTemplate` WHOLE, so a
|
|
2066
|
+
* refresh built without them would PATCH the egress profile off a pod
|
|
2067
|
+
* template that carries it — the replacement pod would come up selected by no
|
|
2068
|
+
* per-profile policy, on a path where nothing re-checks the label, and the
|
|
2069
|
+
* revision stamped beside it would be taken over a template the POST never
|
|
2070
|
+
* writes, so `templateCurrent` would report drift forever.
|
|
2071
|
+
*/
|
|
2072
|
+
export function sandboxPodTemplate(
|
|
2073
|
+
template: SandboxTemplateCopy,
|
|
2074
|
+
sandboxTemplateName: string,
|
|
2075
|
+
runtimeClassName?: string,
|
|
2076
|
+
podLabels?: Readonly<Record<string, string>>,
|
|
2077
|
+
): SandboxPodTemplate {
|
|
2078
|
+
const podTemplate = template.podTemplate
|
|
2079
|
+
const spec =
|
|
2080
|
+
runtimeClassName !== undefined
|
|
2081
|
+
? { ...podTemplate.spec, runtimeClassName }
|
|
2082
|
+
: { ...podTemplate.spec }
|
|
2083
|
+
return {
|
|
2084
|
+
...podTemplate,
|
|
2085
|
+
metadata: {
|
|
2086
|
+
...podTemplate.metadata,
|
|
2087
|
+
labels: sandboxPodLabels(template, sandboxTemplateName, podLabels),
|
|
2088
|
+
},
|
|
2089
|
+
spec,
|
|
2090
|
+
}
|
|
2091
|
+
}
|
|
2092
|
+
|
|
2093
|
+
/**
|
|
2094
|
+
* The labels a directly created Sandbox's pod will carry: whatever the
|
|
2095
|
+
* template declares, plus this backend's own template label, which always
|
|
2096
|
+
* wins because it is the label a policy selector is built to match.
|
|
2097
|
+
*
|
|
2098
|
+
* Its own function because the ingress check has to reason about EXACTLY the
|
|
2099
|
+
* labels {@link buildSandboxBody} stamps, before the POST that stamps them.
|
|
2100
|
+
* Two expressions of the same rule would be one rename away from a check that
|
|
2101
|
+
* verifies a pod nobody creates.
|
|
2102
|
+
*
|
|
2103
|
+
* `extra` is `composeAdditionalPodLabels`'s map — the egress profile today.
|
|
2104
|
+
* It is applied LAST, and so wins over both, for the same reason the template
|
|
2105
|
+
* label wins over the copied template's own: it is a label the translated
|
|
2106
|
+
* policy's selector is built to match, and a pod that matched the selector
|
|
2107
|
+
* only sometimes would be a boundary that applied only sometimes.
|
|
2108
|
+
*/
|
|
2109
|
+
export function sandboxPodLabels(
|
|
2110
|
+
template: SandboxTemplateCopy,
|
|
2111
|
+
sandboxTemplateName: string,
|
|
2112
|
+
extra?: Readonly<Record<string, string>>,
|
|
2113
|
+
): Readonly<Record<string, string>> {
|
|
2114
|
+
return {
|
|
2115
|
+
...template.podTemplate.metadata?.labels,
|
|
2116
|
+
...sandboxTemplateLabel(sandboxTemplateName),
|
|
2117
|
+
...extra,
|
|
2118
|
+
}
|
|
2119
|
+
}
|
|
2120
|
+
|
|
2121
|
+
/** The two halves of a `SandboxTemplate` a directly created Sandbox copies. */
|
|
2122
|
+
export interface SandboxTemplateCopy {
|
|
2123
|
+
readonly podTemplate: SandboxPodTemplate
|
|
2124
|
+
/** Absent when the template declares no disk, which is every task template. */
|
|
2125
|
+
readonly volumeClaimTemplates?: readonly SandboxVolumeClaimTemplate[]
|
|
2126
|
+
}
|
|
2127
|
+
|
|
2128
|
+
export async function readSandboxTemplate(
|
|
2129
|
+
client: KubernetesClient,
|
|
2130
|
+
namespace: string,
|
|
2131
|
+
sandboxTemplateName: string,
|
|
2132
|
+
signal?: AbortSignal,
|
|
2133
|
+
): Promise<SandboxTemplateCopy> {
|
|
2134
|
+
const template = await client.request<SandboxTemplateResource>(
|
|
2135
|
+
'GET',
|
|
2136
|
+
sandboxTemplatePath(namespace, sandboxTemplateName),
|
|
2137
|
+
undefined,
|
|
2138
|
+
signal,
|
|
2139
|
+
)
|
|
2140
|
+
const podTemplate = template?.spec?.podTemplate
|
|
2141
|
+
if (!podTemplate || typeof podTemplate.spec !== 'object' || podTemplate.spec === null) {
|
|
2142
|
+
throw new Error(
|
|
2143
|
+
`kubernetes: SandboxTemplate ${sandboxTemplateName} in namespace ${namespace} carries no spec.podTemplate.spec, so there is nothing to create a pool-less Sandbox from. Sandbox.spec has no templateRef — the podTemplate has to be copied in.`,
|
|
2144
|
+
)
|
|
2145
|
+
}
|
|
2146
|
+
const volumeClaimTemplates = template?.spec?.volumeClaimTemplates
|
|
2147
|
+
return {
|
|
2148
|
+
podTemplate,
|
|
2149
|
+
...(volumeClaimTemplates !== undefined ? { volumeClaimTemplates } : {}),
|
|
2150
|
+
}
|
|
2151
|
+
}
|
|
2152
|
+
|
|
2153
|
+
/**
|
|
2154
|
+
* Where the agent answers.
|
|
2155
|
+
*
|
|
2156
|
+
* Its own function, and under the default mode the Service FQDN wins over a
|
|
2157
|
+
* pod IP, because that address outlives the pod: a suspended-then-resumed
|
|
2158
|
+
* workspace comes back as a new pod with a new IP behind the same name, and
|
|
2159
|
+
* the transport re-resolves the name on every dial. A literal IP baked into a
|
|
2160
|
+
* long-lived handle is the bug that would produce — and it is exactly the bug
|
|
2161
|
+
* `'pod-ip'` accepts, deliberately, in exchange for an address a host outside
|
|
2162
|
+
* the cluster can resolve at all. That mode pays for it by re-reading the IP
|
|
2163
|
+
* on every resume and once after a failed connect.
|
|
2164
|
+
*
|
|
2165
|
+
* `'pod-ip'` takes the address from `pod`, the record the bind token was just
|
|
2166
|
+
* read out of, and NEVER falls back to `binding.podIPs`. The Sandbox's status
|
|
2167
|
+
* is a second source that can name a different pod — the one a resume is
|
|
2168
|
+
* replacing — and an address from one pod with a token from another is the
|
|
2169
|
+
* mismatch that arrives as a flat `unauthorized`.
|
|
2170
|
+
*/
|
|
2171
|
+
export function resolveAgentAddress(
|
|
2172
|
+
binding: KubernetesSandboxBinding,
|
|
2173
|
+
agentPort: number,
|
|
2174
|
+
token: string,
|
|
2175
|
+
options: {
|
|
2176
|
+
readonly mode?: KubernetesAgentAddressMode
|
|
2177
|
+
/** The live pod the token came from. Required by `'pod-ip'`. */
|
|
2178
|
+
readonly podIP?: string
|
|
2179
|
+
} = {},
|
|
2180
|
+
): KubernetesAgentAddress {
|
|
2181
|
+
if ((options.mode ?? 'service') === 'pod-ip') {
|
|
2182
|
+
const podIP = options.podIP
|
|
2183
|
+
if (podIP === undefined || podIP === '') {
|
|
2184
|
+
throw new Error(
|
|
2185
|
+
`kubernetes: sandbox ${binding.name} is configured with agentAddress: 'pod-ip', but the live pod its bind token was read from reported no status.podIP (nor a status.podIPs entry), so there is no address to dial. Every path that binds a pod POLLS for that address until the readiness deadline before this is reached — see readAddressedPod — so a pod that still reports none was never given one: a CNI that did not attach it, or a pod that never got past scheduling. Nothing is taken from the Sandbox's own status here on purpose — that IP may belong to a different pod than the token does.`,
|
|
2186
|
+
)
|
|
2187
|
+
}
|
|
2188
|
+
return { kind: 'tcp', host: podIP, port: agentPort, token }
|
|
2189
|
+
}
|
|
2190
|
+
const host = binding.serviceFQDN ?? binding.podIPs?.[0]
|
|
2191
|
+
if (host === undefined || host === '') {
|
|
2192
|
+
throw new Error(
|
|
2193
|
+
`kubernetes: sandbox ${binding.name} reported Ready with neither a serviceFQDN nor a pod IP, so its agent has no address to dial. Set 'service: true' on the SandboxTemplate the pool is built from.`,
|
|
2194
|
+
)
|
|
2195
|
+
}
|
|
2196
|
+
return { kind: 'tcp', host, port: agentPort, token }
|
|
2197
|
+
}
|
|
2198
|
+
|
|
2199
|
+
/**
|
|
2200
|
+
* Ready-or-not-yet, read off a `SandboxClaim`'s own status — plus the one
|
|
2201
|
+
* "not yet" that is really a "no".
|
|
2202
|
+
*
|
|
2203
|
+
* The rejection check runs BEFORE the readiness check rather than after,
|
|
2204
|
+
* because a refused claim is `Ready=False` forever and the readiness check
|
|
2205
|
+
* cannot tell that from a cold start in progress. See
|
|
2206
|
+
* {@link classifyClaimRejection}.
|
|
2207
|
+
*/
|
|
2208
|
+
function bindingFromClaim(
|
|
2209
|
+
claim: SandboxClaimResource | undefined,
|
|
2210
|
+
claimName: string,
|
|
2211
|
+
metadata?: ClaimPodMetadataContext,
|
|
2212
|
+
): KubernetesSandboxBinding | undefined {
|
|
2213
|
+
const rejection = classifyClaimRejection(claim, claimName, metadata)
|
|
2214
|
+
if (rejection !== undefined) throw rejection
|
|
2215
|
+
if (!isConditionTrue(claim?.status?.conditions, READY_CONDITION)) return undefined
|
|
2216
|
+
const bound = claim?.status?.sandbox
|
|
2217
|
+
// Ready with no bound name is the controller contradicting itself; polling
|
|
2218
|
+
// on would just burn the deadline waiting for a field that is finished.
|
|
2219
|
+
if (!bound?.name) {
|
|
2220
|
+
throw new Error(
|
|
2221
|
+
`kubernetes: SandboxClaim ${claimName} reported Ready but named no sandbox in status.sandbox.name, so there is nothing to address.`,
|
|
2222
|
+
)
|
|
2223
|
+
}
|
|
2224
|
+
return {
|
|
2225
|
+
name: bound.name,
|
|
2226
|
+
...(bound.podIPs !== undefined ? { podIPs: bound.podIPs } : {}),
|
|
2227
|
+
...(bound.serviceFQDN !== undefined ? { serviceFQDN: bound.serviceFQDN } : {}),
|
|
2228
|
+
}
|
|
2229
|
+
}
|
|
2230
|
+
|
|
2231
|
+
/** Ready-or-not-yet, read off a `Sandbox`'s own status. Shared with `workspace.ts`. */
|
|
2232
|
+
export function bindingFromSandbox(
|
|
2233
|
+
sandbox: SandboxResource | undefined,
|
|
2234
|
+
): KubernetesSandboxBinding | undefined {
|
|
2235
|
+
if (!isConditionTrue(sandbox?.status?.conditions, READY_CONDITION)) return undefined
|
|
2236
|
+
const name = sandbox?.metadata?.name
|
|
2237
|
+
if (!name) {
|
|
2238
|
+
throw new Error('kubernetes: Sandbox reported Ready with no metadata.name')
|
|
2239
|
+
}
|
|
2240
|
+
const status = sandbox?.status
|
|
2241
|
+
return {
|
|
2242
|
+
name,
|
|
2243
|
+
...(status?.podIPs !== undefined ? { podIPs: status.podIPs } : {}),
|
|
2244
|
+
...(status?.serviceFQDN !== undefined ? { serviceFQDN: status.serviceFQDN } : {}),
|
|
2245
|
+
...(status?.selector !== undefined ? { podSelector: status.selector } : {}),
|
|
2246
|
+
}
|
|
2247
|
+
}
|
|
2248
|
+
|
|
2249
|
+
/**
|
|
2250
|
+
* The readiness poll ran out of budget — and nothing else. Every OTHER
|
|
2251
|
+
* failure {@link pollForBinding} meets is rethrown as itself, so this class
|
|
2252
|
+
* is an exact answer to "was it the clock?", which a caller that has to
|
|
2253
|
+
* choose between two timeout messages needs and cannot get from the clock.
|
|
2254
|
+
*
|
|
2255
|
+
* Reading `remainingMs()` after the fact is NOT that answer: the expiry timer
|
|
2256
|
+
* and `performance.now()` are different clocks, and a timer that fires a
|
|
2257
|
+
* fraction of a millisecond early leaves a positive remainder behind an
|
|
2258
|
+
* expiry that has already happened.
|
|
2259
|
+
*/
|
|
2260
|
+
export class ReadinessPollTimeout extends Error {
|
|
2261
|
+
override readonly name = 'ReadinessPollTimeout'
|
|
2262
|
+
}
|
|
2263
|
+
|
|
2264
|
+
/**
|
|
2265
|
+
* What a failed readiness read is worth, decided by the caller.
|
|
2266
|
+
*
|
|
2267
|
+
* Optional, and absent means exactly the behaviour every caller had before:
|
|
2268
|
+
* the first failure of any kind ends the poll. `workspace.ts` passes nothing
|
|
2269
|
+
* and is unchanged; the acquire path passes
|
|
2270
|
+
* {@link retryDelayForApiFailure} so one 429 on a shared cluster no longer
|
|
2271
|
+
* fails a create that had fifty-nine seconds of budget left.
|
|
2272
|
+
*/
|
|
2273
|
+
export interface ReadinessPollBehaviour {
|
|
2274
|
+
/**
|
|
2275
|
+
* Milliseconds to wait before reading again, or `undefined` to rethrow.
|
|
2276
|
+
*
|
|
2277
|
+
* The wait is spent on the SAME deadline as everything else in the poll,
|
|
2278
|
+
* so a retry consumes the readiness budget and can never extend it. A hook
|
|
2279
|
+
* that always returns a number therefore still terminates: the clock ends
|
|
2280
|
+
* the loop, not the hook.
|
|
2281
|
+
*/
|
|
2282
|
+
readonly retryDelayFor?: (err: unknown) => number | undefined
|
|
2283
|
+
}
|
|
2284
|
+
|
|
2285
|
+
/**
|
|
2286
|
+
* Poll until `read` reports a binding. `read` returns `undefined` for "not
|
|
2287
|
+
* yet" and throws for a failure worth surfacing; the deadline owns every wait,
|
|
2288
|
+
* including the sleep between attempts, so an expired clock cannot be extended
|
|
2289
|
+
* by one more round trip. Shaped after ACI's `pollForRunningIp`.
|
|
2290
|
+
*
|
|
2291
|
+
* The only failure this raises on its own account is
|
|
2292
|
+
* {@link ReadinessPollTimeout}; anything `read` throws travels out unchanged,
|
|
2293
|
+
* unless `behaviour.retryDelayFor` claims it — in which case it is repeated
|
|
2294
|
+
* inside the same budget and, if the budget then runs out, carried onto the
|
|
2295
|
+
* timeout as its `cause`, so a poll that kept failing still says what it kept
|
|
2296
|
+
* seeing.
|
|
2297
|
+
*/
|
|
2298
|
+
export async function pollForBinding(
|
|
2299
|
+
read: (signal: AbortSignal) => Promise<KubernetesSandboxBinding | undefined>,
|
|
2300
|
+
deadline: OperationDeadline,
|
|
2301
|
+
readiness: { readonly timeoutMs: number; readonly pollIntervalMs: number },
|
|
2302
|
+
label: string,
|
|
2303
|
+
behaviour: ReadinessPollBehaviour = {},
|
|
2304
|
+
): Promise<KubernetesSandboxBinding> {
|
|
2305
|
+
/**
|
|
2306
|
+
* The last failure that was retried rather than raised, and only while it
|
|
2307
|
+
* is still the truth: a read that succeeds clears it, so the timeout
|
|
2308
|
+
* carries a failure the poll was STILL meeting when the budget ran out
|
|
2309
|
+
* rather than a blip it recovered from twenty polls earlier. The
|
|
2310
|
+
* difference is not cosmetic — the acquire's reason is read off this
|
|
2311
|
+
* cause, so a stale one would report an API outage for a poll whose API
|
|
2312
|
+
* was answering fine.
|
|
2313
|
+
*/
|
|
2314
|
+
let lastRetried: unknown
|
|
2315
|
+
while (deadline.remainingMs() > 0) {
|
|
2316
|
+
/** Overrides the poll cadence for one round when the server named one. */
|
|
2317
|
+
let nextDelayMs = readiness.pollIntervalMs
|
|
2318
|
+
try {
|
|
2319
|
+
const binding = await deadline.run(read)
|
|
2320
|
+
lastRetried = undefined
|
|
2321
|
+
if (binding) return binding
|
|
2322
|
+
} catch (err) {
|
|
2323
|
+
if (err instanceof OperationDeadlineExpired) break
|
|
2324
|
+
const retryDelayMs = behaviour.retryDelayFor?.(err)
|
|
2325
|
+
if (retryDelayMs === undefined) throw err
|
|
2326
|
+
lastRetried = err
|
|
2327
|
+
nextDelayMs = retryDelayMs
|
|
2328
|
+
}
|
|
2329
|
+
try {
|
|
2330
|
+
await deadline.delay(nextDelayMs)
|
|
2331
|
+
} catch (err) {
|
|
2332
|
+
if (err instanceof OperationDeadlineExpired) break
|
|
2333
|
+
throw err
|
|
2334
|
+
}
|
|
2335
|
+
}
|
|
2336
|
+
throw new ReadinessPollTimeout(
|
|
2337
|
+
`kubernetes: ${label} never became Ready (${readiness.timeoutMs}ms)${
|
|
2338
|
+
lastRetried !== undefined
|
|
2339
|
+
? `; the last API failure retried inside that budget was: ${
|
|
2340
|
+
lastRetried instanceof Error ? lastRetried.message : String(lastRetried)
|
|
2341
|
+
}`
|
|
2342
|
+
: ''
|
|
2343
|
+
}`,
|
|
2344
|
+
lastRetried !== undefined ? { cause: lastRetried } : undefined,
|
|
2345
|
+
)
|
|
2346
|
+
}
|
|
2347
|
+
|
|
2348
|
+
/**
|
|
2349
|
+
* The live pod behind a bound sandbox: its uid, and — read in the SAME
|
|
2350
|
+
* answer — the IP it can be dialed at.
|
|
2351
|
+
*
|
|
2352
|
+
* One record rather than two reads because the two facts have to describe one
|
|
2353
|
+
* pod. The uid is the agent's bind token and the IP is where that agent
|
|
2354
|
+
* listens; taking them from separate GETs leaves a window in which a resume,
|
|
2355
|
+
* an eviction or a node drain replaces the pod in between, and the handle
|
|
2356
|
+
* then presents pod A's token at pod B's address. The guest answers that with
|
|
2357
|
+
* a flat `unauthorized`, which says nothing about the race that caused it.
|
|
2358
|
+
*/
|
|
2359
|
+
export interface KubernetesBoundPod {
|
|
2360
|
+
/** `metadata.uid` — the agent's bind token. */
|
|
2361
|
+
readonly uid: string
|
|
2362
|
+
/** `status.podIP`. Read by `agentAddress: 'pod-ip'`; absent is legal. */
|
|
2363
|
+
readonly podIP?: string
|
|
2364
|
+
/**
|
|
2365
|
+
* `metadata.labels` — what an ingress policy's `podSelector` actually
|
|
2366
|
+
* matches. Read off the SAME object the uid and the address come from,
|
|
2367
|
+
* for the same reason they are: a policy decision made about one pod and
|
|
2368
|
+
* a connection made to another is the mismatch this record exists to
|
|
2369
|
+
* prevent. The claim path reads it to ask which policies select the bound
|
|
2370
|
+
* pod — a directly created Sandbox's labels are known before its pod
|
|
2371
|
+
* exists — and BOTH paths read it to confirm the pod really carries the
|
|
2372
|
+
* labels this backend asked the controller for, which is the one thing
|
|
2373
|
+
* knowing them in advance cannot establish. See `ingress-policy.ts` and
|
|
2374
|
+
* `assertRequestedPodLabelsObserved`.
|
|
2375
|
+
*/
|
|
2376
|
+
readonly labels?: Readonly<Record<string, string>>
|
|
2377
|
+
}
|
|
2378
|
+
|
|
2379
|
+
/**
|
|
2380
|
+
* Find the pod a sandbox is currently backed by, and read both facts off it.
|
|
2381
|
+
*
|
|
2382
|
+
* `uid` is the per-instance agent bind token; `podIP` is where that agent
|
|
2383
|
+
* listens, and only `agentAddress: 'pod-ip'` reads it.
|
|
2384
|
+
*
|
|
2385
|
+
* The pod is named after its Sandbox in agent-sandbox v1.0.2 — verified
|
|
2386
|
+
* against a running cluster — but that is an observation, not a documented
|
|
2387
|
+
* guarantee, and `Sandbox.status` exposes no pod name to fall back on. So the
|
|
2388
|
+
* fast path is one GET by that name, and the only cost of the name convention
|
|
2389
|
+
* changing upstream is a second round trip through `status.selector`, which is
|
|
2390
|
+
* exactly what the controller publishes the selector for.
|
|
2391
|
+
*/
|
|
2392
|
+
export async function readBoundPod(
|
|
2393
|
+
client: KubernetesClient,
|
|
2394
|
+
namespace: string,
|
|
2395
|
+
binding: KubernetesSandboxBinding,
|
|
2396
|
+
signal?: AbortSignal,
|
|
2397
|
+
): Promise<KubernetesBoundPod> {
|
|
2398
|
+
try {
|
|
2399
|
+
const pod = await client.request<PodResource>(
|
|
2400
|
+
'GET',
|
|
2401
|
+
podPath(namespace, binding.name),
|
|
2402
|
+
undefined,
|
|
2403
|
+
signal,
|
|
2404
|
+
)
|
|
2405
|
+
const uid = pod?.metadata?.uid
|
|
2406
|
+
// `isPodLive` matters on the RESUME path in `workspace.ts`: a resumed
|
|
2407
|
+
// pod keeps its name, so for as long as the outgoing one is
|
|
2408
|
+
// terminating this GET can answer with the pod that is leaving and a
|
|
2409
|
+
// uid the new agent will refuse. On the acquire path nothing is
|
|
2410
|
+
// terminating and the filter never fires.
|
|
2411
|
+
if (uid && isPodLive(pod)) return boundPod(uid, pod)
|
|
2412
|
+
} catch (err) {
|
|
2413
|
+
if (!(err instanceof KubernetesAlreadyGoneError)) throw err
|
|
2414
|
+
}
|
|
2415
|
+
|
|
2416
|
+
const selector =
|
|
2417
|
+
binding.podSelector ?? (await readSandboxSelector(client, namespace, binding, signal))
|
|
2418
|
+
if (selector !== undefined && selector !== '') {
|
|
2419
|
+
const list = await client.request<PodListResource>(
|
|
2420
|
+
'GET',
|
|
2421
|
+
podListPath(namespace, selector),
|
|
2422
|
+
undefined,
|
|
2423
|
+
signal,
|
|
2424
|
+
)
|
|
2425
|
+
for (const pod of list?.items ?? []) {
|
|
2426
|
+
const uid = pod.metadata?.uid
|
|
2427
|
+
if (uid && isPodLive(pod)) return boundPod(uid, pod)
|
|
2428
|
+
}
|
|
2429
|
+
}
|
|
2430
|
+
throw new Error(
|
|
2431
|
+
`kubernetes: could not read a pod uid for sandbox ${binding.name} in namespace ${namespace} — no live pod of that name, and its status.selector matched no live pod either (a pod carrying a deletionTimestamp, or in phase Succeeded/Failed, is never bound to). The pod uid is the agent's bind token, so the sandbox is refused rather than returned unauthenticated.`,
|
|
2432
|
+
)
|
|
2433
|
+
}
|
|
2434
|
+
|
|
2435
|
+
function boundPod(uid: string, pod: PodResource): KubernetesBoundPod {
|
|
2436
|
+
const podIP = readPodIP(pod)
|
|
2437
|
+
const labels = pod.metadata?.labels
|
|
2438
|
+
return {
|
|
2439
|
+
uid,
|
|
2440
|
+
...(podIP !== undefined ? { podIP } : {}),
|
|
2441
|
+
...(labels !== undefined ? { labels } : {}),
|
|
2442
|
+
}
|
|
2443
|
+
}
|
|
2444
|
+
|
|
2445
|
+
/**
|
|
2446
|
+
* {@link readBoundPod}, plus — under `'pod-ip'` only — the wait for an
|
|
2447
|
+
* address to go with the token.
|
|
2448
|
+
*
|
|
2449
|
+
* Under the default mode this is the single read it has always been: one
|
|
2450
|
+
* `GET`, in the same place in the same order, because a Service FQDN is
|
|
2451
|
+
* published with the Sandbox and needs nothing from the pod but its uid.
|
|
2452
|
+
*
|
|
2453
|
+
* `'pod-ip'` has to wait, because a LIVE pod is not yet an ADDRESSED pod. A
|
|
2454
|
+
* pod is created `Pending` and carries no `status.podIP` until the CNI has
|
|
2455
|
+
* finished attaching it, and {@link isPodLive} accepts `Pending` on purpose —
|
|
2456
|
+
* the resume path in `workspace.ts` binds its replacement pod long before that
|
|
2457
|
+
* pod is Ready, because `Ready` stays True across the transition and the uid
|
|
2458
|
+
* is the only transition signal there is. Refusing an address-less pod outright
|
|
2459
|
+
* would therefore fail on the NORMAL path, in milliseconds, with the whole
|
|
2460
|
+
* readiness budget unspent. So "live, no address yet" is polled on the same
|
|
2461
|
+
* deadline as everything else on this path, and
|
|
2462
|
+
* {@link resolveAgentAddress}'s own refusal is left as the post-deadline
|
|
2463
|
+
* backstop for a pod that never gets an address at all.
|
|
2464
|
+
*
|
|
2465
|
+
* A failed READ stays fatal, exactly as it was: this is the acquire path,
|
|
2466
|
+
* where nothing is being replaced and a pod that cannot be read is not a pod
|
|
2467
|
+
* that is about to appear.
|
|
2468
|
+
*
|
|
2469
|
+
* `requiredLabels` is the second thing worth waiting for, and it is waited
|
|
2470
|
+
* for in the SAME loop rather than in a second one: an egress profile is a
|
|
2471
|
+
* label the CONTROLLER patches onto the pod it binds, so a pod read the
|
|
2472
|
+
* instant it was bound can be live, addressed and not yet labelled. Two
|
|
2473
|
+
* loops would be two deadlines and two answers to "is this pod ready to be
|
|
2474
|
+
* admitted". Like the address, an expired clock hands the pod back as it is
|
|
2475
|
+
* — the caller decides whether a missing label is fatal, and on the acquire
|
|
2476
|
+
* path it is: see `assertRequestedPodLabelsObserved`.
|
|
2477
|
+
*/
|
|
2478
|
+
export async function readAddressedPod(
|
|
2479
|
+
client: KubernetesClient,
|
|
2480
|
+
namespace: string,
|
|
2481
|
+
binding: KubernetesSandboxBinding,
|
|
2482
|
+
deadline: OperationDeadline,
|
|
2483
|
+
readiness: { readonly pollIntervalMs: number },
|
|
2484
|
+
mode: KubernetesAgentAddressMode,
|
|
2485
|
+
requiredLabels?: Readonly<Record<string, string>>,
|
|
2486
|
+
): Promise<KubernetesBoundPod> {
|
|
2487
|
+
const required = Object.entries(requiredLabels ?? {})
|
|
2488
|
+
const wanting = (pod: KubernetesBoundPod): boolean =>
|
|
2489
|
+
(mode === 'pod-ip' && pod.podIP === undefined) ||
|
|
2490
|
+
required.some(([key, value]) => pod.labels?.[key] !== value)
|
|
2491
|
+
|
|
2492
|
+
let pod = await deadline.run((signal) => readBoundPod(client, namespace, binding, signal))
|
|
2493
|
+
while (wanting(pod) && deadline.remainingMs() > 0) {
|
|
2494
|
+
try {
|
|
2495
|
+
await deadline.delay(readiness.pollIntervalMs)
|
|
2496
|
+
pod = await deadline.run((signal) => readBoundPod(client, namespace, binding, signal))
|
|
2497
|
+
} catch (err) {
|
|
2498
|
+
// An expired clock hands the incomplete pod back rather than
|
|
2499
|
+
// replacing it with a bare "deadline expired": the caller's
|
|
2500
|
+
// `resolveAgentAddress` (or `assertRequestedPodLabelsObserved`) then
|
|
2501
|
+
// reports WHICH fact never arrived.
|
|
2502
|
+
if (err instanceof OperationDeadlineExpired) break
|
|
2503
|
+
throw err
|
|
2504
|
+
}
|
|
2505
|
+
}
|
|
2506
|
+
return pod
|
|
2507
|
+
}
|
|
2508
|
+
|
|
2509
|
+
/**
|
|
2510
|
+
* The re-read a `'pod-ip'` handle follows a replaced pod with: one live-pod
|
|
2511
|
+
* read, then the same address resolution acquire did.
|
|
2512
|
+
*
|
|
2513
|
+
* Built here rather than inside the transport because finding the pod is a
|
|
2514
|
+
* CONTROL-plane act — the by-name GET, the selector fallback, the
|
|
2515
|
+
* liveness filter — and the transport owns none of that. It is handed over as
|
|
2516
|
+
* a closure so the transport can call it without learning what a Sandbox is.
|
|
2517
|
+
*/
|
|
2518
|
+
export function buildAgentAddressRefresh(
|
|
2519
|
+
client: KubernetesClient,
|
|
2520
|
+
namespace: string,
|
|
2521
|
+
binding: KubernetesSandboxBinding,
|
|
2522
|
+
agentPort: number,
|
|
2523
|
+
mode: KubernetesAgentAddressMode,
|
|
2524
|
+
): (signal?: AbortSignal) => Promise<KubernetesAgentAddress> {
|
|
2525
|
+
return async (signal) => {
|
|
2526
|
+
const pod = await readBoundPod(client, namespace, binding, signal)
|
|
2527
|
+
return resolveAgentAddress(binding, agentPort, pod.uid, {
|
|
2528
|
+
mode,
|
|
2529
|
+
...(pod.podIP !== undefined ? { podIP: pod.podIP } : {}),
|
|
2530
|
+
})
|
|
2531
|
+
}
|
|
2532
|
+
}
|
|
2533
|
+
|
|
2534
|
+
async function readSandboxSelector(
|
|
2535
|
+
client: KubernetesClient,
|
|
2536
|
+
namespace: string,
|
|
2537
|
+
binding: KubernetesSandboxBinding,
|
|
2538
|
+
signal?: AbortSignal,
|
|
2539
|
+
): Promise<string | undefined> {
|
|
2540
|
+
try {
|
|
2541
|
+
const sandbox = await client.request<SandboxResource>(
|
|
2542
|
+
'GET',
|
|
2543
|
+
sandboxPath(namespace, binding.name),
|
|
2544
|
+
undefined,
|
|
2545
|
+
signal,
|
|
2546
|
+
)
|
|
2547
|
+
return sandbox?.status?.selector
|
|
2548
|
+
} catch (err) {
|
|
2549
|
+
if (err instanceof KubernetesAlreadyGoneError) return undefined
|
|
2550
|
+
throw err
|
|
2551
|
+
}
|
|
2552
|
+
}
|
|
2553
|
+
|
|
2554
|
+
/**
|
|
2555
|
+
* Build the Sandbox, prove it is deprivileged, and only then hand it back.
|
|
2556
|
+
*
|
|
2557
|
+
* The probe runs BEFORE `create()` resolves, so a caller never holds a
|
|
2558
|
+
* reference to an under-hardened sandbox — a probe that refuses destroys the
|
|
2559
|
+
* instance on a bounded cleanup budget and rethrows, exactly as a readiness
|
|
2560
|
+
* failure does. There is no configuration that skips it: the whole value of
|
|
2561
|
+
* checking the deprivileging on every acquire rather than once by hand is
|
|
2562
|
+
* that it cannot be forgotten, and an off switch is a way to forget it.
|
|
2563
|
+
*
|
|
2564
|
+
* The probe goes through the Sandbox's own `exec`, not the raw transport, so
|
|
2565
|
+
* it traverses the same reserve/admit/stream/confirm path every later call
|
|
2566
|
+
* will. A sandbox that cannot answer the probe is one a caller could not use
|
|
2567
|
+
* either.
|
|
2568
|
+
*
|
|
2569
|
+
* ## And it runs on a clock
|
|
2570
|
+
*
|
|
2571
|
+
* This is the first thing on the acquire path that talks to the GUEST, and
|
|
2572
|
+
* the readiness deadline that bounded everything before it has already
|
|
2573
|
+
* expired. Left unbounded the probe would inherit the execution controller's
|
|
2574
|
+
* generic defaults instead — a five-minute observation, then a cancel-confirm
|
|
2575
|
+
* and a drain — so a pod whose agent has wedged (out of memory, an event loop
|
|
2576
|
+
* the workload blocked) would keep `create()` pending for minutes past the
|
|
2577
|
+
* caller's `readyTimeoutMs` with nothing reported. It gets its own deadline,
|
|
2578
|
+
* whose expiry takes the same cleanup-and-reject path every other refusal
|
|
2579
|
+
* does, in words that name the hang rather than blame a missing `cat`.
|
|
2580
|
+
*/
|
|
2581
|
+
async function admitProbedSandbox(
|
|
2582
|
+
acquisition: KubernetesAcquisition,
|
|
2583
|
+
config: KubernetesBackendInternalConfig,
|
|
2584
|
+
options: SandboxBackendOptions,
|
|
2585
|
+
probeTimeoutMs: number,
|
|
2586
|
+
/**
|
|
2587
|
+
* Present exactly when `config.egress.perSandbox` is configured — which
|
|
2588
|
+
* is what makes `setNetworkPolicy` present on the handle. See
|
|
2589
|
+
* `per-sandbox-policy.ts`.
|
|
2590
|
+
*/
|
|
2591
|
+
setNetworkPolicy?: (policy: SandboxNetworkPolicy) => Promise<void>,
|
|
2592
|
+
): Promise<Sandbox> {
|
|
2593
|
+
const sandbox = buildKubernetesSandbox({
|
|
2594
|
+
name: acquisition.binding.name,
|
|
2595
|
+
rootDir: options.workingDirectory,
|
|
2596
|
+
transport: new KubernetesAgentTransport(acquisition.agent, {
|
|
2597
|
+
// The backend opts in to the stream heartbeat; the transport
|
|
2598
|
+
// option it sets stays undefined for every other tier.
|
|
2599
|
+
heartbeatMs: resolveStreamHeartbeatMs(config.streamHeartbeatMs),
|
|
2600
|
+
...(acquisition.refreshAgent !== undefined
|
|
2601
|
+
? { refreshHandle: acquisition.refreshAgent }
|
|
2602
|
+
: {}),
|
|
2603
|
+
}),
|
|
2604
|
+
release: acquisition.release,
|
|
2605
|
+
renew: acquisition.renew,
|
|
2606
|
+
ttlSeconds: acquisition.ttlSeconds,
|
|
2607
|
+
...(config.onLeaseRenewalError !== undefined
|
|
2608
|
+
? { onRenewalError: config.onLeaseRenewalError }
|
|
2609
|
+
: {}),
|
|
2610
|
+
...(setNetworkPolicy !== undefined ? { setNetworkPolicy } : {}),
|
|
2611
|
+
})
|
|
2612
|
+
try {
|
|
2613
|
+
await probeSandboxPrivileges(sandbox, acquisition.binding.name, probeTimeoutMs, options.signal)
|
|
2614
|
+
} catch (err) {
|
|
2615
|
+
await runFailureCleanup(async (signal) => {
|
|
2616
|
+
await sandbox.destroy({ signal })
|
|
2617
|
+
})
|
|
2618
|
+
throw err
|
|
2619
|
+
}
|
|
2620
|
+
return sandbox
|
|
2621
|
+
}
|
|
2622
|
+
|
|
2623
|
+
/**
|
|
2624
|
+
* Run the probe against a built Sandbox, on its own clock, and throw if it
|
|
2625
|
+
* refuses. Cleanup is the CALLER's, and the two callers want opposite things:
|
|
2626
|
+
* a task acquire destroys the instance, while `workspace.ts` suspends it,
|
|
2627
|
+
* because deleting a workspace deletes its disk and a probe refusal is not a
|
|
2628
|
+
* reason to lose a caller's files.
|
|
2629
|
+
*/
|
|
2630
|
+
export async function probeSandboxPrivileges(
|
|
2631
|
+
sandbox: Sandbox,
|
|
2632
|
+
sandboxName: string,
|
|
2633
|
+
probeTimeoutMs: number,
|
|
2634
|
+
signal?: AbortSignal,
|
|
2635
|
+
): Promise<void> {
|
|
2636
|
+
// Labelled with the sandbox, so an expiry read off a log line says which
|
|
2637
|
+
// acquire stopped answering — and so the catch below can tell THIS
|
|
2638
|
+
// deadline from any other that might surface through the same exec.
|
|
2639
|
+
const probeLabel = `kubernetes privilege probe ${sandboxName}`
|
|
2640
|
+
try {
|
|
2641
|
+
signal?.throwIfAborted()
|
|
2642
|
+
// The deadline's signal covers both ways this should stop early: it
|
|
2643
|
+
// aborts on expiry, and it aborts with the caller's own reason when
|
|
2644
|
+
// `signal` does. Handing it to `exec` is what releases the guest-side
|
|
2645
|
+
// execution rather than merely abandoning the wait.
|
|
2646
|
+
await new OperationDeadline(probeTimeoutMs, probeLabel, signal).run(
|
|
2647
|
+
async (execSignal) =>
|
|
2648
|
+
await runPrivilegeProbe(
|
|
2649
|
+
async (command, args) => await sandbox.exec(command, args, { signal: execSignal }),
|
|
2650
|
+
sandboxName,
|
|
2651
|
+
),
|
|
2652
|
+
)
|
|
2653
|
+
// An abort that lands WHILE the probe is in flight must not leave a
|
|
2654
|
+
// live sandbox behind: the probe itself may well have finished, and
|
|
2655
|
+
// the caller who cancelled is about to stop holding the reference
|
|
2656
|
+
// that could destroy it. Same cleanup, one branch later.
|
|
2657
|
+
signal?.throwIfAborted()
|
|
2658
|
+
} catch (err) {
|
|
2659
|
+
// A caller who cancelled mid-probe gets THEIR reason, not the probe's
|
|
2660
|
+
// account of a command that was cancelled out from under it.
|
|
2661
|
+
signal?.throwIfAborted()
|
|
2662
|
+
// A probe that ran out of time is a probe that could not run, and is
|
|
2663
|
+
// refused in those words: `OperationDeadlineExpired` on its own would
|
|
2664
|
+
// leave a reader guessing which half of the acquire went quiet.
|
|
2665
|
+
if (err instanceof OperationDeadlineExpired && err.label === probeLabel) {
|
|
2666
|
+
throw privilegeProbeTimedOut(sandboxName, probeTimeoutMs, err)
|
|
2667
|
+
}
|
|
2668
|
+
throw err
|
|
2669
|
+
}
|
|
2670
|
+
}
|