@namzu/sandbox 14.0.0 → 16.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +924 -0
- package/README.md +369 -14
- package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
- package/dist/backends/aci-standby-pool/index.js +13 -1
- package/dist/backends/aci-standby-pool/index.js.map +1 -1
- package/dist/backends/docker/index.d.ts +169 -6
- package/dist/backends/docker/index.d.ts.map +1 -1
- package/dist/backends/docker/index.js +499 -85
- package/dist/backends/docker/index.js.map +1 -1
- package/dist/backends/firecracker/index.d.ts.map +1 -1
- package/dist/backends/firecracker/index.js +12 -2
- package/dist/backends/firecracker/index.js.map +1 -1
- package/dist/backends/firecracker/protocol.d.ts +459 -8
- package/dist/backends/firecracker/protocol.d.ts.map +1 -1
- package/dist/backends/firecracker/protocol.js +136 -0
- package/dist/backends/firecracker/protocol.js.map +1 -1
- package/dist/backends/firecracker/transport.d.ts +539 -6
- package/dist/backends/firecracker/transport.d.ts.map +1 -1
- package/dist/backends/firecracker/transport.js +1171 -24
- package/dist/backends/firecracker/transport.js.map +1 -1
- package/dist/backends/kubernetes/egress-policy.d.ts +1181 -13
- package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -1
- package/dist/backends/kubernetes/egress-policy.js +2350 -31
- package/dist/backends/kubernetes/egress-policy.js.map +1 -1
- package/dist/backends/kubernetes/identity.d.ts +193 -0
- package/dist/backends/kubernetes/identity.d.ts.map +1 -0
- package/dist/backends/kubernetes/identity.js +147 -0
- package/dist/backends/kubernetes/identity.js.map +1 -0
- package/dist/backends/kubernetes/index.d.ts +678 -33
- package/dist/backends/kubernetes/index.d.ts.map +1 -1
- package/dist/backends/kubernetes/index.js +1180 -95
- package/dist/backends/kubernetes/index.js.map +1 -1
- package/dist/backends/kubernetes/ingress-policy.d.ts +375 -0
- package/dist/backends/kubernetes/ingress-policy.d.ts.map +1 -0
- package/dist/backends/kubernetes/ingress-policy.js +1050 -0
- package/dist/backends/kubernetes/ingress-policy.js.map +1 -0
- package/dist/backends/kubernetes/k8s-client.d.ts +213 -4
- package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -1
- package/dist/backends/kubernetes/k8s-client.js +359 -52
- package/dist/backends/kubernetes/k8s-client.js.map +1 -1
- package/dist/backends/kubernetes/lease.d.ts +40 -14
- package/dist/backends/kubernetes/lease.d.ts.map +1 -1
- package/dist/backends/kubernetes/lease.js +68 -18
- package/dist/backends/kubernetes/lease.js.map +1 -1
- package/dist/backends/kubernetes/objects.d.ts +423 -3
- package/dist/backends/kubernetes/objects.d.ts.map +1 -1
- package/dist/backends/kubernetes/objects.js +364 -2
- package/dist/backends/kubernetes/objects.js.map +1 -1
- package/dist/backends/kubernetes/per-sandbox-policy.d.ts +219 -0
- package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -0
- package/dist/backends/kubernetes/per-sandbox-policy.js +375 -0
- package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -0
- package/dist/backends/kubernetes/rbac.d.ts +153 -0
- package/dist/backends/kubernetes/rbac.d.ts.map +1 -0
- package/dist/backends/kubernetes/rbac.js +177 -0
- package/dist/backends/kubernetes/rbac.js.map +1 -0
- package/dist/backends/kubernetes/sandbox.d.ts +81 -14
- package/dist/backends/kubernetes/sandbox.d.ts.map +1 -1
- package/dist/backends/kubernetes/sandbox.js +149 -15
- package/dist/backends/kubernetes/sandbox.js.map +1 -1
- package/dist/backends/kubernetes/transport.d.ts +935 -9
- package/dist/backends/kubernetes/transport.d.ts.map +1 -1
- package/dist/backends/kubernetes/transport.js +1958 -62
- package/dist/backends/kubernetes/transport.js.map +1 -1
- package/dist/backends/kubernetes/workspace.d.ts +1149 -18
- package/dist/backends/kubernetes/workspace.d.ts.map +1 -1
- package/dist/backends/kubernetes/workspace.js +2825 -186
- package/dist/backends/kubernetes/workspace.js.map +1 -1
- package/dist/backends/remote-execution-controller.d.ts +14 -0
- package/dist/backends/remote-execution-controller.d.ts.map +1 -1
- package/dist/backends/remote-execution-controller.js.map +1 -1
- package/dist/index.d.ts +294 -18
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +280 -10
- package/dist/index.js.map +1 -1
- package/dist/testing/sandbox-conformance.d.ts +39 -5
- package/dist/testing/sandbox-conformance.d.ts.map +1 -1
- package/dist/testing/sandbox-conformance.js +436 -5
- package/dist/testing/sandbox-conformance.js.map +1 -1
- package/package.json +3 -3
- package/src/backends/aci-standby-pool/index.ts +16 -1
- package/src/backends/docker/index.ts +617 -100
- package/src/backends/firecracker/index.ts +14 -2
- package/src/backends/firecracker/protocol.ts +514 -6
- package/src/backends/firecracker/transport.ts +1492 -40
- package/src/backends/kubernetes/egress-policy.ts +3334 -55
- package/src/backends/kubernetes/identity.ts +261 -0
- package/src/backends/kubernetes/index.ts +1785 -127
- package/src/backends/kubernetes/ingress-policy.ts +1344 -0
- package/src/backends/kubernetes/k8s-client.ts +444 -54
- package/src/backends/kubernetes/lease.ts +75 -19
- package/src/backends/kubernetes/objects.ts +626 -6
- package/src/backends/kubernetes/per-sandbox-policy.ts +497 -0
- package/src/backends/kubernetes/rbac.ts +192 -0
- package/src/backends/kubernetes/sandbox.ts +218 -20
- package/src/backends/kubernetes/transport.ts +2733 -124
- package/src/backends/kubernetes/workspace.ts +4476 -222
- package/src/backends/remote-execution-controller.ts +14 -0
- package/src/index.ts +668 -19
- package/src/testing/sandbox-conformance.ts +540 -5
|
@@ -77,21 +77,43 @@
|
|
|
77
77
|
* ## Egress
|
|
78
78
|
*
|
|
79
79
|
* `config.egress` is optional and, when set, translated and VERIFIED — never
|
|
80
|
-
* created — by `egress-policy.ts
|
|
81
|
-
*
|
|
82
|
-
*
|
|
83
|
-
*
|
|
84
|
-
*
|
|
85
|
-
*
|
|
86
|
-
*
|
|
80
|
+
* created — by `egress-policy.ts`, in two steps. The NAMED object is GETted
|
|
81
|
+
* and compared to the translation exactly, once, lazily, on the first
|
|
82
|
+
* `create()`, so `buildKubernetesBackend` itself still contacts nothing.
|
|
83
|
+
* Then, because the API server UNIONS every policy selecting a pod, every
|
|
84
|
+
* `NetworkPolicy` in the namespace (and, under `engine: 'cilium'`, every
|
|
85
|
+
* `CiliumNetworkPolicy`) is enumerated against the pod's real labels and the
|
|
86
|
+
* create is refused when any of them lets out more than the translation does
|
|
87
|
+
* — a second policy widens egress however exactly the named one matches, and
|
|
88
|
+
* a `SandboxTemplate`'s own `networkPolicy` block becomes exactly such a
|
|
89
|
+
* policy. `egress.verify: 'named-object-only'` is the opt-out and restores
|
|
90
|
+
* the first step alone. Every Sandbox this file creates directly
|
|
91
|
+
* (`buildSandboxBody`) carries {@link sandboxTemplateLabel} on its
|
|
92
|
+
* podTemplate specifically so that translated policy's `podSelector` has
|
|
93
|
+
* something stable to match — see `objects.ts`'s doc comment on that label
|
|
94
|
+
* for why agent-sandbox's own controller-owned label does not cover this
|
|
95
|
+
* path.
|
|
96
|
+
*
|
|
97
|
+
* ## Ingress
|
|
98
|
+
*
|
|
99
|
+
* `config.ingress` is the opposite default: verification is ON unless a
|
|
100
|
+
* deployment says `'unverified'`. Before the POST for a direct Sandbox, and
|
|
101
|
+
* after the bind for a claimed one, `ingress-policy.ts` lists the namespace's
|
|
102
|
+
* policies and refuses unless one of them actually closes the agent port on
|
|
103
|
+
* the labels this pod carries. Nothing checked that before, while three
|
|
104
|
+
* pieces of shipped text said it was covered — see that module's own doc
|
|
105
|
+
* comment for what was measured.
|
|
87
106
|
*/
|
|
88
107
|
import type { Sandbox } from '@namzu/sdk';
|
|
89
108
|
import type { SandboxBackend, SandboxBackendOptions } from '../../index.js';
|
|
90
109
|
import { OperationDeadline } from '../readiness.js';
|
|
91
|
-
import { type KubernetesEgressConfig } from './egress-policy.js';
|
|
92
|
-
import { type
|
|
93
|
-
import { type
|
|
94
|
-
|
|
110
|
+
import { type EgressProfileLabel, type KubernetesEgressConfig } from './egress-policy.js';
|
|
111
|
+
import { type KubernetesIngressConfig } from './ingress-policy.js';
|
|
112
|
+
import { type KubernetesAccess, type KubernetesClient, type KubernetesClientOptions } from './k8s-client.js';
|
|
113
|
+
import { type SandboxClaimResource, type SandboxPodTemplate, type SandboxResource, type SandboxVolumeClaimTemplate } from './objects.js';
|
|
114
|
+
import { type PerSandboxPolicyOwner } from './per-sandbox-policy.js';
|
|
115
|
+
export type { KubernetesEgressConfig, KubernetesEgressEngine, KubernetesEgressPolicy, KubernetesEgressVerification, KubernetesOnlyEgressPolicy, KubernetesPerSandboxEgressConfig, } from './egress-policy.js';
|
|
116
|
+
export type { KubernetesIngressConfig, KubernetesIngressEngine } from './ingress-policy.js';
|
|
95
117
|
/**
|
|
96
118
|
* How the backend reaches the API server. Two sources, neither needing a YAML
|
|
97
119
|
* parser — see `k8s-client.ts` for the reasoning.
|
|
@@ -119,6 +141,13 @@ export interface KubernetesBackendInternalConfig {
|
|
|
119
141
|
readonly warmPoolName?: string;
|
|
120
142
|
/** TCP port the guest agent listens on. Default {@link DEFAULT_AGENT_PORT}. */
|
|
121
143
|
readonly agentPort?: number;
|
|
144
|
+
/**
|
|
145
|
+
* Which of a sandbox's two addresses the transport dials. Default
|
|
146
|
+
* `'service'` — see {@link KubernetesAgentAddressMode}, which is where
|
|
147
|
+
* the choice is explained, because it is a fact about where the HOST
|
|
148
|
+
* runs rather than about the cluster.
|
|
149
|
+
*/
|
|
150
|
+
readonly agentAddress?: KubernetesAgentAddressMode;
|
|
122
151
|
readonly readyPollIntervalMs?: number;
|
|
123
152
|
readonly readyTimeoutMs?: number;
|
|
124
153
|
/**
|
|
@@ -142,12 +171,107 @@ export interface KubernetesBackendInternalConfig {
|
|
|
142
171
|
/**
|
|
143
172
|
* Egress policy this backend's `NetworkPolicy` (or `CiliumNetworkPolicy`,
|
|
144
173
|
* under `engine: 'cilium'`) is expected to carry. Unset means this backend
|
|
145
|
-
* neither computes nor verifies one
|
|
146
|
-
*
|
|
147
|
-
*
|
|
174
|
+
* neither computes nor verifies one, and OUTBOUND traffic is then whatever
|
|
175
|
+
* the cluster's own policies happen to allow.
|
|
176
|
+
*
|
|
177
|
+
* Deliberately not described as "covered by the SandboxTemplate's managed
|
|
178
|
+
* NetworkPolicy", which is what this comment used to claim: that policy
|
|
179
|
+
* selects `agents.x-k8s.io/sandbox-template-ref-hash`, a label the
|
|
180
|
+
* controller writes only onto a Sandbox adopted out of a `SandboxWarmPool`
|
|
181
|
+
* and never onto one this backend POSTs — so for every pool-less sandbox
|
|
182
|
+
* and every workspace it selects nothing at all. See `ingress-policy.ts`.
|
|
148
183
|
*/
|
|
149
184
|
readonly egress?: KubernetesEgressConfig;
|
|
185
|
+
/**
|
|
186
|
+
* Whether the agent port's INGRESS boundary is verified before a sandbox
|
|
187
|
+
* is created, and against which policy resources.
|
|
188
|
+
*
|
|
189
|
+
* Unset means VERIFY — the one field in this config whose absent value is
|
|
190
|
+
* the strict one, because the deployment that needs the check is the one
|
|
191
|
+
* that would never have set it. `'unverified'` reads no policy and issues
|
|
192
|
+
* no request, for a deployment whose boundary lives somewhere a namespaced
|
|
193
|
+
* Role cannot see. See `ingress-policy.ts`.
|
|
194
|
+
*/
|
|
195
|
+
readonly ingress?: KubernetesIngressConfig;
|
|
196
|
+
/**
|
|
197
|
+
* Bound on every single Kubernetes API request this backend sends. See
|
|
198
|
+
* {@link KubernetesClientOptions.requestTimeoutMs} in `k8s-client.ts`,
|
|
199
|
+
* which owns the default (30 s), the floor (1 s) and the reason there is
|
|
200
|
+
* no value that disables it.
|
|
201
|
+
*/
|
|
202
|
+
readonly apiRequestTimeoutMs?: number;
|
|
203
|
+
/**
|
|
204
|
+
* Interval of the negotiated liveness heartbeat on `openTerminal` and
|
|
205
|
+
* `openTcpConnection` streams. Default
|
|
206
|
+
* {@link DEFAULT_STREAM_HEARTBEAT_MS}; `0` turns it off and restores the
|
|
207
|
+
* pre-heartbeat behaviour exactly. See {@link resolveStreamHeartbeatMs}.
|
|
208
|
+
*/
|
|
209
|
+
readonly streamHeartbeatMs?: number;
|
|
210
|
+
/**
|
|
211
|
+
* Extra labels written onto every `SandboxClaim` this backend POSTs —
|
|
212
|
+
* `metadata.labels`, and nowhere else. Never merged into
|
|
213
|
+
* `additionalPodMetadata`: those are POD labels a running Sandbox and its
|
|
214
|
+
* `NetworkPolicy` selectors read, and a host's own recovery bookkeeping
|
|
215
|
+
* has no business changing what a pod is selected by. Absent means no
|
|
216
|
+
* labels beyond what the controller itself writes, and every claim body
|
|
217
|
+
* this backend sends is byte-for-byte what it always was.
|
|
218
|
+
*
|
|
219
|
+
* The intended use is a host-instance identity — e.g.
|
|
220
|
+
* `{ 'sandbox.namzu.ai/host-instance': hostId }` — so a restarted host
|
|
221
|
+
* can find and {@link releaseKubernetesTaskSandboxes} its predecessor's
|
|
222
|
+
* claims well before `claimTtlSeconds` would reap them on its own.
|
|
223
|
+
*/
|
|
224
|
+
readonly claimLabels?: Record<string, string>;
|
|
150
225
|
}
|
|
226
|
+
/**
|
|
227
|
+
* Default {@link KubernetesBackendInternalConfig.streamHeartbeatMs} — 15 s,
|
|
228
|
+
* so a stream whose peer vanished without a FIN or an RST is given up on
|
|
229
|
+
* within 45 s rather than never.
|
|
230
|
+
*
|
|
231
|
+
* The Kubernetes backend opts IN here; `VsockTransportOptions.heartbeatMs`
|
|
232
|
+
* stays undefined by default, because that transport is shared with the
|
|
233
|
+
* Firecracker tier and a default there would force-close an existing
|
|
234
|
+
* consumer's quiet-but-alive terminal.
|
|
235
|
+
*/
|
|
236
|
+
export declare const DEFAULT_STREAM_HEARTBEAT_MS = 15000;
|
|
237
|
+
/**
|
|
238
|
+
* Validate the configured stream-heartbeat interval, or supply the default.
|
|
239
|
+
*
|
|
240
|
+
* `0` IS accepted here, unlike `apiRequestTimeoutMs`: the heartbeat is a new
|
|
241
|
+
* capability that a deployment behind a middlebox with its own idea about
|
|
242
|
+
* unexpected frames may want off, and turning it off restores exactly the
|
|
243
|
+
* behaviour every release before this one had. An unanswered API request has
|
|
244
|
+
* no such prior behaviour worth restoring.
|
|
245
|
+
*/
|
|
246
|
+
export declare function resolveStreamHeartbeatMs(value: number | undefined): number;
|
|
247
|
+
/**
|
|
248
|
+
* The client options every `createKubernetesClient` call in this backend is
|
|
249
|
+
* built with. One function so the five call sites — the provider, and each
|
|
250
|
+
* of the workspace verbs — cannot drift apart on which bounds they honour.
|
|
251
|
+
*/
|
|
252
|
+
export declare function clientOptions(config: KubernetesBackendInternalConfig): KubernetesClientOptions;
|
|
253
|
+
/**
|
|
254
|
+
* Which address a sandbox's agent is dialed at. A property of where the HOST
|
|
255
|
+
* runs, not of the cluster.
|
|
256
|
+
*
|
|
257
|
+
* - `'service'` (default) — the Sandbox's `status.serviceFQDN`,
|
|
258
|
+
* `<name>.<namespace>.svc.cluster.local`. It outlives the pod: a resumed
|
|
259
|
+
* workspace comes back behind the same name, and every dial re-resolves
|
|
260
|
+
* it. The catch is that only the cluster's own DNS answers it, so this is
|
|
261
|
+
* correct exactly when the host itself runs inside the cluster.
|
|
262
|
+
* - `'pod-ip'` — the bound pod's IP, read from the same `GET` that reads
|
|
263
|
+
* its uid, so the address and the bind token are always one pod's. For a
|
|
264
|
+
* host OUTSIDE the cluster on a routable pod network (a peered VNet, a
|
|
265
|
+
* node-local operator, a CI runner with a route): it needs no cluster
|
|
266
|
+
* resolver at all. The cost is that a pod IP dies with its pod, which is
|
|
267
|
+
* why a resume re-reads it and a connect failure re-reads it once.
|
|
268
|
+
*
|
|
269
|
+
* The cluster still decides whether either address is REACHABLE: `'pod-ip'`
|
|
270
|
+
* additionally needs a NetworkPolicy that admits the host's own address
|
|
271
|
+
* range on the agent port. Neither mode changes the bind token, the
|
|
272
|
+
* privilege probe or egress verification.
|
|
273
|
+
*/
|
|
274
|
+
export type KubernetesAgentAddressMode = 'service' | 'pod-ip';
|
|
151
275
|
/**
|
|
152
276
|
* The address the guest agent answers on, plus the token to present.
|
|
153
277
|
*
|
|
@@ -174,8 +298,33 @@ export interface KubernetesSandboxBinding {
|
|
|
174
298
|
export interface KubernetesAcquisition {
|
|
175
299
|
readonly binding: KubernetesSandboxBinding;
|
|
176
300
|
readonly agent: KubernetesAgentAddress;
|
|
301
|
+
/**
|
|
302
|
+
* Re-read the live pod and recompute {@link agent} from it. Present only
|
|
303
|
+
* under `agentAddress: 'pod-ip'`, where the address is a literal that
|
|
304
|
+
* dies with its pod; a Service FQDN needs no such thing, and handing one
|
|
305
|
+
* over anyway would give the default mode a re-read it never had.
|
|
306
|
+
*/
|
|
307
|
+
readonly refreshAgent?: (signal?: AbortSignal) => Promise<KubernetesAgentAddress>;
|
|
177
308
|
/** API path of the object THIS backend created — the claim, or the Sandbox. */
|
|
178
309
|
readonly ownedPath: string;
|
|
310
|
+
/**
|
|
311
|
+
* The object this backend created, identified well enough to be named in
|
|
312
|
+
* another object's `ownerReferences`: its kind, its name and the uid the
|
|
313
|
+
* API server assigned it.
|
|
314
|
+
*
|
|
315
|
+
* Present only when `config.egress.perSandbox` is configured, because it
|
|
316
|
+
* is read from the create reply and the readiness polls and nothing else
|
|
317
|
+
* needs it — a deployment that never writes a per-sandbox policy should
|
|
318
|
+
* not start carrying a field whose absence would otherwise be a bug.
|
|
319
|
+
*/
|
|
320
|
+
readonly owner?: PerSandboxPolicyOwner;
|
|
321
|
+
/**
|
|
322
|
+
* The VALUE of the per-sandbox selector label on this sandbox's pod —
|
|
323
|
+
* the created object's name, confirmed on the bound pod before the
|
|
324
|
+
* sandbox was admitted. Present under the same condition as
|
|
325
|
+
* {@link owner}.
|
|
326
|
+
*/
|
|
327
|
+
readonly perSandboxLabelValue?: string;
|
|
179
328
|
/** The TTL acquire stamped, which every renewal re-stamps. */
|
|
180
329
|
readonly ttlSeconds: number;
|
|
181
330
|
/** DELETE that object. An already-gone object counts as released. */
|
|
@@ -246,21 +395,229 @@ export declare function assertRuntimeClassIsApplicable(config: {
|
|
|
246
395
|
*/
|
|
247
396
|
export declare function buildKubernetesBackend(config: KubernetesBackendInternalConfig): SandboxBackend;
|
|
248
397
|
/**
|
|
249
|
-
*
|
|
250
|
-
*
|
|
251
|
-
*
|
|
398
|
+
* The two egress checks a create path runs, with their memos.
|
|
399
|
+
*
|
|
400
|
+
* `undefined` when `config.egress` is unset — the whole boundary is one
|
|
401
|
+
* absent object rather than a flag every call site re-reads, the same shape
|
|
402
|
+
* {@link IngressVerifier} uses for its own opt-out.
|
|
252
403
|
*
|
|
253
|
-
* Exported because `workspace.ts` runs the identical
|
|
404
|
+
* Exported because `workspace.ts` runs the identical steps: a workspace does
|
|
254
405
|
* not go through `buildKubernetesBackend`, and a config `egress` honoured on
|
|
255
406
|
* one entry point and ignored on the other would be a silent downgrade of the
|
|
256
|
-
* boundary this backend calls primary.
|
|
257
|
-
*
|
|
258
|
-
*
|
|
259
|
-
*
|
|
407
|
+
* boundary this backend calls primary. It builds its own boundary per create,
|
|
408
|
+
* which is what makes its checks per-call rather than memoized — creating a
|
|
409
|
+
* workspace is a rare, explicit act with nothing to amortise, and a policy
|
|
410
|
+
* deleted since the last call must be noticed.
|
|
260
411
|
*/
|
|
261
|
-
export
|
|
412
|
+
export interface KubernetesEgressBoundary {
|
|
413
|
+
/**
|
|
414
|
+
* Translate `egress.policy` and confirm an operator applied a matching
|
|
415
|
+
* object — the original verify-not-trust step, unchanged, including its
|
|
416
|
+
* exact-match comparison and its once-per-boundary memo.
|
|
417
|
+
*/
|
|
418
|
+
verifyNamedObject(signal?: AbortSignal): Promise<void>;
|
|
419
|
+
/**
|
|
420
|
+
* Enumerate every policy selecting THIS pod and refuse when any of them
|
|
421
|
+
* allows egress the translation does not. A no-op under
|
|
422
|
+
* `egress.verify: 'named-object-only'`.
|
|
423
|
+
*/
|
|
424
|
+
verifyUnion(podLabels: Readonly<Record<string, string>>, subject: string, signal?: AbortSignal): Promise<void>;
|
|
425
|
+
}
|
|
426
|
+
/**
|
|
427
|
+
* How long a union pass is trusted for one label set.
|
|
428
|
+
*
|
|
429
|
+
* Five minutes rather than the backend's lifetime, which is what the
|
|
430
|
+
* named-object check alone used to get: an operator who applies a widening
|
|
431
|
+
* policy at 10:00 should not have it go unnoticed until the host restarts.
|
|
432
|
+
* It is a cache, not a watch — `k8s-client.ts`'s "no watch, no informers, no
|
|
433
|
+
* resourceVersion tracking" invariant is untouched, because the only thing
|
|
434
|
+
* kept across calls is "this exact label set passed at this time".
|
|
435
|
+
*/
|
|
436
|
+
export declare const EGRESS_UNION_CACHE_TTL_MS: number;
|
|
437
|
+
/**
|
|
438
|
+
* Build the egress boundary this config asks for, or nothing at all.
|
|
439
|
+
*
|
|
440
|
+
* `sandboxTemplateName` is the template the caller is actually building from
|
|
441
|
+
* — it decides both the default policy name and the pod label the policy's
|
|
442
|
+
* selector has to match, and a workspace may be built from a different
|
|
443
|
+
* template than the task path's.
|
|
444
|
+
*
|
|
445
|
+
* `now` is injected only so the TTL above can be tested without waiting five
|
|
446
|
+
* minutes; nothing else passes it.
|
|
447
|
+
*/
|
|
448
|
+
export declare function buildEgressBoundary(client: KubernetesClient, config: KubernetesBackendInternalConfig, sandboxTemplateName: string, now?: () => number): KubernetesEgressBoundary | undefined;
|
|
449
|
+
/**
|
|
450
|
+
* What a create path calls to prove the agent port is closed before it hands
|
|
451
|
+
* a sandbox back. `undefined` when `config.ingress` is `'unverified'`, so the
|
|
452
|
+
* opt-out is one absent function rather than a flag every call site re-reads.
|
|
453
|
+
*/
|
|
454
|
+
export type IngressVerifier = (podLabels: Readonly<Record<string, string>>, subject: string, signal?: AbortSignal) => Promise<void>;
|
|
455
|
+
/**
|
|
456
|
+
* Build the ingress check this config asks for, or nothing at all.
|
|
457
|
+
*
|
|
458
|
+
* `cache` is the provider path's per-label-set memo; a workspace passes none,
|
|
459
|
+
* and re-checks on every call for the same reason its egress check does —
|
|
460
|
+
* creating a workspace is a rare, explicit act with nothing to amortise, and
|
|
461
|
+
* a policy deleted since the last call must be noticed.
|
|
462
|
+
*/
|
|
463
|
+
export declare function buildIngressVerifier(client: KubernetesClient, config: KubernetesBackendInternalConfig, cache?: Map<string, Promise<void>>): IngressVerifier | undefined;
|
|
262
464
|
/** Config → the client's own access shape. Shared with `workspace.ts`. */
|
|
263
465
|
export declare function clientAccess(config: KubernetesBackendInternalConfig): KubernetesAccess;
|
|
466
|
+
/**
|
|
467
|
+
* Why an acquire was refused, in the terms an operator acts on rather than
|
|
468
|
+
* the terms the failure happened to arrive in.
|
|
469
|
+
*
|
|
470
|
+
* - `'api-unreachable'` — the API server could not be reached, or kept
|
|
471
|
+
* answering with a status that means "not now": a connect failure, a 429,
|
|
472
|
+
* a 5xx. Retried inside the readiness budget before it ever reaches a
|
|
473
|
+
* caller, so seeing it means the whole budget was spent failing.
|
|
474
|
+
* - `'api-timeout'` — requests were accepted and never answered, until
|
|
475
|
+
* `apiRequestTimeoutMs` gave up on them. Also retried first.
|
|
476
|
+
* - `'forbidden'` — 401 or 403. The host's ServiceAccount cannot do this;
|
|
477
|
+
* no amount of waiting changes that. See the RBAC section of
|
|
478
|
+
* `docs/sdk/kubernetes-sandbox.md`.
|
|
479
|
+
* - `'claim-rejected'` — the controller REFUSED the claim, and said why.
|
|
480
|
+
* {@link KubernetesAcquireError.controllerReason} carries its own word for
|
|
481
|
+
* it. This is the one that used to burn the entire readiness budget before
|
|
482
|
+
* failing.
|
|
483
|
+
* - `'capacity'` — the pod exists and cannot be placed: the scheduler
|
|
484
|
+
* reports `PodScheduled=False` with reason `Unschedulable`. The cluster is
|
|
485
|
+
* full, or nothing matches the template's placement rules.
|
|
486
|
+
* - `'image-pull'` — the pod was placed and its container cannot start
|
|
487
|
+
* because the image will not pull. Permanent until an operator fixes the
|
|
488
|
+
* reference or the pull credential.
|
|
489
|
+
* - `'not-ready'` — none of the above: the readiness budget expired with the
|
|
490
|
+
* cluster reporting nothing wrong. A slow cold start, a webhook, an
|
|
491
|
+
* admission controller, a CNI that never attached the pod.
|
|
492
|
+
*/
|
|
493
|
+
export type KubernetesAcquireFailureReason = 'api-unreachable' | 'api-timeout' | 'forbidden' | 'claim-rejected' | 'capacity' | 'image-pull' | 'not-ready';
|
|
494
|
+
/**
|
|
495
|
+
* An acquire that was refused, carrying WHY in a field rather than in prose.
|
|
496
|
+
*
|
|
497
|
+
* Before this class a burst past node capacity and an API outage were the
|
|
498
|
+
* same plain `Error`, and a host could only tell them apart by matching
|
|
499
|
+
* message text that any release is free to reword. `reason` is the diagnosis,
|
|
500
|
+
* `retryable` is the advice that follows from it, and `cause` is the original
|
|
501
|
+
* failure — unmodified, so a host that already catches
|
|
502
|
+
* {@link ReadinessPollTimeout}, {@link KubernetesApiTimeoutError} or
|
|
503
|
+
* `KubernetesCredentialError` finds it there.
|
|
504
|
+
*
|
|
505
|
+
* `retryable` is about THIS acquire being worth attempting again, not about
|
|
506
|
+
* anything having been retried. Transient API failures are already retried
|
|
507
|
+
* inside the readiness budget, so a `retryable: true` that reaches a caller
|
|
508
|
+
* means the whole budget was spent on them.
|
|
509
|
+
*
|
|
510
|
+
* Not every acquire failure becomes one of these, and that is deliberate: a
|
|
511
|
+
* refusal this class cannot honestly diagnose — a malformed template, a 400
|
|
512
|
+
* from an admission webhook, a controller that reported Ready and named no
|
|
513
|
+
* sandbox — travels out as itself rather than being filed under whichever of
|
|
514
|
+
* the seven reasons is least wrong. `KubernetesApiError` carries the status
|
|
515
|
+
* for those.
|
|
516
|
+
*/
|
|
517
|
+
export declare class KubernetesAcquireError extends Error {
|
|
518
|
+
readonly name = "KubernetesAcquireError";
|
|
519
|
+
readonly reason: KubernetesAcquireFailureReason;
|
|
520
|
+
/** Whether attempting the same acquire again could plausibly succeed. */
|
|
521
|
+
readonly retryable: boolean;
|
|
522
|
+
/**
|
|
523
|
+
* `status.conditions[Ready].reason`, verbatim, when the controller
|
|
524
|
+
* refused the claim — `WarmPoolNotFound`, `TemplateNotFound`,
|
|
525
|
+
* `InvalidMetadata`, `EnvVarsInjectionRejected`. Present only for
|
|
526
|
+
* `'claim-rejected'`.
|
|
527
|
+
*/
|
|
528
|
+
readonly controllerReason?: string;
|
|
529
|
+
/** The controller's own message for the same condition. */
|
|
530
|
+
readonly controllerMessage?: string;
|
|
531
|
+
constructor(details: {
|
|
532
|
+
readonly reason: KubernetesAcquireFailureReason;
|
|
533
|
+
readonly retryable: boolean;
|
|
534
|
+
readonly message: string;
|
|
535
|
+
readonly controllerReason?: string;
|
|
536
|
+
readonly controllerMessage?: string;
|
|
537
|
+
readonly cause?: unknown;
|
|
538
|
+
});
|
|
539
|
+
}
|
|
540
|
+
/**
|
|
541
|
+
* The `status.conditions[Ready].reason` values that mean the controller has
|
|
542
|
+
* DECIDED, so waiting is pointless.
|
|
543
|
+
*
|
|
544
|
+
* ## Where these strings came from
|
|
545
|
+
*
|
|
546
|
+
* Not from the issue that asked for this, and not from upstream source: this
|
|
547
|
+
* repo vendors none of agent-sandbox's Go, so a literal copied out of a
|
|
548
|
+
* changelog is a literal nobody here can check. Each of the four was produced
|
|
549
|
+
* against the deployed controller (kind v1.37.0, agent-sandbox v1.0.2,
|
|
550
|
+
* 2026-09-17) by making the claim it describes and reading the condition
|
|
551
|
+
* back:
|
|
552
|
+
*
|
|
553
|
+
* | reason | how it was produced | the controller's message |
|
|
554
|
+
* |---|---|---|
|
|
555
|
+
* | `WarmPoolNotFound` | claim at a pool that does not exist | `SandboxWarmPool "…" not found` |
|
|
556
|
+
* | `TemplateNotFound` | claim at a pool whose template does not exist | `SandboxTemplate "…" not found` |
|
|
557
|
+
* | `InvalidMetadata` | claim with an `additionalPodMetadata` label outside the allowed domains | `invalid additionalPodMetadata: …` |
|
|
558
|
+
* | `EnvVarsInjectionRejected` | claim with `spec.env` against a template that forbids injection | `environment variable injection rejected: …` |
|
|
559
|
+
*
|
|
560
|
+
* The transient reasons seen on the SAME cluster, which must NOT be in this
|
|
561
|
+
* set, were `DependenciesNotReady` (pod exists, still Pending) and
|
|
562
|
+
* `DependenciesReady` (the Ready=True reason).
|
|
563
|
+
*
|
|
564
|
+
* ## Why an unknown reason is not terminal
|
|
565
|
+
*
|
|
566
|
+
* A wrong literal here fails in one of two ways, and only one of them is
|
|
567
|
+
* recoverable. Too few entries: a rejected claim waits out the readiness
|
|
568
|
+
* budget, which is exactly the behaviour every release before this one had.
|
|
569
|
+
* Too many: an acquire that would have succeeded is refused on a guess. So
|
|
570
|
+
* the set is a closed list of measured strings and everything else falls
|
|
571
|
+
* through to the deadline.
|
|
572
|
+
*/
|
|
573
|
+
export declare const TERMINAL_CLAIM_REASONS: readonly string[];
|
|
574
|
+
/**
|
|
575
|
+
* What this backend asked the controller to put on the claim's pod, carried
|
|
576
|
+
* into the rejection so an `InvalidMetadata` refusal can say what was sent
|
|
577
|
+
* and which knob changes it.
|
|
578
|
+
*
|
|
579
|
+
* Context, never a second decision: whether a claim is refused at all is
|
|
580
|
+
* {@link TERMINAL_CLAIM_REASONS}' answer and nobody else's, so an empty map
|
|
581
|
+
* changes nothing about the class thrown, the reason on it, or when it is
|
|
582
|
+
* raised.
|
|
583
|
+
*/
|
|
584
|
+
export interface ClaimPodMetadataContext {
|
|
585
|
+
readonly namespace: string;
|
|
586
|
+
/** `spec.additionalPodMetadata.labels`, exactly as sent. */
|
|
587
|
+
readonly requestedPodLabels: Readonly<Record<string, string>>;
|
|
588
|
+
/** The egress profile among those labels, when one is configured. */
|
|
589
|
+
readonly profile?: EgressProfileLabel;
|
|
590
|
+
}
|
|
591
|
+
/**
|
|
592
|
+
* A claim the controller has refused, or `undefined` for one it is still
|
|
593
|
+
* working on.
|
|
594
|
+
*
|
|
595
|
+
* `status: 'False'` alone is not a refusal — it is also what a claim looks
|
|
596
|
+
* like for the whole of a cold start — so the REASON decides, against
|
|
597
|
+
* {@link TERMINAL_CLAIM_REASONS}.
|
|
598
|
+
*
|
|
599
|
+
* ONE class comes out of here whatever the reason, and that is deliberate:
|
|
600
|
+
* `InvalidMetadata` is how the controller refuses a pod label whose domain is
|
|
601
|
+
* not on its allowlist — the profile's, today — and a host's `catch` must not
|
|
602
|
+
* have to be written differently depending on whether a profile happens to be
|
|
603
|
+
* configured. `metadata` only decides what rides along as the `cause`: a
|
|
604
|
+
* {@link KubernetesPodLabelsRejectedError} naming the map that was sent
|
|
605
|
+
* and the `config.egress.profileLabelKey` that moves it, which the
|
|
606
|
+
* controller's own message cannot know about.
|
|
607
|
+
*/
|
|
608
|
+
export declare function classifyClaimRejection(claim: SandboxClaimResource | undefined, claimName: string, metadata?: ClaimPodMetadataContext): KubernetesAcquireError | undefined;
|
|
609
|
+
export declare function retryDelayForApiFailure(err: unknown, pollIntervalMs: number): number | undefined;
|
|
610
|
+
/**
|
|
611
|
+
* The refusal a caller sees, given the failure that actually happened and
|
|
612
|
+
* whatever the pod had to say about it.
|
|
613
|
+
*
|
|
614
|
+
* Returns `undefined` for a failure none of the seven reasons describes —
|
|
615
|
+
* see {@link KubernetesAcquireError} for why that is a deliberate hole rather
|
|
616
|
+
* than a missing case. A {@link KubernetesAcquireError} that arrived from
|
|
617
|
+
* deeper in (the claim rejection) is returned unchanged: it is already the
|
|
618
|
+
* diagnosis.
|
|
619
|
+
*/
|
|
620
|
+
export declare function classifyAcquireFailure(err: unknown, podDiagnosis: 'capacity' | 'image-pull' | undefined): KubernetesAcquireError | undefined;
|
|
264
621
|
/**
|
|
265
622
|
* Claim or create, wait for Ready, read the bound identity back, resolve the
|
|
266
623
|
* address and learn the pod's uid — or leave nothing behind trying.
|
|
@@ -268,11 +625,98 @@ export declare function clientAccess(config: KubernetesBackendInternalConfig): K
|
|
|
268
625
|
* Exported because the sandbox surface is built on top of this record rather
|
|
269
626
|
* than beside it: one acquire path, one cleanup path, whatever ends up
|
|
270
627
|
* wrapping them.
|
|
628
|
+
*
|
|
629
|
+
* ## What it refuses with
|
|
630
|
+
*
|
|
631
|
+
* Every refusal this function can diagnose arrives as a
|
|
632
|
+
* {@link KubernetesAcquireError} naming one of seven reasons, with the
|
|
633
|
+
* original failure as its `cause`. Three things stay outside that:
|
|
634
|
+
* configuration refused before anything is created
|
|
635
|
+
* ({@link assertEnforceable}, {@link assertRuntimeClassIsApplicable}), a
|
|
636
|
+
* caller's own abort, and a failure none of the seven reasons honestly
|
|
637
|
+
* describes — see {@link KubernetesAcquireError} for why the last one is a
|
|
638
|
+
* hole on purpose.
|
|
639
|
+
*
|
|
640
|
+
* ## What it retries, and what it will not
|
|
641
|
+
*
|
|
642
|
+
* A readiness GET that fails transiently — a connect failure, a request the
|
|
643
|
+
* API bound gave up on, a 429, a 5xx — is repeated INSIDE the readiness
|
|
644
|
+
* deadline, honouring `Retry-After`. One clock, so a retry spends the budget
|
|
645
|
+
* rather than extending it, and a `create()` cannot outlive the timeout its
|
|
646
|
+
* caller chose. The create POST is never retried: it is not idempotent, and a
|
|
647
|
+
* POST whose answer never arrived may already have committed — which is why
|
|
648
|
+
* cleanup deletes the client-owned name whatever happened.
|
|
271
649
|
*/
|
|
272
650
|
export declare function acquireKubernetesSandbox(client: KubernetesClient, config: KubernetesBackendInternalConfig, options: SandboxBackendOptions, readiness: {
|
|
273
651
|
readonly timeoutMs: number;
|
|
274
652
|
readonly pollIntervalMs: number;
|
|
275
|
-
}): Promise<KubernetesAcquisition>;
|
|
653
|
+
}, verifyIngress?: IngressVerifier | undefined, egressBoundary?: KubernetesEgressBoundary | undefined): Promise<KubernetesAcquisition>;
|
|
654
|
+
/**
|
|
655
|
+
* Options for {@link releaseKubernetesTaskSandboxes}.
|
|
656
|
+
*/
|
|
657
|
+
export interface KubernetesReleaseTaskSandboxesOptions {
|
|
658
|
+
/**
|
|
659
|
+
* Required, and refused if empty — see {@link releaseKubernetesTaskSandboxes}.
|
|
660
|
+
* The same selector syntax a `kubectl get --selector` takes, e.g.
|
|
661
|
+
* `sandbox.namzu.ai/host-instance=host-a`.
|
|
662
|
+
*/
|
|
663
|
+
readonly labelSelector: string;
|
|
664
|
+
readonly signal?: AbortSignal;
|
|
665
|
+
}
|
|
666
|
+
/**
|
|
667
|
+
* Recover a crashed host's claims: LIST every `SandboxClaim` carrying
|
|
668
|
+
* `labelSelector`, `DELETE` each, and report what was removed.
|
|
669
|
+
*
|
|
670
|
+
* This deletes CLAIMS only. The controller's own garbage collection —
|
|
671
|
+
* ownerReferences from claim to the Sandbox it bound, and from Sandbox to
|
|
672
|
+
* Pod and Service — takes the rest down behind it; nothing here reads or
|
|
673
|
+
* touches a Sandbox or a Pod directly. A claim already gone (raced by the
|
|
674
|
+
* controller's own TTL reaper, or a second release call) counts as removed
|
|
675
|
+
* rather than a failure, the same convention every other DELETE in this
|
|
676
|
+
* backend follows.
|
|
677
|
+
*
|
|
678
|
+
* `labelSelector` is REQUIRED and refused, synchronously, before a single
|
|
679
|
+
* request goes out, if it is absent or empty: a release that could fall back
|
|
680
|
+
* to matching every claim (or every claim of the template) would delete a
|
|
681
|
+
* live fleet's work the first time a caller passed one by mistake. There is
|
|
682
|
+
* no default selector for exactly this reason.
|
|
683
|
+
*/
|
|
684
|
+
export declare function releaseKubernetesTaskSandboxes(config: KubernetesBackendInternalConfig, options: KubernetesReleaseTaskSandboxesOptions): Promise<{
|
|
685
|
+
readonly deleted: number;
|
|
686
|
+
readonly names: readonly string[];
|
|
687
|
+
}>;
|
|
688
|
+
/** Result of {@link readKubernetesTaskCapacity}. */
|
|
689
|
+
export interface KubernetesTaskCapacity {
|
|
690
|
+
/** `config.warmPoolName`'s own replica counts, straight off its `status`/`spec`. */
|
|
691
|
+
readonly warmPool: {
|
|
692
|
+
/** `status.readyReplicas`. `0` when the field is absent (a brand-new or empty pool). */
|
|
693
|
+
readonly ready: number;
|
|
694
|
+
/** `spec.replicas`. `0` when the field is absent. */
|
|
695
|
+
readonly desired: number;
|
|
696
|
+
};
|
|
697
|
+
/** Every `SandboxClaim` in the namespace bound to `config.warmPoolName`, whatever its state. */
|
|
698
|
+
readonly activeClaims: number;
|
|
699
|
+
/** Every Pod in the namespace currently in phase `Pending`. */
|
|
700
|
+
readonly pendingPods: number;
|
|
701
|
+
}
|
|
702
|
+
/** Options for {@link readKubernetesTaskCapacity}. */
|
|
703
|
+
export interface KubernetesReadTaskCapacityOptions {
|
|
704
|
+
readonly signal?: AbortSignal;
|
|
705
|
+
}
|
|
706
|
+
/**
|
|
707
|
+
* Read task-pool headroom before admitting more work: three GETs, no writes.
|
|
708
|
+
*
|
|
709
|
+
* `config.warmPoolName` is required — this reads the exact object a claim's
|
|
710
|
+
* `warmPoolRef` names, so a pool-less backend (every create is a direct
|
|
711
|
+
* Sandbox) has no pool to report on. `activeClaims` is every claim in the
|
|
712
|
+
* namespace whose `spec.warmPoolRef.name` matches this pool, counted rather
|
|
713
|
+
* than trusted from a label, because a claim's `warmPoolRef` is the one
|
|
714
|
+
* field the API itself guarantees. `pendingPods` is every Pod in the
|
|
715
|
+
* namespace still in phase `Pending` — a coarse but honest signal of
|
|
716
|
+
* in-flight scale-up the ready-replica count alone does not carry, on
|
|
717
|
+
* either the warm or the pool-less path.
|
|
718
|
+
*/
|
|
719
|
+
export declare function readKubernetesTaskCapacity(config: KubernetesBackendInternalConfig, options?: KubernetesReadTaskCapacityOptions): Promise<KubernetesTaskCapacity>;
|
|
276
720
|
/**
|
|
277
721
|
* What a directly created Sandbox copies out of a `SandboxTemplate`, and the
|
|
278
722
|
* two things it decides for itself.
|
|
@@ -292,6 +736,34 @@ export interface SandboxBodyOptions {
|
|
|
292
736
|
* sets it, because an unbounded task sandbox is a leak.
|
|
293
737
|
*/
|
|
294
738
|
readonly shutdownTime?: string;
|
|
739
|
+
/**
|
|
740
|
+
* Annotations to stamp on the Sandbox's OWN metadata at creation.
|
|
741
|
+
*
|
|
742
|
+
* One caller, and everything it writes is a fact the object has to carry
|
|
743
|
+
* from the moment it exists rather than from its first patch: the holder
|
|
744
|
+
* epoch of a workspace created under one, so there is no window in which
|
|
745
|
+
* it stands unfenced, and the revision of the pod template it was built
|
|
746
|
+
* from, so there is no window in which it claims none. Both are
|
|
747
|
+
* `workspace.ts`'s — see `HOLDER_EPOCH_ANNOTATION_KEY` and
|
|
748
|
+
* `POD_TEMPLATE_HASH_ANNOTATION_KEY`.
|
|
749
|
+
*
|
|
750
|
+
* Absent, the body is byte for byte what it always was, which is what
|
|
751
|
+
* keeps every task sandbox's create unchanged — a task sandbox is
|
|
752
|
+
* ephemeral, so it has no revision to drift from and nothing to fence.
|
|
753
|
+
*/
|
|
754
|
+
readonly annotations?: Readonly<Record<string, string>>;
|
|
755
|
+
/**
|
|
756
|
+
* Extra labels stamped onto the POD template's metadata, beside the
|
|
757
|
+
* template label this body always adds.
|
|
758
|
+
*
|
|
759
|
+
* The same map `buildClaimBody` puts on a claim's
|
|
760
|
+
* `additionalPodMetadata.labels`, from the same
|
|
761
|
+
* `composeAdditionalPodLabels` call — a direct Sandbox has no controller
|
|
762
|
+
* to merge them for it, so the create body stamps them itself and the two
|
|
763
|
+
* paths produce one set of pod labels. Absent or empty, the pod template
|
|
764
|
+
* is byte for byte what it was.
|
|
765
|
+
*/
|
|
766
|
+
readonly podLabels?: Readonly<Record<string, string>>;
|
|
295
767
|
}
|
|
296
768
|
/**
|
|
297
769
|
* The pool-less body. `Sandbox.spec` has no `templateRef` — only a
|
|
@@ -323,6 +795,46 @@ export interface SandboxBodyOptions {
|
|
|
323
795
|
* label the translated policy is built to match.
|
|
324
796
|
*/
|
|
325
797
|
export declare function buildSandboxBody(options: SandboxBodyOptions): Record<string, unknown>;
|
|
798
|
+
/**
|
|
799
|
+
* The `spec.podTemplate` a directly created Sandbox carries: the template's,
|
|
800
|
+
* with this backend's overlays — the template label and whatever
|
|
801
|
+
* {@link composeAdditionalPodLabels} produced ({@link sandboxPodLabels}), and
|
|
802
|
+
* the configured `runtimeClassName`.
|
|
803
|
+
*
|
|
804
|
+
* Its own function because it is now built twice: once into the create POST
|
|
805
|
+
* by {@link buildSandboxBody}, and once into the JSON Patch that refreshes a
|
|
806
|
+
* standing workspace's pod template (`workspace.ts`). Two expressions of the
|
|
807
|
+
* same overlay would drift, and the one that drifted would report a workspace
|
|
808
|
+
* as off-template forever — the hash under
|
|
809
|
+
* `sandbox.namzu.ai/pod-template-hash` is taken over exactly this object, so
|
|
810
|
+
* a second spelling is a second revision.
|
|
811
|
+
*
|
|
812
|
+
* `podLabels` is on this signature rather than only on the create body for
|
|
813
|
+
* exactly that reason. A refresh rewrites `/spec/podTemplate` WHOLE, so a
|
|
814
|
+
* refresh built without them would PATCH the egress profile off a pod
|
|
815
|
+
* template that carries it — the replacement pod would come up selected by no
|
|
816
|
+
* per-profile policy, on a path where nothing re-checks the label, and the
|
|
817
|
+
* revision stamped beside it would be taken over a template the POST never
|
|
818
|
+
* writes, so `templateCurrent` would report drift forever.
|
|
819
|
+
*/
|
|
820
|
+
export declare function sandboxPodTemplate(template: SandboxTemplateCopy, sandboxTemplateName: string, runtimeClassName?: string, podLabels?: Readonly<Record<string, string>>): SandboxPodTemplate;
|
|
821
|
+
/**
|
|
822
|
+
* The labels a directly created Sandbox's pod will carry: whatever the
|
|
823
|
+
* template declares, plus this backend's own template label, which always
|
|
824
|
+
* wins because it is the label a policy selector is built to match.
|
|
825
|
+
*
|
|
826
|
+
* Its own function because the ingress check has to reason about EXACTLY the
|
|
827
|
+
* labels {@link buildSandboxBody} stamps, before the POST that stamps them.
|
|
828
|
+
* Two expressions of the same rule would be one rename away from a check that
|
|
829
|
+
* verifies a pod nobody creates.
|
|
830
|
+
*
|
|
831
|
+
* `extra` is `composeAdditionalPodLabels`'s map — the egress profile today.
|
|
832
|
+
* It is applied LAST, and so wins over both, for the same reason the template
|
|
833
|
+
* label wins over the copied template's own: it is a label the translated
|
|
834
|
+
* policy's selector is built to match, and a pod that matched the selector
|
|
835
|
+
* only sometimes would be a boundary that applied only sometimes.
|
|
836
|
+
*/
|
|
837
|
+
export declare function sandboxPodLabels(template: SandboxTemplateCopy, sandboxTemplateName: string, extra?: Readonly<Record<string, string>>): Readonly<Record<string, string>>;
|
|
326
838
|
/** The two halves of a `SandboxTemplate` a directly created Sandbox copies. */
|
|
327
839
|
export interface SandboxTemplateCopy {
|
|
328
840
|
readonly podTemplate: SandboxPodTemplate;
|
|
@@ -333,27 +845,114 @@ export declare function readSandboxTemplate(client: KubernetesClient, namespace:
|
|
|
333
845
|
/**
|
|
334
846
|
* Where the agent answers.
|
|
335
847
|
*
|
|
336
|
-
* Its own function, and the Service FQDN wins over a
|
|
337
|
-
* address outlives the pod: a suspended-then-resumed
|
|
338
|
-
* new pod with a new IP behind the same name, and
|
|
339
|
-
* the name on every dial. A literal IP baked into a
|
|
340
|
-
* bug that would produce
|
|
848
|
+
* Its own function, and under the default mode the Service FQDN wins over a
|
|
849
|
+
* pod IP, because that address outlives the pod: a suspended-then-resumed
|
|
850
|
+
* workspace comes back as a new pod with a new IP behind the same name, and
|
|
851
|
+
* the transport re-resolves the name on every dial. A literal IP baked into a
|
|
852
|
+
* long-lived handle is the bug that would produce — and it is exactly the bug
|
|
853
|
+
* `'pod-ip'` accepts, deliberately, in exchange for an address a host outside
|
|
854
|
+
* the cluster can resolve at all. That mode pays for it by re-reading the IP
|
|
855
|
+
* on every resume and once after a failed connect.
|
|
856
|
+
*
|
|
857
|
+
* `'pod-ip'` takes the address from `pod`, the record the bind token was just
|
|
858
|
+
* read out of, and NEVER falls back to `binding.podIPs`. The Sandbox's status
|
|
859
|
+
* is a second source that can name a different pod — the one a resume is
|
|
860
|
+
* replacing — and an address from one pod with a token from another is the
|
|
861
|
+
* mismatch that arrives as a flat `unauthorized`.
|
|
341
862
|
*/
|
|
342
|
-
export declare function resolveAgentAddress(binding: KubernetesSandboxBinding, agentPort: number, token: string
|
|
863
|
+
export declare function resolveAgentAddress(binding: KubernetesSandboxBinding, agentPort: number, token: string, options?: {
|
|
864
|
+
readonly mode?: KubernetesAgentAddressMode;
|
|
865
|
+
/** The live pod the token came from. Required by `'pod-ip'`. */
|
|
866
|
+
readonly podIP?: string;
|
|
867
|
+
}): KubernetesAgentAddress;
|
|
343
868
|
/** Ready-or-not-yet, read off a `Sandbox`'s own status. Shared with `workspace.ts`. */
|
|
344
869
|
export declare function bindingFromSandbox(sandbox: SandboxResource | undefined): KubernetesSandboxBinding | undefined;
|
|
870
|
+
/**
|
|
871
|
+
* The readiness poll ran out of budget — and nothing else. Every OTHER
|
|
872
|
+
* failure {@link pollForBinding} meets is rethrown as itself, so this class
|
|
873
|
+
* is an exact answer to "was it the clock?", which a caller that has to
|
|
874
|
+
* choose between two timeout messages needs and cannot get from the clock.
|
|
875
|
+
*
|
|
876
|
+
* Reading `remainingMs()` after the fact is NOT that answer: the expiry timer
|
|
877
|
+
* and `performance.now()` are different clocks, and a timer that fires a
|
|
878
|
+
* fraction of a millisecond early leaves a positive remainder behind an
|
|
879
|
+
* expiry that has already happened.
|
|
880
|
+
*/
|
|
881
|
+
export declare class ReadinessPollTimeout extends Error {
|
|
882
|
+
readonly name = "ReadinessPollTimeout";
|
|
883
|
+
}
|
|
884
|
+
/**
|
|
885
|
+
* What a failed readiness read is worth, decided by the caller.
|
|
886
|
+
*
|
|
887
|
+
* Optional, and absent means exactly the behaviour every caller had before:
|
|
888
|
+
* the first failure of any kind ends the poll. `workspace.ts` passes nothing
|
|
889
|
+
* and is unchanged; the acquire path passes
|
|
890
|
+
* {@link retryDelayForApiFailure} so one 429 on a shared cluster no longer
|
|
891
|
+
* fails a create that had fifty-nine seconds of budget left.
|
|
892
|
+
*/
|
|
893
|
+
export interface ReadinessPollBehaviour {
|
|
894
|
+
/**
|
|
895
|
+
* Milliseconds to wait before reading again, or `undefined` to rethrow.
|
|
896
|
+
*
|
|
897
|
+
* The wait is spent on the SAME deadline as everything else in the poll,
|
|
898
|
+
* so a retry consumes the readiness budget and can never extend it. A hook
|
|
899
|
+
* that always returns a number therefore still terminates: the clock ends
|
|
900
|
+
* the loop, not the hook.
|
|
901
|
+
*/
|
|
902
|
+
readonly retryDelayFor?: (err: unknown) => number | undefined;
|
|
903
|
+
}
|
|
345
904
|
/**
|
|
346
905
|
* Poll until `read` reports a binding. `read` returns `undefined` for "not
|
|
347
906
|
* yet" and throws for a failure worth surfacing; the deadline owns every wait,
|
|
348
907
|
* including the sleep between attempts, so an expired clock cannot be extended
|
|
349
908
|
* by one more round trip. Shaped after ACI's `pollForRunningIp`.
|
|
909
|
+
*
|
|
910
|
+
* The only failure this raises on its own account is
|
|
911
|
+
* {@link ReadinessPollTimeout}; anything `read` throws travels out unchanged,
|
|
912
|
+
* unless `behaviour.retryDelayFor` claims it — in which case it is repeated
|
|
913
|
+
* inside the same budget and, if the budget then runs out, carried onto the
|
|
914
|
+
* timeout as its `cause`, so a poll that kept failing still says what it kept
|
|
915
|
+
* seeing.
|
|
350
916
|
*/
|
|
351
917
|
export declare function pollForBinding(read: (signal: AbortSignal) => Promise<KubernetesSandboxBinding | undefined>, deadline: OperationDeadline, readiness: {
|
|
352
918
|
readonly timeoutMs: number;
|
|
353
919
|
readonly pollIntervalMs: number;
|
|
354
|
-
}, label: string): Promise<KubernetesSandboxBinding>;
|
|
920
|
+
}, label: string, behaviour?: ReadinessPollBehaviour): Promise<KubernetesSandboxBinding>;
|
|
355
921
|
/**
|
|
356
|
-
* The
|
|
922
|
+
* The live pod behind a bound sandbox: its uid, and — read in the SAME
|
|
923
|
+
* answer — the IP it can be dialed at.
|
|
924
|
+
*
|
|
925
|
+
* One record rather than two reads because the two facts have to describe one
|
|
926
|
+
* pod. The uid is the agent's bind token and the IP is where that agent
|
|
927
|
+
* listens; taking them from separate GETs leaves a window in which a resume,
|
|
928
|
+
* an eviction or a node drain replaces the pod in between, and the handle
|
|
929
|
+
* then presents pod A's token at pod B's address. The guest answers that with
|
|
930
|
+
* a flat `unauthorized`, which says nothing about the race that caused it.
|
|
931
|
+
*/
|
|
932
|
+
export interface KubernetesBoundPod {
|
|
933
|
+
/** `metadata.uid` — the agent's bind token. */
|
|
934
|
+
readonly uid: string;
|
|
935
|
+
/** `status.podIP`. Read by `agentAddress: 'pod-ip'`; absent is legal. */
|
|
936
|
+
readonly podIP?: string;
|
|
937
|
+
/**
|
|
938
|
+
* `metadata.labels` — what an ingress policy's `podSelector` actually
|
|
939
|
+
* matches. Read off the SAME object the uid and the address come from,
|
|
940
|
+
* for the same reason they are: a policy decision made about one pod and
|
|
941
|
+
* a connection made to another is the mismatch this record exists to
|
|
942
|
+
* prevent. The claim path reads it to ask which policies select the bound
|
|
943
|
+
* pod — a directly created Sandbox's labels are known before its pod
|
|
944
|
+
* exists — and BOTH paths read it to confirm the pod really carries the
|
|
945
|
+
* labels this backend asked the controller for, which is the one thing
|
|
946
|
+
* knowing them in advance cannot establish. See `ingress-policy.ts` and
|
|
947
|
+
* `assertRequestedPodLabelsObserved`.
|
|
948
|
+
*/
|
|
949
|
+
readonly labels?: Readonly<Record<string, string>>;
|
|
950
|
+
}
|
|
951
|
+
/**
|
|
952
|
+
* Find the pod a sandbox is currently backed by, and read both facts off it.
|
|
953
|
+
*
|
|
954
|
+
* `uid` is the per-instance agent bind token; `podIP` is where that agent
|
|
955
|
+
* listens, and only `agentAddress: 'pod-ip'` reads it.
|
|
357
956
|
*
|
|
358
957
|
* The pod is named after its Sandbox in agent-sandbox v1.0.2 — verified
|
|
359
958
|
* against a running cluster — but that is an observation, not a documented
|
|
@@ -362,7 +961,53 @@ export declare function pollForBinding(read: (signal: AbortSignal) => Promise<Ku
|
|
|
362
961
|
* changing upstream is a second round trip through `status.selector`, which is
|
|
363
962
|
* exactly what the controller publishes the selector for.
|
|
364
963
|
*/
|
|
365
|
-
export declare function
|
|
964
|
+
export declare function readBoundPod(client: KubernetesClient, namespace: string, binding: KubernetesSandboxBinding, signal?: AbortSignal): Promise<KubernetesBoundPod>;
|
|
965
|
+
/**
|
|
966
|
+
* {@link readBoundPod}, plus — under `'pod-ip'` only — the wait for an
|
|
967
|
+
* address to go with the token.
|
|
968
|
+
*
|
|
969
|
+
* Under the default mode this is the single read it has always been: one
|
|
970
|
+
* `GET`, in the same place in the same order, because a Service FQDN is
|
|
971
|
+
* published with the Sandbox and needs nothing from the pod but its uid.
|
|
972
|
+
*
|
|
973
|
+
* `'pod-ip'` has to wait, because a LIVE pod is not yet an ADDRESSED pod. A
|
|
974
|
+
* pod is created `Pending` and carries no `status.podIP` until the CNI has
|
|
975
|
+
* finished attaching it, and {@link isPodLive} accepts `Pending` on purpose —
|
|
976
|
+
* the resume path in `workspace.ts` binds its replacement pod long before that
|
|
977
|
+
* pod is Ready, because `Ready` stays True across the transition and the uid
|
|
978
|
+
* is the only transition signal there is. Refusing an address-less pod outright
|
|
979
|
+
* would therefore fail on the NORMAL path, in milliseconds, with the whole
|
|
980
|
+
* readiness budget unspent. So "live, no address yet" is polled on the same
|
|
981
|
+
* deadline as everything else on this path, and
|
|
982
|
+
* {@link resolveAgentAddress}'s own refusal is left as the post-deadline
|
|
983
|
+
* backstop for a pod that never gets an address at all.
|
|
984
|
+
*
|
|
985
|
+
* A failed READ stays fatal, exactly as it was: this is the acquire path,
|
|
986
|
+
* where nothing is being replaced and a pod that cannot be read is not a pod
|
|
987
|
+
* that is about to appear.
|
|
988
|
+
*
|
|
989
|
+
* `requiredLabels` is the second thing worth waiting for, and it is waited
|
|
990
|
+
* for in the SAME loop rather than in a second one: an egress profile is a
|
|
991
|
+
* label the CONTROLLER patches onto the pod it binds, so a pod read the
|
|
992
|
+
* instant it was bound can be live, addressed and not yet labelled. Two
|
|
993
|
+
* loops would be two deadlines and two answers to "is this pod ready to be
|
|
994
|
+
* admitted". Like the address, an expired clock hands the pod back as it is
|
|
995
|
+
* — the caller decides whether a missing label is fatal, and on the acquire
|
|
996
|
+
* path it is: see `assertRequestedPodLabelsObserved`.
|
|
997
|
+
*/
|
|
998
|
+
export declare function readAddressedPod(client: KubernetesClient, namespace: string, binding: KubernetesSandboxBinding, deadline: OperationDeadline, readiness: {
|
|
999
|
+
readonly pollIntervalMs: number;
|
|
1000
|
+
}, mode: KubernetesAgentAddressMode, requiredLabels?: Readonly<Record<string, string>>): Promise<KubernetesBoundPod>;
|
|
1001
|
+
/**
|
|
1002
|
+
* The re-read a `'pod-ip'` handle follows a replaced pod with: one live-pod
|
|
1003
|
+
* read, then the same address resolution acquire did.
|
|
1004
|
+
*
|
|
1005
|
+
* Built here rather than inside the transport because finding the pod is a
|
|
1006
|
+
* CONTROL-plane act — the by-name GET, the selector fallback, the
|
|
1007
|
+
* liveness filter — and the transport owns none of that. It is handed over as
|
|
1008
|
+
* a closure so the transport can call it without learning what a Sandbox is.
|
|
1009
|
+
*/
|
|
1010
|
+
export declare function buildAgentAddressRefresh(client: KubernetesClient, namespace: string, binding: KubernetesSandboxBinding, agentPort: number, mode: KubernetesAgentAddressMode): (signal?: AbortSignal) => Promise<KubernetesAgentAddress>;
|
|
366
1011
|
/**
|
|
367
1012
|
* Run the probe against a built Sandbox, on its own clock, and throw if it
|
|
368
1013
|
* refuses. Cleanup is the CALLER's, and the two callers want opposite things:
|