@namzu/sandbox 13.0.0 → 15.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (104) hide show
  1. package/CHANGELOG.md +1147 -0
  2. package/README.md +447 -0
  3. package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
  4. package/dist/backends/aci-standby-pool/index.js +13 -1
  5. package/dist/backends/aci-standby-pool/index.js.map +1 -1
  6. package/dist/backends/docker/index.d.ts.map +1 -1
  7. package/dist/backends/docker/index.js +19 -1
  8. package/dist/backends/docker/index.js.map +1 -1
  9. package/dist/backends/firecracker/index.d.ts.map +1 -1
  10. package/dist/backends/firecracker/index.js +12 -2
  11. package/dist/backends/firecracker/index.js.map +1 -1
  12. package/dist/backends/firecracker/protocol.d.ts +481 -8
  13. package/dist/backends/firecracker/protocol.d.ts.map +1 -1
  14. package/dist/backends/firecracker/protocol.js +136 -0
  15. package/dist/backends/firecracker/protocol.js.map +1 -1
  16. package/dist/backends/firecracker/transport.d.ts +642 -14
  17. package/dist/backends/firecracker/transport.d.ts.map +1 -1
  18. package/dist/backends/firecracker/transport.js +1307 -34
  19. package/dist/backends/firecracker/transport.js.map +1 -1
  20. package/dist/backends/kubernetes/egress-policy.d.ts +1296 -0
  21. package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -0
  22. package/dist/backends/kubernetes/egress-policy.js +2458 -0
  23. package/dist/backends/kubernetes/egress-policy.js.map +1 -0
  24. package/dist/backends/kubernetes/identity.d.ts +193 -0
  25. package/dist/backends/kubernetes/identity.d.ts.map +1 -0
  26. package/dist/backends/kubernetes/identity.js +147 -0
  27. package/dist/backends/kubernetes/identity.js.map +1 -0
  28. package/dist/backends/kubernetes/index.d.ts +1019 -0
  29. package/dist/backends/kubernetes/index.d.ts.map +1 -0
  30. package/dist/backends/kubernetes/index.js +1756 -0
  31. package/dist/backends/kubernetes/index.js.map +1 -0
  32. package/dist/backends/kubernetes/ingress-policy.d.ts +375 -0
  33. package/dist/backends/kubernetes/ingress-policy.d.ts.map +1 -0
  34. package/dist/backends/kubernetes/ingress-policy.js +1050 -0
  35. package/dist/backends/kubernetes/ingress-policy.js.map +1 -0
  36. package/dist/backends/kubernetes/k8s-client.d.ts +334 -0
  37. package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -0
  38. package/dist/backends/kubernetes/k8s-client.js +553 -0
  39. package/dist/backends/kubernetes/k8s-client.js.map +1 -0
  40. package/dist/backends/kubernetes/lease.d.ts +145 -0
  41. package/dist/backends/kubernetes/lease.d.ts.map +1 -0
  42. package/dist/backends/kubernetes/lease.js +201 -0
  43. package/dist/backends/kubernetes/lease.js.map +1 -0
  44. package/dist/backends/kubernetes/objects.d.ts +702 -0
  45. package/dist/backends/kubernetes/objects.d.ts.map +1 -0
  46. package/dist/backends/kubernetes/objects.js +518 -0
  47. package/dist/backends/kubernetes/objects.js.map +1 -0
  48. package/dist/backends/kubernetes/per-sandbox-policy.d.ts +219 -0
  49. package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -0
  50. package/dist/backends/kubernetes/per-sandbox-policy.js +407 -0
  51. package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -0
  52. package/dist/backends/kubernetes/privilege-probe.d.ts +136 -0
  53. package/dist/backends/kubernetes/privilege-probe.d.ts.map +1 -0
  54. package/dist/backends/kubernetes/privilege-probe.js +185 -0
  55. package/dist/backends/kubernetes/privilege-probe.js.map +1 -0
  56. package/dist/backends/kubernetes/rbac.d.ts +153 -0
  57. package/dist/backends/kubernetes/rbac.d.ts.map +1 -0
  58. package/dist/backends/kubernetes/rbac.js +177 -0
  59. package/dist/backends/kubernetes/rbac.js.map +1 -0
  60. package/dist/backends/kubernetes/sandbox.d.ts +190 -0
  61. package/dist/backends/kubernetes/sandbox.d.ts.map +1 -0
  62. package/dist/backends/kubernetes/sandbox.js +433 -0
  63. package/dist/backends/kubernetes/sandbox.js.map +1 -0
  64. package/dist/backends/kubernetes/transport.d.ts +1048 -0
  65. package/dist/backends/kubernetes/transport.d.ts.map +1 -0
  66. package/dist/backends/kubernetes/transport.js +2093 -0
  67. package/dist/backends/kubernetes/transport.js.map +1 -0
  68. package/dist/backends/kubernetes/workspace.d.ts +1512 -0
  69. package/dist/backends/kubernetes/workspace.d.ts.map +1 -0
  70. package/dist/backends/kubernetes/workspace.js +3703 -0
  71. package/dist/backends/kubernetes/workspace.js.map +1 -0
  72. package/dist/backends/remote-execution-controller.d.ts +14 -0
  73. package/dist/backends/remote-execution-controller.d.ts.map +1 -1
  74. package/dist/backends/remote-execution-controller.js.map +1 -1
  75. package/dist/index.d.ts +350 -2
  76. package/dist/index.d.ts.map +1 -1
  77. package/dist/index.js +344 -34
  78. package/dist/index.js.map +1 -1
  79. package/dist/testing/sandbox-conformance.d.ts +227 -0
  80. package/dist/testing/sandbox-conformance.d.ts.map +1 -0
  81. package/dist/testing/sandbox-conformance.js +896 -0
  82. package/dist/testing/sandbox-conformance.js.map +1 -0
  83. package/package.json +5 -4
  84. package/src/backends/aci-standby-pool/index.ts +16 -1
  85. package/src/backends/docker/index.ts +22 -1
  86. package/src/backends/firecracker/index.ts +14 -2
  87. package/src/backends/firecracker/protocol.ts +541 -6
  88. package/src/backends/firecracker/transport.ts +1687 -64
  89. package/src/backends/kubernetes/egress-policy.ts +3448 -0
  90. package/src/backends/kubernetes/identity.ts +261 -0
  91. package/src/backends/kubernetes/index.ts +2670 -0
  92. package/src/backends/kubernetes/ingress-policy.ts +1344 -0
  93. package/src/backends/kubernetes/k8s-client.ts +742 -0
  94. package/src/backends/kubernetes/lease.ts +254 -0
  95. package/src/backends/kubernetes/objects.ts +983 -0
  96. package/src/backends/kubernetes/per-sandbox-policy.ts +542 -0
  97. package/src/backends/kubernetes/privilege-probe.ts +261 -0
  98. package/src/backends/kubernetes/rbac.ts +192 -0
  99. package/src/backends/kubernetes/sandbox.ts +593 -0
  100. package/src/backends/kubernetes/transport.ts +2895 -0
  101. package/src/backends/kubernetes/workspace.ts +5640 -0
  102. package/src/backends/remote-execution-controller.ts +14 -0
  103. package/src/index.ts +838 -35
  104. package/src/testing/sandbox-conformance.ts +1202 -0
@@ -0,0 +1,1019 @@
1
+ /**
2
+ * Kubernetes / agent-sandbox backend — acquire, readiness and teardown.
3
+ *
4
+ * Sibling of `firecracker/` and `aci-standby-pool/`: same
5
+ * {@link SandboxBackend} surface, same "the host supplies the credential
6
+ * callback and this package carries no cloud SDK" boundary, a different
7
+ * control plane. Here the control plane is the Kubernetes API server and the
8
+ * warm pool is a `SandboxWarmPool` reconciled by the agent-sandbox controller
9
+ * (kubernetes-sigs/agent-sandbox v1.0.2).
10
+ *
11
+ * Registered as `microvm` because {@link SandboxTier} names the strength of
12
+ * the boundary, not the orchestrator that starts it: a pod scheduled onto a
13
+ * Kata RuntimeClass runs in a hardware-virtualized guest. The tier also keeps
14
+ * the backend out of the container tier's mandatory
15
+ * `ContainerSandboxLayout`, which a remote-copy backend has no use for —
16
+ * Firecracker took the same exemption.
17
+ *
18
+ * ## Two acquire paths, and why they create different kinds
19
+ *
20
+ * - `warmPoolName` set → POST a `SandboxClaim` at the named pool. The
21
+ * controller binds an already-running Sandbox out of the pool, which is
22
+ * what makes a sub-second acquire possible at all.
23
+ * - `warmPoolName` unset → POST a `Sandbox` directly. This is necessitated
24
+ * rather than chosen: `SandboxClaimSpec.warmPoolRef` is a REQUIRED field,
25
+ * so a pool-less claim does not exist in the API.
26
+ *
27
+ * The claim this backend POSTs is PRISTINE: `warmPoolRef` and a lifecycle
28
+ * bound, nothing else. `spec.env` and `spec.volumeClaimTemplates` are
29
+ * available on the claim and are never set, because a claim carrying either
30
+ * is forced to cold-start instead of adopting a pool sandbox — the single
31
+ * most expensive mistake available on this path, and a silent one, since such
32
+ * a claim still works and only the latency shows it. Per-sandbox controls
33
+ * that would need those fields are refused by {@link assertEnforceable}
34
+ * rather than accepted and dropped.
35
+ *
36
+ * ## The bound sandbox is not named after the claim
37
+ *
38
+ * A pool sandbox keeps the generated name the pool gave it when the claim
39
+ * adopts it. The backend therefore reads the bound identity back out of
40
+ * `status.sandbox` and never derives it from the claim's own name. A test
41
+ * covers exactly that asymmetry.
42
+ *
43
+ * ## Credential
44
+ *
45
+ * The per-instance agent bind token is the backing pod's own
46
+ * `metadata.uid`: the host learns it from the API server after readiness, the
47
+ * guest learns it through the downward API (`NAMZU_AGENT_BIND_TOKEN`), and
48
+ * nothing has to be minted, stored or injected at claim time. One GET, no
49
+ * claim mutation, so the warm path stays pristine. A resumed pod is a new pod
50
+ * with a new uid, which is correct: it is a new instance.
51
+ *
52
+ * ## The lease
53
+ *
54
+ * The `shutdownTime` acquire stamps is the leak guard AND, unrenewed, a
55
+ * deadline on the run. The Sandbox handle owns a renewal loop that PATCHes
56
+ * it forward every half-TTL for as long as the handle is alive, and
57
+ * `destroy()` stops it — so a live run keeps its pod and a dead host still
58
+ * costs the cluster exactly one expiry. See `lease.ts`.
59
+ *
60
+ * ## The privilege probe
61
+ *
62
+ * `create()` does not resolve until the guest has reported — and this
63
+ * backend has checked — that it is deprivileged: all four capability masks
64
+ * zero, `NoNewPrivs: 1`, read out of `/proc/self/status` over the agent's
65
+ * `execute` op, on a clock of its own so a guest that goes quiet is refused
66
+ * rather than waited on. A refusal destroys the instance and rejects, so no
67
+ * handle to an under-hardened sandbox escapes. There is no off switch. See
68
+ * `privilege-probe.ts`.
69
+ *
70
+ * ## Not watch
71
+ *
72
+ * Readiness is polled against the shared {@link OperationDeadline}, exactly as
73
+ * ACI polls `provisioningState`. A watch would buy nothing on a path whose
74
+ * whole budget is under a second, and would cost resourceVersion tracking,
75
+ * bookmarks, 410-relist and reconnect backoff.
76
+ *
77
+ * ## Egress
78
+ *
79
+ * `config.egress` is optional and, when set, translated and VERIFIED — never
80
+ * created — by `egress-policy.ts`, in two steps. The NAMED object is GETted
81
+ * and compared to the translation exactly, once, lazily, on the first
82
+ * `create()`, so `buildKubernetesBackend` itself still contacts nothing.
83
+ * Then, because the API server UNIONS every policy selecting a pod, every
84
+ * `NetworkPolicy` in the namespace (and, under `engine: 'cilium'`, every
85
+ * `CiliumNetworkPolicy`) is enumerated against the pod's real labels and the
86
+ * create is refused when any of them lets out more than the translation does
87
+ * — a second policy widens egress however exactly the named one matches, and
88
+ * a `SandboxTemplate`'s own `networkPolicy` block becomes exactly such a
89
+ * policy. `egress.verify: 'named-object-only'` is the opt-out and restores
90
+ * the first step alone. Every Sandbox this file creates directly
91
+ * (`buildSandboxBody`) carries {@link sandboxTemplateLabel} on its
92
+ * podTemplate specifically so that translated policy's `podSelector` has
93
+ * something stable to match — see `objects.ts`'s doc comment on that label
94
+ * for why agent-sandbox's own controller-owned label does not cover this
95
+ * path.
96
+ *
97
+ * ## Ingress
98
+ *
99
+ * `config.ingress` is the opposite default: verification is ON unless a
100
+ * deployment says `'unverified'`. Before the POST for a direct Sandbox, and
101
+ * after the bind for a claimed one, `ingress-policy.ts` lists the namespace's
102
+ * policies and refuses unless one of them actually closes the agent port on
103
+ * the labels this pod carries. Nothing checked that before, while three
104
+ * pieces of shipped text said it was covered — see that module's own doc
105
+ * comment for what was measured.
106
+ */
107
+ import type { Sandbox } from '@namzu/sdk';
108
+ import type { SandboxBackend, SandboxBackendOptions } from '../../index.js';
109
+ import { OperationDeadline } from '../readiness.js';
110
+ import { type EgressProfileLabel, type KubernetesEgressConfig } from './egress-policy.js';
111
+ import { type KubernetesIngressConfig } from './ingress-policy.js';
112
+ import { type KubernetesAccess, type KubernetesClient, type KubernetesClientOptions } from './k8s-client.js';
113
+ import { type SandboxClaimResource, type SandboxPodTemplate, type SandboxResource, type SandboxVolumeClaimTemplate } from './objects.js';
114
+ import { type PerSandboxPolicyOwner } from './per-sandbox-policy.js';
115
+ export type { KubernetesEgressConfig, KubernetesEgressEngine, KubernetesEgressPolicy, KubernetesEgressVerification, KubernetesOnlyEgressPolicy, KubernetesPerSandboxEgressConfig, } from './egress-policy.js';
116
+ export type { KubernetesIngressConfig, KubernetesIngressEngine } from './ingress-policy.js';
117
+ /**
118
+ * How the backend reaches the API server. Two sources, neither needing a YAML
119
+ * parser — see `k8s-client.ts` for the reasoning.
120
+ */
121
+ export type KubernetesClusterAccess = {
122
+ readonly inCluster: true;
123
+ } | {
124
+ readonly inCluster?: false;
125
+ readonly server: string;
126
+ readonly ca?: string | Buffer;
127
+ readonly getToken: () => Promise<string>;
128
+ };
129
+ export interface KubernetesBackendInternalConfig {
130
+ readonly access: KubernetesClusterAccess;
131
+ /** Namespace the claims, sandboxes and pods live in. */
132
+ readonly namespace: string;
133
+ /**
134
+ * SandboxTemplate whose `podTemplate` a POOL-LESS create copies into the
135
+ * Sandbox it posts. The warm path never reads it — the pool's own
136
+ * `sandboxTemplateRef` decides there — but an operator reading this config
137
+ * still learns which template these sandboxes are built from.
138
+ */
139
+ readonly sandboxTemplateName: string;
140
+ /** Named `SandboxWarmPool`. Absent → every create is a direct Sandbox. */
141
+ readonly warmPoolName?: string;
142
+ /** TCP port the guest agent listens on. Default {@link DEFAULT_AGENT_PORT}. */
143
+ readonly agentPort?: number;
144
+ /**
145
+ * Which of a sandbox's two addresses the transport dials. Default
146
+ * `'service'` — see {@link KubernetesAgentAddressMode}, which is where
147
+ * the choice is explained, because it is a fact about where the HOST
148
+ * runs rather than about the cluster.
149
+ */
150
+ readonly agentAddress?: KubernetesAgentAddressMode;
151
+ readonly readyPollIntervalMs?: number;
152
+ readonly readyTimeoutMs?: number;
153
+ /**
154
+ * Lifetime bound written into every created object, and the amount each
155
+ * lease renewal pushes the expiry forward. Default 1 hour.
156
+ */
157
+ readonly claimTtlSeconds?: number;
158
+ /**
159
+ * Every lease-renewal failure that is not "the object is already gone".
160
+ * `@namzu/sandbox` owns no logger and reads none from module scope, so a
161
+ * diagnostic it cannot print is handed to the host that can. Renewal
162
+ * retries on the next tick either way; nothing here changes behaviour.
163
+ */
164
+ readonly onLeaseRenewalError?: (error: unknown) => void;
165
+ /**
166
+ * RuntimeClass for a POOL-LESS create. Refused together with
167
+ * `warmPoolName`: a pooled sandbox's runtime class is fixed by the pool's
168
+ * SandboxTemplate and cannot be chosen per claim.
169
+ */
170
+ readonly runtimeClassName?: string;
171
+ /**
172
+ * Egress policy this backend's `NetworkPolicy` (or `CiliumNetworkPolicy`,
173
+ * under `engine: 'cilium'`) is expected to carry. Unset means this backend
174
+ * neither computes nor verifies one, and OUTBOUND traffic is then whatever
175
+ * the cluster's own policies happen to allow.
176
+ *
177
+ * Deliberately not described as "covered by the SandboxTemplate's managed
178
+ * NetworkPolicy", which is what this comment used to claim: that policy
179
+ * selects `agents.x-k8s.io/sandbox-template-ref-hash`, a label the
180
+ * controller writes only onto a Sandbox adopted out of a `SandboxWarmPool`
181
+ * and never onto one this backend POSTs — so for every pool-less sandbox
182
+ * and every workspace it selects nothing at all. See `ingress-policy.ts`.
183
+ */
184
+ readonly egress?: KubernetesEgressConfig;
185
+ /**
186
+ * Whether the agent port's INGRESS boundary is verified before a sandbox
187
+ * is created, and against which policy resources.
188
+ *
189
+ * Unset means VERIFY — the one field in this config whose absent value is
190
+ * the strict one, because the deployment that needs the check is the one
191
+ * that would never have set it. `'unverified'` reads no policy and issues
192
+ * no request, for a deployment whose boundary lives somewhere a namespaced
193
+ * Role cannot see. See `ingress-policy.ts`.
194
+ */
195
+ readonly ingress?: KubernetesIngressConfig;
196
+ /**
197
+ * Bound on every single Kubernetes API request this backend sends. See
198
+ * {@link KubernetesClientOptions.requestTimeoutMs} in `k8s-client.ts`,
199
+ * which owns the default (30 s), the floor (1 s) and the reason there is
200
+ * no value that disables it.
201
+ */
202
+ readonly apiRequestTimeoutMs?: number;
203
+ /**
204
+ * Interval of the negotiated liveness heartbeat on `openTerminal` and
205
+ * `openTcpConnection` streams. Default
206
+ * {@link DEFAULT_STREAM_HEARTBEAT_MS}; `0` turns it off and restores the
207
+ * pre-heartbeat behaviour exactly. See {@link resolveStreamHeartbeatMs}.
208
+ */
209
+ readonly streamHeartbeatMs?: number;
210
+ /**
211
+ * Extra labels written onto every `SandboxClaim` this backend POSTs —
212
+ * `metadata.labels`, and nowhere else. Never merged into
213
+ * `additionalPodMetadata`: those are POD labels a running Sandbox and its
214
+ * `NetworkPolicy` selectors read, and a host's own recovery bookkeeping
215
+ * has no business changing what a pod is selected by. Absent means no
216
+ * labels beyond what the controller itself writes, and every claim body
217
+ * this backend sends is byte-for-byte what it always was.
218
+ *
219
+ * The intended use is a host-instance identity — e.g.
220
+ * `{ 'sandbox.namzu.ai/host-instance': hostId }` — so a restarted host
221
+ * can find and {@link releaseKubernetesTaskSandboxes} its predecessor's
222
+ * claims well before `claimTtlSeconds` would reap them on its own.
223
+ */
224
+ readonly claimLabels?: Record<string, string>;
225
+ }
226
+ /**
227
+ * Default {@link KubernetesBackendInternalConfig.streamHeartbeatMs} — 15 s,
228
+ * so a stream whose peer vanished without a FIN or an RST is given up on
229
+ * within 45 s rather than never.
230
+ *
231
+ * The Kubernetes backend opts IN here; `VsockTransportOptions.heartbeatMs`
232
+ * stays undefined by default, because that transport is shared with the
233
+ * Firecracker tier and a default there would force-close an existing
234
+ * consumer's quiet-but-alive terminal.
235
+ */
236
+ export declare const DEFAULT_STREAM_HEARTBEAT_MS = 15000;
237
+ /**
238
+ * Validate the configured stream-heartbeat interval, or supply the default.
239
+ *
240
+ * `0` IS accepted here, unlike `apiRequestTimeoutMs`: the heartbeat is a new
241
+ * capability that a deployment behind a middlebox with its own idea about
242
+ * unexpected frames may want off, and turning it off restores exactly the
243
+ * behaviour every release before this one had. An unanswered API request has
244
+ * no such prior behaviour worth restoring.
245
+ */
246
+ export declare function resolveStreamHeartbeatMs(value: number | undefined): number;
247
+ /**
248
+ * The client options every `createKubernetesClient` call in this backend is
249
+ * built with. One function so the five call sites — the provider, and each
250
+ * of the workspace verbs — cannot drift apart on which bounds they honour.
251
+ */
252
+ export declare function clientOptions(config: KubernetesBackendInternalConfig): KubernetesClientOptions;
253
+ /**
254
+ * Which address a sandbox's agent is dialed at. A property of where the HOST
255
+ * runs, not of the cluster.
256
+ *
257
+ * - `'service'` (default) — the Sandbox's `status.serviceFQDN`,
258
+ * `<name>.<namespace>.svc.cluster.local`. It outlives the pod: a resumed
259
+ * workspace comes back behind the same name, and every dial re-resolves
260
+ * it. The catch is that only the cluster's own DNS answers it, so this is
261
+ * correct exactly when the host itself runs inside the cluster.
262
+ * - `'pod-ip'` — the bound pod's IP, read from the same `GET` that reads
263
+ * its uid, so the address and the bind token are always one pod's. For a
264
+ * host OUTSIDE the cluster on a routable pod network (a peered VNet, a
265
+ * node-local operator, a CI runner with a route): it needs no cluster
266
+ * resolver at all. The cost is that a pod IP dies with its pod, which is
267
+ * why a resume re-reads it and a connect failure re-reads it once.
268
+ *
269
+ * The cluster still decides whether either address is REACHABLE: `'pod-ip'`
270
+ * additionally needs a NetworkPolicy that admits the host's own address
271
+ * range on the agent port. Neither mode changes the bind token, the
272
+ * privilege probe or egress verification.
273
+ */
274
+ export type KubernetesAgentAddressMode = 'service' | 'pod-ip';
275
+ /**
276
+ * The address the guest agent answers on, plus the token to present.
277
+ *
278
+ * Structurally the `tcp` arm of the transport's `SandboxAgentHandle`. It is
279
+ * declared here rather than imported so acquire does not depend on the
280
+ * transport landing first; the two are asserted equal where they meet.
281
+ */
282
+ export interface KubernetesAgentAddress {
283
+ readonly kind: 'tcp';
284
+ readonly host: string;
285
+ readonly port: number;
286
+ readonly token: string;
287
+ }
288
+ /** What the controller bound, read back off the object's own status. */
289
+ export interface KubernetesSandboxBinding {
290
+ /** The Sandbox's own name — NOT the claim's. */
291
+ readonly name: string;
292
+ readonly podIPs?: readonly string[];
293
+ readonly serviceFQDN?: string;
294
+ /** `Sandbox.status.selector`, when the path that read it had it for free. */
295
+ readonly podSelector?: string;
296
+ }
297
+ /** One acquired sandbox: what it is, where it answers, how to give it back. */
298
+ export interface KubernetesAcquisition {
299
+ readonly binding: KubernetesSandboxBinding;
300
+ readonly agent: KubernetesAgentAddress;
301
+ /**
302
+ * Re-read the live pod and recompute {@link agent} from it. Present only
303
+ * under `agentAddress: 'pod-ip'`, where the address is a literal that
304
+ * dies with its pod; a Service FQDN needs no such thing, and handing one
305
+ * over anyway would give the default mode a re-read it never had.
306
+ */
307
+ readonly refreshAgent?: (signal?: AbortSignal) => Promise<KubernetesAgentAddress>;
308
+ /** API path of the object THIS backend created — the claim, or the Sandbox. */
309
+ readonly ownedPath: string;
310
+ /**
311
+ * The object this backend created, identified well enough to be named in
312
+ * another object's `ownerReferences`: its kind, its name and the uid the
313
+ * API server assigned it.
314
+ *
315
+ * Present only when `config.egress.perSandbox` is configured, because it
316
+ * is read from the create reply and the readiness polls and nothing else
317
+ * needs it — a deployment that never writes a per-sandbox policy should
318
+ * not start carrying a field whose absence would otherwise be a bug.
319
+ */
320
+ readonly owner?: PerSandboxPolicyOwner;
321
+ /**
322
+ * The VALUE of the per-sandbox selector label on this sandbox's pod —
323
+ * the created object's name, confirmed on the bound pod before the
324
+ * sandbox was admitted. Present under the same condition as
325
+ * {@link owner}.
326
+ */
327
+ readonly perSandboxLabelValue?: string;
328
+ /** The TTL acquire stamped, which every renewal re-stamps. */
329
+ readonly ttlSeconds: number;
330
+ /** DELETE that object. An already-gone object counts as released. */
331
+ release(signal?: AbortSignal): Promise<void>;
332
+ /**
333
+ * Merge-PATCH the created object's expiry forward to `shutdownTime`.
334
+ *
335
+ * On the acquisition rather than in `lease.ts` because only this path
336
+ * knows WHICH object it created and therefore where the field lives: a
337
+ * `SandboxClaim` carries it at `spec.lifecycle.shutdownTime`, a directly
338
+ * created `Sandbox` at `spec.shutdownTime` (v1beta1 as served keeps it at
339
+ * the top of `spec`). A merge patch of the nested object leaves
340
+ * `shutdownPolicy` alone.
341
+ */
342
+ renew(shutdownTime: string, signal?: AbortSignal): Promise<void>;
343
+ }
344
+ /**
345
+ * The same number the Firecracker guest agent listens on over vsock
346
+ * (`DEFAULT_AGENT_VSOCK_PORT`), so one agent has one port across both tiers
347
+ * and a manifest, a NetworkPolicy and a transport can all name it from
348
+ * memory. Unprivileged, and the sandbox pod is not sharing it with anything.
349
+ */
350
+ export declare const DEFAULT_AGENT_PORT = 1024;
351
+ /**
352
+ * How long the privilege probe may take, given the caller's readiness budget.
353
+ *
354
+ * `readyTimeoutMs` bounds the CONTROL plane and has usually expired by the
355
+ * time the probe starts, so the probe cannot share it — but it is still the
356
+ * number the caller chose to describe how long an acquire may take, so the
357
+ * probe is allowed exactly that much again and no more, capped. A caller who
358
+ * asked for a 500 ms acquire gets a 500 ms probe; one who asked for two
359
+ * minutes of cold start still gets {@link PRIVILEGE_PROBE_TIMEOUT_CAP_MS}.
360
+ * Deliberately not a separate config key: a knob whose only correct value is
361
+ * "long enough for one `cat`" is a knob that only ever gets set wrong.
362
+ */
363
+ export declare function resolveProbeTimeoutMs(readyTimeoutMs: number): number;
364
+ /**
365
+ * The readiness bounds every path in this backend polls against — acquire,
366
+ * and `workspace.ts`'s create/suspend/resume. One function so the two cannot
367
+ * drift apart on defaults.
368
+ */
369
+ export declare function resolveKubernetesReadiness(config: {
370
+ readonly readyTimeoutMs?: number;
371
+ readonly readyPollIntervalMs?: number;
372
+ }): {
373
+ readonly timeoutMs: number;
374
+ readonly pollIntervalMs: number;
375
+ };
376
+ export declare function assertEnforceable(options: SandboxBackendOptions): void;
377
+ /**
378
+ * Refuse a runtime class the pool path cannot honour.
379
+ *
380
+ * A pooled sandbox is already running by the time a claim reaches it, under
381
+ * whatever RuntimeClass its SandboxTemplate named. `runtimeClassName` in this
382
+ * config would therefore be read, accepted and ignored — and the thing it
383
+ * selects is the VM boundary, which is the last control to lose quietly.
384
+ */
385
+ export declare function assertRuntimeClassIsApplicable(config: {
386
+ warmPoolName?: string;
387
+ runtimeClassName?: string;
388
+ }): void;
389
+ /**
390
+ * Build a {@link SandboxBackend} against a cluster running the agent-sandbox
391
+ * controller. Construction is synchronous and contacts nothing: readiness
392
+ * bounds and the config refusals are validated here so a misconfiguration
393
+ * surfaces during host wiring rather than mid-run, and the first API call
394
+ * happens on the first `create()`.
395
+ */
396
+ export declare function buildKubernetesBackend(config: KubernetesBackendInternalConfig): SandboxBackend;
397
+ /**
398
+ * The two egress checks a create path runs, with their memos.
399
+ *
400
+ * `undefined` when `config.egress` is unset — the whole boundary is one
401
+ * absent object rather than a flag every call site re-reads, the same shape
402
+ * {@link IngressVerifier} uses for its own opt-out.
403
+ *
404
+ * Exported because `workspace.ts` runs the identical steps: a workspace does
405
+ * not go through `buildKubernetesBackend`, and a config `egress` honoured on
406
+ * one entry point and ignored on the other would be a silent downgrade of the
407
+ * boundary this backend calls primary. It builds its own boundary per create,
408
+ * which is what makes its checks per-call rather than memoized — creating a
409
+ * workspace is a rare, explicit act with nothing to amortise, and a policy
410
+ * deleted since the last call must be noticed.
411
+ */
412
+ export interface KubernetesEgressBoundary {
413
+ /**
414
+ * Translate `egress.policy` and confirm an operator applied a matching
415
+ * object — the original verify-not-trust step, unchanged, including its
416
+ * exact-match comparison and its once-per-boundary memo.
417
+ */
418
+ verifyNamedObject(signal?: AbortSignal): Promise<void>;
419
+ /**
420
+ * Enumerate every policy selecting THIS pod and refuse when any of them
421
+ * allows egress the translation does not. A no-op under
422
+ * `egress.verify: 'named-object-only'`.
423
+ */
424
+ verifyUnion(podLabels: Readonly<Record<string, string>>, subject: string, signal?: AbortSignal): Promise<void>;
425
+ }
426
+ /**
427
+ * How long a union pass is trusted for one label set.
428
+ *
429
+ * Five minutes rather than the backend's lifetime, which is what the
430
+ * named-object check alone used to get: an operator who applies a widening
431
+ * policy at 10:00 should not have it go unnoticed until the host restarts.
432
+ * It is a cache, not a watch — `k8s-client.ts`'s "no watch, no informers, no
433
+ * resourceVersion tracking" invariant is untouched, because the only thing
434
+ * kept across calls is "this exact label set passed at this time".
435
+ */
436
+ export declare const EGRESS_UNION_CACHE_TTL_MS: number;
437
+ /**
438
+ * Build the egress boundary this config asks for, or nothing at all.
439
+ *
440
+ * `sandboxTemplateName` is the template the caller is actually building from
441
+ * — it decides both the default policy name and the pod label the policy's
442
+ * selector has to match, and a workspace may be built from a different
443
+ * template than the task path's.
444
+ *
445
+ * `now` is injected only so the TTL above can be tested without waiting five
446
+ * minutes; nothing else passes it.
447
+ */
448
+ export declare function buildEgressBoundary(client: KubernetesClient, config: KubernetesBackendInternalConfig, sandboxTemplateName: string, now?: () => number): KubernetesEgressBoundary | undefined;
449
+ /**
450
+ * What a create path calls to prove the agent port is closed before it hands
451
+ * a sandbox back. `undefined` when `config.ingress` is `'unverified'`, so the
452
+ * opt-out is one absent function rather than a flag every call site re-reads.
453
+ */
454
+ export type IngressVerifier = (podLabels: Readonly<Record<string, string>>, subject: string, signal?: AbortSignal) => Promise<void>;
455
+ /**
456
+ * Build the ingress check this config asks for, or nothing at all.
457
+ *
458
+ * `cache` is the provider path's per-label-set memo; a workspace passes none,
459
+ * and re-checks on every call for the same reason its egress check does —
460
+ * creating a workspace is a rare, explicit act with nothing to amortise, and
461
+ * a policy deleted since the last call must be noticed.
462
+ */
463
+ export declare function buildIngressVerifier(client: KubernetesClient, config: KubernetesBackendInternalConfig, cache?: Map<string, Promise<void>>): IngressVerifier | undefined;
464
+ /** Config → the client's own access shape. Shared with `workspace.ts`. */
465
+ export declare function clientAccess(config: KubernetesBackendInternalConfig): KubernetesAccess;
466
+ /**
467
+ * Why an acquire was refused, in the terms an operator acts on rather than
468
+ * the terms the failure happened to arrive in.
469
+ *
470
+ * - `'api-unreachable'` — the API server could not be reached, or kept
471
+ * answering with a status that means "not now": a connect failure, a 429,
472
+ * a 5xx. Retried inside the readiness budget before it ever reaches a
473
+ * caller, so seeing it means the whole budget was spent failing.
474
+ * - `'api-timeout'` — requests were accepted and never answered, until
475
+ * `apiRequestTimeoutMs` gave up on them. Also retried first.
476
+ * - `'forbidden'` — 401 or 403. The host's ServiceAccount cannot do this;
477
+ * no amount of waiting changes that. See the RBAC section of
478
+ * `docs/sdk/kubernetes-sandbox.md`.
479
+ * - `'claim-rejected'` — the controller REFUSED the claim, and said why.
480
+ * {@link KubernetesAcquireError.controllerReason} carries its own word for
481
+ * it. This is the one that used to burn the entire readiness budget before
482
+ * failing.
483
+ * - `'capacity'` — the pod exists and cannot be placed: the scheduler
484
+ * reports `PodScheduled=False` with reason `Unschedulable`. The cluster is
485
+ * full, or nothing matches the template's placement rules.
486
+ * - `'image-pull'` — the pod was placed and its container cannot start
487
+ * because the image will not pull. Permanent until an operator fixes the
488
+ * reference or the pull credential.
489
+ * - `'not-ready'` — none of the above: the readiness budget expired with the
490
+ * cluster reporting nothing wrong. A slow cold start, a webhook, an
491
+ * admission controller, a CNI that never attached the pod.
492
+ */
493
+ export type KubernetesAcquireFailureReason = 'api-unreachable' | 'api-timeout' | 'forbidden' | 'claim-rejected' | 'capacity' | 'image-pull' | 'not-ready';
494
+ /**
495
+ * An acquire that was refused, carrying WHY in a field rather than in prose.
496
+ *
497
+ * Before this class a burst past node capacity and an API outage were the
498
+ * same plain `Error`, and a host could only tell them apart by matching
499
+ * message text that any release is free to reword. `reason` is the diagnosis,
500
+ * `retryable` is the advice that follows from it, and `cause` is the original
501
+ * failure — unmodified, so a host that already catches
502
+ * {@link ReadinessPollTimeout}, {@link KubernetesApiTimeoutError} or
503
+ * `KubernetesCredentialError` finds it there.
504
+ *
505
+ * `retryable` is about THIS acquire being worth attempting again, not about
506
+ * anything having been retried. Transient API failures are already retried
507
+ * inside the readiness budget, so a `retryable: true` that reaches a caller
508
+ * means the whole budget was spent on them.
509
+ *
510
+ * Not every acquire failure becomes one of these, and that is deliberate: a
511
+ * refusal this class cannot honestly diagnose — a malformed template, a 400
512
+ * from an admission webhook, a controller that reported Ready and named no
513
+ * sandbox — travels out as itself rather than being filed under whichever of
514
+ * the seven reasons is least wrong. `KubernetesApiError` carries the status
515
+ * for those.
516
+ */
517
+ export declare class KubernetesAcquireError extends Error {
518
+ readonly name = "KubernetesAcquireError";
519
+ readonly reason: KubernetesAcquireFailureReason;
520
+ /** Whether attempting the same acquire again could plausibly succeed. */
521
+ readonly retryable: boolean;
522
+ /**
523
+ * `status.conditions[Ready].reason`, verbatim, when the controller
524
+ * refused the claim — `WarmPoolNotFound`, `TemplateNotFound`,
525
+ * `InvalidMetadata`, `EnvVarsInjectionRejected`. Present only for
526
+ * `'claim-rejected'`.
527
+ */
528
+ readonly controllerReason?: string;
529
+ /** The controller's own message for the same condition. */
530
+ readonly controllerMessage?: string;
531
+ constructor(details: {
532
+ readonly reason: KubernetesAcquireFailureReason;
533
+ readonly retryable: boolean;
534
+ readonly message: string;
535
+ readonly controllerReason?: string;
536
+ readonly controllerMessage?: string;
537
+ readonly cause?: unknown;
538
+ });
539
+ }
540
+ /**
541
+ * The `status.conditions[Ready].reason` values that mean the controller has
542
+ * DECIDED, so waiting is pointless.
543
+ *
544
+ * ## Where these strings came from
545
+ *
546
+ * Not from the issue that asked for this, and not from upstream source: this
547
+ * repo vendors none of agent-sandbox's Go, so a literal copied out of a
548
+ * changelog is a literal nobody here can check. Each of the four was produced
549
+ * against the deployed controller (kind v1.37.0, agent-sandbox v1.0.2,
550
+ * 2026-09-17) by making the claim it describes and reading the condition
551
+ * back:
552
+ *
553
+ * | reason | how it was produced | the controller's message |
554
+ * |---|---|---|
555
+ * | `WarmPoolNotFound` | claim at a pool that does not exist | `SandboxWarmPool "…" not found` |
556
+ * | `TemplateNotFound` | claim at a pool whose template does not exist | `SandboxTemplate "…" not found` |
557
+ * | `InvalidMetadata` | claim with an `additionalPodMetadata` label outside the allowed domains | `invalid additionalPodMetadata: …` |
558
+ * | `EnvVarsInjectionRejected` | claim with `spec.env` against a template that forbids injection | `environment variable injection rejected: …` |
559
+ *
560
+ * The transient reasons seen on the SAME cluster, which must NOT be in this
561
+ * set, were `DependenciesNotReady` (pod exists, still Pending) and
562
+ * `DependenciesReady` (the Ready=True reason).
563
+ *
564
+ * ## Why an unknown reason is not terminal
565
+ *
566
+ * A wrong literal here fails in one of two ways, and only one of them is
567
+ * recoverable. Too few entries: a rejected claim waits out the readiness
568
+ * budget, which is exactly the behaviour every release before this one had.
569
+ * Too many: an acquire that would have succeeded is refused on a guess. So
570
+ * the set is a closed list of measured strings and everything else falls
571
+ * through to the deadline.
572
+ */
573
+ export declare const TERMINAL_CLAIM_REASONS: readonly string[];
574
+ /**
575
+ * What this backend asked the controller to put on the claim's pod, carried
576
+ * into the rejection so an `InvalidMetadata` refusal can say what was sent
577
+ * and which knob changes it.
578
+ *
579
+ * Context, never a second decision: whether a claim is refused at all is
580
+ * {@link TERMINAL_CLAIM_REASONS}' answer and nobody else's, so an empty map
581
+ * changes nothing about the class thrown, the reason on it, or when it is
582
+ * raised.
583
+ */
584
+ export interface ClaimPodMetadataContext {
585
+ readonly namespace: string;
586
+ /** `spec.additionalPodMetadata.labels`, exactly as sent. */
587
+ readonly requestedPodLabels: Readonly<Record<string, string>>;
588
+ /** The egress profile among those labels, when one is configured. */
589
+ readonly profile?: EgressProfileLabel;
590
+ }
591
+ /**
592
+ * A claim the controller has refused, or `undefined` for one it is still
593
+ * working on.
594
+ *
595
+ * `status: 'False'` alone is not a refusal — it is also what a claim looks
596
+ * like for the whole of a cold start — so the REASON decides, against
597
+ * {@link TERMINAL_CLAIM_REASONS}.
598
+ *
599
+ * ONE class comes out of here whatever the reason, and that is deliberate:
600
+ * `InvalidMetadata` is how the controller refuses a pod label whose domain is
601
+ * not on its allowlist — the profile's, today — and a host's `catch` must not
602
+ * have to be written differently depending on whether a profile happens to be
603
+ * configured. `metadata` only decides what rides along as the `cause`: a
604
+ * {@link KubernetesPodLabelsRejectedError} naming the map that was sent
605
+ * and the `config.egress.profileLabelKey` that moves it, which the
606
+ * controller's own message cannot know about.
607
+ */
608
+ export declare function classifyClaimRejection(claim: SandboxClaimResource | undefined, claimName: string, metadata?: ClaimPodMetadataContext): KubernetesAcquireError | undefined;
609
+ export declare function retryDelayForApiFailure(err: unknown, pollIntervalMs: number): number | undefined;
610
+ /**
611
+ * The refusal a caller sees, given the failure that actually happened and
612
+ * whatever the pod had to say about it.
613
+ *
614
+ * Returns `undefined` for a failure none of the seven reasons describes —
615
+ * see {@link KubernetesAcquireError} for why that is a deliberate hole rather
616
+ * than a missing case. A {@link KubernetesAcquireError} that arrived from
617
+ * deeper in (the claim rejection) is returned unchanged: it is already the
618
+ * diagnosis.
619
+ */
620
+ export declare function classifyAcquireFailure(err: unknown, podDiagnosis: 'capacity' | 'image-pull' | undefined): KubernetesAcquireError | undefined;
621
+ /**
622
+ * Claim or create, wait for Ready, read the bound identity back, resolve the
623
+ * address and learn the pod's uid — or leave nothing behind trying.
624
+ *
625
+ * Exported because the sandbox surface is built on top of this record rather
626
+ * than beside it: one acquire path, one cleanup path, whatever ends up
627
+ * wrapping them.
628
+ *
629
+ * ## What it refuses with
630
+ *
631
+ * Every refusal this function can diagnose arrives as a
632
+ * {@link KubernetesAcquireError} naming one of seven reasons, with the
633
+ * original failure as its `cause`. Three things stay outside that:
634
+ * configuration refused before anything is created
635
+ * ({@link assertEnforceable}, {@link assertRuntimeClassIsApplicable}), a
636
+ * caller's own abort, and a failure none of the seven reasons honestly
637
+ * describes — see {@link KubernetesAcquireError} for why the last one is a
638
+ * hole on purpose.
639
+ *
640
+ * ## What it retries, and what it will not
641
+ *
642
+ * A readiness GET that fails transiently — a connect failure, a request the
643
+ * API bound gave up on, a 429, a 5xx — is repeated INSIDE the readiness
644
+ * deadline, honouring `Retry-After`. One clock, so a retry spends the budget
645
+ * rather than extending it, and a `create()` cannot outlive the timeout its
646
+ * caller chose. The create POST is never retried: it is not idempotent, and a
647
+ * POST whose answer never arrived may already have committed — which is why
648
+ * cleanup deletes the client-owned name whatever happened.
649
+ */
650
+ export declare function acquireKubernetesSandbox(client: KubernetesClient, config: KubernetesBackendInternalConfig, options: SandboxBackendOptions, readiness: {
651
+ readonly timeoutMs: number;
652
+ readonly pollIntervalMs: number;
653
+ }, verifyIngress?: IngressVerifier | undefined, egressBoundary?: KubernetesEgressBoundary | undefined): Promise<KubernetesAcquisition>;
654
+ /**
655
+ * Options for {@link releaseKubernetesTaskSandboxes}.
656
+ */
657
+ export interface KubernetesReleaseTaskSandboxesOptions {
658
+ /**
659
+ * Required, and refused if empty — see {@link releaseKubernetesTaskSandboxes}.
660
+ * The same selector syntax a `kubectl get --selector` takes, e.g.
661
+ * `sandbox.namzu.ai/host-instance=host-a`.
662
+ */
663
+ readonly labelSelector: string;
664
+ readonly signal?: AbortSignal;
665
+ }
666
+ /**
667
+ * Recover a crashed host's claims: LIST every `SandboxClaim` carrying
668
+ * `labelSelector`, `DELETE` each, and report what was removed.
669
+ *
670
+ * This deletes CLAIMS only. The controller's own garbage collection —
671
+ * ownerReferences from claim to the Sandbox it bound, and from Sandbox to
672
+ * Pod and Service — takes the rest down behind it; nothing here reads or
673
+ * touches a Sandbox or a Pod directly. A claim already gone (raced by the
674
+ * controller's own TTL reaper, or a second release call) counts as removed
675
+ * rather than a failure, the same convention every other DELETE in this
676
+ * backend follows.
677
+ *
678
+ * `labelSelector` is REQUIRED and refused, synchronously, before a single
679
+ * request goes out, if it is absent or empty: a release that could fall back
680
+ * to matching every claim (or every claim of the template) would delete a
681
+ * live fleet's work the first time a caller passed one by mistake. There is
682
+ * no default selector for exactly this reason.
683
+ */
684
+ export declare function releaseKubernetesTaskSandboxes(config: KubernetesBackendInternalConfig, options: KubernetesReleaseTaskSandboxesOptions): Promise<{
685
+ readonly deleted: number;
686
+ readonly names: readonly string[];
687
+ }>;
688
+ /** Result of {@link readKubernetesTaskCapacity}. */
689
+ export interface KubernetesTaskCapacity {
690
+ /** `config.warmPoolName`'s own replica counts, straight off its `status`/`spec`. */
691
+ readonly warmPool: {
692
+ /** `status.readyReplicas`. `0` when the field is absent (a brand-new or empty pool). */
693
+ readonly ready: number;
694
+ /** `spec.replicas`. `0` when the field is absent. */
695
+ readonly desired: number;
696
+ };
697
+ /** Every `SandboxClaim` in the namespace bound to `config.warmPoolName`, whatever its state. */
698
+ readonly activeClaims: number;
699
+ /** Every Pod in the namespace currently in phase `Pending`. */
700
+ readonly pendingPods: number;
701
+ }
702
+ /** Options for {@link readKubernetesTaskCapacity}. */
703
+ export interface KubernetesReadTaskCapacityOptions {
704
+ readonly signal?: AbortSignal;
705
+ }
706
+ /**
707
+ * Read task-pool headroom before admitting more work: three GETs, no writes.
708
+ *
709
+ * `config.warmPoolName` is required — this reads the exact object a claim's
710
+ * `warmPoolRef` names, so a pool-less backend (every create is a direct
711
+ * Sandbox) has no pool to report on. `activeClaims` is every claim in the
712
+ * namespace whose `spec.warmPoolRef.name` matches this pool, counted rather
713
+ * than trusted from a label, because a claim's `warmPoolRef` is the one
714
+ * field the API itself guarantees. `pendingPods` is every Pod in the
715
+ * namespace still in phase `Pending` — a coarse but honest signal of
716
+ * in-flight scale-up the ready-replica count alone does not carry, on
717
+ * either the warm or the pool-less path.
718
+ */
719
+ export declare function readKubernetesTaskCapacity(config: KubernetesBackendInternalConfig, options?: KubernetesReadTaskCapacityOptions): Promise<KubernetesTaskCapacity>;
720
+ /**
721
+ * What a directly created Sandbox copies out of a `SandboxTemplate`, and the
722
+ * two things it decides for itself.
723
+ */
724
+ export interface SandboxBodyOptions {
725
+ readonly namespace: string;
726
+ readonly name: string;
727
+ readonly template: SandboxTemplateCopy;
728
+ /** The template the copy came from — the value of {@link sandboxTemplateLabel}. */
729
+ readonly sandboxTemplateName: string;
730
+ readonly runtimeClassName?: string;
731
+ /**
732
+ * RFC 3339 expiry, paired with `shutdownPolicy: Delete`. ABSENT means the
733
+ * object carries no expiry at all and nothing reaps it on the wall clock:
734
+ * that is the persistent workspace (`workspace.ts`), which is explicitly
735
+ * managed and must survive a host that stops renewing. Every task sandbox
736
+ * sets it, because an unbounded task sandbox is a leak.
737
+ */
738
+ readonly shutdownTime?: string;
739
+ /**
740
+ * Annotations to stamp on the Sandbox's OWN metadata at creation.
741
+ *
742
+ * One caller, and everything it writes is a fact the object has to carry
743
+ * from the moment it exists rather than from its first patch: the holder
744
+ * epoch of a workspace created under one, so there is no window in which
745
+ * it stands unfenced, and the revision of the pod template it was built
746
+ * from, so there is no window in which it claims none. Both are
747
+ * `workspace.ts`'s — see `HOLDER_EPOCH_ANNOTATION_KEY` and
748
+ * `POD_TEMPLATE_HASH_ANNOTATION_KEY`.
749
+ *
750
+ * Absent, the body is byte for byte what it always was, which is what
751
+ * keeps every task sandbox's create unchanged — a task sandbox is
752
+ * ephemeral, so it has no revision to drift from and nothing to fence.
753
+ */
754
+ readonly annotations?: Readonly<Record<string, string>>;
755
+ /**
756
+ * Extra labels stamped onto the POD template's metadata, beside the
757
+ * template label this body always adds.
758
+ *
759
+ * The same map `buildClaimBody` puts on a claim's
760
+ * `additionalPodMetadata.labels`, from the same
761
+ * `composeAdditionalPodLabels` call — a direct Sandbox has no controller
762
+ * to merge them for it, so the create body stamps them itself and the two
763
+ * paths produce one set of pod labels. Absent or empty, the pod template
764
+ * is byte for byte what it was.
765
+ */
766
+ readonly podLabels?: Readonly<Record<string, string>>;
767
+ }
768
+ /**
769
+ * The pool-less body. `Sandbox.spec` has no `templateRef` — only a
770
+ * SandboxWarmPool consumes a SandboxTemplate — so the template's podTemplate
771
+ * is copied in here by the client.
772
+ *
773
+ * `service: true` is forced rather than inherited: a Sandbox without a Service
774
+ * has no `status.serviceFQDN`, and then the only address left is a pod IP that
775
+ * changes on every resume.
776
+ *
777
+ * `volumeClaimTemplates` is copied VERBATIM when the template declares any.
778
+ * Dropping it would be silent: the Sandbox would come up healthy with no disk,
779
+ * the container's `volumeDevices`/`volumeMounts` entry would fail to resolve
780
+ * (or, worse, resolve to an empty emptyDir on some paths), and the only
781
+ * symptom of a workspace that lost its disk would be that yesterday's files
782
+ * are gone. The controller wires the mount by the entry's own NAME,
783
+ * StatefulSet style, so the copy needs no matching `volumes:` entry and this
784
+ * function adds none.
785
+ *
786
+ * The podTemplate's metadata gains {@link sandboxTemplateLabel}: this Sandbox
787
+ * is created DIRECTLY, never adopted out of a pool, so it never gets
788
+ * agent-sandbox's own controller-owned
789
+ * `agents.x-k8s.io/sandbox-template-ref-hash` label (that is written only on
790
+ * bind). Without a label of its own a direct Sandbox's pod would carry
791
+ * nothing `egress-policy.ts`'s translated `NetworkPolicy` could select it
792
+ * by. Existing labels on the copied template are preserved — this ADDS to
793
+ * them rather than replacing the object outright — but this backend's own
794
+ * key always wins if the template happened to set it too, since this is the
795
+ * label the translated policy is built to match.
796
+ */
797
+ export declare function buildSandboxBody(options: SandboxBodyOptions): Record<string, unknown>;
798
+ /**
799
+ * The `spec.podTemplate` a directly created Sandbox carries: the template's,
800
+ * with this backend's overlays — the template label and whatever
801
+ * {@link composeAdditionalPodLabels} produced ({@link sandboxPodLabels}), and
802
+ * the configured `runtimeClassName`.
803
+ *
804
+ * Its own function because it is now built twice: once into the create POST
805
+ * by {@link buildSandboxBody}, and once into the JSON Patch that refreshes a
806
+ * standing workspace's pod template (`workspace.ts`). Two expressions of the
807
+ * same overlay would drift, and the one that drifted would report a workspace
808
+ * as off-template forever — the hash under
809
+ * `sandbox.namzu.ai/pod-template-hash` is taken over exactly this object, so
810
+ * a second spelling is a second revision.
811
+ *
812
+ * `podLabels` is on this signature rather than only on the create body for
813
+ * exactly that reason. A refresh rewrites `/spec/podTemplate` WHOLE, so a
814
+ * refresh built without them would PATCH the egress profile off a pod
815
+ * template that carries it — the replacement pod would come up selected by no
816
+ * per-profile policy, on a path where nothing re-checks the label, and the
817
+ * revision stamped beside it would be taken over a template the POST never
818
+ * writes, so `templateCurrent` would report drift forever.
819
+ */
820
+ export declare function sandboxPodTemplate(template: SandboxTemplateCopy, sandboxTemplateName: string, runtimeClassName?: string, podLabels?: Readonly<Record<string, string>>): SandboxPodTemplate;
821
+ /**
822
+ * The labels a directly created Sandbox's pod will carry: whatever the
823
+ * template declares, plus this backend's own template label, which always
824
+ * wins because it is the label a policy selector is built to match.
825
+ *
826
+ * Its own function because the ingress check has to reason about EXACTLY the
827
+ * labels {@link buildSandboxBody} stamps, before the POST that stamps them.
828
+ * Two expressions of the same rule would be one rename away from a check that
829
+ * verifies a pod nobody creates.
830
+ *
831
+ * `extra` is `composeAdditionalPodLabels`'s map — the egress profile today.
832
+ * It is applied LAST, and so wins over both, for the same reason the template
833
+ * label wins over the copied template's own: it is a label the translated
834
+ * policy's selector is built to match, and a pod that matched the selector
835
+ * only sometimes would be a boundary that applied only sometimes.
836
+ */
837
+ export declare function sandboxPodLabels(template: SandboxTemplateCopy, sandboxTemplateName: string, extra?: Readonly<Record<string, string>>): Readonly<Record<string, string>>;
838
+ /** The two halves of a `SandboxTemplate` a directly created Sandbox copies. */
839
+ export interface SandboxTemplateCopy {
840
+ readonly podTemplate: SandboxPodTemplate;
841
+ /** Absent when the template declares no disk, which is every task template. */
842
+ readonly volumeClaimTemplates?: readonly SandboxVolumeClaimTemplate[];
843
+ }
844
+ export declare function readSandboxTemplate(client: KubernetesClient, namespace: string, sandboxTemplateName: string, signal?: AbortSignal): Promise<SandboxTemplateCopy>;
845
+ /**
846
+ * Where the agent answers.
847
+ *
848
+ * Its own function, and under the default mode the Service FQDN wins over a
849
+ * pod IP, because that address outlives the pod: a suspended-then-resumed
850
+ * workspace comes back as a new pod with a new IP behind the same name, and
851
+ * the transport re-resolves the name on every dial. A literal IP baked into a
852
+ * long-lived handle is the bug that would produce — and it is exactly the bug
853
+ * `'pod-ip'` accepts, deliberately, in exchange for an address a host outside
854
+ * the cluster can resolve at all. That mode pays for it by re-reading the IP
855
+ * on every resume and once after a failed connect.
856
+ *
857
+ * `'pod-ip'` takes the address from `pod`, the record the bind token was just
858
+ * read out of, and NEVER falls back to `binding.podIPs`. The Sandbox's status
859
+ * is a second source that can name a different pod — the one a resume is
860
+ * replacing — and an address from one pod with a token from another is the
861
+ * mismatch that arrives as a flat `unauthorized`.
862
+ */
863
+ export declare function resolveAgentAddress(binding: KubernetesSandboxBinding, agentPort: number, token: string, options?: {
864
+ readonly mode?: KubernetesAgentAddressMode;
865
+ /** The live pod the token came from. Required by `'pod-ip'`. */
866
+ readonly podIP?: string;
867
+ }): KubernetesAgentAddress;
868
+ /** Ready-or-not-yet, read off a `Sandbox`'s own status. Shared with `workspace.ts`. */
869
+ export declare function bindingFromSandbox(sandbox: SandboxResource | undefined): KubernetesSandboxBinding | undefined;
870
+ /**
871
+ * The readiness poll ran out of budget — and nothing else. Every OTHER
872
+ * failure {@link pollForBinding} meets is rethrown as itself, so this class
873
+ * is an exact answer to "was it the clock?", which a caller that has to
874
+ * choose between two timeout messages needs and cannot get from the clock.
875
+ *
876
+ * Reading `remainingMs()` after the fact is NOT that answer: the expiry timer
877
+ * and `performance.now()` are different clocks, and a timer that fires a
878
+ * fraction of a millisecond early leaves a positive remainder behind an
879
+ * expiry that has already happened.
880
+ */
881
+ export declare class ReadinessPollTimeout extends Error {
882
+ readonly name = "ReadinessPollTimeout";
883
+ }
884
+ /**
885
+ * What a failed readiness read is worth, decided by the caller.
886
+ *
887
+ * Optional, and absent means exactly the behaviour every caller had before:
888
+ * the first failure of any kind ends the poll. `workspace.ts` passes nothing
889
+ * and is unchanged; the acquire path passes
890
+ * {@link retryDelayForApiFailure} so one 429 on a shared cluster no longer
891
+ * fails a create that had fifty-nine seconds of budget left.
892
+ */
893
+ export interface ReadinessPollBehaviour {
894
+ /**
895
+ * Milliseconds to wait before reading again, or `undefined` to rethrow.
896
+ *
897
+ * The wait is spent on the SAME deadline as everything else in the poll,
898
+ * so a retry consumes the readiness budget and can never extend it. A hook
899
+ * that always returns a number therefore still terminates: the clock ends
900
+ * the loop, not the hook.
901
+ */
902
+ readonly retryDelayFor?: (err: unknown) => number | undefined;
903
+ }
904
+ /**
905
+ * Poll until `read` reports a binding. `read` returns `undefined` for "not
906
+ * yet" and throws for a failure worth surfacing; the deadline owns every wait,
907
+ * including the sleep between attempts, so an expired clock cannot be extended
908
+ * by one more round trip. Shaped after ACI's `pollForRunningIp`.
909
+ *
910
+ * The only failure this raises on its own account is
911
+ * {@link ReadinessPollTimeout}; anything `read` throws travels out unchanged,
912
+ * unless `behaviour.retryDelayFor` claims it — in which case it is repeated
913
+ * inside the same budget and, if the budget then runs out, carried onto the
914
+ * timeout as its `cause`, so a poll that kept failing still says what it kept
915
+ * seeing.
916
+ */
917
+ export declare function pollForBinding(read: (signal: AbortSignal) => Promise<KubernetesSandboxBinding | undefined>, deadline: OperationDeadline, readiness: {
918
+ readonly timeoutMs: number;
919
+ readonly pollIntervalMs: number;
920
+ }, label: string, behaviour?: ReadinessPollBehaviour): Promise<KubernetesSandboxBinding>;
921
+ /**
922
+ * The live pod behind a bound sandbox: its uid, and — read in the SAME
923
+ * answer — the IP it can be dialed at.
924
+ *
925
+ * One record rather than two reads because the two facts have to describe one
926
+ * pod. The uid is the agent's bind token and the IP is where that agent
927
+ * listens; taking them from separate GETs leaves a window in which a resume,
928
+ * an eviction or a node drain replaces the pod in between, and the handle
929
+ * then presents pod A's token at pod B's address. The guest answers that with
930
+ * a flat `unauthorized`, which says nothing about the race that caused it.
931
+ */
932
+ export interface KubernetesBoundPod {
933
+ /** `metadata.uid` — the agent's bind token. */
934
+ readonly uid: string;
935
+ /** `status.podIP`. Read by `agentAddress: 'pod-ip'`; absent is legal. */
936
+ readonly podIP?: string;
937
+ /**
938
+ * `metadata.labels` — what an ingress policy's `podSelector` actually
939
+ * matches. Read off the SAME object the uid and the address come from,
940
+ * for the same reason they are: a policy decision made about one pod and
941
+ * a connection made to another is the mismatch this record exists to
942
+ * prevent. The claim path reads it to ask which policies select the bound
943
+ * pod — a directly created Sandbox's labels are known before its pod
944
+ * exists — and BOTH paths read it to confirm the pod really carries the
945
+ * labels this backend asked the controller for, which is the one thing
946
+ * knowing them in advance cannot establish. See `ingress-policy.ts` and
947
+ * `assertRequestedPodLabelsObserved`.
948
+ */
949
+ readonly labels?: Readonly<Record<string, string>>;
950
+ }
951
+ /**
952
+ * Find the pod a sandbox is currently backed by, and read both facts off it.
953
+ *
954
+ * `uid` is the per-instance agent bind token; `podIP` is where that agent
955
+ * listens, and only `agentAddress: 'pod-ip'` reads it.
956
+ *
957
+ * The pod is named after its Sandbox in agent-sandbox v1.0.2 — verified
958
+ * against a running cluster — but that is an observation, not a documented
959
+ * guarantee, and `Sandbox.status` exposes no pod name to fall back on. So the
960
+ * fast path is one GET by that name, and the only cost of the name convention
961
+ * changing upstream is a second round trip through `status.selector`, which is
962
+ * exactly what the controller publishes the selector for.
963
+ */
964
+ export declare function readBoundPod(client: KubernetesClient, namespace: string, binding: KubernetesSandboxBinding, signal?: AbortSignal): Promise<KubernetesBoundPod>;
965
+ /**
966
+ * {@link readBoundPod}, plus — under `'pod-ip'` only — the wait for an
967
+ * address to go with the token.
968
+ *
969
+ * Under the default mode this is the single read it has always been: one
970
+ * `GET`, in the same place in the same order, because a Service FQDN is
971
+ * published with the Sandbox and needs nothing from the pod but its uid.
972
+ *
973
+ * `'pod-ip'` has to wait, because a LIVE pod is not yet an ADDRESSED pod. A
974
+ * pod is created `Pending` and carries no `status.podIP` until the CNI has
975
+ * finished attaching it, and {@link isPodLive} accepts `Pending` on purpose —
976
+ * the resume path in `workspace.ts` binds its replacement pod long before that
977
+ * pod is Ready, because `Ready` stays True across the transition and the uid
978
+ * is the only transition signal there is. Refusing an address-less pod outright
979
+ * would therefore fail on the NORMAL path, in milliseconds, with the whole
980
+ * readiness budget unspent. So "live, no address yet" is polled on the same
981
+ * deadline as everything else on this path, and
982
+ * {@link resolveAgentAddress}'s own refusal is left as the post-deadline
983
+ * backstop for a pod that never gets an address at all.
984
+ *
985
+ * A failed READ stays fatal, exactly as it was: this is the acquire path,
986
+ * where nothing is being replaced and a pod that cannot be read is not a pod
987
+ * that is about to appear.
988
+ *
989
+ * `requiredLabels` is the second thing worth waiting for, and it is waited
990
+ * for in the SAME loop rather than in a second one: an egress profile is a
991
+ * label the CONTROLLER patches onto the pod it binds, so a pod read the
992
+ * instant it was bound can be live, addressed and not yet labelled. Two
993
+ * loops would be two deadlines and two answers to "is this pod ready to be
994
+ * admitted". Like the address, an expired clock hands the pod back as it is
995
+ * — the caller decides whether a missing label is fatal, and on the acquire
996
+ * path it is: see `assertRequestedPodLabelsObserved`.
997
+ */
998
+ export declare function readAddressedPod(client: KubernetesClient, namespace: string, binding: KubernetesSandboxBinding, deadline: OperationDeadline, readiness: {
999
+ readonly pollIntervalMs: number;
1000
+ }, mode: KubernetesAgentAddressMode, requiredLabels?: Readonly<Record<string, string>>): Promise<KubernetesBoundPod>;
1001
+ /**
1002
+ * The re-read a `'pod-ip'` handle follows a replaced pod with: one live-pod
1003
+ * read, then the same address resolution acquire did.
1004
+ *
1005
+ * Built here rather than inside the transport because finding the pod is a
1006
+ * CONTROL-plane act — the by-name GET, the selector fallback, the
1007
+ * liveness filter — and the transport owns none of that. It is handed over as
1008
+ * a closure so the transport can call it without learning what a Sandbox is.
1009
+ */
1010
+ export declare function buildAgentAddressRefresh(client: KubernetesClient, namespace: string, binding: KubernetesSandboxBinding, agentPort: number, mode: KubernetesAgentAddressMode): (signal?: AbortSignal) => Promise<KubernetesAgentAddress>;
1011
+ /**
1012
+ * Run the probe against a built Sandbox, on its own clock, and throw if it
1013
+ * refuses. Cleanup is the CALLER's, and the two callers want opposite things:
1014
+ * a task acquire destroys the instance, while `workspace.ts` suspends it,
1015
+ * because deleting a workspace deletes its disk and a probe refusal is not a
1016
+ * reason to lose a caller's files.
1017
+ */
1018
+ export declare function probeSandboxPrivileges(sandbox: Sandbox, sandboxName: string, probeTimeoutMs: number, signal?: AbortSignal): Promise<void>;
1019
+ //# sourceMappingURL=index.d.ts.map