@namzu/sandbox 14.0.0 → 16.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. package/CHANGELOG.md +924 -0
  2. package/README.md +369 -14
  3. package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
  4. package/dist/backends/aci-standby-pool/index.js +13 -1
  5. package/dist/backends/aci-standby-pool/index.js.map +1 -1
  6. package/dist/backends/docker/index.d.ts +169 -6
  7. package/dist/backends/docker/index.d.ts.map +1 -1
  8. package/dist/backends/docker/index.js +499 -85
  9. package/dist/backends/docker/index.js.map +1 -1
  10. package/dist/backends/firecracker/index.d.ts.map +1 -1
  11. package/dist/backends/firecracker/index.js +12 -2
  12. package/dist/backends/firecracker/index.js.map +1 -1
  13. package/dist/backends/firecracker/protocol.d.ts +459 -8
  14. package/dist/backends/firecracker/protocol.d.ts.map +1 -1
  15. package/dist/backends/firecracker/protocol.js +136 -0
  16. package/dist/backends/firecracker/protocol.js.map +1 -1
  17. package/dist/backends/firecracker/transport.d.ts +539 -6
  18. package/dist/backends/firecracker/transport.d.ts.map +1 -1
  19. package/dist/backends/firecracker/transport.js +1171 -24
  20. package/dist/backends/firecracker/transport.js.map +1 -1
  21. package/dist/backends/kubernetes/egress-policy.d.ts +1181 -13
  22. package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -1
  23. package/dist/backends/kubernetes/egress-policy.js +2350 -31
  24. package/dist/backends/kubernetes/egress-policy.js.map +1 -1
  25. package/dist/backends/kubernetes/identity.d.ts +193 -0
  26. package/dist/backends/kubernetes/identity.d.ts.map +1 -0
  27. package/dist/backends/kubernetes/identity.js +147 -0
  28. package/dist/backends/kubernetes/identity.js.map +1 -0
  29. package/dist/backends/kubernetes/index.d.ts +678 -33
  30. package/dist/backends/kubernetes/index.d.ts.map +1 -1
  31. package/dist/backends/kubernetes/index.js +1180 -95
  32. package/dist/backends/kubernetes/index.js.map +1 -1
  33. package/dist/backends/kubernetes/ingress-policy.d.ts +375 -0
  34. package/dist/backends/kubernetes/ingress-policy.d.ts.map +1 -0
  35. package/dist/backends/kubernetes/ingress-policy.js +1050 -0
  36. package/dist/backends/kubernetes/ingress-policy.js.map +1 -0
  37. package/dist/backends/kubernetes/k8s-client.d.ts +213 -4
  38. package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -1
  39. package/dist/backends/kubernetes/k8s-client.js +359 -52
  40. package/dist/backends/kubernetes/k8s-client.js.map +1 -1
  41. package/dist/backends/kubernetes/lease.d.ts +40 -14
  42. package/dist/backends/kubernetes/lease.d.ts.map +1 -1
  43. package/dist/backends/kubernetes/lease.js +68 -18
  44. package/dist/backends/kubernetes/lease.js.map +1 -1
  45. package/dist/backends/kubernetes/objects.d.ts +423 -3
  46. package/dist/backends/kubernetes/objects.d.ts.map +1 -1
  47. package/dist/backends/kubernetes/objects.js +364 -2
  48. package/dist/backends/kubernetes/objects.js.map +1 -1
  49. package/dist/backends/kubernetes/per-sandbox-policy.d.ts +219 -0
  50. package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -0
  51. package/dist/backends/kubernetes/per-sandbox-policy.js +375 -0
  52. package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -0
  53. package/dist/backends/kubernetes/rbac.d.ts +153 -0
  54. package/dist/backends/kubernetes/rbac.d.ts.map +1 -0
  55. package/dist/backends/kubernetes/rbac.js +177 -0
  56. package/dist/backends/kubernetes/rbac.js.map +1 -0
  57. package/dist/backends/kubernetes/sandbox.d.ts +81 -14
  58. package/dist/backends/kubernetes/sandbox.d.ts.map +1 -1
  59. package/dist/backends/kubernetes/sandbox.js +149 -15
  60. package/dist/backends/kubernetes/sandbox.js.map +1 -1
  61. package/dist/backends/kubernetes/transport.d.ts +935 -9
  62. package/dist/backends/kubernetes/transport.d.ts.map +1 -1
  63. package/dist/backends/kubernetes/transport.js +1958 -62
  64. package/dist/backends/kubernetes/transport.js.map +1 -1
  65. package/dist/backends/kubernetes/workspace.d.ts +1149 -18
  66. package/dist/backends/kubernetes/workspace.d.ts.map +1 -1
  67. package/dist/backends/kubernetes/workspace.js +2825 -186
  68. package/dist/backends/kubernetes/workspace.js.map +1 -1
  69. package/dist/backends/remote-execution-controller.d.ts +14 -0
  70. package/dist/backends/remote-execution-controller.d.ts.map +1 -1
  71. package/dist/backends/remote-execution-controller.js.map +1 -1
  72. package/dist/index.d.ts +294 -18
  73. package/dist/index.d.ts.map +1 -1
  74. package/dist/index.js +280 -10
  75. package/dist/index.js.map +1 -1
  76. package/dist/testing/sandbox-conformance.d.ts +39 -5
  77. package/dist/testing/sandbox-conformance.d.ts.map +1 -1
  78. package/dist/testing/sandbox-conformance.js +436 -5
  79. package/dist/testing/sandbox-conformance.js.map +1 -1
  80. package/package.json +3 -3
  81. package/src/backends/aci-standby-pool/index.ts +16 -1
  82. package/src/backends/docker/index.ts +617 -100
  83. package/src/backends/firecracker/index.ts +14 -2
  84. package/src/backends/firecracker/protocol.ts +514 -6
  85. package/src/backends/firecracker/transport.ts +1492 -40
  86. package/src/backends/kubernetes/egress-policy.ts +3334 -55
  87. package/src/backends/kubernetes/identity.ts +261 -0
  88. package/src/backends/kubernetes/index.ts +1785 -127
  89. package/src/backends/kubernetes/ingress-policy.ts +1344 -0
  90. package/src/backends/kubernetes/k8s-client.ts +444 -54
  91. package/src/backends/kubernetes/lease.ts +75 -19
  92. package/src/backends/kubernetes/objects.ts +626 -6
  93. package/src/backends/kubernetes/per-sandbox-policy.ts +497 -0
  94. package/src/backends/kubernetes/rbac.ts +192 -0
  95. package/src/backends/kubernetes/sandbox.ts +218 -20
  96. package/src/backends/kubernetes/transport.ts +2733 -124
  97. package/src/backends/kubernetes/workspace.ts +4476 -222
  98. package/src/backends/remote-execution-controller.ts +14 -0
  99. package/src/index.ts +668 -19
  100. package/src/testing/sandbox-conformance.ts +540 -5
@@ -77,16 +77,35 @@
77
77
  * ## Egress
78
78
  *
79
79
  * `config.egress` is optional and, when set, translated and VERIFIED — never
80
- * created — by `egress-policy.ts`. Verification happens once, lazily, on the
81
- * first `create()`, so `buildKubernetesBackend` itself still contacts
82
- * nothing. Every Sandbox this file creates directly (`buildSandboxBody`)
83
- * carries {@link sandboxTemplateLabel} on its podTemplate specifically so
84
- * that translated policy's `podSelector` has something stable to match —
85
- * see `objects.ts`'s doc comment on that label for why agent-sandbox's own
86
- * controller-owned label does not cover this path.
80
+ * created — by `egress-policy.ts`, in two steps. The NAMED object is GETted
81
+ * and compared to the translation exactly, once, lazily, on the first
82
+ * `create()`, so `buildKubernetesBackend` itself still contacts nothing.
83
+ * Then, because the API server UNIONS every policy selecting a pod, every
84
+ * `NetworkPolicy` in the namespace (and, under `engine: 'cilium'`, every
85
+ * `CiliumNetworkPolicy`) is enumerated against the pod's real labels and the
86
+ * create is refused when any of them lets out more than the translation does
87
+ * — a second policy widens egress however exactly the named one matches, and
88
+ * a `SandboxTemplate`'s own `networkPolicy` block becomes exactly such a
89
+ * policy. `egress.verify: 'named-object-only'` is the opt-out and restores
90
+ * the first step alone. Every Sandbox this file creates directly
91
+ * (`buildSandboxBody`) carries {@link sandboxTemplateLabel} on its
92
+ * podTemplate specifically so that translated policy's `podSelector` has
93
+ * something stable to match — see `objects.ts`'s doc comment on that label
94
+ * for why agent-sandbox's own controller-owned label does not cover this
95
+ * path.
96
+ *
97
+ * ## Ingress
98
+ *
99
+ * `config.ingress` is the opposite default: verification is ON unless a
100
+ * deployment says `'unverified'`. Before the POST for a direct Sandbox, and
101
+ * after the bind for a claimed one, `ingress-policy.ts` lists the namespace's
102
+ * policies and refuses unless one of them actually closes the agent port on
103
+ * the labels this pod carries. Nothing checked that before, while three
104
+ * pieces of shipped text said it was covered — see that module's own doc
105
+ * comment for what was measured.
87
106
  */
88
107
 
89
- import type { Sandbox } from '@namzu/sdk'
108
+ import type { Sandbox, SandboxNetworkPolicy } from '@namzu/sdk'
90
109
  import { generateSandboxId } from '@namzu/sdk'
91
110
 
92
111
  import type { SandboxBackend, SandboxBackendOptions } from '../../index.js'
@@ -97,46 +116,88 @@ import {
97
116
  runFailureCleanup,
98
117
  } from '../readiness.js'
99
118
  import {
119
+ type EgressProfileLabel,
100
120
  type KubernetesEgressConfig,
121
+ KubernetesPodLabelNotObservedError,
122
+ KubernetesPodLabelsRejectedError,
123
+ type KubernetesTranslatedEgressPolicy,
101
124
  assertEgressPolicyIsEnforceable,
125
+ assertEgressProfileIsUsable,
126
+ assertPerSandboxEgressIsUsable,
127
+ composeAdditionalPodLabels,
102
128
  defaultEgressPolicyName,
129
+ egressProfileLabel,
130
+ egressUnionVerificationEnabled,
131
+ perSandboxEgressLabelKey,
103
132
  translateEgressPolicy,
104
133
  verifyEgressPolicyApplied,
134
+ verifyEgressPolicyUnion,
105
135
  } from './egress-policy.js'
136
+ import {
137
+ type KubernetesIngressConfig,
138
+ ingressVerificationEnabled,
139
+ resolveIngressEngine,
140
+ verifyIngressPolicyApplied,
141
+ } from './ingress-policy.js'
106
142
  import {
107
143
  type KubernetesAccess,
108
144
  KubernetesAlreadyGoneError,
145
+ KubernetesApiError,
146
+ KubernetesApiTimeoutError,
109
147
  type KubernetesClient,
148
+ type KubernetesClientOptions,
149
+ KubernetesCredentialError,
110
150
  createKubernetesClient,
111
151
  } from './k8s-client.js'
112
152
  import {
153
+ type KubernetesCondition,
113
154
  type PodListResource,
114
155
  type PodResource,
115
156
  READY_CONDITION,
116
157
  SANDBOX_API_GROUP,
117
158
  SANDBOX_API_VERSION,
118
159
  SANDBOX_EXTENSIONS_API_GROUP,
160
+ type SandboxClaimListResource,
119
161
  type SandboxClaimResource,
120
162
  type SandboxPodTemplate,
121
163
  type SandboxResource,
122
164
  type SandboxTemplateResource,
123
165
  type SandboxVolumeClaimTemplate,
166
+ type SandboxWarmPoolResource,
124
167
  claimCollectionPath,
168
+ claimListPath,
125
169
  claimPath,
126
170
  isConditionTrue,
127
171
  isPodLive,
172
+ podCollectionPath,
128
173
  podListPath,
129
174
  podPath,
175
+ readPodIP,
130
176
  sandboxCollectionPath,
131
177
  sandboxPath,
132
178
  sandboxTemplateLabel,
133
179
  sandboxTemplatePath,
180
+ warmPoolPath,
134
181
  } from './objects.js'
182
+ import {
183
+ KubernetesOwnerUidMissingError,
184
+ type PerSandboxPolicyOwner,
185
+ buildAdmissionFence,
186
+ buildPerSandboxPolicySetter,
187
+ } from './per-sandbox-policy.js'
135
188
  import { privilegeProbeTimedOut, runPrivilegeProbe } from './privilege-probe.js'
136
189
  import { buildKubernetesSandbox } from './sandbox.js'
137
190
  import { KubernetesAgentTransport } from './transport.js'
138
191
 
139
- export type { KubernetesEgressConfig, KubernetesEgressEngine } from './egress-policy.js'
192
+ export type {
193
+ KubernetesEgressConfig,
194
+ KubernetesEgressEngine,
195
+ KubernetesEgressPolicy,
196
+ KubernetesEgressVerification,
197
+ KubernetesOnlyEgressPolicy,
198
+ KubernetesPerSandboxEgressConfig,
199
+ } from './egress-policy.js'
200
+ export type { KubernetesIngressConfig, KubernetesIngressEngine } from './ingress-policy.js'
140
201
 
141
202
  /**
142
203
  * How the backend reaches the API server. Two sources, neither needing a YAML
@@ -166,6 +227,13 @@ export interface KubernetesBackendInternalConfig {
166
227
  readonly warmPoolName?: string
167
228
  /** TCP port the guest agent listens on. Default {@link DEFAULT_AGENT_PORT}. */
168
229
  readonly agentPort?: number
230
+ /**
231
+ * Which of a sandbox's two addresses the transport dials. Default
232
+ * `'service'` — see {@link KubernetesAgentAddressMode}, which is where
233
+ * the choice is explained, because it is a fact about where the HOST
234
+ * runs rather than about the cluster.
235
+ */
236
+ readonly agentAddress?: KubernetesAgentAddressMode
169
237
  readonly readyPollIntervalMs?: number
170
238
  readonly readyTimeoutMs?: number
171
239
  /**
@@ -189,13 +257,124 @@ export interface KubernetesBackendInternalConfig {
189
257
  /**
190
258
  * Egress policy this backend's `NetworkPolicy` (or `CiliumNetworkPolicy`,
191
259
  * under `engine: 'cilium'`) is expected to carry. Unset means this backend
192
- * neither computes nor verifies one — the cluster's default posture (the
193
- * SandboxTemplate's own managed NetworkPolicy) is all that applies. See
194
- * `egress-policy.ts`.
260
+ * neither computes nor verifies one, and OUTBOUND traffic is then whatever
261
+ * the cluster's own policies happen to allow.
262
+ *
263
+ * Deliberately not described as "covered by the SandboxTemplate's managed
264
+ * NetworkPolicy", which is what this comment used to claim: that policy
265
+ * selects `agents.x-k8s.io/sandbox-template-ref-hash`, a label the
266
+ * controller writes only onto a Sandbox adopted out of a `SandboxWarmPool`
267
+ * and never onto one this backend POSTs — so for every pool-less sandbox
268
+ * and every workspace it selects nothing at all. See `ingress-policy.ts`.
195
269
  */
196
270
  readonly egress?: KubernetesEgressConfig
271
+ /**
272
+ * Whether the agent port's INGRESS boundary is verified before a sandbox
273
+ * is created, and against which policy resources.
274
+ *
275
+ * Unset means VERIFY — the one field in this config whose absent value is
276
+ * the strict one, because the deployment that needs the check is the one
277
+ * that would never have set it. `'unverified'` reads no policy and issues
278
+ * no request, for a deployment whose boundary lives somewhere a namespaced
279
+ * Role cannot see. See `ingress-policy.ts`.
280
+ */
281
+ readonly ingress?: KubernetesIngressConfig
282
+ /**
283
+ * Bound on every single Kubernetes API request this backend sends. See
284
+ * {@link KubernetesClientOptions.requestTimeoutMs} in `k8s-client.ts`,
285
+ * which owns the default (30 s), the floor (1 s) and the reason there is
286
+ * no value that disables it.
287
+ */
288
+ readonly apiRequestTimeoutMs?: number
289
+ /**
290
+ * Interval of the negotiated liveness heartbeat on `openTerminal` and
291
+ * `openTcpConnection` streams. Default
292
+ * {@link DEFAULT_STREAM_HEARTBEAT_MS}; `0` turns it off and restores the
293
+ * pre-heartbeat behaviour exactly. See {@link resolveStreamHeartbeatMs}.
294
+ */
295
+ readonly streamHeartbeatMs?: number
296
+ /**
297
+ * Extra labels written onto every `SandboxClaim` this backend POSTs —
298
+ * `metadata.labels`, and nowhere else. Never merged into
299
+ * `additionalPodMetadata`: those are POD labels a running Sandbox and its
300
+ * `NetworkPolicy` selectors read, and a host's own recovery bookkeeping
301
+ * has no business changing what a pod is selected by. Absent means no
302
+ * labels beyond what the controller itself writes, and every claim body
303
+ * this backend sends is byte-for-byte what it always was.
304
+ *
305
+ * The intended use is a host-instance identity — e.g.
306
+ * `{ 'sandbox.namzu.ai/host-instance': hostId }` — so a restarted host
307
+ * can find and {@link releaseKubernetesTaskSandboxes} its predecessor's
308
+ * claims well before `claimTtlSeconds` would reap them on its own.
309
+ */
310
+ readonly claimLabels?: Record<string, string>
311
+ }
312
+
313
+ /**
314
+ * Default {@link KubernetesBackendInternalConfig.streamHeartbeatMs} — 15 s,
315
+ * so a stream whose peer vanished without a FIN or an RST is given up on
316
+ * within 45 s rather than never.
317
+ *
318
+ * The Kubernetes backend opts IN here; `VsockTransportOptions.heartbeatMs`
319
+ * stays undefined by default, because that transport is shared with the
320
+ * Firecracker tier and a default there would force-close an existing
321
+ * consumer's quiet-but-alive terminal.
322
+ */
323
+ export const DEFAULT_STREAM_HEARTBEAT_MS = 15_000
324
+
325
+ /**
326
+ * Validate the configured stream-heartbeat interval, or supply the default.
327
+ *
328
+ * `0` IS accepted here, unlike `apiRequestTimeoutMs`: the heartbeat is a new
329
+ * capability that a deployment behind a middlebox with its own idea about
330
+ * unexpected frames may want off, and turning it off restores exactly the
331
+ * behaviour every release before this one had. An unanswered API request has
332
+ * no such prior behaviour worth restoring.
333
+ */
334
+ export function resolveStreamHeartbeatMs(value: number | undefined): number {
335
+ if (value === undefined) return DEFAULT_STREAM_HEARTBEAT_MS
336
+ if (!Number.isSafeInteger(value) || value < 0) {
337
+ throw new Error(
338
+ `kubernetes: streamHeartbeatMs must be a non-negative integer, got ${JSON.stringify(
339
+ value,
340
+ )}. Use 0 to send no heartbeats at all, which is how every release before this one behaved; the default is ${DEFAULT_STREAM_HEARTBEAT_MS}ms.`,
341
+ )
342
+ }
343
+ return value
197
344
  }
198
345
 
346
+ /**
347
+ * The client options every `createKubernetesClient` call in this backend is
348
+ * built with. One function so the five call sites — the provider, and each
349
+ * of the workspace verbs — cannot drift apart on which bounds they honour.
350
+ */
351
+ export function clientOptions(config: KubernetesBackendInternalConfig): KubernetesClientOptions {
352
+ return { requestTimeoutMs: config.apiRequestTimeoutMs }
353
+ }
354
+
355
+ /**
356
+ * Which address a sandbox's agent is dialed at. A property of where the HOST
357
+ * runs, not of the cluster.
358
+ *
359
+ * - `'service'` (default) — the Sandbox's `status.serviceFQDN`,
360
+ * `<name>.<namespace>.svc.cluster.local`. It outlives the pod: a resumed
361
+ * workspace comes back behind the same name, and every dial re-resolves
362
+ * it. The catch is that only the cluster's own DNS answers it, so this is
363
+ * correct exactly when the host itself runs inside the cluster.
364
+ * - `'pod-ip'` — the bound pod's IP, read from the same `GET` that reads
365
+ * its uid, so the address and the bind token are always one pod's. For a
366
+ * host OUTSIDE the cluster on a routable pod network (a peered VNet, a
367
+ * node-local operator, a CI runner with a route): it needs no cluster
368
+ * resolver at all. The cost is that a pod IP dies with its pod, which is
369
+ * why a resume re-reads it and a connect failure re-reads it once.
370
+ *
371
+ * The cluster still decides whether either address is REACHABLE: `'pod-ip'`
372
+ * additionally needs a NetworkPolicy that admits the host's own address
373
+ * range on the agent port. Neither mode changes the bind token, the
374
+ * privilege probe or egress verification.
375
+ */
376
+ export type KubernetesAgentAddressMode = 'service' | 'pod-ip'
377
+
199
378
  /**
200
379
  * The address the guest agent answers on, plus the token to present.
201
380
  *
@@ -224,8 +403,33 @@ export interface KubernetesSandboxBinding {
224
403
  export interface KubernetesAcquisition {
225
404
  readonly binding: KubernetesSandboxBinding
226
405
  readonly agent: KubernetesAgentAddress
406
+ /**
407
+ * Re-read the live pod and recompute {@link agent} from it. Present only
408
+ * under `agentAddress: 'pod-ip'`, where the address is a literal that
409
+ * dies with its pod; a Service FQDN needs no such thing, and handing one
410
+ * over anyway would give the default mode a re-read it never had.
411
+ */
412
+ readonly refreshAgent?: (signal?: AbortSignal) => Promise<KubernetesAgentAddress>
227
413
  /** API path of the object THIS backend created — the claim, or the Sandbox. */
228
414
  readonly ownedPath: string
415
+ /**
416
+ * The object this backend created, identified well enough to be named in
417
+ * another object's `ownerReferences`: its kind, its name and the uid the
418
+ * API server assigned it.
419
+ *
420
+ * Present only when `config.egress.perSandbox` is configured, because it
421
+ * is read from the create reply and the readiness polls and nothing else
422
+ * needs it — a deployment that never writes a per-sandbox policy should
423
+ * not start carrying a field whose absence would otherwise be a bug.
424
+ */
425
+ readonly owner?: PerSandboxPolicyOwner
426
+ /**
427
+ * The VALUE of the per-sandbox selector label on this sandbox's pod —
428
+ * the created object's name, confirmed on the bound pod before the
429
+ * sandbox was admitted. Present under the same condition as
430
+ * {@link owner}.
431
+ */
432
+ readonly perSandboxLabelValue?: string
229
433
  /** The TTL acquire stamped, which every renewal re-stamps. */
230
434
  readonly ttlSeconds: number
231
435
  /** DELETE that object. An already-gone object counts as released. */
@@ -378,69 +582,311 @@ export function buildKubernetesBackend(config: KubernetesBackendInternalConfig):
378
582
  // `config.egress.policy.kind` alone, with no API call, so it is refused
379
583
  // here, synchronously, the same moment the two checks above are.
380
584
  if (config.egress) {
381
- assertEgressPolicyIsEnforceable(config.egress.policy, config.egress.engine ?? 'core')
585
+ assertEgressPolicyIsEnforceable(
586
+ config.egress.policy,
587
+ config.egress.engine ?? 'core',
588
+ config.egress.ciliumNarrowing,
589
+ )
382
590
  }
383
- const client = createKubernetesClient(clientAccess(config))
591
+ // And the same for an egress PROFILE that could never be written as a
592
+ // label — a value the API server would reject leaves either a claim
593
+ // nothing binds or a policy nobody can apply, and both are decidable
594
+ // from config alone.
595
+ assertEgressProfileIsUsable(config.egress)
596
+ // And the same for per-sandbox egress: an engine that cannot express a
597
+ // hostname, an unnamed admission fence or a selector key the API server
598
+ // would refuse are all decidable from config alone, and a host that
599
+ // learns any of them from its first `setNetworkPolicy` call learns it an
600
+ // hour into a run that cannot be redone.
601
+ assertPerSandboxEgressIsUsable(config.egress)
602
+ // Resolved here as well as at each session, so a configuration this
603
+ // backend will never honour is refused while `buildKubernetesBackend` is
604
+ // still on the stack rather than on someone's first `create()`.
605
+ resolveStreamHeartbeatMs(config.streamHeartbeatMs)
606
+ const client = createKubernetesClient(clientAccess(config), clientOptions(config))
384
607
  // Verify-not-trust runs once, lazily, on the first `create()` — never here,
385
608
  // because `buildKubernetesBackend` is documented to contact nothing. A
386
609
  // failed attempt is not cached: a transient API error should not wedge
387
- // every later create() behind the same stale rejection forever.
388
- let egressVerification: Promise<void> | undefined
610
+ // every later create() behind the same stale rejection forever. The
611
+ // boundary object holds both halves and both memos; it lives as long as
612
+ // this backend does, which is what makes the named-object check
613
+ // once-per-backend rather than once-per-create.
614
+ const egressBoundary = buildEgressBoundary(client, config, config.sandboxTemplateName)
615
+ // Ingress is cached PER LABEL SET rather than once per backend, because
616
+ // unlike the egress NAMED-object check it is a question about one pod: a
617
+ // pooled sandbox's labels come off the pool's template and a pool-less
618
+ // one's off this config, and a single memo would answer for a pod it never
619
+ // examined. Same failure handling as the egress memo — a failed attempt is
620
+ // dropped, so a transient API error does not wedge every later create()
621
+ // behind it. The egress UNION check is keyed the same way, for the same
622
+ // reason, and additionally expires: see {@link EGRESS_UNION_CACHE_TTL_MS}.
623
+ const ingressVerifier = buildIngressVerifier(client, config, new Map())
624
+ // One fence per backend, so its memo is shared by every sandbox this
625
+ // backend hands out rather than re-proved per handle. `undefined` when
626
+ // per-sandbox egress is not configured, which is what makes
627
+ // `setNetworkPolicy` absent from the handle — presence follows
628
+ // CONFIGURATION and never a runtime probe, so a caller's capability
629
+ // detection cannot depend on when it asked.
630
+ const perSandbox = config.egress?.perSandbox
631
+ const fence = perSandbox === undefined ? undefined : buildAdmissionFence(client, perSandbox)
389
632
  return {
390
633
  tier: 'microvm',
391
634
  name: 'kubernetes',
392
635
  async create(options: SandboxBackendOptions): Promise<Sandbox> {
393
- if (config.egress) {
394
- egressVerification ??= verifyEgressPolicyConfigured(
395
- client,
396
- config.namespace,
397
- config.sandboxTemplateName,
398
- config.egress,
399
- options.signal,
400
- ).catch((err: unknown) => {
401
- egressVerification = undefined
402
- throw err
403
- })
404
- await egressVerification
405
- }
406
- const acquisition = await acquireKubernetesSandbox(client, config, options, readiness)
636
+ await egressBoundary?.verifyNamedObject(options.signal)
637
+ const acquisition = await acquireKubernetesSandbox(
638
+ client,
639
+ config,
640
+ options,
641
+ readiness,
642
+ ingressVerifier,
643
+ egressBoundary,
644
+ )
645
+ const egress = config.egress
646
+ const setNetworkPolicy =
647
+ fence !== undefined &&
648
+ egress?.perSandbox !== undefined &&
649
+ acquisition.owner !== undefined &&
650
+ acquisition.perSandboxLabelValue !== undefined
651
+ ? buildPerSandboxPolicySetter({
652
+ client,
653
+ fence,
654
+ namespace: config.namespace,
655
+ egress: { ...egress, perSandbox: egress.perSandbox },
656
+ owner: acquisition.owner,
657
+ selectorValue: acquisition.perSandboxLabelValue,
658
+ })
659
+ : undefined
407
660
  return await admitProbedSandbox(
408
661
  acquisition,
409
662
  config,
410
663
  options,
411
664
  resolveProbeTimeoutMs(readiness.timeoutMs),
665
+ setNetworkPolicy,
412
666
  )
413
667
  },
414
668
  }
415
669
  }
416
670
 
417
671
  /**
418
- * Translate `egress.policy` and confirm an operator applied a matching
419
- * object — the whole verify-not-trust step, isolated so `create()` above
420
- * stays about ONE thing (memoize-once-per-backend) rather than two.
672
+ * The two egress checks a create path runs, with their memos.
673
+ *
674
+ * `undefined` when `config.egress` is unset — the whole boundary is one
675
+ * absent object rather than a flag every call site re-reads, the same shape
676
+ * {@link IngressVerifier} uses for its own opt-out.
421
677
  *
422
- * Exported because `workspace.ts` runs the identical step: a workspace does
678
+ * Exported because `workspace.ts` runs the identical steps: a workspace does
423
679
  * not go through `buildKubernetesBackend`, and a config `egress` honoured on
424
680
  * one entry point and ignored on the other would be a silent downgrade of the
425
- * boundary this backend calls primary. `sandboxTemplateName` is the template
426
- * the caller is actually building from — it decides both the default policy
427
- * name and the pod label the policy's selector has to match, and a workspace
428
- * may be built from a different template than the task path's.
681
+ * boundary this backend calls primary. It builds its own boundary per create,
682
+ * which is what makes its checks per-call rather than memoized — creating a
683
+ * workspace is a rare, explicit act with nothing to amortise, and a policy
684
+ * deleted since the last call must be noticed.
429
685
  */
430
- export async function verifyEgressPolicyConfigured(
686
+ export interface KubernetesEgressBoundary {
687
+ /**
688
+ * Translate `egress.policy` and confirm an operator applied a matching
689
+ * object — the original verify-not-trust step, unchanged, including its
690
+ * exact-match comparison and its once-per-boundary memo.
691
+ */
692
+ verifyNamedObject(signal?: AbortSignal): Promise<void>
693
+ /**
694
+ * Enumerate every policy selecting THIS pod and refuse when any of them
695
+ * allows egress the translation does not. A no-op under
696
+ * `egress.verify: 'named-object-only'`.
697
+ */
698
+ verifyUnion(
699
+ podLabels: Readonly<Record<string, string>>,
700
+ subject: string,
701
+ signal?: AbortSignal,
702
+ ): Promise<void>
703
+ }
704
+
705
+ /**
706
+ * How long a union pass is trusted for one label set.
707
+ *
708
+ * Five minutes rather than the backend's lifetime, which is what the
709
+ * named-object check alone used to get: an operator who applies a widening
710
+ * policy at 10:00 should not have it go unnoticed until the host restarts.
711
+ * It is a cache, not a watch — `k8s-client.ts`'s "no watch, no informers, no
712
+ * resourceVersion tracking" invariant is untouched, because the only thing
713
+ * kept across calls is "this exact label set passed at this time".
714
+ */
715
+ export const EGRESS_UNION_CACHE_TTL_MS = 5 * 60 * 1_000
716
+
717
+ /** One cached pass, and when it was taken. */
718
+ interface CachedPass {
719
+ readonly at: number
720
+ readonly pending: Promise<void>
721
+ }
722
+
723
+ /**
724
+ * Build the egress boundary this config asks for, or nothing at all.
725
+ *
726
+ * `sandboxTemplateName` is the template the caller is actually building from
727
+ * — it decides both the default policy name and the pod label the policy's
728
+ * selector has to match, and a workspace may be built from a different
729
+ * template than the task path's.
730
+ *
731
+ * `now` is injected only so the TTL above can be tested without waiting five
732
+ * minutes; nothing else passes it.
733
+ */
734
+ export function buildEgressBoundary(
431
735
  client: KubernetesClient,
432
- namespace: string,
736
+ config: KubernetesBackendInternalConfig,
433
737
  sandboxTemplateName: string,
434
- egress: KubernetesEgressConfig,
435
- signal?: AbortSignal,
436
- ): Promise<void> {
738
+ now: () => number = Date.now,
739
+ ): KubernetesEgressBoundary | undefined {
740
+ const egress = config.egress
741
+ if (egress === undefined) return undefined
437
742
  const engine = egress.engine ?? 'core'
438
- const translated = await translateEgressPolicy(egress.policy, engine, {
439
- namespace,
440
- name: egress.networkPolicyName ?? defaultEgressPolicyName(sandboxTemplateName),
743
+ // One resolution of the profile, shared by the policy NAME and the policy
744
+ // SELECTOR: under a profile the default name gains the profile segment
745
+ // (one template under two profiles is two policy objects) and the
746
+ // selector gains the label, and the two must not be able to disagree.
747
+ const profile = egressProfileLabel(egress)
748
+ const target = {
749
+ namespace: config.namespace,
750
+ name: egress.networkPolicyName ?? defaultEgressPolicyName(sandboxTemplateName, profile?.value),
441
751
  sandboxTemplateName,
442
- })
443
- await verifyEgressPolicyApplied(client, translated, signal)
752
+ ...(profile !== undefined ? { profile } : {}),
753
+ }
754
+ // Translated ONCE per boundary, not once per check: a `resolver` policy's
755
+ // `resolve()` is the host's own closure and may cost a network call, and
756
+ // running the two checks against two independently resolved allowlists
757
+ // would compare each against a different translation.
758
+ let translation: Promise<KubernetesTranslatedEgressPolicy> | undefined
759
+ const translate = (): Promise<KubernetesTranslatedEgressPolicy> => {
760
+ translation ??= translateEgressPolicy(
761
+ egress.policy,
762
+ engine,
763
+ target,
764
+ egress.ciliumNarrowing,
765
+ ).catch((err: unknown) => {
766
+ translation = undefined
767
+ throw err
768
+ })
769
+ return translation
770
+ }
771
+ let namedObject: Promise<void> | undefined
772
+ const passes = new Map<string, CachedPass>()
773
+ // The per-sandbox selector label carries a once-ever value, so it is
774
+ // excluded from the memo key — see {@link policyCacheKey}. Resolved once
775
+ // here rather than per check, because the resolver validates as it
776
+ // resolves and a per-check throw would surface from a cache lookup.
777
+ const perSandboxKey = perSandboxEgressLabelKey(egress)
778
+ return {
779
+ async verifyNamedObject(signal) {
780
+ namedObject ??= (async () => {
781
+ await verifyEgressPolicyApplied(client, await translate(), signal)
782
+ })().catch((err: unknown) => {
783
+ namedObject = undefined
784
+ throw err
785
+ })
786
+ await namedObject
787
+ },
788
+ async verifyUnion(podLabels, subject, signal) {
789
+ if (!egressUnionVerificationEnabled(egress)) return
790
+ const translated = await translate()
791
+ const key = policyCacheKey(podLabels, perSandboxKey)
792
+ const cached = passes.get(key)
793
+ if (cached !== undefined && now() - cached.at < EGRESS_UNION_CACHE_TTL_MS) {
794
+ await cached.pending
795
+ return
796
+ }
797
+ const pending = verifyEgressPolicyUnion(
798
+ client,
799
+ translated,
800
+ { namespace: config.namespace, podLabels, engine, subject },
801
+ signal,
802
+ ).catch((err: unknown) => {
803
+ // A failed attempt is never cached — same rule the named-object
804
+ // memo has always had.
805
+ passes.delete(key)
806
+ throw err
807
+ })
808
+ passes.set(key, { at: now(), pending })
809
+ await pending
810
+ },
811
+ }
812
+ }
813
+
814
+ /**
815
+ * What a create path calls to prove the agent port is closed before it hands
816
+ * a sandbox back. `undefined` when `config.ingress` is `'unverified'`, so the
817
+ * opt-out is one absent function rather than a flag every call site re-reads.
818
+ */
819
+ export type IngressVerifier = (
820
+ podLabels: Readonly<Record<string, string>>,
821
+ subject: string,
822
+ signal?: AbortSignal,
823
+ ) => Promise<void>
824
+
825
+ /**
826
+ * Canonical key for one label set — order-independent, so two spellings of
827
+ * the same pod share a memo.
828
+ *
829
+ * `excludeKey` drops the PER-SANDBOX egress label, whose value is unique per
830
+ * acquire. Both memos exist to amortise a namespace-wide policy enumeration
831
+ * across every sandbox a backend produces, and a key that carried a
832
+ * once-ever value would give every acquire a miss and leave an entry behind
833
+ * that nothing ever looks up again — a full enumeration per sandbox, and a
834
+ * Map that grows for the host's whole life.
835
+ *
836
+ * Dropping it is sound at the moment these checks run: the only policy that
837
+ * could select a pod BY that key and value is that sandbox's own, whose name
838
+ * is generated in the same call and which does not exist yet. Every other
839
+ * policy selecting the pod — the operator's baseline, a per-profile one,
840
+ * anything hand-written — selects on the labels that remain, so two pods
841
+ * differing only in this label are the same question. The label itself is
842
+ * still PRESENT in the label set each check is run against; only the memo's
843
+ * key ignores it.
844
+ */
845
+ function policyCacheKey(podLabels: Readonly<Record<string, string>>, excludeKey?: string): string {
846
+ return JSON.stringify(
847
+ Object.entries(podLabels)
848
+ .filter(([k]) => k !== excludeKey)
849
+ .sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0)),
850
+ )
851
+ }
852
+
853
+ /**
854
+ * Build the ingress check this config asks for, or nothing at all.
855
+ *
856
+ * `cache` is the provider path's per-label-set memo; a workspace passes none,
857
+ * and re-checks on every call for the same reason its egress check does —
858
+ * creating a workspace is a rare, explicit act with nothing to amortise, and
859
+ * a policy deleted since the last call must be noticed.
860
+ */
861
+ export function buildIngressVerifier(
862
+ client: KubernetesClient,
863
+ config: KubernetesBackendInternalConfig,
864
+ cache?: Map<string, Promise<void>>,
865
+ ): IngressVerifier | undefined {
866
+ if (!ingressVerificationEnabled(config.ingress)) return undefined
867
+ const engine = resolveIngressEngine(config.ingress, config.egress?.engine)
868
+ const agentPort = config.agentPort ?? DEFAULT_AGENT_PORT
869
+ // Same exclusion, same reason, as the egress union memo above: a
870
+ // once-ever label value in the key would make this memo a per-acquire
871
+ // miss and an unbounded Map. See {@link policyCacheKey}.
872
+ const perSandboxKey = perSandboxEgressLabelKey(config.egress)
873
+ return async (podLabels, subject, signal) => {
874
+ const target = { namespace: config.namespace, podLabels, agentPort, engine, subject }
875
+ if (cache === undefined) {
876
+ await verifyIngressPolicyApplied(client, target, signal)
877
+ return
878
+ }
879
+ const key = policyCacheKey(podLabels, perSandboxKey)
880
+ let pending = cache.get(key)
881
+ if (pending === undefined) {
882
+ pending = verifyIngressPolicyApplied(client, target, signal).catch((err: unknown) => {
883
+ cache.delete(key)
884
+ throw err
885
+ })
886
+ cache.set(key, pending)
887
+ }
888
+ await pending
889
+ }
444
890
  }
445
891
 
446
892
  /** Config → the client's own access shape. Shared with `workspace.ts`. */
@@ -455,6 +901,450 @@ export function clientAccess(config: KubernetesBackendInternalConfig): Kubernete
455
901
  }
456
902
  }
457
903
 
904
+ /**
905
+ * Why an acquire was refused, in the terms an operator acts on rather than
906
+ * the terms the failure happened to arrive in.
907
+ *
908
+ * - `'api-unreachable'` — the API server could not be reached, or kept
909
+ * answering with a status that means "not now": a connect failure, a 429,
910
+ * a 5xx. Retried inside the readiness budget before it ever reaches a
911
+ * caller, so seeing it means the whole budget was spent failing.
912
+ * - `'api-timeout'` — requests were accepted and never answered, until
913
+ * `apiRequestTimeoutMs` gave up on them. Also retried first.
914
+ * - `'forbidden'` — 401 or 403. The host's ServiceAccount cannot do this;
915
+ * no amount of waiting changes that. See the RBAC section of
916
+ * `docs/sdk/kubernetes-sandbox.md`.
917
+ * - `'claim-rejected'` — the controller REFUSED the claim, and said why.
918
+ * {@link KubernetesAcquireError.controllerReason} carries its own word for
919
+ * it. This is the one that used to burn the entire readiness budget before
920
+ * failing.
921
+ * - `'capacity'` — the pod exists and cannot be placed: the scheduler
922
+ * reports `PodScheduled=False` with reason `Unschedulable`. The cluster is
923
+ * full, or nothing matches the template's placement rules.
924
+ * - `'image-pull'` — the pod was placed and its container cannot start
925
+ * because the image will not pull. Permanent until an operator fixes the
926
+ * reference or the pull credential.
927
+ * - `'not-ready'` — none of the above: the readiness budget expired with the
928
+ * cluster reporting nothing wrong. A slow cold start, a webhook, an
929
+ * admission controller, a CNI that never attached the pod.
930
+ */
931
+ export type KubernetesAcquireFailureReason =
932
+ | 'api-unreachable'
933
+ | 'api-timeout'
934
+ | 'forbidden'
935
+ | 'claim-rejected'
936
+ | 'capacity'
937
+ | 'image-pull'
938
+ | 'not-ready'
939
+
940
+ /**
941
+ * An acquire that was refused, carrying WHY in a field rather than in prose.
942
+ *
943
+ * Before this class a burst past node capacity and an API outage were the
944
+ * same plain `Error`, and a host could only tell them apart by matching
945
+ * message text that any release is free to reword. `reason` is the diagnosis,
946
+ * `retryable` is the advice that follows from it, and `cause` is the original
947
+ * failure — unmodified, so a host that already catches
948
+ * {@link ReadinessPollTimeout}, {@link KubernetesApiTimeoutError} or
949
+ * `KubernetesCredentialError` finds it there.
950
+ *
951
+ * `retryable` is about THIS acquire being worth attempting again, not about
952
+ * anything having been retried. Transient API failures are already retried
953
+ * inside the readiness budget, so a `retryable: true` that reaches a caller
954
+ * means the whole budget was spent on them.
955
+ *
956
+ * Not every acquire failure becomes one of these, and that is deliberate: a
957
+ * refusal this class cannot honestly diagnose — a malformed template, a 400
958
+ * from an admission webhook, a controller that reported Ready and named no
959
+ * sandbox — travels out as itself rather than being filed under whichever of
960
+ * the seven reasons is least wrong. `KubernetesApiError` carries the status
961
+ * for those.
962
+ */
963
+ export class KubernetesAcquireError extends Error {
964
+ override readonly name = 'KubernetesAcquireError'
965
+ readonly reason: KubernetesAcquireFailureReason
966
+ /** Whether attempting the same acquire again could plausibly succeed. */
967
+ readonly retryable: boolean
968
+ /**
969
+ * `status.conditions[Ready].reason`, verbatim, when the controller
970
+ * refused the claim — `WarmPoolNotFound`, `TemplateNotFound`,
971
+ * `InvalidMetadata`, `EnvVarsInjectionRejected`. Present only for
972
+ * `'claim-rejected'`.
973
+ */
974
+ readonly controllerReason?: string
975
+ /** The controller's own message for the same condition. */
976
+ readonly controllerMessage?: string
977
+
978
+ constructor(details: {
979
+ readonly reason: KubernetesAcquireFailureReason
980
+ readonly retryable: boolean
981
+ readonly message: string
982
+ readonly controllerReason?: string
983
+ readonly controllerMessage?: string
984
+ readonly cause?: unknown
985
+ }) {
986
+ super(details.message, details.cause !== undefined ? { cause: details.cause } : undefined)
987
+ this.reason = details.reason
988
+ this.retryable = details.retryable
989
+ if (details.controllerReason !== undefined) this.controllerReason = details.controllerReason
990
+ if (details.controllerMessage !== undefined) this.controllerMessage = details.controllerMessage
991
+ }
992
+ }
993
+
994
+ /**
995
+ * The `status.conditions[Ready].reason` values that mean the controller has
996
+ * DECIDED, so waiting is pointless.
997
+ *
998
+ * ## Where these strings came from
999
+ *
1000
+ * Not from the issue that asked for this, and not from upstream source: this
1001
+ * repo vendors none of agent-sandbox's Go, so a literal copied out of a
1002
+ * changelog is a literal nobody here can check. Each of the four was produced
1003
+ * against the deployed controller (kind v1.37.0, agent-sandbox v1.0.2,
1004
+ * 2026-09-17) by making the claim it describes and reading the condition
1005
+ * back:
1006
+ *
1007
+ * | reason | how it was produced | the controller's message |
1008
+ * |---|---|---|
1009
+ * | `WarmPoolNotFound` | claim at a pool that does not exist | `SandboxWarmPool "…" not found` |
1010
+ * | `TemplateNotFound` | claim at a pool whose template does not exist | `SandboxTemplate "…" not found` |
1011
+ * | `InvalidMetadata` | claim with an `additionalPodMetadata` label outside the allowed domains | `invalid additionalPodMetadata: …` |
1012
+ * | `EnvVarsInjectionRejected` | claim with `spec.env` against a template that forbids injection | `environment variable injection rejected: …` |
1013
+ *
1014
+ * The transient reasons seen on the SAME cluster, which must NOT be in this
1015
+ * set, were `DependenciesNotReady` (pod exists, still Pending) and
1016
+ * `DependenciesReady` (the Ready=True reason).
1017
+ *
1018
+ * ## Why an unknown reason is not terminal
1019
+ *
1020
+ * A wrong literal here fails in one of two ways, and only one of them is
1021
+ * recoverable. Too few entries: a rejected claim waits out the readiness
1022
+ * budget, which is exactly the behaviour every release before this one had.
1023
+ * Too many: an acquire that would have succeeded is refused on a guess. So
1024
+ * the set is a closed list of measured strings and everything else falls
1025
+ * through to the deadline.
1026
+ */
1027
+ // Frozen because it is exported from the package root: an array handed to
1028
+ // every consumer is one a cast can push onto, and an entry added there would
1029
+ // change fail-fast for the whole process. The `readonly string[]` annotation
1030
+ // is deliberate rather than `as const` — `includes` on a literal tuple only
1031
+ // accepts the literals, and the whole point is to ask it about a reason no
1032
+ // one here has seen.
1033
+ export const TERMINAL_CLAIM_REASONS: readonly string[] = Object.freeze([
1034
+ 'WarmPoolNotFound',
1035
+ 'TemplateNotFound',
1036
+ 'InvalidMetadata',
1037
+ 'EnvVarsInjectionRejected',
1038
+ ])
1039
+
1040
+ /**
1041
+ * The `Ready` condition a claim reports, whatever its status — the one
1042
+ * {@link isConditionTrue} deliberately cannot return, because it answers a
1043
+ * boolean question and this one needs the reason.
1044
+ */
1045
+ function readyCondition(
1046
+ conditions: readonly KubernetesCondition[] | undefined,
1047
+ ): KubernetesCondition | undefined {
1048
+ return conditions?.find((c) => c.type === READY_CONDITION)
1049
+ }
1050
+
1051
+ /**
1052
+ * What this backend asked the controller to put on the claim's pod, carried
1053
+ * into the rejection so an `InvalidMetadata` refusal can say what was sent
1054
+ * and which knob changes it.
1055
+ *
1056
+ * Context, never a second decision: whether a claim is refused at all is
1057
+ * {@link TERMINAL_CLAIM_REASONS}' answer and nobody else's, so an empty map
1058
+ * changes nothing about the class thrown, the reason on it, or when it is
1059
+ * raised.
1060
+ */
1061
+ export interface ClaimPodMetadataContext {
1062
+ readonly namespace: string
1063
+ /** `spec.additionalPodMetadata.labels`, exactly as sent. */
1064
+ readonly requestedPodLabels: Readonly<Record<string, string>>
1065
+ /** The egress profile among those labels, when one is configured. */
1066
+ readonly profile?: EgressProfileLabel
1067
+ }
1068
+
1069
+ /**
1070
+ * A claim the controller has refused, or `undefined` for one it is still
1071
+ * working on.
1072
+ *
1073
+ * `status: 'False'` alone is not a refusal — it is also what a claim looks
1074
+ * like for the whole of a cold start — so the REASON decides, against
1075
+ * {@link TERMINAL_CLAIM_REASONS}.
1076
+ *
1077
+ * ONE class comes out of here whatever the reason, and that is deliberate:
1078
+ * `InvalidMetadata` is how the controller refuses a pod label whose domain is
1079
+ * not on its allowlist — the profile's, today — and a host's `catch` must not
1080
+ * have to be written differently depending on whether a profile happens to be
1081
+ * configured. `metadata` only decides what rides along as the `cause`: a
1082
+ * {@link KubernetesPodLabelsRejectedError} naming the map that was sent
1083
+ * and the `config.egress.profileLabelKey` that moves it, which the
1084
+ * controller's own message cannot know about.
1085
+ */
1086
+ export function classifyClaimRejection(
1087
+ claim: SandboxClaimResource | undefined,
1088
+ claimName: string,
1089
+ metadata?: ClaimPodMetadataContext,
1090
+ ): KubernetesAcquireError | undefined {
1091
+ const condition = readyCondition(claim?.status?.conditions)
1092
+ if (condition === undefined || condition.status !== 'False') return undefined
1093
+ const reason = condition.reason
1094
+ if (reason === undefined || !TERMINAL_CLAIM_REASONS.includes(reason)) return undefined
1095
+ // Only for the reason the pod metadata can actually cause, and only when
1096
+ // this backend sent any: a `WarmPoolNotFound` carrying a labels-and-
1097
+ // allowlist explanation would send an operator after the wrong thing.
1098
+ const cause =
1099
+ reason === 'InvalidMetadata' &&
1100
+ metadata !== undefined &&
1101
+ Object.keys(metadata.requestedPodLabels).length > 0
1102
+ ? new KubernetesPodLabelsRejectedError(
1103
+ metadata.requestedPodLabels,
1104
+ claimName,
1105
+ metadata.namespace,
1106
+ reason,
1107
+ condition.message ?? '(the controller reported no message)',
1108
+ metadata.profile,
1109
+ )
1110
+ : undefined
1111
+ return new KubernetesAcquireError({
1112
+ reason: 'claim-rejected',
1113
+ retryable: false,
1114
+ controllerReason: reason,
1115
+ ...(condition.message !== undefined ? { controllerMessage: condition.message } : {}),
1116
+ ...(cause !== undefined ? { cause } : {}),
1117
+ message: `kubernetes: the agent-sandbox controller refused SandboxClaim ${claimName} with reason ${reason}${
1118
+ condition.message !== undefined ? `: ${condition.message}` : ''
1119
+ }. That is a decision, not a delay, so the readiness budget was not waited out.${
1120
+ cause !== undefined ? ` ${cause.message}` : ''
1121
+ }`,
1122
+ })
1123
+ }
1124
+
1125
+ /**
1126
+ * How long to wait before repeating a failed readiness read, or `undefined`
1127
+ * when the failure is not worth repeating.
1128
+ *
1129
+ * Retryable: a connect failure (the socket, not the answer), a request the
1130
+ * `apiRequestTimeoutMs` bound gave up on, a 429 (the API server's own
1131
+ * priority-and-fairness queue shedding load) and any 5xx. Not retryable: 401
1132
+ * and 403, which are a decision; 404 and 410, which are an answer; 409, which
1133
+ * a caller resolves by re-reading; and everything this backend threw itself.
1134
+ *
1135
+ * The wait is the poll's own cadence unless the server named one — then its
1136
+ * `Retry-After`, because the server knows when its queue drains and this code
1137
+ * does not. Nothing here consults a clock: the caller's deadline owns the
1138
+ * sleep, so a long `Retry-After` spends the readiness budget rather than
1139
+ * extending it.
1140
+ */
1141
+ /**
1142
+ * The largest delay a timer can hold — `2^31 - 1` ms, Node's own ceiling.
1143
+ * Above it `setTimeout` warns and fires immediately, which is the opposite of
1144
+ * what a long `Retry-After` asked for.
1145
+ */
1146
+ const MAX_RETRY_DELAY_MS = 2_147_483_647
1147
+
1148
+ export function retryDelayForApiFailure(err: unknown, pollIntervalMs: number): number | undefined {
1149
+ if (err instanceof KubernetesApiTimeoutError) return pollIntervalMs
1150
+ if (!(err instanceof KubernetesApiError)) return undefined
1151
+ if (err.transport === 'connect') return pollIntervalMs
1152
+ const status = err.status
1153
+ if (status === undefined) return undefined
1154
+ // `Retry-After` may only ever SLOW the poll down, and only within what a
1155
+ // timer can express. A header of `0`, or one naming a moment already past,
1156
+ // would otherwise turn the retry into a hot loop against a server that is
1157
+ // already shedding load — the caller's own cadence is the rate this loop
1158
+ // runs at when nothing is wrong. And a header naming a moment years away
1159
+ // overflows `setTimeout`, which then fires at once rather than never,
1160
+ // producing the same hot loop from the opposite direction. The readiness
1161
+ // deadline ends the wait either way; the clamp only stops the wait from
1162
+ // silently becoming no wait at all.
1163
+ if (status === 429 || status >= 500) {
1164
+ const asked = err.retryAfterMs ?? pollIntervalMs
1165
+ return Math.min(Math.max(asked, pollIntervalMs), MAX_RETRY_DELAY_MS)
1166
+ }
1167
+ return undefined
1168
+ }
1169
+
1170
+ /**
1171
+ * How long the one diagnostic pod read after a failed acquire may take.
1172
+ *
1173
+ * Same shape and the same argument as `runFailureCleanup`'s grace: the
1174
+ * readiness clock has already expired, so this cannot share it, and a
1175
+ * diagnosis that could hang would keep `create()` pending past the budget the
1176
+ * caller chose — for a nicer error message. One second, and a diagnosis that
1177
+ * does not arrive is simply not made.
1178
+ */
1179
+ const ACQUIRE_DIAGNOSIS_GRACE_MS = 1_000
1180
+
1181
+ /**
1182
+ * The image-pull `status.containerStatuses[].state.waiting.reason` values the
1183
+ * kubelet reports. Measured on kind v1.37.0 (2026-09-17): a container whose
1184
+ * image does not exist waits as `ErrImagePull` for the first attempts and
1185
+ * settles into `ImagePullBackOff`. The other two are the kubelet's names for
1186
+ * a pull that resolved and then failed, and for a registry that cannot be
1187
+ * reached at all.
1188
+ */
1189
+ const IMAGE_PULL_WAITING_REASONS: readonly string[] = [
1190
+ 'ErrImagePull',
1191
+ 'ImagePullBackOff',
1192
+ 'ImageInspectError',
1193
+ 'RegistryUnavailable',
1194
+ ]
1195
+
1196
+ /**
1197
+ * Ask the pod why it is not ready, once, after the budget has already gone.
1198
+ *
1199
+ * Nothing on the healthy path calls this and nothing waits on it: it runs
1200
+ * exactly when an acquire has already failed, and its whole output is a
1201
+ * better {@link KubernetesAcquireFailureReason} than `'not-ready'`. A read
1202
+ * that fails, a pod that is not there and a pod with nothing to say all
1203
+ * produce `undefined`, which leaves the reason where it was.
1204
+ *
1205
+ * It must run BEFORE the cleanup DELETE, because the pod goes away with the
1206
+ * object it belongs to.
1207
+ *
1208
+ * It takes the CALLER's signal where `runFailureCleanup` deliberately does
1209
+ * not: cleanup must finish or the cluster keeps the object, while a diagnosis
1210
+ * is only a better sentence for an error a caller who aborted will never
1211
+ * read.
1212
+ */
1213
+ async function diagnoseUnreadyPod(
1214
+ client: KubernetesClient,
1215
+ namespace: string,
1216
+ podName: string,
1217
+ callerSignal: AbortSignal | undefined,
1218
+ ): Promise<'capacity' | 'image-pull' | undefined> {
1219
+ let pod: PodResource | undefined
1220
+ try {
1221
+ const deadline = new OperationDeadline(
1222
+ ACQUIRE_DIAGNOSIS_GRACE_MS,
1223
+ 'kubernetes acquire diagnosis',
1224
+ callerSignal,
1225
+ )
1226
+ pod = await deadline.run((signal) =>
1227
+ client.request<PodResource>('GET', podPath(namespace, podName), undefined, signal),
1228
+ )
1229
+ } catch {
1230
+ // The acquire failure is the primary one and keeps its reason. A
1231
+ // diagnosis that cannot be made is not a second failure to report.
1232
+ return undefined
1233
+ }
1234
+ const scheduled = pod?.status?.conditions?.find((c) => c.type === 'PodScheduled')
1235
+ if (scheduled?.status === 'False' && scheduled.reason === 'Unschedulable') return 'capacity'
1236
+ for (const container of pod?.status?.containerStatuses ?? []) {
1237
+ const reason = container.state?.waiting?.reason
1238
+ if (reason !== undefined && IMAGE_PULL_WAITING_REASONS.includes(reason)) return 'image-pull'
1239
+ }
1240
+ return undefined
1241
+ }
1242
+
1243
+ /**
1244
+ * The refusal a caller sees, given the failure that actually happened and
1245
+ * whatever the pod had to say about it.
1246
+ *
1247
+ * Returns `undefined` for a failure none of the seven reasons describes —
1248
+ * see {@link KubernetesAcquireError} for why that is a deliberate hole rather
1249
+ * than a missing case. A {@link KubernetesAcquireError} that arrived from
1250
+ * deeper in (the claim rejection) is returned unchanged: it is already the
1251
+ * diagnosis.
1252
+ */
1253
+ export function classifyAcquireFailure(
1254
+ err: unknown,
1255
+ podDiagnosis: 'capacity' | 'image-pull' | undefined,
1256
+ ): KubernetesAcquireError | undefined {
1257
+ if (err instanceof KubernetesAcquireError) return err
1258
+ if (err instanceof KubernetesApiTimeoutError) {
1259
+ return new KubernetesAcquireError({
1260
+ reason: 'api-timeout',
1261
+ retryable: true,
1262
+ message: `kubernetes: the acquire was refused because the API server did not answer in time — ${err.message}`,
1263
+ cause: err,
1264
+ })
1265
+ }
1266
+ if (err instanceof KubernetesCredentialError) {
1267
+ return new KubernetesAcquireError({
1268
+ reason: 'forbidden',
1269
+ retryable: false,
1270
+ message: `kubernetes: the acquire was refused because the API server rejected this host's credential — ${err.message}. Check the host ServiceAccount's Role against the RBAC section of the Kubernetes sandbox documentation.`,
1271
+ cause: err,
1272
+ })
1273
+ }
1274
+ if (err instanceof KubernetesApiError) {
1275
+ // The same predicate the poll retries on, so "worth trying again" has
1276
+ // one definition and a caller cannot be told a failure is retryable
1277
+ // that the poll would have declined to retry. The interval is
1278
+ // irrelevant here — only whether an answer comes back at all.
1279
+ if (retryDelayForApiFailure(err, 1) === undefined) return undefined
1280
+ return new KubernetesAcquireError({
1281
+ reason: 'api-unreachable',
1282
+ retryable: true,
1283
+ message: `kubernetes: the acquire was refused because the API server could not serve it — ${err.message}`,
1284
+ cause: err,
1285
+ })
1286
+ }
1287
+ if (err instanceof ReadinessPollTimeout) {
1288
+ // ORDER MATTERS, and this is the order: what the cluster SAID beats
1289
+ // what the failures suggest. A pod diagnosis is a condition the API
1290
+ // server published about this pod, read after the budget had already
1291
+ // gone; `err.cause` is at best the failure the poll was still meeting
1292
+ // at that moment. On a saturated cluster both are present at once — a
1293
+ // pod nothing can schedule AND an API server shedding load — and
1294
+ // reporting `api-unreachable` there would hide the very reason this
1295
+ // function exists to produce, and would turn `image-pull`'s
1296
+ // `retryable: false` into a `true` that has a host retrying forever
1297
+ // against an image reference that will never resolve. A diagnosis also
1298
+ // cannot be stale in the way a cause can: it only exists because the
1299
+ // API server answered one more read, moments ago.
1300
+ if (podDiagnosis === 'capacity') {
1301
+ return new KubernetesAcquireError({
1302
+ reason: 'capacity',
1303
+ retryable: true,
1304
+ message: `kubernetes: the acquire was refused because its pod could not be scheduled — the cluster reports PodScheduled=False/Unschedulable. ${err.message}`,
1305
+ cause: err,
1306
+ })
1307
+ }
1308
+ if (podDiagnosis === 'image-pull') {
1309
+ return new KubernetesAcquireError({
1310
+ reason: 'image-pull',
1311
+ retryable: false,
1312
+ message: `kubernetes: the acquire was refused because its pod's container image will not pull. ${err.message}`,
1313
+ cause: err,
1314
+ })
1315
+ }
1316
+ // Nothing measured, so the failures the poll kept meeting decide: a
1317
+ // poll that spent its budget retrying API failures did not fail
1318
+ // because the sandbox was slow; it failed because the control plane
1319
+ // was. `pollForBinding` carries the failure it was STILL meeting onto
1320
+ // the timeout, and the reason follows it rather than the timeout.
1321
+ const underlying = classifyAcquireFailure(err.cause, undefined)
1322
+ if (underlying !== undefined) {
1323
+ return new KubernetesAcquireError({
1324
+ reason: underlying.reason,
1325
+ retryable: underlying.retryable,
1326
+ message: underlying.message,
1327
+ cause: err,
1328
+ })
1329
+ }
1330
+ return new KubernetesAcquireError({
1331
+ reason: 'not-ready',
1332
+ retryable: true,
1333
+ message: err.message,
1334
+ cause: err,
1335
+ })
1336
+ }
1337
+ if (err instanceof OperationDeadlineExpired) {
1338
+ return new KubernetesAcquireError({
1339
+ reason: 'not-ready',
1340
+ retryable: true,
1341
+ message: `kubernetes: the acquire ran out of readiness budget — ${err.message}`,
1342
+ cause: err,
1343
+ })
1344
+ }
1345
+ return undefined
1346
+ }
1347
+
458
1348
  /**
459
1349
  * Claim or create, wait for Ready, read the bound identity back, resolve the
460
1350
  * address and learn the pod's uid — or leave nothing behind trying.
@@ -462,12 +1352,39 @@ export function clientAccess(config: KubernetesBackendInternalConfig): Kubernete
462
1352
  * Exported because the sandbox surface is built on top of this record rather
463
1353
  * than beside it: one acquire path, one cleanup path, whatever ends up
464
1354
  * wrapping them.
1355
+ *
1356
+ * ## What it refuses with
1357
+ *
1358
+ * Every refusal this function can diagnose arrives as a
1359
+ * {@link KubernetesAcquireError} naming one of seven reasons, with the
1360
+ * original failure as its `cause`. Three things stay outside that:
1361
+ * configuration refused before anything is created
1362
+ * ({@link assertEnforceable}, {@link assertRuntimeClassIsApplicable}), a
1363
+ * caller's own abort, and a failure none of the seven reasons honestly
1364
+ * describes — see {@link KubernetesAcquireError} for why the last one is a
1365
+ * hole on purpose.
1366
+ *
1367
+ * ## What it retries, and what it will not
1368
+ *
1369
+ * A readiness GET that fails transiently — a connect failure, a request the
1370
+ * API bound gave up on, a 429, a 5xx — is repeated INSIDE the readiness
1371
+ * deadline, honouring `Retry-After`. One clock, so a retry spends the budget
1372
+ * rather than extending it, and a `create()` cannot outlive the timeout its
1373
+ * caller chose. The create POST is never retried: it is not idempotent, and a
1374
+ * POST whose answer never arrived may already have committed — which is why
1375
+ * cleanup deletes the client-owned name whatever happened.
465
1376
  */
466
1377
  export async function acquireKubernetesSandbox(
467
1378
  client: KubernetesClient,
468
1379
  config: KubernetesBackendInternalConfig,
469
1380
  options: SandboxBackendOptions,
470
1381
  readiness: { readonly timeoutMs: number; readonly pollIntervalMs: number },
1382
+ verifyIngress: IngressVerifier | undefined = buildIngressVerifier(client, config),
1383
+ egressBoundary: KubernetesEgressBoundary | undefined = buildEgressBoundary(
1384
+ client,
1385
+ config,
1386
+ config.sandboxTemplateName,
1387
+ ),
471
1388
  ): Promise<KubernetesAcquisition> {
472
1389
  options.signal?.throwIfAborted()
473
1390
  assertEnforceable(options)
@@ -527,70 +1444,241 @@ export async function acquireKubernetesSandbox(
527
1444
  config.warmPoolName !== undefined
528
1445
  ? claimCollectionPath(namespace)
529
1446
  : sandboxCollectionPath(namespace)
530
- const createBody =
531
- config.warmPoolName !== undefined
532
- ? buildClaimBody(namespace, objectName, config.warmPoolName, shutdownTime)
533
- : buildSandboxBody({
534
- namespace,
535
- name: objectName,
536
- template: await deadline.run((signal) =>
537
- readSandboxTemplate(client, namespace, config.sandboxTemplateName, signal),
538
- ),
539
- sandboxTemplateName: config.sandboxTemplateName,
540
- shutdownTime,
541
- ...(config.runtimeClassName !== undefined
542
- ? { runtimeClassName: config.runtimeClassName }
543
- : {}),
544
- })
1447
+ // Composed ONCE, here, and handed to whichever body builder runs below:
1448
+ // the claim's `additionalPodMetadata.labels` and a direct Sandbox's pod
1449
+ // template metadata are the same map, and the translated policy's selector
1450
+ // is built from the same resolution. See `composeAdditionalPodLabels`.
1451
+ // Empty — the only case before a profile is configured — means every body
1452
+ // below is byte for byte what it was.
1453
+ // The per-sandbox selector label rides in the SAME map, through the same
1454
+ // composer, as a second key rather than a second construction — the whole
1455
+ // reason `composeAdditionalPodLabels` takes an `extra`. Its value is the
1456
+ // name of the object this acquire is about to create, which is unique per
1457
+ // acquire and known BEFORE the POST: that is what lets the label travel as
1458
+ // claim-time pod metadata (warm-safe, no cold start) instead of as a patch
1459
+ // to a running pod this backend has no verb for.
1460
+ const perSandboxLabelKey = perSandboxEgressLabelKey(config.egress)
1461
+ const podLabels = composeAdditionalPodLabels(
1462
+ config.egress,
1463
+ perSandboxLabelKey !== undefined ? { [perSandboxLabelKey]: objectName } : undefined,
1464
+ )
1465
+ const profile = egressProfileLabel(config.egress)
1466
+ // Read off the create reply, and off the readiness polls if that reply
1467
+ // carried no object — it is the uid a per-sandbox policy's
1468
+ // `ownerReferences` names and the suffix of its name, so the cluster can
1469
+ // garbage-collect the policy with the object this backend owns.
1470
+ let ownerUid: string | undefined
1471
+ const buildCreateBody = async (): Promise<Record<string, unknown>> => {
1472
+ if (config.warmPoolName !== undefined) {
1473
+ return buildClaimBody({
1474
+ namespace,
1475
+ name: objectName,
1476
+ warmPoolName: config.warmPoolName,
1477
+ shutdownTime,
1478
+ ...(config.claimLabels !== undefined ? { labels: config.claimLabels } : {}),
1479
+ podLabels,
1480
+ })
1481
+ }
1482
+ const template = await deadline.run((signal) =>
1483
+ readSandboxTemplate(client, namespace, config.sandboxTemplateName, signal),
1484
+ )
1485
+ // A direct Sandbox's pod labels are decided HERE, by the body below,
1486
+ // so both network boundaries are checked against the real labels before
1487
+ // anything is created — a refusal leaves no Sandbox and no PVC behind
1488
+ // rather than one of each to clean up. This is EARLIER than the plan
1489
+ // for the egress union check asked for (it said "after binding", for
1490
+ // the bound pod's labels); a pool-less Sandbox's labels are knowable
1491
+ // before the POST, and refusing with nothing created is strictly
1492
+ // better than refusing with an object to clean up.
1493
+ const directPodLabels = sandboxPodLabels(template, config.sandboxTemplateName, podLabels)
1494
+ const directSubject = `to create Sandbox ${objectName} in namespace ${namespace}`
1495
+ if (verifyIngress !== undefined) {
1496
+ await deadline.run((signal) => verifyIngress(directPodLabels, directSubject, signal))
1497
+ }
1498
+ if (egressBoundary !== undefined) {
1499
+ await deadline.run((signal) =>
1500
+ egressBoundary.verifyUnion(directPodLabels, directSubject, signal),
1501
+ )
1502
+ }
1503
+ return buildSandboxBody({
1504
+ namespace,
1505
+ name: objectName,
1506
+ template,
1507
+ sandboxTemplateName: config.sandboxTemplateName,
1508
+ shutdownTime,
1509
+ podLabels,
1510
+ ...(config.runtimeClassName !== undefined
1511
+ ? { runtimeClassName: config.runtimeClassName }
1512
+ : {}),
1513
+ })
1514
+ }
1515
+
1516
+ let createBody: Record<string, unknown>
1517
+ try {
1518
+ createBody = await buildCreateBody()
1519
+ } catch (err) {
1520
+ // Nothing exists yet, so there is nothing to clean up and no pod to
1521
+ // ask — but a 403 on the template read is still a `forbidden` acquire,
1522
+ // and a caller should not have to tell that apart by where it
1523
+ // happened.
1524
+ throw classifyAcquireFailure(err, undefined) ?? err
1525
+ }
1526
+
1527
+ // The pod the diagnosis below asks, when there is one. A directly created
1528
+ // Sandbox is backed by a pod of its own name; a CLAIM's pod is not knowable
1529
+ // until the controller has named a sandbox in `status.sandbox`, which it
1530
+ // does before Ready on a cold start and never on a rejected claim.
1531
+ let diagnosablePodName: string | undefined =
1532
+ config.warmPoolName !== undefined ? undefined : objectName
1533
+ // One definition of "worth trying again", shared by the poll that retries
1534
+ // and the classification that reports — see {@link retryDelayForApiFailure}.
1535
+ const pollBehaviour: ReadinessPollBehaviour = {
1536
+ retryDelayFor: (err) => retryDelayForApiFailure(err, readiness.pollIntervalMs),
1537
+ }
545
1538
 
546
1539
  try {
547
1540
  // Inside the cleanup block: a POST that fails client-side may still have
548
1541
  // committed, so the only safe assumption is that the object exists.
549
- await deadline.run((signal) => client.request('POST', createPath, createBody, signal))
1542
+ const created = await deadline.run((signal) =>
1543
+ client.request<{ readonly metadata?: { readonly uid?: string } }>(
1544
+ 'POST',
1545
+ createPath,
1546
+ createBody,
1547
+ signal,
1548
+ ),
1549
+ )
1550
+ ownerUid ??= created?.metadata?.uid
550
1551
  const binding =
551
1552
  config.warmPoolName !== undefined
552
1553
  ? await pollForBinding(
553
- async (signal) =>
554
- bindingFromClaim(
555
- await client.request<SandboxClaimResource>(
556
- 'GET',
557
- claimPath(namespace, objectName),
558
- undefined,
559
- signal,
560
- ),
561
- objectName,
562
- ),
1554
+ async (signal) => {
1555
+ const claim = await client.request<SandboxClaimResource>(
1556
+ 'GET',
1557
+ claimPath(namespace, objectName),
1558
+ undefined,
1559
+ signal,
1560
+ )
1561
+ diagnosablePodName = claim?.status?.sandbox?.name ?? diagnosablePodName
1562
+ ownerUid ??= claim?.metadata?.uid
1563
+ // The pod labels this backend asked the controller for
1564
+ // travel into the read, not because the fail-fast needs
1565
+ // them — `InvalidMetadata` is already one of
1566
+ // `TERMINAL_CLAIM_REASONS`, and that is the ONE
1567
+ // taxonomy a refused claim is reported under — but so
1568
+ // the refusal can carry what was actually sent and how
1569
+ // to change it. See {@link ClaimPodMetadataContext}.
1570
+ return bindingFromClaim(claim, objectName, {
1571
+ namespace,
1572
+ requestedPodLabels: podLabels,
1573
+ ...(profile !== undefined ? { profile } : {}),
1574
+ })
1575
+ },
563
1576
  deadline,
564
1577
  readiness,
565
1578
  `claim ${objectName}`,
1579
+ pollBehaviour,
566
1580
  )
567
1581
  : await pollForBinding(
568
- async (signal) =>
569
- bindingFromSandbox(
570
- await client.request<SandboxResource>(
571
- 'GET',
572
- sandboxPath(namespace, objectName),
573
- undefined,
574
- signal,
575
- ),
576
- ),
1582
+ async (signal) => {
1583
+ const sandbox = await client.request<SandboxResource>(
1584
+ 'GET',
1585
+ sandboxPath(namespace, objectName),
1586
+ undefined,
1587
+ signal,
1588
+ )
1589
+ ownerUid ??= sandbox?.metadata?.uid
1590
+ return bindingFromSandbox(sandbox)
1591
+ },
577
1592
  deadline,
578
1593
  readiness,
579
1594
  `sandbox ${objectName}`,
1595
+ pollBehaviour,
580
1596
  )
581
1597
 
582
- const token = await deadline.run((signal) =>
583
- readPodBindToken(client, namespace, binding, signal),
1598
+ const agentPort = config.agentPort ?? DEFAULT_AGENT_PORT
1599
+ const mode = config.agentAddress ?? 'service'
1600
+ // One read, two facts: the bind token and — under `'pod-ip'` — the
1601
+ // address, off the same pod. See {@link readAddressedPod}, which under
1602
+ // a configured profile also waits for the label the controller
1603
+ // patches onto the bound pod before anything is admitted.
1604
+ const pod = await readAddressedPod(
1605
+ client,
1606
+ namespace,
1607
+ binding,
1608
+ deadline,
1609
+ readiness,
1610
+ mode,
1611
+ podLabels,
584
1612
  )
1613
+ // The one refusal this capability must not skip: an unlabelled pod
1614
+ // handed back runs under whatever policy DOES select it while the host
1615
+ // believes it is on a narrower profile. This throws INSIDE the try
1616
+ // block, so the cleanup below releases the claim (or deletes the
1617
+ // Sandbox) exactly as a failed privilege probe does — and the probe
1618
+ // itself, which runs in `admitProbedSandbox` after this function
1619
+ // returns, is therefore never reached with the label unobserved.
1620
+ assertRequestedPodLabelsObserved(
1621
+ pod,
1622
+ podLabels,
1623
+ `Sandbox ${binding.name} in namespace ${namespace}`,
1624
+ )
1625
+ // A CLAIMED sandbox's pod was built from the pool's own template, so
1626
+ // its labels are not knowable until the controller has bound one.
1627
+ // Checking here rather than not at all is the trade: a refusal
1628
+ // releases the claim through the cleanup below, which is the same
1629
+ // path a failed privilege probe takes.
1630
+ if (config.warmPoolName !== undefined) {
1631
+ const boundSubject = `the pod bound to Sandbox ${binding.name} in namespace ${namespace}`
1632
+ if (verifyIngress !== undefined) {
1633
+ await deadline.run((signal) => verifyIngress(pod.labels ?? {}, boundSubject, signal))
1634
+ }
1635
+ // The egress union check asks the same question of the same labels
1636
+ // — which policies select THIS pod — so it runs at the same two
1637
+ // points, and a refusal here releases the claim through the cleanup
1638
+ // below, exactly as a failed privilege probe does.
1639
+ if (egressBoundary !== undefined) {
1640
+ await deadline.run((signal) =>
1641
+ egressBoundary.verifyUnion(pod.labels ?? {}, boundSubject, signal),
1642
+ )
1643
+ }
1644
+ }
1645
+ // Only when the capability is configured, and only after the pod has
1646
+ // been confirmed to carry the selector label: a policy written for a
1647
+ // pod that never got the label would select nothing while the caller
1648
+ // was told its egress had been narrowed.
1649
+ const owner =
1650
+ perSandboxLabelKey === undefined
1651
+ ? undefined
1652
+ : {
1653
+ kind: (config.warmPoolName !== undefined ? 'SandboxClaim' : 'Sandbox') as
1654
+ | 'SandboxClaim'
1655
+ | 'Sandbox',
1656
+ name: objectName,
1657
+ uid: assertOwnerUid(ownerUid, objectName, namespace),
1658
+ }
585
1659
  return {
586
1660
  binding,
587
- agent: resolveAgentAddress(binding, config.agentPort ?? DEFAULT_AGENT_PORT, token),
1661
+ agent: resolveAgentAddress(binding, agentPort, pod.uid, {
1662
+ mode,
1663
+ ...(pod.podIP !== undefined ? { podIP: pod.podIP } : {}),
1664
+ }),
1665
+ // Only the literal-address mode gets one — see the field.
1666
+ ...(mode === 'pod-ip'
1667
+ ? { refreshAgent: buildAgentAddressRefresh(client, namespace, binding, agentPort, mode) }
1668
+ : {}),
588
1669
  ownedPath,
1670
+ ...(owner !== undefined ? { owner, perSandboxLabelValue: objectName } : {}),
589
1671
  ttlSeconds,
590
1672
  release,
591
1673
  renew,
592
1674
  }
593
1675
  } catch (err) {
1676
+ // Asked BEFORE cleanup, because the pod goes away with the object, and
1677
+ // only for a timeout — every other failure already knows what it was.
1678
+ const podDiagnosis =
1679
+ err instanceof ReadinessPollTimeout && diagnosablePodName !== undefined
1680
+ ? await diagnoseUnreadyPod(client, namespace, diagnosablePodName, options.signal)
1681
+ : undefined
594
1682
  // One cleanup for every way out of the block above, on its own short
595
1683
  // budget: the readiness clock has already expired in the common case,
596
1684
  // so spending it again would either skip cleanup or leave `create()`
@@ -598,35 +1686,258 @@ export async function acquireKubernetesSandbox(
598
1686
  await runFailureCleanup(async (signal) => {
599
1687
  await release(signal)
600
1688
  })
601
- throw err
1689
+ throw classifyAcquireFailure(err, podDiagnosis) ?? err
602
1690
  }
603
1691
  }
604
1692
 
605
1693
  /**
606
- * The claim body, in full. Everything absent from it is absent on purpose:
607
- * no `env`, no `volumeClaimTemplates`, no `additionalPodMetadata`. Either of
608
- * the first two forces a cold start upstream and takes the warm pool away.
1694
+ * The claim body, in full. Everything absent from it is absent on purpose: no
1695
+ * `env` and no `volumeClaimTemplates`, because either forces a cold start
1696
+ * upstream and takes the warm pool away. `additionalPodMetadata` is the one
1697
+ * piece of claim-time metadata that does NOT — see `podLabels` below.
609
1698
  *
610
1699
  * `shutdownTime` + `shutdownPolicy: 'Delete'` is the leak guard: it bounds the
611
1700
  * object by the wall clock whatever the host does, so a host that dies
612
1701
  * mid-acquire costs the cluster one TTL rather than one leaked sandbox
613
1702
  * forever. `ttlSecondsAfterFinished` deliberately does NOT appear — its timer
614
1703
  * starts from the Finished condition, which a crashed host never reaches.
1704
+ *
1705
+ * `labels` (from {@link KubernetesBackendInternalConfig.claimLabels}) is the
1706
+ * host's own bookkeeping and goes ONLY onto `metadata.labels` — never into
1707
+ * `additionalPodMetadata`, because those are POD labels that change what
1708
+ * selectors match a running sandbox, and a host's crash-recovery identity has
1709
+ * no business doing that. Absent or empty, the body is exactly what it was
1710
+ * before `claimLabels` existed.
1711
+ *
1712
+ * `podLabels` is the other map, on the other object: whatever
1713
+ * `composeAdditionalPodLabels` produced — the egress profile today — which
1714
+ * the controller merges onto the pod it binds. It is the one piece of
1715
+ * claim-time metadata that does NOT cost a cold start, which is why `env` and
1716
+ * `volumeClaimTemplates` are still absent from this body and this is not.
1717
+ * Empty, `additionalPodMetadata` does not appear at all and the body is byte
1718
+ * for byte what it always was.
615
1719
  */
616
- function buildClaimBody(
617
- namespace: string,
618
- name: string,
619
- warmPoolName: string,
620
- shutdownTime: string,
621
- ): Record<string, unknown> {
1720
+ interface ClaimBodyOptions {
1721
+ readonly namespace: string
1722
+ readonly name: string
1723
+ readonly warmPoolName: string
1724
+ readonly shutdownTime: string
1725
+ /** `metadata.labels` on the CLAIM. See above. */
1726
+ readonly labels?: Record<string, string>
1727
+ /** `spec.additionalPodMetadata.labels` — the POD's. See above. */
1728
+ readonly podLabels?: Readonly<Record<string, string>>
1729
+ }
1730
+
1731
+ function buildClaimBody(options: ClaimBodyOptions): Record<string, unknown> {
1732
+ const { labels, podLabels } = options
622
1733
  return {
623
1734
  apiVersion: `${SANDBOX_EXTENSIONS_API_GROUP}/${SANDBOX_API_VERSION}`,
624
1735
  kind: 'SandboxClaim',
625
- metadata: { name, namespace },
1736
+ metadata: {
1737
+ name: options.name,
1738
+ namespace: options.namespace,
1739
+ ...(labels !== undefined && Object.keys(labels).length > 0 ? { labels } : {}),
1740
+ },
626
1741
  spec: {
627
- warmPoolRef: { name: warmPoolName },
628
- lifecycle: { shutdownTime, shutdownPolicy: 'Delete' },
1742
+ warmPoolRef: { name: options.warmPoolName },
1743
+ ...(podLabels !== undefined && Object.keys(podLabels).length > 0
1744
+ ? { additionalPodMetadata: { labels: podLabels } }
1745
+ : {}),
1746
+ lifecycle: { shutdownTime: options.shutdownTime, shutdownPolicy: 'Delete' },
1747
+ },
1748
+ }
1749
+ }
1750
+
1751
+ /**
1752
+ * Refuse a bound pod that does not carry EVERY label this backend asked the
1753
+ * controller to put on it — the egress profile today, and whatever else
1754
+ * `composeAdditionalPodLabels` later contributes to the same map.
1755
+ *
1756
+ * The same set {@link readAddressedPod} waits for, deliberately: a wait that
1757
+ * covered more than the refusal would burn the readiness budget on a label
1758
+ * nothing then checked, and a refusal that covered more than the wait would
1759
+ * refuse a pod that had simply not been patched yet. An empty map (the
1760
+ * unprofiled path) checks nothing and returns.
1761
+ *
1762
+ * Called after {@link readAddressedPod} has already waited on the readiness
1763
+ * deadline, so reaching a missing label here means it never arrived, not that
1764
+ * it had not arrived yet.
1765
+ */
1766
+ function assertRequestedPodLabelsObserved(
1767
+ pod: KubernetesBoundPod,
1768
+ requested: Readonly<Record<string, string>>,
1769
+ subject: string,
1770
+ ): void {
1771
+ for (const [key, value] of Object.entries(requested)) {
1772
+ if (pod.labels?.[key] === value) continue
1773
+ throw new KubernetesPodLabelNotObservedError({ key, value }, subject, pod.labels ?? {})
1774
+ }
1775
+ }
1776
+
1777
+ /**
1778
+ * The uid of the object this acquire created, or a refusal naming what was
1779
+ * missing.
1780
+ *
1781
+ * Only reached when `config.egress.perSandbox` is configured, and only after
1782
+ * a create reply and every readiness poll have been read without one — the
1783
+ * API server assigns `metadata.uid` on admission and returns the object it
1784
+ * created, so an object with no uid is an API server that answered something
1785
+ * other than what it was asked for. Refusing is right: without the uid there
1786
+ * is no owner reference, and a per-sandbox policy with no owner is one the
1787
+ * cluster never collects.
1788
+ */
1789
+ function assertOwnerUid(uid: string | undefined, objectName: string, namespace: string): string {
1790
+ if (uid !== undefined && uid !== '') return uid
1791
+ throw new KubernetesOwnerUidMissingError(objectName, namespace)
1792
+ }
1793
+
1794
+ /**
1795
+ * Options for {@link releaseKubernetesTaskSandboxes}.
1796
+ */
1797
+ export interface KubernetesReleaseTaskSandboxesOptions {
1798
+ /**
1799
+ * Required, and refused if empty — see {@link releaseKubernetesTaskSandboxes}.
1800
+ * The same selector syntax a `kubectl get --selector` takes, e.g.
1801
+ * `sandbox.namzu.ai/host-instance=host-a`.
1802
+ */
1803
+ readonly labelSelector: string
1804
+ readonly signal?: AbortSignal
1805
+ }
1806
+
1807
+ /**
1808
+ * Recover a crashed host's claims: LIST every `SandboxClaim` carrying
1809
+ * `labelSelector`, `DELETE` each, and report what was removed.
1810
+ *
1811
+ * This deletes CLAIMS only. The controller's own garbage collection —
1812
+ * ownerReferences from claim to the Sandbox it bound, and from Sandbox to
1813
+ * Pod and Service — takes the rest down behind it; nothing here reads or
1814
+ * touches a Sandbox or a Pod directly. A claim already gone (raced by the
1815
+ * controller's own TTL reaper, or a second release call) counts as removed
1816
+ * rather than a failure, the same convention every other DELETE in this
1817
+ * backend follows.
1818
+ *
1819
+ * `labelSelector` is REQUIRED and refused, synchronously, before a single
1820
+ * request goes out, if it is absent or empty: a release that could fall back
1821
+ * to matching every claim (or every claim of the template) would delete a
1822
+ * live fleet's work the first time a caller passed one by mistake. There is
1823
+ * no default selector for exactly this reason.
1824
+ */
1825
+ export async function releaseKubernetesTaskSandboxes(
1826
+ config: KubernetesBackendInternalConfig,
1827
+ options: KubernetesReleaseTaskSandboxesOptions,
1828
+ ): Promise<{ readonly deleted: number; readonly names: readonly string[] }> {
1829
+ if (options.labelSelector === '') {
1830
+ throw new Error(
1831
+ 'kubernetes: releaseKubernetesTaskSandboxes requires a non-empty labelSelector — a release with no selector would delete every SandboxClaim in the namespace, including ones a live host still owns. Pass the selector that names only the claims you mean to recover.',
1832
+ )
1833
+ }
1834
+ options.signal?.throwIfAborted()
1835
+ const namespace = config.namespace
1836
+ const client = createKubernetesClient(clientAccess(config), clientOptions(config))
1837
+ const list = await client.request<SandboxClaimListResource>(
1838
+ 'GET',
1839
+ claimListPath(namespace, options.labelSelector),
1840
+ undefined,
1841
+ options.signal,
1842
+ )
1843
+ const names = (list?.items ?? [])
1844
+ .map((claim) => claim.metadata?.name)
1845
+ .filter((name): name is string => typeof name === 'string' && name !== '')
1846
+ // Concurrent, not one at a time: a crash-recovery release can carry a
1847
+ // whole host's worth of claims, and nothing here needs the ordering a
1848
+ // sequential loop would impose — each DELETE is independent and
1849
+ // idempotent (an already-gone claim is tolerated below). A non-tolerated
1850
+ // failure still rejects the whole call, exactly as a sequential loop
1851
+ // would have on its first such failure.
1852
+ await Promise.all(
1853
+ names.map(async (name) => {
1854
+ try {
1855
+ await client.request('DELETE', claimPath(namespace, name), undefined, options.signal)
1856
+ } catch (err) {
1857
+ if (!(err instanceof KubernetesAlreadyGoneError)) throw err
1858
+ }
1859
+ }),
1860
+ )
1861
+ return { deleted: names.length, names }
1862
+ }
1863
+
1864
+ /** Result of {@link readKubernetesTaskCapacity}. */
1865
+ export interface KubernetesTaskCapacity {
1866
+ /** `config.warmPoolName`'s own replica counts, straight off its `status`/`spec`. */
1867
+ readonly warmPool: {
1868
+ /** `status.readyReplicas`. `0` when the field is absent (a brand-new or empty pool). */
1869
+ readonly ready: number
1870
+ /** `spec.replicas`. `0` when the field is absent. */
1871
+ readonly desired: number
1872
+ }
1873
+ /** Every `SandboxClaim` in the namespace bound to `config.warmPoolName`, whatever its state. */
1874
+ readonly activeClaims: number
1875
+ /** Every Pod in the namespace currently in phase `Pending`. */
1876
+ readonly pendingPods: number
1877
+ }
1878
+
1879
+ /** Options for {@link readKubernetesTaskCapacity}. */
1880
+ export interface KubernetesReadTaskCapacityOptions {
1881
+ readonly signal?: AbortSignal
1882
+ }
1883
+
1884
+ /**
1885
+ * Read task-pool headroom before admitting more work: three GETs, no writes.
1886
+ *
1887
+ * `config.warmPoolName` is required — this reads the exact object a claim's
1888
+ * `warmPoolRef` names, so a pool-less backend (every create is a direct
1889
+ * Sandbox) has no pool to report on. `activeClaims` is every claim in the
1890
+ * namespace whose `spec.warmPoolRef.name` matches this pool, counted rather
1891
+ * than trusted from a label, because a claim's `warmPoolRef` is the one
1892
+ * field the API itself guarantees. `pendingPods` is every Pod in the
1893
+ * namespace still in phase `Pending` — a coarse but honest signal of
1894
+ * in-flight scale-up the ready-replica count alone does not carry, on
1895
+ * either the warm or the pool-less path.
1896
+ */
1897
+ export async function readKubernetesTaskCapacity(
1898
+ config: KubernetesBackendInternalConfig,
1899
+ options?: KubernetesReadTaskCapacityOptions,
1900
+ ): Promise<KubernetesTaskCapacity> {
1901
+ options?.signal?.throwIfAborted()
1902
+ if (config.warmPoolName === undefined) {
1903
+ throw new Error(
1904
+ 'kubernetes: readKubernetesTaskCapacity requires config.warmPoolName — there is no SandboxWarmPool to report on for a backend that creates every sandbox directly.',
1905
+ )
1906
+ }
1907
+ const namespace = config.namespace
1908
+ const warmPoolName = config.warmPoolName
1909
+ const client = createKubernetesClient(clientAccess(config), clientOptions(config))
1910
+ const [pool, claims, pods] = await Promise.all([
1911
+ client.request<SandboxWarmPoolResource>(
1912
+ 'GET',
1913
+ warmPoolPath(namespace, warmPoolName),
1914
+ undefined,
1915
+ options?.signal,
1916
+ ),
1917
+ client.request<SandboxClaimListResource>(
1918
+ 'GET',
1919
+ claimCollectionPath(namespace),
1920
+ undefined,
1921
+ options?.signal,
1922
+ ),
1923
+ client.request<PodListResource>(
1924
+ 'GET',
1925
+ podCollectionPath(namespace),
1926
+ undefined,
1927
+ options?.signal,
1928
+ ),
1929
+ ])
1930
+ const activeClaims = (claims?.items ?? []).filter(
1931
+ (claim) => claim.spec?.warmPoolRef?.name === warmPoolName,
1932
+ ).length
1933
+ const pendingPods = (pods?.items ?? []).filter((pod) => pod.status?.phase === 'Pending').length
1934
+ return {
1935
+ warmPool: {
1936
+ ready: pool?.status?.readyReplicas ?? 0,
1937
+ desired: pool?.spec?.replicas ?? 0,
629
1938
  },
1939
+ activeClaims,
1940
+ pendingPods,
630
1941
  }
631
1942
  }
632
1943
 
@@ -649,6 +1960,34 @@ export interface SandboxBodyOptions {
649
1960
  * sets it, because an unbounded task sandbox is a leak.
650
1961
  */
651
1962
  readonly shutdownTime?: string
1963
+ /**
1964
+ * Annotations to stamp on the Sandbox's OWN metadata at creation.
1965
+ *
1966
+ * One caller, and everything it writes is a fact the object has to carry
1967
+ * from the moment it exists rather than from its first patch: the holder
1968
+ * epoch of a workspace created under one, so there is no window in which
1969
+ * it stands unfenced, and the revision of the pod template it was built
1970
+ * from, so there is no window in which it claims none. Both are
1971
+ * `workspace.ts`'s — see `HOLDER_EPOCH_ANNOTATION_KEY` and
1972
+ * `POD_TEMPLATE_HASH_ANNOTATION_KEY`.
1973
+ *
1974
+ * Absent, the body is byte for byte what it always was, which is what
1975
+ * keeps every task sandbox's create unchanged — a task sandbox is
1976
+ * ephemeral, so it has no revision to drift from and nothing to fence.
1977
+ */
1978
+ readonly annotations?: Readonly<Record<string, string>>
1979
+ /**
1980
+ * Extra labels stamped onto the POD template's metadata, beside the
1981
+ * template label this body always adds.
1982
+ *
1983
+ * The same map `buildClaimBody` puts on a claim's
1984
+ * `additionalPodMetadata.labels`, from the same
1985
+ * `composeAdditionalPodLabels` call — a direct Sandbox has no controller
1986
+ * to merge them for it, so the create body stamps them itself and the two
1987
+ * paths produce one set of pod labels. Absent or empty, the pod template
1988
+ * is byte for byte what it was.
1989
+ */
1990
+ readonly podLabels?: Readonly<Record<string, string>>
652
1991
  }
653
1992
 
654
1993
  /**
@@ -681,22 +2020,14 @@ export interface SandboxBodyOptions {
681
2020
  * label the translated policy is built to match.
682
2021
  */
683
2022
  export function buildSandboxBody(options: SandboxBodyOptions): Record<string, unknown> {
684
- const podTemplate = options.template.podTemplate
685
- const spec =
686
- options.runtimeClassName !== undefined
687
- ? { ...podTemplate.spec, runtimeClassName: options.runtimeClassName }
688
- : { ...podTemplate.spec }
689
- const metadata = {
690
- ...podTemplate.metadata,
691
- labels: {
692
- ...podTemplate.metadata?.labels,
693
- ...sandboxTemplateLabel(options.sandboxTemplateName),
694
- },
695
- }
696
2023
  return {
697
2024
  apiVersion: `${SANDBOX_API_GROUP}/${SANDBOX_API_VERSION}`,
698
2025
  kind: 'Sandbox',
699
- metadata: { name: options.name, namespace: options.namespace },
2026
+ metadata: {
2027
+ name: options.name,
2028
+ namespace: options.namespace,
2029
+ ...(options.annotations !== undefined ? { annotations: options.annotations } : {}),
2030
+ },
700
2031
  spec: {
701
2032
  operatingMode: 'Running',
702
2033
  service: true,
@@ -706,11 +2037,87 @@ export function buildSandboxBody(options: SandboxBodyOptions): Record<string, un
706
2037
  ...(options.template.volumeClaimTemplates !== undefined
707
2038
  ? { volumeClaimTemplates: options.template.volumeClaimTemplates }
708
2039
  : {}),
709
- podTemplate: { ...podTemplate, metadata, spec },
2040
+ podTemplate: sandboxPodTemplate(
2041
+ options.template,
2042
+ options.sandboxTemplateName,
2043
+ options.runtimeClassName,
2044
+ options.podLabels,
2045
+ ),
710
2046
  },
711
2047
  }
712
2048
  }
713
2049
 
2050
+ /**
2051
+ * The `spec.podTemplate` a directly created Sandbox carries: the template's,
2052
+ * with this backend's overlays — the template label and whatever
2053
+ * {@link composeAdditionalPodLabels} produced ({@link sandboxPodLabels}), and
2054
+ * the configured `runtimeClassName`.
2055
+ *
2056
+ * Its own function because it is now built twice: once into the create POST
2057
+ * by {@link buildSandboxBody}, and once into the JSON Patch that refreshes a
2058
+ * standing workspace's pod template (`workspace.ts`). Two expressions of the
2059
+ * same overlay would drift, and the one that drifted would report a workspace
2060
+ * as off-template forever — the hash under
2061
+ * `sandbox.namzu.ai/pod-template-hash` is taken over exactly this object, so
2062
+ * a second spelling is a second revision.
2063
+ *
2064
+ * `podLabels` is on this signature rather than only on the create body for
2065
+ * exactly that reason. A refresh rewrites `/spec/podTemplate` WHOLE, so a
2066
+ * refresh built without them would PATCH the egress profile off a pod
2067
+ * template that carries it — the replacement pod would come up selected by no
2068
+ * per-profile policy, on a path where nothing re-checks the label, and the
2069
+ * revision stamped beside it would be taken over a template the POST never
2070
+ * writes, so `templateCurrent` would report drift forever.
2071
+ */
2072
+ export function sandboxPodTemplate(
2073
+ template: SandboxTemplateCopy,
2074
+ sandboxTemplateName: string,
2075
+ runtimeClassName?: string,
2076
+ podLabels?: Readonly<Record<string, string>>,
2077
+ ): SandboxPodTemplate {
2078
+ const podTemplate = template.podTemplate
2079
+ const spec =
2080
+ runtimeClassName !== undefined
2081
+ ? { ...podTemplate.spec, runtimeClassName }
2082
+ : { ...podTemplate.spec }
2083
+ return {
2084
+ ...podTemplate,
2085
+ metadata: {
2086
+ ...podTemplate.metadata,
2087
+ labels: sandboxPodLabels(template, sandboxTemplateName, podLabels),
2088
+ },
2089
+ spec,
2090
+ }
2091
+ }
2092
+
2093
+ /**
2094
+ * The labels a directly created Sandbox's pod will carry: whatever the
2095
+ * template declares, plus this backend's own template label, which always
2096
+ * wins because it is the label a policy selector is built to match.
2097
+ *
2098
+ * Its own function because the ingress check has to reason about EXACTLY the
2099
+ * labels {@link buildSandboxBody} stamps, before the POST that stamps them.
2100
+ * Two expressions of the same rule would be one rename away from a check that
2101
+ * verifies a pod nobody creates.
2102
+ *
2103
+ * `extra` is `composeAdditionalPodLabels`'s map — the egress profile today.
2104
+ * It is applied LAST, and so wins over both, for the same reason the template
2105
+ * label wins over the copied template's own: it is a label the translated
2106
+ * policy's selector is built to match, and a pod that matched the selector
2107
+ * only sometimes would be a boundary that applied only sometimes.
2108
+ */
2109
+ export function sandboxPodLabels(
2110
+ template: SandboxTemplateCopy,
2111
+ sandboxTemplateName: string,
2112
+ extra?: Readonly<Record<string, string>>,
2113
+ ): Readonly<Record<string, string>> {
2114
+ return {
2115
+ ...template.podTemplate.metadata?.labels,
2116
+ ...sandboxTemplateLabel(sandboxTemplateName),
2117
+ ...extra,
2118
+ }
2119
+ }
2120
+
714
2121
  /** The two halves of a `SandboxTemplate` a directly created Sandbox copies. */
715
2122
  export interface SandboxTemplateCopy {
716
2123
  readonly podTemplate: SandboxPodTemplate
@@ -746,17 +2153,40 @@ export async function readSandboxTemplate(
746
2153
  /**
747
2154
  * Where the agent answers.
748
2155
  *
749
- * Its own function, and the Service FQDN wins over a pod IP, because the
750
- * address outlives the pod: a suspended-then-resumed workspace comes back as a
751
- * new pod with a new IP behind the same name, and the transport re-resolves
752
- * the name on every dial. A literal IP baked into a long-lived handle is the
753
- * bug that would produce.
2156
+ * Its own function, and under the default mode the Service FQDN wins over a
2157
+ * pod IP, because that address outlives the pod: a suspended-then-resumed
2158
+ * workspace comes back as a new pod with a new IP behind the same name, and
2159
+ * the transport re-resolves the name on every dial. A literal IP baked into a
2160
+ * long-lived handle is the bug that would produce — and it is exactly the bug
2161
+ * `'pod-ip'` accepts, deliberately, in exchange for an address a host outside
2162
+ * the cluster can resolve at all. That mode pays for it by re-reading the IP
2163
+ * on every resume and once after a failed connect.
2164
+ *
2165
+ * `'pod-ip'` takes the address from `pod`, the record the bind token was just
2166
+ * read out of, and NEVER falls back to `binding.podIPs`. The Sandbox's status
2167
+ * is a second source that can name a different pod — the one a resume is
2168
+ * replacing — and an address from one pod with a token from another is the
2169
+ * mismatch that arrives as a flat `unauthorized`.
754
2170
  */
755
2171
  export function resolveAgentAddress(
756
2172
  binding: KubernetesSandboxBinding,
757
2173
  agentPort: number,
758
2174
  token: string,
2175
+ options: {
2176
+ readonly mode?: KubernetesAgentAddressMode
2177
+ /** The live pod the token came from. Required by `'pod-ip'`. */
2178
+ readonly podIP?: string
2179
+ } = {},
759
2180
  ): KubernetesAgentAddress {
2181
+ if ((options.mode ?? 'service') === 'pod-ip') {
2182
+ const podIP = options.podIP
2183
+ if (podIP === undefined || podIP === '') {
2184
+ throw new Error(
2185
+ `kubernetes: sandbox ${binding.name} is configured with agentAddress: 'pod-ip', but the live pod its bind token was read from reported no status.podIP (nor a status.podIPs entry), so there is no address to dial. Every path that binds a pod POLLS for that address until the readiness deadline before this is reached — see readAddressedPod — so a pod that still reports none was never given one: a CNI that did not attach it, or a pod that never got past scheduling. Nothing is taken from the Sandbox's own status here on purpose — that IP may belong to a different pod than the token does.`,
2186
+ )
2187
+ }
2188
+ return { kind: 'tcp', host: podIP, port: agentPort, token }
2189
+ }
760
2190
  const host = binding.serviceFQDN ?? binding.podIPs?.[0]
761
2191
  if (host === undefined || host === '') {
762
2192
  throw new Error(
@@ -766,10 +2196,22 @@ export function resolveAgentAddress(
766
2196
  return { kind: 'tcp', host, port: agentPort, token }
767
2197
  }
768
2198
 
2199
+ /**
2200
+ * Ready-or-not-yet, read off a `SandboxClaim`'s own status — plus the one
2201
+ * "not yet" that is really a "no".
2202
+ *
2203
+ * The rejection check runs BEFORE the readiness check rather than after,
2204
+ * because a refused claim is `Ready=False` forever and the readiness check
2205
+ * cannot tell that from a cold start in progress. See
2206
+ * {@link classifyClaimRejection}.
2207
+ */
769
2208
  function bindingFromClaim(
770
2209
  claim: SandboxClaimResource | undefined,
771
2210
  claimName: string,
2211
+ metadata?: ClaimPodMetadataContext,
772
2212
  ): KubernetesSandboxBinding | undefined {
2213
+ const rejection = classifyClaimRejection(claim, claimName, metadata)
2214
+ if (rejection !== undefined) throw rejection
773
2215
  if (!isConditionTrue(claim?.status?.conditions, READY_CONDITION)) return undefined
774
2216
  const bound = claim?.status?.sandbox
775
2217
  // Ready with no bound name is the controller contradicting itself; polling
@@ -804,38 +2246,141 @@ export function bindingFromSandbox(
804
2246
  }
805
2247
  }
806
2248
 
2249
+ /**
2250
+ * The readiness poll ran out of budget — and nothing else. Every OTHER
2251
+ * failure {@link pollForBinding} meets is rethrown as itself, so this class
2252
+ * is an exact answer to "was it the clock?", which a caller that has to
2253
+ * choose between two timeout messages needs and cannot get from the clock.
2254
+ *
2255
+ * Reading `remainingMs()` after the fact is NOT that answer: the expiry timer
2256
+ * and `performance.now()` are different clocks, and a timer that fires a
2257
+ * fraction of a millisecond early leaves a positive remainder behind an
2258
+ * expiry that has already happened.
2259
+ */
2260
+ export class ReadinessPollTimeout extends Error {
2261
+ override readonly name = 'ReadinessPollTimeout'
2262
+ }
2263
+
2264
+ /**
2265
+ * What a failed readiness read is worth, decided by the caller.
2266
+ *
2267
+ * Optional, and absent means exactly the behaviour every caller had before:
2268
+ * the first failure of any kind ends the poll. `workspace.ts` passes nothing
2269
+ * and is unchanged; the acquire path passes
2270
+ * {@link retryDelayForApiFailure} so one 429 on a shared cluster no longer
2271
+ * fails a create that had fifty-nine seconds of budget left.
2272
+ */
2273
+ export interface ReadinessPollBehaviour {
2274
+ /**
2275
+ * Milliseconds to wait before reading again, or `undefined` to rethrow.
2276
+ *
2277
+ * The wait is spent on the SAME deadline as everything else in the poll,
2278
+ * so a retry consumes the readiness budget and can never extend it. A hook
2279
+ * that always returns a number therefore still terminates: the clock ends
2280
+ * the loop, not the hook.
2281
+ */
2282
+ readonly retryDelayFor?: (err: unknown) => number | undefined
2283
+ }
2284
+
807
2285
  /**
808
2286
  * Poll until `read` reports a binding. `read` returns `undefined` for "not
809
2287
  * yet" and throws for a failure worth surfacing; the deadline owns every wait,
810
2288
  * including the sleep between attempts, so an expired clock cannot be extended
811
2289
  * by one more round trip. Shaped after ACI's `pollForRunningIp`.
2290
+ *
2291
+ * The only failure this raises on its own account is
2292
+ * {@link ReadinessPollTimeout}; anything `read` throws travels out unchanged,
2293
+ * unless `behaviour.retryDelayFor` claims it — in which case it is repeated
2294
+ * inside the same budget and, if the budget then runs out, carried onto the
2295
+ * timeout as its `cause`, so a poll that kept failing still says what it kept
2296
+ * seeing.
812
2297
  */
813
2298
  export async function pollForBinding(
814
2299
  read: (signal: AbortSignal) => Promise<KubernetesSandboxBinding | undefined>,
815
2300
  deadline: OperationDeadline,
816
2301
  readiness: { readonly timeoutMs: number; readonly pollIntervalMs: number },
817
2302
  label: string,
2303
+ behaviour: ReadinessPollBehaviour = {},
818
2304
  ): Promise<KubernetesSandboxBinding> {
2305
+ /**
2306
+ * The last failure that was retried rather than raised, and only while it
2307
+ * is still the truth: a read that succeeds clears it, so the timeout
2308
+ * carries a failure the poll was STILL meeting when the budget ran out
2309
+ * rather than a blip it recovered from twenty polls earlier. The
2310
+ * difference is not cosmetic — the acquire's reason is read off this
2311
+ * cause, so a stale one would report an API outage for a poll whose API
2312
+ * was answering fine.
2313
+ */
2314
+ let lastRetried: unknown
819
2315
  while (deadline.remainingMs() > 0) {
2316
+ /** Overrides the poll cadence for one round when the server named one. */
2317
+ let nextDelayMs = readiness.pollIntervalMs
820
2318
  try {
821
2319
  const binding = await deadline.run(read)
2320
+ lastRetried = undefined
822
2321
  if (binding) return binding
823
2322
  } catch (err) {
824
2323
  if (err instanceof OperationDeadlineExpired) break
825
- throw err
2324
+ const retryDelayMs = behaviour.retryDelayFor?.(err)
2325
+ if (retryDelayMs === undefined) throw err
2326
+ lastRetried = err
2327
+ nextDelayMs = retryDelayMs
826
2328
  }
827
2329
  try {
828
- await deadline.delay(readiness.pollIntervalMs)
2330
+ await deadline.delay(nextDelayMs)
829
2331
  } catch (err) {
830
2332
  if (err instanceof OperationDeadlineExpired) break
831
2333
  throw err
832
2334
  }
833
2335
  }
834
- throw new Error(`kubernetes: ${label} never became Ready (${readiness.timeoutMs}ms)`)
2336
+ throw new ReadinessPollTimeout(
2337
+ `kubernetes: ${label} never became Ready (${readiness.timeoutMs}ms)${
2338
+ lastRetried !== undefined
2339
+ ? `; the last API failure retried inside that budget was: ${
2340
+ lastRetried instanceof Error ? lastRetried.message : String(lastRetried)
2341
+ }`
2342
+ : ''
2343
+ }`,
2344
+ lastRetried !== undefined ? { cause: lastRetried } : undefined,
2345
+ )
835
2346
  }
836
2347
 
837
2348
  /**
838
- * The per-instance bind token: the backing pod's `metadata.uid`.
2349
+ * The live pod behind a bound sandbox: its uid, and — read in the SAME
2350
+ * answer — the IP it can be dialed at.
2351
+ *
2352
+ * One record rather than two reads because the two facts have to describe one
2353
+ * pod. The uid is the agent's bind token and the IP is where that agent
2354
+ * listens; taking them from separate GETs leaves a window in which a resume,
2355
+ * an eviction or a node drain replaces the pod in between, and the handle
2356
+ * then presents pod A's token at pod B's address. The guest answers that with
2357
+ * a flat `unauthorized`, which says nothing about the race that caused it.
2358
+ */
2359
+ export interface KubernetesBoundPod {
2360
+ /** `metadata.uid` — the agent's bind token. */
2361
+ readonly uid: string
2362
+ /** `status.podIP`. Read by `agentAddress: 'pod-ip'`; absent is legal. */
2363
+ readonly podIP?: string
2364
+ /**
2365
+ * `metadata.labels` — what an ingress policy's `podSelector` actually
2366
+ * matches. Read off the SAME object the uid and the address come from,
2367
+ * for the same reason they are: a policy decision made about one pod and
2368
+ * a connection made to another is the mismatch this record exists to
2369
+ * prevent. The claim path reads it to ask which policies select the bound
2370
+ * pod — a directly created Sandbox's labels are known before its pod
2371
+ * exists — and BOTH paths read it to confirm the pod really carries the
2372
+ * labels this backend asked the controller for, which is the one thing
2373
+ * knowing them in advance cannot establish. See `ingress-policy.ts` and
2374
+ * `assertRequestedPodLabelsObserved`.
2375
+ */
2376
+ readonly labels?: Readonly<Record<string, string>>
2377
+ }
2378
+
2379
+ /**
2380
+ * Find the pod a sandbox is currently backed by, and read both facts off it.
2381
+ *
2382
+ * `uid` is the per-instance agent bind token; `podIP` is where that agent
2383
+ * listens, and only `agentAddress: 'pod-ip'` reads it.
839
2384
  *
840
2385
  * The pod is named after its Sandbox in agent-sandbox v1.0.2 — verified
841
2386
  * against a running cluster — but that is an observation, not a documented
@@ -844,12 +2389,12 @@ export async function pollForBinding(
844
2389
  * changing upstream is a second round trip through `status.selector`, which is
845
2390
  * exactly what the controller publishes the selector for.
846
2391
  */
847
- export async function readPodBindToken(
2392
+ export async function readBoundPod(
848
2393
  client: KubernetesClient,
849
2394
  namespace: string,
850
2395
  binding: KubernetesSandboxBinding,
851
2396
  signal?: AbortSignal,
852
- ): Promise<string> {
2397
+ ): Promise<KubernetesBoundPod> {
853
2398
  try {
854
2399
  const pod = await client.request<PodResource>(
855
2400
  'GET',
@@ -863,7 +2408,7 @@ export async function readPodBindToken(
863
2408
  // terminating this GET can answer with the pod that is leaving and a
864
2409
  // uid the new agent will refuse. On the acquire path nothing is
865
2410
  // terminating and the filter never fires.
866
- if (uid && isPodLive(pod)) return uid
2411
+ if (uid && isPodLive(pod)) return boundPod(uid, pod)
867
2412
  } catch (err) {
868
2413
  if (!(err instanceof KubernetesAlreadyGoneError)) throw err
869
2414
  }
@@ -879,7 +2424,7 @@ export async function readPodBindToken(
879
2424
  )
880
2425
  for (const pod of list?.items ?? []) {
881
2426
  const uid = pod.metadata?.uid
882
- if (uid && isPodLive(pod)) return uid
2427
+ if (uid && isPodLive(pod)) return boundPod(uid, pod)
883
2428
  }
884
2429
  }
885
2430
  throw new Error(
@@ -887,6 +2432,105 @@ export async function readPodBindToken(
887
2432
  )
888
2433
  }
889
2434
 
2435
+ function boundPod(uid: string, pod: PodResource): KubernetesBoundPod {
2436
+ const podIP = readPodIP(pod)
2437
+ const labels = pod.metadata?.labels
2438
+ return {
2439
+ uid,
2440
+ ...(podIP !== undefined ? { podIP } : {}),
2441
+ ...(labels !== undefined ? { labels } : {}),
2442
+ }
2443
+ }
2444
+
2445
+ /**
2446
+ * {@link readBoundPod}, plus — under `'pod-ip'` only — the wait for an
2447
+ * address to go with the token.
2448
+ *
2449
+ * Under the default mode this is the single read it has always been: one
2450
+ * `GET`, in the same place in the same order, because a Service FQDN is
2451
+ * published with the Sandbox and needs nothing from the pod but its uid.
2452
+ *
2453
+ * `'pod-ip'` has to wait, because a LIVE pod is not yet an ADDRESSED pod. A
2454
+ * pod is created `Pending` and carries no `status.podIP` until the CNI has
2455
+ * finished attaching it, and {@link isPodLive} accepts `Pending` on purpose —
2456
+ * the resume path in `workspace.ts` binds its replacement pod long before that
2457
+ * pod is Ready, because `Ready` stays True across the transition and the uid
2458
+ * is the only transition signal there is. Refusing an address-less pod outright
2459
+ * would therefore fail on the NORMAL path, in milliseconds, with the whole
2460
+ * readiness budget unspent. So "live, no address yet" is polled on the same
2461
+ * deadline as everything else on this path, and
2462
+ * {@link resolveAgentAddress}'s own refusal is left as the post-deadline
2463
+ * backstop for a pod that never gets an address at all.
2464
+ *
2465
+ * A failed READ stays fatal, exactly as it was: this is the acquire path,
2466
+ * where nothing is being replaced and a pod that cannot be read is not a pod
2467
+ * that is about to appear.
2468
+ *
2469
+ * `requiredLabels` is the second thing worth waiting for, and it is waited
2470
+ * for in the SAME loop rather than in a second one: an egress profile is a
2471
+ * label the CONTROLLER patches onto the pod it binds, so a pod read the
2472
+ * instant it was bound can be live, addressed and not yet labelled. Two
2473
+ * loops would be two deadlines and two answers to "is this pod ready to be
2474
+ * admitted". Like the address, an expired clock hands the pod back as it is
2475
+ * — the caller decides whether a missing label is fatal, and on the acquire
2476
+ * path it is: see `assertRequestedPodLabelsObserved`.
2477
+ */
2478
+ export async function readAddressedPod(
2479
+ client: KubernetesClient,
2480
+ namespace: string,
2481
+ binding: KubernetesSandboxBinding,
2482
+ deadline: OperationDeadline,
2483
+ readiness: { readonly pollIntervalMs: number },
2484
+ mode: KubernetesAgentAddressMode,
2485
+ requiredLabels?: Readonly<Record<string, string>>,
2486
+ ): Promise<KubernetesBoundPod> {
2487
+ const required = Object.entries(requiredLabels ?? {})
2488
+ const wanting = (pod: KubernetesBoundPod): boolean =>
2489
+ (mode === 'pod-ip' && pod.podIP === undefined) ||
2490
+ required.some(([key, value]) => pod.labels?.[key] !== value)
2491
+
2492
+ let pod = await deadline.run((signal) => readBoundPod(client, namespace, binding, signal))
2493
+ while (wanting(pod) && deadline.remainingMs() > 0) {
2494
+ try {
2495
+ await deadline.delay(readiness.pollIntervalMs)
2496
+ pod = await deadline.run((signal) => readBoundPod(client, namespace, binding, signal))
2497
+ } catch (err) {
2498
+ // An expired clock hands the incomplete pod back rather than
2499
+ // replacing it with a bare "deadline expired": the caller's
2500
+ // `resolveAgentAddress` (or `assertRequestedPodLabelsObserved`) then
2501
+ // reports WHICH fact never arrived.
2502
+ if (err instanceof OperationDeadlineExpired) break
2503
+ throw err
2504
+ }
2505
+ }
2506
+ return pod
2507
+ }
2508
+
2509
+ /**
2510
+ * The re-read a `'pod-ip'` handle follows a replaced pod with: one live-pod
2511
+ * read, then the same address resolution acquire did.
2512
+ *
2513
+ * Built here rather than inside the transport because finding the pod is a
2514
+ * CONTROL-plane act — the by-name GET, the selector fallback, the
2515
+ * liveness filter — and the transport owns none of that. It is handed over as
2516
+ * a closure so the transport can call it without learning what a Sandbox is.
2517
+ */
2518
+ export function buildAgentAddressRefresh(
2519
+ client: KubernetesClient,
2520
+ namespace: string,
2521
+ binding: KubernetesSandboxBinding,
2522
+ agentPort: number,
2523
+ mode: KubernetesAgentAddressMode,
2524
+ ): (signal?: AbortSignal) => Promise<KubernetesAgentAddress> {
2525
+ return async (signal) => {
2526
+ const pod = await readBoundPod(client, namespace, binding, signal)
2527
+ return resolveAgentAddress(binding, agentPort, pod.uid, {
2528
+ mode,
2529
+ ...(pod.podIP !== undefined ? { podIP: pod.podIP } : {}),
2530
+ })
2531
+ }
2532
+ }
2533
+
890
2534
  async function readSandboxSelector(
891
2535
  client: KubernetesClient,
892
2536
  namespace: string,
@@ -939,17 +2583,31 @@ async function admitProbedSandbox(
939
2583
  config: KubernetesBackendInternalConfig,
940
2584
  options: SandboxBackendOptions,
941
2585
  probeTimeoutMs: number,
2586
+ /**
2587
+ * Present exactly when `config.egress.perSandbox` is configured — which
2588
+ * is what makes `setNetworkPolicy` present on the handle. See
2589
+ * `per-sandbox-policy.ts`.
2590
+ */
2591
+ setNetworkPolicy?: (policy: SandboxNetworkPolicy) => Promise<void>,
942
2592
  ): Promise<Sandbox> {
943
2593
  const sandbox = buildKubernetesSandbox({
944
2594
  name: acquisition.binding.name,
945
2595
  rootDir: options.workingDirectory,
946
- transport: new KubernetesAgentTransport(acquisition.agent),
2596
+ transport: new KubernetesAgentTransport(acquisition.agent, {
2597
+ // The backend opts in to the stream heartbeat; the transport
2598
+ // option it sets stays undefined for every other tier.
2599
+ heartbeatMs: resolveStreamHeartbeatMs(config.streamHeartbeatMs),
2600
+ ...(acquisition.refreshAgent !== undefined
2601
+ ? { refreshHandle: acquisition.refreshAgent }
2602
+ : {}),
2603
+ }),
947
2604
  release: acquisition.release,
948
2605
  renew: acquisition.renew,
949
2606
  ttlSeconds: acquisition.ttlSeconds,
950
2607
  ...(config.onLeaseRenewalError !== undefined
951
2608
  ? { onRenewalError: config.onLeaseRenewalError }
952
2609
  : {}),
2610
+ ...(setNetworkPolicy !== undefined ? { setNetworkPolicy } : {}),
953
2611
  })
954
2612
  try {
955
2613
  await probeSandboxPrivileges(sandbox, acquisition.binding.name, probeTimeoutMs, options.signal)