@namzu/sandbox 14.0.0 → 16.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. package/CHANGELOG.md +924 -0
  2. package/README.md +369 -14
  3. package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
  4. package/dist/backends/aci-standby-pool/index.js +13 -1
  5. package/dist/backends/aci-standby-pool/index.js.map +1 -1
  6. package/dist/backends/docker/index.d.ts +169 -6
  7. package/dist/backends/docker/index.d.ts.map +1 -1
  8. package/dist/backends/docker/index.js +499 -85
  9. package/dist/backends/docker/index.js.map +1 -1
  10. package/dist/backends/firecracker/index.d.ts.map +1 -1
  11. package/dist/backends/firecracker/index.js +12 -2
  12. package/dist/backends/firecracker/index.js.map +1 -1
  13. package/dist/backends/firecracker/protocol.d.ts +459 -8
  14. package/dist/backends/firecracker/protocol.d.ts.map +1 -1
  15. package/dist/backends/firecracker/protocol.js +136 -0
  16. package/dist/backends/firecracker/protocol.js.map +1 -1
  17. package/dist/backends/firecracker/transport.d.ts +539 -6
  18. package/dist/backends/firecracker/transport.d.ts.map +1 -1
  19. package/dist/backends/firecracker/transport.js +1171 -24
  20. package/dist/backends/firecracker/transport.js.map +1 -1
  21. package/dist/backends/kubernetes/egress-policy.d.ts +1181 -13
  22. package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -1
  23. package/dist/backends/kubernetes/egress-policy.js +2350 -31
  24. package/dist/backends/kubernetes/egress-policy.js.map +1 -1
  25. package/dist/backends/kubernetes/identity.d.ts +193 -0
  26. package/dist/backends/kubernetes/identity.d.ts.map +1 -0
  27. package/dist/backends/kubernetes/identity.js +147 -0
  28. package/dist/backends/kubernetes/identity.js.map +1 -0
  29. package/dist/backends/kubernetes/index.d.ts +678 -33
  30. package/dist/backends/kubernetes/index.d.ts.map +1 -1
  31. package/dist/backends/kubernetes/index.js +1180 -95
  32. package/dist/backends/kubernetes/index.js.map +1 -1
  33. package/dist/backends/kubernetes/ingress-policy.d.ts +375 -0
  34. package/dist/backends/kubernetes/ingress-policy.d.ts.map +1 -0
  35. package/dist/backends/kubernetes/ingress-policy.js +1050 -0
  36. package/dist/backends/kubernetes/ingress-policy.js.map +1 -0
  37. package/dist/backends/kubernetes/k8s-client.d.ts +213 -4
  38. package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -1
  39. package/dist/backends/kubernetes/k8s-client.js +359 -52
  40. package/dist/backends/kubernetes/k8s-client.js.map +1 -1
  41. package/dist/backends/kubernetes/lease.d.ts +40 -14
  42. package/dist/backends/kubernetes/lease.d.ts.map +1 -1
  43. package/dist/backends/kubernetes/lease.js +68 -18
  44. package/dist/backends/kubernetes/lease.js.map +1 -1
  45. package/dist/backends/kubernetes/objects.d.ts +423 -3
  46. package/dist/backends/kubernetes/objects.d.ts.map +1 -1
  47. package/dist/backends/kubernetes/objects.js +364 -2
  48. package/dist/backends/kubernetes/objects.js.map +1 -1
  49. package/dist/backends/kubernetes/per-sandbox-policy.d.ts +219 -0
  50. package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -0
  51. package/dist/backends/kubernetes/per-sandbox-policy.js +375 -0
  52. package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -0
  53. package/dist/backends/kubernetes/rbac.d.ts +153 -0
  54. package/dist/backends/kubernetes/rbac.d.ts.map +1 -0
  55. package/dist/backends/kubernetes/rbac.js +177 -0
  56. package/dist/backends/kubernetes/rbac.js.map +1 -0
  57. package/dist/backends/kubernetes/sandbox.d.ts +81 -14
  58. package/dist/backends/kubernetes/sandbox.d.ts.map +1 -1
  59. package/dist/backends/kubernetes/sandbox.js +149 -15
  60. package/dist/backends/kubernetes/sandbox.js.map +1 -1
  61. package/dist/backends/kubernetes/transport.d.ts +935 -9
  62. package/dist/backends/kubernetes/transport.d.ts.map +1 -1
  63. package/dist/backends/kubernetes/transport.js +1958 -62
  64. package/dist/backends/kubernetes/transport.js.map +1 -1
  65. package/dist/backends/kubernetes/workspace.d.ts +1149 -18
  66. package/dist/backends/kubernetes/workspace.d.ts.map +1 -1
  67. package/dist/backends/kubernetes/workspace.js +2825 -186
  68. package/dist/backends/kubernetes/workspace.js.map +1 -1
  69. package/dist/backends/remote-execution-controller.d.ts +14 -0
  70. package/dist/backends/remote-execution-controller.d.ts.map +1 -1
  71. package/dist/backends/remote-execution-controller.js.map +1 -1
  72. package/dist/index.d.ts +294 -18
  73. package/dist/index.d.ts.map +1 -1
  74. package/dist/index.js +280 -10
  75. package/dist/index.js.map +1 -1
  76. package/dist/testing/sandbox-conformance.d.ts +39 -5
  77. package/dist/testing/sandbox-conformance.d.ts.map +1 -1
  78. package/dist/testing/sandbox-conformance.js +436 -5
  79. package/dist/testing/sandbox-conformance.js.map +1 -1
  80. package/package.json +3 -3
  81. package/src/backends/aci-standby-pool/index.ts +16 -1
  82. package/src/backends/docker/index.ts +617 -100
  83. package/src/backends/firecracker/index.ts +14 -2
  84. package/src/backends/firecracker/protocol.ts +514 -6
  85. package/src/backends/firecracker/transport.ts +1492 -40
  86. package/src/backends/kubernetes/egress-policy.ts +3334 -55
  87. package/src/backends/kubernetes/identity.ts +261 -0
  88. package/src/backends/kubernetes/index.ts +1785 -127
  89. package/src/backends/kubernetes/ingress-policy.ts +1344 -0
  90. package/src/backends/kubernetes/k8s-client.ts +444 -54
  91. package/src/backends/kubernetes/lease.ts +75 -19
  92. package/src/backends/kubernetes/objects.ts +626 -6
  93. package/src/backends/kubernetes/per-sandbox-policy.ts +497 -0
  94. package/src/backends/kubernetes/rbac.ts +192 -0
  95. package/src/backends/kubernetes/sandbox.ts +218 -20
  96. package/src/backends/kubernetes/transport.ts +2733 -124
  97. package/src/backends/kubernetes/workspace.ts +4476 -222
  98. package/src/backends/remote-execution-controller.ts +14 -0
  99. package/src/index.ts +668 -19
  100. package/src/testing/sandbox-conformance.ts +540 -5
@@ -77,22 +77,79 @@
77
77
  * ## Egress
78
78
  *
79
79
  * `config.egress` is optional and, when set, translated and VERIFIED — never
80
- * created — by `egress-policy.ts`. Verification happens once, lazily, on the
81
- * first `create()`, so `buildKubernetesBackend` itself still contacts
82
- * nothing. Every Sandbox this file creates directly (`buildSandboxBody`)
83
- * carries {@link sandboxTemplateLabel} on its podTemplate specifically so
84
- * that translated policy's `podSelector` has something stable to match —
85
- * see `objects.ts`'s doc comment on that label for why agent-sandbox's own
86
- * controller-owned label does not cover this path.
80
+ * created — by `egress-policy.ts`, in two steps. The NAMED object is GETted
81
+ * and compared to the translation exactly, once, lazily, on the first
82
+ * `create()`, so `buildKubernetesBackend` itself still contacts nothing.
83
+ * Then, because the API server UNIONS every policy selecting a pod, every
84
+ * `NetworkPolicy` in the namespace (and, under `engine: 'cilium'`, every
85
+ * `CiliumNetworkPolicy`) is enumerated against the pod's real labels and the
86
+ * create is refused when any of them lets out more than the translation does
87
+ * — a second policy widens egress however exactly the named one matches, and
88
+ * a `SandboxTemplate`'s own `networkPolicy` block becomes exactly such a
89
+ * policy. `egress.verify: 'named-object-only'` is the opt-out and restores
90
+ * the first step alone. Every Sandbox this file creates directly
91
+ * (`buildSandboxBody`) carries {@link sandboxTemplateLabel} on its
92
+ * podTemplate specifically so that translated policy's `podSelector` has
93
+ * something stable to match — see `objects.ts`'s doc comment on that label
94
+ * for why agent-sandbox's own controller-owned label does not cover this
95
+ * path.
96
+ *
97
+ * ## Ingress
98
+ *
99
+ * `config.ingress` is the opposite default: verification is ON unless a
100
+ * deployment says `'unverified'`. Before the POST for a direct Sandbox, and
101
+ * after the bind for a claimed one, `ingress-policy.ts` lists the namespace's
102
+ * policies and refuses unless one of them actually closes the agent port on
103
+ * the labels this pod carries. Nothing checked that before, while three
104
+ * pieces of shipped text said it was covered — see that module's own doc
105
+ * comment for what was measured.
87
106
  */
88
107
  import { generateSandboxId } from '@namzu/sdk';
89
108
  import { OperationDeadline, OperationDeadlineExpired, resolveReadinessOptions, runFailureCleanup, } from '../readiness.js';
90
- import { assertEgressPolicyIsEnforceable, defaultEgressPolicyName, translateEgressPolicy, verifyEgressPolicyApplied, } from './egress-policy.js';
91
- import { KubernetesAlreadyGoneError, createKubernetesClient, } from './k8s-client.js';
92
- import { READY_CONDITION, SANDBOX_API_GROUP, SANDBOX_API_VERSION, SANDBOX_EXTENSIONS_API_GROUP, claimCollectionPath, claimPath, isConditionTrue, isPodLive, podListPath, podPath, sandboxCollectionPath, sandboxPath, sandboxTemplateLabel, sandboxTemplatePath, } from './objects.js';
109
+ import { KubernetesPodLabelNotObservedError, KubernetesPodLabelsRejectedError, assertEgressPolicyIsEnforceable, assertEgressProfileIsUsable, assertPerSandboxEgressIsUsable, composeAdditionalPodLabels, defaultEgressPolicyName, egressProfileLabel, egressUnionVerificationEnabled, perSandboxEgressLabelKey, translateEgressPolicy, verifyEgressPolicyApplied, verifyEgressPolicyUnion, } from './egress-policy.js';
110
+ import { ingressVerificationEnabled, resolveIngressEngine, verifyIngressPolicyApplied, } from './ingress-policy.js';
111
+ import { KubernetesAlreadyGoneError, KubernetesApiError, KubernetesApiTimeoutError, KubernetesCredentialError, createKubernetesClient, } from './k8s-client.js';
112
+ import { READY_CONDITION, SANDBOX_API_GROUP, SANDBOX_API_VERSION, SANDBOX_EXTENSIONS_API_GROUP, claimCollectionPath, claimListPath, claimPath, isConditionTrue, isPodLive, podCollectionPath, podListPath, podPath, readPodIP, sandboxCollectionPath, sandboxPath, sandboxTemplateLabel, sandboxTemplatePath, warmPoolPath, } from './objects.js';
113
+ import { KubernetesOwnerUidMissingError, buildAdmissionFence, buildPerSandboxPolicySetter, } from './per-sandbox-policy.js';
93
114
  import { privilegeProbeTimedOut, runPrivilegeProbe } from './privilege-probe.js';
94
115
  import { buildKubernetesSandbox } from './sandbox.js';
95
116
  import { KubernetesAgentTransport } from './transport.js';
117
+ /**
118
+ * Default {@link KubernetesBackendInternalConfig.streamHeartbeatMs} — 15 s,
119
+ * so a stream whose peer vanished without a FIN or an RST is given up on
120
+ * within 45 s rather than never.
121
+ *
122
+ * The Kubernetes backend opts IN here; `VsockTransportOptions.heartbeatMs`
123
+ * stays undefined by default, because that transport is shared with the
124
+ * Firecracker tier and a default there would force-close an existing
125
+ * consumer's quiet-but-alive terminal.
126
+ */
127
+ export const DEFAULT_STREAM_HEARTBEAT_MS = 15_000;
128
+ /**
129
+ * Validate the configured stream-heartbeat interval, or supply the default.
130
+ *
131
+ * `0` IS accepted here, unlike `apiRequestTimeoutMs`: the heartbeat is a new
132
+ * capability that a deployment behind a middlebox with its own idea about
133
+ * unexpected frames may want off, and turning it off restores exactly the
134
+ * behaviour every release before this one had. An unanswered API request has
135
+ * no such prior behaviour worth restoring.
136
+ */
137
+ export function resolveStreamHeartbeatMs(value) {
138
+ if (value === undefined)
139
+ return DEFAULT_STREAM_HEARTBEAT_MS;
140
+ if (!Number.isSafeInteger(value) || value < 0) {
141
+ throw new Error(`kubernetes: streamHeartbeatMs must be a non-negative integer, got ${JSON.stringify(value)}. Use 0 to send no heartbeats at all, which is how every release before this one behaved; the default is ${DEFAULT_STREAM_HEARTBEAT_MS}ms.`);
142
+ }
143
+ return value;
144
+ }
145
+ /**
146
+ * The client options every `createKubernetesClient` call in this backend is
147
+ * built with. One function so the five call sites — the provider, and each
148
+ * of the workspace verbs — cannot drift apart on which bounds they honour.
149
+ */
150
+ export function clientOptions(config) {
151
+ return { requestTimeoutMs: config.apiRequestTimeoutMs };
152
+ }
96
153
  /**
97
154
  * The same number the Firecracker guest agent listens on over vsock
98
155
  * (`DEFAULT_AGENT_VSOCK_PORT`), so one agent has one port across both tiers
@@ -209,51 +266,220 @@ export function buildKubernetesBackend(config) {
209
266
  // `config.egress.policy.kind` alone, with no API call, so it is refused
210
267
  // here, synchronously, the same moment the two checks above are.
211
268
  if (config.egress) {
212
- assertEgressPolicyIsEnforceable(config.egress.policy, config.egress.engine ?? 'core');
269
+ assertEgressPolicyIsEnforceable(config.egress.policy, config.egress.engine ?? 'core', config.egress.ciliumNarrowing);
213
270
  }
214
- const client = createKubernetesClient(clientAccess(config));
271
+ // And the same for an egress PROFILE that could never be written as a
272
+ // label — a value the API server would reject leaves either a claim
273
+ // nothing binds or a policy nobody can apply, and both are decidable
274
+ // from config alone.
275
+ assertEgressProfileIsUsable(config.egress);
276
+ // And the same for per-sandbox egress: an engine that cannot express a
277
+ // hostname, an unnamed admission fence or a selector key the API server
278
+ // would refuse are all decidable from config alone, and a host that
279
+ // learns any of them from its first `setNetworkPolicy` call learns it an
280
+ // hour into a run that cannot be redone.
281
+ assertPerSandboxEgressIsUsable(config.egress);
282
+ // Resolved here as well as at each session, so a configuration this
283
+ // backend will never honour is refused while `buildKubernetesBackend` is
284
+ // still on the stack rather than on someone's first `create()`.
285
+ resolveStreamHeartbeatMs(config.streamHeartbeatMs);
286
+ const client = createKubernetesClient(clientAccess(config), clientOptions(config));
215
287
  // Verify-not-trust runs once, lazily, on the first `create()` — never here,
216
288
  // because `buildKubernetesBackend` is documented to contact nothing. A
217
289
  // failed attempt is not cached: a transient API error should not wedge
218
- // every later create() behind the same stale rejection forever.
219
- let egressVerification;
290
+ // every later create() behind the same stale rejection forever. The
291
+ // boundary object holds both halves and both memos; it lives as long as
292
+ // this backend does, which is what makes the named-object check
293
+ // once-per-backend rather than once-per-create.
294
+ const egressBoundary = buildEgressBoundary(client, config, config.sandboxTemplateName);
295
+ // Ingress is cached PER LABEL SET rather than once per backend, because
296
+ // unlike the egress NAMED-object check it is a question about one pod: a
297
+ // pooled sandbox's labels come off the pool's template and a pool-less
298
+ // one's off this config, and a single memo would answer for a pod it never
299
+ // examined. Same failure handling as the egress memo — a failed attempt is
300
+ // dropped, so a transient API error does not wedge every later create()
301
+ // behind it. The egress UNION check is keyed the same way, for the same
302
+ // reason, and additionally expires: see {@link EGRESS_UNION_CACHE_TTL_MS}.
303
+ const ingressVerifier = buildIngressVerifier(client, config, new Map());
304
+ // One fence per backend, so its memo is shared by every sandbox this
305
+ // backend hands out rather than re-proved per handle. `undefined` when
306
+ // per-sandbox egress is not configured, which is what makes
307
+ // `setNetworkPolicy` absent from the handle — presence follows
308
+ // CONFIGURATION and never a runtime probe, so a caller's capability
309
+ // detection cannot depend on when it asked.
310
+ const perSandbox = config.egress?.perSandbox;
311
+ const fence = perSandbox === undefined ? undefined : buildAdmissionFence(client, perSandbox);
220
312
  return {
221
313
  tier: 'microvm',
222
314
  name: 'kubernetes',
223
315
  async create(options) {
224
- if (config.egress) {
225
- egressVerification ??= verifyEgressPolicyConfigured(client, config.namespace, config.sandboxTemplateName, config.egress, options.signal).catch((err) => {
226
- egressVerification = undefined;
227
- throw err;
228
- });
229
- await egressVerification;
230
- }
231
- const acquisition = await acquireKubernetesSandbox(client, config, options, readiness);
232
- return await admitProbedSandbox(acquisition, config, options, resolveProbeTimeoutMs(readiness.timeoutMs));
316
+ await egressBoundary?.verifyNamedObject(options.signal);
317
+ const acquisition = await acquireKubernetesSandbox(client, config, options, readiness, ingressVerifier, egressBoundary);
318
+ const egress = config.egress;
319
+ const setNetworkPolicy = fence !== undefined &&
320
+ egress?.perSandbox !== undefined &&
321
+ acquisition.owner !== undefined &&
322
+ acquisition.perSandboxLabelValue !== undefined
323
+ ? buildPerSandboxPolicySetter({
324
+ client,
325
+ fence,
326
+ namespace: config.namespace,
327
+ egress: { ...egress, perSandbox: egress.perSandbox },
328
+ owner: acquisition.owner,
329
+ selectorValue: acquisition.perSandboxLabelValue,
330
+ })
331
+ : undefined;
332
+ return await admitProbedSandbox(acquisition, config, options, resolveProbeTimeoutMs(readiness.timeoutMs), setNetworkPolicy);
233
333
  },
234
334
  };
235
335
  }
236
336
  /**
237
- * Translate `egress.policy` and confirm an operator applied a matching
238
- * object — the whole verify-not-trust step, isolated so `create()` above
239
- * stays about ONE thing (memoize-once-per-backend) rather than two.
337
+ * How long a union pass is trusted for one label set.
240
338
  *
241
- * Exported because `workspace.ts` runs the identical step: a workspace does
242
- * not go through `buildKubernetesBackend`, and a config `egress` honoured on
243
- * one entry point and ignored on the other would be a silent downgrade of the
244
- * boundary this backend calls primary. `sandboxTemplateName` is the template
245
- * the caller is actually building from — it decides both the default policy
246
- * name and the pod label the policy's selector has to match, and a workspace
247
- * may be built from a different template than the task path's.
339
+ * Five minutes rather than the backend's lifetime, which is what the
340
+ * named-object check alone used to get: an operator who applies a widening
341
+ * policy at 10:00 should not have it go unnoticed until the host restarts.
342
+ * It is a cache, not a watch — `k8s-client.ts`'s "no watch, no informers, no
343
+ * resourceVersion tracking" invariant is untouched, because the only thing
344
+ * kept across calls is "this exact label set passed at this time".
248
345
  */
249
- export async function verifyEgressPolicyConfigured(client, namespace, sandboxTemplateName, egress, signal) {
346
+ export const EGRESS_UNION_CACHE_TTL_MS = 5 * 60 * 1_000;
347
+ /**
348
+ * Build the egress boundary this config asks for, or nothing at all.
349
+ *
350
+ * `sandboxTemplateName` is the template the caller is actually building from
351
+ * — it decides both the default policy name and the pod label the policy's
352
+ * selector has to match, and a workspace may be built from a different
353
+ * template than the task path's.
354
+ *
355
+ * `now` is injected only so the TTL above can be tested without waiting five
356
+ * minutes; nothing else passes it.
357
+ */
358
+ export function buildEgressBoundary(client, config, sandboxTemplateName, now = Date.now) {
359
+ const egress = config.egress;
360
+ if (egress === undefined)
361
+ return undefined;
250
362
  const engine = egress.engine ?? 'core';
251
- const translated = await translateEgressPolicy(egress.policy, engine, {
252
- namespace,
253
- name: egress.networkPolicyName ?? defaultEgressPolicyName(sandboxTemplateName),
363
+ // One resolution of the profile, shared by the policy NAME and the policy
364
+ // SELECTOR: under a profile the default name gains the profile segment
365
+ // (one template under two profiles is two policy objects) and the
366
+ // selector gains the label, and the two must not be able to disagree.
367
+ const profile = egressProfileLabel(egress);
368
+ const target = {
369
+ namespace: config.namespace,
370
+ name: egress.networkPolicyName ?? defaultEgressPolicyName(sandboxTemplateName, profile?.value),
254
371
  sandboxTemplateName,
255
- });
256
- await verifyEgressPolicyApplied(client, translated, signal);
372
+ ...(profile !== undefined ? { profile } : {}),
373
+ };
374
+ // Translated ONCE per boundary, not once per check: a `resolver` policy's
375
+ // `resolve()` is the host's own closure and may cost a network call, and
376
+ // running the two checks against two independently resolved allowlists
377
+ // would compare each against a different translation.
378
+ let translation;
379
+ const translate = () => {
380
+ translation ??= translateEgressPolicy(egress.policy, engine, target, egress.ciliumNarrowing).catch((err) => {
381
+ translation = undefined;
382
+ throw err;
383
+ });
384
+ return translation;
385
+ };
386
+ let namedObject;
387
+ const passes = new Map();
388
+ // The per-sandbox selector label carries a once-ever value, so it is
389
+ // excluded from the memo key — see {@link policyCacheKey}. Resolved once
390
+ // here rather than per check, because the resolver validates as it
391
+ // resolves and a per-check throw would surface from a cache lookup.
392
+ const perSandboxKey = perSandboxEgressLabelKey(egress);
393
+ return {
394
+ async verifyNamedObject(signal) {
395
+ namedObject ??= (async () => {
396
+ await verifyEgressPolicyApplied(client, await translate(), signal);
397
+ })().catch((err) => {
398
+ namedObject = undefined;
399
+ throw err;
400
+ });
401
+ await namedObject;
402
+ },
403
+ async verifyUnion(podLabels, subject, signal) {
404
+ if (!egressUnionVerificationEnabled(egress))
405
+ return;
406
+ const translated = await translate();
407
+ const key = policyCacheKey(podLabels, perSandboxKey);
408
+ const cached = passes.get(key);
409
+ if (cached !== undefined && now() - cached.at < EGRESS_UNION_CACHE_TTL_MS) {
410
+ await cached.pending;
411
+ return;
412
+ }
413
+ const pending = verifyEgressPolicyUnion(client, translated, { namespace: config.namespace, podLabels, engine, subject }, signal).catch((err) => {
414
+ // A failed attempt is never cached — same rule the named-object
415
+ // memo has always had.
416
+ passes.delete(key);
417
+ throw err;
418
+ });
419
+ passes.set(key, { at: now(), pending });
420
+ await pending;
421
+ },
422
+ };
423
+ }
424
+ /**
425
+ * Canonical key for one label set — order-independent, so two spellings of
426
+ * the same pod share a memo.
427
+ *
428
+ * `excludeKey` drops the PER-SANDBOX egress label, whose value is unique per
429
+ * acquire. Both memos exist to amortise a namespace-wide policy enumeration
430
+ * across every sandbox a backend produces, and a key that carried a
431
+ * once-ever value would give every acquire a miss and leave an entry behind
432
+ * that nothing ever looks up again — a full enumeration per sandbox, and a
433
+ * Map that grows for the host's whole life.
434
+ *
435
+ * Dropping it is sound at the moment these checks run: the only policy that
436
+ * could select a pod BY that key and value is that sandbox's own, whose name
437
+ * is generated in the same call and which does not exist yet. Every other
438
+ * policy selecting the pod — the operator's baseline, a per-profile one,
439
+ * anything hand-written — selects on the labels that remain, so two pods
440
+ * differing only in this label are the same question. The label itself is
441
+ * still PRESENT in the label set each check is run against; only the memo's
442
+ * key ignores it.
443
+ */
444
+ function policyCacheKey(podLabels, excludeKey) {
445
+ return JSON.stringify(Object.entries(podLabels)
446
+ .filter(([k]) => k !== excludeKey)
447
+ .sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0)));
448
+ }
449
+ /**
450
+ * Build the ingress check this config asks for, or nothing at all.
451
+ *
452
+ * `cache` is the provider path's per-label-set memo; a workspace passes none,
453
+ * and re-checks on every call for the same reason its egress check does —
454
+ * creating a workspace is a rare, explicit act with nothing to amortise, and
455
+ * a policy deleted since the last call must be noticed.
456
+ */
457
+ export function buildIngressVerifier(client, config, cache) {
458
+ if (!ingressVerificationEnabled(config.ingress))
459
+ return undefined;
460
+ const engine = resolveIngressEngine(config.ingress, config.egress?.engine);
461
+ const agentPort = config.agentPort ?? DEFAULT_AGENT_PORT;
462
+ // Same exclusion, same reason, as the egress union memo above: a
463
+ // once-ever label value in the key would make this memo a per-acquire
464
+ // miss and an unbounded Map. See {@link policyCacheKey}.
465
+ const perSandboxKey = perSandboxEgressLabelKey(config.egress);
466
+ return async (podLabels, subject, signal) => {
467
+ const target = { namespace: config.namespace, podLabels, agentPort, engine, subject };
468
+ if (cache === undefined) {
469
+ await verifyIngressPolicyApplied(client, target, signal);
470
+ return;
471
+ }
472
+ const key = policyCacheKey(podLabels, perSandboxKey);
473
+ let pending = cache.get(key);
474
+ if (pending === undefined) {
475
+ pending = verifyIngressPolicyApplied(client, target, signal).catch((err) => {
476
+ cache.delete(key);
477
+ throw err;
478
+ });
479
+ cache.set(key, pending);
480
+ }
481
+ await pending;
482
+ };
257
483
  }
258
484
  /** Config → the client's own access shape. Shared with `workspace.ts`. */
259
485
  export function clientAccess(config) {
@@ -267,6 +493,359 @@ export function clientAccess(config) {
267
493
  ...(access.ca !== undefined ? { ca: access.ca } : {}),
268
494
  };
269
495
  }
496
+ /**
497
+ * An acquire that was refused, carrying WHY in a field rather than in prose.
498
+ *
499
+ * Before this class a burst past node capacity and an API outage were the
500
+ * same plain `Error`, and a host could only tell them apart by matching
501
+ * message text that any release is free to reword. `reason` is the diagnosis,
502
+ * `retryable` is the advice that follows from it, and `cause` is the original
503
+ * failure — unmodified, so a host that already catches
504
+ * {@link ReadinessPollTimeout}, {@link KubernetesApiTimeoutError} or
505
+ * `KubernetesCredentialError` finds it there.
506
+ *
507
+ * `retryable` is about THIS acquire being worth attempting again, not about
508
+ * anything having been retried. Transient API failures are already retried
509
+ * inside the readiness budget, so a `retryable: true` that reaches a caller
510
+ * means the whole budget was spent on them.
511
+ *
512
+ * Not every acquire failure becomes one of these, and that is deliberate: a
513
+ * refusal this class cannot honestly diagnose — a malformed template, a 400
514
+ * from an admission webhook, a controller that reported Ready and named no
515
+ * sandbox — travels out as itself rather than being filed under whichever of
516
+ * the seven reasons is least wrong. `KubernetesApiError` carries the status
517
+ * for those.
518
+ */
519
+ export class KubernetesAcquireError extends Error {
520
+ name = 'KubernetesAcquireError';
521
+ reason;
522
+ /** Whether attempting the same acquire again could plausibly succeed. */
523
+ retryable;
524
+ /**
525
+ * `status.conditions[Ready].reason`, verbatim, when the controller
526
+ * refused the claim — `WarmPoolNotFound`, `TemplateNotFound`,
527
+ * `InvalidMetadata`, `EnvVarsInjectionRejected`. Present only for
528
+ * `'claim-rejected'`.
529
+ */
530
+ controllerReason;
531
+ /** The controller's own message for the same condition. */
532
+ controllerMessage;
533
+ constructor(details) {
534
+ super(details.message, details.cause !== undefined ? { cause: details.cause } : undefined);
535
+ this.reason = details.reason;
536
+ this.retryable = details.retryable;
537
+ if (details.controllerReason !== undefined)
538
+ this.controllerReason = details.controllerReason;
539
+ if (details.controllerMessage !== undefined)
540
+ this.controllerMessage = details.controllerMessage;
541
+ }
542
+ }
543
+ /**
544
+ * The `status.conditions[Ready].reason` values that mean the controller has
545
+ * DECIDED, so waiting is pointless.
546
+ *
547
+ * ## Where these strings came from
548
+ *
549
+ * Not from the issue that asked for this, and not from upstream source: this
550
+ * repo vendors none of agent-sandbox's Go, so a literal copied out of a
551
+ * changelog is a literal nobody here can check. Each of the four was produced
552
+ * against the deployed controller (kind v1.37.0, agent-sandbox v1.0.2,
553
+ * 2026-09-17) by making the claim it describes and reading the condition
554
+ * back:
555
+ *
556
+ * | reason | how it was produced | the controller's message |
557
+ * |---|---|---|
558
+ * | `WarmPoolNotFound` | claim at a pool that does not exist | `SandboxWarmPool "…" not found` |
559
+ * | `TemplateNotFound` | claim at a pool whose template does not exist | `SandboxTemplate "…" not found` |
560
+ * | `InvalidMetadata` | claim with an `additionalPodMetadata` label outside the allowed domains | `invalid additionalPodMetadata: …` |
561
+ * | `EnvVarsInjectionRejected` | claim with `spec.env` against a template that forbids injection | `environment variable injection rejected: …` |
562
+ *
563
+ * The transient reasons seen on the SAME cluster, which must NOT be in this
564
+ * set, were `DependenciesNotReady` (pod exists, still Pending) and
565
+ * `DependenciesReady` (the Ready=True reason).
566
+ *
567
+ * ## Why an unknown reason is not terminal
568
+ *
569
+ * A wrong literal here fails in one of two ways, and only one of them is
570
+ * recoverable. Too few entries: a rejected claim waits out the readiness
571
+ * budget, which is exactly the behaviour every release before this one had.
572
+ * Too many: an acquire that would have succeeded is refused on a guess. So
573
+ * the set is a closed list of measured strings and everything else falls
574
+ * through to the deadline.
575
+ */
576
+ // Frozen because it is exported from the package root: an array handed to
577
+ // every consumer is one a cast can push onto, and an entry added there would
578
+ // change fail-fast for the whole process. The `readonly string[]` annotation
579
+ // is deliberate rather than `as const` — `includes` on a literal tuple only
580
+ // accepts the literals, and the whole point is to ask it about a reason no
581
+ // one here has seen.
582
+ export const TERMINAL_CLAIM_REASONS = Object.freeze([
583
+ 'WarmPoolNotFound',
584
+ 'TemplateNotFound',
585
+ 'InvalidMetadata',
586
+ 'EnvVarsInjectionRejected',
587
+ ]);
588
+ /**
589
+ * The `Ready` condition a claim reports, whatever its status — the one
590
+ * {@link isConditionTrue} deliberately cannot return, because it answers a
591
+ * boolean question and this one needs the reason.
592
+ */
593
+ function readyCondition(conditions) {
594
+ return conditions?.find((c) => c.type === READY_CONDITION);
595
+ }
596
+ /**
597
+ * A claim the controller has refused, or `undefined` for one it is still
598
+ * working on.
599
+ *
600
+ * `status: 'False'` alone is not a refusal — it is also what a claim looks
601
+ * like for the whole of a cold start — so the REASON decides, against
602
+ * {@link TERMINAL_CLAIM_REASONS}.
603
+ *
604
+ * ONE class comes out of here whatever the reason, and that is deliberate:
605
+ * `InvalidMetadata` is how the controller refuses a pod label whose domain is
606
+ * not on its allowlist — the profile's, today — and a host's `catch` must not
607
+ * have to be written differently depending on whether a profile happens to be
608
+ * configured. `metadata` only decides what rides along as the `cause`: a
609
+ * {@link KubernetesPodLabelsRejectedError} naming the map that was sent
610
+ * and the `config.egress.profileLabelKey` that moves it, which the
611
+ * controller's own message cannot know about.
612
+ */
613
+ export function classifyClaimRejection(claim, claimName, metadata) {
614
+ const condition = readyCondition(claim?.status?.conditions);
615
+ if (condition === undefined || condition.status !== 'False')
616
+ return undefined;
617
+ const reason = condition.reason;
618
+ if (reason === undefined || !TERMINAL_CLAIM_REASONS.includes(reason))
619
+ return undefined;
620
+ // Only for the reason the pod metadata can actually cause, and only when
621
+ // this backend sent any: a `WarmPoolNotFound` carrying a labels-and-
622
+ // allowlist explanation would send an operator after the wrong thing.
623
+ const cause = reason === 'InvalidMetadata' &&
624
+ metadata !== undefined &&
625
+ Object.keys(metadata.requestedPodLabels).length > 0
626
+ ? new KubernetesPodLabelsRejectedError(metadata.requestedPodLabels, claimName, metadata.namespace, reason, condition.message ?? '(the controller reported no message)', metadata.profile)
627
+ : undefined;
628
+ return new KubernetesAcquireError({
629
+ reason: 'claim-rejected',
630
+ retryable: false,
631
+ controllerReason: reason,
632
+ ...(condition.message !== undefined ? { controllerMessage: condition.message } : {}),
633
+ ...(cause !== undefined ? { cause } : {}),
634
+ message: `kubernetes: the agent-sandbox controller refused SandboxClaim ${claimName} with reason ${reason}${condition.message !== undefined ? `: ${condition.message}` : ''}. That is a decision, not a delay, so the readiness budget was not waited out.${cause !== undefined ? ` ${cause.message}` : ''}`,
635
+ });
636
+ }
637
+ /**
638
+ * How long to wait before repeating a failed readiness read, or `undefined`
639
+ * when the failure is not worth repeating.
640
+ *
641
+ * Retryable: a connect failure (the socket, not the answer), a request the
642
+ * `apiRequestTimeoutMs` bound gave up on, a 429 (the API server's own
643
+ * priority-and-fairness queue shedding load) and any 5xx. Not retryable: 401
644
+ * and 403, which are a decision; 404 and 410, which are an answer; 409, which
645
+ * a caller resolves by re-reading; and everything this backend threw itself.
646
+ *
647
+ * The wait is the poll's own cadence unless the server named one — then its
648
+ * `Retry-After`, because the server knows when its queue drains and this code
649
+ * does not. Nothing here consults a clock: the caller's deadline owns the
650
+ * sleep, so a long `Retry-After` spends the readiness budget rather than
651
+ * extending it.
652
+ */
653
+ /**
654
+ * The largest delay a timer can hold — `2^31 - 1` ms, Node's own ceiling.
655
+ * Above it `setTimeout` warns and fires immediately, which is the opposite of
656
+ * what a long `Retry-After` asked for.
657
+ */
658
+ const MAX_RETRY_DELAY_MS = 2_147_483_647;
659
+ export function retryDelayForApiFailure(err, pollIntervalMs) {
660
+ if (err instanceof KubernetesApiTimeoutError)
661
+ return pollIntervalMs;
662
+ if (!(err instanceof KubernetesApiError))
663
+ return undefined;
664
+ if (err.transport === 'connect')
665
+ return pollIntervalMs;
666
+ const status = err.status;
667
+ if (status === undefined)
668
+ return undefined;
669
+ // `Retry-After` may only ever SLOW the poll down, and only within what a
670
+ // timer can express. A header of `0`, or one naming a moment already past,
671
+ // would otherwise turn the retry into a hot loop against a server that is
672
+ // already shedding load — the caller's own cadence is the rate this loop
673
+ // runs at when nothing is wrong. And a header naming a moment years away
674
+ // overflows `setTimeout`, which then fires at once rather than never,
675
+ // producing the same hot loop from the opposite direction. The readiness
676
+ // deadline ends the wait either way; the clamp only stops the wait from
677
+ // silently becoming no wait at all.
678
+ if (status === 429 || status >= 500) {
679
+ const asked = err.retryAfterMs ?? pollIntervalMs;
680
+ return Math.min(Math.max(asked, pollIntervalMs), MAX_RETRY_DELAY_MS);
681
+ }
682
+ return undefined;
683
+ }
684
+ /**
685
+ * How long the one diagnostic pod read after a failed acquire may take.
686
+ *
687
+ * Same shape and the same argument as `runFailureCleanup`'s grace: the
688
+ * readiness clock has already expired, so this cannot share it, and a
689
+ * diagnosis that could hang would keep `create()` pending past the budget the
690
+ * caller chose — for a nicer error message. One second, and a diagnosis that
691
+ * does not arrive is simply not made.
692
+ */
693
+ const ACQUIRE_DIAGNOSIS_GRACE_MS = 1_000;
694
+ /**
695
+ * The image-pull `status.containerStatuses[].state.waiting.reason` values the
696
+ * kubelet reports. Measured on kind v1.37.0 (2026-09-17): a container whose
697
+ * image does not exist waits as `ErrImagePull` for the first attempts and
698
+ * settles into `ImagePullBackOff`. The other two are the kubelet's names for
699
+ * a pull that resolved and then failed, and for a registry that cannot be
700
+ * reached at all.
701
+ */
702
+ const IMAGE_PULL_WAITING_REASONS = [
703
+ 'ErrImagePull',
704
+ 'ImagePullBackOff',
705
+ 'ImageInspectError',
706
+ 'RegistryUnavailable',
707
+ ];
708
+ /**
709
+ * Ask the pod why it is not ready, once, after the budget has already gone.
710
+ *
711
+ * Nothing on the healthy path calls this and nothing waits on it: it runs
712
+ * exactly when an acquire has already failed, and its whole output is a
713
+ * better {@link KubernetesAcquireFailureReason} than `'not-ready'`. A read
714
+ * that fails, a pod that is not there and a pod with nothing to say all
715
+ * produce `undefined`, which leaves the reason where it was.
716
+ *
717
+ * It must run BEFORE the cleanup DELETE, because the pod goes away with the
718
+ * object it belongs to.
719
+ *
720
+ * It takes the CALLER's signal where `runFailureCleanup` deliberately does
721
+ * not: cleanup must finish or the cluster keeps the object, while a diagnosis
722
+ * is only a better sentence for an error a caller who aborted will never
723
+ * read.
724
+ */
725
+ async function diagnoseUnreadyPod(client, namespace, podName, callerSignal) {
726
+ let pod;
727
+ try {
728
+ const deadline = new OperationDeadline(ACQUIRE_DIAGNOSIS_GRACE_MS, 'kubernetes acquire diagnosis', callerSignal);
729
+ pod = await deadline.run((signal) => client.request('GET', podPath(namespace, podName), undefined, signal));
730
+ }
731
+ catch {
732
+ // The acquire failure is the primary one and keeps its reason. A
733
+ // diagnosis that cannot be made is not a second failure to report.
734
+ return undefined;
735
+ }
736
+ const scheduled = pod?.status?.conditions?.find((c) => c.type === 'PodScheduled');
737
+ if (scheduled?.status === 'False' && scheduled.reason === 'Unschedulable')
738
+ return 'capacity';
739
+ for (const container of pod?.status?.containerStatuses ?? []) {
740
+ const reason = container.state?.waiting?.reason;
741
+ if (reason !== undefined && IMAGE_PULL_WAITING_REASONS.includes(reason))
742
+ return 'image-pull';
743
+ }
744
+ return undefined;
745
+ }
746
+ /**
747
+ * The refusal a caller sees, given the failure that actually happened and
748
+ * whatever the pod had to say about it.
749
+ *
750
+ * Returns `undefined` for a failure none of the seven reasons describes —
751
+ * see {@link KubernetesAcquireError} for why that is a deliberate hole rather
752
+ * than a missing case. A {@link KubernetesAcquireError} that arrived from
753
+ * deeper in (the claim rejection) is returned unchanged: it is already the
754
+ * diagnosis.
755
+ */
756
+ export function classifyAcquireFailure(err, podDiagnosis) {
757
+ if (err instanceof KubernetesAcquireError)
758
+ return err;
759
+ if (err instanceof KubernetesApiTimeoutError) {
760
+ return new KubernetesAcquireError({
761
+ reason: 'api-timeout',
762
+ retryable: true,
763
+ message: `kubernetes: the acquire was refused because the API server did not answer in time — ${err.message}`,
764
+ cause: err,
765
+ });
766
+ }
767
+ if (err instanceof KubernetesCredentialError) {
768
+ return new KubernetesAcquireError({
769
+ reason: 'forbidden',
770
+ retryable: false,
771
+ message: `kubernetes: the acquire was refused because the API server rejected this host's credential — ${err.message}. Check the host ServiceAccount's Role against the RBAC section of the Kubernetes sandbox documentation.`,
772
+ cause: err,
773
+ });
774
+ }
775
+ if (err instanceof KubernetesApiError) {
776
+ // The same predicate the poll retries on, so "worth trying again" has
777
+ // one definition and a caller cannot be told a failure is retryable
778
+ // that the poll would have declined to retry. The interval is
779
+ // irrelevant here — only whether an answer comes back at all.
780
+ if (retryDelayForApiFailure(err, 1) === undefined)
781
+ return undefined;
782
+ return new KubernetesAcquireError({
783
+ reason: 'api-unreachable',
784
+ retryable: true,
785
+ message: `kubernetes: the acquire was refused because the API server could not serve it — ${err.message}`,
786
+ cause: err,
787
+ });
788
+ }
789
+ if (err instanceof ReadinessPollTimeout) {
790
+ // ORDER MATTERS, and this is the order: what the cluster SAID beats
791
+ // what the failures suggest. A pod diagnosis is a condition the API
792
+ // server published about this pod, read after the budget had already
793
+ // gone; `err.cause` is at best the failure the poll was still meeting
794
+ // at that moment. On a saturated cluster both are present at once — a
795
+ // pod nothing can schedule AND an API server shedding load — and
796
+ // reporting `api-unreachable` there would hide the very reason this
797
+ // function exists to produce, and would turn `image-pull`'s
798
+ // `retryable: false` into a `true` that has a host retrying forever
799
+ // against an image reference that will never resolve. A diagnosis also
800
+ // cannot be stale in the way a cause can: it only exists because the
801
+ // API server answered one more read, moments ago.
802
+ if (podDiagnosis === 'capacity') {
803
+ return new KubernetesAcquireError({
804
+ reason: 'capacity',
805
+ retryable: true,
806
+ message: `kubernetes: the acquire was refused because its pod could not be scheduled — the cluster reports PodScheduled=False/Unschedulable. ${err.message}`,
807
+ cause: err,
808
+ });
809
+ }
810
+ if (podDiagnosis === 'image-pull') {
811
+ return new KubernetesAcquireError({
812
+ reason: 'image-pull',
813
+ retryable: false,
814
+ message: `kubernetes: the acquire was refused because its pod's container image will not pull. ${err.message}`,
815
+ cause: err,
816
+ });
817
+ }
818
+ // Nothing measured, so the failures the poll kept meeting decide: a
819
+ // poll that spent its budget retrying API failures did not fail
820
+ // because the sandbox was slow; it failed because the control plane
821
+ // was. `pollForBinding` carries the failure it was STILL meeting onto
822
+ // the timeout, and the reason follows it rather than the timeout.
823
+ const underlying = classifyAcquireFailure(err.cause, undefined);
824
+ if (underlying !== undefined) {
825
+ return new KubernetesAcquireError({
826
+ reason: underlying.reason,
827
+ retryable: underlying.retryable,
828
+ message: underlying.message,
829
+ cause: err,
830
+ });
831
+ }
832
+ return new KubernetesAcquireError({
833
+ reason: 'not-ready',
834
+ retryable: true,
835
+ message: err.message,
836
+ cause: err,
837
+ });
838
+ }
839
+ if (err instanceof OperationDeadlineExpired) {
840
+ return new KubernetesAcquireError({
841
+ reason: 'not-ready',
842
+ retryable: true,
843
+ message: `kubernetes: the acquire ran out of readiness budget — ${err.message}`,
844
+ cause: err,
845
+ });
846
+ }
847
+ return undefined;
848
+ }
270
849
  /**
271
850
  * Claim or create, wait for Ready, read the bound identity back, resolve the
272
851
  * address and learn the pod's uid — or leave nothing behind trying.
@@ -274,8 +853,29 @@ export function clientAccess(config) {
274
853
  * Exported because the sandbox surface is built on top of this record rather
275
854
  * than beside it: one acquire path, one cleanup path, whatever ends up
276
855
  * wrapping them.
856
+ *
857
+ * ## What it refuses with
858
+ *
859
+ * Every refusal this function can diagnose arrives as a
860
+ * {@link KubernetesAcquireError} naming one of seven reasons, with the
861
+ * original failure as its `cause`. Three things stay outside that:
862
+ * configuration refused before anything is created
863
+ * ({@link assertEnforceable}, {@link assertRuntimeClassIsApplicable}), a
864
+ * caller's own abort, and a failure none of the seven reasons honestly
865
+ * describes — see {@link KubernetesAcquireError} for why the last one is a
866
+ * hole on purpose.
867
+ *
868
+ * ## What it retries, and what it will not
869
+ *
870
+ * A readiness GET that fails transiently — a connect failure, a request the
871
+ * API bound gave up on, a 429, a 5xx — is repeated INSIDE the readiness
872
+ * deadline, honouring `Retry-After`. One clock, so a retry spends the budget
873
+ * rather than extending it, and a `create()` cannot outlive the timeout its
874
+ * caller chose. The create POST is never retried: it is not idempotent, and a
875
+ * POST whose answer never arrived may already have committed — which is why
876
+ * cleanup deletes the client-owned name whatever happened.
277
877
  */
278
- export async function acquireKubernetesSandbox(client, config, options, readiness) {
878
+ export async function acquireKubernetesSandbox(client, config, options, readiness, verifyIngress = buildIngressVerifier(client, config), egressBoundary = buildEgressBoundary(client, config, config.sandboxTemplateName)) {
279
879
  options.signal?.throwIfAborted();
280
880
  assertEnforceable(options);
281
881
  assertRuntimeClassIsApplicable(config);
@@ -324,36 +924,183 @@ export async function acquireKubernetesSandbox(client, config, options, readines
324
924
  const createPath = config.warmPoolName !== undefined
325
925
  ? claimCollectionPath(namespace)
326
926
  : sandboxCollectionPath(namespace);
327
- const createBody = config.warmPoolName !== undefined
328
- ? buildClaimBody(namespace, objectName, config.warmPoolName, shutdownTime)
329
- : buildSandboxBody({
927
+ // Composed ONCE, here, and handed to whichever body builder runs below:
928
+ // the claim's `additionalPodMetadata.labels` and a direct Sandbox's pod
929
+ // template metadata are the same map, and the translated policy's selector
930
+ // is built from the same resolution. See `composeAdditionalPodLabels`.
931
+ // Empty — the only case before a profile is configured — means every body
932
+ // below is byte for byte what it was.
933
+ // The per-sandbox selector label rides in the SAME map, through the same
934
+ // composer, as a second key rather than a second construction — the whole
935
+ // reason `composeAdditionalPodLabels` takes an `extra`. Its value is the
936
+ // name of the object this acquire is about to create, which is unique per
937
+ // acquire and known BEFORE the POST: that is what lets the label travel as
938
+ // claim-time pod metadata (warm-safe, no cold start) instead of as a patch
939
+ // to a running pod this backend has no verb for.
940
+ const perSandboxLabelKey = perSandboxEgressLabelKey(config.egress);
941
+ const podLabels = composeAdditionalPodLabels(config.egress, perSandboxLabelKey !== undefined ? { [perSandboxLabelKey]: objectName } : undefined);
942
+ const profile = egressProfileLabel(config.egress);
943
+ // Read off the create reply, and off the readiness polls if that reply
944
+ // carried no object — it is the uid a per-sandbox policy's
945
+ // `ownerReferences` names and the suffix of its name, so the cluster can
946
+ // garbage-collect the policy with the object this backend owns.
947
+ let ownerUid;
948
+ const buildCreateBody = async () => {
949
+ if (config.warmPoolName !== undefined) {
950
+ return buildClaimBody({
951
+ namespace,
952
+ name: objectName,
953
+ warmPoolName: config.warmPoolName,
954
+ shutdownTime,
955
+ ...(config.claimLabels !== undefined ? { labels: config.claimLabels } : {}),
956
+ podLabels,
957
+ });
958
+ }
959
+ const template = await deadline.run((signal) => readSandboxTemplate(client, namespace, config.sandboxTemplateName, signal));
960
+ // A direct Sandbox's pod labels are decided HERE, by the body below,
961
+ // so both network boundaries are checked against the real labels before
962
+ // anything is created — a refusal leaves no Sandbox and no PVC behind
963
+ // rather than one of each to clean up. This is EARLIER than the plan
964
+ // for the egress union check asked for (it said "after binding", for
965
+ // the bound pod's labels); a pool-less Sandbox's labels are knowable
966
+ // before the POST, and refusing with nothing created is strictly
967
+ // better than refusing with an object to clean up.
968
+ const directPodLabels = sandboxPodLabels(template, config.sandboxTemplateName, podLabels);
969
+ const directSubject = `to create Sandbox ${objectName} in namespace ${namespace}`;
970
+ if (verifyIngress !== undefined) {
971
+ await deadline.run((signal) => verifyIngress(directPodLabels, directSubject, signal));
972
+ }
973
+ if (egressBoundary !== undefined) {
974
+ await deadline.run((signal) => egressBoundary.verifyUnion(directPodLabels, directSubject, signal));
975
+ }
976
+ return buildSandboxBody({
330
977
  namespace,
331
978
  name: objectName,
332
- template: await deadline.run((signal) => readSandboxTemplate(client, namespace, config.sandboxTemplateName, signal)),
979
+ template,
333
980
  sandboxTemplateName: config.sandboxTemplateName,
334
981
  shutdownTime,
982
+ podLabels,
335
983
  ...(config.runtimeClassName !== undefined
336
984
  ? { runtimeClassName: config.runtimeClassName }
337
985
  : {}),
338
986
  });
987
+ };
988
+ let createBody;
989
+ try {
990
+ createBody = await buildCreateBody();
991
+ }
992
+ catch (err) {
993
+ // Nothing exists yet, so there is nothing to clean up and no pod to
994
+ // ask — but a 403 on the template read is still a `forbidden` acquire,
995
+ // and a caller should not have to tell that apart by where it
996
+ // happened.
997
+ throw classifyAcquireFailure(err, undefined) ?? err;
998
+ }
999
+ // The pod the diagnosis below asks, when there is one. A directly created
1000
+ // Sandbox is backed by a pod of its own name; a CLAIM's pod is not knowable
1001
+ // until the controller has named a sandbox in `status.sandbox`, which it
1002
+ // does before Ready on a cold start and never on a rejected claim.
1003
+ let diagnosablePodName = config.warmPoolName !== undefined ? undefined : objectName;
1004
+ // One definition of "worth trying again", shared by the poll that retries
1005
+ // and the classification that reports — see {@link retryDelayForApiFailure}.
1006
+ const pollBehaviour = {
1007
+ retryDelayFor: (err) => retryDelayForApiFailure(err, readiness.pollIntervalMs),
1008
+ };
339
1009
  try {
340
1010
  // Inside the cleanup block: a POST that fails client-side may still have
341
1011
  // committed, so the only safe assumption is that the object exists.
342
- await deadline.run((signal) => client.request('POST', createPath, createBody, signal));
1012
+ const created = await deadline.run((signal) => client.request('POST', createPath, createBody, signal));
1013
+ ownerUid ??= created?.metadata?.uid;
343
1014
  const binding = config.warmPoolName !== undefined
344
- ? await pollForBinding(async (signal) => bindingFromClaim(await client.request('GET', claimPath(namespace, objectName), undefined, signal), objectName), deadline, readiness, `claim ${objectName}`)
345
- : await pollForBinding(async (signal) => bindingFromSandbox(await client.request('GET', sandboxPath(namespace, objectName), undefined, signal)), deadline, readiness, `sandbox ${objectName}`);
346
- const token = await deadline.run((signal) => readPodBindToken(client, namespace, binding, signal));
1015
+ ? await pollForBinding(async (signal) => {
1016
+ const claim = await client.request('GET', claimPath(namespace, objectName), undefined, signal);
1017
+ diagnosablePodName = claim?.status?.sandbox?.name ?? diagnosablePodName;
1018
+ ownerUid ??= claim?.metadata?.uid;
1019
+ // The pod labels this backend asked the controller for
1020
+ // travel into the read, not because the fail-fast needs
1021
+ // them — `InvalidMetadata` is already one of
1022
+ // `TERMINAL_CLAIM_REASONS`, and that is the ONE
1023
+ // taxonomy a refused claim is reported under — but so
1024
+ // the refusal can carry what was actually sent and how
1025
+ // to change it. See {@link ClaimPodMetadataContext}.
1026
+ return bindingFromClaim(claim, objectName, {
1027
+ namespace,
1028
+ requestedPodLabels: podLabels,
1029
+ ...(profile !== undefined ? { profile } : {}),
1030
+ });
1031
+ }, deadline, readiness, `claim ${objectName}`, pollBehaviour)
1032
+ : await pollForBinding(async (signal) => {
1033
+ const sandbox = await client.request('GET', sandboxPath(namespace, objectName), undefined, signal);
1034
+ ownerUid ??= sandbox?.metadata?.uid;
1035
+ return bindingFromSandbox(sandbox);
1036
+ }, deadline, readiness, `sandbox ${objectName}`, pollBehaviour);
1037
+ const agentPort = config.agentPort ?? DEFAULT_AGENT_PORT;
1038
+ const mode = config.agentAddress ?? 'service';
1039
+ // One read, two facts: the bind token and — under `'pod-ip'` — the
1040
+ // address, off the same pod. See {@link readAddressedPod}, which under
1041
+ // a configured profile also waits for the label the controller
1042
+ // patches onto the bound pod before anything is admitted.
1043
+ const pod = await readAddressedPod(client, namespace, binding, deadline, readiness, mode, podLabels);
1044
+ // The one refusal this capability must not skip: an unlabelled pod
1045
+ // handed back runs under whatever policy DOES select it while the host
1046
+ // believes it is on a narrower profile. This throws INSIDE the try
1047
+ // block, so the cleanup below releases the claim (or deletes the
1048
+ // Sandbox) exactly as a failed privilege probe does — and the probe
1049
+ // itself, which runs in `admitProbedSandbox` after this function
1050
+ // returns, is therefore never reached with the label unobserved.
1051
+ assertRequestedPodLabelsObserved(pod, podLabels, `Sandbox ${binding.name} in namespace ${namespace}`);
1052
+ // A CLAIMED sandbox's pod was built from the pool's own template, so
1053
+ // its labels are not knowable until the controller has bound one.
1054
+ // Checking here rather than not at all is the trade: a refusal
1055
+ // releases the claim through the cleanup below, which is the same
1056
+ // path a failed privilege probe takes.
1057
+ if (config.warmPoolName !== undefined) {
1058
+ const boundSubject = `the pod bound to Sandbox ${binding.name} in namespace ${namespace}`;
1059
+ if (verifyIngress !== undefined) {
1060
+ await deadline.run((signal) => verifyIngress(pod.labels ?? {}, boundSubject, signal));
1061
+ }
1062
+ // The egress union check asks the same question of the same labels
1063
+ // — which policies select THIS pod — so it runs at the same two
1064
+ // points, and a refusal here releases the claim through the cleanup
1065
+ // below, exactly as a failed privilege probe does.
1066
+ if (egressBoundary !== undefined) {
1067
+ await deadline.run((signal) => egressBoundary.verifyUnion(pod.labels ?? {}, boundSubject, signal));
1068
+ }
1069
+ }
1070
+ // Only when the capability is configured, and only after the pod has
1071
+ // been confirmed to carry the selector label: a policy written for a
1072
+ // pod that never got the label would select nothing while the caller
1073
+ // was told its egress had been narrowed.
1074
+ const owner = perSandboxLabelKey === undefined
1075
+ ? undefined
1076
+ : {
1077
+ kind: (config.warmPoolName !== undefined ? 'SandboxClaim' : 'Sandbox'),
1078
+ name: objectName,
1079
+ uid: assertOwnerUid(ownerUid, objectName, namespace),
1080
+ };
347
1081
  return {
348
1082
  binding,
349
- agent: resolveAgentAddress(binding, config.agentPort ?? DEFAULT_AGENT_PORT, token),
1083
+ agent: resolveAgentAddress(binding, agentPort, pod.uid, {
1084
+ mode,
1085
+ ...(pod.podIP !== undefined ? { podIP: pod.podIP } : {}),
1086
+ }),
1087
+ // Only the literal-address mode gets one — see the field.
1088
+ ...(mode === 'pod-ip'
1089
+ ? { refreshAgent: buildAgentAddressRefresh(client, namespace, binding, agentPort, mode) }
1090
+ : {}),
350
1091
  ownedPath,
1092
+ ...(owner !== undefined ? { owner, perSandboxLabelValue: objectName } : {}),
351
1093
  ttlSeconds,
352
1094
  release,
353
1095
  renew,
354
1096
  };
355
1097
  }
356
1098
  catch (err) {
1099
+ // Asked BEFORE cleanup, because the pod goes away with the object, and
1100
+ // only for a timeout — every other failure already knows what it was.
1101
+ const podDiagnosis = err instanceof ReadinessPollTimeout && diagnosablePodName !== undefined
1102
+ ? await diagnoseUnreadyPod(client, namespace, diagnosablePodName, options.signal)
1103
+ : undefined;
357
1104
  // One cleanup for every way out of the block above, on its own short
358
1105
  // budget: the readiness clock has already expired in the common case,
359
1106
  // so spending it again would either skip cleanup or leave `create()`
@@ -361,29 +1108,148 @@ export async function acquireKubernetesSandbox(client, config, options, readines
361
1108
  await runFailureCleanup(async (signal) => {
362
1109
  await release(signal);
363
1110
  });
364
- throw err;
1111
+ throw classifyAcquireFailure(err, podDiagnosis) ?? err;
365
1112
  }
366
1113
  }
367
- /**
368
- * The claim body, in full. Everything absent from it is absent on purpose:
369
- * no `env`, no `volumeClaimTemplates`, no `additionalPodMetadata`. Either of
370
- * the first two forces a cold start upstream and takes the warm pool away.
371
- *
372
- * `shutdownTime` + `shutdownPolicy: 'Delete'` is the leak guard: it bounds the
373
- * object by the wall clock whatever the host does, so a host that dies
374
- * mid-acquire costs the cluster one TTL rather than one leaked sandbox
375
- * forever. `ttlSecondsAfterFinished` deliberately does NOT appear — its timer
376
- * starts from the Finished condition, which a crashed host never reaches.
377
- */
378
- function buildClaimBody(namespace, name, warmPoolName, shutdownTime) {
1114
+ function buildClaimBody(options) {
1115
+ const { labels, podLabels } = options;
379
1116
  return {
380
1117
  apiVersion: `${SANDBOX_EXTENSIONS_API_GROUP}/${SANDBOX_API_VERSION}`,
381
1118
  kind: 'SandboxClaim',
382
- metadata: { name, namespace },
1119
+ metadata: {
1120
+ name: options.name,
1121
+ namespace: options.namespace,
1122
+ ...(labels !== undefined && Object.keys(labels).length > 0 ? { labels } : {}),
1123
+ },
383
1124
  spec: {
384
- warmPoolRef: { name: warmPoolName },
385
- lifecycle: { shutdownTime, shutdownPolicy: 'Delete' },
1125
+ warmPoolRef: { name: options.warmPoolName },
1126
+ ...(podLabels !== undefined && Object.keys(podLabels).length > 0
1127
+ ? { additionalPodMetadata: { labels: podLabels } }
1128
+ : {}),
1129
+ lifecycle: { shutdownTime: options.shutdownTime, shutdownPolicy: 'Delete' },
1130
+ },
1131
+ };
1132
+ }
1133
+ /**
1134
+ * Refuse a bound pod that does not carry EVERY label this backend asked the
1135
+ * controller to put on it — the egress profile today, and whatever else
1136
+ * `composeAdditionalPodLabels` later contributes to the same map.
1137
+ *
1138
+ * The same set {@link readAddressedPod} waits for, deliberately: a wait that
1139
+ * covered more than the refusal would burn the readiness budget on a label
1140
+ * nothing then checked, and a refusal that covered more than the wait would
1141
+ * refuse a pod that had simply not been patched yet. An empty map (the
1142
+ * unprofiled path) checks nothing and returns.
1143
+ *
1144
+ * Called after {@link readAddressedPod} has already waited on the readiness
1145
+ * deadline, so reaching a missing label here means it never arrived, not that
1146
+ * it had not arrived yet.
1147
+ */
1148
+ function assertRequestedPodLabelsObserved(pod, requested, subject) {
1149
+ for (const [key, value] of Object.entries(requested)) {
1150
+ if (pod.labels?.[key] === value)
1151
+ continue;
1152
+ throw new KubernetesPodLabelNotObservedError({ key, value }, subject, pod.labels ?? {});
1153
+ }
1154
+ }
1155
+ /**
1156
+ * The uid of the object this acquire created, or a refusal naming what was
1157
+ * missing.
1158
+ *
1159
+ * Only reached when `config.egress.perSandbox` is configured, and only after
1160
+ * a create reply and every readiness poll have been read without one — the
1161
+ * API server assigns `metadata.uid` on admission and returns the object it
1162
+ * created, so an object with no uid is an API server that answered something
1163
+ * other than what it was asked for. Refusing is right: without the uid there
1164
+ * is no owner reference, and a per-sandbox policy with no owner is one the
1165
+ * cluster never collects.
1166
+ */
1167
+ function assertOwnerUid(uid, objectName, namespace) {
1168
+ if (uid !== undefined && uid !== '')
1169
+ return uid;
1170
+ throw new KubernetesOwnerUidMissingError(objectName, namespace);
1171
+ }
1172
+ /**
1173
+ * Recover a crashed host's claims: LIST every `SandboxClaim` carrying
1174
+ * `labelSelector`, `DELETE` each, and report what was removed.
1175
+ *
1176
+ * This deletes CLAIMS only. The controller's own garbage collection —
1177
+ * ownerReferences from claim to the Sandbox it bound, and from Sandbox to
1178
+ * Pod and Service — takes the rest down behind it; nothing here reads or
1179
+ * touches a Sandbox or a Pod directly. A claim already gone (raced by the
1180
+ * controller's own TTL reaper, or a second release call) counts as removed
1181
+ * rather than a failure, the same convention every other DELETE in this
1182
+ * backend follows.
1183
+ *
1184
+ * `labelSelector` is REQUIRED and refused, synchronously, before a single
1185
+ * request goes out, if it is absent or empty: a release that could fall back
1186
+ * to matching every claim (or every claim of the template) would delete a
1187
+ * live fleet's work the first time a caller passed one by mistake. There is
1188
+ * no default selector for exactly this reason.
1189
+ */
1190
+ export async function releaseKubernetesTaskSandboxes(config, options) {
1191
+ if (options.labelSelector === '') {
1192
+ throw new Error('kubernetes: releaseKubernetesTaskSandboxes requires a non-empty labelSelector — a release with no selector would delete every SandboxClaim in the namespace, including ones a live host still owns. Pass the selector that names only the claims you mean to recover.');
1193
+ }
1194
+ options.signal?.throwIfAborted();
1195
+ const namespace = config.namespace;
1196
+ const client = createKubernetesClient(clientAccess(config), clientOptions(config));
1197
+ const list = await client.request('GET', claimListPath(namespace, options.labelSelector), undefined, options.signal);
1198
+ const names = (list?.items ?? [])
1199
+ .map((claim) => claim.metadata?.name)
1200
+ .filter((name) => typeof name === 'string' && name !== '');
1201
+ // Concurrent, not one at a time: a crash-recovery release can carry a
1202
+ // whole host's worth of claims, and nothing here needs the ordering a
1203
+ // sequential loop would impose — each DELETE is independent and
1204
+ // idempotent (an already-gone claim is tolerated below). A non-tolerated
1205
+ // failure still rejects the whole call, exactly as a sequential loop
1206
+ // would have on its first such failure.
1207
+ await Promise.all(names.map(async (name) => {
1208
+ try {
1209
+ await client.request('DELETE', claimPath(namespace, name), undefined, options.signal);
1210
+ }
1211
+ catch (err) {
1212
+ if (!(err instanceof KubernetesAlreadyGoneError))
1213
+ throw err;
1214
+ }
1215
+ }));
1216
+ return { deleted: names.length, names };
1217
+ }
1218
+ /**
1219
+ * Read task-pool headroom before admitting more work: three GETs, no writes.
1220
+ *
1221
+ * `config.warmPoolName` is required — this reads the exact object a claim's
1222
+ * `warmPoolRef` names, so a pool-less backend (every create is a direct
1223
+ * Sandbox) has no pool to report on. `activeClaims` is every claim in the
1224
+ * namespace whose `spec.warmPoolRef.name` matches this pool, counted rather
1225
+ * than trusted from a label, because a claim's `warmPoolRef` is the one
1226
+ * field the API itself guarantees. `pendingPods` is every Pod in the
1227
+ * namespace still in phase `Pending` — a coarse but honest signal of
1228
+ * in-flight scale-up the ready-replica count alone does not carry, on
1229
+ * either the warm or the pool-less path.
1230
+ */
1231
+ export async function readKubernetesTaskCapacity(config, options) {
1232
+ options?.signal?.throwIfAborted();
1233
+ if (config.warmPoolName === undefined) {
1234
+ throw new Error('kubernetes: readKubernetesTaskCapacity requires config.warmPoolName — there is no SandboxWarmPool to report on for a backend that creates every sandbox directly.');
1235
+ }
1236
+ const namespace = config.namespace;
1237
+ const warmPoolName = config.warmPoolName;
1238
+ const client = createKubernetesClient(clientAccess(config), clientOptions(config));
1239
+ const [pool, claims, pods] = await Promise.all([
1240
+ client.request('GET', warmPoolPath(namespace, warmPoolName), undefined, options?.signal),
1241
+ client.request('GET', claimCollectionPath(namespace), undefined, options?.signal),
1242
+ client.request('GET', podCollectionPath(namespace), undefined, options?.signal),
1243
+ ]);
1244
+ const activeClaims = (claims?.items ?? []).filter((claim) => claim.spec?.warmPoolRef?.name === warmPoolName).length;
1245
+ const pendingPods = (pods?.items ?? []).filter((pod) => pod.status?.phase === 'Pending').length;
1246
+ return {
1247
+ warmPool: {
1248
+ ready: pool?.status?.readyReplicas ?? 0,
1249
+ desired: pool?.spec?.replicas ?? 0,
386
1250
  },
1251
+ activeClaims,
1252
+ pendingPods,
387
1253
  };
388
1254
  }
389
1255
  /**
@@ -416,21 +1282,14 @@ function buildClaimBody(namespace, name, warmPoolName, shutdownTime) {
416
1282
  * label the translated policy is built to match.
417
1283
  */
418
1284
  export function buildSandboxBody(options) {
419
- const podTemplate = options.template.podTemplate;
420
- const spec = options.runtimeClassName !== undefined
421
- ? { ...podTemplate.spec, runtimeClassName: options.runtimeClassName }
422
- : { ...podTemplate.spec };
423
- const metadata = {
424
- ...podTemplate.metadata,
425
- labels: {
426
- ...podTemplate.metadata?.labels,
427
- ...sandboxTemplateLabel(options.sandboxTemplateName),
428
- },
429
- };
430
1285
  return {
431
1286
  apiVersion: `${SANDBOX_API_GROUP}/${SANDBOX_API_VERSION}`,
432
1287
  kind: 'Sandbox',
433
- metadata: { name: options.name, namespace: options.namespace },
1288
+ metadata: {
1289
+ name: options.name,
1290
+ namespace: options.namespace,
1291
+ ...(options.annotations !== undefined ? { annotations: options.annotations } : {}),
1292
+ },
434
1293
  spec: {
435
1294
  operatingMode: 'Running',
436
1295
  service: true,
@@ -440,10 +1299,69 @@ export function buildSandboxBody(options) {
440
1299
  ...(options.template.volumeClaimTemplates !== undefined
441
1300
  ? { volumeClaimTemplates: options.template.volumeClaimTemplates }
442
1301
  : {}),
443
- podTemplate: { ...podTemplate, metadata, spec },
1302
+ podTemplate: sandboxPodTemplate(options.template, options.sandboxTemplateName, options.runtimeClassName, options.podLabels),
444
1303
  },
445
1304
  };
446
1305
  }
1306
+ /**
1307
+ * The `spec.podTemplate` a directly created Sandbox carries: the template's,
1308
+ * with this backend's overlays — the template label and whatever
1309
+ * {@link composeAdditionalPodLabels} produced ({@link sandboxPodLabels}), and
1310
+ * the configured `runtimeClassName`.
1311
+ *
1312
+ * Its own function because it is now built twice: once into the create POST
1313
+ * by {@link buildSandboxBody}, and once into the JSON Patch that refreshes a
1314
+ * standing workspace's pod template (`workspace.ts`). Two expressions of the
1315
+ * same overlay would drift, and the one that drifted would report a workspace
1316
+ * as off-template forever — the hash under
1317
+ * `sandbox.namzu.ai/pod-template-hash` is taken over exactly this object, so
1318
+ * a second spelling is a second revision.
1319
+ *
1320
+ * `podLabels` is on this signature rather than only on the create body for
1321
+ * exactly that reason. A refresh rewrites `/spec/podTemplate` WHOLE, so a
1322
+ * refresh built without them would PATCH the egress profile off a pod
1323
+ * template that carries it — the replacement pod would come up selected by no
1324
+ * per-profile policy, on a path where nothing re-checks the label, and the
1325
+ * revision stamped beside it would be taken over a template the POST never
1326
+ * writes, so `templateCurrent` would report drift forever.
1327
+ */
1328
+ export function sandboxPodTemplate(template, sandboxTemplateName, runtimeClassName, podLabels) {
1329
+ const podTemplate = template.podTemplate;
1330
+ const spec = runtimeClassName !== undefined
1331
+ ? { ...podTemplate.spec, runtimeClassName }
1332
+ : { ...podTemplate.spec };
1333
+ return {
1334
+ ...podTemplate,
1335
+ metadata: {
1336
+ ...podTemplate.metadata,
1337
+ labels: sandboxPodLabels(template, sandboxTemplateName, podLabels),
1338
+ },
1339
+ spec,
1340
+ };
1341
+ }
1342
+ /**
1343
+ * The labels a directly created Sandbox's pod will carry: whatever the
1344
+ * template declares, plus this backend's own template label, which always
1345
+ * wins because it is the label a policy selector is built to match.
1346
+ *
1347
+ * Its own function because the ingress check has to reason about EXACTLY the
1348
+ * labels {@link buildSandboxBody} stamps, before the POST that stamps them.
1349
+ * Two expressions of the same rule would be one rename away from a check that
1350
+ * verifies a pod nobody creates.
1351
+ *
1352
+ * `extra` is `composeAdditionalPodLabels`'s map — the egress profile today.
1353
+ * It is applied LAST, and so wins over both, for the same reason the template
1354
+ * label wins over the copied template's own: it is a label the translated
1355
+ * policy's selector is built to match, and a pod that matched the selector
1356
+ * only sometimes would be a boundary that applied only sometimes.
1357
+ */
1358
+ export function sandboxPodLabels(template, sandboxTemplateName, extra) {
1359
+ return {
1360
+ ...template.podTemplate.metadata?.labels,
1361
+ ...sandboxTemplateLabel(sandboxTemplateName),
1362
+ ...extra,
1363
+ };
1364
+ }
447
1365
  export async function readSandboxTemplate(client, namespace, sandboxTemplateName, signal) {
448
1366
  const template = await client.request('GET', sandboxTemplatePath(namespace, sandboxTemplateName), undefined, signal);
449
1367
  const podTemplate = template?.spec?.podTemplate;
@@ -459,20 +1377,48 @@ export async function readSandboxTemplate(client, namespace, sandboxTemplateName
459
1377
  /**
460
1378
  * Where the agent answers.
461
1379
  *
462
- * Its own function, and the Service FQDN wins over a pod IP, because the
463
- * address outlives the pod: a suspended-then-resumed workspace comes back as a
464
- * new pod with a new IP behind the same name, and the transport re-resolves
465
- * the name on every dial. A literal IP baked into a long-lived handle is the
466
- * bug that would produce.
1380
+ * Its own function, and under the default mode the Service FQDN wins over a
1381
+ * pod IP, because that address outlives the pod: a suspended-then-resumed
1382
+ * workspace comes back as a new pod with a new IP behind the same name, and
1383
+ * the transport re-resolves the name on every dial. A literal IP baked into a
1384
+ * long-lived handle is the bug that would produce — and it is exactly the bug
1385
+ * `'pod-ip'` accepts, deliberately, in exchange for an address a host outside
1386
+ * the cluster can resolve at all. That mode pays for it by re-reading the IP
1387
+ * on every resume and once after a failed connect.
1388
+ *
1389
+ * `'pod-ip'` takes the address from `pod`, the record the bind token was just
1390
+ * read out of, and NEVER falls back to `binding.podIPs`. The Sandbox's status
1391
+ * is a second source that can name a different pod — the one a resume is
1392
+ * replacing — and an address from one pod with a token from another is the
1393
+ * mismatch that arrives as a flat `unauthorized`.
467
1394
  */
468
- export function resolveAgentAddress(binding, agentPort, token) {
1395
+ export function resolveAgentAddress(binding, agentPort, token, options = {}) {
1396
+ if ((options.mode ?? 'service') === 'pod-ip') {
1397
+ const podIP = options.podIP;
1398
+ if (podIP === undefined || podIP === '') {
1399
+ throw new Error(`kubernetes: sandbox ${binding.name} is configured with agentAddress: 'pod-ip', but the live pod its bind token was read from reported no status.podIP (nor a status.podIPs entry), so there is no address to dial. Every path that binds a pod POLLS for that address until the readiness deadline before this is reached — see readAddressedPod — so a pod that still reports none was never given one: a CNI that did not attach it, or a pod that never got past scheduling. Nothing is taken from the Sandbox's own status here on purpose — that IP may belong to a different pod than the token does.`);
1400
+ }
1401
+ return { kind: 'tcp', host: podIP, port: agentPort, token };
1402
+ }
469
1403
  const host = binding.serviceFQDN ?? binding.podIPs?.[0];
470
1404
  if (host === undefined || host === '') {
471
1405
  throw new Error(`kubernetes: sandbox ${binding.name} reported Ready with neither a serviceFQDN nor a pod IP, so its agent has no address to dial. Set 'service: true' on the SandboxTemplate the pool is built from.`);
472
1406
  }
473
1407
  return { kind: 'tcp', host, port: agentPort, token };
474
1408
  }
475
- function bindingFromClaim(claim, claimName) {
1409
+ /**
1410
+ * Ready-or-not-yet, read off a `SandboxClaim`'s own status — plus the one
1411
+ * "not yet" that is really a "no".
1412
+ *
1413
+ * The rejection check runs BEFORE the readiness check rather than after,
1414
+ * because a refused claim is `Ready=False` forever and the readiness check
1415
+ * cannot tell that from a cold start in progress. See
1416
+ * {@link classifyClaimRejection}.
1417
+ */
1418
+ function bindingFromClaim(claim, claimName, metadata) {
1419
+ const rejection = classifyClaimRejection(claim, claimName, metadata);
1420
+ if (rejection !== undefined)
1421
+ throw rejection;
476
1422
  if (!isConditionTrue(claim?.status?.conditions, READY_CONDITION))
477
1423
  return undefined;
478
1424
  const bound = claim?.status?.sandbox;
@@ -503,26 +1449,64 @@ export function bindingFromSandbox(sandbox) {
503
1449
  ...(status?.selector !== undefined ? { podSelector: status.selector } : {}),
504
1450
  };
505
1451
  }
1452
+ /**
1453
+ * The readiness poll ran out of budget — and nothing else. Every OTHER
1454
+ * failure {@link pollForBinding} meets is rethrown as itself, so this class
1455
+ * is an exact answer to "was it the clock?", which a caller that has to
1456
+ * choose between two timeout messages needs and cannot get from the clock.
1457
+ *
1458
+ * Reading `remainingMs()` after the fact is NOT that answer: the expiry timer
1459
+ * and `performance.now()` are different clocks, and a timer that fires a
1460
+ * fraction of a millisecond early leaves a positive remainder behind an
1461
+ * expiry that has already happened.
1462
+ */
1463
+ export class ReadinessPollTimeout extends Error {
1464
+ name = 'ReadinessPollTimeout';
1465
+ }
506
1466
  /**
507
1467
  * Poll until `read` reports a binding. `read` returns `undefined` for "not
508
1468
  * yet" and throws for a failure worth surfacing; the deadline owns every wait,
509
1469
  * including the sleep between attempts, so an expired clock cannot be extended
510
1470
  * by one more round trip. Shaped after ACI's `pollForRunningIp`.
1471
+ *
1472
+ * The only failure this raises on its own account is
1473
+ * {@link ReadinessPollTimeout}; anything `read` throws travels out unchanged,
1474
+ * unless `behaviour.retryDelayFor` claims it — in which case it is repeated
1475
+ * inside the same budget and, if the budget then runs out, carried onto the
1476
+ * timeout as its `cause`, so a poll that kept failing still says what it kept
1477
+ * seeing.
511
1478
  */
512
- export async function pollForBinding(read, deadline, readiness, label) {
1479
+ export async function pollForBinding(read, deadline, readiness, label, behaviour = {}) {
1480
+ /**
1481
+ * The last failure that was retried rather than raised, and only while it
1482
+ * is still the truth: a read that succeeds clears it, so the timeout
1483
+ * carries a failure the poll was STILL meeting when the budget ran out
1484
+ * rather than a blip it recovered from twenty polls earlier. The
1485
+ * difference is not cosmetic — the acquire's reason is read off this
1486
+ * cause, so a stale one would report an API outage for a poll whose API
1487
+ * was answering fine.
1488
+ */
1489
+ let lastRetried;
513
1490
  while (deadline.remainingMs() > 0) {
1491
+ /** Overrides the poll cadence for one round when the server named one. */
1492
+ let nextDelayMs = readiness.pollIntervalMs;
514
1493
  try {
515
1494
  const binding = await deadline.run(read);
1495
+ lastRetried = undefined;
516
1496
  if (binding)
517
1497
  return binding;
518
1498
  }
519
1499
  catch (err) {
520
1500
  if (err instanceof OperationDeadlineExpired)
521
1501
  break;
522
- throw err;
1502
+ const retryDelayMs = behaviour.retryDelayFor?.(err);
1503
+ if (retryDelayMs === undefined)
1504
+ throw err;
1505
+ lastRetried = err;
1506
+ nextDelayMs = retryDelayMs;
523
1507
  }
524
1508
  try {
525
- await deadline.delay(readiness.pollIntervalMs);
1509
+ await deadline.delay(nextDelayMs);
526
1510
  }
527
1511
  catch (err) {
528
1512
  if (err instanceof OperationDeadlineExpired)
@@ -530,10 +1514,15 @@ export async function pollForBinding(read, deadline, readiness, label) {
530
1514
  throw err;
531
1515
  }
532
1516
  }
533
- throw new Error(`kubernetes: ${label} never became Ready (${readiness.timeoutMs}ms)`);
1517
+ throw new ReadinessPollTimeout(`kubernetes: ${label} never became Ready (${readiness.timeoutMs}ms)${lastRetried !== undefined
1518
+ ? `; the last API failure retried inside that budget was: ${lastRetried instanceof Error ? lastRetried.message : String(lastRetried)}`
1519
+ : ''}`, lastRetried !== undefined ? { cause: lastRetried } : undefined);
534
1520
  }
535
1521
  /**
536
- * The per-instance bind token: the backing pod's `metadata.uid`.
1522
+ * Find the pod a sandbox is currently backed by, and read both facts off it.
1523
+ *
1524
+ * `uid` is the per-instance agent bind token; `podIP` is where that agent
1525
+ * listens, and only `agentAddress: 'pod-ip'` reads it.
537
1526
  *
538
1527
  * The pod is named after its Sandbox in agent-sandbox v1.0.2 — verified
539
1528
  * against a running cluster — but that is an observation, not a documented
@@ -542,7 +1531,7 @@ export async function pollForBinding(read, deadline, readiness, label) {
542
1531
  * changing upstream is a second round trip through `status.selector`, which is
543
1532
  * exactly what the controller publishes the selector for.
544
1533
  */
545
- export async function readPodBindToken(client, namespace, binding, signal) {
1534
+ export async function readBoundPod(client, namespace, binding, signal) {
546
1535
  try {
547
1536
  const pod = await client.request('GET', podPath(namespace, binding.name), undefined, signal);
548
1537
  const uid = pod?.metadata?.uid;
@@ -552,7 +1541,7 @@ export async function readPodBindToken(client, namespace, binding, signal) {
552
1541
  // uid the new agent will refuse. On the acquire path nothing is
553
1542
  // terminating and the filter never fires.
554
1543
  if (uid && isPodLive(pod))
555
- return uid;
1544
+ return boundPod(uid, pod);
556
1545
  }
557
1546
  catch (err) {
558
1547
  if (!(err instanceof KubernetesAlreadyGoneError))
@@ -564,11 +1553,93 @@ export async function readPodBindToken(client, namespace, binding, signal) {
564
1553
  for (const pod of list?.items ?? []) {
565
1554
  const uid = pod.metadata?.uid;
566
1555
  if (uid && isPodLive(pod))
567
- return uid;
1556
+ return boundPod(uid, pod);
568
1557
  }
569
1558
  }
570
1559
  throw new Error(`kubernetes: could not read a pod uid for sandbox ${binding.name} in namespace ${namespace} — no live pod of that name, and its status.selector matched no live pod either (a pod carrying a deletionTimestamp, or in phase Succeeded/Failed, is never bound to). The pod uid is the agent's bind token, so the sandbox is refused rather than returned unauthenticated.`);
571
1560
  }
1561
+ function boundPod(uid, pod) {
1562
+ const podIP = readPodIP(pod);
1563
+ const labels = pod.metadata?.labels;
1564
+ return {
1565
+ uid,
1566
+ ...(podIP !== undefined ? { podIP } : {}),
1567
+ ...(labels !== undefined ? { labels } : {}),
1568
+ };
1569
+ }
1570
+ /**
1571
+ * {@link readBoundPod}, plus — under `'pod-ip'` only — the wait for an
1572
+ * address to go with the token.
1573
+ *
1574
+ * Under the default mode this is the single read it has always been: one
1575
+ * `GET`, in the same place in the same order, because a Service FQDN is
1576
+ * published with the Sandbox and needs nothing from the pod but its uid.
1577
+ *
1578
+ * `'pod-ip'` has to wait, because a LIVE pod is not yet an ADDRESSED pod. A
1579
+ * pod is created `Pending` and carries no `status.podIP` until the CNI has
1580
+ * finished attaching it, and {@link isPodLive} accepts `Pending` on purpose —
1581
+ * the resume path in `workspace.ts` binds its replacement pod long before that
1582
+ * pod is Ready, because `Ready` stays True across the transition and the uid
1583
+ * is the only transition signal there is. Refusing an address-less pod outright
1584
+ * would therefore fail on the NORMAL path, in milliseconds, with the whole
1585
+ * readiness budget unspent. So "live, no address yet" is polled on the same
1586
+ * deadline as everything else on this path, and
1587
+ * {@link resolveAgentAddress}'s own refusal is left as the post-deadline
1588
+ * backstop for a pod that never gets an address at all.
1589
+ *
1590
+ * A failed READ stays fatal, exactly as it was: this is the acquire path,
1591
+ * where nothing is being replaced and a pod that cannot be read is not a pod
1592
+ * that is about to appear.
1593
+ *
1594
+ * `requiredLabels` is the second thing worth waiting for, and it is waited
1595
+ * for in the SAME loop rather than in a second one: an egress profile is a
1596
+ * label the CONTROLLER patches onto the pod it binds, so a pod read the
1597
+ * instant it was bound can be live, addressed and not yet labelled. Two
1598
+ * loops would be two deadlines and two answers to "is this pod ready to be
1599
+ * admitted". Like the address, an expired clock hands the pod back as it is
1600
+ * — the caller decides whether a missing label is fatal, and on the acquire
1601
+ * path it is: see `assertRequestedPodLabelsObserved`.
1602
+ */
1603
+ export async function readAddressedPod(client, namespace, binding, deadline, readiness, mode, requiredLabels) {
1604
+ const required = Object.entries(requiredLabels ?? {});
1605
+ const wanting = (pod) => (mode === 'pod-ip' && pod.podIP === undefined) ||
1606
+ required.some(([key, value]) => pod.labels?.[key] !== value);
1607
+ let pod = await deadline.run((signal) => readBoundPod(client, namespace, binding, signal));
1608
+ while (wanting(pod) && deadline.remainingMs() > 0) {
1609
+ try {
1610
+ await deadline.delay(readiness.pollIntervalMs);
1611
+ pod = await deadline.run((signal) => readBoundPod(client, namespace, binding, signal));
1612
+ }
1613
+ catch (err) {
1614
+ // An expired clock hands the incomplete pod back rather than
1615
+ // replacing it with a bare "deadline expired": the caller's
1616
+ // `resolveAgentAddress` (or `assertRequestedPodLabelsObserved`) then
1617
+ // reports WHICH fact never arrived.
1618
+ if (err instanceof OperationDeadlineExpired)
1619
+ break;
1620
+ throw err;
1621
+ }
1622
+ }
1623
+ return pod;
1624
+ }
1625
+ /**
1626
+ * The re-read a `'pod-ip'` handle follows a replaced pod with: one live-pod
1627
+ * read, then the same address resolution acquire did.
1628
+ *
1629
+ * Built here rather than inside the transport because finding the pod is a
1630
+ * CONTROL-plane act — the by-name GET, the selector fallback, the
1631
+ * liveness filter — and the transport owns none of that. It is handed over as
1632
+ * a closure so the transport can call it without learning what a Sandbox is.
1633
+ */
1634
+ export function buildAgentAddressRefresh(client, namespace, binding, agentPort, mode) {
1635
+ return async (signal) => {
1636
+ const pod = await readBoundPod(client, namespace, binding, signal);
1637
+ return resolveAgentAddress(binding, agentPort, pod.uid, {
1638
+ mode,
1639
+ ...(pod.podIP !== undefined ? { podIP: pod.podIP } : {}),
1640
+ });
1641
+ };
1642
+ }
572
1643
  async function readSandboxSelector(client, namespace, binding, signal) {
573
1644
  try {
574
1645
  const sandbox = await client.request('GET', sandboxPath(namespace, binding.name), undefined, signal);
@@ -607,17 +1678,31 @@ async function readSandboxSelector(client, namespace, binding, signal) {
607
1678
  * whose expiry takes the same cleanup-and-reject path every other refusal
608
1679
  * does, in words that name the hang rather than blame a missing `cat`.
609
1680
  */
610
- async function admitProbedSandbox(acquisition, config, options, probeTimeoutMs) {
1681
+ async function admitProbedSandbox(acquisition, config, options, probeTimeoutMs,
1682
+ /**
1683
+ * Present exactly when `config.egress.perSandbox` is configured — which
1684
+ * is what makes `setNetworkPolicy` present on the handle. See
1685
+ * `per-sandbox-policy.ts`.
1686
+ */
1687
+ setNetworkPolicy) {
611
1688
  const sandbox = buildKubernetesSandbox({
612
1689
  name: acquisition.binding.name,
613
1690
  rootDir: options.workingDirectory,
614
- transport: new KubernetesAgentTransport(acquisition.agent),
1691
+ transport: new KubernetesAgentTransport(acquisition.agent, {
1692
+ // The backend opts in to the stream heartbeat; the transport
1693
+ // option it sets stays undefined for every other tier.
1694
+ heartbeatMs: resolveStreamHeartbeatMs(config.streamHeartbeatMs),
1695
+ ...(acquisition.refreshAgent !== undefined
1696
+ ? { refreshHandle: acquisition.refreshAgent }
1697
+ : {}),
1698
+ }),
615
1699
  release: acquisition.release,
616
1700
  renew: acquisition.renew,
617
1701
  ttlSeconds: acquisition.ttlSeconds,
618
1702
  ...(config.onLeaseRenewalError !== undefined
619
1703
  ? { onRenewalError: config.onLeaseRenewalError }
620
1704
  : {}),
1705
+ ...(setNetworkPolicy !== undefined ? { setNetworkPolicy } : {}),
621
1706
  });
622
1707
  try {
623
1708
  await probeSandboxPrivileges(sandbox, acquisition.binding.name, probeTimeoutMs, options.signal);