@namzu/sandbox 14.0.0 → 16.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. package/CHANGELOG.md +924 -0
  2. package/README.md +369 -14
  3. package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
  4. package/dist/backends/aci-standby-pool/index.js +13 -1
  5. package/dist/backends/aci-standby-pool/index.js.map +1 -1
  6. package/dist/backends/docker/index.d.ts +169 -6
  7. package/dist/backends/docker/index.d.ts.map +1 -1
  8. package/dist/backends/docker/index.js +499 -85
  9. package/dist/backends/docker/index.js.map +1 -1
  10. package/dist/backends/firecracker/index.d.ts.map +1 -1
  11. package/dist/backends/firecracker/index.js +12 -2
  12. package/dist/backends/firecracker/index.js.map +1 -1
  13. package/dist/backends/firecracker/protocol.d.ts +459 -8
  14. package/dist/backends/firecracker/protocol.d.ts.map +1 -1
  15. package/dist/backends/firecracker/protocol.js +136 -0
  16. package/dist/backends/firecracker/protocol.js.map +1 -1
  17. package/dist/backends/firecracker/transport.d.ts +539 -6
  18. package/dist/backends/firecracker/transport.d.ts.map +1 -1
  19. package/dist/backends/firecracker/transport.js +1171 -24
  20. package/dist/backends/firecracker/transport.js.map +1 -1
  21. package/dist/backends/kubernetes/egress-policy.d.ts +1181 -13
  22. package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -1
  23. package/dist/backends/kubernetes/egress-policy.js +2350 -31
  24. package/dist/backends/kubernetes/egress-policy.js.map +1 -1
  25. package/dist/backends/kubernetes/identity.d.ts +193 -0
  26. package/dist/backends/kubernetes/identity.d.ts.map +1 -0
  27. package/dist/backends/kubernetes/identity.js +147 -0
  28. package/dist/backends/kubernetes/identity.js.map +1 -0
  29. package/dist/backends/kubernetes/index.d.ts +678 -33
  30. package/dist/backends/kubernetes/index.d.ts.map +1 -1
  31. package/dist/backends/kubernetes/index.js +1180 -95
  32. package/dist/backends/kubernetes/index.js.map +1 -1
  33. package/dist/backends/kubernetes/ingress-policy.d.ts +375 -0
  34. package/dist/backends/kubernetes/ingress-policy.d.ts.map +1 -0
  35. package/dist/backends/kubernetes/ingress-policy.js +1050 -0
  36. package/dist/backends/kubernetes/ingress-policy.js.map +1 -0
  37. package/dist/backends/kubernetes/k8s-client.d.ts +213 -4
  38. package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -1
  39. package/dist/backends/kubernetes/k8s-client.js +359 -52
  40. package/dist/backends/kubernetes/k8s-client.js.map +1 -1
  41. package/dist/backends/kubernetes/lease.d.ts +40 -14
  42. package/dist/backends/kubernetes/lease.d.ts.map +1 -1
  43. package/dist/backends/kubernetes/lease.js +68 -18
  44. package/dist/backends/kubernetes/lease.js.map +1 -1
  45. package/dist/backends/kubernetes/objects.d.ts +423 -3
  46. package/dist/backends/kubernetes/objects.d.ts.map +1 -1
  47. package/dist/backends/kubernetes/objects.js +364 -2
  48. package/dist/backends/kubernetes/objects.js.map +1 -1
  49. package/dist/backends/kubernetes/per-sandbox-policy.d.ts +219 -0
  50. package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -0
  51. package/dist/backends/kubernetes/per-sandbox-policy.js +375 -0
  52. package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -0
  53. package/dist/backends/kubernetes/rbac.d.ts +153 -0
  54. package/dist/backends/kubernetes/rbac.d.ts.map +1 -0
  55. package/dist/backends/kubernetes/rbac.js +177 -0
  56. package/dist/backends/kubernetes/rbac.js.map +1 -0
  57. package/dist/backends/kubernetes/sandbox.d.ts +81 -14
  58. package/dist/backends/kubernetes/sandbox.d.ts.map +1 -1
  59. package/dist/backends/kubernetes/sandbox.js +149 -15
  60. package/dist/backends/kubernetes/sandbox.js.map +1 -1
  61. package/dist/backends/kubernetes/transport.d.ts +935 -9
  62. package/dist/backends/kubernetes/transport.d.ts.map +1 -1
  63. package/dist/backends/kubernetes/transport.js +1958 -62
  64. package/dist/backends/kubernetes/transport.js.map +1 -1
  65. package/dist/backends/kubernetes/workspace.d.ts +1149 -18
  66. package/dist/backends/kubernetes/workspace.d.ts.map +1 -1
  67. package/dist/backends/kubernetes/workspace.js +2825 -186
  68. package/dist/backends/kubernetes/workspace.js.map +1 -1
  69. package/dist/backends/remote-execution-controller.d.ts +14 -0
  70. package/dist/backends/remote-execution-controller.d.ts.map +1 -1
  71. package/dist/backends/remote-execution-controller.js.map +1 -1
  72. package/dist/index.d.ts +294 -18
  73. package/dist/index.d.ts.map +1 -1
  74. package/dist/index.js +280 -10
  75. package/dist/index.js.map +1 -1
  76. package/dist/testing/sandbox-conformance.d.ts +39 -5
  77. package/dist/testing/sandbox-conformance.d.ts.map +1 -1
  78. package/dist/testing/sandbox-conformance.js +436 -5
  79. package/dist/testing/sandbox-conformance.js.map +1 -1
  80. package/package.json +3 -3
  81. package/src/backends/aci-standby-pool/index.ts +16 -1
  82. package/src/backends/docker/index.ts +617 -100
  83. package/src/backends/firecracker/index.ts +14 -2
  84. package/src/backends/firecracker/protocol.ts +514 -6
  85. package/src/backends/firecracker/transport.ts +1492 -40
  86. package/src/backends/kubernetes/egress-policy.ts +3334 -55
  87. package/src/backends/kubernetes/identity.ts +261 -0
  88. package/src/backends/kubernetes/index.ts +1785 -127
  89. package/src/backends/kubernetes/ingress-policy.ts +1344 -0
  90. package/src/backends/kubernetes/k8s-client.ts +444 -54
  91. package/src/backends/kubernetes/lease.ts +75 -19
  92. package/src/backends/kubernetes/objects.ts +626 -6
  93. package/src/backends/kubernetes/per-sandbox-policy.ts +497 -0
  94. package/src/backends/kubernetes/rbac.ts +192 -0
  95. package/src/backends/kubernetes/sandbox.ts +218 -20
  96. package/src/backends/kubernetes/transport.ts +2733 -124
  97. package/src/backends/kubernetes/workspace.ts +4476 -222
  98. package/src/backends/remote-execution-controller.ts +14 -0
  99. package/src/index.ts +668 -19
  100. package/src/testing/sandbox-conformance.ts +540 -5
package/src/index.ts CHANGED
@@ -9,11 +9,29 @@
9
9
  *
10
10
  * Two tiers, each a trust boundary:
11
11
  *
12
- * • `container` — one OCI container per task, seccomp on, tmpfs
13
- * workdir, no network unless asked. The same path on a laptop and
14
- * on a Linux replica anywhere. The tier for trusted prompts and
15
- * contained workloads. Boundary: kernel namespaces, or a
16
- * userspace-kernel runtime where one is installed.
12
+ * • `container` — one OCI container per task, with the container itself
13
+ * as the boundary: kernel namespaces, or a userspace-kernel runtime
14
+ * where one is installed. The same path on a laptop and on a Linux
15
+ * replica anywhere. The tier for trusted prompts and contained
16
+ * workloads. What confines the workload INSIDE the container is
17
+ * applied by each backend and documented where it is applied, not
18
+ * promised here: `container:docker` drops every capability, sets
19
+ * no-new-privileges, gives the container an IPC namespace nothing else can
20
+ * join and mounts its root filesystem read-only over a named writable set
21
+ * (see `backends/docker/index.ts`), while the ACI standby pool takes
22
+ * its controls from the container-group profile its pool was built
23
+ * from. This line used to claim "seccomp on, tmpfs workdir, no
24
+ * network unless asked" on both their behalf, and none of the three
25
+ * is a property this package establishes: nothing here passes a
26
+ * seccomp flag, so what filters a container's syscalls is the daemon's
27
+ * own profile rather than a default this package sets; the directory
28
+ * the agent works in is the layout's `outputs` bind mount — a host
29
+ * directory the run is collected from — rather than a tmpfs that would
30
+ * lose it when the container exits; and the network a container is
31
+ * attached to is a fact about
32
+ * the backend's own configuration — a daemon network whose internals
33
+ * the egress policy is checked against, or the container group's
34
+ * subnet or public address.
17
35
  *
18
36
  * • `microvm` — one hardware-virtualized guest per task. The boundary
19
37
  * to reach for when the prompt itself is adversarial, at the cost
@@ -46,14 +64,27 @@ import { buildDockerBackend, resolveLayout } from './backends/docker/index.js'
46
64
  import { buildFirecrackerBackend } from './backends/firecracker/index.js'
47
65
  import type { KubernetesEgressConfig } from './backends/kubernetes/egress-policy.js'
48
66
  import {
67
+ type KubernetesAgentAddressMode,
49
68
  type KubernetesBackendInternalConfig,
50
69
  type KubernetesClusterAccess,
70
+ type KubernetesReadTaskCapacityOptions,
71
+ type KubernetesReleaseTaskSandboxesOptions,
72
+ type KubernetesTaskCapacity,
51
73
  buildKubernetesBackend,
74
+ readKubernetesTaskCapacity as readTaskCapacityOnCluster,
75
+ releaseKubernetesTaskSandboxes as releaseTaskSandboxesOnCluster,
52
76
  } from './backends/kubernetes/index.js'
77
+ import type { KubernetesIngressConfig } from './backends/kubernetes/ingress-policy.js'
53
78
  import {
54
79
  type KubernetesWorkspace,
55
80
  type KubernetesWorkspaceOptions,
81
+ type KubernetesWorkspaceSummary,
82
+ type KubernetesWorkspaceSuspendOptions,
83
+ type KubernetesWorkspaceTransitionOptions,
56
84
  createKubernetesWorkspace as buildKubernetesWorkspace,
85
+ deleteKubernetesWorkspace as deleteWorkspaceOnCluster,
86
+ listKubernetesWorkspaces as listWorkspacesOnCluster,
87
+ suspendKubernetesWorkspace as suspendWorkspaceOnCluster,
57
88
  } from './backends/kubernetes/workspace.js'
58
89
 
59
90
  // Re-export the layout types so consumers of `@namzu/sandbox` can
@@ -89,24 +120,152 @@ export type {
89
120
  OrchestratorTokenProvider,
90
121
  } from './backends/firecracker/index.js'
91
122
  export {
123
+ AgentDialFailedError,
92
124
  AgentPreauthFrameTooLargeError,
125
+ AgentReadFileStreamUnsupportedError,
126
+ AgentWriteFileTooLargeError,
127
+ DEFAULT_MAX_WRITE_FILE_BYTES,
93
128
  FIRECRACKER_AGENT_PROTOCOL_VERSION,
129
+ GUEST_FRAME_LIMIT_BYTES,
94
130
  type SandboxAgentHandle,
95
131
  TCP_PREAUTH_FRAME_LIMIT_BYTES,
96
132
  type VsockTransportOptions,
97
133
  VsockAgentTransport,
98
134
  } from './backends/firecracker/transport.js'
135
+ // The `write-file` part protocol: the healthz feature string a guest
136
+ // advertises when it can take a body larger than one frame, and the shape
137
+ // of one part. Exported so a host writing its own guest, or asserting what
138
+ // this one advertises, names them rather than repeating the literal.
139
+ export {
140
+ WRITE_FILE_PARTS_FEATURE,
141
+ type WriteFilePart,
142
+ } from './backends/firecracker/protocol.js'
143
+ // The per-stream liveness heartbeat: the feature string a guest advertises
144
+ // when it understands one, the frame both sides send, and how many missed
145
+ // intervals end a stream. Same reason as above — a host asserting what this
146
+ // guest advertises should name the string rather than repeat the literal.
147
+ export {
148
+ MIN_STREAM_HEARTBEAT_MS,
149
+ STREAM_HEARTBEAT_FEATURE,
150
+ STREAM_HEARTBEAT_MAX_ECHO_FACTOR,
151
+ STREAM_HEARTBEAT_MISS_LIMIT,
152
+ type StreamHeartbeat,
153
+ } from './backends/firecracker/protocol.js'
154
+ // The read side of the same arrangement: one healthz feature string for
155
+ // both the ranged `read-file` and the `read-file-stream` op, and the
156
+ // request/event shapes they speak. Exported for the same reason — a host
157
+ // writing its own guest, or asserting what this one advertises, names them
158
+ // rather than repeating the literal.
159
+ export {
160
+ READ_FILE_STREAM_FEATURE,
161
+ type ReadFileStreamEvent,
162
+ type ReadFileStreamRequest,
163
+ } from './backends/firecracker/protocol.js'
99
164
 
100
165
  // Kubernetes (agent-sandbox on any cluster) public surface. The access union
101
166
  // is named by `KubernetesBackendConfig.access`, so a host that builds its own
102
- // credential callback can name what it is passing.
103
- export type { KubernetesClusterAccess } from './backends/kubernetes/index.js'
167
+ // credential callback can name what it is passing; the address mode is named
168
+ // by `KubernetesBackendConfig.agentAddress` and decides whether the agent is
169
+ // dialed at its Service FQDN (in-cluster host) or at its pod IP (a host
170
+ // outside the cluster, on a routable pod network).
171
+ export type {
172
+ KubernetesAgentAddressMode,
173
+ KubernetesClusterAccess,
174
+ } from './backends/kubernetes/index.js'
104
175
  // Egress translation types named by `KubernetesBackendConfig.egress` — see
105
- // `backends/kubernetes/egress-policy.ts` for what each engine can express.
176
+ // `backends/kubernetes/egress-policy.ts` for what each engine can express,
177
+ // and what the two Kubernetes-only kinds (`no-network`, `public-internet`)
178
+ // mean that the shared `EgressPolicy` union has no word for.
106
179
  export type {
180
+ EgressProfileLabel,
181
+ KubernetesCiliumDnsNarrowing,
182
+ KubernetesCiliumEgressNarrowing,
107
183
  KubernetesEgressConfig,
108
184
  KubernetesEgressEngine,
185
+ KubernetesEgressPolicy,
186
+ KubernetesEgressVerification,
187
+ KubernetesOnlyEgressPolicy,
188
+ KubernetesPerSandboxEgressConfig,
189
+ } from './backends/kubernetes/egress-policy.js'
190
+ // Per-sandbox egress — `config.egress.perSandbox`, which makes
191
+ // `Sandbox.setNetworkPolicy` PRESENT on a kubernetes TASK handle instead of
192
+ // omitted. Everything a host needs to configure it and to catch its two
193
+ // refusals by class: a capability declared in a way this backend cannot
194
+ // honour (wiring time), and a write refused because the operator-applied
195
+ // admission fence that bounds it is not there — or cannot be read, which is
196
+ // a different file to fix (call time, nothing written in either case).
197
+ // `KubernetesNetworkPolicyHostError` is the third: an `allowedHosts` entry
198
+ // that is not a hostname, or one the configured narrowing cannot express.
199
+ // `KubernetesOwnerUidMissingError` is the fourth, and the only one raised
200
+ // from an ACQUIRE — the object this backend created reported no uid, so a
201
+ // policy written for it would be an orphan.
202
+ // `KubernetesWorkspacePerSandboxEgressConfigError` is the fifth: the same
203
+ // option reaching `createKubernetesWorkspace`, whose handle never carries
204
+ // `setNetworkPolicy` — refused there rather than accepted and ignored.
205
+ // See `backends/kubernetes/per-sandbox-policy.ts` — and, for
206
+ // `KubernetesNetworkPolicyHostError`, which the config-level translation
207
+ // refuses the same entries with, `backends/kubernetes/egress-policy.ts`.
208
+ export { KubernetesPerSandboxEgressConfigError } from './backends/kubernetes/egress-policy.js'
209
+ export { KubernetesWorkspacePerSandboxEgressConfigError } from './backends/kubernetes/egress-policy.js'
210
+ export { DEFAULT_PER_SANDBOX_EGRESS_LABEL_KEY } from './backends/kubernetes/egress-policy.js'
211
+ export {
212
+ KubernetesAdmissionFenceMissingError,
213
+ KubernetesAdmissionFenceUnreadableError,
214
+ KubernetesNetworkPolicyHostError,
215
+ KubernetesOwnerUidMissingError,
216
+ PER_SANDBOX_POLICY_NAME_PREFIX,
217
+ } from './backends/kubernetes/per-sandbox-policy.js'
218
+ // The default label key `KubernetesEgressConfig.profile` is written under.
219
+ // Exported because an operator has to name that key's DOMAIN in the
220
+ // controller's `allowed-label-domains` allowlist before a profiled claim is
221
+ // accepted, and reading it off the package beats copying a string out of a
222
+ // document.
223
+ export { DEFAULT_EGRESS_PROFILE_LABEL_KEY } from './backends/kubernetes/egress-policy.js'
224
+ // The egress union check's refusal and the shapes it reports, so a host can
225
+ // catch an over-wide policy by class and print which policy it was. Separate
226
+ // from `KubernetesEgressPolicyMismatchError` (the ONE named object drifting)
227
+ // and from `KubernetesIngressPolicyError` (the agent port being reachable):
228
+ // three refusals on one create path, told apart by class.
229
+ export type {
230
+ EgressPolicyRefusal,
231
+ EgressPolicyVerdict,
232
+ ExaminedEgressPolicy,
233
+ } from './backends/kubernetes/egress-policy.js'
234
+ export {
235
+ KubernetesEgressNarrowingUnsupportedError,
236
+ KubernetesEgressPolicyConfigError,
237
+ KubernetesEgressPolicyUnionError,
109
238
  } from './backends/kubernetes/egress-policy.js'
239
+ // What an egress PROFILE adds to the refusals above. Two are thrown: a
240
+ // profile this backend will not emit (while the host is still being wired),
241
+ // and a bound pod that never carried the label this backend asked the
242
+ // controller for — refused rather than admitted, because admitting it would
243
+ // run the sandbox under whatever policy DOES select it. The third is never
244
+ // thrown on its own: a claim the controller refused because a label key's
245
+ // domain is not on its allowlist still comes out as
246
+ // `KubernetesAcquireError { reason: 'claim-rejected' }`, and
247
+ // `KubernetesPodLabelsRejectedError` is that error's `cause`, naming the map
248
+ // that was sent and the config key that moves it.
249
+ export {
250
+ KubernetesEgressProfileConfigError,
251
+ KubernetesPodLabelNotObservedError,
252
+ KubernetesPodLabelsRejectedError,
253
+ } from './backends/kubernetes/egress-policy.js'
254
+ // Ingress verification types named by `KubernetesBackendConfig.ingress`, plus
255
+ // the refusal a create raises when no applied policy closes the agent port —
256
+ // catchable by class, and distinct from every other refusal on that path. See
257
+ // `backends/kubernetes/ingress-policy.ts`.
258
+ export type {
259
+ ExaminedIngressPolicy,
260
+ IngressPolicyRefusal,
261
+ IngressPolicyVerdict,
262
+ KubernetesIngressConfig,
263
+ KubernetesIngressEngine,
264
+ /** @deprecated Renamed to `UnreadPolicySource`; both checks report it. */
265
+ UnreadIngressPolicySource,
266
+ UnreadPolicySource,
267
+ } from './backends/kubernetes/ingress-policy.js'
268
+ export { KubernetesIngressPolicyError } from './backends/kubernetes/ingress-policy.js'
110
269
  // The errors a caller of a kubernetes sandbox has to be able to catch BY
111
270
  // CLASS rather than by matching a message: an acquire refused because the
112
271
  // guest is not deprivileged, a call after the handle ended (this host
@@ -120,22 +279,246 @@ export {
120
279
  KubernetesSandboxDestroyedError,
121
280
  KubernetesSandboxGoneError,
122
281
  } from './backends/kubernetes/sandbox.js'
123
- export { KubernetesAgentUnauthorizedError } from './backends/kubernetes/transport.js'
282
+ export {
283
+ KubernetesAgentAddressUnresolvableError,
284
+ // A guest that has fenced itself refuses one CALL here, not the
285
+ // workspace: the Firecracker tier's mapping of the same refusal retires
286
+ // the sandbox, which on a workspace would take the pod away from every
287
+ // other holder. See `backends/kubernetes/transport.ts`.
288
+ KubernetesAgentRetiringError,
289
+ KubernetesAgentUnauthorizedError,
290
+ } from './backends/kubernetes/transport.js'
291
+ // A workspace command that can outlive the connection watching it: the
292
+ // options that ask for one, and the three refusals a caller has to be able
293
+ // to catch BY CLASS — a guest image too old to keep output, an execution
294
+ // that can no longer be attached to (past retention, or in a replaced
295
+ // pod), and an observation this host gave up on WITHOUT cancelling, which
296
+ // names the id and the byte offset another process resumes from.
297
+ export type {
298
+ KubernetesAttachExecutionOptions,
299
+ KubernetesAttachRefusal,
300
+ KubernetesDetachedExecOptions,
301
+ } from './backends/kubernetes/transport.js'
302
+ export {
303
+ KubernetesExecutionAttachUnsupportedError,
304
+ KubernetesExecutionDetachedError,
305
+ KubernetesExecutionNotAttachableError,
306
+ } from './backends/kubernetes/transport.js'
307
+ // Guest sessions: a workspace terminal or background program that outlives
308
+ // the connection — and the host process — that started it. The options that
309
+ // name one, the rows `listSessions()` returns, the `BackgroundJobOutput`
310
+ // shape `readSession()` answers in, and the three refusals a caller catches
311
+ // BY CLASS: a guest image with no session registry, a session the guest
312
+ // answered about and refused (past retention, or in a replaced pod), and an
313
+ // attachment that ended while its program went on running.
314
+ export type {
315
+ KubernetesAttachTerminalOptions,
316
+ KubernetesOpenTerminalOptions,
317
+ KubernetesReadSessionOptions,
318
+ KubernetesSessionOutput,
319
+ KubernetesSessionRefusal,
320
+ KubernetesSessionSummary,
321
+ KubernetesSessionTerminal,
322
+ KubernetesStartDetachedOptions,
323
+ KubernetesWorkspaceTerminal,
324
+ } from './backends/kubernetes/transport.js'
325
+ export {
326
+ KubernetesSessionRefusedError,
327
+ KubernetesSessionsUnsupportedError,
328
+ } from './backends/kubernetes/transport.js'
329
+ export { AgentSessionDetachedError } from './backends/firecracker/transport.js'
330
+ // The `sessions` healthz feature string, and the session vocabulary its
331
+ // frames use. Same reason as the two feature strings above: a host asserting
332
+ // what an image can do should name the string rather than repeat the literal.
333
+ export {
334
+ SESSIONS_FEATURE,
335
+ type SessionDetachReason,
336
+ type SessionKind,
337
+ type SessionState,
338
+ } from './backends/firecracker/protocol.js'
339
+ // Quiesce: stop every process a workspace's guest is running, while the
340
+ // agent goes on serving, so a capture taken next is one nobody is writing
341
+ // under. The report a host reads, the two refusals it catches by class, and
342
+ // — for the same reason as the feature strings above — the `quiesce` string
343
+ // itself and the scope vocabulary its report is written in.
344
+ export type {
345
+ KubernetesQuiesceOptions,
346
+ KubernetesWorkspaceQuiesceRequest,
347
+ } from './backends/kubernetes/workspace.js'
348
+ export type { KubernetesQuiesceReport } from './backends/kubernetes/transport.js'
349
+ export {
350
+ KubernetesQuiesceUnconfirmedError,
351
+ KubernetesQuiesceUnsupportedError,
352
+ } from './backends/kubernetes/transport.js'
353
+ export {
354
+ QUIESCE_FEATURE,
355
+ type QuiesceScope,
356
+ type QuiescedProcess,
357
+ } from './backends/firecracker/protocol.js'
358
+ // Flush: put a workspace's writes on its device on purpose, rather than
359
+ // leaving them to whatever the guest kernel had written back when the pod
360
+ // stopped. `suspend()` runs it by default; the verb, its options, the report
361
+ // and the three named outcomes are exported for the hosts that flush at a
362
+ // moment of their own — before a snapshot, before a drain — and for the
363
+ // `flush` string itself, for the same reason as the feature strings above.
364
+ // Only ONE of the three is ever thrown at a caller by a suspend
365
+ // (`KubernetesFlushUnconfirmedError`, a guest that answered and could not
366
+ // confirm); the other two are what `onFlushUnsupported` and
367
+ // `onFlushUnreachable` are handed when the suspend goes ahead anyway, and a
368
+ // host that wants to act on either has to be able to name the class.
369
+ export type {
370
+ KubernetesFlushOptions,
371
+ KubernetesWorkspaceFlushRequest,
372
+ } from './backends/kubernetes/workspace.js'
373
+ export type { KubernetesFlushReport } from './backends/kubernetes/transport.js'
374
+ export {
375
+ KubernetesFlushUnconfirmedError,
376
+ KubernetesFlushUnreachableError,
377
+ KubernetesFlushUnsupportedError,
378
+ } from './backends/kubernetes/transport.js'
379
+ export { FLUSH_FEATURE } from './backends/firecracker/protocol.js'
380
+ // The API-request bound and the error it raises. Exported because
381
+ // "distinguishable from a caller abort and from every other failure, by
382
+ // type" is only true for a host that can name the class — and because a
383
+ // host that sets `apiRequestTimeoutMs` wants the default and the floor it is
384
+ // choosing against.
385
+ // `KubernetesHttpMethod` rides along because `KubernetesApiTimeoutError.verb`
386
+ // is one: a caller that can catch the class but cannot name the type of the
387
+ // field it is reading is back to inlining the union or reaching for `any`.
388
+ export {
389
+ DEFAULT_API_REQUEST_TIMEOUT_MS,
390
+ // Every API failure the four classes below do not name: a connect
391
+ // failure, and every non-2xx status outside 401/403/404/409/410. It
392
+ // carries the status and the `Retry-After` the server sent, because a
393
+ // burst past node capacity (429) and an API server that is down (a
394
+ // connect failure) were otherwise the same plain `Error`, separable only
395
+ // by matching a message any release is free to reword. It carries no
396
+ // retry policy — see `backends/kubernetes/index.ts` for who decides that.
397
+ KubernetesApiError,
398
+ type KubernetesApiFailureTransport,
399
+ KubernetesApiTimeoutError,
400
+ type KubernetesHttpMethod,
401
+ // A conditional write the API server would not apply. A host that fences
402
+ // its workspaces with a holder epoch normally catches
403
+ // `KubernetesWorkspacePreconditionError` instead — this one survives only
404
+ // when the object kept changing under the write or the patch body was
405
+ // wrong, and a caller that cannot name the class cannot tell it from a
406
+ // cluster failure.
407
+ KubernetesPatchNotAppliedError,
408
+ MIN_API_REQUEST_TIMEOUT_MS,
409
+ } from './backends/kubernetes/k8s-client.js'
410
+ // The three statuses the client maps to a class of their own, so a caller can
411
+ // treat "already gone" as the state a teardown was asking for, re-read after a
412
+ // 409, and tell a rejected credential from a cluster failure — by class, which
413
+ // is the only way that survives a reworded message. They have been thrown
414
+ // since the backend existed and were reachable only by importing a deep path.
415
+ export {
416
+ KubernetesAlreadyGoneError,
417
+ KubernetesConflictError,
418
+ KubernetesCredentialError,
419
+ } from './backends/kubernetes/k8s-client.js'
420
+ // Why an acquire was refused, as a field rather than as prose: the seven
421
+ // reasons, the class that carries one, and the measured list of controller
422
+ // `Ready=False` reasons that mean "decided" rather than "not yet". A host
423
+ // deciding whether to retry, to fail the run or to page an operator reads
424
+ // `reason` and `retryable`; `cause` is the original failure, so a host that
425
+ // already catches `ReadinessPollTimeout` or `KubernetesApiTimeoutError` finds
426
+ // it there. See `backends/kubernetes/index.ts`.
427
+ export {
428
+ KubernetesAcquireError,
429
+ type KubernetesAcquireFailureReason,
430
+ // The poll's own give-up, by type. It is the `cause` of a `'not-ready'`
431
+ // acquire refusal and is raised directly by the workspace lifecycle, which
432
+ // does not go through acquire.
433
+ ReadinessPollTimeout,
434
+ TERMINAL_CLAIM_REASONS,
435
+ } from './backends/kubernetes/index.js'
436
+ // The three ways egress verification refuses: a policy this engine cannot
437
+ // express, no applied object at all, and an applied object that does not
438
+ // match what this configuration translates to. Catchable by class for the
439
+ // same reason the ingress refusal above is — an operator debugging two
440
+ // default-on refusals in one release should not have to read messages to
441
+ // tell them apart.
442
+ export {
443
+ KubernetesEgressPolicyMismatchError,
444
+ KubernetesEgressPolicyNotAppliedError,
445
+ KubernetesUnenforceableEgressPolicyError,
446
+ } from './backends/kubernetes/egress-policy.js'
447
+ /** Default `KubernetesBackendConfig.streamHeartbeatMs` — see there. */
448
+ export { DEFAULT_STREAM_HEARTBEAT_MS } from './backends/kubernetes/index.js'
449
+ // Crash recovery and headroom for the task path: label a claim with a
450
+ // host-supplied identity (`KubernetesBackendConfig.claimLabels`), find and
451
+ // release a predecessor's claims by that label, and read pool headroom
452
+ // before admitting more work. All three are additive — a host that sets no
453
+ // `claimLabels` and calls neither function sees no change at all.
454
+ export type {
455
+ KubernetesReadTaskCapacityOptions,
456
+ KubernetesReleaseTaskSandboxesOptions,
457
+ KubernetesTaskCapacity,
458
+ } from './backends/kubernetes/index.js'
459
+ // What a CLAIM-ONLY host is allowed to do, as data: the verbs the pool-only
460
+ // path issues, each pinned to its call site in `backends/kubernetes/rbac.ts`.
461
+ // `k8s/manifests/rbac-claimant.yaml` grants exactly this and a test parses
462
+ // that file and compares it here, so an operator who has to prove a live
463
+ // `Role` carries no more than this backend needs compares against the same
464
+ // constant rather than against a list copied out of a page.
465
+ export {
466
+ KUBERNETES_CLAIMANT_RBAC_RULES,
467
+ type KubernetesRbacRule,
468
+ type KubernetesRbacVerb,
469
+ } from './backends/kubernetes/rbac.js'
124
470
  // The persistent workspace: a `Sandbox` that keeps a block disk across a
125
- // suspend, plus the four errors its lifecycle can refuse with — a template
126
- // that cannot carry a disk, a standing object that does not match this
127
- // configuration, a call on a suspended workspace, and a suspend whose pod
128
- // outlived the wait. Declared in `@namzu/sandbox` rather than on the SDK's
129
- // `Sandbox` — see `backends/kubernetes/workspace.ts`.
471
+ // suspend, the union naming how a handle came by its object, plus the four
472
+ // errors its lifecycle can refuse with — a template that cannot carry a disk,
473
+ // a standing object that does not match this configuration, a call on a
474
+ // suspended workspace, and a suspend whose pod outlived the wait. Declared in
475
+ // `@namzu/sandbox` rather than on the SDK's `Sandbox` — see
476
+ // `backends/kubernetes/workspace.ts`.
130
477
  export type {
478
+ KubernetesKillSessionOptions,
131
479
  KubernetesWorkspace,
480
+ KubernetesWorkspaceAgentState,
481
+ KubernetesWorkspaceCancellationNotice,
132
482
  KubernetesWorkspaceDestroyOptions,
133
483
  KubernetesWorkspaceOptions,
484
+ KubernetesWorkspaceOrigin,
485
+ KubernetesWorkspaceStartFailurePolicy,
486
+ KubernetesWorkspaceSummary,
487
+ KubernetesWorkspaceSuspendOptions,
488
+ KubernetesWorkspaceSuspensionNotice,
134
489
  KubernetesWorkspaceTransitionOptions,
135
490
  } from './backends/kubernetes/workspace.js'
491
+ // What a workspace handle is bound to, what it says when the guest behind it
492
+ // is replaced, and the two errors that identity produces. A host that keeps
493
+ // per-workspace state — which processes it started, what is on the disk —
494
+ // subscribes to `onGuestRestart` and compares `identity`; both are useless to
495
+ // a host that cannot name their types.
496
+ export type {
497
+ KubernetesGuestEvidence,
498
+ KubernetesGuestRestart,
499
+ KubernetesGuestRestartReason,
500
+ KubernetesWorkspaceIdentity,
501
+ } from './backends/kubernetes/identity.js'
502
+ export {
503
+ // The command's outcome is unknown AND the guest it ran in is gone. A
504
+ // subclass of `RemoteCancellationUnknownError`, so a host catching the
505
+ // base class keeps catching it; what it adds is which guest the command
506
+ // started on and which one is there now.
507
+ KubernetesWorkspaceGuestGoneError,
508
+ // A different Sandbox now stands under the workspace's deterministic
509
+ // name, or none does. The handle refuses rather than following it — the
510
+ // disk behind the name is not the disk it was opened on.
511
+ KubernetesWorkspaceReplacedError,
512
+ } from './backends/kubernetes/identity.js'
136
513
  export {
137
514
  KubernetesWorkspaceDiskError,
138
515
  KubernetesWorkspaceMismatchError,
516
+ // A lifecycle write refused because this caller's holder epoch has been
517
+ // overtaken: the workspace belongs to another process now, nothing on the
518
+ // cluster changed and nothing about the handle changed. A host that fences
519
+ // its workspaces has to be able to tell this from a cluster failure, which
520
+ // is the whole reason the write is conditional.
521
+ KubernetesWorkspacePreconditionError,
139
522
  KubernetesWorkspaceSuspendTimeoutError,
140
523
  KubernetesWorkspaceSuspendedError,
141
524
  } from './backends/kubernetes/workspace.js'
@@ -320,6 +703,46 @@ export interface ContainerBackendConfig {
320
703
  * collisions with Docker / orchestrator labels.
321
704
  */
322
705
  readonly labels?: Readonly<Record<string, string>>
706
+ /**
707
+ * CPU cores the container may use, rendered as `--cpus`. Unset by
708
+ * default, like `memoryLimitMb` and `maxProcesses`, and for the same
709
+ * reason: the value that is right is a property of the host's machine
710
+ * and of the workload, and a number chosen here would silently throttle
711
+ * runs that finish inside their timeout today.
712
+ *
713
+ * It is set at provider construction rather than per `create()` call,
714
+ * because the documented deployment builds one provider per task — and
715
+ * because the ACI and kubernetes backends cannot apply a per-sandbox CPU
716
+ * limit, so a per-call field would be a control they would have to
717
+ * refuse. See `backends/docker/index.ts` for what it renders.
718
+ */
719
+ readonly cpuLimit?: number
720
+ /**
721
+ * Mount the container's root filesystem read-only. Default `true`.
722
+ *
723
+ * On by default with the paths that stay writable named in
724
+ * `backends/docker/index.ts` (`writableRootfsPaths` extends them). Set it
725
+ * to `false` to make the whole container filesystem writable again, which a
726
+ * host whose image writes somewhere the writable set cannot describe needs,
727
+ * and which is why the switch exists rather than the baseline being
728
+ * unconditional. It gives up that one control: the capability drop,
729
+ * `no-new-privileges` and `--ipc private` are applied to every container
730
+ * whatever this says, so it is not a way back to the previous argv.
731
+ */
732
+ readonly readOnlyRootfs?: boolean
733
+ /**
734
+ * Extra paths to keep writable under `--read-only`, each mounted
735
+ * `--tmpfs`.
736
+ *
737
+ * The default set is the reference image's needs, read off its Dockerfile.
738
+ * A host that points `image` at its own build says what that image needs
739
+ * here, because the backend cannot read an image's writable set and the
740
+ * alternative to asking is guessing. Setting this beside
741
+ * `readOnlyRootfs: false` is refused: with a writable root filesystem the
742
+ * mounts would add nothing, and accepting a control that is not applied is
743
+ * the failure this package refuses everywhere else.
744
+ */
745
+ readonly writableRootfsPaths?: readonly string[]
323
746
  }
324
747
 
325
748
  /**
@@ -480,6 +903,24 @@ export interface KubernetesBackendConfig {
480
903
  readonly warmPoolName?: string
481
904
  /** TCP port the in-pod guest agent listens on. Default 1024. */
482
905
  readonly agentPort?: number
906
+ /**
907
+ * Which of a sandbox's two addresses the transport dials.
908
+ *
909
+ * `'service'` (default) is the Sandbox's `status.serviceFQDN`, which
910
+ * outlives the pod and is re-resolved on every dial — and which ONLY the
911
+ * cluster's own DNS answers. A host running outside the cluster fails
912
+ * every call at name resolution, readiness included, so it reads as a
913
+ * sandbox that never came up.
914
+ *
915
+ * `'pod-ip'` dials the bound pod's IP, read from the same `GET` that
916
+ * reads its bind token. For a host outside the cluster with a route to
917
+ * the pod network. It needs that route and a `NetworkPolicy` admitting
918
+ * the host's address range on {@link agentPort}; the IP dies with its
919
+ * pod, which the backend covers by re-reading it on every resume and once
920
+ * after a connect failure. Nothing else changes: same bind token, same
921
+ * privilege probe, same egress verification.
922
+ */
923
+ readonly agentAddress?: KubernetesAgentAddressMode
483
924
  /** Delay between readiness polls. Default 50ms. */
484
925
  readonly readyPollIntervalMs?: number
485
926
  /** Total deadline from create to an addressed, Ready sandbox. Default 60000ms. */
@@ -495,10 +936,12 @@ export interface KubernetesBackendConfig {
495
936
  *
496
937
  * The handle renews its own `shutdownTime` every half-TTL for as long as
497
938
  * it is alive, so a run that outlives `claimTtlSeconds` keeps its pod.
498
- * A failed renewal is retried on the next tick, half a TTL before
499
- * anything expires; this callback is where the diagnostic goes, because
500
- * `@namzu/sandbox` owns no logger and reads none from module scope.
501
- * Setting it changes nothing about behaviour.
939
+ * A failed renewal is retried on a short capped backoff — starting at one
940
+ * second, not the next half-TTL tick — so a single API blip near a
941
+ * scheduled renewal gets several more chances before anything expires;
942
+ * this callback is where the diagnostic goes, because `@namzu/sandbox`
943
+ * owns no logger and reads none from module scope. Setting it changes
944
+ * nothing about behaviour.
502
945
  */
503
946
  readonly onLeaseRenewalError?: (error: unknown) => void
504
947
  /**
@@ -521,10 +964,89 @@ export interface KubernetesBackendConfig {
521
964
  * enforcement point is one object attached to the template and cannot be
522
965
  * rewritten per running sandbox. `static` and `resolver` — hostname
523
966
  * allowlists — throw a named error at construction unless `engine` is
524
- * `'cilium'`: core `NetworkPolicy` has no FQDN concept at all. See
967
+ * `'cilium'`: core `NetworkPolicy` has no FQDN concept at all.
968
+ *
969
+ * `policy` also takes two kinds that exist only here, because only a
970
+ * `NetworkPolicy` can express them: `{ kind: 'no-network' }` (nothing
971
+ * leaves the pod, the cluster resolver included — which `'deny-all'` never
972
+ * meant, since it allows DNS and a cluster resolver forwards outside
973
+ * names) and `{ kind: 'public-internet', exceptCidrs? }` (the internet,
974
+ * minus the private ranges, carrier-grade NAT, link-local and one cloud
975
+ * platform endpoint). `'deny-all'` and `'allow-all'` emit exactly the
976
+ * manifests they always have.
977
+ *
978
+ * **Setting this now checks the UNION.** Since every policy selecting a
979
+ * pod is unioned by the API server, the check reads the named object AND
980
+ * enumerates the namespace's policies, refusing when any of them lets out
981
+ * more than `policy` does. `verify: 'named-object-only'` restores the
982
+ * single-object check exactly. See
525
983
  * `docs/sdk/kubernetes-sandbox.md`'s egress section.
526
984
  */
527
985
  readonly egress?: KubernetesEgressConfig
986
+ /**
987
+ * Whether this backend proves, before creating a sandbox, that an applied
988
+ * policy actually closes {@link agentPort} on the pod it is about to hand
989
+ * back — and against which policy resources.
990
+ *
991
+ * **Unset means verify.** This is the one field here whose absent value is
992
+ * the strict one, because the deployment that needs the check is the one
993
+ * that would never have switched it on: the guest agent's own source calls
994
+ * the network rule in front of its port the boundary, and until this field
995
+ * existed nothing confirmed there was one.
996
+ *
997
+ * `{ engine: 'cilium' }` also enumerates that CNI's own policy CRD;
998
+ * `engine` otherwise defaults to `egress?.engine ?? 'core'`.
999
+ *
1000
+ * `'unverified'` reads no policy and issues no request. It is the
1001
+ * supported answer for a deployment whose boundary a namespaced Role
1002
+ * cannot see — a cluster-scoped policy, a service mesh, a cloud security
1003
+ * group — and it is a claim the deployment makes on purpose rather than a
1004
+ * default it inherits. See `docs/sdk/kubernetes-sandbox.md`'s ingress
1005
+ * section.
1006
+ */
1007
+ readonly ingress?: KubernetesIngressConfig
1008
+ /**
1009
+ * How long a single Kubernetes API request may take, end to end —
1010
+ * resolving the token, connecting, and reading the reply. Default
1011
+ * `30000`; minimum `1000`; there is no value that turns it off.
1012
+ *
1013
+ * The caller's `signal` is optional everywhere and several of this
1014
+ * backend's requests are SHARED flights that run under whichever caller
1015
+ * arrived first, so a signal-less `destroy()` against an API server that
1016
+ * accepted a request and never answered used to pin every later caller
1017
+ * joined to it. Expiry rejects with `KubernetesApiTimeoutError`, which
1018
+ * says nothing about whether the request was applied — the paths that
1019
+ * send one already cope with not knowing.
1020
+ */
1021
+ readonly apiRequestTimeoutMs?: number
1022
+ /**
1023
+ * Interval of the liveness heartbeat `openTerminal` and
1024
+ * `openTcpConnection` streams negotiate with the guest. Default `15000`;
1025
+ * `0` sends none, which is exactly how every release before this one
1026
+ * behaved.
1027
+ *
1028
+ * A quiet shell is healthy, so nothing replaced the read-idle timer the
1029
+ * transport clears once a stream is ready: a partition that delivered no
1030
+ * FIN and no RST left `exited`/`closed` unresolved on the host and the
1031
+ * shell's process group alive in the guest. Three missed intervals end
1032
+ * the stream on both sides. It is negotiated per stream — the guest
1033
+ * echoes the interval in its `ready` event and sends nothing new unless
1034
+ * it did — so an older guest image behaves exactly as it does today.
1035
+ */
1036
+ readonly streamHeartbeatMs?: number
1037
+ /**
1038
+ * Extra labels written onto every `SandboxClaim` this backend POSTs —
1039
+ * `metadata.labels` only, never the pod's own labels. Unset means no
1040
+ * labels beyond what the controller itself writes, and every claim body
1041
+ * is byte-for-byte what it was before this option existed.
1042
+ *
1043
+ * The intended use is a host-instance identity, so a restarted host can
1044
+ * find and {@link releaseKubernetesTaskSandboxes} a crashed predecessor's
1045
+ * claims well before `claimTtlSeconds` reaps them on its own — see
1046
+ * {@link readKubernetesTaskCapacity} for reading pool headroom
1047
+ * alongside it.
1048
+ */
1049
+ readonly claimLabels?: Record<string, string>
528
1050
  }
529
1051
 
530
1052
  /**
@@ -796,6 +1318,11 @@ function pickBackend(config: SandboxProviderConfig): SandboxBackend {
796
1318
  ...(backend.network !== undefined ? { network: backend.network } : {}),
797
1319
  ...(backend.allowInwardFor !== undefined ? { allowInwardFor: backend.allowInwardFor } : {}),
798
1320
  ...(backend.labels !== undefined ? { labels: backend.labels } : {}),
1321
+ ...(backend.cpuLimit !== undefined ? { cpuLimit: backend.cpuLimit } : {}),
1322
+ ...(backend.readOnlyRootfs !== undefined ? { readOnlyRootfs: backend.readOnlyRootfs } : {}),
1323
+ ...(backend.writableRootfsPaths !== undefined
1324
+ ? { writableRootfsPaths: backend.writableRootfsPaths }
1325
+ : {}),
799
1326
  })
800
1327
  }
801
1328
  if (backend.tier === 'container' && backend.runtime === 'runsc') {
@@ -816,6 +1343,11 @@ function pickBackend(config: SandboxProviderConfig): SandboxBackend {
816
1343
  ...(backend.network !== undefined ? { network: backend.network } : {}),
817
1344
  ...(backend.allowInwardFor !== undefined ? { allowInwardFor: backend.allowInwardFor } : {}),
818
1345
  ...(backend.labels !== undefined ? { labels: backend.labels } : {}),
1346
+ ...(backend.cpuLimit !== undefined ? { cpuLimit: backend.cpuLimit } : {}),
1347
+ ...(backend.readOnlyRootfs !== undefined ? { readOnlyRootfs: backend.readOnlyRootfs } : {}),
1348
+ ...(backend.writableRootfsPaths !== undefined
1349
+ ? { writableRootfsPaths: backend.writableRootfsPaths }
1350
+ : {}),
819
1351
  })
820
1352
  }
821
1353
  // `microvm:self-hosted` targeting the OWNED Azure Firecracker
@@ -871,6 +1403,7 @@ function kubernetesInternalConfig(
871
1403
  sandboxTemplateName: backend.sandboxTemplateName,
872
1404
  ...(backend.warmPoolName !== undefined ? { warmPoolName: backend.warmPoolName } : {}),
873
1405
  ...(backend.agentPort !== undefined ? { agentPort: backend.agentPort } : {}),
1406
+ ...(backend.agentAddress !== undefined ? { agentAddress: backend.agentAddress } : {}),
874
1407
  ...(backend.readyPollIntervalMs !== undefined
875
1408
  ? { readyPollIntervalMs: backend.readyPollIntervalMs }
876
1409
  : {}),
@@ -883,6 +1416,14 @@ function kubernetesInternalConfig(
883
1416
  ? { runtimeClassName: backend.runtimeClassName }
884
1417
  : {}),
885
1418
  ...(backend.egress !== undefined ? { egress: backend.egress } : {}),
1419
+ ...(backend.ingress !== undefined ? { ingress: backend.ingress } : {}),
1420
+ ...(backend.apiRequestTimeoutMs !== undefined
1421
+ ? { apiRequestTimeoutMs: backend.apiRequestTimeoutMs }
1422
+ : {}),
1423
+ ...(backend.streamHeartbeatMs !== undefined
1424
+ ? { streamHeartbeatMs: backend.streamHeartbeatMs }
1425
+ : {}),
1426
+ ...(backend.claimLabels !== undefined ? { claimLabels: backend.claimLabels } : {}),
886
1427
  }
887
1428
  }
888
1429
 
@@ -913,6 +1454,114 @@ export async function createKubernetesWorkspace(
913
1454
  return await buildKubernetesWorkspace(kubernetesInternalConfig(config), options)
914
1455
  }
915
1456
 
1457
+ /**
1458
+ * Every workspace this backend owns in the namespace, read off the objects
1459
+ * and waking none of them.
1460
+ *
1461
+ * The inventory {@link createKubernetesWorkspace} cannot give you: it adopts
1462
+ * AND resumes, so taking stock through it would start a pod for every
1463
+ * suspended workspace it looked at. This issues one GET of the sandboxes
1464
+ * collection and sends no PATCH and no DELETE — a suspended workspace is
1465
+ * still suspended afterwards.
1466
+ *
1467
+ * Needs `list` on `sandboxes` in the namespace, which is the one RBAC verb
1468
+ * the task path did not already require.
1469
+ */
1470
+ export async function listKubernetesWorkspaces(
1471
+ config: KubernetesBackendConfig,
1472
+ options?: KubernetesWorkspaceTransitionOptions,
1473
+ ): Promise<readonly KubernetesWorkspaceSummary[]> {
1474
+ return await listWorkspacesOnCluster(kubernetesInternalConfig(config), options)
1475
+ }
1476
+
1477
+ /**
1478
+ * Delete a workspace by id — the Sandbox, and with it the Pod, the Service
1479
+ * and the PVC — without adopting or resuming it first.
1480
+ *
1481
+ * Exactly what `destroy({ deleteDisk: true })` does to the cluster, with the
1482
+ * same guarantees: an object already gone counts as deleted, and a DELETE
1483
+ * that fails rejects and stays retryable. The files are gone and nothing
1484
+ * brings them back.
1485
+ *
1486
+ * It is the retention verb. Removing a month-old suspended workspace through
1487
+ * a handle meant starting its pod and probing it purely to tell it to go
1488
+ * away; the name is deterministic, so the object never needed opening.
1489
+ */
1490
+ export async function deleteKubernetesWorkspace(
1491
+ config: KubernetesBackendConfig,
1492
+ workspaceId: string,
1493
+ options?: KubernetesWorkspaceTransitionOptions,
1494
+ ): Promise<void> {
1495
+ await deleteWorkspaceOnCluster(kubernetesInternalConfig(config), workspaceId, options)
1496
+ }
1497
+
1498
+ /**
1499
+ * Suspend a workspace by id: send the `operatingMode: Suspended` patch and
1500
+ * wait for the pod to actually stop, without adopting the workspace.
1501
+ *
1502
+ * Resolves only once the pod is gone or in a terminal phase — a suspend is a
1503
+ * promise that the disk is quiesced, and the patch being accepted says only
1504
+ * that the controller has been asked. A pod that outlives `readyTimeoutMs`
1505
+ * rejects with `KubernetesWorkspaceSuspendTimeoutError`, leaving the object
1506
+ * as the patch left it.
1507
+ *
1508
+ * A handle another process is holding is not told. It finds out on its next
1509
+ * call — which fails at the transport and is re-read into a
1510
+ * `KubernetesWorkspaceSuspendedError` — or when that process calls
1511
+ * `refresh()`.
1512
+ *
1513
+ * It takes the suspend options shape and REFUSES `quiesce` rather than
1514
+ * accepting the flag and dropping it: this verb never dials the agent, so
1515
+ * there is no connection here on which anything could be stopped. Quiescing
1516
+ * needs a handle — `createKubernetesWorkspace()`, then
1517
+ * `suspend({ quiesce: true })`.
1518
+ */
1519
+ export async function suspendKubernetesWorkspace(
1520
+ config: KubernetesBackendConfig,
1521
+ workspaceId: string,
1522
+ options?: KubernetesWorkspaceSuspendOptions,
1523
+ ): Promise<void> {
1524
+ await suspendWorkspaceOnCluster(kubernetesInternalConfig(config), workspaceId, options)
1525
+ }
1526
+
1527
+ /**
1528
+ * Recover a crashed host's task-path claims: LIST every `SandboxClaim`
1529
+ * carrying `options.labelSelector`, `DELETE` each, and report what was
1530
+ * removed.
1531
+ *
1532
+ * Deletes claims only — the controller's own ownerReferences take the bound
1533
+ * Sandbox, its Pod and its Service down behind each one; nothing here reads
1534
+ * or touches those objects directly. `labelSelector` is REQUIRED and refused
1535
+ * before any request goes out if it is empty: falling back to matching every
1536
+ * claim would delete a live fleet's work.
1537
+ *
1538
+ * Pairs with `config.claimLabels`: a host stamps its own identity onto every
1539
+ * claim it creates, and a restarted instance passes that same selector here
1540
+ * to reclaim its predecessor's warm-pool capacity well before
1541
+ * `claimTtlSeconds` would reap it on its own.
1542
+ */
1543
+ export async function releaseKubernetesTaskSandboxes(
1544
+ config: KubernetesBackendConfig,
1545
+ options: KubernetesReleaseTaskSandboxesOptions,
1546
+ ): Promise<{ readonly deleted: number; readonly names: readonly string[] }> {
1547
+ return await releaseTaskSandboxesOnCluster(kubernetesInternalConfig(config), options)
1548
+ }
1549
+
1550
+ /**
1551
+ * Read task-pool headroom before admitting more work: three GETs
1552
+ * (`SandboxWarmPool`, the claims collection, the pods collection), no
1553
+ * writes.
1554
+ *
1555
+ * Requires `config.warmPoolName` — a pool-less backend (every create is a
1556
+ * direct Sandbox) has no `SandboxWarmPool` to report on.
1557
+ */
1558
+ export async function readKubernetesTaskCapacity(
1559
+ config: KubernetesBackendConfig,
1560
+ options?: KubernetesReadTaskCapacityOptions,
1561
+ ): Promise<KubernetesTaskCapacity> {
1562
+ return await readTaskCapacityOnCluster(kubernetesInternalConfig(config), options)
1563
+ }
1564
+
916
1565
  /**
917
1566
  * Human-readable backend label for error messages. Returns the
918
1567
  * tier plus the concrete service / runtime when present, e.g.