@namzu/sandbox 13.0.0 → 15.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (104) hide show
  1. package/CHANGELOG.md +1147 -0
  2. package/README.md +447 -0
  3. package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
  4. package/dist/backends/aci-standby-pool/index.js +13 -1
  5. package/dist/backends/aci-standby-pool/index.js.map +1 -1
  6. package/dist/backends/docker/index.d.ts.map +1 -1
  7. package/dist/backends/docker/index.js +19 -1
  8. package/dist/backends/docker/index.js.map +1 -1
  9. package/dist/backends/firecracker/index.d.ts.map +1 -1
  10. package/dist/backends/firecracker/index.js +12 -2
  11. package/dist/backends/firecracker/index.js.map +1 -1
  12. package/dist/backends/firecracker/protocol.d.ts +481 -8
  13. package/dist/backends/firecracker/protocol.d.ts.map +1 -1
  14. package/dist/backends/firecracker/protocol.js +136 -0
  15. package/dist/backends/firecracker/protocol.js.map +1 -1
  16. package/dist/backends/firecracker/transport.d.ts +642 -14
  17. package/dist/backends/firecracker/transport.d.ts.map +1 -1
  18. package/dist/backends/firecracker/transport.js +1307 -34
  19. package/dist/backends/firecracker/transport.js.map +1 -1
  20. package/dist/backends/kubernetes/egress-policy.d.ts +1296 -0
  21. package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -0
  22. package/dist/backends/kubernetes/egress-policy.js +2458 -0
  23. package/dist/backends/kubernetes/egress-policy.js.map +1 -0
  24. package/dist/backends/kubernetes/identity.d.ts +193 -0
  25. package/dist/backends/kubernetes/identity.d.ts.map +1 -0
  26. package/dist/backends/kubernetes/identity.js +147 -0
  27. package/dist/backends/kubernetes/identity.js.map +1 -0
  28. package/dist/backends/kubernetes/index.d.ts +1019 -0
  29. package/dist/backends/kubernetes/index.d.ts.map +1 -0
  30. package/dist/backends/kubernetes/index.js +1756 -0
  31. package/dist/backends/kubernetes/index.js.map +1 -0
  32. package/dist/backends/kubernetes/ingress-policy.d.ts +375 -0
  33. package/dist/backends/kubernetes/ingress-policy.d.ts.map +1 -0
  34. package/dist/backends/kubernetes/ingress-policy.js +1050 -0
  35. package/dist/backends/kubernetes/ingress-policy.js.map +1 -0
  36. package/dist/backends/kubernetes/k8s-client.d.ts +334 -0
  37. package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -0
  38. package/dist/backends/kubernetes/k8s-client.js +553 -0
  39. package/dist/backends/kubernetes/k8s-client.js.map +1 -0
  40. package/dist/backends/kubernetes/lease.d.ts +145 -0
  41. package/dist/backends/kubernetes/lease.d.ts.map +1 -0
  42. package/dist/backends/kubernetes/lease.js +201 -0
  43. package/dist/backends/kubernetes/lease.js.map +1 -0
  44. package/dist/backends/kubernetes/objects.d.ts +702 -0
  45. package/dist/backends/kubernetes/objects.d.ts.map +1 -0
  46. package/dist/backends/kubernetes/objects.js +518 -0
  47. package/dist/backends/kubernetes/objects.js.map +1 -0
  48. package/dist/backends/kubernetes/per-sandbox-policy.d.ts +219 -0
  49. package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -0
  50. package/dist/backends/kubernetes/per-sandbox-policy.js +407 -0
  51. package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -0
  52. package/dist/backends/kubernetes/privilege-probe.d.ts +136 -0
  53. package/dist/backends/kubernetes/privilege-probe.d.ts.map +1 -0
  54. package/dist/backends/kubernetes/privilege-probe.js +185 -0
  55. package/dist/backends/kubernetes/privilege-probe.js.map +1 -0
  56. package/dist/backends/kubernetes/rbac.d.ts +153 -0
  57. package/dist/backends/kubernetes/rbac.d.ts.map +1 -0
  58. package/dist/backends/kubernetes/rbac.js +177 -0
  59. package/dist/backends/kubernetes/rbac.js.map +1 -0
  60. package/dist/backends/kubernetes/sandbox.d.ts +190 -0
  61. package/dist/backends/kubernetes/sandbox.d.ts.map +1 -0
  62. package/dist/backends/kubernetes/sandbox.js +433 -0
  63. package/dist/backends/kubernetes/sandbox.js.map +1 -0
  64. package/dist/backends/kubernetes/transport.d.ts +1048 -0
  65. package/dist/backends/kubernetes/transport.d.ts.map +1 -0
  66. package/dist/backends/kubernetes/transport.js +2093 -0
  67. package/dist/backends/kubernetes/transport.js.map +1 -0
  68. package/dist/backends/kubernetes/workspace.d.ts +1512 -0
  69. package/dist/backends/kubernetes/workspace.d.ts.map +1 -0
  70. package/dist/backends/kubernetes/workspace.js +3703 -0
  71. package/dist/backends/kubernetes/workspace.js.map +1 -0
  72. package/dist/backends/remote-execution-controller.d.ts +14 -0
  73. package/dist/backends/remote-execution-controller.d.ts.map +1 -1
  74. package/dist/backends/remote-execution-controller.js.map +1 -1
  75. package/dist/index.d.ts +350 -2
  76. package/dist/index.d.ts.map +1 -1
  77. package/dist/index.js +344 -34
  78. package/dist/index.js.map +1 -1
  79. package/dist/testing/sandbox-conformance.d.ts +227 -0
  80. package/dist/testing/sandbox-conformance.d.ts.map +1 -0
  81. package/dist/testing/sandbox-conformance.js +896 -0
  82. package/dist/testing/sandbox-conformance.js.map +1 -0
  83. package/package.json +5 -4
  84. package/src/backends/aci-standby-pool/index.ts +16 -1
  85. package/src/backends/docker/index.ts +22 -1
  86. package/src/backends/firecracker/index.ts +14 -2
  87. package/src/backends/firecracker/protocol.ts +541 -6
  88. package/src/backends/firecracker/transport.ts +1687 -64
  89. package/src/backends/kubernetes/egress-policy.ts +3448 -0
  90. package/src/backends/kubernetes/identity.ts +261 -0
  91. package/src/backends/kubernetes/index.ts +2670 -0
  92. package/src/backends/kubernetes/ingress-policy.ts +1344 -0
  93. package/src/backends/kubernetes/k8s-client.ts +742 -0
  94. package/src/backends/kubernetes/lease.ts +254 -0
  95. package/src/backends/kubernetes/objects.ts +983 -0
  96. package/src/backends/kubernetes/per-sandbox-policy.ts +542 -0
  97. package/src/backends/kubernetes/privilege-probe.ts +261 -0
  98. package/src/backends/kubernetes/rbac.ts +192 -0
  99. package/src/backends/kubernetes/sandbox.ts +593 -0
  100. package/src/backends/kubernetes/transport.ts +2895 -0
  101. package/src/backends/kubernetes/workspace.ts +5640 -0
  102. package/src/backends/remote-execution-controller.ts +14 -0
  103. package/src/index.ts +838 -35
  104. package/src/testing/sandbox-conformance.ts +1202 -0
@@ -0,0 +1,593 @@
1
+ /**
2
+ * The {@link Sandbox} a kubernetes acquire hands back: the SDK contract,
3
+ * served over the guest agent's TCP transport, with the lease that keeps the
4
+ * cluster from deleting the pod out from under a long run.
5
+ *
6
+ * Split out of `index.ts` because that file is about the CONTROL plane —
7
+ * claim, poll, address, release — and this one is about the DATA plane, and
8
+ * the two are read for different reasons.
9
+ *
10
+ * ## What it implements, and what it deliberately does not
11
+ *
12
+ * Implemented: `exec` (through the shared {@link RemoteExecutionController},
13
+ * so an `AbortSignal` terminates the guest process rather than abandoning
14
+ * the wait), `writeFile`, `readFile`, `listFiles`, `walkFiles`,
15
+ * `openTerminal`, `openTcpConnection`, `destroy`.
16
+ *
17
+ * Conditionally present, on the same contract read the other way:
18
+ *
19
+ * - `setNetworkPolicy` — present on a TASK handle exactly when the backend
20
+ * was configured with `egress.perSandbox`, which is what gives this host
21
+ * an object to write (one `CiliumNetworkPolicy` per sandbox, owned by the
22
+ * claim), an admission fence bounding what it may write, and the RBAC to
23
+ * write it. Without that configuration there is still no per-running-pod
24
+ * knob to turn, so the method is ABSENT rather than present and throwing —
25
+ * the same rule, applied to a capability that now sometimes exists.
26
+ * Presence follows configuration alone, never a probe of the cluster. See
27
+ * `per-sandbox-policy.ts`.
28
+ *
29
+ * A WORKSPACE handle never carries it: `KubernetesWorkspace`'s create path
30
+ * builds its handle through `buildKubernetesSandbox` without this option,
31
+ * because it composes no per-sandbox pod label and tracks no owner uid for
32
+ * one. That is a property of the PATH and not of the configuration, so it
33
+ * is not answered by omission: `createKubernetesWorkspace` refuses a
34
+ * config carrying `egress.perSandbox` outright
35
+ * (`KubernetesWorkspacePerSandboxEgressConfigError`, before any request)
36
+ * rather than letting the option declare a capability the path never
37
+ * serves — see `workspace.ts`.
38
+ *
39
+ * Absent on purpose, because the SDK's contract says a backend that cannot
40
+ * honour an optional method must omit it rather than accept and ignore:
41
+ *
42
+ * - `spawnDetached` — the guest agent has no op that starts a process and
43
+ * returns it running. A host asking for background jobs must be told no.
44
+ *
45
+ * ## Terminals are owned
46
+ *
47
+ * `openTerminal` is only a compliant implementation if `destroy()` kills and
48
+ * awaits every terminal it returned, so open terminals are tracked and
49
+ * reaped before the object is released — the same thing the Firecracker
50
+ * backend does, for the same contract.
51
+ *
52
+ * ## Two terminal states, one `SandboxStatus`
53
+ *
54
+ * `SandboxStatus` has exactly four members and this change does not widen
55
+ * the SDK's union, so both ways a sandbox ends report `'destroyed'`. They
56
+ * are told apart by the error a later call throws:
57
+ * {@link KubernetesSandboxDestroyedError} (this host released it) and
58
+ * {@link KubernetesSandboxGoneError} (the cluster deleted it — the lease
59
+ * renewal found the object already gone).
60
+ */
61
+
62
+ import type {
63
+ OpenTerminalOptions,
64
+ Sandbox,
65
+ SandboxDestroyOptions,
66
+ SandboxEnvironment,
67
+ SandboxExecOptions,
68
+ SandboxExecResult,
69
+ SandboxFileEntry,
70
+ SandboxId,
71
+ SandboxNetworkPolicy,
72
+ SandboxReadFileOptions,
73
+ SandboxStatus,
74
+ SandboxTcpConnectOptions,
75
+ SandboxTcpConnection,
76
+ SandboxWalkFilesOptions,
77
+ TerminalSession,
78
+ } from '@namzu/sdk'
79
+ import { walkFilesViaExec } from '@namzu/sdk'
80
+
81
+ import { OperationDeadline } from '../readiness.js'
82
+ import {
83
+ RemoteCancellationUnknownError,
84
+ type SandboxRetirementObservation,
85
+ } from '../remote-execution-controller.js'
86
+ import { KubernetesLeaseRenewal } from './lease.js'
87
+ import type { KubernetesAgentTransport } from './transport.js'
88
+
89
+ /**
90
+ * How long a retirement triggered by an unconfirmed cancellation may spend
91
+ * deleting the object. Separate from every other clock on the path: the
92
+ * caller's own deadline has usually already expired by the time this runs,
93
+ * and reusing it would mean skipping teardown exactly when a command of
94
+ * unknown state is still out there.
95
+ */
96
+ const RETIREMENT_TIMEOUT_MS = 15_000
97
+
98
+ /** Thrown by any operation on a sandbox this host already destroyed. */
99
+ export class KubernetesSandboxDestroyedError extends Error {
100
+ override readonly name = 'KubernetesSandboxDestroyedError'
101
+
102
+ constructor(
103
+ readonly operation: string,
104
+ readonly sandboxName: string,
105
+ ) {
106
+ super(
107
+ `kubernetes sandbox ${sandboxName} has been destroyed; ${operation}() cannot be admitted. Acquire a new sandbox — a destroyed one's pod, Service and claim are deleted and its agent address no longer resolves.`,
108
+ )
109
+ }
110
+ }
111
+
112
+ /**
113
+ * Thrown by any operation on a sandbox the CLUSTER removed while this
114
+ * handle still held it — the lease renewal PATCH came back 404/410. Distinct
115
+ * from {@link KubernetesSandboxDestroyedError} because nothing this host did
116
+ * caused it: the object expired, an operator deleted it, or the controller
117
+ * reaped it, and the actionable advice is different.
118
+ */
119
+ export class KubernetesSandboxGoneError extends Error {
120
+ override readonly name = 'KubernetesSandboxGoneError'
121
+
122
+ constructor(
123
+ readonly operation: string,
124
+ readonly sandboxName: string,
125
+ ) {
126
+ super(
127
+ `kubernetes sandbox ${sandboxName} no longer exists on the cluster; ${operation}() cannot be admitted. Its lease renewal found the object already deleted — it expired (spec.lifecycle.shutdownTime), an operator deleted it, or the controller reaped it. Nothing this handle can do brings it back; acquire a new sandbox.`,
128
+ )
129
+ }
130
+ }
131
+
132
+ interface KubernetesSandboxBaseOptions {
133
+ /** The cluster's own name for the bound sandbox — also the sandbox id. */
134
+ readonly name: string
135
+ readonly rootDir: string
136
+ readonly transport: KubernetesAgentTransport
137
+ /** DELETE the object this backend created. Already-gone counts as done. */
138
+ readonly release: (signal?: AbortSignal) => Promise<void>
139
+ /**
140
+ * Decide what an execution whose cancellation could not be CONFIRMED
141
+ * does to this sandbox — called instead of retiring it, and answering
142
+ * the {@link SandboxRetirementObservation} that goes onto the error the
143
+ * caller is about to receive.
144
+ *
145
+ * Unset (the task path, and the Firecracker tier through its own
146
+ * transport) keeps the shared controller's rule verbatim: a command of
147
+ * unknown state is still in that pod, the pod stops being reusable, and
148
+ * the handle retires it through {@link release}. That is right for a
149
+ * disposable object whose disk is scratch.
150
+ *
151
+ * It is wrong for an object that is not disposable. On a workspace
152
+ * `release` is an `operatingMode: Suspended` patch, which makes the
153
+ * controller delete the pod — so eight seconds of network loss under one
154
+ * `exec()` would take every other holder's terminals, dev servers and
155
+ * running commands with it, and no host-side lock can prevent it because
156
+ * no caller issued it. The workspace passes a hook that keeps the pod,
157
+ * diagnoses the agent and says `accepted: false` with a `reason` rather
158
+ * than letting a decision that large be made from inside a failing call.
159
+ *
160
+ * It must not reject; one that does is reported as an unaccepted
161
+ * retirement carrying its own error, so a broken hook cannot replace the
162
+ * error the caller asked about.
163
+ */
164
+ readonly onUnconfirmedCancellation?: (
165
+ error: RemoteCancellationUnknownError,
166
+ ) => Promise<SandboxRetirementObservation>
167
+ /**
168
+ * Narrow this sandbox's egress while it runs — present on the handle
169
+ * EXACTLY when this is passed, and passed by the TASK acquire exactly when
170
+ * `config.egress.perSandbox` is configured.
171
+ *
172
+ * That conditional presence is what the SDK's omit-or-throw contract
173
+ * licenses and what makes it honest here: without the configuration there
174
+ * is no policy object to write, no admission fence bounding what this
175
+ * host may write, and no RBAC grant to write it with, so the method is
176
+ * ABSENT rather than present and throwing. With it, `per-sandbox-policy.ts`
177
+ * writes one `CiliumNetworkPolicy` per sandbox, owned by the object the
178
+ * acquire created.
179
+ *
180
+ * The workspace passes NO such option whatever the config says — its
181
+ * create path composes no per-sandbox pod label and tracks no owner uid
182
+ * for one — and `createKubernetesWorkspace` refuses
183
+ * `config.egress.perSandbox` rather than silently omitting the method the
184
+ * option asks for.
185
+ *
186
+ * Presence depends only on configuration — never on a runtime probe of
187
+ * the cluster — so a caller's capability detection cannot come out
188
+ * differently depending on when it asked.
189
+ */
190
+ readonly setNetworkPolicy?: (policy: SandboxNetworkPolicy) => Promise<void>
191
+ }
192
+
193
+ /**
194
+ * The lease half of the options: a way to move the expiry, and the expiry it
195
+ * is moving. Required TOGETHER, because `renew` without `ttlSeconds` is a
196
+ * renewal loop with nothing to stamp — it would re-stamp `now + 0`, an
197
+ * expiry already in the past, and hand the object straight to the
198
+ * controller's reaper while reporting every tick a success. A pair is the
199
+ * only shape that cannot be half-configured.
200
+ */
201
+ interface KubernetesSandboxLeaseOptions {
202
+ /** PATCH the object's `shutdownTime` forward. See `lease.ts`. */
203
+ readonly renew: (shutdownTime: string, signal?: AbortSignal) => Promise<void>
204
+ /** The TTL acquire stamped; each renewal re-stamps exactly this much. */
205
+ readonly ttlSeconds: number
206
+ readonly onRenewalError?: (error: unknown) => void
207
+ /** Test seam: the renewal loop's base interval. Default: half the TTL. */
208
+ readonly leaseIntervalMs?: number
209
+ }
210
+
211
+ /**
212
+ * The other arm: an object that carries no expiry, so this handle runs no
213
+ * renewal loop at all — the persistent workspace (`workspace.ts`), which is
214
+ * explicitly managed and must outlive a host that stopped renewing. A no-op
215
+ * `renew` would be the wrong way to say that: it would leave a timer ticking
216
+ * forever to do nothing. The lease fields are typed `undefined` rather than
217
+ * omitted so that passing one of them here is a type error and not an
218
+ * excess-property check a spread would slip past.
219
+ */
220
+ interface KubernetesSandboxUnleasedOptions {
221
+ readonly renew?: undefined
222
+ readonly ttlSeconds?: undefined
223
+ readonly onRenewalError?: undefined
224
+ readonly leaseIntervalMs?: undefined
225
+ }
226
+
227
+ export type KubernetesSandboxOptions = KubernetesSandboxBaseOptions &
228
+ (KubernetesSandboxLeaseOptions | KubernetesSandboxUnleasedOptions)
229
+
230
+ function detectEnvironment(): SandboxEnvironment {
231
+ // The guest runs Linux; the enum describes the host-facing shape of the
232
+ // worker, not the isolation technology under it. Firecracker's guest
233
+ // reports the same for the same reason.
234
+ return 'linux-namespace'
235
+ }
236
+
237
+ /**
238
+ * What this backend hands back: the SDK contract, with the optional members
239
+ * it DOES implement narrowed to present, so a caller that composes one —
240
+ * `workspace.ts` wraps this handle — does not have to re-check for a method
241
+ * this file always defines.
242
+ */
243
+ export type KubernetesSandboxHandle = Sandbox &
244
+ Required<Pick<Sandbox, 'openTerminal' | 'openTcpConnection' | 'walkFiles' | 'readFileStream'>>
245
+
246
+ /**
247
+ * Build the handle. It does NOT run the acquire-time privilege probe — that
248
+ * is `create()`'s job in `index.ts`, so that a refusal can destroy this
249
+ * object before any caller has a reference to it, and so this function stays
250
+ * usable by the workspace path that runs its own probe.
251
+ */
252
+ export function buildKubernetesSandbox(options: KubernetesSandboxOptions): KubernetesSandboxHandle {
253
+ // The cluster owns this name. Preserving it verbatim as the sandbox id —
254
+ // as the Firecracker backend preserves its orchestrator's — means a log
255
+ // line carrying an id is also a `kubectl get sandbox` argument.
256
+ const id = options.name as SandboxId
257
+ const transport = options.transport
258
+ // Captured as a const so the conditional member below narrows: the method
259
+ // is on the handle if and only if this is defined, decided once, here.
260
+ const setNetworkPolicy = options.setNetworkPolicy
261
+
262
+ type Lifecycle = 'active' | 'retiring' | 'destroyed' | 'gone'
263
+ let lifecycle: Lifecycle = 'active'
264
+ let activeExecutions = 0
265
+ let teardownPromise: Promise<void> | undefined
266
+ let teardownComplete = false
267
+ let retirementPromise: Promise<SandboxRetirementObservation> | undefined
268
+ const terminals = new Set<TerminalSession>()
269
+
270
+ // No `renew` ⇒ no expiry to move ⇒ no loop. `stop()` on the undefined
271
+ // case is the caller's problem to not have, which is why every use below
272
+ // goes through `lease?.stop()`.
273
+ const renew = options.renew
274
+ const ttlSeconds = options.ttlSeconds
275
+ // The type above already pairs the two. This is the runtime half of the
276
+ // same rule, for a caller that reached here through a cast or from
277
+ // JavaScript: a lease stamping `now + 0` expires the moment it is written,
278
+ // and every tick would report success while the controller deleted the
279
+ // object underneath it.
280
+ if (renew !== undefined && (typeof ttlSeconds !== 'number' || ttlSeconds <= 0)) {
281
+ throw new Error(
282
+ `kubernetes: sandbox ${options.name} was given a lease renewal with ttlSeconds ${String(ttlSeconds)}. A renewal re-stamps shutdownTime as now + ttlSeconds, so a zero or absent TTL stamps an expiry that has already passed and the object is reaped while the loop reports every tick a success. Pass renew and a positive ttlSeconds together, or neither — an object with no expiry (a persistent workspace) runs no renewal loop.`,
283
+ )
284
+ }
285
+ // `ttlSeconds === undefined` is unreachable once `renew` is defined — the
286
+ // throw above saw to that — and is written out anyway because it is what
287
+ // narrows the field to a number for the constructor below.
288
+ const lease =
289
+ renew === undefined || ttlSeconds === undefined
290
+ ? undefined
291
+ : new KubernetesLeaseRenewal({
292
+ ttlSeconds,
293
+ renew,
294
+ onGone: () => {
295
+ // The object is gone; the pod behind the address went with it.
296
+ // Refuse every later call by name rather than let it dial into a
297
+ // connect timeout with nothing to explain it.
298
+ if (lifecycle === 'active') lifecycle = 'gone'
299
+ },
300
+ ...(options.onRenewalError !== undefined
301
+ ? { onRenewalError: options.onRenewalError }
302
+ : {}),
303
+ ...(options.leaseIntervalMs !== undefined ? { intervalMs: options.leaseIntervalMs } : {}),
304
+ })
305
+ lease?.start()
306
+
307
+ const assertAdmissible = (operation: string): void => {
308
+ if (lifecycle === 'active') return
309
+ if (lifecycle === 'gone') throw new KubernetesSandboxGoneError(operation, options.name)
310
+ throw new KubernetesSandboxDestroyedError(operation, options.name)
311
+ }
312
+
313
+ const teardown = (signal?: AbortSignal): Promise<void> => {
314
+ if (lifecycle === 'active') lifecycle = 'retiring'
315
+ lease?.stop()
316
+ if (teardownComplete) return Promise.resolve()
317
+ if (teardownPromise) return teardownPromise
318
+ const shared = options.release(signal).then(
319
+ () => {
320
+ teardownComplete = true
321
+ lifecycle = 'destroyed'
322
+ },
323
+ (error: unknown) => {
324
+ // A failed teardown must stay retryable; keeping the rejected
325
+ // promise would answer every later destroy() with the same
326
+ // stale failure.
327
+ if (teardownPromise === shared) teardownPromise = undefined
328
+ throw error
329
+ },
330
+ )
331
+ teardownPromise = shared
332
+ return shared
333
+ }
334
+
335
+ /**
336
+ * A command whose cancellation the guest could not confirm may still be
337
+ * running in that pod, so the pod stops being reusable. Retire it and
338
+ * report whether the retirement landed, on the error the caller is about
339
+ * to receive.
340
+ */
341
+ const retire = (): Promise<SandboxRetirementObservation> => {
342
+ if (lifecycle === 'active') lifecycle = 'retiring'
343
+ retirementPromise ??= new OperationDeadline(
344
+ RETIREMENT_TIMEOUT_MS,
345
+ `kubernetes sandbox ${options.name} retirement`,
346
+ )
347
+ .run(async (signal) => await teardown(signal))
348
+ .then(() => ({ accepted: true as const }))
349
+ .catch((error: unknown) => ({
350
+ accepted: false as const,
351
+ error: error instanceof Error ? error : new Error(String(error)),
352
+ }))
353
+ return retirementPromise
354
+ }
355
+
356
+ /**
357
+ * What an unconfirmed cancellation does to THIS sandbox: retire it, or
358
+ * whatever the owner's hook decided instead — see
359
+ * {@link KubernetesSandboxBaseOptions.onUnconfirmedCancellation}.
360
+ */
361
+ const observeUnconfirmedCancellation = async (
362
+ error: RemoteCancellationUnknownError,
363
+ ): Promise<SandboxRetirementObservation> => {
364
+ const decide = options.onUnconfirmedCancellation
365
+ if (decide === undefined) return await retire()
366
+ try {
367
+ return await decide(error)
368
+ } catch (hookError: unknown) {
369
+ // The caller is already receiving `error`; a hook that threw must
370
+ // not replace it, and must not be reported as a teardown that was
371
+ // attempted either.
372
+ return {
373
+ accepted: false,
374
+ error: hookError instanceof Error ? hookError : new Error(String(hookError)),
375
+ }
376
+ }
377
+ }
378
+
379
+ const runExecution = async <T>(operation: string, run: () => Promise<T>): Promise<T> => {
380
+ assertAdmissible(operation)
381
+ activeExecutions += 1
382
+ try {
383
+ return await run()
384
+ } catch (error) {
385
+ if (error instanceof RemoteCancellationUnknownError) {
386
+ error.retirement = await observeUnconfirmedCancellation(error)
387
+ }
388
+ throw error
389
+ } finally {
390
+ activeExecutions = Math.max(0, activeExecutions - 1)
391
+ }
392
+ }
393
+
394
+ return {
395
+ id,
396
+ get status(): SandboxStatus {
397
+ // Four members, and no new one: a cluster-side disappearance and a
398
+ // host-side destroy both read as 'destroyed' here and are told
399
+ // apart by the error a later call throws.
400
+ if (lifecycle !== 'active') return 'destroyed'
401
+ return activeExecutions > 0 ? 'busy' : 'ready'
402
+ },
403
+ rootDir: options.rootDir,
404
+ environment: detectEnvironment(),
405
+
406
+ async exec(
407
+ command: string,
408
+ argv?: string[],
409
+ opts?: SandboxExecOptions,
410
+ ): Promise<SandboxExecResult> {
411
+ return await runExecution('exec', async () => await transport.exec(command, argv, opts))
412
+ },
413
+
414
+ /**
415
+ * Every `tcp` request dials a fresh connection, so its envelope is
416
+ * also that connection's first, not-yet-authenticated frame and is
417
+ * bounded by the guest's pre-auth frame ceiling (8 MiB by default).
418
+ * A body above it is no longer a refusal: the transport splits it
419
+ * into parts that each fit, writes them to a temporary sibling of
420
+ * the target and finishes with an atomic rename, so this method
421
+ * takes a body of any size the transport's `maxWriteFileBytes`
422
+ * admits (1 GiB by default). The named refusals that remain are
423
+ * passed through unwrapped so a caller can catch them BY CLASS:
424
+ * `AgentWriteFileTooLargeError` for a body above that bound, and
425
+ * `AgentPreauthFrameTooLargeError` for an oversized body against a
426
+ * guest too old to advertise the part protocol.
427
+ */
428
+ async writeFile(path: string, content: string | Buffer): Promise<void> {
429
+ assertAdmissible('writeFile')
430
+ const buf = Buffer.isBuffer(content) ? content : Buffer.from(content, 'utf8')
431
+ await transport.writeFile(path, buf)
432
+ },
433
+
434
+ /**
435
+ * A whole-file read is served by the guest's streamed op when the
436
+ * guest advertises it, so neither side holds the file's base64 form
437
+ * or its JSON envelope in one piece and a file of any size this
438
+ * workspace's disk holds can be read. `offset`/`length` ask for one
439
+ * slice instead; a guest too old to honour them is refused with
440
+ * `AgentReadFileStreamUnsupportedError` rather than answering with
441
+ * the whole file.
442
+ */
443
+ async readFile(path: string, readOptions?: SandboxReadFileOptions): Promise<Buffer> {
444
+ assertAdmissible('readFile')
445
+ return await transport.readFile(path, readOptions)
446
+ },
447
+
448
+ /**
449
+ * Chunks, in order, with nothing whole at either end — what a host
450
+ * draining a large output file before `destroy()` needs. The
451
+ * admissibility check runs at the call, not per chunk: a workspace
452
+ * suspended mid-stream takes its pod's connection with it, which is
453
+ * what ends the iteration.
454
+ */
455
+ readFileStream(path: string, readOptions?: SandboxReadFileOptions): AsyncIterable<Buffer> {
456
+ assertAdmissible('readFileStream')
457
+ return transport.readFileStream(path, readOptions)
458
+ },
459
+
460
+ // Present only when the backend was configured for per-sandbox egress
461
+ // — see {@link KubernetesSandboxBaseOptions.setNetworkPolicy}. The
462
+ // admissibility gate is this file's, not the writer's, so a destroyed
463
+ // or cluster-removed sandbox refuses BY NAME here, exactly as every
464
+ // other method does, rather than failing at the API server one round
465
+ // trip later.
466
+ ...(setNetworkPolicy !== undefined
467
+ ? {
468
+ setNetworkPolicy: async (policy: SandboxNetworkPolicy): Promise<void> => {
469
+ assertAdmissible('setNetworkPolicy')
470
+ await setNetworkPolicy(policy)
471
+ },
472
+ }
473
+ : {}),
474
+
475
+ async openTerminal(terminalOptions: OpenTerminalOptions): Promise<TerminalSession> {
476
+ assertAdmissible('openTerminal')
477
+ const terminal = await transport.openTerminal(terminalOptions)
478
+ terminals.add(terminal)
479
+ void terminal.exited.finally(() => terminals.delete(terminal))
480
+ return terminal
481
+ },
482
+
483
+ async openTcpConnection(
484
+ connectOptions: SandboxTcpConnectOptions,
485
+ ): Promise<SandboxTcpConnection> {
486
+ assertAdmissible('openTcpConnection')
487
+ return await transport.openTcpConnection(connectOptions)
488
+ },
489
+
490
+ async listFiles(rootPath: string): Promise<readonly SandboxFileEntry[]> {
491
+ return await runExecution('listFiles', async () => {
492
+ // Same wire as docker/aci/firecracker: `find -printf '%p\t%s\n'`,
493
+ // parsed line by line, with a non-zero exit (a root that does
494
+ // not exist yet) mapped to "empty" as the SDK contract asks.
495
+ const result = await transport.exec('find', [rootPath, '-type', 'f', '-printf', '%p\t%s\n'])
496
+ if (result.exitCode !== 0) return []
497
+ const entries: SandboxFileEntry[] = []
498
+ for (const rawLine of result.stdout.split('\n')) {
499
+ if (!rawLine) continue
500
+ const tab = rawLine.indexOf('\t')
501
+ if (tab < 0) continue
502
+ const filePath = rawLine.slice(0, tab)
503
+ const size = Number.parseInt(rawLine.slice(tab + 1), 10)
504
+ if (!filePath || !Number.isFinite(size)) continue
505
+ entries.push({ path: filePath, size })
506
+ }
507
+ return entries
508
+ })
509
+ },
510
+
511
+ /**
512
+ * Bounded, lazy file discovery — the method the SDK's `glob` and
513
+ * `grep` builtins refuse a sandbox for not having.
514
+ *
515
+ * Built on {@link walkFilesViaExec}, the same host-side enumerator the
516
+ * Firecracker and docker backends use, over this transport's `exec`:
517
+ * the guest needs no new agent op, because the walk IS an execution —
518
+ * `node -e` running the SDK's own walk program and streaming one JSONL
519
+ * record per match. The guest image is `node:22-bookworm-slim` (see
520
+ * `k8s/Dockerfile`) and the agent is itself node, so node on the
521
+ * guest's PATH is a precondition of the agent existing rather than a
522
+ * new requirement this method introduces.
523
+ *
524
+ * Ownership is the same as `exec`'s, and deliberately NOT
525
+ * `runExecution`'s: that helper wraps one awaited call, and a walk is a
526
+ * sequence of them. `activeExecutions` is therefore held for the whole
527
+ * walk rather than per entry — `status` reads `busy` from the first
528
+ * `next()` to the last, never flapping between yields — and an
529
+ * unconfirmed cancellation retires this handle exactly as a failed
530
+ * `exec` cancel does, on the same error class and through the same
531
+ * `retire()`.
532
+ *
533
+ * Cancellation: `options.signal` and the consumer's own
534
+ * `iterator.return()` both abort the underlying `exec`, which sends
535
+ * the guest a `cancel-execution` and kills the walk's process group —
536
+ * so breaking out of the loop after five entries leaves nothing
537
+ * running in the pod.
538
+ */
539
+ async *walkFiles(
540
+ rootPath: string,
541
+ walkOptions: SandboxWalkFilesOptions,
542
+ ): AsyncIterable<SandboxFileEntry> {
543
+ assertAdmissible('walkFiles')
544
+ activeExecutions += 1
545
+ try {
546
+ yield* walkFilesViaExec(
547
+ async (command, argv, execOpts) => await transport.exec(command, argv, execOpts),
548
+ rootPath,
549
+ walkOptions,
550
+ )
551
+ } catch (error) {
552
+ // The same rule `runExecution` applies, inlined because a
553
+ // generator cannot be wrapped by it: a command whose
554
+ // cancellation the guest could not confirm may still be running
555
+ // in that pod, so the pod stops being reusable.
556
+ //
557
+ // TWO SITES, ONE RULE. This block and `runExecution`'s must
558
+ // change together — #480, which owns the unconfirmed-cancel
559
+ // rule for this backend, is the next change to both, and a
560
+ // change that lands in one of them is a bug in the other.
561
+ if (error instanceof RemoteCancellationUnknownError) {
562
+ error.retirement = await retire()
563
+ }
564
+ throw error
565
+ } finally {
566
+ activeExecutions = Math.max(0, activeExecutions - 1)
567
+ }
568
+ },
569
+
570
+ async destroy(destroyOptions?: SandboxDestroyOptions): Promise<void> {
571
+ if (retirementPromise) {
572
+ const observation = await retirementPromise
573
+ if (observation.accepted) return
574
+ retirementPromise = undefined
575
+ }
576
+ // A terminal owns an interactive process tree in this pod. Stop and
577
+ // await every one before releasing the object, so the SDK's
578
+ // ownership contract is real rather than best-effort bookkeeping.
579
+ lifecycle = lifecycle === 'active' ? 'retiring' : lifecycle
580
+ lease?.stop()
581
+ const activeTerminals = [...terminals]
582
+ for (const terminal of activeTerminals) terminal.kill('SIGKILL')
583
+ await Promise.allSettled(activeTerminals.map((terminal) => terminal.exited))
584
+ terminals.clear()
585
+ // Deleting the claim cascades to the sandbox it adopted through the
586
+ // ownerReferences the controller re-parents on bind, so one DELETE
587
+ // retires the pod, the Service and the object. An object that is
588
+ // already gone counts as released — that is the state DELETE was
589
+ // asking for.
590
+ await teardown(destroyOptions?.signal)
591
+ },
592
+ }
593
+ }