@namzu/sandbox 13.0.0 → 15.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +1147 -0
- package/README.md +447 -0
- package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
- package/dist/backends/aci-standby-pool/index.js +13 -1
- package/dist/backends/aci-standby-pool/index.js.map +1 -1
- package/dist/backends/docker/index.d.ts.map +1 -1
- package/dist/backends/docker/index.js +19 -1
- package/dist/backends/docker/index.js.map +1 -1
- package/dist/backends/firecracker/index.d.ts.map +1 -1
- package/dist/backends/firecracker/index.js +12 -2
- package/dist/backends/firecracker/index.js.map +1 -1
- package/dist/backends/firecracker/protocol.d.ts +481 -8
- package/dist/backends/firecracker/protocol.d.ts.map +1 -1
- package/dist/backends/firecracker/protocol.js +136 -0
- package/dist/backends/firecracker/protocol.js.map +1 -1
- package/dist/backends/firecracker/transport.d.ts +642 -14
- package/dist/backends/firecracker/transport.d.ts.map +1 -1
- package/dist/backends/firecracker/transport.js +1307 -34
- package/dist/backends/firecracker/transport.js.map +1 -1
- package/dist/backends/kubernetes/egress-policy.d.ts +1296 -0
- package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -0
- package/dist/backends/kubernetes/egress-policy.js +2458 -0
- package/dist/backends/kubernetes/egress-policy.js.map +1 -0
- package/dist/backends/kubernetes/identity.d.ts +193 -0
- package/dist/backends/kubernetes/identity.d.ts.map +1 -0
- package/dist/backends/kubernetes/identity.js +147 -0
- package/dist/backends/kubernetes/identity.js.map +1 -0
- package/dist/backends/kubernetes/index.d.ts +1019 -0
- package/dist/backends/kubernetes/index.d.ts.map +1 -0
- package/dist/backends/kubernetes/index.js +1756 -0
- package/dist/backends/kubernetes/index.js.map +1 -0
- package/dist/backends/kubernetes/ingress-policy.d.ts +375 -0
- package/dist/backends/kubernetes/ingress-policy.d.ts.map +1 -0
- package/dist/backends/kubernetes/ingress-policy.js +1050 -0
- package/dist/backends/kubernetes/ingress-policy.js.map +1 -0
- package/dist/backends/kubernetes/k8s-client.d.ts +334 -0
- package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -0
- package/dist/backends/kubernetes/k8s-client.js +553 -0
- package/dist/backends/kubernetes/k8s-client.js.map +1 -0
- package/dist/backends/kubernetes/lease.d.ts +145 -0
- package/dist/backends/kubernetes/lease.d.ts.map +1 -0
- package/dist/backends/kubernetes/lease.js +201 -0
- package/dist/backends/kubernetes/lease.js.map +1 -0
- package/dist/backends/kubernetes/objects.d.ts +702 -0
- package/dist/backends/kubernetes/objects.d.ts.map +1 -0
- package/dist/backends/kubernetes/objects.js +518 -0
- package/dist/backends/kubernetes/objects.js.map +1 -0
- package/dist/backends/kubernetes/per-sandbox-policy.d.ts +219 -0
- package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -0
- package/dist/backends/kubernetes/per-sandbox-policy.js +407 -0
- package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -0
- package/dist/backends/kubernetes/privilege-probe.d.ts +136 -0
- package/dist/backends/kubernetes/privilege-probe.d.ts.map +1 -0
- package/dist/backends/kubernetes/privilege-probe.js +185 -0
- package/dist/backends/kubernetes/privilege-probe.js.map +1 -0
- package/dist/backends/kubernetes/rbac.d.ts +153 -0
- package/dist/backends/kubernetes/rbac.d.ts.map +1 -0
- package/dist/backends/kubernetes/rbac.js +177 -0
- package/dist/backends/kubernetes/rbac.js.map +1 -0
- package/dist/backends/kubernetes/sandbox.d.ts +190 -0
- package/dist/backends/kubernetes/sandbox.d.ts.map +1 -0
- package/dist/backends/kubernetes/sandbox.js +433 -0
- package/dist/backends/kubernetes/sandbox.js.map +1 -0
- package/dist/backends/kubernetes/transport.d.ts +1048 -0
- package/dist/backends/kubernetes/transport.d.ts.map +1 -0
- package/dist/backends/kubernetes/transport.js +2093 -0
- package/dist/backends/kubernetes/transport.js.map +1 -0
- package/dist/backends/kubernetes/workspace.d.ts +1512 -0
- package/dist/backends/kubernetes/workspace.d.ts.map +1 -0
- package/dist/backends/kubernetes/workspace.js +3703 -0
- package/dist/backends/kubernetes/workspace.js.map +1 -0
- package/dist/backends/remote-execution-controller.d.ts +14 -0
- package/dist/backends/remote-execution-controller.d.ts.map +1 -1
- package/dist/backends/remote-execution-controller.js.map +1 -1
- package/dist/index.d.ts +350 -2
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +344 -34
- package/dist/index.js.map +1 -1
- package/dist/testing/sandbox-conformance.d.ts +227 -0
- package/dist/testing/sandbox-conformance.d.ts.map +1 -0
- package/dist/testing/sandbox-conformance.js +896 -0
- package/dist/testing/sandbox-conformance.js.map +1 -0
- package/package.json +5 -4
- package/src/backends/aci-standby-pool/index.ts +16 -1
- package/src/backends/docker/index.ts +22 -1
- package/src/backends/firecracker/index.ts +14 -2
- package/src/backends/firecracker/protocol.ts +541 -6
- package/src/backends/firecracker/transport.ts +1687 -64
- package/src/backends/kubernetes/egress-policy.ts +3448 -0
- package/src/backends/kubernetes/identity.ts +261 -0
- package/src/backends/kubernetes/index.ts +2670 -0
- package/src/backends/kubernetes/ingress-policy.ts +1344 -0
- package/src/backends/kubernetes/k8s-client.ts +742 -0
- package/src/backends/kubernetes/lease.ts +254 -0
- package/src/backends/kubernetes/objects.ts +983 -0
- package/src/backends/kubernetes/per-sandbox-policy.ts +542 -0
- package/src/backends/kubernetes/privilege-probe.ts +261 -0
- package/src/backends/kubernetes/rbac.ts +192 -0
- package/src/backends/kubernetes/sandbox.ts +593 -0
- package/src/backends/kubernetes/transport.ts +2895 -0
- package/src/backends/kubernetes/workspace.ts +5640 -0
- package/src/backends/remote-execution-controller.ts +14 -0
- package/src/index.ts +838 -35
- package/src/testing/sandbox-conformance.ts +1202 -0
|
@@ -0,0 +1,593 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The {@link Sandbox} a kubernetes acquire hands back: the SDK contract,
|
|
3
|
+
* served over the guest agent's TCP transport, with the lease that keeps the
|
|
4
|
+
* cluster from deleting the pod out from under a long run.
|
|
5
|
+
*
|
|
6
|
+
* Split out of `index.ts` because that file is about the CONTROL plane —
|
|
7
|
+
* claim, poll, address, release — and this one is about the DATA plane, and
|
|
8
|
+
* the two are read for different reasons.
|
|
9
|
+
*
|
|
10
|
+
* ## What it implements, and what it deliberately does not
|
|
11
|
+
*
|
|
12
|
+
* Implemented: `exec` (through the shared {@link RemoteExecutionController},
|
|
13
|
+
* so an `AbortSignal` terminates the guest process rather than abandoning
|
|
14
|
+
* the wait), `writeFile`, `readFile`, `listFiles`, `walkFiles`,
|
|
15
|
+
* `openTerminal`, `openTcpConnection`, `destroy`.
|
|
16
|
+
*
|
|
17
|
+
* Conditionally present, on the same contract read the other way:
|
|
18
|
+
*
|
|
19
|
+
* - `setNetworkPolicy` — present on a TASK handle exactly when the backend
|
|
20
|
+
* was configured with `egress.perSandbox`, which is what gives this host
|
|
21
|
+
* an object to write (one `CiliumNetworkPolicy` per sandbox, owned by the
|
|
22
|
+
* claim), an admission fence bounding what it may write, and the RBAC to
|
|
23
|
+
* write it. Without that configuration there is still no per-running-pod
|
|
24
|
+
* knob to turn, so the method is ABSENT rather than present and throwing —
|
|
25
|
+
* the same rule, applied to a capability that now sometimes exists.
|
|
26
|
+
* Presence follows configuration alone, never a probe of the cluster. See
|
|
27
|
+
* `per-sandbox-policy.ts`.
|
|
28
|
+
*
|
|
29
|
+
* A WORKSPACE handle never carries it: `KubernetesWorkspace`'s create path
|
|
30
|
+
* builds its handle through `buildKubernetesSandbox` without this option,
|
|
31
|
+
* because it composes no per-sandbox pod label and tracks no owner uid for
|
|
32
|
+
* one. That is a property of the PATH and not of the configuration, so it
|
|
33
|
+
* is not answered by omission: `createKubernetesWorkspace` refuses a
|
|
34
|
+
* config carrying `egress.perSandbox` outright
|
|
35
|
+
* (`KubernetesWorkspacePerSandboxEgressConfigError`, before any request)
|
|
36
|
+
* rather than letting the option declare a capability the path never
|
|
37
|
+
* serves — see `workspace.ts`.
|
|
38
|
+
*
|
|
39
|
+
* Absent on purpose, because the SDK's contract says a backend that cannot
|
|
40
|
+
* honour an optional method must omit it rather than accept and ignore:
|
|
41
|
+
*
|
|
42
|
+
* - `spawnDetached` — the guest agent has no op that starts a process and
|
|
43
|
+
* returns it running. A host asking for background jobs must be told no.
|
|
44
|
+
*
|
|
45
|
+
* ## Terminals are owned
|
|
46
|
+
*
|
|
47
|
+
* `openTerminal` is only a compliant implementation if `destroy()` kills and
|
|
48
|
+
* awaits every terminal it returned, so open terminals are tracked and
|
|
49
|
+
* reaped before the object is released — the same thing the Firecracker
|
|
50
|
+
* backend does, for the same contract.
|
|
51
|
+
*
|
|
52
|
+
* ## Two terminal states, one `SandboxStatus`
|
|
53
|
+
*
|
|
54
|
+
* `SandboxStatus` has exactly four members and this change does not widen
|
|
55
|
+
* the SDK's union, so both ways a sandbox ends report `'destroyed'`. They
|
|
56
|
+
* are told apart by the error a later call throws:
|
|
57
|
+
* {@link KubernetesSandboxDestroyedError} (this host released it) and
|
|
58
|
+
* {@link KubernetesSandboxGoneError} (the cluster deleted it — the lease
|
|
59
|
+
* renewal found the object already gone).
|
|
60
|
+
*/
|
|
61
|
+
|
|
62
|
+
import type {
|
|
63
|
+
OpenTerminalOptions,
|
|
64
|
+
Sandbox,
|
|
65
|
+
SandboxDestroyOptions,
|
|
66
|
+
SandboxEnvironment,
|
|
67
|
+
SandboxExecOptions,
|
|
68
|
+
SandboxExecResult,
|
|
69
|
+
SandboxFileEntry,
|
|
70
|
+
SandboxId,
|
|
71
|
+
SandboxNetworkPolicy,
|
|
72
|
+
SandboxReadFileOptions,
|
|
73
|
+
SandboxStatus,
|
|
74
|
+
SandboxTcpConnectOptions,
|
|
75
|
+
SandboxTcpConnection,
|
|
76
|
+
SandboxWalkFilesOptions,
|
|
77
|
+
TerminalSession,
|
|
78
|
+
} from '@namzu/sdk'
|
|
79
|
+
import { walkFilesViaExec } from '@namzu/sdk'
|
|
80
|
+
|
|
81
|
+
import { OperationDeadline } from '../readiness.js'
|
|
82
|
+
import {
|
|
83
|
+
RemoteCancellationUnknownError,
|
|
84
|
+
type SandboxRetirementObservation,
|
|
85
|
+
} from '../remote-execution-controller.js'
|
|
86
|
+
import { KubernetesLeaseRenewal } from './lease.js'
|
|
87
|
+
import type { KubernetesAgentTransport } from './transport.js'
|
|
88
|
+
|
|
89
|
+
/**
|
|
90
|
+
* How long a retirement triggered by an unconfirmed cancellation may spend
|
|
91
|
+
* deleting the object. Separate from every other clock on the path: the
|
|
92
|
+
* caller's own deadline has usually already expired by the time this runs,
|
|
93
|
+
* and reusing it would mean skipping teardown exactly when a command of
|
|
94
|
+
* unknown state is still out there.
|
|
95
|
+
*/
|
|
96
|
+
const RETIREMENT_TIMEOUT_MS = 15_000
|
|
97
|
+
|
|
98
|
+
/** Thrown by any operation on a sandbox this host already destroyed. */
|
|
99
|
+
export class KubernetesSandboxDestroyedError extends Error {
|
|
100
|
+
override readonly name = 'KubernetesSandboxDestroyedError'
|
|
101
|
+
|
|
102
|
+
constructor(
|
|
103
|
+
readonly operation: string,
|
|
104
|
+
readonly sandboxName: string,
|
|
105
|
+
) {
|
|
106
|
+
super(
|
|
107
|
+
`kubernetes sandbox ${sandboxName} has been destroyed; ${operation}() cannot be admitted. Acquire a new sandbox — a destroyed one's pod, Service and claim are deleted and its agent address no longer resolves.`,
|
|
108
|
+
)
|
|
109
|
+
}
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
/**
|
|
113
|
+
* Thrown by any operation on a sandbox the CLUSTER removed while this
|
|
114
|
+
* handle still held it — the lease renewal PATCH came back 404/410. Distinct
|
|
115
|
+
* from {@link KubernetesSandboxDestroyedError} because nothing this host did
|
|
116
|
+
* caused it: the object expired, an operator deleted it, or the controller
|
|
117
|
+
* reaped it, and the actionable advice is different.
|
|
118
|
+
*/
|
|
119
|
+
export class KubernetesSandboxGoneError extends Error {
|
|
120
|
+
override readonly name = 'KubernetesSandboxGoneError'
|
|
121
|
+
|
|
122
|
+
constructor(
|
|
123
|
+
readonly operation: string,
|
|
124
|
+
readonly sandboxName: string,
|
|
125
|
+
) {
|
|
126
|
+
super(
|
|
127
|
+
`kubernetes sandbox ${sandboxName} no longer exists on the cluster; ${operation}() cannot be admitted. Its lease renewal found the object already deleted — it expired (spec.lifecycle.shutdownTime), an operator deleted it, or the controller reaped it. Nothing this handle can do brings it back; acquire a new sandbox.`,
|
|
128
|
+
)
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
interface KubernetesSandboxBaseOptions {
|
|
133
|
+
/** The cluster's own name for the bound sandbox — also the sandbox id. */
|
|
134
|
+
readonly name: string
|
|
135
|
+
readonly rootDir: string
|
|
136
|
+
readonly transport: KubernetesAgentTransport
|
|
137
|
+
/** DELETE the object this backend created. Already-gone counts as done. */
|
|
138
|
+
readonly release: (signal?: AbortSignal) => Promise<void>
|
|
139
|
+
/**
|
|
140
|
+
* Decide what an execution whose cancellation could not be CONFIRMED
|
|
141
|
+
* does to this sandbox — called instead of retiring it, and answering
|
|
142
|
+
* the {@link SandboxRetirementObservation} that goes onto the error the
|
|
143
|
+
* caller is about to receive.
|
|
144
|
+
*
|
|
145
|
+
* Unset (the task path, and the Firecracker tier through its own
|
|
146
|
+
* transport) keeps the shared controller's rule verbatim: a command of
|
|
147
|
+
* unknown state is still in that pod, the pod stops being reusable, and
|
|
148
|
+
* the handle retires it through {@link release}. That is right for a
|
|
149
|
+
* disposable object whose disk is scratch.
|
|
150
|
+
*
|
|
151
|
+
* It is wrong for an object that is not disposable. On a workspace
|
|
152
|
+
* `release` is an `operatingMode: Suspended` patch, which makes the
|
|
153
|
+
* controller delete the pod — so eight seconds of network loss under one
|
|
154
|
+
* `exec()` would take every other holder's terminals, dev servers and
|
|
155
|
+
* running commands with it, and no host-side lock can prevent it because
|
|
156
|
+
* no caller issued it. The workspace passes a hook that keeps the pod,
|
|
157
|
+
* diagnoses the agent and says `accepted: false` with a `reason` rather
|
|
158
|
+
* than letting a decision that large be made from inside a failing call.
|
|
159
|
+
*
|
|
160
|
+
* It must not reject; one that does is reported as an unaccepted
|
|
161
|
+
* retirement carrying its own error, so a broken hook cannot replace the
|
|
162
|
+
* error the caller asked about.
|
|
163
|
+
*/
|
|
164
|
+
readonly onUnconfirmedCancellation?: (
|
|
165
|
+
error: RemoteCancellationUnknownError,
|
|
166
|
+
) => Promise<SandboxRetirementObservation>
|
|
167
|
+
/**
|
|
168
|
+
* Narrow this sandbox's egress while it runs — present on the handle
|
|
169
|
+
* EXACTLY when this is passed, and passed by the TASK acquire exactly when
|
|
170
|
+
* `config.egress.perSandbox` is configured.
|
|
171
|
+
*
|
|
172
|
+
* That conditional presence is what the SDK's omit-or-throw contract
|
|
173
|
+
* licenses and what makes it honest here: without the configuration there
|
|
174
|
+
* is no policy object to write, no admission fence bounding what this
|
|
175
|
+
* host may write, and no RBAC grant to write it with, so the method is
|
|
176
|
+
* ABSENT rather than present and throwing. With it, `per-sandbox-policy.ts`
|
|
177
|
+
* writes one `CiliumNetworkPolicy` per sandbox, owned by the object the
|
|
178
|
+
* acquire created.
|
|
179
|
+
*
|
|
180
|
+
* The workspace passes NO such option whatever the config says — its
|
|
181
|
+
* create path composes no per-sandbox pod label and tracks no owner uid
|
|
182
|
+
* for one — and `createKubernetesWorkspace` refuses
|
|
183
|
+
* `config.egress.perSandbox` rather than silently omitting the method the
|
|
184
|
+
* option asks for.
|
|
185
|
+
*
|
|
186
|
+
* Presence depends only on configuration — never on a runtime probe of
|
|
187
|
+
* the cluster — so a caller's capability detection cannot come out
|
|
188
|
+
* differently depending on when it asked.
|
|
189
|
+
*/
|
|
190
|
+
readonly setNetworkPolicy?: (policy: SandboxNetworkPolicy) => Promise<void>
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
/**
|
|
194
|
+
* The lease half of the options: a way to move the expiry, and the expiry it
|
|
195
|
+
* is moving. Required TOGETHER, because `renew` without `ttlSeconds` is a
|
|
196
|
+
* renewal loop with nothing to stamp — it would re-stamp `now + 0`, an
|
|
197
|
+
* expiry already in the past, and hand the object straight to the
|
|
198
|
+
* controller's reaper while reporting every tick a success. A pair is the
|
|
199
|
+
* only shape that cannot be half-configured.
|
|
200
|
+
*/
|
|
201
|
+
interface KubernetesSandboxLeaseOptions {
|
|
202
|
+
/** PATCH the object's `shutdownTime` forward. See `lease.ts`. */
|
|
203
|
+
readonly renew: (shutdownTime: string, signal?: AbortSignal) => Promise<void>
|
|
204
|
+
/** The TTL acquire stamped; each renewal re-stamps exactly this much. */
|
|
205
|
+
readonly ttlSeconds: number
|
|
206
|
+
readonly onRenewalError?: (error: unknown) => void
|
|
207
|
+
/** Test seam: the renewal loop's base interval. Default: half the TTL. */
|
|
208
|
+
readonly leaseIntervalMs?: number
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
/**
|
|
212
|
+
* The other arm: an object that carries no expiry, so this handle runs no
|
|
213
|
+
* renewal loop at all — the persistent workspace (`workspace.ts`), which is
|
|
214
|
+
* explicitly managed and must outlive a host that stopped renewing. A no-op
|
|
215
|
+
* `renew` would be the wrong way to say that: it would leave a timer ticking
|
|
216
|
+
* forever to do nothing. The lease fields are typed `undefined` rather than
|
|
217
|
+
* omitted so that passing one of them here is a type error and not an
|
|
218
|
+
* excess-property check a spread would slip past.
|
|
219
|
+
*/
|
|
220
|
+
interface KubernetesSandboxUnleasedOptions {
|
|
221
|
+
readonly renew?: undefined
|
|
222
|
+
readonly ttlSeconds?: undefined
|
|
223
|
+
readonly onRenewalError?: undefined
|
|
224
|
+
readonly leaseIntervalMs?: undefined
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
export type KubernetesSandboxOptions = KubernetesSandboxBaseOptions &
|
|
228
|
+
(KubernetesSandboxLeaseOptions | KubernetesSandboxUnleasedOptions)
|
|
229
|
+
|
|
230
|
+
function detectEnvironment(): SandboxEnvironment {
|
|
231
|
+
// The guest runs Linux; the enum describes the host-facing shape of the
|
|
232
|
+
// worker, not the isolation technology under it. Firecracker's guest
|
|
233
|
+
// reports the same for the same reason.
|
|
234
|
+
return 'linux-namespace'
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
/**
|
|
238
|
+
* What this backend hands back: the SDK contract, with the optional members
|
|
239
|
+
* it DOES implement narrowed to present, so a caller that composes one —
|
|
240
|
+
* `workspace.ts` wraps this handle — does not have to re-check for a method
|
|
241
|
+
* this file always defines.
|
|
242
|
+
*/
|
|
243
|
+
export type KubernetesSandboxHandle = Sandbox &
|
|
244
|
+
Required<Pick<Sandbox, 'openTerminal' | 'openTcpConnection' | 'walkFiles' | 'readFileStream'>>
|
|
245
|
+
|
|
246
|
+
/**
|
|
247
|
+
* Build the handle. It does NOT run the acquire-time privilege probe — that
|
|
248
|
+
* is `create()`'s job in `index.ts`, so that a refusal can destroy this
|
|
249
|
+
* object before any caller has a reference to it, and so this function stays
|
|
250
|
+
* usable by the workspace path that runs its own probe.
|
|
251
|
+
*/
|
|
252
|
+
export function buildKubernetesSandbox(options: KubernetesSandboxOptions): KubernetesSandboxHandle {
|
|
253
|
+
// The cluster owns this name. Preserving it verbatim as the sandbox id —
|
|
254
|
+
// as the Firecracker backend preserves its orchestrator's — means a log
|
|
255
|
+
// line carrying an id is also a `kubectl get sandbox` argument.
|
|
256
|
+
const id = options.name as SandboxId
|
|
257
|
+
const transport = options.transport
|
|
258
|
+
// Captured as a const so the conditional member below narrows: the method
|
|
259
|
+
// is on the handle if and only if this is defined, decided once, here.
|
|
260
|
+
const setNetworkPolicy = options.setNetworkPolicy
|
|
261
|
+
|
|
262
|
+
type Lifecycle = 'active' | 'retiring' | 'destroyed' | 'gone'
|
|
263
|
+
let lifecycle: Lifecycle = 'active'
|
|
264
|
+
let activeExecutions = 0
|
|
265
|
+
let teardownPromise: Promise<void> | undefined
|
|
266
|
+
let teardownComplete = false
|
|
267
|
+
let retirementPromise: Promise<SandboxRetirementObservation> | undefined
|
|
268
|
+
const terminals = new Set<TerminalSession>()
|
|
269
|
+
|
|
270
|
+
// No `renew` ⇒ no expiry to move ⇒ no loop. `stop()` on the undefined
|
|
271
|
+
// case is the caller's problem to not have, which is why every use below
|
|
272
|
+
// goes through `lease?.stop()`.
|
|
273
|
+
const renew = options.renew
|
|
274
|
+
const ttlSeconds = options.ttlSeconds
|
|
275
|
+
// The type above already pairs the two. This is the runtime half of the
|
|
276
|
+
// same rule, for a caller that reached here through a cast or from
|
|
277
|
+
// JavaScript: a lease stamping `now + 0` expires the moment it is written,
|
|
278
|
+
// and every tick would report success while the controller deleted the
|
|
279
|
+
// object underneath it.
|
|
280
|
+
if (renew !== undefined && (typeof ttlSeconds !== 'number' || ttlSeconds <= 0)) {
|
|
281
|
+
throw new Error(
|
|
282
|
+
`kubernetes: sandbox ${options.name} was given a lease renewal with ttlSeconds ${String(ttlSeconds)}. A renewal re-stamps shutdownTime as now + ttlSeconds, so a zero or absent TTL stamps an expiry that has already passed and the object is reaped while the loop reports every tick a success. Pass renew and a positive ttlSeconds together, or neither — an object with no expiry (a persistent workspace) runs no renewal loop.`,
|
|
283
|
+
)
|
|
284
|
+
}
|
|
285
|
+
// `ttlSeconds === undefined` is unreachable once `renew` is defined — the
|
|
286
|
+
// throw above saw to that — and is written out anyway because it is what
|
|
287
|
+
// narrows the field to a number for the constructor below.
|
|
288
|
+
const lease =
|
|
289
|
+
renew === undefined || ttlSeconds === undefined
|
|
290
|
+
? undefined
|
|
291
|
+
: new KubernetesLeaseRenewal({
|
|
292
|
+
ttlSeconds,
|
|
293
|
+
renew,
|
|
294
|
+
onGone: () => {
|
|
295
|
+
// The object is gone; the pod behind the address went with it.
|
|
296
|
+
// Refuse every later call by name rather than let it dial into a
|
|
297
|
+
// connect timeout with nothing to explain it.
|
|
298
|
+
if (lifecycle === 'active') lifecycle = 'gone'
|
|
299
|
+
},
|
|
300
|
+
...(options.onRenewalError !== undefined
|
|
301
|
+
? { onRenewalError: options.onRenewalError }
|
|
302
|
+
: {}),
|
|
303
|
+
...(options.leaseIntervalMs !== undefined ? { intervalMs: options.leaseIntervalMs } : {}),
|
|
304
|
+
})
|
|
305
|
+
lease?.start()
|
|
306
|
+
|
|
307
|
+
const assertAdmissible = (operation: string): void => {
|
|
308
|
+
if (lifecycle === 'active') return
|
|
309
|
+
if (lifecycle === 'gone') throw new KubernetesSandboxGoneError(operation, options.name)
|
|
310
|
+
throw new KubernetesSandboxDestroyedError(operation, options.name)
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
const teardown = (signal?: AbortSignal): Promise<void> => {
|
|
314
|
+
if (lifecycle === 'active') lifecycle = 'retiring'
|
|
315
|
+
lease?.stop()
|
|
316
|
+
if (teardownComplete) return Promise.resolve()
|
|
317
|
+
if (teardownPromise) return teardownPromise
|
|
318
|
+
const shared = options.release(signal).then(
|
|
319
|
+
() => {
|
|
320
|
+
teardownComplete = true
|
|
321
|
+
lifecycle = 'destroyed'
|
|
322
|
+
},
|
|
323
|
+
(error: unknown) => {
|
|
324
|
+
// A failed teardown must stay retryable; keeping the rejected
|
|
325
|
+
// promise would answer every later destroy() with the same
|
|
326
|
+
// stale failure.
|
|
327
|
+
if (teardownPromise === shared) teardownPromise = undefined
|
|
328
|
+
throw error
|
|
329
|
+
},
|
|
330
|
+
)
|
|
331
|
+
teardownPromise = shared
|
|
332
|
+
return shared
|
|
333
|
+
}
|
|
334
|
+
|
|
335
|
+
/**
|
|
336
|
+
* A command whose cancellation the guest could not confirm may still be
|
|
337
|
+
* running in that pod, so the pod stops being reusable. Retire it and
|
|
338
|
+
* report whether the retirement landed, on the error the caller is about
|
|
339
|
+
* to receive.
|
|
340
|
+
*/
|
|
341
|
+
const retire = (): Promise<SandboxRetirementObservation> => {
|
|
342
|
+
if (lifecycle === 'active') lifecycle = 'retiring'
|
|
343
|
+
retirementPromise ??= new OperationDeadline(
|
|
344
|
+
RETIREMENT_TIMEOUT_MS,
|
|
345
|
+
`kubernetes sandbox ${options.name} retirement`,
|
|
346
|
+
)
|
|
347
|
+
.run(async (signal) => await teardown(signal))
|
|
348
|
+
.then(() => ({ accepted: true as const }))
|
|
349
|
+
.catch((error: unknown) => ({
|
|
350
|
+
accepted: false as const,
|
|
351
|
+
error: error instanceof Error ? error : new Error(String(error)),
|
|
352
|
+
}))
|
|
353
|
+
return retirementPromise
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
/**
|
|
357
|
+
* What an unconfirmed cancellation does to THIS sandbox: retire it, or
|
|
358
|
+
* whatever the owner's hook decided instead — see
|
|
359
|
+
* {@link KubernetesSandboxBaseOptions.onUnconfirmedCancellation}.
|
|
360
|
+
*/
|
|
361
|
+
const observeUnconfirmedCancellation = async (
|
|
362
|
+
error: RemoteCancellationUnknownError,
|
|
363
|
+
): Promise<SandboxRetirementObservation> => {
|
|
364
|
+
const decide = options.onUnconfirmedCancellation
|
|
365
|
+
if (decide === undefined) return await retire()
|
|
366
|
+
try {
|
|
367
|
+
return await decide(error)
|
|
368
|
+
} catch (hookError: unknown) {
|
|
369
|
+
// The caller is already receiving `error`; a hook that threw must
|
|
370
|
+
// not replace it, and must not be reported as a teardown that was
|
|
371
|
+
// attempted either.
|
|
372
|
+
return {
|
|
373
|
+
accepted: false,
|
|
374
|
+
error: hookError instanceof Error ? hookError : new Error(String(hookError)),
|
|
375
|
+
}
|
|
376
|
+
}
|
|
377
|
+
}
|
|
378
|
+
|
|
379
|
+
const runExecution = async <T>(operation: string, run: () => Promise<T>): Promise<T> => {
|
|
380
|
+
assertAdmissible(operation)
|
|
381
|
+
activeExecutions += 1
|
|
382
|
+
try {
|
|
383
|
+
return await run()
|
|
384
|
+
} catch (error) {
|
|
385
|
+
if (error instanceof RemoteCancellationUnknownError) {
|
|
386
|
+
error.retirement = await observeUnconfirmedCancellation(error)
|
|
387
|
+
}
|
|
388
|
+
throw error
|
|
389
|
+
} finally {
|
|
390
|
+
activeExecutions = Math.max(0, activeExecutions - 1)
|
|
391
|
+
}
|
|
392
|
+
}
|
|
393
|
+
|
|
394
|
+
return {
|
|
395
|
+
id,
|
|
396
|
+
get status(): SandboxStatus {
|
|
397
|
+
// Four members, and no new one: a cluster-side disappearance and a
|
|
398
|
+
// host-side destroy both read as 'destroyed' here and are told
|
|
399
|
+
// apart by the error a later call throws.
|
|
400
|
+
if (lifecycle !== 'active') return 'destroyed'
|
|
401
|
+
return activeExecutions > 0 ? 'busy' : 'ready'
|
|
402
|
+
},
|
|
403
|
+
rootDir: options.rootDir,
|
|
404
|
+
environment: detectEnvironment(),
|
|
405
|
+
|
|
406
|
+
async exec(
|
|
407
|
+
command: string,
|
|
408
|
+
argv?: string[],
|
|
409
|
+
opts?: SandboxExecOptions,
|
|
410
|
+
): Promise<SandboxExecResult> {
|
|
411
|
+
return await runExecution('exec', async () => await transport.exec(command, argv, opts))
|
|
412
|
+
},
|
|
413
|
+
|
|
414
|
+
/**
|
|
415
|
+
* Every `tcp` request dials a fresh connection, so its envelope is
|
|
416
|
+
* also that connection's first, not-yet-authenticated frame and is
|
|
417
|
+
* bounded by the guest's pre-auth frame ceiling (8 MiB by default).
|
|
418
|
+
* A body above it is no longer a refusal: the transport splits it
|
|
419
|
+
* into parts that each fit, writes them to a temporary sibling of
|
|
420
|
+
* the target and finishes with an atomic rename, so this method
|
|
421
|
+
* takes a body of any size the transport's `maxWriteFileBytes`
|
|
422
|
+
* admits (1 GiB by default). The named refusals that remain are
|
|
423
|
+
* passed through unwrapped so a caller can catch them BY CLASS:
|
|
424
|
+
* `AgentWriteFileTooLargeError` for a body above that bound, and
|
|
425
|
+
* `AgentPreauthFrameTooLargeError` for an oversized body against a
|
|
426
|
+
* guest too old to advertise the part protocol.
|
|
427
|
+
*/
|
|
428
|
+
async writeFile(path: string, content: string | Buffer): Promise<void> {
|
|
429
|
+
assertAdmissible('writeFile')
|
|
430
|
+
const buf = Buffer.isBuffer(content) ? content : Buffer.from(content, 'utf8')
|
|
431
|
+
await transport.writeFile(path, buf)
|
|
432
|
+
},
|
|
433
|
+
|
|
434
|
+
/**
|
|
435
|
+
* A whole-file read is served by the guest's streamed op when the
|
|
436
|
+
* guest advertises it, so neither side holds the file's base64 form
|
|
437
|
+
* or its JSON envelope in one piece and a file of any size this
|
|
438
|
+
* workspace's disk holds can be read. `offset`/`length` ask for one
|
|
439
|
+
* slice instead; a guest too old to honour them is refused with
|
|
440
|
+
* `AgentReadFileStreamUnsupportedError` rather than answering with
|
|
441
|
+
* the whole file.
|
|
442
|
+
*/
|
|
443
|
+
async readFile(path: string, readOptions?: SandboxReadFileOptions): Promise<Buffer> {
|
|
444
|
+
assertAdmissible('readFile')
|
|
445
|
+
return await transport.readFile(path, readOptions)
|
|
446
|
+
},
|
|
447
|
+
|
|
448
|
+
/**
|
|
449
|
+
* Chunks, in order, with nothing whole at either end — what a host
|
|
450
|
+
* draining a large output file before `destroy()` needs. The
|
|
451
|
+
* admissibility check runs at the call, not per chunk: a workspace
|
|
452
|
+
* suspended mid-stream takes its pod's connection with it, which is
|
|
453
|
+
* what ends the iteration.
|
|
454
|
+
*/
|
|
455
|
+
readFileStream(path: string, readOptions?: SandboxReadFileOptions): AsyncIterable<Buffer> {
|
|
456
|
+
assertAdmissible('readFileStream')
|
|
457
|
+
return transport.readFileStream(path, readOptions)
|
|
458
|
+
},
|
|
459
|
+
|
|
460
|
+
// Present only when the backend was configured for per-sandbox egress
|
|
461
|
+
// — see {@link KubernetesSandboxBaseOptions.setNetworkPolicy}. The
|
|
462
|
+
// admissibility gate is this file's, not the writer's, so a destroyed
|
|
463
|
+
// or cluster-removed sandbox refuses BY NAME here, exactly as every
|
|
464
|
+
// other method does, rather than failing at the API server one round
|
|
465
|
+
// trip later.
|
|
466
|
+
...(setNetworkPolicy !== undefined
|
|
467
|
+
? {
|
|
468
|
+
setNetworkPolicy: async (policy: SandboxNetworkPolicy): Promise<void> => {
|
|
469
|
+
assertAdmissible('setNetworkPolicy')
|
|
470
|
+
await setNetworkPolicy(policy)
|
|
471
|
+
},
|
|
472
|
+
}
|
|
473
|
+
: {}),
|
|
474
|
+
|
|
475
|
+
async openTerminal(terminalOptions: OpenTerminalOptions): Promise<TerminalSession> {
|
|
476
|
+
assertAdmissible('openTerminal')
|
|
477
|
+
const terminal = await transport.openTerminal(terminalOptions)
|
|
478
|
+
terminals.add(terminal)
|
|
479
|
+
void terminal.exited.finally(() => terminals.delete(terminal))
|
|
480
|
+
return terminal
|
|
481
|
+
},
|
|
482
|
+
|
|
483
|
+
async openTcpConnection(
|
|
484
|
+
connectOptions: SandboxTcpConnectOptions,
|
|
485
|
+
): Promise<SandboxTcpConnection> {
|
|
486
|
+
assertAdmissible('openTcpConnection')
|
|
487
|
+
return await transport.openTcpConnection(connectOptions)
|
|
488
|
+
},
|
|
489
|
+
|
|
490
|
+
async listFiles(rootPath: string): Promise<readonly SandboxFileEntry[]> {
|
|
491
|
+
return await runExecution('listFiles', async () => {
|
|
492
|
+
// Same wire as docker/aci/firecracker: `find -printf '%p\t%s\n'`,
|
|
493
|
+
// parsed line by line, with a non-zero exit (a root that does
|
|
494
|
+
// not exist yet) mapped to "empty" as the SDK contract asks.
|
|
495
|
+
const result = await transport.exec('find', [rootPath, '-type', 'f', '-printf', '%p\t%s\n'])
|
|
496
|
+
if (result.exitCode !== 0) return []
|
|
497
|
+
const entries: SandboxFileEntry[] = []
|
|
498
|
+
for (const rawLine of result.stdout.split('\n')) {
|
|
499
|
+
if (!rawLine) continue
|
|
500
|
+
const tab = rawLine.indexOf('\t')
|
|
501
|
+
if (tab < 0) continue
|
|
502
|
+
const filePath = rawLine.slice(0, tab)
|
|
503
|
+
const size = Number.parseInt(rawLine.slice(tab + 1), 10)
|
|
504
|
+
if (!filePath || !Number.isFinite(size)) continue
|
|
505
|
+
entries.push({ path: filePath, size })
|
|
506
|
+
}
|
|
507
|
+
return entries
|
|
508
|
+
})
|
|
509
|
+
},
|
|
510
|
+
|
|
511
|
+
/**
|
|
512
|
+
* Bounded, lazy file discovery — the method the SDK's `glob` and
|
|
513
|
+
* `grep` builtins refuse a sandbox for not having.
|
|
514
|
+
*
|
|
515
|
+
* Built on {@link walkFilesViaExec}, the same host-side enumerator the
|
|
516
|
+
* Firecracker and docker backends use, over this transport's `exec`:
|
|
517
|
+
* the guest needs no new agent op, because the walk IS an execution —
|
|
518
|
+
* `node -e` running the SDK's own walk program and streaming one JSONL
|
|
519
|
+
* record per match. The guest image is `node:22-bookworm-slim` (see
|
|
520
|
+
* `k8s/Dockerfile`) and the agent is itself node, so node on the
|
|
521
|
+
* guest's PATH is a precondition of the agent existing rather than a
|
|
522
|
+
* new requirement this method introduces.
|
|
523
|
+
*
|
|
524
|
+
* Ownership is the same as `exec`'s, and deliberately NOT
|
|
525
|
+
* `runExecution`'s: that helper wraps one awaited call, and a walk is a
|
|
526
|
+
* sequence of them. `activeExecutions` is therefore held for the whole
|
|
527
|
+
* walk rather than per entry — `status` reads `busy` from the first
|
|
528
|
+
* `next()` to the last, never flapping between yields — and an
|
|
529
|
+
* unconfirmed cancellation retires this handle exactly as a failed
|
|
530
|
+
* `exec` cancel does, on the same error class and through the same
|
|
531
|
+
* `retire()`.
|
|
532
|
+
*
|
|
533
|
+
* Cancellation: `options.signal` and the consumer's own
|
|
534
|
+
* `iterator.return()` both abort the underlying `exec`, which sends
|
|
535
|
+
* the guest a `cancel-execution` and kills the walk's process group —
|
|
536
|
+
* so breaking out of the loop after five entries leaves nothing
|
|
537
|
+
* running in the pod.
|
|
538
|
+
*/
|
|
539
|
+
async *walkFiles(
|
|
540
|
+
rootPath: string,
|
|
541
|
+
walkOptions: SandboxWalkFilesOptions,
|
|
542
|
+
): AsyncIterable<SandboxFileEntry> {
|
|
543
|
+
assertAdmissible('walkFiles')
|
|
544
|
+
activeExecutions += 1
|
|
545
|
+
try {
|
|
546
|
+
yield* walkFilesViaExec(
|
|
547
|
+
async (command, argv, execOpts) => await transport.exec(command, argv, execOpts),
|
|
548
|
+
rootPath,
|
|
549
|
+
walkOptions,
|
|
550
|
+
)
|
|
551
|
+
} catch (error) {
|
|
552
|
+
// The same rule `runExecution` applies, inlined because a
|
|
553
|
+
// generator cannot be wrapped by it: a command whose
|
|
554
|
+
// cancellation the guest could not confirm may still be running
|
|
555
|
+
// in that pod, so the pod stops being reusable.
|
|
556
|
+
//
|
|
557
|
+
// TWO SITES, ONE RULE. This block and `runExecution`'s must
|
|
558
|
+
// change together — #480, which owns the unconfirmed-cancel
|
|
559
|
+
// rule for this backend, is the next change to both, and a
|
|
560
|
+
// change that lands in one of them is a bug in the other.
|
|
561
|
+
if (error instanceof RemoteCancellationUnknownError) {
|
|
562
|
+
error.retirement = await retire()
|
|
563
|
+
}
|
|
564
|
+
throw error
|
|
565
|
+
} finally {
|
|
566
|
+
activeExecutions = Math.max(0, activeExecutions - 1)
|
|
567
|
+
}
|
|
568
|
+
},
|
|
569
|
+
|
|
570
|
+
async destroy(destroyOptions?: SandboxDestroyOptions): Promise<void> {
|
|
571
|
+
if (retirementPromise) {
|
|
572
|
+
const observation = await retirementPromise
|
|
573
|
+
if (observation.accepted) return
|
|
574
|
+
retirementPromise = undefined
|
|
575
|
+
}
|
|
576
|
+
// A terminal owns an interactive process tree in this pod. Stop and
|
|
577
|
+
// await every one before releasing the object, so the SDK's
|
|
578
|
+
// ownership contract is real rather than best-effort bookkeeping.
|
|
579
|
+
lifecycle = lifecycle === 'active' ? 'retiring' : lifecycle
|
|
580
|
+
lease?.stop()
|
|
581
|
+
const activeTerminals = [...terminals]
|
|
582
|
+
for (const terminal of activeTerminals) terminal.kill('SIGKILL')
|
|
583
|
+
await Promise.allSettled(activeTerminals.map((terminal) => terminal.exited))
|
|
584
|
+
terminals.clear()
|
|
585
|
+
// Deleting the claim cascades to the sandbox it adopted through the
|
|
586
|
+
// ownerReferences the controller re-parents on bind, so one DELETE
|
|
587
|
+
// retires the pod, the Service and the object. An object that is
|
|
588
|
+
// already gone counts as released — that is the state DELETE was
|
|
589
|
+
// asking for.
|
|
590
|
+
await teardown(destroyOptions?.signal)
|
|
591
|
+
},
|
|
592
|
+
}
|
|
593
|
+
}
|