@namzu/sandbox 13.0.0 → 14.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +309 -0
- package/README.md +151 -0
- package/dist/backends/firecracker/protocol.d.ts +22 -0
- package/dist/backends/firecracker/protocol.d.ts.map +1 -1
- package/dist/backends/firecracker/protocol.js.map +1 -1
- package/dist/backends/firecracker/transport.d.ts +104 -9
- package/dist/backends/firecracker/transport.d.ts.map +1 -1
- package/dist/backends/firecracker/transport.js +139 -13
- package/dist/backends/firecracker/transport.js.map +1 -1
- package/dist/backends/kubernetes/egress-policy.d.ts +219 -0
- package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -0
- package/dist/backends/kubernetes/egress-policy.js +314 -0
- package/dist/backends/kubernetes/egress-policy.js.map +1 -0
- package/dist/backends/kubernetes/index.d.ts +374 -0
- package/dist/backends/kubernetes/index.d.ts.map +1 -0
- package/dist/backends/kubernetes/index.js +671 -0
- package/dist/backends/kubernetes/index.js.map +1 -0
- package/dist/backends/kubernetes/k8s-client.d.ts +125 -0
- package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -0
- package/dist/backends/kubernetes/k8s-client.js +246 -0
- package/dist/backends/kubernetes/k8s-client.js.map +1 -0
- package/dist/backends/kubernetes/lease.d.ts +119 -0
- package/dist/backends/kubernetes/lease.d.ts.map +1 -0
- package/dist/backends/kubernetes/lease.js +151 -0
- package/dist/backends/kubernetes/lease.js.map +1 -0
- package/dist/backends/kubernetes/objects.d.ts +282 -0
- package/dist/backends/kubernetes/objects.d.ts.map +1 -0
- package/dist/backends/kubernetes/objects.js +156 -0
- package/dist/backends/kubernetes/objects.js.map +1 -0
- package/dist/backends/kubernetes/privilege-probe.d.ts +136 -0
- package/dist/backends/kubernetes/privilege-probe.d.ts.map +1 -0
- package/dist/backends/kubernetes/privilege-probe.js +185 -0
- package/dist/backends/kubernetes/privilege-probe.js.map +1 -0
- package/dist/backends/kubernetes/sandbox.d.ts +123 -0
- package/dist/backends/kubernetes/sandbox.d.ts.map +1 -0
- package/dist/backends/kubernetes/sandbox.js +299 -0
- package/dist/backends/kubernetes/sandbox.js.map +1 -0
- package/dist/backends/kubernetes/transport.d.ts +122 -0
- package/dist/backends/kubernetes/transport.d.ts.map +1 -0
- package/dist/backends/kubernetes/transport.js +197 -0
- package/dist/backends/kubernetes/transport.js.map +1 -0
- package/dist/backends/kubernetes/workspace.d.ts +381 -0
- package/dist/backends/kubernetes/workspace.d.ts.map +1 -0
- package/dist/backends/kubernetes/workspace.js +1064 -0
- package/dist/backends/kubernetes/workspace.js.map +1 -0
- package/dist/index.d.ts +132 -2
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +102 -34
- package/dist/index.js.map +1 -1
- package/dist/testing/sandbox-conformance.d.ts +193 -0
- package/dist/testing/sandbox-conformance.d.ts.map +1 -0
- package/dist/testing/sandbox-conformance.js +465 -0
- package/dist/testing/sandbox-conformance.js.map +1 -0
- package/package.json +5 -4
- package/src/backends/firecracker/protocol.ts +27 -0
- package/src/backends/firecracker/transport.ts +199 -28
- package/src/backends/kubernetes/egress-policy.ts +437 -0
- package/src/backends/kubernetes/index.ts +1012 -0
- package/src/backends/kubernetes/k8s-client.ts +352 -0
- package/src/backends/kubernetes/lease.ts +198 -0
- package/src/backends/kubernetes/objects.ts +363 -0
- package/src/backends/kubernetes/privilege-probe.ts +261 -0
- package/src/backends/kubernetes/sandbox.ts +395 -0
- package/src/backends/kubernetes/transport.ts +286 -0
- package/src/backends/kubernetes/workspace.ts +1386 -0
- package/src/index.ts +257 -35
- package/src/testing/sandbox-conformance.ts +667 -0
|
@@ -0,0 +1,1064 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The persistent workspace: a Sandbox that keeps its disk across suspends.
|
|
3
|
+
*
|
|
4
|
+
* A task sandbox (`index.ts`) is claimed, used and deleted inside one run. A
|
|
5
|
+
* workspace is the opposite object: it is created once, addressed by a name
|
|
6
|
+
* the CALLER chooses, suspended when nobody is using it, resumed days later
|
|
7
|
+
* with yesterday's dependency cache and git checkout still on its disk, and
|
|
8
|
+
* deleted only when someone says so.
|
|
9
|
+
*
|
|
10
|
+
* ## Why this is a directly created Sandbox and never a claim
|
|
11
|
+
*
|
|
12
|
+
* `Sandbox.spec.volumeClaimTemplates` is CEL-immutable on the served CRD
|
|
13
|
+
* ("volumeClaimTemplates is immutable"), and a `SandboxClaim` that carries
|
|
14
|
+
* `spec.volumeClaimTemplates` is forced to cold-start instead of adopting a
|
|
15
|
+
* warm pool sandbox. So the disk has to be in the spec at creation, and the
|
|
16
|
+
* appealing middle road — claim a warm diskless sandbox and attach a disk to
|
|
17
|
+
* it — is not expressible in this API at all. A workspace is therefore a
|
|
18
|
+
* `Sandbox` POSTed directly, with a deterministic name, and the warm pool has
|
|
19
|
+
* nothing to do with it.
|
|
20
|
+
*
|
|
21
|
+
* ## Block, not a filesystem
|
|
22
|
+
*
|
|
23
|
+
* The disk must be `volumeMode: Block`, consumed through the container's
|
|
24
|
+
* `volumeDevices`, and this module REFUSES a template whose disk is anything
|
|
25
|
+
* else. Under a VM-isolating RuntimeClass a `Filesystem` PVC reaches the guest
|
|
26
|
+
* through a host/guest filesystem passthrough (virtio-fs), whose per-file
|
|
27
|
+
* overhead lands squarely on the two things a workspace does all day: walking
|
|
28
|
+
* a dependency tree and touching thousands of small files. Nothing FAILS; the
|
|
29
|
+
* workspace is merely several times slower, and no functional test can see
|
|
30
|
+
* that. A raw block device the guest formats and mounts itself is an ordinary
|
|
31
|
+
* local filesystem inside the VM. See {@link KubernetesWorkspaceDiskError}.
|
|
32
|
+
*
|
|
33
|
+
* ## No lease
|
|
34
|
+
*
|
|
35
|
+
* Every task sandbox carries `shutdownTime` + `shutdownPolicy: Delete`, so a
|
|
36
|
+
* host that dies mid-run costs the cluster one expiry rather than a leak, and
|
|
37
|
+
* its handle renews that expiry for as long as it lives. A workspace carries
|
|
38
|
+
* NEITHER. An expiry on a workspace is a timer that deletes a caller's files,
|
|
39
|
+
* and a renewal loop makes losing them conditional on a host process staying
|
|
40
|
+
* up — exactly backwards for an object whose whole purpose is to outlive the
|
|
41
|
+
* host. A workspace is explicitly managed: it goes away when
|
|
42
|
+
* `destroy({ deleteDisk: true })` says so, and not before.
|
|
43
|
+
*
|
|
44
|
+
* ## Suspend, resume, and what changes across one
|
|
45
|
+
*
|
|
46
|
+
* `suspend()` merge-PATCHes `spec.operatingMode: Suspended`; the controller
|
|
47
|
+
* deletes only the Pod and reconciles PVCs unconditionally on every pass, so
|
|
48
|
+
* the disk survives with the same UID. It then waits on the POD — not on the
|
|
49
|
+
* Sandbox's `Suspended` condition, which upstream documents as lingering True
|
|
50
|
+
* after a resume, and not merely on a `deletionTimestamp`, which appears
|
|
51
|
+
* while the guest is still running. `resume()` PATCHes it back and waits
|
|
52
|
+
* for a new pod — and a resumed pod keeps the sandbox's NAME while getting a
|
|
53
|
+
* new uid and a new IP. Both matter: the uid is the agent's bind token, and
|
|
54
|
+
* the IP is where the agent answers. So resume re-resolves the address, re-
|
|
55
|
+
* reads the uid (skipping the outgoing pod, which is still listed under the
|
|
56
|
+
* same name while it terminates) and rebuilds the transport. Nothing from
|
|
57
|
+
* before the suspend is reused.
|
|
58
|
+
*
|
|
59
|
+
* Between the two, every call refuses with
|
|
60
|
+
* {@link KubernetesWorkspaceSuspendedError} and issues no dial. A dial would
|
|
61
|
+
* be worse than useless: the address still resolves — the Service outlives
|
|
62
|
+
* the pod — so the call would hang until a connect timeout with nothing in
|
|
63
|
+
* the failure naming the suspend.
|
|
64
|
+
*
|
|
65
|
+
* ## Adoption is checked against the object, not against the caller
|
|
66
|
+
*
|
|
67
|
+
* A create that collides with an existing object of the same name ADOPTS it,
|
|
68
|
+
* because the deterministic name is only worth having if coming back is the
|
|
69
|
+
* normal path. What is adopted is then checked against the configuration: the
|
|
70
|
+
* block disk, the `sandbox.namzu.ai/template` pod label (the label an egress
|
|
71
|
+
* NetworkPolicy selects by) and `runtimeClassName` (the VM boundary). A
|
|
72
|
+
* standing object that disagrees with any of them is refused by name rather
|
|
73
|
+
* than driven — see {@link KubernetesWorkspaceMismatchError}. What is NOT
|
|
74
|
+
* checked, and cannot be from here, is whether somebody else is already using
|
|
75
|
+
* it: two host processes can hold handles to one running workspace, and the
|
|
76
|
+
* `destroy()` of either suspends the pod the other is executing in. A
|
|
77
|
+
* workspace id is a name, not a lock.
|
|
78
|
+
*
|
|
79
|
+
* ## There is no delete-compute-keep-disk verb
|
|
80
|
+
*
|
|
81
|
+
* The API has `operatingMode` and it has DELETE. Nothing in between. So
|
|
82
|
+
* `destroy()` with no options, or `deleteDisk: false`, SUSPENDS and leaves
|
|
83
|
+
* the object standing; only `destroy({ deleteDisk: true })` DELETEs the
|
|
84
|
+
* Sandbox, which cascades to the Pod, the Service and the PVC through
|
|
85
|
+
* ownerReferences. The default is the non-destructive one because `destroy()`
|
|
86
|
+
* is what a `finally` block calls, and a `finally` block must not be able to
|
|
87
|
+
* erase a workspace nobody asked to erase.
|
|
88
|
+
*
|
|
89
|
+
* The same holds for the paths nobody asked for at all. A create or resume
|
|
90
|
+
* that fails suspends the object and rethrows rather than cleaning it up (see
|
|
91
|
+
* {@link createKubernetesWorkspace}), and a pod that stops being able to say
|
|
92
|
+
* what happened to a command — the shared execution controller's unconfirmed
|
|
93
|
+
* cancellation, which retires a task sandbox by DELETING it — is retired here
|
|
94
|
+
* by that same suspend patch (see {@link retireSession}). Exactly one DELETE
|
|
95
|
+
* is reachable from this module, and it is the one `deleteDisk: true` asks
|
|
96
|
+
* for.
|
|
97
|
+
*
|
|
98
|
+
* ## A state is committed when the cluster confirms it, never before
|
|
99
|
+
*
|
|
100
|
+
* Both verbs are idempotent, and both are idempotent by EARLY-RETURNING on a
|
|
101
|
+
* state. That makes the moment a state is written the whole correctness
|
|
102
|
+
* question: a handle that marks itself `deleted` before its DELETE lands
|
|
103
|
+
* answers every later `destroy()` from that mark, so a 500 is thrown once and
|
|
104
|
+
* the Sandbox then stands on the cluster with nothing left that would remove
|
|
105
|
+
* it. The same shape on `suspend()` leaves a pod running — and billing —
|
|
106
|
+
* behind a handle that says it is suspended.
|
|
107
|
+
*
|
|
108
|
+
* So `suspended` is written after the patch lands AND the pod is observed
|
|
109
|
+
* stopped, `deleted` after the DELETE resolves (or reports the object already
|
|
110
|
+
* gone), and a request that fails leaves the state it found. Concurrency is
|
|
111
|
+
* covered the other way round, by a single flight per verb: a second caller
|
|
112
|
+
* arriving mid-transition awaits the one in progress instead of sending a
|
|
113
|
+
* second request into the gap the deferred mark opens.
|
|
114
|
+
*/
|
|
115
|
+
import { OperationDeadline, OperationDeadlineExpired, runFailureCleanup } from '../readiness.js';
|
|
116
|
+
import { assertEgressPolicyIsEnforceable } from './egress-policy.js';
|
|
117
|
+
import { DEFAULT_AGENT_PORT, bindingFromSandbox, buildSandboxBody, clientAccess, pollForBinding, probeSandboxPrivileges, readPodBindToken, readSandboxTemplate, resolveAgentAddress, resolveKubernetesReadiness, resolveProbeTimeoutMs, verifyEgressPolicyConfigured, } from './index.js';
|
|
118
|
+
import { KubernetesAlreadyGoneError, KubernetesConflictError, createKubernetesClient, } from './k8s-client.js';
|
|
119
|
+
import { SANDBOX_TEMPLATE_LABEL_KEY, isPodStopped, podPath, sandboxCollectionPath, sandboxPath, } from './objects.js';
|
|
120
|
+
import { KubernetesSandboxDestroyedError, buildKubernetesSandbox, } from './sandbox.js';
|
|
121
|
+
import { KubernetesAgentTransport } from './transport.js';
|
|
122
|
+
/**
|
|
123
|
+
* Thrown when a workspace's `SandboxTemplate` does not describe a block disk
|
|
124
|
+
* this backend is willing to build a workspace on.
|
|
125
|
+
*
|
|
126
|
+
* Named, and thrown before anything is created, because every shape it
|
|
127
|
+
* refuses WORKS: a template with no disk produces a sandbox whose files
|
|
128
|
+
* vanish on the next suspend, and a `Filesystem` disk produces one that keeps
|
|
129
|
+
* its files and is quietly several times slower at the small-file IO a
|
|
130
|
+
* workspace is made of. Neither fails a functional test, so neither can be
|
|
131
|
+
* left to be noticed later.
|
|
132
|
+
*/
|
|
133
|
+
export class KubernetesWorkspaceDiskError extends Error {
|
|
134
|
+
source;
|
|
135
|
+
name = 'KubernetesWorkspaceDiskError';
|
|
136
|
+
constructor(
|
|
137
|
+
/** What was inspected — a `SandboxTemplate`, or an existing Sandbox. */
|
|
138
|
+
source, message) {
|
|
139
|
+
super(message);
|
|
140
|
+
this.source = source;
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
/**
|
|
144
|
+
* Thrown when the Sandbox already standing under a workspace's name was not
|
|
145
|
+
* built the way this caller is configured to build one.
|
|
146
|
+
*
|
|
147
|
+
* Adoption is the NORMAL path — the deterministic name exists so that coming
|
|
148
|
+
* back to a workspace is cheap — and that is exactly why the object handed
|
|
149
|
+
* back is checked against the configuration rather than against the caller's
|
|
150
|
+
* intention. Two controls would otherwise be lost silently, and lost for as
|
|
151
|
+
* long as the workspace lives, which is the longest of anything this backend
|
|
152
|
+
* makes:
|
|
153
|
+
*
|
|
154
|
+
* - the `sandbox.namzu.ai/template` pod label, which is what a translated
|
|
155
|
+
* egress `NetworkPolicy`'s `podSelector` matches. A pod carrying another
|
|
156
|
+
* value — or none — is not selected by the policy this call just verified,
|
|
157
|
+
* so the boundary would report verified while covering nothing.
|
|
158
|
+
* - `runtimeClassName`, which is the VM boundary. The privilege probe cannot
|
|
159
|
+
* stand in for it: `/proc/self/status` reads the same inside a VM guest as
|
|
160
|
+
* it does inside an ordinary shared-kernel container.
|
|
161
|
+
*
|
|
162
|
+
* Nothing is patched to make the standing object match. Its disk may hold a
|
|
163
|
+
* month of the caller's files, and rewriting a live workspace's podTemplate to
|
|
164
|
+
* fit a new configuration is a larger decision than reattaching to it.
|
|
165
|
+
*/
|
|
166
|
+
export class KubernetesWorkspaceMismatchError extends Error {
|
|
167
|
+
sandboxName;
|
|
168
|
+
field;
|
|
169
|
+
expected;
|
|
170
|
+
actual;
|
|
171
|
+
name = 'KubernetesWorkspaceMismatchError';
|
|
172
|
+
constructor(
|
|
173
|
+
/** The Sandbox found standing under this workspace's name. */
|
|
174
|
+
sandboxName,
|
|
175
|
+
/** Which configured field the standing object disagrees with. */
|
|
176
|
+
field,
|
|
177
|
+
/** What the configuration asked for. */
|
|
178
|
+
expected,
|
|
179
|
+
/** What the object carries — absent when it carries nothing at all. */
|
|
180
|
+
actual, message) {
|
|
181
|
+
super(message);
|
|
182
|
+
this.sandboxName = sandboxName;
|
|
183
|
+
this.field = field;
|
|
184
|
+
this.expected = expected;
|
|
185
|
+
this.actual = actual;
|
|
186
|
+
}
|
|
187
|
+
}
|
|
188
|
+
/**
|
|
189
|
+
* Thrown by every operation on a workspace that is currently suspended.
|
|
190
|
+
*
|
|
191
|
+
* Distinct from {@link KubernetesSandboxDestroyedError} because the state is
|
|
192
|
+
* RECOVERABLE and the advice is one word: call `resume()`. Nothing is dialed
|
|
193
|
+
* before it is thrown.
|
|
194
|
+
*/
|
|
195
|
+
export class KubernetesWorkspaceSuspendedError extends Error {
|
|
196
|
+
operation;
|
|
197
|
+
workspaceId;
|
|
198
|
+
sandboxName;
|
|
199
|
+
name = 'KubernetesWorkspaceSuspendedError';
|
|
200
|
+
constructor(operation, workspaceId, sandboxName) {
|
|
201
|
+
super(`kubernetes workspace ${workspaceId} (Sandbox ${sandboxName}) is suspended; ${operation}() cannot be admitted and nothing was dialed. Its pod is deleted and its disk is intact — call resume() to get a new pod, a new address and a new agent token, then retry.`);
|
|
202
|
+
this.operation = operation;
|
|
203
|
+
this.workspaceId = workspaceId;
|
|
204
|
+
this.sandboxName = sandboxName;
|
|
205
|
+
}
|
|
206
|
+
}
|
|
207
|
+
/**
|
|
208
|
+
* Thrown when a suspend's `operatingMode: Suspended` patch was accepted and
|
|
209
|
+
* the pod had still not stopped by the readiness deadline.
|
|
210
|
+
*
|
|
211
|
+
* The workspace is left in the state that is TRUE rather than the one that
|
|
212
|
+
* was asked for: the patch landed, so the pod is on its way out and no call
|
|
213
|
+
* is admitted — but the suspend is not recorded as finished, because it did
|
|
214
|
+
* not finish. A later `suspend()` sends the patch again and waits again
|
|
215
|
+
* instead of returning on a mark this one left behind, and `resume()` still
|
|
216
|
+
* works.
|
|
217
|
+
*
|
|
218
|
+
* Recording it as suspended here is exactly the defect this class exists to
|
|
219
|
+
* make impossible. The guest is still running, still holding the block device
|
|
220
|
+
* open and still writing to it, and "suspended" is a promise that the disk is
|
|
221
|
+
* quiesced — so a handle that made that promise on a wait it lost would let
|
|
222
|
+
* the next caller resume, or delete, a workspace mid-write.
|
|
223
|
+
*/
|
|
224
|
+
export class KubernetesWorkspaceSuspendTimeoutError extends Error {
|
|
225
|
+
workspaceId;
|
|
226
|
+
sandboxName;
|
|
227
|
+
timeoutMs;
|
|
228
|
+
name = 'KubernetesWorkspaceSuspendTimeoutError';
|
|
229
|
+
constructor(workspaceId, sandboxName,
|
|
230
|
+
/** The readiness budget the wait was given, in milliseconds. */
|
|
231
|
+
timeoutMs) {
|
|
232
|
+
super(`kubernetes: workspace ${workspaceId} (Sandbox ${sandboxName}) was patched to operatingMode: Suspended but its pod had still not stopped ${timeoutMs}ms later, so the suspend is UNCONFIRMED and the disk cannot be promised quiesced. A guest whose PID 1 ignores SIGTERM rides out its terminationGracePeriodSeconds first; raise readyTimeoutMs, or make the image exit promptly on SIGTERM. Nothing was deleted and the disk is untouched: no call is admitted while the pod drains, suspend() sends the patch again and waits again, and resume() brings the workspace back.`);
|
|
233
|
+
this.workspaceId = workspaceId;
|
|
234
|
+
this.sandboxName = sandboxName;
|
|
235
|
+
this.timeoutMs = timeoutMs;
|
|
236
|
+
}
|
|
237
|
+
}
|
|
238
|
+
/** Prefix every workspace Sandbox's name carries. */
|
|
239
|
+
export const WORKSPACE_NAME_PREFIX = 'namzu-ws-';
|
|
240
|
+
/** DNS-1123 label: the Sandbox's name is also its Pod's and its Service's. */
|
|
241
|
+
const DNS_LABEL_MAX_LENGTH = 63;
|
|
242
|
+
const MAX_WORKSPACE_ID_LENGTH = DNS_LABEL_MAX_LENGTH - WORKSPACE_NAME_PREFIX.length;
|
|
243
|
+
const WORKSPACE_ID_PATTERN = /^[a-z0-9]([a-z0-9-]*[a-z0-9])?$/;
|
|
244
|
+
/**
|
|
245
|
+
* The Sandbox name for a workspace id.
|
|
246
|
+
*
|
|
247
|
+
* Deterministic on purpose: it is the only way a second host process, or the
|
|
248
|
+
* same one tomorrow, finds the workspace again. Which is also why an id that
|
|
249
|
+
* does not already fit a DNS-1123 label is REFUSED rather than lowercased,
|
|
250
|
+
* stripped or hashed: sanitising maps two ids onto one name, and two callers
|
|
251
|
+
* who believe they have separate workspaces would be sharing one disk.
|
|
252
|
+
* Hashing would fit every id and make the name unreadable in `kubectl get
|
|
253
|
+
* sandbox`, which is most of what the deterministic name is for.
|
|
254
|
+
*/
|
|
255
|
+
export function workspaceSandboxName(workspaceId) {
|
|
256
|
+
if (!WORKSPACE_ID_PATTERN.test(workspaceId) || workspaceId.length > MAX_WORKSPACE_ID_LENGTH) {
|
|
257
|
+
throw new Error(`kubernetes: workspace id ${JSON.stringify(workspaceId)} cannot name a Sandbox. It must be 1-${MAX_WORKSPACE_ID_LENGTH} characters of lowercase letters, digits and '-', starting and ending alphanumeric, so that ${WORKSPACE_NAME_PREFIX}<id> is a legal DNS-1123 label for the Sandbox, its Pod and its Service. The id is refused rather than sanitised because two ids that sanitise to one name would silently share one disk.`);
|
|
258
|
+
}
|
|
259
|
+
return `${WORKSPACE_NAME_PREFIX}${workspaceId}`;
|
|
260
|
+
}
|
|
261
|
+
function readArray(value) {
|
|
262
|
+
return Array.isArray(value) ? value : [];
|
|
263
|
+
}
|
|
264
|
+
function readNames(container, field) {
|
|
265
|
+
if (typeof container !== 'object' || container === null)
|
|
266
|
+
return [];
|
|
267
|
+
const entries = readArray(container[field]);
|
|
268
|
+
const names = [];
|
|
269
|
+
for (const entry of entries) {
|
|
270
|
+
if (typeof entry !== 'object' || entry === null)
|
|
271
|
+
continue;
|
|
272
|
+
const name = entry.name;
|
|
273
|
+
if (typeof name === 'string' && name !== '')
|
|
274
|
+
names.push(name);
|
|
275
|
+
}
|
|
276
|
+
return names;
|
|
277
|
+
}
|
|
278
|
+
/**
|
|
279
|
+
* Every name a pod template's containers claim through `volumeDevices` (a raw
|
|
280
|
+
* block device) or `volumeMounts` (a filesystem), across both container lists.
|
|
281
|
+
* The pod spec is carried opaquely everywhere else in this backend, so it is
|
|
282
|
+
* read defensively here rather than typed: an unexpected shape contributes
|
|
283
|
+
* nothing and lets the named refusal below do the talking.
|
|
284
|
+
*/
|
|
285
|
+
function readVolumeConsumers(podTemplate) {
|
|
286
|
+
const devices = new Set();
|
|
287
|
+
const mounts = new Set();
|
|
288
|
+
const spec = podTemplate?.spec;
|
|
289
|
+
for (const key of ['containers', 'initContainers']) {
|
|
290
|
+
for (const container of readArray(spec?.[key])) {
|
|
291
|
+
for (const name of readNames(container, 'volumeDevices'))
|
|
292
|
+
devices.add(name);
|
|
293
|
+
for (const name of readNames(container, 'volumeMounts'))
|
|
294
|
+
mounts.add(name);
|
|
295
|
+
}
|
|
296
|
+
}
|
|
297
|
+
return { devices, mounts };
|
|
298
|
+
}
|
|
299
|
+
/**
|
|
300
|
+
* Refuse anything that is not a block disk a container actually consumes as
|
|
301
|
+
* one. Every branch here describes a configuration that would work and then
|
|
302
|
+
* disappoint — see {@link KubernetesWorkspaceDiskError}.
|
|
303
|
+
*/
|
|
304
|
+
export function assertBlockModeWorkspaceDisk(source, podTemplate, volumeClaimTemplates) {
|
|
305
|
+
if (volumeClaimTemplates === undefined || volumeClaimTemplates.length === 0) {
|
|
306
|
+
throw new KubernetesWorkspaceDiskError(source, `kubernetes: ${source} declares no spec.volumeClaimTemplates, so a workspace built from it would have no disk and would lose everything on its first suspend — that is a task sandbox, not a workspace. Add a volumeClaimTemplates entry with spec.volumeMode: Block and consume it from the container's volumeDevices.`);
|
|
307
|
+
}
|
|
308
|
+
const { devices, mounts } = readVolumeConsumers(podTemplate);
|
|
309
|
+
for (const entry of volumeClaimTemplates) {
|
|
310
|
+
const name = entry.metadata?.name;
|
|
311
|
+
if (typeof name !== 'string' || name === '') {
|
|
312
|
+
throw new KubernetesWorkspaceDiskError(source, `kubernetes: ${source} declares a spec.volumeClaimTemplates entry with no metadata.name. The controller wires the disk by that name, StatefulSet style — it creates the PVC as <entry name>-<sandbox name> and matches the container's volumeDevices entry against it — so an unnamed entry reaches no container at all.`);
|
|
313
|
+
}
|
|
314
|
+
const volumeMode = entry.spec?.volumeMode;
|
|
315
|
+
if (volumeMode !== 'Block') {
|
|
316
|
+
throw new KubernetesWorkspaceDiskError(source, `kubernetes: ${source} declares volumeClaimTemplate ${JSON.stringify(name)} with volumeMode ${JSON.stringify(volumeMode ?? 'Filesystem (the API default)')}, but a persistent workspace's disk must be volumeMode: Block. Under a VM-isolating RuntimeClass a Filesystem PVC reaches the guest over a host/guest filesystem passthrough (virtio-fs), which pays a round trip per file operation — a dependency tree walk or a git status over a large checkout is several times slower, while nothing fails and no functional test can see it. A Block volume is a raw device the guest formats once and mounts as an ordinary local filesystem. Set spec.volumeMode: Block on this entry and consume it through the container's volumeDevices.`);
|
|
317
|
+
}
|
|
318
|
+
if (mounts.has(name)) {
|
|
319
|
+
throw new KubernetesWorkspaceDiskError(source, `kubernetes: ${source} consumes the Block volumeClaimTemplate ${JSON.stringify(name)} through a container's volumeMounts. A raw block device is claimed through volumeDevices (which gives the container a device node at devicePath); volumeMounts is the filesystem form and the kubelet will refuse the pod. Move the entry to volumeDevices and let the image's entrypoint format and mount the device.`);
|
|
320
|
+
}
|
|
321
|
+
if (!devices.has(name)) {
|
|
322
|
+
throw new KubernetesWorkspaceDiskError(source, `kubernetes: ${source} declares the Block volumeClaimTemplate ${JSON.stringify(name)} but no container claims it through volumeDevices, so the PVC is provisioned and attached to nothing. Add a volumeDevices entry naming ${JSON.stringify(name)} with the devicePath the image's entrypoint formats and mounts.`);
|
|
323
|
+
}
|
|
324
|
+
}
|
|
325
|
+
}
|
|
326
|
+
/**
|
|
327
|
+
* Refuse a standing Sandbox that was built from another `SandboxTemplate`, or
|
|
328
|
+
* that runs without the RuntimeClass this backend is configured for.
|
|
329
|
+
*
|
|
330
|
+
* Read off `spec.podTemplate` — the copy the controller actually runs a pod
|
|
331
|
+
* from — rather than off anything this process decided, because the question
|
|
332
|
+
* is what the POD is, not what the caller meant it to be.
|
|
333
|
+
*
|
|
334
|
+
* The template check is unconditional, `config.egress` set or not: the label
|
|
335
|
+
* is also how an operator reads which template an object came from, and an
|
|
336
|
+
* object whose label says one thing while the caller builds from another is a
|
|
337
|
+
* mix-up worth naming the first time it is seen rather than the first time a
|
|
338
|
+
* policy is switched on. See {@link KubernetesWorkspaceMismatchError}.
|
|
339
|
+
*/
|
|
340
|
+
export function assertAdoptedWorkspaceMatchesConfig(sandboxName, namespace, podTemplate, expected) {
|
|
341
|
+
const label = podTemplate?.metadata?.labels?.[SANDBOX_TEMPLATE_LABEL_KEY];
|
|
342
|
+
if (label !== expected.sandboxTemplateName) {
|
|
343
|
+
throw new KubernetesWorkspaceMismatchError(sandboxName, 'sandboxTemplateName', expected.sandboxTemplateName, label, `kubernetes: Sandbox ${sandboxName} in namespace ${namespace} already exists, and its spec.podTemplate carries ${SANDBOX_TEMPLATE_LABEL_KEY}: ${label === undefined ? '(absent)' : JSON.stringify(label)} rather than ${JSON.stringify(expected.sandboxTemplateName)} — it was built from a different SandboxTemplate, so it is not the workspace this call describes. That label is the one an egress NetworkPolicy's podSelector matches, so adopting this object would hand back a pod the policy verified for ${JSON.stringify(expected.sandboxTemplateName)} does not select, having reported the boundary as verified. Point this workspace at the template the object was built from, or delete the Sandbox — which takes its disk with it — and create it again, or choose another workspaceId.`);
|
|
344
|
+
}
|
|
345
|
+
if (expected.runtimeClassName === undefined)
|
|
346
|
+
return;
|
|
347
|
+
const declared = podTemplate?.spec?.runtimeClassName;
|
|
348
|
+
const actual = typeof declared === 'string' ? declared : undefined;
|
|
349
|
+
if (actual !== expected.runtimeClassName) {
|
|
350
|
+
throw new KubernetesWorkspaceMismatchError(sandboxName, 'runtimeClassName', expected.runtimeClassName, actual, `kubernetes: Sandbox ${sandboxName} in namespace ${namespace} already exists, and its spec.podTemplate.spec.runtimeClassName is ${actual === undefined ? "(absent — the cluster's default runtime)" : JSON.stringify(actual)} rather than the configured ${JSON.stringify(expected.runtimeClassName)}. Running it would put the workspace on that runtime — a shared kernel, if it is the default — while this backend registers itself as tier 'microvm', and nothing downstream would notice: the privilege probe reads /proc/self/status inside the guest and passes identically under a VM and under runc. A standing object's RuntimeClass is not something this backend rewrites underneath a disk it did not create, so the mismatch is named instead. Delete the Sandbox — which takes its disk with it — and create it again under ${JSON.stringify(expected.runtimeClassName)}, or drop runtimeClassName from the config if this object's runtime is the intended one.`);
|
|
351
|
+
}
|
|
352
|
+
}
|
|
353
|
+
/**
|
|
354
|
+
* Create the workspace, or take over the one that is already there.
|
|
355
|
+
*
|
|
356
|
+
* A second call with the same `workspaceId` ADOPTS rather than fails: the
|
|
357
|
+
* POST comes back 409 Conflict, and the object it collided with is this
|
|
358
|
+
* caller's own workspace from an earlier process. The deterministic name is
|
|
359
|
+
* only useful if coming back to it is the normal path. An adopted object is
|
|
360
|
+
* checked against the same block-disk rule a fresh one is, so a Sandbox
|
|
361
|
+
* standing under this name that is not a workspace is refused rather than
|
|
362
|
+
* used.
|
|
363
|
+
*
|
|
364
|
+
* Nothing here is ever deleted on failure. A create that gets as far as an
|
|
365
|
+
* existing object and then fails — a readiness timeout, a privilege probe
|
|
366
|
+
* refusal — SUSPENDS it and rethrows, because the object may be a workspace
|
|
367
|
+
* with a disk full of the caller's files and `deleteDisk` is not a decision
|
|
368
|
+
* a failure path gets to make. That holds even when THIS call POSTed the
|
|
369
|
+
* object and its disk is therefore empty: the 409 above means two processes
|
|
370
|
+
* can be coming up on one name at once, and the one that got the 201 deleting
|
|
371
|
+
* its "own" fresh object would take the disk of the one that adopted it. The
|
|
372
|
+
* cost is named rather than paid: a failed create can leave one suspended
|
|
373
|
+
* Sandbox standing, which the caller finds again under the same deterministic
|
|
374
|
+
* name and nothing reaps for them — see the docs page.
|
|
375
|
+
*/
|
|
376
|
+
export async function createKubernetesWorkspace(config, options) {
|
|
377
|
+
options.signal?.throwIfAborted();
|
|
378
|
+
const readiness = resolveKubernetesReadiness(config);
|
|
379
|
+
const namespace = config.namespace;
|
|
380
|
+
const name = workspaceSandboxName(options.workspaceId);
|
|
381
|
+
const templateName = options.sandboxTemplateName ?? config.sandboxTemplateName;
|
|
382
|
+
const agentPort = config.agentPort ?? DEFAULT_AGENT_PORT;
|
|
383
|
+
const client = createKubernetesClient(clientAccess(config));
|
|
384
|
+
// The same two egress steps `buildKubernetesBackend` runs for a task
|
|
385
|
+
// sandbox, repeated here because a workspace never goes through it. The
|
|
386
|
+
// refusal is synchronous and decided from the policy KIND alone; the
|
|
387
|
+
// verification is a GET of the object an operator was supposed to apply,
|
|
388
|
+
// and neither ever creates or repairs anything — see `egress-policy.ts`.
|
|
389
|
+
//
|
|
390
|
+
// Not optional on this path, and not a copy-paste: a long-lived workspace
|
|
391
|
+
// is the sandbox most likely to be pointed at a network, the NetworkPolicy
|
|
392
|
+
// rather than the bind token is the boundary on its agent port, and a
|
|
393
|
+
// config object refused by one entry point and silently ignored by the
|
|
394
|
+
// other is the exact silent downgrade this translation exists to prevent.
|
|
395
|
+
//
|
|
396
|
+
// Verified against the template this workspace is actually built from:
|
|
397
|
+
// `buildSandboxBody` stamps the pod with THAT template's label and the
|
|
398
|
+
// policy's podSelector matches that label, so a workspace built from a
|
|
399
|
+
// separate workspace template needs its own policy — the task template's
|
|
400
|
+
// does not select it. Deliberately not memoized the way the backend's
|
|
401
|
+
// once-per-backend check is: creating a workspace is a rare, explicit act
|
|
402
|
+
// with nothing to amortise, and re-checking costs one GET.
|
|
403
|
+
if (config.egress) {
|
|
404
|
+
assertEgressPolicyIsEnforceable(config.egress.policy, config.egress.engine ?? 'core');
|
|
405
|
+
await verifyEgressPolicyConfigured(client, namespace, templateName, config.egress, options.signal);
|
|
406
|
+
}
|
|
407
|
+
// Read and validate BEFORE anything is created, so a template that cannot
|
|
408
|
+
// carry a workspace fails with nothing to clean up.
|
|
409
|
+
const template = await readSandboxTemplate(client, namespace, templateName, options.signal);
|
|
410
|
+
assertBlockModeWorkspaceDisk(`SandboxTemplate ${templateName} in namespace ${namespace}`, template.podTemplate, template.volumeClaimTemplates);
|
|
411
|
+
let created = false;
|
|
412
|
+
try {
|
|
413
|
+
await client.request('POST', sandboxCollectionPath(namespace),
|
|
414
|
+
// No shutdownTime: a workspace carries no expiry — see the module
|
|
415
|
+
// comment.
|
|
416
|
+
buildSandboxBody({
|
|
417
|
+
namespace,
|
|
418
|
+
name,
|
|
419
|
+
template,
|
|
420
|
+
sandboxTemplateName: templateName,
|
|
421
|
+
...(config.runtimeClassName !== undefined
|
|
422
|
+
? { runtimeClassName: config.runtimeClassName }
|
|
423
|
+
: {}),
|
|
424
|
+
}), options.signal);
|
|
425
|
+
created = true;
|
|
426
|
+
}
|
|
427
|
+
catch (err) {
|
|
428
|
+
if (!(err instanceof KubernetesConflictError))
|
|
429
|
+
throw err;
|
|
430
|
+
await adoptExistingWorkspace(client, namespace, name, {
|
|
431
|
+
sandboxTemplateName: templateName,
|
|
432
|
+
...(config.runtimeClassName !== undefined
|
|
433
|
+
? { runtimeClassName: config.runtimeClassName }
|
|
434
|
+
: {}),
|
|
435
|
+
}, options.signal);
|
|
436
|
+
}
|
|
437
|
+
return await openWorkspaceHandle({
|
|
438
|
+
client,
|
|
439
|
+
namespace,
|
|
440
|
+
name,
|
|
441
|
+
workspaceId: options.workspaceId,
|
|
442
|
+
rootDir: options.workingDirectory,
|
|
443
|
+
agentPort,
|
|
444
|
+
readiness,
|
|
445
|
+
created,
|
|
446
|
+
...(options.signal !== undefined ? { signal: options.signal } : {}),
|
|
447
|
+
});
|
|
448
|
+
}
|
|
449
|
+
/**
|
|
450
|
+
* Take over a Sandbox that already stands under this workspace's name: check
|
|
451
|
+
* that it really is a block-disk workspace AND that it is the one this
|
|
452
|
+
* configuration describes, then wake it if it is asleep.
|
|
453
|
+
*
|
|
454
|
+
* Everything checked here is checked against the object, because on this path
|
|
455
|
+
* the object is not the one this call built. A create POSTs its own body and
|
|
456
|
+
* knows what is in it; an adopt is handed a pod somebody else's process, or
|
|
457
|
+
* last month's configuration, decided the shape of. The two silent losses are
|
|
458
|
+
* the template label and the RuntimeClass — see
|
|
459
|
+
* {@link KubernetesWorkspaceMismatchError}.
|
|
460
|
+
*
|
|
461
|
+
* The refusals all happen BEFORE the resume patch: an object this call will
|
|
462
|
+
* not use is not woken up on the way to being rejected.
|
|
463
|
+
*/
|
|
464
|
+
async function adoptExistingWorkspace(client, namespace, name, expected, signal) {
|
|
465
|
+
const existing = await client.request('GET', sandboxPath(namespace, name), undefined, signal);
|
|
466
|
+
assertBlockModeWorkspaceDisk(`Sandbox ${name} in namespace ${namespace}`, existing?.spec?.podTemplate, existing?.spec?.volumeClaimTemplates);
|
|
467
|
+
assertAdoptedWorkspaceMatchesConfig(name, namespace, existing?.spec?.podTemplate, expected);
|
|
468
|
+
if (existing?.spec?.operatingMode === 'Suspended') {
|
|
469
|
+
await client.request('PATCH', sandboxPath(namespace, name), RESUME_PATCH, signal);
|
|
470
|
+
}
|
|
471
|
+
}
|
|
472
|
+
/** The two merge patches this module sends, and the only two. */
|
|
473
|
+
const SUSPEND_PATCH = { spec: { operatingMode: 'Suspended' } };
|
|
474
|
+
const RESUME_PATCH = { spec: { operatingMode: 'Running' } };
|
|
475
|
+
async function openWorkspaceHandle(options) {
|
|
476
|
+
const { client, namespace, name, workspaceId, readiness } = options;
|
|
477
|
+
const id = name;
|
|
478
|
+
let state = 'running';
|
|
479
|
+
let session;
|
|
480
|
+
/**
|
|
481
|
+
* The uid of the pod `session` is bound to — the agent's token, and the
|
|
482
|
+
* only way a resume can tell the pod it is waiting for from the one it is
|
|
483
|
+
* replacing. Read once per session and never carried across one.
|
|
484
|
+
*/
|
|
485
|
+
let podUid;
|
|
486
|
+
/**
|
|
487
|
+
* The uid of the pod a suspend patch that LANDED took away, cleared once a
|
|
488
|
+
* resume has bound its replacement.
|
|
489
|
+
*
|
|
490
|
+
* This — not `podUid` — is what a resume excludes; see
|
|
491
|
+
* {@link acquireBoundPod}. It is set from the PATCH rather than from the
|
|
492
|
+
* wait that follows it, so a suspend whose pod outlived `readyTimeoutMs`
|
|
493
|
+
* excludes that pod too: the controller was asked to delete it either
|
|
494
|
+
* way, and a resume arriving while it drains is handed exactly that pod.
|
|
495
|
+
* Every path that sends that patch records it here: `suspend()`, the
|
|
496
|
+
* retirement of a pod that stopped answering, and the cleanup after a
|
|
497
|
+
* failed create or resume — which swallows its own failure, and so
|
|
498
|
+
* records only when the request actually came back. A pod nobody asked
|
|
499
|
+
* the controller to remove has no replacement to wait for, and excluding
|
|
500
|
+
* it would time out a resume whose workspace was perfectly usable.
|
|
501
|
+
*/
|
|
502
|
+
let retiredPodUid;
|
|
503
|
+
const terminals = new Set();
|
|
504
|
+
/**
|
|
505
|
+
* Lifecycle transitions run one at a time. Two of them racing would
|
|
506
|
+
* interleave a suspend patch with a resume's readiness poll and settle on
|
|
507
|
+
* whichever finished last, which is how a workspace ends up marked running
|
|
508
|
+
* with no pod.
|
|
509
|
+
*/
|
|
510
|
+
let queue = Promise.resolve();
|
|
511
|
+
const serialise = (run) => {
|
|
512
|
+
const next = queue.then(run, run);
|
|
513
|
+
queue = next.then(() => undefined, () => undefined);
|
|
514
|
+
return next;
|
|
515
|
+
};
|
|
516
|
+
/**
|
|
517
|
+
* The suspend and the delete currently in flight, so a second caller
|
|
518
|
+
* AWAITS the one that is running rather than queueing another behind it.
|
|
519
|
+
*
|
|
520
|
+
* Serialising is not the same thing and does not cover this. Now that a
|
|
521
|
+
* terminal state is committed only after its request lands, two
|
|
522
|
+
* concurrent `destroy({ deleteDisk: true })` calls would BOTH be admitted
|
|
523
|
+
* under the queue alone — the second finding nothing yet marked and
|
|
524
|
+
* sending a second DELETE — and two concurrent `suspend()` calls would
|
|
525
|
+
* patch and wait twice over. So the single-flight check sits OUTSIDE the
|
|
526
|
+
* queue, where a caller arriving mid-transition can still see it.
|
|
527
|
+
*
|
|
528
|
+
* The first caller's `signal` is the one the shared request runs under;
|
|
529
|
+
* a second caller's is not consulted, which is what sharing means.
|
|
530
|
+
*/
|
|
531
|
+
let pendingSuspend;
|
|
532
|
+
let pendingDelete;
|
|
533
|
+
const deleteSandbox = async (signal) => {
|
|
534
|
+
try {
|
|
535
|
+
await client.request('DELETE', sandboxPath(namespace, name), undefined, signal);
|
|
536
|
+
}
|
|
537
|
+
catch (err) {
|
|
538
|
+
// Already gone is the state DELETE was asking for.
|
|
539
|
+
if (!(err instanceof KubernetesAlreadyGoneError))
|
|
540
|
+
throw err;
|
|
541
|
+
}
|
|
542
|
+
};
|
|
543
|
+
const readBinding = async (pollSignal) => bindingFromSandbox(await client.request('GET', sandboxPath(namespace, name), undefined, pollSignal));
|
|
544
|
+
/**
|
|
545
|
+
* Wait until the Sandbox is Ready AND the pod behind it is a pod this
|
|
546
|
+
* handle is allowed to bind to, then read that pod's uid.
|
|
547
|
+
*
|
|
548
|
+
* On create there is nothing to exclude and this is one poll plus one
|
|
549
|
+
* read, exactly as it was. On RESUME `previousPodUid` is the pod the
|
|
550
|
+
* workspace was suspended from, and excluding it is the whole point:
|
|
551
|
+
* `Ready` is not a transition signal. The controller leaves the condition
|
|
552
|
+
* True across a resume — upstream says the same of `Suspended` in the
|
|
553
|
+
* other direction — so the very first poll after the Running patch can
|
|
554
|
+
* come back Ready while the only pod under that name is still the old
|
|
555
|
+
* one, not yet deleted and not yet carrying a `deletionTimestamp`. Its
|
|
556
|
+
* uid then reads as perfectly live, and the handle binds a token the new
|
|
557
|
+
* agent will refuse, reported as a flat `unauthorized` with nothing
|
|
558
|
+
* pointing at the race. So the uid is polled, under the SAME deadline as
|
|
559
|
+
* everything else on this path, until it is a different pod's.
|
|
560
|
+
*
|
|
561
|
+
* A failed read is fatal on create and merely "not yet" on resume: for a
|
|
562
|
+
* stretch of every resume there is no live pod at all, and
|
|
563
|
+
* {@link readPodBindToken} answers that by throwing rather than returning
|
|
564
|
+
* undefined. The last such error is carried onto the timeout as `cause`,
|
|
565
|
+
* so a resume that never found a pod still says what it kept seeing.
|
|
566
|
+
*/
|
|
567
|
+
const acquireBoundPod = async (deadline, previousPodUid) => {
|
|
568
|
+
const label = `workspace ${workspaceId} (Sandbox ${name})`;
|
|
569
|
+
let lastError;
|
|
570
|
+
while (deadline.remainingMs() > 0) {
|
|
571
|
+
const binding = await pollForBinding(readBinding, deadline, readiness, label);
|
|
572
|
+
let token;
|
|
573
|
+
try {
|
|
574
|
+
token = await deadline.run(async (tokenSignal) => await readPodBindToken(client, namespace, binding, tokenSignal));
|
|
575
|
+
}
|
|
576
|
+
catch (err) {
|
|
577
|
+
if (err instanceof OperationDeadlineExpired)
|
|
578
|
+
break;
|
|
579
|
+
if (previousPodUid === undefined)
|
|
580
|
+
throw err;
|
|
581
|
+
lastError = err;
|
|
582
|
+
token = undefined;
|
|
583
|
+
}
|
|
584
|
+
if (token !== undefined && token !== previousPodUid)
|
|
585
|
+
return { binding, token };
|
|
586
|
+
try {
|
|
587
|
+
await deadline.delay(readiness.pollIntervalMs);
|
|
588
|
+
}
|
|
589
|
+
catch (err) {
|
|
590
|
+
if (err instanceof OperationDeadlineExpired)
|
|
591
|
+
break;
|
|
592
|
+
throw err;
|
|
593
|
+
}
|
|
594
|
+
}
|
|
595
|
+
throw new Error(`kubernetes: workspace ${workspaceId} (Sandbox ${name}) was patched back to operatingMode: Running, but ${readiness.timeoutMs}ms later the only pod behind it was still the one it was suspended from (uid ${previousPodUid}). A resumed pod keeps the sandbox's name and gets a new uid, and that uid is the agent's bind token, so binding to the old pod would present a token the new agent refuses. The Ready condition cannot be waited on instead — the controller leaves it standing across the transition. Raise readyTimeoutMs, or look at why the controller has not replaced the pod.`, lastError !== undefined ? { cause: lastError } : undefined);
|
|
596
|
+
};
|
|
597
|
+
/**
|
|
598
|
+
* Bring up one pod's worth of state: wait for a pod that is not the one
|
|
599
|
+
* being replaced, read its bind token, re-resolve the address, build the
|
|
600
|
+
* transport and prove the guest is deprivileged. Called on create and on
|
|
601
|
+
* every resume, with nothing carried over between them.
|
|
602
|
+
*/
|
|
603
|
+
const startSession = async (label, signal, previousPodUid) => {
|
|
604
|
+
const deadline = new OperationDeadline(readiness.timeoutMs, `kubernetes workspace ${name} ${label}`, signal);
|
|
605
|
+
// Both of these are re-read rather than remembered: a resumed pod keeps
|
|
606
|
+
// the name and changes the uid and the IP, so a handle that reused
|
|
607
|
+
// either would present a token the new agent refuses, at an address
|
|
608
|
+
// whose pod is being deleted.
|
|
609
|
+
const { binding, token } = await acquireBoundPod(deadline, previousPodUid);
|
|
610
|
+
// Recorded before the probe, not after: a probe that refuses suspends
|
|
611
|
+
// this pod, and the resume that follows has to know which pod it is
|
|
612
|
+
// waiting to see replaced.
|
|
613
|
+
podUid = token;
|
|
614
|
+
const address = resolveAgentAddress(binding, options.agentPort, token);
|
|
615
|
+
// A box rather than a `let`, so the callback below can name the handle
|
|
616
|
+
// it belongs to before that handle exists. Nothing can call it in
|
|
617
|
+
// between: `release` is reachable only THROUGH the handle.
|
|
618
|
+
const own = {};
|
|
619
|
+
const inner = buildKubernetesSandbox({
|
|
620
|
+
name,
|
|
621
|
+
rootDir: options.rootDir,
|
|
622
|
+
transport: new KubernetesAgentTransport(address),
|
|
623
|
+
// Deliberately NOT `deleteSandbox` — see {@link retireSession}. On
|
|
624
|
+
// the task path `release` is a DELETE because the object is
|
|
625
|
+
// disposable; here the same callback would erase the caller's disk
|
|
626
|
+
// from a path nobody asked to erase anything.
|
|
627
|
+
release: async (releaseSignal) => {
|
|
628
|
+
await retireSession(own.handle, releaseSignal);
|
|
629
|
+
},
|
|
630
|
+
// No `renew`, and so no lease loop: a workspace carries no expiry.
|
|
631
|
+
});
|
|
632
|
+
own.handle = inner;
|
|
633
|
+
// The same probe every task acquire runs, on every resume as well as on
|
|
634
|
+
// create — a resumed pod is a new pod, from a possibly re-pulled image,
|
|
635
|
+
// and "it was deprivileged last week" is not a check.
|
|
636
|
+
await probeSandboxPrivileges(inner, name, resolveProbeTimeoutMs(readiness.timeoutMs), signal);
|
|
637
|
+
return inner;
|
|
638
|
+
};
|
|
639
|
+
/** Kill and await every terminal this handle returned. */
|
|
640
|
+
const reapTerminals = async () => {
|
|
641
|
+
const open = [...terminals];
|
|
642
|
+
for (const terminal of open)
|
|
643
|
+
terminal.kill('SIGKILL');
|
|
644
|
+
await Promise.allSettled(open.map((terminal) => terminal.exited));
|
|
645
|
+
terminals.clear();
|
|
646
|
+
};
|
|
647
|
+
/**
|
|
648
|
+
* Retire one session's pod WITHOUT deleting anything.
|
|
649
|
+
*
|
|
650
|
+
* This is the inner handle's `release` on a workspace, and the difference
|
|
651
|
+
* from the task path is the entire reason that callback is a parameter
|
|
652
|
+
* rather than a DELETE both paths share. `buildKubernetesSandbox` calls
|
|
653
|
+
* `release` on its own initiative: an execution whose cancellation the
|
|
654
|
+
* guest could not confirm (`RemoteCancellationUnknownError` — a wedged
|
|
655
|
+
* agent, a partitioned pod) leaves a command of unknown state in that
|
|
656
|
+
* pod, so the pod stops being reusable and the handle retires it. For a
|
|
657
|
+
* task sandbox retiring IS deleting, because the object is disposable and
|
|
658
|
+
* its disk is scratch. Here it is not: the DELETE cascades to the PVC, and
|
|
659
|
+
* a cancel that went unconfirmed for eight seconds would take a month of
|
|
660
|
+
* the caller's files with it. `deleteDisk: true` is the only thing in this
|
|
661
|
+
* module that removes a disk, and a failure path is not allowed to become
|
|
662
|
+
* a second one — that is the invariant the whole file is built around.
|
|
663
|
+
*
|
|
664
|
+
* So the pod is retired the way `suspend()` retires one, with the same
|
|
665
|
+
* `operatingMode: Suspended` patch, and the workspace is left
|
|
666
|
+
* `suspending`: nothing is admitted, the disk is untouched, and `resume()`
|
|
667
|
+
* brings up a fresh pod. A patch that FAILS is not swallowed — it travels
|
|
668
|
+
* back out through `retire()` as `retirement.accepted === false` on the
|
|
669
|
+
* error the caller is already receiving, which is what that observation
|
|
670
|
+
* exists to say.
|
|
671
|
+
*
|
|
672
|
+
* `retiring` is the handle the callback was built for. When it is not the
|
|
673
|
+
* current session there is nothing to retire and this is a no-op: a resume
|
|
674
|
+
* has already replaced it, or `deleteNow` dropped it on the way to a
|
|
675
|
+
* DELETE — which must not be preceded by a suspend patch, and says so by
|
|
676
|
+
* dropping it.
|
|
677
|
+
*
|
|
678
|
+
* It never touches the transition queue, and must not: `retire()` is
|
|
679
|
+
* awaited inside the failing `exec()`, and that exec can be the privilege
|
|
680
|
+
* probe of the resume currently HOLDING the queue.
|
|
681
|
+
*/
|
|
682
|
+
const retireSession = async (retiring, signal) => {
|
|
683
|
+
if (retiring === undefined || retiring !== session)
|
|
684
|
+
return;
|
|
685
|
+
session = undefined;
|
|
686
|
+
// Some transition already owns this pod — the suspend that is patching
|
|
687
|
+
// it away, or a create/resume cleanup — and commits its own state when
|
|
688
|
+
// its own request settles. Dropping the session is all there is to do.
|
|
689
|
+
if (state !== 'running')
|
|
690
|
+
return;
|
|
691
|
+
state = 'suspending';
|
|
692
|
+
// Killed but not waited on: the frames go to a pod whose agent has
|
|
693
|
+
// already stopped answering, and whether they are acknowledged must
|
|
694
|
+
// not decide whether the retirement is reported accepted. Every
|
|
695
|
+
// session dies with the pod either way; this is the ownership contract
|
|
696
|
+
// being honoured, not a condition of the patch below. The `catch` is
|
|
697
|
+
// not decoration — a detached chain that rejects with no handler takes
|
|
698
|
+
// the host process down with it.
|
|
699
|
+
void reapTerminals().catch(() => undefined);
|
|
700
|
+
await client.request('PATCH', sandboxPath(namespace, name), SUSPEND_PATCH, signal);
|
|
701
|
+
// The patch landed, so the controller is taking this pod away and the
|
|
702
|
+
// next resume must see it replaced rather than bind it.
|
|
703
|
+
retiredPodUid = podUid;
|
|
704
|
+
};
|
|
705
|
+
/**
|
|
706
|
+
* True once the pod has actually stopped. The POD is asked, and nothing
|
|
707
|
+
* else is consulted or believed.
|
|
708
|
+
*
|
|
709
|
+
* The cheap-looking alternative — the Sandbox's own `Suspended` condition
|
|
710
|
+
* — is unusable, and upstream says so itself: "the controller does not
|
|
711
|
+
* currently remove this condition when the Sandbox is resumed, so a stale
|
|
712
|
+
* Suspended condition may linger after operatingMode returns to Running.
|
|
713
|
+
* Consumers should treat Ready as the authoritative signal and not infer
|
|
714
|
+
* the live operating state from the mere presence of this condition"
|
|
715
|
+
* (`sandbox_types.go`). Reading it would make the second and every later
|
|
716
|
+
* suspend of the same workspace return immediately, on a True left behind
|
|
717
|
+
* by the previous one, while the guest was still running and still
|
|
718
|
+
* writing to the caller's disk — and a `resume()` issued straight after
|
|
719
|
+
* such a false suspend could bind to the pod that is about to be deleted.
|
|
720
|
+
*
|
|
721
|
+
* A `deletionTimestamp` is not the answer either: it is set the moment the
|
|
722
|
+
* DELETE is accepted, and the container goes on running until it exits or
|
|
723
|
+
* `terminationGracePeriodSeconds` expires. Gone (404) or stopped
|
|
724
|
+
* ({@link isPodStopped}) — those are the only two states that mean the
|
|
725
|
+
* disk is quiesced.
|
|
726
|
+
*/
|
|
727
|
+
const isPodRetired = async (signal) => {
|
|
728
|
+
try {
|
|
729
|
+
const pod = await client.request('GET', podPath(namespace, name), undefined, signal);
|
|
730
|
+
return isPodStopped(pod);
|
|
731
|
+
}
|
|
732
|
+
catch (err) {
|
|
733
|
+
if (err instanceof KubernetesAlreadyGoneError)
|
|
734
|
+
return true;
|
|
735
|
+
throw err;
|
|
736
|
+
}
|
|
737
|
+
};
|
|
738
|
+
const awaitPodRetired = async (signal) => {
|
|
739
|
+
const deadline = new OperationDeadline(readiness.timeoutMs, `kubernetes workspace ${name} suspend`, signal);
|
|
740
|
+
while (deadline.remainingMs() > 0) {
|
|
741
|
+
try {
|
|
742
|
+
if (await deadline.run(isPodRetired))
|
|
743
|
+
return;
|
|
744
|
+
await deadline.delay(readiness.pollIntervalMs);
|
|
745
|
+
}
|
|
746
|
+
catch (err) {
|
|
747
|
+
if (err instanceof OperationDeadlineExpired)
|
|
748
|
+
break;
|
|
749
|
+
throw err;
|
|
750
|
+
}
|
|
751
|
+
}
|
|
752
|
+
throw new KubernetesWorkspaceSuspendTimeoutError(workspaceId, name, readiness.timeoutMs);
|
|
753
|
+
};
|
|
754
|
+
/**
|
|
755
|
+
* The suspend, minus the queue — every caller here is already inside it.
|
|
756
|
+
*
|
|
757
|
+
* `suspending` is entered before the patch and `suspended` only after the
|
|
758
|
+
* pod is observed stopped, so the two failures each leave the state that
|
|
759
|
+
* is true: a patch the API server refused leaves the workspace exactly as
|
|
760
|
+
* it was, still serving calls, and a wait that ran out leaves it refusing
|
|
761
|
+
* them with the transition still unfinished. Neither can be returned from
|
|
762
|
+
* by a later `suspend()` as though it had worked.
|
|
763
|
+
*/
|
|
764
|
+
const suspendNow = async (signal) => {
|
|
765
|
+
if (state === 'deleted')
|
|
766
|
+
throw new KubernetesSandboxDestroyedError('suspend', name);
|
|
767
|
+
// `suspending` deliberately falls through: the patch is re-sent and
|
|
768
|
+
// the pod waited for again. Only a CONFIRMED suspend returns here.
|
|
769
|
+
if (state === 'suspended')
|
|
770
|
+
return;
|
|
771
|
+
const before = state;
|
|
772
|
+
// A terminal owns an interactive process tree in a pod that is about
|
|
773
|
+
// to be taken away, so it is stopped first — and stays stopped even if
|
|
774
|
+
// the patch below fails. `suspend()` is a declaration that nobody is
|
|
775
|
+
// using this workspace; killing the sessions that say otherwise is the
|
|
776
|
+
// point of it rather than a cost of it.
|
|
777
|
+
await reapTerminals();
|
|
778
|
+
// From here on nothing new is admitted: the pod is going away, and a
|
|
779
|
+
// call let through would dial an address that still resolves — the
|
|
780
|
+
// Service outlives the pod — and hang on a connect timeout naming
|
|
781
|
+
// nothing. This is NOT the terminal state; a patch that fails puts it
|
|
782
|
+
// straight back.
|
|
783
|
+
state = 'suspending';
|
|
784
|
+
try {
|
|
785
|
+
await client.request('PATCH', sandboxPath(namespace, name), SUSPEND_PATCH, signal);
|
|
786
|
+
}
|
|
787
|
+
catch (err) {
|
|
788
|
+
// Nothing was changed on the cluster, so nothing is changed here:
|
|
789
|
+
// the pod is still running and this handle can still serve it.
|
|
790
|
+
// Marking it suspended would be the defect — every later
|
|
791
|
+
// suspend() and destroy() would return on that mark without ever
|
|
792
|
+
// re-sending the patch, and the pod would run until somebody
|
|
793
|
+
// noticed the bill.
|
|
794
|
+
state = before;
|
|
795
|
+
throw err;
|
|
796
|
+
}
|
|
797
|
+
// The patch landed, so this pod is the controller's to remove and the
|
|
798
|
+
// next resume must see it replaced rather than bind it — whether or
|
|
799
|
+
// not the wait below is still around when it goes.
|
|
800
|
+
retiredPodUid = podUid;
|
|
801
|
+
session = undefined;
|
|
802
|
+
await awaitPodRetired(signal);
|
|
803
|
+
state = 'suspended';
|
|
804
|
+
};
|
|
805
|
+
const resumeNow = async (signal) => {
|
|
806
|
+
if (state === 'deleted')
|
|
807
|
+
throw new KubernetesSandboxDestroyedError('resume', name);
|
|
808
|
+
if (state === 'running')
|
|
809
|
+
return;
|
|
810
|
+
// The pod a landed suspend patch took away is one this resume must see
|
|
811
|
+
// replaced rather than bound — see `acquireBoundPod` and
|
|
812
|
+
// `retiredPodUid`. Where there is none to exclude this is one poll
|
|
813
|
+
// plus one read, exactly as it is on create.
|
|
814
|
+
const replacing = retiredPodUid;
|
|
815
|
+
await client.request('PATCH', sandboxPath(namespace, name), RESUME_PATCH, signal);
|
|
816
|
+
session = await startSessionOrSuspend('resume', signal, replacing);
|
|
817
|
+
state = 'running';
|
|
818
|
+
// Bound, probed and serving: the pod that was excluded is one no
|
|
819
|
+
// answer can name any more, and the next suspend records its own.
|
|
820
|
+
retiredPodUid = undefined;
|
|
821
|
+
};
|
|
822
|
+
/**
|
|
823
|
+
* Suspend, sharing one transition with any caller already inside it.
|
|
824
|
+
*
|
|
825
|
+
* The single-flight slot is taken before the queue so that a second
|
|
826
|
+
* `suspend()` — or the `destroy()` that is a suspend — awaits this one
|
|
827
|
+
* instead of patching and waiting all over again once it finishes. It is
|
|
828
|
+
* released inside the run, before the promise handed to callers settles,
|
|
829
|
+
* so a caller that awaits and then suspends again gets a fresh attempt.
|
|
830
|
+
*/
|
|
831
|
+
const suspendShared = (signal) => {
|
|
832
|
+
pendingSuspend ??= serialise(async () => {
|
|
833
|
+
try {
|
|
834
|
+
await suspendNow(signal);
|
|
835
|
+
}
|
|
836
|
+
finally {
|
|
837
|
+
pendingSuspend = undefined;
|
|
838
|
+
}
|
|
839
|
+
});
|
|
840
|
+
return pendingSuspend;
|
|
841
|
+
};
|
|
842
|
+
/**
|
|
843
|
+
* `destroy()` in its default shape: the suspend above, plus the one thing
|
|
844
|
+
* a destroy owes a caller that a suspend does not — idempotence over a
|
|
845
|
+
* workspace somebody already deleted.
|
|
846
|
+
*
|
|
847
|
+
* `suspendNow` refuses a deleted workspace, and should: asking to suspend
|
|
848
|
+
* an object that no longer exists is a mistake worth hearing about. But
|
|
849
|
+
* `destroy()` is the verb a `finally` block calls, and a body that ends
|
|
850
|
+
* with an explicit `destroy({ deleteDisk: true })` inside such a block
|
|
851
|
+
* must not then be handed a `KubernetesSandboxDestroyedError` naming an
|
|
852
|
+
* operation the caller never typed. What a plain `destroy()` asks for has
|
|
853
|
+
* happened, more thoroughly than it asked.
|
|
854
|
+
*
|
|
855
|
+
* The state is read on both sides of the flight on purpose. Before,
|
|
856
|
+
* for the ordinary sequential case; after, because the delete can land
|
|
857
|
+
* while this call waits its turn — a `destroy()` racing a `destroy({
|
|
858
|
+
* deleteDisk: true })` the queue admitted first must be a no-op in that
|
|
859
|
+
* order too, which is the order it is most likely to be written in.
|
|
860
|
+
*/
|
|
861
|
+
const destroyBySuspending = async (signal) => {
|
|
862
|
+
// Read through a call on both sides. `state` is assigned from other
|
|
863
|
+
// closures, which the checker cannot see, so it takes the first
|
|
864
|
+
// comparison as narrowing the second out of existence — and the second
|
|
865
|
+
// is the one that matters, because it is the one reading a delete that
|
|
866
|
+
// landed while this call was queued.
|
|
867
|
+
const gone = () => state === 'deleted';
|
|
868
|
+
if (gone())
|
|
869
|
+
return;
|
|
870
|
+
try {
|
|
871
|
+
await suspendShared(signal);
|
|
872
|
+
}
|
|
873
|
+
catch (err) {
|
|
874
|
+
if (gone() && err instanceof KubernetesSandboxDestroyedError)
|
|
875
|
+
return;
|
|
876
|
+
throw err;
|
|
877
|
+
}
|
|
878
|
+
};
|
|
879
|
+
/**
|
|
880
|
+
* DELETE the Sandbox, and with it the Pod, the Service and the PVC.
|
|
881
|
+
*
|
|
882
|
+
* The terminal state is committed only once the DELETE has resolved —
|
|
883
|
+
* an object already gone counts, that being the state DELETE was asking
|
|
884
|
+
* for. A DELETE that FAILS leaves the state alone and rethrows, so the
|
|
885
|
+
* caller can retry and the next attempt sends the request again. The
|
|
886
|
+
* inverse — marking `deleted` first — is how an object outlives every
|
|
887
|
+
* handle that could have removed it: the failure is thrown once, and every
|
|
888
|
+
* later `destroy()` resolves immediately on a state nothing established.
|
|
889
|
+
*/
|
|
890
|
+
const deleteNow = async (destroyOptions) => {
|
|
891
|
+
if (state === 'deleted')
|
|
892
|
+
return;
|
|
893
|
+
await reapTerminals();
|
|
894
|
+
const current = session;
|
|
895
|
+
// Dropped BEFORE the handle is torn down, because tearing it down runs
|
|
896
|
+
// its `release` — which on a workspace is {@link retireSession}, a
|
|
897
|
+
// SUSPEND patch. A delete does not want one on the way: the object is
|
|
898
|
+
// going away whole. `retireSession` reads exactly this to know it.
|
|
899
|
+
session = undefined;
|
|
900
|
+
// And the state moves with it. The pod is being taken away however the
|
|
901
|
+
// DELETE goes, and the handle that served it is now torn down, so a
|
|
902
|
+
// DELETE that fails leaves a workspace that admits nothing, says so,
|
|
903
|
+
// and can be resumed or deleted again — rather than one still calling
|
|
904
|
+
// itself `running` with no session behind it, which nothing but
|
|
905
|
+
// another delete could ever get out of.
|
|
906
|
+
if (state === 'running')
|
|
907
|
+
state = 'suspending';
|
|
908
|
+
// Through the inner handle when there is one, so its own terminal
|
|
909
|
+
// reaping and lifecycle bookkeeping run. The DELETE itself is sent
|
|
910
|
+
// here either way, exactly once, and it is retryable: the terminal
|
|
911
|
+
// state below is committed only once it resolves.
|
|
912
|
+
if (current)
|
|
913
|
+
await current.destroy(destroyOptions);
|
|
914
|
+
await deleteSandbox(destroyOptions?.signal);
|
|
915
|
+
state = 'deleted';
|
|
916
|
+
};
|
|
917
|
+
/** Delete, sharing one DELETE with any caller already inside it. */
|
|
918
|
+
const deleteShared = (destroyOptions) => {
|
|
919
|
+
pendingDelete ??= serialise(async () => {
|
|
920
|
+
try {
|
|
921
|
+
await deleteNow(destroyOptions);
|
|
922
|
+
}
|
|
923
|
+
finally {
|
|
924
|
+
pendingDelete = undefined;
|
|
925
|
+
}
|
|
926
|
+
});
|
|
927
|
+
return pendingDelete;
|
|
928
|
+
};
|
|
929
|
+
/**
|
|
930
|
+
* Bring a session up, and put the workspace back to sleep if that fails.
|
|
931
|
+
*
|
|
932
|
+
* Suspend rather than delete, always: the failure might be a probe refusal
|
|
933
|
+
* on a workspace whose disk holds a month of a caller's work, and no
|
|
934
|
+
* failure path in this module is allowed to make that decision. The cost
|
|
935
|
+
* of being wrong the other way is one suspended Sandbox left standing,
|
|
936
|
+
* which the caller finds again under the same deterministic name.
|
|
937
|
+
*/
|
|
938
|
+
const startSessionOrSuspend = async (label, signal, previousPodUid) => {
|
|
939
|
+
try {
|
|
940
|
+
return await startSession(label, signal, previousPodUid);
|
|
941
|
+
}
|
|
942
|
+
catch (err) {
|
|
943
|
+
// `suspending`, not `suspended`: `runFailureCleanup` swallows its
|
|
944
|
+
// own failures so that the primary error stays primary, which
|
|
945
|
+
// means this patch may not have landed and this pod may still be
|
|
946
|
+
// running. The state says the transition is unfinished, so a
|
|
947
|
+
// later suspend() re-sends it rather than believing this one.
|
|
948
|
+
state = 'suspending';
|
|
949
|
+
session = undefined;
|
|
950
|
+
let retired = false;
|
|
951
|
+
await runFailureCleanup(async (cleanupSignal) => {
|
|
952
|
+
await client.request('PATCH', sandboxPath(namespace, name), SUSPEND_PATCH, cleanupSignal);
|
|
953
|
+
retired = true;
|
|
954
|
+
});
|
|
955
|
+
// Only when the patch came back. A resume that got as far as
|
|
956
|
+
// binding a new pod and then failed its probe has retired THAT
|
|
957
|
+
// pod, and the resume after it must wait for the replacement
|
|
958
|
+
// rather than bind the one this cleanup took away. A cleanup whose
|
|
959
|
+
// patch never landed retired nothing and has nothing to exclude.
|
|
960
|
+
if (retired)
|
|
961
|
+
retiredPodUid = podUid;
|
|
962
|
+
throw err;
|
|
963
|
+
}
|
|
964
|
+
};
|
|
965
|
+
const admit = (operation) => {
|
|
966
|
+
if (state === 'deleted')
|
|
967
|
+
throw new KubernetesSandboxDestroyedError(operation, name);
|
|
968
|
+
if (state !== 'running' || session === undefined) {
|
|
969
|
+
throw new KubernetesWorkspaceSuspendedError(operation, workspaceId, name);
|
|
970
|
+
}
|
|
971
|
+
return session;
|
|
972
|
+
};
|
|
973
|
+
// The first session is brought up here so that `createKubernetesWorkspace`
|
|
974
|
+
// resolves with a workspace that is Ready, addressed and probed — the same
|
|
975
|
+
// contract `create()` gives a task sandbox.
|
|
976
|
+
session = await startSessionOrSuspend(options.created ? 'create' : 'adopt', options.signal);
|
|
977
|
+
// Whatever the backend's own Sandbox reports, rather than a second copy of
|
|
978
|
+
// the same constant: it must keep answering after a suspend has taken the
|
|
979
|
+
// handle it came from away.
|
|
980
|
+
const environment = session.environment;
|
|
981
|
+
return {
|
|
982
|
+
id,
|
|
983
|
+
get status() {
|
|
984
|
+
// A suspended workspace reports 'destroyed' because that is the only
|
|
985
|
+
// member of the SDK's four-way union meaning "cannot serve a call".
|
|
986
|
+
// `suspended` below is what tells the recoverable state apart.
|
|
987
|
+
if (state !== 'running' || session === undefined)
|
|
988
|
+
return 'destroyed';
|
|
989
|
+
return session.status;
|
|
990
|
+
},
|
|
991
|
+
get suspended() {
|
|
992
|
+
// `suspending` reads as suspended because that is what a caller
|
|
993
|
+
// can DO about it: no call is admitted and `resume()` is the way
|
|
994
|
+
// back. The difference between the two lives where it matters —
|
|
995
|
+
// in `suspendNow`, which returns early on one and not the other.
|
|
996
|
+
return state === 'suspended' || state === 'suspending';
|
|
997
|
+
},
|
|
998
|
+
rootDir: options.rootDir,
|
|
999
|
+
environment,
|
|
1000
|
+
async exec(command, argv, execOptions) {
|
|
1001
|
+
return await admit('exec').exec(command, argv, execOptions);
|
|
1002
|
+
},
|
|
1003
|
+
async writeFile(path, content) {
|
|
1004
|
+
await admit('writeFile').writeFile(path, content);
|
|
1005
|
+
},
|
|
1006
|
+
async readFile(path) {
|
|
1007
|
+
return await admit('readFile').readFile(path);
|
|
1008
|
+
},
|
|
1009
|
+
async listFiles(rootPath) {
|
|
1010
|
+
return await admit('listFiles').listFiles(rootPath);
|
|
1011
|
+
},
|
|
1012
|
+
async openTerminal(terminalOptions) {
|
|
1013
|
+
const terminal = await admit('openTerminal').openTerminal(terminalOptions);
|
|
1014
|
+
// Tracked HERE as well as by the inner handle, because a suspend
|
|
1015
|
+
// reaps terminals without going through the inner handle's
|
|
1016
|
+
// `destroy()` — the pod is being deleted, and a caller left holding
|
|
1017
|
+
// a session whose exit resolves only when TCP notices is a leak.
|
|
1018
|
+
terminals.add(terminal);
|
|
1019
|
+
// `.catch` after `.finally`, not `void` alone: `exited` belongs to
|
|
1020
|
+
// the CALLER, who may well let it reject, and a bookkeeping chain
|
|
1021
|
+
// hung off it would then reject with no handler and take the host
|
|
1022
|
+
// process down on an unhandled rejection.
|
|
1023
|
+
void terminal.exited
|
|
1024
|
+
.finally(() => {
|
|
1025
|
+
terminals.delete(terminal);
|
|
1026
|
+
})
|
|
1027
|
+
.catch(() => undefined);
|
|
1028
|
+
return terminal;
|
|
1029
|
+
},
|
|
1030
|
+
async openTcpConnection(connectOptions) {
|
|
1031
|
+
return await admit('openTcpConnection').openTcpConnection(connectOptions);
|
|
1032
|
+
},
|
|
1033
|
+
async suspend(transitionOptions) {
|
|
1034
|
+
await suspendShared(transitionOptions?.signal);
|
|
1035
|
+
},
|
|
1036
|
+
async resume(transitionOptions) {
|
|
1037
|
+
// Serialised, and deliberately without a single-flight slot of its
|
|
1038
|
+
// own. The two terminal transitions need one because each commits
|
|
1039
|
+
// its state only after the cluster confirms it, so a second caller
|
|
1040
|
+
// admitted behind the first finds nothing marked and re-sends;
|
|
1041
|
+
// `resumeNow` commits `running` at the END of a transition that
|
|
1042
|
+
// leaves the workspace usable, and returns early on it, so the
|
|
1043
|
+
// second caller finds the work done. Give resume a state it
|
|
1044
|
+
// early-returns on before the cluster confirms it and it will need
|
|
1045
|
+
// a slot as much as they do.
|
|
1046
|
+
await serialise(async () => await resumeNow(transitionOptions?.signal));
|
|
1047
|
+
},
|
|
1048
|
+
async destroy(destroyOptions) {
|
|
1049
|
+
if (destroyOptions?.deleteDisk !== true) {
|
|
1050
|
+
// The default, and the whole point of the default: there is no
|
|
1051
|
+
// delete-compute-keep-disk verb, so the closest thing to one is
|
|
1052
|
+
// a suspend, and `destroy()` in a `finally` must not erase a
|
|
1053
|
+
// workspace nobody asked to erase. It shares the suspend's
|
|
1054
|
+
// single flight, so `destroy()` racing `suspend()` is one
|
|
1055
|
+
// transition rather than two — and it stays idempotent over a
|
|
1056
|
+
// workspace already deleted, where `suspend()` itself refuses.
|
|
1057
|
+
await destroyBySuspending(destroyOptions?.signal);
|
|
1058
|
+
return;
|
|
1059
|
+
}
|
|
1060
|
+
await deleteShared(destroyOptions);
|
|
1061
|
+
},
|
|
1062
|
+
};
|
|
1063
|
+
}
|
|
1064
|
+
//# sourceMappingURL=workspace.js.map
|