@namzu/sandbox 14.0.0 → 15.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +838 -0
- package/README.md +310 -14
- package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
- package/dist/backends/aci-standby-pool/index.js +13 -1
- package/dist/backends/aci-standby-pool/index.js.map +1 -1
- package/dist/backends/docker/index.d.ts.map +1 -1
- package/dist/backends/docker/index.js +19 -1
- package/dist/backends/docker/index.js.map +1 -1
- package/dist/backends/firecracker/index.d.ts.map +1 -1
- package/dist/backends/firecracker/index.js +12 -2
- package/dist/backends/firecracker/index.js.map +1 -1
- package/dist/backends/firecracker/protocol.d.ts +459 -8
- package/dist/backends/firecracker/protocol.d.ts.map +1 -1
- package/dist/backends/firecracker/protocol.js +136 -0
- package/dist/backends/firecracker/protocol.js.map +1 -1
- package/dist/backends/firecracker/transport.d.ts +539 -6
- package/dist/backends/firecracker/transport.d.ts.map +1 -1
- package/dist/backends/firecracker/transport.js +1171 -24
- package/dist/backends/firecracker/transport.js.map +1 -1
- package/dist/backends/kubernetes/egress-policy.d.ts +1088 -11
- package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -1
- package/dist/backends/kubernetes/egress-policy.js +2173 -29
- package/dist/backends/kubernetes/egress-policy.js.map +1 -1
- package/dist/backends/kubernetes/identity.d.ts +193 -0
- package/dist/backends/kubernetes/identity.d.ts.map +1 -0
- package/dist/backends/kubernetes/identity.js +147 -0
- package/dist/backends/kubernetes/identity.js.map +1 -0
- package/dist/backends/kubernetes/index.d.ts +678 -33
- package/dist/backends/kubernetes/index.d.ts.map +1 -1
- package/dist/backends/kubernetes/index.js +1180 -95
- package/dist/backends/kubernetes/index.js.map +1 -1
- package/dist/backends/kubernetes/ingress-policy.d.ts +375 -0
- package/dist/backends/kubernetes/ingress-policy.d.ts.map +1 -0
- package/dist/backends/kubernetes/ingress-policy.js +1050 -0
- package/dist/backends/kubernetes/ingress-policy.js.map +1 -0
- package/dist/backends/kubernetes/k8s-client.d.ts +213 -4
- package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -1
- package/dist/backends/kubernetes/k8s-client.js +359 -52
- package/dist/backends/kubernetes/k8s-client.js.map +1 -1
- package/dist/backends/kubernetes/lease.d.ts +40 -14
- package/dist/backends/kubernetes/lease.d.ts.map +1 -1
- package/dist/backends/kubernetes/lease.js +68 -18
- package/dist/backends/kubernetes/lease.js.map +1 -1
- package/dist/backends/kubernetes/objects.d.ts +423 -3
- package/dist/backends/kubernetes/objects.d.ts.map +1 -1
- package/dist/backends/kubernetes/objects.js +364 -2
- package/dist/backends/kubernetes/objects.js.map +1 -1
- package/dist/backends/kubernetes/per-sandbox-policy.d.ts +219 -0
- package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -0
- package/dist/backends/kubernetes/per-sandbox-policy.js +407 -0
- package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -0
- package/dist/backends/kubernetes/rbac.d.ts +153 -0
- package/dist/backends/kubernetes/rbac.d.ts.map +1 -0
- package/dist/backends/kubernetes/rbac.js +177 -0
- package/dist/backends/kubernetes/rbac.js.map +1 -0
- package/dist/backends/kubernetes/sandbox.d.ts +81 -14
- package/dist/backends/kubernetes/sandbox.d.ts.map +1 -1
- package/dist/backends/kubernetes/sandbox.js +149 -15
- package/dist/backends/kubernetes/sandbox.js.map +1 -1
- package/dist/backends/kubernetes/transport.d.ts +935 -9
- package/dist/backends/kubernetes/transport.d.ts.map +1 -1
- package/dist/backends/kubernetes/transport.js +1958 -62
- package/dist/backends/kubernetes/transport.js.map +1 -1
- package/dist/backends/kubernetes/workspace.d.ts +1149 -18
- package/dist/backends/kubernetes/workspace.d.ts.map +1 -1
- package/dist/backends/kubernetes/workspace.js +2825 -186
- package/dist/backends/kubernetes/workspace.js.map +1 -1
- package/dist/backends/remote-execution-controller.d.ts +14 -0
- package/dist/backends/remote-execution-controller.d.ts.map +1 -1
- package/dist/backends/remote-execution-controller.js.map +1 -1
- package/dist/index.d.ts +231 -13
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +247 -5
- package/dist/index.js.map +1 -1
- package/dist/testing/sandbox-conformance.d.ts +39 -5
- package/dist/testing/sandbox-conformance.d.ts.map +1 -1
- package/dist/testing/sandbox-conformance.js +436 -5
- package/dist/testing/sandbox-conformance.js.map +1 -1
- package/package.json +3 -3
- package/src/backends/aci-standby-pool/index.ts +16 -1
- package/src/backends/docker/index.ts +22 -1
- package/src/backends/firecracker/index.ts +14 -2
- package/src/backends/firecracker/protocol.ts +514 -6
- package/src/backends/firecracker/transport.ts +1492 -40
- package/src/backends/kubernetes/egress-policy.ts +3064 -53
- package/src/backends/kubernetes/identity.ts +261 -0
- package/src/backends/kubernetes/index.ts +1785 -127
- package/src/backends/kubernetes/ingress-policy.ts +1344 -0
- package/src/backends/kubernetes/k8s-client.ts +444 -54
- package/src/backends/kubernetes/lease.ts +75 -19
- package/src/backends/kubernetes/objects.ts +626 -6
- package/src/backends/kubernetes/per-sandbox-policy.ts +542 -0
- package/src/backends/kubernetes/rbac.ts +192 -0
- package/src/backends/kubernetes/sandbox.ts +218 -20
- package/src/backends/kubernetes/transport.ts +2733 -124
- package/src/backends/kubernetes/workspace.ts +4476 -222
- package/src/backends/remote-execution-controller.ts +14 -0
- package/src/index.ts +595 -14
- package/src/testing/sandbox-conformance.ts +540 -5
|
@@ -62,20 +62,64 @@
|
|
|
62
62
|
* the pod — so the call would hang until a connect timeout with nothing in
|
|
63
63
|
* the failure naming the suspend.
|
|
64
64
|
*
|
|
65
|
+
* Both patches also stamp
|
|
66
|
+
* {@link OPERATING_MODE_CHANGED_AT_ANNOTATION_KEY}. Nothing in this module
|
|
67
|
+
* reads it back; it exists so that an inventory can say when a workspace was
|
|
68
|
+
* last put to sleep without waking it up to ask, which the controller's own
|
|
69
|
+
* lingering `Suspended` condition cannot answer. See
|
|
70
|
+
* {@link listKubernetesWorkspaces}.
|
|
71
|
+
*
|
|
72
|
+
* ## Three verbs that never open a workspace, and one that notices
|
|
73
|
+
*
|
|
74
|
+
* {@link createKubernetesWorkspace} adopts AND resumes, which is right for a
|
|
75
|
+
* host about to USE a workspace and wrong for everything else. Deleting a
|
|
76
|
+
* month-old suspended workspace through it means starting a pod, probing it
|
|
77
|
+
* and deleting it again; taking an inventory means waking every suspended
|
|
78
|
+
* object in the namespace. So the three operations that are about the OBJECT
|
|
79
|
+
* are reachable without a handle — {@link listKubernetesWorkspaces},
|
|
80
|
+
* {@link deleteKubernetesWorkspace} and {@link suspendKubernetesWorkspace} —
|
|
81
|
+
* and none of them creates a pod, dials an agent or resumes anything.
|
|
82
|
+
*
|
|
83
|
+
* The other half of the same problem is the handle that was already open when
|
|
84
|
+
* somebody else did one of those. A workspace id is a name, not a lock, so a
|
|
85
|
+
* second process can suspend the workspace this one is holding, and this
|
|
86
|
+
* handle's `state` is a record of what THIS process did. Two things fix that,
|
|
87
|
+
* and both re-read the object rather than guessing:
|
|
88
|
+
* {@link KubernetesWorkspace.refresh} when the caller asks, and the re-read
|
|
89
|
+
* after a call that FAILED at the transport — which is how it would otherwise
|
|
90
|
+
* be found out, as a connect refusal or a flat `unauthorized` naming nothing.
|
|
91
|
+
* Either one records the suspension as UNCONFIRMED (`suspending`, not
|
|
92
|
+
* `suspended`): what was observed is the object's mode, not the pod stopping,
|
|
93
|
+
* and only a wait this process performed can promise the disk is quiesced.
|
|
94
|
+
*
|
|
65
95
|
* ## Adoption is checked against the object, not against the caller
|
|
66
96
|
*
|
|
67
97
|
* A create that collides with an existing object of the same name ADOPTS it,
|
|
68
98
|
* because the deterministic name is only worth having if coming back is the
|
|
69
99
|
* normal path. What is adopted is then checked against the configuration: the
|
|
70
|
-
* block disk, the `sandbox.namzu.ai/template` pod label (the label
|
|
71
|
-
*
|
|
72
|
-
* standing object that disagrees with any of them is refused by
|
|
73
|
-
* than driven — see {@link KubernetesWorkspaceMismatchError}. What is NOT
|
|
100
|
+
* block disk, the `sandbox.namzu.ai/template` pod label (the label the
|
|
101
|
+
* ingress and egress policies select by) and `runtimeClassName` (the VM
|
|
102
|
+
* boundary). A standing object that disagrees with any of them is refused by
|
|
103
|
+
* name rather than driven — see {@link KubernetesWorkspaceMismatchError}. What is NOT
|
|
74
104
|
* checked, and cannot be from here, is whether somebody else is already using
|
|
75
105
|
* it: two host processes can hold handles to one running workspace, and the
|
|
76
106
|
* `destroy()` of either suspends the pod the other is executing in. A
|
|
77
107
|
* workspace id is a name, not a lock.
|
|
78
108
|
*
|
|
109
|
+
* An adopt also has to survive walking in ON a transition, which is the
|
|
110
|
+
* normal way a workspace is found rather than an edge: a suspend that ended
|
|
111
|
+
* in {@link KubernetesWorkspaceSuspendTimeoutError}, a second host coming up
|
|
112
|
+
* during a rollout, a host restarting inside the previous pod's
|
|
113
|
+
* `terminationGracePeriodSeconds`. In all three the pod under the name is
|
|
114
|
+
* draining or already gone, and the replacement has not been created yet. So
|
|
115
|
+
* an adopt that finds the object `Suspended`, or finds its pod carrying a
|
|
116
|
+
* `deletionTimestamp`, WAITS for the replacement under the same readiness
|
|
117
|
+
* budget the resume path waits under, instead of failing on the first read
|
|
118
|
+
* that finds no live pod. What it came by is then reported as `origin`:
|
|
119
|
+
* `created`, `adopted-running` or `resumed` — a host that adopted a pod
|
|
120
|
+
* another process left running needs to know that none of that process's
|
|
121
|
+
* terminals survived it.
|
|
122
|
+
*
|
|
79
123
|
* ## There is no delete-compute-keep-disk verb
|
|
80
124
|
*
|
|
81
125
|
* The API has `operatingMode` and it has DELETE. Nothing in between. So
|
|
@@ -91,9 +135,10 @@
|
|
|
91
135
|
* {@link createKubernetesWorkspace}), and a pod that stops being able to say
|
|
92
136
|
* what happened to a command — the shared execution controller's unconfirmed
|
|
93
137
|
* cancellation, which retires a task sandbox by DELETING it — is retired here
|
|
94
|
-
* by that same suspend patch (see {@link retireSession}). Exactly
|
|
95
|
-
*
|
|
96
|
-
*
|
|
138
|
+
* by that same suspend patch (see {@link retireSession}). Exactly two DELETEs
|
|
139
|
+
* are reachable from this module and both are ASKED FOR by name — the one
|
|
140
|
+
* `deleteDisk: true` sends and the one {@link deleteKubernetesWorkspace} is;
|
|
141
|
+
* no failure path, no `finally`, and no default reaches either.
|
|
97
142
|
*
|
|
98
143
|
* ## A state is committed when the cluster confirms it, never before
|
|
99
144
|
*
|
|
@@ -112,9 +157,12 @@
|
|
|
112
157
|
* arriving mid-transition awaits the one in progress instead of sending a
|
|
113
158
|
* second request into the gap the deferred mark opens.
|
|
114
159
|
*/
|
|
115
|
-
import type {
|
|
160
|
+
import type { Sandbox, SandboxDestroyOptions, SandboxExecResult, SandboxFileEntry, SandboxReadFileOptions, SandboxTcpConnectOptions, SandboxTcpConnection, SandboxWalkFilesOptions } from '@namzu/sdk';
|
|
161
|
+
import { type EgressProfileLabel } from './egress-policy.js';
|
|
162
|
+
import { type KubernetesGuestEvidence, type KubernetesGuestRestart, type KubernetesWorkspaceIdentity } from './identity.js';
|
|
116
163
|
import { type KubernetesBackendInternalConfig } from './index.js';
|
|
117
164
|
import { type SandboxPodTemplate, type SandboxVolumeClaimTemplate } from './objects.js';
|
|
165
|
+
import { type KubernetesAttachExecutionOptions, type KubernetesAttachTerminalOptions, type KubernetesDetachedExecOptions, type KubernetesFlushReport, KubernetesFlushUnreachableError, KubernetesFlushUnsupportedError, type KubernetesOpenTerminalOptions, type KubernetesQuiesceReport, KubernetesQuiesceUnsupportedError, type KubernetesReadSessionOptions, type KubernetesSessionOutput, type KubernetesSessionSummary, type KubernetesSessionTerminal, type KubernetesStartDetachedOptions, type KubernetesWorkspaceTerminal } from './transport.js';
|
|
118
166
|
/**
|
|
119
167
|
* Thrown when a workspace's `SandboxTemplate` does not describe a block disk
|
|
120
168
|
* this backend is willing to build a workspace on.
|
|
@@ -149,6 +197,14 @@ export declare class KubernetesWorkspaceDiskError extends Error {
|
|
|
149
197
|
* egress `NetworkPolicy`'s `podSelector` matches. A pod carrying another
|
|
150
198
|
* value — or none — is not selected by the policy this call just verified,
|
|
151
199
|
* so the boundary would report verified while covering nothing.
|
|
200
|
+
* - the egress PROFILE label, when `config.egress.profile` is set, for
|
|
201
|
+
* exactly the same reason: it is the selector's second half. A standing
|
|
202
|
+
* object built before the profile existed, or under a different one,
|
|
203
|
+
* carries a pod the per-profile policy does not select — and once each
|
|
204
|
+
* profile has its own policy object, as it must, a pod carrying neither
|
|
205
|
+
* key is selected by no egress policy at all. This one is checked only
|
|
206
|
+
* when a profile is configured: with none, the policy this call verified
|
|
207
|
+
* selects the template label alone, which the object does carry.
|
|
152
208
|
* - `runtimeClassName`, which is the VM boundary. The privilege probe cannot
|
|
153
209
|
* stand in for it: `/proc/self/status` reads the same inside a VM guest as
|
|
154
210
|
* it does inside an ordinary shared-kernel container.
|
|
@@ -161,8 +217,12 @@ export declare class KubernetesWorkspaceMismatchError extends Error {
|
|
|
161
217
|
/** The Sandbox found standing under this workspace's name. */
|
|
162
218
|
readonly sandboxName: string;
|
|
163
219
|
/** Which configured field the standing object disagrees with. */
|
|
164
|
-
readonly field: 'sandboxTemplateName' | 'runtimeClassName';
|
|
165
|
-
/**
|
|
220
|
+
readonly field: 'sandboxTemplateName' | 'egressProfile' | 'runtimeClassName';
|
|
221
|
+
/**
|
|
222
|
+
* What the configuration asked for. `key=value` for `egressProfile`,
|
|
223
|
+
* because the KEY is configurable too and the value alone would not
|
|
224
|
+
* say which label was compared.
|
|
225
|
+
*/
|
|
166
226
|
readonly expected: string;
|
|
167
227
|
/** What the object carries — absent when it carries nothing at all. */
|
|
168
228
|
readonly actual: string | undefined;
|
|
@@ -171,25 +231,53 @@ export declare class KubernetesWorkspaceMismatchError extends Error {
|
|
|
171
231
|
/** The Sandbox found standing under this workspace's name. */
|
|
172
232
|
sandboxName: string,
|
|
173
233
|
/** Which configured field the standing object disagrees with. */
|
|
174
|
-
field: 'sandboxTemplateName' | 'runtimeClassName',
|
|
175
|
-
/**
|
|
234
|
+
field: 'sandboxTemplateName' | 'egressProfile' | 'runtimeClassName',
|
|
235
|
+
/**
|
|
236
|
+
* What the configuration asked for. `key=value` for `egressProfile`,
|
|
237
|
+
* because the KEY is configurable too and the value alone would not
|
|
238
|
+
* say which label was compared.
|
|
239
|
+
*/
|
|
176
240
|
expected: string,
|
|
177
241
|
/** What the object carries — absent when it carries nothing at all. */
|
|
178
242
|
actual: string | undefined, message: string);
|
|
179
243
|
}
|
|
244
|
+
/**
|
|
245
|
+
* How a caller came to be told a workspace is suspended.
|
|
246
|
+
*
|
|
247
|
+
* - `admission` — the handle knew before the call, so nothing was dialed.
|
|
248
|
+
* Every suspend this handle performed, and every one it has already
|
|
249
|
+
* noticed, lands here.
|
|
250
|
+
* - `transport` — the call went out and failed, and the re-read that
|
|
251
|
+
* followed found the object `Suspended`. Somebody ELSE suspended this
|
|
252
|
+
* workspace while this handle was holding it; the failure that prompted
|
|
253
|
+
* the re-read is on `cause`.
|
|
254
|
+
*/
|
|
255
|
+
export type KubernetesWorkspaceSuspensionNotice = 'admission' | 'transport';
|
|
180
256
|
/**
|
|
181
257
|
* Thrown by every operation on a workspace that is currently suspended.
|
|
182
258
|
*
|
|
183
259
|
* Distinct from {@link KubernetesSandboxDestroyedError} because the state is
|
|
184
|
-
* RECOVERABLE and the advice is one word: call `resume()`.
|
|
185
|
-
*
|
|
260
|
+
* RECOVERABLE and the advice is one word: call `resume()`.
|
|
261
|
+
*
|
|
262
|
+
* `noticedBy` says which of the two ways the caller got here, and the message
|
|
263
|
+
* changes with it, because "nothing was dialed" is a promise the first one
|
|
264
|
+
* keeps and the second one cannot. A workspace another process suspended is
|
|
265
|
+
* discovered by a call FAILING — the pod is gone, so the dial is refused, or
|
|
266
|
+
* the replacement pod's agent refuses this handle's token — and the reason
|
|
267
|
+
* that is worth converting into this error rather than passing on is that the
|
|
268
|
+
* raw failure names nothing: a flat `unauthorized`, or a connect error against
|
|
269
|
+
* an address that still resolves because the Service outlives the pod.
|
|
186
270
|
*/
|
|
187
271
|
export declare class KubernetesWorkspaceSuspendedError extends Error {
|
|
188
272
|
readonly operation: string;
|
|
189
273
|
readonly workspaceId: string;
|
|
190
274
|
readonly sandboxName: string;
|
|
275
|
+
/** See {@link KubernetesWorkspaceSuspensionNotice}. Defaults to `admission`. */
|
|
276
|
+
readonly noticedBy: KubernetesWorkspaceSuspensionNotice;
|
|
191
277
|
readonly name = "KubernetesWorkspaceSuspendedError";
|
|
192
|
-
constructor(operation: string, workspaceId: string, sandboxName: string
|
|
278
|
+
constructor(operation: string, workspaceId: string, sandboxName: string,
|
|
279
|
+
/** See {@link KubernetesWorkspaceSuspensionNotice}. Defaults to `admission`. */
|
|
280
|
+
noticedBy?: KubernetesWorkspaceSuspensionNotice, options?: ErrorOptions);
|
|
193
281
|
}
|
|
194
282
|
/**
|
|
195
283
|
* Thrown when a suspend's `operatingMode: Suspended` patch was accepted and
|
|
@@ -218,9 +306,337 @@ export declare class KubernetesWorkspaceSuspendTimeoutError extends Error {
|
|
|
218
306
|
/** The readiness budget the wait was given, in milliseconds. */
|
|
219
307
|
timeoutMs: number);
|
|
220
308
|
}
|
|
221
|
-
/**
|
|
309
|
+
/**
|
|
310
|
+
* Thrown when a lifecycle write carried a holder epoch the workspace has
|
|
311
|
+
* already moved past: the request was NOT sent, or was sent and refused, and
|
|
312
|
+
* nothing on the cluster changed either way.
|
|
313
|
+
*
|
|
314
|
+
* The workspace is somebody else's now. The epoch stored on the Sandbox is
|
|
315
|
+
* higher than the one this call carried, which is what a host says when it
|
|
316
|
+
* hands authority over a workspace to another process — see
|
|
317
|
+
* {@link HOLDER_EPOCH_ANNOTATION_KEY}. A superseded holder that suspends,
|
|
318
|
+
* resumes or deletes anyway would be taking the pod, or the disk, away from
|
|
319
|
+
* whoever holds it now.
|
|
320
|
+
*
|
|
321
|
+
* Nothing about the handle changes either. A refused `suspend()` leaves the
|
|
322
|
+
* handle exactly where it was — still `running`, terminals still open,
|
|
323
|
+
* because the refusal is decided BEFORE they are reaped — so a caller that
|
|
324
|
+
* catches this and re-reads its own holder record has lost nothing.
|
|
325
|
+
*
|
|
326
|
+
* `storedEpoch` is `undefined` in the one case the annotation cannot be read
|
|
327
|
+
* at all: it is present and is not a decimal integer, which no release of
|
|
328
|
+
* this backend writes. `storedAnnotation` carries it verbatim so an operator
|
|
329
|
+
* can see what is actually on the object.
|
|
330
|
+
*/
|
|
331
|
+
export declare class KubernetesWorkspacePreconditionError extends Error {
|
|
332
|
+
/** The verb that was refused — `suspend`, `resume`, `destroy`, ... */
|
|
333
|
+
readonly operation: string;
|
|
334
|
+
readonly workspaceId: string;
|
|
335
|
+
readonly sandboxName: string;
|
|
336
|
+
/** The epoch this call carried. */
|
|
337
|
+
readonly epoch: number;
|
|
338
|
+
/** The epoch stored on the Sandbox, or `undefined` if unreadable. */
|
|
339
|
+
readonly storedEpoch: number | undefined;
|
|
340
|
+
/** The annotation exactly as stored, when there is one. */
|
|
341
|
+
readonly storedAnnotation?: string | undefined;
|
|
342
|
+
readonly name = "KubernetesWorkspacePreconditionError";
|
|
343
|
+
constructor(
|
|
344
|
+
/** The verb that was refused — `suspend`, `resume`, `destroy`, ... */
|
|
345
|
+
operation: string, workspaceId: string, sandboxName: string,
|
|
346
|
+
/** The epoch this call carried. */
|
|
347
|
+
epoch: number,
|
|
348
|
+
/** The epoch stored on the Sandbox, or `undefined` if unreadable. */
|
|
349
|
+
storedEpoch: number | undefined,
|
|
350
|
+
/** The annotation exactly as stored, when there is one. */
|
|
351
|
+
storedAnnotation?: string | undefined);
|
|
352
|
+
}
|
|
353
|
+
/**
|
|
354
|
+
* What a start that FAILED is allowed to do to the workspace it was starting
|
|
355
|
+
* in.
|
|
356
|
+
*
|
|
357
|
+
* - `suspend-if-woken` (the default) — send the `operatingMode: Suspended`
|
|
358
|
+
* patch only when this call is the one that moved the mode: it POSTed the
|
|
359
|
+
* object, or its Running patch took the object out of `Suspended`. That
|
|
360
|
+
* keeps the case the rule exists for — a workspace this call WOKE and then
|
|
361
|
+
* failed to start would otherwise be left Running with a pod nobody is
|
|
362
|
+
* using, burning a node until somebody notices — while never taking a pod
|
|
363
|
+
* away from a holder who was already using it.
|
|
364
|
+
* - `leave` — never patch, on any start failure, without exception. For a
|
|
365
|
+
* host that keeps its own holder record and sweeps idle workspaces itself:
|
|
366
|
+
* the cost of being wrong is one Running workspace nobody is in, which
|
|
367
|
+
* that host can already see and already sweeps.
|
|
368
|
+
*
|
|
369
|
+
* Read only by the paths that START a session: {@link
|
|
370
|
+
* createKubernetesWorkspace} and `resume()`. `suspend()`, `refresh()`,
|
|
371
|
+
* `destroy()` and the three standalone verbs start nothing and ignore it.
|
|
372
|
+
*/
|
|
373
|
+
export type KubernetesWorkspaceStartFailurePolicy = 'suspend-if-woken' | 'leave';
|
|
374
|
+
/**
|
|
375
|
+
* What one bounded `healthz` found after a cancellation went unconfirmed —
|
|
376
|
+
* the fact the host needs and the error cannot carry.
|
|
377
|
+
*
|
|
378
|
+
* - `ok` — the agent answered and is serving normally. The command whose
|
|
379
|
+
* cancellation could not be confirmed may still be running in that pod,
|
|
380
|
+
* but nothing is wedged; the workspace goes on working.
|
|
381
|
+
* - `retiring` — the agent has FENCED itself: it could not confirm that a
|
|
382
|
+
* process group was gone, so it refuses every op but `healthz` and
|
|
383
|
+
* `cancel-execution` and will until the pod is replaced. This is the one
|
|
384
|
+
* case where `suspend()` then `resume()` is the cure.
|
|
385
|
+
* - `unreachable` — the probe could not get an answer at all: the pod is
|
|
386
|
+
* gone, the network is out, or the address stopped resolving. Nothing can
|
|
387
|
+
* be concluded about the command, and nothing should be done about the
|
|
388
|
+
* workspace on this alone — the next call finds out, and a foreign suspend
|
|
389
|
+
* is reported as one.
|
|
390
|
+
*/
|
|
391
|
+
export type KubernetesWorkspaceAgentState = 'ok' | 'retiring' | 'unreachable';
|
|
392
|
+
/**
|
|
393
|
+
* What {@link KubernetesWorkspaceOptions.onCancellationUnconfirmed} is told.
|
|
394
|
+
*
|
|
395
|
+
* It is a NOTIFICATION, not a decision point: the `exec()` this came from
|
|
396
|
+
* rejects with `error` whatever the callback does, and nothing the callback
|
|
397
|
+
* does is awaited by the failing call beyond the moment it returns.
|
|
398
|
+
*/
|
|
399
|
+
export interface KubernetesWorkspaceCancellationNotice {
|
|
400
|
+
/** The `RemoteCancellationUnknownError` the caller is about to receive. */
|
|
401
|
+
readonly error: Error;
|
|
402
|
+
/** See {@link KubernetesWorkspaceAgentState}. */
|
|
403
|
+
readonly agent: KubernetesWorkspaceAgentState;
|
|
404
|
+
/**
|
|
405
|
+
* What a bounded look at the pod and the agent process found — see
|
|
406
|
+
* {@link KubernetesGuestEvidence}.
|
|
407
|
+
*
|
|
408
|
+
* It answers the question `agent` cannot: `healthz` says whether SOME
|
|
409
|
+
* agent is serving at that address, and this says whether it is the same
|
|
410
|
+
* one the command was running in. Anything but `same-guest` means the
|
|
411
|
+
* command cannot still be running, because the process tree it belonged
|
|
412
|
+
* to is gone — and that a suspend would take a pod nobody's command is in.
|
|
413
|
+
*/
|
|
414
|
+
readonly guest: KubernetesGuestEvidence;
|
|
415
|
+
/** The guest the command was started on. */
|
|
416
|
+
readonly previous: KubernetesWorkspaceIdentity;
|
|
417
|
+
/** The guest standing under the workspace's name now. */
|
|
418
|
+
readonly current: KubernetesWorkspaceIdentity;
|
|
419
|
+
}
|
|
420
|
+
/**
|
|
421
|
+
* Authority for one workspace operation, owned independently of the run.
|
|
422
|
+
*
|
|
423
|
+
* Carried by the handle's transitions and by the three verbs that reach a
|
|
424
|
+
* workspace without opening one ({@link listKubernetesWorkspaces},
|
|
425
|
+
* {@link deleteKubernetesWorkspace}, {@link suspendKubernetesWorkspace}) —
|
|
426
|
+
* one shape rather than four, because a cancellation scope is what all of
|
|
427
|
+
* them take and the one verb among them that starts a session takes one
|
|
428
|
+
* thing more.
|
|
429
|
+
*/
|
|
222
430
|
export interface KubernetesWorkspaceTransitionOptions {
|
|
223
431
|
readonly signal?: AbortSignal;
|
|
432
|
+
/**
|
|
433
|
+
* Override, for this call only, what a failed start may do to the
|
|
434
|
+
* workspace — see {@link KubernetesWorkspaceStartFailurePolicy}. Defaults
|
|
435
|
+
* to whatever the handle was opened with, which defaults to
|
|
436
|
+
* `suspend-if-woken`.
|
|
437
|
+
*
|
|
438
|
+
* `resume()` is the only verb on this shape that reads it, because it is
|
|
439
|
+
* the only one that starts a session. It is declared here rather than on
|
|
440
|
+
* a resume-only shape so that a host passing one options object to every
|
|
441
|
+
* transition does not have to know which verb consults which field.
|
|
442
|
+
*/
|
|
443
|
+
readonly onStartFailure?: KubernetesWorkspaceStartFailurePolicy;
|
|
444
|
+
/**
|
|
445
|
+
* The holder epoch this call writes under — see
|
|
446
|
+
* {@link HOLDER_EPOCH_ANNOTATION_KEY}.
|
|
447
|
+
*
|
|
448
|
+
* The write applies when the epoch stored on the Sandbox is `<=` this
|
|
449
|
+
* one, and stores this one in the same request; a stored epoch above it
|
|
450
|
+
* refuses the write with {@link KubernetesWorkspacePreconditionError} and
|
|
451
|
+
* changes nothing. Omitted, every request goes out exactly as it did
|
|
452
|
+
* before epochs existed, merge-patch content type included — this fences
|
|
453
|
+
* nothing until a host opts in.
|
|
454
|
+
*
|
|
455
|
+
* Read by every verb on this shape that WRITES: `suspend()`, `resume()`,
|
|
456
|
+
* `destroy()` and the standalone
|
|
457
|
+
* {@link suspendKubernetesWorkspace} / {@link deleteKubernetesWorkspace}.
|
|
458
|
+
* `refresh()`, `cancelExecution()` and {@link listKubernetesWorkspaces}
|
|
459
|
+
* send no write, so there is nothing for an epoch to condition there and
|
|
460
|
+
* they ignore it — the same way `onStartFailure` above is ignored by
|
|
461
|
+
* every verb that starts no session. A list REPORTS each workspace's
|
|
462
|
+
* stored epoch instead, on
|
|
463
|
+
* {@link KubernetesWorkspaceSummary.holderEpoch}.
|
|
464
|
+
*
|
|
465
|
+
* A handle that was opened with one, or last resumed with one, uses it
|
|
466
|
+
* for every write it sends when the call passes none — including the
|
|
467
|
+
* patch a failed start's cleanup sends.
|
|
468
|
+
*/
|
|
469
|
+
readonly epoch?: number;
|
|
470
|
+
/**
|
|
471
|
+
* Write the SandboxTemplate's CURRENT `spec.podTemplate` onto the
|
|
472
|
+
* workspace as it wakes, keeping its disk — see
|
|
473
|
+
* {@link KubernetesWorkspace.templateRevision}.
|
|
474
|
+
*
|
|
475
|
+
* `resume()` is the only verb on this shape that reads it, in the same
|
|
476
|
+
* way `onStartFailure` above is read only by the verb that starts a
|
|
477
|
+
* session. It is honoured on exactly one transition — a workspace
|
|
478
|
+
* observed `Suspended` going back to `Running` — because the controller
|
|
479
|
+
* does not rewrite a pod that already exists, so a Running workspace
|
|
480
|
+
* patched this way would carry a spec describing a pod it is not running.
|
|
481
|
+
* A `resume()` of a workspace that is already awake, and one whose
|
|
482
|
+
* condition loses to another process's resume, send no pod template and
|
|
483
|
+
* bind the pod that is there.
|
|
484
|
+
*
|
|
485
|
+
* Omitted or `false`, `resume()` sends exactly the single merge patch it
|
|
486
|
+
* always sent.
|
|
487
|
+
*/
|
|
488
|
+
readonly refreshPodTemplate?: boolean;
|
|
489
|
+
}
|
|
490
|
+
/**
|
|
491
|
+
* What a SUSPEND takes, which is every transition option plus the one thing
|
|
492
|
+
* only a suspend can do with it.
|
|
493
|
+
*
|
|
494
|
+
* `quiesce` lives here rather than on
|
|
495
|
+
* {@link KubernetesWorkspaceTransitionOptions} so that the verbs which
|
|
496
|
+
* cannot honour it — `resume()`, `refresh()`, {@link listKubernetesWorkspaces}
|
|
497
|
+
* and {@link deleteKubernetesWorkspace}, none of which sends a patch a
|
|
498
|
+
* quiesce could precede — do not accept it and then ignore it. A caller who
|
|
499
|
+
* passed it is about to trust a capture, and a silently dropped flag is the
|
|
500
|
+
* one way this feature could lie.
|
|
501
|
+
*/
|
|
502
|
+
export interface KubernetesWorkspaceSuspendOptions extends KubernetesWorkspaceTransitionOptions {
|
|
503
|
+
/**
|
|
504
|
+
* Stop every process in the guest BEFORE the `Suspended` patch goes out —
|
|
505
|
+
* see {@link KubernetesWorkspace.quiesce}, which this runs.
|
|
506
|
+
*
|
|
507
|
+
* Honoured by the handle's `suspend()` and by the `destroy()` that
|
|
508
|
+
* suspends. The standalone {@link suspendKubernetesWorkspace} takes this
|
|
509
|
+
* same shape and REFUSES the flag rather than ignoring it: it reaches the
|
|
510
|
+
* workspace through the API server alone and never dials the agent, so
|
|
511
|
+
* there is no connection on which it could stop anything. Off by default,
|
|
512
|
+
* so a `suspend()` that does not name it is byte-for-byte the suspend it
|
|
513
|
+
* always was.
|
|
514
|
+
*
|
|
515
|
+
* A guest whose image predates the op keeps today's behaviour: the
|
|
516
|
+
* suspend proceeds and
|
|
517
|
+
* {@link KubernetesWorkspaceOptions.onQuiesceUnsupported} is told. A
|
|
518
|
+
* quiesce the guest ANSWERED and could not confirm is different — the
|
|
519
|
+
* suspend rejects and sends no patch, leaving the workspace running,
|
|
520
|
+
* because the state that is true is the one this handle records. And a
|
|
521
|
+
* suspend ALREADY IN FLIGHT without a quiesce is refused rather than
|
|
522
|
+
* joined — see `suspend()`.
|
|
523
|
+
*/
|
|
524
|
+
readonly quiesce?: KubernetesWorkspaceQuiesceRequest;
|
|
525
|
+
/**
|
|
526
|
+
* Ask the guest to put the workspace's writes on its device before the
|
|
527
|
+
* `Suspended` patch goes out — see {@link KubernetesWorkspace.flush},
|
|
528
|
+
* which this runs. **On by default**, which is the one behaviour this
|
|
529
|
+
* option changes from what `suspend()` used to do.
|
|
530
|
+
*
|
|
531
|
+
* It is on by default because the alternative is the failure it exists
|
|
532
|
+
* to fix: a stopped pod means only that nothing is writing any more, and
|
|
533
|
+
* the page cache of a guest whose pod was taken away is not a thing
|
|
534
|
+
* anything here can go back for. The cost is one round trip and one
|
|
535
|
+
* `syncfs` per suspend.
|
|
536
|
+
*
|
|
537
|
+
* `false` sends the patch with no flush, which is exactly what every
|
|
538
|
+
* release before this one did. Use it when the caller has already
|
|
539
|
+
* flushed (a `flush()` of its own, or a `sync` it ran through `exec`),
|
|
540
|
+
* or when the workspace is about to be deleted anyway.
|
|
541
|
+
*
|
|
542
|
+
* A guest whose image predates the op keeps the old behaviour: the
|
|
543
|
+
* suspend proceeds and
|
|
544
|
+
* {@link KubernetesWorkspaceOptions.onFlushUnsupported} is told. So does
|
|
545
|
+
* a guest that cannot be REACHED to be asked — a crashed or OOM-killed
|
|
546
|
+
* agent, a lost network, a fence — through
|
|
547
|
+
* {@link KubernetesWorkspaceOptions.onFlushUnreachable}: leaving such a
|
|
548
|
+
* workspace running flushes nothing and costs money, and this is the
|
|
549
|
+
* verb that replaces its pod. A flush the guest ANSWERED and could not
|
|
550
|
+
* confirm is the one that is different — the suspend rejects and sends
|
|
551
|
+
* no patch, leaving the workspace running and the caller holding the
|
|
552
|
+
* guest's own message, because a patch sent over unwritten pages is how
|
|
553
|
+
* the data is lost.
|
|
554
|
+
*
|
|
555
|
+
* `{ timeoutMs }` raises what the GUEST may spend inside that `syncfs`,
|
|
556
|
+
* for a workspace that leaves more dirty than the guest's 10000ms
|
|
557
|
+
* default covers. It is the same number `flush({ timeoutMs })` takes.
|
|
558
|
+
*/
|
|
559
|
+
readonly flush?: KubernetesWorkspaceFlushRequest;
|
|
560
|
+
}
|
|
561
|
+
/**
|
|
562
|
+
* `flush()`'s own options.
|
|
563
|
+
*
|
|
564
|
+
* `signal` is the `AbortSignal` that cancels the REQUEST, like every other
|
|
565
|
+
* verb on this handle. `timeoutMs` bounds what the GUEST spends inside the
|
|
566
|
+
* `syncfs`, which is a different clock: aborting the request leaves the
|
|
567
|
+
* guest's own syncfs running, and only the guest's timeout ends that.
|
|
568
|
+
*/
|
|
569
|
+
export interface KubernetesFlushOptions {
|
|
570
|
+
/**
|
|
571
|
+
* How long the guest may spend flushing before it reports the flush
|
|
572
|
+
* unconfirmed. The guest defaults to 10000ms
|
|
573
|
+
* (`NAMZU_AGENT_FLUSH_TIMEOUT_MS`).
|
|
574
|
+
*
|
|
575
|
+
* It is NOT the pod's `terminationGracePeriodSeconds`, which has to
|
|
576
|
+
* cover the flush a stopping pod performs AND the drain beside it.
|
|
577
|
+
*/
|
|
578
|
+
readonly timeoutMs?: number;
|
|
579
|
+
readonly signal?: AbortSignal;
|
|
580
|
+
}
|
|
581
|
+
/**
|
|
582
|
+
* `quiesce()`'s own options.
|
|
583
|
+
*
|
|
584
|
+
* `signal` is the `AbortSignal` that cancels the REQUEST, as it is on every
|
|
585
|
+
* other verb on this handle. There is deliberately no POSIX-signal field to
|
|
586
|
+
* go with it: the escalation is the guest's and is fixed — `SIGTERM`, then
|
|
587
|
+
* `SIGKILL` on whatever is left — because a caller-chosen signal that a
|
|
588
|
+
* process ignores would turn a quiesce into a call that resolves having
|
|
589
|
+
* stopped nothing.
|
|
590
|
+
*/
|
|
591
|
+
export interface KubernetesQuiesceOptions {
|
|
592
|
+
/**
|
|
593
|
+
* How long the guest waits after `SIGTERM` before escalating to
|
|
594
|
+
* `SIGKILL`, per round. The guest refuses a value at or above its own
|
|
595
|
+
* cancel-confirmation timeout (5000ms by default) and defaults to
|
|
596
|
+
* 1000ms.
|
|
597
|
+
*
|
|
598
|
+
* It is NOT the pod's `terminationGracePeriodSeconds`, which bounds how
|
|
599
|
+
* long the kubelet waits after the pod has been asked to stop. This one
|
|
600
|
+
* bounds a round of an op the host called while the pod is still running
|
|
601
|
+
* and still serving.
|
|
602
|
+
*/
|
|
603
|
+
readonly graceMs?: number;
|
|
604
|
+
readonly signal?: AbortSignal;
|
|
605
|
+
}
|
|
606
|
+
/**
|
|
607
|
+
* What `suspend({ quiesce })` and `destroy({ quiesce })` ask for: `true`
|
|
608
|
+
* for the guest's own default window, or an object naming `graceMs`.
|
|
609
|
+
*/
|
|
610
|
+
export type KubernetesWorkspaceQuiesceRequest = boolean | {
|
|
611
|
+
readonly graceMs?: number;
|
|
612
|
+
};
|
|
613
|
+
/**
|
|
614
|
+
* What `suspend({ flush })` and `destroy({ flush })` ask for: `true` (the
|
|
615
|
+
* default) for the guest's own flush timeout, `false` for no flush at all,
|
|
616
|
+
* or an object naming `timeoutMs`.
|
|
617
|
+
*
|
|
618
|
+
* The object shape exists for the workspace whose flush does not fit in the
|
|
619
|
+
* guest's default 10000ms (`NAMZU_AGENT_FLUSH_TIMEOUT_MS`). Without it such
|
|
620
|
+
* a workspace answers `flush_unconfirmed` — the one outcome that STOPS a
|
|
621
|
+
* suspend — and the only ways past it are `flush({ timeoutMs })` followed
|
|
622
|
+
* by `suspend({ flush: false })`, or rebuilding the image's environment.
|
|
623
|
+
* Mirrors {@link KubernetesWorkspaceQuiesceRequest}, which names the
|
|
624
|
+
* guest-side budget of its own op for the same reason.
|
|
625
|
+
*/
|
|
626
|
+
export type KubernetesWorkspaceFlushRequest = boolean | {
|
|
627
|
+
readonly timeoutMs?: number;
|
|
628
|
+
};
|
|
629
|
+
/**
|
|
630
|
+
* `killSession()`'s two signals, which are different things and are named
|
|
631
|
+
* apart for that reason: `signal` is the POSIX signal the guest sends to the
|
|
632
|
+
* session, and `abort` is the `AbortSignal` that cancels the REQUEST.
|
|
633
|
+
* Collapsing them into one field called `signal` is how a caller ends up
|
|
634
|
+
* sending `SIGKILL` to a shell it only meant to stop asking about.
|
|
635
|
+
*/
|
|
636
|
+
export interface KubernetesKillSessionOptions {
|
|
637
|
+
/** `SIGTERM`, `SIGKILL`, `SIGINT` or `SIGHUP`. Default `SIGKILL`. */
|
|
638
|
+
readonly signal?: string;
|
|
639
|
+
readonly abort?: AbortSignal;
|
|
224
640
|
}
|
|
225
641
|
/**
|
|
226
642
|
* `destroy()` on a workspace, with the one field that decides whether the
|
|
@@ -233,7 +649,52 @@ export interface KubernetesWorkspaceDestroyOptions extends SandboxDestroyOptions
|
|
|
233
649
|
* including the default, suspends and leaves the object standing.
|
|
234
650
|
*/
|
|
235
651
|
readonly deleteDisk?: boolean;
|
|
652
|
+
/**
|
|
653
|
+
* The holder epoch this destroy writes under — see
|
|
654
|
+
* {@link KubernetesWorkspaceTransitionOptions.epoch}, which it means
|
|
655
|
+
* exactly the same thing as.
|
|
656
|
+
*
|
|
657
|
+
* It fences both shapes of `destroy()`: the default's suspend patch and
|
|
658
|
+
* `deleteDisk: true`'s DELETE, the latter through a
|
|
659
|
+
* `preconditions.resourceVersion` on the version whose epoch was read. A
|
|
660
|
+
* retention job superseded between its check and its call takes nobody's
|
|
661
|
+
* disk.
|
|
662
|
+
*/
|
|
663
|
+
readonly epoch?: number;
|
|
664
|
+
/**
|
|
665
|
+
* Stop every process in the guest first — see
|
|
666
|
+
* {@link KubernetesWorkspaceSuspendOptions.quiesce}, which this means
|
|
667
|
+
* exactly the same thing as.
|
|
668
|
+
*
|
|
669
|
+
* It is read by the `destroy()` that SUSPENDS, which is the default one.
|
|
670
|
+
* A `destroy({ deleteDisk: true })` ignores it: the disk it would be
|
|
671
|
+
* quiescing is about to be deleted along with everything on it.
|
|
672
|
+
*/
|
|
673
|
+
readonly quiesce?: KubernetesWorkspaceQuiesceRequest;
|
|
674
|
+
/**
|
|
675
|
+
* Put the workspace's writes on its device first — see
|
|
676
|
+
* {@link KubernetesWorkspaceSuspendOptions.flush}, which this means
|
|
677
|
+
* exactly the same thing as, default included.
|
|
678
|
+
*
|
|
679
|
+
* Read by the `destroy()` that SUSPENDS, which is the default one and
|
|
680
|
+
* which keeps the disk. A `destroy({ deleteDisk: true })` ignores it for
|
|
681
|
+
* the same reason it ignores `quiesce`: the disk is about to be deleted.
|
|
682
|
+
*/
|
|
683
|
+
readonly flush?: KubernetesWorkspaceFlushRequest;
|
|
236
684
|
}
|
|
685
|
+
/**
|
|
686
|
+
* How a {@link KubernetesWorkspace} handle came by its workspace.
|
|
687
|
+
*
|
|
688
|
+
* - `created` — this call POSTed the Sandbox. The pod is this process's own
|
|
689
|
+
* and the disk is empty.
|
|
690
|
+
* - `adopted-running` — an object of that name already stood, already
|
|
691
|
+
* Running. The pod predates this handle, and usually predates this
|
|
692
|
+
* process.
|
|
693
|
+
* - `resumed` — an object of that name stood Suspended and this call patched
|
|
694
|
+
* it back to Running. The disk is whatever the last holder left on it; the
|
|
695
|
+
* pod is brand new.
|
|
696
|
+
*/
|
|
697
|
+
export type KubernetesWorkspaceOrigin = 'created' | 'adopted-running' | 'resumed';
|
|
237
698
|
/**
|
|
238
699
|
* A {@link Sandbox} that survives having its compute taken away.
|
|
239
700
|
*
|
|
@@ -244,6 +705,117 @@ export interface KubernetesWorkspaceDestroyOptions extends SandboxDestroyOptions
|
|
|
244
705
|
* `deleteDisk` field on the shared `SandboxDestroyOptions`.
|
|
245
706
|
*/
|
|
246
707
|
export interface KubernetesWorkspace extends Sandbox {
|
|
708
|
+
/**
|
|
709
|
+
* How this handle came by its workspace — see
|
|
710
|
+
* {@link KubernetesWorkspaceOrigin}.
|
|
711
|
+
*
|
|
712
|
+
* Fixed for the handle's life. It answers "what did this call walk into",
|
|
713
|
+
* not "what state is the workspace in now", which is what `suspended` is
|
|
714
|
+
* for; a later suspend/resume cycle does not rewrite it.
|
|
715
|
+
*
|
|
716
|
+
* It is reported because the two adopted values mean a pod this process
|
|
717
|
+
* did not start, and a host reattaching to a workspace another process
|
|
718
|
+
* left behind has to know what of that process's work is still there. On
|
|
719
|
+
* `resumed` the pod is new, so nothing survived but the disk. On
|
|
720
|
+
* `adopted-running` the guest's agent is the same process it was and a
|
|
721
|
+
* detached background command may still be running — but no TERMINAL is:
|
|
722
|
+
* the agent kills a terminal's process group the moment its connection
|
|
723
|
+
* closes, and the previous host's connections closed with the host. A
|
|
724
|
+
* caller that reopens terminals unconditionally on this value is right;
|
|
725
|
+
* one that assumes it can reattach to them is not.
|
|
726
|
+
*/
|
|
727
|
+
readonly origin: KubernetesWorkspaceOrigin;
|
|
728
|
+
/**
|
|
729
|
+
* The revision of the pod template the BOUND Sandbox carries — the value
|
|
730
|
+
* of its `sandbox.namzu.ai/pod-template-hash` annotation.
|
|
731
|
+
*
|
|
732
|
+
* A Sandbox's `spec.podTemplate` is a copy of the SandboxTemplate's,
|
|
733
|
+
* taken once, and the controller builds every replacement pod from that
|
|
734
|
+
* copy — so a workspace kept for weeks runs the pod spec it was created
|
|
735
|
+
* with. This is what the object recorded it was built from.
|
|
736
|
+
*
|
|
737
|
+
* `undefined` for a workspace created before the annotation existed, and
|
|
738
|
+
* for one an older release last wrote. That is reported as unknown rather
|
|
739
|
+
* than guessed from the stored spec: a comparison this backend invents
|
|
740
|
+
* would be a second, weaker copy of the overlay rules.
|
|
741
|
+
*/
|
|
742
|
+
readonly templateRevision: string | undefined;
|
|
743
|
+
/**
|
|
744
|
+
* Whether {@link templateRevision} matched the SandboxTemplate this
|
|
745
|
+
* handle last read.
|
|
746
|
+
*
|
|
747
|
+
* `false` means a suspend and a resume with
|
|
748
|
+
* {@link KubernetesWorkspaceTransitionOptions.refreshPodTemplate} would
|
|
749
|
+
* change what this workspace runs — a new image tag, a memory limit, a
|
|
750
|
+
* grace period, an env entry. A host can schedule that at a moment of its
|
|
751
|
+
* choosing instead of finding out when a release makes the stored image
|
|
752
|
+
* fail every start.
|
|
753
|
+
*
|
|
754
|
+
* A workspace with no recorded revision reads `false`: unknown is not
|
|
755
|
+
* current. The template is read when the handle is opened and again on
|
|
756
|
+
* every refresh, so a `resume()` WITHOUT the option leaves this answering
|
|
757
|
+
* against the last template this handle read rather than paying a GET for
|
|
758
|
+
* one nothing is going to be compared to.
|
|
759
|
+
*/
|
|
760
|
+
readonly templateCurrent: boolean;
|
|
761
|
+
/**
|
|
762
|
+
* The four objects this handle is bound to RIGHT NOW — see
|
|
763
|
+
* {@link KubernetesWorkspaceIdentity}.
|
|
764
|
+
*
|
|
765
|
+
* Read fresh on every access rather than snapshotted, because it moves:
|
|
766
|
+
* `podUid` changes across a suspend/resume and across a pod this handle
|
|
767
|
+
* rebound to, `guestBootId` changes when the kubelet restarts the
|
|
768
|
+
* container, and both are `undefined` while this handle holds no pod — a
|
|
769
|
+
* suspended or destroyed workspace. A resume that has BOUND its pod names
|
|
770
|
+
* it from that moment, probe included, so a listener reading this from
|
|
771
|
+
* inside a `pod-replaced` event raised during a resume is answered with
|
|
772
|
+
* the pod it was just told about rather than with nothing.
|
|
773
|
+
* `sandboxUid` and `volumeClaimUids` do not move for the life of a handle
|
|
774
|
+
* — a handle whose object was replaced refuses rather than following it.
|
|
775
|
+
*
|
|
776
|
+
* A host that keeps per-workspace state compares this across calls, or
|
|
777
|
+
* (better) subscribes with {@link onGuestRestart} and is told.
|
|
778
|
+
*/
|
|
779
|
+
readonly identity: KubernetesWorkspaceIdentity;
|
|
780
|
+
/**
|
|
781
|
+
* Be told when the guest behind this handle is replaced, and get back the
|
|
782
|
+
* function that stops being told.
|
|
783
|
+
*
|
|
784
|
+
* This is the honest half of the rebind. A handle that follows a replaced
|
|
785
|
+
* pod keeps WORKING, which is what a caller wants; what it cannot do is
|
|
786
|
+
* keep the caller's guest state, because every process that pod was
|
|
787
|
+
* running died with it. A host that remembers what it started — a dev
|
|
788
|
+
* server, a watcher, a shell — must subscribe, or it will go on believing
|
|
789
|
+
* in processes that no longer exist while every call succeeds.
|
|
790
|
+
*
|
|
791
|
+
* It fires for a pod the handle rebound to (`pod-replaced`) and for an
|
|
792
|
+
* agent process that was restarted inside the same pod
|
|
793
|
+
* (`container-restarted`, recognised only against a guest that reports a
|
|
794
|
+
* `guestBootId`). It does NOT fire across the caller's own `suspend()`
|
|
795
|
+
* and `resume()`: that pod change was asked for, and a host that issued
|
|
796
|
+
* it already knows its processes are gone. The one exception is a pod
|
|
797
|
+
* replaced by somebody else DURING a resume — between the pod this
|
|
798
|
+
* handle bound and the privilege probe that follows it — where the probe
|
|
799
|
+
* is refused, the handle rebinds, and a `pod-replaced` event is
|
|
800
|
+
* announced. That is a pod change the caller did not ask for, arriving
|
|
801
|
+
* inside a transition it did; reporting it is the point.
|
|
802
|
+
*
|
|
803
|
+
* Every event names both pods, that one included. The payload is built
|
|
804
|
+
* from the uids the routine announcing the move is holding — not read
|
|
805
|
+
* back off a handle whose transition has not finished — so there is no
|
|
806
|
+
* path on which a listener is told its guest moved and not told where
|
|
807
|
+
* from or where to.
|
|
808
|
+
*
|
|
809
|
+
* On a `pod-replaced` event `current.guestBootId` is `undefined`: the
|
|
810
|
+
* replacement has not answered yet, so there is no process to name.
|
|
811
|
+
* `current.podUid` is what moved, and `identity` names the process that
|
|
812
|
+
* answered from there once the call that rebound has returned.
|
|
813
|
+
*
|
|
814
|
+
* Listeners are called synchronously, in subscription order, and a
|
|
815
|
+
* listener that throws is swallowed — the event is a notification and
|
|
816
|
+
* must not fail the call that discovered it.
|
|
817
|
+
*/
|
|
818
|
+
onGuestRestart(listener: (event: KubernetesGuestRestart) => void): () => void;
|
|
247
819
|
/**
|
|
248
820
|
* True from the moment `suspend()` starts until `resume()` finishes.
|
|
249
821
|
*
|
|
@@ -254,8 +826,198 @@ export interface KubernetesWorkspace extends Sandbox {
|
|
|
254
826
|
* SDK's union.
|
|
255
827
|
*/
|
|
256
828
|
readonly suspended: boolean;
|
|
257
|
-
|
|
829
|
+
/**
|
|
830
|
+
* The SDK's `exec`, plus the three Kubernetes-only fields that let a
|
|
831
|
+
* command outlive the connection watching it — see
|
|
832
|
+
* {@link KubernetesDetachedExecOptions}.
|
|
833
|
+
*
|
|
834
|
+
* Without `executionId` and without `detach` this is the SDK's exec
|
|
835
|
+
* exactly: the same reserve-before-admission controller, the same wire
|
|
836
|
+
* request, the same result. The options type only WIDENS what is
|
|
837
|
+
* accepted, so every `SandboxExecOptions` a caller already passes is
|
|
838
|
+
* still a valid argument and nothing about `Sandbox` changed.
|
|
839
|
+
*/
|
|
840
|
+
exec(command: string, argv?: string[], options?: KubernetesDetachedExecOptions): Promise<SandboxExecResult>;
|
|
841
|
+
/**
|
|
842
|
+
* Read a command this workspace is running, or has recently run, by the
|
|
843
|
+
* id it was started with — including one started by a host process that
|
|
844
|
+
* no longer exists.
|
|
845
|
+
*
|
|
846
|
+
* It never signals the command: aborting `signal` stops reading and
|
|
847
|
+
* leaves it running, and closing the connection changes nothing in the
|
|
848
|
+
* guest. Every attach inside the retention window returns the same
|
|
849
|
+
* result. Past it, or in a pod that has since been replaced, it rejects
|
|
850
|
+
* with `KubernetesExecutionNotAttachableError`.
|
|
851
|
+
*/
|
|
852
|
+
attachExecution(executionId: string, options?: KubernetesAttachExecutionOptions): Promise<SandboxExecResult>;
|
|
853
|
+
/**
|
|
854
|
+
* End a command by id, from any process holding the id. Resolves only
|
|
855
|
+
* on a CONFIRMED termination and rejects with
|
|
856
|
+
* `RemoteCancellationUnknownError` otherwise — and rejecting here
|
|
857
|
+
* retires nothing: the workspace, its pod and its disk are left exactly
|
|
858
|
+
* as they are. A caller that wants the pod gone asks for that.
|
|
859
|
+
*/
|
|
860
|
+
cancelExecution(executionId: string, options?: KubernetesWorkspaceTransitionOptions): Promise<void>;
|
|
861
|
+
/**
|
|
862
|
+
* A guest PTY. Without `sessionId` and `persistent: true` this is exactly
|
|
863
|
+
* the terminal it has always been — same wire request, same behaviour —
|
|
864
|
+
* except that its teardown now reaches the whole session rather than only
|
|
865
|
+
* `script`'s process group, so a job the shell backgrounded no longer
|
|
866
|
+
* outlives the terminal that started it.
|
|
867
|
+
*
|
|
868
|
+
* With them the PTY belongs to the pod rather than to this connection:
|
|
869
|
+
* losing the connection DETACHES, and {@link attachTerminal} rejoins the
|
|
870
|
+
* same shell from another process. `exited` on such a terminal rejects
|
|
871
|
+
* with `AgentSessionDetachedError` when the attachment ends and the
|
|
872
|
+
* program does not, because resolving it would claim an exit that never
|
|
873
|
+
* happened.
|
|
874
|
+
*/
|
|
875
|
+
openTerminal(options: KubernetesOpenTerminalOptions): Promise<KubernetesWorkspaceTerminal>;
|
|
876
|
+
/**
|
|
877
|
+
* Rejoin a terminal session by the id it was opened with — from this
|
|
878
|
+
* process or from one that replaced it.
|
|
879
|
+
*
|
|
880
|
+
* The guest replays from `fromOffset` and then follows live, so a reader
|
|
881
|
+
* starting at 0 sees what the shell printed while nobody was watching,
|
|
882
|
+
* and is told how much the ring had to evict. At most one attachment
|
|
883
|
+
* exists at a time: a second attach ends the first by name rather than
|
|
884
|
+
* letting two processes interleave keystrokes into one shell.
|
|
885
|
+
*/
|
|
886
|
+
attachTerminal(sessionId: string, options?: KubernetesAttachTerminalOptions): Promise<KubernetesSessionTerminal>;
|
|
887
|
+
/**
|
|
888
|
+
* Start a program with no terminal, in its own kernel session, with
|
|
889
|
+
* stdin closed and both output streams going into the guest's retained
|
|
890
|
+
* log.
|
|
891
|
+
*
|
|
892
|
+
* This is not the SDK's `spawnDetached` and does not pretend to be: that
|
|
893
|
+
* hands back a host `ChildProcess`, which cannot cross a process
|
|
894
|
+
* boundary, and it stays absent here. What this returns is a NAME, and
|
|
895
|
+
* the name is what another host process comes back with.
|
|
896
|
+
*/
|
|
897
|
+
startDetached(options: KubernetesStartDetachedOptions): Promise<KubernetesSessionSummary>;
|
|
898
|
+
/**
|
|
899
|
+
* Read a session's output from `fromOffset` without attaching to it and
|
|
900
|
+
* without signalling anything — the chunk, the offset to come back with,
|
|
901
|
+
* the bytes the ring dropped before it, and the program's status.
|
|
902
|
+
*/
|
|
903
|
+
readSession(sessionId: string, options?: KubernetesReadSessionOptions): Promise<KubernetesSessionOutput>;
|
|
904
|
+
/**
|
|
905
|
+
* Every session this workspace's pod is holding, running and recently
|
|
906
|
+
* exited. Empty after a suspend and resume: the registry is the pod's
|
|
907
|
+
* memory, and a resumed workspace is a new pod.
|
|
908
|
+
*/
|
|
909
|
+
listSessions(options?: KubernetesWorkspaceTransitionOptions): Promise<readonly KubernetesSessionSummary[]>;
|
|
910
|
+
/**
|
|
911
|
+
* End one session and everything still in it — the shell, the jobs it
|
|
912
|
+
* backgrounded, and whatever a detached session started. Idempotent, and
|
|
913
|
+
* the reply says what the session's state actually is, so a program that
|
|
914
|
+
* ignored a `SIGTERM` is reported still running rather than reported
|
|
915
|
+
* dead.
|
|
916
|
+
*
|
|
917
|
+
* `options.signal` is the POSIX signal to send (`SIGKILL` by default);
|
|
918
|
+
* `options.abort` is the `AbortSignal` that cancels the request.
|
|
919
|
+
*/
|
|
920
|
+
killSession(sessionId: string, options?: KubernetesKillSessionOptions): Promise<KubernetesSessionSummary>;
|
|
258
921
|
openTcpConnection(options: SandboxTcpConnectOptions): Promise<SandboxTcpConnection>;
|
|
922
|
+
/**
|
|
923
|
+
* Narrowed to present, like the two above: the pod behind a workspace runs
|
|
924
|
+
* the same guest agent a task sandbox does, so bounded file discovery is
|
|
925
|
+
* always available here and a caller composing this handle does not have
|
|
926
|
+
* to re-check for a method the backend always defines. The SDK's `glob`
|
|
927
|
+
* and `grep` builtins refuse a sandbox that omits it.
|
|
928
|
+
*/
|
|
929
|
+
walkFiles(rootPath: string, options: SandboxWalkFilesOptions): AsyncIterable<SandboxFileEntry>;
|
|
930
|
+
/**
|
|
931
|
+
* Narrowed to PRESENT, like the three above: the pod behind a workspace
|
|
932
|
+
* runs this repository's agent, so the streamed read is never absent
|
|
933
|
+
* here and a host draining a large output file before `suspend()` or
|
|
934
|
+
* `destroy()` — the use a long-lived workspace exists for — does not
|
|
935
|
+
* have to check for it.
|
|
936
|
+
*/
|
|
937
|
+
readFileStream(path: string, options?: SandboxReadFileOptions): AsyncIterable<Buffer>;
|
|
938
|
+
/**
|
|
939
|
+
* Re-read `spec.operatingMode` and believe it: a workspace ANOTHER
|
|
940
|
+
* process suspended reports `suspended: true` afterwards, and `resume()`
|
|
941
|
+
* brings it back on the same disk.
|
|
942
|
+
*
|
|
943
|
+
* A workspace id is a name, not a lock — two host processes can hold
|
|
944
|
+
* handles to one workspace — and everything else on this handle reports
|
|
945
|
+
* what THIS process did. Without this verb the second process has no way
|
|
946
|
+
* to ask, and its handle goes on claiming to be running with no pod
|
|
947
|
+
* behind it until a call fails.
|
|
948
|
+
*
|
|
949
|
+
* The foreign suspend is recorded as UNCONFIRMED, not as finished: what
|
|
950
|
+
* was observed is the object's mode, not the pod stopping, so a later
|
|
951
|
+
* `suspend()` on this handle still sends its own patch and waits for the
|
|
952
|
+
* pod rather than returning on what this read saw. Every terminal this
|
|
953
|
+
* handle handed out is killed, because the pod they live in is being
|
|
954
|
+
* taken away.
|
|
955
|
+
*
|
|
956
|
+
* It notices a suspension and nothing else. A workspace that reads
|
|
957
|
+
* `Running` while this handle is suspended is NOT taken back — coming
|
|
958
|
+
* back means binding a new pod, reading its token and probing it, which
|
|
959
|
+
* is what `resume()` is. A workspace somebody DELETED rejects with the
|
|
960
|
+
* client's already-gone error and changes nothing here: there is no state
|
|
961
|
+
* on this handle that means "another process deleted it", and inventing
|
|
962
|
+
* one to report it is a larger change than this verb.
|
|
963
|
+
*/
|
|
964
|
+
refresh(options?: KubernetesWorkspaceTransitionOptions): Promise<void>;
|
|
965
|
+
/**
|
|
966
|
+
* Stop every process this workspace's pod is running, and go on serving.
|
|
967
|
+
*
|
|
968
|
+
* `suspend()` on its own reaches two kinds of process: the terminals THIS
|
|
969
|
+
* handle returned, and a command somebody cancelled by id. A terminal
|
|
970
|
+
* another host process opened, a command already in flight, and above all
|
|
971
|
+
* a program that moved into a session of its own with `setsid` and was
|
|
972
|
+
* then reparented away from the agent all keep running — and keep writing
|
|
973
|
+
* to the disk — until the pod stops. By then there is no agent left to
|
|
974
|
+
* read that disk through.
|
|
975
|
+
*
|
|
976
|
+
* So this is the verb for a host that wants the disk still WHILE IT CAN
|
|
977
|
+
* STILL READ IT: afterwards the guest is quiet and the agent is up, so
|
|
978
|
+
* `exec`, `readFile`, `readFileStream` and `writeFile` all work and what
|
|
979
|
+
* they see is a filesystem nobody is writing to. A capture taken here can
|
|
980
|
+
* be trusted; `destroy({ deleteDisk: true })` or
|
|
981
|
+
* {@link deleteKubernetesWorkspace} can then run against a workspace
|
|
982
|
+
* nobody has to wake again to check.
|
|
983
|
+
*
|
|
984
|
+
* What it costs is everything running: an open terminal receives its
|
|
985
|
+
* exit, a running `exec` resolves with a signal in its result, and a
|
|
986
|
+
* session in the registry is reported `exited` rather than detached. A
|
|
987
|
+
* second call straight after the first stops nothing and says so, with an
|
|
988
|
+
* empty list.
|
|
989
|
+
*
|
|
990
|
+
* Admitted only while the workspace is running, and serialised with the
|
|
991
|
+
* lifecycle transitions, so it cannot interleave with a suspend or a
|
|
992
|
+
* resume. It rejects — changing nothing on the cluster — with
|
|
993
|
+
* `KubernetesQuiesceUnconfirmedError` when a process would not stop (the
|
|
994
|
+
* message names its pid), and with `KubernetesQuiesceUnsupportedError`
|
|
995
|
+
* against an image whose agent predates the op.
|
|
996
|
+
*/
|
|
997
|
+
quiesce(options?: KubernetesQuiesceOptions): Promise<KubernetesQuiesceReport>;
|
|
998
|
+
/**
|
|
999
|
+
* Put everything this workspace has written onto its disk, now.
|
|
1000
|
+
*
|
|
1001
|
+
* `syncfs(2)` over the workspace mount inside the guest: every dirty
|
|
1002
|
+
* page of every file on it, whichever process wrote it. That covers what
|
|
1003
|
+
* a COMMAND wrote, which is the half no per-file fsync can reach —
|
|
1004
|
+
* `writeFile` fsyncs its own bytes before it answers, and a compiler's
|
|
1005
|
+
* output belongs to nobody here.
|
|
1006
|
+
*
|
|
1007
|
+
* `suspend()` runs this by default, so a caller that only suspends never
|
|
1008
|
+
* needs it. It is a verb of its own for the moments that are not a
|
|
1009
|
+
* suspend: before a snapshot somebody else takes, before a node is
|
|
1010
|
+
* drained, or beside a {@link quiesce} when a host wants the disk both
|
|
1011
|
+
* still AND written back while the workspace keeps running.
|
|
1012
|
+
*
|
|
1013
|
+
* Admitted only while the workspace is running, and serialised with the
|
|
1014
|
+
* lifecycle transitions like `quiesce()` is. It rejects — changing
|
|
1015
|
+
* nothing on the cluster — with `KubernetesFlushUnconfirmedError` when
|
|
1016
|
+
* the guest answered and could not confirm, and with
|
|
1017
|
+
* `KubernetesFlushUnsupportedError` against an image whose agent
|
|
1018
|
+
* predates the op.
|
|
1019
|
+
*/
|
|
1020
|
+
flush(options?: KubernetesFlushOptions): Promise<KubernetesFlushReport>;
|
|
259
1021
|
/**
|
|
260
1022
|
* Give the compute back and keep the disk. Idempotent: suspending a
|
|
261
1023
|
* workspace whose suspend has been CONFIRMED sends nothing.
|
|
@@ -275,12 +1037,52 @@ export interface KubernetesWorkspace extends Sandbox {
|
|
|
275
1037
|
* A call already in flight when this is called is not cancelled: it fails
|
|
276
1038
|
* at the transport when the pod goes away, rather than with the named
|
|
277
1039
|
* suspended error, which only covers calls admitted from here on.
|
|
1040
|
+
*
|
|
1041
|
+
* It also runs {@link flush} before the patch unless
|
|
1042
|
+
* {@link KubernetesWorkspaceSuspendOptions.flush} says otherwise, so what
|
|
1043
|
+
* the workspace wrote is on the device before the pod is taken away
|
|
1044
|
+
* rather than left to whatever the guest kernel had written back. A flush
|
|
1045
|
+
* the guest ANSWERED and could not confirm rejects and sends NO patch —
|
|
1046
|
+
* the only flush outcome that stops a suspend. An image that cannot
|
|
1047
|
+
* flush at all, and a guest that cannot be reached to be asked, both
|
|
1048
|
+
* keep the old behaviour: the suspend goes ahead and the host is told
|
|
1049
|
+
* through `onFlushUnsupported` or `onFlushUnreachable`, because a
|
|
1050
|
+
* workspace nobody can reach is exactly the one an operator most needs
|
|
1051
|
+
* this verb for.
|
|
1052
|
+
*
|
|
1053
|
+
* `suspend({ quiesce: true })` runs {@link quiesce} first — after this
|
|
1054
|
+
* handle's own terminals are reaped and BEFORE the patch and before that
|
|
1055
|
+
* flush — so the disk is still at the moment the pod is asked to stop
|
|
1056
|
+
* rather than merely by the time it has. A quiesce that cannot be confirmed rejects and sends NO
|
|
1057
|
+
* patch: the workspace stays running and admits calls, which is the rule
|
|
1058
|
+
* this verb already keeps everywhere else — leave the state that is true.
|
|
1059
|
+
*
|
|
1060
|
+
* Sharing a transition has ONE exception, and it is this option. A call
|
|
1061
|
+
* arriving mid-suspend joins the transition in flight rather than
|
|
1062
|
+
* starting one of its own, under the first caller's signal and epoch —
|
|
1063
|
+
* but a `quiesce` the transition in flight is not performing cannot be
|
|
1064
|
+
* joined into: that transition is on its way to patching over a guest
|
|
1065
|
+
* nothing has stopped, and once its state leaves `running` no call is
|
|
1066
|
+
* admitted to stop anything. Such a caller is REJECTED with
|
|
1067
|
+
* {@link KubernetesQuiesceUnconfirmedError} instead of being handed a
|
|
1068
|
+
* resolved suspend it would trust a capture on. A caller whose request
|
|
1069
|
+
* the flight already satisfies still joins it.
|
|
278
1070
|
*/
|
|
279
|
-
suspend(options?:
|
|
1071
|
+
suspend(options?: KubernetesWorkspaceSuspendOptions): Promise<void>;
|
|
280
1072
|
/**
|
|
281
1073
|
* Take a new pod, on a new address, with a new agent token, and prove it
|
|
282
1074
|
* is deprivileged before handing it back. Idempotent: resuming a running
|
|
283
1075
|
* workspace sends nothing.
|
|
1076
|
+
*
|
|
1077
|
+
* With
|
|
1078
|
+
* {@link KubernetesWorkspaceTransitionOptions.refreshPodTemplate}, the
|
|
1079
|
+
* wake-up patch also writes the SandboxTemplate's current
|
|
1080
|
+
* `spec.podTemplate`, so the new pod is built from the template as it
|
|
1081
|
+
* stands rather than as it stood when the workspace was created. The disk
|
|
1082
|
+
* is untouched — `spec.volumeClaimTemplates` is CEL-immutable and is not
|
|
1083
|
+
* in the patch — and a template that no longer claims this workspace's
|
|
1084
|
+
* disk is refused with {@link KubernetesWorkspaceDiskError} before
|
|
1085
|
+
* anything is sent.
|
|
284
1086
|
*/
|
|
285
1087
|
resume(options?: KubernetesWorkspaceTransitionOptions): Promise<void>;
|
|
286
1088
|
/**
|
|
@@ -315,6 +1117,167 @@ export interface KubernetesWorkspaceOptions {
|
|
|
315
1117
|
*/
|
|
316
1118
|
readonly sandboxTemplateName?: string;
|
|
317
1119
|
readonly signal?: AbortSignal;
|
|
1120
|
+
/**
|
|
1121
|
+
* What a failed start may do to the workspace, for this handle and every
|
|
1122
|
+
* transition on it — see {@link KubernetesWorkspaceStartFailurePolicy}.
|
|
1123
|
+
* Default: `suspend-if-woken`.
|
|
1124
|
+
*/
|
|
1125
|
+
readonly onStartFailure?: KubernetesWorkspaceStartFailurePolicy;
|
|
1126
|
+
/**
|
|
1127
|
+
* Told when an execution's cancellation could not be CONFIRMED, with
|
|
1128
|
+
* what a bounded `healthz` found afterwards — see
|
|
1129
|
+
* {@link KubernetesWorkspaceCancellationNotice}.
|
|
1130
|
+
*
|
|
1131
|
+
* Nothing happens to the workspace on this path: the pod keeps running,
|
|
1132
|
+
* no patch is sent, the handle goes on admitting calls, and the failing
|
|
1133
|
+
* `exec()` rejects with `RemoteCancellationUnknownError` carrying
|
|
1134
|
+
* `retirement: { accepted: false, reason: 'workspace-kept' }`. A command
|
|
1135
|
+
* of unknown state is left in that pod, which is a fact the host has to
|
|
1136
|
+
* be able to act on — so it is reported here rather than acted on from
|
|
1137
|
+
* inside a failing call, where the only action available (the Suspended
|
|
1138
|
+
* patch) would delete the pod out from under every other holder.
|
|
1139
|
+
*
|
|
1140
|
+
* A host that wants the old behaviour calls `suspend()` from this
|
|
1141
|
+
* callback. One that wants it only for a genuinely fenced agent calls it
|
|
1142
|
+
* when `agent` is `retiring`.
|
|
1143
|
+
*
|
|
1144
|
+
* Not only from `exec()`. The acquire-time PRIVILEGE PROBE is itself an
|
|
1145
|
+
* execution on the same inner handle, so a create or a resume whose probe
|
|
1146
|
+
* loses its cancellation fires this too — and pays the diagnosis's own
|
|
1147
|
+
* bound before rejecting. A callback that suspends the workspace is
|
|
1148
|
+
* therefore reachable from a failing START as well as from a failing
|
|
1149
|
+
* call, which is worth knowing before it is wired to one.
|
|
1150
|
+
*
|
|
1151
|
+
* In the style of `onLeaseRenewalError`: synchronous, never awaited, and
|
|
1152
|
+
* a callback that throws changes nothing about the error the caller
|
|
1153
|
+
* receives.
|
|
1154
|
+
*/
|
|
1155
|
+
readonly onCancellationUnconfirmed?: (notice: KubernetesWorkspaceCancellationNotice) => void;
|
|
1156
|
+
/**
|
|
1157
|
+
* Told when a `suspend({ quiesce: true })` could not quiesce because the
|
|
1158
|
+
* guest's image predates the op, just before the suspend goes ahead
|
|
1159
|
+
* without one.
|
|
1160
|
+
*
|
|
1161
|
+
* This is the one gap that must not be silent. Everything else about a
|
|
1162
|
+
* quiesce is reported by the call that asked for it — `quiesce()` itself
|
|
1163
|
+
* refuses an image that cannot perform one, and a quiesce the guest
|
|
1164
|
+
* answered and could not confirm rejects the suspend — but a host that
|
|
1165
|
+
* passes `quiesce: true` and gets a suspend anyway would otherwise have
|
|
1166
|
+
* no way to learn that its capture was taken over a guest nothing had
|
|
1167
|
+
* stopped. `@namzu/sandbox` owns no logger and reads none from module
|
|
1168
|
+
* scope, so the diagnostic goes to the host that has one.
|
|
1169
|
+
*
|
|
1170
|
+
* In the style of `onLeaseRenewalError`: synchronous, never awaited, and
|
|
1171
|
+
* a callback that throws changes nothing about the suspend.
|
|
1172
|
+
*/
|
|
1173
|
+
readonly onQuiesceUnsupported?: (error: KubernetesQuiesceUnsupportedError) => void;
|
|
1174
|
+
/**
|
|
1175
|
+
* Told when a `suspend({ quiesce: true })` DID quiesce, and the guest
|
|
1176
|
+
* narrowed the scan to the kernel sessions its own registries own —
|
|
1177
|
+
* `report.scope === 'owned-sessions'` — just before the patch goes out.
|
|
1178
|
+
*
|
|
1179
|
+
* The sibling of `onQuiesceUnsupported`, and it exists for the same
|
|
1180
|
+
* reason: `quiesce()` hands its caller the report and that caller can
|
|
1181
|
+
* read the scope, but a `suspend({ quiesce: true })` returns `void`, so
|
|
1182
|
+
* the one path that cannot read the report would otherwise be the one
|
|
1183
|
+
* that silently under-delivers. A narrowed scan can miss exactly the
|
|
1184
|
+
* process this feature exists for — a program `setsid` moved out of
|
|
1185
|
+
* every session either registry holds — so a capture taken after one is
|
|
1186
|
+
* weaker than a capture taken after a `pid-namespace` scan.
|
|
1187
|
+
*
|
|
1188
|
+
* A pod built from this repo's image never narrows: `k8s/entrypoint.sh`
|
|
1189
|
+
* makes `tini` PID 1 and the agent its child. A derived image that wraps
|
|
1190
|
+
* the agent in something else, or an embedded agent, can.
|
|
1191
|
+
*
|
|
1192
|
+
* In the style of `onLeaseRenewalError`: synchronous, never awaited, and
|
|
1193
|
+
* a callback that throws changes nothing about the suspend.
|
|
1194
|
+
*/
|
|
1195
|
+
readonly onQuiesceNarrowed?: (report: KubernetesQuiesceReport) => void;
|
|
1196
|
+
/**
|
|
1197
|
+
* Told when a `suspend()` could not flush because the guest's image
|
|
1198
|
+
* predates the op, just before the suspend goes ahead without one.
|
|
1199
|
+
*
|
|
1200
|
+
* The exact sibling of {@link onQuiesceUnsupported}, and the one gap on
|
|
1201
|
+
* the flush path that must not be silent for the same reason: a flush is
|
|
1202
|
+
* ON by default here, so a deployment whose images are older than this
|
|
1203
|
+
* release gets a suspend that behaves exactly as it always did, and
|
|
1204
|
+
* nothing would say that the writes it thought were being put on the
|
|
1205
|
+
* device were not. Every other flush failure is reported by the call
|
|
1206
|
+
* itself — an explicit `flush()` refuses an image that cannot do it, and
|
|
1207
|
+
* a flush the guest answered and could not confirm rejects the suspend
|
|
1208
|
+
* before any patch is sent.
|
|
1209
|
+
*
|
|
1210
|
+
* In the style of `onLeaseRenewalError`: synchronous, never awaited, and
|
|
1211
|
+
* a callback that throws changes nothing about the suspend.
|
|
1212
|
+
*/
|
|
1213
|
+
readonly onFlushUnsupported?: (error: KubernetesFlushUnsupportedError) => void;
|
|
1214
|
+
/**
|
|
1215
|
+
* Told when a `suspend()` could not ASK for a flush — the guest could
|
|
1216
|
+
* not be reached, or refused to answer — just before the suspend goes
|
|
1217
|
+
* ahead without one.
|
|
1218
|
+
*
|
|
1219
|
+
* The sibling of {@link onFlushUnsupported}, and the reason `suspend()`
|
|
1220
|
+
* is still the verb it always was. A flush is ON by default, so without
|
|
1221
|
+
* this a workspace whose agent has crashed, been OOM-killed, lost its
|
|
1222
|
+
* network or fenced itself could no longer be suspended at all on the
|
|
1223
|
+
* default path: it would keep running, and keep costing, behind a raw
|
|
1224
|
+
* `ECONNREFUSED` that named neither the flush nor a way past it. A guest
|
|
1225
|
+
* nothing can reach does not become flushable by being left alone, and
|
|
1226
|
+
* `suspend()` then `resume()` is precisely what
|
|
1227
|
+
* `KubernetesAgentRetiringError` tells a caller to do about a fenced
|
|
1228
|
+
* one.
|
|
1229
|
+
*
|
|
1230
|
+
* So the suspend proceeds and the gap is REPORTED. What it means for the
|
|
1231
|
+
* disk is in the error's message: the writes that reached it are the
|
|
1232
|
+
* ones the guest kernel had already written back, plus whatever the
|
|
1233
|
+
* pod's `preStop` hook and the agent's own `SIGTERM` handler manage
|
|
1234
|
+
* while the pod stops. A caller that would rather stop can throw from
|
|
1235
|
+
* here — the callback's exception is swallowed like every other
|
|
1236
|
+
* callback's here, so the way to refuse is to suspend under
|
|
1237
|
+
* `{ flush: false }` only after a `flush()` of its own has succeeded.
|
|
1238
|
+
*
|
|
1239
|
+
* A guest that ANSWERS and cannot confirm is the other case and is not
|
|
1240
|
+
* this one: that rejects the suspend with
|
|
1241
|
+
* {@link KubernetesFlushUnconfirmedError} and sends no patch.
|
|
1242
|
+
*
|
|
1243
|
+
* In the style of `onLeaseRenewalError`: synchronous, never awaited, and
|
|
1244
|
+
* a callback that throws changes nothing about the suspend.
|
|
1245
|
+
*/
|
|
1246
|
+
readonly onFlushUnreachable?: (error: KubernetesFlushUnreachableError) => void;
|
|
1247
|
+
/**
|
|
1248
|
+
* The holder epoch this call opens the workspace under, and the one the
|
|
1249
|
+
* handle keeps — see {@link HOLDER_EPOCH_ANNOTATION_KEY}.
|
|
1250
|
+
*
|
|
1251
|
+
* On the CREATE path the POST stamps it, so the workspace is fenced from
|
|
1252
|
+
* the moment it exists. On the ADOPT path it is checked against the
|
|
1253
|
+
* stored epoch before anything is patched — an opener the workspace has
|
|
1254
|
+
* moved past is refused with
|
|
1255
|
+
* {@link KubernetesWorkspacePreconditionError} and wakes nothing — and
|
|
1256
|
+
* then written, even when the object was already `Running` and today
|
|
1257
|
+
* nothing would be sent: taking a workspace over is exactly the moment
|
|
1258
|
+
* the fence has to move.
|
|
1259
|
+
*
|
|
1260
|
+
* The handle then uses it for every write it sends when the call passes
|
|
1261
|
+
* none: `suspend()`, `destroy()`, and the cleanup patch a failed start
|
|
1262
|
+
* sends. Omitted, nothing is fenced and every request is what it was.
|
|
1263
|
+
*/
|
|
1264
|
+
readonly epoch?: number;
|
|
1265
|
+
/**
|
|
1266
|
+
* On the ADOPT path, write the SandboxTemplate's current
|
|
1267
|
+
* `spec.podTemplate` onto the workspace as it wakes — the same opt-in
|
|
1268
|
+
* {@link KubernetesWorkspaceTransitionOptions.refreshPodTemplate}
|
|
1269
|
+
* describes, reachable from an entry point that needs no handle.
|
|
1270
|
+
*
|
|
1271
|
+
* It exists here because a host that restarted has no handle to call
|
|
1272
|
+
* `resume()` on: the workspace is found again by name, and this call is
|
|
1273
|
+
* the only place the decision can be made. It is honoured on exactly the
|
|
1274
|
+
* transition the handle's is — an object found `Suspended` going back to
|
|
1275
|
+
* `Running` — and is ignored on the CREATE path, where the POST already
|
|
1276
|
+
* carries the template as it stands.
|
|
1277
|
+
*
|
|
1278
|
+
* Omitted or `false`, an adopt behaves exactly as it always did.
|
|
1279
|
+
*/
|
|
1280
|
+
readonly refreshPodTemplate?: boolean;
|
|
318
1281
|
}
|
|
319
1282
|
/** Prefix every workspace Sandbox's name carries. */
|
|
320
1283
|
export declare const WORKSPACE_NAME_PREFIX = "namzu-ws-";
|
|
@@ -349,9 +1312,32 @@ export declare function assertBlockModeWorkspaceDisk(source: string, podTemplate
|
|
|
349
1312
|
* object whose label says one thing while the caller builds from another is a
|
|
350
1313
|
* mix-up worth naming the first time it is seen rather than the first time a
|
|
351
1314
|
* policy is switched on. See {@link KubernetesWorkspaceMismatchError}.
|
|
1315
|
+
*
|
|
1316
|
+
* The PROFILE check is conditional, because a profile is what this
|
|
1317
|
+
* configuration asks for rather than something every object has: it runs when
|
|
1318
|
+
* `config.egress.profile` is set, and then the object's own pod template has
|
|
1319
|
+
* to carry that key with that value — unless this call is about to REWRITE
|
|
1320
|
+
* that pod template (`refreshPodTemplate: true` onto a Suspended object),
|
|
1321
|
+
* which writes the configured profile with the rest of the overlays. That is
|
|
1322
|
+
* the same lifting the `runtimeClassName` check gets, for the same reason and
|
|
1323
|
+
* with the same re-application if the patch does not land.
|
|
1324
|
+
*
|
|
1325
|
+
* Both halves of the policy selector are therefore checked against the object
|
|
1326
|
+
* on this path — which matters because neither network check further up can
|
|
1327
|
+
* do it: the ingress verification and the egress union both run against the
|
|
1328
|
+
* labels the POST *would* stamp, and on this path the POST already lost to a
|
|
1329
|
+
* 409. What this does NOT check is a
|
|
1330
|
+
* label the standing object carries that this configuration does not ask for
|
|
1331
|
+
* — an object built under a profile this config has dropped, say. That pod is
|
|
1332
|
+
* still selected by the unprofiled policy this call verified (a selector is a
|
|
1333
|
+
* subset match), so the boundary it reports holds; what a stale label could
|
|
1334
|
+
* do is bring a SECOND policy into the union, and the adopt path cannot
|
|
1335
|
+
* enumerate the standing object's labels without reading them, which is a
|
|
1336
|
+
* wider change than this refusal.
|
|
352
1337
|
*/
|
|
353
1338
|
export declare function assertAdoptedWorkspaceMatchesConfig(sandboxName: string, namespace: string, podTemplate: SandboxPodTemplate | undefined, expected: {
|
|
354
1339
|
readonly sandboxTemplateName: string;
|
|
1340
|
+
readonly profile?: EgressProfileLabel;
|
|
355
1341
|
readonly runtimeClassName?: string;
|
|
356
1342
|
}): void;
|
|
357
1343
|
/**
|
|
@@ -378,4 +1364,149 @@ export declare function assertAdoptedWorkspaceMatchesConfig(sandboxName: string,
|
|
|
378
1364
|
* name and nothing reaps for them — see the docs page.
|
|
379
1365
|
*/
|
|
380
1366
|
export declare function createKubernetesWorkspace(config: KubernetesBackendInternalConfig, options: KubernetesWorkspaceOptions): Promise<KubernetesWorkspace>;
|
|
1367
|
+
/**
|
|
1368
|
+
* One workspace as an INVENTORY reads it: enough to decide what to do with
|
|
1369
|
+
* it, and not one field that required waking it up.
|
|
1370
|
+
*
|
|
1371
|
+
* Everything here comes off the Sandbox object itself. No pod is read, no
|
|
1372
|
+
* agent is dialed, and no transport exists — which is the whole point, since
|
|
1373
|
+
* the caller this exists for is a retention pass over workspaces that have
|
|
1374
|
+
* been asleep for a month and must stay asleep.
|
|
1375
|
+
*/
|
|
1376
|
+
export interface KubernetesWorkspaceSummary {
|
|
1377
|
+
/** The caller's own id — the Sandbox's name with `namzu-ws-` taken off. */
|
|
1378
|
+
readonly workspaceId: string;
|
|
1379
|
+
/**
|
|
1380
|
+
* `spec.operatingMode`, verbatim. NOT "is a pod running": a `Running`
|
|
1381
|
+
* workspace whose pod is still being created, or has just crashed, reads
|
|
1382
|
+
* `Running` here, because this is the mode the object was last asked to
|
|
1383
|
+
* be in.
|
|
1384
|
+
*/
|
|
1385
|
+
readonly operatingMode: 'Running' | 'Suspended';
|
|
1386
|
+
/** The `SandboxTemplate` this workspace was built from — the pod label. */
|
|
1387
|
+
readonly template: string;
|
|
1388
|
+
/**
|
|
1389
|
+
* `metadata.creationTimestamp`, RFC 3339. Optional only because it is
|
|
1390
|
+
* read off an object rather than promised by this type: the API server
|
|
1391
|
+
* always sets it.
|
|
1392
|
+
*/
|
|
1393
|
+
readonly createdAt?: string;
|
|
1394
|
+
/**
|
|
1395
|
+
* When this backend last patched `spec.operatingMode`, RFC 3339 — the
|
|
1396
|
+
* {@link OPERATING_MODE_CHANGED_AT_ANNOTATION_KEY} annotation.
|
|
1397
|
+
*
|
|
1398
|
+
* ABSENT on a workspace whose mode has never been changed since it was
|
|
1399
|
+
* created, and on one created before this backend started stamping it.
|
|
1400
|
+
* It is reported absent rather than defaulted to `createdAt`, because a
|
|
1401
|
+
* retention rule that deletes "anything not touched for 30 days" must be
|
|
1402
|
+
* able to tell "never suspended" from "suspended a month ago".
|
|
1403
|
+
*/
|
|
1404
|
+
readonly operatingModeChangedAt?: string;
|
|
1405
|
+
/**
|
|
1406
|
+
* The holder epoch stored on the Sandbox — see
|
|
1407
|
+
* {@link HOLDER_EPOCH_ANNOTATION_KEY}.
|
|
1408
|
+
*
|
|
1409
|
+
* `0` on a workspace that carries no such annotation, because that is
|
|
1410
|
+
* what every write compares against: a workspace nobody has fenced is
|
|
1411
|
+
* held at epoch 0 and the next write of any epoch takes it. ABSENT only
|
|
1412
|
+
* when the annotation is present and unreadable — not a decimal integer,
|
|
1413
|
+
* which no release of this backend writes — because reporting that as 0
|
|
1414
|
+
* would tell a retention pass a workspace is free when the next write
|
|
1415
|
+
* against it will be refused.
|
|
1416
|
+
*
|
|
1417
|
+
* Reported rather than conditioned: a list sends no write, so it has
|
|
1418
|
+
* nothing for an epoch to fence, and what an inventory needs is to SEE
|
|
1419
|
+
* the fences it is looking at.
|
|
1420
|
+
*/
|
|
1421
|
+
readonly holderEpoch?: number;
|
|
1422
|
+
}
|
|
1423
|
+
/**
|
|
1424
|
+
* Every workspace this backend owns in the namespace, WITHOUT waking one.
|
|
1425
|
+
*
|
|
1426
|
+
* The verb retention needs. Deleting a workspace nobody has resumed since
|
|
1427
|
+
* last month, or reporting what a namespace is holding, used to require
|
|
1428
|
+
* {@link createKubernetesWorkspace} — which adopts AND resumes: the inventory
|
|
1429
|
+
* pass would start a pod for every suspended workspace it looked at, probe
|
|
1430
|
+
* each one, and then have to put them all back. This issues exactly one GET
|
|
1431
|
+
* of the sandboxes collection and reads the objects.
|
|
1432
|
+
*
|
|
1433
|
+
* ## What counts as a workspace
|
|
1434
|
+
*
|
|
1435
|
+
* Two things together, and both are this backend's own marks:
|
|
1436
|
+
*
|
|
1437
|
+
* - the `namzu-ws-` name prefix — {@link workspaceSandboxName}'s contract,
|
|
1438
|
+
* and the only thing that makes a `workspaceId` recoverable from a
|
|
1439
|
+
* Sandbox at all;
|
|
1440
|
+
* - {@link SANDBOX_TEMPLATE_LABEL_KEY} on `spec.podTemplate.metadata.labels`,
|
|
1441
|
+
* which says this object was built by this backend and names the template
|
|
1442
|
+
* it came from.
|
|
1443
|
+
*
|
|
1444
|
+
* The filtering happens HERE rather than in a `labelSelector` on the request,
|
|
1445
|
+
* and the reason is where that label lives. It is a POD label — the one an
|
|
1446
|
+
* egress `NetworkPolicy`'s `podSelector` matches — written onto
|
|
1447
|
+
* `spec.podTemplate`, while a `labelSelector` on the sandboxes collection
|
|
1448
|
+
* matches the Sandbox's OWN `metadata.labels`, which this backend has never
|
|
1449
|
+
* written. Stamping a second copy up there to make a server-side selector
|
|
1450
|
+
* work would leave every workspace created before that change invisible to
|
|
1451
|
+
* this call, and an inventory that silently omits the oldest objects is worse
|
|
1452
|
+
* than no inventory at all — those are exactly the ones a retention pass is
|
|
1453
|
+
* looking for.
|
|
1454
|
+
*
|
|
1455
|
+
* ## What it never does
|
|
1456
|
+
*
|
|
1457
|
+
* No PATCH, no DELETE, no pod read, no dial. A workspace that was suspended
|
|
1458
|
+
* before this call is suspended after it, and a running one is untouched. The
|
|
1459
|
+
* order is the API server's own (name order); the caller sorts if it cares.
|
|
1460
|
+
*/
|
|
1461
|
+
export declare function listKubernetesWorkspaces(config: KubernetesBackendInternalConfig, options?: KubernetesWorkspaceTransitionOptions): Promise<readonly KubernetesWorkspaceSummary[]>;
|
|
1462
|
+
/**
|
|
1463
|
+
* DELETE a workspace by id, without ever adopting or resuming it.
|
|
1464
|
+
*
|
|
1465
|
+
* The same thing `destroy({ deleteDisk: true })` does to the cluster — one
|
|
1466
|
+
* DELETE of the Sandbox, which cascades to the Pod, the Service and the PVC
|
|
1467
|
+
* through ownerReferences — with the same two guarantees: an object already
|
|
1468
|
+
* gone counts as deleted, that being the state DELETE was asking for, and a
|
|
1469
|
+
* DELETE that FAILS rejects and stays retryable, because nothing here records
|
|
1470
|
+
* a state a retry could early-return on.
|
|
1471
|
+
*
|
|
1472
|
+
* What it does NOT do is the point. Removing a month-old suspended workspace
|
|
1473
|
+
* through a handle meant starting a pod for it first, probing it, and then
|
|
1474
|
+
* deleting the pod again — compute spent, and a guest woken, purely to be
|
|
1475
|
+
* told to go away. The DELETE never needed any of that: the name is
|
|
1476
|
+
* deterministic, so the object can be addressed without being opened.
|
|
1477
|
+
*
|
|
1478
|
+
* It is not gated on the workspace being suspended, and deliberately: a
|
|
1479
|
+
* running workspace's DELETE takes its pod down with it, which is what
|
|
1480
|
+
* deleting a workspace means. A caller that wants the disk quiesced first
|
|
1481
|
+
* calls {@link suspendKubernetesWorkspace} and then this.
|
|
1482
|
+
*/
|
|
1483
|
+
export declare function deleteKubernetesWorkspace(config: KubernetesBackendInternalConfig, workspaceId: string, options?: KubernetesWorkspaceTransitionOptions): Promise<void>;
|
|
1484
|
+
/**
|
|
1485
|
+
* Suspend a workspace by id, without ever adopting it: send the patch, then
|
|
1486
|
+
* wait for the pod to actually stop.
|
|
1487
|
+
*
|
|
1488
|
+
* The handle's own `suspend()` does three things — reap the terminals it
|
|
1489
|
+
* handed out, stop admitting calls, and patch-and-wait. Only the third is
|
|
1490
|
+
* about the CLUSTER, and it is the only one a process holding no handle can
|
|
1491
|
+
* do or needs to. So this is that third thing on its own, for the operator
|
|
1492
|
+
* putting somebody else's workspace to sleep.
|
|
1493
|
+
*
|
|
1494
|
+
* It resolves only once the pod is gone or in a terminal phase, for the same
|
|
1495
|
+
* reason the handle's does: a suspend is a promise that the disk is quiesced,
|
|
1496
|
+
* and the patch being accepted says only that the controller has been asked.
|
|
1497
|
+
* A pod that outlives `readyTimeoutMs` rejects with
|
|
1498
|
+
* {@link KubernetesWorkspaceSuspendTimeoutError} and the object is left
|
|
1499
|
+
* exactly as the patch left it — the next call patches and waits again.
|
|
1500
|
+
*
|
|
1501
|
+
* A workspace that no longer exists rejects with the client's own
|
|
1502
|
+
* already-gone error rather than resolving. Unlike a DELETE, this asks for a
|
|
1503
|
+
* state that cannot be reached: there is no object to suspend, and nothing
|
|
1504
|
+
* the caller believed about it holds.
|
|
1505
|
+
*
|
|
1506
|
+
* Any HANDLE another process is holding on this workspace is not told. It
|
|
1507
|
+
* finds out on its next call, which fails at the transport and is re-read
|
|
1508
|
+
* into a {@link KubernetesWorkspaceSuspendedError} — or the moment that
|
|
1509
|
+
* process calls `refresh()`. See {@link KubernetesWorkspace.refresh}.
|
|
1510
|
+
*/
|
|
1511
|
+
export declare function suspendKubernetesWorkspace(config: KubernetesBackendInternalConfig, workspaceId: string, options?: KubernetesWorkspaceSuspendOptions): Promise<void>;
|
|
381
1512
|
//# sourceMappingURL=workspace.d.ts.map
|