@namzu/sandbox 13.0.0 → 15.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (104) hide show
  1. package/CHANGELOG.md +1147 -0
  2. package/README.md +447 -0
  3. package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
  4. package/dist/backends/aci-standby-pool/index.js +13 -1
  5. package/dist/backends/aci-standby-pool/index.js.map +1 -1
  6. package/dist/backends/docker/index.d.ts.map +1 -1
  7. package/dist/backends/docker/index.js +19 -1
  8. package/dist/backends/docker/index.js.map +1 -1
  9. package/dist/backends/firecracker/index.d.ts.map +1 -1
  10. package/dist/backends/firecracker/index.js +12 -2
  11. package/dist/backends/firecracker/index.js.map +1 -1
  12. package/dist/backends/firecracker/protocol.d.ts +481 -8
  13. package/dist/backends/firecracker/protocol.d.ts.map +1 -1
  14. package/dist/backends/firecracker/protocol.js +136 -0
  15. package/dist/backends/firecracker/protocol.js.map +1 -1
  16. package/dist/backends/firecracker/transport.d.ts +642 -14
  17. package/dist/backends/firecracker/transport.d.ts.map +1 -1
  18. package/dist/backends/firecracker/transport.js +1307 -34
  19. package/dist/backends/firecracker/transport.js.map +1 -1
  20. package/dist/backends/kubernetes/egress-policy.d.ts +1296 -0
  21. package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -0
  22. package/dist/backends/kubernetes/egress-policy.js +2458 -0
  23. package/dist/backends/kubernetes/egress-policy.js.map +1 -0
  24. package/dist/backends/kubernetes/identity.d.ts +193 -0
  25. package/dist/backends/kubernetes/identity.d.ts.map +1 -0
  26. package/dist/backends/kubernetes/identity.js +147 -0
  27. package/dist/backends/kubernetes/identity.js.map +1 -0
  28. package/dist/backends/kubernetes/index.d.ts +1019 -0
  29. package/dist/backends/kubernetes/index.d.ts.map +1 -0
  30. package/dist/backends/kubernetes/index.js +1756 -0
  31. package/dist/backends/kubernetes/index.js.map +1 -0
  32. package/dist/backends/kubernetes/ingress-policy.d.ts +375 -0
  33. package/dist/backends/kubernetes/ingress-policy.d.ts.map +1 -0
  34. package/dist/backends/kubernetes/ingress-policy.js +1050 -0
  35. package/dist/backends/kubernetes/ingress-policy.js.map +1 -0
  36. package/dist/backends/kubernetes/k8s-client.d.ts +334 -0
  37. package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -0
  38. package/dist/backends/kubernetes/k8s-client.js +553 -0
  39. package/dist/backends/kubernetes/k8s-client.js.map +1 -0
  40. package/dist/backends/kubernetes/lease.d.ts +145 -0
  41. package/dist/backends/kubernetes/lease.d.ts.map +1 -0
  42. package/dist/backends/kubernetes/lease.js +201 -0
  43. package/dist/backends/kubernetes/lease.js.map +1 -0
  44. package/dist/backends/kubernetes/objects.d.ts +702 -0
  45. package/dist/backends/kubernetes/objects.d.ts.map +1 -0
  46. package/dist/backends/kubernetes/objects.js +518 -0
  47. package/dist/backends/kubernetes/objects.js.map +1 -0
  48. package/dist/backends/kubernetes/per-sandbox-policy.d.ts +219 -0
  49. package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -0
  50. package/dist/backends/kubernetes/per-sandbox-policy.js +407 -0
  51. package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -0
  52. package/dist/backends/kubernetes/privilege-probe.d.ts +136 -0
  53. package/dist/backends/kubernetes/privilege-probe.d.ts.map +1 -0
  54. package/dist/backends/kubernetes/privilege-probe.js +185 -0
  55. package/dist/backends/kubernetes/privilege-probe.js.map +1 -0
  56. package/dist/backends/kubernetes/rbac.d.ts +153 -0
  57. package/dist/backends/kubernetes/rbac.d.ts.map +1 -0
  58. package/dist/backends/kubernetes/rbac.js +177 -0
  59. package/dist/backends/kubernetes/rbac.js.map +1 -0
  60. package/dist/backends/kubernetes/sandbox.d.ts +190 -0
  61. package/dist/backends/kubernetes/sandbox.d.ts.map +1 -0
  62. package/dist/backends/kubernetes/sandbox.js +433 -0
  63. package/dist/backends/kubernetes/sandbox.js.map +1 -0
  64. package/dist/backends/kubernetes/transport.d.ts +1048 -0
  65. package/dist/backends/kubernetes/transport.d.ts.map +1 -0
  66. package/dist/backends/kubernetes/transport.js +2093 -0
  67. package/dist/backends/kubernetes/transport.js.map +1 -0
  68. package/dist/backends/kubernetes/workspace.d.ts +1512 -0
  69. package/dist/backends/kubernetes/workspace.d.ts.map +1 -0
  70. package/dist/backends/kubernetes/workspace.js +3703 -0
  71. package/dist/backends/kubernetes/workspace.js.map +1 -0
  72. package/dist/backends/remote-execution-controller.d.ts +14 -0
  73. package/dist/backends/remote-execution-controller.d.ts.map +1 -1
  74. package/dist/backends/remote-execution-controller.js.map +1 -1
  75. package/dist/index.d.ts +350 -2
  76. package/dist/index.d.ts.map +1 -1
  77. package/dist/index.js +344 -34
  78. package/dist/index.js.map +1 -1
  79. package/dist/testing/sandbox-conformance.d.ts +227 -0
  80. package/dist/testing/sandbox-conformance.d.ts.map +1 -0
  81. package/dist/testing/sandbox-conformance.js +896 -0
  82. package/dist/testing/sandbox-conformance.js.map +1 -0
  83. package/package.json +5 -4
  84. package/src/backends/aci-standby-pool/index.ts +16 -1
  85. package/src/backends/docker/index.ts +22 -1
  86. package/src/backends/firecracker/index.ts +14 -2
  87. package/src/backends/firecracker/protocol.ts +541 -6
  88. package/src/backends/firecracker/transport.ts +1687 -64
  89. package/src/backends/kubernetes/egress-policy.ts +3448 -0
  90. package/src/backends/kubernetes/identity.ts +261 -0
  91. package/src/backends/kubernetes/index.ts +2670 -0
  92. package/src/backends/kubernetes/ingress-policy.ts +1344 -0
  93. package/src/backends/kubernetes/k8s-client.ts +742 -0
  94. package/src/backends/kubernetes/lease.ts +254 -0
  95. package/src/backends/kubernetes/objects.ts +983 -0
  96. package/src/backends/kubernetes/per-sandbox-policy.ts +542 -0
  97. package/src/backends/kubernetes/privilege-probe.ts +261 -0
  98. package/src/backends/kubernetes/rbac.ts +192 -0
  99. package/src/backends/kubernetes/sandbox.ts +593 -0
  100. package/src/backends/kubernetes/transport.ts +2895 -0
  101. package/src/backends/kubernetes/workspace.ts +5640 -0
  102. package/src/backends/remote-execution-controller.ts +14 -0
  103. package/src/index.ts +838 -35
  104. package/src/testing/sandbox-conformance.ts +1202 -0
@@ -0,0 +1,3703 @@
1
+ /**
2
+ * The persistent workspace: a Sandbox that keeps its disk across suspends.
3
+ *
4
+ * A task sandbox (`index.ts`) is claimed, used and deleted inside one run. A
5
+ * workspace is the opposite object: it is created once, addressed by a name
6
+ * the CALLER chooses, suspended when nobody is using it, resumed days later
7
+ * with yesterday's dependency cache and git checkout still on its disk, and
8
+ * deleted only when someone says so.
9
+ *
10
+ * ## Why this is a directly created Sandbox and never a claim
11
+ *
12
+ * `Sandbox.spec.volumeClaimTemplates` is CEL-immutable on the served CRD
13
+ * ("volumeClaimTemplates is immutable"), and a `SandboxClaim` that carries
14
+ * `spec.volumeClaimTemplates` is forced to cold-start instead of adopting a
15
+ * warm pool sandbox. So the disk has to be in the spec at creation, and the
16
+ * appealing middle road — claim a warm diskless sandbox and attach a disk to
17
+ * it — is not expressible in this API at all. A workspace is therefore a
18
+ * `Sandbox` POSTed directly, with a deterministic name, and the warm pool has
19
+ * nothing to do with it.
20
+ *
21
+ * ## Block, not a filesystem
22
+ *
23
+ * The disk must be `volumeMode: Block`, consumed through the container's
24
+ * `volumeDevices`, and this module REFUSES a template whose disk is anything
25
+ * else. Under a VM-isolating RuntimeClass a `Filesystem` PVC reaches the guest
26
+ * through a host/guest filesystem passthrough (virtio-fs), whose per-file
27
+ * overhead lands squarely on the two things a workspace does all day: walking
28
+ * a dependency tree and touching thousands of small files. Nothing FAILS; the
29
+ * workspace is merely several times slower, and no functional test can see
30
+ * that. A raw block device the guest formats and mounts itself is an ordinary
31
+ * local filesystem inside the VM. See {@link KubernetesWorkspaceDiskError}.
32
+ *
33
+ * ## No lease
34
+ *
35
+ * Every task sandbox carries `shutdownTime` + `shutdownPolicy: Delete`, so a
36
+ * host that dies mid-run costs the cluster one expiry rather than a leak, and
37
+ * its handle renews that expiry for as long as it lives. A workspace carries
38
+ * NEITHER. An expiry on a workspace is a timer that deletes a caller's files,
39
+ * and a renewal loop makes losing them conditional on a host process staying
40
+ * up — exactly backwards for an object whose whole purpose is to outlive the
41
+ * host. A workspace is explicitly managed: it goes away when
42
+ * `destroy({ deleteDisk: true })` says so, and not before.
43
+ *
44
+ * ## Suspend, resume, and what changes across one
45
+ *
46
+ * `suspend()` merge-PATCHes `spec.operatingMode: Suspended`; the controller
47
+ * deletes only the Pod and reconciles PVCs unconditionally on every pass, so
48
+ * the disk survives with the same UID. It then waits on the POD — not on the
49
+ * Sandbox's `Suspended` condition, which upstream documents as lingering True
50
+ * after a resume, and not merely on a `deletionTimestamp`, which appears
51
+ * while the guest is still running. `resume()` PATCHes it back and waits
52
+ * for a new pod — and a resumed pod keeps the sandbox's NAME while getting a
53
+ * new uid and a new IP. Both matter: the uid is the agent's bind token, and
54
+ * the IP is where the agent answers. So resume re-resolves the address, re-
55
+ * reads the uid (skipping the outgoing pod, which is still listed under the
56
+ * same name while it terminates) and rebuilds the transport. Nothing from
57
+ * before the suspend is reused.
58
+ *
59
+ * Between the two, every call refuses with
60
+ * {@link KubernetesWorkspaceSuspendedError} and issues no dial. A dial would
61
+ * be worse than useless: the address still resolves — the Service outlives
62
+ * the pod — so the call would hang until a connect timeout with nothing in
63
+ * the failure naming the suspend.
64
+ *
65
+ * Both patches also stamp
66
+ * {@link OPERATING_MODE_CHANGED_AT_ANNOTATION_KEY}. Nothing in this module
67
+ * reads it back; it exists so that an inventory can say when a workspace was
68
+ * last put to sleep without waking it up to ask, which the controller's own
69
+ * lingering `Suspended` condition cannot answer. See
70
+ * {@link listKubernetesWorkspaces}.
71
+ *
72
+ * ## Three verbs that never open a workspace, and one that notices
73
+ *
74
+ * {@link createKubernetesWorkspace} adopts AND resumes, which is right for a
75
+ * host about to USE a workspace and wrong for everything else. Deleting a
76
+ * month-old suspended workspace through it means starting a pod, probing it
77
+ * and deleting it again; taking an inventory means waking every suspended
78
+ * object in the namespace. So the three operations that are about the OBJECT
79
+ * are reachable without a handle — {@link listKubernetesWorkspaces},
80
+ * {@link deleteKubernetesWorkspace} and {@link suspendKubernetesWorkspace} —
81
+ * and none of them creates a pod, dials an agent or resumes anything.
82
+ *
83
+ * The other half of the same problem is the handle that was already open when
84
+ * somebody else did one of those. A workspace id is a name, not a lock, so a
85
+ * second process can suspend the workspace this one is holding, and this
86
+ * handle's `state` is a record of what THIS process did. Two things fix that,
87
+ * and both re-read the object rather than guessing:
88
+ * {@link KubernetesWorkspace.refresh} when the caller asks, and the re-read
89
+ * after a call that FAILED at the transport — which is how it would otherwise
90
+ * be found out, as a connect refusal or a flat `unauthorized` naming nothing.
91
+ * Either one records the suspension as UNCONFIRMED (`suspending`, not
92
+ * `suspended`): what was observed is the object's mode, not the pod stopping,
93
+ * and only a wait this process performed can promise the disk is quiesced.
94
+ *
95
+ * ## Adoption is checked against the object, not against the caller
96
+ *
97
+ * A create that collides with an existing object of the same name ADOPTS it,
98
+ * because the deterministic name is only worth having if coming back is the
99
+ * normal path. What is adopted is then checked against the configuration: the
100
+ * block disk, the `sandbox.namzu.ai/template` pod label (the label the
101
+ * ingress and egress policies select by) and `runtimeClassName` (the VM
102
+ * boundary). A standing object that disagrees with any of them is refused by
103
+ * name rather than driven — see {@link KubernetesWorkspaceMismatchError}. What is NOT
104
+ * checked, and cannot be from here, is whether somebody else is already using
105
+ * it: two host processes can hold handles to one running workspace, and the
106
+ * `destroy()` of either suspends the pod the other is executing in. A
107
+ * workspace id is a name, not a lock.
108
+ *
109
+ * An adopt also has to survive walking in ON a transition, which is the
110
+ * normal way a workspace is found rather than an edge: a suspend that ended
111
+ * in {@link KubernetesWorkspaceSuspendTimeoutError}, a second host coming up
112
+ * during a rollout, a host restarting inside the previous pod's
113
+ * `terminationGracePeriodSeconds`. In all three the pod under the name is
114
+ * draining or already gone, and the replacement has not been created yet. So
115
+ * an adopt that finds the object `Suspended`, or finds its pod carrying a
116
+ * `deletionTimestamp`, WAITS for the replacement under the same readiness
117
+ * budget the resume path waits under, instead of failing on the first read
118
+ * that finds no live pod. What it came by is then reported as `origin`:
119
+ * `created`, `adopted-running` or `resumed` — a host that adopted a pod
120
+ * another process left running needs to know that none of that process's
121
+ * terminals survived it.
122
+ *
123
+ * ## There is no delete-compute-keep-disk verb
124
+ *
125
+ * The API has `operatingMode` and it has DELETE. Nothing in between. So
126
+ * `destroy()` with no options, or `deleteDisk: false`, SUSPENDS and leaves
127
+ * the object standing; only `destroy({ deleteDisk: true })` DELETEs the
128
+ * Sandbox, which cascades to the Pod, the Service and the PVC through
129
+ * ownerReferences. The default is the non-destructive one because `destroy()`
130
+ * is what a `finally` block calls, and a `finally` block must not be able to
131
+ * erase a workspace nobody asked to erase.
132
+ *
133
+ * The same holds for the paths nobody asked for at all. A create or resume
134
+ * that fails suspends the object and rethrows rather than cleaning it up (see
135
+ * {@link createKubernetesWorkspace}), and a pod that stops being able to say
136
+ * what happened to a command — the shared execution controller's unconfirmed
137
+ * cancellation, which retires a task sandbox by DELETING it — is retired here
138
+ * by that same suspend patch (see {@link retireSession}). Exactly two DELETEs
139
+ * are reachable from this module and both are ASKED FOR by name — the one
140
+ * `deleteDisk: true` sends and the one {@link deleteKubernetesWorkspace} is;
141
+ * no failure path, no `finally`, and no default reaches either.
142
+ *
143
+ * ## A state is committed when the cluster confirms it, never before
144
+ *
145
+ * Both verbs are idempotent, and both are idempotent by EARLY-RETURNING on a
146
+ * state. That makes the moment a state is written the whole correctness
147
+ * question: a handle that marks itself `deleted` before its DELETE lands
148
+ * answers every later `destroy()` from that mark, so a 500 is thrown once and
149
+ * the Sandbox then stands on the cluster with nothing left that would remove
150
+ * it. The same shape on `suspend()` leaves a pod running — and billing —
151
+ * behind a handle that says it is suspended.
152
+ *
153
+ * So `suspended` is written after the patch lands AND the pod is observed
154
+ * stopped, `deleted` after the DELETE resolves (or reports the object already
155
+ * gone), and a request that fails leaves the state it found. Concurrency is
156
+ * covered the other way round, by a single flight per verb: a second caller
157
+ * arriving mid-transition awaits the one in progress instead of sending a
158
+ * second request into the gap the deferred mark opens.
159
+ */
160
+ import { OperationDeadline, OperationDeadlineExpired, runFailureCleanup } from '../readiness.js';
161
+ import { RemoteCancellationUnknownError } from '../remote-execution-controller.js';
162
+ import { assertEgressPolicyIsEnforceable, assertEgressProfileIsUsable, assertWorkspaceCarriesNoPerSandboxEgress, composeAdditionalPodLabels, egressProfileLabel, } from './egress-policy.js';
163
+ import { KubernetesWorkspaceGuestGoneError, KubernetesWorkspaceReplacedError, } from './identity.js';
164
+ import { DEFAULT_AGENT_PORT, ReadinessPollTimeout, bindingFromSandbox, buildEgressBoundary, buildIngressVerifier, buildSandboxBody, clientAccess, clientOptions, pollForBinding, probeSandboxPrivileges, readBoundPod, readSandboxTemplate, resolveAgentAddress, resolveKubernetesReadiness, resolveProbeTimeoutMs, resolveStreamHeartbeatMs, sandboxPodLabels, sandboxPodTemplate, } from './index.js';
165
+ import { KubernetesAlreadyGoneError, KubernetesConflictError, KubernetesPatchNotAppliedError, createKubernetesClient, } from './k8s-client.js';
166
+ import { HOLDER_EPOCH_ANNOTATION_KEY, OPERATING_MODE_CHANGED_AT_ANNOTATION_KEY, POD_TEMPLATE_HASH_ANNOTATION_KEY, SANDBOX_TEMPLATE_LABEL_KEY, buildHolderEpochPatch, holderEpochAllows, isPodStopped, persistentVolumeClaimPath, podPath, podTemplateHash, readHolderEpoch, readPodTemplateHash, sandboxCollectionPath, sandboxPath, } from './objects.js';
167
+ import { KubernetesSandboxDestroyedError, buildKubernetesSandbox, } from './sandbox.js';
168
+ import { KubernetesAgentTransport, KubernetesFlushUnconfirmedError, KubernetesFlushUnreachableError, KubernetesFlushUnsupportedError, KubernetesQuiesceUnconfirmedError, KubernetesQuiesceUnsupportedError, flushUnsupportedError, guestWhenReserved, } from './transport.js';
169
+ /**
170
+ * Thrown when a workspace's `SandboxTemplate` does not describe a block disk
171
+ * this backend is willing to build a workspace on.
172
+ *
173
+ * Named, and thrown before anything is created, because every shape it
174
+ * refuses WORKS: a template with no disk produces a sandbox whose files
175
+ * vanish on the next suspend, and a `Filesystem` disk produces one that keeps
176
+ * its files and is quietly several times slower at the small-file IO a
177
+ * workspace is made of. Neither fails a functional test, so neither can be
178
+ * left to be noticed later.
179
+ */
180
+ export class KubernetesWorkspaceDiskError extends Error {
181
+ source;
182
+ name = 'KubernetesWorkspaceDiskError';
183
+ constructor(
184
+ /** What was inspected — a `SandboxTemplate`, or an existing Sandbox. */
185
+ source, message) {
186
+ super(message);
187
+ this.source = source;
188
+ }
189
+ }
190
+ /**
191
+ * Thrown when the Sandbox already standing under a workspace's name was not
192
+ * built the way this caller is configured to build one.
193
+ *
194
+ * Adoption is the NORMAL path — the deterministic name exists so that coming
195
+ * back to a workspace is cheap — and that is exactly why the object handed
196
+ * back is checked against the configuration rather than against the caller's
197
+ * intention. Two controls would otherwise be lost silently, and lost for as
198
+ * long as the workspace lives, which is the longest of anything this backend
199
+ * makes:
200
+ *
201
+ * - the `sandbox.namzu.ai/template` pod label, which is what a translated
202
+ * egress `NetworkPolicy`'s `podSelector` matches. A pod carrying another
203
+ * value — or none — is not selected by the policy this call just verified,
204
+ * so the boundary would report verified while covering nothing.
205
+ * - the egress PROFILE label, when `config.egress.profile` is set, for
206
+ * exactly the same reason: it is the selector's second half. A standing
207
+ * object built before the profile existed, or under a different one,
208
+ * carries a pod the per-profile policy does not select — and once each
209
+ * profile has its own policy object, as it must, a pod carrying neither
210
+ * key is selected by no egress policy at all. This one is checked only
211
+ * when a profile is configured: with none, the policy this call verified
212
+ * selects the template label alone, which the object does carry.
213
+ * - `runtimeClassName`, which is the VM boundary. The privilege probe cannot
214
+ * stand in for it: `/proc/self/status` reads the same inside a VM guest as
215
+ * it does inside an ordinary shared-kernel container.
216
+ *
217
+ * Nothing is patched to make the standing object match. Its disk may hold a
218
+ * month of the caller's files, and rewriting a live workspace's podTemplate to
219
+ * fit a new configuration is a larger decision than reattaching to it.
220
+ */
221
+ export class KubernetesWorkspaceMismatchError extends Error {
222
+ sandboxName;
223
+ field;
224
+ expected;
225
+ actual;
226
+ name = 'KubernetesWorkspaceMismatchError';
227
+ constructor(
228
+ /** The Sandbox found standing under this workspace's name. */
229
+ sandboxName,
230
+ /** Which configured field the standing object disagrees with. */
231
+ field,
232
+ /**
233
+ * What the configuration asked for. `key=value` for `egressProfile`,
234
+ * because the KEY is configurable too and the value alone would not
235
+ * say which label was compared.
236
+ */
237
+ expected,
238
+ /** What the object carries — absent when it carries nothing at all. */
239
+ actual, message) {
240
+ super(message);
241
+ this.sandboxName = sandboxName;
242
+ this.field = field;
243
+ this.expected = expected;
244
+ this.actual = actual;
245
+ }
246
+ }
247
+ /**
248
+ * Thrown by every operation on a workspace that is currently suspended.
249
+ *
250
+ * Distinct from {@link KubernetesSandboxDestroyedError} because the state is
251
+ * RECOVERABLE and the advice is one word: call `resume()`.
252
+ *
253
+ * `noticedBy` says which of the two ways the caller got here, and the message
254
+ * changes with it, because "nothing was dialed" is a promise the first one
255
+ * keeps and the second one cannot. A workspace another process suspended is
256
+ * discovered by a call FAILING — the pod is gone, so the dial is refused, or
257
+ * the replacement pod's agent refuses this handle's token — and the reason
258
+ * that is worth converting into this error rather than passing on is that the
259
+ * raw failure names nothing: a flat `unauthorized`, or a connect error against
260
+ * an address that still resolves because the Service outlives the pod.
261
+ */
262
+ export class KubernetesWorkspaceSuspendedError extends Error {
263
+ operation;
264
+ workspaceId;
265
+ sandboxName;
266
+ noticedBy;
267
+ name = 'KubernetesWorkspaceSuspendedError';
268
+ constructor(operation, workspaceId, sandboxName,
269
+ /** See {@link KubernetesWorkspaceSuspensionNotice}. Defaults to `admission`. */
270
+ noticedBy = 'admission', options) {
271
+ super(noticedBy === 'admission'
272
+ ? `kubernetes workspace ${workspaceId} (Sandbox ${sandboxName}) is suspended; ${operation}() cannot be admitted and nothing was dialed. Its pod is deleted and its disk is intact — call resume() to get a new pod, a new address and a new agent token, then retry.`
273
+ : `kubernetes workspace ${workspaceId} (Sandbox ${sandboxName}) is suspended; ${operation}() was admitted, failed at the transport, and a re-read of the Sandbox found spec.operatingMode: Suspended — another process suspended this workspace while this handle was holding it. The transport failure is on \`cause\`; it names nothing useful on its own, because the Service outlives the pod and the address still resolves. The disk is intact — call resume() to get a new pod, a new address and a new agent token, then retry.`, options);
274
+ this.operation = operation;
275
+ this.workspaceId = workspaceId;
276
+ this.sandboxName = sandboxName;
277
+ this.noticedBy = noticedBy;
278
+ }
279
+ }
280
+ /**
281
+ * Thrown when a suspend's `operatingMode: Suspended` patch was accepted and
282
+ * the pod had still not stopped by the readiness deadline.
283
+ *
284
+ * The workspace is left in the state that is TRUE rather than the one that
285
+ * was asked for: the patch landed, so the pod is on its way out and no call
286
+ * is admitted — but the suspend is not recorded as finished, because it did
287
+ * not finish. A later `suspend()` sends the patch again and waits again
288
+ * instead of returning on a mark this one left behind, and `resume()` still
289
+ * works.
290
+ *
291
+ * Recording it as suspended here is exactly the defect this class exists to
292
+ * make impossible. The guest is still running, still holding the block device
293
+ * open and still writing to it, and "suspended" is a promise that the disk is
294
+ * quiesced — so a handle that made that promise on a wait it lost would let
295
+ * the next caller resume, or delete, a workspace mid-write.
296
+ */
297
+ export class KubernetesWorkspaceSuspendTimeoutError extends Error {
298
+ workspaceId;
299
+ sandboxName;
300
+ timeoutMs;
301
+ name = 'KubernetesWorkspaceSuspendTimeoutError';
302
+ constructor(workspaceId, sandboxName,
303
+ /** The readiness budget the wait was given, in milliseconds. */
304
+ timeoutMs) {
305
+ super(`kubernetes: workspace ${workspaceId} (Sandbox ${sandboxName}) was patched to operatingMode: Suspended but its pod had still not stopped ${timeoutMs}ms later, so the suspend is UNCONFIRMED and the disk cannot be promised quiesced. A guest whose PID 1 ignores SIGTERM rides out its terminationGracePeriodSeconds first; raise readyTimeoutMs, or make the image exit promptly on SIGTERM. Nothing was deleted and the disk is untouched: no call is admitted while the pod drains, suspend() sends the patch again and waits again, and resume() brings the workspace back.`);
306
+ this.workspaceId = workspaceId;
307
+ this.sandboxName = sandboxName;
308
+ this.timeoutMs = timeoutMs;
309
+ }
310
+ }
311
+ /**
312
+ * Thrown when a lifecycle write carried a holder epoch the workspace has
313
+ * already moved past: the request was NOT sent, or was sent and refused, and
314
+ * nothing on the cluster changed either way.
315
+ *
316
+ * The workspace is somebody else's now. The epoch stored on the Sandbox is
317
+ * higher than the one this call carried, which is what a host says when it
318
+ * hands authority over a workspace to another process — see
319
+ * {@link HOLDER_EPOCH_ANNOTATION_KEY}. A superseded holder that suspends,
320
+ * resumes or deletes anyway would be taking the pod, or the disk, away from
321
+ * whoever holds it now.
322
+ *
323
+ * Nothing about the handle changes either. A refused `suspend()` leaves the
324
+ * handle exactly where it was — still `running`, terminals still open,
325
+ * because the refusal is decided BEFORE they are reaped — so a caller that
326
+ * catches this and re-reads its own holder record has lost nothing.
327
+ *
328
+ * `storedEpoch` is `undefined` in the one case the annotation cannot be read
329
+ * at all: it is present and is not a decimal integer, which no release of
330
+ * this backend writes. `storedAnnotation` carries it verbatim so an operator
331
+ * can see what is actually on the object.
332
+ */
333
+ export class KubernetesWorkspacePreconditionError extends Error {
334
+ operation;
335
+ workspaceId;
336
+ sandboxName;
337
+ epoch;
338
+ storedEpoch;
339
+ storedAnnotation;
340
+ name = 'KubernetesWorkspacePreconditionError';
341
+ constructor(
342
+ /** The verb that was refused — `suspend`, `resume`, `destroy`, ... */
343
+ operation, workspaceId, sandboxName,
344
+ /** The epoch this call carried. */
345
+ epoch,
346
+ /** The epoch stored on the Sandbox, or `undefined` if unreadable. */
347
+ storedEpoch,
348
+ /** The annotation exactly as stored, when there is one. */
349
+ storedAnnotation) {
350
+ super(storedEpoch === undefined
351
+ ? `kubernetes: ${operation}() on workspace ${workspaceId} (Sandbox ${sandboxName}) carried holder epoch ${epoch}, and the Sandbox's ${HOLDER_EPOCH_ANNOTATION_KEY} annotation reads ${JSON.stringify(storedAnnotation)}, which is not a decimal integer. No release of this backend writes that, so it was set by hand or by something else; the write was refused rather than overwriting a fence this code does not understand. Nothing on the cluster changed. Fix the annotation, or remove it to start the workspace's epoch again at 0.`
352
+ : `kubernetes: ${operation}() on workspace ${workspaceId} (Sandbox ${sandboxName}) carried holder epoch ${epoch}, but the Sandbox is held at epoch ${storedEpoch} — another process took this workspace over. Nothing on the cluster changed and nothing about this handle changed: its terminals are still open and it is still admitting calls. A host that raised the epoch elsewhere should stop using this handle; one that believes it is still the holder should re-read its own record first.`);
353
+ this.operation = operation;
354
+ this.workspaceId = workspaceId;
355
+ this.sandboxName = sandboxName;
356
+ this.epoch = epoch;
357
+ this.storedEpoch = storedEpoch;
358
+ this.storedAnnotation = storedAnnotation;
359
+ }
360
+ }
361
+ /**
362
+ * Refuse an epoch this backend cannot honour, at the entry point rather than
363
+ * at the request — the same place {@link resolveRequestTimeoutMs} refuses a
364
+ * timeout, and for the same reason: a configuration that will never work
365
+ * should be named where the caller can still see which call it came from.
366
+ *
367
+ * Non-negative because the stored value is compared as a number and written
368
+ * as a decimal string, and an integer because a fractional epoch would
369
+ * round-trip through the annotation as something other than what was passed.
370
+ */
371
+ function assertHolderEpoch(epoch, where) {
372
+ if (epoch === undefined)
373
+ return undefined;
374
+ if (!Number.isSafeInteger(epoch) || epoch < 0) {
375
+ throw new Error(`kubernetes: ${where} epoch must be a non-negative safe integer, got ${JSON.stringify(epoch)}. It is stored on the Sandbox as a decimal string and compared as a number, so anything else could not be written back as the value that was passed.`);
376
+ }
377
+ return epoch;
378
+ }
379
+ /**
380
+ * Refuse a `quiesce` asked of a verb that has no guest to ask.
381
+ *
382
+ * {@link suspendKubernetesWorkspace} reaches a workspace WITHOUT opening one:
383
+ * it sends a patch and waits for the pod, and never dials the agent. So there
384
+ * is nothing there that could stop a process, and silently ignoring the flag
385
+ * would hand a caller a suspend it believes was preceded by a quiesce —
386
+ * exactly the belief this whole feature exists to make true. The handle's
387
+ * `suspend()` is the verb that can do it, and the message says so.
388
+ */
389
+ function assertNoQuiesceHere(quiesce, where) {
390
+ if (quiesce === undefined || quiesce === false)
391
+ return;
392
+ throw new Error(`kubernetes: ${where} cannot quiesce the guest: it reaches the workspace through the API server alone and never dials the agent, so there is no connection on which to stop anything. Open the workspace with createKubernetesWorkspace() and call suspend({ quiesce: true }) on the handle, or quiesce() and then this. The option is refused rather than ignored, because a caller that passed it is about to trust a capture.`);
393
+ }
394
+ /**
395
+ * Refuse a `flush: true` asked of a verb that has no guest to ask.
396
+ *
397
+ * The same rule as {@link assertNoQuiesceHere} and one difference worth
398
+ * saying out loud: a flush is ON by default on the handle's `suspend()`, so
399
+ * the DEFAULT reaching {@link suspendKubernetesWorkspace} cannot be an
400
+ * error — it is what every caller of that verb has always passed. Only an
401
+ * EXPLICIT flush — `true`, or an object naming a `timeoutMs` — is refused,
402
+ * because only that caller believes the writes are being put on the device.
403
+ * `flush: false` is accepted and means what it says here: this verb never
404
+ * flushes, and its own documentation is where that is stated rather than
405
+ * hidden behind a thrown error nobody sees.
406
+ */
407
+ function assertNoFlushHere(flush, where) {
408
+ if (flush === undefined || flush === false)
409
+ return;
410
+ throw new Error(`kubernetes: ${where} cannot flush the guest: it reaches the workspace through the API server alone and never dials the agent, so there is no connection on which to ask for a syncfs. What the disk keeps is whatever the guest kernel had already written back. Open the workspace with createKubernetesWorkspace() and call suspend() on the handle, which flushes by default. The option is refused rather than ignored, because a caller that passed it is about to trust the disk.`);
411
+ }
412
+ /** Prefix every workspace Sandbox's name carries. */
413
+ export const WORKSPACE_NAME_PREFIX = 'namzu-ws-';
414
+ /** DNS-1123 label: the Sandbox's name is also its Pod's and its Service's. */
415
+ const DNS_LABEL_MAX_LENGTH = 63;
416
+ const MAX_WORKSPACE_ID_LENGTH = DNS_LABEL_MAX_LENGTH - WORKSPACE_NAME_PREFIX.length;
417
+ const WORKSPACE_ID_PATTERN = /^[a-z0-9]([a-z0-9-]*[a-z0-9])?$/;
418
+ /**
419
+ * The Sandbox name for a workspace id.
420
+ *
421
+ * Deterministic on purpose: it is the only way a second host process, or the
422
+ * same one tomorrow, finds the workspace again. Which is also why an id that
423
+ * does not already fit a DNS-1123 label is REFUSED rather than lowercased,
424
+ * stripped or hashed: sanitising maps two ids onto one name, and two callers
425
+ * who believe they have separate workspaces would be sharing one disk.
426
+ * Hashing would fit every id and make the name unreadable in `kubectl get
427
+ * sandbox`, which is most of what the deterministic name is for.
428
+ */
429
+ export function workspaceSandboxName(workspaceId) {
430
+ if (!WORKSPACE_ID_PATTERN.test(workspaceId) || workspaceId.length > MAX_WORKSPACE_ID_LENGTH) {
431
+ throw new Error(`kubernetes: workspace id ${JSON.stringify(workspaceId)} cannot name a Sandbox. It must be 1-${MAX_WORKSPACE_ID_LENGTH} characters of lowercase letters, digits and '-', starting and ending alphanumeric, so that ${WORKSPACE_NAME_PREFIX}<id> is a legal DNS-1123 label for the Sandbox, its Pod and its Service. The id is refused rather than sanitised because two ids that sanitise to one name would silently share one disk.`);
432
+ }
433
+ return `${WORKSPACE_NAME_PREFIX}${workspaceId}`;
434
+ }
435
+ function readArray(value) {
436
+ return Array.isArray(value) ? value : [];
437
+ }
438
+ function readNames(container, field) {
439
+ if (typeof container !== 'object' || container === null)
440
+ return [];
441
+ const entries = readArray(container[field]);
442
+ const names = [];
443
+ for (const entry of entries) {
444
+ if (typeof entry !== 'object' || entry === null)
445
+ continue;
446
+ const name = entry.name;
447
+ if (typeof name === 'string' && name !== '')
448
+ names.push(name);
449
+ }
450
+ return names;
451
+ }
452
+ /**
453
+ * Every name a pod template's containers claim through `volumeDevices` (a raw
454
+ * block device) or `volumeMounts` (a filesystem), across both container lists.
455
+ * The pod spec is carried opaquely everywhere else in this backend, so it is
456
+ * read defensively here rather than typed: an unexpected shape contributes
457
+ * nothing and lets the named refusal below do the talking.
458
+ */
459
+ function readVolumeConsumers(podTemplate) {
460
+ const devices = new Set();
461
+ const mounts = new Set();
462
+ const spec = podTemplate?.spec;
463
+ for (const key of ['containers', 'initContainers']) {
464
+ for (const container of readArray(spec?.[key])) {
465
+ for (const name of readNames(container, 'volumeDevices'))
466
+ devices.add(name);
467
+ for (const name of readNames(container, 'volumeMounts'))
468
+ mounts.add(name);
469
+ }
470
+ }
471
+ return { devices, mounts };
472
+ }
473
+ /**
474
+ * Refuse anything that is not a block disk a container actually consumes as
475
+ * one. Every branch here describes a configuration that would work and then
476
+ * disappoint — see {@link KubernetesWorkspaceDiskError}.
477
+ */
478
+ export function assertBlockModeWorkspaceDisk(source, podTemplate, volumeClaimTemplates) {
479
+ if (volumeClaimTemplates === undefined || volumeClaimTemplates.length === 0) {
480
+ throw new KubernetesWorkspaceDiskError(source, `kubernetes: ${source} declares no spec.volumeClaimTemplates, so a workspace built from it would have no disk and would lose everything on its first suspend — that is a task sandbox, not a workspace. Add a volumeClaimTemplates entry with spec.volumeMode: Block and consume it from the container's volumeDevices.`);
481
+ }
482
+ const { devices, mounts } = readVolumeConsumers(podTemplate);
483
+ for (const entry of volumeClaimTemplates) {
484
+ const name = entry.metadata?.name;
485
+ if (typeof name !== 'string' || name === '') {
486
+ throw new KubernetesWorkspaceDiskError(source, `kubernetes: ${source} declares a spec.volumeClaimTemplates entry with no metadata.name. The controller wires the disk by that name, StatefulSet style — it creates the PVC as <entry name>-<sandbox name> and matches the container's volumeDevices entry against it — so an unnamed entry reaches no container at all.`);
487
+ }
488
+ const volumeMode = entry.spec?.volumeMode;
489
+ if (volumeMode !== 'Block') {
490
+ throw new KubernetesWorkspaceDiskError(source, `kubernetes: ${source} declares volumeClaimTemplate ${JSON.stringify(name)} with volumeMode ${JSON.stringify(volumeMode ?? 'Filesystem (the API default)')}, but a persistent workspace's disk must be volumeMode: Block. Under a VM-isolating RuntimeClass a Filesystem PVC reaches the guest over a host/guest filesystem passthrough (virtio-fs), which pays a round trip per file operation — a dependency tree walk or a git status over a large checkout is several times slower, while nothing fails and no functional test can see it. A Block volume is a raw device the guest formats once and mounts as an ordinary local filesystem. Set spec.volumeMode: Block on this entry and consume it through the container's volumeDevices.`);
491
+ }
492
+ if (mounts.has(name)) {
493
+ throw new KubernetesWorkspaceDiskError(source, `kubernetes: ${source} consumes the Block volumeClaimTemplate ${JSON.stringify(name)} through a container's volumeMounts. A raw block device is claimed through volumeDevices (which gives the container a device node at devicePath); volumeMounts is the filesystem form and the kubelet will refuse the pod. Move the entry to volumeDevices and let the image's entrypoint format and mount the device.`);
494
+ }
495
+ if (!devices.has(name)) {
496
+ throw new KubernetesWorkspaceDiskError(source, `kubernetes: ${source} declares the Block volumeClaimTemplate ${JSON.stringify(name)} but no container claims it through volumeDevices, so the PVC is provisioned and attached to nothing. Add a volumeDevices entry naming ${JSON.stringify(name)} with the devicePath the image's entrypoint formats and mounts.`);
497
+ }
498
+ }
499
+ }
500
+ /** The `metadata.name` of each claim, in declaration order, blanks dropped. */
501
+ function volumeClaimTemplateNames(volumeClaimTemplates) {
502
+ const names = [];
503
+ for (const entry of volumeClaimTemplates ?? []) {
504
+ const name = entry.metadata?.name;
505
+ if (typeof name === 'string' && name !== '')
506
+ names.push(name);
507
+ }
508
+ return names;
509
+ }
510
+ /**
511
+ * Refuse a refresh whose template declares a DIFFERENT set of disks than the
512
+ * Sandbox it would be written onto.
513
+ *
514
+ * Only a refresh can arrive here, and that is the whole point. On the create
515
+ * path the pod template and the `volumeClaimTemplates` come out of the same
516
+ * `SandboxTemplate` in the same read, so the two sides cannot disagree. On a
517
+ * refresh they are two different objects: `spec.volumeClaimTemplates` is
518
+ * CEL-immutable, so it is never in the patch and stays whatever the create
519
+ * POST froze, while `spec.podTemplate` becomes whatever the template says
520
+ * TODAY.
521
+ *
522
+ * {@link assertBlockModeWorkspaceDisk} asks one half of the question — is
523
+ * every disk this Sandbox HAS still claimed as a block device by the pod
524
+ * template about to land. This asks the other half, which nothing else can
525
+ * see: does the template claim a disk this Sandbox does not have. Adding a
526
+ * second `volumeClaimTemplates` entry with its matching `volumeDevices` entry
527
+ * is the ordinary way to give a workspace another disk, and a template edited
528
+ * that way is internally consistent — it passes every check made against
529
+ * itself, and every branch of the check above, because the original disk is
530
+ * still claimed. Written onto a standing workspace it produces a pod spec
531
+ * naming a device the Sandbox has no PVC for. The shipped workspace template
532
+ * declares no `spec.volumes`, so there is no other source for that name: the
533
+ * controller would be left unable to build a valid pod, the workspace would
534
+ * sit `Running` with nothing coming up, and the caller would see a bind
535
+ * timeout naming nothing. Recovery would mean suspending, reverting the
536
+ * template and refreshing again — the object having been left permanently
537
+ * describing a disk that does not exist.
538
+ *
539
+ * Comparing the NAMES is the whole rule, because the name is what the
540
+ * controller wires by (StatefulSet style, `<entry name>-<sandbox name>`). An
541
+ * edit to an existing entry's other fields — its size, its storage class,
542
+ * its `volumeMode` — simply does not apply to a standing workspace, and is
543
+ * not refused here: the disk it describes is the disk that is already there.
544
+ * Only a template pointed at a workspace whose disks it cannot describe at
545
+ * all is a mistake worth stopping.
546
+ *
547
+ * What neither check catches, and deliberately: a template whose containers
548
+ * claim a `volumeDevices` name that is in NEITHER its own
549
+ * `volumeClaimTemplates` NOR the Sandbox's. The check above asks only that
550
+ * every disk the Sandbox HAS is still claimed, and this one asks only about
551
+ * names the template's `volumeClaimTemplates` adds, so a device claimed out
552
+ * of nowhere passes both. That template is equally broken on the CREATE path,
553
+ * which has no check either and would produce the same unbuildable pod on a
554
+ * brand-new workspace — it is a template that is wrong about itself, not a
555
+ * template that is wrong about this workspace, and this pair of refusals is
556
+ * only about the second kind.
557
+ */
558
+ function assertRefreshedTemplateClaimsTheSameDisks(source, templateVolumeClaimTemplates, sandboxVolumeClaimTemplates) {
559
+ const wanted = volumeClaimTemplateNames(templateVolumeClaimTemplates);
560
+ const held = new Set(volumeClaimTemplateNames(sandboxVolumeClaimTemplates));
561
+ const extra = wanted.filter((name) => !held.has(name));
562
+ if (extra.length === 0)
563
+ return;
564
+ throw new KubernetesWorkspaceDiskError(source, `kubernetes: ${source} declares the volumeClaimTemplate${extra.length === 1 ? '' : 's'} ${extra.map((name) => JSON.stringify(name)).join(', ')}, which this workspace's Sandbox does not have — it was created with ${[...held].map((name) => JSON.stringify(name)).join(', ') || 'none'} and spec.volumeClaimTemplates is CEL-immutable, so no refresh can add a disk to a workspace that already exists. Writing this template's pod spec onto it would claim a device node backed by no PVC, and the controller would never build a valid pod. Refresh this workspace against a template that declares the disks it already has, or create a new workspace from this template and migrate the data.`);
565
+ }
566
+ /**
567
+ * Refuse a standing Sandbox that was built from another `SandboxTemplate`, or
568
+ * that runs without the RuntimeClass this backend is configured for.
569
+ *
570
+ * Read off `spec.podTemplate` — the copy the controller actually runs a pod
571
+ * from — rather than off anything this process decided, because the question
572
+ * is what the POD is, not what the caller meant it to be.
573
+ *
574
+ * The template check is unconditional, `config.egress` set or not: the label
575
+ * is also how an operator reads which template an object came from, and an
576
+ * object whose label says one thing while the caller builds from another is a
577
+ * mix-up worth naming the first time it is seen rather than the first time a
578
+ * policy is switched on. See {@link KubernetesWorkspaceMismatchError}.
579
+ *
580
+ * The PROFILE check is conditional, because a profile is what this
581
+ * configuration asks for rather than something every object has: it runs when
582
+ * `config.egress.profile` is set, and then the object's own pod template has
583
+ * to carry that key with that value — unless this call is about to REWRITE
584
+ * that pod template (`refreshPodTemplate: true` onto a Suspended object),
585
+ * which writes the configured profile with the rest of the overlays. That is
586
+ * the same lifting the `runtimeClassName` check gets, for the same reason and
587
+ * with the same re-application if the patch does not land.
588
+ *
589
+ * Both halves of the policy selector are therefore checked against the object
590
+ * on this path — which matters because neither network check further up can
591
+ * do it: the ingress verification and the egress union both run against the
592
+ * labels the POST *would* stamp, and on this path the POST already lost to a
593
+ * 409. What this does NOT check is a
594
+ * label the standing object carries that this configuration does not ask for
595
+ * — an object built under a profile this config has dropped, say. That pod is
596
+ * still selected by the unprofiled policy this call verified (a selector is a
597
+ * subset match), so the boundary it reports holds; what a stale label could
598
+ * do is bring a SECOND policy into the union, and the adopt path cannot
599
+ * enumerate the standing object's labels without reading them, which is a
600
+ * wider change than this refusal.
601
+ */
602
+ export function assertAdoptedWorkspaceMatchesConfig(sandboxName, namespace, podTemplate, expected) {
603
+ const label = podTemplate?.metadata?.labels?.[SANDBOX_TEMPLATE_LABEL_KEY];
604
+ if (label !== expected.sandboxTemplateName) {
605
+ throw new KubernetesWorkspaceMismatchError(sandboxName, 'sandboxTemplateName', expected.sandboxTemplateName, label, `kubernetes: Sandbox ${sandboxName} in namespace ${namespace} already exists, and its spec.podTemplate carries ${SANDBOX_TEMPLATE_LABEL_KEY}: ${label === undefined ? '(absent)' : JSON.stringify(label)} rather than ${JSON.stringify(expected.sandboxTemplateName)} — it was built from a different SandboxTemplate, so it is not the workspace this call describes. That label is the one an egress NetworkPolicy's podSelector matches, so adopting this object would hand back a pod the policy verified for ${JSON.stringify(expected.sandboxTemplateName)} does not select, having reported the boundary as verified. Point this workspace at the template the object was built from, or delete the Sandbox — which takes its disk with it — and create it again, or choose another workspaceId.`);
606
+ }
607
+ const profile = expected.profile;
608
+ if (profile !== undefined) {
609
+ const carried = podTemplate?.metadata?.labels?.[profile.key];
610
+ if (carried !== profile.value) {
611
+ throw new KubernetesWorkspaceMismatchError(sandboxName, 'egressProfile', `${profile.key}=${profile.value}`, carried === undefined ? undefined : `${profile.key}=${carried}`, `kubernetes: Sandbox ${sandboxName} in namespace ${namespace} already exists, and its spec.podTemplate carries ${profile.key}: ${carried === undefined ? '(absent)' : JSON.stringify(carried)} rather than the configured egress profile ${JSON.stringify(profile.value)}. The translated policy's podSelector matches that label as well as the template one, so this object's pod is not selected by the policy this call just verified — and since each profile needs its own policy object, a pod carrying no profile label at all is selected by none of them, while create() would have reported the boundary verified. A standing workspace's pod labels are not rewritten underneath it by an ordinary adopt, so the mismatch is named instead. Three ways out, and only the last one costs the disk: reopen it with refreshPodTemplate: true, which rewrites spec.podTemplate — profile label and all — on the one Suspended → Running transition and is the supported way to move a workspace between profiles; or point config.egress.profile at ${carried === undefined ? 'the profile this object was built under (none was)' : JSON.stringify(carried)}; or delete the Sandbox — which takes its disk with it — and create it again under ${JSON.stringify(profile.value)}.`);
612
+ }
613
+ }
614
+ if (expected.runtimeClassName === undefined)
615
+ return;
616
+ const declared = podTemplate?.spec?.runtimeClassName;
617
+ const actual = typeof declared === 'string' ? declared : undefined;
618
+ if (actual !== expected.runtimeClassName) {
619
+ throw new KubernetesWorkspaceMismatchError(sandboxName, 'runtimeClassName', expected.runtimeClassName, actual, `kubernetes: Sandbox ${sandboxName} in namespace ${namespace} already exists, and its spec.podTemplate.spec.runtimeClassName is ${actual === undefined ? "(absent — the cluster's default runtime)" : JSON.stringify(actual)} rather than the configured ${JSON.stringify(expected.runtimeClassName)}. Running it would put the workspace on that runtime — a shared kernel, if it is the default — while this backend registers itself as tier 'microvm', and nothing downstream would notice: the privilege probe reads /proc/self/status inside the guest and passes identically under a VM and under runc. A standing object's RuntimeClass is not something this backend rewrites underneath a disk it did not create, so the mismatch is named instead. Delete the Sandbox — which takes its disk with it — and create it again under ${JSON.stringify(expected.runtimeClassName)}, or drop runtimeClassName from the config if this object's runtime is the intended one.`);
620
+ }
621
+ }
622
+ /**
623
+ * Create the workspace, or take over the one that is already there.
624
+ *
625
+ * A second call with the same `workspaceId` ADOPTS rather than fails: the
626
+ * POST comes back 409 Conflict, and the object it collided with is this
627
+ * caller's own workspace from an earlier process. The deterministic name is
628
+ * only useful if coming back to it is the normal path. An adopted object is
629
+ * checked against the same block-disk rule a fresh one is, so a Sandbox
630
+ * standing under this name that is not a workspace is refused rather than
631
+ * used.
632
+ *
633
+ * Nothing here is ever deleted on failure. A create that gets as far as an
634
+ * existing object and then fails — a readiness timeout, a privilege probe
635
+ * refusal — SUSPENDS it and rethrows, because the object may be a workspace
636
+ * with a disk full of the caller's files and `deleteDisk` is not a decision
637
+ * a failure path gets to make. That holds even when THIS call POSTed the
638
+ * object and its disk is therefore empty: the 409 above means two processes
639
+ * can be coming up on one name at once, and the one that got the 201 deleting
640
+ * its "own" fresh object would take the disk of the one that adopted it. The
641
+ * cost is named rather than paid: a failed create can leave one suspended
642
+ * Sandbox standing, which the caller finds again under the same deterministic
643
+ * name and nothing reaps for them — see the docs page.
644
+ */
645
+ export async function createKubernetesWorkspace(config, options) {
646
+ options.signal?.throwIfAborted();
647
+ // `config.egress.perSandbox` is the one part of `config.egress` this path
648
+ // cannot honour at all, and it is refused FIRST — before readiness, before
649
+ // the client, before any request. It configures `setNetworkPolicy`, which
650
+ // a workspace handle never carries: nothing in the create path below
651
+ // composes a per-sandbox pod label or tracks an owner uid for one, and
652
+ // the option exists to make the method present on a TASK handle. Accepting
653
+ // it here and omitting the method is the silent downgrade every other
654
+ // refusal in `egress-policy.ts` exists to prevent, so it is refused by
655
+ // name rather than validated: `assertPerSandboxEgressIsUsable` would
656
+ // check a config this path serves no purpose for and report it as usable.
657
+ assertWorkspaceCarriesNoPerSandboxEgress(config.egress);
658
+ const readiness = resolveKubernetesReadiness(config);
659
+ const namespace = config.namespace;
660
+ const name = workspaceSandboxName(options.workspaceId);
661
+ const templateName = options.sandboxTemplateName ?? config.sandboxTemplateName;
662
+ const agentPort = config.agentPort ?? DEFAULT_AGENT_PORT;
663
+ const agentAddress = config.agentAddress ?? 'service';
664
+ // Resolved before anything is POSTed, so a configuration this backend
665
+ // will never honour is refused rather than leaving an object behind.
666
+ const streamHeartbeatMs = resolveStreamHeartbeatMs(config.streamHeartbeatMs);
667
+ const epoch = assertHolderEpoch(options.epoch, 'createKubernetesWorkspace');
668
+ const client = createKubernetesClient(clientAccess(config), clientOptions(config));
669
+ // The same two egress steps `buildKubernetesBackend` runs for a task
670
+ // sandbox, repeated here because a workspace never goes through it. The
671
+ // refusal is synchronous and decided from the policy KIND alone; the
672
+ // verification is a GET of the object an operator was supposed to apply,
673
+ // and neither ever creates or repairs anything — see `egress-policy.ts`.
674
+ //
675
+ // Not optional on this path, and not a copy-paste: a long-lived workspace
676
+ // is the sandbox most likely to be pointed at a network, the NetworkPolicy
677
+ // rather than the bind token is the boundary on its agent port, and a
678
+ // config object refused by one entry point and silently ignored by the
679
+ // other is the exact silent downgrade this translation exists to prevent.
680
+ //
681
+ // Verified against the template this workspace is actually built from:
682
+ // `buildSandboxBody` stamps the pod with THAT template's label and the
683
+ // policy's podSelector matches that label, so a workspace built from a
684
+ // separate workspace template needs its own policy — the task template's
685
+ // does not select it. Deliberately not memoized the way the backend's
686
+ // once-per-backend check is: the boundary object is built per create here,
687
+ // so both of its memos live exactly as long as this call — creating a
688
+ // workspace is a rare, explicit act with nothing to amortise, and a policy
689
+ // deleted since the last call has to be noticed. The union half runs
690
+ // further down, against the labels the POST is about to stamp — the pod's
691
+ // own labels on the create path. On the ADOPT path the POST loses to a
692
+ // 409 and those labels are not the standing pod's, so the two a policy
693
+ // selector is built from, the template label and the profile, are checked
694
+ // against the standing object itself instead: see
695
+ // `assertAdoptedWorkspaceMatchesConfig`.
696
+ //
697
+ // Before the boundary is built, because building it resolves the profile:
698
+ // a profile this backend would never emit is a wiring error, and it reads
699
+ // as one when it is refused in its own words rather than out of a policy
700
+ // name.
701
+ assertEgressProfileIsUsable(config.egress);
702
+ const egressBoundary = buildEgressBoundary(client, config, templateName);
703
+ if (config.egress) {
704
+ assertEgressPolicyIsEnforceable(config.egress.policy, config.egress.engine ?? 'core', config.egress.ciliumNarrowing);
705
+ await egressBoundary?.verifyNamedObject(options.signal);
706
+ }
707
+ // Read and validate BEFORE anything is created, so a template that cannot
708
+ // carry a workspace fails with nothing to clean up.
709
+ const template = await readSandboxTemplate(client, namespace, templateName, options.signal);
710
+ assertBlockModeWorkspaceDisk(`SandboxTemplate ${templateName} in namespace ${namespace}`, template.podTemplate, template.volumeClaimTemplates);
711
+ // And the other half of the boundary the egress block above calls
712
+ // primary: an ingress policy that actually closes the agent port on the
713
+ // labels this pod will carry. Checked before the POST, so a refusal leaves
714
+ // no Sandbox and no PVC behind — and, on the adopt path, sends no resume
715
+ // patch: a workspace whose port stopped being covered is refused asleep
716
+ // rather than woken up to be refused. `sandboxPodLabels` is the same function the
717
+ // create body stamps its labels with, so the check cannot verify a pod
718
+ // nobody creates. Not memoized, for the reason the egress check above is
719
+ // not: this is a rare, explicit act with nothing to amortise, and a policy
720
+ // deleted since the last call has to be noticed.
721
+ // A workspace is a Sandbox this backend POSTs directly, so an egress
722
+ // PROFILE has to be stamped by this body rather than merged by the
723
+ // controller — and the same map has to reach both the check below and the
724
+ // POST further down, or the check would verify a pod nobody creates. One
725
+ // composer, one call, both call sites. A workspace ADOPTED from a previous
726
+ // create keeps the pod labels it was created with — an ordinary adopt
727
+ // patches no pod template — so the adopt below REFUSES an object whose
728
+ // profile label is not the configured one rather than handing back a pod
729
+ // the verified policy does not select. The way to move an existing
730
+ // workspace onto a new profile is `refreshPodTemplate: true`, which
731
+ // rewrites `/spec/podTemplate` with exactly these labels.
732
+ const profile = egressProfileLabel(config.egress);
733
+ const profilePodLabels = composeAdditionalPodLabels(config.egress);
734
+ const workspacePodLabels = sandboxPodLabels(template, templateName, profilePodLabels);
735
+ const workspaceSubject = `to open workspace ${options.workspaceId} as Sandbox ${name} in namespace ${namespace}`;
736
+ const verifyIngress = buildIngressVerifier(client, config);
737
+ if (verifyIngress !== undefined) {
738
+ await verifyIngress(workspacePodLabels, workspaceSubject, options.signal);
739
+ }
740
+ // And the egress half of the same question, against the same labels: the
741
+ // named object matching exactly says nothing about what a SECOND policy
742
+ // selecting these pods lets out, and a long-lived workspace is the sandbox
743
+ // most likely to be pointed at a network. Checked before the POST for the
744
+ // reason the ingress check is, and on the adopt path before any resume
745
+ // patch — a workspace whose egress stopped being bounded is refused asleep
746
+ // rather than woken up to be refused.
747
+ await egressBoundary?.verifyUnion(workspacePodLabels, workspaceSubject, options.signal);
748
+ // The pod template this call would write, and its revision — built once,
749
+ // from the same overlays `buildSandboxBody` applies below, because the
750
+ // hash has to be taken over exactly the object that lands on the cluster.
751
+ // `profilePodLabels` is one of those overlays: a refresh rewrites
752
+ // `/spec/podTemplate` WHOLE, so one built without them would PATCH the
753
+ // profile label off a workspace that carries it, and the revision stamped
754
+ // by the POST would be taken over a template the POST never wrote.
755
+ const refresh = buildPodTemplateRefresh(template, templateName, config.runtimeClassName, profilePodLabels);
756
+ let adopted;
757
+ try {
758
+ await client.request('POST', sandboxCollectionPath(namespace),
759
+ // No shutdownTime: a workspace carries no expiry — see the module
760
+ // comment.
761
+ buildSandboxBody({
762
+ namespace,
763
+ name,
764
+ template,
765
+ sandboxTemplateName: templateName,
766
+ podLabels: profilePodLabels,
767
+ // Stamped by the POST itself rather than by a patch after it,
768
+ // so a workspace created under an epoch is fenced from the
769
+ // moment it exists — there is no window in which the object
770
+ // stands unfenced and a second opener could take it. The
771
+ // pod-template revision rides along for the same reason: an
772
+ // object that recorded what it was built from on its second
773
+ // request would have a window in which it claimed none.
774
+ annotations: {
775
+ ...(epoch !== undefined ? { [HOLDER_EPOCH_ANNOTATION_KEY]: String(epoch) } : {}),
776
+ [POD_TEMPLATE_HASH_ANNOTATION_KEY]: refresh.hash,
777
+ },
778
+ ...(config.runtimeClassName !== undefined
779
+ ? { runtimeClassName: config.runtimeClassName }
780
+ : {}),
781
+ }), options.signal);
782
+ }
783
+ catch (err) {
784
+ if (!(err instanceof KubernetesConflictError))
785
+ throw err;
786
+ adopted = await adoptExistingWorkspace(client, namespace, name, options.workspaceId, {
787
+ sandboxTemplateName: templateName,
788
+ // The same resolution the create body and the policy selector
789
+ // were built from, so the object is checked against exactly
790
+ // the label this call would have stamped.
791
+ ...(profile !== undefined ? { profile } : {}),
792
+ ...(config.runtimeClassName !== undefined
793
+ ? { runtimeClassName: config.runtimeClassName }
794
+ : {}),
795
+ }, epoch, options.refreshPodTemplate === true ? refresh : undefined, options.signal);
796
+ }
797
+ return await openWorkspaceHandle({
798
+ client,
799
+ namespace,
800
+ name,
801
+ workspaceId: options.workspaceId,
802
+ rootDir: options.workingDirectory,
803
+ agentPort,
804
+ agentAddress,
805
+ streamHeartbeatMs,
806
+ readiness,
807
+ origin: adopted === undefined ? 'created' : adopted.resumed ? 'resumed' : 'adopted-running',
808
+ sandboxTemplateName: templateName,
809
+ ...(config.runtimeClassName !== undefined ? { runtimeClassName: config.runtimeClassName } : {}),
810
+ // Carried for the reason the template name and the RuntimeClass are:
811
+ // a `resume({ refreshPodTemplate: true })` rebuilds the pod template
812
+ // long after this call, and a refresh rebuilt without these labels
813
+ // would patch the egress profile off the very pod the configured
814
+ // policy selects.
815
+ ...(Object.keys(profilePodLabels).length > 0 ? { podLabels: profilePodLabels } : {}),
816
+ // A create wrote the revision it just computed; an adopt reports
817
+ // whatever the standing object carries, which is `undefined` for one
818
+ // created before the annotation existed.
819
+ ...(adopted === undefined
820
+ ? { templateRevision: refresh.hash }
821
+ : adopted.templateRevision !== undefined
822
+ ? { templateRevision: adopted.templateRevision }
823
+ : {}),
824
+ currentTemplateHash: refresh.hash,
825
+ ...(adopted?.awaitReplacement === true ? { awaitReplacement: true } : {}),
826
+ ...(epoch !== undefined ? { epoch } : {}),
827
+ ...(adopted?.drainingPodUid !== undefined ? { drainingPodUid: adopted.drainingPodUid } : {}),
828
+ ...(options.onStartFailure !== undefined ? { onStartFailure: options.onStartFailure } : {}),
829
+ ...(options.onCancellationUnconfirmed !== undefined
830
+ ? { onCancellationUnconfirmed: options.onCancellationUnconfirmed }
831
+ : {}),
832
+ ...(options.onQuiesceUnsupported !== undefined
833
+ ? { onQuiesceUnsupported: options.onQuiesceUnsupported }
834
+ : {}),
835
+ ...(options.onQuiesceNarrowed !== undefined
836
+ ? { onQuiesceNarrowed: options.onQuiesceNarrowed }
837
+ : {}),
838
+ ...(options.onFlushUnsupported !== undefined
839
+ ? { onFlushUnsupported: options.onFlushUnsupported }
840
+ : {}),
841
+ ...(options.onFlushUnreachable !== undefined
842
+ ? { onFlushUnreachable: options.onFlushUnreachable }
843
+ : {}),
844
+ ...(options.signal !== undefined ? { signal: options.signal } : {}),
845
+ });
846
+ }
847
+ /**
848
+ * Take over a Sandbox that already stands under this workspace's name: check
849
+ * that it really is a block-disk workspace AND that it is the one this
850
+ * configuration describes, then wake it if it is asleep.
851
+ *
852
+ * Everything checked here is checked against the object, because on this path
853
+ * the object is not the one this call built. A create POSTs its own body and
854
+ * knows what is in it; an adopt is handed a pod somebody else's process, or
855
+ * last month's configuration, decided the shape of. The silent losses are the
856
+ * template label, the egress profile label when one is configured, and the
857
+ * RuntimeClass — see {@link KubernetesWorkspaceMismatchError}.
858
+ *
859
+ * The refusals all happen BEFORE the resume patch: an object this call will
860
+ * not use is not woken up on the way to being rejected.
861
+ *
862
+ * What it OBSERVES, and hands back, is whether the pod this workspace is
863
+ * about to be bound to is a pod that already exists. Both answers here mean
864
+ * it is not: an object found `Suspended` has had its pod taken away, and a
865
+ * pod already carrying a `deletionTimestamp` is one the controller is in the
866
+ * middle of taking away. Either way the pod this handle will serve has not
867
+ * been created yet, and the first bind attempt has to WAIT for it rather than
868
+ * fail on the read that finds no live pod — see {@link PodBindPolicy}.
869
+ */
870
+ async function adoptExistingWorkspace(client, namespace, name, workspaceId, expected, epoch, refresh, signal) {
871
+ const target = { client, namespace, name, workspaceId };
872
+ const existing = await readSandboxObject(target, signal);
873
+ assertBlockModeWorkspaceDisk(`Sandbox ${name} in namespace ${namespace}`, existing?.spec?.podTemplate, existing?.spec?.volumeClaimTemplates);
874
+ const suspended = operatingModeOf(existing) === 'Suspended';
875
+ // A refresh is a Suspended→Running write and nothing else, so an object
876
+ // found Running is adopted exactly as it always was — see
877
+ // {@link KubernetesWorkspaceTransitionOptions.refreshPodTemplate}.
878
+ const refreshing = refresh !== undefined && suspended;
879
+ assertAdoptedWorkspaceMatchesConfig(name, namespace, existing?.spec?.podTemplate, {
880
+ sandboxTemplateName: expected.sandboxTemplateName,
881
+ // The RuntimeClass and egress-profile refusals are lifted for a call
882
+ // that is ABOUT TO WRITE them, and only for as long as that stays
883
+ // true: if the patch below does not land, both are re-applied against
884
+ // the object as it then stands, so a pod on the wrong runtime — or
885
+ // under a profile no policy selects — is never bound. They are lifted
886
+ // together because a refresh writes them together: the patched pod
887
+ // template is `sandboxPodTemplate`'s, overlays and all, so the
888
+ // profile label lands with the class. The TEMPLATE refusal is never
889
+ // lifted — a refresh rewrites a workspace's pod spec, it never moves
890
+ // the workspace to another template.
891
+ ...(refreshing
892
+ ? {}
893
+ : {
894
+ ...(expected.profile !== undefined ? { profile: expected.profile } : {}),
895
+ ...(expected.runtimeClassName !== undefined
896
+ ? { runtimeClassName: expected.runtimeClassName }
897
+ : {}),
898
+ }),
899
+ });
900
+ const reading = readHolderEpoch(existing?.metadata);
901
+ // With the adopt's other refusals, and for the same reason: an object this
902
+ // call will not use is not woken up on the way to being rejected. A
903
+ // superseded opener patches nothing and starts no pod.
904
+ if (epoch !== undefined)
905
+ assertHolderEpochAllows(target, 'createKubernetesWorkspace', epoch, reading);
906
+ if (refreshing && refresh !== undefined) {
907
+ // Both BEFORE the patch, and both against the SANDBOX's
908
+ // volumeClaimTemplates rather than the template's: those are
909
+ // CEL-immutable and are not in the patch, so the disks a refresh can
910
+ // be applied to are the disks this workspace already has, and the two
911
+ // questions worth asking are whether the template still claims them
912
+ // and whether it claims anything else.
913
+ const source = `SandboxTemplate ${expected.sandboxTemplateName} in namespace ${namespace}, refreshing Sandbox ${name}`;
914
+ assertRefreshedTemplateClaimsTheSameDisks(source, refresh.volumeClaimTemplates, existing?.spec?.volumeClaimTemplates);
915
+ // A template that stopped claiming this workspace's disk would come up
916
+ // healthy with the disk attached to nothing, and the only symptom
917
+ // would be that yesterday's files are gone.
918
+ assertBlockModeWorkspaceDisk(source, refresh.podTemplate, existing?.spec?.volumeClaimTemplates);
919
+ }
920
+ let templateRevision = readPodTemplateHash(existing?.metadata);
921
+ // Read BEFORE the resume patch, so what is recorded is the state this
922
+ // adopt WALKED INTO rather than one it provoked. It costs one GET on a
923
+ // path that is a rare, explicit act with nothing to amortise — the same
924
+ // trade the egress verification above makes — and it buys the one fact
925
+ // nothing else on this path can supply: whether the pod standing under
926
+ // this name is on its way out.
927
+ const drainingPodUid = await readDrainingPodUid(client, namespace, name, signal);
928
+ let resumed = suspended;
929
+ /** Somebody else's resume beat this call's patch — see `awaitReplacement`. */
930
+ let racedResume = false;
931
+ if (refreshing && refresh !== undefined) {
932
+ const result = await writeRefreshedPodTemplate(target, 'createKubernetesWorkspace', refresh, epoch, reading, signal);
933
+ if (result.applied) {
934
+ templateRevision = refresh.hash;
935
+ }
936
+ else {
937
+ // Another process resumed it first: nothing was written, so this
938
+ // is an adopt of a Running object and behaves like one — the
939
+ // RuntimeClass refusal applies again, against the object as it
940
+ // now stands, and the pod that is there is bound unchanged.
941
+ resumed = false;
942
+ racedResume = true;
943
+ assertAdoptedWorkspaceMatchesConfig(name, namespace, result.sandbox.spec?.podTemplate, expected);
944
+ templateRevision = readPodTemplateHash(result.sandbox.metadata);
945
+ if (epoch !== undefined) {
946
+ await writeOperatingMode(target, 'createKubernetesWorkspace', undefined, epoch, signal, readHolderEpoch(result.sandbox.metadata));
947
+ }
948
+ }
949
+ }
950
+ else if (resumed) {
951
+ await writeOperatingMode(target, 'createKubernetesWorkspace', 'Running', epoch, signal, reading);
952
+ }
953
+ else if (epoch !== undefined) {
954
+ // The one write this path did not use to make. Adopting a RUNNING
955
+ // workspace sent nothing at all, which is precisely what left race 1
956
+ // open: the new holder took the workspace over and left no trace on
957
+ // the object, so a superseded holder's later suspend had nothing to
958
+ // be refused by. Taking a workspace over is the moment the fence
959
+ // moves, whether or not the mode moves with it.
960
+ await writeOperatingMode(target, 'createKubernetesWorkspace', undefined, epoch, signal, reading);
961
+ }
962
+ return {
963
+ resumed,
964
+ ...(templateRevision !== undefined ? { templateRevision } : {}),
965
+ ...(racedResume ? { awaitReplacement: true } : {}),
966
+ ...(drainingPodUid !== undefined ? { drainingPodUid } : {}),
967
+ };
968
+ }
969
+ /**
970
+ * The uid of the pod standing under this Sandbox's name IF it is draining,
971
+ * and `undefined` for every other answer — no pod, or a pod that is fine.
972
+ *
973
+ * One GET by the Sandbox's name, the same convention
974
+ * {@link readBoundPod}'s fast path uses: the pod is named after its
975
+ * Sandbox in the controller this backend targets. No selector fallback,
976
+ * because the cost of missing a draining pod here is one adopt that fails the
977
+ * way it does today rather than a wrong answer, while a list on every adopt
978
+ * is a round trip paid by every caller for a state most of them are not in.
979
+ *
980
+ * A 404 is "no pod", which is the ordinary answer on a suspended workspace.
981
+ * Anything else is rethrown: a host that cannot read pods cannot read a bind
982
+ * token either, and hearing it here names the real problem.
983
+ */
984
+ async function readDrainingPodUid(client, namespace, name, signal) {
985
+ let pod;
986
+ try {
987
+ pod = await client.request('GET', podPath(namespace, name), undefined, signal);
988
+ }
989
+ catch (err) {
990
+ if (err instanceof KubernetesAlreadyGoneError)
991
+ return undefined;
992
+ throw err;
993
+ }
994
+ // `!= null` for the same reason `isPodLive` uses it: an explicit JSON null
995
+ // must not read as "terminating".
996
+ if (pod?.metadata?.deletionTimestamp == null)
997
+ return undefined;
998
+ const uid = pod.metadata.uid;
999
+ return typeof uid === 'string' && uid !== '' ? uid : undefined;
1000
+ }
1001
+ /**
1002
+ * The UNFENCED operating-mode merge patch — what this module sends when a
1003
+ * call carries no holder epoch, which is every call that carried none before
1004
+ * epochs existed, byte for byte and content type included. The fenced form is
1005
+ * `objects.ts`'s `buildHolderEpochPatch`; both go out through
1006
+ * {@link writeOperatingMode} and nothing else builds either.
1007
+ *
1008
+ * Built per call rather than held as constants because each one stamps
1009
+ * {@link OPERATING_MODE_CHANGED_AT_ANNOTATION_KEY} with the moment it was
1010
+ * sent. That annotation is what
1011
+ * {@link listKubernetesWorkspaces} reports as `operatingModeChangedAt`, and
1012
+ * the patch is the only place the fact exists: the controller's `Suspended`
1013
+ * condition lingers True across a resume, so neither its presence nor its
1014
+ * `lastTransitionTime` can be read as "when did this change" — see the
1015
+ * annotation's own comment.
1016
+ *
1017
+ * The body stays otherwise minimal, and a JSON merge patch (RFC 7386, which
1018
+ * is what `k8s-client.ts` sends) recurses into `metadata.annotations` rather
1019
+ * than replacing the map, so a Sandbox carrying annotations somebody else put
1020
+ * there keeps them.
1021
+ */
1022
+ function operatingModePatch(mode) {
1023
+ return {
1024
+ metadata: {
1025
+ annotations: { [OPERATING_MODE_CHANGED_AT_ANNOTATION_KEY]: new Date().toISOString() },
1026
+ },
1027
+ spec: { operatingMode: mode },
1028
+ };
1029
+ }
1030
+ /**
1031
+ * How many times one fenced write may be re-read and re-sent before it gives
1032
+ * up.
1033
+ *
1034
+ * A failed `test` is only worth retrying when the value the patch tested has
1035
+ * actually MOVED, and the loop checks that on every attempt (see
1036
+ * {@link writeOperatingMode}), so this is not a "retry until it works" budget
1037
+ * — it is the ceiling on how many times a workspace may legitimately change
1038
+ * underneath one call. Three is one more than the case that really happens: a
1039
+ * controller status write moving `resourceVersion` between the read and the
1040
+ * write of a workspace that carries no epoch annotation yet, which needs one
1041
+ * re-read and then succeeds.
1042
+ */
1043
+ const HOLDER_EPOCH_WRITE_ATTEMPTS = 3;
1044
+ /** One GET of the Sandbox, which is where every condition is read from. */
1045
+ async function readSandboxObject(target, signal) {
1046
+ return await target.client.request('GET', sandboxPath(target.namespace, target.name), undefined, signal);
1047
+ }
1048
+ /** `spec.operatingMode`, with the CRD's own default for an absent one. */
1049
+ function operatingModeOf(sandbox) {
1050
+ return sandbox?.spec?.operatingMode === 'Suspended' ? 'Suspended' : 'Running';
1051
+ }
1052
+ /** Refuse a write the stored epoch has moved past. Changes nothing anywhere. */
1053
+ function assertHolderEpochAllows(target, operation, epoch, reading) {
1054
+ if (holderEpochAllows(reading, epoch))
1055
+ return;
1056
+ throw new KubernetesWorkspacePreconditionError(operation, target.workspaceId, target.name, epoch, reading.epoch, reading.annotation);
1057
+ }
1058
+ /**
1059
+ * Read the object and refuse the write if this caller has been superseded,
1060
+ * BEFORE the caller does anything it cannot undo.
1061
+ *
1062
+ * The refusal it raises is the same one the write itself would raise; what
1063
+ * this buys is WHEN. `suspend()` kills every terminal it handed out and
1064
+ * `destroy()` tears its session down, both before any request goes out, so a
1065
+ * superseded holder that found out from the write would have taken its own
1066
+ * caller's sessions away on a write that never applied. The reading it hands
1067
+ * back is then used as the first attempt's condition, so the gate costs no
1068
+ * extra round trip.
1069
+ *
1070
+ * With no epoch there is nothing to check and nothing is read: the request
1071
+ * log of an unfenced call is what it always was.
1072
+ */
1073
+ async function readHolderEpochGate(target, operation, epoch, signal) {
1074
+ if (epoch === undefined)
1075
+ return undefined;
1076
+ const reading = readHolderEpoch((await readSandboxObject(target, signal))?.metadata);
1077
+ assertHolderEpochAllows(target, operation, epoch, reading);
1078
+ return reading;
1079
+ }
1080
+ /**
1081
+ * The single send path for every `spec.operatingMode` write this backend
1082
+ * makes, fenced or not.
1083
+ *
1084
+ * Unfenced (`epoch === undefined`) it sends the merge patch it always sent,
1085
+ * content type included, and a `mode` of `undefined` sends nothing at all —
1086
+ * there is no such thing as an unfenced write with no mutation in it.
1087
+ *
1088
+ * Fenced, it is one request: {@link buildHolderEpochPatch} puts the `test`
1089
+ * and the mutation in the same body, so nothing can fit between them. The
1090
+ * loop around it is NOT a retry-until-it-works — the API server answers every
1091
+ * unapplied JSON patch with the same opaque 422 whether the `test` failed or
1092
+ * the body was wrong (see {@link KubernetesPatchNotAppliedError}), so the
1093
+ * only honest way to tell them apart is to look: re-read the object, and if
1094
+ * the value this patch tested is still exactly what it tested, the patch did
1095
+ * not lose a race and is simply wrong, so the error stands. If it HAS moved,
1096
+ * the new reading is checked against this call's epoch like any other — a
1097
+ * holder that overtook this one is refused, and a controller status write
1098
+ * that only moved `resourceVersion` is retried on the fresh read.
1099
+ *
1100
+ * `mode` omitted with an epoch present is the stamp-only write: take the
1101
+ * workspace over without changing what it is doing. It deliberately does not
1102
+ * stamp {@link OPERATING_MODE_CHANGED_AT_ANNOTATION_KEY} — the mode did not
1103
+ * change, and that annotation is an inventory column that has to stay true.
1104
+ */
1105
+ async function writeOperatingMode(target, operation, mode, epoch, signal, gate) {
1106
+ const path = sandboxPath(target.namespace, target.name);
1107
+ if (epoch === undefined) {
1108
+ if (mode === undefined)
1109
+ return;
1110
+ await target.client.request('PATCH', path, operatingModePatch(mode), signal);
1111
+ return;
1112
+ }
1113
+ let reading = gate ?? readHolderEpoch((await readSandboxObject(target, signal))?.metadata);
1114
+ for (let attempt = 1;; attempt += 1) {
1115
+ assertHolderEpochAllows(target, operation, epoch, reading);
1116
+ const patch = buildHolderEpochPatch({
1117
+ reading,
1118
+ epoch,
1119
+ ...(mode !== undefined
1120
+ ? { operatingMode: mode, operatingModeChangedAt: new Date().toISOString() }
1121
+ : {}),
1122
+ });
1123
+ try {
1124
+ await target.client.request('PATCH', path, patch, signal, 'json');
1125
+ return;
1126
+ }
1127
+ catch (err) {
1128
+ if (!(err instanceof KubernetesPatchNotAppliedError))
1129
+ throw err;
1130
+ if (attempt >= HOLDER_EPOCH_WRITE_ATTEMPTS)
1131
+ throw err;
1132
+ const next = readHolderEpoch((await readSandboxObject(target, signal))?.metadata);
1133
+ const unmoved = next.annotation === reading.annotation &&
1134
+ next.resourceVersion === reading.resourceVersion &&
1135
+ next.hasAnnotations === reading.hasAnnotations;
1136
+ if (unmoved)
1137
+ throw err;
1138
+ reading = next;
1139
+ }
1140
+ }
1141
+ }
1142
+ /**
1143
+ * The overlays of `buildSandboxBody`'s create body, plus their revision.
1144
+ *
1145
+ * `podLabels` is `composeAdditionalPodLabels`'s map — the egress profile
1146
+ * today — and it is an overlay like the other two rather than an extra: the
1147
+ * patch this feeds replaces `/spec/podTemplate` whole, so a refresh that left
1148
+ * it out would REMOVE the profile label from a workspace created with it. The
1149
+ * replacement pod would come up selected by no per-profile policy, on a path
1150
+ * where nothing re-reads the label, and the hash would disagree with what
1151
+ * every create POSTs, so `templateCurrent` would report drift that no refresh
1152
+ * could clear.
1153
+ */
1154
+ function buildPodTemplateRefresh(template, sandboxTemplateName, runtimeClassName, podLabels) {
1155
+ const podTemplate = sandboxPodTemplate(template, sandboxTemplateName, runtimeClassName, podLabels);
1156
+ return {
1157
+ podTemplate,
1158
+ hash: podTemplateHash(podTemplate),
1159
+ ...(template.volumeClaimTemplates !== undefined
1160
+ ? { volumeClaimTemplates: template.volumeClaimTemplates }
1161
+ : {}),
1162
+ };
1163
+ }
1164
+ /**
1165
+ * The Suspended→Running write that also rewrites `spec.podTemplate`: ONE
1166
+ * conditional patch, built by the one builder this backend has.
1167
+ *
1168
+ * `test /spec/operatingMode == "Suspended"` is what enforces "only on a
1169
+ * suspended workspace" — not the read that preceded it, which another
1170
+ * process's resume can invalidate in the time it takes to send this. When the
1171
+ * caller also carries a holder epoch, that clause rides in the SAME body
1172
+ * (`buildHolderEpochPatch`'s `tests` seam), so the two conditions are one
1173
+ * request and there is no window between them.
1174
+ *
1175
+ * The discrimination is the one W8 measured and has to be repeated here
1176
+ * rather than shared with {@link writeOperatingMode}, because the two want
1177
+ * opposite things from the same 422: a real API server answers an unapplied
1178
+ * JSON patch identically whether a `test` failed or the body was malformed,
1179
+ * so the only way to tell is to re-read the object.
1180
+ *
1181
+ * - the object is GONE ⇒ the re-read 404s and says so, and that error is
1182
+ * what the caller hears. Neither clause can be said to have refused the
1183
+ * patch, and there is no pod to fall back to.
1184
+ * - the mode is no longer `Suspended` ⇒ the mode clause is what refused it.
1185
+ * That is an outcome, not an error: report it, after checking that this
1186
+ * caller has not ALSO been superseded, which is a refusal.
1187
+ * - the mode is still `Suspended` and the call carries no epoch ⇒ the only
1188
+ * condition in the body was true, so the body is what the server refused.
1189
+ * Nothing to retry.
1190
+ * - otherwise the epoch clause is the suspect: a stored epoch above this
1191
+ * caller's refuses it by name, one that merely moved is retried on the
1192
+ * fresh reading, and an object that did not move at all means the body is
1193
+ * wrong and the error stands.
1194
+ */
1195
+ async function writeRefreshedPodTemplate(target, operation, refresh, epoch, gate, signal) {
1196
+ const path = sandboxPath(target.namespace, target.name);
1197
+ let reading = gate;
1198
+ for (let attempt = 1;; attempt += 1) {
1199
+ if (epoch !== undefined)
1200
+ assertHolderEpochAllows(target, operation, epoch, reading);
1201
+ const patch = buildHolderEpochPatch({
1202
+ reading,
1203
+ ...(epoch !== undefined ? { epoch } : {}),
1204
+ tests: [{ op: 'test', path: '/spec/operatingMode', value: 'Suspended' }],
1205
+ annotations: { [POD_TEMPLATE_HASH_ANNOTATION_KEY]: refresh.hash },
1206
+ podTemplate: refresh.podTemplate,
1207
+ operatingMode: 'Running',
1208
+ operatingModeChangedAt: new Date().toISOString(),
1209
+ });
1210
+ try {
1211
+ await target.client.request('PATCH', path, patch, signal, 'json');
1212
+ return { applied: true };
1213
+ }
1214
+ catch (err) {
1215
+ if (!(err instanceof KubernetesPatchNotAppliedError))
1216
+ throw err;
1217
+ const sandbox = await readSandboxObject(target, signal);
1218
+ // An object DELETED in this window does not arrive here as
1219
+ // `undefined`: the GET 404s and `readSandboxObject` throws
1220
+ // `KubernetesAlreadyGoneError`, which is the truthful answer and is
1221
+ // let through. `undefined` is the API server answering 200 with no
1222
+ // body — a shape this code has no reading of. It must not fall
1223
+ // through, because `operatingModeOf(undefined)` is `'Running'` and
1224
+ // the branch below means "somebody else's pod is standing under
1225
+ // this name", which would send the caller on to bind a pod nothing
1226
+ // established was there, after refusals made against an object
1227
+ // nobody read. The original error says the one thing that is known
1228
+ // — the patch did not apply and nothing changed.
1229
+ if (sandbox === undefined)
1230
+ throw err;
1231
+ const next = readHolderEpoch(sandbox.metadata);
1232
+ if (operatingModeOf(sandbox) !== 'Suspended') {
1233
+ // Somebody resumed it first. If they also took the workspace
1234
+ // over, this caller hears THAT rather than being handed a pod
1235
+ // it is no longer entitled to bind.
1236
+ if (epoch !== undefined)
1237
+ assertHolderEpochAllows(target, operation, epoch, next);
1238
+ return { applied: false, sandbox };
1239
+ }
1240
+ if (epoch === undefined)
1241
+ throw err;
1242
+ if (attempt >= HOLDER_EPOCH_WRITE_ATTEMPTS)
1243
+ throw err;
1244
+ const unmoved = next.annotation === reading.annotation &&
1245
+ next.resourceVersion === reading.resourceVersion &&
1246
+ next.hasAnnotations === reading.hasAnnotations;
1247
+ if (unmoved)
1248
+ throw err;
1249
+ reading = next;
1250
+ }
1251
+ }
1252
+ }
1253
+ /**
1254
+ * DELETE the Sandbox, fenced by the epoch when the caller supplied one.
1255
+ *
1256
+ * The condition here cannot be a JSON Patch `test`, because a DELETE has no
1257
+ * patch body — so it is `preconditions.resourceVersion` on the very version
1258
+ * whose epoch was just read, which the API server answers with a 409 naming
1259
+ * both versions when it no longer matches (measured). Anything that touched
1260
+ * the object between the read and the DELETE — another holder raising the
1261
+ * epoch included — moves that version, so the DELETE is refused and re-read
1262
+ * rather than taking a disk on a stale view.
1263
+ *
1264
+ * Unfenced it is the bodyless DELETE it always was. Already gone counts as
1265
+ * deleted either way, that being the state DELETE was asking for.
1266
+ */
1267
+ async function deleteSandboxObject(target, operation, epoch, signal, gate) {
1268
+ const path = sandboxPath(target.namespace, target.name);
1269
+ if (epoch === undefined) {
1270
+ try {
1271
+ await target.client.request('DELETE', path, undefined, signal);
1272
+ }
1273
+ catch (err) {
1274
+ if (!(err instanceof KubernetesAlreadyGoneError))
1275
+ throw err;
1276
+ }
1277
+ return;
1278
+ }
1279
+ let reading = gate;
1280
+ for (let attempt = 1;; attempt += 1) {
1281
+ if (reading === undefined) {
1282
+ let sandbox;
1283
+ try {
1284
+ sandbox = await readSandboxObject(target, signal);
1285
+ }
1286
+ catch (err) {
1287
+ if (err instanceof KubernetesAlreadyGoneError)
1288
+ return;
1289
+ throw err;
1290
+ }
1291
+ reading = readHolderEpoch(sandbox?.metadata);
1292
+ }
1293
+ assertHolderEpochAllows(target, operation, epoch, reading);
1294
+ if (reading.resourceVersion === undefined) {
1295
+ // Unreachable against a real API server, which sets it on every
1296
+ // object it serves — and a bodyless DELETE here would be an
1297
+ // UNFENCED delete of somebody's disk, which is the one thing this
1298
+ // path may never silently become.
1299
+ throw new Error(`kubernetes: cannot delete workspace ${target.workspaceId} (Sandbox ${target.name}) under holder epoch ${epoch}: the object came back with no metadata.resourceVersion, so there is no precondition to send and the DELETE would be unconditional.`);
1300
+ }
1301
+ try {
1302
+ await target.client.request('DELETE', path, {
1303
+ apiVersion: 'v1',
1304
+ kind: 'DeleteOptions',
1305
+ preconditions: { resourceVersion: reading.resourceVersion },
1306
+ }, signal);
1307
+ return;
1308
+ }
1309
+ catch (err) {
1310
+ if (err instanceof KubernetesAlreadyGoneError)
1311
+ return;
1312
+ if (!(err instanceof KubernetesConflictError))
1313
+ throw err;
1314
+ if (attempt >= HOLDER_EPOCH_WRITE_ATTEMPTS)
1315
+ throw err;
1316
+ reading = undefined;
1317
+ }
1318
+ }
1319
+ }
1320
+ /** Shared tail of every bind timeout but the resume-specific one. */
1321
+ const NO_BINDABLE_POD_ADVICE = "A pod's uid is the agent's bind token, so a pod carrying a deletionTimestamp — or one in a terminal phase — is never bound to: its uid is a token the pod's replacement will refuse. The Ready condition cannot be waited on instead, because the controller leaves it standing across a transition. Raise readyTimeoutMs, or look at why the controller has not brought a pod up.";
1322
+ /**
1323
+ * True once the pod has actually stopped. The POD is asked, and nothing else
1324
+ * is consulted or believed.
1325
+ *
1326
+ * The cheap-looking alternative — the Sandbox's own `Suspended` condition —
1327
+ * is unusable, and upstream says so itself: "the controller does not
1328
+ * currently remove this condition when the Sandbox is resumed, so a stale
1329
+ * Suspended condition may linger after operatingMode returns to Running.
1330
+ * Consumers should treat Ready as the authoritative signal and not infer the
1331
+ * live operating state from the mere presence of this condition"
1332
+ * (`sandbox_types.go`). Reading it would make the second and every later
1333
+ * suspend of the same workspace return immediately, on a True left behind by
1334
+ * the previous one, while the guest was still running and still writing to
1335
+ * the caller's disk — and a `resume()` issued straight after such a false
1336
+ * suspend could bind to the pod that is about to be deleted.
1337
+ *
1338
+ * A `deletionTimestamp` is not the answer either: it is set the moment the
1339
+ * DELETE is accepted, and the container goes on running until it exits or
1340
+ * `terminationGracePeriodSeconds` expires. Gone (404) or stopped
1341
+ * ({@link isPodStopped}) — those are the only two states that mean the disk
1342
+ * is quiesced.
1343
+ */
1344
+ async function isPodRetired(client, namespace, name, signal) {
1345
+ try {
1346
+ const pod = await client.request('GET', podPath(namespace, name), undefined, signal);
1347
+ return isPodStopped(pod);
1348
+ }
1349
+ catch (err) {
1350
+ if (err instanceof KubernetesAlreadyGoneError)
1351
+ return true;
1352
+ throw err;
1353
+ }
1354
+ }
1355
+ /**
1356
+ * Poll {@link isPodRetired} until it answers true, or refuse with
1357
+ * {@link KubernetesWorkspaceSuspendTimeoutError}.
1358
+ *
1359
+ * At module scope, and taking its subject as arguments, because BOTH suspend
1360
+ * paths owe the caller the same wait: the handle's `suspend()`, which has a
1361
+ * session to tear down first, and {@link suspendKubernetesWorkspace}, which
1362
+ * has no handle at all. A suspend that resolved on the patch alone would
1363
+ * promise a quiesced disk it had not waited for, and that promise must not
1364
+ * depend on which entry point was used.
1365
+ */
1366
+ async function awaitPodRetired(client, namespace, name, workspaceId, readiness, signal) {
1367
+ const deadline = new OperationDeadline(readiness.timeoutMs, `kubernetes workspace ${name} suspend`, signal);
1368
+ while (deadline.remainingMs() > 0) {
1369
+ try {
1370
+ if (await deadline.run(async (pollSignal) => await isPodRetired(client, namespace, name, pollSignal)))
1371
+ return;
1372
+ await deadline.delay(readiness.pollIntervalMs);
1373
+ }
1374
+ catch (err) {
1375
+ if (err instanceof OperationDeadlineExpired)
1376
+ break;
1377
+ throw err;
1378
+ }
1379
+ }
1380
+ throw new KubernetesWorkspaceSuspendTimeoutError(workspaceId, name, readiness.timeoutMs);
1381
+ }
1382
+ /**
1383
+ * What `spec.operatingMode` says RIGHT NOW, read off the object and nothing
1384
+ * else.
1385
+ *
1386
+ * `Running` when the field is absent, which is the CRD's own default. Every
1387
+ * Sandbox this backend creates sets it explicitly, so an absent value means
1388
+ * an object somebody else made — and the API's default for it is Running.
1389
+ */
1390
+ async function readOperatingMode(client, namespace, name, signal) {
1391
+ const sandbox = await client.request('GET', sandboxPath(namespace, name), undefined, signal);
1392
+ return operatingModeOf(sandbox);
1393
+ }
1394
+ /**
1395
+ * Every workspace this backend owns in the namespace, WITHOUT waking one.
1396
+ *
1397
+ * The verb retention needs. Deleting a workspace nobody has resumed since
1398
+ * last month, or reporting what a namespace is holding, used to require
1399
+ * {@link createKubernetesWorkspace} — which adopts AND resumes: the inventory
1400
+ * pass would start a pod for every suspended workspace it looked at, probe
1401
+ * each one, and then have to put them all back. This issues exactly one GET
1402
+ * of the sandboxes collection and reads the objects.
1403
+ *
1404
+ * ## What counts as a workspace
1405
+ *
1406
+ * Two things together, and both are this backend's own marks:
1407
+ *
1408
+ * - the `namzu-ws-` name prefix — {@link workspaceSandboxName}'s contract,
1409
+ * and the only thing that makes a `workspaceId` recoverable from a
1410
+ * Sandbox at all;
1411
+ * - {@link SANDBOX_TEMPLATE_LABEL_KEY} on `spec.podTemplate.metadata.labels`,
1412
+ * which says this object was built by this backend and names the template
1413
+ * it came from.
1414
+ *
1415
+ * The filtering happens HERE rather than in a `labelSelector` on the request,
1416
+ * and the reason is where that label lives. It is a POD label — the one an
1417
+ * egress `NetworkPolicy`'s `podSelector` matches — written onto
1418
+ * `spec.podTemplate`, while a `labelSelector` on the sandboxes collection
1419
+ * matches the Sandbox's OWN `metadata.labels`, which this backend has never
1420
+ * written. Stamping a second copy up there to make a server-side selector
1421
+ * work would leave every workspace created before that change invisible to
1422
+ * this call, and an inventory that silently omits the oldest objects is worse
1423
+ * than no inventory at all — those are exactly the ones a retention pass is
1424
+ * looking for.
1425
+ *
1426
+ * ## What it never does
1427
+ *
1428
+ * No PATCH, no DELETE, no pod read, no dial. A workspace that was suspended
1429
+ * before this call is suspended after it, and a running one is untouched. The
1430
+ * order is the API server's own (name order); the caller sorts if it cares.
1431
+ */
1432
+ export async function listKubernetesWorkspaces(config, options) {
1433
+ options?.signal?.throwIfAborted();
1434
+ const namespace = config.namespace;
1435
+ const client = createKubernetesClient(clientAccess(config), clientOptions(config));
1436
+ const list = await client.request('GET', sandboxCollectionPath(namespace), undefined, options?.signal);
1437
+ const summaries = [];
1438
+ for (const sandbox of list?.items ?? []) {
1439
+ const name = sandbox?.metadata?.name;
1440
+ if (typeof name !== 'string' || !name.startsWith(WORKSPACE_NAME_PREFIX))
1441
+ continue;
1442
+ const template = sandbox?.spec?.podTemplate?.metadata?.labels?.[SANDBOX_TEMPLATE_LABEL_KEY];
1443
+ if (typeof template !== 'string' || template === '')
1444
+ continue;
1445
+ const createdAt = sandbox?.metadata?.creationTimestamp;
1446
+ const changedAt = sandbox?.metadata?.annotations?.[OPERATING_MODE_CHANGED_AT_ANNOTATION_KEY];
1447
+ const holder = readHolderEpoch(sandbox?.metadata);
1448
+ summaries.push({
1449
+ workspaceId: name.slice(WORKSPACE_NAME_PREFIX.length),
1450
+ operatingMode: operatingModeOf(sandbox),
1451
+ template,
1452
+ ...(typeof createdAt === 'string' && createdAt !== '' ? { createdAt } : {}),
1453
+ ...(typeof changedAt === 'string' && changedAt !== ''
1454
+ ? { operatingModeChangedAt: changedAt }
1455
+ : {}),
1456
+ ...(holder.epoch !== undefined ? { holderEpoch: holder.epoch } : {}),
1457
+ });
1458
+ }
1459
+ return summaries;
1460
+ }
1461
+ /**
1462
+ * DELETE a workspace by id, without ever adopting or resuming it.
1463
+ *
1464
+ * The same thing `destroy({ deleteDisk: true })` does to the cluster — one
1465
+ * DELETE of the Sandbox, which cascades to the Pod, the Service and the PVC
1466
+ * through ownerReferences — with the same two guarantees: an object already
1467
+ * gone counts as deleted, that being the state DELETE was asking for, and a
1468
+ * DELETE that FAILS rejects and stays retryable, because nothing here records
1469
+ * a state a retry could early-return on.
1470
+ *
1471
+ * What it does NOT do is the point. Removing a month-old suspended workspace
1472
+ * through a handle meant starting a pod for it first, probing it, and then
1473
+ * deleting the pod again — compute spent, and a guest woken, purely to be
1474
+ * told to go away. The DELETE never needed any of that: the name is
1475
+ * deterministic, so the object can be addressed without being opened.
1476
+ *
1477
+ * It is not gated on the workspace being suspended, and deliberately: a
1478
+ * running workspace's DELETE takes its pod down with it, which is what
1479
+ * deleting a workspace means. A caller that wants the disk quiesced first
1480
+ * calls {@link suspendKubernetesWorkspace} and then this.
1481
+ */
1482
+ export async function deleteKubernetesWorkspace(config, workspaceId, options) {
1483
+ options?.signal?.throwIfAborted();
1484
+ const namespace = config.namespace;
1485
+ const name = workspaceSandboxName(workspaceId);
1486
+ const epoch = assertHolderEpoch(options?.epoch, 'deleteKubernetesWorkspace');
1487
+ const client = createKubernetesClient(clientAccess(config), clientOptions(config));
1488
+ // The fence reaches the standalone verbs too, and this is the one it
1489
+ // matters most on: a retention job superseded between deciding to delete a
1490
+ // workspace and calling this would otherwise take the new holder's disk,
1491
+ // and nothing brings a disk back.
1492
+ await deleteSandboxObject({ client, namespace, name, workspaceId }, 'deleteKubernetesWorkspace', epoch, options?.signal);
1493
+ }
1494
+ /**
1495
+ * Suspend a workspace by id, without ever adopting it: send the patch, then
1496
+ * wait for the pod to actually stop.
1497
+ *
1498
+ * The handle's own `suspend()` does three things — reap the terminals it
1499
+ * handed out, stop admitting calls, and patch-and-wait. Only the third is
1500
+ * about the CLUSTER, and it is the only one a process holding no handle can
1501
+ * do or needs to. So this is that third thing on its own, for the operator
1502
+ * putting somebody else's workspace to sleep.
1503
+ *
1504
+ * It resolves only once the pod is gone or in a terminal phase, for the same
1505
+ * reason the handle's does: a suspend is a promise that the disk is quiesced,
1506
+ * and the patch being accepted says only that the controller has been asked.
1507
+ * A pod that outlives `readyTimeoutMs` rejects with
1508
+ * {@link KubernetesWorkspaceSuspendTimeoutError} and the object is left
1509
+ * exactly as the patch left it — the next call patches and waits again.
1510
+ *
1511
+ * A workspace that no longer exists rejects with the client's own
1512
+ * already-gone error rather than resolving. Unlike a DELETE, this asks for a
1513
+ * state that cannot be reached: there is no object to suspend, and nothing
1514
+ * the caller believed about it holds.
1515
+ *
1516
+ * Any HANDLE another process is holding on this workspace is not told. It
1517
+ * finds out on its next call, which fails at the transport and is re-read
1518
+ * into a {@link KubernetesWorkspaceSuspendedError} — or the moment that
1519
+ * process calls `refresh()`. See {@link KubernetesWorkspace.refresh}.
1520
+ */
1521
+ export async function suspendKubernetesWorkspace(config, workspaceId, options) {
1522
+ options?.signal?.throwIfAborted();
1523
+ const namespace = config.namespace;
1524
+ const name = workspaceSandboxName(workspaceId);
1525
+ const epoch = assertHolderEpoch(options?.epoch, 'suspendKubernetesWorkspace');
1526
+ assertNoQuiesceHere(options?.quiesce, 'suspendKubernetesWorkspace');
1527
+ assertNoFlushHere(options?.flush, 'suspendKubernetesWorkspace');
1528
+ const readiness = resolveKubernetesReadiness(config);
1529
+ const client = createKubernetesClient(clientAccess(config), clientOptions(config));
1530
+ await writeOperatingMode({ client, namespace, name, workspaceId }, 'suspendKubernetesWorkspace', 'Suspended', epoch, options?.signal);
1531
+ await awaitPodRetired(client, namespace, name, workspaceId, readiness, options?.signal);
1532
+ }
1533
+ /**
1534
+ * How long the diagnostic `healthz` after an unconfirmed cancellation may
1535
+ * take before the agent is reported `unreachable`.
1536
+ *
1537
+ * Its own clock, short, and unrelated to every other budget on the path. The
1538
+ * caller's deadline has usually already expired by the time this runs — the
1539
+ * shared controller has just spent its whole eight-second cancel-confirm
1540
+ * window — and the readiness budget is for waiting out a pod that is coming
1541
+ * up, which is not what this is asking. This asks one question of a pod that
1542
+ * is already there, and an answer that has not arrived in five seconds is
1543
+ * not going to change what the host does with it.
1544
+ */
1545
+ const CANCELLATION_DIAGNOSIS_TIMEOUT_MS = 5_000;
1546
+ /**
1547
+ * The identity evidence gathered about one unconfirmed cancellation, keyed
1548
+ * by the error the caller is about to receive.
1549
+ *
1550
+ * A WeakMap rather than a field, because the error class belongs to the
1551
+ * shared execution controller and this is one backend's evidence ABOUT it:
1552
+ * adding a Kubernetes-shaped property to `RemoteCancellationUnknownError`
1553
+ * would put a field on the Firecracker tier's errors that nothing there can
1554
+ * ever fill. Weak, so an error nobody kept takes its entry with it.
1555
+ *
1556
+ * It is written where the diagnosis happens ({@link
1557
+ * KubernetesWorkspace.exec}'s retirement hook) and read once, where the error
1558
+ * passes back through the handle, which is the only place that can rename it.
1559
+ */
1560
+ const guestGoneEvidence = new WeakMap();
1561
+ /**
1562
+ * The `reason` an unconfirmed cancellation reports on a workspace — see
1563
+ * {@link SandboxRetirementObservation.reason}. Not "the patch failed": no
1564
+ * patch was attempted, and the pod is standing on purpose.
1565
+ */
1566
+ const WORKSPACE_KEPT_REASON = 'workspace-kept';
1567
+ async function openWorkspaceHandle(options) {
1568
+ const { client, namespace, name, workspaceId, readiness } = options;
1569
+ const id = name;
1570
+ /** What every fenced write addresses, and what a refusal names. */
1571
+ const target = { client, namespace, name, workspaceId };
1572
+ /**
1573
+ * The epoch this handle writes under when a call passes none.
1574
+ *
1575
+ * It is the one the handle was opened with, and it moves to whatever a
1576
+ * later call wrote under successfully — never down, because a write only
1577
+ * succeeds when its epoch was at least the stored one. Letting it lag
1578
+ * behind a write this handle itself made would be the worst of both:
1579
+ * every later `suspend()` of its own would be refused by its own stamp.
1580
+ */
1581
+ let heldEpoch = options.epoch;
1582
+ /**
1583
+ * The pod-template revision the BOUND object carries, and the one the
1584
+ * template this handle last read would produce. Both move together on a
1585
+ * refresh; a plain `resume()` re-reads only the first, off the `GET` it
1586
+ * already makes.
1587
+ */
1588
+ let templateRevision = options.templateRevision;
1589
+ let currentTemplateHash = options.currentTemplateHash;
1590
+ /** The handle's own policy; a transition may override it for one call. */
1591
+ const defaultStartFailure = options.onStartFailure ?? 'suspend-if-woken';
1592
+ let state = 'running';
1593
+ let session;
1594
+ /**
1595
+ * The uid of the pod `session` is bound to — the agent's token, and the
1596
+ * only way a resume can tell the pod it is waiting for from the one it is
1597
+ * replacing. Read once per session and never carried across one.
1598
+ */
1599
+ let podUid;
1600
+ /**
1601
+ * Which session `podUid` belongs to. Bumped when a session starts AND
1602
+ * when one is dropped ({@link dropSession}), so the number identifies the
1603
+ * LIVE session and nothing else — a transport whose session has been
1604
+ * retired never matches it again. Read by {@link refreshBoundPod}, which
1605
+ * is where the desync it prevents is described, and by
1606
+ * {@link boundToPod}.
1607
+ */
1608
+ let sessionSeq = 0;
1609
+ /**
1610
+ * The session `podUid` was written for — see {@link boundToPod}.
1611
+ *
1612
+ * Deliberately NOT `sessionSeq` itself: a dropped session bumps that
1613
+ * counter and writes nothing else, so "the pod belongs to the session
1614
+ * that is live" is a comparison rather than a flag anybody has to
1615
+ * remember to clear. It starts before the first session so that a handle
1616
+ * which has not bound anything yet is bound to nothing.
1617
+ */
1618
+ let podSession = -1;
1619
+ /**
1620
+ * The uid of the pod a suspend patch that LANDED took away, cleared once a
1621
+ * resume has bound its replacement.
1622
+ *
1623
+ * This — not `podUid` — is what a resume excludes; see
1624
+ * {@link acquireBoundPod}. It is set from the PATCH rather than from the
1625
+ * wait that follows it, so a suspend whose pod outlived `readyTimeoutMs`
1626
+ * excludes that pod too: the controller was asked to delete it either
1627
+ * way, and a resume arriving while it drains is handed exactly that pod.
1628
+ * Every path that sends that patch records it here: `suspend()`, the
1629
+ * retirement of a pod that stopped answering, and the cleanup after a
1630
+ * failed create or resume — which swallows its own failure, and so
1631
+ * records only when the request actually came back. A pod nobody asked
1632
+ * the controller to remove has no replacement to wait for, and excluding
1633
+ * it would time out a resume whose workspace was perfectly usable.
1634
+ */
1635
+ let retiredPodUid;
1636
+ /**
1637
+ * The Sandbox object this handle was opened on, and the disks behind it.
1638
+ *
1639
+ * `sandboxUid` is taken from the readiness GET the first bind already
1640
+ * makes — the object was read, its uid was in the reply, and until now it
1641
+ * was thrown away. It is what a rebind compares against: the workspace
1642
+ * name is DETERMINISTIC, so an object deleted and recreated stands under
1643
+ * the same name with an empty disk, and following a pod behind a
1644
+ * different uid would hand a caller a stranger's workspace.
1645
+ *
1646
+ * `volumeClaimUids` is read ONCE, at the first bind, and not again. The
1647
+ * PVCs belong to the Sandbox: while `sandboxUid` has not moved they have
1648
+ * not either, so re-reading them on every resume would be a GET per
1649
+ * volume per transition for an answer that cannot have changed. It stays
1650
+ * empty when the Role does not grant `get` on `persistentvolumeclaims` —
1651
+ * an existing deployment upgrading into this release must not have every
1652
+ * `createKubernetesWorkspace` start failing on a verb its Role has never
1653
+ * had.
1654
+ */
1655
+ let sandboxUid;
1656
+ let volumeClaimUids = {};
1657
+ let volumeClaimsRead = false;
1658
+ /**
1659
+ * The agent PROCESS the handle is talking to: the boot id of the last
1660
+ * reply that carried one.
1661
+ *
1662
+ * Cleared whenever the pod changes — a session start, and a rebind that
1663
+ * followed a replacement — which is what keeps the caller's own
1664
+ * `suspend()`/`resume()` from being reported as a restart: the new pod's
1665
+ * first reply establishes a baseline instead of being compared against
1666
+ * the old pod's.
1667
+ *
1668
+ * It is deliberately NOT a baseline for the cancellation diagnosis.
1669
+ * Whether the guest a COMMAND was running in is gone is a question about
1670
+ * that command, and the boot id it reserved against is kept per execution
1671
+ * by the transport ({@link guestBootIdWhenReserved}); a handle-wide
1672
+ * baseline would answer "restarted" for every command started after a
1673
+ * restart the handle survived.
1674
+ *
1675
+ * Against a guest too old to report one it stays `undefined` for ever,
1676
+ * and nothing here fires. That is the interop contract: a missing boot id
1677
+ * is "this guest cannot tell me", never "it changed".
1678
+ */
1679
+ let guestBootId;
1680
+ /** Subscribers to {@link KubernetesWorkspace.onGuestRestart}. */
1681
+ const restartListeners = new Set();
1682
+ const terminals = new Set();
1683
+ /**
1684
+ * This handle's disk, around ONE named pod and the process inside it.
1685
+ *
1686
+ * Every identity this file hands out is built here, which is what keeps a
1687
+ * payload from being blanked by a state the handle happens to be in: a
1688
+ * routine that is HOLDING a pod uid reports that uid, and the only
1689
+ * question left is which pod it holds.
1690
+ */
1691
+ const identityOn = (pod, boot) => ({
1692
+ sandboxUid,
1693
+ volumeClaimUids: { ...volumeClaimUids },
1694
+ podUid: pod,
1695
+ guestBootId: boot,
1696
+ });
1697
+ /**
1698
+ * Whether `podUid` still names the pod this handle is TALKING to.
1699
+ *
1700
+ * The question is about the SESSION and never about `state`, and the
1701
+ * difference is not academic: a transition is not the absence of a pod.
1702
+ * {@link startSession} binds one, records it and runs the privilege probe
1703
+ * against it while `state` is still `'suspended'`, so a handle that
1704
+ * answered "no pod" for the length of a resume would blank exactly the
1705
+ * announcement this backend exists to make — a pod somebody else replaced
1706
+ * underneath a resume, discovered when that probe is refused.
1707
+ *
1708
+ * Every path that gives a pod back drops the session first
1709
+ * ({@link dropSession}), which bumps `sessionSeq` and leaves this false;
1710
+ * every path that takes one binds through `startSession`, which sets both
1711
+ * together. So there is no window in which this says yes about a pod
1712
+ * nothing is talking to.
1713
+ */
1714
+ const boundToPod = () => podUid !== undefined && podSession === sessionSeq;
1715
+ /** This handle's four objects, read fresh — see {@link KubernetesWorkspaceIdentity}. */
1716
+ const identityNow = () =>
1717
+ // A handle with no session has no pod and so no agent process.
1718
+ // Reporting the ones it HAD would be the single most misleading thing
1719
+ // this object could say: the pod is deleted and those ids name nothing.
1720
+ boundToPod() ? identityOn(podUid, guestBootId) : identityOn(undefined, undefined);
1721
+ /**
1722
+ * This handle's identity with the pod a LOOK actually FOUND, naming an
1723
+ * agent process only when that pod is the one the handle's boot id came
1724
+ * from.
1725
+ *
1726
+ * The two halves have to come from the same pod or the answer is a
1727
+ * fabrication: the boot id of the pod this handle is bound to, printed
1728
+ * beside the uid of a replacement nobody has heard a word from, names a
1729
+ * process that pod never ran. `undefined` is what "nothing has answered
1730
+ * from there yet" looks like, and it is the only honest thing to say.
1731
+ */
1732
+ const guestSeen = (uid) => {
1733
+ const live = identityNow();
1734
+ return {
1735
+ ...live,
1736
+ podUid: uid,
1737
+ guestBootId: uid !== undefined && uid === live.podUid ? live.guestBootId : undefined,
1738
+ };
1739
+ };
1740
+ /**
1741
+ * This handle's identity as one COMMAND saw it: the pod its reservation
1742
+ * was accepted by and the process inside it, where the transport kept
1743
+ * them ({@link guestWhenReserved}), and the handle's own where it did not.
1744
+ *
1745
+ * What a diagnosis compares against has to be the command's guest and not
1746
+ * the handle's. The handle moves — it follows a replaced pod, and the
1747
+ * failing call's own `cancel-execution` refusal already advanced its boot
1748
+ * id on the way in — so reading it here would report the guest that is
1749
+ * running NOW as the guest that died.
1750
+ */
1751
+ const identityWhenReserved = (reserved) => {
1752
+ const live = identityNow();
1753
+ if (reserved === undefined)
1754
+ return live;
1755
+ const pod = reserved.podUid ?? live.podUid;
1756
+ return {
1757
+ ...live,
1758
+ podUid: pod,
1759
+ // Never the handle's boot id for a DIFFERENT pod than the one
1760
+ // being named — see {@link guestSeen}.
1761
+ guestBootId: reserved.guestBootId ?? (pod === live.podUid ? live.guestBootId : undefined),
1762
+ };
1763
+ };
1764
+ /**
1765
+ * Tell every subscriber, and let none of them fail the call that noticed.
1766
+ *
1767
+ * Synchronous and in subscription order, so a host can keep its own book
1768
+ * up to date before the call that discovered the restart returns. A
1769
+ * listener that throws is swallowed for the reason every notification
1770
+ * callback in this file is: the caller's result is already decided, and a
1771
+ * host's bookkeeping bug must not become the workspace's error.
1772
+ */
1773
+ const announceGuestRestart = (event) => {
1774
+ for (const listener of [...restartListeners]) {
1775
+ try {
1776
+ listener(event);
1777
+ }
1778
+ catch {
1779
+ // See above.
1780
+ }
1781
+ }
1782
+ };
1783
+ /**
1784
+ * Follow the agent PROCESS behind every authenticated reply.
1785
+ *
1786
+ * Installed on the transport for both address modes, because this is the
1787
+ * one thing no address and no token can reveal: the kubelet restarts a
1788
+ * crashed container INSIDE the same pod, so the uid — and therefore the
1789
+ * bind token — is unchanged and every call keeps working, while every
1790
+ * process the caller started is gone. Only the guest's own boot id says
1791
+ * so, and only on replies it was already sending.
1792
+ *
1793
+ * It never fires for the FIRST reply after the pod changed: `guestBootId`
1794
+ * is undefined then, and the first value is the baseline rather than a
1795
+ * change. That is what makes the caller's own suspend/resume silent.
1796
+ *
1797
+ * `generation` guards it exactly as {@link refreshBoundPod}'s does, and
1798
+ * for the same reason: a reply from a session this handle has already let
1799
+ * go says nothing about the guest it is bound to now. A late frame from a
1800
+ * retired transport would otherwise set the boot id back to the dead
1801
+ * pod's, announce a restart nobody had, and make the live pod's next
1802
+ * reply announce a second one.
1803
+ *
1804
+ * `from` — the bind token of the wire the reply arrived on — is the same
1805
+ * guard one level down, and a REBIND needs it because it does not retire
1806
+ * the session: it swaps the pod under a session that goes on. The
1807
+ * transport keeps no lock on the wire it leaves, so the outgoing pod can
1808
+ * still answer a call dispatched to it (a `write-file` mid-flight, a pod
1809
+ * inside its termination grace period) after this handle has followed the
1810
+ * replacement. Taken as this session's, that reply would seed the
1811
+ * replacement's baseline with the DEPARTED pod's process — so
1812
+ * `workspace.identity`, and the `container-restarted` event the next real
1813
+ * reply then fires, would name a process that never ran in the pod beside
1814
+ * it. A pod the handle has left says nothing at all, which is exactly
1815
+ * what a retired session's reply says.
1816
+ */
1817
+ const observeGuestReply = (generation, from, reply) => {
1818
+ if (generation !== sessionSeq)
1819
+ return;
1820
+ if (from !== podUid)
1821
+ return;
1822
+ const seen = reply.guestBootId;
1823
+ if (typeof seen !== 'string' || seen === '')
1824
+ return;
1825
+ if (guestBootId === undefined) {
1826
+ guestBootId = seen;
1827
+ return;
1828
+ }
1829
+ if (seen === guestBootId)
1830
+ return;
1831
+ // Both halves name the pod this reply came from, taken from the
1832
+ // variable rather than from {@link identityNow}: the two guards above
1833
+ // have already established that `podUid` IS the pod that answered,
1834
+ // and an announcement about a guest must not be able to come out
1835
+ // naming no guest at all.
1836
+ const previous = identityOn(podUid, guestBootId);
1837
+ guestBootId = seen;
1838
+ announceGuestRestart({
1839
+ reason: 'container-restarted',
1840
+ previous,
1841
+ current: identityOn(podUid, seen),
1842
+ });
1843
+ };
1844
+ /**
1845
+ * Let go of one terminal this handle handed out, on the way to giving the
1846
+ * pod back.
1847
+ *
1848
+ * A connection-bound terminal is KILLED, exactly as it always was: its
1849
+ * program lives on this connection and the pod is going away. A session
1850
+ * ATTACHMENT is only detached — the program is the registry's, not this
1851
+ * connection's, and this path exists to stop a caller waiting on an
1852
+ * `exited` that would otherwise resolve only when TCP notices, not to
1853
+ * decide the program's fate. Either way the session dies with the pod;
1854
+ * what differs is whether this handle claims to have ended it.
1855
+ */
1856
+ const releaseTerminal = (terminal) => {
1857
+ if (typeof terminal.detach === 'function')
1858
+ terminal.detach();
1859
+ else
1860
+ terminal.kill('SIGKILL');
1861
+ };
1862
+ /**
1863
+ * The transport behind each session's inner handle.
1864
+ *
1865
+ * The detach/attach ops are the workspace's own surface, not the SDK
1866
+ * `Sandbox`'s, so they are reached on the transport rather than through
1867
+ * the inner handle — which also keeps them off the inner handle's
1868
+ * automatic-retirement path, where an unconfirmed cancel takes the pod
1869
+ * away. A detached command losing its connection must cost the
1870
+ * workspace nothing, so it must not travel that road at all.
1871
+ *
1872
+ * Keyed by handle rather than kept in a `let`, so there is no window in
1873
+ * which `session` and the transport disagree: `admitted()` hands out the
1874
+ * live handle, and the transport looked up from it is that handle's own
1875
+ * or nothing.
1876
+ */
1877
+ const sessionTransports = new WeakMap();
1878
+ /** The live session's transport, refused by name if there is none. */
1879
+ const admittedTransport = (operation, handle) => {
1880
+ const transport = sessionTransports.get(handle);
1881
+ if (transport === undefined) {
1882
+ throw new KubernetesWorkspaceSuspendedError(operation, workspaceId, name);
1883
+ }
1884
+ return transport;
1885
+ };
1886
+ /**
1887
+ * Lifecycle transitions run one at a time. Two of them racing would
1888
+ * interleave a suspend patch with a resume's readiness poll and settle on
1889
+ * whichever finished last, which is how a workspace ends up marked running
1890
+ * with no pod.
1891
+ */
1892
+ let queue = Promise.resolve();
1893
+ const serialise = (run) => {
1894
+ const next = queue.then(run, run);
1895
+ queue = next.then(() => undefined, () => undefined);
1896
+ return next;
1897
+ };
1898
+ /**
1899
+ * The suspend and the delete currently in flight, so a second caller
1900
+ * AWAITS the one that is running rather than queueing another behind it.
1901
+ *
1902
+ * Serialising is not the same thing and does not cover this. Now that a
1903
+ * terminal state is committed only after its request lands, two
1904
+ * concurrent `destroy({ deleteDisk: true })` calls would BOTH be admitted
1905
+ * under the queue alone — the second finding nothing yet marked and
1906
+ * sending a second DELETE — and two concurrent `suspend()` calls would
1907
+ * patch and wait twice over. So the single-flight check sits OUTSIDE the
1908
+ * queue, where a caller arriving mid-transition can still see it.
1909
+ *
1910
+ * The first caller's `signal` is the one the shared request runs under;
1911
+ * a second caller's is not consulted, which is what sharing means.
1912
+ */
1913
+ let pendingSuspend;
1914
+ /**
1915
+ * Whether the suspend in flight is quiescing the guest first.
1916
+ *
1917
+ * Kept beside the promise because it is the ONE part of a caller's
1918
+ * request that a joining caller cannot inherit. `signal` and `epoch` are
1919
+ * authority and lifetime — the first caller's to give, and a second
1920
+ * caller joining a transition under them is what sharing means. A
1921
+ * `quiesce` is a promise about the guest, and a suspend already patching
1922
+ * over processes nobody stopped cannot keep it retroactively.
1923
+ */
1924
+ let pendingSuspendQuiesces = false;
1925
+ /**
1926
+ * Whether the suspend in flight is flushing the guest's writes first.
1927
+ *
1928
+ * Beside {@link pendingSuspendQuiesces} and for its reason: a flush is a
1929
+ * promise about the DISK, and a suspend already patching cannot keep it
1930
+ * afterwards — once its state leaves `running` no call is admitted to
1931
+ * ask the guest for anything. The flag reads `true` for almost every
1932
+ * flight, because a flush is what `suspend()` does unless a caller turns
1933
+ * it off; it is `false` only behind a deliberate `flush: false`.
1934
+ */
1935
+ let pendingSuspendFlushes = false;
1936
+ let pendingDelete;
1937
+ const deleteSandbox = async (signal, epoch, gate) =>
1938
+ // Already gone is the state DELETE was asking for, fenced or not.
1939
+ await deleteSandboxObject(target, 'destroy', epoch, signal, gate);
1940
+ /**
1941
+ * The last Sandbox object a readiness poll read, kept for the two facts
1942
+ * {@link bindingFromSandbox} does not carry: `metadata.uid` and the
1943
+ * `volumeClaimTemplates` entry names.
1944
+ *
1945
+ * Kept rather than re-read. Every bind already GETs this object, both
1946
+ * fields were in the reply and were being discarded, and asking for them
1947
+ * again would be a second GET of a thing already in hand — and a second
1948
+ * answer that could disagree with the one the bind acted on.
1949
+ */
1950
+ let lastSandbox;
1951
+ const readBinding = async (pollSignal) => {
1952
+ const sandbox = await client.request('GET', sandboxPath(namespace, name), undefined, pollSignal);
1953
+ lastSandbox = sandbox;
1954
+ return bindingFromSandbox(sandbox);
1955
+ };
1956
+ /**
1957
+ * Read the uid of each `volumeClaimTemplates` entry's PVC, once per
1958
+ * handle, and never fail the workspace over it.
1959
+ *
1960
+ * The disk is the thing a workspace IS, and a host whose records say "id
1961
+ * X holds a month of work" needs to be able to tell that disk from a
1962
+ * different disk behind the same name. `sandboxUid` already catches the
1963
+ * ordinary case (delete and recreate takes the PVCs with it, because the
1964
+ * Sandbox owns them); this is what makes the claim checkable
1965
+ * independently, and what a host compares across processes.
1966
+ *
1967
+ * Best-effort ON PURPOSE. It needs `get` on `persistentvolumeclaims`,
1968
+ * which this release adds to the shipped Role and which no Role from an
1969
+ * earlier release has. A 403 here must cost a deployment nothing but this
1970
+ * one report — refusing to open a workspace because an OPTIONAL identity
1971
+ * field could not be read would turn a documentation change into an
1972
+ * outage.
1973
+ */
1974
+ const readVolumeClaimUids = async (signal) => {
1975
+ if (volumeClaimsRead)
1976
+ return;
1977
+ const entries = lastSandbox?.spec?.volumeClaimTemplates ?? [];
1978
+ const uids = {};
1979
+ for (const entry of entries) {
1980
+ const claimName = entry?.metadata?.name;
1981
+ if (typeof claimName !== 'string' || claimName === '')
1982
+ continue;
1983
+ try {
1984
+ const claim = await client.request('GET', persistentVolumeClaimPath(namespace, name, claimName), undefined, signal);
1985
+ const uid = claim?.metadata?.uid;
1986
+ if (typeof uid === 'string' && uid !== '')
1987
+ uids[claimName] = uid;
1988
+ }
1989
+ catch {
1990
+ // See above: an unreadable PVC leaves the entry out and
1991
+ // nothing else. `volumeClaimUids` says what could be read,
1992
+ // never what was guessed.
1993
+ }
1994
+ }
1995
+ // Both written only once the loop has finished, and in this order. The
1996
+ // per-PVC `catch` above covers the request and nothing else, so an
1997
+ // abort — or a claim name the path builder refuses — leaves through
1998
+ // here; latching the flag on the way IN would have left the uids
1999
+ // empty for the handle's life with no read left that could fill them.
2000
+ volumeClaimUids = uids;
2001
+ volumeClaimsRead = true;
2002
+ };
2003
+ /**
2004
+ * The budget ran out with no pod this handle was allowed to bind, worded
2005
+ * for the transition that ran out.
2006
+ *
2007
+ * Three different operator problems arrive here and one sentence cannot
2008
+ * serve them: a resume whose controller never replaced the pod, an adopt
2009
+ * that walked in on a pod still riding out its termination grace period,
2010
+ * and a create whose pod never came up at all. The pod the wait was
2011
+ * excluding is named whenever there is one, because "which pod is still
2012
+ * there" is the first thing anyone looks up next.
2013
+ *
2014
+ * A fourth arrives only under `agentAddress: 'pod-ip'`, and it takes
2015
+ * precedence over all of them: the bind DID find its pod, and what never
2016
+ * turned up was that pod's address. Reporting "no pod this handle could
2017
+ * bind to had appeared" about a pod the wait had already bound would send
2018
+ * an operator after the controller for something the CNI never finished.
2019
+ */
2020
+ const bindTimedOut = (policy, lastError, addresslessPodUid) => {
2021
+ const cause = lastError !== undefined ? { cause: lastError } : undefined;
2022
+ const subject = `kubernetes: workspace ${workspaceId} (Sandbox ${name})`;
2023
+ const budget = `${readiness.timeoutMs}ms`;
2024
+ if (addresslessPodUid !== undefined) {
2025
+ return new Error(`${subject} is configured with agentAddress: 'pod-ip', and ${budget} after it bound pod ${addresslessPodUid} that pod still reported no status.podIP, so there is no address to dial it at. A pod is given its IP when the CNI has finished attaching it, so a pod that never gets one has not been given a network at all. Raise readyTimeoutMs, look at the pod's events — or run with the default agentAddress: 'service' if this host is inside the cluster.`, cause);
2026
+ }
2027
+ if (policy.transition === 'resume' && policy.retiring !== undefined) {
2028
+ return new Error(`${subject} was patched back to operatingMode: Running, but ${budget} later the only pod behind it was still the one it was suspended from (uid ${policy.retiring}). A resumed pod keeps the sandbox's name and gets a new uid, and that uid is the agent's bind token, so binding to the old pod would present a token the new agent refuses. The Ready condition cannot be waited on instead — the controller leaves it standing across the transition. Raise readyTimeoutMs, or look at why the controller has not replaced the pod.`, cause);
2029
+ }
2030
+ if (policy.transition === 'adopt' && policy.retiring !== undefined) {
2031
+ return new Error(`${subject} was adopted while the previous pod (uid ${policy.retiring}) was still TERMINATING, and ${budget} later the controller had still not replaced it. A guest whose PID 1 ignores SIGTERM rides out its terminationGracePeriodSeconds before the replacement is created, so the adopt waits for the new pod under the same readiness budget as everything else on this path. ${NO_BINDABLE_POD_ADVICE}`, cause);
2032
+ }
2033
+ if (policy.transition === 'adopt' && policy.awaitReplacement) {
2034
+ return new Error(`${subject} was adopted from operatingMode: Suspended and patched back to Running, but ${budget} later no pod this handle could bind to had appeared — the pod it was suspended from is most likely still terminating. ${NO_BINDABLE_POD_ADVICE}`, cause);
2035
+ }
2036
+ const opened = policy.transition === 'create'
2037
+ ? 'was created'
2038
+ : policy.transition === 'adopt'
2039
+ ? 'was adopted while it was Running'
2040
+ : 'was patched back to operatingMode: Running';
2041
+ return new Error(`${subject} ${opened}, but ${budget} later no pod this handle could bind to had appeared. ${NO_BINDABLE_POD_ADVICE}`, cause);
2042
+ };
2043
+ /**
2044
+ * Wait until the Sandbox is Ready AND the pod behind it is a pod this
2045
+ * handle is allowed to bind to — and, under `agentAddress: 'pod-ip'`, one
2046
+ * that has an address — then read both facts off it.
2047
+ *
2048
+ * On create there is nothing to exclude and nothing to wait for, and this
2049
+ * is one poll plus one read, exactly as it was. Everywhere else the pod
2050
+ * behind the name is MOVING, and `policy` says how: `retiring` is the pod
2051
+ * this bind must see replaced, and `awaitReplacement` says whether a read
2052
+ * that finds no live pod at all is "not yet" — see {@link PodBindPolicy}.
2053
+ *
2054
+ * Excluding a pod is the whole point wherever one is named, because
2055
+ * `Ready` is not a transition signal. The controller leaves the condition
2056
+ * True across a resume — upstream says the same of `Suspended` in the
2057
+ * other direction — so the very first poll after the Running patch can
2058
+ * come back Ready while the only pod under that name is still the old
2059
+ * one, not yet deleted and not yet carrying a `deletionTimestamp`. Its
2060
+ * uid then reads as perfectly live, and the handle binds a token the new
2061
+ * agent will refuse, reported as a flat `unauthorized` with nothing
2062
+ * pointing at the race. So the uid is polled, under the SAME deadline as
2063
+ * everything else on this path, until it is a different pod's.
2064
+ *
2065
+ * An ADDRESS-less pod is a third kind of "not yet", and only under
2066
+ * `'pod-ip'`. The replacement pod is picked up while it is still
2067
+ * `Pending` — that is the point of excluding by uid rather than waiting
2068
+ * for Ready — and a `Pending` pod has no `status.podIP` until the CNI has
2069
+ * attached it. Every real pod passes through that window, so refusing
2070
+ * there would fail the ordinary resume in milliseconds with the whole
2071
+ * budget unspent. It is waited out here, on the same deadline, and the
2072
+ * timeout below names the pod that never got an address.
2073
+ *
2074
+ * The last failed read is carried onto the timeout as `cause`, so a wait
2075
+ * that never found a pod still says what it kept seeing.
2076
+ */
2077
+ const acquireBoundPod = async (deadline, policy) => {
2078
+ const label = `workspace ${workspaceId} (Sandbox ${name})`;
2079
+ const needsPodIP = options.agentAddress === 'pod-ip';
2080
+ let lastError;
2081
+ /** The last live-but-address-less pod seen, for the timeout's words. */
2082
+ let addresslessPodUid;
2083
+ /**
2084
+ * Whether the Sandbox has been seen Ready at least once.
2085
+ *
2086
+ * It decides who gets to report a budget that ran out inside the
2087
+ * readiness poll. Before the first Ready, "never became Ready" is the
2088
+ * true sentence and `pollForBinding`'s own error is the right one. After
2089
+ * it, the clock was being spent waiting for a POD, and blaming a
2090
+ * condition that has been True the whole time points the operator at
2091
+ * the one thing that is not the problem — see {@link bindTimedOut}.
2092
+ *
2093
+ * It rewrites nothing else: a readiness read that failed while the
2094
+ * budget still had time left failed on its own account.
2095
+ */
2096
+ let seenReady = false;
2097
+ while (deadline.remainingMs() > 0) {
2098
+ let binding;
2099
+ try {
2100
+ binding = await pollForBinding(readBinding, deadline, readiness, label);
2101
+ }
2102
+ catch (err) {
2103
+ // Only a clock that has actually run out is rewritten. A
2104
+ // readiness read that fails for any other reason — the API
2105
+ // server refused it, the caller aborted — is that call's own
2106
+ // failure and travels out unchanged. That matters most while a
2107
+ // `'pod-ip'` bind is waiting for an address: replacing a 5xx
2108
+ // with "the CNI never gave the pod an address" sends an operator
2109
+ // to the pod's events for a fault that was never the pod's, and
2110
+ // claims an elapsed time that never elapsed.
2111
+ //
2112
+ // The error is what says which happened, and only the error
2113
+ // can: an expiry does not arrive here as an
2114
+ // `OperationDeadlineExpired` — {@link pollForBinding} catches
2115
+ // that itself and reports it in its own words, as
2116
+ // {@link ReadinessPollTimeout} — and asking the clock instead
2117
+ // is wrong by a fraction of a millisecond, because the expiry
2118
+ // timer and `performance.now()` are different clocks and the
2119
+ // timer can fire first.
2120
+ if (!seenReady || !(err instanceof ReadinessPollTimeout))
2121
+ throw err;
2122
+ // Kept only if no pod read has failed yet: what the wait kept
2123
+ // seeing is more useful as the cause than the clock running out.
2124
+ lastError ??= err;
2125
+ break;
2126
+ }
2127
+ seenReady = true;
2128
+ let pod;
2129
+ try {
2130
+ pod = await deadline.run(async (tokenSignal) => await readBoundPod(client, namespace, binding, tokenSignal));
2131
+ }
2132
+ catch (err) {
2133
+ if (err instanceof OperationDeadlineExpired)
2134
+ break;
2135
+ if (!policy.awaitReplacement)
2136
+ throw err;
2137
+ lastError = err;
2138
+ pod = undefined;
2139
+ }
2140
+ // The IP travels WITH the uid, from this same read: a resumed pod
2141
+ // is a new pod at a new address, so a `'pod-ip'` session that
2142
+ // carried yesterday's IP would dial a pod that no longer exists.
2143
+ if (pod !== undefined && pod.uid !== policy.retiring) {
2144
+ if (!needsPodIP || pod.podIP !== undefined)
2145
+ return { binding, pod };
2146
+ addresslessPodUid = pod.uid;
2147
+ }
2148
+ try {
2149
+ await deadline.delay(readiness.pollIntervalMs);
2150
+ }
2151
+ catch (err) {
2152
+ if (err instanceof OperationDeadlineExpired)
2153
+ break;
2154
+ throw err;
2155
+ }
2156
+ }
2157
+ throw bindTimedOut(policy, lastError, addresslessPodUid);
2158
+ };
2159
+ /**
2160
+ * The ONE re-read behind every rebind: the routine the transport runs
2161
+ * when a call failed in a way that proves it ran nothing in the guest.
2162
+ *
2163
+ * It serves both triggers and both address modes, and generalising it is
2164
+ * the whole of this workstream's transport change. `'pod-ip'` needed it
2165
+ * first, for a dial that could not connect because the address died with
2166
+ * its pod. The DEFAULT `'service'` mode needs it for the opposite
2167
+ * symptom: the Service FQDN outlives the pod and resolves to the
2168
+ * replacement, so the dial succeeds and the new agent refuses the old
2169
+ * pod's uid — a flat `unauthorized` that used to be the end of the
2170
+ * handle.
2171
+ *
2172
+ * What it decides, in order, and why each answer is the only safe one:
2173
+ *
2174
+ * - **A different Sandbox uid, or no Sandbox at all** ⇒
2175
+ * {@link KubernetesWorkspaceReplacedError}, and never a rebind. The
2176
+ * name is deterministic, so this is a workspace somebody deleted and
2177
+ * created again: an EMPTY disk behind a name whose records say
2178
+ * otherwise. The transport rethrows this one rather than swallowing
2179
+ * it, because it is a verdict about the object and not a failed
2180
+ * diagnosis.
2181
+ * - **Suspended** ⇒ the current handle, unchanged. There is no pod to
2182
+ * bind and nothing here to say about it; the failing call's own
2183
+ * {@link admitted} re-read is what turns this into a
2184
+ * `KubernetesWorkspaceSuspendedError` naming the foreign suspend.
2185
+ * - **Running, same uid, a live pod** ⇒ that pod's address and token.
2186
+ * The transport installs it only if the token actually moved, so an
2187
+ * unchanged pod leaves the caller's original error standing — a pod
2188
+ * that is still there and still refusing is a guest problem, and
2189
+ * replacing that error with a later one would hide it.
2190
+ *
2191
+ * `generation` is what serialises this with the transitions WITHOUT
2192
+ * taking their queue — which it must not, because the privilege probe
2193
+ * runs inside a transition and a rebind that waited on the queue would
2194
+ * deadlock the resume holding it. Every transition bumps `sessionSeq`
2195
+ * before it changes the pod ({@link dropSession}, and `startSession`
2196
+ * itself), so a re-read that lands after a suspend or a resume no longer
2197
+ * matches and writes nothing: the transport may still swap the handle of
2198
+ * a session nothing is using, which costs nobody anything, and the
2199
+ * handle's own `podUid` — which `retiredPodUid` is stamped from — is
2200
+ * never written by a session that has been retired.
2201
+ */
2202
+ const refreshBoundPod = (binding, generation, opened) => {
2203
+ /**
2204
+ * What the transport is dialing right now, so "nothing changed" can
2205
+ * be said by handing back exactly that.
2206
+ *
2207
+ * It must be the CURRENT one and not the one this session opened
2208
+ * with: the transport rebinds whenever the token it is given differs
2209
+ * from the token it holds, so answering a later re-read with the
2210
+ * original address would drag a handle that has already followed a
2211
+ * replacement back onto the pod it left.
2212
+ */
2213
+ let dialing = opened;
2214
+ return async (signal) => {
2215
+ let sandbox;
2216
+ try {
2217
+ sandbox = await readSandboxObject(target, signal);
2218
+ }
2219
+ catch (err) {
2220
+ // A DELETE cascades to the disk, so a name with no object
2221
+ // behind it is not "not yet" — it is the end of this
2222
+ // workspace, and the one answer that must never be followed.
2223
+ if (err instanceof KubernetesAlreadyGoneError) {
2224
+ throw new KubernetesWorkspaceReplacedError(workspaceId, name, sandboxUid, undefined, {
2225
+ cause: err,
2226
+ });
2227
+ }
2228
+ throw err;
2229
+ }
2230
+ const standing = sandbox?.metadata?.uid;
2231
+ // Read before anything else, and refused before anything else: a
2232
+ // disk that is not this handle's disk is the one answer no retry
2233
+ // and no rebind may follow.
2234
+ if (sandbox === undefined || (sandboxUid !== undefined && standing !== sandboxUid)) {
2235
+ throw new KubernetesWorkspaceReplacedError(workspaceId, name, sandboxUid, standing);
2236
+ }
2237
+ // Somebody else suspended it. There is no pod, and saying so here
2238
+ // would be a worse error than the one `admitted` is about to
2239
+ // produce, which names the suspension and what to do about it.
2240
+ if (operatingModeOf(sandbox) === 'Suspended')
2241
+ return dialing;
2242
+ const refreshed = bindingFromSandbox(sandbox) ?? binding;
2243
+ const pod = await readBoundPod(client, namespace, refreshed, signal);
2244
+ const next = resolveAgentAddress(refreshed, options.agentPort, pod.uid, {
2245
+ mode: options.agentAddress,
2246
+ ...(pod.podIP !== undefined ? { podIP: pod.podIP } : {}),
2247
+ });
2248
+ dialing = next;
2249
+ if (generation === sessionSeq && pod.uid !== podUid) {
2250
+ // The pod this handle was bound to a moment ago, and the last
2251
+ // process heard from INSIDE it — read out of the variables
2252
+ // this routine is holding rather than assembled from the
2253
+ // handle's state. That is the whole difference on the one
2254
+ // path this announcement matters most: a pod replaced during
2255
+ // a RESUME is discovered by the privilege probe, which runs
2256
+ // while `state` is still `'suspended'`, and an identity built
2257
+ // from a state gate would announce a pod replacement while
2258
+ // naming neither pod.
2259
+ const previous = identityOn(podUid, guestBootId);
2260
+ podUid = next.token;
2261
+ podSession = sessionSeq;
2262
+ // A new pod is a new agent process, and the boot id it will
2263
+ // report is not the one the handle has been comparing
2264
+ // against. Clearing it is what stops the first reply from the
2265
+ // replacement being announced a second time as a container
2266
+ // restart — and it is why `current.guestBootId` on the event
2267
+ // below is `undefined`: the replacement has not answered yet,
2268
+ // and naming a process nobody has heard from would be a
2269
+ // guess. A listener reads `current.podUid` for what moved,
2270
+ // and `workspace.identity` once its own call has returned for
2271
+ // the process that answered from there.
2272
+ guestBootId = undefined;
2273
+ announceGuestRestart({
2274
+ reason: 'pod-replaced',
2275
+ previous,
2276
+ current: identityOn(next.token, undefined),
2277
+ });
2278
+ }
2279
+ return next;
2280
+ };
2281
+ };
2282
+ /**
2283
+ * Bring up one pod's worth of state: wait for a pod that is not the one
2284
+ * being replaced, read its bind token, re-resolve the address, build the
2285
+ * transport and prove the guest is deprivileged. Called on create and on
2286
+ * every resume, with nothing carried over between them.
2287
+ */
2288
+ const startSession = async (policy, signal) => {
2289
+ const deadline = new OperationDeadline(readiness.timeoutMs, `kubernetes workspace ${name} ${policy.transition}`, signal);
2290
+ // Both of these are re-read rather than remembered: a resumed pod keeps
2291
+ // the name and changes the uid and the IP, so a handle that reused
2292
+ // either would present a token the new agent refuses, at an address
2293
+ // whose pod is being deleted.
2294
+ const { binding, pod } = await acquireBoundPod(deadline, policy);
2295
+ const token = pod.uid;
2296
+ // From the readiness GET the bind just made, not from a read of its
2297
+ // own — see `lastSandbox`. Fixed for the handle's life: a later bind
2298
+ // that found a DIFFERENT object would have been refused by
2299
+ // `refreshBoundPod` long before it got here.
2300
+ sandboxUid ??= lastSandbox?.metadata?.uid;
2301
+ // Recorded before the probe, not after: a probe that refuses suspends
2302
+ // this pod, and the resume that follows has to know which pod it is
2303
+ // waiting to see replaced.
2304
+ podUid = token;
2305
+ // A new pod is a new agent process. Clearing it is what makes the
2306
+ // caller's own suspend/resume silent: the first reply from the new
2307
+ // guest establishes an identity rather than differing from the old
2308
+ // one's.
2309
+ guestBootId = undefined;
2310
+ sessionSeq += 1;
2311
+ // Written together with the counter, so that `podUid` belongs to THIS
2312
+ // session for as long as this session is the live one — which is what
2313
+ // {@link boundToPod} asks, and what makes a handle mid-resume report
2314
+ // the pod it has just bound rather than nothing at all.
2315
+ podSession = sessionSeq;
2316
+ const generation = sessionSeq;
2317
+ const address = resolveAgentAddress(binding, options.agentPort, token, {
2318
+ mode: options.agentAddress,
2319
+ ...(pod.podIP !== undefined ? { podIP: pod.podIP } : {}),
2320
+ });
2321
+ // A box rather than a `let`, so the callback below can name the handle
2322
+ // it belongs to before that handle exists. Nothing can call it in
2323
+ // between: `release` is reachable only THROUGH the handle.
2324
+ const own = {};
2325
+ const transport = new KubernetesAgentTransport(address, {
2326
+ // The backend opts in to the stream heartbeat; the transport
2327
+ // option it sets stays undefined for every other tier.
2328
+ heartbeatMs: options.streamHeartbeatMs,
2329
+ // Installed for BOTH address modes now. Under `'pod-ip'` it
2330
+ // follows an address that died with its pod; under the default
2331
+ // `'service'` mode the address is fine and the TOKEN is what
2332
+ // moved — see {@link refreshBoundPod}.
2333
+ refreshHandle: refreshBoundPod(binding, generation, address),
2334
+ // And the one fact no address and no token can carry: which
2335
+ // agent PROCESS answered. The two values beside it say whose
2336
+ // answer it is — this session's, and this session's CURRENT pod's
2337
+ // — because neither a retired session's transport nor the wire a
2338
+ // rebind left behind stops answering when it stops counting.
2339
+ onGuestReply: (reply, from) => observeGuestReply(generation, from, reply),
2340
+ });
2341
+ const inner = buildKubernetesSandbox({
2342
+ name,
2343
+ rootDir: options.rootDir,
2344
+ transport,
2345
+ // Deliberately NOT `deleteSandbox` — see {@link retireSession}. On
2346
+ // the task path `release` is a DELETE because the object is
2347
+ // disposable; here the same callback would erase the caller's disk
2348
+ // from a path nobody asked to erase anything.
2349
+ release: async (releaseSignal) => {
2350
+ await retireSession(own.handle, releaseSignal);
2351
+ },
2352
+ // And `release` is not reached from the unconfirmed-cancellation
2353
+ // path at all any more: this hook answers it instead, keeping the
2354
+ // pod — see {@link keepPodOnUnconfirmedCancellation}.
2355
+ onUnconfirmedCancellation: async (error) => await keepPodOnUnconfirmedCancellation(transport, error),
2356
+ // No `renew`, and so no lease loop: a workspace carries no expiry.
2357
+ });
2358
+ own.handle = inner;
2359
+ sessionTransports.set(inner, transport);
2360
+ // The same probe every task acquire runs, on every resume as well as on
2361
+ // create — a resumed pod is a new pod, from a possibly re-pulled image,
2362
+ // and "it was deprivileged last week" is not a check.
2363
+ await probeSandboxPrivileges(inner, name, resolveProbeTimeoutMs(readiness.timeoutMs), signal);
2364
+ // After the probe, so a workspace that is about to be refused never
2365
+ // spends a round trip per volume on an identity nobody will read; and
2366
+ // only once per handle, for the reason `readVolumeClaimUids` gives.
2367
+ await readVolumeClaimUids(signal);
2368
+ return inner;
2369
+ };
2370
+ /**
2371
+ * Drop the live session, and with it the right of anything still holding
2372
+ * that session's transport to report a pod.
2373
+ *
2374
+ * The two happen together or the guard in {@link refreshBoundPod} is a
2375
+ * lie: a retired session's transport can still be unwinding a call, and a
2376
+ * re-read that lands after the drop would write `podUid` for a session
2377
+ * nothing is using — including in the window before a suspend stamps
2378
+ * `retiredPodUid` from it.
2379
+ */
2380
+ const dropSession = () => {
2381
+ session = undefined;
2382
+ sessionSeq += 1;
2383
+ };
2384
+ /** Kill and await every terminal this handle returned. */
2385
+ const reapTerminals = async () => {
2386
+ const open = [...terminals];
2387
+ for (const terminal of open)
2388
+ releaseTerminal(terminal);
2389
+ await Promise.allSettled(open.map((terminal) => terminal.exited));
2390
+ terminals.clear();
2391
+ };
2392
+ /**
2393
+ * What an execution whose cancellation could not be CONFIRMED does to a
2394
+ * workspace: nothing to the cluster, and one bounded question to the
2395
+ * agent.
2396
+ *
2397
+ * `buildKubernetesSandbox` used to retire the sandbox here on its own
2398
+ * initiative — the shared controller's rule is that a command of unknown
2399
+ * state makes the pod unusable, and on a task sandbox retiring means
2400
+ * DELETING a disposable object with a scratch disk. On a workspace the
2401
+ * same decision reached {@link retireSession} and sent an
2402
+ * `operatingMode: Suspended` patch, which makes the controller delete the
2403
+ * pod. So eight seconds of network loss under one `exec()` — or a pod
2404
+ * evicted under an in-flight command, which can never confirm anything —
2405
+ * took every holder's terminals, dev servers and running commands away,
2406
+ * on a decision no caller issued and no host-side lock could prevent.
2407
+ *
2408
+ * The pod is therefore KEPT, and what the host is told instead is the one
2409
+ * thing the error cannot carry: whether the agent is still serving
2410
+ * (`ok`), has fenced itself (`retiring`), or could not be reached at all.
2411
+ * Today's `healthz()` boolean collapses the last two, which is why this
2412
+ * asks {@link KubernetesAgentTransport.agentHealth} instead.
2413
+ *
2414
+ * The `exec()` still rejects with `RemoteCancellationUnknownError`, now
2415
+ * carrying `accepted: false` with `reason: 'workspace-kept'` so a host
2416
+ * cannot read it as a patch that was attempted and failed. A host that
2417
+ * wants the old behaviour calls `suspend()` from the callback.
2418
+ */
2419
+ const keepPodOnUnconfirmedCancellation = async (transport, error) => {
2420
+ // The guest THIS command reserved against, kept per execution by the
2421
+ // transport. Never the handle's own pod and process: a restart the
2422
+ // handle survived an hour ago is not evidence about a command started
2423
+ // after it, and a pod another call has ALREADY rebound to is not the
2424
+ // pod this command's processes died with. Reading the handle would
2425
+ // answer `container-restarted` for every later unconfirmed
2426
+ // cancellation in the session and would name the replacement as the
2427
+ // guest that died.
2428
+ const reserved = guestWhenReserved(error);
2429
+ const previous = identityWhenReserved(reserved);
2430
+ const agent = await diagnoseAgent(transport);
2431
+ const { evidence, current } = await diagnoseGuest(previous, reserved?.guestBootId);
2432
+ // Recorded against the error itself, so the call that is about to
2433
+ // receive it can name the diagnosis rather than repeat the work —
2434
+ // see {@link admitted}. A WeakMap rather than a field on the error:
2435
+ // the class belongs to the shared controller and this is one
2436
+ // backend's evidence about it.
2437
+ if (evidence !== 'same-guest' && evidence !== 'unknown') {
2438
+ guestGoneEvidence.set(error, { evidence, previous, current });
2439
+ }
2440
+ try {
2441
+ options.onCancellationUnconfirmed?.({ error, agent, guest: evidence, previous, current });
2442
+ }
2443
+ catch {
2444
+ // A host's callback is not allowed to change the error the caller
2445
+ // is already receiving, and a throwing one must not become an
2446
+ // unaccepted retirement carrying somebody else's failure.
2447
+ }
2448
+ return { accepted: false, reason: WORKSPACE_KEPT_REASON };
2449
+ };
2450
+ /**
2451
+ * Was the guest the command was running in replaced while it ran?
2452
+ *
2453
+ * This is the DIAGNOSIS half of the unconfirmed-cancellation rule, and it
2454
+ * is deliberately only that. Nothing here patches, deletes or suspends
2455
+ * anything — a `Suspended` patch cannot stop a command whose pod is
2456
+ * already gone, and sending one would take the replacement pod away from
2457
+ * every other holder. What the host gets instead is the evidence:
2458
+ * `healthz` (beside this, in {@link diagnoseAgent}) says whether SOME
2459
+ * agent is serving at that address; this says whether it is the same one.
2460
+ *
2461
+ * The boot id is asked first because it is free — it rode in on replies
2462
+ * the handle already received, the `cancel-execution` refusal included —
2463
+ * and because it is the only evidence for the case no API read can see: a
2464
+ * container restarted in place keeps the pod, the uid and the token, and
2465
+ * changes nothing an API server would report. `reservedIn` is the guest
2466
+ * THIS command reserved against, not the one this session opened on: the
2467
+ * question is whether the command's own guest went away, so a restart the
2468
+ * handle survived before the command started is not evidence about it.
2469
+ *
2470
+ * Bounded by its own short deadline and run WITHOUT the caller's signal,
2471
+ * for the reason {@link diagnoseAgent} gives: the caller's signal is
2472
+ * quite possibly what started the cancellation.
2473
+ *
2474
+ * Every failure answers `unknown`, never a guess. This evidence decides
2475
+ * whether the caller is told its guest is gone, and saying so on the
2476
+ * strength of one 500 on a pod GET would be worse than saying nothing.
2477
+ */
2478
+ const diagnoseGuest = async (previous, reservedIn) => {
2479
+ // A handle that is not RUNNING is inside a transition it asked for:
2480
+ // the pod going away IS that transition, and a caller who typed
2481
+ // `suspend()` under a command of their own is not being told their
2482
+ // guest vanished. Only a handle that still believes it is running has
2483
+ // a question here — which is the case the whole diagnosis is for.
2484
+ if (state !== 'running')
2485
+ return { evidence: 'unknown', current: identityNow() };
2486
+ const bound = previous.podUid;
2487
+ if (bound === undefined)
2488
+ return { evidence: 'unknown', current: identityNow() };
2489
+ // The one piece of evidence that needs no API read: the handle has
2490
+ // heard from a DIFFERENT agent process than the one this command
2491
+ // reserved against. Computed here and CONSULTED below, after the pod
2492
+ // read, because it cannot tell a container restarted in place from a
2493
+ // pod another call rebound to — both leave the handle talking to a
2494
+ // process the command never reserved against, and only the pod read
2495
+ // says which happened. It is the answer when no read succeeds at all.
2496
+ const restartedInPlace = reservedIn !== undefined && guestBootId !== undefined && guestBootId !== reservedIn;
2497
+ try {
2498
+ return await new OperationDeadline(CANCELLATION_DIAGNOSIS_TIMEOUT_MS, `kubernetes workspace ${name} guest diagnosis`).run(async (deadlineSignal) => {
2499
+ const sandbox = await readSandboxObject(target, deadlineSignal);
2500
+ // No object, or one somebody replaced: not this workspace any
2501
+ // more, and not something to claim a verdict about here — the
2502
+ // next call's rebind refuses it by name.
2503
+ if (sandbox === undefined || sandbox.metadata?.uid !== previous.sandboxUid) {
2504
+ return { evidence: 'unknown', current: identityNow() };
2505
+ }
2506
+ // Somebody suspended it, so there is genuinely no pod — but
2507
+ // naming the mode is #473's job and not this one's. The
2508
+ // evidence is reported to the host's callback; the ERROR the
2509
+ // caller receives is decided by {@link admitted}, which asks
2510
+ // about a foreign suspend BEFORE it renames anything, so this
2511
+ // answer never pre-empts `KubernetesWorkspaceSuspendedError`.
2512
+ if (operatingModeOf(sandbox) === 'Suspended') {
2513
+ return { evidence: 'pod-gone', current: guestSeen(undefined) };
2514
+ }
2515
+ const binding = bindingFromSandbox(sandbox);
2516
+ if (binding === undefined) {
2517
+ return { evidence: 'pod-gone', current: guestSeen(undefined) };
2518
+ }
2519
+ const pod = await readBoundPod(client, namespace, binding, deadlineSignal);
2520
+ // A DIFFERENT pod outranks the boot id. Both say the command's
2521
+ // guest is gone; only this one says where it went, and calling
2522
+ // a replacement a restarted container would tell a host the
2523
+ // address and the token still work when neither does.
2524
+ if (pod.uid !== bound) {
2525
+ return { evidence: 'pod-replaced', current: guestSeen(pod.uid) };
2526
+ }
2527
+ if (restartedInPlace) {
2528
+ return { evidence: 'container-restarted', current: guestSeen(pod.uid) };
2529
+ }
2530
+ return { evidence: 'same-guest', current: guestSeen(pod.uid) };
2531
+ });
2532
+ }
2533
+ catch {
2534
+ // Including `readBoundPod`'s own refusal, which means "no live pod
2535
+ // AND no pod its selector matched" but is also what an API server
2536
+ // returning 500 produces. One error for two facts is not evidence
2537
+ // — but a boot id that already moved is evidence of its own, and
2538
+ // it was never the API server's to confirm.
2539
+ if (restartedInPlace)
2540
+ return { evidence: 'container-restarted', current: identityNow() };
2541
+ return { evidence: 'unknown', current: identityNow() };
2542
+ }
2543
+ };
2544
+ /**
2545
+ * One bounded `healthz` over a fresh connection, and never anything else.
2546
+ *
2547
+ * Bounded by its own short deadline rather than by the caller's: the
2548
+ * caller's signal is quite possibly what started the cancellation in the
2549
+ * first place, and a diagnosis that inherited it would report every
2550
+ * aborted call as an unreachable agent. Fresh, because every request on
2551
+ * this transport dials fresh — there is no cached socket to reuse and no
2552
+ * pooled connection whose state could answer for the pod.
2553
+ *
2554
+ * A failure is an ANSWER here, not an error to propagate: an agent that
2555
+ * cannot be reached is exactly the `unreachable` case, and this whole
2556
+ * routine exists to report rather than to decide.
2557
+ *
2558
+ * `unreachable` also covers a reply that is neither: the agent's
2559
+ * connection gate refuses a `healthz` that arrived over one
2560
+ * unauthenticated connection too many with a named `{ ok: false, error }`
2561
+ * and no fence flag. The probe got no answer about the agent's health
2562
+ * there, which is what `unreachable` means — reading it as `retiring`
2563
+ * would advise a `suspend()` on a healthy pod, and as `ok` would claim a
2564
+ * pod is serving on a reply that said the opposite.
2565
+ */
2566
+ const diagnoseAgent = async (transport) => {
2567
+ try {
2568
+ const health = await new OperationDeadline(CANCELLATION_DIAGNOSIS_TIMEOUT_MS, `kubernetes workspace ${name} agent diagnosis`).run(async (deadlineSignal) => await transport.agentHealth(deadlineSignal));
2569
+ return health.retiring ? 'retiring' : health.ok ? 'ok' : 'unreachable';
2570
+ }
2571
+ catch {
2572
+ return 'unreachable';
2573
+ }
2574
+ };
2575
+ /**
2576
+ * Retire one session's pod WITHOUT deleting anything.
2577
+ *
2578
+ * This is the inner handle's `release` on a workspace, and the difference
2579
+ * from the task path is the entire reason that callback is a parameter
2580
+ * rather than a DELETE both paths share: for a task sandbox retiring IS
2581
+ * deleting, because the object is disposable and its disk is scratch.
2582
+ * Here it is not — the DELETE cascades to the PVC, and nothing but a
2583
+ * caller naming a disk (`deleteDisk: true`, or
2584
+ * {@link deleteKubernetesWorkspace}) is allowed to remove one. That is the
2585
+ * invariant the whole file is built around.
2586
+ *
2587
+ * So the pod is retired the way `suspend()` retires one, with the same
2588
+ * `operatingMode: Suspended` patch, and the workspace is left
2589
+ * `suspending`: nothing is admitted, the disk is untouched, and `resume()`
2590
+ * brings up a fresh pod. A patch that FAILS is not swallowed — it travels
2591
+ * back out through the inner handle's `destroy()` on the error the caller
2592
+ * is already receiving.
2593
+ *
2594
+ * The path that used to reach this — an unconfirmed cancellation — no
2595
+ * longer does: see {@link keepPodOnUnconfirmedCancellation}. What is left
2596
+ * is the inner handle's own `destroy()`, and the workspace's delete drops
2597
+ * the session before it calls that, so in practice this is a guard rather
2598
+ * than a verb. It stays because the callback is required and because a
2599
+ * `release` that silently did nothing would be a worse answer than one
2600
+ * that does the safe thing if a future path ever reaches it.
2601
+ *
2602
+ * `retiring` is the handle the callback was built for. When it is not the
2603
+ * current session there is nothing to retire and this is a no-op: a resume
2604
+ * has already replaced it, or `deleteNow` dropped it on the way to a
2605
+ * DELETE — which must not be preceded by a suspend patch, and says so by
2606
+ * dropping it.
2607
+ *
2608
+ * It never touches the transition queue, and must not: it is awaited
2609
+ * inside a failing call, and that call can be the privilege probe of the
2610
+ * resume currently HOLDING the queue.
2611
+ */
2612
+ const retireSession = async (retiring, signal) => {
2613
+ if (retiring === undefined || retiring !== session)
2614
+ return;
2615
+ dropSession();
2616
+ // Some transition already owns this pod — the suspend that is patching
2617
+ // it away, or a create/resume cleanup — and commits its own state when
2618
+ // its own request settles. Dropping the session is all there is to do.
2619
+ if (state !== 'running')
2620
+ return;
2621
+ state = 'suspending';
2622
+ // Killed but not waited on: the frames go to a pod whose agent has
2623
+ // already stopped answering, and whether they are acknowledged must
2624
+ // not decide whether the retirement is reported accepted. Every
2625
+ // session dies with the pod either way; this is the ownership contract
2626
+ // being honoured, not a condition of the patch below. The `catch` is
2627
+ // not decoration — a detached chain that rejects with no handler takes
2628
+ // the host process down with it.
2629
+ void reapTerminals().catch(() => undefined);
2630
+ // Fenced by whatever this handle holds, like every other write it
2631
+ // sends on its own: a superseded handle's release must not suspend the
2632
+ // pod the new holder is using. The refusal travels out on the error
2633
+ // the caller is already receiving, exactly as a refused patch does.
2634
+ await writeOperatingMode(target, 'retire', 'Suspended', heldEpoch, signal);
2635
+ // The patch landed, so the controller is taking this pod away and the
2636
+ // next resume must see it replaced rather than bind it.
2637
+ retiredPodUid = podUid;
2638
+ };
2639
+ /**
2640
+ * The suspend, minus the queue — every caller here is already inside it.
2641
+ *
2642
+ * `suspending` is entered before the patch and `suspended` only after the
2643
+ * pod is observed stopped, so the two failures each leave the state that
2644
+ * is true: a patch the API server refused leaves the workspace exactly as
2645
+ * it was, still serving calls, and a wait that ran out leaves it refusing
2646
+ * them with the transition still unfinished. Neither can be returned from
2647
+ * by a later `suspend()` as though it had worked.
2648
+ */
2649
+ const suspendNow = async (signal, epoch, quiesceRequest, flushRequest) => {
2650
+ if (state === 'deleted')
2651
+ throw new KubernetesSandboxDestroyedError('suspend', name);
2652
+ // `suspending` deliberately falls through: the patch is re-sent and
2653
+ // the pod waited for again. Only a CONFIRMED suspend returns here.
2654
+ if (state === 'suspended')
2655
+ return;
2656
+ const before = state;
2657
+ // BEFORE the terminals are reaped, which is the whole point of doing
2658
+ // the read here rather than letting the patch below carry the refusal
2659
+ // on its own: `reapTerminals` SIGKILLs every session this handle
2660
+ // handed out, and a superseded holder that learned it was superseded
2661
+ // from the write would already have taken its own caller's terminals
2662
+ // away on a write that never applied. The reading is reused as the
2663
+ // patch's condition, so the gate costs no extra round trip.
2664
+ const gate = await readHolderEpochGate(target, 'suspend', epoch, signal);
2665
+ // A terminal owns an interactive process tree in a pod that is about
2666
+ // to be taken away, so it is stopped first — and stays stopped even if
2667
+ // the patch below fails. `suspend()` is a declaration that nobody is
2668
+ // using this workspace; killing the sessions that say otherwise is the
2669
+ // point of it rather than a cost of it.
2670
+ await reapTerminals();
2671
+ // And then, if the caller asked for it, everything else in the guest:
2672
+ // the terminals another handle opened, the commands already running,
2673
+ // and the programs that left every session this handle knows about.
2674
+ // HERE and nowhere later — `admit` refuses every call the moment the
2675
+ // state leaves `running` below, so a quiesce after that point could not
2676
+ // reach the pod it is quiescing. A quiesce that cannot be confirmed
2677
+ // throws from here, before `state` has moved and before any patch has
2678
+ // been sent: the pod is still running, this handle can still serve it,
2679
+ // and the caller is told which pid would not stop.
2680
+ //
2681
+ // A re-sent suspend (`state === 'suspending'`, the fall-through above)
2682
+ // does NOT re-quiesce: its patch has already landed, its session was
2683
+ // dropped with it, and there is no admitted call left to ask through.
2684
+ if (quiesceRequest !== undefined && quiesceRequest !== false && state === 'running') {
2685
+ await quiesceBeforeSuspend(quiesceRequest, signal);
2686
+ }
2687
+ // And then the disk, which is the thing this patch is about to take
2688
+ // away the only means of reading. HERE for the same reason the
2689
+ // quiesce is here — after the guest is quiet, so the flush covers
2690
+ // what the processes just stopped had written, and before `state`
2691
+ // moves, because nothing is admitted afterwards. A flush the guest
2692
+ // answered and could not confirm throws from here, with no patch
2693
+ // sent: the pod is still running and the caller can decide.
2694
+ //
2695
+ // A re-sent suspend does not re-flush, for the reason a re-sent
2696
+ // suspend does not re-quiesce: its patch has already landed and
2697
+ // there is no admitted call left to ask through.
2698
+ if (flushRequest !== false && state === 'running') {
2699
+ await flushBeforeSuspend(flushRequest ?? true, signal);
2700
+ }
2701
+ // From here on nothing new is admitted: the pod is going away, and a
2702
+ // call let through would dial an address that still resolves — the
2703
+ // Service outlives the pod — and hang on a connect timeout naming
2704
+ // nothing. This is NOT the terminal state; a patch that fails puts it
2705
+ // straight back.
2706
+ state = 'suspending';
2707
+ try {
2708
+ await writeOperatingMode(target, 'suspend', 'Suspended', epoch, signal, gate);
2709
+ }
2710
+ catch (err) {
2711
+ // Nothing was changed on the cluster, so nothing is changed here:
2712
+ // the pod is still running and this handle can still serve it.
2713
+ // Marking it suspended would be the defect — every later
2714
+ // suspend() and destroy() would return on that mark without ever
2715
+ // re-sending the patch, and the pod would run until somebody
2716
+ // noticed the bill.
2717
+ //
2718
+ // Unless the SESSION went away while this call was still short of
2719
+ // the patch, which is the one case where that premise is false.
2720
+ // The quiesce and the flush above are admitted calls, and an
2721
+ // admitted call that fails asks once whether somebody else has
2722
+ // suspended this workspace; when the answer is yes,
2723
+ // `noticeSuspendedElsewhere` records `suspending` and DROPS the
2724
+ // session, and `flushBeforeSuspend` reports that rather than
2725
+ // refusing — so the patch can be reached with no session behind
2726
+ // it, and then be refused by the epoch the other holder bumped.
2727
+ // Writing `running` back over that would leave a handle nothing
2728
+ // can use and nothing can recover: `admit` refuses every call
2729
+ // (there is no session), `status` reads 'destroyed' for the same
2730
+ // reason, `suspended` reads FALSE because only `state` decides it,
2731
+ // and `resume()` returns early on 'running' without starting one.
2732
+ // What `noticeSuspendedElsewhere` wrote is what is true, so it
2733
+ // stands.
2734
+ state = session === undefined ? 'suspending' : before;
2735
+ throw err;
2736
+ }
2737
+ // The patch landed, so this pod is the controller's to remove and the
2738
+ // next resume must see it replaced rather than bind it — whether or
2739
+ // not the wait below is still around when it goes.
2740
+ retiredPodUid = podUid;
2741
+ if (epoch !== undefined)
2742
+ heldEpoch = epoch;
2743
+ dropSession();
2744
+ await awaitPodRetired(client, namespace, name, workspaceId, readiness, signal);
2745
+ state = 'suspended';
2746
+ };
2747
+ /**
2748
+ * Wake the workspace if it is asleep, then bring a session up on it.
2749
+ *
2750
+ * The mode is READ before the patch rather than patched blind, for the
2751
+ * same reason {@link adoptExistingWorkspace} reads it: a workspace this
2752
+ * handle believes is suspended may already have been resumed by another
2753
+ * process, whose pod is serving its terminals right now. Patching Running
2754
+ * over Running would change nothing on the cluster but would make this
2755
+ * call the apparent author of a mode change it did not make — and a start
2756
+ * that then failed would suspend somebody else's live pod. It also keeps
2757
+ * {@link OPERATING_MODE_CHANGED_AT_ANNOTATION_KEY} honest: the annotation
2758
+ * says when the mode last CHANGED, and a no-op patch would restamp it.
2759
+ *
2760
+ * The extra GET is one round trip on a rare, explicit transition, which
2761
+ * is the same trade the adopt path already makes.
2762
+ */
2763
+ const resumeNow = async (signal, onStartFailure = defaultStartFailure, epoch, refreshPodTemplate = false) => {
2764
+ if (state === 'deleted')
2765
+ throw new KubernetesSandboxDestroyedError('resume', name);
2766
+ if (state === 'running') {
2767
+ // Resuming a running workspace sends nothing — unless it carries
2768
+ // an epoch, in which case the ONE thing it is asking for is the
2769
+ // thing that still has to happen: take this workspace over. That
2770
+ // is a write, even though the mode does not move, and a `resume()`
2771
+ // that silently dropped it would leave the new holder unfenced.
2772
+ if (epoch === undefined)
2773
+ return;
2774
+ await writeOperatingMode(target, 'resume', undefined, epoch, signal);
2775
+ heldEpoch = epoch;
2776
+ return;
2777
+ }
2778
+ // The pod a landed suspend patch took away is one this resume must see
2779
+ // replaced rather than bound — see `acquireBoundPod` and
2780
+ // `retiredPodUid`. Where there is none to exclude this is one poll
2781
+ // plus one read, exactly as it is on create.
2782
+ const replacing = retiredPodUid;
2783
+ // One GET, read twice: the mode decides whether a patch is sent at all
2784
+ // and the epoch decides whether it may be. Reading both off the same
2785
+ // object rather than off two GETs is what keeps an unfenced resume's
2786
+ // request log identical to what it always was.
2787
+ const current = await readSandboxObject(target, signal);
2788
+ let woke = operatingModeOf(current) === 'Suspended';
2789
+ const reading = readHolderEpoch(current?.metadata);
2790
+ if (epoch !== undefined)
2791
+ assertHolderEpochAllows(target, 'resume', epoch, reading);
2792
+ // Off the GET this path already makes, so an unfenced resume that asks
2793
+ // for no refresh sends and reads exactly what it always did: whatever
2794
+ // the object says it was built from, including a refresh another
2795
+ // process applied while this handle slept.
2796
+ templateRevision = readPodTemplateHash(current?.metadata);
2797
+ if (woke && refreshPodTemplate) {
2798
+ // The template is re-read HERE rather than remembered from the
2799
+ // open: the whole point of the option is to pick up an edit made
2800
+ // since, and a cached copy would refresh a workspace onto the
2801
+ // template as it stood when the handle was created.
2802
+ const template = await readSandboxTemplate(client, namespace, options.sandboxTemplateName, signal);
2803
+ const refresh = buildPodTemplateRefresh(template, options.sandboxTemplateName, options.runtimeClassName, options.podLabels);
2804
+ currentTemplateHash = refresh.hash;
2805
+ // Before any patch, and against the SANDBOX's own disk — both of
2806
+ // them, in the same order as the adopt path.
2807
+ const source = `SandboxTemplate ${options.sandboxTemplateName} in namespace ${namespace}, refreshing Sandbox ${name}`;
2808
+ assertRefreshedTemplateClaimsTheSameDisks(source, refresh.volumeClaimTemplates, current?.spec?.volumeClaimTemplates);
2809
+ assertBlockModeWorkspaceDisk(source, refresh.podTemplate, current?.spec?.volumeClaimTemplates);
2810
+ const result = await writeRefreshedPodTemplate(target, 'resume', refresh, epoch, reading, signal);
2811
+ if (result.applied) {
2812
+ templateRevision = refresh.hash;
2813
+ }
2814
+ else {
2815
+ // Another process resumed it first. Nothing was written, so
2816
+ // this call did not wake the workspace and a start that fails
2817
+ // must not put somebody else's live pod back to sleep.
2818
+ woke = false;
2819
+ templateRevision = readPodTemplateHash(result.sandbox.metadata);
2820
+ if (epoch !== undefined) {
2821
+ await writeOperatingMode(target, 'resume', undefined, epoch, signal, readHolderEpoch(result.sandbox.metadata));
2822
+ }
2823
+ }
2824
+ }
2825
+ else if (woke) {
2826
+ await writeOperatingMode(target, 'resume', 'Running', epoch, signal, reading);
2827
+ }
2828
+ else if (epoch !== undefined) {
2829
+ // Somebody else resumed it first, so there is no mode change to
2830
+ // make — but this call is still the one taking the workspace over,
2831
+ // and the stamp is what says so.
2832
+ await writeOperatingMode(target, 'resume', undefined, epoch, signal, reading);
2833
+ }
2834
+ if (epoch !== undefined)
2835
+ heldEpoch = epoch;
2836
+ session = await startSessionOrSuspend({
2837
+ transition: 'resume',
2838
+ awaitReplacement: replacing !== undefined,
2839
+ ...(replacing !== undefined ? { retiring: replacing } : {}),
2840
+ }, { woke, policy: onStartFailure, ...(heldEpoch !== undefined ? { epoch: heldEpoch } : {}) }, signal);
2841
+ state = 'running';
2842
+ // Bound, probed and serving: the pod that was excluded is one no
2843
+ // answer can name any more, and the next suspend records its own.
2844
+ retiredPodUid = undefined;
2845
+ };
2846
+ /**
2847
+ * Suspend, sharing one transition with any caller already inside it.
2848
+ *
2849
+ * The single-flight slot is taken before the queue so that a second
2850
+ * `suspend()` — or the `destroy()` that is a suspend — awaits this one
2851
+ * instead of patching and waiting all over again once it finishes. It is
2852
+ * released inside the run, before the promise handed to callers settles,
2853
+ * so a caller that awaits and then suspends again gets a fresh attempt.
2854
+ *
2855
+ * Joining is only honest while the transition in flight does everything
2856
+ * the joining caller asked for. It carries the first caller's signal and
2857
+ * epoch, and that is what sharing means — but a caller that asked for a
2858
+ * `quiesce` is about to trust a capture, and a suspend already in flight
2859
+ * WITHOUT one is on its way to patching over a guest nothing stopped. A
2860
+ * quiesce cannot be added to it afterwards either: `admit` refuses every
2861
+ * call the moment that transition's state leaves `running`. So such a
2862
+ * caller is refused, by the same class an unconfirmable quiesce rejects
2863
+ * with and for the same reason — everything this feature cannot deliver,
2864
+ * it says out loud.
2865
+ */
2866
+ const suspendShared = (signal, epoch, quiesceRequest, flushRequest) => {
2867
+ const wantsQuiesce = quiesceRequest !== undefined && quiesceRequest !== false;
2868
+ const wantsFlush = flushRequest !== false;
2869
+ if (pendingSuspend !== undefined) {
2870
+ if (wantsFlush && !pendingSuspendFlushes) {
2871
+ return Promise.reject(new KubernetesFlushUnconfirmedError('suspend_already_in_flight', `kubernetes: workspace '${workspaceId}' is already suspending under a call that asked for no flush ('flush: false'), and a flush cannot be added to a transition in flight — once that transition's state leaves 'running', no call is admitted to ask the guest for anything. Nothing was sent on this call's behalf: await the suspend in flight and treat the disk as one whose last writes may not have reached it, or flush() before the suspend next time. Refused rather than joined, because a caller that let the flush default on is about to trust the disk.`));
2872
+ }
2873
+ if (wantsQuiesce && !pendingSuspendQuiesces) {
2874
+ return Promise.reject(new KubernetesQuiesceUnconfirmedError('suspend_already_in_flight', `kubernetes: workspace '${workspaceId}' is already suspending under a call that did not ask for a quiesce, and a quiesce cannot be added to a transition in flight — once that transition's state leaves 'running', no call is admitted to stop anything in the guest. Nothing was sent and nothing was stopped on this call's behalf: await the suspend in flight and treat the disk as one that was written to, or call quiesce() before the suspend next time. Refused rather than joined, because a caller that passed 'quiesce' is about to trust a capture.`));
2875
+ }
2876
+ return pendingSuspend;
2877
+ }
2878
+ pendingSuspendQuiesces = wantsQuiesce;
2879
+ pendingSuspendFlushes = wantsFlush;
2880
+ pendingSuspend = serialise(async () => {
2881
+ try {
2882
+ await suspendNow(signal, epoch, quiesceRequest, flushRequest);
2883
+ }
2884
+ finally {
2885
+ pendingSuspend = undefined;
2886
+ pendingSuspendQuiesces = false;
2887
+ pendingSuspendFlushes = false;
2888
+ }
2889
+ });
2890
+ return pendingSuspend;
2891
+ };
2892
+ /**
2893
+ * `destroy()` in its default shape: the suspend above, plus the one thing
2894
+ * a destroy owes a caller that a suspend does not — idempotence over a
2895
+ * workspace somebody already deleted.
2896
+ *
2897
+ * `suspendNow` refuses a deleted workspace, and should: asking to suspend
2898
+ * an object that no longer exists is a mistake worth hearing about. But
2899
+ * `destroy()` is the verb a `finally` block calls, and a body that ends
2900
+ * with an explicit `destroy({ deleteDisk: true })` inside such a block
2901
+ * must not then be handed a `KubernetesSandboxDestroyedError` naming an
2902
+ * operation the caller never typed. What a plain `destroy()` asks for has
2903
+ * happened, more thoroughly than it asked.
2904
+ *
2905
+ * The state is read on both sides of the flight on purpose. Before,
2906
+ * for the ordinary sequential case; after, because the delete can land
2907
+ * while this call waits its turn — a `destroy()` racing a `destroy({
2908
+ * deleteDisk: true })` the queue admitted first must be a no-op in that
2909
+ * order too, which is the order it is most likely to be written in.
2910
+ */
2911
+ const destroyBySuspending = async (signal, epoch, quiesceRequest, flushRequest) => {
2912
+ // Read through a call on both sides. `state` is assigned from other
2913
+ // closures, which the checker cannot see, so it takes the first
2914
+ // comparison as narrowing the second out of existence — and the second
2915
+ // is the one that matters, because it is the one reading a delete that
2916
+ // landed while this call was queued.
2917
+ const gone = () => state === 'deleted';
2918
+ if (gone())
2919
+ return;
2920
+ try {
2921
+ await suspendShared(signal, epoch, quiesceRequest, flushRequest);
2922
+ }
2923
+ catch (err) {
2924
+ if (gone() && err instanceof KubernetesSandboxDestroyedError)
2925
+ return;
2926
+ throw err;
2927
+ }
2928
+ };
2929
+ /**
2930
+ * DELETE the Sandbox, and with it the Pod, the Service and the PVC.
2931
+ *
2932
+ * The terminal state is committed only once the DELETE has resolved —
2933
+ * an object already gone counts, that being the state DELETE was asking
2934
+ * for. A DELETE that FAILS leaves the state alone and rethrows, so the
2935
+ * caller can retry and the next attempt sends the request again. The
2936
+ * inverse — marking `deleted` first — is how an object outlives every
2937
+ * handle that could have removed it: the failure is thrown once, and every
2938
+ * later `destroy()` resolves immediately on a state nothing established.
2939
+ */
2940
+ const deleteNow = async (destroyOptions) => {
2941
+ if (state === 'deleted')
2942
+ return;
2943
+ const epoch = destroyOptions?.epoch ?? heldEpoch;
2944
+ // Before the terminals are reaped and before the session is torn
2945
+ // down, for the same reason `suspendNow` gates there: a refused
2946
+ // destroy must leave this handle exactly as it found it. The reading
2947
+ // is carried into the DELETE as its `preconditions.resourceVersion`,
2948
+ // so the gate costs no extra round trip.
2949
+ //
2950
+ // An object that is already gone is not a refusal: already gone is the
2951
+ // state DELETE was asking for, and an unfenced destroy has always
2952
+ // resolved on it. Reading the epoch must not turn that into a
2953
+ // rejection — the fence exists to stop a write, and there is no write
2954
+ // left to stop.
2955
+ let gate;
2956
+ try {
2957
+ gate = await readHolderEpochGate(target, 'destroy', epoch, destroyOptions?.signal);
2958
+ }
2959
+ catch (err) {
2960
+ if (!(err instanceof KubernetesAlreadyGoneError))
2961
+ throw err;
2962
+ state = 'deleted';
2963
+ dropSession();
2964
+ // Killed but not awaited, as every other path that learns the pod
2965
+ // is gone does it: `exited` on a session whose pod no longer
2966
+ // exists resolves only when TCP notices. The `catch` is not
2967
+ // decoration — a detached chain that rejects with no handler takes
2968
+ // the host process down.
2969
+ void reapTerminals().catch(() => undefined);
2970
+ return;
2971
+ }
2972
+ await reapTerminals();
2973
+ const current = session;
2974
+ // Dropped BEFORE the handle is torn down, because tearing it down runs
2975
+ // its `release` — which on a workspace is {@link retireSession}, a
2976
+ // SUSPEND patch. A delete does not want one on the way: the object is
2977
+ // going away whole. `retireSession` reads exactly this to know it.
2978
+ dropSession();
2979
+ // And the state moves with it. The pod is being taken away however the
2980
+ // DELETE goes, and the handle that served it is now torn down, so a
2981
+ // DELETE that fails leaves a workspace that admits nothing, says so,
2982
+ // and can be resumed or deleted again — rather than one still calling
2983
+ // itself `running` with no session behind it, which nothing but
2984
+ // another delete could ever get out of.
2985
+ if (state === 'running')
2986
+ state = 'suspending';
2987
+ // Through the inner handle when there is one, so its own terminal
2988
+ // reaping and lifecycle bookkeeping run. The DELETE itself is sent
2989
+ // here either way, exactly once, and it is retryable: the terminal
2990
+ // state below is committed only once it resolves.
2991
+ if (current)
2992
+ await current.destroy(destroyOptions);
2993
+ await deleteSandbox(destroyOptions?.signal, epoch, gate);
2994
+ if (epoch !== undefined)
2995
+ heldEpoch = epoch;
2996
+ state = 'deleted';
2997
+ };
2998
+ /** Delete, sharing one DELETE with any caller already inside it. */
2999
+ const deleteShared = (destroyOptions) => {
3000
+ pendingDelete ??= serialise(async () => {
3001
+ try {
3002
+ await deleteNow(destroyOptions);
3003
+ }
3004
+ finally {
3005
+ pendingDelete = undefined;
3006
+ }
3007
+ });
3008
+ return pendingDelete;
3009
+ };
3010
+ /**
3011
+ * Bring a session up, and put the workspace back to sleep ONLY if this
3012
+ * call is the one that woke it.
3013
+ *
3014
+ * The failures that reach here are not failures OF the workspace. A
3015
+ * caller's signal aborting during readiness, one 5xx or 429 on a Sandbox
3016
+ * or pod GET (the client does not retry), a privilege probe that overran
3017
+ * its own deadline — every one of them can happen to a second process
3018
+ * adopting a workspace the first process is happily using, and the patch
3019
+ * that used to go out on all of them makes the controller delete that
3020
+ * pod. So the question asked here is not "did the start fail" but "did
3021
+ * THIS call move `operatingMode`", which is what `woke` carries:
3022
+ *
3023
+ * - a create POSTed the object, so the pod exists because of this call;
3024
+ * - an adopt or a `resume()` that found the object `Suspended` sent the
3025
+ * Running patch that asked for the pod;
3026
+ * - an adopt of an object that was already Running, or a `resume()` that
3027
+ * found somebody had already resumed it, moved nothing and so has
3028
+ * nothing to put back.
3029
+ *
3030
+ * The first two keep suspending, and must: a workspace this call woke and
3031
+ * then failed to start is left Running with a pod nobody is using,
3032
+ * burning a node until somebody notices. The third sends nothing and
3033
+ * rethrows, which is the whole point of the rule.
3034
+ *
3035
+ * Suspend rather than delete, always, wherever it does patch: the failure
3036
+ * might be a probe refusal on a workspace whose disk holds a month of a
3037
+ * caller's work, and no failure path in this module is allowed to make
3038
+ * that decision. The cost of being wrong the other way is one suspended
3039
+ * Sandbox left standing, which the caller finds again under the same
3040
+ * deterministic name.
3041
+ *
3042
+ * The read that decided `woke` and the patch below are deliberately NOT
3043
+ * one conditional write. Between them another process can change the
3044
+ * object, and this call would then suspend a mode change it did not make
3045
+ * after all — a narrow window, and the honest way to close it is a
3046
+ * condition ON the write rather than a second read here. Until there is
3047
+ * one, a patch that was not refused is treated as this call's own. This
3048
+ * comment is the single place a conditional write has to tighten.
3049
+ */
3050
+ const startSessionOrSuspend = async (policy, wake, signal) => {
3051
+ try {
3052
+ return await startSession(policy, signal);
3053
+ }
3054
+ catch (err) {
3055
+ // `suspending`, not `suspended`, and set whether or not a patch
3056
+ // goes out below. This handle has no session either way, so it can
3057
+ // serve nothing and `resume()` is the way back — and where a patch
3058
+ // IS sent, `runFailureCleanup` swallows its own failures so that
3059
+ // the primary error stays primary, which means the patch may not
3060
+ // have landed and the pod may still be running. Only a CONFIRMED
3061
+ // suspend is ever recorded as `suspended`.
3062
+ state = 'suspending';
3063
+ dropSession();
3064
+ if (!wake.woke || wake.policy === 'leave')
3065
+ throw err;
3066
+ let retired = false;
3067
+ await runFailureCleanup(async (cleanupSignal) => {
3068
+ // Fenced by the epoch this transition carried, which closes
3069
+ // the window the comment above names: between the read that
3070
+ // decided `woke` and this patch, another holder can take the
3071
+ // workspace, and an unconditional suspend here would stop the
3072
+ // pod that holder is now using. A refusal is swallowed by
3073
+ // `runFailureCleanup` like every other cleanup failure, and
3074
+ // the primary error is still what the caller receives.
3075
+ await writeOperatingMode(target, 'start-cleanup', 'Suspended', wake.epoch, cleanupSignal);
3076
+ retired = true;
3077
+ });
3078
+ // Only when the patch came back. A resume that got as far as
3079
+ // binding a new pod and then failed its probe has retired THAT
3080
+ // pod, and the resume after it must wait for the replacement
3081
+ // rather than bind the one this cleanup took away. A cleanup whose
3082
+ // patch never landed retired nothing and has nothing to exclude.
3083
+ if (retired)
3084
+ retiredPodUid = podUid;
3085
+ throw err;
3086
+ }
3087
+ };
3088
+ const admit = (operation) => {
3089
+ if (state === 'deleted')
3090
+ throw new KubernetesSandboxDestroyedError(operation, name);
3091
+ if (state !== 'running' || session === undefined) {
3092
+ throw new KubernetesWorkspaceSuspendedError(operation, workspaceId, name);
3093
+ }
3094
+ return session;
3095
+ };
3096
+ /**
3097
+ * Adopt a suspension this handle did not perform, if the object really is
3098
+ * suspended. Answers whether it was.
3099
+ *
3100
+ * Recorded as `suspending` rather than `suspended`, and the distinction is
3101
+ * the same one the whole module turns on: what was observed is the
3102
+ * object's MODE, not the pod stopping. The other process's wait may still
3103
+ * be running, or may have run out. A `suspended` mark here would let this
3104
+ * handle's next `suspend()` return on it and promise a quiesced disk
3105
+ * nobody in this process ever waited for.
3106
+ *
3107
+ * `retiredPodUid` is set for the same reason `suspendNow` sets it: a
3108
+ * suspend patch has landed — somebody else's — so the controller is taking
3109
+ * this pod away, and the resume that follows must see it REPLACED rather
3110
+ * than bind the uid the new agent will refuse.
3111
+ *
3112
+ * It never touches the transition queue, and must not: the call-failure
3113
+ * path below runs inside a rejecting `exec()`, and that exec can be the
3114
+ * privilege probe of a resume that is currently HOLDING the queue. The
3115
+ * `state !== 'running'` guard is what keeps it out of a transition's way —
3116
+ * every transition leaves `running` before it does anything.
3117
+ */
3118
+ const noticeSuspendedElsewhere = async (signal) => {
3119
+ if (state !== 'running')
3120
+ return false;
3121
+ if ((await readOperatingMode(client, namespace, name, signal)) !== 'Suspended')
3122
+ return false;
3123
+ if (state !== 'running')
3124
+ return false;
3125
+ state = 'suspending';
3126
+ dropSession();
3127
+ retiredPodUid = podUid;
3128
+ // Killed but not awaited, exactly as `retireSession` does it and for
3129
+ // the same reason: the frames go to a pod that is being deleted, and
3130
+ // `exited` would otherwise resolve only when TCP notices. The `catch`
3131
+ // is not decoration — a detached chain that rejects with no handler
3132
+ // takes the host process down.
3133
+ void reapTerminals().catch(() => undefined);
3134
+ return true;
3135
+ };
3136
+ /**
3137
+ * Run one admitted call and, if it fails, ask ONCE whether the workspace
3138
+ * has been suspended out from under this handle.
3139
+ *
3140
+ * The failure a foreign suspend produces is not recognisable on its own.
3141
+ * The pod is gone, so the dial is refused — against an address that still
3142
+ * resolves, because the Service outlives the pod — or, if a replacement
3143
+ * pod is already up, the guest answers a flat `unauthorized` because this
3144
+ * handle is presenting the retired pod's uid. Neither says "suspended",
3145
+ * and a caller looking at either has no reason to try `resume()`.
3146
+ *
3147
+ * So the object is re-read, and only then: once per failed call, never
3148
+ * speculatively, and never on a call that succeeded. The re-read is a
3149
+ * diagnostic and behaves like one — it runs without the caller's signal
3150
+ * (which is quite possibly what aborted the call in the first place), and
3151
+ * a re-read that itself fails hands back the caller's own error rather
3152
+ * than replacing it with a second one about the API server.
3153
+ *
3154
+ * That question is asked BEFORE the unconfirmed-cancellation diagnosis is
3155
+ * turned into an error, and the order is load-bearing. A foreign suspend
3156
+ * takes the pod away, so it looks exactly like a gone guest and produces
3157
+ * that diagnosis too — but only one of the two answers lets the caller
3158
+ * recover: `KubernetesWorkspaceSuspendedError` says what happened and
3159
+ * leaves the handle suspended, so `resume()` brings a pod back. The
3160
+ * diagnosis is the answer for a workspace that is still Running.
3161
+ */
3162
+ const admitted = async (operation, run) => {
3163
+ const current = admit(operation);
3164
+ try {
3165
+ return await run(current);
3166
+ }
3167
+ catch (err) {
3168
+ // The diagnosis, if one was taken — read here and ACTED ON last.
3169
+ // A foreign suspend is also a gone guest, and it is the answer
3170
+ // that outranks: #473 promises a caller that somebody else
3171
+ // suspended the workspace hears it by name, and that the handle
3172
+ // adopts the suspension so `resume()` works. Renaming first would
3173
+ // leave the handle marked running with no pod, `suspended` false
3174
+ // and `resume()` a silent no-op. So the suspend question is asked
3175
+ // first, and this is the answer when it comes back `false`.
3176
+ const diagnosis = err instanceof Error ? guestGoneEvidence.get(err) : undefined;
3177
+ if (diagnosis !== undefined && err instanceof Error)
3178
+ guestGoneEvidence.delete(err);
3179
+ if (state === 'running') {
3180
+ let suspended = false;
3181
+ try {
3182
+ suspended = await noticeSuspendedElsewhere();
3183
+ }
3184
+ catch {
3185
+ // A re-read that itself fails is not a better error than
3186
+ // the one the caller is already holding: fall through and
3187
+ // give them that, named if a diagnosis was taken.
3188
+ suspended = false;
3189
+ }
3190
+ if (suspended) {
3191
+ throw new KubernetesWorkspaceSuspendedError(operation, workspaceId, name, 'transport', {
3192
+ cause: err,
3193
+ });
3194
+ }
3195
+ }
3196
+ // An unconfirmed cancellation whose guest is demonstrably gone
3197
+ // leaves as an error that SAYS so and carries both identities. It
3198
+ // is a subclass of the error it replaces, so nothing that catches
3199
+ // the base class stops catching it, and the rule is unchanged —
3200
+ // the outcome is still unknown and still must not be retried.
3201
+ // Nothing was patched to get here and nothing is patched on the
3202
+ // way out.
3203
+ if (diagnosis !== undefined) {
3204
+ const named = new KubernetesWorkspaceGuestGoneError(workspaceId, name, diagnosis.evidence, diagnosis.previous, diagnosis.current, { cause: err });
3205
+ if (err instanceof RemoteCancellationUnknownError)
3206
+ named.retirement = err.retirement;
3207
+ throw named;
3208
+ }
3209
+ throw err;
3210
+ }
3211
+ };
3212
+ /**
3213
+ * {@link admitted}, for a call that hands back an ITERABLE rather than a
3214
+ * promise.
3215
+ *
3216
+ * Admission is taken HERE, synchronously, and only the iteration lives in
3217
+ * the generator below: an async generator's body does not run until
3218
+ * something pulls from it, and a suspended workspace has to refuse
3219
+ * `readFileStream(...)` where the caller wrote it — the same place, and
3220
+ * with the same error, as every other data-plane call. The task sandbox's
3221
+ * own `readFileStream` checks admissibility at the call for the same
3222
+ * reason.
3223
+ *
3224
+ * The one diagnostic re-read is otherwise identical, and it covers the
3225
+ * whole stream rather than its first pull. A workspace suspended halfway
3226
+ * through a long read takes its pod's connection with it, and what
3227
+ * surfaces is a socket that closed early: as unrecognisable on its own as
3228
+ * the failures `admitted` exists to name.
3229
+ *
3230
+ * `yield*` rather than a hand-rolled loop so that a consumer's `break`
3231
+ * still reaches the transport's generator, whose `finally` is what
3232
+ * destroys the socket and makes the guest release the file descriptor.
3233
+ */
3234
+ const admittedStream = (operation, open) => {
3235
+ const source = open(admit(operation));
3236
+ return (async function* stream() {
3237
+ try {
3238
+ yield* source;
3239
+ }
3240
+ catch (err) {
3241
+ if (state !== 'running')
3242
+ throw err;
3243
+ let suspended = false;
3244
+ try {
3245
+ suspended = await noticeSuspendedElsewhere();
3246
+ }
3247
+ catch {
3248
+ throw err;
3249
+ }
3250
+ if (!suspended)
3251
+ throw err;
3252
+ throw new KubernetesWorkspaceSuspendedError(operation, workspaceId, name, 'transport', {
3253
+ cause: err,
3254
+ });
3255
+ }
3256
+ })();
3257
+ };
3258
+ /**
3259
+ * One quiesce, through the live session's transport.
3260
+ *
3261
+ * It goes through {@link admitted} like every other data-plane call, so a
3262
+ * workspace another process suspended underneath this handle is named as
3263
+ * suspended rather than reported as a quiesce that failed at the
3264
+ * transport. It is NOT serialised here: both callers are already holding
3265
+ * the transition queue — the public verb takes it, and `suspendNow` runs
3266
+ * inside it.
3267
+ */
3268
+ const quiesceNow = async (quiesceOptions) => await admitted('quiesce', async (handle) => await admittedTransport('quiesce', handle).quiesce(quiesceOptions?.graceMs !== undefined ? { graceMs: quiesceOptions.graceMs } : {}, quiesceOptions?.signal));
3269
+ /**
3270
+ * Tell the host about a gap a suspend went ahead over.
3271
+ *
3272
+ * Synchronous, never awaited, and a callback that throws changes
3273
+ * nothing: these are reports about a transition that has already been
3274
+ * decided, and a host's diagnostic is not allowed to decide it again.
3275
+ */
3276
+ const tellHost = (callback, value) => {
3277
+ try {
3278
+ callback?.(value);
3279
+ }
3280
+ catch {
3281
+ // A host's callback is not allowed to decide whether the suspend
3282
+ // it was only being told about goes ahead.
3283
+ }
3284
+ };
3285
+ /**
3286
+ * One flush, through the live session's transport.
3287
+ *
3288
+ * Through {@link admitted} like every other data-plane call, so a
3289
+ * workspace another process suspended underneath this handle is named as
3290
+ * suspended rather than reported as a flush that failed at the
3291
+ * transport. Not serialised here: both callers already hold the
3292
+ * transition queue — the public verb takes it, and `suspendNow` runs
3293
+ * inside it.
3294
+ */
3295
+ const flushIfSupported = async (flushOptions) => await admitted('flush', async (handle) => {
3296
+ const transport = admittedTransport('flush', handle);
3297
+ // Asked INSIDE the admitted call and answered with `undefined`
3298
+ // rather than by throwing, because an unsupported feature is an
3299
+ // answer from the guest and not a failure of the connection —
3300
+ // and `admitted`'s failure path treats every error as one, which
3301
+ // means an extra GET of the Sandbox and, on a workspace somebody
3302
+ // else had already suspended, a `KubernetesWorkspaceSuspendedError`
3303
+ // wrapping a refusal that has nothing to do with it.
3304
+ if (!(await transport.supportsFlush(flushOptions?.signal)))
3305
+ return undefined;
3306
+ return await transport.flush(flushOptions?.timeoutMs !== undefined ? { timeoutMs: flushOptions.timeoutMs } : {}, flushOptions?.signal);
3307
+ });
3308
+ const flushNow = async (flushOptions) => {
3309
+ const report = await flushIfSupported(flushOptions);
3310
+ // The explicit verb refuses, where `suspend()` degrades: a caller
3311
+ // that asked for a flush by name is told its image cannot do one.
3312
+ if (report === undefined)
3313
+ throw flushUnsupportedError();
3314
+ return report;
3315
+ };
3316
+ /**
3317
+ * The flush `suspend()` performs by default, and the ONE failure that
3318
+ * stops the suspend.
3319
+ *
3320
+ * That one is a guest which ANSWERED and could not confirm: it is alive,
3321
+ * it tried, and its writes may never reach the device — so no patch goes
3322
+ * out, the pod keeps serving, and the caller holds the guest's own
3323
+ * message and can retry, read the data out, or suspend under
3324
+ * `{ flush: false }` deliberately.
3325
+ *
3326
+ * Everything else is REPORTED and the suspend goes ahead, because a
3327
+ * flush is on by DEFAULT and `suspend()` is the verb an operator reaches
3328
+ * for when a workspace has gone wrong. Two shapes reach here:
3329
+ *
3330
+ * - An image that cannot flush: one whose agent predates the op, or one
3331
+ * whose agent has the op and no `sync` to run it with and says so.
3332
+ * Refusing over it would mean this release could not suspend any
3333
+ * workspace built from such an image without the caller changing its
3334
+ * own code. {@link KubernetesWorkspaceOptions.onFlushUnsupported} is
3335
+ * told.
3336
+ * - A guest that could not be REACHED — a refused dial, a connect
3337
+ * timeout, a rejected token, an agent that has fenced itself, a
3338
+ * workspace somebody else has already suspended. None of those become
3339
+ * flushable by leaving the pod running, and refusing would take the
3340
+ * lifecycle verb away from precisely the crashed, OOM-killed or
3341
+ * fenced workspace it is needed for — which would leave it running,
3342
+ * and billing, behind a raw `ECONNREFUSED` naming neither the flush
3343
+ * nor a way past it. `suspend()` then `resume()` is also what
3344
+ * `KubernetesAgentRetiringError` tells a caller to do about a fence.
3345
+ * {@link KubernetesWorkspaceOptions.onFlushUnreachable} is told, with
3346
+ * the transport's own error as its `cause`.
3347
+ *
3348
+ * The caller's own abort is not a flush verdict and travels untouched: a
3349
+ * caller that withdrew its suspend gets the abort, not a suspend that
3350
+ * went ahead without the flush it asked for.
3351
+ */
3352
+ const flushBeforeSuspend = async (request, signal) => {
3353
+ const timeoutMs = typeof request === 'object' ? request.timeoutMs : undefined;
3354
+ let report;
3355
+ try {
3356
+ report = await flushIfSupported({
3357
+ ...(timeoutMs !== undefined ? { timeoutMs } : {}),
3358
+ ...(signal !== undefined ? { signal } : {}),
3359
+ });
3360
+ }
3361
+ catch (err) {
3362
+ if (err instanceof KubernetesFlushUnconfirmedError)
3363
+ throw err;
3364
+ if (signal?.aborted === true)
3365
+ throw err;
3366
+ // A guest that ANSWERED that it cannot flush — an advertised op
3367
+ // its image has no program to run — is the unsupported gap under
3368
+ // another name, not an unreachable guest, and the host hears it
3369
+ // on the callback that means that.
3370
+ if (err instanceof KubernetesFlushUnsupportedError) {
3371
+ tellHost(options.onFlushUnsupported, err);
3372
+ return;
3373
+ }
3374
+ tellHost(options.onFlushUnreachable, new KubernetesFlushUnreachableError(`kubernetes: workspace '${workspaceId}' could not be asked to flush before its pod stopped (${err instanceof Error ? err.message : String(err)}). The suspend went ahead rather than leaving a workspace nobody can reach running: a guest that answers nothing does not become flushable by keeping its pod, and this is the verb that replaces the pod. What is on the disk is what the guest kernel had already written back, plus whatever the pod's preStop hook and the agent's own SIGTERM handler manage while it stops. A caller that would rather stop should flush() first and pass 'flush: false' only once that has succeeded.`, { cause: err }));
3375
+ return;
3376
+ }
3377
+ if (report !== undefined)
3378
+ return;
3379
+ tellHost(options.onFlushUnsupported, flushUnsupportedError());
3380
+ };
3381
+ /**
3382
+ * The quiesce `suspend({ quiesce })` performs, and the ONE failure it
3383
+ * does not pass on.
3384
+ *
3385
+ * An image whose agent predates the op cannot be asked, and refusing to
3386
+ * suspend over that would make the option unusable against every pod
3387
+ * built before this release — a caller could not even suspend such a
3388
+ * workspace without changing its own code. So the suspend goes ahead, as
3389
+ * it always did, and the host is TOLD through
3390
+ * {@link KubernetesWorkspaceOptions.onQuiesceUnsupported}; the gap is
3391
+ * reported rather than either hidden or turned into a refusal.
3392
+ *
3393
+ * Every other failure travels: a guest that answered and could not
3394
+ * confirm has processes still writing to the disk, and a suspend that
3395
+ * patched anyway would take the pod away while they did.
3396
+ */
3397
+ const quiesceBeforeSuspend = async (request, signal) => {
3398
+ const graceMs = typeof request === 'object' ? request.graceMs : undefined;
3399
+ try {
3400
+ const report = await quiesceNow({
3401
+ ...(graceMs !== undefined ? { graceMs } : {}),
3402
+ ...(signal !== undefined ? { signal } : {}),
3403
+ });
3404
+ // This call answers `void`, so the scope in the report would go
3405
+ // nowhere — and a narrowed scan can miss exactly the process the
3406
+ // quiesce was asked for. Told, for the same reason the
3407
+ // unsupported gap is.
3408
+ if (report.scope !== 'pid-namespace') {
3409
+ try {
3410
+ options.onQuiesceNarrowed?.(report);
3411
+ }
3412
+ catch {
3413
+ // A host's callback is not allowed to decide whether the
3414
+ // suspend it was only being told about goes ahead.
3415
+ }
3416
+ }
3417
+ }
3418
+ catch (err) {
3419
+ if (!(err instanceof KubernetesQuiesceUnsupportedError))
3420
+ throw err;
3421
+ try {
3422
+ options.onQuiesceUnsupported?.(err);
3423
+ }
3424
+ catch {
3425
+ // A host's callback is not allowed to decide whether the suspend
3426
+ // it was only being told about goes ahead.
3427
+ }
3428
+ }
3429
+ };
3430
+ /**
3431
+ * How the FIRST bind is allowed to behave, read off how this handle came
3432
+ * by its object.
3433
+ *
3434
+ * A create POSTed the Sandbox itself: no pod existed a moment ago, no pod
3435
+ * is being replaced, and a read that finds none is a failure to report
3436
+ * rather than a state to wait out. An adopt is the opposite by default —
3437
+ * the pod is somebody else's, possibly on its way out — and the two
3438
+ * shapes that say so are the object standing `Suspended` (its pod has
3439
+ * been taken away, and the resume patch this adopt just sent is what asks
3440
+ * for the replacement) and a pod already carrying a `deletionTimestamp`
3441
+ * (the controller is taking it away now). Both of those leave a window
3442
+ * with no live pod under the name at all, which is exactly the window a
3443
+ * host restarting inside a previous pod's terminationGracePeriodSeconds
3444
+ * arrives in.
3445
+ *
3446
+ * An adopt of an object that was Running with a healthy pod keeps the
3447
+ * create path's behaviour: nothing is being replaced, so nothing is
3448
+ * waited for.
3449
+ */
3450
+ const initialBindPolicy = options.origin === 'created'
3451
+ ? { transition: 'create', awaitReplacement: false }
3452
+ : {
3453
+ transition: 'adopt',
3454
+ awaitReplacement: options.origin === 'resumed' ||
3455
+ options.awaitReplacement === true ||
3456
+ options.drainingPodUid !== undefined,
3457
+ ...(options.drainingPodUid !== undefined ? { retiring: options.drainingPodUid } : {}),
3458
+ };
3459
+ // The first session is brought up here so that `createKubernetesWorkspace`
3460
+ // resolves with a workspace that is Ready, addressed and probed — the same
3461
+ // contract `create()` gives a task sandbox.
3462
+ session = await startSessionOrSuspend(initialBindPolicy,
3463
+ // A create POSTed the object and an adopt that found it `Suspended`
3464
+ // patched it Running; an adopt of an object that was already Running
3465
+ // moved nothing, and a start that fails on it must leave the pod its
3466
+ // holder is using exactly where it found it.
3467
+ {
3468
+ woke: options.origin !== 'adopted-running',
3469
+ policy: defaultStartFailure,
3470
+ ...(heldEpoch !== undefined ? { epoch: heldEpoch } : {}),
3471
+ }, options.signal);
3472
+ // Whatever the backend's own Sandbox reports, rather than a second copy of
3473
+ // the same constant: it must keep answering after a suspend has taken the
3474
+ // handle it came from away.
3475
+ const environment = session.environment;
3476
+ return {
3477
+ id,
3478
+ origin: options.origin,
3479
+ get templateRevision() {
3480
+ return templateRevision;
3481
+ },
3482
+ get templateCurrent() {
3483
+ // An object that recorded no revision is NOT current: unknown is
3484
+ // not a match, and reporting it as one would tell a host there is
3485
+ // nothing to refresh on exactly the workspaces created before
3486
+ // anything recorded what they were built from.
3487
+ return templateRevision !== undefined && templateRevision === currentTemplateHash;
3488
+ },
3489
+ get identity() {
3490
+ return identityNow();
3491
+ },
3492
+ onGuestRestart(listener) {
3493
+ restartListeners.add(listener);
3494
+ return () => {
3495
+ restartListeners.delete(listener);
3496
+ };
3497
+ },
3498
+ get status() {
3499
+ // A suspended workspace reports 'destroyed' because that is the only
3500
+ // member of the SDK's four-way union meaning "cannot serve a call".
3501
+ // `suspended` below is what tells the recoverable state apart.
3502
+ if (state !== 'running' || session === undefined)
3503
+ return 'destroyed';
3504
+ return session.status;
3505
+ },
3506
+ get suspended() {
3507
+ // `suspending` reads as suspended because that is what a caller
3508
+ // can DO about it: no call is admitted and `resume()` is the way
3509
+ // back. The difference between the two lives where it matters —
3510
+ // in `suspendNow`, which returns early on one and not the other.
3511
+ return state === 'suspended' || state === 'suspending';
3512
+ },
3513
+ rootDir: options.rootDir,
3514
+ environment,
3515
+ async exec(command, argv, execOptions) {
3516
+ // The default path, untouched: the SDK's exec through the inner
3517
+ // handle, the shared execution controller, and the retirement
3518
+ // behaviour that comes with it.
3519
+ if (execOptions?.detach !== true && execOptions?.executionId === undefined) {
3520
+ return await admitted('exec', async (handle) => await handle.exec(command, argv, execOptions));
3521
+ }
3522
+ return await admitted('exec', async (handle) => await admittedTransport('exec', handle).execDetached(command, argv, execOptions));
3523
+ },
3524
+ async attachExecution(executionId, attachOptions) {
3525
+ return await admitted('attachExecution', async (handle) => await admittedTransport('attachExecution', handle).attachExecution(executionId, attachOptions));
3526
+ },
3527
+ async cancelExecution(executionId, transitionOptions) {
3528
+ await admitted('cancelExecution', async (handle) => await admittedTransport('cancelExecution', handle).cancelExecution(executionId, transitionOptions?.signal));
3529
+ },
3530
+ async writeFile(path, content) {
3531
+ await admitted('writeFile', async (handle) => await handle.writeFile(path, content));
3532
+ },
3533
+ /**
3534
+ * `readOptions` is FORWARDED, and that is the whole of the
3535
+ * requirement: a backend that takes `offset`/`length` and answers
3536
+ * with the whole file has given a wrong answer, not a degraded one
3537
+ * (`Sandbox.readFile` in `@namzu/sdk` says so, and the transport
3538
+ * refuses rather than downgrades against a guest too old to honour
3539
+ * them). A workspace that dropped them would do exactly that, on the
3540
+ * surface most likely to be pointed at a file worth ranging.
3541
+ */
3542
+ async readFile(path, readOptions) {
3543
+ return await admitted('readFile', async (handle) => await handle.readFile(path, readOptions));
3544
+ },
3545
+ /**
3546
+ * Draining a large output file before `suspend()` or `destroy()`
3547
+ * without holding it — the use a long-lived workspace exists for, and
3548
+ * the reason this is narrowed to present on
3549
+ * {@link KubernetesWorkspace} rather than left optional.
3550
+ *
3551
+ * Admission runs where the caller wrote the call, not at the first
3552
+ * pull, exactly as the task sandbox does it; the suspended-elsewhere
3553
+ * diagnostic runs on a failure at any point in the stream, because a
3554
+ * workspace suspended halfway through takes its pod's connection with
3555
+ * it and leaves only a socket that closed early.
3556
+ */
3557
+ readFileStream(path, readOptions) {
3558
+ return admittedStream('readFileStream', (handle) => handle.readFileStream(path, readOptions));
3559
+ },
3560
+ async listFiles(rootPath) {
3561
+ return await admitted('listFiles', async (handle) => await handle.listFiles(rootPath));
3562
+ },
3563
+ /**
3564
+ * Admitted ONCE, when the consumer asks for the first entry, and then
3565
+ * delegated to the inner handle with `yield*`.
3566
+ *
3567
+ * {@link admit} is the synchronous gate every data-plane call on this
3568
+ * handle passes, so a workspace this process knows to be suspended
3569
+ * refuses a walk exactly as it refuses `readFile` — the same error
3570
+ * class, the same `noticedBy: 'admission'`, this operation's own name —
3571
+ * and nothing is dialed. What it deliberately does NOT do is re-admit
3572
+ * per entry: a suspend that lands mid-walk surfaces as the transport
3573
+ * failure it is, and this handle learns it was suspended elsewhere on
3574
+ * its next data-plane call, which is where {@link admitted}'s one-shot
3575
+ * diagnosis lives for every other operation.
3576
+ *
3577
+ * `yield*` is also what makes cancellation work with no code of its own
3578
+ * here: a consumer breaking out of its `for await` runs this generator's
3579
+ * `return()`, and the delegation forwards it to the inner walk — which
3580
+ * is the call that terminates the guest's walk process.
3581
+ *
3582
+ * Busy accounting is not repeated either: the inner handle holds one
3583
+ * execution for the whole walk, so `status` stays `busy` from the first
3584
+ * entry to the last rather than flapping between them.
3585
+ */
3586
+ async *walkFiles(rootPath, walkOptions) {
3587
+ yield* admit('walkFiles').walkFiles(rootPath, walkOptions);
3588
+ },
3589
+ async openTerminal(terminalOptions) {
3590
+ // A session terminal goes through the TRANSPORT rather than the
3591
+ // inner handle, for the reason `sessionTransports` states: the
3592
+ // session ops are the workspace's own surface, and keeping them
3593
+ // off the inner handle's automatic-retirement path is what makes
3594
+ // "losing a connection costs the workspace nothing" true here too.
3595
+ const terminal = terminalOptions.sessionId === undefined && terminalOptions.persistent !== true
3596
+ ? await admitted('openTerminal', async (handle) => await handle.openTerminal(terminalOptions))
3597
+ : await admitted('openTerminal', async (handle) => await admittedTransport('openTerminal', handle).openTerminal(terminalOptions));
3598
+ // Tracked HERE as well as by the inner handle, because a suspend
3599
+ // reaps terminals without going through the inner handle's
3600
+ // `destroy()` — the pod is being deleted, and a caller left holding
3601
+ // a session whose exit resolves only when TCP notices is a leak.
3602
+ terminals.add(terminal);
3603
+ // `.catch` after `.finally`, not `void` alone: `exited` belongs to
3604
+ // the CALLER, who may well let it reject, and a bookkeeping chain
3605
+ // hung off it would then reject with no handler and take the host
3606
+ // process down on an unhandled rejection.
3607
+ void terminal.exited
3608
+ .finally(() => {
3609
+ terminals.delete(terminal);
3610
+ })
3611
+ .catch(() => undefined);
3612
+ return terminal;
3613
+ },
3614
+ async attachTerminal(sessionId, attachOptions) {
3615
+ const terminal = await admitted('attachTerminal', async (handle) => await admittedTransport('attachTerminal', handle).attachSession(sessionId, attachOptions));
3616
+ // Tracked exactly like an `openTerminal` result, and released the
3617
+ // same way: a suspend detaches it rather than killing the shell.
3618
+ terminals.add(terminal);
3619
+ void terminal.exited
3620
+ .finally(() => {
3621
+ terminals.delete(terminal);
3622
+ })
3623
+ .catch(() => undefined);
3624
+ return terminal;
3625
+ },
3626
+ async startDetached(detachedOptions) {
3627
+ return await admitted('startDetached', async (handle) => await admittedTransport('startDetached', handle).startDetached(detachedOptions));
3628
+ },
3629
+ async readSession(sessionId, readOptions) {
3630
+ return await admitted('readSession', async (handle) => await admittedTransport('readSession', handle).readSession(sessionId, readOptions));
3631
+ },
3632
+ async listSessions(transitionOptions) {
3633
+ return await admitted('listSessions', async (handle) => await admittedTransport('listSessions', handle).listSessions(transitionOptions?.signal));
3634
+ },
3635
+ async killSession(sessionId, killOptions) {
3636
+ return await admitted('killSession', async (handle) => await admittedTransport('killSession', handle).killSession(sessionId, killOptions ?? {}));
3637
+ },
3638
+ async openTcpConnection(connectOptions) {
3639
+ return await admitted('openTcpConnection', async (handle) => await handle.openTcpConnection(connectOptions));
3640
+ },
3641
+ async refresh(transitionOptions) {
3642
+ // Serialised, unlike the call-failure path, because nothing calls
3643
+ // this from inside a transition: it is a caller's own verb, and
3644
+ // running it between transitions rather than through one keeps it
3645
+ // from reading a mode a resume is halfway through changing.
3646
+ await serialise(async () => {
3647
+ await noticeSuspendedElsewhere(transitionOptions?.signal);
3648
+ });
3649
+ },
3650
+ async quiesce(quiesceOptions) {
3651
+ // Serialised, like the transitions it is meant to precede: a quiesce
3652
+ // racing the suspend that follows it would be a quiesce of a pod the
3653
+ // patch has already taken away, and one racing a resume would ask
3654
+ // the old pod to stop the new one's processes.
3655
+ return await serialise(async () => await quiesceNow(quiesceOptions));
3656
+ },
3657
+ async flush(flushOptions) {
3658
+ // Serialised for the reason `quiesce()` is: a flush racing the
3659
+ // suspend that follows it would be flushing a pod the patch has
3660
+ // already taken away, and one racing a resume would ask the old
3661
+ // pod about the new one's disk.
3662
+ return await serialise(async () => await flushNow(flushOptions));
3663
+ },
3664
+ async suspend(transitionOptions) {
3665
+ // The single flight takes the FIRST caller's epoch, exactly as it
3666
+ // takes the first caller's signal: a second caller arriving
3667
+ // mid-suspend is joining that transition rather than starting one
3668
+ // of its own, and one transition can only be written under one
3669
+ // authority. Its `quiesce` is the exception `suspendShared`
3670
+ // explains — a guarantee about the guest, not an authority, and
3671
+ // one a transition already patching cannot be given.
3672
+ await suspendShared(transitionOptions?.signal, assertHolderEpoch(transitionOptions?.epoch, 'suspend') ?? heldEpoch, transitionOptions?.quiesce, transitionOptions?.flush);
3673
+ },
3674
+ async resume(transitionOptions) {
3675
+ // Serialised, and deliberately without a single-flight slot of its
3676
+ // own. The two terminal transitions need one because each commits
3677
+ // its state only after the cluster confirms it, so a second caller
3678
+ // admitted behind the first finds nothing marked and re-sends;
3679
+ // `resumeNow` commits `running` at the END of a transition that
3680
+ // leaves the workspace usable, and returns early on it, so the
3681
+ // second caller finds the work done. Give resume a state it
3682
+ // early-returns on before the cluster confirms it and it will need
3683
+ // a slot as much as they do.
3684
+ await serialise(async () => await resumeNow(transitionOptions?.signal, transitionOptions?.onStartFailure ?? defaultStartFailure, assertHolderEpoch(transitionOptions?.epoch, 'resume') ?? heldEpoch, transitionOptions?.refreshPodTemplate === true));
3685
+ },
3686
+ async destroy(destroyOptions) {
3687
+ assertHolderEpoch(destroyOptions?.epoch, 'destroy');
3688
+ if (destroyOptions?.deleteDisk !== true) {
3689
+ // The default, and the whole point of the default: there is no
3690
+ // delete-compute-keep-disk verb, so the closest thing to one is
3691
+ // a suspend, and `destroy()` in a `finally` must not erase a
3692
+ // workspace nobody asked to erase. It shares the suspend's
3693
+ // single flight, so `destroy()` racing `suspend()` is one
3694
+ // transition rather than two — and it stays idempotent over a
3695
+ // workspace already deleted, where `suspend()` itself refuses.
3696
+ await destroyBySuspending(destroyOptions?.signal, destroyOptions?.epoch ?? heldEpoch, destroyOptions?.quiesce, destroyOptions?.flush);
3697
+ return;
3698
+ }
3699
+ await deleteShared(destroyOptions);
3700
+ },
3701
+ };
3702
+ }
3703
+ //# sourceMappingURL=workspace.js.map