@namzu/sandbox 14.0.0 → 15.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (99) hide show
  1. package/CHANGELOG.md +838 -0
  2. package/README.md +310 -14
  3. package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
  4. package/dist/backends/aci-standby-pool/index.js +13 -1
  5. package/dist/backends/aci-standby-pool/index.js.map +1 -1
  6. package/dist/backends/docker/index.d.ts.map +1 -1
  7. package/dist/backends/docker/index.js +19 -1
  8. package/dist/backends/docker/index.js.map +1 -1
  9. package/dist/backends/firecracker/index.d.ts.map +1 -1
  10. package/dist/backends/firecracker/index.js +12 -2
  11. package/dist/backends/firecracker/index.js.map +1 -1
  12. package/dist/backends/firecracker/protocol.d.ts +459 -8
  13. package/dist/backends/firecracker/protocol.d.ts.map +1 -1
  14. package/dist/backends/firecracker/protocol.js +136 -0
  15. package/dist/backends/firecracker/protocol.js.map +1 -1
  16. package/dist/backends/firecracker/transport.d.ts +539 -6
  17. package/dist/backends/firecracker/transport.d.ts.map +1 -1
  18. package/dist/backends/firecracker/transport.js +1171 -24
  19. package/dist/backends/firecracker/transport.js.map +1 -1
  20. package/dist/backends/kubernetes/egress-policy.d.ts +1088 -11
  21. package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -1
  22. package/dist/backends/kubernetes/egress-policy.js +2173 -29
  23. package/dist/backends/kubernetes/egress-policy.js.map +1 -1
  24. package/dist/backends/kubernetes/identity.d.ts +193 -0
  25. package/dist/backends/kubernetes/identity.d.ts.map +1 -0
  26. package/dist/backends/kubernetes/identity.js +147 -0
  27. package/dist/backends/kubernetes/identity.js.map +1 -0
  28. package/dist/backends/kubernetes/index.d.ts +678 -33
  29. package/dist/backends/kubernetes/index.d.ts.map +1 -1
  30. package/dist/backends/kubernetes/index.js +1180 -95
  31. package/dist/backends/kubernetes/index.js.map +1 -1
  32. package/dist/backends/kubernetes/ingress-policy.d.ts +375 -0
  33. package/dist/backends/kubernetes/ingress-policy.d.ts.map +1 -0
  34. package/dist/backends/kubernetes/ingress-policy.js +1050 -0
  35. package/dist/backends/kubernetes/ingress-policy.js.map +1 -0
  36. package/dist/backends/kubernetes/k8s-client.d.ts +213 -4
  37. package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -1
  38. package/dist/backends/kubernetes/k8s-client.js +359 -52
  39. package/dist/backends/kubernetes/k8s-client.js.map +1 -1
  40. package/dist/backends/kubernetes/lease.d.ts +40 -14
  41. package/dist/backends/kubernetes/lease.d.ts.map +1 -1
  42. package/dist/backends/kubernetes/lease.js +68 -18
  43. package/dist/backends/kubernetes/lease.js.map +1 -1
  44. package/dist/backends/kubernetes/objects.d.ts +423 -3
  45. package/dist/backends/kubernetes/objects.d.ts.map +1 -1
  46. package/dist/backends/kubernetes/objects.js +364 -2
  47. package/dist/backends/kubernetes/objects.js.map +1 -1
  48. package/dist/backends/kubernetes/per-sandbox-policy.d.ts +219 -0
  49. package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -0
  50. package/dist/backends/kubernetes/per-sandbox-policy.js +407 -0
  51. package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -0
  52. package/dist/backends/kubernetes/rbac.d.ts +153 -0
  53. package/dist/backends/kubernetes/rbac.d.ts.map +1 -0
  54. package/dist/backends/kubernetes/rbac.js +177 -0
  55. package/dist/backends/kubernetes/rbac.js.map +1 -0
  56. package/dist/backends/kubernetes/sandbox.d.ts +81 -14
  57. package/dist/backends/kubernetes/sandbox.d.ts.map +1 -1
  58. package/dist/backends/kubernetes/sandbox.js +149 -15
  59. package/dist/backends/kubernetes/sandbox.js.map +1 -1
  60. package/dist/backends/kubernetes/transport.d.ts +935 -9
  61. package/dist/backends/kubernetes/transport.d.ts.map +1 -1
  62. package/dist/backends/kubernetes/transport.js +1958 -62
  63. package/dist/backends/kubernetes/transport.js.map +1 -1
  64. package/dist/backends/kubernetes/workspace.d.ts +1149 -18
  65. package/dist/backends/kubernetes/workspace.d.ts.map +1 -1
  66. package/dist/backends/kubernetes/workspace.js +2825 -186
  67. package/dist/backends/kubernetes/workspace.js.map +1 -1
  68. package/dist/backends/remote-execution-controller.d.ts +14 -0
  69. package/dist/backends/remote-execution-controller.d.ts.map +1 -1
  70. package/dist/backends/remote-execution-controller.js.map +1 -1
  71. package/dist/index.d.ts +231 -13
  72. package/dist/index.d.ts.map +1 -1
  73. package/dist/index.js +247 -5
  74. package/dist/index.js.map +1 -1
  75. package/dist/testing/sandbox-conformance.d.ts +39 -5
  76. package/dist/testing/sandbox-conformance.d.ts.map +1 -1
  77. package/dist/testing/sandbox-conformance.js +436 -5
  78. package/dist/testing/sandbox-conformance.js.map +1 -1
  79. package/package.json +3 -3
  80. package/src/backends/aci-standby-pool/index.ts +16 -1
  81. package/src/backends/docker/index.ts +22 -1
  82. package/src/backends/firecracker/index.ts +14 -2
  83. package/src/backends/firecracker/protocol.ts +514 -6
  84. package/src/backends/firecracker/transport.ts +1492 -40
  85. package/src/backends/kubernetes/egress-policy.ts +3064 -53
  86. package/src/backends/kubernetes/identity.ts +261 -0
  87. package/src/backends/kubernetes/index.ts +1785 -127
  88. package/src/backends/kubernetes/ingress-policy.ts +1344 -0
  89. package/src/backends/kubernetes/k8s-client.ts +444 -54
  90. package/src/backends/kubernetes/lease.ts +75 -19
  91. package/src/backends/kubernetes/objects.ts +626 -6
  92. package/src/backends/kubernetes/per-sandbox-policy.ts +542 -0
  93. package/src/backends/kubernetes/rbac.ts +192 -0
  94. package/src/backends/kubernetes/sandbox.ts +218 -20
  95. package/src/backends/kubernetes/transport.ts +2733 -124
  96. package/src/backends/kubernetes/workspace.ts +4476 -222
  97. package/src/backends/remote-execution-controller.ts +14 -0
  98. package/src/index.ts +595 -14
  99. package/src/testing/sandbox-conformance.ts +540 -5
@@ -62,20 +62,64 @@
62
62
  * the pod — so the call would hang until a connect timeout with nothing in
63
63
  * the failure naming the suspend.
64
64
  *
65
+ * Both patches also stamp
66
+ * {@link OPERATING_MODE_CHANGED_AT_ANNOTATION_KEY}. Nothing in this module
67
+ * reads it back; it exists so that an inventory can say when a workspace was
68
+ * last put to sleep without waking it up to ask, which the controller's own
69
+ * lingering `Suspended` condition cannot answer. See
70
+ * {@link listKubernetesWorkspaces}.
71
+ *
72
+ * ## Three verbs that never open a workspace, and one that notices
73
+ *
74
+ * {@link createKubernetesWorkspace} adopts AND resumes, which is right for a
75
+ * host about to USE a workspace and wrong for everything else. Deleting a
76
+ * month-old suspended workspace through it means starting a pod, probing it
77
+ * and deleting it again; taking an inventory means waking every suspended
78
+ * object in the namespace. So the three operations that are about the OBJECT
79
+ * are reachable without a handle — {@link listKubernetesWorkspaces},
80
+ * {@link deleteKubernetesWorkspace} and {@link suspendKubernetesWorkspace} —
81
+ * and none of them creates a pod, dials an agent or resumes anything.
82
+ *
83
+ * The other half of the same problem is the handle that was already open when
84
+ * somebody else did one of those. A workspace id is a name, not a lock, so a
85
+ * second process can suspend the workspace this one is holding, and this
86
+ * handle's `state` is a record of what THIS process did. Two things fix that,
87
+ * and both re-read the object rather than guessing:
88
+ * {@link KubernetesWorkspace.refresh} when the caller asks, and the re-read
89
+ * after a call that FAILED at the transport — which is how it would otherwise
90
+ * be found out, as a connect refusal or a flat `unauthorized` naming nothing.
91
+ * Either one records the suspension as UNCONFIRMED (`suspending`, not
92
+ * `suspended`): what was observed is the object's mode, not the pod stopping,
93
+ * and only a wait this process performed can promise the disk is quiesced.
94
+ *
65
95
  * ## Adoption is checked against the object, not against the caller
66
96
  *
67
97
  * A create that collides with an existing object of the same name ADOPTS it,
68
98
  * because the deterministic name is only worth having if coming back is the
69
99
  * normal path. What is adopted is then checked against the configuration: the
70
- * block disk, the `sandbox.namzu.ai/template` pod label (the label an egress
71
- * NetworkPolicy selects by) and `runtimeClassName` (the VM boundary). A
72
- * standing object that disagrees with any of them is refused by name rather
73
- * than driven — see {@link KubernetesWorkspaceMismatchError}. What is NOT
100
+ * block disk, the `sandbox.namzu.ai/template` pod label (the label the
101
+ * ingress and egress policies select by) and `runtimeClassName` (the VM
102
+ * boundary). A standing object that disagrees with any of them is refused by
103
+ * name rather than driven — see {@link KubernetesWorkspaceMismatchError}. What is NOT
74
104
  * checked, and cannot be from here, is whether somebody else is already using
75
105
  * it: two host processes can hold handles to one running workspace, and the
76
106
  * `destroy()` of either suspends the pod the other is executing in. A
77
107
  * workspace id is a name, not a lock.
78
108
  *
109
+ * An adopt also has to survive walking in ON a transition, which is the
110
+ * normal way a workspace is found rather than an edge: a suspend that ended
111
+ * in {@link KubernetesWorkspaceSuspendTimeoutError}, a second host coming up
112
+ * during a rollout, a host restarting inside the previous pod's
113
+ * `terminationGracePeriodSeconds`. In all three the pod under the name is
114
+ * draining or already gone, and the replacement has not been created yet. So
115
+ * an adopt that finds the object `Suspended`, or finds its pod carrying a
116
+ * `deletionTimestamp`, WAITS for the replacement under the same readiness
117
+ * budget the resume path waits under, instead of failing on the first read
118
+ * that finds no live pod. What it came by is then reported as `origin`:
119
+ * `created`, `adopted-running` or `resumed` — a host that adopted a pod
120
+ * another process left running needs to know that none of that process's
121
+ * terminals survived it.
122
+ *
79
123
  * ## There is no delete-compute-keep-disk verb
80
124
  *
81
125
  * The API has `operatingMode` and it has DELETE. Nothing in between. So
@@ -91,9 +135,10 @@
91
135
  * {@link createKubernetesWorkspace}), and a pod that stops being able to say
92
136
  * what happened to a command — the shared execution controller's unconfirmed
93
137
  * cancellation, which retires a task sandbox by DELETING it — is retired here
94
- * by that same suspend patch (see {@link retireSession}). Exactly one DELETE
95
- * is reachable from this module, and it is the one `deleteDisk: true` asks
96
- * for.
138
+ * by that same suspend patch (see {@link retireSession}). Exactly two DELETEs
139
+ * are reachable from this module and both are ASKED FOR by name — the one
140
+ * `deleteDisk: true` sends and the one {@link deleteKubernetesWorkspace} is;
141
+ * no failure path, no `finally`, and no default reaches either.
97
142
  *
98
143
  * ## A state is committed when the cluster confirms it, never before
99
144
  *
@@ -113,12 +158,14 @@
113
158
  * second request into the gap the deferred mark opens.
114
159
  */
115
160
  import { OperationDeadline, OperationDeadlineExpired, runFailureCleanup } from '../readiness.js';
116
- import { assertEgressPolicyIsEnforceable } from './egress-policy.js';
117
- import { DEFAULT_AGENT_PORT, bindingFromSandbox, buildSandboxBody, clientAccess, pollForBinding, probeSandboxPrivileges, readPodBindToken, readSandboxTemplate, resolveAgentAddress, resolveKubernetesReadiness, resolveProbeTimeoutMs, verifyEgressPolicyConfigured, } from './index.js';
118
- import { KubernetesAlreadyGoneError, KubernetesConflictError, createKubernetesClient, } from './k8s-client.js';
119
- import { SANDBOX_TEMPLATE_LABEL_KEY, isPodStopped, podPath, sandboxCollectionPath, sandboxPath, } from './objects.js';
161
+ import { RemoteCancellationUnknownError } from '../remote-execution-controller.js';
162
+ import { assertEgressPolicyIsEnforceable, assertEgressProfileIsUsable, assertWorkspaceCarriesNoPerSandboxEgress, composeAdditionalPodLabels, egressProfileLabel, } from './egress-policy.js';
163
+ import { KubernetesWorkspaceGuestGoneError, KubernetesWorkspaceReplacedError, } from './identity.js';
164
+ import { DEFAULT_AGENT_PORT, ReadinessPollTimeout, bindingFromSandbox, buildEgressBoundary, buildIngressVerifier, buildSandboxBody, clientAccess, clientOptions, pollForBinding, probeSandboxPrivileges, readBoundPod, readSandboxTemplate, resolveAgentAddress, resolveKubernetesReadiness, resolveProbeTimeoutMs, resolveStreamHeartbeatMs, sandboxPodLabels, sandboxPodTemplate, } from './index.js';
165
+ import { KubernetesAlreadyGoneError, KubernetesConflictError, KubernetesPatchNotAppliedError, createKubernetesClient, } from './k8s-client.js';
166
+ import { HOLDER_EPOCH_ANNOTATION_KEY, OPERATING_MODE_CHANGED_AT_ANNOTATION_KEY, POD_TEMPLATE_HASH_ANNOTATION_KEY, SANDBOX_TEMPLATE_LABEL_KEY, buildHolderEpochPatch, holderEpochAllows, isPodStopped, persistentVolumeClaimPath, podPath, podTemplateHash, readHolderEpoch, readPodTemplateHash, sandboxCollectionPath, sandboxPath, } from './objects.js';
120
167
  import { KubernetesSandboxDestroyedError, buildKubernetesSandbox, } from './sandbox.js';
121
- import { KubernetesAgentTransport } from './transport.js';
168
+ import { KubernetesAgentTransport, KubernetesFlushUnconfirmedError, KubernetesFlushUnreachableError, KubernetesFlushUnsupportedError, KubernetesQuiesceUnconfirmedError, KubernetesQuiesceUnsupportedError, flushUnsupportedError, guestWhenReserved, } from './transport.js';
122
169
  /**
123
170
  * Thrown when a workspace's `SandboxTemplate` does not describe a block disk
124
171
  * this backend is willing to build a workspace on.
@@ -155,6 +202,14 @@ export class KubernetesWorkspaceDiskError extends Error {
155
202
  * egress `NetworkPolicy`'s `podSelector` matches. A pod carrying another
156
203
  * value — or none — is not selected by the policy this call just verified,
157
204
  * so the boundary would report verified while covering nothing.
205
+ * - the egress PROFILE label, when `config.egress.profile` is set, for
206
+ * exactly the same reason: it is the selector's second half. A standing
207
+ * object built before the profile existed, or under a different one,
208
+ * carries a pod the per-profile policy does not select — and once each
209
+ * profile has its own policy object, as it must, a pod carrying neither
210
+ * key is selected by no egress policy at all. This one is checked only
211
+ * when a profile is configured: with none, the policy this call verified
212
+ * selects the template label alone, which the object does carry.
158
213
  * - `runtimeClassName`, which is the VM boundary. The privilege probe cannot
159
214
  * stand in for it: `/proc/self/status` reads the same inside a VM guest as
160
215
  * it does inside an ordinary shared-kernel container.
@@ -174,7 +229,11 @@ export class KubernetesWorkspaceMismatchError extends Error {
174
229
  sandboxName,
175
230
  /** Which configured field the standing object disagrees with. */
176
231
  field,
177
- /** What the configuration asked for. */
232
+ /**
233
+ * What the configuration asked for. `key=value` for `egressProfile`,
234
+ * because the KEY is configurable too and the value alone would not
235
+ * say which label was compared.
236
+ */
178
237
  expected,
179
238
  /** What the object carries — absent when it carries nothing at all. */
180
239
  actual, message) {
@@ -189,19 +248,33 @@ export class KubernetesWorkspaceMismatchError extends Error {
189
248
  * Thrown by every operation on a workspace that is currently suspended.
190
249
  *
191
250
  * Distinct from {@link KubernetesSandboxDestroyedError} because the state is
192
- * RECOVERABLE and the advice is one word: call `resume()`. Nothing is dialed
193
- * before it is thrown.
251
+ * RECOVERABLE and the advice is one word: call `resume()`.
252
+ *
253
+ * `noticedBy` says which of the two ways the caller got here, and the message
254
+ * changes with it, because "nothing was dialed" is a promise the first one
255
+ * keeps and the second one cannot. A workspace another process suspended is
256
+ * discovered by a call FAILING — the pod is gone, so the dial is refused, or
257
+ * the replacement pod's agent refuses this handle's token — and the reason
258
+ * that is worth converting into this error rather than passing on is that the
259
+ * raw failure names nothing: a flat `unauthorized`, or a connect error against
260
+ * an address that still resolves because the Service outlives the pod.
194
261
  */
195
262
  export class KubernetesWorkspaceSuspendedError extends Error {
196
263
  operation;
197
264
  workspaceId;
198
265
  sandboxName;
266
+ noticedBy;
199
267
  name = 'KubernetesWorkspaceSuspendedError';
200
- constructor(operation, workspaceId, sandboxName) {
201
- super(`kubernetes workspace ${workspaceId} (Sandbox ${sandboxName}) is suspended; ${operation}() cannot be admitted and nothing was dialed. Its pod is deleted and its disk is intact — call resume() to get a new pod, a new address and a new agent token, then retry.`);
268
+ constructor(operation, workspaceId, sandboxName,
269
+ /** See {@link KubernetesWorkspaceSuspensionNotice}. Defaults to `admission`. */
270
+ noticedBy = 'admission', options) {
271
+ super(noticedBy === 'admission'
272
+ ? `kubernetes workspace ${workspaceId} (Sandbox ${sandboxName}) is suspended; ${operation}() cannot be admitted and nothing was dialed. Its pod is deleted and its disk is intact — call resume() to get a new pod, a new address and a new agent token, then retry.`
273
+ : `kubernetes workspace ${workspaceId} (Sandbox ${sandboxName}) is suspended; ${operation}() was admitted, failed at the transport, and a re-read of the Sandbox found spec.operatingMode: Suspended — another process suspended this workspace while this handle was holding it. The transport failure is on \`cause\`; it names nothing useful on its own, because the Service outlives the pod and the address still resolves. The disk is intact — call resume() to get a new pod, a new address and a new agent token, then retry.`, options);
202
274
  this.operation = operation;
203
275
  this.workspaceId = workspaceId;
204
276
  this.sandboxName = sandboxName;
277
+ this.noticedBy = noticedBy;
205
278
  }
206
279
  }
207
280
  /**
@@ -235,6 +308,107 @@ export class KubernetesWorkspaceSuspendTimeoutError extends Error {
235
308
  this.timeoutMs = timeoutMs;
236
309
  }
237
310
  }
311
+ /**
312
+ * Thrown when a lifecycle write carried a holder epoch the workspace has
313
+ * already moved past: the request was NOT sent, or was sent and refused, and
314
+ * nothing on the cluster changed either way.
315
+ *
316
+ * The workspace is somebody else's now. The epoch stored on the Sandbox is
317
+ * higher than the one this call carried, which is what a host says when it
318
+ * hands authority over a workspace to another process — see
319
+ * {@link HOLDER_EPOCH_ANNOTATION_KEY}. A superseded holder that suspends,
320
+ * resumes or deletes anyway would be taking the pod, or the disk, away from
321
+ * whoever holds it now.
322
+ *
323
+ * Nothing about the handle changes either. A refused `suspend()` leaves the
324
+ * handle exactly where it was — still `running`, terminals still open,
325
+ * because the refusal is decided BEFORE they are reaped — so a caller that
326
+ * catches this and re-reads its own holder record has lost nothing.
327
+ *
328
+ * `storedEpoch` is `undefined` in the one case the annotation cannot be read
329
+ * at all: it is present and is not a decimal integer, which no release of
330
+ * this backend writes. `storedAnnotation` carries it verbatim so an operator
331
+ * can see what is actually on the object.
332
+ */
333
+ export class KubernetesWorkspacePreconditionError extends Error {
334
+ operation;
335
+ workspaceId;
336
+ sandboxName;
337
+ epoch;
338
+ storedEpoch;
339
+ storedAnnotation;
340
+ name = 'KubernetesWorkspacePreconditionError';
341
+ constructor(
342
+ /** The verb that was refused — `suspend`, `resume`, `destroy`, ... */
343
+ operation, workspaceId, sandboxName,
344
+ /** The epoch this call carried. */
345
+ epoch,
346
+ /** The epoch stored on the Sandbox, or `undefined` if unreadable. */
347
+ storedEpoch,
348
+ /** The annotation exactly as stored, when there is one. */
349
+ storedAnnotation) {
350
+ super(storedEpoch === undefined
351
+ ? `kubernetes: ${operation}() on workspace ${workspaceId} (Sandbox ${sandboxName}) carried holder epoch ${epoch}, and the Sandbox's ${HOLDER_EPOCH_ANNOTATION_KEY} annotation reads ${JSON.stringify(storedAnnotation)}, which is not a decimal integer. No release of this backend writes that, so it was set by hand or by something else; the write was refused rather than overwriting a fence this code does not understand. Nothing on the cluster changed. Fix the annotation, or remove it to start the workspace's epoch again at 0.`
352
+ : `kubernetes: ${operation}() on workspace ${workspaceId} (Sandbox ${sandboxName}) carried holder epoch ${epoch}, but the Sandbox is held at epoch ${storedEpoch} — another process took this workspace over. Nothing on the cluster changed and nothing about this handle changed: its terminals are still open and it is still admitting calls. A host that raised the epoch elsewhere should stop using this handle; one that believes it is still the holder should re-read its own record first.`);
353
+ this.operation = operation;
354
+ this.workspaceId = workspaceId;
355
+ this.sandboxName = sandboxName;
356
+ this.epoch = epoch;
357
+ this.storedEpoch = storedEpoch;
358
+ this.storedAnnotation = storedAnnotation;
359
+ }
360
+ }
361
+ /**
362
+ * Refuse an epoch this backend cannot honour, at the entry point rather than
363
+ * at the request — the same place {@link resolveRequestTimeoutMs} refuses a
364
+ * timeout, and for the same reason: a configuration that will never work
365
+ * should be named where the caller can still see which call it came from.
366
+ *
367
+ * Non-negative because the stored value is compared as a number and written
368
+ * as a decimal string, and an integer because a fractional epoch would
369
+ * round-trip through the annotation as something other than what was passed.
370
+ */
371
+ function assertHolderEpoch(epoch, where) {
372
+ if (epoch === undefined)
373
+ return undefined;
374
+ if (!Number.isSafeInteger(epoch) || epoch < 0) {
375
+ throw new Error(`kubernetes: ${where} epoch must be a non-negative safe integer, got ${JSON.stringify(epoch)}. It is stored on the Sandbox as a decimal string and compared as a number, so anything else could not be written back as the value that was passed.`);
376
+ }
377
+ return epoch;
378
+ }
379
+ /**
380
+ * Refuse a `quiesce` asked of a verb that has no guest to ask.
381
+ *
382
+ * {@link suspendKubernetesWorkspace} reaches a workspace WITHOUT opening one:
383
+ * it sends a patch and waits for the pod, and never dials the agent. So there
384
+ * is nothing there that could stop a process, and silently ignoring the flag
385
+ * would hand a caller a suspend it believes was preceded by a quiesce —
386
+ * exactly the belief this whole feature exists to make true. The handle's
387
+ * `suspend()` is the verb that can do it, and the message says so.
388
+ */
389
+ function assertNoQuiesceHere(quiesce, where) {
390
+ if (quiesce === undefined || quiesce === false)
391
+ return;
392
+ throw new Error(`kubernetes: ${where} cannot quiesce the guest: it reaches the workspace through the API server alone and never dials the agent, so there is no connection on which to stop anything. Open the workspace with createKubernetesWorkspace() and call suspend({ quiesce: true }) on the handle, or quiesce() and then this. The option is refused rather than ignored, because a caller that passed it is about to trust a capture.`);
393
+ }
394
+ /**
395
+ * Refuse a `flush: true` asked of a verb that has no guest to ask.
396
+ *
397
+ * The same rule as {@link assertNoQuiesceHere} and one difference worth
398
+ * saying out loud: a flush is ON by default on the handle's `suspend()`, so
399
+ * the DEFAULT reaching {@link suspendKubernetesWorkspace} cannot be an
400
+ * error — it is what every caller of that verb has always passed. Only an
401
+ * EXPLICIT flush — `true`, or an object naming a `timeoutMs` — is refused,
402
+ * because only that caller believes the writes are being put on the device.
403
+ * `flush: false` is accepted and means what it says here: this verb never
404
+ * flushes, and its own documentation is where that is stated rather than
405
+ * hidden behind a thrown error nobody sees.
406
+ */
407
+ function assertNoFlushHere(flush, where) {
408
+ if (flush === undefined || flush === false)
409
+ return;
410
+ throw new Error(`kubernetes: ${where} cannot flush the guest: it reaches the workspace through the API server alone and never dials the agent, so there is no connection on which to ask for a syncfs. What the disk keeps is whatever the guest kernel had already written back. Open the workspace with createKubernetesWorkspace() and call suspend() on the handle, which flushes by default. The option is refused rather than ignored, because a caller that passed it is about to trust the disk.`);
411
+ }
238
412
  /** Prefix every workspace Sandbox's name carries. */
239
413
  export const WORKSPACE_NAME_PREFIX = 'namzu-ws-';
240
414
  /** DNS-1123 label: the Sandbox's name is also its Pod's and its Service's. */
@@ -323,6 +497,72 @@ export function assertBlockModeWorkspaceDisk(source, podTemplate, volumeClaimTem
323
497
  }
324
498
  }
325
499
  }
500
+ /** The `metadata.name` of each claim, in declaration order, blanks dropped. */
501
+ function volumeClaimTemplateNames(volumeClaimTemplates) {
502
+ const names = [];
503
+ for (const entry of volumeClaimTemplates ?? []) {
504
+ const name = entry.metadata?.name;
505
+ if (typeof name === 'string' && name !== '')
506
+ names.push(name);
507
+ }
508
+ return names;
509
+ }
510
+ /**
511
+ * Refuse a refresh whose template declares a DIFFERENT set of disks than the
512
+ * Sandbox it would be written onto.
513
+ *
514
+ * Only a refresh can arrive here, and that is the whole point. On the create
515
+ * path the pod template and the `volumeClaimTemplates` come out of the same
516
+ * `SandboxTemplate` in the same read, so the two sides cannot disagree. On a
517
+ * refresh they are two different objects: `spec.volumeClaimTemplates` is
518
+ * CEL-immutable, so it is never in the patch and stays whatever the create
519
+ * POST froze, while `spec.podTemplate` becomes whatever the template says
520
+ * TODAY.
521
+ *
522
+ * {@link assertBlockModeWorkspaceDisk} asks one half of the question — is
523
+ * every disk this Sandbox HAS still claimed as a block device by the pod
524
+ * template about to land. This asks the other half, which nothing else can
525
+ * see: does the template claim a disk this Sandbox does not have. Adding a
526
+ * second `volumeClaimTemplates` entry with its matching `volumeDevices` entry
527
+ * is the ordinary way to give a workspace another disk, and a template edited
528
+ * that way is internally consistent — it passes every check made against
529
+ * itself, and every branch of the check above, because the original disk is
530
+ * still claimed. Written onto a standing workspace it produces a pod spec
531
+ * naming a device the Sandbox has no PVC for. The shipped workspace template
532
+ * declares no `spec.volumes`, so there is no other source for that name: the
533
+ * controller would be left unable to build a valid pod, the workspace would
534
+ * sit `Running` with nothing coming up, and the caller would see a bind
535
+ * timeout naming nothing. Recovery would mean suspending, reverting the
536
+ * template and refreshing again — the object having been left permanently
537
+ * describing a disk that does not exist.
538
+ *
539
+ * Comparing the NAMES is the whole rule, because the name is what the
540
+ * controller wires by (StatefulSet style, `<entry name>-<sandbox name>`). An
541
+ * edit to an existing entry's other fields — its size, its storage class,
542
+ * its `volumeMode` — simply does not apply to a standing workspace, and is
543
+ * not refused here: the disk it describes is the disk that is already there.
544
+ * Only a template pointed at a workspace whose disks it cannot describe at
545
+ * all is a mistake worth stopping.
546
+ *
547
+ * What neither check catches, and deliberately: a template whose containers
548
+ * claim a `volumeDevices` name that is in NEITHER its own
549
+ * `volumeClaimTemplates` NOR the Sandbox's. The check above asks only that
550
+ * every disk the Sandbox HAS is still claimed, and this one asks only about
551
+ * names the template's `volumeClaimTemplates` adds, so a device claimed out
552
+ * of nowhere passes both. That template is equally broken on the CREATE path,
553
+ * which has no check either and would produce the same unbuildable pod on a
554
+ * brand-new workspace — it is a template that is wrong about itself, not a
555
+ * template that is wrong about this workspace, and this pair of refusals is
556
+ * only about the second kind.
557
+ */
558
+ function assertRefreshedTemplateClaimsTheSameDisks(source, templateVolumeClaimTemplates, sandboxVolumeClaimTemplates) {
559
+ const wanted = volumeClaimTemplateNames(templateVolumeClaimTemplates);
560
+ const held = new Set(volumeClaimTemplateNames(sandboxVolumeClaimTemplates));
561
+ const extra = wanted.filter((name) => !held.has(name));
562
+ if (extra.length === 0)
563
+ return;
564
+ throw new KubernetesWorkspaceDiskError(source, `kubernetes: ${source} declares the volumeClaimTemplate${extra.length === 1 ? '' : 's'} ${extra.map((name) => JSON.stringify(name)).join(', ')}, which this workspace's Sandbox does not have — it was created with ${[...held].map((name) => JSON.stringify(name)).join(', ') || 'none'} and spec.volumeClaimTemplates is CEL-immutable, so no refresh can add a disk to a workspace that already exists. Writing this template's pod spec onto it would claim a device node backed by no PVC, and the controller would never build a valid pod. Refresh this workspace against a template that declares the disks it already has, or create a new workspace from this template and migrate the data.`);
565
+ }
326
566
  /**
327
567
  * Refuse a standing Sandbox that was built from another `SandboxTemplate`, or
328
568
  * that runs without the RuntimeClass this backend is configured for.
@@ -336,12 +576,41 @@ export function assertBlockModeWorkspaceDisk(source, podTemplate, volumeClaimTem
336
576
  * object whose label says one thing while the caller builds from another is a
337
577
  * mix-up worth naming the first time it is seen rather than the first time a
338
578
  * policy is switched on. See {@link KubernetesWorkspaceMismatchError}.
579
+ *
580
+ * The PROFILE check is conditional, because a profile is what this
581
+ * configuration asks for rather than something every object has: it runs when
582
+ * `config.egress.profile` is set, and then the object's own pod template has
583
+ * to carry that key with that value — unless this call is about to REWRITE
584
+ * that pod template (`refreshPodTemplate: true` onto a Suspended object),
585
+ * which writes the configured profile with the rest of the overlays. That is
586
+ * the same lifting the `runtimeClassName` check gets, for the same reason and
587
+ * with the same re-application if the patch does not land.
588
+ *
589
+ * Both halves of the policy selector are therefore checked against the object
590
+ * on this path — which matters because neither network check further up can
591
+ * do it: the ingress verification and the egress union both run against the
592
+ * labels the POST *would* stamp, and on this path the POST already lost to a
593
+ * 409. What this does NOT check is a
594
+ * label the standing object carries that this configuration does not ask for
595
+ * — an object built under a profile this config has dropped, say. That pod is
596
+ * still selected by the unprofiled policy this call verified (a selector is a
597
+ * subset match), so the boundary it reports holds; what a stale label could
598
+ * do is bring a SECOND policy into the union, and the adopt path cannot
599
+ * enumerate the standing object's labels without reading them, which is a
600
+ * wider change than this refusal.
339
601
  */
340
602
  export function assertAdoptedWorkspaceMatchesConfig(sandboxName, namespace, podTemplate, expected) {
341
603
  const label = podTemplate?.metadata?.labels?.[SANDBOX_TEMPLATE_LABEL_KEY];
342
604
  if (label !== expected.sandboxTemplateName) {
343
605
  throw new KubernetesWorkspaceMismatchError(sandboxName, 'sandboxTemplateName', expected.sandboxTemplateName, label, `kubernetes: Sandbox ${sandboxName} in namespace ${namespace} already exists, and its spec.podTemplate carries ${SANDBOX_TEMPLATE_LABEL_KEY}: ${label === undefined ? '(absent)' : JSON.stringify(label)} rather than ${JSON.stringify(expected.sandboxTemplateName)} — it was built from a different SandboxTemplate, so it is not the workspace this call describes. That label is the one an egress NetworkPolicy's podSelector matches, so adopting this object would hand back a pod the policy verified for ${JSON.stringify(expected.sandboxTemplateName)} does not select, having reported the boundary as verified. Point this workspace at the template the object was built from, or delete the Sandbox — which takes its disk with it — and create it again, or choose another workspaceId.`);
344
606
  }
607
+ const profile = expected.profile;
608
+ if (profile !== undefined) {
609
+ const carried = podTemplate?.metadata?.labels?.[profile.key];
610
+ if (carried !== profile.value) {
611
+ throw new KubernetesWorkspaceMismatchError(sandboxName, 'egressProfile', `${profile.key}=${profile.value}`, carried === undefined ? undefined : `${profile.key}=${carried}`, `kubernetes: Sandbox ${sandboxName} in namespace ${namespace} already exists, and its spec.podTemplate carries ${profile.key}: ${carried === undefined ? '(absent)' : JSON.stringify(carried)} rather than the configured egress profile ${JSON.stringify(profile.value)}. The translated policy's podSelector matches that label as well as the template one, so this object's pod is not selected by the policy this call just verified — and since each profile needs its own policy object, a pod carrying no profile label at all is selected by none of them, while create() would have reported the boundary verified. A standing workspace's pod labels are not rewritten underneath it by an ordinary adopt, so the mismatch is named instead. Three ways out, and only the last one costs the disk: reopen it with refreshPodTemplate: true, which rewrites spec.podTemplate — profile label and all — on the one Suspended → Running transition and is the supported way to move a workspace between profiles; or point config.egress.profile at ${carried === undefined ? 'the profile this object was built under (none was)' : JSON.stringify(carried)}; or delete the Sandbox — which takes its disk with it — and create it again under ${JSON.stringify(profile.value)}.`);
612
+ }
613
+ }
345
614
  if (expected.runtimeClassName === undefined)
346
615
  return;
347
616
  const declared = podTemplate?.spec?.runtimeClassName;
@@ -375,12 +644,28 @@ export function assertAdoptedWorkspaceMatchesConfig(sandboxName, namespace, podT
375
644
  */
376
645
  export async function createKubernetesWorkspace(config, options) {
377
646
  options.signal?.throwIfAborted();
647
+ // `config.egress.perSandbox` is the one part of `config.egress` this path
648
+ // cannot honour at all, and it is refused FIRST — before readiness, before
649
+ // the client, before any request. It configures `setNetworkPolicy`, which
650
+ // a workspace handle never carries: nothing in the create path below
651
+ // composes a per-sandbox pod label or tracks an owner uid for one, and
652
+ // the option exists to make the method present on a TASK handle. Accepting
653
+ // it here and omitting the method is the silent downgrade every other
654
+ // refusal in `egress-policy.ts` exists to prevent, so it is refused by
655
+ // name rather than validated: `assertPerSandboxEgressIsUsable` would
656
+ // check a config this path serves no purpose for and report it as usable.
657
+ assertWorkspaceCarriesNoPerSandboxEgress(config.egress);
378
658
  const readiness = resolveKubernetesReadiness(config);
379
659
  const namespace = config.namespace;
380
660
  const name = workspaceSandboxName(options.workspaceId);
381
661
  const templateName = options.sandboxTemplateName ?? config.sandboxTemplateName;
382
662
  const agentPort = config.agentPort ?? DEFAULT_AGENT_PORT;
383
- const client = createKubernetesClient(clientAccess(config));
663
+ const agentAddress = config.agentAddress ?? 'service';
664
+ // Resolved before anything is POSTed, so a configuration this backend
665
+ // will never honour is refused rather than leaving an object behind.
666
+ const streamHeartbeatMs = resolveStreamHeartbeatMs(config.streamHeartbeatMs);
667
+ const epoch = assertHolderEpoch(options.epoch, 'createKubernetesWorkspace');
668
+ const client = createKubernetesClient(clientAccess(config), clientOptions(config));
384
669
  // The same two egress steps `buildKubernetesBackend` runs for a task
385
670
  // sandbox, repeated here because a workspace never goes through it. The
386
671
  // refusal is synchronous and decided from the policy KIND alone; the
@@ -398,17 +683,77 @@ export async function createKubernetesWorkspace(config, options) {
398
683
  // policy's podSelector matches that label, so a workspace built from a
399
684
  // separate workspace template needs its own policy — the task template's
400
685
  // does not select it. Deliberately not memoized the way the backend's
401
- // once-per-backend check is: creating a workspace is a rare, explicit act
402
- // with nothing to amortise, and re-checking costs one GET.
686
+ // once-per-backend check is: the boundary object is built per create here,
687
+ // so both of its memos live exactly as long as this call — creating a
688
+ // workspace is a rare, explicit act with nothing to amortise, and a policy
689
+ // deleted since the last call has to be noticed. The union half runs
690
+ // further down, against the labels the POST is about to stamp — the pod's
691
+ // own labels on the create path. On the ADOPT path the POST loses to a
692
+ // 409 and those labels are not the standing pod's, so the two a policy
693
+ // selector is built from, the template label and the profile, are checked
694
+ // against the standing object itself instead: see
695
+ // `assertAdoptedWorkspaceMatchesConfig`.
696
+ //
697
+ // Before the boundary is built, because building it resolves the profile:
698
+ // a profile this backend would never emit is a wiring error, and it reads
699
+ // as one when it is refused in its own words rather than out of a policy
700
+ // name.
701
+ assertEgressProfileIsUsable(config.egress);
702
+ const egressBoundary = buildEgressBoundary(client, config, templateName);
403
703
  if (config.egress) {
404
- assertEgressPolicyIsEnforceable(config.egress.policy, config.egress.engine ?? 'core');
405
- await verifyEgressPolicyConfigured(client, namespace, templateName, config.egress, options.signal);
704
+ assertEgressPolicyIsEnforceable(config.egress.policy, config.egress.engine ?? 'core', config.egress.ciliumNarrowing);
705
+ await egressBoundary?.verifyNamedObject(options.signal);
406
706
  }
407
707
  // Read and validate BEFORE anything is created, so a template that cannot
408
708
  // carry a workspace fails with nothing to clean up.
409
709
  const template = await readSandboxTemplate(client, namespace, templateName, options.signal);
410
710
  assertBlockModeWorkspaceDisk(`SandboxTemplate ${templateName} in namespace ${namespace}`, template.podTemplate, template.volumeClaimTemplates);
411
- let created = false;
711
+ // And the other half of the boundary the egress block above calls
712
+ // primary: an ingress policy that actually closes the agent port on the
713
+ // labels this pod will carry. Checked before the POST, so a refusal leaves
714
+ // no Sandbox and no PVC behind — and, on the adopt path, sends no resume
715
+ // patch: a workspace whose port stopped being covered is refused asleep
716
+ // rather than woken up to be refused. `sandboxPodLabels` is the same function the
717
+ // create body stamps its labels with, so the check cannot verify a pod
718
+ // nobody creates. Not memoized, for the reason the egress check above is
719
+ // not: this is a rare, explicit act with nothing to amortise, and a policy
720
+ // deleted since the last call has to be noticed.
721
+ // A workspace is a Sandbox this backend POSTs directly, so an egress
722
+ // PROFILE has to be stamped by this body rather than merged by the
723
+ // controller — and the same map has to reach both the check below and the
724
+ // POST further down, or the check would verify a pod nobody creates. One
725
+ // composer, one call, both call sites. A workspace ADOPTED from a previous
726
+ // create keeps the pod labels it was created with — an ordinary adopt
727
+ // patches no pod template — so the adopt below REFUSES an object whose
728
+ // profile label is not the configured one rather than handing back a pod
729
+ // the verified policy does not select. The way to move an existing
730
+ // workspace onto a new profile is `refreshPodTemplate: true`, which
731
+ // rewrites `/spec/podTemplate` with exactly these labels.
732
+ const profile = egressProfileLabel(config.egress);
733
+ const profilePodLabels = composeAdditionalPodLabels(config.egress);
734
+ const workspacePodLabels = sandboxPodLabels(template, templateName, profilePodLabels);
735
+ const workspaceSubject = `to open workspace ${options.workspaceId} as Sandbox ${name} in namespace ${namespace}`;
736
+ const verifyIngress = buildIngressVerifier(client, config);
737
+ if (verifyIngress !== undefined) {
738
+ await verifyIngress(workspacePodLabels, workspaceSubject, options.signal);
739
+ }
740
+ // And the egress half of the same question, against the same labels: the
741
+ // named object matching exactly says nothing about what a SECOND policy
742
+ // selecting these pods lets out, and a long-lived workspace is the sandbox
743
+ // most likely to be pointed at a network. Checked before the POST for the
744
+ // reason the ingress check is, and on the adopt path before any resume
745
+ // patch — a workspace whose egress stopped being bounded is refused asleep
746
+ // rather than woken up to be refused.
747
+ await egressBoundary?.verifyUnion(workspacePodLabels, workspaceSubject, options.signal);
748
+ // The pod template this call would write, and its revision — built once,
749
+ // from the same overlays `buildSandboxBody` applies below, because the
750
+ // hash has to be taken over exactly the object that lands on the cluster.
751
+ // `profilePodLabels` is one of those overlays: a refresh rewrites
752
+ // `/spec/podTemplate` WHOLE, so one built without them would PATCH the
753
+ // profile label off a workspace that carries it, and the revision stamped
754
+ // by the POST would be taken over a template the POST never wrote.
755
+ const refresh = buildPodTemplateRefresh(template, templateName, config.runtimeClassName, profilePodLabels);
756
+ let adopted;
412
757
  try {
413
758
  await client.request('POST', sandboxCollectionPath(namespace),
414
759
  // No shutdownTime: a workspace carries no expiry — see the module
@@ -418,21 +763,36 @@ export async function createKubernetesWorkspace(config, options) {
418
763
  name,
419
764
  template,
420
765
  sandboxTemplateName: templateName,
766
+ podLabels: profilePodLabels,
767
+ // Stamped by the POST itself rather than by a patch after it,
768
+ // so a workspace created under an epoch is fenced from the
769
+ // moment it exists — there is no window in which the object
770
+ // stands unfenced and a second opener could take it. The
771
+ // pod-template revision rides along for the same reason: an
772
+ // object that recorded what it was built from on its second
773
+ // request would have a window in which it claimed none.
774
+ annotations: {
775
+ ...(epoch !== undefined ? { [HOLDER_EPOCH_ANNOTATION_KEY]: String(epoch) } : {}),
776
+ [POD_TEMPLATE_HASH_ANNOTATION_KEY]: refresh.hash,
777
+ },
421
778
  ...(config.runtimeClassName !== undefined
422
779
  ? { runtimeClassName: config.runtimeClassName }
423
780
  : {}),
424
781
  }), options.signal);
425
- created = true;
426
782
  }
427
783
  catch (err) {
428
784
  if (!(err instanceof KubernetesConflictError))
429
785
  throw err;
430
- await adoptExistingWorkspace(client, namespace, name, {
786
+ adopted = await adoptExistingWorkspace(client, namespace, name, options.workspaceId, {
431
787
  sandboxTemplateName: templateName,
788
+ // The same resolution the create body and the policy selector
789
+ // were built from, so the object is checked against exactly
790
+ // the label this call would have stamped.
791
+ ...(profile !== undefined ? { profile } : {}),
432
792
  ...(config.runtimeClassName !== undefined
433
793
  ? { runtimeClassName: config.runtimeClassName }
434
794
  : {}),
435
- }, options.signal);
795
+ }, epoch, options.refreshPodTemplate === true ? refresh : undefined, options.signal);
436
796
  }
437
797
  return await openWorkspaceHandle({
438
798
  client,
@@ -441,8 +801,46 @@ export async function createKubernetesWorkspace(config, options) {
441
801
  workspaceId: options.workspaceId,
442
802
  rootDir: options.workingDirectory,
443
803
  agentPort,
804
+ agentAddress,
805
+ streamHeartbeatMs,
444
806
  readiness,
445
- created,
807
+ origin: adopted === undefined ? 'created' : adopted.resumed ? 'resumed' : 'adopted-running',
808
+ sandboxTemplateName: templateName,
809
+ ...(config.runtimeClassName !== undefined ? { runtimeClassName: config.runtimeClassName } : {}),
810
+ // Carried for the reason the template name and the RuntimeClass are:
811
+ // a `resume({ refreshPodTemplate: true })` rebuilds the pod template
812
+ // long after this call, and a refresh rebuilt without these labels
813
+ // would patch the egress profile off the very pod the configured
814
+ // policy selects.
815
+ ...(Object.keys(profilePodLabels).length > 0 ? { podLabels: profilePodLabels } : {}),
816
+ // A create wrote the revision it just computed; an adopt reports
817
+ // whatever the standing object carries, which is `undefined` for one
818
+ // created before the annotation existed.
819
+ ...(adopted === undefined
820
+ ? { templateRevision: refresh.hash }
821
+ : adopted.templateRevision !== undefined
822
+ ? { templateRevision: adopted.templateRevision }
823
+ : {}),
824
+ currentTemplateHash: refresh.hash,
825
+ ...(adopted?.awaitReplacement === true ? { awaitReplacement: true } : {}),
826
+ ...(epoch !== undefined ? { epoch } : {}),
827
+ ...(adopted?.drainingPodUid !== undefined ? { drainingPodUid: adopted.drainingPodUid } : {}),
828
+ ...(options.onStartFailure !== undefined ? { onStartFailure: options.onStartFailure } : {}),
829
+ ...(options.onCancellationUnconfirmed !== undefined
830
+ ? { onCancellationUnconfirmed: options.onCancellationUnconfirmed }
831
+ : {}),
832
+ ...(options.onQuiesceUnsupported !== undefined
833
+ ? { onQuiesceUnsupported: options.onQuiesceUnsupported }
834
+ : {}),
835
+ ...(options.onQuiesceNarrowed !== undefined
836
+ ? { onQuiesceNarrowed: options.onQuiesceNarrowed }
837
+ : {}),
838
+ ...(options.onFlushUnsupported !== undefined
839
+ ? { onFlushUnsupported: options.onFlushUnsupported }
840
+ : {}),
841
+ ...(options.onFlushUnreachable !== undefined
842
+ ? { onFlushUnreachable: options.onFlushUnreachable }
843
+ : {}),
446
844
  ...(options.signal !== undefined ? { signal: options.signal } : {}),
447
845
  });
448
846
  }
@@ -454,27 +852,743 @@ export async function createKubernetesWorkspace(config, options) {
454
852
  * Everything checked here is checked against the object, because on this path
455
853
  * the object is not the one this call built. A create POSTs its own body and
456
854
  * knows what is in it; an adopt is handed a pod somebody else's process, or
457
- * last month's configuration, decided the shape of. The two silent losses are
458
- * the template label and the RuntimeClass — see
459
- * {@link KubernetesWorkspaceMismatchError}.
855
+ * last month's configuration, decided the shape of. The silent losses are the
856
+ * template label, the egress profile label when one is configured, and the
857
+ * RuntimeClass — see {@link KubernetesWorkspaceMismatchError}.
460
858
  *
461
859
  * The refusals all happen BEFORE the resume patch: an object this call will
462
860
  * not use is not woken up on the way to being rejected.
861
+ *
862
+ * What it OBSERVES, and hands back, is whether the pod this workspace is
863
+ * about to be bound to is a pod that already exists. Both answers here mean
864
+ * it is not: an object found `Suspended` has had its pod taken away, and a
865
+ * pod already carrying a `deletionTimestamp` is one the controller is in the
866
+ * middle of taking away. Either way the pod this handle will serve has not
867
+ * been created yet, and the first bind attempt has to WAIT for it rather than
868
+ * fail on the read that finds no live pod — see {@link PodBindPolicy}.
463
869
  */
464
- async function adoptExistingWorkspace(client, namespace, name, expected, signal) {
465
- const existing = await client.request('GET', sandboxPath(namespace, name), undefined, signal);
870
+ async function adoptExistingWorkspace(client, namespace, name, workspaceId, expected, epoch, refresh, signal) {
871
+ const target = { client, namespace, name, workspaceId };
872
+ const existing = await readSandboxObject(target, signal);
466
873
  assertBlockModeWorkspaceDisk(`Sandbox ${name} in namespace ${namespace}`, existing?.spec?.podTemplate, existing?.spec?.volumeClaimTemplates);
467
- assertAdoptedWorkspaceMatchesConfig(name, namespace, existing?.spec?.podTemplate, expected);
468
- if (existing?.spec?.operatingMode === 'Suspended') {
469
- await client.request('PATCH', sandboxPath(namespace, name), RESUME_PATCH, signal);
874
+ const suspended = operatingModeOf(existing) === 'Suspended';
875
+ // A refresh is a Suspended→Running write and nothing else, so an object
876
+ // found Running is adopted exactly as it always was — see
877
+ // {@link KubernetesWorkspaceTransitionOptions.refreshPodTemplate}.
878
+ const refreshing = refresh !== undefined && suspended;
879
+ assertAdoptedWorkspaceMatchesConfig(name, namespace, existing?.spec?.podTemplate, {
880
+ sandboxTemplateName: expected.sandboxTemplateName,
881
+ // The RuntimeClass and egress-profile refusals are lifted for a call
882
+ // that is ABOUT TO WRITE them, and only for as long as that stays
883
+ // true: if the patch below does not land, both are re-applied against
884
+ // the object as it then stands, so a pod on the wrong runtime — or
885
+ // under a profile no policy selects — is never bound. They are lifted
886
+ // together because a refresh writes them together: the patched pod
887
+ // template is `sandboxPodTemplate`'s, overlays and all, so the
888
+ // profile label lands with the class. The TEMPLATE refusal is never
889
+ // lifted — a refresh rewrites a workspace's pod spec, it never moves
890
+ // the workspace to another template.
891
+ ...(refreshing
892
+ ? {}
893
+ : {
894
+ ...(expected.profile !== undefined ? { profile: expected.profile } : {}),
895
+ ...(expected.runtimeClassName !== undefined
896
+ ? { runtimeClassName: expected.runtimeClassName }
897
+ : {}),
898
+ }),
899
+ });
900
+ const reading = readHolderEpoch(existing?.metadata);
901
+ // With the adopt's other refusals, and for the same reason: an object this
902
+ // call will not use is not woken up on the way to being rejected. A
903
+ // superseded opener patches nothing and starts no pod.
904
+ if (epoch !== undefined)
905
+ assertHolderEpochAllows(target, 'createKubernetesWorkspace', epoch, reading);
906
+ if (refreshing && refresh !== undefined) {
907
+ // Both BEFORE the patch, and both against the SANDBOX's
908
+ // volumeClaimTemplates rather than the template's: those are
909
+ // CEL-immutable and are not in the patch, so the disks a refresh can
910
+ // be applied to are the disks this workspace already has, and the two
911
+ // questions worth asking are whether the template still claims them
912
+ // and whether it claims anything else.
913
+ const source = `SandboxTemplate ${expected.sandboxTemplateName} in namespace ${namespace}, refreshing Sandbox ${name}`;
914
+ assertRefreshedTemplateClaimsTheSameDisks(source, refresh.volumeClaimTemplates, existing?.spec?.volumeClaimTemplates);
915
+ // A template that stopped claiming this workspace's disk would come up
916
+ // healthy with the disk attached to nothing, and the only symptom
917
+ // would be that yesterday's files are gone.
918
+ assertBlockModeWorkspaceDisk(source, refresh.podTemplate, existing?.spec?.volumeClaimTemplates);
919
+ }
920
+ let templateRevision = readPodTemplateHash(existing?.metadata);
921
+ // Read BEFORE the resume patch, so what is recorded is the state this
922
+ // adopt WALKED INTO rather than one it provoked. It costs one GET on a
923
+ // path that is a rare, explicit act with nothing to amortise — the same
924
+ // trade the egress verification above makes — and it buys the one fact
925
+ // nothing else on this path can supply: whether the pod standing under
926
+ // this name is on its way out.
927
+ const drainingPodUid = await readDrainingPodUid(client, namespace, name, signal);
928
+ let resumed = suspended;
929
+ /** Somebody else's resume beat this call's patch — see `awaitReplacement`. */
930
+ let racedResume = false;
931
+ if (refreshing && refresh !== undefined) {
932
+ const result = await writeRefreshedPodTemplate(target, 'createKubernetesWorkspace', refresh, epoch, reading, signal);
933
+ if (result.applied) {
934
+ templateRevision = refresh.hash;
935
+ }
936
+ else {
937
+ // Another process resumed it first: nothing was written, so this
938
+ // is an adopt of a Running object and behaves like one — the
939
+ // RuntimeClass refusal applies again, against the object as it
940
+ // now stands, and the pod that is there is bound unchanged.
941
+ resumed = false;
942
+ racedResume = true;
943
+ assertAdoptedWorkspaceMatchesConfig(name, namespace, result.sandbox.spec?.podTemplate, expected);
944
+ templateRevision = readPodTemplateHash(result.sandbox.metadata);
945
+ if (epoch !== undefined) {
946
+ await writeOperatingMode(target, 'createKubernetesWorkspace', undefined, epoch, signal, readHolderEpoch(result.sandbox.metadata));
947
+ }
948
+ }
949
+ }
950
+ else if (resumed) {
951
+ await writeOperatingMode(target, 'createKubernetesWorkspace', 'Running', epoch, signal, reading);
952
+ }
953
+ else if (epoch !== undefined) {
954
+ // The one write this path did not use to make. Adopting a RUNNING
955
+ // workspace sent nothing at all, which is precisely what left race 1
956
+ // open: the new holder took the workspace over and left no trace on
957
+ // the object, so a superseded holder's later suspend had nothing to
958
+ // be refused by. Taking a workspace over is the moment the fence
959
+ // moves, whether or not the mode moves with it.
960
+ await writeOperatingMode(target, 'createKubernetesWorkspace', undefined, epoch, signal, reading);
961
+ }
962
+ return {
963
+ resumed,
964
+ ...(templateRevision !== undefined ? { templateRevision } : {}),
965
+ ...(racedResume ? { awaitReplacement: true } : {}),
966
+ ...(drainingPodUid !== undefined ? { drainingPodUid } : {}),
967
+ };
968
+ }
969
+ /**
970
+ * The uid of the pod standing under this Sandbox's name IF it is draining,
971
+ * and `undefined` for every other answer — no pod, or a pod that is fine.
972
+ *
973
+ * One GET by the Sandbox's name, the same convention
974
+ * {@link readBoundPod}'s fast path uses: the pod is named after its
975
+ * Sandbox in the controller this backend targets. No selector fallback,
976
+ * because the cost of missing a draining pod here is one adopt that fails the
977
+ * way it does today rather than a wrong answer, while a list on every adopt
978
+ * is a round trip paid by every caller for a state most of them are not in.
979
+ *
980
+ * A 404 is "no pod", which is the ordinary answer on a suspended workspace.
981
+ * Anything else is rethrown: a host that cannot read pods cannot read a bind
982
+ * token either, and hearing it here names the real problem.
983
+ */
984
+ async function readDrainingPodUid(client, namespace, name, signal) {
985
+ let pod;
986
+ try {
987
+ pod = await client.request('GET', podPath(namespace, name), undefined, signal);
988
+ }
989
+ catch (err) {
990
+ if (err instanceof KubernetesAlreadyGoneError)
991
+ return undefined;
992
+ throw err;
993
+ }
994
+ // `!= null` for the same reason `isPodLive` uses it: an explicit JSON null
995
+ // must not read as "terminating".
996
+ if (pod?.metadata?.deletionTimestamp == null)
997
+ return undefined;
998
+ const uid = pod.metadata.uid;
999
+ return typeof uid === 'string' && uid !== '' ? uid : undefined;
1000
+ }
1001
+ /**
1002
+ * The UNFENCED operating-mode merge patch — what this module sends when a
1003
+ * call carries no holder epoch, which is every call that carried none before
1004
+ * epochs existed, byte for byte and content type included. The fenced form is
1005
+ * `objects.ts`'s `buildHolderEpochPatch`; both go out through
1006
+ * {@link writeOperatingMode} and nothing else builds either.
1007
+ *
1008
+ * Built per call rather than held as constants because each one stamps
1009
+ * {@link OPERATING_MODE_CHANGED_AT_ANNOTATION_KEY} with the moment it was
1010
+ * sent. That annotation is what
1011
+ * {@link listKubernetesWorkspaces} reports as `operatingModeChangedAt`, and
1012
+ * the patch is the only place the fact exists: the controller's `Suspended`
1013
+ * condition lingers True across a resume, so neither its presence nor its
1014
+ * `lastTransitionTime` can be read as "when did this change" — see the
1015
+ * annotation's own comment.
1016
+ *
1017
+ * The body stays otherwise minimal, and a JSON merge patch (RFC 7386, which
1018
+ * is what `k8s-client.ts` sends) recurses into `metadata.annotations` rather
1019
+ * than replacing the map, so a Sandbox carrying annotations somebody else put
1020
+ * there keeps them.
1021
+ */
1022
+ function operatingModePatch(mode) {
1023
+ return {
1024
+ metadata: {
1025
+ annotations: { [OPERATING_MODE_CHANGED_AT_ANNOTATION_KEY]: new Date().toISOString() },
1026
+ },
1027
+ spec: { operatingMode: mode },
1028
+ };
1029
+ }
1030
+ /**
1031
+ * How many times one fenced write may be re-read and re-sent before it gives
1032
+ * up.
1033
+ *
1034
+ * A failed `test` is only worth retrying when the value the patch tested has
1035
+ * actually MOVED, and the loop checks that on every attempt (see
1036
+ * {@link writeOperatingMode}), so this is not a "retry until it works" budget
1037
+ * — it is the ceiling on how many times a workspace may legitimately change
1038
+ * underneath one call. Three is one more than the case that really happens: a
1039
+ * controller status write moving `resourceVersion` between the read and the
1040
+ * write of a workspace that carries no epoch annotation yet, which needs one
1041
+ * re-read and then succeeds.
1042
+ */
1043
+ const HOLDER_EPOCH_WRITE_ATTEMPTS = 3;
1044
+ /** One GET of the Sandbox, which is where every condition is read from. */
1045
+ async function readSandboxObject(target, signal) {
1046
+ return await target.client.request('GET', sandboxPath(target.namespace, target.name), undefined, signal);
1047
+ }
1048
+ /** `spec.operatingMode`, with the CRD's own default for an absent one. */
1049
+ function operatingModeOf(sandbox) {
1050
+ return sandbox?.spec?.operatingMode === 'Suspended' ? 'Suspended' : 'Running';
1051
+ }
1052
+ /** Refuse a write the stored epoch has moved past. Changes nothing anywhere. */
1053
+ function assertHolderEpochAllows(target, operation, epoch, reading) {
1054
+ if (holderEpochAllows(reading, epoch))
1055
+ return;
1056
+ throw new KubernetesWorkspacePreconditionError(operation, target.workspaceId, target.name, epoch, reading.epoch, reading.annotation);
1057
+ }
1058
+ /**
1059
+ * Read the object and refuse the write if this caller has been superseded,
1060
+ * BEFORE the caller does anything it cannot undo.
1061
+ *
1062
+ * The refusal it raises is the same one the write itself would raise; what
1063
+ * this buys is WHEN. `suspend()` kills every terminal it handed out and
1064
+ * `destroy()` tears its session down, both before any request goes out, so a
1065
+ * superseded holder that found out from the write would have taken its own
1066
+ * caller's sessions away on a write that never applied. The reading it hands
1067
+ * back is then used as the first attempt's condition, so the gate costs no
1068
+ * extra round trip.
1069
+ *
1070
+ * With no epoch there is nothing to check and nothing is read: the request
1071
+ * log of an unfenced call is what it always was.
1072
+ */
1073
+ async function readHolderEpochGate(target, operation, epoch, signal) {
1074
+ if (epoch === undefined)
1075
+ return undefined;
1076
+ const reading = readHolderEpoch((await readSandboxObject(target, signal))?.metadata);
1077
+ assertHolderEpochAllows(target, operation, epoch, reading);
1078
+ return reading;
1079
+ }
1080
+ /**
1081
+ * The single send path for every `spec.operatingMode` write this backend
1082
+ * makes, fenced or not.
1083
+ *
1084
+ * Unfenced (`epoch === undefined`) it sends the merge patch it always sent,
1085
+ * content type included, and a `mode` of `undefined` sends nothing at all —
1086
+ * there is no such thing as an unfenced write with no mutation in it.
1087
+ *
1088
+ * Fenced, it is one request: {@link buildHolderEpochPatch} puts the `test`
1089
+ * and the mutation in the same body, so nothing can fit between them. The
1090
+ * loop around it is NOT a retry-until-it-works — the API server answers every
1091
+ * unapplied JSON patch with the same opaque 422 whether the `test` failed or
1092
+ * the body was wrong (see {@link KubernetesPatchNotAppliedError}), so the
1093
+ * only honest way to tell them apart is to look: re-read the object, and if
1094
+ * the value this patch tested is still exactly what it tested, the patch did
1095
+ * not lose a race and is simply wrong, so the error stands. If it HAS moved,
1096
+ * the new reading is checked against this call's epoch like any other — a
1097
+ * holder that overtook this one is refused, and a controller status write
1098
+ * that only moved `resourceVersion` is retried on the fresh read.
1099
+ *
1100
+ * `mode` omitted with an epoch present is the stamp-only write: take the
1101
+ * workspace over without changing what it is doing. It deliberately does not
1102
+ * stamp {@link OPERATING_MODE_CHANGED_AT_ANNOTATION_KEY} — the mode did not
1103
+ * change, and that annotation is an inventory column that has to stay true.
1104
+ */
1105
+ async function writeOperatingMode(target, operation, mode, epoch, signal, gate) {
1106
+ const path = sandboxPath(target.namespace, target.name);
1107
+ if (epoch === undefined) {
1108
+ if (mode === undefined)
1109
+ return;
1110
+ await target.client.request('PATCH', path, operatingModePatch(mode), signal);
1111
+ return;
1112
+ }
1113
+ let reading = gate ?? readHolderEpoch((await readSandboxObject(target, signal))?.metadata);
1114
+ for (let attempt = 1;; attempt += 1) {
1115
+ assertHolderEpochAllows(target, operation, epoch, reading);
1116
+ const patch = buildHolderEpochPatch({
1117
+ reading,
1118
+ epoch,
1119
+ ...(mode !== undefined
1120
+ ? { operatingMode: mode, operatingModeChangedAt: new Date().toISOString() }
1121
+ : {}),
1122
+ });
1123
+ try {
1124
+ await target.client.request('PATCH', path, patch, signal, 'json');
1125
+ return;
1126
+ }
1127
+ catch (err) {
1128
+ if (!(err instanceof KubernetesPatchNotAppliedError))
1129
+ throw err;
1130
+ if (attempt >= HOLDER_EPOCH_WRITE_ATTEMPTS)
1131
+ throw err;
1132
+ const next = readHolderEpoch((await readSandboxObject(target, signal))?.metadata);
1133
+ const unmoved = next.annotation === reading.annotation &&
1134
+ next.resourceVersion === reading.resourceVersion &&
1135
+ next.hasAnnotations === reading.hasAnnotations;
1136
+ if (unmoved)
1137
+ throw err;
1138
+ reading = next;
1139
+ }
1140
+ }
1141
+ }
1142
+ /**
1143
+ * The overlays of `buildSandboxBody`'s create body, plus their revision.
1144
+ *
1145
+ * `podLabels` is `composeAdditionalPodLabels`'s map — the egress profile
1146
+ * today — and it is an overlay like the other two rather than an extra: the
1147
+ * patch this feeds replaces `/spec/podTemplate` whole, so a refresh that left
1148
+ * it out would REMOVE the profile label from a workspace created with it. The
1149
+ * replacement pod would come up selected by no per-profile policy, on a path
1150
+ * where nothing re-reads the label, and the hash would disagree with what
1151
+ * every create POSTs, so `templateCurrent` would report drift that no refresh
1152
+ * could clear.
1153
+ */
1154
+ function buildPodTemplateRefresh(template, sandboxTemplateName, runtimeClassName, podLabels) {
1155
+ const podTemplate = sandboxPodTemplate(template, sandboxTemplateName, runtimeClassName, podLabels);
1156
+ return {
1157
+ podTemplate,
1158
+ hash: podTemplateHash(podTemplate),
1159
+ ...(template.volumeClaimTemplates !== undefined
1160
+ ? { volumeClaimTemplates: template.volumeClaimTemplates }
1161
+ : {}),
1162
+ };
1163
+ }
1164
+ /**
1165
+ * The Suspended→Running write that also rewrites `spec.podTemplate`: ONE
1166
+ * conditional patch, built by the one builder this backend has.
1167
+ *
1168
+ * `test /spec/operatingMode == "Suspended"` is what enforces "only on a
1169
+ * suspended workspace" — not the read that preceded it, which another
1170
+ * process's resume can invalidate in the time it takes to send this. When the
1171
+ * caller also carries a holder epoch, that clause rides in the SAME body
1172
+ * (`buildHolderEpochPatch`'s `tests` seam), so the two conditions are one
1173
+ * request and there is no window between them.
1174
+ *
1175
+ * The discrimination is the one W8 measured and has to be repeated here
1176
+ * rather than shared with {@link writeOperatingMode}, because the two want
1177
+ * opposite things from the same 422: a real API server answers an unapplied
1178
+ * JSON patch identically whether a `test` failed or the body was malformed,
1179
+ * so the only way to tell is to re-read the object.
1180
+ *
1181
+ * - the object is GONE ⇒ the re-read 404s and says so, and that error is
1182
+ * what the caller hears. Neither clause can be said to have refused the
1183
+ * patch, and there is no pod to fall back to.
1184
+ * - the mode is no longer `Suspended` ⇒ the mode clause is what refused it.
1185
+ * That is an outcome, not an error: report it, after checking that this
1186
+ * caller has not ALSO been superseded, which is a refusal.
1187
+ * - the mode is still `Suspended` and the call carries no epoch ⇒ the only
1188
+ * condition in the body was true, so the body is what the server refused.
1189
+ * Nothing to retry.
1190
+ * - otherwise the epoch clause is the suspect: a stored epoch above this
1191
+ * caller's refuses it by name, one that merely moved is retried on the
1192
+ * fresh reading, and an object that did not move at all means the body is
1193
+ * wrong and the error stands.
1194
+ */
1195
+ async function writeRefreshedPodTemplate(target, operation, refresh, epoch, gate, signal) {
1196
+ const path = sandboxPath(target.namespace, target.name);
1197
+ let reading = gate;
1198
+ for (let attempt = 1;; attempt += 1) {
1199
+ if (epoch !== undefined)
1200
+ assertHolderEpochAllows(target, operation, epoch, reading);
1201
+ const patch = buildHolderEpochPatch({
1202
+ reading,
1203
+ ...(epoch !== undefined ? { epoch } : {}),
1204
+ tests: [{ op: 'test', path: '/spec/operatingMode', value: 'Suspended' }],
1205
+ annotations: { [POD_TEMPLATE_HASH_ANNOTATION_KEY]: refresh.hash },
1206
+ podTemplate: refresh.podTemplate,
1207
+ operatingMode: 'Running',
1208
+ operatingModeChangedAt: new Date().toISOString(),
1209
+ });
1210
+ try {
1211
+ await target.client.request('PATCH', path, patch, signal, 'json');
1212
+ return { applied: true };
1213
+ }
1214
+ catch (err) {
1215
+ if (!(err instanceof KubernetesPatchNotAppliedError))
1216
+ throw err;
1217
+ const sandbox = await readSandboxObject(target, signal);
1218
+ // An object DELETED in this window does not arrive here as
1219
+ // `undefined`: the GET 404s and `readSandboxObject` throws
1220
+ // `KubernetesAlreadyGoneError`, which is the truthful answer and is
1221
+ // let through. `undefined` is the API server answering 200 with no
1222
+ // body — a shape this code has no reading of. It must not fall
1223
+ // through, because `operatingModeOf(undefined)` is `'Running'` and
1224
+ // the branch below means "somebody else's pod is standing under
1225
+ // this name", which would send the caller on to bind a pod nothing
1226
+ // established was there, after refusals made against an object
1227
+ // nobody read. The original error says the one thing that is known
1228
+ // — the patch did not apply and nothing changed.
1229
+ if (sandbox === undefined)
1230
+ throw err;
1231
+ const next = readHolderEpoch(sandbox.metadata);
1232
+ if (operatingModeOf(sandbox) !== 'Suspended') {
1233
+ // Somebody resumed it first. If they also took the workspace
1234
+ // over, this caller hears THAT rather than being handed a pod
1235
+ // it is no longer entitled to bind.
1236
+ if (epoch !== undefined)
1237
+ assertHolderEpochAllows(target, operation, epoch, next);
1238
+ return { applied: false, sandbox };
1239
+ }
1240
+ if (epoch === undefined)
1241
+ throw err;
1242
+ if (attempt >= HOLDER_EPOCH_WRITE_ATTEMPTS)
1243
+ throw err;
1244
+ const unmoved = next.annotation === reading.annotation &&
1245
+ next.resourceVersion === reading.resourceVersion &&
1246
+ next.hasAnnotations === reading.hasAnnotations;
1247
+ if (unmoved)
1248
+ throw err;
1249
+ reading = next;
1250
+ }
1251
+ }
1252
+ }
1253
+ /**
1254
+ * DELETE the Sandbox, fenced by the epoch when the caller supplied one.
1255
+ *
1256
+ * The condition here cannot be a JSON Patch `test`, because a DELETE has no
1257
+ * patch body — so it is `preconditions.resourceVersion` on the very version
1258
+ * whose epoch was just read, which the API server answers with a 409 naming
1259
+ * both versions when it no longer matches (measured). Anything that touched
1260
+ * the object between the read and the DELETE — another holder raising the
1261
+ * epoch included — moves that version, so the DELETE is refused and re-read
1262
+ * rather than taking a disk on a stale view.
1263
+ *
1264
+ * Unfenced it is the bodyless DELETE it always was. Already gone counts as
1265
+ * deleted either way, that being the state DELETE was asking for.
1266
+ */
1267
+ async function deleteSandboxObject(target, operation, epoch, signal, gate) {
1268
+ const path = sandboxPath(target.namespace, target.name);
1269
+ if (epoch === undefined) {
1270
+ try {
1271
+ await target.client.request('DELETE', path, undefined, signal);
1272
+ }
1273
+ catch (err) {
1274
+ if (!(err instanceof KubernetesAlreadyGoneError))
1275
+ throw err;
1276
+ }
1277
+ return;
1278
+ }
1279
+ let reading = gate;
1280
+ for (let attempt = 1;; attempt += 1) {
1281
+ if (reading === undefined) {
1282
+ let sandbox;
1283
+ try {
1284
+ sandbox = await readSandboxObject(target, signal);
1285
+ }
1286
+ catch (err) {
1287
+ if (err instanceof KubernetesAlreadyGoneError)
1288
+ return;
1289
+ throw err;
1290
+ }
1291
+ reading = readHolderEpoch(sandbox?.metadata);
1292
+ }
1293
+ assertHolderEpochAllows(target, operation, epoch, reading);
1294
+ if (reading.resourceVersion === undefined) {
1295
+ // Unreachable against a real API server, which sets it on every
1296
+ // object it serves — and a bodyless DELETE here would be an
1297
+ // UNFENCED delete of somebody's disk, which is the one thing this
1298
+ // path may never silently become.
1299
+ throw new Error(`kubernetes: cannot delete workspace ${target.workspaceId} (Sandbox ${target.name}) under holder epoch ${epoch}: the object came back with no metadata.resourceVersion, so there is no precondition to send and the DELETE would be unconditional.`);
1300
+ }
1301
+ try {
1302
+ await target.client.request('DELETE', path, {
1303
+ apiVersion: 'v1',
1304
+ kind: 'DeleteOptions',
1305
+ preconditions: { resourceVersion: reading.resourceVersion },
1306
+ }, signal);
1307
+ return;
1308
+ }
1309
+ catch (err) {
1310
+ if (err instanceof KubernetesAlreadyGoneError)
1311
+ return;
1312
+ if (!(err instanceof KubernetesConflictError))
1313
+ throw err;
1314
+ if (attempt >= HOLDER_EPOCH_WRITE_ATTEMPTS)
1315
+ throw err;
1316
+ reading = undefined;
1317
+ }
1318
+ }
1319
+ }
1320
+ /** Shared tail of every bind timeout but the resume-specific one. */
1321
+ const NO_BINDABLE_POD_ADVICE = "A pod's uid is the agent's bind token, so a pod carrying a deletionTimestamp — or one in a terminal phase — is never bound to: its uid is a token the pod's replacement will refuse. The Ready condition cannot be waited on instead, because the controller leaves it standing across a transition. Raise readyTimeoutMs, or look at why the controller has not brought a pod up.";
1322
+ /**
1323
+ * True once the pod has actually stopped. The POD is asked, and nothing else
1324
+ * is consulted or believed.
1325
+ *
1326
+ * The cheap-looking alternative — the Sandbox's own `Suspended` condition —
1327
+ * is unusable, and upstream says so itself: "the controller does not
1328
+ * currently remove this condition when the Sandbox is resumed, so a stale
1329
+ * Suspended condition may linger after operatingMode returns to Running.
1330
+ * Consumers should treat Ready as the authoritative signal and not infer the
1331
+ * live operating state from the mere presence of this condition"
1332
+ * (`sandbox_types.go`). Reading it would make the second and every later
1333
+ * suspend of the same workspace return immediately, on a True left behind by
1334
+ * the previous one, while the guest was still running and still writing to
1335
+ * the caller's disk — and a `resume()` issued straight after such a false
1336
+ * suspend could bind to the pod that is about to be deleted.
1337
+ *
1338
+ * A `deletionTimestamp` is not the answer either: it is set the moment the
1339
+ * DELETE is accepted, and the container goes on running until it exits or
1340
+ * `terminationGracePeriodSeconds` expires. Gone (404) or stopped
1341
+ * ({@link isPodStopped}) — those are the only two states that mean the disk
1342
+ * is quiesced.
1343
+ */
1344
+ async function isPodRetired(client, namespace, name, signal) {
1345
+ try {
1346
+ const pod = await client.request('GET', podPath(namespace, name), undefined, signal);
1347
+ return isPodStopped(pod);
1348
+ }
1349
+ catch (err) {
1350
+ if (err instanceof KubernetesAlreadyGoneError)
1351
+ return true;
1352
+ throw err;
1353
+ }
1354
+ }
1355
+ /**
1356
+ * Poll {@link isPodRetired} until it answers true, or refuse with
1357
+ * {@link KubernetesWorkspaceSuspendTimeoutError}.
1358
+ *
1359
+ * At module scope, and taking its subject as arguments, because BOTH suspend
1360
+ * paths owe the caller the same wait: the handle's `suspend()`, which has a
1361
+ * session to tear down first, and {@link suspendKubernetesWorkspace}, which
1362
+ * has no handle at all. A suspend that resolved on the patch alone would
1363
+ * promise a quiesced disk it had not waited for, and that promise must not
1364
+ * depend on which entry point was used.
1365
+ */
1366
+ async function awaitPodRetired(client, namespace, name, workspaceId, readiness, signal) {
1367
+ const deadline = new OperationDeadline(readiness.timeoutMs, `kubernetes workspace ${name} suspend`, signal);
1368
+ while (deadline.remainingMs() > 0) {
1369
+ try {
1370
+ if (await deadline.run(async (pollSignal) => await isPodRetired(client, namespace, name, pollSignal)))
1371
+ return;
1372
+ await deadline.delay(readiness.pollIntervalMs);
1373
+ }
1374
+ catch (err) {
1375
+ if (err instanceof OperationDeadlineExpired)
1376
+ break;
1377
+ throw err;
1378
+ }
1379
+ }
1380
+ throw new KubernetesWorkspaceSuspendTimeoutError(workspaceId, name, readiness.timeoutMs);
1381
+ }
1382
+ /**
1383
+ * What `spec.operatingMode` says RIGHT NOW, read off the object and nothing
1384
+ * else.
1385
+ *
1386
+ * `Running` when the field is absent, which is the CRD's own default. Every
1387
+ * Sandbox this backend creates sets it explicitly, so an absent value means
1388
+ * an object somebody else made — and the API's default for it is Running.
1389
+ */
1390
+ async function readOperatingMode(client, namespace, name, signal) {
1391
+ const sandbox = await client.request('GET', sandboxPath(namespace, name), undefined, signal);
1392
+ return operatingModeOf(sandbox);
1393
+ }
1394
+ /**
1395
+ * Every workspace this backend owns in the namespace, WITHOUT waking one.
1396
+ *
1397
+ * The verb retention needs. Deleting a workspace nobody has resumed since
1398
+ * last month, or reporting what a namespace is holding, used to require
1399
+ * {@link createKubernetesWorkspace} — which adopts AND resumes: the inventory
1400
+ * pass would start a pod for every suspended workspace it looked at, probe
1401
+ * each one, and then have to put them all back. This issues exactly one GET
1402
+ * of the sandboxes collection and reads the objects.
1403
+ *
1404
+ * ## What counts as a workspace
1405
+ *
1406
+ * Two things together, and both are this backend's own marks:
1407
+ *
1408
+ * - the `namzu-ws-` name prefix — {@link workspaceSandboxName}'s contract,
1409
+ * and the only thing that makes a `workspaceId` recoverable from a
1410
+ * Sandbox at all;
1411
+ * - {@link SANDBOX_TEMPLATE_LABEL_KEY} on `spec.podTemplate.metadata.labels`,
1412
+ * which says this object was built by this backend and names the template
1413
+ * it came from.
1414
+ *
1415
+ * The filtering happens HERE rather than in a `labelSelector` on the request,
1416
+ * and the reason is where that label lives. It is a POD label — the one an
1417
+ * egress `NetworkPolicy`'s `podSelector` matches — written onto
1418
+ * `spec.podTemplate`, while a `labelSelector` on the sandboxes collection
1419
+ * matches the Sandbox's OWN `metadata.labels`, which this backend has never
1420
+ * written. Stamping a second copy up there to make a server-side selector
1421
+ * work would leave every workspace created before that change invisible to
1422
+ * this call, and an inventory that silently omits the oldest objects is worse
1423
+ * than no inventory at all — those are exactly the ones a retention pass is
1424
+ * looking for.
1425
+ *
1426
+ * ## What it never does
1427
+ *
1428
+ * No PATCH, no DELETE, no pod read, no dial. A workspace that was suspended
1429
+ * before this call is suspended after it, and a running one is untouched. The
1430
+ * order is the API server's own (name order); the caller sorts if it cares.
1431
+ */
1432
+ export async function listKubernetesWorkspaces(config, options) {
1433
+ options?.signal?.throwIfAborted();
1434
+ const namespace = config.namespace;
1435
+ const client = createKubernetesClient(clientAccess(config), clientOptions(config));
1436
+ const list = await client.request('GET', sandboxCollectionPath(namespace), undefined, options?.signal);
1437
+ const summaries = [];
1438
+ for (const sandbox of list?.items ?? []) {
1439
+ const name = sandbox?.metadata?.name;
1440
+ if (typeof name !== 'string' || !name.startsWith(WORKSPACE_NAME_PREFIX))
1441
+ continue;
1442
+ const template = sandbox?.spec?.podTemplate?.metadata?.labels?.[SANDBOX_TEMPLATE_LABEL_KEY];
1443
+ if (typeof template !== 'string' || template === '')
1444
+ continue;
1445
+ const createdAt = sandbox?.metadata?.creationTimestamp;
1446
+ const changedAt = sandbox?.metadata?.annotations?.[OPERATING_MODE_CHANGED_AT_ANNOTATION_KEY];
1447
+ const holder = readHolderEpoch(sandbox?.metadata);
1448
+ summaries.push({
1449
+ workspaceId: name.slice(WORKSPACE_NAME_PREFIX.length),
1450
+ operatingMode: operatingModeOf(sandbox),
1451
+ template,
1452
+ ...(typeof createdAt === 'string' && createdAt !== '' ? { createdAt } : {}),
1453
+ ...(typeof changedAt === 'string' && changedAt !== ''
1454
+ ? { operatingModeChangedAt: changedAt }
1455
+ : {}),
1456
+ ...(holder.epoch !== undefined ? { holderEpoch: holder.epoch } : {}),
1457
+ });
470
1458
  }
1459
+ return summaries;
1460
+ }
1461
+ /**
1462
+ * DELETE a workspace by id, without ever adopting or resuming it.
1463
+ *
1464
+ * The same thing `destroy({ deleteDisk: true })` does to the cluster — one
1465
+ * DELETE of the Sandbox, which cascades to the Pod, the Service and the PVC
1466
+ * through ownerReferences — with the same two guarantees: an object already
1467
+ * gone counts as deleted, that being the state DELETE was asking for, and a
1468
+ * DELETE that FAILS rejects and stays retryable, because nothing here records
1469
+ * a state a retry could early-return on.
1470
+ *
1471
+ * What it does NOT do is the point. Removing a month-old suspended workspace
1472
+ * through a handle meant starting a pod for it first, probing it, and then
1473
+ * deleting the pod again — compute spent, and a guest woken, purely to be
1474
+ * told to go away. The DELETE never needed any of that: the name is
1475
+ * deterministic, so the object can be addressed without being opened.
1476
+ *
1477
+ * It is not gated on the workspace being suspended, and deliberately: a
1478
+ * running workspace's DELETE takes its pod down with it, which is what
1479
+ * deleting a workspace means. A caller that wants the disk quiesced first
1480
+ * calls {@link suspendKubernetesWorkspace} and then this.
1481
+ */
1482
+ export async function deleteKubernetesWorkspace(config, workspaceId, options) {
1483
+ options?.signal?.throwIfAborted();
1484
+ const namespace = config.namespace;
1485
+ const name = workspaceSandboxName(workspaceId);
1486
+ const epoch = assertHolderEpoch(options?.epoch, 'deleteKubernetesWorkspace');
1487
+ const client = createKubernetesClient(clientAccess(config), clientOptions(config));
1488
+ // The fence reaches the standalone verbs too, and this is the one it
1489
+ // matters most on: a retention job superseded between deciding to delete a
1490
+ // workspace and calling this would otherwise take the new holder's disk,
1491
+ // and nothing brings a disk back.
1492
+ await deleteSandboxObject({ client, namespace, name, workspaceId }, 'deleteKubernetesWorkspace', epoch, options?.signal);
1493
+ }
1494
+ /**
1495
+ * Suspend a workspace by id, without ever adopting it: send the patch, then
1496
+ * wait for the pod to actually stop.
1497
+ *
1498
+ * The handle's own `suspend()` does three things — reap the terminals it
1499
+ * handed out, stop admitting calls, and patch-and-wait. Only the third is
1500
+ * about the CLUSTER, and it is the only one a process holding no handle can
1501
+ * do or needs to. So this is that third thing on its own, for the operator
1502
+ * putting somebody else's workspace to sleep.
1503
+ *
1504
+ * It resolves only once the pod is gone or in a terminal phase, for the same
1505
+ * reason the handle's does: a suspend is a promise that the disk is quiesced,
1506
+ * and the patch being accepted says only that the controller has been asked.
1507
+ * A pod that outlives `readyTimeoutMs` rejects with
1508
+ * {@link KubernetesWorkspaceSuspendTimeoutError} and the object is left
1509
+ * exactly as the patch left it — the next call patches and waits again.
1510
+ *
1511
+ * A workspace that no longer exists rejects with the client's own
1512
+ * already-gone error rather than resolving. Unlike a DELETE, this asks for a
1513
+ * state that cannot be reached: there is no object to suspend, and nothing
1514
+ * the caller believed about it holds.
1515
+ *
1516
+ * Any HANDLE another process is holding on this workspace is not told. It
1517
+ * finds out on its next call, which fails at the transport and is re-read
1518
+ * into a {@link KubernetesWorkspaceSuspendedError} — or the moment that
1519
+ * process calls `refresh()`. See {@link KubernetesWorkspace.refresh}.
1520
+ */
1521
+ export async function suspendKubernetesWorkspace(config, workspaceId, options) {
1522
+ options?.signal?.throwIfAborted();
1523
+ const namespace = config.namespace;
1524
+ const name = workspaceSandboxName(workspaceId);
1525
+ const epoch = assertHolderEpoch(options?.epoch, 'suspendKubernetesWorkspace');
1526
+ assertNoQuiesceHere(options?.quiesce, 'suspendKubernetesWorkspace');
1527
+ assertNoFlushHere(options?.flush, 'suspendKubernetesWorkspace');
1528
+ const readiness = resolveKubernetesReadiness(config);
1529
+ const client = createKubernetesClient(clientAccess(config), clientOptions(config));
1530
+ await writeOperatingMode({ client, namespace, name, workspaceId }, 'suspendKubernetesWorkspace', 'Suspended', epoch, options?.signal);
1531
+ await awaitPodRetired(client, namespace, name, workspaceId, readiness, options?.signal);
471
1532
  }
472
- /** The two merge patches this module sends, and the only two. */
473
- const SUSPEND_PATCH = { spec: { operatingMode: 'Suspended' } };
474
- const RESUME_PATCH = { spec: { operatingMode: 'Running' } };
1533
+ /**
1534
+ * How long the diagnostic `healthz` after an unconfirmed cancellation may
1535
+ * take before the agent is reported `unreachable`.
1536
+ *
1537
+ * Its own clock, short, and unrelated to every other budget on the path. The
1538
+ * caller's deadline has usually already expired by the time this runs — the
1539
+ * shared controller has just spent its whole eight-second cancel-confirm
1540
+ * window — and the readiness budget is for waiting out a pod that is coming
1541
+ * up, which is not what this is asking. This asks one question of a pod that
1542
+ * is already there, and an answer that has not arrived in five seconds is
1543
+ * not going to change what the host does with it.
1544
+ */
1545
+ const CANCELLATION_DIAGNOSIS_TIMEOUT_MS = 5_000;
1546
+ /**
1547
+ * The identity evidence gathered about one unconfirmed cancellation, keyed
1548
+ * by the error the caller is about to receive.
1549
+ *
1550
+ * A WeakMap rather than a field, because the error class belongs to the
1551
+ * shared execution controller and this is one backend's evidence ABOUT it:
1552
+ * adding a Kubernetes-shaped property to `RemoteCancellationUnknownError`
1553
+ * would put a field on the Firecracker tier's errors that nothing there can
1554
+ * ever fill. Weak, so an error nobody kept takes its entry with it.
1555
+ *
1556
+ * It is written where the diagnosis happens ({@link
1557
+ * KubernetesWorkspace.exec}'s retirement hook) and read once, where the error
1558
+ * passes back through the handle, which is the only place that can rename it.
1559
+ */
1560
+ const guestGoneEvidence = new WeakMap();
1561
+ /**
1562
+ * The `reason` an unconfirmed cancellation reports on a workspace — see
1563
+ * {@link SandboxRetirementObservation.reason}. Not "the patch failed": no
1564
+ * patch was attempted, and the pod is standing on purpose.
1565
+ */
1566
+ const WORKSPACE_KEPT_REASON = 'workspace-kept';
475
1567
  async function openWorkspaceHandle(options) {
476
1568
  const { client, namespace, name, workspaceId, readiness } = options;
477
1569
  const id = name;
1570
+ /** What every fenced write addresses, and what a refusal names. */
1571
+ const target = { client, namespace, name, workspaceId };
1572
+ /**
1573
+ * The epoch this handle writes under when a call passes none.
1574
+ *
1575
+ * It is the one the handle was opened with, and it moves to whatever a
1576
+ * later call wrote under successfully — never down, because a write only
1577
+ * succeeds when its epoch was at least the stored one. Letting it lag
1578
+ * behind a write this handle itself made would be the worst of both:
1579
+ * every later `suspend()` of its own would be refused by its own stamp.
1580
+ */
1581
+ let heldEpoch = options.epoch;
1582
+ /**
1583
+ * The pod-template revision the BOUND object carries, and the one the
1584
+ * template this handle last read would produce. Both move together on a
1585
+ * refresh; a plain `resume()` re-reads only the first, off the `GET` it
1586
+ * already makes.
1587
+ */
1588
+ let templateRevision = options.templateRevision;
1589
+ let currentTemplateHash = options.currentTemplateHash;
1590
+ /** The handle's own policy; a transition may override it for one call. */
1591
+ const defaultStartFailure = options.onStartFailure ?? 'suspend-if-woken';
478
1592
  let state = 'running';
479
1593
  let session;
480
1594
  /**
@@ -483,6 +1597,25 @@ async function openWorkspaceHandle(options) {
483
1597
  * replacing. Read once per session and never carried across one.
484
1598
  */
485
1599
  let podUid;
1600
+ /**
1601
+ * Which session `podUid` belongs to. Bumped when a session starts AND
1602
+ * when one is dropped ({@link dropSession}), so the number identifies the
1603
+ * LIVE session and nothing else — a transport whose session has been
1604
+ * retired never matches it again. Read by {@link refreshBoundPod}, which
1605
+ * is where the desync it prevents is described, and by
1606
+ * {@link boundToPod}.
1607
+ */
1608
+ let sessionSeq = 0;
1609
+ /**
1610
+ * The session `podUid` was written for — see {@link boundToPod}.
1611
+ *
1612
+ * Deliberately NOT `sessionSeq` itself: a dropped session bumps that
1613
+ * counter and writes nothing else, so "the pod belongs to the session
1614
+ * that is live" is a comparison rather than a flag anybody has to
1615
+ * remember to clear. It starts before the first session so that a handle
1616
+ * which has not bound anything yet is bound to nothing.
1617
+ */
1618
+ let podSession = -1;
486
1619
  /**
487
1620
  * The uid of the pod a suspend patch that LANDED took away, cleared once a
488
1621
  * resume has bound its replacement.
@@ -500,13 +1633,262 @@ async function openWorkspaceHandle(options) {
500
1633
  * it would time out a resume whose workspace was perfectly usable.
501
1634
  */
502
1635
  let retiredPodUid;
503
- const terminals = new Set();
504
1636
  /**
505
- * Lifecycle transitions run one at a time. Two of them racing would
506
- * interleave a suspend patch with a resume's readiness poll and settle on
507
- * whichever finished last, which is how a workspace ends up marked running
508
- * with no pod.
509
- */
1637
+ * The Sandbox object this handle was opened on, and the disks behind it.
1638
+ *
1639
+ * `sandboxUid` is taken from the readiness GET the first bind already
1640
+ * makes — the object was read, its uid was in the reply, and until now it
1641
+ * was thrown away. It is what a rebind compares against: the workspace
1642
+ * name is DETERMINISTIC, so an object deleted and recreated stands under
1643
+ * the same name with an empty disk, and following a pod behind a
1644
+ * different uid would hand a caller a stranger's workspace.
1645
+ *
1646
+ * `volumeClaimUids` is read ONCE, at the first bind, and not again. The
1647
+ * PVCs belong to the Sandbox: while `sandboxUid` has not moved they have
1648
+ * not either, so re-reading them on every resume would be a GET per
1649
+ * volume per transition for an answer that cannot have changed. It stays
1650
+ * empty when the Role does not grant `get` on `persistentvolumeclaims` —
1651
+ * an existing deployment upgrading into this release must not have every
1652
+ * `createKubernetesWorkspace` start failing on a verb its Role has never
1653
+ * had.
1654
+ */
1655
+ let sandboxUid;
1656
+ let volumeClaimUids = {};
1657
+ let volumeClaimsRead = false;
1658
+ /**
1659
+ * The agent PROCESS the handle is talking to: the boot id of the last
1660
+ * reply that carried one.
1661
+ *
1662
+ * Cleared whenever the pod changes — a session start, and a rebind that
1663
+ * followed a replacement — which is what keeps the caller's own
1664
+ * `suspend()`/`resume()` from being reported as a restart: the new pod's
1665
+ * first reply establishes a baseline instead of being compared against
1666
+ * the old pod's.
1667
+ *
1668
+ * It is deliberately NOT a baseline for the cancellation diagnosis.
1669
+ * Whether the guest a COMMAND was running in is gone is a question about
1670
+ * that command, and the boot id it reserved against is kept per execution
1671
+ * by the transport ({@link guestBootIdWhenReserved}); a handle-wide
1672
+ * baseline would answer "restarted" for every command started after a
1673
+ * restart the handle survived.
1674
+ *
1675
+ * Against a guest too old to report one it stays `undefined` for ever,
1676
+ * and nothing here fires. That is the interop contract: a missing boot id
1677
+ * is "this guest cannot tell me", never "it changed".
1678
+ */
1679
+ let guestBootId;
1680
+ /** Subscribers to {@link KubernetesWorkspace.onGuestRestart}. */
1681
+ const restartListeners = new Set();
1682
+ const terminals = new Set();
1683
+ /**
1684
+ * This handle's disk, around ONE named pod and the process inside it.
1685
+ *
1686
+ * Every identity this file hands out is built here, which is what keeps a
1687
+ * payload from being blanked by a state the handle happens to be in: a
1688
+ * routine that is HOLDING a pod uid reports that uid, and the only
1689
+ * question left is which pod it holds.
1690
+ */
1691
+ const identityOn = (pod, boot) => ({
1692
+ sandboxUid,
1693
+ volumeClaimUids: { ...volumeClaimUids },
1694
+ podUid: pod,
1695
+ guestBootId: boot,
1696
+ });
1697
+ /**
1698
+ * Whether `podUid` still names the pod this handle is TALKING to.
1699
+ *
1700
+ * The question is about the SESSION and never about `state`, and the
1701
+ * difference is not academic: a transition is not the absence of a pod.
1702
+ * {@link startSession} binds one, records it and runs the privilege probe
1703
+ * against it while `state` is still `'suspended'`, so a handle that
1704
+ * answered "no pod" for the length of a resume would blank exactly the
1705
+ * announcement this backend exists to make — a pod somebody else replaced
1706
+ * underneath a resume, discovered when that probe is refused.
1707
+ *
1708
+ * Every path that gives a pod back drops the session first
1709
+ * ({@link dropSession}), which bumps `sessionSeq` and leaves this false;
1710
+ * every path that takes one binds through `startSession`, which sets both
1711
+ * together. So there is no window in which this says yes about a pod
1712
+ * nothing is talking to.
1713
+ */
1714
+ const boundToPod = () => podUid !== undefined && podSession === sessionSeq;
1715
+ /** This handle's four objects, read fresh — see {@link KubernetesWorkspaceIdentity}. */
1716
+ const identityNow = () =>
1717
+ // A handle with no session has no pod and so no agent process.
1718
+ // Reporting the ones it HAD would be the single most misleading thing
1719
+ // this object could say: the pod is deleted and those ids name nothing.
1720
+ boundToPod() ? identityOn(podUid, guestBootId) : identityOn(undefined, undefined);
1721
+ /**
1722
+ * This handle's identity with the pod a LOOK actually FOUND, naming an
1723
+ * agent process only when that pod is the one the handle's boot id came
1724
+ * from.
1725
+ *
1726
+ * The two halves have to come from the same pod or the answer is a
1727
+ * fabrication: the boot id of the pod this handle is bound to, printed
1728
+ * beside the uid of a replacement nobody has heard a word from, names a
1729
+ * process that pod never ran. `undefined` is what "nothing has answered
1730
+ * from there yet" looks like, and it is the only honest thing to say.
1731
+ */
1732
+ const guestSeen = (uid) => {
1733
+ const live = identityNow();
1734
+ return {
1735
+ ...live,
1736
+ podUid: uid,
1737
+ guestBootId: uid !== undefined && uid === live.podUid ? live.guestBootId : undefined,
1738
+ };
1739
+ };
1740
+ /**
1741
+ * This handle's identity as one COMMAND saw it: the pod its reservation
1742
+ * was accepted by and the process inside it, where the transport kept
1743
+ * them ({@link guestWhenReserved}), and the handle's own where it did not.
1744
+ *
1745
+ * What a diagnosis compares against has to be the command's guest and not
1746
+ * the handle's. The handle moves — it follows a replaced pod, and the
1747
+ * failing call's own `cancel-execution` refusal already advanced its boot
1748
+ * id on the way in — so reading it here would report the guest that is
1749
+ * running NOW as the guest that died.
1750
+ */
1751
+ const identityWhenReserved = (reserved) => {
1752
+ const live = identityNow();
1753
+ if (reserved === undefined)
1754
+ return live;
1755
+ const pod = reserved.podUid ?? live.podUid;
1756
+ return {
1757
+ ...live,
1758
+ podUid: pod,
1759
+ // Never the handle's boot id for a DIFFERENT pod than the one
1760
+ // being named — see {@link guestSeen}.
1761
+ guestBootId: reserved.guestBootId ?? (pod === live.podUid ? live.guestBootId : undefined),
1762
+ };
1763
+ };
1764
+ /**
1765
+ * Tell every subscriber, and let none of them fail the call that noticed.
1766
+ *
1767
+ * Synchronous and in subscription order, so a host can keep its own book
1768
+ * up to date before the call that discovered the restart returns. A
1769
+ * listener that throws is swallowed for the reason every notification
1770
+ * callback in this file is: the caller's result is already decided, and a
1771
+ * host's bookkeeping bug must not become the workspace's error.
1772
+ */
1773
+ const announceGuestRestart = (event) => {
1774
+ for (const listener of [...restartListeners]) {
1775
+ try {
1776
+ listener(event);
1777
+ }
1778
+ catch {
1779
+ // See above.
1780
+ }
1781
+ }
1782
+ };
1783
+ /**
1784
+ * Follow the agent PROCESS behind every authenticated reply.
1785
+ *
1786
+ * Installed on the transport for both address modes, because this is the
1787
+ * one thing no address and no token can reveal: the kubelet restarts a
1788
+ * crashed container INSIDE the same pod, so the uid — and therefore the
1789
+ * bind token — is unchanged and every call keeps working, while every
1790
+ * process the caller started is gone. Only the guest's own boot id says
1791
+ * so, and only on replies it was already sending.
1792
+ *
1793
+ * It never fires for the FIRST reply after the pod changed: `guestBootId`
1794
+ * is undefined then, and the first value is the baseline rather than a
1795
+ * change. That is what makes the caller's own suspend/resume silent.
1796
+ *
1797
+ * `generation` guards it exactly as {@link refreshBoundPod}'s does, and
1798
+ * for the same reason: a reply from a session this handle has already let
1799
+ * go says nothing about the guest it is bound to now. A late frame from a
1800
+ * retired transport would otherwise set the boot id back to the dead
1801
+ * pod's, announce a restart nobody had, and make the live pod's next
1802
+ * reply announce a second one.
1803
+ *
1804
+ * `from` — the bind token of the wire the reply arrived on — is the same
1805
+ * guard one level down, and a REBIND needs it because it does not retire
1806
+ * the session: it swaps the pod under a session that goes on. The
1807
+ * transport keeps no lock on the wire it leaves, so the outgoing pod can
1808
+ * still answer a call dispatched to it (a `write-file` mid-flight, a pod
1809
+ * inside its termination grace period) after this handle has followed the
1810
+ * replacement. Taken as this session's, that reply would seed the
1811
+ * replacement's baseline with the DEPARTED pod's process — so
1812
+ * `workspace.identity`, and the `container-restarted` event the next real
1813
+ * reply then fires, would name a process that never ran in the pod beside
1814
+ * it. A pod the handle has left says nothing at all, which is exactly
1815
+ * what a retired session's reply says.
1816
+ */
1817
+ const observeGuestReply = (generation, from, reply) => {
1818
+ if (generation !== sessionSeq)
1819
+ return;
1820
+ if (from !== podUid)
1821
+ return;
1822
+ const seen = reply.guestBootId;
1823
+ if (typeof seen !== 'string' || seen === '')
1824
+ return;
1825
+ if (guestBootId === undefined) {
1826
+ guestBootId = seen;
1827
+ return;
1828
+ }
1829
+ if (seen === guestBootId)
1830
+ return;
1831
+ // Both halves name the pod this reply came from, taken from the
1832
+ // variable rather than from {@link identityNow}: the two guards above
1833
+ // have already established that `podUid` IS the pod that answered,
1834
+ // and an announcement about a guest must not be able to come out
1835
+ // naming no guest at all.
1836
+ const previous = identityOn(podUid, guestBootId);
1837
+ guestBootId = seen;
1838
+ announceGuestRestart({
1839
+ reason: 'container-restarted',
1840
+ previous,
1841
+ current: identityOn(podUid, seen),
1842
+ });
1843
+ };
1844
+ /**
1845
+ * Let go of one terminal this handle handed out, on the way to giving the
1846
+ * pod back.
1847
+ *
1848
+ * A connection-bound terminal is KILLED, exactly as it always was: its
1849
+ * program lives on this connection and the pod is going away. A session
1850
+ * ATTACHMENT is only detached — the program is the registry's, not this
1851
+ * connection's, and this path exists to stop a caller waiting on an
1852
+ * `exited` that would otherwise resolve only when TCP notices, not to
1853
+ * decide the program's fate. Either way the session dies with the pod;
1854
+ * what differs is whether this handle claims to have ended it.
1855
+ */
1856
+ const releaseTerminal = (terminal) => {
1857
+ if (typeof terminal.detach === 'function')
1858
+ terminal.detach();
1859
+ else
1860
+ terminal.kill('SIGKILL');
1861
+ };
1862
+ /**
1863
+ * The transport behind each session's inner handle.
1864
+ *
1865
+ * The detach/attach ops are the workspace's own surface, not the SDK
1866
+ * `Sandbox`'s, so they are reached on the transport rather than through
1867
+ * the inner handle — which also keeps them off the inner handle's
1868
+ * automatic-retirement path, where an unconfirmed cancel takes the pod
1869
+ * away. A detached command losing its connection must cost the
1870
+ * workspace nothing, so it must not travel that road at all.
1871
+ *
1872
+ * Keyed by handle rather than kept in a `let`, so there is no window in
1873
+ * which `session` and the transport disagree: `admitted()` hands out the
1874
+ * live handle, and the transport looked up from it is that handle's own
1875
+ * or nothing.
1876
+ */
1877
+ const sessionTransports = new WeakMap();
1878
+ /** The live session's transport, refused by name if there is none. */
1879
+ const admittedTransport = (operation, handle) => {
1880
+ const transport = sessionTransports.get(handle);
1881
+ if (transport === undefined) {
1882
+ throw new KubernetesWorkspaceSuspendedError(operation, workspaceId, name);
1883
+ }
1884
+ return transport;
1885
+ };
1886
+ /**
1887
+ * Lifecycle transitions run one at a time. Two of them racing would
1888
+ * interleave a suspend patch with a resume's readiness poll and settle on
1889
+ * whichever finished last, which is how a workspace ends up marked running
1890
+ * with no pod.
1891
+ */
510
1892
  let queue = Promise.resolve();
511
1893
  const serialise = (run) => {
512
1894
  const next = queue.then(run, run);
@@ -529,25 +1911,147 @@ async function openWorkspaceHandle(options) {
529
1911
  * a second caller's is not consulted, which is what sharing means.
530
1912
  */
531
1913
  let pendingSuspend;
1914
+ /**
1915
+ * Whether the suspend in flight is quiescing the guest first.
1916
+ *
1917
+ * Kept beside the promise because it is the ONE part of a caller's
1918
+ * request that a joining caller cannot inherit. `signal` and `epoch` are
1919
+ * authority and lifetime — the first caller's to give, and a second
1920
+ * caller joining a transition under them is what sharing means. A
1921
+ * `quiesce` is a promise about the guest, and a suspend already patching
1922
+ * over processes nobody stopped cannot keep it retroactively.
1923
+ */
1924
+ let pendingSuspendQuiesces = false;
1925
+ /**
1926
+ * Whether the suspend in flight is flushing the guest's writes first.
1927
+ *
1928
+ * Beside {@link pendingSuspendQuiesces} and for its reason: a flush is a
1929
+ * promise about the DISK, and a suspend already patching cannot keep it
1930
+ * afterwards — once its state leaves `running` no call is admitted to
1931
+ * ask the guest for anything. The flag reads `true` for almost every
1932
+ * flight, because a flush is what `suspend()` does unless a caller turns
1933
+ * it off; it is `false` only behind a deliberate `flush: false`.
1934
+ */
1935
+ let pendingSuspendFlushes = false;
532
1936
  let pendingDelete;
533
- const deleteSandbox = async (signal) => {
534
- try {
535
- await client.request('DELETE', sandboxPath(namespace, name), undefined, signal);
1937
+ const deleteSandbox = async (signal, epoch, gate) =>
1938
+ // Already gone is the state DELETE was asking for, fenced or not.
1939
+ await deleteSandboxObject(target, 'destroy', epoch, signal, gate);
1940
+ /**
1941
+ * The last Sandbox object a readiness poll read, kept for the two facts
1942
+ * {@link bindingFromSandbox} does not carry: `metadata.uid` and the
1943
+ * `volumeClaimTemplates` entry names.
1944
+ *
1945
+ * Kept rather than re-read. Every bind already GETs this object, both
1946
+ * fields were in the reply and were being discarded, and asking for them
1947
+ * again would be a second GET of a thing already in hand — and a second
1948
+ * answer that could disagree with the one the bind acted on.
1949
+ */
1950
+ let lastSandbox;
1951
+ const readBinding = async (pollSignal) => {
1952
+ const sandbox = await client.request('GET', sandboxPath(namespace, name), undefined, pollSignal);
1953
+ lastSandbox = sandbox;
1954
+ return bindingFromSandbox(sandbox);
1955
+ };
1956
+ /**
1957
+ * Read the uid of each `volumeClaimTemplates` entry's PVC, once per
1958
+ * handle, and never fail the workspace over it.
1959
+ *
1960
+ * The disk is the thing a workspace IS, and a host whose records say "id
1961
+ * X holds a month of work" needs to be able to tell that disk from a
1962
+ * different disk behind the same name. `sandboxUid` already catches the
1963
+ * ordinary case (delete and recreate takes the PVCs with it, because the
1964
+ * Sandbox owns them); this is what makes the claim checkable
1965
+ * independently, and what a host compares across processes.
1966
+ *
1967
+ * Best-effort ON PURPOSE. It needs `get` on `persistentvolumeclaims`,
1968
+ * which this release adds to the shipped Role and which no Role from an
1969
+ * earlier release has. A 403 here must cost a deployment nothing but this
1970
+ * one report — refusing to open a workspace because an OPTIONAL identity
1971
+ * field could not be read would turn a documentation change into an
1972
+ * outage.
1973
+ */
1974
+ const readVolumeClaimUids = async (signal) => {
1975
+ if (volumeClaimsRead)
1976
+ return;
1977
+ const entries = lastSandbox?.spec?.volumeClaimTemplates ?? [];
1978
+ const uids = {};
1979
+ for (const entry of entries) {
1980
+ const claimName = entry?.metadata?.name;
1981
+ if (typeof claimName !== 'string' || claimName === '')
1982
+ continue;
1983
+ try {
1984
+ const claim = await client.request('GET', persistentVolumeClaimPath(namespace, name, claimName), undefined, signal);
1985
+ const uid = claim?.metadata?.uid;
1986
+ if (typeof uid === 'string' && uid !== '')
1987
+ uids[claimName] = uid;
1988
+ }
1989
+ catch {
1990
+ // See above: an unreadable PVC leaves the entry out and
1991
+ // nothing else. `volumeClaimUids` says what could be read,
1992
+ // never what was guessed.
1993
+ }
536
1994
  }
537
- catch (err) {
538
- // Already gone is the state DELETE was asking for.
539
- if (!(err instanceof KubernetesAlreadyGoneError))
540
- throw err;
1995
+ // Both written only once the loop has finished, and in this order. The
1996
+ // per-PVC `catch` above covers the request and nothing else, so an
1997
+ // abort — or a claim name the path builder refuses — leaves through
1998
+ // here; latching the flag on the way IN would have left the uids
1999
+ // empty for the handle's life with no read left that could fill them.
2000
+ volumeClaimUids = uids;
2001
+ volumeClaimsRead = true;
2002
+ };
2003
+ /**
2004
+ * The budget ran out with no pod this handle was allowed to bind, worded
2005
+ * for the transition that ran out.
2006
+ *
2007
+ * Three different operator problems arrive here and one sentence cannot
2008
+ * serve them: a resume whose controller never replaced the pod, an adopt
2009
+ * that walked in on a pod still riding out its termination grace period,
2010
+ * and a create whose pod never came up at all. The pod the wait was
2011
+ * excluding is named whenever there is one, because "which pod is still
2012
+ * there" is the first thing anyone looks up next.
2013
+ *
2014
+ * A fourth arrives only under `agentAddress: 'pod-ip'`, and it takes
2015
+ * precedence over all of them: the bind DID find its pod, and what never
2016
+ * turned up was that pod's address. Reporting "no pod this handle could
2017
+ * bind to had appeared" about a pod the wait had already bound would send
2018
+ * an operator after the controller for something the CNI never finished.
2019
+ */
2020
+ const bindTimedOut = (policy, lastError, addresslessPodUid) => {
2021
+ const cause = lastError !== undefined ? { cause: lastError } : undefined;
2022
+ const subject = `kubernetes: workspace ${workspaceId} (Sandbox ${name})`;
2023
+ const budget = `${readiness.timeoutMs}ms`;
2024
+ if (addresslessPodUid !== undefined) {
2025
+ return new Error(`${subject} is configured with agentAddress: 'pod-ip', and ${budget} after it bound pod ${addresslessPodUid} that pod still reported no status.podIP, so there is no address to dial it at. A pod is given its IP when the CNI has finished attaching it, so a pod that never gets one has not been given a network at all. Raise readyTimeoutMs, look at the pod's events — or run with the default agentAddress: 'service' if this host is inside the cluster.`, cause);
2026
+ }
2027
+ if (policy.transition === 'resume' && policy.retiring !== undefined) {
2028
+ return new Error(`${subject} was patched back to operatingMode: Running, but ${budget} later the only pod behind it was still the one it was suspended from (uid ${policy.retiring}). A resumed pod keeps the sandbox's name and gets a new uid, and that uid is the agent's bind token, so binding to the old pod would present a token the new agent refuses. The Ready condition cannot be waited on instead — the controller leaves it standing across the transition. Raise readyTimeoutMs, or look at why the controller has not replaced the pod.`, cause);
541
2029
  }
2030
+ if (policy.transition === 'adopt' && policy.retiring !== undefined) {
2031
+ return new Error(`${subject} was adopted while the previous pod (uid ${policy.retiring}) was still TERMINATING, and ${budget} later the controller had still not replaced it. A guest whose PID 1 ignores SIGTERM rides out its terminationGracePeriodSeconds before the replacement is created, so the adopt waits for the new pod under the same readiness budget as everything else on this path. ${NO_BINDABLE_POD_ADVICE}`, cause);
2032
+ }
2033
+ if (policy.transition === 'adopt' && policy.awaitReplacement) {
2034
+ return new Error(`${subject} was adopted from operatingMode: Suspended and patched back to Running, but ${budget} later no pod this handle could bind to had appeared — the pod it was suspended from is most likely still terminating. ${NO_BINDABLE_POD_ADVICE}`, cause);
2035
+ }
2036
+ const opened = policy.transition === 'create'
2037
+ ? 'was created'
2038
+ : policy.transition === 'adopt'
2039
+ ? 'was adopted while it was Running'
2040
+ : 'was patched back to operatingMode: Running';
2041
+ return new Error(`${subject} ${opened}, but ${budget} later no pod this handle could bind to had appeared. ${NO_BINDABLE_POD_ADVICE}`, cause);
542
2042
  };
543
- const readBinding = async (pollSignal) => bindingFromSandbox(await client.request('GET', sandboxPath(namespace, name), undefined, pollSignal));
544
2043
  /**
545
2044
  * Wait until the Sandbox is Ready AND the pod behind it is a pod this
546
- * handle is allowed to bind to, then read that pod's uid.
2045
+ * handle is allowed to bind to — and, under `agentAddress: 'pod-ip'`, one
2046
+ * that has an address — then read both facts off it.
2047
+ *
2048
+ * On create there is nothing to exclude and nothing to wait for, and this
2049
+ * is one poll plus one read, exactly as it was. Everywhere else the pod
2050
+ * behind the name is MOVING, and `policy` says how: `retiring` is the pod
2051
+ * this bind must see replaced, and `awaitReplacement` says whether a read
2052
+ * that finds no live pod at all is "not yet" — see {@link PodBindPolicy}.
547
2053
  *
548
- * On create there is nothing to exclude and this is one poll plus one
549
- * read, exactly as it was. On RESUME `previousPodUid` is the pod the
550
- * workspace was suspended from, and excluding it is the whole point:
2054
+ * Excluding a pod is the whole point wherever one is named, because
551
2055
  * `Ready` is not a transition signal. The controller leaves the condition
552
2056
  * True across a resume — upstream says the same of `Suspended` in the
553
2057
  * other direction — so the very first poll after the Running patch can
@@ -558,31 +2062,89 @@ async function openWorkspaceHandle(options) {
558
2062
  * pointing at the race. So the uid is polled, under the SAME deadline as
559
2063
  * everything else on this path, until it is a different pod's.
560
2064
  *
561
- * A failed read is fatal on create and merely "not yet" on resume: for a
562
- * stretch of every resume there is no live pod at all, and
563
- * {@link readPodBindToken} answers that by throwing rather than returning
564
- * undefined. The last such error is carried onto the timeout as `cause`,
565
- * so a resume that never found a pod still says what it kept seeing.
2065
+ * An ADDRESS-less pod is a third kind of "not yet", and only under
2066
+ * `'pod-ip'`. The replacement pod is picked up while it is still
2067
+ * `Pending` — that is the point of excluding by uid rather than waiting
2068
+ * for Ready — and a `Pending` pod has no `status.podIP` until the CNI has
2069
+ * attached it. Every real pod passes through that window, so refusing
2070
+ * there would fail the ordinary resume in milliseconds with the whole
2071
+ * budget unspent. It is waited out here, on the same deadline, and the
2072
+ * timeout below names the pod that never got an address.
2073
+ *
2074
+ * The last failed read is carried onto the timeout as `cause`, so a wait
2075
+ * that never found a pod still says what it kept seeing.
566
2076
  */
567
- const acquireBoundPod = async (deadline, previousPodUid) => {
2077
+ const acquireBoundPod = async (deadline, policy) => {
568
2078
  const label = `workspace ${workspaceId} (Sandbox ${name})`;
2079
+ const needsPodIP = options.agentAddress === 'pod-ip';
569
2080
  let lastError;
2081
+ /** The last live-but-address-less pod seen, for the timeout's words. */
2082
+ let addresslessPodUid;
2083
+ /**
2084
+ * Whether the Sandbox has been seen Ready at least once.
2085
+ *
2086
+ * It decides who gets to report a budget that ran out inside the
2087
+ * readiness poll. Before the first Ready, "never became Ready" is the
2088
+ * true sentence and `pollForBinding`'s own error is the right one. After
2089
+ * it, the clock was being spent waiting for a POD, and blaming a
2090
+ * condition that has been True the whole time points the operator at
2091
+ * the one thing that is not the problem — see {@link bindTimedOut}.
2092
+ *
2093
+ * It rewrites nothing else: a readiness read that failed while the
2094
+ * budget still had time left failed on its own account.
2095
+ */
2096
+ let seenReady = false;
570
2097
  while (deadline.remainingMs() > 0) {
571
- const binding = await pollForBinding(readBinding, deadline, readiness, label);
572
- let token;
2098
+ let binding;
2099
+ try {
2100
+ binding = await pollForBinding(readBinding, deadline, readiness, label);
2101
+ }
2102
+ catch (err) {
2103
+ // Only a clock that has actually run out is rewritten. A
2104
+ // readiness read that fails for any other reason — the API
2105
+ // server refused it, the caller aborted — is that call's own
2106
+ // failure and travels out unchanged. That matters most while a
2107
+ // `'pod-ip'` bind is waiting for an address: replacing a 5xx
2108
+ // with "the CNI never gave the pod an address" sends an operator
2109
+ // to the pod's events for a fault that was never the pod's, and
2110
+ // claims an elapsed time that never elapsed.
2111
+ //
2112
+ // The error is what says which happened, and only the error
2113
+ // can: an expiry does not arrive here as an
2114
+ // `OperationDeadlineExpired` — {@link pollForBinding} catches
2115
+ // that itself and reports it in its own words, as
2116
+ // {@link ReadinessPollTimeout} — and asking the clock instead
2117
+ // is wrong by a fraction of a millisecond, because the expiry
2118
+ // timer and `performance.now()` are different clocks and the
2119
+ // timer can fire first.
2120
+ if (!seenReady || !(err instanceof ReadinessPollTimeout))
2121
+ throw err;
2122
+ // Kept only if no pod read has failed yet: what the wait kept
2123
+ // seeing is more useful as the cause than the clock running out.
2124
+ lastError ??= err;
2125
+ break;
2126
+ }
2127
+ seenReady = true;
2128
+ let pod;
573
2129
  try {
574
- token = await deadline.run(async (tokenSignal) => await readPodBindToken(client, namespace, binding, tokenSignal));
2130
+ pod = await deadline.run(async (tokenSignal) => await readBoundPod(client, namespace, binding, tokenSignal));
575
2131
  }
576
2132
  catch (err) {
577
2133
  if (err instanceof OperationDeadlineExpired)
578
2134
  break;
579
- if (previousPodUid === undefined)
2135
+ if (!policy.awaitReplacement)
580
2136
  throw err;
581
2137
  lastError = err;
582
- token = undefined;
2138
+ pod = undefined;
2139
+ }
2140
+ // The IP travels WITH the uid, from this same read: a resumed pod
2141
+ // is a new pod at a new address, so a `'pod-ip'` session that
2142
+ // carried yesterday's IP would dial a pod that no longer exists.
2143
+ if (pod !== undefined && pod.uid !== policy.retiring) {
2144
+ if (!needsPodIP || pod.podIP !== undefined)
2145
+ return { binding, pod };
2146
+ addresslessPodUid = pod.uid;
583
2147
  }
584
- if (token !== undefined && token !== previousPodUid)
585
- return { binding, token };
586
2148
  try {
587
2149
  await deadline.delay(readiness.pollIntervalMs);
588
2150
  }
@@ -592,7 +2154,130 @@ async function openWorkspaceHandle(options) {
592
2154
  throw err;
593
2155
  }
594
2156
  }
595
- throw new Error(`kubernetes: workspace ${workspaceId} (Sandbox ${name}) was patched back to operatingMode: Running, but ${readiness.timeoutMs}ms later the only pod behind it was still the one it was suspended from (uid ${previousPodUid}). A resumed pod keeps the sandbox's name and gets a new uid, and that uid is the agent's bind token, so binding to the old pod would present a token the new agent refuses. The Ready condition cannot be waited on instead — the controller leaves it standing across the transition. Raise readyTimeoutMs, or look at why the controller has not replaced the pod.`, lastError !== undefined ? { cause: lastError } : undefined);
2157
+ throw bindTimedOut(policy, lastError, addresslessPodUid);
2158
+ };
2159
+ /**
2160
+ * The ONE re-read behind every rebind: the routine the transport runs
2161
+ * when a call failed in a way that proves it ran nothing in the guest.
2162
+ *
2163
+ * It serves both triggers and both address modes, and generalising it is
2164
+ * the whole of this workstream's transport change. `'pod-ip'` needed it
2165
+ * first, for a dial that could not connect because the address died with
2166
+ * its pod. The DEFAULT `'service'` mode needs it for the opposite
2167
+ * symptom: the Service FQDN outlives the pod and resolves to the
2168
+ * replacement, so the dial succeeds and the new agent refuses the old
2169
+ * pod's uid — a flat `unauthorized` that used to be the end of the
2170
+ * handle.
2171
+ *
2172
+ * What it decides, in order, and why each answer is the only safe one:
2173
+ *
2174
+ * - **A different Sandbox uid, or no Sandbox at all** ⇒
2175
+ * {@link KubernetesWorkspaceReplacedError}, and never a rebind. The
2176
+ * name is deterministic, so this is a workspace somebody deleted and
2177
+ * created again: an EMPTY disk behind a name whose records say
2178
+ * otherwise. The transport rethrows this one rather than swallowing
2179
+ * it, because it is a verdict about the object and not a failed
2180
+ * diagnosis.
2181
+ * - **Suspended** ⇒ the current handle, unchanged. There is no pod to
2182
+ * bind and nothing here to say about it; the failing call's own
2183
+ * {@link admitted} re-read is what turns this into a
2184
+ * `KubernetesWorkspaceSuspendedError` naming the foreign suspend.
2185
+ * - **Running, same uid, a live pod** ⇒ that pod's address and token.
2186
+ * The transport installs it only if the token actually moved, so an
2187
+ * unchanged pod leaves the caller's original error standing — a pod
2188
+ * that is still there and still refusing is a guest problem, and
2189
+ * replacing that error with a later one would hide it.
2190
+ *
2191
+ * `generation` is what serialises this with the transitions WITHOUT
2192
+ * taking their queue — which it must not, because the privilege probe
2193
+ * runs inside a transition and a rebind that waited on the queue would
2194
+ * deadlock the resume holding it. Every transition bumps `sessionSeq`
2195
+ * before it changes the pod ({@link dropSession}, and `startSession`
2196
+ * itself), so a re-read that lands after a suspend or a resume no longer
2197
+ * matches and writes nothing: the transport may still swap the handle of
2198
+ * a session nothing is using, which costs nobody anything, and the
2199
+ * handle's own `podUid` — which `retiredPodUid` is stamped from — is
2200
+ * never written by a session that has been retired.
2201
+ */
2202
+ const refreshBoundPod = (binding, generation, opened) => {
2203
+ /**
2204
+ * What the transport is dialing right now, so "nothing changed" can
2205
+ * be said by handing back exactly that.
2206
+ *
2207
+ * It must be the CURRENT one and not the one this session opened
2208
+ * with: the transport rebinds whenever the token it is given differs
2209
+ * from the token it holds, so answering a later re-read with the
2210
+ * original address would drag a handle that has already followed a
2211
+ * replacement back onto the pod it left.
2212
+ */
2213
+ let dialing = opened;
2214
+ return async (signal) => {
2215
+ let sandbox;
2216
+ try {
2217
+ sandbox = await readSandboxObject(target, signal);
2218
+ }
2219
+ catch (err) {
2220
+ // A DELETE cascades to the disk, so a name with no object
2221
+ // behind it is not "not yet" — it is the end of this
2222
+ // workspace, and the one answer that must never be followed.
2223
+ if (err instanceof KubernetesAlreadyGoneError) {
2224
+ throw new KubernetesWorkspaceReplacedError(workspaceId, name, sandboxUid, undefined, {
2225
+ cause: err,
2226
+ });
2227
+ }
2228
+ throw err;
2229
+ }
2230
+ const standing = sandbox?.metadata?.uid;
2231
+ // Read before anything else, and refused before anything else: a
2232
+ // disk that is not this handle's disk is the one answer no retry
2233
+ // and no rebind may follow.
2234
+ if (sandbox === undefined || (sandboxUid !== undefined && standing !== sandboxUid)) {
2235
+ throw new KubernetesWorkspaceReplacedError(workspaceId, name, sandboxUid, standing);
2236
+ }
2237
+ // Somebody else suspended it. There is no pod, and saying so here
2238
+ // would be a worse error than the one `admitted` is about to
2239
+ // produce, which names the suspension and what to do about it.
2240
+ if (operatingModeOf(sandbox) === 'Suspended')
2241
+ return dialing;
2242
+ const refreshed = bindingFromSandbox(sandbox) ?? binding;
2243
+ const pod = await readBoundPod(client, namespace, refreshed, signal);
2244
+ const next = resolveAgentAddress(refreshed, options.agentPort, pod.uid, {
2245
+ mode: options.agentAddress,
2246
+ ...(pod.podIP !== undefined ? { podIP: pod.podIP } : {}),
2247
+ });
2248
+ dialing = next;
2249
+ if (generation === sessionSeq && pod.uid !== podUid) {
2250
+ // The pod this handle was bound to a moment ago, and the last
2251
+ // process heard from INSIDE it — read out of the variables
2252
+ // this routine is holding rather than assembled from the
2253
+ // handle's state. That is the whole difference on the one
2254
+ // path this announcement matters most: a pod replaced during
2255
+ // a RESUME is discovered by the privilege probe, which runs
2256
+ // while `state` is still `'suspended'`, and an identity built
2257
+ // from a state gate would announce a pod replacement while
2258
+ // naming neither pod.
2259
+ const previous = identityOn(podUid, guestBootId);
2260
+ podUid = next.token;
2261
+ podSession = sessionSeq;
2262
+ // A new pod is a new agent process, and the boot id it will
2263
+ // report is not the one the handle has been comparing
2264
+ // against. Clearing it is what stops the first reply from the
2265
+ // replacement being announced a second time as a container
2266
+ // restart — and it is why `current.guestBootId` on the event
2267
+ // below is `undefined`: the replacement has not answered yet,
2268
+ // and naming a process nobody has heard from would be a
2269
+ // guess. A listener reads `current.podUid` for what moved,
2270
+ // and `workspace.identity` once its own call has returned for
2271
+ // the process that answered from there.
2272
+ guestBootId = undefined;
2273
+ announceGuestRestart({
2274
+ reason: 'pod-replaced',
2275
+ previous,
2276
+ current: identityOn(next.token, undefined),
2277
+ });
2278
+ }
2279
+ return next;
2280
+ };
596
2281
  };
597
2282
  /**
598
2283
  * Bring up one pod's worth of state: wait for a pod that is not the one
@@ -600,26 +2285,63 @@ async function openWorkspaceHandle(options) {
600
2285
  * transport and prove the guest is deprivileged. Called on create and on
601
2286
  * every resume, with nothing carried over between them.
602
2287
  */
603
- const startSession = async (label, signal, previousPodUid) => {
604
- const deadline = new OperationDeadline(readiness.timeoutMs, `kubernetes workspace ${name} ${label}`, signal);
2288
+ const startSession = async (policy, signal) => {
2289
+ const deadline = new OperationDeadline(readiness.timeoutMs, `kubernetes workspace ${name} ${policy.transition}`, signal);
605
2290
  // Both of these are re-read rather than remembered: a resumed pod keeps
606
2291
  // the name and changes the uid and the IP, so a handle that reused
607
2292
  // either would present a token the new agent refuses, at an address
608
2293
  // whose pod is being deleted.
609
- const { binding, token } = await acquireBoundPod(deadline, previousPodUid);
2294
+ const { binding, pod } = await acquireBoundPod(deadline, policy);
2295
+ const token = pod.uid;
2296
+ // From the readiness GET the bind just made, not from a read of its
2297
+ // own — see `lastSandbox`. Fixed for the handle's life: a later bind
2298
+ // that found a DIFFERENT object would have been refused by
2299
+ // `refreshBoundPod` long before it got here.
2300
+ sandboxUid ??= lastSandbox?.metadata?.uid;
610
2301
  // Recorded before the probe, not after: a probe that refuses suspends
611
2302
  // this pod, and the resume that follows has to know which pod it is
612
2303
  // waiting to see replaced.
613
2304
  podUid = token;
614
- const address = resolveAgentAddress(binding, options.agentPort, token);
2305
+ // A new pod is a new agent process. Clearing it is what makes the
2306
+ // caller's own suspend/resume silent: the first reply from the new
2307
+ // guest establishes an identity rather than differing from the old
2308
+ // one's.
2309
+ guestBootId = undefined;
2310
+ sessionSeq += 1;
2311
+ // Written together with the counter, so that `podUid` belongs to THIS
2312
+ // session for as long as this session is the live one — which is what
2313
+ // {@link boundToPod} asks, and what makes a handle mid-resume report
2314
+ // the pod it has just bound rather than nothing at all.
2315
+ podSession = sessionSeq;
2316
+ const generation = sessionSeq;
2317
+ const address = resolveAgentAddress(binding, options.agentPort, token, {
2318
+ mode: options.agentAddress,
2319
+ ...(pod.podIP !== undefined ? { podIP: pod.podIP } : {}),
2320
+ });
615
2321
  // A box rather than a `let`, so the callback below can name the handle
616
2322
  // it belongs to before that handle exists. Nothing can call it in
617
2323
  // between: `release` is reachable only THROUGH the handle.
618
2324
  const own = {};
2325
+ const transport = new KubernetesAgentTransport(address, {
2326
+ // The backend opts in to the stream heartbeat; the transport
2327
+ // option it sets stays undefined for every other tier.
2328
+ heartbeatMs: options.streamHeartbeatMs,
2329
+ // Installed for BOTH address modes now. Under `'pod-ip'` it
2330
+ // follows an address that died with its pod; under the default
2331
+ // `'service'` mode the address is fine and the TOKEN is what
2332
+ // moved — see {@link refreshBoundPod}.
2333
+ refreshHandle: refreshBoundPod(binding, generation, address),
2334
+ // And the one fact no address and no token can carry: which
2335
+ // agent PROCESS answered. The two values beside it say whose
2336
+ // answer it is — this session's, and this session's CURRENT pod's
2337
+ // — because neither a retired session's transport nor the wire a
2338
+ // rebind left behind stops answering when it stops counting.
2339
+ onGuestReply: (reply, from) => observeGuestReply(generation, from, reply),
2340
+ });
619
2341
  const inner = buildKubernetesSandbox({
620
2342
  name,
621
2343
  rootDir: options.rootDir,
622
- transport: new KubernetesAgentTransport(address),
2344
+ transport,
623
2345
  // Deliberately NOT `deleteSandbox` — see {@link retireSession}. On
624
2346
  // the task path `release` is a DELETE because the object is
625
2347
  // disposable; here the same callback would erase the caller's disk
@@ -627,47 +2349,255 @@ async function openWorkspaceHandle(options) {
627
2349
  release: async (releaseSignal) => {
628
2350
  await retireSession(own.handle, releaseSignal);
629
2351
  },
2352
+ // And `release` is not reached from the unconfirmed-cancellation
2353
+ // path at all any more: this hook answers it instead, keeping the
2354
+ // pod — see {@link keepPodOnUnconfirmedCancellation}.
2355
+ onUnconfirmedCancellation: async (error) => await keepPodOnUnconfirmedCancellation(transport, error),
630
2356
  // No `renew`, and so no lease loop: a workspace carries no expiry.
631
2357
  });
632
2358
  own.handle = inner;
2359
+ sessionTransports.set(inner, transport);
633
2360
  // The same probe every task acquire runs, on every resume as well as on
634
2361
  // create — a resumed pod is a new pod, from a possibly re-pulled image,
635
2362
  // and "it was deprivileged last week" is not a check.
636
2363
  await probeSandboxPrivileges(inner, name, resolveProbeTimeoutMs(readiness.timeoutMs), signal);
2364
+ // After the probe, so a workspace that is about to be refused never
2365
+ // spends a round trip per volume on an identity nobody will read; and
2366
+ // only once per handle, for the reason `readVolumeClaimUids` gives.
2367
+ await readVolumeClaimUids(signal);
637
2368
  return inner;
638
2369
  };
2370
+ /**
2371
+ * Drop the live session, and with it the right of anything still holding
2372
+ * that session's transport to report a pod.
2373
+ *
2374
+ * The two happen together or the guard in {@link refreshBoundPod} is a
2375
+ * lie: a retired session's transport can still be unwinding a call, and a
2376
+ * re-read that lands after the drop would write `podUid` for a session
2377
+ * nothing is using — including in the window before a suspend stamps
2378
+ * `retiredPodUid` from it.
2379
+ */
2380
+ const dropSession = () => {
2381
+ session = undefined;
2382
+ sessionSeq += 1;
2383
+ };
639
2384
  /** Kill and await every terminal this handle returned. */
640
2385
  const reapTerminals = async () => {
641
2386
  const open = [...terminals];
642
2387
  for (const terminal of open)
643
- terminal.kill('SIGKILL');
2388
+ releaseTerminal(terminal);
644
2389
  await Promise.allSettled(open.map((terminal) => terminal.exited));
645
2390
  terminals.clear();
646
2391
  };
2392
+ /**
2393
+ * What an execution whose cancellation could not be CONFIRMED does to a
2394
+ * workspace: nothing to the cluster, and one bounded question to the
2395
+ * agent.
2396
+ *
2397
+ * `buildKubernetesSandbox` used to retire the sandbox here on its own
2398
+ * initiative — the shared controller's rule is that a command of unknown
2399
+ * state makes the pod unusable, and on a task sandbox retiring means
2400
+ * DELETING a disposable object with a scratch disk. On a workspace the
2401
+ * same decision reached {@link retireSession} and sent an
2402
+ * `operatingMode: Suspended` patch, which makes the controller delete the
2403
+ * pod. So eight seconds of network loss under one `exec()` — or a pod
2404
+ * evicted under an in-flight command, which can never confirm anything —
2405
+ * took every holder's terminals, dev servers and running commands away,
2406
+ * on a decision no caller issued and no host-side lock could prevent.
2407
+ *
2408
+ * The pod is therefore KEPT, and what the host is told instead is the one
2409
+ * thing the error cannot carry: whether the agent is still serving
2410
+ * (`ok`), has fenced itself (`retiring`), or could not be reached at all.
2411
+ * Today's `healthz()` boolean collapses the last two, which is why this
2412
+ * asks {@link KubernetesAgentTransport.agentHealth} instead.
2413
+ *
2414
+ * The `exec()` still rejects with `RemoteCancellationUnknownError`, now
2415
+ * carrying `accepted: false` with `reason: 'workspace-kept'` so a host
2416
+ * cannot read it as a patch that was attempted and failed. A host that
2417
+ * wants the old behaviour calls `suspend()` from the callback.
2418
+ */
2419
+ const keepPodOnUnconfirmedCancellation = async (transport, error) => {
2420
+ // The guest THIS command reserved against, kept per execution by the
2421
+ // transport. Never the handle's own pod and process: a restart the
2422
+ // handle survived an hour ago is not evidence about a command started
2423
+ // after it, and a pod another call has ALREADY rebound to is not the
2424
+ // pod this command's processes died with. Reading the handle would
2425
+ // answer `container-restarted` for every later unconfirmed
2426
+ // cancellation in the session and would name the replacement as the
2427
+ // guest that died.
2428
+ const reserved = guestWhenReserved(error);
2429
+ const previous = identityWhenReserved(reserved);
2430
+ const agent = await diagnoseAgent(transport);
2431
+ const { evidence, current } = await diagnoseGuest(previous, reserved?.guestBootId);
2432
+ // Recorded against the error itself, so the call that is about to
2433
+ // receive it can name the diagnosis rather than repeat the work —
2434
+ // see {@link admitted}. A WeakMap rather than a field on the error:
2435
+ // the class belongs to the shared controller and this is one
2436
+ // backend's evidence about it.
2437
+ if (evidence !== 'same-guest' && evidence !== 'unknown') {
2438
+ guestGoneEvidence.set(error, { evidence, previous, current });
2439
+ }
2440
+ try {
2441
+ options.onCancellationUnconfirmed?.({ error, agent, guest: evidence, previous, current });
2442
+ }
2443
+ catch {
2444
+ // A host's callback is not allowed to change the error the caller
2445
+ // is already receiving, and a throwing one must not become an
2446
+ // unaccepted retirement carrying somebody else's failure.
2447
+ }
2448
+ return { accepted: false, reason: WORKSPACE_KEPT_REASON };
2449
+ };
2450
+ /**
2451
+ * Was the guest the command was running in replaced while it ran?
2452
+ *
2453
+ * This is the DIAGNOSIS half of the unconfirmed-cancellation rule, and it
2454
+ * is deliberately only that. Nothing here patches, deletes or suspends
2455
+ * anything — a `Suspended` patch cannot stop a command whose pod is
2456
+ * already gone, and sending one would take the replacement pod away from
2457
+ * every other holder. What the host gets instead is the evidence:
2458
+ * `healthz` (beside this, in {@link diagnoseAgent}) says whether SOME
2459
+ * agent is serving at that address; this says whether it is the same one.
2460
+ *
2461
+ * The boot id is asked first because it is free — it rode in on replies
2462
+ * the handle already received, the `cancel-execution` refusal included —
2463
+ * and because it is the only evidence for the case no API read can see: a
2464
+ * container restarted in place keeps the pod, the uid and the token, and
2465
+ * changes nothing an API server would report. `reservedIn` is the guest
2466
+ * THIS command reserved against, not the one this session opened on: the
2467
+ * question is whether the command's own guest went away, so a restart the
2468
+ * handle survived before the command started is not evidence about it.
2469
+ *
2470
+ * Bounded by its own short deadline and run WITHOUT the caller's signal,
2471
+ * for the reason {@link diagnoseAgent} gives: the caller's signal is
2472
+ * quite possibly what started the cancellation.
2473
+ *
2474
+ * Every failure answers `unknown`, never a guess. This evidence decides
2475
+ * whether the caller is told its guest is gone, and saying so on the
2476
+ * strength of one 500 on a pod GET would be worse than saying nothing.
2477
+ */
2478
+ const diagnoseGuest = async (previous, reservedIn) => {
2479
+ // A handle that is not RUNNING is inside a transition it asked for:
2480
+ // the pod going away IS that transition, and a caller who typed
2481
+ // `suspend()` under a command of their own is not being told their
2482
+ // guest vanished. Only a handle that still believes it is running has
2483
+ // a question here — which is the case the whole diagnosis is for.
2484
+ if (state !== 'running')
2485
+ return { evidence: 'unknown', current: identityNow() };
2486
+ const bound = previous.podUid;
2487
+ if (bound === undefined)
2488
+ return { evidence: 'unknown', current: identityNow() };
2489
+ // The one piece of evidence that needs no API read: the handle has
2490
+ // heard from a DIFFERENT agent process than the one this command
2491
+ // reserved against. Computed here and CONSULTED below, after the pod
2492
+ // read, because it cannot tell a container restarted in place from a
2493
+ // pod another call rebound to — both leave the handle talking to a
2494
+ // process the command never reserved against, and only the pod read
2495
+ // says which happened. It is the answer when no read succeeds at all.
2496
+ const restartedInPlace = reservedIn !== undefined && guestBootId !== undefined && guestBootId !== reservedIn;
2497
+ try {
2498
+ return await new OperationDeadline(CANCELLATION_DIAGNOSIS_TIMEOUT_MS, `kubernetes workspace ${name} guest diagnosis`).run(async (deadlineSignal) => {
2499
+ const sandbox = await readSandboxObject(target, deadlineSignal);
2500
+ // No object, or one somebody replaced: not this workspace any
2501
+ // more, and not something to claim a verdict about here — the
2502
+ // next call's rebind refuses it by name.
2503
+ if (sandbox === undefined || sandbox.metadata?.uid !== previous.sandboxUid) {
2504
+ return { evidence: 'unknown', current: identityNow() };
2505
+ }
2506
+ // Somebody suspended it, so there is genuinely no pod — but
2507
+ // naming the mode is #473's job and not this one's. The
2508
+ // evidence is reported to the host's callback; the ERROR the
2509
+ // caller receives is decided by {@link admitted}, which asks
2510
+ // about a foreign suspend BEFORE it renames anything, so this
2511
+ // answer never pre-empts `KubernetesWorkspaceSuspendedError`.
2512
+ if (operatingModeOf(sandbox) === 'Suspended') {
2513
+ return { evidence: 'pod-gone', current: guestSeen(undefined) };
2514
+ }
2515
+ const binding = bindingFromSandbox(sandbox);
2516
+ if (binding === undefined) {
2517
+ return { evidence: 'pod-gone', current: guestSeen(undefined) };
2518
+ }
2519
+ const pod = await readBoundPod(client, namespace, binding, deadlineSignal);
2520
+ // A DIFFERENT pod outranks the boot id. Both say the command's
2521
+ // guest is gone; only this one says where it went, and calling
2522
+ // a replacement a restarted container would tell a host the
2523
+ // address and the token still work when neither does.
2524
+ if (pod.uid !== bound) {
2525
+ return { evidence: 'pod-replaced', current: guestSeen(pod.uid) };
2526
+ }
2527
+ if (restartedInPlace) {
2528
+ return { evidence: 'container-restarted', current: guestSeen(pod.uid) };
2529
+ }
2530
+ return { evidence: 'same-guest', current: guestSeen(pod.uid) };
2531
+ });
2532
+ }
2533
+ catch {
2534
+ // Including `readBoundPod`'s own refusal, which means "no live pod
2535
+ // AND no pod its selector matched" but is also what an API server
2536
+ // returning 500 produces. One error for two facts is not evidence
2537
+ // — but a boot id that already moved is evidence of its own, and
2538
+ // it was never the API server's to confirm.
2539
+ if (restartedInPlace)
2540
+ return { evidence: 'container-restarted', current: identityNow() };
2541
+ return { evidence: 'unknown', current: identityNow() };
2542
+ }
2543
+ };
2544
+ /**
2545
+ * One bounded `healthz` over a fresh connection, and never anything else.
2546
+ *
2547
+ * Bounded by its own short deadline rather than by the caller's: the
2548
+ * caller's signal is quite possibly what started the cancellation in the
2549
+ * first place, and a diagnosis that inherited it would report every
2550
+ * aborted call as an unreachable agent. Fresh, because every request on
2551
+ * this transport dials fresh — there is no cached socket to reuse and no
2552
+ * pooled connection whose state could answer for the pod.
2553
+ *
2554
+ * A failure is an ANSWER here, not an error to propagate: an agent that
2555
+ * cannot be reached is exactly the `unreachable` case, and this whole
2556
+ * routine exists to report rather than to decide.
2557
+ *
2558
+ * `unreachable` also covers a reply that is neither: the agent's
2559
+ * connection gate refuses a `healthz` that arrived over one
2560
+ * unauthenticated connection too many with a named `{ ok: false, error }`
2561
+ * and no fence flag. The probe got no answer about the agent's health
2562
+ * there, which is what `unreachable` means — reading it as `retiring`
2563
+ * would advise a `suspend()` on a healthy pod, and as `ok` would claim a
2564
+ * pod is serving on a reply that said the opposite.
2565
+ */
2566
+ const diagnoseAgent = async (transport) => {
2567
+ try {
2568
+ const health = await new OperationDeadline(CANCELLATION_DIAGNOSIS_TIMEOUT_MS, `kubernetes workspace ${name} agent diagnosis`).run(async (deadlineSignal) => await transport.agentHealth(deadlineSignal));
2569
+ return health.retiring ? 'retiring' : health.ok ? 'ok' : 'unreachable';
2570
+ }
2571
+ catch {
2572
+ return 'unreachable';
2573
+ }
2574
+ };
647
2575
  /**
648
2576
  * Retire one session's pod WITHOUT deleting anything.
649
2577
  *
650
2578
  * This is the inner handle's `release` on a workspace, and the difference
651
2579
  * from the task path is the entire reason that callback is a parameter
652
- * rather than a DELETE both paths share. `buildKubernetesSandbox` calls
653
- * `release` on its own initiative: an execution whose cancellation the
654
- * guest could not confirm (`RemoteCancellationUnknownError` — a wedged
655
- * agent, a partitioned pod) leaves a command of unknown state in that
656
- * pod, so the pod stops being reusable and the handle retires it. For a
657
- * task sandbox retiring IS deleting, because the object is disposable and
658
- * its disk is scratch. Here it is not: the DELETE cascades to the PVC, and
659
- * a cancel that went unconfirmed for eight seconds would take a month of
660
- * the caller's files with it. `deleteDisk: true` is the only thing in this
661
- * module that removes a disk, and a failure path is not allowed to become
662
- * a second one — that is the invariant the whole file is built around.
2580
+ * rather than a DELETE both paths share: for a task sandbox retiring IS
2581
+ * deleting, because the object is disposable and its disk is scratch.
2582
+ * Here it is not — the DELETE cascades to the PVC, and nothing but a
2583
+ * caller naming a disk (`deleteDisk: true`, or
2584
+ * {@link deleteKubernetesWorkspace}) is allowed to remove one. That is the
2585
+ * invariant the whole file is built around.
663
2586
  *
664
2587
  * So the pod is retired the way `suspend()` retires one, with the same
665
2588
  * `operatingMode: Suspended` patch, and the workspace is left
666
2589
  * `suspending`: nothing is admitted, the disk is untouched, and `resume()`
667
2590
  * brings up a fresh pod. A patch that FAILS is not swallowed — it travels
668
- * back out through `retire()` as `retirement.accepted === false` on the
669
- * error the caller is already receiving, which is what that observation
670
- * exists to say.
2591
+ * back out through the inner handle's `destroy()` on the error the caller
2592
+ * is already receiving.
2593
+ *
2594
+ * The path that used to reach this — an unconfirmed cancellation — no
2595
+ * longer does: see {@link keepPodOnUnconfirmedCancellation}. What is left
2596
+ * is the inner handle's own `destroy()`, and the workspace's delete drops
2597
+ * the session before it calls that, so in practice this is a guard rather
2598
+ * than a verb. It stays because the callback is required and because a
2599
+ * `release` that silently did nothing would be a worse answer than one
2600
+ * that does the safe thing if a future path ever reaches it.
671
2601
  *
672
2602
  * `retiring` is the handle the callback was built for. When it is not the
673
2603
  * current session there is nothing to retire and this is a no-op: a resume
@@ -675,14 +2605,14 @@ async function openWorkspaceHandle(options) {
675
2605
  * DELETE — which must not be preceded by a suspend patch, and says so by
676
2606
  * dropping it.
677
2607
  *
678
- * It never touches the transition queue, and must not: `retire()` is
679
- * awaited inside the failing `exec()`, and that exec can be the privilege
680
- * probe of the resume currently HOLDING the queue.
2608
+ * It never touches the transition queue, and must not: it is awaited
2609
+ * inside a failing call, and that call can be the privilege probe of the
2610
+ * resume currently HOLDING the queue.
681
2611
  */
682
2612
  const retireSession = async (retiring, signal) => {
683
2613
  if (retiring === undefined || retiring !== session)
684
2614
  return;
685
- session = undefined;
2615
+ dropSession();
686
2616
  // Some transition already owns this pod — the suspend that is patching
687
2617
  // it away, or a create/resume cleanup — and commits its own state when
688
2618
  // its own request settles. Dropping the session is all there is to do.
@@ -697,60 +2627,15 @@ async function openWorkspaceHandle(options) {
697
2627
  // not decoration — a detached chain that rejects with no handler takes
698
2628
  // the host process down with it.
699
2629
  void reapTerminals().catch(() => undefined);
700
- await client.request('PATCH', sandboxPath(namespace, name), SUSPEND_PATCH, signal);
2630
+ // Fenced by whatever this handle holds, like every other write it
2631
+ // sends on its own: a superseded handle's release must not suspend the
2632
+ // pod the new holder is using. The refusal travels out on the error
2633
+ // the caller is already receiving, exactly as a refused patch does.
2634
+ await writeOperatingMode(target, 'retire', 'Suspended', heldEpoch, signal);
701
2635
  // The patch landed, so the controller is taking this pod away and the
702
2636
  // next resume must see it replaced rather than bind it.
703
2637
  retiredPodUid = podUid;
704
2638
  };
705
- /**
706
- * True once the pod has actually stopped. The POD is asked, and nothing
707
- * else is consulted or believed.
708
- *
709
- * The cheap-looking alternative — the Sandbox's own `Suspended` condition
710
- * — is unusable, and upstream says so itself: "the controller does not
711
- * currently remove this condition when the Sandbox is resumed, so a stale
712
- * Suspended condition may linger after operatingMode returns to Running.
713
- * Consumers should treat Ready as the authoritative signal and not infer
714
- * the live operating state from the mere presence of this condition"
715
- * (`sandbox_types.go`). Reading it would make the second and every later
716
- * suspend of the same workspace return immediately, on a True left behind
717
- * by the previous one, while the guest was still running and still
718
- * writing to the caller's disk — and a `resume()` issued straight after
719
- * such a false suspend could bind to the pod that is about to be deleted.
720
- *
721
- * A `deletionTimestamp` is not the answer either: it is set the moment the
722
- * DELETE is accepted, and the container goes on running until it exits or
723
- * `terminationGracePeriodSeconds` expires. Gone (404) or stopped
724
- * ({@link isPodStopped}) — those are the only two states that mean the
725
- * disk is quiesced.
726
- */
727
- const isPodRetired = async (signal) => {
728
- try {
729
- const pod = await client.request('GET', podPath(namespace, name), undefined, signal);
730
- return isPodStopped(pod);
731
- }
732
- catch (err) {
733
- if (err instanceof KubernetesAlreadyGoneError)
734
- return true;
735
- throw err;
736
- }
737
- };
738
- const awaitPodRetired = async (signal) => {
739
- const deadline = new OperationDeadline(readiness.timeoutMs, `kubernetes workspace ${name} suspend`, signal);
740
- while (deadline.remainingMs() > 0) {
741
- try {
742
- if (await deadline.run(isPodRetired))
743
- return;
744
- await deadline.delay(readiness.pollIntervalMs);
745
- }
746
- catch (err) {
747
- if (err instanceof OperationDeadlineExpired)
748
- break;
749
- throw err;
750
- }
751
- }
752
- throw new KubernetesWorkspaceSuspendTimeoutError(workspaceId, name, readiness.timeoutMs);
753
- };
754
2639
  /**
755
2640
  * The suspend, minus the queue — every caller here is already inside it.
756
2641
  *
@@ -761,7 +2646,7 @@ async function openWorkspaceHandle(options) {
761
2646
  * them with the transition still unfinished. Neither can be returned from
762
2647
  * by a later `suspend()` as though it had worked.
763
2648
  */
764
- const suspendNow = async (signal) => {
2649
+ const suspendNow = async (signal, epoch, quiesceRequest, flushRequest) => {
765
2650
  if (state === 'deleted')
766
2651
  throw new KubernetesSandboxDestroyedError('suspend', name);
767
2652
  // `suspending` deliberately falls through: the patch is re-sent and
@@ -769,12 +2654,50 @@ async function openWorkspaceHandle(options) {
769
2654
  if (state === 'suspended')
770
2655
  return;
771
2656
  const before = state;
2657
+ // BEFORE the terminals are reaped, which is the whole point of doing
2658
+ // the read here rather than letting the patch below carry the refusal
2659
+ // on its own: `reapTerminals` SIGKILLs every session this handle
2660
+ // handed out, and a superseded holder that learned it was superseded
2661
+ // from the write would already have taken its own caller's terminals
2662
+ // away on a write that never applied. The reading is reused as the
2663
+ // patch's condition, so the gate costs no extra round trip.
2664
+ const gate = await readHolderEpochGate(target, 'suspend', epoch, signal);
772
2665
  // A terminal owns an interactive process tree in a pod that is about
773
2666
  // to be taken away, so it is stopped first — and stays stopped even if
774
2667
  // the patch below fails. `suspend()` is a declaration that nobody is
775
2668
  // using this workspace; killing the sessions that say otherwise is the
776
2669
  // point of it rather than a cost of it.
777
2670
  await reapTerminals();
2671
+ // And then, if the caller asked for it, everything else in the guest:
2672
+ // the terminals another handle opened, the commands already running,
2673
+ // and the programs that left every session this handle knows about.
2674
+ // HERE and nowhere later — `admit` refuses every call the moment the
2675
+ // state leaves `running` below, so a quiesce after that point could not
2676
+ // reach the pod it is quiescing. A quiesce that cannot be confirmed
2677
+ // throws from here, before `state` has moved and before any patch has
2678
+ // been sent: the pod is still running, this handle can still serve it,
2679
+ // and the caller is told which pid would not stop.
2680
+ //
2681
+ // A re-sent suspend (`state === 'suspending'`, the fall-through above)
2682
+ // does NOT re-quiesce: its patch has already landed, its session was
2683
+ // dropped with it, and there is no admitted call left to ask through.
2684
+ if (quiesceRequest !== undefined && quiesceRequest !== false && state === 'running') {
2685
+ await quiesceBeforeSuspend(quiesceRequest, signal);
2686
+ }
2687
+ // And then the disk, which is the thing this patch is about to take
2688
+ // away the only means of reading. HERE for the same reason the
2689
+ // quiesce is here — after the guest is quiet, so the flush covers
2690
+ // what the processes just stopped had written, and before `state`
2691
+ // moves, because nothing is admitted afterwards. A flush the guest
2692
+ // answered and could not confirm throws from here, with no patch
2693
+ // sent: the pod is still running and the caller can decide.
2694
+ //
2695
+ // A re-sent suspend does not re-flush, for the reason a re-sent
2696
+ // suspend does not re-quiesce: its patch has already landed and
2697
+ // there is no admitted call left to ask through.
2698
+ if (flushRequest !== false && state === 'running') {
2699
+ await flushBeforeSuspend(flushRequest ?? true, signal);
2700
+ }
778
2701
  // From here on nothing new is admitted: the pod is going away, and a
779
2702
  // call let through would dial an address that still resolves — the
780
2703
  // Service outlives the pod — and hang on a connect timeout naming
@@ -782,7 +2705,7 @@ async function openWorkspaceHandle(options) {
782
2705
  // straight back.
783
2706
  state = 'suspending';
784
2707
  try {
785
- await client.request('PATCH', sandboxPath(namespace, name), SUSPEND_PATCH, signal);
2708
+ await writeOperatingMode(target, 'suspend', 'Suspended', epoch, signal, gate);
786
2709
  }
787
2710
  catch (err) {
788
2711
  // Nothing was changed on the cluster, so nothing is changed here:
@@ -791,29 +2714,130 @@ async function openWorkspaceHandle(options) {
791
2714
  // suspend() and destroy() would return on that mark without ever
792
2715
  // re-sending the patch, and the pod would run until somebody
793
2716
  // noticed the bill.
794
- state = before;
2717
+ //
2718
+ // Unless the SESSION went away while this call was still short of
2719
+ // the patch, which is the one case where that premise is false.
2720
+ // The quiesce and the flush above are admitted calls, and an
2721
+ // admitted call that fails asks once whether somebody else has
2722
+ // suspended this workspace; when the answer is yes,
2723
+ // `noticeSuspendedElsewhere` records `suspending` and DROPS the
2724
+ // session, and `flushBeforeSuspend` reports that rather than
2725
+ // refusing — so the patch can be reached with no session behind
2726
+ // it, and then be refused by the epoch the other holder bumped.
2727
+ // Writing `running` back over that would leave a handle nothing
2728
+ // can use and nothing can recover: `admit` refuses every call
2729
+ // (there is no session), `status` reads 'destroyed' for the same
2730
+ // reason, `suspended` reads FALSE because only `state` decides it,
2731
+ // and `resume()` returns early on 'running' without starting one.
2732
+ // What `noticeSuspendedElsewhere` wrote is what is true, so it
2733
+ // stands.
2734
+ state = session === undefined ? 'suspending' : before;
795
2735
  throw err;
796
2736
  }
797
2737
  // The patch landed, so this pod is the controller's to remove and the
798
2738
  // next resume must see it replaced rather than bind it — whether or
799
2739
  // not the wait below is still around when it goes.
800
2740
  retiredPodUid = podUid;
801
- session = undefined;
802
- await awaitPodRetired(signal);
2741
+ if (epoch !== undefined)
2742
+ heldEpoch = epoch;
2743
+ dropSession();
2744
+ await awaitPodRetired(client, namespace, name, workspaceId, readiness, signal);
803
2745
  state = 'suspended';
804
2746
  };
805
- const resumeNow = async (signal) => {
2747
+ /**
2748
+ * Wake the workspace if it is asleep, then bring a session up on it.
2749
+ *
2750
+ * The mode is READ before the patch rather than patched blind, for the
2751
+ * same reason {@link adoptExistingWorkspace} reads it: a workspace this
2752
+ * handle believes is suspended may already have been resumed by another
2753
+ * process, whose pod is serving its terminals right now. Patching Running
2754
+ * over Running would change nothing on the cluster but would make this
2755
+ * call the apparent author of a mode change it did not make — and a start
2756
+ * that then failed would suspend somebody else's live pod. It also keeps
2757
+ * {@link OPERATING_MODE_CHANGED_AT_ANNOTATION_KEY} honest: the annotation
2758
+ * says when the mode last CHANGED, and a no-op patch would restamp it.
2759
+ *
2760
+ * The extra GET is one round trip on a rare, explicit transition, which
2761
+ * is the same trade the adopt path already makes.
2762
+ */
2763
+ const resumeNow = async (signal, onStartFailure = defaultStartFailure, epoch, refreshPodTemplate = false) => {
806
2764
  if (state === 'deleted')
807
2765
  throw new KubernetesSandboxDestroyedError('resume', name);
808
- if (state === 'running')
2766
+ if (state === 'running') {
2767
+ // Resuming a running workspace sends nothing — unless it carries
2768
+ // an epoch, in which case the ONE thing it is asking for is the
2769
+ // thing that still has to happen: take this workspace over. That
2770
+ // is a write, even though the mode does not move, and a `resume()`
2771
+ // that silently dropped it would leave the new holder unfenced.
2772
+ if (epoch === undefined)
2773
+ return;
2774
+ await writeOperatingMode(target, 'resume', undefined, epoch, signal);
2775
+ heldEpoch = epoch;
809
2776
  return;
2777
+ }
810
2778
  // The pod a landed suspend patch took away is one this resume must see
811
2779
  // replaced rather than bound — see `acquireBoundPod` and
812
2780
  // `retiredPodUid`. Where there is none to exclude this is one poll
813
2781
  // plus one read, exactly as it is on create.
814
2782
  const replacing = retiredPodUid;
815
- await client.request('PATCH', sandboxPath(namespace, name), RESUME_PATCH, signal);
816
- session = await startSessionOrSuspend('resume', signal, replacing);
2783
+ // One GET, read twice: the mode decides whether a patch is sent at all
2784
+ // and the epoch decides whether it may be. Reading both off the same
2785
+ // object rather than off two GETs is what keeps an unfenced resume's
2786
+ // request log identical to what it always was.
2787
+ const current = await readSandboxObject(target, signal);
2788
+ let woke = operatingModeOf(current) === 'Suspended';
2789
+ const reading = readHolderEpoch(current?.metadata);
2790
+ if (epoch !== undefined)
2791
+ assertHolderEpochAllows(target, 'resume', epoch, reading);
2792
+ // Off the GET this path already makes, so an unfenced resume that asks
2793
+ // for no refresh sends and reads exactly what it always did: whatever
2794
+ // the object says it was built from, including a refresh another
2795
+ // process applied while this handle slept.
2796
+ templateRevision = readPodTemplateHash(current?.metadata);
2797
+ if (woke && refreshPodTemplate) {
2798
+ // The template is re-read HERE rather than remembered from the
2799
+ // open: the whole point of the option is to pick up an edit made
2800
+ // since, and a cached copy would refresh a workspace onto the
2801
+ // template as it stood when the handle was created.
2802
+ const template = await readSandboxTemplate(client, namespace, options.sandboxTemplateName, signal);
2803
+ const refresh = buildPodTemplateRefresh(template, options.sandboxTemplateName, options.runtimeClassName, options.podLabels);
2804
+ currentTemplateHash = refresh.hash;
2805
+ // Before any patch, and against the SANDBOX's own disk — both of
2806
+ // them, in the same order as the adopt path.
2807
+ const source = `SandboxTemplate ${options.sandboxTemplateName} in namespace ${namespace}, refreshing Sandbox ${name}`;
2808
+ assertRefreshedTemplateClaimsTheSameDisks(source, refresh.volumeClaimTemplates, current?.spec?.volumeClaimTemplates);
2809
+ assertBlockModeWorkspaceDisk(source, refresh.podTemplate, current?.spec?.volumeClaimTemplates);
2810
+ const result = await writeRefreshedPodTemplate(target, 'resume', refresh, epoch, reading, signal);
2811
+ if (result.applied) {
2812
+ templateRevision = refresh.hash;
2813
+ }
2814
+ else {
2815
+ // Another process resumed it first. Nothing was written, so
2816
+ // this call did not wake the workspace and a start that fails
2817
+ // must not put somebody else's live pod back to sleep.
2818
+ woke = false;
2819
+ templateRevision = readPodTemplateHash(result.sandbox.metadata);
2820
+ if (epoch !== undefined) {
2821
+ await writeOperatingMode(target, 'resume', undefined, epoch, signal, readHolderEpoch(result.sandbox.metadata));
2822
+ }
2823
+ }
2824
+ }
2825
+ else if (woke) {
2826
+ await writeOperatingMode(target, 'resume', 'Running', epoch, signal, reading);
2827
+ }
2828
+ else if (epoch !== undefined) {
2829
+ // Somebody else resumed it first, so there is no mode change to
2830
+ // make — but this call is still the one taking the workspace over,
2831
+ // and the stamp is what says so.
2832
+ await writeOperatingMode(target, 'resume', undefined, epoch, signal, reading);
2833
+ }
2834
+ if (epoch !== undefined)
2835
+ heldEpoch = epoch;
2836
+ session = await startSessionOrSuspend({
2837
+ transition: 'resume',
2838
+ awaitReplacement: replacing !== undefined,
2839
+ ...(replacing !== undefined ? { retiring: replacing } : {}),
2840
+ }, { woke, policy: onStartFailure, ...(heldEpoch !== undefined ? { epoch: heldEpoch } : {}) }, signal);
817
2841
  state = 'running';
818
2842
  // Bound, probed and serving: the pod that was excluded is one no
819
2843
  // answer can name any more, and the next suspend records its own.
@@ -827,14 +2851,40 @@ async function openWorkspaceHandle(options) {
827
2851
  * instead of patching and waiting all over again once it finishes. It is
828
2852
  * released inside the run, before the promise handed to callers settles,
829
2853
  * so a caller that awaits and then suspends again gets a fresh attempt.
2854
+ *
2855
+ * Joining is only honest while the transition in flight does everything
2856
+ * the joining caller asked for. It carries the first caller's signal and
2857
+ * epoch, and that is what sharing means — but a caller that asked for a
2858
+ * `quiesce` is about to trust a capture, and a suspend already in flight
2859
+ * WITHOUT one is on its way to patching over a guest nothing stopped. A
2860
+ * quiesce cannot be added to it afterwards either: `admit` refuses every
2861
+ * call the moment that transition's state leaves `running`. So such a
2862
+ * caller is refused, by the same class an unconfirmable quiesce rejects
2863
+ * with and for the same reason — everything this feature cannot deliver,
2864
+ * it says out loud.
830
2865
  */
831
- const suspendShared = (signal) => {
832
- pendingSuspend ??= serialise(async () => {
2866
+ const suspendShared = (signal, epoch, quiesceRequest, flushRequest) => {
2867
+ const wantsQuiesce = quiesceRequest !== undefined && quiesceRequest !== false;
2868
+ const wantsFlush = flushRequest !== false;
2869
+ if (pendingSuspend !== undefined) {
2870
+ if (wantsFlush && !pendingSuspendFlushes) {
2871
+ return Promise.reject(new KubernetesFlushUnconfirmedError('suspend_already_in_flight', `kubernetes: workspace '${workspaceId}' is already suspending under a call that asked for no flush ('flush: false'), and a flush cannot be added to a transition in flight — once that transition's state leaves 'running', no call is admitted to ask the guest for anything. Nothing was sent on this call's behalf: await the suspend in flight and treat the disk as one whose last writes may not have reached it, or flush() before the suspend next time. Refused rather than joined, because a caller that let the flush default on is about to trust the disk.`));
2872
+ }
2873
+ if (wantsQuiesce && !pendingSuspendQuiesces) {
2874
+ return Promise.reject(new KubernetesQuiesceUnconfirmedError('suspend_already_in_flight', `kubernetes: workspace '${workspaceId}' is already suspending under a call that did not ask for a quiesce, and a quiesce cannot be added to a transition in flight — once that transition's state leaves 'running', no call is admitted to stop anything in the guest. Nothing was sent and nothing was stopped on this call's behalf: await the suspend in flight and treat the disk as one that was written to, or call quiesce() before the suspend next time. Refused rather than joined, because a caller that passed 'quiesce' is about to trust a capture.`));
2875
+ }
2876
+ return pendingSuspend;
2877
+ }
2878
+ pendingSuspendQuiesces = wantsQuiesce;
2879
+ pendingSuspendFlushes = wantsFlush;
2880
+ pendingSuspend = serialise(async () => {
833
2881
  try {
834
- await suspendNow(signal);
2882
+ await suspendNow(signal, epoch, quiesceRequest, flushRequest);
835
2883
  }
836
2884
  finally {
837
2885
  pendingSuspend = undefined;
2886
+ pendingSuspendQuiesces = false;
2887
+ pendingSuspendFlushes = false;
838
2888
  }
839
2889
  });
840
2890
  return pendingSuspend;
@@ -858,7 +2908,7 @@ async function openWorkspaceHandle(options) {
858
2908
  * deleteDisk: true })` the queue admitted first must be a no-op in that
859
2909
  * order too, which is the order it is most likely to be written in.
860
2910
  */
861
- const destroyBySuspending = async (signal) => {
2911
+ const destroyBySuspending = async (signal, epoch, quiesceRequest, flushRequest) => {
862
2912
  // Read through a call on both sides. `state` is assigned from other
863
2913
  // closures, which the checker cannot see, so it takes the first
864
2914
  // comparison as narrowing the second out of existence — and the second
@@ -868,7 +2918,7 @@ async function openWorkspaceHandle(options) {
868
2918
  if (gone())
869
2919
  return;
870
2920
  try {
871
- await suspendShared(signal);
2921
+ await suspendShared(signal, epoch, quiesceRequest, flushRequest);
872
2922
  }
873
2923
  catch (err) {
874
2924
  if (gone() && err instanceof KubernetesSandboxDestroyedError)
@@ -890,13 +2940,42 @@ async function openWorkspaceHandle(options) {
890
2940
  const deleteNow = async (destroyOptions) => {
891
2941
  if (state === 'deleted')
892
2942
  return;
2943
+ const epoch = destroyOptions?.epoch ?? heldEpoch;
2944
+ // Before the terminals are reaped and before the session is torn
2945
+ // down, for the same reason `suspendNow` gates there: a refused
2946
+ // destroy must leave this handle exactly as it found it. The reading
2947
+ // is carried into the DELETE as its `preconditions.resourceVersion`,
2948
+ // so the gate costs no extra round trip.
2949
+ //
2950
+ // An object that is already gone is not a refusal: already gone is the
2951
+ // state DELETE was asking for, and an unfenced destroy has always
2952
+ // resolved on it. Reading the epoch must not turn that into a
2953
+ // rejection — the fence exists to stop a write, and there is no write
2954
+ // left to stop.
2955
+ let gate;
2956
+ try {
2957
+ gate = await readHolderEpochGate(target, 'destroy', epoch, destroyOptions?.signal);
2958
+ }
2959
+ catch (err) {
2960
+ if (!(err instanceof KubernetesAlreadyGoneError))
2961
+ throw err;
2962
+ state = 'deleted';
2963
+ dropSession();
2964
+ // Killed but not awaited, as every other path that learns the pod
2965
+ // is gone does it: `exited` on a session whose pod no longer
2966
+ // exists resolves only when TCP notices. The `catch` is not
2967
+ // decoration — a detached chain that rejects with no handler takes
2968
+ // the host process down.
2969
+ void reapTerminals().catch(() => undefined);
2970
+ return;
2971
+ }
893
2972
  await reapTerminals();
894
2973
  const current = session;
895
2974
  // Dropped BEFORE the handle is torn down, because tearing it down runs
896
2975
  // its `release` — which on a workspace is {@link retireSession}, a
897
2976
  // SUSPEND patch. A delete does not want one on the way: the object is
898
2977
  // going away whole. `retireSession` reads exactly this to know it.
899
- session = undefined;
2978
+ dropSession();
900
2979
  // And the state moves with it. The pod is being taken away however the
901
2980
  // DELETE goes, and the handle that served it is now torn down, so a
902
2981
  // DELETE that fails leaves a workspace that admits nothing, says so,
@@ -911,7 +2990,9 @@ async function openWorkspaceHandle(options) {
911
2990
  // state below is committed only once it resolves.
912
2991
  if (current)
913
2992
  await current.destroy(destroyOptions);
914
- await deleteSandbox(destroyOptions?.signal);
2993
+ await deleteSandbox(destroyOptions?.signal, epoch, gate);
2994
+ if (epoch !== undefined)
2995
+ heldEpoch = epoch;
915
2996
  state = 'deleted';
916
2997
  };
917
2998
  /** Delete, sharing one DELETE with any caller already inside it. */
@@ -927,29 +3008,71 @@ async function openWorkspaceHandle(options) {
927
3008
  return pendingDelete;
928
3009
  };
929
3010
  /**
930
- * Bring a session up, and put the workspace back to sleep if that fails.
3011
+ * Bring a session up, and put the workspace back to sleep ONLY if this
3012
+ * call is the one that woke it.
3013
+ *
3014
+ * The failures that reach here are not failures OF the workspace. A
3015
+ * caller's signal aborting during readiness, one 5xx or 429 on a Sandbox
3016
+ * or pod GET (the client does not retry), a privilege probe that overran
3017
+ * its own deadline — every one of them can happen to a second process
3018
+ * adopting a workspace the first process is happily using, and the patch
3019
+ * that used to go out on all of them makes the controller delete that
3020
+ * pod. So the question asked here is not "did the start fail" but "did
3021
+ * THIS call move `operatingMode`", which is what `woke` carries:
3022
+ *
3023
+ * - a create POSTed the object, so the pod exists because of this call;
3024
+ * - an adopt or a `resume()` that found the object `Suspended` sent the
3025
+ * Running patch that asked for the pod;
3026
+ * - an adopt of an object that was already Running, or a `resume()` that
3027
+ * found somebody had already resumed it, moved nothing and so has
3028
+ * nothing to put back.
3029
+ *
3030
+ * The first two keep suspending, and must: a workspace this call woke and
3031
+ * then failed to start is left Running with a pod nobody is using,
3032
+ * burning a node until somebody notices. The third sends nothing and
3033
+ * rethrows, which is the whole point of the rule.
931
3034
  *
932
- * Suspend rather than delete, always: the failure might be a probe refusal
933
- * on a workspace whose disk holds a month of a caller's work, and no
934
- * failure path in this module is allowed to make that decision. The cost
935
- * of being wrong the other way is one suspended Sandbox left standing,
936
- * which the caller finds again under the same deterministic name.
3035
+ * Suspend rather than delete, always, wherever it does patch: the failure
3036
+ * might be a probe refusal on a workspace whose disk holds a month of a
3037
+ * caller's work, and no failure path in this module is allowed to make
3038
+ * that decision. The cost of being wrong the other way is one suspended
3039
+ * Sandbox left standing, which the caller finds again under the same
3040
+ * deterministic name.
3041
+ *
3042
+ * The read that decided `woke` and the patch below are deliberately NOT
3043
+ * one conditional write. Between them another process can change the
3044
+ * object, and this call would then suspend a mode change it did not make
3045
+ * after all — a narrow window, and the honest way to close it is a
3046
+ * condition ON the write rather than a second read here. Until there is
3047
+ * one, a patch that was not refused is treated as this call's own. This
3048
+ * comment is the single place a conditional write has to tighten.
937
3049
  */
938
- const startSessionOrSuspend = async (label, signal, previousPodUid) => {
3050
+ const startSessionOrSuspend = async (policy, wake, signal) => {
939
3051
  try {
940
- return await startSession(label, signal, previousPodUid);
3052
+ return await startSession(policy, signal);
941
3053
  }
942
3054
  catch (err) {
943
- // `suspending`, not `suspended`: `runFailureCleanup` swallows its
944
- // own failures so that the primary error stays primary, which
945
- // means this patch may not have landed and this pod may still be
946
- // running. The state says the transition is unfinished, so a
947
- // later suspend() re-sends it rather than believing this one.
3055
+ // `suspending`, not `suspended`, and set whether or not a patch
3056
+ // goes out below. This handle has no session either way, so it can
3057
+ // serve nothing and `resume()` is the way back — and where a patch
3058
+ // IS sent, `runFailureCleanup` swallows its own failures so that
3059
+ // the primary error stays primary, which means the patch may not
3060
+ // have landed and the pod may still be running. Only a CONFIRMED
3061
+ // suspend is ever recorded as `suspended`.
948
3062
  state = 'suspending';
949
- session = undefined;
3063
+ dropSession();
3064
+ if (!wake.woke || wake.policy === 'leave')
3065
+ throw err;
950
3066
  let retired = false;
951
3067
  await runFailureCleanup(async (cleanupSignal) => {
952
- await client.request('PATCH', sandboxPath(namespace, name), SUSPEND_PATCH, cleanupSignal);
3068
+ // Fenced by the epoch this transition carried, which closes
3069
+ // the window the comment above names: between the read that
3070
+ // decided `woke` and this patch, another holder can take the
3071
+ // workspace, and an unconditional suspend here would stop the
3072
+ // pod that holder is now using. A refusal is swallowed by
3073
+ // `runFailureCleanup` like every other cleanup failure, and
3074
+ // the primary error is still what the caller receives.
3075
+ await writeOperatingMode(target, 'start-cleanup', 'Suspended', wake.epoch, cleanupSignal);
953
3076
  retired = true;
954
3077
  });
955
3078
  // Only when the patch came back. A resume that got as far as
@@ -970,16 +3093,408 @@ async function openWorkspaceHandle(options) {
970
3093
  }
971
3094
  return session;
972
3095
  };
3096
+ /**
3097
+ * Adopt a suspension this handle did not perform, if the object really is
3098
+ * suspended. Answers whether it was.
3099
+ *
3100
+ * Recorded as `suspending` rather than `suspended`, and the distinction is
3101
+ * the same one the whole module turns on: what was observed is the
3102
+ * object's MODE, not the pod stopping. The other process's wait may still
3103
+ * be running, or may have run out. A `suspended` mark here would let this
3104
+ * handle's next `suspend()` return on it and promise a quiesced disk
3105
+ * nobody in this process ever waited for.
3106
+ *
3107
+ * `retiredPodUid` is set for the same reason `suspendNow` sets it: a
3108
+ * suspend patch has landed — somebody else's — so the controller is taking
3109
+ * this pod away, and the resume that follows must see it REPLACED rather
3110
+ * than bind the uid the new agent will refuse.
3111
+ *
3112
+ * It never touches the transition queue, and must not: the call-failure
3113
+ * path below runs inside a rejecting `exec()`, and that exec can be the
3114
+ * privilege probe of a resume that is currently HOLDING the queue. The
3115
+ * `state !== 'running'` guard is what keeps it out of a transition's way —
3116
+ * every transition leaves `running` before it does anything.
3117
+ */
3118
+ const noticeSuspendedElsewhere = async (signal) => {
3119
+ if (state !== 'running')
3120
+ return false;
3121
+ if ((await readOperatingMode(client, namespace, name, signal)) !== 'Suspended')
3122
+ return false;
3123
+ if (state !== 'running')
3124
+ return false;
3125
+ state = 'suspending';
3126
+ dropSession();
3127
+ retiredPodUid = podUid;
3128
+ // Killed but not awaited, exactly as `retireSession` does it and for
3129
+ // the same reason: the frames go to a pod that is being deleted, and
3130
+ // `exited` would otherwise resolve only when TCP notices. The `catch`
3131
+ // is not decoration — a detached chain that rejects with no handler
3132
+ // takes the host process down.
3133
+ void reapTerminals().catch(() => undefined);
3134
+ return true;
3135
+ };
3136
+ /**
3137
+ * Run one admitted call and, if it fails, ask ONCE whether the workspace
3138
+ * has been suspended out from under this handle.
3139
+ *
3140
+ * The failure a foreign suspend produces is not recognisable on its own.
3141
+ * The pod is gone, so the dial is refused — against an address that still
3142
+ * resolves, because the Service outlives the pod — or, if a replacement
3143
+ * pod is already up, the guest answers a flat `unauthorized` because this
3144
+ * handle is presenting the retired pod's uid. Neither says "suspended",
3145
+ * and a caller looking at either has no reason to try `resume()`.
3146
+ *
3147
+ * So the object is re-read, and only then: once per failed call, never
3148
+ * speculatively, and never on a call that succeeded. The re-read is a
3149
+ * diagnostic and behaves like one — it runs without the caller's signal
3150
+ * (which is quite possibly what aborted the call in the first place), and
3151
+ * a re-read that itself fails hands back the caller's own error rather
3152
+ * than replacing it with a second one about the API server.
3153
+ *
3154
+ * That question is asked BEFORE the unconfirmed-cancellation diagnosis is
3155
+ * turned into an error, and the order is load-bearing. A foreign suspend
3156
+ * takes the pod away, so it looks exactly like a gone guest and produces
3157
+ * that diagnosis too — but only one of the two answers lets the caller
3158
+ * recover: `KubernetesWorkspaceSuspendedError` says what happened and
3159
+ * leaves the handle suspended, so `resume()` brings a pod back. The
3160
+ * diagnosis is the answer for a workspace that is still Running.
3161
+ */
3162
+ const admitted = async (operation, run) => {
3163
+ const current = admit(operation);
3164
+ try {
3165
+ return await run(current);
3166
+ }
3167
+ catch (err) {
3168
+ // The diagnosis, if one was taken — read here and ACTED ON last.
3169
+ // A foreign suspend is also a gone guest, and it is the answer
3170
+ // that outranks: #473 promises a caller that somebody else
3171
+ // suspended the workspace hears it by name, and that the handle
3172
+ // adopts the suspension so `resume()` works. Renaming first would
3173
+ // leave the handle marked running with no pod, `suspended` false
3174
+ // and `resume()` a silent no-op. So the suspend question is asked
3175
+ // first, and this is the answer when it comes back `false`.
3176
+ const diagnosis = err instanceof Error ? guestGoneEvidence.get(err) : undefined;
3177
+ if (diagnosis !== undefined && err instanceof Error)
3178
+ guestGoneEvidence.delete(err);
3179
+ if (state === 'running') {
3180
+ let suspended = false;
3181
+ try {
3182
+ suspended = await noticeSuspendedElsewhere();
3183
+ }
3184
+ catch {
3185
+ // A re-read that itself fails is not a better error than
3186
+ // the one the caller is already holding: fall through and
3187
+ // give them that, named if a diagnosis was taken.
3188
+ suspended = false;
3189
+ }
3190
+ if (suspended) {
3191
+ throw new KubernetesWorkspaceSuspendedError(operation, workspaceId, name, 'transport', {
3192
+ cause: err,
3193
+ });
3194
+ }
3195
+ }
3196
+ // An unconfirmed cancellation whose guest is demonstrably gone
3197
+ // leaves as an error that SAYS so and carries both identities. It
3198
+ // is a subclass of the error it replaces, so nothing that catches
3199
+ // the base class stops catching it, and the rule is unchanged —
3200
+ // the outcome is still unknown and still must not be retried.
3201
+ // Nothing was patched to get here and nothing is patched on the
3202
+ // way out.
3203
+ if (diagnosis !== undefined) {
3204
+ const named = new KubernetesWorkspaceGuestGoneError(workspaceId, name, diagnosis.evidence, diagnosis.previous, diagnosis.current, { cause: err });
3205
+ if (err instanceof RemoteCancellationUnknownError)
3206
+ named.retirement = err.retirement;
3207
+ throw named;
3208
+ }
3209
+ throw err;
3210
+ }
3211
+ };
3212
+ /**
3213
+ * {@link admitted}, for a call that hands back an ITERABLE rather than a
3214
+ * promise.
3215
+ *
3216
+ * Admission is taken HERE, synchronously, and only the iteration lives in
3217
+ * the generator below: an async generator's body does not run until
3218
+ * something pulls from it, and a suspended workspace has to refuse
3219
+ * `readFileStream(...)` where the caller wrote it — the same place, and
3220
+ * with the same error, as every other data-plane call. The task sandbox's
3221
+ * own `readFileStream` checks admissibility at the call for the same
3222
+ * reason.
3223
+ *
3224
+ * The one diagnostic re-read is otherwise identical, and it covers the
3225
+ * whole stream rather than its first pull. A workspace suspended halfway
3226
+ * through a long read takes its pod's connection with it, and what
3227
+ * surfaces is a socket that closed early: as unrecognisable on its own as
3228
+ * the failures `admitted` exists to name.
3229
+ *
3230
+ * `yield*` rather than a hand-rolled loop so that a consumer's `break`
3231
+ * still reaches the transport's generator, whose `finally` is what
3232
+ * destroys the socket and makes the guest release the file descriptor.
3233
+ */
3234
+ const admittedStream = (operation, open) => {
3235
+ const source = open(admit(operation));
3236
+ return (async function* stream() {
3237
+ try {
3238
+ yield* source;
3239
+ }
3240
+ catch (err) {
3241
+ if (state !== 'running')
3242
+ throw err;
3243
+ let suspended = false;
3244
+ try {
3245
+ suspended = await noticeSuspendedElsewhere();
3246
+ }
3247
+ catch {
3248
+ throw err;
3249
+ }
3250
+ if (!suspended)
3251
+ throw err;
3252
+ throw new KubernetesWorkspaceSuspendedError(operation, workspaceId, name, 'transport', {
3253
+ cause: err,
3254
+ });
3255
+ }
3256
+ })();
3257
+ };
3258
+ /**
3259
+ * One quiesce, through the live session's transport.
3260
+ *
3261
+ * It goes through {@link admitted} like every other data-plane call, so a
3262
+ * workspace another process suspended underneath this handle is named as
3263
+ * suspended rather than reported as a quiesce that failed at the
3264
+ * transport. It is NOT serialised here: both callers are already holding
3265
+ * the transition queue — the public verb takes it, and `suspendNow` runs
3266
+ * inside it.
3267
+ */
3268
+ const quiesceNow = async (quiesceOptions) => await admitted('quiesce', async (handle) => await admittedTransport('quiesce', handle).quiesce(quiesceOptions?.graceMs !== undefined ? { graceMs: quiesceOptions.graceMs } : {}, quiesceOptions?.signal));
3269
+ /**
3270
+ * Tell the host about a gap a suspend went ahead over.
3271
+ *
3272
+ * Synchronous, never awaited, and a callback that throws changes
3273
+ * nothing: these are reports about a transition that has already been
3274
+ * decided, and a host's diagnostic is not allowed to decide it again.
3275
+ */
3276
+ const tellHost = (callback, value) => {
3277
+ try {
3278
+ callback?.(value);
3279
+ }
3280
+ catch {
3281
+ // A host's callback is not allowed to decide whether the suspend
3282
+ // it was only being told about goes ahead.
3283
+ }
3284
+ };
3285
+ /**
3286
+ * One flush, through the live session's transport.
3287
+ *
3288
+ * Through {@link admitted} like every other data-plane call, so a
3289
+ * workspace another process suspended underneath this handle is named as
3290
+ * suspended rather than reported as a flush that failed at the
3291
+ * transport. Not serialised here: both callers already hold the
3292
+ * transition queue — the public verb takes it, and `suspendNow` runs
3293
+ * inside it.
3294
+ */
3295
+ const flushIfSupported = async (flushOptions) => await admitted('flush', async (handle) => {
3296
+ const transport = admittedTransport('flush', handle);
3297
+ // Asked INSIDE the admitted call and answered with `undefined`
3298
+ // rather than by throwing, because an unsupported feature is an
3299
+ // answer from the guest and not a failure of the connection —
3300
+ // and `admitted`'s failure path treats every error as one, which
3301
+ // means an extra GET of the Sandbox and, on a workspace somebody
3302
+ // else had already suspended, a `KubernetesWorkspaceSuspendedError`
3303
+ // wrapping a refusal that has nothing to do with it.
3304
+ if (!(await transport.supportsFlush(flushOptions?.signal)))
3305
+ return undefined;
3306
+ return await transport.flush(flushOptions?.timeoutMs !== undefined ? { timeoutMs: flushOptions.timeoutMs } : {}, flushOptions?.signal);
3307
+ });
3308
+ const flushNow = async (flushOptions) => {
3309
+ const report = await flushIfSupported(flushOptions);
3310
+ // The explicit verb refuses, where `suspend()` degrades: a caller
3311
+ // that asked for a flush by name is told its image cannot do one.
3312
+ if (report === undefined)
3313
+ throw flushUnsupportedError();
3314
+ return report;
3315
+ };
3316
+ /**
3317
+ * The flush `suspend()` performs by default, and the ONE failure that
3318
+ * stops the suspend.
3319
+ *
3320
+ * That one is a guest which ANSWERED and could not confirm: it is alive,
3321
+ * it tried, and its writes may never reach the device — so no patch goes
3322
+ * out, the pod keeps serving, and the caller holds the guest's own
3323
+ * message and can retry, read the data out, or suspend under
3324
+ * `{ flush: false }` deliberately.
3325
+ *
3326
+ * Everything else is REPORTED and the suspend goes ahead, because a
3327
+ * flush is on by DEFAULT and `suspend()` is the verb an operator reaches
3328
+ * for when a workspace has gone wrong. Two shapes reach here:
3329
+ *
3330
+ * - An image that cannot flush: one whose agent predates the op, or one
3331
+ * whose agent has the op and no `sync` to run it with and says so.
3332
+ * Refusing over it would mean this release could not suspend any
3333
+ * workspace built from such an image without the caller changing its
3334
+ * own code. {@link KubernetesWorkspaceOptions.onFlushUnsupported} is
3335
+ * told.
3336
+ * - A guest that could not be REACHED — a refused dial, a connect
3337
+ * timeout, a rejected token, an agent that has fenced itself, a
3338
+ * workspace somebody else has already suspended. None of those become
3339
+ * flushable by leaving the pod running, and refusing would take the
3340
+ * lifecycle verb away from precisely the crashed, OOM-killed or
3341
+ * fenced workspace it is needed for — which would leave it running,
3342
+ * and billing, behind a raw `ECONNREFUSED` naming neither the flush
3343
+ * nor a way past it. `suspend()` then `resume()` is also what
3344
+ * `KubernetesAgentRetiringError` tells a caller to do about a fence.
3345
+ * {@link KubernetesWorkspaceOptions.onFlushUnreachable} is told, with
3346
+ * the transport's own error as its `cause`.
3347
+ *
3348
+ * The caller's own abort is not a flush verdict and travels untouched: a
3349
+ * caller that withdrew its suspend gets the abort, not a suspend that
3350
+ * went ahead without the flush it asked for.
3351
+ */
3352
+ const flushBeforeSuspend = async (request, signal) => {
3353
+ const timeoutMs = typeof request === 'object' ? request.timeoutMs : undefined;
3354
+ let report;
3355
+ try {
3356
+ report = await flushIfSupported({
3357
+ ...(timeoutMs !== undefined ? { timeoutMs } : {}),
3358
+ ...(signal !== undefined ? { signal } : {}),
3359
+ });
3360
+ }
3361
+ catch (err) {
3362
+ if (err instanceof KubernetesFlushUnconfirmedError)
3363
+ throw err;
3364
+ if (signal?.aborted === true)
3365
+ throw err;
3366
+ // A guest that ANSWERED that it cannot flush — an advertised op
3367
+ // its image has no program to run — is the unsupported gap under
3368
+ // another name, not an unreachable guest, and the host hears it
3369
+ // on the callback that means that.
3370
+ if (err instanceof KubernetesFlushUnsupportedError) {
3371
+ tellHost(options.onFlushUnsupported, err);
3372
+ return;
3373
+ }
3374
+ tellHost(options.onFlushUnreachable, new KubernetesFlushUnreachableError(`kubernetes: workspace '${workspaceId}' could not be asked to flush before its pod stopped (${err instanceof Error ? err.message : String(err)}). The suspend went ahead rather than leaving a workspace nobody can reach running: a guest that answers nothing does not become flushable by keeping its pod, and this is the verb that replaces the pod. What is on the disk is what the guest kernel had already written back, plus whatever the pod's preStop hook and the agent's own SIGTERM handler manage while it stops. A caller that would rather stop should flush() first and pass 'flush: false' only once that has succeeded.`, { cause: err }));
3375
+ return;
3376
+ }
3377
+ if (report !== undefined)
3378
+ return;
3379
+ tellHost(options.onFlushUnsupported, flushUnsupportedError());
3380
+ };
3381
+ /**
3382
+ * The quiesce `suspend({ quiesce })` performs, and the ONE failure it
3383
+ * does not pass on.
3384
+ *
3385
+ * An image whose agent predates the op cannot be asked, and refusing to
3386
+ * suspend over that would make the option unusable against every pod
3387
+ * built before this release — a caller could not even suspend such a
3388
+ * workspace without changing its own code. So the suspend goes ahead, as
3389
+ * it always did, and the host is TOLD through
3390
+ * {@link KubernetesWorkspaceOptions.onQuiesceUnsupported}; the gap is
3391
+ * reported rather than either hidden or turned into a refusal.
3392
+ *
3393
+ * Every other failure travels: a guest that answered and could not
3394
+ * confirm has processes still writing to the disk, and a suspend that
3395
+ * patched anyway would take the pod away while they did.
3396
+ */
3397
+ const quiesceBeforeSuspend = async (request, signal) => {
3398
+ const graceMs = typeof request === 'object' ? request.graceMs : undefined;
3399
+ try {
3400
+ const report = await quiesceNow({
3401
+ ...(graceMs !== undefined ? { graceMs } : {}),
3402
+ ...(signal !== undefined ? { signal } : {}),
3403
+ });
3404
+ // This call answers `void`, so the scope in the report would go
3405
+ // nowhere — and a narrowed scan can miss exactly the process the
3406
+ // quiesce was asked for. Told, for the same reason the
3407
+ // unsupported gap is.
3408
+ if (report.scope !== 'pid-namespace') {
3409
+ try {
3410
+ options.onQuiesceNarrowed?.(report);
3411
+ }
3412
+ catch {
3413
+ // A host's callback is not allowed to decide whether the
3414
+ // suspend it was only being told about goes ahead.
3415
+ }
3416
+ }
3417
+ }
3418
+ catch (err) {
3419
+ if (!(err instanceof KubernetesQuiesceUnsupportedError))
3420
+ throw err;
3421
+ try {
3422
+ options.onQuiesceUnsupported?.(err);
3423
+ }
3424
+ catch {
3425
+ // A host's callback is not allowed to decide whether the suspend
3426
+ // it was only being told about goes ahead.
3427
+ }
3428
+ }
3429
+ };
3430
+ /**
3431
+ * How the FIRST bind is allowed to behave, read off how this handle came
3432
+ * by its object.
3433
+ *
3434
+ * A create POSTed the Sandbox itself: no pod existed a moment ago, no pod
3435
+ * is being replaced, and a read that finds none is a failure to report
3436
+ * rather than a state to wait out. An adopt is the opposite by default —
3437
+ * the pod is somebody else's, possibly on its way out — and the two
3438
+ * shapes that say so are the object standing `Suspended` (its pod has
3439
+ * been taken away, and the resume patch this adopt just sent is what asks
3440
+ * for the replacement) and a pod already carrying a `deletionTimestamp`
3441
+ * (the controller is taking it away now). Both of those leave a window
3442
+ * with no live pod under the name at all, which is exactly the window a
3443
+ * host restarting inside a previous pod's terminationGracePeriodSeconds
3444
+ * arrives in.
3445
+ *
3446
+ * An adopt of an object that was Running with a healthy pod keeps the
3447
+ * create path's behaviour: nothing is being replaced, so nothing is
3448
+ * waited for.
3449
+ */
3450
+ const initialBindPolicy = options.origin === 'created'
3451
+ ? { transition: 'create', awaitReplacement: false }
3452
+ : {
3453
+ transition: 'adopt',
3454
+ awaitReplacement: options.origin === 'resumed' ||
3455
+ options.awaitReplacement === true ||
3456
+ options.drainingPodUid !== undefined,
3457
+ ...(options.drainingPodUid !== undefined ? { retiring: options.drainingPodUid } : {}),
3458
+ };
973
3459
  // The first session is brought up here so that `createKubernetesWorkspace`
974
3460
  // resolves with a workspace that is Ready, addressed and probed — the same
975
3461
  // contract `create()` gives a task sandbox.
976
- session = await startSessionOrSuspend(options.created ? 'create' : 'adopt', options.signal);
3462
+ session = await startSessionOrSuspend(initialBindPolicy,
3463
+ // A create POSTed the object and an adopt that found it `Suspended`
3464
+ // patched it Running; an adopt of an object that was already Running
3465
+ // moved nothing, and a start that fails on it must leave the pod its
3466
+ // holder is using exactly where it found it.
3467
+ {
3468
+ woke: options.origin !== 'adopted-running',
3469
+ policy: defaultStartFailure,
3470
+ ...(heldEpoch !== undefined ? { epoch: heldEpoch } : {}),
3471
+ }, options.signal);
977
3472
  // Whatever the backend's own Sandbox reports, rather than a second copy of
978
3473
  // the same constant: it must keep answering after a suspend has taken the
979
3474
  // handle it came from away.
980
3475
  const environment = session.environment;
981
3476
  return {
982
3477
  id,
3478
+ origin: options.origin,
3479
+ get templateRevision() {
3480
+ return templateRevision;
3481
+ },
3482
+ get templateCurrent() {
3483
+ // An object that recorded no revision is NOT current: unknown is
3484
+ // not a match, and reporting it as one would tell a host there is
3485
+ // nothing to refresh on exactly the workspaces created before
3486
+ // anything recorded what they were built from.
3487
+ return templateRevision !== undefined && templateRevision === currentTemplateHash;
3488
+ },
3489
+ get identity() {
3490
+ return identityNow();
3491
+ },
3492
+ onGuestRestart(listener) {
3493
+ restartListeners.add(listener);
3494
+ return () => {
3495
+ restartListeners.delete(listener);
3496
+ };
3497
+ },
983
3498
  get status() {
984
3499
  // A suspended workspace reports 'destroyed' because that is the only
985
3500
  // member of the SDK's four-way union meaning "cannot serve a call".
@@ -998,19 +3513,88 @@ async function openWorkspaceHandle(options) {
998
3513
  rootDir: options.rootDir,
999
3514
  environment,
1000
3515
  async exec(command, argv, execOptions) {
1001
- return await admit('exec').exec(command, argv, execOptions);
3516
+ // The default path, untouched: the SDK's exec through the inner
3517
+ // handle, the shared execution controller, and the retirement
3518
+ // behaviour that comes with it.
3519
+ if (execOptions?.detach !== true && execOptions?.executionId === undefined) {
3520
+ return await admitted('exec', async (handle) => await handle.exec(command, argv, execOptions));
3521
+ }
3522
+ return await admitted('exec', async (handle) => await admittedTransport('exec', handle).execDetached(command, argv, execOptions));
3523
+ },
3524
+ async attachExecution(executionId, attachOptions) {
3525
+ return await admitted('attachExecution', async (handle) => await admittedTransport('attachExecution', handle).attachExecution(executionId, attachOptions));
3526
+ },
3527
+ async cancelExecution(executionId, transitionOptions) {
3528
+ await admitted('cancelExecution', async (handle) => await admittedTransport('cancelExecution', handle).cancelExecution(executionId, transitionOptions?.signal));
1002
3529
  },
1003
3530
  async writeFile(path, content) {
1004
- await admit('writeFile').writeFile(path, content);
3531
+ await admitted('writeFile', async (handle) => await handle.writeFile(path, content));
3532
+ },
3533
+ /**
3534
+ * `readOptions` is FORWARDED, and that is the whole of the
3535
+ * requirement: a backend that takes `offset`/`length` and answers
3536
+ * with the whole file has given a wrong answer, not a degraded one
3537
+ * (`Sandbox.readFile` in `@namzu/sdk` says so, and the transport
3538
+ * refuses rather than downgrades against a guest too old to honour
3539
+ * them). A workspace that dropped them would do exactly that, on the
3540
+ * surface most likely to be pointed at a file worth ranging.
3541
+ */
3542
+ async readFile(path, readOptions) {
3543
+ return await admitted('readFile', async (handle) => await handle.readFile(path, readOptions));
1005
3544
  },
1006
- async readFile(path) {
1007
- return await admit('readFile').readFile(path);
3545
+ /**
3546
+ * Draining a large output file before `suspend()` or `destroy()`
3547
+ * without holding it — the use a long-lived workspace exists for, and
3548
+ * the reason this is narrowed to present on
3549
+ * {@link KubernetesWorkspace} rather than left optional.
3550
+ *
3551
+ * Admission runs where the caller wrote the call, not at the first
3552
+ * pull, exactly as the task sandbox does it; the suspended-elsewhere
3553
+ * diagnostic runs on a failure at any point in the stream, because a
3554
+ * workspace suspended halfway through takes its pod's connection with
3555
+ * it and leaves only a socket that closed early.
3556
+ */
3557
+ readFileStream(path, readOptions) {
3558
+ return admittedStream('readFileStream', (handle) => handle.readFileStream(path, readOptions));
1008
3559
  },
1009
3560
  async listFiles(rootPath) {
1010
- return await admit('listFiles').listFiles(rootPath);
3561
+ return await admitted('listFiles', async (handle) => await handle.listFiles(rootPath));
3562
+ },
3563
+ /**
3564
+ * Admitted ONCE, when the consumer asks for the first entry, and then
3565
+ * delegated to the inner handle with `yield*`.
3566
+ *
3567
+ * {@link admit} is the synchronous gate every data-plane call on this
3568
+ * handle passes, so a workspace this process knows to be suspended
3569
+ * refuses a walk exactly as it refuses `readFile` — the same error
3570
+ * class, the same `noticedBy: 'admission'`, this operation's own name —
3571
+ * and nothing is dialed. What it deliberately does NOT do is re-admit
3572
+ * per entry: a suspend that lands mid-walk surfaces as the transport
3573
+ * failure it is, and this handle learns it was suspended elsewhere on
3574
+ * its next data-plane call, which is where {@link admitted}'s one-shot
3575
+ * diagnosis lives for every other operation.
3576
+ *
3577
+ * `yield*` is also what makes cancellation work with no code of its own
3578
+ * here: a consumer breaking out of its `for await` runs this generator's
3579
+ * `return()`, and the delegation forwards it to the inner walk — which
3580
+ * is the call that terminates the guest's walk process.
3581
+ *
3582
+ * Busy accounting is not repeated either: the inner handle holds one
3583
+ * execution for the whole walk, so `status` stays `busy` from the first
3584
+ * entry to the last rather than flapping between them.
3585
+ */
3586
+ async *walkFiles(rootPath, walkOptions) {
3587
+ yield* admit('walkFiles').walkFiles(rootPath, walkOptions);
1011
3588
  },
1012
3589
  async openTerminal(terminalOptions) {
1013
- const terminal = await admit('openTerminal').openTerminal(terminalOptions);
3590
+ // A session terminal goes through the TRANSPORT rather than the
3591
+ // inner handle, for the reason `sessionTransports` states: the
3592
+ // session ops are the workspace's own surface, and keeping them
3593
+ // off the inner handle's automatic-retirement path is what makes
3594
+ // "losing a connection costs the workspace nothing" true here too.
3595
+ const terminal = terminalOptions.sessionId === undefined && terminalOptions.persistent !== true
3596
+ ? await admitted('openTerminal', async (handle) => await handle.openTerminal(terminalOptions))
3597
+ : await admitted('openTerminal', async (handle) => await admittedTransport('openTerminal', handle).openTerminal(terminalOptions));
1014
3598
  // Tracked HERE as well as by the inner handle, because a suspend
1015
3599
  // reaps terminals without going through the inner handle's
1016
3600
  // `destroy()` — the pod is being deleted, and a caller left holding
@@ -1027,11 +3611,65 @@ async function openWorkspaceHandle(options) {
1027
3611
  .catch(() => undefined);
1028
3612
  return terminal;
1029
3613
  },
3614
+ async attachTerminal(sessionId, attachOptions) {
3615
+ const terminal = await admitted('attachTerminal', async (handle) => await admittedTransport('attachTerminal', handle).attachSession(sessionId, attachOptions));
3616
+ // Tracked exactly like an `openTerminal` result, and released the
3617
+ // same way: a suspend detaches it rather than killing the shell.
3618
+ terminals.add(terminal);
3619
+ void terminal.exited
3620
+ .finally(() => {
3621
+ terminals.delete(terminal);
3622
+ })
3623
+ .catch(() => undefined);
3624
+ return terminal;
3625
+ },
3626
+ async startDetached(detachedOptions) {
3627
+ return await admitted('startDetached', async (handle) => await admittedTransport('startDetached', handle).startDetached(detachedOptions));
3628
+ },
3629
+ async readSession(sessionId, readOptions) {
3630
+ return await admitted('readSession', async (handle) => await admittedTransport('readSession', handle).readSession(sessionId, readOptions));
3631
+ },
3632
+ async listSessions(transitionOptions) {
3633
+ return await admitted('listSessions', async (handle) => await admittedTransport('listSessions', handle).listSessions(transitionOptions?.signal));
3634
+ },
3635
+ async killSession(sessionId, killOptions) {
3636
+ return await admitted('killSession', async (handle) => await admittedTransport('killSession', handle).killSession(sessionId, killOptions ?? {}));
3637
+ },
1030
3638
  async openTcpConnection(connectOptions) {
1031
- return await admit('openTcpConnection').openTcpConnection(connectOptions);
3639
+ return await admitted('openTcpConnection', async (handle) => await handle.openTcpConnection(connectOptions));
3640
+ },
3641
+ async refresh(transitionOptions) {
3642
+ // Serialised, unlike the call-failure path, because nothing calls
3643
+ // this from inside a transition: it is a caller's own verb, and
3644
+ // running it between transitions rather than through one keeps it
3645
+ // from reading a mode a resume is halfway through changing.
3646
+ await serialise(async () => {
3647
+ await noticeSuspendedElsewhere(transitionOptions?.signal);
3648
+ });
3649
+ },
3650
+ async quiesce(quiesceOptions) {
3651
+ // Serialised, like the transitions it is meant to precede: a quiesce
3652
+ // racing the suspend that follows it would be a quiesce of a pod the
3653
+ // patch has already taken away, and one racing a resume would ask
3654
+ // the old pod to stop the new one's processes.
3655
+ return await serialise(async () => await quiesceNow(quiesceOptions));
3656
+ },
3657
+ async flush(flushOptions) {
3658
+ // Serialised for the reason `quiesce()` is: a flush racing the
3659
+ // suspend that follows it would be flushing a pod the patch has
3660
+ // already taken away, and one racing a resume would ask the old
3661
+ // pod about the new one's disk.
3662
+ return await serialise(async () => await flushNow(flushOptions));
1032
3663
  },
1033
3664
  async suspend(transitionOptions) {
1034
- await suspendShared(transitionOptions?.signal);
3665
+ // The single flight takes the FIRST caller's epoch, exactly as it
3666
+ // takes the first caller's signal: a second caller arriving
3667
+ // mid-suspend is joining that transition rather than starting one
3668
+ // of its own, and one transition can only be written under one
3669
+ // authority. Its `quiesce` is the exception `suspendShared`
3670
+ // explains — a guarantee about the guest, not an authority, and
3671
+ // one a transition already patching cannot be given.
3672
+ await suspendShared(transitionOptions?.signal, assertHolderEpoch(transitionOptions?.epoch, 'suspend') ?? heldEpoch, transitionOptions?.quiesce, transitionOptions?.flush);
1035
3673
  },
1036
3674
  async resume(transitionOptions) {
1037
3675
  // Serialised, and deliberately without a single-flight slot of its
@@ -1043,9 +3681,10 @@ async function openWorkspaceHandle(options) {
1043
3681
  // second caller finds the work done. Give resume a state it
1044
3682
  // early-returns on before the cluster confirms it and it will need
1045
3683
  // a slot as much as they do.
1046
- await serialise(async () => await resumeNow(transitionOptions?.signal));
3684
+ await serialise(async () => await resumeNow(transitionOptions?.signal, transitionOptions?.onStartFailure ?? defaultStartFailure, assertHolderEpoch(transitionOptions?.epoch, 'resume') ?? heldEpoch, transitionOptions?.refreshPodTemplate === true));
1047
3685
  },
1048
3686
  async destroy(destroyOptions) {
3687
+ assertHolderEpoch(destroyOptions?.epoch, 'destroy');
1049
3688
  if (destroyOptions?.deleteDisk !== true) {
1050
3689
  // The default, and the whole point of the default: there is no
1051
3690
  // delete-compute-keep-disk verb, so the closest thing to one is
@@ -1054,7 +3693,7 @@ async function openWorkspaceHandle(options) {
1054
3693
  // single flight, so `destroy()` racing `suspend()` is one
1055
3694
  // transition rather than two — and it stays idempotent over a
1056
3695
  // workspace already deleted, where `suspend()` itself refuses.
1057
- await destroyBySuspending(destroyOptions?.signal);
3696
+ await destroyBySuspending(destroyOptions?.signal, destroyOptions?.epoch ?? heldEpoch, destroyOptions?.quiesce, destroyOptions?.flush);
1058
3697
  return;
1059
3698
  }
1060
3699
  await deleteShared(destroyOptions);