@namzu/sandbox 14.0.0 → 16.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. package/CHANGELOG.md +924 -0
  2. package/README.md +369 -14
  3. package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
  4. package/dist/backends/aci-standby-pool/index.js +13 -1
  5. package/dist/backends/aci-standby-pool/index.js.map +1 -1
  6. package/dist/backends/docker/index.d.ts +169 -6
  7. package/dist/backends/docker/index.d.ts.map +1 -1
  8. package/dist/backends/docker/index.js +499 -85
  9. package/dist/backends/docker/index.js.map +1 -1
  10. package/dist/backends/firecracker/index.d.ts.map +1 -1
  11. package/dist/backends/firecracker/index.js +12 -2
  12. package/dist/backends/firecracker/index.js.map +1 -1
  13. package/dist/backends/firecracker/protocol.d.ts +459 -8
  14. package/dist/backends/firecracker/protocol.d.ts.map +1 -1
  15. package/dist/backends/firecracker/protocol.js +136 -0
  16. package/dist/backends/firecracker/protocol.js.map +1 -1
  17. package/dist/backends/firecracker/transport.d.ts +539 -6
  18. package/dist/backends/firecracker/transport.d.ts.map +1 -1
  19. package/dist/backends/firecracker/transport.js +1171 -24
  20. package/dist/backends/firecracker/transport.js.map +1 -1
  21. package/dist/backends/kubernetes/egress-policy.d.ts +1181 -13
  22. package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -1
  23. package/dist/backends/kubernetes/egress-policy.js +2350 -31
  24. package/dist/backends/kubernetes/egress-policy.js.map +1 -1
  25. package/dist/backends/kubernetes/identity.d.ts +193 -0
  26. package/dist/backends/kubernetes/identity.d.ts.map +1 -0
  27. package/dist/backends/kubernetes/identity.js +147 -0
  28. package/dist/backends/kubernetes/identity.js.map +1 -0
  29. package/dist/backends/kubernetes/index.d.ts +678 -33
  30. package/dist/backends/kubernetes/index.d.ts.map +1 -1
  31. package/dist/backends/kubernetes/index.js +1180 -95
  32. package/dist/backends/kubernetes/index.js.map +1 -1
  33. package/dist/backends/kubernetes/ingress-policy.d.ts +375 -0
  34. package/dist/backends/kubernetes/ingress-policy.d.ts.map +1 -0
  35. package/dist/backends/kubernetes/ingress-policy.js +1050 -0
  36. package/dist/backends/kubernetes/ingress-policy.js.map +1 -0
  37. package/dist/backends/kubernetes/k8s-client.d.ts +213 -4
  38. package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -1
  39. package/dist/backends/kubernetes/k8s-client.js +359 -52
  40. package/dist/backends/kubernetes/k8s-client.js.map +1 -1
  41. package/dist/backends/kubernetes/lease.d.ts +40 -14
  42. package/dist/backends/kubernetes/lease.d.ts.map +1 -1
  43. package/dist/backends/kubernetes/lease.js +68 -18
  44. package/dist/backends/kubernetes/lease.js.map +1 -1
  45. package/dist/backends/kubernetes/objects.d.ts +423 -3
  46. package/dist/backends/kubernetes/objects.d.ts.map +1 -1
  47. package/dist/backends/kubernetes/objects.js +364 -2
  48. package/dist/backends/kubernetes/objects.js.map +1 -1
  49. package/dist/backends/kubernetes/per-sandbox-policy.d.ts +219 -0
  50. package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -0
  51. package/dist/backends/kubernetes/per-sandbox-policy.js +375 -0
  52. package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -0
  53. package/dist/backends/kubernetes/rbac.d.ts +153 -0
  54. package/dist/backends/kubernetes/rbac.d.ts.map +1 -0
  55. package/dist/backends/kubernetes/rbac.js +177 -0
  56. package/dist/backends/kubernetes/rbac.js.map +1 -0
  57. package/dist/backends/kubernetes/sandbox.d.ts +81 -14
  58. package/dist/backends/kubernetes/sandbox.d.ts.map +1 -1
  59. package/dist/backends/kubernetes/sandbox.js +149 -15
  60. package/dist/backends/kubernetes/sandbox.js.map +1 -1
  61. package/dist/backends/kubernetes/transport.d.ts +935 -9
  62. package/dist/backends/kubernetes/transport.d.ts.map +1 -1
  63. package/dist/backends/kubernetes/transport.js +1958 -62
  64. package/dist/backends/kubernetes/transport.js.map +1 -1
  65. package/dist/backends/kubernetes/workspace.d.ts +1149 -18
  66. package/dist/backends/kubernetes/workspace.d.ts.map +1 -1
  67. package/dist/backends/kubernetes/workspace.js +2825 -186
  68. package/dist/backends/kubernetes/workspace.js.map +1 -1
  69. package/dist/backends/remote-execution-controller.d.ts +14 -0
  70. package/dist/backends/remote-execution-controller.d.ts.map +1 -1
  71. package/dist/backends/remote-execution-controller.js.map +1 -1
  72. package/dist/index.d.ts +294 -18
  73. package/dist/index.d.ts.map +1 -1
  74. package/dist/index.js +280 -10
  75. package/dist/index.js.map +1 -1
  76. package/dist/testing/sandbox-conformance.d.ts +39 -5
  77. package/dist/testing/sandbox-conformance.d.ts.map +1 -1
  78. package/dist/testing/sandbox-conformance.js +436 -5
  79. package/dist/testing/sandbox-conformance.js.map +1 -1
  80. package/package.json +3 -3
  81. package/src/backends/aci-standby-pool/index.ts +16 -1
  82. package/src/backends/docker/index.ts +617 -100
  83. package/src/backends/firecracker/index.ts +14 -2
  84. package/src/backends/firecracker/protocol.ts +514 -6
  85. package/src/backends/firecracker/transport.ts +1492 -40
  86. package/src/backends/kubernetes/egress-policy.ts +3334 -55
  87. package/src/backends/kubernetes/identity.ts +261 -0
  88. package/src/backends/kubernetes/index.ts +1785 -127
  89. package/src/backends/kubernetes/ingress-policy.ts +1344 -0
  90. package/src/backends/kubernetes/k8s-client.ts +444 -54
  91. package/src/backends/kubernetes/lease.ts +75 -19
  92. package/src/backends/kubernetes/objects.ts +626 -6
  93. package/src/backends/kubernetes/per-sandbox-policy.ts +497 -0
  94. package/src/backends/kubernetes/rbac.ts +192 -0
  95. package/src/backends/kubernetes/sandbox.ts +218 -20
  96. package/src/backends/kubernetes/transport.ts +2733 -124
  97. package/src/backends/kubernetes/workspace.ts +4476 -222
  98. package/src/backends/remote-execution-controller.ts +14 -0
  99. package/src/index.ts +668 -19
  100. package/src/testing/sandbox-conformance.ts +540 -5
@@ -24,26 +24,53 @@
24
24
  * measured, not guessed at.
25
25
  */
26
26
 
27
+ import { randomUUID } from 'node:crypto'
28
+ import net from 'node:net'
29
+
27
30
  import type {
31
+ BackgroundJobStatus,
28
32
  OpenTerminalOptions,
29
33
  SandboxExecOptions,
30
34
  SandboxExecResult,
35
+ SandboxReadFileOptions,
31
36
  SandboxTcpConnectOptions,
32
37
  SandboxTcpConnection,
33
38
  TerminalSession,
34
39
  } from '@namzu/sdk'
35
40
 
36
- import type { ExecRequest } from '../firecracker/protocol.js'
37
41
  import {
42
+ EXECUTION_ATTACH_FEATURE,
43
+ type ExecRequest,
44
+ ExecResultAccumulator,
45
+ FLUSH_FEATURE,
46
+ type GuestReplyIdentity,
47
+ QUIESCE_FEATURE,
48
+ type QuiesceScope,
49
+ type QuiescedProcess,
50
+ SESSIONS_FEATURE,
51
+ type SessionKind,
52
+ type SessionState,
53
+ parseExecEvent,
54
+ } from '../firecracker/protocol.js'
55
+ import {
56
+ AgentDialFailedError,
38
57
  type AgentRequest,
58
+ type AgentTerminalStream,
39
59
  type SandboxAgentHandle,
40
60
  VsockAgentTransport,
41
61
  type VsockTransportOptions,
42
62
  } from '../firecracker/transport.js'
43
63
  import {
64
+ type RemoteCancellationAcknowledgement,
65
+ RemoteCancellationUnknownError,
66
+ RemoteCommandError,
44
67
  type RemoteExecutionAdapter,
45
68
  RemoteExecutionController,
69
+ RemoteProtocolError,
70
+ RemoteResultIncompleteError,
71
+ type RemoteTerminalMetadata,
46
72
  } from '../remote-execution-controller.js'
73
+ import { KubernetesWorkspaceReplacedError } from './identity.js'
47
74
 
48
75
  /** The one {@link SandboxAgentHandle} arm this backend ever constructs. */
49
76
  export type KubernetesAgentHandle = Extract<SandboxAgentHandle, { kind: 'tcp' }>
@@ -57,23 +84,355 @@ export type KubernetesAgentHandle = Extract<SandboxAgentHandle, { kind: 'tcp' }>
57
84
  */
58
85
  export class KubernetesAgentUnauthorizedError extends Error {
59
86
  constructor(
60
- message = 'kubernetes tcp transport: the guest agent rejected this connection’s token (unauthorized)',
87
+ message = 'kubernetes tcp transport: the guest agent rejected this connection’s token (unauthorized). The token is the bound pod’s metadata.uid and the guest checks it BEFORE dispatch, so the request ran nothing at all; a handle that can re-read its pod follows the replacement and retries once, and this error is what stands when there is nothing to follow — the same pod is still there and still refusing, or the re-read could not be made.',
88
+ options?: ErrorOptions,
61
89
  ) {
62
- super(message)
90
+ super(message, options)
63
91
  this.name = 'KubernetesAgentUnauthorizedError'
64
92
  }
65
93
  }
66
94
 
95
+ /**
96
+ * Thrown when the guest agent refuses a request because it has FENCED
97
+ * ITSELF: an earlier process group's shutdown could not be confirmed, so it
98
+ * answers every op but `healthz` and `cancel-execution` with `agent_retiring`
99
+ * and will go on doing so until the pod is replaced.
100
+ *
101
+ * Its own class, and deliberately NOT the Firecracker tier's mapping of the
102
+ * same refusal. There a fenced agent becomes {@link
103
+ * RemoteCancellationUnknownError}, which is correct for a disposable microVM:
104
+ * the shared controller's rule is that the sandbox stops being reusable, and
105
+ * on that tier retiring one means deleting scratch. On a workspace the same
106
+ * error would retire the handle and take the pod — and with it every other
107
+ * holder's terminals, dev servers and running commands — away from callers
108
+ * who did nothing but share a workspace with the command that wedged.
109
+ *
110
+ * So this error refuses the ONE call rather than the workspace: the handle is
111
+ * not retired, nothing is patched, and every other holder's pod stays where it
112
+ * was. What it does NOT claim is that the next call will work. The fence is
113
+ * the GUEST's, and `dispatch` gates it ahead of every data-plane branch, so
114
+ * `readFile`, `writeFile`, `openTerminal` and `openTcpConnection` meet the
115
+ * same refusal on the wire — under their own paths' error shapes, since only
116
+ * the two control ops come through `requestChecked`. Only a new pod clears
117
+ * it, which is why the message names the verbs that REPLACE the pod, on both
118
+ * tiers that use this transport, and leaves the timing to the host: on a
119
+ * workspace those are `suspend()` then `resume()`, and they take the live
120
+ * sessions in that pod down with them.
121
+ */
122
+ export class KubernetesAgentRetiringError extends Error {
123
+ override readonly name = 'KubernetesAgentRetiringError'
124
+
125
+ constructor(
126
+ message = 'kubernetes tcp transport: the guest agent has fenced itself (agent_retiring) because an earlier process group’s shutdown could not be confirmed. A command of unknown state may still be running in that pod and nothing on this side can end it: only a new pod clears the fence, and until the pod is replaced every call except healthz and cancel-execution meets this same refusal — reads, writes, terminals and tcp connections included. Nothing was changed on the cluster by this refusal and this handle was not retired. On a persistent workspace, call suspend() and then resume() when you are ready for the live terminals and background processes in that pod to go down; on a task sandbox, destroy() it and take another.',
127
+ ) {
128
+ super(message)
129
+ }
130
+ }
131
+
132
+ /**
133
+ * Thrown when the agent's address could not be RESOLVED — the dial never
134
+ * reached a socket because the name has no answer here.
135
+ *
136
+ * Its own class, and its own message, because this is the one failure whose
137
+ * cause is the deployment's shape rather than anything the cluster did: a
138
+ * Service FQDN resolves through cluster DNS and nowhere else, so a host
139
+ * outside the cluster fails every call at name resolution and reads the
140
+ * result as a sandbox that never came up. The fix is a configuration field,
141
+ * so the error names it.
142
+ */
143
+ export class KubernetesAgentAddressUnresolvableError extends Error {
144
+ override readonly name = 'KubernetesAgentAddressUnresolvableError'
145
+
146
+ constructor(
147
+ /** The host that did not resolve — normally a `*.svc.cluster.local`. */
148
+ readonly host: string,
149
+ message: string,
150
+ options?: { cause?: unknown },
151
+ ) {
152
+ super(message, options)
153
+ }
154
+ }
155
+
156
+ /** Every `code` and message in an error's own chain, cause by cause. */
157
+ function errorChain(error: unknown): { codes: string[]; messages: string[] } {
158
+ const codes: string[] = []
159
+ const messages: string[] = []
160
+ let current: unknown = error
161
+ // Bounded rather than `while (current)`: a cause cycle is a hang, and no
162
+ // real chain on this path is more than three deep.
163
+ for (let depth = 0; depth < 8 && current instanceof Error; depth += 1) {
164
+ const code = (current as { code?: unknown }).code
165
+ if (typeof code === 'string') codes.push(code)
166
+ messages.push(current.message)
167
+ current = current.cause
168
+ }
169
+ return { codes, messages }
170
+ }
171
+
172
+ /**
173
+ * Connect-shaped: the failure came out of the DIAL, so NOTHING was sent to
174
+ * the guest.
175
+ *
176
+ * That last part is what makes a retry safe after the handle follows a
177
+ * replaced pod — a reserved execution, a half-written file or a terminal
178
+ * cannot be sitting in a pod this connection never opened — and it is why the
179
+ * test is WHERE the error came from rather than which `errno` it carries.
180
+ * A code cannot say that much: `ETIMEDOUT` is what the kernel raises when a
181
+ * connect attempt gets no answer AND what it raises on an ESTABLISHED socket
182
+ * that has run out of retransmits — the second of those happens mid-request,
183
+ * with bytes already delivered, and retrying it is not safe.
184
+ *
185
+ * {@link AgentDialFailedError} is what {@link VsockAgentTransport}'s dial
186
+ * throws when it gives up without a socket, so the question is asked of the
187
+ * error's own type, anywhere in its cause chain. Not of its text: a guest
188
+ * answers with text, and a command's stderr quoting "could not connect to
189
+ * agent" would otherwise be read as this process's own dial failing.
190
+ */
191
+ function isConnectFailure(error: unknown): boolean {
192
+ let current: unknown = error
193
+ // Bounded for the same reason {@link errorChain} is: a cause cycle is a
194
+ // hang, and no real chain on this path is more than three deep.
195
+ for (let depth = 0; depth < 8 && current instanceof Error; depth += 1) {
196
+ if (current instanceof AgentDialFailedError) return true
197
+ current = current.cause
198
+ }
199
+ return false
200
+ }
201
+
202
+ /**
203
+ * What one attempt's dials actually did, recorded as they happen.
204
+ *
205
+ * The marker above is the primary test and travels with the error, which is
206
+ * enough for every operation that hands its failure straight back. `exec()`
207
+ * does not: {@link RemoteExecutionController} bounds its control requests
208
+ * (`bounded`, 2000ms by default) by RACING the operation against a timer, so
209
+ * when the dial is still inside its 30s connect-retry budget at 2s the caller
210
+ * is given the timer's bare `… reservation exceeded 2000ms` Error and the
211
+ * dial's own failure — marker, `code` and all — is discarded rather than
212
+ * wrapped. Classifying that error is impossible; the only way to know a
213
+ * socket was never established is to have watched the dials.
214
+ *
215
+ * What is watched is that a connect was ATTEMPTED, not that one failed. The
216
+ * bound is 2000ms and the dial's own connect timer is 5000ms by default, so
217
+ * the shape this exists for — a pod IP whose SYN is dropped rather than
218
+ * refused, which is what a released address on a routed pod network does —
219
+ * is ABORTED by the bound before the attempt has failed at all. A watch of
220
+ * failures alone is blind to exactly the half of the input space that made
221
+ * the bound a problem in the first place; a fast `ECONNREFUSED` is the easy
222
+ * half.
223
+ *
224
+ * `connected` is the half that makes the retry safe. It is set by the dial's
225
+ * own success callback, so "a connect was attempted and no dial ever handed
226
+ * back a socket" means literally nothing was sent to the guest during this
227
+ * attempt — the same guarantee {@link AgentDialFailedError} carries, read
228
+ * from the other end. It is as strong as it sounds on this arm: the `tcp`
229
+ * dial resolves only on the socket's own `connect` event, so there is no
230
+ * moment at which a byte has been written and `connected` is still false.
231
+ */
232
+ interface DialWatch {
233
+ /** At least one connect attempt was made — it may not have settled. */
234
+ attempted: boolean
235
+ /** At least one connect attempt failed, with an error to read. */
236
+ failed: boolean
237
+ /** At least one dial handed back a connected socket. */
238
+ connected: boolean
239
+ /** The last connect attempt's own error, which the bound may swallow. */
240
+ lastError?: unknown
241
+ }
242
+
243
+ /** Nothing this attempt sent ever reached a socket. */
244
+ function neverConnected(dials: DialWatch | undefined): boolean {
245
+ return dials?.attempted === true && !dials.connected
246
+ }
247
+
248
+ /**
249
+ * Outcomes no rebind may retry, whatever the failure underneath them looks
250
+ * like.
251
+ *
252
+ * Both are the controller's way of saying the REMOTE state is unknown: a
253
+ * cancellation it could not confirm, a confirmed termination whose result
254
+ * stream broke. Their own messages interpolate the underlying failure, so a
255
+ * dial that broke mid-command travels as the `cause` of one of them — and a
256
+ * retry there would re-run a command that may already have run, against a
257
+ * disk that followed the pod. `RemoteCancellationUnknownError`
258
+ * says so in its own words: "do not automatically retry the command".
259
+ */
260
+ function isUnretryableOutcome(error: unknown): boolean {
261
+ return (
262
+ error instanceof RemoteCancellationUnknownError || error instanceof RemoteResultIncompleteError
263
+ )
264
+ }
265
+
266
+ /**
267
+ * Name-resolution-shaped: `getaddrinfo` refused the handle's host.
268
+ *
269
+ * Asked only of a DIAL failure, and only of a handle whose host is a name —
270
+ * both gates are at the one call site, {@link
271
+ * KubernetesAgentTransport.withRebind}. The message match below is
272
+ * load-bearing (the retry wrapper's own Error does not carry the `code`
273
+ * forward) and a substring test is exactly as strong as the text it is given,
274
+ * so it is never asked of an arbitrary error: a `readFile` that fails because
275
+ * the guest reported a path containing `ENOTFOUND`, or an exec whose output is
276
+ * quoted into a message, is not a resolver failure and must not be rewritten
277
+ * into one. `net.connect` never consults a resolver for a literal either, so a
278
+ * `pod-ip` handle keeps its one re-read however an unrelated error is worded.
279
+ */
280
+ const DNS_ERROR_CODES = new Set(['ENOTFOUND', 'EAI_AGAIN'])
281
+
282
+ function isNameResolutionFailure(error: unknown): boolean {
283
+ const { codes, messages } = errorChain(error)
284
+ if (codes.some((code) => DNS_ERROR_CODES.has(code))) return true
285
+ // `net.connect` surfaces a `getaddrinfo ENOTFOUND <host>` message whose
286
+ // `code` the retry wrapper's own Error does not carry forward.
287
+ return messages.some((message) => message.includes('ENOTFOUND') || message.includes('EAI_AGAIN'))
288
+ }
289
+
290
+ /**
291
+ * The resolver said the name does not exist — as opposed to saying it could
292
+ * not answer right now.
293
+ *
294
+ * Only this half is treated as permanent, and the difference is the default
295
+ * mode's whole retry budget. `EAI_AGAIN` is BY DEFINITION "temporary failure
296
+ * in name resolution": it is the shape a CoreDNS restart or a conntrack race
297
+ * produces for a host INSIDE the cluster, where the Service FQDN is correct
298
+ * and waiting is exactly the cure — the 30s budget exists to ride over that,
299
+ * and advice to set `agentAddress: 'pod-ip'` would be wrong for that host.
300
+ * `ENOTFOUND` is the out-of-cluster symptom this mode exists for: a resolver
301
+ * that has answered, definitively, that the name is not a name here, and no
302
+ * amount of re-asking it changes that.
303
+ *
304
+ * A temporary failure that outlives the budget still ends as
305
+ * {@link KubernetesAgentAddressUnresolvableError}, so the diagnosis is
306
+ * delayed rather than lost.
307
+ */
308
+ function isMissingNameFailure(error: unknown): boolean {
309
+ const { codes, messages } = errorChain(error)
310
+ if (codes.includes('ENOTFOUND')) return true
311
+ return messages.some((message) => message.includes('ENOTFOUND'))
312
+ }
313
+
67
314
  /** True for the wire shape `agent.cjs` sends when a token is rejected. */
68
315
  function isUnauthorized(response: unknown): boolean {
316
+ return refusedWith(response, 'unauthorized')
317
+ }
318
+
319
+ /**
320
+ * True for the wire shape `agent.cjs` sends when it has fenced itself —
321
+ * `dispatch`'s gate, and `handleReserveExecution`'s own earlier one.
322
+ */
323
+ function isAgentRetiring(response: unknown): boolean {
324
+ return refusedWith(response, 'agent_retiring')
325
+ }
326
+
327
+ /** One refusal envelope: `{ ok: false, error: <name> }`, and nothing else. */
328
+ function refusedWith(response: unknown, error: string): boolean {
69
329
  if (!response || typeof response !== 'object') return false
70
330
  const value = response as { ok?: unknown; error?: unknown }
71
- return value.ok === false && value.error === 'unauthorized'
331
+ return value.ok === false && value.error === error
332
+ }
333
+
334
+ /**
335
+ * The guest REFUSED this handle's token — whichever of the two shapes that
336
+ * refusal happens to arrive in.
337
+ *
338
+ * `reserve-execution` and `cancel-execution` come through
339
+ * {@link requestChecked}, so their refusal is already a
340
+ * {@link KubernetesAgentUnauthorizedError}. Nothing else does:
341
+ * `writeFile`, `readFile`, `openTerminal` and `openTcpConnection` are the
342
+ * shared Firecracker transport's own paths, and each of them throws the
343
+ * guest's error NAME as a plain `Error` — `unauthorized` and nothing more.
344
+ * Two shapes for one fact is why a host could never hang a single recovery
345
+ * off it, and this is where they become one.
346
+ *
347
+ * The message test is exact, never a substring, and that is deliberate: the
348
+ * guest's refusal envelope carries the bare token `unauthorized` as its whole
349
+ * `error` field, while a read whose PATH contains the word, or an exec whose
350
+ * output is quoted into a message, does not equal it. The chain is walked
351
+ * because the retry wrapper and the stream paths wrap rather than replace.
352
+ */
353
+ function isUnauthorizedRefusal(error: unknown): boolean {
354
+ let current: unknown = error
355
+ for (let depth = 0; depth < 8 && current instanceof Error; depth += 1) {
356
+ if (current instanceof KubernetesAgentUnauthorizedError) return true
357
+ if (current.message === 'unauthorized') return true
358
+ current = current.cause
359
+ }
360
+ return false
361
+ }
362
+
363
+ /**
364
+ * The one shape a refused call ends in when no rebind could fix it.
365
+ *
366
+ * An error that is already the named class travels unchanged — replacing it
367
+ * would lose its `cause` chain for nothing — and every other shape is wrapped
368
+ * with the original on `cause`, so the plain `Error('unauthorized')` a
369
+ * `writeFile` used to end with is still readable underneath.
370
+ */
371
+ function asUnauthorized(error: unknown): Error {
372
+ if (error instanceof KubernetesAgentUnauthorizedError) return error
373
+ return new KubernetesAgentUnauthorizedError(undefined, { cause: error })
374
+ }
375
+
376
+ /** The optional boot id on one guest reply, and nothing read from any other shape. */
377
+ function guestBootIdOf(reply: unknown): string | undefined {
378
+ if (!reply || typeof reply !== 'object') return undefined
379
+ const value = (reply as { guestBootId?: unknown }).guestBootId
380
+ return typeof value === 'string' && value !== '' ? value : undefined
381
+ }
382
+
383
+ /**
384
+ * The GUEST that RESERVED each execution whose cancellation could not be
385
+ * confirmed: the pod its reservation was accepted by, and the agent PROCESS
386
+ * inside that pod.
387
+ *
388
+ * Per EXECUTION and never per session, which is the whole reason it is kept
389
+ * here rather than read off the handle when the diagnosis runs. A workspace
390
+ * handle outlives many commands and follows a replaced pod, so what it is
391
+ * bound to when a diagnosis runs is not what a command that failed minutes
392
+ * ago was running in: a container the kubelet restarted an hour ago says
393
+ * nothing about a command started after it, and a pod ANOTHER call already
394
+ * rebound to is not the pod this command's processes died with. A handle-wide
395
+ * baseline would report both as "the guest your command was running in is
396
+ * gone", about a command that is very likely still running, and would name
397
+ * the replacement as the guest that died.
398
+ *
399
+ * What is recorded is the bind token this attempt presented — which IS the
400
+ * pod's `metadata.uid`, the same value the handle reports as
401
+ * `identity.podUid` — and the boot id the `reserve-execution` reply carried,
402
+ * because the guest that accepted the reservation is the guest that ran the
403
+ * command. Both are stamped onto the error on its way out of {@link
404
+ * KubernetesAgentTransport.exec}, the last frame that still knows which
405
+ * execution an error belongs to.
406
+ *
407
+ * Weak, so an error nobody kept takes its entry with it. The boot id is
408
+ * absent for a guest too old to report one, and the whole entry is absent for
409
+ * an error no `exec()` produced: the diagnosis then falls back to what the
410
+ * handle itself is bound to, which is the honest answer rather than a guess.
411
+ */
412
+ const executionGuest = new WeakMap<Error, KubernetesReservedGuest>()
413
+
414
+ /** The pod and process one execution reserved on — see {@link executionGuest}. */
415
+ export interface KubernetesReservedGuest {
416
+ /** The bind token the attempt presented, which is the pod's uid. */
417
+ readonly podUid?: string
418
+ /** The boot id the `reserve-execution` reply carried, when it carried one. */
419
+ readonly guestBootId?: string
420
+ }
421
+
422
+ /**
423
+ * The guest that reserved the execution this error came from — see
424
+ * {@link executionGuest}. `undefined` when the error is not an execution
425
+ * failure this transport produced.
426
+ */
427
+ export function guestWhenReserved(error: unknown): KubernetesReservedGuest | undefined {
428
+ return error instanceof Error ? executionGuest.get(error) : undefined
72
429
  }
73
430
 
74
431
  /**
75
- * One framed control request, with the guest's `unauthorized` refusal
76
- * turned into {@link KubernetesAgentUnauthorizedError}.
432
+ * One framed control request, with the guest's two NAMED refusals turned
433
+ * into errors a caller can catch by class: {@link
434
+ * KubernetesAgentUnauthorizedError} for a rejected token, and {@link
435
+ * KubernetesAgentRetiringError} for an agent that has fenced itself.
77
436
  *
78
437
  * BOTH control requests go through here — `reserve-execution` and
79
438
  * `cancel-execution` — because the two are read by the same caller for
@@ -84,6 +443,16 @@ function isUnauthorized(response: unknown): boolean {
84
443
  * rejected token read as a transport failure would therefore spend that
85
444
  * window re-sending a request that can never succeed, and end by
86
445
  * describing a wrong credential as an ambiguous outcome.
446
+ *
447
+ * The fenced-agent refusal is the one a SECOND holder meets. `dispatch`
448
+ * lets `cancel-execution` through the fence and refuses everything else, so
449
+ * `agent_retiring` reaches here from `reserve-execution` — which without
450
+ * this mapping is parsed as a reservation and rejected as
451
+ * `RemoteProtocolError: remote sandbox returned an invalid execution
452
+ * reservation`, a message about a wire shape for a pod that is telling the
453
+ * truth about itself. Named here rather than in the shared controller
454
+ * because the two tiers answer it differently: see {@link
455
+ * KubernetesAgentRetiringError}.
87
456
  */
88
457
  async function requestChecked(
89
458
  wire: VsockAgentTransport,
@@ -92,6 +461,7 @@ async function requestChecked(
92
461
  ): Promise<unknown> {
93
462
  const response = await wire.request(request, signal)
94
463
  if (isUnauthorized(response)) throw new KubernetesAgentUnauthorizedError()
464
+ if (isAgentRetiring(response)) throw new KubernetesAgentRetiringError()
95
465
  return response
96
466
  }
97
467
 
@@ -119,161 +489,1956 @@ export interface KubernetesTransportTiming {
119
489
  readonly drainMs: number
120
490
  }
121
491
 
122
- export interface KubernetesTransportOptions extends VsockTransportOptions {
492
+ /**
493
+ * What one `healthz` reply says about the agent, with the fence kept rather
494
+ * than collapsed into a boolean.
495
+ *
496
+ * {@link VsockAgentTransport.healthz} answers `false` both for an agent that
497
+ * did not reply and for one that replied "I have fenced myself", and those
498
+ * are opposite facts for a host deciding what to do next: the first is a pod
499
+ * that may be perfectly fine a second from now, the second is a pod that will
500
+ * refuse every call until it is replaced.
501
+ */
502
+ export interface KubernetesAgentHealth {
503
+ /** The reply's own `ok` — `true` only for an agent serving normally. */
504
+ readonly ok: boolean
505
+ /**
506
+ * The agent has fenced itself and only a new pod clears it.
507
+ *
508
+ * Read from the reply's own `retiring` flag and from nothing else. A
509
+ * not-`ok` reply without it is NOT inferred to be a fence: the connection
510
+ * gate answers a `healthz` that arrived over one unauthenticated
511
+ * connection too many, or behind an exhausted pre-auth buffer, with a
512
+ * named `{ ok: false, error }` and no flag — and a caller told `retiring`
513
+ * there would suspend and resume a perfectly healthy pod. So `ok: false`
514
+ * with `retiring: false` is its own answer: this reply says nothing about
515
+ * whether the agent is serving.
516
+ */
517
+ readonly retiring: boolean
518
+ }
519
+
520
+ /**
521
+ * `permanentDialFailure` is deliberately NOT inherited: this transport sets
522
+ * its own (see the constructor), so advertising the field would be offering a
523
+ * caller a predicate that is silently overwritten. `onGuestReply` is
524
+ * re-declared rather than inherited, with the pod the reply came FROM — see
525
+ * below.
526
+ */
527
+ export interface KubernetesTransportOptions
528
+ extends Omit<VsockTransportOptions, 'permanentDialFailure' | 'onGuestReply'> {
123
529
  /**
124
530
  * Fires once per completed `exec()` call (success or failure) with
125
531
  * the four phase durations above. The payload is exactly those four
126
532
  * numbers — never the token, never a command, argv, or output.
127
533
  */
128
534
  readonly onTiming?: (timing: KubernetesTransportTiming) => void
535
+ /**
536
+ * Re-read the live pod behind this sandbox and hand back the address
537
+ * and token it answers on NOW.
538
+ *
539
+ * Set only by the `pod-ip` address mode, where the handle carries a
540
+ * literal IP that dies with its pod; a Service FQDN needs none of this
541
+ * because the name outlives the pod and the dial re-resolves it every
542
+ * call. Consulted at most ONCE per call, and only after a dial that
543
+ * failed at connect — see {@link KubernetesAgentTransport}.
544
+ */
545
+ readonly refreshHandle?: (signal?: AbortSignal) => Promise<KubernetesAgentHandle>
546
+ /**
547
+ * {@link VsockTransportOptions.onGuestReply}, plus the one thing the
548
+ * shared transport cannot say and this one always can: the bind token of
549
+ * the WIRE the reply arrived on, which is the uid of the pod that
550
+ * answered.
551
+ *
552
+ * A rebind builds a replacement wire and does not close the wire it
553
+ * leaves, so a call dispatched to the outgoing pod can still be answered
554
+ * BY it — a long `write-file`, a `read-file`, a pod inside its
555
+ * termination grace period — minutes after this transport has followed
556
+ * the replacement. That reply is a true statement about a pod this
557
+ * transport is no longer bound to, and a listener that could not tell the
558
+ * two apart would pair the departed pod's agent process with the
559
+ * replacement's uid: an identity naming a process that never ran there.
560
+ *
561
+ * Bound per wire, so the token is the one the reply's own connection
562
+ * presented and never the one the transport happens to hold now —
563
+ * `undefined` on a handle carrying no token at all, which is a wire that
564
+ * cannot name the pod that answered and so proves nothing about it.
565
+ */
566
+ readonly onGuestReply?: (reply: GuestReplyIdentity, podUid: string | undefined) => void
129
567
  }
130
568
 
569
+ // --- detachable executions (#479) -----------------------------------------
570
+
571
+ /** How long a detached `exec()` keeps trying to get its stream back. */
572
+ const DEFAULT_REATTACH_WINDOW_MS = 30_000
573
+ /** Pause between reattach attempts, so a refusing port is not hot-looped. */
574
+ const REATTACH_RETRY_DELAY_MS = 250
575
+ /** The guest's own default, mirrored so an observation bound exists. */
576
+ const DEFAULT_EXECUTION_TIMEOUT_MS = 5 * 60 * 1_000
577
+ /** Slack over the command's own timeout, as `executeRaw` allows itself. */
578
+ const EXECUTION_OBSERVATION_GRACE_MS = 10_000
579
+ /** How long a confirmed cancel is retried before it is reported unknown. */
580
+ const CANCEL_CONFIRM_WINDOW_MS = 8_000
581
+ const CANCEL_ATTEMPT_TIMEOUT_MS = 2_000
582
+ const MAX_TIMER_DELAY_MS = 2_147_483_647
583
+
131
584
  /**
132
- * The kubernetes backend's dialable transport: a `tcp` handle plus a
133
- * `RemoteExecutionAdapter` built from it, so `exec()` gets the same
134
- * reserve-before-admission behaviour (cancellation, timeout ownership,
135
- * "do not infer complete output on an ambiguous cancel") every other
136
- * remote backend gets, while every network operation is delegated to
137
- * {@link VsockAgentTransport} for the actual dial/frame/token work.
585
+ * Thrown before a command is admitted, when the caller asked for a
586
+ * detachable execution and the guest does not advertise
587
+ * {@link EXECUTION_ATTACH_FEATURE}.
588
+ *
589
+ * Refused rather than downgraded: a caller that asked for detach is about
590
+ * to rely on being able to come back for the output, and running the
591
+ * command anyway would keep nothing and tell nobody.
138
592
  */
139
- export class KubernetesAgentTransport {
140
- private readonly handle: KubernetesAgentHandle
141
- private readonly transportOptions: VsockTransportOptions
142
- private readonly onTiming?: (timing: KubernetesTransportTiming) => void
143
- /** Simple pass-through operations share one transport instance. */
144
- private readonly wire: VsockAgentTransport
593
+ export class KubernetesExecutionAttachUnsupportedError extends Error {
594
+ override readonly name = 'KubernetesExecutionAttachUnsupportedError'
145
595
 
146
- constructor(handle: KubernetesAgentHandle, options: KubernetesTransportOptions = {}) {
147
- const { onTiming, ...transportOptions } = options
148
- this.handle = handle
149
- this.transportOptions = transportOptions
150
- this.onTiming = onTiming
151
- this.wire = new VsockAgentTransport(handle, transportOptions)
596
+ constructor(
597
+ readonly feature: string,
598
+ message: string,
599
+ ) {
600
+ super(message)
152
601
  }
602
+ }
153
603
 
154
- /** Readiness probe — never requires a token; see `protocol.ts`. */
155
- async healthz(signal?: AbortSignal): Promise<boolean> {
156
- return await this.wire.healthz(signal)
157
- }
604
+ /** Why an `attach-execution` could not be served. */
605
+ export type KubernetesAttachRefusal =
606
+ | 'unknown_execution'
607
+ | 'output_not_retained'
608
+ | 'invalid_offset'
609
+ | 'invalid_execution_id'
610
+ | 'agent_retiring'
611
+ | 'unknown'
158
612
 
159
- /** Poll until `healthz` succeeds or the timeout elapses. */
160
- async waitForReady(
161
- timeoutMs: number,
162
- pollIntervalMs: number,
163
- signal?: AbortSignal,
164
- ): Promise<void> {
165
- return await this.wire.waitForReady(timeoutMs, pollIntervalMs, signal)
166
- }
613
+ /**
614
+ * Thrown when the guest ANSWERED an attach and refused it — the execution
615
+ * is past its retention, ran in a pod that has since been replaced, never
616
+ * asked for its output to be kept, or the offset names bytes it does not
617
+ * have.
618
+ *
619
+ * Distinct from a transport failure on purpose: a refusal will not become a
620
+ * success by being retried, so the reattach loop stops on it instead of
621
+ * spending its whole window re-asking a question already answered.
622
+ */
623
+ export class KubernetesExecutionNotAttachableError extends Error {
624
+ override readonly name = 'KubernetesExecutionNotAttachableError'
167
625
 
168
626
  /**
169
- * The raw `reserve-execution` primitive, exposed directly (rather than
170
- * only reachable as a side effect of `exec()`) so the reservation
171
- * round trip is independently observable and testable against the
172
- * real guest.
627
+ * The state the guest reported for this execution, when it reported
628
+ * one. `'reserved'` is the one that changes what a caller should do:
629
+ * the command was never started, so nothing is running.
173
630
  */
174
- async reserve(signal?: AbortSignal): Promise<unknown> {
175
- return await requestChecked(this.wire, { op: 'reserve-execution' }, signal)
631
+ readonly executionState: string | undefined
632
+
633
+ constructor(
634
+ readonly executionId: string,
635
+ readonly reason: KubernetesAttachRefusal,
636
+ message: string,
637
+ options?: { cause?: unknown; state?: string },
638
+ ) {
639
+ super(message, options)
640
+ this.executionState = options?.state
176
641
  }
642
+ }
177
643
 
178
- /**
179
- * The raw `cancel-execution` primitive, exposed for the same reason
180
- * {@link reserve} is: it is a control request with its own refusal
181
- * semantics, and proving those against the real guest should not require
182
- * driving a whole cancelled `exec()` to reach it.
183
- */
184
- async cancel(executionId: string, signal?: AbortSignal): Promise<unknown> {
185
- return await requestChecked(
186
- this.wire,
187
- { op: 'cancel-execution', body: { executionId } },
188
- signal,
189
- )
644
+ /**
645
+ * Thrown when a detached `exec()` gave up OBSERVING a command that is, as
646
+ * far as this host knows, still the guest's to run.
647
+ *
648
+ * The two fields are what makes it recoverable rather than merely a
649
+ * failure: `executionId` names the command to a second host process, and
650
+ * `outputOffset` is the byte the next `attachExecution` should resume from
651
+ * so nothing is read twice and no gap is invented.
652
+ *
653
+ * Nothing on this path cancels to reconcile. That is the whole point of
654
+ * the feature: a reset connection used to cost the workspace its pod, and
655
+ * a command the host has stopped watching is not a command that has to
656
+ * die.
657
+ */
658
+ export class KubernetesExecutionDetachedError extends Error {
659
+ override readonly name = 'KubernetesExecutionDetachedError'
660
+
661
+ constructor(
662
+ readonly executionId: string,
663
+ readonly outputOffset: number,
664
+ message: string,
665
+ options?: { cause?: unknown },
666
+ ) {
667
+ super(message, options)
190
668
  }
669
+ }
670
+
671
+ /** What the guest reports about a finished execution on an attach. */
672
+ interface AttachTerminal {
673
+ readonly outcome: 'completed' | 'cancelled' | 'failed'
674
+ readonly result?: RemoteTerminalMetadata
675
+ readonly error?: string
676
+ }
677
+
678
+ /**
679
+ * Everything one detached observation has read so far, across however many
680
+ * connections it took.
681
+ *
682
+ * `offset` is the guest's own byte offset into the execution's retained
683
+ * log, and it is the reason this is a mutable cursor rather than a return
684
+ * value: a reattach resumes from it, and every chunk a previous connection
685
+ * delivered has to be behind it.
686
+ */
687
+ interface OutputCursor {
688
+ offset: number
689
+ stdout: string
690
+ stderr: string
691
+ /** Bytes the guest had already evicted when a reattach asked for them. */
692
+ droppedBytes: number
693
+ }
694
+
695
+ /** One `stdout_delta`/`stderr_delta` payload, from either op's stream. */
696
+ function deltaStream(type: unknown): 'stdout' | 'stderr' | undefined {
697
+ if (type === 'stdout_delta') return 'stdout'
698
+ if (type === 'stderr_delta') return 'stderr'
699
+ return undefined
700
+ }
701
+
702
+ function isAttachRefusal(value: string): value is KubernetesAttachRefusal {
703
+ return (
704
+ value === 'unknown_execution' ||
705
+ value === 'output_not_retained' ||
706
+ value === 'invalid_offset' ||
707
+ value === 'invalid_execution_id' ||
708
+ value === 'agent_retiring'
709
+ )
710
+ }
191
711
 
192
- async writeFile(path: string, content: Buffer): Promise<void> {
193
- return await this.wire.writeFile(path, content)
712
+ function terminalMetadataOrThrow(value: unknown, executionId: string): RemoteTerminalMetadata {
713
+ if (!value || typeof value !== 'object') {
714
+ throw new RemoteProtocolError(
715
+ `kubernetes: the guest ended the attach stream for ${executionId} without terminal metadata`,
716
+ )
194
717
  }
718
+ return value as RemoteTerminalMetadata
719
+ }
720
+
721
+ function pause(ms: number): Promise<void> {
722
+ return new Promise((resolve) => {
723
+ const timer = setTimeout(resolve, ms)
724
+ timer.unref?.()
725
+ })
726
+ }
727
+
728
+ /**
729
+ * The id shape the guest enforces, mirrored here so a caller-chosen id is
730
+ * refused locally with a message that says what the shape is, rather than
731
+ * as an `invalid_execution_id` frame after a round trip.
732
+ */
733
+ const EXECUTION_ID_PATTERN =
734
+ /^exec_[0-9a-f]{8}-[0-9a-f]{4}-[1-5][0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$/i
735
+
736
+ function assertExecutionId(executionId: string): void {
737
+ if (EXECUTION_ID_PATTERN.test(executionId)) return
738
+ throw new RemoteProtocolError(
739
+ `kubernetes: ${JSON.stringify(executionId)} is not a valid execution id. The guest accepts exec_<uuid> and nothing else, so that an id minted by one host process is recognisable to another.`,
740
+ )
741
+ }
195
742
 
196
- async readFile(path: string): Promise<Buffer> {
197
- return await this.wire.readFile(path)
743
+ /**
744
+ * Fold one delta frame into the cursor, taking the offset FROM THE GUEST,
745
+ * and pass the output on to the caller.
746
+ *
747
+ * The host never derives an offset from the string it received, on either
748
+ * stream, and that is the whole of this function's reason to exist.
749
+ * `Buffer.toString('utf8')` over a chunk that ends mid-character does not
750
+ * preserve byte length — an incomplete sequence decodes to U+FFFD, which
751
+ * is WIDER than the bytes it replaced — so a cursor advanced by
752
+ * `Buffer.byteLength(data)` runs ahead of the guest's retained log the
753
+ * first time a multi-byte character straddles a read boundary. A drifted
754
+ * cursor is not a cosmetic error: the reattach either resumes past bytes
755
+ * that are then never delivered and never reported (a silent hole in a
756
+ * result whose truncation flags both read `false`) or names an offset the
757
+ * guest never had and is refused `invalid_offset`, which costs the caller
758
+ * the feature entirely. The guest stamps `nextOffset` on every delta of a
759
+ * retained execution, on the execute stream and the attach stream alike.
760
+ */
761
+ function applyDelta(
762
+ cursor: OutputCursor,
763
+ executionId: string,
764
+ stream: 'stdout' | 'stderr',
765
+ event: Record<string, unknown>,
766
+ onOutput?: SandboxExecOptions['onOutput'],
767
+ ): void {
768
+ const data = typeof event.data === 'string' ? event.data : ''
769
+ const nextOffset = Number(event.nextOffset)
770
+ if (!Number.isFinite(nextOffset)) {
771
+ throw new RemoteProtocolError(
772
+ `kubernetes: the guest sent a ${stream} delta for execution ${executionId} without the byte offset a reattach resumes from. Every guest advertising '${EXECUTION_ATTACH_FEATURE}' stamps them on a retained execution's output; rebuild the workspace image from this Namzu release.`,
773
+ )
198
774
  }
775
+ cursor.offset = nextOffset
776
+ if (stream === 'stdout') cursor.stdout += data
777
+ else cursor.stderr += data
778
+ onOutput?.({ stream, data })
779
+ }
780
+
781
+ /**
782
+ * The guest's refusal to start a command on an id that is no longer
783
+ * `reserved` — another host process got its `execute` in first.
784
+ *
785
+ * It reads as terminal (the guest answered, and answering again will not
786
+ * change it) and it is the one case where that is the wrong conclusion:
787
+ * the command this call asked for EXISTS, so the caller gets it by
788
+ * attaching rather than an error about a race it does not care about.
789
+ */
790
+ function lostTheStartRace(error: unknown): boolean {
791
+ return error instanceof RemoteCommandError && error.message.startsWith('execution_not_reserved')
792
+ }
793
+
794
+ /**
795
+ * Errors a reattach must NOT spend its window re-asking about: the guest
796
+ * answered, and the answer will be the same next time.
797
+ */
798
+ function isTerminalAttachError(error: unknown): boolean {
799
+ return (
800
+ error instanceof KubernetesExecutionNotAttachableError ||
801
+ error instanceof KubernetesAgentUnauthorizedError ||
802
+ error instanceof KubernetesAgentAddressUnresolvableError ||
803
+ error instanceof RemoteCommandError ||
804
+ error instanceof RemoteProtocolError
805
+ )
806
+ }
199
807
 
200
- async openTerminal(options: OpenTerminalOptions): Promise<TerminalSession> {
201
- return await this.wire.openTerminal(options)
808
+ /** The guest's `cancel-execution` reply, refusals told apart from blips. */
809
+ function parseCancellationReply(
810
+ executionId: string,
811
+ response: unknown,
812
+ ): RemoteCancellationAcknowledgement {
813
+ const reply = (response ?? {}) as Record<string, unknown>
814
+ if (reply.ok === true) {
815
+ const state = String(reply.state ?? '')
816
+ if (state === 'cancelled' || state === 'completed' || state === 'failed') {
817
+ return reply as unknown as RemoteCancellationAcknowledgement
818
+ }
819
+ throw new RemoteProtocolError(
820
+ `kubernetes: the guest acknowledged cancelling ${executionId} with an unknown state ${JSON.stringify(reply.state)}`,
821
+ )
202
822
  }
823
+ const error = typeof reply.error === 'string' ? reply.error : 'unknown'
824
+ if (error === 'unknown_execution' || error === 'invalid_execution_id') {
825
+ throw new KubernetesExecutionNotAttachableError(
826
+ executionId,
827
+ error,
828
+ `kubernetes: the guest holds no execution ${executionId} to cancel (${error}). A record is kept only for its retention window and is lost when the pod is replaced.`,
829
+ )
830
+ }
831
+ throw new Error(`kubernetes: the guest refused to cancel ${executionId}: ${error}`)
832
+ }
203
833
 
204
- async openTcpConnection(options: SandboxTcpConnectOptions): Promise<SandboxTcpConnection> {
205
- return await this.wire.openTcpConnection(options)
834
+ /**
835
+ * The SDK-shaped result of an attached observation.
836
+ *
837
+ * A reported gap sets BOTH truncation flags. The retained log is one
838
+ * interleaved space, so bytes lost out of it cannot be attributed to
839
+ * stdout or to stderr, and the contract already has exactly one way to say
840
+ * "this output is not all of it". Saying it on one stream only would be a
841
+ * guess; saying it on neither would hand back a short stream that looks
842
+ * complete, which is the thing this design refuses to do.
843
+ */
844
+ function resultFromAttachTerminal(
845
+ executionId: string,
846
+ terminal: AttachTerminal,
847
+ cursor: OutputCursor,
848
+ ): SandboxExecResult {
849
+ if (terminal.outcome === 'failed' && terminal.result === undefined) {
850
+ throw new RemoteCommandError(
851
+ terminal.error ?? `the guest reported execution ${executionId} as failed`,
852
+ )
206
853
  }
854
+ const metadata = terminalMetadataOrThrow(terminal.result, executionId)
855
+ const lost = cursor.droppedBytes > 0
856
+ return {
857
+ exitCode: metadata.exitCode,
858
+ stdout: cursor.stdout,
859
+ stderr: cursor.stderr,
860
+ ...(metadata.signal !== undefined ? { signal: metadata.signal } : {}),
861
+ timedOut: metadata.timedOut === true,
862
+ durationMs: metadata.durationMs,
863
+ stdoutTruncated: metadata.stdoutTruncated === true || lost,
864
+ stderrTruncated: metadata.stderrTruncated === true || lost,
865
+ }
866
+ }
207
867
 
868
+ /**
869
+ * What a caller passes to read an execution it did not necessarily start.
870
+ *
871
+ * `signal` here is an OBSERVATION signal, not the SDK's command signal:
872
+ * aborting it stops reading and leaves the command running. That is the
873
+ * opposite of `SandboxExecOptions.signal`, and it is why this type does not
874
+ * extend it.
875
+ */
876
+ export interface KubernetesAttachExecutionOptions {
877
+ /** Byte offset to resume from. Default 0 — the whole retained log. */
878
+ readonly fromOffset?: number
879
+ readonly onOutput?: SandboxExecOptions['onOutput']
880
+ /** Stops OBSERVING. Never cancels; see {@link KubernetesAgentTransport.cancelExecution}. */
881
+ readonly signal?: AbortSignal
208
882
  /**
209
- * Run one command through a fresh, call-scoped adapter + controller.
210
- * Fresh per call — not shared instance state — because
211
- * {@link VsockAgentTransport} itself carries no cross-call connection
212
- * state (it dials fresh every time), so building one per `exec()` is
213
- * free and makes concurrent `exec()` calls on the same
214
- * `KubernetesAgentTransport` correctly independent: each gets its own
215
- * `onDial` closure and its own timing accumulator, with no shared
216
- * mutable field for two in-flight calls to race on.
883
+ * Called when the guest reports that bytes the caller asked for had
884
+ * already been evicted from the retained log. The result's truncation
885
+ * flags say the same thing; this says how much.
217
886
  */
218
- async exec(
219
- command: string,
220
- argv?: string[],
221
- opts?: SandboxExecOptions,
222
- ): Promise<SandboxExecResult> {
223
- let dialMs = 0
224
- let reserveMs = 0
225
- let executeMs = 0
226
- let executeSettledAt = 0
887
+ readonly onGap?: (gap: {
888
+ readonly executionId: string
889
+ readonly fromOffset: number
890
+ readonly droppedBytes: number
891
+ }) => void
892
+ }
227
893
 
228
- const timedWire = new VsockAgentTransport(this.handle, {
229
- ...this.transportOptions,
230
- onDial: (ms) => {
231
- dialMs += ms
232
- },
233
- })
894
+ /**
895
+ * An `exec()` whose observation can be lost and taken up again — on this
896
+ * handle or in another host process — instead of costing the command its
897
+ * life.
898
+ *
899
+ * Deliberately NOT on the SDK's `SandboxExecOptions`: every other backend
900
+ * would then have to answer for a field it cannot honour, and the SDK's
901
+ * exec contract stays exactly what it was. This is a Kubernetes workspace
902
+ * surface, layered over the shared options type rather than widening it.
903
+ */
904
+ export interface KubernetesDetachedExecOptions
905
+ extends SandboxExecOptions,
906
+ Pick<KubernetesAttachExecutionOptions, 'onGap'> {
907
+ /**
908
+ * The id this command is known by, to this process and to any other.
909
+ * Minted here when absent. Reserving an id the guest still holds does
910
+ * NOT start a second command: the call attaches to the one that exists,
911
+ * which is what makes a retried start idempotent for as long as the
912
+ * record lives.
913
+ */
914
+ readonly executionId?: string
915
+ /**
916
+ * Ask the guest to retain this command's output so the observation can
917
+ * be resumed. Setting `executionId` implies it; the flag is what a
918
+ * caller that does not care about the id passes.
919
+ */
920
+ readonly detach?: boolean
921
+ /**
922
+ * Stop observing and leave the command running — for a host that is
923
+ * shutting down. Rejects with {@link KubernetesExecutionDetachedError},
924
+ * which names the id and the offset to resume from.
925
+ *
926
+ * The opposite of `signal`, which keeps the SDK contract and terminates.
927
+ */
928
+ readonly detachSignal?: AbortSignal
929
+ /**
930
+ * How long a lost connection is retried before the call gives up and
931
+ * reports itself detached. Default 30s.
932
+ */
933
+ readonly reattachWindowMs?: number
934
+ }
234
935
 
235
- const adapter: RemoteExecutionAdapter<Pick<ExecRequest, 'stdin' | 'maxOutputBytes'>> = {
236
- label: 'kubernetes pod-network agent',
237
- reserve: async (signal) => {
238
- const startedAt = Date.now()
239
- try {
240
- return await requestChecked(timedWire, { op: 'reserve-execution' }, signal)
241
- } finally {
242
- reserveMs += Date.now() - startedAt
243
- }
244
- },
245
- // Checked exactly like `reserve` — see `requestChecked`.
246
- cancel: async (executionId, signal) =>
247
- await requestChecked(timedWire, { op: 'cancel-execution', body: { executionId } }, signal),
248
- execute: async (executionId, cmd, execArgv, execOpts, signal, context) => {
249
- const startedAt = Date.now()
250
- try {
251
- return await timedWire.executeStreamed(
252
- {
253
- ...(executionId ? { executionId } : {}),
254
- command: cmd,
255
- args: execArgv ?? [],
256
- ...(execOpts?.cwd !== undefined ? { cwd: execOpts.cwd } : {}),
257
- ...(execOpts?.env !== undefined ? { env: execOpts.env } : {}),
258
- ...(execOpts?.timeout !== undefined ? { timeoutMs: execOpts.timeout } : {}),
259
- ...(context?.stdin !== undefined ? { stdin: context.stdin } : {}),
260
- ...(context?.maxOutputBytes !== undefined
261
- ? { maxOutputBytes: context.maxOutputBytes }
262
- : {}),
263
- },
264
- execOpts,
265
- signal,
266
- )
267
- } finally {
268
- executeMs += Date.now() - startedAt
269
- executeSettledAt = Date.now()
270
- }
936
+ // --- guest sessions (#478) ------------------------------------------------
937
+
938
+ /**
939
+ * Thrown before anything is started, when the caller asked for a session and
940
+ * the guest does not advertise {@link SESSIONS_FEATURE}.
941
+ *
942
+ * Refused, never downgraded to a connection-bound terminal. A caller that
943
+ * asked for a session is about to rely on coming back to it after its own
944
+ * process has been replaced; handing it one that dies with the socket would
945
+ * look like it worked until the one moment it was needed.
946
+ */
947
+ export class KubernetesSessionsUnsupportedError extends Error {
948
+ override readonly name = 'KubernetesSessionsUnsupportedError'
949
+
950
+ constructor(
951
+ readonly feature: string,
952
+ message: string,
953
+ ) {
954
+ super(message)
955
+ }
956
+ }
957
+
958
+ /** Why the guest refused a session request. */
959
+ export type KubernetesSessionRefusal =
960
+ | 'unknown_session'
961
+ | 'invalid_session_id'
962
+ | 'invalid_offset'
963
+ | 'session_exists'
964
+ | 'session_capacity'
965
+ | 'missing_command'
966
+ | 'spawn_failed'
967
+ | 'agent_retiring'
968
+ | 'unknown'
969
+
970
+ const SESSION_REFUSALS = new Set<KubernetesSessionRefusal>([
971
+ 'unknown_session',
972
+ 'invalid_session_id',
973
+ 'invalid_offset',
974
+ 'session_exists',
975
+ 'session_capacity',
976
+ 'missing_command',
977
+ 'spawn_failed',
978
+ 'agent_retiring',
979
+ ])
980
+
981
+ function isSessionRefusal(value: string): value is KubernetesSessionRefusal {
982
+ return SESSION_REFUSALS.has(value as KubernetesSessionRefusal)
983
+ }
984
+
985
+ /**
986
+ * Thrown when the guest ANSWERED and refused: the session is past its
987
+ * retention, ran in a pod that has since been replaced, the id is already
988
+ * taken, or the offset names bytes it does not have.
989
+ *
990
+ * Distinct from a transport failure for the same reason
991
+ * {@link KubernetesExecutionNotAttachableError} is: a refusal does not
992
+ * become a success by being retried.
993
+ */
994
+ export class KubernetesSessionRefusedError extends Error {
995
+ override readonly name = 'KubernetesSessionRefusedError'
996
+
997
+ constructor(
998
+ readonly sessionId: string,
999
+ readonly reason: KubernetesSessionRefusal,
1000
+ message: string,
1001
+ options?: { cause?: unknown },
1002
+ ) {
1003
+ super(message, options)
1004
+ }
1005
+ }
1006
+
1007
+ /** One row of {@link KubernetesAgentTransport.listSessions}. */
1008
+ export interface KubernetesSessionSummary {
1009
+ readonly sessionId: string
1010
+ readonly kind: SessionKind
1011
+ /** The program, as it was asked for. Never the environment it was given. */
1012
+ readonly command: string
1013
+ readonly args: readonly string[]
1014
+ readonly startedAt: number
1015
+ readonly lastInputAt?: number
1016
+ readonly lastOutputAt?: number
1017
+ /** Pass as `fromOffset` to read everything this session has printed since. */
1018
+ readonly nextOffset: number
1019
+ /** Bytes the ring has evicted over this session's life. */
1020
+ readonly droppedBytes: number
1021
+ readonly state: SessionState
1022
+ /** Whether a host process is attached to it right now. */
1023
+ readonly attached: boolean
1024
+ readonly exitCode?: number
1025
+ readonly signal?: number
1026
+ }
1027
+
1028
+ /**
1029
+ * One read of a session's retained output, in the SDK's
1030
+ * `BackgroundJobOutput` shape — deliberately, because it answers the same
1031
+ * question for the same kind of consumer and a second vocabulary for
1032
+ * "here is the next chunk and here is what you missed" helps nobody.
1033
+ */
1034
+ export interface KubernetesSessionOutput {
1035
+ readonly chunk: string
1036
+ readonly nextOffset: number
1037
+ readonly droppedBytes: number
1038
+ readonly status: BackgroundJobStatus
1039
+ readonly exitCode?: number
1040
+ }
1041
+
1042
+ /**
1043
+ * A terminal on a workspace, with the three things only a SESSION's reader
1044
+ * needs. On a connection-bound terminal the two optional members are absent,
1045
+ * which is the honest answer: there is no session to name and nothing to
1046
+ * detach from.
1047
+ */
1048
+ export interface KubernetesWorkspaceTerminal extends TerminalSession {
1049
+ /** Present exactly when this terminal belongs to a guest session. */
1050
+ readonly sessionId?: string
1051
+ /** One past the newest retained byte delivered so far. */
1052
+ nextOffset?(): number | undefined
1053
+ /**
1054
+ * Stop reading and leave the program running — the opposite of
1055
+ * {@link TerminalSession.kill}. `exited` then rejects with
1056
+ * `AgentSessionDetachedError`, because a resolved `exited` would claim
1057
+ * an exit that did not happen.
1058
+ */
1059
+ detach?(): void
1060
+ }
1061
+
1062
+ /**
1063
+ * What an ATTACH hands back: the same terminal, with the three session
1064
+ * members present rather than optional. `openTerminal` returns the looser
1065
+ * type because it serves both shapes and a connection-bound terminal
1066
+ * genuinely has no session to name.
1067
+ */
1068
+ export interface KubernetesSessionTerminal extends KubernetesWorkspaceTerminal {
1069
+ readonly sessionId: string
1070
+ nextOffset(): number | undefined
1071
+ detach(): void
1072
+ }
1073
+
1074
+ /** `openTerminal` on a workspace, widened by the two session fields. */
1075
+ export interface KubernetesOpenTerminalOptions extends OpenTerminalOptions {
1076
+ /**
1077
+ * Name this terminal so a later host process can find it again.
1078
+ * Requires {@link persistent}; on its own it names nothing.
1079
+ */
1080
+ readonly sessionId?: string
1081
+ /**
1082
+ * Hand the PTY to the guest's session registry rather than to this
1083
+ * connection. Losing the connection then DETACHES — no signal is sent,
1084
+ * and the program ends when it exits, on `killSession`, or when the pod
1085
+ * stops.
1086
+ */
1087
+ readonly persistent?: boolean
1088
+ }
1089
+
1090
+ /** Rejoin a terminal session that is already running. */
1091
+ export interface KubernetesAttachTerminalOptions {
1092
+ /** Byte offset to replay from. Default 0 — everything the ring still holds. */
1093
+ readonly fromOffset?: number
1094
+ /** Resize the PTY on attach, for a reader whose window is a different shape. */
1095
+ readonly size?: { readonly cols: number; readonly rows: number }
1096
+ }
1097
+
1098
+ /** Start a program with no terminal, which only a kill or the pod ends. */
1099
+ export interface KubernetesStartDetachedOptions {
1100
+ readonly sessionId: string
1101
+ readonly command: string
1102
+ readonly args?: readonly string[]
1103
+ readonly cwd?: string
1104
+ readonly env?: Record<string, string>
1105
+ }
1106
+
1107
+ /** Read a session's retained output without attaching to it. */
1108
+ export interface KubernetesReadSessionOptions {
1109
+ readonly fromOffset?: number
1110
+ }
1111
+
1112
+ function sessionNumber(value: unknown, fallback = 0): number {
1113
+ const parsed = Number(value)
1114
+ return Number.isFinite(parsed) ? parsed : fallback
1115
+ }
1116
+
1117
+ /** How long one `readSession` may spend reading a bounded, one-shot reply. */
1118
+ const SESSION_READ_TIMEOUT_MS = 30_000
1119
+
1120
+ /**
1121
+ * How long the SHARED re-read behind a rebind may take before it is given up
1122
+ * on — two API GETs on a client that sets no per-request timeout.
1123
+ *
1124
+ * It replaces the caller's signal rather than joining it, because the read is
1125
+ * shared: see {@link KubernetesAgentTransport.rebind}. Generous enough that a
1126
+ * busy API server still answers, short enough that a call refused by a
1127
+ * replacement is not held behind a hung one.
1128
+ */
1129
+ const REBIND_READ_TIMEOUT_MS = 10_000
1130
+
1131
+ /**
1132
+ * The same shape the guest enforces, checked here so a bad id is a local
1133
+ * error naming the rule rather than a round trip that comes back
1134
+ * `invalid_session_id`.
1135
+ */
1136
+ const SESSION_ID_PATTERN = /^[A-Za-z0-9][A-Za-z0-9._-]{0,63}$/
1137
+
1138
+ function assertSessionId(sessionId: string): void {
1139
+ if (!SESSION_ID_PATTERN.test(sessionId)) {
1140
+ throw new Error(
1141
+ `kubernetes: ${JSON.stringify(sessionId)} cannot name a session. It must be 1-64 characters of letters, digits, '.', '_' or '-', starting alphanumeric. The id is the ONLY way another host process finds this session again, so it is refused rather than sanitised.`,
1142
+ )
1143
+ }
1144
+ }
1145
+
1146
+ /**
1147
+ * Both session fields or neither.
1148
+ *
1149
+ * A `sessionId` without `persistent` would open a terminal that dies with
1150
+ * its connection under a name nothing can use, and `persistent` without an
1151
+ * id would open one nobody can ever find. Either alone is a mistake worth a
1152
+ * message rather than a surprise.
1153
+ */
1154
+ function assertSessionOpen(options: KubernetesOpenTerminalOptions): string {
1155
+ if (options.sessionId === undefined || options.persistent !== true) {
1156
+ throw new Error(
1157
+ 'kubernetes: a persistent terminal needs both `sessionId` and `persistent: true`. An id without `persistent` opens a connection-bound terminal under a name nothing can attach to, and `persistent` without an id opens one nobody can find again.',
1158
+ )
1159
+ }
1160
+ assertSessionId(options.sessionId)
1161
+ return options.sessionId
1162
+ }
1163
+
1164
+ /** The workspace-facing terminal around one open session stream. */
1165
+ function sessionTerminal(
1166
+ stream: AgentTerminalStream,
1167
+ sessionId: string,
1168
+ ): KubernetesSessionTerminal {
1169
+ return {
1170
+ ...stream.session,
1171
+ sessionId,
1172
+ nextOffset: () => stream.nextOffset(),
1173
+ detach: () => stream.detach(),
1174
+ }
1175
+ }
1176
+
1177
+ /** The guest's row, structurally validated. */
1178
+ function parseSessionSummary(value: unknown): KubernetesSessionSummary {
1179
+ if (!value || typeof value !== 'object') {
1180
+ throw new RemoteProtocolError('kubernetes: the guest sent a session row that is not an object')
1181
+ }
1182
+ const row = value as Record<string, unknown>
1183
+ if (typeof row.sessionId !== 'string') {
1184
+ throw new RemoteProtocolError('kubernetes: the guest sent a session row with no sessionId')
1185
+ }
1186
+ const kind = row.kind === 'detached' ? 'detached' : 'terminal'
1187
+ const state = row.state === 'exited' ? 'exited' : 'running'
1188
+ return {
1189
+ sessionId: row.sessionId,
1190
+ kind,
1191
+ command: typeof row.command === 'string' ? row.command : '',
1192
+ args: Array.isArray(row.args) ? row.args.map(String) : [],
1193
+ startedAt: sessionNumber(row.startedAt),
1194
+ ...(row.lastInputAt !== undefined ? { lastInputAt: sessionNumber(row.lastInputAt) } : {}),
1195
+ ...(row.lastOutputAt !== undefined ? { lastOutputAt: sessionNumber(row.lastOutputAt) } : {}),
1196
+ nextOffset: sessionNumber(row.nextOffset),
1197
+ droppedBytes: sessionNumber(row.droppedBytes),
1198
+ state,
1199
+ attached: row.attached === true,
1200
+ ...(typeof row.exitCode === 'number' ? { exitCode: row.exitCode } : {}),
1201
+ ...(typeof row.signal === 'number' ? { signal: row.signal } : {}),
1202
+ }
1203
+ }
1204
+
1205
+ /**
1206
+ * The SDK's three-way job status, from the guest's two-way state plus the
1207
+ * signal. A program the kernel stopped is `killed`, not `exited`: the
1208
+ * distinction is the whole reason `BackgroundJobStatus` has three members.
1209
+ */
1210
+ function sessionStatus(summary: {
1211
+ readonly state: SessionState
1212
+ readonly signal?: number
1213
+ }): BackgroundJobStatus {
1214
+ if (summary.state === 'running') return 'running'
1215
+ return summary.signal !== undefined ? 'killed' : 'exited'
1216
+ }
1217
+
1218
+ /** The refusal shape every session op answers a bad request with. */
1219
+ function sessionRefusal(
1220
+ sessionId: string,
1221
+ reply: Record<string, unknown>,
1222
+ operation: string,
1223
+ cause?: unknown,
1224
+ ): KubernetesSessionRefusedError {
1225
+ const code = typeof reply.error === 'string' ? reply.error : 'unknown'
1226
+ const detail = typeof reply.message === 'string' ? ` ${reply.message}` : ''
1227
+ return new KubernetesSessionRefusedError(
1228
+ sessionId,
1229
+ isSessionRefusal(code) ? code : 'unknown',
1230
+ `kubernetes: the guest refused ${operation} for session ${sessionId} (${code}).${detail} A session lives in the pod's memory only: it is lost when the pod is replaced, and an exited one is kept for the window NAMZU_AGENT_SESSION_TERMINAL_TTL_MS names.`,
1231
+ cause !== undefined ? { cause } : undefined,
1232
+ )
1233
+ }
1234
+
1235
+ /**
1236
+ * The same refusal, arriving on a STREAM rather than in a reply.
1237
+ *
1238
+ * `openTerminal` and `attachSession` do not get a `{ ok: false }` body: the
1239
+ * guest refuses them with an `error` FRAME, which the vsock transport
1240
+ * surfaces as a plain `Error` carrying the guest's code as its message. Left
1241
+ * alone, those two would be the only session verbs a caller could not catch
1242
+ * by class — so the code is recognised here, at the one boundary where it is
1243
+ * still recognisable, and everything else (a dial failure, an idle timeout)
1244
+ * is handed back untouched.
1245
+ */
1246
+ function sessionStreamFailure(sessionId: string, operation: string, error: unknown): unknown {
1247
+ if (!(error instanceof Error) || !isSessionRefusal(error.message)) return error
1248
+ return sessionRefusal(sessionId, { error: error.message }, operation, error)
1249
+ }
1250
+
1251
+ /**
1252
+ * The kubernetes backend's dialable transport: a `tcp` handle plus a
1253
+ * `RemoteExecutionAdapter` built from it, so `exec()` gets the same
1254
+ * reserve-before-admission behaviour (cancellation, timeout ownership,
1255
+ * "do not infer complete output on an ambiguous cancel") every other
1256
+ * remote backend gets, while every network operation is delegated to
1257
+ * {@link VsockAgentTransport} for the actual dial/frame/token work.
1258
+ */
1259
+ // --- quiesce (#485) -------------------------------------------------------
1260
+
1261
+ /**
1262
+ * Thrown when a quiesce was asked of a guest that cannot perform one:
1263
+ * either its `healthz` does not advertise {@link QUIESCE_FEATURE}, or it
1264
+ * answered `unknown_op: quiesce`.
1265
+ *
1266
+ * Refused rather than treated as "nothing was running". A caller asks for a
1267
+ * quiesce because it is about to read the disk and needs it still; an image
1268
+ * that cannot stop its processes has to say so, not resolve with an empty
1269
+ * list that reads exactly like a guest which had nothing to stop.
1270
+ */
1271
+ export class KubernetesQuiesceUnsupportedError extends Error {
1272
+ override readonly name = 'KubernetesQuiesceUnsupportedError'
1273
+
1274
+ constructor(
1275
+ readonly feature: string,
1276
+ message: string,
1277
+ ) {
1278
+ super(message)
1279
+ }
1280
+ }
1281
+
1282
+ /**
1283
+ * Thrown when nothing can promise the guest is quiet.
1284
+ *
1285
+ * Usually because the guest ANSWERED and said so — a process survived
1286
+ * `SIGKILL`, the scan could not be performed, or the op ran out of its own
1287
+ * deadline — and then the guest's own message names the pid. The workspace
1288
+ * handle raises the same class for the one case that never reaches a guest:
1289
+ * a `suspend({ quiesce: true })` arriving while a suspend WITHOUT a quiesce
1290
+ * is already in flight, whose patch has gone over processes nobody stopped
1291
+ * (`reason: 'suspend_already_in_flight'`). Both mean the one thing a caller
1292
+ * has to act on: do not trust a capture taken now.
1293
+ *
1294
+ * Its own class for the same reason {@link KubernetesSessionRefusedError}
1295
+ * is: this is an answer, not a transport failure, and retrying it is a
1296
+ * decision the caller makes with the pid in hand rather than one a wrapper
1297
+ * makes on its behalf.
1298
+ */
1299
+ export class KubernetesQuiesceUnconfirmedError extends Error {
1300
+ override readonly name = 'KubernetesQuiesceUnconfirmedError'
1301
+
1302
+ constructor(
1303
+ readonly reason: string,
1304
+ message: string,
1305
+ ) {
1306
+ super(message)
1307
+ }
1308
+ }
1309
+
1310
+ /** What the guest stopped, and how widely it was allowed to look. */
1311
+ export interface KubernetesQuiesceReport {
1312
+ /** Every process signalled, with the last signal it was actually sent. */
1313
+ readonly stopped: readonly QuiescedProcess[]
1314
+ /** See {@link QuiesceScope}. `owned-sessions` is the narrowed one. */
1315
+ readonly scope: QuiesceScope
1316
+ /** The per-round SIGTERM window the guest used, after its own clamp. */
1317
+ readonly graceMs: number
1318
+ /** Scan-and-signal passes. `0` means nothing was running. */
1319
+ readonly rounds: number
1320
+ }
1321
+
1322
+ function quiesceReport(reply: Record<string, unknown>): KubernetesQuiesceReport {
1323
+ const stopped = Array.isArray(reply.stopped) ? reply.stopped : []
1324
+ return {
1325
+ stopped: stopped.flatMap((entry): QuiescedProcess[] => {
1326
+ if (!entry || typeof entry !== 'object') return []
1327
+ const row = entry as Record<string, unknown>
1328
+ if (typeof row.pid !== 'number') return []
1329
+ return [
1330
+ {
1331
+ pid: row.pid,
1332
+ command: typeof row.command === 'string' ? row.command : '',
1333
+ signal: row.signal === 'SIGKILL' ? 'SIGKILL' : 'SIGTERM',
1334
+ },
1335
+ ]
1336
+ }),
1337
+ // The WIDER claim has to be said in so many words. A reply carrying
1338
+ // a scope this host does not recognise reads as the narrow one,
1339
+ // because the only wrong answer here is telling a caller its guest
1340
+ // was swept completely when nothing says so.
1341
+ scope: reply.scope === 'pid-namespace' ? 'pid-namespace' : 'owned-sessions',
1342
+ graceMs: typeof reply.graceMs === 'number' ? reply.graceMs : 0,
1343
+ rounds: typeof reply.rounds === 'number' ? reply.rounds : 0,
1344
+ }
1345
+ }
1346
+
1347
+ // --- flush (#484) ---------------------------------------------------------
1348
+
1349
+ /**
1350
+ * Thrown when a flush was asked of a guest that cannot perform one: either
1351
+ * its `healthz` does not advertise {@link FLUSH_FEATURE}, or it answered
1352
+ * `unknown_op: flush`.
1353
+ *
1354
+ * Refused rather than treated as "the disk is already flushed", for the
1355
+ * same reason {@link KubernetesQuiesceUnsupportedError} is not treated as
1356
+ * "nothing was running". The one caller that does NOT pass this on is
1357
+ * `suspend()`, which goes ahead and tells the host through
1358
+ * `onFlushUnsupported`: an image built before this op is a deployment that
1359
+ * has to be able to suspend its workspaces, not one that has to be stopped.
1360
+ */
1361
+ export class KubernetesFlushUnsupportedError extends Error {
1362
+ override readonly name = 'KubernetesFlushUnsupportedError'
1363
+
1364
+ constructor(
1365
+ readonly feature: string,
1366
+ message: string,
1367
+ ) {
1368
+ super(message)
1369
+ }
1370
+ }
1371
+
1372
+ /**
1373
+ * Thrown when nothing can promise the workspace's writes are on the device.
1374
+ *
1375
+ * The guest ANSWERED and said so — the `syncfs` failed, or ran past its own
1376
+ * timeout — and its message says which. Its own class for the reason every
1377
+ * other answered refusal here has one: this is a fact about the disk, not a
1378
+ * transport failure, and a caller holding it decides whether to retry,
1379
+ * suspend anyway, or leave the workspace running.
1380
+ */
1381
+ export class KubernetesFlushUnconfirmedError extends Error {
1382
+ override readonly name = 'KubernetesFlushUnconfirmedError'
1383
+
1384
+ constructor(
1385
+ readonly reason: string,
1386
+ message: string,
1387
+ ) {
1388
+ super(message)
1389
+ }
1390
+ }
1391
+
1392
+ /**
1393
+ * Carried to {@link KubernetesWorkspaceOptions.onFlushUnreachable} when a
1394
+ * `suspend()` could not ASK for a flush at all — the dial failed, the
1395
+ * connection timed out, the token was refused, or the agent has fenced
1396
+ * itself — and the suspend went ahead without one.
1397
+ *
1398
+ * Never thrown at a caller. The difference between this and
1399
+ * {@link KubernetesFlushUnconfirmedError} is the difference between a guest
1400
+ * that could not be reached and a guest that answered: a guest that
1401
+ * answered is alive, and stopping the suspend gives its caller something to
1402
+ * do about it, while a guest nothing can reach will not become flushable by
1403
+ * leaving its pod running — and refusing to suspend over it would take away
1404
+ * the one verb an operator reaches for when a workspace is wedged, the verb
1405
+ * {@link KubernetesAgentRetiringError} itself names as the way out.
1406
+ *
1407
+ * So the suspend proceeds, the host is told, and the message says what the
1408
+ * disk is resting on instead: whatever the guest kernel had already written
1409
+ * back, plus the pod's own `preStop` hook and the agent's `SIGTERM` handler
1410
+ * if either of them still runs.
1411
+ */
1412
+ export class KubernetesFlushUnreachableError extends Error {
1413
+ override readonly name = 'KubernetesFlushUnreachableError'
1414
+ }
1415
+
1416
+ /** What a flush did, as the guest measured it. */
1417
+ export interface KubernetesFlushReport {
1418
+ /** How long the `syncfs` took inside the guest. */
1419
+ readonly durationMs: number
1420
+ /** The mount the guest flushed — its workspace root. */
1421
+ readonly workspace: string
1422
+ }
1423
+
1424
+ /**
1425
+ * The refusal a guest that cannot flush earns, built in one place.
1426
+ *
1427
+ * Exported because the WORKSPACE has to be able to hand this exact error to
1428
+ * `onFlushUnsupported` on the one path that does not throw it — a
1429
+ * `suspend()` against an older image — and a second copy of the message
1430
+ * would be a second thing to keep true.
1431
+ */
1432
+ export function flushUnsupportedError(): KubernetesFlushUnsupportedError {
1433
+ return new KubernetesFlushUnsupportedError(
1434
+ FLUSH_FEATURE,
1435
+ `kubernetes: this workspace's guest agent does not advertise the '${FLUSH_FEATURE}' healthz feature, so nothing here can make its writes reach the disk before the pod stops — what survives is whatever the guest kernel had already written back. The request is refused rather than answered as a flush that happened. Rebuild the workspace image from this Namzu release.`,
1436
+ )
1437
+ }
1438
+
1439
+ function flushReport(reply: Record<string, unknown>): KubernetesFlushReport {
1440
+ return {
1441
+ durationMs: typeof reply.durationMs === 'number' ? reply.durationMs : 0,
1442
+ workspace: typeof reply.workspace === 'string' ? reply.workspace : '',
1443
+ }
1444
+ }
1445
+
1446
+ export class KubernetesAgentTransport {
1447
+ /**
1448
+ * Mutable: the handle follows a replaced pod — see {@link rebind}.
1449
+ *
1450
+ * Under `'pod-ip'` both the address and the token move, because the
1451
+ * address is a literal that died with its pod. Under the default
1452
+ * `'service'` mode the host is a Service FQDN that outlives the pod and
1453
+ * resolves to the replacement on its own, so what moves is the TOKEN
1454
+ * alone — which is the entire reason a `service` handle needed a rebind
1455
+ * at all: the dial keeps working and the guest refuses every call.
1456
+ */
1457
+ private handle: KubernetesAgentHandle
1458
+ /** The re-read currently in flight, so concurrent refusals share one. */
1459
+ private rebinding?: Promise<boolean>
1460
+ /**
1461
+ * How many times this transport has moved to a different pod. Captured
1462
+ * before every attempt and compared after it fails, so a call refused by
1463
+ * the pod it was bound to — arriving after ANOTHER call's rebind already
1464
+ * installed the replacement — retries on the handle that has moved
1465
+ * instead of re-reading to be told nothing changed.
1466
+ */
1467
+ private rebindSeq = 0
1468
+ private readonly transportOptions: VsockTransportOptions
1469
+ /** Bound to each wire's own pod by {@link wireOptions}. */
1470
+ private readonly onGuestReply?: (reply: GuestReplyIdentity, podUid: string | undefined) => void
1471
+ private readonly onTiming?: (timing: KubernetesTransportTiming) => void
1472
+ private readonly refreshHandle?: (signal?: AbortSignal) => Promise<KubernetesAgentHandle>
1473
+ /** Simple pass-through operations share one transport instance. */
1474
+ private wire: VsockAgentTransport
1475
+
1476
+ constructor(handle: KubernetesAgentHandle, options: KubernetesTransportOptions = {}) {
1477
+ const { onTiming, refreshHandle, onGuestReply, ...transportOptions } = options
1478
+ this.handle = handle
1479
+ // Held apart from the options the wires are built from: it is the one
1480
+ // hook whose payload depends on WHICH wire read the reply, so
1481
+ // {@link wireOptions} binds it per wire rather than spreading it.
1482
+ this.onGuestReply = onGuestReply
1483
+ this.transportOptions = {
1484
+ ...transportOptions,
1485
+ // A NAME the resolver says does not EXIST is the one connect
1486
+ // failure waiting cannot cure, and the reason it must not be
1487
+ // waited on is that the wait destroys the diagnosis. A resolver
1488
+ // that merely could not answer (`EAI_AGAIN`) is not that — see
1489
+ // {@link isMissingNameFailure} — and keeps the whole budget.
1490
+ //
1491
+ // The dial's retry budget is 30 s and the privilege probe's
1492
+ // deadline is at most 15 s, so a host with no cluster resolver
1493
+ // never reaches the wrapper below: the probe's clock expires first
1494
+ // and `create()` rejects saying the guest "accepted the connection
1495
+ // and did not answer", about a connection that was never made. The
1496
+ // address here comes off a Sandbox that reported Ready, so its
1497
+ // Service exists and its record is published — a name that fails
1498
+ // to resolve against that is a fact about THIS host, not a race
1499
+ // with the controller.
1500
+ permanentDialFailure: (err) => this.dialFailureIsPermanent(err),
1501
+ }
1502
+ this.onTiming = onTiming
1503
+ this.refreshHandle = refreshHandle
1504
+ this.wire = new VsockAgentTransport(handle, this.wireOptions(handle))
1505
+ }
1506
+
1507
+ /**
1508
+ * The shared transport's options for ONE wire, with the reply observer
1509
+ * bound to the pod that wire is talking to.
1510
+ *
1511
+ * Every `VsockAgentTransport` this class builds goes through here — the
1512
+ * first one, the one a rebind installs, and the per-attempt one `exec()`
1513
+ * times — because a wire outlives the moment it was current: the pod it
1514
+ * dials can answer a call after a rebind has already moved
1515
+ * {@link handle} on, and the observer has to be told the pod that
1516
+ * ANSWERED rather than the pod this transport now holds. Capturing the
1517
+ * handle in the closure is what makes the two different values.
1518
+ */
1519
+ private wireOptions(handle: KubernetesAgentHandle): VsockTransportOptions {
1520
+ const observe = this.onGuestReply
1521
+ if (observe === undefined) return this.transportOptions
1522
+ return {
1523
+ ...this.transportOptions,
1524
+ onGuestReply: (reply) => {
1525
+ observe(reply, handle.token)
271
1526
  },
272
1527
  }
1528
+ }
1529
+
1530
+ /**
1531
+ * The address this transport is dialing right now. Diagnostics only —
1532
+ * host and port, and deliberately NOT the handle itself: this is read
1533
+ * into a log line or an assertion, and the bind token has no business
1534
+ * travelling with either.
1535
+ */
1536
+ get address(): { readonly host: string; readonly port: number } {
1537
+ return { host: this.handle.host, port: this.handle.port }
1538
+ }
1539
+
1540
+ /**
1541
+ * Run one operation, and give the handle exactly one chance to follow a
1542
+ * pod that was replaced underneath it.
1543
+ *
1544
+ * TWO failures reach a rebind, and they are the only two, because they
1545
+ * are the only two that prove the operation ran NOTHING in the guest:
1546
+ *
1547
+ * - **the dial failed.** No socket was ever established, so no byte was
1548
+ * sent. This is the trigger the `'pod-ip'` mode was built on: the
1549
+ * address is a literal that dies with its pod.
1550
+ * - **the guest REFUSED the token** (`unauthorized`, in either of the
1551
+ * shapes {@link isUnauthorizedRefusal} unifies). The agent checks the
1552
+ * token before `dispatch`, so a refused request reached the guest and
1553
+ * was thrown away unread. This is the trigger that matters under the
1554
+ * DEFAULT `'service'` mode, where the Service FQDN outlives the pod
1555
+ * and keeps resolving: the dial SUCCEEDS against the replacement and
1556
+ * the refusal is the only thing that says the pod moved.
1557
+ *
1558
+ * Everything else is the guest's answer to a request it did receive, and
1559
+ * nothing here may reinterpret it — not as a resolver problem, and not as
1560
+ * a reason to repeat work the guest has already begun.
1561
+ *
1562
+ * From there both arms run the SAME routine, which is the whole of this
1563
+ * change to it: {@link rebind} reads the pod once, and a DIFFERENT uid
1564
+ * means the controller replaced it (a resume, an eviction, a node drain),
1565
+ * so the handle takes the new address AND the new token together and the
1566
+ * operation is retried once.
1567
+ *
1568
+ * One branch belongs to the dial arm alone, and is taken before the
1569
+ * re-read: the handle's host is a NAME and the dial gave up at
1570
+ * resolution, so this is a Service FQDN and this host has no resolver for
1571
+ * it. Re-reading the pod would change nothing — the next dial would ask
1572
+ * the same resolver the same question — so the error is replaced with one
1573
+ * that names the FQDN and the configuration field that fixes it. It is
1574
+ * never asked of a refusal, which came back over a connection that
1575
+ * plainly worked, and the name check is not decoration either: a
1576
+ * `pod-ip` handle must keep its one re-read however an unrelated error
1577
+ * happens to be worded.
1578
+ *
1579
+ * Anything else — a re-read that finds the SAME pod, or one that fails —
1580
+ * leaves the original error standing. A pod that is still there and still
1581
+ * refusing is a guest problem, and replacing that error with a second,
1582
+ * later one would hide it. The one exception is a re-read that comes back
1583
+ * with a verdict about the OBJECT rather than a failed diagnosis
1584
+ * ({@link KubernetesWorkspaceReplacedError}): the disk behind the name is
1585
+ * not this handle's disk, which outranks whatever uncovered it.
1586
+ *
1587
+ * `signal` is the caller's own, and it decides one thing only: a call that
1588
+ * has ALREADY been cancelled is not owed a re-read, because the retry it
1589
+ * would buy would abort before it left. It is never handed to the re-read
1590
+ * itself, which is shared and runs under a bound of its own — see
1591
+ * {@link rebind}.
1592
+ *
1593
+ * `dials` is how `exec()` answers the first question at all. Its failure
1594
+ * can arrive as a bare timeout from the execution controller's own bound,
1595
+ * with the dial's error discarded rather than wrapped, so that path
1596
+ * watches its dials instead of reading its error — see {@link DialWatch}.
1597
+ * Every other operation hands back the dial's own error and passes none.
1598
+ *
1599
+ * A refusal that no rebind could fix leaves as {@link
1600
+ * KubernetesAgentUnauthorizedError} whichever shape it arrived in — the
1601
+ * unification the whole recovery hangs off, applied at the one point
1602
+ * where "nothing else can be done about it" is known.
1603
+ */
1604
+ private async withRebind<T>(
1605
+ run: () => Promise<T>,
1606
+ signal?: AbortSignal,
1607
+ dials?: DialWatch,
1608
+ ): Promise<T> {
1609
+ // Read BEFORE the attempt: everything below asks whether the handle
1610
+ // has moved SINCE this call was dispatched.
1611
+ const boundAt = this.rebindSeq
1612
+ try {
1613
+ return await run()
1614
+ } catch (err) {
1615
+ if (isUnretryableOutcome(err)) throw err
1616
+ const refused = isUnauthorizedRefusal(err)
1617
+ // A refusal came back over a connection that was made, so it is
1618
+ // never also a dial failure; asking both questions of one error
1619
+ // and letting the dial arm win would send it to the resolver
1620
+ // branch, which is about a connection that was never attempted.
1621
+ const fromTheDial = !refused && (isConnectFailure(err) || neverConnected(dials))
1622
+ if (!refused && !fromTheDial) throw err
1623
+ if (fromTheDial && this.dialsAName() && this.failedToResolve(err, dials)) {
1624
+ throw this.unresolvable(err)
1625
+ }
1626
+ // See above: nothing to buy with a re-read here, and the shared
1627
+ // one must not be started on behalf of a call that is gone.
1628
+ if (signal?.aborted === true) throw refused ? asUnauthorized(err) : err
1629
+ if (!(await this.rebind(boundAt))) throw refused ? asUnauthorized(err) : err
1630
+ try {
1631
+ return await run()
1632
+ } catch (retryErr) {
1633
+ // The retry is the last attempt either way; a second refusal
1634
+ // (the replacement is refusing too) still leaves as one shape.
1635
+ throw refused && isUnauthorizedRefusal(retryErr) ? asUnauthorized(retryErr) : retryErr
1636
+ }
1637
+ }
1638
+ }
1639
+
1640
+ /**
1641
+ * Whether the dial gave up at name resolution.
1642
+ *
1643
+ * Asked of the caller's error first, and of the watched dial's own error
1644
+ * only when nothing ever connected — the case where the error the caller
1645
+ * holds is a bound's timer rather than the failure that caused it. A
1646
+ * watched attempt that DID connect is never consulted: its error is the
1647
+ * guest's answer, however it happens to be worded.
1648
+ */
1649
+ private failedToResolve(error: unknown, dials: DialWatch | undefined): boolean {
1650
+ if (isNameResolutionFailure(error)) return true
1651
+ return neverConnected(dials) && isNameResolutionFailure(dials?.lastError)
1652
+ }
1653
+
1654
+ /**
1655
+ * The one connect failure waiting cannot cure — see the constructor.
1656
+ * A method rather than a closure because the watched `exec()` dials wrap
1657
+ * it, and a caller-visible answer must not depend on which wire asked.
1658
+ */
1659
+ private dialFailureIsPermanent(error: unknown): boolean {
1660
+ return this.dialsAName() && isMissingNameFailure(error)
1661
+ }
1662
+
1663
+ /**
1664
+ * One re-read, shared by every call that needs one at the same moment.
1665
+ * True when the handle now points at a DIFFERENT pod than the one the
1666
+ * caller's attempt was dispatched on — whether this call's own re-read
1667
+ * moved it or another call's already had.
1668
+ *
1669
+ * Single-flight, and that is not an optimisation. A pod replaced
1670
+ * underneath a busy handle refuses EVERY call in flight at once, and one
1671
+ * re-read per refused call would be a burst of Sandbox and pod GETs, each
1672
+ * one racing the others to install a handle — with the last to finish
1673
+ * winning, which is not necessarily the last to read. Sharing the promise
1674
+ * makes the burst one round trip and one installation, in the order the
1675
+ * API answered.
1676
+ *
1677
+ * The slot is cleared inside the shared run, before the promise settles,
1678
+ * so a caller that awaits it and is refused AGAIN gets a fresh re-read
1679
+ * rather than the answer to the previous question.
1680
+ *
1681
+ * Because it is shared it runs under {@link REBIND_READ_TIMEOUT_MS} and
1682
+ * under NO caller's signal. A caller that aborts while the shared read is
1683
+ * in flight would otherwise abort it for everyone, and every call the
1684
+ * replacement refused would fail with its original refusal although the
1685
+ * handle could have followed the pod. Its own bound is what the aborting
1686
+ * caller was owed — nobody waits on a re-read for longer than that — and
1687
+ * the read is two GETs nobody is billed for twice.
1688
+ */
1689
+ private async rebind(boundAt?: number): Promise<boolean> {
1690
+ // Another call already followed the replacement while this one was in
1691
+ // flight, so the pod that refused this call is the pod it was bound
1692
+ // to and the answer a re-read would give is already installed. Asking
1693
+ // again would compare the new token against itself, conclude nothing
1694
+ // moved, and fail a call the handle can now serve.
1695
+ if (boundAt !== undefined && this.rebindSeq !== boundAt) return true
1696
+ const refresh = this.refreshHandle
1697
+ if (refresh === undefined) return false
1698
+ this.rebinding ??= this.runRebind(refresh, AbortSignal.timeout(REBIND_READ_TIMEOUT_MS))
1699
+ return await this.rebinding
1700
+ }
273
1701
 
274
- const controller = new RemoteExecutionController(adapter)
1702
+ private async runRebind(
1703
+ refresh: (signal?: AbortSignal) => Promise<KubernetesAgentHandle>,
1704
+ signal?: AbortSignal,
1705
+ ): Promise<boolean> {
275
1706
  try {
276
- return await controller.exec(command, argv, opts)
1707
+ let next: KubernetesAgentHandle
1708
+ try {
1709
+ next = await refresh(signal)
1710
+ } catch (err) {
1711
+ // A verdict about the OBJECT is not a failed diagnosis and
1712
+ // must not be swallowed: it says the disk behind the name is
1713
+ // not this handle's disk, which is a far more important thing
1714
+ // to report than the refusal that uncovered it — and it is the
1715
+ // one answer that must never be followed by a retry.
1716
+ if (err instanceof KubernetesWorkspaceReplacedError) throw err
1717
+ // Everything else: the re-read is a diagnosis, not an
1718
+ // operation. A pod that cannot be read is not a better error
1719
+ // than the failure the caller is already holding.
1720
+ return false
1721
+ }
1722
+ if (next.token === this.handle.token) return false
1723
+ this.rebindSeq += 1
1724
+ this.handle = next
1725
+ this.wire = new VsockAgentTransport(next, this.wireOptions(next))
1726
+ return true
1727
+ } finally {
1728
+ this.rebinding = undefined
1729
+ }
1730
+ }
1731
+
1732
+ /** Whether the current handle's host goes through a resolver at all. */
1733
+ private dialsAName(): boolean {
1734
+ return net.isIP(this.handle.host) === 0
1735
+ }
1736
+
1737
+ /** The DNS-shaped failure, in words that name the way out of it. */
1738
+ private unresolvable(cause: unknown): Error {
1739
+ const host = this.handle.host
1740
+ return new KubernetesAgentAddressUnresolvableError(
1741
+ host,
1742
+ `kubernetes: the guest agent's address ${host}:${this.handle.port} did not resolve (ENOTFOUND/EAI_AGAIN), so no connection was attempted. That is a Kubernetes Service FQDN and only the cluster's own DNS answers it: a host running OUTSIDE the cluster — a VNet peer, a CI runner, a laptop — fails every call here, readiness probes included, and the symptom looks like a sandbox that never came up. Set agentAddress: 'pod-ip' on the kubernetes backend config to dial the bound pod's IP instead, which needs a pod network routable from this host and a NetworkPolicy admitting its address range.`,
1743
+ { cause },
1744
+ )
1745
+ }
1746
+
1747
+ /**
1748
+ * Readiness probe — never requires a token; see `protocol.ts`.
1749
+ *
1750
+ * Deliberately NOT wrapped in {@link withRebind}: `healthz` answers a
1751
+ * failed dial with `false` rather than by throwing, so there is no error
1752
+ * to classify and nothing for a re-read to be triggered by. A caller that
1753
+ * wants the reason asks for it by making a real call.
1754
+ */
1755
+ async healthz(signal?: AbortSignal): Promise<boolean> {
1756
+ return await this.wire.healthz(signal)
1757
+ }
1758
+
1759
+ /**
1760
+ * Poll until `healthz` succeeds or the timeout elapses. Unwrapped for the
1761
+ * same reason, and for one more: it already owns a retry loop, so a
1762
+ * connect failure here is not a single failed dial but a whole budget of
1763
+ * them.
1764
+ */
1765
+ async waitForReady(
1766
+ timeoutMs: number,
1767
+ pollIntervalMs: number,
1768
+ signal?: AbortSignal,
1769
+ ): Promise<void> {
1770
+ return await this.wire.waitForReady(timeoutMs, pollIntervalMs, signal)
1771
+ }
1772
+
1773
+ /**
1774
+ * Ask the agent how it is, and keep the two "not ok" answers apart —
1775
+ * see {@link KubernetesAgentHealth}.
1776
+ *
1777
+ * Deliberately NOT wrapped in {@link withRebind}: this is a diagnostic
1778
+ * about the pod this handle is bound to RIGHT NOW, and a rebind would
1779
+ * silently answer it about a different pod. A caller that wants to know
1780
+ * whether the agent it was talking to has fenced itself would then be
1781
+ * told about the replacement, which is a different question with a
1782
+ * different answer.
1783
+ *
1784
+ * It throws whatever the dial or the read threw. An agent that cannot be
1785
+ * reached has no health to report, and inventing one here would turn
1786
+ * "unreachable" into "fine".
1787
+ */
1788
+ async agentHealth(signal?: AbortSignal): Promise<KubernetesAgentHealth> {
1789
+ const reply = await this.wire.request<{ ok?: unknown; retiring?: unknown }>(
1790
+ { op: 'healthz' },
1791
+ signal,
1792
+ )
1793
+ return { ok: reply?.ok === true, retiring: reply?.retiring === true }
1794
+ }
1795
+
1796
+ /**
1797
+ * The raw `reserve-execution` primitive, exposed directly (rather than
1798
+ * only reachable as a side effect of `exec()`) so the reservation
1799
+ * round trip is independently observable and testable against the
1800
+ * real guest.
1801
+ */
1802
+ async reserve(signal?: AbortSignal): Promise<unknown> {
1803
+ return await this.withRebind(
1804
+ async () => await requestChecked(this.wire, { op: 'reserve-execution' }, signal),
1805
+ signal,
1806
+ )
1807
+ }
1808
+
1809
+ /**
1810
+ * The raw `cancel-execution` primitive, exposed for the same reason
1811
+ * {@link reserve} is: it is a control request with its own refusal
1812
+ * semantics, and proving those against the real guest should not require
1813
+ * driving a whole cancelled `exec()` to reach it.
1814
+ */
1815
+ async cancel(executionId: string, signal?: AbortSignal): Promise<unknown> {
1816
+ return await this.withRebind(
1817
+ async () =>
1818
+ await requestChecked(this.wire, { op: 'cancel-execution', body: { executionId } }, signal),
1819
+ signal,
1820
+ )
1821
+ }
1822
+
1823
+ /**
1824
+ * Delegated, `signal` included: a body larger than one pre-auth frame
1825
+ * is written as a sequence of parts by {@link VsockAgentTransport}
1826
+ * itself, and a cancelled sequence has to be able to stop mid-way and
1827
+ * take its temp file with it.
1828
+ *
1829
+ * One transport instance, deliberately — {@link wire} is shared by
1830
+ * every simple pass-through op — so the one `healthz` probe that asks
1831
+ * the guest whether it can take parts is asked once for this sandbox,
1832
+ * not once per large write.
1833
+ *
1834
+ * Wrapped in {@link withRebind} like every other guest-dialling op,
1835
+ * the multi-part route included, and for both of its triggers: a retry
1836
+ * starts a fresh sequence under a new temp name
1837
+ * (`writeFilePartTempPath` mints a UUID per call), so a sequence
1838
+ * abandoned mid-way on the replaced pod — because the dial failed, or
1839
+ * because the replacement refused this handle's token before reading a
1840
+ * byte — cannot collide with the retry's offsets and never touched the
1841
+ * target. At worst it leaves one orphan temp file behind on the
1842
+ * workspace volume.
1843
+ */
1844
+ async writeFile(path: string, content: Buffer, signal?: AbortSignal): Promise<void> {
1845
+ return await this.withRebind(
1846
+ async () => await this.wire.writeFile(path, content, signal),
1847
+ signal,
1848
+ )
1849
+ }
1850
+
1851
+ async readFile(path: string, options?: SandboxReadFileOptions): Promise<Buffer> {
1852
+ return await this.withRebind(
1853
+ async () => await this.wire.readFile(path, options),
1854
+ options?.signal,
1855
+ )
1856
+ }
1857
+
1858
+ /**
1859
+ * Delegated, and rebound exactly once — but only around the FIRST
1860
+ * chunk.
1861
+ *
1862
+ * That is the whole of what {@link withRebind} can honestly cover here.
1863
+ * Its retry is safe because nothing reached the guest, and once a chunk
1864
+ * has been yielded that is no longer true: re-dialing a replaced pod
1865
+ * mid-stream would restart the file from its beginning, and the
1866
+ * consumer — which has already taken the bytes and cannot give them
1867
+ * back — would silently concatenate a duplicate prefix. So the first
1868
+ * pull carries the dial, the rebind and the retry; everything after it
1869
+ * fails as itself.
1870
+ */
1871
+ async *readFileStream(
1872
+ path: string,
1873
+ options?: SandboxReadFileOptions,
1874
+ ): AsyncGenerator<Buffer, void, undefined> {
1875
+ const started = await this.withRebind(async () => {
1876
+ const iterator = this.wire.readFileStream(path, options)[Symbol.asyncIterator]()
1877
+ try {
1878
+ return { iterator, first: await iterator.next() }
1879
+ } catch (error) {
1880
+ // The abandoned generator's own `finally` has already run by
1881
+ // the time its `next()` rejects, so the socket is down; the
1882
+ // `return()` is belt and braces for an implementation that
1883
+ // rejected without finishing.
1884
+ await iterator.return?.(undefined).catch(() => undefined)
1885
+ throw error
1886
+ }
1887
+ }, options?.signal)
1888
+ const { iterator, first } = started
1889
+ try {
1890
+ if (first.done === true) return
1891
+ yield first.value
1892
+ for (;;) {
1893
+ const next = await iterator.next()
1894
+ if (next.done === true) return
1895
+ yield next.value
1896
+ }
1897
+ } finally {
1898
+ await iterator.return?.(undefined).catch(() => undefined)
1899
+ }
1900
+ }
1901
+
1902
+ /**
1903
+ * A guest PTY. Without `sessionId`/`persistent` this is exactly the
1904
+ * terminal it has always been, down to the wire request.
1905
+ *
1906
+ * With them the PTY belongs to the guest's session registry: losing this
1907
+ * connection detaches rather than killing, a later process rejoins it
1908
+ * with {@link attachSession}, and the capability is verified against the
1909
+ * guest's `healthz` features BEFORE the shell is started — never
1910
+ * downgraded to a connection-bound terminal, which would look like it
1911
+ * worked until the rollout it exists for.
1912
+ */
1913
+ async openTerminal(options: KubernetesOpenTerminalOptions): Promise<KubernetesWorkspaceTerminal> {
1914
+ if (options.sessionId === undefined && options.persistent !== true) {
1915
+ return await this.withRebind(async () => await this.wire.openTerminal(options))
1916
+ }
1917
+ const sessionId = assertSessionOpen(options)
1918
+ await this.assertSessionsSupported()
1919
+ return await this.sessionStream(
1920
+ sessionId,
1921
+ 'openTerminal',
1922
+ async () => await this.wire.openSessionTerminal(options),
1923
+ )
1924
+ }
1925
+
1926
+ /**
1927
+ * One open of a session stream, with the guest's refusal mapped to
1928
+ * {@link KubernetesSessionRefusedError} — see {@link sessionStreamFailure}.
1929
+ */
1930
+ private async sessionStream(
1931
+ sessionId: string,
1932
+ operation: string,
1933
+ open: () => Promise<AgentTerminalStream>,
1934
+ ): Promise<KubernetesSessionTerminal> {
1935
+ try {
1936
+ return sessionTerminal(await this.withRebind(open), sessionId)
1937
+ } catch (error) {
1938
+ throw sessionStreamFailure(sessionId, operation, error)
1939
+ }
1940
+ }
1941
+
1942
+ /**
1943
+ * Rejoin a terminal session, replaying from `fromOffset` and then
1944
+ * following it live.
1945
+ *
1946
+ * The guest allows one attachment per session and ends the previous one
1947
+ * by name, so two host processes cannot interleave keystrokes into one
1948
+ * shell. Losing this connection detaches; ending the program is
1949
+ * {@link killSession} and nothing else.
1950
+ */
1951
+ async attachSession(
1952
+ sessionId: string,
1953
+ options: KubernetesAttachTerminalOptions = {},
1954
+ ): Promise<KubernetesSessionTerminal> {
1955
+ assertSessionId(sessionId)
1956
+ await this.assertSessionsSupported()
1957
+ return await this.sessionStream(
1958
+ sessionId,
1959
+ 'attachSession',
1960
+ async () =>
1961
+ await this.wire.attachSessionTerminal({
1962
+ sessionId,
1963
+ ...(options.fromOffset !== undefined ? { fromOffset: options.fromOffset } : {}),
1964
+ ...(options.size !== undefined
1965
+ ? { cols: options.size.cols, rows: options.size.rows }
1966
+ : {}),
1967
+ }),
1968
+ )
1969
+ }
1970
+
1971
+ /**
1972
+ * Start a program with no terminal at all, in its own kernel session,
1973
+ * with stdin closed and both output streams going into the guest's
1974
+ * retained log.
1975
+ *
1976
+ * It is not the SDK's `spawnDetached` and deliberately does not pretend
1977
+ * to be: that one hands back a host `ChildProcess`, which cannot cross a
1978
+ * process boundary. This returns a NAME, and the name is what a
1979
+ * redeployed host comes back with.
1980
+ */
1981
+ async startDetached(
1982
+ options: KubernetesStartDetachedOptions,
1983
+ signal?: AbortSignal,
1984
+ ): Promise<KubernetesSessionSummary> {
1985
+ assertSessionId(options.sessionId)
1986
+ await this.assertSessionsSupported(signal)
1987
+ const reply = (await this.withRebind(
1988
+ async () =>
1989
+ await requestChecked(
1990
+ this.wire,
1991
+ {
1992
+ op: 'start-detached',
1993
+ body: {
1994
+ sessionId: options.sessionId,
1995
+ command: options.command,
1996
+ ...(options.args !== undefined ? { args: options.args } : {}),
1997
+ ...(options.cwd !== undefined ? { cwd: options.cwd } : {}),
1998
+ ...(options.env !== undefined ? { env: { ...options.env } } : {}),
1999
+ },
2000
+ },
2001
+ signal,
2002
+ ),
2003
+ signal,
2004
+ )) as Record<string, unknown>
2005
+ if (reply.ok !== true) throw sessionRefusal(options.sessionId, reply, 'startDetached')
2006
+ return parseSessionSummary(reply)
2007
+ }
2008
+
2009
+ /** Every session this pod's agent is holding, running and recently exited. */
2010
+ async listSessions(signal?: AbortSignal): Promise<readonly KubernetesSessionSummary[]> {
2011
+ await this.assertSessionsSupported(signal)
2012
+ const reply = (await this.withRebind(
2013
+ async () => await requestChecked(this.wire, { op: 'list-sessions' }, signal),
2014
+ signal,
2015
+ )) as Record<string, unknown>
2016
+ if (reply.ok !== true) {
2017
+ throw new RemoteProtocolError(
2018
+ `kubernetes: the guest refused to list sessions: ${String(reply.error ?? 'no reason given')}`,
2019
+ )
2020
+ }
2021
+ if (!Array.isArray(reply.sessions)) {
2022
+ throw new RemoteProtocolError(
2023
+ 'kubernetes: the guest sent a session list that is not an array',
2024
+ )
2025
+ }
2026
+ return reply.sessions.map(parseSessionSummary)
2027
+ }
2028
+
2029
+ /**
2030
+ * End one session and everything still in it — the shell, its
2031
+ * backgrounded jobs, and the program a detached session started.
2032
+ *
2033
+ * Idempotent: a session that has already exited answers with what it
2034
+ * exited with. The reply carries the session's state, so a program that
2035
+ * ignored a `SIGTERM` and outlived the guest's confirm window is
2036
+ * reported still running rather than reported dead.
2037
+ */
2038
+ async killSession(
2039
+ sessionId: string,
2040
+ options: { readonly signal?: string; readonly abort?: AbortSignal } = {},
2041
+ ): Promise<KubernetesSessionSummary> {
2042
+ assertSessionId(sessionId)
2043
+ await this.assertSessionsSupported(options.abort)
2044
+ const reply = (await this.withRebind(
2045
+ async () =>
2046
+ await requestChecked(
2047
+ this.wire,
2048
+ {
2049
+ op: 'kill-session',
2050
+ body: {
2051
+ sessionId,
2052
+ ...(options.signal !== undefined ? { signal: options.signal } : {}),
2053
+ },
2054
+ },
2055
+ options.abort,
2056
+ ),
2057
+ options.abort,
2058
+ )) as Record<string, unknown>
2059
+ if (reply.ok !== true) throw sessionRefusal(sessionId, reply, 'killSession')
2060
+ return parseSessionSummary(reply)
2061
+ }
2062
+
2063
+ /**
2064
+ * Read what a session has printed since `fromOffset`, without attaching
2065
+ * to it and without signalling anything.
2066
+ *
2067
+ * One request, one answer, in the SDK's `BackgroundJobOutput` shape: the
2068
+ * chunk, the offset to come back with, the bytes the ring dropped before
2069
+ * it, and the program's status. A caller polling in a loop can neither
2070
+ * re-read nor skip, because the offset is the guest's own.
2071
+ */
2072
+ async readSession(
2073
+ sessionId: string,
2074
+ options: KubernetesReadSessionOptions = {},
2075
+ signal?: AbortSignal,
2076
+ ): Promise<KubernetesSessionOutput> {
2077
+ assertSessionId(sessionId)
2078
+ await this.assertSessionsSupported(signal)
2079
+ const fromOffset = options.fromOffset ?? 0
2080
+ let chunk = ''
2081
+ let nextOffset = fromOffset
2082
+ let droppedBytes = 0
2083
+ let state: SessionState = 'running'
2084
+ let exitCode: number | undefined
2085
+ let exitSignal: number | undefined
2086
+ let refusal: KubernetesSessionRefusedError | undefined
2087
+ // Accumulating INSIDE a `withRebind` is safe for the one reason that
2088
+ // matters: the wrapper retries only a failure that delivered no
2089
+ // session frame. A dial that failed sent nothing at all, and a token
2090
+ // the guest REFUSED is answered before `dispatch` with that refusal
2091
+ // as the connection's first and only frame — which the handler above
2092
+ // turns into an error rather than accumulating. Either way a retry
2093
+ // starts from an untouched chunk and the offset the caller asked
2094
+ // for.
2095
+ await this.withRebind(
2096
+ async () =>
2097
+ await this.wire.streamFramedRequest(
2098
+ { op: 'attach-session', body: { sessionId, fromOffset, follow: false } },
2099
+ (event) => {
2100
+ if (isUnauthorized(event)) throw new KubernetesAgentUnauthorizedError()
2101
+ if (event.type === 'ready') {
2102
+ nextOffset = sessionNumber(event.nextOffset, nextOffset)
2103
+ droppedBytes = sessionNumber(event.droppedBytes)
2104
+ state = event.state === 'exited' ? 'exited' : 'running'
2105
+ if (typeof event.exitCode === 'number') exitCode = event.exitCode
2106
+ if (typeof event.signal === 'number') exitSignal = event.signal
2107
+ return
2108
+ }
2109
+ if (event.type === 'data') {
2110
+ chunk += String(event.data ?? '')
2111
+ nextOffset = sessionNumber(event.nextOffset, nextOffset)
2112
+ return
2113
+ }
2114
+ if (event.type === 'error') {
2115
+ refusal = sessionRefusal(
2116
+ sessionId,
2117
+ {
2118
+ error: event.error,
2119
+ ...(event.message !== undefined ? { message: event.message } : {}),
2120
+ },
2121
+ 'readSession',
2122
+ )
2123
+ return
2124
+ }
2125
+ throw new RemoteProtocolError(
2126
+ `kubernetes: the guest sent an unexpected frame reading session ${sessionId}: ${JSON.stringify(event).slice(0, 200)}`,
2127
+ )
2128
+ },
2129
+ { observationTimeoutMs: SESSION_READ_TIMEOUT_MS },
2130
+ signal,
2131
+ ),
2132
+ signal,
2133
+ )
2134
+ if (refusal !== undefined) throw refusal
2135
+ return {
2136
+ chunk,
2137
+ nextOffset,
2138
+ droppedBytes,
2139
+ status: sessionStatus({ state, ...(exitSignal !== undefined ? { signal: exitSignal } : {}) }),
2140
+ ...(exitCode !== undefined ? { exitCode } : {}),
2141
+ }
2142
+ }
2143
+
2144
+ /**
2145
+ * Stop every process this pod's guest is running, and keep the agent.
2146
+ *
2147
+ * The point is the pair. Stopping the POD stops its processes too, but
2148
+ * leaves nothing to read the disk through, so a host that wants a capture
2149
+ * it can trust has to wake the workspace again and check. After this the
2150
+ * guest is quiet and still serving: `exec`, `readFile` and `writeFile`
2151
+ * all work, and what they see is a filesystem nobody is writing to.
2152
+ *
2153
+ * Open terminals and running commands end as a side effect and report it
2154
+ * through their own exit and result paths — a terminal receives its exit,
2155
+ * an `exec` resolves with a signal in its result. Rejecting with
2156
+ * {@link KubernetesQuiesceUnconfirmedError} is the honest failure: a
2157
+ * process would not stop, and the reply names its pid.
2158
+ */
2159
+ async quiesce(
2160
+ options: { readonly graceMs?: number } = {},
2161
+ signal?: AbortSignal,
2162
+ ): Promise<KubernetesQuiesceReport> {
2163
+ await this.assertQuiesceSupported(signal)
2164
+ const reply = (await this.withRebind(
2165
+ async () =>
2166
+ await requestChecked(
2167
+ this.wire,
2168
+ {
2169
+ op: 'quiesce',
2170
+ body: { ...(options.graceMs !== undefined ? { graceMs: options.graceMs } : {}) },
2171
+ },
2172
+ signal,
2173
+ ),
2174
+ signal,
2175
+ )) as Record<string, unknown>
2176
+ if (reply.ok === true) return quiesceReport(reply)
2177
+ const refusal = typeof reply.error === 'string' ? reply.error : 'no reason given'
2178
+ // The guest advertised the feature and then did not know the op. That
2179
+ // is one image, not two, so it is the unsupported error rather than a
2180
+ // second name for the same fact — see {@link assertQuiesceSupported}.
2181
+ if (refusal.startsWith('unknown_op')) throw this.quiesceUnsupported()
2182
+ const detail = typeof reply.message === 'string' ? `: ${reply.message}` : ''
2183
+ throw new KubernetesQuiesceUnconfirmedError(
2184
+ refusal,
2185
+ `kubernetes: the guest could not confirm that every process it is running has stopped (${refusal})${detail}. Nothing on the cluster was changed and the pod is still serving; a capture taken now may not be consistent.`,
2186
+ )
2187
+ }
2188
+
2189
+ /**
2190
+ * Put everything this workspace has written onto its device.
2191
+ *
2192
+ * The disk a workspace keeps is whatever the guest kernel happened to
2193
+ * write back. Nothing in this backend ever asked for more than that: a
2194
+ * `suspend()` patched the pod away and waited for it to stop, and a
2195
+ * stopped pod means only that nothing is writing any more — not that
2196
+ * what was written arrived. This is the call that closes the difference,
2197
+ * and it is `syncfs(2)` over the whole workspace mount, so it covers
2198
+ * what a COMMAND wrote as well as what `writeFile` did (`writeFile`
2199
+ * fsyncs its own bytes before it answers; a compiler's output is nobody's
2200
+ * to fsync).
2201
+ *
2202
+ * Rejecting with {@link KubernetesFlushUnconfirmedError} is the honest
2203
+ * failure, exactly as an unconfirmed quiesce is: the caller is usually
2204
+ * about to take the pod away, and "the flush did not run" has to be
2205
+ * distinguishable from "the flush ran".
2206
+ */
2207
+ async flush(
2208
+ options: { readonly timeoutMs?: number } = {},
2209
+ signal?: AbortSignal,
2210
+ ): Promise<KubernetesFlushReport> {
2211
+ await this.assertFlushSupported(signal)
2212
+ const reply = (await this.withRebind(
2213
+ async () =>
2214
+ await requestChecked(
2215
+ this.wire,
2216
+ {
2217
+ op: 'flush',
2218
+ body: { ...(options.timeoutMs !== undefined ? { timeoutMs: options.timeoutMs } : {}) },
2219
+ },
2220
+ signal,
2221
+ ),
2222
+ signal,
2223
+ )) as Record<string, unknown>
2224
+ if (reply.ok === true) return flushReport(reply)
2225
+ const refusal = typeof reply.error === 'string' ? reply.error : 'no reason given'
2226
+ // Advertised and then not known: one image, not two — the same
2227
+ // reading {@link quiesce} gives that answer. `flush_unsupported` and
2228
+ // `flush_unsupported_platform` join it because they say the same
2229
+ // thing in the guest's own words: this image cannot perform a flush,
2230
+ // now or ever. Read as an unconfirmed flush instead, they would stop
2231
+ // every default `suspend()` against such an image permanently, which
2232
+ // is exactly what the feature advertisement exists to avoid.
2233
+ if (refusal.startsWith('unknown_op') || refusal.startsWith('flush_unsupported')) {
2234
+ throw this.flushUnsupported()
2235
+ }
2236
+ const detail = typeof reply.message === 'string' ? `: ${reply.message}` : ''
2237
+ throw new KubernetesFlushUnconfirmedError(
2238
+ refusal,
2239
+ `kubernetes: the guest could not confirm that the workspace's writes reached its disk (${refusal})${detail}. Nothing on the cluster was changed and the pod is still serving; suspending or deleting the pod now may lose whatever had not been written back.`,
2240
+ )
2241
+ }
2242
+
2243
+ /**
2244
+ * Whether this guest can flush at all — asked, rather than assumed,
2245
+ * because a `suspend()` against an older image keeps today's behaviour
2246
+ * and reports the gap instead of refusing to suspend.
2247
+ */
2248
+ async supportsFlush(signal?: AbortSignal): Promise<boolean> {
2249
+ return (await this.wire.guestFeatures(signal)).includes(FLUSH_FEATURE)
2250
+ }
2251
+
2252
+ private async assertFlushSupported(signal?: AbortSignal): Promise<void> {
2253
+ if (await this.supportsFlush(signal)) return
2254
+ throw this.flushUnsupported()
2255
+ }
2256
+
2257
+ private flushUnsupported(): KubernetesFlushUnsupportedError {
2258
+ return flushUnsupportedError()
2259
+ }
2260
+
2261
+ /**
2262
+ * Whether this guest can quiesce at all — asked, rather than assumed,
2263
+ * because `suspend({ quiesce: true })` against an older image keeps
2264
+ * today's behaviour and reports the gap instead of refusing to suspend.
2265
+ */
2266
+ async supportsQuiesce(signal?: AbortSignal): Promise<boolean> {
2267
+ return (await this.wire.guestFeatures(signal)).includes(QUIESCE_FEATURE)
2268
+ }
2269
+
2270
+ private async assertQuiesceSupported(signal?: AbortSignal): Promise<void> {
2271
+ if (await this.supportsQuiesce(signal)) return
2272
+ throw this.quiesceUnsupported()
2273
+ }
2274
+
2275
+ private quiesceUnsupported(): KubernetesQuiesceUnsupportedError {
2276
+ return new KubernetesQuiesceUnsupportedError(
2277
+ QUIESCE_FEATURE,
2278
+ `kubernetes: this workspace's guest agent does not advertise the '${QUIESCE_FEATURE}' healthz feature, so nothing here can stop the processes it is running — a terminal another handle opened, a command already in flight, or a program that moved into a session of its own all keep writing. The request is refused rather than answered with an empty list, which would read as a guest that had nothing to stop. Rebuild the workspace image from this Namzu release.`,
2279
+ )
2280
+ }
2281
+
2282
+ /**
2283
+ * Whether this guest keeps a session registry at all, asked once per
2284
+ * transport and only when a caller wants one.
2285
+ */
2286
+ private async assertSessionsSupported(signal?: AbortSignal): Promise<void> {
2287
+ const features = await this.wire.guestFeatures(signal)
2288
+ if (features.includes(SESSIONS_FEATURE)) return
2289
+ throw new KubernetesSessionsUnsupportedError(
2290
+ SESSIONS_FEATURE,
2291
+ `kubernetes: this workspace's guest agent does not advertise the '${SESSIONS_FEATURE}' healthz feature, so a terminal opened here would die with this connection and a detached program could not be named, read or killed. The request is refused rather than served as a connection-bound terminal. Rebuild the workspace image from this Namzu release.`,
2292
+ )
2293
+ }
2294
+
2295
+ async openTcpConnection(options: SandboxTcpConnectOptions): Promise<SandboxTcpConnection> {
2296
+ return await this.withRebind(async () => await this.wire.openTcpConnection(options))
2297
+ }
2298
+
2299
+ /**
2300
+ * Run one command through a fresh, call-scoped adapter + controller.
2301
+ * Fresh per call — not shared instance state — because
2302
+ * {@link VsockAgentTransport} itself carries no cross-call connection
2303
+ * state (it dials fresh every time), so building one per `exec()` is
2304
+ * free and makes concurrent `exec()` calls on the same
2305
+ * `KubernetesAgentTransport` correctly independent: each gets its own
2306
+ * `onDial` closure and its own timing accumulator, with no shared
2307
+ * mutable field for two in-flight calls to race on.
2308
+ */
2309
+ async exec(
2310
+ command: string,
2311
+ argv?: string[],
2312
+ opts?: SandboxExecOptions,
2313
+ ): Promise<SandboxExecResult> {
2314
+ let dialMs = 0
2315
+ let reserveMs = 0
2316
+ let executeMs = 0
2317
+ let executeSettledAt = 0
2318
+ // The guest that ACCEPTED the reservation, which is the guest that
2319
+ // runs the command: its pod (the bind token this attempt presented)
2320
+ // and its process (the boot id the reply carried). Written per
2321
+ // ATTEMPT, so a retry that followed a replaced pod is remembered
2322
+ // against the pod it actually reserved on and not the one that
2323
+ // refused it — see {@link executionGuest}.
2324
+ let reservedIn: string | undefined
2325
+ let reservedOn: string | undefined
2326
+
2327
+ // What this attempt's dials did, for the one classification `exec()`
2328
+ // cannot make from its error: the controller bounds `reserve` at 2s
2329
+ // and hands back its own timer's Error, so a dial still inside its
2330
+ // connect-retry budget is reported as a reservation that took too
2331
+ // long, with the connect failure discarded. See {@link DialWatch}.
2332
+ const dials: DialWatch = { attempted: false, failed: false, connected: false }
2333
+
2334
+ // Built per ATTEMPT, from `this.handle` as it stands when the attempt
2335
+ // starts: a retry that follows a replaced pod has to dial the new
2336
+ // address, and the timing accumulators above outlive both attempts so
2337
+ // the caller still sees one call's total. The watch is reset here and
2338
+ // not there — it describes the attempt, and the second attempt is a
2339
+ // different pod's.
2340
+ const buildAttempt = (): Promise<SandboxExecResult> => {
2341
+ dials.attempted = false
2342
+ dials.failed = false
2343
+ dials.connected = false
2344
+ dials.lastError = undefined
2345
+ // The handle as it stands NOW, read once: the attempt dials this
2346
+ // address with this token, and another call's rebind must not
2347
+ // change what this attempt records it reserved on.
2348
+ const boundTo = this.handle
2349
+ const timedWire = new VsockAgentTransport(boundTo, {
2350
+ ...this.wireOptions(boundTo),
2351
+ // Fires before each connect, so an attempt the controller's
2352
+ // bound aborts mid-connect is still on the record — see
2353
+ // {@link DialWatch}.
2354
+ onDialAttempt: () => {
2355
+ dials.attempted = true
2356
+ },
2357
+ onDial: (ms) => {
2358
+ dials.connected = true
2359
+ dialMs += ms
2360
+ },
2361
+ // Consulted by the dial on every FAILED connect attempt, which
2362
+ // is where the error the bound swallows is kept. The answer
2363
+ // itself is the transport's own, unchanged.
2364
+ permanentDialFailure: (err) => {
2365
+ dials.failed = true
2366
+ dials.lastError = err
2367
+ return this.dialFailureIsPermanent(err)
2368
+ },
2369
+ })
2370
+
2371
+ const adapter: RemoteExecutionAdapter<Pick<ExecRequest, 'stdin' | 'maxOutputBytes'>> = {
2372
+ label: 'kubernetes pod-network agent',
2373
+ reserve: async (signal) => {
2374
+ const startedAt = Date.now()
2375
+ try {
2376
+ const reply = await requestChecked(timedWire, { op: 'reserve-execution' }, signal)
2377
+ // Overwritten, never merged: the LAST reservation is the
2378
+ // one the command ran under, and a replacement guest
2379
+ // that reports no boot id must leave no evidence
2380
+ // behind rather than inherit its predecessor's.
2381
+ reservedIn = guestBootIdOf(reply)
2382
+ reservedOn = boundTo.token
2383
+ return reply
2384
+ } finally {
2385
+ reserveMs += Date.now() - startedAt
2386
+ }
2387
+ },
2388
+ // Checked exactly like `reserve` — see `requestChecked`.
2389
+ cancel: async (executionId, signal) =>
2390
+ await requestChecked(
2391
+ timedWire,
2392
+ { op: 'cancel-execution', body: { executionId } },
2393
+ signal,
2394
+ ),
2395
+ execute: async (executionId, cmd, execArgv, execOpts, signal, context) => {
2396
+ const startedAt = Date.now()
2397
+ try {
2398
+ return await timedWire.executeStreamed(
2399
+ {
2400
+ ...(executionId ? { executionId } : {}),
2401
+ command: cmd,
2402
+ args: execArgv ?? [],
2403
+ ...(execOpts?.cwd !== undefined ? { cwd: execOpts.cwd } : {}),
2404
+ ...(execOpts?.env !== undefined ? { env: execOpts.env } : {}),
2405
+ ...(execOpts?.timeout !== undefined ? { timeoutMs: execOpts.timeout } : {}),
2406
+ ...(context?.stdin !== undefined ? { stdin: context.stdin } : {}),
2407
+ ...(context?.maxOutputBytes !== undefined
2408
+ ? { maxOutputBytes: context.maxOutputBytes }
2409
+ : {}),
2410
+ },
2411
+ execOpts,
2412
+ signal,
2413
+ )
2414
+ } finally {
2415
+ executeMs += Date.now() - startedAt
2416
+ executeSettledAt = Date.now()
2417
+ }
2418
+ },
2419
+ }
2420
+
2421
+ return new RemoteExecutionController(adapter).exec(command, argv, opts)
2422
+ }
2423
+
2424
+ try {
2425
+ return await this.withRebind(buildAttempt, opts?.signal, dials)
2426
+ } catch (error) {
2427
+ // Stamped here and nowhere else. The retirement hook one frame up
2428
+ // is handed the error ALONE, so this is the last point at which
2429
+ // "which execution is this" is still known — and the question the
2430
+ // hook has to answer is about this command's guest, not the
2431
+ // handle's.
2432
+ if (
2433
+ error instanceof RemoteCancellationUnknownError &&
2434
+ (reservedOn !== undefined || reservedIn !== undefined)
2435
+ ) {
2436
+ executionGuest.set(error, {
2437
+ ...(reservedOn !== undefined ? { podUid: reservedOn } : {}),
2438
+ ...(reservedIn !== undefined ? { guestBootId: reservedIn } : {}),
2439
+ })
2440
+ }
2441
+ throw error
277
2442
  } finally {
278
2443
  this.onTiming?.({
279
2444
  dialMs,
@@ -283,4 +2448,448 @@ export class KubernetesAgentTransport {
283
2448
  })
284
2449
  }
285
2450
  }
2451
+
2452
+ /**
2453
+ * Whether this guest implements the detach/attach ops at all, asked once
2454
+ * per transport and only when a caller wants them.
2455
+ */
2456
+ private async assertExecutionAttachSupported(signal?: AbortSignal): Promise<void> {
2457
+ const features = await this.wire.guestFeatures(signal)
2458
+ if (features.includes(EXECUTION_ATTACH_FEATURE)) return
2459
+ throw new KubernetesExecutionAttachUnsupportedError(
2460
+ EXECUTION_ATTACH_FEATURE,
2461
+ `kubernetes: this workspace's guest agent does not advertise the '${EXECUTION_ATTACH_FEATURE}' healthz feature, so a command started here would keep no output and could not be reattached to. The request is refused before the command is admitted rather than run as an ordinary exec. Rebuild the workspace image from this Namzu release.`,
2462
+ )
2463
+ }
2464
+
2465
+ /** `reserve-execution` for a caller-named id, with its reported state. */
2466
+ private async reserveDetached(
2467
+ executionId: string,
2468
+ signal?: AbortSignal,
2469
+ ): Promise<{ readonly state: string }> {
2470
+ const response = (await this.withRebind(
2471
+ async () =>
2472
+ await requestChecked(this.wire, { op: 'reserve-execution', body: { executionId } }, signal),
2473
+ signal,
2474
+ )) as Record<string, unknown>
2475
+ if (response.ok !== true || response.executionId !== executionId) {
2476
+ throw new RemoteProtocolError(
2477
+ `kubernetes: the guest refused the reservation for ${executionId}: ${String(response.error ?? 'no reason given')}`,
2478
+ )
2479
+ }
2480
+ return { state: typeof response.state === 'string' ? response.state : 'reserved' }
2481
+ }
2482
+
2483
+ /**
2484
+ * The `cancel-execution` control path, retried for its whole confirm
2485
+ * window and reported UNKNOWN rather than as a failure if none of the
2486
+ * attempts got an answer — the same rule the shared execution
2487
+ * controller applies, because a command whose termination nobody
2488
+ * confirmed is not a command anybody may call dead.
2489
+ *
2490
+ * Nothing on the detach path calls this to reconcile a lost connection.
2491
+ * It runs when the CALLER asked for it: `SandboxExecOptions.signal`
2492
+ * aborting, or {@link cancelExecution}.
2493
+ */
2494
+ private async confirmCancel(
2495
+ executionId: string,
2496
+ signal?: AbortSignal,
2497
+ ): Promise<RemoteCancellationAcknowledgement> {
2498
+ const deadlineAt = Date.now() + CANCEL_CONFIRM_WINDOW_MS
2499
+ let lastError: unknown = new Error('no cancellation attempt completed')
2500
+ while (Date.now() < deadlineAt) {
2501
+ const attempt = new AbortController()
2502
+ const onAbort = () => attempt.abort(signal?.reason)
2503
+ signal?.addEventListener('abort', onAbort, { once: true })
2504
+ const timer = setTimeout(
2505
+ () =>
2506
+ attempt.abort(new Error(`cancellation attempt exceeded ${CANCEL_ATTEMPT_TIMEOUT_MS}ms`)),
2507
+ Math.min(CANCEL_ATTEMPT_TIMEOUT_MS, Math.max(1, deadlineAt - Date.now())),
2508
+ )
2509
+ timer.unref?.()
2510
+ try {
2511
+ return parseCancellationReply(executionId, await this.cancel(executionId, attempt.signal))
2512
+ } catch (error) {
2513
+ if (error instanceof KubernetesExecutionNotAttachableError) throw error
2514
+ if (error instanceof KubernetesAgentUnauthorizedError) throw error
2515
+ lastError = error
2516
+ const remaining = deadlineAt - Date.now()
2517
+ if (remaining > 0) await pause(Math.min(50, remaining))
2518
+ } finally {
2519
+ clearTimeout(timer)
2520
+ signal?.removeEventListener('abort', onAbort)
2521
+ }
2522
+ }
2523
+ throw new RemoteCancellationUnknownError(
2524
+ `Remote sandbox cancellation could not be confirmed for ${executionId}: ${lastError instanceof Error ? lastError.message : String(lastError)}. The remote outcome is unknown; do not automatically retry the command.`,
2525
+ { cause: lastError },
2526
+ )
2527
+ }
2528
+
2529
+ /**
2530
+ * End a command by id, from any host process holding the id and the
2531
+ * bind token. Resolves only on a CONFIRMED termination.
2532
+ *
2533
+ * The outcome is not reported here — a cancelled execution's result and
2534
+ * whatever output it managed is read back through
2535
+ * {@link attachExecution}, which is the op that exists for reading.
2536
+ */
2537
+ async cancelExecution(executionId: string, signal?: AbortSignal): Promise<void> {
2538
+ assertExecutionId(executionId)
2539
+ await this.confirmCancel(executionId, signal)
2540
+ }
2541
+
2542
+ /**
2543
+ * Observe a command that is already the guest's, from `fromOffset` on,
2544
+ * and resolve with its result.
2545
+ *
2546
+ * It never signals the command. Aborting `signal` stops OBSERVING and
2547
+ * rejects with {@link KubernetesExecutionDetachedError}; it does not
2548
+ * cancel, and the command goes on running. Ending a command is
2549
+ * {@link cancelExecution} and nothing else.
2550
+ */
2551
+ async attachExecution(
2552
+ executionId: string,
2553
+ options: KubernetesAttachExecutionOptions = {},
2554
+ ): Promise<SandboxExecResult> {
2555
+ assertExecutionId(executionId)
2556
+ await this.assertExecutionAttachSupported(options.signal)
2557
+ const cursor: OutputCursor = {
2558
+ offset: options.fromOffset ?? 0,
2559
+ stdout: '',
2560
+ stderr: '',
2561
+ droppedBytes: 0,
2562
+ }
2563
+ const observation = new AbortController()
2564
+ const onDetach = () => observation.abort(options.signal?.reason)
2565
+ options.signal?.addEventListener('abort', onDetach, { once: true })
2566
+ if (options.signal?.aborted) onDetach()
2567
+ try {
2568
+ return await this.attachOnce(executionId, cursor, options, observation.signal)
2569
+ } catch (error) {
2570
+ if (options.signal?.aborted) throw this.detached(executionId, cursor, error)
2571
+ throw error
2572
+ } finally {
2573
+ options.signal?.removeEventListener('abort', onDetach)
2574
+ observation.abort(new Error('kubernetes: attach observation finished'))
2575
+ }
2576
+ }
2577
+
2578
+ /**
2579
+ * Run one command whose observation can outlive this connection, and —
2580
+ * when the connection is what failed — get it back rather than killing
2581
+ * the command to reconcile.
2582
+ *
2583
+ * The order is exactly: refuse if the guest cannot keep output, reserve
2584
+ * the id, and only then admit the command. Reserving the id the CALLER
2585
+ * named is what makes a retried start idempotent: a second call with the
2586
+ * same id inside retention finds the execution already running or
2587
+ * finished, sends no `execute`, and attaches to the one that exists.
2588
+ *
2589
+ * `SandboxExecOptions.signal` keeps its contract — aborting it runs the
2590
+ * confirmed cancel — and `detachSignal` is its opposite: it ends the
2591
+ * observation and leaves the command alone, for a host that is shutting
2592
+ * down and wants its work to survive the rollout.
2593
+ */
2594
+ async execDetached(
2595
+ command: string,
2596
+ argv?: string[],
2597
+ opts: KubernetesDetachedExecOptions = {},
2598
+ ): Promise<SandboxExecResult> {
2599
+ const executionId = opts.executionId ?? `exec_${randomUUID()}`
2600
+ assertExecutionId(executionId)
2601
+ const startedAt = Date.now()
2602
+ if (opts.signal?.aborted) {
2603
+ return {
2604
+ exitCode: 1,
2605
+ stdout: '',
2606
+ stderr: '',
2607
+ timedOut: false,
2608
+ durationMs: Math.max(0, Date.now() - startedAt),
2609
+ stdoutTruncated: false,
2610
+ stderrTruncated: false,
2611
+ }
2612
+ }
2613
+ await this.assertExecutionAttachSupported(opts.signal)
2614
+
2615
+ const cursor: OutputCursor = { offset: 0, stdout: '', stderr: '', droppedBytes: 0 }
2616
+ const observationTimeoutMs = Math.min(
2617
+ MAX_TIMER_DELAY_MS,
2618
+ (typeof opts.timeout === 'number' && Number.isFinite(opts.timeout) && opts.timeout > 0
2619
+ ? opts.timeout
2620
+ : DEFAULT_EXECUTION_TIMEOUT_MS) + EXECUTION_OBSERVATION_GRACE_MS,
2621
+ )
2622
+ const deadlineAt = Date.now() + observationTimeoutMs
2623
+
2624
+ // The caller's abort runs the CONFIRMED cancel, in the background,
2625
+ // while the observation keeps reading: the guest answers the cancel
2626
+ // on its own connection and ends this one with the terminal frame, so
2627
+ // aborting produces a result rather than a severed stream.
2628
+ let cancelFailure: unknown
2629
+ let cancelling = false
2630
+ const cancelNow = (): void => {
2631
+ if (cancelling) return
2632
+ cancelling = true
2633
+ void this.confirmCancel(executionId).catch((error: unknown) => {
2634
+ cancelFailure = error
2635
+ })
2636
+ }
2637
+ const onAbort = () => cancelNow()
2638
+ opts.signal?.addEventListener('abort', onAbort, { once: true })
2639
+
2640
+ // Aborting this stops the READING and nothing else. It is never the
2641
+ // caller's `signal`: destroying a socket does not end a guest command,
2642
+ // and a host that treats it as though it did is exactly how a network
2643
+ // blip used to cost a workspace its pod.
2644
+ const observation = new AbortController()
2645
+ const onDetach = () => observation.abort(new Error('kubernetes: observation detached'))
2646
+ opts.detachSignal?.addEventListener('abort', onDetach, { once: true })
2647
+ if (opts.detachSignal?.aborted) onDetach()
2648
+
2649
+ try {
2650
+ let lastError: unknown
2651
+ const reservation = await this.reserveDetached(executionId, opts.signal)
2652
+ if (reservation.state === 'reserved') {
2653
+ try {
2654
+ return await this.executeRetained(
2655
+ executionId,
2656
+ {
2657
+ executionId,
2658
+ command,
2659
+ args: argv ?? [],
2660
+ ...(opts.cwd !== undefined ? { cwd: opts.cwd } : {}),
2661
+ ...(opts.env !== undefined ? { env: opts.env } : {}),
2662
+ ...(opts.timeout !== undefined ? { timeoutMs: opts.timeout } : {}),
2663
+ retainOutput: true,
2664
+ },
2665
+ cursor,
2666
+ opts,
2667
+ observation.signal,
2668
+ observationTimeoutMs,
2669
+ )
2670
+ } catch (error) {
2671
+ lastError = error
2672
+ // A second host process that reserved the same id and won
2673
+ // the race owns the command now. Its refusal says so, and
2674
+ // the answer is to ATTACH to the command that exists —
2675
+ // reporting a failure here would be a lie about an id
2676
+ // whose command is running.
2677
+ if (!lostTheStartRace(error) && isTerminalAttachError(error)) throw error
2678
+ }
2679
+ }
2680
+
2681
+ // The window bounds GETTING BACK, and only that. It is armed as an
2682
+ // abort rather than checked between attempts because a single
2683
+ // attempt is not short: the dial carries its own connect-retry
2684
+ // budget, so a peer that is refusing connections would otherwise
2685
+ // be waited on for that whole budget inside one attempt and the
2686
+ // caller's bound would never be consulted. Once an attach has
2687
+ // actually attached the window is disarmed — a command that is
2688
+ // being read successfully is not something to give up on — and it
2689
+ // is re-armed if that connection dies in its turn.
2690
+ const reattachWindowMs = opts.reattachWindowMs ?? DEFAULT_REATTACH_WINDOW_MS
2691
+ let windowEndsAt = Date.now() + reattachWindowMs
2692
+ for (;;) {
2693
+ if (opts.detachSignal?.aborted) break
2694
+ const remainingMs = windowEndsAt - Date.now()
2695
+ if (remainingMs <= 0) break
2696
+ const window = new AbortController()
2697
+ const windowTimer = setTimeout(
2698
+ () =>
2699
+ window.abort(
2700
+ new Error(
2701
+ `kubernetes: could not reattach to execution ${executionId} within ${reattachWindowMs}ms`,
2702
+ ),
2703
+ ),
2704
+ remainingMs,
2705
+ )
2706
+ windowTimer.unref?.()
2707
+ let reattached = false
2708
+ try {
2709
+ return await this.attachOnce(
2710
+ executionId,
2711
+ cursor,
2712
+ opts,
2713
+ AbortSignal.any([observation.signal, window.signal]),
2714
+ deadlineAt,
2715
+ () => {
2716
+ reattached = true
2717
+ clearTimeout(windowTimer)
2718
+ },
2719
+ )
2720
+ } catch (error) {
2721
+ lastError = error
2722
+ if (isTerminalAttachError(error)) break
2723
+ if (opts.detachSignal?.aborted) break
2724
+ if (reattached) windowEndsAt = Date.now() + reattachWindowMs
2725
+ else if (window.signal.aborted) break
2726
+ await pause(REATTACH_RETRY_DELAY_MS)
2727
+ } finally {
2728
+ clearTimeout(windowTimer)
2729
+ }
2730
+ }
2731
+ throw this.detached(executionId, cursor, cancelFailure ?? lastError)
2732
+ } finally {
2733
+ opts.signal?.removeEventListener('abort', onAbort)
2734
+ opts.detachSignal?.removeEventListener('abort', onDetach)
2735
+ observation.abort(new Error('kubernetes: detached execution observation finished'))
2736
+ }
2737
+ }
2738
+
2739
+ /**
2740
+ * The `execute` leg of a detached run, read frame by frame.
2741
+ *
2742
+ * It deliberately does NOT go through `executeStreamed`: that path
2743
+ * hands its caller `{stream, data}` and nothing else, and the byte
2744
+ * offsets this cursor lives on are on the frames themselves. Reading
2745
+ * them here is what lets a reattach resume exactly where this
2746
+ * connection stopped, on output whose decoded length is not its byte
2747
+ * length — see {@link applyDelta}. The frame union and its validation
2748
+ * are the shared ones, so an ordinary exec and a detached one never
2749
+ * disagree about what the guest said.
2750
+ */
2751
+ private async executeRetained(
2752
+ executionId: string,
2753
+ body: ExecRequest,
2754
+ cursor: OutputCursor,
2755
+ opts: KubernetesDetachedExecOptions,
2756
+ signal: AbortSignal,
2757
+ observationTimeoutMs: number,
2758
+ ): Promise<SandboxExecResult> {
2759
+ // No `onOutput` on the accumulator: this reads the frames, so the
2760
+ // caller is called exactly once per chunk, from `applyDelta`.
2761
+ const accumulator = new ExecResultAccumulator(Date.now())
2762
+ await this.wire.streamFramedRequest(
2763
+ { op: 'execute', body },
2764
+ (frame) => {
2765
+ if (isUnauthorized(frame)) throw new KubernetesAgentUnauthorizedError()
2766
+ const event = parseExecEvent(frame)
2767
+ const stream = deltaStream(event.type)
2768
+ if (stream !== undefined) applyDelta(cursor, executionId, stream, frame, opts.onOutput)
2769
+ accumulator.push(event)
2770
+ },
2771
+ { observationTimeoutMs },
2772
+ signal,
2773
+ )
2774
+ if (!accumulator.done) {
2775
+ throw new RemoteProtocolError(
2776
+ `kubernetes: the execute stream for ${executionId} ended without a result`,
2777
+ )
2778
+ }
2779
+ return accumulator.finish()
2780
+ }
2781
+
2782
+ /** One `attach-execution` stream, read to its terminal frame. */
2783
+ private async attachOnce(
2784
+ executionId: string,
2785
+ cursor: OutputCursor,
2786
+ opts: KubernetesAttachExecutionOptions,
2787
+ signal: AbortSignal,
2788
+ deadlineAt?: number,
2789
+ onAttached?: () => void,
2790
+ ): Promise<SandboxExecResult> {
2791
+ let terminal: AttachTerminal | undefined
2792
+ let refusal: KubernetesExecutionNotAttachableError | undefined
2793
+ const observationTimeoutMs =
2794
+ deadlineAt === undefined
2795
+ ? MAX_TIMER_DELAY_MS
2796
+ : Math.max(1, Math.min(MAX_TIMER_DELAY_MS, deadlineAt - Date.now()))
2797
+ await this.wire.streamFramedRequest(
2798
+ { op: 'attach-execution', body: { executionId, fromOffset: cursor.offset } },
2799
+ (event) => {
2800
+ if (isUnauthorized(event)) throw new KubernetesAgentUnauthorizedError()
2801
+ const type = event.type
2802
+ if (type === 'attached') {
2803
+ onAttached?.()
2804
+ const from = Number(event.fromOffset)
2805
+ if (Number.isFinite(from)) cursor.offset = from
2806
+ const dropped = Number(event.droppedBytes ?? 0)
2807
+ if (Number.isFinite(dropped) && dropped > 0) {
2808
+ cursor.droppedBytes += dropped
2809
+ opts.onGap?.({ executionId, fromOffset: cursor.offset, droppedBytes: dropped })
2810
+ }
2811
+ return
2812
+ }
2813
+ const stream = deltaStream(type)
2814
+ if (stream !== undefined) {
2815
+ applyDelta(cursor, executionId, stream, event, opts.onOutput)
2816
+ return
2817
+ }
2818
+ if (type === 'attach_result') {
2819
+ const next = Number(event.nextOffset)
2820
+ if (Number.isFinite(next)) cursor.offset = next
2821
+ const outcome = String(event.outcome)
2822
+ terminal = {
2823
+ outcome:
2824
+ outcome === 'cancelled' || outcome === 'failed'
2825
+ ? (outcome as 'cancelled' | 'failed')
2826
+ : 'completed',
2827
+ ...(event.result !== undefined
2828
+ ? { result: terminalMetadataOrThrow(event.result, executionId) }
2829
+ : {}),
2830
+ ...(typeof event.error === 'string' ? { error: event.error } : {}),
2831
+ }
2832
+ return
2833
+ }
2834
+ if (type === 'error') {
2835
+ const code = typeof event.error === 'string' ? event.error : 'unknown'
2836
+ const state = typeof event.state === 'string' ? event.state : undefined
2837
+ // A refusal for an execution the guest still holds as
2838
+ // `reserved` is the one case that is not about retention:
2839
+ // the command was never started, so there is nothing
2840
+ // running and nothing to come back for. Saying anything
2841
+ // else here would send a caller looking for a process
2842
+ // that does not exist.
2843
+ const because =
2844
+ state === 'reserved'
2845
+ ? 'The guest holds it as reserved and never started it, so no command is running and there is nothing to reattach to.'
2846
+ : 'A record is kept for the retention window configured by NAMZU_AGENT_EXECUTION_RETAINED_TTL_MS and is lost when the pod is replaced; only a command started with detach keeps its output at all.'
2847
+ refusal = new KubernetesExecutionNotAttachableError(
2848
+ executionId,
2849
+ isAttachRefusal(code) ? code : 'unknown',
2850
+ `kubernetes: the guest refused to attach to execution ${executionId} (${code}). ${because}`,
2851
+ { state },
2852
+ )
2853
+ return
2854
+ }
2855
+ throw new RemoteProtocolError(
2856
+ `kubernetes: the guest sent an unexpected frame on the attach stream for ${executionId}: ${JSON.stringify(event).slice(0, 200)}`,
2857
+ )
2858
+ },
2859
+ { observationTimeoutMs },
2860
+ signal,
2861
+ )
2862
+ if (refusal !== undefined) throw refusal
2863
+ if (terminal === undefined) {
2864
+ throw new RemoteProtocolError(
2865
+ `kubernetes: the attach stream for ${executionId} ended without a terminal frame`,
2866
+ )
2867
+ }
2868
+ return resultFromAttachTerminal(executionId, terminal, cursor)
2869
+ }
2870
+
2871
+ /** The one error a lost observation ends with. */
2872
+ private detached(
2873
+ executionId: string,
2874
+ cursor: OutputCursor,
2875
+ cause: unknown,
2876
+ ): KubernetesExecutionDetachedError {
2877
+ // The guest can tell us the command never started — the reservation
2878
+ // is still `reserved`, so the `execute` never reached it. Then
2879
+ // there is nothing running, nothing to reattach to and nothing to
2880
+ // cancel, and promising otherwise sends the caller after a process
2881
+ // that does not exist. Every other cause leaves the command's fate
2882
+ // genuinely unknown to this host, which is what the rest says.
2883
+ const neverStarted =
2884
+ cause instanceof KubernetesExecutionNotAttachableError && cause.executionState === 'reserved'
2885
+ const advice = neverStarted
2886
+ ? 'the guest still holds it as RESERVED, so the command never started: nothing is running, and starting it again with the same id is safe.'
2887
+ : `it may still be running in the workspace pod. Reattach with attachExecution('${executionId}', { fromOffset: ${cursor.offset} }), from this process or another one, or end it with cancelExecution('${executionId}').`
2888
+ return new KubernetesExecutionDetachedError(
2889
+ executionId,
2890
+ cursor.offset,
2891
+ `kubernetes: stopped observing execution ${executionId} after ${cursor.offset} bytes of output, and did NOT cancel it — ${advice} Cause: ${cause instanceof Error ? cause.message : String(cause)}`,
2892
+ { cause },
2893
+ )
2894
+ }
286
2895
  }