@namzu/sandbox 14.0.0 → 15.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (99) hide show
  1. package/CHANGELOG.md +838 -0
  2. package/README.md +310 -14
  3. package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
  4. package/dist/backends/aci-standby-pool/index.js +13 -1
  5. package/dist/backends/aci-standby-pool/index.js.map +1 -1
  6. package/dist/backends/docker/index.d.ts.map +1 -1
  7. package/dist/backends/docker/index.js +19 -1
  8. package/dist/backends/docker/index.js.map +1 -1
  9. package/dist/backends/firecracker/index.d.ts.map +1 -1
  10. package/dist/backends/firecracker/index.js +12 -2
  11. package/dist/backends/firecracker/index.js.map +1 -1
  12. package/dist/backends/firecracker/protocol.d.ts +459 -8
  13. package/dist/backends/firecracker/protocol.d.ts.map +1 -1
  14. package/dist/backends/firecracker/protocol.js +136 -0
  15. package/dist/backends/firecracker/protocol.js.map +1 -1
  16. package/dist/backends/firecracker/transport.d.ts +539 -6
  17. package/dist/backends/firecracker/transport.d.ts.map +1 -1
  18. package/dist/backends/firecracker/transport.js +1171 -24
  19. package/dist/backends/firecracker/transport.js.map +1 -1
  20. package/dist/backends/kubernetes/egress-policy.d.ts +1088 -11
  21. package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -1
  22. package/dist/backends/kubernetes/egress-policy.js +2173 -29
  23. package/dist/backends/kubernetes/egress-policy.js.map +1 -1
  24. package/dist/backends/kubernetes/identity.d.ts +193 -0
  25. package/dist/backends/kubernetes/identity.d.ts.map +1 -0
  26. package/dist/backends/kubernetes/identity.js +147 -0
  27. package/dist/backends/kubernetes/identity.js.map +1 -0
  28. package/dist/backends/kubernetes/index.d.ts +678 -33
  29. package/dist/backends/kubernetes/index.d.ts.map +1 -1
  30. package/dist/backends/kubernetes/index.js +1180 -95
  31. package/dist/backends/kubernetes/index.js.map +1 -1
  32. package/dist/backends/kubernetes/ingress-policy.d.ts +375 -0
  33. package/dist/backends/kubernetes/ingress-policy.d.ts.map +1 -0
  34. package/dist/backends/kubernetes/ingress-policy.js +1050 -0
  35. package/dist/backends/kubernetes/ingress-policy.js.map +1 -0
  36. package/dist/backends/kubernetes/k8s-client.d.ts +213 -4
  37. package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -1
  38. package/dist/backends/kubernetes/k8s-client.js +359 -52
  39. package/dist/backends/kubernetes/k8s-client.js.map +1 -1
  40. package/dist/backends/kubernetes/lease.d.ts +40 -14
  41. package/dist/backends/kubernetes/lease.d.ts.map +1 -1
  42. package/dist/backends/kubernetes/lease.js +68 -18
  43. package/dist/backends/kubernetes/lease.js.map +1 -1
  44. package/dist/backends/kubernetes/objects.d.ts +423 -3
  45. package/dist/backends/kubernetes/objects.d.ts.map +1 -1
  46. package/dist/backends/kubernetes/objects.js +364 -2
  47. package/dist/backends/kubernetes/objects.js.map +1 -1
  48. package/dist/backends/kubernetes/per-sandbox-policy.d.ts +219 -0
  49. package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -0
  50. package/dist/backends/kubernetes/per-sandbox-policy.js +407 -0
  51. package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -0
  52. package/dist/backends/kubernetes/rbac.d.ts +153 -0
  53. package/dist/backends/kubernetes/rbac.d.ts.map +1 -0
  54. package/dist/backends/kubernetes/rbac.js +177 -0
  55. package/dist/backends/kubernetes/rbac.js.map +1 -0
  56. package/dist/backends/kubernetes/sandbox.d.ts +81 -14
  57. package/dist/backends/kubernetes/sandbox.d.ts.map +1 -1
  58. package/dist/backends/kubernetes/sandbox.js +149 -15
  59. package/dist/backends/kubernetes/sandbox.js.map +1 -1
  60. package/dist/backends/kubernetes/transport.d.ts +935 -9
  61. package/dist/backends/kubernetes/transport.d.ts.map +1 -1
  62. package/dist/backends/kubernetes/transport.js +1958 -62
  63. package/dist/backends/kubernetes/transport.js.map +1 -1
  64. package/dist/backends/kubernetes/workspace.d.ts +1149 -18
  65. package/dist/backends/kubernetes/workspace.d.ts.map +1 -1
  66. package/dist/backends/kubernetes/workspace.js +2825 -186
  67. package/dist/backends/kubernetes/workspace.js.map +1 -1
  68. package/dist/backends/remote-execution-controller.d.ts +14 -0
  69. package/dist/backends/remote-execution-controller.d.ts.map +1 -1
  70. package/dist/backends/remote-execution-controller.js.map +1 -1
  71. package/dist/index.d.ts +231 -13
  72. package/dist/index.d.ts.map +1 -1
  73. package/dist/index.js +247 -5
  74. package/dist/index.js.map +1 -1
  75. package/dist/testing/sandbox-conformance.d.ts +39 -5
  76. package/dist/testing/sandbox-conformance.d.ts.map +1 -1
  77. package/dist/testing/sandbox-conformance.js +436 -5
  78. package/dist/testing/sandbox-conformance.js.map +1 -1
  79. package/package.json +3 -3
  80. package/src/backends/aci-standby-pool/index.ts +16 -1
  81. package/src/backends/docker/index.ts +22 -1
  82. package/src/backends/firecracker/index.ts +14 -2
  83. package/src/backends/firecracker/protocol.ts +514 -6
  84. package/src/backends/firecracker/transport.ts +1492 -40
  85. package/src/backends/kubernetes/egress-policy.ts +3064 -53
  86. package/src/backends/kubernetes/identity.ts +261 -0
  87. package/src/backends/kubernetes/index.ts +1785 -127
  88. package/src/backends/kubernetes/ingress-policy.ts +1344 -0
  89. package/src/backends/kubernetes/k8s-client.ts +444 -54
  90. package/src/backends/kubernetes/lease.ts +75 -19
  91. package/src/backends/kubernetes/objects.ts +626 -6
  92. package/src/backends/kubernetes/per-sandbox-policy.ts +542 -0
  93. package/src/backends/kubernetes/rbac.ts +192 -0
  94. package/src/backends/kubernetes/sandbox.ts +218 -20
  95. package/src/backends/kubernetes/transport.ts +2733 -124
  96. package/src/backends/kubernetes/workspace.ts +4476 -222
  97. package/src/backends/remote-execution-controller.ts +14 -0
  98. package/src/index.ts +595 -14
  99. package/src/testing/sandbox-conformance.ts +540 -5
@@ -23,8 +23,12 @@
23
23
  * sub-second warm-acquire target this backend is judged against must be
24
24
  * measured, not guessed at.
25
25
  */
26
- import { VsockAgentTransport, } from '../firecracker/transport.js';
27
- import { RemoteExecutionController, } from '../remote-execution-controller.js';
26
+ import { randomUUID } from 'node:crypto';
27
+ import net from 'node:net';
28
+ import { EXECUTION_ATTACH_FEATURE, ExecResultAccumulator, FLUSH_FEATURE, QUIESCE_FEATURE, SESSIONS_FEATURE, parseExecEvent, } from '../firecracker/protocol.js';
29
+ import { AgentDialFailedError, VsockAgentTransport, } from '../firecracker/transport.js';
30
+ import { RemoteCancellationUnknownError, RemoteCommandError, RemoteExecutionController, RemoteProtocolError, RemoteResultIncompleteError, } from '../remote-execution-controller.js';
31
+ import { KubernetesWorkspaceReplacedError } from './identity.js';
28
32
  /**
29
33
  * Thrown when the guest agent refuses a request because the handle's
30
34
  * token does not match what the pod is bound to. Named distinctly from
@@ -33,21 +37,288 @@ import { RemoteExecutionController, } from '../remote-execution-controller.js';
33
37
  * unexpected".
34
38
  */
35
39
  export class KubernetesAgentUnauthorizedError extends Error {
36
- constructor(message = 'kubernetes tcp transport: the guest agent rejected this connection’s token (unauthorized)') {
37
- super(message);
40
+ constructor(message = 'kubernetes tcp transport: the guest agent rejected this connection’s token (unauthorized). The token is the bound pod’s metadata.uid and the guest checks it BEFORE dispatch, so the request ran nothing at all; a handle that can re-read its pod follows the replacement and retries once, and this error is what stands when there is nothing to follow — the same pod is still there and still refusing, or the re-read could not be made.', options) {
41
+ super(message, options);
38
42
  this.name = 'KubernetesAgentUnauthorizedError';
39
43
  }
40
44
  }
45
+ /**
46
+ * Thrown when the guest agent refuses a request because it has FENCED
47
+ * ITSELF: an earlier process group's shutdown could not be confirmed, so it
48
+ * answers every op but `healthz` and `cancel-execution` with `agent_retiring`
49
+ * and will go on doing so until the pod is replaced.
50
+ *
51
+ * Its own class, and deliberately NOT the Firecracker tier's mapping of the
52
+ * same refusal. There a fenced agent becomes {@link
53
+ * RemoteCancellationUnknownError}, which is correct for a disposable microVM:
54
+ * the shared controller's rule is that the sandbox stops being reusable, and
55
+ * on that tier retiring one means deleting scratch. On a workspace the same
56
+ * error would retire the handle and take the pod — and with it every other
57
+ * holder's terminals, dev servers and running commands — away from callers
58
+ * who did nothing but share a workspace with the command that wedged.
59
+ *
60
+ * So this error refuses the ONE call rather than the workspace: the handle is
61
+ * not retired, nothing is patched, and every other holder's pod stays where it
62
+ * was. What it does NOT claim is that the next call will work. The fence is
63
+ * the GUEST's, and `dispatch` gates it ahead of every data-plane branch, so
64
+ * `readFile`, `writeFile`, `openTerminal` and `openTcpConnection` meet the
65
+ * same refusal on the wire — under their own paths' error shapes, since only
66
+ * the two control ops come through `requestChecked`. Only a new pod clears
67
+ * it, which is why the message names the verbs that REPLACE the pod, on both
68
+ * tiers that use this transport, and leaves the timing to the host: on a
69
+ * workspace those are `suspend()` then `resume()`, and they take the live
70
+ * sessions in that pod down with them.
71
+ */
72
+ export class KubernetesAgentRetiringError extends Error {
73
+ name = 'KubernetesAgentRetiringError';
74
+ constructor(message = 'kubernetes tcp transport: the guest agent has fenced itself (agent_retiring) because an earlier process group’s shutdown could not be confirmed. A command of unknown state may still be running in that pod and nothing on this side can end it: only a new pod clears the fence, and until the pod is replaced every call except healthz and cancel-execution meets this same refusal — reads, writes, terminals and tcp connections included. Nothing was changed on the cluster by this refusal and this handle was not retired. On a persistent workspace, call suspend() and then resume() when you are ready for the live terminals and background processes in that pod to go down; on a task sandbox, destroy() it and take another.') {
75
+ super(message);
76
+ }
77
+ }
78
+ /**
79
+ * Thrown when the agent's address could not be RESOLVED — the dial never
80
+ * reached a socket because the name has no answer here.
81
+ *
82
+ * Its own class, and its own message, because this is the one failure whose
83
+ * cause is the deployment's shape rather than anything the cluster did: a
84
+ * Service FQDN resolves through cluster DNS and nowhere else, so a host
85
+ * outside the cluster fails every call at name resolution and reads the
86
+ * result as a sandbox that never came up. The fix is a configuration field,
87
+ * so the error names it.
88
+ */
89
+ export class KubernetesAgentAddressUnresolvableError extends Error {
90
+ host;
91
+ name = 'KubernetesAgentAddressUnresolvableError';
92
+ constructor(
93
+ /** The host that did not resolve — normally a `*.svc.cluster.local`. */
94
+ host, message, options) {
95
+ super(message, options);
96
+ this.host = host;
97
+ }
98
+ }
99
+ /** Every `code` and message in an error's own chain, cause by cause. */
100
+ function errorChain(error) {
101
+ const codes = [];
102
+ const messages = [];
103
+ let current = error;
104
+ // Bounded rather than `while (current)`: a cause cycle is a hang, and no
105
+ // real chain on this path is more than three deep.
106
+ for (let depth = 0; depth < 8 && current instanceof Error; depth += 1) {
107
+ const code = current.code;
108
+ if (typeof code === 'string')
109
+ codes.push(code);
110
+ messages.push(current.message);
111
+ current = current.cause;
112
+ }
113
+ return { codes, messages };
114
+ }
115
+ /**
116
+ * Connect-shaped: the failure came out of the DIAL, so NOTHING was sent to
117
+ * the guest.
118
+ *
119
+ * That last part is what makes a retry safe after the handle follows a
120
+ * replaced pod — a reserved execution, a half-written file or a terminal
121
+ * cannot be sitting in a pod this connection never opened — and it is why the
122
+ * test is WHERE the error came from rather than which `errno` it carries.
123
+ * A code cannot say that much: `ETIMEDOUT` is what the kernel raises when a
124
+ * connect attempt gets no answer AND what it raises on an ESTABLISHED socket
125
+ * that has run out of retransmits — the second of those happens mid-request,
126
+ * with bytes already delivered, and retrying it is not safe.
127
+ *
128
+ * {@link AgentDialFailedError} is what {@link VsockAgentTransport}'s dial
129
+ * throws when it gives up without a socket, so the question is asked of the
130
+ * error's own type, anywhere in its cause chain. Not of its text: a guest
131
+ * answers with text, and a command's stderr quoting "could not connect to
132
+ * agent" would otherwise be read as this process's own dial failing.
133
+ */
134
+ function isConnectFailure(error) {
135
+ let current = error;
136
+ // Bounded for the same reason {@link errorChain} is: a cause cycle is a
137
+ // hang, and no real chain on this path is more than three deep.
138
+ for (let depth = 0; depth < 8 && current instanceof Error; depth += 1) {
139
+ if (current instanceof AgentDialFailedError)
140
+ return true;
141
+ current = current.cause;
142
+ }
143
+ return false;
144
+ }
145
+ /** Nothing this attempt sent ever reached a socket. */
146
+ function neverConnected(dials) {
147
+ return dials?.attempted === true && !dials.connected;
148
+ }
149
+ /**
150
+ * Outcomes no rebind may retry, whatever the failure underneath them looks
151
+ * like.
152
+ *
153
+ * Both are the controller's way of saying the REMOTE state is unknown: a
154
+ * cancellation it could not confirm, a confirmed termination whose result
155
+ * stream broke. Their own messages interpolate the underlying failure, so a
156
+ * dial that broke mid-command travels as the `cause` of one of them — and a
157
+ * retry there would re-run a command that may already have run, against a
158
+ * disk that followed the pod. `RemoteCancellationUnknownError`
159
+ * says so in its own words: "do not automatically retry the command".
160
+ */
161
+ function isUnretryableOutcome(error) {
162
+ return (error instanceof RemoteCancellationUnknownError || error instanceof RemoteResultIncompleteError);
163
+ }
164
+ /**
165
+ * Name-resolution-shaped: `getaddrinfo` refused the handle's host.
166
+ *
167
+ * Asked only of a DIAL failure, and only of a handle whose host is a name —
168
+ * both gates are at the one call site, {@link
169
+ * KubernetesAgentTransport.withRebind}. The message match below is
170
+ * load-bearing (the retry wrapper's own Error does not carry the `code`
171
+ * forward) and a substring test is exactly as strong as the text it is given,
172
+ * so it is never asked of an arbitrary error: a `readFile` that fails because
173
+ * the guest reported a path containing `ENOTFOUND`, or an exec whose output is
174
+ * quoted into a message, is not a resolver failure and must not be rewritten
175
+ * into one. `net.connect` never consults a resolver for a literal either, so a
176
+ * `pod-ip` handle keeps its one re-read however an unrelated error is worded.
177
+ */
178
+ const DNS_ERROR_CODES = new Set(['ENOTFOUND', 'EAI_AGAIN']);
179
+ function isNameResolutionFailure(error) {
180
+ const { codes, messages } = errorChain(error);
181
+ if (codes.some((code) => DNS_ERROR_CODES.has(code)))
182
+ return true;
183
+ // `net.connect` surfaces a `getaddrinfo ENOTFOUND <host>` message whose
184
+ // `code` the retry wrapper's own Error does not carry forward.
185
+ return messages.some((message) => message.includes('ENOTFOUND') || message.includes('EAI_AGAIN'));
186
+ }
187
+ /**
188
+ * The resolver said the name does not exist — as opposed to saying it could
189
+ * not answer right now.
190
+ *
191
+ * Only this half is treated as permanent, and the difference is the default
192
+ * mode's whole retry budget. `EAI_AGAIN` is BY DEFINITION "temporary failure
193
+ * in name resolution": it is the shape a CoreDNS restart or a conntrack race
194
+ * produces for a host INSIDE the cluster, where the Service FQDN is correct
195
+ * and waiting is exactly the cure — the 30s budget exists to ride over that,
196
+ * and advice to set `agentAddress: 'pod-ip'` would be wrong for that host.
197
+ * `ENOTFOUND` is the out-of-cluster symptom this mode exists for: a resolver
198
+ * that has answered, definitively, that the name is not a name here, and no
199
+ * amount of re-asking it changes that.
200
+ *
201
+ * A temporary failure that outlives the budget still ends as
202
+ * {@link KubernetesAgentAddressUnresolvableError}, so the diagnosis is
203
+ * delayed rather than lost.
204
+ */
205
+ function isMissingNameFailure(error) {
206
+ const { codes, messages } = errorChain(error);
207
+ if (codes.includes('ENOTFOUND'))
208
+ return true;
209
+ return messages.some((message) => message.includes('ENOTFOUND'));
210
+ }
41
211
  /** True for the wire shape `agent.cjs` sends when a token is rejected. */
42
212
  function isUnauthorized(response) {
213
+ return refusedWith(response, 'unauthorized');
214
+ }
215
+ /**
216
+ * True for the wire shape `agent.cjs` sends when it has fenced itself —
217
+ * `dispatch`'s gate, and `handleReserveExecution`'s own earlier one.
218
+ */
219
+ function isAgentRetiring(response) {
220
+ return refusedWith(response, 'agent_retiring');
221
+ }
222
+ /** One refusal envelope: `{ ok: false, error: <name> }`, and nothing else. */
223
+ function refusedWith(response, error) {
43
224
  if (!response || typeof response !== 'object')
44
225
  return false;
45
226
  const value = response;
46
- return value.ok === false && value.error === 'unauthorized';
227
+ return value.ok === false && value.error === error;
47
228
  }
48
229
  /**
49
- * One framed control request, with the guest's `unauthorized` refusal
50
- * turned into {@link KubernetesAgentUnauthorizedError}.
230
+ * The guest REFUSED this handle's token — whichever of the two shapes that
231
+ * refusal happens to arrive in.
232
+ *
233
+ * `reserve-execution` and `cancel-execution` come through
234
+ * {@link requestChecked}, so their refusal is already a
235
+ * {@link KubernetesAgentUnauthorizedError}. Nothing else does:
236
+ * `writeFile`, `readFile`, `openTerminal` and `openTcpConnection` are the
237
+ * shared Firecracker transport's own paths, and each of them throws the
238
+ * guest's error NAME as a plain `Error` — `unauthorized` and nothing more.
239
+ * Two shapes for one fact is why a host could never hang a single recovery
240
+ * off it, and this is where they become one.
241
+ *
242
+ * The message test is exact, never a substring, and that is deliberate: the
243
+ * guest's refusal envelope carries the bare token `unauthorized` as its whole
244
+ * `error` field, while a read whose PATH contains the word, or an exec whose
245
+ * output is quoted into a message, does not equal it. The chain is walked
246
+ * because the retry wrapper and the stream paths wrap rather than replace.
247
+ */
248
+ function isUnauthorizedRefusal(error) {
249
+ let current = error;
250
+ for (let depth = 0; depth < 8 && current instanceof Error; depth += 1) {
251
+ if (current instanceof KubernetesAgentUnauthorizedError)
252
+ return true;
253
+ if (current.message === 'unauthorized')
254
+ return true;
255
+ current = current.cause;
256
+ }
257
+ return false;
258
+ }
259
+ /**
260
+ * The one shape a refused call ends in when no rebind could fix it.
261
+ *
262
+ * An error that is already the named class travels unchanged — replacing it
263
+ * would lose its `cause` chain for nothing — and every other shape is wrapped
264
+ * with the original on `cause`, so the plain `Error('unauthorized')` a
265
+ * `writeFile` used to end with is still readable underneath.
266
+ */
267
+ function asUnauthorized(error) {
268
+ if (error instanceof KubernetesAgentUnauthorizedError)
269
+ return error;
270
+ return new KubernetesAgentUnauthorizedError(undefined, { cause: error });
271
+ }
272
+ /** The optional boot id on one guest reply, and nothing read from any other shape. */
273
+ function guestBootIdOf(reply) {
274
+ if (!reply || typeof reply !== 'object')
275
+ return undefined;
276
+ const value = reply.guestBootId;
277
+ return typeof value === 'string' && value !== '' ? value : undefined;
278
+ }
279
+ /**
280
+ * The GUEST that RESERVED each execution whose cancellation could not be
281
+ * confirmed: the pod its reservation was accepted by, and the agent PROCESS
282
+ * inside that pod.
283
+ *
284
+ * Per EXECUTION and never per session, which is the whole reason it is kept
285
+ * here rather than read off the handle when the diagnosis runs. A workspace
286
+ * handle outlives many commands and follows a replaced pod, so what it is
287
+ * bound to when a diagnosis runs is not what a command that failed minutes
288
+ * ago was running in: a container the kubelet restarted an hour ago says
289
+ * nothing about a command started after it, and a pod ANOTHER call already
290
+ * rebound to is not the pod this command's processes died with. A handle-wide
291
+ * baseline would report both as "the guest your command was running in is
292
+ * gone", about a command that is very likely still running, and would name
293
+ * the replacement as the guest that died.
294
+ *
295
+ * What is recorded is the bind token this attempt presented — which IS the
296
+ * pod's `metadata.uid`, the same value the handle reports as
297
+ * `identity.podUid` — and the boot id the `reserve-execution` reply carried,
298
+ * because the guest that accepted the reservation is the guest that ran the
299
+ * command. Both are stamped onto the error on its way out of {@link
300
+ * KubernetesAgentTransport.exec}, the last frame that still knows which
301
+ * execution an error belongs to.
302
+ *
303
+ * Weak, so an error nobody kept takes its entry with it. The boot id is
304
+ * absent for a guest too old to report one, and the whole entry is absent for
305
+ * an error no `exec()` produced: the diagnosis then falls back to what the
306
+ * handle itself is bound to, which is the honest answer rather than a guess.
307
+ */
308
+ const executionGuest = new WeakMap();
309
+ /**
310
+ * The guest that reserved the execution this error came from — see
311
+ * {@link executionGuest}. `undefined` when the error is not an execution
312
+ * failure this transport produced.
313
+ */
314
+ export function guestWhenReserved(error) {
315
+ return error instanceof Error ? executionGuest.get(error) : undefined;
316
+ }
317
+ /**
318
+ * One framed control request, with the guest's two NAMED refusals turned
319
+ * into errors a caller can catch by class: {@link
320
+ * KubernetesAgentUnauthorizedError} for a rejected token, and {@link
321
+ * KubernetesAgentRetiringError} for an agent that has fenced itself.
51
322
  *
52
323
  * BOTH control requests go through here — `reserve-execution` and
53
324
  * `cancel-execution` — because the two are read by the same caller for
@@ -58,13 +329,401 @@ function isUnauthorized(response) {
58
329
  * rejected token read as a transport failure would therefore spend that
59
330
  * window re-sending a request that can never succeed, and end by
60
331
  * describing a wrong credential as an ambiguous outcome.
332
+ *
333
+ * The fenced-agent refusal is the one a SECOND holder meets. `dispatch`
334
+ * lets `cancel-execution` through the fence and refuses everything else, so
335
+ * `agent_retiring` reaches here from `reserve-execution` — which without
336
+ * this mapping is parsed as a reservation and rejected as
337
+ * `RemoteProtocolError: remote sandbox returned an invalid execution
338
+ * reservation`, a message about a wire shape for a pod that is telling the
339
+ * truth about itself. Named here rather than in the shared controller
340
+ * because the two tiers answer it differently: see {@link
341
+ * KubernetesAgentRetiringError}.
61
342
  */
62
343
  async function requestChecked(wire, request, signal) {
63
344
  const response = await wire.request(request, signal);
64
345
  if (isUnauthorized(response))
65
346
  throw new KubernetesAgentUnauthorizedError();
347
+ if (isAgentRetiring(response))
348
+ throw new KubernetesAgentRetiringError();
66
349
  return response;
67
350
  }
351
+ // --- detachable executions (#479) -----------------------------------------
352
+ /** How long a detached `exec()` keeps trying to get its stream back. */
353
+ const DEFAULT_REATTACH_WINDOW_MS = 30_000;
354
+ /** Pause between reattach attempts, so a refusing port is not hot-looped. */
355
+ const REATTACH_RETRY_DELAY_MS = 250;
356
+ /** The guest's own default, mirrored so an observation bound exists. */
357
+ const DEFAULT_EXECUTION_TIMEOUT_MS = 5 * 60 * 1_000;
358
+ /** Slack over the command's own timeout, as `executeRaw` allows itself. */
359
+ const EXECUTION_OBSERVATION_GRACE_MS = 10_000;
360
+ /** How long a confirmed cancel is retried before it is reported unknown. */
361
+ const CANCEL_CONFIRM_WINDOW_MS = 8_000;
362
+ const CANCEL_ATTEMPT_TIMEOUT_MS = 2_000;
363
+ const MAX_TIMER_DELAY_MS = 2_147_483_647;
364
+ /**
365
+ * Thrown before a command is admitted, when the caller asked for a
366
+ * detachable execution and the guest does not advertise
367
+ * {@link EXECUTION_ATTACH_FEATURE}.
368
+ *
369
+ * Refused rather than downgraded: a caller that asked for detach is about
370
+ * to rely on being able to come back for the output, and running the
371
+ * command anyway would keep nothing and tell nobody.
372
+ */
373
+ export class KubernetesExecutionAttachUnsupportedError extends Error {
374
+ feature;
375
+ name = 'KubernetesExecutionAttachUnsupportedError';
376
+ constructor(feature, message) {
377
+ super(message);
378
+ this.feature = feature;
379
+ }
380
+ }
381
+ /**
382
+ * Thrown when the guest ANSWERED an attach and refused it — the execution
383
+ * is past its retention, ran in a pod that has since been replaced, never
384
+ * asked for its output to be kept, or the offset names bytes it does not
385
+ * have.
386
+ *
387
+ * Distinct from a transport failure on purpose: a refusal will not become a
388
+ * success by being retried, so the reattach loop stops on it instead of
389
+ * spending its whole window re-asking a question already answered.
390
+ */
391
+ export class KubernetesExecutionNotAttachableError extends Error {
392
+ executionId;
393
+ reason;
394
+ name = 'KubernetesExecutionNotAttachableError';
395
+ /**
396
+ * The state the guest reported for this execution, when it reported
397
+ * one. `'reserved'` is the one that changes what a caller should do:
398
+ * the command was never started, so nothing is running.
399
+ */
400
+ executionState;
401
+ constructor(executionId, reason, message, options) {
402
+ super(message, options);
403
+ this.executionId = executionId;
404
+ this.reason = reason;
405
+ this.executionState = options?.state;
406
+ }
407
+ }
408
+ /**
409
+ * Thrown when a detached `exec()` gave up OBSERVING a command that is, as
410
+ * far as this host knows, still the guest's to run.
411
+ *
412
+ * The two fields are what makes it recoverable rather than merely a
413
+ * failure: `executionId` names the command to a second host process, and
414
+ * `outputOffset` is the byte the next `attachExecution` should resume from
415
+ * so nothing is read twice and no gap is invented.
416
+ *
417
+ * Nothing on this path cancels to reconcile. That is the whole point of
418
+ * the feature: a reset connection used to cost the workspace its pod, and
419
+ * a command the host has stopped watching is not a command that has to
420
+ * die.
421
+ */
422
+ export class KubernetesExecutionDetachedError extends Error {
423
+ executionId;
424
+ outputOffset;
425
+ name = 'KubernetesExecutionDetachedError';
426
+ constructor(executionId, outputOffset, message, options) {
427
+ super(message, options);
428
+ this.executionId = executionId;
429
+ this.outputOffset = outputOffset;
430
+ }
431
+ }
432
+ /** One `stdout_delta`/`stderr_delta` payload, from either op's stream. */
433
+ function deltaStream(type) {
434
+ if (type === 'stdout_delta')
435
+ return 'stdout';
436
+ if (type === 'stderr_delta')
437
+ return 'stderr';
438
+ return undefined;
439
+ }
440
+ function isAttachRefusal(value) {
441
+ return (value === 'unknown_execution' ||
442
+ value === 'output_not_retained' ||
443
+ value === 'invalid_offset' ||
444
+ value === 'invalid_execution_id' ||
445
+ value === 'agent_retiring');
446
+ }
447
+ function terminalMetadataOrThrow(value, executionId) {
448
+ if (!value || typeof value !== 'object') {
449
+ throw new RemoteProtocolError(`kubernetes: the guest ended the attach stream for ${executionId} without terminal metadata`);
450
+ }
451
+ return value;
452
+ }
453
+ function pause(ms) {
454
+ return new Promise((resolve) => {
455
+ const timer = setTimeout(resolve, ms);
456
+ timer.unref?.();
457
+ });
458
+ }
459
+ /**
460
+ * The id shape the guest enforces, mirrored here so a caller-chosen id is
461
+ * refused locally with a message that says what the shape is, rather than
462
+ * as an `invalid_execution_id` frame after a round trip.
463
+ */
464
+ const EXECUTION_ID_PATTERN = /^exec_[0-9a-f]{8}-[0-9a-f]{4}-[1-5][0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$/i;
465
+ function assertExecutionId(executionId) {
466
+ if (EXECUTION_ID_PATTERN.test(executionId))
467
+ return;
468
+ throw new RemoteProtocolError(`kubernetes: ${JSON.stringify(executionId)} is not a valid execution id. The guest accepts exec_<uuid> and nothing else, so that an id minted by one host process is recognisable to another.`);
469
+ }
470
+ /**
471
+ * Fold one delta frame into the cursor, taking the offset FROM THE GUEST,
472
+ * and pass the output on to the caller.
473
+ *
474
+ * The host never derives an offset from the string it received, on either
475
+ * stream, and that is the whole of this function's reason to exist.
476
+ * `Buffer.toString('utf8')` over a chunk that ends mid-character does not
477
+ * preserve byte length — an incomplete sequence decodes to U+FFFD, which
478
+ * is WIDER than the bytes it replaced — so a cursor advanced by
479
+ * `Buffer.byteLength(data)` runs ahead of the guest's retained log the
480
+ * first time a multi-byte character straddles a read boundary. A drifted
481
+ * cursor is not a cosmetic error: the reattach either resumes past bytes
482
+ * that are then never delivered and never reported (a silent hole in a
483
+ * result whose truncation flags both read `false`) or names an offset the
484
+ * guest never had and is refused `invalid_offset`, which costs the caller
485
+ * the feature entirely. The guest stamps `nextOffset` on every delta of a
486
+ * retained execution, on the execute stream and the attach stream alike.
487
+ */
488
+ function applyDelta(cursor, executionId, stream, event, onOutput) {
489
+ const data = typeof event.data === 'string' ? event.data : '';
490
+ const nextOffset = Number(event.nextOffset);
491
+ if (!Number.isFinite(nextOffset)) {
492
+ throw new RemoteProtocolError(`kubernetes: the guest sent a ${stream} delta for execution ${executionId} without the byte offset a reattach resumes from. Every guest advertising '${EXECUTION_ATTACH_FEATURE}' stamps them on a retained execution's output; rebuild the workspace image from this Namzu release.`);
493
+ }
494
+ cursor.offset = nextOffset;
495
+ if (stream === 'stdout')
496
+ cursor.stdout += data;
497
+ else
498
+ cursor.stderr += data;
499
+ onOutput?.({ stream, data });
500
+ }
501
+ /**
502
+ * The guest's refusal to start a command on an id that is no longer
503
+ * `reserved` — another host process got its `execute` in first.
504
+ *
505
+ * It reads as terminal (the guest answered, and answering again will not
506
+ * change it) and it is the one case where that is the wrong conclusion:
507
+ * the command this call asked for EXISTS, so the caller gets it by
508
+ * attaching rather than an error about a race it does not care about.
509
+ */
510
+ function lostTheStartRace(error) {
511
+ return error instanceof RemoteCommandError && error.message.startsWith('execution_not_reserved');
512
+ }
513
+ /**
514
+ * Errors a reattach must NOT spend its window re-asking about: the guest
515
+ * answered, and the answer will be the same next time.
516
+ */
517
+ function isTerminalAttachError(error) {
518
+ return (error instanceof KubernetesExecutionNotAttachableError ||
519
+ error instanceof KubernetesAgentUnauthorizedError ||
520
+ error instanceof KubernetesAgentAddressUnresolvableError ||
521
+ error instanceof RemoteCommandError ||
522
+ error instanceof RemoteProtocolError);
523
+ }
524
+ /** The guest's `cancel-execution` reply, refusals told apart from blips. */
525
+ function parseCancellationReply(executionId, response) {
526
+ const reply = (response ?? {});
527
+ if (reply.ok === true) {
528
+ const state = String(reply.state ?? '');
529
+ if (state === 'cancelled' || state === 'completed' || state === 'failed') {
530
+ return reply;
531
+ }
532
+ throw new RemoteProtocolError(`kubernetes: the guest acknowledged cancelling ${executionId} with an unknown state ${JSON.stringify(reply.state)}`);
533
+ }
534
+ const error = typeof reply.error === 'string' ? reply.error : 'unknown';
535
+ if (error === 'unknown_execution' || error === 'invalid_execution_id') {
536
+ throw new KubernetesExecutionNotAttachableError(executionId, error, `kubernetes: the guest holds no execution ${executionId} to cancel (${error}). A record is kept only for its retention window and is lost when the pod is replaced.`);
537
+ }
538
+ throw new Error(`kubernetes: the guest refused to cancel ${executionId}: ${error}`);
539
+ }
540
+ /**
541
+ * The SDK-shaped result of an attached observation.
542
+ *
543
+ * A reported gap sets BOTH truncation flags. The retained log is one
544
+ * interleaved space, so bytes lost out of it cannot be attributed to
545
+ * stdout or to stderr, and the contract already has exactly one way to say
546
+ * "this output is not all of it". Saying it on one stream only would be a
547
+ * guess; saying it on neither would hand back a short stream that looks
548
+ * complete, which is the thing this design refuses to do.
549
+ */
550
+ function resultFromAttachTerminal(executionId, terminal, cursor) {
551
+ if (terminal.outcome === 'failed' && terminal.result === undefined) {
552
+ throw new RemoteCommandError(terminal.error ?? `the guest reported execution ${executionId} as failed`);
553
+ }
554
+ const metadata = terminalMetadataOrThrow(terminal.result, executionId);
555
+ const lost = cursor.droppedBytes > 0;
556
+ return {
557
+ exitCode: metadata.exitCode,
558
+ stdout: cursor.stdout,
559
+ stderr: cursor.stderr,
560
+ ...(metadata.signal !== undefined ? { signal: metadata.signal } : {}),
561
+ timedOut: metadata.timedOut === true,
562
+ durationMs: metadata.durationMs,
563
+ stdoutTruncated: metadata.stdoutTruncated === true || lost,
564
+ stderrTruncated: metadata.stderrTruncated === true || lost,
565
+ };
566
+ }
567
+ // --- guest sessions (#478) ------------------------------------------------
568
+ /**
569
+ * Thrown before anything is started, when the caller asked for a session and
570
+ * the guest does not advertise {@link SESSIONS_FEATURE}.
571
+ *
572
+ * Refused, never downgraded to a connection-bound terminal. A caller that
573
+ * asked for a session is about to rely on coming back to it after its own
574
+ * process has been replaced; handing it one that dies with the socket would
575
+ * look like it worked until the one moment it was needed.
576
+ */
577
+ export class KubernetesSessionsUnsupportedError extends Error {
578
+ feature;
579
+ name = 'KubernetesSessionsUnsupportedError';
580
+ constructor(feature, message) {
581
+ super(message);
582
+ this.feature = feature;
583
+ }
584
+ }
585
+ const SESSION_REFUSALS = new Set([
586
+ 'unknown_session',
587
+ 'invalid_session_id',
588
+ 'invalid_offset',
589
+ 'session_exists',
590
+ 'session_capacity',
591
+ 'missing_command',
592
+ 'spawn_failed',
593
+ 'agent_retiring',
594
+ ]);
595
+ function isSessionRefusal(value) {
596
+ return SESSION_REFUSALS.has(value);
597
+ }
598
+ /**
599
+ * Thrown when the guest ANSWERED and refused: the session is past its
600
+ * retention, ran in a pod that has since been replaced, the id is already
601
+ * taken, or the offset names bytes it does not have.
602
+ *
603
+ * Distinct from a transport failure for the same reason
604
+ * {@link KubernetesExecutionNotAttachableError} is: a refusal does not
605
+ * become a success by being retried.
606
+ */
607
+ export class KubernetesSessionRefusedError extends Error {
608
+ sessionId;
609
+ reason;
610
+ name = 'KubernetesSessionRefusedError';
611
+ constructor(sessionId, reason, message, options) {
612
+ super(message, options);
613
+ this.sessionId = sessionId;
614
+ this.reason = reason;
615
+ }
616
+ }
617
+ function sessionNumber(value, fallback = 0) {
618
+ const parsed = Number(value);
619
+ return Number.isFinite(parsed) ? parsed : fallback;
620
+ }
621
+ /** How long one `readSession` may spend reading a bounded, one-shot reply. */
622
+ const SESSION_READ_TIMEOUT_MS = 30_000;
623
+ /**
624
+ * How long the SHARED re-read behind a rebind may take before it is given up
625
+ * on — two API GETs on a client that sets no per-request timeout.
626
+ *
627
+ * It replaces the caller's signal rather than joining it, because the read is
628
+ * shared: see {@link KubernetesAgentTransport.rebind}. Generous enough that a
629
+ * busy API server still answers, short enough that a call refused by a
630
+ * replacement is not held behind a hung one.
631
+ */
632
+ const REBIND_READ_TIMEOUT_MS = 10_000;
633
+ /**
634
+ * The same shape the guest enforces, checked here so a bad id is a local
635
+ * error naming the rule rather than a round trip that comes back
636
+ * `invalid_session_id`.
637
+ */
638
+ const SESSION_ID_PATTERN = /^[A-Za-z0-9][A-Za-z0-9._-]{0,63}$/;
639
+ function assertSessionId(sessionId) {
640
+ if (!SESSION_ID_PATTERN.test(sessionId)) {
641
+ throw new Error(`kubernetes: ${JSON.stringify(sessionId)} cannot name a session. It must be 1-64 characters of letters, digits, '.', '_' or '-', starting alphanumeric. The id is the ONLY way another host process finds this session again, so it is refused rather than sanitised.`);
642
+ }
643
+ }
644
+ /**
645
+ * Both session fields or neither.
646
+ *
647
+ * A `sessionId` without `persistent` would open a terminal that dies with
648
+ * its connection under a name nothing can use, and `persistent` without an
649
+ * id would open one nobody can ever find. Either alone is a mistake worth a
650
+ * message rather than a surprise.
651
+ */
652
+ function assertSessionOpen(options) {
653
+ if (options.sessionId === undefined || options.persistent !== true) {
654
+ throw new Error('kubernetes: a persistent terminal needs both `sessionId` and `persistent: true`. An id without `persistent` opens a connection-bound terminal under a name nothing can attach to, and `persistent` without an id opens one nobody can find again.');
655
+ }
656
+ assertSessionId(options.sessionId);
657
+ return options.sessionId;
658
+ }
659
+ /** The workspace-facing terminal around one open session stream. */
660
+ function sessionTerminal(stream, sessionId) {
661
+ return {
662
+ ...stream.session,
663
+ sessionId,
664
+ nextOffset: () => stream.nextOffset(),
665
+ detach: () => stream.detach(),
666
+ };
667
+ }
668
+ /** The guest's row, structurally validated. */
669
+ function parseSessionSummary(value) {
670
+ if (!value || typeof value !== 'object') {
671
+ throw new RemoteProtocolError('kubernetes: the guest sent a session row that is not an object');
672
+ }
673
+ const row = value;
674
+ if (typeof row.sessionId !== 'string') {
675
+ throw new RemoteProtocolError('kubernetes: the guest sent a session row with no sessionId');
676
+ }
677
+ const kind = row.kind === 'detached' ? 'detached' : 'terminal';
678
+ const state = row.state === 'exited' ? 'exited' : 'running';
679
+ return {
680
+ sessionId: row.sessionId,
681
+ kind,
682
+ command: typeof row.command === 'string' ? row.command : '',
683
+ args: Array.isArray(row.args) ? row.args.map(String) : [],
684
+ startedAt: sessionNumber(row.startedAt),
685
+ ...(row.lastInputAt !== undefined ? { lastInputAt: sessionNumber(row.lastInputAt) } : {}),
686
+ ...(row.lastOutputAt !== undefined ? { lastOutputAt: sessionNumber(row.lastOutputAt) } : {}),
687
+ nextOffset: sessionNumber(row.nextOffset),
688
+ droppedBytes: sessionNumber(row.droppedBytes),
689
+ state,
690
+ attached: row.attached === true,
691
+ ...(typeof row.exitCode === 'number' ? { exitCode: row.exitCode } : {}),
692
+ ...(typeof row.signal === 'number' ? { signal: row.signal } : {}),
693
+ };
694
+ }
695
+ /**
696
+ * The SDK's three-way job status, from the guest's two-way state plus the
697
+ * signal. A program the kernel stopped is `killed`, not `exited`: the
698
+ * distinction is the whole reason `BackgroundJobStatus` has three members.
699
+ */
700
+ function sessionStatus(summary) {
701
+ if (summary.state === 'running')
702
+ return 'running';
703
+ return summary.signal !== undefined ? 'killed' : 'exited';
704
+ }
705
+ /** The refusal shape every session op answers a bad request with. */
706
+ function sessionRefusal(sessionId, reply, operation, cause) {
707
+ const code = typeof reply.error === 'string' ? reply.error : 'unknown';
708
+ const detail = typeof reply.message === 'string' ? ` ${reply.message}` : '';
709
+ return new KubernetesSessionRefusedError(sessionId, isSessionRefusal(code) ? code : 'unknown', `kubernetes: the guest refused ${operation} for session ${sessionId} (${code}).${detail} A session lives in the pod's memory only: it is lost when the pod is replaced, and an exited one is kept for the window NAMZU_AGENT_SESSION_TERMINAL_TTL_MS names.`, cause !== undefined ? { cause } : undefined);
710
+ }
711
+ /**
712
+ * The same refusal, arriving on a STREAM rather than in a reply.
713
+ *
714
+ * `openTerminal` and `attachSession` do not get a `{ ok: false }` body: the
715
+ * guest refuses them with an `error` FRAME, which the vsock transport
716
+ * surfaces as a plain `Error` carrying the guest's code as its message. Left
717
+ * alone, those two would be the only session verbs a caller could not catch
718
+ * by class — so the code is recognised here, at the one boundary where it is
719
+ * still recognisable, and everything else (a dial failure, an idle timeout)
720
+ * is handed back untouched.
721
+ */
722
+ function sessionStreamFailure(sessionId, operation, error) {
723
+ if (!(error instanceof Error) || !isSessionRefusal(error.message))
724
+ return error;
725
+ return sessionRefusal(sessionId, { error: error.message }, operation, error);
726
+ }
68
727
  /**
69
728
  * The kubernetes backend's dialable transport: a `tcp` handle plus a
70
729
  * `RemoteExecutionAdapter` built from it, so `exec()` gets the same
@@ -73,27 +732,486 @@ async function requestChecked(wire, request, signal) {
73
732
  * remote backend gets, while every network operation is delegated to
74
733
  * {@link VsockAgentTransport} for the actual dial/frame/token work.
75
734
  */
735
+ // --- quiesce (#485) -------------------------------------------------------
736
+ /**
737
+ * Thrown when a quiesce was asked of a guest that cannot perform one:
738
+ * either its `healthz` does not advertise {@link QUIESCE_FEATURE}, or it
739
+ * answered `unknown_op: quiesce`.
740
+ *
741
+ * Refused rather than treated as "nothing was running". A caller asks for a
742
+ * quiesce because it is about to read the disk and needs it still; an image
743
+ * that cannot stop its processes has to say so, not resolve with an empty
744
+ * list that reads exactly like a guest which had nothing to stop.
745
+ */
746
+ export class KubernetesQuiesceUnsupportedError extends Error {
747
+ feature;
748
+ name = 'KubernetesQuiesceUnsupportedError';
749
+ constructor(feature, message) {
750
+ super(message);
751
+ this.feature = feature;
752
+ }
753
+ }
754
+ /**
755
+ * Thrown when nothing can promise the guest is quiet.
756
+ *
757
+ * Usually because the guest ANSWERED and said so — a process survived
758
+ * `SIGKILL`, the scan could not be performed, or the op ran out of its own
759
+ * deadline — and then the guest's own message names the pid. The workspace
760
+ * handle raises the same class for the one case that never reaches a guest:
761
+ * a `suspend({ quiesce: true })` arriving while a suspend WITHOUT a quiesce
762
+ * is already in flight, whose patch has gone over processes nobody stopped
763
+ * (`reason: 'suspend_already_in_flight'`). Both mean the one thing a caller
764
+ * has to act on: do not trust a capture taken now.
765
+ *
766
+ * Its own class for the same reason {@link KubernetesSessionRefusedError}
767
+ * is: this is an answer, not a transport failure, and retrying it is a
768
+ * decision the caller makes with the pid in hand rather than one a wrapper
769
+ * makes on its behalf.
770
+ */
771
+ export class KubernetesQuiesceUnconfirmedError extends Error {
772
+ reason;
773
+ name = 'KubernetesQuiesceUnconfirmedError';
774
+ constructor(reason, message) {
775
+ super(message);
776
+ this.reason = reason;
777
+ }
778
+ }
779
+ function quiesceReport(reply) {
780
+ const stopped = Array.isArray(reply.stopped) ? reply.stopped : [];
781
+ return {
782
+ stopped: stopped.flatMap((entry) => {
783
+ if (!entry || typeof entry !== 'object')
784
+ return [];
785
+ const row = entry;
786
+ if (typeof row.pid !== 'number')
787
+ return [];
788
+ return [
789
+ {
790
+ pid: row.pid,
791
+ command: typeof row.command === 'string' ? row.command : '',
792
+ signal: row.signal === 'SIGKILL' ? 'SIGKILL' : 'SIGTERM',
793
+ },
794
+ ];
795
+ }),
796
+ // The WIDER claim has to be said in so many words. A reply carrying
797
+ // a scope this host does not recognise reads as the narrow one,
798
+ // because the only wrong answer here is telling a caller its guest
799
+ // was swept completely when nothing says so.
800
+ scope: reply.scope === 'pid-namespace' ? 'pid-namespace' : 'owned-sessions',
801
+ graceMs: typeof reply.graceMs === 'number' ? reply.graceMs : 0,
802
+ rounds: typeof reply.rounds === 'number' ? reply.rounds : 0,
803
+ };
804
+ }
805
+ // --- flush (#484) ---------------------------------------------------------
806
+ /**
807
+ * Thrown when a flush was asked of a guest that cannot perform one: either
808
+ * its `healthz` does not advertise {@link FLUSH_FEATURE}, or it answered
809
+ * `unknown_op: flush`.
810
+ *
811
+ * Refused rather than treated as "the disk is already flushed", for the
812
+ * same reason {@link KubernetesQuiesceUnsupportedError} is not treated as
813
+ * "nothing was running". The one caller that does NOT pass this on is
814
+ * `suspend()`, which goes ahead and tells the host through
815
+ * `onFlushUnsupported`: an image built before this op is a deployment that
816
+ * has to be able to suspend its workspaces, not one that has to be stopped.
817
+ */
818
+ export class KubernetesFlushUnsupportedError extends Error {
819
+ feature;
820
+ name = 'KubernetesFlushUnsupportedError';
821
+ constructor(feature, message) {
822
+ super(message);
823
+ this.feature = feature;
824
+ }
825
+ }
826
+ /**
827
+ * Thrown when nothing can promise the workspace's writes are on the device.
828
+ *
829
+ * The guest ANSWERED and said so — the `syncfs` failed, or ran past its own
830
+ * timeout — and its message says which. Its own class for the reason every
831
+ * other answered refusal here has one: this is a fact about the disk, not a
832
+ * transport failure, and a caller holding it decides whether to retry,
833
+ * suspend anyway, or leave the workspace running.
834
+ */
835
+ export class KubernetesFlushUnconfirmedError extends Error {
836
+ reason;
837
+ name = 'KubernetesFlushUnconfirmedError';
838
+ constructor(reason, message) {
839
+ super(message);
840
+ this.reason = reason;
841
+ }
842
+ }
843
+ /**
844
+ * Carried to {@link KubernetesWorkspaceOptions.onFlushUnreachable} when a
845
+ * `suspend()` could not ASK for a flush at all — the dial failed, the
846
+ * connection timed out, the token was refused, or the agent has fenced
847
+ * itself — and the suspend went ahead without one.
848
+ *
849
+ * Never thrown at a caller. The difference between this and
850
+ * {@link KubernetesFlushUnconfirmedError} is the difference between a guest
851
+ * that could not be reached and a guest that answered: a guest that
852
+ * answered is alive, and stopping the suspend gives its caller something to
853
+ * do about it, while a guest nothing can reach will not become flushable by
854
+ * leaving its pod running — and refusing to suspend over it would take away
855
+ * the one verb an operator reaches for when a workspace is wedged, the verb
856
+ * {@link KubernetesAgentRetiringError} itself names as the way out.
857
+ *
858
+ * So the suspend proceeds, the host is told, and the message says what the
859
+ * disk is resting on instead: whatever the guest kernel had already written
860
+ * back, plus the pod's own `preStop` hook and the agent's `SIGTERM` handler
861
+ * if either of them still runs.
862
+ */
863
+ export class KubernetesFlushUnreachableError extends Error {
864
+ name = 'KubernetesFlushUnreachableError';
865
+ }
866
+ /**
867
+ * The refusal a guest that cannot flush earns, built in one place.
868
+ *
869
+ * Exported because the WORKSPACE has to be able to hand this exact error to
870
+ * `onFlushUnsupported` on the one path that does not throw it — a
871
+ * `suspend()` against an older image — and a second copy of the message
872
+ * would be a second thing to keep true.
873
+ */
874
+ export function flushUnsupportedError() {
875
+ return new KubernetesFlushUnsupportedError(FLUSH_FEATURE, `kubernetes: this workspace's guest agent does not advertise the '${FLUSH_FEATURE}' healthz feature, so nothing here can make its writes reach the disk before the pod stops — what survives is whatever the guest kernel had already written back. The request is refused rather than answered as a flush that happened. Rebuild the workspace image from this Namzu release.`);
876
+ }
877
+ function flushReport(reply) {
878
+ return {
879
+ durationMs: typeof reply.durationMs === 'number' ? reply.durationMs : 0,
880
+ workspace: typeof reply.workspace === 'string' ? reply.workspace : '',
881
+ };
882
+ }
76
883
  export class KubernetesAgentTransport {
884
+ /**
885
+ * Mutable: the handle follows a replaced pod — see {@link rebind}.
886
+ *
887
+ * Under `'pod-ip'` both the address and the token move, because the
888
+ * address is a literal that died with its pod. Under the default
889
+ * `'service'` mode the host is a Service FQDN that outlives the pod and
890
+ * resolves to the replacement on its own, so what moves is the TOKEN
891
+ * alone — which is the entire reason a `service` handle needed a rebind
892
+ * at all: the dial keeps working and the guest refuses every call.
893
+ */
77
894
  handle;
895
+ /** The re-read currently in flight, so concurrent refusals share one. */
896
+ rebinding;
897
+ /**
898
+ * How many times this transport has moved to a different pod. Captured
899
+ * before every attempt and compared after it fails, so a call refused by
900
+ * the pod it was bound to — arriving after ANOTHER call's rebind already
901
+ * installed the replacement — retries on the handle that has moved
902
+ * instead of re-reading to be told nothing changed.
903
+ */
904
+ rebindSeq = 0;
78
905
  transportOptions;
906
+ /** Bound to each wire's own pod by {@link wireOptions}. */
907
+ onGuestReply;
79
908
  onTiming;
909
+ refreshHandle;
80
910
  /** Simple pass-through operations share one transport instance. */
81
911
  wire;
82
912
  constructor(handle, options = {}) {
83
- const { onTiming, ...transportOptions } = options;
913
+ const { onTiming, refreshHandle, onGuestReply, ...transportOptions } = options;
84
914
  this.handle = handle;
85
- this.transportOptions = transportOptions;
915
+ // Held apart from the options the wires are built from: it is the one
916
+ // hook whose payload depends on WHICH wire read the reply, so
917
+ // {@link wireOptions} binds it per wire rather than spreading it.
918
+ this.onGuestReply = onGuestReply;
919
+ this.transportOptions = {
920
+ ...transportOptions,
921
+ // A NAME the resolver says does not EXIST is the one connect
922
+ // failure waiting cannot cure, and the reason it must not be
923
+ // waited on is that the wait destroys the diagnosis. A resolver
924
+ // that merely could not answer (`EAI_AGAIN`) is not that — see
925
+ // {@link isMissingNameFailure} — and keeps the whole budget.
926
+ //
927
+ // The dial's retry budget is 30 s and the privilege probe's
928
+ // deadline is at most 15 s, so a host with no cluster resolver
929
+ // never reaches the wrapper below: the probe's clock expires first
930
+ // and `create()` rejects saying the guest "accepted the connection
931
+ // and did not answer", about a connection that was never made. The
932
+ // address here comes off a Sandbox that reported Ready, so its
933
+ // Service exists and its record is published — a name that fails
934
+ // to resolve against that is a fact about THIS host, not a race
935
+ // with the controller.
936
+ permanentDialFailure: (err) => this.dialFailureIsPermanent(err),
937
+ };
86
938
  this.onTiming = onTiming;
87
- this.wire = new VsockAgentTransport(handle, transportOptions);
939
+ this.refreshHandle = refreshHandle;
940
+ this.wire = new VsockAgentTransport(handle, this.wireOptions(handle));
88
941
  }
89
- /** Readiness probe — never requires a token; see `protocol.ts`. */
942
+ /**
943
+ * The shared transport's options for ONE wire, with the reply observer
944
+ * bound to the pod that wire is talking to.
945
+ *
946
+ * Every `VsockAgentTransport` this class builds goes through here — the
947
+ * first one, the one a rebind installs, and the per-attempt one `exec()`
948
+ * times — because a wire outlives the moment it was current: the pod it
949
+ * dials can answer a call after a rebind has already moved
950
+ * {@link handle} on, and the observer has to be told the pod that
951
+ * ANSWERED rather than the pod this transport now holds. Capturing the
952
+ * handle in the closure is what makes the two different values.
953
+ */
954
+ wireOptions(handle) {
955
+ const observe = this.onGuestReply;
956
+ if (observe === undefined)
957
+ return this.transportOptions;
958
+ return {
959
+ ...this.transportOptions,
960
+ onGuestReply: (reply) => {
961
+ observe(reply, handle.token);
962
+ },
963
+ };
964
+ }
965
+ /**
966
+ * The address this transport is dialing right now. Diagnostics only —
967
+ * host and port, and deliberately NOT the handle itself: this is read
968
+ * into a log line or an assertion, and the bind token has no business
969
+ * travelling with either.
970
+ */
971
+ get address() {
972
+ return { host: this.handle.host, port: this.handle.port };
973
+ }
974
+ /**
975
+ * Run one operation, and give the handle exactly one chance to follow a
976
+ * pod that was replaced underneath it.
977
+ *
978
+ * TWO failures reach a rebind, and they are the only two, because they
979
+ * are the only two that prove the operation ran NOTHING in the guest:
980
+ *
981
+ * - **the dial failed.** No socket was ever established, so no byte was
982
+ * sent. This is the trigger the `'pod-ip'` mode was built on: the
983
+ * address is a literal that dies with its pod.
984
+ * - **the guest REFUSED the token** (`unauthorized`, in either of the
985
+ * shapes {@link isUnauthorizedRefusal} unifies). The agent checks the
986
+ * token before `dispatch`, so a refused request reached the guest and
987
+ * was thrown away unread. This is the trigger that matters under the
988
+ * DEFAULT `'service'` mode, where the Service FQDN outlives the pod
989
+ * and keeps resolving: the dial SUCCEEDS against the replacement and
990
+ * the refusal is the only thing that says the pod moved.
991
+ *
992
+ * Everything else is the guest's answer to a request it did receive, and
993
+ * nothing here may reinterpret it — not as a resolver problem, and not as
994
+ * a reason to repeat work the guest has already begun.
995
+ *
996
+ * From there both arms run the SAME routine, which is the whole of this
997
+ * change to it: {@link rebind} reads the pod once, and a DIFFERENT uid
998
+ * means the controller replaced it (a resume, an eviction, a node drain),
999
+ * so the handle takes the new address AND the new token together and the
1000
+ * operation is retried once.
1001
+ *
1002
+ * One branch belongs to the dial arm alone, and is taken before the
1003
+ * re-read: the handle's host is a NAME and the dial gave up at
1004
+ * resolution, so this is a Service FQDN and this host has no resolver for
1005
+ * it. Re-reading the pod would change nothing — the next dial would ask
1006
+ * the same resolver the same question — so the error is replaced with one
1007
+ * that names the FQDN and the configuration field that fixes it. It is
1008
+ * never asked of a refusal, which came back over a connection that
1009
+ * plainly worked, and the name check is not decoration either: a
1010
+ * `pod-ip` handle must keep its one re-read however an unrelated error
1011
+ * happens to be worded.
1012
+ *
1013
+ * Anything else — a re-read that finds the SAME pod, or one that fails —
1014
+ * leaves the original error standing. A pod that is still there and still
1015
+ * refusing is a guest problem, and replacing that error with a second,
1016
+ * later one would hide it. The one exception is a re-read that comes back
1017
+ * with a verdict about the OBJECT rather than a failed diagnosis
1018
+ * ({@link KubernetesWorkspaceReplacedError}): the disk behind the name is
1019
+ * not this handle's disk, which outranks whatever uncovered it.
1020
+ *
1021
+ * `signal` is the caller's own, and it decides one thing only: a call that
1022
+ * has ALREADY been cancelled is not owed a re-read, because the retry it
1023
+ * would buy would abort before it left. It is never handed to the re-read
1024
+ * itself, which is shared and runs under a bound of its own — see
1025
+ * {@link rebind}.
1026
+ *
1027
+ * `dials` is how `exec()` answers the first question at all. Its failure
1028
+ * can arrive as a bare timeout from the execution controller's own bound,
1029
+ * with the dial's error discarded rather than wrapped, so that path
1030
+ * watches its dials instead of reading its error — see {@link DialWatch}.
1031
+ * Every other operation hands back the dial's own error and passes none.
1032
+ *
1033
+ * A refusal that no rebind could fix leaves as {@link
1034
+ * KubernetesAgentUnauthorizedError} whichever shape it arrived in — the
1035
+ * unification the whole recovery hangs off, applied at the one point
1036
+ * where "nothing else can be done about it" is known.
1037
+ */
1038
+ async withRebind(run, signal, dials) {
1039
+ // Read BEFORE the attempt: everything below asks whether the handle
1040
+ // has moved SINCE this call was dispatched.
1041
+ const boundAt = this.rebindSeq;
1042
+ try {
1043
+ return await run();
1044
+ }
1045
+ catch (err) {
1046
+ if (isUnretryableOutcome(err))
1047
+ throw err;
1048
+ const refused = isUnauthorizedRefusal(err);
1049
+ // A refusal came back over a connection that was made, so it is
1050
+ // never also a dial failure; asking both questions of one error
1051
+ // and letting the dial arm win would send it to the resolver
1052
+ // branch, which is about a connection that was never attempted.
1053
+ const fromTheDial = !refused && (isConnectFailure(err) || neverConnected(dials));
1054
+ if (!refused && !fromTheDial)
1055
+ throw err;
1056
+ if (fromTheDial && this.dialsAName() && this.failedToResolve(err, dials)) {
1057
+ throw this.unresolvable(err);
1058
+ }
1059
+ // See above: nothing to buy with a re-read here, and the shared
1060
+ // one must not be started on behalf of a call that is gone.
1061
+ if (signal?.aborted === true)
1062
+ throw refused ? asUnauthorized(err) : err;
1063
+ if (!(await this.rebind(boundAt)))
1064
+ throw refused ? asUnauthorized(err) : err;
1065
+ try {
1066
+ return await run();
1067
+ }
1068
+ catch (retryErr) {
1069
+ // The retry is the last attempt either way; a second refusal
1070
+ // (the replacement is refusing too) still leaves as one shape.
1071
+ throw refused && isUnauthorizedRefusal(retryErr) ? asUnauthorized(retryErr) : retryErr;
1072
+ }
1073
+ }
1074
+ }
1075
+ /**
1076
+ * Whether the dial gave up at name resolution.
1077
+ *
1078
+ * Asked of the caller's error first, and of the watched dial's own error
1079
+ * only when nothing ever connected — the case where the error the caller
1080
+ * holds is a bound's timer rather than the failure that caused it. A
1081
+ * watched attempt that DID connect is never consulted: its error is the
1082
+ * guest's answer, however it happens to be worded.
1083
+ */
1084
+ failedToResolve(error, dials) {
1085
+ if (isNameResolutionFailure(error))
1086
+ return true;
1087
+ return neverConnected(dials) && isNameResolutionFailure(dials?.lastError);
1088
+ }
1089
+ /**
1090
+ * The one connect failure waiting cannot cure — see the constructor.
1091
+ * A method rather than a closure because the watched `exec()` dials wrap
1092
+ * it, and a caller-visible answer must not depend on which wire asked.
1093
+ */
1094
+ dialFailureIsPermanent(error) {
1095
+ return this.dialsAName() && isMissingNameFailure(error);
1096
+ }
1097
+ /**
1098
+ * One re-read, shared by every call that needs one at the same moment.
1099
+ * True when the handle now points at a DIFFERENT pod than the one the
1100
+ * caller's attempt was dispatched on — whether this call's own re-read
1101
+ * moved it or another call's already had.
1102
+ *
1103
+ * Single-flight, and that is not an optimisation. A pod replaced
1104
+ * underneath a busy handle refuses EVERY call in flight at once, and one
1105
+ * re-read per refused call would be a burst of Sandbox and pod GETs, each
1106
+ * one racing the others to install a handle — with the last to finish
1107
+ * winning, which is not necessarily the last to read. Sharing the promise
1108
+ * makes the burst one round trip and one installation, in the order the
1109
+ * API answered.
1110
+ *
1111
+ * The slot is cleared inside the shared run, before the promise settles,
1112
+ * so a caller that awaits it and is refused AGAIN gets a fresh re-read
1113
+ * rather than the answer to the previous question.
1114
+ *
1115
+ * Because it is shared it runs under {@link REBIND_READ_TIMEOUT_MS} and
1116
+ * under NO caller's signal. A caller that aborts while the shared read is
1117
+ * in flight would otherwise abort it for everyone, and every call the
1118
+ * replacement refused would fail with its original refusal although the
1119
+ * handle could have followed the pod. Its own bound is what the aborting
1120
+ * caller was owed — nobody waits on a re-read for longer than that — and
1121
+ * the read is two GETs nobody is billed for twice.
1122
+ */
1123
+ async rebind(boundAt) {
1124
+ // Another call already followed the replacement while this one was in
1125
+ // flight, so the pod that refused this call is the pod it was bound
1126
+ // to and the answer a re-read would give is already installed. Asking
1127
+ // again would compare the new token against itself, conclude nothing
1128
+ // moved, and fail a call the handle can now serve.
1129
+ if (boundAt !== undefined && this.rebindSeq !== boundAt)
1130
+ return true;
1131
+ const refresh = this.refreshHandle;
1132
+ if (refresh === undefined)
1133
+ return false;
1134
+ this.rebinding ??= this.runRebind(refresh, AbortSignal.timeout(REBIND_READ_TIMEOUT_MS));
1135
+ return await this.rebinding;
1136
+ }
1137
+ async runRebind(refresh, signal) {
1138
+ try {
1139
+ let next;
1140
+ try {
1141
+ next = await refresh(signal);
1142
+ }
1143
+ catch (err) {
1144
+ // A verdict about the OBJECT is not a failed diagnosis and
1145
+ // must not be swallowed: it says the disk behind the name is
1146
+ // not this handle's disk, which is a far more important thing
1147
+ // to report than the refusal that uncovered it — and it is the
1148
+ // one answer that must never be followed by a retry.
1149
+ if (err instanceof KubernetesWorkspaceReplacedError)
1150
+ throw err;
1151
+ // Everything else: the re-read is a diagnosis, not an
1152
+ // operation. A pod that cannot be read is not a better error
1153
+ // than the failure the caller is already holding.
1154
+ return false;
1155
+ }
1156
+ if (next.token === this.handle.token)
1157
+ return false;
1158
+ this.rebindSeq += 1;
1159
+ this.handle = next;
1160
+ this.wire = new VsockAgentTransport(next, this.wireOptions(next));
1161
+ return true;
1162
+ }
1163
+ finally {
1164
+ this.rebinding = undefined;
1165
+ }
1166
+ }
1167
+ /** Whether the current handle's host goes through a resolver at all. */
1168
+ dialsAName() {
1169
+ return net.isIP(this.handle.host) === 0;
1170
+ }
1171
+ /** The DNS-shaped failure, in words that name the way out of it. */
1172
+ unresolvable(cause) {
1173
+ const host = this.handle.host;
1174
+ return new KubernetesAgentAddressUnresolvableError(host, `kubernetes: the guest agent's address ${host}:${this.handle.port} did not resolve (ENOTFOUND/EAI_AGAIN), so no connection was attempted. That is a Kubernetes Service FQDN and only the cluster's own DNS answers it: a host running OUTSIDE the cluster — a VNet peer, a CI runner, a laptop — fails every call here, readiness probes included, and the symptom looks like a sandbox that never came up. Set agentAddress: 'pod-ip' on the kubernetes backend config to dial the bound pod's IP instead, which needs a pod network routable from this host and a NetworkPolicy admitting its address range.`, { cause });
1175
+ }
1176
+ /**
1177
+ * Readiness probe — never requires a token; see `protocol.ts`.
1178
+ *
1179
+ * Deliberately NOT wrapped in {@link withRebind}: `healthz` answers a
1180
+ * failed dial with `false` rather than by throwing, so there is no error
1181
+ * to classify and nothing for a re-read to be triggered by. A caller that
1182
+ * wants the reason asks for it by making a real call.
1183
+ */
90
1184
  async healthz(signal) {
91
1185
  return await this.wire.healthz(signal);
92
1186
  }
93
- /** Poll until `healthz` succeeds or the timeout elapses. */
1187
+ /**
1188
+ * Poll until `healthz` succeeds or the timeout elapses. Unwrapped for the
1189
+ * same reason, and for one more: it already owns a retry loop, so a
1190
+ * connect failure here is not a single failed dial but a whole budget of
1191
+ * them.
1192
+ */
94
1193
  async waitForReady(timeoutMs, pollIntervalMs, signal) {
95
1194
  return await this.wire.waitForReady(timeoutMs, pollIntervalMs, signal);
96
1195
  }
1196
+ /**
1197
+ * Ask the agent how it is, and keep the two "not ok" answers apart —
1198
+ * see {@link KubernetesAgentHealth}.
1199
+ *
1200
+ * Deliberately NOT wrapped in {@link withRebind}: this is a diagnostic
1201
+ * about the pod this handle is bound to RIGHT NOW, and a rebind would
1202
+ * silently answer it about a different pod. A caller that wants to know
1203
+ * whether the agent it was talking to has fenced itself would then be
1204
+ * told about the replacement, which is a different question with a
1205
+ * different answer.
1206
+ *
1207
+ * It throws whatever the dial or the read threw. An agent that cannot be
1208
+ * reached has no health to report, and inventing one here would turn
1209
+ * "unreachable" into "fine".
1210
+ */
1211
+ async agentHealth(signal) {
1212
+ const reply = await this.wire.request({ op: 'healthz' }, signal);
1213
+ return { ok: reply?.ok === true, retiring: reply?.retiring === true };
1214
+ }
97
1215
  /**
98
1216
  * The raw `reserve-execution` primitive, exposed directly (rather than
99
1217
  * only reachable as a side effect of `exec()`) so the reservation
@@ -101,7 +1219,7 @@ export class KubernetesAgentTransport {
101
1219
  * real guest.
102
1220
  */
103
1221
  async reserve(signal) {
104
- return await requestChecked(this.wire, { op: 'reserve-execution' }, signal);
1222
+ return await this.withRebind(async () => await requestChecked(this.wire, { op: 'reserve-execution' }, signal), signal);
105
1223
  }
106
1224
  /**
107
1225
  * The raw `cancel-execution` primitive, exposed for the same reason
@@ -110,19 +1228,373 @@ export class KubernetesAgentTransport {
110
1228
  * driving a whole cancelled `exec()` to reach it.
111
1229
  */
112
1230
  async cancel(executionId, signal) {
113
- return await requestChecked(this.wire, { op: 'cancel-execution', body: { executionId } }, signal);
1231
+ return await this.withRebind(async () => await requestChecked(this.wire, { op: 'cancel-execution', body: { executionId } }, signal), signal);
114
1232
  }
115
- async writeFile(path, content) {
116
- return await this.wire.writeFile(path, content);
1233
+ /**
1234
+ * Delegated, `signal` included: a body larger than one pre-auth frame
1235
+ * is written as a sequence of parts by {@link VsockAgentTransport}
1236
+ * itself, and a cancelled sequence has to be able to stop mid-way and
1237
+ * take its temp file with it.
1238
+ *
1239
+ * One transport instance, deliberately — {@link wire} is shared by
1240
+ * every simple pass-through op — so the one `healthz` probe that asks
1241
+ * the guest whether it can take parts is asked once for this sandbox,
1242
+ * not once per large write.
1243
+ *
1244
+ * Wrapped in {@link withRebind} like every other guest-dialling op,
1245
+ * the multi-part route included, and for both of its triggers: a retry
1246
+ * starts a fresh sequence under a new temp name
1247
+ * (`writeFilePartTempPath` mints a UUID per call), so a sequence
1248
+ * abandoned mid-way on the replaced pod — because the dial failed, or
1249
+ * because the replacement refused this handle's token before reading a
1250
+ * byte — cannot collide with the retry's offsets and never touched the
1251
+ * target. At worst it leaves one orphan temp file behind on the
1252
+ * workspace volume.
1253
+ */
1254
+ async writeFile(path, content, signal) {
1255
+ return await this.withRebind(async () => await this.wire.writeFile(path, content, signal), signal);
1256
+ }
1257
+ async readFile(path, options) {
1258
+ return await this.withRebind(async () => await this.wire.readFile(path, options), options?.signal);
117
1259
  }
118
- async readFile(path) {
119
- return await this.wire.readFile(path);
1260
+ /**
1261
+ * Delegated, and rebound exactly once — but only around the FIRST
1262
+ * chunk.
1263
+ *
1264
+ * That is the whole of what {@link withRebind} can honestly cover here.
1265
+ * Its retry is safe because nothing reached the guest, and once a chunk
1266
+ * has been yielded that is no longer true: re-dialing a replaced pod
1267
+ * mid-stream would restart the file from its beginning, and the
1268
+ * consumer — which has already taken the bytes and cannot give them
1269
+ * back — would silently concatenate a duplicate prefix. So the first
1270
+ * pull carries the dial, the rebind and the retry; everything after it
1271
+ * fails as itself.
1272
+ */
1273
+ async *readFileStream(path, options) {
1274
+ const started = await this.withRebind(async () => {
1275
+ const iterator = this.wire.readFileStream(path, options)[Symbol.asyncIterator]();
1276
+ try {
1277
+ return { iterator, first: await iterator.next() };
1278
+ }
1279
+ catch (error) {
1280
+ // The abandoned generator's own `finally` has already run by
1281
+ // the time its `next()` rejects, so the socket is down; the
1282
+ // `return()` is belt and braces for an implementation that
1283
+ // rejected without finishing.
1284
+ await iterator.return?.(undefined).catch(() => undefined);
1285
+ throw error;
1286
+ }
1287
+ }, options?.signal);
1288
+ const { iterator, first } = started;
1289
+ try {
1290
+ if (first.done === true)
1291
+ return;
1292
+ yield first.value;
1293
+ for (;;) {
1294
+ const next = await iterator.next();
1295
+ if (next.done === true)
1296
+ return;
1297
+ yield next.value;
1298
+ }
1299
+ }
1300
+ finally {
1301
+ await iterator.return?.(undefined).catch(() => undefined);
1302
+ }
120
1303
  }
1304
+ /**
1305
+ * A guest PTY. Without `sessionId`/`persistent` this is exactly the
1306
+ * terminal it has always been, down to the wire request.
1307
+ *
1308
+ * With them the PTY belongs to the guest's session registry: losing this
1309
+ * connection detaches rather than killing, a later process rejoins it
1310
+ * with {@link attachSession}, and the capability is verified against the
1311
+ * guest's `healthz` features BEFORE the shell is started — never
1312
+ * downgraded to a connection-bound terminal, which would look like it
1313
+ * worked until the rollout it exists for.
1314
+ */
121
1315
  async openTerminal(options) {
122
- return await this.wire.openTerminal(options);
1316
+ if (options.sessionId === undefined && options.persistent !== true) {
1317
+ return await this.withRebind(async () => await this.wire.openTerminal(options));
1318
+ }
1319
+ const sessionId = assertSessionOpen(options);
1320
+ await this.assertSessionsSupported();
1321
+ return await this.sessionStream(sessionId, 'openTerminal', async () => await this.wire.openSessionTerminal(options));
1322
+ }
1323
+ /**
1324
+ * One open of a session stream, with the guest's refusal mapped to
1325
+ * {@link KubernetesSessionRefusedError} — see {@link sessionStreamFailure}.
1326
+ */
1327
+ async sessionStream(sessionId, operation, open) {
1328
+ try {
1329
+ return sessionTerminal(await this.withRebind(open), sessionId);
1330
+ }
1331
+ catch (error) {
1332
+ throw sessionStreamFailure(sessionId, operation, error);
1333
+ }
1334
+ }
1335
+ /**
1336
+ * Rejoin a terminal session, replaying from `fromOffset` and then
1337
+ * following it live.
1338
+ *
1339
+ * The guest allows one attachment per session and ends the previous one
1340
+ * by name, so two host processes cannot interleave keystrokes into one
1341
+ * shell. Losing this connection detaches; ending the program is
1342
+ * {@link killSession} and nothing else.
1343
+ */
1344
+ async attachSession(sessionId, options = {}) {
1345
+ assertSessionId(sessionId);
1346
+ await this.assertSessionsSupported();
1347
+ return await this.sessionStream(sessionId, 'attachSession', async () => await this.wire.attachSessionTerminal({
1348
+ sessionId,
1349
+ ...(options.fromOffset !== undefined ? { fromOffset: options.fromOffset } : {}),
1350
+ ...(options.size !== undefined
1351
+ ? { cols: options.size.cols, rows: options.size.rows }
1352
+ : {}),
1353
+ }));
1354
+ }
1355
+ /**
1356
+ * Start a program with no terminal at all, in its own kernel session,
1357
+ * with stdin closed and both output streams going into the guest's
1358
+ * retained log.
1359
+ *
1360
+ * It is not the SDK's `spawnDetached` and deliberately does not pretend
1361
+ * to be: that one hands back a host `ChildProcess`, which cannot cross a
1362
+ * process boundary. This returns a NAME, and the name is what a
1363
+ * redeployed host comes back with.
1364
+ */
1365
+ async startDetached(options, signal) {
1366
+ assertSessionId(options.sessionId);
1367
+ await this.assertSessionsSupported(signal);
1368
+ const reply = (await this.withRebind(async () => await requestChecked(this.wire, {
1369
+ op: 'start-detached',
1370
+ body: {
1371
+ sessionId: options.sessionId,
1372
+ command: options.command,
1373
+ ...(options.args !== undefined ? { args: options.args } : {}),
1374
+ ...(options.cwd !== undefined ? { cwd: options.cwd } : {}),
1375
+ ...(options.env !== undefined ? { env: { ...options.env } } : {}),
1376
+ },
1377
+ }, signal), signal));
1378
+ if (reply.ok !== true)
1379
+ throw sessionRefusal(options.sessionId, reply, 'startDetached');
1380
+ return parseSessionSummary(reply);
1381
+ }
1382
+ /** Every session this pod's agent is holding, running and recently exited. */
1383
+ async listSessions(signal) {
1384
+ await this.assertSessionsSupported(signal);
1385
+ const reply = (await this.withRebind(async () => await requestChecked(this.wire, { op: 'list-sessions' }, signal), signal));
1386
+ if (reply.ok !== true) {
1387
+ throw new RemoteProtocolError(`kubernetes: the guest refused to list sessions: ${String(reply.error ?? 'no reason given')}`);
1388
+ }
1389
+ if (!Array.isArray(reply.sessions)) {
1390
+ throw new RemoteProtocolError('kubernetes: the guest sent a session list that is not an array');
1391
+ }
1392
+ return reply.sessions.map(parseSessionSummary);
1393
+ }
1394
+ /**
1395
+ * End one session and everything still in it — the shell, its
1396
+ * backgrounded jobs, and the program a detached session started.
1397
+ *
1398
+ * Idempotent: a session that has already exited answers with what it
1399
+ * exited with. The reply carries the session's state, so a program that
1400
+ * ignored a `SIGTERM` and outlived the guest's confirm window is
1401
+ * reported still running rather than reported dead.
1402
+ */
1403
+ async killSession(sessionId, options = {}) {
1404
+ assertSessionId(sessionId);
1405
+ await this.assertSessionsSupported(options.abort);
1406
+ const reply = (await this.withRebind(async () => await requestChecked(this.wire, {
1407
+ op: 'kill-session',
1408
+ body: {
1409
+ sessionId,
1410
+ ...(options.signal !== undefined ? { signal: options.signal } : {}),
1411
+ },
1412
+ }, options.abort), options.abort));
1413
+ if (reply.ok !== true)
1414
+ throw sessionRefusal(sessionId, reply, 'killSession');
1415
+ return parseSessionSummary(reply);
1416
+ }
1417
+ /**
1418
+ * Read what a session has printed since `fromOffset`, without attaching
1419
+ * to it and without signalling anything.
1420
+ *
1421
+ * One request, one answer, in the SDK's `BackgroundJobOutput` shape: the
1422
+ * chunk, the offset to come back with, the bytes the ring dropped before
1423
+ * it, and the program's status. A caller polling in a loop can neither
1424
+ * re-read nor skip, because the offset is the guest's own.
1425
+ */
1426
+ async readSession(sessionId, options = {}, signal) {
1427
+ assertSessionId(sessionId);
1428
+ await this.assertSessionsSupported(signal);
1429
+ const fromOffset = options.fromOffset ?? 0;
1430
+ let chunk = '';
1431
+ let nextOffset = fromOffset;
1432
+ let droppedBytes = 0;
1433
+ let state = 'running';
1434
+ let exitCode;
1435
+ let exitSignal;
1436
+ let refusal;
1437
+ // Accumulating INSIDE a `withRebind` is safe for the one reason that
1438
+ // matters: the wrapper retries only a failure that delivered no
1439
+ // session frame. A dial that failed sent nothing at all, and a token
1440
+ // the guest REFUSED is answered before `dispatch` with that refusal
1441
+ // as the connection's first and only frame — which the handler above
1442
+ // turns into an error rather than accumulating. Either way a retry
1443
+ // starts from an untouched chunk and the offset the caller asked
1444
+ // for.
1445
+ await this.withRebind(async () => await this.wire.streamFramedRequest({ op: 'attach-session', body: { sessionId, fromOffset, follow: false } }, (event) => {
1446
+ if (isUnauthorized(event))
1447
+ throw new KubernetesAgentUnauthorizedError();
1448
+ if (event.type === 'ready') {
1449
+ nextOffset = sessionNumber(event.nextOffset, nextOffset);
1450
+ droppedBytes = sessionNumber(event.droppedBytes);
1451
+ state = event.state === 'exited' ? 'exited' : 'running';
1452
+ if (typeof event.exitCode === 'number')
1453
+ exitCode = event.exitCode;
1454
+ if (typeof event.signal === 'number')
1455
+ exitSignal = event.signal;
1456
+ return;
1457
+ }
1458
+ if (event.type === 'data') {
1459
+ chunk += String(event.data ?? '');
1460
+ nextOffset = sessionNumber(event.nextOffset, nextOffset);
1461
+ return;
1462
+ }
1463
+ if (event.type === 'error') {
1464
+ refusal = sessionRefusal(sessionId, {
1465
+ error: event.error,
1466
+ ...(event.message !== undefined ? { message: event.message } : {}),
1467
+ }, 'readSession');
1468
+ return;
1469
+ }
1470
+ throw new RemoteProtocolError(`kubernetes: the guest sent an unexpected frame reading session ${sessionId}: ${JSON.stringify(event).slice(0, 200)}`);
1471
+ }, { observationTimeoutMs: SESSION_READ_TIMEOUT_MS }, signal), signal);
1472
+ if (refusal !== undefined)
1473
+ throw refusal;
1474
+ return {
1475
+ chunk,
1476
+ nextOffset,
1477
+ droppedBytes,
1478
+ status: sessionStatus({ state, ...(exitSignal !== undefined ? { signal: exitSignal } : {}) }),
1479
+ ...(exitCode !== undefined ? { exitCode } : {}),
1480
+ };
1481
+ }
1482
+ /**
1483
+ * Stop every process this pod's guest is running, and keep the agent.
1484
+ *
1485
+ * The point is the pair. Stopping the POD stops its processes too, but
1486
+ * leaves nothing to read the disk through, so a host that wants a capture
1487
+ * it can trust has to wake the workspace again and check. After this the
1488
+ * guest is quiet and still serving: `exec`, `readFile` and `writeFile`
1489
+ * all work, and what they see is a filesystem nobody is writing to.
1490
+ *
1491
+ * Open terminals and running commands end as a side effect and report it
1492
+ * through their own exit and result paths — a terminal receives its exit,
1493
+ * an `exec` resolves with a signal in its result. Rejecting with
1494
+ * {@link KubernetesQuiesceUnconfirmedError} is the honest failure: a
1495
+ * process would not stop, and the reply names its pid.
1496
+ */
1497
+ async quiesce(options = {}, signal) {
1498
+ await this.assertQuiesceSupported(signal);
1499
+ const reply = (await this.withRebind(async () => await requestChecked(this.wire, {
1500
+ op: 'quiesce',
1501
+ body: { ...(options.graceMs !== undefined ? { graceMs: options.graceMs } : {}) },
1502
+ }, signal), signal));
1503
+ if (reply.ok === true)
1504
+ return quiesceReport(reply);
1505
+ const refusal = typeof reply.error === 'string' ? reply.error : 'no reason given';
1506
+ // The guest advertised the feature and then did not know the op. That
1507
+ // is one image, not two, so it is the unsupported error rather than a
1508
+ // second name for the same fact — see {@link assertQuiesceSupported}.
1509
+ if (refusal.startsWith('unknown_op'))
1510
+ throw this.quiesceUnsupported();
1511
+ const detail = typeof reply.message === 'string' ? `: ${reply.message}` : '';
1512
+ throw new KubernetesQuiesceUnconfirmedError(refusal, `kubernetes: the guest could not confirm that every process it is running has stopped (${refusal})${detail}. Nothing on the cluster was changed and the pod is still serving; a capture taken now may not be consistent.`);
1513
+ }
1514
+ /**
1515
+ * Put everything this workspace has written onto its device.
1516
+ *
1517
+ * The disk a workspace keeps is whatever the guest kernel happened to
1518
+ * write back. Nothing in this backend ever asked for more than that: a
1519
+ * `suspend()` patched the pod away and waited for it to stop, and a
1520
+ * stopped pod means only that nothing is writing any more — not that
1521
+ * what was written arrived. This is the call that closes the difference,
1522
+ * and it is `syncfs(2)` over the whole workspace mount, so it covers
1523
+ * what a COMMAND wrote as well as what `writeFile` did (`writeFile`
1524
+ * fsyncs its own bytes before it answers; a compiler's output is nobody's
1525
+ * to fsync).
1526
+ *
1527
+ * Rejecting with {@link KubernetesFlushUnconfirmedError} is the honest
1528
+ * failure, exactly as an unconfirmed quiesce is: the caller is usually
1529
+ * about to take the pod away, and "the flush did not run" has to be
1530
+ * distinguishable from "the flush ran".
1531
+ */
1532
+ async flush(options = {}, signal) {
1533
+ await this.assertFlushSupported(signal);
1534
+ const reply = (await this.withRebind(async () => await requestChecked(this.wire, {
1535
+ op: 'flush',
1536
+ body: { ...(options.timeoutMs !== undefined ? { timeoutMs: options.timeoutMs } : {}) },
1537
+ }, signal), signal));
1538
+ if (reply.ok === true)
1539
+ return flushReport(reply);
1540
+ const refusal = typeof reply.error === 'string' ? reply.error : 'no reason given';
1541
+ // Advertised and then not known: one image, not two — the same
1542
+ // reading {@link quiesce} gives that answer. `flush_unsupported` and
1543
+ // `flush_unsupported_platform` join it because they say the same
1544
+ // thing in the guest's own words: this image cannot perform a flush,
1545
+ // now or ever. Read as an unconfirmed flush instead, they would stop
1546
+ // every default `suspend()` against such an image permanently, which
1547
+ // is exactly what the feature advertisement exists to avoid.
1548
+ if (refusal.startsWith('unknown_op') || refusal.startsWith('flush_unsupported')) {
1549
+ throw this.flushUnsupported();
1550
+ }
1551
+ const detail = typeof reply.message === 'string' ? `: ${reply.message}` : '';
1552
+ throw new KubernetesFlushUnconfirmedError(refusal, `kubernetes: the guest could not confirm that the workspace's writes reached its disk (${refusal})${detail}. Nothing on the cluster was changed and the pod is still serving; suspending or deleting the pod now may lose whatever had not been written back.`);
1553
+ }
1554
+ /**
1555
+ * Whether this guest can flush at all — asked, rather than assumed,
1556
+ * because a `suspend()` against an older image keeps today's behaviour
1557
+ * and reports the gap instead of refusing to suspend.
1558
+ */
1559
+ async supportsFlush(signal) {
1560
+ return (await this.wire.guestFeatures(signal)).includes(FLUSH_FEATURE);
1561
+ }
1562
+ async assertFlushSupported(signal) {
1563
+ if (await this.supportsFlush(signal))
1564
+ return;
1565
+ throw this.flushUnsupported();
1566
+ }
1567
+ flushUnsupported() {
1568
+ return flushUnsupportedError();
1569
+ }
1570
+ /**
1571
+ * Whether this guest can quiesce at all — asked, rather than assumed,
1572
+ * because `suspend({ quiesce: true })` against an older image keeps
1573
+ * today's behaviour and reports the gap instead of refusing to suspend.
1574
+ */
1575
+ async supportsQuiesce(signal) {
1576
+ return (await this.wire.guestFeatures(signal)).includes(QUIESCE_FEATURE);
1577
+ }
1578
+ async assertQuiesceSupported(signal) {
1579
+ if (await this.supportsQuiesce(signal))
1580
+ return;
1581
+ throw this.quiesceUnsupported();
1582
+ }
1583
+ quiesceUnsupported() {
1584
+ return new KubernetesQuiesceUnsupportedError(QUIESCE_FEATURE, `kubernetes: this workspace's guest agent does not advertise the '${QUIESCE_FEATURE}' healthz feature, so nothing here can stop the processes it is running — a terminal another handle opened, a command already in flight, or a program that moved into a session of its own all keep writing. The request is refused rather than answered with an empty list, which would read as a guest that had nothing to stop. Rebuild the workspace image from this Namzu release.`);
1585
+ }
1586
+ /**
1587
+ * Whether this guest keeps a session registry at all, asked once per
1588
+ * transport and only when a caller wants one.
1589
+ */
1590
+ async assertSessionsSupported(signal) {
1591
+ const features = await this.wire.guestFeatures(signal);
1592
+ if (features.includes(SESSIONS_FEATURE))
1593
+ return;
1594
+ throw new KubernetesSessionsUnsupportedError(SESSIONS_FEATURE, `kubernetes: this workspace's guest agent does not advertise the '${SESSIONS_FEATURE}' healthz feature, so a terminal opened here would die with this connection and a detached program could not be named, read or killed. The request is refused rather than served as a connection-bound terminal. Rebuild the workspace image from this Namzu release.`);
123
1595
  }
124
1596
  async openTcpConnection(options) {
125
- return await this.wire.openTcpConnection(options);
1597
+ return await this.withRebind(async () => await this.wire.openTcpConnection(options));
126
1598
  }
127
1599
  /**
128
1600
  * Run one command through a fresh, call-scoped adapter + controller.
@@ -139,50 +1611,117 @@ export class KubernetesAgentTransport {
139
1611
  let reserveMs = 0;
140
1612
  let executeMs = 0;
141
1613
  let executeSettledAt = 0;
142
- const timedWire = new VsockAgentTransport(this.handle, {
143
- ...this.transportOptions,
144
- onDial: (ms) => {
145
- dialMs += ms;
146
- },
147
- });
148
- const adapter = {
149
- label: 'kubernetes pod-network agent',
150
- reserve: async (signal) => {
151
- const startedAt = Date.now();
152
- try {
153
- return await requestChecked(timedWire, { op: 'reserve-execution' }, signal);
154
- }
155
- finally {
156
- reserveMs += Date.now() - startedAt;
157
- }
158
- },
159
- // Checked exactly like `reserve` — see `requestChecked`.
160
- cancel: async (executionId, signal) => await requestChecked(timedWire, { op: 'cancel-execution', body: { executionId } }, signal),
161
- execute: async (executionId, cmd, execArgv, execOpts, signal, context) => {
162
- const startedAt = Date.now();
163
- try {
164
- return await timedWire.executeStreamed({
165
- ...(executionId ? { executionId } : {}),
166
- command: cmd,
167
- args: execArgv ?? [],
168
- ...(execOpts?.cwd !== undefined ? { cwd: execOpts.cwd } : {}),
169
- ...(execOpts?.env !== undefined ? { env: execOpts.env } : {}),
170
- ...(execOpts?.timeout !== undefined ? { timeoutMs: execOpts.timeout } : {}),
171
- ...(context?.stdin !== undefined ? { stdin: context.stdin } : {}),
172
- ...(context?.maxOutputBytes !== undefined
173
- ? { maxOutputBytes: context.maxOutputBytes }
174
- : {}),
175
- }, execOpts, signal);
176
- }
177
- finally {
178
- executeMs += Date.now() - startedAt;
179
- executeSettledAt = Date.now();
180
- }
181
- },
1614
+ // The guest that ACCEPTED the reservation, which is the guest that
1615
+ // runs the command: its pod (the bind token this attempt presented)
1616
+ // and its process (the boot id the reply carried). Written per
1617
+ // ATTEMPT, so a retry that followed a replaced pod is remembered
1618
+ // against the pod it actually reserved on and not the one that
1619
+ // refused it — see {@link executionGuest}.
1620
+ let reservedIn;
1621
+ let reservedOn;
1622
+ // What this attempt's dials did, for the one classification `exec()`
1623
+ // cannot make from its error: the controller bounds `reserve` at 2s
1624
+ // and hands back its own timer's Error, so a dial still inside its
1625
+ // connect-retry budget is reported as a reservation that took too
1626
+ // long, with the connect failure discarded. See {@link DialWatch}.
1627
+ const dials = { attempted: false, failed: false, connected: false };
1628
+ // Built per ATTEMPT, from `this.handle` as it stands when the attempt
1629
+ // starts: a retry that follows a replaced pod has to dial the new
1630
+ // address, and the timing accumulators above outlive both attempts so
1631
+ // the caller still sees one call's total. The watch is reset here and
1632
+ // not there — it describes the attempt, and the second attempt is a
1633
+ // different pod's.
1634
+ const buildAttempt = () => {
1635
+ dials.attempted = false;
1636
+ dials.failed = false;
1637
+ dials.connected = false;
1638
+ dials.lastError = undefined;
1639
+ // The handle as it stands NOW, read once: the attempt dials this
1640
+ // address with this token, and another call's rebind must not
1641
+ // change what this attempt records it reserved on.
1642
+ const boundTo = this.handle;
1643
+ const timedWire = new VsockAgentTransport(boundTo, {
1644
+ ...this.wireOptions(boundTo),
1645
+ // Fires before each connect, so an attempt the controller's
1646
+ // bound aborts mid-connect is still on the record — see
1647
+ // {@link DialWatch}.
1648
+ onDialAttempt: () => {
1649
+ dials.attempted = true;
1650
+ },
1651
+ onDial: (ms) => {
1652
+ dials.connected = true;
1653
+ dialMs += ms;
1654
+ },
1655
+ // Consulted by the dial on every FAILED connect attempt, which
1656
+ // is where the error the bound swallows is kept. The answer
1657
+ // itself is the transport's own, unchanged.
1658
+ permanentDialFailure: (err) => {
1659
+ dials.failed = true;
1660
+ dials.lastError = err;
1661
+ return this.dialFailureIsPermanent(err);
1662
+ },
1663
+ });
1664
+ const adapter = {
1665
+ label: 'kubernetes pod-network agent',
1666
+ reserve: async (signal) => {
1667
+ const startedAt = Date.now();
1668
+ try {
1669
+ const reply = await requestChecked(timedWire, { op: 'reserve-execution' }, signal);
1670
+ // Overwritten, never merged: the LAST reservation is the
1671
+ // one the command ran under, and a replacement guest
1672
+ // that reports no boot id must leave no evidence
1673
+ // behind rather than inherit its predecessor's.
1674
+ reservedIn = guestBootIdOf(reply);
1675
+ reservedOn = boundTo.token;
1676
+ return reply;
1677
+ }
1678
+ finally {
1679
+ reserveMs += Date.now() - startedAt;
1680
+ }
1681
+ },
1682
+ // Checked exactly like `reserve` — see `requestChecked`.
1683
+ cancel: async (executionId, signal) => await requestChecked(timedWire, { op: 'cancel-execution', body: { executionId } }, signal),
1684
+ execute: async (executionId, cmd, execArgv, execOpts, signal, context) => {
1685
+ const startedAt = Date.now();
1686
+ try {
1687
+ return await timedWire.executeStreamed({
1688
+ ...(executionId ? { executionId } : {}),
1689
+ command: cmd,
1690
+ args: execArgv ?? [],
1691
+ ...(execOpts?.cwd !== undefined ? { cwd: execOpts.cwd } : {}),
1692
+ ...(execOpts?.env !== undefined ? { env: execOpts.env } : {}),
1693
+ ...(execOpts?.timeout !== undefined ? { timeoutMs: execOpts.timeout } : {}),
1694
+ ...(context?.stdin !== undefined ? { stdin: context.stdin } : {}),
1695
+ ...(context?.maxOutputBytes !== undefined
1696
+ ? { maxOutputBytes: context.maxOutputBytes }
1697
+ : {}),
1698
+ }, execOpts, signal);
1699
+ }
1700
+ finally {
1701
+ executeMs += Date.now() - startedAt;
1702
+ executeSettledAt = Date.now();
1703
+ }
1704
+ },
1705
+ };
1706
+ return new RemoteExecutionController(adapter).exec(command, argv, opts);
182
1707
  };
183
- const controller = new RemoteExecutionController(adapter);
184
1708
  try {
185
- return await controller.exec(command, argv, opts);
1709
+ return await this.withRebind(buildAttempt, opts?.signal, dials);
1710
+ }
1711
+ catch (error) {
1712
+ // Stamped here and nowhere else. The retirement hook one frame up
1713
+ // is handed the error ALONE, so this is the last point at which
1714
+ // "which execution is this" is still known — and the question the
1715
+ // hook has to answer is about this command's guest, not the
1716
+ // handle's.
1717
+ if (error instanceof RemoteCancellationUnknownError &&
1718
+ (reservedOn !== undefined || reservedIn !== undefined)) {
1719
+ executionGuest.set(error, {
1720
+ ...(reservedOn !== undefined ? { podUid: reservedOn } : {}),
1721
+ ...(reservedIn !== undefined ? { guestBootId: reservedIn } : {}),
1722
+ });
1723
+ }
1724
+ throw error;
186
1725
  }
187
1726
  finally {
188
1727
  this.onTiming?.({
@@ -193,5 +1732,362 @@ export class KubernetesAgentTransport {
193
1732
  });
194
1733
  }
195
1734
  }
1735
+ /**
1736
+ * Whether this guest implements the detach/attach ops at all, asked once
1737
+ * per transport and only when a caller wants them.
1738
+ */
1739
+ async assertExecutionAttachSupported(signal) {
1740
+ const features = await this.wire.guestFeatures(signal);
1741
+ if (features.includes(EXECUTION_ATTACH_FEATURE))
1742
+ return;
1743
+ throw new KubernetesExecutionAttachUnsupportedError(EXECUTION_ATTACH_FEATURE, `kubernetes: this workspace's guest agent does not advertise the '${EXECUTION_ATTACH_FEATURE}' healthz feature, so a command started here would keep no output and could not be reattached to. The request is refused before the command is admitted rather than run as an ordinary exec. Rebuild the workspace image from this Namzu release.`);
1744
+ }
1745
+ /** `reserve-execution` for a caller-named id, with its reported state. */
1746
+ async reserveDetached(executionId, signal) {
1747
+ const response = (await this.withRebind(async () => await requestChecked(this.wire, { op: 'reserve-execution', body: { executionId } }, signal), signal));
1748
+ if (response.ok !== true || response.executionId !== executionId) {
1749
+ throw new RemoteProtocolError(`kubernetes: the guest refused the reservation for ${executionId}: ${String(response.error ?? 'no reason given')}`);
1750
+ }
1751
+ return { state: typeof response.state === 'string' ? response.state : 'reserved' };
1752
+ }
1753
+ /**
1754
+ * The `cancel-execution` control path, retried for its whole confirm
1755
+ * window and reported UNKNOWN rather than as a failure if none of the
1756
+ * attempts got an answer — the same rule the shared execution
1757
+ * controller applies, because a command whose termination nobody
1758
+ * confirmed is not a command anybody may call dead.
1759
+ *
1760
+ * Nothing on the detach path calls this to reconcile a lost connection.
1761
+ * It runs when the CALLER asked for it: `SandboxExecOptions.signal`
1762
+ * aborting, or {@link cancelExecution}.
1763
+ */
1764
+ async confirmCancel(executionId, signal) {
1765
+ const deadlineAt = Date.now() + CANCEL_CONFIRM_WINDOW_MS;
1766
+ let lastError = new Error('no cancellation attempt completed');
1767
+ while (Date.now() < deadlineAt) {
1768
+ const attempt = new AbortController();
1769
+ const onAbort = () => attempt.abort(signal?.reason);
1770
+ signal?.addEventListener('abort', onAbort, { once: true });
1771
+ const timer = setTimeout(() => attempt.abort(new Error(`cancellation attempt exceeded ${CANCEL_ATTEMPT_TIMEOUT_MS}ms`)), Math.min(CANCEL_ATTEMPT_TIMEOUT_MS, Math.max(1, deadlineAt - Date.now())));
1772
+ timer.unref?.();
1773
+ try {
1774
+ return parseCancellationReply(executionId, await this.cancel(executionId, attempt.signal));
1775
+ }
1776
+ catch (error) {
1777
+ if (error instanceof KubernetesExecutionNotAttachableError)
1778
+ throw error;
1779
+ if (error instanceof KubernetesAgentUnauthorizedError)
1780
+ throw error;
1781
+ lastError = error;
1782
+ const remaining = deadlineAt - Date.now();
1783
+ if (remaining > 0)
1784
+ await pause(Math.min(50, remaining));
1785
+ }
1786
+ finally {
1787
+ clearTimeout(timer);
1788
+ signal?.removeEventListener('abort', onAbort);
1789
+ }
1790
+ }
1791
+ throw new RemoteCancellationUnknownError(`Remote sandbox cancellation could not be confirmed for ${executionId}: ${lastError instanceof Error ? lastError.message : String(lastError)}. The remote outcome is unknown; do not automatically retry the command.`, { cause: lastError });
1792
+ }
1793
+ /**
1794
+ * End a command by id, from any host process holding the id and the
1795
+ * bind token. Resolves only on a CONFIRMED termination.
1796
+ *
1797
+ * The outcome is not reported here — a cancelled execution's result and
1798
+ * whatever output it managed is read back through
1799
+ * {@link attachExecution}, which is the op that exists for reading.
1800
+ */
1801
+ async cancelExecution(executionId, signal) {
1802
+ assertExecutionId(executionId);
1803
+ await this.confirmCancel(executionId, signal);
1804
+ }
1805
+ /**
1806
+ * Observe a command that is already the guest's, from `fromOffset` on,
1807
+ * and resolve with its result.
1808
+ *
1809
+ * It never signals the command. Aborting `signal` stops OBSERVING and
1810
+ * rejects with {@link KubernetesExecutionDetachedError}; it does not
1811
+ * cancel, and the command goes on running. Ending a command is
1812
+ * {@link cancelExecution} and nothing else.
1813
+ */
1814
+ async attachExecution(executionId, options = {}) {
1815
+ assertExecutionId(executionId);
1816
+ await this.assertExecutionAttachSupported(options.signal);
1817
+ const cursor = {
1818
+ offset: options.fromOffset ?? 0,
1819
+ stdout: '',
1820
+ stderr: '',
1821
+ droppedBytes: 0,
1822
+ };
1823
+ const observation = new AbortController();
1824
+ const onDetach = () => observation.abort(options.signal?.reason);
1825
+ options.signal?.addEventListener('abort', onDetach, { once: true });
1826
+ if (options.signal?.aborted)
1827
+ onDetach();
1828
+ try {
1829
+ return await this.attachOnce(executionId, cursor, options, observation.signal);
1830
+ }
1831
+ catch (error) {
1832
+ if (options.signal?.aborted)
1833
+ throw this.detached(executionId, cursor, error);
1834
+ throw error;
1835
+ }
1836
+ finally {
1837
+ options.signal?.removeEventListener('abort', onDetach);
1838
+ observation.abort(new Error('kubernetes: attach observation finished'));
1839
+ }
1840
+ }
1841
+ /**
1842
+ * Run one command whose observation can outlive this connection, and —
1843
+ * when the connection is what failed — get it back rather than killing
1844
+ * the command to reconcile.
1845
+ *
1846
+ * The order is exactly: refuse if the guest cannot keep output, reserve
1847
+ * the id, and only then admit the command. Reserving the id the CALLER
1848
+ * named is what makes a retried start idempotent: a second call with the
1849
+ * same id inside retention finds the execution already running or
1850
+ * finished, sends no `execute`, and attaches to the one that exists.
1851
+ *
1852
+ * `SandboxExecOptions.signal` keeps its contract — aborting it runs the
1853
+ * confirmed cancel — and `detachSignal` is its opposite: it ends the
1854
+ * observation and leaves the command alone, for a host that is shutting
1855
+ * down and wants its work to survive the rollout.
1856
+ */
1857
+ async execDetached(command, argv, opts = {}) {
1858
+ const executionId = opts.executionId ?? `exec_${randomUUID()}`;
1859
+ assertExecutionId(executionId);
1860
+ const startedAt = Date.now();
1861
+ if (opts.signal?.aborted) {
1862
+ return {
1863
+ exitCode: 1,
1864
+ stdout: '',
1865
+ stderr: '',
1866
+ timedOut: false,
1867
+ durationMs: Math.max(0, Date.now() - startedAt),
1868
+ stdoutTruncated: false,
1869
+ stderrTruncated: false,
1870
+ };
1871
+ }
1872
+ await this.assertExecutionAttachSupported(opts.signal);
1873
+ const cursor = { offset: 0, stdout: '', stderr: '', droppedBytes: 0 };
1874
+ const observationTimeoutMs = Math.min(MAX_TIMER_DELAY_MS, (typeof opts.timeout === 'number' && Number.isFinite(opts.timeout) && opts.timeout > 0
1875
+ ? opts.timeout
1876
+ : DEFAULT_EXECUTION_TIMEOUT_MS) + EXECUTION_OBSERVATION_GRACE_MS);
1877
+ const deadlineAt = Date.now() + observationTimeoutMs;
1878
+ // The caller's abort runs the CONFIRMED cancel, in the background,
1879
+ // while the observation keeps reading: the guest answers the cancel
1880
+ // on its own connection and ends this one with the terminal frame, so
1881
+ // aborting produces a result rather than a severed stream.
1882
+ let cancelFailure;
1883
+ let cancelling = false;
1884
+ const cancelNow = () => {
1885
+ if (cancelling)
1886
+ return;
1887
+ cancelling = true;
1888
+ void this.confirmCancel(executionId).catch((error) => {
1889
+ cancelFailure = error;
1890
+ });
1891
+ };
1892
+ const onAbort = () => cancelNow();
1893
+ opts.signal?.addEventListener('abort', onAbort, { once: true });
1894
+ // Aborting this stops the READING and nothing else. It is never the
1895
+ // caller's `signal`: destroying a socket does not end a guest command,
1896
+ // and a host that treats it as though it did is exactly how a network
1897
+ // blip used to cost a workspace its pod.
1898
+ const observation = new AbortController();
1899
+ const onDetach = () => observation.abort(new Error('kubernetes: observation detached'));
1900
+ opts.detachSignal?.addEventListener('abort', onDetach, { once: true });
1901
+ if (opts.detachSignal?.aborted)
1902
+ onDetach();
1903
+ try {
1904
+ let lastError;
1905
+ const reservation = await this.reserveDetached(executionId, opts.signal);
1906
+ if (reservation.state === 'reserved') {
1907
+ try {
1908
+ return await this.executeRetained(executionId, {
1909
+ executionId,
1910
+ command,
1911
+ args: argv ?? [],
1912
+ ...(opts.cwd !== undefined ? { cwd: opts.cwd } : {}),
1913
+ ...(opts.env !== undefined ? { env: opts.env } : {}),
1914
+ ...(opts.timeout !== undefined ? { timeoutMs: opts.timeout } : {}),
1915
+ retainOutput: true,
1916
+ }, cursor, opts, observation.signal, observationTimeoutMs);
1917
+ }
1918
+ catch (error) {
1919
+ lastError = error;
1920
+ // A second host process that reserved the same id and won
1921
+ // the race owns the command now. Its refusal says so, and
1922
+ // the answer is to ATTACH to the command that exists —
1923
+ // reporting a failure here would be a lie about an id
1924
+ // whose command is running.
1925
+ if (!lostTheStartRace(error) && isTerminalAttachError(error))
1926
+ throw error;
1927
+ }
1928
+ }
1929
+ // The window bounds GETTING BACK, and only that. It is armed as an
1930
+ // abort rather than checked between attempts because a single
1931
+ // attempt is not short: the dial carries its own connect-retry
1932
+ // budget, so a peer that is refusing connections would otherwise
1933
+ // be waited on for that whole budget inside one attempt and the
1934
+ // caller's bound would never be consulted. Once an attach has
1935
+ // actually attached the window is disarmed — a command that is
1936
+ // being read successfully is not something to give up on — and it
1937
+ // is re-armed if that connection dies in its turn.
1938
+ const reattachWindowMs = opts.reattachWindowMs ?? DEFAULT_REATTACH_WINDOW_MS;
1939
+ let windowEndsAt = Date.now() + reattachWindowMs;
1940
+ for (;;) {
1941
+ if (opts.detachSignal?.aborted)
1942
+ break;
1943
+ const remainingMs = windowEndsAt - Date.now();
1944
+ if (remainingMs <= 0)
1945
+ break;
1946
+ const window = new AbortController();
1947
+ const windowTimer = setTimeout(() => window.abort(new Error(`kubernetes: could not reattach to execution ${executionId} within ${reattachWindowMs}ms`)), remainingMs);
1948
+ windowTimer.unref?.();
1949
+ let reattached = false;
1950
+ try {
1951
+ return await this.attachOnce(executionId, cursor, opts, AbortSignal.any([observation.signal, window.signal]), deadlineAt, () => {
1952
+ reattached = true;
1953
+ clearTimeout(windowTimer);
1954
+ });
1955
+ }
1956
+ catch (error) {
1957
+ lastError = error;
1958
+ if (isTerminalAttachError(error))
1959
+ break;
1960
+ if (opts.detachSignal?.aborted)
1961
+ break;
1962
+ if (reattached)
1963
+ windowEndsAt = Date.now() + reattachWindowMs;
1964
+ else if (window.signal.aborted)
1965
+ break;
1966
+ await pause(REATTACH_RETRY_DELAY_MS);
1967
+ }
1968
+ finally {
1969
+ clearTimeout(windowTimer);
1970
+ }
1971
+ }
1972
+ throw this.detached(executionId, cursor, cancelFailure ?? lastError);
1973
+ }
1974
+ finally {
1975
+ opts.signal?.removeEventListener('abort', onAbort);
1976
+ opts.detachSignal?.removeEventListener('abort', onDetach);
1977
+ observation.abort(new Error('kubernetes: detached execution observation finished'));
1978
+ }
1979
+ }
1980
+ /**
1981
+ * The `execute` leg of a detached run, read frame by frame.
1982
+ *
1983
+ * It deliberately does NOT go through `executeStreamed`: that path
1984
+ * hands its caller `{stream, data}` and nothing else, and the byte
1985
+ * offsets this cursor lives on are on the frames themselves. Reading
1986
+ * them here is what lets a reattach resume exactly where this
1987
+ * connection stopped, on output whose decoded length is not its byte
1988
+ * length — see {@link applyDelta}. The frame union and its validation
1989
+ * are the shared ones, so an ordinary exec and a detached one never
1990
+ * disagree about what the guest said.
1991
+ */
1992
+ async executeRetained(executionId, body, cursor, opts, signal, observationTimeoutMs) {
1993
+ // No `onOutput` on the accumulator: this reads the frames, so the
1994
+ // caller is called exactly once per chunk, from `applyDelta`.
1995
+ const accumulator = new ExecResultAccumulator(Date.now());
1996
+ await this.wire.streamFramedRequest({ op: 'execute', body }, (frame) => {
1997
+ if (isUnauthorized(frame))
1998
+ throw new KubernetesAgentUnauthorizedError();
1999
+ const event = parseExecEvent(frame);
2000
+ const stream = deltaStream(event.type);
2001
+ if (stream !== undefined)
2002
+ applyDelta(cursor, executionId, stream, frame, opts.onOutput);
2003
+ accumulator.push(event);
2004
+ }, { observationTimeoutMs }, signal);
2005
+ if (!accumulator.done) {
2006
+ throw new RemoteProtocolError(`kubernetes: the execute stream for ${executionId} ended without a result`);
2007
+ }
2008
+ return accumulator.finish();
2009
+ }
2010
+ /** One `attach-execution` stream, read to its terminal frame. */
2011
+ async attachOnce(executionId, cursor, opts, signal, deadlineAt, onAttached) {
2012
+ let terminal;
2013
+ let refusal;
2014
+ const observationTimeoutMs = deadlineAt === undefined
2015
+ ? MAX_TIMER_DELAY_MS
2016
+ : Math.max(1, Math.min(MAX_TIMER_DELAY_MS, deadlineAt - Date.now()));
2017
+ await this.wire.streamFramedRequest({ op: 'attach-execution', body: { executionId, fromOffset: cursor.offset } }, (event) => {
2018
+ if (isUnauthorized(event))
2019
+ throw new KubernetesAgentUnauthorizedError();
2020
+ const type = event.type;
2021
+ if (type === 'attached') {
2022
+ onAttached?.();
2023
+ const from = Number(event.fromOffset);
2024
+ if (Number.isFinite(from))
2025
+ cursor.offset = from;
2026
+ const dropped = Number(event.droppedBytes ?? 0);
2027
+ if (Number.isFinite(dropped) && dropped > 0) {
2028
+ cursor.droppedBytes += dropped;
2029
+ opts.onGap?.({ executionId, fromOffset: cursor.offset, droppedBytes: dropped });
2030
+ }
2031
+ return;
2032
+ }
2033
+ const stream = deltaStream(type);
2034
+ if (stream !== undefined) {
2035
+ applyDelta(cursor, executionId, stream, event, opts.onOutput);
2036
+ return;
2037
+ }
2038
+ if (type === 'attach_result') {
2039
+ const next = Number(event.nextOffset);
2040
+ if (Number.isFinite(next))
2041
+ cursor.offset = next;
2042
+ const outcome = String(event.outcome);
2043
+ terminal = {
2044
+ outcome: outcome === 'cancelled' || outcome === 'failed'
2045
+ ? outcome
2046
+ : 'completed',
2047
+ ...(event.result !== undefined
2048
+ ? { result: terminalMetadataOrThrow(event.result, executionId) }
2049
+ : {}),
2050
+ ...(typeof event.error === 'string' ? { error: event.error } : {}),
2051
+ };
2052
+ return;
2053
+ }
2054
+ if (type === 'error') {
2055
+ const code = typeof event.error === 'string' ? event.error : 'unknown';
2056
+ const state = typeof event.state === 'string' ? event.state : undefined;
2057
+ // A refusal for an execution the guest still holds as
2058
+ // `reserved` is the one case that is not about retention:
2059
+ // the command was never started, so there is nothing
2060
+ // running and nothing to come back for. Saying anything
2061
+ // else here would send a caller looking for a process
2062
+ // that does not exist.
2063
+ const because = state === 'reserved'
2064
+ ? 'The guest holds it as reserved and never started it, so no command is running and there is nothing to reattach to.'
2065
+ : 'A record is kept for the retention window configured by NAMZU_AGENT_EXECUTION_RETAINED_TTL_MS and is lost when the pod is replaced; only a command started with detach keeps its output at all.';
2066
+ refusal = new KubernetesExecutionNotAttachableError(executionId, isAttachRefusal(code) ? code : 'unknown', `kubernetes: the guest refused to attach to execution ${executionId} (${code}). ${because}`, { state });
2067
+ return;
2068
+ }
2069
+ throw new RemoteProtocolError(`kubernetes: the guest sent an unexpected frame on the attach stream for ${executionId}: ${JSON.stringify(event).slice(0, 200)}`);
2070
+ }, { observationTimeoutMs }, signal);
2071
+ if (refusal !== undefined)
2072
+ throw refusal;
2073
+ if (terminal === undefined) {
2074
+ throw new RemoteProtocolError(`kubernetes: the attach stream for ${executionId} ended without a terminal frame`);
2075
+ }
2076
+ return resultFromAttachTerminal(executionId, terminal, cursor);
2077
+ }
2078
+ /** The one error a lost observation ends with. */
2079
+ detached(executionId, cursor, cause) {
2080
+ // The guest can tell us the command never started — the reservation
2081
+ // is still `reserved`, so the `execute` never reached it. Then
2082
+ // there is nothing running, nothing to reattach to and nothing to
2083
+ // cancel, and promising otherwise sends the caller after a process
2084
+ // that does not exist. Every other cause leaves the command's fate
2085
+ // genuinely unknown to this host, which is what the rest says.
2086
+ const neverStarted = cause instanceof KubernetesExecutionNotAttachableError && cause.executionState === 'reserved';
2087
+ const advice = neverStarted
2088
+ ? 'the guest still holds it as RESERVED, so the command never started: nothing is running, and starting it again with the same id is safe.'
2089
+ : `it may still be running in the workspace pod. Reattach with attachExecution('${executionId}', { fromOffset: ${cursor.offset} }), from this process or another one, or end it with cancelExecution('${executionId}').`;
2090
+ return new KubernetesExecutionDetachedError(executionId, cursor.offset, `kubernetes: stopped observing execution ${executionId} after ${cursor.offset} bytes of output, and did NOT cancel it — ${advice} Cause: ${cause instanceof Error ? cause.message : String(cause)}`, { cause });
2091
+ }
196
2092
  }
197
2093
  //# sourceMappingURL=transport.js.map