@namzu/sandbox 13.0.0 → 15.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (104) hide show
  1. package/CHANGELOG.md +1147 -0
  2. package/README.md +447 -0
  3. package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
  4. package/dist/backends/aci-standby-pool/index.js +13 -1
  5. package/dist/backends/aci-standby-pool/index.js.map +1 -1
  6. package/dist/backends/docker/index.d.ts.map +1 -1
  7. package/dist/backends/docker/index.js +19 -1
  8. package/dist/backends/docker/index.js.map +1 -1
  9. package/dist/backends/firecracker/index.d.ts.map +1 -1
  10. package/dist/backends/firecracker/index.js +12 -2
  11. package/dist/backends/firecracker/index.js.map +1 -1
  12. package/dist/backends/firecracker/protocol.d.ts +481 -8
  13. package/dist/backends/firecracker/protocol.d.ts.map +1 -1
  14. package/dist/backends/firecracker/protocol.js +136 -0
  15. package/dist/backends/firecracker/protocol.js.map +1 -1
  16. package/dist/backends/firecracker/transport.d.ts +642 -14
  17. package/dist/backends/firecracker/transport.d.ts.map +1 -1
  18. package/dist/backends/firecracker/transport.js +1307 -34
  19. package/dist/backends/firecracker/transport.js.map +1 -1
  20. package/dist/backends/kubernetes/egress-policy.d.ts +1296 -0
  21. package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -0
  22. package/dist/backends/kubernetes/egress-policy.js +2458 -0
  23. package/dist/backends/kubernetes/egress-policy.js.map +1 -0
  24. package/dist/backends/kubernetes/identity.d.ts +193 -0
  25. package/dist/backends/kubernetes/identity.d.ts.map +1 -0
  26. package/dist/backends/kubernetes/identity.js +147 -0
  27. package/dist/backends/kubernetes/identity.js.map +1 -0
  28. package/dist/backends/kubernetes/index.d.ts +1019 -0
  29. package/dist/backends/kubernetes/index.d.ts.map +1 -0
  30. package/dist/backends/kubernetes/index.js +1756 -0
  31. package/dist/backends/kubernetes/index.js.map +1 -0
  32. package/dist/backends/kubernetes/ingress-policy.d.ts +375 -0
  33. package/dist/backends/kubernetes/ingress-policy.d.ts.map +1 -0
  34. package/dist/backends/kubernetes/ingress-policy.js +1050 -0
  35. package/dist/backends/kubernetes/ingress-policy.js.map +1 -0
  36. package/dist/backends/kubernetes/k8s-client.d.ts +334 -0
  37. package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -0
  38. package/dist/backends/kubernetes/k8s-client.js +553 -0
  39. package/dist/backends/kubernetes/k8s-client.js.map +1 -0
  40. package/dist/backends/kubernetes/lease.d.ts +145 -0
  41. package/dist/backends/kubernetes/lease.d.ts.map +1 -0
  42. package/dist/backends/kubernetes/lease.js +201 -0
  43. package/dist/backends/kubernetes/lease.js.map +1 -0
  44. package/dist/backends/kubernetes/objects.d.ts +702 -0
  45. package/dist/backends/kubernetes/objects.d.ts.map +1 -0
  46. package/dist/backends/kubernetes/objects.js +518 -0
  47. package/dist/backends/kubernetes/objects.js.map +1 -0
  48. package/dist/backends/kubernetes/per-sandbox-policy.d.ts +219 -0
  49. package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -0
  50. package/dist/backends/kubernetes/per-sandbox-policy.js +407 -0
  51. package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -0
  52. package/dist/backends/kubernetes/privilege-probe.d.ts +136 -0
  53. package/dist/backends/kubernetes/privilege-probe.d.ts.map +1 -0
  54. package/dist/backends/kubernetes/privilege-probe.js +185 -0
  55. package/dist/backends/kubernetes/privilege-probe.js.map +1 -0
  56. package/dist/backends/kubernetes/rbac.d.ts +153 -0
  57. package/dist/backends/kubernetes/rbac.d.ts.map +1 -0
  58. package/dist/backends/kubernetes/rbac.js +177 -0
  59. package/dist/backends/kubernetes/rbac.js.map +1 -0
  60. package/dist/backends/kubernetes/sandbox.d.ts +190 -0
  61. package/dist/backends/kubernetes/sandbox.d.ts.map +1 -0
  62. package/dist/backends/kubernetes/sandbox.js +433 -0
  63. package/dist/backends/kubernetes/sandbox.js.map +1 -0
  64. package/dist/backends/kubernetes/transport.d.ts +1048 -0
  65. package/dist/backends/kubernetes/transport.d.ts.map +1 -0
  66. package/dist/backends/kubernetes/transport.js +2093 -0
  67. package/dist/backends/kubernetes/transport.js.map +1 -0
  68. package/dist/backends/kubernetes/workspace.d.ts +1512 -0
  69. package/dist/backends/kubernetes/workspace.d.ts.map +1 -0
  70. package/dist/backends/kubernetes/workspace.js +3703 -0
  71. package/dist/backends/kubernetes/workspace.js.map +1 -0
  72. package/dist/backends/remote-execution-controller.d.ts +14 -0
  73. package/dist/backends/remote-execution-controller.d.ts.map +1 -1
  74. package/dist/backends/remote-execution-controller.js.map +1 -1
  75. package/dist/index.d.ts +350 -2
  76. package/dist/index.d.ts.map +1 -1
  77. package/dist/index.js +344 -34
  78. package/dist/index.js.map +1 -1
  79. package/dist/testing/sandbox-conformance.d.ts +227 -0
  80. package/dist/testing/sandbox-conformance.d.ts.map +1 -0
  81. package/dist/testing/sandbox-conformance.js +896 -0
  82. package/dist/testing/sandbox-conformance.js.map +1 -0
  83. package/package.json +5 -4
  84. package/src/backends/aci-standby-pool/index.ts +16 -1
  85. package/src/backends/docker/index.ts +22 -1
  86. package/src/backends/firecracker/index.ts +14 -2
  87. package/src/backends/firecracker/protocol.ts +541 -6
  88. package/src/backends/firecracker/transport.ts +1687 -64
  89. package/src/backends/kubernetes/egress-policy.ts +3448 -0
  90. package/src/backends/kubernetes/identity.ts +261 -0
  91. package/src/backends/kubernetes/index.ts +2670 -0
  92. package/src/backends/kubernetes/ingress-policy.ts +1344 -0
  93. package/src/backends/kubernetes/k8s-client.ts +742 -0
  94. package/src/backends/kubernetes/lease.ts +254 -0
  95. package/src/backends/kubernetes/objects.ts +983 -0
  96. package/src/backends/kubernetes/per-sandbox-policy.ts +542 -0
  97. package/src/backends/kubernetes/privilege-probe.ts +261 -0
  98. package/src/backends/kubernetes/rbac.ts +192 -0
  99. package/src/backends/kubernetes/sandbox.ts +593 -0
  100. package/src/backends/kubernetes/transport.ts +2895 -0
  101. package/src/backends/kubernetes/workspace.ts +5640 -0
  102. package/src/backends/remote-execution-controller.ts +14 -0
  103. package/src/index.ts +838 -35
  104. package/src/testing/sandbox-conformance.ts +1202 -0
@@ -52,8 +52,8 @@
52
52
  * socket to be silently severed by a resume), which makes the
53
53
  * transport resume-survivable by construction.
54
54
  */
55
- import type { OpenTerminalOptions, SandboxExecOptions, SandboxExecResult, SandboxTcpConnectOptions, SandboxTcpConnection, TerminalSession } from '@namzu/sdk';
56
- import { type ExecRequest, type ReadFileRequest, type TcpConnectRequest, type TerminalOpenRequest, type WriteFileRequest } from './protocol.js';
55
+ import type { OpenTerminalOptions, SandboxExecOptions, SandboxExecResult, SandboxReadFileOptions, SandboxTcpConnectOptions, SandboxTcpConnection, TerminalSession } from '@namzu/sdk';
56
+ import { type AgentRequestCredential, type AttachSessionRequest, type ExecRequest, type FlushRequest, type GuestReplyIdentity, type KillSessionRequest, type QuiesceRequest, type ReadFileRequest, type ReadFileStreamRequest, type StartDetachedRequest, type TcpConnectRequest, type TerminalOpenRequest, type TerminalReadyEvent, type WriteFileRequest } from './protocol.js';
57
57
  /**
58
58
  * An addressable agent endpoint. The orchestrator hands one of these
59
59
  * back per sandbox (`create()` response → `vsock endpoint`).
@@ -79,6 +79,18 @@ import { type ExecRequest, type ReadFileRequest, type TcpConnectRequest, type Te
79
79
  * (`tls.TLSSocket` is a `net.Socket`). The container-app NEVER sees
80
80
  * a host-local `udsPath`; the cert material is injected by the
81
81
  * Vandal host layer, never returned by the orchestrator.
82
+ * - `tcp` — the guest agent listening directly on a routed pod
83
+ * network (the kubernetes backend). The dialer does a plain
84
+ * `net.connect({ host, port })` — no relay, no routing preamble, no
85
+ * ack: there is nothing between the host and the guest's own listen
86
+ * socket to route through, so the framing loop starts on the very
87
+ * first byte, exactly as the `unix` arm's does. `host` may be a
88
+ * Service FQDN rather than a literal IP; every call dials fresh (see
89
+ * `dial()` below), so DNS is re-resolved on every request and a
90
+ * resumed pod's new address is picked up for free. `token` rides in
91
+ * each request envelope (see `AgentRequestCredential` in
92
+ * `protocol.ts`) because a routed listener authenticates what the
93
+ * vsock/unix control channel never had to.
82
94
  */
83
95
  export type SandboxAgentHandle = {
84
96
  readonly kind: 'unix';
@@ -98,6 +110,11 @@ export type SandboxAgentHandle = {
98
110
  readonly key: string | Buffer;
99
111
  readonly servername?: string;
100
112
  };
113
+ } | {
114
+ readonly kind: 'tcp';
115
+ readonly host: string;
116
+ readonly port: number;
117
+ readonly token?: string;
101
118
  };
102
119
  /**
103
120
  * The mTLS cert material the consumer injects onto a wire `mtls` handle (the
@@ -112,12 +129,17 @@ export interface MtlsClientMaterial {
112
129
  readonly servername?: string;
113
130
  }
114
131
  /**
115
- * The WIRE shape of an agent handle as the orchestrator returns it. Identical
116
- * to {@link SandboxAgentHandle} EXCEPT the `mtls` arm omits the `tls` cert
117
- * block: the orchestrator returns only host/port/sandboxId, and the consumer
118
- * (Vandal host layer) merges the cert material in (see `normalizeHandle`)
119
- * before constructing the transport. The `unix`/`vsock` arms are unchanged
120
- * (they carry no cert material).
132
+ * The WIRE shape of an agent handle as the FIRECRACKER orchestrator
133
+ * returns it. Identical to {@link SandboxAgentHandle} EXCEPT the `mtls`
134
+ * arm omits the `tls` cert block: the orchestrator returns only
135
+ * host/port/sandboxId, and the consumer (Vandal host layer) merges the
136
+ * cert material in (see `normalizeHandle`) before constructing the
137
+ * transport. The `unix`/`vsock` arms are unchanged (they carry no cert
138
+ * material). There is deliberately no `tcp` arm here: that kind belongs
139
+ * to the kubernetes backend, which builds its {@link SandboxAgentHandle}
140
+ * directly (host/port from the claimed Sandbox's status, token from the
141
+ * pod's own identity) and never goes through this orchestrator wire
142
+ * shape or `normalizeHandle`.
121
143
  */
122
144
  export type WireSandboxAgentHandle = {
123
145
  readonly kind: 'unix';
@@ -137,31 +159,60 @@ export type WireSandboxAgentHandle = {
137
159
  * used the URL path (`/execute`, `/read-file`, `/write-file`,
138
160
  * `/healthz`) — over vsock the same selector rides in the framed JSON.
139
161
  */
140
- export type AgentRequest = {
162
+ export type AgentRequest = ({
141
163
  readonly op: 'execute';
142
164
  readonly body: ExecRequest;
143
165
  } | {
144
166
  readonly op: 'reserve-execution';
167
+ readonly body?: {
168
+ readonly executionId: string;
169
+ };
145
170
  } | {
146
171
  readonly op: 'cancel-execution';
147
172
  readonly body: {
148
173
  readonly executionId: string;
149
174
  };
175
+ } | {
176
+ readonly op: 'attach-execution';
177
+ readonly body: {
178
+ readonly executionId: string;
179
+ readonly fromOffset?: number;
180
+ };
150
181
  } | {
151
182
  readonly op: 'read-file';
152
183
  readonly body: ReadFileRequest;
184
+ } | {
185
+ readonly op: 'read-file-stream';
186
+ readonly body: ReadFileStreamRequest;
153
187
  } | {
154
188
  readonly op: 'write-file';
155
189
  readonly body: WriteFileRequest;
156
190
  } | {
157
191
  readonly op: 'terminal';
158
192
  readonly body: TerminalOpenRequest;
193
+ } | {
194
+ readonly op: 'attach-session';
195
+ readonly body: AttachSessionRequest;
196
+ } | {
197
+ readonly op: 'start-detached';
198
+ readonly body: StartDetachedRequest;
199
+ } | {
200
+ readonly op: 'list-sessions';
201
+ } | {
202
+ readonly op: 'kill-session';
203
+ readonly body: KillSessionRequest;
159
204
  } | {
160
205
  readonly op: 'tcp-connect';
161
206
  readonly body: TcpConnectRequest;
207
+ } | {
208
+ readonly op: 'quiesce';
209
+ readonly body: QuiesceRequest;
210
+ } | {
211
+ readonly op: 'flush';
212
+ readonly body: FlushRequest;
162
213
  } | {
163
214
  readonly op: 'healthz';
164
- };
215
+ }) & AgentRequestCredential;
165
216
  export interface VsockTransportOptions {
166
217
  /** Per-attempt connect + handshake timeout. Default 5000ms. */
167
218
  readonly connectTimeoutMs?: number;
@@ -177,6 +228,253 @@ export interface VsockTransportOptions {
177
228
  * against the agent's fresh listen socket. Default 60000ms.
178
229
  */
179
230
  readonly readIdleTimeoutMs?: number;
231
+ /**
232
+ * Interval, in milliseconds, of the per-stream liveness heartbeat on
233
+ * `openTerminal` and `openTcpConnection`. See `protocol.ts`'s
234
+ * `StreamHeartbeat` for the negotiation and what a heartbeat does and
235
+ * does not prove.
236
+ *
237
+ * **Undefined by default, and deliberately so.** This transport is
238
+ * shared with the Firecracker tier, where a default would force-close an
239
+ * existing consumer's quiet-but-alive terminal after three intervals —
240
+ * a changed default for a tier that asked for nothing. The Kubernetes
241
+ * backend opts in (`backends/kubernetes/index.ts`'s
242
+ * `DEFAULT_STREAM_HEARTBEAT_MS`); every other caller that passes nothing
243
+ * sends and expects exactly the frames it always did.
244
+ *
245
+ * A value of `0` or less is the same as leaving it out.
246
+ */
247
+ readonly heartbeatMs?: number;
248
+ /**
249
+ * Fires once per successful dial with how long the connect took, in
250
+ * milliseconds. Never fires with the handle's `token` or any request
251
+ * content — a bare number. Used by callers that build their own
252
+ * `RemoteExecutionAdapter` on top of this transport (the kubernetes
253
+ * backend's `KubernetesAgentTransport`) to attribute wall time; the
254
+ * vsock/mtls/unix arms are free to ignore it.
255
+ */
256
+ readonly onDial?: (durationMs: number) => void;
257
+ /**
258
+ * Fires once per connect ATTEMPT, immediately before it is made, and
259
+ * carries nothing at all.
260
+ *
261
+ * {@link onDial} above only ever fires for an attempt that SUCCEEDED, and
262
+ * a caller that has to know a socket was never established cannot learn
263
+ * it from the attempt's error either: an attempt can be aborted — by its
264
+ * caller, or by a deadline shorter than {@link connectTimeoutMs} — before
265
+ * it has failed, and the abort is what the caller is then holding. Paired
266
+ * with `onDial`, this says "a dial was attempted and none of them handed
267
+ * back a socket", which on the `tcp` arm is exactly "nothing reached the
268
+ * guest": that arm resolves only on the socket's own `connect` event, so
269
+ * no byte can have been sent before `onDial` fired.
270
+ *
271
+ * Used by the kubernetes backend's `KubernetesAgentTransport` to decide
272
+ * whether a failed `exec()` may be retried against a replaced pod. The
273
+ * vsock/mtls/unix arms are free to ignore it.
274
+ */
275
+ readonly onDialAttempt?: () => void;
276
+ /**
277
+ * Fires once for every reply this transport reads that the guest
278
+ * answered on an AUTHENTICATED basis — one control/file reply per
279
+ * `request`, and the opening `ready` frame of a terminal, a session
280
+ * attachment or a TCP stream.
281
+ *
282
+ * It carries the reply itself, read only for the optional identity
283
+ * fields `protocol.ts` documents ({@link GUEST_BOOT_ID_FEATURE}), and it
284
+ * is an OBSERVER: it cannot change the reply, it is called after the
285
+ * reply has been accepted, and a listener that throws is that listener's
286
+ * problem — never the caller's, whose result is already decided.
287
+ *
288
+ * Absent by default, which is what keeps this shared transport's
289
+ * behaviour identical for the Firecracker tier. The kubernetes backend
290
+ * sets it to follow the guest PROCESS behind a handle whose pod uid — its
291
+ * bind token — cannot change when the container is restarted in place.
292
+ */
293
+ readonly onGuestReply?: (reply: GuestReplyIdentity) => void;
294
+ /**
295
+ * The largest `writeFile` body this transport will accept, in raw
296
+ * bytes. Default {@link DEFAULT_MAX_WRITE_FILE_BYTES} (1 GiB).
297
+ *
298
+ * A body above one frame is written in parts (see
299
+ * {@link VsockAgentTransport.writeFile}), so nothing about the wire
300
+ * stops a caller handing over a body larger than the guest's disk or
301
+ * this process's heap. This is the bound that says no first, by a
302
+ * number the caller chose, with {@link AgentWriteFileTooLargeError}
303
+ * naming it — rather than by an out-of-memory or an ENOSPC halfway
304
+ * through a sequence of parts.
305
+ *
306
+ * Checked before the route is chosen, so it caps EVERY body — a value
307
+ * set below what one frame carries caps the small single-frame writes
308
+ * too, which is the range a host capping what a caller may push into a
309
+ * workspace would most plausibly set it to.
310
+ */
311
+ readonly maxWriteFileBytes?: number;
312
+ /**
313
+ * Raw bytes per part when a `writeFile` body is written in parts.
314
+ * Defaults to the largest part one frame can carry, and is clamped
315
+ * DOWN to that: a value above what a frame admits is not a way to
316
+ * send a bigger frame.
317
+ *
318
+ * Setting it also lowers the size at which a body is split at all,
319
+ * so a suite can exercise a multi-part write without allocating one.
320
+ * Leave it unset in production: the default is the fewest round trips
321
+ * the pre-auth ceiling allows.
322
+ */
323
+ readonly writeFilePartBytes?: number;
324
+ /**
325
+ * Say that a failed connect attempt will NOT fix itself, so the retry
326
+ * budget above is not worth spending on it.
327
+ *
328
+ * Absent by default, which keeps every dial retrying for the whole budget
329
+ * exactly as it always has — the budget exists because the common connect
330
+ * failure IS transient (an agent re-listening after a resume answers
331
+ * `ECONNREFUSED` for a moment). The exception is a failure that is about
332
+ * the CALLER's own environment rather than the guest's: the kubernetes
333
+ * backend's Service FQDN does not resolve on a host with no cluster DNS,
334
+ * and re-asking the same resolver the same question for half a minute
335
+ * only ensures the caller's own deadline expires first and reports a
336
+ * timeout in place of the diagnosis.
337
+ *
338
+ * Called with each attempt's error, before the backoff. Returning true
339
+ * ends the loop; the error thrown is the same wrapper as an exhausted
340
+ * budget, carrying that attempt's failure as its `cause`.
341
+ */
342
+ readonly permanentDialFailure?: (error: unknown) => boolean;
343
+ }
344
+ /**
345
+ * The guest agent's default pre-auth frame ceiling for a routed (`tcp`)
346
+ * connection — mirrors `agent.cjs`'s `MAX_PREAUTH_FRAME_BYTES`
347
+ * (`NAMZU_AGENT_MAX_PREAUTH_FRAME_BYTES`, default 8 MiB). The gate on
348
+ * that side cannot run until a whole frame is parsed (the credential
349
+ * rides inside the envelope), so it bounds what an UNAUTHENTICATED
350
+ * connection's first frame may announce. This transport dials fresh per
351
+ * call — there is no persistent, already-authenticated connection to
352
+ * reuse — so EVERY `tcp` request is that connection's first frame, and
353
+ * this ceiling is therefore the effective per-request budget, not just
354
+ * a one-time cost paid on first use.
355
+ *
356
+ * Checked here, client-side, BEFORE dialing: an oversized request fails
357
+ * fast with a clear error instead of opening a connection the guest is
358
+ * going to refuse anyway. Chunking a `write-file` body across multiple
359
+ * frames would lift this ceiling; it is a documented follow-up, not
360
+ * implemented by this transport.
361
+ */
362
+ export declare const TCP_PREAUTH_FRAME_LIMIT_BYTES: number;
363
+ /**
364
+ * The guest agent's default ceiling on ANY frame, pre-auth or not —
365
+ * mirrors `agent.cjs`'s `MAX_FRAME_BYTES` (`NAMZU_AGENT_MAX_FRAME_BYTES`,
366
+ * default 256 MiB). It is the budget the `unix`/`vsock`/`mtls` arms are
367
+ * bounded by, since none of them runs a credential gate and so none of
368
+ * them ever pays the smaller pre-auth price.
369
+ */
370
+ export declare const GUEST_FRAME_LIMIT_BYTES: number;
371
+ /** Default {@link VsockTransportOptions.maxWriteFileBytes} — 1 GiB. */
372
+ export declare const DEFAULT_MAX_WRITE_FILE_BYTES: number;
373
+ /**
374
+ * Idle time before the kernel sends its first TCP keepalive probe on a
375
+ * `tcp`-arm connection, host side. 15 s, matching the Kubernetes backend's
376
+ * default heartbeat interval, so the two bounds do not disagree about how
377
+ * long a silent connection is allowed to look healthy.
378
+ */
379
+ export declare const TCP_KEEPALIVE_INITIAL_DELAY_MS = 15000;
380
+ /**
381
+ * Thrown when a `tcp`-handle request's framed envelope (op + body +
382
+ * token) would exceed {@link TCP_PREAUTH_FRAME_LIMIT_BYTES}. Named so a
383
+ * caller can distinguish "this body needs chunking" from every other
384
+ * transport failure.
385
+ */
386
+ export declare class AgentPreauthFrameTooLargeError extends Error {
387
+ constructor(message: string);
388
+ }
389
+ /**
390
+ * Thrown when a `writeFile` body exceeds
391
+ * {@link VsockTransportOptions.maxWriteFileBytes}. Distinct from
392
+ * {@link AgentPreauthFrameTooLargeError}: that one says the WIRE cannot
393
+ * carry this in one frame (and, since the part protocol, only ever fires
394
+ * when the guest cannot carry it in several either), this one says the
395
+ * HOST was configured not to send a body this large at all.
396
+ */
397
+ export declare class AgentWriteFileTooLargeError extends Error {
398
+ constructor(message: string);
399
+ }
400
+ /**
401
+ * Thrown when a read this transport cannot serve on the old whole-file op
402
+ * is asked of a guest that does not advertise
403
+ * {@link READ_FILE_STREAM_FEATURE} — a ranged `readFile`, or any
404
+ * `readFileStream`.
405
+ *
406
+ * A refusal rather than a fallback, and that is the whole point of the
407
+ * class: an agent that predates the feature IGNORES `offset`/`length` and
408
+ * answers with the WHOLE file, so silently taking the old path would hand
409
+ * a caller the entire file where it asked for a slice — a wrong answer
410
+ * dressed as a degraded one. Named so a host can tell "rebuild the guest
411
+ * image" apart from "that file is not there".
412
+ */
413
+ export declare class AgentReadFileStreamUnsupportedError extends Error {
414
+ constructor(message: string);
415
+ }
416
+ /**
417
+ * Thrown by {@link VsockAgentTransport}'s dial when it gives up without a
418
+ * socket — the retry budget spent, or the failure declared one waiting cannot
419
+ * cure. The underlying connect failure is its `cause`.
420
+ *
421
+ * A class, and not only a phrase in the message, because callers classify on
422
+ * it: "the failure came out of the dial" means NOTHING was sent to the guest,
423
+ * which is what makes a retry against a replaced pod safe (the kubernetes
424
+ * backend's `agentAddress: 'pod-ip'` re-read). A message a guest can quote
425
+ * back — a command's stderr, a path, a proxy's own error — cannot be allowed
426
+ * to claim that.
427
+ */
428
+ /**
429
+ * The two fields that turn a connection-bound terminal into a session, on
430
+ * the host's side of {@link VsockAgentTransport.openTerminal}.
431
+ *
432
+ * Both are optional and both are ignored by a guest that predates the
433
+ * session registry, which is why the caller — never this transport — is the
434
+ * one that checks the guest advertises the capability first.
435
+ */
436
+ export interface SessionTerminalOpen {
437
+ readonly sessionId?: string;
438
+ readonly persistent?: boolean;
439
+ }
440
+ /**
441
+ * One open terminal stream, plus what only a SESSION's reader needs: the
442
+ * guest's opening frame, the byte offset to come back at, and a way to stop
443
+ * reading without signalling the program.
444
+ */
445
+ export interface AgentTerminalStream {
446
+ readonly session: TerminalSession;
447
+ /** The guest's `ready` frame. Carries the session fields, when there are any. */
448
+ readonly ready: TerminalReadyEvent;
449
+ /** One past the newest retained byte this stream has delivered, if any. */
450
+ nextOffset(): number | undefined;
451
+ /**
452
+ * End this attachment locally. Nothing is signalled in the guest: the
453
+ * program goes on running and its output goes on filling the retained
454
+ * log, which is the entire difference between this and `kill`.
455
+ */
456
+ detach(): void;
457
+ }
458
+ /**
459
+ * Thrown — as the rejection of a session terminal's `exited` — when the
460
+ * attachment ended and the program did not.
461
+ *
462
+ * `exited` may not RESOLVE here: a resolved `exited` says the program is
463
+ * over, and reporting `exitCode: -1` for a shell that is still running in
464
+ * the pod is exactly the confusion this whole feature exists to remove. The
465
+ * offset is carried because it is what the next attach resumes from.
466
+ */
467
+ export declare class AgentSessionDetachedError extends Error {
468
+ readonly nextOffset: number | undefined;
469
+ readonly name = "AgentSessionDetachedError";
470
+ constructor(nextOffset: number | undefined, message: string, options?: {
471
+ cause?: unknown;
472
+ });
473
+ }
474
+ export declare class AgentDialFailedError extends Error {
475
+ constructor(message: string, options?: {
476
+ cause?: unknown;
477
+ });
180
478
  }
181
479
  /** Exact guest wire version accepted by this Firecracker transport. */
182
480
  export declare const FIRECRACKER_AGENT_PROTOCOL_VERSION: 2;
@@ -185,11 +483,28 @@ declare function frame(payload: string): Buffer;
185
483
  * Incremental frame reader. Feed it socket chunks; it yields complete
186
484
  * payloads. A zero-length frame is the exec stream terminator and is
187
485
  * surfaced as an empty string so the caller can stop.
486
+ *
487
+ * It accumulates into ONE buffer it grows geometrically, with a read
488
+ * cursor, rather than re-`concat`ing every arriving chunk onto a fresh
489
+ * allocation. The distinction only matters for a large frame, where it is
490
+ * the difference between linear and quadratic: a `read-file` reply for a
491
+ * 64 MiB file arrives as ~1400 socket chunks, and copying everything
492
+ * received so far onto each one of them spent half a minute of memcpy on
493
+ * a reply the socket delivered in under a second. Identical framing,
494
+ * identical errors, identical `bufferedBytes` — only the copying changes.
188
495
  */
189
496
  declare class FrameReader {
190
497
  private buf;
498
+ /** First byte not yet handed out as part of a frame. */
499
+ private start;
500
+ /** One past the last byte received. */
501
+ private end;
191
502
  push(chunk: Buffer): string[];
192
503
  get bufferedBytes(): number;
504
+ /** Copy `chunk` in, growing (and first compacting) only when needed. */
505
+ private append;
506
+ /** Mark `bytes` from the read cursor as consumed. */
507
+ private consume;
193
508
  }
194
509
  /**
195
510
  * The transport. One instance per sandbox handle; every request opens
@@ -203,13 +518,31 @@ export declare class VsockAgentTransport {
203
518
  private readonly connectRetryBudgetMs;
204
519
  private readonly connectRetryIntervalMs;
205
520
  private readonly readIdleTimeoutMs;
521
+ /** Undefined → this transport negotiates no heartbeat at all. */
522
+ private readonly heartbeatMs?;
523
+ private readonly onDial?;
524
+ private readonly onDialAttempt?;
525
+ private readonly onGuestReply?;
526
+ private readonly maxWriteFileBytes;
527
+ private readonly writeFilePartBytes?;
528
+ /**
529
+ * What the guest advertised in `healthz`, cached for this handle's
530
+ * lifetime. A pod does not swap its agent binary while it is running,
531
+ * so the probe is asked once per transport and only when something
532
+ * actually depends on a capability — an ordinary write, exec, read or
533
+ * terminal pays nothing for it.
534
+ */
535
+ private guestFeatureList?;
536
+ private readonly permanentDialFailure?;
206
537
  private readonly executionController;
207
538
  constructor(handle: SandboxAgentHandle, options?: VsockTransportOptions);
208
539
  /**
209
540
  * Dial the agent with the resume-survival retry budget. Resolves a
210
541
  * connected, post-handshake socket. Retries connect/handshake
211
542
  * failures (ECONNREFUSED while the agent re-listens after a resume,
212
- * a dropped CONNECT ack) until the budget is exhausted.
543
+ * a dropped CONNECT ack) until the budget is exhausted — or until
544
+ * {@link VsockTransportOptions.permanentDialFailure} says this particular
545
+ * failure is not one waiting will cure.
213
546
  */
214
547
  private dial;
215
548
  private connectOnce;
@@ -236,6 +569,37 @@ export declare class VsockAgentTransport {
236
569
  * CONNECT-ack path) before resolving — today it does not.
237
570
  */
238
571
  private connectOnceMtls;
572
+ /**
573
+ * Dial a `tcp` handle: a plain `net.connect({ host, port })`. NO
574
+ * routing preamble and NO ack — there is no relay to route through
575
+ * and no handshake line the guest expects, so the socket is handed
576
+ * to the framing loop the instant it connects, exactly like the
577
+ * `unix` arm above (whose body this deliberately does not touch).
578
+ */
579
+ private connectOnceTcp;
580
+ /**
581
+ * Fold the handle's credential into a request envelope. Only the
582
+ * `tcp` arm carries a token — the vsock/unix/mtls control channels
583
+ * are host↔guest only and the agent authenticates nothing there, so
584
+ * a request built for those arms passes through unchanged.
585
+ */
586
+ private withCredential;
587
+ /**
588
+ * Refuse an oversized `tcp` envelope BEFORE dialing. See
589
+ * {@link TCP_PREAUTH_FRAME_LIMIT_BYTES}. A no-op for every other
590
+ * handle kind, which the guest never gates on frame size pre-auth.
591
+ */
592
+ private assertPreauthBudget;
593
+ /**
594
+ * Hand one accepted reply to {@link VsockTransportOptions.onGuestReply},
595
+ * and never let the listener's failure reach the caller.
596
+ *
597
+ * The caller's result is already decided by the time this runs — the
598
+ * reply parsed, the frame accounted for — so a hook that throws must not
599
+ * turn a successful read into a failed one. Swallowing is the only
600
+ * behaviour that keeps an optional observer optional.
601
+ */
602
+ private observeGuestReply;
239
603
  /**
240
604
  * Send one framed request and read one framed JSON reply (file-IO +
241
605
  * healthz). Applies the read-idle timeout so a post-resume hung read
@@ -256,6 +620,20 @@ export declare class VsockAgentTransport {
256
620
  */
257
621
  execute(body: ExecRequest, opts?: SandboxExecOptions, signal?: AbortSignal): Promise<SandboxExecResult>;
258
622
  exec(command: string, argv?: string[], opts?: SandboxExecOptions): Promise<SandboxExecResult>;
623
+ /**
624
+ * The raw `/execute` primitive with NO admission/reservation
625
+ * semantics: dial, send one framed request, accumulate the streamed
626
+ * NDJSON reply. Public (unlike the identical-in-spirit
627
+ * {@link reserveExecution}/{@link cancelExecution}, reached through
628
+ * the already-public {@link request}) so a caller running its OWN
629
+ * {@link RemoteExecutionController} — the kubernetes backend's
630
+ * adapter — can reuse this exact dial + framing rather than
631
+ * reimplementing it, while still supplying that controller its own
632
+ * `reserve`/`cancel`/`execute` triple as {@link RemoteExecutionAdapter}
633
+ * requires. Callers that just want a reserve-before-admission `exec`
634
+ * should use {@link execute} or {@link exec} instead.
635
+ */
636
+ executeStreamed(body: ExecRequest, opts?: SandboxExecOptions, signal?: AbortSignal): Promise<SandboxExecResult>;
259
637
  private reserveExecution;
260
638
  private cancelExecution;
261
639
  /** Readiness probe. A healthy guest must also speak the exact host protocol. */
@@ -267,8 +645,210 @@ export declare class VsockAgentTransport {
267
645
  * post-create readiness fence.
268
646
  */
269
647
  waitForReady(timeoutMs: number, pollIntervalMs: number, signal?: AbortSignal): Promise<void>;
270
- writeFile(path: string, content: Buffer): Promise<void>;
271
- readFile(path: string): Promise<Buffer>;
648
+ /**
649
+ * Write a whole file into the guest workspace.
650
+ *
651
+ * A body that fits one frame goes as it always has: a single
652
+ * `write-file` envelope carrying the base64 content, one round trip,
653
+ * byte-for-byte the request this transport has always sent.
654
+ *
655
+ * A body that does NOT fit is the case this exists for. On the `tcp`
656
+ * arm every request dials a fresh connection, so every request is that
657
+ * connection's first, not-yet-authenticated frame (the credential
658
+ * rides in the envelope) and is bounded by
659
+ * {@link TCP_PREAUTH_FRAME_LIMIT_BYTES} — about 5.9 MiB of file
660
+ * content — on EVERY call, not once. Raising the guest's
661
+ * `NAMZU_AGENT_MAX_PREAUTH_FRAME_BYTES` trades away the pre-auth
662
+ * budget that ceiling exists to bound, so the fix is on this side:
663
+ * split the body into parts that each fit, append them to a temporary
664
+ * SIBLING of the target inside the same workspace jail, and finish
665
+ * with an atomic `rename` onto the target.
666
+ *
667
+ * What that buys, and why the shape is what it is:
668
+ *
669
+ * - **A reader never sees a half-written file.** The target changes
670
+ * exactly once, in the final part's `rename`. A sequence that dies
671
+ * at part 3 of 9 leaves the target exactly as it was — including
672
+ * not existing.
673
+ * - **A lost or duplicated part is detected, not written.** Each part
674
+ * names the offset it starts at and the guest refuses it unless
675
+ * that equals the temp file's current size.
676
+ * - **An abandoned sequence cleans up after itself.** An abort or a
677
+ * transport failure removes the temp file (best effort — a peer
678
+ * that has gone away cannot be asked to) and rejects.
679
+ * - **Parts go out sequentially on fresh connections**, which is what
680
+ * the offset check assumes and what keeps the guest's pre-auth
681
+ * connection pool holding one of this caller's sockets at a time.
682
+ *
683
+ * The guest must ADVERTISE the capability (`${WRITE_FILE_PARTS_FEATURE}`
684
+ * in its `healthz` reply) before a single part is sent. An agent that
685
+ * predates the part protocol would read a part's `content` as a whole
686
+ * file; it never receives one, and an oversized body against such a
687
+ * guest still fails with the named
688
+ * {@link AgentPreauthFrameTooLargeError} it always did.
689
+ *
690
+ * {@link VsockTransportOptions.maxWriteFileBytes} is checked FIRST, on
691
+ * every body and before anything about the wire is considered: it is a
692
+ * bound on what a caller may push into a workspace, not a bound on
693
+ * multi-part writes, so a host that lowers it below the frame budget
694
+ * gets the cap it asked for rather than none.
695
+ */
696
+ writeFile(path: string, content: Buffer, signal?: AbortSignal): Promise<void>;
697
+ /** Today's single-frame write, unchanged — see {@link writeFile}. */
698
+ private writeFileWhole;
699
+ /**
700
+ * The frame budget one request has on this handle: the pre-auth
701
+ * ceiling on the credentialed `tcp` arm, the guest's global frame
702
+ * ceiling on the host-local arms, which authenticate nothing and so
703
+ * never pay the smaller price.
704
+ */
705
+ private singleFrameBudgetBytes;
706
+ /**
707
+ * Exact framed size of a `write-file` envelope whose `content` field
708
+ * holds `contentBytes` raw bytes base64-encoded — WITHOUT encoding
709
+ * them, so sizing a 1 GiB body costs nothing and never builds a string
710
+ * longer than V8 permits.
711
+ *
712
+ * Exact rather than approximate because the base64 alphabet contains
713
+ * no character `JSON.stringify` escapes, so the encoded content
714
+ * contributes precisely its own length to the envelope and the rest of
715
+ * the envelope (paths, the token, the part fields) is measured as it
716
+ * will actually be serialized.
717
+ */
718
+ private writeFileEnvelopeBytes;
719
+ /**
720
+ * Ask the guest whether it implements the part protocol. Cached for
721
+ * the transport's lifetime; a transport failure propagates rather than
722
+ * reading as "not supported", because answering a broken connection
723
+ * with a too-large error would name the wrong cause.
724
+ */
725
+ private guestSupportsWriteFileParts;
726
+ /**
727
+ * The capability strings this guest advertises in `healthz`, cached for
728
+ * this transport's lifetime.
729
+ *
730
+ * One list, asked once, for every optional op: the write-file part
731
+ * protocol and the execution-attach ops both read it, and a second
732
+ * cache would mean a second probe against a guest that answers both
733
+ * questions in one reply. A transport FAILURE propagates rather than
734
+ * reading as "not supported", because answering a broken connection
735
+ * with a capability refusal would name the wrong cause.
736
+ */
737
+ guestFeatures(signal?: AbortSignal): Promise<readonly string[]>;
738
+ /**
739
+ * Send one framed request and read a STREAM of framed JSON events until
740
+ * the agent's zero-length terminator, handing each event to `onEvent`.
741
+ *
742
+ * The generic half of what {@link executeRaw} does, without any of its
743
+ * opinions about what the events mean: `executeRaw` owns the exec
744
+ * NDJSON union and the {@link ExecResultAccumulator}, and this owns
745
+ * dial, framing, the terminator, the observation bound and the
746
+ * post-terminator close. Additive — nothing already shipped calls it —
747
+ * so the Firecracker tier's behaviour is untouched and a later streamed
748
+ * op reuses it rather than writing a fourth copy of this loop.
749
+ *
750
+ * There is deliberately NO read-idle timeout. A stream that exists to
751
+ * follow a long, quiet command must not be torn down for being quiet;
752
+ * the whole observation is bounded by `observationTimeoutMs` instead,
753
+ * exactly as an `execute` stream is.
754
+ */
755
+ streamFramedRequest(req: AgentRequest, onEvent: (event: Record<string, unknown>) => void, options: {
756
+ readonly observationTimeoutMs: number;
757
+ }, signal?: AbortSignal): Promise<void>;
758
+ /** The refusal for a body no frame can carry and no guest can take in parts. */
759
+ private oversizedWriteFileError;
760
+ /**
761
+ * Write `content` to `target` as a sequence of parts. See
762
+ * {@link writeFile} for why this shape.
763
+ */
764
+ private writeFileInParts;
765
+ /**
766
+ * Remove an abandoned part file. Best effort BY CONTRACT: the reason
767
+ * the sequence failed is frequently that the guest is unreachable, and
768
+ * a cleanup that threw would replace the caller's real error — the one
769
+ * that says why the write failed — with a second one about tidying up.
770
+ */
771
+ private discardWriteFileTemp;
772
+ /**
773
+ * Read a file out of the guest.
774
+ *
775
+ * Three shapes, decided by what the guest advertises and what the
776
+ * caller asked for:
777
+ *
778
+ * - **No options, guest advertises {@link READ_FILE_STREAM_FEATURE}** —
779
+ * served by {@link readFileStream} and concatenated here. Neither
780
+ * side ever holds the base64 form or the JSON envelope whole, so the
781
+ * ~384 MiB ceiling (V8 refuses a string longer than `0x1fffffe8`
782
+ * characters, which is what a base64-encoded file of that size
783
+ * needs) is gone and the guest's peak stops tracking the file's
784
+ * size. The result is still one `Buffer`, because that is what this
785
+ * method returns; a caller that must not hold even that iterates
786
+ * {@link readFileStream} directly.
787
+ * - **No options, guest does not advertise it** — today's single
788
+ * whole-file reply, byte for byte, with today's ceiling.
789
+ * - **`offset` and `length`** — one ranged `read-file`: a single round
790
+ * trip for a single slice, which is the point of asking for one.
791
+ * `offset` WITHOUT `length` is an unbounded tail, so it goes through
792
+ * the stream instead; the guest refuses an uncapped range on
793
+ * `read-file` for exactly that reason.
794
+ *
795
+ * A ranged read against a guest that does not advertise the feature is
796
+ * REFUSED with {@link AgentReadFileStreamUnsupportedError} rather than
797
+ * downgraded: such an agent ignores `offset`/`length` and answers with
798
+ * the whole file, which the caller would read as its slice.
799
+ */
800
+ readFile(path: string, options?: SandboxReadFileOptions): Promise<Buffer>;
801
+ /** Today's single whole-file reply, unchanged — see {@link readFile}. */
802
+ private readFileWhole;
803
+ /**
804
+ * One bounded slice, in one round trip.
805
+ *
806
+ * The guest's own ceiling on a range (`NAMZU_AGENT_READ_FILE_RANGE_BYTES`,
807
+ * 1 MiB by default) is not mirrored here and deliberately so: it is the
808
+ * DEPLOYMENT's number, a host that guessed it would refuse ranges the
809
+ * guest would have served, and the guest's refusal already names the
810
+ * variable that raises it.
811
+ */
812
+ private readFileRange;
813
+ /**
814
+ * Read a file as an ordered sequence of chunks, so neither side holds
815
+ * the whole of it.
816
+ *
817
+ * The guest sends `meta`, then `data` frames, then `end`, then the
818
+ * zero-length terminator — the same terminated-stream shape `execute`
819
+ * uses. Two bounds keep this side's heap flat while the guest's stays
820
+ * flat on its own: the socket is PAUSED once
821
+ * {@link READ_FILE_STREAM_HIGH_WATER_BYTES} of decoded chunks are
822
+ * waiting for a slow consumer, and the guest itself waits for each
823
+ * `data` frame to drain before it reads the next one.
824
+ *
825
+ * Leaving the loop early — `break`, an exception, an aborted
826
+ * `options.signal` — destroys the socket in the generator's `finally`,
827
+ * which is what makes the guest close its fd: it sees the connection go
828
+ * and releases the descriptor rather than leaking one per abandoned
829
+ * read.
830
+ *
831
+ * Refuses a guest that does not advertise
832
+ * {@link READ_FILE_STREAM_FEATURE} before dialing, with
833
+ * {@link AgentReadFileStreamUnsupportedError}.
834
+ */
835
+ readFileStream(path: string, options?: SandboxReadFileOptions): AsyncGenerator<Buffer, void, undefined>;
836
+ /**
837
+ * The stream itself. Private, and one argument wider than
838
+ * {@link readFileStream}: `onMeta` fires once, with the guest's `meta`
839
+ * frame, before the first chunk is yielded, which is how
840
+ * {@link readFile} sizes its destination buffer without a second round
841
+ * trip and without a public parameter nobody outside this class should
842
+ * pass.
843
+ */
844
+ private readFileFrames;
845
+ /**
846
+ * Ask the guest whether it implements ranged and streamed reads.
847
+ * Cached for the transport's lifetime, exactly as
848
+ * {@link guestSupportsWriteFileParts} is, and for the same reason: a
849
+ * pod does not swap its agent binary while it is running.
850
+ */
851
+ private guestSupportsReadFileStream;
272
852
  /**
273
853
  * Open a real PTY owned by the in-VM agent.
274
854
  *
@@ -278,7 +858,55 @@ export declare class VsockAgentTransport {
278
858
  * browser never reaches this transport directly; the runtime gateway owns
279
859
  * the session and its authenticated WebSocket attachment.
280
860
  */
281
- openTerminal(options: OpenTerminalOptions): Promise<TerminalSession>;
861
+ openTerminal(options: OpenTerminalOptions & SessionTerminalOpen): Promise<TerminalSession>;
862
+ /**
863
+ * The same open, handing back the session's own handles as well as the
864
+ * `TerminalSession` — the offset to come back at, and the detach that
865
+ * ends the attachment without signalling the program.
866
+ *
867
+ * Exactly one code path serves both: a persistent terminal is not a
868
+ * second kind of terminal, it is the same stream with a different answer
869
+ * to "what does a closed connection mean".
870
+ */
871
+ openSessionTerminal(options: OpenTerminalOptions & SessionTerminalOpen): Promise<AgentTerminalStream>;
872
+ /**
873
+ * The framed, bidirectional stream behind every terminal this transport
874
+ * opens — the one the `terminal` op starts, and the one `attach-session`
875
+ * joins to a terminal that is already running.
876
+ *
877
+ * Parameterised rather than copied, because the two differ in exactly two
878
+ * places and everything else — the dial, the framing, the ready
879
+ * handshake, the read-idle timer that is cleared once a shell may
880
+ * legitimately go quiet, the heartbeat, the output buffering before the
881
+ * first listener, the kill grace — has to behave identically or a
882
+ * reattached terminal is a second terminal implementation with its own
883
+ * bugs. The two differences:
884
+ *
885
+ * - **`detachable`.** For a connection-bound terminal a lost stream IS
886
+ * the end of the program, and `exited` resolves with `exitCode: -1`
887
+ * exactly as it always has. For a session attachment it is not: the
888
+ * program is still running in the pod, so `exited` REJECTS with
889
+ * {@link AgentSessionDetachedError} rather than reporting an exit that
890
+ * did not happen. The rejection is pre-handled here so a caller that
891
+ * only reads output cannot take the host process down with an
892
+ * unhandled rejection.
893
+ * - **the offsets.** A session stream's frames carry their place in the
894
+ * guest's retained log, and {@link AgentTerminalStream.nextOffset} is
895
+ * what a reattach resumes from. It is never computed from the decoded
896
+ * text: a chunk that ends mid-character decodes wider than the bytes
897
+ * it replaced.
898
+ */
899
+ private openTerminalStream;
900
+ /**
901
+ * Join a terminal session that is already running in the guest, replaying
902
+ * what it printed from `fromOffset` before following it live.
903
+ *
904
+ * The guest allows ONE attachment per session and ends the previous one
905
+ * by name, so two host processes cannot interleave keystrokes into one
906
+ * shell. Nothing here signals the program: releasing this stream is a
907
+ * detach, and ending the session is `kill-session`.
908
+ */
909
+ attachSessionTerminal(request: AttachSessionRequest): Promise<AgentTerminalStream>;
282
910
  /** Open one TCP stream to a service listening on guest loopback. */
283
911
  openTcpConnection(options: SandboxTcpConnectOptions): Promise<SandboxTcpConnection>;
284
912
  }