@namzu/sandbox 14.0.0 → 16.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. package/CHANGELOG.md +924 -0
  2. package/README.md +369 -14
  3. package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
  4. package/dist/backends/aci-standby-pool/index.js +13 -1
  5. package/dist/backends/aci-standby-pool/index.js.map +1 -1
  6. package/dist/backends/docker/index.d.ts +169 -6
  7. package/dist/backends/docker/index.d.ts.map +1 -1
  8. package/dist/backends/docker/index.js +499 -85
  9. package/dist/backends/docker/index.js.map +1 -1
  10. package/dist/backends/firecracker/index.d.ts.map +1 -1
  11. package/dist/backends/firecracker/index.js +12 -2
  12. package/dist/backends/firecracker/index.js.map +1 -1
  13. package/dist/backends/firecracker/protocol.d.ts +459 -8
  14. package/dist/backends/firecracker/protocol.d.ts.map +1 -1
  15. package/dist/backends/firecracker/protocol.js +136 -0
  16. package/dist/backends/firecracker/protocol.js.map +1 -1
  17. package/dist/backends/firecracker/transport.d.ts +539 -6
  18. package/dist/backends/firecracker/transport.d.ts.map +1 -1
  19. package/dist/backends/firecracker/transport.js +1171 -24
  20. package/dist/backends/firecracker/transport.js.map +1 -1
  21. package/dist/backends/kubernetes/egress-policy.d.ts +1181 -13
  22. package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -1
  23. package/dist/backends/kubernetes/egress-policy.js +2350 -31
  24. package/dist/backends/kubernetes/egress-policy.js.map +1 -1
  25. package/dist/backends/kubernetes/identity.d.ts +193 -0
  26. package/dist/backends/kubernetes/identity.d.ts.map +1 -0
  27. package/dist/backends/kubernetes/identity.js +147 -0
  28. package/dist/backends/kubernetes/identity.js.map +1 -0
  29. package/dist/backends/kubernetes/index.d.ts +678 -33
  30. package/dist/backends/kubernetes/index.d.ts.map +1 -1
  31. package/dist/backends/kubernetes/index.js +1180 -95
  32. package/dist/backends/kubernetes/index.js.map +1 -1
  33. package/dist/backends/kubernetes/ingress-policy.d.ts +375 -0
  34. package/dist/backends/kubernetes/ingress-policy.d.ts.map +1 -0
  35. package/dist/backends/kubernetes/ingress-policy.js +1050 -0
  36. package/dist/backends/kubernetes/ingress-policy.js.map +1 -0
  37. package/dist/backends/kubernetes/k8s-client.d.ts +213 -4
  38. package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -1
  39. package/dist/backends/kubernetes/k8s-client.js +359 -52
  40. package/dist/backends/kubernetes/k8s-client.js.map +1 -1
  41. package/dist/backends/kubernetes/lease.d.ts +40 -14
  42. package/dist/backends/kubernetes/lease.d.ts.map +1 -1
  43. package/dist/backends/kubernetes/lease.js +68 -18
  44. package/dist/backends/kubernetes/lease.js.map +1 -1
  45. package/dist/backends/kubernetes/objects.d.ts +423 -3
  46. package/dist/backends/kubernetes/objects.d.ts.map +1 -1
  47. package/dist/backends/kubernetes/objects.js +364 -2
  48. package/dist/backends/kubernetes/objects.js.map +1 -1
  49. package/dist/backends/kubernetes/per-sandbox-policy.d.ts +219 -0
  50. package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -0
  51. package/dist/backends/kubernetes/per-sandbox-policy.js +375 -0
  52. package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -0
  53. package/dist/backends/kubernetes/rbac.d.ts +153 -0
  54. package/dist/backends/kubernetes/rbac.d.ts.map +1 -0
  55. package/dist/backends/kubernetes/rbac.js +177 -0
  56. package/dist/backends/kubernetes/rbac.js.map +1 -0
  57. package/dist/backends/kubernetes/sandbox.d.ts +81 -14
  58. package/dist/backends/kubernetes/sandbox.d.ts.map +1 -1
  59. package/dist/backends/kubernetes/sandbox.js +149 -15
  60. package/dist/backends/kubernetes/sandbox.js.map +1 -1
  61. package/dist/backends/kubernetes/transport.d.ts +935 -9
  62. package/dist/backends/kubernetes/transport.d.ts.map +1 -1
  63. package/dist/backends/kubernetes/transport.js +1958 -62
  64. package/dist/backends/kubernetes/transport.js.map +1 -1
  65. package/dist/backends/kubernetes/workspace.d.ts +1149 -18
  66. package/dist/backends/kubernetes/workspace.d.ts.map +1 -1
  67. package/dist/backends/kubernetes/workspace.js +2825 -186
  68. package/dist/backends/kubernetes/workspace.js.map +1 -1
  69. package/dist/backends/remote-execution-controller.d.ts +14 -0
  70. package/dist/backends/remote-execution-controller.d.ts.map +1 -1
  71. package/dist/backends/remote-execution-controller.js.map +1 -1
  72. package/dist/index.d.ts +294 -18
  73. package/dist/index.d.ts.map +1 -1
  74. package/dist/index.js +280 -10
  75. package/dist/index.js.map +1 -1
  76. package/dist/testing/sandbox-conformance.d.ts +39 -5
  77. package/dist/testing/sandbox-conformance.d.ts.map +1 -1
  78. package/dist/testing/sandbox-conformance.js +436 -5
  79. package/dist/testing/sandbox-conformance.js.map +1 -1
  80. package/package.json +3 -3
  81. package/src/backends/aci-standby-pool/index.ts +16 -1
  82. package/src/backends/docker/index.ts +617 -100
  83. package/src/backends/firecracker/index.ts +14 -2
  84. package/src/backends/firecracker/protocol.ts +514 -6
  85. package/src/backends/firecracker/transport.ts +1492 -40
  86. package/src/backends/kubernetes/egress-policy.ts +3334 -55
  87. package/src/backends/kubernetes/identity.ts +261 -0
  88. package/src/backends/kubernetes/index.ts +1785 -127
  89. package/src/backends/kubernetes/ingress-policy.ts +1344 -0
  90. package/src/backends/kubernetes/k8s-client.ts +444 -54
  91. package/src/backends/kubernetes/lease.ts +75 -19
  92. package/src/backends/kubernetes/objects.ts +626 -6
  93. package/src/backends/kubernetes/per-sandbox-policy.ts +497 -0
  94. package/src/backends/kubernetes/rbac.ts +192 -0
  95. package/src/backends/kubernetes/sandbox.ts +218 -20
  96. package/src/backends/kubernetes/transport.ts +2733 -124
  97. package/src/backends/kubernetes/workspace.ts +4476 -222
  98. package/src/backends/remote-execution-controller.ts +14 -0
  99. package/src/index.ts +668 -19
  100. package/src/testing/sandbox-conformance.ts +540 -5
@@ -52,8 +52,8 @@
52
52
  * socket to be silently severed by a resume), which makes the
53
53
  * transport resume-survivable by construction.
54
54
  */
55
- import type { OpenTerminalOptions, SandboxExecOptions, SandboxExecResult, SandboxTcpConnectOptions, SandboxTcpConnection, TerminalSession } from '@namzu/sdk';
56
- import { type AgentRequestCredential, type ExecRequest, type ReadFileRequest, type TcpConnectRequest, type TerminalOpenRequest, type WriteFileRequest } from './protocol.js';
55
+ import type { OpenTerminalOptions, SandboxExecOptions, SandboxExecResult, SandboxReadFileOptions, SandboxTcpConnectOptions, SandboxTcpConnection, TerminalSession } from '@namzu/sdk';
56
+ import { type AgentRequestCredential, type AttachSessionRequest, type ExecRequest, type FlushRequest, type GuestReplyIdentity, type KillSessionRequest, type QuiesceRequest, type ReadFileRequest, type ReadFileStreamRequest, type StartDetachedRequest, type TcpConnectRequest, type TerminalOpenRequest, type TerminalReadyEvent, type WriteFileRequest } from './protocol.js';
57
57
  /**
58
58
  * An addressable agent endpoint. The orchestrator hands one of these
59
59
  * back per sandbox (`create()` response → `vsock endpoint`).
@@ -164,23 +164,52 @@ export type AgentRequest = ({
164
164
  readonly body: ExecRequest;
165
165
  } | {
166
166
  readonly op: 'reserve-execution';
167
+ readonly body?: {
168
+ readonly executionId: string;
169
+ };
167
170
  } | {
168
171
  readonly op: 'cancel-execution';
169
172
  readonly body: {
170
173
  readonly executionId: string;
171
174
  };
175
+ } | {
176
+ readonly op: 'attach-execution';
177
+ readonly body: {
178
+ readonly executionId: string;
179
+ readonly fromOffset?: number;
180
+ };
172
181
  } | {
173
182
  readonly op: 'read-file';
174
183
  readonly body: ReadFileRequest;
184
+ } | {
185
+ readonly op: 'read-file-stream';
186
+ readonly body: ReadFileStreamRequest;
175
187
  } | {
176
188
  readonly op: 'write-file';
177
189
  readonly body: WriteFileRequest;
178
190
  } | {
179
191
  readonly op: 'terminal';
180
192
  readonly body: TerminalOpenRequest;
193
+ } | {
194
+ readonly op: 'attach-session';
195
+ readonly body: AttachSessionRequest;
196
+ } | {
197
+ readonly op: 'start-detached';
198
+ readonly body: StartDetachedRequest;
199
+ } | {
200
+ readonly op: 'list-sessions';
201
+ } | {
202
+ readonly op: 'kill-session';
203
+ readonly body: KillSessionRequest;
181
204
  } | {
182
205
  readonly op: 'tcp-connect';
183
206
  readonly body: TcpConnectRequest;
207
+ } | {
208
+ readonly op: 'quiesce';
209
+ readonly body: QuiesceRequest;
210
+ } | {
211
+ readonly op: 'flush';
212
+ readonly body: FlushRequest;
184
213
  } | {
185
214
  readonly op: 'healthz';
186
215
  }) & AgentRequestCredential;
@@ -199,6 +228,23 @@ export interface VsockTransportOptions {
199
228
  * against the agent's fresh listen socket. Default 60000ms.
200
229
  */
201
230
  readonly readIdleTimeoutMs?: number;
231
+ /**
232
+ * Interval, in milliseconds, of the per-stream liveness heartbeat on
233
+ * `openTerminal` and `openTcpConnection`. See `protocol.ts`'s
234
+ * `StreamHeartbeat` for the negotiation and what a heartbeat does and
235
+ * does not prove.
236
+ *
237
+ * **Undefined by default, and deliberately so.** This transport is
238
+ * shared with the Firecracker tier, where a default would force-close an
239
+ * existing consumer's quiet-but-alive terminal after three intervals —
240
+ * a changed default for a tier that asked for nothing. The Kubernetes
241
+ * backend opts in (`backends/kubernetes/index.ts`'s
242
+ * `DEFAULT_STREAM_HEARTBEAT_MS`); every other caller that passes nothing
243
+ * sends and expects exactly the frames it always did.
244
+ *
245
+ * A value of `0` or less is the same as leaving it out.
246
+ */
247
+ readonly heartbeatMs?: number;
202
248
  /**
203
249
  * Fires once per successful dial with how long the connect took, in
204
250
  * milliseconds. Never fires with the handle's `token` or any request
@@ -208,6 +254,92 @@ export interface VsockTransportOptions {
208
254
  * vsock/mtls/unix arms are free to ignore it.
209
255
  */
210
256
  readonly onDial?: (durationMs: number) => void;
257
+ /**
258
+ * Fires once per connect ATTEMPT, immediately before it is made, and
259
+ * carries nothing at all.
260
+ *
261
+ * {@link onDial} above only ever fires for an attempt that SUCCEEDED, and
262
+ * a caller that has to know a socket was never established cannot learn
263
+ * it from the attempt's error either: an attempt can be aborted — by its
264
+ * caller, or by a deadline shorter than {@link connectTimeoutMs} — before
265
+ * it has failed, and the abort is what the caller is then holding. Paired
266
+ * with `onDial`, this says "a dial was attempted and none of them handed
267
+ * back a socket", which on the `tcp` arm is exactly "nothing reached the
268
+ * guest": that arm resolves only on the socket's own `connect` event, so
269
+ * no byte can have been sent before `onDial` fired.
270
+ *
271
+ * Used by the kubernetes backend's `KubernetesAgentTransport` to decide
272
+ * whether a failed `exec()` may be retried against a replaced pod. The
273
+ * vsock/mtls/unix arms are free to ignore it.
274
+ */
275
+ readonly onDialAttempt?: () => void;
276
+ /**
277
+ * Fires once for every reply this transport reads that the guest
278
+ * answered on an AUTHENTICATED basis — one control/file reply per
279
+ * `request`, and the opening `ready` frame of a terminal, a session
280
+ * attachment or a TCP stream.
281
+ *
282
+ * It carries the reply itself, read only for the optional identity
283
+ * fields `protocol.ts` documents ({@link GUEST_BOOT_ID_FEATURE}), and it
284
+ * is an OBSERVER: it cannot change the reply, it is called after the
285
+ * reply has been accepted, and a listener that throws is that listener's
286
+ * problem — never the caller's, whose result is already decided.
287
+ *
288
+ * Absent by default, which is what keeps this shared transport's
289
+ * behaviour identical for the Firecracker tier. The kubernetes backend
290
+ * sets it to follow the guest PROCESS behind a handle whose pod uid — its
291
+ * bind token — cannot change when the container is restarted in place.
292
+ */
293
+ readonly onGuestReply?: (reply: GuestReplyIdentity) => void;
294
+ /**
295
+ * The largest `writeFile` body this transport will accept, in raw
296
+ * bytes. Default {@link DEFAULT_MAX_WRITE_FILE_BYTES} (1 GiB).
297
+ *
298
+ * A body above one frame is written in parts (see
299
+ * {@link VsockAgentTransport.writeFile}), so nothing about the wire
300
+ * stops a caller handing over a body larger than the guest's disk or
301
+ * this process's heap. This is the bound that says no first, by a
302
+ * number the caller chose, with {@link AgentWriteFileTooLargeError}
303
+ * naming it — rather than by an out-of-memory or an ENOSPC halfway
304
+ * through a sequence of parts.
305
+ *
306
+ * Checked before the route is chosen, so it caps EVERY body — a value
307
+ * set below what one frame carries caps the small single-frame writes
308
+ * too, which is the range a host capping what a caller may push into a
309
+ * workspace would most plausibly set it to.
310
+ */
311
+ readonly maxWriteFileBytes?: number;
312
+ /**
313
+ * Raw bytes per part when a `writeFile` body is written in parts.
314
+ * Defaults to the largest part one frame can carry, and is clamped
315
+ * DOWN to that: a value above what a frame admits is not a way to
316
+ * send a bigger frame.
317
+ *
318
+ * Setting it also lowers the size at which a body is split at all,
319
+ * so a suite can exercise a multi-part write without allocating one.
320
+ * Leave it unset in production: the default is the fewest round trips
321
+ * the pre-auth ceiling allows.
322
+ */
323
+ readonly writeFilePartBytes?: number;
324
+ /**
325
+ * Say that a failed connect attempt will NOT fix itself, so the retry
326
+ * budget above is not worth spending on it.
327
+ *
328
+ * Absent by default, which keeps every dial retrying for the whole budget
329
+ * exactly as it always has — the budget exists because the common connect
330
+ * failure IS transient (an agent re-listening after a resume answers
331
+ * `ECONNREFUSED` for a moment). The exception is a failure that is about
332
+ * the CALLER's own environment rather than the guest's: the kubernetes
333
+ * backend's Service FQDN does not resolve on a host with no cluster DNS,
334
+ * and re-asking the same resolver the same question for half a minute
335
+ * only ensures the caller's own deadline expires first and reports a
336
+ * timeout in place of the diagnosis.
337
+ *
338
+ * Called with each attempt's error, before the backoff. Returning true
339
+ * ends the loop; the error thrown is the same wrapper as an exhausted
340
+ * budget, carrying that attempt's failure as its `cause`.
341
+ */
342
+ readonly permanentDialFailure?: (error: unknown) => boolean;
211
343
  }
212
344
  /**
213
345
  * The guest agent's default pre-auth frame ceiling for a routed (`tcp`)
@@ -228,6 +360,23 @@ export interface VsockTransportOptions {
228
360
  * implemented by this transport.
229
361
  */
230
362
  export declare const TCP_PREAUTH_FRAME_LIMIT_BYTES: number;
363
+ /**
364
+ * The guest agent's default ceiling on ANY frame, pre-auth or not —
365
+ * mirrors `agent.cjs`'s `MAX_FRAME_BYTES` (`NAMZU_AGENT_MAX_FRAME_BYTES`,
366
+ * default 256 MiB). It is the budget the `unix`/`vsock`/`mtls` arms are
367
+ * bounded by, since none of them runs a credential gate and so none of
368
+ * them ever pays the smaller pre-auth price.
369
+ */
370
+ export declare const GUEST_FRAME_LIMIT_BYTES: number;
371
+ /** Default {@link VsockTransportOptions.maxWriteFileBytes} — 1 GiB. */
372
+ export declare const DEFAULT_MAX_WRITE_FILE_BYTES: number;
373
+ /**
374
+ * Idle time before the kernel sends its first TCP keepalive probe on a
375
+ * `tcp`-arm connection, host side. 15 s, matching the Kubernetes backend's
376
+ * default heartbeat interval, so the two bounds do not disagree about how
377
+ * long a silent connection is allowed to look healthy.
378
+ */
379
+ export declare const TCP_KEEPALIVE_INITIAL_DELAY_MS = 15000;
231
380
  /**
232
381
  * Thrown when a `tcp`-handle request's framed envelope (op + body +
233
382
  * token) would exceed {@link TCP_PREAUTH_FRAME_LIMIT_BYTES}. Named so a
@@ -237,6 +386,96 @@ export declare const TCP_PREAUTH_FRAME_LIMIT_BYTES: number;
237
386
  export declare class AgentPreauthFrameTooLargeError extends Error {
238
387
  constructor(message: string);
239
388
  }
389
+ /**
390
+ * Thrown when a `writeFile` body exceeds
391
+ * {@link VsockTransportOptions.maxWriteFileBytes}. Distinct from
392
+ * {@link AgentPreauthFrameTooLargeError}: that one says the WIRE cannot
393
+ * carry this in one frame (and, since the part protocol, only ever fires
394
+ * when the guest cannot carry it in several either), this one says the
395
+ * HOST was configured not to send a body this large at all.
396
+ */
397
+ export declare class AgentWriteFileTooLargeError extends Error {
398
+ constructor(message: string);
399
+ }
400
+ /**
401
+ * Thrown when a read this transport cannot serve on the old whole-file op
402
+ * is asked of a guest that does not advertise
403
+ * {@link READ_FILE_STREAM_FEATURE} — a ranged `readFile`, or any
404
+ * `readFileStream`.
405
+ *
406
+ * A refusal rather than a fallback, and that is the whole point of the
407
+ * class: an agent that predates the feature IGNORES `offset`/`length` and
408
+ * answers with the WHOLE file, so silently taking the old path would hand
409
+ * a caller the entire file where it asked for a slice — a wrong answer
410
+ * dressed as a degraded one. Named so a host can tell "rebuild the guest
411
+ * image" apart from "that file is not there".
412
+ */
413
+ export declare class AgentReadFileStreamUnsupportedError extends Error {
414
+ constructor(message: string);
415
+ }
416
+ /**
417
+ * Thrown by {@link VsockAgentTransport}'s dial when it gives up without a
418
+ * socket — the retry budget spent, or the failure declared one waiting cannot
419
+ * cure. The underlying connect failure is its `cause`.
420
+ *
421
+ * A class, and not only a phrase in the message, because callers classify on
422
+ * it: "the failure came out of the dial" means NOTHING was sent to the guest,
423
+ * which is what makes a retry against a replaced pod safe (the kubernetes
424
+ * backend's `agentAddress: 'pod-ip'` re-read). A message a guest can quote
425
+ * back — a command's stderr, a path, a proxy's own error — cannot be allowed
426
+ * to claim that.
427
+ */
428
+ /**
429
+ * The two fields that turn a connection-bound terminal into a session, on
430
+ * the host's side of {@link VsockAgentTransport.openTerminal}.
431
+ *
432
+ * Both are optional and both are ignored by a guest that predates the
433
+ * session registry, which is why the caller — never this transport — is the
434
+ * one that checks the guest advertises the capability first.
435
+ */
436
+ export interface SessionTerminalOpen {
437
+ readonly sessionId?: string;
438
+ readonly persistent?: boolean;
439
+ }
440
+ /**
441
+ * One open terminal stream, plus what only a SESSION's reader needs: the
442
+ * guest's opening frame, the byte offset to come back at, and a way to stop
443
+ * reading without signalling the program.
444
+ */
445
+ export interface AgentTerminalStream {
446
+ readonly session: TerminalSession;
447
+ /** The guest's `ready` frame. Carries the session fields, when there are any. */
448
+ readonly ready: TerminalReadyEvent;
449
+ /** One past the newest retained byte this stream has delivered, if any. */
450
+ nextOffset(): number | undefined;
451
+ /**
452
+ * End this attachment locally. Nothing is signalled in the guest: the
453
+ * program goes on running and its output goes on filling the retained
454
+ * log, which is the entire difference between this and `kill`.
455
+ */
456
+ detach(): void;
457
+ }
458
+ /**
459
+ * Thrown — as the rejection of a session terminal's `exited` — when the
460
+ * attachment ended and the program did not.
461
+ *
462
+ * `exited` may not RESOLVE here: a resolved `exited` says the program is
463
+ * over, and reporting `exitCode: -1` for a shell that is still running in
464
+ * the pod is exactly the confusion this whole feature exists to remove. The
465
+ * offset is carried because it is what the next attach resumes from.
466
+ */
467
+ export declare class AgentSessionDetachedError extends Error {
468
+ readonly nextOffset: number | undefined;
469
+ readonly name = "AgentSessionDetachedError";
470
+ constructor(nextOffset: number | undefined, message: string, options?: {
471
+ cause?: unknown;
472
+ });
473
+ }
474
+ export declare class AgentDialFailedError extends Error {
475
+ constructor(message: string, options?: {
476
+ cause?: unknown;
477
+ });
478
+ }
240
479
  /** Exact guest wire version accepted by this Firecracker transport. */
241
480
  export declare const FIRECRACKER_AGENT_PROTOCOL_VERSION: 2;
242
481
  declare function frame(payload: string): Buffer;
@@ -244,11 +483,28 @@ declare function frame(payload: string): Buffer;
244
483
  * Incremental frame reader. Feed it socket chunks; it yields complete
245
484
  * payloads. A zero-length frame is the exec stream terminator and is
246
485
  * surfaced as an empty string so the caller can stop.
486
+ *
487
+ * It accumulates into ONE buffer it grows geometrically, with a read
488
+ * cursor, rather than re-`concat`ing every arriving chunk onto a fresh
489
+ * allocation. The distinction only matters for a large frame, where it is
490
+ * the difference between linear and quadratic: a `read-file` reply for a
491
+ * 64 MiB file arrives as ~1400 socket chunks, and copying everything
492
+ * received so far onto each one of them spent half a minute of memcpy on
493
+ * a reply the socket delivered in under a second. Identical framing,
494
+ * identical errors, identical `bufferedBytes` — only the copying changes.
247
495
  */
248
496
  declare class FrameReader {
249
497
  private buf;
498
+ /** First byte not yet handed out as part of a frame. */
499
+ private start;
500
+ /** One past the last byte received. */
501
+ private end;
250
502
  push(chunk: Buffer): string[];
251
503
  get bufferedBytes(): number;
504
+ /** Copy `chunk` in, growing (and first compacting) only when needed. */
505
+ private append;
506
+ /** Mark `bytes` from the read cursor as consumed. */
507
+ private consume;
252
508
  }
253
509
  /**
254
510
  * The transport. One instance per sandbox handle; every request opens
@@ -262,14 +518,31 @@ export declare class VsockAgentTransport {
262
518
  private readonly connectRetryBudgetMs;
263
519
  private readonly connectRetryIntervalMs;
264
520
  private readonly readIdleTimeoutMs;
521
+ /** Undefined → this transport negotiates no heartbeat at all. */
522
+ private readonly heartbeatMs?;
265
523
  private readonly onDial?;
524
+ private readonly onDialAttempt?;
525
+ private readonly onGuestReply?;
526
+ private readonly maxWriteFileBytes;
527
+ private readonly writeFilePartBytes?;
528
+ /**
529
+ * What the guest advertised in `healthz`, cached for this handle's
530
+ * lifetime. A pod does not swap its agent binary while it is running,
531
+ * so the probe is asked once per transport and only when something
532
+ * actually depends on a capability — an ordinary write, exec, read or
533
+ * terminal pays nothing for it.
534
+ */
535
+ private guestFeatureList?;
536
+ private readonly permanentDialFailure?;
266
537
  private readonly executionController;
267
538
  constructor(handle: SandboxAgentHandle, options?: VsockTransportOptions);
268
539
  /**
269
540
  * Dial the agent with the resume-survival retry budget. Resolves a
270
541
  * connected, post-handshake socket. Retries connect/handshake
271
542
  * failures (ECONNREFUSED while the agent re-listens after a resume,
272
- * a dropped CONNECT ack) until the budget is exhausted.
543
+ * a dropped CONNECT ack) until the budget is exhausted — or until
544
+ * {@link VsockTransportOptions.permanentDialFailure} says this particular
545
+ * failure is not one waiting will cure.
273
546
  */
274
547
  private dial;
275
548
  private connectOnce;
@@ -317,6 +590,16 @@ export declare class VsockAgentTransport {
317
590
  * handle kind, which the guest never gates on frame size pre-auth.
318
591
  */
319
592
  private assertPreauthBudget;
593
+ /**
594
+ * Hand one accepted reply to {@link VsockTransportOptions.onGuestReply},
595
+ * and never let the listener's failure reach the caller.
596
+ *
597
+ * The caller's result is already decided by the time this runs — the
598
+ * reply parsed, the frame accounted for — so a hook that throws must not
599
+ * turn a successful read into a failed one. Swallowing is the only
600
+ * behaviour that keeps an optional observer optional.
601
+ */
602
+ private observeGuestReply;
320
603
  /**
321
604
  * Send one framed request and read one framed JSON reply (file-IO +
322
605
  * healthz). Applies the read-idle timeout so a post-resume hung read
@@ -362,8 +645,210 @@ export declare class VsockAgentTransport {
362
645
  * post-create readiness fence.
363
646
  */
364
647
  waitForReady(timeoutMs: number, pollIntervalMs: number, signal?: AbortSignal): Promise<void>;
365
- writeFile(path: string, content: Buffer): Promise<void>;
366
- readFile(path: string): Promise<Buffer>;
648
+ /**
649
+ * Write a whole file into the guest workspace.
650
+ *
651
+ * A body that fits one frame goes as it always has: a single
652
+ * `write-file` envelope carrying the base64 content, one round trip,
653
+ * byte-for-byte the request this transport has always sent.
654
+ *
655
+ * A body that does NOT fit is the case this exists for. On the `tcp`
656
+ * arm every request dials a fresh connection, so every request is that
657
+ * connection's first, not-yet-authenticated frame (the credential
658
+ * rides in the envelope) and is bounded by
659
+ * {@link TCP_PREAUTH_FRAME_LIMIT_BYTES} — about 5.9 MiB of file
660
+ * content — on EVERY call, not once. Raising the guest's
661
+ * `NAMZU_AGENT_MAX_PREAUTH_FRAME_BYTES` trades away the pre-auth
662
+ * budget that ceiling exists to bound, so the fix is on this side:
663
+ * split the body into parts that each fit, append them to a temporary
664
+ * SIBLING of the target inside the same workspace jail, and finish
665
+ * with an atomic `rename` onto the target.
666
+ *
667
+ * What that buys, and why the shape is what it is:
668
+ *
669
+ * - **A reader never sees a half-written file.** The target changes
670
+ * exactly once, in the final part's `rename`. A sequence that dies
671
+ * at part 3 of 9 leaves the target exactly as it was — including
672
+ * not existing.
673
+ * - **A lost or duplicated part is detected, not written.** Each part
674
+ * names the offset it starts at and the guest refuses it unless
675
+ * that equals the temp file's current size.
676
+ * - **An abandoned sequence cleans up after itself.** An abort or a
677
+ * transport failure removes the temp file (best effort — a peer
678
+ * that has gone away cannot be asked to) and rejects.
679
+ * - **Parts go out sequentially on fresh connections**, which is what
680
+ * the offset check assumes and what keeps the guest's pre-auth
681
+ * connection pool holding one of this caller's sockets at a time.
682
+ *
683
+ * The guest must ADVERTISE the capability (`${WRITE_FILE_PARTS_FEATURE}`
684
+ * in its `healthz` reply) before a single part is sent. An agent that
685
+ * predates the part protocol would read a part's `content` as a whole
686
+ * file; it never receives one, and an oversized body against such a
687
+ * guest still fails with the named
688
+ * {@link AgentPreauthFrameTooLargeError} it always did.
689
+ *
690
+ * {@link VsockTransportOptions.maxWriteFileBytes} is checked FIRST, on
691
+ * every body and before anything about the wire is considered: it is a
692
+ * bound on what a caller may push into a workspace, not a bound on
693
+ * multi-part writes, so a host that lowers it below the frame budget
694
+ * gets the cap it asked for rather than none.
695
+ */
696
+ writeFile(path: string, content: Buffer, signal?: AbortSignal): Promise<void>;
697
+ /** Today's single-frame write, unchanged — see {@link writeFile}. */
698
+ private writeFileWhole;
699
+ /**
700
+ * The frame budget one request has on this handle: the pre-auth
701
+ * ceiling on the credentialed `tcp` arm, the guest's global frame
702
+ * ceiling on the host-local arms, which authenticate nothing and so
703
+ * never pay the smaller price.
704
+ */
705
+ private singleFrameBudgetBytes;
706
+ /**
707
+ * Exact framed size of a `write-file` envelope whose `content` field
708
+ * holds `contentBytes` raw bytes base64-encoded — WITHOUT encoding
709
+ * them, so sizing a 1 GiB body costs nothing and never builds a string
710
+ * longer than V8 permits.
711
+ *
712
+ * Exact rather than approximate because the base64 alphabet contains
713
+ * no character `JSON.stringify` escapes, so the encoded content
714
+ * contributes precisely its own length to the envelope and the rest of
715
+ * the envelope (paths, the token, the part fields) is measured as it
716
+ * will actually be serialized.
717
+ */
718
+ private writeFileEnvelopeBytes;
719
+ /**
720
+ * Ask the guest whether it implements the part protocol. Cached for
721
+ * the transport's lifetime; a transport failure propagates rather than
722
+ * reading as "not supported", because answering a broken connection
723
+ * with a too-large error would name the wrong cause.
724
+ */
725
+ private guestSupportsWriteFileParts;
726
+ /**
727
+ * The capability strings this guest advertises in `healthz`, cached for
728
+ * this transport's lifetime.
729
+ *
730
+ * One list, asked once, for every optional op: the write-file part
731
+ * protocol and the execution-attach ops both read it, and a second
732
+ * cache would mean a second probe against a guest that answers both
733
+ * questions in one reply. A transport FAILURE propagates rather than
734
+ * reading as "not supported", because answering a broken connection
735
+ * with a capability refusal would name the wrong cause.
736
+ */
737
+ guestFeatures(signal?: AbortSignal): Promise<readonly string[]>;
738
+ /**
739
+ * Send one framed request and read a STREAM of framed JSON events until
740
+ * the agent's zero-length terminator, handing each event to `onEvent`.
741
+ *
742
+ * The generic half of what {@link executeRaw} does, without any of its
743
+ * opinions about what the events mean: `executeRaw` owns the exec
744
+ * NDJSON union and the {@link ExecResultAccumulator}, and this owns
745
+ * dial, framing, the terminator, the observation bound and the
746
+ * post-terminator close. Additive — nothing already shipped calls it —
747
+ * so the Firecracker tier's behaviour is untouched and a later streamed
748
+ * op reuses it rather than writing a fourth copy of this loop.
749
+ *
750
+ * There is deliberately NO read-idle timeout. A stream that exists to
751
+ * follow a long, quiet command must not be torn down for being quiet;
752
+ * the whole observation is bounded by `observationTimeoutMs` instead,
753
+ * exactly as an `execute` stream is.
754
+ */
755
+ streamFramedRequest(req: AgentRequest, onEvent: (event: Record<string, unknown>) => void, options: {
756
+ readonly observationTimeoutMs: number;
757
+ }, signal?: AbortSignal): Promise<void>;
758
+ /** The refusal for a body no frame can carry and no guest can take in parts. */
759
+ private oversizedWriteFileError;
760
+ /**
761
+ * Write `content` to `target` as a sequence of parts. See
762
+ * {@link writeFile} for why this shape.
763
+ */
764
+ private writeFileInParts;
765
+ /**
766
+ * Remove an abandoned part file. Best effort BY CONTRACT: the reason
767
+ * the sequence failed is frequently that the guest is unreachable, and
768
+ * a cleanup that threw would replace the caller's real error — the one
769
+ * that says why the write failed — with a second one about tidying up.
770
+ */
771
+ private discardWriteFileTemp;
772
+ /**
773
+ * Read a file out of the guest.
774
+ *
775
+ * Three shapes, decided by what the guest advertises and what the
776
+ * caller asked for:
777
+ *
778
+ * - **No options, guest advertises {@link READ_FILE_STREAM_FEATURE}** —
779
+ * served by {@link readFileStream} and concatenated here. Neither
780
+ * side ever holds the base64 form or the JSON envelope whole, so the
781
+ * ~384 MiB ceiling (V8 refuses a string longer than `0x1fffffe8`
782
+ * characters, which is what a base64-encoded file of that size
783
+ * needs) is gone and the guest's peak stops tracking the file's
784
+ * size. The result is still one `Buffer`, because that is what this
785
+ * method returns; a caller that must not hold even that iterates
786
+ * {@link readFileStream} directly.
787
+ * - **No options, guest does not advertise it** — today's single
788
+ * whole-file reply, byte for byte, with today's ceiling.
789
+ * - **`offset` and `length`** — one ranged `read-file`: a single round
790
+ * trip for a single slice, which is the point of asking for one.
791
+ * `offset` WITHOUT `length` is an unbounded tail, so it goes through
792
+ * the stream instead; the guest refuses an uncapped range on
793
+ * `read-file` for exactly that reason.
794
+ *
795
+ * A ranged read against a guest that does not advertise the feature is
796
+ * REFUSED with {@link AgentReadFileStreamUnsupportedError} rather than
797
+ * downgraded: such an agent ignores `offset`/`length` and answers with
798
+ * the whole file, which the caller would read as its slice.
799
+ */
800
+ readFile(path: string, options?: SandboxReadFileOptions): Promise<Buffer>;
801
+ /** Today's single whole-file reply, unchanged — see {@link readFile}. */
802
+ private readFileWhole;
803
+ /**
804
+ * One bounded slice, in one round trip.
805
+ *
806
+ * The guest's own ceiling on a range (`NAMZU_AGENT_READ_FILE_RANGE_BYTES`,
807
+ * 1 MiB by default) is not mirrored here and deliberately so: it is the
808
+ * DEPLOYMENT's number, a host that guessed it would refuse ranges the
809
+ * guest would have served, and the guest's refusal already names the
810
+ * variable that raises it.
811
+ */
812
+ private readFileRange;
813
+ /**
814
+ * Read a file as an ordered sequence of chunks, so neither side holds
815
+ * the whole of it.
816
+ *
817
+ * The guest sends `meta`, then `data` frames, then `end`, then the
818
+ * zero-length terminator — the same terminated-stream shape `execute`
819
+ * uses. Two bounds keep this side's heap flat while the guest's stays
820
+ * flat on its own: the socket is PAUSED once
821
+ * {@link READ_FILE_STREAM_HIGH_WATER_BYTES} of decoded chunks are
822
+ * waiting for a slow consumer, and the guest itself waits for each
823
+ * `data` frame to drain before it reads the next one.
824
+ *
825
+ * Leaving the loop early — `break`, an exception, an aborted
826
+ * `options.signal` — destroys the socket in the generator's `finally`,
827
+ * which is what makes the guest close its fd: it sees the connection go
828
+ * and releases the descriptor rather than leaking one per abandoned
829
+ * read.
830
+ *
831
+ * Refuses a guest that does not advertise
832
+ * {@link READ_FILE_STREAM_FEATURE} before dialing, with
833
+ * {@link AgentReadFileStreamUnsupportedError}.
834
+ */
835
+ readFileStream(path: string, options?: SandboxReadFileOptions): AsyncGenerator<Buffer, void, undefined>;
836
+ /**
837
+ * The stream itself. Private, and one argument wider than
838
+ * {@link readFileStream}: `onMeta` fires once, with the guest's `meta`
839
+ * frame, before the first chunk is yielded, which is how
840
+ * {@link readFile} sizes its destination buffer without a second round
841
+ * trip and without a public parameter nobody outside this class should
842
+ * pass.
843
+ */
844
+ private readFileFrames;
845
+ /**
846
+ * Ask the guest whether it implements ranged and streamed reads.
847
+ * Cached for the transport's lifetime, exactly as
848
+ * {@link guestSupportsWriteFileParts} is, and for the same reason: a
849
+ * pod does not swap its agent binary while it is running.
850
+ */
851
+ private guestSupportsReadFileStream;
367
852
  /**
368
853
  * Open a real PTY owned by the in-VM agent.
369
854
  *
@@ -373,7 +858,55 @@ export declare class VsockAgentTransport {
373
858
  * browser never reaches this transport directly; the runtime gateway owns
374
859
  * the session and its authenticated WebSocket attachment.
375
860
  */
376
- openTerminal(options: OpenTerminalOptions): Promise<TerminalSession>;
861
+ openTerminal(options: OpenTerminalOptions & SessionTerminalOpen): Promise<TerminalSession>;
862
+ /**
863
+ * The same open, handing back the session's own handles as well as the
864
+ * `TerminalSession` — the offset to come back at, and the detach that
865
+ * ends the attachment without signalling the program.
866
+ *
867
+ * Exactly one code path serves both: a persistent terminal is not a
868
+ * second kind of terminal, it is the same stream with a different answer
869
+ * to "what does a closed connection mean".
870
+ */
871
+ openSessionTerminal(options: OpenTerminalOptions & SessionTerminalOpen): Promise<AgentTerminalStream>;
872
+ /**
873
+ * The framed, bidirectional stream behind every terminal this transport
874
+ * opens — the one the `terminal` op starts, and the one `attach-session`
875
+ * joins to a terminal that is already running.
876
+ *
877
+ * Parameterised rather than copied, because the two differ in exactly two
878
+ * places and everything else — the dial, the framing, the ready
879
+ * handshake, the read-idle timer that is cleared once a shell may
880
+ * legitimately go quiet, the heartbeat, the output buffering before the
881
+ * first listener, the kill grace — has to behave identically or a
882
+ * reattached terminal is a second terminal implementation with its own
883
+ * bugs. The two differences:
884
+ *
885
+ * - **`detachable`.** For a connection-bound terminal a lost stream IS
886
+ * the end of the program, and `exited` resolves with `exitCode: -1`
887
+ * exactly as it always has. For a session attachment it is not: the
888
+ * program is still running in the pod, so `exited` REJECTS with
889
+ * {@link AgentSessionDetachedError} rather than reporting an exit that
890
+ * did not happen. The rejection is pre-handled here so a caller that
891
+ * only reads output cannot take the host process down with an
892
+ * unhandled rejection.
893
+ * - **the offsets.** A session stream's frames carry their place in the
894
+ * guest's retained log, and {@link AgentTerminalStream.nextOffset} is
895
+ * what a reattach resumes from. It is never computed from the decoded
896
+ * text: a chunk that ends mid-character decodes wider than the bytes
897
+ * it replaced.
898
+ */
899
+ private openTerminalStream;
900
+ /**
901
+ * Join a terminal session that is already running in the guest, replaying
902
+ * what it printed from `fromOffset` before following it live.
903
+ *
904
+ * The guest allows ONE attachment per session and ends the previous one
905
+ * by name, so two host processes cannot interleave keystrokes into one
906
+ * shell. Nothing here signals the program: releasing this stream is a
907
+ * detach, and ending the session is `kill-session`.
908
+ */
909
+ attachSessionTerminal(request: AttachSessionRequest): Promise<AgentTerminalStream>;
377
910
  /** Open one TCP stream to a service listening on guest loopback. */
378
911
  openTcpConnection(options: SandboxTcpConnectOptions): Promise<SandboxTcpConnection>;
379
912
  }