@namzu/sandbox 14.0.0 → 16.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. package/CHANGELOG.md +924 -0
  2. package/README.md +369 -14
  3. package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
  4. package/dist/backends/aci-standby-pool/index.js +13 -1
  5. package/dist/backends/aci-standby-pool/index.js.map +1 -1
  6. package/dist/backends/docker/index.d.ts +169 -6
  7. package/dist/backends/docker/index.d.ts.map +1 -1
  8. package/dist/backends/docker/index.js +499 -85
  9. package/dist/backends/docker/index.js.map +1 -1
  10. package/dist/backends/firecracker/index.d.ts.map +1 -1
  11. package/dist/backends/firecracker/index.js +12 -2
  12. package/dist/backends/firecracker/index.js.map +1 -1
  13. package/dist/backends/firecracker/protocol.d.ts +459 -8
  14. package/dist/backends/firecracker/protocol.d.ts.map +1 -1
  15. package/dist/backends/firecracker/protocol.js +136 -0
  16. package/dist/backends/firecracker/protocol.js.map +1 -1
  17. package/dist/backends/firecracker/transport.d.ts +539 -6
  18. package/dist/backends/firecracker/transport.d.ts.map +1 -1
  19. package/dist/backends/firecracker/transport.js +1171 -24
  20. package/dist/backends/firecracker/transport.js.map +1 -1
  21. package/dist/backends/kubernetes/egress-policy.d.ts +1181 -13
  22. package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -1
  23. package/dist/backends/kubernetes/egress-policy.js +2350 -31
  24. package/dist/backends/kubernetes/egress-policy.js.map +1 -1
  25. package/dist/backends/kubernetes/identity.d.ts +193 -0
  26. package/dist/backends/kubernetes/identity.d.ts.map +1 -0
  27. package/dist/backends/kubernetes/identity.js +147 -0
  28. package/dist/backends/kubernetes/identity.js.map +1 -0
  29. package/dist/backends/kubernetes/index.d.ts +678 -33
  30. package/dist/backends/kubernetes/index.d.ts.map +1 -1
  31. package/dist/backends/kubernetes/index.js +1180 -95
  32. package/dist/backends/kubernetes/index.js.map +1 -1
  33. package/dist/backends/kubernetes/ingress-policy.d.ts +375 -0
  34. package/dist/backends/kubernetes/ingress-policy.d.ts.map +1 -0
  35. package/dist/backends/kubernetes/ingress-policy.js +1050 -0
  36. package/dist/backends/kubernetes/ingress-policy.js.map +1 -0
  37. package/dist/backends/kubernetes/k8s-client.d.ts +213 -4
  38. package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -1
  39. package/dist/backends/kubernetes/k8s-client.js +359 -52
  40. package/dist/backends/kubernetes/k8s-client.js.map +1 -1
  41. package/dist/backends/kubernetes/lease.d.ts +40 -14
  42. package/dist/backends/kubernetes/lease.d.ts.map +1 -1
  43. package/dist/backends/kubernetes/lease.js +68 -18
  44. package/dist/backends/kubernetes/lease.js.map +1 -1
  45. package/dist/backends/kubernetes/objects.d.ts +423 -3
  46. package/dist/backends/kubernetes/objects.d.ts.map +1 -1
  47. package/dist/backends/kubernetes/objects.js +364 -2
  48. package/dist/backends/kubernetes/objects.js.map +1 -1
  49. package/dist/backends/kubernetes/per-sandbox-policy.d.ts +219 -0
  50. package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -0
  51. package/dist/backends/kubernetes/per-sandbox-policy.js +375 -0
  52. package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -0
  53. package/dist/backends/kubernetes/rbac.d.ts +153 -0
  54. package/dist/backends/kubernetes/rbac.d.ts.map +1 -0
  55. package/dist/backends/kubernetes/rbac.js +177 -0
  56. package/dist/backends/kubernetes/rbac.js.map +1 -0
  57. package/dist/backends/kubernetes/sandbox.d.ts +81 -14
  58. package/dist/backends/kubernetes/sandbox.d.ts.map +1 -1
  59. package/dist/backends/kubernetes/sandbox.js +149 -15
  60. package/dist/backends/kubernetes/sandbox.js.map +1 -1
  61. package/dist/backends/kubernetes/transport.d.ts +935 -9
  62. package/dist/backends/kubernetes/transport.d.ts.map +1 -1
  63. package/dist/backends/kubernetes/transport.js +1958 -62
  64. package/dist/backends/kubernetes/transport.js.map +1 -1
  65. package/dist/backends/kubernetes/workspace.d.ts +1149 -18
  66. package/dist/backends/kubernetes/workspace.d.ts.map +1 -1
  67. package/dist/backends/kubernetes/workspace.js +2825 -186
  68. package/dist/backends/kubernetes/workspace.js.map +1 -1
  69. package/dist/backends/remote-execution-controller.d.ts +14 -0
  70. package/dist/backends/remote-execution-controller.d.ts.map +1 -1
  71. package/dist/backends/remote-execution-controller.js.map +1 -1
  72. package/dist/index.d.ts +294 -18
  73. package/dist/index.d.ts.map +1 -1
  74. package/dist/index.js +280 -10
  75. package/dist/index.js.map +1 -1
  76. package/dist/testing/sandbox-conformance.d.ts +39 -5
  77. package/dist/testing/sandbox-conformance.d.ts.map +1 -1
  78. package/dist/testing/sandbox-conformance.js +436 -5
  79. package/dist/testing/sandbox-conformance.js.map +1 -1
  80. package/package.json +3 -3
  81. package/src/backends/aci-standby-pool/index.ts +16 -1
  82. package/src/backends/docker/index.ts +617 -100
  83. package/src/backends/firecracker/index.ts +14 -2
  84. package/src/backends/firecracker/protocol.ts +514 -6
  85. package/src/backends/firecracker/transport.ts +1492 -40
  86. package/src/backends/kubernetes/egress-policy.ts +3334 -55
  87. package/src/backends/kubernetes/identity.ts +261 -0
  88. package/src/backends/kubernetes/index.ts +1785 -127
  89. package/src/backends/kubernetes/ingress-policy.ts +1344 -0
  90. package/src/backends/kubernetes/k8s-client.ts +444 -54
  91. package/src/backends/kubernetes/lease.ts +75 -19
  92. package/src/backends/kubernetes/objects.ts +626 -6
  93. package/src/backends/kubernetes/per-sandbox-policy.ts +497 -0
  94. package/src/backends/kubernetes/rbac.ts +192 -0
  95. package/src/backends/kubernetes/sandbox.ts +218 -20
  96. package/src/backends/kubernetes/transport.ts +2733 -124
  97. package/src/backends/kubernetes/workspace.ts +4476 -222
  98. package/src/backends/remote-execution-controller.ts +14 -0
  99. package/src/index.ts +668 -19
  100. package/src/testing/sandbox-conformance.ts +540 -5
@@ -23,7 +23,8 @@
23
23
  * sub-second warm-acquire target this backend is judged against must be
24
24
  * measured, not guessed at.
25
25
  */
26
- import type { OpenTerminalOptions, SandboxExecOptions, SandboxExecResult, SandboxTcpConnectOptions, SandboxTcpConnection, TerminalSession } from '@namzu/sdk';
26
+ import type { BackgroundJobStatus, OpenTerminalOptions, SandboxExecOptions, SandboxExecResult, SandboxReadFileOptions, SandboxTcpConnectOptions, SandboxTcpConnection, TerminalSession } from '@namzu/sdk';
27
+ import { type GuestReplyIdentity, type QuiesceScope, type QuiescedProcess, type SessionKind, type SessionState } from '../firecracker/protocol.js';
27
28
  import { type SandboxAgentHandle, type VsockTransportOptions } from '../firecracker/transport.js';
28
29
  /** The one {@link SandboxAgentHandle} arm this backend ever constructs. */
29
30
  export type KubernetesAgentHandle = Extract<SandboxAgentHandle, {
@@ -37,8 +38,73 @@ export type KubernetesAgentHandle = Extract<SandboxAgentHandle, {
37
38
  * unexpected".
38
39
  */
39
40
  export declare class KubernetesAgentUnauthorizedError extends Error {
41
+ constructor(message?: string, options?: ErrorOptions);
42
+ }
43
+ /**
44
+ * Thrown when the guest agent refuses a request because it has FENCED
45
+ * ITSELF: an earlier process group's shutdown could not be confirmed, so it
46
+ * answers every op but `healthz` and `cancel-execution` with `agent_retiring`
47
+ * and will go on doing so until the pod is replaced.
48
+ *
49
+ * Its own class, and deliberately NOT the Firecracker tier's mapping of the
50
+ * same refusal. There a fenced agent becomes {@link
51
+ * RemoteCancellationUnknownError}, which is correct for a disposable microVM:
52
+ * the shared controller's rule is that the sandbox stops being reusable, and
53
+ * on that tier retiring one means deleting scratch. On a workspace the same
54
+ * error would retire the handle and take the pod — and with it every other
55
+ * holder's terminals, dev servers and running commands — away from callers
56
+ * who did nothing but share a workspace with the command that wedged.
57
+ *
58
+ * So this error refuses the ONE call rather than the workspace: the handle is
59
+ * not retired, nothing is patched, and every other holder's pod stays where it
60
+ * was. What it does NOT claim is that the next call will work. The fence is
61
+ * the GUEST's, and `dispatch` gates it ahead of every data-plane branch, so
62
+ * `readFile`, `writeFile`, `openTerminal` and `openTcpConnection` meet the
63
+ * same refusal on the wire — under their own paths' error shapes, since only
64
+ * the two control ops come through `requestChecked`. Only a new pod clears
65
+ * it, which is why the message names the verbs that REPLACE the pod, on both
66
+ * tiers that use this transport, and leaves the timing to the host: on a
67
+ * workspace those are `suspend()` then `resume()`, and they take the live
68
+ * sessions in that pod down with them.
69
+ */
70
+ export declare class KubernetesAgentRetiringError extends Error {
71
+ readonly name = "KubernetesAgentRetiringError";
40
72
  constructor(message?: string);
41
73
  }
74
+ /**
75
+ * Thrown when the agent's address could not be RESOLVED — the dial never
76
+ * reached a socket because the name has no answer here.
77
+ *
78
+ * Its own class, and its own message, because this is the one failure whose
79
+ * cause is the deployment's shape rather than anything the cluster did: a
80
+ * Service FQDN resolves through cluster DNS and nowhere else, so a host
81
+ * outside the cluster fails every call at name resolution and reads the
82
+ * result as a sandbox that never came up. The fix is a configuration field,
83
+ * so the error names it.
84
+ */
85
+ export declare class KubernetesAgentAddressUnresolvableError extends Error {
86
+ /** The host that did not resolve — normally a `*.svc.cluster.local`. */
87
+ readonly host: string;
88
+ readonly name = "KubernetesAgentAddressUnresolvableError";
89
+ constructor(
90
+ /** The host that did not resolve — normally a `*.svc.cluster.local`. */
91
+ host: string, message: string, options?: {
92
+ cause?: unknown;
93
+ });
94
+ }
95
+ /** The pod and process one execution reserved on — see {@link executionGuest}. */
96
+ export interface KubernetesReservedGuest {
97
+ /** The bind token the attempt presented, which is the pod's uid. */
98
+ readonly podUid?: string;
99
+ /** The boot id the `reserve-execution` reply carried, when it carried one. */
100
+ readonly guestBootId?: string;
101
+ }
102
+ /**
103
+ * The guest that reserved the execution this error came from — see
104
+ * {@link executionGuest}. `undefined` when the error is not an execution
105
+ * failure this transport produced.
106
+ */
107
+ export declare function guestWhenReserved(error: unknown): KubernetesReservedGuest | undefined;
42
108
  /**
43
109
  * One `exec()` call's wall-time breakdown. Durations are NOT a partition
44
110
  * of a single total — `reserveMs` and `executeMs` each include their OWN
@@ -62,13 +128,339 @@ export interface KubernetesTransportTiming {
62
128
  */
63
129
  readonly drainMs: number;
64
130
  }
65
- export interface KubernetesTransportOptions extends VsockTransportOptions {
131
+ /**
132
+ * What one `healthz` reply says about the agent, with the fence kept rather
133
+ * than collapsed into a boolean.
134
+ *
135
+ * {@link VsockAgentTransport.healthz} answers `false` both for an agent that
136
+ * did not reply and for one that replied "I have fenced myself", and those
137
+ * are opposite facts for a host deciding what to do next: the first is a pod
138
+ * that may be perfectly fine a second from now, the second is a pod that will
139
+ * refuse every call until it is replaced.
140
+ */
141
+ export interface KubernetesAgentHealth {
142
+ /** The reply's own `ok` — `true` only for an agent serving normally. */
143
+ readonly ok: boolean;
144
+ /**
145
+ * The agent has fenced itself and only a new pod clears it.
146
+ *
147
+ * Read from the reply's own `retiring` flag and from nothing else. A
148
+ * not-`ok` reply without it is NOT inferred to be a fence: the connection
149
+ * gate answers a `healthz` that arrived over one unauthenticated
150
+ * connection too many, or behind an exhausted pre-auth buffer, with a
151
+ * named `{ ok: false, error }` and no flag — and a caller told `retiring`
152
+ * there would suspend and resume a perfectly healthy pod. So `ok: false`
153
+ * with `retiring: false` is its own answer: this reply says nothing about
154
+ * whether the agent is serving.
155
+ */
156
+ readonly retiring: boolean;
157
+ }
158
+ /**
159
+ * `permanentDialFailure` is deliberately NOT inherited: this transport sets
160
+ * its own (see the constructor), so advertising the field would be offering a
161
+ * caller a predicate that is silently overwritten. `onGuestReply` is
162
+ * re-declared rather than inherited, with the pod the reply came FROM — see
163
+ * below.
164
+ */
165
+ export interface KubernetesTransportOptions extends Omit<VsockTransportOptions, 'permanentDialFailure' | 'onGuestReply'> {
66
166
  /**
67
167
  * Fires once per completed `exec()` call (success or failure) with
68
168
  * the four phase durations above. The payload is exactly those four
69
169
  * numbers — never the token, never a command, argv, or output.
70
170
  */
71
171
  readonly onTiming?: (timing: KubernetesTransportTiming) => void;
172
+ /**
173
+ * Re-read the live pod behind this sandbox and hand back the address
174
+ * and token it answers on NOW.
175
+ *
176
+ * Set only by the `pod-ip` address mode, where the handle carries a
177
+ * literal IP that dies with its pod; a Service FQDN needs none of this
178
+ * because the name outlives the pod and the dial re-resolves it every
179
+ * call. Consulted at most ONCE per call, and only after a dial that
180
+ * failed at connect — see {@link KubernetesAgentTransport}.
181
+ */
182
+ readonly refreshHandle?: (signal?: AbortSignal) => Promise<KubernetesAgentHandle>;
183
+ /**
184
+ * {@link VsockTransportOptions.onGuestReply}, plus the one thing the
185
+ * shared transport cannot say and this one always can: the bind token of
186
+ * the WIRE the reply arrived on, which is the uid of the pod that
187
+ * answered.
188
+ *
189
+ * A rebind builds a replacement wire and does not close the wire it
190
+ * leaves, so a call dispatched to the outgoing pod can still be answered
191
+ * BY it — a long `write-file`, a `read-file`, a pod inside its
192
+ * termination grace period — minutes after this transport has followed
193
+ * the replacement. That reply is a true statement about a pod this
194
+ * transport is no longer bound to, and a listener that could not tell the
195
+ * two apart would pair the departed pod's agent process with the
196
+ * replacement's uid: an identity naming a process that never ran there.
197
+ *
198
+ * Bound per wire, so the token is the one the reply's own connection
199
+ * presented and never the one the transport happens to hold now —
200
+ * `undefined` on a handle carrying no token at all, which is a wire that
201
+ * cannot name the pod that answered and so proves nothing about it.
202
+ */
203
+ readonly onGuestReply?: (reply: GuestReplyIdentity, podUid: string | undefined) => void;
204
+ }
205
+ /**
206
+ * Thrown before a command is admitted, when the caller asked for a
207
+ * detachable execution and the guest does not advertise
208
+ * {@link EXECUTION_ATTACH_FEATURE}.
209
+ *
210
+ * Refused rather than downgraded: a caller that asked for detach is about
211
+ * to rely on being able to come back for the output, and running the
212
+ * command anyway would keep nothing and tell nobody.
213
+ */
214
+ export declare class KubernetesExecutionAttachUnsupportedError extends Error {
215
+ readonly feature: string;
216
+ readonly name = "KubernetesExecutionAttachUnsupportedError";
217
+ constructor(feature: string, message: string);
218
+ }
219
+ /** Why an `attach-execution` could not be served. */
220
+ export type KubernetesAttachRefusal = 'unknown_execution' | 'output_not_retained' | 'invalid_offset' | 'invalid_execution_id' | 'agent_retiring' | 'unknown';
221
+ /**
222
+ * Thrown when the guest ANSWERED an attach and refused it — the execution
223
+ * is past its retention, ran in a pod that has since been replaced, never
224
+ * asked for its output to be kept, or the offset names bytes it does not
225
+ * have.
226
+ *
227
+ * Distinct from a transport failure on purpose: a refusal will not become a
228
+ * success by being retried, so the reattach loop stops on it instead of
229
+ * spending its whole window re-asking a question already answered.
230
+ */
231
+ export declare class KubernetesExecutionNotAttachableError extends Error {
232
+ readonly executionId: string;
233
+ readonly reason: KubernetesAttachRefusal;
234
+ readonly name = "KubernetesExecutionNotAttachableError";
235
+ /**
236
+ * The state the guest reported for this execution, when it reported
237
+ * one. `'reserved'` is the one that changes what a caller should do:
238
+ * the command was never started, so nothing is running.
239
+ */
240
+ readonly executionState: string | undefined;
241
+ constructor(executionId: string, reason: KubernetesAttachRefusal, message: string, options?: {
242
+ cause?: unknown;
243
+ state?: string;
244
+ });
245
+ }
246
+ /**
247
+ * Thrown when a detached `exec()` gave up OBSERVING a command that is, as
248
+ * far as this host knows, still the guest's to run.
249
+ *
250
+ * The two fields are what makes it recoverable rather than merely a
251
+ * failure: `executionId` names the command to a second host process, and
252
+ * `outputOffset` is the byte the next `attachExecution` should resume from
253
+ * so nothing is read twice and no gap is invented.
254
+ *
255
+ * Nothing on this path cancels to reconcile. That is the whole point of
256
+ * the feature: a reset connection used to cost the workspace its pod, and
257
+ * a command the host has stopped watching is not a command that has to
258
+ * die.
259
+ */
260
+ export declare class KubernetesExecutionDetachedError extends Error {
261
+ readonly executionId: string;
262
+ readonly outputOffset: number;
263
+ readonly name = "KubernetesExecutionDetachedError";
264
+ constructor(executionId: string, outputOffset: number, message: string, options?: {
265
+ cause?: unknown;
266
+ });
267
+ }
268
+ /**
269
+ * What a caller passes to read an execution it did not necessarily start.
270
+ *
271
+ * `signal` here is an OBSERVATION signal, not the SDK's command signal:
272
+ * aborting it stops reading and leaves the command running. That is the
273
+ * opposite of `SandboxExecOptions.signal`, and it is why this type does not
274
+ * extend it.
275
+ */
276
+ export interface KubernetesAttachExecutionOptions {
277
+ /** Byte offset to resume from. Default 0 — the whole retained log. */
278
+ readonly fromOffset?: number;
279
+ readonly onOutput?: SandboxExecOptions['onOutput'];
280
+ /** Stops OBSERVING. Never cancels; see {@link KubernetesAgentTransport.cancelExecution}. */
281
+ readonly signal?: AbortSignal;
282
+ /**
283
+ * Called when the guest reports that bytes the caller asked for had
284
+ * already been evicted from the retained log. The result's truncation
285
+ * flags say the same thing; this says how much.
286
+ */
287
+ readonly onGap?: (gap: {
288
+ readonly executionId: string;
289
+ readonly fromOffset: number;
290
+ readonly droppedBytes: number;
291
+ }) => void;
292
+ }
293
+ /**
294
+ * An `exec()` whose observation can be lost and taken up again — on this
295
+ * handle or in another host process — instead of costing the command its
296
+ * life.
297
+ *
298
+ * Deliberately NOT on the SDK's `SandboxExecOptions`: every other backend
299
+ * would then have to answer for a field it cannot honour, and the SDK's
300
+ * exec contract stays exactly what it was. This is a Kubernetes workspace
301
+ * surface, layered over the shared options type rather than widening it.
302
+ */
303
+ export interface KubernetesDetachedExecOptions extends SandboxExecOptions, Pick<KubernetesAttachExecutionOptions, 'onGap'> {
304
+ /**
305
+ * The id this command is known by, to this process and to any other.
306
+ * Minted here when absent. Reserving an id the guest still holds does
307
+ * NOT start a second command: the call attaches to the one that exists,
308
+ * which is what makes a retried start idempotent for as long as the
309
+ * record lives.
310
+ */
311
+ readonly executionId?: string;
312
+ /**
313
+ * Ask the guest to retain this command's output so the observation can
314
+ * be resumed. Setting `executionId` implies it; the flag is what a
315
+ * caller that does not care about the id passes.
316
+ */
317
+ readonly detach?: boolean;
318
+ /**
319
+ * Stop observing and leave the command running — for a host that is
320
+ * shutting down. Rejects with {@link KubernetesExecutionDetachedError},
321
+ * which names the id and the offset to resume from.
322
+ *
323
+ * The opposite of `signal`, which keeps the SDK contract and terminates.
324
+ */
325
+ readonly detachSignal?: AbortSignal;
326
+ /**
327
+ * How long a lost connection is retried before the call gives up and
328
+ * reports itself detached. Default 30s.
329
+ */
330
+ readonly reattachWindowMs?: number;
331
+ }
332
+ /**
333
+ * Thrown before anything is started, when the caller asked for a session and
334
+ * the guest does not advertise {@link SESSIONS_FEATURE}.
335
+ *
336
+ * Refused, never downgraded to a connection-bound terminal. A caller that
337
+ * asked for a session is about to rely on coming back to it after its own
338
+ * process has been replaced; handing it one that dies with the socket would
339
+ * look like it worked until the one moment it was needed.
340
+ */
341
+ export declare class KubernetesSessionsUnsupportedError extends Error {
342
+ readonly feature: string;
343
+ readonly name = "KubernetesSessionsUnsupportedError";
344
+ constructor(feature: string, message: string);
345
+ }
346
+ /** Why the guest refused a session request. */
347
+ export type KubernetesSessionRefusal = 'unknown_session' | 'invalid_session_id' | 'invalid_offset' | 'session_exists' | 'session_capacity' | 'missing_command' | 'spawn_failed' | 'agent_retiring' | 'unknown';
348
+ /**
349
+ * Thrown when the guest ANSWERED and refused: the session is past its
350
+ * retention, ran in a pod that has since been replaced, the id is already
351
+ * taken, or the offset names bytes it does not have.
352
+ *
353
+ * Distinct from a transport failure for the same reason
354
+ * {@link KubernetesExecutionNotAttachableError} is: a refusal does not
355
+ * become a success by being retried.
356
+ */
357
+ export declare class KubernetesSessionRefusedError extends Error {
358
+ readonly sessionId: string;
359
+ readonly reason: KubernetesSessionRefusal;
360
+ readonly name = "KubernetesSessionRefusedError";
361
+ constructor(sessionId: string, reason: KubernetesSessionRefusal, message: string, options?: {
362
+ cause?: unknown;
363
+ });
364
+ }
365
+ /** One row of {@link KubernetesAgentTransport.listSessions}. */
366
+ export interface KubernetesSessionSummary {
367
+ readonly sessionId: string;
368
+ readonly kind: SessionKind;
369
+ /** The program, as it was asked for. Never the environment it was given. */
370
+ readonly command: string;
371
+ readonly args: readonly string[];
372
+ readonly startedAt: number;
373
+ readonly lastInputAt?: number;
374
+ readonly lastOutputAt?: number;
375
+ /** Pass as `fromOffset` to read everything this session has printed since. */
376
+ readonly nextOffset: number;
377
+ /** Bytes the ring has evicted over this session's life. */
378
+ readonly droppedBytes: number;
379
+ readonly state: SessionState;
380
+ /** Whether a host process is attached to it right now. */
381
+ readonly attached: boolean;
382
+ readonly exitCode?: number;
383
+ readonly signal?: number;
384
+ }
385
+ /**
386
+ * One read of a session's retained output, in the SDK's
387
+ * `BackgroundJobOutput` shape — deliberately, because it answers the same
388
+ * question for the same kind of consumer and a second vocabulary for
389
+ * "here is the next chunk and here is what you missed" helps nobody.
390
+ */
391
+ export interface KubernetesSessionOutput {
392
+ readonly chunk: string;
393
+ readonly nextOffset: number;
394
+ readonly droppedBytes: number;
395
+ readonly status: BackgroundJobStatus;
396
+ readonly exitCode?: number;
397
+ }
398
+ /**
399
+ * A terminal on a workspace, with the three things only a SESSION's reader
400
+ * needs. On a connection-bound terminal the two optional members are absent,
401
+ * which is the honest answer: there is no session to name and nothing to
402
+ * detach from.
403
+ */
404
+ export interface KubernetesWorkspaceTerminal extends TerminalSession {
405
+ /** Present exactly when this terminal belongs to a guest session. */
406
+ readonly sessionId?: string;
407
+ /** One past the newest retained byte delivered so far. */
408
+ nextOffset?(): number | undefined;
409
+ /**
410
+ * Stop reading and leave the program running — the opposite of
411
+ * {@link TerminalSession.kill}. `exited` then rejects with
412
+ * `AgentSessionDetachedError`, because a resolved `exited` would claim
413
+ * an exit that did not happen.
414
+ */
415
+ detach?(): void;
416
+ }
417
+ /**
418
+ * What an ATTACH hands back: the same terminal, with the three session
419
+ * members present rather than optional. `openTerminal` returns the looser
420
+ * type because it serves both shapes and a connection-bound terminal
421
+ * genuinely has no session to name.
422
+ */
423
+ export interface KubernetesSessionTerminal extends KubernetesWorkspaceTerminal {
424
+ readonly sessionId: string;
425
+ nextOffset(): number | undefined;
426
+ detach(): void;
427
+ }
428
+ /** `openTerminal` on a workspace, widened by the two session fields. */
429
+ export interface KubernetesOpenTerminalOptions extends OpenTerminalOptions {
430
+ /**
431
+ * Name this terminal so a later host process can find it again.
432
+ * Requires {@link persistent}; on its own it names nothing.
433
+ */
434
+ readonly sessionId?: string;
435
+ /**
436
+ * Hand the PTY to the guest's session registry rather than to this
437
+ * connection. Losing the connection then DETACHES — no signal is sent,
438
+ * and the program ends when it exits, on `killSession`, or when the pod
439
+ * stops.
440
+ */
441
+ readonly persistent?: boolean;
442
+ }
443
+ /** Rejoin a terminal session that is already running. */
444
+ export interface KubernetesAttachTerminalOptions {
445
+ /** Byte offset to replay from. Default 0 — everything the ring still holds. */
446
+ readonly fromOffset?: number;
447
+ /** Resize the PTY on attach, for a reader whose window is a different shape. */
448
+ readonly size?: {
449
+ readonly cols: number;
450
+ readonly rows: number;
451
+ };
452
+ }
453
+ /** Start a program with no terminal, which only a kill or the pod ends. */
454
+ export interface KubernetesStartDetachedOptions {
455
+ readonly sessionId: string;
456
+ readonly command: string;
457
+ readonly args?: readonly string[];
458
+ readonly cwd?: string;
459
+ readonly env?: Record<string, string>;
460
+ }
461
+ /** Read a session's retained output without attaching to it. */
462
+ export interface KubernetesReadSessionOptions {
463
+ readonly fromOffset?: number;
72
464
  }
73
465
  /**
74
466
  * The kubernetes backend's dialable transport: a `tcp` handle plus a
@@ -78,17 +470,322 @@ export interface KubernetesTransportOptions extends VsockTransportOptions {
78
470
  * remote backend gets, while every network operation is delegated to
79
471
  * {@link VsockAgentTransport} for the actual dial/frame/token work.
80
472
  */
473
+ /**
474
+ * Thrown when a quiesce was asked of a guest that cannot perform one:
475
+ * either its `healthz` does not advertise {@link QUIESCE_FEATURE}, or it
476
+ * answered `unknown_op: quiesce`.
477
+ *
478
+ * Refused rather than treated as "nothing was running". A caller asks for a
479
+ * quiesce because it is about to read the disk and needs it still; an image
480
+ * that cannot stop its processes has to say so, not resolve with an empty
481
+ * list that reads exactly like a guest which had nothing to stop.
482
+ */
483
+ export declare class KubernetesQuiesceUnsupportedError extends Error {
484
+ readonly feature: string;
485
+ readonly name = "KubernetesQuiesceUnsupportedError";
486
+ constructor(feature: string, message: string);
487
+ }
488
+ /**
489
+ * Thrown when nothing can promise the guest is quiet.
490
+ *
491
+ * Usually because the guest ANSWERED and said so — a process survived
492
+ * `SIGKILL`, the scan could not be performed, or the op ran out of its own
493
+ * deadline — and then the guest's own message names the pid. The workspace
494
+ * handle raises the same class for the one case that never reaches a guest:
495
+ * a `suspend({ quiesce: true })` arriving while a suspend WITHOUT a quiesce
496
+ * is already in flight, whose patch has gone over processes nobody stopped
497
+ * (`reason: 'suspend_already_in_flight'`). Both mean the one thing a caller
498
+ * has to act on: do not trust a capture taken now.
499
+ *
500
+ * Its own class for the same reason {@link KubernetesSessionRefusedError}
501
+ * is: this is an answer, not a transport failure, and retrying it is a
502
+ * decision the caller makes with the pid in hand rather than one a wrapper
503
+ * makes on its behalf.
504
+ */
505
+ export declare class KubernetesQuiesceUnconfirmedError extends Error {
506
+ readonly reason: string;
507
+ readonly name = "KubernetesQuiesceUnconfirmedError";
508
+ constructor(reason: string, message: string);
509
+ }
510
+ /** What the guest stopped, and how widely it was allowed to look. */
511
+ export interface KubernetesQuiesceReport {
512
+ /** Every process signalled, with the last signal it was actually sent. */
513
+ readonly stopped: readonly QuiescedProcess[];
514
+ /** See {@link QuiesceScope}. `owned-sessions` is the narrowed one. */
515
+ readonly scope: QuiesceScope;
516
+ /** The per-round SIGTERM window the guest used, after its own clamp. */
517
+ readonly graceMs: number;
518
+ /** Scan-and-signal passes. `0` means nothing was running. */
519
+ readonly rounds: number;
520
+ }
521
+ /**
522
+ * Thrown when a flush was asked of a guest that cannot perform one: either
523
+ * its `healthz` does not advertise {@link FLUSH_FEATURE}, or it answered
524
+ * `unknown_op: flush`.
525
+ *
526
+ * Refused rather than treated as "the disk is already flushed", for the
527
+ * same reason {@link KubernetesQuiesceUnsupportedError} is not treated as
528
+ * "nothing was running". The one caller that does NOT pass this on is
529
+ * `suspend()`, which goes ahead and tells the host through
530
+ * `onFlushUnsupported`: an image built before this op is a deployment that
531
+ * has to be able to suspend its workspaces, not one that has to be stopped.
532
+ */
533
+ export declare class KubernetesFlushUnsupportedError extends Error {
534
+ readonly feature: string;
535
+ readonly name = "KubernetesFlushUnsupportedError";
536
+ constructor(feature: string, message: string);
537
+ }
538
+ /**
539
+ * Thrown when nothing can promise the workspace's writes are on the device.
540
+ *
541
+ * The guest ANSWERED and said so — the `syncfs` failed, or ran past its own
542
+ * timeout — and its message says which. Its own class for the reason every
543
+ * other answered refusal here has one: this is a fact about the disk, not a
544
+ * transport failure, and a caller holding it decides whether to retry,
545
+ * suspend anyway, or leave the workspace running.
546
+ */
547
+ export declare class KubernetesFlushUnconfirmedError extends Error {
548
+ readonly reason: string;
549
+ readonly name = "KubernetesFlushUnconfirmedError";
550
+ constructor(reason: string, message: string);
551
+ }
552
+ /**
553
+ * Carried to {@link KubernetesWorkspaceOptions.onFlushUnreachable} when a
554
+ * `suspend()` could not ASK for a flush at all — the dial failed, the
555
+ * connection timed out, the token was refused, or the agent has fenced
556
+ * itself — and the suspend went ahead without one.
557
+ *
558
+ * Never thrown at a caller. The difference between this and
559
+ * {@link KubernetesFlushUnconfirmedError} is the difference between a guest
560
+ * that could not be reached and a guest that answered: a guest that
561
+ * answered is alive, and stopping the suspend gives its caller something to
562
+ * do about it, while a guest nothing can reach will not become flushable by
563
+ * leaving its pod running — and refusing to suspend over it would take away
564
+ * the one verb an operator reaches for when a workspace is wedged, the verb
565
+ * {@link KubernetesAgentRetiringError} itself names as the way out.
566
+ *
567
+ * So the suspend proceeds, the host is told, and the message says what the
568
+ * disk is resting on instead: whatever the guest kernel had already written
569
+ * back, plus the pod's own `preStop` hook and the agent's `SIGTERM` handler
570
+ * if either of them still runs.
571
+ */
572
+ export declare class KubernetesFlushUnreachableError extends Error {
573
+ readonly name = "KubernetesFlushUnreachableError";
574
+ }
575
+ /** What a flush did, as the guest measured it. */
576
+ export interface KubernetesFlushReport {
577
+ /** How long the `syncfs` took inside the guest. */
578
+ readonly durationMs: number;
579
+ /** The mount the guest flushed — its workspace root. */
580
+ readonly workspace: string;
581
+ }
582
+ /**
583
+ * The refusal a guest that cannot flush earns, built in one place.
584
+ *
585
+ * Exported because the WORKSPACE has to be able to hand this exact error to
586
+ * `onFlushUnsupported` on the one path that does not throw it — a
587
+ * `suspend()` against an older image — and a second copy of the message
588
+ * would be a second thing to keep true.
589
+ */
590
+ export declare function flushUnsupportedError(): KubernetesFlushUnsupportedError;
81
591
  export declare class KubernetesAgentTransport {
82
- private readonly handle;
592
+ /**
593
+ * Mutable: the handle follows a replaced pod — see {@link rebind}.
594
+ *
595
+ * Under `'pod-ip'` both the address and the token move, because the
596
+ * address is a literal that died with its pod. Under the default
597
+ * `'service'` mode the host is a Service FQDN that outlives the pod and
598
+ * resolves to the replacement on its own, so what moves is the TOKEN
599
+ * alone — which is the entire reason a `service` handle needed a rebind
600
+ * at all: the dial keeps working and the guest refuses every call.
601
+ */
602
+ private handle;
603
+ /** The re-read currently in flight, so concurrent refusals share one. */
604
+ private rebinding?;
605
+ /**
606
+ * How many times this transport has moved to a different pod. Captured
607
+ * before every attempt and compared after it fails, so a call refused by
608
+ * the pod it was bound to — arriving after ANOTHER call's rebind already
609
+ * installed the replacement — retries on the handle that has moved
610
+ * instead of re-reading to be told nothing changed.
611
+ */
612
+ private rebindSeq;
83
613
  private readonly transportOptions;
614
+ /** Bound to each wire's own pod by {@link wireOptions}. */
615
+ private readonly onGuestReply?;
84
616
  private readonly onTiming?;
617
+ private readonly refreshHandle?;
85
618
  /** Simple pass-through operations share one transport instance. */
86
- private readonly wire;
619
+ private wire;
87
620
  constructor(handle: KubernetesAgentHandle, options?: KubernetesTransportOptions);
88
- /** Readiness probe — never requires a token; see `protocol.ts`. */
621
+ /**
622
+ * The shared transport's options for ONE wire, with the reply observer
623
+ * bound to the pod that wire is talking to.
624
+ *
625
+ * Every `VsockAgentTransport` this class builds goes through here — the
626
+ * first one, the one a rebind installs, and the per-attempt one `exec()`
627
+ * times — because a wire outlives the moment it was current: the pod it
628
+ * dials can answer a call after a rebind has already moved
629
+ * {@link handle} on, and the observer has to be told the pod that
630
+ * ANSWERED rather than the pod this transport now holds. Capturing the
631
+ * handle in the closure is what makes the two different values.
632
+ */
633
+ private wireOptions;
634
+ /**
635
+ * The address this transport is dialing right now. Diagnostics only —
636
+ * host and port, and deliberately NOT the handle itself: this is read
637
+ * into a log line or an assertion, and the bind token has no business
638
+ * travelling with either.
639
+ */
640
+ get address(): {
641
+ readonly host: string;
642
+ readonly port: number;
643
+ };
644
+ /**
645
+ * Run one operation, and give the handle exactly one chance to follow a
646
+ * pod that was replaced underneath it.
647
+ *
648
+ * TWO failures reach a rebind, and they are the only two, because they
649
+ * are the only two that prove the operation ran NOTHING in the guest:
650
+ *
651
+ * - **the dial failed.** No socket was ever established, so no byte was
652
+ * sent. This is the trigger the `'pod-ip'` mode was built on: the
653
+ * address is a literal that dies with its pod.
654
+ * - **the guest REFUSED the token** (`unauthorized`, in either of the
655
+ * shapes {@link isUnauthorizedRefusal} unifies). The agent checks the
656
+ * token before `dispatch`, so a refused request reached the guest and
657
+ * was thrown away unread. This is the trigger that matters under the
658
+ * DEFAULT `'service'` mode, where the Service FQDN outlives the pod
659
+ * and keeps resolving: the dial SUCCEEDS against the replacement and
660
+ * the refusal is the only thing that says the pod moved.
661
+ *
662
+ * Everything else is the guest's answer to a request it did receive, and
663
+ * nothing here may reinterpret it — not as a resolver problem, and not as
664
+ * a reason to repeat work the guest has already begun.
665
+ *
666
+ * From there both arms run the SAME routine, which is the whole of this
667
+ * change to it: {@link rebind} reads the pod once, and a DIFFERENT uid
668
+ * means the controller replaced it (a resume, an eviction, a node drain),
669
+ * so the handle takes the new address AND the new token together and the
670
+ * operation is retried once.
671
+ *
672
+ * One branch belongs to the dial arm alone, and is taken before the
673
+ * re-read: the handle's host is a NAME and the dial gave up at
674
+ * resolution, so this is a Service FQDN and this host has no resolver for
675
+ * it. Re-reading the pod would change nothing — the next dial would ask
676
+ * the same resolver the same question — so the error is replaced with one
677
+ * that names the FQDN and the configuration field that fixes it. It is
678
+ * never asked of a refusal, which came back over a connection that
679
+ * plainly worked, and the name check is not decoration either: a
680
+ * `pod-ip` handle must keep its one re-read however an unrelated error
681
+ * happens to be worded.
682
+ *
683
+ * Anything else — a re-read that finds the SAME pod, or one that fails —
684
+ * leaves the original error standing. A pod that is still there and still
685
+ * refusing is a guest problem, and replacing that error with a second,
686
+ * later one would hide it. The one exception is a re-read that comes back
687
+ * with a verdict about the OBJECT rather than a failed diagnosis
688
+ * ({@link KubernetesWorkspaceReplacedError}): the disk behind the name is
689
+ * not this handle's disk, which outranks whatever uncovered it.
690
+ *
691
+ * `signal` is the caller's own, and it decides one thing only: a call that
692
+ * has ALREADY been cancelled is not owed a re-read, because the retry it
693
+ * would buy would abort before it left. It is never handed to the re-read
694
+ * itself, which is shared and runs under a bound of its own — see
695
+ * {@link rebind}.
696
+ *
697
+ * `dials` is how `exec()` answers the first question at all. Its failure
698
+ * can arrive as a bare timeout from the execution controller's own bound,
699
+ * with the dial's error discarded rather than wrapped, so that path
700
+ * watches its dials instead of reading its error — see {@link DialWatch}.
701
+ * Every other operation hands back the dial's own error and passes none.
702
+ *
703
+ * A refusal that no rebind could fix leaves as {@link
704
+ * KubernetesAgentUnauthorizedError} whichever shape it arrived in — the
705
+ * unification the whole recovery hangs off, applied at the one point
706
+ * where "nothing else can be done about it" is known.
707
+ */
708
+ private withRebind;
709
+ /**
710
+ * Whether the dial gave up at name resolution.
711
+ *
712
+ * Asked of the caller's error first, and of the watched dial's own error
713
+ * only when nothing ever connected — the case where the error the caller
714
+ * holds is a bound's timer rather than the failure that caused it. A
715
+ * watched attempt that DID connect is never consulted: its error is the
716
+ * guest's answer, however it happens to be worded.
717
+ */
718
+ private failedToResolve;
719
+ /**
720
+ * The one connect failure waiting cannot cure — see the constructor.
721
+ * A method rather than a closure because the watched `exec()` dials wrap
722
+ * it, and a caller-visible answer must not depend on which wire asked.
723
+ */
724
+ private dialFailureIsPermanent;
725
+ /**
726
+ * One re-read, shared by every call that needs one at the same moment.
727
+ * True when the handle now points at a DIFFERENT pod than the one the
728
+ * caller's attempt was dispatched on — whether this call's own re-read
729
+ * moved it or another call's already had.
730
+ *
731
+ * Single-flight, and that is not an optimisation. A pod replaced
732
+ * underneath a busy handle refuses EVERY call in flight at once, and one
733
+ * re-read per refused call would be a burst of Sandbox and pod GETs, each
734
+ * one racing the others to install a handle — with the last to finish
735
+ * winning, which is not necessarily the last to read. Sharing the promise
736
+ * makes the burst one round trip and one installation, in the order the
737
+ * API answered.
738
+ *
739
+ * The slot is cleared inside the shared run, before the promise settles,
740
+ * so a caller that awaits it and is refused AGAIN gets a fresh re-read
741
+ * rather than the answer to the previous question.
742
+ *
743
+ * Because it is shared it runs under {@link REBIND_READ_TIMEOUT_MS} and
744
+ * under NO caller's signal. A caller that aborts while the shared read is
745
+ * in flight would otherwise abort it for everyone, and every call the
746
+ * replacement refused would fail with its original refusal although the
747
+ * handle could have followed the pod. Its own bound is what the aborting
748
+ * caller was owed — nobody waits on a re-read for longer than that — and
749
+ * the read is two GETs nobody is billed for twice.
750
+ */
751
+ private rebind;
752
+ private runRebind;
753
+ /** Whether the current handle's host goes through a resolver at all. */
754
+ private dialsAName;
755
+ /** The DNS-shaped failure, in words that name the way out of it. */
756
+ private unresolvable;
757
+ /**
758
+ * Readiness probe — never requires a token; see `protocol.ts`.
759
+ *
760
+ * Deliberately NOT wrapped in {@link withRebind}: `healthz` answers a
761
+ * failed dial with `false` rather than by throwing, so there is no error
762
+ * to classify and nothing for a re-read to be triggered by. A caller that
763
+ * wants the reason asks for it by making a real call.
764
+ */
89
765
  healthz(signal?: AbortSignal): Promise<boolean>;
90
- /** Poll until `healthz` succeeds or the timeout elapses. */
766
+ /**
767
+ * Poll until `healthz` succeeds or the timeout elapses. Unwrapped for the
768
+ * same reason, and for one more: it already owns a retry loop, so a
769
+ * connect failure here is not a single failed dial but a whole budget of
770
+ * them.
771
+ */
91
772
  waitForReady(timeoutMs: number, pollIntervalMs: number, signal?: AbortSignal): Promise<void>;
773
+ /**
774
+ * Ask the agent how it is, and keep the two "not ok" answers apart —
775
+ * see {@link KubernetesAgentHealth}.
776
+ *
777
+ * Deliberately NOT wrapped in {@link withRebind}: this is a diagnostic
778
+ * about the pod this handle is bound to RIGHT NOW, and a rebind would
779
+ * silently answer it about a different pod. A caller that wants to know
780
+ * whether the agent it was talking to has fenced itself would then be
781
+ * told about the replacement, which is a different question with a
782
+ * different answer.
783
+ *
784
+ * It throws whatever the dial or the read threw. An agent that cannot be
785
+ * reached has no health to report, and inventing one here would turn
786
+ * "unreachable" into "fine".
787
+ */
788
+ agentHealth(signal?: AbortSignal): Promise<KubernetesAgentHealth>;
92
789
  /**
93
790
  * The raw `reserve-execution` primitive, exposed directly (rather than
94
791
  * only reachable as a side effect of `exec()`) so the reservation
@@ -103,9 +800,166 @@ export declare class KubernetesAgentTransport {
103
800
  * driving a whole cancelled `exec()` to reach it.
104
801
  */
105
802
  cancel(executionId: string, signal?: AbortSignal): Promise<unknown>;
106
- writeFile(path: string, content: Buffer): Promise<void>;
107
- readFile(path: string): Promise<Buffer>;
108
- openTerminal(options: OpenTerminalOptions): Promise<TerminalSession>;
803
+ /**
804
+ * Delegated, `signal` included: a body larger than one pre-auth frame
805
+ * is written as a sequence of parts by {@link VsockAgentTransport}
806
+ * itself, and a cancelled sequence has to be able to stop mid-way and
807
+ * take its temp file with it.
808
+ *
809
+ * One transport instance, deliberately — {@link wire} is shared by
810
+ * every simple pass-through op — so the one `healthz` probe that asks
811
+ * the guest whether it can take parts is asked once for this sandbox,
812
+ * not once per large write.
813
+ *
814
+ * Wrapped in {@link withRebind} like every other guest-dialling op,
815
+ * the multi-part route included, and for both of its triggers: a retry
816
+ * starts a fresh sequence under a new temp name
817
+ * (`writeFilePartTempPath` mints a UUID per call), so a sequence
818
+ * abandoned mid-way on the replaced pod — because the dial failed, or
819
+ * because the replacement refused this handle's token before reading a
820
+ * byte — cannot collide with the retry's offsets and never touched the
821
+ * target. At worst it leaves one orphan temp file behind on the
822
+ * workspace volume.
823
+ */
824
+ writeFile(path: string, content: Buffer, signal?: AbortSignal): Promise<void>;
825
+ readFile(path: string, options?: SandboxReadFileOptions): Promise<Buffer>;
826
+ /**
827
+ * Delegated, and rebound exactly once — but only around the FIRST
828
+ * chunk.
829
+ *
830
+ * That is the whole of what {@link withRebind} can honestly cover here.
831
+ * Its retry is safe because nothing reached the guest, and once a chunk
832
+ * has been yielded that is no longer true: re-dialing a replaced pod
833
+ * mid-stream would restart the file from its beginning, and the
834
+ * consumer — which has already taken the bytes and cannot give them
835
+ * back — would silently concatenate a duplicate prefix. So the first
836
+ * pull carries the dial, the rebind and the retry; everything after it
837
+ * fails as itself.
838
+ */
839
+ readFileStream(path: string, options?: SandboxReadFileOptions): AsyncGenerator<Buffer, void, undefined>;
840
+ /**
841
+ * A guest PTY. Without `sessionId`/`persistent` this is exactly the
842
+ * terminal it has always been, down to the wire request.
843
+ *
844
+ * With them the PTY belongs to the guest's session registry: losing this
845
+ * connection detaches rather than killing, a later process rejoins it
846
+ * with {@link attachSession}, and the capability is verified against the
847
+ * guest's `healthz` features BEFORE the shell is started — never
848
+ * downgraded to a connection-bound terminal, which would look like it
849
+ * worked until the rollout it exists for.
850
+ */
851
+ openTerminal(options: KubernetesOpenTerminalOptions): Promise<KubernetesWorkspaceTerminal>;
852
+ /**
853
+ * One open of a session stream, with the guest's refusal mapped to
854
+ * {@link KubernetesSessionRefusedError} — see {@link sessionStreamFailure}.
855
+ */
856
+ private sessionStream;
857
+ /**
858
+ * Rejoin a terminal session, replaying from `fromOffset` and then
859
+ * following it live.
860
+ *
861
+ * The guest allows one attachment per session and ends the previous one
862
+ * by name, so two host processes cannot interleave keystrokes into one
863
+ * shell. Losing this connection detaches; ending the program is
864
+ * {@link killSession} and nothing else.
865
+ */
866
+ attachSession(sessionId: string, options?: KubernetesAttachTerminalOptions): Promise<KubernetesSessionTerminal>;
867
+ /**
868
+ * Start a program with no terminal at all, in its own kernel session,
869
+ * with stdin closed and both output streams going into the guest's
870
+ * retained log.
871
+ *
872
+ * It is not the SDK's `spawnDetached` and deliberately does not pretend
873
+ * to be: that one hands back a host `ChildProcess`, which cannot cross a
874
+ * process boundary. This returns a NAME, and the name is what a
875
+ * redeployed host comes back with.
876
+ */
877
+ startDetached(options: KubernetesStartDetachedOptions, signal?: AbortSignal): Promise<KubernetesSessionSummary>;
878
+ /** Every session this pod's agent is holding, running and recently exited. */
879
+ listSessions(signal?: AbortSignal): Promise<readonly KubernetesSessionSummary[]>;
880
+ /**
881
+ * End one session and everything still in it — the shell, its
882
+ * backgrounded jobs, and the program a detached session started.
883
+ *
884
+ * Idempotent: a session that has already exited answers with what it
885
+ * exited with. The reply carries the session's state, so a program that
886
+ * ignored a `SIGTERM` and outlived the guest's confirm window is
887
+ * reported still running rather than reported dead.
888
+ */
889
+ killSession(sessionId: string, options?: {
890
+ readonly signal?: string;
891
+ readonly abort?: AbortSignal;
892
+ }): Promise<KubernetesSessionSummary>;
893
+ /**
894
+ * Read what a session has printed since `fromOffset`, without attaching
895
+ * to it and without signalling anything.
896
+ *
897
+ * One request, one answer, in the SDK's `BackgroundJobOutput` shape: the
898
+ * chunk, the offset to come back with, the bytes the ring dropped before
899
+ * it, and the program's status. A caller polling in a loop can neither
900
+ * re-read nor skip, because the offset is the guest's own.
901
+ */
902
+ readSession(sessionId: string, options?: KubernetesReadSessionOptions, signal?: AbortSignal): Promise<KubernetesSessionOutput>;
903
+ /**
904
+ * Stop every process this pod's guest is running, and keep the agent.
905
+ *
906
+ * The point is the pair. Stopping the POD stops its processes too, but
907
+ * leaves nothing to read the disk through, so a host that wants a capture
908
+ * it can trust has to wake the workspace again and check. After this the
909
+ * guest is quiet and still serving: `exec`, `readFile` and `writeFile`
910
+ * all work, and what they see is a filesystem nobody is writing to.
911
+ *
912
+ * Open terminals and running commands end as a side effect and report it
913
+ * through their own exit and result paths — a terminal receives its exit,
914
+ * an `exec` resolves with a signal in its result. Rejecting with
915
+ * {@link KubernetesQuiesceUnconfirmedError} is the honest failure: a
916
+ * process would not stop, and the reply names its pid.
917
+ */
918
+ quiesce(options?: {
919
+ readonly graceMs?: number;
920
+ }, signal?: AbortSignal): Promise<KubernetesQuiesceReport>;
921
+ /**
922
+ * Put everything this workspace has written onto its device.
923
+ *
924
+ * The disk a workspace keeps is whatever the guest kernel happened to
925
+ * write back. Nothing in this backend ever asked for more than that: a
926
+ * `suspend()` patched the pod away and waited for it to stop, and a
927
+ * stopped pod means only that nothing is writing any more — not that
928
+ * what was written arrived. This is the call that closes the difference,
929
+ * and it is `syncfs(2)` over the whole workspace mount, so it covers
930
+ * what a COMMAND wrote as well as what `writeFile` did (`writeFile`
931
+ * fsyncs its own bytes before it answers; a compiler's output is nobody's
932
+ * to fsync).
933
+ *
934
+ * Rejecting with {@link KubernetesFlushUnconfirmedError} is the honest
935
+ * failure, exactly as an unconfirmed quiesce is: the caller is usually
936
+ * about to take the pod away, and "the flush did not run" has to be
937
+ * distinguishable from "the flush ran".
938
+ */
939
+ flush(options?: {
940
+ readonly timeoutMs?: number;
941
+ }, signal?: AbortSignal): Promise<KubernetesFlushReport>;
942
+ /**
943
+ * Whether this guest can flush at all — asked, rather than assumed,
944
+ * because a `suspend()` against an older image keeps today's behaviour
945
+ * and reports the gap instead of refusing to suspend.
946
+ */
947
+ supportsFlush(signal?: AbortSignal): Promise<boolean>;
948
+ private assertFlushSupported;
949
+ private flushUnsupported;
950
+ /**
951
+ * Whether this guest can quiesce at all — asked, rather than assumed,
952
+ * because `suspend({ quiesce: true })` against an older image keeps
953
+ * today's behaviour and reports the gap instead of refusing to suspend.
954
+ */
955
+ supportsQuiesce(signal?: AbortSignal): Promise<boolean>;
956
+ private assertQuiesceSupported;
957
+ private quiesceUnsupported;
958
+ /**
959
+ * Whether this guest keeps a session registry at all, asked once per
960
+ * transport and only when a caller wants one.
961
+ */
962
+ private assertSessionsSupported;
109
963
  openTcpConnection(options: SandboxTcpConnectOptions): Promise<SandboxTcpConnection>;
110
964
  /**
111
965
  * Run one command through a fresh, call-scoped adapter + controller.
@@ -118,5 +972,77 @@ export declare class KubernetesAgentTransport {
118
972
  * mutable field for two in-flight calls to race on.
119
973
  */
120
974
  exec(command: string, argv?: string[], opts?: SandboxExecOptions): Promise<SandboxExecResult>;
975
+ /**
976
+ * Whether this guest implements the detach/attach ops at all, asked once
977
+ * per transport and only when a caller wants them.
978
+ */
979
+ private assertExecutionAttachSupported;
980
+ /** `reserve-execution` for a caller-named id, with its reported state. */
981
+ private reserveDetached;
982
+ /**
983
+ * The `cancel-execution` control path, retried for its whole confirm
984
+ * window and reported UNKNOWN rather than as a failure if none of the
985
+ * attempts got an answer — the same rule the shared execution
986
+ * controller applies, because a command whose termination nobody
987
+ * confirmed is not a command anybody may call dead.
988
+ *
989
+ * Nothing on the detach path calls this to reconcile a lost connection.
990
+ * It runs when the CALLER asked for it: `SandboxExecOptions.signal`
991
+ * aborting, or {@link cancelExecution}.
992
+ */
993
+ private confirmCancel;
994
+ /**
995
+ * End a command by id, from any host process holding the id and the
996
+ * bind token. Resolves only on a CONFIRMED termination.
997
+ *
998
+ * The outcome is not reported here — a cancelled execution's result and
999
+ * whatever output it managed is read back through
1000
+ * {@link attachExecution}, which is the op that exists for reading.
1001
+ */
1002
+ cancelExecution(executionId: string, signal?: AbortSignal): Promise<void>;
1003
+ /**
1004
+ * Observe a command that is already the guest's, from `fromOffset` on,
1005
+ * and resolve with its result.
1006
+ *
1007
+ * It never signals the command. Aborting `signal` stops OBSERVING and
1008
+ * rejects with {@link KubernetesExecutionDetachedError}; it does not
1009
+ * cancel, and the command goes on running. Ending a command is
1010
+ * {@link cancelExecution} and nothing else.
1011
+ */
1012
+ attachExecution(executionId: string, options?: KubernetesAttachExecutionOptions): Promise<SandboxExecResult>;
1013
+ /**
1014
+ * Run one command whose observation can outlive this connection, and —
1015
+ * when the connection is what failed — get it back rather than killing
1016
+ * the command to reconcile.
1017
+ *
1018
+ * The order is exactly: refuse if the guest cannot keep output, reserve
1019
+ * the id, and only then admit the command. Reserving the id the CALLER
1020
+ * named is what makes a retried start idempotent: a second call with the
1021
+ * same id inside retention finds the execution already running or
1022
+ * finished, sends no `execute`, and attaches to the one that exists.
1023
+ *
1024
+ * `SandboxExecOptions.signal` keeps its contract — aborting it runs the
1025
+ * confirmed cancel — and `detachSignal` is its opposite: it ends the
1026
+ * observation and leaves the command alone, for a host that is shutting
1027
+ * down and wants its work to survive the rollout.
1028
+ */
1029
+ execDetached(command: string, argv?: string[], opts?: KubernetesDetachedExecOptions): Promise<SandboxExecResult>;
1030
+ /**
1031
+ * The `execute` leg of a detached run, read frame by frame.
1032
+ *
1033
+ * It deliberately does NOT go through `executeStreamed`: that path
1034
+ * hands its caller `{stream, data}` and nothing else, and the byte
1035
+ * offsets this cursor lives on are on the frames themselves. Reading
1036
+ * them here is what lets a reattach resume exactly where this
1037
+ * connection stopped, on output whose decoded length is not its byte
1038
+ * length — see {@link applyDelta}. The frame union and its validation
1039
+ * are the shared ones, so an ordinary exec and a detached one never
1040
+ * disagree about what the guest said.
1041
+ */
1042
+ private executeRetained;
1043
+ /** One `attach-execution` stream, read to its terminal frame. */
1044
+ private attachOnce;
1045
+ /** The one error a lost observation ends with. */
1046
+ private detached;
121
1047
  }
122
1048
  //# sourceMappingURL=transport.d.ts.map