@namzu/sandbox 14.0.0 → 16.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +924 -0
- package/README.md +369 -14
- package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
- package/dist/backends/aci-standby-pool/index.js +13 -1
- package/dist/backends/aci-standby-pool/index.js.map +1 -1
- package/dist/backends/docker/index.d.ts +169 -6
- package/dist/backends/docker/index.d.ts.map +1 -1
- package/dist/backends/docker/index.js +499 -85
- package/dist/backends/docker/index.js.map +1 -1
- package/dist/backends/firecracker/index.d.ts.map +1 -1
- package/dist/backends/firecracker/index.js +12 -2
- package/dist/backends/firecracker/index.js.map +1 -1
- package/dist/backends/firecracker/protocol.d.ts +459 -8
- package/dist/backends/firecracker/protocol.d.ts.map +1 -1
- package/dist/backends/firecracker/protocol.js +136 -0
- package/dist/backends/firecracker/protocol.js.map +1 -1
- package/dist/backends/firecracker/transport.d.ts +539 -6
- package/dist/backends/firecracker/transport.d.ts.map +1 -1
- package/dist/backends/firecracker/transport.js +1171 -24
- package/dist/backends/firecracker/transport.js.map +1 -1
- package/dist/backends/kubernetes/egress-policy.d.ts +1181 -13
- package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -1
- package/dist/backends/kubernetes/egress-policy.js +2350 -31
- package/dist/backends/kubernetes/egress-policy.js.map +1 -1
- package/dist/backends/kubernetes/identity.d.ts +193 -0
- package/dist/backends/kubernetes/identity.d.ts.map +1 -0
- package/dist/backends/kubernetes/identity.js +147 -0
- package/dist/backends/kubernetes/identity.js.map +1 -0
- package/dist/backends/kubernetes/index.d.ts +678 -33
- package/dist/backends/kubernetes/index.d.ts.map +1 -1
- package/dist/backends/kubernetes/index.js +1180 -95
- package/dist/backends/kubernetes/index.js.map +1 -1
- package/dist/backends/kubernetes/ingress-policy.d.ts +375 -0
- package/dist/backends/kubernetes/ingress-policy.d.ts.map +1 -0
- package/dist/backends/kubernetes/ingress-policy.js +1050 -0
- package/dist/backends/kubernetes/ingress-policy.js.map +1 -0
- package/dist/backends/kubernetes/k8s-client.d.ts +213 -4
- package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -1
- package/dist/backends/kubernetes/k8s-client.js +359 -52
- package/dist/backends/kubernetes/k8s-client.js.map +1 -1
- package/dist/backends/kubernetes/lease.d.ts +40 -14
- package/dist/backends/kubernetes/lease.d.ts.map +1 -1
- package/dist/backends/kubernetes/lease.js +68 -18
- package/dist/backends/kubernetes/lease.js.map +1 -1
- package/dist/backends/kubernetes/objects.d.ts +423 -3
- package/dist/backends/kubernetes/objects.d.ts.map +1 -1
- package/dist/backends/kubernetes/objects.js +364 -2
- package/dist/backends/kubernetes/objects.js.map +1 -1
- package/dist/backends/kubernetes/per-sandbox-policy.d.ts +219 -0
- package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -0
- package/dist/backends/kubernetes/per-sandbox-policy.js +375 -0
- package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -0
- package/dist/backends/kubernetes/rbac.d.ts +153 -0
- package/dist/backends/kubernetes/rbac.d.ts.map +1 -0
- package/dist/backends/kubernetes/rbac.js +177 -0
- package/dist/backends/kubernetes/rbac.js.map +1 -0
- package/dist/backends/kubernetes/sandbox.d.ts +81 -14
- package/dist/backends/kubernetes/sandbox.d.ts.map +1 -1
- package/dist/backends/kubernetes/sandbox.js +149 -15
- package/dist/backends/kubernetes/sandbox.js.map +1 -1
- package/dist/backends/kubernetes/transport.d.ts +935 -9
- package/dist/backends/kubernetes/transport.d.ts.map +1 -1
- package/dist/backends/kubernetes/transport.js +1958 -62
- package/dist/backends/kubernetes/transport.js.map +1 -1
- package/dist/backends/kubernetes/workspace.d.ts +1149 -18
- package/dist/backends/kubernetes/workspace.d.ts.map +1 -1
- package/dist/backends/kubernetes/workspace.js +2825 -186
- package/dist/backends/kubernetes/workspace.js.map +1 -1
- package/dist/backends/remote-execution-controller.d.ts +14 -0
- package/dist/backends/remote-execution-controller.d.ts.map +1 -1
- package/dist/backends/remote-execution-controller.js.map +1 -1
- package/dist/index.d.ts +294 -18
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +280 -10
- package/dist/index.js.map +1 -1
- package/dist/testing/sandbox-conformance.d.ts +39 -5
- package/dist/testing/sandbox-conformance.d.ts.map +1 -1
- package/dist/testing/sandbox-conformance.js +436 -5
- package/dist/testing/sandbox-conformance.js.map +1 -1
- package/package.json +3 -3
- package/src/backends/aci-standby-pool/index.ts +16 -1
- package/src/backends/docker/index.ts +617 -100
- package/src/backends/firecracker/index.ts +14 -2
- package/src/backends/firecracker/protocol.ts +514 -6
- package/src/backends/firecracker/transport.ts +1492 -40
- package/src/backends/kubernetes/egress-policy.ts +3334 -55
- package/src/backends/kubernetes/identity.ts +261 -0
- package/src/backends/kubernetes/index.ts +1785 -127
- package/src/backends/kubernetes/ingress-policy.ts +1344 -0
- package/src/backends/kubernetes/k8s-client.ts +444 -54
- package/src/backends/kubernetes/lease.ts +75 -19
- package/src/backends/kubernetes/objects.ts +626 -6
- package/src/backends/kubernetes/per-sandbox-policy.ts +497 -0
- package/src/backends/kubernetes/rbac.ts +192 -0
- package/src/backends/kubernetes/sandbox.ts +218 -20
- package/src/backends/kubernetes/transport.ts +2733 -124
- package/src/backends/kubernetes/workspace.ts +4476 -222
- package/src/backends/remote-execution-controller.ts +14 -0
- package/src/index.ts +668 -19
- package/src/testing/sandbox-conformance.ts +540 -5
|
@@ -23,7 +23,8 @@
|
|
|
23
23
|
* sub-second warm-acquire target this backend is judged against must be
|
|
24
24
|
* measured, not guessed at.
|
|
25
25
|
*/
|
|
26
|
-
import type { OpenTerminalOptions, SandboxExecOptions, SandboxExecResult, SandboxTcpConnectOptions, SandboxTcpConnection, TerminalSession } from '@namzu/sdk';
|
|
26
|
+
import type { BackgroundJobStatus, OpenTerminalOptions, SandboxExecOptions, SandboxExecResult, SandboxReadFileOptions, SandboxTcpConnectOptions, SandboxTcpConnection, TerminalSession } from '@namzu/sdk';
|
|
27
|
+
import { type GuestReplyIdentity, type QuiesceScope, type QuiescedProcess, type SessionKind, type SessionState } from '../firecracker/protocol.js';
|
|
27
28
|
import { type SandboxAgentHandle, type VsockTransportOptions } from '../firecracker/transport.js';
|
|
28
29
|
/** The one {@link SandboxAgentHandle} arm this backend ever constructs. */
|
|
29
30
|
export type KubernetesAgentHandle = Extract<SandboxAgentHandle, {
|
|
@@ -37,8 +38,73 @@ export type KubernetesAgentHandle = Extract<SandboxAgentHandle, {
|
|
|
37
38
|
* unexpected".
|
|
38
39
|
*/
|
|
39
40
|
export declare class KubernetesAgentUnauthorizedError extends Error {
|
|
41
|
+
constructor(message?: string, options?: ErrorOptions);
|
|
42
|
+
}
|
|
43
|
+
/**
|
|
44
|
+
* Thrown when the guest agent refuses a request because it has FENCED
|
|
45
|
+
* ITSELF: an earlier process group's shutdown could not be confirmed, so it
|
|
46
|
+
* answers every op but `healthz` and `cancel-execution` with `agent_retiring`
|
|
47
|
+
* and will go on doing so until the pod is replaced.
|
|
48
|
+
*
|
|
49
|
+
* Its own class, and deliberately NOT the Firecracker tier's mapping of the
|
|
50
|
+
* same refusal. There a fenced agent becomes {@link
|
|
51
|
+
* RemoteCancellationUnknownError}, which is correct for a disposable microVM:
|
|
52
|
+
* the shared controller's rule is that the sandbox stops being reusable, and
|
|
53
|
+
* on that tier retiring one means deleting scratch. On a workspace the same
|
|
54
|
+
* error would retire the handle and take the pod — and with it every other
|
|
55
|
+
* holder's terminals, dev servers and running commands — away from callers
|
|
56
|
+
* who did nothing but share a workspace with the command that wedged.
|
|
57
|
+
*
|
|
58
|
+
* So this error refuses the ONE call rather than the workspace: the handle is
|
|
59
|
+
* not retired, nothing is patched, and every other holder's pod stays where it
|
|
60
|
+
* was. What it does NOT claim is that the next call will work. The fence is
|
|
61
|
+
* the GUEST's, and `dispatch` gates it ahead of every data-plane branch, so
|
|
62
|
+
* `readFile`, `writeFile`, `openTerminal` and `openTcpConnection` meet the
|
|
63
|
+
* same refusal on the wire — under their own paths' error shapes, since only
|
|
64
|
+
* the two control ops come through `requestChecked`. Only a new pod clears
|
|
65
|
+
* it, which is why the message names the verbs that REPLACE the pod, on both
|
|
66
|
+
* tiers that use this transport, and leaves the timing to the host: on a
|
|
67
|
+
* workspace those are `suspend()` then `resume()`, and they take the live
|
|
68
|
+
* sessions in that pod down with them.
|
|
69
|
+
*/
|
|
70
|
+
export declare class KubernetesAgentRetiringError extends Error {
|
|
71
|
+
readonly name = "KubernetesAgentRetiringError";
|
|
40
72
|
constructor(message?: string);
|
|
41
73
|
}
|
|
74
|
+
/**
|
|
75
|
+
* Thrown when the agent's address could not be RESOLVED — the dial never
|
|
76
|
+
* reached a socket because the name has no answer here.
|
|
77
|
+
*
|
|
78
|
+
* Its own class, and its own message, because this is the one failure whose
|
|
79
|
+
* cause is the deployment's shape rather than anything the cluster did: a
|
|
80
|
+
* Service FQDN resolves through cluster DNS and nowhere else, so a host
|
|
81
|
+
* outside the cluster fails every call at name resolution and reads the
|
|
82
|
+
* result as a sandbox that never came up. The fix is a configuration field,
|
|
83
|
+
* so the error names it.
|
|
84
|
+
*/
|
|
85
|
+
export declare class KubernetesAgentAddressUnresolvableError extends Error {
|
|
86
|
+
/** The host that did not resolve — normally a `*.svc.cluster.local`. */
|
|
87
|
+
readonly host: string;
|
|
88
|
+
readonly name = "KubernetesAgentAddressUnresolvableError";
|
|
89
|
+
constructor(
|
|
90
|
+
/** The host that did not resolve — normally a `*.svc.cluster.local`. */
|
|
91
|
+
host: string, message: string, options?: {
|
|
92
|
+
cause?: unknown;
|
|
93
|
+
});
|
|
94
|
+
}
|
|
95
|
+
/** The pod and process one execution reserved on — see {@link executionGuest}. */
|
|
96
|
+
export interface KubernetesReservedGuest {
|
|
97
|
+
/** The bind token the attempt presented, which is the pod's uid. */
|
|
98
|
+
readonly podUid?: string;
|
|
99
|
+
/** The boot id the `reserve-execution` reply carried, when it carried one. */
|
|
100
|
+
readonly guestBootId?: string;
|
|
101
|
+
}
|
|
102
|
+
/**
|
|
103
|
+
* The guest that reserved the execution this error came from — see
|
|
104
|
+
* {@link executionGuest}. `undefined` when the error is not an execution
|
|
105
|
+
* failure this transport produced.
|
|
106
|
+
*/
|
|
107
|
+
export declare function guestWhenReserved(error: unknown): KubernetesReservedGuest | undefined;
|
|
42
108
|
/**
|
|
43
109
|
* One `exec()` call's wall-time breakdown. Durations are NOT a partition
|
|
44
110
|
* of a single total — `reserveMs` and `executeMs` each include their OWN
|
|
@@ -62,13 +128,339 @@ export interface KubernetesTransportTiming {
|
|
|
62
128
|
*/
|
|
63
129
|
readonly drainMs: number;
|
|
64
130
|
}
|
|
65
|
-
|
|
131
|
+
/**
|
|
132
|
+
* What one `healthz` reply says about the agent, with the fence kept rather
|
|
133
|
+
* than collapsed into a boolean.
|
|
134
|
+
*
|
|
135
|
+
* {@link VsockAgentTransport.healthz} answers `false` both for an agent that
|
|
136
|
+
* did not reply and for one that replied "I have fenced myself", and those
|
|
137
|
+
* are opposite facts for a host deciding what to do next: the first is a pod
|
|
138
|
+
* that may be perfectly fine a second from now, the second is a pod that will
|
|
139
|
+
* refuse every call until it is replaced.
|
|
140
|
+
*/
|
|
141
|
+
export interface KubernetesAgentHealth {
|
|
142
|
+
/** The reply's own `ok` — `true` only for an agent serving normally. */
|
|
143
|
+
readonly ok: boolean;
|
|
144
|
+
/**
|
|
145
|
+
* The agent has fenced itself and only a new pod clears it.
|
|
146
|
+
*
|
|
147
|
+
* Read from the reply's own `retiring` flag and from nothing else. A
|
|
148
|
+
* not-`ok` reply without it is NOT inferred to be a fence: the connection
|
|
149
|
+
* gate answers a `healthz` that arrived over one unauthenticated
|
|
150
|
+
* connection too many, or behind an exhausted pre-auth buffer, with a
|
|
151
|
+
* named `{ ok: false, error }` and no flag — and a caller told `retiring`
|
|
152
|
+
* there would suspend and resume a perfectly healthy pod. So `ok: false`
|
|
153
|
+
* with `retiring: false` is its own answer: this reply says nothing about
|
|
154
|
+
* whether the agent is serving.
|
|
155
|
+
*/
|
|
156
|
+
readonly retiring: boolean;
|
|
157
|
+
}
|
|
158
|
+
/**
|
|
159
|
+
* `permanentDialFailure` is deliberately NOT inherited: this transport sets
|
|
160
|
+
* its own (see the constructor), so advertising the field would be offering a
|
|
161
|
+
* caller a predicate that is silently overwritten. `onGuestReply` is
|
|
162
|
+
* re-declared rather than inherited, with the pod the reply came FROM — see
|
|
163
|
+
* below.
|
|
164
|
+
*/
|
|
165
|
+
export interface KubernetesTransportOptions extends Omit<VsockTransportOptions, 'permanentDialFailure' | 'onGuestReply'> {
|
|
66
166
|
/**
|
|
67
167
|
* Fires once per completed `exec()` call (success or failure) with
|
|
68
168
|
* the four phase durations above. The payload is exactly those four
|
|
69
169
|
* numbers — never the token, never a command, argv, or output.
|
|
70
170
|
*/
|
|
71
171
|
readonly onTiming?: (timing: KubernetesTransportTiming) => void;
|
|
172
|
+
/**
|
|
173
|
+
* Re-read the live pod behind this sandbox and hand back the address
|
|
174
|
+
* and token it answers on NOW.
|
|
175
|
+
*
|
|
176
|
+
* Set only by the `pod-ip` address mode, where the handle carries a
|
|
177
|
+
* literal IP that dies with its pod; a Service FQDN needs none of this
|
|
178
|
+
* because the name outlives the pod and the dial re-resolves it every
|
|
179
|
+
* call. Consulted at most ONCE per call, and only after a dial that
|
|
180
|
+
* failed at connect — see {@link KubernetesAgentTransport}.
|
|
181
|
+
*/
|
|
182
|
+
readonly refreshHandle?: (signal?: AbortSignal) => Promise<KubernetesAgentHandle>;
|
|
183
|
+
/**
|
|
184
|
+
* {@link VsockTransportOptions.onGuestReply}, plus the one thing the
|
|
185
|
+
* shared transport cannot say and this one always can: the bind token of
|
|
186
|
+
* the WIRE the reply arrived on, which is the uid of the pod that
|
|
187
|
+
* answered.
|
|
188
|
+
*
|
|
189
|
+
* A rebind builds a replacement wire and does not close the wire it
|
|
190
|
+
* leaves, so a call dispatched to the outgoing pod can still be answered
|
|
191
|
+
* BY it — a long `write-file`, a `read-file`, a pod inside its
|
|
192
|
+
* termination grace period — minutes after this transport has followed
|
|
193
|
+
* the replacement. That reply is a true statement about a pod this
|
|
194
|
+
* transport is no longer bound to, and a listener that could not tell the
|
|
195
|
+
* two apart would pair the departed pod's agent process with the
|
|
196
|
+
* replacement's uid: an identity naming a process that never ran there.
|
|
197
|
+
*
|
|
198
|
+
* Bound per wire, so the token is the one the reply's own connection
|
|
199
|
+
* presented and never the one the transport happens to hold now —
|
|
200
|
+
* `undefined` on a handle carrying no token at all, which is a wire that
|
|
201
|
+
* cannot name the pod that answered and so proves nothing about it.
|
|
202
|
+
*/
|
|
203
|
+
readonly onGuestReply?: (reply: GuestReplyIdentity, podUid: string | undefined) => void;
|
|
204
|
+
}
|
|
205
|
+
/**
|
|
206
|
+
* Thrown before a command is admitted, when the caller asked for a
|
|
207
|
+
* detachable execution and the guest does not advertise
|
|
208
|
+
* {@link EXECUTION_ATTACH_FEATURE}.
|
|
209
|
+
*
|
|
210
|
+
* Refused rather than downgraded: a caller that asked for detach is about
|
|
211
|
+
* to rely on being able to come back for the output, and running the
|
|
212
|
+
* command anyway would keep nothing and tell nobody.
|
|
213
|
+
*/
|
|
214
|
+
export declare class KubernetesExecutionAttachUnsupportedError extends Error {
|
|
215
|
+
readonly feature: string;
|
|
216
|
+
readonly name = "KubernetesExecutionAttachUnsupportedError";
|
|
217
|
+
constructor(feature: string, message: string);
|
|
218
|
+
}
|
|
219
|
+
/** Why an `attach-execution` could not be served. */
|
|
220
|
+
export type KubernetesAttachRefusal = 'unknown_execution' | 'output_not_retained' | 'invalid_offset' | 'invalid_execution_id' | 'agent_retiring' | 'unknown';
|
|
221
|
+
/**
|
|
222
|
+
* Thrown when the guest ANSWERED an attach and refused it — the execution
|
|
223
|
+
* is past its retention, ran in a pod that has since been replaced, never
|
|
224
|
+
* asked for its output to be kept, or the offset names bytes it does not
|
|
225
|
+
* have.
|
|
226
|
+
*
|
|
227
|
+
* Distinct from a transport failure on purpose: a refusal will not become a
|
|
228
|
+
* success by being retried, so the reattach loop stops on it instead of
|
|
229
|
+
* spending its whole window re-asking a question already answered.
|
|
230
|
+
*/
|
|
231
|
+
export declare class KubernetesExecutionNotAttachableError extends Error {
|
|
232
|
+
readonly executionId: string;
|
|
233
|
+
readonly reason: KubernetesAttachRefusal;
|
|
234
|
+
readonly name = "KubernetesExecutionNotAttachableError";
|
|
235
|
+
/**
|
|
236
|
+
* The state the guest reported for this execution, when it reported
|
|
237
|
+
* one. `'reserved'` is the one that changes what a caller should do:
|
|
238
|
+
* the command was never started, so nothing is running.
|
|
239
|
+
*/
|
|
240
|
+
readonly executionState: string | undefined;
|
|
241
|
+
constructor(executionId: string, reason: KubernetesAttachRefusal, message: string, options?: {
|
|
242
|
+
cause?: unknown;
|
|
243
|
+
state?: string;
|
|
244
|
+
});
|
|
245
|
+
}
|
|
246
|
+
/**
|
|
247
|
+
* Thrown when a detached `exec()` gave up OBSERVING a command that is, as
|
|
248
|
+
* far as this host knows, still the guest's to run.
|
|
249
|
+
*
|
|
250
|
+
* The two fields are what makes it recoverable rather than merely a
|
|
251
|
+
* failure: `executionId` names the command to a second host process, and
|
|
252
|
+
* `outputOffset` is the byte the next `attachExecution` should resume from
|
|
253
|
+
* so nothing is read twice and no gap is invented.
|
|
254
|
+
*
|
|
255
|
+
* Nothing on this path cancels to reconcile. That is the whole point of
|
|
256
|
+
* the feature: a reset connection used to cost the workspace its pod, and
|
|
257
|
+
* a command the host has stopped watching is not a command that has to
|
|
258
|
+
* die.
|
|
259
|
+
*/
|
|
260
|
+
export declare class KubernetesExecutionDetachedError extends Error {
|
|
261
|
+
readonly executionId: string;
|
|
262
|
+
readonly outputOffset: number;
|
|
263
|
+
readonly name = "KubernetesExecutionDetachedError";
|
|
264
|
+
constructor(executionId: string, outputOffset: number, message: string, options?: {
|
|
265
|
+
cause?: unknown;
|
|
266
|
+
});
|
|
267
|
+
}
|
|
268
|
+
/**
|
|
269
|
+
* What a caller passes to read an execution it did not necessarily start.
|
|
270
|
+
*
|
|
271
|
+
* `signal` here is an OBSERVATION signal, not the SDK's command signal:
|
|
272
|
+
* aborting it stops reading and leaves the command running. That is the
|
|
273
|
+
* opposite of `SandboxExecOptions.signal`, and it is why this type does not
|
|
274
|
+
* extend it.
|
|
275
|
+
*/
|
|
276
|
+
export interface KubernetesAttachExecutionOptions {
|
|
277
|
+
/** Byte offset to resume from. Default 0 — the whole retained log. */
|
|
278
|
+
readonly fromOffset?: number;
|
|
279
|
+
readonly onOutput?: SandboxExecOptions['onOutput'];
|
|
280
|
+
/** Stops OBSERVING. Never cancels; see {@link KubernetesAgentTransport.cancelExecution}. */
|
|
281
|
+
readonly signal?: AbortSignal;
|
|
282
|
+
/**
|
|
283
|
+
* Called when the guest reports that bytes the caller asked for had
|
|
284
|
+
* already been evicted from the retained log. The result's truncation
|
|
285
|
+
* flags say the same thing; this says how much.
|
|
286
|
+
*/
|
|
287
|
+
readonly onGap?: (gap: {
|
|
288
|
+
readonly executionId: string;
|
|
289
|
+
readonly fromOffset: number;
|
|
290
|
+
readonly droppedBytes: number;
|
|
291
|
+
}) => void;
|
|
292
|
+
}
|
|
293
|
+
/**
|
|
294
|
+
* An `exec()` whose observation can be lost and taken up again — on this
|
|
295
|
+
* handle or in another host process — instead of costing the command its
|
|
296
|
+
* life.
|
|
297
|
+
*
|
|
298
|
+
* Deliberately NOT on the SDK's `SandboxExecOptions`: every other backend
|
|
299
|
+
* would then have to answer for a field it cannot honour, and the SDK's
|
|
300
|
+
* exec contract stays exactly what it was. This is a Kubernetes workspace
|
|
301
|
+
* surface, layered over the shared options type rather than widening it.
|
|
302
|
+
*/
|
|
303
|
+
export interface KubernetesDetachedExecOptions extends SandboxExecOptions, Pick<KubernetesAttachExecutionOptions, 'onGap'> {
|
|
304
|
+
/**
|
|
305
|
+
* The id this command is known by, to this process and to any other.
|
|
306
|
+
* Minted here when absent. Reserving an id the guest still holds does
|
|
307
|
+
* NOT start a second command: the call attaches to the one that exists,
|
|
308
|
+
* which is what makes a retried start idempotent for as long as the
|
|
309
|
+
* record lives.
|
|
310
|
+
*/
|
|
311
|
+
readonly executionId?: string;
|
|
312
|
+
/**
|
|
313
|
+
* Ask the guest to retain this command's output so the observation can
|
|
314
|
+
* be resumed. Setting `executionId` implies it; the flag is what a
|
|
315
|
+
* caller that does not care about the id passes.
|
|
316
|
+
*/
|
|
317
|
+
readonly detach?: boolean;
|
|
318
|
+
/**
|
|
319
|
+
* Stop observing and leave the command running — for a host that is
|
|
320
|
+
* shutting down. Rejects with {@link KubernetesExecutionDetachedError},
|
|
321
|
+
* which names the id and the offset to resume from.
|
|
322
|
+
*
|
|
323
|
+
* The opposite of `signal`, which keeps the SDK contract and terminates.
|
|
324
|
+
*/
|
|
325
|
+
readonly detachSignal?: AbortSignal;
|
|
326
|
+
/**
|
|
327
|
+
* How long a lost connection is retried before the call gives up and
|
|
328
|
+
* reports itself detached. Default 30s.
|
|
329
|
+
*/
|
|
330
|
+
readonly reattachWindowMs?: number;
|
|
331
|
+
}
|
|
332
|
+
/**
|
|
333
|
+
* Thrown before anything is started, when the caller asked for a session and
|
|
334
|
+
* the guest does not advertise {@link SESSIONS_FEATURE}.
|
|
335
|
+
*
|
|
336
|
+
* Refused, never downgraded to a connection-bound terminal. A caller that
|
|
337
|
+
* asked for a session is about to rely on coming back to it after its own
|
|
338
|
+
* process has been replaced; handing it one that dies with the socket would
|
|
339
|
+
* look like it worked until the one moment it was needed.
|
|
340
|
+
*/
|
|
341
|
+
export declare class KubernetesSessionsUnsupportedError extends Error {
|
|
342
|
+
readonly feature: string;
|
|
343
|
+
readonly name = "KubernetesSessionsUnsupportedError";
|
|
344
|
+
constructor(feature: string, message: string);
|
|
345
|
+
}
|
|
346
|
+
/** Why the guest refused a session request. */
|
|
347
|
+
export type KubernetesSessionRefusal = 'unknown_session' | 'invalid_session_id' | 'invalid_offset' | 'session_exists' | 'session_capacity' | 'missing_command' | 'spawn_failed' | 'agent_retiring' | 'unknown';
|
|
348
|
+
/**
|
|
349
|
+
* Thrown when the guest ANSWERED and refused: the session is past its
|
|
350
|
+
* retention, ran in a pod that has since been replaced, the id is already
|
|
351
|
+
* taken, or the offset names bytes it does not have.
|
|
352
|
+
*
|
|
353
|
+
* Distinct from a transport failure for the same reason
|
|
354
|
+
* {@link KubernetesExecutionNotAttachableError} is: a refusal does not
|
|
355
|
+
* become a success by being retried.
|
|
356
|
+
*/
|
|
357
|
+
export declare class KubernetesSessionRefusedError extends Error {
|
|
358
|
+
readonly sessionId: string;
|
|
359
|
+
readonly reason: KubernetesSessionRefusal;
|
|
360
|
+
readonly name = "KubernetesSessionRefusedError";
|
|
361
|
+
constructor(sessionId: string, reason: KubernetesSessionRefusal, message: string, options?: {
|
|
362
|
+
cause?: unknown;
|
|
363
|
+
});
|
|
364
|
+
}
|
|
365
|
+
/** One row of {@link KubernetesAgentTransport.listSessions}. */
|
|
366
|
+
export interface KubernetesSessionSummary {
|
|
367
|
+
readonly sessionId: string;
|
|
368
|
+
readonly kind: SessionKind;
|
|
369
|
+
/** The program, as it was asked for. Never the environment it was given. */
|
|
370
|
+
readonly command: string;
|
|
371
|
+
readonly args: readonly string[];
|
|
372
|
+
readonly startedAt: number;
|
|
373
|
+
readonly lastInputAt?: number;
|
|
374
|
+
readonly lastOutputAt?: number;
|
|
375
|
+
/** Pass as `fromOffset` to read everything this session has printed since. */
|
|
376
|
+
readonly nextOffset: number;
|
|
377
|
+
/** Bytes the ring has evicted over this session's life. */
|
|
378
|
+
readonly droppedBytes: number;
|
|
379
|
+
readonly state: SessionState;
|
|
380
|
+
/** Whether a host process is attached to it right now. */
|
|
381
|
+
readonly attached: boolean;
|
|
382
|
+
readonly exitCode?: number;
|
|
383
|
+
readonly signal?: number;
|
|
384
|
+
}
|
|
385
|
+
/**
|
|
386
|
+
* One read of a session's retained output, in the SDK's
|
|
387
|
+
* `BackgroundJobOutput` shape — deliberately, because it answers the same
|
|
388
|
+
* question for the same kind of consumer and a second vocabulary for
|
|
389
|
+
* "here is the next chunk and here is what you missed" helps nobody.
|
|
390
|
+
*/
|
|
391
|
+
export interface KubernetesSessionOutput {
|
|
392
|
+
readonly chunk: string;
|
|
393
|
+
readonly nextOffset: number;
|
|
394
|
+
readonly droppedBytes: number;
|
|
395
|
+
readonly status: BackgroundJobStatus;
|
|
396
|
+
readonly exitCode?: number;
|
|
397
|
+
}
|
|
398
|
+
/**
|
|
399
|
+
* A terminal on a workspace, with the three things only a SESSION's reader
|
|
400
|
+
* needs. On a connection-bound terminal the two optional members are absent,
|
|
401
|
+
* which is the honest answer: there is no session to name and nothing to
|
|
402
|
+
* detach from.
|
|
403
|
+
*/
|
|
404
|
+
export interface KubernetesWorkspaceTerminal extends TerminalSession {
|
|
405
|
+
/** Present exactly when this terminal belongs to a guest session. */
|
|
406
|
+
readonly sessionId?: string;
|
|
407
|
+
/** One past the newest retained byte delivered so far. */
|
|
408
|
+
nextOffset?(): number | undefined;
|
|
409
|
+
/**
|
|
410
|
+
* Stop reading and leave the program running — the opposite of
|
|
411
|
+
* {@link TerminalSession.kill}. `exited` then rejects with
|
|
412
|
+
* `AgentSessionDetachedError`, because a resolved `exited` would claim
|
|
413
|
+
* an exit that did not happen.
|
|
414
|
+
*/
|
|
415
|
+
detach?(): void;
|
|
416
|
+
}
|
|
417
|
+
/**
|
|
418
|
+
* What an ATTACH hands back: the same terminal, with the three session
|
|
419
|
+
* members present rather than optional. `openTerminal` returns the looser
|
|
420
|
+
* type because it serves both shapes and a connection-bound terminal
|
|
421
|
+
* genuinely has no session to name.
|
|
422
|
+
*/
|
|
423
|
+
export interface KubernetesSessionTerminal extends KubernetesWorkspaceTerminal {
|
|
424
|
+
readonly sessionId: string;
|
|
425
|
+
nextOffset(): number | undefined;
|
|
426
|
+
detach(): void;
|
|
427
|
+
}
|
|
428
|
+
/** `openTerminal` on a workspace, widened by the two session fields. */
|
|
429
|
+
export interface KubernetesOpenTerminalOptions extends OpenTerminalOptions {
|
|
430
|
+
/**
|
|
431
|
+
* Name this terminal so a later host process can find it again.
|
|
432
|
+
* Requires {@link persistent}; on its own it names nothing.
|
|
433
|
+
*/
|
|
434
|
+
readonly sessionId?: string;
|
|
435
|
+
/**
|
|
436
|
+
* Hand the PTY to the guest's session registry rather than to this
|
|
437
|
+
* connection. Losing the connection then DETACHES — no signal is sent,
|
|
438
|
+
* and the program ends when it exits, on `killSession`, or when the pod
|
|
439
|
+
* stops.
|
|
440
|
+
*/
|
|
441
|
+
readonly persistent?: boolean;
|
|
442
|
+
}
|
|
443
|
+
/** Rejoin a terminal session that is already running. */
|
|
444
|
+
export interface KubernetesAttachTerminalOptions {
|
|
445
|
+
/** Byte offset to replay from. Default 0 — everything the ring still holds. */
|
|
446
|
+
readonly fromOffset?: number;
|
|
447
|
+
/** Resize the PTY on attach, for a reader whose window is a different shape. */
|
|
448
|
+
readonly size?: {
|
|
449
|
+
readonly cols: number;
|
|
450
|
+
readonly rows: number;
|
|
451
|
+
};
|
|
452
|
+
}
|
|
453
|
+
/** Start a program with no terminal, which only a kill or the pod ends. */
|
|
454
|
+
export interface KubernetesStartDetachedOptions {
|
|
455
|
+
readonly sessionId: string;
|
|
456
|
+
readonly command: string;
|
|
457
|
+
readonly args?: readonly string[];
|
|
458
|
+
readonly cwd?: string;
|
|
459
|
+
readonly env?: Record<string, string>;
|
|
460
|
+
}
|
|
461
|
+
/** Read a session's retained output without attaching to it. */
|
|
462
|
+
export interface KubernetesReadSessionOptions {
|
|
463
|
+
readonly fromOffset?: number;
|
|
72
464
|
}
|
|
73
465
|
/**
|
|
74
466
|
* The kubernetes backend's dialable transport: a `tcp` handle plus a
|
|
@@ -78,17 +470,322 @@ export interface KubernetesTransportOptions extends VsockTransportOptions {
|
|
|
78
470
|
* remote backend gets, while every network operation is delegated to
|
|
79
471
|
* {@link VsockAgentTransport} for the actual dial/frame/token work.
|
|
80
472
|
*/
|
|
473
|
+
/**
|
|
474
|
+
* Thrown when a quiesce was asked of a guest that cannot perform one:
|
|
475
|
+
* either its `healthz` does not advertise {@link QUIESCE_FEATURE}, or it
|
|
476
|
+
* answered `unknown_op: quiesce`.
|
|
477
|
+
*
|
|
478
|
+
* Refused rather than treated as "nothing was running". A caller asks for a
|
|
479
|
+
* quiesce because it is about to read the disk and needs it still; an image
|
|
480
|
+
* that cannot stop its processes has to say so, not resolve with an empty
|
|
481
|
+
* list that reads exactly like a guest which had nothing to stop.
|
|
482
|
+
*/
|
|
483
|
+
export declare class KubernetesQuiesceUnsupportedError extends Error {
|
|
484
|
+
readonly feature: string;
|
|
485
|
+
readonly name = "KubernetesQuiesceUnsupportedError";
|
|
486
|
+
constructor(feature: string, message: string);
|
|
487
|
+
}
|
|
488
|
+
/**
|
|
489
|
+
* Thrown when nothing can promise the guest is quiet.
|
|
490
|
+
*
|
|
491
|
+
* Usually because the guest ANSWERED and said so — a process survived
|
|
492
|
+
* `SIGKILL`, the scan could not be performed, or the op ran out of its own
|
|
493
|
+
* deadline — and then the guest's own message names the pid. The workspace
|
|
494
|
+
* handle raises the same class for the one case that never reaches a guest:
|
|
495
|
+
* a `suspend({ quiesce: true })` arriving while a suspend WITHOUT a quiesce
|
|
496
|
+
* is already in flight, whose patch has gone over processes nobody stopped
|
|
497
|
+
* (`reason: 'suspend_already_in_flight'`). Both mean the one thing a caller
|
|
498
|
+
* has to act on: do not trust a capture taken now.
|
|
499
|
+
*
|
|
500
|
+
* Its own class for the same reason {@link KubernetesSessionRefusedError}
|
|
501
|
+
* is: this is an answer, not a transport failure, and retrying it is a
|
|
502
|
+
* decision the caller makes with the pid in hand rather than one a wrapper
|
|
503
|
+
* makes on its behalf.
|
|
504
|
+
*/
|
|
505
|
+
export declare class KubernetesQuiesceUnconfirmedError extends Error {
|
|
506
|
+
readonly reason: string;
|
|
507
|
+
readonly name = "KubernetesQuiesceUnconfirmedError";
|
|
508
|
+
constructor(reason: string, message: string);
|
|
509
|
+
}
|
|
510
|
+
/** What the guest stopped, and how widely it was allowed to look. */
|
|
511
|
+
export interface KubernetesQuiesceReport {
|
|
512
|
+
/** Every process signalled, with the last signal it was actually sent. */
|
|
513
|
+
readonly stopped: readonly QuiescedProcess[];
|
|
514
|
+
/** See {@link QuiesceScope}. `owned-sessions` is the narrowed one. */
|
|
515
|
+
readonly scope: QuiesceScope;
|
|
516
|
+
/** The per-round SIGTERM window the guest used, after its own clamp. */
|
|
517
|
+
readonly graceMs: number;
|
|
518
|
+
/** Scan-and-signal passes. `0` means nothing was running. */
|
|
519
|
+
readonly rounds: number;
|
|
520
|
+
}
|
|
521
|
+
/**
|
|
522
|
+
* Thrown when a flush was asked of a guest that cannot perform one: either
|
|
523
|
+
* its `healthz` does not advertise {@link FLUSH_FEATURE}, or it answered
|
|
524
|
+
* `unknown_op: flush`.
|
|
525
|
+
*
|
|
526
|
+
* Refused rather than treated as "the disk is already flushed", for the
|
|
527
|
+
* same reason {@link KubernetesQuiesceUnsupportedError} is not treated as
|
|
528
|
+
* "nothing was running". The one caller that does NOT pass this on is
|
|
529
|
+
* `suspend()`, which goes ahead and tells the host through
|
|
530
|
+
* `onFlushUnsupported`: an image built before this op is a deployment that
|
|
531
|
+
* has to be able to suspend its workspaces, not one that has to be stopped.
|
|
532
|
+
*/
|
|
533
|
+
export declare class KubernetesFlushUnsupportedError extends Error {
|
|
534
|
+
readonly feature: string;
|
|
535
|
+
readonly name = "KubernetesFlushUnsupportedError";
|
|
536
|
+
constructor(feature: string, message: string);
|
|
537
|
+
}
|
|
538
|
+
/**
|
|
539
|
+
* Thrown when nothing can promise the workspace's writes are on the device.
|
|
540
|
+
*
|
|
541
|
+
* The guest ANSWERED and said so — the `syncfs` failed, or ran past its own
|
|
542
|
+
* timeout — and its message says which. Its own class for the reason every
|
|
543
|
+
* other answered refusal here has one: this is a fact about the disk, not a
|
|
544
|
+
* transport failure, and a caller holding it decides whether to retry,
|
|
545
|
+
* suspend anyway, or leave the workspace running.
|
|
546
|
+
*/
|
|
547
|
+
export declare class KubernetesFlushUnconfirmedError extends Error {
|
|
548
|
+
readonly reason: string;
|
|
549
|
+
readonly name = "KubernetesFlushUnconfirmedError";
|
|
550
|
+
constructor(reason: string, message: string);
|
|
551
|
+
}
|
|
552
|
+
/**
|
|
553
|
+
* Carried to {@link KubernetesWorkspaceOptions.onFlushUnreachable} when a
|
|
554
|
+
* `suspend()` could not ASK for a flush at all — the dial failed, the
|
|
555
|
+
* connection timed out, the token was refused, or the agent has fenced
|
|
556
|
+
* itself — and the suspend went ahead without one.
|
|
557
|
+
*
|
|
558
|
+
* Never thrown at a caller. The difference between this and
|
|
559
|
+
* {@link KubernetesFlushUnconfirmedError} is the difference between a guest
|
|
560
|
+
* that could not be reached and a guest that answered: a guest that
|
|
561
|
+
* answered is alive, and stopping the suspend gives its caller something to
|
|
562
|
+
* do about it, while a guest nothing can reach will not become flushable by
|
|
563
|
+
* leaving its pod running — and refusing to suspend over it would take away
|
|
564
|
+
* the one verb an operator reaches for when a workspace is wedged, the verb
|
|
565
|
+
* {@link KubernetesAgentRetiringError} itself names as the way out.
|
|
566
|
+
*
|
|
567
|
+
* So the suspend proceeds, the host is told, and the message says what the
|
|
568
|
+
* disk is resting on instead: whatever the guest kernel had already written
|
|
569
|
+
* back, plus the pod's own `preStop` hook and the agent's `SIGTERM` handler
|
|
570
|
+
* if either of them still runs.
|
|
571
|
+
*/
|
|
572
|
+
export declare class KubernetesFlushUnreachableError extends Error {
|
|
573
|
+
readonly name = "KubernetesFlushUnreachableError";
|
|
574
|
+
}
|
|
575
|
+
/** What a flush did, as the guest measured it. */
|
|
576
|
+
export interface KubernetesFlushReport {
|
|
577
|
+
/** How long the `syncfs` took inside the guest. */
|
|
578
|
+
readonly durationMs: number;
|
|
579
|
+
/** The mount the guest flushed — its workspace root. */
|
|
580
|
+
readonly workspace: string;
|
|
581
|
+
}
|
|
582
|
+
/**
|
|
583
|
+
* The refusal a guest that cannot flush earns, built in one place.
|
|
584
|
+
*
|
|
585
|
+
* Exported because the WORKSPACE has to be able to hand this exact error to
|
|
586
|
+
* `onFlushUnsupported` on the one path that does not throw it — a
|
|
587
|
+
* `suspend()` against an older image — and a second copy of the message
|
|
588
|
+
* would be a second thing to keep true.
|
|
589
|
+
*/
|
|
590
|
+
export declare function flushUnsupportedError(): KubernetesFlushUnsupportedError;
|
|
81
591
|
export declare class KubernetesAgentTransport {
|
|
82
|
-
|
|
592
|
+
/**
|
|
593
|
+
* Mutable: the handle follows a replaced pod — see {@link rebind}.
|
|
594
|
+
*
|
|
595
|
+
* Under `'pod-ip'` both the address and the token move, because the
|
|
596
|
+
* address is a literal that died with its pod. Under the default
|
|
597
|
+
* `'service'` mode the host is a Service FQDN that outlives the pod and
|
|
598
|
+
* resolves to the replacement on its own, so what moves is the TOKEN
|
|
599
|
+
* alone — which is the entire reason a `service` handle needed a rebind
|
|
600
|
+
* at all: the dial keeps working and the guest refuses every call.
|
|
601
|
+
*/
|
|
602
|
+
private handle;
|
|
603
|
+
/** The re-read currently in flight, so concurrent refusals share one. */
|
|
604
|
+
private rebinding?;
|
|
605
|
+
/**
|
|
606
|
+
* How many times this transport has moved to a different pod. Captured
|
|
607
|
+
* before every attempt and compared after it fails, so a call refused by
|
|
608
|
+
* the pod it was bound to — arriving after ANOTHER call's rebind already
|
|
609
|
+
* installed the replacement — retries on the handle that has moved
|
|
610
|
+
* instead of re-reading to be told nothing changed.
|
|
611
|
+
*/
|
|
612
|
+
private rebindSeq;
|
|
83
613
|
private readonly transportOptions;
|
|
614
|
+
/** Bound to each wire's own pod by {@link wireOptions}. */
|
|
615
|
+
private readonly onGuestReply?;
|
|
84
616
|
private readonly onTiming?;
|
|
617
|
+
private readonly refreshHandle?;
|
|
85
618
|
/** Simple pass-through operations share one transport instance. */
|
|
86
|
-
private
|
|
619
|
+
private wire;
|
|
87
620
|
constructor(handle: KubernetesAgentHandle, options?: KubernetesTransportOptions);
|
|
88
|
-
/**
|
|
621
|
+
/**
|
|
622
|
+
* The shared transport's options for ONE wire, with the reply observer
|
|
623
|
+
* bound to the pod that wire is talking to.
|
|
624
|
+
*
|
|
625
|
+
* Every `VsockAgentTransport` this class builds goes through here — the
|
|
626
|
+
* first one, the one a rebind installs, and the per-attempt one `exec()`
|
|
627
|
+
* times — because a wire outlives the moment it was current: the pod it
|
|
628
|
+
* dials can answer a call after a rebind has already moved
|
|
629
|
+
* {@link handle} on, and the observer has to be told the pod that
|
|
630
|
+
* ANSWERED rather than the pod this transport now holds. Capturing the
|
|
631
|
+
* handle in the closure is what makes the two different values.
|
|
632
|
+
*/
|
|
633
|
+
private wireOptions;
|
|
634
|
+
/**
|
|
635
|
+
* The address this transport is dialing right now. Diagnostics only —
|
|
636
|
+
* host and port, and deliberately NOT the handle itself: this is read
|
|
637
|
+
* into a log line or an assertion, and the bind token has no business
|
|
638
|
+
* travelling with either.
|
|
639
|
+
*/
|
|
640
|
+
get address(): {
|
|
641
|
+
readonly host: string;
|
|
642
|
+
readonly port: number;
|
|
643
|
+
};
|
|
644
|
+
/**
|
|
645
|
+
* Run one operation, and give the handle exactly one chance to follow a
|
|
646
|
+
* pod that was replaced underneath it.
|
|
647
|
+
*
|
|
648
|
+
* TWO failures reach a rebind, and they are the only two, because they
|
|
649
|
+
* are the only two that prove the operation ran NOTHING in the guest:
|
|
650
|
+
*
|
|
651
|
+
* - **the dial failed.** No socket was ever established, so no byte was
|
|
652
|
+
* sent. This is the trigger the `'pod-ip'` mode was built on: the
|
|
653
|
+
* address is a literal that dies with its pod.
|
|
654
|
+
* - **the guest REFUSED the token** (`unauthorized`, in either of the
|
|
655
|
+
* shapes {@link isUnauthorizedRefusal} unifies). The agent checks the
|
|
656
|
+
* token before `dispatch`, so a refused request reached the guest and
|
|
657
|
+
* was thrown away unread. This is the trigger that matters under the
|
|
658
|
+
* DEFAULT `'service'` mode, where the Service FQDN outlives the pod
|
|
659
|
+
* and keeps resolving: the dial SUCCEEDS against the replacement and
|
|
660
|
+
* the refusal is the only thing that says the pod moved.
|
|
661
|
+
*
|
|
662
|
+
* Everything else is the guest's answer to a request it did receive, and
|
|
663
|
+
* nothing here may reinterpret it — not as a resolver problem, and not as
|
|
664
|
+
* a reason to repeat work the guest has already begun.
|
|
665
|
+
*
|
|
666
|
+
* From there both arms run the SAME routine, which is the whole of this
|
|
667
|
+
* change to it: {@link rebind} reads the pod once, and a DIFFERENT uid
|
|
668
|
+
* means the controller replaced it (a resume, an eviction, a node drain),
|
|
669
|
+
* so the handle takes the new address AND the new token together and the
|
|
670
|
+
* operation is retried once.
|
|
671
|
+
*
|
|
672
|
+
* One branch belongs to the dial arm alone, and is taken before the
|
|
673
|
+
* re-read: the handle's host is a NAME and the dial gave up at
|
|
674
|
+
* resolution, so this is a Service FQDN and this host has no resolver for
|
|
675
|
+
* it. Re-reading the pod would change nothing — the next dial would ask
|
|
676
|
+
* the same resolver the same question — so the error is replaced with one
|
|
677
|
+
* that names the FQDN and the configuration field that fixes it. It is
|
|
678
|
+
* never asked of a refusal, which came back over a connection that
|
|
679
|
+
* plainly worked, and the name check is not decoration either: a
|
|
680
|
+
* `pod-ip` handle must keep its one re-read however an unrelated error
|
|
681
|
+
* happens to be worded.
|
|
682
|
+
*
|
|
683
|
+
* Anything else — a re-read that finds the SAME pod, or one that fails —
|
|
684
|
+
* leaves the original error standing. A pod that is still there and still
|
|
685
|
+
* refusing is a guest problem, and replacing that error with a second,
|
|
686
|
+
* later one would hide it. The one exception is a re-read that comes back
|
|
687
|
+
* with a verdict about the OBJECT rather than a failed diagnosis
|
|
688
|
+
* ({@link KubernetesWorkspaceReplacedError}): the disk behind the name is
|
|
689
|
+
* not this handle's disk, which outranks whatever uncovered it.
|
|
690
|
+
*
|
|
691
|
+
* `signal` is the caller's own, and it decides one thing only: a call that
|
|
692
|
+
* has ALREADY been cancelled is not owed a re-read, because the retry it
|
|
693
|
+
* would buy would abort before it left. It is never handed to the re-read
|
|
694
|
+
* itself, which is shared and runs under a bound of its own — see
|
|
695
|
+
* {@link rebind}.
|
|
696
|
+
*
|
|
697
|
+
* `dials` is how `exec()` answers the first question at all. Its failure
|
|
698
|
+
* can arrive as a bare timeout from the execution controller's own bound,
|
|
699
|
+
* with the dial's error discarded rather than wrapped, so that path
|
|
700
|
+
* watches its dials instead of reading its error — see {@link DialWatch}.
|
|
701
|
+
* Every other operation hands back the dial's own error and passes none.
|
|
702
|
+
*
|
|
703
|
+
* A refusal that no rebind could fix leaves as {@link
|
|
704
|
+
* KubernetesAgentUnauthorizedError} whichever shape it arrived in — the
|
|
705
|
+
* unification the whole recovery hangs off, applied at the one point
|
|
706
|
+
* where "nothing else can be done about it" is known.
|
|
707
|
+
*/
|
|
708
|
+
private withRebind;
|
|
709
|
+
/**
|
|
710
|
+
* Whether the dial gave up at name resolution.
|
|
711
|
+
*
|
|
712
|
+
* Asked of the caller's error first, and of the watched dial's own error
|
|
713
|
+
* only when nothing ever connected — the case where the error the caller
|
|
714
|
+
* holds is a bound's timer rather than the failure that caused it. A
|
|
715
|
+
* watched attempt that DID connect is never consulted: its error is the
|
|
716
|
+
* guest's answer, however it happens to be worded.
|
|
717
|
+
*/
|
|
718
|
+
private failedToResolve;
|
|
719
|
+
/**
|
|
720
|
+
* The one connect failure waiting cannot cure — see the constructor.
|
|
721
|
+
* A method rather than a closure because the watched `exec()` dials wrap
|
|
722
|
+
* it, and a caller-visible answer must not depend on which wire asked.
|
|
723
|
+
*/
|
|
724
|
+
private dialFailureIsPermanent;
|
|
725
|
+
/**
|
|
726
|
+
* One re-read, shared by every call that needs one at the same moment.
|
|
727
|
+
* True when the handle now points at a DIFFERENT pod than the one the
|
|
728
|
+
* caller's attempt was dispatched on — whether this call's own re-read
|
|
729
|
+
* moved it or another call's already had.
|
|
730
|
+
*
|
|
731
|
+
* Single-flight, and that is not an optimisation. A pod replaced
|
|
732
|
+
* underneath a busy handle refuses EVERY call in flight at once, and one
|
|
733
|
+
* re-read per refused call would be a burst of Sandbox and pod GETs, each
|
|
734
|
+
* one racing the others to install a handle — with the last to finish
|
|
735
|
+
* winning, which is not necessarily the last to read. Sharing the promise
|
|
736
|
+
* makes the burst one round trip and one installation, in the order the
|
|
737
|
+
* API answered.
|
|
738
|
+
*
|
|
739
|
+
* The slot is cleared inside the shared run, before the promise settles,
|
|
740
|
+
* so a caller that awaits it and is refused AGAIN gets a fresh re-read
|
|
741
|
+
* rather than the answer to the previous question.
|
|
742
|
+
*
|
|
743
|
+
* Because it is shared it runs under {@link REBIND_READ_TIMEOUT_MS} and
|
|
744
|
+
* under NO caller's signal. A caller that aborts while the shared read is
|
|
745
|
+
* in flight would otherwise abort it for everyone, and every call the
|
|
746
|
+
* replacement refused would fail with its original refusal although the
|
|
747
|
+
* handle could have followed the pod. Its own bound is what the aborting
|
|
748
|
+
* caller was owed — nobody waits on a re-read for longer than that — and
|
|
749
|
+
* the read is two GETs nobody is billed for twice.
|
|
750
|
+
*/
|
|
751
|
+
private rebind;
|
|
752
|
+
private runRebind;
|
|
753
|
+
/** Whether the current handle's host goes through a resolver at all. */
|
|
754
|
+
private dialsAName;
|
|
755
|
+
/** The DNS-shaped failure, in words that name the way out of it. */
|
|
756
|
+
private unresolvable;
|
|
757
|
+
/**
|
|
758
|
+
* Readiness probe — never requires a token; see `protocol.ts`.
|
|
759
|
+
*
|
|
760
|
+
* Deliberately NOT wrapped in {@link withRebind}: `healthz` answers a
|
|
761
|
+
* failed dial with `false` rather than by throwing, so there is no error
|
|
762
|
+
* to classify and nothing for a re-read to be triggered by. A caller that
|
|
763
|
+
* wants the reason asks for it by making a real call.
|
|
764
|
+
*/
|
|
89
765
|
healthz(signal?: AbortSignal): Promise<boolean>;
|
|
90
|
-
/**
|
|
766
|
+
/**
|
|
767
|
+
* Poll until `healthz` succeeds or the timeout elapses. Unwrapped for the
|
|
768
|
+
* same reason, and for one more: it already owns a retry loop, so a
|
|
769
|
+
* connect failure here is not a single failed dial but a whole budget of
|
|
770
|
+
* them.
|
|
771
|
+
*/
|
|
91
772
|
waitForReady(timeoutMs: number, pollIntervalMs: number, signal?: AbortSignal): Promise<void>;
|
|
773
|
+
/**
|
|
774
|
+
* Ask the agent how it is, and keep the two "not ok" answers apart —
|
|
775
|
+
* see {@link KubernetesAgentHealth}.
|
|
776
|
+
*
|
|
777
|
+
* Deliberately NOT wrapped in {@link withRebind}: this is a diagnostic
|
|
778
|
+
* about the pod this handle is bound to RIGHT NOW, and a rebind would
|
|
779
|
+
* silently answer it about a different pod. A caller that wants to know
|
|
780
|
+
* whether the agent it was talking to has fenced itself would then be
|
|
781
|
+
* told about the replacement, which is a different question with a
|
|
782
|
+
* different answer.
|
|
783
|
+
*
|
|
784
|
+
* It throws whatever the dial or the read threw. An agent that cannot be
|
|
785
|
+
* reached has no health to report, and inventing one here would turn
|
|
786
|
+
* "unreachable" into "fine".
|
|
787
|
+
*/
|
|
788
|
+
agentHealth(signal?: AbortSignal): Promise<KubernetesAgentHealth>;
|
|
92
789
|
/**
|
|
93
790
|
* The raw `reserve-execution` primitive, exposed directly (rather than
|
|
94
791
|
* only reachable as a side effect of `exec()`) so the reservation
|
|
@@ -103,9 +800,166 @@ export declare class KubernetesAgentTransport {
|
|
|
103
800
|
* driving a whole cancelled `exec()` to reach it.
|
|
104
801
|
*/
|
|
105
802
|
cancel(executionId: string, signal?: AbortSignal): Promise<unknown>;
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
803
|
+
/**
|
|
804
|
+
* Delegated, `signal` included: a body larger than one pre-auth frame
|
|
805
|
+
* is written as a sequence of parts by {@link VsockAgentTransport}
|
|
806
|
+
* itself, and a cancelled sequence has to be able to stop mid-way and
|
|
807
|
+
* take its temp file with it.
|
|
808
|
+
*
|
|
809
|
+
* One transport instance, deliberately — {@link wire} is shared by
|
|
810
|
+
* every simple pass-through op — so the one `healthz` probe that asks
|
|
811
|
+
* the guest whether it can take parts is asked once for this sandbox,
|
|
812
|
+
* not once per large write.
|
|
813
|
+
*
|
|
814
|
+
* Wrapped in {@link withRebind} like every other guest-dialling op,
|
|
815
|
+
* the multi-part route included, and for both of its triggers: a retry
|
|
816
|
+
* starts a fresh sequence under a new temp name
|
|
817
|
+
* (`writeFilePartTempPath` mints a UUID per call), so a sequence
|
|
818
|
+
* abandoned mid-way on the replaced pod — because the dial failed, or
|
|
819
|
+
* because the replacement refused this handle's token before reading a
|
|
820
|
+
* byte — cannot collide with the retry's offsets and never touched the
|
|
821
|
+
* target. At worst it leaves one orphan temp file behind on the
|
|
822
|
+
* workspace volume.
|
|
823
|
+
*/
|
|
824
|
+
writeFile(path: string, content: Buffer, signal?: AbortSignal): Promise<void>;
|
|
825
|
+
readFile(path: string, options?: SandboxReadFileOptions): Promise<Buffer>;
|
|
826
|
+
/**
|
|
827
|
+
* Delegated, and rebound exactly once — but only around the FIRST
|
|
828
|
+
* chunk.
|
|
829
|
+
*
|
|
830
|
+
* That is the whole of what {@link withRebind} can honestly cover here.
|
|
831
|
+
* Its retry is safe because nothing reached the guest, and once a chunk
|
|
832
|
+
* has been yielded that is no longer true: re-dialing a replaced pod
|
|
833
|
+
* mid-stream would restart the file from its beginning, and the
|
|
834
|
+
* consumer — which has already taken the bytes and cannot give them
|
|
835
|
+
* back — would silently concatenate a duplicate prefix. So the first
|
|
836
|
+
* pull carries the dial, the rebind and the retry; everything after it
|
|
837
|
+
* fails as itself.
|
|
838
|
+
*/
|
|
839
|
+
readFileStream(path: string, options?: SandboxReadFileOptions): AsyncGenerator<Buffer, void, undefined>;
|
|
840
|
+
/**
|
|
841
|
+
* A guest PTY. Without `sessionId`/`persistent` this is exactly the
|
|
842
|
+
* terminal it has always been, down to the wire request.
|
|
843
|
+
*
|
|
844
|
+
* With them the PTY belongs to the guest's session registry: losing this
|
|
845
|
+
* connection detaches rather than killing, a later process rejoins it
|
|
846
|
+
* with {@link attachSession}, and the capability is verified against the
|
|
847
|
+
* guest's `healthz` features BEFORE the shell is started — never
|
|
848
|
+
* downgraded to a connection-bound terminal, which would look like it
|
|
849
|
+
* worked until the rollout it exists for.
|
|
850
|
+
*/
|
|
851
|
+
openTerminal(options: KubernetesOpenTerminalOptions): Promise<KubernetesWorkspaceTerminal>;
|
|
852
|
+
/**
|
|
853
|
+
* One open of a session stream, with the guest's refusal mapped to
|
|
854
|
+
* {@link KubernetesSessionRefusedError} — see {@link sessionStreamFailure}.
|
|
855
|
+
*/
|
|
856
|
+
private sessionStream;
|
|
857
|
+
/**
|
|
858
|
+
* Rejoin a terminal session, replaying from `fromOffset` and then
|
|
859
|
+
* following it live.
|
|
860
|
+
*
|
|
861
|
+
* The guest allows one attachment per session and ends the previous one
|
|
862
|
+
* by name, so two host processes cannot interleave keystrokes into one
|
|
863
|
+
* shell. Losing this connection detaches; ending the program is
|
|
864
|
+
* {@link killSession} and nothing else.
|
|
865
|
+
*/
|
|
866
|
+
attachSession(sessionId: string, options?: KubernetesAttachTerminalOptions): Promise<KubernetesSessionTerminal>;
|
|
867
|
+
/**
|
|
868
|
+
* Start a program with no terminal at all, in its own kernel session,
|
|
869
|
+
* with stdin closed and both output streams going into the guest's
|
|
870
|
+
* retained log.
|
|
871
|
+
*
|
|
872
|
+
* It is not the SDK's `spawnDetached` and deliberately does not pretend
|
|
873
|
+
* to be: that one hands back a host `ChildProcess`, which cannot cross a
|
|
874
|
+
* process boundary. This returns a NAME, and the name is what a
|
|
875
|
+
* redeployed host comes back with.
|
|
876
|
+
*/
|
|
877
|
+
startDetached(options: KubernetesStartDetachedOptions, signal?: AbortSignal): Promise<KubernetesSessionSummary>;
|
|
878
|
+
/** Every session this pod's agent is holding, running and recently exited. */
|
|
879
|
+
listSessions(signal?: AbortSignal): Promise<readonly KubernetesSessionSummary[]>;
|
|
880
|
+
/**
|
|
881
|
+
* End one session and everything still in it — the shell, its
|
|
882
|
+
* backgrounded jobs, and the program a detached session started.
|
|
883
|
+
*
|
|
884
|
+
* Idempotent: a session that has already exited answers with what it
|
|
885
|
+
* exited with. The reply carries the session's state, so a program that
|
|
886
|
+
* ignored a `SIGTERM` and outlived the guest's confirm window is
|
|
887
|
+
* reported still running rather than reported dead.
|
|
888
|
+
*/
|
|
889
|
+
killSession(sessionId: string, options?: {
|
|
890
|
+
readonly signal?: string;
|
|
891
|
+
readonly abort?: AbortSignal;
|
|
892
|
+
}): Promise<KubernetesSessionSummary>;
|
|
893
|
+
/**
|
|
894
|
+
* Read what a session has printed since `fromOffset`, without attaching
|
|
895
|
+
* to it and without signalling anything.
|
|
896
|
+
*
|
|
897
|
+
* One request, one answer, in the SDK's `BackgroundJobOutput` shape: the
|
|
898
|
+
* chunk, the offset to come back with, the bytes the ring dropped before
|
|
899
|
+
* it, and the program's status. A caller polling in a loop can neither
|
|
900
|
+
* re-read nor skip, because the offset is the guest's own.
|
|
901
|
+
*/
|
|
902
|
+
readSession(sessionId: string, options?: KubernetesReadSessionOptions, signal?: AbortSignal): Promise<KubernetesSessionOutput>;
|
|
903
|
+
/**
|
|
904
|
+
* Stop every process this pod's guest is running, and keep the agent.
|
|
905
|
+
*
|
|
906
|
+
* The point is the pair. Stopping the POD stops its processes too, but
|
|
907
|
+
* leaves nothing to read the disk through, so a host that wants a capture
|
|
908
|
+
* it can trust has to wake the workspace again and check. After this the
|
|
909
|
+
* guest is quiet and still serving: `exec`, `readFile` and `writeFile`
|
|
910
|
+
* all work, and what they see is a filesystem nobody is writing to.
|
|
911
|
+
*
|
|
912
|
+
* Open terminals and running commands end as a side effect and report it
|
|
913
|
+
* through their own exit and result paths — a terminal receives its exit,
|
|
914
|
+
* an `exec` resolves with a signal in its result. Rejecting with
|
|
915
|
+
* {@link KubernetesQuiesceUnconfirmedError} is the honest failure: a
|
|
916
|
+
* process would not stop, and the reply names its pid.
|
|
917
|
+
*/
|
|
918
|
+
quiesce(options?: {
|
|
919
|
+
readonly graceMs?: number;
|
|
920
|
+
}, signal?: AbortSignal): Promise<KubernetesQuiesceReport>;
|
|
921
|
+
/**
|
|
922
|
+
* Put everything this workspace has written onto its device.
|
|
923
|
+
*
|
|
924
|
+
* The disk a workspace keeps is whatever the guest kernel happened to
|
|
925
|
+
* write back. Nothing in this backend ever asked for more than that: a
|
|
926
|
+
* `suspend()` patched the pod away and waited for it to stop, and a
|
|
927
|
+
* stopped pod means only that nothing is writing any more — not that
|
|
928
|
+
* what was written arrived. This is the call that closes the difference,
|
|
929
|
+
* and it is `syncfs(2)` over the whole workspace mount, so it covers
|
|
930
|
+
* what a COMMAND wrote as well as what `writeFile` did (`writeFile`
|
|
931
|
+
* fsyncs its own bytes before it answers; a compiler's output is nobody's
|
|
932
|
+
* to fsync).
|
|
933
|
+
*
|
|
934
|
+
* Rejecting with {@link KubernetesFlushUnconfirmedError} is the honest
|
|
935
|
+
* failure, exactly as an unconfirmed quiesce is: the caller is usually
|
|
936
|
+
* about to take the pod away, and "the flush did not run" has to be
|
|
937
|
+
* distinguishable from "the flush ran".
|
|
938
|
+
*/
|
|
939
|
+
flush(options?: {
|
|
940
|
+
readonly timeoutMs?: number;
|
|
941
|
+
}, signal?: AbortSignal): Promise<KubernetesFlushReport>;
|
|
942
|
+
/**
|
|
943
|
+
* Whether this guest can flush at all — asked, rather than assumed,
|
|
944
|
+
* because a `suspend()` against an older image keeps today's behaviour
|
|
945
|
+
* and reports the gap instead of refusing to suspend.
|
|
946
|
+
*/
|
|
947
|
+
supportsFlush(signal?: AbortSignal): Promise<boolean>;
|
|
948
|
+
private assertFlushSupported;
|
|
949
|
+
private flushUnsupported;
|
|
950
|
+
/**
|
|
951
|
+
* Whether this guest can quiesce at all — asked, rather than assumed,
|
|
952
|
+
* because `suspend({ quiesce: true })` against an older image keeps
|
|
953
|
+
* today's behaviour and reports the gap instead of refusing to suspend.
|
|
954
|
+
*/
|
|
955
|
+
supportsQuiesce(signal?: AbortSignal): Promise<boolean>;
|
|
956
|
+
private assertQuiesceSupported;
|
|
957
|
+
private quiesceUnsupported;
|
|
958
|
+
/**
|
|
959
|
+
* Whether this guest keeps a session registry at all, asked once per
|
|
960
|
+
* transport and only when a caller wants one.
|
|
961
|
+
*/
|
|
962
|
+
private assertSessionsSupported;
|
|
109
963
|
openTcpConnection(options: SandboxTcpConnectOptions): Promise<SandboxTcpConnection>;
|
|
110
964
|
/**
|
|
111
965
|
* Run one command through a fresh, call-scoped adapter + controller.
|
|
@@ -118,5 +972,77 @@ export declare class KubernetesAgentTransport {
|
|
|
118
972
|
* mutable field for two in-flight calls to race on.
|
|
119
973
|
*/
|
|
120
974
|
exec(command: string, argv?: string[], opts?: SandboxExecOptions): Promise<SandboxExecResult>;
|
|
975
|
+
/**
|
|
976
|
+
* Whether this guest implements the detach/attach ops at all, asked once
|
|
977
|
+
* per transport and only when a caller wants them.
|
|
978
|
+
*/
|
|
979
|
+
private assertExecutionAttachSupported;
|
|
980
|
+
/** `reserve-execution` for a caller-named id, with its reported state. */
|
|
981
|
+
private reserveDetached;
|
|
982
|
+
/**
|
|
983
|
+
* The `cancel-execution` control path, retried for its whole confirm
|
|
984
|
+
* window and reported UNKNOWN rather than as a failure if none of the
|
|
985
|
+
* attempts got an answer — the same rule the shared execution
|
|
986
|
+
* controller applies, because a command whose termination nobody
|
|
987
|
+
* confirmed is not a command anybody may call dead.
|
|
988
|
+
*
|
|
989
|
+
* Nothing on the detach path calls this to reconcile a lost connection.
|
|
990
|
+
* It runs when the CALLER asked for it: `SandboxExecOptions.signal`
|
|
991
|
+
* aborting, or {@link cancelExecution}.
|
|
992
|
+
*/
|
|
993
|
+
private confirmCancel;
|
|
994
|
+
/**
|
|
995
|
+
* End a command by id, from any host process holding the id and the
|
|
996
|
+
* bind token. Resolves only on a CONFIRMED termination.
|
|
997
|
+
*
|
|
998
|
+
* The outcome is not reported here — a cancelled execution's result and
|
|
999
|
+
* whatever output it managed is read back through
|
|
1000
|
+
* {@link attachExecution}, which is the op that exists for reading.
|
|
1001
|
+
*/
|
|
1002
|
+
cancelExecution(executionId: string, signal?: AbortSignal): Promise<void>;
|
|
1003
|
+
/**
|
|
1004
|
+
* Observe a command that is already the guest's, from `fromOffset` on,
|
|
1005
|
+
* and resolve with its result.
|
|
1006
|
+
*
|
|
1007
|
+
* It never signals the command. Aborting `signal` stops OBSERVING and
|
|
1008
|
+
* rejects with {@link KubernetesExecutionDetachedError}; it does not
|
|
1009
|
+
* cancel, and the command goes on running. Ending a command is
|
|
1010
|
+
* {@link cancelExecution} and nothing else.
|
|
1011
|
+
*/
|
|
1012
|
+
attachExecution(executionId: string, options?: KubernetesAttachExecutionOptions): Promise<SandboxExecResult>;
|
|
1013
|
+
/**
|
|
1014
|
+
* Run one command whose observation can outlive this connection, and —
|
|
1015
|
+
* when the connection is what failed — get it back rather than killing
|
|
1016
|
+
* the command to reconcile.
|
|
1017
|
+
*
|
|
1018
|
+
* The order is exactly: refuse if the guest cannot keep output, reserve
|
|
1019
|
+
* the id, and only then admit the command. Reserving the id the CALLER
|
|
1020
|
+
* named is what makes a retried start idempotent: a second call with the
|
|
1021
|
+
* same id inside retention finds the execution already running or
|
|
1022
|
+
* finished, sends no `execute`, and attaches to the one that exists.
|
|
1023
|
+
*
|
|
1024
|
+
* `SandboxExecOptions.signal` keeps its contract — aborting it runs the
|
|
1025
|
+
* confirmed cancel — and `detachSignal` is its opposite: it ends the
|
|
1026
|
+
* observation and leaves the command alone, for a host that is shutting
|
|
1027
|
+
* down and wants its work to survive the rollout.
|
|
1028
|
+
*/
|
|
1029
|
+
execDetached(command: string, argv?: string[], opts?: KubernetesDetachedExecOptions): Promise<SandboxExecResult>;
|
|
1030
|
+
/**
|
|
1031
|
+
* The `execute` leg of a detached run, read frame by frame.
|
|
1032
|
+
*
|
|
1033
|
+
* It deliberately does NOT go through `executeStreamed`: that path
|
|
1034
|
+
* hands its caller `{stream, data}` and nothing else, and the byte
|
|
1035
|
+
* offsets this cursor lives on are on the frames themselves. Reading
|
|
1036
|
+
* them here is what lets a reattach resume exactly where this
|
|
1037
|
+
* connection stopped, on output whose decoded length is not its byte
|
|
1038
|
+
* length — see {@link applyDelta}. The frame union and its validation
|
|
1039
|
+
* are the shared ones, so an ordinary exec and a detached one never
|
|
1040
|
+
* disagree about what the guest said.
|
|
1041
|
+
*/
|
|
1042
|
+
private executeRetained;
|
|
1043
|
+
/** One `attach-execution` stream, read to its terminal frame. */
|
|
1044
|
+
private attachOnce;
|
|
1045
|
+
/** The one error a lost observation ends with. */
|
|
1046
|
+
private detached;
|
|
121
1047
|
}
|
|
122
1048
|
//# sourceMappingURL=transport.d.ts.map
|