@namzu/sandbox 14.0.0 → 15.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +838 -0
- package/README.md +310 -14
- package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
- package/dist/backends/aci-standby-pool/index.js +13 -1
- package/dist/backends/aci-standby-pool/index.js.map +1 -1
- package/dist/backends/docker/index.d.ts.map +1 -1
- package/dist/backends/docker/index.js +19 -1
- package/dist/backends/docker/index.js.map +1 -1
- package/dist/backends/firecracker/index.d.ts.map +1 -1
- package/dist/backends/firecracker/index.js +12 -2
- package/dist/backends/firecracker/index.js.map +1 -1
- package/dist/backends/firecracker/protocol.d.ts +459 -8
- package/dist/backends/firecracker/protocol.d.ts.map +1 -1
- package/dist/backends/firecracker/protocol.js +136 -0
- package/dist/backends/firecracker/protocol.js.map +1 -1
- package/dist/backends/firecracker/transport.d.ts +539 -6
- package/dist/backends/firecracker/transport.d.ts.map +1 -1
- package/dist/backends/firecracker/transport.js +1171 -24
- package/dist/backends/firecracker/transport.js.map +1 -1
- package/dist/backends/kubernetes/egress-policy.d.ts +1088 -11
- package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -1
- package/dist/backends/kubernetes/egress-policy.js +2173 -29
- package/dist/backends/kubernetes/egress-policy.js.map +1 -1
- package/dist/backends/kubernetes/identity.d.ts +193 -0
- package/dist/backends/kubernetes/identity.d.ts.map +1 -0
- package/dist/backends/kubernetes/identity.js +147 -0
- package/dist/backends/kubernetes/identity.js.map +1 -0
- package/dist/backends/kubernetes/index.d.ts +678 -33
- package/dist/backends/kubernetes/index.d.ts.map +1 -1
- package/dist/backends/kubernetes/index.js +1180 -95
- package/dist/backends/kubernetes/index.js.map +1 -1
- package/dist/backends/kubernetes/ingress-policy.d.ts +375 -0
- package/dist/backends/kubernetes/ingress-policy.d.ts.map +1 -0
- package/dist/backends/kubernetes/ingress-policy.js +1050 -0
- package/dist/backends/kubernetes/ingress-policy.js.map +1 -0
- package/dist/backends/kubernetes/k8s-client.d.ts +213 -4
- package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -1
- package/dist/backends/kubernetes/k8s-client.js +359 -52
- package/dist/backends/kubernetes/k8s-client.js.map +1 -1
- package/dist/backends/kubernetes/lease.d.ts +40 -14
- package/dist/backends/kubernetes/lease.d.ts.map +1 -1
- package/dist/backends/kubernetes/lease.js +68 -18
- package/dist/backends/kubernetes/lease.js.map +1 -1
- package/dist/backends/kubernetes/objects.d.ts +423 -3
- package/dist/backends/kubernetes/objects.d.ts.map +1 -1
- package/dist/backends/kubernetes/objects.js +364 -2
- package/dist/backends/kubernetes/objects.js.map +1 -1
- package/dist/backends/kubernetes/per-sandbox-policy.d.ts +219 -0
- package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -0
- package/dist/backends/kubernetes/per-sandbox-policy.js +407 -0
- package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -0
- package/dist/backends/kubernetes/rbac.d.ts +153 -0
- package/dist/backends/kubernetes/rbac.d.ts.map +1 -0
- package/dist/backends/kubernetes/rbac.js +177 -0
- package/dist/backends/kubernetes/rbac.js.map +1 -0
- package/dist/backends/kubernetes/sandbox.d.ts +81 -14
- package/dist/backends/kubernetes/sandbox.d.ts.map +1 -1
- package/dist/backends/kubernetes/sandbox.js +149 -15
- package/dist/backends/kubernetes/sandbox.js.map +1 -1
- package/dist/backends/kubernetes/transport.d.ts +935 -9
- package/dist/backends/kubernetes/transport.d.ts.map +1 -1
- package/dist/backends/kubernetes/transport.js +1958 -62
- package/dist/backends/kubernetes/transport.js.map +1 -1
- package/dist/backends/kubernetes/workspace.d.ts +1149 -18
- package/dist/backends/kubernetes/workspace.d.ts.map +1 -1
- package/dist/backends/kubernetes/workspace.js +2825 -186
- package/dist/backends/kubernetes/workspace.js.map +1 -1
- package/dist/backends/remote-execution-controller.d.ts +14 -0
- package/dist/backends/remote-execution-controller.d.ts.map +1 -1
- package/dist/backends/remote-execution-controller.js.map +1 -1
- package/dist/index.d.ts +231 -13
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +247 -5
- package/dist/index.js.map +1 -1
- package/dist/testing/sandbox-conformance.d.ts +39 -5
- package/dist/testing/sandbox-conformance.d.ts.map +1 -1
- package/dist/testing/sandbox-conformance.js +436 -5
- package/dist/testing/sandbox-conformance.js.map +1 -1
- package/package.json +3 -3
- package/src/backends/aci-standby-pool/index.ts +16 -1
- package/src/backends/docker/index.ts +22 -1
- package/src/backends/firecracker/index.ts +14 -2
- package/src/backends/firecracker/protocol.ts +514 -6
- package/src/backends/firecracker/transport.ts +1492 -40
- package/src/backends/kubernetes/egress-policy.ts +3064 -53
- package/src/backends/kubernetes/identity.ts +261 -0
- package/src/backends/kubernetes/index.ts +1785 -127
- package/src/backends/kubernetes/ingress-policy.ts +1344 -0
- package/src/backends/kubernetes/k8s-client.ts +444 -54
- package/src/backends/kubernetes/lease.ts +75 -19
- package/src/backends/kubernetes/objects.ts +626 -6
- package/src/backends/kubernetes/per-sandbox-policy.ts +542 -0
- package/src/backends/kubernetes/rbac.ts +192 -0
- package/src/backends/kubernetes/sandbox.ts +218 -20
- package/src/backends/kubernetes/transport.ts +2733 -124
- package/src/backends/kubernetes/workspace.ts +4476 -222
- package/src/backends/remote-execution-controller.ts +14 -0
- package/src/index.ts +595 -14
- package/src/testing/sandbox-conformance.ts +540 -5
|
@@ -24,26 +24,53 @@
|
|
|
24
24
|
* measured, not guessed at.
|
|
25
25
|
*/
|
|
26
26
|
|
|
27
|
+
import { randomUUID } from 'node:crypto'
|
|
28
|
+
import net from 'node:net'
|
|
29
|
+
|
|
27
30
|
import type {
|
|
31
|
+
BackgroundJobStatus,
|
|
28
32
|
OpenTerminalOptions,
|
|
29
33
|
SandboxExecOptions,
|
|
30
34
|
SandboxExecResult,
|
|
35
|
+
SandboxReadFileOptions,
|
|
31
36
|
SandboxTcpConnectOptions,
|
|
32
37
|
SandboxTcpConnection,
|
|
33
38
|
TerminalSession,
|
|
34
39
|
} from '@namzu/sdk'
|
|
35
40
|
|
|
36
|
-
import type { ExecRequest } from '../firecracker/protocol.js'
|
|
37
41
|
import {
|
|
42
|
+
EXECUTION_ATTACH_FEATURE,
|
|
43
|
+
type ExecRequest,
|
|
44
|
+
ExecResultAccumulator,
|
|
45
|
+
FLUSH_FEATURE,
|
|
46
|
+
type GuestReplyIdentity,
|
|
47
|
+
QUIESCE_FEATURE,
|
|
48
|
+
type QuiesceScope,
|
|
49
|
+
type QuiescedProcess,
|
|
50
|
+
SESSIONS_FEATURE,
|
|
51
|
+
type SessionKind,
|
|
52
|
+
type SessionState,
|
|
53
|
+
parseExecEvent,
|
|
54
|
+
} from '../firecracker/protocol.js'
|
|
55
|
+
import {
|
|
56
|
+
AgentDialFailedError,
|
|
38
57
|
type AgentRequest,
|
|
58
|
+
type AgentTerminalStream,
|
|
39
59
|
type SandboxAgentHandle,
|
|
40
60
|
VsockAgentTransport,
|
|
41
61
|
type VsockTransportOptions,
|
|
42
62
|
} from '../firecracker/transport.js'
|
|
43
63
|
import {
|
|
64
|
+
type RemoteCancellationAcknowledgement,
|
|
65
|
+
RemoteCancellationUnknownError,
|
|
66
|
+
RemoteCommandError,
|
|
44
67
|
type RemoteExecutionAdapter,
|
|
45
68
|
RemoteExecutionController,
|
|
69
|
+
RemoteProtocolError,
|
|
70
|
+
RemoteResultIncompleteError,
|
|
71
|
+
type RemoteTerminalMetadata,
|
|
46
72
|
} from '../remote-execution-controller.js'
|
|
73
|
+
import { KubernetesWorkspaceReplacedError } from './identity.js'
|
|
47
74
|
|
|
48
75
|
/** The one {@link SandboxAgentHandle} arm this backend ever constructs. */
|
|
49
76
|
export type KubernetesAgentHandle = Extract<SandboxAgentHandle, { kind: 'tcp' }>
|
|
@@ -57,23 +84,355 @@ export type KubernetesAgentHandle = Extract<SandboxAgentHandle, { kind: 'tcp' }>
|
|
|
57
84
|
*/
|
|
58
85
|
export class KubernetesAgentUnauthorizedError extends Error {
|
|
59
86
|
constructor(
|
|
60
|
-
message = 'kubernetes tcp transport: the guest agent rejected this connection’s token (unauthorized)',
|
|
87
|
+
message = 'kubernetes tcp transport: the guest agent rejected this connection’s token (unauthorized). The token is the bound pod’s metadata.uid and the guest checks it BEFORE dispatch, so the request ran nothing at all; a handle that can re-read its pod follows the replacement and retries once, and this error is what stands when there is nothing to follow — the same pod is still there and still refusing, or the re-read could not be made.',
|
|
88
|
+
options?: ErrorOptions,
|
|
61
89
|
) {
|
|
62
|
-
super(message)
|
|
90
|
+
super(message, options)
|
|
63
91
|
this.name = 'KubernetesAgentUnauthorizedError'
|
|
64
92
|
}
|
|
65
93
|
}
|
|
66
94
|
|
|
95
|
+
/**
|
|
96
|
+
* Thrown when the guest agent refuses a request because it has FENCED
|
|
97
|
+
* ITSELF: an earlier process group's shutdown could not be confirmed, so it
|
|
98
|
+
* answers every op but `healthz` and `cancel-execution` with `agent_retiring`
|
|
99
|
+
* and will go on doing so until the pod is replaced.
|
|
100
|
+
*
|
|
101
|
+
* Its own class, and deliberately NOT the Firecracker tier's mapping of the
|
|
102
|
+
* same refusal. There a fenced agent becomes {@link
|
|
103
|
+
* RemoteCancellationUnknownError}, which is correct for a disposable microVM:
|
|
104
|
+
* the shared controller's rule is that the sandbox stops being reusable, and
|
|
105
|
+
* on that tier retiring one means deleting scratch. On a workspace the same
|
|
106
|
+
* error would retire the handle and take the pod — and with it every other
|
|
107
|
+
* holder's terminals, dev servers and running commands — away from callers
|
|
108
|
+
* who did nothing but share a workspace with the command that wedged.
|
|
109
|
+
*
|
|
110
|
+
* So this error refuses the ONE call rather than the workspace: the handle is
|
|
111
|
+
* not retired, nothing is patched, and every other holder's pod stays where it
|
|
112
|
+
* was. What it does NOT claim is that the next call will work. The fence is
|
|
113
|
+
* the GUEST's, and `dispatch` gates it ahead of every data-plane branch, so
|
|
114
|
+
* `readFile`, `writeFile`, `openTerminal` and `openTcpConnection` meet the
|
|
115
|
+
* same refusal on the wire — under their own paths' error shapes, since only
|
|
116
|
+
* the two control ops come through `requestChecked`. Only a new pod clears
|
|
117
|
+
* it, which is why the message names the verbs that REPLACE the pod, on both
|
|
118
|
+
* tiers that use this transport, and leaves the timing to the host: on a
|
|
119
|
+
* workspace those are `suspend()` then `resume()`, and they take the live
|
|
120
|
+
* sessions in that pod down with them.
|
|
121
|
+
*/
|
|
122
|
+
export class KubernetesAgentRetiringError extends Error {
|
|
123
|
+
override readonly name = 'KubernetesAgentRetiringError'
|
|
124
|
+
|
|
125
|
+
constructor(
|
|
126
|
+
message = 'kubernetes tcp transport: the guest agent has fenced itself (agent_retiring) because an earlier process group’s shutdown could not be confirmed. A command of unknown state may still be running in that pod and nothing on this side can end it: only a new pod clears the fence, and until the pod is replaced every call except healthz and cancel-execution meets this same refusal — reads, writes, terminals and tcp connections included. Nothing was changed on the cluster by this refusal and this handle was not retired. On a persistent workspace, call suspend() and then resume() when you are ready for the live terminals and background processes in that pod to go down; on a task sandbox, destroy() it and take another.',
|
|
127
|
+
) {
|
|
128
|
+
super(message)
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
/**
|
|
133
|
+
* Thrown when the agent's address could not be RESOLVED — the dial never
|
|
134
|
+
* reached a socket because the name has no answer here.
|
|
135
|
+
*
|
|
136
|
+
* Its own class, and its own message, because this is the one failure whose
|
|
137
|
+
* cause is the deployment's shape rather than anything the cluster did: a
|
|
138
|
+
* Service FQDN resolves through cluster DNS and nowhere else, so a host
|
|
139
|
+
* outside the cluster fails every call at name resolution and reads the
|
|
140
|
+
* result as a sandbox that never came up. The fix is a configuration field,
|
|
141
|
+
* so the error names it.
|
|
142
|
+
*/
|
|
143
|
+
export class KubernetesAgentAddressUnresolvableError extends Error {
|
|
144
|
+
override readonly name = 'KubernetesAgentAddressUnresolvableError'
|
|
145
|
+
|
|
146
|
+
constructor(
|
|
147
|
+
/** The host that did not resolve — normally a `*.svc.cluster.local`. */
|
|
148
|
+
readonly host: string,
|
|
149
|
+
message: string,
|
|
150
|
+
options?: { cause?: unknown },
|
|
151
|
+
) {
|
|
152
|
+
super(message, options)
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
/** Every `code` and message in an error's own chain, cause by cause. */
|
|
157
|
+
function errorChain(error: unknown): { codes: string[]; messages: string[] } {
|
|
158
|
+
const codes: string[] = []
|
|
159
|
+
const messages: string[] = []
|
|
160
|
+
let current: unknown = error
|
|
161
|
+
// Bounded rather than `while (current)`: a cause cycle is a hang, and no
|
|
162
|
+
// real chain on this path is more than three deep.
|
|
163
|
+
for (let depth = 0; depth < 8 && current instanceof Error; depth += 1) {
|
|
164
|
+
const code = (current as { code?: unknown }).code
|
|
165
|
+
if (typeof code === 'string') codes.push(code)
|
|
166
|
+
messages.push(current.message)
|
|
167
|
+
current = current.cause
|
|
168
|
+
}
|
|
169
|
+
return { codes, messages }
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
/**
|
|
173
|
+
* Connect-shaped: the failure came out of the DIAL, so NOTHING was sent to
|
|
174
|
+
* the guest.
|
|
175
|
+
*
|
|
176
|
+
* That last part is what makes a retry safe after the handle follows a
|
|
177
|
+
* replaced pod — a reserved execution, a half-written file or a terminal
|
|
178
|
+
* cannot be sitting in a pod this connection never opened — and it is why the
|
|
179
|
+
* test is WHERE the error came from rather than which `errno` it carries.
|
|
180
|
+
* A code cannot say that much: `ETIMEDOUT` is what the kernel raises when a
|
|
181
|
+
* connect attempt gets no answer AND what it raises on an ESTABLISHED socket
|
|
182
|
+
* that has run out of retransmits — the second of those happens mid-request,
|
|
183
|
+
* with bytes already delivered, and retrying it is not safe.
|
|
184
|
+
*
|
|
185
|
+
* {@link AgentDialFailedError} is what {@link VsockAgentTransport}'s dial
|
|
186
|
+
* throws when it gives up without a socket, so the question is asked of the
|
|
187
|
+
* error's own type, anywhere in its cause chain. Not of its text: a guest
|
|
188
|
+
* answers with text, and a command's stderr quoting "could not connect to
|
|
189
|
+
* agent" would otherwise be read as this process's own dial failing.
|
|
190
|
+
*/
|
|
191
|
+
function isConnectFailure(error: unknown): boolean {
|
|
192
|
+
let current: unknown = error
|
|
193
|
+
// Bounded for the same reason {@link errorChain} is: a cause cycle is a
|
|
194
|
+
// hang, and no real chain on this path is more than three deep.
|
|
195
|
+
for (let depth = 0; depth < 8 && current instanceof Error; depth += 1) {
|
|
196
|
+
if (current instanceof AgentDialFailedError) return true
|
|
197
|
+
current = current.cause
|
|
198
|
+
}
|
|
199
|
+
return false
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
/**
|
|
203
|
+
* What one attempt's dials actually did, recorded as they happen.
|
|
204
|
+
*
|
|
205
|
+
* The marker above is the primary test and travels with the error, which is
|
|
206
|
+
* enough for every operation that hands its failure straight back. `exec()`
|
|
207
|
+
* does not: {@link RemoteExecutionController} bounds its control requests
|
|
208
|
+
* (`bounded`, 2000ms by default) by RACING the operation against a timer, so
|
|
209
|
+
* when the dial is still inside its 30s connect-retry budget at 2s the caller
|
|
210
|
+
* is given the timer's bare `… reservation exceeded 2000ms` Error and the
|
|
211
|
+
* dial's own failure — marker, `code` and all — is discarded rather than
|
|
212
|
+
* wrapped. Classifying that error is impossible; the only way to know a
|
|
213
|
+
* socket was never established is to have watched the dials.
|
|
214
|
+
*
|
|
215
|
+
* What is watched is that a connect was ATTEMPTED, not that one failed. The
|
|
216
|
+
* bound is 2000ms and the dial's own connect timer is 5000ms by default, so
|
|
217
|
+
* the shape this exists for — a pod IP whose SYN is dropped rather than
|
|
218
|
+
* refused, which is what a released address on a routed pod network does —
|
|
219
|
+
* is ABORTED by the bound before the attempt has failed at all. A watch of
|
|
220
|
+
* failures alone is blind to exactly the half of the input space that made
|
|
221
|
+
* the bound a problem in the first place; a fast `ECONNREFUSED` is the easy
|
|
222
|
+
* half.
|
|
223
|
+
*
|
|
224
|
+
* `connected` is the half that makes the retry safe. It is set by the dial's
|
|
225
|
+
* own success callback, so "a connect was attempted and no dial ever handed
|
|
226
|
+
* back a socket" means literally nothing was sent to the guest during this
|
|
227
|
+
* attempt — the same guarantee {@link AgentDialFailedError} carries, read
|
|
228
|
+
* from the other end. It is as strong as it sounds on this arm: the `tcp`
|
|
229
|
+
* dial resolves only on the socket's own `connect` event, so there is no
|
|
230
|
+
* moment at which a byte has been written and `connected` is still false.
|
|
231
|
+
*/
|
|
232
|
+
interface DialWatch {
|
|
233
|
+
/** At least one connect attempt was made — it may not have settled. */
|
|
234
|
+
attempted: boolean
|
|
235
|
+
/** At least one connect attempt failed, with an error to read. */
|
|
236
|
+
failed: boolean
|
|
237
|
+
/** At least one dial handed back a connected socket. */
|
|
238
|
+
connected: boolean
|
|
239
|
+
/** The last connect attempt's own error, which the bound may swallow. */
|
|
240
|
+
lastError?: unknown
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
/** Nothing this attempt sent ever reached a socket. */
|
|
244
|
+
function neverConnected(dials: DialWatch | undefined): boolean {
|
|
245
|
+
return dials?.attempted === true && !dials.connected
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
/**
|
|
249
|
+
* Outcomes no rebind may retry, whatever the failure underneath them looks
|
|
250
|
+
* like.
|
|
251
|
+
*
|
|
252
|
+
* Both are the controller's way of saying the REMOTE state is unknown: a
|
|
253
|
+
* cancellation it could not confirm, a confirmed termination whose result
|
|
254
|
+
* stream broke. Their own messages interpolate the underlying failure, so a
|
|
255
|
+
* dial that broke mid-command travels as the `cause` of one of them — and a
|
|
256
|
+
* retry there would re-run a command that may already have run, against a
|
|
257
|
+
* disk that followed the pod. `RemoteCancellationUnknownError`
|
|
258
|
+
* says so in its own words: "do not automatically retry the command".
|
|
259
|
+
*/
|
|
260
|
+
function isUnretryableOutcome(error: unknown): boolean {
|
|
261
|
+
return (
|
|
262
|
+
error instanceof RemoteCancellationUnknownError || error instanceof RemoteResultIncompleteError
|
|
263
|
+
)
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
/**
|
|
267
|
+
* Name-resolution-shaped: `getaddrinfo` refused the handle's host.
|
|
268
|
+
*
|
|
269
|
+
* Asked only of a DIAL failure, and only of a handle whose host is a name —
|
|
270
|
+
* both gates are at the one call site, {@link
|
|
271
|
+
* KubernetesAgentTransport.withRebind}. The message match below is
|
|
272
|
+
* load-bearing (the retry wrapper's own Error does not carry the `code`
|
|
273
|
+
* forward) and a substring test is exactly as strong as the text it is given,
|
|
274
|
+
* so it is never asked of an arbitrary error: a `readFile` that fails because
|
|
275
|
+
* the guest reported a path containing `ENOTFOUND`, or an exec whose output is
|
|
276
|
+
* quoted into a message, is not a resolver failure and must not be rewritten
|
|
277
|
+
* into one. `net.connect` never consults a resolver for a literal either, so a
|
|
278
|
+
* `pod-ip` handle keeps its one re-read however an unrelated error is worded.
|
|
279
|
+
*/
|
|
280
|
+
const DNS_ERROR_CODES = new Set(['ENOTFOUND', 'EAI_AGAIN'])
|
|
281
|
+
|
|
282
|
+
function isNameResolutionFailure(error: unknown): boolean {
|
|
283
|
+
const { codes, messages } = errorChain(error)
|
|
284
|
+
if (codes.some((code) => DNS_ERROR_CODES.has(code))) return true
|
|
285
|
+
// `net.connect` surfaces a `getaddrinfo ENOTFOUND <host>` message whose
|
|
286
|
+
// `code` the retry wrapper's own Error does not carry forward.
|
|
287
|
+
return messages.some((message) => message.includes('ENOTFOUND') || message.includes('EAI_AGAIN'))
|
|
288
|
+
}
|
|
289
|
+
|
|
290
|
+
/**
|
|
291
|
+
* The resolver said the name does not exist — as opposed to saying it could
|
|
292
|
+
* not answer right now.
|
|
293
|
+
*
|
|
294
|
+
* Only this half is treated as permanent, and the difference is the default
|
|
295
|
+
* mode's whole retry budget. `EAI_AGAIN` is BY DEFINITION "temporary failure
|
|
296
|
+
* in name resolution": it is the shape a CoreDNS restart or a conntrack race
|
|
297
|
+
* produces for a host INSIDE the cluster, where the Service FQDN is correct
|
|
298
|
+
* and waiting is exactly the cure — the 30s budget exists to ride over that,
|
|
299
|
+
* and advice to set `agentAddress: 'pod-ip'` would be wrong for that host.
|
|
300
|
+
* `ENOTFOUND` is the out-of-cluster symptom this mode exists for: a resolver
|
|
301
|
+
* that has answered, definitively, that the name is not a name here, and no
|
|
302
|
+
* amount of re-asking it changes that.
|
|
303
|
+
*
|
|
304
|
+
* A temporary failure that outlives the budget still ends as
|
|
305
|
+
* {@link KubernetesAgentAddressUnresolvableError}, so the diagnosis is
|
|
306
|
+
* delayed rather than lost.
|
|
307
|
+
*/
|
|
308
|
+
function isMissingNameFailure(error: unknown): boolean {
|
|
309
|
+
const { codes, messages } = errorChain(error)
|
|
310
|
+
if (codes.includes('ENOTFOUND')) return true
|
|
311
|
+
return messages.some((message) => message.includes('ENOTFOUND'))
|
|
312
|
+
}
|
|
313
|
+
|
|
67
314
|
/** True for the wire shape `agent.cjs` sends when a token is rejected. */
|
|
68
315
|
function isUnauthorized(response: unknown): boolean {
|
|
316
|
+
return refusedWith(response, 'unauthorized')
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
/**
|
|
320
|
+
* True for the wire shape `agent.cjs` sends when it has fenced itself —
|
|
321
|
+
* `dispatch`'s gate, and `handleReserveExecution`'s own earlier one.
|
|
322
|
+
*/
|
|
323
|
+
function isAgentRetiring(response: unknown): boolean {
|
|
324
|
+
return refusedWith(response, 'agent_retiring')
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
/** One refusal envelope: `{ ok: false, error: <name> }`, and nothing else. */
|
|
328
|
+
function refusedWith(response: unknown, error: string): boolean {
|
|
69
329
|
if (!response || typeof response !== 'object') return false
|
|
70
330
|
const value = response as { ok?: unknown; error?: unknown }
|
|
71
|
-
return value.ok === false && value.error ===
|
|
331
|
+
return value.ok === false && value.error === error
|
|
332
|
+
}
|
|
333
|
+
|
|
334
|
+
/**
|
|
335
|
+
* The guest REFUSED this handle's token — whichever of the two shapes that
|
|
336
|
+
* refusal happens to arrive in.
|
|
337
|
+
*
|
|
338
|
+
* `reserve-execution` and `cancel-execution` come through
|
|
339
|
+
* {@link requestChecked}, so their refusal is already a
|
|
340
|
+
* {@link KubernetesAgentUnauthorizedError}. Nothing else does:
|
|
341
|
+
* `writeFile`, `readFile`, `openTerminal` and `openTcpConnection` are the
|
|
342
|
+
* shared Firecracker transport's own paths, and each of them throws the
|
|
343
|
+
* guest's error NAME as a plain `Error` — `unauthorized` and nothing more.
|
|
344
|
+
* Two shapes for one fact is why a host could never hang a single recovery
|
|
345
|
+
* off it, and this is where they become one.
|
|
346
|
+
*
|
|
347
|
+
* The message test is exact, never a substring, and that is deliberate: the
|
|
348
|
+
* guest's refusal envelope carries the bare token `unauthorized` as its whole
|
|
349
|
+
* `error` field, while a read whose PATH contains the word, or an exec whose
|
|
350
|
+
* output is quoted into a message, does not equal it. The chain is walked
|
|
351
|
+
* because the retry wrapper and the stream paths wrap rather than replace.
|
|
352
|
+
*/
|
|
353
|
+
function isUnauthorizedRefusal(error: unknown): boolean {
|
|
354
|
+
let current: unknown = error
|
|
355
|
+
for (let depth = 0; depth < 8 && current instanceof Error; depth += 1) {
|
|
356
|
+
if (current instanceof KubernetesAgentUnauthorizedError) return true
|
|
357
|
+
if (current.message === 'unauthorized') return true
|
|
358
|
+
current = current.cause
|
|
359
|
+
}
|
|
360
|
+
return false
|
|
361
|
+
}
|
|
362
|
+
|
|
363
|
+
/**
|
|
364
|
+
* The one shape a refused call ends in when no rebind could fix it.
|
|
365
|
+
*
|
|
366
|
+
* An error that is already the named class travels unchanged — replacing it
|
|
367
|
+
* would lose its `cause` chain for nothing — and every other shape is wrapped
|
|
368
|
+
* with the original on `cause`, so the plain `Error('unauthorized')` a
|
|
369
|
+
* `writeFile` used to end with is still readable underneath.
|
|
370
|
+
*/
|
|
371
|
+
function asUnauthorized(error: unknown): Error {
|
|
372
|
+
if (error instanceof KubernetesAgentUnauthorizedError) return error
|
|
373
|
+
return new KubernetesAgentUnauthorizedError(undefined, { cause: error })
|
|
374
|
+
}
|
|
375
|
+
|
|
376
|
+
/** The optional boot id on one guest reply, and nothing read from any other shape. */
|
|
377
|
+
function guestBootIdOf(reply: unknown): string | undefined {
|
|
378
|
+
if (!reply || typeof reply !== 'object') return undefined
|
|
379
|
+
const value = (reply as { guestBootId?: unknown }).guestBootId
|
|
380
|
+
return typeof value === 'string' && value !== '' ? value : undefined
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
/**
|
|
384
|
+
* The GUEST that RESERVED each execution whose cancellation could not be
|
|
385
|
+
* confirmed: the pod its reservation was accepted by, and the agent PROCESS
|
|
386
|
+
* inside that pod.
|
|
387
|
+
*
|
|
388
|
+
* Per EXECUTION and never per session, which is the whole reason it is kept
|
|
389
|
+
* here rather than read off the handle when the diagnosis runs. A workspace
|
|
390
|
+
* handle outlives many commands and follows a replaced pod, so what it is
|
|
391
|
+
* bound to when a diagnosis runs is not what a command that failed minutes
|
|
392
|
+
* ago was running in: a container the kubelet restarted an hour ago says
|
|
393
|
+
* nothing about a command started after it, and a pod ANOTHER call already
|
|
394
|
+
* rebound to is not the pod this command's processes died with. A handle-wide
|
|
395
|
+
* baseline would report both as "the guest your command was running in is
|
|
396
|
+
* gone", about a command that is very likely still running, and would name
|
|
397
|
+
* the replacement as the guest that died.
|
|
398
|
+
*
|
|
399
|
+
* What is recorded is the bind token this attempt presented — which IS the
|
|
400
|
+
* pod's `metadata.uid`, the same value the handle reports as
|
|
401
|
+
* `identity.podUid` — and the boot id the `reserve-execution` reply carried,
|
|
402
|
+
* because the guest that accepted the reservation is the guest that ran the
|
|
403
|
+
* command. Both are stamped onto the error on its way out of {@link
|
|
404
|
+
* KubernetesAgentTransport.exec}, the last frame that still knows which
|
|
405
|
+
* execution an error belongs to.
|
|
406
|
+
*
|
|
407
|
+
* Weak, so an error nobody kept takes its entry with it. The boot id is
|
|
408
|
+
* absent for a guest too old to report one, and the whole entry is absent for
|
|
409
|
+
* an error no `exec()` produced: the diagnosis then falls back to what the
|
|
410
|
+
* handle itself is bound to, which is the honest answer rather than a guess.
|
|
411
|
+
*/
|
|
412
|
+
const executionGuest = new WeakMap<Error, KubernetesReservedGuest>()
|
|
413
|
+
|
|
414
|
+
/** The pod and process one execution reserved on — see {@link executionGuest}. */
|
|
415
|
+
export interface KubernetesReservedGuest {
|
|
416
|
+
/** The bind token the attempt presented, which is the pod's uid. */
|
|
417
|
+
readonly podUid?: string
|
|
418
|
+
/** The boot id the `reserve-execution` reply carried, when it carried one. */
|
|
419
|
+
readonly guestBootId?: string
|
|
420
|
+
}
|
|
421
|
+
|
|
422
|
+
/**
|
|
423
|
+
* The guest that reserved the execution this error came from — see
|
|
424
|
+
* {@link executionGuest}. `undefined` when the error is not an execution
|
|
425
|
+
* failure this transport produced.
|
|
426
|
+
*/
|
|
427
|
+
export function guestWhenReserved(error: unknown): KubernetesReservedGuest | undefined {
|
|
428
|
+
return error instanceof Error ? executionGuest.get(error) : undefined
|
|
72
429
|
}
|
|
73
430
|
|
|
74
431
|
/**
|
|
75
|
-
* One framed control request, with the guest's
|
|
76
|
-
*
|
|
432
|
+
* One framed control request, with the guest's two NAMED refusals turned
|
|
433
|
+
* into errors a caller can catch by class: {@link
|
|
434
|
+
* KubernetesAgentUnauthorizedError} for a rejected token, and {@link
|
|
435
|
+
* KubernetesAgentRetiringError} for an agent that has fenced itself.
|
|
77
436
|
*
|
|
78
437
|
* BOTH control requests go through here — `reserve-execution` and
|
|
79
438
|
* `cancel-execution` — because the two are read by the same caller for
|
|
@@ -84,6 +443,16 @@ function isUnauthorized(response: unknown): boolean {
|
|
|
84
443
|
* rejected token read as a transport failure would therefore spend that
|
|
85
444
|
* window re-sending a request that can never succeed, and end by
|
|
86
445
|
* describing a wrong credential as an ambiguous outcome.
|
|
446
|
+
*
|
|
447
|
+
* The fenced-agent refusal is the one a SECOND holder meets. `dispatch`
|
|
448
|
+
* lets `cancel-execution` through the fence and refuses everything else, so
|
|
449
|
+
* `agent_retiring` reaches here from `reserve-execution` — which without
|
|
450
|
+
* this mapping is parsed as a reservation and rejected as
|
|
451
|
+
* `RemoteProtocolError: remote sandbox returned an invalid execution
|
|
452
|
+
* reservation`, a message about a wire shape for a pod that is telling the
|
|
453
|
+
* truth about itself. Named here rather than in the shared controller
|
|
454
|
+
* because the two tiers answer it differently: see {@link
|
|
455
|
+
* KubernetesAgentRetiringError}.
|
|
87
456
|
*/
|
|
88
457
|
async function requestChecked(
|
|
89
458
|
wire: VsockAgentTransport,
|
|
@@ -92,6 +461,7 @@ async function requestChecked(
|
|
|
92
461
|
): Promise<unknown> {
|
|
93
462
|
const response = await wire.request(request, signal)
|
|
94
463
|
if (isUnauthorized(response)) throw new KubernetesAgentUnauthorizedError()
|
|
464
|
+
if (isAgentRetiring(response)) throw new KubernetesAgentRetiringError()
|
|
95
465
|
return response
|
|
96
466
|
}
|
|
97
467
|
|
|
@@ -119,161 +489,1956 @@ export interface KubernetesTransportTiming {
|
|
|
119
489
|
readonly drainMs: number
|
|
120
490
|
}
|
|
121
491
|
|
|
122
|
-
|
|
492
|
+
/**
|
|
493
|
+
* What one `healthz` reply says about the agent, with the fence kept rather
|
|
494
|
+
* than collapsed into a boolean.
|
|
495
|
+
*
|
|
496
|
+
* {@link VsockAgentTransport.healthz} answers `false` both for an agent that
|
|
497
|
+
* did not reply and for one that replied "I have fenced myself", and those
|
|
498
|
+
* are opposite facts for a host deciding what to do next: the first is a pod
|
|
499
|
+
* that may be perfectly fine a second from now, the second is a pod that will
|
|
500
|
+
* refuse every call until it is replaced.
|
|
501
|
+
*/
|
|
502
|
+
export interface KubernetesAgentHealth {
|
|
503
|
+
/** The reply's own `ok` — `true` only for an agent serving normally. */
|
|
504
|
+
readonly ok: boolean
|
|
505
|
+
/**
|
|
506
|
+
* The agent has fenced itself and only a new pod clears it.
|
|
507
|
+
*
|
|
508
|
+
* Read from the reply's own `retiring` flag and from nothing else. A
|
|
509
|
+
* not-`ok` reply without it is NOT inferred to be a fence: the connection
|
|
510
|
+
* gate answers a `healthz` that arrived over one unauthenticated
|
|
511
|
+
* connection too many, or behind an exhausted pre-auth buffer, with a
|
|
512
|
+
* named `{ ok: false, error }` and no flag — and a caller told `retiring`
|
|
513
|
+
* there would suspend and resume a perfectly healthy pod. So `ok: false`
|
|
514
|
+
* with `retiring: false` is its own answer: this reply says nothing about
|
|
515
|
+
* whether the agent is serving.
|
|
516
|
+
*/
|
|
517
|
+
readonly retiring: boolean
|
|
518
|
+
}
|
|
519
|
+
|
|
520
|
+
/**
|
|
521
|
+
* `permanentDialFailure` is deliberately NOT inherited: this transport sets
|
|
522
|
+
* its own (see the constructor), so advertising the field would be offering a
|
|
523
|
+
* caller a predicate that is silently overwritten. `onGuestReply` is
|
|
524
|
+
* re-declared rather than inherited, with the pod the reply came FROM — see
|
|
525
|
+
* below.
|
|
526
|
+
*/
|
|
527
|
+
export interface KubernetesTransportOptions
|
|
528
|
+
extends Omit<VsockTransportOptions, 'permanentDialFailure' | 'onGuestReply'> {
|
|
123
529
|
/**
|
|
124
530
|
* Fires once per completed `exec()` call (success or failure) with
|
|
125
531
|
* the four phase durations above. The payload is exactly those four
|
|
126
532
|
* numbers — never the token, never a command, argv, or output.
|
|
127
533
|
*/
|
|
128
534
|
readonly onTiming?: (timing: KubernetesTransportTiming) => void
|
|
535
|
+
/**
|
|
536
|
+
* Re-read the live pod behind this sandbox and hand back the address
|
|
537
|
+
* and token it answers on NOW.
|
|
538
|
+
*
|
|
539
|
+
* Set only by the `pod-ip` address mode, where the handle carries a
|
|
540
|
+
* literal IP that dies with its pod; a Service FQDN needs none of this
|
|
541
|
+
* because the name outlives the pod and the dial re-resolves it every
|
|
542
|
+
* call. Consulted at most ONCE per call, and only after a dial that
|
|
543
|
+
* failed at connect — see {@link KubernetesAgentTransport}.
|
|
544
|
+
*/
|
|
545
|
+
readonly refreshHandle?: (signal?: AbortSignal) => Promise<KubernetesAgentHandle>
|
|
546
|
+
/**
|
|
547
|
+
* {@link VsockTransportOptions.onGuestReply}, plus the one thing the
|
|
548
|
+
* shared transport cannot say and this one always can: the bind token of
|
|
549
|
+
* the WIRE the reply arrived on, which is the uid of the pod that
|
|
550
|
+
* answered.
|
|
551
|
+
*
|
|
552
|
+
* A rebind builds a replacement wire and does not close the wire it
|
|
553
|
+
* leaves, so a call dispatched to the outgoing pod can still be answered
|
|
554
|
+
* BY it — a long `write-file`, a `read-file`, a pod inside its
|
|
555
|
+
* termination grace period — minutes after this transport has followed
|
|
556
|
+
* the replacement. That reply is a true statement about a pod this
|
|
557
|
+
* transport is no longer bound to, and a listener that could not tell the
|
|
558
|
+
* two apart would pair the departed pod's agent process with the
|
|
559
|
+
* replacement's uid: an identity naming a process that never ran there.
|
|
560
|
+
*
|
|
561
|
+
* Bound per wire, so the token is the one the reply's own connection
|
|
562
|
+
* presented and never the one the transport happens to hold now —
|
|
563
|
+
* `undefined` on a handle carrying no token at all, which is a wire that
|
|
564
|
+
* cannot name the pod that answered and so proves nothing about it.
|
|
565
|
+
*/
|
|
566
|
+
readonly onGuestReply?: (reply: GuestReplyIdentity, podUid: string | undefined) => void
|
|
129
567
|
}
|
|
130
568
|
|
|
569
|
+
// --- detachable executions (#479) -----------------------------------------
|
|
570
|
+
|
|
571
|
+
/** How long a detached `exec()` keeps trying to get its stream back. */
|
|
572
|
+
const DEFAULT_REATTACH_WINDOW_MS = 30_000
|
|
573
|
+
/** Pause between reattach attempts, so a refusing port is not hot-looped. */
|
|
574
|
+
const REATTACH_RETRY_DELAY_MS = 250
|
|
575
|
+
/** The guest's own default, mirrored so an observation bound exists. */
|
|
576
|
+
const DEFAULT_EXECUTION_TIMEOUT_MS = 5 * 60 * 1_000
|
|
577
|
+
/** Slack over the command's own timeout, as `executeRaw` allows itself. */
|
|
578
|
+
const EXECUTION_OBSERVATION_GRACE_MS = 10_000
|
|
579
|
+
/** How long a confirmed cancel is retried before it is reported unknown. */
|
|
580
|
+
const CANCEL_CONFIRM_WINDOW_MS = 8_000
|
|
581
|
+
const CANCEL_ATTEMPT_TIMEOUT_MS = 2_000
|
|
582
|
+
const MAX_TIMER_DELAY_MS = 2_147_483_647
|
|
583
|
+
|
|
131
584
|
/**
|
|
132
|
-
*
|
|
133
|
-
*
|
|
134
|
-
*
|
|
135
|
-
*
|
|
136
|
-
*
|
|
137
|
-
*
|
|
585
|
+
* Thrown before a command is admitted, when the caller asked for a
|
|
586
|
+
* detachable execution and the guest does not advertise
|
|
587
|
+
* {@link EXECUTION_ATTACH_FEATURE}.
|
|
588
|
+
*
|
|
589
|
+
* Refused rather than downgraded: a caller that asked for detach is about
|
|
590
|
+
* to rely on being able to come back for the output, and running the
|
|
591
|
+
* command anyway would keep nothing and tell nobody.
|
|
138
592
|
*/
|
|
139
|
-
export class
|
|
140
|
-
|
|
141
|
-
private readonly transportOptions: VsockTransportOptions
|
|
142
|
-
private readonly onTiming?: (timing: KubernetesTransportTiming) => void
|
|
143
|
-
/** Simple pass-through operations share one transport instance. */
|
|
144
|
-
private readonly wire: VsockAgentTransport
|
|
593
|
+
export class KubernetesExecutionAttachUnsupportedError extends Error {
|
|
594
|
+
override readonly name = 'KubernetesExecutionAttachUnsupportedError'
|
|
145
595
|
|
|
146
|
-
constructor(
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
this.wire = new VsockAgentTransport(handle, transportOptions)
|
|
596
|
+
constructor(
|
|
597
|
+
readonly feature: string,
|
|
598
|
+
message: string,
|
|
599
|
+
) {
|
|
600
|
+
super(message)
|
|
152
601
|
}
|
|
602
|
+
}
|
|
153
603
|
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
604
|
+
/** Why an `attach-execution` could not be served. */
|
|
605
|
+
export type KubernetesAttachRefusal =
|
|
606
|
+
| 'unknown_execution'
|
|
607
|
+
| 'output_not_retained'
|
|
608
|
+
| 'invalid_offset'
|
|
609
|
+
| 'invalid_execution_id'
|
|
610
|
+
| 'agent_retiring'
|
|
611
|
+
| 'unknown'
|
|
158
612
|
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
613
|
+
/**
|
|
614
|
+
* Thrown when the guest ANSWERED an attach and refused it — the execution
|
|
615
|
+
* is past its retention, ran in a pod that has since been replaced, never
|
|
616
|
+
* asked for its output to be kept, or the offset names bytes it does not
|
|
617
|
+
* have.
|
|
618
|
+
*
|
|
619
|
+
* Distinct from a transport failure on purpose: a refusal will not become a
|
|
620
|
+
* success by being retried, so the reattach loop stops on it instead of
|
|
621
|
+
* spending its whole window re-asking a question already answered.
|
|
622
|
+
*/
|
|
623
|
+
export class KubernetesExecutionNotAttachableError extends Error {
|
|
624
|
+
override readonly name = 'KubernetesExecutionNotAttachableError'
|
|
167
625
|
|
|
168
626
|
/**
|
|
169
|
-
* The
|
|
170
|
-
*
|
|
171
|
-
*
|
|
172
|
-
* real guest.
|
|
627
|
+
* The state the guest reported for this execution, when it reported
|
|
628
|
+
* one. `'reserved'` is the one that changes what a caller should do:
|
|
629
|
+
* the command was never started, so nothing is running.
|
|
173
630
|
*/
|
|
174
|
-
|
|
175
|
-
|
|
631
|
+
readonly executionState: string | undefined
|
|
632
|
+
|
|
633
|
+
constructor(
|
|
634
|
+
readonly executionId: string,
|
|
635
|
+
readonly reason: KubernetesAttachRefusal,
|
|
636
|
+
message: string,
|
|
637
|
+
options?: { cause?: unknown; state?: string },
|
|
638
|
+
) {
|
|
639
|
+
super(message, options)
|
|
640
|
+
this.executionState = options?.state
|
|
176
641
|
}
|
|
642
|
+
}
|
|
177
643
|
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
644
|
+
/**
|
|
645
|
+
* Thrown when a detached `exec()` gave up OBSERVING a command that is, as
|
|
646
|
+
* far as this host knows, still the guest's to run.
|
|
647
|
+
*
|
|
648
|
+
* The two fields are what makes it recoverable rather than merely a
|
|
649
|
+
* failure: `executionId` names the command to a second host process, and
|
|
650
|
+
* `outputOffset` is the byte the next `attachExecution` should resume from
|
|
651
|
+
* so nothing is read twice and no gap is invented.
|
|
652
|
+
*
|
|
653
|
+
* Nothing on this path cancels to reconcile. That is the whole point of
|
|
654
|
+
* the feature: a reset connection used to cost the workspace its pod, and
|
|
655
|
+
* a command the host has stopped watching is not a command that has to
|
|
656
|
+
* die.
|
|
657
|
+
*/
|
|
658
|
+
export class KubernetesExecutionDetachedError extends Error {
|
|
659
|
+
override readonly name = 'KubernetesExecutionDetachedError'
|
|
660
|
+
|
|
661
|
+
constructor(
|
|
662
|
+
readonly executionId: string,
|
|
663
|
+
readonly outputOffset: number,
|
|
664
|
+
message: string,
|
|
665
|
+
options?: { cause?: unknown },
|
|
666
|
+
) {
|
|
667
|
+
super(message, options)
|
|
190
668
|
}
|
|
669
|
+
}
|
|
670
|
+
|
|
671
|
+
/** What the guest reports about a finished execution on an attach. */
|
|
672
|
+
interface AttachTerminal {
|
|
673
|
+
readonly outcome: 'completed' | 'cancelled' | 'failed'
|
|
674
|
+
readonly result?: RemoteTerminalMetadata
|
|
675
|
+
readonly error?: string
|
|
676
|
+
}
|
|
677
|
+
|
|
678
|
+
/**
|
|
679
|
+
* Everything one detached observation has read so far, across however many
|
|
680
|
+
* connections it took.
|
|
681
|
+
*
|
|
682
|
+
* `offset` is the guest's own byte offset into the execution's retained
|
|
683
|
+
* log, and it is the reason this is a mutable cursor rather than a return
|
|
684
|
+
* value: a reattach resumes from it, and every chunk a previous connection
|
|
685
|
+
* delivered has to be behind it.
|
|
686
|
+
*/
|
|
687
|
+
interface OutputCursor {
|
|
688
|
+
offset: number
|
|
689
|
+
stdout: string
|
|
690
|
+
stderr: string
|
|
691
|
+
/** Bytes the guest had already evicted when a reattach asked for them. */
|
|
692
|
+
droppedBytes: number
|
|
693
|
+
}
|
|
694
|
+
|
|
695
|
+
/** One `stdout_delta`/`stderr_delta` payload, from either op's stream. */
|
|
696
|
+
function deltaStream(type: unknown): 'stdout' | 'stderr' | undefined {
|
|
697
|
+
if (type === 'stdout_delta') return 'stdout'
|
|
698
|
+
if (type === 'stderr_delta') return 'stderr'
|
|
699
|
+
return undefined
|
|
700
|
+
}
|
|
701
|
+
|
|
702
|
+
function isAttachRefusal(value: string): value is KubernetesAttachRefusal {
|
|
703
|
+
return (
|
|
704
|
+
value === 'unknown_execution' ||
|
|
705
|
+
value === 'output_not_retained' ||
|
|
706
|
+
value === 'invalid_offset' ||
|
|
707
|
+
value === 'invalid_execution_id' ||
|
|
708
|
+
value === 'agent_retiring'
|
|
709
|
+
)
|
|
710
|
+
}
|
|
191
711
|
|
|
192
|
-
|
|
193
|
-
|
|
712
|
+
function terminalMetadataOrThrow(value: unknown, executionId: string): RemoteTerminalMetadata {
|
|
713
|
+
if (!value || typeof value !== 'object') {
|
|
714
|
+
throw new RemoteProtocolError(
|
|
715
|
+
`kubernetes: the guest ended the attach stream for ${executionId} without terminal metadata`,
|
|
716
|
+
)
|
|
194
717
|
}
|
|
718
|
+
return value as RemoteTerminalMetadata
|
|
719
|
+
}
|
|
720
|
+
|
|
721
|
+
function pause(ms: number): Promise<void> {
|
|
722
|
+
return new Promise((resolve) => {
|
|
723
|
+
const timer = setTimeout(resolve, ms)
|
|
724
|
+
timer.unref?.()
|
|
725
|
+
})
|
|
726
|
+
}
|
|
727
|
+
|
|
728
|
+
/**
|
|
729
|
+
* The id shape the guest enforces, mirrored here so a caller-chosen id is
|
|
730
|
+
* refused locally with a message that says what the shape is, rather than
|
|
731
|
+
* as an `invalid_execution_id` frame after a round trip.
|
|
732
|
+
*/
|
|
733
|
+
const EXECUTION_ID_PATTERN =
|
|
734
|
+
/^exec_[0-9a-f]{8}-[0-9a-f]{4}-[1-5][0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$/i
|
|
735
|
+
|
|
736
|
+
function assertExecutionId(executionId: string): void {
|
|
737
|
+
if (EXECUTION_ID_PATTERN.test(executionId)) return
|
|
738
|
+
throw new RemoteProtocolError(
|
|
739
|
+
`kubernetes: ${JSON.stringify(executionId)} is not a valid execution id. The guest accepts exec_<uuid> and nothing else, so that an id minted by one host process is recognisable to another.`,
|
|
740
|
+
)
|
|
741
|
+
}
|
|
195
742
|
|
|
196
|
-
|
|
197
|
-
|
|
743
|
+
/**
|
|
744
|
+
* Fold one delta frame into the cursor, taking the offset FROM THE GUEST,
|
|
745
|
+
* and pass the output on to the caller.
|
|
746
|
+
*
|
|
747
|
+
* The host never derives an offset from the string it received, on either
|
|
748
|
+
* stream, and that is the whole of this function's reason to exist.
|
|
749
|
+
* `Buffer.toString('utf8')` over a chunk that ends mid-character does not
|
|
750
|
+
* preserve byte length — an incomplete sequence decodes to U+FFFD, which
|
|
751
|
+
* is WIDER than the bytes it replaced — so a cursor advanced by
|
|
752
|
+
* `Buffer.byteLength(data)` runs ahead of the guest's retained log the
|
|
753
|
+
* first time a multi-byte character straddles a read boundary. A drifted
|
|
754
|
+
* cursor is not a cosmetic error: the reattach either resumes past bytes
|
|
755
|
+
* that are then never delivered and never reported (a silent hole in a
|
|
756
|
+
* result whose truncation flags both read `false`) or names an offset the
|
|
757
|
+
* guest never had and is refused `invalid_offset`, which costs the caller
|
|
758
|
+
* the feature entirely. The guest stamps `nextOffset` on every delta of a
|
|
759
|
+
* retained execution, on the execute stream and the attach stream alike.
|
|
760
|
+
*/
|
|
761
|
+
function applyDelta(
|
|
762
|
+
cursor: OutputCursor,
|
|
763
|
+
executionId: string,
|
|
764
|
+
stream: 'stdout' | 'stderr',
|
|
765
|
+
event: Record<string, unknown>,
|
|
766
|
+
onOutput?: SandboxExecOptions['onOutput'],
|
|
767
|
+
): void {
|
|
768
|
+
const data = typeof event.data === 'string' ? event.data : ''
|
|
769
|
+
const nextOffset = Number(event.nextOffset)
|
|
770
|
+
if (!Number.isFinite(nextOffset)) {
|
|
771
|
+
throw new RemoteProtocolError(
|
|
772
|
+
`kubernetes: the guest sent a ${stream} delta for execution ${executionId} without the byte offset a reattach resumes from. Every guest advertising '${EXECUTION_ATTACH_FEATURE}' stamps them on a retained execution's output; rebuild the workspace image from this Namzu release.`,
|
|
773
|
+
)
|
|
198
774
|
}
|
|
775
|
+
cursor.offset = nextOffset
|
|
776
|
+
if (stream === 'stdout') cursor.stdout += data
|
|
777
|
+
else cursor.stderr += data
|
|
778
|
+
onOutput?.({ stream, data })
|
|
779
|
+
}
|
|
780
|
+
|
|
781
|
+
/**
|
|
782
|
+
* The guest's refusal to start a command on an id that is no longer
|
|
783
|
+
* `reserved` — another host process got its `execute` in first.
|
|
784
|
+
*
|
|
785
|
+
* It reads as terminal (the guest answered, and answering again will not
|
|
786
|
+
* change it) and it is the one case where that is the wrong conclusion:
|
|
787
|
+
* the command this call asked for EXISTS, so the caller gets it by
|
|
788
|
+
* attaching rather than an error about a race it does not care about.
|
|
789
|
+
*/
|
|
790
|
+
function lostTheStartRace(error: unknown): boolean {
|
|
791
|
+
return error instanceof RemoteCommandError && error.message.startsWith('execution_not_reserved')
|
|
792
|
+
}
|
|
793
|
+
|
|
794
|
+
/**
|
|
795
|
+
* Errors a reattach must NOT spend its window re-asking about: the guest
|
|
796
|
+
* answered, and the answer will be the same next time.
|
|
797
|
+
*/
|
|
798
|
+
function isTerminalAttachError(error: unknown): boolean {
|
|
799
|
+
return (
|
|
800
|
+
error instanceof KubernetesExecutionNotAttachableError ||
|
|
801
|
+
error instanceof KubernetesAgentUnauthorizedError ||
|
|
802
|
+
error instanceof KubernetesAgentAddressUnresolvableError ||
|
|
803
|
+
error instanceof RemoteCommandError ||
|
|
804
|
+
error instanceof RemoteProtocolError
|
|
805
|
+
)
|
|
806
|
+
}
|
|
199
807
|
|
|
200
|
-
|
|
201
|
-
|
|
808
|
+
/** The guest's `cancel-execution` reply, refusals told apart from blips. */
|
|
809
|
+
function parseCancellationReply(
|
|
810
|
+
executionId: string,
|
|
811
|
+
response: unknown,
|
|
812
|
+
): RemoteCancellationAcknowledgement {
|
|
813
|
+
const reply = (response ?? {}) as Record<string, unknown>
|
|
814
|
+
if (reply.ok === true) {
|
|
815
|
+
const state = String(reply.state ?? '')
|
|
816
|
+
if (state === 'cancelled' || state === 'completed' || state === 'failed') {
|
|
817
|
+
return reply as unknown as RemoteCancellationAcknowledgement
|
|
818
|
+
}
|
|
819
|
+
throw new RemoteProtocolError(
|
|
820
|
+
`kubernetes: the guest acknowledged cancelling ${executionId} with an unknown state ${JSON.stringify(reply.state)}`,
|
|
821
|
+
)
|
|
202
822
|
}
|
|
823
|
+
const error = typeof reply.error === 'string' ? reply.error : 'unknown'
|
|
824
|
+
if (error === 'unknown_execution' || error === 'invalid_execution_id') {
|
|
825
|
+
throw new KubernetesExecutionNotAttachableError(
|
|
826
|
+
executionId,
|
|
827
|
+
error,
|
|
828
|
+
`kubernetes: the guest holds no execution ${executionId} to cancel (${error}). A record is kept only for its retention window and is lost when the pod is replaced.`,
|
|
829
|
+
)
|
|
830
|
+
}
|
|
831
|
+
throw new Error(`kubernetes: the guest refused to cancel ${executionId}: ${error}`)
|
|
832
|
+
}
|
|
203
833
|
|
|
204
|
-
|
|
205
|
-
|
|
834
|
+
/**
|
|
835
|
+
* The SDK-shaped result of an attached observation.
|
|
836
|
+
*
|
|
837
|
+
* A reported gap sets BOTH truncation flags. The retained log is one
|
|
838
|
+
* interleaved space, so bytes lost out of it cannot be attributed to
|
|
839
|
+
* stdout or to stderr, and the contract already has exactly one way to say
|
|
840
|
+
* "this output is not all of it". Saying it on one stream only would be a
|
|
841
|
+
* guess; saying it on neither would hand back a short stream that looks
|
|
842
|
+
* complete, which is the thing this design refuses to do.
|
|
843
|
+
*/
|
|
844
|
+
function resultFromAttachTerminal(
|
|
845
|
+
executionId: string,
|
|
846
|
+
terminal: AttachTerminal,
|
|
847
|
+
cursor: OutputCursor,
|
|
848
|
+
): SandboxExecResult {
|
|
849
|
+
if (terminal.outcome === 'failed' && terminal.result === undefined) {
|
|
850
|
+
throw new RemoteCommandError(
|
|
851
|
+
terminal.error ?? `the guest reported execution ${executionId} as failed`,
|
|
852
|
+
)
|
|
206
853
|
}
|
|
854
|
+
const metadata = terminalMetadataOrThrow(terminal.result, executionId)
|
|
855
|
+
const lost = cursor.droppedBytes > 0
|
|
856
|
+
return {
|
|
857
|
+
exitCode: metadata.exitCode,
|
|
858
|
+
stdout: cursor.stdout,
|
|
859
|
+
stderr: cursor.stderr,
|
|
860
|
+
...(metadata.signal !== undefined ? { signal: metadata.signal } : {}),
|
|
861
|
+
timedOut: metadata.timedOut === true,
|
|
862
|
+
durationMs: metadata.durationMs,
|
|
863
|
+
stdoutTruncated: metadata.stdoutTruncated === true || lost,
|
|
864
|
+
stderrTruncated: metadata.stderrTruncated === true || lost,
|
|
865
|
+
}
|
|
866
|
+
}
|
|
207
867
|
|
|
868
|
+
/**
|
|
869
|
+
* What a caller passes to read an execution it did not necessarily start.
|
|
870
|
+
*
|
|
871
|
+
* `signal` here is an OBSERVATION signal, not the SDK's command signal:
|
|
872
|
+
* aborting it stops reading and leaves the command running. That is the
|
|
873
|
+
* opposite of `SandboxExecOptions.signal`, and it is why this type does not
|
|
874
|
+
* extend it.
|
|
875
|
+
*/
|
|
876
|
+
export interface KubernetesAttachExecutionOptions {
|
|
877
|
+
/** Byte offset to resume from. Default 0 — the whole retained log. */
|
|
878
|
+
readonly fromOffset?: number
|
|
879
|
+
readonly onOutput?: SandboxExecOptions['onOutput']
|
|
880
|
+
/** Stops OBSERVING. Never cancels; see {@link KubernetesAgentTransport.cancelExecution}. */
|
|
881
|
+
readonly signal?: AbortSignal
|
|
208
882
|
/**
|
|
209
|
-
*
|
|
210
|
-
*
|
|
211
|
-
*
|
|
212
|
-
* state (it dials fresh every time), so building one per `exec()` is
|
|
213
|
-
* free and makes concurrent `exec()` calls on the same
|
|
214
|
-
* `KubernetesAgentTransport` correctly independent: each gets its own
|
|
215
|
-
* `onDial` closure and its own timing accumulator, with no shared
|
|
216
|
-
* mutable field for two in-flight calls to race on.
|
|
883
|
+
* Called when the guest reports that bytes the caller asked for had
|
|
884
|
+
* already been evicted from the retained log. The result's truncation
|
|
885
|
+
* flags say the same thing; this says how much.
|
|
217
886
|
*/
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
)
|
|
223
|
-
|
|
224
|
-
let reserveMs = 0
|
|
225
|
-
let executeMs = 0
|
|
226
|
-
let executeSettledAt = 0
|
|
887
|
+
readonly onGap?: (gap: {
|
|
888
|
+
readonly executionId: string
|
|
889
|
+
readonly fromOffset: number
|
|
890
|
+
readonly droppedBytes: number
|
|
891
|
+
}) => void
|
|
892
|
+
}
|
|
227
893
|
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
894
|
+
/**
|
|
895
|
+
* An `exec()` whose observation can be lost and taken up again — on this
|
|
896
|
+
* handle or in another host process — instead of costing the command its
|
|
897
|
+
* life.
|
|
898
|
+
*
|
|
899
|
+
* Deliberately NOT on the SDK's `SandboxExecOptions`: every other backend
|
|
900
|
+
* would then have to answer for a field it cannot honour, and the SDK's
|
|
901
|
+
* exec contract stays exactly what it was. This is a Kubernetes workspace
|
|
902
|
+
* surface, layered over the shared options type rather than widening it.
|
|
903
|
+
*/
|
|
904
|
+
export interface KubernetesDetachedExecOptions
|
|
905
|
+
extends SandboxExecOptions,
|
|
906
|
+
Pick<KubernetesAttachExecutionOptions, 'onGap'> {
|
|
907
|
+
/**
|
|
908
|
+
* The id this command is known by, to this process and to any other.
|
|
909
|
+
* Minted here when absent. Reserving an id the guest still holds does
|
|
910
|
+
* NOT start a second command: the call attaches to the one that exists,
|
|
911
|
+
* which is what makes a retried start idempotent for as long as the
|
|
912
|
+
* record lives.
|
|
913
|
+
*/
|
|
914
|
+
readonly executionId?: string
|
|
915
|
+
/**
|
|
916
|
+
* Ask the guest to retain this command's output so the observation can
|
|
917
|
+
* be resumed. Setting `executionId` implies it; the flag is what a
|
|
918
|
+
* caller that does not care about the id passes.
|
|
919
|
+
*/
|
|
920
|
+
readonly detach?: boolean
|
|
921
|
+
/**
|
|
922
|
+
* Stop observing and leave the command running — for a host that is
|
|
923
|
+
* shutting down. Rejects with {@link KubernetesExecutionDetachedError},
|
|
924
|
+
* which names the id and the offset to resume from.
|
|
925
|
+
*
|
|
926
|
+
* The opposite of `signal`, which keeps the SDK contract and terminates.
|
|
927
|
+
*/
|
|
928
|
+
readonly detachSignal?: AbortSignal
|
|
929
|
+
/**
|
|
930
|
+
* How long a lost connection is retried before the call gives up and
|
|
931
|
+
* reports itself detached. Default 30s.
|
|
932
|
+
*/
|
|
933
|
+
readonly reattachWindowMs?: number
|
|
934
|
+
}
|
|
234
935
|
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
936
|
+
// --- guest sessions (#478) ------------------------------------------------
|
|
937
|
+
|
|
938
|
+
/**
|
|
939
|
+
* Thrown before anything is started, when the caller asked for a session and
|
|
940
|
+
* the guest does not advertise {@link SESSIONS_FEATURE}.
|
|
941
|
+
*
|
|
942
|
+
* Refused, never downgraded to a connection-bound terminal. A caller that
|
|
943
|
+
* asked for a session is about to rely on coming back to it after its own
|
|
944
|
+
* process has been replaced; handing it one that dies with the socket would
|
|
945
|
+
* look like it worked until the one moment it was needed.
|
|
946
|
+
*/
|
|
947
|
+
export class KubernetesSessionsUnsupportedError extends Error {
|
|
948
|
+
override readonly name = 'KubernetesSessionsUnsupportedError'
|
|
949
|
+
|
|
950
|
+
constructor(
|
|
951
|
+
readonly feature: string,
|
|
952
|
+
message: string,
|
|
953
|
+
) {
|
|
954
|
+
super(message)
|
|
955
|
+
}
|
|
956
|
+
}
|
|
957
|
+
|
|
958
|
+
/** Why the guest refused a session request. */
|
|
959
|
+
export type KubernetesSessionRefusal =
|
|
960
|
+
| 'unknown_session'
|
|
961
|
+
| 'invalid_session_id'
|
|
962
|
+
| 'invalid_offset'
|
|
963
|
+
| 'session_exists'
|
|
964
|
+
| 'session_capacity'
|
|
965
|
+
| 'missing_command'
|
|
966
|
+
| 'spawn_failed'
|
|
967
|
+
| 'agent_retiring'
|
|
968
|
+
| 'unknown'
|
|
969
|
+
|
|
970
|
+
const SESSION_REFUSALS = new Set<KubernetesSessionRefusal>([
|
|
971
|
+
'unknown_session',
|
|
972
|
+
'invalid_session_id',
|
|
973
|
+
'invalid_offset',
|
|
974
|
+
'session_exists',
|
|
975
|
+
'session_capacity',
|
|
976
|
+
'missing_command',
|
|
977
|
+
'spawn_failed',
|
|
978
|
+
'agent_retiring',
|
|
979
|
+
])
|
|
980
|
+
|
|
981
|
+
function isSessionRefusal(value: string): value is KubernetesSessionRefusal {
|
|
982
|
+
return SESSION_REFUSALS.has(value as KubernetesSessionRefusal)
|
|
983
|
+
}
|
|
984
|
+
|
|
985
|
+
/**
|
|
986
|
+
* Thrown when the guest ANSWERED and refused: the session is past its
|
|
987
|
+
* retention, ran in a pod that has since been replaced, the id is already
|
|
988
|
+
* taken, or the offset names bytes it does not have.
|
|
989
|
+
*
|
|
990
|
+
* Distinct from a transport failure for the same reason
|
|
991
|
+
* {@link KubernetesExecutionNotAttachableError} is: a refusal does not
|
|
992
|
+
* become a success by being retried.
|
|
993
|
+
*/
|
|
994
|
+
export class KubernetesSessionRefusedError extends Error {
|
|
995
|
+
override readonly name = 'KubernetesSessionRefusedError'
|
|
996
|
+
|
|
997
|
+
constructor(
|
|
998
|
+
readonly sessionId: string,
|
|
999
|
+
readonly reason: KubernetesSessionRefusal,
|
|
1000
|
+
message: string,
|
|
1001
|
+
options?: { cause?: unknown },
|
|
1002
|
+
) {
|
|
1003
|
+
super(message, options)
|
|
1004
|
+
}
|
|
1005
|
+
}
|
|
1006
|
+
|
|
1007
|
+
/** One row of {@link KubernetesAgentTransport.listSessions}. */
|
|
1008
|
+
export interface KubernetesSessionSummary {
|
|
1009
|
+
readonly sessionId: string
|
|
1010
|
+
readonly kind: SessionKind
|
|
1011
|
+
/** The program, as it was asked for. Never the environment it was given. */
|
|
1012
|
+
readonly command: string
|
|
1013
|
+
readonly args: readonly string[]
|
|
1014
|
+
readonly startedAt: number
|
|
1015
|
+
readonly lastInputAt?: number
|
|
1016
|
+
readonly lastOutputAt?: number
|
|
1017
|
+
/** Pass as `fromOffset` to read everything this session has printed since. */
|
|
1018
|
+
readonly nextOffset: number
|
|
1019
|
+
/** Bytes the ring has evicted over this session's life. */
|
|
1020
|
+
readonly droppedBytes: number
|
|
1021
|
+
readonly state: SessionState
|
|
1022
|
+
/** Whether a host process is attached to it right now. */
|
|
1023
|
+
readonly attached: boolean
|
|
1024
|
+
readonly exitCode?: number
|
|
1025
|
+
readonly signal?: number
|
|
1026
|
+
}
|
|
1027
|
+
|
|
1028
|
+
/**
|
|
1029
|
+
* One read of a session's retained output, in the SDK's
|
|
1030
|
+
* `BackgroundJobOutput` shape — deliberately, because it answers the same
|
|
1031
|
+
* question for the same kind of consumer and a second vocabulary for
|
|
1032
|
+
* "here is the next chunk and here is what you missed" helps nobody.
|
|
1033
|
+
*/
|
|
1034
|
+
export interface KubernetesSessionOutput {
|
|
1035
|
+
readonly chunk: string
|
|
1036
|
+
readonly nextOffset: number
|
|
1037
|
+
readonly droppedBytes: number
|
|
1038
|
+
readonly status: BackgroundJobStatus
|
|
1039
|
+
readonly exitCode?: number
|
|
1040
|
+
}
|
|
1041
|
+
|
|
1042
|
+
/**
|
|
1043
|
+
* A terminal on a workspace, with the three things only a SESSION's reader
|
|
1044
|
+
* needs. On a connection-bound terminal the two optional members are absent,
|
|
1045
|
+
* which is the honest answer: there is no session to name and nothing to
|
|
1046
|
+
* detach from.
|
|
1047
|
+
*/
|
|
1048
|
+
export interface KubernetesWorkspaceTerminal extends TerminalSession {
|
|
1049
|
+
/** Present exactly when this terminal belongs to a guest session. */
|
|
1050
|
+
readonly sessionId?: string
|
|
1051
|
+
/** One past the newest retained byte delivered so far. */
|
|
1052
|
+
nextOffset?(): number | undefined
|
|
1053
|
+
/**
|
|
1054
|
+
* Stop reading and leave the program running — the opposite of
|
|
1055
|
+
* {@link TerminalSession.kill}. `exited` then rejects with
|
|
1056
|
+
* `AgentSessionDetachedError`, because a resolved `exited` would claim
|
|
1057
|
+
* an exit that did not happen.
|
|
1058
|
+
*/
|
|
1059
|
+
detach?(): void
|
|
1060
|
+
}
|
|
1061
|
+
|
|
1062
|
+
/**
|
|
1063
|
+
* What an ATTACH hands back: the same terminal, with the three session
|
|
1064
|
+
* members present rather than optional. `openTerminal` returns the looser
|
|
1065
|
+
* type because it serves both shapes and a connection-bound terminal
|
|
1066
|
+
* genuinely has no session to name.
|
|
1067
|
+
*/
|
|
1068
|
+
export interface KubernetesSessionTerminal extends KubernetesWorkspaceTerminal {
|
|
1069
|
+
readonly sessionId: string
|
|
1070
|
+
nextOffset(): number | undefined
|
|
1071
|
+
detach(): void
|
|
1072
|
+
}
|
|
1073
|
+
|
|
1074
|
+
/** `openTerminal` on a workspace, widened by the two session fields. */
|
|
1075
|
+
export interface KubernetesOpenTerminalOptions extends OpenTerminalOptions {
|
|
1076
|
+
/**
|
|
1077
|
+
* Name this terminal so a later host process can find it again.
|
|
1078
|
+
* Requires {@link persistent}; on its own it names nothing.
|
|
1079
|
+
*/
|
|
1080
|
+
readonly sessionId?: string
|
|
1081
|
+
/**
|
|
1082
|
+
* Hand the PTY to the guest's session registry rather than to this
|
|
1083
|
+
* connection. Losing the connection then DETACHES — no signal is sent,
|
|
1084
|
+
* and the program ends when it exits, on `killSession`, or when the pod
|
|
1085
|
+
* stops.
|
|
1086
|
+
*/
|
|
1087
|
+
readonly persistent?: boolean
|
|
1088
|
+
}
|
|
1089
|
+
|
|
1090
|
+
/** Rejoin a terminal session that is already running. */
|
|
1091
|
+
export interface KubernetesAttachTerminalOptions {
|
|
1092
|
+
/** Byte offset to replay from. Default 0 — everything the ring still holds. */
|
|
1093
|
+
readonly fromOffset?: number
|
|
1094
|
+
/** Resize the PTY on attach, for a reader whose window is a different shape. */
|
|
1095
|
+
readonly size?: { readonly cols: number; readonly rows: number }
|
|
1096
|
+
}
|
|
1097
|
+
|
|
1098
|
+
/** Start a program with no terminal, which only a kill or the pod ends. */
|
|
1099
|
+
export interface KubernetesStartDetachedOptions {
|
|
1100
|
+
readonly sessionId: string
|
|
1101
|
+
readonly command: string
|
|
1102
|
+
readonly args?: readonly string[]
|
|
1103
|
+
readonly cwd?: string
|
|
1104
|
+
readonly env?: Record<string, string>
|
|
1105
|
+
}
|
|
1106
|
+
|
|
1107
|
+
/** Read a session's retained output without attaching to it. */
|
|
1108
|
+
export interface KubernetesReadSessionOptions {
|
|
1109
|
+
readonly fromOffset?: number
|
|
1110
|
+
}
|
|
1111
|
+
|
|
1112
|
+
function sessionNumber(value: unknown, fallback = 0): number {
|
|
1113
|
+
const parsed = Number(value)
|
|
1114
|
+
return Number.isFinite(parsed) ? parsed : fallback
|
|
1115
|
+
}
|
|
1116
|
+
|
|
1117
|
+
/** How long one `readSession` may spend reading a bounded, one-shot reply. */
|
|
1118
|
+
const SESSION_READ_TIMEOUT_MS = 30_000
|
|
1119
|
+
|
|
1120
|
+
/**
|
|
1121
|
+
* How long the SHARED re-read behind a rebind may take before it is given up
|
|
1122
|
+
* on — two API GETs on a client that sets no per-request timeout.
|
|
1123
|
+
*
|
|
1124
|
+
* It replaces the caller's signal rather than joining it, because the read is
|
|
1125
|
+
* shared: see {@link KubernetesAgentTransport.rebind}. Generous enough that a
|
|
1126
|
+
* busy API server still answers, short enough that a call refused by a
|
|
1127
|
+
* replacement is not held behind a hung one.
|
|
1128
|
+
*/
|
|
1129
|
+
const REBIND_READ_TIMEOUT_MS = 10_000
|
|
1130
|
+
|
|
1131
|
+
/**
|
|
1132
|
+
* The same shape the guest enforces, checked here so a bad id is a local
|
|
1133
|
+
* error naming the rule rather than a round trip that comes back
|
|
1134
|
+
* `invalid_session_id`.
|
|
1135
|
+
*/
|
|
1136
|
+
const SESSION_ID_PATTERN = /^[A-Za-z0-9][A-Za-z0-9._-]{0,63}$/
|
|
1137
|
+
|
|
1138
|
+
function assertSessionId(sessionId: string): void {
|
|
1139
|
+
if (!SESSION_ID_PATTERN.test(sessionId)) {
|
|
1140
|
+
throw new Error(
|
|
1141
|
+
`kubernetes: ${JSON.stringify(sessionId)} cannot name a session. It must be 1-64 characters of letters, digits, '.', '_' or '-', starting alphanumeric. The id is the ONLY way another host process finds this session again, so it is refused rather than sanitised.`,
|
|
1142
|
+
)
|
|
1143
|
+
}
|
|
1144
|
+
}
|
|
1145
|
+
|
|
1146
|
+
/**
|
|
1147
|
+
* Both session fields or neither.
|
|
1148
|
+
*
|
|
1149
|
+
* A `sessionId` without `persistent` would open a terminal that dies with
|
|
1150
|
+
* its connection under a name nothing can use, and `persistent` without an
|
|
1151
|
+
* id would open one nobody can ever find. Either alone is a mistake worth a
|
|
1152
|
+
* message rather than a surprise.
|
|
1153
|
+
*/
|
|
1154
|
+
function assertSessionOpen(options: KubernetesOpenTerminalOptions): string {
|
|
1155
|
+
if (options.sessionId === undefined || options.persistent !== true) {
|
|
1156
|
+
throw new Error(
|
|
1157
|
+
'kubernetes: a persistent terminal needs both `sessionId` and `persistent: true`. An id without `persistent` opens a connection-bound terminal under a name nothing can attach to, and `persistent` without an id opens one nobody can find again.',
|
|
1158
|
+
)
|
|
1159
|
+
}
|
|
1160
|
+
assertSessionId(options.sessionId)
|
|
1161
|
+
return options.sessionId
|
|
1162
|
+
}
|
|
1163
|
+
|
|
1164
|
+
/** The workspace-facing terminal around one open session stream. */
|
|
1165
|
+
function sessionTerminal(
|
|
1166
|
+
stream: AgentTerminalStream,
|
|
1167
|
+
sessionId: string,
|
|
1168
|
+
): KubernetesSessionTerminal {
|
|
1169
|
+
return {
|
|
1170
|
+
...stream.session,
|
|
1171
|
+
sessionId,
|
|
1172
|
+
nextOffset: () => stream.nextOffset(),
|
|
1173
|
+
detach: () => stream.detach(),
|
|
1174
|
+
}
|
|
1175
|
+
}
|
|
1176
|
+
|
|
1177
|
+
/** The guest's row, structurally validated. */
|
|
1178
|
+
function parseSessionSummary(value: unknown): KubernetesSessionSummary {
|
|
1179
|
+
if (!value || typeof value !== 'object') {
|
|
1180
|
+
throw new RemoteProtocolError('kubernetes: the guest sent a session row that is not an object')
|
|
1181
|
+
}
|
|
1182
|
+
const row = value as Record<string, unknown>
|
|
1183
|
+
if (typeof row.sessionId !== 'string') {
|
|
1184
|
+
throw new RemoteProtocolError('kubernetes: the guest sent a session row with no sessionId')
|
|
1185
|
+
}
|
|
1186
|
+
const kind = row.kind === 'detached' ? 'detached' : 'terminal'
|
|
1187
|
+
const state = row.state === 'exited' ? 'exited' : 'running'
|
|
1188
|
+
return {
|
|
1189
|
+
sessionId: row.sessionId,
|
|
1190
|
+
kind,
|
|
1191
|
+
command: typeof row.command === 'string' ? row.command : '',
|
|
1192
|
+
args: Array.isArray(row.args) ? row.args.map(String) : [],
|
|
1193
|
+
startedAt: sessionNumber(row.startedAt),
|
|
1194
|
+
...(row.lastInputAt !== undefined ? { lastInputAt: sessionNumber(row.lastInputAt) } : {}),
|
|
1195
|
+
...(row.lastOutputAt !== undefined ? { lastOutputAt: sessionNumber(row.lastOutputAt) } : {}),
|
|
1196
|
+
nextOffset: sessionNumber(row.nextOffset),
|
|
1197
|
+
droppedBytes: sessionNumber(row.droppedBytes),
|
|
1198
|
+
state,
|
|
1199
|
+
attached: row.attached === true,
|
|
1200
|
+
...(typeof row.exitCode === 'number' ? { exitCode: row.exitCode } : {}),
|
|
1201
|
+
...(typeof row.signal === 'number' ? { signal: row.signal } : {}),
|
|
1202
|
+
}
|
|
1203
|
+
}
|
|
1204
|
+
|
|
1205
|
+
/**
|
|
1206
|
+
* The SDK's three-way job status, from the guest's two-way state plus the
|
|
1207
|
+
* signal. A program the kernel stopped is `killed`, not `exited`: the
|
|
1208
|
+
* distinction is the whole reason `BackgroundJobStatus` has three members.
|
|
1209
|
+
*/
|
|
1210
|
+
function sessionStatus(summary: {
|
|
1211
|
+
readonly state: SessionState
|
|
1212
|
+
readonly signal?: number
|
|
1213
|
+
}): BackgroundJobStatus {
|
|
1214
|
+
if (summary.state === 'running') return 'running'
|
|
1215
|
+
return summary.signal !== undefined ? 'killed' : 'exited'
|
|
1216
|
+
}
|
|
1217
|
+
|
|
1218
|
+
/** The refusal shape every session op answers a bad request with. */
|
|
1219
|
+
function sessionRefusal(
|
|
1220
|
+
sessionId: string,
|
|
1221
|
+
reply: Record<string, unknown>,
|
|
1222
|
+
operation: string,
|
|
1223
|
+
cause?: unknown,
|
|
1224
|
+
): KubernetesSessionRefusedError {
|
|
1225
|
+
const code = typeof reply.error === 'string' ? reply.error : 'unknown'
|
|
1226
|
+
const detail = typeof reply.message === 'string' ? ` ${reply.message}` : ''
|
|
1227
|
+
return new KubernetesSessionRefusedError(
|
|
1228
|
+
sessionId,
|
|
1229
|
+
isSessionRefusal(code) ? code : 'unknown',
|
|
1230
|
+
`kubernetes: the guest refused ${operation} for session ${sessionId} (${code}).${detail} A session lives in the pod's memory only: it is lost when the pod is replaced, and an exited one is kept for the window NAMZU_AGENT_SESSION_TERMINAL_TTL_MS names.`,
|
|
1231
|
+
cause !== undefined ? { cause } : undefined,
|
|
1232
|
+
)
|
|
1233
|
+
}
|
|
1234
|
+
|
|
1235
|
+
/**
|
|
1236
|
+
* The same refusal, arriving on a STREAM rather than in a reply.
|
|
1237
|
+
*
|
|
1238
|
+
* `openTerminal` and `attachSession` do not get a `{ ok: false }` body: the
|
|
1239
|
+
* guest refuses them with an `error` FRAME, which the vsock transport
|
|
1240
|
+
* surfaces as a plain `Error` carrying the guest's code as its message. Left
|
|
1241
|
+
* alone, those two would be the only session verbs a caller could not catch
|
|
1242
|
+
* by class — so the code is recognised here, at the one boundary where it is
|
|
1243
|
+
* still recognisable, and everything else (a dial failure, an idle timeout)
|
|
1244
|
+
* is handed back untouched.
|
|
1245
|
+
*/
|
|
1246
|
+
function sessionStreamFailure(sessionId: string, operation: string, error: unknown): unknown {
|
|
1247
|
+
if (!(error instanceof Error) || !isSessionRefusal(error.message)) return error
|
|
1248
|
+
return sessionRefusal(sessionId, { error: error.message }, operation, error)
|
|
1249
|
+
}
|
|
1250
|
+
|
|
1251
|
+
/**
|
|
1252
|
+
* The kubernetes backend's dialable transport: a `tcp` handle plus a
|
|
1253
|
+
* `RemoteExecutionAdapter` built from it, so `exec()` gets the same
|
|
1254
|
+
* reserve-before-admission behaviour (cancellation, timeout ownership,
|
|
1255
|
+
* "do not infer complete output on an ambiguous cancel") every other
|
|
1256
|
+
* remote backend gets, while every network operation is delegated to
|
|
1257
|
+
* {@link VsockAgentTransport} for the actual dial/frame/token work.
|
|
1258
|
+
*/
|
|
1259
|
+
// --- quiesce (#485) -------------------------------------------------------
|
|
1260
|
+
|
|
1261
|
+
/**
|
|
1262
|
+
* Thrown when a quiesce was asked of a guest that cannot perform one:
|
|
1263
|
+
* either its `healthz` does not advertise {@link QUIESCE_FEATURE}, or it
|
|
1264
|
+
* answered `unknown_op: quiesce`.
|
|
1265
|
+
*
|
|
1266
|
+
* Refused rather than treated as "nothing was running". A caller asks for a
|
|
1267
|
+
* quiesce because it is about to read the disk and needs it still; an image
|
|
1268
|
+
* that cannot stop its processes has to say so, not resolve with an empty
|
|
1269
|
+
* list that reads exactly like a guest which had nothing to stop.
|
|
1270
|
+
*/
|
|
1271
|
+
export class KubernetesQuiesceUnsupportedError extends Error {
|
|
1272
|
+
override readonly name = 'KubernetesQuiesceUnsupportedError'
|
|
1273
|
+
|
|
1274
|
+
constructor(
|
|
1275
|
+
readonly feature: string,
|
|
1276
|
+
message: string,
|
|
1277
|
+
) {
|
|
1278
|
+
super(message)
|
|
1279
|
+
}
|
|
1280
|
+
}
|
|
1281
|
+
|
|
1282
|
+
/**
|
|
1283
|
+
* Thrown when nothing can promise the guest is quiet.
|
|
1284
|
+
*
|
|
1285
|
+
* Usually because the guest ANSWERED and said so — a process survived
|
|
1286
|
+
* `SIGKILL`, the scan could not be performed, or the op ran out of its own
|
|
1287
|
+
* deadline — and then the guest's own message names the pid. The workspace
|
|
1288
|
+
* handle raises the same class for the one case that never reaches a guest:
|
|
1289
|
+
* a `suspend({ quiesce: true })` arriving while a suspend WITHOUT a quiesce
|
|
1290
|
+
* is already in flight, whose patch has gone over processes nobody stopped
|
|
1291
|
+
* (`reason: 'suspend_already_in_flight'`). Both mean the one thing a caller
|
|
1292
|
+
* has to act on: do not trust a capture taken now.
|
|
1293
|
+
*
|
|
1294
|
+
* Its own class for the same reason {@link KubernetesSessionRefusedError}
|
|
1295
|
+
* is: this is an answer, not a transport failure, and retrying it is a
|
|
1296
|
+
* decision the caller makes with the pid in hand rather than one a wrapper
|
|
1297
|
+
* makes on its behalf.
|
|
1298
|
+
*/
|
|
1299
|
+
export class KubernetesQuiesceUnconfirmedError extends Error {
|
|
1300
|
+
override readonly name = 'KubernetesQuiesceUnconfirmedError'
|
|
1301
|
+
|
|
1302
|
+
constructor(
|
|
1303
|
+
readonly reason: string,
|
|
1304
|
+
message: string,
|
|
1305
|
+
) {
|
|
1306
|
+
super(message)
|
|
1307
|
+
}
|
|
1308
|
+
}
|
|
1309
|
+
|
|
1310
|
+
/** What the guest stopped, and how widely it was allowed to look. */
|
|
1311
|
+
export interface KubernetesQuiesceReport {
|
|
1312
|
+
/** Every process signalled, with the last signal it was actually sent. */
|
|
1313
|
+
readonly stopped: readonly QuiescedProcess[]
|
|
1314
|
+
/** See {@link QuiesceScope}. `owned-sessions` is the narrowed one. */
|
|
1315
|
+
readonly scope: QuiesceScope
|
|
1316
|
+
/** The per-round SIGTERM window the guest used, after its own clamp. */
|
|
1317
|
+
readonly graceMs: number
|
|
1318
|
+
/** Scan-and-signal passes. `0` means nothing was running. */
|
|
1319
|
+
readonly rounds: number
|
|
1320
|
+
}
|
|
1321
|
+
|
|
1322
|
+
function quiesceReport(reply: Record<string, unknown>): KubernetesQuiesceReport {
|
|
1323
|
+
const stopped = Array.isArray(reply.stopped) ? reply.stopped : []
|
|
1324
|
+
return {
|
|
1325
|
+
stopped: stopped.flatMap((entry): QuiescedProcess[] => {
|
|
1326
|
+
if (!entry || typeof entry !== 'object') return []
|
|
1327
|
+
const row = entry as Record<string, unknown>
|
|
1328
|
+
if (typeof row.pid !== 'number') return []
|
|
1329
|
+
return [
|
|
1330
|
+
{
|
|
1331
|
+
pid: row.pid,
|
|
1332
|
+
command: typeof row.command === 'string' ? row.command : '',
|
|
1333
|
+
signal: row.signal === 'SIGKILL' ? 'SIGKILL' : 'SIGTERM',
|
|
1334
|
+
},
|
|
1335
|
+
]
|
|
1336
|
+
}),
|
|
1337
|
+
// The WIDER claim has to be said in so many words. A reply carrying
|
|
1338
|
+
// a scope this host does not recognise reads as the narrow one,
|
|
1339
|
+
// because the only wrong answer here is telling a caller its guest
|
|
1340
|
+
// was swept completely when nothing says so.
|
|
1341
|
+
scope: reply.scope === 'pid-namespace' ? 'pid-namespace' : 'owned-sessions',
|
|
1342
|
+
graceMs: typeof reply.graceMs === 'number' ? reply.graceMs : 0,
|
|
1343
|
+
rounds: typeof reply.rounds === 'number' ? reply.rounds : 0,
|
|
1344
|
+
}
|
|
1345
|
+
}
|
|
1346
|
+
|
|
1347
|
+
// --- flush (#484) ---------------------------------------------------------
|
|
1348
|
+
|
|
1349
|
+
/**
|
|
1350
|
+
* Thrown when a flush was asked of a guest that cannot perform one: either
|
|
1351
|
+
* its `healthz` does not advertise {@link FLUSH_FEATURE}, or it answered
|
|
1352
|
+
* `unknown_op: flush`.
|
|
1353
|
+
*
|
|
1354
|
+
* Refused rather than treated as "the disk is already flushed", for the
|
|
1355
|
+
* same reason {@link KubernetesQuiesceUnsupportedError} is not treated as
|
|
1356
|
+
* "nothing was running". The one caller that does NOT pass this on is
|
|
1357
|
+
* `suspend()`, which goes ahead and tells the host through
|
|
1358
|
+
* `onFlushUnsupported`: an image built before this op is a deployment that
|
|
1359
|
+
* has to be able to suspend its workspaces, not one that has to be stopped.
|
|
1360
|
+
*/
|
|
1361
|
+
export class KubernetesFlushUnsupportedError extends Error {
|
|
1362
|
+
override readonly name = 'KubernetesFlushUnsupportedError'
|
|
1363
|
+
|
|
1364
|
+
constructor(
|
|
1365
|
+
readonly feature: string,
|
|
1366
|
+
message: string,
|
|
1367
|
+
) {
|
|
1368
|
+
super(message)
|
|
1369
|
+
}
|
|
1370
|
+
}
|
|
1371
|
+
|
|
1372
|
+
/**
|
|
1373
|
+
* Thrown when nothing can promise the workspace's writes are on the device.
|
|
1374
|
+
*
|
|
1375
|
+
* The guest ANSWERED and said so — the `syncfs` failed, or ran past its own
|
|
1376
|
+
* timeout — and its message says which. Its own class for the reason every
|
|
1377
|
+
* other answered refusal here has one: this is a fact about the disk, not a
|
|
1378
|
+
* transport failure, and a caller holding it decides whether to retry,
|
|
1379
|
+
* suspend anyway, or leave the workspace running.
|
|
1380
|
+
*/
|
|
1381
|
+
export class KubernetesFlushUnconfirmedError extends Error {
|
|
1382
|
+
override readonly name = 'KubernetesFlushUnconfirmedError'
|
|
1383
|
+
|
|
1384
|
+
constructor(
|
|
1385
|
+
readonly reason: string,
|
|
1386
|
+
message: string,
|
|
1387
|
+
) {
|
|
1388
|
+
super(message)
|
|
1389
|
+
}
|
|
1390
|
+
}
|
|
1391
|
+
|
|
1392
|
+
/**
|
|
1393
|
+
* Carried to {@link KubernetesWorkspaceOptions.onFlushUnreachable} when a
|
|
1394
|
+
* `suspend()` could not ASK for a flush at all — the dial failed, the
|
|
1395
|
+
* connection timed out, the token was refused, or the agent has fenced
|
|
1396
|
+
* itself — and the suspend went ahead without one.
|
|
1397
|
+
*
|
|
1398
|
+
* Never thrown at a caller. The difference between this and
|
|
1399
|
+
* {@link KubernetesFlushUnconfirmedError} is the difference between a guest
|
|
1400
|
+
* that could not be reached and a guest that answered: a guest that
|
|
1401
|
+
* answered is alive, and stopping the suspend gives its caller something to
|
|
1402
|
+
* do about it, while a guest nothing can reach will not become flushable by
|
|
1403
|
+
* leaving its pod running — and refusing to suspend over it would take away
|
|
1404
|
+
* the one verb an operator reaches for when a workspace is wedged, the verb
|
|
1405
|
+
* {@link KubernetesAgentRetiringError} itself names as the way out.
|
|
1406
|
+
*
|
|
1407
|
+
* So the suspend proceeds, the host is told, and the message says what the
|
|
1408
|
+
* disk is resting on instead: whatever the guest kernel had already written
|
|
1409
|
+
* back, plus the pod's own `preStop` hook and the agent's `SIGTERM` handler
|
|
1410
|
+
* if either of them still runs.
|
|
1411
|
+
*/
|
|
1412
|
+
export class KubernetesFlushUnreachableError extends Error {
|
|
1413
|
+
override readonly name = 'KubernetesFlushUnreachableError'
|
|
1414
|
+
}
|
|
1415
|
+
|
|
1416
|
+
/** What a flush did, as the guest measured it. */
|
|
1417
|
+
export interface KubernetesFlushReport {
|
|
1418
|
+
/** How long the `syncfs` took inside the guest. */
|
|
1419
|
+
readonly durationMs: number
|
|
1420
|
+
/** The mount the guest flushed — its workspace root. */
|
|
1421
|
+
readonly workspace: string
|
|
1422
|
+
}
|
|
1423
|
+
|
|
1424
|
+
/**
|
|
1425
|
+
* The refusal a guest that cannot flush earns, built in one place.
|
|
1426
|
+
*
|
|
1427
|
+
* Exported because the WORKSPACE has to be able to hand this exact error to
|
|
1428
|
+
* `onFlushUnsupported` on the one path that does not throw it — a
|
|
1429
|
+
* `suspend()` against an older image — and a second copy of the message
|
|
1430
|
+
* would be a second thing to keep true.
|
|
1431
|
+
*/
|
|
1432
|
+
export function flushUnsupportedError(): KubernetesFlushUnsupportedError {
|
|
1433
|
+
return new KubernetesFlushUnsupportedError(
|
|
1434
|
+
FLUSH_FEATURE,
|
|
1435
|
+
`kubernetes: this workspace's guest agent does not advertise the '${FLUSH_FEATURE}' healthz feature, so nothing here can make its writes reach the disk before the pod stops — what survives is whatever the guest kernel had already written back. The request is refused rather than answered as a flush that happened. Rebuild the workspace image from this Namzu release.`,
|
|
1436
|
+
)
|
|
1437
|
+
}
|
|
1438
|
+
|
|
1439
|
+
function flushReport(reply: Record<string, unknown>): KubernetesFlushReport {
|
|
1440
|
+
return {
|
|
1441
|
+
durationMs: typeof reply.durationMs === 'number' ? reply.durationMs : 0,
|
|
1442
|
+
workspace: typeof reply.workspace === 'string' ? reply.workspace : '',
|
|
1443
|
+
}
|
|
1444
|
+
}
|
|
1445
|
+
|
|
1446
|
+
export class KubernetesAgentTransport {
|
|
1447
|
+
/**
|
|
1448
|
+
* Mutable: the handle follows a replaced pod — see {@link rebind}.
|
|
1449
|
+
*
|
|
1450
|
+
* Under `'pod-ip'` both the address and the token move, because the
|
|
1451
|
+
* address is a literal that died with its pod. Under the default
|
|
1452
|
+
* `'service'` mode the host is a Service FQDN that outlives the pod and
|
|
1453
|
+
* resolves to the replacement on its own, so what moves is the TOKEN
|
|
1454
|
+
* alone — which is the entire reason a `service` handle needed a rebind
|
|
1455
|
+
* at all: the dial keeps working and the guest refuses every call.
|
|
1456
|
+
*/
|
|
1457
|
+
private handle: KubernetesAgentHandle
|
|
1458
|
+
/** The re-read currently in flight, so concurrent refusals share one. */
|
|
1459
|
+
private rebinding?: Promise<boolean>
|
|
1460
|
+
/**
|
|
1461
|
+
* How many times this transport has moved to a different pod. Captured
|
|
1462
|
+
* before every attempt and compared after it fails, so a call refused by
|
|
1463
|
+
* the pod it was bound to — arriving after ANOTHER call's rebind already
|
|
1464
|
+
* installed the replacement — retries on the handle that has moved
|
|
1465
|
+
* instead of re-reading to be told nothing changed.
|
|
1466
|
+
*/
|
|
1467
|
+
private rebindSeq = 0
|
|
1468
|
+
private readonly transportOptions: VsockTransportOptions
|
|
1469
|
+
/** Bound to each wire's own pod by {@link wireOptions}. */
|
|
1470
|
+
private readonly onGuestReply?: (reply: GuestReplyIdentity, podUid: string | undefined) => void
|
|
1471
|
+
private readonly onTiming?: (timing: KubernetesTransportTiming) => void
|
|
1472
|
+
private readonly refreshHandle?: (signal?: AbortSignal) => Promise<KubernetesAgentHandle>
|
|
1473
|
+
/** Simple pass-through operations share one transport instance. */
|
|
1474
|
+
private wire: VsockAgentTransport
|
|
1475
|
+
|
|
1476
|
+
constructor(handle: KubernetesAgentHandle, options: KubernetesTransportOptions = {}) {
|
|
1477
|
+
const { onTiming, refreshHandle, onGuestReply, ...transportOptions } = options
|
|
1478
|
+
this.handle = handle
|
|
1479
|
+
// Held apart from the options the wires are built from: it is the one
|
|
1480
|
+
// hook whose payload depends on WHICH wire read the reply, so
|
|
1481
|
+
// {@link wireOptions} binds it per wire rather than spreading it.
|
|
1482
|
+
this.onGuestReply = onGuestReply
|
|
1483
|
+
this.transportOptions = {
|
|
1484
|
+
...transportOptions,
|
|
1485
|
+
// A NAME the resolver says does not EXIST is the one connect
|
|
1486
|
+
// failure waiting cannot cure, and the reason it must not be
|
|
1487
|
+
// waited on is that the wait destroys the diagnosis. A resolver
|
|
1488
|
+
// that merely could not answer (`EAI_AGAIN`) is not that — see
|
|
1489
|
+
// {@link isMissingNameFailure} — and keeps the whole budget.
|
|
1490
|
+
//
|
|
1491
|
+
// The dial's retry budget is 30 s and the privilege probe's
|
|
1492
|
+
// deadline is at most 15 s, so a host with no cluster resolver
|
|
1493
|
+
// never reaches the wrapper below: the probe's clock expires first
|
|
1494
|
+
// and `create()` rejects saying the guest "accepted the connection
|
|
1495
|
+
// and did not answer", about a connection that was never made. The
|
|
1496
|
+
// address here comes off a Sandbox that reported Ready, so its
|
|
1497
|
+
// Service exists and its record is published — a name that fails
|
|
1498
|
+
// to resolve against that is a fact about THIS host, not a race
|
|
1499
|
+
// with the controller.
|
|
1500
|
+
permanentDialFailure: (err) => this.dialFailureIsPermanent(err),
|
|
1501
|
+
}
|
|
1502
|
+
this.onTiming = onTiming
|
|
1503
|
+
this.refreshHandle = refreshHandle
|
|
1504
|
+
this.wire = new VsockAgentTransport(handle, this.wireOptions(handle))
|
|
1505
|
+
}
|
|
1506
|
+
|
|
1507
|
+
/**
|
|
1508
|
+
* The shared transport's options for ONE wire, with the reply observer
|
|
1509
|
+
* bound to the pod that wire is talking to.
|
|
1510
|
+
*
|
|
1511
|
+
* Every `VsockAgentTransport` this class builds goes through here — the
|
|
1512
|
+
* first one, the one a rebind installs, and the per-attempt one `exec()`
|
|
1513
|
+
* times — because a wire outlives the moment it was current: the pod it
|
|
1514
|
+
* dials can answer a call after a rebind has already moved
|
|
1515
|
+
* {@link handle} on, and the observer has to be told the pod that
|
|
1516
|
+
* ANSWERED rather than the pod this transport now holds. Capturing the
|
|
1517
|
+
* handle in the closure is what makes the two different values.
|
|
1518
|
+
*/
|
|
1519
|
+
private wireOptions(handle: KubernetesAgentHandle): VsockTransportOptions {
|
|
1520
|
+
const observe = this.onGuestReply
|
|
1521
|
+
if (observe === undefined) return this.transportOptions
|
|
1522
|
+
return {
|
|
1523
|
+
...this.transportOptions,
|
|
1524
|
+
onGuestReply: (reply) => {
|
|
1525
|
+
observe(reply, handle.token)
|
|
271
1526
|
},
|
|
272
1527
|
}
|
|
1528
|
+
}
|
|
1529
|
+
|
|
1530
|
+
/**
|
|
1531
|
+
* The address this transport is dialing right now. Diagnostics only —
|
|
1532
|
+
* host and port, and deliberately NOT the handle itself: this is read
|
|
1533
|
+
* into a log line or an assertion, and the bind token has no business
|
|
1534
|
+
* travelling with either.
|
|
1535
|
+
*/
|
|
1536
|
+
get address(): { readonly host: string; readonly port: number } {
|
|
1537
|
+
return { host: this.handle.host, port: this.handle.port }
|
|
1538
|
+
}
|
|
1539
|
+
|
|
1540
|
+
/**
|
|
1541
|
+
* Run one operation, and give the handle exactly one chance to follow a
|
|
1542
|
+
* pod that was replaced underneath it.
|
|
1543
|
+
*
|
|
1544
|
+
* TWO failures reach a rebind, and they are the only two, because they
|
|
1545
|
+
* are the only two that prove the operation ran NOTHING in the guest:
|
|
1546
|
+
*
|
|
1547
|
+
* - **the dial failed.** No socket was ever established, so no byte was
|
|
1548
|
+
* sent. This is the trigger the `'pod-ip'` mode was built on: the
|
|
1549
|
+
* address is a literal that dies with its pod.
|
|
1550
|
+
* - **the guest REFUSED the token** (`unauthorized`, in either of the
|
|
1551
|
+
* shapes {@link isUnauthorizedRefusal} unifies). The agent checks the
|
|
1552
|
+
* token before `dispatch`, so a refused request reached the guest and
|
|
1553
|
+
* was thrown away unread. This is the trigger that matters under the
|
|
1554
|
+
* DEFAULT `'service'` mode, where the Service FQDN outlives the pod
|
|
1555
|
+
* and keeps resolving: the dial SUCCEEDS against the replacement and
|
|
1556
|
+
* the refusal is the only thing that says the pod moved.
|
|
1557
|
+
*
|
|
1558
|
+
* Everything else is the guest's answer to a request it did receive, and
|
|
1559
|
+
* nothing here may reinterpret it — not as a resolver problem, and not as
|
|
1560
|
+
* a reason to repeat work the guest has already begun.
|
|
1561
|
+
*
|
|
1562
|
+
* From there both arms run the SAME routine, which is the whole of this
|
|
1563
|
+
* change to it: {@link rebind} reads the pod once, and a DIFFERENT uid
|
|
1564
|
+
* means the controller replaced it (a resume, an eviction, a node drain),
|
|
1565
|
+
* so the handle takes the new address AND the new token together and the
|
|
1566
|
+
* operation is retried once.
|
|
1567
|
+
*
|
|
1568
|
+
* One branch belongs to the dial arm alone, and is taken before the
|
|
1569
|
+
* re-read: the handle's host is a NAME and the dial gave up at
|
|
1570
|
+
* resolution, so this is a Service FQDN and this host has no resolver for
|
|
1571
|
+
* it. Re-reading the pod would change nothing — the next dial would ask
|
|
1572
|
+
* the same resolver the same question — so the error is replaced with one
|
|
1573
|
+
* that names the FQDN and the configuration field that fixes it. It is
|
|
1574
|
+
* never asked of a refusal, which came back over a connection that
|
|
1575
|
+
* plainly worked, and the name check is not decoration either: a
|
|
1576
|
+
* `pod-ip` handle must keep its one re-read however an unrelated error
|
|
1577
|
+
* happens to be worded.
|
|
1578
|
+
*
|
|
1579
|
+
* Anything else — a re-read that finds the SAME pod, or one that fails —
|
|
1580
|
+
* leaves the original error standing. A pod that is still there and still
|
|
1581
|
+
* refusing is a guest problem, and replacing that error with a second,
|
|
1582
|
+
* later one would hide it. The one exception is a re-read that comes back
|
|
1583
|
+
* with a verdict about the OBJECT rather than a failed diagnosis
|
|
1584
|
+
* ({@link KubernetesWorkspaceReplacedError}): the disk behind the name is
|
|
1585
|
+
* not this handle's disk, which outranks whatever uncovered it.
|
|
1586
|
+
*
|
|
1587
|
+
* `signal` is the caller's own, and it decides one thing only: a call that
|
|
1588
|
+
* has ALREADY been cancelled is not owed a re-read, because the retry it
|
|
1589
|
+
* would buy would abort before it left. It is never handed to the re-read
|
|
1590
|
+
* itself, which is shared and runs under a bound of its own — see
|
|
1591
|
+
* {@link rebind}.
|
|
1592
|
+
*
|
|
1593
|
+
* `dials` is how `exec()` answers the first question at all. Its failure
|
|
1594
|
+
* can arrive as a bare timeout from the execution controller's own bound,
|
|
1595
|
+
* with the dial's error discarded rather than wrapped, so that path
|
|
1596
|
+
* watches its dials instead of reading its error — see {@link DialWatch}.
|
|
1597
|
+
* Every other operation hands back the dial's own error and passes none.
|
|
1598
|
+
*
|
|
1599
|
+
* A refusal that no rebind could fix leaves as {@link
|
|
1600
|
+
* KubernetesAgentUnauthorizedError} whichever shape it arrived in — the
|
|
1601
|
+
* unification the whole recovery hangs off, applied at the one point
|
|
1602
|
+
* where "nothing else can be done about it" is known.
|
|
1603
|
+
*/
|
|
1604
|
+
private async withRebind<T>(
|
|
1605
|
+
run: () => Promise<T>,
|
|
1606
|
+
signal?: AbortSignal,
|
|
1607
|
+
dials?: DialWatch,
|
|
1608
|
+
): Promise<T> {
|
|
1609
|
+
// Read BEFORE the attempt: everything below asks whether the handle
|
|
1610
|
+
// has moved SINCE this call was dispatched.
|
|
1611
|
+
const boundAt = this.rebindSeq
|
|
1612
|
+
try {
|
|
1613
|
+
return await run()
|
|
1614
|
+
} catch (err) {
|
|
1615
|
+
if (isUnretryableOutcome(err)) throw err
|
|
1616
|
+
const refused = isUnauthorizedRefusal(err)
|
|
1617
|
+
// A refusal came back over a connection that was made, so it is
|
|
1618
|
+
// never also a dial failure; asking both questions of one error
|
|
1619
|
+
// and letting the dial arm win would send it to the resolver
|
|
1620
|
+
// branch, which is about a connection that was never attempted.
|
|
1621
|
+
const fromTheDial = !refused && (isConnectFailure(err) || neverConnected(dials))
|
|
1622
|
+
if (!refused && !fromTheDial) throw err
|
|
1623
|
+
if (fromTheDial && this.dialsAName() && this.failedToResolve(err, dials)) {
|
|
1624
|
+
throw this.unresolvable(err)
|
|
1625
|
+
}
|
|
1626
|
+
// See above: nothing to buy with a re-read here, and the shared
|
|
1627
|
+
// one must not be started on behalf of a call that is gone.
|
|
1628
|
+
if (signal?.aborted === true) throw refused ? asUnauthorized(err) : err
|
|
1629
|
+
if (!(await this.rebind(boundAt))) throw refused ? asUnauthorized(err) : err
|
|
1630
|
+
try {
|
|
1631
|
+
return await run()
|
|
1632
|
+
} catch (retryErr) {
|
|
1633
|
+
// The retry is the last attempt either way; a second refusal
|
|
1634
|
+
// (the replacement is refusing too) still leaves as one shape.
|
|
1635
|
+
throw refused && isUnauthorizedRefusal(retryErr) ? asUnauthorized(retryErr) : retryErr
|
|
1636
|
+
}
|
|
1637
|
+
}
|
|
1638
|
+
}
|
|
1639
|
+
|
|
1640
|
+
/**
|
|
1641
|
+
* Whether the dial gave up at name resolution.
|
|
1642
|
+
*
|
|
1643
|
+
* Asked of the caller's error first, and of the watched dial's own error
|
|
1644
|
+
* only when nothing ever connected — the case where the error the caller
|
|
1645
|
+
* holds is a bound's timer rather than the failure that caused it. A
|
|
1646
|
+
* watched attempt that DID connect is never consulted: its error is the
|
|
1647
|
+
* guest's answer, however it happens to be worded.
|
|
1648
|
+
*/
|
|
1649
|
+
private failedToResolve(error: unknown, dials: DialWatch | undefined): boolean {
|
|
1650
|
+
if (isNameResolutionFailure(error)) return true
|
|
1651
|
+
return neverConnected(dials) && isNameResolutionFailure(dials?.lastError)
|
|
1652
|
+
}
|
|
1653
|
+
|
|
1654
|
+
/**
|
|
1655
|
+
* The one connect failure waiting cannot cure — see the constructor.
|
|
1656
|
+
* A method rather than a closure because the watched `exec()` dials wrap
|
|
1657
|
+
* it, and a caller-visible answer must not depend on which wire asked.
|
|
1658
|
+
*/
|
|
1659
|
+
private dialFailureIsPermanent(error: unknown): boolean {
|
|
1660
|
+
return this.dialsAName() && isMissingNameFailure(error)
|
|
1661
|
+
}
|
|
1662
|
+
|
|
1663
|
+
/**
|
|
1664
|
+
* One re-read, shared by every call that needs one at the same moment.
|
|
1665
|
+
* True when the handle now points at a DIFFERENT pod than the one the
|
|
1666
|
+
* caller's attempt was dispatched on — whether this call's own re-read
|
|
1667
|
+
* moved it or another call's already had.
|
|
1668
|
+
*
|
|
1669
|
+
* Single-flight, and that is not an optimisation. A pod replaced
|
|
1670
|
+
* underneath a busy handle refuses EVERY call in flight at once, and one
|
|
1671
|
+
* re-read per refused call would be a burst of Sandbox and pod GETs, each
|
|
1672
|
+
* one racing the others to install a handle — with the last to finish
|
|
1673
|
+
* winning, which is not necessarily the last to read. Sharing the promise
|
|
1674
|
+
* makes the burst one round trip and one installation, in the order the
|
|
1675
|
+
* API answered.
|
|
1676
|
+
*
|
|
1677
|
+
* The slot is cleared inside the shared run, before the promise settles,
|
|
1678
|
+
* so a caller that awaits it and is refused AGAIN gets a fresh re-read
|
|
1679
|
+
* rather than the answer to the previous question.
|
|
1680
|
+
*
|
|
1681
|
+
* Because it is shared it runs under {@link REBIND_READ_TIMEOUT_MS} and
|
|
1682
|
+
* under NO caller's signal. A caller that aborts while the shared read is
|
|
1683
|
+
* in flight would otherwise abort it for everyone, and every call the
|
|
1684
|
+
* replacement refused would fail with its original refusal although the
|
|
1685
|
+
* handle could have followed the pod. Its own bound is what the aborting
|
|
1686
|
+
* caller was owed — nobody waits on a re-read for longer than that — and
|
|
1687
|
+
* the read is two GETs nobody is billed for twice.
|
|
1688
|
+
*/
|
|
1689
|
+
private async rebind(boundAt?: number): Promise<boolean> {
|
|
1690
|
+
// Another call already followed the replacement while this one was in
|
|
1691
|
+
// flight, so the pod that refused this call is the pod it was bound
|
|
1692
|
+
// to and the answer a re-read would give is already installed. Asking
|
|
1693
|
+
// again would compare the new token against itself, conclude nothing
|
|
1694
|
+
// moved, and fail a call the handle can now serve.
|
|
1695
|
+
if (boundAt !== undefined && this.rebindSeq !== boundAt) return true
|
|
1696
|
+
const refresh = this.refreshHandle
|
|
1697
|
+
if (refresh === undefined) return false
|
|
1698
|
+
this.rebinding ??= this.runRebind(refresh, AbortSignal.timeout(REBIND_READ_TIMEOUT_MS))
|
|
1699
|
+
return await this.rebinding
|
|
1700
|
+
}
|
|
273
1701
|
|
|
274
|
-
|
|
1702
|
+
private async runRebind(
|
|
1703
|
+
refresh: (signal?: AbortSignal) => Promise<KubernetesAgentHandle>,
|
|
1704
|
+
signal?: AbortSignal,
|
|
1705
|
+
): Promise<boolean> {
|
|
275
1706
|
try {
|
|
276
|
-
|
|
1707
|
+
let next: KubernetesAgentHandle
|
|
1708
|
+
try {
|
|
1709
|
+
next = await refresh(signal)
|
|
1710
|
+
} catch (err) {
|
|
1711
|
+
// A verdict about the OBJECT is not a failed diagnosis and
|
|
1712
|
+
// must not be swallowed: it says the disk behind the name is
|
|
1713
|
+
// not this handle's disk, which is a far more important thing
|
|
1714
|
+
// to report than the refusal that uncovered it — and it is the
|
|
1715
|
+
// one answer that must never be followed by a retry.
|
|
1716
|
+
if (err instanceof KubernetesWorkspaceReplacedError) throw err
|
|
1717
|
+
// Everything else: the re-read is a diagnosis, not an
|
|
1718
|
+
// operation. A pod that cannot be read is not a better error
|
|
1719
|
+
// than the failure the caller is already holding.
|
|
1720
|
+
return false
|
|
1721
|
+
}
|
|
1722
|
+
if (next.token === this.handle.token) return false
|
|
1723
|
+
this.rebindSeq += 1
|
|
1724
|
+
this.handle = next
|
|
1725
|
+
this.wire = new VsockAgentTransport(next, this.wireOptions(next))
|
|
1726
|
+
return true
|
|
1727
|
+
} finally {
|
|
1728
|
+
this.rebinding = undefined
|
|
1729
|
+
}
|
|
1730
|
+
}
|
|
1731
|
+
|
|
1732
|
+
/** Whether the current handle's host goes through a resolver at all. */
|
|
1733
|
+
private dialsAName(): boolean {
|
|
1734
|
+
return net.isIP(this.handle.host) === 0
|
|
1735
|
+
}
|
|
1736
|
+
|
|
1737
|
+
/** The DNS-shaped failure, in words that name the way out of it. */
|
|
1738
|
+
private unresolvable(cause: unknown): Error {
|
|
1739
|
+
const host = this.handle.host
|
|
1740
|
+
return new KubernetesAgentAddressUnresolvableError(
|
|
1741
|
+
host,
|
|
1742
|
+
`kubernetes: the guest agent's address ${host}:${this.handle.port} did not resolve (ENOTFOUND/EAI_AGAIN), so no connection was attempted. That is a Kubernetes Service FQDN and only the cluster's own DNS answers it: a host running OUTSIDE the cluster — a VNet peer, a CI runner, a laptop — fails every call here, readiness probes included, and the symptom looks like a sandbox that never came up. Set agentAddress: 'pod-ip' on the kubernetes backend config to dial the bound pod's IP instead, which needs a pod network routable from this host and a NetworkPolicy admitting its address range.`,
|
|
1743
|
+
{ cause },
|
|
1744
|
+
)
|
|
1745
|
+
}
|
|
1746
|
+
|
|
1747
|
+
/**
|
|
1748
|
+
* Readiness probe — never requires a token; see `protocol.ts`.
|
|
1749
|
+
*
|
|
1750
|
+
* Deliberately NOT wrapped in {@link withRebind}: `healthz` answers a
|
|
1751
|
+
* failed dial with `false` rather than by throwing, so there is no error
|
|
1752
|
+
* to classify and nothing for a re-read to be triggered by. A caller that
|
|
1753
|
+
* wants the reason asks for it by making a real call.
|
|
1754
|
+
*/
|
|
1755
|
+
async healthz(signal?: AbortSignal): Promise<boolean> {
|
|
1756
|
+
return await this.wire.healthz(signal)
|
|
1757
|
+
}
|
|
1758
|
+
|
|
1759
|
+
/**
|
|
1760
|
+
* Poll until `healthz` succeeds or the timeout elapses. Unwrapped for the
|
|
1761
|
+
* same reason, and for one more: it already owns a retry loop, so a
|
|
1762
|
+
* connect failure here is not a single failed dial but a whole budget of
|
|
1763
|
+
* them.
|
|
1764
|
+
*/
|
|
1765
|
+
async waitForReady(
|
|
1766
|
+
timeoutMs: number,
|
|
1767
|
+
pollIntervalMs: number,
|
|
1768
|
+
signal?: AbortSignal,
|
|
1769
|
+
): Promise<void> {
|
|
1770
|
+
return await this.wire.waitForReady(timeoutMs, pollIntervalMs, signal)
|
|
1771
|
+
}
|
|
1772
|
+
|
|
1773
|
+
/**
|
|
1774
|
+
* Ask the agent how it is, and keep the two "not ok" answers apart —
|
|
1775
|
+
* see {@link KubernetesAgentHealth}.
|
|
1776
|
+
*
|
|
1777
|
+
* Deliberately NOT wrapped in {@link withRebind}: this is a diagnostic
|
|
1778
|
+
* about the pod this handle is bound to RIGHT NOW, and a rebind would
|
|
1779
|
+
* silently answer it about a different pod. A caller that wants to know
|
|
1780
|
+
* whether the agent it was talking to has fenced itself would then be
|
|
1781
|
+
* told about the replacement, which is a different question with a
|
|
1782
|
+
* different answer.
|
|
1783
|
+
*
|
|
1784
|
+
* It throws whatever the dial or the read threw. An agent that cannot be
|
|
1785
|
+
* reached has no health to report, and inventing one here would turn
|
|
1786
|
+
* "unreachable" into "fine".
|
|
1787
|
+
*/
|
|
1788
|
+
async agentHealth(signal?: AbortSignal): Promise<KubernetesAgentHealth> {
|
|
1789
|
+
const reply = await this.wire.request<{ ok?: unknown; retiring?: unknown }>(
|
|
1790
|
+
{ op: 'healthz' },
|
|
1791
|
+
signal,
|
|
1792
|
+
)
|
|
1793
|
+
return { ok: reply?.ok === true, retiring: reply?.retiring === true }
|
|
1794
|
+
}
|
|
1795
|
+
|
|
1796
|
+
/**
|
|
1797
|
+
* The raw `reserve-execution` primitive, exposed directly (rather than
|
|
1798
|
+
* only reachable as a side effect of `exec()`) so the reservation
|
|
1799
|
+
* round trip is independently observable and testable against the
|
|
1800
|
+
* real guest.
|
|
1801
|
+
*/
|
|
1802
|
+
async reserve(signal?: AbortSignal): Promise<unknown> {
|
|
1803
|
+
return await this.withRebind(
|
|
1804
|
+
async () => await requestChecked(this.wire, { op: 'reserve-execution' }, signal),
|
|
1805
|
+
signal,
|
|
1806
|
+
)
|
|
1807
|
+
}
|
|
1808
|
+
|
|
1809
|
+
/**
|
|
1810
|
+
* The raw `cancel-execution` primitive, exposed for the same reason
|
|
1811
|
+
* {@link reserve} is: it is a control request with its own refusal
|
|
1812
|
+
* semantics, and proving those against the real guest should not require
|
|
1813
|
+
* driving a whole cancelled `exec()` to reach it.
|
|
1814
|
+
*/
|
|
1815
|
+
async cancel(executionId: string, signal?: AbortSignal): Promise<unknown> {
|
|
1816
|
+
return await this.withRebind(
|
|
1817
|
+
async () =>
|
|
1818
|
+
await requestChecked(this.wire, { op: 'cancel-execution', body: { executionId } }, signal),
|
|
1819
|
+
signal,
|
|
1820
|
+
)
|
|
1821
|
+
}
|
|
1822
|
+
|
|
1823
|
+
/**
|
|
1824
|
+
* Delegated, `signal` included: a body larger than one pre-auth frame
|
|
1825
|
+
* is written as a sequence of parts by {@link VsockAgentTransport}
|
|
1826
|
+
* itself, and a cancelled sequence has to be able to stop mid-way and
|
|
1827
|
+
* take its temp file with it.
|
|
1828
|
+
*
|
|
1829
|
+
* One transport instance, deliberately — {@link wire} is shared by
|
|
1830
|
+
* every simple pass-through op — so the one `healthz` probe that asks
|
|
1831
|
+
* the guest whether it can take parts is asked once for this sandbox,
|
|
1832
|
+
* not once per large write.
|
|
1833
|
+
*
|
|
1834
|
+
* Wrapped in {@link withRebind} like every other guest-dialling op,
|
|
1835
|
+
* the multi-part route included, and for both of its triggers: a retry
|
|
1836
|
+
* starts a fresh sequence under a new temp name
|
|
1837
|
+
* (`writeFilePartTempPath` mints a UUID per call), so a sequence
|
|
1838
|
+
* abandoned mid-way on the replaced pod — because the dial failed, or
|
|
1839
|
+
* because the replacement refused this handle's token before reading a
|
|
1840
|
+
* byte — cannot collide with the retry's offsets and never touched the
|
|
1841
|
+
* target. At worst it leaves one orphan temp file behind on the
|
|
1842
|
+
* workspace volume.
|
|
1843
|
+
*/
|
|
1844
|
+
async writeFile(path: string, content: Buffer, signal?: AbortSignal): Promise<void> {
|
|
1845
|
+
return await this.withRebind(
|
|
1846
|
+
async () => await this.wire.writeFile(path, content, signal),
|
|
1847
|
+
signal,
|
|
1848
|
+
)
|
|
1849
|
+
}
|
|
1850
|
+
|
|
1851
|
+
async readFile(path: string, options?: SandboxReadFileOptions): Promise<Buffer> {
|
|
1852
|
+
return await this.withRebind(
|
|
1853
|
+
async () => await this.wire.readFile(path, options),
|
|
1854
|
+
options?.signal,
|
|
1855
|
+
)
|
|
1856
|
+
}
|
|
1857
|
+
|
|
1858
|
+
/**
|
|
1859
|
+
* Delegated, and rebound exactly once — but only around the FIRST
|
|
1860
|
+
* chunk.
|
|
1861
|
+
*
|
|
1862
|
+
* That is the whole of what {@link withRebind} can honestly cover here.
|
|
1863
|
+
* Its retry is safe because nothing reached the guest, and once a chunk
|
|
1864
|
+
* has been yielded that is no longer true: re-dialing a replaced pod
|
|
1865
|
+
* mid-stream would restart the file from its beginning, and the
|
|
1866
|
+
* consumer — which has already taken the bytes and cannot give them
|
|
1867
|
+
* back — would silently concatenate a duplicate prefix. So the first
|
|
1868
|
+
* pull carries the dial, the rebind and the retry; everything after it
|
|
1869
|
+
* fails as itself.
|
|
1870
|
+
*/
|
|
1871
|
+
async *readFileStream(
|
|
1872
|
+
path: string,
|
|
1873
|
+
options?: SandboxReadFileOptions,
|
|
1874
|
+
): AsyncGenerator<Buffer, void, undefined> {
|
|
1875
|
+
const started = await this.withRebind(async () => {
|
|
1876
|
+
const iterator = this.wire.readFileStream(path, options)[Symbol.asyncIterator]()
|
|
1877
|
+
try {
|
|
1878
|
+
return { iterator, first: await iterator.next() }
|
|
1879
|
+
} catch (error) {
|
|
1880
|
+
// The abandoned generator's own `finally` has already run by
|
|
1881
|
+
// the time its `next()` rejects, so the socket is down; the
|
|
1882
|
+
// `return()` is belt and braces for an implementation that
|
|
1883
|
+
// rejected without finishing.
|
|
1884
|
+
await iterator.return?.(undefined).catch(() => undefined)
|
|
1885
|
+
throw error
|
|
1886
|
+
}
|
|
1887
|
+
}, options?.signal)
|
|
1888
|
+
const { iterator, first } = started
|
|
1889
|
+
try {
|
|
1890
|
+
if (first.done === true) return
|
|
1891
|
+
yield first.value
|
|
1892
|
+
for (;;) {
|
|
1893
|
+
const next = await iterator.next()
|
|
1894
|
+
if (next.done === true) return
|
|
1895
|
+
yield next.value
|
|
1896
|
+
}
|
|
1897
|
+
} finally {
|
|
1898
|
+
await iterator.return?.(undefined).catch(() => undefined)
|
|
1899
|
+
}
|
|
1900
|
+
}
|
|
1901
|
+
|
|
1902
|
+
/**
|
|
1903
|
+
* A guest PTY. Without `sessionId`/`persistent` this is exactly the
|
|
1904
|
+
* terminal it has always been, down to the wire request.
|
|
1905
|
+
*
|
|
1906
|
+
* With them the PTY belongs to the guest's session registry: losing this
|
|
1907
|
+
* connection detaches rather than killing, a later process rejoins it
|
|
1908
|
+
* with {@link attachSession}, and the capability is verified against the
|
|
1909
|
+
* guest's `healthz` features BEFORE the shell is started — never
|
|
1910
|
+
* downgraded to a connection-bound terminal, which would look like it
|
|
1911
|
+
* worked until the rollout it exists for.
|
|
1912
|
+
*/
|
|
1913
|
+
async openTerminal(options: KubernetesOpenTerminalOptions): Promise<KubernetesWorkspaceTerminal> {
|
|
1914
|
+
if (options.sessionId === undefined && options.persistent !== true) {
|
|
1915
|
+
return await this.withRebind(async () => await this.wire.openTerminal(options))
|
|
1916
|
+
}
|
|
1917
|
+
const sessionId = assertSessionOpen(options)
|
|
1918
|
+
await this.assertSessionsSupported()
|
|
1919
|
+
return await this.sessionStream(
|
|
1920
|
+
sessionId,
|
|
1921
|
+
'openTerminal',
|
|
1922
|
+
async () => await this.wire.openSessionTerminal(options),
|
|
1923
|
+
)
|
|
1924
|
+
}
|
|
1925
|
+
|
|
1926
|
+
/**
|
|
1927
|
+
* One open of a session stream, with the guest's refusal mapped to
|
|
1928
|
+
* {@link KubernetesSessionRefusedError} — see {@link sessionStreamFailure}.
|
|
1929
|
+
*/
|
|
1930
|
+
private async sessionStream(
|
|
1931
|
+
sessionId: string,
|
|
1932
|
+
operation: string,
|
|
1933
|
+
open: () => Promise<AgentTerminalStream>,
|
|
1934
|
+
): Promise<KubernetesSessionTerminal> {
|
|
1935
|
+
try {
|
|
1936
|
+
return sessionTerminal(await this.withRebind(open), sessionId)
|
|
1937
|
+
} catch (error) {
|
|
1938
|
+
throw sessionStreamFailure(sessionId, operation, error)
|
|
1939
|
+
}
|
|
1940
|
+
}
|
|
1941
|
+
|
|
1942
|
+
/**
|
|
1943
|
+
* Rejoin a terminal session, replaying from `fromOffset` and then
|
|
1944
|
+
* following it live.
|
|
1945
|
+
*
|
|
1946
|
+
* The guest allows one attachment per session and ends the previous one
|
|
1947
|
+
* by name, so two host processes cannot interleave keystrokes into one
|
|
1948
|
+
* shell. Losing this connection detaches; ending the program is
|
|
1949
|
+
* {@link killSession} and nothing else.
|
|
1950
|
+
*/
|
|
1951
|
+
async attachSession(
|
|
1952
|
+
sessionId: string,
|
|
1953
|
+
options: KubernetesAttachTerminalOptions = {},
|
|
1954
|
+
): Promise<KubernetesSessionTerminal> {
|
|
1955
|
+
assertSessionId(sessionId)
|
|
1956
|
+
await this.assertSessionsSupported()
|
|
1957
|
+
return await this.sessionStream(
|
|
1958
|
+
sessionId,
|
|
1959
|
+
'attachSession',
|
|
1960
|
+
async () =>
|
|
1961
|
+
await this.wire.attachSessionTerminal({
|
|
1962
|
+
sessionId,
|
|
1963
|
+
...(options.fromOffset !== undefined ? { fromOffset: options.fromOffset } : {}),
|
|
1964
|
+
...(options.size !== undefined
|
|
1965
|
+
? { cols: options.size.cols, rows: options.size.rows }
|
|
1966
|
+
: {}),
|
|
1967
|
+
}),
|
|
1968
|
+
)
|
|
1969
|
+
}
|
|
1970
|
+
|
|
1971
|
+
/**
|
|
1972
|
+
* Start a program with no terminal at all, in its own kernel session,
|
|
1973
|
+
* with stdin closed and both output streams going into the guest's
|
|
1974
|
+
* retained log.
|
|
1975
|
+
*
|
|
1976
|
+
* It is not the SDK's `spawnDetached` and deliberately does not pretend
|
|
1977
|
+
* to be: that one hands back a host `ChildProcess`, which cannot cross a
|
|
1978
|
+
* process boundary. This returns a NAME, and the name is what a
|
|
1979
|
+
* redeployed host comes back with.
|
|
1980
|
+
*/
|
|
1981
|
+
async startDetached(
|
|
1982
|
+
options: KubernetesStartDetachedOptions,
|
|
1983
|
+
signal?: AbortSignal,
|
|
1984
|
+
): Promise<KubernetesSessionSummary> {
|
|
1985
|
+
assertSessionId(options.sessionId)
|
|
1986
|
+
await this.assertSessionsSupported(signal)
|
|
1987
|
+
const reply = (await this.withRebind(
|
|
1988
|
+
async () =>
|
|
1989
|
+
await requestChecked(
|
|
1990
|
+
this.wire,
|
|
1991
|
+
{
|
|
1992
|
+
op: 'start-detached',
|
|
1993
|
+
body: {
|
|
1994
|
+
sessionId: options.sessionId,
|
|
1995
|
+
command: options.command,
|
|
1996
|
+
...(options.args !== undefined ? { args: options.args } : {}),
|
|
1997
|
+
...(options.cwd !== undefined ? { cwd: options.cwd } : {}),
|
|
1998
|
+
...(options.env !== undefined ? { env: { ...options.env } } : {}),
|
|
1999
|
+
},
|
|
2000
|
+
},
|
|
2001
|
+
signal,
|
|
2002
|
+
),
|
|
2003
|
+
signal,
|
|
2004
|
+
)) as Record<string, unknown>
|
|
2005
|
+
if (reply.ok !== true) throw sessionRefusal(options.sessionId, reply, 'startDetached')
|
|
2006
|
+
return parseSessionSummary(reply)
|
|
2007
|
+
}
|
|
2008
|
+
|
|
2009
|
+
/** Every session this pod's agent is holding, running and recently exited. */
|
|
2010
|
+
async listSessions(signal?: AbortSignal): Promise<readonly KubernetesSessionSummary[]> {
|
|
2011
|
+
await this.assertSessionsSupported(signal)
|
|
2012
|
+
const reply = (await this.withRebind(
|
|
2013
|
+
async () => await requestChecked(this.wire, { op: 'list-sessions' }, signal),
|
|
2014
|
+
signal,
|
|
2015
|
+
)) as Record<string, unknown>
|
|
2016
|
+
if (reply.ok !== true) {
|
|
2017
|
+
throw new RemoteProtocolError(
|
|
2018
|
+
`kubernetes: the guest refused to list sessions: ${String(reply.error ?? 'no reason given')}`,
|
|
2019
|
+
)
|
|
2020
|
+
}
|
|
2021
|
+
if (!Array.isArray(reply.sessions)) {
|
|
2022
|
+
throw new RemoteProtocolError(
|
|
2023
|
+
'kubernetes: the guest sent a session list that is not an array',
|
|
2024
|
+
)
|
|
2025
|
+
}
|
|
2026
|
+
return reply.sessions.map(parseSessionSummary)
|
|
2027
|
+
}
|
|
2028
|
+
|
|
2029
|
+
/**
|
|
2030
|
+
* End one session and everything still in it — the shell, its
|
|
2031
|
+
* backgrounded jobs, and the program a detached session started.
|
|
2032
|
+
*
|
|
2033
|
+
* Idempotent: a session that has already exited answers with what it
|
|
2034
|
+
* exited with. The reply carries the session's state, so a program that
|
|
2035
|
+
* ignored a `SIGTERM` and outlived the guest's confirm window is
|
|
2036
|
+
* reported still running rather than reported dead.
|
|
2037
|
+
*/
|
|
2038
|
+
async killSession(
|
|
2039
|
+
sessionId: string,
|
|
2040
|
+
options: { readonly signal?: string; readonly abort?: AbortSignal } = {},
|
|
2041
|
+
): Promise<KubernetesSessionSummary> {
|
|
2042
|
+
assertSessionId(sessionId)
|
|
2043
|
+
await this.assertSessionsSupported(options.abort)
|
|
2044
|
+
const reply = (await this.withRebind(
|
|
2045
|
+
async () =>
|
|
2046
|
+
await requestChecked(
|
|
2047
|
+
this.wire,
|
|
2048
|
+
{
|
|
2049
|
+
op: 'kill-session',
|
|
2050
|
+
body: {
|
|
2051
|
+
sessionId,
|
|
2052
|
+
...(options.signal !== undefined ? { signal: options.signal } : {}),
|
|
2053
|
+
},
|
|
2054
|
+
},
|
|
2055
|
+
options.abort,
|
|
2056
|
+
),
|
|
2057
|
+
options.abort,
|
|
2058
|
+
)) as Record<string, unknown>
|
|
2059
|
+
if (reply.ok !== true) throw sessionRefusal(sessionId, reply, 'killSession')
|
|
2060
|
+
return parseSessionSummary(reply)
|
|
2061
|
+
}
|
|
2062
|
+
|
|
2063
|
+
/**
|
|
2064
|
+
* Read what a session has printed since `fromOffset`, without attaching
|
|
2065
|
+
* to it and without signalling anything.
|
|
2066
|
+
*
|
|
2067
|
+
* One request, one answer, in the SDK's `BackgroundJobOutput` shape: the
|
|
2068
|
+
* chunk, the offset to come back with, the bytes the ring dropped before
|
|
2069
|
+
* it, and the program's status. A caller polling in a loop can neither
|
|
2070
|
+
* re-read nor skip, because the offset is the guest's own.
|
|
2071
|
+
*/
|
|
2072
|
+
async readSession(
|
|
2073
|
+
sessionId: string,
|
|
2074
|
+
options: KubernetesReadSessionOptions = {},
|
|
2075
|
+
signal?: AbortSignal,
|
|
2076
|
+
): Promise<KubernetesSessionOutput> {
|
|
2077
|
+
assertSessionId(sessionId)
|
|
2078
|
+
await this.assertSessionsSupported(signal)
|
|
2079
|
+
const fromOffset = options.fromOffset ?? 0
|
|
2080
|
+
let chunk = ''
|
|
2081
|
+
let nextOffset = fromOffset
|
|
2082
|
+
let droppedBytes = 0
|
|
2083
|
+
let state: SessionState = 'running'
|
|
2084
|
+
let exitCode: number | undefined
|
|
2085
|
+
let exitSignal: number | undefined
|
|
2086
|
+
let refusal: KubernetesSessionRefusedError | undefined
|
|
2087
|
+
// Accumulating INSIDE a `withRebind` is safe for the one reason that
|
|
2088
|
+
// matters: the wrapper retries only a failure that delivered no
|
|
2089
|
+
// session frame. A dial that failed sent nothing at all, and a token
|
|
2090
|
+
// the guest REFUSED is answered before `dispatch` with that refusal
|
|
2091
|
+
// as the connection's first and only frame — which the handler above
|
|
2092
|
+
// turns into an error rather than accumulating. Either way a retry
|
|
2093
|
+
// starts from an untouched chunk and the offset the caller asked
|
|
2094
|
+
// for.
|
|
2095
|
+
await this.withRebind(
|
|
2096
|
+
async () =>
|
|
2097
|
+
await this.wire.streamFramedRequest(
|
|
2098
|
+
{ op: 'attach-session', body: { sessionId, fromOffset, follow: false } },
|
|
2099
|
+
(event) => {
|
|
2100
|
+
if (isUnauthorized(event)) throw new KubernetesAgentUnauthorizedError()
|
|
2101
|
+
if (event.type === 'ready') {
|
|
2102
|
+
nextOffset = sessionNumber(event.nextOffset, nextOffset)
|
|
2103
|
+
droppedBytes = sessionNumber(event.droppedBytes)
|
|
2104
|
+
state = event.state === 'exited' ? 'exited' : 'running'
|
|
2105
|
+
if (typeof event.exitCode === 'number') exitCode = event.exitCode
|
|
2106
|
+
if (typeof event.signal === 'number') exitSignal = event.signal
|
|
2107
|
+
return
|
|
2108
|
+
}
|
|
2109
|
+
if (event.type === 'data') {
|
|
2110
|
+
chunk += String(event.data ?? '')
|
|
2111
|
+
nextOffset = sessionNumber(event.nextOffset, nextOffset)
|
|
2112
|
+
return
|
|
2113
|
+
}
|
|
2114
|
+
if (event.type === 'error') {
|
|
2115
|
+
refusal = sessionRefusal(
|
|
2116
|
+
sessionId,
|
|
2117
|
+
{
|
|
2118
|
+
error: event.error,
|
|
2119
|
+
...(event.message !== undefined ? { message: event.message } : {}),
|
|
2120
|
+
},
|
|
2121
|
+
'readSession',
|
|
2122
|
+
)
|
|
2123
|
+
return
|
|
2124
|
+
}
|
|
2125
|
+
throw new RemoteProtocolError(
|
|
2126
|
+
`kubernetes: the guest sent an unexpected frame reading session ${sessionId}: ${JSON.stringify(event).slice(0, 200)}`,
|
|
2127
|
+
)
|
|
2128
|
+
},
|
|
2129
|
+
{ observationTimeoutMs: SESSION_READ_TIMEOUT_MS },
|
|
2130
|
+
signal,
|
|
2131
|
+
),
|
|
2132
|
+
signal,
|
|
2133
|
+
)
|
|
2134
|
+
if (refusal !== undefined) throw refusal
|
|
2135
|
+
return {
|
|
2136
|
+
chunk,
|
|
2137
|
+
nextOffset,
|
|
2138
|
+
droppedBytes,
|
|
2139
|
+
status: sessionStatus({ state, ...(exitSignal !== undefined ? { signal: exitSignal } : {}) }),
|
|
2140
|
+
...(exitCode !== undefined ? { exitCode } : {}),
|
|
2141
|
+
}
|
|
2142
|
+
}
|
|
2143
|
+
|
|
2144
|
+
/**
|
|
2145
|
+
* Stop every process this pod's guest is running, and keep the agent.
|
|
2146
|
+
*
|
|
2147
|
+
* The point is the pair. Stopping the POD stops its processes too, but
|
|
2148
|
+
* leaves nothing to read the disk through, so a host that wants a capture
|
|
2149
|
+
* it can trust has to wake the workspace again and check. After this the
|
|
2150
|
+
* guest is quiet and still serving: `exec`, `readFile` and `writeFile`
|
|
2151
|
+
* all work, and what they see is a filesystem nobody is writing to.
|
|
2152
|
+
*
|
|
2153
|
+
* Open terminals and running commands end as a side effect and report it
|
|
2154
|
+
* through their own exit and result paths — a terminal receives its exit,
|
|
2155
|
+
* an `exec` resolves with a signal in its result. Rejecting with
|
|
2156
|
+
* {@link KubernetesQuiesceUnconfirmedError} is the honest failure: a
|
|
2157
|
+
* process would not stop, and the reply names its pid.
|
|
2158
|
+
*/
|
|
2159
|
+
async quiesce(
|
|
2160
|
+
options: { readonly graceMs?: number } = {},
|
|
2161
|
+
signal?: AbortSignal,
|
|
2162
|
+
): Promise<KubernetesQuiesceReport> {
|
|
2163
|
+
await this.assertQuiesceSupported(signal)
|
|
2164
|
+
const reply = (await this.withRebind(
|
|
2165
|
+
async () =>
|
|
2166
|
+
await requestChecked(
|
|
2167
|
+
this.wire,
|
|
2168
|
+
{
|
|
2169
|
+
op: 'quiesce',
|
|
2170
|
+
body: { ...(options.graceMs !== undefined ? { graceMs: options.graceMs } : {}) },
|
|
2171
|
+
},
|
|
2172
|
+
signal,
|
|
2173
|
+
),
|
|
2174
|
+
signal,
|
|
2175
|
+
)) as Record<string, unknown>
|
|
2176
|
+
if (reply.ok === true) return quiesceReport(reply)
|
|
2177
|
+
const refusal = typeof reply.error === 'string' ? reply.error : 'no reason given'
|
|
2178
|
+
// The guest advertised the feature and then did not know the op. That
|
|
2179
|
+
// is one image, not two, so it is the unsupported error rather than a
|
|
2180
|
+
// second name for the same fact — see {@link assertQuiesceSupported}.
|
|
2181
|
+
if (refusal.startsWith('unknown_op')) throw this.quiesceUnsupported()
|
|
2182
|
+
const detail = typeof reply.message === 'string' ? `: ${reply.message}` : ''
|
|
2183
|
+
throw new KubernetesQuiesceUnconfirmedError(
|
|
2184
|
+
refusal,
|
|
2185
|
+
`kubernetes: the guest could not confirm that every process it is running has stopped (${refusal})${detail}. Nothing on the cluster was changed and the pod is still serving; a capture taken now may not be consistent.`,
|
|
2186
|
+
)
|
|
2187
|
+
}
|
|
2188
|
+
|
|
2189
|
+
/**
|
|
2190
|
+
* Put everything this workspace has written onto its device.
|
|
2191
|
+
*
|
|
2192
|
+
* The disk a workspace keeps is whatever the guest kernel happened to
|
|
2193
|
+
* write back. Nothing in this backend ever asked for more than that: a
|
|
2194
|
+
* `suspend()` patched the pod away and waited for it to stop, and a
|
|
2195
|
+
* stopped pod means only that nothing is writing any more — not that
|
|
2196
|
+
* what was written arrived. This is the call that closes the difference,
|
|
2197
|
+
* and it is `syncfs(2)` over the whole workspace mount, so it covers
|
|
2198
|
+
* what a COMMAND wrote as well as what `writeFile` did (`writeFile`
|
|
2199
|
+
* fsyncs its own bytes before it answers; a compiler's output is nobody's
|
|
2200
|
+
* to fsync).
|
|
2201
|
+
*
|
|
2202
|
+
* Rejecting with {@link KubernetesFlushUnconfirmedError} is the honest
|
|
2203
|
+
* failure, exactly as an unconfirmed quiesce is: the caller is usually
|
|
2204
|
+
* about to take the pod away, and "the flush did not run" has to be
|
|
2205
|
+
* distinguishable from "the flush ran".
|
|
2206
|
+
*/
|
|
2207
|
+
async flush(
|
|
2208
|
+
options: { readonly timeoutMs?: number } = {},
|
|
2209
|
+
signal?: AbortSignal,
|
|
2210
|
+
): Promise<KubernetesFlushReport> {
|
|
2211
|
+
await this.assertFlushSupported(signal)
|
|
2212
|
+
const reply = (await this.withRebind(
|
|
2213
|
+
async () =>
|
|
2214
|
+
await requestChecked(
|
|
2215
|
+
this.wire,
|
|
2216
|
+
{
|
|
2217
|
+
op: 'flush',
|
|
2218
|
+
body: { ...(options.timeoutMs !== undefined ? { timeoutMs: options.timeoutMs } : {}) },
|
|
2219
|
+
},
|
|
2220
|
+
signal,
|
|
2221
|
+
),
|
|
2222
|
+
signal,
|
|
2223
|
+
)) as Record<string, unknown>
|
|
2224
|
+
if (reply.ok === true) return flushReport(reply)
|
|
2225
|
+
const refusal = typeof reply.error === 'string' ? reply.error : 'no reason given'
|
|
2226
|
+
// Advertised and then not known: one image, not two — the same
|
|
2227
|
+
// reading {@link quiesce} gives that answer. `flush_unsupported` and
|
|
2228
|
+
// `flush_unsupported_platform` join it because they say the same
|
|
2229
|
+
// thing in the guest's own words: this image cannot perform a flush,
|
|
2230
|
+
// now or ever. Read as an unconfirmed flush instead, they would stop
|
|
2231
|
+
// every default `suspend()` against such an image permanently, which
|
|
2232
|
+
// is exactly what the feature advertisement exists to avoid.
|
|
2233
|
+
if (refusal.startsWith('unknown_op') || refusal.startsWith('flush_unsupported')) {
|
|
2234
|
+
throw this.flushUnsupported()
|
|
2235
|
+
}
|
|
2236
|
+
const detail = typeof reply.message === 'string' ? `: ${reply.message}` : ''
|
|
2237
|
+
throw new KubernetesFlushUnconfirmedError(
|
|
2238
|
+
refusal,
|
|
2239
|
+
`kubernetes: the guest could not confirm that the workspace's writes reached its disk (${refusal})${detail}. Nothing on the cluster was changed and the pod is still serving; suspending or deleting the pod now may lose whatever had not been written back.`,
|
|
2240
|
+
)
|
|
2241
|
+
}
|
|
2242
|
+
|
|
2243
|
+
/**
|
|
2244
|
+
* Whether this guest can flush at all — asked, rather than assumed,
|
|
2245
|
+
* because a `suspend()` against an older image keeps today's behaviour
|
|
2246
|
+
* and reports the gap instead of refusing to suspend.
|
|
2247
|
+
*/
|
|
2248
|
+
async supportsFlush(signal?: AbortSignal): Promise<boolean> {
|
|
2249
|
+
return (await this.wire.guestFeatures(signal)).includes(FLUSH_FEATURE)
|
|
2250
|
+
}
|
|
2251
|
+
|
|
2252
|
+
private async assertFlushSupported(signal?: AbortSignal): Promise<void> {
|
|
2253
|
+
if (await this.supportsFlush(signal)) return
|
|
2254
|
+
throw this.flushUnsupported()
|
|
2255
|
+
}
|
|
2256
|
+
|
|
2257
|
+
private flushUnsupported(): KubernetesFlushUnsupportedError {
|
|
2258
|
+
return flushUnsupportedError()
|
|
2259
|
+
}
|
|
2260
|
+
|
|
2261
|
+
/**
|
|
2262
|
+
* Whether this guest can quiesce at all — asked, rather than assumed,
|
|
2263
|
+
* because `suspend({ quiesce: true })` against an older image keeps
|
|
2264
|
+
* today's behaviour and reports the gap instead of refusing to suspend.
|
|
2265
|
+
*/
|
|
2266
|
+
async supportsQuiesce(signal?: AbortSignal): Promise<boolean> {
|
|
2267
|
+
return (await this.wire.guestFeatures(signal)).includes(QUIESCE_FEATURE)
|
|
2268
|
+
}
|
|
2269
|
+
|
|
2270
|
+
private async assertQuiesceSupported(signal?: AbortSignal): Promise<void> {
|
|
2271
|
+
if (await this.supportsQuiesce(signal)) return
|
|
2272
|
+
throw this.quiesceUnsupported()
|
|
2273
|
+
}
|
|
2274
|
+
|
|
2275
|
+
private quiesceUnsupported(): KubernetesQuiesceUnsupportedError {
|
|
2276
|
+
return new KubernetesQuiesceUnsupportedError(
|
|
2277
|
+
QUIESCE_FEATURE,
|
|
2278
|
+
`kubernetes: this workspace's guest agent does not advertise the '${QUIESCE_FEATURE}' healthz feature, so nothing here can stop the processes it is running — a terminal another handle opened, a command already in flight, or a program that moved into a session of its own all keep writing. The request is refused rather than answered with an empty list, which would read as a guest that had nothing to stop. Rebuild the workspace image from this Namzu release.`,
|
|
2279
|
+
)
|
|
2280
|
+
}
|
|
2281
|
+
|
|
2282
|
+
/**
|
|
2283
|
+
* Whether this guest keeps a session registry at all, asked once per
|
|
2284
|
+
* transport and only when a caller wants one.
|
|
2285
|
+
*/
|
|
2286
|
+
private async assertSessionsSupported(signal?: AbortSignal): Promise<void> {
|
|
2287
|
+
const features = await this.wire.guestFeatures(signal)
|
|
2288
|
+
if (features.includes(SESSIONS_FEATURE)) return
|
|
2289
|
+
throw new KubernetesSessionsUnsupportedError(
|
|
2290
|
+
SESSIONS_FEATURE,
|
|
2291
|
+
`kubernetes: this workspace's guest agent does not advertise the '${SESSIONS_FEATURE}' healthz feature, so a terminal opened here would die with this connection and a detached program could not be named, read or killed. The request is refused rather than served as a connection-bound terminal. Rebuild the workspace image from this Namzu release.`,
|
|
2292
|
+
)
|
|
2293
|
+
}
|
|
2294
|
+
|
|
2295
|
+
async openTcpConnection(options: SandboxTcpConnectOptions): Promise<SandboxTcpConnection> {
|
|
2296
|
+
return await this.withRebind(async () => await this.wire.openTcpConnection(options))
|
|
2297
|
+
}
|
|
2298
|
+
|
|
2299
|
+
/**
|
|
2300
|
+
* Run one command through a fresh, call-scoped adapter + controller.
|
|
2301
|
+
* Fresh per call — not shared instance state — because
|
|
2302
|
+
* {@link VsockAgentTransport} itself carries no cross-call connection
|
|
2303
|
+
* state (it dials fresh every time), so building one per `exec()` is
|
|
2304
|
+
* free and makes concurrent `exec()` calls on the same
|
|
2305
|
+
* `KubernetesAgentTransport` correctly independent: each gets its own
|
|
2306
|
+
* `onDial` closure and its own timing accumulator, with no shared
|
|
2307
|
+
* mutable field for two in-flight calls to race on.
|
|
2308
|
+
*/
|
|
2309
|
+
async exec(
|
|
2310
|
+
command: string,
|
|
2311
|
+
argv?: string[],
|
|
2312
|
+
opts?: SandboxExecOptions,
|
|
2313
|
+
): Promise<SandboxExecResult> {
|
|
2314
|
+
let dialMs = 0
|
|
2315
|
+
let reserveMs = 0
|
|
2316
|
+
let executeMs = 0
|
|
2317
|
+
let executeSettledAt = 0
|
|
2318
|
+
// The guest that ACCEPTED the reservation, which is the guest that
|
|
2319
|
+
// runs the command: its pod (the bind token this attempt presented)
|
|
2320
|
+
// and its process (the boot id the reply carried). Written per
|
|
2321
|
+
// ATTEMPT, so a retry that followed a replaced pod is remembered
|
|
2322
|
+
// against the pod it actually reserved on and not the one that
|
|
2323
|
+
// refused it — see {@link executionGuest}.
|
|
2324
|
+
let reservedIn: string | undefined
|
|
2325
|
+
let reservedOn: string | undefined
|
|
2326
|
+
|
|
2327
|
+
// What this attempt's dials did, for the one classification `exec()`
|
|
2328
|
+
// cannot make from its error: the controller bounds `reserve` at 2s
|
|
2329
|
+
// and hands back its own timer's Error, so a dial still inside its
|
|
2330
|
+
// connect-retry budget is reported as a reservation that took too
|
|
2331
|
+
// long, with the connect failure discarded. See {@link DialWatch}.
|
|
2332
|
+
const dials: DialWatch = { attempted: false, failed: false, connected: false }
|
|
2333
|
+
|
|
2334
|
+
// Built per ATTEMPT, from `this.handle` as it stands when the attempt
|
|
2335
|
+
// starts: a retry that follows a replaced pod has to dial the new
|
|
2336
|
+
// address, and the timing accumulators above outlive both attempts so
|
|
2337
|
+
// the caller still sees one call's total. The watch is reset here and
|
|
2338
|
+
// not there — it describes the attempt, and the second attempt is a
|
|
2339
|
+
// different pod's.
|
|
2340
|
+
const buildAttempt = (): Promise<SandboxExecResult> => {
|
|
2341
|
+
dials.attempted = false
|
|
2342
|
+
dials.failed = false
|
|
2343
|
+
dials.connected = false
|
|
2344
|
+
dials.lastError = undefined
|
|
2345
|
+
// The handle as it stands NOW, read once: the attempt dials this
|
|
2346
|
+
// address with this token, and another call's rebind must not
|
|
2347
|
+
// change what this attempt records it reserved on.
|
|
2348
|
+
const boundTo = this.handle
|
|
2349
|
+
const timedWire = new VsockAgentTransport(boundTo, {
|
|
2350
|
+
...this.wireOptions(boundTo),
|
|
2351
|
+
// Fires before each connect, so an attempt the controller's
|
|
2352
|
+
// bound aborts mid-connect is still on the record — see
|
|
2353
|
+
// {@link DialWatch}.
|
|
2354
|
+
onDialAttempt: () => {
|
|
2355
|
+
dials.attempted = true
|
|
2356
|
+
},
|
|
2357
|
+
onDial: (ms) => {
|
|
2358
|
+
dials.connected = true
|
|
2359
|
+
dialMs += ms
|
|
2360
|
+
},
|
|
2361
|
+
// Consulted by the dial on every FAILED connect attempt, which
|
|
2362
|
+
// is where the error the bound swallows is kept. The answer
|
|
2363
|
+
// itself is the transport's own, unchanged.
|
|
2364
|
+
permanentDialFailure: (err) => {
|
|
2365
|
+
dials.failed = true
|
|
2366
|
+
dials.lastError = err
|
|
2367
|
+
return this.dialFailureIsPermanent(err)
|
|
2368
|
+
},
|
|
2369
|
+
})
|
|
2370
|
+
|
|
2371
|
+
const adapter: RemoteExecutionAdapter<Pick<ExecRequest, 'stdin' | 'maxOutputBytes'>> = {
|
|
2372
|
+
label: 'kubernetes pod-network agent',
|
|
2373
|
+
reserve: async (signal) => {
|
|
2374
|
+
const startedAt = Date.now()
|
|
2375
|
+
try {
|
|
2376
|
+
const reply = await requestChecked(timedWire, { op: 'reserve-execution' }, signal)
|
|
2377
|
+
// Overwritten, never merged: the LAST reservation is the
|
|
2378
|
+
// one the command ran under, and a replacement guest
|
|
2379
|
+
// that reports no boot id must leave no evidence
|
|
2380
|
+
// behind rather than inherit its predecessor's.
|
|
2381
|
+
reservedIn = guestBootIdOf(reply)
|
|
2382
|
+
reservedOn = boundTo.token
|
|
2383
|
+
return reply
|
|
2384
|
+
} finally {
|
|
2385
|
+
reserveMs += Date.now() - startedAt
|
|
2386
|
+
}
|
|
2387
|
+
},
|
|
2388
|
+
// Checked exactly like `reserve` — see `requestChecked`.
|
|
2389
|
+
cancel: async (executionId, signal) =>
|
|
2390
|
+
await requestChecked(
|
|
2391
|
+
timedWire,
|
|
2392
|
+
{ op: 'cancel-execution', body: { executionId } },
|
|
2393
|
+
signal,
|
|
2394
|
+
),
|
|
2395
|
+
execute: async (executionId, cmd, execArgv, execOpts, signal, context) => {
|
|
2396
|
+
const startedAt = Date.now()
|
|
2397
|
+
try {
|
|
2398
|
+
return await timedWire.executeStreamed(
|
|
2399
|
+
{
|
|
2400
|
+
...(executionId ? { executionId } : {}),
|
|
2401
|
+
command: cmd,
|
|
2402
|
+
args: execArgv ?? [],
|
|
2403
|
+
...(execOpts?.cwd !== undefined ? { cwd: execOpts.cwd } : {}),
|
|
2404
|
+
...(execOpts?.env !== undefined ? { env: execOpts.env } : {}),
|
|
2405
|
+
...(execOpts?.timeout !== undefined ? { timeoutMs: execOpts.timeout } : {}),
|
|
2406
|
+
...(context?.stdin !== undefined ? { stdin: context.stdin } : {}),
|
|
2407
|
+
...(context?.maxOutputBytes !== undefined
|
|
2408
|
+
? { maxOutputBytes: context.maxOutputBytes }
|
|
2409
|
+
: {}),
|
|
2410
|
+
},
|
|
2411
|
+
execOpts,
|
|
2412
|
+
signal,
|
|
2413
|
+
)
|
|
2414
|
+
} finally {
|
|
2415
|
+
executeMs += Date.now() - startedAt
|
|
2416
|
+
executeSettledAt = Date.now()
|
|
2417
|
+
}
|
|
2418
|
+
},
|
|
2419
|
+
}
|
|
2420
|
+
|
|
2421
|
+
return new RemoteExecutionController(adapter).exec(command, argv, opts)
|
|
2422
|
+
}
|
|
2423
|
+
|
|
2424
|
+
try {
|
|
2425
|
+
return await this.withRebind(buildAttempt, opts?.signal, dials)
|
|
2426
|
+
} catch (error) {
|
|
2427
|
+
// Stamped here and nowhere else. The retirement hook one frame up
|
|
2428
|
+
// is handed the error ALONE, so this is the last point at which
|
|
2429
|
+
// "which execution is this" is still known — and the question the
|
|
2430
|
+
// hook has to answer is about this command's guest, not the
|
|
2431
|
+
// handle's.
|
|
2432
|
+
if (
|
|
2433
|
+
error instanceof RemoteCancellationUnknownError &&
|
|
2434
|
+
(reservedOn !== undefined || reservedIn !== undefined)
|
|
2435
|
+
) {
|
|
2436
|
+
executionGuest.set(error, {
|
|
2437
|
+
...(reservedOn !== undefined ? { podUid: reservedOn } : {}),
|
|
2438
|
+
...(reservedIn !== undefined ? { guestBootId: reservedIn } : {}),
|
|
2439
|
+
})
|
|
2440
|
+
}
|
|
2441
|
+
throw error
|
|
277
2442
|
} finally {
|
|
278
2443
|
this.onTiming?.({
|
|
279
2444
|
dialMs,
|
|
@@ -283,4 +2448,448 @@ export class KubernetesAgentTransport {
|
|
|
283
2448
|
})
|
|
284
2449
|
}
|
|
285
2450
|
}
|
|
2451
|
+
|
|
2452
|
+
/**
|
|
2453
|
+
* Whether this guest implements the detach/attach ops at all, asked once
|
|
2454
|
+
* per transport and only when a caller wants them.
|
|
2455
|
+
*/
|
|
2456
|
+
private async assertExecutionAttachSupported(signal?: AbortSignal): Promise<void> {
|
|
2457
|
+
const features = await this.wire.guestFeatures(signal)
|
|
2458
|
+
if (features.includes(EXECUTION_ATTACH_FEATURE)) return
|
|
2459
|
+
throw new KubernetesExecutionAttachUnsupportedError(
|
|
2460
|
+
EXECUTION_ATTACH_FEATURE,
|
|
2461
|
+
`kubernetes: this workspace's guest agent does not advertise the '${EXECUTION_ATTACH_FEATURE}' healthz feature, so a command started here would keep no output and could not be reattached to. The request is refused before the command is admitted rather than run as an ordinary exec. Rebuild the workspace image from this Namzu release.`,
|
|
2462
|
+
)
|
|
2463
|
+
}
|
|
2464
|
+
|
|
2465
|
+
/** `reserve-execution` for a caller-named id, with its reported state. */
|
|
2466
|
+
private async reserveDetached(
|
|
2467
|
+
executionId: string,
|
|
2468
|
+
signal?: AbortSignal,
|
|
2469
|
+
): Promise<{ readonly state: string }> {
|
|
2470
|
+
const response = (await this.withRebind(
|
|
2471
|
+
async () =>
|
|
2472
|
+
await requestChecked(this.wire, { op: 'reserve-execution', body: { executionId } }, signal),
|
|
2473
|
+
signal,
|
|
2474
|
+
)) as Record<string, unknown>
|
|
2475
|
+
if (response.ok !== true || response.executionId !== executionId) {
|
|
2476
|
+
throw new RemoteProtocolError(
|
|
2477
|
+
`kubernetes: the guest refused the reservation for ${executionId}: ${String(response.error ?? 'no reason given')}`,
|
|
2478
|
+
)
|
|
2479
|
+
}
|
|
2480
|
+
return { state: typeof response.state === 'string' ? response.state : 'reserved' }
|
|
2481
|
+
}
|
|
2482
|
+
|
|
2483
|
+
/**
|
|
2484
|
+
* The `cancel-execution` control path, retried for its whole confirm
|
|
2485
|
+
* window and reported UNKNOWN rather than as a failure if none of the
|
|
2486
|
+
* attempts got an answer — the same rule the shared execution
|
|
2487
|
+
* controller applies, because a command whose termination nobody
|
|
2488
|
+
* confirmed is not a command anybody may call dead.
|
|
2489
|
+
*
|
|
2490
|
+
* Nothing on the detach path calls this to reconcile a lost connection.
|
|
2491
|
+
* It runs when the CALLER asked for it: `SandboxExecOptions.signal`
|
|
2492
|
+
* aborting, or {@link cancelExecution}.
|
|
2493
|
+
*/
|
|
2494
|
+
private async confirmCancel(
|
|
2495
|
+
executionId: string,
|
|
2496
|
+
signal?: AbortSignal,
|
|
2497
|
+
): Promise<RemoteCancellationAcknowledgement> {
|
|
2498
|
+
const deadlineAt = Date.now() + CANCEL_CONFIRM_WINDOW_MS
|
|
2499
|
+
let lastError: unknown = new Error('no cancellation attempt completed')
|
|
2500
|
+
while (Date.now() < deadlineAt) {
|
|
2501
|
+
const attempt = new AbortController()
|
|
2502
|
+
const onAbort = () => attempt.abort(signal?.reason)
|
|
2503
|
+
signal?.addEventListener('abort', onAbort, { once: true })
|
|
2504
|
+
const timer = setTimeout(
|
|
2505
|
+
() =>
|
|
2506
|
+
attempt.abort(new Error(`cancellation attempt exceeded ${CANCEL_ATTEMPT_TIMEOUT_MS}ms`)),
|
|
2507
|
+
Math.min(CANCEL_ATTEMPT_TIMEOUT_MS, Math.max(1, deadlineAt - Date.now())),
|
|
2508
|
+
)
|
|
2509
|
+
timer.unref?.()
|
|
2510
|
+
try {
|
|
2511
|
+
return parseCancellationReply(executionId, await this.cancel(executionId, attempt.signal))
|
|
2512
|
+
} catch (error) {
|
|
2513
|
+
if (error instanceof KubernetesExecutionNotAttachableError) throw error
|
|
2514
|
+
if (error instanceof KubernetesAgentUnauthorizedError) throw error
|
|
2515
|
+
lastError = error
|
|
2516
|
+
const remaining = deadlineAt - Date.now()
|
|
2517
|
+
if (remaining > 0) await pause(Math.min(50, remaining))
|
|
2518
|
+
} finally {
|
|
2519
|
+
clearTimeout(timer)
|
|
2520
|
+
signal?.removeEventListener('abort', onAbort)
|
|
2521
|
+
}
|
|
2522
|
+
}
|
|
2523
|
+
throw new RemoteCancellationUnknownError(
|
|
2524
|
+
`Remote sandbox cancellation could not be confirmed for ${executionId}: ${lastError instanceof Error ? lastError.message : String(lastError)}. The remote outcome is unknown; do not automatically retry the command.`,
|
|
2525
|
+
{ cause: lastError },
|
|
2526
|
+
)
|
|
2527
|
+
}
|
|
2528
|
+
|
|
2529
|
+
/**
|
|
2530
|
+
* End a command by id, from any host process holding the id and the
|
|
2531
|
+
* bind token. Resolves only on a CONFIRMED termination.
|
|
2532
|
+
*
|
|
2533
|
+
* The outcome is not reported here — a cancelled execution's result and
|
|
2534
|
+
* whatever output it managed is read back through
|
|
2535
|
+
* {@link attachExecution}, which is the op that exists for reading.
|
|
2536
|
+
*/
|
|
2537
|
+
async cancelExecution(executionId: string, signal?: AbortSignal): Promise<void> {
|
|
2538
|
+
assertExecutionId(executionId)
|
|
2539
|
+
await this.confirmCancel(executionId, signal)
|
|
2540
|
+
}
|
|
2541
|
+
|
|
2542
|
+
/**
|
|
2543
|
+
* Observe a command that is already the guest's, from `fromOffset` on,
|
|
2544
|
+
* and resolve with its result.
|
|
2545
|
+
*
|
|
2546
|
+
* It never signals the command. Aborting `signal` stops OBSERVING and
|
|
2547
|
+
* rejects with {@link KubernetesExecutionDetachedError}; it does not
|
|
2548
|
+
* cancel, and the command goes on running. Ending a command is
|
|
2549
|
+
* {@link cancelExecution} and nothing else.
|
|
2550
|
+
*/
|
|
2551
|
+
async attachExecution(
|
|
2552
|
+
executionId: string,
|
|
2553
|
+
options: KubernetesAttachExecutionOptions = {},
|
|
2554
|
+
): Promise<SandboxExecResult> {
|
|
2555
|
+
assertExecutionId(executionId)
|
|
2556
|
+
await this.assertExecutionAttachSupported(options.signal)
|
|
2557
|
+
const cursor: OutputCursor = {
|
|
2558
|
+
offset: options.fromOffset ?? 0,
|
|
2559
|
+
stdout: '',
|
|
2560
|
+
stderr: '',
|
|
2561
|
+
droppedBytes: 0,
|
|
2562
|
+
}
|
|
2563
|
+
const observation = new AbortController()
|
|
2564
|
+
const onDetach = () => observation.abort(options.signal?.reason)
|
|
2565
|
+
options.signal?.addEventListener('abort', onDetach, { once: true })
|
|
2566
|
+
if (options.signal?.aborted) onDetach()
|
|
2567
|
+
try {
|
|
2568
|
+
return await this.attachOnce(executionId, cursor, options, observation.signal)
|
|
2569
|
+
} catch (error) {
|
|
2570
|
+
if (options.signal?.aborted) throw this.detached(executionId, cursor, error)
|
|
2571
|
+
throw error
|
|
2572
|
+
} finally {
|
|
2573
|
+
options.signal?.removeEventListener('abort', onDetach)
|
|
2574
|
+
observation.abort(new Error('kubernetes: attach observation finished'))
|
|
2575
|
+
}
|
|
2576
|
+
}
|
|
2577
|
+
|
|
2578
|
+
/**
|
|
2579
|
+
* Run one command whose observation can outlive this connection, and —
|
|
2580
|
+
* when the connection is what failed — get it back rather than killing
|
|
2581
|
+
* the command to reconcile.
|
|
2582
|
+
*
|
|
2583
|
+
* The order is exactly: refuse if the guest cannot keep output, reserve
|
|
2584
|
+
* the id, and only then admit the command. Reserving the id the CALLER
|
|
2585
|
+
* named is what makes a retried start idempotent: a second call with the
|
|
2586
|
+
* same id inside retention finds the execution already running or
|
|
2587
|
+
* finished, sends no `execute`, and attaches to the one that exists.
|
|
2588
|
+
*
|
|
2589
|
+
* `SandboxExecOptions.signal` keeps its contract — aborting it runs the
|
|
2590
|
+
* confirmed cancel — and `detachSignal` is its opposite: it ends the
|
|
2591
|
+
* observation and leaves the command alone, for a host that is shutting
|
|
2592
|
+
* down and wants its work to survive the rollout.
|
|
2593
|
+
*/
|
|
2594
|
+
async execDetached(
|
|
2595
|
+
command: string,
|
|
2596
|
+
argv?: string[],
|
|
2597
|
+
opts: KubernetesDetachedExecOptions = {},
|
|
2598
|
+
): Promise<SandboxExecResult> {
|
|
2599
|
+
const executionId = opts.executionId ?? `exec_${randomUUID()}`
|
|
2600
|
+
assertExecutionId(executionId)
|
|
2601
|
+
const startedAt = Date.now()
|
|
2602
|
+
if (opts.signal?.aborted) {
|
|
2603
|
+
return {
|
|
2604
|
+
exitCode: 1,
|
|
2605
|
+
stdout: '',
|
|
2606
|
+
stderr: '',
|
|
2607
|
+
timedOut: false,
|
|
2608
|
+
durationMs: Math.max(0, Date.now() - startedAt),
|
|
2609
|
+
stdoutTruncated: false,
|
|
2610
|
+
stderrTruncated: false,
|
|
2611
|
+
}
|
|
2612
|
+
}
|
|
2613
|
+
await this.assertExecutionAttachSupported(opts.signal)
|
|
2614
|
+
|
|
2615
|
+
const cursor: OutputCursor = { offset: 0, stdout: '', stderr: '', droppedBytes: 0 }
|
|
2616
|
+
const observationTimeoutMs = Math.min(
|
|
2617
|
+
MAX_TIMER_DELAY_MS,
|
|
2618
|
+
(typeof opts.timeout === 'number' && Number.isFinite(opts.timeout) && opts.timeout > 0
|
|
2619
|
+
? opts.timeout
|
|
2620
|
+
: DEFAULT_EXECUTION_TIMEOUT_MS) + EXECUTION_OBSERVATION_GRACE_MS,
|
|
2621
|
+
)
|
|
2622
|
+
const deadlineAt = Date.now() + observationTimeoutMs
|
|
2623
|
+
|
|
2624
|
+
// The caller's abort runs the CONFIRMED cancel, in the background,
|
|
2625
|
+
// while the observation keeps reading: the guest answers the cancel
|
|
2626
|
+
// on its own connection and ends this one with the terminal frame, so
|
|
2627
|
+
// aborting produces a result rather than a severed stream.
|
|
2628
|
+
let cancelFailure: unknown
|
|
2629
|
+
let cancelling = false
|
|
2630
|
+
const cancelNow = (): void => {
|
|
2631
|
+
if (cancelling) return
|
|
2632
|
+
cancelling = true
|
|
2633
|
+
void this.confirmCancel(executionId).catch((error: unknown) => {
|
|
2634
|
+
cancelFailure = error
|
|
2635
|
+
})
|
|
2636
|
+
}
|
|
2637
|
+
const onAbort = () => cancelNow()
|
|
2638
|
+
opts.signal?.addEventListener('abort', onAbort, { once: true })
|
|
2639
|
+
|
|
2640
|
+
// Aborting this stops the READING and nothing else. It is never the
|
|
2641
|
+
// caller's `signal`: destroying a socket does not end a guest command,
|
|
2642
|
+
// and a host that treats it as though it did is exactly how a network
|
|
2643
|
+
// blip used to cost a workspace its pod.
|
|
2644
|
+
const observation = new AbortController()
|
|
2645
|
+
const onDetach = () => observation.abort(new Error('kubernetes: observation detached'))
|
|
2646
|
+
opts.detachSignal?.addEventListener('abort', onDetach, { once: true })
|
|
2647
|
+
if (opts.detachSignal?.aborted) onDetach()
|
|
2648
|
+
|
|
2649
|
+
try {
|
|
2650
|
+
let lastError: unknown
|
|
2651
|
+
const reservation = await this.reserveDetached(executionId, opts.signal)
|
|
2652
|
+
if (reservation.state === 'reserved') {
|
|
2653
|
+
try {
|
|
2654
|
+
return await this.executeRetained(
|
|
2655
|
+
executionId,
|
|
2656
|
+
{
|
|
2657
|
+
executionId,
|
|
2658
|
+
command,
|
|
2659
|
+
args: argv ?? [],
|
|
2660
|
+
...(opts.cwd !== undefined ? { cwd: opts.cwd } : {}),
|
|
2661
|
+
...(opts.env !== undefined ? { env: opts.env } : {}),
|
|
2662
|
+
...(opts.timeout !== undefined ? { timeoutMs: opts.timeout } : {}),
|
|
2663
|
+
retainOutput: true,
|
|
2664
|
+
},
|
|
2665
|
+
cursor,
|
|
2666
|
+
opts,
|
|
2667
|
+
observation.signal,
|
|
2668
|
+
observationTimeoutMs,
|
|
2669
|
+
)
|
|
2670
|
+
} catch (error) {
|
|
2671
|
+
lastError = error
|
|
2672
|
+
// A second host process that reserved the same id and won
|
|
2673
|
+
// the race owns the command now. Its refusal says so, and
|
|
2674
|
+
// the answer is to ATTACH to the command that exists —
|
|
2675
|
+
// reporting a failure here would be a lie about an id
|
|
2676
|
+
// whose command is running.
|
|
2677
|
+
if (!lostTheStartRace(error) && isTerminalAttachError(error)) throw error
|
|
2678
|
+
}
|
|
2679
|
+
}
|
|
2680
|
+
|
|
2681
|
+
// The window bounds GETTING BACK, and only that. It is armed as an
|
|
2682
|
+
// abort rather than checked between attempts because a single
|
|
2683
|
+
// attempt is not short: the dial carries its own connect-retry
|
|
2684
|
+
// budget, so a peer that is refusing connections would otherwise
|
|
2685
|
+
// be waited on for that whole budget inside one attempt and the
|
|
2686
|
+
// caller's bound would never be consulted. Once an attach has
|
|
2687
|
+
// actually attached the window is disarmed — a command that is
|
|
2688
|
+
// being read successfully is not something to give up on — and it
|
|
2689
|
+
// is re-armed if that connection dies in its turn.
|
|
2690
|
+
const reattachWindowMs = opts.reattachWindowMs ?? DEFAULT_REATTACH_WINDOW_MS
|
|
2691
|
+
let windowEndsAt = Date.now() + reattachWindowMs
|
|
2692
|
+
for (;;) {
|
|
2693
|
+
if (opts.detachSignal?.aborted) break
|
|
2694
|
+
const remainingMs = windowEndsAt - Date.now()
|
|
2695
|
+
if (remainingMs <= 0) break
|
|
2696
|
+
const window = new AbortController()
|
|
2697
|
+
const windowTimer = setTimeout(
|
|
2698
|
+
() =>
|
|
2699
|
+
window.abort(
|
|
2700
|
+
new Error(
|
|
2701
|
+
`kubernetes: could not reattach to execution ${executionId} within ${reattachWindowMs}ms`,
|
|
2702
|
+
),
|
|
2703
|
+
),
|
|
2704
|
+
remainingMs,
|
|
2705
|
+
)
|
|
2706
|
+
windowTimer.unref?.()
|
|
2707
|
+
let reattached = false
|
|
2708
|
+
try {
|
|
2709
|
+
return await this.attachOnce(
|
|
2710
|
+
executionId,
|
|
2711
|
+
cursor,
|
|
2712
|
+
opts,
|
|
2713
|
+
AbortSignal.any([observation.signal, window.signal]),
|
|
2714
|
+
deadlineAt,
|
|
2715
|
+
() => {
|
|
2716
|
+
reattached = true
|
|
2717
|
+
clearTimeout(windowTimer)
|
|
2718
|
+
},
|
|
2719
|
+
)
|
|
2720
|
+
} catch (error) {
|
|
2721
|
+
lastError = error
|
|
2722
|
+
if (isTerminalAttachError(error)) break
|
|
2723
|
+
if (opts.detachSignal?.aborted) break
|
|
2724
|
+
if (reattached) windowEndsAt = Date.now() + reattachWindowMs
|
|
2725
|
+
else if (window.signal.aborted) break
|
|
2726
|
+
await pause(REATTACH_RETRY_DELAY_MS)
|
|
2727
|
+
} finally {
|
|
2728
|
+
clearTimeout(windowTimer)
|
|
2729
|
+
}
|
|
2730
|
+
}
|
|
2731
|
+
throw this.detached(executionId, cursor, cancelFailure ?? lastError)
|
|
2732
|
+
} finally {
|
|
2733
|
+
opts.signal?.removeEventListener('abort', onAbort)
|
|
2734
|
+
opts.detachSignal?.removeEventListener('abort', onDetach)
|
|
2735
|
+
observation.abort(new Error('kubernetes: detached execution observation finished'))
|
|
2736
|
+
}
|
|
2737
|
+
}
|
|
2738
|
+
|
|
2739
|
+
/**
|
|
2740
|
+
* The `execute` leg of a detached run, read frame by frame.
|
|
2741
|
+
*
|
|
2742
|
+
* It deliberately does NOT go through `executeStreamed`: that path
|
|
2743
|
+
* hands its caller `{stream, data}` and nothing else, and the byte
|
|
2744
|
+
* offsets this cursor lives on are on the frames themselves. Reading
|
|
2745
|
+
* them here is what lets a reattach resume exactly where this
|
|
2746
|
+
* connection stopped, on output whose decoded length is not its byte
|
|
2747
|
+
* length — see {@link applyDelta}. The frame union and its validation
|
|
2748
|
+
* are the shared ones, so an ordinary exec and a detached one never
|
|
2749
|
+
* disagree about what the guest said.
|
|
2750
|
+
*/
|
|
2751
|
+
private async executeRetained(
|
|
2752
|
+
executionId: string,
|
|
2753
|
+
body: ExecRequest,
|
|
2754
|
+
cursor: OutputCursor,
|
|
2755
|
+
opts: KubernetesDetachedExecOptions,
|
|
2756
|
+
signal: AbortSignal,
|
|
2757
|
+
observationTimeoutMs: number,
|
|
2758
|
+
): Promise<SandboxExecResult> {
|
|
2759
|
+
// No `onOutput` on the accumulator: this reads the frames, so the
|
|
2760
|
+
// caller is called exactly once per chunk, from `applyDelta`.
|
|
2761
|
+
const accumulator = new ExecResultAccumulator(Date.now())
|
|
2762
|
+
await this.wire.streamFramedRequest(
|
|
2763
|
+
{ op: 'execute', body },
|
|
2764
|
+
(frame) => {
|
|
2765
|
+
if (isUnauthorized(frame)) throw new KubernetesAgentUnauthorizedError()
|
|
2766
|
+
const event = parseExecEvent(frame)
|
|
2767
|
+
const stream = deltaStream(event.type)
|
|
2768
|
+
if (stream !== undefined) applyDelta(cursor, executionId, stream, frame, opts.onOutput)
|
|
2769
|
+
accumulator.push(event)
|
|
2770
|
+
},
|
|
2771
|
+
{ observationTimeoutMs },
|
|
2772
|
+
signal,
|
|
2773
|
+
)
|
|
2774
|
+
if (!accumulator.done) {
|
|
2775
|
+
throw new RemoteProtocolError(
|
|
2776
|
+
`kubernetes: the execute stream for ${executionId} ended without a result`,
|
|
2777
|
+
)
|
|
2778
|
+
}
|
|
2779
|
+
return accumulator.finish()
|
|
2780
|
+
}
|
|
2781
|
+
|
|
2782
|
+
/** One `attach-execution` stream, read to its terminal frame. */
|
|
2783
|
+
private async attachOnce(
|
|
2784
|
+
executionId: string,
|
|
2785
|
+
cursor: OutputCursor,
|
|
2786
|
+
opts: KubernetesAttachExecutionOptions,
|
|
2787
|
+
signal: AbortSignal,
|
|
2788
|
+
deadlineAt?: number,
|
|
2789
|
+
onAttached?: () => void,
|
|
2790
|
+
): Promise<SandboxExecResult> {
|
|
2791
|
+
let terminal: AttachTerminal | undefined
|
|
2792
|
+
let refusal: KubernetesExecutionNotAttachableError | undefined
|
|
2793
|
+
const observationTimeoutMs =
|
|
2794
|
+
deadlineAt === undefined
|
|
2795
|
+
? MAX_TIMER_DELAY_MS
|
|
2796
|
+
: Math.max(1, Math.min(MAX_TIMER_DELAY_MS, deadlineAt - Date.now()))
|
|
2797
|
+
await this.wire.streamFramedRequest(
|
|
2798
|
+
{ op: 'attach-execution', body: { executionId, fromOffset: cursor.offset } },
|
|
2799
|
+
(event) => {
|
|
2800
|
+
if (isUnauthorized(event)) throw new KubernetesAgentUnauthorizedError()
|
|
2801
|
+
const type = event.type
|
|
2802
|
+
if (type === 'attached') {
|
|
2803
|
+
onAttached?.()
|
|
2804
|
+
const from = Number(event.fromOffset)
|
|
2805
|
+
if (Number.isFinite(from)) cursor.offset = from
|
|
2806
|
+
const dropped = Number(event.droppedBytes ?? 0)
|
|
2807
|
+
if (Number.isFinite(dropped) && dropped > 0) {
|
|
2808
|
+
cursor.droppedBytes += dropped
|
|
2809
|
+
opts.onGap?.({ executionId, fromOffset: cursor.offset, droppedBytes: dropped })
|
|
2810
|
+
}
|
|
2811
|
+
return
|
|
2812
|
+
}
|
|
2813
|
+
const stream = deltaStream(type)
|
|
2814
|
+
if (stream !== undefined) {
|
|
2815
|
+
applyDelta(cursor, executionId, stream, event, opts.onOutput)
|
|
2816
|
+
return
|
|
2817
|
+
}
|
|
2818
|
+
if (type === 'attach_result') {
|
|
2819
|
+
const next = Number(event.nextOffset)
|
|
2820
|
+
if (Number.isFinite(next)) cursor.offset = next
|
|
2821
|
+
const outcome = String(event.outcome)
|
|
2822
|
+
terminal = {
|
|
2823
|
+
outcome:
|
|
2824
|
+
outcome === 'cancelled' || outcome === 'failed'
|
|
2825
|
+
? (outcome as 'cancelled' | 'failed')
|
|
2826
|
+
: 'completed',
|
|
2827
|
+
...(event.result !== undefined
|
|
2828
|
+
? { result: terminalMetadataOrThrow(event.result, executionId) }
|
|
2829
|
+
: {}),
|
|
2830
|
+
...(typeof event.error === 'string' ? { error: event.error } : {}),
|
|
2831
|
+
}
|
|
2832
|
+
return
|
|
2833
|
+
}
|
|
2834
|
+
if (type === 'error') {
|
|
2835
|
+
const code = typeof event.error === 'string' ? event.error : 'unknown'
|
|
2836
|
+
const state = typeof event.state === 'string' ? event.state : undefined
|
|
2837
|
+
// A refusal for an execution the guest still holds as
|
|
2838
|
+
// `reserved` is the one case that is not about retention:
|
|
2839
|
+
// the command was never started, so there is nothing
|
|
2840
|
+
// running and nothing to come back for. Saying anything
|
|
2841
|
+
// else here would send a caller looking for a process
|
|
2842
|
+
// that does not exist.
|
|
2843
|
+
const because =
|
|
2844
|
+
state === 'reserved'
|
|
2845
|
+
? 'The guest holds it as reserved and never started it, so no command is running and there is nothing to reattach to.'
|
|
2846
|
+
: 'A record is kept for the retention window configured by NAMZU_AGENT_EXECUTION_RETAINED_TTL_MS and is lost when the pod is replaced; only a command started with detach keeps its output at all.'
|
|
2847
|
+
refusal = new KubernetesExecutionNotAttachableError(
|
|
2848
|
+
executionId,
|
|
2849
|
+
isAttachRefusal(code) ? code : 'unknown',
|
|
2850
|
+
`kubernetes: the guest refused to attach to execution ${executionId} (${code}). ${because}`,
|
|
2851
|
+
{ state },
|
|
2852
|
+
)
|
|
2853
|
+
return
|
|
2854
|
+
}
|
|
2855
|
+
throw new RemoteProtocolError(
|
|
2856
|
+
`kubernetes: the guest sent an unexpected frame on the attach stream for ${executionId}: ${JSON.stringify(event).slice(0, 200)}`,
|
|
2857
|
+
)
|
|
2858
|
+
},
|
|
2859
|
+
{ observationTimeoutMs },
|
|
2860
|
+
signal,
|
|
2861
|
+
)
|
|
2862
|
+
if (refusal !== undefined) throw refusal
|
|
2863
|
+
if (terminal === undefined) {
|
|
2864
|
+
throw new RemoteProtocolError(
|
|
2865
|
+
`kubernetes: the attach stream for ${executionId} ended without a terminal frame`,
|
|
2866
|
+
)
|
|
2867
|
+
}
|
|
2868
|
+
return resultFromAttachTerminal(executionId, terminal, cursor)
|
|
2869
|
+
}
|
|
2870
|
+
|
|
2871
|
+
/** The one error a lost observation ends with. */
|
|
2872
|
+
private detached(
|
|
2873
|
+
executionId: string,
|
|
2874
|
+
cursor: OutputCursor,
|
|
2875
|
+
cause: unknown,
|
|
2876
|
+
): KubernetesExecutionDetachedError {
|
|
2877
|
+
// The guest can tell us the command never started — the reservation
|
|
2878
|
+
// is still `reserved`, so the `execute` never reached it. Then
|
|
2879
|
+
// there is nothing running, nothing to reattach to and nothing to
|
|
2880
|
+
// cancel, and promising otherwise sends the caller after a process
|
|
2881
|
+
// that does not exist. Every other cause leaves the command's fate
|
|
2882
|
+
// genuinely unknown to this host, which is what the rest says.
|
|
2883
|
+
const neverStarted =
|
|
2884
|
+
cause instanceof KubernetesExecutionNotAttachableError && cause.executionState === 'reserved'
|
|
2885
|
+
const advice = neverStarted
|
|
2886
|
+
? 'the guest still holds it as RESERVED, so the command never started: nothing is running, and starting it again with the same id is safe.'
|
|
2887
|
+
: `it may still be running in the workspace pod. Reattach with attachExecution('${executionId}', { fromOffset: ${cursor.offset} }), from this process or another one, or end it with cancelExecution('${executionId}').`
|
|
2888
|
+
return new KubernetesExecutionDetachedError(
|
|
2889
|
+
executionId,
|
|
2890
|
+
cursor.offset,
|
|
2891
|
+
`kubernetes: stopped observing execution ${executionId} after ${cursor.offset} bytes of output, and did NOT cancel it — ${advice} Cause: ${cause instanceof Error ? cause.message : String(cause)}`,
|
|
2892
|
+
{ cause },
|
|
2893
|
+
)
|
|
2894
|
+
}
|
|
286
2895
|
}
|