@namzu/sandbox 13.0.0 → 15.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +1147 -0
- package/README.md +447 -0
- package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
- package/dist/backends/aci-standby-pool/index.js +13 -1
- package/dist/backends/aci-standby-pool/index.js.map +1 -1
- package/dist/backends/docker/index.d.ts.map +1 -1
- package/dist/backends/docker/index.js +19 -1
- package/dist/backends/docker/index.js.map +1 -1
- package/dist/backends/firecracker/index.d.ts.map +1 -1
- package/dist/backends/firecracker/index.js +12 -2
- package/dist/backends/firecracker/index.js.map +1 -1
- package/dist/backends/firecracker/protocol.d.ts +481 -8
- package/dist/backends/firecracker/protocol.d.ts.map +1 -1
- package/dist/backends/firecracker/protocol.js +136 -0
- package/dist/backends/firecracker/protocol.js.map +1 -1
- package/dist/backends/firecracker/transport.d.ts +642 -14
- package/dist/backends/firecracker/transport.d.ts.map +1 -1
- package/dist/backends/firecracker/transport.js +1307 -34
- package/dist/backends/firecracker/transport.js.map +1 -1
- package/dist/backends/kubernetes/egress-policy.d.ts +1296 -0
- package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -0
- package/dist/backends/kubernetes/egress-policy.js +2458 -0
- package/dist/backends/kubernetes/egress-policy.js.map +1 -0
- package/dist/backends/kubernetes/identity.d.ts +193 -0
- package/dist/backends/kubernetes/identity.d.ts.map +1 -0
- package/dist/backends/kubernetes/identity.js +147 -0
- package/dist/backends/kubernetes/identity.js.map +1 -0
- package/dist/backends/kubernetes/index.d.ts +1019 -0
- package/dist/backends/kubernetes/index.d.ts.map +1 -0
- package/dist/backends/kubernetes/index.js +1756 -0
- package/dist/backends/kubernetes/index.js.map +1 -0
- package/dist/backends/kubernetes/ingress-policy.d.ts +375 -0
- package/dist/backends/kubernetes/ingress-policy.d.ts.map +1 -0
- package/dist/backends/kubernetes/ingress-policy.js +1050 -0
- package/dist/backends/kubernetes/ingress-policy.js.map +1 -0
- package/dist/backends/kubernetes/k8s-client.d.ts +334 -0
- package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -0
- package/dist/backends/kubernetes/k8s-client.js +553 -0
- package/dist/backends/kubernetes/k8s-client.js.map +1 -0
- package/dist/backends/kubernetes/lease.d.ts +145 -0
- package/dist/backends/kubernetes/lease.d.ts.map +1 -0
- package/dist/backends/kubernetes/lease.js +201 -0
- package/dist/backends/kubernetes/lease.js.map +1 -0
- package/dist/backends/kubernetes/objects.d.ts +702 -0
- package/dist/backends/kubernetes/objects.d.ts.map +1 -0
- package/dist/backends/kubernetes/objects.js +518 -0
- package/dist/backends/kubernetes/objects.js.map +1 -0
- package/dist/backends/kubernetes/per-sandbox-policy.d.ts +219 -0
- package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -0
- package/dist/backends/kubernetes/per-sandbox-policy.js +407 -0
- package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -0
- package/dist/backends/kubernetes/privilege-probe.d.ts +136 -0
- package/dist/backends/kubernetes/privilege-probe.d.ts.map +1 -0
- package/dist/backends/kubernetes/privilege-probe.js +185 -0
- package/dist/backends/kubernetes/privilege-probe.js.map +1 -0
- package/dist/backends/kubernetes/rbac.d.ts +153 -0
- package/dist/backends/kubernetes/rbac.d.ts.map +1 -0
- package/dist/backends/kubernetes/rbac.js +177 -0
- package/dist/backends/kubernetes/rbac.js.map +1 -0
- package/dist/backends/kubernetes/sandbox.d.ts +190 -0
- package/dist/backends/kubernetes/sandbox.d.ts.map +1 -0
- package/dist/backends/kubernetes/sandbox.js +433 -0
- package/dist/backends/kubernetes/sandbox.js.map +1 -0
- package/dist/backends/kubernetes/transport.d.ts +1048 -0
- package/dist/backends/kubernetes/transport.d.ts.map +1 -0
- package/dist/backends/kubernetes/transport.js +2093 -0
- package/dist/backends/kubernetes/transport.js.map +1 -0
- package/dist/backends/kubernetes/workspace.d.ts +1512 -0
- package/dist/backends/kubernetes/workspace.d.ts.map +1 -0
- package/dist/backends/kubernetes/workspace.js +3703 -0
- package/dist/backends/kubernetes/workspace.js.map +1 -0
- package/dist/backends/remote-execution-controller.d.ts +14 -0
- package/dist/backends/remote-execution-controller.d.ts.map +1 -1
- package/dist/backends/remote-execution-controller.js.map +1 -1
- package/dist/index.d.ts +350 -2
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +344 -34
- package/dist/index.js.map +1 -1
- package/dist/testing/sandbox-conformance.d.ts +227 -0
- package/dist/testing/sandbox-conformance.d.ts.map +1 -0
- package/dist/testing/sandbox-conformance.js +896 -0
- package/dist/testing/sandbox-conformance.js.map +1 -0
- package/package.json +5 -4
- package/src/backends/aci-standby-pool/index.ts +16 -1
- package/src/backends/docker/index.ts +22 -1
- package/src/backends/firecracker/index.ts +14 -2
- package/src/backends/firecracker/protocol.ts +541 -6
- package/src/backends/firecracker/transport.ts +1687 -64
- package/src/backends/kubernetes/egress-policy.ts +3448 -0
- package/src/backends/kubernetes/identity.ts +261 -0
- package/src/backends/kubernetes/index.ts +2670 -0
- package/src/backends/kubernetes/ingress-policy.ts +1344 -0
- package/src/backends/kubernetes/k8s-client.ts +742 -0
- package/src/backends/kubernetes/lease.ts +254 -0
- package/src/backends/kubernetes/objects.ts +983 -0
- package/src/backends/kubernetes/per-sandbox-policy.ts +542 -0
- package/src/backends/kubernetes/privilege-probe.ts +261 -0
- package/src/backends/kubernetes/rbac.ts +192 -0
- package/src/backends/kubernetes/sandbox.ts +593 -0
- package/src/backends/kubernetes/transport.ts +2895 -0
- package/src/backends/kubernetes/workspace.ts +5640 -0
- package/src/backends/remote-execution-controller.ts +14 -0
- package/src/index.ts +838 -35
- package/src/testing/sandbox-conformance.ts +1202 -0
package/src/index.ts
CHANGED
|
@@ -44,6 +44,30 @@ import type {
|
|
|
44
44
|
import { buildAciStandbyPoolBackend } from './backends/aci-standby-pool/index.js'
|
|
45
45
|
import { buildDockerBackend, resolveLayout } from './backends/docker/index.js'
|
|
46
46
|
import { buildFirecrackerBackend } from './backends/firecracker/index.js'
|
|
47
|
+
import type { KubernetesEgressConfig } from './backends/kubernetes/egress-policy.js'
|
|
48
|
+
import {
|
|
49
|
+
type KubernetesAgentAddressMode,
|
|
50
|
+
type KubernetesBackendInternalConfig,
|
|
51
|
+
type KubernetesClusterAccess,
|
|
52
|
+
type KubernetesReadTaskCapacityOptions,
|
|
53
|
+
type KubernetesReleaseTaskSandboxesOptions,
|
|
54
|
+
type KubernetesTaskCapacity,
|
|
55
|
+
buildKubernetesBackend,
|
|
56
|
+
readKubernetesTaskCapacity as readTaskCapacityOnCluster,
|
|
57
|
+
releaseKubernetesTaskSandboxes as releaseTaskSandboxesOnCluster,
|
|
58
|
+
} from './backends/kubernetes/index.js'
|
|
59
|
+
import type { KubernetesIngressConfig } from './backends/kubernetes/ingress-policy.js'
|
|
60
|
+
import {
|
|
61
|
+
type KubernetesWorkspace,
|
|
62
|
+
type KubernetesWorkspaceOptions,
|
|
63
|
+
type KubernetesWorkspaceSummary,
|
|
64
|
+
type KubernetesWorkspaceSuspendOptions,
|
|
65
|
+
type KubernetesWorkspaceTransitionOptions,
|
|
66
|
+
createKubernetesWorkspace as buildKubernetesWorkspace,
|
|
67
|
+
deleteKubernetesWorkspace as deleteWorkspaceOnCluster,
|
|
68
|
+
listKubernetesWorkspaces as listWorkspacesOnCluster,
|
|
69
|
+
suspendKubernetesWorkspace as suspendWorkspaceOnCluster,
|
|
70
|
+
} from './backends/kubernetes/workspace.js'
|
|
47
71
|
|
|
48
72
|
// Re-export the layout types so consumers of `@namzu/sandbox` can
|
|
49
73
|
// import them without also depending on `@namzu/sdk`. The canonical
|
|
@@ -78,11 +102,408 @@ export type {
|
|
|
78
102
|
OrchestratorTokenProvider,
|
|
79
103
|
} from './backends/firecracker/index.js'
|
|
80
104
|
export {
|
|
105
|
+
AgentDialFailedError,
|
|
106
|
+
AgentPreauthFrameTooLargeError,
|
|
107
|
+
AgentReadFileStreamUnsupportedError,
|
|
108
|
+
AgentWriteFileTooLargeError,
|
|
109
|
+
DEFAULT_MAX_WRITE_FILE_BYTES,
|
|
81
110
|
FIRECRACKER_AGENT_PROTOCOL_VERSION,
|
|
111
|
+
GUEST_FRAME_LIMIT_BYTES,
|
|
82
112
|
type SandboxAgentHandle,
|
|
113
|
+
TCP_PREAUTH_FRAME_LIMIT_BYTES,
|
|
83
114
|
type VsockTransportOptions,
|
|
84
115
|
VsockAgentTransport,
|
|
85
116
|
} from './backends/firecracker/transport.js'
|
|
117
|
+
// The `write-file` part protocol: the healthz feature string a guest
|
|
118
|
+
// advertises when it can take a body larger than one frame, and the shape
|
|
119
|
+
// of one part. Exported so a host writing its own guest, or asserting what
|
|
120
|
+
// this one advertises, names them rather than repeating the literal.
|
|
121
|
+
export {
|
|
122
|
+
WRITE_FILE_PARTS_FEATURE,
|
|
123
|
+
type WriteFilePart,
|
|
124
|
+
} from './backends/firecracker/protocol.js'
|
|
125
|
+
// The per-stream liveness heartbeat: the feature string a guest advertises
|
|
126
|
+
// when it understands one, the frame both sides send, and how many missed
|
|
127
|
+
// intervals end a stream. Same reason as above — a host asserting what this
|
|
128
|
+
// guest advertises should name the string rather than repeat the literal.
|
|
129
|
+
export {
|
|
130
|
+
MIN_STREAM_HEARTBEAT_MS,
|
|
131
|
+
STREAM_HEARTBEAT_FEATURE,
|
|
132
|
+
STREAM_HEARTBEAT_MAX_ECHO_FACTOR,
|
|
133
|
+
STREAM_HEARTBEAT_MISS_LIMIT,
|
|
134
|
+
type StreamHeartbeat,
|
|
135
|
+
} from './backends/firecracker/protocol.js'
|
|
136
|
+
// The read side of the same arrangement: one healthz feature string for
|
|
137
|
+
// both the ranged `read-file` and the `read-file-stream` op, and the
|
|
138
|
+
// request/event shapes they speak. Exported for the same reason — a host
|
|
139
|
+
// writing its own guest, or asserting what this one advertises, names them
|
|
140
|
+
// rather than repeating the literal.
|
|
141
|
+
export {
|
|
142
|
+
READ_FILE_STREAM_FEATURE,
|
|
143
|
+
type ReadFileStreamEvent,
|
|
144
|
+
type ReadFileStreamRequest,
|
|
145
|
+
} from './backends/firecracker/protocol.js'
|
|
146
|
+
|
|
147
|
+
// Kubernetes (agent-sandbox on any cluster) public surface. The access union
|
|
148
|
+
// is named by `KubernetesBackendConfig.access`, so a host that builds its own
|
|
149
|
+
// credential callback can name what it is passing; the address mode is named
|
|
150
|
+
// by `KubernetesBackendConfig.agentAddress` and decides whether the agent is
|
|
151
|
+
// dialed at its Service FQDN (in-cluster host) or at its pod IP (a host
|
|
152
|
+
// outside the cluster, on a routable pod network).
|
|
153
|
+
export type {
|
|
154
|
+
KubernetesAgentAddressMode,
|
|
155
|
+
KubernetesClusterAccess,
|
|
156
|
+
} from './backends/kubernetes/index.js'
|
|
157
|
+
// Egress translation types named by `KubernetesBackendConfig.egress` — see
|
|
158
|
+
// `backends/kubernetes/egress-policy.ts` for what each engine can express,
|
|
159
|
+
// and what the two Kubernetes-only kinds (`no-network`, `public-internet`)
|
|
160
|
+
// mean that the shared `EgressPolicy` union has no word for.
|
|
161
|
+
export type {
|
|
162
|
+
EgressProfileLabel,
|
|
163
|
+
KubernetesCiliumDnsNarrowing,
|
|
164
|
+
KubernetesCiliumEgressNarrowing,
|
|
165
|
+
KubernetesEgressConfig,
|
|
166
|
+
KubernetesEgressEngine,
|
|
167
|
+
KubernetesEgressPolicy,
|
|
168
|
+
KubernetesEgressVerification,
|
|
169
|
+
KubernetesOnlyEgressPolicy,
|
|
170
|
+
KubernetesPerSandboxEgressConfig,
|
|
171
|
+
} from './backends/kubernetes/egress-policy.js'
|
|
172
|
+
// Per-sandbox egress — `config.egress.perSandbox`, which makes
|
|
173
|
+
// `Sandbox.setNetworkPolicy` PRESENT on a kubernetes TASK handle instead of
|
|
174
|
+
// omitted. Everything a host needs to configure it and to catch its two
|
|
175
|
+
// refusals by class: a capability declared in a way this backend cannot
|
|
176
|
+
// honour (wiring time), and a write refused because the operator-applied
|
|
177
|
+
// admission fence that bounds it is not there — or cannot be read, which is
|
|
178
|
+
// a different file to fix (call time, nothing written in either case).
|
|
179
|
+
// `KubernetesNetworkPolicyHostError` is the third: an `allowedHosts` entry
|
|
180
|
+
// that is not a hostname, or one the configured narrowing cannot express.
|
|
181
|
+
// `KubernetesOwnerUidMissingError` is the fourth, and the only one raised
|
|
182
|
+
// from an ACQUIRE — the object this backend created reported no uid, so a
|
|
183
|
+
// policy written for it would be an orphan.
|
|
184
|
+
// `KubernetesWorkspacePerSandboxEgressConfigError` is the fifth: the same
|
|
185
|
+
// option reaching `createKubernetesWorkspace`, whose handle never carries
|
|
186
|
+
// `setNetworkPolicy` — refused there rather than accepted and ignored.
|
|
187
|
+
// See `backends/kubernetes/per-sandbox-policy.ts` — and, for
|
|
188
|
+
// `KubernetesNetworkPolicyHostError`, which the config-level translation
|
|
189
|
+
// refuses the same entries with, `backends/kubernetes/egress-policy.ts`.
|
|
190
|
+
export { KubernetesPerSandboxEgressConfigError } from './backends/kubernetes/egress-policy.js'
|
|
191
|
+
export { KubernetesWorkspacePerSandboxEgressConfigError } from './backends/kubernetes/egress-policy.js'
|
|
192
|
+
export { DEFAULT_PER_SANDBOX_EGRESS_LABEL_KEY } from './backends/kubernetes/egress-policy.js'
|
|
193
|
+
export {
|
|
194
|
+
KubernetesAdmissionFenceMissingError,
|
|
195
|
+
KubernetesAdmissionFenceUnreadableError,
|
|
196
|
+
KubernetesNetworkPolicyHostError,
|
|
197
|
+
KubernetesOwnerUidMissingError,
|
|
198
|
+
PER_SANDBOX_POLICY_NAME_PREFIX,
|
|
199
|
+
} from './backends/kubernetes/per-sandbox-policy.js'
|
|
200
|
+
// The default label key `KubernetesEgressConfig.profile` is written under.
|
|
201
|
+
// Exported because an operator has to name that key's DOMAIN in the
|
|
202
|
+
// controller's `allowed-label-domains` allowlist before a profiled claim is
|
|
203
|
+
// accepted, and reading it off the package beats copying a string out of a
|
|
204
|
+
// document.
|
|
205
|
+
export { DEFAULT_EGRESS_PROFILE_LABEL_KEY } from './backends/kubernetes/egress-policy.js'
|
|
206
|
+
// The egress union check's refusal and the shapes it reports, so a host can
|
|
207
|
+
// catch an over-wide policy by class and print which policy it was. Separate
|
|
208
|
+
// from `KubernetesEgressPolicyMismatchError` (the ONE named object drifting)
|
|
209
|
+
// and from `KubernetesIngressPolicyError` (the agent port being reachable):
|
|
210
|
+
// three refusals on one create path, told apart by class.
|
|
211
|
+
export type {
|
|
212
|
+
EgressPolicyRefusal,
|
|
213
|
+
EgressPolicyVerdict,
|
|
214
|
+
ExaminedEgressPolicy,
|
|
215
|
+
} from './backends/kubernetes/egress-policy.js'
|
|
216
|
+
export {
|
|
217
|
+
KubernetesEgressNarrowingUnsupportedError,
|
|
218
|
+
KubernetesEgressPolicyConfigError,
|
|
219
|
+
KubernetesEgressPolicyUnionError,
|
|
220
|
+
} from './backends/kubernetes/egress-policy.js'
|
|
221
|
+
// What an egress PROFILE adds to the refusals above. Two are thrown: a
|
|
222
|
+
// profile this backend will not emit (while the host is still being wired),
|
|
223
|
+
// and a bound pod that never carried the label this backend asked the
|
|
224
|
+
// controller for — refused rather than admitted, because admitting it would
|
|
225
|
+
// run the sandbox under whatever policy DOES select it. The third is never
|
|
226
|
+
// thrown on its own: a claim the controller refused because a label key's
|
|
227
|
+
// domain is not on its allowlist still comes out as
|
|
228
|
+
// `KubernetesAcquireError { reason: 'claim-rejected' }`, and
|
|
229
|
+
// `KubernetesPodLabelsRejectedError` is that error's `cause`, naming the map
|
|
230
|
+
// that was sent and the config key that moves it.
|
|
231
|
+
export {
|
|
232
|
+
KubernetesEgressProfileConfigError,
|
|
233
|
+
KubernetesPodLabelNotObservedError,
|
|
234
|
+
KubernetesPodLabelsRejectedError,
|
|
235
|
+
} from './backends/kubernetes/egress-policy.js'
|
|
236
|
+
// Ingress verification types named by `KubernetesBackendConfig.ingress`, plus
|
|
237
|
+
// the refusal a create raises when no applied policy closes the agent port —
|
|
238
|
+
// catchable by class, and distinct from every other refusal on that path. See
|
|
239
|
+
// `backends/kubernetes/ingress-policy.ts`.
|
|
240
|
+
export type {
|
|
241
|
+
ExaminedIngressPolicy,
|
|
242
|
+
IngressPolicyRefusal,
|
|
243
|
+
IngressPolicyVerdict,
|
|
244
|
+
KubernetesIngressConfig,
|
|
245
|
+
KubernetesIngressEngine,
|
|
246
|
+
/** @deprecated Renamed to `UnreadPolicySource`; both checks report it. */
|
|
247
|
+
UnreadIngressPolicySource,
|
|
248
|
+
UnreadPolicySource,
|
|
249
|
+
} from './backends/kubernetes/ingress-policy.js'
|
|
250
|
+
export { KubernetesIngressPolicyError } from './backends/kubernetes/ingress-policy.js'
|
|
251
|
+
// The errors a caller of a kubernetes sandbox has to be able to catch BY
|
|
252
|
+
// CLASS rather than by matching a message: an acquire refused because the
|
|
253
|
+
// guest is not deprivileged, a call after the handle ended (this host
|
|
254
|
+
// destroyed it, or the cluster deleted it), and a rejected agent token.
|
|
255
|
+
export {
|
|
256
|
+
KubernetesPrivilegeProbeError,
|
|
257
|
+
type PrivilegeProbeFailure,
|
|
258
|
+
type ProcStatusPrivileges,
|
|
259
|
+
} from './backends/kubernetes/privilege-probe.js'
|
|
260
|
+
export {
|
|
261
|
+
KubernetesSandboxDestroyedError,
|
|
262
|
+
KubernetesSandboxGoneError,
|
|
263
|
+
} from './backends/kubernetes/sandbox.js'
|
|
264
|
+
export {
|
|
265
|
+
KubernetesAgentAddressUnresolvableError,
|
|
266
|
+
// A guest that has fenced itself refuses one CALL here, not the
|
|
267
|
+
// workspace: the Firecracker tier's mapping of the same refusal retires
|
|
268
|
+
// the sandbox, which on a workspace would take the pod away from every
|
|
269
|
+
// other holder. See `backends/kubernetes/transport.ts`.
|
|
270
|
+
KubernetesAgentRetiringError,
|
|
271
|
+
KubernetesAgentUnauthorizedError,
|
|
272
|
+
} from './backends/kubernetes/transport.js'
|
|
273
|
+
// A workspace command that can outlive the connection watching it: the
|
|
274
|
+
// options that ask for one, and the three refusals a caller has to be able
|
|
275
|
+
// to catch BY CLASS — a guest image too old to keep output, an execution
|
|
276
|
+
// that can no longer be attached to (past retention, or in a replaced
|
|
277
|
+
// pod), and an observation this host gave up on WITHOUT cancelling, which
|
|
278
|
+
// names the id and the byte offset another process resumes from.
|
|
279
|
+
export type {
|
|
280
|
+
KubernetesAttachExecutionOptions,
|
|
281
|
+
KubernetesAttachRefusal,
|
|
282
|
+
KubernetesDetachedExecOptions,
|
|
283
|
+
} from './backends/kubernetes/transport.js'
|
|
284
|
+
export {
|
|
285
|
+
KubernetesExecutionAttachUnsupportedError,
|
|
286
|
+
KubernetesExecutionDetachedError,
|
|
287
|
+
KubernetesExecutionNotAttachableError,
|
|
288
|
+
} from './backends/kubernetes/transport.js'
|
|
289
|
+
// Guest sessions: a workspace terminal or background program that outlives
|
|
290
|
+
// the connection — and the host process — that started it. The options that
|
|
291
|
+
// name one, the rows `listSessions()` returns, the `BackgroundJobOutput`
|
|
292
|
+
// shape `readSession()` answers in, and the three refusals a caller catches
|
|
293
|
+
// BY CLASS: a guest image with no session registry, a session the guest
|
|
294
|
+
// answered about and refused (past retention, or in a replaced pod), and an
|
|
295
|
+
// attachment that ended while its program went on running.
|
|
296
|
+
export type {
|
|
297
|
+
KubernetesAttachTerminalOptions,
|
|
298
|
+
KubernetesOpenTerminalOptions,
|
|
299
|
+
KubernetesReadSessionOptions,
|
|
300
|
+
KubernetesSessionOutput,
|
|
301
|
+
KubernetesSessionRefusal,
|
|
302
|
+
KubernetesSessionSummary,
|
|
303
|
+
KubernetesSessionTerminal,
|
|
304
|
+
KubernetesStartDetachedOptions,
|
|
305
|
+
KubernetesWorkspaceTerminal,
|
|
306
|
+
} from './backends/kubernetes/transport.js'
|
|
307
|
+
export {
|
|
308
|
+
KubernetesSessionRefusedError,
|
|
309
|
+
KubernetesSessionsUnsupportedError,
|
|
310
|
+
} from './backends/kubernetes/transport.js'
|
|
311
|
+
export { AgentSessionDetachedError } from './backends/firecracker/transport.js'
|
|
312
|
+
// The `sessions` healthz feature string, and the session vocabulary its
|
|
313
|
+
// frames use. Same reason as the two feature strings above: a host asserting
|
|
314
|
+
// what an image can do should name the string rather than repeat the literal.
|
|
315
|
+
export {
|
|
316
|
+
SESSIONS_FEATURE,
|
|
317
|
+
type SessionDetachReason,
|
|
318
|
+
type SessionKind,
|
|
319
|
+
type SessionState,
|
|
320
|
+
} from './backends/firecracker/protocol.js'
|
|
321
|
+
// Quiesce: stop every process a workspace's guest is running, while the
|
|
322
|
+
// agent goes on serving, so a capture taken next is one nobody is writing
|
|
323
|
+
// under. The report a host reads, the two refusals it catches by class, and
|
|
324
|
+
// — for the same reason as the feature strings above — the `quiesce` string
|
|
325
|
+
// itself and the scope vocabulary its report is written in.
|
|
326
|
+
export type {
|
|
327
|
+
KubernetesQuiesceOptions,
|
|
328
|
+
KubernetesWorkspaceQuiesceRequest,
|
|
329
|
+
} from './backends/kubernetes/workspace.js'
|
|
330
|
+
export type { KubernetesQuiesceReport } from './backends/kubernetes/transport.js'
|
|
331
|
+
export {
|
|
332
|
+
KubernetesQuiesceUnconfirmedError,
|
|
333
|
+
KubernetesQuiesceUnsupportedError,
|
|
334
|
+
} from './backends/kubernetes/transport.js'
|
|
335
|
+
export {
|
|
336
|
+
QUIESCE_FEATURE,
|
|
337
|
+
type QuiesceScope,
|
|
338
|
+
type QuiescedProcess,
|
|
339
|
+
} from './backends/firecracker/protocol.js'
|
|
340
|
+
// Flush: put a workspace's writes on its device on purpose, rather than
|
|
341
|
+
// leaving them to whatever the guest kernel had written back when the pod
|
|
342
|
+
// stopped. `suspend()` runs it by default; the verb, its options, the report
|
|
343
|
+
// and the three named outcomes are exported for the hosts that flush at a
|
|
344
|
+
// moment of their own — before a snapshot, before a drain — and for the
|
|
345
|
+
// `flush` string itself, for the same reason as the feature strings above.
|
|
346
|
+
// Only ONE of the three is ever thrown at a caller by a suspend
|
|
347
|
+
// (`KubernetesFlushUnconfirmedError`, a guest that answered and could not
|
|
348
|
+
// confirm); the other two are what `onFlushUnsupported` and
|
|
349
|
+
// `onFlushUnreachable` are handed when the suspend goes ahead anyway, and a
|
|
350
|
+
// host that wants to act on either has to be able to name the class.
|
|
351
|
+
export type {
|
|
352
|
+
KubernetesFlushOptions,
|
|
353
|
+
KubernetesWorkspaceFlushRequest,
|
|
354
|
+
} from './backends/kubernetes/workspace.js'
|
|
355
|
+
export type { KubernetesFlushReport } from './backends/kubernetes/transport.js'
|
|
356
|
+
export {
|
|
357
|
+
KubernetesFlushUnconfirmedError,
|
|
358
|
+
KubernetesFlushUnreachableError,
|
|
359
|
+
KubernetesFlushUnsupportedError,
|
|
360
|
+
} from './backends/kubernetes/transport.js'
|
|
361
|
+
export { FLUSH_FEATURE } from './backends/firecracker/protocol.js'
|
|
362
|
+
// The API-request bound and the error it raises. Exported because
|
|
363
|
+
// "distinguishable from a caller abort and from every other failure, by
|
|
364
|
+
// type" is only true for a host that can name the class — and because a
|
|
365
|
+
// host that sets `apiRequestTimeoutMs` wants the default and the floor it is
|
|
366
|
+
// choosing against.
|
|
367
|
+
// `KubernetesHttpMethod` rides along because `KubernetesApiTimeoutError.verb`
|
|
368
|
+
// is one: a caller that can catch the class but cannot name the type of the
|
|
369
|
+
// field it is reading is back to inlining the union or reaching for `any`.
|
|
370
|
+
export {
|
|
371
|
+
DEFAULT_API_REQUEST_TIMEOUT_MS,
|
|
372
|
+
// Every API failure the four classes below do not name: a connect
|
|
373
|
+
// failure, and every non-2xx status outside 401/403/404/409/410. It
|
|
374
|
+
// carries the status and the `Retry-After` the server sent, because a
|
|
375
|
+
// burst past node capacity (429) and an API server that is down (a
|
|
376
|
+
// connect failure) were otherwise the same plain `Error`, separable only
|
|
377
|
+
// by matching a message any release is free to reword. It carries no
|
|
378
|
+
// retry policy — see `backends/kubernetes/index.ts` for who decides that.
|
|
379
|
+
KubernetesApiError,
|
|
380
|
+
type KubernetesApiFailureTransport,
|
|
381
|
+
KubernetesApiTimeoutError,
|
|
382
|
+
type KubernetesHttpMethod,
|
|
383
|
+
// A conditional write the API server would not apply. A host that fences
|
|
384
|
+
// its workspaces with a holder epoch normally catches
|
|
385
|
+
// `KubernetesWorkspacePreconditionError` instead — this one survives only
|
|
386
|
+
// when the object kept changing under the write or the patch body was
|
|
387
|
+
// wrong, and a caller that cannot name the class cannot tell it from a
|
|
388
|
+
// cluster failure.
|
|
389
|
+
KubernetesPatchNotAppliedError,
|
|
390
|
+
MIN_API_REQUEST_TIMEOUT_MS,
|
|
391
|
+
} from './backends/kubernetes/k8s-client.js'
|
|
392
|
+
// The three statuses the client maps to a class of their own, so a caller can
|
|
393
|
+
// treat "already gone" as the state a teardown was asking for, re-read after a
|
|
394
|
+
// 409, and tell a rejected credential from a cluster failure — by class, which
|
|
395
|
+
// is the only way that survives a reworded message. They have been thrown
|
|
396
|
+
// since the backend existed and were reachable only by importing a deep path.
|
|
397
|
+
export {
|
|
398
|
+
KubernetesAlreadyGoneError,
|
|
399
|
+
KubernetesConflictError,
|
|
400
|
+
KubernetesCredentialError,
|
|
401
|
+
} from './backends/kubernetes/k8s-client.js'
|
|
402
|
+
// Why an acquire was refused, as a field rather than as prose: the seven
|
|
403
|
+
// reasons, the class that carries one, and the measured list of controller
|
|
404
|
+
// `Ready=False` reasons that mean "decided" rather than "not yet". A host
|
|
405
|
+
// deciding whether to retry, to fail the run or to page an operator reads
|
|
406
|
+
// `reason` and `retryable`; `cause` is the original failure, so a host that
|
|
407
|
+
// already catches `ReadinessPollTimeout` or `KubernetesApiTimeoutError` finds
|
|
408
|
+
// it there. See `backends/kubernetes/index.ts`.
|
|
409
|
+
export {
|
|
410
|
+
KubernetesAcquireError,
|
|
411
|
+
type KubernetesAcquireFailureReason,
|
|
412
|
+
// The poll's own give-up, by type. It is the `cause` of a `'not-ready'`
|
|
413
|
+
// acquire refusal and is raised directly by the workspace lifecycle, which
|
|
414
|
+
// does not go through acquire.
|
|
415
|
+
ReadinessPollTimeout,
|
|
416
|
+
TERMINAL_CLAIM_REASONS,
|
|
417
|
+
} from './backends/kubernetes/index.js'
|
|
418
|
+
// The three ways egress verification refuses: a policy this engine cannot
|
|
419
|
+
// express, no applied object at all, and an applied object that does not
|
|
420
|
+
// match what this configuration translates to. Catchable by class for the
|
|
421
|
+
// same reason the ingress refusal above is — an operator debugging two
|
|
422
|
+
// default-on refusals in one release should not have to read messages to
|
|
423
|
+
// tell them apart.
|
|
424
|
+
export {
|
|
425
|
+
KubernetesEgressPolicyMismatchError,
|
|
426
|
+
KubernetesEgressPolicyNotAppliedError,
|
|
427
|
+
KubernetesUnenforceableEgressPolicyError,
|
|
428
|
+
} from './backends/kubernetes/egress-policy.js'
|
|
429
|
+
/** Default `KubernetesBackendConfig.streamHeartbeatMs` — see there. */
|
|
430
|
+
export { DEFAULT_STREAM_HEARTBEAT_MS } from './backends/kubernetes/index.js'
|
|
431
|
+
// Crash recovery and headroom for the task path: label a claim with a
|
|
432
|
+
// host-supplied identity (`KubernetesBackendConfig.claimLabels`), find and
|
|
433
|
+
// release a predecessor's claims by that label, and read pool headroom
|
|
434
|
+
// before admitting more work. All three are additive — a host that sets no
|
|
435
|
+
// `claimLabels` and calls neither function sees no change at all.
|
|
436
|
+
export type {
|
|
437
|
+
KubernetesReadTaskCapacityOptions,
|
|
438
|
+
KubernetesReleaseTaskSandboxesOptions,
|
|
439
|
+
KubernetesTaskCapacity,
|
|
440
|
+
} from './backends/kubernetes/index.js'
|
|
441
|
+
// What a CLAIM-ONLY host is allowed to do, as data: the verbs the pool-only
|
|
442
|
+
// path issues, each pinned to its call site in `backends/kubernetes/rbac.ts`.
|
|
443
|
+
// `k8s/manifests/rbac-claimant.yaml` grants exactly this and a test parses
|
|
444
|
+
// that file and compares it here, so an operator who has to prove a live
|
|
445
|
+
// `Role` carries no more than this backend needs compares against the same
|
|
446
|
+
// constant rather than against a list copied out of a page.
|
|
447
|
+
export {
|
|
448
|
+
KUBERNETES_CLAIMANT_RBAC_RULES,
|
|
449
|
+
type KubernetesRbacRule,
|
|
450
|
+
type KubernetesRbacVerb,
|
|
451
|
+
} from './backends/kubernetes/rbac.js'
|
|
452
|
+
// The persistent workspace: a `Sandbox` that keeps a block disk across a
|
|
453
|
+
// suspend, the union naming how a handle came by its object, plus the four
|
|
454
|
+
// errors its lifecycle can refuse with — a template that cannot carry a disk,
|
|
455
|
+
// a standing object that does not match this configuration, a call on a
|
|
456
|
+
// suspended workspace, and a suspend whose pod outlived the wait. Declared in
|
|
457
|
+
// `@namzu/sandbox` rather than on the SDK's `Sandbox` — see
|
|
458
|
+
// `backends/kubernetes/workspace.ts`.
|
|
459
|
+
export type {
|
|
460
|
+
KubernetesKillSessionOptions,
|
|
461
|
+
KubernetesWorkspace,
|
|
462
|
+
KubernetesWorkspaceAgentState,
|
|
463
|
+
KubernetesWorkspaceCancellationNotice,
|
|
464
|
+
KubernetesWorkspaceDestroyOptions,
|
|
465
|
+
KubernetesWorkspaceOptions,
|
|
466
|
+
KubernetesWorkspaceOrigin,
|
|
467
|
+
KubernetesWorkspaceStartFailurePolicy,
|
|
468
|
+
KubernetesWorkspaceSummary,
|
|
469
|
+
KubernetesWorkspaceSuspendOptions,
|
|
470
|
+
KubernetesWorkspaceSuspensionNotice,
|
|
471
|
+
KubernetesWorkspaceTransitionOptions,
|
|
472
|
+
} from './backends/kubernetes/workspace.js'
|
|
473
|
+
// What a workspace handle is bound to, what it says when the guest behind it
|
|
474
|
+
// is replaced, and the two errors that identity produces. A host that keeps
|
|
475
|
+
// per-workspace state — which processes it started, what is on the disk —
|
|
476
|
+
// subscribes to `onGuestRestart` and compares `identity`; both are useless to
|
|
477
|
+
// a host that cannot name their types.
|
|
478
|
+
export type {
|
|
479
|
+
KubernetesGuestEvidence,
|
|
480
|
+
KubernetesGuestRestart,
|
|
481
|
+
KubernetesGuestRestartReason,
|
|
482
|
+
KubernetesWorkspaceIdentity,
|
|
483
|
+
} from './backends/kubernetes/identity.js'
|
|
484
|
+
export {
|
|
485
|
+
// The command's outcome is unknown AND the guest it ran in is gone. A
|
|
486
|
+
// subclass of `RemoteCancellationUnknownError`, so a host catching the
|
|
487
|
+
// base class keeps catching it; what it adds is which guest the command
|
|
488
|
+
// started on and which one is there now.
|
|
489
|
+
KubernetesWorkspaceGuestGoneError,
|
|
490
|
+
// A different Sandbox now stands under the workspace's deterministic
|
|
491
|
+
// name, or none does. The handle refuses rather than following it — the
|
|
492
|
+
// disk behind the name is not the disk it was opened on.
|
|
493
|
+
KubernetesWorkspaceReplacedError,
|
|
494
|
+
} from './backends/kubernetes/identity.js'
|
|
495
|
+
export {
|
|
496
|
+
KubernetesWorkspaceDiskError,
|
|
497
|
+
KubernetesWorkspaceMismatchError,
|
|
498
|
+
// A lifecycle write refused because this caller's holder epoch has been
|
|
499
|
+
// overtaken: the workspace belongs to another process now, nothing on the
|
|
500
|
+
// cluster changed and nothing about the handle changed. A host that fences
|
|
501
|
+
// its workspaces has to be able to tell this from a cluster failure, which
|
|
502
|
+
// is the whole reason the write is conditional.
|
|
503
|
+
KubernetesWorkspacePreconditionError,
|
|
504
|
+
KubernetesWorkspaceSuspendTimeoutError,
|
|
505
|
+
KubernetesWorkspaceSuspendedError,
|
|
506
|
+
} from './backends/kubernetes/workspace.js'
|
|
86
507
|
|
|
87
508
|
// ---------------------------------------------------------------------------
|
|
88
509
|
// Backend strategy
|
|
@@ -120,6 +541,7 @@ export type SandboxBackendConfig =
|
|
|
120
541
|
| ContainerBackendConfig
|
|
121
542
|
| ACIStandbyPoolBackendConfig
|
|
122
543
|
| MicroVMBackendConfig
|
|
544
|
+
| KubernetesBackendConfig
|
|
123
545
|
|
|
124
546
|
/**
|
|
125
547
|
* Azure Container Instances Standby Pool backend. Container tier,
|
|
@@ -375,6 +797,200 @@ export interface AgentSnapshotRef {
|
|
|
375
797
|
readonly version: string
|
|
376
798
|
}
|
|
377
799
|
|
|
800
|
+
/**
|
|
801
|
+
* `microvm` tier, against a Kubernetes cluster running the agent-sandbox
|
|
802
|
+
* controller (kubernetes-sigs/agent-sandbox) with a VM-isolating
|
|
803
|
+
* RuntimeClass such as Kata.
|
|
804
|
+
*
|
|
805
|
+
* `microvm` because the tier names the strength of the boundary rather than
|
|
806
|
+
* the orchestrator behind it: a pod scheduled onto a Kata RuntimeClass runs
|
|
807
|
+
* in a hardware-virtualized guest, and the same field on the same tier is how
|
|
808
|
+
* a host says "give me a VM, I do not care who starts it".
|
|
809
|
+
*
|
|
810
|
+
* Sandboxes are claimed out of a `SandboxWarmPool` when {@link warmPoolName}
|
|
811
|
+
* names one, which is what makes the acquire sub-second; without it every
|
|
812
|
+
* create is a `Sandbox` built from {@link sandboxTemplateName}'s podTemplate
|
|
813
|
+
* and pays a full pod start. This package speaks the API server with bare
|
|
814
|
+
* `fetch` and carries no Kubernetes client dependency: credentials arrive
|
|
815
|
+
* through {@link access}, and kubeconfig parsing (context merging,
|
|
816
|
+
* exec credential plugins) stays in the host that owns it.
|
|
817
|
+
*/
|
|
818
|
+
export interface KubernetesBackendConfig {
|
|
819
|
+
readonly tier: 'microvm'
|
|
820
|
+
readonly service: 'kubernetes'
|
|
821
|
+
/** Namespace the claims, sandboxes and their pods live in. */
|
|
822
|
+
readonly namespace: string
|
|
823
|
+
/**
|
|
824
|
+
* How to reach the API server. `{ inCluster: true }` reads the projected
|
|
825
|
+
* ServiceAccount volume and the kubelet's `KUBERNETES_SERVICE_*` env, which
|
|
826
|
+
* is the production path; otherwise the host supplies the server URL, an
|
|
827
|
+
* optional cluster CA and a `getToken()` callback — the same boundary this
|
|
828
|
+
* package already draws for ACI's `getArmToken` and Firecracker's
|
|
829
|
+
* `getToken`.
|
|
830
|
+
*/
|
|
831
|
+
readonly access: KubernetesClusterAccess
|
|
832
|
+
/**
|
|
833
|
+
* `SandboxTemplate` whose `podTemplate` a POOL-LESS create copies into the
|
|
834
|
+
* `Sandbox` it posts. Required because `Sandbox.spec` has no `templateRef`
|
|
835
|
+
* — only a `SandboxWarmPool` references a template — so the pod spec has to
|
|
836
|
+
* be carried across by the client. The warm path does not read it; the
|
|
837
|
+
* pool's own `sandboxTemplateRef` decides there.
|
|
838
|
+
*/
|
|
839
|
+
readonly sandboxTemplateName: string
|
|
840
|
+
/**
|
|
841
|
+
* `SandboxWarmPool` to claim from. Absent → every create posts a `Sandbox`
|
|
842
|
+
* directly, because `SandboxClaim.spec.warmPoolRef` is a required field and
|
|
843
|
+
* a pool-less claim does not exist in the API.
|
|
844
|
+
*/
|
|
845
|
+
readonly warmPoolName?: string
|
|
846
|
+
/** TCP port the in-pod guest agent listens on. Default 1024. */
|
|
847
|
+
readonly agentPort?: number
|
|
848
|
+
/**
|
|
849
|
+
* Which of a sandbox's two addresses the transport dials.
|
|
850
|
+
*
|
|
851
|
+
* `'service'` (default) is the Sandbox's `status.serviceFQDN`, which
|
|
852
|
+
* outlives the pod and is re-resolved on every dial — and which ONLY the
|
|
853
|
+
* cluster's own DNS answers. A host running outside the cluster fails
|
|
854
|
+
* every call at name resolution, readiness included, so it reads as a
|
|
855
|
+
* sandbox that never came up.
|
|
856
|
+
*
|
|
857
|
+
* `'pod-ip'` dials the bound pod's IP, read from the same `GET` that
|
|
858
|
+
* reads its bind token. For a host outside the cluster with a route to
|
|
859
|
+
* the pod network. It needs that route and a `NetworkPolicy` admitting
|
|
860
|
+
* the host's address range on {@link agentPort}; the IP dies with its
|
|
861
|
+
* pod, which the backend covers by re-reading it on every resume and once
|
|
862
|
+
* after a connect failure. Nothing else changes: same bind token, same
|
|
863
|
+
* privilege probe, same egress verification.
|
|
864
|
+
*/
|
|
865
|
+
readonly agentAddress?: KubernetesAgentAddressMode
|
|
866
|
+
/** Delay between readiness polls. Default 50ms. */
|
|
867
|
+
readonly readyPollIntervalMs?: number
|
|
868
|
+
/** Total deadline from create to an addressed, Ready sandbox. Default 60000ms. */
|
|
869
|
+
readonly readyTimeoutMs?: number
|
|
870
|
+
/**
|
|
871
|
+
* Wall-clock lifetime written into every object this backend creates, so a
|
|
872
|
+
* host that dies mid-run costs the cluster one expiry rather than a leaked
|
|
873
|
+
* sandbox. Default 3600.
|
|
874
|
+
*/
|
|
875
|
+
readonly claimTtlSeconds?: number
|
|
876
|
+
/**
|
|
877
|
+
* Every lease-renewal failure that is not "the object is already gone".
|
|
878
|
+
*
|
|
879
|
+
* The handle renews its own `shutdownTime` every half-TTL for as long as
|
|
880
|
+
* it is alive, so a run that outlives `claimTtlSeconds` keeps its pod.
|
|
881
|
+
* A failed renewal is retried on a short capped backoff — starting at one
|
|
882
|
+
* second, not the next half-TTL tick — so a single API blip near a
|
|
883
|
+
* scheduled renewal gets several more chances before anything expires;
|
|
884
|
+
* this callback is where the diagnostic goes, because `@namzu/sandbox`
|
|
885
|
+
* owns no logger and reads none from module scope. Setting it changes
|
|
886
|
+
* nothing about behaviour.
|
|
887
|
+
*/
|
|
888
|
+
readonly onLeaseRenewalError?: (error: unknown) => void
|
|
889
|
+
/**
|
|
890
|
+
* RuntimeClass for a POOL-LESS create. Refused together with
|
|
891
|
+
* {@link warmPoolName}: a pooled sandbox is already running under the
|
|
892
|
+
* RuntimeClass its `SandboxTemplate` named, and a claim cannot change it —
|
|
893
|
+
* so accepting it there would quietly drop the choice of VM boundary.
|
|
894
|
+
*/
|
|
895
|
+
readonly runtimeClassName?: string
|
|
896
|
+
/**
|
|
897
|
+
* Egress policy this backend expects an operator to have applied as a
|
|
898
|
+
* `NetworkPolicy` (or, under `engine: 'cilium'`, a `CiliumNetworkPolicy`)
|
|
899
|
+
* scoped to every Sandbox this backend produces. Unset means this backend
|
|
900
|
+
* neither computes nor checks one — the cluster's default posture (the
|
|
901
|
+
* `SandboxTemplate`'s own managed `NetworkPolicy`) is all that applies.
|
|
902
|
+
*
|
|
903
|
+
* This is a CONFIG-level, whole-backend policy, not a per-`create()` one:
|
|
904
|
+
* `SandboxBackendOptions.egress` is still refused by name (see
|
|
905
|
+
* `backends/kubernetes/index.ts`'s `assertEnforceable`), because the
|
|
906
|
+
* enforcement point is one object attached to the template and cannot be
|
|
907
|
+
* rewritten per running sandbox. `static` and `resolver` — hostname
|
|
908
|
+
* allowlists — throw a named error at construction unless `engine` is
|
|
909
|
+
* `'cilium'`: core `NetworkPolicy` has no FQDN concept at all.
|
|
910
|
+
*
|
|
911
|
+
* `policy` also takes two kinds that exist only here, because only a
|
|
912
|
+
* `NetworkPolicy` can express them: `{ kind: 'no-network' }` (nothing
|
|
913
|
+
* leaves the pod, the cluster resolver included — which `'deny-all'` never
|
|
914
|
+
* meant, since it allows DNS and a cluster resolver forwards outside
|
|
915
|
+
* names) and `{ kind: 'public-internet', exceptCidrs? }` (the internet,
|
|
916
|
+
* minus the private ranges, carrier-grade NAT, link-local and one cloud
|
|
917
|
+
* platform endpoint). `'deny-all'` and `'allow-all'` emit exactly the
|
|
918
|
+
* manifests they always have.
|
|
919
|
+
*
|
|
920
|
+
* **Setting this now checks the UNION.** Since every policy selecting a
|
|
921
|
+
* pod is unioned by the API server, the check reads the named object AND
|
|
922
|
+
* enumerates the namespace's policies, refusing when any of them lets out
|
|
923
|
+
* more than `policy` does. `verify: 'named-object-only'` restores the
|
|
924
|
+
* single-object check exactly. See
|
|
925
|
+
* `docs/sdk/kubernetes-sandbox.md`'s egress section.
|
|
926
|
+
*/
|
|
927
|
+
readonly egress?: KubernetesEgressConfig
|
|
928
|
+
/**
|
|
929
|
+
* Whether this backend proves, before creating a sandbox, that an applied
|
|
930
|
+
* policy actually closes {@link agentPort} on the pod it is about to hand
|
|
931
|
+
* back — and against which policy resources.
|
|
932
|
+
*
|
|
933
|
+
* **Unset means verify.** This is the one field here whose absent value is
|
|
934
|
+
* the strict one, because the deployment that needs the check is the one
|
|
935
|
+
* that would never have switched it on: the guest agent's own source calls
|
|
936
|
+
* the network rule in front of its port the boundary, and until this field
|
|
937
|
+
* existed nothing confirmed there was one.
|
|
938
|
+
*
|
|
939
|
+
* `{ engine: 'cilium' }` also enumerates that CNI's own policy CRD;
|
|
940
|
+
* `engine` otherwise defaults to `egress?.engine ?? 'core'`.
|
|
941
|
+
*
|
|
942
|
+
* `'unverified'` reads no policy and issues no request. It is the
|
|
943
|
+
* supported answer for a deployment whose boundary a namespaced Role
|
|
944
|
+
* cannot see — a cluster-scoped policy, a service mesh, a cloud security
|
|
945
|
+
* group — and it is a claim the deployment makes on purpose rather than a
|
|
946
|
+
* default it inherits. See `docs/sdk/kubernetes-sandbox.md`'s ingress
|
|
947
|
+
* section.
|
|
948
|
+
*/
|
|
949
|
+
readonly ingress?: KubernetesIngressConfig
|
|
950
|
+
/**
|
|
951
|
+
* How long a single Kubernetes API request may take, end to end —
|
|
952
|
+
* resolving the token, connecting, and reading the reply. Default
|
|
953
|
+
* `30000`; minimum `1000`; there is no value that turns it off.
|
|
954
|
+
*
|
|
955
|
+
* The caller's `signal` is optional everywhere and several of this
|
|
956
|
+
* backend's requests are SHARED flights that run under whichever caller
|
|
957
|
+
* arrived first, so a signal-less `destroy()` against an API server that
|
|
958
|
+
* accepted a request and never answered used to pin every later caller
|
|
959
|
+
* joined to it. Expiry rejects with `KubernetesApiTimeoutError`, which
|
|
960
|
+
* says nothing about whether the request was applied — the paths that
|
|
961
|
+
* send one already cope with not knowing.
|
|
962
|
+
*/
|
|
963
|
+
readonly apiRequestTimeoutMs?: number
|
|
964
|
+
/**
|
|
965
|
+
* Interval of the liveness heartbeat `openTerminal` and
|
|
966
|
+
* `openTcpConnection` streams negotiate with the guest. Default `15000`;
|
|
967
|
+
* `0` sends none, which is exactly how every release before this one
|
|
968
|
+
* behaved.
|
|
969
|
+
*
|
|
970
|
+
* A quiet shell is healthy, so nothing replaced the read-idle timer the
|
|
971
|
+
* transport clears once a stream is ready: a partition that delivered no
|
|
972
|
+
* FIN and no RST left `exited`/`closed` unresolved on the host and the
|
|
973
|
+
* shell's process group alive in the guest. Three missed intervals end
|
|
974
|
+
* the stream on both sides. It is negotiated per stream — the guest
|
|
975
|
+
* echoes the interval in its `ready` event and sends nothing new unless
|
|
976
|
+
* it did — so an older guest image behaves exactly as it does today.
|
|
977
|
+
*/
|
|
978
|
+
readonly streamHeartbeatMs?: number
|
|
979
|
+
/**
|
|
980
|
+
* Extra labels written onto every `SandboxClaim` this backend POSTs —
|
|
981
|
+
* `metadata.labels` only, never the pod's own labels. Unset means no
|
|
982
|
+
* labels beyond what the controller itself writes, and every claim body
|
|
983
|
+
* is byte-for-byte what it was before this option existed.
|
|
984
|
+
*
|
|
985
|
+
* The intended use is a host-instance identity, so a restarted host can
|
|
986
|
+
* find and {@link releaseKubernetesTaskSandboxes} a crashed predecessor's
|
|
987
|
+
* claims well before `claimTtlSeconds` reaps them on its own — see
|
|
988
|
+
* {@link readKubernetesTaskCapacity} for reading pool headroom
|
|
989
|
+
* alongside it.
|
|
990
|
+
*/
|
|
991
|
+
readonly claimLabels?: Record<string, string>
|
|
992
|
+
}
|
|
993
|
+
|
|
378
994
|
/**
|
|
379
995
|
* Egress allowlist resolution. Host-supplied policy decides whether
|
|
380
996
|
* an outbound request is allowed before the proxy opens a socket.
|
|
@@ -505,9 +1121,16 @@ export type SandboxProviderConfig =
|
|
|
505
1121
|
readonly backend: ContainerBackendConfig
|
|
506
1122
|
readonly layout: ContainerSandboxLayout
|
|
507
1123
|
})
|
|
1124
|
+
| (SandboxProviderConfigBase & {
|
|
1125
|
+
readonly backend: ACIStandbyPoolBackendConfig
|
|
1126
|
+
readonly layout: ContainerSandboxLayout
|
|
1127
|
+
})
|
|
508
1128
|
| (SandboxProviderConfigBase & {
|
|
509
1129
|
readonly backend: MicroVMBackendConfig
|
|
510
1130
|
})
|
|
1131
|
+
| (SandboxProviderConfigBase & {
|
|
1132
|
+
readonly backend: KubernetesBackendConfig
|
|
1133
|
+
})
|
|
511
1134
|
|
|
512
1135
|
interface SandboxProviderConfigBase {
|
|
513
1136
|
readonly defaultEgress?: EgressPolicy
|
|
@@ -580,6 +1203,40 @@ export function createSandboxProvider(config: SandboxProviderConfig): SandboxPro
|
|
|
580
1203
|
|
|
581
1204
|
function pickBackend(config: SandboxProviderConfig): SandboxBackend {
|
|
582
1205
|
const backend = config.backend
|
|
1206
|
+
// Checked ahead of the `docker` default below: `ACIStandbyPoolBackendConfig`
|
|
1207
|
+
// is a real arm of `SandboxProviderConfig` (see the discriminated union
|
|
1208
|
+
// above), discriminated from `ContainerBackendConfig` by `runtime`. A
|
|
1209
|
+
// plain equality check here narrows `backend` to the ACI shape with no
|
|
1210
|
+
// cast, and — because this branch always returns — narrows it AWAY for
|
|
1211
|
+
// every check below, so the `docker` branch's `backend.runtime ?? 'docker'`
|
|
1212
|
+
// still sees only `ContainerBackendConfig`.
|
|
1213
|
+
if (backend.tier === 'container' && backend.runtime === 'aci-standby-pool') {
|
|
1214
|
+
const layout = (config as Extract<SandboxProviderConfig, { layout: ContainerSandboxLayout }>)
|
|
1215
|
+
.layout
|
|
1216
|
+
const resolved = resolveLayout(layout)
|
|
1217
|
+
return buildAciStandbyPoolBackend({
|
|
1218
|
+
subscriptionId: backend.subscriptionId,
|
|
1219
|
+
resourceGroup: backend.resourceGroup,
|
|
1220
|
+
location: backend.location,
|
|
1221
|
+
standbyPoolResourceId: backend.standbyPoolResourceId,
|
|
1222
|
+
containerGroupProfileResourceId: backend.containerGroupProfileResourceId,
|
|
1223
|
+
...(backend.containerGroupProfileRevision !== undefined
|
|
1224
|
+
? { containerGroupProfileRevision: backend.containerGroupProfileRevision }
|
|
1225
|
+
: {}),
|
|
1226
|
+
layout: resolved,
|
|
1227
|
+
getArmToken: backend.getArmToken,
|
|
1228
|
+
...(backend.subnetId !== undefined ? { subnetId: backend.subnetId } : {}),
|
|
1229
|
+
...(backend.readyPollIntervalMs !== undefined
|
|
1230
|
+
? { readyPollIntervalMs: backend.readyPollIntervalMs }
|
|
1231
|
+
: {}),
|
|
1232
|
+
...(backend.readyTimeoutMs !== undefined ? { readyTimeoutMs: backend.readyTimeoutMs } : {}),
|
|
1233
|
+
...(backend.workerPort !== undefined ? { workerPort: backend.workerPort } : {}),
|
|
1234
|
+
...(backend.armApiVersion !== undefined ? { armApiVersion: backend.armApiVersion } : {}),
|
|
1235
|
+
...(backend.containerNamePrefix !== undefined
|
|
1236
|
+
? { containerNamePrefix: backend.containerNamePrefix }
|
|
1237
|
+
: {}),
|
|
1238
|
+
})
|
|
1239
|
+
}
|
|
583
1240
|
if (backend.tier === 'container' && (backend.runtime ?? 'docker') === 'docker') {
|
|
584
1241
|
// `layout` is required for container-tier backends by the
|
|
585
1242
|
// discriminated union — narrow safely without a non-null
|
|
@@ -605,41 +1262,6 @@ function pickBackend(config: SandboxProviderConfig): SandboxBackend {
|
|
|
605
1262
|
...(backend.labels !== undefined ? { labels: backend.labels } : {}),
|
|
606
1263
|
})
|
|
607
1264
|
}
|
|
608
|
-
if (
|
|
609
|
-
backend.tier === 'container' &&
|
|
610
|
-
(backend as unknown as { runtime?: string }).runtime === 'aci-standby-pool'
|
|
611
|
-
) {
|
|
612
|
-
const aciBackend = backend as unknown as ACIStandbyPoolBackendConfig
|
|
613
|
-
const layout = (config as Extract<SandboxProviderConfig, { layout: ContainerSandboxLayout }>)
|
|
614
|
-
.layout
|
|
615
|
-
const resolved = resolveLayout(layout)
|
|
616
|
-
return buildAciStandbyPoolBackend({
|
|
617
|
-
subscriptionId: aciBackend.subscriptionId,
|
|
618
|
-
resourceGroup: aciBackend.resourceGroup,
|
|
619
|
-
location: aciBackend.location,
|
|
620
|
-
standbyPoolResourceId: aciBackend.standbyPoolResourceId,
|
|
621
|
-
containerGroupProfileResourceId: aciBackend.containerGroupProfileResourceId,
|
|
622
|
-
...(aciBackend.containerGroupProfileRevision !== undefined
|
|
623
|
-
? { containerGroupProfileRevision: aciBackend.containerGroupProfileRevision }
|
|
624
|
-
: {}),
|
|
625
|
-
layout: resolved,
|
|
626
|
-
getArmToken: aciBackend.getArmToken,
|
|
627
|
-
...(aciBackend.subnetId !== undefined ? { subnetId: aciBackend.subnetId } : {}),
|
|
628
|
-
...(aciBackend.readyPollIntervalMs !== undefined
|
|
629
|
-
? { readyPollIntervalMs: aciBackend.readyPollIntervalMs }
|
|
630
|
-
: {}),
|
|
631
|
-
...(aciBackend.readyTimeoutMs !== undefined
|
|
632
|
-
? { readyTimeoutMs: aciBackend.readyTimeoutMs }
|
|
633
|
-
: {}),
|
|
634
|
-
...(aciBackend.workerPort !== undefined ? { workerPort: aciBackend.workerPort } : {}),
|
|
635
|
-
...(aciBackend.armApiVersion !== undefined
|
|
636
|
-
? { armApiVersion: aciBackend.armApiVersion }
|
|
637
|
-
: {}),
|
|
638
|
-
...(aciBackend.containerNamePrefix !== undefined
|
|
639
|
-
? { containerNamePrefix: aciBackend.containerNamePrefix }
|
|
640
|
-
: {}),
|
|
641
|
-
})
|
|
642
|
-
}
|
|
643
1265
|
if (backend.tier === 'container' && backend.runtime === 'runsc') {
|
|
644
1266
|
const layout = (config as Extract<SandboxProviderConfig, { layout: ContainerSandboxLayout }>)
|
|
645
1267
|
.layout
|
|
@@ -688,9 +1310,190 @@ function pickBackend(config: SandboxProviderConfig): SandboxBackend {
|
|
|
688
1310
|
: {}),
|
|
689
1311
|
})
|
|
690
1312
|
}
|
|
1313
|
+
// `microvm:kubernetes` — agent-sandbox on any cluster. Reached through a
|
|
1314
|
+
// real arm of `SandboxProviderConfig`, so `backend` narrows here and every
|
|
1315
|
+
// field below is read off the narrowed type, same as the ACI and docker
|
|
1316
|
+
// branches above.
|
|
1317
|
+
if (backend.tier === 'microvm' && backend.service === 'kubernetes') {
|
|
1318
|
+
return buildKubernetesBackend(kubernetesInternalConfig(backend))
|
|
1319
|
+
}
|
|
691
1320
|
throw new SandboxBackendNotImplementedError(describeBackend(backend))
|
|
692
1321
|
}
|
|
693
1322
|
|
|
1323
|
+
/**
|
|
1324
|
+
* Public config → the kubernetes backend's own. One function so the two
|
|
1325
|
+
* entry points that build against a cluster — {@link createSandboxProvider}
|
|
1326
|
+
* for task sandboxes and {@link createKubernetesWorkspace} for persistent
|
|
1327
|
+
* ones — cannot drift apart on which fields they forward.
|
|
1328
|
+
*/
|
|
1329
|
+
function kubernetesInternalConfig(
|
|
1330
|
+
backend: KubernetesBackendConfig,
|
|
1331
|
+
): KubernetesBackendInternalConfig {
|
|
1332
|
+
return {
|
|
1333
|
+
access: backend.access,
|
|
1334
|
+
namespace: backend.namespace,
|
|
1335
|
+
sandboxTemplateName: backend.sandboxTemplateName,
|
|
1336
|
+
...(backend.warmPoolName !== undefined ? { warmPoolName: backend.warmPoolName } : {}),
|
|
1337
|
+
...(backend.agentPort !== undefined ? { agentPort: backend.agentPort } : {}),
|
|
1338
|
+
...(backend.agentAddress !== undefined ? { agentAddress: backend.agentAddress } : {}),
|
|
1339
|
+
...(backend.readyPollIntervalMs !== undefined
|
|
1340
|
+
? { readyPollIntervalMs: backend.readyPollIntervalMs }
|
|
1341
|
+
: {}),
|
|
1342
|
+
...(backend.readyTimeoutMs !== undefined ? { readyTimeoutMs: backend.readyTimeoutMs } : {}),
|
|
1343
|
+
...(backend.claimTtlSeconds !== undefined ? { claimTtlSeconds: backend.claimTtlSeconds } : {}),
|
|
1344
|
+
...(backend.onLeaseRenewalError !== undefined
|
|
1345
|
+
? { onLeaseRenewalError: backend.onLeaseRenewalError }
|
|
1346
|
+
: {}),
|
|
1347
|
+
...(backend.runtimeClassName !== undefined
|
|
1348
|
+
? { runtimeClassName: backend.runtimeClassName }
|
|
1349
|
+
: {}),
|
|
1350
|
+
...(backend.egress !== undefined ? { egress: backend.egress } : {}),
|
|
1351
|
+
...(backend.ingress !== undefined ? { ingress: backend.ingress } : {}),
|
|
1352
|
+
...(backend.apiRequestTimeoutMs !== undefined
|
|
1353
|
+
? { apiRequestTimeoutMs: backend.apiRequestTimeoutMs }
|
|
1354
|
+
: {}),
|
|
1355
|
+
...(backend.streamHeartbeatMs !== undefined
|
|
1356
|
+
? { streamHeartbeatMs: backend.streamHeartbeatMs }
|
|
1357
|
+
: {}),
|
|
1358
|
+
...(backend.claimLabels !== undefined ? { claimLabels: backend.claimLabels } : {}),
|
|
1359
|
+
}
|
|
1360
|
+
}
|
|
1361
|
+
|
|
1362
|
+
/**
|
|
1363
|
+
* Create — or reattach to — a persistent workspace on a cluster running the
|
|
1364
|
+
* agent-sandbox controller.
|
|
1365
|
+
*
|
|
1366
|
+
* A workspace is the other half of this backend, and deliberately not
|
|
1367
|
+
* something {@link createSandboxProvider} can hand out: a `SandboxProvider`
|
|
1368
|
+
* promises an EPHEMERAL sandbox per run (`workspaceModes: ['ephemeral']`),
|
|
1369
|
+
* while this returns one object with a name the caller chose, a disk that
|
|
1370
|
+
* survives a suspend, and a lifetime nothing reaps on a timer. It is its own
|
|
1371
|
+
* verb so that the difference is visible at the call site.
|
|
1372
|
+
*
|
|
1373
|
+
* `config.warmPoolName` is ignored here: a workspace is always a `Sandbox`
|
|
1374
|
+
* POSTed directly, because a claim cannot carry the immutable disk spec.
|
|
1375
|
+
* `config.sandboxTemplateName` is the default template, and
|
|
1376
|
+
* `options.sandboxTemplateName` overrides it — a deployment normally has a
|
|
1377
|
+
* task template with no disk and a workspace template with a block one.
|
|
1378
|
+
*
|
|
1379
|
+
* Resolves once the workspace is Ready, addressed and has proved it is
|
|
1380
|
+
* deprivileged, exactly as `provider.create()` does for a task sandbox.
|
|
1381
|
+
*/
|
|
1382
|
+
export async function createKubernetesWorkspace(
|
|
1383
|
+
config: KubernetesBackendConfig,
|
|
1384
|
+
options: KubernetesWorkspaceOptions,
|
|
1385
|
+
): Promise<KubernetesWorkspace> {
|
|
1386
|
+
return await buildKubernetesWorkspace(kubernetesInternalConfig(config), options)
|
|
1387
|
+
}
|
|
1388
|
+
|
|
1389
|
+
/**
|
|
1390
|
+
* Every workspace this backend owns in the namespace, read off the objects
|
|
1391
|
+
* and waking none of them.
|
|
1392
|
+
*
|
|
1393
|
+
* The inventory {@link createKubernetesWorkspace} cannot give you: it adopts
|
|
1394
|
+
* AND resumes, so taking stock through it would start a pod for every
|
|
1395
|
+
* suspended workspace it looked at. This issues one GET of the sandboxes
|
|
1396
|
+
* collection and sends no PATCH and no DELETE — a suspended workspace is
|
|
1397
|
+
* still suspended afterwards.
|
|
1398
|
+
*
|
|
1399
|
+
* Needs `list` on `sandboxes` in the namespace, which is the one RBAC verb
|
|
1400
|
+
* the task path did not already require.
|
|
1401
|
+
*/
|
|
1402
|
+
export async function listKubernetesWorkspaces(
|
|
1403
|
+
config: KubernetesBackendConfig,
|
|
1404
|
+
options?: KubernetesWorkspaceTransitionOptions,
|
|
1405
|
+
): Promise<readonly KubernetesWorkspaceSummary[]> {
|
|
1406
|
+
return await listWorkspacesOnCluster(kubernetesInternalConfig(config), options)
|
|
1407
|
+
}
|
|
1408
|
+
|
|
1409
|
+
/**
|
|
1410
|
+
* Delete a workspace by id — the Sandbox, and with it the Pod, the Service
|
|
1411
|
+
* and the PVC — without adopting or resuming it first.
|
|
1412
|
+
*
|
|
1413
|
+
* Exactly what `destroy({ deleteDisk: true })` does to the cluster, with the
|
|
1414
|
+
* same guarantees: an object already gone counts as deleted, and a DELETE
|
|
1415
|
+
* that fails rejects and stays retryable. The files are gone and nothing
|
|
1416
|
+
* brings them back.
|
|
1417
|
+
*
|
|
1418
|
+
* It is the retention verb. Removing a month-old suspended workspace through
|
|
1419
|
+
* a handle meant starting its pod and probing it purely to tell it to go
|
|
1420
|
+
* away; the name is deterministic, so the object never needed opening.
|
|
1421
|
+
*/
|
|
1422
|
+
export async function deleteKubernetesWorkspace(
|
|
1423
|
+
config: KubernetesBackendConfig,
|
|
1424
|
+
workspaceId: string,
|
|
1425
|
+
options?: KubernetesWorkspaceTransitionOptions,
|
|
1426
|
+
): Promise<void> {
|
|
1427
|
+
await deleteWorkspaceOnCluster(kubernetesInternalConfig(config), workspaceId, options)
|
|
1428
|
+
}
|
|
1429
|
+
|
|
1430
|
+
/**
|
|
1431
|
+
* Suspend a workspace by id: send the `operatingMode: Suspended` patch and
|
|
1432
|
+
* wait for the pod to actually stop, without adopting the workspace.
|
|
1433
|
+
*
|
|
1434
|
+
* Resolves only once the pod is gone or in a terminal phase — a suspend is a
|
|
1435
|
+
* promise that the disk is quiesced, and the patch being accepted says only
|
|
1436
|
+
* that the controller has been asked. A pod that outlives `readyTimeoutMs`
|
|
1437
|
+
* rejects with `KubernetesWorkspaceSuspendTimeoutError`, leaving the object
|
|
1438
|
+
* as the patch left it.
|
|
1439
|
+
*
|
|
1440
|
+
* A handle another process is holding is not told. It finds out on its next
|
|
1441
|
+
* call — which fails at the transport and is re-read into a
|
|
1442
|
+
* `KubernetesWorkspaceSuspendedError` — or when that process calls
|
|
1443
|
+
* `refresh()`.
|
|
1444
|
+
*
|
|
1445
|
+
* It takes the suspend options shape and REFUSES `quiesce` rather than
|
|
1446
|
+
* accepting the flag and dropping it: this verb never dials the agent, so
|
|
1447
|
+
* there is no connection here on which anything could be stopped. Quiescing
|
|
1448
|
+
* needs a handle — `createKubernetesWorkspace()`, then
|
|
1449
|
+
* `suspend({ quiesce: true })`.
|
|
1450
|
+
*/
|
|
1451
|
+
export async function suspendKubernetesWorkspace(
|
|
1452
|
+
config: KubernetesBackendConfig,
|
|
1453
|
+
workspaceId: string,
|
|
1454
|
+
options?: KubernetesWorkspaceSuspendOptions,
|
|
1455
|
+
): Promise<void> {
|
|
1456
|
+
await suspendWorkspaceOnCluster(kubernetesInternalConfig(config), workspaceId, options)
|
|
1457
|
+
}
|
|
1458
|
+
|
|
1459
|
+
/**
|
|
1460
|
+
* Recover a crashed host's task-path claims: LIST every `SandboxClaim`
|
|
1461
|
+
* carrying `options.labelSelector`, `DELETE` each, and report what was
|
|
1462
|
+
* removed.
|
|
1463
|
+
*
|
|
1464
|
+
* Deletes claims only — the controller's own ownerReferences take the bound
|
|
1465
|
+
* Sandbox, its Pod and its Service down behind each one; nothing here reads
|
|
1466
|
+
* or touches those objects directly. `labelSelector` is REQUIRED and refused
|
|
1467
|
+
* before any request goes out if it is empty: falling back to matching every
|
|
1468
|
+
* claim would delete a live fleet's work.
|
|
1469
|
+
*
|
|
1470
|
+
* Pairs with `config.claimLabels`: a host stamps its own identity onto every
|
|
1471
|
+
* claim it creates, and a restarted instance passes that same selector here
|
|
1472
|
+
* to reclaim its predecessor's warm-pool capacity well before
|
|
1473
|
+
* `claimTtlSeconds` would reap it on its own.
|
|
1474
|
+
*/
|
|
1475
|
+
export async function releaseKubernetesTaskSandboxes(
|
|
1476
|
+
config: KubernetesBackendConfig,
|
|
1477
|
+
options: KubernetesReleaseTaskSandboxesOptions,
|
|
1478
|
+
): Promise<{ readonly deleted: number; readonly names: readonly string[] }> {
|
|
1479
|
+
return await releaseTaskSandboxesOnCluster(kubernetesInternalConfig(config), options)
|
|
1480
|
+
}
|
|
1481
|
+
|
|
1482
|
+
/**
|
|
1483
|
+
* Read task-pool headroom before admitting more work: three GETs
|
|
1484
|
+
* (`SandboxWarmPool`, the claims collection, the pods collection), no
|
|
1485
|
+
* writes.
|
|
1486
|
+
*
|
|
1487
|
+
* Requires `config.warmPoolName` — a pool-less backend (every create is a
|
|
1488
|
+
* direct Sandbox) has no `SandboxWarmPool` to report on.
|
|
1489
|
+
*/
|
|
1490
|
+
export async function readKubernetesTaskCapacity(
|
|
1491
|
+
config: KubernetesBackendConfig,
|
|
1492
|
+
options?: KubernetesReadTaskCapacityOptions,
|
|
1493
|
+
): Promise<KubernetesTaskCapacity> {
|
|
1494
|
+
return await readTaskCapacityOnCluster(kubernetesInternalConfig(config), options)
|
|
1495
|
+
}
|
|
1496
|
+
|
|
694
1497
|
/**
|
|
695
1498
|
* Human-readable backend label for error messages. Returns the
|
|
696
1499
|
* tier plus the concrete service / runtime when present, e.g.
|