@namzu/sandbox 13.0.0 → 14.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. package/CHANGELOG.md +309 -0
  2. package/README.md +151 -0
  3. package/dist/backends/firecracker/protocol.d.ts +22 -0
  4. package/dist/backends/firecracker/protocol.d.ts.map +1 -1
  5. package/dist/backends/firecracker/protocol.js.map +1 -1
  6. package/dist/backends/firecracker/transport.d.ts +104 -9
  7. package/dist/backends/firecracker/transport.d.ts.map +1 -1
  8. package/dist/backends/firecracker/transport.js +139 -13
  9. package/dist/backends/firecracker/transport.js.map +1 -1
  10. package/dist/backends/kubernetes/egress-policy.d.ts +219 -0
  11. package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -0
  12. package/dist/backends/kubernetes/egress-policy.js +314 -0
  13. package/dist/backends/kubernetes/egress-policy.js.map +1 -0
  14. package/dist/backends/kubernetes/index.d.ts +374 -0
  15. package/dist/backends/kubernetes/index.d.ts.map +1 -0
  16. package/dist/backends/kubernetes/index.js +671 -0
  17. package/dist/backends/kubernetes/index.js.map +1 -0
  18. package/dist/backends/kubernetes/k8s-client.d.ts +125 -0
  19. package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -0
  20. package/dist/backends/kubernetes/k8s-client.js +246 -0
  21. package/dist/backends/kubernetes/k8s-client.js.map +1 -0
  22. package/dist/backends/kubernetes/lease.d.ts +119 -0
  23. package/dist/backends/kubernetes/lease.d.ts.map +1 -0
  24. package/dist/backends/kubernetes/lease.js +151 -0
  25. package/dist/backends/kubernetes/lease.js.map +1 -0
  26. package/dist/backends/kubernetes/objects.d.ts +282 -0
  27. package/dist/backends/kubernetes/objects.d.ts.map +1 -0
  28. package/dist/backends/kubernetes/objects.js +156 -0
  29. package/dist/backends/kubernetes/objects.js.map +1 -0
  30. package/dist/backends/kubernetes/privilege-probe.d.ts +136 -0
  31. package/dist/backends/kubernetes/privilege-probe.d.ts.map +1 -0
  32. package/dist/backends/kubernetes/privilege-probe.js +185 -0
  33. package/dist/backends/kubernetes/privilege-probe.js.map +1 -0
  34. package/dist/backends/kubernetes/sandbox.d.ts +123 -0
  35. package/dist/backends/kubernetes/sandbox.d.ts.map +1 -0
  36. package/dist/backends/kubernetes/sandbox.js +299 -0
  37. package/dist/backends/kubernetes/sandbox.js.map +1 -0
  38. package/dist/backends/kubernetes/transport.d.ts +122 -0
  39. package/dist/backends/kubernetes/transport.d.ts.map +1 -0
  40. package/dist/backends/kubernetes/transport.js +197 -0
  41. package/dist/backends/kubernetes/transport.js.map +1 -0
  42. package/dist/backends/kubernetes/workspace.d.ts +381 -0
  43. package/dist/backends/kubernetes/workspace.d.ts.map +1 -0
  44. package/dist/backends/kubernetes/workspace.js +1064 -0
  45. package/dist/backends/kubernetes/workspace.js.map +1 -0
  46. package/dist/index.d.ts +132 -2
  47. package/dist/index.d.ts.map +1 -1
  48. package/dist/index.js +102 -34
  49. package/dist/index.js.map +1 -1
  50. package/dist/testing/sandbox-conformance.d.ts +193 -0
  51. package/dist/testing/sandbox-conformance.d.ts.map +1 -0
  52. package/dist/testing/sandbox-conformance.js +465 -0
  53. package/dist/testing/sandbox-conformance.js.map +1 -0
  54. package/package.json +5 -4
  55. package/src/backends/firecracker/protocol.ts +27 -0
  56. package/src/backends/firecracker/transport.ts +199 -28
  57. package/src/backends/kubernetes/egress-policy.ts +437 -0
  58. package/src/backends/kubernetes/index.ts +1012 -0
  59. package/src/backends/kubernetes/k8s-client.ts +352 -0
  60. package/src/backends/kubernetes/lease.ts +198 -0
  61. package/src/backends/kubernetes/objects.ts +363 -0
  62. package/src/backends/kubernetes/privilege-probe.ts +261 -0
  63. package/src/backends/kubernetes/sandbox.ts +395 -0
  64. package/src/backends/kubernetes/transport.ts +286 -0
  65. package/src/backends/kubernetes/workspace.ts +1386 -0
  66. package/src/index.ts +257 -35
  67. package/src/testing/sandbox-conformance.ts +667 -0
@@ -0,0 +1,363 @@
1
+ /**
2
+ * Wire shapes and paths for the agent-sandbox CRDs this backend touches.
3
+ *
4
+ * Every field name here was read off the CRDs a real cluster serves —
5
+ * `kubectl get crd -o json` against agent-sandbox v1.0.2 on a kind cluster —
6
+ * and cross-checked against the upstream `sandbox_types.go` /
7
+ * `sandboxclaim_types.go` doc comments. Where the two disagree the served CRD
8
+ * wins, because it is what the API server validates against: the Go source on
9
+ * upstream main has already moved `Sandbox`'s `shutdownTime` /
10
+ * `shutdownPolicy` under a `lifecycle` block, and v1beta1 as served still
11
+ * carries them at the top of `spec`.
12
+ *
13
+ * The shapes are deliberately partial. This backend reads four fields out of a
14
+ * Sandbox status and writes three into a claim spec; typing the rest of a
15
+ * PodSpec would be a vendored copy of the core API that goes stale on its own
16
+ * schedule. A pod template read from a SandboxTemplate is carried through as
17
+ * an opaque record for exactly that reason — it is copied, never interpreted.
18
+ *
19
+ * Two groups, both `v1beta1` and both singular-versioned today:
20
+ * - `agents.x-k8s.io` → sandboxes
21
+ * - `extensions.agents.x-k8s.io` → sandboxtemplates, sandboxwarmpools,
22
+ * sandboxclaims
23
+ */
24
+
25
+ /** Group serving the `Sandbox` kind. */
26
+ export const SANDBOX_API_GROUP = 'agents.x-k8s.io'
27
+ /** Group serving `SandboxTemplate`, `SandboxWarmPool` and `SandboxClaim`. */
28
+ export const SANDBOX_EXTENSIONS_API_GROUP = 'extensions.agents.x-k8s.io'
29
+ /** The only version either group serves in agent-sandbox v1.0.2. */
30
+ export const SANDBOX_API_VERSION = 'v1beta1'
31
+
32
+ /** `status.conditions[].type` both kinds report readiness under. */
33
+ export const READY_CONDITION = 'Ready'
34
+
35
+ // The controller also reports a `Suspended` condition, and this backend
36
+ // deliberately does NOT model or read it. Upstream's own `sandbox_types.go`
37
+ // says why: "the controller does not currently remove this condition when the
38
+ // Sandbox is resumed, so a stale Suspended condition may linger after
39
+ // operatingMode returns to Running. Consumers should treat Ready as the
40
+ // authoritative signal and not infer the live operating state from the mere
41
+ // presence of this condition." A suspend that waited on it would return on the
42
+ // True left behind by the previous suspend, while the guest was still running
43
+ // and still writing. `workspace.ts` waits on the pod instead — see
44
+ // `isPodStopped` below.
45
+
46
+ export interface KubernetesObjectMeta {
47
+ readonly name?: string
48
+ readonly namespace?: string
49
+ readonly uid?: string
50
+ /**
51
+ * Set the moment a DELETE is accepted, long before the object goes away.
52
+ * A pod that carries one is on its way out and must never be bound to —
53
+ * see {@link isPodLive}.
54
+ */
55
+ readonly deletionTimestamp?: string
56
+ readonly labels?: Readonly<Record<string, string>>
57
+ readonly annotations?: Readonly<Record<string, string>>
58
+ }
59
+
60
+ /** `metav1.Condition`, as both CRDs embed it. */
61
+ export interface KubernetesCondition {
62
+ readonly type: string
63
+ readonly status: 'True' | 'False' | 'Unknown'
64
+ readonly reason?: string
65
+ readonly message?: string
66
+ readonly lastTransitionTime?: string
67
+ }
68
+
69
+ /**
70
+ * True only for an explicit `status: 'True'`. An absent condition, an
71
+ * `Unknown` and a `False` are all "not yet", never "assume so" — the
72
+ * controller writes `Unknown` while it is still deciding.
73
+ */
74
+ export function isConditionTrue(
75
+ conditions: readonly KubernetesCondition[] | undefined,
76
+ type: string,
77
+ ): boolean {
78
+ return conditions?.some((c) => c.type === type && c.status === 'True') === true
79
+ }
80
+
81
+ /**
82
+ * `SandboxClaim.spec.lifecycle`.
83
+ *
84
+ * `shutdownTime` is the only one of the three that bounds a claim whose owner
85
+ * disappeared: the controller deletes the claim's resources once the wall
86
+ * clock reaches it, whatever the claim is doing. `ttlSecondsAfterFinished`
87
+ * reads like the leak guard and is not one — upstream's own comment says "the
88
+ * timer starts from the mirrored Finished condition's LastTransitionTime", so
89
+ * a claim whose host crashed before finishing never starts that clock.
90
+ */
91
+ export interface SandboxClaimLifecycle {
92
+ /** RFC 3339. Absolute expiry; the claim never expires without it. */
93
+ readonly shutdownTime?: string
94
+ /**
95
+ * What happens to the claim OBJECT at expiry. `Retain` (the CRD default)
96
+ * deletes the Sandbox, Pod and Service but leaves the claim behind, so a
97
+ * host that crashes daily accumulates claims forever.
98
+ */
99
+ readonly shutdownPolicy?: 'Delete' | 'DeleteForeground' | 'Retain'
100
+ readonly ttlSecondsAfterFinished?: number
101
+ }
102
+
103
+ /**
104
+ * `SandboxClaim.spec`. `warmPoolRef` is REQUIRED by the CRD, which is why
105
+ * there is no such thing as a pool-less claim and the no-pool path has to
106
+ * create a Sandbox directly.
107
+ *
108
+ * `env` and `volumeClaimTemplates` exist on this spec and are deliberately
109
+ * absent from this type: setting either forces the claim to cold-start rather
110
+ * than adopt a warm pool sandbox, which is the one thing the warm path exists
111
+ * to avoid. A field that cannot be named cannot be set by accident.
112
+ */
113
+ export interface SandboxClaimResourceSpec {
114
+ readonly warmPoolRef: { readonly name: string }
115
+ readonly lifecycle?: SandboxClaimLifecycle
116
+ }
117
+
118
+ /**
119
+ * `SandboxClaim.status`. `sandbox` is the whole reason the claim path reads
120
+ * status back: an adopted pool sandbox keeps the generated name the pool gave
121
+ * it, so the bound object is routinely NOT named after the claim.
122
+ */
123
+ export interface SandboxClaimResourceStatus {
124
+ readonly conditions?: readonly KubernetesCondition[]
125
+ readonly sandbox?: {
126
+ readonly name?: string
127
+ readonly podIPs?: readonly string[]
128
+ readonly serviceFQDN?: string
129
+ }
130
+ }
131
+
132
+ export interface SandboxClaimResource {
133
+ readonly apiVersion?: string
134
+ readonly kind?: string
135
+ readonly metadata?: KubernetesObjectMeta
136
+ readonly spec?: SandboxClaimResourceSpec
137
+ readonly status?: SandboxClaimResourceStatus
138
+ }
139
+
140
+ /**
141
+ * `podTemplate` on a Sandbox or a SandboxTemplate. `spec` is a core `PodSpec`,
142
+ * carried opaquely: this backend copies one from a template into a Sandbox and
143
+ * overlays at most `runtimeClassName`.
144
+ */
145
+ export interface SandboxPodTemplate {
146
+ readonly metadata?: KubernetesObjectMeta
147
+ readonly spec: Readonly<Record<string, unknown>>
148
+ }
149
+
150
+ /**
151
+ * One `spec.volumeClaimTemplates` entry.
152
+ *
153
+ * Partial in the same way every other shape here is: a workspace's disk is
154
+ * COPIED verbatim from the `SandboxTemplate` that declares it, and only the
155
+ * two fields this backend has to reason about are named — the entry's own
156
+ * `metadata.name`, which is how the controller wires the mount (StatefulSet
157
+ * style: the PVC is created as `<entry name>-<sandbox name>` and no explicit
158
+ * `volumes:` entry is needed in the podTemplate), and `spec.volumeMode`,
159
+ * which decides whether the guest gets a raw block device or a filesystem
160
+ * passthrough. The index signatures carry everything else across untouched.
161
+ */
162
+ export interface SandboxVolumeClaimTemplate {
163
+ readonly metadata?: KubernetesObjectMeta
164
+ readonly spec?: {
165
+ readonly volumeMode?: string
166
+ readonly [field: string]: unknown
167
+ }
168
+ readonly [field: string]: unknown
169
+ }
170
+
171
+ export interface SandboxResourceSpec {
172
+ readonly operatingMode?: 'Running' | 'Suspended'
173
+ readonly podTemplate: SandboxPodTemplate
174
+ /** Create a headless Service, and with it a `status.serviceFQDN`. */
175
+ readonly service?: boolean
176
+ readonly shutdownPolicy?: 'Delete' | 'Retain'
177
+ /** RFC 3339, top-level on v1beta1 Sandbox (NOT under `lifecycle`). */
178
+ readonly shutdownTime?: string
179
+ /**
180
+ * CEL-immutable on the served CRD ("volumeClaimTemplates is immutable"),
181
+ * which is why a workspace's disk has to be in the spec from creation and
182
+ * cannot be attached to a sandbox that is already running.
183
+ */
184
+ readonly volumeClaimTemplates?: readonly SandboxVolumeClaimTemplate[]
185
+ }
186
+
187
+ /**
188
+ * `Sandbox.status`. Note what is NOT here: a pod name. The backing pod is
189
+ * named after the Sandbox itself in v1.0.2, and `selector` — a serialised
190
+ * label selector, e.g. `agents.x-k8s.io/sandbox-name-hash=<hash>` — is the
191
+ * only thing in the API that finds the pod without relying on that.
192
+ */
193
+ export interface SandboxResourceStatus {
194
+ readonly conditions?: readonly KubernetesCondition[]
195
+ readonly nodeName?: string
196
+ readonly podIPs?: readonly string[]
197
+ readonly selector?: string
198
+ readonly service?: string
199
+ readonly serviceFQDN?: string
200
+ }
201
+
202
+ export interface SandboxResource {
203
+ readonly apiVersion?: string
204
+ readonly kind?: string
205
+ readonly metadata?: KubernetesObjectMeta
206
+ readonly spec?: SandboxResourceSpec
207
+ readonly status?: SandboxResourceStatus
208
+ }
209
+
210
+ export interface SandboxTemplateResource {
211
+ readonly metadata?: KubernetesObjectMeta
212
+ readonly spec?: {
213
+ readonly podTemplate?: SandboxPodTemplate
214
+ readonly service?: boolean
215
+ readonly volumeClaimTemplates?: readonly SandboxVolumeClaimTemplate[]
216
+ }
217
+ }
218
+
219
+ /**
220
+ * `metadata.uid` is the per-instance agent bind token; the other two fields
221
+ * exist only to answer "is this the pod that uid belongs to, or the one being
222
+ * deleted?" — see {@link isPodLive}.
223
+ */
224
+ export interface PodResource {
225
+ readonly metadata?: KubernetesObjectMeta
226
+ readonly status?: { readonly phase?: string }
227
+ }
228
+
229
+ export interface PodListResource {
230
+ readonly items?: readonly PodResource[]
231
+ }
232
+
233
+ /**
234
+ * A pod whose uid is still worth binding to: not being deleted, and not in a
235
+ * phase it cannot leave.
236
+ *
237
+ * The case this exists for is resume. A resumed sandbox's pod keeps the
238
+ * SAME NAME and gets a new uid and a new IP, so while the outgoing pod is
239
+ * terminating a `GET` by that name answers with the pod on its way out, and a
240
+ * list by the sandbox's selector returns every pod still carrying its labels,
241
+ * that one included. Binding to the terminating pod's uid produces a token the
242
+ * new agent refuses, and the failure arrives as a flat `unauthorized` with
243
+ * nothing pointing at the race that caused it.
244
+ */
245
+ export function isPodLive(pod: PodResource | undefined): boolean {
246
+ if (!pod?.metadata) return false
247
+ // `!= null`, not `!== undefined`: an explicit JSON null would otherwise
248
+ // read as "terminating" and make a perfectly healthy pod unbindable. The
249
+ // API server omits the field rather than nulling it, so this never fires
250
+ // against a real cluster — it costs nothing not to depend on that.
251
+ if (pod.metadata.deletionTimestamp != null) return false
252
+ const phase = pod.status?.phase
253
+ return phase !== 'Succeeded' && phase !== 'Failed'
254
+ }
255
+
256
+ /**
257
+ * A pod whose containers have stopped: it is still an object, and nothing in
258
+ * it is executing any more.
259
+ *
260
+ * NOT the negation of {@link isPodLive}, and the gap between the two is the
261
+ * whole point. A terminating pod — `deletionTimestamp` set, phase still
262
+ * `Running` — is not live (never bind to it: its uid is about to stop being
263
+ * a valid token) and not stopped either (its process is still running, and on
264
+ * a workspace it is still writing to the caller's block device until it exits
265
+ * or `terminationGracePeriodSeconds` runs out). A suspend that treated the
266
+ * timestamp as "gone" would resolve mid-drain and promise a quiesced disk it
267
+ * had not waited for.
268
+ */
269
+ export function isPodStopped(pod: PodResource | undefined): boolean {
270
+ const phase = pod?.status?.phase
271
+ return phase === 'Succeeded' || phase === 'Failed'
272
+ }
273
+
274
+ /**
275
+ * Backend-owned pod label naming which `SandboxTemplate` a Sandbox's pod was
276
+ * built from.
277
+ *
278
+ * agent-sandbox's OWN template-adoption controller selects pods by a
279
+ * controller-owned label, `agents.x-k8s.io/sandbox-template-ref-hash` — but
280
+ * that label is written only onto a Sandbox ADOPTED out of a
281
+ * `SandboxWarmPool` (the controller re-parents ownership and re-labels on
282
+ * bind). A Sandbox this backend POSTs directly (the pool-less path in
283
+ * `index.ts`'s `buildSandboxBody`) is never adopted, so it never gets that
284
+ * label — a direct Sandbox's pod would carry nothing a `NetworkPolicy`
285
+ * could reliably select it by. This backend writes its own label instead, on
286
+ * every Sandbox it creates, pooled or direct, so `egress-policy.ts`'s
287
+ * translated `NetworkPolicy` has one selector that always matches.
288
+ *
289
+ * `namzu.ai` matches the published domain (`packages/cli/package.json`'s
290
+ * `homepage`); there is no pre-existing Kubernetes label or annotation
291
+ * prefix anywhere in this repo to follow instead — the closest existing
292
+ * convention, `NAMZU_AGENT_*` / `NAMZU_SANDBOX_*` env vars, is not a
293
+ * label-safe shape.
294
+ *
295
+ * W8's `SandboxTemplate` manifests MUST set this same label (value = the
296
+ * template's own name) on their `podTemplate.metadata.labels`, so a POOLED
297
+ * sandbox's pod carries it too — the pool's pods are built from that
298
+ * template's `podTemplate` directly, not through `buildSandboxBody`, so
299
+ * nothing here can put it there for them. Skipping that step means a
300
+ * translated `NetworkPolicy`'s `podSelector` matches only sandboxes this
301
+ * backend created directly and none of the pooled ones — see
302
+ * `docs/sdk/kubernetes-sandbox.md`'s egress section.
303
+ */
304
+ export const SANDBOX_TEMPLATE_LABEL_KEY = 'sandbox.namzu.ai/template'
305
+
306
+ /** `{ [SANDBOX_TEMPLATE_LABEL_KEY]: sandboxTemplateName }`, as a matchLabels-ready object. */
307
+ export function sandboxTemplateLabel(
308
+ sandboxTemplateName: string,
309
+ ): Readonly<Record<string, string>> {
310
+ return { [SANDBOX_TEMPLATE_LABEL_KEY]: sandboxTemplateName }
311
+ }
312
+
313
+ /** Core `NetworkPolicy` — a stock resource, no CRD. */
314
+ export const CORE_NETWORK_POLICY_API_GROUP = 'networking.k8s.io'
315
+ export const CORE_NETWORK_POLICY_API_VERSION = 'v1'
316
+
317
+ /**
318
+ * Cilium's FQDN-capable policy CRD. Only reached when
319
+ * `KubernetesEgressConfig.engine` is explicitly `'cilium'` — see
320
+ * `egress-policy.ts`.
321
+ */
322
+ export const CILIUM_NETWORK_POLICY_API_GROUP = 'cilium.io'
323
+ export const CILIUM_NETWORK_POLICY_API_VERSION = 'v2'
324
+
325
+ function segment(value: string): string {
326
+ return encodeURIComponent(value)
327
+ }
328
+
329
+ export function claimCollectionPath(namespace: string): string {
330
+ return `/apis/${SANDBOX_EXTENSIONS_API_GROUP}/${SANDBOX_API_VERSION}/namespaces/${segment(namespace)}/sandboxclaims`
331
+ }
332
+
333
+ export function claimPath(namespace: string, name: string): string {
334
+ return `${claimCollectionPath(namespace)}/${segment(name)}`
335
+ }
336
+
337
+ export function sandboxCollectionPath(namespace: string): string {
338
+ return `/apis/${SANDBOX_API_GROUP}/${SANDBOX_API_VERSION}/namespaces/${segment(namespace)}/sandboxes`
339
+ }
340
+
341
+ export function sandboxPath(namespace: string, name: string): string {
342
+ return `${sandboxCollectionPath(namespace)}/${segment(name)}`
343
+ }
344
+
345
+ export function sandboxTemplatePath(namespace: string, name: string): string {
346
+ return `/apis/${SANDBOX_EXTENSIONS_API_GROUP}/${SANDBOX_API_VERSION}/namespaces/${segment(namespace)}/sandboxtemplates/${segment(name)}`
347
+ }
348
+
349
+ export function podPath(namespace: string, name: string): string {
350
+ return `/api/v1/namespaces/${segment(namespace)}/pods/${segment(name)}`
351
+ }
352
+
353
+ export function podListPath(namespace: string, labelSelector: string): string {
354
+ return `/api/v1/namespaces/${segment(namespace)}/pods?labelSelector=${encodeURIComponent(labelSelector)}`
355
+ }
356
+
357
+ export function networkPolicyPath(namespace: string, name: string): string {
358
+ return `/apis/${CORE_NETWORK_POLICY_API_GROUP}/${CORE_NETWORK_POLICY_API_VERSION}/namespaces/${segment(namespace)}/networkpolicies/${segment(name)}`
359
+ }
360
+
361
+ export function ciliumNetworkPolicyPath(namespace: string, name: string): string {
362
+ return `/apis/${CILIUM_NETWORK_POLICY_API_GROUP}/${CILIUM_NETWORK_POLICY_API_VERSION}/namespaces/${segment(namespace)}/ciliumnetworkpolicies/${segment(name)}`
363
+ }
@@ -0,0 +1,261 @@
1
+ /**
2
+ * Acquire-time privilege probe: prove the guest process really was
3
+ * deprivileged, rather than assume the entrypoint did its job.
4
+ *
5
+ * The image's entrypoint mounts the workspace as root and then
6
+ * `exec setpriv --reuid --regid --clear-groups --inh-caps=-all
7
+ * --bounding-set=-all --no-new-privs -- tini -- node agent.cjs` (`tini` is
8
+ * the container's pid 1 so it can reap an orphan `agent.cjs` itself never
9
+ * spawned — see `k8s/entrypoint.sh` and the Dockerfile). Nothing in
10
+ * `agent.cjs` knows about any of that, and nothing on the host can see it
11
+ * either — an image built from an older entrypoint, a `RuntimeClass` change,
12
+ * a hand-edited `SandboxTemplate` all produce a sandbox that works perfectly
13
+ * and is not deprivileged. So the backend asks the guest, once, before it
14
+ * hands a caller a handle.
15
+ *
16
+ * ## Why `execute`, and why not `read-file`
17
+ *
18
+ * The guest's `read-file` resolves every path against `READ_ROOTS`
19
+ * (`WORKSPACE_ROOT` only), so it cannot reach `/proc` at all, and widening
20
+ * `READ_ROOTS` to make the probe work would hand every caller of `readFile`
21
+ * a window into the guest's process tree for the sake of one diagnostic.
22
+ * `handleExecute` jails only `cwd`, so a command whose ARGUMENT is an
23
+ * absolute path outside the workspace runs fine. The probe therefore spends
24
+ * one `exec` and touches no jail.
25
+ *
26
+ * ## Why all four masks
27
+ *
28
+ * Checking `CapEff` alone is a true-looking answer: an ordinary unprivileged
29
+ * process shows `CapEff: 0000000000000000` whether or not its bounding set
30
+ * was ever dropped, so a container running as uid 0 with the full bounding
31
+ * set still passes. `CapBnd` is the one that says a capability can never be
32
+ * regained; `CapInh` and `CapPrm` close the two ways one could be carried
33
+ * across an exec. `NoNewPrivs: 1` is what makes a setuid binary inside the
34
+ * guest unable to raise any of it back.
35
+ *
36
+ * ## Why every failure is a refusal
37
+ *
38
+ * A probe that could not run, one that never answered at all, output that
39
+ * could not be parsed and a process that is genuinely privileged are all
40
+ * reasons NOT to hand back a handle, and they are separated only in the error
41
+ * TEXT — a distroless image with no `cat` on `PATH` must be diagnosable as
42
+ * exactly that rather than read as a hardening failure. There is no
43
+ * configuration that turns this off.
44
+ *
45
+ * ## The clock is the caller's, and it lives in `index.ts`
46
+ *
47
+ * Nothing here has a timeout: the probe is one `exec`, and the budget it may
48
+ * spend belongs to the `create()` that ordered it. `admitProbedSandbox` runs
49
+ * it under an `OperationDeadline` and turns an expiry into
50
+ * {@link privilegeProbeTimedOut}, so a guest that accepts the connection and
51
+ * then goes quiet is refused on the caller's clock rather than on the
52
+ * execution controller's five-minute generic default.
53
+ */
54
+
55
+ import type { SandboxExecResult } from '@namzu/sdk'
56
+
57
+ /**
58
+ * The probe command. `cat` rather than an absolute `/bin/cat` so a guest
59
+ * that keeps its coreutils somewhere else still answers, and rather than a
60
+ * shell so there is no quoting to get wrong. A guest without it fails with
61
+ * a spawn error the refusal repeats verbatim.
62
+ */
63
+ export const PRIVILEGE_PROBE_COMMAND = 'cat'
64
+
65
+ /** `/proc/self/status` — of the process the guest agent spawns, which
66
+ * inherits exactly the agent's own credentials and capability masks. */
67
+ export const PRIVILEGE_PROBE_ARGS: readonly string[] = ['/proc/self/status']
68
+
69
+ /** Why a probe refused. The text says the same thing in words. */
70
+ export type PrivilegeProbeFailure =
71
+ /** The `exec` failed, exited non-zero, or never answered at all. */
72
+ | 'probe-failed'
73
+ /** It ran, but its output is not a readable `/proc/<pid>/status`. */
74
+ | 'unreadable-output'
75
+ /** It ran, it parsed, and the process has capabilities it should not. */
76
+ | 'privileged'
77
+
78
+ /**
79
+ * Raised by {@link runPrivilegeProbe} and {@link parseProcStatus}. The
80
+ * `reason` is the machine-readable form of the distinction the message
81
+ * draws in prose: `'privileged'` means the guest is under-hardened, and the
82
+ * other two mean the backend could not tell.
83
+ */
84
+ export class KubernetesPrivilegeProbeError extends Error {
85
+ override readonly name = 'KubernetesPrivilegeProbeError'
86
+
87
+ constructor(
88
+ readonly reason: PrivilegeProbeFailure,
89
+ message: string,
90
+ options?: ErrorOptions,
91
+ ) {
92
+ super(message, options)
93
+ }
94
+ }
95
+
96
+ /**
97
+ * The five fields the probe reads, parsed. The four capability masks are
98
+ * `bigint` because a capability mask is 64 bits wide and `Number` stops
99
+ * being exact at 53 — `000001ffffffffff` is only 41 bits today, but a mask
100
+ * that silently rounds is precisely the bug this whole module exists to
101
+ * catch.
102
+ */
103
+ export interface ProcStatusPrivileges {
104
+ readonly capInh: bigint
105
+ readonly capPrm: bigint
106
+ readonly capEff: bigint
107
+ readonly capBnd: bigint
108
+ /** `prctl(PR_GET_NO_NEW_PRIVS)`, 0 or 1 as the kernel prints it. */
109
+ readonly noNewPrivs: number
110
+ }
111
+
112
+ const CAPABILITY_FIELDS = ['CapInh', 'CapPrm', 'CapEff', 'CapBnd'] as const
113
+ const NO_NEW_PRIVS_FIELD = 'NoNewPrivs'
114
+
115
+ /** Clip a value before it goes into an error message. */
116
+ function clip(value: string): string {
117
+ return value.length > 40 ? `${value.slice(0, 40)}…` : value
118
+ }
119
+
120
+ function readField(text: string, field: string): string {
121
+ for (const rawLine of text.split(/\r?\n/)) {
122
+ const colon = rawLine.indexOf(':')
123
+ if (colon < 0) continue
124
+ if (rawLine.slice(0, colon).trim() !== field) continue
125
+ return rawLine.slice(colon + 1).trim()
126
+ }
127
+ throw new KubernetesPrivilegeProbeError(
128
+ 'unreadable-output',
129
+ `the privilege probe ran but its output carries no ${field} line, so this sandbox's privileges could not be read. The probe is \`${PRIVILEGE_PROBE_COMMAND} ${PRIVILEGE_PROBE_ARGS.join(' ')}\` — a guest whose /proc is not mounted, or whose kernel does not publish ${field}, cannot be admitted, because an unreadable answer is not a safe one.`,
130
+ )
131
+ }
132
+
133
+ /**
134
+ * Parse `/proc/<pid>/status` into the five fields that decide admission.
135
+ *
136
+ * Pure: no transport, no clock, no I/O. Everything it cannot read is a
137
+ * throw, never a default — a zero substituted for a missing mask is the one
138
+ * mistake that would make this function report hardening that is not there.
139
+ */
140
+ export function parseProcStatus(text: string): ProcStatusPrivileges {
141
+ const masks = CAPABILITY_FIELDS.map((field) => {
142
+ const raw = readField(text, field)
143
+ // The kernel prints a bare, fixed-width hex mask with no `0x`. A
144
+ // `0x` prefix, a sign, whitespace inside, or anything non-hex means
145
+ // this is not the file this parser thinks it is.
146
+ if (!/^[0-9a-fA-F]+$/.test(raw)) {
147
+ throw new KubernetesPrivilegeProbeError(
148
+ 'unreadable-output',
149
+ `the privilege probe ran but ${field} is ${JSON.stringify(clip(raw))}, which is not the bare hexadecimal capability mask /proc/<pid>/status publishes, so this sandbox's privileges could not be read.`,
150
+ )
151
+ }
152
+ return BigInt(`0x${raw}`)
153
+ })
154
+
155
+ const noNewPrivsRaw = readField(text, NO_NEW_PRIVS_FIELD)
156
+ if (!/^\d+$/.test(noNewPrivsRaw)) {
157
+ throw new KubernetesPrivilegeProbeError(
158
+ 'unreadable-output',
159
+ `the privilege probe ran but ${NO_NEW_PRIVS_FIELD} is ${JSON.stringify(clip(noNewPrivsRaw))}, which is not the integer /proc/<pid>/status publishes, so this sandbox's privileges could not be read.`,
160
+ )
161
+ }
162
+
163
+ return {
164
+ capInh: masks[0] as bigint,
165
+ capPrm: masks[1] as bigint,
166
+ capEff: masks[2] as bigint,
167
+ capBnd: masks[3] as bigint,
168
+ noNewPrivs: Number(noNewPrivsRaw),
169
+ }
170
+ }
171
+
172
+ /**
173
+ * Admit only an all-zero capability set with `no_new_privs` set. Every
174
+ * non-zero mask is named in the refusal, because "one of them is set" sends
175
+ * the reader back to the guest to find out which.
176
+ */
177
+ export function assertDeprivileged(privileges: ProcStatusPrivileges, sandboxName: string): void {
178
+ const offenders: string[] = []
179
+ if (privileges.capInh !== 0n) offenders.push(`CapInh=${privileges.capInh.toString(16)}`)
180
+ if (privileges.capPrm !== 0n) offenders.push(`CapPrm=${privileges.capPrm.toString(16)}`)
181
+ if (privileges.capEff !== 0n) offenders.push(`CapEff=${privileges.capEff.toString(16)}`)
182
+ if (privileges.capBnd !== 0n) offenders.push(`CapBnd=${privileges.capBnd.toString(16)}`)
183
+ if (privileges.noNewPrivs !== 1) offenders.push(`NoNewPrivs=${privileges.noNewPrivs}`)
184
+ if (offenders.length === 0) return
185
+
186
+ throw new KubernetesPrivilegeProbeError(
187
+ 'privileged',
188
+ `kubernetes sandbox ${sandboxName} is PRIVILEGED and was refused: ${offenders.join(
189
+ ', ',
190
+ )} (every capability mask must be 0 and NoNewPrivs must be 1). The probe ran and was read successfully — this is the guest's real state, not a diagnostic failure. The image's entrypoint is expected to end with \`exec setpriv --reuid --regid --clear-groups --inh-caps=-all --bounding-set=-all --no-new-privs -- tini -- node agent.cjs\`; a sandbox that reaches this message is running with capabilities the workload could use.`,
191
+ )
192
+ }
193
+
194
+ /**
195
+ * Run the probe over an already-built sandbox's `exec` and admit or refuse.
196
+ *
197
+ * `run` is the sandbox's own `exec`, not the raw transport, so the probe
198
+ * traverses exactly the path every later call will: reserve, admit, stream,
199
+ * confirm. A probe that cannot get through this is a sandbox a caller
200
+ * cannot use either.
201
+ */
202
+ export async function runPrivilegeProbe(
203
+ run: (command: string, args: string[]) => Promise<SandboxExecResult>,
204
+ sandboxName: string,
205
+ ): Promise<ProcStatusPrivileges> {
206
+ let result: SandboxExecResult
207
+ try {
208
+ result = await run(PRIVILEGE_PROBE_COMMAND, [...PRIVILEGE_PROBE_ARGS])
209
+ } catch (error) {
210
+ throw new KubernetesPrivilegeProbeError(
211
+ 'probe-failed',
212
+ `the privilege probe could not run in kubernetes sandbox ${sandboxName}: ${
213
+ error instanceof Error ? error.message : String(error)
214
+ }. The probe is \`${PRIVILEGE_PROBE_COMMAND} ${PRIVILEGE_PROBE_ARGS.join(
215
+ ' ',
216
+ )}\`; an image without it on PATH cannot be admitted, because a sandbox whose privileges cannot be checked is refused rather than trusted.`,
217
+ { cause: error },
218
+ )
219
+ }
220
+ if (result.exitCode !== 0) {
221
+ throw new KubernetesPrivilegeProbeError(
222
+ 'probe-failed',
223
+ `the privilege probe could not run in kubernetes sandbox ${sandboxName}: \`${PRIVILEGE_PROBE_COMMAND} ${PRIVILEGE_PROBE_ARGS.join(
224
+ ' ',
225
+ )}\` exited ${result.exitCode}${
226
+ result.stderr.trim() ? ` (${clip(result.stderr.trim())})` : ''
227
+ }. This is a diagnostic failure, not a privilege failure — the sandbox is refused because its state is unknown.`,
228
+ )
229
+ }
230
+
231
+ const privileges = parseProcStatus(result.stdout)
232
+ assertDeprivileged(privileges, sandboxName)
233
+ return privileges
234
+ }
235
+
236
+ /**
237
+ * The refusal for a probe that never answered.
238
+ *
239
+ * A guest that accepts the TCP connection and then goes quiet — an agent
240
+ * process out of memory, an event loop blocked by the workload, a container
241
+ * alive with a listener that has stopped reading — cannot be told apart from
242
+ * a healthy one by the wire alone, so the caller's acquire budget is the only
243
+ * thing that ends the wait. That expiry is a `'probe-failed'` like any other
244
+ * way the probe could not run, but it gets its own words: "the deadline
245
+ * expired" on its own says nothing about WHICH half of the acquire stopped
246
+ * answering, and a reader who sees `cat` blamed for a hang goes looking for a
247
+ * missing binary that is not missing.
248
+ */
249
+ export function privilegeProbeTimedOut(
250
+ sandboxName: string,
251
+ timeoutMs: number,
252
+ cause: unknown,
253
+ ): KubernetesPrivilegeProbeError {
254
+ return new KubernetesPrivilegeProbeError(
255
+ 'probe-failed',
256
+ `the privilege probe could not run in kubernetes sandbox ${sandboxName}: the guest accepted the connection and did not answer \`${PRIVILEGE_PROBE_COMMAND} ${PRIVILEGE_PROBE_ARGS.join(
257
+ ' ',
258
+ )}\` within ${timeoutMs} ms, so the probe was abandoned. This is a diagnostic failure, not a privilege failure — the sandbox is refused because its state is unknown. A wedged agent (out of memory, an event loop blocked by the workload) looks exactly like this from the host; check the pod's logs before raising the acquire budget.`,
259
+ { cause },
260
+ )
261
+ }