@namzu/sandbox 13.0.0 → 15.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (104) hide show
  1. package/CHANGELOG.md +1147 -0
  2. package/README.md +447 -0
  3. package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
  4. package/dist/backends/aci-standby-pool/index.js +13 -1
  5. package/dist/backends/aci-standby-pool/index.js.map +1 -1
  6. package/dist/backends/docker/index.d.ts.map +1 -1
  7. package/dist/backends/docker/index.js +19 -1
  8. package/dist/backends/docker/index.js.map +1 -1
  9. package/dist/backends/firecracker/index.d.ts.map +1 -1
  10. package/dist/backends/firecracker/index.js +12 -2
  11. package/dist/backends/firecracker/index.js.map +1 -1
  12. package/dist/backends/firecracker/protocol.d.ts +481 -8
  13. package/dist/backends/firecracker/protocol.d.ts.map +1 -1
  14. package/dist/backends/firecracker/protocol.js +136 -0
  15. package/dist/backends/firecracker/protocol.js.map +1 -1
  16. package/dist/backends/firecracker/transport.d.ts +642 -14
  17. package/dist/backends/firecracker/transport.d.ts.map +1 -1
  18. package/dist/backends/firecracker/transport.js +1307 -34
  19. package/dist/backends/firecracker/transport.js.map +1 -1
  20. package/dist/backends/kubernetes/egress-policy.d.ts +1296 -0
  21. package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -0
  22. package/dist/backends/kubernetes/egress-policy.js +2458 -0
  23. package/dist/backends/kubernetes/egress-policy.js.map +1 -0
  24. package/dist/backends/kubernetes/identity.d.ts +193 -0
  25. package/dist/backends/kubernetes/identity.d.ts.map +1 -0
  26. package/dist/backends/kubernetes/identity.js +147 -0
  27. package/dist/backends/kubernetes/identity.js.map +1 -0
  28. package/dist/backends/kubernetes/index.d.ts +1019 -0
  29. package/dist/backends/kubernetes/index.d.ts.map +1 -0
  30. package/dist/backends/kubernetes/index.js +1756 -0
  31. package/dist/backends/kubernetes/index.js.map +1 -0
  32. package/dist/backends/kubernetes/ingress-policy.d.ts +375 -0
  33. package/dist/backends/kubernetes/ingress-policy.d.ts.map +1 -0
  34. package/dist/backends/kubernetes/ingress-policy.js +1050 -0
  35. package/dist/backends/kubernetes/ingress-policy.js.map +1 -0
  36. package/dist/backends/kubernetes/k8s-client.d.ts +334 -0
  37. package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -0
  38. package/dist/backends/kubernetes/k8s-client.js +553 -0
  39. package/dist/backends/kubernetes/k8s-client.js.map +1 -0
  40. package/dist/backends/kubernetes/lease.d.ts +145 -0
  41. package/dist/backends/kubernetes/lease.d.ts.map +1 -0
  42. package/dist/backends/kubernetes/lease.js +201 -0
  43. package/dist/backends/kubernetes/lease.js.map +1 -0
  44. package/dist/backends/kubernetes/objects.d.ts +702 -0
  45. package/dist/backends/kubernetes/objects.d.ts.map +1 -0
  46. package/dist/backends/kubernetes/objects.js +518 -0
  47. package/dist/backends/kubernetes/objects.js.map +1 -0
  48. package/dist/backends/kubernetes/per-sandbox-policy.d.ts +219 -0
  49. package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -0
  50. package/dist/backends/kubernetes/per-sandbox-policy.js +407 -0
  51. package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -0
  52. package/dist/backends/kubernetes/privilege-probe.d.ts +136 -0
  53. package/dist/backends/kubernetes/privilege-probe.d.ts.map +1 -0
  54. package/dist/backends/kubernetes/privilege-probe.js +185 -0
  55. package/dist/backends/kubernetes/privilege-probe.js.map +1 -0
  56. package/dist/backends/kubernetes/rbac.d.ts +153 -0
  57. package/dist/backends/kubernetes/rbac.d.ts.map +1 -0
  58. package/dist/backends/kubernetes/rbac.js +177 -0
  59. package/dist/backends/kubernetes/rbac.js.map +1 -0
  60. package/dist/backends/kubernetes/sandbox.d.ts +190 -0
  61. package/dist/backends/kubernetes/sandbox.d.ts.map +1 -0
  62. package/dist/backends/kubernetes/sandbox.js +433 -0
  63. package/dist/backends/kubernetes/sandbox.js.map +1 -0
  64. package/dist/backends/kubernetes/transport.d.ts +1048 -0
  65. package/dist/backends/kubernetes/transport.d.ts.map +1 -0
  66. package/dist/backends/kubernetes/transport.js +2093 -0
  67. package/dist/backends/kubernetes/transport.js.map +1 -0
  68. package/dist/backends/kubernetes/workspace.d.ts +1512 -0
  69. package/dist/backends/kubernetes/workspace.d.ts.map +1 -0
  70. package/dist/backends/kubernetes/workspace.js +3703 -0
  71. package/dist/backends/kubernetes/workspace.js.map +1 -0
  72. package/dist/backends/remote-execution-controller.d.ts +14 -0
  73. package/dist/backends/remote-execution-controller.d.ts.map +1 -1
  74. package/dist/backends/remote-execution-controller.js.map +1 -1
  75. package/dist/index.d.ts +350 -2
  76. package/dist/index.d.ts.map +1 -1
  77. package/dist/index.js +344 -34
  78. package/dist/index.js.map +1 -1
  79. package/dist/testing/sandbox-conformance.d.ts +227 -0
  80. package/dist/testing/sandbox-conformance.d.ts.map +1 -0
  81. package/dist/testing/sandbox-conformance.js +896 -0
  82. package/dist/testing/sandbox-conformance.js.map +1 -0
  83. package/package.json +5 -4
  84. package/src/backends/aci-standby-pool/index.ts +16 -1
  85. package/src/backends/docker/index.ts +22 -1
  86. package/src/backends/firecracker/index.ts +14 -2
  87. package/src/backends/firecracker/protocol.ts +541 -6
  88. package/src/backends/firecracker/transport.ts +1687 -64
  89. package/src/backends/kubernetes/egress-policy.ts +3448 -0
  90. package/src/backends/kubernetes/identity.ts +261 -0
  91. package/src/backends/kubernetes/index.ts +2670 -0
  92. package/src/backends/kubernetes/ingress-policy.ts +1344 -0
  93. package/src/backends/kubernetes/k8s-client.ts +742 -0
  94. package/src/backends/kubernetes/lease.ts +254 -0
  95. package/src/backends/kubernetes/objects.ts +983 -0
  96. package/src/backends/kubernetes/per-sandbox-policy.ts +542 -0
  97. package/src/backends/kubernetes/privilege-probe.ts +261 -0
  98. package/src/backends/kubernetes/rbac.ts +192 -0
  99. package/src/backends/kubernetes/sandbox.ts +593 -0
  100. package/src/backends/kubernetes/transport.ts +2895 -0
  101. package/src/backends/kubernetes/workspace.ts +5640 -0
  102. package/src/backends/remote-execution-controller.ts +14 -0
  103. package/src/index.ts +838 -35
  104. package/src/testing/sandbox-conformance.ts +1202 -0
package/CHANGELOG.md CHANGED
@@ -1,5 +1,1152 @@
1
1
  # @namzu/sandbox
2
2
 
3
+ ## 15.0.0
4
+
5
+ ### Major Changes
6
+
7
+ - a5c19bf: A Kubernetes API request that the API server accepts and never answers now fails after 30 seconds instead of hanging forever.
8
+
9
+ **What changes for you.** Every request `createKubernetesClient` sends carries its own bound — `apiRequestTimeoutMs`, default `30000` — on top of whatever `AbortSignal` the caller passed. It covers `getToken()`, the connection and reading the body, on both the `fetch` and the `node:https` path. Expiry rejects with the new `KubernetesApiTimeoutError`, which carries the verb, the resource path and the bound that expired, and never the bearer token; it is deliberately not folded into the generic `kubernetes … failed:` error, so a caller can tell a timeout from a refusal. A caller's own abort behaves exactly as it did. **To keep something closer to the old behaviour on a genuinely slow cluster, raise the number** — `apiRequestTimeoutMs: 120000`, say. There is no value that disables the bound: `0` and anything below the `1000` floor are refused at construction.
10
+
11
+ **Why a signal was never enough, and why this is worth a major.** `signal` is optional on every one of these calls, and several of them are single-flight promises that run under whichever caller arrived first and never consult a later one's. A workspace `suspend()` is one: a plain `destroy()` joins it, a `resume()` queues behind it, and while it is in flight the handle's state is `suspending`, so every data-plane call is already refused. One signal-less call against an unanswering API server therefore pinned the entire handle, and a host calling `destroy()` during shutdown hung until it was killed. The same shape applies to the task backend's shared egress verification and to a joined teardown. Putting the bound in the client covers those, the standalone workspace verbs and any verb added later, with no wiring at each call site.
12
+
13
+ A timeout cannot tell whether the request was applied, and nothing pretends otherwise: `suspend()` restores the state it saw and re-sends its idempotent patch, a create `POST` that timed out but did land is adopted through the 409 path, and a `DELETE` that had already applied counts as done.
14
+
15
+ **Also in this release, and opt-in rather than a changed default: stream liveness.** Once `openTerminal` or `openTcpConnection` reported ready, the transport cleared its read-idle timer — correctly, since a quiet shell is healthy — and nothing replaced it, so a peer that vanished without a FIN or an RST left `exited`/`closed` unresolved on the host and the shell's process group alive in the guest until the pod stopped. Streams now trade a `{ type: 'heartbeat' }` frame every `streamHeartbeatMs` (new on the Kubernetes backend config, default `15000`, `0` sends none). Three consecutive intervals with nothing at all arriving end the stream: the host resolves `exited` with `exitCode: -1`, exactly as a closed socket already produces, and resolves `closed`; the guest runs the same cleanup a closed socket runs. Both sides count bytes rather than whole frames, so a large frame still on its way proves the peer is there; silence while a side has paused reading for backpressure is not counted, in either direction; and each side polls at a quarter of the interval, so a dead stream is noticed within three intervals plus at most one more tick — 45 seconds plus up to 3.75 more at the default. TCP keepalive is enabled on both ends of the routed connection as well.
16
+
17
+ The heartbeat is **negotiated per stream and off unless both peers asked**, so no existing stream's behaviour changes: the open request carries the interval, an agent that implements it echoes the interval it will use in its `ready` event and only then starts sending, and the host only starts once that echo arrived. The echo is a number from the pod and the host times its own watchdog with it, so the host honours it only between `100` ms and four times what it asked for; this agent clamps to the same floor before echoing, so an honest echo is never altered. An agent built before this change ignores the unknown field and echoes nothing; an older host never asks and is therefore never sent a frame type it would treat as a protocol error. The guest advertises `stream-heartbeat` in its `healthz` `features` and the guest wire protocol version is unchanged, so no host and no golden image has to roll with this release.
18
+
19
+ **The Firecracker tier is untouched.** `VsockTransportOptions.heartbeatMs` is new and undefined by default; only the Kubernetes backend opts in. A default on the shared transport would have force-closed an existing consumer's quiet-but-alive terminal after 45 seconds, which is a changed default for a tier this change does not claim. Keepalive is set on the routed `tcp` arm only.
20
+
21
+ New on `KubernetesBackendConfig`: `apiRequestTimeoutMs` and `streamHeartbeatMs`. Newly exported from `@namzu/sandbox`, all additive: `KubernetesApiTimeoutError` and the type of its `verb`, `KubernetesHttpMethod`; `DEFAULT_API_REQUEST_TIMEOUT_MS` and `MIN_API_REQUEST_TIMEOUT_MS`; `DEFAULT_STREAM_HEARTBEAT_MS`; and, for a host writing its own guest or asserting what this one advertises, `STREAM_HEARTBEAT_FEATURE`, `STREAM_HEARTBEAT_MISS_LIMIT`, `MIN_STREAM_HEARTBEAT_MS`, `STREAM_HEARTBEAT_MAX_ECHO_FACTOR` and the type `StreamHeartbeat`. `createKubernetesClient` takes an optional second `KubernetesClientOptions` argument. Nothing on the SDK's `Sandbox` changes, so no `@namzu/sdk` changeset accompanies this.
22
+
23
+ - 3a6651c: A Kubernetes acquire that fails now says why, in a field. `create()` rejects with `KubernetesAcquireError` rather than a plain `Error` or a bare `ReadinessPollTimeout`, and a transient API failure during readiness is retried inside the readiness budget instead of ending the acquire on sight.
24
+
25
+ **What breaks.** Two things, and both are about what a caller catches or how long it waits.
26
+
27
+ 1. **The thrown type changed.** A `create()` that ran out of readiness budget used to reject with `ReadinessPollTimeout`; it now rejects with `KubernetesAcquireError` carrying `reason: 'not-ready'`, with the `ReadinessPollTimeout` as its `cause`. A host doing `catch (e) { if (e instanceof ReadinessPollTimeout) … }` around `create()` must look at `e.cause`, or — better — switch to `e.reason`. A host matching message text does not survive at all: the message for a diagnosed refusal is new. `KubernetesWorkspace` is unaffected — the workspace lifecycle does not go through acquire and still raises `ReadinessPollTimeout` directly.
28
+
29
+ 2. **A doomed `create()` can now take the full `readyTimeoutMs` where it used to fail in milliseconds.** A readiness `GET` that fails with a connect error, an `apiRequestTimeoutMs` expiry, a 429 or a 5xx is repeated inside the existing readiness deadline, honouring the server's `Retry-After` (which may only slow the poll down, never speed it past `readyPollIntervalMs`). One clock, so a retry spends the budget rather than extending it and `create()` still cannot outlive the timeout its caller chose — but with the default 60 s budget, an acquire against an API server that is down now rejects after 60 s instead of after one round trip. A host with its own outer timeout will notice. To keep the old latency, lower `readyTimeoutMs`. The create `POST` is never retried: it is not idempotent, and a POST whose answer never arrived may already have committed.
30
+
31
+ **What you get for it.** `KubernetesAcquireError.reason` is one of `api-unreachable`, `api-timeout`, `forbidden`, `claim-rejected`, `capacity`, `image-pull` or `not-ready`, with a `retryable` flag and the original failure as `cause`. A burst past node capacity and an API outage were previously the same plain `Error`, separable only by matching message text any release is free to reword.
32
+
33
+ And a claim the controller has already **refused** no longer waits out the clock. Four `status.conditions[Ready]` reasons mean decided rather than not-yet — `WarmPoolNotFound`, `TemplateNotFound`, `InvalidMetadata`, `EnvVarsInjectionRejected` — and meeting one ends the acquire on the first read, carrying the controller's own reason and message on `controllerReason`/`controllerMessage`. Measured against agent-sandbox v1.0.2 on Kubernetes v1.37.0: a claim naming a warm pool that does not exist is refused in 233 ms against a 60 000 ms budget, with the claim deleted behind it. Those four strings were read off a live controller rather than copied from a changelog, and any other reason still falls through to the deadline — an unrecognised one is never guessed at.
34
+
35
+ `capacity` (`PodScheduled=False`/`Unschedulable`) and `image-pull` come from a single pod `GET` made only after the budget has already gone, and before the cleanup `DELETE`. When a pod condition and a retried API failure both describe one refusal — the ordinary state of a saturated cluster — the pod condition decides, because it is what the cluster published about this pod; and a failure the poll recovered from is forgotten rather than carried to the end, so it can neither rename a diagnosed refusal nor be quoted by a poll that simply ran out of time. Nothing on the successful path reads anything it did not read before: a clean acquire issues exactly the requests it issued in the previous release, pinned by a request-log test.
36
+
37
+ **Newly exported from the package root**, because catching by class is not possible from outside otherwise: `KubernetesAcquireError` and its `KubernetesAcquireFailureReason`, `TERMINAL_CLAIM_REASONS`, `ReadinessPollTimeout`, `KubernetesApiError` (new — it carries the verb, path, `status`, the `Retry-After` the server sent, and whether the failure was a connect or a status), `KubernetesApiFailureTransport`, `KubernetesCredentialError`, `KubernetesAlreadyGoneError`, `KubernetesConflictError`, and the three egress refusals `KubernetesUnenforceableEgressPolicyError`, `KubernetesEgressPolicyNotAppliedError` and `KubernetesEgressPolicyMismatchError`. Nothing was renamed or removed, and `@namzu/sdk` is untouched.
38
+
39
+ **A refusal this taxonomy cannot honestly diagnose is not filed under the least wrong reason** — a malformed template, a 400 from an admission webhook, a controller that reported `Ready` and named no sandbox all travel out as themselves. That is deliberate: a `reason` that meant "something else" would be worth nothing to the host reading it.
40
+
41
+ - ff6134f: The Kubernetes backend now refuses to create a sandbox when ANY policy selecting its pods lets out more than `config.egress` says — not just when the one object it GETs by name has drifted. A deployment with a second `NetworkPolicy` over those pods, including the controller-managed one a `SandboxTemplate`'s own `networkPolicy` block becomes, used to work and now fails with `KubernetesEgressPolicyUnionError`; `egress: { policy, verify: 'named-object-only' }` restores the previous single-object check exactly. This is the SECOND default-on refusal in this release — the other is `KubernetesIngressPolicyError`, which is about the agent port being reachable INBOUND. They are distinct classes, and each message opens by naming what it refused.
42
+
43
+ What was open before: verification GETted one object, compared it to the translation exactly, and memoized the pass for the whole life of the backend. The API server UNIONS every policy selecting a pod — traffic leaves if any of them allows it — so that check could only ever prove one policy was not the problem. In agent-sandbox v1.0.2 a `SandboxTemplate` that sets `networkPolicy` has its egress translated verbatim into a managed policy, and a template that omits the block gets a controller default allowing `0.0.0.0/0` minus RFC 1918 and `169.254.0.0/16` INSTEAD — not underneath. So deleting a `networkPolicy` block opened the internet (and left `100.64.0.0/10`, `127.0.0.0/8` and `168.63.129.16/32` reachable) while `namzu-task-egress` still verified perfectly. The shipped `sandboxtemplate-task.yaml` said that managed baseline "still applies underneath"; it does not, and that comment is corrected here.
44
+
45
+ Two new Kubernetes-only kinds on `KubernetesEgressConfig.policy`. `{ kind: 'no-network' }` emits `policyTypes: ['Egress']` with no rule at all: nothing leaves the pod, the cluster's own resolver included, which is what `deny-all` never meant — `deny-all` allows UDP/TCP 53 to `kube-system`, and a cluster resolver forwards outside names upstream. A `no-network` sandbox resolves nothing; the agent needs no resolver because the host dials in, but a workload that resolves anything fails, which is the point. `{ kind: 'public-internet', exceptCidrs? }` emits DNS to the resolver's own pods plus `0.0.0.0/0` except `10.0.0.0/8`, `172.16.0.0/12`, `192.168.0.0/16`, `100.64.0.0/10`, `169.254.0.0/16`, `127.0.0.0/8` and `168.63.129.16/32`, and `::/0` except `fc00::/7`, `fe80::/10` and `::1/128` — out, but not sideways to the node, the service network, the API server, instance metadata or another sandbox pod. `exceptCidrs` adds to that list and is routed to the block of its own address family; an entry that is not a CIDR is refused at construction with the new `KubernetesEgressPolicyConfigError`.
46
+
47
+ **`deny-all` and `allow-all` emit byte-for-byte what they always have.** Nothing about them changed, deliberately: verification of the named object is an exact match, so tightening either translation would stop every already-applied policy from verifying and fail every `create()` until an operator re-applied it. A test pins both manifests by deep equality. The tier-wide `EgressPolicy` union is untouched, so nothing changes for the process, container, Firecracker or ACI tiers and `@namzu/sdk` is not affected.
48
+
49
+ What runs now, by default, wherever `config.egress` is set: the named-object check exactly as before, and then, at the same two points the ingress check already runs — before the POST for a directly created Sandbox, after the bind for a claimed one, and before any resume patch on `createKubernetesWorkspace` — a LIST of the namespace's `NetworkPolicy` objects (and, under `engine: 'cilium'`, that CNI's CRD), evaluated against the pod's real labels. It refuses when a selecting policy allows a destination the translation does not (`refusal: 'policy-widens-egress'`; under `no-network`, any egress rule at all), when nothing selecting the pod puts it in egress default-deny so the translation bounds nothing (`'no-enforcing-policy'`, skipped under `allow-all`), or when a policy, peer, port or collection cannot be read (`'not-evaluable'`). The refusal carries the pod's labels and every policy examined with a verdict each. A pass is cached per label set for at most five minutes — not for the backend's lifetime — and a failure is never cached. Subset is decided conservatively: a second policy passes only when the check can show it is inside the translation, and anything it cannot place inside refuses.
50
+
51
+ To take this upgrade: run one `create()` against each namespace and read the refusal if there is one — it names every policy it examined, so the fix is the line that says `widens-egress`. In practice that means removing or narrowing the placeholder rule in `k8s/manifests/networkpolicy.yaml`, and keeping a template's inline `networkPolicy` egress inside whatever `config.egress.policy` names. `k8s/manifests/rbac.yaml` already grants `list` on `networkpolicies` (and on `ciliumnetworkpolicies`) for the ingress check; without it this check refuses with `not-evaluable` naming the missing verb. A host that cannot be granted it, or that accepts the union it has, sets `egress.verify: 'named-object-only'` and says so on purpose.
52
+
53
+ Also new on the public surface: `KubernetesOnlyEgressPolicy`, `KubernetesEgressPolicy`, `KubernetesEgressVerification`, `KubernetesEgressPolicyUnionError`, `KubernetesEgressPolicyConfigError` and the `EgressPolicyRefusal` / `EgressPolicyVerdict` / `ExaminedEgressPolicy` reporting types; `UnreadIngressPolicySource` is renamed `UnreadPolicySource` because both checks now report it, with the old name kept as a `@deprecated` alias. What no test proves: enforcement. `kind` accepts every `NetworkPolicy` and enforces none and runs no Cilium data plane, so every "it was blocked" probe there passes for the wrong reason — `k8s/scripts/egress-check.mjs` is the live check and runs two positive controls before it reads any result. Nothing about the guest wire protocol, the bind token, the privilege probe, ingress verification, suspend, resume or deletion changed.
54
+
55
+ - 984d966: A Kubernetes workspace's writes are now put on its disk before its pod stops, and `suspend()` asks for that by default.
56
+
57
+ **What was wrong.** Nothing in the guest agent or the Kubernetes backend ever called `sync`, `syncfs` or `fsync`. A `write-file` answered `ok` as soon as the bytes were in the guest's page cache, and `suspend()` patched the pod away and waited for it to stop — a wait both the code and the docs treated as the point the disk was quiesced. It is not: a stopped pod means only that nothing is writing any more, and whether the guest's dirty pages reached the device depends on how the runtime tore the guest down. Measured on a cluster with a VM runtime class: the container was killed with exit 137 about a second into a five-second stop, while `kill -TERM 1` from inside the guest ended it cleanly in about a second.
58
+
59
+ **Three operator-visible changes, in the order they will be noticed.**
60
+
61
+ 1. **`sandboxtemplate-workspace.yaml`'s `terminationGracePeriodSeconds` moves from 5 to 30, and the container gains a `preStop` hook** (`/entrypoint.sh prestop`: `sync -f` the workspace, signal pid 1, wait for it). **The 30 is a budget, not a measurement** — 15 s of it is the agent's own `NAMZU_AGENT_SHUTDOWN_DEADLINE_MS`, the rest is for the hook and the kubelet — and it must be measured on your runtime, because it depends on the runtime class, the storage class and how much a workload leaves dirty. On the runtime measured above the stop lasted the whole grace period even though the container was gone in a second, so **this is roughly what every `suspend()` will cost there**, up from about 7 s. To keep the old behaviour, set `terminationGracePeriodSeconds: 5` in your own copy of the template and drop the `lifecycle` block; re-applying the manifest does not change an existing workspace, which carries its own copy of `spec.podTemplate`. Expect a `FailedPreStopHook` warning on every stop that worked: the hook ends pid 1 of the container's pid namespace and the kernel then SIGKILLs the hook with it, so the event records the success path rather than a defect. The hook's own wait is derived from the agent's bound rather than fixed — `ceil(NAMZU_AGENT_SHUTDOWN_DEADLINE_MS / 1000) + 2` seconds, 17 as shipped — because the kubelet runs the hook, waits for it, and only then signals pid 1: a shorter wait would end the hook mid-drain and hand the agent a stop signal sent only because the hook gave up. Raise `NAMZU_AGENT_SHUTDOWN_DEADLINE_MS` and the hook's wait follows; `NAMZU_PRESTOP_WAIT_SECONDS` overrides the derivation for an operator who wants to own the relationship. The hook signals pid 1 only after `/proc/1/comm` says pid 1 is the `tini` this image starts — a derived image that boots a different init sets `NAMZU_PRESTOP_INIT_NAME` or gets no signal (and a line on stderr) instead of an arbitrary process being killed. Note also that the grace-period countdown includes the hook's own unbounded `sync -f`, so the number to clear is `sync-time + NAMZU_AGENT_SHUTDOWN_DEADLINE_MS`, not 15 s alone. **`sandboxtemplate-task.yaml` keeps its own 5 s grace period and now sets `NAMZU_AGENT_SHUTDOWN_DEADLINE_MS: 3000` inside it** — the same pair read the other way. The agent's `SIGTERM` handler is a bounded drain rather than the prompt exit it used to be, and with that variable unset its 15000 ms default outlived a task pod's 5 s grace period three times over: a task pod with anything still running at stop time was SIGKILLed mid-drain instead of exiting 0 on its own bound. The task template deliberately carries no `preStop` hook, and cannot at this grace period — the wait a hook derives from a 3000 ms deadline, `ceil(3000 / 1000) + 2` = 5 s, is the whole of it, leaving the kubelet no room after the drain — and a diskless task pod has nothing persistent for the hook's `sync -f` to flush in any case. What its 3 s bounds is a graceful stop of the guest's own processes: close the listener, quiesce what is running, exit 0.
62
+ 2. **`suspend()` and the `destroy()` that suspends now send the guest a `flush` op and wait for the reply before the `Suspended` patch.** Exactly ONE outcome stops the suspend: a guest that ANSWERED and could not confirm rejects with the new `KubernetesFlushUnconfirmedError` and **sends no patch at all** — the workspace stays running and serving, and the caller holds the guest's own message. That is a new way for `suspend()` to fail, and `suspend({ flush: false })` is byte-for-byte the suspend of every previous release. Everything else is reported and the suspend goes ahead, because `suspend()` is the verb an operator reaches for when a workspace has gone wrong: an image that cannot flush (an agent predating the op, or one with no `sync` to run) tells the host through the new `KubernetesWorkspaceOptions.onFlushUnsupported`, and a guest that cannot be REACHED to be asked — crashed, OOM-killed, off the network, or fenced with `agent_retiring` — through the new `onFlushUnreachable`, carrying the transport's error as its `cause`. Reaching an unreachable guest costs the transport's connect-retry budget (30 s) before the suspend goes on, so suspending a wedged workspace is slower than it was, but it still works: refusing there would leave exactly that workspace running and billing. A suspend already in flight that asked for no flush refuses a joining caller that wants one, the way an in-flight suspend already refuses a joining `quiesce` — which includes a plain `destroy()`, so that is the one shape in which `destroy()` in a `finally` block is not a no-op; `destroy({ flush: false })` joins such a transition deliberately. `suspend({ flush: { timeoutMs } })` and `destroy({ flush: { timeoutMs } })` raise what the GUEST may spend inside the `syncfs`, for a workspace that leaves more dirty than its 10000 ms default covers; the option is `boolean | { timeoutMs }`, the shape `quiesce` already had for `graceMs`, and the new `KubernetesWorkspaceFlushRequest` type is exported for it.
63
+ 3. **`writeFile` resolves later, and replaces rather than overwrites.** A reply of `ok` now means the bytes are on the device: the guest writes a temp sibling, `fsync`s it, renames it onto the target and `fsync`s the directory. Every write of any size is therefore atomic at the target — a write that fails leaves the previous contents rather than a truncated file — and the target's mode is carried onto the replacement, on a whole-body write and on a sequence's final part alike (a mode the guest cannot carry over fails the write rather than renaming a file that has a different one). The cost is one `fsync` per write (one per part sequence for a chunked body, on the last part, which covers the whole file). Four consequences of the rename now reach writes of every size, where before they reached only a part sequence — a body above about 5.9 MiB of raw content, the point at which one frame stops holding it, since 8 MiB is the frame's ceiling and not the body's: hard links to the target are broken (the other name keeps the old inode), the replacement is owned by the agent's uid whatever the previous file's owner was, the CONTAINING DIRECTORY must now be writable (overwriting a writable file inside a directory the agent's uid cannot write to used to succeed and now fails when the temp sibling is opened), and a `.namzu-write-….part` sibling is briefly visible to a `listFiles` or `walkFiles` racing the write. A fifth is the single-frame path's alone, and is one a part sequence never had: a basename over about 200 characters fails with `ENAMETOOLONG`, because the temp sibling's name is the target's plus 55 bytes — the host names a part sequence's temp file itself, with the target's basename truncated to 96 characters, so those never carried this one either before or after.
64
+
65
+ **Also new, and additive:** `KubernetesWorkspace.flush(options?)` for the moments that are not a suspend (before a snapshot, before a drain), with `KubernetesFlushOptions`, `KubernetesFlushReport`, `KubernetesFlushUnsupportedError`, `KubernetesFlushUnreachableError` and the `FLUSH_FEATURE` string exported alongside it. The standalone `suspendKubernetesWorkspace` never dials a guest, so it cannot flush: it refuses an explicit `flush: true` rather than ignoring it. The agent's `SIGTERM` handler now stops accepting connections, stops every process the guest is running (the routine `quiesce` already used — there is one implementation of that, not two), flushes and exits 0, all bounded by `NAMZU_AGENT_SHUTDOWN_DEADLINE_MS` (15000 ms) with `NAMZU_AGENT_FLUSH_TIMEOUT_MS` (10000 ms) bounding one flush. That deadline is the only bound: a repeat `SIGTERM` is logged and ignored rather than exiting at once, because in a stopping pod the second signal is the kubelet's own — sent the moment the `preStop` hook returns — and not anybody asking for a shorter wait. `SIGINT`, `SIGQUIT` and `SIGKILL` are unhandled and still end the process immediately. The guest wire protocol version is unchanged: `flush` is an additive op advertised in the `healthz` `features` list, so no host and no image has to roll together with this release — and a guest advertises it only if it can perform one, since the flush runs `sync -f` and a derived image that strips coreutils has the code and no `sync`. A bare `sync` is deliberately not a fallback anywhere here, in the agent or in the hook: it flushes every mounted filesystem, and on a runtime sharing the host kernel that is the node's disks.
66
+
67
+ **What is not proven.** That a kubelet runs a `preStop` hook under the runtime class the shipped manifests name was never measured — the three mechanisms above are deliberately independent for that reason — and no test can prove the device wrote what the kernel handed it. `k8s/scripts/suspend-resume.mjs` now covers a 5 MiB `writeFile` and a 100 MiB command-written file issued immediately before the suspend, and prints the suspend's own duration; that is the measurement to run on a real cluster before trusting the grace period this ships with.
68
+
69
+ - 87c5337: The Kubernetes-backend guest entrypoint (`packages/sandbox/k8s/entrypoint.sh`) now resolves and exports a writable `HOME`, `USER`, `LOGNAME`, `XDG_CACHE_HOME` and `XDG_CONFIG_HOME` before starting the guest agent, on both the root and non-root exec paths.
70
+
71
+ **Before:** `setpriv --reuid=/--regid=` (the exec every pod ends with) changes only the running process's credentials, never the environment, so `HOME` stayed whatever it was before the drop — `/root` on every pod, since nothing in the container ever ran as anyone else first. `/root` is not readable or writable by the de-privileged agent uid (verified in #469), and `agent.cjs`'s `childEnvironment` copies every non-`NAMZU_AGENT_`/`NAMZU_SANDBOX_` environment variable into every `execute` and terminal child, so the broken value reached every process a task ever started: LibreOffice without `-env:UserInstallation`, `pip install --user`, npm's cache, and the fontconfig/matplotlib caches all failed against it.
72
+
73
+ **After:** `entrypoint.sh` resolves a home once, before either exec site. It first tries `getent passwd "$AGENT_UID"` field 6 — the shipped image's own `useradd --create-home` already creates this directory (`/home/namzu` as built) owned by the agent uid, so this is the common case, and `USER`/`LOGNAME` come from that same passwd entry. If `getent` is missing, the uid has no entry, the entry names a directory under the workspace root, or the directory still cannot be made to belong to the agent uid, it falls back to `/tmp/namzu-home-$AGENT_UID` (created fresh, mode `0700`, owned by the agent uid, `USER`/`LOGNAME` set to `namzu`) — never under the workspace root, so it can never appear in `listFiles`, `walkFiles` or an archive. Only if both fail does the pod refuse to start. `agent.cjs` itself is unchanged: the propagation mechanism this relies on (`childEnvironment`) already carried `HOME` correctly — it just never had a correct value to carry.
74
+
75
+ **Major, not patch:** this changes the guest's exported environment on the NORMAL path, for every task and workspace pod, not only in an error configuration. A derived image that depended on `HOME=/root` inside the guest — the only thing running as root before the privilege drop could have relied on — must now set `HOME` (and `USER`/`LOGNAME`/`XDG_*`, if it depends on those too) itself, after its own `FROM`, since this resolution runs unconditionally on every boot and always wins over whatever the base image set. A derived image that changes `AGENT_UID` needs no changes: it either already ships a passwd entry for that uid naming a directory it owns, or falls back to `/tmp` automatically.
76
+
77
+ `k8s/scripts/capability-check.mjs` now also prints `$HOME`, whether it exists, and whether a probe write succeeded, informationally — like its existing `Seccomp` and set-id lines, this never fails the check itself.
78
+
79
+ - 29c510d: The Kubernetes-backend task `SandboxTemplate` (`packages/sandbox/k8s/manifests/sandboxtemplate-task.yaml`, and the kind overlay's `patch-task-no-runtimeclass.yaml`/`patch-workspace-filesystem-pvc.yaml`) no longer runs its container `privileged`. It now runs as `runAsUser: 1001`, `runAsGroup: 1001`, `runAsNonRoot: true`, `allowPrivilegeEscalation: false`, `capabilities: { drop: [ALL] }` and `seccompProfile: { type: RuntimeDefault }` — the container-level shape of the Pod Security Standards `restricted` profile.
80
+
81
+ **Before:** every task sandbox pod started as root with the full capability bounding set and no seccomp filter possible (Kubernetes runs a `privileged` container unconfined regardless of any `seccompProfile` named alongside it). The guest agent itself was still deprivileged by `entrypoint.sh`'s `setpriv` before it ever ran — the acquire-time privilege probe (`src/backends/kubernetes/privilege-probe.ts`) verified that on every acquire, and still does — but anything else that ran in the pod (most concretely, a `kubectl exec` shell, since the image sets no `USER`) got uid 0 and every capability the pod granted, for no runtime reason: the task template mounts no block device, so `entrypoint.sh`'s only root-requiring step never ran for it.
82
+
83
+ **After:** a task pod never has root or a capability at any point in its life. `entrypoint.sh` now branches on its own `id -u`: non-root, it execs `setpriv --no-new-privs -- tini -- node agent.cjs` directly (skipping the `--reuid`/`--regid`/`--clear-groups`/`--inh-caps`/`--bounding-set` flags a non-root process cannot run anyway), and refuses outright, naming the uid, if a device is set on a non-root pod — a misconfiguration that used to look like a silently-skipped format. `sandboxtemplate-workspace.yaml` is unaffected: it keeps `privileged: true`, because its device branch genuinely needs `CAP_SYS_ADMIN` to `blkid`/`mkfs.ext4`/`mount` a raw block device before dropping every capability itself.
84
+
85
+ `k8s/Dockerfile` also strips every setuid/setgid bit its own packages carry (`util-linux`/`e2fsprogs` on `node:22-bookworm-slim` ship `su`, `mount`, `umount`, `passwd`, `chsh`, `chfn`, `gpasswd`, `newgrp`, `chage`, `expiry` and `/usr/sbin/unix_chkpwd` set-id) — defence in depth for a process that ever runs non-root without `--no-new-privs` set some other way, not something the entrypoint or the probe depend on. A task image `FROM`ing this one and layering its own packages on top must repeat that `find`/`chmod` step after its own installs.
86
+
87
+ **Major, not patch:** a task pod's default runtime capabilities changed. A task workload that relied on root or an ambient Linux capability inside a task sandbox — mounting something itself, binding a privileged port, `CAP_NET_RAW`, writing to a path only root owns, invoking a setuid binary the image used to ship — worked before this change and fails now. To keep the old behaviour, fork `sandboxtemplate-task.yaml` (or patch it after applying) back to `privileged: true`, understanding that this also re-opens the capability surface the acquire-time privilege probe never covered (a `kubectl exec` shell, or anything else that bypasses the guest agent's own `setpriv`). The guest agent's own process tree is unaffected either way — it was already fully deprivileged by `entrypoint.sh` before this change, and the privilege probe's admission rule (all four capability masks zero, `NoNewPrivs` 1) is unchanged.
88
+
89
+ No code on the published npm package's surface changed — `packages/sandbox`'s `files` array packs only `dist` and `src`, and none of `k8s/` is in it. The major bump is for the shipped Kubernetes deployment artifacts a consumer applies directly to their own cluster, which this repository documents and versions as part of `@namzu/sandbox`.
90
+
91
+ - edf1d79: The Kubernetes backend now refuses to create a sandbox whose agent port no applied policy closes. A deployment that never applied `packages/sandbox/k8s/manifests/networkpolicy.yaml` used to work, insecurely, and now fails with `KubernetesIngressPolicyError`; `ingress: 'unverified'` on the backend config restores the old behaviour for a deployment whose boundary lives somewhere a namespaced `Role` cannot read.
92
+
93
+ What was open before: every `Sandbox` this backend POSTs directly — that is every persistent workspace and every task sandbox created with `warmPoolName` unset — carried no ingress policy at all unless an operator had applied that one manifest, and nothing checked. The templates' inline `networkPolicy` block looked like coverage and was not: the controller translates it into a policy selecting `agents.x-k8s.io/sandbox-template-ref-hash`, a label it writes only onto a Sandbox adopted out of a `SandboxWarmPool`. Measured on a managed cluster running a policy-capable CNI, pods created that way had enforcement on egress only, and TCP 1024 answered — with zero policy drops — from a pod in another namespace, a pod on another node, an unlabelled pod in the sandbox namespace, and host-network pods on both nodes. The guest agent's own source calls the network rule in front of that port the boundary; the bind token was the only thing actually in the way.
94
+
95
+ What runs now, by default: before the POST for a directly created Sandbox (so a refusal leaves no Sandbox and no PVC behind), after the bind for a claimed one (where a refusal releases the claim), and before the POST on every `createKubernetesWorkspace` — including the adopt path, so a workspace whose port stopped being covered is refused asleep rather than woken up to be refused. The check LISTS the namespace's policies and evaluates their selectors against the pod's real labels, never a policy name, and passes only when at least one ingress-enforcing policy selects the pod and none of them admits a wide-open peer on the agent port. Policies union, so one open rule fails the check however many closed ones sit beside it. The refusal carries the pod's labels, the port and every policy examined with a verdict each, plus `unread` — the collections it could not enumerate, so a refused (403) or unserved (404) list is reported as such instead of as a namespace that holds no policy, and the action it names is granting the missing verb rather than applying a policy the check never got to look for. Every wire field is read defensively in one direction: one that arrives as something the schema does not declare (a `spec.ingress` that is not a list of rules, a `podSelector` that is not a selector, a Cilium `fromCIDR` that is a string) refuses rather than being read as its empty value, and a `CiliumNetworkPolicy` carrying `enableDefaultDeny: { ingress: false }` cannot count as coverage — it allows without isolating the endpoint — though what it admits is still read as an opening.
96
+
97
+ To take this upgrade: apply `k8s/manifests/networkpolicy.yaml` (it is in the README's apply order already and in the kind overlay), and add `list` on `networkpolicies` to the host's `Role` — `k8s/manifests/rbac.yaml` now grants it, next to the `get` the egress check uses. Under `ingress: { engine: 'cilium' }` the same two verbs are needed on `ciliumnetworkpolicies`. A cluster that closes the port through a cluster-scoped policy, a service mesh or a cloud security group sets `ingress: 'unverified'`, which reads no policy and issues no request; that is a supported configuration, stated on purpose rather than inherited.
98
+
99
+ Also new on the public surface: `KubernetesIngressConfig`, `KubernetesIngressEngine`, `KubernetesIngressPolicyError` and its `IngressPolicyRefusal` / `IngressPolicyVerdict` / `ExaminedIngressPolicy` / `UnreadIngressPolicySource` reporting types, plus `KubernetesBackendConfig.ingress`. This refusal is distinct by class from every other one a create can raise, so an open agent port is never confused with an unenforceable egress policy or a slow API server. Nothing about the guest wire protocol, the bind token, the privilege probe, egress translation, suspend, resume or deletion changed.
100
+
101
+ - f6bfb7d: Kubernetes workspace failure paths no longer suspend a workspace that is in use. Two observable defaults change, and a suspend on this backend deletes the pod — every terminal, dev server and running command in it, for every process holding the workspace.
102
+
103
+ **A start that fails no longer always suspends.** `createKubernetesWorkspace` and `KubernetesWorkspace.resume()` used to send `operatingMode: Suspended` on any error while bringing a session up. The errors that reach that path include a caller's `AbortSignal` firing during readiness, a single 5xx or 429 on a Sandbox or pod `GET` (the client does not retry), and a privilege probe that overran its deadline — none of which is a fault of the workspace, and all of which a _second_ process meets while the _first_ is executing in the pod. A workspace id is a name and not a lock, so that second process is the ordinary case: a restart, or a second revision during a rollout.
104
+
105
+ A failed start now patches only when the call itself moved the mode — it `POST`ed the object, or its `Running` patch took the object out of `Suspended`. An adopt of an already-running workspace, and a `resume()` that finds another process has already woken it, write nothing and rethrow. `resume()` reads `spec.operatingMode` before patching instead of patching blind, so it no longer claims authorship of a wake it did not perform (and no longer restamps `sandbox.namzu.ai/operating-mode-changed-at` for a mode that did not change).
106
+
107
+ _To keep the old behaviour there is nothing to do for the case it was right about_: a workspace this call woke and then failed to start is still suspended, because leaving it `Running` with a pod nobody is using burns a node. If you want **no** patch on any start failure — you keep your own holder record and sweep idle workspaces yourself — pass `onStartFailure: 'leave'`, on `KubernetesWorkspaceOptions` for the handle or on `KubernetesWorkspaceTransitionOptions` for one `resume()`. The old blanket behaviour (suspend on every start failure, including another process's workspace) is deliberately not offered.
108
+
109
+ One thing a failed start still does, unchanged, is leave the handle without a session: after a `resume()` that rejected without writing anything, `workspace.suspended` reads `true` although the cluster is `Running`. It is the handle's own state, not a claim about the object, and another `resume()` is the way back — `refresh()` reports only a suspension somebody else performed.
110
+
111
+ **An unconfirmed cancellation no longer retires a workspace or sends a patch.** When an execution's cancellation could not be confirmed within the shared controller's eight-second window, the handle retired itself — which on a workspace meant the same `Suspended` patch. Eight seconds of network loss under one `exec()`, or a pod evicted under an in-flight command, therefore stopped the pod for everyone. Nothing is written now:
112
+
113
+ - `exec()` still rejects with `RemoteCancellationUnknownError`;
114
+ - it carries `retirement: { accepted: false, reason: 'workspace-kept' }` instead of `{ accepted: true }`. `reason` is a new optional field on `SandboxRetirementObservation`; `error` is absent, because nothing was attempted. **Code that reads `retirement.accepted` to mean "the pod was stopped" must read `reason` as well**;
115
+ - the handle is not retired — `suspended` stays `false` and the next call is admitted;
116
+ - one bounded `healthz` over a fresh connection reports through the new `onCancellationUnconfirmed({ error, agent })` on `KubernetesWorkspaceOptions`, where `agent` is `'ok'`, `'retiring'` (the guest fenced itself and only a new pod clears it) or `'unreachable'`.
117
+
118
+ _To restore the old behaviour, call `suspend()` from `onCancellationUnconfirmed`._ Calling it only when `agent === 'retiring'` restores it for the case it was actually diagnosing.
119
+
120
+ **A fenced agent is named.** A reservation refused with `agent_retiring` on this backend now rejects with the new `KubernetesAgentRetiringError` instead of `RemoteProtocolError: remote sandbox returned an invalid execution reservation`. Unlike the Firecracker tier's mapping of the same refusal it does not retire the handle: that mapping is `RemoteCancellationUnknownError`, which would take the workspace's pod away. Not retiring the handle is a statement about _this side_ — nothing was patched and the workspace is still `Running`; the guest's fence gates every op but `healthz` and `cancel-execution`, so reads, writes, terminals and tcp connections meet it too until the pod is replaced, and the message names the verbs that replace it on each tier (`suspend()` then `resume()` on a workspace, `destroy()` and a new sandbox on the task path).
121
+
122
+ Nothing here adds a `DELETE`. `destroy({ deleteDisk: true })` and `deleteKubernetesWorkspace` remain the only paths that remove a disk.
123
+
124
+ Also new on the public surface: `KubernetesWorkspaceStartFailurePolicy`, `KubernetesWorkspaceAgentState`, `KubernetesWorkspaceCancellationNotice`.
125
+
126
+ - 1d46ef4: A Kubernetes workspace call that used to fail with `unauthorized` after its pod was replaced now succeeds — against a **different guest**, where every process the caller started is gone. That is the changed default, and a host that keeps per-workspace state must subscribe to the new `KubernetesWorkspace.onGuestRestart(listener)` rather than assume continuity.
127
+
128
+ Why the handle rebinds at all: the agent's bind token is the bound pod's `metadata.uid`, read once per session, and under the default `agentAddress: 'service'` the Service FQDN outlives the pod. So after an eviction, a node drain, or a `suspend()`/`resume()` another process performed, the dial kept succeeding against the replacement pod and its agent refused every call. Nothing recovered from that, and the refusal arrived in two unrelated shapes: `KubernetesAgentUnauthorizedError` for `exec` and `listFiles`, a bare `Error('unauthorized')` for `writeFile`, `readFile`, `openTerminal` and `openTcpConnection`.
129
+
130
+ **What changed.** A refused call re-reads the `Sandbox` once and, when it finds the same `sandboxUid` and a live pod with a different uid, takes the new address and token and retries the call **once** — safe because the guest checks the token before dispatch, so a refused request ran nothing at all. The same pod uid leaves the original error standing. A **different** `sandboxUid`, or no `Sandbox` at all, now raises the new `KubernetesWorkspaceReplacedError` and never rebinds: the workspace name is deterministic, so that is a different object with an empty disk. Every operation that cannot be rebound now rejects with `KubernetesAgentUnauthorizedError`, so the bare-`Error` shapes are gone — code matching `error.message === 'unauthorized'` on those four calls must catch the class instead.
131
+
132
+ **What is additive.** `KubernetesWorkspace.identity` (`{ sandboxUid, volumeClaimUids, podUid, guestBootId }`); `onGuestRestart`, which fires `pod-replaced` and `container-restarted` events and does not fire across the caller's own `suspend()`/`resume()` (the one exception being a pod somebody else replaced during that resume, where both halves of the payload still name their pod — `previous` the one the resume had bound, `current` the one it rebound to); `KubernetesWorkspaceReplacedError` and `KubernetesWorkspaceGuestGoneError`; and three fields on the `onCancellationUnconfirmed` notice (`guest`, `previous`, `current`). An unconfirmed cancellation whose guest is demonstrably gone now rejects with `KubernetesWorkspaceGuestGoneError`, which **extends** `RemoteCancellationUnknownError` — every existing `instanceof` and every read of `retirement` keeps working, and the rule is unchanged: the outcome is unknown and the command must not be retried automatically. The diagnosis is per command, and so is the identity it carries: `previous` is the guest the COMMAND reserved on — the pod its `reserve-execution` was accepted by and the agent process inside it — so neither a restart the handle survived earlier nor a pod another call has already rebound to is blamed on it, and `current` names an agent process only when that process belongs to the pod it names. A workspace somebody else suspended still rejects with `KubernetesWorkspaceSuspendedError`, not this class, so the handle adopts the suspension and `resume()` works. Nothing on this path patches the cluster, exactly as before.
133
+
134
+ **Guest and cluster.** The agent stamps an optional `guestBootId` on `reserve-execution`, `cancel-execution` (including `unknown_execution`), `read-file`, `write-file`, and the terminal and TCP `ready` frames, advertised as `guest-boot-id` in `healthz` features. The wire protocol version is unchanged, so an older guest image reports nothing and a newer host keeps working against it — it simply loses the container-restart signal. The shipped `k8s/manifests/rbac.yaml` gains `get` on `persistentvolumeclaims`, used only to report `volumeClaimUids`; re-applying it is optional, since a Role without it leaves that map empty and changes nothing else.
135
+
136
+ **To keep the old behaviour** — a refused call staying refused rather than rebinding — there is no flag: subscribe to `onGuestRestart` and drop your handle from the listener when `reason` is `pod-replaced`.
137
+
138
+ - b3db254: **Closing a terminal now kills everything the terminal started.** That is the
139
+ one change here you did not ask for, and it is why this is a major. Before it,
140
+ tearing a terminal down sent `SIGKILL` to the process group of util-linux
141
+ `script` — and `script` starts the shell in a **new session**, so the kill
142
+ reached `script` alone. `script`, the shell and the foreground job then died of
143
+ the PTY hanging up, but a job backgrounded with `&` was never signalled: it
144
+ kept running with no terminal, holding its port, reachable by no op, until the
145
+ pod stopped. The agent's own comment claimed that kill reached "the shell and
146
+ every descendant". It now does: both a session kill and a plain terminal's
147
+ teardown signal every process still in the kernel session the shell was
148
+ started in, found through `/proc`. **If you were relying on that leak** —
149
+ starting a dev server with `&` inside a terminal and expecting it to survive
150
+ the terminal — move it to `startDetached` below, which is the verb for a
151
+ program meant to outlive its caller. Nothing else about a plain `openTerminal`
152
+ changed, down to the wire request it sends.
153
+
154
+ **What is new: a workspace terminal or program can outlive the host process.**
155
+ A workspace is built to outlive the host — it carries no lease for exactly
156
+ that reason — but the processes inside it were not. A terminal belonged to one
157
+ connection, so a deploy, a crash or an OOM kill tore down every terminal the
158
+ host had open. Replay was buffered in the host process, so its successor had
159
+ neither the output nor a way to name the terminal. And nothing could run
160
+ outside a terminal at all: `exec` caps at thirty minutes and kills the process
161
+ group when the cap fires.
162
+
163
+ All of it is on `KubernetesWorkspace`. `@namzu/sdk` is unchanged, and so is
164
+ every other backend, the Firecracker tier included.
165
+
166
+ - `openTerminal({ ...options, sessionId, persistent: true })` hands the PTY to
167
+ the guest's session registry. Closing the connection then DETACHES and sends
168
+ no signal of any kind; the session ends when its program exits, on
169
+ `killSession`, or when the pod stops.
170
+ - `attachTerminal(sessionId, { fromOffset, size })` rejoins it from any
171
+ process, replaying what it missed and then following live, with input and
172
+ resize working after the attach. At most one attachment exists at a time: a
173
+ second attach ends the first by name, so two host processes cannot
174
+ interleave keystrokes into one shell.
175
+ - `startDetached({ sessionId, command, args, cwd, env })` starts a program with
176
+ no PTY, stdin closed, in its own kernel session. `readSession(sessionId, {
177
+ fromOffset })` answers in the SDK's `BackgroundJobOutput` shape (`chunk`,
178
+ `nextOffset`, `droppedBytes`, `status`, `exitCode`) — and a read is not an
179
+ attachment: it displaces nobody and signals nothing, so polling a shell's
180
+ tail leaves the terminal reading it alone. `listSessions()` names what is
181
+ running, and `killSession(sessionId, { signal })` ends one and everything
182
+ still in it; `signal` is one of `SIGTERM`, `SIGKILL`, `SIGINT` or `SIGHUP`
183
+ and anything else is coerced to `SIGTERM`, the same on every connection.
184
+ - Exported: `KubernetesSessionsUnsupportedError`,
185
+ `KubernetesSessionRefusedError`, `AgentSessionDetachedError`,
186
+ `SESSIONS_FEATURE`, plus the option and row types
187
+ (`KubernetesOpenTerminalOptions`, `KubernetesAttachTerminalOptions`,
188
+ `KubernetesStartDetachedOptions`, `KubernetesReadSessionOptions`,
189
+ `KubernetesKillSessionOptions`, `KubernetesSessionSummary`,
190
+ `KubernetesSessionOutput`, `KubernetesWorkspaceTerminal`,
191
+ `KubernetesSessionTerminal`, `KubernetesSessionRefusal`, `SessionKind`,
192
+ `SessionState`, `SessionDetachReason`).
193
+
194
+ **`exited` on a session terminal can reject.** When the attachment ends and the
195
+ program does not — the connection was lost, or another process took the
196
+ session — it rejects with `AgentSessionDetachedError`, carrying the byte offset
197
+ to come back at. Resolving it would report an exit that never happened, which
198
+ is the confusion this whole feature exists to remove. A connection-bound
199
+ terminal's `exited` is unchanged.
200
+
201
+ **What you have to know before relying on it.** The registry is the pod's
202
+ memory and is never written to disk, so `listSessions()` is empty after
203
+ `suspend()` and `resume()`, and after any eviction, node drain or restart:
204
+ this makes a program survive the HOST, not the pod. A session's output ring is
205
+ the same `OutputLog` a detached execution uses — one monotonic byte-offset
206
+ space, eviction reported as `droppedBytes`, never a shorter stream that looks
207
+ complete — and output is read into it whether or not anybody is attached, so a
208
+ program with no reader never blocks on a full PTY. No signal can follow a
209
+ process that called `setsid` for itself: it has left the session, and nothing
210
+ short of a PID namespace or a cgroup reaches it.
211
+
212
+ **`spawnDetached` is still absent, deliberately.** It returns a host
213
+ `ChildProcess` synchronously and its consumer keeps jobs in a map inside one
214
+ host process, so it cannot express a hand-off between processes. `startDetached`
215
+ has a different name because it does a different thing: it returns a NAME, and
216
+ the name is what a redeployed host comes back with.
217
+
218
+ **Redeploy the workspace image to get it.** The guest advertises `sessions` in
219
+ its `healthz` features, and a host asking for any session verb against an older
220
+ image is refused with `KubernetesSessionsUnsupportedError` — never downgraded
221
+ to a connection-bound terminal. The guest wire protocol version is deliberately
222
+ **unchanged**: `sessionId`, `persistent` and the four new ops
223
+ (`attach-session`, `start-detached`, `list-sessions`, `kill-session`) are all
224
+ additive, so no host and no image has to roll together with this release.
225
+
226
+ Three variables join the shipped workspace template's `env` block at their
227
+ defaults, so deleting them changes nothing: `NAMZU_AGENT_MAX_SESSIONS` (16
228
+ sessions at once), `NAMZU_AGENT_SESSION_LOG_BYTES` (1 MiB of retained output
229
+ each) and `NAMZU_AGENT_SESSION_TERMINAL_TTL_MS` (10 minutes an exited
230
+ session's record and output outlive it). The first two bound what the registry
231
+ can cost the container's 512Mi; a session exists only when a caller names one,
232
+ so a deployment that never asks for one pays nothing.
233
+
234
+ ### Minor Changes
235
+
236
+ - b52609b: Adopting a Kubernetes workspace whose previous pod is still terminating now waits for the replacement instead of failing.
237
+
238
+ `createKubernetesWorkspace` on a name that already exists adopts the standing object, and the process that does so arrives at a moment the previous one did not choose: a `suspend()` that ended in `KubernetesWorkspaceSuspendTimeoutError` (the patch landed; the guest is riding out its `terminationGracePeriodSeconds`), two hosts coming up on one workspace during a rollout, or a host restarting inside that grace period. In each of those the only pod under the Sandbox's name carries a `deletionTimestamp` and is never bound to — its uid is the agent's bind token and the pod's replacement refuses it — while the replacement has not been created yet, so the bind-token read threw and the adopt rethrew it on the spot. It now polls for a live pod under the same `readyTimeoutMs` budget the resume path already polled under, and a budget that runs out names the pod that was still terminating rather than reporting a generic missing uid.
239
+
240
+ An adopt of an object that was Running with a healthy pod is unchanged, and so is the created path: nothing is being replaced there, so a pod that cannot be read is still reported on the first read rather than waited out for the whole budget. Nothing about suspend, resume, deletion or the guest wire protocol changed, and no default moved.
241
+
242
+ New on the public surface, and the reason this is `minor` rather than `patch`: `KubernetesWorkspace.origin`, typed by the new `KubernetesWorkspaceOrigin` union — `'created'`, `'adopted-running'` or `'resumed'`. It is fixed for the handle's life and says what the call walked into, not what state the workspace is in now (`suspended` is for that). A host reattaching to a workspace another process left behind needs it: on `'resumed'` the pod is brand new and only the disk survived, while on `'adopted-running'` the pod is the one the previous holder was using — a detached command may still be running in it, but no terminal is, because the guest agent kills a terminal's process group the moment its connection closes and a dead host's connections closed with it. Code that only reads a `KubernetesWorkspace` needs no change; code that implements the interface must add the field.
243
+
244
+ - a75f670: `config.egress.ciliumNarrowing` lets a `static`/`resolver` egress allowlist under `engine: 'cilium'` limit ports, DNS names and TLS server names instead of allowing an address, any port and any resolvable name. This is additive: with `ciliumNarrowing` unset, the emitted `CiliumNetworkPolicy` is byte-for-byte what it always was, pinned by a deep-equality test, so an already-applied policy keeps verifying after this upgrade.
245
+
246
+ **Why.** Measured on AKS with Cilium 1.18.11: an unnarrowed allowlist's `toFQDNs` rule allows the ADDRESS a name resolved to, and one CDN address can serve many unrelated sites — `curl --resolve example.com:443:<registry.npmjs.org's own address>` returned the wrong site's content. It sets no `toPorts`, so `github.com:22` accepted a connection. And its DNS-visibility rule allows any name to resolve at all (`rules.dns: [{ matchPattern: '*' }]`), so DNS itself is an open channel out. `verifyEgressPolicyApplied` compares the applied object to the translation exactly, so an operator could not tighten any of this on the cluster without every `createKubernetesWorkspace` and provider `create()` failing.
247
+
248
+ **What's new**, all opt-in on `KubernetesEgressConfig.ciliumNarrowing`:
249
+
250
+ - `ports` (a default port list) and `hostPorts` (per-host overrides) — emitted as `toPorts` on each host's own `toFQDNs` rule. Setting either switches the translation from one shared `toFQDNs` rule naming every host to one rule PER host, so ports can differ host by host.
251
+ - `tlsServerNames: true` adds `serverNames: [<host>]` to a host's TLS ports (default `[443]`, override with `tlsPorts`) — SNI enforcement, which needs Cilium's L7 proxy. A host with no port configured is limited to its TLS ports rather than left open, because a `serverNames` rule needs a port to attach to.
252
+ - `dnsNames` (`true` or `{ namespace?, clusterDomain?, searchSuffixes? }`) replaces the DNS-visibility rule's `matchPattern: '*'` with an exact `matchName` per allowed host plus the host under `<namespace>.svc.<clusterDomain>`, `svc.<clusterDomain>`, `<clusterDomain>` (`cluster.local` default) and any configured extra search suffixes — exact names because Cilium's `matchName` does not match across a `.`.
253
+
254
+ Setting any of these under `engine: 'core'`, or with a `deny-all`/`no-network`/`allow-all`/`public-internet` policy, throws the new `KubernetesEgressNarrowingUnsupportedError` synchronously — both from `buildKubernetesBackend` and from `createKubernetesWorkspace`, before any request — because narrowing only means something next to a hostname allowlist Cilium enforces. A port outside `1-65535`, or an empty `clusterDomain`/search suffix, is refused at construction with `KubernetesEgressPolicyConfigError`.
255
+
256
+ Every new field becomes part of what `verifyEgressPolicyApplied` requires — no new comparison logic, it already deep-equals the whole `spec.egress` — so an applied object missing a configured `toPorts`, DNS name or `serverNames` entry throws `KubernetesEgressPolicyMismatchError` naming it. The union check that reads every OTHER policy selecting the pod now tracks ports per allowed hostname too, so a second `CiliumNetworkPolicy` naming an allowed host on a wider port (or with no `toPorts` at all) is caught as `policy-widens-egress` even though the hostname itself is on the allowlist.
257
+
258
+ A `ports`, `hostPorts[host]` or `tlsPorts` list set to an explicit empty array (`ports: []`) is now refused at construction with `KubernetesEgressPolicyConfigError` — it was neither "no restriction" (that's what omitting the field means) nor usable, and would have emitted a `toPorts` shape the API server rejects on apply.
259
+
260
+ **A shipped-manifest change you may need to make, now enforced rather than only documented.** `packages/sandbox/k8s/manifests/networkpolicy.yaml` and both `sandboxtemplate-{task,workspace}.yaml` templates' managed `networkPolicy` grant kube-dns on port 53 at plain L4, with no L7 rule. Cilium's own rule-precedence says an L4-only rule cancels the L7 portion of a similar rule that carries one — so if you turn `ciliumNarrowing.dnsNames` on, that plain rule defeats it: every name resolves again regardless of the narrower allowlist. Each shipped file says, at the rule itself, to delete it when `ciliumNarrowing.dnsNames` is set. Port and TLS-server-name narrowing need no manifest change; the translated `CiliumNetworkPolicy` grants the DNS access those two need on its own.
261
+
262
+ That plain rule reaches the exact same peer and port the narrowed `CiliumNetworkPolicy` allows for DNS, so mere reachability cannot distinguish them. The union check (above) now can: `EgressAllowance` carries the exact DNS names a narrowed translation restricts lookups to, and the check reads a candidate rule's own DNS restriction (or the fact that a plain `NetworkPolicy` has none at all) before deciding it is within bounds. Leave the shipped rule in place with `ciliumNarrowing.dnsNames` on, and — under the default `egress.verify: 'union'` — the next `create()` now refuses with `policy-widens-egress` naming that policy, rather than silently letting the narrowing do nothing. `egress.verify: 'named-object-only'` does not run this check, so a deployment on that setting must still delete the rule by hand.
263
+
264
+ **Unproven here, and the changeset says so rather than implying otherwise:** real Cilium L7 enforcement of any of these three options. The `kind` cluster this repository tests against runs no Cilium data plane, so a "the port is closed" probe there would pass for the wrong reason regardless of whether narrowing works; `k8s/scripts/egress-check.mjs` does not probe the hostname allowlist, narrowed or not, and its README section says so. Confirming enforcement needs a real Cilium cluster, a positive control (an allowed name still resolving and connecting) alongside the negative one, and ideally a `cilium policy trace`/BPF policy dump showing the narrowed rule is the one in force.
265
+
266
+ - f024b6b: The Kubernetes backend can now label its own task-path claims with a
267
+ host-supplied identity, recover a crashed predecessor's claims by that label,
268
+ and read warm-pool headroom before admitting more work. All additive; no
269
+ default changed.
270
+
271
+ A `SandboxClaim`'s own name is client-generated per acquire, so nothing about
272
+ one said which host process created it. A host killed by a deploy, an OOM or
273
+ a lost node left every claim it held running until `claimTtlSeconds` reaped
274
+ it — an hour by default — and its replacement had no way to find, let alone
275
+ release, them sooner. Three new pieces of surface close that gap:
276
+
277
+ - `claimLabels?: Record<string, string>` on the Kubernetes backend config is
278
+ written into every `SandboxClaim`'s `metadata.labels` only — never into
279
+ `additionalPodMetadata`, so a running Sandbox's pod labels and any
280
+ `NetworkPolicy` selecting by them are unaffected.
281
+ - `releaseKubernetesTaskSandboxes(config, { labelSelector, signal })` LISTs
282
+ claims by `labelSelector` and DELETEs each one, returning
283
+ `{ deleted, names }`. `labelSelector` is **required** and refused, before a
284
+ single request goes out, if it is absent or empty: a release that fell back
285
+ to matching every claim would delete a live fleet's work.
286
+ - `readKubernetesTaskCapacity(config, { signal })` is three GETs and no
287
+ writes — the configured `SandboxWarmPool`, the claims collection filtered
288
+ to that pool, and the pods collection counted by `Pending` phase — into
289
+ `{ warmPool: { ready, desired }, activeClaims, pendingPods }`. Requires
290
+ `warmPoolName`; there is no pool to report on for a backend that creates
291
+ every sandbox directly.
292
+
293
+ The shipped `k8s/manifests/rbac.yaml` gains `list` on `sandboxclaims` — the
294
+ verb both new functions need to find claims by label instead of by a name
295
+ they already know — so a deployment that does not re-apply it gets a `403`
296
+ from either function alone, and never from `create()`. `sandboxwarmpools:
297
+ get` is unchanged but is now a documented backend need rather than only a
298
+ diagnostic script's.
299
+
300
+ A host setting no `claimLabels` and calling neither new function sends the
301
+ exact requests it always has.
302
+
303
+ - b07cebc: A second, deliberately narrower Kubernetes `Role` now ships for hosts that acquire only from a warm pool, and the verb list it grants is exported from the package root as data.
304
+
305
+ `packages/sandbox/k8s/manifests/rbac-claimant.yaml` is a `ServiceAccount` + `Role` + `RoleBinding` named `namzu-sandbox-claimant`, written in the same shape as the existing `rbac.yaml`'s. Bind a host that sets `warmPoolName` and never creates a `Sandbox` of its own to this ServiceAccount and it can claim, hold, release and enumerate pooled task sandboxes — nothing else.
306
+
307
+ Why it exists: the existing Role grants `sandboxes: create`, and a namespace that also runs the privileged workspace template has to allow privileged pods, so any identity holding that Role can POST a `Sandbox` with an arbitrary pod spec — privileged, `hostPath`-mounting, or on another `RuntimeClass`. RBAC decides per verb and never per object shape, so the only way to take that capability away from a host that does not need it is to not grant it. The claimant Role grants `create`/`get`/`list`/`patch`/`delete` on `sandboxclaims`, `get` on `sandboxes`, `get` on `sandboxwarmpools`, `get`/`list` on `pods`, and `get`/`list` on `networkpolicies` and `ciliumnetworkpolicies` — and withholds every `sandboxes` write, every write on a policy resource, and the `sandboxtemplates` and `persistentvolumeclaims` reads, each because no call site on the pool-only path issues it.
308
+
309
+ **What reaches a consumer through npm is the constant, not the manifest.** `k8s/` is outside the package's `files` array and has never been published, so the new `Role` is a repository artifact. New on the public surface, and the reason this is `minor` rather than `patch`: `KUBERNETES_CLAIMANT_RBAC_RULES`, exported from `@namzu/sandbox`'s entry point, with its `KubernetesRbacRule` and `KubernetesRbacVerb` types. It is the pool-only path's verb list with a call site named against every entry, and a host that wants to prove its own applied `Role` carries no more than this backend needs can compare against it instead of against a list copied out of a page. Nothing existing changed name, shape or default: no export was removed or narrowed, no config key moved, and a host that creates sandboxes directly keeps using `rbac.yaml` exactly as before.
310
+
311
+ The direct-create `ValidatingAdmissionPolicy` example that would bound what a `sandboxes: create` holder may POST is not in this change — it lands with the per-sandbox capability work. The manifest's header states that neither of the two cluster-level claims that go with it (a claimant host passing the contract suite and the acquire-p50 script, and `kubectl auth can-i create sandboxes` answering `no`) has been measured; the shipped test parses both Role files and compares verb sets, reads the pool-only path's own sources and resolves every literal `client.request('<METHOD>', …)` there to the `(apiGroup, resource, verb)` triple its path builder addresses — failing on any request it cannot trace, including a builder call with anything appended to it (a `pods/log`-style subresource is authorized separately from its parent), a NAME the source writes a second path to, in any of the four spellings the guard reads WHEREVER they are written rather than only where they open a statement (a plain assignment `path = …` — the one-line branch included — a compound one, a `for (path of …)` binding, a destructuring target `({ path } = …)`) — as a `const`/`let`/`var`-declared one, through which the appended spelling reaches a request too, and as a wrapper's own path parameter, through which the call sites that resolve it stop being what it sends; the first used to resolve to the initializer's triple in silence, a request in no function body it can match, a path that is a parameter of a body whose call sites it has not been told about, and a wrapper declared for such a path in a file the scan does not read — pins the file set it reads to the directory — any other `.ts` under `src/backends/kubernetes/` that reaches the client, calling `client.request(…)` itself in either spelling the scan reads (the literal one, or a computed `client['request']`) OR naming `listPolicies`, the one wrapper the scan knows, whose declaration is found in either shape the scan reads — fails until it is scanned or declared off the path with its reason — and contacts no API server. What it cannot see: the files declared off the path (`workspace.ts`, `k8s-client.ts`, `transport.ts`) are never resolved to triples, a wrapper call site in one of them included, since call sites are read from the scanned files only; a wrapper reached by another NAME is outside that file predicate, which matches the wrapper's name and not its identity, so a file importing a re-export of `listPolicies` — or calling a helper one level further out that calls it — is neither scanned nor required to be declared; a path arriving through an expression-bodied arrow's parameter is outside what it matches; a body it could not register — a class method, or an object literal's shorthand method, declared inside one of the bodies it reads — is attributed to the body that encloses it, so the nested body's own parameters are invisible; and the assignment guard it now has is by name in the file it resolves in rather than by scope, so a binding it registers nowhere shadows nothing there — a destructuring parameter (`function f({ path })`) or a write made to an exported binding from another module — and an assignment to a same-named local in an unrelated function of the same file refuses the request too, a false refusal rather than an escape; a parameter carrying a default (`function f(path = …)`, `function f({ path } = {})`) is written with an `=`, so the guard reads it and refuses the request rather than resolving it; and a destructuring pattern nested inside another (`({ a: { path } } = …)`) is read by neither spelling — measured green, and listed as a hole rather than claimed as read. That list is the SHORT form, not the list: it is carried in full in the header of `packages/sandbox/k8s/__tests__/manifests.test.ts`, which also holds the bullet `a path assembled by an operation the scan does not model`. Three shapes it used to pass in silence fail it now rather than being listed here: a declared wrapper written as `const NAME = … =>`, whose call sites resolved to nothing because the declaration was looked up with a `function`-keyword pattern; a file whose only route to the client is the computed `client['request']` spelling, which the file predicate's literal half could not match; and a name a source writes a second path to, in the two shapes it can reach a request through — a declared one, whose request used to be resolved from the initializer, and a wrapper's own path parameter, whose request used to be resolved from its call sites. Both are the appended-subresource escape the argument spelling already failed on.
312
+
313
+ - 7488339: The Kubernetes backend can now select an egress PROFILE per claim, so one
314
+ `SandboxWarmPool` serves several enforced network modes. **It needs an
315
+ operator action on the cluster before it works:** a profile is a pod label,
316
+ and the agent-sandbox controller refuses a claim whose label key is outside
317
+ the `allowed-label-domains` key of the `agent-sandbox-config` ConfigMap in
318
+ the controller's own namespace (built-in default `sandbox.users.io`), while
319
+ this backend's default key is `sandbox.namzu.ai/egress-profile`. Add the
320
+ domain there, or set `egress.profileLabelKey` to a key already allowed.
321
+ Everything is additive: with no `egress.profile` set, every emitted body,
322
+ selector, policy name and request is byte for byte what it was.
323
+
324
+ Egress was one policy per backend: the translated policy's selector is the
325
+ template label alone, and a per-`create()` override is refused. Every network
326
+ mode therefore needed its own `SandboxTemplate`, its own `SandboxWarmPool`
327
+ and its own policy, and every warm replica is a full pod reservation
328
+ multiplied by the number of modes.
329
+
330
+ `KubernetesEgressConfig` gains two fields:
331
+
332
+ - `profile?: string` — a DNS-1123 label value, e.g. `none` or `internet`.
333
+ Set, it is written onto the `SandboxClaim`'s
334
+ `spec.additionalPodMetadata.labels`, onto a directly created `Sandbox`'s
335
+ (and a workspace's) pod template, and into the translated policy's
336
+ `podSelector`/`endpointSelector`. The default policy name becomes
337
+ `${sandboxTemplateName}-${profile}-egress`.
338
+ - `profileLabelKey?: string` — the key it is written under. Defaults to
339
+ `DEFAULT_EGRESS_PROFILE_LABEL_KEY` (`sandbox.namzu.ai/egress-profile`),
340
+ now exported.
341
+
342
+ One thing to plan for when adopting it: **each profile needs its own applied
343
+ policy object.** A deployment that sets `profile` while leaving the policy an
344
+ operator applied selecting the template label alone gets
345
+ `KubernetesEgressPolicyMismatchError` on the first `create()` — by design,
346
+ because the selector is part of the exact-match verification, and a profile
347
+ whose policy does not select it would bound nothing.
348
+
349
+ Two new refusals are thrown, each catchable by class:
350
+ `KubernetesEgressProfileConfigError` (synchronous, while the host is being
351
+ wired, for a profile or key this backend will not emit — including
352
+ `sandbox.namzu.ai/template`, which would overwrite the template label, and a
353
+ `${template}-${profile}-egress` past the 253-character object-name limit) and
354
+ `KubernetesPodLabelNotObservedError` (the bound pod never carried the label;
355
+ the claim is released rather than a sandbox handed back, because an
356
+ unlabelled pod would run under whatever policy does select it).
357
+
358
+ **A claim the controller refuses for its metadata does NOT get a taxonomy of
359
+ its own.** `InvalidMetadata` is already one of the terminal claim reasons an
360
+ acquire fails fast on, so it still rejects with `KubernetesAcquireError`
361
+ (`reason: 'claim-rejected'`, `retryable: false`) whether or not a profile is
362
+ configured — an existing `catch` keeps working unchanged. The new
363
+ `KubernetesPodLabelsRejectedError` is exported but never thrown on its own: it
364
+ rides as that error's `cause`, carrying the pod labels this backend sent
365
+ (`requestedPodLabels`) and the `egress.profileLabelKey` that moves the
366
+ offending key to an allowed domain, neither of which the controller's own
367
+ message can know.
368
+
369
+ A third refusal is an existing class gaining a value: a workspace whose
370
+ standing `Sandbox` does not carry the configured profile on its pod template
371
+ is **not adopted**. `KubernetesWorkspaceMismatchError.field` gains
372
+ `'egressProfile'` beside `'sandboxTemplateName'` and `'runtimeClassName'`
373
+ (additive — a host matching on the two existing values is unaffected), and
374
+ the refusal lands before any resume patch, so the workspace is left asleep
375
+ rather than woken up to be rejected. The way to move an existing workspace
376
+ onto a profile is `refreshPodTemplate: true` on `resume()` or
377
+ `createKubernetesWorkspace`, which rewrites `spec.podTemplate` — profile label
378
+ included — on the one Suspended → Running transition; the refusal is lifted
379
+ for a call that is about to write the configured profile, exactly as the
380
+ `runtimeClassName` refusal is, and re-applied if the patch does not land. A
381
+ workspace that is neither refreshed nor recreated refuses to open rather than
382
+ opening under whatever policy still selects it.
383
+
384
+ Measured against agent-sandbox v1.0.2 on Kubernetes v1.37: six claims out of
385
+ one two-replica pool, alternating two profile values, each bound a replica
386
+ that already existed, in 47–61 ms, with the controller patching the label
387
+ onto the running pod and into the `Sandbox`'s own `spec.podTemplate`. What
388
+ was NOT measured anywhere in this repo is the egress difference between two
389
+ profiles — that is enforcement, and it needs a cluster that enforces.
390
+
391
+ - 260547f: A Kubernetes workspace command can now keep running when the connection
392
+ watching it is lost, and be picked up again by execution id — from this host
393
+ process or another one.
394
+
395
+ Before this, a command's output and result were tied to the one connection
396
+ that started it. Any socket reset made the host cancel the command to
397
+ reconcile, and a cancel it could not confirm within eight seconds retired the
398
+ pod, taking every open terminal with it. If the host process exited instead,
399
+ nothing cancelled at all: the command ran on, its output had been written only
400
+ to a closed socket and kept nowhere, and the only op that returned its record
401
+ terminated a running command to hand it over. A host that ran installs, builds
402
+ or test suites from a redeployable process had to hold every rollout until
403
+ nothing was in flight.
404
+
405
+ **Nothing changes unless you ask for it.** Without `executionId` and without
406
+ `detach`, `exec()` sends byte-for-byte the wire request it always sent, keeps
407
+ no output, and behaves exactly as it did. `SandboxExecOptions` is untouched, so
408
+ the SDK's exec contract and every other backend — the Firecracker tier
409
+ included — are unaffected, and `@namzu/sdk` is unchanged.
410
+
411
+ **What is new, all on the Kubernetes workspace handle.**
412
+
413
+ - `exec(command, argv, { executionId, detach, detachSignal, reattachWindowMs,
414
+ onGap })`. When the exec connection fails, the handle reattaches from the
415
+ last byte offset it received. Only getting back is bounded
416
+ (`reattachWindowMs`, 30 s by default), and the bound is disarmed once an
417
+ attach succeeds. If it cannot get back, `exec()` rejects with
418
+ `KubernetesExecutionDetachedError`, carrying the `executionId` and the
419
+ `outputOffset` to resume from. **It never sends a cancel to reconcile**, so a
420
+ lost connection now costs the workspace nothing — no patch, no suspend, no
421
+ pod. `signal` keeps its SDK contract and terminates; `detachSignal` ends the
422
+ watching and leaves the command running.
423
+ - `attachExecution(executionId, { fromOffset, onOutput, onGap, signal })`,
424
+ resolving to the same `SandboxExecResult`. It never signals the command:
425
+ aborting its signal detaches. Every attach inside the retention window
426
+ returns the same result. A refusal arrives as
427
+ `KubernetesExecutionNotAttachableError`, whose `executionState` — when the
428
+ guest reported one — tells the two answers apart that matter:
429
+ `'reserved'` means the command never started, so nothing is running.
430
+ - `cancelExecution(executionId)`, the confirmed-cancel path from any process
431
+ holding the id. A cancellation it could not confirm rejects and retires
432
+ nothing.
433
+ - Exported: `KubernetesExecutionDetachedError`,
434
+ `KubernetesExecutionNotAttachableError`,
435
+ `KubernetesExecutionAttachUnsupportedError`,
436
+ `KubernetesDetachedExecOptions`, `KubernetesAttachExecutionOptions`,
437
+ `KubernetesAttachRefusal`.
438
+
439
+ **What you have to know before relying on it.** Starting the same
440
+ `executionId` twice runs the command once — but only while the guest still
441
+ holds the record. That ends at the retention window
442
+ (`NAMZU_AGENT_EXECUTION_RETAINED_TTL_MS`, 10 minutes) and at pod replacement,
443
+ which takes the whole registry with it; after either, the same id starts a
444
+ fresh command. A reader that asks for output the guest has already evicted is
445
+ told how many bytes it lost through `onGap`, and the result carries
446
+ `stdoutTruncated` and `stderrTruncated` — both, because the retained log is one
447
+ interleaved space.
448
+
449
+ **Redeploy the workspace image to get it.** The guest advertises
450
+ `execution-attach` in its `healthz` features and a host asking for a detachable
451
+ command against an older image is refused before the command is admitted. The
452
+ guest wire protocol version is deliberately **unchanged**, so no host and no
453
+ image has to roll together with this release.
454
+
455
+ **`NAMZU_SANDBOX_MAX_TIMEOUT_MS`.** The guest's ceiling on a caller's `timeout`
456
+ was a hard 30-minute constant; it is now read from this variable — the same one
457
+ the container worker has always read for the same limit — and named in the
458
+ refusal. The default is still 30 minutes, so an unconfigured deployment refuses
459
+ exactly what it always refused. Set it to run a suite longer than that without
460
+ rebuilding the image.
461
+
462
+ Four variables join the shipped workspace template's `env` block at their
463
+ defaults, so deleting them changes nothing: `NAMZU_SANDBOX_MAX_TIMEOUT_MS`,
464
+ `NAMZU_AGENT_EXECUTION_LOG_BYTES` (1 MiB of retained output per execution),
465
+ `NAMZU_AGENT_MAX_RETAINED_OUTPUT_LOGS` (32 executions retaining at once) and
466
+ `NAMZU_AGENT_EXECUTION_RETAINED_TTL_MS`. The last three bound what retention
467
+ can cost the container's 512Mi; only a command that asked for it retains
468
+ anything.
469
+
470
+ - 5a2c25a: Kubernetes sandboxes can now be dialed at their pod IP, so a host outside the cluster can reach the guest agent at all.
471
+
472
+ Before, every sandbox and workspace was addressed as `<name>.<namespace>.svc.cluster.local`, which only the cluster's own DNS resolves. A host running outside the cluster — a peered VNet, an operator on a node, a CI runner with a route to the pod network — failed every call at name resolution, readiness and the acquire-time privilege probe included, so `create()` rejected on its readiness budget describing a timeout while the cluster was fine.
473
+
474
+ `KubernetesBackendConfig.agentAddress` takes `'service'` or `'pod-ip'` and **defaults to `'service'`, which is exactly the previous behaviour** — same address, same API requests, same dial. Nothing changes for a host running inside the cluster, and no existing configuration needs to be touched.
475
+
476
+ `'pod-ip'` dials the bound pod's IP, read from the same `GET` that reads the pod's uid, so the address and the bind token always describe one pod. It needs a pod network routable from the host and a `NetworkPolicy` admitting the host's address range on `agentPort`; the bind token, the privilege probe and egress verification are unchanged. A pod is `Pending` and has no IP until the CNI attaches it, so in this mode an acquire and a resume both WAIT for the address on the readiness budget they already own, and refuse only when it never arrives. Because a pod IP dies with its pod, every `resume()` re-reads it, and a dial that fails at connect re-reads the live pod once: a new uid means the controller replaced the pod, so the handle follows the new address and token and retries — safe because the failure came from the dial, so nothing reached the guest — while an unchanged pod leaves the original error standing. That covers every operation that dials the guest — `exec` and `listFiles`, `readFile`, `writeFile` including a body written in parts, `openTerminal`, `openTcpConnection` — with one exception: a command whose cancellation the guest could not confirm is never retried, because its outcome is unknown. `exec` and `listFiles` get there by watching what their dials did rather than by reading their error: the shared execution controller bounds a control request at 2 s and a dial's connect timer is 5 s, so a released pod IP that drops the SYN — the failure shape this mode is most likely to meet — is aborted by the bound before it has failed, leaving the caller a bare `… reservation exceeded 2000ms` with the dial's own failure discarded. "A connect was attempted and none handed back a socket" is the same "nothing reached the guest" guarantee, read from the other end.
477
+
478
+ A dial of a Service FQDN that fails with `ENOTFOUND`/`EAI_AGAIN` now throws the new, exported `KubernetesAgentAddressUnresolvableError`, naming the FQDN and pointing at `agentAddress: 'pod-ip'`, instead of surfacing as an unexplained readiness timeout. An `ENOTFOUND` also stops retrying at once rather than spending its 30-second connect budget re-asking a resolver that has already answered definitively: that budget outlasts the privilege probe's own deadline, which is how the diagnosis used to be lost. An `EAI_AGAIN` — "temporary failure in name resolution", the shape a CoreDNS restart produces for an in-cluster host — keeps the whole budget and is named this way only if the budget runs out. `VsockTransportOptions` gains two optional hooks this is built on — the `permanentDialFailure` predicate, and `onDialAttempt`, which fires once per connect attempt before it is made so a caller can tell "no socket was ever established" from "the attempt had not failed yet". Absent, as both are everywhere else, every dial behaves exactly as before. `AgentDialFailedError` is now exported too: `VsockAgentTransport`'s dial throws it when it gives up without a socket, which is how "nothing reached the guest" is established by type rather than by reading an error's text.
479
+
480
+ Minor, not major: one optional config field, two optional transport options and two new exported error classes, all additive, with the default behaviour byte-for-byte what it was. The guest wire protocol is untouched, so no pod image needs rebuilding.
481
+
482
+ - b14e068: `walkFiles` is now implemented on Kubernetes task sandboxes and on
483
+ `KubernetesWorkspace`, and the shared sandbox conformance suite gains three
484
+ sections.
485
+
486
+ The SDK's `glob` and `grep` builtins refuse any sandbox that omits
487
+ `walkFiles`, and both are in the default builtin set — so a host that
488
+ registered them and moved from the Firecracker or docker backend to Kubernetes
489
+ lost both tools with no change on its own side. It no longer does. Nothing
490
+ about the guest wire protocol changed and no new agent op was added: this is
491
+ the SDK's own `walkFilesViaExec` running the SDK's walk program through the
492
+ existing `execute`, the same way the Firecracker and docker backends have
493
+ always done it, so all three answer a search identically for one tree. The
494
+ shipped guest image needed no change either (`node:22-bookworm-slim`; the
495
+ agent is itself node).
496
+
497
+ The walk is an execution. It counts as busy for its whole duration rather than
498
+ per entry, `options.signal` and breaking out of the iterator both terminate the
499
+ guest's walk process, and a cancellation the guest cannot confirm retires the
500
+ sandbox exactly as a failed `exec` cancel does. On a workspace it passes the
501
+ same admission gate as every other data-plane call, so a suspended workspace
502
+ refuses a walk with `KubernetesWorkspaceSuspendedError` naming `walkFiles`,
503
+ exactly as it refuses `readFile`, without dialing anything.
504
+
505
+ `SANDBOX_CONTRACT_VERSION` moves from `2` to `3`: `defineSandboxConformance`
506
+ now also covers `walkFiles` (`maxEntries`, `maxDepth`, `includeHidden`, a
507
+ missing root, symlinks not followed, `ERR_FILE_WALK_LIMIT` on an exhausted
508
+ budget, and a refusal once destroyed), several `exec` calls at once on one
509
+ sandbox with no cross-talk between their results, and an `exec` whose `timeout`
510
+ produces a timed-out result and really terminates the command. A backend author
511
+ running the suite should expect the new cases; `walkFiles` is optional on
512
+ `Sandbox`, so a backend that omits it skips that section rather than failing
513
+ it, and no new case needs a guest feature that did not already exist.
514
+
515
+ - 7bf9425: Kubernetes workspaces can be listed, suspended and deleted without waking them, and a handle can see a suspend another process performed.
516
+
517
+ `createKubernetesWorkspace` adopts AND resumes, and until now it was the only way in. So there was no path for the two things that are about the OBJECT rather than the guest: retention — deleting a month-old suspended workspace meant starting its pod and probing it purely to tell it to go away — and hand-off, where a second process could not discover a holder-less running workspace, and a handle opened before another process suspended the workspace went on reporting `suspended: false` over a pod that was gone.
518
+
519
+ Three new verbs reach a workspace without opening one, and none of them creates a pod, dials an agent or resumes anything. `listKubernetesWorkspaces(config, { signal })` issues one GET of the sandboxes collection and returns a `KubernetesWorkspaceSummary` per workspace — `workspaceId`, `operatingMode`, `template`, `createdAt`, `operatingModeChangedAt` — sending no PATCH and no DELETE. `deleteKubernetesWorkspace(config, workspaceId, { signal })` DELETEs by the deterministic name with the same guarantees `destroy({ deleteDisk: true })` gives: an object already gone counts as deleted, and a DELETE that fails rejects and stays retryable. `suspendKubernetesWorkspace(config, workspaceId, { signal })` sends the suspend patch and waits for the pod to actually stop, rejecting with `KubernetesWorkspaceSuspendTimeoutError` if it outlives `readyTimeoutMs`.
520
+
521
+ `operatingModeChangedAt` comes from a new annotation, `sandbox.namzu.ai/operating-mode-changed-at`, that the suspend and resume patches now stamp. Nothing already on the object answers the question — upstream leaves the `Suspended` condition True across a resume, so neither it nor its `lastTransitionTime` says when the mode last changed. A workspace whose mode has never changed carries no annotation and is reported without one rather than defaulted to `createdAt`, so a retention rule can tell "never suspended" from "suspended a month ago".
522
+
523
+ On the handle: `KubernetesWorkspace.refresh()` re-reads `spec.operatingMode`, so a workspace another process suspended reports `suspended: true` and `resume()` brings it back on the same disk. A call that FAILS while the handle still believes it is running now also re-reads the object once, and if it is suspended the caller gets `KubernetesWorkspaceSuspendedError` — carrying the transport failure on `cause` — instead of the flat `unauthorized` or connect refusal that named nothing. That error's new `noticedBy` field (`'admission' | 'transport'`, typed by `KubernetesWorkspaceSuspensionNotice`) says which of the two happened, and its message changes with it, because "nothing was dialed" is a promise only the first keeps. A foreign suspend is recorded as UNCONFIRMED: what was observed is the object's mode, not the pod stopping, so this handle's own `suspend()` still patches and waits.
524
+
525
+ **Two things to do on upgrade.** The host's Role needs `list` on `sandboxes` in the sandbox namespace — `packages/sandbox/k8s/manifests/rbac.yaml` grants it, and `listKubernetesWorkspaces` is the only caller; without it that one verb gets a 403 naming it and nothing else changes. And code that IMPLEMENTS `KubernetesWorkspace` rather than merely consuming one must add `refresh()`. Nothing else moved: no default changed, the guest wire protocol is untouched, and the only difference on the wire is the annotation the two existing patches now carry.
526
+
527
+ - e2024bc: `Sandbox.setNetworkPolicy` is now implemented on Kubernetes sandboxes — but
528
+ only on a TASK handle, only when the backend is configured with
529
+ `egress.perSandbox`, and only on a cluster where **both operator
530
+ prerequisites** are in place: the applied
531
+ `ValidatingAdmissionPolicy` and its binding
532
+ (`packages/sandbox/k8s/manifests/validatingadmissionpolicy-cilium.yaml`),
533
+ which the backend proves before its first write, and the opt-in RBAC file
534
+ (`rbac-per-sandbox-egress.yaml`), whose policy-write verbs the default
535
+ `rbac.yaml` Role deliberately still withholds. A host enabling the option
536
+ without the fence gets `KubernetesAdmissionFenceMissingError` and nothing is
537
+ written; without the RBAC it gets a `403` naming the verb. Everything is
538
+ additive: with no `egress.perSandbox`, the method is absent exactly as before
539
+ and no new request is issued on any path.
540
+
541
+ A `KubernetesWorkspace` handle never carries the method — nothing in its
542
+ create path composes a per-sandbox pod label or tracks an owner uid for one —
543
+ so `createKubernetesWorkspace` **refuses** a config carrying
544
+ `egress.perSandbox`, with `KubernetesWorkspacePerSandboxEgressConfigError`
545
+ and no request sent, rather than accepting a capability it would never serve.
546
+ A host that creates workspaces and task sandboxes from one config object
547
+ passes that call a config without `perSandbox`; task sandboxes are unaffected.
548
+
549
+ Live per-sandbox egress is the SDK's contract for "fetch the repository with
550
+ a token, then narrow before running what the repository contains", and the
551
+ Kubernetes backend omitted it: egress was one policy for every sandbox the
552
+ backend produced, so a per-tenant host list meant an operator-applied policy,
553
+ its own template and its own warm pool, per list.
554
+
555
+ Configured, each `setNetworkPolicy({ allowedHosts })` writes one
556
+ `CiliumNetworkPolicy` for that sandbox alone: named `namzu-sbx-<uid>` after
557
+ the `SandboxClaim` (or `Sandbox`) this backend created, selecting one
558
+ per-sandbox pod label carried in the same claim-time
559
+ `additionalPodMetadata.labels` map the egress profile travels in and confirmed
560
+ on the bound pod before the sandbox is handed back, owned by that object
561
+ through `ownerReferences` so `destroy()` lets the cluster collect it, and
562
+ allowing the cluster-DNS rule plus one `toFQDNs` rule for the list —
563
+ `api.example.com` for a host, `matchName` plus `matchPattern: '*.example.com'`
564
+ for a `.example.com` entry, carrying the same port/DNS-name/TLS-server-name
565
+ narrowing options the config-level allowlist has. `dnsNames` narrowing goes
566
+ one step further here than at the config level, because this translation is
567
+ the one that expands the domain form: an expanded entry also gets a
568
+ `matchPattern: '*.<host>.<suffix>'` for every cluster search suffix, since a
569
+ guest resolving `a.example.com` tries those suffixes first under the default
570
+ `ndots: 5` and a lookup the DNS proxy refuses can fail the whole resolution.
571
+ Both halves of that are new — the exact-host branch emits `<host>.<suffix>`
572
+ `matchName`s and never a pattern — and it is reachable only under
573
+ `perSandbox.narrowing`, where the entry really does admit subdomains. The call
574
+ resolves only after a read-back deep-equals what it sent, through the
575
+ comparator the named-object check already used. `setNetworkPolicy([])` deletes
576
+ that one object, leaving the configured baseline in force rather than no
577
+ policy at all.
578
+
579
+ Policies UNION, so a per-sandbox list ADDS to whatever `egress.policy`
580
+ translated to and only narrows when that baseline denies: pair `perSandbox`
581
+ with `no-network` or `deny-all` if `setNetworkPolicy` is to be the boundary
582
+ rather than an addition to one. Allowlist entries are hostnames, and letter
583
+ case is canonicalised rather than refused; a URL, a port suffix, an explicit
584
+ glob, an IP address and a whole public suffix such as `.com` are each refused
585
+ by name before anything is sent, the last two more strictly than the docker
586
+ backend's proxy.
587
+
588
+ One message correction rides along: `KubernetesEgressPolicyConfigError` now
589
+ spells its `field` relative to `config.egress` rather than to
590
+ `config.egress.policy`, because the fields that reach it live at both levels
591
+ — `policy.exceptCidrs` on the policy, `ciliumNarrowing` and
592
+ `perSandbox.narrowing` beside it — and the old prefix named a key that does
593
+ not exist for two of the three. A caller matching that error's `field` on the
594
+ literal `'exceptCidrs'` should match `'policy.exceptCidrs'` instead; nothing
595
+ about which values are refused has changed.
596
+
597
+ One combination is refused rather than translated: **a `.domain` allowlist
598
+ entry together with `tlsServerNames`**, with `KubernetesNetworkPolicyHostError`
599
+ and nothing written. A TLS server name is one exact SNI value a handshake
600
+ presents, while `.example.com` means that domain and every subdomain of it, so
601
+ `serverNames: ['.example.com']` is a value no handshake ever presents and
602
+ `['example.com']` would deny every subdomain the same rule's `toFQDNs` half
603
+ admits — the object is admitted by the fence and reads back deep-equal to what
604
+ was sent, so the call would report success and deny the domain it was asked to
605
+ allow. Refused at the translation, which covers `config.egress.ciliumNarrowing`
606
+ on the config-level allowlist as well as `perSandbox.narrowing`, and again —
607
+ earlier, before the fence is read — by the per-sandbox writer.
608
+
609
+ **The refusal's message is path-aware**, because the entry is a different
610
+ thing on each path and one sentence cannot be true of both. The per-sandbox
611
+ translation expands `.example.com` into a name plus a `*.example.com` pattern,
612
+ so its message says exactly that, and the remedy it offers — "list the exact
613
+ hosts, or leave `tlsServerNames` off for a domain list" — is a real repair
614
+ there. The CONFIG-level translation expands nothing: the entry reaches the
615
+ object as the literal `matchName: '.example.com'`, which no DNS answer carries,
616
+ so it admits nothing with the option on or off, and `serverNames` on top of it
617
+ is a second, independent denial. Its message says that instead of offering a
618
+ remedy that is not one. Which values are refused is unchanged; only what a
619
+ refused caller is told to do about it is. Each message names the caller's OWN
620
+ field — `config.egress.perSandbox.narrowing` for the expanding one,
621
+ `config.egress.ciliumNarrowing` for the config-level one — and the
622
+ config-level one omits the closing sentence saying what an entry means, since
623
+ that is the grammar its translation does not apply. (That a config-level
624
+ `.domain` entry matches no answer at all is a pre-existing defect of this
625
+ shipped translation: it is left exactly as it was and deferred to its own
626
+ change.)
627
+
628
+ A fence read the API server REFUSES (a `401`/`403` — most often the namespaced
629
+ `Role` applied without the `ClusterRole` in the same file, since both admission
630
+ objects are cluster-scoped) is now its own refusal,
631
+ `KubernetesAdmissionFenceUnreadableError`, rather than being reported as a
632
+ missing fence: `404` means the object is not there, `403` means this host may
633
+ not look, and the two send an operator to different files. Nothing is written
634
+ in either case.
635
+
636
+ New exports: `KubernetesAdmissionFenceUnreadableError`,
637
+ `KubernetesPerSandboxEgressConfig`,
638
+ `KubernetesPerSandboxEgressConfigError`, `KubernetesAdmissionFenceMissingError`,
639
+ `KubernetesNetworkPolicyHostError`, `KubernetesOwnerUidMissingError`,
640
+ `KubernetesWorkspacePerSandboxEgressConfigError`,
641
+ `DEFAULT_PER_SANDBOX_EGRESS_LABEL_KEY`,
642
+ `PER_SANDBOX_POLICY_NAME_PREFIX`. `egress.perSandbox.engine: 'core'` is
643
+ refused synchronously while the host is being wired, because core
644
+ `NetworkPolicy` has no hostname concept to translate an allowlist into. The
645
+ per-sandbox selector key defaults to `sandbox.namzu.ai/per-sandbox-egress`,
646
+ which carries the same controller `allowed-label-domains` prerequisite the
647
+ egress profile's key does, and the shipped admission policy pins that KEY
648
+ alongside the owner's name as the value — the value check alone would let a
649
+ claim named `namzu-task` select on the shared template label.
650
+
651
+ **Enforcement was not measured in this repository.** What was measured, on a
652
+ local single-node cluster running the upstream `CiliumNetworkPolicy` CRD with
653
+ no data plane at all: that the exact body written is admitted by the shipped
654
+ fence and its replacement merge-patch is too; that the fence refuses a name
655
+ without the prefix, a name whose suffix is not its owner's uid, a missing or
656
+ foreign owner, `blockOwnerDeletion: true`, a two-label selector, a one-label
657
+ selector pointed at any pod but the owner's own, a `toFQDNs` entry matching
658
+ every name (`'*'`, `'*.*'`, `'*.com'`), the kube-dns rule moved off port 53,
659
+ `toEntities`, `toCIDR`, an ingress rule, a `specs` list, a widening patch and
660
+ a `DELETE` of the operator's own policy; that 51 concurrent claims produced 51
661
+ distinct policies with no cross-writes; and that deleting a claim collected
662
+ exactly its policy (about 100 ms) while the operator's unowned policy
663
+ survived. That an allowed host answers and a disallowed one does not is a
664
+ property of a CNI data plane and needs a real Cilium cluster with a positive
665
+ control.
666
+
667
+ - feaeaba: Read a large file out of a sandbox without the sandbox holding it.
668
+
669
+ `readFile` used to load the whole file in the guest, base64-encode it into one
670
+ JSON object and write that object as a single frame, so the file buffer, the
671
+ base64 string, the JSON string and two frame buffers all existed at once —
672
+ about 7.7x the file, inside the container the workload shares. A 64 MiB read
673
+ grew the guest agent by 405 MiB against the shipped workspace template's
674
+ `512Mi` limit, and a file of about 384 MiB or more could not be read at all:
675
+ its base64 string exceeds V8's `0x1fffffe8`-character ceiling, so the call
676
+ failed with `Cannot create a string longer than 0x1fffffe8 characters`. That
677
+ call now succeeds.
678
+
679
+ - `readFile(path, { offset, length, signal })` reads one slice. The guest
680
+ `pread`s at that position and answers with the whole file's size beside the
681
+ bytes; a range past the end returns what exists. A slice above
682
+ `NAMZU_AGENT_READ_FILE_RANGE_BYTES` (1 MiB default) is refused, not
683
+ shortened.
684
+ - `readFileStream(path, options?)` returns an `AsyncIterable<Buffer>` over a
685
+ new `read-file-stream` guest op. A 1 GiB read grows the guest by about
686
+ 12 MiB.
687
+ - `readFile(path)` with no options is served by that stream against a guest
688
+ that advertises the capability, so callers lose the ceiling without changing
689
+ a line. It still returns one `Buffer`; iterate `readFileStream` to avoid even
690
+ that copy.
691
+
692
+ The guest opts in. `agent.cjs` advertises `read-file-stream` in its `healthz`
693
+ reply, and against a guest that does not, a whole-file read takes the
694
+ unchanged single-frame path while a ranged read or a `readFileStream` throws
695
+ the new `AgentReadFileStreamUnsupportedError` before dialing — an agent that
696
+ predates the feature ignores `offset`/`length` and answers with the whole
697
+ file, which you would otherwise read as your slice. Rebuild the guest image
698
+ from this release to get the new behaviour; nothing forces you to, and the
699
+ guest wire protocol version is unchanged.
700
+
701
+ A `KubernetesWorkspace` has both shapes too, which is where they matter most:
702
+ draining a large output file before `suspend()` or `destroy()` is what a
703
+ long-lived workspace is for. Its `readFile` forwards `offset`/`length` to the
704
+ guest, and `readFileStream` is present on the interface rather than optional.
705
+ Both refuse a suspended workspace by name, as every other data-plane call does.
706
+
707
+ The two backends that cannot serve a range now REFUSE one rather than ignoring
708
+ it: the docker and standby-pool workers answer whole files only, so
709
+ `readFile(path, { offset })` against either throws instead of handing back the
710
+ file. Previously those backends declared the one-parameter form, which type
711
+ checks and silently discards the range. Both also pass `options.signal` to the
712
+ request they make. Nothing that compiled before breaks — no caller could pass
713
+ the parameter until this release.
714
+
715
+ Three guest rules to know if you write to this wire yourself. A range must ask
716
+ for `base64` (`read_file_range_requires_base64`): a `utf8` slice at an
717
+ arbitrary offset can split a multi-byte character. `read-file-stream` serves
718
+ regular files only (`read_file_stream_not_a_regular_file`), so a whole-file
719
+ read of a fifo or a device node — reachable only if you set
720
+ `NAMZU_SANDBOX_READ_ROOTS` — is now refused rather than attempted, and a file
721
+ that shrinks under the open fd fails the read instead of coming back short.
722
+ A regular file that `stat` reports as zero bytes and that still has content,
723
+ the procfs shape, is read to EOF by both new shapes rather than answered as
724
+ empty.
725
+
726
+ New exports: `AgentReadFileStreamUnsupportedError`, `READ_FILE_STREAM_FEATURE`,
727
+ `ReadFileStreamRequest`, `ReadFileStreamEvent`. New guest environment
728
+ variables: `NAMZU_AGENT_READ_FILE_RANGE_BYTES` (1 MiB),
729
+ `NAMZU_AGENT_READ_FILE_STREAM_CHUNK_BYTES` (256 KiB). `defineSandboxConformance`
730
+ gains `supportsRangedAndStreamedReads`, default `false`, which gates two new
731
+ cases; those two deliberately did not raise `SANDBOX_CONTRACT_VERSION` beyond
732
+ the `3` the `walkFiles`, concurrent-`exec` and `exec`-timeout sections took it
733
+ to, so a backend that passes those three still passes the suite without
734
+ implementing either read shape.
735
+
736
+ - 30da35d: Kubernetes workspaces can be woken onto the current `SandboxTemplate`, keeping their disk.
737
+
738
+ A `Sandbox` carries its own copy of `spec.podTemplate`, taken once in the create `POST`, and agent-sandbox builds every replacement pod from that copy rather than from the template. A workspace kept for weeks therefore ran the pod spec it was created with: a new image tag, a memory limit, a `terminationGracePeriodSeconds` or an env change reached only workspaces created after the edit. After a release that changes the agent's wire protocol this is not cosmetic — every session start runs the privilege probe through an `exec`, the execution controller refuses a reservation whose `protocolVersion` is not the host's, and every adopt and every resume of a workspace pinning the old image fails. Until now the only way out was `destroy({ deleteDisk: true })`, which deletes the PVC.
739
+
740
+ `refreshPodTemplate` is opt-in on both entry points and is honoured on exactly one transition, Suspended → Running:
741
+
742
+ ```ts
743
+ await workspace.resume({ refreshPodTemplate: true });
744
+ // or, for a host that restarted and holds no handle:
745
+ await createKubernetesWorkspace(config, {
746
+ workspaceId,
747
+ workingDirectory,
748
+ refreshPodTemplate: true,
749
+ });
750
+ ```
751
+
752
+ It sends one `application/json-patch+json` body — `test /spec/operatingMode == "Suspended"`, then the annotations, `/spec/podTemplate` and `/spec/operatingMode` — so the condition and the write cannot be separated, and a configured holder epoch's `test` travels in that same body rather than in a second request. A JSON Patch rather than a merge patch on purpose: a merge patch recurses into maps, so a `nodeSelector` entry the template dropped would survive on the object. The two writes go up as `add` rather than `replace`, which is the same write on a member that is already there (RFC 6902 §4.1) and the only one of the two that lands on a member that is not — `spec.operatingMode` is absent on a `Sandbox` that has never been suspended.
753
+
754
+ The disk is the one part of a template a refresh cannot apply: `spec.volumeClaimTemplates` is CEL-immutable, so it is never in the patch and the PVC is untouched. Two refusals guard that, both `KubernetesWorkspaceDiskError`, both before anything is sent, both leaving the workspace suspended — a template that no longer claims this workspace's disk through `volumeDevices`, and a template that declares a disk this workspace does not have. The second is the one to know about: adding a second `volumeClaimTemplates` entry with its matching `volumeDevices` entry is a valid edit that a NEW workspace would honour, and refreshing an existing workspace onto it would write a pod spec claiming a device node backed by no PVC. Give an existing workspace another disk by creating a new one from the new template and migrating the data.
755
+
756
+ Nothing happens to a Running `Sandbox`: v1.0.2 does not rewrite a pod that already exists, so an adopt that finds the workspace Running — and a `test` that loses to another process's resume — binds the pod that is there, unchanged. The `sandboxTemplateName` mismatch refusal still always applies; the `runtimeClassName` one is lifted only for a call whose patch lands, because that patch is what writes the configured class.
757
+
758
+ Two new readonly fields on `KubernetesWorkspace`, `templateRevision` and `templateCurrent`, report whether a workspace is still on the template it was built from. They are REQUIRED members of that exported interface, so anything that implements `KubernetesWorkspace` by hand — a test double, in practice — needs both before it compiles again; nothing that merely consumes a handle is affected, which is why this is a minor rather than a major. They read a new `sandbox.namzu.ai/pod-template-hash` annotation that every workspace create `POST` now stamps; a workspace created before this release carries none and reports `undefined` and `false`, which reads as unknown rather than as current. Task sandboxes are unchanged — their create body gains nothing.
759
+
760
+ One thing is asserted rather than measured, and is stated here so an operator can weigh it rather than discover it: that the sandbox controller builds the replacement pod from `spec.podTemplate` AS REWRITTEN. That is upstream behaviour, taken from its source, and it was not reproduced against a live cluster for this change — not because the experiment is hard, but because the environment this change was written in refuses cluster writes; the run was attempted there and could not be made. It is a small run and it needs neither a disk nor a workspace: create a bare `Sandbox` with any pod template and no `volumeClaimTemplates`, wait for its pod, suspend it, send exactly the patch above with a changed image tag, a changed `terminationGracePeriodSeconds` and one `nodeSelector` key dropped, and read the replacement pod's own `spec` back. Anyone with write access to a cluster running the controller can settle it.
761
+
762
+ The `test` clause bounds the cost of the premise being wrong: a refresh that never reaches the pod is a missed refresh, never a wrong write, and nothing in this release changes what a workspace runs without one. What would be wrong is `templateCurrent`, which would report a workspace as current while its pod ran the old spec — so treat it as a scheduling hint until that run is made, not as a statement about a running process. The disk-preserving round trip through a real block-mode PVC (file digests match, PVC uid unchanged) is unmeasured for the separate reason that it needs block storage; what is proven is that `spec.volumeClaimTemplates` never appears in the patch and that the stored claims are byte-identical afterwards.
763
+
764
+ Without the option nothing changes: `resume()` sends the same single merge patch it always sent, an adopt behaves exactly as it did, both adopt refusals apply as they did, and no new RBAC or guest protocol change is involved.
765
+
766
+ - 7948fc0: Kubernetes workspace lifecycle writes can now be fenced with a holder epoch, so a superseded host process cannot suspend, resume or delete a workspace another process has taken over.
767
+
768
+ A host that drives one workspace from more than one process — a rollout overlap, a restart, a retention job beside a request handler — usually already keeps a monotonic holder epoch of its own. It fenced nothing on the cluster: "check my epoch, then call `suspend()`" is check-then-act, the write that follows is a separate request, and the API server accepted it. A late `suspend()` stopped the new holder's pod, a late `destroy({ deleteDisk: true })` took the disk, and a late adopt woke a workspace that had just been suspended.
769
+
770
+ Pass `epoch` and the condition travels in the same request as the write. It is accepted on `createKubernetesWorkspace`, `workspace.suspend()`, `workspace.resume()`, `workspace.destroy()`, `suspendKubernetesWorkspace` and `deleteKubernetesWorkspace`; a handle keeps the epoch it was opened or last resumed with and writes under it whenever a call passes none, including the cleanup patch a failed start sends. A write carrying epoch `e` applies when the epoch stored on the `Sandbox` is `<= e` and stores `e` in the same request; a stored epoch above `e` rejects with the new `KubernetesWorkspacePreconditionError` (`operation`, `workspaceId`, `sandboxName`, `epoch`, `storedEpoch`) and changes nothing — not on the cluster, and not on the handle, because the refusal is decided before `suspend()` reaps its terminals or `destroy()` tears its session down. `createKubernetesWorkspace` with an epoch now also writes it when it adopts a workspace that is already `Running`, and `resume({ epoch })` writes it on a workspace that is already running; both of those used to send nothing at all, which is exactly what left a new holder invisible to the process it had superseded.
771
+
772
+ **Nothing changes for a caller that passes no epoch**, which is why this is a minor rather than a major: every request is byte for byte what it was, `application/merge-patch+json` included, and an unfenced write is not a write with epoch 0 — it carries no condition and still applies to a workspace held at 7. A workspace with no annotation reads as epoch 0, so existing workspaces accept their first epoch-carrying write.
773
+
774
+ Also new: `KubernetesWorkspaceSummary.holderEpoch`, so a retention pass can see the fences it is looking at without waking anything; `KubernetesPatchNotAppliedError` for a conditional patch the API server would not apply; and a `patchType` argument on the internal API client's `request()`. No RBAC change is needed — the `patch` verb covers every patch type and the shipped `Role` already grants `patch` and `delete` on `sandboxes`.
775
+
776
+ - 3162371: Stop every process in a Kubernetes workspace's guest before a capture, with the new opt-in `quiesce` op and `KubernetesWorkspace.quiesce()`.
777
+
778
+ A `suspend()` promises a quiesced disk, and until now that promise was only kept once the pod had stopped — by which point there is no agent left to read the disk through. What `suspend()` itself reaches is narrower than it looks: the terminals **that handle** returned, plus an execution somebody cancelled by id. A terminal another host process opened, an `exec` already in flight, and above all a program that moved into a session of its own with `setsid` and was then reparented away from the agent all kept running, and kept writing, into the drain. A host taking a final capture could not make it exact, and before deleting a suspended workspace it had to wake it and check.
779
+
780
+ `workspace.quiesce({ graceMs })` stops all of it and **leaves the agent serving**, so the next `exec`, `readFile`, `readFileStream` or `writeFile` reads a filesystem nobody is writing under. It marks every running execution before it signals anything (the mark is what keeps the agent from fencing itself on a group leader that dies first, which would refuse the very capture the quiesce was for), scans `/proc` rather than its own children, skips PID 1, itself and its own kernel session, and signals in rounds — `SIGTERM`, `graceMs`, `SIGKILL` on what is left — until a pass finds nothing. A process still present after `SIGKILL` rejects the call with `KubernetesQuiesceUnconfirmedError` naming its pid; it never resolves optimistically. `suspend({ quiesce: true })` and `destroy({ quiesce: true })` run it after this handle's terminals are reaped and before the `Suspended` patch, and a quiesce that cannot be confirmed sends **no patch**, leaving the workspace running and admitting calls. Concurrent suspends still share one transition, with one exception worth knowing before you wire this to a shutdown path: a `suspend({ quiesce: true })` arriving while a suspend WITHOUT a quiesce is already in flight is rejected with `KubernetesQuiesceUnconfirmedError` (`reason: 'suspend_already_in_flight'`) rather than joining it, because that transition's patch has gone over a guest nothing stopped and no later call can make it still. A caller the flight already satisfies joins it as before.
781
+
782
+ Nothing changes for a caller that does not ask. `suspend()`, `destroy()` and the standalone `suspendKubernetesWorkspace` without the option send exactly the requests they sent before (`suspendKubernetesWorkspace` refuses the option outright, since it never dials the agent and could not honour it), and the guest wire protocol version is unchanged: `quiesce` is an additive op advertised in `healthz` as `quiesce`, so no image and no host has to roll together with this release. An explicit `quiesce()` against an image whose agent predates the op is refused with `KubernetesQuiesceUnsupportedError` rather than answered with an empty list that would read like a guest with nothing to stop; a `suspend({ quiesce: true })` against that image suspends as it always did and tells the new `KubernetesWorkspaceOptions.onQuiesceUnsupported` callback, so the gap is reported rather than hidden.
783
+
784
+ New surface: `KubernetesWorkspace.quiesce`, `KubernetesQuiesceOptions`, `KubernetesQuiesceReport`, `KubernetesWorkspaceQuiesceRequest`, the new `KubernetesWorkspaceSuspendOptions` (`KubernetesWorkspaceTransitionOptions` plus `quiesce`, taken by `suspend()` and by `suspendKubernetesWorkspace`, so that `resume()`, `refresh()`, `listKubernetesWorkspaces` and `deleteKubernetesWorkspace` do not accept a flag they could only ignore), `quiesce` on `KubernetesWorkspaceDestroyOptions`, `onQuiesceUnsupported` and `onQuiesceNarrowed` on `KubernetesWorkspaceOptions`, `KubernetesQuiesceUnsupportedError`, `KubernetesQuiesceUnconfirmedError`, and the wire vocabulary `QUIESCE_FEATURE`, `QuiesceScope` and `QuiescedProcess`. Nothing is added to `@namzu/sdk`'s `Sandbox`.
785
+
786
+ One caveat worth reading before you rely on it: the general scan covers the guest's PID namespace, which in a pod is the container and nothing else, and the agent performs it only when it is the init of that namespace or was started by it — the shape `k8s/entrypoint.sh` gives it, where `tini` is PID 1 and the agent is its child. An agent that is neither narrows itself to the kernel sessions its own registries own and reports `scope: 'owned-sessions'`, which can miss exactly the program this op exists for. The scope is always in the report — and because `suspend({ quiesce: true })` answers `void` and cannot read one, a narrowed scan reaches that caller through `onQuiesceNarrowed` instead.
787
+
788
+ - 5f0222a: `Sandbox.writeFile` now writes a file of any size over the Kubernetes (`tcp`) transport, instead of refusing anything above about 5.9 MiB.
789
+
790
+ Before: every `tcp` request dials a fresh connection and the guest's credential rides inside the request envelope, so every request was that connection's first, not-yet-authenticated frame and was bounded by the guest's pre-auth frame ceiling (`NAMZU_AGENT_MAX_PREAUTH_FRAME_BYTES`, 8 MiB) on _every_ call. A `write-file` carries its whole body base64-encoded in that one frame, so a body above ~5.9 MiB raw threw `AgentPreauthFrameTooLargeError` before dialing. The only workaround was raising the guest's ceiling — trading away the pre-auth budget that bound exists to enforce — which made seeding a repository archive into a workspace impractical.
791
+
792
+ Now: a body that does not fit one frame is split into parts that each do. The parts are appended to a temporary _sibling_ of the target inside the same workspace jail, each naming the byte offset it starts at, and the sequence finishes with an atomic `rename` onto the target. So a reader never observes a half-written file, a failed or cancelled write leaves the target exactly as it was (including not existing), a part that went missing or arrived twice is refused by the guest rather than written in the wrong place, and an abandoned sequence removes its temp file on a best-effort basis. Parts go out sequentially, each on its own connection, so the guest's pre-auth connection pool never holds more than one of this caller's sockets.
793
+
794
+ A body that already fits one frame is unaffected: the same single request, byte for byte.
795
+
796
+ The guest opts in. `agent.cjs` advertises `features: ['write-file-parts']` in its `healthz` reply and the host sends a part only to a guest that did, because an agent that predates the field would read a part's content as a whole file. An oversized body against such a guest still fails with the named `AgentPreauthFrameTooLargeError`, whose message now says the guest is what is missing. The guest wire protocol version is deliberately unchanged (`2`): `part` is an optional field on an existing op, so no host and no guest image has to roll together with this release. `part.discard` — the one verb on `write-file` that removes a file — reaches only the agent's own `.namzu-write-….part` files, a `part` that is present but is not an object is refused rather than served as the plain write it resembles, and a final part's `renameTo` is jailed before any byte is written, so a refused target leaves the temp file untouched. A part whose bytes did not all reach the disk is refused (`write_part_short_write`) rather than renamed: one `pwrite` answers a write that crosses the volume's free space or an `RLIMIT_FSIZE` with a short count and no error, and on the final part a truncated temp file would otherwise be renamed onto the target — destroying the file the rename exists to protect while the caller is told the write failed.
797
+
798
+ New on `VsockTransportOptions`, both optional: `maxWriteFileBytes` (default 1 GiB) is the body size the host refuses outright, with the new `AgentWriteFileTooLargeError` naming it — a bound stated up front rather than an out-of-memory partway through a sequence, and checked before the route is chosen, so a value below what one frame carries caps the ordinary single-frame writes too; `writeFilePartBytes` sets the bytes per part, defaulting to the largest a frame admits. Newly exported from `@namzu/sandbox`: `AgentWriteFileTooLargeError`, `DEFAULT_MAX_WRITE_FILE_BYTES`, `GUEST_FRAME_LIMIT_BYTES`, `WRITE_FILE_PARTS_FEATURE` and the `WriteFilePart` type. `VsockAgentTransport.writeFile` and `KubernetesAgentTransport.writeFile` take an optional trailing `AbortSignal`.
799
+
800
+ Also in this release: the host-side frame reader accumulates into one geometrically-grown buffer instead of re-concatenating every arriving socket chunk onto a fresh allocation. Reading a 64 MiB file back spent 31 seconds of memory copying on a reply the socket delivered in under one; it now takes about 0.7s. Same framing, same errors — only the copying changed.
801
+
802
+ Minor, not major: every existing call keeps its behaviour and its types. The one thing a caller could observe differently is that a `writeFile` above ~5.9 MiB now succeeds where it used to throw, and `SANDBOX_CONTRACT_VERSION` moved from 1 to 2 because `defineSandboxConformance` gained a case for it — a backend run against the suite must now serve a body larger than one wire frame.
803
+
804
+ ### Patch Changes
805
+
806
+ - ddf5a3c: A failed Kubernetes lease renewal now retries on a short capped backoff — one second, doubling, capped at whichever is smaller of thirty seconds or a twentieth of the TTL — instead of waiting a full half-TTL for the next attempt.
807
+
808
+ `KubernetesLeaseRenewal.tick()` used to schedule its next attempt a full jittered half-TTL after every outcome, success or failure alike. A renewal failure landed its retry 0.9–1.1 × TTL after the last success, while the object's `shutdownTime` was exactly one TTL after that same success: a single API blip at renewal time expired a live claim with roughly 50% probability, and the controller deleted the pod out from under whatever command was still running in it. The fix does not change what a success does — the loop still renews every half-TTL and reports nothing — only how quickly it comes back after a failure, so a short outage around a scheduled renewal now gets several attempts inside the window that actually matters instead of one. A renewal that finds the object already gone (404/410) still stops the loop immediately, exactly as before.
809
+
810
+ No public API changed — `LeaseRenewalOptions` gained no new field, and no default a caller configures moved. This is a bug fix to already-documented behaviour (`onLeaseRenewalError`'s doc comment and the [lease renewal](docs/sdk/kubernetes-sandbox.md#the-lease) page both promised the resilience this now actually provides), so it ships as `patch`.
811
+
812
+ - 8d97927: The Kubernetes-backend guest entrypoint (`packages/sandbox/k8s/entrypoint.sh`) no longer treats every `blkid` failure as "the workspace disk is empty."
813
+
814
+ Previously `blkid`'s exit status was folded away with `2>/dev/null || true`, so a `blkid` that was missing from `PATH` (127), not executable (126), erroring (4), or answering ambiguously (8) looked identical to a device that genuinely carries no filesystem — and the entrypoint ran `mkfs.ext4 -F` on it either way. On an image whose `PATH` omits `blkid`'s directory, or that ships a broken `blkid`, that reformats an already-populated, resumed workspace disk instead of refusing to touch it.
815
+
816
+ The entrypoint now keeps `blkid`'s exit status and acts on it: only status 2 ("no filesystem found", `blkid(8)`) may lead to `mkfs`, and only once a raw `dd` read confirms the device can actually be probed (status 2 also covers "blkid could not read the device at all"). Every other outcome — a missing tool, a non-2/non-0 status, or a 0 exit with no printed type — aborts the pod instead, naming the status and leaving `blkid`'s own stderr on the container log rather than discarding it. `blkid`, `dd`, `mkfs.ext4`, `mount`, `chown` and `setpriv` are each checked with `command -v` up front, so a missing tool is named explicitly rather than surfacing as a silent format.
817
+
818
+ No public API changed — this is guest-image/entrypoint behaviour for the `microvm`/`kubernetes` sandbox backend's deployment artifacts, which are not part of the published npm package (`packages/sandbox/k8s/` is excluded from `files`). A host running a workspace template built from the shipped image should rebuild it to pick up the fix.
819
+
820
+ - 29493a3: The shipped Kubernetes sandbox host `Role` (`packages/sandbox/k8s/manifests/rbac.yaml`) now grants `get` on `ciliumnetworkpolicies` (`cilium.io`), matching what `docs/sdk/kubernetes-sandbox.md`'s RBAC section already documented.
821
+
822
+ Before: the Role granted `get` on `networkpolicies` (`networking.k8s.io`) only. A deployment configuring `config.egress.engine: 'cilium'` has `verifyEgressPolicyConfigured` read a `CiliumNetworkPolicy` instead, before every `createKubernetesWorkspace` call and a provider's first `create()` — so the shipped Role 403'd on exactly the path the docs said it covered.
823
+
824
+ A cluster with no Cilium CRDs installed simply never matches the added rule, so this changes nothing for the default `'core'` engine. No code, type or default changed — patch.
825
+
826
+ - Updated dependencies [bd32216]
827
+ - Updated dependencies [c272993]
828
+ - Updated dependencies [c99f088]
829
+ - Updated dependencies [d0227e2]
830
+ - Updated dependencies [359b27f]
831
+ - Updated dependencies [5663108]
832
+ - Updated dependencies [165fd64]
833
+ - Updated dependencies [93f8d1e]
834
+ - Updated dependencies [7694a82]
835
+ - Updated dependencies [6283f8d]
836
+ - Updated dependencies [449642e]
837
+ - Updated dependencies [3e6980d]
838
+ - Updated dependencies [feaeaba]
839
+ - @namzu/sdk@41.0.0
840
+
841
+ ## 14.0.0
842
+
843
+ ### Minor Changes
844
+
845
+ - cd43cfe: `SandboxProviderConfig` gains a real arm for `ACIStandbyPoolBackendConfig` (paired with the `ContainerSandboxLayout` it requires, same as the plain container arm). Previously the exported union only covered `ContainerBackendConfig`, `MicroVMBackendConfig` and `KubernetesBackendConfig`, so `createSandboxProvider({ backend: { tier: 'container', runtime: 'aci-standby-pool', … } })` did not type-check even though the backend was fully implemented and `pickBackend` already dispatched to it internally through two `as unknown as` casts. That call now type-checks with no cast.
846
+
847
+ **Minor, not patch:** this is a backward-compatible widening of an exported input type — every config that type-checked before still does, and the only change is that a config shape the runtime already accepted is now also accepted by the type checker. That is additive public surface (a new union arm a consumer's own type-level code can observe), not an implementation-only correction, so it does not qualify for patch under this repo's rule that patch is reserved for changes that leave the public surface untouched.
848
+
849
+ No runtime behavior changed: the ACI backend's construction, options and defaults are exactly what they were.
850
+
851
+ - f54f6f1: The microVM guest agent can now listen on a TCP port and require a per-instance
852
+ token. Both are opt-in through the environment and both are absent from every
853
+ shipped backend's configuration today, so an existing Firecracker deployment
854
+ behaves exactly as it did: with neither variable set, the agent authenticates
855
+ nothing and listens exactly where it listened before.
856
+
857
+ `NAMZU_AGENT_TCP_PORT` is a third listen mode, after `NAMZU_AGENT_UNIX_PATH`
858
+ and the inherited vsock descriptor and in that order, binding `0.0.0.0` for a
859
+ deployment that reaches the guest over a routed network rather than a
860
+ host-local socket. It fails closed: set with neither token variable it is
861
+ refused at startup, naming both, instead of binding an unauthenticated
862
+ listener. Framing, ops, execution leases, terminals, loopback TCP and
863
+ file IO are unchanged — only the listen address differs. A guest configured
864
+ with none of the three still refuses to start, and the message now names all
865
+ three.
866
+
867
+ `NAMZU_AGENT_BIND_TOKEN` makes every op except `healthz` present that exact
868
+ token in the request envelope, from the first frame of a connection; anything
869
+ else is answered `unauthorized` and the connection is closed before a handler
870
+ runs. Comparison is constant-time over a fixed-width digest, so neither the
871
+ value nor its length is learnable by probing. `NAMZU_AGENT_REQUIRE_TOKEN`
872
+ without a preset token is a fallback that binds to the first token seen and
873
+ refuses every other for the life of the process. `healthz` never requires a
874
+ token and never echoes one, so readiness probing needs no secret. An empty
875
+ `NAMZU_AGENT_BIND_TOKEN` is refused at startup in every mode — that is what a
876
+ downward-API injection looks like when it resolved to nothing.
877
+
878
+ A refused connection is destroyed rather than `end()`ed, so a refused peer can
879
+ no longer keep streaming into the agent's frame buffer over the readable half
880
+ that `end()` leaves open. Alongside it, bounds on what an unauthenticated peer
881
+ may spend before the gate — which cannot run until a whole frame is parsed,
882
+ because the credential rides inside the envelope.
883
+ `NAMZU_AGENT_MAX_FRAME_BYTES` (default 256 MiB) caps the length ANY frame
884
+ header may announce, on every listen mode, where the 8-hex prefix used to allow
885
+ 4 GiB. `NAMZU_AGENT_REFUSAL_FLUSH_GRACE_MS` (default 1s) likewise applies
886
+ everywhere: it is how long a refusal frame may take to reach the wire before
887
+ the socket is destroyed regardless, so a peer that stops reading cannot hold a
888
+ refusal open. `NAMZU_AGENT_MAX_PREAUTH_FRAME_BYTES` (default 8 MiB),
889
+ `NAMZU_AGENT_MAX_PREAUTH_CONNECTIONS` (default 64),
890
+ `NAMZU_AGENT_MAX_PREAUTH_BUFFER_BYTES` (default 32 MiB),
891
+ `NAMZU_AGENT_PREAUTH_IDLE_TIMEOUT_MS` (default 10s) and
892
+ `NAMZU_AGENT_PREAUTH_DEADLINE_MS` (default 10s) apply only in the token
893
+ modes, so the Firecracker path sees none of them; together they bound what an
894
+ unauthenticated peer can make the agent hold to one number rather than to a
895
+ number per connection. One bound needs no variable and applies everywhere: a
896
+ frame header is exactly nine bytes, so a peer streaming bytes with no newline
897
+ in them is refused on the ninth rather than buffered against a newline that
898
+ never arrives. Note what the pre-auth cap
899
+ costs in a token mode: a `write-file` body shares the first frame with the
900
+ token, so that cap is the ceiling on the body — about 6 MiB of file content at
901
+ the default — and a body above it is refused `frame_too_large`, naming the
902
+ limit and the variable to raise.
903
+
904
+ Two of those pre-auth bounds are shaped by a slow loris rather than by a flood,
905
+ and an operator who tunes them should know which is which.
906
+ `NAMZU_AGENT_PREAUTH_IDLE_TIMEOUT_MS` is an idle timer that every byte resets,
907
+ so on its own it retires only a silent connection;
908
+ `NAMZU_AGENT_PREAUTH_DEADLINE_MS` runs from accept, is reset by nothing, and is
909
+ the bound a peer dripping one byte every few seconds actually meets. And a full
910
+ pre-auth pool evicts its **oldest** unauthenticated connection — answering it
911
+ `too_many_unauthenticated_connections` — rather than refusing the arrival, so a
912
+ poolful of squatters can no longer decide that nobody else, `healthz` included,
913
+ gets served. None of this stops a peer that can reach the port from causing
914
+ churn; the NetworkPolicy ingress rule in front of that port is the boundary,
915
+ and these bounds are defence in depth behind it.
916
+
917
+ The guest protocol version is deliberately NOT bumped: `token` is an optional,
918
+ additive envelope field, and the wire is otherwise byte-for-byte what it was.
919
+ Taking this release therefore requires no coupled rollout — no golden image has
920
+ to be rebuilt and no host has to be redeployed in step with it. A deployment
921
+ that wants the new modes turns them on in its own pod or image environment.
922
+
923
+ Read `NAMZU_AGENT_BIND_TOKEN` as an instance credential, not an isolation
924
+ boundary: the agent and the workload share a uid after deprivileging, so a
925
+ workload can read the agent's own environment out of `/proc`. It is
926
+ per-instance for that reason, and the network rule in front of the agent port
927
+ is the boundary it sits behind.
928
+
929
+ One behaviour changes for every existing deployment, Firecracker included, and
930
+ it is the reason to read this entry before upgrading. The `terminal` op used to
931
+ hand its shell the agent's whole environment; it now gets the same scrubbed
932
+ environment an `execute` child has always had, with every `NAMZU_AGENT_*` and
933
+ `NAMZU_SANDBOX_*` variable removed, and so does the `stty` resize helper behind
934
+ it. That closed a hole this release would otherwise have opened — the bind
935
+ token was visible in an interactive shell — and it means a terminal session no
936
+ longer sees the agent's own settings, such as `NAMZU_SANDBOX_WORKSPACE`. A
937
+ workload that needs a value in its terminal passes it in `env` on the
938
+ `openTerminal` call, which still wins over everything else, `TERM` included.
939
+ The bump stays `minor`: nothing exported changes shape, and the environment a
940
+ terminal is handed is guest-internal behaviour rather than a typed API, but it
941
+ is a change a terminal user can observe.
942
+
943
+ - 5604f73: A Kubernetes backend that claims VM-isolated sandboxes out of an [agent-sandbox](https://github.com/kubernetes-sigs/agent-sandbox) warm pool. New exported types `KubernetesBackendConfig` and `KubernetesClusterAccess`; `SandboxBackendConfig` and `SandboxProviderConfig` each gain an arm for it, so `createSandboxProvider({ backend: { tier: 'microvm', service: 'kubernetes', … } })` type-checks with no cast. Nothing existing changes shape.
944
+
945
+ **Take this upgrade for the new backend, not for a complete one.** This changeset covers acquire, readiness, address resolution and teardown; the execution surface, the acquire-time privilege probe and the lease arrive in the same release under their own changesets, and persistent workspaces (suspend/resume with a block-mode disk) and the cluster manifests follow in later ones. Every other backend is untouched.
946
+
947
+ What it does today, on a cluster running agent-sandbox v1.0.2 with a VM-isolating `RuntimeClass`:
948
+
949
+ - **Warm claim, or a direct Sandbox.** With `warmPoolName`, `create()` POSTs a `SandboxClaim` at that pool and the controller binds an already-running sandbox. Without it, it POSTs a `Sandbox` built from `sandboxTemplateName`'s pod template — necessitated rather than offered, since `SandboxClaim.spec.warmPoolRef` is required and a pool-less claim does not exist in the API.
950
+ - **The claim is pristine.** `spec.env` and `spec.volumeClaimTemplates` are never set, because a claim carrying either is forced to cold-start upstream instead of adopting a pool sandbox. It would still work; it would just stop being fast. Per-sandbox `env`, `memoryLimitMb`, `maxProcesses` and `egress` are therefore refused by name instead of accepted and dropped — set them on the `SandboxTemplate` the pool is built from.
951
+ - **The bound sandbox's own identity.** A pool sandbox keeps the name the pool generated for it, so the backend reads `status.sandbox` back rather than assuming the claim's name; the sandbox `id` is that cluster name, which makes an id in a log line a `kubectl get sandbox` argument.
952
+ - **Nothing left behind.** Every created object carries an absolute `shutdownTime` (default one hour, `claimTtlSeconds`) plus `shutdownPolicy: Delete`, so a host that dies mid-run costs one expiry rather than a leaked sandbox — `ttlSecondsAfterFinished` deliberately is not used, because its timer starts from a `Finished` condition a crashed host never reaches. Every failure on the create path deletes what it created on a separate short budget; an object already gone counts as released.
953
+ - **A per-instance agent credential.** The pod's own `metadata.uid`, read with one `GET` after readiness and delivered to the guest through the downward API. No claim mutation, so the warm path stays pristine.
954
+
955
+ Credentials arrive through `access`: `{ inCluster: true }` reads the projected ServiceAccount volume, and anything else supplies `{ server, ca?, getToken }`. There is no kubeconfig parsing in the package and no new dependency — `@namzu/sandbox` still declares no `dependencies` key.
956
+
957
+ - 437e3d3: The Kubernetes backend now ships the cluster-side half of itself: the guest image, its entrypoint, the `RuntimeClass`/`SandboxTemplate`/`SandboxWarmPool`/`NetworkPolicy`/RBAC manifests, and five scripts that measure the five acceptance criteria against a live cluster — all under `packages/sandbox/k8s/`, which is **not part of the published package** (the `files` array still packs only `dist` and `src`; `npm pack --dry-run` confirms it). None of that is a consumer-visible surface. What IS:
958
+
959
+ **The guest agent now exits cleanly on `SIGTERM` — bumped `minor`, not `patch`, for exactly this.** `k8s/entrypoint.sh` `exec`s straight into `agent/agent.cjs`, so in a real pod the agent is pid 1 of the container's own pid namespace, and Linux leaves a signal whose default action is "terminate" un-applied for pid 1 unless the process installs its own handler. With none registered, an operator deleting a sandbox watched it ride out the pod's full `terminationGracePeriodSeconds` before `SIGKILL` finally landed — every delete paid that tax, cluster-wide, whatever backend dialed the agent. The new handler closes the listener and calls `process.exit(0)` the moment `SIGTERM` arrives. It is not a graceful drain: an in-flight `exec` or an open terminal gets no grace window, the same "gone" a caller already has to handle from a pod the cluster removed out from under it. This is additive guest behavior on a file the Firecracker tier also runs in production — nothing about the vsock/unix paths changes, and `REMOTE_EXECUTION_PROTOCOL_VERSION`/`FIRECRACKER_AGENT_PROTOCOL_VERSION` are untouched — but it is a real, observable change to when a pod actually terminates, which is why this is `minor` rather than `patch`.
960
+
961
+ **Documented for the first time, no behavior change:** a confirmed `exec` cancellation on this backend resolves with the terminal signal/exit code the shared `RemoteExecutionController` observed — the same contract the Firecracker tier's `exec` already honors, now stated on `docs/sdk/kubernetes-sandbox.md` rather than left implicit.
962
+
963
+ Everything else in this change is infrastructure an operator applies by hand — see `packages/sandbox/k8s/README.md` for the apply order and how to run each acceptance script, and `docs/sdk/kubernetes-sandbox.md`'s new deployment section for what each one measures. The five acceptance numbers themselves are not yet in that table; they are gathered by running those scripts against a real Kata cluster, not by this change.
964
+
965
+ - 432db25: The Kubernetes backend's `KubernetesBackendConfig` gains an optional `egress` block: `{ policy: EgressPolicy; networkPolicyName?: string; engine?: 'core' | 'cilium' }` (`engine` defaults to `'core'`). New exported types `KubernetesEgressConfig` and `KubernetesEgressEngine`. Nothing existing changes shape — `egress` is additive and optional, and every Sandbox this backend produces now carries a `sandbox.namzu.ai/template` label it did not carry before, which is additive metadata rather than a behavior change for an existing caller.
966
+
967
+ **What it does.** `deny-all` and `allow-all` translate into a `NetworkPolicy` (egress rules that always leave the cluster's own DNS reachable, even under `deny-all`). `static` and `resolver` — hostname allowlists — are **refused at construction**, before any API call, naming the policy kind and what the cluster needs: core Kubernetes `NetworkPolicy` has no FQDN concept at all. Declaring `engine: 'cilium'` turns that refusal into an emission: a `CiliumNetworkPolicy` with a `toFQDNs` entry for every allowed host. This backend never emits `HTTP_PROXY`/`HTTPS_PROXY` as a substitute for an unenforceable policy — that is the container tier's still-open gap, not repeated here.
968
+
969
+ **Verify, never trust.** This backend never creates the `NetworkPolicy`/`CiliumNetworkPolicy` itself — like the docker backend's network, it is operator-applied. The first `create()` after construction (not `createSandboxProvider`, which still contacts nothing) `GET`s the object named by `networkPolicyName` (default `${sandboxTemplateName}-egress`) and refuses to proceed on a 404 or a shape mismatch, naming the field that is wrong. It runs once per backend and is not cached across a failure, so fixing the cluster and creating again retries it.
970
+
971
+ **Take this upgrade for the label, even without using `egress`.** Every Sandbox this backend creates directly now carries `sandbox.namzu.ai/template: <sandboxTemplateName>` on its pod — agent-sandbox's own controller-owned template label is written only on a Sandbox adopted out of a `SandboxWarmPool`, never on one this backend POSTs directly, so a `NetworkPolicy` an operator writes against a direct Sandbox should select by this new label. A `SandboxWarmPool`'s own `SandboxTemplate` needs the same label added to its `podTemplate.metadata.labels` for a pooled sandbox to match it too — this backend has no path to add it after the fact.
972
+
973
+ RBAC: when `config.egress` is set, the ServiceAccount also needs `get` on `networkpolicies` (`networking.k8s.io`), or `get` on `ciliumnetworkpolicies` (`cilium.io`) under `engine: 'cilium'`.
974
+
975
+ - a3ae83e: The Kubernetes backend now returns a working `Sandbox`, refuses to hand one back until the guest has proved it is deprivileged, and keeps a long run's pod from expiring underneath it.
976
+
977
+ **The execution surface exists.** `exec`, `writeFile`, `readFile`, `listFiles`, `openTerminal` and `openTcpConnection` no longer throw `KubernetesAgentTransportPendingError` — that error is gone, and so is the reason for it. `exec` runs through the same reserve-before-admission controller every other remote backend uses, so an `AbortSignal` terminates the guest process and the peer confirms the termination rather than the host abandoning the wait. `destroy()` kills and awaits every terminal it handed out before releasing the object, which is what makes offering `openTerminal` compliant at all; it is idempotent, and an object something else already reaped counts as released. Every call after it throws `KubernetesSandboxDestroyedError` naming the operation.
978
+
979
+ **`setNetworkPolicy`, `spawnDetached` and `walkFiles` are absent, deliberately.** Egress here is a `NetworkPolicy` on the pool's `SandboxTemplate` and there is no per-running-pod knob, the guest agent has no detached-spawn op, and bounded search is not in this batch. The SDK's contract says a backend that cannot honour an optional method must omit it rather than accept it and quietly do nothing; a test asserts each stays absent.
980
+
981
+ **Every acquire now proves the guest is deprivileged.** Before `create()` resolves, the backend reads `/proc/self/status` through the agent's `execute` op and refuses unless `CapInh`, `CapPrm`, `CapEff` and `CapBnd` are ALL zero and `NoNewPrivs` is 1. Checking `CapEff` alone would pass a container running as uid 0 with the full bounding set. A refusal destroys the instance and rejects, so no handle to an under-hardened sandbox escapes, and the error distinguishes "the probe could not run" (a minimal image with no `cat`) from "the process is privileged". **There is no configuration that turns this off.** If your image's entrypoint does not end with `exec setpriv --reuid --regid --clear-groups --inh-caps=-all --bounding-set=-all --no-new-privs -- node agent.cjs`, or equivalent, `create()` will now reject where it previously returned. The probe carries its own deadline — `min(readyTimeoutMs, 15s)` — so a pod whose agent has wedged (out of memory, an event loop the workload blocked) is refused on your acquire budget rather than held open for the execution controller's five-minute default; budget for a `create()` that can take `readyTimeoutMs` plus that again in the worst case. The probe is a check that the deprivileging happened, not a boundary against a guest that is already compromised — it asks the agent to report its own `/proc/self/status`. The boundary is still the VM and the `NetworkPolicy`.
982
+
983
+ **A handle renews its own lease.** The absolute `shutdownTime` this backend stamps on every object it creates bounds a leak; unrenewed it also bounded the RUN, so a session outliving `claimTtlSeconds` (default one hour) had its pod deleted mid-command. The handle now merge-PATCHes that expiry a full TTL forward every half TTL, jittered ±10%, and `destroy()` stops it. **This adds an RBAC requirement**: the ServiceAccount needs `patch` on `sandboxclaims` and `sandboxes` in the sandbox namespace, alongside the verbs it already needed. A renewal that fails is reported to the new optional `KubernetesBackendConfig.onLeaseRenewalError` and retried on the next tick; each PATCH is bounded on its own clock (a quarter of the interval, capped at 30 seconds), so an API server that accepts a renewal and never answers it is abandoned and retried rather than parking the loop and letting the lease expire in silence; and one that finds the object already deleted stops the loop and marks the handle gone, so later calls throw `KubernetesSandboxGoneError` instead of dialing a pod that no longer exists. A handle you drop without calling `destroy()` keeps renewing for as long as the process lives, so `destroy()` is now load-bearing for cleanup inside a long-lived host.
984
+
985
+ New on the public surface: `KubernetesPrivilegeProbeError` (with `reason`, `PrivilegeProbeFailure` and `ProcStatusPrivileges`), `KubernetesSandboxDestroyedError`, `KubernetesSandboxGoneError`, `KubernetesAgentUnauthorizedError`, `AgentPreauthFrameTooLargeError`, `TCP_PREAUTH_FRAME_LIMIT_BYTES`, and `KubernetesBackendConfig.onLeaseRenewalError`. Removed: `KubernetesAgentTransportPendingError`, which was never exported from the package entry point and could only ever be thrown by a method that now works.
986
+
987
+ One limit worth knowing before you write a large file: every request to the guest dials a fresh connection, so every request is that connection's first — not-yet-authenticated — frame and is bounded by the guest's pre-auth ceiling (8 MiB by default) on every call, not once. A `writeFile` whose base64 body would exceed it throws `AgentPreauthFrameTooLargeError` before dialing, naming the limit; in practice bodies above about 5.9 MiB raw do not fit. Chunking is not implemented — raise `NAMZU_AGENT_MAX_PREAUTH_FRAME_BYTES` in the deployment, or split the write.
988
+
989
+ Every other backend is untouched.
990
+
991
+ - 53c526c: `SandboxAgentHandle` (re-exported from the package's public entry point)
992
+ gains a fourth arm: `{ kind: 'tcp', host, port, token }`. `VsockAgentTransport`
993
+ (also public) gains a new `executeStreamed()` method, and its
994
+ `VsockTransportOptions` gain an optional `onDial` callback. All three
995
+ changes are additive and backward compatible — existing `unix`/`vsock`/`mtls`
996
+ handles, existing `VsockAgentTransport` callers, and the guest protocol are
997
+ untouched (`firecracker/__tests__/transport.test.ts` passes unmodified) — but
998
+ they are genuinely new surface in the compiled `.d.ts`, not yet constructible
999
+ by anything outside `packages/sandbox/src/backends/` until a later workstream
1000
+ wires the kubernetes backend up to them.
1001
+
1002
+ Alongside this, a new (package-internal, not yet exported) `KubernetesAgentTransport`
1003
+ in `src/backends/kubernetes/transport.ts` dials the guest agent (`agent/agent.cjs`)
1004
+ directly over a routed pod network for the upcoming kubernetes backend.
1005
+
1006
+ The `tcp` dialer is a plain `net.connect({ host, port })` per call — no
1007
+ routing preamble, no ack, no cached socket or IP — so a `host` that is a
1008
+ Kubernetes Service FQDN is re-resolved on every request and a resumed
1009
+ pod's new address costs nothing extra. The handle's optional `token`
1010
+ rides in each request envelope (the credential field the guest agent
1011
+ already accepts); a wrong token surfaces as a named
1012
+ `KubernetesAgentUnauthorizedError` rather than a generic protocol error.
1013
+
1014
+ Because every `tcp` request dials a fresh connection, that connection's
1015
+ first frame is also the one the guest agent has not authenticated yet,
1016
+ so it is bound by the agent's pre-auth frame ceiling (8 MiB by default)
1017
+ on every call, not just on first use. This transport now checks an
1018
+ outgoing envelope's size against that ceiling BEFORE dialing and throws
1019
+ a named `AgentPreauthFrameTooLargeError` naming the limit, instead of
1020
+ opening a connection the agent would refuse anyway. Chunking a large
1021
+ `write-file` body across multiple frames is a documented follow-up, not
1022
+ implemented here.
1023
+
1024
+ `KubernetesAgentTransport` also accepts an optional `onTiming` callback
1025
+ reporting a completed `exec()` call's dial/reserve/execute/drain
1026
+ durations (never the token, command, or output), so the kubernetes
1027
+ backend's sub-second warm-acquire target can be measured rather than
1028
+ assumed.
1029
+
1030
+ - 50de52b: The Kubernetes backend can now keep a workspace: a sandbox with a block-mode disk that survives being suspended.
1031
+
1032
+ **New verb, not a new provider.** `createKubernetesWorkspace(config, options)` returns a `KubernetesWorkspace` — the SDK's `Sandbox`, plus `suspend()`, `resume()`, a `suspended` flag, and a `destroy()` that takes `deleteDisk`. It is separate from `createSandboxProvider` because a `SandboxProvider` promises an ephemeral sandbox per run and this promises the opposite; `warmPoolName` is ignored, since a workspace is always a `Sandbox` POSTed directly. `@namzu/sdk`'s `Sandbox` is untouched: no new `SandboxStatus` member, no `suspend?()`/`resume?()` on the shared contract, no `deleteDisk` on the shared `SandboxDestroyOptions`.
1033
+
1034
+ **`destroy()` keeps the disk.** The API has `operatingMode` and it has DELETE, and nothing in between, so there is no delete-compute-keep-disk verb to offer. `destroy()` and `destroy({ deleteDisk: false })` SUSPEND and leave the object standing; only `destroy({ deleteDisk: true })` DELETEs the Sandbox and cascades to its Pod, Service and PVC. The default is the non-destructive one because `destroy()` is what a `finally` block calls. No failure path ever deletes: a create or resume that fails after the object exists suspends it and rethrows — including a create that POSTed the object itself, because two processes can be coming up on one name at once and the one that got the `201` would otherwise delete the disk the other just adopted. The named cost: a failed create can leave one suspended `Sandbox` and its PVC standing, which nothing reaps and which the caller finds again under the same name.
1035
+
1036
+ **A workspace carries no lease.** Unlike a task sandbox it gets no `shutdownTime`, no `shutdownPolicy: Delete` and no renewal loop. An expiry on a workspace is a timer that deletes your files, and a renewal loop makes keeping them conditional on a host process staying up. The trade is explicit: **nothing reaps a workspace you abandon** — the PVC stands until someone calls `destroy({ deleteDisk: true })` or deletes the Sandbox by hand.
1037
+
1038
+ **The disk is fixed at creation and must be `volumeMode: Block` — every entry of it.** `Sandbox.spec.volumeClaimTemplates` is CEL-immutable and a `SandboxClaim` carrying one is forced to cold-start, so "claim a warm diskless sandbox and attach a disk later" is not expressible in this API; resizing is out of scope for the same reason. The `SandboxTemplate` a workspace is built from must declare at least one `volumeClaimTemplates` entry, every entry must be `Block`, and every entry must be claimed through a container's `volumeDevices` rather than `volumeMounts`. Anything else throws the new `KubernetesWorkspaceDiskError` before anything is created — each refused shape otherwise WORKS: no disk gives you a sandbox whose files vanish on the first suspend, and a `Filesystem` PVC under a VM-isolating RuntimeClass reaches the guest over a filesystem passthrough that pays a round trip per file operation, so a dependency-tree walk is several times slower and nothing fails.
1039
+
1040
+ **`config.egress` applies to a workspace too.** `createKubernetesWorkspace` runs the same two steps a provider `create()` runs: a `static`/`resolver` hostname allowlist with no FQDN-capable `engine` is refused synchronously, before any request, and the `NetworkPolicy` an operator applied is fetched and matched against the translation before anything is created. It is checked against the template the workspace is built from (`options.sandboxTemplateName`, falling back to `config.sandboxTemplateName`), because that name is the pod label the policy's `podSelector` matches — a deployment with a separate workspace template needs a policy object for it (`<workspace template>-egress` by default), and the task template's does not cover a workspace pod. Unlike the provider's once-per-backend check, this one runs on every call.
1041
+
1042
+ **A suspend waits for the pod, and only the pod.** `suspend()` resolves once the pod is gone or in a terminal phase — not on the Sandbox's `Suspended` condition, which upstream documents as lingering True after a resume, and not on a `deletionTimestamp`, which is set while the guest is still running and still writing to the disk. A pod that outlives `readyTimeoutMs` rejects with the new `KubernetesWorkspaceSuspendTimeoutError` rather than resolving early. A call already in flight when `suspend()` starts is not cancelled: it fails at the transport, not with the suspended error.
1043
+
1044
+ **A state is recorded when the cluster confirms it, never before.** Both verbs are idempotent by early-returning on a recorded state, so the moment that record is written decides what happens to a failure. `suspended` is written after the patch lands AND the pod is observed stopped; `deleted` after the DELETE resolves, or reports the object already gone. A request that fails leaves the state it found and rethrows, so you can retry: a refused suspend patch leaves the workspace running and still serving calls, a suspend whose pod outlived the wait admits no call but is not recorded as finished, and a failed DELETE does not answer your retry "already deleted" while the `Sandbox`, its pod and its PVC stand on the cluster with nothing left that would remove them. Concurrency is covered the other way round, with a single flight per verb: a second `suspend()` or `destroy()` arriving mid-transition awaits the one in progress instead of sending a second request into the window the deferred record opens, and `destroy()` with no options shares the suspend's flight because it is a suspend — the shared request running under the FIRST caller's `signal`, since that is what sharing one request means. `destroy()` is idempotent across the two shapes as well as within each: a plain `destroy()` on a workspace already removed by `destroy({ deleteDisk: true })` is a no-op in either admission order, because `destroy()` is what a `finally` block calls and what it asks for has happened; an explicit `suspend()` on a deleted workspace still throws.
1045
+
1046
+ **Nothing but `deleteDisk: true` deletes — including the path you never call.** When an execution's cancellation cannot be confirmed (a wedged agent, a partitioned pod, the `cancel-execution` window closing with no answer), the shared execution controller retires the pod that command was left in. On a task sandbox retiring means DELETEing the object, correctly — it is disposable and its disk is scratch. A workspace is retired by the same `operatingMode: Suspended` patch `suspend()` sends: the `exec()` still rejects, carrying `retirement: { accepted: true }` once that patch lands (`accepted: false`, with the error, when it does not), the workspace then reads `suspended: true` and admits nothing, and `resume()` brings up a fresh pod on the same disk. A `destroy({ deleteDisk: true })` whose DELETE fails leaves the same shape — session torn down, nothing admitted, `suspended: true` — because neither the delete nor a suspend reached the cluster, so both `resume()` and a retried delete are open to you.
1047
+
1048
+ **A resume rebuilds the address and the token.** A resumed pod keeps the sandbox's name and gets a new uid and a new IP, so `resume()` re-resolves the address, re-reads the bind token and rebuilds the transport, skipping any pod carrying a `deletionTimestamp` or in a terminal phase — while the outgoing pod terminates, a `GET` by name can still answer with it and a selector list can return it beside the new one. `Ready` is not a transition signal either — the controller leaves it standing across a resume the way it leaves `Suspended` standing — so the uid is POLLED under `readyTimeoutMs` until a live pod with a uid different from the one the last landed suspend patch retired is found, rather than read once from a status that has not caught up. That covers a resume issued straight after a suspend whose pod outlived its own wait, which arrives mid-drain with no live pod of that name to read at all. Every patch that lands is recorded the same way — an explicit `suspend()`, the retirement above, and the cleanup after a failed create or resume, which swallows its own failure and therefore records only when the request actually came back; a pod nobody asked the controller to remove has no replacement to wait for, and excluding it would time out a resume whose workspace was perfectly usable. What visibly moves depends on the Service: the token is always new, the pod IP always changes, and a `status.serviceFQDN` address does not, because the Service outlives the pod. The acquire-time privilege probe runs again on every resume. Between a suspend and a resume every call throws the new `KubernetesWorkspaceSuspendedError` and issues no dial, because the Service outlives the pod and a dial would hang on a connect timeout that names nothing. `status` reports `destroyed` while suspended — `SandboxStatus` has no suspended member — and `suspended` is what tells the recoverable state from the final one.
1049
+
1050
+ **Calling it twice reattaches — and what is adopted is checked.** The Sandbox is named `namzu-ws-<workspaceId>`, so a second process finds the same workspace; a create that collides adopts the existing object and resumes it if it was asleep. Because an adopt is handed an object this call did not build, the object is checked against the configuration before it is woken: it must carry a block disk, its pod template must carry `sandbox.namzu.ai/template` for the template this call builds from, and — when `runtimeClassName` is configured — it must already run under that class. A disagreement throws the new `KubernetesWorkspaceMismatchError` and nothing is patched, woken or dialed. The label check runs whether or not `config.egress` is set, since that label is what a policy's `podSelector` matches and adopting an object built from another template would hand back a pod the verified policy does not select; the RuntimeClass check is there because the privilege probe cannot see a missing VM boundary (`/proc/self/status` reads the same under Kata and under runc). The fix is to point the workspace at the template it was built from, or to delete the Sandbox — which takes its disk with it — and create it again.
1051
+
1052
+ A `workspaceId` that is not already a legal DNS-1123 label is refused rather than sanitised, because two ids that sanitise to one name would silently share one disk. It is a name and not a lock: two host processes can adopt one running workspace, and either one's `destroy()` suspends the pod the other is executing in.
1053
+
1054
+ **Fixed on the task path, in the same change:** a pool-less `create()` copied the `SandboxTemplate`'s `podTemplate` and dropped its `volumeClaimTemplates`, so a task template declaring a disk produced a healthy Sandbox with no disk and a container naming a volume that did not exist. Both paths now copy the template's `volumeClaimTemplates` verbatim. If you have been working around that by declaring no disk on a task template, nothing changes; if you declared one and wondered where it went, it now arrives.
1055
+
1056
+ New on the public surface: `createKubernetesWorkspace`, `KubernetesWorkspace`, `KubernetesWorkspaceOptions`, `KubernetesWorkspaceDestroyOptions`, `KubernetesWorkspaceTransitionOptions`, `KubernetesWorkspaceDiskError`, `KubernetesWorkspaceMismatchError`, `KubernetesWorkspaceSuspendTimeoutError`, `KubernetesWorkspaceSuspendedError`. RBAC is unchanged — a workspace uses verbs the task path already needed. Every other backend is untouched.
1057
+
1058
+ - a70c936: A `Sandbox` contract conformance suite: `defineSandboxConformance` (`packages/sandbox/src/testing/sandbox-conformance.ts`) asserts `exec`'s exit codes and streamed output, the `AbortSignal` contract (the process is genuinely terminated, never a resolved result that reads as an unaborted success), a `writeFile`/`readFile` round trip including binary content, `listFiles`, `openTerminal` ownership on `destroy()`, `openTcpConnection` to guest loopback and its refusal of a non-loopback host, destroy idempotence, and every call failing once destroyed. It takes its `describe`/`it`/`expect` and a factory producing a fresh `Sandbox` as arguments, the same shape `@namzu/sdk/testing`'s checkpoint-store and provider-driver suites already use, so `@namzu/sandbox` gains no test dependency from shipping it and a caller can run it against a recording harness.
1059
+
1060
+ **New exported surface, within the package only.** `@namzu/sandbox` has no `testing` subpath in its published `exports` map, and this change does not add one — that is a deliberate, separate decision. A caller inside this monorepo imports `defineSandboxConformance` by relative path (`src/testing/sandbox-conformance.js`), exactly as the two new test files below do. It is minor rather than patch because it is new, intentionally-public TypeScript surface a consumer with access to the package's source can import and depend on, even though nothing in `@namzu/sandbox`'s npm entry point changes shape.
1061
+
1062
+ **Proven against two backends, not one.** `packages/sandbox/src/backends/kubernetes/__tests__/conformance.test.ts` and `packages/sandbox/src/backends/firecracker/__tests__/conformance.test.ts` both run the identical suite — the kubernetes backend over a real `agent/agent.cjs` on a loopback TCP socket, Firecracker over the same agent on its existing unix-domain-socket fixture — which is what makes it a contract suite rather than one backend's tests wearing a new name. `packages/sandbox/src/testing/__tests__/conformance-fails-a-broken-sandbox.test.ts` is the suite's own negative test: three deliberately broken `Sandbox`s (resolves `exec` after abort, `destroy()` leaves a terminal running, `readFile` returns corrupted bytes) each fail it by name.
1063
+
1064
+ Nothing existing changes shape — every other export, every shipped backend's behavior, is untouched.
1065
+
1066
+ ### Patch Changes
1067
+
1068
+ - 4b0e7ad: `defineSandboxConformance`'s `openTcpConnection` positive case now starts its echo listener INSIDE the guest, through `openTerminal`, instead of on the orchestrator/test process's own loopback. The old fixture only ever proved anything for a backend whose "guest" happened to share that loopback with the test process (a Firecracker unit test over a local socket, a fake-agent-in-process kubernetes test) — it could never pass against a real remote sandbox, which cannot dial the orchestrator's loopback at all. Confirmed in-cluster: this case now passes against a live kubernetes backend acquisition on a real kind cluster, where it previously failed with `connect ECONNREFUSED`.
1069
+
1070
+ Two new optional fields on `SandboxConformanceOptions` — `guestCanRunNode` and `guestListenerCommand` — let a backend whose guest cannot run a listener this way skip the case with a stated reason (its own title) rather than fail spuriously; both default to the existing behavior (node is assumed available, since every shipped backend's guest agent already runs on node), so no existing caller of `defineSandboxConformance` needs to change anything.
1071
+
1072
+ Patch, not minor: this module has no `testing` subpath in `@namzu/sandbox`'s own `exports` map (see the file's own doc comment) — a caller reaches it only by relative path within the monorepo, as `backends/kubernetes/__tests__/conformance.test.ts` and `backends/firecracker/__tests__/conformance.test.ts` already do — so this is not yet public surface, and the added options are additive and optional regardless.
1073
+
1074
+ Also corrects a self-contradictory doc comment on `Sandbox.openTerminal` (`@namzu/sdk`): it told an implementer to both "throw" and "omit" for a guest that cannot provide one. It now says only "omit", matching `Sandbox.openTcpConnection`'s own wording and this suite's documented skip-if-unavailable convention. Comment-only; no type or behavior changed.
1075
+
1076
+ - 13d01db: Adds an internal Kubernetes API client (`backends/kubernetes/k8s-client.ts`, not yet exported from the package entrypoint) for an in-progress Kubernetes/Kata sandbox backend. It speaks the API server with bare `fetch`, falling back to `node:https` only when a custom cluster CA is supplied, and bootstraps in-cluster credentials straight from the projected ServiceAccount volume — the same zero-dependency pattern the ACI and Firecracker backends already use. `@namzu/sandbox` still declares no `dependencies` key.
1077
+ - 77adb50: Fix both the kubernetes/Firecracker guest agent (`agent/agent.cjs`) and the
1078
+ container-tier HTTP worker (`worker/server.js`) reporting a cancelled
1079
+ `exec()` as a clean, unaborted-looking success when the target process
1080
+ ignores `SIGTERM` but happens to finish on its own before the cancel grace
1081
+ window elapses. Both peers' `terminateAndConfirm` only escalated to
1082
+ `SIGKILL` if the owned process group was still alive at the end of that
1083
+ window, with nothing checking that the exit was actually caused by the
1084
+ signal — so an ignoring process whose natural runtime was under the grace
1085
+ period ran to completion untouched, in violation of
1086
+ `SandboxExecOptions.signal`'s contract ("must terminate the owned process
1087
+ ... never silently ignore the signal and let the command run to
1088
+ completion"). This is the same mechanism on both transports, found on the
1089
+ guest agent first (issue #469's kind conformance run) and confirmed to
1090
+ exist verbatim in the worker once looked for.
1091
+
1092
+ The default (previously `2000`ms on both) is now `250`ms for
1093
+ `NAMZU_AGENT_CANCEL_GRACE_MS` (agent) and `NAMZU_SANDBOX_CANCEL_GRACE_MS`
1094
+ (worker) — comfortably under the shared conformance suite's adversarial
1095
+ fixture (a command that finishes on its own in ~400ms) while still enough
1096
+ for a fast, well-behaved SIGTERM handler's cleanup. A deployment that
1097
+ genuinely needs a longer window for cooperative shutdown sets either
1098
+ variable explicitly; both were already, and remain, overridable. A new
1099
+ regression test on each transport deliberately leaves its own grace
1100
+ variable unset — the one thing every other suite on that transport
1101
+ overrides — and fails against the old default, passes against the new one.
1102
+
1103
+ **Defect 2, found alongside the agent fix, is now MITIGATED for the
1104
+ Kubernetes backend's shipped image, not merely documented:** in a real pod
1105
+ the agent used to run as the container's PID 1 with no subreaper
1106
+ (`packages/sandbox/k8s/entrypoint.sh` `exec`ed straight into it), so a
1107
+ background job forked by a cancelled `sh -c` command (`... &`) reparented
1108
+ to the agent on `SIGKILL` and was never reaped — Node's `child_process`
1109
+ only `waitpid()`s the children it spawned itself — running that
1110
+ cancellation out the full `RemoteExecutionController` cancel-confirm window
1111
+ (8s by default) and tearing the sandbox down instead of confirming.
1112
+ `k8s/entrypoint.sh` now execs into `tini` (installed in `k8s/Dockerfile`)
1113
+ as the container's real PID 1 and subreaper, with the guest agent as its
1114
+ child; `tini` reaps the orphan and forwards `SIGTERM` to the agent exactly
1115
+ as before. **A host building its own image from `agent.cjs` directly rather
1116
+ than from `k8s/Dockerfile` must still provide its own subreaper as PID 1**
1117
+ — this fix lives in the shipped image, not in the agent itself, since
1118
+ `agent.cjs` cannot know what pid it is. Root-caused with live evidence in
1119
+ `research/k8s-sandbox/kind-e2e-results.md` ("Defect 2", `2026-09-16`) and
1120
+ re-verified in-cluster against the new image in
1121
+ `research/k8s-sandbox/abort-case-recheck-results.json`.
1122
+
1123
+ **Patch, not minor or major:** `agent/agent.cjs` and `worker/server.js` are
1124
+ guest/worker files that run INSIDE a sandbox or container, never imported
1125
+ by a consumer of the published `@namzu/sandbox` tarball (`npm pack
1126
+ --dry-run` packs only `dist` and `src`); the k8s image's `Dockerfile` and
1127
+ `entrypoint.sh` are likewise deployment artifacts, not the package's own
1128
+ `exports`. Both defaults that changed are guest-/worker-internal timings
1129
+ with no public type or exported symbol affected, and both changes narrow
1130
+ when a cancellation escalates to `SIGKILL` and add a real subreaper —
1131
+ strictly tightening what `SandboxExecOptions.signal`'s contract already
1132
+ promised, never loosening it, so no caller-visible behavior a consumer
1133
+ could have depended on gets worse.
1134
+
1135
+ - Updated dependencies [68e535b]
1136
+ - Updated dependencies [a9e4b19]
1137
+ - Updated dependencies [a54dc71]
1138
+ - Updated dependencies [86a3818]
1139
+ - Updated dependencies [03630cd]
1140
+ - Updated dependencies [f33c62b]
1141
+ - Updated dependencies [a8df193]
1142
+ - Updated dependencies [8bfe291]
1143
+ - Updated dependencies [6ae4072]
1144
+ - Updated dependencies [dd8702d]
1145
+ - Updated dependencies [92ab1d9]
1146
+ - Updated dependencies [e6d6d1e]
1147
+ - Updated dependencies [7ca8c7d]
1148
+ - @namzu/sdk@40.0.0
1149
+
3
1150
  ## 13.0.0
4
1151
 
5
1152
  ### Patch Changes