@namzu/sandbox 14.0.0 → 15.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (99) hide show
  1. package/CHANGELOG.md +838 -0
  2. package/README.md +310 -14
  3. package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
  4. package/dist/backends/aci-standby-pool/index.js +13 -1
  5. package/dist/backends/aci-standby-pool/index.js.map +1 -1
  6. package/dist/backends/docker/index.d.ts.map +1 -1
  7. package/dist/backends/docker/index.js +19 -1
  8. package/dist/backends/docker/index.js.map +1 -1
  9. package/dist/backends/firecracker/index.d.ts.map +1 -1
  10. package/dist/backends/firecracker/index.js +12 -2
  11. package/dist/backends/firecracker/index.js.map +1 -1
  12. package/dist/backends/firecracker/protocol.d.ts +459 -8
  13. package/dist/backends/firecracker/protocol.d.ts.map +1 -1
  14. package/dist/backends/firecracker/protocol.js +136 -0
  15. package/dist/backends/firecracker/protocol.js.map +1 -1
  16. package/dist/backends/firecracker/transport.d.ts +539 -6
  17. package/dist/backends/firecracker/transport.d.ts.map +1 -1
  18. package/dist/backends/firecracker/transport.js +1171 -24
  19. package/dist/backends/firecracker/transport.js.map +1 -1
  20. package/dist/backends/kubernetes/egress-policy.d.ts +1088 -11
  21. package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -1
  22. package/dist/backends/kubernetes/egress-policy.js +2173 -29
  23. package/dist/backends/kubernetes/egress-policy.js.map +1 -1
  24. package/dist/backends/kubernetes/identity.d.ts +193 -0
  25. package/dist/backends/kubernetes/identity.d.ts.map +1 -0
  26. package/dist/backends/kubernetes/identity.js +147 -0
  27. package/dist/backends/kubernetes/identity.js.map +1 -0
  28. package/dist/backends/kubernetes/index.d.ts +678 -33
  29. package/dist/backends/kubernetes/index.d.ts.map +1 -1
  30. package/dist/backends/kubernetes/index.js +1180 -95
  31. package/dist/backends/kubernetes/index.js.map +1 -1
  32. package/dist/backends/kubernetes/ingress-policy.d.ts +375 -0
  33. package/dist/backends/kubernetes/ingress-policy.d.ts.map +1 -0
  34. package/dist/backends/kubernetes/ingress-policy.js +1050 -0
  35. package/dist/backends/kubernetes/ingress-policy.js.map +1 -0
  36. package/dist/backends/kubernetes/k8s-client.d.ts +213 -4
  37. package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -1
  38. package/dist/backends/kubernetes/k8s-client.js +359 -52
  39. package/dist/backends/kubernetes/k8s-client.js.map +1 -1
  40. package/dist/backends/kubernetes/lease.d.ts +40 -14
  41. package/dist/backends/kubernetes/lease.d.ts.map +1 -1
  42. package/dist/backends/kubernetes/lease.js +68 -18
  43. package/dist/backends/kubernetes/lease.js.map +1 -1
  44. package/dist/backends/kubernetes/objects.d.ts +423 -3
  45. package/dist/backends/kubernetes/objects.d.ts.map +1 -1
  46. package/dist/backends/kubernetes/objects.js +364 -2
  47. package/dist/backends/kubernetes/objects.js.map +1 -1
  48. package/dist/backends/kubernetes/per-sandbox-policy.d.ts +219 -0
  49. package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -0
  50. package/dist/backends/kubernetes/per-sandbox-policy.js +407 -0
  51. package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -0
  52. package/dist/backends/kubernetes/rbac.d.ts +153 -0
  53. package/dist/backends/kubernetes/rbac.d.ts.map +1 -0
  54. package/dist/backends/kubernetes/rbac.js +177 -0
  55. package/dist/backends/kubernetes/rbac.js.map +1 -0
  56. package/dist/backends/kubernetes/sandbox.d.ts +81 -14
  57. package/dist/backends/kubernetes/sandbox.d.ts.map +1 -1
  58. package/dist/backends/kubernetes/sandbox.js +149 -15
  59. package/dist/backends/kubernetes/sandbox.js.map +1 -1
  60. package/dist/backends/kubernetes/transport.d.ts +935 -9
  61. package/dist/backends/kubernetes/transport.d.ts.map +1 -1
  62. package/dist/backends/kubernetes/transport.js +1958 -62
  63. package/dist/backends/kubernetes/transport.js.map +1 -1
  64. package/dist/backends/kubernetes/workspace.d.ts +1149 -18
  65. package/dist/backends/kubernetes/workspace.d.ts.map +1 -1
  66. package/dist/backends/kubernetes/workspace.js +2825 -186
  67. package/dist/backends/kubernetes/workspace.js.map +1 -1
  68. package/dist/backends/remote-execution-controller.d.ts +14 -0
  69. package/dist/backends/remote-execution-controller.d.ts.map +1 -1
  70. package/dist/backends/remote-execution-controller.js.map +1 -1
  71. package/dist/index.d.ts +231 -13
  72. package/dist/index.d.ts.map +1 -1
  73. package/dist/index.js +247 -5
  74. package/dist/index.js.map +1 -1
  75. package/dist/testing/sandbox-conformance.d.ts +39 -5
  76. package/dist/testing/sandbox-conformance.d.ts.map +1 -1
  77. package/dist/testing/sandbox-conformance.js +436 -5
  78. package/dist/testing/sandbox-conformance.js.map +1 -1
  79. package/package.json +3 -3
  80. package/src/backends/aci-standby-pool/index.ts +16 -1
  81. package/src/backends/docker/index.ts +22 -1
  82. package/src/backends/firecracker/index.ts +14 -2
  83. package/src/backends/firecracker/protocol.ts +514 -6
  84. package/src/backends/firecracker/transport.ts +1492 -40
  85. package/src/backends/kubernetes/egress-policy.ts +3064 -53
  86. package/src/backends/kubernetes/identity.ts +261 -0
  87. package/src/backends/kubernetes/index.ts +1785 -127
  88. package/src/backends/kubernetes/ingress-policy.ts +1344 -0
  89. package/src/backends/kubernetes/k8s-client.ts +444 -54
  90. package/src/backends/kubernetes/lease.ts +75 -19
  91. package/src/backends/kubernetes/objects.ts +626 -6
  92. package/src/backends/kubernetes/per-sandbox-policy.ts +542 -0
  93. package/src/backends/kubernetes/rbac.ts +192 -0
  94. package/src/backends/kubernetes/sandbox.ts +218 -20
  95. package/src/backends/kubernetes/transport.ts +2733 -124
  96. package/src/backends/kubernetes/workspace.ts +4476 -222
  97. package/src/backends/remote-execution-controller.ts +14 -0
  98. package/src/index.ts +595 -14
  99. package/src/testing/sandbox-conformance.ts +540 -5
@@ -0,0 +1,542 @@
1
+ /**
2
+ * Per-sandbox egress: one policy object per LIVE sandbox, written by this
3
+ * host while the sandbox runs, so `Sandbox.setNetworkPolicy` means something
4
+ * here instead of being omitted.
5
+ *
6
+ * ## Why this is a separate file from `egress-policy.ts`
7
+ *
8
+ * That one is about the boundary an OPERATOR applies and this backend only
9
+ * ever reads: one object per backend, translated from config, verified and
10
+ * never written. This one is about an object this backend CREATES, replaces
11
+ * and lets the cluster collect — a different lifecycle, a different RBAC
12
+ * grant, and the only place in `@namzu/sandbox` that writes a policy at all.
13
+ * They share the manifest builder and the read-back comparator, which is the
14
+ * point: two spellings of "what a namzu egress policy looks like" would be
15
+ * one comparing itself against the other's shape.
16
+ *
17
+ * ## The shape
18
+ *
19
+ * - **Name** `namzu-sbx-<uid>`, where the uid is the object the acquire
20
+ * created for this sandbox — the `SandboxClaim` on the pooled path, the
21
+ * `Sandbox` on the pool-less one. Unique per sandbox by construction, and
22
+ * derivable from the owner reference alone, which is what lets the
23
+ * admission policy check the two against each other.
24
+ * - **Selector** one per-sandbox pod label, composed by
25
+ * `composeAdditionalPodLabels` — the ONE composer — and carried onto the
26
+ * pod as claim-time metadata before the pod exists. Not patched onto a
27
+ * running pod: this backend holds no write verb on pods, and a label added
28
+ * after the fact would leave a window in which the policy selected nothing.
29
+ * - **Owner** an `ownerReferences` entry naming that same object, so
30
+ * `destroy()` deletes the claim and the cluster's garbage collector
31
+ * removes the policy. This backend issues no deletion of its own on
32
+ * teardown, which is what keeps a crashed host from leaking policies.
33
+ * - **Rules** the DNS-visibility rule plus one `toFQDNs` rule per allowed
34
+ * host — `SandboxNetworkPolicy.allowedHosts`'s own grammar, where
35
+ * `.example.com` means the domain and its subdomains. A `.domain` entry
36
+ * and `narrowing.tlsServerNames` are refused TOGETHER: a TLS server name
37
+ * is one exact SNI value and an expanded entry is a name plus a pattern,
38
+ * so no single value means both and the emitted policy would deny the
39
+ * domain it claims to allow.
40
+ *
41
+ * ## The fence is not a nicety
42
+ *
43
+ * Writing these needs `create`/`patch`/`delete` on the namespace's
44
+ * `ciliumnetworkpolicies`, and RBAC has no way to say "only the objects you
45
+ * own". A host holding those verbs could widen — or simply delete — the
46
+ * operator's own baseline policy. So before the FIRST write, this module
47
+ * proves an operator applied a `ValidatingAdmissionPolicy` and its binding,
48
+ * and refuses with {@link KubernetesAdmissionFenceMissingError}, having
49
+ * written nothing, if either is absent. The shipped example
50
+ * (`k8s/manifests/validatingadmissionpolicy-cilium.yaml`) bounds the host
51
+ * identity to this name prefix, an owner reference whose uid the name has to
52
+ * match, ONE selector label whose VALUE is that owner's own name (the shape
53
+ * alone would not bound anything — one label is exactly what selects every
54
+ * sandbox pod in the namespace), `toFQDNs` entries that each name something
55
+ * (`matchPattern: '*'` is a `toFQDNs` rule and is not an allowlist), the
56
+ * kube-dns rule on port 53, no address-based or entity-based peers and no
57
+ * ingress.
58
+ *
59
+ * ## Union, not replacement
60
+ *
61
+ * The cluster unions every policy selecting a pod, so what this writes ADDS
62
+ * to whatever `config.egress.policy` translated to. Under a `no-network` or
63
+ * `deny-all` baseline that makes the per-sandbox list the pod's whole
64
+ * boundary, which is the deployment this capability is for; under
65
+ * `allow-all` it adds nothing to a pod that could already reach everything,
66
+ * and `setNetworkPolicy([])` there does not deny everything the way
67
+ * `SandboxNetworkPolicy` describes. Nothing refuses that wiring — a
68
+ * permissive baseline with per-sandbox additions is a legitimate deployment
69
+ * — but it is not the one the SDK's sentence is about.
70
+ *
71
+ * ## What is NOT proven anywhere in this repository
72
+ *
73
+ * That any of it is ENFORCED. Enforcement is one CNI's data plane, and the
74
+ * only cluster available here runs none — every object below is proved
75
+ * against a fake API server, and its lifecycle, ownership, garbage collection
76
+ * and admission refusals against a local single-node cluster. "The allowed
77
+ * host answers and the disallowed one does not" is a statement about a
78
+ * network, and it needs a cluster that enforces plus a positive control.
79
+ */
80
+
81
+ import type { SandboxNetworkPolicy } from '@namzu/sdk'
82
+
83
+ import {
84
+ type KubernetesEgressConfig,
85
+ KubernetesNetworkPolicyHostError,
86
+ type KubernetesPerSandboxEgressConfig,
87
+ type KubernetesTranslatedEgressPolicy,
88
+ PER_SANDBOX_NARROWING_REFUSAL,
89
+ assertHostsFitNarrowing,
90
+ buildCiliumEgressManifest,
91
+ perSandboxEgressLabelKey,
92
+ verifyEgressPolicyApplied,
93
+ } from './egress-policy.js'
94
+ import {
95
+ KubernetesAlreadyGoneError,
96
+ type KubernetesClient,
97
+ KubernetesConflictError,
98
+ KubernetesCredentialError,
99
+ } from './k8s-client.js'
100
+ import {
101
+ type KubernetesOwnerReference,
102
+ SANDBOX_API_GROUP,
103
+ SANDBOX_API_VERSION,
104
+ SANDBOX_EXTENSIONS_API_GROUP,
105
+ ciliumNetworkPolicyCollectionPath,
106
+ ciliumNetworkPolicyPath,
107
+ validatingAdmissionPolicyBindingPath,
108
+ validatingAdmissionPolicyPath,
109
+ } from './objects.js'
110
+
111
+ /**
112
+ * Every per-sandbox policy's name starts with this, and the shipped admission
113
+ * policy refuses one that does not.
114
+ *
115
+ * It is a CONSTANT rather than an option because it is half of a contract
116
+ * with an object an operator applies: a configurable prefix would be a
117
+ * configuration the fence could not know about, and a fence that trusted the
118
+ * host to tell it what to allow would not be one.
119
+ */
120
+ export const PER_SANDBOX_POLICY_NAME_PREFIX = 'namzu-sbx-'
121
+
122
+ /** `namzu-sbx-<uid>` — the only name this backend ever writes a policy under. */
123
+ export function perSandboxPolicyName(ownerUid: string): string {
124
+ return `${PER_SANDBOX_POLICY_NAME_PREFIX}${ownerUid}`
125
+ }
126
+
127
+ /**
128
+ * The object a per-sandbox policy belongs to: what the acquire created, which
129
+ * is what the cluster deletes when the sandbox is destroyed.
130
+ *
131
+ * Two kinds, because this backend has two acquire paths and both must be able
132
+ * to carry the capability: a `SandboxClaim` when the sandbox came out of a
133
+ * warm pool, the `Sandbox` itself when it did not. The claim is the one the
134
+ * issue names; the direct object works identically for garbage collection,
135
+ * and refusing the pool-less path would be refusing it for a reason that is
136
+ * about the pool rather than about the policy.
137
+ */
138
+ export interface PerSandboxPolicyOwner {
139
+ readonly kind: 'SandboxClaim' | 'Sandbox'
140
+ readonly name: string
141
+ /** `metadata.uid`, as the API server assigned it. Also the policy's name suffix. */
142
+ readonly uid: string
143
+ }
144
+
145
+ /** The `ownerReferences` entry that makes the cluster collect the policy. */
146
+ export function perSandboxPolicyOwnerReference(
147
+ owner: PerSandboxPolicyOwner,
148
+ ): KubernetesOwnerReference {
149
+ return {
150
+ apiVersion: `${
151
+ owner.kind === 'SandboxClaim' ? SANDBOX_EXTENSIONS_API_GROUP : SANDBOX_API_GROUP
152
+ }/${SANDBOX_API_VERSION}`,
153
+ kind: owner.kind,
154
+ name: owner.name,
155
+ uid: owner.uid,
156
+ }
157
+ }
158
+
159
+ /**
160
+ * Raised when the fence check is REFUSED rather than answered — a 401/403 on
161
+ * one of the two cluster-scoped admission objects.
162
+ *
163
+ * Its own class rather than a reuse of
164
+ * {@link KubernetesAdmissionFenceMissingError}, because the two ask for
165
+ * different actions and conflating them would send an operator to apply a
166
+ * fence that may already be there: 404 means the object does not exist, 403
167
+ * means this host may not look. The most common cause is applying
168
+ * `rbac-per-sandbox-egress.yaml`'s namespaced Role without the ClusterRole in
169
+ * the same file, so the message names that file rather than the manifest the
170
+ * missing-fence error names.
171
+ */
172
+ export class KubernetesAdmissionFenceUnreadableError extends Error {
173
+ override readonly name = 'KubernetesAdmissionFenceUnreadableError'
174
+
175
+ constructor(
176
+ readonly resourceKind: 'ValidatingAdmissionPolicy' | 'ValidatingAdmissionPolicyBinding',
177
+ readonly objectName: string,
178
+ readonly path: string,
179
+ cause: unknown,
180
+ ) {
181
+ super(
182
+ `kubernetes: config.egress.perSandbox is configured but this host may not READ the ${resourceKind} named ${objectName} (${path}), so it cannot prove the fence that bounds what it writes. Both admission objects are cluster-scoped: the namespaced Role in k8s/manifests/rbac-per-sandbox-egress.yaml is not enough on its own, and the ClusterRole and ClusterRoleBinding in that same file are what grant the read. Nothing was written.`,
183
+ { cause },
184
+ )
185
+ }
186
+ }
187
+
188
+ /**
189
+ * Raised when the admission fence is not in place — the policy, its binding,
190
+ * or both — and therefore before anything has been written.
191
+ *
192
+ * Refusing rather than proceeding is the whole design: the fence is what
193
+ * makes the write verbs safe to hold, so a deployment that granted them and
194
+ * has not applied the fence is in exactly the state the capability exists to
195
+ * avoid. A `ValidatingAdmissionPolicy` with no BINDING is inert — it validates
196
+ * nothing at all — which is why the binding is checked separately rather than
197
+ * assumed from the policy's existence.
198
+ */
199
+ export class KubernetesAdmissionFenceMissingError extends Error {
200
+ override readonly name = 'KubernetesAdmissionFenceMissingError'
201
+
202
+ constructor(
203
+ readonly resourceKind: 'ValidatingAdmissionPolicy' | 'ValidatingAdmissionPolicyBinding',
204
+ readonly objectName: string,
205
+ readonly path: string,
206
+ ) {
207
+ super(
208
+ `kubernetes: config.egress.perSandbox is configured but no ${resourceKind} named ${objectName} exists (${path}), so nothing would bound what this host writes to the namespace's CiliumNetworkPolicies. Apply k8s/manifests/validatingadmissionpolicy-cilium.yaml (and its binding) before calling setNetworkPolicy — the host holds create/patch/delete on every policy in the namespace, and the fence is what limits that to objects named ${PER_SANDBOX_POLICY_NAME_PREFIX}<owner uid>, selected by one label, carrying only toFQDNs and the cluster-DNS rule. Nothing was written.`,
209
+ )
210
+ }
211
+ }
212
+
213
+ /**
214
+ * Raised for an `allowedHosts` entry this backend will not translate —
215
+ * re-exported rather than defined here.
216
+ *
217
+ * It lives in `egress-policy.ts` beside {@link assertHostsFitNarrowing},
218
+ * because the config-level `ciliumNarrowing` translation raises the same
219
+ * class for the same entry, and this module imports that one: defining it
220
+ * here would make the two modules import each other. The identifier is
221
+ * re-exported so every existing import path — including the package index —
222
+ * keeps resolving.
223
+ */
224
+ export { KubernetesNetworkPolicyHostError }
225
+
226
+ /** A DNS name, lowercase, no scheme, no port, no wildcard. */
227
+ const DNS_NAME = /^[a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*$/
228
+
229
+ /**
230
+ * One `allowedHosts` entry, validated and canonicalised to the bytes a
231
+ * `CiliumNetworkPolicy` should carry.
232
+ *
233
+ * Lowercasing is not leniency: DNS names are case-insensitive, so
234
+ * `API.example.com` and `api.example.com` name the same host, and Cilium's
235
+ * `matchName` is compared against what the DNS proxy saw — lowercase. The
236
+ * docker backend accepts either case, and a list that worked there and threw
237
+ * here would be a portability trap with no boundary behind it.
238
+ */
239
+ function normalizeHost(entry: string): string {
240
+ if (typeof entry !== 'string') {
241
+ throw new KubernetesNetworkPolicyHostError(String(entry), 'is not a string')
242
+ }
243
+ const canonical = entry.toLowerCase()
244
+ assertUsableHost(canonical)
245
+ return canonical
246
+ }
247
+
248
+ function assertUsableHost(entry: string): void {
249
+ if (typeof entry !== 'string' || entry === '') {
250
+ throw new KubernetesNetworkPolicyHostError(String(entry), 'is empty')
251
+ }
252
+ const bare = entry.startsWith('.') ? entry.slice(1) : entry
253
+ if (bare === '') {
254
+ throw new KubernetesNetworkPolicyHostError(entry, 'names no domain after its leading dot')
255
+ }
256
+ if (entry.includes('*')) {
257
+ throw new KubernetesNetworkPolicyHostError(
258
+ entry,
259
+ "contains a glob; a domain and its subdomains are written with a leading dot ('.example.com'), which becomes matchName plus matchPattern",
260
+ )
261
+ }
262
+ if (!DNS_NAME.test(bare)) {
263
+ throw new KubernetesNetworkPolicyHostError(
264
+ entry,
265
+ 'is not a DNS name (a scheme, a path, a port suffix and an IP address all land here; letter case is canonicalised before this check, so it is never the cause)',
266
+ )
267
+ }
268
+ if (bare.length > 253) {
269
+ throw new KubernetesNetworkPolicyHostError(entry, 'is longer than a DNS name may be')
270
+ }
271
+ // A leading-dot entry becomes `matchPattern: '*.<domain>'`, and a
272
+ // single-label domain there is a whole public suffix — `.com`, `.org`.
273
+ // That is not an allowlist entry, and the shipped admission fence refuses
274
+ // the pattern it would produce, so refusing it HERE is what turns an
275
+ // opaque 403 from the API server into an error naming the entry.
276
+ if (entry.startsWith('.') && !bare.includes('.')) {
277
+ throw new KubernetesNetworkPolicyHostError(
278
+ entry,
279
+ "names a whole top-level domain ('.com' means every name under it); a domain entry needs at least two labels, as in '.example.com', and the shipped admission policy refuses the '*.com' pattern this would emit",
280
+ )
281
+ }
282
+ // An IPv4 literal passes the grammar above — every label is digits, and
283
+ // digits are legal in a DNS label. It is still not a hostname: a DNS
284
+ // top-level label is never all-numeric, and Cilium's `matchName` is
285
+ // compared against names the DNS proxy SAW, which an address never is. A
286
+ // policy carrying one is admitted and matches nothing, which reads from
287
+ // outside exactly like a policy that is working.
288
+ const lastLabel = bare.slice(bare.lastIndexOf('.') + 1)
289
+ if (/^[0-9]+$/.test(lastLabel)) {
290
+ throw new KubernetesNetworkPolicyHostError(
291
+ entry,
292
+ 'ends in an all-numeric label, so it is an address rather than a hostname; toFQDNs matches names a DNS lookup returned, and an address is never one of them (use config.egress.policy for address-based egress)',
293
+ )
294
+ }
295
+ }
296
+
297
+ /**
298
+ * Raised when the object an acquire created reported no `metadata.uid`.
299
+ *
300
+ * Its own exported class rather than a bare `Error` because every other
301
+ * refusal this capability adds is catchable by class, and this one refuses an
302
+ * ACQUIRE: a caller that wants to fall back to a sandbox without per-sandbox
303
+ * egress has to be able to tell it from a readiness timeout.
304
+ */
305
+ export class KubernetesOwnerUidMissingError extends Error {
306
+ override readonly name = 'KubernetesOwnerUidMissingError'
307
+
308
+ constructor(
309
+ readonly objectName: string,
310
+ readonly namespace: string,
311
+ ) {
312
+ super(
313
+ `kubernetes: ${objectName} in namespace ${namespace} was created but reported no metadata.uid, in its create reply or in any readiness read. config.egress.perSandbox needs that uid: it is what a per-sandbox CiliumNetworkPolicy names in its ownerReferences (so the cluster deletes the policy with the object) and the suffix of the policy's own name. Refusing rather than handing back a sandbox whose setNetworkPolicy would write an orphan.`,
314
+ )
315
+ }
316
+ }
317
+
318
+ /**
319
+ * The once-per-backend proof that the fence exists.
320
+ *
321
+ * Memoized on SUCCESS only, exactly as the egress named-object check is: a
322
+ * transient API failure must not wedge every later `setNetworkPolicy` behind
323
+ * a stale rejection, and an operator who applies the fence after the first
324
+ * refusal gets the next call through without restarting the host.
325
+ */
326
+ export interface PerSandboxPolicyFence {
327
+ assertApplied(signal?: AbortSignal): Promise<void>
328
+ }
329
+
330
+ export function buildAdmissionFence(
331
+ client: KubernetesClient,
332
+ perSandbox: KubernetesPerSandboxEgressConfig,
333
+ ): PerSandboxPolicyFence {
334
+ const policyName = perSandbox.admissionPolicyName
335
+ const bindingName = perSandbox.admissionPolicyBindingName ?? `${policyName}-binding`
336
+ let applied: Promise<void> | undefined
337
+ const read = async (
338
+ kind: 'ValidatingAdmissionPolicy' | 'ValidatingAdmissionPolicyBinding',
339
+ name: string,
340
+ path: string,
341
+ signal?: AbortSignal,
342
+ ): Promise<void> => {
343
+ try {
344
+ await client.request('GET', path, undefined, signal)
345
+ } catch (err) {
346
+ if (err instanceof KubernetesAlreadyGoneError) {
347
+ throw new KubernetesAdmissionFenceMissingError(kind, name, path)
348
+ }
349
+ // A 401/403 is NOT "the fence is missing": the object may be there
350
+ // and this host simply may not read it. Both refuse the write, and
351
+ // they send an operator to different files.
352
+ if (err instanceof KubernetesCredentialError) {
353
+ throw new KubernetesAdmissionFenceUnreadableError(kind, name, path, err)
354
+ }
355
+ throw err
356
+ }
357
+ }
358
+ return {
359
+ async assertApplied(signal) {
360
+ applied ??= (async () => {
361
+ await read(
362
+ 'ValidatingAdmissionPolicy',
363
+ policyName,
364
+ validatingAdmissionPolicyPath(policyName),
365
+ signal,
366
+ )
367
+ await read(
368
+ 'ValidatingAdmissionPolicyBinding',
369
+ bindingName,
370
+ validatingAdmissionPolicyBindingPath(bindingName),
371
+ signal,
372
+ )
373
+ })().catch((err: unknown) => {
374
+ applied = undefined
375
+ throw err
376
+ })
377
+ await applied
378
+ },
379
+ }
380
+ }
381
+
382
+ /** What one sandbox's policy writer needs to know. */
383
+ export interface PerSandboxPolicySetterOptions {
384
+ readonly client: KubernetesClient
385
+ readonly fence: PerSandboxPolicyFence
386
+ readonly namespace: string
387
+ readonly egress: KubernetesEgressConfig & {
388
+ readonly perSandbox: KubernetesPerSandboxEgressConfig
389
+ }
390
+ /** The object the acquire created, which owns the policy. */
391
+ readonly owner: PerSandboxPolicyOwner
392
+ /**
393
+ * The per-sandbox label's VALUE on this sandbox's pod — the name of the
394
+ * object above, put there by `composeAdditionalPodLabels` and CONFIRMED on
395
+ * the bound pod before the sandbox was admitted. The key comes from
396
+ * config, through the one resolver.
397
+ */
398
+ readonly selectorValue: string
399
+ }
400
+
401
+ /**
402
+ * Build the `setNetworkPolicy` this backend attaches to a sandbox handle.
403
+ *
404
+ * Calls are SERIALIZED per sandbox. Two overlapping calls would otherwise
405
+ * interleave create, replace and read-back against one object, and the loser
406
+ * would resolve having verified the winner's rules — a caller told its policy
407
+ * was applied when a different one is in force is the exact failure the
408
+ * read-back exists to prevent.
409
+ */
410
+ export function buildPerSandboxPolicySetter(
411
+ options: PerSandboxPolicySetterOptions,
412
+ ): (policy: SandboxNetworkPolicy) => Promise<void> {
413
+ const { client, fence, namespace, egress, owner, selectorValue } = options
414
+ const perSandbox = egress.perSandbox
415
+ const labelKey = perSandboxEgressLabelKey(egress)
416
+ if (labelKey === undefined) {
417
+ // Unreachable: the type above requires `perSandbox`, and the resolver
418
+ // returns a key whenever it is present.
419
+ throw new Error('kubernetes: per-sandbox egress is not configured')
420
+ }
421
+ const name = perSandboxPolicyName(owner.uid)
422
+ const path = ciliumNetworkPolicyPath(namespace, name)
423
+ let queue: Promise<void> = Promise.resolve()
424
+
425
+ const translate = (allowedHosts: readonly string[]): KubernetesTranslatedEgressPolicy =>
426
+ buildCiliumEgressManifest({
427
+ namespace,
428
+ name,
429
+ selectorLabels: { [labelKey]: selectorValue },
430
+ allowedHosts,
431
+ // The CONFIGURED kind this stands in for: a host-supplied
432
+ // allowlist is exactly what `'static'` means, and a refusal that
433
+ // named anything else would send a reader to the wrong config key.
434
+ policyKind: 'static',
435
+ ...(perSandbox.narrowing !== undefined ? { narrowing: perSandbox.narrowing } : {}),
436
+ ownerReferences: [perSandboxPolicyOwnerReference(owner)],
437
+ expandDomains: true,
438
+ })
439
+
440
+ const write = async (translated: KubernetesTranslatedEgressPolicy): Promise<void> => {
441
+ const body = translated.manifest
442
+ try {
443
+ await client.request('POST', ciliumNetworkPolicyCollectionPath(namespace), body)
444
+ } catch (err) {
445
+ // The object already exists — this sandbox is narrowing its egress
446
+ // a second time. A merge patch replaces `spec.egress` wholesale
447
+ // (RFC 7386 replaces arrays), which is what a REPLACEMENT policy
448
+ // needs, and the read-back below proves it rather than assuming it.
449
+ if (!(err instanceof KubernetesConflictError)) throw err
450
+ await client.request('PATCH', path, body)
451
+ }
452
+ // Resolve only once the cluster's own copy deep-equals the
453
+ // translation, through the comparator the config-level check already
454
+ // uses. A policy the API server accepted and mutated — a defaulting
455
+ // webhook, an operator's own automation — would otherwise be reported
456
+ // as the policy the caller asked for.
457
+ //
458
+ // One inherited wording to know about: if the object is GONE by the
459
+ // time it is read back, the shared comparator raises
460
+ // `KubernetesEgressPolicyNotAppliedError`, whose message asks an
461
+ // operator to apply the manifest this backend computed. Here that is
462
+ // not the action — the object was written a moment ago, so it being
463
+ // gone means its owner was deleted underneath the call and this
464
+ // sandbox is already being destroyed. The refusal is still right; only
465
+ // its advice belongs to the other caller.
466
+ await verifyEgressPolicyApplied(client, translated)
467
+ }
468
+
469
+ const remove = async (): Promise<void> => {
470
+ try {
471
+ await client.request('DELETE', path)
472
+ } catch (err) {
473
+ // Already gone is the state DELETE was asking for — the cluster
474
+ // may have collected it with the claim while this call was in
475
+ // flight.
476
+ if (!(err instanceof KubernetesAlreadyGoneError)) throw err
477
+ }
478
+ }
479
+
480
+ return async (policy: SandboxNetworkPolicy): Promise<void> => {
481
+ // Validated BEFORE the queue, so a malformed list is refused with
482
+ // nothing written and nothing waited for. `normalizeHost` also
483
+ // lowercases: DNS names are case-insensitive, the docker backend
484
+ // accepts either case, and a `CiliumNetworkPolicy` matches names the
485
+ // resolver returns — which are lowercase. Refusing `API.example.com`
486
+ // on one backend and honouring it on another would be a portability
487
+ // trap, so it is canonicalised here instead.
488
+ const hosts = [...(policy?.allowedHosts ?? [])].map(normalizeHost)
489
+ // And refused here rather than emitted: an entry the CONFIGURED
490
+ // narrowing cannot express is one whose policy would read as applied
491
+ // and deny what it names. The shared guard refuses it again inside the
492
+ // translation, so every caller is covered; this call is the earlier
493
+ // one, before the fence is read and before anything is queued behind a
494
+ // previous call.
495
+ assertHostsFitNarrowing(
496
+ hosts,
497
+ perSandbox.narrowing,
498
+ // The per-sandbox context, from `egress-policy.ts` rather than
499
+ // rebuilt here: it is the SAME value `buildCiliumEgressManifest`
500
+ // refuses by on this path (`expandDomains: true`), so the earlier
501
+ // check and the translation's own cannot name different fields —
502
+ // and the sentence it carries is the true one here, where the entry
503
+ // really does become a name plus a `*.domain` pattern. The
504
+ // config-level wording, where nothing is expanded, would be a lie
505
+ // on this path: leaving the option off here does allow the domain
506
+ // and its subdomains.
507
+ PER_SANDBOX_NARROWING_REFUSAL,
508
+ )
509
+ // A previous call's failure belongs to that caller; this one still runs,
510
+ // in order behind it.
511
+ const run = queue.then(async () => {
512
+ // Before EVERY write — the DELETE below included — and before
513
+ // every one until the fence check succeeds: nothing is sent while
514
+ // the fence is missing.
515
+ //
516
+ // The DELETE is checked too because the plan's invariant is about
517
+ // writes, not about widening: a host that never proved the fence
518
+ // must not reach the namespace's policies with any verb it holds.
519
+ // The cost is a deployment that REMOVES the fence mid-run, whose
520
+ // `setNetworkPolicy([])` is then refused and whose sandbox keeps
521
+ // the wider per-sandbox allowance until it is destroyed and the
522
+ // cluster collects the policy with its owner. That is a loud,
523
+ // named refusal the caller can act on, which is the better half
524
+ // of the trade against a silent unfenced write.
525
+ await fence.assertApplied()
526
+ // An EMPTY list is not "no policy": it deletes this sandbox's own
527
+ // object and leaves the configured baseline — whatever
528
+ // `config.egress.policy` translated to, plus every other policy
529
+ // selecting the pod — in force.
530
+ if (hosts.length === 0) {
531
+ await remove()
532
+ return
533
+ }
534
+ await write(translate(hosts))
535
+ })
536
+ queue = run.then(
537
+ () => undefined,
538
+ () => undefined,
539
+ )
540
+ await run
541
+ }
542
+ }