@namzu/sandbox 14.0.0 → 16.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. package/CHANGELOG.md +924 -0
  2. package/README.md +369 -14
  3. package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
  4. package/dist/backends/aci-standby-pool/index.js +13 -1
  5. package/dist/backends/aci-standby-pool/index.js.map +1 -1
  6. package/dist/backends/docker/index.d.ts +169 -6
  7. package/dist/backends/docker/index.d.ts.map +1 -1
  8. package/dist/backends/docker/index.js +499 -85
  9. package/dist/backends/docker/index.js.map +1 -1
  10. package/dist/backends/firecracker/index.d.ts.map +1 -1
  11. package/dist/backends/firecracker/index.js +12 -2
  12. package/dist/backends/firecracker/index.js.map +1 -1
  13. package/dist/backends/firecracker/protocol.d.ts +459 -8
  14. package/dist/backends/firecracker/protocol.d.ts.map +1 -1
  15. package/dist/backends/firecracker/protocol.js +136 -0
  16. package/dist/backends/firecracker/protocol.js.map +1 -1
  17. package/dist/backends/firecracker/transport.d.ts +539 -6
  18. package/dist/backends/firecracker/transport.d.ts.map +1 -1
  19. package/dist/backends/firecracker/transport.js +1171 -24
  20. package/dist/backends/firecracker/transport.js.map +1 -1
  21. package/dist/backends/kubernetes/egress-policy.d.ts +1181 -13
  22. package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -1
  23. package/dist/backends/kubernetes/egress-policy.js +2350 -31
  24. package/dist/backends/kubernetes/egress-policy.js.map +1 -1
  25. package/dist/backends/kubernetes/identity.d.ts +193 -0
  26. package/dist/backends/kubernetes/identity.d.ts.map +1 -0
  27. package/dist/backends/kubernetes/identity.js +147 -0
  28. package/dist/backends/kubernetes/identity.js.map +1 -0
  29. package/dist/backends/kubernetes/index.d.ts +678 -33
  30. package/dist/backends/kubernetes/index.d.ts.map +1 -1
  31. package/dist/backends/kubernetes/index.js +1180 -95
  32. package/dist/backends/kubernetes/index.js.map +1 -1
  33. package/dist/backends/kubernetes/ingress-policy.d.ts +375 -0
  34. package/dist/backends/kubernetes/ingress-policy.d.ts.map +1 -0
  35. package/dist/backends/kubernetes/ingress-policy.js +1050 -0
  36. package/dist/backends/kubernetes/ingress-policy.js.map +1 -0
  37. package/dist/backends/kubernetes/k8s-client.d.ts +213 -4
  38. package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -1
  39. package/dist/backends/kubernetes/k8s-client.js +359 -52
  40. package/dist/backends/kubernetes/k8s-client.js.map +1 -1
  41. package/dist/backends/kubernetes/lease.d.ts +40 -14
  42. package/dist/backends/kubernetes/lease.d.ts.map +1 -1
  43. package/dist/backends/kubernetes/lease.js +68 -18
  44. package/dist/backends/kubernetes/lease.js.map +1 -1
  45. package/dist/backends/kubernetes/objects.d.ts +423 -3
  46. package/dist/backends/kubernetes/objects.d.ts.map +1 -1
  47. package/dist/backends/kubernetes/objects.js +364 -2
  48. package/dist/backends/kubernetes/objects.js.map +1 -1
  49. package/dist/backends/kubernetes/per-sandbox-policy.d.ts +219 -0
  50. package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -0
  51. package/dist/backends/kubernetes/per-sandbox-policy.js +375 -0
  52. package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -0
  53. package/dist/backends/kubernetes/rbac.d.ts +153 -0
  54. package/dist/backends/kubernetes/rbac.d.ts.map +1 -0
  55. package/dist/backends/kubernetes/rbac.js +177 -0
  56. package/dist/backends/kubernetes/rbac.js.map +1 -0
  57. package/dist/backends/kubernetes/sandbox.d.ts +81 -14
  58. package/dist/backends/kubernetes/sandbox.d.ts.map +1 -1
  59. package/dist/backends/kubernetes/sandbox.js +149 -15
  60. package/dist/backends/kubernetes/sandbox.js.map +1 -1
  61. package/dist/backends/kubernetes/transport.d.ts +935 -9
  62. package/dist/backends/kubernetes/transport.d.ts.map +1 -1
  63. package/dist/backends/kubernetes/transport.js +1958 -62
  64. package/dist/backends/kubernetes/transport.js.map +1 -1
  65. package/dist/backends/kubernetes/workspace.d.ts +1149 -18
  66. package/dist/backends/kubernetes/workspace.d.ts.map +1 -1
  67. package/dist/backends/kubernetes/workspace.js +2825 -186
  68. package/dist/backends/kubernetes/workspace.js.map +1 -1
  69. package/dist/backends/remote-execution-controller.d.ts +14 -0
  70. package/dist/backends/remote-execution-controller.d.ts.map +1 -1
  71. package/dist/backends/remote-execution-controller.js.map +1 -1
  72. package/dist/index.d.ts +294 -18
  73. package/dist/index.d.ts.map +1 -1
  74. package/dist/index.js +280 -10
  75. package/dist/index.js.map +1 -1
  76. package/dist/testing/sandbox-conformance.d.ts +39 -5
  77. package/dist/testing/sandbox-conformance.d.ts.map +1 -1
  78. package/dist/testing/sandbox-conformance.js +436 -5
  79. package/dist/testing/sandbox-conformance.js.map +1 -1
  80. package/package.json +3 -3
  81. package/src/backends/aci-standby-pool/index.ts +16 -1
  82. package/src/backends/docker/index.ts +617 -100
  83. package/src/backends/firecracker/index.ts +14 -2
  84. package/src/backends/firecracker/protocol.ts +514 -6
  85. package/src/backends/firecracker/transport.ts +1492 -40
  86. package/src/backends/kubernetes/egress-policy.ts +3334 -55
  87. package/src/backends/kubernetes/identity.ts +261 -0
  88. package/src/backends/kubernetes/index.ts +1785 -127
  89. package/src/backends/kubernetes/ingress-policy.ts +1344 -0
  90. package/src/backends/kubernetes/k8s-client.ts +444 -54
  91. package/src/backends/kubernetes/lease.ts +75 -19
  92. package/src/backends/kubernetes/objects.ts +626 -6
  93. package/src/backends/kubernetes/per-sandbox-policy.ts +497 -0
  94. package/src/backends/kubernetes/rbac.ts +192 -0
  95. package/src/backends/kubernetes/sandbox.ts +218 -20
  96. package/src/backends/kubernetes/transport.ts +2733 -124
  97. package/src/backends/kubernetes/workspace.ts +4476 -222
  98. package/src/backends/remote-execution-controller.ts +14 -0
  99. package/src/index.ts +668 -19
  100. package/src/testing/sandbox-conformance.ts +540 -5
@@ -0,0 +1,497 @@
1
+ /**
2
+ * Per-sandbox egress: one policy object per LIVE sandbox, written by this
3
+ * host while the sandbox runs, so `Sandbox.setNetworkPolicy` means something
4
+ * here instead of being omitted.
5
+ *
6
+ * ## Why this is a separate file from `egress-policy.ts`
7
+ *
8
+ * That one is about the boundary an OPERATOR applies and this backend only
9
+ * ever reads: one object per backend, translated from config, verified and
10
+ * never written. This one is about an object this backend CREATES, replaces
11
+ * and lets the cluster collect — a different lifecycle, a different RBAC
12
+ * grant, and the only place in `@namzu/sandbox` that writes a policy at all.
13
+ * They share the manifest builder and the read-back comparator, which is the
14
+ * point: two spellings of "what a namzu egress policy looks like" would be
15
+ * one comparing itself against the other's shape.
16
+ *
17
+ * ## The shape
18
+ *
19
+ * - **Name** `namzu-sbx-<uid>`, where the uid is the object the acquire
20
+ * created for this sandbox — the `SandboxClaim` on the pooled path, the
21
+ * `Sandbox` on the pool-less one. Unique per sandbox by construction, and
22
+ * derivable from the owner reference alone, which is what lets the
23
+ * admission policy check the two against each other.
24
+ * - **Selector** one per-sandbox pod label, composed by
25
+ * `composeAdditionalPodLabels` — the ONE composer — and carried onto the
26
+ * pod as claim-time metadata before the pod exists. Not patched onto a
27
+ * running pod: this backend holds no write verb on pods, and a label added
28
+ * after the fact would leave a window in which the policy selected nothing.
29
+ * - **Owner** an `ownerReferences` entry naming that same object, so
30
+ * `destroy()` deletes the claim and the cluster's garbage collector
31
+ * removes the policy. This backend issues no deletion of its own on
32
+ * teardown, which is what keeps a crashed host from leaking policies.
33
+ * - **Rules** the DNS-visibility rule plus one `toFQDNs` rule per allowed
34
+ * host — `SandboxNetworkPolicy.allowedHosts`'s own grammar, where
35
+ * `.example.com` means the domain and its subdomains. A `.domain` entry
36
+ * and `narrowing.tlsServerNames` are refused TOGETHER: a TLS server name
37
+ * is one exact SNI value and an expanded entry is a name plus a pattern,
38
+ * so no single value means both and the emitted policy would deny the
39
+ * domain it claims to allow.
40
+ *
41
+ * ## The fence is not a nicety
42
+ *
43
+ * Writing these needs `create`/`patch`/`delete` on the namespace's
44
+ * `ciliumnetworkpolicies`, and RBAC has no way to say "only the objects you
45
+ * own". A host holding those verbs could widen — or simply delete — the
46
+ * operator's own baseline policy. So before the FIRST write, this module
47
+ * proves an operator applied a `ValidatingAdmissionPolicy` and its binding,
48
+ * and refuses with {@link KubernetesAdmissionFenceMissingError}, having
49
+ * written nothing, if either is absent. The shipped example
50
+ * (`k8s/manifests/validatingadmissionpolicy-cilium.yaml`) bounds the host
51
+ * identity to this name prefix, an owner reference whose uid the name has to
52
+ * match, ONE selector label whose VALUE is that owner's own name (the shape
53
+ * alone would not bound anything — one label is exactly what selects every
54
+ * sandbox pod in the namespace), `toFQDNs` entries that each name something
55
+ * (`matchPattern: '*'` is a `toFQDNs` rule and is not an allowlist), the
56
+ * kube-dns rule on port 53, no address-based or entity-based peers and no
57
+ * ingress.
58
+ *
59
+ * ## Union, not replacement
60
+ *
61
+ * The cluster unions every policy selecting a pod, so what this writes ADDS
62
+ * to whatever `config.egress.policy` translated to. Under a `no-network` or
63
+ * `deny-all` baseline that makes the per-sandbox list the pod's whole
64
+ * boundary, which is the deployment this capability is for; under
65
+ * `allow-all` it adds nothing to a pod that could already reach everything,
66
+ * and `setNetworkPolicy([])` there does not deny everything the way
67
+ * `SandboxNetworkPolicy` describes. Nothing refuses that wiring — a
68
+ * permissive baseline with per-sandbox additions is a legitimate deployment
69
+ * — but it is not the one the SDK's sentence is about.
70
+ *
71
+ * ## What is NOT proven anywhere in this repository
72
+ *
73
+ * That any of it is ENFORCED. Enforcement is one CNI's data plane, and the
74
+ * only cluster available here runs none — every object below is proved
75
+ * against a fake API server, and its lifecycle, ownership, garbage collection
76
+ * and admission refusals against a local single-node cluster. "The allowed
77
+ * host answers and the disallowed one does not" is a statement about a
78
+ * network, and it needs a cluster that enforces plus a positive control.
79
+ */
80
+
81
+ import type { SandboxNetworkPolicy } from '@namzu/sdk'
82
+
83
+ import {
84
+ type KubernetesEgressConfig,
85
+ KubernetesNetworkPolicyHostError,
86
+ type KubernetesPerSandboxEgressConfig,
87
+ type KubernetesTranslatedEgressPolicy,
88
+ PER_SANDBOX_NARROWING_REFUSAL,
89
+ assertHostsFitNarrowing,
90
+ assertUsableHost,
91
+ buildCiliumEgressManifest,
92
+ perSandboxEgressLabelKey,
93
+ verifyEgressPolicyApplied,
94
+ } from './egress-policy.js'
95
+ import {
96
+ KubernetesAlreadyGoneError,
97
+ type KubernetesClient,
98
+ KubernetesConflictError,
99
+ KubernetesCredentialError,
100
+ } from './k8s-client.js'
101
+ import {
102
+ type KubernetesOwnerReference,
103
+ SANDBOX_API_GROUP,
104
+ SANDBOX_API_VERSION,
105
+ SANDBOX_EXTENSIONS_API_GROUP,
106
+ ciliumNetworkPolicyCollectionPath,
107
+ ciliumNetworkPolicyPath,
108
+ validatingAdmissionPolicyBindingPath,
109
+ validatingAdmissionPolicyPath,
110
+ } from './objects.js'
111
+
112
+ /**
113
+ * Every per-sandbox policy's name starts with this, and the shipped admission
114
+ * policy refuses one that does not.
115
+ *
116
+ * It is a CONSTANT rather than an option because it is half of a contract
117
+ * with an object an operator applies: a configurable prefix would be a
118
+ * configuration the fence could not know about, and a fence that trusted the
119
+ * host to tell it what to allow would not be one.
120
+ */
121
+ export const PER_SANDBOX_POLICY_NAME_PREFIX = 'namzu-sbx-'
122
+
123
+ /** `namzu-sbx-<uid>` — the only name this backend ever writes a policy under. */
124
+ export function perSandboxPolicyName(ownerUid: string): string {
125
+ return `${PER_SANDBOX_POLICY_NAME_PREFIX}${ownerUid}`
126
+ }
127
+
128
+ /**
129
+ * The object a per-sandbox policy belongs to: what the acquire created, which
130
+ * is what the cluster deletes when the sandbox is destroyed.
131
+ *
132
+ * Two kinds, because this backend has two acquire paths and both must be able
133
+ * to carry the capability: a `SandboxClaim` when the sandbox came out of a
134
+ * warm pool, the `Sandbox` itself when it did not. The claim is the one the
135
+ * issue names; the direct object works identically for garbage collection,
136
+ * and refusing the pool-less path would be refusing it for a reason that is
137
+ * about the pool rather than about the policy.
138
+ */
139
+ export interface PerSandboxPolicyOwner {
140
+ readonly kind: 'SandboxClaim' | 'Sandbox'
141
+ readonly name: string
142
+ /** `metadata.uid`, as the API server assigned it. Also the policy's name suffix. */
143
+ readonly uid: string
144
+ }
145
+
146
+ /** The `ownerReferences` entry that makes the cluster collect the policy. */
147
+ export function perSandboxPolicyOwnerReference(
148
+ owner: PerSandboxPolicyOwner,
149
+ ): KubernetesOwnerReference {
150
+ return {
151
+ apiVersion: `${
152
+ owner.kind === 'SandboxClaim' ? SANDBOX_EXTENSIONS_API_GROUP : SANDBOX_API_GROUP
153
+ }/${SANDBOX_API_VERSION}`,
154
+ kind: owner.kind,
155
+ name: owner.name,
156
+ uid: owner.uid,
157
+ }
158
+ }
159
+
160
+ /**
161
+ * Raised when the fence check is REFUSED rather than answered — a 401/403 on
162
+ * one of the two cluster-scoped admission objects.
163
+ *
164
+ * Its own class rather than a reuse of
165
+ * {@link KubernetesAdmissionFenceMissingError}, because the two ask for
166
+ * different actions and conflating them would send an operator to apply a
167
+ * fence that may already be there: 404 means the object does not exist, 403
168
+ * means this host may not look. The most common cause is applying
169
+ * `rbac-per-sandbox-egress.yaml`'s namespaced Role without the ClusterRole in
170
+ * the same file, so the message names that file rather than the manifest the
171
+ * missing-fence error names.
172
+ */
173
+ export class KubernetesAdmissionFenceUnreadableError extends Error {
174
+ override readonly name = 'KubernetesAdmissionFenceUnreadableError'
175
+
176
+ constructor(
177
+ readonly resourceKind: 'ValidatingAdmissionPolicy' | 'ValidatingAdmissionPolicyBinding',
178
+ readonly objectName: string,
179
+ readonly path: string,
180
+ cause: unknown,
181
+ ) {
182
+ super(
183
+ `kubernetes: config.egress.perSandbox is configured but this host may not READ the ${resourceKind} named ${objectName} (${path}), so it cannot prove the fence that bounds what it writes. Both admission objects are cluster-scoped: the namespaced Role in k8s/manifests/rbac-per-sandbox-egress.yaml is not enough on its own, and the ClusterRole and ClusterRoleBinding in that same file are what grant the read. Nothing was written.`,
184
+ { cause },
185
+ )
186
+ }
187
+ }
188
+
189
+ /**
190
+ * Raised when the admission fence is not in place — the policy, its binding,
191
+ * or both — and therefore before anything has been written.
192
+ *
193
+ * Refusing rather than proceeding is the whole design: the fence is what
194
+ * makes the write verbs safe to hold, so a deployment that granted them and
195
+ * has not applied the fence is in exactly the state the capability exists to
196
+ * avoid. A `ValidatingAdmissionPolicy` with no BINDING is inert — it validates
197
+ * nothing at all — which is why the binding is checked separately rather than
198
+ * assumed from the policy's existence.
199
+ */
200
+ export class KubernetesAdmissionFenceMissingError extends Error {
201
+ override readonly name = 'KubernetesAdmissionFenceMissingError'
202
+
203
+ constructor(
204
+ readonly resourceKind: 'ValidatingAdmissionPolicy' | 'ValidatingAdmissionPolicyBinding',
205
+ readonly objectName: string,
206
+ readonly path: string,
207
+ ) {
208
+ super(
209
+ `kubernetes: config.egress.perSandbox is configured but no ${resourceKind} named ${objectName} exists (${path}), so nothing would bound what this host writes to the namespace's CiliumNetworkPolicies. Apply k8s/manifests/validatingadmissionpolicy-cilium.yaml (and its binding) before calling setNetworkPolicy — the host holds create/patch/delete on every policy in the namespace, and the fence is what limits that to objects named ${PER_SANDBOX_POLICY_NAME_PREFIX}<owner uid>, selected by one label, carrying only toFQDNs and the cluster-DNS rule. Nothing was written.`,
210
+ )
211
+ }
212
+ }
213
+
214
+ /**
215
+ * Raised for an `allowedHosts` entry this backend will not translate —
216
+ * re-exported rather than defined here.
217
+ *
218
+ * It lives in `egress-policy.ts` beside {@link assertHostsFitNarrowing} and
219
+ * {@link assertUsableHost}, because the config-level translation raises the
220
+ * same class for the same entry and this module imports both checks from
221
+ * there: defining it here would make the two modules import each other. The
222
+ * identifier is re-exported so every existing import path — including the
223
+ * package index — keeps resolving.
224
+ */
225
+ export { KubernetesNetworkPolicyHostError }
226
+
227
+ /**
228
+ * One `allowedHosts` entry, validated and canonicalised to the bytes a
229
+ * `CiliumNetworkPolicy` should carry.
230
+ *
231
+ * Lowercasing is not leniency: DNS names are case-insensitive, so
232
+ * `API.example.com` and `api.example.com` name the same host, and Cilium's
233
+ * `matchName` is compared against what the DNS proxy saw — lowercase. The
234
+ * docker backend accepts either case, and a list that worked there and threw
235
+ * here would be a portability trap with no boundary behind it.
236
+ *
237
+ * The grammar itself is {@link assertUsableHost}'s, in `egress-policy.ts`
238
+ * beside the error it throws, because it is no longer only this writer's: the
239
+ * shared translation calls it too, so a config-level allowlist is refused the
240
+ * same entries this path refuses. What is HERE is the part that is this
241
+ * writer's alone — the canonicalisation, applied before the fence is read and
242
+ * before anything is queued behind a previous call.
243
+ */
244
+ function normalizeHost(entry: string): string {
245
+ if (typeof entry !== 'string') {
246
+ throw new KubernetesNetworkPolicyHostError(String(entry), 'is not a string')
247
+ }
248
+ const canonical = entry.toLowerCase()
249
+ assertUsableHost(canonical)
250
+ return canonical
251
+ }
252
+
253
+ /**
254
+ * Raised when the object an acquire created reported no `metadata.uid`.
255
+ *
256
+ * Its own exported class rather than a bare `Error` because every other
257
+ * refusal this capability adds is catchable by class, and this one refuses an
258
+ * ACQUIRE: a caller that wants to fall back to a sandbox without per-sandbox
259
+ * egress has to be able to tell it from a readiness timeout.
260
+ */
261
+ export class KubernetesOwnerUidMissingError extends Error {
262
+ override readonly name = 'KubernetesOwnerUidMissingError'
263
+
264
+ constructor(
265
+ readonly objectName: string,
266
+ readonly namespace: string,
267
+ ) {
268
+ super(
269
+ `kubernetes: ${objectName} in namespace ${namespace} was created but reported no metadata.uid, in its create reply or in any readiness read. config.egress.perSandbox needs that uid: it is what a per-sandbox CiliumNetworkPolicy names in its ownerReferences (so the cluster deletes the policy with the object) and the suffix of the policy's own name. Refusing rather than handing back a sandbox whose setNetworkPolicy would write an orphan.`,
270
+ )
271
+ }
272
+ }
273
+
274
+ /**
275
+ * The once-per-backend proof that the fence exists.
276
+ *
277
+ * Memoized on SUCCESS only, exactly as the egress named-object check is: a
278
+ * transient API failure must not wedge every later `setNetworkPolicy` behind
279
+ * a stale rejection, and an operator who applies the fence after the first
280
+ * refusal gets the next call through without restarting the host.
281
+ */
282
+ export interface PerSandboxPolicyFence {
283
+ assertApplied(signal?: AbortSignal): Promise<void>
284
+ }
285
+
286
+ export function buildAdmissionFence(
287
+ client: KubernetesClient,
288
+ perSandbox: KubernetesPerSandboxEgressConfig,
289
+ ): PerSandboxPolicyFence {
290
+ const policyName = perSandbox.admissionPolicyName
291
+ const bindingName = perSandbox.admissionPolicyBindingName ?? `${policyName}-binding`
292
+ let applied: Promise<void> | undefined
293
+ const read = async (
294
+ kind: 'ValidatingAdmissionPolicy' | 'ValidatingAdmissionPolicyBinding',
295
+ name: string,
296
+ path: string,
297
+ signal?: AbortSignal,
298
+ ): Promise<void> => {
299
+ try {
300
+ await client.request('GET', path, undefined, signal)
301
+ } catch (err) {
302
+ if (err instanceof KubernetesAlreadyGoneError) {
303
+ throw new KubernetesAdmissionFenceMissingError(kind, name, path)
304
+ }
305
+ // A 401/403 is NOT "the fence is missing": the object may be there
306
+ // and this host simply may not read it. Both refuse the write, and
307
+ // they send an operator to different files.
308
+ if (err instanceof KubernetesCredentialError) {
309
+ throw new KubernetesAdmissionFenceUnreadableError(kind, name, path, err)
310
+ }
311
+ throw err
312
+ }
313
+ }
314
+ return {
315
+ async assertApplied(signal) {
316
+ applied ??= (async () => {
317
+ await read(
318
+ 'ValidatingAdmissionPolicy',
319
+ policyName,
320
+ validatingAdmissionPolicyPath(policyName),
321
+ signal,
322
+ )
323
+ await read(
324
+ 'ValidatingAdmissionPolicyBinding',
325
+ bindingName,
326
+ validatingAdmissionPolicyBindingPath(bindingName),
327
+ signal,
328
+ )
329
+ })().catch((err: unknown) => {
330
+ applied = undefined
331
+ throw err
332
+ })
333
+ await applied
334
+ },
335
+ }
336
+ }
337
+
338
+ /** What one sandbox's policy writer needs to know. */
339
+ export interface PerSandboxPolicySetterOptions {
340
+ readonly client: KubernetesClient
341
+ readonly fence: PerSandboxPolicyFence
342
+ readonly namespace: string
343
+ readonly egress: KubernetesEgressConfig & {
344
+ readonly perSandbox: KubernetesPerSandboxEgressConfig
345
+ }
346
+ /** The object the acquire created, which owns the policy. */
347
+ readonly owner: PerSandboxPolicyOwner
348
+ /**
349
+ * The per-sandbox label's VALUE on this sandbox's pod — the name of the
350
+ * object above, put there by `composeAdditionalPodLabels` and CONFIRMED on
351
+ * the bound pod before the sandbox was admitted. The key comes from
352
+ * config, through the one resolver.
353
+ */
354
+ readonly selectorValue: string
355
+ }
356
+
357
+ /**
358
+ * Build the `setNetworkPolicy` this backend attaches to a sandbox handle.
359
+ *
360
+ * Calls are SERIALIZED per sandbox. Two overlapping calls would otherwise
361
+ * interleave create, replace and read-back against one object, and the loser
362
+ * would resolve having verified the winner's rules — a caller told its policy
363
+ * was applied when a different one is in force is the exact failure the
364
+ * read-back exists to prevent.
365
+ */
366
+ export function buildPerSandboxPolicySetter(
367
+ options: PerSandboxPolicySetterOptions,
368
+ ): (policy: SandboxNetworkPolicy) => Promise<void> {
369
+ const { client, fence, namespace, egress, owner, selectorValue } = options
370
+ const perSandbox = egress.perSandbox
371
+ const labelKey = perSandboxEgressLabelKey(egress)
372
+ if (labelKey === undefined) {
373
+ // Unreachable: the type above requires `perSandbox`, and the resolver
374
+ // returns a key whenever it is present.
375
+ throw new Error('kubernetes: per-sandbox egress is not configured')
376
+ }
377
+ const name = perSandboxPolicyName(owner.uid)
378
+ const path = ciliumNetworkPolicyPath(namespace, name)
379
+ let queue: Promise<void> = Promise.resolve()
380
+
381
+ const translate = (allowedHosts: readonly string[]): KubernetesTranslatedEgressPolicy =>
382
+ buildCiliumEgressManifest({
383
+ namespace,
384
+ name,
385
+ selectorLabels: { [labelKey]: selectorValue },
386
+ allowedHosts,
387
+ // The CONFIGURED kind this stands in for: a host-supplied
388
+ // allowlist is exactly what `'static'` means, and a refusal that
389
+ // named anything else would send a reader to the wrong config key.
390
+ policyKind: 'static',
391
+ ...(perSandbox.narrowing !== undefined ? { narrowing: perSandbox.narrowing } : {}),
392
+ ownerReferences: [perSandboxPolicyOwnerReference(owner)],
393
+ refusalContext: PER_SANDBOX_NARROWING_REFUSAL,
394
+ })
395
+
396
+ const write = async (translated: KubernetesTranslatedEgressPolicy): Promise<void> => {
397
+ const body = translated.manifest
398
+ try {
399
+ await client.request('POST', ciliumNetworkPolicyCollectionPath(namespace), body)
400
+ } catch (err) {
401
+ // The object already exists — this sandbox is narrowing its egress
402
+ // a second time. A merge patch replaces `spec.egress` wholesale
403
+ // (RFC 7386 replaces arrays), which is what a REPLACEMENT policy
404
+ // needs, and the read-back below proves it rather than assuming it.
405
+ if (!(err instanceof KubernetesConflictError)) throw err
406
+ await client.request('PATCH', path, body)
407
+ }
408
+ // Resolve only once the cluster's own copy deep-equals the
409
+ // translation, through the comparator the config-level check already
410
+ // uses. A policy the API server accepted and mutated — a defaulting
411
+ // webhook, an operator's own automation — would otherwise be reported
412
+ // as the policy the caller asked for.
413
+ //
414
+ // One inherited wording to know about: if the object is GONE by the
415
+ // time it is read back, the shared comparator raises
416
+ // `KubernetesEgressPolicyNotAppliedError`, whose message asks an
417
+ // operator to apply the manifest this backend computed. Here that is
418
+ // not the action — the object was written a moment ago, so it being
419
+ // gone means its owner was deleted underneath the call and this
420
+ // sandbox is already being destroyed. The refusal is still right; only
421
+ // its advice belongs to the other caller.
422
+ await verifyEgressPolicyApplied(client, translated)
423
+ }
424
+
425
+ const remove = async (): Promise<void> => {
426
+ try {
427
+ await client.request('DELETE', path)
428
+ } catch (err) {
429
+ // Already gone is the state DELETE was asking for — the cluster
430
+ // may have collected it with the claim while this call was in
431
+ // flight.
432
+ if (!(err instanceof KubernetesAlreadyGoneError)) throw err
433
+ }
434
+ }
435
+
436
+ return async (policy: SandboxNetworkPolicy): Promise<void> => {
437
+ // Validated BEFORE the queue, so a malformed list is refused with
438
+ // nothing written and nothing waited for. `normalizeHost` also
439
+ // lowercases: DNS names are case-insensitive, the docker backend
440
+ // accepts either case, and a `CiliumNetworkPolicy` matches names the
441
+ // resolver returns — which are lowercase. Refusing `API.example.com`
442
+ // on one backend and honouring it on another would be a portability
443
+ // trap, so it is canonicalised here instead.
444
+ const hosts = [...(policy?.allowedHosts ?? [])].map(normalizeHost)
445
+ // And refused here rather than emitted: an entry the CONFIGURED
446
+ // narrowing cannot express is one whose policy would read as applied
447
+ // and deny what it names. The shared guard refuses it again inside the
448
+ // translation, so every caller is covered; this call is the earlier
449
+ // one, before the fence is read and before anything is queued behind a
450
+ // previous call.
451
+ assertHostsFitNarrowing(
452
+ hosts,
453
+ perSandbox.narrowing,
454
+ // The per-sandbox context, from `egress-policy.ts` rather than
455
+ // rebuilt here: it is the SAME value `buildCiliumEgressManifest`
456
+ // refuses by on this path (`refusalContext`), so the earlier check
457
+ // and the translation's own cannot name different fields — and the
458
+ // sentence it carries is the true one here, where the entry really
459
+ // does become a name plus a `*.domain` pattern. Sending this caller
460
+ // to `config.egress.ciliumNarrowing` would name a field a
461
+ // `perSandbox` host never set.
462
+ PER_SANDBOX_NARROWING_REFUSAL,
463
+ )
464
+ // A previous call's failure belongs to that caller; this one still runs,
465
+ // in order behind it.
466
+ const run = queue.then(async () => {
467
+ // Before EVERY write — the DELETE below included — and before
468
+ // every one until the fence check succeeds: nothing is sent while
469
+ // the fence is missing.
470
+ //
471
+ // The DELETE is checked too because the plan's invariant is about
472
+ // writes, not about widening: a host that never proved the fence
473
+ // must not reach the namespace's policies with any verb it holds.
474
+ // The cost is a deployment that REMOVES the fence mid-run, whose
475
+ // `setNetworkPolicy([])` is then refused and whose sandbox keeps
476
+ // the wider per-sandbox allowance until it is destroyed and the
477
+ // cluster collects the policy with its owner. That is a loud,
478
+ // named refusal the caller can act on, which is the better half
479
+ // of the trade against a silent unfenced write.
480
+ await fence.assertApplied()
481
+ // An EMPTY list is not "no policy": it deletes this sandbox's own
482
+ // object and leaves the configured baseline — whatever
483
+ // `config.egress.policy` translated to, plus every other policy
484
+ // selecting the pod — in force.
485
+ if (hosts.length === 0) {
486
+ await remove()
487
+ return
488
+ }
489
+ await write(translate(hosts))
490
+ })
491
+ queue = run.then(
492
+ () => undefined,
493
+ () => undefined,
494
+ )
495
+ await run
496
+ }
497
+ }