@namzu/sandbox 14.0.0 → 16.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. package/CHANGELOG.md +924 -0
  2. package/README.md +369 -14
  3. package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
  4. package/dist/backends/aci-standby-pool/index.js +13 -1
  5. package/dist/backends/aci-standby-pool/index.js.map +1 -1
  6. package/dist/backends/docker/index.d.ts +169 -6
  7. package/dist/backends/docker/index.d.ts.map +1 -1
  8. package/dist/backends/docker/index.js +499 -85
  9. package/dist/backends/docker/index.js.map +1 -1
  10. package/dist/backends/firecracker/index.d.ts.map +1 -1
  11. package/dist/backends/firecracker/index.js +12 -2
  12. package/dist/backends/firecracker/index.js.map +1 -1
  13. package/dist/backends/firecracker/protocol.d.ts +459 -8
  14. package/dist/backends/firecracker/protocol.d.ts.map +1 -1
  15. package/dist/backends/firecracker/protocol.js +136 -0
  16. package/dist/backends/firecracker/protocol.js.map +1 -1
  17. package/dist/backends/firecracker/transport.d.ts +539 -6
  18. package/dist/backends/firecracker/transport.d.ts.map +1 -1
  19. package/dist/backends/firecracker/transport.js +1171 -24
  20. package/dist/backends/firecracker/transport.js.map +1 -1
  21. package/dist/backends/kubernetes/egress-policy.d.ts +1181 -13
  22. package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -1
  23. package/dist/backends/kubernetes/egress-policy.js +2350 -31
  24. package/dist/backends/kubernetes/egress-policy.js.map +1 -1
  25. package/dist/backends/kubernetes/identity.d.ts +193 -0
  26. package/dist/backends/kubernetes/identity.d.ts.map +1 -0
  27. package/dist/backends/kubernetes/identity.js +147 -0
  28. package/dist/backends/kubernetes/identity.js.map +1 -0
  29. package/dist/backends/kubernetes/index.d.ts +678 -33
  30. package/dist/backends/kubernetes/index.d.ts.map +1 -1
  31. package/dist/backends/kubernetes/index.js +1180 -95
  32. package/dist/backends/kubernetes/index.js.map +1 -1
  33. package/dist/backends/kubernetes/ingress-policy.d.ts +375 -0
  34. package/dist/backends/kubernetes/ingress-policy.d.ts.map +1 -0
  35. package/dist/backends/kubernetes/ingress-policy.js +1050 -0
  36. package/dist/backends/kubernetes/ingress-policy.js.map +1 -0
  37. package/dist/backends/kubernetes/k8s-client.d.ts +213 -4
  38. package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -1
  39. package/dist/backends/kubernetes/k8s-client.js +359 -52
  40. package/dist/backends/kubernetes/k8s-client.js.map +1 -1
  41. package/dist/backends/kubernetes/lease.d.ts +40 -14
  42. package/dist/backends/kubernetes/lease.d.ts.map +1 -1
  43. package/dist/backends/kubernetes/lease.js +68 -18
  44. package/dist/backends/kubernetes/lease.js.map +1 -1
  45. package/dist/backends/kubernetes/objects.d.ts +423 -3
  46. package/dist/backends/kubernetes/objects.d.ts.map +1 -1
  47. package/dist/backends/kubernetes/objects.js +364 -2
  48. package/dist/backends/kubernetes/objects.js.map +1 -1
  49. package/dist/backends/kubernetes/per-sandbox-policy.d.ts +219 -0
  50. package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -0
  51. package/dist/backends/kubernetes/per-sandbox-policy.js +375 -0
  52. package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -0
  53. package/dist/backends/kubernetes/rbac.d.ts +153 -0
  54. package/dist/backends/kubernetes/rbac.d.ts.map +1 -0
  55. package/dist/backends/kubernetes/rbac.js +177 -0
  56. package/dist/backends/kubernetes/rbac.js.map +1 -0
  57. package/dist/backends/kubernetes/sandbox.d.ts +81 -14
  58. package/dist/backends/kubernetes/sandbox.d.ts.map +1 -1
  59. package/dist/backends/kubernetes/sandbox.js +149 -15
  60. package/dist/backends/kubernetes/sandbox.js.map +1 -1
  61. package/dist/backends/kubernetes/transport.d.ts +935 -9
  62. package/dist/backends/kubernetes/transport.d.ts.map +1 -1
  63. package/dist/backends/kubernetes/transport.js +1958 -62
  64. package/dist/backends/kubernetes/transport.js.map +1 -1
  65. package/dist/backends/kubernetes/workspace.d.ts +1149 -18
  66. package/dist/backends/kubernetes/workspace.d.ts.map +1 -1
  67. package/dist/backends/kubernetes/workspace.js +2825 -186
  68. package/dist/backends/kubernetes/workspace.js.map +1 -1
  69. package/dist/backends/remote-execution-controller.d.ts +14 -0
  70. package/dist/backends/remote-execution-controller.d.ts.map +1 -1
  71. package/dist/backends/remote-execution-controller.js.map +1 -1
  72. package/dist/index.d.ts +294 -18
  73. package/dist/index.d.ts.map +1 -1
  74. package/dist/index.js +280 -10
  75. package/dist/index.js.map +1 -1
  76. package/dist/testing/sandbox-conformance.d.ts +39 -5
  77. package/dist/testing/sandbox-conformance.d.ts.map +1 -1
  78. package/dist/testing/sandbox-conformance.js +436 -5
  79. package/dist/testing/sandbox-conformance.js.map +1 -1
  80. package/package.json +3 -3
  81. package/src/backends/aci-standby-pool/index.ts +16 -1
  82. package/src/backends/docker/index.ts +617 -100
  83. package/src/backends/firecracker/index.ts +14 -2
  84. package/src/backends/firecracker/protocol.ts +514 -6
  85. package/src/backends/firecracker/transport.ts +1492 -40
  86. package/src/backends/kubernetes/egress-policy.ts +3334 -55
  87. package/src/backends/kubernetes/identity.ts +261 -0
  88. package/src/backends/kubernetes/index.ts +1785 -127
  89. package/src/backends/kubernetes/ingress-policy.ts +1344 -0
  90. package/src/backends/kubernetes/k8s-client.ts +444 -54
  91. package/src/backends/kubernetes/lease.ts +75 -19
  92. package/src/backends/kubernetes/objects.ts +626 -6
  93. package/src/backends/kubernetes/per-sandbox-policy.ts +497 -0
  94. package/src/backends/kubernetes/rbac.ts +192 -0
  95. package/src/backends/kubernetes/sandbox.ts +218 -20
  96. package/src/backends/kubernetes/transport.ts +2733 -124
  97. package/src/backends/kubernetes/workspace.ts +4476 -222
  98. package/src/backends/remote-execution-controller.ts +14 -0
  99. package/src/index.ts +668 -19
  100. package/src/testing/sandbox-conformance.ts +540 -5
@@ -78,13 +78,36 @@
78
78
 
79
79
  import { isDeepStrictEqual } from 'node:util'
80
80
  import type { EgressPolicy } from '../../index.js'
81
+ // The policy-enumeration and shape-reading primitives, shared with the
82
+ // ingress direction. There is one enumeration of the policies selecting a pod
83
+ // in this package and one reading of what a peer is; see that module's
84
+ // "Shared with the egress direction".
85
+ import {
86
+ type SelectorMatch,
87
+ UnreadPolicyCollection,
88
+ type UnreadPolicySource,
89
+ ciliumIdentityLabels,
90
+ ciliumSelectorKey,
91
+ corePeerIsWideOpen,
92
+ formatLabels,
93
+ isRecord,
94
+ listPolicies,
95
+ matchesLabelSelector,
96
+ policyName,
97
+ readList,
98
+ selectorIsReadable,
99
+ } from './ingress-policy.js'
81
100
  import { KubernetesAlreadyGoneError, type KubernetesClient } from './k8s-client.js'
101
+ import type { KubernetesOwnerReference } from './objects.js'
82
102
  import {
83
103
  CILIUM_NETWORK_POLICY_API_GROUP,
84
104
  CILIUM_NETWORK_POLICY_API_VERSION,
85
105
  CORE_NETWORK_POLICY_API_GROUP,
86
106
  CORE_NETWORK_POLICY_API_VERSION,
107
+ SANDBOX_TEMPLATE_LABEL_KEY,
108
+ ciliumNetworkPolicyCollectionPath,
87
109
  ciliumNetworkPolicyPath,
110
+ networkPolicyCollectionPath,
88
111
  networkPolicyPath,
89
112
  sandboxTemplateLabel,
90
113
  } from './objects.js'
@@ -97,6 +120,160 @@ import {
97
120
  */
98
121
  export type KubernetesEgressEngine = 'core' | 'cilium'
99
122
 
123
+ /**
124
+ * The two egress kinds that exist only here, because only a `NetworkPolicy`
125
+ * can express them and the shared {@link EgressPolicy} union describes what
126
+ * every tier can carry.
127
+ *
128
+ * - `'no-network'` is the one the shared union has no word for: NOTHING
129
+ * leaves the pod, the cluster's own resolver included. `'deny-all'` is not
130
+ * that and never was — it emits {@link CLUSTER_DNS_EGRESS_RULE}, and a
131
+ * cluster resolver forwards outside names upstream, so a `'deny-all'`
132
+ * sandbox keeps a channel out through DNS. `'deny-all'` is deliberately
133
+ * NOT tightened into this: its emitted manifest is byte-identical to
134
+ * every release before this one, because verification of the named object
135
+ * is an exact match and changing the translation would fail every
136
+ * `create()` on every deployment that already applied a policy until an
137
+ * operator re-applied it. A workload under `'no-network'` resolves
138
+ * nothing at all — that is the point, and the agent needs no resolver
139
+ * because the host dials in.
140
+ * - `'public-internet'` is `'allow-all'` minus everything that is not the
141
+ * public internet: the private ranges, the carrier-grade NAT range, the
142
+ * link-local range that carries cloud instance metadata, loopback, the
143
+ * platform endpoint some clouds answer on, and IPv6's equivalents. It
144
+ * exists because `'allow-all'` reaches the node, the API server, the
145
+ * service network and every other sandbox pod, and nothing between the
146
+ * two said "out, but not sideways".
147
+ *
148
+ * `exceptCidrs` adds to the excluded list; it never removes from it. A CIDR
149
+ * that is not one this check can parse is refused at construction rather
150
+ * than emitted into a manifest the API server would reject on apply.
151
+ */
152
+ export type KubernetesOnlyEgressPolicy =
153
+ | { readonly kind: 'no-network' }
154
+ | {
155
+ readonly kind: 'public-internet'
156
+ readonly exceptCidrs?: readonly string[]
157
+ }
158
+
159
+ /**
160
+ * What {@link KubernetesEgressConfig.policy} accepts: the shared
161
+ * {@link EgressPolicy} union plus the two {@link KubernetesOnlyEgressPolicy}
162
+ * kinds. The shared union itself is untouched — a kind no other tier can
163
+ * enforce does not belong in the type every tier reads.
164
+ */
165
+ export type KubernetesEgressPolicy = EgressPolicy | KubernetesOnlyEgressPolicy
166
+
167
+ /**
168
+ * How thoroughly the applied boundary is checked before a sandbox is handed
169
+ * back.
170
+ *
171
+ * - `'union'` (the default) reads the NAMED object exactly as before AND
172
+ * enumerates every `NetworkPolicy` — and, under `engine: 'cilium'`, every
173
+ * `CiliumNetworkPolicy` — in the namespace, refusing when any OTHER policy
174
+ * that selects the pod allows egress the configured translation does not.
175
+ * The named object is left to the exact comparison, which is stronger than
176
+ * anything this rule could conclude about it — see
177
+ * {@link EgressPolicyDocument.namedObject} — and still counts as the policy
178
+ * that default-denies, so it is the enumeration's subject rather than an
179
+ * exception to it. That is not belt-and-braces: the API server UNIONS every
180
+ * policy selecting a pod, so a second policy widens egress however exactly
181
+ * the named one matches, and a `SandboxTemplate`'s own `networkPolicy`
182
+ * block becomes exactly such a policy.
183
+ * - `'named-object-only'` is the documented opt-out, and restores the
184
+ * previous behaviour exactly: one GET of the named object, memoized for
185
+ * the backend's lifetime, and no enumeration. For a deployment whose
186
+ * other policies a namespaced Role cannot read, or which accepts the
187
+ * union it has. It mirrors `ingress: 'unverified'` — a claim a deployment
188
+ * makes on purpose rather than a default it inherits.
189
+ */
190
+ export type KubernetesEgressVerification = 'union' | 'named-object-only'
191
+
192
+ /**
193
+ * How {@link KubernetesCiliumEgressNarrowing.dnsNames} narrows the kube-dns
194
+ * L7 rule. `true` uses every default below; an object customises them.
195
+ *
196
+ * The exact-name list a narrowed policy emits is, for each allowed host,
197
+ * the bare name plus the host under `<namespace>.svc.<clusterDomain>`,
198
+ * `svc.<clusterDomain>` and `<clusterDomain>` — because Cilium's `matchName`
199
+ * is an EXACT name and does not match across a `.` the way `matchPattern`
200
+ * does, and a search-list query for `github.com.svc.cluster.local` is a
201
+ * DIFFERENT name than `github.com`. `searchSuffixes` adds more: kubelet
202
+ * appends the node's own search domains, which this backend cannot see, so a
203
+ * deployment whose nodes carry extra search domains lists them here or a
204
+ * search-list query for one of them is refused by the DNS proxy — and, per
205
+ * Cilium's own docs, some images (musl/Alpine) stop trying the search list
206
+ * entirely the first time that happens, breaking the bare name lookup too.
207
+ */
208
+ export interface KubernetesCiliumDnsNarrowing {
209
+ /** Defaults to the egress target's own namespace (where the sandbox pods run). */
210
+ readonly namespace?: string
211
+ /** Defaults to `'cluster.local'`. */
212
+ readonly clusterDomain?: string
213
+ /** Appended after the three built-in suffixes, not replacing them. */
214
+ readonly searchSuffixes?: readonly string[]
215
+ }
216
+
217
+ /**
218
+ * Opt-in narrowing of a `static`/`resolver` hostname allowlist under
219
+ * `engine: 'cilium'` — see the module doc and `#490`. Every field here is
220
+ * OFF unless set, and setting none of them leaves the translation
221
+ * byte-for-byte what it always emitted: that is the compatibility guarantee
222
+ * a deployment with an already-applied policy relies on.
223
+ *
224
+ * Setting any field switches the translation from one shared `toFQDNs` rule
225
+ * naming every host to one `toFQDNs` rule PER HOST, so ports and server
226
+ * names can differ host by host. Refused synchronously (alongside
227
+ * {@link assertEgressPolicyIsEnforceable}'s existing refusals) unless
228
+ * `engine` is `'cilium'` and `policy.kind` is `'static'` or `'resolver'`.
229
+ */
230
+ export interface KubernetesCiliumEgressNarrowing {
231
+ /**
232
+ * TCP ports allowed to every host that has no entry in `hostPorts`, e.g.
233
+ * `[443]`. Leaving both this and `hostPorts` unset — with `dnsNames` and
234
+ * `tlsServerNames` also unset — means no port narrowing: a host's
235
+ * `toFQDNs` rule carries no `toPorts` at all, exactly as the
236
+ * unnarrowed translation emits today (every port reachable).
237
+ */
238
+ readonly ports?: readonly number[]
239
+ /**
240
+ * Per-host TCP port overrides, keyed by the host exactly as it appears in
241
+ * `allowedHosts` or a `resolver`'s result. A host with no entry here
242
+ * falls back to `ports`.
243
+ */
244
+ readonly hostPorts?: Readonly<Record<string, readonly number[]>>
245
+ /**
246
+ * Narrow the kube-dns L7 rule from `rules.dns: [{ matchPattern: '*' }]`
247
+ * to an exact `matchName` per allowed host, and per host-plus-suffix. See
248
+ * {@link KubernetesCiliumDnsNarrowing}.
249
+ */
250
+ readonly dnsNames?: boolean | KubernetesCiliumDnsNarrowing
251
+ /**
252
+ * Add `serverNames: [<host>]` to each host's TLS ports (SNI enforcement,
253
+ * which needs Cilium's L7 proxy — see the module doc). A host with no
254
+ * port configured in `ports`/`hostPorts` is limited to `tlsPorts` rather
255
+ * than left with no port restriction at all, because a `serverNames`
256
+ * rule needs a port to attach to.
257
+ *
258
+ * REFUSED for a `.domain` allowlist entry, with
259
+ * {@link KubernetesNetworkPolicyHostError} and nothing written: a server
260
+ * name is one exact SNI value a handshake presents, while the entry means
261
+ * a name PLUS its subdomains, and no single value means both — see
262
+ * {@link assertHostsFitNarrowing}. The refusal is the entry's, not this
263
+ * option's, so it fires wherever `tlsServerNames` can be set:
264
+ * `config.egress.ciliumNarrowing` on the config-level allowlist and
265
+ * `config.egress.perSandbox.narrowing` on the per-sandbox one.
266
+ */
267
+ readonly tlsServerNames?: boolean
268
+ /**
269
+ * Which of a host's configured ports are TLS ports, for
270
+ * `tlsServerNames`: those get `serverNames` on their own `toPorts`
271
+ * entry, the rest (if any) get a separate entry with none. Defaults to
272
+ * `[443]`. Meaningless unless `tlsServerNames` is set.
273
+ */
274
+ readonly tlsPorts?: readonly number[]
275
+ }
276
+
100
277
  /**
101
278
  * The config-level egress hook on {@link KubernetesBackendConfig}.
102
279
  *
@@ -109,7 +286,7 @@ export type KubernetesEgressEngine = 'core' | 'cilium'
109
286
  * the one egress policy the whole backend enforces.
110
287
  */
111
288
  export interface KubernetesEgressConfig {
112
- readonly policy: EgressPolicy
289
+ readonly policy: KubernetesEgressPolicy
113
290
  /**
114
291
  * Name of the `NetworkPolicy` (or `CiliumNetworkPolicy`, under the
115
292
  * `'cilium'` engine) an operator applied. Defaults to
@@ -119,6 +296,663 @@ export interface KubernetesEgressConfig {
119
296
  readonly networkPolicyName?: string
120
297
  /** Default `'core'`. See the type doc. */
121
298
  readonly engine?: KubernetesEgressEngine
299
+ /** Default `'union'`. See {@link KubernetesEgressVerification}. */
300
+ readonly verify?: KubernetesEgressVerification
301
+ /**
302
+ * Opt-in port/DNS-name/TLS-server-name narrowing for a `static`/
303
+ * `resolver` allowlist under `engine: 'cilium'`. Unset (the default)
304
+ * emits exactly what every release before this one did. See
305
+ * {@link KubernetesCiliumEgressNarrowing}.
306
+ */
307
+ readonly ciliumNarrowing?: KubernetesCiliumEgressNarrowing
308
+ /**
309
+ * Which egress PROFILE the sandboxes this backend produces run under —
310
+ * a DNS-1123 label value such as `none` or `internet`.
311
+ *
312
+ * Unset (the default) is the single-profile world this backend has
313
+ * always had: one policy per backend, selected by the template label
314
+ * alone, and every emitted body, selector and policy name byte-identical
315
+ * to the release before profiles existed.
316
+ *
317
+ * SET, it becomes a pod LABEL — {@link profileLabelKey} is its key —
318
+ * which travels three places at once: onto the `SandboxClaim`'s
319
+ * `additionalPodMetadata.labels`, onto a directly created Sandbox's pod
320
+ * template, and into the translated policy's own selector. That is what
321
+ * lets ONE warm pool serve several network modes: a claim carrying a
322
+ * profile label adopts a warm replica and the controller patches the
323
+ * label onto the running pod, with no cold start and no second pool —
324
+ * measured on agent-sandbox v1.0.2 (two profiles out of one two-replica
325
+ * pool, every adopt under 70 ms, each bound pod a replica that already
326
+ * existed). A label is the only claim-time metadata that is warm-safe:
327
+ * `env` and `volumeClaimTemplates` force a cold start, which is why
328
+ * neither appears on the claim body.
329
+ *
330
+ * OPERATOR PREREQUISITE, and it is not optional: the controller refuses
331
+ * a claim whose label key sits outside its `allowed-label-domains`
332
+ * allowlist (the `agent-sandbox-config` ConfigMap in the controller's
333
+ * namespace; default `sandbox.users.io`), so the DEFAULT key below is
334
+ * refused by a stock controller until an operator adds
335
+ * `sandbox.namzu.ai` to that key — or sets {@link profileLabelKey} to
336
+ * something already allowed. That refusal is fast and carries the
337
+ * controller's own reason and message: it is the acquire's ordinary
338
+ * `claim-rejected` failure, with a
339
+ * {@link KubernetesPodLabelsRejectedError} as its cause naming the labels
340
+ * that were sent and the key that moves them.
341
+ */
342
+ readonly profile?: string
343
+ /**
344
+ * Label key {@link profile} is written under. Defaults to
345
+ * {@link DEFAULT_EGRESS_PROFILE_LABEL_KEY}.
346
+ *
347
+ * Worth setting to `sandbox.users.io/egress-profile` on a cluster whose
348
+ * controller still carries the stock `allowed-label-domains` — that
349
+ * domain is upstream's own default and needs no ConfigMap edit at all.
350
+ */
351
+ readonly profileLabelKey?: string
352
+ /**
353
+ * Opt in to PER-SANDBOX egress: one `CiliumNetworkPolicy` per live
354
+ * sandbox, written by this host when a caller calls
355
+ * `Sandbox.setNetworkPolicy`, owned by the object the acquire created so
356
+ * the cluster garbage-collects it.
357
+ *
358
+ * Unset (the default) is every release before this one: `setNetworkPolicy`
359
+ * is ABSENT from the handle, exactly as the SDK's omit-or-throw contract
360
+ * asks of a backend that cannot honour an optional method, and this
361
+ * backend issues no policy write of any kind. Presence depends only on
362
+ * this field — never on a runtime probe — so a host can decide what it
363
+ * has from its own config rather than from a call that might fail.
364
+ *
365
+ * TWO OPERATOR PREREQUISITES, and neither is a nicety: the admission
366
+ * policy named below (with its binding) has to exist, and the host's
367
+ * ServiceAccount needs the write verbs the default Role deliberately does
368
+ * not grant. See {@link KubernetesPerSandboxEgressConfig}.
369
+ */
370
+ readonly perSandbox?: KubernetesPerSandboxEgressConfig
371
+ }
372
+
373
+ /**
374
+ * Per-sandbox egress: what it takes to let a host narrow ONE live sandbox's
375
+ * egress without touching any other sandbox's.
376
+ *
377
+ * The mechanism is a `CiliumNetworkPolicy` per sandbox, named
378
+ * `namzu-sbx-<uid of the object this backend created>`, selecting that one
379
+ * sandbox's pod through a per-sandbox label, and carrying an
380
+ * `ownerReferences` entry naming that same object — so `destroy()` deletes
381
+ * the claim (or the Sandbox) and the cluster's garbage collector removes the
382
+ * policy, with this backend issuing no deletion of its own.
383
+ *
384
+ * It is NOT the docker backend's mechanism. That one is an egress PROXY with
385
+ * no relationship to any policy object; the only two things worth taking from
386
+ * it are its refusal discipline and `SandboxNetworkPolicy.allowedHosts`'s
387
+ * `.domain`-prefix wildcard semantics.
388
+ *
389
+ * A per-sandbox list ADDS to {@link KubernetesEgressConfig.policy}, because
390
+ * the cluster unions every policy that selects a pod. That makes it the whole
391
+ * boundary under a `'no-network'` or `'deny-all'` baseline — the deployment
392
+ * this is for — and a no-op addition under `'allow-all'`, where
393
+ * `setNetworkPolicy([])` also does not deny everything the way
394
+ * `SandboxNetworkPolicy` describes. Neither combination is refused; pair this
395
+ * with a denying `policy.kind` when `setNetworkPolicy` is meant to BE the
396
+ * boundary rather than to widen one.
397
+ */
398
+ export interface KubernetesPerSandboxEgressConfig {
399
+ /**
400
+ * Must be `'cilium'`. `'core'` is accepted by the TYPE and refused
401
+ * SYNCHRONOUSLY during host wiring, the same way
402
+ * {@link assertEgressPolicyIsEnforceable} refuses a hostname allowlist
403
+ * with no FQDN-capable engine: core `NetworkPolicy` has no hostname
404
+ * concept at all, so there is nothing for an `allowedHosts` list to
405
+ * become. It is spelled out in the type rather than fixed to the one
406
+ * legal value so the refusal can name what was configured.
407
+ */
408
+ readonly engine: KubernetesEgressEngine
409
+ /**
410
+ * Name of the operator-applied `ValidatingAdmissionPolicy` that bounds
411
+ * what this host may write. Checked — with its binding — BEFORE the first
412
+ * write, and the write is refused with nothing sent if either object is
413
+ * missing.
414
+ *
415
+ * The fence is the whole reason this capability can be granted at all:
416
+ * the RBAC it needs is `create`/`patch`/`delete` on the namespace's
417
+ * `ciliumnetworkpolicies`, which without a fence would let a compromised
418
+ * host widen or delete the operator's own baseline policy. The shipped
419
+ * example is `k8s/manifests/validatingadmissionpolicy-cilium.yaml`.
420
+ */
421
+ readonly admissionPolicyName: string
422
+ /**
423
+ * Name of the `ValidatingAdmissionPolicyBinding` that ATTACHES the policy
424
+ * above. Defaults to `${admissionPolicyName}-binding`, which is what the
425
+ * shipped manifest names it.
426
+ *
427
+ * Checked separately because a `ValidatingAdmissionPolicy` with no
428
+ * binding validates nothing at all — it is inert, and an inert fence
429
+ * reads exactly like an enforced one from the object alone.
430
+ */
431
+ readonly admissionPolicyBindingName?: string
432
+ /**
433
+ * Label key the per-sandbox selector is written under. Defaults to
434
+ * {@link DEFAULT_PER_SANDBOX_EGRESS_LABEL_KEY}.
435
+ *
436
+ * Its VALUE is the name of the object this backend created for the
437
+ * sandbox, so it is unique per acquire and known before the create POST —
438
+ * which is what lets it travel as claim-time pod metadata through
439
+ * {@link composeAdditionalPodLabels}, the ONE composer, rather than as a
440
+ * patch to a running pod this backend has no verb for.
441
+ *
442
+ * Same operator prerequisite as {@link KubernetesEgressConfig.profile}:
443
+ * the key's domain has to be in the controller's `allowed-label-domains`
444
+ * allowlist, or the claim is refused with the controller's own reason.
445
+ */
446
+ readonly labelKey?: string
447
+ /**
448
+ * Port, DNS-name and TLS-server-name narrowing applied to the policies
449
+ * `setNetworkPolicy` writes — the same shape as
450
+ * {@link KubernetesEgressConfig.ciliumNarrowing}, and deliberately its own
451
+ * field rather than a reuse of it: that one narrows the ONE config-level
452
+ * policy, which is only a hostname allowlist at all when
453
+ * `policy.kind` is `'static'` or `'resolver'`, while these narrow the
454
+ * per-sandbox policies of a deployment whose baseline is usually
455
+ * `'deny-all'` or `'no-network'`.
456
+ *
457
+ * Unset (the default) emits one `toFQDNs` rule per host with no
458
+ * `toPorts` at all — every port on an allowed host reachable, which is
459
+ * what `SandboxNetworkPolicy` itself says (it names hosts, not ports).
460
+ *
461
+ * `hostPorts` is keyed by the `allowedHosts` entry AS WRITTEN, leading
462
+ * dot included: a `.example.com` entry needs the key `.example.com`, even
463
+ * though the rule it produces says `example.com` plus `*.example.com`.
464
+ *
465
+ * `tlsServerNames` and a `.domain` entry are REFUSED together, with
466
+ * {@link KubernetesNetworkPolicyHostError} and nothing written. A TLS
467
+ * server name is one exact SNI value; the expanded entry is a name and a
468
+ * pattern, and no single SNI value means both — `example.com` would deny
469
+ * every subdomain the policy claims to allow. List the exact hosts, or
470
+ * leave `tlsServerNames` off for a domain list.
471
+ */
472
+ readonly narrowing?: KubernetesCiliumEgressNarrowing
473
+ }
474
+
475
+ /**
476
+ * Default {@link KubernetesEgressConfig.profileLabelKey} — this backend's
477
+ * own label domain, matching `SANDBOX_TEMPLATE_LABEL_KEY`'s prefix so
478
+ * every label a namzu host puts on a sandbox pod reads as one family.
479
+ *
480
+ * It is deliberately NOT upstream's `sandbox.users.io`: a key in someone
481
+ * else's domain is a key someone else may define differently. The cost is
482
+ * the ConfigMap edit named on {@link KubernetesEgressConfig.profile}, and
483
+ * the controller's refusal spells that edit out itself.
484
+ */
485
+ export const DEFAULT_EGRESS_PROFILE_LABEL_KEY = 'sandbox.namzu.ai/egress-profile'
486
+
487
+ /** One resolved profile label: the key it is written under and its value. */
488
+ export interface EgressProfileLabel {
489
+ readonly key: string
490
+ readonly value: string
491
+ }
492
+
493
+ /**
494
+ * Kubernetes' own label-value grammar, narrowed to DNS-1123: lowercase
495
+ * alphanumerics and `-`, starting and ending alphanumeric, at most 63
496
+ * characters.
497
+ *
498
+ * Narrower than what a label value may legally hold (`_` and `.` are legal
499
+ * there, and uppercase is too) because the profile is also a NAME: it goes
500
+ * into `${template}-${profile}-egress`, which has to be a legal object name,
501
+ * and a value that is legal as a label but not as a name would produce a
502
+ * policy an operator cannot apply.
503
+ */
504
+ const DNS_1123_LABEL = /^[a-z0-9]([-a-z0-9]{0,61}[a-z0-9])?$/
505
+
506
+ /**
507
+ * A label KEY: an optional DNS-subdomain prefix, a `/`, then a name segment
508
+ * of at most 63 characters. Exactly what the API server enforces, checked
509
+ * here so a typo is refused during host wiring rather than as a claim the
510
+ * controller rejects one round trip later.
511
+ */
512
+ const LABEL_KEY_NAME = /^[A-Za-z0-9]([-A-Za-z0-9_.]{0,61}[A-Za-z0-9])?$/
513
+ const LABEL_KEY_PREFIX = /^[a-z0-9]([-a-z0-9.]{0,251}[a-z0-9])?$/
514
+
515
+ /**
516
+ * Named refusal for an egress PROFILE this backend can read but not use.
517
+ * Sibling of {@link KubernetesEgressPolicyConfigError} rather than a reuse of
518
+ * it: that one names a field under `config.egress.policy` and this one names
519
+ * a field beside it, and an operator reading either should not have to work
520
+ * out which level of the config the path belongs to.
521
+ *
522
+ * Thrown SYNCHRONOUSLY from `buildKubernetesBackend` and
523
+ * `createKubernetesWorkspace`, so a misconfigured profile surfaces during
524
+ * host wiring rather than on the first `create()`.
525
+ */
526
+ export class KubernetesEgressProfileConfigError extends Error {
527
+ override readonly name = 'KubernetesEgressProfileConfigError'
528
+
529
+ constructor(
530
+ readonly field: 'profile' | 'profileLabelKey',
531
+ readonly value: string,
532
+ reason: string,
533
+ ) {
534
+ super(
535
+ `kubernetes: config.egress.${field} is unusable: ${JSON.stringify(value)} ${reason}. The profile travels onto a SandboxClaim's additionalPodMetadata.labels, onto a directly created Sandbox's pod template and into the translated policy's own selector, so a value the API server would reject leaves either a claim nothing binds or a policy nobody can apply. Refusing here rather than emitting it.`,
536
+ )
537
+ }
538
+ }
539
+
540
+ /**
541
+ * The profile label this config asks for, or nothing at all — the ONE place
542
+ * the key/value pair is derived, so the claim body, the Sandbox pod
543
+ * template, the policy selector and the policy name cannot disagree about
544
+ * what the profile is.
545
+ *
546
+ * Validates as it resolves: the value has to be a DNS-1123 label and the key
547
+ * a legal label key, both refused with {@link KubernetesEgressProfileConfigError}.
548
+ */
549
+ export function egressProfileLabel(
550
+ egress: KubernetesEgressConfig | undefined,
551
+ ): EgressProfileLabel | undefined {
552
+ // The KEY is validated whenever it is present, profile or no profile. A
553
+ // key set without a value is a half-finished configuration — the value is
554
+ // usually the next line someone writes — and reporting the typo only once
555
+ // the profile arrives is reporting it at the second edit rather than the
556
+ // first.
557
+ const configured = egress?.profileLabelKey
558
+ if (configured !== undefined) assertUsableProfileLabelKey(configured)
559
+ const value = egress?.profile
560
+ if (value === undefined) return undefined
561
+ if (!DNS_1123_LABEL.test(value)) {
562
+ throw new KubernetesEgressProfileConfigError(
563
+ 'profile',
564
+ value,
565
+ 'is not a DNS-1123 label (lowercase letters, digits and dashes, starting and ending alphanumeric, at most 63 characters)',
566
+ )
567
+ }
568
+ const key = configured ?? DEFAULT_EGRESS_PROFILE_LABEL_KEY
569
+ return { key, value }
570
+ }
571
+
572
+ /**
573
+ * Why a label KEY is unusable, or `undefined` when it is fine — the grammar
574
+ * half of the two checks below, shared because the API server's rule is the
575
+ * same whichever of this backend's label keys is being configured and two
576
+ * spellings of it would be two rules.
577
+ */
578
+ function labelKeyProblem(key: string): string | undefined {
579
+ const slash = key.indexOf('/')
580
+ const name = slash === -1 ? key : key.slice(slash + 1)
581
+ const prefix = slash === -1 ? undefined : key.slice(0, slash)
582
+ if (
583
+ !LABEL_KEY_NAME.test(name) ||
584
+ (prefix !== undefined && !LABEL_KEY_PREFIX.test(prefix)) ||
585
+ key.indexOf('/', slash + 1) !== -1
586
+ ) {
587
+ return 'is not a Kubernetes label key (an optional DNS-subdomain prefix, a single slash, then a name of at most 63 characters)'
588
+ }
589
+ return undefined
590
+ }
591
+
592
+ /**
593
+ * Refuse a `profileLabelKey` the API server would not take, or that this
594
+ * backend already uses for something else.
595
+ */
596
+ function assertUsableProfileLabelKey(key: string): void {
597
+ const problem = labelKeyProblem(key)
598
+ if (problem !== undefined) {
599
+ throw new KubernetesEgressProfileConfigError('profileLabelKey', key, problem)
600
+ }
601
+ // The one legal key that must not be used: it is the key this backend
602
+ // stamps the SandboxTemplate name under, and the profile label is applied
603
+ // LAST (see `sandboxPodLabels`), so this key would overwrite the template
604
+ // label on every pod this backend creates — and the translated policy's
605
+ // selector, built from the same resolution, would agree with it. Both
606
+ // halves would be wrong together, which is exactly the shape nothing else
607
+ // would catch.
608
+ if (key === SANDBOX_TEMPLATE_LABEL_KEY) {
609
+ throw new KubernetesEgressProfileConfigError(
610
+ 'profileLabelKey',
611
+ key,
612
+ 'is the key this backend writes the SandboxTemplate name under, and a profile label is applied last — a pod would carry the profile value where its template label belongs, and the policy selector built from the same resolution would match it anyway',
613
+ )
614
+ }
615
+ }
616
+
617
+ /**
618
+ * Validate the profile without needing its value — the wiring-time hook, so
619
+ * `buildKubernetesBackend` and `createKubernetesWorkspace` refuse a bad
620
+ * profile the same moment they refuse an unenforceable policy.
621
+ */
622
+ export function assertEgressProfileIsUsable(egress: KubernetesEgressConfig | undefined): void {
623
+ egressProfileLabel(egress)
624
+ }
625
+
626
+ /**
627
+ * Default {@link KubernetesPerSandboxEgressConfig.labelKey} — the key the
628
+ * per-sandbox policy's `endpointSelector` matches on.
629
+ *
630
+ * In this backend's own label domain, for the same reason the profile key is
631
+ * (see {@link DEFAULT_EGRESS_PROFILE_LABEL_KEY}), and carrying the same
632
+ * operator prerequisite: the controller's `allowed-label-domains` allowlist
633
+ * has to admit `sandbox.namzu.ai`, or every claim carrying it is refused with
634
+ * {@link KubernetesPodLabelsRejectedError} as the cause. A deployment that
635
+ * would rather not edit that ConfigMap sets this to a key under
636
+ * `sandbox.users.io`, which is upstream's own default domain.
637
+ */
638
+ export const DEFAULT_PER_SANDBOX_EGRESS_LABEL_KEY = 'sandbox.namzu.ai/per-sandbox-egress'
639
+
640
+ /**
641
+ * Named refusal for a `config.egress.perSandbox` this backend can read but
642
+ * not honour. Thrown SYNCHRONOUSLY from `buildKubernetesBackend`, beside the
643
+ * policy and profile refusals, so a host that has mis-declared the capability
644
+ * learns it during wiring rather than from the first `setNetworkPolicy` call
645
+ * — which may be an hour into a run, after work that cannot be redone.
646
+ *
647
+ * Its own class rather than a reuse of {@link KubernetesEgressProfileConfigError}
648
+ * for the reason that one is not a reuse of
649
+ * {@link KubernetesEgressPolicyConfigError}: an operator reading a refusal
650
+ * should not have to work out which level of `config.egress` the named field
651
+ * belongs to.
652
+ */
653
+ export class KubernetesPerSandboxEgressConfigError extends Error {
654
+ override readonly name = 'KubernetesPerSandboxEgressConfigError'
655
+
656
+ constructor(
657
+ readonly field: 'engine' | 'admissionPolicyName' | 'admissionPolicyBindingName' | 'labelKey',
658
+ readonly value: string,
659
+ reason: string,
660
+ ) {
661
+ super(
662
+ `kubernetes: config.egress.perSandbox.${field} is unusable: ${JSON.stringify(value)} ${reason}. Per-sandbox egress writes one CiliumNetworkPolicy per live sandbox, selected by a per-sandbox pod label and fenced by an operator-applied ValidatingAdmissionPolicy, so a value this backend cannot honour would leave either a policy that selects nothing or a write the fence refuses. Refusing during host wiring rather than at the first setNetworkPolicy call.`,
663
+ )
664
+ }
665
+ }
666
+
667
+ /**
668
+ * The per-sandbox selector label KEY this config asks for, or nothing at all
669
+ * when `perSandbox` is unset — the one place the default is applied, so the
670
+ * pod label, the policy selector and the admission policy's own expectation
671
+ * cannot disagree.
672
+ *
673
+ * Validates as it resolves, exactly as {@link egressProfileLabel} does.
674
+ */
675
+ export function perSandboxEgressLabelKey(
676
+ egress: KubernetesEgressConfig | undefined,
677
+ ): string | undefined {
678
+ const perSandbox = egress?.perSandbox
679
+ if (perSandbox === undefined) return undefined
680
+ const key = perSandbox.labelKey ?? DEFAULT_PER_SANDBOX_EGRESS_LABEL_KEY
681
+ const problem = labelKeyProblem(key)
682
+ if (problem !== undefined) {
683
+ throw new KubernetesPerSandboxEgressConfigError('labelKey', key, problem)
684
+ }
685
+ if (key === SANDBOX_TEMPLATE_LABEL_KEY) {
686
+ throw new KubernetesPerSandboxEgressConfigError(
687
+ 'labelKey',
688
+ key,
689
+ 'is the key this backend writes the SandboxTemplate name under, so a pod would carry a sandbox name where its template label belongs and every policy selecting the template would stop selecting it',
690
+ )
691
+ }
692
+ // The profile's key is refused from the other direction too — see
693
+ // `composeAdditionalPodLabels` — but naming it HERE names the field a
694
+ // reader has to change, which the generic collision message cannot.
695
+ const profile = egressProfileLabel(egress)
696
+ if (profile !== undefined && key === profile.key) {
697
+ throw new KubernetesPerSandboxEgressConfigError(
698
+ 'labelKey',
699
+ key,
700
+ 'is also config.egress.profileLabelKey, and both are written into the SAME pod-label map — one value would silently replace the other, and whichever lost would be a label a policy selector still expects',
701
+ )
702
+ }
703
+ return key
704
+ }
705
+
706
+ /**
707
+ * Validate `config.egress.perSandbox` in full — the wiring-time hook, so a
708
+ * mis-declared capability is refused the same moment an unenforceable policy
709
+ * or an unusable profile is.
710
+ *
711
+ * `engine: 'core'` is the refusal the SDK's contract cares about: core
712
+ * `NetworkPolicy` cannot express a hostname at all, so a host that configured
713
+ * it would be told "policy applied" about an object that could never carry
714
+ * the allowlist. It is refused here rather than from the method, which means
715
+ * a `'core'` deployment never gets a handle carrying the method in the first
716
+ * place.
717
+ */
718
+ export function assertPerSandboxEgressIsUsable(egress: KubernetesEgressConfig | undefined): void {
719
+ const perSandbox = egress?.perSandbox
720
+ if (perSandbox === undefined) return
721
+ if (perSandbox.engine !== 'cilium') {
722
+ throw new KubernetesPerSandboxEgressConfigError(
723
+ 'engine',
724
+ perSandbox.engine,
725
+ "is not 'cilium'; a per-sandbox allowlist is a list of HOSTNAMES, and core NetworkPolicy has only ipBlock, podSelector and namespaceSelector — it has no hostname concept to translate one into",
726
+ )
727
+ }
728
+ // `admissionPolicyName` is REQUIRED by the type; the cast is for the
729
+ // caller reaching here from JavaScript, or through a cast of its own.
730
+ // There is no default and there must not be one: the fence is what makes
731
+ // the policy-write RBAC safe to grant, and a host that could skip naming
732
+ // it would hold create/patch/delete on every CiliumNetworkPolicy in the
733
+ // namespace with nothing bounding what it writes.
734
+ const names = [
735
+ ['admissionPolicyName', perSandbox.admissionPolicyName as string | undefined, true],
736
+ ['admissionPolicyBindingName', perSandbox.admissionPolicyBindingName, false],
737
+ ] as const
738
+ for (const [field, value, required] of names) {
739
+ if (value === undefined) {
740
+ if (!required) continue
741
+ throw new KubernetesPerSandboxEgressConfigError(
742
+ field,
743
+ '',
744
+ 'is required, and has no default: an unnamed fence is an unchecked one',
745
+ )
746
+ }
747
+ if (value === '' || value.trim() !== value) {
748
+ throw new KubernetesPerSandboxEgressConfigError(
749
+ field,
750
+ value,
751
+ 'is not an object name (an empty or space-padded name matches nothing, so the fence check would refuse every write)',
752
+ )
753
+ }
754
+ }
755
+ if (perSandbox.narrowing !== undefined) {
756
+ assertCiliumNarrowingIsUsable(perSandbox.narrowing, 'perSandbox.narrowing')
757
+ }
758
+ perSandboxEgressLabelKey(egress)
759
+ }
760
+
761
+ /**
762
+ * Named refusal for `config.egress.perSandbox` reaching an entry point that
763
+ * can never carry the capability it configures: `createKubernetesWorkspace`.
764
+ *
765
+ * `perSandbox` exists to make `Sandbox.setNetworkPolicy` PRESENT on a TASK
766
+ * handle, where the acquire creates the object each policy is named after and
767
+ * owned by and stamps the per-sandbox pod label its selector matches, and
768
+ * where the RBAC and the admission fence are the ones that bound the write.
769
+ * A workspace's create path does none of that — it composes no per-sandbox
770
+ * pod label and tracks no owner uid for one — so on that path the option
771
+ * would be accepted and mean nothing at all: exactly the "declared and
772
+ * silently unused" configuration this module refuses everywhere else. The
773
+ * alternative, omitting the method and saying nothing, is how a host comes to
774
+ * believe it narrowed a workspace's egress.
775
+ *
776
+ * NOT a variant of {@link KubernetesPerSandboxEgressConfigError}, which is
777
+ * about a `perSandbox` value this backend cannot honour ANYWHERE (an engine
778
+ * with no hostname concept, an unnamed fence). This one is about a `perSandbox`
779
+ * value it honours perfectly well on the other entry point, so a caller
780
+ * catching the sibling for a typo'd value does not also catch a correct
781
+ * configuration aimed at the wrong path.
782
+ */
783
+ export class KubernetesWorkspacePerSandboxEgressConfigError extends Error {
784
+ override readonly name = 'KubernetesWorkspacePerSandboxEgressConfigError'
785
+
786
+ constructor() {
787
+ super(
788
+ `kubernetes: config.egress.perSandbox is set, but a KubernetesWorkspace cannot carry the capability it configures: Sandbox.setNetworkPolicy is implemented on a TASK handle, whose acquire creates the object each per-sandbox policy is named after and owned by and stamps the pod label its selector matches, while a workspace's create path composes no per-sandbox pod label and tracks no owner uid for one. Accepting the option here would declare a capability this path never serves — the silent downgrade this backend refuses on every other entry point — so it is refused before anything is sent. Drop egress.perSandbox from the configuration workspaces are created from (a backend serving task sandboxes can keep it), or create a task sandbox with it, where the method is present.`,
789
+ )
790
+ }
791
+ }
792
+
793
+ /**
794
+ * Refuse `config.egress.perSandbox` on the entry point that cannot carry it —
795
+ * `createKubernetesWorkspace`. See
796
+ * {@link KubernetesWorkspacePerSandboxEgressConfigError}.
797
+ *
798
+ * Deliberately not {@link assertPerSandboxEgressIsUsable}, which validates the
799
+ * same config for the path that DOES honour it: a workspace has nothing to
800
+ * validate the option for, whatever its `engine` or `admissionPolicyName` say,
801
+ * because it serves no `setNetworkPolicy` at all. Calling that validator here
802
+ * instead would be worse than saying nothing — it would report a
803
+ * `perSandbox` this deployment can use as if it were in use.
804
+ */
805
+ export function assertWorkspaceCarriesNoPerSandboxEgress(
806
+ egress: KubernetesEgressConfig | undefined,
807
+ ): void {
808
+ if (egress?.perSandbox === undefined) return
809
+ throw new KubernetesWorkspacePerSandboxEgressConfigError()
810
+ }
811
+
812
+ /**
813
+ * The ONE composer for a sandbox pod's `additionalPodMetadata.labels`.
814
+ *
815
+ * Every label this backend asks the controller to put on a POD is built
816
+ * here: the egress profile today, and whatever a later capability
817
+ * contributes through `extra` (a per-sandbox policy selector, for one). Two
818
+ * independent constructions would be two answers to "what labels is this pod
819
+ * selected by", and the policy selector is built from the same resolution —
820
+ * so a second builder would be a pod bound under a policy nobody checked.
821
+ *
822
+ * Deliberately NOT where `KubernetesBackendInternalConfig.claimLabels`
823
+ * goes. Those are a host's own bookkeeping on the CLAIM object's
824
+ * `metadata.labels`; putting them on the pod would change what selectors
825
+ * match a running sandbox, which is a different question on a different
826
+ * object.
827
+ *
828
+ * Returns an empty object when nothing applies, which every caller reads as
829
+ * "emit nothing at all" — that is what keeps an unprofiled body byte-identical.
830
+ *
831
+ * A key in `extra` that is ALSO the profile's is refused rather than merged
832
+ * either way round. Whichever won, the loser would be a label the translated
833
+ * policy's selector still expects: the profile's selector is built from this
834
+ * same resolution, so a pod carrying the other value is selected by no
835
+ * per-profile policy while `create()` reported the boundary verified. It is
836
+ * the same failure {@link egressProfileLabel} refuses the template key for,
837
+ * reached from the other direction.
838
+ */
839
+ export function composeAdditionalPodLabels(
840
+ egress: KubernetesEgressConfig | undefined,
841
+ extra?: Readonly<Record<string, string>>,
842
+ ): Readonly<Record<string, string>> {
843
+ const profile = egressProfileLabel(egress)
844
+ if (profile !== undefined && extra !== undefined && profile.key in extra) {
845
+ throw new KubernetesEgressProfileConfigError(
846
+ 'profileLabelKey',
847
+ profile.key,
848
+ `is also the key another capability contributes to the same pod-label map (as ${JSON.stringify(extra[profile.key])}), and one of the two values would silently replace the other`,
849
+ )
850
+ }
851
+ return {
852
+ ...(profile !== undefined ? { [profile.key]: profile.value } : {}),
853
+ ...extra,
854
+ }
855
+ }
856
+
857
+ /**
858
+ * Why a claim the controller refused with `InvalidMetadata` was refused, in
859
+ * this backend's own terms — in practice a pod label whose domain is not in
860
+ * the controller's `allowed-label-domains` allowlist, which today means the
861
+ * egress profile's.
862
+ *
863
+ * NOT what an acquire throws. A refused claim comes out of `create()` as
864
+ * `KubernetesAcquireError { reason: 'claim-rejected' }` whether or not a
865
+ * profile is configured — one condition, one taxonomy, one `catch` — and this
866
+ * rides as that error's `cause`. Two classes for one controller condition
867
+ * would make a host's error handling correct or incorrect depending on
868
+ * whether `config.egress.profile` happened to be set.
869
+ *
870
+ * What it adds to the acquire error is what the controller cannot know: the
871
+ * map this backend actually sent, and the `config.egress.profileLabelKey`
872
+ * that moves the offending key to an allowed domain. It carries the whole map
873
+ * rather than the profile alone, and `profile` is optional, because the map is
874
+ * {@link composeAdditionalPodLabels}'s — the profile is the only thing in it
875
+ * today, and a later capability adding a second key would otherwise get an
876
+ * explanation that named a label it did not send.
877
+ *
878
+ * It carries the controller's OWN `reason` and `message` rather than a
879
+ * translation of them: the message agent-sandbox v1.0.2 writes names the
880
+ * offending key, the domain, the ConfigMap key to edit and its default, and
881
+ * no paraphrase of it would be as useful. Measured verbatim against a kind
882
+ * cluster running that controller:
883
+ *
884
+ * > invalid additionalPodMetadata: failed to validate label
885
+ * > "sandbox.namzu.ai/egress-profile": label domain "sandbox.namzu.ai" is
886
+ * > not in the allowlist (configure the allowed-label-domains key of the
887
+ * > agent-sandbox-config ConfigMap in the controller namespace; default:
888
+ * > sandbox.users.io)
889
+ *
890
+ * The refusal is raised as soon as that condition is read rather than after
891
+ * the readiness budget, because `InvalidMetadata` is one of
892
+ * `TERMINAL_CLAIM_REASONS` — nothing about it becomes true by waiting — and
893
+ * the claim is deleted on the way out, so a misconfigured profile costs one
894
+ * round trip rather than a minute of polling.
895
+ */
896
+ export class KubernetesPodLabelsRejectedError extends Error {
897
+ override readonly name = 'KubernetesPodLabelsRejectedError'
898
+
899
+ constructor(
900
+ /** Every label this backend put on the claim's `additionalPodMetadata`. */
901
+ readonly requestedPodLabels: Readonly<Record<string, string>>,
902
+ readonly claimName: string,
903
+ readonly namespace: string,
904
+ /** The controller's own condition `reason`, e.g. `InvalidMetadata`. */
905
+ readonly controllerReason: string,
906
+ /** The controller's own condition `message`, verbatim. */
907
+ readonly controllerMessage: string,
908
+ /** The egress profile among those labels, when one is configured. */
909
+ readonly profile?: EgressProfileLabel,
910
+ ) {
911
+ super(
912
+ `kubernetes: the controller refused SandboxClaim ${claimName} in namespace ${namespace} carrying the pod labels ${formatLabels(requestedPodLabels)} — ${controllerReason}: ${controllerMessage}. Those labels are written onto the claim's additionalPodMetadata.labels, so each key's domain has to appear in the controller's allowed-label-domains allowlist; add it there, or move the offending key to a domain that is already allowed${profile === undefined ? '' : ` (config.egress.profileLabelKey, for the egress profile ${profile.key}=${profile.value})`}. The claim has been deleted.`,
913
+ )
914
+ }
915
+ }
916
+
917
+ /**
918
+ * Named refusal for a bound pod that never carried a label this backend asked
919
+ * the controller to put on it — the egress profile's, today the only one
920
+ * {@link composeAdditionalPodLabels} produces, which is why the class is named
921
+ * for the LABEL rather than for the profile.
922
+ *
923
+ * This is the one failure this capability must not have quietly. An
924
+ * unlabelled pod handed back is a sandbox running under the DEFAULT policy
925
+ * while the host believes it is on a narrower profile — the translated
926
+ * policy's selector includes the profile label, so a pod without it is
927
+ * selected by neither this profile's policy nor, necessarily, anything else.
928
+ * Refusing is correct and waiting is correct; proceeding is not, so the
929
+ * acquire releases what it claimed and raises this instead.
930
+ *
931
+ * `missingLabel` is the pair that never arrived rather than "the profile", so
932
+ * the refusal stays true for whatever a later capability contributes to that
933
+ * same map: the wait in `readAddressedPod` already covers every entry of it,
934
+ * and this refusal covers exactly the same set.
935
+ */
936
+ export class KubernetesPodLabelNotObservedError extends Error {
937
+ override readonly name = 'KubernetesPodLabelNotObservedError'
938
+
939
+ constructor(
940
+ /** The label this backend requested and never saw on the bound pod. */
941
+ readonly missingLabel: EgressProfileLabel,
942
+ readonly subject: string,
943
+ readonly observedLabels: Readonly<Record<string, string>>,
944
+ ) {
945
+ super(
946
+ `kubernetes: refusing ${subject} — its pod never carried the label ${missingLabel.key}=${missingLabel.value} this backend asked the controller to put on it, within the readiness budget; the labels it did carry are ${formatLabels(observedLabels)}. The translated policy's selector includes that label, so admitting this pod would run it under whatever policy DOES select it rather than under the configured profile. The controller patches a claim's additionalPodMetadata.labels onto the pod it binds — a pod that never got them means the claim's metadata was not applied. Nothing was handed back and the claim was released.`,
947
+ )
948
+ }
949
+ }
950
+
951
+ /** `true` unless the deployment asked for the single-object check by name. */
952
+ export function egressUnionVerificationEnabled(
953
+ egress: KubernetesEgressConfig | undefined,
954
+ ): boolean {
955
+ return egress !== undefined && egress.verify !== 'named-object-only'
122
956
  }
123
957
 
124
958
  /** Where a translated policy is targeted, and what its `podSelector` names. */
@@ -132,6 +966,33 @@ export interface EgressPolicyTarget {
132
966
  * carries it.
133
967
  */
134
968
  readonly sandboxTemplateName: string
969
+ /**
970
+ * The egress PROFILE label this policy also selects, when
971
+ * {@link KubernetesEgressConfig.profile} is set. Absent, the selector is
972
+ * the template label alone and every translation is what it always was.
973
+ *
974
+ * It is part of the SELECTOR rather than a second policy because that is
975
+ * what makes one warm pool serve several modes: pods out of one template
976
+ * carry one template label and differ only by this one, so
977
+ * `${template}-none-egress` selects exactly the `none` pods and
978
+ * `${template}-internet-egress` exactly the `internet` ones.
979
+ */
980
+ readonly profile?: EgressProfileLabel
981
+ }
982
+
983
+ /**
984
+ * What a translated policy's `podSelector`/`endpointSelector` matches: the
985
+ * template label, plus the profile label when one is configured. One
986
+ * function so the two manifest builders below and every reader of a
987
+ * translation agree on the selector down to the key order.
988
+ */
989
+ export function egressPolicySelectorLabels(
990
+ target: EgressPolicyTarget,
991
+ ): Readonly<Record<string, string>> {
992
+ return {
993
+ ...sandboxTemplateLabel(target.sandboxTemplateName),
994
+ ...(target.profile !== undefined ? { [target.profile.key]: target.profile.value } : {}),
995
+ }
135
996
  }
136
997
 
137
998
  /**
@@ -146,14 +1007,49 @@ export interface EgressPolicyTarget {
146
1007
  */
147
1008
  export interface KubernetesTranslatedEgressPolicy {
148
1009
  readonly kind: 'NetworkPolicy' | 'CiliumNetworkPolicy'
1010
+ /**
1011
+ * The CONFIGURED kind this came from, carried so a refusal can name what
1012
+ * the deployment asked for (`no-network`, `public-internet`, …) rather
1013
+ * than only the resource it produced.
1014
+ */
1015
+ readonly policyKind: KubernetesEgressPolicy['kind']
149
1016
  readonly namespace: string
150
1017
  readonly name: string
151
1018
  readonly manifest: Readonly<Record<string, unknown>>
152
1019
  }
153
1020
 
154
- /** `${sandboxTemplateName}-egress`, the name {@link KubernetesEgressConfig.networkPolicyName} defaults to. */
155
- export function defaultEgressPolicyName(sandboxTemplateName: string): string {
156
- return `${sandboxTemplateName}-egress`
1021
+ /**
1022
+ * The longest name a Kubernetes object may carry. A `NetworkPolicy` name is a
1023
+ * DNS subdomain, so 253 characters, and the API server refuses anything
1024
+ * longer.
1025
+ */
1026
+ const MAX_OBJECT_NAME_LENGTH = 253
1027
+
1028
+ /**
1029
+ * `${sandboxTemplateName}-egress`, the name
1030
+ * {@link KubernetesEgressConfig.networkPolicyName} defaults to — or
1031
+ * `${sandboxTemplateName}-${profile}-egress` when a profile is configured,
1032
+ * because one template under two profiles needs two policy objects and a
1033
+ * single default name would have the second silently verify against the
1034
+ * first's manifest.
1035
+ *
1036
+ * The profile is bounded at 63 characters on its own, but the CONCATENATION
1037
+ * is what has to be a legal object name, and only this function knows both
1038
+ * halves. Refused here, during host wiring, rather than as an API-server
1039
+ * rejection on the first policy GET of the first `create()`: `networkPolicyName`
1040
+ * is the way out and it is a config field, so this is a config error.
1041
+ */
1042
+ export function defaultEgressPolicyName(sandboxTemplateName: string, profile?: string): string {
1043
+ if (profile === undefined) return `${sandboxTemplateName}-egress`
1044
+ const name = `${sandboxTemplateName}-${profile}-egress`
1045
+ if (name.length > MAX_OBJECT_NAME_LENGTH) {
1046
+ throw new KubernetesEgressProfileConfigError(
1047
+ 'profile',
1048
+ profile,
1049
+ `makes the default egress policy name ${JSON.stringify(name)} ${name.length} characters long, past the ${MAX_OBJECT_NAME_LENGTH} an object name may carry — shorten the profile or the SandboxTemplate name, or set config.egress.networkPolicyName yourself`,
1050
+ )
1051
+ }
1052
+ return name
157
1053
  }
158
1054
 
159
1055
  /**
@@ -174,45 +1070,461 @@ export class KubernetesUnenforceableEgressPolicyError extends Error {
174
1070
  }
175
1071
 
176
1072
  /**
177
- * Synchronous, no-I/O precondition: can `policy.kind` be enforced under
178
- * `engine` at all. Deliberately decided from the KIND alone — a `resolver`
179
- * policy's `resolve()` is never invoked here, both because calling it just to
180
- * prove a refusal would be wasted work (and possibly a side effect the host
181
- * did not expect yet) and because this has to stay callable synchronously
182
- * from `buildKubernetesBackend`, which contacts nothing.
1073
+ * Named refusal for a config value this backend can read but not use — today
1074
+ * only a `public-internet` `exceptCidrs` entry that is not a CIDR. Thrown
1075
+ * SYNCHRONOUSLY from `buildKubernetesBackend`, beside
1076
+ * {@link KubernetesUnenforceableEgressPolicyError}, so a typo surfaces
1077
+ * during host wiring rather than as a policy the API server rejects when an
1078
+ * operator applies it.
183
1079
  */
184
- export function assertEgressPolicyIsEnforceable(
185
- policy: EgressPolicy,
186
- engine: KubernetesEgressEngine,
187
- ): void {
188
- if ((policy.kind === 'static' || policy.kind === 'resolver') && engine !== 'cilium') {
189
- throw new KubernetesUnenforceableEgressPolicyError(policy.kind)
1080
+ export class KubernetesEgressPolicyConfigError extends Error {
1081
+ override readonly name = 'KubernetesEgressPolicyConfigError'
1082
+
1083
+ constructor(
1084
+ readonly field: string,
1085
+ reason: string,
1086
+ ) {
1087
+ // `field` is relative to `config.egress`, not to `config.egress.policy`:
1088
+ // the fields that land here live at BOTH levels — `policy.exceptCidrs`
1089
+ // on the policy, `ciliumNarrowing` and `perSandbox.narrowing` beside
1090
+ // it — and a prefix that assumed one of them sent a reader to a key
1091
+ // that does not exist.
1092
+ super(`kubernetes: config.egress.${field} is unusable: ${reason}`)
190
1093
  }
191
1094
  }
192
1095
 
193
1096
  /**
194
- * The cluster DNS egress rule every translated `NetworkPolicy` carries,
195
- * including on `deny-all`. See the module doc's "Why a NetworkPolicy always
196
- * allows the cluster's own DNS" section.
1097
+ * Named refusal for `config.egress.ciliumNarrowing` set on a policy/engine
1098
+ * combination it does not apply to. Thrown SYNCHRONOUSLY from
1099
+ * {@link assertEgressPolicyIsEnforceable}, beside
1100
+ * {@link KubernetesUnenforceableEgressPolicyError}, so a narrowing option
1101
+ * that would silently do nothing is refused at host-wiring time rather than
1102
+ * accepted and ignored.
197
1103
  */
198
- const CLUSTER_DNS_EGRESS_RULE = {
199
- to: [{ namespaceSelector: { matchLabels: { 'kubernetes.io/metadata.name': 'kube-system' } } }],
200
- ports: [
201
- { protocol: 'UDP', port: 53 },
202
- { protocol: 'TCP', port: 53 },
203
- ],
204
- } as const
1104
+ export class KubernetesEgressNarrowingUnsupportedError extends Error {
1105
+ override readonly name = 'KubernetesEgressNarrowingUnsupportedError'
1106
+
1107
+ constructor(
1108
+ readonly policyKind: KubernetesEgressPolicy['kind'],
1109
+ readonly engine: KubernetesEgressEngine,
1110
+ ) {
1111
+ super(
1112
+ `kubernetes: config.egress.ciliumNarrowing is set, but config.egress.policy is '${policyKind}' under engine ${JSON.stringify(engine)}. Port, DNS-name and TLS-server-name narrowing only apply to a 'static' or 'resolver' hostname allowlist under engine: 'cilium' — set engine: 'cilium' with one of those policy kinds, or remove ciliumNarrowing. Refusing rather than silently ignoring an option that would never be applied.`,
1113
+ )
1114
+ }
1115
+ }
205
1116
 
206
1117
  /**
207
- * Cilium requires DNS lookups to be explicitly allowed AND made visible to
208
- * the agent before `toFQDNs` enforcement can match anything a name resolves
209
- * to: without a preceding rule granting DNS and turning on visibility, the
210
- * sandbox's own lookups for the allowed hosts are invisible to Cilium's
211
- * FQDN-to-IP mapping and `toFQDNs` matches nothing, which would make a
212
- * `static`/`resolver` policy fail closed for every host rather than only the
213
- * disallowed ones.
1118
+ * The grammar a host refusal closes with: what an `allowedHosts` entry is.
1119
+ * Every translation in this module implements it, which is why nothing turns
1120
+ * it off — see {@link KubernetesNetworkPolicyHostError}.
1121
+ */
1122
+ const HOSTNAME_GRAMMAR_SENTENCE =
1123
+ "Entries are hostnames: 'api.example.com' for one host, '.example.com' for that domain and its subdomains."
1124
+
1125
+ /**
1126
+ * Raised for an `allowedHosts` entry this backend will not translate.
214
1127
  *
215
- * Shape verified against Cilium's own worked example — the `toFQDNs`
1128
+ * Named for the ENTRY rather than for a config field, because entries arrive
1129
+ * from two directions: a host's `Sandbox.setNetworkPolicy` call, and the
1130
+ * `allowedHosts` of a `static`/`resolver` `config.egress.policy`.
1131
+ * `SandboxNetworkPolicy` names HOSTS: `api.example.com` for one host,
1132
+ * `.example.com` for a domain and its subdomains. A glob, a URL, a port
1133
+ * suffix or an address is refused rather than emitted, because a
1134
+ * `CiliumNetworkPolicy` carrying one is an object the API server rejects on
1135
+ * apply — or, worse, accepts as a name that resolves to nothing, which reads
1136
+ * from the outside exactly like a policy that is working.
1137
+ */
1138
+ export class KubernetesNetworkPolicyHostError extends Error {
1139
+ override readonly name = 'KubernetesNetworkPolicyHostError'
1140
+
1141
+ constructor(
1142
+ readonly host: string,
1143
+ reason: string,
1144
+ ) {
1145
+ super(
1146
+ `kubernetes: the allowedHosts entry ${JSON.stringify(host)} is refused: it ${reason}. ${HOSTNAME_GRAMMAR_SENTENCE} Nothing was written.`,
1147
+ )
1148
+ }
1149
+ }
1150
+
1151
+ /** A DNS name, lowercase, no scheme, no port, no wildcard. */
1152
+ const DNS_NAME = /^[a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*$/
1153
+
1154
+ /**
1155
+ * The one grammar an `allowedHosts` entry has to satisfy, on EVERY path that
1156
+ * emits one.
1157
+ *
1158
+ * It lives here, beside the class it throws, rather than in the per-sandbox
1159
+ * module that first needed it: it is called from {@link
1160
+ * buildCiliumEgressManifest} — the single builder both translations go
1161
+ * through — so the config-level allowlist and a `setNetworkPolicy` list are
1162
+ * refused the SAME entries, by name, with nothing emitted. That is the
1163
+ * contract {@link KubernetesNetworkPolicyHostError} already states — it
1164
+ * names entries rather than a config field, for exactly this reason — and a
1165
+ * grammar only ONE of the two translations enforces is not one the other has:
1166
+ * that is how the config-level translation came to pass `'.com'`, `'.'`,
1167
+ * `'..example.com'`, `'*'` and an IP literal straight into the emitted
1168
+ * `toFQDNs`, where the per-sandbox path refused every one of them.
1169
+ *
1170
+ * The per-sandbox writer still calls it too — {@link normalizeHost}, which
1171
+ * lowercases and then calls this — because there it also CANONICALISES the
1172
+ * bytes, before the fence is read and before anything is queued.
1173
+ */
1174
+ export function assertUsableHost(entry: string): void {
1175
+ // The two reasons the per-sandbox writer's own type check used to give
1176
+ // first, kept apart here so an untyped caller meets the same sentence
1177
+ // whichever path reached this — see the note on lowercasing below.
1178
+ if (typeof entry !== 'string') {
1179
+ throw new KubernetesNetworkPolicyHostError(String(entry), 'is not a string')
1180
+ }
1181
+ if (entry === '') {
1182
+ throw new KubernetesNetworkPolicyHostError(entry, 'is empty')
1183
+ }
1184
+ const bare = entry.startsWith('.') ? entry.slice(1) : entry
1185
+ if (bare === '') {
1186
+ throw new KubernetesNetworkPolicyHostError(entry, 'names no domain after its leading dot')
1187
+ }
1188
+ if (entry.includes('*')) {
1189
+ throw new KubernetesNetworkPolicyHostError(
1190
+ entry,
1191
+ "contains a glob; a domain and its subdomains are written with a leading dot ('.example.com'), which becomes matchName plus matchPattern",
1192
+ )
1193
+ }
1194
+ if (!DNS_NAME.test(bare)) {
1195
+ throw new KubernetesNetworkPolicyHostError(
1196
+ entry,
1197
+ 'is not a DNS name (a scheme, a path, a port suffix and an IP address all land here; letter case is canonicalised before this check, so it is never the cause)',
1198
+ )
1199
+ }
1200
+ if (bare.length > 253) {
1201
+ throw new KubernetesNetworkPolicyHostError(entry, 'is longer than a DNS name may be')
1202
+ }
1203
+ // A leading-dot entry becomes `matchPattern: '*.<domain>'`, and a
1204
+ // single-label domain there is a whole public suffix — `.com`, `.org`.
1205
+ // That is not an allowlist entry: it admits every name under a registry
1206
+ // the caller does not control. Cilium would match it if it were applied,
1207
+ // which is why this is a refusal rather than a lenient pass, and the
1208
+ // shipped admission fence refuses the same pattern for the per-sandbox
1209
+ // writes it binds (its `matchConditions` scope it to the sandbox host's
1210
+ // ServiceAccount, so an operator applying the config-level object is not
1211
+ // covered by it — one more reason the refusal has to come from here).
1212
+ if (entry.startsWith('.') && !bare.includes('.')) {
1213
+ throw new KubernetesNetworkPolicyHostError(
1214
+ entry,
1215
+ "names a whole top-level domain ('.com' means every name under it); a domain entry needs at least two labels, as in '.example.com', and the '*.com' pattern this would emit is one the shipped admission policy refuses for the per-sandbox writes it binds",
1216
+ )
1217
+ }
1218
+ // An IPv4 literal passes the grammar above — every label is digits, and
1219
+ // digits are legal in a DNS label. It is still not a hostname: a DNS
1220
+ // top-level label is never all-numeric, and Cilium's `matchName` is
1221
+ // compared against names the DNS proxy SAW, which an address never is. A
1222
+ // policy carrying one is admitted and matches nothing, which reads from
1223
+ // outside exactly like a policy that is working.
1224
+ const lastLabel = bare.slice(bare.lastIndexOf('.') + 1)
1225
+ if (/^[0-9]+$/.test(lastLabel)) {
1226
+ throw new KubernetesNetworkPolicyHostError(
1227
+ entry,
1228
+ 'ends in an all-numeric label, so it is an address rather than a hostname; toFQDNs matches names a DNS lookup returned, and an address is never one of them (use config.egress.policy for address-based egress)',
1229
+ )
1230
+ }
1231
+ }
1232
+
1233
+ /**
1234
+ * Every entry of one allowlist, judged by {@link assertUsableHost} — called
1235
+ * from {@link buildCiliumEgressManifest} before anything is emitted, so BOTH
1236
+ * translations refuse the same entries.
1237
+ *
1238
+ * The entry is lowercased for the DECISION and not for the bytes. The
1239
+ * per-sandbox writer canonicalises before it validates (`normalizeHost`), so
1240
+ * judging the lowercased form is what makes the two paths agree on WHICH
1241
+ * entries are refused — while the emitted `matchName` stays the string the
1242
+ * caller wrote, exactly as every release before this one emitted it. This
1243
+ * call refuses what no translation can express, not what one of them would
1244
+ * spell differently.
1245
+ */
1246
+ function assertHostsAreUsable(allowedHosts: readonly string[]): void {
1247
+ for (const host of allowedHosts) {
1248
+ assertUsableHost(typeof host === 'string' ? host.toLowerCase() : host)
1249
+ }
1250
+ }
1251
+
1252
+ /**
1253
+ * Every entry in `narrowing.ports`, `narrowing.hostPorts` and
1254
+ * `narrowing.tlsPorts` is a port the API server will actually accept, none of
1255
+ * those three is an explicitly empty array, and every DNS suffix
1256
+ * `narrowing.dnsNames` names is a non-empty string — checked here,
1257
+ * synchronously, for the same reason {@link assertEgressPolicyIsEnforceable}
1258
+ * checks `exceptCidrs`: a value the API server rejects on apply would leave a
1259
+ * policy that never verifies.
1260
+ */
1261
+ /**
1262
+ * Refuse a narrowing the API server would reject on apply, naming the field
1263
+ * that has to change.
1264
+ *
1265
+ * `fieldPath` is how the refusal spells the option's location, because the
1266
+ * same shape is configurable in two places now: `ciliumNarrowing` narrows the
1267
+ * CONFIG-level translated policy, and `perSandbox.narrowing` narrows the
1268
+ * per-sandbox policies a host writes through `setNetworkPolicy`. One
1269
+ * validator, because they are one shape and two would disagree the first time
1270
+ * either grew a field.
1271
+ */
1272
+ export function assertCiliumNarrowingIsUsable(
1273
+ narrowing: KubernetesCiliumEgressNarrowing,
1274
+ fieldPath = 'ciliumNarrowing',
1275
+ ): void {
1276
+ const assertValidPort = (port: unknown, field: string): void => {
1277
+ if (typeof port !== 'number' || !Number.isInteger(port) || port < 1 || port > 65535) {
1278
+ throw new KubernetesEgressPolicyConfigError(
1279
+ field,
1280
+ `${JSON.stringify(port)} is not a TCP port a NetworkPolicy/CiliumNetworkPolicy can carry (an integer 1-65535)`,
1281
+ )
1282
+ }
1283
+ }
1284
+ // `undefined` means "no restriction on this list" and is fine; `[]` is
1285
+ // neither that nor a usable restriction — `narrowedHostFqdnRule` would
1286
+ // emit `toPorts: [{ ports: [] }]` for it, a shape the API server rejects
1287
+ // on apply. Refuse it here rather than let a typo (`ports: []` where
1288
+ // `ports` was meant to be omitted) reach the cluster as a policy that
1289
+ // never verifies.
1290
+ const assertNotEmptyPortList = (ports: readonly number[] | undefined, field: string): void => {
1291
+ if (ports !== undefined && ports.length === 0) {
1292
+ throw new KubernetesEgressPolicyConfigError(
1293
+ field,
1294
+ 'must not be an empty array — omit the field entirely for no port restriction, or list at least one port',
1295
+ )
1296
+ }
1297
+ }
1298
+ assertNotEmptyPortList(narrowing.ports, `${fieldPath}.ports`)
1299
+ for (const port of narrowing.ports ?? []) assertValidPort(port, `${fieldPath}.ports`)
1300
+ for (const [host, ports] of Object.entries(narrowing.hostPorts ?? {})) {
1301
+ assertNotEmptyPortList(ports, `${fieldPath}.hostPorts[${JSON.stringify(host)}]`)
1302
+ for (const port of ports) {
1303
+ assertValidPort(port, `${fieldPath}.hostPorts[${JSON.stringify(host)}]`)
1304
+ }
1305
+ }
1306
+ assertNotEmptyPortList(narrowing.tlsPorts, `${fieldPath}.tlsPorts`)
1307
+ for (const port of narrowing.tlsPorts ?? []) assertValidPort(port, `${fieldPath}.tlsPorts`)
1308
+ if (typeof narrowing.dnsNames === 'object') {
1309
+ const { clusterDomain, searchSuffixes } = narrowing.dnsNames
1310
+ if (clusterDomain !== undefined && clusterDomain.trim() === '') {
1311
+ throw new KubernetesEgressPolicyConfigError(
1312
+ `${fieldPath}.dnsNames.clusterDomain`,
1313
+ 'must not be an empty string',
1314
+ )
1315
+ }
1316
+ for (const suffix of searchSuffixes ?? []) {
1317
+ if (typeof suffix !== 'string' || suffix.trim() === '') {
1318
+ throw new KubernetesEgressPolicyConfigError(
1319
+ `${fieldPath}.dnsNames.searchSuffixes`,
1320
+ `${JSON.stringify(suffix)} is not a usable DNS suffix`,
1321
+ )
1322
+ }
1323
+ }
1324
+ }
1325
+ }
1326
+
1327
+ /** Does `narrowing` actually ask for anything, or is every field off/absent. */
1328
+ function ciliumNarrowingIsActive(
1329
+ narrowing: KubernetesCiliumEgressNarrowing | undefined,
1330
+ ): narrowing is KubernetesCiliumEgressNarrowing {
1331
+ if (narrowing === undefined) return false
1332
+ if (narrowing.ports !== undefined) return true
1333
+ if (narrowing.hostPorts !== undefined && Object.keys(narrowing.hostPorts).length > 0) return true
1334
+ if (dnsNarrowingIsActive(narrowing.dnsNames)) return true
1335
+ if (narrowing.tlsServerNames === true) return true
1336
+ return false
1337
+ }
1338
+
1339
+ function dnsNarrowingIsActive(dnsNames: KubernetesCiliumEgressNarrowing['dnsNames']): boolean {
1340
+ return activeDnsNarrowing(dnsNames) !== undefined
1341
+ }
1342
+
1343
+ /** `dnsNames` reduced to the config it names, or `undefined` when it is off. `true` reduces to every default. */
1344
+ function activeDnsNarrowing(
1345
+ dnsNames: KubernetesCiliumEgressNarrowing['dnsNames'],
1346
+ ): KubernetesCiliumDnsNarrowing | undefined {
1347
+ if (dnsNames === true) return {}
1348
+ if (typeof dnsNames === 'object' && dnsNames !== null) return dnsNames
1349
+ return undefined
1350
+ }
1351
+
1352
+ /**
1353
+ * Synchronous, no-I/O precondition: can `policy.kind` be enforced under
1354
+ * `engine` at all, and — if `narrowing` is set — does it apply to this
1355
+ * policy/engine combination. Deliberately decided from the KIND alone — a
1356
+ * `resolver` policy's `resolve()` is never invoked here, both because calling
1357
+ * it just to prove a refusal would be wasted work (and possibly a side effect
1358
+ * the host did not expect yet) and because this has to stay callable
1359
+ * synchronously from `buildKubernetesBackend`, which contacts nothing.
1360
+ */
1361
+ export function assertEgressPolicyIsEnforceable(
1362
+ policy: KubernetesEgressPolicy,
1363
+ engine: KubernetesEgressEngine,
1364
+ narrowing?: KubernetesCiliumEgressNarrowing,
1365
+ ): void {
1366
+ if ((policy.kind === 'static' || policy.kind === 'resolver') && engine !== 'cilium') {
1367
+ throw new KubernetesUnenforceableEgressPolicyError(policy.kind)
1368
+ }
1369
+ if (policy.kind === 'public-internet') {
1370
+ for (const cidr of policy.exceptCidrs ?? []) {
1371
+ if (parseCidr(cidr) === undefined) {
1372
+ throw new KubernetesEgressPolicyConfigError(
1373
+ 'policy.exceptCidrs',
1374
+ `${JSON.stringify(cidr)} is not an IPv4 or IPv6 CIDR this backend can read. A NetworkPolicy ipBlock.except entry has to be a CIDR inside the block it carves out of, and an entry the API server rejects on apply would leave a policy that never verifies.`,
1375
+ )
1376
+ }
1377
+ }
1378
+ }
1379
+ if (ciliumNarrowingIsActive(narrowing)) {
1380
+ if (engine !== 'cilium' || (policy.kind !== 'static' && policy.kind !== 'resolver')) {
1381
+ throw new KubernetesEgressNarrowingUnsupportedError(policy.kind, engine)
1382
+ }
1383
+ assertCiliumNarrowingIsUsable(narrowing)
1384
+ }
1385
+ }
1386
+
1387
+ /**
1388
+ * The cluster DNS egress rule every translated `NetworkPolicy` carries,
1389
+ * including on `deny-all`. See the module doc's "Why a NetworkPolicy always
1390
+ * allows the cluster's own DNS" section.
1391
+ */
1392
+ const CLUSTER_DNS_EGRESS_RULE = {
1393
+ to: [
1394
+ {
1395
+ namespaceSelector: {
1396
+ matchLabels: { 'kubernetes.io/metadata.name': 'kube-system' },
1397
+ },
1398
+ },
1399
+ ],
1400
+ ports: [
1401
+ { protocol: 'UDP', port: 53 },
1402
+ { protocol: 'TCP', port: 53 },
1403
+ ],
1404
+ } as const
1405
+
1406
+ /**
1407
+ * DNS to the cluster resolver's own pods, which is narrower than
1408
+ * {@link CLUSTER_DNS_EGRESS_RULE}'s whole-namespace rule and is what
1409
+ * `'public-internet'` emits: that kind exists to name a destination set
1410
+ * precisely, so it names the resolver precisely too. `kube-system` +
1411
+ * `k8s-app: kube-dns` is the pairing CoreDNS ships under on every cluster
1412
+ * this was checked against, and the same pairing the repo's own
1413
+ * `k8s/manifests/networkpolicy.yaml` baseline already uses.
1414
+ *
1415
+ * `'deny-all'` keeps {@link CLUSTER_DNS_EGRESS_RULE} instead. It is not
1416
+ * changed to this one: its emitted manifest is pinned byte-for-byte, because
1417
+ * verification of the named object is an exact match and any change to the
1418
+ * translation fails every `create()` on every deployment that already
1419
+ * applied a policy.
1420
+ */
1421
+ const KUBE_DNS_EGRESS_RULE = {
1422
+ to: [
1423
+ {
1424
+ namespaceSelector: {
1425
+ matchLabels: { 'kubernetes.io/metadata.name': 'kube-system' },
1426
+ },
1427
+ podSelector: { matchLabels: { 'k8s-app': 'kube-dns' } },
1428
+ },
1429
+ ],
1430
+ ports: [
1431
+ { protocol: 'UDP', port: 53 },
1432
+ { protocol: 'TCP', port: 53 },
1433
+ ],
1434
+ } as const
1435
+
1436
+ /**
1437
+ * What `'public-internet'` carves out of `0.0.0.0/0`, and why each entry is
1438
+ * here. Order is part of the emitted manifest and therefore part of what
1439
+ * verification compares, so it is fixed rather than sorted at build time.
1440
+ *
1441
+ * - `10.0.0.0/8`, `172.16.0.0/12`, `192.168.0.0/16` — RFC 1918. The cluster
1442
+ * network, the node network and every other sandbox pod live in one of
1443
+ * them on every deployment this was checked against.
1444
+ * - `100.64.0.0/10` — RFC 6598 carrier-grade NAT, which several managed
1445
+ * Kubernetes offerings hand to pods or nodes. Missing from the policy the
1446
+ * agent-sandbox controller writes when a template declares no
1447
+ * `networkPolicy`, which is exactly why this check compares except lists
1448
+ * rather than trusting that a policy "looks restrictive".
1449
+ * - `169.254.0.0/16` — link-local, and with it `169.254.169.254`, the
1450
+ * instance-metadata address whose credentials are the reason a sandbox
1451
+ * reaching "only the internet" still must not reach sideways.
1452
+ * - `127.0.0.0/8` — loopback. Not routable off the pod, but a policy that
1453
+ * says "the public internet" should not say it admits loopback either.
1454
+ * - `168.63.129.16/32` — one cloud's platform endpoint, a single address
1455
+ * outside every range above that answers DNS and instance services.
1456
+ */
1457
+ const PUBLIC_INTERNET_EXCLUDED_IPV4_CIDRS = [
1458
+ '10.0.0.0/8',
1459
+ '172.16.0.0/12',
1460
+ '192.168.0.0/16',
1461
+ '100.64.0.0/10',
1462
+ '169.254.0.0/16',
1463
+ '127.0.0.0/8',
1464
+ '168.63.129.16/32',
1465
+ ] as const
1466
+
1467
+ /**
1468
+ * The IPv6 equivalents: unique-local (`fc00::/7`), link-local
1469
+ * (`fe80::/10`, which carries the v6 spelling of instance metadata) and
1470
+ * loopback (`::1/128`). A cluster with no IPv6 at all is unaffected by the
1471
+ * rule's presence — it names destinations nothing routes to.
1472
+ */
1473
+ const PUBLIC_INTERNET_EXCLUDED_IPV6_CIDRS = ['fc00::/7', 'fe80::/10', '::1/128'] as const
1474
+
1475
+ /**
1476
+ * The one rule `'public-internet'` adds beside DNS: everything, minus the
1477
+ * lists above, minus whatever the deployment added. No `ports`, because the
1478
+ * kind is a statement about DESTINATIONS — a deployment that also wants to
1479
+ * bound ports states that in its own applied policy, which this check then
1480
+ * accepts as narrower.
1481
+ *
1482
+ * A deployment's own `exceptCidrs` are routed to the block of their own
1483
+ * address family: a `NetworkPolicy` requires every `except` entry to sit
1484
+ * inside the `cidr` it carves out of, so an IPv6 entry under the IPv4 block
1485
+ * is rejected on apply.
1486
+ */
1487
+ function publicInternetEgressRule(
1488
+ exceptCidrs: readonly string[] | undefined,
1489
+ ): Readonly<Record<string, unknown>> {
1490
+ const extraV4: string[] = []
1491
+ const extraV6: string[] = []
1492
+ for (const cidr of exceptCidrs ?? []) {
1493
+ const parsed = parseCidr(cidr)
1494
+ // Unparseable entries are refused by `assertEgressPolicyIsEnforceable`
1495
+ // at construction; reaching here with one would mean this function was
1496
+ // called around it, so it is dropped rather than emitted.
1497
+ if (parsed === undefined) continue
1498
+ ;(parsed.version === 4 ? extraV4 : extraV6).push(cidr)
1499
+ }
1500
+ return {
1501
+ to: [
1502
+ {
1503
+ ipBlock: {
1504
+ cidr: '0.0.0.0/0',
1505
+ except: [...PUBLIC_INTERNET_EXCLUDED_IPV4_CIDRS, ...extraV4],
1506
+ },
1507
+ },
1508
+ {
1509
+ ipBlock: {
1510
+ cidr: '::/0',
1511
+ except: [...PUBLIC_INTERNET_EXCLUDED_IPV6_CIDRS, ...extraV6],
1512
+ },
1513
+ },
1514
+ ],
1515
+ }
1516
+ }
1517
+
1518
+ /**
1519
+ * Cilium requires DNS lookups to be explicitly allowed AND made visible to
1520
+ * the agent before `toFQDNs` enforcement can match anything a name resolves
1521
+ * to: without a preceding rule granting DNS and turning on visibility, the
1522
+ * sandbox's own lookups for the allowed hosts are invisible to Cilium's
1523
+ * FQDN-to-IP mapping and `toFQDNs` matches nothing, which would make a
1524
+ * `static`/`resolver` policy fail closed for every host rather than only the
1525
+ * disallowed ones.
1526
+ *
1527
+ * Shape verified against Cilium's own worked example — the `toFQDNs`
216
1528
  * `CiliumNetworkPolicy` at https://docs.cilium.io/en/stable/security/dns/
217
1529
  * (the "DNS Based" policy walkthrough) allows egress to `kube-dns` on port 53
218
1530
  * with `rules.dns: [{ matchPattern: "*" }]` alongside its own `toFQDNs` rule
@@ -240,10 +1552,12 @@ const CILIUM_DNS_VISIBILITY_RULE = {
240
1552
 
241
1553
  function buildCoreNetworkPolicy(
242
1554
  target: EgressPolicyTarget,
1555
+ policyKind: KubernetesEgressPolicy['kind'],
243
1556
  egress: readonly Readonly<Record<string, unknown>>[],
244
1557
  ): KubernetesTranslatedEgressPolicy {
245
1558
  return {
246
1559
  kind: 'NetworkPolicy',
1560
+ policyKind,
247
1561
  namespace: target.namespace,
248
1562
  name: target.name,
249
1563
  manifest: {
@@ -251,7 +1565,9 @@ function buildCoreNetworkPolicy(
251
1565
  kind: 'NetworkPolicy',
252
1566
  metadata: { name: target.name, namespace: target.namespace },
253
1567
  spec: {
254
- podSelector: { matchLabels: sandboxTemplateLabel(target.sandboxTemplateName) },
1568
+ podSelector: {
1569
+ matchLabels: egressPolicySelectorLabels(target),
1570
+ },
255
1571
  policyTypes: ['Egress'],
256
1572
  egress,
257
1573
  },
@@ -259,29 +1575,394 @@ function buildCoreNetworkPolicy(
259
1575
  }
260
1576
  }
261
1577
 
262
- function buildCiliumNetworkPolicy(
263
- target: EgressPolicyTarget,
1578
+ /** `[443]` — the TLS-port default for both `tlsServerNames` and `tlsPorts`. */
1579
+ const DEFAULT_TLS_PORTS = [443] as const
1580
+
1581
+ /** `{ port: '443', protocol: 'TCP' }` — Cilium's port shape, ports spelled as strings. */
1582
+ function ciliumPortEntry(port: number): Readonly<Record<string, unknown>> {
1583
+ return { port: String(port), protocol: 'TCP' }
1584
+ }
1585
+
1586
+ /**
1587
+ * Which field a host refusal sends the reader to — the ONE thing about a
1588
+ * refusal that is still a fact about the caller rather than about the entry.
1589
+ *
1590
+ * It was two facts while the two translations differed in what they emitted:
1591
+ * the sentence, which followed from whether the translation expanded a
1592
+ * leading-dot entry, and the field path, which followed from which field the
1593
+ * caller set. The config-level translation used to expand nothing, so it was
1594
+ * both a different sentence and a different field; it now expands exactly as
1595
+ * the per-sandbox one does, one entry no longer means two things, and the
1596
+ * sentence is the same one on every path — so the field path is all that is
1597
+ * left to get right, which is why it is the only field here.
1598
+ *
1599
+ * It still travels as one value rather than as a loose string argument,
1600
+ * because there are exactly two legal ones and both are named below: a refusal
1601
+ * is raised from the translation AND, earlier still, from the per-sandbox
1602
+ * writer before it reads the fence, and the two have to name the same field
1603
+ * for one entry however the earlier check is reached or skipped.
1604
+ */
1605
+ export interface HostsFitNarrowingContext {
1606
+ /**
1607
+ * The field the refusal sends the reader to, spelled as the caller set it
1608
+ * — the full path, as every other refusal in this module spells one.
1609
+ */
1610
+ readonly fieldPath: string
1611
+ }
1612
+
1613
+ /**
1614
+ * The refusal context of the PER-SANDBOX translation — `egress.perSandbox`,
1615
+ * whose allowlist comes from a host's own `setNetworkPolicy` call.
1616
+ *
1617
+ * Shared deliberately: the per-sandbox writer refuses a host by this context
1618
+ * before it reads the fence, and {@link buildCiliumEgressManifest} refuses the
1619
+ * same host by the same context on the way through the translation, so the two
1620
+ * cannot name different fields for one entry however the earlier check is
1621
+ * reached or skipped.
1622
+ */
1623
+ export const PER_SANDBOX_NARROWING_REFUSAL: HostsFitNarrowingContext = {
1624
+ fieldPath: 'config.egress.perSandbox.narrowing',
1625
+ }
1626
+
1627
+ /**
1628
+ * The refusal context of the config-level translation — `config.egress.policy`
1629
+ * and its `config.egress.ciliumNarrowing`.
1630
+ */
1631
+ export const CONFIG_LEVEL_NARROWING_REFUSAL: HostsFitNarrowingContext = {
1632
+ fieldPath: 'config.egress.ciliumNarrowing',
1633
+ }
1634
+
1635
+ /**
1636
+ * Refuse the one allowlist entry a narrowing option cannot express.
1637
+ *
1638
+ * `tlsServerNames` puts the entry on the rule as a TLS server name — one
1639
+ * exact SNI value a handshake presents — and a `.domain` entry is a set of
1640
+ * names no single SNI value means. Either way the emitted object would be
1641
+ * applied without complaint — nothing refuses an operator's apply: the
1642
+ * shipped admission fence's `matchConditions` scope it to the sandbox host's
1643
+ * ServiceAccount, so the config-level object is not matched by it — read back
1644
+ * deep-equal to what was sent, and deny what the caller asked to allow — the
1645
+ * failure every other refusal in this module exists to prevent. Refused
1646
+ * rather than translated another way,
1647
+ * because both other ways are guesses: dropping the subdomains silently
1648
+ * narrows what the caller asked for, and a wildcard SNI value is not
1649
+ * something the Cilium versions these manifests are written against are
1650
+ * known here to match — an SNI that matches nothing denies just as
1651
+ * completely, and more quietly.
1652
+ *
1653
+ * One sentence, because one entry now means one thing: both translations turn
1654
+ * `.domain` into `matchName: domain` PLUS `matchPattern: '*.domain'` — see
1655
+ * {@link ciliumFqdnEntries} — and no single SNI value means that pair.
1656
+ * `domain` alone denies every subdomain the pattern admits, and the entry as
1657
+ * written is not a name any handshake presents at all.
1658
+ *
1659
+ * `context` is the field the refusal sends the reader to — see
1660
+ * {@link HostsFitNarrowingContext}. A reader has to be sent to the field they
1661
+ * actually set: passed as a loose string, an expanding caller was told to
1662
+ * repair `config.egress.ciliumNarrowing` with the remedy that belongs to
1663
+ * `config.egress.perSandbox.narrowing`, a message whose only repair is a field
1664
+ * the entry never came from. Called from BOTH — the translation itself, so
1665
+ * every caller is covered, and the per-sandbox writer, which refuses earlier
1666
+ * still, before it reads the fence.
1667
+ */
1668
+ export function assertHostsFitNarrowing(
1669
+ allowedHosts: readonly string[],
1670
+ narrowing: KubernetesCiliumEgressNarrowing | undefined,
1671
+ context: HostsFitNarrowingContext,
1672
+ ): void {
1673
+ if (narrowing?.tlsServerNames !== true) return
1674
+ for (const host of allowedHosts) {
1675
+ if (!host.startsWith('.')) continue
1676
+ throw new KubernetesNetworkPolicyHostError(
1677
+ host,
1678
+ `names a domain and its subdomains while ${context.fieldPath}.tlsServerNames is on; a TLS server name is one exact SNI value a handshake presents, and this entry becomes a name plus a '*.domain' pattern, which no single value means — list the exact hosts, or leave tlsServerNames off for a domain list`,
1679
+ )
1680
+ }
1681
+ }
1682
+
1683
+ /**
1684
+ * The port list a narrowed translation applies to `host`, before splitting
1685
+ * out the TLS subset — see {@link KubernetesCiliumEgressNarrowing.ports}.
1686
+ * `undefined` means no port restriction: the host's `toFQDNs` rule carries
1687
+ * no `toPorts` at all.
1688
+ */
1689
+ function narrowedHostPorts(
1690
+ host: string,
1691
+ narrowing: KubernetesCiliumEgressNarrowing,
1692
+ ): readonly number[] | undefined {
1693
+ const explicit = narrowing.hostPorts?.[host] ?? narrowing.ports
1694
+ if (explicit !== undefined) return explicit
1695
+ // `serverNames` needs a port to attach to — leaving this host with no
1696
+ // port at all would mean either no `serverNames` rule (silently dropping
1697
+ // the option) or one with no `toPorts`, which Cilium rejects on apply.
1698
+ if (narrowing.tlsServerNames === true) return narrowing.tlsPorts ?? DEFAULT_TLS_PORTS
1699
+ return undefined
1700
+ }
1701
+
1702
+ /**
1703
+ * One host's `toFQDNs` rule under narrowing: the allowlist entry's `toFQDNs`
1704
+ * entries, plus `toPorts` split into a `serverNames`-bearing entry for the
1705
+ * host's TLS ports and a plain entry for whatever is left, when
1706
+ * `tlsServerNames` is on.
1707
+ *
1708
+ * `fqdns` is {@link ciliumFqdnEntries} of the host's allowlist entry, passed
1709
+ * in rather than derived here because the entry and the host are not always
1710
+ * the same string: `.example.com` becomes `example.com` plus
1711
+ * `*.example.com`. It is REQUIRED, with no `[{ matchName: host }]` default —
1712
+ * a caller that omitted it would emit the entry unexpanded, which is the
1713
+ * exact object this translation stopped producing.
1714
+ *
1715
+ * `serverNames` carries the allowlist ENTRY as written while the `toFQDNs`
1716
+ * half carries the expansion of it. A TLS server name is one exact SNI value
1717
+ * a handshake presents, and a `.example.com` entry — whose `toFQDNs` half is
1718
+ * `example.com` PLUS `*.example.com` — has no single SNI value meaning that
1719
+ * set: `example.com` would deny every subdomain the pattern admits, and
1720
+ * `.example.com` is not a name any handshake ever presents. So
1721
+ * {@link assertHostsFitNarrowing} refuses that one combination before this
1722
+ * rule is built, from `buildCiliumEgressManifest` for every caller and from
1723
+ * the per-sandbox writer before it reads the fence — which is why this
1724
+ * function can still take `host` and the entry as one string.
1725
+ */
1726
+ function narrowedHostFqdnRule(
1727
+ host: string,
1728
+ narrowing: KubernetesCiliumEgressNarrowing,
1729
+ fqdns: readonly Readonly<Record<string, string>>[],
1730
+ ): Readonly<Record<string, unknown>> {
1731
+ const ports = narrowedHostPorts(host, narrowing)
1732
+ if (ports === undefined) return { toFQDNs: fqdns }
1733
+ if (narrowing.tlsServerNames !== true) {
1734
+ return { toFQDNs: fqdns, toPorts: [{ ports: ports.map(ciliumPortEntry) }] }
1735
+ }
1736
+ const tlsPorts = narrowing.tlsPorts ?? DEFAULT_TLS_PORTS
1737
+ const tlsSubset = ports.filter((port) => tlsPorts.includes(port))
1738
+ const rest = ports.filter((port) => !tlsPorts.includes(port))
1739
+ const toPorts: Record<string, unknown>[] = []
1740
+ if (tlsSubset.length > 0) {
1741
+ toPorts.push({ ports: tlsSubset.map(ciliumPortEntry), serverNames: [host] })
1742
+ }
1743
+ if (rest.length > 0) toPorts.push({ ports: rest.map(ciliumPortEntry) })
1744
+ return {
1745
+ toFQDNs: fqdns,
1746
+ ...(toPorts.length > 0 ? { toPorts } : {}),
1747
+ }
1748
+ }
1749
+
1750
+ /**
1751
+ * One allowlist entry, expanded to the `toFQDNs` entries it means.
1752
+ *
1753
+ * `SandboxNetworkPolicy.allowedHosts`'s own grammar, which the SDK states —
1754
+ * `packages/sdk/src/types/sandbox/index.ts`'s `allowedHosts` — and which
1755
+ * `src/egress/allowlist.ts` implements for the docker backend: `api.example.com`
1756
+ * is that host, and `.example.com` is the domain AND its subdomains. Cilium's
1757
+ * `matchName` is an exact name and does not match across a `.`, so the domain
1758
+ * form needs the name plus a `matchPattern` — `*.example.com` alone would
1759
+ * admit `a.example.com` and not `example.com` itself.
1760
+ *
1761
+ * Used by BOTH translations. The config-level one used to emit a leading-dot
1762
+ * entry verbatim instead, which no DNS answer carries and so denied the
1763
+ * domain and every subdomain it was asked to allow, after a create that
1764
+ * reported success — but "the config-level bytes are pinned" was never a
1765
+ * contract the pinned bytes honoured, and one entry has to mean one thing on
1766
+ * every backend.
1767
+ */
1768
+ export function ciliumFqdnEntries(entry: string): readonly Readonly<Record<string, string>>[] {
1769
+ if (!entry.startsWith('.')) return [{ matchName: entry }]
1770
+ const domain = entry.slice(1)
1771
+ return [{ matchName: domain }, { matchPattern: `*.${domain}` }]
1772
+ }
1773
+
1774
+ /**
1775
+ * The DNS-visibility rule narrowed to the names an allowlist entry admits:
1776
+ * an exact `matchName` per allowed host plus the host under every search
1777
+ * suffix, and — for an entry that admits subdomains — the `matchPattern` that
1778
+ * admits them, under the bare name and under every suffix as well. Replaces
1779
+ * {@link CILIUM_DNS_VISIBILITY_RULE}'s `matchPattern: '*'`. See
1780
+ * {@link KubernetesCiliumDnsNarrowing}.
1781
+ *
1782
+ * Both halves follow from {@link ciliumFqdnEntries}: the DNS proxy has to see
1783
+ * exactly what the `toFQDNs` rule admits, or the half that learns addresses
1784
+ * from a lookup never sees one the other half allows.
1785
+ */
1786
+ function narrowedDnsVisibilityRule(
264
1787
  allowedHosts: readonly string[],
1788
+ fallbackNamespace: string,
1789
+ dnsNames: KubernetesCiliumDnsNarrowing,
1790
+ ): Readonly<Record<string, unknown>> {
1791
+ const namespace = dnsNames.namespace ?? fallbackNamespace
1792
+ const clusterDomain = dnsNames.clusterDomain ?? 'cluster.local'
1793
+ const suffixes = [
1794
+ `${namespace}.svc.${clusterDomain}`,
1795
+ `svc.${clusterDomain}`,
1796
+ clusterDomain,
1797
+ ...(dnsNames.searchSuffixes ?? []),
1798
+ ]
1799
+ const matchNames: Record<string, string>[] = []
1800
+ for (const entry of allowedHosts) {
1801
+ // A `.domain` entry resolves through its subdomains as well, so the
1802
+ // DNS proxy has to be allowed to SEE those lookups or `toFQDNs` never
1803
+ // learns the addresses they resolve to.
1804
+ const wildcard = entry.startsWith('.')
1805
+ const host = wildcard ? entry.slice(1) : entry
1806
+ matchNames.push({ matchName: host })
1807
+ if (wildcard) matchNames.push({ matchPattern: `*.${host}` })
1808
+ for (const suffix of suffixes) {
1809
+ matchNames.push({ matchName: `${host}.${suffix}` })
1810
+ // The suffixed names are here because a guest resolving a name
1811
+ // with fewer dots than the cluster's `ndots` tries the search
1812
+ // suffixes FIRST, and a lookup the DNS proxy refuses is not an
1813
+ // NXDOMAIN the resolver walks past — it can fail the whole
1814
+ // resolution. An entry that admits subdomains needs the pattern
1815
+ // under each suffix too, or `a.example.com` fails on its first
1816
+ // search-suffix attempt under a rule that allows it.
1817
+ if (wildcard) matchNames.push({ matchPattern: `*.${host}.${suffix}` })
1818
+ }
1819
+ }
1820
+ return {
1821
+ toEndpoints: CILIUM_DNS_VISIBILITY_RULE.toEndpoints,
1822
+ toPorts: [
1823
+ {
1824
+ ports: [{ port: '53', protocol: 'ANY' }],
1825
+ rules: { dns: matchNames },
1826
+ },
1827
+ ],
1828
+ }
1829
+ }
1830
+
1831
+ /**
1832
+ * Everything a `CiliumNetworkPolicy` for a hostname allowlist is built from.
1833
+ *
1834
+ * ONE builder, two callers: the config-level translation below, whose
1835
+ * selector is the template (and profile) label, and the per-sandbox policy in
1836
+ * `per-sandbox-policy.ts`, whose selector is one per-sandbox label and which
1837
+ * additionally carries an `ownerReferences` entry so the cluster
1838
+ * garbage-collects it. A second builder would be a second answer to what a
1839
+ * namzu egress policy looks like, and the read-back comparator would then be
1840
+ * verifying one of them against the other's shape.
1841
+ *
1842
+ * The two callers now differ in nothing this interface cannot spell out: the
1843
+ * same entries become the same bytes and a refusal uses the same sentence on
1844
+ * both, so what is left of "which caller this is" is
1845
+ * {@link CiliumEgressManifestOptions.refusalContext} — the field a refusal
1846
+ * names.
1847
+ */
1848
+ export interface CiliumEgressManifestOptions {
1849
+ readonly namespace: string
1850
+ readonly name: string
1851
+ /** What `spec.endpointSelector.matchLabels` carries, verbatim. */
1852
+ readonly selectorLabels: Readonly<Record<string, string>>
1853
+ readonly allowedHosts: readonly string[]
1854
+ /** The CONFIGURED kind this came from, for a refusal's wording. */
1855
+ readonly policyKind: KubernetesEgressPolicy['kind']
1856
+ readonly narrowing?: KubernetesCiliumEgressNarrowing
1857
+ /** `metadata.ownerReferences`. Absent ⇒ the metadata is what it always was. */
1858
+ readonly ownerReferences?: readonly KubernetesOwnerReference[]
1859
+ /**
1860
+ * The field a refusal for one of these hosts sends its reader to — see
1861
+ * {@link HostsFitNarrowingContext}. REQUIRED, not defaulted: there is no
1862
+ * way to tell from inside this function which config the caller's hosts
1863
+ * came from, and a refusal that names a field the caller never set is a
1864
+ * remedy that does not exist for them.
1865
+ *
1866
+ * The bytes are not this option's business any more, and neither is the
1867
+ * sentence a refusal uses: there is one translation of an allowlist entry,
1868
+ * so both are the same for every caller whatever this says.
1869
+ */
1870
+ readonly refusalContext: HostsFitNarrowingContext
1871
+ }
1872
+
1873
+ export function buildCiliumEgressManifest(
1874
+ options: CiliumEgressManifestOptions,
265
1875
  ): KubernetesTranslatedEgressPolicy {
1876
+ const { narrowing, allowedHosts, refusalContext } = options
1877
+ // Before anything is emitted, and for EVERY caller — which is the point:
1878
+ // an entry this backend will not translate is one whose object would be
1879
+ // either rejected on apply or, worse, accepted as a name that matches
1880
+ // nothing while the create reports success. The config-level allowlist
1881
+ // used to reach this builder unvalidated, so `['.com']`, `['.']`,
1882
+ // `['..example.com']`, `['*']` and an address were each emitted into
1883
+ // `toFQDNs` — the first three as a `matchPattern` no fence covers on that
1884
+ // path (the shipped one is scoped to the sandbox host's ServiceAccount),
1885
+ // which turned a fail-closed no-op into a silent grant of every name under
1886
+ // a public suffix. See {@link assertHostsAreUsable}.
1887
+ assertHostsAreUsable(allowedHosts)
1888
+ // A leading-dot entry under `tlsServerNames`
1889
+ // would become `serverNames: ['.domain']` — not a name any handshake
1890
+ // presents — and `['domain']` would deny every subdomain the `toFQDNs`
1891
+ // half of the same rule admits; the object goes through with nothing
1892
+ // objecting to it, and reads back deep-equal to what was sent, so nothing
1893
+ // downstream would ever report it. (On the config-level path nothing
1894
+ // objects to it at all: the shipped fence is scoped to the sandbox host's
1895
+ // ServiceAccount and an operator's apply is not matched by it.)
1896
+ //
1897
+ // What this call guarantees, per caller shape: the per-sandbox writer
1898
+ // refuses the same hosts earlier still, by name and by the same context
1899
+ // (`per-sandbox-policy.ts`), and this call covers them again if that check
1900
+ // is ever reached later or skipped; the config-level translation has no
1901
+ // earlier check of its own and relies on this one entirely; and a direct
1902
+ // call to this function is covered here too, whatever context it passes.
1903
+ //
1904
+ // The context decides ONE thing, because one thing is left to decide —
1905
+ // which field the refusal names. It is required rather than derived, since
1906
+ // nothing here can see the caller's config, and derived from the bytes is
1907
+ // exactly how an expanding caller came to be sent to the config-level
1908
+ // field for a repair that only exists under `perSandbox.narrowing`.
1909
+ assertHostsFitNarrowing(allowedHosts, narrowing, refusalContext)
1910
+ // Unnarrowed is the exact shape every release before #490 emitted — kept
1911
+ // as its own branch, untouched, rather than folded into the narrowed one
1912
+ // with every option defaulted off, so the byte-identical guarantee does
1913
+ // not depend on the narrowed code path happening to reduce to it.
1914
+ const narrowed = ciliumNarrowingIsActive(narrowing)
1915
+ const activeDns = narrowed ? activeDnsNarrowing(narrowing.dnsNames) : undefined
1916
+ const dnsRule =
1917
+ activeDns !== undefined
1918
+ ? narrowedDnsVisibilityRule(allowedHosts, options.namespace, activeDns)
1919
+ : CILIUM_DNS_VISIBILITY_RULE
1920
+ const hostRules = narrowed
1921
+ ? allowedHosts.map((host) => narrowedHostFqdnRule(host, narrowing, ciliumFqdnEntries(host)))
1922
+ : [{ toFQDNs: allowedHosts.flatMap(ciliumFqdnEntries) }]
1923
+
266
1924
  return {
267
1925
  kind: 'CiliumNetworkPolicy',
268
- namespace: target.namespace,
269
- name: target.name,
1926
+ policyKind: options.policyKind,
1927
+ namespace: options.namespace,
1928
+ name: options.name,
270
1929
  manifest: {
271
1930
  apiVersion: `${CILIUM_NETWORK_POLICY_API_GROUP}/${CILIUM_NETWORK_POLICY_API_VERSION}`,
272
1931
  kind: 'CiliumNetworkPolicy',
273
- metadata: { name: target.name, namespace: target.namespace },
1932
+ metadata: {
1933
+ name: options.name,
1934
+ namespace: options.namespace,
1935
+ ...(options.ownerReferences !== undefined
1936
+ ? { ownerReferences: options.ownerReferences }
1937
+ : {}),
1938
+ },
274
1939
  spec: {
275
- endpointSelector: { matchLabels: sandboxTemplateLabel(target.sandboxTemplateName) },
276
- egress: [
277
- CILIUM_DNS_VISIBILITY_RULE,
278
- { toFQDNs: allowedHosts.map((host) => ({ matchName: host })) },
279
- ],
1940
+ endpointSelector: {
1941
+ matchLabels: options.selectorLabels,
1942
+ },
1943
+ egress: [dnsRule, ...hostRules],
280
1944
  },
281
1945
  },
282
1946
  }
283
1947
  }
284
1948
 
1949
+ function buildCiliumNetworkPolicy(
1950
+ target: EgressPolicyTarget,
1951
+ policyKind: KubernetesEgressPolicy['kind'],
1952
+ allowedHosts: readonly string[],
1953
+ narrowing: KubernetesCiliumEgressNarrowing | undefined,
1954
+ ): KubernetesTranslatedEgressPolicy {
1955
+ return buildCiliumEgressManifest({
1956
+ namespace: target.namespace,
1957
+ name: target.name,
1958
+ selectorLabels: egressPolicySelectorLabels(target),
1959
+ allowedHosts,
1960
+ policyKind,
1961
+ ...(narrowing !== undefined ? { narrowing } : {}),
1962
+ refusalContext: CONFIG_LEVEL_NARROWING_REFUSAL,
1963
+ })
1964
+ }
1965
+
285
1966
  /**
286
1967
  * The pure translation: an {@link EgressPolicy} plus the declared
287
1968
  * {@link KubernetesEgressEngine} in, the concrete manifest this backend can
@@ -300,27 +1981,39 @@ function buildCiliumNetworkPolicy(
300
1981
  * value nothing downstream re-applies to the cluster.
301
1982
  */
302
1983
  export async function translateEgressPolicy(
303
- policy: EgressPolicy,
1984
+ policy: KubernetesEgressPolicy,
304
1985
  engine: KubernetesEgressEngine,
305
1986
  target: EgressPolicyTarget,
1987
+ ciliumNarrowing?: KubernetesCiliumEgressNarrowing,
306
1988
  ): Promise<KubernetesTranslatedEgressPolicy> {
307
- assertEgressPolicyIsEnforceable(policy, engine)
1989
+ assertEgressPolicyIsEnforceable(policy, engine, ciliumNarrowing)
308
1990
 
309
1991
  switch (policy.kind) {
310
1992
  case 'deny-all':
311
- return buildCoreNetworkPolicy(target, [CLUSTER_DNS_EGRESS_RULE])
1993
+ return buildCoreNetworkPolicy(target, 'deny-all', [CLUSTER_DNS_EGRESS_RULE])
1994
+ case 'no-network':
1995
+ // `policyTypes: ['Egress']` with an EMPTY rule list is the API's own
1996
+ // spelling of "this pod sends nothing": the pod is in egress
1997
+ // default-deny and no rule lets anything back out. Not even DNS —
1998
+ // see the kind's own doc.
1999
+ return buildCoreNetworkPolicy(target, 'no-network', [])
2000
+ case 'public-internet':
2001
+ return buildCoreNetworkPolicy(target, 'public-internet', [
2002
+ KUBE_DNS_EGRESS_RULE,
2003
+ publicInternetEgressRule(policy.exceptCidrs),
2004
+ ])
312
2005
  case 'allow-all':
313
2006
  // No `to`/`ports` on an egress rule matches every destination and
314
2007
  // every port. `CLUSTER_DNS_EGRESS_RULE` is a strict subset of this,
315
2008
  // so it is folded in rather than listed twice.
316
- return buildCoreNetworkPolicy(target, [{}])
2009
+ return buildCoreNetworkPolicy(target, 'allow-all', [{}])
317
2010
  case 'static':
318
2011
  // `assertEgressPolicyIsEnforceable` already threw above unless
319
2012
  // `engine === 'cilium'`, so reaching here means it did not.
320
- return buildCiliumNetworkPolicy(target, policy.allowedHosts)
2013
+ return buildCiliumNetworkPolicy(target, 'static', policy.allowedHosts, ciliumNarrowing)
321
2014
  case 'resolver': {
322
2015
  const allowedHosts = await policy.resolve()
323
- return buildCiliumNetworkPolicy(target, allowedHosts)
2016
+ return buildCiliumNetworkPolicy(target, 'resolver', allowedHosts, ciliumNarrowing)
324
2017
  }
325
2018
  default: {
326
2019
  const exhaustive: never = policy
@@ -373,14 +2066,28 @@ export class KubernetesEgressPolicyMismatchError extends Error {
373
2066
  * `../docker/index.ts`'s `assertNetworkCarriesThePolicy`, which inspects the
374
2067
  * daemon's own `{{.Internal}}` flag instead of trusting a network's name.
375
2068
  *
376
- * Checks exactly three things, each named separately in a mismatch so an
377
- * operator sees which one to fix:
2069
+ * Checks four things, each named separately in a mismatch so an operator sees
2070
+ * which one to fix:
2071
+ * - the object carries NO `specs` list. A `CiliumNetworkPolicy` carries
2072
+ * EITHER one `spec` or a `specs` list and a rule in either one enforces,
2073
+ * while every check below reads `spec` alone — so an object carrying both
2074
+ * would be compared on half of what it enforces. This is refused and not
2075
+ * read, because a comparison that accepts rules it never looked at is the
2076
+ * one answer this function must not give; the shipped admission fence
2077
+ * refuses a `specs` list for the same reason. The translation never emits
2078
+ * one, so its presence is drift and not an alternative spelling;
378
2079
  * - the selector (`podSelector` for `NetworkPolicy`, `endpointSelector` for
379
2080
  * `CiliumNetworkPolicy`) carries the expected template label:
380
2081
  * - `NetworkPolicy` additionally declares `policyTypes` including
381
2082
  * `'Egress'` — a `NetworkPolicy` with an `egress` array but no `'Egress'`
382
2083
  * in `policyTypes` enforces nothing on egress at all;
383
- * - the `egress` rule array matches the translation exactly.
2084
+ * - the `egress` rule array matches the translation exactly;
2085
+ * - and, ONLY when the translation carries `metadata.ownerReferences` (the
2086
+ * per-sandbox policies of `per-sandbox-policy.ts`, never the operator's
2087
+ * own object), that the live object still carries each of them — an owner
2088
+ * reference dropped between the write and the read is a policy the
2089
+ * cluster will never collect with the sandbox it belongs to, which is the
2090
+ * leak the reference exists to prevent, and it is invisible in `spec`.
384
2091
  *
385
2092
  * Never mutates and never creates — a 404/410 is refused, not repaired.
386
2093
  */
@@ -394,7 +2101,13 @@ export async function verifyEgressPolicyApplied(
394
2101
  ? ciliumNetworkPolicyPath(translated.namespace, translated.name)
395
2102
  : networkPolicyPath(translated.namespace, translated.name)
396
2103
 
397
- let resource: { readonly spec?: Readonly<Record<string, unknown>> } | undefined
2104
+ let resource:
2105
+ | {
2106
+ readonly spec?: Readonly<Record<string, unknown>>
2107
+ readonly specs?: unknown
2108
+ readonly metadata?: Readonly<Record<string, unknown>>
2109
+ }
2110
+ | undefined
398
2111
  try {
399
2112
  resource = await client.request('GET', path, undefined, signal)
400
2113
  } catch (err) {
@@ -408,6 +2121,33 @@ export async function verifyEgressPolicyApplied(
408
2121
  const actualSpec = resource?.spec ?? {}
409
2122
  const selectorKey = translated.kind === 'CiliumNetworkPolicy' ? 'endpointSelector' : 'podSelector'
410
2123
 
2124
+ // First, and before anything is compared: an object carrying a `specs`
2125
+ // list is refused rather than read. Everything below reads `spec`, and a
2126
+ // `CiliumNetworkPolicy` rule in a `specs` entry enforces exactly as one in
2127
+ // `spec` does — so comparing `spec` and reporting a match would be
2128
+ // accepting rules this function never looked at, which is the one answer it
2129
+ // must not give. `readCiliumEgressPolicies` reads both spellings and
2130
+ // `decideEgressUnion` refuses a `specs` entry that widens; this is the
2131
+ // other half of the same rule, for the callers that run this check ALONE —
2132
+ // `verify: 'named-object-only'`, and the per-sandbox read-back — and it is
2133
+ // what makes the named-object exemption in `decideEgressUnion` sound rather
2134
+ // than merely narrower. The translation never emits a `specs` list, and the
2135
+ // shipped admission fence refuses one for the same reason.
2136
+ const actualSpecs = resource?.specs
2137
+ if (actualSpecs !== undefined && actualSpecs !== null) {
2138
+ throw new KubernetesEgressPolicyMismatchError(
2139
+ translated.kind,
2140
+ path,
2141
+ `specs is ${JSON.stringify(actualSpecs)}, and the translation never emits one — ${
2142
+ translated.kind === 'CiliumNetworkPolicy'
2143
+ ? 'a CiliumNetworkPolicy carries EITHER one spec or a specs list, and a rule in either one enforces'
2144
+ : 'a NetworkPolicy has no specs field at all'
2145
+ }, while this check reads spec.${selectorKey}${
2146
+ translated.kind === 'NetworkPolicy' ? ', spec.policyTypes' : ''
2147
+ } and spec.egress and nothing else — so anything a specs list enforces is egress this comparison never read`,
2148
+ )
2149
+ }
2150
+
411
2151
  if (!isDeepStrictEqual(actualSpec[selectorKey], expectedSpec[selectorKey])) {
412
2152
  throw new KubernetesEgressPolicyMismatchError(
413
2153
  translated.kind,
@@ -434,4 +2174,1543 @@ export async function verifyEgressPolicyApplied(
434
2174
  `spec.egress is ${JSON.stringify(actualSpec.egress)}, expected ${JSON.stringify(expectedSpec.egress)}`,
435
2175
  )
436
2176
  }
2177
+
2178
+ // Last, and only for a translation that asked for one: an object written
2179
+ // with an owner reference and read back without it is a policy that
2180
+ // outlives its sandbox. The operator-applied objects this function's other
2181
+ // caller checks carry none, so they never reach this branch.
2182
+ const expectedOwners = (
2183
+ translated.manifest as { readonly metadata?: { readonly ownerReferences?: readonly unknown[] } }
2184
+ ).metadata?.ownerReferences
2185
+ if (expectedOwners !== undefined) {
2186
+ const actualOwners = (resource?.metadata as { readonly ownerReferences?: readonly unknown[] })
2187
+ ?.ownerReferences
2188
+ const held = Array.isArray(actualOwners) ? actualOwners : []
2189
+ for (const owner of expectedOwners) {
2190
+ if (held.some((entry) => isDeepStrictEqual(entry, owner))) continue
2191
+ throw new KubernetesEgressPolicyMismatchError(
2192
+ translated.kind,
2193
+ path,
2194
+ `metadata.ownerReferences is ${JSON.stringify(actualOwners)}, expected it to carry ${JSON.stringify(owner)} — without that entry the cluster never collects this policy with the object it belongs to, so it would outlive the sandbox it was written for`,
2195
+ )
2196
+ }
2197
+ }
2198
+ }
2199
+
2200
+ // ---------------------------------------------------------------------------
2201
+ // CIDR arithmetic
2202
+ // ---------------------------------------------------------------------------
2203
+ //
2204
+ // Enough of it to answer one question: is this policy's block inside the
2205
+ // block the translation allows. Written here rather than taken from a
2206
+ // dependency because `packages/sandbox` declares no runtime dependency and
2207
+ // this is forty lines of integer arithmetic.
2208
+
2209
+ /** A CIDR as a masked base address and a prefix length. */
2210
+ export interface ParsedCidr {
2211
+ readonly version: 4 | 6
2212
+ readonly base: bigint
2213
+ readonly bits: number
2214
+ }
2215
+
2216
+ function parseIpv4(text: string): bigint | undefined {
2217
+ const parts = text.split('.')
2218
+ if (parts.length !== 4) return undefined
2219
+ let value = 0n
2220
+ for (const part of parts) {
2221
+ if (!/^\d{1,3}$/.test(part)) return undefined
2222
+ const octet = Number(part)
2223
+ if (octet > 255) return undefined
2224
+ value = (value << 8n) | BigInt(octet)
2225
+ }
2226
+ return value
2227
+ }
2228
+
2229
+ function parseIpv6(text: string): bigint | undefined {
2230
+ // An embedded IPv4 tail (`::ffff:10.0.0.1`) is legal and is how a
2231
+ // dual-stack cluster spells a v4 address in a v6 field.
2232
+ let head = text
2233
+ let tail = 0n
2234
+ let tailGroups = 0
2235
+ const lastColon = text.lastIndexOf(':')
2236
+ if (lastColon >= 0 && text.slice(lastColon + 1).includes('.')) {
2237
+ const embedded = parseIpv4(text.slice(lastColon + 1))
2238
+ if (embedded === undefined) return undefined
2239
+ head = text.slice(0, lastColon + 1)
2240
+ tail = embedded
2241
+ tailGroups = 2
2242
+ }
2243
+ const halves = head.split('::')
2244
+ if (halves.length > 2) return undefined
2245
+ const readGroups = (part: string): bigint[] | undefined => {
2246
+ if (part === '') return []
2247
+ const groups: bigint[] = []
2248
+ for (const group of part.split(':')) {
2249
+ if (group === '') continue
2250
+ if (!/^[0-9a-fA-F]{1,4}$/.test(group)) return undefined
2251
+ groups.push(BigInt(Number.parseInt(group, 16)))
2252
+ }
2253
+ return groups
2254
+ }
2255
+ const left = readGroups(halves[0] ?? '')
2256
+ const right = readGroups(halves[1] ?? '')
2257
+ if (left === undefined || right === undefined) return undefined
2258
+ const present = left.length + right.length + tailGroups
2259
+ if (present > 8) return undefined
2260
+ if (halves.length === 1 && present !== 8) return undefined
2261
+ const zeros = 8 - present
2262
+ const groups = [...left, ...Array.from({ length: zeros }, () => 0n), ...right]
2263
+ let value = 0n
2264
+ for (const group of groups) value = (value << 16n) | group
2265
+ if (tailGroups === 2) value = (value << 32n) | tail
2266
+ return value
2267
+ }
2268
+
2269
+ /**
2270
+ * A CIDR, or `undefined` when it is not one this check can read. Host bits
2271
+ * are masked off rather than rejected: `10.0.0.1/8` and `10.0.0.0/8` name the
2272
+ * same block, and an operator who wrote the first meant the second.
2273
+ */
2274
+ export function parseCidr(text: unknown): ParsedCidr | undefined {
2275
+ if (typeof text !== 'string') return undefined
2276
+ const slash = text.indexOf('/')
2277
+ if (slash < 0) return undefined
2278
+ const address = text.slice(0, slash)
2279
+ const prefix = text.slice(slash + 1)
2280
+ if (!/^\d{1,3}$/.test(prefix)) return undefined
2281
+ const bits = Number(prefix)
2282
+ if (address.includes(':')) {
2283
+ const value = parseIpv6(address)
2284
+ if (value === undefined || bits > 128) return undefined
2285
+ return { version: 6, base: maskAddress(value, bits, 128), bits }
2286
+ }
2287
+ const value = parseIpv4(address)
2288
+ if (value === undefined || bits > 32) return undefined
2289
+ return { version: 4, base: maskAddress(value, bits, 32), bits }
2290
+ }
2291
+
2292
+ function maskAddress(value: bigint, bits: number, width: number): bigint {
2293
+ if (bits === 0) return 0n
2294
+ const host = BigInt(width - bits)
2295
+ return (value >> host) << host
2296
+ }
2297
+
2298
+ /** Every address of `inner` is an address of `outer`. */
2299
+ function cidrContains(outer: ParsedCidr, inner: ParsedCidr): boolean {
2300
+ if (outer.version !== inner.version) return false
2301
+ if (outer.bits > inner.bits) return false
2302
+ const width = outer.version === 4 ? 32 : 128
2303
+ return maskAddress(inner.base, outer.bits, width) === outer.base
2304
+ }
2305
+
2306
+ /** They share at least one address — i.e. one contains the other. */
2307
+ function cidrsOverlap(a: ParsedCidr, b: ParsedCidr): boolean {
2308
+ return cidrContains(a, b) || cidrContains(b, a)
2309
+ }
2310
+
2311
+ // ---------------------------------------------------------------------------
2312
+ // What the configured translation permits
2313
+ // ---------------------------------------------------------------------------
2314
+
2315
+ /** One port, or a range of them, on one protocol. `start` absent is every port. */
2316
+ interface PortRange {
2317
+ /** `undefined` is EVERY protocol — how Cilium spells `ANY`. Core defaults to TCP. */
2318
+ readonly protocol?: string
2319
+ readonly start?: number
2320
+ readonly end?: number
2321
+ }
2322
+
2323
+ /** A port list, `'all'` for the absent/empty one, `'unreadable'` for a shape this check cannot read. */
2324
+ type PortSet = 'all' | readonly PortRange[] | 'unreadable'
2325
+
2326
+ /** A destination, as either side of the comparison names it. */
2327
+ type PolicyPeer =
2328
+ | { readonly kind: 'everything' }
2329
+ | {
2330
+ readonly kind: 'cidr'
2331
+ readonly cidr: ParsedCidr
2332
+ readonly except: readonly ParsedCidr[]
2333
+ readonly text: string
2334
+ }
2335
+ | {
2336
+ readonly kind: 'selector'
2337
+ readonly namespaceSelector?: unknown
2338
+ readonly podSelector?: unknown
2339
+ readonly text: string
2340
+ }
2341
+
2342
+ interface AllowedDestination {
2343
+ readonly peer: PolicyPeer
2344
+ readonly ports: 'all' | readonly PortRange[]
2345
+ }
2346
+
2347
+ /** One `toFQDNs` entry the translation allows, with the ports it allows it on. */
2348
+ interface AllowedFqdn {
2349
+ readonly host: string
2350
+ readonly ports: 'all' | readonly PortRange[]
2351
+ }
2352
+
2353
+ /**
2354
+ * The cluster resolver's identity as a `PolicyPeer` — shared by
2355
+ * {@link egressAllowance} (reading a Cilium DNS-visibility rule into its
2356
+ * core-shaped equivalent) and {@link reachesResolverAtDnsPort} (deciding
2357
+ * whether some OTHER policy's rule reaches it), so there is one definition of
2358
+ * "this peer is the cluster resolver" rather than two that could drift apart.
2359
+ */
2360
+ const RESOLVER_PEER: PolicyPeer = {
2361
+ kind: 'selector',
2362
+ namespaceSelector: {
2363
+ matchLabels: { 'kubernetes.io/metadata.name': 'kube-system' },
2364
+ },
2365
+ podSelector: { matchLabels: { 'k8s-app': 'kube-dns' } },
2366
+ text: 'the cluster resolver',
2367
+ }
2368
+
2369
+ /**
2370
+ * Everything the configured translation lets out, in the one shape the union
2371
+ * check compares against.
2372
+ *
2373
+ * Built from the manifest {@link translateEgressPolicy} just produced for
2374
+ * every core kind, so there is no second statement anywhere of what
2375
+ * `deny-all` or `public-internet` permit — the emitted object IS the
2376
+ * statement, and a test asserts each translation is within its own
2377
+ * allowance. The Cilium kinds are the one exception and say so: a
2378
+ * `CiliumNetworkPolicy`'s DNS-visibility rule is read into the core-shaped
2379
+ * destination it is equivalent to (UDP/TCP 53 to the resolver's pods), so
2380
+ * that a core `NetworkPolicy` allowing exactly cluster DNS is not reported as
2381
+ * widening a hostname allowlist that already allows it.
2382
+ */
2383
+ export interface EgressAllowance {
2384
+ readonly destinations: readonly AllowedDestination[]
2385
+ /**
2386
+ * `toFQDNs` names a `static`/`resolver` translation allows, each with the
2387
+ * ports it allows that name on (`'all'` for the unnarrowed shape, which
2388
+ * puts no `toPorts` on its `toFQDNs` rule at all). Empty for every core
2389
+ * kind. A name can appear more than once — one entry per `toFQDNs` rule
2390
+ * naming it — and it is allowed on a port if ANY entry covers that port,
2391
+ * the same "any matching destination" rule {@link destinationIsAllowed}
2392
+ * applies to CIDR and selector peers.
2393
+ */
2394
+ readonly fqdns: readonly AllowedFqdn[]
2395
+ /**
2396
+ * The exact DNS names a `static`/`resolver` translation's kube-dns rule
2397
+ * restricts LOOKUPS to when `ciliumNarrowing.dnsNames` is set — the
2398
+ * `matchName` list {@link narrowedDnsVisibilityRule} builds. `'all'` for
2399
+ * every translation that does not narrow DNS: every core kind (which
2400
+ * cannot express an L7 DNS restriction at all) and an unnarrowed
2401
+ * `static`/`resolver` (`rules.dns: [{ matchPattern: '*' }]`).
2402
+ *
2403
+ * `destinations` alone cannot carry this: reachability to the resolver's
2404
+ * peer and port is the same whether or not DNS is narrowed, so a peer/port
2405
+ * check reports a plain kube-dns rule as `within` a narrowed translation
2406
+ * exactly as it would an unnarrowed one. {@link reachesResolverAtDnsPort}
2407
+ * is the check that actually reads this field, in both
2408
+ * {@link coreEgressRuleVerdict} and {@link ciliumEgressRuleVerdict}.
2409
+ */
2410
+ readonly dnsNarrowedTo: 'all' | readonly string[]
2411
+ readonly permitsEverything: boolean
2412
+ readonly permitsNothing: boolean
2413
+ /** The configured kind, for the refusal message. */
2414
+ readonly policyKind: KubernetesEgressPolicy['kind']
2415
+ /**
2416
+ * The object this allowance is the allowance OF — `translated.kind` and
2417
+ * `translated.name`, the object `config.egress` translates to and an
2418
+ * operator applies.
2419
+ *
2420
+ * The union rule reads every OTHER policy in the namespace against this
2421
+ * allowance, and this pair is what says which of the listed objects is the
2422
+ * named one, so {@link decideEgressUnion} can leave that object's `spec`
2423
+ * document to {@link verifyEgressPolicyApplied} instead of judging it here.
2424
+ * See {@link EgressPolicyDocument.namedObject} for what that comparison
2425
+ * reads, which documents it covers, and why the exemption is not an
2426
+ * optimisation.
2427
+ */
2428
+ readonly namedObject: {
2429
+ readonly kind: 'NetworkPolicy' | 'CiliumNetworkPolicy'
2430
+ readonly name: string
2431
+ }
2432
+ }
2433
+
2434
+ function readPortRanges(ports: unknown, defaultProtocol: string | undefined): PortSet {
2435
+ const entries = readList(ports)
2436
+ if (entries === 'unreadable') return 'unreadable'
2437
+ // Absent or empty `ports` on an egress rule means EVERY port.
2438
+ if (entries === undefined || entries.length === 0) return 'all'
2439
+ const ranges: PortRange[] = []
2440
+ for (const entry of entries) {
2441
+ if (!isRecord(entry)) return 'unreadable'
2442
+ const rawProtocol = entry.protocol
2443
+ if (rawProtocol !== undefined && typeof rawProtocol !== 'string') return 'unreadable'
2444
+ const protocol =
2445
+ rawProtocol === undefined
2446
+ ? defaultProtocol
2447
+ : rawProtocol.toUpperCase() === 'ANY'
2448
+ ? undefined
2449
+ : rawProtocol.toUpperCase()
2450
+ const endPort = entry.endPort
2451
+ if (endPort !== undefined && typeof endPort !== 'number') return 'unreadable'
2452
+ const raw = entry.port
2453
+ if (raw === undefined) {
2454
+ ranges.push(protocol === undefined ? {} : { protocol })
2455
+ continue
2456
+ }
2457
+ const port = typeof raw === 'number' ? raw : typeof raw === 'string' ? Number(raw) : Number.NaN
2458
+ // A NAMED container port. Resolving it needs the destination pod's own
2459
+ // container spec, which this check never has.
2460
+ if (!Number.isInteger(port)) return 'unreadable'
2461
+ ranges.push({
2462
+ ...(protocol !== undefined ? { protocol } : {}),
2463
+ start: port,
2464
+ ...(typeof endPort === 'number' ? { end: endPort } : {}),
2465
+ })
2466
+ }
2467
+ return ranges
2468
+ }
2469
+
2470
+ function readCorePeer(peer: unknown): PolicyPeer | undefined {
2471
+ if (!isRecord(peer)) return undefined
2472
+ const { ipBlock, podSelector, namespaceSelector } = peer
2473
+ if (ipBlock !== undefined) {
2474
+ if (!isRecord(ipBlock)) return undefined
2475
+ const cidr = parseCidr(ipBlock.cidr)
2476
+ if (cidr === undefined) return undefined
2477
+ const rawExcept = readList(ipBlock.except)
2478
+ if (rawExcept === 'unreadable') return undefined
2479
+ const except: ParsedCidr[] = []
2480
+ for (const entry of rawExcept ?? []) {
2481
+ const parsed = parseCidr(entry)
2482
+ if (parsed === undefined) return undefined
2483
+ except.push(parsed)
2484
+ }
2485
+ return { kind: 'cidr', cidr, except, text: JSON.stringify(ipBlock) }
2486
+ }
2487
+ if (podSelector === undefined && namespaceSelector === undefined) {
2488
+ // A peer naming none of the three constrains nothing. The API server
2489
+ // rejects it on admission, so an object carrying one did not come from
2490
+ // there and is not read as anything.
2491
+ return undefined
2492
+ }
2493
+ if (!selectorIsReadable(podSelector) || !selectorIsReadable(namespaceSelector)) return undefined
2494
+ return {
2495
+ kind: 'selector',
2496
+ ...(namespaceSelector !== undefined ? { namespaceSelector } : {}),
2497
+ ...(podSelector !== undefined ? { podSelector } : {}),
2498
+ text: JSON.stringify(peer),
2499
+ }
2500
+ }
2501
+
2502
+ /**
2503
+ * The allowance a translated policy expresses. See {@link EgressAllowance}.
2504
+ */
2505
+ export function egressAllowance(translated: KubernetesTranslatedEgressPolicy): EgressAllowance {
2506
+ const spec = (translated.manifest as { readonly spec: Record<string, unknown> }).spec
2507
+ // This module built the manifest a line ago, so `spec.egress` is always a
2508
+ // list here; the narrowing exists so the allowance is derived from the
2509
+ // object itself rather than from a second statement of what each kind
2510
+ // permits, and an impossible shape yields an allowance that permits
2511
+ // nothing rather than one that permits anything.
2512
+ const emitted = readList(spec.egress)
2513
+ const rules = emitted === undefined || emitted === 'unreadable' ? [] : emitted
2514
+ if (translated.kind === 'CiliumNetworkPolicy') {
2515
+ const fqdns: AllowedFqdn[] = []
2516
+ // The DNS-visibility rule is the one entry among `rules` whose
2517
+ // `toEndpoints` names the cluster resolver — narrowed or not, it always
2518
+ // reuses `CILIUM_DNS_VISIBILITY_RULE.toEndpoints` verbatim (see
2519
+ // `narrowedDnsVisibilityRule` and `buildCiliumNetworkPolicy`), so a deep
2520
+ // equality check finds it regardless of which branch built this rule.
2521
+ const dnsVisibilityRule = rules.find(
2522
+ (rule): rule is Readonly<Record<string, unknown>> =>
2523
+ isRecord(rule) &&
2524
+ isDeepStrictEqual(rule.toEndpoints, CILIUM_DNS_VISIBILITY_RULE.toEndpoints),
2525
+ )
2526
+ for (const rule of rules) {
2527
+ if (!isRecord(rule)) continue
2528
+ const hostNames = readList(rule.toFQDNs)
2529
+ if (hostNames === undefined || hostNames === 'unreadable') continue
2530
+ // This module built `rule` a few lines above (see the comment at the
2531
+ // top of this function), so its `toPorts` is always in the shape
2532
+ // `readCiliumRulePorts` reads — 'unreadable' cannot happen for a
2533
+ // manifest this file emitted, and 'all' is the safe fallback if it
2534
+ // somehow did, since that is what an absent `toPorts` also means.
2535
+ const portsRead = readCiliumRulePorts(rule)
2536
+ const ports = portsRead.ok ? portsRead.ports : 'all'
2537
+ for (const entry of hostNames) {
2538
+ if (isRecord(entry) && typeof entry.matchName === 'string') {
2539
+ fqdns.push({ host: entry.matchName, ports })
2540
+ }
2541
+ }
2542
+ }
2543
+ // This module built `dnsVisibilityRule` above (when it exists at all —
2544
+ // every Cilium translation this function reads carries one), so
2545
+ // `readCiliumRuleDnsNames` failing to read it, or reading a wildcard,
2546
+ // cannot mean anything other than "not narrowed": 'all' is the safe
2547
+ // fallback, the same reasoning `readCiliumRulePorts`'s own fallback above
2548
+ // already relies on.
2549
+ const dnsNamesRead =
2550
+ dnsVisibilityRule === undefined ? undefined : readCiliumRuleDnsNames(dnsVisibilityRule)
2551
+ const dnsNarrowedTo: 'all' | readonly string[] =
2552
+ dnsNamesRead?.ok === true && dnsNamesRead.names !== 'all' ? dnsNamesRead.names : 'all'
2553
+ return {
2554
+ // The core-shaped reading of CILIUM_DNS_VISIBILITY_RULE — the one
2555
+ // place in this module where one resource's rule is restated in the
2556
+ // other's vocabulary, so that a core policy allowing exactly cluster
2557
+ // DNS is not reported as widening a translation that already allows
2558
+ // it. Nothing else about a Cilium translation is restated: an
2559
+ // allowlist of names has no core spelling at all.
2560
+ destinations: [{ peer: RESOLVER_PEER, ports: [{ start: 53 }] }],
2561
+ fqdns,
2562
+ dnsNarrowedTo,
2563
+ permitsEverything: false,
2564
+ permitsNothing: false,
2565
+ policyKind: translated.policyKind,
2566
+ namedObject: { kind: translated.kind, name: translated.name },
2567
+ }
2568
+ }
2569
+ const destinations: AllowedDestination[] = []
2570
+ let permitsEverything = false
2571
+ for (const rule of rules) {
2572
+ if (!isRecord(rule)) continue
2573
+ const ports = readPortRanges(rule.ports, 'TCP')
2574
+ if (ports === 'unreadable') continue
2575
+ const to = readList(rule.to)
2576
+ if (to === 'unreadable') continue
2577
+ if (to === undefined || to.length === 0) {
2578
+ destinations.push({ peer: { kind: 'everything' }, ports })
2579
+ if (ports === 'all') permitsEverything = true
2580
+ continue
2581
+ }
2582
+ for (const peer of to) {
2583
+ const read = readCorePeer(peer)
2584
+ if (read !== undefined) destinations.push({ peer: read, ports })
2585
+ }
2586
+ }
2587
+ return {
2588
+ destinations,
2589
+ fqdns: [],
2590
+ // A core `NetworkPolicy` translation never narrows DNS — it has no L7
2591
+ // concept to narrow with — so the DNS-widening check in
2592
+ // `coreEgressRuleVerdict`/`ciliumEgressRuleVerdict` never fires against
2593
+ // this allowance.
2594
+ dnsNarrowedTo: 'all',
2595
+ permitsEverything,
2596
+ permitsNothing: rules.length === 0,
2597
+ policyKind: translated.policyKind,
2598
+ namedObject: { kind: translated.kind, name: translated.name },
2599
+ }
2600
+ }
2601
+
2602
+ // ---------------------------------------------------------------------------
2603
+ // Is one policy's rule inside the translation
2604
+ // ---------------------------------------------------------------------------
2605
+
2606
+ function protocolCovers(allowed: string | undefined, wanted: string | undefined): boolean {
2607
+ // `undefined` on the allowed side is EVERY protocol; on the wanted side it
2608
+ // is also every protocol, which only an every-protocol allowance covers.
2609
+ return allowed === undefined || allowed === wanted
2610
+ }
2611
+
2612
+ function portRangeCovers(allowed: PortRange, wanted: PortRange): boolean {
2613
+ if (!protocolCovers(allowed.protocol, wanted.protocol)) return false
2614
+ if (allowed.start === undefined) return true
2615
+ if (wanted.start === undefined) return false
2616
+ const allowedEnd = allowed.end ?? allowed.start
2617
+ const wantedEnd = wanted.end ?? wanted.start
2618
+ return wanted.start >= allowed.start && wantedEnd <= allowedEnd
2619
+ }
2620
+
2621
+ function portsCover(allowed: 'all' | readonly PortRange[], wanted: PortRange): boolean {
2622
+ if (allowed === 'all') return true
2623
+ return allowed.some((range) => portRangeCovers(range, wanted))
2624
+ }
2625
+
2626
+ /** Every constraint `outer` places, `inner` places too — so `inner` selects a subset. */
2627
+ function selectorIsNarrower(outer: unknown, inner: unknown): boolean {
2628
+ if (outer === undefined) return true
2629
+ if (!isRecord(outer)) return false
2630
+ const outerLabels = outer.matchLabels
2631
+ const outerExpressions = readList(outer.matchExpressions)
2632
+ if (outerExpressions === 'unreadable') return false
2633
+ if (outerLabels === undefined && (outerExpressions ?? []).length === 0) {
2634
+ // An EMPTY selector is "everything of this kind", which every selector
2635
+ // of that kind is inside.
2636
+ return true
2637
+ }
2638
+ if (inner === undefined || !isRecord(inner)) return false
2639
+ if (outerLabels !== undefined) {
2640
+ if (!isRecord(outerLabels)) return false
2641
+ const innerLabels = inner.matchLabels
2642
+ if (!isRecord(innerLabels)) return false
2643
+ for (const [key, value] of Object.entries(outerLabels)) {
2644
+ if (!Object.hasOwn(innerLabels, key) || innerLabels[key] !== value) return false
2645
+ }
2646
+ }
2647
+ const innerExpressions = readList(inner.matchExpressions)
2648
+ if (innerExpressions === 'unreadable') return false
2649
+ for (const expression of outerExpressions ?? []) {
2650
+ if (!(innerExpressions ?? []).some((candidate) => isDeepStrictEqual(candidate, expression))) {
2651
+ return false
2652
+ }
2653
+ }
2654
+ return true
2655
+ }
2656
+
2657
+ /** Is every address `wanted` admits also admitted by `allowed`. */
2658
+ function peerIsWithin(allowed: PolicyPeer, wanted: PolicyPeer): boolean {
2659
+ if (allowed.kind === 'everything') return true
2660
+ if (wanted.kind === 'everything') return false
2661
+ if (allowed.kind === 'cidr') {
2662
+ if (wanted.kind !== 'cidr') return false
2663
+ if (!cidrContains(allowed.cidr, wanted.cidr)) return false
2664
+ // Every hole the allowance carves out of its block has to be a hole in
2665
+ // this peer too, or this peer reaches an address the translation does
2666
+ // not allow. Coverage by a UNION of the peer's own `except` entries is
2667
+ // not attempted: one entry has to contain it.
2668
+ for (const hole of allowed.except) {
2669
+ if (!cidrsOverlap(hole, wanted.cidr)) continue
2670
+ if (!wanted.except.some((own) => cidrContains(own, hole))) return false
2671
+ }
2672
+ return true
2673
+ }
2674
+ if (wanted.kind !== 'selector') return false
2675
+ // A namespaceSelector the allowance omits means "this namespace"; a peer
2676
+ // that names namespaces reaches further than that.
2677
+ if (allowed.namespaceSelector === undefined && wanted.namespaceSelector !== undefined) {
2678
+ return false
2679
+ }
2680
+ return (
2681
+ selectorIsNarrower(allowed.namespaceSelector, wanted.namespaceSelector) &&
2682
+ selectorIsNarrower(allowed.podSelector, wanted.podSelector)
2683
+ )
2684
+ }
2685
+
2686
+ function destinationIsAllowed(
2687
+ allowance: EgressAllowance,
2688
+ peer: PolicyPeer,
2689
+ ports: 'all' | readonly PortRange[],
2690
+ ): boolean {
2691
+ const wanted: readonly PortRange[] = ports === 'all' ? [{}] : ports
2692
+ return wanted.every((range) =>
2693
+ allowance.destinations.some(
2694
+ (destination) => peerIsWithin(destination.peer, peer) && portsCover(destination.ports, range),
2695
+ ),
2696
+ )
2697
+ }
2698
+
2699
+ /** Is every port `wanted` reaches on `host` also allowed by some `fqdns` entry naming it. */
2700
+ function fqdnIsAllowed(
2701
+ allowance: EgressAllowance,
2702
+ host: string,
2703
+ ports: 'all' | readonly PortRange[],
2704
+ ): boolean {
2705
+ const wanted: readonly PortRange[] = ports === 'all' ? [{}] : ports
2706
+ return wanted.every((range) =>
2707
+ allowance.fqdns.some((entry) => entry.host === host && portsCover(entry.ports, range)),
2708
+ )
2709
+ }
2710
+
2711
+ /** Does `ports` (as a CANDIDATE rule's own port list) reach TCP or UDP 53 at all. */
2712
+ function coversDnsPort(ports: 'all' | readonly PortRange[]): boolean {
2713
+ if (ports === 'all') return true
2714
+ return ports.some((range) => {
2715
+ if (range.protocol !== undefined && range.protocol !== 'UDP' && range.protocol !== 'TCP') {
2716
+ return false
2717
+ }
2718
+ if (range.start === undefined) return true
2719
+ return range.start <= 53 && (range.end ?? range.start) >= 53
2720
+ })
2721
+ }
2722
+
2723
+ /**
2724
+ * Does a candidate rule's `peer`+`ports` reach the cluster resolver on the DNS
2725
+ * port at all — the precondition for the DNS-widening check both
2726
+ * {@link coreEgressRuleVerdict} and {@link ciliumEgressRuleVerdict} apply
2727
+ * before falling back to the ordinary peer/port `destinationIsAllowed` check.
2728
+ */
2729
+ function reachesResolverAtDnsPort(peer: PolicyPeer, ports: 'all' | readonly PortRange[]): boolean {
2730
+ return peerIsWithin(peer, RESOLVER_PEER) && coversDnsPort(ports)
2731
+ }
2732
+
2733
+ /** One egress rule, judged against the translation. */
2734
+ export interface EgressRuleVerdict {
2735
+ readonly beyond: boolean | 'unknown'
2736
+ readonly detail?: string
2737
+ }
2738
+
2739
+ function describePeer(peer: PolicyPeer): string {
2740
+ if (peer.kind === 'everything') return 'every destination'
2741
+ return peer.text
2742
+ }
2743
+
2744
+ function describePorts(ports: 'all' | readonly PortRange[]): string {
2745
+ if (ports === 'all') return 'every port'
2746
+ return ports
2747
+ .map((range) =>
2748
+ range.start === undefined
2749
+ ? `every ${range.protocol ?? ''} port`.trim()
2750
+ : `${range.protocol ?? 'any'} ${range.start}${range.end !== undefined ? `-${range.end}` : ''}`,
2751
+ )
2752
+ .join(', ')
2753
+ }
2754
+
2755
+ function coreEgressRuleVerdict(rule: unknown, allowance: EgressAllowance): EgressRuleVerdict {
2756
+ // Under a translation that permits nothing, the rule's own shape does not
2757
+ // matter and is not read: ANY egress rule on a policy selecting this pod
2758
+ // lets something out.
2759
+ if (allowance.permitsNothing) {
2760
+ return {
2761
+ beyond: true,
2762
+ detail: 'an egress rule, where the configured translation permits no egress at all',
2763
+ }
2764
+ }
2765
+ if (!isRecord(rule))
2766
+ return {
2767
+ beyond: 'unknown',
2768
+ detail: 'an egress rule that is not an object',
2769
+ }
2770
+ const ports = readPortRanges(rule.ports, 'TCP')
2771
+ if (ports === 'unreadable') {
2772
+ return {
2773
+ beyond: 'unknown',
2774
+ detail: 'a ports entry this check cannot read',
2775
+ }
2776
+ }
2777
+ const to = readList(rule.to)
2778
+ if (to === 'unreadable') {
2779
+ return { beyond: 'unknown', detail: "a 'to' that is not a list of peers" }
2780
+ }
2781
+ if (to === undefined || to.length === 0) {
2782
+ // No `to` on an egress rule means EVERY destination.
2783
+ return allowance.permitsEverything
2784
+ ? { beyond: false }
2785
+ : {
2786
+ beyond: true,
2787
+ detail: `no 'to' peers, so every destination is reachable on ${describePorts(ports)}`,
2788
+ }
2789
+ }
2790
+ for (const peer of to) {
2791
+ const read = readCorePeer(peer)
2792
+ if (read === undefined) {
2793
+ return {
2794
+ beyond: 'unknown',
2795
+ detail: `a 'to' peer this check cannot read (${JSON.stringify(peer)})`,
2796
+ }
2797
+ }
2798
+ // A plain `NetworkPolicy` has no L7 concept, so it cannot express the DNS
2799
+ // restriction `ciliumNarrowing.dnsNames` narrows to — a core rule
2800
+ // reaching the resolver on the DNS port always resolves every name, and
2801
+ // that is wider than a narrowed translation whatever its own peer/port
2802
+ // shape says. This has to be decided BEFORE `destinationIsAllowed`
2803
+ // below: that check only reasons about reachability, and our own
2804
+ // translation's `destinations` entry for the resolver is peer/port-only
2805
+ // too, so a plain kube-dns rule would otherwise read as `within` a
2806
+ // translation it actually resolves every name for.
2807
+ if (allowance.dnsNarrowedTo !== 'all' && reachesResolverAtDnsPort(read, ports)) {
2808
+ return {
2809
+ beyond: true,
2810
+ detail: `a 'to' peer ${describePeer(read)} on ${describePorts(ports)} reaching the cluster resolver's DNS port with no DNS-name restriction — a plain NetworkPolicy cannot narrow lookups the way the configured translation's ciliumNarrowing.dnsNames does, so this rule resolves every name the narrowed policy does not`,
2811
+ }
2812
+ }
2813
+ if (!destinationIsAllowed(allowance, read, ports)) {
2814
+ const wideOpen = corePeerIsWideOpen(peer) === true
2815
+ return {
2816
+ beyond: true,
2817
+ detail: `${wideOpen ? 'a wide-open ' : 'a '}'to' peer ${describePeer(read)} on ${describePorts(ports)}, which a '${allowance.policyKind}' translation does not allow`,
2818
+ }
2819
+ }
2820
+ }
2821
+ return { beyond: false }
2822
+ }
2823
+
2824
+ /** Every `to…` field the Cilium CRD declares. A rule naming none of them is port-only. */
2825
+ const CILIUM_DESTINATION_FIELDS = [
2826
+ 'toEndpoints',
2827
+ 'toEntities',
2828
+ 'toCIDR',
2829
+ 'toCIDRSet',
2830
+ 'toFQDNs',
2831
+ 'toServices',
2832
+ 'toGroups',
2833
+ 'toNodes',
2834
+ ] as const
2835
+
2836
+ /**
2837
+ * The label set a Cilium endpoint selector names, read back as the core
2838
+ * `namespaceSelector`/`podSelector` pair it is equivalent to, so the one
2839
+ * `peerIsWithin` serves both resource kinds. `undefined` when the selector
2840
+ * uses anything this reading cannot map — a label source that is not a pod
2841
+ * label, or a match expression.
2842
+ */
2843
+ function ciliumEndpointPeer(selector: unknown): PolicyPeer | undefined {
2844
+ if (!isRecord(selector)) return undefined
2845
+ if (selector.matchExpressions !== undefined) return undefined
2846
+ const matchLabels = selector.matchLabels
2847
+ if (!isRecord(matchLabels)) return undefined
2848
+ const podLabels: Record<string, string> = {}
2849
+ let namespace: string | undefined
2850
+ for (const [rawKey, value] of Object.entries(matchLabels)) {
2851
+ if (typeof value !== 'string') return undefined
2852
+ const key = ciliumSelectorKey(rawKey)
2853
+ if (key === undefined) return undefined
2854
+ if (key === 'io.kubernetes.pod.namespace') {
2855
+ namespace = value
2856
+ continue
2857
+ }
2858
+ podLabels[key] = value
2859
+ }
2860
+ return {
2861
+ kind: 'selector',
2862
+ ...(namespace !== undefined
2863
+ ? {
2864
+ namespaceSelector: {
2865
+ matchLabels: { 'kubernetes.io/metadata.name': namespace },
2866
+ },
2867
+ }
2868
+ : {}),
2869
+ podSelector: { matchLabels: podLabels },
2870
+ text: JSON.stringify(selector),
2871
+ }
2872
+ }
2873
+
2874
+ /**
2875
+ * A Cilium rule's `toPorts`, read into the same {@link PortSet} core rules
2876
+ * use. Shared by {@link ciliumEgressRuleVerdict} (a CANDIDATE rule read off
2877
+ * the cluster) and {@link egressAllowance} (OUR OWN translated rule) so
2878
+ * there is one reading of "what ports does this Cilium rule reach", not two
2879
+ * that could silently disagree.
2880
+ */
2881
+ type CiliumRulePorts =
2882
+ | { readonly ok: true; readonly ports: 'all' | readonly PortRange[] }
2883
+ | { readonly ok: false; readonly detail: string }
2884
+
2885
+ function readCiliumRulePorts(rule: Readonly<Record<string, unknown>>): CiliumRulePorts {
2886
+ // A Cilium rule carries its ports one level deeper, and a port entry with
2887
+ // no protocol means ANY rather than TCP.
2888
+ const toPorts = readList(rule.toPorts)
2889
+ if (toPorts === 'unreadable') {
2890
+ return { ok: false, detail: 'a toPorts that is not a list' }
2891
+ }
2892
+ const portEntries: unknown[] = []
2893
+ for (const entry of toPorts ?? []) {
2894
+ if (!isRecord(entry)) {
2895
+ return { ok: false, detail: 'a toPorts entry that is not an object' }
2896
+ }
2897
+ const list = readList(entry.ports)
2898
+ if (list === 'unreadable') {
2899
+ return { ok: false, detail: 'a toPorts ports field that is not a list' }
2900
+ }
2901
+ // An entry with no `ports` at all bounds nothing, so the rule reaches
2902
+ // every port — exactly what an absent `toPorts` means.
2903
+ if (list === undefined || list.length === 0) {
2904
+ portEntries.length = 0
2905
+ break
2906
+ }
2907
+ portEntries.push(...list)
2908
+ }
2909
+ const ports = readPortRanges(portEntries, undefined)
2910
+ if (ports === 'unreadable') {
2911
+ return { ok: false, detail: 'a toPorts entry this check cannot read' }
2912
+ }
2913
+ return { ok: true, ports }
2914
+ }
2915
+
2916
+ /**
2917
+ * A Cilium rule's `toPorts[].rules.dns`, read into the exact-name list it
2918
+ * restricts lookups to, or `'all'` when the rule does not restrict DNS at
2919
+ * all. Shared by {@link egressAllowance} (reading OUR OWN narrowed
2920
+ * DNS-visibility rule into {@link EgressAllowance.dnsNarrowedTo}) and
2921
+ * {@link ciliumEgressRuleVerdict} (deciding whether a CANDIDATE rule's own
2922
+ * restriction is narrow enough to not widen it) — one reading of "what names
2923
+ * does this Cilium rule let resolve", not two that could disagree.
2924
+ *
2925
+ * A `toPorts` entry with no `rules` at all, or a `rules.dns` entry carrying
2926
+ * `matchPattern` rather than `matchName`, both read as `'all'`: an absent L7
2927
+ * restriction resolves every name by definition, and this check does not
2928
+ * attempt to decide whether some wildcard pattern is a subset of an exact
2929
+ * name list — `'all'` is the conservative (never under-counts a widening)
2930
+ * answer for a shape it cannot reduce further.
2931
+ */
2932
+ function readCiliumRuleDnsNames(
2933
+ rule: Readonly<Record<string, unknown>>,
2934
+ ): { readonly ok: true; readonly names: 'all' | readonly string[] } | { readonly ok: false } {
2935
+ const toPorts = readList(rule.toPorts)
2936
+ if (toPorts === 'unreadable') return { ok: false }
2937
+ const names: string[] = []
2938
+ for (const entry of toPorts ?? []) {
2939
+ if (!isRecord(entry)) return { ok: false }
2940
+ if (entry.rules === undefined) return { ok: true, names: 'all' }
2941
+ if (!isRecord(entry.rules)) return { ok: false }
2942
+ const dns = readList(entry.rules.dns)
2943
+ if (dns === 'unreadable') return { ok: false }
2944
+ if (dns === undefined) return { ok: true, names: 'all' }
2945
+ for (const item of dns) {
2946
+ if (!isRecord(item)) return { ok: false }
2947
+ if (typeof item.matchName === 'string') {
2948
+ names.push(item.matchName)
2949
+ continue
2950
+ }
2951
+ // `matchPattern` (Cilium's glob syntax) or anything else this reading
2952
+ // does not recognise — both are read as unrestricted rather than
2953
+ // guessed at, per the doc comment above.
2954
+ return { ok: true, names: 'all' }
2955
+ }
2956
+ }
2957
+ return { ok: true, names }
2958
+ }
2959
+
2960
+ function ciliumEgressRuleVerdict(rule: unknown, allowance: EgressAllowance): EgressRuleVerdict {
2961
+ if (allowance.permitsNothing) {
2962
+ return {
2963
+ beyond: true,
2964
+ detail: 'an egress rule, where the configured translation permits no egress at all',
2965
+ }
2966
+ }
2967
+ if (!isRecord(rule))
2968
+ return {
2969
+ beyond: 'unknown',
2970
+ detail: 'an egress rule that is not an object',
2971
+ }
2972
+ const portsRead = readCiliumRulePorts(rule)
2973
+ if (!portsRead.ok) {
2974
+ return { beyond: 'unknown', detail: portsRead.detail }
2975
+ }
2976
+ const ports = portsRead.ports
2977
+ const fields = new Map<(typeof CILIUM_DESTINATION_FIELDS)[number], readonly unknown[]>()
2978
+ for (const field of CILIUM_DESTINATION_FIELDS) {
2979
+ const list = readList(rule[field])
2980
+ if (list === 'unreadable') {
2981
+ return {
2982
+ beyond: 'unknown',
2983
+ detail: `a ${field} that is not a list of peers`,
2984
+ }
2985
+ }
2986
+ if (list !== undefined && list.length > 0) fields.set(field, list)
2987
+ }
2988
+ if (fields.size === 0) {
2989
+ return allowance.permitsEverything
2990
+ ? { beyond: false }
2991
+ : {
2992
+ beyond: true,
2993
+ detail: `a port-only egress rule, so every destination is reachable on ${describePorts(ports)}`,
2994
+ }
2995
+ }
2996
+ for (const entity of fields.get('toEntities') ?? []) {
2997
+ if (typeof entity !== 'string') {
2998
+ return {
2999
+ beyond: 'unknown',
3000
+ detail: 'a toEntities entry that is not an entity name',
3001
+ }
3002
+ }
3003
+ if (!allowance.permitsEverything) {
3004
+ return {
3005
+ beyond: true,
3006
+ detail: `toEntities '${entity}', which a '${allowance.policyKind}' translation does not allow`,
3007
+ }
3008
+ }
3009
+ }
3010
+ for (const field of ['toCIDR', 'toCIDRSet'] as const) {
3011
+ for (const entry of fields.get(field) ?? []) {
3012
+ const peer =
3013
+ field === 'toCIDR'
3014
+ ? readCorePeer({ ipBlock: { cidr: entry } })
3015
+ : isRecord(entry)
3016
+ ? readCorePeer({
3017
+ ipBlock: { cidr: entry.cidr, except: entry.except },
3018
+ })
3019
+ : undefined
3020
+ if (peer === undefined) {
3021
+ return {
3022
+ beyond: 'unknown',
3023
+ detail: `a ${field} entry this check cannot read`,
3024
+ }
3025
+ }
3026
+ if (!destinationIsAllowed(allowance, peer, ports)) {
3027
+ return {
3028
+ beyond: true,
3029
+ detail: `${field} ${describePeer(peer)} on ${describePorts(ports)}, which a '${allowance.policyKind}' translation does not allow`,
3030
+ }
3031
+ }
3032
+ }
3033
+ }
3034
+ for (const entry of fields.get('toFQDNs') ?? []) {
3035
+ if (!isRecord(entry)) {
3036
+ return {
3037
+ beyond: 'unknown',
3038
+ detail: 'a toFQDNs entry that is not an object',
3039
+ }
3040
+ }
3041
+ const matchName = entry.matchName
3042
+ if (typeof matchName !== 'string' || !fqdnIsAllowed(allowance, matchName, ports)) {
3043
+ return {
3044
+ beyond: true,
3045
+ detail: `toFQDNs ${JSON.stringify(entry)} on ${describePorts(ports)}, which a '${allowance.policyKind}' translation does not allow`,
3046
+ }
3047
+ }
3048
+ }
3049
+ for (const selector of fields.get('toEndpoints') ?? []) {
3050
+ const peer = ciliumEndpointPeer(selector)
3051
+ if (peer === undefined) {
3052
+ return {
3053
+ beyond: 'unknown',
3054
+ detail: 'a toEndpoints entry this check cannot read as a selector',
3055
+ }
3056
+ }
3057
+ // See the matching comment in `coreEgressRuleVerdict`: reachability alone
3058
+ // cannot tell a plain kube-dns rule from a narrowed one, so this has to
3059
+ // run before `destinationIsAllowed` below. Unlike a core rule, a Cilium
3060
+ // one CAN narrow DNS on its own `toPorts.rules.dns` — read it and accept
3061
+ // the rule only when what it names is a subset of what our own
3062
+ // translation narrows to.
3063
+ if (allowance.dnsNarrowedTo !== 'all' && reachesResolverAtDnsPort(peer, ports)) {
3064
+ const narrowedTo = allowance.dnsNarrowedTo
3065
+ const candidate = readCiliumRuleDnsNames(rule)
3066
+ const isSubset =
3067
+ candidate.ok &&
3068
+ candidate.names !== 'all' &&
3069
+ candidate.names.every((n) => narrowedTo.includes(n))
3070
+ if (!isSubset) {
3071
+ return {
3072
+ beyond: true,
3073
+ detail: `toEndpoints ${describePeer(peer)} on ${describePorts(ports)} reaching the cluster resolver's DNS port with ${candidate.ok && candidate.names !== 'all' ? 'a rules.dns list this check cannot confirm is a subset of' : 'no rules.dns restriction at least as narrow as'} the configured translation's ciliumNarrowing.dnsNames`,
3074
+ }
3075
+ }
3076
+ continue
3077
+ }
3078
+ if (!destinationIsAllowed(allowance, peer, ports)) {
3079
+ return {
3080
+ beyond: true,
3081
+ detail: `toEndpoints ${describePeer(peer)} on ${describePorts(ports)}, which a '${allowance.policyKind}' translation does not allow`,
3082
+ }
3083
+ }
3084
+ }
3085
+ for (const field of ['toServices', 'toGroups', 'toNodes'] as const) {
3086
+ if (fields.has(field)) {
3087
+ return {
3088
+ beyond: 'unknown',
3089
+ detail: `a ${field} rule, whose destinations this check cannot enumerate`,
3090
+ }
3091
+ }
3092
+ }
3093
+ return { beyond: false }
3094
+ }
3095
+
3096
+ // ---------------------------------------------------------------------------
3097
+ // Normalising the policies that select this pod
3098
+ // ---------------------------------------------------------------------------
3099
+
3100
+ /** The pod the union check is about. */
3101
+ export interface EgressVerificationTarget {
3102
+ readonly namespace: string
3103
+ /**
3104
+ * The pod's REAL labels — for a directly created Sandbox the labels the
3105
+ * create body stamps (known before the POST, so a refusal leaves no
3106
+ * Sandbox and no PVC behind), for a claimed one the bound pod's own
3107
+ * `metadata.labels`. The same value the ingress check is given, for the
3108
+ * same reason: a name proves an object exists, a label is what a selector
3109
+ * actually matches.
3110
+ */
3111
+ readonly podLabels: Readonly<Record<string, string>>
3112
+ readonly engine: KubernetesEgressEngine
3113
+ /** How the refusal names the thing being created, e.g. `Sandbox namzu-ws-demo`. */
3114
+ readonly subject: string
3115
+ }
3116
+
3117
+ /** What one examined policy turned out to be. One line of the refusal. */
3118
+ export type EgressPolicyVerdict =
3119
+ /** Selects the pod and lets out nothing the translation does not. */
3120
+ | 'within'
3121
+ /** Selects the pod and allows egress the translation does not. */
3122
+ | 'widens-egress'
3123
+ /** Its selector does not match the pod's labels. */
3124
+ | 'does-not-select'
3125
+ /** Selects the pod but does not enforce egress, so its egress block is inert. */
3126
+ | 'not-egress-scoped'
3127
+ /** Contains something this check cannot decide. */
3128
+ | 'not-evaluable'
3129
+
3130
+ /** One policy, as the refusal reports it. */
3131
+ export interface ExaminedEgressPolicy {
3132
+ readonly kind: 'NetworkPolicy' | 'CiliumNetworkPolicy'
3133
+ readonly name: string
3134
+ readonly verdict: EgressPolicyVerdict
3135
+ /** Why, for every verdict that is not a plain match or non-match. */
3136
+ readonly detail?: string
3137
+ }
3138
+
3139
+ /**
3140
+ * Which of the three refusals this is — three different operator actions, so
3141
+ * they are carried apart rather than folded into one message:
3142
+ *
3143
+ * - `policy-widens-egress` — narrow or delete the policy that lets more out
3144
+ * than `config.egress` says.
3145
+ * - `no-enforcing-policy` — nothing puts this pod in egress default-deny, so
3146
+ * the translation is not the boundary; apply a policy that selects these
3147
+ * labels.
3148
+ * - `not-evaluable` — this check cannot decide; grant the missing verb, fix
3149
+ * the unreadable policy, or declare `egress.verify: 'named-object-only'`.
3150
+ */
3151
+ export type EgressPolicyRefusal = 'policy-widens-egress' | 'no-enforcing-policy' | 'not-evaluable'
3152
+
3153
+ /**
3154
+ * One policy, reduced to the three questions the union rule asks: does it
3155
+ * select this pod, does it put it in egress default-deny, and does anything
3156
+ * it allows fall outside the configured translation.
3157
+ */
3158
+ export interface EgressPolicyDocument {
3159
+ readonly kind: 'NetworkPolicy' | 'CiliumNetworkPolicy'
3160
+ readonly name: string
3161
+ readonly selects: SelectorMatch
3162
+ /**
3163
+ * Does this policy put the pod into egress DEFAULT-DENY — the only thing
3164
+ * that makes the translation a boundary at all. A core policy whose
3165
+ * `policyTypes` leaves Egress out does not (and the API server ignores its
3166
+ * `egress` block outright), and neither does a Cilium rule carrying
3167
+ * `enableDefaultDeny.egress: false`.
3168
+ */
3169
+ readonly enforcesEgress: boolean
3170
+ readonly rules: readonly EgressRuleVerdict[]
3171
+ /** Set when the OBJECT could not be read. It decides alone. */
3172
+ readonly unreadable?: string
3173
+ /**
3174
+ * This document is the one {@link verifyEgressPolicyApplied} compared to
3175
+ * the translation on this same create path, before this check runs: the
3176
+ * document built from the named object's `spec`, and nothing else.
3177
+ *
3178
+ * Marked so {@link decideEgressUnion} does not judge its RULES, because
3179
+ * the translation's own allowance cannot express one of them.
3180
+ * `egressAllowance` records a `toFQDNs` entry by its `matchName`, while a
3181
+ * `.domain` entry the translation emits is a `matchName` PLUS a
3182
+ * `matchPattern` — so the pattern half read as widening and the check
3183
+ * refused the very object it had just told the operator to apply,
3184
+ * permanently and on every `create()`.
3185
+ *
3186
+ * What the exempting comparison actually reads, because this marking is
3187
+ * only as sound as it is: `spec.podSelector`/`spec.endpointSelector`,
3188
+ * `spec.policyTypes` (core), `spec.egress`, and any `metadata.ownerReferences`
3189
+ * the translation carries. It is NOT a comparison of the whole object, so
3190
+ * the marking is scoped to what it covers:
3191
+ *
3192
+ * - Only the document built from `item.spec` is marked. A document built
3193
+ * from an `item.specs` entry is judged by {@link decideEgressUnion} like
3194
+ * any other object's — a rule in either spelling enforces, so a `specs`
3195
+ * entry that allows more than the translation is refused by name.
3196
+ * - {@link verifyEgressPolicyApplied} REFUSES a live object carrying a
3197
+ * `specs` list at all, which is what makes the first point airtight
3198
+ * rather than merely narrower: the translation never emits one, and half
3199
+ * of what such an object enforces would be rules that comparison never
3200
+ * read.
3201
+ *
3202
+ * What is NOT redundant, and the reason the document stays in the
3203
+ * enumeration at all, is whether it puts this pod in egress default-deny:
3204
+ * dropping it would report a deployment whose only applied policy is the
3205
+ * named one as having no boundary at all, which is a deployment that
3206
+ * verifies today.
3207
+ *
3208
+ * The cost is stated rather than hidden: the named check is memoized on
3209
+ * success for the boundary's lifetime, so a named object that drifts WIDER
3210
+ * after its own check has passed is no longer caught by this one — it is
3211
+ * caught by the named comparison at the next construction. Before this
3212
+ * exemption there was a second net under a five-minute cache; the
3213
+ * alternative is a check that refuses the object it has just told an
3214
+ * operator to apply.
3215
+ */
3216
+ readonly namedObject?: boolean
3217
+ }
3218
+
3219
+ function unreadableEgressPolicy(
3220
+ kind: EgressPolicyDocument['kind'],
3221
+ name: string,
3222
+ detail: string,
3223
+ ): EgressPolicyDocument {
3224
+ return {
3225
+ kind,
3226
+ name,
3227
+ selects: 'unknown',
3228
+ enforcesEgress: false,
3229
+ rules: [],
3230
+ unreadable: detail,
3231
+ }
3232
+ }
3233
+
3234
+ /**
3235
+ * Is this the object `config.egress` named — see
3236
+ * {@link EgressPolicyDocument.namedObject}.
3237
+ *
3238
+ * Read off the allowance rather than passed in, so every caller of
3239
+ * {@link readCoreEgressPolicies}/{@link readCiliumEgressPolicies} that built
3240
+ * its allowance with {@link egressAllowance} gets the marking without having
3241
+ * to remember a second argument, and the readers stay a function of "(items,
3242
+ * this pod, what the translation allows)".
3243
+ *
3244
+ * For a core `NetworkPolicy` the name is the whole answer. For a
3245
+ * `CiliumNetworkPolicy` it is not: the CRD carries EITHER one `spec` or a
3246
+ * `specs` list, `verifyEgressPolicyApplied` reads the former and refuses an
3247
+ * object carrying the latter, so the reader marks only the document built
3248
+ * from `item.spec` and lets a `specs` entry be judged by the union rule like
3249
+ * any other object's rules — see {@link EgressPolicyDocument.namedObject}.
3250
+ *
3251
+ * Deliberately NOT applied to an unreadable document: an object at the named
3252
+ * name that could not be read keeps its `not-evaluable` verdict, which refuses
3253
+ * — a fail-closed answer for a shape whose exact comparison already refused it
3254
+ * earlier on the create path.
3255
+ */
3256
+ function isNamedObject(
3257
+ allowance: EgressAllowance,
3258
+ kind: EgressPolicyDocument['kind'],
3259
+ name: string,
3260
+ ): boolean {
3261
+ return allowance.namedObject.kind === kind && allowance.namedObject.name === name
3262
+ }
3263
+
3264
+ /** Every core `NetworkPolicy` in the list, reduced to {@link EgressPolicyDocument}. */
3265
+ export function readCoreEgressPolicies(
3266
+ items: readonly unknown[],
3267
+ target: EgressVerificationTarget,
3268
+ allowance: EgressAllowance,
3269
+ ): EgressPolicyDocument[] {
3270
+ return items.map((item, index) => {
3271
+ const name = policyName(item, index)
3272
+ if (!isRecord(item)) {
3273
+ return unreadableEgressPolicy(
3274
+ 'NetworkPolicy',
3275
+ name,
3276
+ 'a list entry that is not a policy object',
3277
+ )
3278
+ }
3279
+ const spec = item.spec
3280
+ if (!isRecord(spec)) {
3281
+ return unreadableEgressPolicy('NetworkPolicy', name, 'a spec that is not an object')
3282
+ }
3283
+ const policyTypes = readList(spec.policyTypes)
3284
+ if (policyTypes === 'unreadable') {
3285
+ return unreadableEgressPolicy('NetworkPolicy', name, 'a spec.policyTypes that is not a list')
3286
+ }
3287
+ const rules = readList(spec.egress)
3288
+ if (rules === 'unreadable') {
3289
+ return unreadableEgressPolicy(
3290
+ 'NetworkPolicy',
3291
+ name,
3292
+ 'a spec.egress that is not a list of rules',
3293
+ )
3294
+ }
3295
+ // An ABSENT `policyTypes` is defaulted by the API server from the blocks
3296
+ // the object carries: Egress appears exactly when `spec.egress` does.
3297
+ // This is the opposite of the ingress default, where Egress's absence is
3298
+ // the thing that has to be spelled out.
3299
+ const enforcesEgress =
3300
+ policyTypes === undefined ? rules !== undefined : policyTypes.includes('Egress')
3301
+ return {
3302
+ kind: 'NetworkPolicy',
3303
+ name,
3304
+ ...(isNamedObject(allowance, 'NetworkPolicy', name) ? { namedObject: true } : {}),
3305
+ selects: matchesLabelSelector(spec.podSelector, target.podLabels),
3306
+ enforcesEgress,
3307
+ // An `egress` block under a `policyTypes` that leaves Egress out is
3308
+ // ignored by the API server itself: it neither bounds nor widens.
3309
+ rules: enforcesEgress
3310
+ ? (rules ?? []).map((rule) => coreEgressRuleVerdict(rule, allowance))
3311
+ : [],
3312
+ }
3313
+ })
3314
+ }
3315
+
3316
+ /** Every `CiliumNetworkPolicy` in the list, reduced the same way. */
3317
+ export function readCiliumEgressPolicies(
3318
+ items: readonly unknown[],
3319
+ target: EgressVerificationTarget,
3320
+ allowance: EgressAllowance,
3321
+ ): EgressPolicyDocument[] {
3322
+ const identity = ciliumIdentityLabels(target.podLabels, target.namespace)
3323
+ const documents: EgressPolicyDocument[] = []
3324
+ for (const [index, item] of items.entries()) {
3325
+ const name = policyName(item, index)
3326
+ if (!isRecord(item)) {
3327
+ documents.push(
3328
+ unreadableEgressPolicy(
3329
+ 'CiliumNetworkPolicy',
3330
+ name,
3331
+ 'a list entry that is not a policy object',
3332
+ ),
3333
+ )
3334
+ continue
3335
+ }
3336
+ // The CRD carries EITHER one `spec` or a `specs` list, and a rule in
3337
+ // either enforces. Reading only `spec` would miss a whole policy.
3338
+ //
3339
+ // The two spellings are NOT equal where the named-object exemption is
3340
+ // concerned — see {@link EgressPolicyDocument.namedObject}. The exact
3341
+ // comparison the exemption leans on reads `spec` and refuses an object
3342
+ // carrying a `specs` list, so `spec` is the one document that earns the
3343
+ // marking; a `specs` entry is read as an ordinary policy's rules and is
3344
+ // judged by the union rule, which is what refuses one that widens.
3345
+ const named = isNamedObject(allowance, 'CiliumNetworkPolicy', name)
3346
+ const specDocuments: Array<{ readonly spec: unknown; readonly isTheSpec: boolean }> = []
3347
+ if (item.spec !== undefined && item.spec !== null) {
3348
+ specDocuments.push({ spec: item.spec, isTheSpec: true })
3349
+ }
3350
+ const more = readList(item.specs)
3351
+ if (more === 'unreadable') {
3352
+ documents.push(
3353
+ unreadableEgressPolicy(
3354
+ 'CiliumNetworkPolicy',
3355
+ name,
3356
+ 'a specs that is not a list of rule specs',
3357
+ ),
3358
+ )
3359
+ continue
3360
+ }
3361
+ for (const spec of more ?? []) specDocuments.push({ spec, isTheSpec: false })
3362
+ if (specDocuments.length === 0) {
3363
+ documents.push(
3364
+ unreadableEgressPolicy('CiliumNetworkPolicy', name, 'neither a spec nor a specs list'),
3365
+ )
3366
+ continue
3367
+ }
3368
+ for (const { spec, isTheSpec } of specDocuments) {
3369
+ const document = readCiliumEgressRuleSpec(spec, name, identity, allowance, named && isTheSpec)
3370
+ if (document !== undefined) documents.push(document)
3371
+ }
3372
+ }
3373
+ return documents
3374
+ }
3375
+
3376
+ /**
3377
+ * One `spec`/`specs` entry. `undefined` when it is node-scoped — see below.
3378
+ *
3379
+ * `isTheNamedSpec` is true only for the document built from the named
3380
+ * object's own `spec` — the one {@link verifyEgressPolicyApplied} compared to
3381
+ * the translation. A `specs` entry never is; see
3382
+ * {@link EgressPolicyDocument.namedObject} for why that distinction is
3383
+ * load-bearing rather than bookkeeping.
3384
+ */
3385
+ function readCiliumEgressRuleSpec(
3386
+ spec: unknown,
3387
+ name: string,
3388
+ identity: Readonly<Record<string, string>>,
3389
+ allowance: EgressAllowance,
3390
+ isTheNamedSpec: boolean,
3391
+ ): EgressPolicyDocument | undefined {
3392
+ const unreadable = (detail: string) => unreadableEgressPolicy('CiliumNetworkPolicy', name, detail)
3393
+ if (!isRecord(spec)) return unreadable('a rule spec that is not an object')
3394
+ // A node-scoped rule selects nodes, never pods.
3395
+ if (spec.nodeSelector !== undefined) return undefined
3396
+ const rules = readList(spec.egress)
3397
+ if (rules === 'unreadable') return unreadable('a spec.egress that is not a list of rules')
3398
+ const egressDeny = readList(spec.egressDeny)
3399
+ if (egressDeny === 'unreadable')
3400
+ return unreadable('a spec.egressDeny that is not a list of rules')
3401
+ let enforcesEgress = rules !== undefined || egressDeny !== undefined
3402
+ const enableDefaultDeny = spec.enableDefaultDeny
3403
+ if (enableDefaultDeny !== undefined) {
3404
+ if (!isRecord(enableDefaultDeny))
3405
+ return unreadable('an enableDefaultDeny that is not an object')
3406
+ const forEgress = enableDefaultDeny.egress
3407
+ if (forEgress !== undefined && typeof forEgress !== 'boolean') {
3408
+ return unreadable('an enableDefaultDeny.egress that is not a boolean')
3409
+ }
3410
+ if (forEgress === false) enforcesEgress = false
3411
+ }
3412
+ return {
3413
+ kind: 'CiliumNetworkPolicy',
3414
+ name,
3415
+ ...(isTheNamedSpec ? { namedObject: true } : {}),
3416
+ selects: matchesLabelSelector(spec.endpointSelector, identity, ciliumSelectorKey),
3417
+ enforcesEgress,
3418
+ // `egressDeny` rules are not read: a deny rule can only narrow what
3419
+ // leaves the pod, and this check refuses widening. Allow rules are read
3420
+ // whatever `enableDefaultDeny` says, because what a rule admits it
3421
+ // admits — it simply may not be the policy that default-denies.
3422
+ rules: (rules ?? []).map((rule) => ciliumEgressRuleVerdict(rule, allowance)),
3423
+ }
3424
+ }
3425
+
3426
+ // ---------------------------------------------------------------------------
3427
+ // The decision
3428
+ // ---------------------------------------------------------------------------
3429
+
3430
+ export interface EgressUnionDecision {
3431
+ readonly examined: readonly ExaminedEgressPolicy[]
3432
+ /** Absent when every selecting policy stays inside the translation. */
3433
+ readonly refusal?: {
3434
+ readonly kind: EgressPolicyRefusal
3435
+ readonly summary: string
3436
+ }
3437
+ }
3438
+
3439
+ /**
3440
+ * The union rule, applied. Pure — no I/O, no client, no clock — so every
3441
+ * shape that has to be refused can be asserted one per test.
3442
+ *
3443
+ * Kubernetes UNIONS every policy selecting a pod: traffic leaves if ANY
3444
+ * selecting policy allows it. So one policy allowing more than the
3445
+ * translation is the finding however many narrower ones sit beside it, and a
3446
+ * pod no policy default-denies has no egress boundary at all whatever the
3447
+ * named object says.
3448
+ *
3449
+ * The named object's `spec` is the one document whose rules are not judged
3450
+ * here — see {@link EgressPolicyDocument.namedObject} — and it was compared to
3451
+ * the translation by {@link verifyEgressPolicyApplied} on the same create path
3452
+ * just before. Its SELECTOR and its egress-scoping still decide, which is what
3453
+ * keeps "nothing bounds this pod" answerable for a deployment whose only
3454
+ * policy is that one; and a document built from a `specs` entry is judged here
3455
+ * like any other object's, because the exempting comparison refuses an object
3456
+ * carrying a `specs` list rather than reading one.
3457
+ */
3458
+ export function decideEgressUnion(
3459
+ documents: readonly EgressPolicyDocument[],
3460
+ allowance: EgressAllowance,
3461
+ ): EgressUnionDecision {
3462
+ const examined: ExaminedEgressPolicy[] = []
3463
+ let enforcing = 0
3464
+ let widening: ExaminedEgressPolicy | undefined
3465
+ let undecided: ExaminedEgressPolicy | undefined
3466
+
3467
+ for (const document of documents) {
3468
+ const base = { kind: document.kind, name: document.name } as const
3469
+ if (document.unreadable !== undefined) {
3470
+ const entry: ExaminedEgressPolicy = {
3471
+ ...base,
3472
+ verdict: 'not-evaluable',
3473
+ detail: document.unreadable,
3474
+ }
3475
+ examined.push(entry)
3476
+ undecided ??= entry
3477
+ continue
3478
+ }
3479
+ if (document.selects === 'no') {
3480
+ examined.push({ ...base, verdict: 'does-not-select' })
3481
+ continue
3482
+ }
3483
+ if (document.selects === 'unknown') {
3484
+ const entry: ExaminedEgressPolicy = {
3485
+ ...base,
3486
+ verdict: 'not-evaluable',
3487
+ detail: 'its selector uses something this check cannot evaluate against pod labels',
3488
+ }
3489
+ examined.push(entry)
3490
+ undecided ??= entry
3491
+ continue
3492
+ }
3493
+ // The named object's `spec` rules were just compared to the translation
3494
+ // by `verifyEgressPolicyApplied`, field by field, which is the one
3495
+ // judgement about them that a `toFQDNs`-by-`matchName` allowance cannot
3496
+ // reproduce — see {@link EgressPolicyDocument.namedObject}. So the union
3497
+ // rule does not judge them; judging them here is what made the check
3498
+ // refuse the object it had just told an operator to apply. Everything
3499
+ // below still applies, which is how a pod whose only policy is the named
3500
+ // one is still known to be in egress default-deny. Only that `spec`
3501
+ // document carries the marking: a `specs` entry is judged here like any
3502
+ // other rules.
3503
+ if (document.namedObject !== true) {
3504
+ const beyondRule = document.rules.find((rule) => rule.beyond === true)
3505
+ if (beyondRule !== undefined) {
3506
+ const entry: ExaminedEgressPolicy = {
3507
+ ...base,
3508
+ verdict: 'widens-egress',
3509
+ ...(beyondRule.detail !== undefined ? { detail: beyondRule.detail } : {}),
3510
+ }
3511
+ examined.push(entry)
3512
+ widening ??= entry
3513
+ continue
3514
+ }
3515
+ const unknownRule = document.rules.find((rule) => rule.beyond === 'unknown')
3516
+ if (unknownRule !== undefined) {
3517
+ const entry: ExaminedEgressPolicy = {
3518
+ ...base,
3519
+ verdict: 'not-evaluable',
3520
+ ...(unknownRule.detail !== undefined ? { detail: unknownRule.detail } : {}),
3521
+ }
3522
+ examined.push(entry)
3523
+ undecided ??= entry
3524
+ continue
3525
+ }
3526
+ }
3527
+ if (!document.enforcesEgress) {
3528
+ examined.push({
3529
+ ...base,
3530
+ verdict: 'not-egress-scoped',
3531
+ detail: 'it selects the pod but default-denies nothing on egress',
3532
+ })
3533
+ continue
3534
+ }
3535
+ examined.push({ ...base, verdict: 'within' })
3536
+ enforcing += 1
3537
+ }
3538
+
3539
+ if (widening !== undefined) {
3540
+ return {
3541
+ examined,
3542
+ refusal: {
3543
+ kind: 'policy-widens-egress',
3544
+ summary: `${widening.kind}/${widening.name} selects this pod and allows ${widening.detail ?? 'egress the configured translation does not'}.`,
3545
+ },
3546
+ }
3547
+ }
3548
+ if (undecided !== undefined) {
3549
+ return {
3550
+ examined,
3551
+ refusal: {
3552
+ kind: 'not-evaluable',
3553
+ summary: `${undecided.kind}/${undecided.name} contains ${undecided.detail ?? 'something this check cannot evaluate'}, so what this pod may reach cannot be decided from the cluster's own objects.`,
3554
+ },
3555
+ }
3556
+ }
3557
+ // `allow-all` asks for no boundary, so a pod nothing default-denies is
3558
+ // exactly what it configured; every other kind needs a policy that
3559
+ // actually puts this pod in egress default-deny, or the translation is a
3560
+ // manifest nobody enforces.
3561
+ if (enforcing === 0 && !allowance.permitsEverything) {
3562
+ return {
3563
+ examined,
3564
+ refusal: {
3565
+ kind: 'no-enforcing-policy',
3566
+ summary: `no applied policy puts this pod in egress default-deny, so a '${allowance.policyKind}' translation bounds nothing it sends.`,
3567
+ },
3568
+ }
3569
+ }
3570
+ return { examined }
3571
+ }
3572
+
3573
+ /**
3574
+ * The named refusal. Distinct from {@link KubernetesEgressPolicyMismatchError}
3575
+ * — which is about the ONE named object drifting from the translation — and
3576
+ * from `KubernetesIngressPolicyError`, which is about the agent port being
3577
+ * reachable. An operator debugging a release that ships all three tells them
3578
+ * apart by class and by the first clause of the message.
3579
+ *
3580
+ * It carries the pod's labels and EVERY policy examined, with a verdict each,
3581
+ * because that list is the operator's whole debugging session: "why does my
3582
+ * policy not count?" is answered by the line saying it did not select these
3583
+ * labels.
3584
+ */
3585
+ export class KubernetesEgressPolicyUnionError extends Error {
3586
+ override readonly name = 'KubernetesEgressPolicyUnionError'
3587
+
3588
+ constructor(
3589
+ readonly refusal: EgressPolicyRefusal,
3590
+ readonly subject: string,
3591
+ readonly podLabels: Readonly<Record<string, string>>,
3592
+ readonly policyKind: KubernetesEgressPolicy['kind'],
3593
+ readonly examined: readonly ExaminedEgressPolicy[],
3594
+ summary: string,
3595
+ /** Empty on every decision made from policies that WERE read. */
3596
+ readonly unread: readonly UnreadPolicySource[] = [],
3597
+ ) {
3598
+ super(
3599
+ `kubernetes: refusing ${subject} — ${summary} config.egress.policy is '${policyKind}' and the pod's labels are ${formatLabels(podLabels)}. ${formatExaminedEgress(examined, unread)} Kubernetes UNIONS every policy selecting a pod, so what leaves this pod is everything ANY of them allows — a second policy widens egress however exactly the named object matches. ${formatEgressRemedy(refusal, unread)}`,
3600
+ )
3601
+ }
3602
+ }
3603
+
3604
+ function formatExaminedEgress(
3605
+ examined: readonly ExaminedEgressPolicy[],
3606
+ unread: readonly UnreadPolicySource[],
3607
+ ): string {
3608
+ const lines = examined.map((entry) => {
3609
+ const detail = entry.detail !== undefined ? ` (${entry.detail})` : ''
3610
+ return `${entry.kind}/${entry.name}: ${entry.verdict}${detail}`
3611
+ })
3612
+ const read =
3613
+ lines.length === 0
3614
+ ? 'No policy was read from the collections this check could enumerate.'
3615
+ : `Policies examined: ${lines.join('; ')}.`
3616
+ if (unread.length === 0) return read
3617
+ const missing = unread
3618
+ .map((source) => `${source.resource} at ${source.path} (${source.why}: ${source.reason})`)
3619
+ .join('; ')
3620
+ return `${read} NOT read, so nothing below is a statement about it: ${missing}.`
3621
+ }
3622
+
3623
+ function formatEgressRemedy(
3624
+ refusal: EgressPolicyRefusal,
3625
+ unread: readonly UnreadPolicySource[],
3626
+ ): string {
3627
+ if (refusal === 'not-evaluable') {
3628
+ const forbidden = unread.some((source) => source.why === 'forbidden')
3629
+ return `${forbidden ? "Grant this backend's ServiceAccount 'list' on that resource, " : 'Fix or remove the policy named above, '}or set egress.verify: 'named-object-only' to check only the one named object, as every release before this one did.`
3630
+ }
3631
+ if (refusal === 'no-enforcing-policy') {
3632
+ return "Apply a NetworkPolicy whose podSelector matches the labels above and whose policyTypes includes 'Egress' (packages/sandbox/k8s/manifests/networkpolicy.yaml is this repo's baseline), or set egress.verify: 'named-object-only' if the boundary lives somewhere a namespaced Role cannot read."
3633
+ }
3634
+ return "Narrow or delete the policy named above so that nothing selecting these pods allows more than config.egress does, or set egress.verify: 'named-object-only' to go back to checking only the named object."
3635
+ }
3636
+
3637
+ // ---------------------------------------------------------------------------
3638
+ // The I/O half
3639
+ // ---------------------------------------------------------------------------
3640
+
3641
+ /**
3642
+ * Verify-not-trust, widened from one object to the union: list the
3643
+ * namespace's policies, evaluate every one that selects this pod against the
3644
+ * configured translation, and refuse unless nothing lets out more than
3645
+ * `config.egress` says. The named object's `spec` document is the one
3646
+ * exception, and it was compared to the translation by
3647
+ * {@link verifyEgressPolicyApplied} on this same create path just before this
3648
+ * one — see {@link EgressPolicyDocument.namedObject}: a Cilium object that
3649
+ * carries a `specs` list is refused there outright, so a `specs` entry is
3650
+ * judged HERE, like any other object's rules.
3651
+ *
3652
+ * Runs beside the ingress check on every create path — before the POST for a
3653
+ * directly created Sandbox, where the labels are known and a refusal leaves
3654
+ * nothing behind, and after the bind for a claimed one, where the pool's own
3655
+ * template decides the labels and a refusal releases the claim through the
3656
+ * acquire path's cleanup.
3657
+ */
3658
+ export async function verifyEgressPolicyUnion(
3659
+ client: KubernetesClient,
3660
+ translated: KubernetesTranslatedEgressPolicy,
3661
+ target: EgressVerificationTarget,
3662
+ signal?: AbortSignal,
3663
+ ): Promise<void> {
3664
+ const allowance = egressAllowance(translated)
3665
+ const documents: EgressPolicyDocument[] = []
3666
+ try {
3667
+ documents.push(
3668
+ ...readCoreEgressPolicies(
3669
+ await listPolicies(
3670
+ client,
3671
+ networkPolicyCollectionPath(target.namespace),
3672
+ 'networkpolicies',
3673
+ signal,
3674
+ ),
3675
+ target,
3676
+ allowance,
3677
+ ),
3678
+ )
3679
+ if (target.engine === 'cilium') {
3680
+ documents.push(
3681
+ ...readCiliumEgressPolicies(
3682
+ await listPolicies(
3683
+ client,
3684
+ ciliumNetworkPolicyCollectionPath(target.namespace),
3685
+ 'ciliumnetworkpolicies',
3686
+ signal,
3687
+ ),
3688
+ target,
3689
+ allowance,
3690
+ ),
3691
+ )
3692
+ }
3693
+ } catch (err) {
3694
+ if (!(err instanceof UnreadPolicyCollection)) throw err
3695
+ throw new KubernetesEgressPolicyUnionError(
3696
+ 'not-evaluable',
3697
+ target.subject,
3698
+ target.podLabels,
3699
+ allowance.policyKind,
3700
+ decideEgressUnion(documents, allowance).examined,
3701
+ err.summary,
3702
+ [err.source],
3703
+ )
3704
+ }
3705
+
3706
+ const decision = decideEgressUnion(documents, allowance)
3707
+ if (decision.refusal === undefined) return
3708
+ throw new KubernetesEgressPolicyUnionError(
3709
+ decision.refusal.kind,
3710
+ target.subject,
3711
+ target.podLabels,
3712
+ allowance.policyKind,
3713
+ decision.examined,
3714
+ decision.refusal.summary,
3715
+ )
437
3716
  }