@namzu/sandbox 14.0.0 → 15.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +838 -0
- package/README.md +310 -14
- package/dist/backends/aci-standby-pool/index.d.ts.map +1 -1
- package/dist/backends/aci-standby-pool/index.js +13 -1
- package/dist/backends/aci-standby-pool/index.js.map +1 -1
- package/dist/backends/docker/index.d.ts.map +1 -1
- package/dist/backends/docker/index.js +19 -1
- package/dist/backends/docker/index.js.map +1 -1
- package/dist/backends/firecracker/index.d.ts.map +1 -1
- package/dist/backends/firecracker/index.js +12 -2
- package/dist/backends/firecracker/index.js.map +1 -1
- package/dist/backends/firecracker/protocol.d.ts +459 -8
- package/dist/backends/firecracker/protocol.d.ts.map +1 -1
- package/dist/backends/firecracker/protocol.js +136 -0
- package/dist/backends/firecracker/protocol.js.map +1 -1
- package/dist/backends/firecracker/transport.d.ts +539 -6
- package/dist/backends/firecracker/transport.d.ts.map +1 -1
- package/dist/backends/firecracker/transport.js +1171 -24
- package/dist/backends/firecracker/transport.js.map +1 -1
- package/dist/backends/kubernetes/egress-policy.d.ts +1088 -11
- package/dist/backends/kubernetes/egress-policy.d.ts.map +1 -1
- package/dist/backends/kubernetes/egress-policy.js +2173 -29
- package/dist/backends/kubernetes/egress-policy.js.map +1 -1
- package/dist/backends/kubernetes/identity.d.ts +193 -0
- package/dist/backends/kubernetes/identity.d.ts.map +1 -0
- package/dist/backends/kubernetes/identity.js +147 -0
- package/dist/backends/kubernetes/identity.js.map +1 -0
- package/dist/backends/kubernetes/index.d.ts +678 -33
- package/dist/backends/kubernetes/index.d.ts.map +1 -1
- package/dist/backends/kubernetes/index.js +1180 -95
- package/dist/backends/kubernetes/index.js.map +1 -1
- package/dist/backends/kubernetes/ingress-policy.d.ts +375 -0
- package/dist/backends/kubernetes/ingress-policy.d.ts.map +1 -0
- package/dist/backends/kubernetes/ingress-policy.js +1050 -0
- package/dist/backends/kubernetes/ingress-policy.js.map +1 -0
- package/dist/backends/kubernetes/k8s-client.d.ts +213 -4
- package/dist/backends/kubernetes/k8s-client.d.ts.map +1 -1
- package/dist/backends/kubernetes/k8s-client.js +359 -52
- package/dist/backends/kubernetes/k8s-client.js.map +1 -1
- package/dist/backends/kubernetes/lease.d.ts +40 -14
- package/dist/backends/kubernetes/lease.d.ts.map +1 -1
- package/dist/backends/kubernetes/lease.js +68 -18
- package/dist/backends/kubernetes/lease.js.map +1 -1
- package/dist/backends/kubernetes/objects.d.ts +423 -3
- package/dist/backends/kubernetes/objects.d.ts.map +1 -1
- package/dist/backends/kubernetes/objects.js +364 -2
- package/dist/backends/kubernetes/objects.js.map +1 -1
- package/dist/backends/kubernetes/per-sandbox-policy.d.ts +219 -0
- package/dist/backends/kubernetes/per-sandbox-policy.d.ts.map +1 -0
- package/dist/backends/kubernetes/per-sandbox-policy.js +407 -0
- package/dist/backends/kubernetes/per-sandbox-policy.js.map +1 -0
- package/dist/backends/kubernetes/rbac.d.ts +153 -0
- package/dist/backends/kubernetes/rbac.d.ts.map +1 -0
- package/dist/backends/kubernetes/rbac.js +177 -0
- package/dist/backends/kubernetes/rbac.js.map +1 -0
- package/dist/backends/kubernetes/sandbox.d.ts +81 -14
- package/dist/backends/kubernetes/sandbox.d.ts.map +1 -1
- package/dist/backends/kubernetes/sandbox.js +149 -15
- package/dist/backends/kubernetes/sandbox.js.map +1 -1
- package/dist/backends/kubernetes/transport.d.ts +935 -9
- package/dist/backends/kubernetes/transport.d.ts.map +1 -1
- package/dist/backends/kubernetes/transport.js +1958 -62
- package/dist/backends/kubernetes/transport.js.map +1 -1
- package/dist/backends/kubernetes/workspace.d.ts +1149 -18
- package/dist/backends/kubernetes/workspace.d.ts.map +1 -1
- package/dist/backends/kubernetes/workspace.js +2825 -186
- package/dist/backends/kubernetes/workspace.js.map +1 -1
- package/dist/backends/remote-execution-controller.d.ts +14 -0
- package/dist/backends/remote-execution-controller.d.ts.map +1 -1
- package/dist/backends/remote-execution-controller.js.map +1 -1
- package/dist/index.d.ts +231 -13
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +247 -5
- package/dist/index.js.map +1 -1
- package/dist/testing/sandbox-conformance.d.ts +39 -5
- package/dist/testing/sandbox-conformance.d.ts.map +1 -1
- package/dist/testing/sandbox-conformance.js +436 -5
- package/dist/testing/sandbox-conformance.js.map +1 -1
- package/package.json +3 -3
- package/src/backends/aci-standby-pool/index.ts +16 -1
- package/src/backends/docker/index.ts +22 -1
- package/src/backends/firecracker/index.ts +14 -2
- package/src/backends/firecracker/protocol.ts +514 -6
- package/src/backends/firecracker/transport.ts +1492 -40
- package/src/backends/kubernetes/egress-policy.ts +3064 -53
- package/src/backends/kubernetes/identity.ts +261 -0
- package/src/backends/kubernetes/index.ts +1785 -127
- package/src/backends/kubernetes/ingress-policy.ts +1344 -0
- package/src/backends/kubernetes/k8s-client.ts +444 -54
- package/src/backends/kubernetes/lease.ts +75 -19
- package/src/backends/kubernetes/objects.ts +626 -6
- package/src/backends/kubernetes/per-sandbox-policy.ts +542 -0
- package/src/backends/kubernetes/rbac.ts +192 -0
- package/src/backends/kubernetes/sandbox.ts +218 -20
- package/src/backends/kubernetes/transport.ts +2733 -124
- package/src/backends/kubernetes/workspace.ts +4476 -222
- package/src/backends/remote-execution-controller.ts +14 -0
- package/src/index.ts +595 -14
- package/src/testing/sandbox-conformance.ts +540 -5
|
@@ -78,13 +78,36 @@
|
|
|
78
78
|
|
|
79
79
|
import { isDeepStrictEqual } from 'node:util'
|
|
80
80
|
import type { EgressPolicy } from '../../index.js'
|
|
81
|
+
// The policy-enumeration and shape-reading primitives, shared with the
|
|
82
|
+
// ingress direction. There is one enumeration of the policies selecting a pod
|
|
83
|
+
// in this package and one reading of what a peer is; see that module's
|
|
84
|
+
// "Shared with the egress direction".
|
|
85
|
+
import {
|
|
86
|
+
type SelectorMatch,
|
|
87
|
+
UnreadPolicyCollection,
|
|
88
|
+
type UnreadPolicySource,
|
|
89
|
+
ciliumIdentityLabels,
|
|
90
|
+
ciliumSelectorKey,
|
|
91
|
+
corePeerIsWideOpen,
|
|
92
|
+
formatLabels,
|
|
93
|
+
isRecord,
|
|
94
|
+
listPolicies,
|
|
95
|
+
matchesLabelSelector,
|
|
96
|
+
policyName,
|
|
97
|
+
readList,
|
|
98
|
+
selectorIsReadable,
|
|
99
|
+
} from './ingress-policy.js'
|
|
81
100
|
import { KubernetesAlreadyGoneError, type KubernetesClient } from './k8s-client.js'
|
|
101
|
+
import type { KubernetesOwnerReference } from './objects.js'
|
|
82
102
|
import {
|
|
83
103
|
CILIUM_NETWORK_POLICY_API_GROUP,
|
|
84
104
|
CILIUM_NETWORK_POLICY_API_VERSION,
|
|
85
105
|
CORE_NETWORK_POLICY_API_GROUP,
|
|
86
106
|
CORE_NETWORK_POLICY_API_VERSION,
|
|
107
|
+
SANDBOX_TEMPLATE_LABEL_KEY,
|
|
108
|
+
ciliumNetworkPolicyCollectionPath,
|
|
87
109
|
ciliumNetworkPolicyPath,
|
|
110
|
+
networkPolicyCollectionPath,
|
|
88
111
|
networkPolicyPath,
|
|
89
112
|
sandboxTemplateLabel,
|
|
90
113
|
} from './objects.js'
|
|
@@ -97,6 +120,156 @@ import {
|
|
|
97
120
|
*/
|
|
98
121
|
export type KubernetesEgressEngine = 'core' | 'cilium'
|
|
99
122
|
|
|
123
|
+
/**
|
|
124
|
+
* The two egress kinds that exist only here, because only a `NetworkPolicy`
|
|
125
|
+
* can express them and the shared {@link EgressPolicy} union describes what
|
|
126
|
+
* every tier can carry.
|
|
127
|
+
*
|
|
128
|
+
* - `'no-network'` is the one the shared union has no word for: NOTHING
|
|
129
|
+
* leaves the pod, the cluster's own resolver included. `'deny-all'` is not
|
|
130
|
+
* that and never was — it emits {@link CLUSTER_DNS_EGRESS_RULE}, and a
|
|
131
|
+
* cluster resolver forwards outside names upstream, so a `'deny-all'`
|
|
132
|
+
* sandbox keeps a channel out through DNS. `'deny-all'` is deliberately
|
|
133
|
+
* NOT tightened into this: its emitted manifest is byte-identical to
|
|
134
|
+
* every release before this one, because verification of the named object
|
|
135
|
+
* is an exact match and changing the translation would fail every
|
|
136
|
+
* `create()` on every deployment that already applied a policy until an
|
|
137
|
+
* operator re-applied it. A workload under `'no-network'` resolves
|
|
138
|
+
* nothing at all — that is the point, and the agent needs no resolver
|
|
139
|
+
* because the host dials in.
|
|
140
|
+
* - `'public-internet'` is `'allow-all'` minus everything that is not the
|
|
141
|
+
* public internet: the private ranges, the carrier-grade NAT range, the
|
|
142
|
+
* link-local range that carries cloud instance metadata, loopback, the
|
|
143
|
+
* platform endpoint some clouds answer on, and IPv6's equivalents. It
|
|
144
|
+
* exists because `'allow-all'` reaches the node, the API server, the
|
|
145
|
+
* service network and every other sandbox pod, and nothing between the
|
|
146
|
+
* two said "out, but not sideways".
|
|
147
|
+
*
|
|
148
|
+
* `exceptCidrs` adds to the excluded list; it never removes from it. A CIDR
|
|
149
|
+
* that is not one this check can parse is refused at construction rather
|
|
150
|
+
* than emitted into a manifest the API server would reject on apply.
|
|
151
|
+
*/
|
|
152
|
+
export type KubernetesOnlyEgressPolicy =
|
|
153
|
+
| { readonly kind: 'no-network' }
|
|
154
|
+
| {
|
|
155
|
+
readonly kind: 'public-internet'
|
|
156
|
+
readonly exceptCidrs?: readonly string[]
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
/**
|
|
160
|
+
* What {@link KubernetesEgressConfig.policy} accepts: the shared
|
|
161
|
+
* {@link EgressPolicy} union plus the two {@link KubernetesOnlyEgressPolicy}
|
|
162
|
+
* kinds. The shared union itself is untouched — a kind no other tier can
|
|
163
|
+
* enforce does not belong in the type every tier reads.
|
|
164
|
+
*/
|
|
165
|
+
export type KubernetesEgressPolicy = EgressPolicy | KubernetesOnlyEgressPolicy
|
|
166
|
+
|
|
167
|
+
/**
|
|
168
|
+
* How thoroughly the applied boundary is checked before a sandbox is handed
|
|
169
|
+
* back.
|
|
170
|
+
*
|
|
171
|
+
* - `'union'` (the default) reads the NAMED object exactly as before AND
|
|
172
|
+
* enumerates every `NetworkPolicy` — and, under `engine: 'cilium'`, every
|
|
173
|
+
* `CiliumNetworkPolicy` — in the namespace, refusing when any policy that
|
|
174
|
+
* selects the pod allows egress the configured translation does not. That
|
|
175
|
+
* is not belt-and-braces: the API server UNIONS every policy selecting a
|
|
176
|
+
* pod, so a second policy widens egress however exactly the named one
|
|
177
|
+
* matches, and a `SandboxTemplate`'s own `networkPolicy` block becomes
|
|
178
|
+
* exactly such a policy.
|
|
179
|
+
* - `'named-object-only'` is the documented opt-out, and restores the
|
|
180
|
+
* previous behaviour exactly: one GET of the named object, memoized for
|
|
181
|
+
* the backend's lifetime, and no enumeration. For a deployment whose
|
|
182
|
+
* other policies a namespaced Role cannot read, or which accepts the
|
|
183
|
+
* union it has. It mirrors `ingress: 'unverified'` — a claim a deployment
|
|
184
|
+
* makes on purpose rather than a default it inherits.
|
|
185
|
+
*/
|
|
186
|
+
export type KubernetesEgressVerification = 'union' | 'named-object-only'
|
|
187
|
+
|
|
188
|
+
/**
|
|
189
|
+
* How {@link KubernetesCiliumEgressNarrowing.dnsNames} narrows the kube-dns
|
|
190
|
+
* L7 rule. `true` uses every default below; an object customises them.
|
|
191
|
+
*
|
|
192
|
+
* The exact-name list a narrowed policy emits is, for each allowed host,
|
|
193
|
+
* the bare name plus the host under `<namespace>.svc.<clusterDomain>`,
|
|
194
|
+
* `svc.<clusterDomain>` and `<clusterDomain>` — because Cilium's `matchName`
|
|
195
|
+
* is an EXACT name and does not match across a `.` the way `matchPattern`
|
|
196
|
+
* does, and a search-list query for `github.com.svc.cluster.local` is a
|
|
197
|
+
* DIFFERENT name than `github.com`. `searchSuffixes` adds more: kubelet
|
|
198
|
+
* appends the node's own search domains, which this backend cannot see, so a
|
|
199
|
+
* deployment whose nodes carry extra search domains lists them here or a
|
|
200
|
+
* search-list query for one of them is refused by the DNS proxy — and, per
|
|
201
|
+
* Cilium's own docs, some images (musl/Alpine) stop trying the search list
|
|
202
|
+
* entirely the first time that happens, breaking the bare name lookup too.
|
|
203
|
+
*/
|
|
204
|
+
export interface KubernetesCiliumDnsNarrowing {
|
|
205
|
+
/** Defaults to the egress target's own namespace (where the sandbox pods run). */
|
|
206
|
+
readonly namespace?: string
|
|
207
|
+
/** Defaults to `'cluster.local'`. */
|
|
208
|
+
readonly clusterDomain?: string
|
|
209
|
+
/** Appended after the three built-in suffixes, not replacing them. */
|
|
210
|
+
readonly searchSuffixes?: readonly string[]
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
/**
|
|
214
|
+
* Opt-in narrowing of a `static`/`resolver` hostname allowlist under
|
|
215
|
+
* `engine: 'cilium'` — see the module doc and `#490`. Every field here is
|
|
216
|
+
* OFF unless set, and setting none of them leaves the translation
|
|
217
|
+
* byte-for-byte what it always emitted: that is the compatibility guarantee
|
|
218
|
+
* a deployment with an already-applied policy relies on.
|
|
219
|
+
*
|
|
220
|
+
* Setting any field switches the translation from one shared `toFQDNs` rule
|
|
221
|
+
* naming every host to one `toFQDNs` rule PER HOST, so ports and server
|
|
222
|
+
* names can differ host by host. Refused synchronously (alongside
|
|
223
|
+
* {@link assertEgressPolicyIsEnforceable}'s existing refusals) unless
|
|
224
|
+
* `engine` is `'cilium'` and `policy.kind` is `'static'` or `'resolver'`.
|
|
225
|
+
*/
|
|
226
|
+
export interface KubernetesCiliumEgressNarrowing {
|
|
227
|
+
/**
|
|
228
|
+
* TCP ports allowed to every host that has no entry in `hostPorts`, e.g.
|
|
229
|
+
* `[443]`. Leaving both this and `hostPorts` unset — with `dnsNames` and
|
|
230
|
+
* `tlsServerNames` also unset — means no port narrowing: a host's
|
|
231
|
+
* `toFQDNs` rule carries no `toPorts` at all, exactly as the
|
|
232
|
+
* unnarrowed translation emits today (every port reachable).
|
|
233
|
+
*/
|
|
234
|
+
readonly ports?: readonly number[]
|
|
235
|
+
/**
|
|
236
|
+
* Per-host TCP port overrides, keyed by the host exactly as it appears in
|
|
237
|
+
* `allowedHosts` or a `resolver`'s result. A host with no entry here
|
|
238
|
+
* falls back to `ports`.
|
|
239
|
+
*/
|
|
240
|
+
readonly hostPorts?: Readonly<Record<string, readonly number[]>>
|
|
241
|
+
/**
|
|
242
|
+
* Narrow the kube-dns L7 rule from `rules.dns: [{ matchPattern: '*' }]`
|
|
243
|
+
* to an exact `matchName` per allowed host, and per host-plus-suffix. See
|
|
244
|
+
* {@link KubernetesCiliumDnsNarrowing}.
|
|
245
|
+
*/
|
|
246
|
+
readonly dnsNames?: boolean | KubernetesCiliumDnsNarrowing
|
|
247
|
+
/**
|
|
248
|
+
* Add `serverNames: [<host>]` to each host's TLS ports (SNI enforcement,
|
|
249
|
+
* which needs Cilium's L7 proxy — see the module doc). A host with no
|
|
250
|
+
* port configured in `ports`/`hostPorts` is limited to `tlsPorts` rather
|
|
251
|
+
* than left with no port restriction at all, because a `serverNames`
|
|
252
|
+
* rule needs a port to attach to.
|
|
253
|
+
*
|
|
254
|
+
* REFUSED for a `.domain` allowlist entry, with
|
|
255
|
+
* {@link KubernetesNetworkPolicyHostError} and nothing written: a server
|
|
256
|
+
* name is one exact SNI value a handshake presents, while the entry means
|
|
257
|
+
* a name PLUS its subdomains, and no single value means both — see
|
|
258
|
+
* {@link assertHostsFitNarrowing}. The refusal is the entry's, not this
|
|
259
|
+
* option's, so it fires wherever `tlsServerNames` can be set:
|
|
260
|
+
* `config.egress.ciliumNarrowing` on the config-level allowlist and
|
|
261
|
+
* `config.egress.perSandbox.narrowing` on the per-sandbox one.
|
|
262
|
+
*/
|
|
263
|
+
readonly tlsServerNames?: boolean
|
|
264
|
+
/**
|
|
265
|
+
* Which of a host's configured ports are TLS ports, for
|
|
266
|
+
* `tlsServerNames`: those get `serverNames` on their own `toPorts`
|
|
267
|
+
* entry, the rest (if any) get a separate entry with none. Defaults to
|
|
268
|
+
* `[443]`. Meaningless unless `tlsServerNames` is set.
|
|
269
|
+
*/
|
|
270
|
+
readonly tlsPorts?: readonly number[]
|
|
271
|
+
}
|
|
272
|
+
|
|
100
273
|
/**
|
|
101
274
|
* The config-level egress hook on {@link KubernetesBackendConfig}.
|
|
102
275
|
*
|
|
@@ -109,7 +282,7 @@ export type KubernetesEgressEngine = 'core' | 'cilium'
|
|
|
109
282
|
* the one egress policy the whole backend enforces.
|
|
110
283
|
*/
|
|
111
284
|
export interface KubernetesEgressConfig {
|
|
112
|
-
readonly policy:
|
|
285
|
+
readonly policy: KubernetesEgressPolicy
|
|
113
286
|
/**
|
|
114
287
|
* Name of the `NetworkPolicy` (or `CiliumNetworkPolicy`, under the
|
|
115
288
|
* `'cilium'` engine) an operator applied. Defaults to
|
|
@@ -119,6 +292,663 @@ export interface KubernetesEgressConfig {
|
|
|
119
292
|
readonly networkPolicyName?: string
|
|
120
293
|
/** Default `'core'`. See the type doc. */
|
|
121
294
|
readonly engine?: KubernetesEgressEngine
|
|
295
|
+
/** Default `'union'`. See {@link KubernetesEgressVerification}. */
|
|
296
|
+
readonly verify?: KubernetesEgressVerification
|
|
297
|
+
/**
|
|
298
|
+
* Opt-in port/DNS-name/TLS-server-name narrowing for a `static`/
|
|
299
|
+
* `resolver` allowlist under `engine: 'cilium'`. Unset (the default)
|
|
300
|
+
* emits exactly what every release before this one did. See
|
|
301
|
+
* {@link KubernetesCiliumEgressNarrowing}.
|
|
302
|
+
*/
|
|
303
|
+
readonly ciliumNarrowing?: KubernetesCiliumEgressNarrowing
|
|
304
|
+
/**
|
|
305
|
+
* Which egress PROFILE the sandboxes this backend produces run under —
|
|
306
|
+
* a DNS-1123 label value such as `none` or `internet`.
|
|
307
|
+
*
|
|
308
|
+
* Unset (the default) is the single-profile world this backend has
|
|
309
|
+
* always had: one policy per backend, selected by the template label
|
|
310
|
+
* alone, and every emitted body, selector and policy name byte-identical
|
|
311
|
+
* to the release before profiles existed.
|
|
312
|
+
*
|
|
313
|
+
* SET, it becomes a pod LABEL — {@link profileLabelKey} is its key —
|
|
314
|
+
* which travels three places at once: onto the `SandboxClaim`'s
|
|
315
|
+
* `additionalPodMetadata.labels`, onto a directly created Sandbox's pod
|
|
316
|
+
* template, and into the translated policy's own selector. That is what
|
|
317
|
+
* lets ONE warm pool serve several network modes: a claim carrying a
|
|
318
|
+
* profile label adopts a warm replica and the controller patches the
|
|
319
|
+
* label onto the running pod, with no cold start and no second pool —
|
|
320
|
+
* measured on agent-sandbox v1.0.2 (two profiles out of one two-replica
|
|
321
|
+
* pool, every adopt under 70 ms, each bound pod a replica that already
|
|
322
|
+
* existed). A label is the only claim-time metadata that is warm-safe:
|
|
323
|
+
* `env` and `volumeClaimTemplates` force a cold start, which is why
|
|
324
|
+
* neither appears on the claim body.
|
|
325
|
+
*
|
|
326
|
+
* OPERATOR PREREQUISITE, and it is not optional: the controller refuses
|
|
327
|
+
* a claim whose label key sits outside its `allowed-label-domains`
|
|
328
|
+
* allowlist (the `agent-sandbox-config` ConfigMap in the controller's
|
|
329
|
+
* namespace; default `sandbox.users.io`), so the DEFAULT key below is
|
|
330
|
+
* refused by a stock controller until an operator adds
|
|
331
|
+
* `sandbox.namzu.ai` to that key — or sets {@link profileLabelKey} to
|
|
332
|
+
* something already allowed. That refusal is fast and carries the
|
|
333
|
+
* controller's own reason and message: it is the acquire's ordinary
|
|
334
|
+
* `claim-rejected` failure, with a
|
|
335
|
+
* {@link KubernetesPodLabelsRejectedError} as its cause naming the labels
|
|
336
|
+
* that were sent and the key that moves them.
|
|
337
|
+
*/
|
|
338
|
+
readonly profile?: string
|
|
339
|
+
/**
|
|
340
|
+
* Label key {@link profile} is written under. Defaults to
|
|
341
|
+
* {@link DEFAULT_EGRESS_PROFILE_LABEL_KEY}.
|
|
342
|
+
*
|
|
343
|
+
* Worth setting to `sandbox.users.io/egress-profile` on a cluster whose
|
|
344
|
+
* controller still carries the stock `allowed-label-domains` — that
|
|
345
|
+
* domain is upstream's own default and needs no ConfigMap edit at all.
|
|
346
|
+
*/
|
|
347
|
+
readonly profileLabelKey?: string
|
|
348
|
+
/**
|
|
349
|
+
* Opt in to PER-SANDBOX egress: one `CiliumNetworkPolicy` per live
|
|
350
|
+
* sandbox, written by this host when a caller calls
|
|
351
|
+
* `Sandbox.setNetworkPolicy`, owned by the object the acquire created so
|
|
352
|
+
* the cluster garbage-collects it.
|
|
353
|
+
*
|
|
354
|
+
* Unset (the default) is every release before this one: `setNetworkPolicy`
|
|
355
|
+
* is ABSENT from the handle, exactly as the SDK's omit-or-throw contract
|
|
356
|
+
* asks of a backend that cannot honour an optional method, and this
|
|
357
|
+
* backend issues no policy write of any kind. Presence depends only on
|
|
358
|
+
* this field — never on a runtime probe — so a host can decide what it
|
|
359
|
+
* has from its own config rather than from a call that might fail.
|
|
360
|
+
*
|
|
361
|
+
* TWO OPERATOR PREREQUISITES, and neither is a nicety: the admission
|
|
362
|
+
* policy named below (with its binding) has to exist, and the host's
|
|
363
|
+
* ServiceAccount needs the write verbs the default Role deliberately does
|
|
364
|
+
* not grant. See {@link KubernetesPerSandboxEgressConfig}.
|
|
365
|
+
*/
|
|
366
|
+
readonly perSandbox?: KubernetesPerSandboxEgressConfig
|
|
367
|
+
}
|
|
368
|
+
|
|
369
|
+
/**
|
|
370
|
+
* Per-sandbox egress: what it takes to let a host narrow ONE live sandbox's
|
|
371
|
+
* egress without touching any other sandbox's.
|
|
372
|
+
*
|
|
373
|
+
* The mechanism is a `CiliumNetworkPolicy` per sandbox, named
|
|
374
|
+
* `namzu-sbx-<uid of the object this backend created>`, selecting that one
|
|
375
|
+
* sandbox's pod through a per-sandbox label, and carrying an
|
|
376
|
+
* `ownerReferences` entry naming that same object — so `destroy()` deletes
|
|
377
|
+
* the claim (or the Sandbox) and the cluster's garbage collector removes the
|
|
378
|
+
* policy, with this backend issuing no deletion of its own.
|
|
379
|
+
*
|
|
380
|
+
* It is NOT the docker backend's mechanism. That one is an egress PROXY with
|
|
381
|
+
* no relationship to any policy object; the only two things worth taking from
|
|
382
|
+
* it are its refusal discipline and `SandboxNetworkPolicy.allowedHosts`'s
|
|
383
|
+
* `.domain`-prefix wildcard semantics.
|
|
384
|
+
*
|
|
385
|
+
* A per-sandbox list ADDS to {@link KubernetesEgressConfig.policy}, because
|
|
386
|
+
* the cluster unions every policy that selects a pod. That makes it the whole
|
|
387
|
+
* boundary under a `'no-network'` or `'deny-all'` baseline — the deployment
|
|
388
|
+
* this is for — and a no-op addition under `'allow-all'`, where
|
|
389
|
+
* `setNetworkPolicy([])` also does not deny everything the way
|
|
390
|
+
* `SandboxNetworkPolicy` describes. Neither combination is refused; pair this
|
|
391
|
+
* with a denying `policy.kind` when `setNetworkPolicy` is meant to BE the
|
|
392
|
+
* boundary rather than to widen one.
|
|
393
|
+
*/
|
|
394
|
+
export interface KubernetesPerSandboxEgressConfig {
|
|
395
|
+
/**
|
|
396
|
+
* Must be `'cilium'`. `'core'` is accepted by the TYPE and refused
|
|
397
|
+
* SYNCHRONOUSLY during host wiring, the same way
|
|
398
|
+
* {@link assertEgressPolicyIsEnforceable} refuses a hostname allowlist
|
|
399
|
+
* with no FQDN-capable engine: core `NetworkPolicy` has no hostname
|
|
400
|
+
* concept at all, so there is nothing for an `allowedHosts` list to
|
|
401
|
+
* become. It is spelled out in the type rather than fixed to the one
|
|
402
|
+
* legal value so the refusal can name what was configured.
|
|
403
|
+
*/
|
|
404
|
+
readonly engine: KubernetesEgressEngine
|
|
405
|
+
/**
|
|
406
|
+
* Name of the operator-applied `ValidatingAdmissionPolicy` that bounds
|
|
407
|
+
* what this host may write. Checked — with its binding — BEFORE the first
|
|
408
|
+
* write, and the write is refused with nothing sent if either object is
|
|
409
|
+
* missing.
|
|
410
|
+
*
|
|
411
|
+
* The fence is the whole reason this capability can be granted at all:
|
|
412
|
+
* the RBAC it needs is `create`/`patch`/`delete` on the namespace's
|
|
413
|
+
* `ciliumnetworkpolicies`, which without a fence would let a compromised
|
|
414
|
+
* host widen or delete the operator's own baseline policy. The shipped
|
|
415
|
+
* example is `k8s/manifests/validatingadmissionpolicy-cilium.yaml`.
|
|
416
|
+
*/
|
|
417
|
+
readonly admissionPolicyName: string
|
|
418
|
+
/**
|
|
419
|
+
* Name of the `ValidatingAdmissionPolicyBinding` that ATTACHES the policy
|
|
420
|
+
* above. Defaults to `${admissionPolicyName}-binding`, which is what the
|
|
421
|
+
* shipped manifest names it.
|
|
422
|
+
*
|
|
423
|
+
* Checked separately because a `ValidatingAdmissionPolicy` with no
|
|
424
|
+
* binding validates nothing at all — it is inert, and an inert fence
|
|
425
|
+
* reads exactly like an enforced one from the object alone.
|
|
426
|
+
*/
|
|
427
|
+
readonly admissionPolicyBindingName?: string
|
|
428
|
+
/**
|
|
429
|
+
* Label key the per-sandbox selector is written under. Defaults to
|
|
430
|
+
* {@link DEFAULT_PER_SANDBOX_EGRESS_LABEL_KEY}.
|
|
431
|
+
*
|
|
432
|
+
* Its VALUE is the name of the object this backend created for the
|
|
433
|
+
* sandbox, so it is unique per acquire and known before the create POST —
|
|
434
|
+
* which is what lets it travel as claim-time pod metadata through
|
|
435
|
+
* {@link composeAdditionalPodLabels}, the ONE composer, rather than as a
|
|
436
|
+
* patch to a running pod this backend has no verb for.
|
|
437
|
+
*
|
|
438
|
+
* Same operator prerequisite as {@link KubernetesEgressConfig.profile}:
|
|
439
|
+
* the key's domain has to be in the controller's `allowed-label-domains`
|
|
440
|
+
* allowlist, or the claim is refused with the controller's own reason.
|
|
441
|
+
*/
|
|
442
|
+
readonly labelKey?: string
|
|
443
|
+
/**
|
|
444
|
+
* Port, DNS-name and TLS-server-name narrowing applied to the policies
|
|
445
|
+
* `setNetworkPolicy` writes — the same shape as
|
|
446
|
+
* {@link KubernetesEgressConfig.ciliumNarrowing}, and deliberately its own
|
|
447
|
+
* field rather than a reuse of it: that one narrows the ONE config-level
|
|
448
|
+
* policy, which is only a hostname allowlist at all when
|
|
449
|
+
* `policy.kind` is `'static'` or `'resolver'`, while these narrow the
|
|
450
|
+
* per-sandbox policies of a deployment whose baseline is usually
|
|
451
|
+
* `'deny-all'` or `'no-network'`.
|
|
452
|
+
*
|
|
453
|
+
* Unset (the default) emits one `toFQDNs` rule per host with no
|
|
454
|
+
* `toPorts` at all — every port on an allowed host reachable, which is
|
|
455
|
+
* what `SandboxNetworkPolicy` itself says (it names hosts, not ports).
|
|
456
|
+
*
|
|
457
|
+
* `hostPorts` is keyed by the `allowedHosts` entry AS WRITTEN, leading
|
|
458
|
+
* dot included: a `.example.com` entry needs the key `.example.com`, even
|
|
459
|
+
* though the rule it produces says `example.com` plus `*.example.com`.
|
|
460
|
+
*
|
|
461
|
+
* `tlsServerNames` and a `.domain` entry are REFUSED together, with
|
|
462
|
+
* {@link KubernetesNetworkPolicyHostError} and nothing written. A TLS
|
|
463
|
+
* server name is one exact SNI value; the expanded entry is a name and a
|
|
464
|
+
* pattern, and no single SNI value means both — `example.com` would deny
|
|
465
|
+
* every subdomain the policy claims to allow. List the exact hosts, or
|
|
466
|
+
* leave `tlsServerNames` off for a domain list.
|
|
467
|
+
*/
|
|
468
|
+
readonly narrowing?: KubernetesCiliumEgressNarrowing
|
|
469
|
+
}
|
|
470
|
+
|
|
471
|
+
/**
|
|
472
|
+
* Default {@link KubernetesEgressConfig.profileLabelKey} — this backend's
|
|
473
|
+
* own label domain, matching `SANDBOX_TEMPLATE_LABEL_KEY`'s prefix so
|
|
474
|
+
* every label a namzu host puts on a sandbox pod reads as one family.
|
|
475
|
+
*
|
|
476
|
+
* It is deliberately NOT upstream's `sandbox.users.io`: a key in someone
|
|
477
|
+
* else's domain is a key someone else may define differently. The cost is
|
|
478
|
+
* the ConfigMap edit named on {@link KubernetesEgressConfig.profile}, and
|
|
479
|
+
* the controller's refusal spells that edit out itself.
|
|
480
|
+
*/
|
|
481
|
+
export const DEFAULT_EGRESS_PROFILE_LABEL_KEY = 'sandbox.namzu.ai/egress-profile'
|
|
482
|
+
|
|
483
|
+
/** One resolved profile label: the key it is written under and its value. */
|
|
484
|
+
export interface EgressProfileLabel {
|
|
485
|
+
readonly key: string
|
|
486
|
+
readonly value: string
|
|
487
|
+
}
|
|
488
|
+
|
|
489
|
+
/**
|
|
490
|
+
* Kubernetes' own label-value grammar, narrowed to DNS-1123: lowercase
|
|
491
|
+
* alphanumerics and `-`, starting and ending alphanumeric, at most 63
|
|
492
|
+
* characters.
|
|
493
|
+
*
|
|
494
|
+
* Narrower than what a label value may legally hold (`_` and `.` are legal
|
|
495
|
+
* there, and uppercase is too) because the profile is also a NAME: it goes
|
|
496
|
+
* into `${template}-${profile}-egress`, which has to be a legal object name,
|
|
497
|
+
* and a value that is legal as a label but not as a name would produce a
|
|
498
|
+
* policy an operator cannot apply.
|
|
499
|
+
*/
|
|
500
|
+
const DNS_1123_LABEL = /^[a-z0-9]([-a-z0-9]{0,61}[a-z0-9])?$/
|
|
501
|
+
|
|
502
|
+
/**
|
|
503
|
+
* A label KEY: an optional DNS-subdomain prefix, a `/`, then a name segment
|
|
504
|
+
* of at most 63 characters. Exactly what the API server enforces, checked
|
|
505
|
+
* here so a typo is refused during host wiring rather than as a claim the
|
|
506
|
+
* controller rejects one round trip later.
|
|
507
|
+
*/
|
|
508
|
+
const LABEL_KEY_NAME = /^[A-Za-z0-9]([-A-Za-z0-9_.]{0,61}[A-Za-z0-9])?$/
|
|
509
|
+
const LABEL_KEY_PREFIX = /^[a-z0-9]([-a-z0-9.]{0,251}[a-z0-9])?$/
|
|
510
|
+
|
|
511
|
+
/**
|
|
512
|
+
* Named refusal for an egress PROFILE this backend can read but not use.
|
|
513
|
+
* Sibling of {@link KubernetesEgressPolicyConfigError} rather than a reuse of
|
|
514
|
+
* it: that one names a field under `config.egress.policy` and this one names
|
|
515
|
+
* a field beside it, and an operator reading either should not have to work
|
|
516
|
+
* out which level of the config the path belongs to.
|
|
517
|
+
*
|
|
518
|
+
* Thrown SYNCHRONOUSLY from `buildKubernetesBackend` and
|
|
519
|
+
* `createKubernetesWorkspace`, so a misconfigured profile surfaces during
|
|
520
|
+
* host wiring rather than on the first `create()`.
|
|
521
|
+
*/
|
|
522
|
+
export class KubernetesEgressProfileConfigError extends Error {
|
|
523
|
+
override readonly name = 'KubernetesEgressProfileConfigError'
|
|
524
|
+
|
|
525
|
+
constructor(
|
|
526
|
+
readonly field: 'profile' | 'profileLabelKey',
|
|
527
|
+
readonly value: string,
|
|
528
|
+
reason: string,
|
|
529
|
+
) {
|
|
530
|
+
super(
|
|
531
|
+
`kubernetes: config.egress.${field} is unusable: ${JSON.stringify(value)} ${reason}. The profile travels onto a SandboxClaim's additionalPodMetadata.labels, onto a directly created Sandbox's pod template and into the translated policy's own selector, so a value the API server would reject leaves either a claim nothing binds or a policy nobody can apply. Refusing here rather than emitting it.`,
|
|
532
|
+
)
|
|
533
|
+
}
|
|
534
|
+
}
|
|
535
|
+
|
|
536
|
+
/**
|
|
537
|
+
* The profile label this config asks for, or nothing at all — the ONE place
|
|
538
|
+
* the key/value pair is derived, so the claim body, the Sandbox pod
|
|
539
|
+
* template, the policy selector and the policy name cannot disagree about
|
|
540
|
+
* what the profile is.
|
|
541
|
+
*
|
|
542
|
+
* Validates as it resolves: the value has to be a DNS-1123 label and the key
|
|
543
|
+
* a legal label key, both refused with {@link KubernetesEgressProfileConfigError}.
|
|
544
|
+
*/
|
|
545
|
+
export function egressProfileLabel(
|
|
546
|
+
egress: KubernetesEgressConfig | undefined,
|
|
547
|
+
): EgressProfileLabel | undefined {
|
|
548
|
+
// The KEY is validated whenever it is present, profile or no profile. A
|
|
549
|
+
// key set without a value is a half-finished configuration — the value is
|
|
550
|
+
// usually the next line someone writes — and reporting the typo only once
|
|
551
|
+
// the profile arrives is reporting it at the second edit rather than the
|
|
552
|
+
// first.
|
|
553
|
+
const configured = egress?.profileLabelKey
|
|
554
|
+
if (configured !== undefined) assertUsableProfileLabelKey(configured)
|
|
555
|
+
const value = egress?.profile
|
|
556
|
+
if (value === undefined) return undefined
|
|
557
|
+
if (!DNS_1123_LABEL.test(value)) {
|
|
558
|
+
throw new KubernetesEgressProfileConfigError(
|
|
559
|
+
'profile',
|
|
560
|
+
value,
|
|
561
|
+
'is not a DNS-1123 label (lowercase letters, digits and dashes, starting and ending alphanumeric, at most 63 characters)',
|
|
562
|
+
)
|
|
563
|
+
}
|
|
564
|
+
const key = configured ?? DEFAULT_EGRESS_PROFILE_LABEL_KEY
|
|
565
|
+
return { key, value }
|
|
566
|
+
}
|
|
567
|
+
|
|
568
|
+
/**
|
|
569
|
+
* Why a label KEY is unusable, or `undefined` when it is fine — the grammar
|
|
570
|
+
* half of the two checks below, shared because the API server's rule is the
|
|
571
|
+
* same whichever of this backend's label keys is being configured and two
|
|
572
|
+
* spellings of it would be two rules.
|
|
573
|
+
*/
|
|
574
|
+
function labelKeyProblem(key: string): string | undefined {
|
|
575
|
+
const slash = key.indexOf('/')
|
|
576
|
+
const name = slash === -1 ? key : key.slice(slash + 1)
|
|
577
|
+
const prefix = slash === -1 ? undefined : key.slice(0, slash)
|
|
578
|
+
if (
|
|
579
|
+
!LABEL_KEY_NAME.test(name) ||
|
|
580
|
+
(prefix !== undefined && !LABEL_KEY_PREFIX.test(prefix)) ||
|
|
581
|
+
key.indexOf('/', slash + 1) !== -1
|
|
582
|
+
) {
|
|
583
|
+
return 'is not a Kubernetes label key (an optional DNS-subdomain prefix, a single slash, then a name of at most 63 characters)'
|
|
584
|
+
}
|
|
585
|
+
return undefined
|
|
586
|
+
}
|
|
587
|
+
|
|
588
|
+
/**
|
|
589
|
+
* Refuse a `profileLabelKey` the API server would not take, or that this
|
|
590
|
+
* backend already uses for something else.
|
|
591
|
+
*/
|
|
592
|
+
function assertUsableProfileLabelKey(key: string): void {
|
|
593
|
+
const problem = labelKeyProblem(key)
|
|
594
|
+
if (problem !== undefined) {
|
|
595
|
+
throw new KubernetesEgressProfileConfigError('profileLabelKey', key, problem)
|
|
596
|
+
}
|
|
597
|
+
// The one legal key that must not be used: it is the key this backend
|
|
598
|
+
// stamps the SandboxTemplate name under, and the profile label is applied
|
|
599
|
+
// LAST (see `sandboxPodLabels`), so this key would overwrite the template
|
|
600
|
+
// label on every pod this backend creates — and the translated policy's
|
|
601
|
+
// selector, built from the same resolution, would agree with it. Both
|
|
602
|
+
// halves would be wrong together, which is exactly the shape nothing else
|
|
603
|
+
// would catch.
|
|
604
|
+
if (key === SANDBOX_TEMPLATE_LABEL_KEY) {
|
|
605
|
+
throw new KubernetesEgressProfileConfigError(
|
|
606
|
+
'profileLabelKey',
|
|
607
|
+
key,
|
|
608
|
+
'is the key this backend writes the SandboxTemplate name under, and a profile label is applied last — a pod would carry the profile value where its template label belongs, and the policy selector built from the same resolution would match it anyway',
|
|
609
|
+
)
|
|
610
|
+
}
|
|
611
|
+
}
|
|
612
|
+
|
|
613
|
+
/**
|
|
614
|
+
* Validate the profile without needing its value — the wiring-time hook, so
|
|
615
|
+
* `buildKubernetesBackend` and `createKubernetesWorkspace` refuse a bad
|
|
616
|
+
* profile the same moment they refuse an unenforceable policy.
|
|
617
|
+
*/
|
|
618
|
+
export function assertEgressProfileIsUsable(egress: KubernetesEgressConfig | undefined): void {
|
|
619
|
+
egressProfileLabel(egress)
|
|
620
|
+
}
|
|
621
|
+
|
|
622
|
+
/**
|
|
623
|
+
* Default {@link KubernetesPerSandboxEgressConfig.labelKey} — the key the
|
|
624
|
+
* per-sandbox policy's `endpointSelector` matches on.
|
|
625
|
+
*
|
|
626
|
+
* In this backend's own label domain, for the same reason the profile key is
|
|
627
|
+
* (see {@link DEFAULT_EGRESS_PROFILE_LABEL_KEY}), and carrying the same
|
|
628
|
+
* operator prerequisite: the controller's `allowed-label-domains` allowlist
|
|
629
|
+
* has to admit `sandbox.namzu.ai`, or every claim carrying it is refused with
|
|
630
|
+
* {@link KubernetesPodLabelsRejectedError} as the cause. A deployment that
|
|
631
|
+
* would rather not edit that ConfigMap sets this to a key under
|
|
632
|
+
* `sandbox.users.io`, which is upstream's own default domain.
|
|
633
|
+
*/
|
|
634
|
+
export const DEFAULT_PER_SANDBOX_EGRESS_LABEL_KEY = 'sandbox.namzu.ai/per-sandbox-egress'
|
|
635
|
+
|
|
636
|
+
/**
|
|
637
|
+
* Named refusal for a `config.egress.perSandbox` this backend can read but
|
|
638
|
+
* not honour. Thrown SYNCHRONOUSLY from `buildKubernetesBackend`, beside the
|
|
639
|
+
* policy and profile refusals, so a host that has mis-declared the capability
|
|
640
|
+
* learns it during wiring rather than from the first `setNetworkPolicy` call
|
|
641
|
+
* — which may be an hour into a run, after work that cannot be redone.
|
|
642
|
+
*
|
|
643
|
+
* Its own class rather than a reuse of {@link KubernetesEgressProfileConfigError}
|
|
644
|
+
* for the reason that one is not a reuse of
|
|
645
|
+
* {@link KubernetesEgressPolicyConfigError}: an operator reading a refusal
|
|
646
|
+
* should not have to work out which level of `config.egress` the named field
|
|
647
|
+
* belongs to.
|
|
648
|
+
*/
|
|
649
|
+
export class KubernetesPerSandboxEgressConfigError extends Error {
|
|
650
|
+
override readonly name = 'KubernetesPerSandboxEgressConfigError'
|
|
651
|
+
|
|
652
|
+
constructor(
|
|
653
|
+
readonly field: 'engine' | 'admissionPolicyName' | 'admissionPolicyBindingName' | 'labelKey',
|
|
654
|
+
readonly value: string,
|
|
655
|
+
reason: string,
|
|
656
|
+
) {
|
|
657
|
+
super(
|
|
658
|
+
`kubernetes: config.egress.perSandbox.${field} is unusable: ${JSON.stringify(value)} ${reason}. Per-sandbox egress writes one CiliumNetworkPolicy per live sandbox, selected by a per-sandbox pod label and fenced by an operator-applied ValidatingAdmissionPolicy, so a value this backend cannot honour would leave either a policy that selects nothing or a write the fence refuses. Refusing during host wiring rather than at the first setNetworkPolicy call.`,
|
|
659
|
+
)
|
|
660
|
+
}
|
|
661
|
+
}
|
|
662
|
+
|
|
663
|
+
/**
|
|
664
|
+
* The per-sandbox selector label KEY this config asks for, or nothing at all
|
|
665
|
+
* when `perSandbox` is unset — the one place the default is applied, so the
|
|
666
|
+
* pod label, the policy selector and the admission policy's own expectation
|
|
667
|
+
* cannot disagree.
|
|
668
|
+
*
|
|
669
|
+
* Validates as it resolves, exactly as {@link egressProfileLabel} does.
|
|
670
|
+
*/
|
|
671
|
+
export function perSandboxEgressLabelKey(
|
|
672
|
+
egress: KubernetesEgressConfig | undefined,
|
|
673
|
+
): string | undefined {
|
|
674
|
+
const perSandbox = egress?.perSandbox
|
|
675
|
+
if (perSandbox === undefined) return undefined
|
|
676
|
+
const key = perSandbox.labelKey ?? DEFAULT_PER_SANDBOX_EGRESS_LABEL_KEY
|
|
677
|
+
const problem = labelKeyProblem(key)
|
|
678
|
+
if (problem !== undefined) {
|
|
679
|
+
throw new KubernetesPerSandboxEgressConfigError('labelKey', key, problem)
|
|
680
|
+
}
|
|
681
|
+
if (key === SANDBOX_TEMPLATE_LABEL_KEY) {
|
|
682
|
+
throw new KubernetesPerSandboxEgressConfigError(
|
|
683
|
+
'labelKey',
|
|
684
|
+
key,
|
|
685
|
+
'is the key this backend writes the SandboxTemplate name under, so a pod would carry a sandbox name where its template label belongs and every policy selecting the template would stop selecting it',
|
|
686
|
+
)
|
|
687
|
+
}
|
|
688
|
+
// The profile's key is refused from the other direction too — see
|
|
689
|
+
// `composeAdditionalPodLabels` — but naming it HERE names the field a
|
|
690
|
+
// reader has to change, which the generic collision message cannot.
|
|
691
|
+
const profile = egressProfileLabel(egress)
|
|
692
|
+
if (profile !== undefined && key === profile.key) {
|
|
693
|
+
throw new KubernetesPerSandboxEgressConfigError(
|
|
694
|
+
'labelKey',
|
|
695
|
+
key,
|
|
696
|
+
'is also config.egress.profileLabelKey, and both are written into the SAME pod-label map — one value would silently replace the other, and whichever lost would be a label a policy selector still expects',
|
|
697
|
+
)
|
|
698
|
+
}
|
|
699
|
+
return key
|
|
700
|
+
}
|
|
701
|
+
|
|
702
|
+
/**
|
|
703
|
+
* Validate `config.egress.perSandbox` in full — the wiring-time hook, so a
|
|
704
|
+
* mis-declared capability is refused the same moment an unenforceable policy
|
|
705
|
+
* or an unusable profile is.
|
|
706
|
+
*
|
|
707
|
+
* `engine: 'core'` is the refusal the SDK's contract cares about: core
|
|
708
|
+
* `NetworkPolicy` cannot express a hostname at all, so a host that configured
|
|
709
|
+
* it would be told "policy applied" about an object that could never carry
|
|
710
|
+
* the allowlist. It is refused here rather than from the method, which means
|
|
711
|
+
* a `'core'` deployment never gets a handle carrying the method in the first
|
|
712
|
+
* place.
|
|
713
|
+
*/
|
|
714
|
+
export function assertPerSandboxEgressIsUsable(egress: KubernetesEgressConfig | undefined): void {
|
|
715
|
+
const perSandbox = egress?.perSandbox
|
|
716
|
+
if (perSandbox === undefined) return
|
|
717
|
+
if (perSandbox.engine !== 'cilium') {
|
|
718
|
+
throw new KubernetesPerSandboxEgressConfigError(
|
|
719
|
+
'engine',
|
|
720
|
+
perSandbox.engine,
|
|
721
|
+
"is not 'cilium'; a per-sandbox allowlist is a list of HOSTNAMES, and core NetworkPolicy has only ipBlock, podSelector and namespaceSelector — it has no hostname concept to translate one into",
|
|
722
|
+
)
|
|
723
|
+
}
|
|
724
|
+
// `admissionPolicyName` is REQUIRED by the type; the cast is for the
|
|
725
|
+
// caller reaching here from JavaScript, or through a cast of its own.
|
|
726
|
+
// There is no default and there must not be one: the fence is what makes
|
|
727
|
+
// the policy-write RBAC safe to grant, and a host that could skip naming
|
|
728
|
+
// it would hold create/patch/delete on every CiliumNetworkPolicy in the
|
|
729
|
+
// namespace with nothing bounding what it writes.
|
|
730
|
+
const names = [
|
|
731
|
+
['admissionPolicyName', perSandbox.admissionPolicyName as string | undefined, true],
|
|
732
|
+
['admissionPolicyBindingName', perSandbox.admissionPolicyBindingName, false],
|
|
733
|
+
] as const
|
|
734
|
+
for (const [field, value, required] of names) {
|
|
735
|
+
if (value === undefined) {
|
|
736
|
+
if (!required) continue
|
|
737
|
+
throw new KubernetesPerSandboxEgressConfigError(
|
|
738
|
+
field,
|
|
739
|
+
'',
|
|
740
|
+
'is required, and has no default: an unnamed fence is an unchecked one',
|
|
741
|
+
)
|
|
742
|
+
}
|
|
743
|
+
if (value === '' || value.trim() !== value) {
|
|
744
|
+
throw new KubernetesPerSandboxEgressConfigError(
|
|
745
|
+
field,
|
|
746
|
+
value,
|
|
747
|
+
'is not an object name (an empty or space-padded name matches nothing, so the fence check would refuse every write)',
|
|
748
|
+
)
|
|
749
|
+
}
|
|
750
|
+
}
|
|
751
|
+
if (perSandbox.narrowing !== undefined) {
|
|
752
|
+
assertCiliumNarrowingIsUsable(perSandbox.narrowing, 'perSandbox.narrowing')
|
|
753
|
+
}
|
|
754
|
+
perSandboxEgressLabelKey(egress)
|
|
755
|
+
}
|
|
756
|
+
|
|
757
|
+
/**
|
|
758
|
+
* Named refusal for `config.egress.perSandbox` reaching an entry point that
|
|
759
|
+
* can never carry the capability it configures: `createKubernetesWorkspace`.
|
|
760
|
+
*
|
|
761
|
+
* `perSandbox` exists to make `Sandbox.setNetworkPolicy` PRESENT on a TASK
|
|
762
|
+
* handle, where the acquire creates the object each policy is named after and
|
|
763
|
+
* owned by and stamps the per-sandbox pod label its selector matches, and
|
|
764
|
+
* where the RBAC and the admission fence are the ones that bound the write.
|
|
765
|
+
* A workspace's create path does none of that — it composes no per-sandbox
|
|
766
|
+
* pod label and tracks no owner uid for one — so on that path the option
|
|
767
|
+
* would be accepted and mean nothing at all: exactly the "declared and
|
|
768
|
+
* silently unused" configuration this module refuses everywhere else. The
|
|
769
|
+
* alternative, omitting the method and saying nothing, is how a host comes to
|
|
770
|
+
* believe it narrowed a workspace's egress.
|
|
771
|
+
*
|
|
772
|
+
* NOT a variant of {@link KubernetesPerSandboxEgressConfigError}, which is
|
|
773
|
+
* about a `perSandbox` value this backend cannot honour ANYWHERE (an engine
|
|
774
|
+
* with no hostname concept, an unnamed fence). This one is about a `perSandbox`
|
|
775
|
+
* value it honours perfectly well on the other entry point, so a caller
|
|
776
|
+
* catching the sibling for a typo'd value does not also catch a correct
|
|
777
|
+
* configuration aimed at the wrong path.
|
|
778
|
+
*/
|
|
779
|
+
export class KubernetesWorkspacePerSandboxEgressConfigError extends Error {
|
|
780
|
+
override readonly name = 'KubernetesWorkspacePerSandboxEgressConfigError'
|
|
781
|
+
|
|
782
|
+
constructor() {
|
|
783
|
+
super(
|
|
784
|
+
`kubernetes: config.egress.perSandbox is set, but a KubernetesWorkspace cannot carry the capability it configures: Sandbox.setNetworkPolicy is implemented on a TASK handle, whose acquire creates the object each per-sandbox policy is named after and owned by and stamps the pod label its selector matches, while a workspace's create path composes no per-sandbox pod label and tracks no owner uid for one. Accepting the option here would declare a capability this path never serves — the silent downgrade this backend refuses on every other entry point — so it is refused before anything is sent. Drop egress.perSandbox from the configuration workspaces are created from (a backend serving task sandboxes can keep it), or create a task sandbox with it, where the method is present.`,
|
|
785
|
+
)
|
|
786
|
+
}
|
|
787
|
+
}
|
|
788
|
+
|
|
789
|
+
/**
|
|
790
|
+
* Refuse `config.egress.perSandbox` on the entry point that cannot carry it —
|
|
791
|
+
* `createKubernetesWorkspace`. See
|
|
792
|
+
* {@link KubernetesWorkspacePerSandboxEgressConfigError}.
|
|
793
|
+
*
|
|
794
|
+
* Deliberately not {@link assertPerSandboxEgressIsUsable}, which validates the
|
|
795
|
+
* same config for the path that DOES honour it: a workspace has nothing to
|
|
796
|
+
* validate the option for, whatever its `engine` or `admissionPolicyName` say,
|
|
797
|
+
* because it serves no `setNetworkPolicy` at all. Calling that validator here
|
|
798
|
+
* instead would be worse than saying nothing — it would report a
|
|
799
|
+
* `perSandbox` this deployment can use as if it were in use.
|
|
800
|
+
*/
|
|
801
|
+
export function assertWorkspaceCarriesNoPerSandboxEgress(
|
|
802
|
+
egress: KubernetesEgressConfig | undefined,
|
|
803
|
+
): void {
|
|
804
|
+
if (egress?.perSandbox === undefined) return
|
|
805
|
+
throw new KubernetesWorkspacePerSandboxEgressConfigError()
|
|
806
|
+
}
|
|
807
|
+
|
|
808
|
+
/**
|
|
809
|
+
* The ONE composer for a sandbox pod's `additionalPodMetadata.labels`.
|
|
810
|
+
*
|
|
811
|
+
* Every label this backend asks the controller to put on a POD is built
|
|
812
|
+
* here: the egress profile today, and whatever a later capability
|
|
813
|
+
* contributes through `extra` (a per-sandbox policy selector, for one). Two
|
|
814
|
+
* independent constructions would be two answers to "what labels is this pod
|
|
815
|
+
* selected by", and the policy selector is built from the same resolution —
|
|
816
|
+
* so a second builder would be a pod bound under a policy nobody checked.
|
|
817
|
+
*
|
|
818
|
+
* Deliberately NOT where `KubernetesBackendInternalConfig.claimLabels`
|
|
819
|
+
* goes. Those are a host's own bookkeeping on the CLAIM object's
|
|
820
|
+
* `metadata.labels`; putting them on the pod would change what selectors
|
|
821
|
+
* match a running sandbox, which is a different question on a different
|
|
822
|
+
* object.
|
|
823
|
+
*
|
|
824
|
+
* Returns an empty object when nothing applies, which every caller reads as
|
|
825
|
+
* "emit nothing at all" — that is what keeps an unprofiled body byte-identical.
|
|
826
|
+
*
|
|
827
|
+
* A key in `extra` that is ALSO the profile's is refused rather than merged
|
|
828
|
+
* either way round. Whichever won, the loser would be a label the translated
|
|
829
|
+
* policy's selector still expects: the profile's selector is built from this
|
|
830
|
+
* same resolution, so a pod carrying the other value is selected by no
|
|
831
|
+
* per-profile policy while `create()` reported the boundary verified. It is
|
|
832
|
+
* the same failure {@link egressProfileLabel} refuses the template key for,
|
|
833
|
+
* reached from the other direction.
|
|
834
|
+
*/
|
|
835
|
+
export function composeAdditionalPodLabels(
|
|
836
|
+
egress: KubernetesEgressConfig | undefined,
|
|
837
|
+
extra?: Readonly<Record<string, string>>,
|
|
838
|
+
): Readonly<Record<string, string>> {
|
|
839
|
+
const profile = egressProfileLabel(egress)
|
|
840
|
+
if (profile !== undefined && extra !== undefined && profile.key in extra) {
|
|
841
|
+
throw new KubernetesEgressProfileConfigError(
|
|
842
|
+
'profileLabelKey',
|
|
843
|
+
profile.key,
|
|
844
|
+
`is also the key another capability contributes to the same pod-label map (as ${JSON.stringify(extra[profile.key])}), and one of the two values would silently replace the other`,
|
|
845
|
+
)
|
|
846
|
+
}
|
|
847
|
+
return {
|
|
848
|
+
...(profile !== undefined ? { [profile.key]: profile.value } : {}),
|
|
849
|
+
...extra,
|
|
850
|
+
}
|
|
851
|
+
}
|
|
852
|
+
|
|
853
|
+
/**
|
|
854
|
+
* Why a claim the controller refused with `InvalidMetadata` was refused, in
|
|
855
|
+
* this backend's own terms — in practice a pod label whose domain is not in
|
|
856
|
+
* the controller's `allowed-label-domains` allowlist, which today means the
|
|
857
|
+
* egress profile's.
|
|
858
|
+
*
|
|
859
|
+
* NOT what an acquire throws. A refused claim comes out of `create()` as
|
|
860
|
+
* `KubernetesAcquireError { reason: 'claim-rejected' }` whether or not a
|
|
861
|
+
* profile is configured — one condition, one taxonomy, one `catch` — and this
|
|
862
|
+
* rides as that error's `cause`. Two classes for one controller condition
|
|
863
|
+
* would make a host's error handling correct or incorrect depending on
|
|
864
|
+
* whether `config.egress.profile` happened to be set.
|
|
865
|
+
*
|
|
866
|
+
* What it adds to the acquire error is what the controller cannot know: the
|
|
867
|
+
* map this backend actually sent, and the `config.egress.profileLabelKey`
|
|
868
|
+
* that moves the offending key to an allowed domain. It carries the whole map
|
|
869
|
+
* rather than the profile alone, and `profile` is optional, because the map is
|
|
870
|
+
* {@link composeAdditionalPodLabels}'s — the profile is the only thing in it
|
|
871
|
+
* today, and a later capability adding a second key would otherwise get an
|
|
872
|
+
* explanation that named a label it did not send.
|
|
873
|
+
*
|
|
874
|
+
* It carries the controller's OWN `reason` and `message` rather than a
|
|
875
|
+
* translation of them: the message agent-sandbox v1.0.2 writes names the
|
|
876
|
+
* offending key, the domain, the ConfigMap key to edit and its default, and
|
|
877
|
+
* no paraphrase of it would be as useful. Measured verbatim against a kind
|
|
878
|
+
* cluster running that controller:
|
|
879
|
+
*
|
|
880
|
+
* > invalid additionalPodMetadata: failed to validate label
|
|
881
|
+
* > "sandbox.namzu.ai/egress-profile": label domain "sandbox.namzu.ai" is
|
|
882
|
+
* > not in the allowlist (configure the allowed-label-domains key of the
|
|
883
|
+
* > agent-sandbox-config ConfigMap in the controller namespace; default:
|
|
884
|
+
* > sandbox.users.io)
|
|
885
|
+
*
|
|
886
|
+
* The refusal is raised as soon as that condition is read rather than after
|
|
887
|
+
* the readiness budget, because `InvalidMetadata` is one of
|
|
888
|
+
* `TERMINAL_CLAIM_REASONS` — nothing about it becomes true by waiting — and
|
|
889
|
+
* the claim is deleted on the way out, so a misconfigured profile costs one
|
|
890
|
+
* round trip rather than a minute of polling.
|
|
891
|
+
*/
|
|
892
|
+
export class KubernetesPodLabelsRejectedError extends Error {
|
|
893
|
+
override readonly name = 'KubernetesPodLabelsRejectedError'
|
|
894
|
+
|
|
895
|
+
constructor(
|
|
896
|
+
/** Every label this backend put on the claim's `additionalPodMetadata`. */
|
|
897
|
+
readonly requestedPodLabels: Readonly<Record<string, string>>,
|
|
898
|
+
readonly claimName: string,
|
|
899
|
+
readonly namespace: string,
|
|
900
|
+
/** The controller's own condition `reason`, e.g. `InvalidMetadata`. */
|
|
901
|
+
readonly controllerReason: string,
|
|
902
|
+
/** The controller's own condition `message`, verbatim. */
|
|
903
|
+
readonly controllerMessage: string,
|
|
904
|
+
/** The egress profile among those labels, when one is configured. */
|
|
905
|
+
readonly profile?: EgressProfileLabel,
|
|
906
|
+
) {
|
|
907
|
+
super(
|
|
908
|
+
`kubernetes: the controller refused SandboxClaim ${claimName} in namespace ${namespace} carrying the pod labels ${formatLabels(requestedPodLabels)} — ${controllerReason}: ${controllerMessage}. Those labels are written onto the claim's additionalPodMetadata.labels, so each key's domain has to appear in the controller's allowed-label-domains allowlist; add it there, or move the offending key to a domain that is already allowed${profile === undefined ? '' : ` (config.egress.profileLabelKey, for the egress profile ${profile.key}=${profile.value})`}. The claim has been deleted.`,
|
|
909
|
+
)
|
|
910
|
+
}
|
|
911
|
+
}
|
|
912
|
+
|
|
913
|
+
/**
|
|
914
|
+
* Named refusal for a bound pod that never carried a label this backend asked
|
|
915
|
+
* the controller to put on it — the egress profile's, today the only one
|
|
916
|
+
* {@link composeAdditionalPodLabels} produces, which is why the class is named
|
|
917
|
+
* for the LABEL rather than for the profile.
|
|
918
|
+
*
|
|
919
|
+
* This is the one failure this capability must not have quietly. An
|
|
920
|
+
* unlabelled pod handed back is a sandbox running under the DEFAULT policy
|
|
921
|
+
* while the host believes it is on a narrower profile — the translated
|
|
922
|
+
* policy's selector includes the profile label, so a pod without it is
|
|
923
|
+
* selected by neither this profile's policy nor, necessarily, anything else.
|
|
924
|
+
* Refusing is correct and waiting is correct; proceeding is not, so the
|
|
925
|
+
* acquire releases what it claimed and raises this instead.
|
|
926
|
+
*
|
|
927
|
+
* `missingLabel` is the pair that never arrived rather than "the profile", so
|
|
928
|
+
* the refusal stays true for whatever a later capability contributes to that
|
|
929
|
+
* same map: the wait in `readAddressedPod` already covers every entry of it,
|
|
930
|
+
* and this refusal covers exactly the same set.
|
|
931
|
+
*/
|
|
932
|
+
export class KubernetesPodLabelNotObservedError extends Error {
|
|
933
|
+
override readonly name = 'KubernetesPodLabelNotObservedError'
|
|
934
|
+
|
|
935
|
+
constructor(
|
|
936
|
+
/** The label this backend requested and never saw on the bound pod. */
|
|
937
|
+
readonly missingLabel: EgressProfileLabel,
|
|
938
|
+
readonly subject: string,
|
|
939
|
+
readonly observedLabels: Readonly<Record<string, string>>,
|
|
940
|
+
) {
|
|
941
|
+
super(
|
|
942
|
+
`kubernetes: refusing ${subject} — its pod never carried the label ${missingLabel.key}=${missingLabel.value} this backend asked the controller to put on it, within the readiness budget; the labels it did carry are ${formatLabels(observedLabels)}. The translated policy's selector includes that label, so admitting this pod would run it under whatever policy DOES select it rather than under the configured profile. The controller patches a claim's additionalPodMetadata.labels onto the pod it binds — a pod that never got them means the claim's metadata was not applied. Nothing was handed back and the claim was released.`,
|
|
943
|
+
)
|
|
944
|
+
}
|
|
945
|
+
}
|
|
946
|
+
|
|
947
|
+
/** `true` unless the deployment asked for the single-object check by name. */
|
|
948
|
+
export function egressUnionVerificationEnabled(
|
|
949
|
+
egress: KubernetesEgressConfig | undefined,
|
|
950
|
+
): boolean {
|
|
951
|
+
return egress !== undefined && egress.verify !== 'named-object-only'
|
|
122
952
|
}
|
|
123
953
|
|
|
124
954
|
/** Where a translated policy is targeted, and what its `podSelector` names. */
|
|
@@ -132,6 +962,33 @@ export interface EgressPolicyTarget {
|
|
|
132
962
|
* carries it.
|
|
133
963
|
*/
|
|
134
964
|
readonly sandboxTemplateName: string
|
|
965
|
+
/**
|
|
966
|
+
* The egress PROFILE label this policy also selects, when
|
|
967
|
+
* {@link KubernetesEgressConfig.profile} is set. Absent, the selector is
|
|
968
|
+
* the template label alone and every translation is what it always was.
|
|
969
|
+
*
|
|
970
|
+
* It is part of the SELECTOR rather than a second policy because that is
|
|
971
|
+
* what makes one warm pool serve several modes: pods out of one template
|
|
972
|
+
* carry one template label and differ only by this one, so
|
|
973
|
+
* `${template}-none-egress` selects exactly the `none` pods and
|
|
974
|
+
* `${template}-internet-egress` exactly the `internet` ones.
|
|
975
|
+
*/
|
|
976
|
+
readonly profile?: EgressProfileLabel
|
|
977
|
+
}
|
|
978
|
+
|
|
979
|
+
/**
|
|
980
|
+
* What a translated policy's `podSelector`/`endpointSelector` matches: the
|
|
981
|
+
* template label, plus the profile label when one is configured. One
|
|
982
|
+
* function so the two manifest builders below and every reader of a
|
|
983
|
+
* translation agree on the selector down to the key order.
|
|
984
|
+
*/
|
|
985
|
+
export function egressPolicySelectorLabels(
|
|
986
|
+
target: EgressPolicyTarget,
|
|
987
|
+
): Readonly<Record<string, string>> {
|
|
988
|
+
return {
|
|
989
|
+
...sandboxTemplateLabel(target.sandboxTemplateName),
|
|
990
|
+
...(target.profile !== undefined ? { [target.profile.key]: target.profile.value } : {}),
|
|
991
|
+
}
|
|
135
992
|
}
|
|
136
993
|
|
|
137
994
|
/**
|
|
@@ -146,14 +1003,49 @@ export interface EgressPolicyTarget {
|
|
|
146
1003
|
*/
|
|
147
1004
|
export interface KubernetesTranslatedEgressPolicy {
|
|
148
1005
|
readonly kind: 'NetworkPolicy' | 'CiliumNetworkPolicy'
|
|
1006
|
+
/**
|
|
1007
|
+
* The CONFIGURED kind this came from, carried so a refusal can name what
|
|
1008
|
+
* the deployment asked for (`no-network`, `public-internet`, …) rather
|
|
1009
|
+
* than only the resource it produced.
|
|
1010
|
+
*/
|
|
1011
|
+
readonly policyKind: KubernetesEgressPolicy['kind']
|
|
149
1012
|
readonly namespace: string
|
|
150
1013
|
readonly name: string
|
|
151
1014
|
readonly manifest: Readonly<Record<string, unknown>>
|
|
152
1015
|
}
|
|
153
1016
|
|
|
154
|
-
/**
|
|
155
|
-
|
|
156
|
-
|
|
1017
|
+
/**
|
|
1018
|
+
* The longest name a Kubernetes object may carry. A `NetworkPolicy` name is a
|
|
1019
|
+
* DNS subdomain, so 253 characters, and the API server refuses anything
|
|
1020
|
+
* longer.
|
|
1021
|
+
*/
|
|
1022
|
+
const MAX_OBJECT_NAME_LENGTH = 253
|
|
1023
|
+
|
|
1024
|
+
/**
|
|
1025
|
+
* `${sandboxTemplateName}-egress`, the name
|
|
1026
|
+
* {@link KubernetesEgressConfig.networkPolicyName} defaults to — or
|
|
1027
|
+
* `${sandboxTemplateName}-${profile}-egress` when a profile is configured,
|
|
1028
|
+
* because one template under two profiles needs two policy objects and a
|
|
1029
|
+
* single default name would have the second silently verify against the
|
|
1030
|
+
* first's manifest.
|
|
1031
|
+
*
|
|
1032
|
+
* The profile is bounded at 63 characters on its own, but the CONCATENATION
|
|
1033
|
+
* is what has to be a legal object name, and only this function knows both
|
|
1034
|
+
* halves. Refused here, during host wiring, rather than as an API-server
|
|
1035
|
+
* rejection on the first policy GET of the first `create()`: `networkPolicyName`
|
|
1036
|
+
* is the way out and it is a config field, so this is a config error.
|
|
1037
|
+
*/
|
|
1038
|
+
export function defaultEgressPolicyName(sandboxTemplateName: string, profile?: string): string {
|
|
1039
|
+
if (profile === undefined) return `${sandboxTemplateName}-egress`
|
|
1040
|
+
const name = `${sandboxTemplateName}-${profile}-egress`
|
|
1041
|
+
if (name.length > MAX_OBJECT_NAME_LENGTH) {
|
|
1042
|
+
throw new KubernetesEgressProfileConfigError(
|
|
1043
|
+
'profile',
|
|
1044
|
+
profile,
|
|
1045
|
+
`makes the default egress policy name ${JSON.stringify(name)} ${name.length} characters long, past the ${MAX_OBJECT_NAME_LENGTH} an object name may carry — shorten the profile or the SandboxTemplate name, or set config.egress.networkPolicyName yourself`,
|
|
1046
|
+
)
|
|
1047
|
+
}
|
|
1048
|
+
return name
|
|
157
1049
|
}
|
|
158
1050
|
|
|
159
1051
|
/**
|
|
@@ -174,45 +1066,372 @@ export class KubernetesUnenforceableEgressPolicyError extends Error {
|
|
|
174
1066
|
}
|
|
175
1067
|
|
|
176
1068
|
/**
|
|
177
|
-
*
|
|
178
|
-
* `
|
|
179
|
-
*
|
|
180
|
-
*
|
|
181
|
-
*
|
|
182
|
-
*
|
|
1069
|
+
* Named refusal for a config value this backend can read but not use — today
|
|
1070
|
+
* only a `public-internet` `exceptCidrs` entry that is not a CIDR. Thrown
|
|
1071
|
+
* SYNCHRONOUSLY from `buildKubernetesBackend`, beside
|
|
1072
|
+
* {@link KubernetesUnenforceableEgressPolicyError}, so a typo surfaces
|
|
1073
|
+
* during host wiring rather than as a policy the API server rejects when an
|
|
1074
|
+
* operator applies it.
|
|
183
1075
|
*/
|
|
184
|
-
export
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
1076
|
+
export class KubernetesEgressPolicyConfigError extends Error {
|
|
1077
|
+
override readonly name = 'KubernetesEgressPolicyConfigError'
|
|
1078
|
+
|
|
1079
|
+
constructor(
|
|
1080
|
+
readonly field: string,
|
|
1081
|
+
reason: string,
|
|
1082
|
+
) {
|
|
1083
|
+
// `field` is relative to `config.egress`, not to `config.egress.policy`:
|
|
1084
|
+
// the fields that land here live at BOTH levels — `policy.exceptCidrs`
|
|
1085
|
+
// on the policy, `ciliumNarrowing` and `perSandbox.narrowing` beside
|
|
1086
|
+
// it — and a prefix that assumed one of them sent a reader to a key
|
|
1087
|
+
// that does not exist.
|
|
1088
|
+
super(`kubernetes: config.egress.${field} is unusable: ${reason}`)
|
|
190
1089
|
}
|
|
191
1090
|
}
|
|
192
1091
|
|
|
193
1092
|
/**
|
|
194
|
-
*
|
|
195
|
-
*
|
|
196
|
-
*
|
|
1093
|
+
* Named refusal for `config.egress.ciliumNarrowing` set on a policy/engine
|
|
1094
|
+
* combination it does not apply to. Thrown SYNCHRONOUSLY from
|
|
1095
|
+
* {@link assertEgressPolicyIsEnforceable}, beside
|
|
1096
|
+
* {@link KubernetesUnenforceableEgressPolicyError}, so a narrowing option
|
|
1097
|
+
* that would silently do nothing is refused at host-wiring time rather than
|
|
1098
|
+
* accepted and ignored.
|
|
197
1099
|
*/
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
1100
|
+
export class KubernetesEgressNarrowingUnsupportedError extends Error {
|
|
1101
|
+
override readonly name = 'KubernetesEgressNarrowingUnsupportedError'
|
|
1102
|
+
|
|
1103
|
+
constructor(
|
|
1104
|
+
readonly policyKind: KubernetesEgressPolicy['kind'],
|
|
1105
|
+
readonly engine: KubernetesEgressEngine,
|
|
1106
|
+
) {
|
|
1107
|
+
super(
|
|
1108
|
+
`kubernetes: config.egress.ciliumNarrowing is set, but config.egress.policy is '${policyKind}' under engine ${JSON.stringify(engine)}. Port, DNS-name and TLS-server-name narrowing only apply to a 'static' or 'resolver' hostname allowlist under engine: 'cilium' — set engine: 'cilium' with one of those policy kinds, or remove ciliumNarrowing. Refusing rather than silently ignoring an option that would never be applied.`,
|
|
1109
|
+
)
|
|
1110
|
+
}
|
|
1111
|
+
}
|
|
205
1112
|
|
|
206
1113
|
/**
|
|
207
|
-
*
|
|
208
|
-
*
|
|
209
|
-
*
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
1114
|
+
* The grammar a host refusal normally closes with: what an `allowedHosts`
|
|
1115
|
+
* entry is. Stated only where the translation actually implements it — see
|
|
1116
|
+
* `stateHostnameGrammar` on {@link KubernetesNetworkPolicyHostError}.
|
|
1117
|
+
*/
|
|
1118
|
+
const HOSTNAME_GRAMMAR_SENTENCE =
|
|
1119
|
+
"Entries are hostnames: 'api.example.com' for one host, '.example.com' for that domain and its subdomains."
|
|
1120
|
+
|
|
1121
|
+
/**
|
|
1122
|
+
* Raised for an `allowedHosts` entry this backend will not translate.
|
|
214
1123
|
*
|
|
215
|
-
*
|
|
1124
|
+
* Named for the ENTRY rather than for a config field, because entries arrive
|
|
1125
|
+
* from two directions: a host's `Sandbox.setNetworkPolicy` call, and the
|
|
1126
|
+
* `allowedHosts` of a `static`/`resolver` `config.egress.policy`.
|
|
1127
|
+
* `SandboxNetworkPolicy` names HOSTS: `api.example.com` for one host,
|
|
1128
|
+
* `.example.com` for a domain and its subdomains. A glob, a URL, a port
|
|
1129
|
+
* suffix or an address is refused rather than emitted, because a
|
|
1130
|
+
* `CiliumNetworkPolicy` carrying one is an object the API server rejects on
|
|
1131
|
+
* apply — or, worse, accepts as a name that resolves to nothing, which reads
|
|
1132
|
+
* from the outside exactly like a policy that is working.
|
|
1133
|
+
*
|
|
1134
|
+
* `stateHostnameGrammar` is the one part of the message a caller can turn
|
|
1135
|
+
* off, because NOT every refusal can make that claim. It is false for the
|
|
1136
|
+
* config-level refusal of a leading-dot entry under `tlsServerNames`, whose
|
|
1137
|
+
* body has just said that the entry reaches the object as a literal
|
|
1138
|
+
* `matchName` admitting nothing with the option on or off: closing that
|
|
1139
|
+
* message with "'.example.com' for that domain and its subdomains" would
|
|
1140
|
+
* re-assert, as implemented, the very grammar whose absence is why the entry
|
|
1141
|
+
* is refused. Every other refusal keeps the sentence, the expanding branch of
|
|
1142
|
+
* the same one included.
|
|
1143
|
+
*/
|
|
1144
|
+
export class KubernetesNetworkPolicyHostError extends Error {
|
|
1145
|
+
override readonly name = 'KubernetesNetworkPolicyHostError'
|
|
1146
|
+
|
|
1147
|
+
constructor(
|
|
1148
|
+
readonly host: string,
|
|
1149
|
+
reason: string,
|
|
1150
|
+
options: { readonly stateHostnameGrammar?: boolean } = {},
|
|
1151
|
+
) {
|
|
1152
|
+
const grammar = options.stateHostnameGrammar === false ? '' : ` ${HOSTNAME_GRAMMAR_SENTENCE}`
|
|
1153
|
+
super(
|
|
1154
|
+
`kubernetes: the allowedHosts entry ${JSON.stringify(host)} is refused: it ${reason}.${grammar} Nothing was written.`,
|
|
1155
|
+
)
|
|
1156
|
+
}
|
|
1157
|
+
}
|
|
1158
|
+
|
|
1159
|
+
/**
|
|
1160
|
+
* Every entry in `narrowing.ports`, `narrowing.hostPorts` and
|
|
1161
|
+
* `narrowing.tlsPorts` is a port the API server will actually accept, none of
|
|
1162
|
+
* those three is an explicitly empty array, and every DNS suffix
|
|
1163
|
+
* `narrowing.dnsNames` names is a non-empty string — checked here,
|
|
1164
|
+
* synchronously, for the same reason {@link assertEgressPolicyIsEnforceable}
|
|
1165
|
+
* checks `exceptCidrs`: a value the API server rejects on apply would leave a
|
|
1166
|
+
* policy that never verifies.
|
|
1167
|
+
*/
|
|
1168
|
+
/**
|
|
1169
|
+
* Refuse a narrowing the API server would reject on apply, naming the field
|
|
1170
|
+
* that has to change.
|
|
1171
|
+
*
|
|
1172
|
+
* `fieldPath` is how the refusal spells the option's location, because the
|
|
1173
|
+
* same shape is configurable in two places now: `ciliumNarrowing` narrows the
|
|
1174
|
+
* CONFIG-level translated policy, and `perSandbox.narrowing` narrows the
|
|
1175
|
+
* per-sandbox policies a host writes through `setNetworkPolicy`. One
|
|
1176
|
+
* validator, because they are one shape and two would disagree the first time
|
|
1177
|
+
* either grew a field.
|
|
1178
|
+
*/
|
|
1179
|
+
export function assertCiliumNarrowingIsUsable(
|
|
1180
|
+
narrowing: KubernetesCiliumEgressNarrowing,
|
|
1181
|
+
fieldPath = 'ciliumNarrowing',
|
|
1182
|
+
): void {
|
|
1183
|
+
const assertValidPort = (port: unknown, field: string): void => {
|
|
1184
|
+
if (typeof port !== 'number' || !Number.isInteger(port) || port < 1 || port > 65535) {
|
|
1185
|
+
throw new KubernetesEgressPolicyConfigError(
|
|
1186
|
+
field,
|
|
1187
|
+
`${JSON.stringify(port)} is not a TCP port a NetworkPolicy/CiliumNetworkPolicy can carry (an integer 1-65535)`,
|
|
1188
|
+
)
|
|
1189
|
+
}
|
|
1190
|
+
}
|
|
1191
|
+
// `undefined` means "no restriction on this list" and is fine; `[]` is
|
|
1192
|
+
// neither that nor a usable restriction — `narrowedHostFqdnRule` would
|
|
1193
|
+
// emit `toPorts: [{ ports: [] }]` for it, a shape the API server rejects
|
|
1194
|
+
// on apply. Refuse it here rather than let a typo (`ports: []` where
|
|
1195
|
+
// `ports` was meant to be omitted) reach the cluster as a policy that
|
|
1196
|
+
// never verifies.
|
|
1197
|
+
const assertNotEmptyPortList = (ports: readonly number[] | undefined, field: string): void => {
|
|
1198
|
+
if (ports !== undefined && ports.length === 0) {
|
|
1199
|
+
throw new KubernetesEgressPolicyConfigError(
|
|
1200
|
+
field,
|
|
1201
|
+
'must not be an empty array — omit the field entirely for no port restriction, or list at least one port',
|
|
1202
|
+
)
|
|
1203
|
+
}
|
|
1204
|
+
}
|
|
1205
|
+
assertNotEmptyPortList(narrowing.ports, `${fieldPath}.ports`)
|
|
1206
|
+
for (const port of narrowing.ports ?? []) assertValidPort(port, `${fieldPath}.ports`)
|
|
1207
|
+
for (const [host, ports] of Object.entries(narrowing.hostPorts ?? {})) {
|
|
1208
|
+
assertNotEmptyPortList(ports, `${fieldPath}.hostPorts[${JSON.stringify(host)}]`)
|
|
1209
|
+
for (const port of ports) {
|
|
1210
|
+
assertValidPort(port, `${fieldPath}.hostPorts[${JSON.stringify(host)}]`)
|
|
1211
|
+
}
|
|
1212
|
+
}
|
|
1213
|
+
assertNotEmptyPortList(narrowing.tlsPorts, `${fieldPath}.tlsPorts`)
|
|
1214
|
+
for (const port of narrowing.tlsPorts ?? []) assertValidPort(port, `${fieldPath}.tlsPorts`)
|
|
1215
|
+
if (typeof narrowing.dnsNames === 'object') {
|
|
1216
|
+
const { clusterDomain, searchSuffixes } = narrowing.dnsNames
|
|
1217
|
+
if (clusterDomain !== undefined && clusterDomain.trim() === '') {
|
|
1218
|
+
throw new KubernetesEgressPolicyConfigError(
|
|
1219
|
+
`${fieldPath}.dnsNames.clusterDomain`,
|
|
1220
|
+
'must not be an empty string',
|
|
1221
|
+
)
|
|
1222
|
+
}
|
|
1223
|
+
for (const suffix of searchSuffixes ?? []) {
|
|
1224
|
+
if (typeof suffix !== 'string' || suffix.trim() === '') {
|
|
1225
|
+
throw new KubernetesEgressPolicyConfigError(
|
|
1226
|
+
`${fieldPath}.dnsNames.searchSuffixes`,
|
|
1227
|
+
`${JSON.stringify(suffix)} is not a usable DNS suffix`,
|
|
1228
|
+
)
|
|
1229
|
+
}
|
|
1230
|
+
}
|
|
1231
|
+
}
|
|
1232
|
+
}
|
|
1233
|
+
|
|
1234
|
+
/** Does `narrowing` actually ask for anything, or is every field off/absent. */
|
|
1235
|
+
function ciliumNarrowingIsActive(
|
|
1236
|
+
narrowing: KubernetesCiliumEgressNarrowing | undefined,
|
|
1237
|
+
): narrowing is KubernetesCiliumEgressNarrowing {
|
|
1238
|
+
if (narrowing === undefined) return false
|
|
1239
|
+
if (narrowing.ports !== undefined) return true
|
|
1240
|
+
if (narrowing.hostPorts !== undefined && Object.keys(narrowing.hostPorts).length > 0) return true
|
|
1241
|
+
if (dnsNarrowingIsActive(narrowing.dnsNames)) return true
|
|
1242
|
+
if (narrowing.tlsServerNames === true) return true
|
|
1243
|
+
return false
|
|
1244
|
+
}
|
|
1245
|
+
|
|
1246
|
+
function dnsNarrowingIsActive(dnsNames: KubernetesCiliumEgressNarrowing['dnsNames']): boolean {
|
|
1247
|
+
return activeDnsNarrowing(dnsNames) !== undefined
|
|
1248
|
+
}
|
|
1249
|
+
|
|
1250
|
+
/** `dnsNames` reduced to the config it names, or `undefined` when it is off. `true` reduces to every default. */
|
|
1251
|
+
function activeDnsNarrowing(
|
|
1252
|
+
dnsNames: KubernetesCiliumEgressNarrowing['dnsNames'],
|
|
1253
|
+
): KubernetesCiliumDnsNarrowing | undefined {
|
|
1254
|
+
if (dnsNames === true) return {}
|
|
1255
|
+
if (typeof dnsNames === 'object' && dnsNames !== null) return dnsNames
|
|
1256
|
+
return undefined
|
|
1257
|
+
}
|
|
1258
|
+
|
|
1259
|
+
/**
|
|
1260
|
+
* Synchronous, no-I/O precondition: can `policy.kind` be enforced under
|
|
1261
|
+
* `engine` at all, and — if `narrowing` is set — does it apply to this
|
|
1262
|
+
* policy/engine combination. Deliberately decided from the KIND alone — a
|
|
1263
|
+
* `resolver` policy's `resolve()` is never invoked here, both because calling
|
|
1264
|
+
* it just to prove a refusal would be wasted work (and possibly a side effect
|
|
1265
|
+
* the host did not expect yet) and because this has to stay callable
|
|
1266
|
+
* synchronously from `buildKubernetesBackend`, which contacts nothing.
|
|
1267
|
+
*/
|
|
1268
|
+
export function assertEgressPolicyIsEnforceable(
|
|
1269
|
+
policy: KubernetesEgressPolicy,
|
|
1270
|
+
engine: KubernetesEgressEngine,
|
|
1271
|
+
narrowing?: KubernetesCiliumEgressNarrowing,
|
|
1272
|
+
): void {
|
|
1273
|
+
if ((policy.kind === 'static' || policy.kind === 'resolver') && engine !== 'cilium') {
|
|
1274
|
+
throw new KubernetesUnenforceableEgressPolicyError(policy.kind)
|
|
1275
|
+
}
|
|
1276
|
+
if (policy.kind === 'public-internet') {
|
|
1277
|
+
for (const cidr of policy.exceptCidrs ?? []) {
|
|
1278
|
+
if (parseCidr(cidr) === undefined) {
|
|
1279
|
+
throw new KubernetesEgressPolicyConfigError(
|
|
1280
|
+
'policy.exceptCidrs',
|
|
1281
|
+
`${JSON.stringify(cidr)} is not an IPv4 or IPv6 CIDR this backend can read. A NetworkPolicy ipBlock.except entry has to be a CIDR inside the block it carves out of, and an entry the API server rejects on apply would leave a policy that never verifies.`,
|
|
1282
|
+
)
|
|
1283
|
+
}
|
|
1284
|
+
}
|
|
1285
|
+
}
|
|
1286
|
+
if (ciliumNarrowingIsActive(narrowing)) {
|
|
1287
|
+
if (engine !== 'cilium' || (policy.kind !== 'static' && policy.kind !== 'resolver')) {
|
|
1288
|
+
throw new KubernetesEgressNarrowingUnsupportedError(policy.kind, engine)
|
|
1289
|
+
}
|
|
1290
|
+
assertCiliumNarrowingIsUsable(narrowing)
|
|
1291
|
+
}
|
|
1292
|
+
}
|
|
1293
|
+
|
|
1294
|
+
/**
|
|
1295
|
+
* The cluster DNS egress rule every translated `NetworkPolicy` carries,
|
|
1296
|
+
* including on `deny-all`. See the module doc's "Why a NetworkPolicy always
|
|
1297
|
+
* allows the cluster's own DNS" section.
|
|
1298
|
+
*/
|
|
1299
|
+
const CLUSTER_DNS_EGRESS_RULE = {
|
|
1300
|
+
to: [
|
|
1301
|
+
{
|
|
1302
|
+
namespaceSelector: {
|
|
1303
|
+
matchLabels: { 'kubernetes.io/metadata.name': 'kube-system' },
|
|
1304
|
+
},
|
|
1305
|
+
},
|
|
1306
|
+
],
|
|
1307
|
+
ports: [
|
|
1308
|
+
{ protocol: 'UDP', port: 53 },
|
|
1309
|
+
{ protocol: 'TCP', port: 53 },
|
|
1310
|
+
],
|
|
1311
|
+
} as const
|
|
1312
|
+
|
|
1313
|
+
/**
|
|
1314
|
+
* DNS to the cluster resolver's own pods, which is narrower than
|
|
1315
|
+
* {@link CLUSTER_DNS_EGRESS_RULE}'s whole-namespace rule and is what
|
|
1316
|
+
* `'public-internet'` emits: that kind exists to name a destination set
|
|
1317
|
+
* precisely, so it names the resolver precisely too. `kube-system` +
|
|
1318
|
+
* `k8s-app: kube-dns` is the pairing CoreDNS ships under on every cluster
|
|
1319
|
+
* this was checked against, and the same pairing the repo's own
|
|
1320
|
+
* `k8s/manifests/networkpolicy.yaml` baseline already uses.
|
|
1321
|
+
*
|
|
1322
|
+
* `'deny-all'` keeps {@link CLUSTER_DNS_EGRESS_RULE} instead. It is not
|
|
1323
|
+
* changed to this one: its emitted manifest is pinned byte-for-byte, because
|
|
1324
|
+
* verification of the named object is an exact match and any change to the
|
|
1325
|
+
* translation fails every `create()` on every deployment that already
|
|
1326
|
+
* applied a policy.
|
|
1327
|
+
*/
|
|
1328
|
+
const KUBE_DNS_EGRESS_RULE = {
|
|
1329
|
+
to: [
|
|
1330
|
+
{
|
|
1331
|
+
namespaceSelector: {
|
|
1332
|
+
matchLabels: { 'kubernetes.io/metadata.name': 'kube-system' },
|
|
1333
|
+
},
|
|
1334
|
+
podSelector: { matchLabels: { 'k8s-app': 'kube-dns' } },
|
|
1335
|
+
},
|
|
1336
|
+
],
|
|
1337
|
+
ports: [
|
|
1338
|
+
{ protocol: 'UDP', port: 53 },
|
|
1339
|
+
{ protocol: 'TCP', port: 53 },
|
|
1340
|
+
],
|
|
1341
|
+
} as const
|
|
1342
|
+
|
|
1343
|
+
/**
|
|
1344
|
+
* What `'public-internet'` carves out of `0.0.0.0/0`, and why each entry is
|
|
1345
|
+
* here. Order is part of the emitted manifest and therefore part of what
|
|
1346
|
+
* verification compares, so it is fixed rather than sorted at build time.
|
|
1347
|
+
*
|
|
1348
|
+
* - `10.0.0.0/8`, `172.16.0.0/12`, `192.168.0.0/16` — RFC 1918. The cluster
|
|
1349
|
+
* network, the node network and every other sandbox pod live in one of
|
|
1350
|
+
* them on every deployment this was checked against.
|
|
1351
|
+
* - `100.64.0.0/10` — RFC 6598 carrier-grade NAT, which several managed
|
|
1352
|
+
* Kubernetes offerings hand to pods or nodes. Missing from the policy the
|
|
1353
|
+
* agent-sandbox controller writes when a template declares no
|
|
1354
|
+
* `networkPolicy`, which is exactly why this check compares except lists
|
|
1355
|
+
* rather than trusting that a policy "looks restrictive".
|
|
1356
|
+
* - `169.254.0.0/16` — link-local, and with it `169.254.169.254`, the
|
|
1357
|
+
* instance-metadata address whose credentials are the reason a sandbox
|
|
1358
|
+
* reaching "only the internet" still must not reach sideways.
|
|
1359
|
+
* - `127.0.0.0/8` — loopback. Not routable off the pod, but a policy that
|
|
1360
|
+
* says "the public internet" should not say it admits loopback either.
|
|
1361
|
+
* - `168.63.129.16/32` — one cloud's platform endpoint, a single address
|
|
1362
|
+
* outside every range above that answers DNS and instance services.
|
|
1363
|
+
*/
|
|
1364
|
+
const PUBLIC_INTERNET_EXCLUDED_IPV4_CIDRS = [
|
|
1365
|
+
'10.0.0.0/8',
|
|
1366
|
+
'172.16.0.0/12',
|
|
1367
|
+
'192.168.0.0/16',
|
|
1368
|
+
'100.64.0.0/10',
|
|
1369
|
+
'169.254.0.0/16',
|
|
1370
|
+
'127.0.0.0/8',
|
|
1371
|
+
'168.63.129.16/32',
|
|
1372
|
+
] as const
|
|
1373
|
+
|
|
1374
|
+
/**
|
|
1375
|
+
* The IPv6 equivalents: unique-local (`fc00::/7`), link-local
|
|
1376
|
+
* (`fe80::/10`, which carries the v6 spelling of instance metadata) and
|
|
1377
|
+
* loopback (`::1/128`). A cluster with no IPv6 at all is unaffected by the
|
|
1378
|
+
* rule's presence — it names destinations nothing routes to.
|
|
1379
|
+
*/
|
|
1380
|
+
const PUBLIC_INTERNET_EXCLUDED_IPV6_CIDRS = ['fc00::/7', 'fe80::/10', '::1/128'] as const
|
|
1381
|
+
|
|
1382
|
+
/**
|
|
1383
|
+
* The one rule `'public-internet'` adds beside DNS: everything, minus the
|
|
1384
|
+
* lists above, minus whatever the deployment added. No `ports`, because the
|
|
1385
|
+
* kind is a statement about DESTINATIONS — a deployment that also wants to
|
|
1386
|
+
* bound ports states that in its own applied policy, which this check then
|
|
1387
|
+
* accepts as narrower.
|
|
1388
|
+
*
|
|
1389
|
+
* A deployment's own `exceptCidrs` are routed to the block of their own
|
|
1390
|
+
* address family: a `NetworkPolicy` requires every `except` entry to sit
|
|
1391
|
+
* inside the `cidr` it carves out of, so an IPv6 entry under the IPv4 block
|
|
1392
|
+
* is rejected on apply.
|
|
1393
|
+
*/
|
|
1394
|
+
function publicInternetEgressRule(
|
|
1395
|
+
exceptCidrs: readonly string[] | undefined,
|
|
1396
|
+
): Readonly<Record<string, unknown>> {
|
|
1397
|
+
const extraV4: string[] = []
|
|
1398
|
+
const extraV6: string[] = []
|
|
1399
|
+
for (const cidr of exceptCidrs ?? []) {
|
|
1400
|
+
const parsed = parseCidr(cidr)
|
|
1401
|
+
// Unparseable entries are refused by `assertEgressPolicyIsEnforceable`
|
|
1402
|
+
// at construction; reaching here with one would mean this function was
|
|
1403
|
+
// called around it, so it is dropped rather than emitted.
|
|
1404
|
+
if (parsed === undefined) continue
|
|
1405
|
+
;(parsed.version === 4 ? extraV4 : extraV6).push(cidr)
|
|
1406
|
+
}
|
|
1407
|
+
return {
|
|
1408
|
+
to: [
|
|
1409
|
+
{
|
|
1410
|
+
ipBlock: {
|
|
1411
|
+
cidr: '0.0.0.0/0',
|
|
1412
|
+
except: [...PUBLIC_INTERNET_EXCLUDED_IPV4_CIDRS, ...extraV4],
|
|
1413
|
+
},
|
|
1414
|
+
},
|
|
1415
|
+
{
|
|
1416
|
+
ipBlock: {
|
|
1417
|
+
cidr: '::/0',
|
|
1418
|
+
except: [...PUBLIC_INTERNET_EXCLUDED_IPV6_CIDRS, ...extraV6],
|
|
1419
|
+
},
|
|
1420
|
+
},
|
|
1421
|
+
],
|
|
1422
|
+
}
|
|
1423
|
+
}
|
|
1424
|
+
|
|
1425
|
+
/**
|
|
1426
|
+
* Cilium requires DNS lookups to be explicitly allowed AND made visible to
|
|
1427
|
+
* the agent before `toFQDNs` enforcement can match anything a name resolves
|
|
1428
|
+
* to: without a preceding rule granting DNS and turning on visibility, the
|
|
1429
|
+
* sandbox's own lookups for the allowed hosts are invisible to Cilium's
|
|
1430
|
+
* FQDN-to-IP mapping and `toFQDNs` matches nothing, which would make a
|
|
1431
|
+
* `static`/`resolver` policy fail closed for every host rather than only the
|
|
1432
|
+
* disallowed ones.
|
|
1433
|
+
*
|
|
1434
|
+
* Shape verified against Cilium's own worked example — the `toFQDNs`
|
|
216
1435
|
* `CiliumNetworkPolicy` at https://docs.cilium.io/en/stable/security/dns/
|
|
217
1436
|
* (the "DNS Based" policy walkthrough) allows egress to `kube-dns` on port 53
|
|
218
1437
|
* with `rules.dns: [{ matchPattern: "*" }]` alongside its own `toFQDNs` rule
|
|
@@ -240,10 +1459,12 @@ const CILIUM_DNS_VISIBILITY_RULE = {
|
|
|
240
1459
|
|
|
241
1460
|
function buildCoreNetworkPolicy(
|
|
242
1461
|
target: EgressPolicyTarget,
|
|
1462
|
+
policyKind: KubernetesEgressPolicy['kind'],
|
|
243
1463
|
egress: readonly Readonly<Record<string, unknown>>[],
|
|
244
1464
|
): KubernetesTranslatedEgressPolicy {
|
|
245
1465
|
return {
|
|
246
1466
|
kind: 'NetworkPolicy',
|
|
1467
|
+
policyKind,
|
|
247
1468
|
namespace: target.namespace,
|
|
248
1469
|
name: target.name,
|
|
249
1470
|
manifest: {
|
|
@@ -251,7 +1472,9 @@ function buildCoreNetworkPolicy(
|
|
|
251
1472
|
kind: 'NetworkPolicy',
|
|
252
1473
|
metadata: { name: target.name, namespace: target.namespace },
|
|
253
1474
|
spec: {
|
|
254
|
-
podSelector: {
|
|
1475
|
+
podSelector: {
|
|
1476
|
+
matchLabels: egressPolicySelectorLabels(target),
|
|
1477
|
+
},
|
|
255
1478
|
policyTypes: ['Egress'],
|
|
256
1479
|
egress,
|
|
257
1480
|
},
|
|
@@ -259,29 +1482,395 @@ function buildCoreNetworkPolicy(
|
|
|
259
1482
|
}
|
|
260
1483
|
}
|
|
261
1484
|
|
|
262
|
-
|
|
263
|
-
|
|
1485
|
+
/** `[443]` — the TLS-port default for both `tlsServerNames` and `tlsPorts`. */
|
|
1486
|
+
const DEFAULT_TLS_PORTS = [443] as const
|
|
1487
|
+
|
|
1488
|
+
/** `{ port: '443', protocol: 'TCP' }` — Cilium's port shape, ports spelled as strings. */
|
|
1489
|
+
function ciliumPortEntry(port: number): Readonly<Record<string, unknown>> {
|
|
1490
|
+
return { port: String(port), protocol: 'TCP' }
|
|
1491
|
+
}
|
|
1492
|
+
|
|
1493
|
+
/**
|
|
1494
|
+
* Which translation a host refusal is being spelled for, as ONE value.
|
|
1495
|
+
*
|
|
1496
|
+
* The refusal has two halves that have to agree — the sentence, which follows
|
|
1497
|
+
* from whether the translation expands a leading-dot entry, and the field
|
|
1498
|
+
* path, which follows from which field the caller set — and both are facts
|
|
1499
|
+
* about the SAME thing: the translation the caller is on. Passed as two
|
|
1500
|
+
* arguments they are decided from two inputs, and an expanding caller was
|
|
1501
|
+
* told to repair `config.egress.ciliumNarrowing` with the remedy that belongs
|
|
1502
|
+
* to `config.egress.perSandbox.narrowing` — a message sending a reader to a
|
|
1503
|
+
* field the entry never came from. Neither half means anything without the
|
|
1504
|
+
* other, so they travel together.
|
|
1505
|
+
*/
|
|
1506
|
+
export interface HostsFitNarrowingContext {
|
|
1507
|
+
/**
|
|
1508
|
+
* The field the refusal sends the reader to, spelled as the caller set it
|
|
1509
|
+
* — the full path, as every other refusal in this module spells one.
|
|
1510
|
+
*/
|
|
1511
|
+
readonly fieldPath: string
|
|
1512
|
+
/**
|
|
1513
|
+
* Whether this translation turns a leading-dot entry into a `matchName`
|
|
1514
|
+
* plus a `matchPattern` — {@link ciliumFqdnEntries}. `true` for the
|
|
1515
|
+
* per-sandbox translation, `false` for the config-level one, whose emitted
|
|
1516
|
+
* bytes are pinned. Decides the refusal's sentence and its closing grammar
|
|
1517
|
+
* as well as the bytes, and all three are the same fact about one entry.
|
|
1518
|
+
*/
|
|
1519
|
+
readonly expandsDottedEntries: boolean
|
|
1520
|
+
}
|
|
1521
|
+
|
|
1522
|
+
/**
|
|
1523
|
+
* The refusal context of the PER-SANDBOX translation — the one that expands,
|
|
1524
|
+
* and so the one `expandDomains: true` builds.
|
|
1525
|
+
*
|
|
1526
|
+
* Shared deliberately: the per-sandbox writer refuses a host by this context
|
|
1527
|
+
* before it reads the fence, and {@link buildCiliumEgressManifest} refuses the
|
|
1528
|
+
* same host by the same context on the way through the translation, so the two
|
|
1529
|
+
* cannot name different fields for one entry however the earlier check is
|
|
1530
|
+
* reached or skipped.
|
|
1531
|
+
*/
|
|
1532
|
+
export const PER_SANDBOX_NARROWING_REFUSAL: HostsFitNarrowingContext = {
|
|
1533
|
+
fieldPath: 'config.egress.perSandbox.narrowing',
|
|
1534
|
+
expandsDottedEntries: true,
|
|
1535
|
+
}
|
|
1536
|
+
|
|
1537
|
+
/**
|
|
1538
|
+
* The refusal context of the config-level translation — `config.egress.policy`
|
|
1539
|
+
* and its `config.egress.ciliumNarrowing`, which expands nothing.
|
|
1540
|
+
*/
|
|
1541
|
+
export const CONFIG_LEVEL_NARROWING_REFUSAL: HostsFitNarrowingContext = {
|
|
1542
|
+
fieldPath: 'config.egress.ciliumNarrowing',
|
|
1543
|
+
expandsDottedEntries: false,
|
|
1544
|
+
}
|
|
1545
|
+
|
|
1546
|
+
/**
|
|
1547
|
+
* Refuse the one allowlist entry a narrowing option cannot express.
|
|
1548
|
+
*
|
|
1549
|
+
* `tlsServerNames` puts the entry on the rule as a TLS server name — one
|
|
1550
|
+
* exact SNI value a handshake presents — and a `.domain` entry is a set of
|
|
1551
|
+
* names no single SNI value means. Either way the emitted object would be
|
|
1552
|
+
* admitted by the shipped fence, read back deep-equal to what was sent, and
|
|
1553
|
+
* deny what the caller asked to allow — the failure every other refusal in
|
|
1554
|
+
* this module exists to prevent. Refused rather than translated another way,
|
|
1555
|
+
* because both other ways are guesses: dropping the subdomains silently
|
|
1556
|
+
* narrows what the caller asked for, and a wildcard SNI value is not
|
|
1557
|
+
* something the Cilium versions these manifests are written against are
|
|
1558
|
+
* known here to match — an SNI that matches nothing denies just as
|
|
1559
|
+
* completely, and more quietly.
|
|
1560
|
+
*
|
|
1561
|
+
* The entry is refused for a DIFFERENT reason on each of the two paths, and
|
|
1562
|
+
* the message says which one the caller is on, because the same sentence
|
|
1563
|
+
* cannot be true of both:
|
|
1564
|
+
*
|
|
1565
|
+
* - `expandDomains` — the per-sandbox writer's translation, which turns
|
|
1566
|
+
* `.domain` into `matchName: domain` PLUS `matchPattern: '*.domain'`. No
|
|
1567
|
+
* single SNI value means that pair: `domain` alone denies every subdomain
|
|
1568
|
+
* the pattern admits, and the entry as written is not a name any handshake
|
|
1569
|
+
* presents at all.
|
|
1570
|
+
* - unexpanded — the config-level translation, which expands nothing. There
|
|
1571
|
+
* the entry reaches the object as the literal `matchName: '.domain'`,
|
|
1572
|
+
* which no DNS answer carries, so it already admits nothing; `serverNames`
|
|
1573
|
+
* on top of it is a second, independent denial. Telling this caller to
|
|
1574
|
+
* "leave tlsServerNames off for a domain list" would name a repair that is
|
|
1575
|
+
* not one — the config-level `.domain` entry denies the domain and every
|
|
1576
|
+
* subdomain with or without the option (a pre-existing defect of this
|
|
1577
|
+
* translation, deferred to its own change) — so it is not offered here.
|
|
1578
|
+
*
|
|
1579
|
+
* `context` is WHICH translation the refusal is being spelled for, as one
|
|
1580
|
+
* value — see {@link HostsFitNarrowingContext}. A reader has to be sent to
|
|
1581
|
+
* the field they actually set, and the sentence they are sent by has to be
|
|
1582
|
+
* true of the translation they are on, and both follow from that one fact:
|
|
1583
|
+
* passed separately, an expanding caller gets the per-sandbox remedy attached
|
|
1584
|
+
* to the config-level field name, a message whose only repair is a field the
|
|
1585
|
+
* entry never came from. Called from BOTH — the translation itself, so every
|
|
1586
|
+
* caller is covered, and the per-sandbox writer, which refuses earlier still,
|
|
1587
|
+
* before it reads the fence.
|
|
1588
|
+
*/
|
|
1589
|
+
export function assertHostsFitNarrowing(
|
|
264
1590
|
allowedHosts: readonly string[],
|
|
1591
|
+
narrowing: KubernetesCiliumEgressNarrowing | undefined,
|
|
1592
|
+
context: HostsFitNarrowingContext,
|
|
1593
|
+
): void {
|
|
1594
|
+
if (narrowing?.tlsServerNames !== true) return
|
|
1595
|
+
for (const host of allowedHosts) {
|
|
1596
|
+
if (!host.startsWith('.')) continue
|
|
1597
|
+
throw new KubernetesNetworkPolicyHostError(
|
|
1598
|
+
host,
|
|
1599
|
+
context.expandsDottedEntries
|
|
1600
|
+
? `names a domain and its subdomains while ${context.fieldPath}.tlsServerNames is on; a TLS server name is one exact SNI value a handshake presents, and this entry becomes a name plus a '*.domain' pattern, which no single value means — list the exact hosts, or leave tlsServerNames off for a domain list`
|
|
1601
|
+
: `names a domain and its subdomains while ${context.fieldPath}.tlsServerNames is on; a TLS server name is one exact SNI value a handshake presents, and this translation does not expand a leading-dot entry — it reaches the object as the literal matchName ${JSON.stringify(host)}, which no DNS answer carries — so the entry admits nothing as written, and serverNames on top of it is a second, independent denial. List the exact hosts this policy should allow; leaving tlsServerNames off does not repair the entry here`,
|
|
1602
|
+
// The closing grammar rides the same input as the sentence, and for
|
|
1603
|
+
// the same reason: the unexpanded body has just said this entry
|
|
1604
|
+
// admits nothing as written, so a tail asserting that
|
|
1605
|
+
// `.example.com` means the domain and its subdomains would state,
|
|
1606
|
+
// as implemented, the grammar the refusal exists because this
|
|
1607
|
+
// translation does not apply to it.
|
|
1608
|
+
{ stateHostnameGrammar: context.expandsDottedEntries },
|
|
1609
|
+
)
|
|
1610
|
+
}
|
|
1611
|
+
}
|
|
1612
|
+
|
|
1613
|
+
/**
|
|
1614
|
+
* The port list a narrowed translation applies to `host`, before splitting
|
|
1615
|
+
* out the TLS subset — see {@link KubernetesCiliumEgressNarrowing.ports}.
|
|
1616
|
+
* `undefined` means no port restriction: the host's `toFQDNs` rule carries
|
|
1617
|
+
* no `toPorts` at all.
|
|
1618
|
+
*/
|
|
1619
|
+
function narrowedHostPorts(
|
|
1620
|
+
host: string,
|
|
1621
|
+
narrowing: KubernetesCiliumEgressNarrowing,
|
|
1622
|
+
): readonly number[] | undefined {
|
|
1623
|
+
const explicit = narrowing.hostPorts?.[host] ?? narrowing.ports
|
|
1624
|
+
if (explicit !== undefined) return explicit
|
|
1625
|
+
// `serverNames` needs a port to attach to — leaving this host with no
|
|
1626
|
+
// port at all would mean either no `serverNames` rule (silently dropping
|
|
1627
|
+
// the option) or one with no `toPorts`, which Cilium rejects on apply.
|
|
1628
|
+
if (narrowing.tlsServerNames === true) return narrowing.tlsPorts ?? DEFAULT_TLS_PORTS
|
|
1629
|
+
return undefined
|
|
1630
|
+
}
|
|
1631
|
+
|
|
1632
|
+
/**
|
|
1633
|
+
* One host's `toFQDNs` rule under narrowing: the allowlist entry's `toFQDNs`
|
|
1634
|
+
* entries, plus `toPorts` split into a `serverNames`-bearing entry for the
|
|
1635
|
+
* host's TLS ports and a plain entry for whatever is left, when
|
|
1636
|
+
* `tlsServerNames` is on.
|
|
1637
|
+
*
|
|
1638
|
+
* `serverNames` carries the allowlist ENTRY as written, which is the same
|
|
1639
|
+
* string as the `toFQDNs` entry only while nothing expanded it. A TLS server
|
|
1640
|
+
* name is one exact SNI value a handshake presents, and a `.example.com`
|
|
1641
|
+
* entry — whose `toFQDNs` half is `example.com` PLUS `*.example.com` — has no
|
|
1642
|
+
* single SNI value meaning that set: `example.com` would deny every subdomain
|
|
1643
|
+
* the pattern admits, and `.example.com` is not a name any handshake ever
|
|
1644
|
+
* presents. So {@link assertHostsFitNarrowing} refuses that one combination
|
|
1645
|
+
* before this rule is built, from `buildCiliumEgressManifest` for every
|
|
1646
|
+
* caller and from the per-sandbox writer before it reads the fence — which is
|
|
1647
|
+
* why this function can take `host` and the entry as one string.
|
|
1648
|
+
*/
|
|
1649
|
+
function narrowedHostFqdnRule(
|
|
1650
|
+
host: string,
|
|
1651
|
+
narrowing: KubernetesCiliumEgressNarrowing,
|
|
1652
|
+
fqdns: readonly Readonly<Record<string, string>>[] = [{ matchName: host }],
|
|
1653
|
+
): Readonly<Record<string, unknown>> {
|
|
1654
|
+
const ports = narrowedHostPorts(host, narrowing)
|
|
1655
|
+
if (ports === undefined) return { toFQDNs: fqdns }
|
|
1656
|
+
if (narrowing.tlsServerNames !== true) {
|
|
1657
|
+
return { toFQDNs: fqdns, toPorts: [{ ports: ports.map(ciliumPortEntry) }] }
|
|
1658
|
+
}
|
|
1659
|
+
const tlsPorts = narrowing.tlsPorts ?? DEFAULT_TLS_PORTS
|
|
1660
|
+
const tlsSubset = ports.filter((port) => tlsPorts.includes(port))
|
|
1661
|
+
const rest = ports.filter((port) => !tlsPorts.includes(port))
|
|
1662
|
+
const toPorts: Record<string, unknown>[] = []
|
|
1663
|
+
if (tlsSubset.length > 0) {
|
|
1664
|
+
toPorts.push({ ports: tlsSubset.map(ciliumPortEntry), serverNames: [host] })
|
|
1665
|
+
}
|
|
1666
|
+
if (rest.length > 0) toPorts.push({ ports: rest.map(ciliumPortEntry) })
|
|
1667
|
+
return {
|
|
1668
|
+
toFQDNs: fqdns,
|
|
1669
|
+
...(toPorts.length > 0 ? { toPorts } : {}),
|
|
1670
|
+
}
|
|
1671
|
+
}
|
|
1672
|
+
|
|
1673
|
+
/**
|
|
1674
|
+
* One allowlist entry, expanded to the `toFQDNs` entries it means.
|
|
1675
|
+
*
|
|
1676
|
+
* `SandboxNetworkPolicy.allowedHosts`'s own grammar, which the SDK states and
|
|
1677
|
+
* the docker backend implements: `api.example.com` is that host, and
|
|
1678
|
+
* `.example.com` is the domain AND its subdomains. Cilium's `matchName` is an
|
|
1679
|
+
* exact name and does not match across a `.`, so the domain form needs the
|
|
1680
|
+
* name plus a `matchPattern` — `*.example.com` alone would admit
|
|
1681
|
+
* `a.example.com` and not `example.com` itself.
|
|
1682
|
+
*
|
|
1683
|
+
* Used by the PER-SANDBOX policy only. The config-level translation
|
|
1684
|
+
* (`config.egress.policy`) deliberately does not expand anything: what it
|
|
1685
|
+
* emits for a given config is pinned byte-for-byte, because verification of
|
|
1686
|
+
* the named object is an exact match and a changed translation fails every
|
|
1687
|
+
* `create()` on every deployment that already applied a policy.
|
|
1688
|
+
*/
|
|
1689
|
+
export function ciliumFqdnEntries(entry: string): readonly Readonly<Record<string, string>>[] {
|
|
1690
|
+
if (!entry.startsWith('.')) return [{ matchName: entry }]
|
|
1691
|
+
const domain = entry.slice(1)
|
|
1692
|
+
return [{ matchName: domain }, { matchPattern: `*.${domain}` }]
|
|
1693
|
+
}
|
|
1694
|
+
|
|
1695
|
+
/**
|
|
1696
|
+
* The DNS-visibility rule narrowed to an exact `matchName` per allowed host
|
|
1697
|
+
* plus the host under every search suffix, replacing
|
|
1698
|
+
* {@link CILIUM_DNS_VISIBILITY_RULE}'s `matchPattern: '*'`. See
|
|
1699
|
+
* {@link KubernetesCiliumDnsNarrowing}.
|
|
1700
|
+
*/
|
|
1701
|
+
function narrowedDnsVisibilityRule(
|
|
1702
|
+
allowedHosts: readonly string[],
|
|
1703
|
+
fallbackNamespace: string,
|
|
1704
|
+
dnsNames: KubernetesCiliumDnsNarrowing,
|
|
1705
|
+
expandDomains = false,
|
|
1706
|
+
): Readonly<Record<string, unknown>> {
|
|
1707
|
+
const namespace = dnsNames.namespace ?? fallbackNamespace
|
|
1708
|
+
const clusterDomain = dnsNames.clusterDomain ?? 'cluster.local'
|
|
1709
|
+
const suffixes = [
|
|
1710
|
+
`${namespace}.svc.${clusterDomain}`,
|
|
1711
|
+
`svc.${clusterDomain}`,
|
|
1712
|
+
clusterDomain,
|
|
1713
|
+
...(dnsNames.searchSuffixes ?? []),
|
|
1714
|
+
]
|
|
1715
|
+
const matchNames: Record<string, string>[] = []
|
|
1716
|
+
for (const entry of allowedHosts) {
|
|
1717
|
+
// A `.domain` entry resolves through its subdomains as well, so the
|
|
1718
|
+
// DNS proxy has to be allowed to SEE those lookups or `toFQDNs` never
|
|
1719
|
+
// learns the addresses they resolve to. Off for the config-level
|
|
1720
|
+
// translation, whose emitted bytes are pinned: `expandDomains` is
|
|
1721
|
+
// false there and `host === entry`, so this loop is what it always
|
|
1722
|
+
// was.
|
|
1723
|
+
const wildcard = expandDomains && entry.startsWith('.')
|
|
1724
|
+
const host = wildcard ? entry.slice(1) : entry
|
|
1725
|
+
matchNames.push({ matchName: host })
|
|
1726
|
+
if (wildcard) matchNames.push({ matchPattern: `*.${host}` })
|
|
1727
|
+
for (const suffix of suffixes) {
|
|
1728
|
+
matchNames.push({ matchName: `${host}.${suffix}` })
|
|
1729
|
+
// The suffixed names are here because a guest resolving a name
|
|
1730
|
+
// with fewer dots than the cluster's `ndots` tries the search
|
|
1731
|
+
// suffixes FIRST, and a lookup the DNS proxy refuses is not an
|
|
1732
|
+
// NXDOMAIN the resolver walks past — it can fail the whole
|
|
1733
|
+
// resolution. An EXPANDED entry admits subdomains, so its
|
|
1734
|
+
// subdomains need the same treatment, or `a.example.com` fails on
|
|
1735
|
+
// its first search-suffix attempt under a rule that allows it.
|
|
1736
|
+
if (wildcard) matchNames.push({ matchPattern: `*.${host}.${suffix}` })
|
|
1737
|
+
}
|
|
1738
|
+
}
|
|
1739
|
+
return {
|
|
1740
|
+
toEndpoints: CILIUM_DNS_VISIBILITY_RULE.toEndpoints,
|
|
1741
|
+
toPorts: [
|
|
1742
|
+
{
|
|
1743
|
+
ports: [{ port: '53', protocol: 'ANY' }],
|
|
1744
|
+
rules: { dns: matchNames },
|
|
1745
|
+
},
|
|
1746
|
+
],
|
|
1747
|
+
}
|
|
1748
|
+
}
|
|
1749
|
+
|
|
1750
|
+
/**
|
|
1751
|
+
* Everything a `CiliumNetworkPolicy` for a hostname allowlist is built from.
|
|
1752
|
+
*
|
|
1753
|
+
* ONE builder, two callers: the config-level translation below, whose
|
|
1754
|
+
* selector is the template (and profile) label and whose emitted bytes are
|
|
1755
|
+
* pinned, and the per-sandbox policy in `per-sandbox-policy.ts`, whose
|
|
1756
|
+
* selector is one per-sandbox label and which additionally carries an
|
|
1757
|
+
* `ownerReferences` entry so the cluster garbage-collects it. A second
|
|
1758
|
+
* builder would be a second answer to what a namzu egress policy looks like,
|
|
1759
|
+
* and the read-back comparator would then be verifying one of them against
|
|
1760
|
+
* the other's shape.
|
|
1761
|
+
*/
|
|
1762
|
+
export interface CiliumEgressManifestOptions {
|
|
1763
|
+
readonly namespace: string
|
|
1764
|
+
readonly name: string
|
|
1765
|
+
/** What `spec.endpointSelector.matchLabels` carries, verbatim. */
|
|
1766
|
+
readonly selectorLabels: Readonly<Record<string, string>>
|
|
1767
|
+
readonly allowedHosts: readonly string[]
|
|
1768
|
+
/** The CONFIGURED kind this came from, for a refusal's wording. */
|
|
1769
|
+
readonly policyKind: KubernetesEgressPolicy['kind']
|
|
1770
|
+
readonly narrowing?: KubernetesCiliumEgressNarrowing
|
|
1771
|
+
/** `metadata.ownerReferences`. Absent ⇒ the metadata is what it always was. */
|
|
1772
|
+
readonly ownerReferences?: readonly KubernetesOwnerReference[]
|
|
1773
|
+
/**
|
|
1774
|
+
* Expand a leading-dot entry into `matchName` plus `matchPattern` — see
|
|
1775
|
+
* {@link ciliumFqdnEntries}. Off by default, because the config-level
|
|
1776
|
+
* translation's emitted bytes are pinned.
|
|
1777
|
+
*
|
|
1778
|
+
* This option IS the translation, not a formatting flag on it: it selects
|
|
1779
|
+
* the emitted bytes, the sentence a refusal carries and the field that
|
|
1780
|
+
* refusal names, all from one value — see
|
|
1781
|
+
* {@link HostsFitNarrowingContext}. There is no way to ask for one without
|
|
1782
|
+
* the others, and none should be added.
|
|
1783
|
+
*/
|
|
1784
|
+
readonly expandDomains?: boolean
|
|
1785
|
+
}
|
|
1786
|
+
|
|
1787
|
+
export function buildCiliumEgressManifest(
|
|
1788
|
+
options: CiliumEgressManifestOptions,
|
|
265
1789
|
): KubernetesTranslatedEgressPolicy {
|
|
1790
|
+
const { narrowing, allowedHosts, expandDomains = false } = options
|
|
1791
|
+
// Before anything is emitted. A leading-dot entry under `tlsServerNames`
|
|
1792
|
+
// would become `serverNames: ['.domain']` — not a name any handshake
|
|
1793
|
+
// presents — and `['domain']` would deny every subdomain the `toFQDNs`
|
|
1794
|
+
// half of the same rule admits; the object is admitted by the shipped
|
|
1795
|
+
// fence and reads back deep-equal to what was sent, so nothing downstream
|
|
1796
|
+
// would ever report it.
|
|
1797
|
+
//
|
|
1798
|
+
// What this call guarantees, per caller shape: the per-sandbox writer
|
|
1799
|
+
// refuses the same hosts earlier still, by name and by the same context
|
|
1800
|
+
// (`per-sandbox-policy.ts`), and this call covers them again if that check
|
|
1801
|
+
// is ever reached later or skipped; the config-level translation has no
|
|
1802
|
+
// earlier check of its own and relies on this one entirely; and a direct
|
|
1803
|
+
// call to this function is covered here too, whatever its `expandDomains`.
|
|
1804
|
+
//
|
|
1805
|
+
// ONE input decides all of it. `expandDomains` is not a formatting flag —
|
|
1806
|
+
// it is which translation this is, and so which field a refused caller
|
|
1807
|
+
// actually set, which sentence is true of the entry on that path, and
|
|
1808
|
+
// whether the translation applies the hostname grammar at all. Deriving
|
|
1809
|
+
// the field path from anything else is how an expanding caller came to be
|
|
1810
|
+
// sent to the config-level field for a repair that only exists under
|
|
1811
|
+
// `perSandbox.narrowing`.
|
|
1812
|
+
assertHostsFitNarrowing(
|
|
1813
|
+
allowedHosts,
|
|
1814
|
+
narrowing,
|
|
1815
|
+
expandDomains ? PER_SANDBOX_NARROWING_REFUSAL : CONFIG_LEVEL_NARROWING_REFUSAL,
|
|
1816
|
+
)
|
|
1817
|
+
// Unnarrowed is the exact shape every release before #490 emitted — kept
|
|
1818
|
+
// as its own branch, untouched, rather than folded into the narrowed one
|
|
1819
|
+
// with every option defaulted off, so the byte-identical guarantee does
|
|
1820
|
+
// not depend on the narrowed code path happening to reduce to it.
|
|
1821
|
+
const narrowed = ciliumNarrowingIsActive(narrowing)
|
|
1822
|
+
const activeDns = narrowed ? activeDnsNarrowing(narrowing.dnsNames) : undefined
|
|
1823
|
+
const dnsRule =
|
|
1824
|
+
activeDns !== undefined
|
|
1825
|
+
? narrowedDnsVisibilityRule(allowedHosts, options.namespace, activeDns, expandDomains)
|
|
1826
|
+
: CILIUM_DNS_VISIBILITY_RULE
|
|
1827
|
+
const entriesFor = (entry: string): readonly Readonly<Record<string, string>>[] =>
|
|
1828
|
+
expandDomains ? ciliumFqdnEntries(entry) : [{ matchName: entry }]
|
|
1829
|
+
const hostRules = narrowed
|
|
1830
|
+
? allowedHosts.map((host) => narrowedHostFqdnRule(host, narrowing, entriesFor(host)))
|
|
1831
|
+
: [{ toFQDNs: allowedHosts.flatMap(entriesFor) }]
|
|
1832
|
+
|
|
266
1833
|
return {
|
|
267
1834
|
kind: 'CiliumNetworkPolicy',
|
|
268
|
-
|
|
269
|
-
|
|
1835
|
+
policyKind: options.policyKind,
|
|
1836
|
+
namespace: options.namespace,
|
|
1837
|
+
name: options.name,
|
|
270
1838
|
manifest: {
|
|
271
1839
|
apiVersion: `${CILIUM_NETWORK_POLICY_API_GROUP}/${CILIUM_NETWORK_POLICY_API_VERSION}`,
|
|
272
1840
|
kind: 'CiliumNetworkPolicy',
|
|
273
|
-
metadata: {
|
|
1841
|
+
metadata: {
|
|
1842
|
+
name: options.name,
|
|
1843
|
+
namespace: options.namespace,
|
|
1844
|
+
...(options.ownerReferences !== undefined
|
|
1845
|
+
? { ownerReferences: options.ownerReferences }
|
|
1846
|
+
: {}),
|
|
1847
|
+
},
|
|
274
1848
|
spec: {
|
|
275
|
-
endpointSelector: {
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
],
|
|
1849
|
+
endpointSelector: {
|
|
1850
|
+
matchLabels: options.selectorLabels,
|
|
1851
|
+
},
|
|
1852
|
+
egress: [dnsRule, ...hostRules],
|
|
280
1853
|
},
|
|
281
1854
|
},
|
|
282
1855
|
}
|
|
283
1856
|
}
|
|
284
1857
|
|
|
1858
|
+
function buildCiliumNetworkPolicy(
|
|
1859
|
+
target: EgressPolicyTarget,
|
|
1860
|
+
policyKind: KubernetesEgressPolicy['kind'],
|
|
1861
|
+
allowedHosts: readonly string[],
|
|
1862
|
+
narrowing: KubernetesCiliumEgressNarrowing | undefined,
|
|
1863
|
+
): KubernetesTranslatedEgressPolicy {
|
|
1864
|
+
return buildCiliumEgressManifest({
|
|
1865
|
+
namespace: target.namespace,
|
|
1866
|
+
name: target.name,
|
|
1867
|
+
selectorLabels: egressPolicySelectorLabels(target),
|
|
1868
|
+
allowedHosts,
|
|
1869
|
+
policyKind,
|
|
1870
|
+
...(narrowing !== undefined ? { narrowing } : {}),
|
|
1871
|
+
})
|
|
1872
|
+
}
|
|
1873
|
+
|
|
285
1874
|
/**
|
|
286
1875
|
* The pure translation: an {@link EgressPolicy} plus the declared
|
|
287
1876
|
* {@link KubernetesEgressEngine} in, the concrete manifest this backend can
|
|
@@ -300,27 +1889,39 @@ function buildCiliumNetworkPolicy(
|
|
|
300
1889
|
* value nothing downstream re-applies to the cluster.
|
|
301
1890
|
*/
|
|
302
1891
|
export async function translateEgressPolicy(
|
|
303
|
-
policy:
|
|
1892
|
+
policy: KubernetesEgressPolicy,
|
|
304
1893
|
engine: KubernetesEgressEngine,
|
|
305
1894
|
target: EgressPolicyTarget,
|
|
1895
|
+
ciliumNarrowing?: KubernetesCiliumEgressNarrowing,
|
|
306
1896
|
): Promise<KubernetesTranslatedEgressPolicy> {
|
|
307
|
-
assertEgressPolicyIsEnforceable(policy, engine)
|
|
1897
|
+
assertEgressPolicyIsEnforceable(policy, engine, ciliumNarrowing)
|
|
308
1898
|
|
|
309
1899
|
switch (policy.kind) {
|
|
310
1900
|
case 'deny-all':
|
|
311
|
-
return buildCoreNetworkPolicy(target, [CLUSTER_DNS_EGRESS_RULE])
|
|
1901
|
+
return buildCoreNetworkPolicy(target, 'deny-all', [CLUSTER_DNS_EGRESS_RULE])
|
|
1902
|
+
case 'no-network':
|
|
1903
|
+
// `policyTypes: ['Egress']` with an EMPTY rule list is the API's own
|
|
1904
|
+
// spelling of "this pod sends nothing": the pod is in egress
|
|
1905
|
+
// default-deny and no rule lets anything back out. Not even DNS —
|
|
1906
|
+
// see the kind's own doc.
|
|
1907
|
+
return buildCoreNetworkPolicy(target, 'no-network', [])
|
|
1908
|
+
case 'public-internet':
|
|
1909
|
+
return buildCoreNetworkPolicy(target, 'public-internet', [
|
|
1910
|
+
KUBE_DNS_EGRESS_RULE,
|
|
1911
|
+
publicInternetEgressRule(policy.exceptCidrs),
|
|
1912
|
+
])
|
|
312
1913
|
case 'allow-all':
|
|
313
1914
|
// No `to`/`ports` on an egress rule matches every destination and
|
|
314
1915
|
// every port. `CLUSTER_DNS_EGRESS_RULE` is a strict subset of this,
|
|
315
1916
|
// so it is folded in rather than listed twice.
|
|
316
|
-
return buildCoreNetworkPolicy(target, [{}])
|
|
1917
|
+
return buildCoreNetworkPolicy(target, 'allow-all', [{}])
|
|
317
1918
|
case 'static':
|
|
318
1919
|
// `assertEgressPolicyIsEnforceable` already threw above unless
|
|
319
1920
|
// `engine === 'cilium'`, so reaching here means it did not.
|
|
320
|
-
return buildCiliumNetworkPolicy(target, policy.allowedHosts)
|
|
1921
|
+
return buildCiliumNetworkPolicy(target, 'static', policy.allowedHosts, ciliumNarrowing)
|
|
321
1922
|
case 'resolver': {
|
|
322
1923
|
const allowedHosts = await policy.resolve()
|
|
323
|
-
return buildCiliumNetworkPolicy(target, allowedHosts)
|
|
1924
|
+
return buildCiliumNetworkPolicy(target, 'resolver', allowedHosts, ciliumNarrowing)
|
|
324
1925
|
}
|
|
325
1926
|
default: {
|
|
326
1927
|
const exhaustive: never = policy
|
|
@@ -380,7 +1981,13 @@ export class KubernetesEgressPolicyMismatchError extends Error {
|
|
|
380
1981
|
* - `NetworkPolicy` additionally declares `policyTypes` including
|
|
381
1982
|
* `'Egress'` — a `NetworkPolicy` with an `egress` array but no `'Egress'`
|
|
382
1983
|
* in `policyTypes` enforces nothing on egress at all;
|
|
383
|
-
* - the `egress` rule array matches the translation exactly
|
|
1984
|
+
* - the `egress` rule array matches the translation exactly;
|
|
1985
|
+
* - and, ONLY when the translation carries `metadata.ownerReferences` (the
|
|
1986
|
+
* per-sandbox policies of `per-sandbox-policy.ts`, never the operator's
|
|
1987
|
+
* own object), that the live object still carries each of them — an owner
|
|
1988
|
+
* reference dropped between the write and the read is a policy the
|
|
1989
|
+
* cluster will never collect with the sandbox it belongs to, which is the
|
|
1990
|
+
* leak the reference exists to prevent, and it is invisible in `spec`.
|
|
384
1991
|
*
|
|
385
1992
|
* Never mutates and never creates — a 404/410 is refused, not repaired.
|
|
386
1993
|
*/
|
|
@@ -394,7 +2001,12 @@ export async function verifyEgressPolicyApplied(
|
|
|
394
2001
|
? ciliumNetworkPolicyPath(translated.namespace, translated.name)
|
|
395
2002
|
: networkPolicyPath(translated.namespace, translated.name)
|
|
396
2003
|
|
|
397
|
-
let resource:
|
|
2004
|
+
let resource:
|
|
2005
|
+
| {
|
|
2006
|
+
readonly spec?: Readonly<Record<string, unknown>>
|
|
2007
|
+
readonly metadata?: Readonly<Record<string, unknown>>
|
|
2008
|
+
}
|
|
2009
|
+
| undefined
|
|
398
2010
|
try {
|
|
399
2011
|
resource = await client.request('GET', path, undefined, signal)
|
|
400
2012
|
} catch (err) {
|
|
@@ -434,4 +2046,1403 @@ export async function verifyEgressPolicyApplied(
|
|
|
434
2046
|
`spec.egress is ${JSON.stringify(actualSpec.egress)}, expected ${JSON.stringify(expectedSpec.egress)}`,
|
|
435
2047
|
)
|
|
436
2048
|
}
|
|
2049
|
+
|
|
2050
|
+
// Last, and only for a translation that asked for one: an object written
|
|
2051
|
+
// with an owner reference and read back without it is a policy that
|
|
2052
|
+
// outlives its sandbox. The operator-applied objects this function's other
|
|
2053
|
+
// caller checks carry none, so they never reach this branch.
|
|
2054
|
+
const expectedOwners = (
|
|
2055
|
+
translated.manifest as { readonly metadata?: { readonly ownerReferences?: readonly unknown[] } }
|
|
2056
|
+
).metadata?.ownerReferences
|
|
2057
|
+
if (expectedOwners !== undefined) {
|
|
2058
|
+
const actualOwners = (resource?.metadata as { readonly ownerReferences?: readonly unknown[] })
|
|
2059
|
+
?.ownerReferences
|
|
2060
|
+
const held = Array.isArray(actualOwners) ? actualOwners : []
|
|
2061
|
+
for (const owner of expectedOwners) {
|
|
2062
|
+
if (held.some((entry) => isDeepStrictEqual(entry, owner))) continue
|
|
2063
|
+
throw new KubernetesEgressPolicyMismatchError(
|
|
2064
|
+
translated.kind,
|
|
2065
|
+
path,
|
|
2066
|
+
`metadata.ownerReferences is ${JSON.stringify(actualOwners)}, expected it to carry ${JSON.stringify(owner)} — without that entry the cluster never collects this policy with the object it belongs to, so it would outlive the sandbox it was written for`,
|
|
2067
|
+
)
|
|
2068
|
+
}
|
|
2069
|
+
}
|
|
2070
|
+
}
|
|
2071
|
+
|
|
2072
|
+
// ---------------------------------------------------------------------------
|
|
2073
|
+
// CIDR arithmetic
|
|
2074
|
+
// ---------------------------------------------------------------------------
|
|
2075
|
+
//
|
|
2076
|
+
// Enough of it to answer one question: is this policy's block inside the
|
|
2077
|
+
// block the translation allows. Written here rather than taken from a
|
|
2078
|
+
// dependency because `packages/sandbox` declares no runtime dependency and
|
|
2079
|
+
// this is forty lines of integer arithmetic.
|
|
2080
|
+
|
|
2081
|
+
/** A CIDR as a masked base address and a prefix length. */
|
|
2082
|
+
export interface ParsedCidr {
|
|
2083
|
+
readonly version: 4 | 6
|
|
2084
|
+
readonly base: bigint
|
|
2085
|
+
readonly bits: number
|
|
2086
|
+
}
|
|
2087
|
+
|
|
2088
|
+
function parseIpv4(text: string): bigint | undefined {
|
|
2089
|
+
const parts = text.split('.')
|
|
2090
|
+
if (parts.length !== 4) return undefined
|
|
2091
|
+
let value = 0n
|
|
2092
|
+
for (const part of parts) {
|
|
2093
|
+
if (!/^\d{1,3}$/.test(part)) return undefined
|
|
2094
|
+
const octet = Number(part)
|
|
2095
|
+
if (octet > 255) return undefined
|
|
2096
|
+
value = (value << 8n) | BigInt(octet)
|
|
2097
|
+
}
|
|
2098
|
+
return value
|
|
2099
|
+
}
|
|
2100
|
+
|
|
2101
|
+
function parseIpv6(text: string): bigint | undefined {
|
|
2102
|
+
// An embedded IPv4 tail (`::ffff:10.0.0.1`) is legal and is how a
|
|
2103
|
+
// dual-stack cluster spells a v4 address in a v6 field.
|
|
2104
|
+
let head = text
|
|
2105
|
+
let tail = 0n
|
|
2106
|
+
let tailGroups = 0
|
|
2107
|
+
const lastColon = text.lastIndexOf(':')
|
|
2108
|
+
if (lastColon >= 0 && text.slice(lastColon + 1).includes('.')) {
|
|
2109
|
+
const embedded = parseIpv4(text.slice(lastColon + 1))
|
|
2110
|
+
if (embedded === undefined) return undefined
|
|
2111
|
+
head = text.slice(0, lastColon + 1)
|
|
2112
|
+
tail = embedded
|
|
2113
|
+
tailGroups = 2
|
|
2114
|
+
}
|
|
2115
|
+
const halves = head.split('::')
|
|
2116
|
+
if (halves.length > 2) return undefined
|
|
2117
|
+
const readGroups = (part: string): bigint[] | undefined => {
|
|
2118
|
+
if (part === '') return []
|
|
2119
|
+
const groups: bigint[] = []
|
|
2120
|
+
for (const group of part.split(':')) {
|
|
2121
|
+
if (group === '') continue
|
|
2122
|
+
if (!/^[0-9a-fA-F]{1,4}$/.test(group)) return undefined
|
|
2123
|
+
groups.push(BigInt(Number.parseInt(group, 16)))
|
|
2124
|
+
}
|
|
2125
|
+
return groups
|
|
2126
|
+
}
|
|
2127
|
+
const left = readGroups(halves[0] ?? '')
|
|
2128
|
+
const right = readGroups(halves[1] ?? '')
|
|
2129
|
+
if (left === undefined || right === undefined) return undefined
|
|
2130
|
+
const present = left.length + right.length + tailGroups
|
|
2131
|
+
if (present > 8) return undefined
|
|
2132
|
+
if (halves.length === 1 && present !== 8) return undefined
|
|
2133
|
+
const zeros = 8 - present
|
|
2134
|
+
const groups = [...left, ...Array.from({ length: zeros }, () => 0n), ...right]
|
|
2135
|
+
let value = 0n
|
|
2136
|
+
for (const group of groups) value = (value << 16n) | group
|
|
2137
|
+
if (tailGroups === 2) value = (value << 32n) | tail
|
|
2138
|
+
return value
|
|
2139
|
+
}
|
|
2140
|
+
|
|
2141
|
+
/**
|
|
2142
|
+
* A CIDR, or `undefined` when it is not one this check can read. Host bits
|
|
2143
|
+
* are masked off rather than rejected: `10.0.0.1/8` and `10.0.0.0/8` name the
|
|
2144
|
+
* same block, and an operator who wrote the first meant the second.
|
|
2145
|
+
*/
|
|
2146
|
+
export function parseCidr(text: unknown): ParsedCidr | undefined {
|
|
2147
|
+
if (typeof text !== 'string') return undefined
|
|
2148
|
+
const slash = text.indexOf('/')
|
|
2149
|
+
if (slash < 0) return undefined
|
|
2150
|
+
const address = text.slice(0, slash)
|
|
2151
|
+
const prefix = text.slice(slash + 1)
|
|
2152
|
+
if (!/^\d{1,3}$/.test(prefix)) return undefined
|
|
2153
|
+
const bits = Number(prefix)
|
|
2154
|
+
if (address.includes(':')) {
|
|
2155
|
+
const value = parseIpv6(address)
|
|
2156
|
+
if (value === undefined || bits > 128) return undefined
|
|
2157
|
+
return { version: 6, base: maskAddress(value, bits, 128), bits }
|
|
2158
|
+
}
|
|
2159
|
+
const value = parseIpv4(address)
|
|
2160
|
+
if (value === undefined || bits > 32) return undefined
|
|
2161
|
+
return { version: 4, base: maskAddress(value, bits, 32), bits }
|
|
2162
|
+
}
|
|
2163
|
+
|
|
2164
|
+
function maskAddress(value: bigint, bits: number, width: number): bigint {
|
|
2165
|
+
if (bits === 0) return 0n
|
|
2166
|
+
const host = BigInt(width - bits)
|
|
2167
|
+
return (value >> host) << host
|
|
2168
|
+
}
|
|
2169
|
+
|
|
2170
|
+
/** Every address of `inner` is an address of `outer`. */
|
|
2171
|
+
function cidrContains(outer: ParsedCidr, inner: ParsedCidr): boolean {
|
|
2172
|
+
if (outer.version !== inner.version) return false
|
|
2173
|
+
if (outer.bits > inner.bits) return false
|
|
2174
|
+
const width = outer.version === 4 ? 32 : 128
|
|
2175
|
+
return maskAddress(inner.base, outer.bits, width) === outer.base
|
|
2176
|
+
}
|
|
2177
|
+
|
|
2178
|
+
/** They share at least one address — i.e. one contains the other. */
|
|
2179
|
+
function cidrsOverlap(a: ParsedCidr, b: ParsedCidr): boolean {
|
|
2180
|
+
return cidrContains(a, b) || cidrContains(b, a)
|
|
2181
|
+
}
|
|
2182
|
+
|
|
2183
|
+
// ---------------------------------------------------------------------------
|
|
2184
|
+
// What the configured translation permits
|
|
2185
|
+
// ---------------------------------------------------------------------------
|
|
2186
|
+
|
|
2187
|
+
/** One port, or a range of them, on one protocol. `start` absent is every port. */
|
|
2188
|
+
interface PortRange {
|
|
2189
|
+
/** `undefined` is EVERY protocol — how Cilium spells `ANY`. Core defaults to TCP. */
|
|
2190
|
+
readonly protocol?: string
|
|
2191
|
+
readonly start?: number
|
|
2192
|
+
readonly end?: number
|
|
2193
|
+
}
|
|
2194
|
+
|
|
2195
|
+
/** A port list, `'all'` for the absent/empty one, `'unreadable'` for a shape this check cannot read. */
|
|
2196
|
+
type PortSet = 'all' | readonly PortRange[] | 'unreadable'
|
|
2197
|
+
|
|
2198
|
+
/** A destination, as either side of the comparison names it. */
|
|
2199
|
+
type PolicyPeer =
|
|
2200
|
+
| { readonly kind: 'everything' }
|
|
2201
|
+
| {
|
|
2202
|
+
readonly kind: 'cidr'
|
|
2203
|
+
readonly cidr: ParsedCidr
|
|
2204
|
+
readonly except: readonly ParsedCidr[]
|
|
2205
|
+
readonly text: string
|
|
2206
|
+
}
|
|
2207
|
+
| {
|
|
2208
|
+
readonly kind: 'selector'
|
|
2209
|
+
readonly namespaceSelector?: unknown
|
|
2210
|
+
readonly podSelector?: unknown
|
|
2211
|
+
readonly text: string
|
|
2212
|
+
}
|
|
2213
|
+
|
|
2214
|
+
interface AllowedDestination {
|
|
2215
|
+
readonly peer: PolicyPeer
|
|
2216
|
+
readonly ports: 'all' | readonly PortRange[]
|
|
2217
|
+
}
|
|
2218
|
+
|
|
2219
|
+
/** One `toFQDNs` entry the translation allows, with the ports it allows it on. */
|
|
2220
|
+
interface AllowedFqdn {
|
|
2221
|
+
readonly host: string
|
|
2222
|
+
readonly ports: 'all' | readonly PortRange[]
|
|
2223
|
+
}
|
|
2224
|
+
|
|
2225
|
+
/**
|
|
2226
|
+
* The cluster resolver's identity as a `PolicyPeer` — shared by
|
|
2227
|
+
* {@link egressAllowance} (reading a Cilium DNS-visibility rule into its
|
|
2228
|
+
* core-shaped equivalent) and {@link reachesResolverAtDnsPort} (deciding
|
|
2229
|
+
* whether some OTHER policy's rule reaches it), so there is one definition of
|
|
2230
|
+
* "this peer is the cluster resolver" rather than two that could drift apart.
|
|
2231
|
+
*/
|
|
2232
|
+
const RESOLVER_PEER: PolicyPeer = {
|
|
2233
|
+
kind: 'selector',
|
|
2234
|
+
namespaceSelector: {
|
|
2235
|
+
matchLabels: { 'kubernetes.io/metadata.name': 'kube-system' },
|
|
2236
|
+
},
|
|
2237
|
+
podSelector: { matchLabels: { 'k8s-app': 'kube-dns' } },
|
|
2238
|
+
text: 'the cluster resolver',
|
|
2239
|
+
}
|
|
2240
|
+
|
|
2241
|
+
/**
|
|
2242
|
+
* Everything the configured translation lets out, in the one shape the union
|
|
2243
|
+
* check compares against.
|
|
2244
|
+
*
|
|
2245
|
+
* Built from the manifest {@link translateEgressPolicy} just produced for
|
|
2246
|
+
* every core kind, so there is no second statement anywhere of what
|
|
2247
|
+
* `deny-all` or `public-internet` permit — the emitted object IS the
|
|
2248
|
+
* statement, and a test asserts each translation is within its own
|
|
2249
|
+
* allowance. The Cilium kinds are the one exception and say so: a
|
|
2250
|
+
* `CiliumNetworkPolicy`'s DNS-visibility rule is read into the core-shaped
|
|
2251
|
+
* destination it is equivalent to (UDP/TCP 53 to the resolver's pods), so
|
|
2252
|
+
* that a core `NetworkPolicy` allowing exactly cluster DNS is not reported as
|
|
2253
|
+
* widening a hostname allowlist that already allows it.
|
|
2254
|
+
*/
|
|
2255
|
+
export interface EgressAllowance {
|
|
2256
|
+
readonly destinations: readonly AllowedDestination[]
|
|
2257
|
+
/**
|
|
2258
|
+
* `toFQDNs` names a `static`/`resolver` translation allows, each with the
|
|
2259
|
+
* ports it allows that name on (`'all'` for the unnarrowed shape, which
|
|
2260
|
+
* puts no `toPorts` on its `toFQDNs` rule at all). Empty for every core
|
|
2261
|
+
* kind. A name can appear more than once — one entry per `toFQDNs` rule
|
|
2262
|
+
* naming it — and it is allowed on a port if ANY entry covers that port,
|
|
2263
|
+
* the same "any matching destination" rule {@link destinationIsAllowed}
|
|
2264
|
+
* applies to CIDR and selector peers.
|
|
2265
|
+
*/
|
|
2266
|
+
readonly fqdns: readonly AllowedFqdn[]
|
|
2267
|
+
/**
|
|
2268
|
+
* The exact DNS names a `static`/`resolver` translation's kube-dns rule
|
|
2269
|
+
* restricts LOOKUPS to when `ciliumNarrowing.dnsNames` is set — the
|
|
2270
|
+
* `matchName` list {@link narrowedDnsVisibilityRule} builds. `'all'` for
|
|
2271
|
+
* every translation that does not narrow DNS: every core kind (which
|
|
2272
|
+
* cannot express an L7 DNS restriction at all) and an unnarrowed
|
|
2273
|
+
* `static`/`resolver` (`rules.dns: [{ matchPattern: '*' }]`).
|
|
2274
|
+
*
|
|
2275
|
+
* `destinations` alone cannot carry this: reachability to the resolver's
|
|
2276
|
+
* peer and port is the same whether or not DNS is narrowed, so a peer/port
|
|
2277
|
+
* check reports a plain kube-dns rule as `within` a narrowed translation
|
|
2278
|
+
* exactly as it would an unnarrowed one. {@link reachesResolverAtDnsPort}
|
|
2279
|
+
* is the check that actually reads this field, in both
|
|
2280
|
+
* {@link coreEgressRuleVerdict} and {@link ciliumEgressRuleVerdict}.
|
|
2281
|
+
*/
|
|
2282
|
+
readonly dnsNarrowedTo: 'all' | readonly string[]
|
|
2283
|
+
readonly permitsEverything: boolean
|
|
2284
|
+
readonly permitsNothing: boolean
|
|
2285
|
+
/** The configured kind, for the refusal message. */
|
|
2286
|
+
readonly policyKind: KubernetesEgressPolicy['kind']
|
|
2287
|
+
}
|
|
2288
|
+
|
|
2289
|
+
function readPortRanges(ports: unknown, defaultProtocol: string | undefined): PortSet {
|
|
2290
|
+
const entries = readList(ports)
|
|
2291
|
+
if (entries === 'unreadable') return 'unreadable'
|
|
2292
|
+
// Absent or empty `ports` on an egress rule means EVERY port.
|
|
2293
|
+
if (entries === undefined || entries.length === 0) return 'all'
|
|
2294
|
+
const ranges: PortRange[] = []
|
|
2295
|
+
for (const entry of entries) {
|
|
2296
|
+
if (!isRecord(entry)) return 'unreadable'
|
|
2297
|
+
const rawProtocol = entry.protocol
|
|
2298
|
+
if (rawProtocol !== undefined && typeof rawProtocol !== 'string') return 'unreadable'
|
|
2299
|
+
const protocol =
|
|
2300
|
+
rawProtocol === undefined
|
|
2301
|
+
? defaultProtocol
|
|
2302
|
+
: rawProtocol.toUpperCase() === 'ANY'
|
|
2303
|
+
? undefined
|
|
2304
|
+
: rawProtocol.toUpperCase()
|
|
2305
|
+
const endPort = entry.endPort
|
|
2306
|
+
if (endPort !== undefined && typeof endPort !== 'number') return 'unreadable'
|
|
2307
|
+
const raw = entry.port
|
|
2308
|
+
if (raw === undefined) {
|
|
2309
|
+
ranges.push(protocol === undefined ? {} : { protocol })
|
|
2310
|
+
continue
|
|
2311
|
+
}
|
|
2312
|
+
const port = typeof raw === 'number' ? raw : typeof raw === 'string' ? Number(raw) : Number.NaN
|
|
2313
|
+
// A NAMED container port. Resolving it needs the destination pod's own
|
|
2314
|
+
// container spec, which this check never has.
|
|
2315
|
+
if (!Number.isInteger(port)) return 'unreadable'
|
|
2316
|
+
ranges.push({
|
|
2317
|
+
...(protocol !== undefined ? { protocol } : {}),
|
|
2318
|
+
start: port,
|
|
2319
|
+
...(typeof endPort === 'number' ? { end: endPort } : {}),
|
|
2320
|
+
})
|
|
2321
|
+
}
|
|
2322
|
+
return ranges
|
|
2323
|
+
}
|
|
2324
|
+
|
|
2325
|
+
function readCorePeer(peer: unknown): PolicyPeer | undefined {
|
|
2326
|
+
if (!isRecord(peer)) return undefined
|
|
2327
|
+
const { ipBlock, podSelector, namespaceSelector } = peer
|
|
2328
|
+
if (ipBlock !== undefined) {
|
|
2329
|
+
if (!isRecord(ipBlock)) return undefined
|
|
2330
|
+
const cidr = parseCidr(ipBlock.cidr)
|
|
2331
|
+
if (cidr === undefined) return undefined
|
|
2332
|
+
const rawExcept = readList(ipBlock.except)
|
|
2333
|
+
if (rawExcept === 'unreadable') return undefined
|
|
2334
|
+
const except: ParsedCidr[] = []
|
|
2335
|
+
for (const entry of rawExcept ?? []) {
|
|
2336
|
+
const parsed = parseCidr(entry)
|
|
2337
|
+
if (parsed === undefined) return undefined
|
|
2338
|
+
except.push(parsed)
|
|
2339
|
+
}
|
|
2340
|
+
return { kind: 'cidr', cidr, except, text: JSON.stringify(ipBlock) }
|
|
2341
|
+
}
|
|
2342
|
+
if (podSelector === undefined && namespaceSelector === undefined) {
|
|
2343
|
+
// A peer naming none of the three constrains nothing. The API server
|
|
2344
|
+
// rejects it on admission, so an object carrying one did not come from
|
|
2345
|
+
// there and is not read as anything.
|
|
2346
|
+
return undefined
|
|
2347
|
+
}
|
|
2348
|
+
if (!selectorIsReadable(podSelector) || !selectorIsReadable(namespaceSelector)) return undefined
|
|
2349
|
+
return {
|
|
2350
|
+
kind: 'selector',
|
|
2351
|
+
...(namespaceSelector !== undefined ? { namespaceSelector } : {}),
|
|
2352
|
+
...(podSelector !== undefined ? { podSelector } : {}),
|
|
2353
|
+
text: JSON.stringify(peer),
|
|
2354
|
+
}
|
|
2355
|
+
}
|
|
2356
|
+
|
|
2357
|
+
/**
|
|
2358
|
+
* The allowance a translated policy expresses. See {@link EgressAllowance}.
|
|
2359
|
+
*/
|
|
2360
|
+
export function egressAllowance(translated: KubernetesTranslatedEgressPolicy): EgressAllowance {
|
|
2361
|
+
const spec = (translated.manifest as { readonly spec: Record<string, unknown> }).spec
|
|
2362
|
+
// This module built the manifest a line ago, so `spec.egress` is always a
|
|
2363
|
+
// list here; the narrowing exists so the allowance is derived from the
|
|
2364
|
+
// object itself rather than from a second statement of what each kind
|
|
2365
|
+
// permits, and an impossible shape yields an allowance that permits
|
|
2366
|
+
// nothing rather than one that permits anything.
|
|
2367
|
+
const emitted = readList(spec.egress)
|
|
2368
|
+
const rules = emitted === undefined || emitted === 'unreadable' ? [] : emitted
|
|
2369
|
+
if (translated.kind === 'CiliumNetworkPolicy') {
|
|
2370
|
+
const fqdns: AllowedFqdn[] = []
|
|
2371
|
+
// The DNS-visibility rule is the one entry among `rules` whose
|
|
2372
|
+
// `toEndpoints` names the cluster resolver — narrowed or not, it always
|
|
2373
|
+
// reuses `CILIUM_DNS_VISIBILITY_RULE.toEndpoints` verbatim (see
|
|
2374
|
+
// `narrowedDnsVisibilityRule` and `buildCiliumNetworkPolicy`), so a deep
|
|
2375
|
+
// equality check finds it regardless of which branch built this rule.
|
|
2376
|
+
const dnsVisibilityRule = rules.find(
|
|
2377
|
+
(rule): rule is Readonly<Record<string, unknown>> =>
|
|
2378
|
+
isRecord(rule) &&
|
|
2379
|
+
isDeepStrictEqual(rule.toEndpoints, CILIUM_DNS_VISIBILITY_RULE.toEndpoints),
|
|
2380
|
+
)
|
|
2381
|
+
for (const rule of rules) {
|
|
2382
|
+
if (!isRecord(rule)) continue
|
|
2383
|
+
const hostNames = readList(rule.toFQDNs)
|
|
2384
|
+
if (hostNames === undefined || hostNames === 'unreadable') continue
|
|
2385
|
+
// This module built `rule` a few lines above (see the comment at the
|
|
2386
|
+
// top of this function), so its `toPorts` is always in the shape
|
|
2387
|
+
// `readCiliumRulePorts` reads — 'unreadable' cannot happen for a
|
|
2388
|
+
// manifest this file emitted, and 'all' is the safe fallback if it
|
|
2389
|
+
// somehow did, since that is what an absent `toPorts` also means.
|
|
2390
|
+
const portsRead = readCiliumRulePorts(rule)
|
|
2391
|
+
const ports = portsRead.ok ? portsRead.ports : 'all'
|
|
2392
|
+
for (const entry of hostNames) {
|
|
2393
|
+
if (isRecord(entry) && typeof entry.matchName === 'string') {
|
|
2394
|
+
fqdns.push({ host: entry.matchName, ports })
|
|
2395
|
+
}
|
|
2396
|
+
}
|
|
2397
|
+
}
|
|
2398
|
+
// This module built `dnsVisibilityRule` above (when it exists at all —
|
|
2399
|
+
// every Cilium translation this function reads carries one), so
|
|
2400
|
+
// `readCiliumRuleDnsNames` failing to read it, or reading a wildcard,
|
|
2401
|
+
// cannot mean anything other than "not narrowed": 'all' is the safe
|
|
2402
|
+
// fallback, the same reasoning `readCiliumRulePorts`'s own fallback above
|
|
2403
|
+
// already relies on.
|
|
2404
|
+
const dnsNamesRead =
|
|
2405
|
+
dnsVisibilityRule === undefined ? undefined : readCiliumRuleDnsNames(dnsVisibilityRule)
|
|
2406
|
+
const dnsNarrowedTo: 'all' | readonly string[] =
|
|
2407
|
+
dnsNamesRead?.ok === true && dnsNamesRead.names !== 'all' ? dnsNamesRead.names : 'all'
|
|
2408
|
+
return {
|
|
2409
|
+
// The core-shaped reading of CILIUM_DNS_VISIBILITY_RULE — the one
|
|
2410
|
+
// place in this module where one resource's rule is restated in the
|
|
2411
|
+
// other's vocabulary, so that a core policy allowing exactly cluster
|
|
2412
|
+
// DNS is not reported as widening a translation that already allows
|
|
2413
|
+
// it. Nothing else about a Cilium translation is restated: an
|
|
2414
|
+
// allowlist of names has no core spelling at all.
|
|
2415
|
+
destinations: [{ peer: RESOLVER_PEER, ports: [{ start: 53 }] }],
|
|
2416
|
+
fqdns,
|
|
2417
|
+
dnsNarrowedTo,
|
|
2418
|
+
permitsEverything: false,
|
|
2419
|
+
permitsNothing: false,
|
|
2420
|
+
policyKind: translated.policyKind,
|
|
2421
|
+
}
|
|
2422
|
+
}
|
|
2423
|
+
const destinations: AllowedDestination[] = []
|
|
2424
|
+
let permitsEverything = false
|
|
2425
|
+
for (const rule of rules) {
|
|
2426
|
+
if (!isRecord(rule)) continue
|
|
2427
|
+
const ports = readPortRanges(rule.ports, 'TCP')
|
|
2428
|
+
if (ports === 'unreadable') continue
|
|
2429
|
+
const to = readList(rule.to)
|
|
2430
|
+
if (to === 'unreadable') continue
|
|
2431
|
+
if (to === undefined || to.length === 0) {
|
|
2432
|
+
destinations.push({ peer: { kind: 'everything' }, ports })
|
|
2433
|
+
if (ports === 'all') permitsEverything = true
|
|
2434
|
+
continue
|
|
2435
|
+
}
|
|
2436
|
+
for (const peer of to) {
|
|
2437
|
+
const read = readCorePeer(peer)
|
|
2438
|
+
if (read !== undefined) destinations.push({ peer: read, ports })
|
|
2439
|
+
}
|
|
2440
|
+
}
|
|
2441
|
+
return {
|
|
2442
|
+
destinations,
|
|
2443
|
+
fqdns: [],
|
|
2444
|
+
// A core `NetworkPolicy` translation never narrows DNS — it has no L7
|
|
2445
|
+
// concept to narrow with — so the DNS-widening check in
|
|
2446
|
+
// `coreEgressRuleVerdict`/`ciliumEgressRuleVerdict` never fires against
|
|
2447
|
+
// this allowance.
|
|
2448
|
+
dnsNarrowedTo: 'all',
|
|
2449
|
+
permitsEverything,
|
|
2450
|
+
permitsNothing: rules.length === 0,
|
|
2451
|
+
policyKind: translated.policyKind,
|
|
2452
|
+
}
|
|
2453
|
+
}
|
|
2454
|
+
|
|
2455
|
+
// ---------------------------------------------------------------------------
|
|
2456
|
+
// Is one policy's rule inside the translation
|
|
2457
|
+
// ---------------------------------------------------------------------------
|
|
2458
|
+
|
|
2459
|
+
function protocolCovers(allowed: string | undefined, wanted: string | undefined): boolean {
|
|
2460
|
+
// `undefined` on the allowed side is EVERY protocol; on the wanted side it
|
|
2461
|
+
// is also every protocol, which only an every-protocol allowance covers.
|
|
2462
|
+
return allowed === undefined || allowed === wanted
|
|
2463
|
+
}
|
|
2464
|
+
|
|
2465
|
+
function portRangeCovers(allowed: PortRange, wanted: PortRange): boolean {
|
|
2466
|
+
if (!protocolCovers(allowed.protocol, wanted.protocol)) return false
|
|
2467
|
+
if (allowed.start === undefined) return true
|
|
2468
|
+
if (wanted.start === undefined) return false
|
|
2469
|
+
const allowedEnd = allowed.end ?? allowed.start
|
|
2470
|
+
const wantedEnd = wanted.end ?? wanted.start
|
|
2471
|
+
return wanted.start >= allowed.start && wantedEnd <= allowedEnd
|
|
2472
|
+
}
|
|
2473
|
+
|
|
2474
|
+
function portsCover(allowed: 'all' | readonly PortRange[], wanted: PortRange): boolean {
|
|
2475
|
+
if (allowed === 'all') return true
|
|
2476
|
+
return allowed.some((range) => portRangeCovers(range, wanted))
|
|
2477
|
+
}
|
|
2478
|
+
|
|
2479
|
+
/** Every constraint `outer` places, `inner` places too — so `inner` selects a subset. */
|
|
2480
|
+
function selectorIsNarrower(outer: unknown, inner: unknown): boolean {
|
|
2481
|
+
if (outer === undefined) return true
|
|
2482
|
+
if (!isRecord(outer)) return false
|
|
2483
|
+
const outerLabels = outer.matchLabels
|
|
2484
|
+
const outerExpressions = readList(outer.matchExpressions)
|
|
2485
|
+
if (outerExpressions === 'unreadable') return false
|
|
2486
|
+
if (outerLabels === undefined && (outerExpressions ?? []).length === 0) {
|
|
2487
|
+
// An EMPTY selector is "everything of this kind", which every selector
|
|
2488
|
+
// of that kind is inside.
|
|
2489
|
+
return true
|
|
2490
|
+
}
|
|
2491
|
+
if (inner === undefined || !isRecord(inner)) return false
|
|
2492
|
+
if (outerLabels !== undefined) {
|
|
2493
|
+
if (!isRecord(outerLabels)) return false
|
|
2494
|
+
const innerLabels = inner.matchLabels
|
|
2495
|
+
if (!isRecord(innerLabels)) return false
|
|
2496
|
+
for (const [key, value] of Object.entries(outerLabels)) {
|
|
2497
|
+
if (!Object.hasOwn(innerLabels, key) || innerLabels[key] !== value) return false
|
|
2498
|
+
}
|
|
2499
|
+
}
|
|
2500
|
+
const innerExpressions = readList(inner.matchExpressions)
|
|
2501
|
+
if (innerExpressions === 'unreadable') return false
|
|
2502
|
+
for (const expression of outerExpressions ?? []) {
|
|
2503
|
+
if (!(innerExpressions ?? []).some((candidate) => isDeepStrictEqual(candidate, expression))) {
|
|
2504
|
+
return false
|
|
2505
|
+
}
|
|
2506
|
+
}
|
|
2507
|
+
return true
|
|
2508
|
+
}
|
|
2509
|
+
|
|
2510
|
+
/** Is every address `wanted` admits also admitted by `allowed`. */
|
|
2511
|
+
function peerIsWithin(allowed: PolicyPeer, wanted: PolicyPeer): boolean {
|
|
2512
|
+
if (allowed.kind === 'everything') return true
|
|
2513
|
+
if (wanted.kind === 'everything') return false
|
|
2514
|
+
if (allowed.kind === 'cidr') {
|
|
2515
|
+
if (wanted.kind !== 'cidr') return false
|
|
2516
|
+
if (!cidrContains(allowed.cidr, wanted.cidr)) return false
|
|
2517
|
+
// Every hole the allowance carves out of its block has to be a hole in
|
|
2518
|
+
// this peer too, or this peer reaches an address the translation does
|
|
2519
|
+
// not allow. Coverage by a UNION of the peer's own `except` entries is
|
|
2520
|
+
// not attempted: one entry has to contain it.
|
|
2521
|
+
for (const hole of allowed.except) {
|
|
2522
|
+
if (!cidrsOverlap(hole, wanted.cidr)) continue
|
|
2523
|
+
if (!wanted.except.some((own) => cidrContains(own, hole))) return false
|
|
2524
|
+
}
|
|
2525
|
+
return true
|
|
2526
|
+
}
|
|
2527
|
+
if (wanted.kind !== 'selector') return false
|
|
2528
|
+
// A namespaceSelector the allowance omits means "this namespace"; a peer
|
|
2529
|
+
// that names namespaces reaches further than that.
|
|
2530
|
+
if (allowed.namespaceSelector === undefined && wanted.namespaceSelector !== undefined) {
|
|
2531
|
+
return false
|
|
2532
|
+
}
|
|
2533
|
+
return (
|
|
2534
|
+
selectorIsNarrower(allowed.namespaceSelector, wanted.namespaceSelector) &&
|
|
2535
|
+
selectorIsNarrower(allowed.podSelector, wanted.podSelector)
|
|
2536
|
+
)
|
|
2537
|
+
}
|
|
2538
|
+
|
|
2539
|
+
function destinationIsAllowed(
|
|
2540
|
+
allowance: EgressAllowance,
|
|
2541
|
+
peer: PolicyPeer,
|
|
2542
|
+
ports: 'all' | readonly PortRange[],
|
|
2543
|
+
): boolean {
|
|
2544
|
+
const wanted: readonly PortRange[] = ports === 'all' ? [{}] : ports
|
|
2545
|
+
return wanted.every((range) =>
|
|
2546
|
+
allowance.destinations.some(
|
|
2547
|
+
(destination) => peerIsWithin(destination.peer, peer) && portsCover(destination.ports, range),
|
|
2548
|
+
),
|
|
2549
|
+
)
|
|
2550
|
+
}
|
|
2551
|
+
|
|
2552
|
+
/** Is every port `wanted` reaches on `host` also allowed by some `fqdns` entry naming it. */
|
|
2553
|
+
function fqdnIsAllowed(
|
|
2554
|
+
allowance: EgressAllowance,
|
|
2555
|
+
host: string,
|
|
2556
|
+
ports: 'all' | readonly PortRange[],
|
|
2557
|
+
): boolean {
|
|
2558
|
+
const wanted: readonly PortRange[] = ports === 'all' ? [{}] : ports
|
|
2559
|
+
return wanted.every((range) =>
|
|
2560
|
+
allowance.fqdns.some((entry) => entry.host === host && portsCover(entry.ports, range)),
|
|
2561
|
+
)
|
|
2562
|
+
}
|
|
2563
|
+
|
|
2564
|
+
/** Does `ports` (as a CANDIDATE rule's own port list) reach TCP or UDP 53 at all. */
|
|
2565
|
+
function coversDnsPort(ports: 'all' | readonly PortRange[]): boolean {
|
|
2566
|
+
if (ports === 'all') return true
|
|
2567
|
+
return ports.some((range) => {
|
|
2568
|
+
if (range.protocol !== undefined && range.protocol !== 'UDP' && range.protocol !== 'TCP') {
|
|
2569
|
+
return false
|
|
2570
|
+
}
|
|
2571
|
+
if (range.start === undefined) return true
|
|
2572
|
+
return range.start <= 53 && (range.end ?? range.start) >= 53
|
|
2573
|
+
})
|
|
2574
|
+
}
|
|
2575
|
+
|
|
2576
|
+
/**
|
|
2577
|
+
* Does a candidate rule's `peer`+`ports` reach the cluster resolver on the DNS
|
|
2578
|
+
* port at all — the precondition for the DNS-widening check both
|
|
2579
|
+
* {@link coreEgressRuleVerdict} and {@link ciliumEgressRuleVerdict} apply
|
|
2580
|
+
* before falling back to the ordinary peer/port `destinationIsAllowed` check.
|
|
2581
|
+
*/
|
|
2582
|
+
function reachesResolverAtDnsPort(peer: PolicyPeer, ports: 'all' | readonly PortRange[]): boolean {
|
|
2583
|
+
return peerIsWithin(peer, RESOLVER_PEER) && coversDnsPort(ports)
|
|
2584
|
+
}
|
|
2585
|
+
|
|
2586
|
+
/** One egress rule, judged against the translation. */
|
|
2587
|
+
export interface EgressRuleVerdict {
|
|
2588
|
+
readonly beyond: boolean | 'unknown'
|
|
2589
|
+
readonly detail?: string
|
|
2590
|
+
}
|
|
2591
|
+
|
|
2592
|
+
function describePeer(peer: PolicyPeer): string {
|
|
2593
|
+
if (peer.kind === 'everything') return 'every destination'
|
|
2594
|
+
return peer.text
|
|
2595
|
+
}
|
|
2596
|
+
|
|
2597
|
+
function describePorts(ports: 'all' | readonly PortRange[]): string {
|
|
2598
|
+
if (ports === 'all') return 'every port'
|
|
2599
|
+
return ports
|
|
2600
|
+
.map((range) =>
|
|
2601
|
+
range.start === undefined
|
|
2602
|
+
? `every ${range.protocol ?? ''} port`.trim()
|
|
2603
|
+
: `${range.protocol ?? 'any'} ${range.start}${range.end !== undefined ? `-${range.end}` : ''}`,
|
|
2604
|
+
)
|
|
2605
|
+
.join(', ')
|
|
2606
|
+
}
|
|
2607
|
+
|
|
2608
|
+
function coreEgressRuleVerdict(rule: unknown, allowance: EgressAllowance): EgressRuleVerdict {
|
|
2609
|
+
// Under a translation that permits nothing, the rule's own shape does not
|
|
2610
|
+
// matter and is not read: ANY egress rule on a policy selecting this pod
|
|
2611
|
+
// lets something out.
|
|
2612
|
+
if (allowance.permitsNothing) {
|
|
2613
|
+
return {
|
|
2614
|
+
beyond: true,
|
|
2615
|
+
detail: 'an egress rule, where the configured translation permits no egress at all',
|
|
2616
|
+
}
|
|
2617
|
+
}
|
|
2618
|
+
if (!isRecord(rule))
|
|
2619
|
+
return {
|
|
2620
|
+
beyond: 'unknown',
|
|
2621
|
+
detail: 'an egress rule that is not an object',
|
|
2622
|
+
}
|
|
2623
|
+
const ports = readPortRanges(rule.ports, 'TCP')
|
|
2624
|
+
if (ports === 'unreadable') {
|
|
2625
|
+
return {
|
|
2626
|
+
beyond: 'unknown',
|
|
2627
|
+
detail: 'a ports entry this check cannot read',
|
|
2628
|
+
}
|
|
2629
|
+
}
|
|
2630
|
+
const to = readList(rule.to)
|
|
2631
|
+
if (to === 'unreadable') {
|
|
2632
|
+
return { beyond: 'unknown', detail: "a 'to' that is not a list of peers" }
|
|
2633
|
+
}
|
|
2634
|
+
if (to === undefined || to.length === 0) {
|
|
2635
|
+
// No `to` on an egress rule means EVERY destination.
|
|
2636
|
+
return allowance.permitsEverything
|
|
2637
|
+
? { beyond: false }
|
|
2638
|
+
: {
|
|
2639
|
+
beyond: true,
|
|
2640
|
+
detail: `no 'to' peers, so every destination is reachable on ${describePorts(ports)}`,
|
|
2641
|
+
}
|
|
2642
|
+
}
|
|
2643
|
+
for (const peer of to) {
|
|
2644
|
+
const read = readCorePeer(peer)
|
|
2645
|
+
if (read === undefined) {
|
|
2646
|
+
return {
|
|
2647
|
+
beyond: 'unknown',
|
|
2648
|
+
detail: `a 'to' peer this check cannot read (${JSON.stringify(peer)})`,
|
|
2649
|
+
}
|
|
2650
|
+
}
|
|
2651
|
+
// A plain `NetworkPolicy` has no L7 concept, so it cannot express the DNS
|
|
2652
|
+
// restriction `ciliumNarrowing.dnsNames` narrows to — a core rule
|
|
2653
|
+
// reaching the resolver on the DNS port always resolves every name, and
|
|
2654
|
+
// that is wider than a narrowed translation whatever its own peer/port
|
|
2655
|
+
// shape says. This has to be decided BEFORE `destinationIsAllowed`
|
|
2656
|
+
// below: that check only reasons about reachability, and our own
|
|
2657
|
+
// translation's `destinations` entry for the resolver is peer/port-only
|
|
2658
|
+
// too, so a plain kube-dns rule would otherwise read as `within` a
|
|
2659
|
+
// translation it actually resolves every name for.
|
|
2660
|
+
if (allowance.dnsNarrowedTo !== 'all' && reachesResolverAtDnsPort(read, ports)) {
|
|
2661
|
+
return {
|
|
2662
|
+
beyond: true,
|
|
2663
|
+
detail: `a 'to' peer ${describePeer(read)} on ${describePorts(ports)} reaching the cluster resolver's DNS port with no DNS-name restriction — a plain NetworkPolicy cannot narrow lookups the way the configured translation's ciliumNarrowing.dnsNames does, so this rule resolves every name the narrowed policy does not`,
|
|
2664
|
+
}
|
|
2665
|
+
}
|
|
2666
|
+
if (!destinationIsAllowed(allowance, read, ports)) {
|
|
2667
|
+
const wideOpen = corePeerIsWideOpen(peer) === true
|
|
2668
|
+
return {
|
|
2669
|
+
beyond: true,
|
|
2670
|
+
detail: `${wideOpen ? 'a wide-open ' : 'a '}'to' peer ${describePeer(read)} on ${describePorts(ports)}, which a '${allowance.policyKind}' translation does not allow`,
|
|
2671
|
+
}
|
|
2672
|
+
}
|
|
2673
|
+
}
|
|
2674
|
+
return { beyond: false }
|
|
2675
|
+
}
|
|
2676
|
+
|
|
2677
|
+
/** Every `to…` field the Cilium CRD declares. A rule naming none of them is port-only. */
|
|
2678
|
+
const CILIUM_DESTINATION_FIELDS = [
|
|
2679
|
+
'toEndpoints',
|
|
2680
|
+
'toEntities',
|
|
2681
|
+
'toCIDR',
|
|
2682
|
+
'toCIDRSet',
|
|
2683
|
+
'toFQDNs',
|
|
2684
|
+
'toServices',
|
|
2685
|
+
'toGroups',
|
|
2686
|
+
'toNodes',
|
|
2687
|
+
] as const
|
|
2688
|
+
|
|
2689
|
+
/**
|
|
2690
|
+
* The label set a Cilium endpoint selector names, read back as the core
|
|
2691
|
+
* `namespaceSelector`/`podSelector` pair it is equivalent to, so the one
|
|
2692
|
+
* `peerIsWithin` serves both resource kinds. `undefined` when the selector
|
|
2693
|
+
* uses anything this reading cannot map — a label source that is not a pod
|
|
2694
|
+
* label, or a match expression.
|
|
2695
|
+
*/
|
|
2696
|
+
function ciliumEndpointPeer(selector: unknown): PolicyPeer | undefined {
|
|
2697
|
+
if (!isRecord(selector)) return undefined
|
|
2698
|
+
if (selector.matchExpressions !== undefined) return undefined
|
|
2699
|
+
const matchLabels = selector.matchLabels
|
|
2700
|
+
if (!isRecord(matchLabels)) return undefined
|
|
2701
|
+
const podLabels: Record<string, string> = {}
|
|
2702
|
+
let namespace: string | undefined
|
|
2703
|
+
for (const [rawKey, value] of Object.entries(matchLabels)) {
|
|
2704
|
+
if (typeof value !== 'string') return undefined
|
|
2705
|
+
const key = ciliumSelectorKey(rawKey)
|
|
2706
|
+
if (key === undefined) return undefined
|
|
2707
|
+
if (key === 'io.kubernetes.pod.namespace') {
|
|
2708
|
+
namespace = value
|
|
2709
|
+
continue
|
|
2710
|
+
}
|
|
2711
|
+
podLabels[key] = value
|
|
2712
|
+
}
|
|
2713
|
+
return {
|
|
2714
|
+
kind: 'selector',
|
|
2715
|
+
...(namespace !== undefined
|
|
2716
|
+
? {
|
|
2717
|
+
namespaceSelector: {
|
|
2718
|
+
matchLabels: { 'kubernetes.io/metadata.name': namespace },
|
|
2719
|
+
},
|
|
2720
|
+
}
|
|
2721
|
+
: {}),
|
|
2722
|
+
podSelector: { matchLabels: podLabels },
|
|
2723
|
+
text: JSON.stringify(selector),
|
|
2724
|
+
}
|
|
2725
|
+
}
|
|
2726
|
+
|
|
2727
|
+
/**
|
|
2728
|
+
* A Cilium rule's `toPorts`, read into the same {@link PortSet} core rules
|
|
2729
|
+
* use. Shared by {@link ciliumEgressRuleVerdict} (a CANDIDATE rule read off
|
|
2730
|
+
* the cluster) and {@link egressAllowance} (OUR OWN translated rule) so
|
|
2731
|
+
* there is one reading of "what ports does this Cilium rule reach", not two
|
|
2732
|
+
* that could silently disagree.
|
|
2733
|
+
*/
|
|
2734
|
+
type CiliumRulePorts =
|
|
2735
|
+
| { readonly ok: true; readonly ports: 'all' | readonly PortRange[] }
|
|
2736
|
+
| { readonly ok: false; readonly detail: string }
|
|
2737
|
+
|
|
2738
|
+
function readCiliumRulePorts(rule: Readonly<Record<string, unknown>>): CiliumRulePorts {
|
|
2739
|
+
// A Cilium rule carries its ports one level deeper, and a port entry with
|
|
2740
|
+
// no protocol means ANY rather than TCP.
|
|
2741
|
+
const toPorts = readList(rule.toPorts)
|
|
2742
|
+
if (toPorts === 'unreadable') {
|
|
2743
|
+
return { ok: false, detail: 'a toPorts that is not a list' }
|
|
2744
|
+
}
|
|
2745
|
+
const portEntries: unknown[] = []
|
|
2746
|
+
for (const entry of toPorts ?? []) {
|
|
2747
|
+
if (!isRecord(entry)) {
|
|
2748
|
+
return { ok: false, detail: 'a toPorts entry that is not an object' }
|
|
2749
|
+
}
|
|
2750
|
+
const list = readList(entry.ports)
|
|
2751
|
+
if (list === 'unreadable') {
|
|
2752
|
+
return { ok: false, detail: 'a toPorts ports field that is not a list' }
|
|
2753
|
+
}
|
|
2754
|
+
// An entry with no `ports` at all bounds nothing, so the rule reaches
|
|
2755
|
+
// every port — exactly what an absent `toPorts` means.
|
|
2756
|
+
if (list === undefined || list.length === 0) {
|
|
2757
|
+
portEntries.length = 0
|
|
2758
|
+
break
|
|
2759
|
+
}
|
|
2760
|
+
portEntries.push(...list)
|
|
2761
|
+
}
|
|
2762
|
+
const ports = readPortRanges(portEntries, undefined)
|
|
2763
|
+
if (ports === 'unreadable') {
|
|
2764
|
+
return { ok: false, detail: 'a toPorts entry this check cannot read' }
|
|
2765
|
+
}
|
|
2766
|
+
return { ok: true, ports }
|
|
2767
|
+
}
|
|
2768
|
+
|
|
2769
|
+
/**
|
|
2770
|
+
* A Cilium rule's `toPorts[].rules.dns`, read into the exact-name list it
|
|
2771
|
+
* restricts lookups to, or `'all'` when the rule does not restrict DNS at
|
|
2772
|
+
* all. Shared by {@link egressAllowance} (reading OUR OWN narrowed
|
|
2773
|
+
* DNS-visibility rule into {@link EgressAllowance.dnsNarrowedTo}) and
|
|
2774
|
+
* {@link ciliumEgressRuleVerdict} (deciding whether a CANDIDATE rule's own
|
|
2775
|
+
* restriction is narrow enough to not widen it) — one reading of "what names
|
|
2776
|
+
* does this Cilium rule let resolve", not two that could disagree.
|
|
2777
|
+
*
|
|
2778
|
+
* A `toPorts` entry with no `rules` at all, or a `rules.dns` entry carrying
|
|
2779
|
+
* `matchPattern` rather than `matchName`, both read as `'all'`: an absent L7
|
|
2780
|
+
* restriction resolves every name by definition, and this check does not
|
|
2781
|
+
* attempt to decide whether some wildcard pattern is a subset of an exact
|
|
2782
|
+
* name list — `'all'` is the conservative (never under-counts a widening)
|
|
2783
|
+
* answer for a shape it cannot reduce further.
|
|
2784
|
+
*/
|
|
2785
|
+
function readCiliumRuleDnsNames(
|
|
2786
|
+
rule: Readonly<Record<string, unknown>>,
|
|
2787
|
+
): { readonly ok: true; readonly names: 'all' | readonly string[] } | { readonly ok: false } {
|
|
2788
|
+
const toPorts = readList(rule.toPorts)
|
|
2789
|
+
if (toPorts === 'unreadable') return { ok: false }
|
|
2790
|
+
const names: string[] = []
|
|
2791
|
+
for (const entry of toPorts ?? []) {
|
|
2792
|
+
if (!isRecord(entry)) return { ok: false }
|
|
2793
|
+
if (entry.rules === undefined) return { ok: true, names: 'all' }
|
|
2794
|
+
if (!isRecord(entry.rules)) return { ok: false }
|
|
2795
|
+
const dns = readList(entry.rules.dns)
|
|
2796
|
+
if (dns === 'unreadable') return { ok: false }
|
|
2797
|
+
if (dns === undefined) return { ok: true, names: 'all' }
|
|
2798
|
+
for (const item of dns) {
|
|
2799
|
+
if (!isRecord(item)) return { ok: false }
|
|
2800
|
+
if (typeof item.matchName === 'string') {
|
|
2801
|
+
names.push(item.matchName)
|
|
2802
|
+
continue
|
|
2803
|
+
}
|
|
2804
|
+
// `matchPattern` (Cilium's glob syntax) or anything else this reading
|
|
2805
|
+
// does not recognise — both are read as unrestricted rather than
|
|
2806
|
+
// guessed at, per the doc comment above.
|
|
2807
|
+
return { ok: true, names: 'all' }
|
|
2808
|
+
}
|
|
2809
|
+
}
|
|
2810
|
+
return { ok: true, names }
|
|
2811
|
+
}
|
|
2812
|
+
|
|
2813
|
+
function ciliumEgressRuleVerdict(rule: unknown, allowance: EgressAllowance): EgressRuleVerdict {
|
|
2814
|
+
if (allowance.permitsNothing) {
|
|
2815
|
+
return {
|
|
2816
|
+
beyond: true,
|
|
2817
|
+
detail: 'an egress rule, where the configured translation permits no egress at all',
|
|
2818
|
+
}
|
|
2819
|
+
}
|
|
2820
|
+
if (!isRecord(rule))
|
|
2821
|
+
return {
|
|
2822
|
+
beyond: 'unknown',
|
|
2823
|
+
detail: 'an egress rule that is not an object',
|
|
2824
|
+
}
|
|
2825
|
+
const portsRead = readCiliumRulePorts(rule)
|
|
2826
|
+
if (!portsRead.ok) {
|
|
2827
|
+
return { beyond: 'unknown', detail: portsRead.detail }
|
|
2828
|
+
}
|
|
2829
|
+
const ports = portsRead.ports
|
|
2830
|
+
const fields = new Map<(typeof CILIUM_DESTINATION_FIELDS)[number], readonly unknown[]>()
|
|
2831
|
+
for (const field of CILIUM_DESTINATION_FIELDS) {
|
|
2832
|
+
const list = readList(rule[field])
|
|
2833
|
+
if (list === 'unreadable') {
|
|
2834
|
+
return {
|
|
2835
|
+
beyond: 'unknown',
|
|
2836
|
+
detail: `a ${field} that is not a list of peers`,
|
|
2837
|
+
}
|
|
2838
|
+
}
|
|
2839
|
+
if (list !== undefined && list.length > 0) fields.set(field, list)
|
|
2840
|
+
}
|
|
2841
|
+
if (fields.size === 0) {
|
|
2842
|
+
return allowance.permitsEverything
|
|
2843
|
+
? { beyond: false }
|
|
2844
|
+
: {
|
|
2845
|
+
beyond: true,
|
|
2846
|
+
detail: `a port-only egress rule, so every destination is reachable on ${describePorts(ports)}`,
|
|
2847
|
+
}
|
|
2848
|
+
}
|
|
2849
|
+
for (const entity of fields.get('toEntities') ?? []) {
|
|
2850
|
+
if (typeof entity !== 'string') {
|
|
2851
|
+
return {
|
|
2852
|
+
beyond: 'unknown',
|
|
2853
|
+
detail: 'a toEntities entry that is not an entity name',
|
|
2854
|
+
}
|
|
2855
|
+
}
|
|
2856
|
+
if (!allowance.permitsEverything) {
|
|
2857
|
+
return {
|
|
2858
|
+
beyond: true,
|
|
2859
|
+
detail: `toEntities '${entity}', which a '${allowance.policyKind}' translation does not allow`,
|
|
2860
|
+
}
|
|
2861
|
+
}
|
|
2862
|
+
}
|
|
2863
|
+
for (const field of ['toCIDR', 'toCIDRSet'] as const) {
|
|
2864
|
+
for (const entry of fields.get(field) ?? []) {
|
|
2865
|
+
const peer =
|
|
2866
|
+
field === 'toCIDR'
|
|
2867
|
+
? readCorePeer({ ipBlock: { cidr: entry } })
|
|
2868
|
+
: isRecord(entry)
|
|
2869
|
+
? readCorePeer({
|
|
2870
|
+
ipBlock: { cidr: entry.cidr, except: entry.except },
|
|
2871
|
+
})
|
|
2872
|
+
: undefined
|
|
2873
|
+
if (peer === undefined) {
|
|
2874
|
+
return {
|
|
2875
|
+
beyond: 'unknown',
|
|
2876
|
+
detail: `a ${field} entry this check cannot read`,
|
|
2877
|
+
}
|
|
2878
|
+
}
|
|
2879
|
+
if (!destinationIsAllowed(allowance, peer, ports)) {
|
|
2880
|
+
return {
|
|
2881
|
+
beyond: true,
|
|
2882
|
+
detail: `${field} ${describePeer(peer)} on ${describePorts(ports)}, which a '${allowance.policyKind}' translation does not allow`,
|
|
2883
|
+
}
|
|
2884
|
+
}
|
|
2885
|
+
}
|
|
2886
|
+
}
|
|
2887
|
+
for (const entry of fields.get('toFQDNs') ?? []) {
|
|
2888
|
+
if (!isRecord(entry)) {
|
|
2889
|
+
return {
|
|
2890
|
+
beyond: 'unknown',
|
|
2891
|
+
detail: 'a toFQDNs entry that is not an object',
|
|
2892
|
+
}
|
|
2893
|
+
}
|
|
2894
|
+
const matchName = entry.matchName
|
|
2895
|
+
if (typeof matchName !== 'string' || !fqdnIsAllowed(allowance, matchName, ports)) {
|
|
2896
|
+
return {
|
|
2897
|
+
beyond: true,
|
|
2898
|
+
detail: `toFQDNs ${JSON.stringify(entry)} on ${describePorts(ports)}, which a '${allowance.policyKind}' translation does not allow`,
|
|
2899
|
+
}
|
|
2900
|
+
}
|
|
2901
|
+
}
|
|
2902
|
+
for (const selector of fields.get('toEndpoints') ?? []) {
|
|
2903
|
+
const peer = ciliumEndpointPeer(selector)
|
|
2904
|
+
if (peer === undefined) {
|
|
2905
|
+
return {
|
|
2906
|
+
beyond: 'unknown',
|
|
2907
|
+
detail: 'a toEndpoints entry this check cannot read as a selector',
|
|
2908
|
+
}
|
|
2909
|
+
}
|
|
2910
|
+
// See the matching comment in `coreEgressRuleVerdict`: reachability alone
|
|
2911
|
+
// cannot tell a plain kube-dns rule from a narrowed one, so this has to
|
|
2912
|
+
// run before `destinationIsAllowed` below. Unlike a core rule, a Cilium
|
|
2913
|
+
// one CAN narrow DNS on its own `toPorts.rules.dns` — read it and accept
|
|
2914
|
+
// the rule only when what it names is a subset of what our own
|
|
2915
|
+
// translation narrows to.
|
|
2916
|
+
if (allowance.dnsNarrowedTo !== 'all' && reachesResolverAtDnsPort(peer, ports)) {
|
|
2917
|
+
const narrowedTo = allowance.dnsNarrowedTo
|
|
2918
|
+
const candidate = readCiliumRuleDnsNames(rule)
|
|
2919
|
+
const isSubset =
|
|
2920
|
+
candidate.ok &&
|
|
2921
|
+
candidate.names !== 'all' &&
|
|
2922
|
+
candidate.names.every((n) => narrowedTo.includes(n))
|
|
2923
|
+
if (!isSubset) {
|
|
2924
|
+
return {
|
|
2925
|
+
beyond: true,
|
|
2926
|
+
detail: `toEndpoints ${describePeer(peer)} on ${describePorts(ports)} reaching the cluster resolver's DNS port with ${candidate.ok && candidate.names !== 'all' ? 'a rules.dns list this check cannot confirm is a subset of' : 'no rules.dns restriction at least as narrow as'} the configured translation's ciliumNarrowing.dnsNames`,
|
|
2927
|
+
}
|
|
2928
|
+
}
|
|
2929
|
+
continue
|
|
2930
|
+
}
|
|
2931
|
+
if (!destinationIsAllowed(allowance, peer, ports)) {
|
|
2932
|
+
return {
|
|
2933
|
+
beyond: true,
|
|
2934
|
+
detail: `toEndpoints ${describePeer(peer)} on ${describePorts(ports)}, which a '${allowance.policyKind}' translation does not allow`,
|
|
2935
|
+
}
|
|
2936
|
+
}
|
|
2937
|
+
}
|
|
2938
|
+
for (const field of ['toServices', 'toGroups', 'toNodes'] as const) {
|
|
2939
|
+
if (fields.has(field)) {
|
|
2940
|
+
return {
|
|
2941
|
+
beyond: 'unknown',
|
|
2942
|
+
detail: `a ${field} rule, whose destinations this check cannot enumerate`,
|
|
2943
|
+
}
|
|
2944
|
+
}
|
|
2945
|
+
}
|
|
2946
|
+
return { beyond: false }
|
|
2947
|
+
}
|
|
2948
|
+
|
|
2949
|
+
// ---------------------------------------------------------------------------
|
|
2950
|
+
// Normalising the policies that select this pod
|
|
2951
|
+
// ---------------------------------------------------------------------------
|
|
2952
|
+
|
|
2953
|
+
/** The pod the union check is about. */
|
|
2954
|
+
export interface EgressVerificationTarget {
|
|
2955
|
+
readonly namespace: string
|
|
2956
|
+
/**
|
|
2957
|
+
* The pod's REAL labels — for a directly created Sandbox the labels the
|
|
2958
|
+
* create body stamps (known before the POST, so a refusal leaves no
|
|
2959
|
+
* Sandbox and no PVC behind), for a claimed one the bound pod's own
|
|
2960
|
+
* `metadata.labels`. The same value the ingress check is given, for the
|
|
2961
|
+
* same reason: a name proves an object exists, a label is what a selector
|
|
2962
|
+
* actually matches.
|
|
2963
|
+
*/
|
|
2964
|
+
readonly podLabels: Readonly<Record<string, string>>
|
|
2965
|
+
readonly engine: KubernetesEgressEngine
|
|
2966
|
+
/** How the refusal names the thing being created, e.g. `Sandbox namzu-ws-demo`. */
|
|
2967
|
+
readonly subject: string
|
|
2968
|
+
}
|
|
2969
|
+
|
|
2970
|
+
/** What one examined policy turned out to be. One line of the refusal. */
|
|
2971
|
+
export type EgressPolicyVerdict =
|
|
2972
|
+
/** Selects the pod and lets out nothing the translation does not. */
|
|
2973
|
+
| 'within'
|
|
2974
|
+
/** Selects the pod and allows egress the translation does not. */
|
|
2975
|
+
| 'widens-egress'
|
|
2976
|
+
/** Its selector does not match the pod's labels. */
|
|
2977
|
+
| 'does-not-select'
|
|
2978
|
+
/** Selects the pod but does not enforce egress, so its egress block is inert. */
|
|
2979
|
+
| 'not-egress-scoped'
|
|
2980
|
+
/** Contains something this check cannot decide. */
|
|
2981
|
+
| 'not-evaluable'
|
|
2982
|
+
|
|
2983
|
+
/** One policy, as the refusal reports it. */
|
|
2984
|
+
export interface ExaminedEgressPolicy {
|
|
2985
|
+
readonly kind: 'NetworkPolicy' | 'CiliumNetworkPolicy'
|
|
2986
|
+
readonly name: string
|
|
2987
|
+
readonly verdict: EgressPolicyVerdict
|
|
2988
|
+
/** Why, for every verdict that is not a plain match or non-match. */
|
|
2989
|
+
readonly detail?: string
|
|
2990
|
+
}
|
|
2991
|
+
|
|
2992
|
+
/**
|
|
2993
|
+
* Which of the three refusals this is — three different operator actions, so
|
|
2994
|
+
* they are carried apart rather than folded into one message:
|
|
2995
|
+
*
|
|
2996
|
+
* - `policy-widens-egress` — narrow or delete the policy that lets more out
|
|
2997
|
+
* than `config.egress` says.
|
|
2998
|
+
* - `no-enforcing-policy` — nothing puts this pod in egress default-deny, so
|
|
2999
|
+
* the translation is not the boundary; apply a policy that selects these
|
|
3000
|
+
* labels.
|
|
3001
|
+
* - `not-evaluable` — this check cannot decide; grant the missing verb, fix
|
|
3002
|
+
* the unreadable policy, or declare `egress.verify: 'named-object-only'`.
|
|
3003
|
+
*/
|
|
3004
|
+
export type EgressPolicyRefusal = 'policy-widens-egress' | 'no-enforcing-policy' | 'not-evaluable'
|
|
3005
|
+
|
|
3006
|
+
/**
|
|
3007
|
+
* One policy, reduced to the three questions the union rule asks: does it
|
|
3008
|
+
* select this pod, does it put it in egress default-deny, and does anything
|
|
3009
|
+
* it allows fall outside the configured translation.
|
|
3010
|
+
*/
|
|
3011
|
+
export interface EgressPolicyDocument {
|
|
3012
|
+
readonly kind: 'NetworkPolicy' | 'CiliumNetworkPolicy'
|
|
3013
|
+
readonly name: string
|
|
3014
|
+
readonly selects: SelectorMatch
|
|
3015
|
+
/**
|
|
3016
|
+
* Does this policy put the pod into egress DEFAULT-DENY — the only thing
|
|
3017
|
+
* that makes the translation a boundary at all. A core policy whose
|
|
3018
|
+
* `policyTypes` leaves Egress out does not (and the API server ignores its
|
|
3019
|
+
* `egress` block outright), and neither does a Cilium rule carrying
|
|
3020
|
+
* `enableDefaultDeny.egress: false`.
|
|
3021
|
+
*/
|
|
3022
|
+
readonly enforcesEgress: boolean
|
|
3023
|
+
readonly rules: readonly EgressRuleVerdict[]
|
|
3024
|
+
/** Set when the OBJECT could not be read. It decides alone. */
|
|
3025
|
+
readonly unreadable?: string
|
|
3026
|
+
}
|
|
3027
|
+
|
|
3028
|
+
function unreadableEgressPolicy(
|
|
3029
|
+
kind: EgressPolicyDocument['kind'],
|
|
3030
|
+
name: string,
|
|
3031
|
+
detail: string,
|
|
3032
|
+
): EgressPolicyDocument {
|
|
3033
|
+
return {
|
|
3034
|
+
kind,
|
|
3035
|
+
name,
|
|
3036
|
+
selects: 'unknown',
|
|
3037
|
+
enforcesEgress: false,
|
|
3038
|
+
rules: [],
|
|
3039
|
+
unreadable: detail,
|
|
3040
|
+
}
|
|
3041
|
+
}
|
|
3042
|
+
|
|
3043
|
+
/** Every core `NetworkPolicy` in the list, reduced to {@link EgressPolicyDocument}. */
|
|
3044
|
+
export function readCoreEgressPolicies(
|
|
3045
|
+
items: readonly unknown[],
|
|
3046
|
+
target: EgressVerificationTarget,
|
|
3047
|
+
allowance: EgressAllowance,
|
|
3048
|
+
): EgressPolicyDocument[] {
|
|
3049
|
+
return items.map((item, index) => {
|
|
3050
|
+
const name = policyName(item, index)
|
|
3051
|
+
if (!isRecord(item)) {
|
|
3052
|
+
return unreadableEgressPolicy(
|
|
3053
|
+
'NetworkPolicy',
|
|
3054
|
+
name,
|
|
3055
|
+
'a list entry that is not a policy object',
|
|
3056
|
+
)
|
|
3057
|
+
}
|
|
3058
|
+
const spec = item.spec
|
|
3059
|
+
if (!isRecord(spec)) {
|
|
3060
|
+
return unreadableEgressPolicy('NetworkPolicy', name, 'a spec that is not an object')
|
|
3061
|
+
}
|
|
3062
|
+
const policyTypes = readList(spec.policyTypes)
|
|
3063
|
+
if (policyTypes === 'unreadable') {
|
|
3064
|
+
return unreadableEgressPolicy('NetworkPolicy', name, 'a spec.policyTypes that is not a list')
|
|
3065
|
+
}
|
|
3066
|
+
const rules = readList(spec.egress)
|
|
3067
|
+
if (rules === 'unreadable') {
|
|
3068
|
+
return unreadableEgressPolicy(
|
|
3069
|
+
'NetworkPolicy',
|
|
3070
|
+
name,
|
|
3071
|
+
'a spec.egress that is not a list of rules',
|
|
3072
|
+
)
|
|
3073
|
+
}
|
|
3074
|
+
// An ABSENT `policyTypes` is defaulted by the API server from the blocks
|
|
3075
|
+
// the object carries: Egress appears exactly when `spec.egress` does.
|
|
3076
|
+
// This is the opposite of the ingress default, where Egress's absence is
|
|
3077
|
+
// the thing that has to be spelled out.
|
|
3078
|
+
const enforcesEgress =
|
|
3079
|
+
policyTypes === undefined ? rules !== undefined : policyTypes.includes('Egress')
|
|
3080
|
+
return {
|
|
3081
|
+
kind: 'NetworkPolicy',
|
|
3082
|
+
name,
|
|
3083
|
+
selects: matchesLabelSelector(spec.podSelector, target.podLabels),
|
|
3084
|
+
enforcesEgress,
|
|
3085
|
+
// An `egress` block under a `policyTypes` that leaves Egress out is
|
|
3086
|
+
// ignored by the API server itself: it neither bounds nor widens.
|
|
3087
|
+
rules: enforcesEgress
|
|
3088
|
+
? (rules ?? []).map((rule) => coreEgressRuleVerdict(rule, allowance))
|
|
3089
|
+
: [],
|
|
3090
|
+
}
|
|
3091
|
+
})
|
|
3092
|
+
}
|
|
3093
|
+
|
|
3094
|
+
/** Every `CiliumNetworkPolicy` in the list, reduced the same way. */
|
|
3095
|
+
export function readCiliumEgressPolicies(
|
|
3096
|
+
items: readonly unknown[],
|
|
3097
|
+
target: EgressVerificationTarget,
|
|
3098
|
+
allowance: EgressAllowance,
|
|
3099
|
+
): EgressPolicyDocument[] {
|
|
3100
|
+
const identity = ciliumIdentityLabels(target.podLabels, target.namespace)
|
|
3101
|
+
const documents: EgressPolicyDocument[] = []
|
|
3102
|
+
for (const [index, item] of items.entries()) {
|
|
3103
|
+
const name = policyName(item, index)
|
|
3104
|
+
if (!isRecord(item)) {
|
|
3105
|
+
documents.push(
|
|
3106
|
+
unreadableEgressPolicy(
|
|
3107
|
+
'CiliumNetworkPolicy',
|
|
3108
|
+
name,
|
|
3109
|
+
'a list entry that is not a policy object',
|
|
3110
|
+
),
|
|
3111
|
+
)
|
|
3112
|
+
continue
|
|
3113
|
+
}
|
|
3114
|
+
// The CRD carries EITHER one `spec` or a `specs` list, and a rule in
|
|
3115
|
+
// either enforces. Reading only `spec` would miss a whole policy.
|
|
3116
|
+
const specs: unknown[] = []
|
|
3117
|
+
if (item.spec !== undefined && item.spec !== null) specs.push(item.spec)
|
|
3118
|
+
const more = readList(item.specs)
|
|
3119
|
+
if (more === 'unreadable') {
|
|
3120
|
+
documents.push(
|
|
3121
|
+
unreadableEgressPolicy(
|
|
3122
|
+
'CiliumNetworkPolicy',
|
|
3123
|
+
name,
|
|
3124
|
+
'a specs that is not a list of rule specs',
|
|
3125
|
+
),
|
|
3126
|
+
)
|
|
3127
|
+
continue
|
|
3128
|
+
}
|
|
3129
|
+
for (const spec of more ?? []) specs.push(spec)
|
|
3130
|
+
if (specs.length === 0) {
|
|
3131
|
+
documents.push(
|
|
3132
|
+
unreadableEgressPolicy('CiliumNetworkPolicy', name, 'neither a spec nor a specs list'),
|
|
3133
|
+
)
|
|
3134
|
+
continue
|
|
3135
|
+
}
|
|
3136
|
+
for (const spec of specs) {
|
|
3137
|
+
const document = readCiliumEgressRuleSpec(spec, name, identity, allowance)
|
|
3138
|
+
if (document !== undefined) documents.push(document)
|
|
3139
|
+
}
|
|
3140
|
+
}
|
|
3141
|
+
return documents
|
|
3142
|
+
}
|
|
3143
|
+
|
|
3144
|
+
/** One `spec`/`specs` entry. `undefined` when it is node-scoped — see below. */
|
|
3145
|
+
function readCiliumEgressRuleSpec(
|
|
3146
|
+
spec: unknown,
|
|
3147
|
+
name: string,
|
|
3148
|
+
identity: Readonly<Record<string, string>>,
|
|
3149
|
+
allowance: EgressAllowance,
|
|
3150
|
+
): EgressPolicyDocument | undefined {
|
|
3151
|
+
const unreadable = (detail: string) => unreadableEgressPolicy('CiliumNetworkPolicy', name, detail)
|
|
3152
|
+
if (!isRecord(spec)) return unreadable('a rule spec that is not an object')
|
|
3153
|
+
// A node-scoped rule selects nodes, never pods.
|
|
3154
|
+
if (spec.nodeSelector !== undefined) return undefined
|
|
3155
|
+
const rules = readList(spec.egress)
|
|
3156
|
+
if (rules === 'unreadable') return unreadable('a spec.egress that is not a list of rules')
|
|
3157
|
+
const egressDeny = readList(spec.egressDeny)
|
|
3158
|
+
if (egressDeny === 'unreadable')
|
|
3159
|
+
return unreadable('a spec.egressDeny that is not a list of rules')
|
|
3160
|
+
let enforcesEgress = rules !== undefined || egressDeny !== undefined
|
|
3161
|
+
const enableDefaultDeny = spec.enableDefaultDeny
|
|
3162
|
+
if (enableDefaultDeny !== undefined) {
|
|
3163
|
+
if (!isRecord(enableDefaultDeny))
|
|
3164
|
+
return unreadable('an enableDefaultDeny that is not an object')
|
|
3165
|
+
const forEgress = enableDefaultDeny.egress
|
|
3166
|
+
if (forEgress !== undefined && typeof forEgress !== 'boolean') {
|
|
3167
|
+
return unreadable('an enableDefaultDeny.egress that is not a boolean')
|
|
3168
|
+
}
|
|
3169
|
+
if (forEgress === false) enforcesEgress = false
|
|
3170
|
+
}
|
|
3171
|
+
return {
|
|
3172
|
+
kind: 'CiliumNetworkPolicy',
|
|
3173
|
+
name,
|
|
3174
|
+
selects: matchesLabelSelector(spec.endpointSelector, identity, ciliumSelectorKey),
|
|
3175
|
+
enforcesEgress,
|
|
3176
|
+
// `egressDeny` rules are not read: a deny rule can only narrow what
|
|
3177
|
+
// leaves the pod, and this check refuses widening. Allow rules are read
|
|
3178
|
+
// whatever `enableDefaultDeny` says, because what a rule admits it
|
|
3179
|
+
// admits — it simply may not be the policy that default-denies.
|
|
3180
|
+
rules: (rules ?? []).map((rule) => ciliumEgressRuleVerdict(rule, allowance)),
|
|
3181
|
+
}
|
|
3182
|
+
}
|
|
3183
|
+
|
|
3184
|
+
// ---------------------------------------------------------------------------
|
|
3185
|
+
// The decision
|
|
3186
|
+
// ---------------------------------------------------------------------------
|
|
3187
|
+
|
|
3188
|
+
export interface EgressUnionDecision {
|
|
3189
|
+
readonly examined: readonly ExaminedEgressPolicy[]
|
|
3190
|
+
/** Absent when every selecting policy stays inside the translation. */
|
|
3191
|
+
readonly refusal?: {
|
|
3192
|
+
readonly kind: EgressPolicyRefusal
|
|
3193
|
+
readonly summary: string
|
|
3194
|
+
}
|
|
3195
|
+
}
|
|
3196
|
+
|
|
3197
|
+
/**
|
|
3198
|
+
* The union rule, applied. Pure — no I/O, no client, no clock — so every
|
|
3199
|
+
* shape that has to be refused can be asserted one per test.
|
|
3200
|
+
*
|
|
3201
|
+
* Kubernetes UNIONS every policy selecting a pod: traffic leaves if ANY
|
|
3202
|
+
* selecting policy allows it. So one policy allowing more than the
|
|
3203
|
+
* translation is the finding however many narrower ones sit beside it, and a
|
|
3204
|
+
* pod no policy default-denies has no egress boundary at all whatever the
|
|
3205
|
+
* named object says.
|
|
3206
|
+
*/
|
|
3207
|
+
export function decideEgressUnion(
|
|
3208
|
+
documents: readonly EgressPolicyDocument[],
|
|
3209
|
+
allowance: EgressAllowance,
|
|
3210
|
+
): EgressUnionDecision {
|
|
3211
|
+
const examined: ExaminedEgressPolicy[] = []
|
|
3212
|
+
let enforcing = 0
|
|
3213
|
+
let widening: ExaminedEgressPolicy | undefined
|
|
3214
|
+
let undecided: ExaminedEgressPolicy | undefined
|
|
3215
|
+
|
|
3216
|
+
for (const document of documents) {
|
|
3217
|
+
const base = { kind: document.kind, name: document.name } as const
|
|
3218
|
+
if (document.unreadable !== undefined) {
|
|
3219
|
+
const entry: ExaminedEgressPolicy = {
|
|
3220
|
+
...base,
|
|
3221
|
+
verdict: 'not-evaluable',
|
|
3222
|
+
detail: document.unreadable,
|
|
3223
|
+
}
|
|
3224
|
+
examined.push(entry)
|
|
3225
|
+
undecided ??= entry
|
|
3226
|
+
continue
|
|
3227
|
+
}
|
|
3228
|
+
if (document.selects === 'no') {
|
|
3229
|
+
examined.push({ ...base, verdict: 'does-not-select' })
|
|
3230
|
+
continue
|
|
3231
|
+
}
|
|
3232
|
+
if (document.selects === 'unknown') {
|
|
3233
|
+
const entry: ExaminedEgressPolicy = {
|
|
3234
|
+
...base,
|
|
3235
|
+
verdict: 'not-evaluable',
|
|
3236
|
+
detail: 'its selector uses something this check cannot evaluate against pod labels',
|
|
3237
|
+
}
|
|
3238
|
+
examined.push(entry)
|
|
3239
|
+
undecided ??= entry
|
|
3240
|
+
continue
|
|
3241
|
+
}
|
|
3242
|
+
const beyondRule = document.rules.find((rule) => rule.beyond === true)
|
|
3243
|
+
if (beyondRule !== undefined) {
|
|
3244
|
+
const entry: ExaminedEgressPolicy = {
|
|
3245
|
+
...base,
|
|
3246
|
+
verdict: 'widens-egress',
|
|
3247
|
+
...(beyondRule.detail !== undefined ? { detail: beyondRule.detail } : {}),
|
|
3248
|
+
}
|
|
3249
|
+
examined.push(entry)
|
|
3250
|
+
widening ??= entry
|
|
3251
|
+
continue
|
|
3252
|
+
}
|
|
3253
|
+
const unknownRule = document.rules.find((rule) => rule.beyond === 'unknown')
|
|
3254
|
+
if (unknownRule !== undefined) {
|
|
3255
|
+
const entry: ExaminedEgressPolicy = {
|
|
3256
|
+
...base,
|
|
3257
|
+
verdict: 'not-evaluable',
|
|
3258
|
+
...(unknownRule.detail !== undefined ? { detail: unknownRule.detail } : {}),
|
|
3259
|
+
}
|
|
3260
|
+
examined.push(entry)
|
|
3261
|
+
undecided ??= entry
|
|
3262
|
+
continue
|
|
3263
|
+
}
|
|
3264
|
+
if (!document.enforcesEgress) {
|
|
3265
|
+
examined.push({
|
|
3266
|
+
...base,
|
|
3267
|
+
verdict: 'not-egress-scoped',
|
|
3268
|
+
detail: 'it selects the pod but default-denies nothing on egress',
|
|
3269
|
+
})
|
|
3270
|
+
continue
|
|
3271
|
+
}
|
|
3272
|
+
examined.push({ ...base, verdict: 'within' })
|
|
3273
|
+
enforcing += 1
|
|
3274
|
+
}
|
|
3275
|
+
|
|
3276
|
+
if (widening !== undefined) {
|
|
3277
|
+
return {
|
|
3278
|
+
examined,
|
|
3279
|
+
refusal: {
|
|
3280
|
+
kind: 'policy-widens-egress',
|
|
3281
|
+
summary: `${widening.kind}/${widening.name} selects this pod and allows ${widening.detail ?? 'egress the configured translation does not'}.`,
|
|
3282
|
+
},
|
|
3283
|
+
}
|
|
3284
|
+
}
|
|
3285
|
+
if (undecided !== undefined) {
|
|
3286
|
+
return {
|
|
3287
|
+
examined,
|
|
3288
|
+
refusal: {
|
|
3289
|
+
kind: 'not-evaluable',
|
|
3290
|
+
summary: `${undecided.kind}/${undecided.name} contains ${undecided.detail ?? 'something this check cannot evaluate'}, so what this pod may reach cannot be decided from the cluster's own objects.`,
|
|
3291
|
+
},
|
|
3292
|
+
}
|
|
3293
|
+
}
|
|
3294
|
+
// `allow-all` asks for no boundary, so a pod nothing default-denies is
|
|
3295
|
+
// exactly what it configured; every other kind needs a policy that
|
|
3296
|
+
// actually puts this pod in egress default-deny, or the translation is a
|
|
3297
|
+
// manifest nobody enforces.
|
|
3298
|
+
if (enforcing === 0 && !allowance.permitsEverything) {
|
|
3299
|
+
return {
|
|
3300
|
+
examined,
|
|
3301
|
+
refusal: {
|
|
3302
|
+
kind: 'no-enforcing-policy',
|
|
3303
|
+
summary: `no applied policy puts this pod in egress default-deny, so a '${allowance.policyKind}' translation bounds nothing it sends.`,
|
|
3304
|
+
},
|
|
3305
|
+
}
|
|
3306
|
+
}
|
|
3307
|
+
return { examined }
|
|
3308
|
+
}
|
|
3309
|
+
|
|
3310
|
+
/**
|
|
3311
|
+
* The named refusal. Distinct from {@link KubernetesEgressPolicyMismatchError}
|
|
3312
|
+
* — which is about the ONE named object drifting from the translation — and
|
|
3313
|
+
* from `KubernetesIngressPolicyError`, which is about the agent port being
|
|
3314
|
+
* reachable. An operator debugging a release that ships all three tells them
|
|
3315
|
+
* apart by class and by the first clause of the message.
|
|
3316
|
+
*
|
|
3317
|
+
* It carries the pod's labels and EVERY policy examined, with a verdict each,
|
|
3318
|
+
* because that list is the operator's whole debugging session: "why does my
|
|
3319
|
+
* policy not count?" is answered by the line saying it did not select these
|
|
3320
|
+
* labels.
|
|
3321
|
+
*/
|
|
3322
|
+
export class KubernetesEgressPolicyUnionError extends Error {
|
|
3323
|
+
override readonly name = 'KubernetesEgressPolicyUnionError'
|
|
3324
|
+
|
|
3325
|
+
constructor(
|
|
3326
|
+
readonly refusal: EgressPolicyRefusal,
|
|
3327
|
+
readonly subject: string,
|
|
3328
|
+
readonly podLabels: Readonly<Record<string, string>>,
|
|
3329
|
+
readonly policyKind: KubernetesEgressPolicy['kind'],
|
|
3330
|
+
readonly examined: readonly ExaminedEgressPolicy[],
|
|
3331
|
+
summary: string,
|
|
3332
|
+
/** Empty on every decision made from policies that WERE read. */
|
|
3333
|
+
readonly unread: readonly UnreadPolicySource[] = [],
|
|
3334
|
+
) {
|
|
3335
|
+
super(
|
|
3336
|
+
`kubernetes: refusing ${subject} — ${summary} config.egress.policy is '${policyKind}' and the pod's labels are ${formatLabels(podLabels)}. ${formatExaminedEgress(examined, unread)} Kubernetes UNIONS every policy selecting a pod, so what leaves this pod is everything ANY of them allows — a second policy widens egress however exactly the named object matches. ${formatEgressRemedy(refusal, unread)}`,
|
|
3337
|
+
)
|
|
3338
|
+
}
|
|
3339
|
+
}
|
|
3340
|
+
|
|
3341
|
+
function formatExaminedEgress(
|
|
3342
|
+
examined: readonly ExaminedEgressPolicy[],
|
|
3343
|
+
unread: readonly UnreadPolicySource[],
|
|
3344
|
+
): string {
|
|
3345
|
+
const lines = examined.map((entry) => {
|
|
3346
|
+
const detail = entry.detail !== undefined ? ` (${entry.detail})` : ''
|
|
3347
|
+
return `${entry.kind}/${entry.name}: ${entry.verdict}${detail}`
|
|
3348
|
+
})
|
|
3349
|
+
const read =
|
|
3350
|
+
lines.length === 0
|
|
3351
|
+
? 'No policy was read from the collections this check could enumerate.'
|
|
3352
|
+
: `Policies examined: ${lines.join('; ')}.`
|
|
3353
|
+
if (unread.length === 0) return read
|
|
3354
|
+
const missing = unread
|
|
3355
|
+
.map((source) => `${source.resource} at ${source.path} (${source.why}: ${source.reason})`)
|
|
3356
|
+
.join('; ')
|
|
3357
|
+
return `${read} NOT read, so nothing below is a statement about it: ${missing}.`
|
|
3358
|
+
}
|
|
3359
|
+
|
|
3360
|
+
function formatEgressRemedy(
|
|
3361
|
+
refusal: EgressPolicyRefusal,
|
|
3362
|
+
unread: readonly UnreadPolicySource[],
|
|
3363
|
+
): string {
|
|
3364
|
+
if (refusal === 'not-evaluable') {
|
|
3365
|
+
const forbidden = unread.some((source) => source.why === 'forbidden')
|
|
3366
|
+
return `${forbidden ? "Grant this backend's ServiceAccount 'list' on that resource, " : 'Fix or remove the policy named above, '}or set egress.verify: 'named-object-only' to check only the one named object, as every release before this one did.`
|
|
3367
|
+
}
|
|
3368
|
+
if (refusal === 'no-enforcing-policy') {
|
|
3369
|
+
return "Apply a NetworkPolicy whose podSelector matches the labels above and whose policyTypes includes 'Egress' (packages/sandbox/k8s/manifests/networkpolicy.yaml is this repo's baseline), or set egress.verify: 'named-object-only' if the boundary lives somewhere a namespaced Role cannot read."
|
|
3370
|
+
}
|
|
3371
|
+
return "Narrow or delete the policy named above so that nothing selecting these pods allows more than config.egress does, or set egress.verify: 'named-object-only' to go back to checking only the named object."
|
|
3372
|
+
}
|
|
3373
|
+
|
|
3374
|
+
// ---------------------------------------------------------------------------
|
|
3375
|
+
// The I/O half
|
|
3376
|
+
// ---------------------------------------------------------------------------
|
|
3377
|
+
|
|
3378
|
+
/**
|
|
3379
|
+
* Verify-not-trust, widened from one object to the union: list the
|
|
3380
|
+
* namespace's policies, evaluate every one that selects this pod against the
|
|
3381
|
+
* configured translation, and refuse unless nothing lets out more than
|
|
3382
|
+
* `config.egress` says.
|
|
3383
|
+
*
|
|
3384
|
+
* Runs beside the ingress check on every create path — before the POST for a
|
|
3385
|
+
* directly created Sandbox, where the labels are known and a refusal leaves
|
|
3386
|
+
* nothing behind, and after the bind for a claimed one, where the pool's own
|
|
3387
|
+
* template decides the labels and a refusal releases the claim through the
|
|
3388
|
+
* acquire path's cleanup.
|
|
3389
|
+
*/
|
|
3390
|
+
export async function verifyEgressPolicyUnion(
|
|
3391
|
+
client: KubernetesClient,
|
|
3392
|
+
translated: KubernetesTranslatedEgressPolicy,
|
|
3393
|
+
target: EgressVerificationTarget,
|
|
3394
|
+
signal?: AbortSignal,
|
|
3395
|
+
): Promise<void> {
|
|
3396
|
+
const allowance = egressAllowance(translated)
|
|
3397
|
+
const documents: EgressPolicyDocument[] = []
|
|
3398
|
+
try {
|
|
3399
|
+
documents.push(
|
|
3400
|
+
...readCoreEgressPolicies(
|
|
3401
|
+
await listPolicies(
|
|
3402
|
+
client,
|
|
3403
|
+
networkPolicyCollectionPath(target.namespace),
|
|
3404
|
+
'networkpolicies',
|
|
3405
|
+
signal,
|
|
3406
|
+
),
|
|
3407
|
+
target,
|
|
3408
|
+
allowance,
|
|
3409
|
+
),
|
|
3410
|
+
)
|
|
3411
|
+
if (target.engine === 'cilium') {
|
|
3412
|
+
documents.push(
|
|
3413
|
+
...readCiliumEgressPolicies(
|
|
3414
|
+
await listPolicies(
|
|
3415
|
+
client,
|
|
3416
|
+
ciliumNetworkPolicyCollectionPath(target.namespace),
|
|
3417
|
+
'ciliumnetworkpolicies',
|
|
3418
|
+
signal,
|
|
3419
|
+
),
|
|
3420
|
+
target,
|
|
3421
|
+
allowance,
|
|
3422
|
+
),
|
|
3423
|
+
)
|
|
3424
|
+
}
|
|
3425
|
+
} catch (err) {
|
|
3426
|
+
if (!(err instanceof UnreadPolicyCollection)) throw err
|
|
3427
|
+
throw new KubernetesEgressPolicyUnionError(
|
|
3428
|
+
'not-evaluable',
|
|
3429
|
+
target.subject,
|
|
3430
|
+
target.podLabels,
|
|
3431
|
+
allowance.policyKind,
|
|
3432
|
+
decideEgressUnion(documents, allowance).examined,
|
|
3433
|
+
err.summary,
|
|
3434
|
+
[err.source],
|
|
3435
|
+
)
|
|
3436
|
+
}
|
|
3437
|
+
|
|
3438
|
+
const decision = decideEgressUnion(documents, allowance)
|
|
3439
|
+
if (decision.refusal === undefined) return
|
|
3440
|
+
throw new KubernetesEgressPolicyUnionError(
|
|
3441
|
+
decision.refusal.kind,
|
|
3442
|
+
target.subject,
|
|
3443
|
+
target.podLabels,
|
|
3444
|
+
allowance.policyKind,
|
|
3445
|
+
decision.examined,
|
|
3446
|
+
decision.refusal.summary,
|
|
3447
|
+
)
|
|
437
3448
|
}
|