@push.rocks/smartnftables 4.0.0 → 4.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/changelog.md CHANGED
@@ -1,5 +1,21 @@
1
1
  # Changelog
2
2
 
3
+ ## 2026-10-01 - 4.1.0
4
+
5
+ ### Features
6
+
7
+ - Add optional exclusive forwarding to the schema-v2 `hostTransit` scope: `exclusiveForwarding: true` ends the scope's forward chain in one unconditional drop, so the host forwards exactly the scope's leased, published and symmetric handoff flows and their replies, and every other forwarded packet drops, IPv4 and IPv6, on every pair of links, including flows established before the member was applied. It is the host-wide forwarding default-deny that Docker's FORWARD policy DROP used to provide, for hosts where this table is the only forwarding owner: a drop in any table's forward chain is final, so it also denies forwarding other owners accept. With no `handoffs` the scope forwards nothing. The drop is created, verified, adopted, recovered, replaced and released with the table in one batch; this package sets no sysctl. Absent and `false` are the same canonical policy, digest and compiled bytes. `ManagedDockerForwarding.prepare()` refuses an exclusive barrier as `INVALID`, because its drop would deny Docker's own container forwarding.
8
+
9
+ ### Maintenance
10
+
11
+ - Release tooling currency: `packageManager` `pnpm@12.6.0`, `@git.zone/cli` 8.1.0, `@git.zone/tsrust` 2.0.1 and `@git.zone/tsrun` `^3.0.2`; `@git.zone/tsbuild` 5.0.0 and `@git.zone/tstest` 6.3.2 are already current, and `@types/node` stays at 26.6.2 because 26.6.3 is younger than seven days. tsrust 2 builds only the host architecture in a plain `tsrust`, so the new `build:release` script builds both configured musl targets with `tsrust --configured-targets` and checks the set with `tsrust verify`, and `release.preflight.buildCommand` runs it, so a release still ships the amd64 and arm64 binaries. The 36 lock entries recorded with SHA-1 integrity are restored to the SHA-512 integrity registry.npmjs.org publishes, after each SHA-1 matched the npmjs shasum.
12
+
13
+ ## 2026-09-28 - 4.0.1
14
+
15
+ ### Fixes
16
+
17
+ - Accept a `hostTransit` scope whose uplink addresses lie outside `protection.prefixes`. Every local address had to be inside the protected union, including the uplink lease and `snatAddress`, but the protected authority is cluster-wide and covers pools, resolvers, platform endpoints and the VPN, not a node's uplink lease, which is usually public; so every node with an uplink outside the plan was refused `INVALID` on each preparation. The host barrier now protects the uplink addresses itself: each one no protected prefix covers joins `protected_out` and `protected_back` as a `/32`, after the authority's prefixes, so a forwarded flow whose live or originally tracked destination is the host's uplink lease, or one sourced from it toward a handoff, is still denied, and the scope compiles exactly as if the authority had named that `/32`. Handoff addresses must still be protected, and a platform endpoint on an uplink address must still be protected and declared in `localPlatformEndpoints`. A scope whose uplink is covered keeps its exact digest and compiled bytes.
18
+
3
19
  ## 2026-09-28 - 4.0.0
4
20
 
5
21
  ### Breaking Changes
@@ -1,13 +1,13 @@
1
1
  {
2
2
  "format": "tsrust.build-provenance.v2",
3
- "binarySha256": "e879efc88997e38bb45c045fae690c3a07a3ff515c2a2581689466e3088ca236",
3
+ "binarySha256": "31265adcd734d687d19c98508332ec591f36afd9581c27a3c7b7fac647620b07",
4
4
  "buildInfo": {
5
5
  "projectName": "@push.rocks/smartnftables",
6
- "projectVersion": "4.0.0",
7
- "gitCommit": "7c66795f5e0edea87f2c9f7f1f70455cdf7e6bab",
6
+ "projectVersion": "4.1.0",
7
+ "gitCommit": "71cf6d5ee0d229b03b01f85901f5d50984b570a3",
8
8
  "gitDirty": false,
9
- "builtAt": "2026-09-28T03:45:45.694Z",
10
- "tsrustVersion": "1.15.1",
9
+ "builtAt": "2026-10-01T18:44:25.782Z",
10
+ "tsrustVersion": "2.0.1",
11
11
  "binary": "smartnftables",
12
12
  "target": "linux_amd64_musl"
13
13
  }
@@ -1,13 +1,13 @@
1
1
  {
2
2
  "format": "tsrust.build-provenance.v2",
3
- "binarySha256": "268d3371a7986e7f1efcd6fb7fdd02b04635060a3e845c0ef45baec75e16a32a",
3
+ "binarySha256": "a6d37a21727fed45c2a4079261930682fbb136b67ac1b24971015725b2b4ada5",
4
4
  "buildInfo": {
5
5
  "projectName": "@push.rocks/smartnftables",
6
- "projectVersion": "4.0.0",
7
- "gitCommit": "7c66795f5e0edea87f2c9f7f1f70455cdf7e6bab",
6
+ "projectVersion": "4.1.0",
7
+ "gitCommit": "71cf6d5ee0d229b03b01f85901f5d50984b570a3",
8
8
  "gitDirty": false,
9
- "builtAt": "2026-09-28T03:45:55.540Z",
10
- "tsrustVersion": "1.15.1",
9
+ "builtAt": "2026-10-01T18:44:34.745Z",
10
+ "tsrustVersion": "2.0.1",
11
11
  "binary": "smartnftables",
12
12
  "target": "linux_arm64_musl"
13
13
  }
@@ -3,7 +3,7 @@
3
3
  */
4
4
  export const commitinfo = {
5
5
  name: '@push.rocks/smartnftables',
6
- version: '4.0.0',
6
+ version: '4.1.0',
7
7
  description: 'A TypeScript module for managing nftables rules including NAT, firewall, and rate limiting with a high-level API.'
8
8
  };
9
9
  //# sourceMappingURL=data:application/json;base64,eyJ2ZXJzaW9uIjozLCJmaWxlIjoiMDBfY29tbWl0aW5mb19kYXRhLmpzIiwic291cmNlUm9vdCI6IiIsInNvdXJjZXMiOlsiLi4vdHMvMDBfY29tbWl0aW5mb19kYXRhLnRzIl0sIm5hbWVzIjpbXSwibWFwcGluZ3MiOiJBQUFBOztHQUVHO0FBQ0gsTUFBTSxDQUFDLE1BQU0sVUFBVSxHQUFHO0lBQ3hCLElBQUksRUFBRSwyQkFBMkI7SUFDakMsT0FBTyxFQUFFLE9BQU87SUFDaEIsV0FBVyxFQUFFLG1IQUFtSDtDQUNqSSxDQUFBIn0=
@@ -175,6 +175,8 @@ export interface IManagedNftHostTransitScopeV2 {
175
175
  link: IManagedNftLocalLinkV2;
176
176
  allocations: IManagedNftHandoffAllocationV2[];
177
177
  }>;
178
+ /** Its addresses need not lie in `protection.prefixes`; the host barrier protects each
179
+ * uncovered one as a `/32` destination. Handoff addresses must be protected. */
178
180
  uplink: IManagedNftLocalLinkV2;
179
181
  /** Exact current address present on uplink. No masquerade or default-route inference. */
180
182
  snatAddress: string;
@@ -194,6 +196,11 @@ export interface IManagedNftHostTransitScopeV2 {
194
196
  * leased source address and a source port of the allocation's range for the endpoint's protocol; its
195
197
  * ESTABLISHED replies return. Absent and empty are the same canonical policy, digest and compiled graph. */
196
198
  localPlatformEndpoints?: string[];
199
+ /** Optional: this table is the host's only forwarding owner. Every forwarded packet the scope does not
200
+ * admit (its leased, published and symmetric handoff flows) drops, IPv4 and IPv6, on every pair of
201
+ * interfaces, including forwarding that another table would accept. With no `handoffs` the host forwards
202
+ * nothing. It sets no sysctl. Absent and `false` are the same canonical policy, digest and compiled graph. */
203
+ exclusiveForwarding?: boolean;
197
204
  }
198
205
  /** One loopback TCP service that only one local user may dial. Every packet the host sends to the
199
206
  * exact address and port from a socket of any other user is rejected with a TCP reset, after ordinary
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@push.rocks/smartnftables",
3
- "version": "4.0.0",
3
+ "version": "4.1.0",
4
4
  "private": false,
5
5
  "description": "A TypeScript module for managing nftables rules including NAT, firewall, and rate limiting with a high-level API.",
6
6
  "main": "dist_ts/index.js",
@@ -9,10 +9,10 @@
9
9
  "author": "Task Venture Capital GmbH",
10
10
  "license": "MIT",
11
11
  "devDependencies": {
12
- "@git.zone/cli": "7.3.0",
12
+ "@git.zone/cli": "8.1.0",
13
13
  "@git.zone/tsbuild": "^5.0.0",
14
- "@git.zone/tsrun": "^3.0.1",
15
- "@git.zone/tsrust": "1.15.1",
14
+ "@git.zone/tsrun": "^3.0.2",
15
+ "@git.zone/tsrust": "2.0.1",
16
16
  "@git.zone/tstest": "6.3.2",
17
17
  "@types/node": "26.6.2",
18
18
  "typescript": "^7.0.2"
@@ -66,6 +66,7 @@
66
66
  "scripts": {
67
67
  "test": "(tstest test/**/test*.ts --verbose --timeout 60)",
68
68
  "build": "(tsbuild tsfolders) && (tsrust)",
69
+ "build:release": "(tsbuild tsfolders) && (tsrust --configured-targets) && (tsrust verify)",
69
70
  "format": "(gitzone format)",
70
71
  "buildDocs": "tsdoc"
71
72
  }
package/readme.md CHANGED
@@ -321,7 +321,7 @@ allocation-pool guard policy kinds.
321
321
  | V2 scope | Required authority and behavior |
322
322
  | --- | --- |
323
323
  | `routerEgress` | Private `endpoints` and `rules`, one exact `links` binding per endpoint, a separate veth `handoff`, `protection`, and active `generations`. Private veth/TUN/local DNS and egress share one table so terminal private denial cannot override a separate egress table. Optional `publishedPorts` add the inbound second hop from the handoff to a workload endpoint, one port or a port range; both directions are classified into the default conntrack zone ahead of every leased classifier, so a published endpoint port is dedicated to its publication and never becomes leased egress. A `symmetric` publication also lets the workload open flows from its published ports. Optional `hostGrants` forward exact host-origin flows from the handoff to a workload endpoint. Optional `workloadGrants` let one workload open one exact port of another, one way. |
324
- | `hostTransit` | Exact handoff `link`/`allocations` pairs, complete `protection`, an explicit veth or Ethernet `uplink`, and its current `snatAddress`. It checks each handoff's leased source address and protocol/port range, default conntrack zone, direction, uplink, and protected destinations before outer SNAT. Optional `publishedPorts` add inbound uplink destination NAT inside the same generation, one port or a port range, and a `symmetric` publication also carries the workload's own flows from its published ports out through the uplink. Optional `hostGrants` let the host's own address on a handoff dial exact workload ports. Optional `localPlatformEndpoints` serve platform endpoints on the host's own addresses to leased flows. |
324
+ | `hostTransit` | Exact handoff `link`/`allocations` pairs, complete `protection`, an explicit veth or Ethernet `uplink`, and its current `snatAddress`. It checks each handoff's leased source address and protocol/port range, default conntrack zone, direction, uplink, and protected destinations before outer SNAT. Optional `publishedPorts` add inbound uplink destination NAT inside the same generation, one port or a port range, and a `symmetric` publication also carries the workload's own flows from its published ports out through the uplink. Optional `hostGrants` let the host's own address on a handoff dial exact workload ports. Optional `localPlatformEndpoints` serve platform endpoints on the host's own addresses to leased flows. Optional `exclusiveForwarding` makes the table the host's only forwarding owner: every other forwarded packet drops. |
325
325
  | `allocationPoolGuard` | An authenticated `authorityDigest` and complete current allocation-pool `prefixes`. Installs host-wide IPv4 destination denial before any handoff exists, without link, uplink or SNAT dependencies. Optional `hostGrants` are its only exceptions. Optional `localTcpPortOwners` restrict loopback TCP ports to one local user each. |
326
326
 
327
327
  `allocationPoolGuard` accepts 1–64 canonical, disjoint RFC1918 prefixes. Supply
@@ -333,7 +333,9 @@ source, then return; all remaining packets check their current destination. This
333
333
  denies DNAT into or away from a pool and untracked current-destination traffic,
334
334
  while permitting reverse-SNAT replies to authorized transit sources. It grants no
335
335
  egress permissions; exact `hostTransit` and router policies remain separate owners.
336
- Unrelated host traffic and IPv6 retain their existing behavior.
336
+ Unrelated host traffic and IPv6 retain their existing behavior; the guard never denies
337
+ forwarding between other links. On a host that forwards only for the handoffs, see
338
+ [Exclusive forwarding](#exclusive-forwarding).
337
339
 
338
340
  For example, a guard policy is `{ schemaVersion: 2, revision: 1, scope: {
339
341
  kind: 'allocationPoolGuard', authorityDigest, prefixes: ['10.240.0.0/16',
@@ -369,6 +371,15 @@ so foreign DNAT cannot turn a protected destination into a public exception or
369
371
  redirect public traffic into protected space. INPUT diversion, local OUTPUT into
370
372
  handoffs, unmatched handoff traffic and IPv6 forwarding are denied.
371
373
 
374
+ Every handoff address and platform endpoint lies in the protected prefixes. The
375
+ `hostTransit` uplink addresses need not: the authority is the platform's address
376
+ space, and a node's uplink lease and `snatAddress` is usually a public address
377
+ outside it. The host barrier protects the uplink addresses itself, adding each one
378
+ no protected prefix covers as a `/32` after the authority's prefixes, so a
379
+ forwarded flow whose current or original destination is the host's uplink address,
380
+ or one sourced from it toward a handoff, is denied like any protected destination.
381
+ A scope whose uplink the prefixes already cover compiles exactly as before.
382
+
372
383
  #### Published host ports
373
384
 
374
385
  `hostTransit.publishedPorts` is optional. Absent and empty are the same canonical
@@ -494,7 +505,8 @@ the ESTABLISHED original direction, a NEW opening per protocol with opening TCP
494
505
  restricted to SYN with FIN/RST/ACK clear, and the ESTABLISHED reply. The handoff
495
506
  ingress classification precedes the
496
507
  barrier's protected-source denial, because the uplink or management network that
497
- reaches a published port is itself protected space. The endpoint ingress
508
+ reaches a published port can itself be protected space (an uplink address outside
509
+ the protected prefixes is still guarded as a /32). The endpoint ingress
498
510
  classification matches the exact workload link, address and published endpoint
499
511
  port, and precedes every leased classifier of every generation: raw runs before
500
512
  conntrack, so a client whose source port equals a leased grant's destination port
@@ -642,6 +654,61 @@ rules. Absent and empty are the same canonical policy, digest and compiled bytes
642
654
  caller keeps a declared endpoint's address outside every guarded allocation pool and
643
655
  owns the listener.
644
656
 
657
+ #### Exclusive forwarding
658
+
659
+ A host that forwards IPv4 (`net.ipv4.conf.all.forwarding=1`) routes between all of its
660
+ links. On a Docker host, Docker set that switch and also the iptables FORWARD policy
661
+ DROP, and `ManagedDockerForwarding` admits the handoff flows through it. On a host
662
+ without Docker nothing denies forwarding that does not touch a handoff: host transit
663
+ governs only handoff traffic, and the allocation-pool guard only pool destinations,
664
+ so a LAN peer could use the host as its gateway, the uplink could hairpin, and two
665
+ other links could exchange traffic. `hostTransit.exclusiveForwarding: true` closes
666
+ that path inside the barrier's own table:
667
+
668
+ ```typescript
669
+ const transit: IManagedNftPolicyV2 = { schemaVersion: 2, revision, scope: {
670
+ kind: 'hostTransit', protection, handoffs, uplink, snatAddress, exclusiveForwarding: true } };
671
+ ```
672
+
673
+ The forward chain ends in one unconditional drop, after every rule of the scope.
674
+ The host then forwards exactly what the scope admits: the leased flows of each
675
+ handoff to the uplink with their ESTABLISHED replies, the inbound publications and
676
+ their replies, and the symmetric publications' flows. Everything else that reaches
677
+ the forward hook drops, whatever its links: IPv4 between the uplink, a LAN or any
678
+ other link, a hairpin through the uplink, flows that were established before the
679
+ member was applied, and forwarded IPv6. There is no ESTABLISHED/RELATED bypass, as
680
+ everywhere in this scope, so ICMP errors for leased flows stay denied as before.
681
+ Allocation pools need no exception: pool addresses are routed inside the router
682
+ namespace and leave it translated to the transit source, so on the host only
683
+ transit addresses cross the forward hook, and the pool guard denies pool
684
+ destinations there anyway. IPv6 is dropped because the scope's packet model is
685
+ IPv4 only; a host whose IPv6 forwarding is off never presents IPv6 to the hook.
686
+ Input, output and NAT are unchanged, and so are the host's own flows: host grants
687
+ and host-local platform endpoints are INPUT and OUTPUT traffic. With no `handoffs`
688
+ the scope forwards nothing at all.
689
+
690
+ nftables runs the forward base chain of every table, and a drop in any one of them
691
+ is final, while an accept only ends its own chain. The member therefore drops
692
+ forwarding that another owner (Docker, a container or VM bridge, a VPN router, a
693
+ second routing daemon) accepts, and accepting in this table still cannot override
694
+ another owner's drop. Use it only on a host where this table is the sole forwarding
695
+ owner; on a Docker host keep it absent and use the Docker contribution below, which
696
+ refuses an exclusive barrier. The
697
+ member does not fence flowtable offload, packet queues, proxies or anything outside
698
+ the forward hook.
699
+
700
+ The drop is part of the scope's graph: it is created, verified, adopted, recovered,
701
+ replaced and released with the rest of the table in one batch, and a replacement
702
+ without the member or `release()` removes it, so forwarding resumes for every link
703
+ the moment the table goes. This package sets no sysctl. The denial holds only while
704
+ an exclusive table is applied, so the caller enables host forwarding only after an
705
+ exclusive scope is enforced and disables it before releasing that scope, or keeps
706
+ an exclusive scope applied between generations (with no handoffs it is a pure
707
+ host-wide forwarding denial) and replaces it in place. Persistent forwarding in
708
+ `sysctl.d` would make the host forward unfiltered after every boot until the scope
709
+ is applied again, so leave the boot default off. Absent and `false` are the same
710
+ canonical policy, digest and compiled bytes; the member adds one rule.
711
+
645
712
  #### Workload grants
646
713
 
647
714
  `routerEgress.workloadGrants` is optional: a stateful one-way flow between two
@@ -845,7 +912,9 @@ an explicit `/32` has different hidden match bytes in these frontends. A
845
912
  semantically equal rewrite with a different exact representation rejects, as does
846
913
  an early ACCEPT that the frontend reconstructs as the same command.
847
914
  The contribution never creates, flushes, adopts or deletes a Docker table/chain,
848
- and never changes a host forwarding policy.
915
+ and never changes a host forwarding policy. `prepare()` refuses a barrier with
916
+ `exclusiveForwarding` as `INVALID`: its drop would also deny Docker's own container
917
+ forwarding.
849
918
 
850
919
  Each contributed rule binds the handoff and uplink names, current and original
851
920
  transit source, exact leased TCP/UDP port range, packet direction and connection
@@ -960,6 +1029,14 @@ another destination workload, the destination's TCP and UDP openings toward the
960
1029
  source stay dark; withdrawal stops new flows and the established one; release reopens
961
1030
  the path. A set of 1024 grants persists across owner loss, is re-verified element by
962
1031
  element and survives lost-ACK replay.
1032
+ Exclusive forwarding is qualified on a host namespace with a handoff, an uplink and
1033
+ a LAN link: against positive controls with no policy and with the scope without the
1034
+ member, the member denies the LAN's IPv4 to the uplink peer, the peer's way back into
1035
+ the LAN, the LAN's IPv6 and a LAN flow established before it, while leased egress
1036
+ still leaves translated and an unleased port stays denied; without handoffs leased
1037
+ egress drops too; replacing it with a scope without the member and releasing an
1038
+ exclusive scope both reopen the LAN path. The same probe fails against a compiler
1039
+ without the drop.
963
1040
  Loopback TCP port owners are qualified against pre-policy positive controls on the
964
1041
  same listeners: the owning uid (root, and uid 1000 for a second port) connects, every
965
1042
  other uid is reset with `ECONNREFUSED`, an unowned port and another loopback address
@@ -63,7 +63,11 @@ impl Policy {
63
63
  return Err(Error::Invalid);
64
64
  }
65
65
  self.barrier.prepared.validate()?;
66
- self.scope()?;
66
+ // The contribution admits handoff flows through Docker's FORWARD path; an
67
+ // exclusive barrier would drop Docker's own container forwarding.
68
+ if self.scope()?.exclusive_forwarding {
69
+ return Err(Error::Invalid);
70
+ }
67
71
  let receipt = &self.barrier.receipt;
68
72
  let handle = receipt.table_handle.parse::<u64>().map_err(|_| Error::Invalid)?;
69
73
  if handle == 0 || handle.to_string() != receipt.table_handle
@@ -30,6 +30,9 @@ fn docker_prepare_rejects_router_receipts_bad_hashes_and_unbounded_restore() {
30
30
  let mut value=policy();value.schema_version=2;assert!(value.prepare("v1.8.11 (nf_tables)".into()).is_err());
31
31
  let mut value=policy();value.barrier.receipt.digest="sha256:wrong".into();assert!(value.prepare("v1.8.11 (nf_tables)".into()).is_err());
32
32
  let mut value=policy();value.barrier.prepared.policy=crate::managed::Policy::Egress(serde_json::from_value(crate::egress::tests::router()).unwrap());assert!(value.prepare("v1.8.11 (nf_tables)".into()).is_err());
33
+ let mut exclusive=crate::egress::tests::host();exclusive["scope"]["exclusiveForwarding"]=json!(true);
34
+ let prepared:crate::managed::Prepared=serde_json::from_value::<crate::egress::Policy>(exclusive).unwrap().prepare().unwrap().into();
35
+ let mut value=policy();value.barrier.receipt.digest=prepared.digest.clone();value.barrier.prepared=prepared;assert!(value.prepare("v1.8.11 (nf_tables)".into()).is_err());
33
36
  let value=serde_json::to_value(policy()).unwrap();
34
37
  let mut extra=value.clone();extra["rules"]=json!([]);assert!(serde_json::from_value::<policy::Policy>(extra).is_err());
35
38
  let mut extra=value;extra["barrier"]["receipt"]["tableHandle"]=json!("01");assert!(serde_json::from_value::<policy::Policy>(extra).unwrap().prepare("v1.8.11 (nf_tables)".into()).is_err());
@@ -1,5 +1,22 @@
1
1
  use super::*;
2
2
 
3
+ /// The host's own uplink addresses are protected destinations of this scope. A
4
+ /// forwarded packet never reaches them, so a live or originally tracked tuple
5
+ /// holding one is a translation into the host, and a forwarded one sourced from
6
+ /// one is a spoof. The protected authority is cluster-wide and need not cover a
7
+ /// node's uplink lease, which is usually public, so every uplink address outside
8
+ /// it joins the barrier as a /32, in the uplink's canonical address order. An address the authority
9
+ /// covers meets its prefix already, so a covered uplink compiles exactly as
10
+ /// before.
11
+ fn uplink_barrier(scope: &HostScope) -> Result<Vec<String>> {
12
+ let mut barrier = Vec::new();
13
+ for ip in &scope.uplink.required_ipv4_addresses {
14
+ if !covers_address(&scope.protection.prefixes, ip)? {
15
+ barrier.push(format!("{ip}/32"));
16
+ }
17
+ }
18
+ Ok(barrier)
19
+ }
3
20
  fn protection(program: &mut Program<'_>, scope: &HostScope, reply: bool) -> Result<()> {
4
21
  let chain = if reply {
5
22
  "protected_back"
@@ -17,7 +34,8 @@ fn protection(program: &mut Program<'_>, scope: &HostScope, reply: bool) -> Resu
17
34
  expressions.extend(original_port(endpoint.port, false));
18
35
  program.end(chain, expressions, -5)?;
19
36
  }
20
- for prefix in &scope.protection.prefixes {
37
+ let barrier = uplink_barrier(scope)?;
38
+ for prefix in scope.protection.prefixes.iter().chain(&barrier) {
21
39
  for original in [false, true] {
22
40
  let mut expressions = ipv4();
23
41
  expressions.extend(if original {
@@ -254,5 +272,14 @@ pub(super) fn compile(program: &mut Program<'_>, scope: &HostScope) -> Result<()
254
272
  program.deny_link("forward", &handoff.link, true)?;
255
273
  program.deny_link("forward", &handoff.link, false)?;
256
274
  }
275
+ // Exclusive forwarding: everything the forward chain has not accepted by now
276
+ // drops, of every family and on every pair of interfaces, so the host
277
+ // forwards exactly this scope's leased, published and symmetric handoff
278
+ // flows. nftables runs every table's forward base chain, and a drop in any
279
+ // one is final: this denies forwarding that other owners would accept, and
280
+ // accepting here still cannot override their drops.
281
+ if scope.exclusive_forwarding {
282
+ program.end("forward", vec![], 0)?;
283
+ }
257
284
  Ok(())
258
285
  }
@@ -210,6 +210,11 @@ pub struct HostScope {
210
210
  /// bound links. Absent and empty are the same canonical policy.
211
211
  #[serde(default, skip_serializing_if = "Vec::is_empty")]
212
212
  pub local_platform_endpoints: Vec<String>,
213
+ /// This table is the host's only forwarding owner: every forwarded packet,
214
+ /// IPv4 or IPv6, that the scope does not admit drops, instead of passing to
215
+ /// other owners. Absent and false are the same canonical policy.
216
+ #[serde(default, skip_serializing_if = "std::ops::Not::not")]
217
+ pub exclusive_forwarding: bool,
213
218
  }
214
219
  /// One loopback TCP service that only one local user may dial: packets the host
215
220
  /// sends to the exact address and port from a socket of any other user reject,
@@ -1102,13 +1107,14 @@ fn normalize_host(value: &mut HostScope) -> Result<()> {
1102
1107
  .ok_or(Error::Invalid)?;
1103
1108
  require(local_addresses.contains(&endpoint.address))?;
1104
1109
  }
1110
+ // Handoff addresses lie in protected space, checked per handoff below. An
1111
+ // uplink address need not: the uplink lease and SNAT address is the host's
1112
+ // public side, usually outside the protected authority, and the compiler
1113
+ // adds each uncovered one to the host barrier as a protected destination.
1105
1114
  for ip in &local_addresses {
1106
- require(
1107
- covers_address(&value.protection.prefixes, ip)?
1108
- && value.protection.platform_endpoints.iter().all(|endpoint| {
1109
- endpoint.address != *ip || value.local_platform_endpoints.contains(&endpoint.id)
1110
- }),
1111
- )?;
1115
+ require(value.protection.platform_endpoints.iter().all(|endpoint| {
1116
+ endpoint.address != *ip || value.local_platform_endpoints.contains(&endpoint.id)
1117
+ }))?;
1112
1118
  }
1113
1119
  for handoff in &mut value.handoffs {
1114
1120
  normalize_link(&mut handoff.link)?;
@@ -1167,6 +1173,9 @@ fn normalize_host(value: &mut HostScope) -> Result<()> {
1167
1173
  )
1168
1174
  }
1169
1175
 
1176
+ #[cfg(test)]
1177
+ #[path = "egress_exclusiveforwarding_tests.rs"]
1178
+ mod exclusiveforwarding_tests;
1170
1179
  #[cfg(test)]
1171
1180
  #[path = "egress_hostgrant_tests.rs"]
1172
1181
  pub(crate) mod hostgrant_tests;
@@ -1186,6 +1195,9 @@ mod routerequivalence_tests;
1186
1195
  #[path = "egress_tests.rs"]
1187
1196
  pub(crate) mod tests;
1188
1197
  #[cfg(test)]
1198
+ #[path = "egress_uplink_tests.rs"]
1199
+ mod uplink_tests;
1200
+ #[cfg(test)]
1189
1201
  #[path = "egress_workloadgrant_tests.rs"]
1190
1202
  pub(crate) mod workloadgrant_tests;
1191
1203
  impl Policy {
@@ -0,0 +1,256 @@
1
+ //! Exclusive forwarding in host transit: the host forwards exactly the scope's
2
+ //! leased, published and symmetric handoff flows, and every other forwarded
3
+ //! packet drops instead of passing to other owners.
4
+ use super::hostgrant_tests::{
5
+ bytes, chains, decide, normalized, opening, prepared, Flow, ACK, ESTABLISHED,
6
+ };
7
+ use super::tests::{host, pool_guard, published_host, published_port, router};
8
+ use super::Prepared;
9
+ use crate::policy::verdict;
10
+ use serde_json::{json, Value};
11
+
12
+ const TRANSIT: [u8; 4] = [10, 240, 0, 2];
13
+ const UPLINK: [u8; 4] = [192, 0, 2, 2];
14
+ const PUBLIC: [u8; 4] = [198, 51, 100, 200];
15
+ const CLIENT: [u8; 4] = [198, 51, 100, 9];
16
+ const LAN_PEER: [u8; 4] = [172, 16, 0, 9];
17
+ const HANDOFF: Option<(u32, &str)> = Some((4, "handoff"));
18
+ const UPLINK_LINK: Option<(u32, &str)> = Some((2, "ens18"));
19
+ /// Another host link that no scope binds: a management LAN, a second uplink.
20
+ const LAN: Option<(u32, &str)> = Some((7, "lan0"));
21
+
22
+ fn exclusive(mut value: Value) -> Value {
23
+ value["scope"]["exclusiveForwarding"] = json!(true);
24
+ value
25
+ }
26
+ /// The fixture with one symmetric UDP publication of the transit address.
27
+ fn published() -> Value {
28
+ let mut port = published_port("udp", 5060, 5060);
29
+ port["symmetric"] = json!(true);
30
+ published_host(json!([published_port("tcp", 443, 8443), port]))
31
+ }
32
+ /// One forwarded packet; the originally tracked tuple equals the live one.
33
+ fn forwarded(
34
+ input: Option<(u32, &'static str)>,
35
+ output: Option<(u32, &'static str)>,
36
+ (source, source_port): ([u8; 4], u16),
37
+ (destination, destination_port): ([u8; 4], u16),
38
+ ) -> Flow {
39
+ Flow {
40
+ input,
41
+ output,
42
+ source,
43
+ destination,
44
+ source_port,
45
+ destination_port,
46
+ original: (source, destination, source_port, destination_port),
47
+ ..opening(input, output)
48
+ }
49
+ }
50
+ /// The answer to `flow` on the reverse links.
51
+ fn answer(flow: Flow) -> Flow {
52
+ Flow {
53
+ input: flow.output,
54
+ output: flow.input,
55
+ source: flow.destination,
56
+ destination: flow.source,
57
+ source_port: flow.destination_port,
58
+ destination_port: flow.source_port,
59
+ tcp_flags: ACK,
60
+ state: ESTABLISHED,
61
+ reply: true,
62
+ ..flow
63
+ }
64
+ }
65
+ /// Forwarded traffic no scope admits: the host as a gateway between its other
66
+ /// links and the uplink, a hairpin back out of the uplink, and traffic between
67
+ /// two links the scope does not bind, in both directions and both protocols.
68
+ fn unadmitted() -> Vec<Flow> {
69
+ let mut result = Vec::new();
70
+ for flow in [
71
+ forwarded(LAN, UPLINK_LINK, (LAN_PEER, 40000), (PUBLIC, 443)),
72
+ forwarded(UPLINK_LINK, LAN, (CLIENT, 40000), (LAN_PEER, 22)),
73
+ forwarded(UPLINK_LINK, UPLINK_LINK, (CLIENT, 40000), (PUBLIC, 443)),
74
+ forwarded(
75
+ LAN,
76
+ Some((8, "lan1")),
77
+ (LAN_PEER, 40000),
78
+ ([172, 17, 0, 9], 80),
79
+ ),
80
+ ] {
81
+ result.extend([
82
+ flow,
83
+ answer(flow),
84
+ Flow {
85
+ protocol: 17,
86
+ ..flow
87
+ },
88
+ ]);
89
+ }
90
+ result
91
+ }
92
+ /// Every flow the scope forwards: leased TCP and UDP egress with replies,
93
+ /// published inbound TCP and UDP with replies, and symmetric UDP outbound.
94
+ fn admitted() -> Vec<Flow> {
95
+ let leased = forwarded(HANDOFF, UPLINK_LINK, (TRANSIT, 10500), (PUBLIC, 443));
96
+ let inbound = Flow {
97
+ original: (CLIENT, UPLINK, 40000, 443),
98
+ ..forwarded(UPLINK_LINK, HANDOFF, (CLIENT, 40000), (TRANSIT, 8443))
99
+ };
100
+ let signalling = Flow {
101
+ protocol: 17,
102
+ original: (CLIENT, UPLINK, 40001, 5060),
103
+ ..forwarded(UPLINK_LINK, HANDOFF, (CLIENT, 40001), (TRANSIT, 5060))
104
+ };
105
+ let symmetric = Flow {
106
+ protocol: 17,
107
+ ..forwarded(HANDOFF, UPLINK_LINK, (TRANSIT, 5060), (CLIENT, 5060))
108
+ };
109
+ let mut result = Vec::new();
110
+ for flow in [
111
+ leased,
112
+ Flow {
113
+ protocol: 17,
114
+ ..leased
115
+ },
116
+ inbound,
117
+ signalling,
118
+ symmetric,
119
+ ] {
120
+ result.extend([
121
+ flow,
122
+ Flow {
123
+ state: ESTABLISHED,
124
+ tcp_flags: ACK,
125
+ ..flow
126
+ },
127
+ answer(flow),
128
+ ]);
129
+ }
130
+ result
131
+ }
132
+
133
+ #[test]
134
+ fn exclusive_forwarding_absent_and_false_keep_the_host_digest_and_compiled_bytes() {
135
+ let baseline = prepared(host());
136
+ assert!(!serde_json::to_string(&baseline.policy)
137
+ .unwrap()
138
+ .contains("exclusiveForwarding"));
139
+ let mut value = host();
140
+ value["scope"]["exclusiveForwarding"] = json!(false);
141
+ let explicit = prepared(value);
142
+ assert_eq!(explicit, baseline);
143
+ assert_eq!(bytes(&explicit), bytes(&baseline));
144
+ let with = prepared(exclusive(host()));
145
+ assert_ne!(with.digest, baseline.digest);
146
+ assert!(serde_json::to_string(&with.policy)
147
+ .unwrap()
148
+ .contains("\"exclusiveForwarding\":true"));
149
+ }
150
+
151
+ #[test]
152
+ fn exclusive_forwarding_appends_one_unconditional_forward_drop_and_nothing_else() {
153
+ for fixture in [host(), published()] {
154
+ let baseline = chains(&prepared(fixture.clone()));
155
+ let with = chains(&prepared(exclusive(fixture)));
156
+ assert_eq!(
157
+ with.keys().collect::<Vec<_>>(),
158
+ baseline.keys().collect::<Vec<_>>()
159
+ );
160
+ for (name, rules) in &baseline {
161
+ if name != "forward" {
162
+ assert_eq!(&with[name], rules, "{name}");
163
+ }
164
+ }
165
+ // The previous forward rules keep their order and the drop follows them.
166
+ // It has no family or interface match: forwarded IPv6 and every link
167
+ // pair drop like forwarded IPv4.
168
+ let forward = &with["forward"];
169
+ assert_eq!(&forward[..forward.len() - 1], &baseline["forward"][..]);
170
+ assert_eq!(forward.last().unwrap(), &vec![verdict(0, None)]);
171
+ }
172
+ }
173
+
174
+ #[test]
175
+ fn exclusive_forwarding_denies_every_unadmitted_forward_and_keeps_every_admitted_one() {
176
+ for fixture in [host(), published()] {
177
+ let baseline = prepared(fixture.clone());
178
+ let with = prepared(exclusive(fixture));
179
+ // Without the member another owner's forwarding passes this table.
180
+ for (index, flow) in unadmitted().iter().enumerate() {
181
+ let decision = decide(&baseline, "forward", flow);
182
+ assert_eq!(
183
+ (decision.verdict, decision.chain.as_str()),
184
+ (1, "policy"),
185
+ "baseline {index}"
186
+ );
187
+ let decision = decide(&with, "forward", flow);
188
+ assert_eq!(
189
+ (decision.verdict, decision.chain.as_str()),
190
+ (0, "forward"),
191
+ "exclusive {index}"
192
+ );
193
+ }
194
+ // Every flow the scope forwards keeps its exact decision, and so does
195
+ // every handoff flow it already denies.
196
+ let handoff = [
197
+ forwarded(HANDOFF, LAN, (TRANSIT, 10500), (LAN_PEER, 443)),
198
+ forwarded(LAN, HANDOFF, (LAN_PEER, 40000), (TRANSIT, 8443)),
199
+ forwarded(HANDOFF, UPLINK_LINK, (TRANSIT, 9999), (PUBLIC, 443)),
200
+ forwarded(
201
+ HANDOFF,
202
+ UPLINK_LINK,
203
+ (TRANSIT, 10500),
204
+ ([10, 250, 0, 2], 443),
205
+ ),
206
+ ];
207
+ for (index, flow) in admitted().iter().chain(&handoff).enumerate() {
208
+ assert_eq!(
209
+ decide(&with, "forward", flow),
210
+ decide(&baseline, "forward", flow),
211
+ "flow {index}"
212
+ );
213
+ }
214
+ }
215
+ // The positive controls are real admissions of the published fixture.
216
+ let published = prepared(exclusive(published()));
217
+ for (index, flow) in admitted().iter().enumerate() {
218
+ assert_eq!(
219
+ decide(&published, "forward", flow).verdict,
220
+ 1,
221
+ "admitted {index}"
222
+ );
223
+ }
224
+ }
225
+
226
+ #[test]
227
+ fn exclusive_forwarding_without_handoffs_forwards_nothing() {
228
+ let mut value = exclusive(host());
229
+ value["scope"]["handoffs"] = json!([]);
230
+ let empty = prepared(value);
231
+ assert_eq!(chains(&empty)["forward"], vec![vec![verdict(0, None)]]);
232
+ let leased = forwarded(HANDOFF, UPLINK_LINK, (TRANSIT, 10500), (PUBLIC, 443));
233
+ for flow in unadmitted().into_iter().chain([leased, answer(leased)]) {
234
+ assert_eq!(decide(&empty, "forward", &flow).verdict, 0);
235
+ }
236
+ // Without the member an empty scope forwards everything, as before.
237
+ let mut value = host();
238
+ value["scope"]["handoffs"] = json!([]);
239
+ let open = prepared(value);
240
+ assert!(chains(&open)["forward"].is_empty());
241
+ assert_eq!(decide(&open, "forward", &leased).verdict, 1);
242
+ }
243
+
244
+ #[test]
245
+ fn exclusive_forwarding_is_a_host_transit_boolean_only() {
246
+ for fixture in [router(), pool_guard()] {
247
+ assert!(normalized(exclusive(fixture)).is_err());
248
+ }
249
+ for invalid in [json!(1), json!("true"), json!(null)] {
250
+ let mut value = host();
251
+ value["scope"]["exclusiveForwarding"] = invalid;
252
+ assert!(normalized(value).is_err());
253
+ }
254
+ let restored: Prepared = prepared(exclusive(host()));
255
+ assert_eq!(restored.clone().policy.prepare().unwrap(), restored);
256
+ }
@@ -0,0 +1,159 @@
1
+ //! Host transit on an uplink whose own addresses lie outside the protected
2
+ //! authority: the cluster-wide authority covers pools, resolvers, platform
3
+ //! endpoints and the VPN, not a node's uplink lease, which is usually public.
4
+ //! The scope prepares, and the host barrier protects the uplink addresses itself.
5
+ use super::hostgrant_tests::{
6
+ chains, decide, normalized, opening, prepared, Flow, ACK, ESTABLISHED,
7
+ };
8
+ use super::tests::host;
9
+ use serde_json::{json, Value};
10
+
11
+ const TRANSIT: [u8; 4] = [10, 240, 0, 2];
12
+ const UPLINK: [u8; 4] = [198, 51, 100, 7];
13
+ const PUBLIC: [u8; 4] = [198, 51, 100, 200];
14
+ const HANDOFF_LINK: Option<(u32, &str)> = Some((4, "handoff"));
15
+ const UPLINK_LINK: Option<(u32, &str)> = Some((2, "ens18"));
16
+
17
+ /// The `preparePolicy` request Pallet 35.1.0 sent pallet-guard on a lab node,
18
+ /// captured on the native IPC (lab-only private addresses). Its uplink lease and
19
+ /// SNAT address 10.92.26.30 lies outside the protected prefix 10.240.0.0/12; the
20
+ /// two other uplink addresses hold host-local platform endpoints inside it.
21
+ fn captured() -> Value {
22
+ json!({"schemaVersion":2,"revision":1,"scope":{"kind":"hostTransit",
23
+ "protection":{
24
+ "authorityDigest":"sha256:f436a15c8c8f4d4f0f23bc9fc69f45a8c9137a9586db1d2c2be8c09dac3467c9",
25
+ "prefixes":["10.240.0.0/12"],
26
+ "platformEndpoints":[
27
+ {"id":"5aaa706266a3c4e9b0cc99eb354496a732f88cdf3d36fd5911ab6fbb98ef7489","address":"10.244.0.30","port":8443,"protocol":"udp"},
28
+ {"id":"0298aa0b89b4248e058b7a279abc9d9c1793d2d012752432e95881367b43d6ac","address":"10.245.0.30","port":27017,"protocol":"tcp"},
29
+ {"id":"f0808a59e876a51c1a8a5b8cf38e0fc323dd5309dec3a8890468a1afc69dc9ff","address":"10.245.0.30","port":9000,"protocol":"tcp"},
30
+ {"id":"f4744ddb8e7d63883a7fbd4635f3ee4a031650effaaf5c13a1c664182b2b0501","address":"10.245.0.30","port":8443,"protocol":"tcp"},
31
+ {"id":"8933bdd0997dc2b230b0be7960c1708a57d72a0fa9481ec4aa56b9311ae68751","address":"10.243.0.1","port":53,"protocol":"tcp"},
32
+ {"id":"6e88da7a803f69d77d8d8fd86c5952ae87c26c3172b2f9a751662793e0ee8a7e","address":"10.243.0.1","port":53,"protocol":"udp"}]},
33
+ "handoffs":[{
34
+ "link":{"interfaceIndex":11,"interfaceName":"phc05f15afde9d","interfaceKind":"veth",
35
+ "macAddress":"4a:d5:5b:95:91:fa","interfaceLinkIndex":2,"requiredIpv4Addresses":["10.240.0.0"]},
36
+ "allocations":[{
37
+ "lease":{"id":"049eda6820b222e7e371bb4d1e24a325ac92422cff1900443c1e74a9efbb5d5e","generation":1,
38
+ "digest":"sha256:9a5635b213263796ac3b883cce942a0ca45161d9412ec6c6c9a5d748c22a3a7c"},
39
+ "transitSourceAddress":"10.240.0.1",
40
+ "sourcePortRanges":[{"protocol":"tcp","first":1024,"last":65535},{"protocol":"udp","first":1024,"last":65535}]}]}],
41
+ "uplink":{"interfaceIndex":2,"interfaceName":"enp1s0","interfaceKind":"ethernet",
42
+ "macAddress":"52:54:00:92:26:30","interfaceLinkIndex":2,
43
+ "requiredIpv4Addresses":["10.244.0.30","10.245.0.30","10.92.26.30"]},
44
+ "snatAddress":"10.92.26.30",
45
+ "localPlatformEndpoints":[
46
+ "0298aa0b89b4248e058b7a279abc9d9c1793d2d012752432e95881367b43d6ac",
47
+ "5aaa706266a3c4e9b0cc99eb354496a732f88cdf3d36fd5911ab6fbb98ef7489",
48
+ "f0808a59e876a51c1a8a5b8cf38e0fc323dd5309dec3a8890468a1afc69dc9ff",
49
+ "f4744ddb8e7d63883a7fbd4635f3ee4a031650effaaf5c13a1c664182b2b0501"]}})
50
+ }
51
+ /// The host fixture on a public uplink lease outside every protected prefix.
52
+ fn public() -> Value {
53
+ let mut value = host();
54
+ value["scope"]["uplink"]["requiredIpv4Addresses"] = json!(["198.51.100.7"]);
55
+ value["scope"]["snatAddress"] = json!("198.51.100.7");
56
+ value
57
+ }
58
+ /// A leased opening from the handoff out through the uplink, to `destination`
59
+ /// in the live tuple and `original` in the originally tracked one.
60
+ fn egress(destination: [u8; 4], original: [u8; 4]) -> Flow {
61
+ Flow {
62
+ source: TRANSIT,
63
+ destination,
64
+ source_port: 10500,
65
+ destination_port: 443,
66
+ original: (TRANSIT, original, 10500, 443),
67
+ ..opening(HANDOFF_LINK, UPLINK_LINK)
68
+ }
69
+ }
70
+
71
+ #[test]
72
+ fn host_transit_prepares_the_captured_lab_request_with_an_unprotected_uplink_lease() {
73
+ // Refused INVALID by 4.0.0: the uplink lease was required to be protected.
74
+ let lab = prepared(captured());
75
+ // The lease joins the barrier as a /32, exactly as if the authority had
76
+ // named it; the covered uplink addresses meet their prefix already.
77
+ let mut value = captured();
78
+ value["scope"]["protection"]["prefixes"] = json!(["10.240.0.0/12", "10.92.26.30/32"]);
79
+ assert_eq!(chains(&lab), chains(&prepared(value)));
80
+ }
81
+
82
+ #[test]
83
+ fn host_transit_uplink_addresses_need_no_protected_prefix_but_handoff_addresses_do() {
84
+ assert!(normalized(public()).is_ok());
85
+ // A handoff address outside the authority still refuses; the control
86
+ // covering it prepares.
87
+ let handoff = |prefix: &str| {
88
+ let mut value = public();
89
+ value["scope"]["protection"]["prefixes"] =
90
+ json!([prefix, "192.0.2.0/24", "203.0.113.10/32"]);
91
+ normalized(value)
92
+ };
93
+ assert!(handoff("10.240.0.2/32").is_err());
94
+ assert!(handoff("10.240.0.0/30").is_ok());
95
+ // A platform endpoint on the uplink lease lies in protected space, and it
96
+ // still refuses unless declared host-local.
97
+ let mut value = public();
98
+ value["scope"]["protection"]["prefixes"] =
99
+ json!(["10.0.0.0/8", "192.0.2.0/24", "198.51.100.7/32"]);
100
+ value["scope"]["protection"]["platformEndpoints"] = json!([
101
+ {"id":"hub","address":"198.51.100.7","protocol":"tcp","port":8443}]);
102
+ assert!(normalized(value.clone()).is_err());
103
+ value["scope"]["localPlatformEndpoints"] = json!(["hub"]);
104
+ assert!(normalized(value).is_ok());
105
+ }
106
+
107
+ #[test]
108
+ fn host_transit_barrier_protects_an_unprotected_uplink_lease_as_a_destination() {
109
+ let with = prepared(public());
110
+ // Positive control: leased egress to public space passes, including a flow
111
+ // another owner translated between two public destinations.
112
+ for flow in [egress(PUBLIC, PUBLIC), egress(PUBLIC, [198, 51, 100, 201])] {
113
+ assert_eq!(decide(&with, "forward", &flow).verdict, 1);
114
+ }
115
+ // A flow whose live or originally tracked destination is the host's own
116
+ // uplink lease is a translation into the host: the barrier refuses it.
117
+ for flow in [egress(PUBLIC, UPLINK), egress(UPLINK, PUBLIC)] {
118
+ let decision = decide(&with, "forward", &flow);
119
+ assert_eq!(
120
+ (decision.verdict, decision.chain.as_str()),
121
+ (0, "protected_out")
122
+ );
123
+ }
124
+ // A packet forwarded toward the handoff from the uplink lease is a spoof.
125
+ let reply = Flow {
126
+ input: UPLINK_LINK,
127
+ output: HANDOFF_LINK,
128
+ source: UPLINK,
129
+ destination: TRANSIT,
130
+ source_port: 443,
131
+ destination_port: 10500,
132
+ tcp_flags: ACK,
133
+ state: ESTABLISHED,
134
+ reply: true,
135
+ ..egress(PUBLIC, PUBLIC)
136
+ };
137
+ let decision = decide(&with, "forward", &reply);
138
+ assert_eq!(
139
+ (decision.verdict, decision.chain.as_str()),
140
+ (0, "protected_back")
141
+ );
142
+ let reply = Flow {
143
+ source: PUBLIC,
144
+ ..reply
145
+ };
146
+ assert_eq!(decide(&with, "forward", &reply).verdict, 1);
147
+ }
148
+
149
+ #[test]
150
+ fn host_transit_covered_uplink_compiles_exactly_the_authority_barrier() {
151
+ // The fixture's uplink lies inside the authority: no address joins the
152
+ // barrier, and its compiled bytes stay frozen by the golden host test.
153
+ let covered = chains(&prepared(host()));
154
+ let public = chains(&prepared(public()));
155
+ for name in ["protected_out", "protected_back"] {
156
+ // One /32 adds a live and an originally tracked destination check.
157
+ assert_eq!(public[name].len(), covered[name].len() + 2, "{name}");
158
+ }
159
+ }
@@ -0,0 +1,200 @@
1
+ //! Exclusive forwarding qualified on a host namespace that forwards IPv4 and
2
+ //! IPv6 between a handoff, an uplink and a LAN link no scope binds.
3
+ use super::egress_traffic_tests::{connect, host_policy, ns_ip, roundtrip, socket};
4
+ use super::host_traffic_tests::{control, ipv6_pair};
5
+ use super::*;
6
+ use crate::egress;
7
+ use std::net::UdpSocket;
8
+ use std::time::Duration;
9
+
10
+ const SOURCE: &str = "xf_source";
11
+ const HOST: &str = "xf_host";
12
+ const PEER: &str = "xf_peer";
13
+ const LAN: &str = "xf_lan";
14
+
15
+ /// The fixture host policy at `revision`, with the member and the handoffs
16
+ /// chosen. It reads the bound links, so it runs in the host namespace.
17
+ fn policy(revision: u64, exclusive: bool, handoffs: bool) -> Prepared {
18
+ let mut policy = host_policy();
19
+ policy.revision = revision;
20
+ let egress::Scope::HostTransit(scope) = &mut policy.scope else {
21
+ panic!()
22
+ };
23
+ scope.exclusive_forwarding = exclusive;
24
+ if !handoffs {
25
+ scope.handoffs.clear();
26
+ }
27
+ let target: Prepared = policy.prepare().unwrap().into();
28
+ target
29
+ .validate_interfaces()
30
+ .expect("exclusive forwarding fixture binding");
31
+ target
32
+ }
33
+ /// A UDP socket in `ns` on an address built at run time.
34
+ fn bound(ns: &str, address: String) -> UdpSocket {
35
+ let socket = in_namespace(ns, move || UdpSocket::bind(address).unwrap());
36
+ socket
37
+ .set_read_timeout(Some(Duration::from_millis(250)))
38
+ .unwrap();
39
+ socket
40
+ }
41
+ /// Every forward this host performs for the LAN link: the host as the LAN's
42
+ /// gateway to the uplink peer, the peer's way back into the LAN, and the same
43
+ /// gateway for IPv6. Fresh ports per call give each probe its own conntrack entry.
44
+ fn lan_probes(port: u16, allowed: bool) {
45
+ for (from, source, to, destination) in [
46
+ (
47
+ LAN,
48
+ format!("172.16.0.2:{port}"),
49
+ PEER,
50
+ format!("192.0.2.3:{port}"),
51
+ ),
52
+ (
53
+ PEER,
54
+ format!("192.0.2.3:{}", port + 1),
55
+ LAN,
56
+ format!("172.16.0.2:{}", port + 1),
57
+ ),
58
+ (
59
+ LAN,
60
+ format!("[fd00:16::2]:{port}"),
61
+ PEER,
62
+ format!("[2001:db8:190::3]:{port}"),
63
+ ),
64
+ ] {
65
+ let target = bound(to, destination.clone());
66
+ control(&bound(from, source), &destination, &target, allowed);
67
+ }
68
+ }
69
+ /// Replaces the applied policy with `policy(revision, exclusive, handoffs)`,
70
+ /// bound in the host namespace that holds the links.
71
+ fn reconcile(
72
+ owner: Owner,
73
+ previous: Applied,
74
+ (revision, exclusive, handoffs): (u64, bool, bool),
75
+ ) -> (Owner, Applied) {
76
+ in_namespace(HOST, move || {
77
+ let mut owner = owner;
78
+ let applied = owner
79
+ .reconcile(Transition {
80
+ previous: Some(previous),
81
+ target: policy(revision, exclusive, handoffs),
82
+ })
83
+ .unwrap();
84
+ assert!(owner.inspect().enforced);
85
+ (owner, applied)
86
+ })
87
+ }
88
+
89
+ #[test]
90
+ #[ignore = "requires disposable isolated native qualification guest"]
91
+ fn v2_host_exclusive_forwarding_forwards_only_the_scope_and_withdraws_cleanly() {
92
+ isolated();
93
+ for name in [SOURCE, HOST, PEER, LAN] {
94
+ ip(&["netns", "add", name]);
95
+ ns_ip(name, &["link", "set", "lo", "up"]);
96
+ in_namespace(name, || {
97
+ std::fs::write("/proc/sys/net/ipv4/ip_forward", "1").unwrap();
98
+ std::fs::write("/proc/sys/net/ipv6/conf/all/forwarding", "1").unwrap();
99
+ });
100
+ }
101
+ connect(
102
+ SOURCE,
103
+ "source",
104
+ "10.240.0.2/30",
105
+ HOST,
106
+ "router",
107
+ "10.240.0.1/30",
108
+ );
109
+ connect(
110
+ HOST,
111
+ "uplink",
112
+ "192.0.2.2/24",
113
+ PEER,
114
+ "underlay",
115
+ "192.0.2.3/24",
116
+ );
117
+ connect(
118
+ HOST,
119
+ "lanlink",
120
+ "172.16.0.1/24",
121
+ LAN,
122
+ "lanpeer",
123
+ "172.16.0.2/24",
124
+ );
125
+ ns_ip(SOURCE, &["route", "add", "default", "via", "10.240.0.1"]);
126
+ ns_ip(PEER, &["route", "add", "default", "via", "192.0.2.2"]);
127
+ ns_ip(LAN, &["route", "add", "default", "via", "172.16.0.1"]);
128
+ ipv6_pair(
129
+ HOST,
130
+ "uplink",
131
+ "2001:db8:190::108",
132
+ PEER,
133
+ "underlay",
134
+ "2001:db8:190::3",
135
+ );
136
+ ipv6_pair(HOST, "lanlink", "fd00:16::1", LAN, "lanpeer", "fd00:16::2");
137
+ ns_ip(
138
+ PEER,
139
+ &["-6", "route", "add", "default", "via", "2001:db8:190::108"],
140
+ );
141
+ ns_ip(LAN, &["-6", "route", "add", "default", "via", "fd00:16::1"]);
142
+ // The leased egress path: the transit source and a leased port to the
143
+ // uplink peer's platform endpoint, answered through outer SNAT.
144
+ let leased = socket(SOURCE, "10.240.0.2:10000");
145
+ let platform = socket(PEER, "192.0.2.3:5300");
146
+ let unleased = socket(SOURCE, "10.240.0.2:9999");
147
+ // Positive controls: without any policy the host forwards for the LAN in
148
+ // both directions and both families.
149
+ lan_probes(41000, true);
150
+ // The scope without the member leaves forwarding outside its handoffs to
151
+ // other owners, so every later denial is caused by the member.
152
+ let (owner, applied) = in_namespace(HOST, || {
153
+ let mut owner = Owner::new(options("host_exclusive")).unwrap();
154
+ let applied = owner
155
+ .reconcile(Transition {
156
+ previous: None,
157
+ target: policy(1, false, true),
158
+ })
159
+ .unwrap();
160
+ assert!(owner.inspect().enforced);
161
+ (owner, applied)
162
+ });
163
+ lan_probes(41010, true);
164
+ assert_eq!(roundtrip(&leased, &platform).ip().to_string(), "192.0.2.2");
165
+ // A LAN flow already established through the host before the member.
166
+ let established = socket(LAN, "172.16.0.2:41100");
167
+ let established_target = socket(PEER, "192.0.2.3:41100");
168
+ assert_eq!(
169
+ roundtrip(&established, &established_target)
170
+ .ip()
171
+ .to_string(),
172
+ "172.16.0.2"
173
+ );
174
+ // Exclusive: only the scope's leased handoff flows still cross the host.
175
+ let (owner, applied) = reconcile(owner, applied, (2, true, true));
176
+ lan_probes(41020, false);
177
+ control(&established, "192.0.2.3:41100", &established_target, false);
178
+ assert_eq!(roundtrip(&leased, &platform).ip().to_string(), "192.0.2.2");
179
+ control(&unleased, "192.0.2.3:5300", &platform, false);
180
+ // Without handoffs the exclusive scope forwards nothing at all.
181
+ let (owner, applied) = reconcile(owner, applied, (3, true, false));
182
+ lan_probes(41030, false);
183
+ control(&leased, "192.0.2.3:5300", &platform, false);
184
+ // Replacing it with a scope without the member reopens the LAN path.
185
+ let (owner, applied) = reconcile(owner, applied, (4, false, true));
186
+ lan_probes(41040, true);
187
+ assert_eq!(roundtrip(&leased, &platform).ip().to_string(), "192.0.2.2");
188
+ // Release of an exclusive scope reopens it as well.
189
+ let (owner, applied) = reconcile(owner, applied, (5, true, true));
190
+ lan_probes(41050, false);
191
+ in_namespace(HOST, move || {
192
+ let mut owner = owner;
193
+ assert!(owner.inspect().enforced);
194
+ owner.release(applied).unwrap();
195
+ });
196
+ lan_probes(41060, true);
197
+ let reopened = socket(SOURCE, "10.240.0.2:10001");
198
+ control(&reopened, "192.0.2.3:5300", &platform, true);
199
+ println!("V2_HOST_EXCLUSIVE_FORWARDING_PROOF positive_controls=true member_absent_lan_forwarded=true lan_ipv4_denied=true lan_reverse_denied=true lan_ipv6_denied=true established_foreign_flow_stopped=true leased_egress_kept=true unleased_port_denied=true empty_scope_forwards_nothing=true replacement_reopens=true release_reopens=true");
200
+ }
@@ -121,7 +121,7 @@ pub(super) fn control(source: &UdpSocket, destination: &str, target: &UdpSocket,
121
121
  assert!(target.recv(&mut [0; 64]).is_err());
122
122
  }
123
123
  }
124
- fn ipv6_pair(a: &str, aname: &str, aip: &str, b: &str, bname: &str, bip: &str) {
124
+ pub(super) fn ipv6_pair(a: &str, aname: &str, aip: &str, b: &str, bname: &str, bip: &str) {
125
125
  let aname_owned = aname.to_owned();
126
126
  let bname_owned = bname.to_owned();
127
127
  let amac = in_namespace(a, move || packet_fixture::mac(&aname_owned));
@@ -21,6 +21,9 @@ mod packet_fixture;
21
21
  #[path = "owner_host_traffic_tests.rs"]
22
22
  mod host_traffic_tests;
23
23
 
24
+ #[path = "owner_exclusiveforwarding_tests.rs"]
25
+ mod exclusiveforwarding_tests;
26
+
24
27
  #[path = "owner_router_traffic_tests.rs"]
25
28
  mod router_traffic_tests;
26
29
 
@@ -3,6 +3,6 @@
3
3
  */
4
4
  export const commitinfo = {
5
5
  name: '@push.rocks/smartnftables',
6
- version: '4.0.0',
6
+ version: '4.1.0',
7
7
  description: 'A TypeScript module for managing nftables rules including NAT, firewall, and rate limiting with a high-level API.'
8
8
  }
@@ -172,6 +172,8 @@ export interface IManagedNftHostTransitScopeV2 {
172
172
  kind: 'hostTransit';
173
173
  protection: IManagedNftProtectionV2;
174
174
  handoffs: Array<{ link: IManagedNftLocalLinkV2; allocations: IManagedNftHandoffAllocationV2[] }>;
175
+ /** Its addresses need not lie in `protection.prefixes`; the host barrier protects each
176
+ * uncovered one as a `/32` destination. Handoff addresses must be protected. */
175
177
  uplink: IManagedNftLocalLinkV2;
176
178
  /** Exact current address present on uplink. No masquerade or default-route inference. */
177
179
  snatAddress: string;
@@ -191,6 +193,11 @@ export interface IManagedNftHostTransitScopeV2 {
191
193
  * leased source address and a source port of the allocation's range for the endpoint's protocol; its
192
194
  * ESTABLISHED replies return. Absent and empty are the same canonical policy, digest and compiled graph. */
193
195
  localPlatformEndpoints?: string[];
196
+ /** Optional: this table is the host's only forwarding owner. Every forwarded packet the scope does not
197
+ * admit (its leased, published and symmetric handoff flows) drops, IPv4 and IPv6, on every pair of
198
+ * interfaces, including forwarding that another table would accept. With no `handoffs` the host forwards
199
+ * nothing. It sets no sysctl. Absent and `false` are the same canonical policy, digest and compiled graph. */
200
+ exclusiveForwarding?: boolean;
194
201
  }
195
202
 
196
203
  /** One loopback TCP service that only one local user may dial. Every packet the host sends to the