@push.rocks/smartnftables 2.5.2 → 2.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/changelog.md CHANGED
@@ -1,5 +1,17 @@
1
1
  # Changelog
2
2
 
3
+ ## 2026-09-25 - 2.6.0
4
+
5
+ ### Features
6
+
7
+ - Add optional loopback TCP port owners to the schema-v2 `allocationPoolGuard` scope: `localTcpPortOwners` entries `{ address, port, uid }` (an exact IPv4 loopback host address, one port and one owning uid, at most 8) compile, ahead of the guard, an OUTPUT rule that rejects with a TCP reset every packet to that address and port from a socket whose uid differs (`meta skuid != uid`, after ordinary output destination NAT) and an INPUT rule that drops the address and port arriving on any interface but loopback. Owner verification now admits the TCP-reset `reject` expression and reads it back like every other operand, so the entries are verified on apply, retained, adopted, recovered and released with the table. Absent and empty keep the previous canonical policy, digest and compiled bytes. The native qualification guest loads the `nf_reject_ipv4`, `nf_reject_ipv6`, `nft_reject` and `nft_reject_inet` modules.
8
+ - Add optional host-local platform endpoints to the schema-v2 `hostTransit` scope: `localPlatformEndpoints` lists ids of `protection.platformEndpoints` the host serves on an exact address of its uplink or a handoff link (verified through rtnetlink at apply, recovery and inspection). INPUT admits leased flows from their exact handoff with the lease's transit source address and a source port in its range for the endpoint's protocol, to the exact endpoint tuple in the live and originally tracked tuple (NEW with a SYN-only TCP opening, or ESTABLISHED), and OUTPUT admits only the ESTABLISHED reply; translated flows and host-origin openings stay denied. The refusal of a platform endpoint on a host address is lifted only for declared ids. Absent and empty keep the previous canonical policy, digest and compiled bytes.
9
+ - Add optional stateful one-way workload grants to the schema-v2 `routerEgress` scope: `workloadGrants` entries `{ protocol, sourceAddress, destinationAddress, destinationPort }` let one veth workload endpoint open one exact port of another. Four constant FORWARD admissions look up the incoming and outgoing link with their addresses in a CONSTANT `workload_link` set and the protocols with the live and originally tracked tuple in a CONSTANT `workload_grant` set, in the default conntrack zone; the destination only answers (ESTABLISHED replies) and can never open toward the source. Up to 1024 grants, sharing the scope's set element budget with host grants. Absent and empty keep the previous canonical policy, digest and compiled bytes.
10
+
11
+ ### Maintenance
12
+
13
+ - Bring the release-time tooling current: `@git.zone/tsrust` 1.14.0 → 1.15.0. `@types/node` stays 26.6.1 and pnpm 12.4.2, because their newer releases are younger than the seven-day rule; `@git.zone/cli` 7.2.2, tsbuild 5.0.0, tsrun 3.0.0 and tstest 6.2.0 are already current. The native notices verify unchanged.
14
+
3
15
  ## 2026-09-24 - 2.5.2
4
16
 
5
17
  ### Maintenance
@@ -1,13 +1,13 @@
1
1
  {
2
2
  "format": "tsrust.build-provenance.v2",
3
- "binarySha256": "8171a73d23d26828e29fc1a403ec08a4a24664d7b0f625d90217f59cb760638e",
3
+ "binarySha256": "8b0883d60ba14c1971ce50bd7ec45ea4cbb8a79c292e9063166fed168779dd52",
4
4
  "buildInfo": {
5
5
  "projectName": "@push.rocks/smartnftables",
6
- "projectVersion": "2.5.2",
7
- "gitCommit": "af81d179977c8903fc265a0b66a0b6c52c2b7e04",
6
+ "projectVersion": "2.6.0",
7
+ "gitCommit": "8032aac9d7b1b78650f0947574e5aab783b2d5db",
8
8
  "gitDirty": false,
9
- "builtAt": "2026-09-24T23:25:26.155Z",
10
- "tsrustVersion": "1.14.0",
9
+ "builtAt": "2026-09-25T13:42:00.180Z",
10
+ "tsrustVersion": "1.15.0",
11
11
  "binary": "smartnftables",
12
12
  "target": "linux_amd64_musl"
13
13
  }
@@ -1,13 +1,13 @@
1
1
  {
2
2
  "format": "tsrust.build-provenance.v2",
3
- "binarySha256": "05b1bfa1d470edec18ced9a47b2426fad72801b81ac8e29f3a8c3dc9cb1787f7",
3
+ "binarySha256": "86d2a679f4f1968b9ae9e0a3681572f5c4a6dc913993747abf5422f3adf4916b",
4
4
  "buildInfo": {
5
5
  "projectName": "@push.rocks/smartnftables",
6
- "projectVersion": "2.5.2",
7
- "gitCommit": "af81d179977c8903fc265a0b66a0b6c52c2b7e04",
6
+ "projectVersion": "2.6.0",
7
+ "gitCommit": "8032aac9d7b1b78650f0947574e5aab783b2d5db",
8
8
  "gitDirty": false,
9
- "builtAt": "2026-09-24T23:25:35.983Z",
10
- "tsrustVersion": "1.14.0",
9
+ "builtAt": "2026-09-25T13:42:10.633Z",
10
+ "tsrustVersion": "1.15.0",
11
11
  "binary": "smartnftables",
12
12
  "target": "linux_arm64_musl"
13
13
  }
@@ -3,7 +3,7 @@
3
3
  */
4
4
  export const commitinfo = {
5
5
  name: '@push.rocks/smartnftables',
6
- version: '2.5.2',
6
+ version: '2.6.0',
7
7
  description: 'A TypeScript module for managing nftables rules including NAT, firewall, and rate limiting with a high-level API.'
8
8
  };
9
9
  //# sourceMappingURL=data:application/json;base64,eyJ2ZXJzaW9uIjozLCJmaWxlIjoiMDBfY29tbWl0aW5mb19kYXRhLmpzIiwic291cmNlUm9vdCI6IiIsInNvdXJjZXMiOlsiLi4vdHMvMDBfY29tbWl0aW5mb19kYXRhLnRzIl0sIm5hbWVzIjpbXSwibWFwcGluZ3MiOiJBQUFBOztHQUVHO0FBQ0gsTUFBTSxDQUFDLE1BQU0sVUFBVSxHQUFHO0lBQ3hCLElBQUksRUFBRSwyQkFBMkI7SUFDakMsT0FBTyxFQUFFLE9BQU87SUFDaEIsV0FBVyxFQUFFLG1IQUFtSDtDQUNqSSxDQUFBIn0=
@@ -91,6 +91,21 @@ export interface IManagedNftHostGrant {
91
91
  /** The workload's exact listening port, 1–65535; never a range. */
92
92
  destinationPort: number;
93
93
  }
94
+ /** One stateful one-way flow between two workloads behind the same router: the source workload opens
95
+ * connections to one exact port of the destination workload, which may only answer. Both addresses are
96
+ * exact current addresses of two different veth workload endpoints of the scope. It admits the tuple
97
+ * (a TCP opening only with SYN) and its ESTABLISHED replies in the default conntrack zone and nothing
98
+ * else: no reverse opening, no translation and no source-port selection. Grant the reverse direction as
99
+ * a second grant. */
100
+ export interface IManagedNftWorkloadGrant {
101
+ protocol: 'tcp' | 'udp';
102
+ /** Exact unicast address of the dialling workload endpoint, never a prefix. */
103
+ sourceAddress: string;
104
+ /** Exact unicast address of the dialled workload endpoint, never a prefix. */
105
+ destinationAddress: string;
106
+ /** The destination workload's exact listening port, 1–65535; never a range. */
107
+ destinationPort: number;
108
+ }
94
109
  /** One combined private/TUN/local/egress table; never compose a second ACCEPT over v1 terminal drops. */
95
110
  export interface IManagedNftRouterEgressScopeV2 {
96
111
  kind: 'routerEgress';
@@ -114,6 +129,10 @@ export interface IManagedNftRouterEgressScopeV2 {
114
129
  * workload. Absent and empty are the same canonical policy, digest and compiled graph; at most 1024,
115
130
  * held in a named set behind a constant number of rules. */
116
131
  hostGrants?: IManagedNftHostGrant[];
132
+ /** Optional stateful one-way workload-to-workload grants. Absent and empty are the same canonical
133
+ * policy, digest and compiled graph; at most 1024, held in two named sets behind four rules. They
134
+ * share the scope's set element budget with `hostGrants`. */
135
+ workloadGrants?: IManagedNftWorkloadGrant[];
117
136
  }
118
137
  /** One inbound uplink publication, compiled inside the same host-transit generation. */
119
138
  export interface IManagedNftPublishedPortV2 {
@@ -147,6 +166,27 @@ export interface IManagedNftHostTransitScopeV2 {
147
166
  * host-local, a platform endpoint nor a leased transit source address. Absent and empty are the
148
167
  * same canonical policy; at most 1024, held in a named set behind a constant number of rules. */
149
168
  hostGrants?: IManagedNftHostGrant[];
169
+ /** Optional ids of `protection.platformEndpoints` this host serves itself, on an exact current address
170
+ * of the uplink or of a handoff link (rtnetlink verifies it at apply, recovery and inspection). Only a
171
+ * declared endpoint may hold a host address. Leased flows reach it from their exact handoff with the
172
+ * leased source address and a source port of the allocation's range for the endpoint's protocol; its
173
+ * ESTABLISHED replies return. Absent and empty are the same canonical policy, digest and compiled graph. */
174
+ localPlatformEndpoints?: string[];
175
+ }
176
+ /** One loopback TCP service that only one local user may dial. Every packet the host sends to the
177
+ * exact address and port from a socket of any other user is rejected with a TCP reset, after ordinary
178
+ * output destination NAT; the address and port arriving on any interface but loopback are dropped.
179
+ * The service must bind exactly this address: a wildcard listener stays reachable through others. */
180
+ export interface IManagedNftLocalTcpPortOwner {
181
+ /** Exact IPv4 loopback host address in 127.0.0.0/8, e.g. `127.0.0.1`. Never a prefix or wildcard. */
182
+ address: string;
183
+ /** The service's exact listening port, 1–65535; never a range. */
184
+ port: number;
185
+ /** The only user id whose sockets may send to the port: the socket file's file-system uid as
186
+ * the user namespace owning the network namespace sees it, 0–4294967294. Packets without a socket file carry no uid and
187
+ * are not matched: kernel replies, the teardown of an orphaned socket, and in-kernel sockets such as NFS or CIFS clients,
188
+ * which only privileged mounts create. */
189
+ uid: number;
150
190
  }
151
191
  /** Host-wide IPv4 destination denial for caller-authenticated private allocation pools.
152
192
  * It has no link dependency; exact host grants are its only exceptions. It is not allocation-release or boot-order proof. */
@@ -159,6 +199,10 @@ export interface IManagedNftAllocationPoolGuardScopeV2 {
159
199
  * listed pool, and their replies. Absent and empty are the same canonical policy; at most 1024,
160
200
  * held in a named set behind a constant number of rules. */
161
201
  hostGrants?: IManagedNftHostGrant[];
202
+ /** Optional host-wide loopback port ownership, compiled ahead of the guard in the same table.
203
+ * Absent and empty are the same canonical policy, digest and compiled graph; at most 8, one per
204
+ * exact address and port. */
205
+ localTcpPortOwners?: IManagedNftLocalTcpPortOwner[];
162
206
  }
163
207
  export interface IManagedNftPolicyV2 {
164
208
  schemaVersion: 2;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@push.rocks/smartnftables",
3
- "version": "2.5.2",
3
+ "version": "2.6.0",
4
4
  "private": false,
5
5
  "description": "A TypeScript module for managing nftables rules including NAT, firewall, and rate limiting with a high-level API.",
6
6
  "main": "dist_ts/index.js",
@@ -12,7 +12,7 @@
12
12
  "@git.zone/cli": "7.2.2",
13
13
  "@git.zone/tsbuild": "^5.0.0",
14
14
  "@git.zone/tsrun": "^3.0.0",
15
- "@git.zone/tsrust": "1.14.0",
15
+ "@git.zone/tsrust": "1.15.0",
16
16
  "@git.zone/tstest": "6.2.0",
17
17
  "@types/node": "26.6.1",
18
18
  "typescript": "^7.0.2"
package/readme.md CHANGED
@@ -264,9 +264,9 @@ allocation-pool guard policy kinds.
264
264
 
265
265
  | V2 scope | Required authority and behavior |
266
266
  | --- | --- |
267
- | `routerEgress` | Private `endpoints` and `rules`, one exact `links` binding per endpoint, a separate veth `handoff`, `protection`, and active `generations`. Private veth/TUN/local DNS and egress share one table so terminal private denial cannot override a separate egress table. Optional `publishedPorts` add the inbound second hop from the handoff to a workload endpoint; both directions are classified into the default conntrack zone ahead of every leased classifier, so a published endpoint port is dedicated to its publication and never becomes leased egress. Optional `hostGrants` forward exact host-origin flows from the handoff to a workload endpoint. |
268
- | `hostTransit` | Exact handoff `link`/`allocations` pairs, complete `protection`, an explicit veth or Ethernet `uplink`, and its current `snatAddress`. It checks each handoff's leased source address and protocol/port range, default conntrack zone, direction, uplink, and protected destinations before outer SNAT. Optional `publishedPorts` add inbound uplink destination NAT inside the same generation. Optional `hostGrants` let the host's own address on a handoff dial exact workload ports. |
269
- | `allocationPoolGuard` | An authenticated `authorityDigest` and complete current allocation-pool `prefixes`. Installs host-wide IPv4 destination denial before any handoff exists, without link, uplink or SNAT dependencies. Optional `hostGrants` are its only exceptions. |
267
+ | `routerEgress` | Private `endpoints` and `rules`, one exact `links` binding per endpoint, a separate veth `handoff`, `protection`, and active `generations`. Private veth/TUN/local DNS and egress share one table so terminal private denial cannot override a separate egress table. Optional `publishedPorts` add the inbound second hop from the handoff to a workload endpoint; both directions are classified into the default conntrack zone ahead of every leased classifier, so a published endpoint port is dedicated to its publication and never becomes leased egress. Optional `hostGrants` forward exact host-origin flows from the handoff to a workload endpoint. Optional `workloadGrants` let one workload open one exact port of another, one way. |
268
+ | `hostTransit` | Exact handoff `link`/`allocations` pairs, complete `protection`, an explicit veth or Ethernet `uplink`, and its current `snatAddress`. It checks each handoff's leased source address and protocol/port range, default conntrack zone, direction, uplink, and protected destinations before outer SNAT. Optional `publishedPorts` add inbound uplink destination NAT inside the same generation. Optional `hostGrants` let the host's own address on a handoff dial exact workload ports. Optional `localPlatformEndpoints` serve platform endpoints on the host's own addresses to leased flows. |
269
+ | `allocationPoolGuard` | An authenticated `authorityDigest` and complete current allocation-pool `prefixes`. Installs host-wide IPv4 destination denial before any handoff exists, without link, uplink or SNAT dependencies. Optional `hostGrants` are its only exceptions. Optional `localTcpPortOwners` restrict loopback TCP ports to one local user each. |
270
270
 
271
271
  `allocationPoolGuard` accepts 1–64 canonical, disjoint RFC1918 prefixes. Supply
272
272
  the actual allocation pools, not the broader protected union containing management
@@ -474,6 +474,119 @@ hop by hop from the workload side (router, host transit, guard) and withdraw the
474
474
  in the reverse order; each table is atomic on its own, and a partially applied
475
475
  grant only stays dark.
476
476
 
477
+ #### Host-local platform endpoints
478
+
479
+ A platform endpoint normally lives elsewhere and host transit forwards leased
480
+ flows to it. When the host serves an endpoint itself — a relay listener, a data
481
+ port or a VPN hub on an address the host holds — the leased flow ends in the
482
+ host's INPUT instead, where every handoff arrival is denied. The optional
483
+ `hostTransit.localPlatformEndpoints` lists the ids of `protection.platformEndpoints`
484
+ the host serves:
485
+
486
+ ```typescript
487
+ protection: { authorityDigest, prefixes: ['10.0.0.0/8', '192.0.2.0/24'], platformEndpoints: [
488
+ { id: 'relay', address: '192.0.2.2', protocol: 'tcp', port: 8443 }] },
489
+ localPlatformEndpoints: ['relay'],
490
+ ```
491
+
492
+ Each id names an existing endpoint whose address is an exact `requiredIpv4Addresses`
493
+ entry of the uplink or of a handoff link, so rtnetlink verifies that the host holds it
494
+ at apply, recovery and inspection. The refusal of a platform endpoint on a host
495
+ address is lifted only for declared ids; unknown, duplicate or unheld ids reject.
496
+ The router scope grants its workloads the endpoint like any platform endpoint
497
+ (`destination: { kind: 'platform', endpointId }`).
498
+
499
+ INPUT admits the original direction only from the exact handoff that carries a lease,
500
+ with the lease's transit source address and a source port in one of its ranges for
501
+ the endpoint's protocol, in the default conntrack zone, to the exact endpoint address
502
+ and port in both the live and the originally tracked tuple, NEW (a TCP opening only
503
+ with SYN) or ESTABLISHED. OUTPUT admits only the ESTABLISHED reply back through that
504
+ handoff. A flow translated in transit, another source address or port, another
505
+ endpoint port or protocol and any host-origin opening toward the transit address keep
506
+ meeting the handoff denials. The source port is checked against the leased range in
507
+ both tuples, not for equality between them. Each leased range adds one INPUT and one
508
+ OUTPUT jump when an endpoint of its protocol is declared, and each endpoint adds three
509
+ rules. Absent and empty are the same canonical policy, digest and compiled bytes. The
510
+ caller keeps a declared endpoint's address outside every guarded allocation pool and
511
+ owns the listener.
512
+
513
+ #### Workload grants
514
+
515
+ `routerEgress.workloadGrants` is optional: a stateful one-way flow between two
516
+ workloads behind the same router, such as an ingress workload reaching a target
517
+ workload's service port across organizations. Each entry is `{ protocol,
518
+ sourceAddress, destinationAddress, destinationPort }`: `tcp` or `udp`, the exact
519
+ current addresses of two different veth workload endpoints of the scope and one port
520
+ 1–65535. Prefixes, wildcards, ranges, router-local and platform-endpoint addresses,
521
+ TUN endpoints, both ends on one endpoint and duplicates reject. The source opens and
522
+ the destination only answers; the reverse direction is a second grant.
523
+
524
+ Four FORWARD admissions follow the private anti-spoof guards whatever the number of
525
+ grants: ESTABLISHED original direction, a NEW UDP opening, a NEW TCP opening
526
+ restricted to SYN with FIN/RST/ACK clear, and the ESTABLISHED reply, each in the
527
+ default conntrack zone. Every admission looks up the incoming link with the source
528
+ address and the outgoing link with the destination address (index and name) in the
529
+ CONSTANT set `workload_link`, and the protocols with the live and originally tracked
530
+ tuple in the CONSTANT set `workload_grant`, so a spoofed, renamed or translated flow
531
+ never matches and the destination can never open toward the source. Endpoint prefixes
532
+ are protected, so both directions stay in the default zone ahead of every leased
533
+ classifier. Owner verification reads both sets back element by element. Up to 1024
534
+ grants fit; the sets share the scope's 100,000-byte element budget with `hostGrants`,
535
+ so 1024 of each do not fit together and preparation refuses that before any kernel
536
+ work. Absent and empty are the same canonical policy, digest and compiled bytes, and
537
+ withdrawal is an ordinary atomic replacement that also stops established flows.
538
+
539
+ These flows need no private `rules`. A private rule is stateless: answering through
540
+ private rules needs a reverse rule, which would also let the destination open toward
541
+ the source.
542
+
543
+ #### Loopback TCP port owners
544
+
545
+ `allocationPoolGuard.localTcpPortOwners` is optional and restricts a loopback TCP
546
+ service to one local user, such as a container runtime's CRI stream server on
547
+ `127.0.0.1:10010` that only root may reach. Each entry is `{ address, port, uid }`:
548
+ one exact IPv4 loopback host address in `127.0.0.0/8` (not the network or broadcast
549
+ address), one port 1–65535 and one uid 0–4294967294. Prefixes, wildcards, IPv6,
550
+ ranges, uid lists and a second entry for the same address and port reject; the
551
+ limit is 8 entries. Absent and empty are the same canonical policy, so a guard
552
+ without owners keeps its exact previous digest and compiled bytes.
553
+
554
+ ```typescript
555
+ const guard: IManagedNftPolicyV2 = { schemaVersion: 2, revision: 1, scope: {
556
+ kind: 'allocationPoolGuard', authorityDigest, prefixes: ['10.240.0.0/16'],
557
+ localTcpPortOwners: [{ address: '127.0.0.1', port: 10010, uid: 0 }] } };
558
+ ```
559
+
560
+ Each entry compiles two rules at the head of the guard's base chains, ahead of the
561
+ host grant admissions and the pool guard, so nothing earlier in the table accepts
562
+ past them:
563
+
564
+ - OUTPUT (filter priority 0, after ordinary output destination NAT): IPv4 TCP to the
565
+ exact address and port with `meta skuid != uid` is rejected with a TCP reset, so
566
+ another user's `connect()` fails at once with `ECONNREFUSED`. This covers every
567
+ packet, not only the opening, and a flow a foreign output DNAT redirects to the
568
+ port. `meta skuid` is the socket file's file-system uid as the user namespace owning the
569
+ network namespace sees it, the initial one on a host: a process in a user namespace
570
+ that shares the host network namespace is matched by its host uid, and a user
571
+ namespace's root is not uid 0. A packet without a socket
572
+ file carries no uid and does not match: a kernel reply, the teardown of an orphaned
573
+ socket, or an in-kernel socket such as an NFS or CIFS client, which only a privileged
574
+ mount creates.
575
+ - INPUT: the exact address and port arriving on any interface but loopback (ifindex
576
+ 1 in every network namespace) are dropped, so `route_localnet` or a prerouting DNAT
577
+ cannot expose the service to another host.
578
+
579
+ The rules match the exact destination address. The service must bind exactly that
580
+ address; a wildcard listener stays reachable through the host's other addresses.
581
+ IPv6 is not covered, so the service must not listen on `::1` or `::`. A socket keeps
582
+ the uid it was created with, so a root process that hands its connected socket to
583
+ another user delegates that connection. The reject expression needs the kernel's
584
+ `nft_reject_inet` module (autoloaded on stock kernels). Owner verification reads
585
+ back every rule, including the comparison operator, the uid and the reject type, so
586
+ apply, inspection, adoption and recovery reject a changed rule. The entries are
587
+ applied, retained, recovered and released with the rest of the guard's table; after
588
+ a reboot the caller applies its retained intent again, like every other member.
589
+
477
590
  Input bounds are 1024 grants per scope, the maximum a network projection carries.
478
591
  Elements have their own 100,000-byte budget beside the 100,000-byte rule graph, and
479
592
  are sent 256 to a message. At 1024 grants the reference router needs about 94 KB
@@ -632,6 +745,32 @@ withdrawal in every scope stops new flows and the established one. A full guard
632
745
  set persists across owner loss, is re-verified element by element before adoption,
633
746
  survives lost-ACK replay and is replaced and released like the rest of the graph,
634
747
  and another socket cannot add or delete one of its elements.
748
+ Host-local platform endpoints are qualified from a router namespace against
749
+ pre-policy positive controls: without the member the handoff denials keep both a TCP
750
+ endpoint on the uplink address and a UDP endpoint on the handoff address dark; with
751
+ it the exact leased flows reach them with the transit source preserved, while a
752
+ source port outside the lease or in the other protocol's range, an unleased source
753
+ address, another endpoint port, a flow a foreign prerouting DNAT translated onto the
754
+ endpoint and the host's own opening toward the transit address stay dark; withdrawal
755
+ stops new flows and the established one, and release reopens the path. A graph with
756
+ served endpoints persists across owner loss and survives lost-ACK replay.
757
+ Workload grants are qualified between three workload namespaces behind the router,
758
+ each with leased public egress on every port of both protocols: against pre-policy
759
+ positive controls, the grant opens exactly the source workload's TCP and UDP flows to
760
+ the granted ports with the source preserved; another port, another source workload,
761
+ another destination workload, the destination's TCP and UDP openings toward the
762
+ source stay dark; withdrawal stops new flows and the established one; release reopens
763
+ the path. A set of 1024 grants persists across owner loss, is re-verified element by
764
+ element and survives lost-ACK replay.
765
+ Loopback TCP port owners are qualified against pre-policy positive controls on the
766
+ same listeners: the owning uid (root, and uid 1000 for a second port) connects, every
767
+ other uid is reset with `ECONNREFUSED`, an unowned port and another loopback address
768
+ stay open to every user, a foreign output DNAT to the owned port stays closed to
769
+ another user, and an uplink arrival translated to loopback with `route_localnet`
770
+ enabled is dropped. The policy keeps enforcing after the owner detaches and its
771
+ process is gone, a fresh owner verifies and adopts the exact graph, and release
772
+ reopens every probe. The maximum of eight owners beside the guard persists across
773
+ owner loss and survives lost-ACK replay.
635
774
  The arm64 binary is cross-built; native packet qualification is currently x86_64.
636
775
  Kernel 6.8 is unsupported. No production activation is implied by these tests.
637
776
 
@@ -10,6 +10,8 @@ mod hostgrant;
10
10
  mod poolguard;
11
11
  #[path = "egress.router.rs"]
12
12
  mod router;
13
+ #[path = "egress.workloadgrant.rs"]
14
+ mod workloadgrant;
13
15
 
14
16
  /// Set elements of one target. Together with the 100,000-byte rule graph and the
15
17
  /// handle-only deletion of a previous maximum graph this bounds one replacement
@@ -36,10 +36,25 @@ fn envelope(
36
36
  allocation: &Allocation,
37
37
  range: &PortRange,
38
38
  reply: bool,
39
+ ) -> Result<Vec<Attr>> {
40
+ leased(handoff, Some(&scope.uplink), allocation, range, reply)
41
+ }
42
+ /// One leased range of one allocation on its exact handoff: the default zone, the
43
+ /// direction, the protocols, and the transit source address and port range in
44
+ /// both the live and the originally tracked tuple. A forwarded envelope adds the
45
+ /// uplink; a host-local one has none.
46
+ fn leased(
47
+ handoff: &HostHandoff,
48
+ uplink: Option<&LocalLink>,
49
+ allocation: &Allocation,
50
+ range: &PortRange,
51
+ reply: bool,
39
52
  ) -> Result<Vec<Attr>> {
40
53
  let mut expressions = ipv4();
41
54
  expressions.extend(link(&handoff.link, !reply));
42
- expressions.extend(link(&scope.uplink, reply));
55
+ if let Some(uplink) = uplink {
56
+ expressions.extend(link(uplink, reply));
57
+ }
43
58
  expressions.extend(meta(16, vec![protocol(&range.protocol)]));
44
59
  expressions.extend(ct(10, None, vec![protocol(&range.protocol)]));
45
60
  expressions.extend(ct(17, None, 0_u16.to_ne_bytes().to_vec()));
@@ -114,6 +129,70 @@ fn published(program: &mut Program<'_>, scope: &HostScope) -> Result<()> {
114
129
  }
115
130
  Ok(())
116
131
  }
132
+ /// The exact platform endpoints served on host addresses. Every packet is
133
+ /// checked against the live and the originally tracked destination tuple, so a
134
+ /// flow rewritten in transit never matches.
135
+ fn local_endpoints(scope: &HostScope) -> Vec<&PlatformEndpoint> {
136
+ scope
137
+ .protection
138
+ .platform_endpoints
139
+ .iter()
140
+ .filter(|endpoint| scope.local_platform_endpoints.contains(&endpoint.id))
141
+ .collect()
142
+ }
143
+ fn local_endpoint(endpoint: &PlatformEndpoint, reply: bool) -> Result<Vec<Attr>> {
144
+ let mut expressions = meta(16, vec![protocol(&endpoint.protocol)]);
145
+ expressions.extend(ct(10, None, vec![protocol(&endpoint.protocol)]));
146
+ let address = format!("{}/32", endpoint.address);
147
+ expressions.extend(current_address(&address, reply)?);
148
+ expressions.extend(current_port(endpoint.port, reply));
149
+ expressions.extend(original_address(&address, false)?);
150
+ expressions.extend(original_port(endpoint.port, false));
151
+ Ok(expressions)
152
+ }
153
+ /// Leased flows into host-local platform endpoints, the reverse of a host grant:
154
+ /// INPUT admits the original direction only from an exact handoff with a leased
155
+ /// source address and port range of that handoff, NEW (a TCP opening only with
156
+ /// SYN) or ESTABLISHED; OUTPUT admits only the ESTABLISHED reply back to it. Each
157
+ /// leased range jumps once per direction, and the endpoint chains hold one tuple
158
+ /// check per endpoint, so the graph grows with ranges plus endpoints.
159
+ fn local_platform(program: &mut Program<'_>, scope: &HostScope) -> Result<()> {
160
+ let endpoints = local_endpoints(scope);
161
+ for endpoint in &endpoints {
162
+ let original = local_endpoint(endpoint, false)?;
163
+ for connection_state in [2, 8] {
164
+ let mut expressions = original.clone();
165
+ expressions.extend(state(connection_state));
166
+ if connection_state == 8 && endpoint.protocol == "tcp" {
167
+ expressions.extend(opening_tcp());
168
+ }
169
+ program.end("local_platform_in", expressions, 1)?;
170
+ }
171
+ let mut reply = local_endpoint(endpoint, true)?;
172
+ reply.extend(state(2));
173
+ program.end("local_platform_back", reply, 1)?;
174
+ }
175
+ for handoff in &scope.handoffs {
176
+ for allocation in &handoff.allocations {
177
+ for range in &allocation.source_port_ranges {
178
+ if endpoints
179
+ .iter()
180
+ .all(|endpoint| endpoint.protocol != range.protocol)
181
+ {
182
+ continue;
183
+ }
184
+ for (chain, target, reply) in [
185
+ ("input", "local_platform_in", false),
186
+ ("output", "local_platform_back", true),
187
+ ] {
188
+ let expressions = leased(handoff, None, allocation, range, reply)?;
189
+ program.jump(chain, expressions, target)?;
190
+ }
191
+ }
192
+ }
193
+ }
194
+ Ok(())
195
+ }
117
196
  pub(super) fn compile(program: &mut Program<'_>, scope: &HostScope) -> Result<()> {
118
197
  // Filtering runs after destination NAT. Docker has independent filter hooks.
119
198
  // This table cannot override another owner's DROP; Docker contribution and
@@ -135,6 +214,11 @@ pub(super) fn compile(program: &mut Program<'_>, scope: &HostScope) -> Result<()
135
214
  for name in ["protected_out", "protected_back"] {
136
215
  program.chain(name, None)?;
137
216
  }
217
+ if !scope.local_platform_endpoints.is_empty() {
218
+ for name in ["local_platform_in", "local_platform_back"] {
219
+ program.chain(name, None)?;
220
+ }
221
+ }
138
222
  if !scope.host_grants.is_empty() {
139
223
  // Validation admits only an address of exactly one handoff link.
140
224
  hostgrant::tuple_set(program, &scope.host_grants, |grant| {
@@ -163,6 +247,9 @@ pub(super) fn compile(program: &mut Program<'_>, scope: &HostScope) -> Result<()
163
247
  if !scope.host_grants.is_empty() {
164
248
  hostgrant::admissions(program, ["output", "input"], true, |_| vec![])?;
165
249
  }
250
+ // Leased flows into host-local platform endpoints precede the same terminal
251
+ // handoff denials, which remain for everything else.
252
+ local_platform(program, scope)?;
166
253
  for handoff in &scope.handoffs {
167
254
  program.jump("forward", link(&handoff.link, true), "protected_out")?;
168
255
  program.jump("forward", link(&handoff.link, false), "protected_back")?;
@@ -13,7 +13,7 @@ pub(super) const ARRIVAL_SET: &str = "host_grant_arrival";
13
13
 
14
14
  /// Register numbers as Linux dumps them: 16-byte aligned registers use the
15
15
  /// legacy NFT_REG_1..4 names, every other 4-byte register NFT_REG32_xx.
16
- fn register(index: u32) -> u32 {
16
+ pub(super) fn register(index: u32) -> u32 {
17
17
  if index % 4 == 0 {
18
18
  index / 4
19
19
  } else {
@@ -24,10 +24,10 @@ fn store(name: &str, mut data: Vec<Attr>, index: u32) -> Attr {
24
24
  data.insert(0, Attr::u32(1, register(index)));
25
25
  expr(name, data)
26
26
  }
27
- fn meta_to(key: u32, index: u32) -> Attr {
27
+ pub(super) fn meta_to(key: u32, index: u32) -> Attr {
28
28
  store("meta", vec![Attr::u32(2, key)], index)
29
29
  }
30
- fn payload_to(base: u32, offset: u32, length: u32, index: u32) -> Attr {
30
+ pub(super) fn payload_to(base: u32, offset: u32, length: u32, index: u32) -> Attr {
31
31
  store(
32
32
  "payload",
33
33
  vec![
@@ -45,7 +45,7 @@ fn ct_to(key: u32, original: bool, index: u32) -> Attr {
45
45
  }
46
46
  store("ct", data, index)
47
47
  }
48
- fn lookup(set: &str) -> Attr {
48
+ pub(super) fn lookup(set: &str) -> Attr {
49
49
  expr(
50
50
  "lookup",
51
51
  vec![
@@ -61,10 +61,10 @@ fn word(bytes: &[u8]) -> Vec<u8> {
61
61
  result
62
62
  }
63
63
 
64
- /// Key loads for one direction, then the lookup. `linked` adds the variable
65
- /// interface: the outgoing interface of the original direction, the incoming
66
- /// interface of the reply.
67
- fn tuple_lookup(linked: bool, reply: bool) -> Vec<Attr> {
64
+ /// Key loads for one direction, then the lookup in `set`. `linked` adds the
65
+ /// variable interface: the outgoing interface of the original direction, the
66
+ /// incoming interface of the reply.
67
+ pub(super) fn tuple_lookup(set: &str, linked: bool, reply: bool) -> Vec<Attr> {
68
68
  let mut result = Vec::new();
69
69
  let mut index = 4;
70
70
  if linked {
@@ -82,10 +82,10 @@ fn tuple_lookup(linked: bool, reply: bool) -> Vec<Attr> {
82
82
  result.push(ct_to(19, true, index + 5));
83
83
  result.push(ct_to(20, true, index + 6));
84
84
  result.push(ct_to(12, true, index + 7));
85
- result.push(lookup(TUPLE_SET));
85
+ result.push(lookup(set));
86
86
  result
87
87
  }
88
- fn tuple_key(grant: &HostGrant, link: Option<&LocalLink>) -> Result<Vec<u8>> {
88
+ pub(super) fn tuple_key(grant: &HostGrant, link: Option<&LocalLink>) -> Result<Vec<u8>> {
89
89
  let protocol = word(&[protocol(&grant.protocol)]);
90
90
  let source = address(&grant.source_address)?.to_be_bytes();
91
91
  let destination = address(&grant.destination_address)?.to_be_bytes();
@@ -170,7 +170,7 @@ pub(super) fn admissions(
170
170
  result
171
171
  };
172
172
  let mut established = envelope(false, 2);
173
- established.extend(tuple_lookup(linked, false));
173
+ established.extend(tuple_lookup(TUPLE_SET, linked, false));
174
174
  program.end(chains[0], established, 1)?;
175
175
  for transport in ["udp", "tcp"] {
176
176
  let mut opening = envelope(false, 8);
@@ -178,10 +178,10 @@ pub(super) fn admissions(
178
178
  if transport == "tcp" {
179
179
  opening.extend(opening_tcp());
180
180
  }
181
- opening.extend(tuple_lookup(linked, false));
181
+ opening.extend(tuple_lookup(TUPLE_SET, linked, false));
182
182
  program.end(chains[0], opening, 1)?;
183
183
  }
184
184
  let mut reply = envelope(true, 2);
185
- reply.extend(tuple_lookup(linked, true));
185
+ reply.extend(tuple_lookup(TUPLE_SET, linked, true));
186
186
  program.end(chains[1], reply, 1)
187
187
  }
@@ -7,6 +7,10 @@ pub(super) fn compile(program: &mut Program<'_>, scope: &AllocationPoolGuardScop
7
7
  for name in ["pool_guard", "pool_reply"] {
8
8
  program.chain(name, None)?;
9
9
  }
10
+ // First in the base chains, so nothing earlier in this table accepts past them.
11
+ for owner in &scope.local_tcp_port_owners {
12
+ local_tcp_port_owner(program, owner)?;
13
+ }
10
14
  // The only exceptions: an exact host-origin flow into a pool and its reply,
11
15
  // ahead of the guard. Everything else, including a translated flow that merely
12
16
  // arrives at a granted tuple, still meets the unconditional guard below.
@@ -35,3 +39,44 @@ pub(super) fn compile(program: &mut Program<'_>, scope: &AllocationPoolGuardScop
35
39
  program.end("pool_guard", vec![], -5)?;
36
40
  program.end("pool_reply", vec![], -5)
37
41
  }
42
+
43
+ /// The exact loopback TCP destination: IPv4, TCP, address and port.
44
+ fn local_tcp_port(owner: &LocalTcpPortOwner) -> Result<Vec<Attr>> {
45
+ let mut result = ipv4();
46
+ result.extend(meta(16, vec![protocol("tcp")]));
47
+ result.push(load_payload(1, 16, 4));
48
+ result.push(compare(address(&owner.address)?.to_be_bytes().to_vec()));
49
+ result.extend(current_port(owner.port, false));
50
+ Ok(result)
51
+ }
52
+ /// `meta <key> != value`. A key the packet does not carry breaks the rule.
53
+ fn meta_differs(key: u32, value: Vec<u8>) -> Vec<Attr> {
54
+ vec![
55
+ expr("meta", vec![Attr::u32(1, 1), Attr::u32(2, key)]),
56
+ expr(
57
+ "cmp",
58
+ vec![
59
+ Attr::u32(1, 1),
60
+ Attr::u32(2, 1),
61
+ Attr::nested(3, vec![Attr::bytes(1, value)]),
62
+ ],
63
+ ),
64
+ ]
65
+ }
66
+ /// OUTPUT at filter priority 0 sees the destination after ordinary destination
67
+ /// NAT and rejects, with a TCP reset, every packet from a socket whose file
68
+ /// system uid (as the user namespace owning the network namespace sees it) is
69
+ /// not the owner. A packet without a user socket (a kernel reply, an orphaned
70
+ /// socket's teardown) carries no uid and breaks the rule; it cannot open a
71
+ /// connection. INPUT drops the destination arriving on any interface but
72
+ /// loopback (ifindex 1 in every namespace), which `route_localnet` or
73
+ /// prerouting DNAT could otherwise admit.
74
+ fn local_tcp_port_owner(program: &mut Program<'_>, owner: &LocalTcpPortOwner) -> Result<()> {
75
+ let mut output = local_tcp_port(owner)?;
76
+ output.extend(meta_differs(10, owner.uid.to_ne_bytes().to_vec()));
77
+ output.push(expr("reject", vec![Attr::u32(1, 1)]));
78
+ program.rule("output", output)?;
79
+ let mut input = local_tcp_port(owner)?;
80
+ input.extend(meta_differs(4, 1_u32.to_ne_bytes().to_vec()));
81
+ program.end("input", input, 0)
82
+ }
@@ -50,7 +50,7 @@ fn raw_return(scope: &RouterScope, grant: &Grant, generation: &Generation) -> Re
50
50
  /// The exact veth link of the workload endpoint that holds an endpoint address.
51
51
  /// Validation admits only a current endpoint address, and endpoint prefixes are
52
52
  /// pairwise disjoint, so one publication or host grant never spans two links.
53
- fn endpoint_link<'a>(scope: &'a RouterScope, address: &str) -> Result<&'a LocalLink> {
53
+ pub(super) fn endpoint_link<'a>(scope: &'a RouterScope, address: &str) -> Result<&'a LocalLink> {
54
54
  let target = format!("{address}/32");
55
55
  let endpoint = scope
56
56
  .endpoints
@@ -365,6 +365,7 @@ pub(super) fn compile(program: &mut Program<'_>, scope: &RouterScope, revision:
365
365
  program.chain(name, None)?;
366
366
  }
367
367
  host_grant_sets(program, scope)?;
368
+ workloadgrant::sets(program, scope)?;
368
369
  let private = policy::Policy {
369
370
  schema_version: 1,
370
371
  revision,
@@ -397,6 +398,7 @@ pub(super) fn compile(program: &mut Program<'_>, scope: &RouterScope, revision:
397
398
  if chain == "forward" {
398
399
  published_admissions(program, scope)?;
399
400
  host_grant_admissions(program, scope)?;
401
+ workloadgrant::admissions(program, scope)?;
400
402
  }
401
403
  for expressions in parts.grants {
402
404
  program.rule(chain, expressions)?;