@push.rocks/smartnftables 2.3.0 → 2.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/changelog.md +19 -0
  2. package/dist_rust/smartnftables_linux_amd64_musl +0 -0
  3. package/dist_rust/smartnftables_linux_amd64_musl.tsrust-build.json +5 -5
  4. package/dist_rust/smartnftables_linux_arm64_musl +0 -0
  5. package/dist_rust/smartnftables_linux_arm64_musl.tsrust-build.json +5 -5
  6. package/dist_ts/00_commitinfo_data.js +1 -1
  7. package/dist_ts/classes.manageddockerforwarding.js +3 -3
  8. package/dist_ts/classes.managednftables.d.ts +5 -0
  9. package/dist_ts/classes.managednftables.js +13 -6
  10. package/dist_ts/managed.egress.types.d.ts +53 -1
  11. package/package.json +7 -7
  12. package/readme.md +153 -4
  13. package/rust/src/egress.compile.rs +49 -1
  14. package/rust/src/egress.host.rs +22 -0
  15. package/rust/src/egress.hostgrant.rs +187 -0
  16. package/rust/src/egress.poolguard.rs +7 -0
  17. package/rust/src/egress.router.rs +189 -3
  18. package/rust/src/egress.rs +244 -1
  19. package/rust/src/egress_hostgrant_tests.rs +948 -0
  20. package/rust/src/egress_tests.rs +414 -0
  21. package/rust/src/main.rs +6 -2
  22. package/rust/src/owner.rs +116 -4
  23. package/rust/src/owner_egress_tests.rs +1 -1
  24. package/rust/src/owner_egress_traffic_tests.rs +3 -3
  25. package/rust/src/owner_host_traffic_tests.rs +4 -4
  26. package/rust/src/owner_hostgrant_traffic_tests.rs +280 -0
  27. package/rust/src/owner_identity_tests.rs +125 -1
  28. package/rust/src/owner_poolguard_tests.rs +69 -0
  29. package/rust/src/owner_router_traffic_tests.rs +393 -0
  30. package/rust/src/owner_tests.rs +6 -0
  31. package/rust/src/policy.rs +1 -1
  32. package/rust/src/wire.rs +15 -1
  33. package/ts/00_commitinfo_data.ts +1 -1
  34. package/ts/classes.manageddockerforwarding.ts +2 -2
  35. package/ts/classes.managednftables.ts +13 -5
  36. package/ts/managed.egress.types.ts +55 -1
@@ -0,0 +1,393 @@
1
+ use super::egress_traffic_tests::{allocation, binding, connect, host_policy, ns_ip, protection};
2
+ use super::egress_traffic_tests::{roundtrip, socket};
3
+ use super::host_traffic_tests::{chain_names, control, dark_udp, published_udp};
4
+ use super::*;
5
+ use crate::egress;
6
+ use serde_json::json;
7
+ use std::io::{Read, Write};
8
+ use std::net::{TcpListener, TcpStream};
9
+ use std::os::fd::FromRawFd;
10
+ use std::time::Duration;
11
+
12
+ const WORK: &str = "rp_work";
13
+ const ROUTER: &str = "rp_router";
14
+ const HOST: &str = "rp_host";
15
+ const PEER: &str = "rp_peer";
16
+
17
+ /// One workload veth behind the router, one leased generation, and the inbound
18
+ /// publications of the second hop. The first hop is the host-transit publication.
19
+ fn router_policy(published: bool) -> egress::Policy {
20
+ let link = binding("work", "10.241.0.1", "veth");
21
+ let mut generation = allocation();
22
+ generation["conntrackZone"] = json!(17);
23
+ generation["conntrackLabel"] = json!("a".repeat(32));
24
+ // Both leased grants claim the same public address and a destination port a
25
+ // published client may legitimately use as its source port, so the generation
26
+ // and the publications compete for the workload's outgoing packets.
27
+ generation["grants"] = json!([
28
+ {"sourceEndpoint":"work","sourcePrefix":"10.241.0.2/32","destination":{"kind":"public","prefix":"203.0.113.2/32"},
29
+ "protocol":"udp","sourcePort":null,"destinationPort":42000,"sourcePortRange":{"first":10000,"last":10015}},
30
+ {"sourceEndpoint":"work","sourcePrefix":"10.241.0.2/32","destination":{"kind":"public","prefix":"203.0.113.2/32"},
31
+ "protocol":"tcp","sourcePort":null,"destinationPort":443,"sourcePortRange":{"first":10016,"last":10031}}
32
+ ]);
33
+ let ports = if published {
34
+ json!([
35
+ {"protocol":"tcp","transitPort":8443,"endpointPort":9443,
36
+ "endpointAddress":"10.241.0.2","transitSourceAddress":"10.240.0.2"},
37
+ {"protocol":"udp","transitPort":8125,"endpointPort":9125,
38
+ "endpointAddress":"10.241.0.2","transitSourceAddress":"10.240.0.2"}
39
+ ])
40
+ } else {
41
+ json!([])
42
+ };
43
+ serde_json::from_value(json!({"schemaVersion":2,"revision":1,"scope":{"kind":"routerEgress",
44
+ "endpoints":[{"id":"work","interfaceIndex":link["interfaceIndex"],"interfaceName":"work",
45
+ "interfaceKind":"veth","sourcePrefixes":["10.241.0.2/32"]}],
46
+ "rules":[],"links":[link],"handoff":binding("handoff","10.240.0.2","veth"),
47
+ "protection":protection(),"generations":[generation],"publishedPorts":ports}}))
48
+ .unwrap()
49
+ }
50
+ fn sockaddr(value: &str) -> libc::sockaddr_in {
51
+ let parsed: std::net::SocketAddrV4 = value.parse().unwrap();
52
+ let mut raw: libc::sockaddr_in = unsafe { std::mem::zeroed() };
53
+ raw.sin_family = libc::AF_INET as u16;
54
+ raw.sin_port = parsed.port().to_be();
55
+ raw.sin_addr.s_addr = u32::from(*parsed.ip()).to_be();
56
+ raw
57
+ }
58
+ /// A TCP client with an exact source address and port, which std cannot express.
59
+ /// The collision this qualifies is the client's source port, so it must be chosen
60
+ /// and not left to the ephemeral range.
61
+ pub(super) fn connect_from(source: &str, destination: &str) -> std::io::Result<TcpStream> {
62
+ let raw = unsafe { libc::socket(libc::AF_INET, libc::SOCK_STREAM | libc::SOCK_CLOEXEC, 0) };
63
+ assert!(raw >= 0, "socket: {}", std::io::Error::last_os_error());
64
+ let stream = unsafe { TcpStream::from_raw_fd(raw) };
65
+ let bound = sockaddr(source);
66
+ assert_eq!(
67
+ unsafe {
68
+ libc::bind(
69
+ raw,
70
+ (&bound as *const libc::sockaddr_in).cast(),
71
+ std::mem::size_of_val(&bound) as _,
72
+ )
73
+ },
74
+ 0,
75
+ "bind: {}",
76
+ std::io::Error::last_os_error()
77
+ );
78
+ stream.set_nonblocking(true)?;
79
+ let target = sockaddr(destination);
80
+ if unsafe {
81
+ libc::connect(
82
+ raw,
83
+ (&target as *const libc::sockaddr_in).cast(),
84
+ std::mem::size_of_val(&target) as _,
85
+ )
86
+ } != 0
87
+ {
88
+ let error = std::io::Error::last_os_error();
89
+ if error.raw_os_error() != Some(libc::EINPROGRESS) {
90
+ return Err(error);
91
+ }
92
+ let mut waiting = libc::pollfd {
93
+ fd: raw,
94
+ events: libc::POLLOUT,
95
+ revents: 0,
96
+ };
97
+ assert!(unsafe { libc::poll(&mut waiting, 1, 1000) } >= 0);
98
+ let mut code: libc::c_int = 0;
99
+ let mut size = std::mem::size_of_val(&code) as libc::socklen_t;
100
+ assert_eq!(
101
+ unsafe {
102
+ libc::getsockopt(
103
+ raw,
104
+ libc::SOL_SOCKET,
105
+ libc::SO_ERROR,
106
+ (&mut code as *mut libc::c_int).cast(),
107
+ &mut size,
108
+ )
109
+ },
110
+ 0
111
+ );
112
+ if code != 0 || waiting.revents & libc::POLLOUT == 0 {
113
+ return Err(std::io::Error::from_raw_os_error(if code != 0 {
114
+ code
115
+ } else {
116
+ libc::ETIMEDOUT
117
+ }));
118
+ }
119
+ }
120
+ stream.set_nonblocking(false)?;
121
+ Ok(stream)
122
+ }
123
+ /// The host publications of the first hop: two with a router counterpart, and one
124
+ /// whose transit port the router never publishes.
125
+ fn published_host_policy() -> egress::Policy {
126
+ let mut policy = host_policy();
127
+ let egress::Scope::HostTransit(scope) = &mut policy.scope else {
128
+ panic!()
129
+ };
130
+ scope.published_ports = [("tcp", 443, 8443), ("udp", 8125, 8125), ("udp", 8126, 8126)]
131
+ .into_iter()
132
+ .map(|(protocol, host_port, target_port)| egress::PublishedPort {
133
+ protocol: protocol.into(),
134
+ host_port,
135
+ target_port,
136
+ target_address: "10.240.0.2".into(),
137
+ host_ip: "192.0.2.2".into(),
138
+ })
139
+ .collect();
140
+ policy
141
+ }
142
+
143
+ #[test]
144
+ #[ignore = "requires disposable isolated native qualification guest"]
145
+ fn v2_router_publishes_transit_ports_to_the_workload_and_removes_them_with_the_generation() {
146
+ isolated();
147
+ for name in [WORK, ROUTER, HOST, PEER] {
148
+ ip(&["netns", "add", name]);
149
+ ns_ip(name, &["link", "set", "lo", "up"]);
150
+ in_namespace(name, || {
151
+ std::fs::write("/proc/sys/net/ipv4/ip_forward", "1").unwrap()
152
+ });
153
+ }
154
+ connect(
155
+ WORK,
156
+ "client",
157
+ "10.241.0.2/24",
158
+ ROUTER,
159
+ "work",
160
+ "10.241.0.1/24",
161
+ );
162
+ connect(
163
+ ROUTER,
164
+ "handoff",
165
+ "10.240.0.2/30",
166
+ HOST,
167
+ "router",
168
+ "10.240.0.1/30",
169
+ );
170
+ connect(
171
+ HOST,
172
+ "uplink",
173
+ "192.0.2.2/24",
174
+ PEER,
175
+ "underlay",
176
+ "192.0.2.3/24",
177
+ );
178
+ ns_ip(WORK, &["route", "add", "default", "via", "10.241.0.1"]);
179
+ ns_ip(ROUTER, &["route", "add", "default", "via", "10.240.0.1"]);
180
+ // The unpublished workload route exists throughout, so every later denial is
181
+ // caused by policy and not by a missing route.
182
+ ns_ip(HOST, &["route", "add", "10.241.0.0/24", "via", "10.240.0.2"]);
183
+ ns_ip(
184
+ HOST,
185
+ &["route", "add", "203.0.113.2/32", "via", "192.0.2.3"],
186
+ );
187
+ ns_ip(PEER, &["address", "add", "203.0.113.2/32", "dev", "lo"]);
188
+ ns_ip(PEER, &["route", "add", "default", "via", "192.0.2.2"]);
189
+ let workload_udp = socket(WORK, "10.241.0.2:9125");
190
+ let workload_direct = socket(WORK, "10.241.0.2:8444");
191
+ let workload_unpublished = socket(WORK, "10.241.0.2:9126");
192
+ let work_egress = socket(WORK, "10.241.0.2:40001");
193
+ let public = socket(PEER, "203.0.113.2:42000");
194
+ // Every probe uses its own client port: an unreplied conntrack entry would
195
+ // otherwise carry its null translation across the policy change.
196
+ let control_client = socket(PEER, "192.0.2.3:40010");
197
+ let client_udp = socket(PEER, "192.0.2.3:40000");
198
+ // Positive controls: without policy both hops forward to the workload, and the
199
+ // published address is not reachable, so every later result is caused by policy.
200
+ control(&control_client, "10.241.0.2:8444", &workload_direct, true);
201
+ dark_udp(
202
+ &socket(PEER, "192.0.2.3:40011"),
203
+ &workload_udp,
204
+ "192.0.2.2:8125",
205
+ );
206
+ let (host_owner, host_applied) = in_namespace(HOST, || {
207
+ let mut owner = Owner::new(options("router_published_host")).unwrap();
208
+ let target: Prepared = published_host_policy().prepare().unwrap().into();
209
+ target
210
+ .validate_interfaces()
211
+ .expect("published host fixture binding");
212
+ let applied = owner
213
+ .reconcile(Transition {
214
+ previous: None,
215
+ target,
216
+ })
217
+ .unwrap();
218
+ assert!(owner.inspect().enforced);
219
+ (owner, applied)
220
+ });
221
+ let (router_owner, router_applied, names) = in_namespace(ROUTER, || {
222
+ let mut owner = Owner::new(options("router_published")).unwrap();
223
+ let target: Prepared = router_policy(true).prepare().unwrap().into();
224
+ target
225
+ .validate_interfaces()
226
+ .expect("published router fixture binding");
227
+ let applied = owner
228
+ .reconcile(Transition {
229
+ previous: None,
230
+ target,
231
+ })
232
+ .unwrap();
233
+ assert!(owner.inspect().enforced);
234
+ let names = chain_names(&mut owner);
235
+ (owner, applied, names)
236
+ });
237
+ assert!(names.contains(&"pre".to_string()));
238
+ // Two hops of destination NAT deliver UDP to the workload; the workload sees the
239
+ // actual client and the client sees the published uplink address back.
240
+ published_udp(
241
+ &client_udp,
242
+ &workload_udp,
243
+ "192.0.2.2:8125",
244
+ "192.0.2.3:40000",
245
+ );
246
+ // The fixture observes ingress only, so the translated opening is read where it
247
+ // arrives: the workload side of the last veth.
248
+ let observer = in_namespace(WORK, || packet_fixture::PacketSocket::open("client"));
249
+ let listener = in_namespace(WORK, || TcpListener::bind("10.241.0.2:9443").unwrap());
250
+ let mut client = in_namespace(PEER, || {
251
+ TcpStream::connect_timeout(&"192.0.2.2:443".parse().unwrap(), Duration::from_secs(1))
252
+ .expect("published tcp opening")
253
+ });
254
+ let (mut server, peer) = listener.accept().unwrap();
255
+ assert_eq!(peer.ip().to_string(), "192.0.2.3");
256
+ assert_eq!(client.peer_addr().unwrap().to_string(), "192.0.2.2:443");
257
+ for stream in [&client, &server] {
258
+ stream.set_read_timeout(Some(Duration::from_secs(1))).unwrap();
259
+ }
260
+ client.write_all(b"ping").unwrap();
261
+ server.read_exact(&mut [0; 4]).unwrap();
262
+ server.write_all(b"pong").unwrap();
263
+ client.read_exact(&mut [0; 4]).unwrap();
264
+ let syn = observer
265
+ .matching(|packet| {
266
+ packet.len() >= 40
267
+ && packet[9] == 6
268
+ && packet[16..20] == [10, 241, 0, 2]
269
+ && packet[22..24] == 9443_u16.to_be_bytes()
270
+ && packet[33] & 0x17 == 0x02
271
+ })
272
+ .expect("translated opening at the workload");
273
+ assert_eq!(syn[12..16], [192, 0, 2, 3]);
274
+ // Leased egress is unaffected by the publication in the same generation.
275
+ let outer = roundtrip(&work_egress, &public);
276
+ assert_eq!(outer.ip().to_string(), "192.0.2.2");
277
+ assert!((10000..=10015).contains(&outer.port()));
278
+ // A client whose source port equals a leased grant's destination port on that
279
+ // grant's public address is the exact collision raw classification must
280
+ // survive: the workload's published packets carry the same tuple, and raw runs
281
+ // before conntrack can separate the two flows. The publication must keep its
282
+ // own translation, so the reply arrives from the published address and port.
283
+ published_udp(&public, &workload_udp, "192.0.2.2:8125", "203.0.113.2:42000");
284
+ let mut colliding = in_namespace(PEER, || connect_from("203.0.113.2:443", "192.0.2.2:443"))
285
+ .expect("published tcp opening from a leased destination port");
286
+ let (mut served, observed) = listener.accept().unwrap();
287
+ assert_eq!(observed.to_string(), "203.0.113.2:443");
288
+ assert_eq!(colliding.peer_addr().unwrap().to_string(), "192.0.2.2:443");
289
+ for stream in [&colliding, &served] {
290
+ stream.set_read_timeout(Some(Duration::from_secs(1))).unwrap();
291
+ }
292
+ colliding.write_all(b"ping").unwrap();
293
+ served.read_exact(&mut [0; 4]).unwrap();
294
+ served.write_all(b"pong").unwrap();
295
+ colliding.read_exact(&mut [0; 4]).unwrap();
296
+ drop(colliding);
297
+ drop(served);
298
+ // Inject the second hop directly: only SYN with FIN/RST/ACK clear may open the
299
+ // publication, and an unpublished transit port stays behind the router barrier.
300
+ let injector = in_namespace(HOST, || packet_fixture::PacketSocket::open("router"));
301
+ let handoff_mac = in_namespace(ROUTER, || packet_fixture::mac("handoff"));
302
+ let arrived = |port: u16, flags: u8| {
303
+ observer
304
+ .matching(|packet| {
305
+ packet.len() >= 40
306
+ && packet[9] == 6
307
+ && packet[16..20] == [10, 241, 0, 2]
308
+ && packet[20..22] == port.to_be_bytes()
309
+ && packet[22..24] == 9443_u16.to_be_bytes()
310
+ && packet[24..28] == 100_u32.to_be_bytes()
311
+ && packet[33] == flags
312
+ })
313
+ .is_some()
314
+ };
315
+ for (port, flags) in [(40050_u16, 0x10_u8), (40051, 0x04), (40052, 0x12)] {
316
+ injector.send(
317
+ handoff_mac,
318
+ &packet_fixture::tcp([192, 0, 2, 3], [10, 240, 0, 2], port, 8443, flags),
319
+ );
320
+ assert!(!arrived(port, flags), "{flags:#04x}");
321
+ }
322
+ // The same injector opens the publication with an exact SYN, so the denials
323
+ // above are caused by the opening flags and not by the fixture.
324
+ injector.send(
325
+ handoff_mac,
326
+ &packet_fixture::tcp([192, 0, 2, 3], [10, 240, 0, 2], 40053, 8443, 0x02),
327
+ );
328
+ assert!(arrived(40053, 0x02));
329
+ injector.send(
330
+ handoff_mac,
331
+ &packet_fixture::tcp([192, 0, 2, 3], [10, 240, 0, 2], 40054, 9999, 0x02),
332
+ );
333
+ assert!(observer
334
+ .matching(|packet| packet.len() >= 40
335
+ && packet[9] == 6
336
+ && packet[20..22] == 40054_u16.to_be_bytes()
337
+ && packet[24..28] == 100_u32.to_be_bytes())
338
+ .is_none());
339
+ // A host publication whose transit port the router never published cannot pass
340
+ // the router barrier, while the published port beside it keeps working.
341
+ dark_udp(
342
+ &socket(PEER, "192.0.2.3:40012"),
343
+ &workload_unpublished,
344
+ "192.0.2.2:8126",
345
+ );
346
+ control(&control_client, "10.241.0.2:8444", &workload_direct, false);
347
+ drop(client);
348
+ drop(server);
349
+ // Withdrawing the router publication is an ordinary generation replacement. The
350
+ // host publication stays installed, so this isolates the second hop exactly.
351
+ let (mut router_owner, rolled_back, names) = in_namespace(ROUTER, {
352
+ let mut owner = router_owner;
353
+ move || {
354
+ let mut withdrawn = router_policy(false);
355
+ withdrawn.revision = 2;
356
+ let applied = owner
357
+ .reconcile(Transition {
358
+ previous: Some(router_applied),
359
+ target: withdrawn.prepare().unwrap().into(),
360
+ })
361
+ .unwrap();
362
+ assert!(owner.inspect().enforced);
363
+ let names = chain_names(&mut owner);
364
+ (owner, applied, names)
365
+ }
366
+ });
367
+ assert!(!names.contains(&"pre".to_string()));
368
+ dark_udp(
369
+ &socket(PEER, "192.0.2.3:40003"),
370
+ &workload_udp,
371
+ "192.0.2.2:8125",
372
+ );
373
+ assert!(in_namespace(PEER, || TcpStream::connect_timeout(
374
+ &"192.0.2.2:443".parse().unwrap(),
375
+ Duration::from_secs(1)
376
+ ))
377
+ .is_err());
378
+ in_namespace(ROUTER, move || {
379
+ assert!(router_owner.inspect().enforced);
380
+ router_owner.release(rolled_back).unwrap();
381
+ });
382
+ dark_udp(
383
+ &socket(PEER, "192.0.2.3:40004"),
384
+ &workload_udp,
385
+ "192.0.2.2:8125",
386
+ );
387
+ in_namespace(HOST, move || {
388
+ let mut owner = host_owner;
389
+ assert!(owner.inspect().enforced);
390
+ owner.release(host_applied).unwrap();
391
+ });
392
+ println!("V2_ROUTER_PUBLISHED_PROOF two_hop_udp_delivered=true two_hop_tcp_delivered=true client_address_preserved=true reverse_translation_both_hops=true syn_only_opening=true nonsyn_opening_denied=true injector_positive_control=true unpublished_transit_port_denied=true unpublished_workload_port_denied=true leased_egress_unaffected=true leased_destination_port_client_tcp=true leased_destination_port_client_udp=true barrier_positive_control=true withdrawn_chain_absent=true withdrawn_port_dark=true released_port_dark=true");
393
+ }
@@ -21,9 +21,15 @@ mod packet_fixture;
21
21
  #[path = "owner_host_traffic_tests.rs"]
22
22
  mod host_traffic_tests;
23
23
 
24
+ #[path = "owner_router_traffic_tests.rs"]
25
+ mod router_traffic_tests;
26
+
24
27
  #[path = "owner_poolguard_tests.rs"]
25
28
  mod poolguard_tests;
26
29
 
30
+ #[path = "owner_hostgrant_traffic_tests.rs"]
31
+ mod hostgrant_traffic_tests;
32
+
27
33
  #[path = "owner_detach_tests.rs"]
28
34
  mod detach_tests;
29
35
 
@@ -195,7 +195,7 @@ impl Policy {
195
195
  };
196
196
  // Compile during preparation; an oversized native batch never reaches the kernel.
197
197
  let program = prepared.program(&format!("snft_{}", "x".repeat(59)))?;
198
- // Reserve enough of the 240 KiB batch for deleting a previous maximum
198
+ // Reserve enough of the batch for deleting a previous maximum
199
199
  // graph, then installing this complete graph in the same transaction.
200
200
  if program.len() > 768
201
201
  || program
package/rust/src/wire.rs CHANGED
@@ -24,6 +24,10 @@ impl Attr {
24
24
  pub fn nested(kind: u16, value: Vec<Attr>) -> Self {
25
25
  Self::bytes(kind | 0x8000, encode_attrs(&value))
26
26
  }
27
+ /// Netlink attribute lengths are 16 bits; a longer value cannot be encoded.
28
+ pub fn encodable(&self) -> bool {
29
+ self.value.len() <= usize::from(u16::MAX) - 4
30
+ }
27
31
  pub fn id(&self) -> u16 {
28
32
  self.kind & 0x3fff
29
33
  }
@@ -31,6 +35,9 @@ impl Attr {
31
35
  pub fn encode_attrs(attrs: &[Attr]) -> Vec<u8> {
32
36
  let mut result = Vec::new();
33
37
  for attr in attrs {
38
+ // Every caller bounds its attributes far below this; a truncated length
39
+ // would make the kernel parse a different message.
40
+ assert!(attr.encodable(), "netlink attribute exceeds 65,531 bytes");
34
41
  let len = 4 + attr.value.len();
35
42
  result.extend((len as u16).to_ne_bytes());
36
43
  result.extend(attr.kind.to_ne_bytes());
@@ -117,6 +124,13 @@ struct Message {
117
124
  sequence: u32,
118
125
  payload: Vec<u8>,
119
126
  }
127
+ /// One replacement batch: handle-only deletion of a previous maximum graph
128
+ /// (768 operations, about 96 KB), a 100,000-byte target rule graph and 100,000
129
+ /// bytes of target set elements, with framing. Netlink refuses a message larger
130
+ /// than the socket send buffer; Linux's default `net.core.wmem_max` of 212,992
131
+ /// gives a 425,984-byte buffer, and this limit needs at least 160,016.
132
+ pub const BATCH_BYTES: usize = 320_000;
133
+
120
134
  pub struct Socket {
121
135
  fd: OwnedFd,
122
136
  pub port: u32,
@@ -213,7 +227,7 @@ impl Socket {
213
227
  Ok((sequence, bytes))
214
228
  }
215
229
  fn send(&self, bytes: &[u8]) -> Result<()> {
216
- if bytes.len() > 240_000 {
230
+ if bytes.len() > BATCH_BYTES {
217
231
  return Err(Error::Invalid);
218
232
  }
219
233
  let mut address: libc::sockaddr_nl = unsafe { std::mem::zeroed() };
@@ -3,6 +3,6 @@
3
3
  */
4
4
  export const commitinfo = {
5
5
  name: '@push.rocks/smartnftables',
6
- version: '2.3.0',
6
+ version: '2.5.0',
7
7
  description: 'A TypeScript module for managing nftables rules including NAT, firewall, and rate limiting with a high-level API.'
8
8
  }
@@ -1,5 +1,5 @@
1
1
  import * as plugins from './plugins.js';
2
- import { snapshot, ManagedNftablesError } from './classes.managednftables.js';
2
+ import { snapshot, managedIpcBytes, ManagedNftablesError } from './classes.managednftables.js';
3
3
  import type { IManagedDockerForwardingOptions, IManagedDockerForwardingStatus, IManagedDockerForwardingPolicy, IPreparedManagedDockerForwardingPolicy, IManagedDockerForwardingTransition, IAppliedManagedDockerForwardingPolicy, TManagedDockerForwardingCommands } from './managed.docker.types.js';
4
4
 
5
5
  /** One node-level Docker DOCKER-USER contribution. Caller owns durable intents and handoff lifetimes. */
@@ -39,7 +39,7 @@ export class ManagedDockerForwarding {
39
39
  localPaths: [], searchSystemPath: false,
40
40
  cliArgs: ['--docker-forwarding', ...(this.#options.networkNamespaceFd === undefined ? [] : ['--network-namespace-fd', '3'])],
41
41
  inheritedFileDescriptors: this.#options.networkNamespaceFd === undefined ? undefined : [this.#options.networkNamespaceFd],
42
- readyTimeoutMs: 3000, requestTimeoutMs: 30_000, maxPayloadSize: 262_144,
42
+ readyTimeoutMs: 3000, requestTimeoutMs: 30_000, maxPayloadSize: managedIpcBytes,
43
43
  maxPendingRequestsByMethod: { prepareDockerForwarding: 1, openDockerForwarding: 1, reconcileDockerForwarding: 1, inspectDockerForwarding: 1, releaseDockerForwarding: 1, closeDockerForwarding: 1 },
44
44
  });
45
45
  this.#bridge.on('exit', () => {
@@ -6,16 +6,24 @@ export class ManagedNftablesError extends Error {
6
6
  constructor(public readonly code: string) { super(`Managed nftables ${code}.`); }
7
7
  }
8
8
 
9
+ /** Bounds of the private native IPC. A status carries up to three complete policies (applied,
10
+ * pending previous and pending target); a policy with the maximum 1024 host grants serializes to
11
+ * about 111 KB and 5,300 values, so these bounds hold three of them with room for the rest of a
12
+ * large scope. */
13
+ export const managedIpcBytes = 1_048_576;
14
+ const snapshotBytes = 1_000_000;
15
+ const snapshotValues = 65_536;
16
+
9
17
  /** Capture inert input without evaluating getters, proxies, coercions or toJSON. */
10
18
  export function snapshot<T>(input: T): T {
11
19
  let count = 0;
12
20
  let stringBytes = 0;
13
21
  const string = (value: string): string => {
14
- if (value.length > 240_000 || (stringBytes += Buffer.byteLength(value)) > 240_000) throw new ManagedNftablesError('INVALID');
22
+ if (value.length > snapshotBytes || (stringBytes += Buffer.byteLength(value)) > snapshotBytes) throw new ManagedNftablesError('INVALID');
15
23
  return value;
16
24
  };
17
25
  const copy = (value: unknown, depth: number): unknown => {
18
- if (++count > 16_384 || depth > 20) throw new ManagedNftablesError('INVALID');
26
+ if (++count > snapshotValues || depth > 20) throw new ManagedNftablesError('INVALID');
19
27
  if (typeof value === 'string') return string(value);
20
28
  if (value === null || typeof value === 'boolean') return value;
21
29
  if (typeof value === 'number' && Number.isSafeInteger(value)) return value;
@@ -23,7 +31,7 @@ export function snapshot<T>(input: T): T {
23
31
  const array = Array.isArray(value);
24
32
  if (Object.getPrototypeOf(value) !== (array ? Array.prototype : Object.prototype)) throw new ManagedNftablesError('INVALID');
25
33
  const keys = Reflect.ownKeys(value);
26
- if (keys.length > 16_384 - count) throw new ManagedNftablesError('INVALID');
34
+ if (keys.length > snapshotValues - count) throw new ManagedNftablesError('INVALID');
27
35
  for (const key of keys) if (typeof key === 'string') string(key);
28
36
  const descriptors = Object.getOwnPropertyDescriptors(value);
29
37
  const result: Record<string, unknown> | unknown[] = array ? [] : {};
@@ -38,7 +46,7 @@ export function snapshot<T>(input: T): T {
38
46
  return result;
39
47
  };
40
48
  const result = copy(input, 0) as T;
41
- if (Buffer.byteLength(JSON.stringify(result)) > 240_000) throw new ManagedNftablesError('INVALID');
49
+ if (Buffer.byteLength(JSON.stringify(result)) > snapshotBytes) throw new ManagedNftablesError('INVALID');
42
50
  return result;
43
51
  }
44
52
 
@@ -97,7 +105,7 @@ export class ManagedNftables<TPolicy extends TManagedNftPolicy = IManagedNftPoli
97
105
  localPaths: [], searchSystemPath: false,
98
106
  cliArgs: ['--management', ...(this.#options.networkNamespaceFd === undefined ? [] : ['--network-namespace-fd', '3'])],
99
107
  inheritedFileDescriptors: this.#options.networkNamespaceFd === undefined ? undefined : [this.#options.networkNamespaceFd],
100
- readyTimeoutMs: 3000, requestTimeoutMs: 30_000, maxPayloadSize: 262_144,
108
+ readyTimeoutMs: 3000, requestTimeoutMs: 30_000, maxPayloadSize: managedIpcBytes,
101
109
  maxPendingRequestsByMethod: { preparePolicy: 1, openOwner: 1, reconcilePolicy: 1, inspectPolicy: 1, releasePolicy: 1, releasePolicyTransition: 1, detachPolicy: 1, closeOwner: 1 },
102
110
  });
103
111
  this.#bridge.on('exit', () => {
@@ -55,6 +55,39 @@ export interface IManagedNftEgressGenerationV2 extends IManagedNftHandoffAllocat
55
55
  grants: IManagedNftEgressGrantV2[];
56
56
  }
57
57
 
58
+ /** One inbound handoff publication, compiled inside the same router generation.
59
+ * It is the second hop of a published host port: the host translates to
60
+ * `transitSourceAddress:transitPort`, this translates on to the workload. */
61
+ export interface IManagedNftRouterPublishedPortV2 {
62
+ protocol: 'tcp' | 'udp';
63
+ /** Published handoff port, 1–65535 and unique per protocol across the scope.
64
+ * It is the `targetPort` of the matching host-transit publication. */
65
+ transitPort: number;
66
+ endpointPort: number;
67
+ /** Exact current workload endpoint address behind one veth endpoint of this
68
+ * scope. Never a router-local address, a gateway or a platform endpoint. */
69
+ endpointAddress: string;
70
+ /** Exact leased transit source address of one current generation, present on
71
+ * `handoff`. No wildcard, secondary-address or route inference. */
72
+ transitSourceAddress: string;
73
+ }
74
+
75
+ /** One exact host-origin flow: the node's own host network namespace dials one port of one local
76
+ * workload at its lease address, from the transit host address of the current handoff. Compile the
77
+ * same grant into every scope the flow crosses (`allocationPoolGuard` and `hostTransit` on the host,
78
+ * `routerEgress` in the router namespace); each scope binds it to the authority it proves. It admits
79
+ * that tuple and its replies in the default conntrack zone and nothing else: no loopback, no
80
+ * translation, no source-port selection and no workload-origin flow. */
81
+ export interface IManagedNftHostGrant {
82
+ protocol: 'tcp' | 'udp';
83
+ /** Exact unicast address, never a prefix: the host's own address on the handoff. */
84
+ sourceAddress: string;
85
+ /** Exact unicast workload lease address, never a prefix. */
86
+ destinationAddress: string;
87
+ /** The workload's exact listening port, 1–65535; never a range. */
88
+ destinationPort: number;
89
+ }
90
+
58
91
  /** One combined private/TUN/local/egress table; never compose a second ACCEPT over v1 terminal drops. */
59
92
  export interface IManagedNftRouterEgressScopeV2 {
60
93
  kind: 'routerEgress';
@@ -66,6 +99,18 @@ export interface IManagedNftRouterEgressScopeV2 {
66
99
  protection: IManagedNftProtectionV2;
67
100
  /** Empty retains private/local enforcement while denying external egress. */
68
101
  generations: IManagedNftEgressGenerationV2[];
102
+ /** Optional inbound destination NAT toward a workload endpoint. Absent and empty
103
+ * are the same canonical policy and keep the compiled graph unchanged. Every entry
104
+ * compiles its own handoff DNAT, the handoff ingress classification and the exact
105
+ * forward and reply admission in this generation, so a failed apply, a target
106
+ * without it, or release removes it with the generation. */
107
+ publishedPorts?: IManagedNftRouterPublishedPortV2[];
108
+ /** Optional host-origin grants forwarded from the handoff to a workload endpoint. Each
109
+ * `destinationAddress` is an exact current veth endpoint address of this scope; each
110
+ * `sourceAddress` is protected space that is neither router-local, a platform endpoint nor a
111
+ * workload. Absent and empty are the same canonical policy, digest and compiled graph; at most 1024,
112
+ * held in a named set behind a constant number of rules. */
113
+ hostGrants?: IManagedNftHostGrant[];
69
114
  }
70
115
 
71
116
  /** One inbound uplink publication, compiled inside the same host-transit generation. */
@@ -93,15 +138,24 @@ export interface IManagedNftHostTransitScopeV2 {
93
138
  * exact forward and reply admission in this generation, so a failed apply, a target
94
139
  * without it, or release removes it with the generation. */
95
140
  publishedPorts?: IManagedNftPublishedPortV2[];
141
+ /** Optional host-origin grants leaving through a handoff. Each `sourceAddress` is an exact current
142
+ * address of exactly one handoff link; each `destinationAddress` is protected space that is neither
143
+ * host-local, a platform endpoint nor a leased transit source address. Absent and empty are the
144
+ * same canonical policy; at most 1024, held in a named set behind a constant number of rules. */
145
+ hostGrants?: IManagedNftHostGrant[];
96
146
  }
97
147
 
98
148
  /** Host-wide IPv4 destination denial for caller-authenticated private allocation pools.
99
- * This has no link dependency or allow grants. It is not allocation-release or boot-order proof. */
149
+ * It has no link dependency; exact host grants are its only exceptions. It is not allocation-release or boot-order proof. */
100
150
  export interface IManagedNftAllocationPoolGuardScopeV2 {
101
151
  kind: 'allocationPoolGuard';
102
152
  authorityDigest: string;
103
153
  /** Complete current pool list: 1–64 canonical, disjoint RFC1918 prefixes. */
104
154
  prefixes: string[];
155
+ /** The only exceptions to the guard: exact host-origin flows whose `destinationAddress` lies in a
156
+ * listed pool, and their replies. Absent and empty are the same canonical policy; at most 1024,
157
+ * held in a named set behind a constant number of rules. */
158
+ hostGrants?: IManagedNftHostGrant[];
105
159
  }
106
160
 
107
161
  export interface IManagedNftPolicyV2 {