@push.rocks/smartnftables 2.2.0 → 2.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,393 @@
1
+ use super::egress_traffic_tests::{allocation, binding, connect, host_policy, ns_ip, protection};
2
+ use super::egress_traffic_tests::{roundtrip, socket};
3
+ use super::host_traffic_tests::{chain_names, control, dark_udp, published_udp};
4
+ use super::*;
5
+ use crate::egress;
6
+ use serde_json::json;
7
+ use std::io::{Read, Write};
8
+ use std::net::{TcpListener, TcpStream};
9
+ use std::os::fd::FromRawFd;
10
+ use std::time::Duration;
11
+
12
+ const WORK: &str = "rp_work";
13
+ const ROUTER: &str = "rp_router";
14
+ const HOST: &str = "rp_host";
15
+ const PEER: &str = "rp_peer";
16
+
17
+ /// One workload veth behind the router, one leased generation, and the inbound
18
+ /// publications of the second hop. The first hop is the host-transit publication.
19
+ fn router_policy(published: bool) -> egress::Policy {
20
+ let link = binding("work", "10.241.0.1", "veth");
21
+ let mut generation = allocation();
22
+ generation["conntrackZone"] = json!(17);
23
+ generation["conntrackLabel"] = json!("a".repeat(32));
24
+ // Both leased grants claim the same public address and a destination port a
25
+ // published client may legitimately use as its source port, so the generation
26
+ // and the publications compete for the workload's outgoing packets.
27
+ generation["grants"] = json!([
28
+ {"sourceEndpoint":"work","sourcePrefix":"10.241.0.2/32","destination":{"kind":"public","prefix":"203.0.113.2/32"},
29
+ "protocol":"udp","sourcePort":null,"destinationPort":42000,"sourcePortRange":{"first":10000,"last":10015}},
30
+ {"sourceEndpoint":"work","sourcePrefix":"10.241.0.2/32","destination":{"kind":"public","prefix":"203.0.113.2/32"},
31
+ "protocol":"tcp","sourcePort":null,"destinationPort":443,"sourcePortRange":{"first":10016,"last":10031}}
32
+ ]);
33
+ let ports = if published {
34
+ json!([
35
+ {"protocol":"tcp","transitPort":8443,"endpointPort":9443,
36
+ "endpointAddress":"10.241.0.2","transitSourceAddress":"10.240.0.2"},
37
+ {"protocol":"udp","transitPort":8125,"endpointPort":9125,
38
+ "endpointAddress":"10.241.0.2","transitSourceAddress":"10.240.0.2"}
39
+ ])
40
+ } else {
41
+ json!([])
42
+ };
43
+ serde_json::from_value(json!({"schemaVersion":2,"revision":1,"scope":{"kind":"routerEgress",
44
+ "endpoints":[{"id":"work","interfaceIndex":link["interfaceIndex"],"interfaceName":"work",
45
+ "interfaceKind":"veth","sourcePrefixes":["10.241.0.2/32"]}],
46
+ "rules":[],"links":[link],"handoff":binding("handoff","10.240.0.2","veth"),
47
+ "protection":protection(),"generations":[generation],"publishedPorts":ports}}))
48
+ .unwrap()
49
+ }
50
+ fn sockaddr(value: &str) -> libc::sockaddr_in {
51
+ let parsed: std::net::SocketAddrV4 = value.parse().unwrap();
52
+ let mut raw: libc::sockaddr_in = unsafe { std::mem::zeroed() };
53
+ raw.sin_family = libc::AF_INET as u16;
54
+ raw.sin_port = parsed.port().to_be();
55
+ raw.sin_addr.s_addr = u32::from(*parsed.ip()).to_be();
56
+ raw
57
+ }
58
+ /// A TCP client with an exact source address and port, which std cannot express.
59
+ /// The collision this qualifies is the client's source port, so it must be chosen
60
+ /// and not left to the ephemeral range.
61
+ fn connect_from(source: &str, destination: &str) -> std::io::Result<TcpStream> {
62
+ let raw = unsafe { libc::socket(libc::AF_INET, libc::SOCK_STREAM | libc::SOCK_CLOEXEC, 0) };
63
+ assert!(raw >= 0, "socket: {}", std::io::Error::last_os_error());
64
+ let stream = unsafe { TcpStream::from_raw_fd(raw) };
65
+ let bound = sockaddr(source);
66
+ assert_eq!(
67
+ unsafe {
68
+ libc::bind(
69
+ raw,
70
+ (&bound as *const libc::sockaddr_in).cast(),
71
+ std::mem::size_of_val(&bound) as _,
72
+ )
73
+ },
74
+ 0,
75
+ "bind: {}",
76
+ std::io::Error::last_os_error()
77
+ );
78
+ stream.set_nonblocking(true)?;
79
+ let target = sockaddr(destination);
80
+ if unsafe {
81
+ libc::connect(
82
+ raw,
83
+ (&target as *const libc::sockaddr_in).cast(),
84
+ std::mem::size_of_val(&target) as _,
85
+ )
86
+ } != 0
87
+ {
88
+ let error = std::io::Error::last_os_error();
89
+ if error.raw_os_error() != Some(libc::EINPROGRESS) {
90
+ return Err(error);
91
+ }
92
+ let mut waiting = libc::pollfd {
93
+ fd: raw,
94
+ events: libc::POLLOUT,
95
+ revents: 0,
96
+ };
97
+ assert!(unsafe { libc::poll(&mut waiting, 1, 1000) } >= 0);
98
+ let mut code: libc::c_int = 0;
99
+ let mut size = std::mem::size_of_val(&code) as libc::socklen_t;
100
+ assert_eq!(
101
+ unsafe {
102
+ libc::getsockopt(
103
+ raw,
104
+ libc::SOL_SOCKET,
105
+ libc::SO_ERROR,
106
+ (&mut code as *mut libc::c_int).cast(),
107
+ &mut size,
108
+ )
109
+ },
110
+ 0
111
+ );
112
+ if code != 0 || waiting.revents & libc::POLLOUT == 0 {
113
+ return Err(std::io::Error::from_raw_os_error(if code != 0 {
114
+ code
115
+ } else {
116
+ libc::ETIMEDOUT
117
+ }));
118
+ }
119
+ }
120
+ stream.set_nonblocking(false)?;
121
+ Ok(stream)
122
+ }
123
+ /// The host publications of the first hop: two with a router counterpart, and one
124
+ /// whose transit port the router never publishes.
125
+ fn published_host_policy() -> egress::Policy {
126
+ let mut policy = host_policy();
127
+ let egress::Scope::HostTransit(scope) = &mut policy.scope else {
128
+ panic!()
129
+ };
130
+ scope.published_ports = [("tcp", 443, 8443), ("udp", 8125, 8125), ("udp", 8126, 8126)]
131
+ .into_iter()
132
+ .map(|(protocol, host_port, target_port)| egress::PublishedPort {
133
+ protocol: protocol.into(),
134
+ host_port,
135
+ target_port,
136
+ target_address: "10.240.0.2".into(),
137
+ host_ip: "192.0.2.2".into(),
138
+ })
139
+ .collect();
140
+ policy
141
+ }
142
+
143
+ #[test]
144
+ #[ignore = "requires disposable isolated native qualification guest"]
145
+ fn v2_router_publishes_transit_ports_to_the_workload_and_removes_them_with_the_generation() {
146
+ isolated();
147
+ for name in [WORK, ROUTER, HOST, PEER] {
148
+ ip(&["netns", "add", name]);
149
+ ns_ip(name, &["link", "set", "lo", "up"]);
150
+ in_namespace(name, || {
151
+ std::fs::write("/proc/sys/net/ipv4/ip_forward", "1").unwrap()
152
+ });
153
+ }
154
+ connect(
155
+ WORK,
156
+ "client",
157
+ "10.241.0.2/24",
158
+ ROUTER,
159
+ "work",
160
+ "10.241.0.1/24",
161
+ );
162
+ connect(
163
+ ROUTER,
164
+ "handoff",
165
+ "10.240.0.2/30",
166
+ HOST,
167
+ "router",
168
+ "10.240.0.1/30",
169
+ );
170
+ connect(
171
+ HOST,
172
+ "uplink",
173
+ "192.0.2.2/24",
174
+ PEER,
175
+ "underlay",
176
+ "192.0.2.3/24",
177
+ );
178
+ ns_ip(WORK, &["route", "add", "default", "via", "10.241.0.1"]);
179
+ ns_ip(ROUTER, &["route", "add", "default", "via", "10.240.0.1"]);
180
+ // The unpublished workload route exists throughout, so every later denial is
181
+ // caused by policy and not by a missing route.
182
+ ns_ip(HOST, &["route", "add", "10.241.0.0/24", "via", "10.240.0.2"]);
183
+ ns_ip(
184
+ HOST,
185
+ &["route", "add", "203.0.113.2/32", "via", "192.0.2.3"],
186
+ );
187
+ ns_ip(PEER, &["address", "add", "203.0.113.2/32", "dev", "lo"]);
188
+ ns_ip(PEER, &["route", "add", "default", "via", "192.0.2.2"]);
189
+ let workload_udp = socket(WORK, "10.241.0.2:9125");
190
+ let workload_direct = socket(WORK, "10.241.0.2:8444");
191
+ let workload_unpublished = socket(WORK, "10.241.0.2:9126");
192
+ let work_egress = socket(WORK, "10.241.0.2:40001");
193
+ let public = socket(PEER, "203.0.113.2:42000");
194
+ // Every probe uses its own client port: an unreplied conntrack entry would
195
+ // otherwise carry its null translation across the policy change.
196
+ let control_client = socket(PEER, "192.0.2.3:40010");
197
+ let client_udp = socket(PEER, "192.0.2.3:40000");
198
+ // Positive controls: without policy both hops forward to the workload, and the
199
+ // published address is not reachable, so every later result is caused by policy.
200
+ control(&control_client, "10.241.0.2:8444", &workload_direct, true);
201
+ dark_udp(
202
+ &socket(PEER, "192.0.2.3:40011"),
203
+ &workload_udp,
204
+ "192.0.2.2:8125",
205
+ );
206
+ let (host_owner, host_applied) = in_namespace(HOST, || {
207
+ let mut owner = Owner::new(options("router_published_host")).unwrap();
208
+ let target: Prepared = published_host_policy().prepare().unwrap().into();
209
+ target
210
+ .validate_interfaces()
211
+ .expect("published host fixture binding");
212
+ let applied = owner
213
+ .reconcile(Transition {
214
+ previous: None,
215
+ target,
216
+ })
217
+ .unwrap();
218
+ assert!(owner.inspect().enforced);
219
+ (owner, applied)
220
+ });
221
+ let (router_owner, router_applied, names) = in_namespace(ROUTER, || {
222
+ let mut owner = Owner::new(options("router_published")).unwrap();
223
+ let target: Prepared = router_policy(true).prepare().unwrap().into();
224
+ target
225
+ .validate_interfaces()
226
+ .expect("published router fixture binding");
227
+ let applied = owner
228
+ .reconcile(Transition {
229
+ previous: None,
230
+ target,
231
+ })
232
+ .unwrap();
233
+ assert!(owner.inspect().enforced);
234
+ let names = chain_names(&mut owner);
235
+ (owner, applied, names)
236
+ });
237
+ assert!(names.contains(&"pre".to_string()));
238
+ // Two hops of destination NAT deliver UDP to the workload; the workload sees the
239
+ // actual client and the client sees the published uplink address back.
240
+ published_udp(
241
+ &client_udp,
242
+ &workload_udp,
243
+ "192.0.2.2:8125",
244
+ "192.0.2.3:40000",
245
+ );
246
+ // The fixture observes ingress only, so the translated opening is read where it
247
+ // arrives: the workload side of the last veth.
248
+ let observer = in_namespace(WORK, || packet_fixture::PacketSocket::open("client"));
249
+ let listener = in_namespace(WORK, || TcpListener::bind("10.241.0.2:9443").unwrap());
250
+ let mut client = in_namespace(PEER, || {
251
+ TcpStream::connect_timeout(&"192.0.2.2:443".parse().unwrap(), Duration::from_secs(1))
252
+ .expect("published tcp opening")
253
+ });
254
+ let (mut server, peer) = listener.accept().unwrap();
255
+ assert_eq!(peer.ip().to_string(), "192.0.2.3");
256
+ assert_eq!(client.peer_addr().unwrap().to_string(), "192.0.2.2:443");
257
+ for stream in [&client, &server] {
258
+ stream.set_read_timeout(Some(Duration::from_secs(1))).unwrap();
259
+ }
260
+ client.write_all(b"ping").unwrap();
261
+ server.read_exact(&mut [0; 4]).unwrap();
262
+ server.write_all(b"pong").unwrap();
263
+ client.read_exact(&mut [0; 4]).unwrap();
264
+ let syn = observer
265
+ .matching(|packet| {
266
+ packet.len() >= 40
267
+ && packet[9] == 6
268
+ && packet[16..20] == [10, 241, 0, 2]
269
+ && packet[22..24] == 9443_u16.to_be_bytes()
270
+ && packet[33] & 0x17 == 0x02
271
+ })
272
+ .expect("translated opening at the workload");
273
+ assert_eq!(syn[12..16], [192, 0, 2, 3]);
274
+ // Leased egress is unaffected by the publication in the same generation.
275
+ let outer = roundtrip(&work_egress, &public);
276
+ assert_eq!(outer.ip().to_string(), "192.0.2.2");
277
+ assert!((10000..=10015).contains(&outer.port()));
278
+ // A client whose source port equals a leased grant's destination port on that
279
+ // grant's public address is the exact collision raw classification must
280
+ // survive: the workload's published packets carry the same tuple, and raw runs
281
+ // before conntrack can separate the two flows. The publication must keep its
282
+ // own translation, so the reply arrives from the published address and port.
283
+ published_udp(&public, &workload_udp, "192.0.2.2:8125", "203.0.113.2:42000");
284
+ let mut colliding = in_namespace(PEER, || connect_from("203.0.113.2:443", "192.0.2.2:443"))
285
+ .expect("published tcp opening from a leased destination port");
286
+ let (mut served, observed) = listener.accept().unwrap();
287
+ assert_eq!(observed.to_string(), "203.0.113.2:443");
288
+ assert_eq!(colliding.peer_addr().unwrap().to_string(), "192.0.2.2:443");
289
+ for stream in [&colliding, &served] {
290
+ stream.set_read_timeout(Some(Duration::from_secs(1))).unwrap();
291
+ }
292
+ colliding.write_all(b"ping").unwrap();
293
+ served.read_exact(&mut [0; 4]).unwrap();
294
+ served.write_all(b"pong").unwrap();
295
+ colliding.read_exact(&mut [0; 4]).unwrap();
296
+ drop(colliding);
297
+ drop(served);
298
+ // Inject the second hop directly: only SYN with FIN/RST/ACK clear may open the
299
+ // publication, and an unpublished transit port stays behind the router barrier.
300
+ let injector = in_namespace(HOST, || packet_fixture::PacketSocket::open("router"));
301
+ let handoff_mac = in_namespace(ROUTER, || packet_fixture::mac("handoff"));
302
+ let arrived = |port: u16, flags: u8| {
303
+ observer
304
+ .matching(|packet| {
305
+ packet.len() >= 40
306
+ && packet[9] == 6
307
+ && packet[16..20] == [10, 241, 0, 2]
308
+ && packet[20..22] == port.to_be_bytes()
309
+ && packet[22..24] == 9443_u16.to_be_bytes()
310
+ && packet[24..28] == 100_u32.to_be_bytes()
311
+ && packet[33] == flags
312
+ })
313
+ .is_some()
314
+ };
315
+ for (port, flags) in [(40050_u16, 0x10_u8), (40051, 0x04), (40052, 0x12)] {
316
+ injector.send(
317
+ handoff_mac,
318
+ &packet_fixture::tcp([192, 0, 2, 3], [10, 240, 0, 2], port, 8443, flags),
319
+ );
320
+ assert!(!arrived(port, flags), "{flags:#04x}");
321
+ }
322
+ // The same injector opens the publication with an exact SYN, so the denials
323
+ // above are caused by the opening flags and not by the fixture.
324
+ injector.send(
325
+ handoff_mac,
326
+ &packet_fixture::tcp([192, 0, 2, 3], [10, 240, 0, 2], 40053, 8443, 0x02),
327
+ );
328
+ assert!(arrived(40053, 0x02));
329
+ injector.send(
330
+ handoff_mac,
331
+ &packet_fixture::tcp([192, 0, 2, 3], [10, 240, 0, 2], 40054, 9999, 0x02),
332
+ );
333
+ assert!(observer
334
+ .matching(|packet| packet.len() >= 40
335
+ && packet[9] == 6
336
+ && packet[20..22] == 40054_u16.to_be_bytes()
337
+ && packet[24..28] == 100_u32.to_be_bytes())
338
+ .is_none());
339
+ // A host publication whose transit port the router never published cannot pass
340
+ // the router barrier, while the published port beside it keeps working.
341
+ dark_udp(
342
+ &socket(PEER, "192.0.2.3:40012"),
343
+ &workload_unpublished,
344
+ "192.0.2.2:8126",
345
+ );
346
+ control(&control_client, "10.241.0.2:8444", &workload_direct, false);
347
+ drop(client);
348
+ drop(server);
349
+ // Withdrawing the router publication is an ordinary generation replacement. The
350
+ // host publication stays installed, so this isolates the second hop exactly.
351
+ let (mut router_owner, rolled_back, names) = in_namespace(ROUTER, {
352
+ let mut owner = router_owner;
353
+ move || {
354
+ let mut withdrawn = router_policy(false);
355
+ withdrawn.revision = 2;
356
+ let applied = owner
357
+ .reconcile(Transition {
358
+ previous: Some(router_applied),
359
+ target: withdrawn.prepare().unwrap().into(),
360
+ })
361
+ .unwrap();
362
+ assert!(owner.inspect().enforced);
363
+ let names = chain_names(&mut owner);
364
+ (owner, applied, names)
365
+ }
366
+ });
367
+ assert!(!names.contains(&"pre".to_string()));
368
+ dark_udp(
369
+ &socket(PEER, "192.0.2.3:40003"),
370
+ &workload_udp,
371
+ "192.0.2.2:8125",
372
+ );
373
+ assert!(in_namespace(PEER, || TcpStream::connect_timeout(
374
+ &"192.0.2.2:443".parse().unwrap(),
375
+ Duration::from_secs(1)
376
+ ))
377
+ .is_err());
378
+ in_namespace(ROUTER, move || {
379
+ assert!(router_owner.inspect().enforced);
380
+ router_owner.release(rolled_back).unwrap();
381
+ });
382
+ dark_udp(
383
+ &socket(PEER, "192.0.2.3:40004"),
384
+ &workload_udp,
385
+ "192.0.2.2:8125",
386
+ );
387
+ in_namespace(HOST, move || {
388
+ let mut owner = host_owner;
389
+ assert!(owner.inspect().enforced);
390
+ owner.release(host_applied).unwrap();
391
+ });
392
+ println!("V2_ROUTER_PUBLISHED_PROOF two_hop_udp_delivered=true two_hop_tcp_delivered=true client_address_preserved=true reverse_translation_both_hops=true syn_only_opening=true nonsyn_opening_denied=true injector_positive_control=true unpublished_transit_port_denied=true unpublished_workload_port_denied=true leased_egress_unaffected=true leased_destination_port_client_tcp=true leased_destination_port_client_udp=true barrier_positive_control=true withdrawn_chain_absent=true withdrawn_port_dark=true released_port_dark=true");
393
+ }
@@ -21,6 +21,9 @@ mod packet_fixture;
21
21
  #[path = "owner_host_traffic_tests.rs"]
22
22
  mod host_traffic_tests;
23
23
 
24
+ #[path = "owner_router_traffic_tests.rs"]
25
+ mod router_traffic_tests;
26
+
24
27
  #[path = "owner_poolguard_tests.rs"]
25
28
  mod poolguard_tests;
26
29
 
@@ -3,6 +3,6 @@
3
3
  */
4
4
  export const commitinfo = {
5
5
  name: '@push.rocks/smartnftables',
6
- version: '2.2.0',
6
+ version: '2.4.0',
7
7
  description: 'A TypeScript module for managing nftables rules including NAT, firewall, and rate limiting with a high-level API.'
8
8
  }
@@ -55,6 +55,23 @@ export interface IManagedNftEgressGenerationV2 extends IManagedNftHandoffAllocat
55
55
  grants: IManagedNftEgressGrantV2[];
56
56
  }
57
57
 
58
+ /** One inbound handoff publication, compiled inside the same router generation.
59
+ * It is the second hop of a published host port: the host translates to
60
+ * `transitSourceAddress:transitPort`, this translates on to the workload. */
61
+ export interface IManagedNftRouterPublishedPortV2 {
62
+ protocol: 'tcp' | 'udp';
63
+ /** Published handoff port, 1–65535 and unique per protocol across the scope.
64
+ * It is the `targetPort` of the matching host-transit publication. */
65
+ transitPort: number;
66
+ endpointPort: number;
67
+ /** Exact current workload endpoint address behind one veth endpoint of this
68
+ * scope. Never a router-local address, a gateway or a platform endpoint. */
69
+ endpointAddress: string;
70
+ /** Exact leased transit source address of one current generation, present on
71
+ * `handoff`. No wildcard, secondary-address or route inference. */
72
+ transitSourceAddress: string;
73
+ }
74
+
58
75
  /** One combined private/TUN/local/egress table; never compose a second ACCEPT over v1 terminal drops. */
59
76
  export interface IManagedNftRouterEgressScopeV2 {
60
77
  kind: 'routerEgress';
@@ -66,6 +83,24 @@ export interface IManagedNftRouterEgressScopeV2 {
66
83
  protection: IManagedNftProtectionV2;
67
84
  /** Empty retains private/local enforcement while denying external egress. */
68
85
  generations: IManagedNftEgressGenerationV2[];
86
+ /** Optional inbound destination NAT toward a workload endpoint. Absent and empty
87
+ * are the same canonical policy and keep the compiled graph unchanged. Every entry
88
+ * compiles its own handoff DNAT, the handoff ingress classification and the exact
89
+ * forward and reply admission in this generation, so a failed apply, a target
90
+ * without it, or release removes it with the generation. */
91
+ publishedPorts?: IManagedNftRouterPublishedPortV2[];
92
+ }
93
+
94
+ /** One inbound uplink publication, compiled inside the same host-transit generation. */
95
+ export interface IManagedNftPublishedPortV2 {
96
+ protocol: 'tcp' | 'udp';
97
+ /** Published uplink port, 1–65535 and unique per protocol across the scope. */
98
+ hostPort: number;
99
+ targetPort: number;
100
+ /** Exact leased transit source address of one current allocation; it selects the handoff. */
101
+ targetAddress: string;
102
+ /** Exact current uplink address that publishes this port. No wildcard or route inference. */
103
+ hostIp: string;
69
104
  }
70
105
 
71
106
  /** Shared-host capture barrier and explicit outer SNAT; Docker forwarding remains a separate owner. */
@@ -76,6 +111,11 @@ export interface IManagedNftHostTransitScopeV2 {
76
111
  uplink: IManagedNftLocalLinkV2;
77
112
  /** Exact current address present on uplink. No masquerade or default-route inference. */
78
113
  snatAddress: string;
114
+ /** Optional inbound destination NAT. Absent and empty are the same canonical policy and
115
+ * keep the compiled graph unchanged. Every entry compiles its own uplink DNAT plus the
116
+ * exact forward and reply admission in this generation, so a failed apply, a target
117
+ * without it, or release removes it with the generation. */
118
+ publishedPorts?: IManagedNftPublishedPortV2[];
79
119
  }
80
120
 
81
121
  /** Host-wide IPv4 destination denial for caller-authenticated private allocation pools.