@push.rocks/smartnftables 2.2.0 → 2.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,7 +1,10 @@
1
1
  use super::egress_traffic_tests::{connect, denied, host_policy, ns_ip, roundtrip, socket};
2
2
  use super::*;
3
+ use crate::egress;
3
4
  use crate::policy::{compare, expr, meta};
4
- use std::net::UdpSocket;
5
+ use std::io::{Read, Write};
6
+ use std::net::{TcpListener, TcpStream, UdpSocket};
7
+ use std::time::Duration;
5
8
 
6
9
  const SOURCE: &str = "hs_source";
7
10
  const HOST: &str = "hs_host";
@@ -110,7 +113,7 @@ fn hostile_capture_fixture() {
110
113
  // This is a persistent *foreign fixture* table. Drop its process socket;
111
114
  // native managed-table ownership never adopts or removes this graph.
112
115
  }
113
- fn control(source: &UdpSocket, destination: &str, target: &UdpSocket, allowed: bool) {
116
+ pub(super) fn control(source: &UdpSocket, destination: &str, target: &UdpSocket, allowed: bool) {
114
117
  assert_eq!(source.send_to(b"probe", destination).unwrap(), 5);
115
118
  if allowed {
116
119
  assert_eq!(target.recv(&mut [0; 64]).unwrap(), 5);
@@ -297,3 +300,223 @@ fn v2_host_rejects_foreign_dnat_zone_local_diversion_and_ipv6() {
297
300
  });
298
301
  println!("V2_HOST_PROOF original_protected_dnat_denied=true current_protected_dnat_denied=true nonzero_zone_denied=true input_diversion_denied=true output_spoof_denied=true ipv6_forward_denied=true unleased_port_denied=true exact_platform_allowed=true positive_controls=true");
299
302
  }
303
+
304
+ const PUBLISH_SOURCE: &str = "hp_source";
305
+ const PUBLISH_HOST: &str = "hp_host";
306
+ const PUBLISH_PEER: &str = "hp_peer";
307
+
308
+ pub(super) fn chain_names(owner: &mut Owner) -> Vec<String> {
309
+ owner
310
+ .graph()
311
+ .unwrap()
312
+ .expect("owned table present")
313
+ .chains
314
+ .iter()
315
+ .map(|chain| wire::text(chain, 3).unwrap())
316
+ .collect()
317
+ }
318
+ /// Inbound publication through the uplink: the workload must observe the actual
319
+ /// client address, and the client must observe the published address back.
320
+ pub(super) fn published_udp(client: &UdpSocket, workload: &UdpSocket, published: &str, client_address: &str) {
321
+ let mut bytes = [0; 64];
322
+ assert_eq!(client.send_to(b"query", published).unwrap(), 5);
323
+ let (size, observed) = workload.recv_from(&mut bytes).expect("published delivery");
324
+ assert_eq!(&bytes[..size], b"query");
325
+ assert_eq!(observed.to_string(), client_address);
326
+ assert_eq!(workload.send_to(b"reply", observed).unwrap(), 5);
327
+ let (size, from) = client.recv_from(&mut bytes).expect("published reply");
328
+ assert_eq!(&bytes[..size], b"reply");
329
+ assert_eq!(from.to_string(), published);
330
+ }
331
+ pub(super) fn dark_udp(client: &UdpSocket, workload: &UdpSocket, published: &str) {
332
+ assert_eq!(client.send_to(b"query", published).unwrap(), 5);
333
+ assert!(workload.recv(&mut [0; 64]).is_err());
334
+ }
335
+
336
+ #[test]
337
+ #[ignore = "requires disposable isolated native qualification guest"]
338
+ fn v2_host_publishes_exact_uplink_ports_and_removes_them_with_the_generation() {
339
+ isolated();
340
+ for name in [PUBLISH_SOURCE, PUBLISH_HOST, PUBLISH_PEER] {
341
+ ip(&["netns", "add", name]);
342
+ ns_ip(name, &["link", "set", "lo", "up"]);
343
+ in_namespace(name, || {
344
+ std::fs::write("/proc/sys/net/ipv4/ip_forward", "1").unwrap()
345
+ });
346
+ }
347
+ connect(
348
+ PUBLISH_SOURCE,
349
+ "source",
350
+ "10.240.0.2/30",
351
+ PUBLISH_HOST,
352
+ "router",
353
+ "10.240.0.1/30",
354
+ );
355
+ connect(
356
+ PUBLISH_HOST,
357
+ "uplink",
358
+ "192.0.2.2/24",
359
+ PUBLISH_PEER,
360
+ "underlay",
361
+ "192.0.2.3/24",
362
+ );
363
+ ns_ip(PUBLISH_SOURCE, &["route", "add", "default", "via", "10.240.0.1"]);
364
+ ns_ip(PUBLISH_PEER, &["route", "add", "default", "via", "192.0.2.2"]);
365
+ let workload_udp = socket(PUBLISH_SOURCE, "10.240.0.2:8125");
366
+ let workload_direct = socket(PUBLISH_SOURCE, "10.240.0.2:8444");
367
+ let workload_unpublished = socket(PUBLISH_SOURCE, "10.240.0.2:9999");
368
+ // Every probe uses its own client port: an unreplied conntrack entry would
369
+ // otherwise carry its null translation across the policy change.
370
+ let control_client = socket(PUBLISH_PEER, "192.0.2.3:40010");
371
+ let client_udp = socket(PUBLISH_PEER, "192.0.2.3:40000");
372
+ // Positive controls: without policy the workload is reachable directly, and
373
+ // the published address is not, so every later result is caused by the policy.
374
+ control(&control_client, "10.240.0.2:8444", &workload_direct, true);
375
+ dark_udp(
376
+ &socket(PUBLISH_PEER, "192.0.2.3:40011"),
377
+ &workload_udp,
378
+ "192.0.2.2:8125",
379
+ );
380
+ let (owner, applied, published_policy, names) = in_namespace(PUBLISH_HOST, || {
381
+ let mut policy = host_policy();
382
+ let egress::Scope::HostTransit(scope) = &mut policy.scope else {
383
+ panic!()
384
+ };
385
+ scope.published_ports = vec![
386
+ egress::PublishedPort {
387
+ protocol: "tcp".into(),
388
+ host_port: 443,
389
+ target_port: 8443,
390
+ target_address: "10.240.0.2".into(),
391
+ host_ip: "192.0.2.2".into(),
392
+ },
393
+ egress::PublishedPort {
394
+ protocol: "udp".into(),
395
+ host_port: 8125,
396
+ target_port: 8125,
397
+ target_address: "10.240.0.2".into(),
398
+ host_ip: "192.0.2.2".into(),
399
+ },
400
+ ];
401
+ let mut owner = Owner::new(options("host_published")).unwrap();
402
+ let target: Prepared = policy.clone().prepare().unwrap().into();
403
+ target
404
+ .validate_interfaces()
405
+ .expect("published host fixture binding");
406
+ let applied = owner
407
+ .reconcile(Transition {
408
+ previous: None,
409
+ target,
410
+ })
411
+ .unwrap();
412
+ assert!(owner.inspect().enforced);
413
+ let names = chain_names(&mut owner);
414
+ (owner, applied, policy, names)
415
+ });
416
+ assert!(names.contains(&"pre".to_string()));
417
+ // The fixture observes ingress only, so the translated opening is read where
418
+ // it arrives: the workload side of the handoff.
419
+ let observer = in_namespace(PUBLISH_SOURCE, || packet_fixture::PacketSocket::open("source"));
420
+ published_udp(&client_udp, &workload_udp, "192.0.2.2:8125", "192.0.2.3:40000");
421
+ let listener = in_namespace(PUBLISH_SOURCE, || {
422
+ TcpListener::bind("10.240.0.2:8443").unwrap()
423
+ });
424
+ let mut client = in_namespace(PUBLISH_PEER, || {
425
+ TcpStream::connect_timeout(&"192.0.2.2:443".parse().unwrap(), Duration::from_secs(1))
426
+ .expect("published tcp opening")
427
+ });
428
+ let (mut server, peer) = listener.accept().unwrap();
429
+ assert_eq!(peer.ip().to_string(), "192.0.2.3");
430
+ assert_eq!(client.peer_addr().unwrap().to_string(), "192.0.2.2:443");
431
+ for stream in [&client, &server] {
432
+ stream.set_read_timeout(Some(Duration::from_secs(1))).unwrap();
433
+ }
434
+ client.write_all(b"ping").unwrap();
435
+ server.read_exact(&mut [0; 4]).unwrap();
436
+ server.write_all(b"pong").unwrap();
437
+ client.read_exact(&mut [0; 4]).unwrap();
438
+ // The translated opening is observable on the handoff, with the actual client
439
+ // address preserved and SYN only.
440
+ let syn = observer
441
+ .matching(|packet| {
442
+ packet.len() >= 40
443
+ && packet[9] == 6
444
+ && packet[16..20] == [10, 240, 0, 2]
445
+ && packet[22..24] == 8443_u16.to_be_bytes()
446
+ && packet[33] & 0x17 == 0x02
447
+ })
448
+ .expect("translated opening on the handoff");
449
+ assert_eq!(syn[12..16], [192, 0, 2, 3]);
450
+ // An opening that is not SYN cannot acquire the publication.
451
+ let injector = in_namespace(PUBLISH_PEER, || {
452
+ packet_fixture::PacketSocket::open("underlay")
453
+ });
454
+ let uplink_mac = in_namespace(PUBLISH_HOST, || packet_fixture::mac("uplink"));
455
+ for (port, flags) in [(40050_u16, 0x10_u8), (40051, 0x04), (40052, 0x12)] {
456
+ injector.send(
457
+ uplink_mac,
458
+ &packet_fixture::tcp([192, 0, 2, 3], [192, 0, 2, 2], port, 443, flags),
459
+ );
460
+ // The exact injected flow, never the live connection's own acknowledgements.
461
+ assert!(
462
+ observer
463
+ .matching(|packet| packet.len() >= 40
464
+ && packet[9] == 6
465
+ && packet[16..20] == [10, 240, 0, 2]
466
+ && packet[20..22] == port.to_be_bytes()
467
+ && packet[22..24] == 8443_u16.to_be_bytes()
468
+ && packet[24..28] == 100_u32.to_be_bytes()
469
+ && packet[33] == flags)
470
+ .is_none(),
471
+ "{flags:#04x}"
472
+ );
473
+ }
474
+ // The capture barrier still denies every unpublished path to the workload.
475
+ control(&control_client, "10.240.0.2:8444", &workload_direct, false);
476
+ let unpublished_client = socket(PUBLISH_PEER, "192.0.2.3:40002");
477
+ dark_udp(&unpublished_client, &workload_unpublished, "192.0.2.2:9999");
478
+ assert!(in_namespace(PUBLISH_PEER, || TcpStream::connect_timeout(
479
+ &"192.0.2.2:444".parse().unwrap(),
480
+ Duration::from_secs(1)
481
+ ))
482
+ .is_err());
483
+ drop(client);
484
+ drop(server);
485
+ // Withdrawing the publication is an ordinary generation replacement.
486
+ let (mut owner, rolled_back, names) = in_namespace(PUBLISH_HOST, {
487
+ let mut owner = owner;
488
+ move || {
489
+ let mut withdrawn = published_policy;
490
+ withdrawn.revision = 2;
491
+ let egress::Scope::HostTransit(scope) = &mut withdrawn.scope else {
492
+ panic!()
493
+ };
494
+ scope.published_ports.clear();
495
+ let applied = owner
496
+ .reconcile(Transition {
497
+ previous: Some(applied),
498
+ target: withdrawn.prepare().unwrap().into(),
499
+ })
500
+ .unwrap();
501
+ assert!(owner.inspect().enforced);
502
+ let names = chain_names(&mut owner);
503
+ (owner, applied, names)
504
+ }
505
+ });
506
+ assert!(!names.contains(&"pre".to_string()));
507
+ let dark_client = socket(PUBLISH_PEER, "192.0.2.3:40003");
508
+ dark_udp(&dark_client, &workload_udp, "192.0.2.2:8125");
509
+ assert!(in_namespace(PUBLISH_PEER, || TcpStream::connect_timeout(
510
+ &"192.0.2.2:443".parse().unwrap(),
511
+ Duration::from_secs(1)
512
+ ))
513
+ .is_err());
514
+ control(&control_client, "10.240.0.2:8444", &workload_direct, false);
515
+ in_namespace(PUBLISH_HOST, move || {
516
+ assert!(owner.inspect().enforced);
517
+ owner.release(rolled_back).unwrap();
518
+ });
519
+ let released_client = socket(PUBLISH_PEER, "192.0.2.3:40004");
520
+ dark_udp(&released_client, &workload_udp, "192.0.2.2:8125");
521
+ println!("V2_HOST_PUBLISHED_PROOF udp_published_delivered=true tcp_published_delivered=true client_address_preserved=true reverse_translation=true syn_only_opening=true nonsyn_opening_denied=true unpublished_workload_port_denied=true unpublished_uplink_port_denied=true barrier_positive_control=true withdrawn_chain_absent=true withdrawn_port_dark=true released_port_dark=true");
522
+ }
@@ -167,3 +167,85 @@ fn egress_identity_rejects_unknown_duplicate_or_incomplete_operands() {
167
167
  assert!(rule_identity(&rule(vec![expression(name, extra)])).is_err());
168
168
  }
169
169
  }
170
+
171
+ /// A published host-transit generation must survive the same apply-time graph
172
+ /// identity as the rest of the table: the kernel dump is compared operand by
173
+ /// operand, so destination NAT and its admissions cannot be silently rewritten.
174
+ #[test]
175
+ fn published_host_graph_identity_preserves_destination_nat_and_admission_operands() {
176
+ let compiled = |target_port: u16| {
177
+ let value = crate::egress::tests::published_host(serde_json::json!([
178
+ crate::egress::tests::published_port("tcp", 443, target_port)
179
+ ]));
180
+ let prepared = serde_json::from_value::<crate::egress::Policy>(value)
181
+ .unwrap()
182
+ .prepare()
183
+ .unwrap();
184
+ let mut chains = Vec::new();
185
+ let mut rules = Vec::new();
186
+ for (kind, attributes) in prepared.program("snft_identity").unwrap() {
187
+ if kind == 3 {
188
+ chains.push(super::chain_identity(&attributes).unwrap());
189
+ } else {
190
+ rules.push(rule_identity(&attributes).unwrap());
191
+ }
192
+ }
193
+ (chains, rules)
194
+ };
195
+ let (chains, rules) = compiled(8443);
196
+ assert_eq!(rules.iter().filter(|(chain, _)| chain == "pre").count(), 1);
197
+ assert_eq!(chains.len(), 7);
198
+ let translation = rules.iter().find(|(chain, _)| chain == "pre").unwrap();
199
+ let (_, changed) = compiled(9443);
200
+ let other = changed.iter().find(|(chain, _)| chain == "pre").unwrap();
201
+ assert_ne!(translation.1, other.1);
202
+ assert_eq!(
203
+ rules.iter().filter(|(chain, _)| chain == "forward").count(),
204
+ changed.iter().filter(|(chain, _)| chain == "forward").count()
205
+ );
206
+ }
207
+
208
+ /// A published router generation must survive the same apply-time graph identity
209
+ /// as the rest of the table: the kernel dump is compared operand by operand, so
210
+ /// the workload translation and its admissions cannot be silently rewritten.
211
+ #[test]
212
+ fn published_router_graph_identity_preserves_destination_nat_and_admission_operands() {
213
+ let compiled = |endpoint_port: u16| {
214
+ let value = crate::egress::tests::published_router(serde_json::json!([
215
+ crate::egress::tests::router_published_port("tcp", 443, endpoint_port)
216
+ ]));
217
+ let prepared = serde_json::from_value::<crate::egress::Policy>(value)
218
+ .unwrap()
219
+ .prepare()
220
+ .unwrap();
221
+ let mut chains = Vec::new();
222
+ let mut rules = Vec::new();
223
+ for (kind, attributes) in prepared.program("snft_identity").unwrap() {
224
+ if kind == 3 {
225
+ chains.push(super::chain_identity(&attributes).unwrap());
226
+ } else {
227
+ rules.push(rule_identity(&attributes).unwrap());
228
+ }
229
+ }
230
+ (chains, rules)
231
+ };
232
+ let (chains, rules) = compiled(8443);
233
+ assert_eq!(rules.iter().filter(|(chain, _)| chain == "pre").count(), 1);
234
+ assert_eq!(chains.len(), 16);
235
+ let translation = rules.iter().find(|(chain, _)| chain == "pre").unwrap();
236
+ let (_, changed) = compiled(9443);
237
+ let other = changed.iter().find(|(chain, _)| chain == "pre").unwrap();
238
+ assert_ne!(translation.1, other.1);
239
+ assert_eq!(
240
+ rules.iter().filter(|(chain, _)| chain == "forward").count(),
241
+ changed.iter().filter(|(chain, _)| chain == "forward").count()
242
+ );
243
+ // The handoff ingress classification is part of the same identity.
244
+ assert_eq!(
245
+ rules.iter().filter(|(chain, _)| chain == "raw_return").count(),
246
+ changed
247
+ .iter()
248
+ .filter(|(chain, _)| chain == "raw_return")
249
+ .count()
250
+ );
251
+ }