@push.rocks/smartnftables 2.1.0 → 2.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -110,6 +110,15 @@ pub struct HostHandoff {
110
110
  pub link: LocalLink,
111
111
  pub allocations: Vec<Allocation>,
112
112
  }
113
+ #[derive(Clone, Debug, Deserialize, Serialize, PartialEq, Eq, PartialOrd, Ord)]
114
+ #[serde(rename_all = "camelCase", deny_unknown_fields)]
115
+ pub struct PublishedPort {
116
+ pub protocol: String,
117
+ pub host_port: u16,
118
+ pub target_port: u16,
119
+ pub target_address: String,
120
+ pub host_ip: String,
121
+ }
113
122
  #[derive(Clone, Debug, Deserialize, Serialize, PartialEq, Eq)]
114
123
  #[serde(rename_all = "camelCase", deny_unknown_fields)]
115
124
  pub struct HostScope {
@@ -117,6 +126,10 @@ pub struct HostScope {
117
126
  pub handoffs: Vec<HostHandoff>,
118
127
  pub uplink: LocalLink,
119
128
  pub snat_address: String,
129
+ /// Absent and empty are the same canonical policy, so an unpublished scope
130
+ /// keeps its exact previous digest and compiled graph.
131
+ #[serde(default, skip_serializing_if = "Vec::is_empty")]
132
+ pub published_ports: Vec<PublishedPort>,
120
133
  }
121
134
  #[derive(Clone, Debug, Deserialize, Serialize, PartialEq, Eq)]
122
135
  #[serde(rename_all = "camelCase", deny_unknown_fields)]
@@ -556,6 +569,53 @@ fn normalize_pool_guard(value: &mut AllocationPoolGuardScope) -> Result<()> {
556
569
  }
557
570
  Ok(())
558
571
  }
572
+ /// Inbound publication is bounded by the same authority the scope already proves:
573
+ /// an exact current uplink address, an exact leased transit source address, and no
574
+ /// overlap with the leased outbound source ports translated to that same address.
575
+ fn normalize_published(value: &mut HostScope) -> Result<()> {
576
+ let mut ports = std::mem::take(&mut value.published_ports);
577
+ require(ports.len() <= 64)?;
578
+ ports.sort();
579
+ for (index, port) in ports.iter().enumerate() {
580
+ require(
581
+ protocol(&port.protocol)
582
+ && port.host_port > 0
583
+ && port.target_port > 0
584
+ && (index == 0
585
+ || ports[index - 1].protocol != port.protocol
586
+ || ports[index - 1].host_port != port.host_port),
587
+ )?;
588
+ // rtnetlink verifies this exact uplink address at apply, recovery and
589
+ // inspection. No wildcard, secondary-link or default-route inference.
590
+ require(value.uplink.required_ipv4_addresses.contains(&port.host_ip))?;
591
+ // Only a current leased transit source address can be published; that
592
+ // binding is what proves the exact handoff link of the forward admission.
593
+ require(
594
+ value
595
+ .handoffs
596
+ .iter()
597
+ .flat_map(|handoff| handoff.allocations.iter())
598
+ .any(|allocation| allocation.transit_source_address == port.target_address),
599
+ )?;
600
+ if port.host_ip == value.snat_address {
601
+ for handoff in &value.handoffs {
602
+ for allocation in &handoff.allocations {
603
+ for range in &allocation.source_port_ranges {
604
+ // Outer SNAT translates to this same address, so a published
605
+ // port inside a leased outbound range is a translation conflict.
606
+ require(
607
+ range.protocol != port.protocol
608
+ || port.host_port < range.first
609
+ || port.host_port > range.last,
610
+ )?;
611
+ }
612
+ }
613
+ }
614
+ }
615
+ }
616
+ value.published_ports = ports;
617
+ Ok(())
618
+ }
559
619
  fn normalize_host(value: &mut HostScope) -> Result<()> {
560
620
  normalize_protection(&mut value.protection)?;
561
621
  normalize_link(&mut value.uplink)?;
@@ -622,6 +682,7 @@ fn normalize_host(value: &mut HostScope) -> Result<()> {
622
682
  }
623
683
  }
624
684
  }
685
+ normalize_published(value)?;
625
686
  unique_links(
626
687
  value
627
688
  .handoffs
@@ -1,4 +1,4 @@
1
- use super::{Policy, Scope};
1
+ use super::{Policy, Prepared, Scope};
2
2
  use serde_json::{json, Value};
3
3
 
4
4
  fn link(index: u32, name: &str, kind: &str, address: &str) -> Value {
@@ -483,3 +483,196 @@ fn allocation_pool_guard_maximum_fits_the_atomic_graph_without_link_bindings() {
483
483
  let excessive: Vec<_> = (0..65).map(|part| format!("10.{part}.0.0/16")).collect();
484
484
  assert!(normalized(changed(pool_guard(), "/scope/prefixes", json!(excessive))).is_err());
485
485
  }
486
+
487
+ pub(crate) fn published_port(protocol: &str, host_port: u16, target_port: u16) -> Value {
488
+ json!({"protocol":protocol,"hostPort":host_port,"targetPort":target_port,
489
+ "targetAddress":"10.240.0.2","hostIp":"192.0.2.2"})
490
+ }
491
+ pub(crate) fn published_host(ports: Value) -> Value {
492
+ let mut value = host();
493
+ value["scope"]["publishedPorts"] = ports;
494
+ value
495
+ }
496
+ fn prepared(value: Value) -> Prepared {
497
+ normalized(value).unwrap().prepare().unwrap()
498
+ }
499
+ fn program_bytes(prepared: &Prepared, table: &str) -> Vec<u8> {
500
+ let mut bytes = Vec::new();
501
+ for (kind, attributes) in prepared.program(table).unwrap() {
502
+ let encoded = crate::wire::encode_attrs(&attributes);
503
+ bytes.extend(kind.to_be_bytes());
504
+ bytes.extend((encoded.len() as u32).to_be_bytes());
505
+ bytes.extend(encoded);
506
+ }
507
+ bytes
508
+ }
509
+ /// Chain names in creation order, then every rule as (chain, expression names).
510
+ fn graph(prepared: &Prepared, table: &str) -> (Vec<String>, Vec<(String, Vec<String>)>) {
511
+ let mut chains = Vec::new();
512
+ let mut rules = Vec::new();
513
+ for (kind, attributes) in prepared.program(table).unwrap() {
514
+ assert_eq!(crate::wire::text(&attributes, 1).unwrap(), table);
515
+ if kind == 3 {
516
+ chains.push(crate::wire::text(&attributes, 3).unwrap());
517
+ continue;
518
+ }
519
+ let expressions = crate::wire::attrs(&crate::wire::one(&attributes, 4).unwrap().value).unwrap();
520
+ rules.push((
521
+ crate::wire::text(&attributes, 2).unwrap(),
522
+ expressions
523
+ .iter()
524
+ .map(|expression| {
525
+ let fields = crate::wire::attrs(&expression.value).unwrap();
526
+ crate::wire::text(&fields, 1).unwrap()
527
+ })
528
+ .collect(),
529
+ ));
530
+ }
531
+ (chains, rules)
532
+ }
533
+
534
+ #[test]
535
+ fn host_v2_without_published_ports_keeps_its_exact_digest_and_compiled_bytes() {
536
+ use sha2::{Digest, Sha256};
537
+ let baseline = prepared(host());
538
+ // Frozen before the optional publication field existed.
539
+ assert_eq!(
540
+ baseline.digest,
541
+ "sha256:b24cd2bc4e812014cabd71a77f57a2fe085c4b83d4e7b7939ccfca28949c6848"
542
+ );
543
+ let bytes = program_bytes(&baseline, "snft_v2_host_golden");
544
+ assert_eq!(bytes.len(), 24932);
545
+ assert_eq!(
546
+ format!("{:x}", Sha256::digest(&bytes)),
547
+ "91735de50a831184648bf79cd7a840f7af5264f3c6ec461f5412b3fc1d1ef6ec"
548
+ );
549
+ let serialized = serde_json::to_string(&baseline.policy).unwrap();
550
+ assert!(!serialized.contains("publishedPorts"));
551
+ // An explicit empty publication is the same canonical policy, digest and graph.
552
+ assert_eq!(prepared(published_host(json!([]))), baseline);
553
+ let (chains, rules) = graph(&baseline, "snft_v2_host_golden");
554
+ assert!(!chains.contains(&"pre".to_string()));
555
+ assert!(rules.iter().all(|(chain, _)| chain != "pre"));
556
+ }
557
+
558
+ #[test]
559
+ fn host_v2_published_ports_compile_uplink_dnat_and_exact_forward_admission() {
560
+ use sha2::{Digest, Sha256};
561
+ let baseline = graph(&prepared(host()), "snft_v2_host_golden");
562
+ let value = published_host(json!([
563
+ published_port("tcp", 443, 8443),
564
+ published_port("udp", 8125, 8125)
565
+ ]));
566
+ let published = prepared(value.clone());
567
+ assert_ne!(published.digest, prepared(host()).digest);
568
+ // Publication order is canonical, so a reordered list is the same policy.
569
+ let mut reordered = value;
570
+ reordered["scope"]["publishedPorts"]
571
+ .as_array_mut()
572
+ .unwrap()
573
+ .reverse();
574
+ assert_eq!(prepared(reordered), published);
575
+ let (chains, rules) = graph(&published, "snft_v2_host_golden");
576
+ assert_eq!(chains.len(), baseline.0.len() + 1);
577
+ assert_eq!(chains.iter().filter(|name| *name == "pre").count(), 1);
578
+ assert_eq!(rules.len(), baseline.1.len() + 8);
579
+ let translations: Vec<_> = rules.iter().filter(|(chain, _)| chain == "pre").collect();
580
+ assert_eq!(translations.len(), 2);
581
+ for (_, expressions) in &translations {
582
+ assert_eq!(expressions.last().unwrap(), "nat");
583
+ assert_eq!(expressions.iter().filter(|name| *name == "immediate").count(), 2);
584
+ }
585
+ // The three admissions precede the capture barrier jumps of their handoff.
586
+ let forward: Vec<_> = rules
587
+ .iter()
588
+ .filter(|(chain, _)| chain == "forward")
589
+ .map(|(_, expressions)| expressions.len())
590
+ .collect();
591
+ assert_eq!(forward.len(), baseline.1.iter().filter(|(chain, _)| chain == "forward").count() + 6);
592
+ let (_, first) = rules.iter().find(|(chain, _)| chain == "forward").unwrap();
593
+ assert!(first.contains(&"ct".to_string()) && first.len() > 4);
594
+ let bytes = program_bytes(&published, "snft_v2_host_golden");
595
+ assert_eq!(bytes.len(), 36322);
596
+ assert_eq!(
597
+ format!("{:x}", Sha256::digest(&bytes)),
598
+ "1d3485bb739ac3601642610eba465b80c9b5ac34df22a5374cfbc8773623710b"
599
+ );
600
+ }
601
+
602
+ #[test]
603
+ fn host_v2_published_ports_are_owned_by_their_generation_and_leave_with_it() {
604
+ let published = prepared(published_host(json!([published_port("tcp", 443, 8443)])));
605
+ let rollback = prepared(host());
606
+ let (chains, rules) = graph(&published, "snft_v2_host_rollback");
607
+ assert!(chains.contains(&"pre".to_string()));
608
+ assert_eq!(rules.iter().filter(|(chain, _)| chain == "pre").count(), 1);
609
+ // Every operation names the owned table, and the rollback target compiles the
610
+ // same graph as before publication; reconcile deletes the complete previous
611
+ // graph and re-adds the target in one batch, so the publication cannot survive.
612
+ let (rollback_chains, rollback_rules) = graph(&rollback, "snft_v2_host_rollback");
613
+ assert!(!rollback_chains.contains(&"pre".to_string()));
614
+ assert!(rollback_rules.iter().all(|(chain, _)| chain != "pre"));
615
+ assert_eq!(rollback_chains.len() + 1, chains.len());
616
+ assert_eq!(rollback_rules.len() + 4, rules.len());
617
+ assert_ne!(published.digest, rollback.digest);
618
+ }
619
+
620
+ #[test]
621
+ fn host_v2_published_ports_reject_unbound_targets_conflicts_and_out_of_range_ports() {
622
+ assert!(normalized(published_host(json!([published_port("tcp", 443, 8443)]))).is_ok());
623
+ let mut unknown = published_port("tcp", 443, 8443);
624
+ unknown["hostInterface"] = json!("ens18");
625
+ let mut second_address = published_host(json!([
626
+ changed(published_port("tcp", 443, 8443), "/hostIp", json!("192.0.2.5"))
627
+ ]));
628
+ second_address["scope"]["uplink"]["requiredIpv4Addresses"] = json!(["192.0.2.2", "192.0.2.5"]);
629
+ let mut duplicate_across_addresses = second_address.clone();
630
+ duplicate_across_addresses["scope"]["publishedPorts"]
631
+ .as_array_mut()
632
+ .unwrap()
633
+ .push(published_port("tcp", 443, 9443));
634
+ assert!(normalized(second_address).is_ok());
635
+ for invalid in [
636
+ published_host(json!([unknown])),
637
+ published_host(json!([published_port("sctp", 443, 8443)])),
638
+ published_host(json!([published_port("tcp", 0, 8443)])),
639
+ published_host(json!([published_port("tcp", 443, 0)])),
640
+ published_host(json!([changed(published_port("tcp", 443, 8443), "/hostPort", json!(65536))])),
641
+ published_host(json!([changed(published_port("tcp", 443, 8443), "/targetPort", json!(65536))])),
642
+ published_host(json!([changed(published_port("tcp", 443, 8443), "/hostPort", json!(-1))])),
643
+ // The published address must be an exact current uplink address.
644
+ published_host(json!([changed(published_port("tcp", 443, 8443), "/hostIp", json!("192.0.2.5"))])),
645
+ published_host(json!([changed(published_port("tcp", 443, 8443), "/hostIp", json!("10.240.0.1"))])),
646
+ // The target must be a current leased transit source address.
647
+ published_host(json!([changed(published_port("tcp", 443, 8443), "/targetAddress", json!("10.240.0.3"))])),
648
+ published_host(json!([changed(published_port("tcp", 443, 8443), "/targetAddress", json!("10.240.0.1"))])),
649
+ published_host(json!([changed(published_port("tcp", 443, 8443), "/targetAddress", json!("192.0.2.2"))])),
650
+ // A published port cannot collide with the leased outbound SNAT ranges.
651
+ published_host(json!([published_port("tcp", 10500, 8443)])),
652
+ published_host(json!([published_port("udp", 10000, 8443)])),
653
+ // One protocol/port is published exactly once, including across addresses.
654
+ published_host(json!([published_port("tcp", 443, 8443), published_port("tcp", 443, 9443)])),
655
+ duplicate_across_addresses,
656
+ published_host(json!((0..65).map(|index| published_port("tcp", 20000 + index, 8443)).collect::<Vec<_>>())),
657
+ ] {
658
+ assert!(normalized(invalid).is_err());
659
+ }
660
+ // Bounded input still meets the atomic reserve first: every publication costs
661
+ // one translation and three admissions, so prepare rejects before any kernel work.
662
+ let ports = |count: u16| -> Value {
663
+ json!((0..count)
664
+ .map(|index| published_port("tcp", 20000 + index, 8443))
665
+ .collect::<Vec<_>>())
666
+ };
667
+ let prepared = prepared(published_host(ports(12)));
668
+ let program = prepared
669
+ .program(&format!("snft_{}", "x".repeat(59)))
670
+ .unwrap();
671
+ let bytes: usize = program
672
+ .iter()
673
+ .map(|(_, attributes)| crate::wire::encode_attrs(attributes).len() + 20)
674
+ .sum();
675
+ assert!(program.len() <= 768 && bytes <= 100_000);
676
+ assert!(normalized(published_host(ports(13))).unwrap().prepare().is_err());
677
+ println!("PUBLISHED_MAX ports=12 operations={} bytes={bytes}", program.len());
678
+ }
package/rust/src/main.rs CHANGED
@@ -91,6 +91,10 @@ fn dispatch(owner: &mut Option<owner::Owner>, request: Request) -> Result<serde_
91
91
  .release(decode(request.params)?)?;
92
92
  Ok(serde_json::json!({"released":true}))
93
93
  }
94
+ "releasePolicyTransition" => {
95
+ owner.as_mut().ok_or(Error::Conflict)?.release_transition(decode(request.params)?)?;
96
+ Ok(serde_json::json!({"released":true}))
97
+ }
94
98
  "detachPolicy" => {
95
99
  owner.as_mut().ok_or(Error::Conflict)?.detach(decode(request.params)?)?;
96
100
  Ok(serde_json::json!({"detached":true}))
@@ -5,7 +5,7 @@ impl Owner {
5
5
  /// orphan after an ACK was lost, including through a fresh native process.
6
6
  pub fn detach(&mut self, expected: Applied) -> Result<()> {
7
7
  self.validate_receipt(&expected)?;
8
- if self.released || self.pending.is_some() {
8
+ if self.released || self.pending.is_some() || self.transition_release.is_some() {
9
9
  return Err(Error::Conflict);
10
10
  }
11
11
  if let Some(retention) = &self.retention {
package/rust/src/owner.rs CHANGED
@@ -44,6 +44,12 @@ pub struct Transition {
44
44
  pub previous: Option<Applied>,
45
45
  pub target: Prepared,
46
46
  }
47
+ #[derive(Clone, Debug, Deserialize, Serialize, PartialEq, Eq)]
48
+ #[serde(rename_all = "camelCase", deny_unknown_fields)]
49
+ pub struct TransitionRelease {
50
+ pub identity: Identity,
51
+ pub transition: Transition,
52
+ }
47
53
  #[derive(Clone, Debug, Serialize)]
48
54
  #[serde(rename_all = "camelCase")]
49
55
  pub struct Status {
@@ -69,6 +75,7 @@ pub struct Owner {
69
75
  error: Option<Error>,
70
76
  released: bool,
71
77
  retention: Option<Retention>,
78
+ transition_release: Option<TransitionRelease>,
72
79
  }
73
80
 
74
81
  struct Retention {
@@ -78,6 +85,8 @@ struct Retention {
78
85
 
79
86
  #[path = "owner.detach.rs"]
80
87
  mod detach;
88
+ #[path = "owner.transitionrelease.rs"]
89
+ mod transitionrelease;
81
90
 
82
91
  pub(crate) fn context() -> Result<(String, String, String)> {
83
92
  let boot = std::fs::read_to_string("/proc/sys/kernel/random/boot_id")
@@ -164,6 +173,7 @@ impl Owner {
164
173
  error: None,
165
174
  released: false,
166
175
  retention: None,
176
+ transition_release: None,
167
177
  })
168
178
  }
169
179
  fn socket(&mut self) -> Result<&mut Socket> {
@@ -360,7 +370,7 @@ impl Owner {
360
370
  })
361
371
  }
362
372
  pub fn reconcile(&mut self, transition: Transition) -> Result<Applied> {
363
- if self.released || self.retention.is_some() {
373
+ if self.released || self.retention.is_some() || self.transition_release.is_some() {
364
374
  return Err(Error::Conflict);
365
375
  }
366
376
  transition.target.validate()?;
@@ -463,6 +473,9 @@ impl Owner {
463
473
  self.applied(&graph, transition.target.clone())
464
474
  }
465
475
  pub fn inspect(&mut self) -> Status {
476
+ if self.transition_release.is_some() {
477
+ return self.inspect_transition_release();
478
+ }
466
479
  if self.retention.is_some() {
467
480
  return self.inspect_retained();
468
481
  }
@@ -503,7 +516,7 @@ impl Owner {
503
516
  }
504
517
  }
505
518
  pub fn release(&mut self, expected: Applied) -> Result<()> {
506
- if self.retention.is_some() {
519
+ if self.retention.is_some() || self.transition_release.is_some() {
507
520
  return Err(Error::Conflict);
508
521
  }
509
522
  self.validate_receipt(&expected)?;
@@ -561,6 +574,9 @@ impl Owner {
561
574
  }
562
575
  }
563
576
  pub fn close(&mut self) -> Result<()> {
577
+ if let Some(request) = self.transition_release.clone() {
578
+ return self.release_transition(request);
579
+ }
564
580
  if self.retention.is_some() {
565
581
  return Err(Error::Conflict);
566
582
  }
@@ -0,0 +1,107 @@
1
+ use super::*;
2
+
3
+ impl Owner {
4
+ /// Terminal cleanup of an exact caller-retained transition. Unlike reconcile,
5
+ /// this never creates/replaces policy and does not require surviving interfaces.
6
+ pub fn release_transition(&mut self, request: TransitionRelease) -> Result<()> {
7
+ if self.retention.is_some() || request.identity != self.identity {
8
+ return Err(Error::Conflict);
9
+ }
10
+ request.transition.target.validate()?;
11
+ if let Some(previous) = &request.transition.previous {
12
+ self.validate_receipt(previous)?;
13
+ if previous.prepared.policy.kind() != request.transition.target.policy.kind()
14
+ || previous.prepared.policy.revision()
15
+ >= request.transition.target.policy.revision()
16
+ {
17
+ return Err(Error::Conflict);
18
+ }
19
+ }
20
+ if let Some(selected) = &self.transition_release {
21
+ if selected != &request {
22
+ return Err(Error::Conflict);
23
+ }
24
+ if self.released {
25
+ return Ok(());
26
+ }
27
+ } else {
28
+ if self.released
29
+ || self
30
+ .pending
31
+ .as_ref()
32
+ .is_some_and(|pending| pending != &request.transition)
33
+ || self.applied.as_ref().is_some_and(|applied| {
34
+ Some(applied) != request.transition.previous.as_ref()
35
+ && applied.prepared != request.transition.target
36
+ })
37
+ {
38
+ return Err(Error::Conflict);
39
+ }
40
+ // Capture terminal intent before any query, ownership claim or delete.
41
+ self.transition_release = Some(request.clone());
42
+ }
43
+ let result = self.release_transition_inner(&request.transition);
44
+ match result {
45
+ Ok(()) => {
46
+ self.released = true;
47
+ self.pending = None;
48
+ self.error = None;
49
+ self.socket = None;
50
+ Ok(())
51
+ }
52
+ Err(error) => {
53
+ self.error = Some(error);
54
+ Err(error)
55
+ }
56
+ }
57
+ }
58
+
59
+ fn release_transition_inner(&mut self, transition: &Transition) -> Result<()> {
60
+ let Some(graph) = self.graph()? else {
61
+ return Ok(());
62
+ };
63
+ // Also bind any handle this process already observed before an ambiguous
64
+ // delete. A replacement under the same name must never inherit authority.
65
+ if let Some(applied) = &self.applied {
66
+ self.binding(&graph, Some(applied))?;
67
+ }
68
+ let graph = self.adopt(graph, transition)?;
69
+ let prepared = if self.matches(&graph, &transition.target)? {
70
+ transition.target.clone()
71
+ } else {
72
+ transition
73
+ .previous
74
+ .as_ref()
75
+ .ok_or(Error::Conflict)?
76
+ .prepared
77
+ .clone()
78
+ };
79
+ self.applied = Some(self.applied(&graph, prepared)?);
80
+ self.socket()?.batch(
81
+ graph.generation,
82
+ vec![(2, 0, vec![Attr::u64(4, wire::handle(&graph.table, 4)?)])],
83
+ )?;
84
+ if self.graph()?.is_some() {
85
+ return Err(Error::Conflict);
86
+ }
87
+ Ok(())
88
+ }
89
+
90
+ pub(super) fn inspect_transition_release(&self) -> Status {
91
+ // Cleanup intent cannot grant enforcement, including after a failed
92
+ // attempt that acquired OWNER on an otherwise valid persistent graph.
93
+ Status {
94
+ identity: self.identity.clone(),
95
+ state: if self.released {
96
+ "released"
97
+ } else {
98
+ "failed-owned"
99
+ }
100
+ .into(),
101
+ enforced: false,
102
+ applied: self.applied.clone(),
103
+ pending: self.pending.clone(),
104
+ error: self.error.map(|error| error.code().into()),
105
+ }
106
+ }
107
+ }