@push.rocks/smartnftables 2.4.0 → 2.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. package/changelog.md +12 -0
  2. package/dist_rust/smartnftables_linux_amd64_musl +0 -0
  3. package/dist_rust/smartnftables_linux_amd64_musl.tsrust-build.json +5 -5
  4. package/dist_rust/smartnftables_linux_arm64_musl +0 -0
  5. package/dist_rust/smartnftables_linux_arm64_musl.tsrust-build.json +5 -5
  6. package/dist_ts/00_commitinfo_data.js +1 -1
  7. package/dist_ts/classes.manageddockerforwarding.js +3 -3
  8. package/dist_ts/classes.managednftables.d.ts +5 -0
  9. package/dist_ts/classes.managednftables.js +13 -6
  10. package/dist_ts/managed.egress.types.d.ts +31 -1
  11. package/package.json +7 -7
  12. package/readme.md +91 -3
  13. package/rust/src/egress.compile.rs +49 -1
  14. package/rust/src/egress.host.rs +22 -0
  15. package/rust/src/egress.hostgrant.rs +187 -0
  16. package/rust/src/egress.poolguard.rs +7 -0
  17. package/rust/src/egress.router.rs +45 -4
  18. package/rust/src/egress.rs +150 -1
  19. package/rust/src/egress_hostgrant_tests.rs +948 -0
  20. package/rust/src/main.rs +6 -2
  21. package/rust/src/owner.rs +116 -4
  22. package/rust/src/owner_egress_tests.rs +1 -1
  23. package/rust/src/owner_hostgrant_traffic_tests.rs +280 -0
  24. package/rust/src/owner_identity_tests.rs +80 -1
  25. package/rust/src/owner_poolguard_tests.rs +69 -0
  26. package/rust/src/owner_router_traffic_tests.rs +1 -1
  27. package/rust/src/owner_tests.rs +3 -0
  28. package/rust/src/policy.rs +1 -1
  29. package/rust/src/wire.rs +15 -1
  30. package/ts/00_commitinfo_data.ts +1 -1
  31. package/ts/classes.manageddockerforwarding.ts +2 -2
  32. package/ts/classes.managednftables.ts +13 -5
  33. package/ts/managed.egress.types.ts +32 -1
package/rust/src/main.rs CHANGED
@@ -11,6 +11,10 @@ mod wire;
11
11
  use serde::Deserialize;
12
12
  use std::io::{BufRead, Read, Write};
13
13
 
14
+ /// One request line from the facade; the facade's `managedIpcBytes` bounds the
15
+ /// same lines. A transition carries two complete policies.
16
+ pub(crate) const IPC_BYTES: usize = 1_048_576;
17
+
14
18
  #[derive(Clone, Copy, Debug, PartialEq, Eq)]
15
19
  pub enum Error {
16
20
  Invalid,
@@ -141,12 +145,12 @@ fn main() {
141
145
  loop {
142
146
  let mut bytes = Vec::new();
143
147
  let read = std::io::Read::by_ref(&mut input)
144
- .take(262_145)
148
+ .take(IPC_BYTES as u64 + 1)
145
149
  .read_until(b'\n', &mut bytes);
146
150
  if matches!(read, Ok(0)) {
147
151
  break;
148
152
  }
149
- if read.is_err() || bytes.len() > 262_144 || bytes.last() != Some(&b'\n') {
153
+ if read.is_err() || bytes.len() > IPC_BYTES || bytes.last() != Some(&b'\n') {
150
154
  std::process::exit(2);
151
155
  }
152
156
  let request: Request = match serde_json::from_slice(&bytes) {
package/rust/src/owner.rs CHANGED
@@ -66,6 +66,8 @@ struct Graph {
66
66
  table: Vec<Attr>,
67
67
  chains: Vec<Vec<Attr>>,
68
68
  rules: Vec<Vec<Attr>>,
69
+ /// Each named set with the key of every element it holds.
70
+ sets: Vec<(Vec<Attr>, Vec<Vec<u8>>)>,
69
71
  }
70
72
  pub struct Owner {
71
73
  identity: Identity,
@@ -220,8 +222,31 @@ impl Owner {
220
222
  }
221
223
  }
222
224
  let rules = socket.query(7, vec![Attr::string(1, &table_name)], true)?;
225
+ // Set dumps select the table. Every element is read back, so a set
226
+ // whose content differs from the compiled target never verifies.
227
+ let mut sets = Vec::new();
228
+ for set in socket.query(10, vec![Attr::string(1, &table_name)], true)? {
229
+ if wire::text(&set, 1)? != table_name {
230
+ return Err(Error::Conflict);
231
+ }
232
+ let name = wire::text(&set, 2)?;
233
+ let mut keys = Vec::new();
234
+ for message in socket.query(
235
+ 13,
236
+ vec![Attr::string(1, &table_name), Attr::string(2, &name)],
237
+ true,
238
+ )? {
239
+ if wire::text(&message, 1)? != table_name || wire::text(&message, 2)? != name {
240
+ return Err(Error::Conflict);
241
+ }
242
+ if let Ok(list) = wire::one(&message, 3) {
243
+ keys.extend(element_keys(&wire::attrs(&list.value)?)?);
244
+ }
245
+ }
246
+ sets.push((set, keys));
247
+ }
223
248
  // The managed compiler never creates any of these. Foreign objects reject.
224
- for operation in [10, 19, 23] {
249
+ for operation in [19, 23] {
225
250
  if !socket
226
251
  .query(operation, vec![Attr::string(1, &table_name)], true)?
227
252
  .is_empty()
@@ -235,6 +260,7 @@ impl Owner {
235
260
  table,
236
261
  chains,
237
262
  rules,
263
+ sets,
238
264
  }));
239
265
  }
240
266
  }
@@ -310,7 +336,37 @@ impl Owner {
310
336
  }
311
337
  result
312
338
  };
313
- Ok(expected_chains == actual_chains && regroup(expected_rules) == regroup(actual_rules))
339
+ // Element order is the set backend's; element keys are the identity.
340
+ let mut expected_sets = std::collections::BTreeMap::<String, (Vec<u8>, Vec<Vec<u8>>)>::new();
341
+ for (kind, attrs) in &program {
342
+ match kind {
343
+ 9 => {
344
+ expected_sets.insert(wire::text(attrs, 2)?, (set_identity(attrs)?, Vec::new()));
345
+ }
346
+ 12 => {
347
+ let entry = expected_sets
348
+ .get_mut(&wire::text(attrs, 2)?)
349
+ .ok_or(Error::Invalid)?;
350
+ entry.1.extend(element_keys(&wire::attrs(&wire::one(attrs, 3)?.value)?)?);
351
+ }
352
+ _ => {}
353
+ }
354
+ }
355
+ let mut actual_sets = std::collections::BTreeMap::new();
356
+ for (attrs, keys) in &graph.sets {
357
+ if actual_sets
358
+ .insert(wire::text(attrs, 2)?, (set_identity(attrs)?, keys.clone()))
359
+ .is_some()
360
+ {
361
+ return Err(Error::Conflict);
362
+ }
363
+ }
364
+ for entry in expected_sets.values_mut().chain(actual_sets.values_mut()) {
365
+ entry.1.sort();
366
+ }
367
+ Ok(expected_chains == actual_chains
368
+ && regroup(expected_rules) == regroup(actual_rules)
369
+ && expected_sets == actual_sets)
314
370
  }
315
371
  fn adopt(&mut self, graph: Graph, transition: &Transition) -> Result<Graph> {
316
372
  self.binding(&graph, transition.previous.as_ref())?;
@@ -437,6 +493,17 @@ impl Owner {
437
493
  ],
438
494
  ));
439
495
  }
496
+ // After every rule that binds it; deleting a set deletes its elements.
497
+ for (set, _) in &graph.sets {
498
+ operations.push((
499
+ 11,
500
+ 0,
501
+ vec![
502
+ Attr::string(1, &table_name),
503
+ Attr::u64(16, wire::handle(set, 16)?),
504
+ ],
505
+ ));
506
+ }
440
507
  graph.generation
441
508
  } else {
442
509
  if transition.previous.is_some() || self.applied.is_some() {
@@ -454,9 +521,10 @@ impl Owner {
454
521
  self.socket()?.generation()?
455
522
  };
456
523
  for (operation, attributes) in transition.target.program(&table_name)? {
524
+ // Chains, sets and elements are created exclusively; rules append.
457
525
  operations.push((
458
526
  operation,
459
- if operation == 3 { 0x600 } else { 0xc00 },
527
+ if operation == 6 { 0xc00 } else { 0x600 },
460
528
  attributes,
461
529
  ));
462
530
  }
@@ -612,6 +680,41 @@ fn chain_identity(attributes: &[Attr]) -> Result<Vec<u8>> {
612
680
  }
613
681
  serde_json::to_vec(&wire::canonical(&identity, false)?).map_err(|_| Error::Protocol)
614
682
  }
683
+ /// A set is its table, name, flags, key type and key length. The handle, the
684
+ /// informational backend type and count, the transaction ID and an empty
685
+ /// description are not identity; anything else rejects.
686
+ fn set_identity(attributes: &[Attr]) -> Result<Vec<u8>> {
687
+ let mut identity = Vec::new();
688
+ for attribute in attributes {
689
+ match attribute.id() {
690
+ 10 | 16 | 19 | 20 => {}
691
+ 9 if wire::attrs(&attribute.value)?.is_empty() => {}
692
+ 1..=5 => identity.push(attribute.clone()),
693
+ _ => return Err(Error::Conflict),
694
+ }
695
+ }
696
+ for id in [1, 2, 3, 4, 5] {
697
+ wire::one(&identity, id)?;
698
+ }
699
+ serde_json::to_vec(&wire::canonical(&identity, false)?).map_err(|_| Error::Protocol)
700
+ }
701
+ /// Only plain keys: an element with data, flags, timeouts, expressions or any
702
+ /// other extension is not one the compiler created.
703
+ fn element_keys(list: &[Attr]) -> Result<Vec<Vec<u8>>> {
704
+ let mut keys = Vec::new();
705
+ for element in list {
706
+ let fields = wire::attrs(&element.value)?;
707
+ if element.id() != 1 || fields.len() != 1 {
708
+ return Err(Error::Conflict);
709
+ }
710
+ let key = wire::attrs(&wire::one(&fields, 1)?.value)?;
711
+ if key.len() != 1 {
712
+ return Err(Error::Conflict);
713
+ }
714
+ keys.push(wire::one(&key, 1)?.value.clone());
715
+ }
716
+ Ok(keys)
717
+ }
615
718
  fn rule_identity(attributes: &[Attr]) -> Result<(String, Vec<u8>)> {
616
719
  let chain = wire::text(attributes, 2)?;
617
720
  if attributes
@@ -638,7 +741,7 @@ fn rule_identity(attributes: &[Attr]) -> Result<(String, Vec<u8>)> {
638
741
  "cmp" => &[3],
639
742
  "bitwise" => &[4, 5],
640
743
  "range" => &[3, 4],
641
- "meta" | "payload" | "ct" | "nat" => &[],
744
+ "meta" | "payload" | "ct" | "nat" | "lookup" => &[],
642
745
  _ => return Err(Error::Conflict),
643
746
  };
644
747
  for attribute in &mut data {
@@ -669,6 +772,7 @@ fn rule_identity(attributes: &[Attr]) -> Result<(String, Vec<u8>)> {
669
772
  /// neither unknown fields nor default repair may hide a changed kernel rule.
670
773
  fn validate_egress_expression(name: &str, data: &[Attr]) -> Result<()> {
671
774
  let allowed: &[u16] = match name {
775
+ "lookup" => &[1, 2, 5],
672
776
  "ct" | "range" => &[1, 2, 3, 4],
673
777
  "nat" => &[1, 2, 3, 4, 5, 6, 7],
674
778
  _ => return Ok(()),
@@ -680,6 +784,14 @@ fn validate_egress_expression(name: &str, data: &[Attr]) -> Result<()> {
680
784
  wire::one(data, attribute.id())?; // Duplicate fields are never canonical identity.
681
785
  }
682
786
  match name {
787
+ "lookup" => {
788
+ // A plain membership test of a named set: no map, no inversion.
789
+ wire::text(data, 1)?;
790
+ wire::number(data, 2)?;
791
+ if data.len() != 3 || wire::number(data, 5)? != 0 {
792
+ return Err(Error::Conflict);
793
+ }
794
+ }
683
795
  "ct" => {
684
796
  let read = data.iter().any(|item| item.id() == 1);
685
797
  let write = data.iter().any(|item| item.id() == 4);
@@ -85,7 +85,7 @@ pub(super) fn recover(name: &str, policy: egress::Policy) {
85
85
  chain_identity(actual),
86
86
  "chain {name}"
87
87
  );
88
- } else {
88
+ } else if kind == 6 {
89
89
  let stamp = &wire::one(&expected, 7).unwrap().value;
90
90
  let actual = graph
91
91
  .rules
@@ -0,0 +1,280 @@
1
+ use super::egress_traffic_tests::{
2
+ allocation, binding, connect, denied, host_policy, ns_ip, protection,
3
+ };
4
+ use super::egress_traffic_tests::{roundtrip, socket};
5
+ use super::router_traffic_tests::connect_from;
6
+ use super::*;
7
+ use crate::egress;
8
+ use serde_json::json;
9
+ use std::io::{Read, Write};
10
+ use std::net::{TcpListener, TcpStream};
11
+ use std::time::Duration;
12
+
13
+ const WORK: &str = "hg_work";
14
+ const ROUTER: &str = "hg_router";
15
+ const HOST: &str = "hg_host";
16
+ const PEER: &str = "hg_peer";
17
+
18
+ /// The contract maximum: the two probed grants among 1022 others, so every
19
+ /// admission is a lookup in a full set. TCP and UDP 30000–30510 are members.
20
+ fn grants() -> Vec<egress::HostGrant> {
21
+ let grant = |protocol: &str, port: u16| egress::HostGrant {
22
+ protocol: protocol.into(),
23
+ source_address: "10.240.0.1".into(),
24
+ destination_address: "10.241.0.2".into(),
25
+ destination_port: port,
26
+ };
27
+ let mut result = vec![grant("tcp", 8080), grant("udp", 5353)];
28
+ for port in 30000..=30510 {
29
+ result.push(grant("tcp", port));
30
+ result.push(grant("udp", port));
31
+ }
32
+ assert_eq!(result.len(), 1024);
33
+ result
34
+ }
35
+ /// One workload behind the router with leased public egress on every port of
36
+ /// both protocols, exactly as a node composes it, so a leased classifier would
37
+ /// claim the workload's replies if host grants were not default-zone flows.
38
+ fn router_policy(revision: u64, granted: bool) -> egress::Policy {
39
+ let link = binding("work", "10.241.0.1", "veth");
40
+ let mut generation = allocation();
41
+ generation["conntrackZone"] = json!(17);
42
+ generation["conntrackLabel"] = json!("a".repeat(32));
43
+ generation["grants"] = json!([
44
+ {"sourceEndpoint":"work","sourcePrefix":"10.241.0.2/32","destination":{"kind":"public","prefix":"0.0.0.0/0"},
45
+ "protocol":"udp","sourcePort":null,"destinationPort":null,"sourcePortRange":{"first":10000,"last":10015}},
46
+ {"sourceEndpoint":"work","sourcePrefix":"10.241.0.2/32","destination":{"kind":"public","prefix":"0.0.0.0/0"},
47
+ "protocol":"tcp","sourcePort":null,"destinationPort":null,"sourcePortRange":{"first":10016,"last":10031}}
48
+ ]);
49
+ let mut policy: egress::Policy =
50
+ serde_json::from_value(json!({"schemaVersion":2,"revision":revision,
51
+ "scope":{"kind":"routerEgress",
52
+ "endpoints":[{"id":"work","interfaceIndex":link["interfaceIndex"],"interfaceName":"work",
53
+ "interfaceKind":"veth","sourcePrefixes":["10.241.0.2/32"]}],
54
+ "rules":[],"links":[link],"handoff":binding("handoff","10.240.0.2","veth"),
55
+ "protection":protection(),"generations":[generation]}}))
56
+ .unwrap();
57
+ let egress::Scope::RouterEgress(scope) = &mut policy.scope else {
58
+ panic!()
59
+ };
60
+ scope.host_grants = if granted { grants() } else { vec![] };
61
+ policy
62
+ }
63
+ fn transit_policy(revision: u64, granted: bool) -> egress::Policy {
64
+ let mut policy = host_policy();
65
+ policy.revision = revision;
66
+ let egress::Scope::HostTransit(scope) = &mut policy.scope else {
67
+ panic!()
68
+ };
69
+ scope.host_grants = if granted { grants() } else { vec![] };
70
+ policy
71
+ }
72
+ fn guard_policy(revision: u64, granted: bool) -> egress::Policy {
73
+ let mut policy: egress::Policy = serde_json::from_value(egress::tests::pool_guard()).unwrap();
74
+ policy.revision = revision;
75
+ let egress::Scope::AllocationPoolGuard(scope) = &mut policy.scope else {
76
+ panic!()
77
+ };
78
+ scope.host_grants = if granted { grants() } else { vec![] };
79
+ policy
80
+ }
81
+ /// One policy owner in its own namespace; every transition runs there.
82
+ struct Hop {
83
+ namespace: &'static str,
84
+ owner: Option<Owner>,
85
+ applied: Option<Applied>,
86
+ }
87
+ impl Hop {
88
+ fn new(namespace: &'static str, table: &'static str) -> Self {
89
+ let owner = in_namespace(namespace, move || Owner::new(options(table)).unwrap());
90
+ Self {
91
+ namespace,
92
+ owner: Some(owner),
93
+ applied: None,
94
+ }
95
+ }
96
+ fn apply(&mut self, build: fn(u64, bool) -> egress::Policy, revision: u64, granted: bool) {
97
+ let mut owner = self.owner.take().unwrap();
98
+ let previous = self.applied.take();
99
+ let (owner, applied) = in_namespace(self.namespace, move || {
100
+ let target: Prepared = build(revision, granted).prepare().unwrap().into();
101
+ target
102
+ .validate_interfaces()
103
+ .expect("host grant fixture binding");
104
+ let applied = owner.reconcile(Transition { previous, target }).unwrap();
105
+ assert!(owner.inspect().enforced);
106
+ (owner, applied)
107
+ });
108
+ self.owner = Some(owner);
109
+ self.applied = Some(applied);
110
+ }
111
+ fn release(mut self) {
112
+ let mut owner = self.owner.take().unwrap();
113
+ let applied = self.applied.take().unwrap();
114
+ in_namespace(self.namespace, move || {
115
+ assert!(owner.inspect().enforced);
116
+ owner.release(applied).unwrap();
117
+ });
118
+ }
119
+ }
120
+ /// A host-origin TCP flow from an exact source address. On success the workload
121
+ /// must observe that exact source and exchange data both ways.
122
+ fn dial(
123
+ source: &'static str,
124
+ destination: &'static str,
125
+ listener: &TcpListener,
126
+ ) -> Option<(TcpStream, TcpStream)> {
127
+ let client = in_namespace(HOST, move || connect_from(source, destination)).ok()?;
128
+ let (server, peer) = listener.accept().unwrap();
129
+ assert_eq!(peer.ip().to_string(), source.split(':').next().unwrap());
130
+ let (mut client, mut server) = (client, server);
131
+ for stream in [&client, &server] {
132
+ stream
133
+ .set_read_timeout(Some(Duration::from_secs(1)))
134
+ .unwrap();
135
+ }
136
+ client.write_all(b"ping").unwrap();
137
+ server.read_exact(&mut [0; 4]).unwrap();
138
+ server.write_all(b"pong").unwrap();
139
+ client.read_exact(&mut [0; 4]).unwrap();
140
+ Some((client, server))
141
+ }
142
+ fn dark(source: &'static str, destination: &'static str, listener: &TcpListener) {
143
+ assert!(
144
+ dial(source, destination, listener).is_none(),
145
+ "{source} -> {destination}"
146
+ );
147
+ }
148
+
149
+ #[test]
150
+ #[ignore = "requires disposable isolated native qualification guest"]
151
+ fn v2_host_grant_opens_exactly_the_host_to_workload_port_through_guard_transit_and_router() {
152
+ isolated();
153
+ for name in [WORK, ROUTER, HOST, PEER] {
154
+ ip(&["netns", "add", name]);
155
+ ns_ip(name, &["link", "set", "lo", "up"]);
156
+ in_namespace(name, || {
157
+ std::fs::write("/proc/sys/net/ipv4/ip_forward", "1").unwrap()
158
+ });
159
+ }
160
+ connect(
161
+ WORK,
162
+ "client",
163
+ "10.241.0.2/24",
164
+ ROUTER,
165
+ "work",
166
+ "10.241.0.1/24",
167
+ );
168
+ connect(
169
+ ROUTER,
170
+ "handoff",
171
+ "10.240.0.2/30",
172
+ HOST,
173
+ "router",
174
+ "10.240.0.1/30",
175
+ );
176
+ connect(
177
+ HOST,
178
+ "uplink",
179
+ "192.0.2.2/24",
180
+ PEER,
181
+ "underlay",
182
+ "192.0.2.3/24",
183
+ );
184
+ ns_ip(WORK, &["route", "add", "default", "via", "10.241.0.1"]);
185
+ ns_ip(ROUTER, &["route", "add", "default", "via", "10.240.0.1"]);
186
+ // The workload route exists throughout, so every later denial is caused by
187
+ // policy and not by a missing route.
188
+ ns_ip(
189
+ HOST,
190
+ &["route", "add", "10.241.0.0/24", "via", "10.240.0.2"],
191
+ );
192
+ let granted_tcp = in_namespace(WORK, || TcpListener::bind("10.241.0.2:8080").unwrap());
193
+ let other_tcp = in_namespace(WORK, || TcpListener::bind("10.241.0.2:8081").unwrap());
194
+ let member_tcp = in_namespace(WORK, || TcpListener::bind("10.241.0.2:30300").unwrap());
195
+ let neighbour_tcp = in_namespace(WORK, || TcpListener::bind("10.241.0.2:30511").unwrap());
196
+ let host_tcp = in_namespace(HOST, || TcpListener::bind("10.240.0.1:9000").unwrap());
197
+ let granted_udp = socket(WORK, "10.241.0.2:5353");
198
+ let other_udp = socket(WORK, "10.241.0.2:5354");
199
+ let protocol_udp = socket(WORK, "10.241.0.2:8080");
200
+ // Every UDP probe uses its own socket: an unreplied conntrack entry would
201
+ // otherwise carry an earlier decision across a policy change.
202
+ let host_udp = || socket(HOST, "10.240.0.1:0");
203
+ // Positive controls without policy: every probe below can succeed, including
204
+ // the foreign source address and the workload-origin dial of the host.
205
+ assert!(dial("10.240.0.1:0", "10.241.0.2:8080", &granted_tcp).is_some());
206
+ assert!(dial("10.240.0.1:0", "10.241.0.2:8081", &other_tcp).is_some());
207
+ assert!(dial("192.0.2.2:0", "10.241.0.2:8080", &granted_tcp).is_some());
208
+ assert!(dial("10.240.0.1:0", "10.241.0.2:30511", &neighbour_tcp).is_some());
209
+ roundtrip(&host_udp(), &granted_udp);
210
+ roundtrip(&host_udp(), &other_udp);
211
+ roundtrip(&host_udp(), &protocol_udp);
212
+ let workload_dial = |expected: bool| {
213
+ let result = in_namespace(WORK, || {
214
+ TcpStream::connect_timeout(&"10.240.0.1:9000".parse().unwrap(), Duration::from_secs(1))
215
+ });
216
+ assert_eq!(result.is_ok(), expected);
217
+ if expected {
218
+ host_tcp.accept().unwrap();
219
+ }
220
+ };
221
+ workload_dial(true);
222
+
223
+ let mut guard = Hop::new(HOST, "host_grant_guard");
224
+ let mut transit = Hop::new(HOST, "host_grant_transit");
225
+ let mut router = Hop::new(ROUTER, "host_grant_router");
226
+ guard.apply(guard_policy, 1, false);
227
+ transit.apply(transit_policy, 1, false);
228
+ router.apply(router_policy, 1, false);
229
+ // Without grants the host never reaches the workload pool.
230
+ dark("10.240.0.1:0", "10.241.0.2:8080", &granted_tcp);
231
+ denied(&host_udp(), &granted_udp, true);
232
+
233
+ // Transit and router grants alone do not pass the host-wide guard.
234
+ transit.apply(transit_policy, 2, true);
235
+ router.apply(router_policy, 2, true);
236
+ dark("10.240.0.1:0", "10.241.0.2:8080", &granted_tcp);
237
+ denied(&host_udp(), &granted_udp, true);
238
+
239
+ // With the guard exception the exact tuples pass all three tables, and the
240
+ // workload's replies return in the default zone despite leased egress on
241
+ // every port of both protocols.
242
+ guard.apply(guard_policy, 2, true);
243
+ let (mut established, mut served) =
244
+ dial("10.240.0.1:0", "10.241.0.2:8080", &granted_tcp).expect("granted host tcp");
245
+ let observed = roundtrip(&host_udp(), &granted_udp);
246
+ assert_eq!(observed.ip().to_string(), "10.240.0.1");
247
+ // Any member of the full set passes; the tuple just outside it does not.
248
+ assert!(dial("10.240.0.1:0", "10.241.0.2:30300", &member_tcp).is_some());
249
+ dark("10.240.0.1:0", "10.241.0.2:30511", &neighbour_tcp);
250
+ // Another port, protocol, source address or direction stays blocked.
251
+ dark("10.240.0.1:0", "10.241.0.2:8081", &other_tcp);
252
+ denied(&host_udp(), &other_udp, true);
253
+ denied(&host_udp(), &protocol_udp, true);
254
+ dark("192.0.2.2:0", "10.241.0.2:8080", &granted_tcp);
255
+ workload_dial(false);
256
+
257
+ // The router hop is independently required: withdrawing only its grants
258
+ // closes the path behind the host's own tables, and re-granting reopens it.
259
+ router.apply(router_policy, 3, false);
260
+ dark("10.240.0.1:0", "10.241.0.2:8080", &granted_tcp);
261
+ denied(&host_udp(), &granted_udp, false);
262
+ router.apply(router_policy, 4, true);
263
+ assert!(dial("10.240.0.1:0", "10.241.0.2:8080", &granted_tcp).is_some());
264
+
265
+ // Withdrawal in every scope closes new flows and the established one.
266
+ guard.apply(guard_policy, 3, false);
267
+ transit.apply(transit_policy, 3, false);
268
+ router.apply(router_policy, 5, false);
269
+ dark("10.240.0.1:0", "10.241.0.2:8080", &granted_tcp);
270
+ denied(&host_udp(), &granted_udp, true);
271
+ established.write_all(b"late").unwrap();
272
+ assert!(served.read_exact(&mut [0; 4]).is_err());
273
+ drop(established);
274
+ drop(served);
275
+
276
+ router.release();
277
+ transit.release();
278
+ guard.release();
279
+ println!("V2_HOST_GRANT_PROOF grants_per_scope=1024 set_member=true set_neighbour_denied=true positive_controls=true ungranted_dark=true guard_required=true exact_tcp=true exact_udp=true source_preserved=true other_port_denied=true other_protocol_denied=true other_source_denied=true workload_origin_denied=true router_required=true regranted=true withdrawn_dark=true established_withdrawn=true");
280
+ }
@@ -1,4 +1,4 @@
1
- use super::rule_identity;
1
+ use super::{element_keys, rule_identity, set_identity};
2
2
  use crate::wire::{self, Attr};
3
3
 
4
4
  fn expression(name: &str, data: Vec<Attr>) -> Attr {
@@ -249,3 +249,82 @@ fn published_router_graph_identity_preserves_destination_nat_and_admission_opera
249
249
  .count()
250
250
  );
251
251
  }
252
+
253
+ /// A host-grant set as Linux dumps it: the compiled creation plus the kernel's
254
+ /// handle, an empty description, the backend type and the element count.
255
+ fn dumped_set(created: &[Attr]) -> Vec<Attr> {
256
+ let mut dump: Vec<Attr> = created.iter().filter(|attribute| attribute.id() != 10).cloned().collect();
257
+ dump.push(Attr::u64(16, 7));
258
+ dump.push(Attr::nested(9, vec![]));
259
+ dump.push(Attr::string(19, "nft_rhash_type"));
260
+ dump.push(Attr::u32(20, 1024));
261
+ dump
262
+ }
263
+
264
+ #[test]
265
+ fn host_grant_sets_elements_and_lookups_have_exact_graph_identity() {
266
+ let prepared = serde_json::from_value::<crate::egress::Policy>(
267
+ crate::egress::hostgrant_tests::granted(
268
+ crate::egress::tests::router(),
269
+ serde_json::json!([crate::egress::hostgrant_tests::host_grant("tcp", "10.240.0.1", "10.241.0.2", 8080)]),
270
+ ),
271
+ )
272
+ .unwrap()
273
+ .prepare()
274
+ .unwrap();
275
+ let program = prepared.program("snft_identity").unwrap();
276
+ let sets: Vec<_> = program.iter().filter(|(kind, _)| *kind == 9).map(|(_, attributes)| attributes.clone()).collect();
277
+ assert_eq!(sets.len(), 2);
278
+ for created in &sets {
279
+ // The kernel's informational attributes are not identity.
280
+ assert_eq!(set_identity(created).unwrap(), set_identity(&dumped_set(created)).unwrap());
281
+ // A changed key length, flag or any unknown attribute is a different set.
282
+ for (id, value) in [(5, Attr::u32(5, 4)), (3, Attr::u32(3, 0))] {
283
+ let mut changed: Vec<_> = created.iter().filter(|attribute| attribute.id() != id).cloned().collect();
284
+ changed.push(value);
285
+ assert_ne!(set_identity(&changed).unwrap(), set_identity(created).unwrap());
286
+ }
287
+ for foreign in [
288
+ Attr::u64(11, 1000),
289
+ Attr::bytes(13, b"foreign".to_vec()),
290
+ Attr::nested(9, vec![Attr::u32(1, 64)]),
291
+ ] {
292
+ let mut changed = dumped_set(created);
293
+ changed.push(foreign);
294
+ assert!(set_identity(&changed).is_err());
295
+ }
296
+ }
297
+ // Elements are plain keys; any extension rejects.
298
+ let (_, elements) = program.iter().find(|(kind, _)| *kind == 12).unwrap();
299
+ let list = wire::attrs(&wire::one(elements, 3).unwrap().value).unwrap();
300
+ assert_eq!(element_keys(&list).unwrap().len(), 1);
301
+ let key = wire::attrs(&list[0].value).unwrap();
302
+ for extension in [Attr::u32(3, 1), Attr::nested(2, vec![Attr::bytes(1, vec![1; 4])])] {
303
+ let mut changed = key.clone();
304
+ changed.push(extension);
305
+ assert!(element_keys(&[Attr::nested(1, changed)]).is_err());
306
+ }
307
+ // Every lookup names its set with a plain, uninverted membership test, and
308
+ // the key registers are part of identity.
309
+ let lookups: Vec<_> = program
310
+ .iter()
311
+ .filter(|(kind, attributes)| *kind == 6 && rule_identity(attributes).unwrap().1.windows(6).any(|window| window == b"lookup"))
312
+ .collect();
313
+ assert_eq!(lookups.len(), 5);
314
+ let lookup = |data: Vec<Attr>| rule(vec![expression("lookup", data)]);
315
+ let exact = vec![Attr::string(1, "host_grant"), Attr::u32(2, 1), Attr::u32(5, 0)];
316
+ let identity = rule_identity(&lookup(exact.clone())).unwrap();
317
+ for changed in [
318
+ vec![Attr::string(1, "host_grant_arrival"), Attr::u32(2, 1), Attr::u32(5, 0)],
319
+ vec![Attr::string(1, "host_grant"), Attr::u32(2, 9), Attr::u32(5, 0)],
320
+ ] {
321
+ assert_ne!(rule_identity(&lookup(changed)).unwrap(), identity);
322
+ }
323
+ for invalid in [
324
+ vec![Attr::string(1, "host_grant"), Attr::u32(2, 1), Attr::u32(5, 1)],
325
+ vec![Attr::string(1, "host_grant"), Attr::u32(2, 1)],
326
+ vec![Attr::string(1, "host_grant"), Attr::u32(2, 1), Attr::u32(3, 2), Attr::u32(5, 0)],
327
+ ] {
328
+ assert!(rule_identity(&lookup(invalid)).is_err());
329
+ }
330
+ }
@@ -24,6 +24,75 @@ fn allocation_pool_guard_maximum_graph_persist_and_lost_ack_recovery() {
24
24
  super::egress_tests::recover("pools_v2", serde_json::from_value(policy).unwrap());
25
25
  }
26
26
 
27
+ /// A full host-grant set persists, is re-verified element by element by a fresh
28
+ /// owner, and is replaced and released exactly like the rest of the graph.
29
+ #[test]
30
+ #[ignore = "requires disposable isolated native qualification guest"]
31
+ fn allocation_pool_guard_host_grant_set_persist_and_lost_ack_recovery() {
32
+ isolated();
33
+ let grants: Vec<_> = (0..1024_u16)
34
+ .map(|index| crate::egress::hostgrant_tests::host_grant(
35
+ if index % 2 == 0 { "tcp" } else { "udp" },
36
+ "10.240.0.1",
37
+ "10.241.0.2",
38
+ 20000 + index / 2,
39
+ ))
40
+ .collect();
41
+ let policy = crate::egress::hostgrant_tests::granted(egress::tests::pool_guard(), json!(grants));
42
+ super::egress_tests::recover("grant_pools_v2", serde_json::from_value(policy).unwrap());
43
+ }
44
+
45
+ /// The set is CONSTANT and bound: even after the owning socket is gone, another
46
+ /// privileged socket cannot add or remove a grant, and a fresh owner verifies
47
+ /// the exact element set before adopting it.
48
+ #[test]
49
+ #[ignore = "requires disposable isolated native qualification guest"]
50
+ fn allocation_pool_guard_host_grant_set_rejects_foreign_element_changes() {
51
+ isolated();
52
+ let grant = crate::egress::hostgrant_tests::host_grant("tcp", "10.240.0.1", "10.241.0.2", 8080);
53
+ let policy: egress::Policy = serde_json::from_value(crate::egress::hostgrant_tests::granted(
54
+ egress::tests::pool_guard(),
55
+ json!([grant]),
56
+ ))
57
+ .unwrap();
58
+ let prepared: Prepared = policy.prepare().unwrap().into();
59
+ let mut owner = Owner::new(options("grant_pool_foreign")).unwrap();
60
+ let applied = owner
61
+ .reconcile(Transition { previous: None, target: prepared.clone() })
62
+ .unwrap();
63
+ drop(owner);
64
+ let table = "snft_grant_pool_foreign";
65
+ let (_, elements) = prepared
66
+ .program(table)
67
+ .unwrap()
68
+ .into_iter()
69
+ .find(|(kind, _)| *kind == 12)
70
+ .unwrap();
71
+ let list = wire::attrs(&wire::one(&elements, 3).unwrap().value).unwrap();
72
+ let key = wire::attrs(&wire::attrs(&list[0].value).unwrap()[0].value).unwrap()[0].value.clone();
73
+ let mut foreign = key.clone();
74
+ foreign[16] ^= 1;
75
+ let element = |key: Vec<u8>| {
76
+ vec![
77
+ Attr::string(1, table),
78
+ Attr::string(2, "host_grant"),
79
+ Attr::nested(3, vec![Attr::nested(1, vec![Attr::nested(1, vec![Attr::bytes(1, key)])])]),
80
+ ]
81
+ };
82
+ let mut socket = wire::Socket::open().unwrap();
83
+ for (operation, key) in [(12, foreign), (14, key)] {
84
+ let generation = socket.generation().unwrap();
85
+ assert!(socket.batch(generation, vec![(operation, 0x600, element(key))]).is_err());
86
+ }
87
+ let mut owner = Owner::new(options("grant_pool_foreign")).unwrap();
88
+ let recovered = owner
89
+ .reconcile(Transition { previous: None, target: prepared })
90
+ .unwrap();
91
+ assert_eq!(recovered.receipt.table_handle, applied.receipt.table_handle);
92
+ assert!(owner.inspect().enforced);
93
+ owner.release(recovered).unwrap();
94
+ }
95
+
27
96
  // Foreign capture controls exist only in this offline guest. They are never
28
97
  // composed into or removed with the managed policy under test.
29
98
  fn foreign_captures() {
@@ -58,7 +58,7 @@ fn sockaddr(value: &str) -> libc::sockaddr_in {
58
58
  /// A TCP client with an exact source address and port, which std cannot express.
59
59
  /// The collision this qualifies is the client's source port, so it must be chosen
60
60
  /// and not left to the ephemeral range.
61
- fn connect_from(source: &str, destination: &str) -> std::io::Result<TcpStream> {
61
+ pub(super) fn connect_from(source: &str, destination: &str) -> std::io::Result<TcpStream> {
62
62
  let raw = unsafe { libc::socket(libc::AF_INET, libc::SOCK_STREAM | libc::SOCK_CLOEXEC, 0) };
63
63
  assert!(raw >= 0, "socket: {}", std::io::Error::last_os_error());
64
64
  let stream = unsafe { TcpStream::from_raw_fd(raw) };