@push.rocks/smartnftables 2.4.0 → 2.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/changelog.md +19 -0
- package/dist_rust/smartnftables_linux_amd64_musl +0 -0
- package/dist_rust/smartnftables_linux_amd64_musl.tsrust-build.json +5 -5
- package/dist_rust/smartnftables_linux_arm64_musl +0 -0
- package/dist_rust/smartnftables_linux_arm64_musl.tsrust-build.json +5 -5
- package/dist_ts/00_commitinfo_data.js +1 -1
- package/dist_ts/classes.manageddockerforwarding.js +3 -3
- package/dist_ts/classes.managednftables.d.ts +5 -0
- package/dist_ts/classes.managednftables.js +13 -6
- package/dist_ts/managed.egress.types.d.ts +31 -1
- package/package.json +8 -8
- package/readme.md +91 -3
- package/rust/src/egress.compile.rs +49 -1
- package/rust/src/egress.host.rs +22 -0
- package/rust/src/egress.hostgrant.rs +187 -0
- package/rust/src/egress.poolguard.rs +7 -0
- package/rust/src/egress.router.rs +45 -4
- package/rust/src/egress.rs +150 -1
- package/rust/src/egress_hostgrant_tests.rs +948 -0
- package/rust/src/main.rs +6 -2
- package/rust/src/owner.rs +116 -4
- package/rust/src/owner_egress_tests.rs +1 -1
- package/rust/src/owner_hostgrant_traffic_tests.rs +280 -0
- package/rust/src/owner_identity_tests.rs +80 -1
- package/rust/src/owner_poolguard_tests.rs +69 -0
- package/rust/src/owner_router_traffic_tests.rs +1 -1
- package/rust/src/owner_tests.rs +3 -0
- package/rust/src/policy.rs +1 -1
- package/rust/src/wire.rs +15 -1
- package/ts/00_commitinfo_data.ts +1 -1
- package/ts/classes.manageddockerforwarding.ts +2 -2
- package/ts/classes.managednftables.ts +13 -5
- package/ts/managed.egress.types.ts +32 -1
package/rust/src/main.rs
CHANGED
|
@@ -11,6 +11,10 @@ mod wire;
|
|
|
11
11
|
use serde::Deserialize;
|
|
12
12
|
use std::io::{BufRead, Read, Write};
|
|
13
13
|
|
|
14
|
+
/// One request line from the facade; the facade's `managedIpcBytes` bounds the
|
|
15
|
+
/// same lines. A transition carries two complete policies.
|
|
16
|
+
pub(crate) const IPC_BYTES: usize = 1_048_576;
|
|
17
|
+
|
|
14
18
|
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
|
15
19
|
pub enum Error {
|
|
16
20
|
Invalid,
|
|
@@ -141,12 +145,12 @@ fn main() {
|
|
|
141
145
|
loop {
|
|
142
146
|
let mut bytes = Vec::new();
|
|
143
147
|
let read = std::io::Read::by_ref(&mut input)
|
|
144
|
-
.take(
|
|
148
|
+
.take(IPC_BYTES as u64 + 1)
|
|
145
149
|
.read_until(b'\n', &mut bytes);
|
|
146
150
|
if matches!(read, Ok(0)) {
|
|
147
151
|
break;
|
|
148
152
|
}
|
|
149
|
-
if read.is_err() || bytes.len() >
|
|
153
|
+
if read.is_err() || bytes.len() > IPC_BYTES || bytes.last() != Some(&b'\n') {
|
|
150
154
|
std::process::exit(2);
|
|
151
155
|
}
|
|
152
156
|
let request: Request = match serde_json::from_slice(&bytes) {
|
package/rust/src/owner.rs
CHANGED
|
@@ -66,6 +66,8 @@ struct Graph {
|
|
|
66
66
|
table: Vec<Attr>,
|
|
67
67
|
chains: Vec<Vec<Attr>>,
|
|
68
68
|
rules: Vec<Vec<Attr>>,
|
|
69
|
+
/// Each named set with the key of every element it holds.
|
|
70
|
+
sets: Vec<(Vec<Attr>, Vec<Vec<u8>>)>,
|
|
69
71
|
}
|
|
70
72
|
pub struct Owner {
|
|
71
73
|
identity: Identity,
|
|
@@ -220,8 +222,31 @@ impl Owner {
|
|
|
220
222
|
}
|
|
221
223
|
}
|
|
222
224
|
let rules = socket.query(7, vec![Attr::string(1, &table_name)], true)?;
|
|
225
|
+
// Set dumps select the table. Every element is read back, so a set
|
|
226
|
+
// whose content differs from the compiled target never verifies.
|
|
227
|
+
let mut sets = Vec::new();
|
|
228
|
+
for set in socket.query(10, vec![Attr::string(1, &table_name)], true)? {
|
|
229
|
+
if wire::text(&set, 1)? != table_name {
|
|
230
|
+
return Err(Error::Conflict);
|
|
231
|
+
}
|
|
232
|
+
let name = wire::text(&set, 2)?;
|
|
233
|
+
let mut keys = Vec::new();
|
|
234
|
+
for message in socket.query(
|
|
235
|
+
13,
|
|
236
|
+
vec![Attr::string(1, &table_name), Attr::string(2, &name)],
|
|
237
|
+
true,
|
|
238
|
+
)? {
|
|
239
|
+
if wire::text(&message, 1)? != table_name || wire::text(&message, 2)? != name {
|
|
240
|
+
return Err(Error::Conflict);
|
|
241
|
+
}
|
|
242
|
+
if let Ok(list) = wire::one(&message, 3) {
|
|
243
|
+
keys.extend(element_keys(&wire::attrs(&list.value)?)?);
|
|
244
|
+
}
|
|
245
|
+
}
|
|
246
|
+
sets.push((set, keys));
|
|
247
|
+
}
|
|
223
248
|
// The managed compiler never creates any of these. Foreign objects reject.
|
|
224
|
-
for operation in [
|
|
249
|
+
for operation in [19, 23] {
|
|
225
250
|
if !socket
|
|
226
251
|
.query(operation, vec![Attr::string(1, &table_name)], true)?
|
|
227
252
|
.is_empty()
|
|
@@ -235,6 +260,7 @@ impl Owner {
|
|
|
235
260
|
table,
|
|
236
261
|
chains,
|
|
237
262
|
rules,
|
|
263
|
+
sets,
|
|
238
264
|
}));
|
|
239
265
|
}
|
|
240
266
|
}
|
|
@@ -310,7 +336,37 @@ impl Owner {
|
|
|
310
336
|
}
|
|
311
337
|
result
|
|
312
338
|
};
|
|
313
|
-
|
|
339
|
+
// Element order is the set backend's; element keys are the identity.
|
|
340
|
+
let mut expected_sets = std::collections::BTreeMap::<String, (Vec<u8>, Vec<Vec<u8>>)>::new();
|
|
341
|
+
for (kind, attrs) in &program {
|
|
342
|
+
match kind {
|
|
343
|
+
9 => {
|
|
344
|
+
expected_sets.insert(wire::text(attrs, 2)?, (set_identity(attrs)?, Vec::new()));
|
|
345
|
+
}
|
|
346
|
+
12 => {
|
|
347
|
+
let entry = expected_sets
|
|
348
|
+
.get_mut(&wire::text(attrs, 2)?)
|
|
349
|
+
.ok_or(Error::Invalid)?;
|
|
350
|
+
entry.1.extend(element_keys(&wire::attrs(&wire::one(attrs, 3)?.value)?)?);
|
|
351
|
+
}
|
|
352
|
+
_ => {}
|
|
353
|
+
}
|
|
354
|
+
}
|
|
355
|
+
let mut actual_sets = std::collections::BTreeMap::new();
|
|
356
|
+
for (attrs, keys) in &graph.sets {
|
|
357
|
+
if actual_sets
|
|
358
|
+
.insert(wire::text(attrs, 2)?, (set_identity(attrs)?, keys.clone()))
|
|
359
|
+
.is_some()
|
|
360
|
+
{
|
|
361
|
+
return Err(Error::Conflict);
|
|
362
|
+
}
|
|
363
|
+
}
|
|
364
|
+
for entry in expected_sets.values_mut().chain(actual_sets.values_mut()) {
|
|
365
|
+
entry.1.sort();
|
|
366
|
+
}
|
|
367
|
+
Ok(expected_chains == actual_chains
|
|
368
|
+
&& regroup(expected_rules) == regroup(actual_rules)
|
|
369
|
+
&& expected_sets == actual_sets)
|
|
314
370
|
}
|
|
315
371
|
fn adopt(&mut self, graph: Graph, transition: &Transition) -> Result<Graph> {
|
|
316
372
|
self.binding(&graph, transition.previous.as_ref())?;
|
|
@@ -437,6 +493,17 @@ impl Owner {
|
|
|
437
493
|
],
|
|
438
494
|
));
|
|
439
495
|
}
|
|
496
|
+
// After every rule that binds it; deleting a set deletes its elements.
|
|
497
|
+
for (set, _) in &graph.sets {
|
|
498
|
+
operations.push((
|
|
499
|
+
11,
|
|
500
|
+
0,
|
|
501
|
+
vec![
|
|
502
|
+
Attr::string(1, &table_name),
|
|
503
|
+
Attr::u64(16, wire::handle(set, 16)?),
|
|
504
|
+
],
|
|
505
|
+
));
|
|
506
|
+
}
|
|
440
507
|
graph.generation
|
|
441
508
|
} else {
|
|
442
509
|
if transition.previous.is_some() || self.applied.is_some() {
|
|
@@ -454,9 +521,10 @@ impl Owner {
|
|
|
454
521
|
self.socket()?.generation()?
|
|
455
522
|
};
|
|
456
523
|
for (operation, attributes) in transition.target.program(&table_name)? {
|
|
524
|
+
// Chains, sets and elements are created exclusively; rules append.
|
|
457
525
|
operations.push((
|
|
458
526
|
operation,
|
|
459
|
-
if operation ==
|
|
527
|
+
if operation == 6 { 0xc00 } else { 0x600 },
|
|
460
528
|
attributes,
|
|
461
529
|
));
|
|
462
530
|
}
|
|
@@ -612,6 +680,41 @@ fn chain_identity(attributes: &[Attr]) -> Result<Vec<u8>> {
|
|
|
612
680
|
}
|
|
613
681
|
serde_json::to_vec(&wire::canonical(&identity, false)?).map_err(|_| Error::Protocol)
|
|
614
682
|
}
|
|
683
|
+
/// A set is its table, name, flags, key type and key length. The handle, the
|
|
684
|
+
/// informational backend type and count, the transaction ID and an empty
|
|
685
|
+
/// description are not identity; anything else rejects.
|
|
686
|
+
fn set_identity(attributes: &[Attr]) -> Result<Vec<u8>> {
|
|
687
|
+
let mut identity = Vec::new();
|
|
688
|
+
for attribute in attributes {
|
|
689
|
+
match attribute.id() {
|
|
690
|
+
10 | 16 | 19 | 20 => {}
|
|
691
|
+
9 if wire::attrs(&attribute.value)?.is_empty() => {}
|
|
692
|
+
1..=5 => identity.push(attribute.clone()),
|
|
693
|
+
_ => return Err(Error::Conflict),
|
|
694
|
+
}
|
|
695
|
+
}
|
|
696
|
+
for id in [1, 2, 3, 4, 5] {
|
|
697
|
+
wire::one(&identity, id)?;
|
|
698
|
+
}
|
|
699
|
+
serde_json::to_vec(&wire::canonical(&identity, false)?).map_err(|_| Error::Protocol)
|
|
700
|
+
}
|
|
701
|
+
/// Only plain keys: an element with data, flags, timeouts, expressions or any
|
|
702
|
+
/// other extension is not one the compiler created.
|
|
703
|
+
fn element_keys(list: &[Attr]) -> Result<Vec<Vec<u8>>> {
|
|
704
|
+
let mut keys = Vec::new();
|
|
705
|
+
for element in list {
|
|
706
|
+
let fields = wire::attrs(&element.value)?;
|
|
707
|
+
if element.id() != 1 || fields.len() != 1 {
|
|
708
|
+
return Err(Error::Conflict);
|
|
709
|
+
}
|
|
710
|
+
let key = wire::attrs(&wire::one(&fields, 1)?.value)?;
|
|
711
|
+
if key.len() != 1 {
|
|
712
|
+
return Err(Error::Conflict);
|
|
713
|
+
}
|
|
714
|
+
keys.push(wire::one(&key, 1)?.value.clone());
|
|
715
|
+
}
|
|
716
|
+
Ok(keys)
|
|
717
|
+
}
|
|
615
718
|
fn rule_identity(attributes: &[Attr]) -> Result<(String, Vec<u8>)> {
|
|
616
719
|
let chain = wire::text(attributes, 2)?;
|
|
617
720
|
if attributes
|
|
@@ -638,7 +741,7 @@ fn rule_identity(attributes: &[Attr]) -> Result<(String, Vec<u8>)> {
|
|
|
638
741
|
"cmp" => &[3],
|
|
639
742
|
"bitwise" => &[4, 5],
|
|
640
743
|
"range" => &[3, 4],
|
|
641
|
-
"meta" | "payload" | "ct" | "nat" => &[],
|
|
744
|
+
"meta" | "payload" | "ct" | "nat" | "lookup" => &[],
|
|
642
745
|
_ => return Err(Error::Conflict),
|
|
643
746
|
};
|
|
644
747
|
for attribute in &mut data {
|
|
@@ -669,6 +772,7 @@ fn rule_identity(attributes: &[Attr]) -> Result<(String, Vec<u8>)> {
|
|
|
669
772
|
/// neither unknown fields nor default repair may hide a changed kernel rule.
|
|
670
773
|
fn validate_egress_expression(name: &str, data: &[Attr]) -> Result<()> {
|
|
671
774
|
let allowed: &[u16] = match name {
|
|
775
|
+
"lookup" => &[1, 2, 5],
|
|
672
776
|
"ct" | "range" => &[1, 2, 3, 4],
|
|
673
777
|
"nat" => &[1, 2, 3, 4, 5, 6, 7],
|
|
674
778
|
_ => return Ok(()),
|
|
@@ -680,6 +784,14 @@ fn validate_egress_expression(name: &str, data: &[Attr]) -> Result<()> {
|
|
|
680
784
|
wire::one(data, attribute.id())?; // Duplicate fields are never canonical identity.
|
|
681
785
|
}
|
|
682
786
|
match name {
|
|
787
|
+
"lookup" => {
|
|
788
|
+
// A plain membership test of a named set: no map, no inversion.
|
|
789
|
+
wire::text(data, 1)?;
|
|
790
|
+
wire::number(data, 2)?;
|
|
791
|
+
if data.len() != 3 || wire::number(data, 5)? != 0 {
|
|
792
|
+
return Err(Error::Conflict);
|
|
793
|
+
}
|
|
794
|
+
}
|
|
683
795
|
"ct" => {
|
|
684
796
|
let read = data.iter().any(|item| item.id() == 1);
|
|
685
797
|
let write = data.iter().any(|item| item.id() == 4);
|
|
@@ -0,0 +1,280 @@
|
|
|
1
|
+
use super::egress_traffic_tests::{
|
|
2
|
+
allocation, binding, connect, denied, host_policy, ns_ip, protection,
|
|
3
|
+
};
|
|
4
|
+
use super::egress_traffic_tests::{roundtrip, socket};
|
|
5
|
+
use super::router_traffic_tests::connect_from;
|
|
6
|
+
use super::*;
|
|
7
|
+
use crate::egress;
|
|
8
|
+
use serde_json::json;
|
|
9
|
+
use std::io::{Read, Write};
|
|
10
|
+
use std::net::{TcpListener, TcpStream};
|
|
11
|
+
use std::time::Duration;
|
|
12
|
+
|
|
13
|
+
const WORK: &str = "hg_work";
|
|
14
|
+
const ROUTER: &str = "hg_router";
|
|
15
|
+
const HOST: &str = "hg_host";
|
|
16
|
+
const PEER: &str = "hg_peer";
|
|
17
|
+
|
|
18
|
+
/// The contract maximum: the two probed grants among 1022 others, so every
|
|
19
|
+
/// admission is a lookup in a full set. TCP and UDP 30000–30510 are members.
|
|
20
|
+
fn grants() -> Vec<egress::HostGrant> {
|
|
21
|
+
let grant = |protocol: &str, port: u16| egress::HostGrant {
|
|
22
|
+
protocol: protocol.into(),
|
|
23
|
+
source_address: "10.240.0.1".into(),
|
|
24
|
+
destination_address: "10.241.0.2".into(),
|
|
25
|
+
destination_port: port,
|
|
26
|
+
};
|
|
27
|
+
let mut result = vec![grant("tcp", 8080), grant("udp", 5353)];
|
|
28
|
+
for port in 30000..=30510 {
|
|
29
|
+
result.push(grant("tcp", port));
|
|
30
|
+
result.push(grant("udp", port));
|
|
31
|
+
}
|
|
32
|
+
assert_eq!(result.len(), 1024);
|
|
33
|
+
result
|
|
34
|
+
}
|
|
35
|
+
/// One workload behind the router with leased public egress on every port of
|
|
36
|
+
/// both protocols, exactly as a node composes it, so a leased classifier would
|
|
37
|
+
/// claim the workload's replies if host grants were not default-zone flows.
|
|
38
|
+
fn router_policy(revision: u64, granted: bool) -> egress::Policy {
|
|
39
|
+
let link = binding("work", "10.241.0.1", "veth");
|
|
40
|
+
let mut generation = allocation();
|
|
41
|
+
generation["conntrackZone"] = json!(17);
|
|
42
|
+
generation["conntrackLabel"] = json!("a".repeat(32));
|
|
43
|
+
generation["grants"] = json!([
|
|
44
|
+
{"sourceEndpoint":"work","sourcePrefix":"10.241.0.2/32","destination":{"kind":"public","prefix":"0.0.0.0/0"},
|
|
45
|
+
"protocol":"udp","sourcePort":null,"destinationPort":null,"sourcePortRange":{"first":10000,"last":10015}},
|
|
46
|
+
{"sourceEndpoint":"work","sourcePrefix":"10.241.0.2/32","destination":{"kind":"public","prefix":"0.0.0.0/0"},
|
|
47
|
+
"protocol":"tcp","sourcePort":null,"destinationPort":null,"sourcePortRange":{"first":10016,"last":10031}}
|
|
48
|
+
]);
|
|
49
|
+
let mut policy: egress::Policy =
|
|
50
|
+
serde_json::from_value(json!({"schemaVersion":2,"revision":revision,
|
|
51
|
+
"scope":{"kind":"routerEgress",
|
|
52
|
+
"endpoints":[{"id":"work","interfaceIndex":link["interfaceIndex"],"interfaceName":"work",
|
|
53
|
+
"interfaceKind":"veth","sourcePrefixes":["10.241.0.2/32"]}],
|
|
54
|
+
"rules":[],"links":[link],"handoff":binding("handoff","10.240.0.2","veth"),
|
|
55
|
+
"protection":protection(),"generations":[generation]}}))
|
|
56
|
+
.unwrap();
|
|
57
|
+
let egress::Scope::RouterEgress(scope) = &mut policy.scope else {
|
|
58
|
+
panic!()
|
|
59
|
+
};
|
|
60
|
+
scope.host_grants = if granted { grants() } else { vec![] };
|
|
61
|
+
policy
|
|
62
|
+
}
|
|
63
|
+
fn transit_policy(revision: u64, granted: bool) -> egress::Policy {
|
|
64
|
+
let mut policy = host_policy();
|
|
65
|
+
policy.revision = revision;
|
|
66
|
+
let egress::Scope::HostTransit(scope) = &mut policy.scope else {
|
|
67
|
+
panic!()
|
|
68
|
+
};
|
|
69
|
+
scope.host_grants = if granted { grants() } else { vec![] };
|
|
70
|
+
policy
|
|
71
|
+
}
|
|
72
|
+
fn guard_policy(revision: u64, granted: bool) -> egress::Policy {
|
|
73
|
+
let mut policy: egress::Policy = serde_json::from_value(egress::tests::pool_guard()).unwrap();
|
|
74
|
+
policy.revision = revision;
|
|
75
|
+
let egress::Scope::AllocationPoolGuard(scope) = &mut policy.scope else {
|
|
76
|
+
panic!()
|
|
77
|
+
};
|
|
78
|
+
scope.host_grants = if granted { grants() } else { vec![] };
|
|
79
|
+
policy
|
|
80
|
+
}
|
|
81
|
+
/// One policy owner in its own namespace; every transition runs there.
|
|
82
|
+
struct Hop {
|
|
83
|
+
namespace: &'static str,
|
|
84
|
+
owner: Option<Owner>,
|
|
85
|
+
applied: Option<Applied>,
|
|
86
|
+
}
|
|
87
|
+
impl Hop {
|
|
88
|
+
fn new(namespace: &'static str, table: &'static str) -> Self {
|
|
89
|
+
let owner = in_namespace(namespace, move || Owner::new(options(table)).unwrap());
|
|
90
|
+
Self {
|
|
91
|
+
namespace,
|
|
92
|
+
owner: Some(owner),
|
|
93
|
+
applied: None,
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
fn apply(&mut self, build: fn(u64, bool) -> egress::Policy, revision: u64, granted: bool) {
|
|
97
|
+
let mut owner = self.owner.take().unwrap();
|
|
98
|
+
let previous = self.applied.take();
|
|
99
|
+
let (owner, applied) = in_namespace(self.namespace, move || {
|
|
100
|
+
let target: Prepared = build(revision, granted).prepare().unwrap().into();
|
|
101
|
+
target
|
|
102
|
+
.validate_interfaces()
|
|
103
|
+
.expect("host grant fixture binding");
|
|
104
|
+
let applied = owner.reconcile(Transition { previous, target }).unwrap();
|
|
105
|
+
assert!(owner.inspect().enforced);
|
|
106
|
+
(owner, applied)
|
|
107
|
+
});
|
|
108
|
+
self.owner = Some(owner);
|
|
109
|
+
self.applied = Some(applied);
|
|
110
|
+
}
|
|
111
|
+
fn release(mut self) {
|
|
112
|
+
let mut owner = self.owner.take().unwrap();
|
|
113
|
+
let applied = self.applied.take().unwrap();
|
|
114
|
+
in_namespace(self.namespace, move || {
|
|
115
|
+
assert!(owner.inspect().enforced);
|
|
116
|
+
owner.release(applied).unwrap();
|
|
117
|
+
});
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
/// A host-origin TCP flow from an exact source address. On success the workload
|
|
121
|
+
/// must observe that exact source and exchange data both ways.
|
|
122
|
+
fn dial(
|
|
123
|
+
source: &'static str,
|
|
124
|
+
destination: &'static str,
|
|
125
|
+
listener: &TcpListener,
|
|
126
|
+
) -> Option<(TcpStream, TcpStream)> {
|
|
127
|
+
let client = in_namespace(HOST, move || connect_from(source, destination)).ok()?;
|
|
128
|
+
let (server, peer) = listener.accept().unwrap();
|
|
129
|
+
assert_eq!(peer.ip().to_string(), source.split(':').next().unwrap());
|
|
130
|
+
let (mut client, mut server) = (client, server);
|
|
131
|
+
for stream in [&client, &server] {
|
|
132
|
+
stream
|
|
133
|
+
.set_read_timeout(Some(Duration::from_secs(1)))
|
|
134
|
+
.unwrap();
|
|
135
|
+
}
|
|
136
|
+
client.write_all(b"ping").unwrap();
|
|
137
|
+
server.read_exact(&mut [0; 4]).unwrap();
|
|
138
|
+
server.write_all(b"pong").unwrap();
|
|
139
|
+
client.read_exact(&mut [0; 4]).unwrap();
|
|
140
|
+
Some((client, server))
|
|
141
|
+
}
|
|
142
|
+
fn dark(source: &'static str, destination: &'static str, listener: &TcpListener) {
|
|
143
|
+
assert!(
|
|
144
|
+
dial(source, destination, listener).is_none(),
|
|
145
|
+
"{source} -> {destination}"
|
|
146
|
+
);
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
#[test]
|
|
150
|
+
#[ignore = "requires disposable isolated native qualification guest"]
|
|
151
|
+
fn v2_host_grant_opens_exactly_the_host_to_workload_port_through_guard_transit_and_router() {
|
|
152
|
+
isolated();
|
|
153
|
+
for name in [WORK, ROUTER, HOST, PEER] {
|
|
154
|
+
ip(&["netns", "add", name]);
|
|
155
|
+
ns_ip(name, &["link", "set", "lo", "up"]);
|
|
156
|
+
in_namespace(name, || {
|
|
157
|
+
std::fs::write("/proc/sys/net/ipv4/ip_forward", "1").unwrap()
|
|
158
|
+
});
|
|
159
|
+
}
|
|
160
|
+
connect(
|
|
161
|
+
WORK,
|
|
162
|
+
"client",
|
|
163
|
+
"10.241.0.2/24",
|
|
164
|
+
ROUTER,
|
|
165
|
+
"work",
|
|
166
|
+
"10.241.0.1/24",
|
|
167
|
+
);
|
|
168
|
+
connect(
|
|
169
|
+
ROUTER,
|
|
170
|
+
"handoff",
|
|
171
|
+
"10.240.0.2/30",
|
|
172
|
+
HOST,
|
|
173
|
+
"router",
|
|
174
|
+
"10.240.0.1/30",
|
|
175
|
+
);
|
|
176
|
+
connect(
|
|
177
|
+
HOST,
|
|
178
|
+
"uplink",
|
|
179
|
+
"192.0.2.2/24",
|
|
180
|
+
PEER,
|
|
181
|
+
"underlay",
|
|
182
|
+
"192.0.2.3/24",
|
|
183
|
+
);
|
|
184
|
+
ns_ip(WORK, &["route", "add", "default", "via", "10.241.0.1"]);
|
|
185
|
+
ns_ip(ROUTER, &["route", "add", "default", "via", "10.240.0.1"]);
|
|
186
|
+
// The workload route exists throughout, so every later denial is caused by
|
|
187
|
+
// policy and not by a missing route.
|
|
188
|
+
ns_ip(
|
|
189
|
+
HOST,
|
|
190
|
+
&["route", "add", "10.241.0.0/24", "via", "10.240.0.2"],
|
|
191
|
+
);
|
|
192
|
+
let granted_tcp = in_namespace(WORK, || TcpListener::bind("10.241.0.2:8080").unwrap());
|
|
193
|
+
let other_tcp = in_namespace(WORK, || TcpListener::bind("10.241.0.2:8081").unwrap());
|
|
194
|
+
let member_tcp = in_namespace(WORK, || TcpListener::bind("10.241.0.2:30300").unwrap());
|
|
195
|
+
let neighbour_tcp = in_namespace(WORK, || TcpListener::bind("10.241.0.2:30511").unwrap());
|
|
196
|
+
let host_tcp = in_namespace(HOST, || TcpListener::bind("10.240.0.1:9000").unwrap());
|
|
197
|
+
let granted_udp = socket(WORK, "10.241.0.2:5353");
|
|
198
|
+
let other_udp = socket(WORK, "10.241.0.2:5354");
|
|
199
|
+
let protocol_udp = socket(WORK, "10.241.0.2:8080");
|
|
200
|
+
// Every UDP probe uses its own socket: an unreplied conntrack entry would
|
|
201
|
+
// otherwise carry an earlier decision across a policy change.
|
|
202
|
+
let host_udp = || socket(HOST, "10.240.0.1:0");
|
|
203
|
+
// Positive controls without policy: every probe below can succeed, including
|
|
204
|
+
// the foreign source address and the workload-origin dial of the host.
|
|
205
|
+
assert!(dial("10.240.0.1:0", "10.241.0.2:8080", &granted_tcp).is_some());
|
|
206
|
+
assert!(dial("10.240.0.1:0", "10.241.0.2:8081", &other_tcp).is_some());
|
|
207
|
+
assert!(dial("192.0.2.2:0", "10.241.0.2:8080", &granted_tcp).is_some());
|
|
208
|
+
assert!(dial("10.240.0.1:0", "10.241.0.2:30511", &neighbour_tcp).is_some());
|
|
209
|
+
roundtrip(&host_udp(), &granted_udp);
|
|
210
|
+
roundtrip(&host_udp(), &other_udp);
|
|
211
|
+
roundtrip(&host_udp(), &protocol_udp);
|
|
212
|
+
let workload_dial = |expected: bool| {
|
|
213
|
+
let result = in_namespace(WORK, || {
|
|
214
|
+
TcpStream::connect_timeout(&"10.240.0.1:9000".parse().unwrap(), Duration::from_secs(1))
|
|
215
|
+
});
|
|
216
|
+
assert_eq!(result.is_ok(), expected);
|
|
217
|
+
if expected {
|
|
218
|
+
host_tcp.accept().unwrap();
|
|
219
|
+
}
|
|
220
|
+
};
|
|
221
|
+
workload_dial(true);
|
|
222
|
+
|
|
223
|
+
let mut guard = Hop::new(HOST, "host_grant_guard");
|
|
224
|
+
let mut transit = Hop::new(HOST, "host_grant_transit");
|
|
225
|
+
let mut router = Hop::new(ROUTER, "host_grant_router");
|
|
226
|
+
guard.apply(guard_policy, 1, false);
|
|
227
|
+
transit.apply(transit_policy, 1, false);
|
|
228
|
+
router.apply(router_policy, 1, false);
|
|
229
|
+
// Without grants the host never reaches the workload pool.
|
|
230
|
+
dark("10.240.0.1:0", "10.241.0.2:8080", &granted_tcp);
|
|
231
|
+
denied(&host_udp(), &granted_udp, true);
|
|
232
|
+
|
|
233
|
+
// Transit and router grants alone do not pass the host-wide guard.
|
|
234
|
+
transit.apply(transit_policy, 2, true);
|
|
235
|
+
router.apply(router_policy, 2, true);
|
|
236
|
+
dark("10.240.0.1:0", "10.241.0.2:8080", &granted_tcp);
|
|
237
|
+
denied(&host_udp(), &granted_udp, true);
|
|
238
|
+
|
|
239
|
+
// With the guard exception the exact tuples pass all three tables, and the
|
|
240
|
+
// workload's replies return in the default zone despite leased egress on
|
|
241
|
+
// every port of both protocols.
|
|
242
|
+
guard.apply(guard_policy, 2, true);
|
|
243
|
+
let (mut established, mut served) =
|
|
244
|
+
dial("10.240.0.1:0", "10.241.0.2:8080", &granted_tcp).expect("granted host tcp");
|
|
245
|
+
let observed = roundtrip(&host_udp(), &granted_udp);
|
|
246
|
+
assert_eq!(observed.ip().to_string(), "10.240.0.1");
|
|
247
|
+
// Any member of the full set passes; the tuple just outside it does not.
|
|
248
|
+
assert!(dial("10.240.0.1:0", "10.241.0.2:30300", &member_tcp).is_some());
|
|
249
|
+
dark("10.240.0.1:0", "10.241.0.2:30511", &neighbour_tcp);
|
|
250
|
+
// Another port, protocol, source address or direction stays blocked.
|
|
251
|
+
dark("10.240.0.1:0", "10.241.0.2:8081", &other_tcp);
|
|
252
|
+
denied(&host_udp(), &other_udp, true);
|
|
253
|
+
denied(&host_udp(), &protocol_udp, true);
|
|
254
|
+
dark("192.0.2.2:0", "10.241.0.2:8080", &granted_tcp);
|
|
255
|
+
workload_dial(false);
|
|
256
|
+
|
|
257
|
+
// The router hop is independently required: withdrawing only its grants
|
|
258
|
+
// closes the path behind the host's own tables, and re-granting reopens it.
|
|
259
|
+
router.apply(router_policy, 3, false);
|
|
260
|
+
dark("10.240.0.1:0", "10.241.0.2:8080", &granted_tcp);
|
|
261
|
+
denied(&host_udp(), &granted_udp, false);
|
|
262
|
+
router.apply(router_policy, 4, true);
|
|
263
|
+
assert!(dial("10.240.0.1:0", "10.241.0.2:8080", &granted_tcp).is_some());
|
|
264
|
+
|
|
265
|
+
// Withdrawal in every scope closes new flows and the established one.
|
|
266
|
+
guard.apply(guard_policy, 3, false);
|
|
267
|
+
transit.apply(transit_policy, 3, false);
|
|
268
|
+
router.apply(router_policy, 5, false);
|
|
269
|
+
dark("10.240.0.1:0", "10.241.0.2:8080", &granted_tcp);
|
|
270
|
+
denied(&host_udp(), &granted_udp, true);
|
|
271
|
+
established.write_all(b"late").unwrap();
|
|
272
|
+
assert!(served.read_exact(&mut [0; 4]).is_err());
|
|
273
|
+
drop(established);
|
|
274
|
+
drop(served);
|
|
275
|
+
|
|
276
|
+
router.release();
|
|
277
|
+
transit.release();
|
|
278
|
+
guard.release();
|
|
279
|
+
println!("V2_HOST_GRANT_PROOF grants_per_scope=1024 set_member=true set_neighbour_denied=true positive_controls=true ungranted_dark=true guard_required=true exact_tcp=true exact_udp=true source_preserved=true other_port_denied=true other_protocol_denied=true other_source_denied=true workload_origin_denied=true router_required=true regranted=true withdrawn_dark=true established_withdrawn=true");
|
|
280
|
+
}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
use super::rule_identity;
|
|
1
|
+
use super::{element_keys, rule_identity, set_identity};
|
|
2
2
|
use crate::wire::{self, Attr};
|
|
3
3
|
|
|
4
4
|
fn expression(name: &str, data: Vec<Attr>) -> Attr {
|
|
@@ -249,3 +249,82 @@ fn published_router_graph_identity_preserves_destination_nat_and_admission_opera
|
|
|
249
249
|
.count()
|
|
250
250
|
);
|
|
251
251
|
}
|
|
252
|
+
|
|
253
|
+
/// A host-grant set as Linux dumps it: the compiled creation plus the kernel's
|
|
254
|
+
/// handle, an empty description, the backend type and the element count.
|
|
255
|
+
fn dumped_set(created: &[Attr]) -> Vec<Attr> {
|
|
256
|
+
let mut dump: Vec<Attr> = created.iter().filter(|attribute| attribute.id() != 10).cloned().collect();
|
|
257
|
+
dump.push(Attr::u64(16, 7));
|
|
258
|
+
dump.push(Attr::nested(9, vec![]));
|
|
259
|
+
dump.push(Attr::string(19, "nft_rhash_type"));
|
|
260
|
+
dump.push(Attr::u32(20, 1024));
|
|
261
|
+
dump
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
#[test]
|
|
265
|
+
fn host_grant_sets_elements_and_lookups_have_exact_graph_identity() {
|
|
266
|
+
let prepared = serde_json::from_value::<crate::egress::Policy>(
|
|
267
|
+
crate::egress::hostgrant_tests::granted(
|
|
268
|
+
crate::egress::tests::router(),
|
|
269
|
+
serde_json::json!([crate::egress::hostgrant_tests::host_grant("tcp", "10.240.0.1", "10.241.0.2", 8080)]),
|
|
270
|
+
),
|
|
271
|
+
)
|
|
272
|
+
.unwrap()
|
|
273
|
+
.prepare()
|
|
274
|
+
.unwrap();
|
|
275
|
+
let program = prepared.program("snft_identity").unwrap();
|
|
276
|
+
let sets: Vec<_> = program.iter().filter(|(kind, _)| *kind == 9).map(|(_, attributes)| attributes.clone()).collect();
|
|
277
|
+
assert_eq!(sets.len(), 2);
|
|
278
|
+
for created in &sets {
|
|
279
|
+
// The kernel's informational attributes are not identity.
|
|
280
|
+
assert_eq!(set_identity(created).unwrap(), set_identity(&dumped_set(created)).unwrap());
|
|
281
|
+
// A changed key length, flag or any unknown attribute is a different set.
|
|
282
|
+
for (id, value) in [(5, Attr::u32(5, 4)), (3, Attr::u32(3, 0))] {
|
|
283
|
+
let mut changed: Vec<_> = created.iter().filter(|attribute| attribute.id() != id).cloned().collect();
|
|
284
|
+
changed.push(value);
|
|
285
|
+
assert_ne!(set_identity(&changed).unwrap(), set_identity(created).unwrap());
|
|
286
|
+
}
|
|
287
|
+
for foreign in [
|
|
288
|
+
Attr::u64(11, 1000),
|
|
289
|
+
Attr::bytes(13, b"foreign".to_vec()),
|
|
290
|
+
Attr::nested(9, vec![Attr::u32(1, 64)]),
|
|
291
|
+
] {
|
|
292
|
+
let mut changed = dumped_set(created);
|
|
293
|
+
changed.push(foreign);
|
|
294
|
+
assert!(set_identity(&changed).is_err());
|
|
295
|
+
}
|
|
296
|
+
}
|
|
297
|
+
// Elements are plain keys; any extension rejects.
|
|
298
|
+
let (_, elements) = program.iter().find(|(kind, _)| *kind == 12).unwrap();
|
|
299
|
+
let list = wire::attrs(&wire::one(elements, 3).unwrap().value).unwrap();
|
|
300
|
+
assert_eq!(element_keys(&list).unwrap().len(), 1);
|
|
301
|
+
let key = wire::attrs(&list[0].value).unwrap();
|
|
302
|
+
for extension in [Attr::u32(3, 1), Attr::nested(2, vec![Attr::bytes(1, vec![1; 4])])] {
|
|
303
|
+
let mut changed = key.clone();
|
|
304
|
+
changed.push(extension);
|
|
305
|
+
assert!(element_keys(&[Attr::nested(1, changed)]).is_err());
|
|
306
|
+
}
|
|
307
|
+
// Every lookup names its set with a plain, uninverted membership test, and
|
|
308
|
+
// the key registers are part of identity.
|
|
309
|
+
let lookups: Vec<_> = program
|
|
310
|
+
.iter()
|
|
311
|
+
.filter(|(kind, attributes)| *kind == 6 && rule_identity(attributes).unwrap().1.windows(6).any(|window| window == b"lookup"))
|
|
312
|
+
.collect();
|
|
313
|
+
assert_eq!(lookups.len(), 5);
|
|
314
|
+
let lookup = |data: Vec<Attr>| rule(vec![expression("lookup", data)]);
|
|
315
|
+
let exact = vec![Attr::string(1, "host_grant"), Attr::u32(2, 1), Attr::u32(5, 0)];
|
|
316
|
+
let identity = rule_identity(&lookup(exact.clone())).unwrap();
|
|
317
|
+
for changed in [
|
|
318
|
+
vec![Attr::string(1, "host_grant_arrival"), Attr::u32(2, 1), Attr::u32(5, 0)],
|
|
319
|
+
vec![Attr::string(1, "host_grant"), Attr::u32(2, 9), Attr::u32(5, 0)],
|
|
320
|
+
] {
|
|
321
|
+
assert_ne!(rule_identity(&lookup(changed)).unwrap(), identity);
|
|
322
|
+
}
|
|
323
|
+
for invalid in [
|
|
324
|
+
vec![Attr::string(1, "host_grant"), Attr::u32(2, 1), Attr::u32(5, 1)],
|
|
325
|
+
vec![Attr::string(1, "host_grant"), Attr::u32(2, 1)],
|
|
326
|
+
vec![Attr::string(1, "host_grant"), Attr::u32(2, 1), Attr::u32(3, 2), Attr::u32(5, 0)],
|
|
327
|
+
] {
|
|
328
|
+
assert!(rule_identity(&lookup(invalid)).is_err());
|
|
329
|
+
}
|
|
330
|
+
}
|
|
@@ -24,6 +24,75 @@ fn allocation_pool_guard_maximum_graph_persist_and_lost_ack_recovery() {
|
|
|
24
24
|
super::egress_tests::recover("pools_v2", serde_json::from_value(policy).unwrap());
|
|
25
25
|
}
|
|
26
26
|
|
|
27
|
+
/// A full host-grant set persists, is re-verified element by element by a fresh
|
|
28
|
+
/// owner, and is replaced and released exactly like the rest of the graph.
|
|
29
|
+
#[test]
|
|
30
|
+
#[ignore = "requires disposable isolated native qualification guest"]
|
|
31
|
+
fn allocation_pool_guard_host_grant_set_persist_and_lost_ack_recovery() {
|
|
32
|
+
isolated();
|
|
33
|
+
let grants: Vec<_> = (0..1024_u16)
|
|
34
|
+
.map(|index| crate::egress::hostgrant_tests::host_grant(
|
|
35
|
+
if index % 2 == 0 { "tcp" } else { "udp" },
|
|
36
|
+
"10.240.0.1",
|
|
37
|
+
"10.241.0.2",
|
|
38
|
+
20000 + index / 2,
|
|
39
|
+
))
|
|
40
|
+
.collect();
|
|
41
|
+
let policy = crate::egress::hostgrant_tests::granted(egress::tests::pool_guard(), json!(grants));
|
|
42
|
+
super::egress_tests::recover("grant_pools_v2", serde_json::from_value(policy).unwrap());
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
/// The set is CONSTANT and bound: even after the owning socket is gone, another
|
|
46
|
+
/// privileged socket cannot add or remove a grant, and a fresh owner verifies
|
|
47
|
+
/// the exact element set before adopting it.
|
|
48
|
+
#[test]
|
|
49
|
+
#[ignore = "requires disposable isolated native qualification guest"]
|
|
50
|
+
fn allocation_pool_guard_host_grant_set_rejects_foreign_element_changes() {
|
|
51
|
+
isolated();
|
|
52
|
+
let grant = crate::egress::hostgrant_tests::host_grant("tcp", "10.240.0.1", "10.241.0.2", 8080);
|
|
53
|
+
let policy: egress::Policy = serde_json::from_value(crate::egress::hostgrant_tests::granted(
|
|
54
|
+
egress::tests::pool_guard(),
|
|
55
|
+
json!([grant]),
|
|
56
|
+
))
|
|
57
|
+
.unwrap();
|
|
58
|
+
let prepared: Prepared = policy.prepare().unwrap().into();
|
|
59
|
+
let mut owner = Owner::new(options("grant_pool_foreign")).unwrap();
|
|
60
|
+
let applied = owner
|
|
61
|
+
.reconcile(Transition { previous: None, target: prepared.clone() })
|
|
62
|
+
.unwrap();
|
|
63
|
+
drop(owner);
|
|
64
|
+
let table = "snft_grant_pool_foreign";
|
|
65
|
+
let (_, elements) = prepared
|
|
66
|
+
.program(table)
|
|
67
|
+
.unwrap()
|
|
68
|
+
.into_iter()
|
|
69
|
+
.find(|(kind, _)| *kind == 12)
|
|
70
|
+
.unwrap();
|
|
71
|
+
let list = wire::attrs(&wire::one(&elements, 3).unwrap().value).unwrap();
|
|
72
|
+
let key = wire::attrs(&wire::attrs(&list[0].value).unwrap()[0].value).unwrap()[0].value.clone();
|
|
73
|
+
let mut foreign = key.clone();
|
|
74
|
+
foreign[16] ^= 1;
|
|
75
|
+
let element = |key: Vec<u8>| {
|
|
76
|
+
vec![
|
|
77
|
+
Attr::string(1, table),
|
|
78
|
+
Attr::string(2, "host_grant"),
|
|
79
|
+
Attr::nested(3, vec![Attr::nested(1, vec![Attr::nested(1, vec![Attr::bytes(1, key)])])]),
|
|
80
|
+
]
|
|
81
|
+
};
|
|
82
|
+
let mut socket = wire::Socket::open().unwrap();
|
|
83
|
+
for (operation, key) in [(12, foreign), (14, key)] {
|
|
84
|
+
let generation = socket.generation().unwrap();
|
|
85
|
+
assert!(socket.batch(generation, vec![(operation, 0x600, element(key))]).is_err());
|
|
86
|
+
}
|
|
87
|
+
let mut owner = Owner::new(options("grant_pool_foreign")).unwrap();
|
|
88
|
+
let recovered = owner
|
|
89
|
+
.reconcile(Transition { previous: None, target: prepared })
|
|
90
|
+
.unwrap();
|
|
91
|
+
assert_eq!(recovered.receipt.table_handle, applied.receipt.table_handle);
|
|
92
|
+
assert!(owner.inspect().enforced);
|
|
93
|
+
owner.release(recovered).unwrap();
|
|
94
|
+
}
|
|
95
|
+
|
|
27
96
|
// Foreign capture controls exist only in this offline guest. They are never
|
|
28
97
|
// composed into or removed with the managed policy under test.
|
|
29
98
|
fn foreign_captures() {
|
|
@@ -58,7 +58,7 @@ fn sockaddr(value: &str) -> libc::sockaddr_in {
|
|
|
58
58
|
/// A TCP client with an exact source address and port, which std cannot express.
|
|
59
59
|
/// The collision this qualifies is the client's source port, so it must be chosen
|
|
60
60
|
/// and not left to the ephemeral range.
|
|
61
|
-
fn connect_from(source: &str, destination: &str) -> std::io::Result<TcpStream> {
|
|
61
|
+
pub(super) fn connect_from(source: &str, destination: &str) -> std::io::Result<TcpStream> {
|
|
62
62
|
let raw = unsafe { libc::socket(libc::AF_INET, libc::SOCK_STREAM | libc::SOCK_CLOEXEC, 0) };
|
|
63
63
|
assert!(raw >= 0, "socket: {}", std::io::Error::last_os_error());
|
|
64
64
|
let stream = unsafe { TcpStream::from_raw_fd(raw) };
|