@push.rocks/smartnftables 2.3.0 → 2.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/changelog.md +19 -0
- package/dist_rust/smartnftables_linux_amd64_musl +0 -0
- package/dist_rust/smartnftables_linux_amd64_musl.tsrust-build.json +5 -5
- package/dist_rust/smartnftables_linux_arm64_musl +0 -0
- package/dist_rust/smartnftables_linux_arm64_musl.tsrust-build.json +5 -5
- package/dist_ts/00_commitinfo_data.js +1 -1
- package/dist_ts/classes.manageddockerforwarding.js +3 -3
- package/dist_ts/classes.managednftables.d.ts +5 -0
- package/dist_ts/classes.managednftables.js +13 -6
- package/dist_ts/managed.egress.types.d.ts +53 -1
- package/package.json +7 -7
- package/readme.md +153 -4
- package/rust/src/egress.compile.rs +49 -1
- package/rust/src/egress.host.rs +22 -0
- package/rust/src/egress.hostgrant.rs +187 -0
- package/rust/src/egress.poolguard.rs +7 -0
- package/rust/src/egress.router.rs +189 -3
- package/rust/src/egress.rs +244 -1
- package/rust/src/egress_hostgrant_tests.rs +948 -0
- package/rust/src/egress_tests.rs +414 -0
- package/rust/src/main.rs +6 -2
- package/rust/src/owner.rs +116 -4
- package/rust/src/owner_egress_tests.rs +1 -1
- package/rust/src/owner_egress_traffic_tests.rs +3 -3
- package/rust/src/owner_host_traffic_tests.rs +4 -4
- package/rust/src/owner_hostgrant_traffic_tests.rs +280 -0
- package/rust/src/owner_identity_tests.rs +125 -1
- package/rust/src/owner_poolguard_tests.rs +69 -0
- package/rust/src/owner_router_traffic_tests.rs +393 -0
- package/rust/src/owner_tests.rs +6 -0
- package/rust/src/policy.rs +1 -1
- package/rust/src/wire.rs +15 -1
- package/ts/00_commitinfo_data.ts +1 -1
- package/ts/classes.manageddockerforwarding.ts +2 -2
- package/ts/classes.managednftables.ts +13 -5
- package/ts/managed.egress.types.ts +55 -1
|
@@ -0,0 +1,393 @@
|
|
|
1
|
+
use super::egress_traffic_tests::{allocation, binding, connect, host_policy, ns_ip, protection};
|
|
2
|
+
use super::egress_traffic_tests::{roundtrip, socket};
|
|
3
|
+
use super::host_traffic_tests::{chain_names, control, dark_udp, published_udp};
|
|
4
|
+
use super::*;
|
|
5
|
+
use crate::egress;
|
|
6
|
+
use serde_json::json;
|
|
7
|
+
use std::io::{Read, Write};
|
|
8
|
+
use std::net::{TcpListener, TcpStream};
|
|
9
|
+
use std::os::fd::FromRawFd;
|
|
10
|
+
use std::time::Duration;
|
|
11
|
+
|
|
12
|
+
const WORK: &str = "rp_work";
|
|
13
|
+
const ROUTER: &str = "rp_router";
|
|
14
|
+
const HOST: &str = "rp_host";
|
|
15
|
+
const PEER: &str = "rp_peer";
|
|
16
|
+
|
|
17
|
+
/// One workload veth behind the router, one leased generation, and the inbound
|
|
18
|
+
/// publications of the second hop. The first hop is the host-transit publication.
|
|
19
|
+
fn router_policy(published: bool) -> egress::Policy {
|
|
20
|
+
let link = binding("work", "10.241.0.1", "veth");
|
|
21
|
+
let mut generation = allocation();
|
|
22
|
+
generation["conntrackZone"] = json!(17);
|
|
23
|
+
generation["conntrackLabel"] = json!("a".repeat(32));
|
|
24
|
+
// Both leased grants claim the same public address and a destination port a
|
|
25
|
+
// published client may legitimately use as its source port, so the generation
|
|
26
|
+
// and the publications compete for the workload's outgoing packets.
|
|
27
|
+
generation["grants"] = json!([
|
|
28
|
+
{"sourceEndpoint":"work","sourcePrefix":"10.241.0.2/32","destination":{"kind":"public","prefix":"203.0.113.2/32"},
|
|
29
|
+
"protocol":"udp","sourcePort":null,"destinationPort":42000,"sourcePortRange":{"first":10000,"last":10015}},
|
|
30
|
+
{"sourceEndpoint":"work","sourcePrefix":"10.241.0.2/32","destination":{"kind":"public","prefix":"203.0.113.2/32"},
|
|
31
|
+
"protocol":"tcp","sourcePort":null,"destinationPort":443,"sourcePortRange":{"first":10016,"last":10031}}
|
|
32
|
+
]);
|
|
33
|
+
let ports = if published {
|
|
34
|
+
json!([
|
|
35
|
+
{"protocol":"tcp","transitPort":8443,"endpointPort":9443,
|
|
36
|
+
"endpointAddress":"10.241.0.2","transitSourceAddress":"10.240.0.2"},
|
|
37
|
+
{"protocol":"udp","transitPort":8125,"endpointPort":9125,
|
|
38
|
+
"endpointAddress":"10.241.0.2","transitSourceAddress":"10.240.0.2"}
|
|
39
|
+
])
|
|
40
|
+
} else {
|
|
41
|
+
json!([])
|
|
42
|
+
};
|
|
43
|
+
serde_json::from_value(json!({"schemaVersion":2,"revision":1,"scope":{"kind":"routerEgress",
|
|
44
|
+
"endpoints":[{"id":"work","interfaceIndex":link["interfaceIndex"],"interfaceName":"work",
|
|
45
|
+
"interfaceKind":"veth","sourcePrefixes":["10.241.0.2/32"]}],
|
|
46
|
+
"rules":[],"links":[link],"handoff":binding("handoff","10.240.0.2","veth"),
|
|
47
|
+
"protection":protection(),"generations":[generation],"publishedPorts":ports}}))
|
|
48
|
+
.unwrap()
|
|
49
|
+
}
|
|
50
|
+
fn sockaddr(value: &str) -> libc::sockaddr_in {
|
|
51
|
+
let parsed: std::net::SocketAddrV4 = value.parse().unwrap();
|
|
52
|
+
let mut raw: libc::sockaddr_in = unsafe { std::mem::zeroed() };
|
|
53
|
+
raw.sin_family = libc::AF_INET as u16;
|
|
54
|
+
raw.sin_port = parsed.port().to_be();
|
|
55
|
+
raw.sin_addr.s_addr = u32::from(*parsed.ip()).to_be();
|
|
56
|
+
raw
|
|
57
|
+
}
|
|
58
|
+
/// A TCP client with an exact source address and port, which std cannot express.
|
|
59
|
+
/// The collision this qualifies is the client's source port, so it must be chosen
|
|
60
|
+
/// and not left to the ephemeral range.
|
|
61
|
+
pub(super) fn connect_from(source: &str, destination: &str) -> std::io::Result<TcpStream> {
|
|
62
|
+
let raw = unsafe { libc::socket(libc::AF_INET, libc::SOCK_STREAM | libc::SOCK_CLOEXEC, 0) };
|
|
63
|
+
assert!(raw >= 0, "socket: {}", std::io::Error::last_os_error());
|
|
64
|
+
let stream = unsafe { TcpStream::from_raw_fd(raw) };
|
|
65
|
+
let bound = sockaddr(source);
|
|
66
|
+
assert_eq!(
|
|
67
|
+
unsafe {
|
|
68
|
+
libc::bind(
|
|
69
|
+
raw,
|
|
70
|
+
(&bound as *const libc::sockaddr_in).cast(),
|
|
71
|
+
std::mem::size_of_val(&bound) as _,
|
|
72
|
+
)
|
|
73
|
+
},
|
|
74
|
+
0,
|
|
75
|
+
"bind: {}",
|
|
76
|
+
std::io::Error::last_os_error()
|
|
77
|
+
);
|
|
78
|
+
stream.set_nonblocking(true)?;
|
|
79
|
+
let target = sockaddr(destination);
|
|
80
|
+
if unsafe {
|
|
81
|
+
libc::connect(
|
|
82
|
+
raw,
|
|
83
|
+
(&target as *const libc::sockaddr_in).cast(),
|
|
84
|
+
std::mem::size_of_val(&target) as _,
|
|
85
|
+
)
|
|
86
|
+
} != 0
|
|
87
|
+
{
|
|
88
|
+
let error = std::io::Error::last_os_error();
|
|
89
|
+
if error.raw_os_error() != Some(libc::EINPROGRESS) {
|
|
90
|
+
return Err(error);
|
|
91
|
+
}
|
|
92
|
+
let mut waiting = libc::pollfd {
|
|
93
|
+
fd: raw,
|
|
94
|
+
events: libc::POLLOUT,
|
|
95
|
+
revents: 0,
|
|
96
|
+
};
|
|
97
|
+
assert!(unsafe { libc::poll(&mut waiting, 1, 1000) } >= 0);
|
|
98
|
+
let mut code: libc::c_int = 0;
|
|
99
|
+
let mut size = std::mem::size_of_val(&code) as libc::socklen_t;
|
|
100
|
+
assert_eq!(
|
|
101
|
+
unsafe {
|
|
102
|
+
libc::getsockopt(
|
|
103
|
+
raw,
|
|
104
|
+
libc::SOL_SOCKET,
|
|
105
|
+
libc::SO_ERROR,
|
|
106
|
+
(&mut code as *mut libc::c_int).cast(),
|
|
107
|
+
&mut size,
|
|
108
|
+
)
|
|
109
|
+
},
|
|
110
|
+
0
|
|
111
|
+
);
|
|
112
|
+
if code != 0 || waiting.revents & libc::POLLOUT == 0 {
|
|
113
|
+
return Err(std::io::Error::from_raw_os_error(if code != 0 {
|
|
114
|
+
code
|
|
115
|
+
} else {
|
|
116
|
+
libc::ETIMEDOUT
|
|
117
|
+
}));
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
stream.set_nonblocking(false)?;
|
|
121
|
+
Ok(stream)
|
|
122
|
+
}
|
|
123
|
+
/// The host publications of the first hop: two with a router counterpart, and one
|
|
124
|
+
/// whose transit port the router never publishes.
|
|
125
|
+
fn published_host_policy() -> egress::Policy {
|
|
126
|
+
let mut policy = host_policy();
|
|
127
|
+
let egress::Scope::HostTransit(scope) = &mut policy.scope else {
|
|
128
|
+
panic!()
|
|
129
|
+
};
|
|
130
|
+
scope.published_ports = [("tcp", 443, 8443), ("udp", 8125, 8125), ("udp", 8126, 8126)]
|
|
131
|
+
.into_iter()
|
|
132
|
+
.map(|(protocol, host_port, target_port)| egress::PublishedPort {
|
|
133
|
+
protocol: protocol.into(),
|
|
134
|
+
host_port,
|
|
135
|
+
target_port,
|
|
136
|
+
target_address: "10.240.0.2".into(),
|
|
137
|
+
host_ip: "192.0.2.2".into(),
|
|
138
|
+
})
|
|
139
|
+
.collect();
|
|
140
|
+
policy
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
#[test]
|
|
144
|
+
#[ignore = "requires disposable isolated native qualification guest"]
|
|
145
|
+
fn v2_router_publishes_transit_ports_to_the_workload_and_removes_them_with_the_generation() {
|
|
146
|
+
isolated();
|
|
147
|
+
for name in [WORK, ROUTER, HOST, PEER] {
|
|
148
|
+
ip(&["netns", "add", name]);
|
|
149
|
+
ns_ip(name, &["link", "set", "lo", "up"]);
|
|
150
|
+
in_namespace(name, || {
|
|
151
|
+
std::fs::write("/proc/sys/net/ipv4/ip_forward", "1").unwrap()
|
|
152
|
+
});
|
|
153
|
+
}
|
|
154
|
+
connect(
|
|
155
|
+
WORK,
|
|
156
|
+
"client",
|
|
157
|
+
"10.241.0.2/24",
|
|
158
|
+
ROUTER,
|
|
159
|
+
"work",
|
|
160
|
+
"10.241.0.1/24",
|
|
161
|
+
);
|
|
162
|
+
connect(
|
|
163
|
+
ROUTER,
|
|
164
|
+
"handoff",
|
|
165
|
+
"10.240.0.2/30",
|
|
166
|
+
HOST,
|
|
167
|
+
"router",
|
|
168
|
+
"10.240.0.1/30",
|
|
169
|
+
);
|
|
170
|
+
connect(
|
|
171
|
+
HOST,
|
|
172
|
+
"uplink",
|
|
173
|
+
"192.0.2.2/24",
|
|
174
|
+
PEER,
|
|
175
|
+
"underlay",
|
|
176
|
+
"192.0.2.3/24",
|
|
177
|
+
);
|
|
178
|
+
ns_ip(WORK, &["route", "add", "default", "via", "10.241.0.1"]);
|
|
179
|
+
ns_ip(ROUTER, &["route", "add", "default", "via", "10.240.0.1"]);
|
|
180
|
+
// The unpublished workload route exists throughout, so every later denial is
|
|
181
|
+
// caused by policy and not by a missing route.
|
|
182
|
+
ns_ip(HOST, &["route", "add", "10.241.0.0/24", "via", "10.240.0.2"]);
|
|
183
|
+
ns_ip(
|
|
184
|
+
HOST,
|
|
185
|
+
&["route", "add", "203.0.113.2/32", "via", "192.0.2.3"],
|
|
186
|
+
);
|
|
187
|
+
ns_ip(PEER, &["address", "add", "203.0.113.2/32", "dev", "lo"]);
|
|
188
|
+
ns_ip(PEER, &["route", "add", "default", "via", "192.0.2.2"]);
|
|
189
|
+
let workload_udp = socket(WORK, "10.241.0.2:9125");
|
|
190
|
+
let workload_direct = socket(WORK, "10.241.0.2:8444");
|
|
191
|
+
let workload_unpublished = socket(WORK, "10.241.0.2:9126");
|
|
192
|
+
let work_egress = socket(WORK, "10.241.0.2:40001");
|
|
193
|
+
let public = socket(PEER, "203.0.113.2:42000");
|
|
194
|
+
// Every probe uses its own client port: an unreplied conntrack entry would
|
|
195
|
+
// otherwise carry its null translation across the policy change.
|
|
196
|
+
let control_client = socket(PEER, "192.0.2.3:40010");
|
|
197
|
+
let client_udp = socket(PEER, "192.0.2.3:40000");
|
|
198
|
+
// Positive controls: without policy both hops forward to the workload, and the
|
|
199
|
+
// published address is not reachable, so every later result is caused by policy.
|
|
200
|
+
control(&control_client, "10.241.0.2:8444", &workload_direct, true);
|
|
201
|
+
dark_udp(
|
|
202
|
+
&socket(PEER, "192.0.2.3:40011"),
|
|
203
|
+
&workload_udp,
|
|
204
|
+
"192.0.2.2:8125",
|
|
205
|
+
);
|
|
206
|
+
let (host_owner, host_applied) = in_namespace(HOST, || {
|
|
207
|
+
let mut owner = Owner::new(options("router_published_host")).unwrap();
|
|
208
|
+
let target: Prepared = published_host_policy().prepare().unwrap().into();
|
|
209
|
+
target
|
|
210
|
+
.validate_interfaces()
|
|
211
|
+
.expect("published host fixture binding");
|
|
212
|
+
let applied = owner
|
|
213
|
+
.reconcile(Transition {
|
|
214
|
+
previous: None,
|
|
215
|
+
target,
|
|
216
|
+
})
|
|
217
|
+
.unwrap();
|
|
218
|
+
assert!(owner.inspect().enforced);
|
|
219
|
+
(owner, applied)
|
|
220
|
+
});
|
|
221
|
+
let (router_owner, router_applied, names) = in_namespace(ROUTER, || {
|
|
222
|
+
let mut owner = Owner::new(options("router_published")).unwrap();
|
|
223
|
+
let target: Prepared = router_policy(true).prepare().unwrap().into();
|
|
224
|
+
target
|
|
225
|
+
.validate_interfaces()
|
|
226
|
+
.expect("published router fixture binding");
|
|
227
|
+
let applied = owner
|
|
228
|
+
.reconcile(Transition {
|
|
229
|
+
previous: None,
|
|
230
|
+
target,
|
|
231
|
+
})
|
|
232
|
+
.unwrap();
|
|
233
|
+
assert!(owner.inspect().enforced);
|
|
234
|
+
let names = chain_names(&mut owner);
|
|
235
|
+
(owner, applied, names)
|
|
236
|
+
});
|
|
237
|
+
assert!(names.contains(&"pre".to_string()));
|
|
238
|
+
// Two hops of destination NAT deliver UDP to the workload; the workload sees the
|
|
239
|
+
// actual client and the client sees the published uplink address back.
|
|
240
|
+
published_udp(
|
|
241
|
+
&client_udp,
|
|
242
|
+
&workload_udp,
|
|
243
|
+
"192.0.2.2:8125",
|
|
244
|
+
"192.0.2.3:40000",
|
|
245
|
+
);
|
|
246
|
+
// The fixture observes ingress only, so the translated opening is read where it
|
|
247
|
+
// arrives: the workload side of the last veth.
|
|
248
|
+
let observer = in_namespace(WORK, || packet_fixture::PacketSocket::open("client"));
|
|
249
|
+
let listener = in_namespace(WORK, || TcpListener::bind("10.241.0.2:9443").unwrap());
|
|
250
|
+
let mut client = in_namespace(PEER, || {
|
|
251
|
+
TcpStream::connect_timeout(&"192.0.2.2:443".parse().unwrap(), Duration::from_secs(1))
|
|
252
|
+
.expect("published tcp opening")
|
|
253
|
+
});
|
|
254
|
+
let (mut server, peer) = listener.accept().unwrap();
|
|
255
|
+
assert_eq!(peer.ip().to_string(), "192.0.2.3");
|
|
256
|
+
assert_eq!(client.peer_addr().unwrap().to_string(), "192.0.2.2:443");
|
|
257
|
+
for stream in [&client, &server] {
|
|
258
|
+
stream.set_read_timeout(Some(Duration::from_secs(1))).unwrap();
|
|
259
|
+
}
|
|
260
|
+
client.write_all(b"ping").unwrap();
|
|
261
|
+
server.read_exact(&mut [0; 4]).unwrap();
|
|
262
|
+
server.write_all(b"pong").unwrap();
|
|
263
|
+
client.read_exact(&mut [0; 4]).unwrap();
|
|
264
|
+
let syn = observer
|
|
265
|
+
.matching(|packet| {
|
|
266
|
+
packet.len() >= 40
|
|
267
|
+
&& packet[9] == 6
|
|
268
|
+
&& packet[16..20] == [10, 241, 0, 2]
|
|
269
|
+
&& packet[22..24] == 9443_u16.to_be_bytes()
|
|
270
|
+
&& packet[33] & 0x17 == 0x02
|
|
271
|
+
})
|
|
272
|
+
.expect("translated opening at the workload");
|
|
273
|
+
assert_eq!(syn[12..16], [192, 0, 2, 3]);
|
|
274
|
+
// Leased egress is unaffected by the publication in the same generation.
|
|
275
|
+
let outer = roundtrip(&work_egress, &public);
|
|
276
|
+
assert_eq!(outer.ip().to_string(), "192.0.2.2");
|
|
277
|
+
assert!((10000..=10015).contains(&outer.port()));
|
|
278
|
+
// A client whose source port equals a leased grant's destination port on that
|
|
279
|
+
// grant's public address is the exact collision raw classification must
|
|
280
|
+
// survive: the workload's published packets carry the same tuple, and raw runs
|
|
281
|
+
// before conntrack can separate the two flows. The publication must keep its
|
|
282
|
+
// own translation, so the reply arrives from the published address and port.
|
|
283
|
+
published_udp(&public, &workload_udp, "192.0.2.2:8125", "203.0.113.2:42000");
|
|
284
|
+
let mut colliding = in_namespace(PEER, || connect_from("203.0.113.2:443", "192.0.2.2:443"))
|
|
285
|
+
.expect("published tcp opening from a leased destination port");
|
|
286
|
+
let (mut served, observed) = listener.accept().unwrap();
|
|
287
|
+
assert_eq!(observed.to_string(), "203.0.113.2:443");
|
|
288
|
+
assert_eq!(colliding.peer_addr().unwrap().to_string(), "192.0.2.2:443");
|
|
289
|
+
for stream in [&colliding, &served] {
|
|
290
|
+
stream.set_read_timeout(Some(Duration::from_secs(1))).unwrap();
|
|
291
|
+
}
|
|
292
|
+
colliding.write_all(b"ping").unwrap();
|
|
293
|
+
served.read_exact(&mut [0; 4]).unwrap();
|
|
294
|
+
served.write_all(b"pong").unwrap();
|
|
295
|
+
colliding.read_exact(&mut [0; 4]).unwrap();
|
|
296
|
+
drop(colliding);
|
|
297
|
+
drop(served);
|
|
298
|
+
// Inject the second hop directly: only SYN with FIN/RST/ACK clear may open the
|
|
299
|
+
// publication, and an unpublished transit port stays behind the router barrier.
|
|
300
|
+
let injector = in_namespace(HOST, || packet_fixture::PacketSocket::open("router"));
|
|
301
|
+
let handoff_mac = in_namespace(ROUTER, || packet_fixture::mac("handoff"));
|
|
302
|
+
let arrived = |port: u16, flags: u8| {
|
|
303
|
+
observer
|
|
304
|
+
.matching(|packet| {
|
|
305
|
+
packet.len() >= 40
|
|
306
|
+
&& packet[9] == 6
|
|
307
|
+
&& packet[16..20] == [10, 241, 0, 2]
|
|
308
|
+
&& packet[20..22] == port.to_be_bytes()
|
|
309
|
+
&& packet[22..24] == 9443_u16.to_be_bytes()
|
|
310
|
+
&& packet[24..28] == 100_u32.to_be_bytes()
|
|
311
|
+
&& packet[33] == flags
|
|
312
|
+
})
|
|
313
|
+
.is_some()
|
|
314
|
+
};
|
|
315
|
+
for (port, flags) in [(40050_u16, 0x10_u8), (40051, 0x04), (40052, 0x12)] {
|
|
316
|
+
injector.send(
|
|
317
|
+
handoff_mac,
|
|
318
|
+
&packet_fixture::tcp([192, 0, 2, 3], [10, 240, 0, 2], port, 8443, flags),
|
|
319
|
+
);
|
|
320
|
+
assert!(!arrived(port, flags), "{flags:#04x}");
|
|
321
|
+
}
|
|
322
|
+
// The same injector opens the publication with an exact SYN, so the denials
|
|
323
|
+
// above are caused by the opening flags and not by the fixture.
|
|
324
|
+
injector.send(
|
|
325
|
+
handoff_mac,
|
|
326
|
+
&packet_fixture::tcp([192, 0, 2, 3], [10, 240, 0, 2], 40053, 8443, 0x02),
|
|
327
|
+
);
|
|
328
|
+
assert!(arrived(40053, 0x02));
|
|
329
|
+
injector.send(
|
|
330
|
+
handoff_mac,
|
|
331
|
+
&packet_fixture::tcp([192, 0, 2, 3], [10, 240, 0, 2], 40054, 9999, 0x02),
|
|
332
|
+
);
|
|
333
|
+
assert!(observer
|
|
334
|
+
.matching(|packet| packet.len() >= 40
|
|
335
|
+
&& packet[9] == 6
|
|
336
|
+
&& packet[20..22] == 40054_u16.to_be_bytes()
|
|
337
|
+
&& packet[24..28] == 100_u32.to_be_bytes())
|
|
338
|
+
.is_none());
|
|
339
|
+
// A host publication whose transit port the router never published cannot pass
|
|
340
|
+
// the router barrier, while the published port beside it keeps working.
|
|
341
|
+
dark_udp(
|
|
342
|
+
&socket(PEER, "192.0.2.3:40012"),
|
|
343
|
+
&workload_unpublished,
|
|
344
|
+
"192.0.2.2:8126",
|
|
345
|
+
);
|
|
346
|
+
control(&control_client, "10.241.0.2:8444", &workload_direct, false);
|
|
347
|
+
drop(client);
|
|
348
|
+
drop(server);
|
|
349
|
+
// Withdrawing the router publication is an ordinary generation replacement. The
|
|
350
|
+
// host publication stays installed, so this isolates the second hop exactly.
|
|
351
|
+
let (mut router_owner, rolled_back, names) = in_namespace(ROUTER, {
|
|
352
|
+
let mut owner = router_owner;
|
|
353
|
+
move || {
|
|
354
|
+
let mut withdrawn = router_policy(false);
|
|
355
|
+
withdrawn.revision = 2;
|
|
356
|
+
let applied = owner
|
|
357
|
+
.reconcile(Transition {
|
|
358
|
+
previous: Some(router_applied),
|
|
359
|
+
target: withdrawn.prepare().unwrap().into(),
|
|
360
|
+
})
|
|
361
|
+
.unwrap();
|
|
362
|
+
assert!(owner.inspect().enforced);
|
|
363
|
+
let names = chain_names(&mut owner);
|
|
364
|
+
(owner, applied, names)
|
|
365
|
+
}
|
|
366
|
+
});
|
|
367
|
+
assert!(!names.contains(&"pre".to_string()));
|
|
368
|
+
dark_udp(
|
|
369
|
+
&socket(PEER, "192.0.2.3:40003"),
|
|
370
|
+
&workload_udp,
|
|
371
|
+
"192.0.2.2:8125",
|
|
372
|
+
);
|
|
373
|
+
assert!(in_namespace(PEER, || TcpStream::connect_timeout(
|
|
374
|
+
&"192.0.2.2:443".parse().unwrap(),
|
|
375
|
+
Duration::from_secs(1)
|
|
376
|
+
))
|
|
377
|
+
.is_err());
|
|
378
|
+
in_namespace(ROUTER, move || {
|
|
379
|
+
assert!(router_owner.inspect().enforced);
|
|
380
|
+
router_owner.release(rolled_back).unwrap();
|
|
381
|
+
});
|
|
382
|
+
dark_udp(
|
|
383
|
+
&socket(PEER, "192.0.2.3:40004"),
|
|
384
|
+
&workload_udp,
|
|
385
|
+
"192.0.2.2:8125",
|
|
386
|
+
);
|
|
387
|
+
in_namespace(HOST, move || {
|
|
388
|
+
let mut owner = host_owner;
|
|
389
|
+
assert!(owner.inspect().enforced);
|
|
390
|
+
owner.release(host_applied).unwrap();
|
|
391
|
+
});
|
|
392
|
+
println!("V2_ROUTER_PUBLISHED_PROOF two_hop_udp_delivered=true two_hop_tcp_delivered=true client_address_preserved=true reverse_translation_both_hops=true syn_only_opening=true nonsyn_opening_denied=true injector_positive_control=true unpublished_transit_port_denied=true unpublished_workload_port_denied=true leased_egress_unaffected=true leased_destination_port_client_tcp=true leased_destination_port_client_udp=true barrier_positive_control=true withdrawn_chain_absent=true withdrawn_port_dark=true released_port_dark=true");
|
|
393
|
+
}
|
package/rust/src/owner_tests.rs
CHANGED
|
@@ -21,9 +21,15 @@ mod packet_fixture;
|
|
|
21
21
|
#[path = "owner_host_traffic_tests.rs"]
|
|
22
22
|
mod host_traffic_tests;
|
|
23
23
|
|
|
24
|
+
#[path = "owner_router_traffic_tests.rs"]
|
|
25
|
+
mod router_traffic_tests;
|
|
26
|
+
|
|
24
27
|
#[path = "owner_poolguard_tests.rs"]
|
|
25
28
|
mod poolguard_tests;
|
|
26
29
|
|
|
30
|
+
#[path = "owner_hostgrant_traffic_tests.rs"]
|
|
31
|
+
mod hostgrant_traffic_tests;
|
|
32
|
+
|
|
27
33
|
#[path = "owner_detach_tests.rs"]
|
|
28
34
|
mod detach_tests;
|
|
29
35
|
|
package/rust/src/policy.rs
CHANGED
|
@@ -195,7 +195,7 @@ impl Policy {
|
|
|
195
195
|
};
|
|
196
196
|
// Compile during preparation; an oversized native batch never reaches the kernel.
|
|
197
197
|
let program = prepared.program(&format!("snft_{}", "x".repeat(59)))?;
|
|
198
|
-
// Reserve enough of the
|
|
198
|
+
// Reserve enough of the batch for deleting a previous maximum
|
|
199
199
|
// graph, then installing this complete graph in the same transaction.
|
|
200
200
|
if program.len() > 768
|
|
201
201
|
|| program
|
package/rust/src/wire.rs
CHANGED
|
@@ -24,6 +24,10 @@ impl Attr {
|
|
|
24
24
|
pub fn nested(kind: u16, value: Vec<Attr>) -> Self {
|
|
25
25
|
Self::bytes(kind | 0x8000, encode_attrs(&value))
|
|
26
26
|
}
|
|
27
|
+
/// Netlink attribute lengths are 16 bits; a longer value cannot be encoded.
|
|
28
|
+
pub fn encodable(&self) -> bool {
|
|
29
|
+
self.value.len() <= usize::from(u16::MAX) - 4
|
|
30
|
+
}
|
|
27
31
|
pub fn id(&self) -> u16 {
|
|
28
32
|
self.kind & 0x3fff
|
|
29
33
|
}
|
|
@@ -31,6 +35,9 @@ impl Attr {
|
|
|
31
35
|
pub fn encode_attrs(attrs: &[Attr]) -> Vec<u8> {
|
|
32
36
|
let mut result = Vec::new();
|
|
33
37
|
for attr in attrs {
|
|
38
|
+
// Every caller bounds its attributes far below this; a truncated length
|
|
39
|
+
// would make the kernel parse a different message.
|
|
40
|
+
assert!(attr.encodable(), "netlink attribute exceeds 65,531 bytes");
|
|
34
41
|
let len = 4 + attr.value.len();
|
|
35
42
|
result.extend((len as u16).to_ne_bytes());
|
|
36
43
|
result.extend(attr.kind.to_ne_bytes());
|
|
@@ -117,6 +124,13 @@ struct Message {
|
|
|
117
124
|
sequence: u32,
|
|
118
125
|
payload: Vec<u8>,
|
|
119
126
|
}
|
|
127
|
+
/// One replacement batch: handle-only deletion of a previous maximum graph
|
|
128
|
+
/// (768 operations, about 96 KB), a 100,000-byte target rule graph and 100,000
|
|
129
|
+
/// bytes of target set elements, with framing. Netlink refuses a message larger
|
|
130
|
+
/// than the socket send buffer; Linux's default `net.core.wmem_max` of 212,992
|
|
131
|
+
/// gives a 425,984-byte buffer, and this limit needs at least 160,016.
|
|
132
|
+
pub const BATCH_BYTES: usize = 320_000;
|
|
133
|
+
|
|
120
134
|
pub struct Socket {
|
|
121
135
|
fd: OwnedFd,
|
|
122
136
|
pub port: u32,
|
|
@@ -213,7 +227,7 @@ impl Socket {
|
|
|
213
227
|
Ok((sequence, bytes))
|
|
214
228
|
}
|
|
215
229
|
fn send(&self, bytes: &[u8]) -> Result<()> {
|
|
216
|
-
if bytes.len() >
|
|
230
|
+
if bytes.len() > BATCH_BYTES {
|
|
217
231
|
return Err(Error::Invalid);
|
|
218
232
|
}
|
|
219
233
|
let mut address: libc::sockaddr_nl = unsafe { std::mem::zeroed() };
|
package/ts/00_commitinfo_data.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import * as plugins from './plugins.js';
|
|
2
|
-
import { snapshot, ManagedNftablesError } from './classes.managednftables.js';
|
|
2
|
+
import { snapshot, managedIpcBytes, ManagedNftablesError } from './classes.managednftables.js';
|
|
3
3
|
import type { IManagedDockerForwardingOptions, IManagedDockerForwardingStatus, IManagedDockerForwardingPolicy, IPreparedManagedDockerForwardingPolicy, IManagedDockerForwardingTransition, IAppliedManagedDockerForwardingPolicy, TManagedDockerForwardingCommands } from './managed.docker.types.js';
|
|
4
4
|
|
|
5
5
|
/** One node-level Docker DOCKER-USER contribution. Caller owns durable intents and handoff lifetimes. */
|
|
@@ -39,7 +39,7 @@ export class ManagedDockerForwarding {
|
|
|
39
39
|
localPaths: [], searchSystemPath: false,
|
|
40
40
|
cliArgs: ['--docker-forwarding', ...(this.#options.networkNamespaceFd === undefined ? [] : ['--network-namespace-fd', '3'])],
|
|
41
41
|
inheritedFileDescriptors: this.#options.networkNamespaceFd === undefined ? undefined : [this.#options.networkNamespaceFd],
|
|
42
|
-
readyTimeoutMs: 3000, requestTimeoutMs: 30_000, maxPayloadSize:
|
|
42
|
+
readyTimeoutMs: 3000, requestTimeoutMs: 30_000, maxPayloadSize: managedIpcBytes,
|
|
43
43
|
maxPendingRequestsByMethod: { prepareDockerForwarding: 1, openDockerForwarding: 1, reconcileDockerForwarding: 1, inspectDockerForwarding: 1, releaseDockerForwarding: 1, closeDockerForwarding: 1 },
|
|
44
44
|
});
|
|
45
45
|
this.#bridge.on('exit', () => {
|
|
@@ -6,16 +6,24 @@ export class ManagedNftablesError extends Error {
|
|
|
6
6
|
constructor(public readonly code: string) { super(`Managed nftables ${code}.`); }
|
|
7
7
|
}
|
|
8
8
|
|
|
9
|
+
/** Bounds of the private native IPC. A status carries up to three complete policies (applied,
|
|
10
|
+
* pending previous and pending target); a policy with the maximum 1024 host grants serializes to
|
|
11
|
+
* about 111 KB and 5,300 values, so these bounds hold three of them with room for the rest of a
|
|
12
|
+
* large scope. */
|
|
13
|
+
export const managedIpcBytes = 1_048_576;
|
|
14
|
+
const snapshotBytes = 1_000_000;
|
|
15
|
+
const snapshotValues = 65_536;
|
|
16
|
+
|
|
9
17
|
/** Capture inert input without evaluating getters, proxies, coercions or toJSON. */
|
|
10
18
|
export function snapshot<T>(input: T): T {
|
|
11
19
|
let count = 0;
|
|
12
20
|
let stringBytes = 0;
|
|
13
21
|
const string = (value: string): string => {
|
|
14
|
-
if (value.length >
|
|
22
|
+
if (value.length > snapshotBytes || (stringBytes += Buffer.byteLength(value)) > snapshotBytes) throw new ManagedNftablesError('INVALID');
|
|
15
23
|
return value;
|
|
16
24
|
};
|
|
17
25
|
const copy = (value: unknown, depth: number): unknown => {
|
|
18
|
-
if (++count >
|
|
26
|
+
if (++count > snapshotValues || depth > 20) throw new ManagedNftablesError('INVALID');
|
|
19
27
|
if (typeof value === 'string') return string(value);
|
|
20
28
|
if (value === null || typeof value === 'boolean') return value;
|
|
21
29
|
if (typeof value === 'number' && Number.isSafeInteger(value)) return value;
|
|
@@ -23,7 +31,7 @@ export function snapshot<T>(input: T): T {
|
|
|
23
31
|
const array = Array.isArray(value);
|
|
24
32
|
if (Object.getPrototypeOf(value) !== (array ? Array.prototype : Object.prototype)) throw new ManagedNftablesError('INVALID');
|
|
25
33
|
const keys = Reflect.ownKeys(value);
|
|
26
|
-
if (keys.length >
|
|
34
|
+
if (keys.length > snapshotValues - count) throw new ManagedNftablesError('INVALID');
|
|
27
35
|
for (const key of keys) if (typeof key === 'string') string(key);
|
|
28
36
|
const descriptors = Object.getOwnPropertyDescriptors(value);
|
|
29
37
|
const result: Record<string, unknown> | unknown[] = array ? [] : {};
|
|
@@ -38,7 +46,7 @@ export function snapshot<T>(input: T): T {
|
|
|
38
46
|
return result;
|
|
39
47
|
};
|
|
40
48
|
const result = copy(input, 0) as T;
|
|
41
|
-
if (Buffer.byteLength(JSON.stringify(result)) >
|
|
49
|
+
if (Buffer.byteLength(JSON.stringify(result)) > snapshotBytes) throw new ManagedNftablesError('INVALID');
|
|
42
50
|
return result;
|
|
43
51
|
}
|
|
44
52
|
|
|
@@ -97,7 +105,7 @@ export class ManagedNftables<TPolicy extends TManagedNftPolicy = IManagedNftPoli
|
|
|
97
105
|
localPaths: [], searchSystemPath: false,
|
|
98
106
|
cliArgs: ['--management', ...(this.#options.networkNamespaceFd === undefined ? [] : ['--network-namespace-fd', '3'])],
|
|
99
107
|
inheritedFileDescriptors: this.#options.networkNamespaceFd === undefined ? undefined : [this.#options.networkNamespaceFd],
|
|
100
|
-
readyTimeoutMs: 3000, requestTimeoutMs: 30_000, maxPayloadSize:
|
|
108
|
+
readyTimeoutMs: 3000, requestTimeoutMs: 30_000, maxPayloadSize: managedIpcBytes,
|
|
101
109
|
maxPendingRequestsByMethod: { preparePolicy: 1, openOwner: 1, reconcilePolicy: 1, inspectPolicy: 1, releasePolicy: 1, releasePolicyTransition: 1, detachPolicy: 1, closeOwner: 1 },
|
|
102
110
|
});
|
|
103
111
|
this.#bridge.on('exit', () => {
|
|
@@ -55,6 +55,39 @@ export interface IManagedNftEgressGenerationV2 extends IManagedNftHandoffAllocat
|
|
|
55
55
|
grants: IManagedNftEgressGrantV2[];
|
|
56
56
|
}
|
|
57
57
|
|
|
58
|
+
/** One inbound handoff publication, compiled inside the same router generation.
|
|
59
|
+
* It is the second hop of a published host port: the host translates to
|
|
60
|
+
* `transitSourceAddress:transitPort`, this translates on to the workload. */
|
|
61
|
+
export interface IManagedNftRouterPublishedPortV2 {
|
|
62
|
+
protocol: 'tcp' | 'udp';
|
|
63
|
+
/** Published handoff port, 1–65535 and unique per protocol across the scope.
|
|
64
|
+
* It is the `targetPort` of the matching host-transit publication. */
|
|
65
|
+
transitPort: number;
|
|
66
|
+
endpointPort: number;
|
|
67
|
+
/** Exact current workload endpoint address behind one veth endpoint of this
|
|
68
|
+
* scope. Never a router-local address, a gateway or a platform endpoint. */
|
|
69
|
+
endpointAddress: string;
|
|
70
|
+
/** Exact leased transit source address of one current generation, present on
|
|
71
|
+
* `handoff`. No wildcard, secondary-address or route inference. */
|
|
72
|
+
transitSourceAddress: string;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/** One exact host-origin flow: the node's own host network namespace dials one port of one local
|
|
76
|
+
* workload at its lease address, from the transit host address of the current handoff. Compile the
|
|
77
|
+
* same grant into every scope the flow crosses (`allocationPoolGuard` and `hostTransit` on the host,
|
|
78
|
+
* `routerEgress` in the router namespace); each scope binds it to the authority it proves. It admits
|
|
79
|
+
* that tuple and its replies in the default conntrack zone and nothing else: no loopback, no
|
|
80
|
+
* translation, no source-port selection and no workload-origin flow. */
|
|
81
|
+
export interface IManagedNftHostGrant {
|
|
82
|
+
protocol: 'tcp' | 'udp';
|
|
83
|
+
/** Exact unicast address, never a prefix: the host's own address on the handoff. */
|
|
84
|
+
sourceAddress: string;
|
|
85
|
+
/** Exact unicast workload lease address, never a prefix. */
|
|
86
|
+
destinationAddress: string;
|
|
87
|
+
/** The workload's exact listening port, 1–65535; never a range. */
|
|
88
|
+
destinationPort: number;
|
|
89
|
+
}
|
|
90
|
+
|
|
58
91
|
/** One combined private/TUN/local/egress table; never compose a second ACCEPT over v1 terminal drops. */
|
|
59
92
|
export interface IManagedNftRouterEgressScopeV2 {
|
|
60
93
|
kind: 'routerEgress';
|
|
@@ -66,6 +99,18 @@ export interface IManagedNftRouterEgressScopeV2 {
|
|
|
66
99
|
protection: IManagedNftProtectionV2;
|
|
67
100
|
/** Empty retains private/local enforcement while denying external egress. */
|
|
68
101
|
generations: IManagedNftEgressGenerationV2[];
|
|
102
|
+
/** Optional inbound destination NAT toward a workload endpoint. Absent and empty
|
|
103
|
+
* are the same canonical policy and keep the compiled graph unchanged. Every entry
|
|
104
|
+
* compiles its own handoff DNAT, the handoff ingress classification and the exact
|
|
105
|
+
* forward and reply admission in this generation, so a failed apply, a target
|
|
106
|
+
* without it, or release removes it with the generation. */
|
|
107
|
+
publishedPorts?: IManagedNftRouterPublishedPortV2[];
|
|
108
|
+
/** Optional host-origin grants forwarded from the handoff to a workload endpoint. Each
|
|
109
|
+
* `destinationAddress` is an exact current veth endpoint address of this scope; each
|
|
110
|
+
* `sourceAddress` is protected space that is neither router-local, a platform endpoint nor a
|
|
111
|
+
* workload. Absent and empty are the same canonical policy, digest and compiled graph; at most 1024,
|
|
112
|
+
* held in a named set behind a constant number of rules. */
|
|
113
|
+
hostGrants?: IManagedNftHostGrant[];
|
|
69
114
|
}
|
|
70
115
|
|
|
71
116
|
/** One inbound uplink publication, compiled inside the same host-transit generation. */
|
|
@@ -93,15 +138,24 @@ export interface IManagedNftHostTransitScopeV2 {
|
|
|
93
138
|
* exact forward and reply admission in this generation, so a failed apply, a target
|
|
94
139
|
* without it, or release removes it with the generation. */
|
|
95
140
|
publishedPorts?: IManagedNftPublishedPortV2[];
|
|
141
|
+
/** Optional host-origin grants leaving through a handoff. Each `sourceAddress` is an exact current
|
|
142
|
+
* address of exactly one handoff link; each `destinationAddress` is protected space that is neither
|
|
143
|
+
* host-local, a platform endpoint nor a leased transit source address. Absent and empty are the
|
|
144
|
+
* same canonical policy; at most 1024, held in a named set behind a constant number of rules. */
|
|
145
|
+
hostGrants?: IManagedNftHostGrant[];
|
|
96
146
|
}
|
|
97
147
|
|
|
98
148
|
/** Host-wide IPv4 destination denial for caller-authenticated private allocation pools.
|
|
99
|
-
*
|
|
149
|
+
* It has no link dependency; exact host grants are its only exceptions. It is not allocation-release or boot-order proof. */
|
|
100
150
|
export interface IManagedNftAllocationPoolGuardScopeV2 {
|
|
101
151
|
kind: 'allocationPoolGuard';
|
|
102
152
|
authorityDigest: string;
|
|
103
153
|
/** Complete current pool list: 1–64 canonical, disjoint RFC1918 prefixes. */
|
|
104
154
|
prefixes: string[];
|
|
155
|
+
/** The only exceptions to the guard: exact host-origin flows whose `destinationAddress` lies in a
|
|
156
|
+
* listed pool, and their replies. Absent and empty are the same canonical policy; at most 1024,
|
|
157
|
+
* held in a named set behind a constant number of rules. */
|
|
158
|
+
hostGrants?: IManagedNftHostGrant[];
|
|
105
159
|
}
|
|
106
160
|
|
|
107
161
|
export interface IManagedNftPolicyV2 {
|