@push.rocks/smartnftables 1.4.0 → 1.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/changelog.md +26 -0
- package/dist_rust/smartnftables_linux_amd64_musl +0 -0
- package/dist_rust/smartnftables_linux_amd64_musl.tsrust-build.json +4 -4
- package/dist_rust/smartnftables_linux_arm64_musl +0 -0
- package/dist_rust/smartnftables_linux_arm64_musl.tsrust-build.json +4 -4
- package/dist_ts/00_commitinfo_data.js +1 -1
- package/dist_ts/classes.manageddockerforwarding.d.ts +17 -0
- package/dist_ts/classes.manageddockerforwarding.js +128 -0
- package/dist_ts/classes.managednftables.d.ts +9 -7
- package/dist_ts/classes.managednftables.js +2 -2
- package/dist_ts/index.d.ts +3 -0
- package/dist_ts/index.js +3 -1
- package/dist_ts/managed.docker.types.d.ts +78 -0
- package/dist_ts/managed.docker.types.js +2 -0
- package/dist_ts/managed.egress.types.d.ts +91 -0
- package/dist_ts/managed.egress.types.js +2 -0
- package/dist_ts/managed.types.d.ts +20 -18
- package/package.json +2 -2
- package/readme.md +143 -6
- package/rust/src/docker.frontend.rs +203 -0
- package/rust/src/docker.graph.rs +129 -0
- package/rust/src/docker.policy.rs +140 -0
- package/rust/src/docker.rs +191 -0
- package/rust/src/docker_process_tests.rs +29 -0
- package/rust/src/docker_tests.rs +148 -0
- package/rust/src/egress.compile.rs +288 -0
- package/rust/src/egress.host.rs +103 -0
- package/rust/src/egress.router.rs +231 -0
- package/rust/src/egress.rs +682 -0
- package/rust/src/egress_tests.rs +332 -0
- package/rust/src/main.rs +13 -6
- package/rust/src/managed.rs +105 -0
- package/rust/src/owner.rs +115 -7
- package/rust/src/owner_coexistence_tests.rs +77 -0
- package/rust/src/owner_egress_identity_tests.rs +155 -0
- package/rust/src/owner_egress_tests.rs +153 -0
- package/rust/src/owner_egress_traffic_tests.rs +470 -0
- package/rust/src/owner_host_traffic_tests.rs +305 -0
- package/rust/src/owner_identity_tests.rs +169 -0
- package/rust/src/owner_link_tests.rs +119 -0
- package/rust/src/owner_packet_fixture.rs +200 -0
- package/rust/src/owner_tests.rs +33 -7
- package/rust/src/policy.rs +120 -81
- package/rust/src/tests.rs +22 -0
- package/rust/src/wire.links.rs +279 -0
- package/rust/src/wire.rs +16 -66
- package/ts/00_commitinfo_data.ts +1 -1
- package/ts/classes.manageddockerforwarding.ts +113 -0
- package/ts/classes.managednftables.ts +13 -13
- package/ts/index.ts +3 -0
- package/ts/managed.docker.types.ts +49 -0
- package/ts/managed.egress.types.ts +85 -0
- package/ts/managed.types.ts +20 -16
package/package.json
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@push.rocks/smartnftables",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.6.0",
|
|
4
4
|
"private": false,
|
|
5
5
|
"description": "A TypeScript module for managing nftables rules including NAT, firewall, and rate limiting with a high-level API.",
|
|
6
6
|
"main": "dist_ts/index.js",
|
|
7
7
|
"typings": "dist_ts/index.d.ts",
|
|
8
8
|
"type": "module",
|
|
9
|
-
"author": "
|
|
9
|
+
"author": "Task Venture Capital GmbH",
|
|
10
10
|
"license": "MIT",
|
|
11
11
|
"devDependencies": {
|
|
12
12
|
"@git.zone/cli": "6.13.2",
|
package/readme.md
CHANGED
|
@@ -43,7 +43,7 @@ namespace rejects. Await complete policy and process cleanup before releasing
|
|
|
43
43
|
the caller's final namespace owner. Namespace entry does not create uplink
|
|
44
44
|
authority, drain conntrack flows or authorize address and port reuse.
|
|
45
45
|
|
|
46
|
-
The Linux x86_64 musl candidate
|
|
46
|
+
The Linux x86_64 musl candidate passes isolated Linux 6.18.35 kernel tests,
|
|
47
47
|
including inherited namespace entry, retained policy after process loss, recovery
|
|
48
48
|
in the original namespace, rejection in another namespace, and denied entry
|
|
49
49
|
before readiness. ARM binaries are built; privileged ARM and complete Pallet
|
|
@@ -115,7 +115,7 @@ grants remain outstanding.
|
|
|
115
115
|
| `release(applied)` | Verifies the exact retained graph and confirms deletion; repeats are idempotent. |
|
|
116
116
|
| `close()` | Stops command admission, joins admitted policy operations, confirms owned-table deletion, then joins the native process. Failure retains the owner for explicit recovery. |
|
|
117
117
|
|
|
118
|
-
|
|
118
|
+
Schema-v1 policies are directed. Return traffic requires its own grant; there is no broad
|
|
119
119
|
connection-tracking bypass. A null endpoint denotes the actual local host, with an
|
|
120
120
|
explicit prefix that cannot overlap any endpoint. Such a grant applies only to
|
|
121
121
|
input/output. Forwarding requires two explicit endpoint references; a relay TUN
|
|
@@ -142,7 +142,10 @@ uses the same caller-retained owner, instance, table and boot/namespace receipt.
|
|
|
142
142
|
Recovery checks the complete graph before and after orphan adoption. The kernel's
|
|
143
143
|
wrapping generation counter only fences concurrent transactions; it is never a
|
|
144
144
|
durable revision or sufficient ownership proof. Foreign owners, unexpected tables,
|
|
145
|
-
chains, rules, sets or objects reject without deletion.
|
|
145
|
+
chains, rules, sets or objects inside the owned table reject without deletion.
|
|
146
|
+
Other tables coexist independently. Linux's family-wide chain dump is explicitly
|
|
147
|
+
filtered by table identity before graph comparison; foreign chains are never
|
|
148
|
+
included in replacement or cleanup.
|
|
146
149
|
|
|
147
150
|
PERSIST deliberately retains accepted static grants if the native process crashes.
|
|
148
151
|
It supplies no wall-clock lease or timed revocation. The caller must keep revocation
|
|
@@ -159,15 +162,149 @@ workload before treating revocation as complete or reusing its addresses and
|
|
|
159
162
|
interfaces. Table deletion does not release IP allocation authority or clean up
|
|
160
163
|
conntrack/NAT state. This restriction also applies to private veth/TUN forwarding.
|
|
161
164
|
|
|
165
|
+
### Combined router egress and host transit
|
|
166
|
+
|
|
167
|
+
`ManagedNftables<IManagedNftPolicyV2>` accepts the exported schema-v2 policy.
|
|
168
|
+
The prepared/applied/transition/status interfaces accept the same policy type
|
|
169
|
+
parameter; existing callers default to the unchanged schema-v1 contract. V1
|
|
170
|
+
canonical digests and compiled bytes are preserved. V2 uses its own hash domain.
|
|
171
|
+
An owner cannot transition between v1 private, v2 router, and v2 host policy kinds.
|
|
172
|
+
|
|
173
|
+
| V2 scope | Required authority and behavior |
|
|
174
|
+
| --- | --- |
|
|
175
|
+
| `routerEgress` | Private `endpoints` and `rules`, one exact `links` binding per endpoint, a separate veth `handoff`, `protection`, and active `generations`. Private veth/TUN/local DNS and egress share one table so terminal private denial cannot override a separate egress table. |
|
|
176
|
+
| `hostTransit` | Exact handoff `link`/`allocations` pairs, complete `protection`, an explicit veth or Ethernet `uplink`, and its current `snatAddress`. It checks each handoff's leased source address and protocol/port range, default conntrack zone, direction, uplink, and protected destinations before outer SNAT. |
|
|
177
|
+
|
|
178
|
+
Local bindings include name, index, kind, MAC (null for L3 TUN), interface-link
|
|
179
|
+
index, and required IPv4 addresses. RTM_GETLINK/RTM_GETADDR verify those local
|
|
180
|
+
facts at apply, recovery and inspection. Ethernet must be unbridged driver-backed
|
|
181
|
+
Ethernet, including VirtIO; virtual VLAN/bond/bridge/dummy kinds are not inferred
|
|
182
|
+
uplinks. The caller retains actual peer namespaces and link-generation ownership.
|
|
183
|
+
These serialized facts are not native lifetime capabilities or a continuous
|
|
184
|
+
link-change monitor. Address, DHCP and route changes require caller fencing.
|
|
185
|
+
|
|
186
|
+
`protection` carries an authority digest, non-overlapping protected IPv4 prefixes
|
|
187
|
+
and exact platform endpoint IDs/address/protocol/port tuples. The caller must
|
|
188
|
+
authenticate and supply complete authority. Hashing does not prove completeness.
|
|
189
|
+
A public grant means its declared IPv4 prefix excluding the protected union;
|
|
190
|
+
only an explicit platform endpoint grant admits a protected destination. Host
|
|
191
|
+
checks cover both the current packet destination and original conntrack destination,
|
|
192
|
+
so foreign DNAT cannot turn a protected destination into a public exception or
|
|
193
|
+
redirect public traffic into protected space. INPUT diversion, local OUTPUT into
|
|
194
|
+
handoffs, unmatched handoff traffic and IPv6 forwarding are denied.
|
|
195
|
+
|
|
196
|
+
Each router generation binds an immutable lease reference, transit source address,
|
|
197
|
+
leased TCP/UDP source-port ranges, a nonzero conntrack zone and a 16-byte label.
|
|
198
|
+
Each directed grant selects one exact leased protocol range for SNAT. A grant's
|
|
199
|
+
source is a workload veth or the actual router-local host with an exact assigned
|
|
200
|
+
IPv4 source address; TUN endpoints retain private routing only. Router-local DNS
|
|
201
|
+
and relay transports therefore need explicit local-origin grants. Raw PREROUTING
|
|
202
|
+
and OUTPUT classify before conntrack; filter rules validate current and original
|
|
203
|
+
tuples, links, direction, zone, state and generation label on every packet.
|
|
204
|
+
Opening TCP requires SYN with FIN/RST/ACK clear and permits ECN negotiation.
|
|
205
|
+
Replies must match the labelled original flow and current directed grant.
|
|
206
|
+
There is no broad ESTABLISHED/RELATED bypass.
|
|
207
|
+
|
|
208
|
+
Active overlapping classifiers, duplicate zones/labels and conflicting handoff
|
|
209
|
+
allocations reject. Empty router generations retain private routing while denying
|
|
210
|
+
egress. Input bounds include 32 private endpoints, 128 private rules, 32 active
|
|
211
|
+
generations, 128 total egress grants, 128 protected prefixes, 96 platform endpoints,
|
|
212
|
+
16 ranges per allocation, and 32 host handoffs/active allocations. The complete
|
|
213
|
+
compiled graph still must fit 768 operations and 100,000 bytes; cross-products can
|
|
214
|
+
reach that limit before individual input limits. Exhausted source-port ranges
|
|
215
|
+
drop new flows rather than allocate outside the lease.
|
|
216
|
+
|
|
217
|
+
The caller must coordinate router and host apply order, retain complete intents
|
|
218
|
+
and receipts, authenticate protected authority, prevent reuse of quarantined
|
|
219
|
+
leases, and qualify other packet owners. An ACCEPT in this table cannot override
|
|
220
|
+
Docker's independent FORWARD DROP. `ManagedDockerForwarding`, described below,
|
|
221
|
+
owns the separate DOCKER-USER contribution. The caller must also control deferred packets, proxy/BPF/
|
|
222
|
+
offload paths and changing network authority. No table receipt proves flow
|
|
223
|
+
drainage, conntrack cleanup, elapsed-time expiry, or safe address/port/zone reuse.
|
|
224
|
+
|
|
225
|
+
### Docker forwarding contribution
|
|
226
|
+
|
|
227
|
+
`ManagedDockerForwarding` admits the leased handoff traffic from an exact applied
|
|
228
|
+
schema-v2 `hostTransit` receipt through Docker's existing IPv4 forwarding path.
|
|
229
|
+
It reads and verifies that dedicated barrier's complete table and local links,
|
|
230
|
+
without adopting its ownership. The contribution uses Docker's supported
|
|
231
|
+
[DOCKER-USER extension point](https://docs.docker.com/engine/network/firewall-iptables/).
|
|
232
|
+
Docker's native nftables backend has no equivalent user chain and is unsupported.
|
|
233
|
+
|
|
234
|
+
The host must provide root-owned `/usr/sbin/iptables-nft`,
|
|
235
|
+
`/usr/sbin/iptables-nft-save` and `/usr/sbin/iptables-nft-restore`, using the same
|
|
236
|
+
1.8.10-or-newer 1.8-series `nf_tables` frontend. `prepare()` captures the exact
|
|
237
|
+
frontend version in its digest. Apply, inspection and recovery require that same
|
|
238
|
+
version. Subprocesses use fixed argv, a clean environment, bounded input/output,
|
|
239
|
+
one shared eight-second frontend deadline per request, and joined termination.
|
|
240
|
+
No shell or raw rule API is exposed.
|
|
241
|
+
|
|
242
|
+
One node-level contribution owns a contiguous leading block in DOCKER-USER.
|
|
243
|
+
FORWARD must already have policy DROP and its first two unconditional jumps must
|
|
244
|
+
be DOCKER-USER and DOCKER-FORWARD. Their exact ordered raw graphs are checked
|
|
245
|
+
without claiming their ownership. Other managed contribution owners, altered or
|
|
246
|
+
duplicated owned rules, missing chains and changed placement reject. Unrelated
|
|
247
|
+
rules following the owned block and Docker's own chains remain separate owners.
|
|
248
|
+
Saved text verifies placement, count and normalized policy. Native netlink reads
|
|
249
|
+
verify every contributed rule's complete ordered expression graph, including
|
|
250
|
+
register flow and all match bytes; only counter values and attribute encoding
|
|
251
|
+
order/flags are normalized. An exact xtables `-C` check also verifies each compiled
|
|
252
|
+
deletion specification. These reads must share an unchanged ruleset generation;
|
|
253
|
+
concurrent Docker reconciliation can therefore make inspection inconclusive.
|
|
254
|
+
Conntrack source addresses use the canonical bare IPv4 form because
|
|
255
|
+
an explicit `/32` has different hidden match bytes in these frontends. A
|
|
256
|
+
semantically equal rewrite with a different exact representation rejects, as does
|
|
257
|
+
an early ACCEPT that the frontend reconstructs as the same command.
|
|
258
|
+
The contribution never creates, flushes, adopts or deletes a Docker table/chain,
|
|
259
|
+
and never changes a host forwarding policy.
|
|
260
|
+
|
|
261
|
+
Each contributed rule binds the handoff and uplink names, current and original
|
|
262
|
+
transit source, exact leased TCP/UDP port range, packet direction and connection
|
|
263
|
+
state. New TCP flows require SYN with FIN/RST/ACK clear; reverse traffic requires
|
|
264
|
+
ESTABLISHED. The exact v2 host barrier independently checks interface indices,
|
|
265
|
+
MAC/iflink facts, the default host conntrack zone, protected original/current
|
|
266
|
+
destinations and explicit outer SNAT. A complete contribution is bounded to 192
|
|
267
|
+
rules and 100,000 generated command bytes.
|
|
268
|
+
|
|
269
|
+
Create the class with `ownerId`, a caller-retained `instanceId` of at least 16
|
|
270
|
+
identifier characters, and optional `binaryPath`/`networkNamespaceFd`. After
|
|
271
|
+
`start()`, call `prepare({ schemaVersion: 1, revision, barrier })`, where `barrier`
|
|
272
|
+
is the exact applied host policy. Persist the complete `{ previous, target }`
|
|
273
|
+
transition before `reconcile()`. Keep every affected handoff link down through
|
|
274
|
+
mutations; retain those exact links until cleanup finishes. The native owner
|
|
275
|
+
verifies this down state before and after its single no-flush restore transaction.
|
|
276
|
+
Exact target replay can recover a lost response without rewriting live rules.
|
|
277
|
+
|
|
278
|
+
`inspect().present` confirms the contribution and its referenced barrier, not
|
|
279
|
+
the complete host packet path or durable controller authority. Errors retain
|
|
280
|
+
failed-owned state and pending intent. `release(applied)` and `close()` delete
|
|
281
|
+
only exact contributed rules under the same link fence. A cold boot needs a new
|
|
282
|
+
barrier receipt and caller-authorized replay; old boot receipts cannot authorize
|
|
283
|
+
native operations. The caller owns restart ordering, disjoint retained leases,
|
|
284
|
+
exclusive privileged mutation authority and activation. The frontend transaction
|
|
285
|
+
does not provide compare-and-swap against other privileged processes.
|
|
286
|
+
|
|
162
287
|
### Native qualification
|
|
163
288
|
|
|
164
|
-
|
|
165
|
-
|
|
289
|
+
The separate [Docker fixture](test/native/readme.docker.md) exercises the native
|
|
290
|
+
contribution with pinned Docker packages on offline Ubuntu 26.04 and Ubuntu
|
|
291
|
+
24.04.4/HWE guests, including retained-state cold boots and published workloads.
|
|
292
|
+
Its checked-in input manifest and per-run receipts identify the tested artifacts.
|
|
293
|
+
|
|
294
|
+
`cargo test --manifest-path rust/Cargo.toml --locked` runs unprivileged compiler,
|
|
295
|
+
snapshot and bounded-subprocess tests. The ignored native cases require the explicitly marked disposable guest
|
|
166
296
|
from `test/native/qualify.py`; requesting them on an ordinary host fails its scope
|
|
167
297
|
check. Qualification uses an offline Linux 6.18.35 x86_64 guest with no host disks,
|
|
168
|
-
mounts or network
|
|
298
|
+
mounts or external network backend. One VirtIO device connects only to a singleton
|
|
299
|
+
QEMU-internal hub; packet-path tests use guest-owned namespaces, veth pairs and TUN.
|
|
300
|
+
It covers
|
|
169
301
|
UDP/TCP grants, direct-IP denial, source spoofing, IPv6 denial, renamed interfaces,
|
|
170
302
|
foreign ownership, atomic rollback, generation conflicts and lost-ACK recovery.
|
|
303
|
+
V2 tests exercise complete graph persistence/recovery, private and local DNS,
|
|
304
|
+
local TCP, TUN isolation, observed UDP/TCP handoff ranges, range exhaustion,
|
|
305
|
+
fragment reassembly, ECN SYN and invalid opening flags, protected current/original
|
|
306
|
+
DNAT barriers, nonzero foreign zones, local diversion, IPv6 denial, revocation
|
|
307
|
+
and restoration with disjoint generation ranges while old flows remain retained.
|
|
171
308
|
The arm64 binary is cross-built; native packet qualification is currently x86_64.
|
|
172
309
|
Kernel 6.8 is unsupported. No production activation is implied by these tests.
|
|
173
310
|
|
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
use super::policy::{Options, Rule};
|
|
2
|
+
use crate::{Error, Result};
|
|
3
|
+
use std::collections::{BTreeMap, BTreeSet};
|
|
4
|
+
use std::io::{Read, Write};
|
|
5
|
+
use std::os::fd::AsRawFd;
|
|
6
|
+
use std::os::unix::{fs::MetadataExt, process::CommandExt};
|
|
7
|
+
use std::process::{Command, Stdio};
|
|
8
|
+
use std::time::{Duration, Instant};
|
|
9
|
+
|
|
10
|
+
const IPTABLES: &str = "/usr/sbin/iptables-nft";
|
|
11
|
+
const SAVE: &str = "/usr/sbin/iptables-nft-save";
|
|
12
|
+
const RESTORE: &str = "/usr/sbin/iptables-nft-restore";
|
|
13
|
+
const MAX_OUTPUT: usize = 2_097_152;
|
|
14
|
+
#[cfg(test)]
|
|
15
|
+
#[path="docker_process_tests.rs"]
|
|
16
|
+
mod process_tests;
|
|
17
|
+
|
|
18
|
+
fn nonblocking(fd: i32) -> Result<()> {
|
|
19
|
+
let flags = unsafe { libc::fcntl(fd, libc::F_GETFL) };
|
|
20
|
+
if flags < 0 || unsafe { libc::fcntl(fd,libc::F_SETFL,flags|libc::O_NONBLOCK) } < 0 {return Err(Error::Unavailable);}
|
|
21
|
+
Ok(())
|
|
22
|
+
}
|
|
23
|
+
pub fn deadline() -> Instant { Instant::now()+Duration::from_secs(8) }
|
|
24
|
+
fn command(path: &str, args: &[&str], input: &[u8], deadline: Instant) -> Result<String> {
|
|
25
|
+
if Instant::now()>=deadline {return Err(Error::Unavailable);}
|
|
26
|
+
let metadata=std::fs::metadata(path).map_err(|_|Error::Unavailable)?;
|
|
27
|
+
if !metadata.is_file() || metadata.uid()!=0 || metadata.mode() & 0o022 != 0 { return Err(Error::Permission); }
|
|
28
|
+
let mut child=Command::new(path).args(args).env_clear().env("PATH","/usr/sbin:/usr/bin:/sbin:/bin")
|
|
29
|
+
.env("LC_ALL","C").stdin(Stdio::piped()).stdout(Stdio::piped()).stderr(Stdio::piped()).process_group(0)
|
|
30
|
+
.spawn().map_err(|_|Error::Unavailable)?;
|
|
31
|
+
let result=(|| {
|
|
32
|
+
let mut stdin=child.stdin.take();
|
|
33
|
+
let mut stdout=child.stdout.take().ok_or(Error::Unavailable)?;
|
|
34
|
+
let mut stderr=child.stderr.take().ok_or(Error::Unavailable)?;
|
|
35
|
+
for fd in [stdin.as_ref().ok_or(Error::Unavailable)?.as_raw_fd(),stdout.as_raw_fd(),stderr.as_raw_fd()] {nonblocking(fd)?;}
|
|
36
|
+
let mut written=0;let mut output=Vec::new();let mut errors=Vec::new();
|
|
37
|
+
let mut out_eof=false;let mut err_eof=false;let mut status=None;
|
|
38
|
+
loop {
|
|
39
|
+
if Instant::now()>=deadline {return Err(Error::Unavailable);}
|
|
40
|
+
if written==input.len() {stdin=None;}
|
|
41
|
+
if let Some(pipe)=stdin.as_mut() {
|
|
42
|
+
match pipe.write(&input[written..]) {
|
|
43
|
+
Ok(0)=>return Err(Error::Unavailable),Ok(count)=>written+=count,
|
|
44
|
+
Err(error) if error.kind()==std::io::ErrorKind::WouldBlock=>{},
|
|
45
|
+
Err(_)=>return Err(Error::Unavailable),
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
for (pipe,bytes,eof) in [(&mut stdout as &mut dyn Read,&mut output,&mut out_eof),(&mut stderr as &mut dyn Read,&mut errors,&mut err_eof)] {
|
|
49
|
+
let mut buffer=[0;8192];
|
|
50
|
+
loop {
|
|
51
|
+
match pipe.read(&mut buffer) {
|
|
52
|
+
Ok(0)=>{*eof=true;break;},
|
|
53
|
+
Ok(count)=>{if bytes.len()+count>MAX_OUTPUT{return Err(Error::Protocol);}bytes.extend_from_slice(&buffer[..count]);},
|
|
54
|
+
Err(error) if error.kind()==std::io::ErrorKind::WouldBlock=>break,
|
|
55
|
+
Err(_)=>return Err(Error::Unavailable),
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
if status.is_none() {status=child.try_wait().map_err(|_|Error::Unavailable)?;}
|
|
60
|
+
if let Some(status)=status.filter(|_|out_eof&&err_eof) {
|
|
61
|
+
if !status.success() || written!=input.len() {
|
|
62
|
+
if path==IPTABLES && args.get(2)==Some(&"-C") && status.code()==Some(1)
|
|
63
|
+
&& errors==b"iptables: Bad rule (does a matching rule exist in that chain?).\n" {
|
|
64
|
+
return Err(Error::Conflict);
|
|
65
|
+
}
|
|
66
|
+
return Err(Error::Unavailable);
|
|
67
|
+
}
|
|
68
|
+
return String::from_utf8(output).map_err(|_|Error::Protocol);
|
|
69
|
+
}
|
|
70
|
+
std::thread::sleep(Duration::from_millis(2));
|
|
71
|
+
}
|
|
72
|
+
})();
|
|
73
|
+
if result.is_err() {
|
|
74
|
+
// Stop and join this exact mutator before reporting an inconclusive ACK.
|
|
75
|
+
if child.try_wait().ok().flatten().is_none() {
|
|
76
|
+
unsafe {libc::kill(-(child.id() as i32),libc::SIGKILL);}
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
child.wait().map_err(|_|Error::Unavailable)?;
|
|
80
|
+
result
|
|
81
|
+
}
|
|
82
|
+
pub fn version(deadline: Instant) -> Result<String> {
|
|
83
|
+
let mut version=None;
|
|
84
|
+
for path in [IPTABLES,SAVE,RESTORE] {
|
|
85
|
+
let output=command(path,&["--version"],&[],deadline)?;
|
|
86
|
+
let parts:Vec<_>=output.split_whitespace().collect();
|
|
87
|
+
if parts.len()!=3 {return Err(Error::UnsupportedBackend);}
|
|
88
|
+
let observed=format!("{} {}",parts[1],parts[2]);
|
|
89
|
+
super::policy::validate_backend(&observed)?;
|
|
90
|
+
if version.as_ref().is_some_and(|v|v!=&observed) {return Err(Error::Conflict);}
|
|
91
|
+
version=Some(observed);
|
|
92
|
+
}
|
|
93
|
+
version.ok_or(Error::Unavailable)
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
/// Tokenize save output only; no token is ever evaluated by a shell.
|
|
97
|
+
fn tokens(line: &str) -> Result<Vec<String>> {
|
|
98
|
+
if line.len()>16_384 {return Err(Error::Protocol);}
|
|
99
|
+
let mut values=Vec::new();let mut word=String::new();let mut quoted=false;let mut escaped=false;let mut active=false;
|
|
100
|
+
for byte in line.bytes() {
|
|
101
|
+
if !(32..127).contains(&byte) {return Err(Error::Protocol);}
|
|
102
|
+
if escaped {word.push(byte as char);escaped=false;active=true;continue;}
|
|
103
|
+
if byte==b'\\' {escaped=true;continue;}
|
|
104
|
+
if byte==b'"' {quoted=!quoted;active=true;continue;}
|
|
105
|
+
if byte==b' ' && !quoted {if active {values.push(std::mem::take(&mut word));active=false;}} else {word.push(byte as char);active=true;}
|
|
106
|
+
}
|
|
107
|
+
if quoted||escaped {return Err(Error::Protocol);}
|
|
108
|
+
if active {values.push(word);}
|
|
109
|
+
if values.len()>512 {return Err(Error::Protocol);}
|
|
110
|
+
Ok(values)
|
|
111
|
+
}
|
|
112
|
+
fn normalized(args: &[String]) -> Result<Vec<(String, Vec<String>)>> {
|
|
113
|
+
let mut values=BTreeMap::new();let mut modules=BTreeSet::new();let mut index=0;
|
|
114
|
+
while index<args.len() {
|
|
115
|
+
let key=&args[index];let count=if key=="--tcp-flags" {2} else {1};
|
|
116
|
+
if !["-i","-o","-s","-d","-p","-m","--sport","--dport","--tcp-flags","--ctstate","--ctproto","--ctorigsrc","--ctorigsrcport","--ctdir","--comment","-j"].contains(&key.as_str()) {return Err(Error::Conflict);}
|
|
117
|
+
let mut value=args.get(index+1..index+1+count).ok_or(Error::Conflict)?.to_vec();index+=1+count;
|
|
118
|
+
if key=="-m" {if !modules.insert(value[0].clone()) {return Err(Error::Conflict);}continue;}
|
|
119
|
+
if key=="--ctorigsrc" && !value[0].contains('/') {value[0].push_str("/32");}
|
|
120
|
+
if key=="--ctproto" {value[0]=match value[0].as_str() {"tcp"=>"6".into(),"udp"=>"17".into(),_=>value[0].clone()};}
|
|
121
|
+
if values.insert(key.clone(),value).is_some() {return Err(Error::Conflict);}
|
|
122
|
+
}
|
|
123
|
+
values.insert("-m".into(),modules.into_iter().collect());
|
|
124
|
+
Ok(values.into_iter().collect())
|
|
125
|
+
}
|
|
126
|
+
pub struct Snapshot {
|
|
127
|
+
pub backend: String,
|
|
128
|
+
pub rows: Vec<Vec<String>>,
|
|
129
|
+
pub user: Vec<Vec<String>>,
|
|
130
|
+
raw: Option<super::graph::Snapshot>,
|
|
131
|
+
}
|
|
132
|
+
impl Snapshot {
|
|
133
|
+
pub fn read(deadline: Instant) -> Result<Self> {
|
|
134
|
+
let backend=version(deadline)?;
|
|
135
|
+
let (raw,text)=super::graph::Snapshot::read(||command(SAVE,&["-t","filter"],&[],deadline))?;
|
|
136
|
+
let mut snapshot=Self::parse(backend,&text)?;
|
|
137
|
+
snapshot.raw=Some(raw);
|
|
138
|
+
Ok(snapshot)
|
|
139
|
+
}
|
|
140
|
+
pub fn parse(backend: String, text: &str) -> Result<Self> {
|
|
141
|
+
if text.len()>MAX_OUTPUT {return Err(Error::Protocol);}
|
|
142
|
+
let mut table=false;let mut committed=false;let mut chains=BTreeMap::new();let mut rows=Vec::new();
|
|
143
|
+
for line in text.lines().filter(|line|!line.starts_with('#')&&!line.is_empty()) {
|
|
144
|
+
if line=="*filter" {if table||committed {return Err(Error::Conflict);}table=true;continue;}
|
|
145
|
+
if line=="COMMIT" {if !table||committed {return Err(Error::Conflict);}committed=true;continue;}
|
|
146
|
+
if !table||committed {return Err(Error::Conflict);}
|
|
147
|
+
let row=tokens(line)?;
|
|
148
|
+
if line.starts_with(':') {
|
|
149
|
+
if row.len()!=3 || chains.insert(row[0][1..].to_string(),row[1].clone()).is_some() {return Err(Error::Conflict);}
|
|
150
|
+
} else {
|
|
151
|
+
if row.len()<2 || row[0]!="-A" || rows.len()>=4096 {return Err(Error::Conflict);}
|
|
152
|
+
rows.push(row);
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
if !committed || chains.get("FORWARD").map(String::as_str)!=Some("DROP")
|
|
156
|
+
|| chains.get("DOCKER-USER").map(String::as_str)!=Some("-")
|
|
157
|
+
|| chains.get("DOCKER-FORWARD").map(String::as_str)!=Some("-") {return Err(Error::Conflict);}
|
|
158
|
+
let forward:Vec<_>=rows.iter().filter(|r|r[1]=="FORWARD").collect();
|
|
159
|
+
for (index,target) in ["DOCKER-USER","DOCKER-FORWARD"].iter().enumerate() {
|
|
160
|
+
if forward.get(index).map(|r|r.iter().map(String::as_str).collect::<Vec<_>>())!=Some(vec!["-A","FORWARD","-j",target]) {return Err(Error::Conflict);}
|
|
161
|
+
}
|
|
162
|
+
let user=rows.iter().filter(|r|r[1]=="DOCKER-USER").cloned().collect();
|
|
163
|
+
Ok(Self {backend,rows,user,raw:None})
|
|
164
|
+
}
|
|
165
|
+
/// Text proves placement/count and logical policy; -C additionally checks
|
|
166
|
+
/// match-extension bytes hidden by save output. Deletion uses the same spec.
|
|
167
|
+
pub fn verified_matches(&self, options: &Options, expected: &[Rule], deadline: Instant) -> Result<bool> {
|
|
168
|
+
if !self.matches(options,expected)? {return Ok(false);}
|
|
169
|
+
let raw=self.raw.as_ref().ok_or(Error::Conflict)?;
|
|
170
|
+
if !raw.matches(expected)? {return Ok(false);}
|
|
171
|
+
for rule in expected {
|
|
172
|
+
let mut args=vec!["-t","filter","-C","DOCKER-USER"];
|
|
173
|
+
args.extend(rule.args.iter().map(String::as_str));
|
|
174
|
+
match command(IPTABLES,&args,&[],deadline) {
|
|
175
|
+
Ok(_)=>{},Err(Error::Conflict)=>return Ok(false),Err(error)=>return Err(error),
|
|
176
|
+
}
|
|
177
|
+
}
|
|
178
|
+
raw.unchanged()
|
|
179
|
+
}
|
|
180
|
+
pub fn matches(&self, options: &Options, expected: &[Rule]) -> Result<bool> {
|
|
181
|
+
let prefix=format!("snftd1:{}:",options.owner_id);
|
|
182
|
+
// One node-level contribution owns the leading block. Do not prepend
|
|
183
|
+
// another managed owner and silently invalidate its placement proof.
|
|
184
|
+
if self.rows.iter().any(|r|r.iter().any(|t|t.starts_with("snftd1:")&&!t.starts_with(&prefix))) {return Err(Error::ForeignOwner);}
|
|
185
|
+
let owned:Vec<_>=self.rows.iter().filter(|r|r.iter().any(|t|t.starts_with(&prefix))).collect();
|
|
186
|
+
if owned.len()!=expected.len() || self.user.len()<expected.len() {return Ok(false);}
|
|
187
|
+
for (index,rule) in expected.iter().enumerate() {
|
|
188
|
+
let row=&self.user[index];
|
|
189
|
+
if row[1]!="DOCKER-USER" || !row.iter().any(|t|t==&rule.comment)
|
|
190
|
+
|| normalized(&row[2..])?!=normalized(&rule.args)? {return Ok(false);}
|
|
191
|
+
}
|
|
192
|
+
Ok(true)
|
|
193
|
+
}
|
|
194
|
+
}
|
|
195
|
+
pub fn replace(previous: &[Rule], target: &[Rule], deadline: Instant) -> Result<()> {
|
|
196
|
+
let mut batch="*filter\n".to_string();
|
|
197
|
+
for rule in previous {batch.push_str(&format!("-D DOCKER-USER {}\n",rule.args.join(" ")));}
|
|
198
|
+
for rule in target.iter().rev() {batch.push_str(&format!("-I DOCKER-USER 1 {}\n",rule.args.join(" ")));}
|
|
199
|
+
batch.push_str("COMMIT\n");
|
|
200
|
+
if batch.len()>210_000 {return Err(Error::Invalid);}
|
|
201
|
+
command(RESTORE,&["--noflush","--wait","5"],batch.as_bytes(),deadline)?;
|
|
202
|
+
Ok(())
|
|
203
|
+
}
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
//! Ordered iptables-nft graph identity. Command-state reconstruction loses order.
|
|
2
|
+
use super::policy::Rule;
|
|
3
|
+
use crate::{policy::{compare, expr, meta, verdict}, wire::{self, Attr, Socket}, Error, Result};
|
|
4
|
+
|
|
5
|
+
fn value<'a>(rule: &'a Rule, key: &str) -> Result<&'a str> {
|
|
6
|
+
rule.args.windows(2).find(|pair|pair[0]==key).map(|pair|pair[1].as_str()).ok_or(Error::Invalid)
|
|
7
|
+
}
|
|
8
|
+
fn payload(base: u32, offset: u32, length: u32) -> Attr {
|
|
9
|
+
expr("payload",vec![Attr::u32(1,1),Attr::u32(2,base),Attr::u32(3,offset),Attr::u32(4,length)])
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
/// Fixed Linux UAPI / xtables 1.8.10 and 1.8.11 constructors, not a raw rule API.
|
|
13
|
+
pub(super) fn expected(rule: &Rule) -> Result<Vec<Attr>> {
|
|
14
|
+
let reply=value(rule,"--ctdir")?=="REPLY";
|
|
15
|
+
let new=value(rule,"--ctstate")?=="NEW";
|
|
16
|
+
let protocol=match value(rule,"-p")? {"tcp"=>6_u16,"udp"=>17,_=>return Err(Error::Invalid)};
|
|
17
|
+
let address=value(rule,"--ctorigsrc")?.parse::<std::net::Ipv4Addr>().map_err(|_|Error::Invalid)?.octets();
|
|
18
|
+
let ports=value(rule,"--ctorigsrcport")?;
|
|
19
|
+
let (first,last)=ports.split_once(':').unwrap_or((ports,ports));
|
|
20
|
+
let first=first.parse::<u16>().map_err(|_|Error::Invalid)?;
|
|
21
|
+
let last=last.parse::<u16>().map_err(|_|Error::Invalid)?;
|
|
22
|
+
let mut result=vec![payload(1,if reply {16} else {12},4),compare(address.to_vec())];
|
|
23
|
+
for (key,argument) in [(6,"-i"),(7,"-o")] {
|
|
24
|
+
result.extend(meta(key,[value(rule,argument)?.as_bytes(),&[0]].concat()));
|
|
25
|
+
}
|
|
26
|
+
result.extend([payload(1,9,1),compare(vec![protocol as u8])]);
|
|
27
|
+
if new && protocol==6 {
|
|
28
|
+
result.extend([payload(2,13,1),expr("bitwise",vec![
|
|
29
|
+
Attr::u32(1,1),Attr::u32(2,1),Attr::u32(3,1),Attr::u32(6,0),
|
|
30
|
+
Attr::nested(4,vec![Attr::bytes(1,vec![0x17])]),Attr::nested(5,vec![Attr::bytes(1,vec![0])]),
|
|
31
|
+
]),compare(vec![2])]);
|
|
32
|
+
}
|
|
33
|
+
result.push(payload(2,if reply {2} else {0},2));
|
|
34
|
+
result.push(if first==last {compare(first.to_be_bytes().to_vec())} else {
|
|
35
|
+
expr("range",vec![Attr::u32(1,1),Attr::u32(2,0),
|
|
36
|
+
Attr::nested(3,vec![Attr::bytes(1,first.to_be_bytes().to_vec())]),
|
|
37
|
+
Attr::nested(4,vec![Attr::bytes(1,last.to_be_bytes().to_vec())])])
|
|
38
|
+
});
|
|
39
|
+
// xt_conntrack_mtinfo3 is 164 bytes, XT_ALIGN'ed to 168 on both shipped
|
|
40
|
+
// 64-bit Linux targets. All fields and padding are initialized explicitly.
|
|
41
|
+
let mut conntrack=vec![0;168];
|
|
42
|
+
conntrack[..4].copy_from_slice(&address);
|
|
43
|
+
conntrack[16..32].fill(0xff); // canonical bare-host nf_inet_addr mask
|
|
44
|
+
for (offset,number) in [(136,protocol),(138,first),(146,0x1107),
|
|
45
|
+
(148,if reply {0x1000} else {0}),(150,if new {8} else {2}),(154,last)] {
|
|
46
|
+
conntrack[offset..offset+2].copy_from_slice(&number.to_ne_bytes());
|
|
47
|
+
}
|
|
48
|
+
result.push(expr("match",vec![Attr::string(1,"conntrack"),Attr::u32(2,3),Attr::bytes(3,conntrack)]));
|
|
49
|
+
let mut comment=vec![0;256];
|
|
50
|
+
if rule.comment.len()>=comment.len() {return Err(Error::Invalid);}
|
|
51
|
+
comment[..rule.comment.len()].copy_from_slice(rule.comment.as_bytes());
|
|
52
|
+
result.push(expr("match",vec![Attr::string(1,"comment"),Attr::u32(2,0),Attr::bytes(3,comment)]));
|
|
53
|
+
result.push(expr("counter",vec![Attr::u64(1,0),Attr::u64(2,0)]));
|
|
54
|
+
result.push(verdict(1,None));
|
|
55
|
+
Ok(result)
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/// Kernel dumps may omit nested flags and reorder fields; list order is exact.
|
|
59
|
+
fn equivalent(actual: &[Attr], expected: &[Attr], ordered: bool) -> Result<bool> {
|
|
60
|
+
if actual.len()!=expected.len() {return Ok(false);}
|
|
61
|
+
let mut actual=actual.iter().collect::<Vec<_>>();let mut expected=expected.iter().collect::<Vec<_>>();
|
|
62
|
+
if !ordered {actual.sort_by_key(|a|a.id());expected.sort_by_key(|a|a.id());}
|
|
63
|
+
for (actual,expected) in actual.into_iter().zip(expected) {
|
|
64
|
+
if actual.id()!=expected.id() {return Ok(false);}
|
|
65
|
+
if expected.kind & 0x8000 != 0 {
|
|
66
|
+
let expected=wire::attrs(&expected.value)?;
|
|
67
|
+
if !equivalent(&wire::attrs(&actual.value)?,&expected,expected.iter().all(|a|a.id()==1))? {return Ok(false);}
|
|
68
|
+
} else if actual.value!=expected.value {return Ok(false);}
|
|
69
|
+
}
|
|
70
|
+
Ok(true)
|
|
71
|
+
}
|
|
72
|
+
pub(super) fn matches_rule(row: &[Attr], rule: &Rule) -> Result<bool> {
|
|
73
|
+
matches_graph(row,"DOCKER-USER",&expected(rule)?)
|
|
74
|
+
}
|
|
75
|
+
pub(super) fn matches_forward(row: &[Attr], target: &str) -> Result<bool> {
|
|
76
|
+
if !["DOCKER-USER","DOCKER-FORWARD"].contains(&target) {return Err(Error::Invalid);}
|
|
77
|
+
matches_graph(row,"FORWARD",&[expr("counter",vec![Attr::u64(1,0),Attr::u64(2,0)]),verdict(-3,Some(target))])
|
|
78
|
+
}
|
|
79
|
+
fn matches_graph(row: &[Attr], chain: &str, expected: &[Attr]) -> Result<bool> {
|
|
80
|
+
if row.iter().any(|a|![1,2,3,4,6,8].contains(&a.id()))
|
|
81
|
+
|| wire::text(row,1)?!="filter" || wire::text(row,2)?!=chain
|
|
82
|
+
|| wire::handle(row,3)?==0 {return Ok(false);}
|
|
83
|
+
for id in [6,8] {
|
|
84
|
+
if row.iter().any(|a|a.id()==id) {
|
|
85
|
+
let attr=wire::one(row,id)?;
|
|
86
|
+
if id==6 {wire::handle(row,id)?;} else if !attr.value.is_empty() {return Ok(false);}
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
let mut expressions=wire::attrs(&wire::one(row,4)?.value)?;
|
|
90
|
+
for expression in &mut expressions {
|
|
91
|
+
let mut fields=wire::attrs(&expression.value)?;
|
|
92
|
+
if wire::text(&fields,1)?=="counter" {
|
|
93
|
+
let mut data=wire::attrs(&wire::one(&fields,2)?.value)?;
|
|
94
|
+
// Counter values change without a ruleset generation change. No
|
|
95
|
+
// expression, including the counter itself, may move or disappear.
|
|
96
|
+
data.retain(|a|!(a.id()==3 && a.value.is_empty()));
|
|
97
|
+
if data.len()!=2 {return Ok(false);}
|
|
98
|
+
for id in [1,2] {wire::handle(&data,id)?;}
|
|
99
|
+
for attribute in &mut data {attribute.value=0_u64.to_be_bytes().to_vec();}
|
|
100
|
+
let field=fields.iter_mut().find(|a|a.id()==2).ok_or(Error::Protocol)?;
|
|
101
|
+
*field=Attr::nested(2,data);
|
|
102
|
+
expression.value=wire::encode_attrs(&fields);
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
equivalent(&expressions,expected,true)
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
pub struct Snapshot { generation: u32, rows: Vec<Vec<Attr>> }
|
|
109
|
+
impl Snapshot {
|
|
110
|
+
pub fn read(save: impl FnOnce()->Result<String>) -> Result<(Self,String)> {
|
|
111
|
+
let mut socket=Socket::open()?;let generation=socket.generation()?;
|
|
112
|
+
let text=save()?;
|
|
113
|
+
let rows=socket.query_ipv4(7,vec![Attr::string(1,"filter"),Attr::string(2,"DOCKER-USER")],true)?;
|
|
114
|
+
let forward=socket.query_ipv4(7,vec![Attr::string(1,"filter"),Attr::string(2,"FORWARD")],true)?;
|
|
115
|
+
if socket.generation()!=Ok(generation) {return Err(Error::Conflict);}
|
|
116
|
+
for (index,target) in ["DOCKER-USER","DOCKER-FORWARD"].iter().enumerate() {
|
|
117
|
+
if !matches_forward(forward.get(index).ok_or(Error::Conflict)?,target)? {return Err(Error::Conflict);}
|
|
118
|
+
}
|
|
119
|
+
Ok((Self {generation,rows},text))
|
|
120
|
+
}
|
|
121
|
+
pub fn matches(&self, expected: &[Rule]) -> Result<bool> {
|
|
122
|
+
if self.rows.len()<expected.len() {return Ok(false);}
|
|
123
|
+
for (row,rule) in self.rows.iter().zip(expected) {
|
|
124
|
+
if !matches_rule(row,rule)? {return Ok(false);}
|
|
125
|
+
}
|
|
126
|
+
Ok(true)
|
|
127
|
+
}
|
|
128
|
+
pub fn unchanged(&self) -> Result<bool> {Ok(Socket::open()?.generation()?==self.generation)}
|
|
129
|
+
}
|