@push.rocks/smartnftables 1.4.0 → 1.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. package/changelog.md +26 -0
  2. package/dist_rust/smartnftables_linux_amd64_musl +0 -0
  3. package/dist_rust/smartnftables_linux_amd64_musl.tsrust-build.json +4 -4
  4. package/dist_rust/smartnftables_linux_arm64_musl +0 -0
  5. package/dist_rust/smartnftables_linux_arm64_musl.tsrust-build.json +4 -4
  6. package/dist_ts/00_commitinfo_data.js +1 -1
  7. package/dist_ts/classes.manageddockerforwarding.d.ts +17 -0
  8. package/dist_ts/classes.manageddockerforwarding.js +128 -0
  9. package/dist_ts/classes.managednftables.d.ts +9 -7
  10. package/dist_ts/classes.managednftables.js +2 -2
  11. package/dist_ts/index.d.ts +3 -0
  12. package/dist_ts/index.js +3 -1
  13. package/dist_ts/managed.docker.types.d.ts +78 -0
  14. package/dist_ts/managed.docker.types.js +2 -0
  15. package/dist_ts/managed.egress.types.d.ts +91 -0
  16. package/dist_ts/managed.egress.types.js +2 -0
  17. package/dist_ts/managed.types.d.ts +20 -18
  18. package/package.json +2 -2
  19. package/readme.md +143 -6
  20. package/rust/src/docker.frontend.rs +203 -0
  21. package/rust/src/docker.graph.rs +129 -0
  22. package/rust/src/docker.policy.rs +140 -0
  23. package/rust/src/docker.rs +191 -0
  24. package/rust/src/docker_process_tests.rs +29 -0
  25. package/rust/src/docker_tests.rs +148 -0
  26. package/rust/src/egress.compile.rs +288 -0
  27. package/rust/src/egress.host.rs +103 -0
  28. package/rust/src/egress.router.rs +231 -0
  29. package/rust/src/egress.rs +682 -0
  30. package/rust/src/egress_tests.rs +332 -0
  31. package/rust/src/main.rs +13 -6
  32. package/rust/src/managed.rs +105 -0
  33. package/rust/src/owner.rs +115 -7
  34. package/rust/src/owner_coexistence_tests.rs +77 -0
  35. package/rust/src/owner_egress_identity_tests.rs +155 -0
  36. package/rust/src/owner_egress_tests.rs +153 -0
  37. package/rust/src/owner_egress_traffic_tests.rs +470 -0
  38. package/rust/src/owner_host_traffic_tests.rs +305 -0
  39. package/rust/src/owner_identity_tests.rs +169 -0
  40. package/rust/src/owner_link_tests.rs +119 -0
  41. package/rust/src/owner_packet_fixture.rs +200 -0
  42. package/rust/src/owner_tests.rs +33 -7
  43. package/rust/src/policy.rs +120 -81
  44. package/rust/src/tests.rs +22 -0
  45. package/rust/src/wire.links.rs +279 -0
  46. package/rust/src/wire.rs +16 -66
  47. package/ts/00_commitinfo_data.ts +1 -1
  48. package/ts/classes.manageddockerforwarding.ts +113 -0
  49. package/ts/classes.managednftables.ts +13 -13
  50. package/ts/index.ts +3 -0
  51. package/ts/managed.docker.types.ts +49 -0
  52. package/ts/managed.egress.types.ts +85 -0
  53. package/ts/managed.types.ts +20 -16
package/package.json CHANGED
@@ -1,12 +1,12 @@
1
1
  {
2
2
  "name": "@push.rocks/smartnftables",
3
- "version": "1.4.0",
3
+ "version": "1.6.0",
4
4
  "private": false,
5
5
  "description": "A TypeScript module for managing nftables rules including NAT, firewall, and rate limiting with a high-level API.",
6
6
  "main": "dist_ts/index.js",
7
7
  "typings": "dist_ts/index.d.ts",
8
8
  "type": "module",
9
- "author": "Lossless GmbH",
9
+ "author": "Task Venture Capital GmbH",
10
10
  "license": "MIT",
11
11
  "devDependencies": {
12
12
  "@git.zone/cli": "6.13.2",
package/readme.md CHANGED
@@ -43,7 +43,7 @@ namespace rejects. Await complete policy and process cleanup before releasing
43
43
  the caller's final namespace owner. Namespace entry does not create uplink
44
44
  authority, drain conntrack flows or authorize address and port reuse.
45
45
 
46
- The Linux x86_64 musl candidate passed nine isolated Linux 6.18.35 kernel tests,
46
+ The Linux x86_64 musl candidate passes isolated Linux 6.18.35 kernel tests,
47
47
  including inherited namespace entry, retained policy after process loss, recovery
48
48
  in the original namespace, rejection in another namespace, and denied entry
49
49
  before readiness. ARM binaries are built; privileged ARM and complete Pallet
@@ -115,7 +115,7 @@ grants remain outstanding.
115
115
  | `release(applied)` | Verifies the exact retained graph and confirms deletion; repeats are idempotent. |
116
116
  | `close()` | Stops command admission, joins admitted policy operations, confirms owned-table deletion, then joins the native process. Failure retains the owner for explicit recovery. |
117
117
 
118
- Policies are directed. Return traffic requires its own grant; there is no broad
118
+ Schema-v1 policies are directed. Return traffic requires its own grant; there is no broad
119
119
  connection-tracking bypass. A null endpoint denotes the actual local host, with an
120
120
  explicit prefix that cannot overlap any endpoint. Such a grant applies only to
121
121
  input/output. Forwarding requires two explicit endpoint references; a relay TUN
@@ -142,7 +142,10 @@ uses the same caller-retained owner, instance, table and boot/namespace receipt.
142
142
  Recovery checks the complete graph before and after orphan adoption. The kernel's
143
143
  wrapping generation counter only fences concurrent transactions; it is never a
144
144
  durable revision or sufficient ownership proof. Foreign owners, unexpected tables,
145
- chains, rules, sets or objects reject without deletion.
145
+ chains, rules, sets or objects inside the owned table reject without deletion.
146
+ Other tables coexist independently. Linux's family-wide chain dump is explicitly
147
+ filtered by table identity before graph comparison; foreign chains are never
148
+ included in replacement or cleanup.
146
149
 
147
150
  PERSIST deliberately retains accepted static grants if the native process crashes.
148
151
  It supplies no wall-clock lease or timed revocation. The caller must keep revocation
@@ -159,15 +162,149 @@ workload before treating revocation as complete or reusing its addresses and
159
162
  interfaces. Table deletion does not release IP allocation authority or clean up
160
163
  conntrack/NAT state. This restriction also applies to private veth/TUN forwarding.
161
164
 
165
+ ### Combined router egress and host transit
166
+
167
+ `ManagedNftables<IManagedNftPolicyV2>` accepts the exported schema-v2 policy.
168
+ The prepared/applied/transition/status interfaces accept the same policy type
169
+ parameter; existing callers default to the unchanged schema-v1 contract. V1
170
+ canonical digests and compiled bytes are preserved. V2 uses its own hash domain.
171
+ An owner cannot transition between v1 private, v2 router, and v2 host policy kinds.
172
+
173
+ | V2 scope | Required authority and behavior |
174
+ | --- | --- |
175
+ | `routerEgress` | Private `endpoints` and `rules`, one exact `links` binding per endpoint, a separate veth `handoff`, `protection`, and active `generations`. Private veth/TUN/local DNS and egress share one table so terminal private denial cannot override a separate egress table. |
176
+ | `hostTransit` | Exact handoff `link`/`allocations` pairs, complete `protection`, an explicit veth or Ethernet `uplink`, and its current `snatAddress`. It checks each handoff's leased source address and protocol/port range, default conntrack zone, direction, uplink, and protected destinations before outer SNAT. |
177
+
178
+ Local bindings include name, index, kind, MAC (null for L3 TUN), interface-link
179
+ index, and required IPv4 addresses. RTM_GETLINK/RTM_GETADDR verify those local
180
+ facts at apply, recovery and inspection. Ethernet must be unbridged driver-backed
181
+ Ethernet, including VirtIO; virtual VLAN/bond/bridge/dummy kinds are not inferred
182
+ uplinks. The caller retains actual peer namespaces and link-generation ownership.
183
+ These serialized facts are not native lifetime capabilities or a continuous
184
+ link-change monitor. Address, DHCP and route changes require caller fencing.
185
+
186
+ `protection` carries an authority digest, non-overlapping protected IPv4 prefixes
187
+ and exact platform endpoint IDs/address/protocol/port tuples. The caller must
188
+ authenticate and supply complete authority. Hashing does not prove completeness.
189
+ A public grant means its declared IPv4 prefix excluding the protected union;
190
+ only an explicit platform endpoint grant admits a protected destination. Host
191
+ checks cover both the current packet destination and original conntrack destination,
192
+ so foreign DNAT cannot turn a protected destination into a public exception or
193
+ redirect public traffic into protected space. INPUT diversion, local OUTPUT into
194
+ handoffs, unmatched handoff traffic and IPv6 forwarding are denied.
195
+
196
+ Each router generation binds an immutable lease reference, transit source address,
197
+ leased TCP/UDP source-port ranges, a nonzero conntrack zone and a 16-byte label.
198
+ Each directed grant selects one exact leased protocol range for SNAT. A grant's
199
+ source is a workload veth or the actual router-local host with an exact assigned
200
+ IPv4 source address; TUN endpoints retain private routing only. Router-local DNS
201
+ and relay transports therefore need explicit local-origin grants. Raw PREROUTING
202
+ and OUTPUT classify before conntrack; filter rules validate current and original
203
+ tuples, links, direction, zone, state and generation label on every packet.
204
+ Opening TCP requires SYN with FIN/RST/ACK clear and permits ECN negotiation.
205
+ Replies must match the labelled original flow and current directed grant.
206
+ There is no broad ESTABLISHED/RELATED bypass.
207
+
208
+ Active overlapping classifiers, duplicate zones/labels and conflicting handoff
209
+ allocations reject. Empty router generations retain private routing while denying
210
+ egress. Input bounds include 32 private endpoints, 128 private rules, 32 active
211
+ generations, 128 total egress grants, 128 protected prefixes, 96 platform endpoints,
212
+ 16 ranges per allocation, and 32 host handoffs/active allocations. The complete
213
+ compiled graph still must fit 768 operations and 100,000 bytes; cross-products can
214
+ reach that limit before individual input limits. Exhausted source-port ranges
215
+ drop new flows rather than allocate outside the lease.
216
+
217
+ The caller must coordinate router and host apply order, retain complete intents
218
+ and receipts, authenticate protected authority, prevent reuse of quarantined
219
+ leases, and qualify other packet owners. An ACCEPT in this table cannot override
220
+ Docker's independent FORWARD DROP. `ManagedDockerForwarding`, described below,
221
+ owns the separate DOCKER-USER contribution. The caller must also control deferred packets, proxy/BPF/
222
+ offload paths and changing network authority. No table receipt proves flow
223
+ drainage, conntrack cleanup, elapsed-time expiry, or safe address/port/zone reuse.
224
+
225
+ ### Docker forwarding contribution
226
+
227
+ `ManagedDockerForwarding` admits the leased handoff traffic from an exact applied
228
+ schema-v2 `hostTransit` receipt through Docker's existing IPv4 forwarding path.
229
+ It reads and verifies that dedicated barrier's complete table and local links,
230
+ without adopting its ownership. The contribution uses Docker's supported
231
+ [DOCKER-USER extension point](https://docs.docker.com/engine/network/firewall-iptables/).
232
+ Docker's native nftables backend has no equivalent user chain and is unsupported.
233
+
234
+ The host must provide root-owned `/usr/sbin/iptables-nft`,
235
+ `/usr/sbin/iptables-nft-save` and `/usr/sbin/iptables-nft-restore`, using the same
236
+ 1.8.10-or-newer 1.8-series `nf_tables` frontend. `prepare()` captures the exact
237
+ frontend version in its digest. Apply, inspection and recovery require that same
238
+ version. Subprocesses use fixed argv, a clean environment, bounded input/output,
239
+ one shared eight-second frontend deadline per request, and joined termination.
240
+ No shell or raw rule API is exposed.
241
+
242
+ One node-level contribution owns a contiguous leading block in DOCKER-USER.
243
+ FORWARD must already have policy DROP and its first two unconditional jumps must
244
+ be DOCKER-USER and DOCKER-FORWARD. Their exact ordered raw graphs are checked
245
+ without claiming their ownership. Other managed contribution owners, altered or
246
+ duplicated owned rules, missing chains and changed placement reject. Unrelated
247
+ rules following the owned block and Docker's own chains remain separate owners.
248
+ Saved text verifies placement, count and normalized policy. Native netlink reads
249
+ verify every contributed rule's complete ordered expression graph, including
250
+ register flow and all match bytes; only counter values and attribute encoding
251
+ order/flags are normalized. An exact xtables `-C` check also verifies each compiled
252
+ deletion specification. These reads must share an unchanged ruleset generation;
253
+ concurrent Docker reconciliation can therefore make inspection inconclusive.
254
+ Conntrack source addresses use the canonical bare IPv4 form because
255
+ an explicit `/32` has different hidden match bytes in these frontends. A
256
+ semantically equal rewrite with a different exact representation rejects, as does
257
+ an early ACCEPT that the frontend reconstructs as the same command.
258
+ The contribution never creates, flushes, adopts or deletes a Docker table/chain,
259
+ and never changes a host forwarding policy.
260
+
261
+ Each contributed rule binds the handoff and uplink names, current and original
262
+ transit source, exact leased TCP/UDP port range, packet direction and connection
263
+ state. New TCP flows require SYN with FIN/RST/ACK clear; reverse traffic requires
264
+ ESTABLISHED. The exact v2 host barrier independently checks interface indices,
265
+ MAC/iflink facts, the default host conntrack zone, protected original/current
266
+ destinations and explicit outer SNAT. A complete contribution is bounded to 192
267
+ rules and 100,000 generated command bytes.
268
+
269
+ Create the class with `ownerId`, a caller-retained `instanceId` of at least 16
270
+ identifier characters, and optional `binaryPath`/`networkNamespaceFd`. After
271
+ `start()`, call `prepare({ schemaVersion: 1, revision, barrier })`, where `barrier`
272
+ is the exact applied host policy. Persist the complete `{ previous, target }`
273
+ transition before `reconcile()`. Keep every affected handoff link down through
274
+ mutations; retain those exact links until cleanup finishes. The native owner
275
+ verifies this down state before and after its single no-flush restore transaction.
276
+ Exact target replay can recover a lost response without rewriting live rules.
277
+
278
+ `inspect().present` confirms the contribution and its referenced barrier, not
279
+ the complete host packet path or durable controller authority. Errors retain
280
+ failed-owned state and pending intent. `release(applied)` and `close()` delete
281
+ only exact contributed rules under the same link fence. A cold boot needs a new
282
+ barrier receipt and caller-authorized replay; old boot receipts cannot authorize
283
+ native operations. The caller owns restart ordering, disjoint retained leases,
284
+ exclusive privileged mutation authority and activation. The frontend transaction
285
+ does not provide compare-and-swap against other privileged processes.
286
+
162
287
  ### Native qualification
163
288
 
164
- `cargo test --manifest-path rust/Cargo.toml --locked` runs only unprivileged pure
165
- tests. The ignored native cases require the explicitly marked disposable guest
289
+ The separate [Docker fixture](test/native/readme.docker.md) exercises the native
290
+ contribution with pinned Docker packages on offline Ubuntu 26.04 and Ubuntu
291
+ 24.04.4/HWE guests, including retained-state cold boots and published workloads.
292
+ Its checked-in input manifest and per-run receipts identify the tested artifacts.
293
+
294
+ `cargo test --manifest-path rust/Cargo.toml --locked` runs unprivileged compiler,
295
+ snapshot and bounded-subprocess tests. The ignored native cases require the explicitly marked disposable guest
166
296
  from `test/native/qualify.py`; requesting them on an ordinary host fails its scope
167
297
  check. Qualification uses an offline Linux 6.18.35 x86_64 guest with no host disks,
168
- mounts or network devices, and guest-owned network namespaces/veth pairs. It covers
298
+ mounts or external network backend. One VirtIO device connects only to a singleton
299
+ QEMU-internal hub; packet-path tests use guest-owned namespaces, veth pairs and TUN.
300
+ It covers
169
301
  UDP/TCP grants, direct-IP denial, source spoofing, IPv6 denial, renamed interfaces,
170
302
  foreign ownership, atomic rollback, generation conflicts and lost-ACK recovery.
303
+ V2 tests exercise complete graph persistence/recovery, private and local DNS,
304
+ local TCP, TUN isolation, observed UDP/TCP handoff ranges, range exhaustion,
305
+ fragment reassembly, ECN SYN and invalid opening flags, protected current/original
306
+ DNAT barriers, nonzero foreign zones, local diversion, IPv6 denial, revocation
307
+ and restoration with disjoint generation ranges while old flows remain retained.
171
308
  The arm64 binary is cross-built; native packet qualification is currently x86_64.
172
309
  Kernel 6.8 is unsupported. No production activation is implied by these tests.
173
310
 
@@ -0,0 +1,203 @@
1
+ use super::policy::{Options, Rule};
2
+ use crate::{Error, Result};
3
+ use std::collections::{BTreeMap, BTreeSet};
4
+ use std::io::{Read, Write};
5
+ use std::os::fd::AsRawFd;
6
+ use std::os::unix::{fs::MetadataExt, process::CommandExt};
7
+ use std::process::{Command, Stdio};
8
+ use std::time::{Duration, Instant};
9
+
10
+ const IPTABLES: &str = "/usr/sbin/iptables-nft";
11
+ const SAVE: &str = "/usr/sbin/iptables-nft-save";
12
+ const RESTORE: &str = "/usr/sbin/iptables-nft-restore";
13
+ const MAX_OUTPUT: usize = 2_097_152;
14
+ #[cfg(test)]
15
+ #[path="docker_process_tests.rs"]
16
+ mod process_tests;
17
+
18
+ fn nonblocking(fd: i32) -> Result<()> {
19
+ let flags = unsafe { libc::fcntl(fd, libc::F_GETFL) };
20
+ if flags < 0 || unsafe { libc::fcntl(fd,libc::F_SETFL,flags|libc::O_NONBLOCK) } < 0 {return Err(Error::Unavailable);}
21
+ Ok(())
22
+ }
23
+ pub fn deadline() -> Instant { Instant::now()+Duration::from_secs(8) }
24
+ fn command(path: &str, args: &[&str], input: &[u8], deadline: Instant) -> Result<String> {
25
+ if Instant::now()>=deadline {return Err(Error::Unavailable);}
26
+ let metadata=std::fs::metadata(path).map_err(|_|Error::Unavailable)?;
27
+ if !metadata.is_file() || metadata.uid()!=0 || metadata.mode() & 0o022 != 0 { return Err(Error::Permission); }
28
+ let mut child=Command::new(path).args(args).env_clear().env("PATH","/usr/sbin:/usr/bin:/sbin:/bin")
29
+ .env("LC_ALL","C").stdin(Stdio::piped()).stdout(Stdio::piped()).stderr(Stdio::piped()).process_group(0)
30
+ .spawn().map_err(|_|Error::Unavailable)?;
31
+ let result=(|| {
32
+ let mut stdin=child.stdin.take();
33
+ let mut stdout=child.stdout.take().ok_or(Error::Unavailable)?;
34
+ let mut stderr=child.stderr.take().ok_or(Error::Unavailable)?;
35
+ for fd in [stdin.as_ref().ok_or(Error::Unavailable)?.as_raw_fd(),stdout.as_raw_fd(),stderr.as_raw_fd()] {nonblocking(fd)?;}
36
+ let mut written=0;let mut output=Vec::new();let mut errors=Vec::new();
37
+ let mut out_eof=false;let mut err_eof=false;let mut status=None;
38
+ loop {
39
+ if Instant::now()>=deadline {return Err(Error::Unavailable);}
40
+ if written==input.len() {stdin=None;}
41
+ if let Some(pipe)=stdin.as_mut() {
42
+ match pipe.write(&input[written..]) {
43
+ Ok(0)=>return Err(Error::Unavailable),Ok(count)=>written+=count,
44
+ Err(error) if error.kind()==std::io::ErrorKind::WouldBlock=>{},
45
+ Err(_)=>return Err(Error::Unavailable),
46
+ }
47
+ }
48
+ for (pipe,bytes,eof) in [(&mut stdout as &mut dyn Read,&mut output,&mut out_eof),(&mut stderr as &mut dyn Read,&mut errors,&mut err_eof)] {
49
+ let mut buffer=[0;8192];
50
+ loop {
51
+ match pipe.read(&mut buffer) {
52
+ Ok(0)=>{*eof=true;break;},
53
+ Ok(count)=>{if bytes.len()+count>MAX_OUTPUT{return Err(Error::Protocol);}bytes.extend_from_slice(&buffer[..count]);},
54
+ Err(error) if error.kind()==std::io::ErrorKind::WouldBlock=>break,
55
+ Err(_)=>return Err(Error::Unavailable),
56
+ }
57
+ }
58
+ }
59
+ if status.is_none() {status=child.try_wait().map_err(|_|Error::Unavailable)?;}
60
+ if let Some(status)=status.filter(|_|out_eof&&err_eof) {
61
+ if !status.success() || written!=input.len() {
62
+ if path==IPTABLES && args.get(2)==Some(&"-C") && status.code()==Some(1)
63
+ && errors==b"iptables: Bad rule (does a matching rule exist in that chain?).\n" {
64
+ return Err(Error::Conflict);
65
+ }
66
+ return Err(Error::Unavailable);
67
+ }
68
+ return String::from_utf8(output).map_err(|_|Error::Protocol);
69
+ }
70
+ std::thread::sleep(Duration::from_millis(2));
71
+ }
72
+ })();
73
+ if result.is_err() {
74
+ // Stop and join this exact mutator before reporting an inconclusive ACK.
75
+ if child.try_wait().ok().flatten().is_none() {
76
+ unsafe {libc::kill(-(child.id() as i32),libc::SIGKILL);}
77
+ }
78
+ }
79
+ child.wait().map_err(|_|Error::Unavailable)?;
80
+ result
81
+ }
82
+ pub fn version(deadline: Instant) -> Result<String> {
83
+ let mut version=None;
84
+ for path in [IPTABLES,SAVE,RESTORE] {
85
+ let output=command(path,&["--version"],&[],deadline)?;
86
+ let parts:Vec<_>=output.split_whitespace().collect();
87
+ if parts.len()!=3 {return Err(Error::UnsupportedBackend);}
88
+ let observed=format!("{} {}",parts[1],parts[2]);
89
+ super::policy::validate_backend(&observed)?;
90
+ if version.as_ref().is_some_and(|v|v!=&observed) {return Err(Error::Conflict);}
91
+ version=Some(observed);
92
+ }
93
+ version.ok_or(Error::Unavailable)
94
+ }
95
+
96
+ /// Tokenize save output only; no token is ever evaluated by a shell.
97
+ fn tokens(line: &str) -> Result<Vec<String>> {
98
+ if line.len()>16_384 {return Err(Error::Protocol);}
99
+ let mut values=Vec::new();let mut word=String::new();let mut quoted=false;let mut escaped=false;let mut active=false;
100
+ for byte in line.bytes() {
101
+ if !(32..127).contains(&byte) {return Err(Error::Protocol);}
102
+ if escaped {word.push(byte as char);escaped=false;active=true;continue;}
103
+ if byte==b'\\' {escaped=true;continue;}
104
+ if byte==b'"' {quoted=!quoted;active=true;continue;}
105
+ if byte==b' ' && !quoted {if active {values.push(std::mem::take(&mut word));active=false;}} else {word.push(byte as char);active=true;}
106
+ }
107
+ if quoted||escaped {return Err(Error::Protocol);}
108
+ if active {values.push(word);}
109
+ if values.len()>512 {return Err(Error::Protocol);}
110
+ Ok(values)
111
+ }
112
+ fn normalized(args: &[String]) -> Result<Vec<(String, Vec<String>)>> {
113
+ let mut values=BTreeMap::new();let mut modules=BTreeSet::new();let mut index=0;
114
+ while index<args.len() {
115
+ let key=&args[index];let count=if key=="--tcp-flags" {2} else {1};
116
+ if !["-i","-o","-s","-d","-p","-m","--sport","--dport","--tcp-flags","--ctstate","--ctproto","--ctorigsrc","--ctorigsrcport","--ctdir","--comment","-j"].contains(&key.as_str()) {return Err(Error::Conflict);}
117
+ let mut value=args.get(index+1..index+1+count).ok_or(Error::Conflict)?.to_vec();index+=1+count;
118
+ if key=="-m" {if !modules.insert(value[0].clone()) {return Err(Error::Conflict);}continue;}
119
+ if key=="--ctorigsrc" && !value[0].contains('/') {value[0].push_str("/32");}
120
+ if key=="--ctproto" {value[0]=match value[0].as_str() {"tcp"=>"6".into(),"udp"=>"17".into(),_=>value[0].clone()};}
121
+ if values.insert(key.clone(),value).is_some() {return Err(Error::Conflict);}
122
+ }
123
+ values.insert("-m".into(),modules.into_iter().collect());
124
+ Ok(values.into_iter().collect())
125
+ }
126
+ pub struct Snapshot {
127
+ pub backend: String,
128
+ pub rows: Vec<Vec<String>>,
129
+ pub user: Vec<Vec<String>>,
130
+ raw: Option<super::graph::Snapshot>,
131
+ }
132
+ impl Snapshot {
133
+ pub fn read(deadline: Instant) -> Result<Self> {
134
+ let backend=version(deadline)?;
135
+ let (raw,text)=super::graph::Snapshot::read(||command(SAVE,&["-t","filter"],&[],deadline))?;
136
+ let mut snapshot=Self::parse(backend,&text)?;
137
+ snapshot.raw=Some(raw);
138
+ Ok(snapshot)
139
+ }
140
+ pub fn parse(backend: String, text: &str) -> Result<Self> {
141
+ if text.len()>MAX_OUTPUT {return Err(Error::Protocol);}
142
+ let mut table=false;let mut committed=false;let mut chains=BTreeMap::new();let mut rows=Vec::new();
143
+ for line in text.lines().filter(|line|!line.starts_with('#')&&!line.is_empty()) {
144
+ if line=="*filter" {if table||committed {return Err(Error::Conflict);}table=true;continue;}
145
+ if line=="COMMIT" {if !table||committed {return Err(Error::Conflict);}committed=true;continue;}
146
+ if !table||committed {return Err(Error::Conflict);}
147
+ let row=tokens(line)?;
148
+ if line.starts_with(':') {
149
+ if row.len()!=3 || chains.insert(row[0][1..].to_string(),row[1].clone()).is_some() {return Err(Error::Conflict);}
150
+ } else {
151
+ if row.len()<2 || row[0]!="-A" || rows.len()>=4096 {return Err(Error::Conflict);}
152
+ rows.push(row);
153
+ }
154
+ }
155
+ if !committed || chains.get("FORWARD").map(String::as_str)!=Some("DROP")
156
+ || chains.get("DOCKER-USER").map(String::as_str)!=Some("-")
157
+ || chains.get("DOCKER-FORWARD").map(String::as_str)!=Some("-") {return Err(Error::Conflict);}
158
+ let forward:Vec<_>=rows.iter().filter(|r|r[1]=="FORWARD").collect();
159
+ for (index,target) in ["DOCKER-USER","DOCKER-FORWARD"].iter().enumerate() {
160
+ if forward.get(index).map(|r|r.iter().map(String::as_str).collect::<Vec<_>>())!=Some(vec!["-A","FORWARD","-j",target]) {return Err(Error::Conflict);}
161
+ }
162
+ let user=rows.iter().filter(|r|r[1]=="DOCKER-USER").cloned().collect();
163
+ Ok(Self {backend,rows,user,raw:None})
164
+ }
165
+ /// Text proves placement/count and logical policy; -C additionally checks
166
+ /// match-extension bytes hidden by save output. Deletion uses the same spec.
167
+ pub fn verified_matches(&self, options: &Options, expected: &[Rule], deadline: Instant) -> Result<bool> {
168
+ if !self.matches(options,expected)? {return Ok(false);}
169
+ let raw=self.raw.as_ref().ok_or(Error::Conflict)?;
170
+ if !raw.matches(expected)? {return Ok(false);}
171
+ for rule in expected {
172
+ let mut args=vec!["-t","filter","-C","DOCKER-USER"];
173
+ args.extend(rule.args.iter().map(String::as_str));
174
+ match command(IPTABLES,&args,&[],deadline) {
175
+ Ok(_)=>{},Err(Error::Conflict)=>return Ok(false),Err(error)=>return Err(error),
176
+ }
177
+ }
178
+ raw.unchanged()
179
+ }
180
+ pub fn matches(&self, options: &Options, expected: &[Rule]) -> Result<bool> {
181
+ let prefix=format!("snftd1:{}:",options.owner_id);
182
+ // One node-level contribution owns the leading block. Do not prepend
183
+ // another managed owner and silently invalidate its placement proof.
184
+ if self.rows.iter().any(|r|r.iter().any(|t|t.starts_with("snftd1:")&&!t.starts_with(&prefix))) {return Err(Error::ForeignOwner);}
185
+ let owned:Vec<_>=self.rows.iter().filter(|r|r.iter().any(|t|t.starts_with(&prefix))).collect();
186
+ if owned.len()!=expected.len() || self.user.len()<expected.len() {return Ok(false);}
187
+ for (index,rule) in expected.iter().enumerate() {
188
+ let row=&self.user[index];
189
+ if row[1]!="DOCKER-USER" || !row.iter().any(|t|t==&rule.comment)
190
+ || normalized(&row[2..])?!=normalized(&rule.args)? {return Ok(false);}
191
+ }
192
+ Ok(true)
193
+ }
194
+ }
195
+ pub fn replace(previous: &[Rule], target: &[Rule], deadline: Instant) -> Result<()> {
196
+ let mut batch="*filter\n".to_string();
197
+ for rule in previous {batch.push_str(&format!("-D DOCKER-USER {}\n",rule.args.join(" ")));}
198
+ for rule in target.iter().rev() {batch.push_str(&format!("-I DOCKER-USER 1 {}\n",rule.args.join(" ")));}
199
+ batch.push_str("COMMIT\n");
200
+ if batch.len()>210_000 {return Err(Error::Invalid);}
201
+ command(RESTORE,&["--noflush","--wait","5"],batch.as_bytes(),deadline)?;
202
+ Ok(())
203
+ }
@@ -0,0 +1,129 @@
1
+ //! Ordered iptables-nft graph identity. Command-state reconstruction loses order.
2
+ use super::policy::Rule;
3
+ use crate::{policy::{compare, expr, meta, verdict}, wire::{self, Attr, Socket}, Error, Result};
4
+
5
+ fn value<'a>(rule: &'a Rule, key: &str) -> Result<&'a str> {
6
+ rule.args.windows(2).find(|pair|pair[0]==key).map(|pair|pair[1].as_str()).ok_or(Error::Invalid)
7
+ }
8
+ fn payload(base: u32, offset: u32, length: u32) -> Attr {
9
+ expr("payload",vec![Attr::u32(1,1),Attr::u32(2,base),Attr::u32(3,offset),Attr::u32(4,length)])
10
+ }
11
+
12
+ /// Fixed Linux UAPI / xtables 1.8.10 and 1.8.11 constructors, not a raw rule API.
13
+ pub(super) fn expected(rule: &Rule) -> Result<Vec<Attr>> {
14
+ let reply=value(rule,"--ctdir")?=="REPLY";
15
+ let new=value(rule,"--ctstate")?=="NEW";
16
+ let protocol=match value(rule,"-p")? {"tcp"=>6_u16,"udp"=>17,_=>return Err(Error::Invalid)};
17
+ let address=value(rule,"--ctorigsrc")?.parse::<std::net::Ipv4Addr>().map_err(|_|Error::Invalid)?.octets();
18
+ let ports=value(rule,"--ctorigsrcport")?;
19
+ let (first,last)=ports.split_once(':').unwrap_or((ports,ports));
20
+ let first=first.parse::<u16>().map_err(|_|Error::Invalid)?;
21
+ let last=last.parse::<u16>().map_err(|_|Error::Invalid)?;
22
+ let mut result=vec![payload(1,if reply {16} else {12},4),compare(address.to_vec())];
23
+ for (key,argument) in [(6,"-i"),(7,"-o")] {
24
+ result.extend(meta(key,[value(rule,argument)?.as_bytes(),&[0]].concat()));
25
+ }
26
+ result.extend([payload(1,9,1),compare(vec![protocol as u8])]);
27
+ if new && protocol==6 {
28
+ result.extend([payload(2,13,1),expr("bitwise",vec![
29
+ Attr::u32(1,1),Attr::u32(2,1),Attr::u32(3,1),Attr::u32(6,0),
30
+ Attr::nested(4,vec![Attr::bytes(1,vec![0x17])]),Attr::nested(5,vec![Attr::bytes(1,vec![0])]),
31
+ ]),compare(vec![2])]);
32
+ }
33
+ result.push(payload(2,if reply {2} else {0},2));
34
+ result.push(if first==last {compare(first.to_be_bytes().to_vec())} else {
35
+ expr("range",vec![Attr::u32(1,1),Attr::u32(2,0),
36
+ Attr::nested(3,vec![Attr::bytes(1,first.to_be_bytes().to_vec())]),
37
+ Attr::nested(4,vec![Attr::bytes(1,last.to_be_bytes().to_vec())])])
38
+ });
39
+ // xt_conntrack_mtinfo3 is 164 bytes, XT_ALIGN'ed to 168 on both shipped
40
+ // 64-bit Linux targets. All fields and padding are initialized explicitly.
41
+ let mut conntrack=vec![0;168];
42
+ conntrack[..4].copy_from_slice(&address);
43
+ conntrack[16..32].fill(0xff); // canonical bare-host nf_inet_addr mask
44
+ for (offset,number) in [(136,protocol),(138,first),(146,0x1107),
45
+ (148,if reply {0x1000} else {0}),(150,if new {8} else {2}),(154,last)] {
46
+ conntrack[offset..offset+2].copy_from_slice(&number.to_ne_bytes());
47
+ }
48
+ result.push(expr("match",vec![Attr::string(1,"conntrack"),Attr::u32(2,3),Attr::bytes(3,conntrack)]));
49
+ let mut comment=vec![0;256];
50
+ if rule.comment.len()>=comment.len() {return Err(Error::Invalid);}
51
+ comment[..rule.comment.len()].copy_from_slice(rule.comment.as_bytes());
52
+ result.push(expr("match",vec![Attr::string(1,"comment"),Attr::u32(2,0),Attr::bytes(3,comment)]));
53
+ result.push(expr("counter",vec![Attr::u64(1,0),Attr::u64(2,0)]));
54
+ result.push(verdict(1,None));
55
+ Ok(result)
56
+ }
57
+
58
+ /// Kernel dumps may omit nested flags and reorder fields; list order is exact.
59
+ fn equivalent(actual: &[Attr], expected: &[Attr], ordered: bool) -> Result<bool> {
60
+ if actual.len()!=expected.len() {return Ok(false);}
61
+ let mut actual=actual.iter().collect::<Vec<_>>();let mut expected=expected.iter().collect::<Vec<_>>();
62
+ if !ordered {actual.sort_by_key(|a|a.id());expected.sort_by_key(|a|a.id());}
63
+ for (actual,expected) in actual.into_iter().zip(expected) {
64
+ if actual.id()!=expected.id() {return Ok(false);}
65
+ if expected.kind & 0x8000 != 0 {
66
+ let expected=wire::attrs(&expected.value)?;
67
+ if !equivalent(&wire::attrs(&actual.value)?,&expected,expected.iter().all(|a|a.id()==1))? {return Ok(false);}
68
+ } else if actual.value!=expected.value {return Ok(false);}
69
+ }
70
+ Ok(true)
71
+ }
72
+ pub(super) fn matches_rule(row: &[Attr], rule: &Rule) -> Result<bool> {
73
+ matches_graph(row,"DOCKER-USER",&expected(rule)?)
74
+ }
75
+ pub(super) fn matches_forward(row: &[Attr], target: &str) -> Result<bool> {
76
+ if !["DOCKER-USER","DOCKER-FORWARD"].contains(&target) {return Err(Error::Invalid);}
77
+ matches_graph(row,"FORWARD",&[expr("counter",vec![Attr::u64(1,0),Attr::u64(2,0)]),verdict(-3,Some(target))])
78
+ }
79
+ fn matches_graph(row: &[Attr], chain: &str, expected: &[Attr]) -> Result<bool> {
80
+ if row.iter().any(|a|![1,2,3,4,6,8].contains(&a.id()))
81
+ || wire::text(row,1)?!="filter" || wire::text(row,2)?!=chain
82
+ || wire::handle(row,3)?==0 {return Ok(false);}
83
+ for id in [6,8] {
84
+ if row.iter().any(|a|a.id()==id) {
85
+ let attr=wire::one(row,id)?;
86
+ if id==6 {wire::handle(row,id)?;} else if !attr.value.is_empty() {return Ok(false);}
87
+ }
88
+ }
89
+ let mut expressions=wire::attrs(&wire::one(row,4)?.value)?;
90
+ for expression in &mut expressions {
91
+ let mut fields=wire::attrs(&expression.value)?;
92
+ if wire::text(&fields,1)?=="counter" {
93
+ let mut data=wire::attrs(&wire::one(&fields,2)?.value)?;
94
+ // Counter values change without a ruleset generation change. No
95
+ // expression, including the counter itself, may move or disappear.
96
+ data.retain(|a|!(a.id()==3 && a.value.is_empty()));
97
+ if data.len()!=2 {return Ok(false);}
98
+ for id in [1,2] {wire::handle(&data,id)?;}
99
+ for attribute in &mut data {attribute.value=0_u64.to_be_bytes().to_vec();}
100
+ let field=fields.iter_mut().find(|a|a.id()==2).ok_or(Error::Protocol)?;
101
+ *field=Attr::nested(2,data);
102
+ expression.value=wire::encode_attrs(&fields);
103
+ }
104
+ }
105
+ equivalent(&expressions,expected,true)
106
+ }
107
+
108
+ pub struct Snapshot { generation: u32, rows: Vec<Vec<Attr>> }
109
+ impl Snapshot {
110
+ pub fn read(save: impl FnOnce()->Result<String>) -> Result<(Self,String)> {
111
+ let mut socket=Socket::open()?;let generation=socket.generation()?;
112
+ let text=save()?;
113
+ let rows=socket.query_ipv4(7,vec![Attr::string(1,"filter"),Attr::string(2,"DOCKER-USER")],true)?;
114
+ let forward=socket.query_ipv4(7,vec![Attr::string(1,"filter"),Attr::string(2,"FORWARD")],true)?;
115
+ if socket.generation()!=Ok(generation) {return Err(Error::Conflict);}
116
+ for (index,target) in ["DOCKER-USER","DOCKER-FORWARD"].iter().enumerate() {
117
+ if !matches_forward(forward.get(index).ok_or(Error::Conflict)?,target)? {return Err(Error::Conflict);}
118
+ }
119
+ Ok((Self {generation,rows},text))
120
+ }
121
+ pub fn matches(&self, expected: &[Rule]) -> Result<bool> {
122
+ if self.rows.len()<expected.len() {return Ok(false);}
123
+ for (row,rule) in self.rows.iter().zip(expected) {
124
+ if !matches_rule(row,rule)? {return Ok(false);}
125
+ }
126
+ Ok(true)
127
+ }
128
+ pub fn unchanged(&self) -> Result<bool> {Ok(Socket::open()?.generation()?==self.generation)}
129
+ }