@push.rocks/smartnftables 4.1.0 → 4.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. package/changelog.md +23 -0
  2. package/dist_rust/smartnftables_linux_amd64_musl +0 -0
  3. package/dist_rust/smartnftables_linux_amd64_musl.tsrust-build.json +5 -5
  4. package/dist_rust/smartnftables_linux_arm64_musl +0 -0
  5. package/dist_rust/smartnftables_linux_arm64_musl.tsrust-build.json +5 -5
  6. package/dist_ts/00_commitinfo_data.js +1 -1
  7. package/dist_ts/classes.manageddockerforwarding.d.ts +4 -1
  8. package/dist_ts/classes.manageddockerforwarding.js +8 -5
  9. package/dist_ts/classes.managednftables.d.ts +8 -2
  10. package/dist_ts/classes.managednftables.js +11 -6
  11. package/dist_ts/forwardhooks.d.ts +8 -0
  12. package/dist_ts/forwardhooks.js +42 -0
  13. package/dist_ts/index.d.ts +2 -0
  14. package/dist_ts/index.js +2 -1
  15. package/dist_ts/managed.docker.types.d.ts +2 -1
  16. package/dist_ts/managed.forwardhooks.types.d.ts +35 -0
  17. package/dist_ts/managed.forwardhooks.types.js +2 -0
  18. package/package.json +4 -4
  19. package/readme.md +80 -17
  20. package/rust/src/docker.frontend.rs +47 -9
  21. package/rust/src/docker.graph.rs +53 -14
  22. package/rust/src/docker.policy.rs +90 -31
  23. package/rust/src/docker.rs +48 -20
  24. package/rust/src/docker_tests.rs +206 -2
  25. package/rust/src/egress.host.rs +44 -45
  26. package/rust/src/egress.rs +56 -0
  27. package/rust/src/forwardhooks.rs +176 -0
  28. package/rust/src/forwardhooks_tests.rs +136 -0
  29. package/rust/src/main.rs +12 -8
  30. package/rust/src/wire.links.rs +28 -6
  31. package/rust/src/wire.rs +22 -2
  32. package/ts/00_commitinfo_data.ts +1 -1
  33. package/ts/classes.manageddockerforwarding.ts +7 -4
  34. package/ts/classes.managednftables.ts +13 -5
  35. package/ts/forwardhooks.ts +39 -0
  36. package/ts/index.ts +2 -0
  37. package/ts/managed.docker.types.ts +2 -1
  38. package/ts/managed.forwardhooks.types.ts +26 -0
package/readme.md CHANGED
@@ -297,7 +297,9 @@ endpoint, addresses per link, protected prefixes, platform endpoints, active
297
297
  generations, handoffs and allocations, ranges per allocation, guarded pools, the
298
298
  IPC and capture limits) describes the shape of the topology and stays `INVALID`.
299
299
  Other codes, and the facade's own local rejections, carry neither `reason` nor
300
- `details`. A native refusal outside this shape is `PROTOCOL`.
300
+ `details`. A native refusal outside this shape is `PROTOCOL`. A failed `close()`
301
+ rejects `CLEANUP_UNCONFIRMED`, whose `cause` is the `ManagedNftablesError` that left
302
+ cleanup unconfirmed, with the native code when the native owner refused.
301
303
 
302
304
  ```typescript
303
305
  try {
@@ -452,8 +454,8 @@ under host grants. 64 symmetric ranges take about 19,600 element bytes.
452
454
  The caller still owns the route to `targetAddress` through that handoff, the
453
455
  workload listener, and every other packet owner on the host. As with leased egress,
454
456
  an ACCEPT here cannot override Docker's independent FORWARD DROP; the
455
- `ManagedDockerForwarding` contribution below derives only leased handoff egress from
456
- the barrier and currently carries no inbound publication.
457
+ `ManagedDockerForwarding` contribution below admits every inbound and symmetric
458
+ publication of the barrier through it, with the leased flows.
457
459
 
458
460
  Each router generation binds an immutable lease reference, transit source address,
459
461
  leased TCP/UDP source-port ranges, a nonzero conntrack zone and a 16-byte label.
@@ -693,7 +695,8 @@ forwarding that another owner (Docker, a container or VM bridge, a VPN router, a
693
695
  second routing daemon) accepts, and accepting in this table still cannot override
694
696
  another owner's drop. Use it only on a host where this table is the sole forwarding
695
697
  owner; on a Docker host keep it absent and use the Docker contribution below, which
696
- refuses an exclusive barrier. The
698
+ refuses an exclusive barrier. `inspectForwardHooks()`, described below, reports
699
+ the other forward owners of the namespace for that decision. The
697
700
  member does not fence flowtable offload, packet queues, proxies or anything outside
698
701
  the forward hook.
699
702
 
@@ -880,7 +883,7 @@ drainage, conntrack cleanup, elapsed-time expiry, or safe address/port/zone reus
880
883
 
881
884
  ### Docker forwarding contribution
882
885
 
883
- `ManagedDockerForwarding` admits the leased handoff traffic from an exact applied
886
+ `ManagedDockerForwarding` admits exactly the forwarded flows of an exact applied
884
887
  schema-v2 `hostTransit` receipt through Docker's existing IPv4 forwarding path.
885
888
  It reads and verifies that dedicated barrier's complete table and local links,
886
889
  without adopting its ownership. The contribution uses Docker's supported
@@ -916,32 +919,92 @@ and never changes a host forwarding policy. `prepare()` refuses a barrier with
916
919
  `exclusiveForwarding` as `INVALID`: its drop would also deny Docker's own container
917
920
  forwarding.
918
921
 
919
- Each contributed rule binds the handoff and uplink names, current and original
920
- transit source, exact leased TCP/UDP port range, packet direction and connection
921
- state. New TCP flows require SYN with FIN/RST/ACK clear; reverse traffic requires
922
- ESTABLISHED. The exact v2 host barrier independently checks interface indices,
923
- MAC/iflink facts, the default host conntrack zone, protected original/current
924
- destinations and explicit outer SNAT. A complete contribution is bounded to 192
925
- rules and 100,000 generated command bytes.
922
+ The contribution and the barrier's forward chain derive from one enumeration of
923
+ the scope's forwarded flows, so the contribution admits exactly the flow kinds the
924
+ non-exclusive barrier admits, each with its replies:
925
+
926
+ | Flow kind | Original direction | Live tuple | Conntrack original tuple |
927
+ | --- | --- | --- | --- |
928
+ | Leased range | handoff → uplink | transit source address and leased port range | same source |
929
+ | Inbound publication | uplink → handoff, after the barrier's DNAT | inside target address and port(s) as destination | published `hostIp` and port(s) as destination |
930
+ | Symmetric publication | handoff → uplink, before the barrier's SNAT | target address and port(s) as source | same source |
931
+
932
+ Every flow kind has three rules: the NEW opening (TCP only with SYN and
933
+ FIN/RST/ACK clear), the ESTABLISHED original direction and the ESTABLISHED reply
934
+ with swapped interfaces and tuple sides. A router's own egress, such as a VPN client
935
+ in the router namespace dialling its hub, reaches the host as a leased flow: the
936
+ router translates it to the transit address and a leased port range, so it is
937
+ admitted by the leased rules exactly when the barrier admits it. Host grants and
938
+ host-local platform endpoints use INPUT/OUTPUT and never cross FORWARD. Each rule
939
+ binds the handoff and uplink names. The exact v2 host barrier independently checks
940
+ interface indices, MAC/iflink facts, the default host conntrack zone, protected
941
+ original/current destinations and explicit outer SNAT; it governs every forwarded
942
+ packet on its handoffs, so the host forwards the intersection, which is the
943
+ barrier's admitted set. A scope without publications keeps its exact 4.1.0 rules,
944
+ comments and graphs. A complete contribution is bounded to 192 rules and 100,000
945
+ generated command bytes: three rules per leased range and per publication, and
946
+ three more per symmetric publication. A larger scope is refused as `EXHAUSTED`
947
+ (`restoreRules`/`restoreBytes`) before any subprocess.
926
948
 
927
949
  Create the class with `ownerId`, a caller-retained `instanceId` of at least 16
928
950
  identifier characters, and optional `binaryPath`/`networkNamespaceFd`. After
929
951
  `start()`, call `prepare({ schemaVersion: 1, revision, barrier })`, where `barrier`
930
952
  is the exact applied host policy. Persist the complete `{ previous, target }`
931
- transition before `reconcile()`. Keep every affected handoff link down through
932
- mutations; retain those exact links until cleanup finishes. The native owner
933
- verifies this down state before and after its single no-flush restore transaction.
934
- Exact target replay can recover a lost response without rewriting live rules.
953
+ transition before `reconcile()`. The native owner fences the handoff links before
954
+ and after its single no-flush restore transaction, which deletes the previous block
955
+ and inserts the target block in one atomic nf_tables commit. A handoff link that
956
+ `previous` and `target` both bind exactly may stay up, so a publication or lease
957
+ amendment of a running generation applies in place: every flow both contributions
958
+ admit passes throughout, a flow only the target admits passes from the commit, and
959
+ the referenced barrier must already be the target. Every handoff link the
960
+ transition adds or drops, and every link of a first apply, must be down; retain
961
+ those exact links until cleanup finishes. Exact target replay can recover a lost
962
+ response without rewriting live rules.
935
963
 
936
964
  `inspect().present` confirms the contribution and its referenced barrier, not
937
965
  the complete host packet path or durable controller authority. Errors retain
938
966
  failed-owned state and pending intent. `release(applied)` and `close()` delete
939
- only exact contributed rules under the same link fence. A cold boot needs a new
967
+ only exact contributed rules. `close()` first reads the filter table through the
968
+ same generation-consistent frontend and netlink read, without requiring Docker's
969
+ forwarding path: when no saved rule in any chain and no native DOCKER-USER or
970
+ FORWARD rule carries this owner's comment prefix, cleanup is confirmed. So an owner
971
+ whose first apply was refused, for example on a host with FORWARD policy ACCEPT or
972
+ without Docker, closes on that host. A read that fails rejects
973
+ `CLEANUP_UNCONFIRMED` whose `cause` carries the native code (`UNAVAILABLE`,
974
+ `UNSUPPORTED_BACKEND`, `PERMISSION`, `CONFLICT`); it is never reported as cleanup.
975
+ The fence of `release(applied)` and `close()` requires every handoff link down or gone:
976
+ a link whose index no link holds any more, such as a veth that left with its router
977
+ namespace when the node process ended, counts as down, so a restarted process in
978
+ the same boot can release the previous contribution before applying the next one.
979
+ A different link at a bound index still rejects. A cold boot needs a new
940
980
  barrier receipt and caller-authorized replay; old boot receipts cannot authorize
941
981
  native operations. The caller owns restart ordering, disjoint retained leases,
942
982
  exclusive privileged mutation authority and activation. The frontend transaction
943
983
  does not provide compare-and-swap against other privileged processes.
944
984
 
985
+ ### Forward hook owners
986
+
987
+ `inspectForwardHooks({ binaryPath?, networkNamespaceFd? })` reads every
988
+ packet-filter owner on the routed forward path of one network namespace and
989
+ changes nothing. One native process takes the read and exits:
990
+
991
+ - `chains`: every nf_tables base chain on the forward hook of the `ip`, `ip6` and
992
+ `inet` families, with `family`, `table`, `tableHandle`, `chain`, `type`,
993
+ `priority` and `policy`, sorted by family, table and chain. Managed tables are
994
+ included; compare `table` and `tableHandle` with a receipt's
995
+ `identity.tableName` and `tableHandle` to find your own.
996
+ - `docker`: whether Docker's iptables-nft `DOCKER-USER` and `DOCKER-FORWARD`
997
+ chains exist in table `ip filter`.
998
+ - `legacyTables`: the registered legacy xtables `ipv4` and `ipv6` tables. Their
999
+ hooks, including FORWARD, are not nf_tables objects and never appear in `chains`.
1000
+
1001
+ The table and chain dumps share one ruleset generation, or the read is
1002
+ `CONFLICT`. Without `CAP_NET_ADMIN` in the namespace it is `PERMISSION`. Bridge
1003
+ and netdev hooks do not see routed packets and are not reported. Guard
1004
+ `exclusiveForwarding` with it: refuse while any chain other than your own barrier's,
1005
+ either Docker chain or any legacy table is present. The read is a point in time;
1006
+ the caller still owns exclusive authority over who may add a forward owner later.
1007
+
945
1008
  ### Native qualification
946
1009
 
947
1010
  The separate [Docker fixture](test/native/readme.docker.md) exercises the native
@@ -113,10 +113,10 @@ fn normalized(args: &[String]) -> Result<Vec<(String, Vec<String>)>> {
113
113
  let mut values=BTreeMap::new();let mut modules=BTreeSet::new();let mut index=0;
114
114
  while index<args.len() {
115
115
  let key=&args[index];let count=if key=="--tcp-flags" {2} else {1};
116
- if !["-i","-o","-s","-d","-p","-m","--sport","--dport","--tcp-flags","--ctstate","--ctproto","--ctorigsrc","--ctorigsrcport","--ctdir","--comment","-j"].contains(&key.as_str()) {return Err(Error::Conflict);}
116
+ if !["-i","-o","-s","-d","-p","-m","--sport","--dport","--tcp-flags","--ctstate","--ctproto","--ctorigsrc","--ctorigsrcport","--ctorigdst","--ctorigdstport","--ctdir","--comment","-j"].contains(&key.as_str()) {return Err(Error::Conflict);}
117
117
  let mut value=args.get(index+1..index+1+count).ok_or(Error::Conflict)?.to_vec();index+=1+count;
118
118
  if key=="-m" {if !modules.insert(value[0].clone()) {return Err(Error::Conflict);}continue;}
119
- if key=="--ctorigsrc" && !value[0].contains('/') {value[0].push_str("/32");}
119
+ if (key=="--ctorigsrc" || key=="--ctorigdst") && !value[0].contains('/') {value[0].push_str("/32");}
120
120
  if key=="--ctproto" {value[0]=match value[0].as_str() {"tcp"=>"6".into(),"udp"=>"17".into(),_=>value[0].clone()};}
121
121
  if values.insert(key.clone(),value).is_some() {return Err(Error::Conflict);}
122
122
  }
@@ -125,19 +125,39 @@ fn normalized(args: &[String]) -> Result<Vec<(String, Vec<String>)>> {
125
125
  }
126
126
  pub struct Snapshot {
127
127
  pub backend: String,
128
+ chains: BTreeMap<String, String>,
128
129
  pub rows: Vec<Vec<String>>,
129
130
  pub user: Vec<Vec<String>>,
130
131
  raw: Option<super::graph::Snapshot>,
131
132
  }
132
133
  impl Snapshot {
134
+ /// Docker's admitted forwarding path. Apply, inspection and every deletion
135
+ /// proof read through it.
133
136
  pub fn read(deadline: Instant) -> Result<Self> {
137
+ let snapshot=Self::observe(deadline)?;
138
+ snapshot.path()?;
139
+ snapshot.raw.as_ref().ok_or(Error::Conflict)?.path()?;
140
+ Ok(snapshot)
141
+ }
142
+ /// The same generation-consistent read of the filter table, without
143
+ /// requiring Docker's forwarding path. It proves only whether this owner
144
+ /// holds any rule there, never a placement or a deletion.
145
+ pub fn observe(deadline: Instant) -> Result<Self> {
134
146
  let backend=version(deadline)?;
135
147
  let (raw,text)=super::graph::Snapshot::read(||command(SAVE,&["-t","filter"],&[],deadline))?;
136
- let mut snapshot=Self::parse(backend,&text)?;
148
+ let mut snapshot=Self::table(backend,&text)?;
137
149
  snapshot.raw=Some(raw);
138
150
  Ok(snapshot)
139
151
  }
152
+ /// The text half of `read`.
153
+ #[cfg(test)]
140
154
  pub fn parse(backend: String, text: &str) -> Result<Self> {
155
+ let snapshot=Self::table(backend,text)?;
156
+ snapshot.path()?;
157
+ Ok(snapshot)
158
+ }
159
+ /// One complete saved filter table: its chains and rules, in order.
160
+ pub(super) fn table(backend: String, text: &str) -> Result<Self> {
141
161
  if text.len()>MAX_OUTPUT {return Err(Error::Protocol);}
142
162
  let mut table=false;let mut committed=false;let mut chains=BTreeMap::new();let mut rows=Vec::new();
143
163
  for line in text.lines().filter(|line|!line.starts_with('#')&&!line.is_empty()) {
@@ -152,15 +172,33 @@ impl Snapshot {
152
172
  rows.push(row);
153
173
  }
154
174
  }
155
- if !committed || chains.get("FORWARD").map(String::as_str)!=Some("DROP")
156
- || chains.get("DOCKER-USER").map(String::as_str)!=Some("-")
157
- || chains.get("DOCKER-FORWARD").map(String::as_str)!=Some("-") {return Err(Error::Conflict);}
158
- let forward:Vec<_>=rows.iter().filter(|r|r[1]=="FORWARD").collect();
175
+ if !committed {return Err(Error::Conflict);}
176
+ let user=rows.iter().filter(|r|r[1]=="DOCKER-USER").cloned().collect();
177
+ Ok(Self {backend,chains,rows,user,raw:None})
178
+ }
179
+ /// FORWARD has policy DROP and dispatches first to DOCKER-USER, then to
180
+ /// DOCKER-FORWARD.
181
+ fn path(&self) -> Result<()> {
182
+ if self.chains.get("FORWARD").map(String::as_str)!=Some("DROP")
183
+ || self.chains.get("DOCKER-USER").map(String::as_str)!=Some("-")
184
+ || self.chains.get("DOCKER-FORWARD").map(String::as_str)!=Some("-") {return Err(Error::Conflict);}
185
+ let forward:Vec<_>=self.rows.iter().filter(|r|r[1]=="FORWARD").collect();
159
186
  for (index,target) in ["DOCKER-USER","DOCKER-FORWARD"].iter().enumerate() {
160
187
  if forward.get(index).map(|r|r.iter().map(String::as_str).collect::<Vec<_>>())!=Some(vec!["-A","FORWARD","-j",target]) {return Err(Error::Conflict);}
161
188
  }
162
- let user=rows.iter().filter(|r|r[1]=="DOCKER-USER").cloned().collect();
163
- Ok(Self {backend,rows,user,raw:None})
189
+ Ok(())
190
+ }
191
+ /// No rule of this owner exists in the filter table: neither saved text in
192
+ /// any chain nor native DOCKER-USER/FORWARD attributes carry its comment
193
+ /// prefix. Rules of other owners are not this owner's to prove.
194
+ pub fn owns_nothing(&self, options: &Options) -> Result<bool> {
195
+ let prefix=format!("snftd1:{}:",options.owner_id);
196
+ let raw=self.raw.as_ref().ok_or(Error::Conflict)?;
197
+ Ok(!self.mentions(&prefix) && !raw.mentions(prefix.as_bytes()))
198
+ }
199
+ /// Whether any saved rule, in any chain, carries a token starting with `prefix`.
200
+ pub(super) fn mentions(&self, prefix: &str) -> bool {
201
+ self.rows.iter().any(|r|r.iter().any(|t|t.starts_with(prefix)))
164
202
  }
165
203
  /// Text proves placement/count and logical policy; -C additionally checks
166
204
  /// match-extension bytes hidden by save output. Deletion uses the same spec.
@@ -9,17 +9,36 @@ fn payload(base: u32, offset: u32, length: u32) -> Attr {
9
9
  expr("payload",vec![Attr::u32(1,1),Attr::u32(2,base),Attr::u32(3,offset),Attr::u32(4,length)])
10
10
  }
11
11
 
12
+ fn span(value: &str) -> Result<(u16, u16)> {
13
+ let (first,last)=value.split_once(':').unwrap_or((value,value));
14
+ Ok((first.parse::<u16>().map_err(|_|Error::Invalid)?,last.parse::<u16>().map_err(|_|Error::Invalid)?))
15
+ }
16
+ /// The one key of a rule among alternatives, with its value.
17
+ fn either<'a>(rule: &'a Rule, keys: [&'static str; 2]) -> Result<(usize, &'a str)> {
18
+ let mut found=keys.iter().enumerate().filter_map(|(index,key)|value(rule,key).ok().map(|value|(index,value)));
19
+ let result=found.next().ok_or(Error::Invalid)?;
20
+ if found.next().is_some() {return Err(Error::Invalid);}
21
+ Ok(result)
22
+ }
23
+
12
24
  /// Fixed Linux UAPI / xtables 1.8.10 and 1.8.11 constructors, not a raw rule API.
13
25
  pub(super) fn expected(rule: &Rule) -> Result<Vec<Attr>> {
14
26
  let reply=value(rule,"--ctdir")?=="REPLY";
15
27
  let new=value(rule,"--ctstate")?=="NEW";
16
28
  let protocol=match value(rule,"-p")? {"tcp"=>6_u16,"udp"=>17,_=>return Err(Error::Invalid)};
17
- let address=value(rule,"--ctorigsrc")?.parse::<std::net::Ipv4Addr>().map_err(|_|Error::Invalid)?.octets();
18
- let ports=value(rule,"--ctorigsrcport")?;
19
- let (first,last)=ports.split_once(':').unwrap_or((ports,ports));
20
- let first=first.parse::<u16>().map_err(|_|Error::Invalid)?;
21
- let last=last.parse::<u16>().map_err(|_|Error::Invalid)?;
22
- let mut result=vec![payload(1,if reply {16} else {12},4),compare(address.to_vec())];
29
+ // The packet's own source (0) or destination (1) address and port.
30
+ let (side,live)=either(rule,["-s","-d"])?;
31
+ let live=live.strip_suffix("/32").ok_or(Error::Invalid)?.parse::<std::net::Ipv4Addr>().map_err(|_|Error::Invalid)?.octets();
32
+ let (port_side,ports)=either(rule,["--sport","--dport"])?;
33
+ if port_side!=side {return Err(Error::Invalid);}
34
+ let (first,last)=span(ports)?;
35
+ // The originally tracked source (0) or destination (1) address and port.
36
+ let (tracked_side,tracked)=either(rule,["--ctorigsrc","--ctorigdst"])?;
37
+ let tracked=tracked.parse::<std::net::Ipv4Addr>().map_err(|_|Error::Invalid)?.octets();
38
+ let (tracked_port_side,tracked_ports)=either(rule,["--ctorigsrcport","--ctorigdstport"])?;
39
+ if tracked_port_side!=tracked_side {return Err(Error::Invalid);}
40
+ let (tracked_first,tracked_last)=span(tracked_ports)?;
41
+ let mut result=vec![payload(1,if side==0 {12} else {16},4),compare(live.to_vec())];
23
42
  for (key,argument) in [(6,"-i"),(7,"-o")] {
24
43
  result.extend(meta(key,[value(rule,argument)?.as_bytes(),&[0]].concat()));
25
44
  }
@@ -30,7 +49,7 @@ pub(super) fn expected(rule: &Rule) -> Result<Vec<Attr>> {
30
49
  Attr::nested(4,vec![Attr::bytes(1,vec![0x17])]),Attr::nested(5,vec![Attr::bytes(1,vec![0])]),
31
50
  ]),compare(vec![2])]);
32
51
  }
33
- result.push(payload(2,if reply {2} else {0},2));
52
+ result.push(payload(2,if side==0 {0} else {2},2));
34
53
  result.push(if first==last {compare(first.to_be_bytes().to_vec())} else {
35
54
  expr("range",vec![Attr::u32(1,1),Attr::u32(2,0),
36
55
  Attr::nested(3,vec![Attr::bytes(1,first.to_be_bytes().to_vec())]),
@@ -38,11 +57,17 @@ pub(super) fn expected(rule: &Rule) -> Result<Vec<Attr>> {
38
57
  });
39
58
  // xt_conntrack_mtinfo3 is 164 bytes, XT_ALIGN'ed to 168 on both shipped
40
59
  // 64-bit Linux targets. All fields and padding are initialized explicitly.
60
+ // The original source address/mask are at 0/16, its ports at 138/154; the
61
+ // original destination's at 32/48 and 140/156.
41
62
  let mut conntrack=vec![0;168];
42
- conntrack[..4].copy_from_slice(&address);
43
- conntrack[16..32].fill(0xff); // canonical bare-host nf_inet_addr mask
44
- for (offset,number) in [(136,protocol),(138,first),(146,0x1107),
45
- (148,if reply {0x1000} else {0}),(150,if new {8} else {2}),(154,last)] {
63
+ let address=32*tracked_side;
64
+ conntrack[address..address+4].copy_from_slice(&tracked);
65
+ conntrack[address+16..address+32].fill(0xff); // canonical bare-host nf_inet_addr mask
66
+ // STATE|PROTO|DIRECTION with ORIGSRC|ORIGSRC_PORT or ORIGDST|ORIGDST_PORT.
67
+ let flags=0x1003|if tracked_side==0 {0x0104} else {0x0208};
68
+ let port=2*tracked_side;
69
+ for (offset,number) in [(136,protocol),(138+port,tracked_first),(146,flags),
70
+ (148,if reply {0x1000} else {0}),(150,if new {8} else {2}),(154+port,tracked_last)] {
46
71
  conntrack[offset..offset+2].copy_from_slice(&number.to_ne_bytes());
47
72
  }
48
73
  result.push(expr("match",vec![Attr::string(1,"conntrack"),Attr::u32(2,3),Attr::bytes(3,conntrack)]));
@@ -105,19 +130,33 @@ fn matches_graph(row: &[Attr], chain: &str, expected: &[Attr]) -> Result<bool> {
105
130
  equivalent(&expressions,expected,true)
106
131
  }
107
132
 
108
- pub struct Snapshot { generation: u32, rows: Vec<Vec<Attr>> }
133
+ /// Whether any rule carries `marker` anywhere in its native attributes. An owner
134
+ /// comment is stored verbatim, whether the frontend can render the rule or not.
135
+ pub(super) fn mentions<'a>(rows: impl IntoIterator<Item=&'a Vec<Attr>>, marker: &[u8]) -> bool {
136
+ rows.into_iter().any(|row|wire::encode_attrs(row).windows(marker.len()).any(|window|window==marker))
137
+ }
138
+
139
+ pub struct Snapshot { generation: u32, rows: Vec<Vec<Attr>>, forward: Vec<Vec<Attr>> }
109
140
  impl Snapshot {
141
+ /// The saved text and the native DOCKER-USER and FORWARD rules under one
142
+ /// unchanged ruleset generation. A missing table or chain dumps no rules.
110
143
  pub fn read(save: impl FnOnce()->Result<String>) -> Result<(Self,String)> {
111
144
  let mut socket=Socket::open()?;let generation=socket.generation()?;
112
145
  let text=save()?;
113
146
  let rows=socket.query_ipv4(7,vec![Attr::string(1,"filter"),Attr::string(2,"DOCKER-USER")],true)?;
114
147
  let forward=socket.query_ipv4(7,vec![Attr::string(1,"filter"),Attr::string(2,"FORWARD")],true)?;
115
148
  if socket.generation()!=Ok(generation) {return Err(Error::Conflict);}
149
+ Ok((Self {generation,rows,forward},text))
150
+ }
151
+ /// Docker's forwarding path: FORWARD's first two rules are its exact
152
+ /// unconditional jumps to DOCKER-USER and then DOCKER-FORWARD.
153
+ pub fn path(&self) -> Result<()> {
116
154
  for (index,target) in ["DOCKER-USER","DOCKER-FORWARD"].iter().enumerate() {
117
- if !matches_forward(forward.get(index).ok_or(Error::Conflict)?,target)? {return Err(Error::Conflict);}
155
+ if !matches_forward(self.forward.get(index).ok_or(Error::Conflict)?,target)? {return Err(Error::Conflict);}
118
156
  }
119
- Ok((Self {generation,rows},text))
157
+ Ok(())
120
158
  }
159
+ pub fn mentions(&self, marker: &[u8]) -> bool {mentions(self.rows.iter().chain(&self.forward),marker)}
121
160
  pub fn matches(&self, expected: &[Rule]) -> Result<bool> {
122
161
  if self.rows.len()<expected.len() {return Ok(false);}
123
162
  for (row,rule) in self.rows.iter().zip(expected) {
@@ -99,46 +99,105 @@ pub struct Rule {
99
99
  fn port_range(first: u16, last: u16) -> String {
100
100
  if first == last { first.to_string() } else { format!("{first}:{last}") }
101
101
  }
102
+ /// One side of a packet or of its originally tracked tuple: an exact address
103
+ /// and an inclusive port span.
104
+ #[derive(Clone, Copy)]
105
+ struct Tuple<'a> {
106
+ address: &'a str,
107
+ first: u16,
108
+ last: u16,
109
+ }
110
+ /// One flow class of the barrier's forward chain, in its original direction:
111
+ /// it enters on `input`, leaves on `output`, carries `live` as its source
112
+ /// (`source`) or destination, and conntrack tracked it with `tracked` as its
113
+ /// original source (`source`) or destination. A leased or symmetric flow is
114
+ /// keyed on its own source, before outer source NAT; an inbound publication on
115
+ /// its translated inside destination and its published outside destination.
116
+ struct Flow<'a> {
117
+ input: &'a str,
118
+ output: &'a str,
119
+ protocol: &'a str,
120
+ source: bool,
121
+ live: Tuple<'a>,
122
+ tracked: Tuple<'a>,
123
+ }
124
+ impl Flow<'_> {
125
+ /// The ESTABLISHED original direction, a NEW opening (TCP only with SYN and
126
+ /// FIN/RST/ACK clear) and the ESTABLISHED reply, as the barrier admits them.
127
+ fn rules(&self, mut push: impl FnMut(Vec<String>) -> Result<()>) -> Result<()> {
128
+ for state in ["NEW", "ESTABLISHED", "REPLY"] {
129
+ let reply = state == "REPLY";
130
+ // The reply swaps the interfaces and the packet's tuple sides; the
131
+ // originally tracked tuple stays the same.
132
+ let source = self.source != reply;
133
+ let mut args: Vec<String> = vec![
134
+ "-i".into(), if reply { self.output } else { self.input }.into(),
135
+ "-o".into(), if reply { self.input } else { self.output }.into(),
136
+ if source { "-s" } else { "-d" }.into(), format!("{}/32", self.live.address),
137
+ "-p".into(), self.protocol.into(), "-m".into(), self.protocol.into(),
138
+ if source { "--sport" } else { "--dport" }.into(), port_range(self.live.first, self.live.last),
139
+ ];
140
+ if state == "NEW" && self.protocol == "tcp" {
141
+ args.extend(["--tcp-flags","FIN,SYN,RST,ACK","SYN"].map(String::from));
142
+ }
143
+ let (address, port) = if self.source { ("--ctorigsrc", "--ctorigsrcport") } else { ("--ctorigdst", "--ctorigdstport") };
144
+ args.extend(["-m","conntrack","--ctstate",if reply {"ESTABLISHED"} else {state},"--ctproto",if self.protocol == "tcp" {"6"} else {"17"},address].map(String::from));
145
+ // Canonical xtables host form: explicit /32 has different unused
146
+ // conntrack mask bytes and does not survive a save/restore round trip.
147
+ args.push(self.tracked.address.into());
148
+ args.push(port.into());args.push(port_range(self.tracked.first,self.tracked.last));
149
+ args.extend(["--ctdir",if reply {"REPLY"} else {"ORIGINAL"}].map(String::from));
150
+ push(args)?;
151
+ }
152
+ Ok(())
153
+ }
154
+ }
102
155
  impl Prepared {
103
156
  pub fn validate(&self) -> Result<()> {
104
157
  if self.policy.clone().prepare(self.backend.clone())? != *self { return Err(Error::Invalid); }
105
158
  Ok(())
106
159
  }
160
+ /// Every forwarded flow the barrier admits, from the scope's one forwarding
161
+ /// enumeration: leased ranges first, then each publication's inbound flows
162
+ /// and, when symmetric, the flows it opens from its target ports.
163
+ fn flows(&self) -> Result<Vec<Flow<'_>>> {
164
+ let scope = self.policy.scope()?;
165
+ let forwarding = scope.forwarding()?;
166
+ let uplink = scope.uplink.interface_name.as_str();
167
+ let mut flows = Vec::new();
168
+ for leased in &forwarding.leased {
169
+ let tuple = Tuple { address:&leased.allocation.transit_source_address, first:leased.range.first, last:leased.range.last };
170
+ flows.push(Flow { input:&leased.handoff.link.interface_name, output:uplink, protocol:&leased.range.protocol,
171
+ source:true, live:tuple, tracked:tuple });
172
+ }
173
+ for published in &forwarding.published {
174
+ let port = published.port;
175
+ let inside = Tuple { address:&port.target_address, first:port.target_port, last:port.target_last() };
176
+ let outside = Tuple { address:&port.host_ip, first:port.host_port, last:port.host_last() };
177
+ let link = published.link.interface_name.as_str();
178
+ flows.push(Flow { input:uplink, output:link, protocol:&port.protocol, source:false, live:inside, tracked:outside });
179
+ if port.symmetric {
180
+ flows.push(Flow { input:link, output:uplink, protocol:&port.protocol, source:true, live:inside, tracked:inside });
181
+ }
182
+ }
183
+ Ok(flows)
184
+ }
107
185
  pub fn rules(&self, options: &Options) -> Result<Vec<Rule>> {
108
186
  options.validate()?;
109
- let scope = self.policy.scope()?;
187
+ let digest = self.digest.strip_prefix("sha256:").ok_or(Error::Invalid)?;
110
188
  let mut result = Vec::new();
111
189
  let mut bytes = 0;
112
- for handoff in &scope.handoffs {
113
- for allocation in &handoff.allocations {
114
- for range in &allocation.source_port_ranges {
115
- for state in ["NEW", "ESTABLISHED", "REPLY"] {
116
- let reply = state == "REPLY";
117
- let mut args: Vec<String> = vec![
118
- "-i".into(), if reply { scope.uplink.interface_name.clone() } else { handoff.link.interface_name.clone() },
119
- "-o".into(), if reply { handoff.link.interface_name.clone() } else { scope.uplink.interface_name.clone() },
120
- if reply { "-d" } else { "-s" }.into(), format!("{}/32", allocation.transit_source_address),
121
- "-p".into(),range.protocol.clone(),"-m".into(),range.protocol.clone(),
122
- if reply { "--dport" } else { "--sport" }.into(),port_range(range.first,range.last),
123
- ];
124
- if state == "NEW" && range.protocol == "tcp" {
125
- args.extend(["--tcp-flags","FIN,SYN,RST,ACK","SYN"].map(String::from));
126
- }
127
- args.extend(["-m","conntrack","--ctstate",if reply {"ESTABLISHED"} else {state},"--ctproto",if range.protocol == "tcp" {"6"} else {"17"},"--ctorigsrc"].map(String::from));
128
- // Canonical xtables host form: explicit /32 has different unused
129
- // conntrack mask bytes and does not survive a save/restore round trip.
130
- args.push(allocation.transit_source_address.clone());
131
- args.push("--ctorigsrcport".into());args.push(port_range(range.first,range.last));
132
- args.extend(["--ctdir",if reply {"REPLY"} else {"ORIGINAL"},"-m","comment","--comment"].map(String::from));
133
- let comment = format!("snftd1:{}:{}:{}:{}",options.owner_id,options.instance_id,self.digest.strip_prefix("sha256:").ok_or(Error::Invalid)?,result.len());
134
- args.push(comment.clone());args.extend(["-j","ACCEPT"].map(String::from));
135
- bytes += args.iter().map(|arg|arg.len()+1).sum::<usize>()+32;
136
- crate::capacity("restoreRules", 192, result.len() + 1)?;
137
- crate::capacity("restoreBytes", 100_000, bytes)?;
138
- result.push(Rule {comment,args});
139
- }
140
- }
141
- }
190
+ for flow in self.flows()? {
191
+ flow.rules(|mut args| {
192
+ let comment = format!("snftd1:{}:{}:{}:{}",options.owner_id,options.instance_id,digest,result.len());
193
+ args.extend(["-m","comment","--comment"].map(String::from));
194
+ args.push(comment.clone());args.extend(["-j","ACCEPT"].map(String::from));
195
+ bytes += args.iter().map(|arg|arg.len()+1).sum::<usize>()+32;
196
+ crate::capacity("restoreRules", 192, result.len() + 1)?;
197
+ crate::capacity("restoreBytes", 100_000, bytes)?;
198
+ result.push(Rule {comment,args});
199
+ Ok(())
200
+ })?;
142
201
  }
143
202
  Ok(result)
144
203
  }
@@ -71,13 +71,38 @@ impl Owner {
71
71
  || expected.receipt.backend!=expected.prepared.backend {return Err(Error::Conflict);}
72
72
  Ok(())
73
73
  }
74
- fn fence(target: &Prepared) -> Result<()> {
75
- let scope=target.policy.scope()?;
76
- wire::verify_bound_interfaces_down(scope.handoffs.iter().map(|handoff| {
77
- let link=&handoff.link;
78
- wire::BoundInterface {index:link.interface_index,name:&link.interface_name,kind:&link.interface_kind,
79
- mac:link.mac_address.as_deref(),link_index:link.interface_link_index,addresses:&link.required_ipv4_addresses}
80
- }))
74
+ fn handoffs(prepared: &Prepared) -> Result<Vec<&crate::egress::LocalLink>> {
75
+ Ok(prepared.policy.scope()?.handoffs.iter().map(|handoff|&handoff.link).collect())
76
+ }
77
+ fn bound(link: &crate::egress::LocalLink) -> wire::BoundInterface<'_> {
78
+ wire::BoundInterface {index:link.interface_index,name:&link.interface_name,kind:&link.interface_kind,
79
+ mac:link.mac_address.as_deref(),link_index:link.interface_link_index,addresses:&link.required_ipv4_addresses}
80
+ }
81
+ /// The handoff links of a transition: those both generations bind exactly,
82
+ /// and those it adds or drops.
83
+ fn fenced<'a>(previous: Option<&'a Prepared>, target: &'a Prepared) -> Result<(Vec<&'a crate::egress::LocalLink>, Vec<&'a crate::egress::LocalLink>)> {
84
+ let previous=previous.map(Self::handoffs).transpose()?.unwrap_or_default();
85
+ let target=Self::handoffs(target)?;
86
+ let kept=target.iter().copied().filter(|link|previous.contains(link)).collect();
87
+ let changed=previous.iter().copied().filter(|link|!target.contains(link))
88
+ .chain(target.iter().copied().filter(|link|!previous.contains(link))).collect();
89
+ Ok((kept,changed))
90
+ }
91
+ /// Reconcile fence, before and after the single restore transaction. A handoff
92
+ /// link both generations bind exactly may carry packets: the transaction swaps
93
+ /// the owned block atomically, so a flow both contributions admit passes
94
+ /// throughout, and the referenced barrier already is the target. Every link
95
+ /// the transition adds or drops must be down.
96
+ fn fence(previous: Option<&Prepared>, target: &Prepared) -> Result<()> {
97
+ let (kept,changed)=Self::fenced(previous,target)?;
98
+ wire::verify_bound_interfaces(kept.into_iter().map(Self::bound))?;
99
+ wire::verify_bound_interfaces_down(changed.into_iter().map(Self::bound))
100
+ }
101
+ /// Cleanup fence: deletion only withdraws admission. A handoff link that no
102
+ /// longer exists, such as a veth that left with its router namespace, counts
103
+ /// as down; one that exists must still be the exact bound link, and down.
104
+ fn cleanup_fence(prepared: &Prepared) -> Result<()> {
105
+ wire::verify_bound_interfaces_down_or_absent(Self::handoffs(prepared)?.into_iter().map(Self::bound))
81
106
  }
82
107
  fn applied(&self, prepared: Prepared, backend: String) -> Applied {
83
108
  Applied {receipt:Receipt {identity:self.identity.clone(),backend,revision:prepared.policy.revision,digest:prepared.digest.clone()},prepared}
@@ -102,15 +127,14 @@ impl Owner {
102
127
  return Ok(self.applied(transition.target.clone(),snapshot.backend));
103
128
  }
104
129
  if !snapshot.verified_matches(&self.options,&previous_rules,deadline)? {return Err(Error::Conflict);}
105
- if let Some(previous)=&transition.previous {Self::fence(&previous.prepared)?;}
106
- Self::fence(&transition.target)?;
130
+ let previous=transition.previous.as_ref().map(|v|&v.prepared);
131
+ Self::fence(previous,&transition.target)?;
107
132
  frontend::replace(&previous_rules,&target_rules,deadline)?;
108
133
  self.context()?;
109
134
  let current=frontend::Snapshot::read(deadline)?;
110
135
  if current.backend!=snapshot.backend || !current.verified_matches(&self.options,&target_rules,deadline)? {return Err(Error::Conflict);}
111
136
  owner::Owner::verify_external(&transition.target.policy.barrier)?;
112
- if let Some(previous)=&transition.previous {Self::fence(&previous.prepared)?;}
113
- Self::fence(&transition.target)?;
137
+ Self::fence(previous,&transition.target)?;
114
138
  Ok(self.applied(transition.target.clone(),current.backend))
115
139
  })();
116
140
  match result {
@@ -141,18 +165,25 @@ impl Owner {
141
165
  if current.verified_matches(&self.options,&[],deadline)? {return Ok(());}
142
166
  let rules=expected.prepared.rules(&self.options)?;
143
167
  if !current.verified_matches(&self.options,&rules,deadline)? {return Err(Error::Conflict);}
144
- Self::fence(&expected.prepared)?;
168
+ Self::cleanup_fence(&expected.prepared)?;
145
169
  frontend::replace(&rules,&[],deadline)?;
146
170
  self.context()?;
147
171
  let after=frontend::Snapshot::read(deadline)?;
148
172
  if after.backend!=current.backend || !after.verified_matches(&self.options,&[],deadline)? {return Err(Error::Conflict);}
149
- Self::fence(&expected.prepared)
173
+ Self::cleanup_fence(&expected.prepared)
150
174
  })();
151
175
  match result {Ok(())=>{self.applied=Some(expected);self.error=None;self.released=true;Ok(())},Err(error)=>{self.error=Some(error);Err(error)}}
152
176
  }
153
177
  fn close(&mut self, deadline: std::time::Instant) -> Result<()> {
178
+ self.context()?;
179
+ // Holding no rule is complete cleanup, whatever was refused before: the
180
+ // same generation-consistent read proves it without Docker's forwarding
181
+ // path. A read that fails is that read's error, never cleanup.
182
+ if frontend::Snapshot::observe(deadline)?.owns_nothing(&self.options)? {
183
+ self.pending=None;self.released=true;self.error=None;
184
+ return Ok(());
185
+ }
154
186
  if let Some(pending)=self.pending.clone() {
155
- self.context()?;
156
187
  let current=frontend::Snapshot::read(deadline)?;
157
188
  if current.backend!=pending.target.backend {return Err(Error::Conflict);}
158
189
  let known=if current.verified_matches(&self.options,&pending.target.rules(&self.options)?,deadline)? {
@@ -163,12 +194,9 @@ impl Owner {
163
194
  } else if current.verified_matches(&self.options,&[],deadline)? {None} else {return Err(Error::Conflict);};
164
195
  self.applied=known;self.pending=None;
165
196
  }
166
- if let Some(applied)=self.applied.clone() {self.release(applied,deadline)?;} else {
167
- self.context()?;
168
- if !frontend::Snapshot::read(deadline)?.verified_matches(&self.options,&[],deadline)? {return Err(Error::Conflict);}
169
- self.released=true;self.error=None;
170
- }
171
- Ok(())
197
+ // Rules of this owner without a receipt for them are never deleted.
198
+ let applied=self.applied.clone().ok_or(Error::Conflict)?;
199
+ self.release(applied,deadline)
172
200
  }
173
201
  }
174
202
  pub fn dispatch(owner: &mut Option<Owner>, request: Request) -> Result<serde_json::Value> {