@push.rocks/smartnftables 4.1.0 → 4.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/changelog.md +23 -0
- package/dist_rust/smartnftables_linux_amd64_musl +0 -0
- package/dist_rust/smartnftables_linux_amd64_musl.tsrust-build.json +5 -5
- package/dist_rust/smartnftables_linux_arm64_musl +0 -0
- package/dist_rust/smartnftables_linux_arm64_musl.tsrust-build.json +5 -5
- package/dist_ts/00_commitinfo_data.js +1 -1
- package/dist_ts/classes.manageddockerforwarding.d.ts +4 -1
- package/dist_ts/classes.manageddockerforwarding.js +8 -5
- package/dist_ts/classes.managednftables.d.ts +8 -2
- package/dist_ts/classes.managednftables.js +11 -6
- package/dist_ts/forwardhooks.d.ts +8 -0
- package/dist_ts/forwardhooks.js +42 -0
- package/dist_ts/index.d.ts +2 -0
- package/dist_ts/index.js +2 -1
- package/dist_ts/managed.docker.types.d.ts +2 -1
- package/dist_ts/managed.forwardhooks.types.d.ts +35 -0
- package/dist_ts/managed.forwardhooks.types.js +2 -0
- package/package.json +4 -4
- package/readme.md +80 -17
- package/rust/src/docker.frontend.rs +47 -9
- package/rust/src/docker.graph.rs +53 -14
- package/rust/src/docker.policy.rs +90 -31
- package/rust/src/docker.rs +48 -20
- package/rust/src/docker_tests.rs +206 -2
- package/rust/src/egress.host.rs +44 -45
- package/rust/src/egress.rs +56 -0
- package/rust/src/forwardhooks.rs +176 -0
- package/rust/src/forwardhooks_tests.rs +136 -0
- package/rust/src/main.rs +12 -8
- package/rust/src/wire.links.rs +28 -6
- package/rust/src/wire.rs +22 -2
- package/ts/00_commitinfo_data.ts +1 -1
- package/ts/classes.manageddockerforwarding.ts +7 -4
- package/ts/classes.managednftables.ts +13 -5
- package/ts/forwardhooks.ts +39 -0
- package/ts/index.ts +2 -0
- package/ts/managed.docker.types.ts +2 -1
- package/ts/managed.forwardhooks.types.ts +26 -0
package/readme.md
CHANGED
|
@@ -297,7 +297,9 @@ endpoint, addresses per link, protected prefixes, platform endpoints, active
|
|
|
297
297
|
generations, handoffs and allocations, ranges per allocation, guarded pools, the
|
|
298
298
|
IPC and capture limits) describes the shape of the topology and stays `INVALID`.
|
|
299
299
|
Other codes, and the facade's own local rejections, carry neither `reason` nor
|
|
300
|
-
`details`. A native refusal outside this shape is `PROTOCOL`.
|
|
300
|
+
`details`. A native refusal outside this shape is `PROTOCOL`. A failed `close()`
|
|
301
|
+
rejects `CLEANUP_UNCONFIRMED`, whose `cause` is the `ManagedNftablesError` that left
|
|
302
|
+
cleanup unconfirmed, with the native code when the native owner refused.
|
|
301
303
|
|
|
302
304
|
```typescript
|
|
303
305
|
try {
|
|
@@ -452,8 +454,8 @@ under host grants. 64 symmetric ranges take about 19,600 element bytes.
|
|
|
452
454
|
The caller still owns the route to `targetAddress` through that handoff, the
|
|
453
455
|
workload listener, and every other packet owner on the host. As with leased egress,
|
|
454
456
|
an ACCEPT here cannot override Docker's independent FORWARD DROP; the
|
|
455
|
-
`ManagedDockerForwarding` contribution below
|
|
456
|
-
the barrier
|
|
457
|
+
`ManagedDockerForwarding` contribution below admits every inbound and symmetric
|
|
458
|
+
publication of the barrier through it, with the leased flows.
|
|
457
459
|
|
|
458
460
|
Each router generation binds an immutable lease reference, transit source address,
|
|
459
461
|
leased TCP/UDP source-port ranges, a nonzero conntrack zone and a 16-byte label.
|
|
@@ -693,7 +695,8 @@ forwarding that another owner (Docker, a container or VM bridge, a VPN router, a
|
|
|
693
695
|
second routing daemon) accepts, and accepting in this table still cannot override
|
|
694
696
|
another owner's drop. Use it only on a host where this table is the sole forwarding
|
|
695
697
|
owner; on a Docker host keep it absent and use the Docker contribution below, which
|
|
696
|
-
refuses an exclusive barrier.
|
|
698
|
+
refuses an exclusive barrier. `inspectForwardHooks()`, described below, reports
|
|
699
|
+
the other forward owners of the namespace for that decision. The
|
|
697
700
|
member does not fence flowtable offload, packet queues, proxies or anything outside
|
|
698
701
|
the forward hook.
|
|
699
702
|
|
|
@@ -880,7 +883,7 @@ drainage, conntrack cleanup, elapsed-time expiry, or safe address/port/zone reus
|
|
|
880
883
|
|
|
881
884
|
### Docker forwarding contribution
|
|
882
885
|
|
|
883
|
-
`ManagedDockerForwarding` admits the
|
|
886
|
+
`ManagedDockerForwarding` admits exactly the forwarded flows of an exact applied
|
|
884
887
|
schema-v2 `hostTransit` receipt through Docker's existing IPv4 forwarding path.
|
|
885
888
|
It reads and verifies that dedicated barrier's complete table and local links,
|
|
886
889
|
without adopting its ownership. The contribution uses Docker's supported
|
|
@@ -916,32 +919,92 @@ and never changes a host forwarding policy. `prepare()` refuses a barrier with
|
|
|
916
919
|
`exclusiveForwarding` as `INVALID`: its drop would also deny Docker's own container
|
|
917
920
|
forwarding.
|
|
918
921
|
|
|
919
|
-
|
|
920
|
-
|
|
921
|
-
|
|
922
|
-
|
|
923
|
-
|
|
924
|
-
|
|
925
|
-
|
|
922
|
+
The contribution and the barrier's forward chain derive from one enumeration of
|
|
923
|
+
the scope's forwarded flows, so the contribution admits exactly the flow kinds the
|
|
924
|
+
non-exclusive barrier admits, each with its replies:
|
|
925
|
+
|
|
926
|
+
| Flow kind | Original direction | Live tuple | Conntrack original tuple |
|
|
927
|
+
| --- | --- | --- | --- |
|
|
928
|
+
| Leased range | handoff → uplink | transit source address and leased port range | same source |
|
|
929
|
+
| Inbound publication | uplink → handoff, after the barrier's DNAT | inside target address and port(s) as destination | published `hostIp` and port(s) as destination |
|
|
930
|
+
| Symmetric publication | handoff → uplink, before the barrier's SNAT | target address and port(s) as source | same source |
|
|
931
|
+
|
|
932
|
+
Every flow kind has three rules: the NEW opening (TCP only with SYN and
|
|
933
|
+
FIN/RST/ACK clear), the ESTABLISHED original direction and the ESTABLISHED reply
|
|
934
|
+
with swapped interfaces and tuple sides. A router's own egress, such as a VPN client
|
|
935
|
+
in the router namespace dialling its hub, reaches the host as a leased flow: the
|
|
936
|
+
router translates it to the transit address and a leased port range, so it is
|
|
937
|
+
admitted by the leased rules exactly when the barrier admits it. Host grants and
|
|
938
|
+
host-local platform endpoints use INPUT/OUTPUT and never cross FORWARD. Each rule
|
|
939
|
+
binds the handoff and uplink names. The exact v2 host barrier independently checks
|
|
940
|
+
interface indices, MAC/iflink facts, the default host conntrack zone, protected
|
|
941
|
+
original/current destinations and explicit outer SNAT; it governs every forwarded
|
|
942
|
+
packet on its handoffs, so the host forwards the intersection, which is the
|
|
943
|
+
barrier's admitted set. A scope without publications keeps its exact 4.1.0 rules,
|
|
944
|
+
comments and graphs. A complete contribution is bounded to 192 rules and 100,000
|
|
945
|
+
generated command bytes: three rules per leased range and per publication, and
|
|
946
|
+
three more per symmetric publication. A larger scope is refused as `EXHAUSTED`
|
|
947
|
+
(`restoreRules`/`restoreBytes`) before any subprocess.
|
|
926
948
|
|
|
927
949
|
Create the class with `ownerId`, a caller-retained `instanceId` of at least 16
|
|
928
950
|
identifier characters, and optional `binaryPath`/`networkNamespaceFd`. After
|
|
929
951
|
`start()`, call `prepare({ schemaVersion: 1, revision, barrier })`, where `barrier`
|
|
930
952
|
is the exact applied host policy. Persist the complete `{ previous, target }`
|
|
931
|
-
transition before `reconcile()`.
|
|
932
|
-
|
|
933
|
-
|
|
934
|
-
|
|
953
|
+
transition before `reconcile()`. The native owner fences the handoff links before
|
|
954
|
+
and after its single no-flush restore transaction, which deletes the previous block
|
|
955
|
+
and inserts the target block in one atomic nf_tables commit. A handoff link that
|
|
956
|
+
`previous` and `target` both bind exactly may stay up, so a publication or lease
|
|
957
|
+
amendment of a running generation applies in place: every flow both contributions
|
|
958
|
+
admit passes throughout, a flow only the target admits passes from the commit, and
|
|
959
|
+
the referenced barrier must already be the target. Every handoff link the
|
|
960
|
+
transition adds or drops, and every link of a first apply, must be down; retain
|
|
961
|
+
those exact links until cleanup finishes. Exact target replay can recover a lost
|
|
962
|
+
response without rewriting live rules.
|
|
935
963
|
|
|
936
964
|
`inspect().present` confirms the contribution and its referenced barrier, not
|
|
937
965
|
the complete host packet path or durable controller authority. Errors retain
|
|
938
966
|
failed-owned state and pending intent. `release(applied)` and `close()` delete
|
|
939
|
-
only exact contributed rules
|
|
967
|
+
only exact contributed rules. `close()` first reads the filter table through the
|
|
968
|
+
same generation-consistent frontend and netlink read, without requiring Docker's
|
|
969
|
+
forwarding path: when no saved rule in any chain and no native DOCKER-USER or
|
|
970
|
+
FORWARD rule carries this owner's comment prefix, cleanup is confirmed. So an owner
|
|
971
|
+
whose first apply was refused, for example on a host with FORWARD policy ACCEPT or
|
|
972
|
+
without Docker, closes on that host. A read that fails rejects
|
|
973
|
+
`CLEANUP_UNCONFIRMED` whose `cause` carries the native code (`UNAVAILABLE`,
|
|
974
|
+
`UNSUPPORTED_BACKEND`, `PERMISSION`, `CONFLICT`); it is never reported as cleanup.
|
|
975
|
+
The fence of `release(applied)` and `close()` requires every handoff link down or gone:
|
|
976
|
+
a link whose index no link holds any more, such as a veth that left with its router
|
|
977
|
+
namespace when the node process ended, counts as down, so a restarted process in
|
|
978
|
+
the same boot can release the previous contribution before applying the next one.
|
|
979
|
+
A different link at a bound index still rejects. A cold boot needs a new
|
|
940
980
|
barrier receipt and caller-authorized replay; old boot receipts cannot authorize
|
|
941
981
|
native operations. The caller owns restart ordering, disjoint retained leases,
|
|
942
982
|
exclusive privileged mutation authority and activation. The frontend transaction
|
|
943
983
|
does not provide compare-and-swap against other privileged processes.
|
|
944
984
|
|
|
985
|
+
### Forward hook owners
|
|
986
|
+
|
|
987
|
+
`inspectForwardHooks({ binaryPath?, networkNamespaceFd? })` reads every
|
|
988
|
+
packet-filter owner on the routed forward path of one network namespace and
|
|
989
|
+
changes nothing. One native process takes the read and exits:
|
|
990
|
+
|
|
991
|
+
- `chains`: every nf_tables base chain on the forward hook of the `ip`, `ip6` and
|
|
992
|
+
`inet` families, with `family`, `table`, `tableHandle`, `chain`, `type`,
|
|
993
|
+
`priority` and `policy`, sorted by family, table and chain. Managed tables are
|
|
994
|
+
included; compare `table` and `tableHandle` with a receipt's
|
|
995
|
+
`identity.tableName` and `tableHandle` to find your own.
|
|
996
|
+
- `docker`: whether Docker's iptables-nft `DOCKER-USER` and `DOCKER-FORWARD`
|
|
997
|
+
chains exist in table `ip filter`.
|
|
998
|
+
- `legacyTables`: the registered legacy xtables `ipv4` and `ipv6` tables. Their
|
|
999
|
+
hooks, including FORWARD, are not nf_tables objects and never appear in `chains`.
|
|
1000
|
+
|
|
1001
|
+
The table and chain dumps share one ruleset generation, or the read is
|
|
1002
|
+
`CONFLICT`. Without `CAP_NET_ADMIN` in the namespace it is `PERMISSION`. Bridge
|
|
1003
|
+
and netdev hooks do not see routed packets and are not reported. Guard
|
|
1004
|
+
`exclusiveForwarding` with it: refuse while any chain other than your own barrier's,
|
|
1005
|
+
either Docker chain or any legacy table is present. The read is a point in time;
|
|
1006
|
+
the caller still owns exclusive authority over who may add a forward owner later.
|
|
1007
|
+
|
|
945
1008
|
### Native qualification
|
|
946
1009
|
|
|
947
1010
|
The separate [Docker fixture](test/native/readme.docker.md) exercises the native
|
|
@@ -113,10 +113,10 @@ fn normalized(args: &[String]) -> Result<Vec<(String, Vec<String>)>> {
|
|
|
113
113
|
let mut values=BTreeMap::new();let mut modules=BTreeSet::new();let mut index=0;
|
|
114
114
|
while index<args.len() {
|
|
115
115
|
let key=&args[index];let count=if key=="--tcp-flags" {2} else {1};
|
|
116
|
-
if !["-i","-o","-s","-d","-p","-m","--sport","--dport","--tcp-flags","--ctstate","--ctproto","--ctorigsrc","--ctorigsrcport","--ctdir","--comment","-j"].contains(&key.as_str()) {return Err(Error::Conflict);}
|
|
116
|
+
if !["-i","-o","-s","-d","-p","-m","--sport","--dport","--tcp-flags","--ctstate","--ctproto","--ctorigsrc","--ctorigsrcport","--ctorigdst","--ctorigdstport","--ctdir","--comment","-j"].contains(&key.as_str()) {return Err(Error::Conflict);}
|
|
117
117
|
let mut value=args.get(index+1..index+1+count).ok_or(Error::Conflict)?.to_vec();index+=1+count;
|
|
118
118
|
if key=="-m" {if !modules.insert(value[0].clone()) {return Err(Error::Conflict);}continue;}
|
|
119
|
-
if key=="--ctorigsrc" && !value[0].contains('/') {value[0].push_str("/32");}
|
|
119
|
+
if (key=="--ctorigsrc" || key=="--ctorigdst") && !value[0].contains('/') {value[0].push_str("/32");}
|
|
120
120
|
if key=="--ctproto" {value[0]=match value[0].as_str() {"tcp"=>"6".into(),"udp"=>"17".into(),_=>value[0].clone()};}
|
|
121
121
|
if values.insert(key.clone(),value).is_some() {return Err(Error::Conflict);}
|
|
122
122
|
}
|
|
@@ -125,19 +125,39 @@ fn normalized(args: &[String]) -> Result<Vec<(String, Vec<String>)>> {
|
|
|
125
125
|
}
|
|
126
126
|
pub struct Snapshot {
|
|
127
127
|
pub backend: String,
|
|
128
|
+
chains: BTreeMap<String, String>,
|
|
128
129
|
pub rows: Vec<Vec<String>>,
|
|
129
130
|
pub user: Vec<Vec<String>>,
|
|
130
131
|
raw: Option<super::graph::Snapshot>,
|
|
131
132
|
}
|
|
132
133
|
impl Snapshot {
|
|
134
|
+
/// Docker's admitted forwarding path. Apply, inspection and every deletion
|
|
135
|
+
/// proof read through it.
|
|
133
136
|
pub fn read(deadline: Instant) -> Result<Self> {
|
|
137
|
+
let snapshot=Self::observe(deadline)?;
|
|
138
|
+
snapshot.path()?;
|
|
139
|
+
snapshot.raw.as_ref().ok_or(Error::Conflict)?.path()?;
|
|
140
|
+
Ok(snapshot)
|
|
141
|
+
}
|
|
142
|
+
/// The same generation-consistent read of the filter table, without
|
|
143
|
+
/// requiring Docker's forwarding path. It proves only whether this owner
|
|
144
|
+
/// holds any rule there, never a placement or a deletion.
|
|
145
|
+
pub fn observe(deadline: Instant) -> Result<Self> {
|
|
134
146
|
let backend=version(deadline)?;
|
|
135
147
|
let (raw,text)=super::graph::Snapshot::read(||command(SAVE,&["-t","filter"],&[],deadline))?;
|
|
136
|
-
let mut snapshot=Self::
|
|
148
|
+
let mut snapshot=Self::table(backend,&text)?;
|
|
137
149
|
snapshot.raw=Some(raw);
|
|
138
150
|
Ok(snapshot)
|
|
139
151
|
}
|
|
152
|
+
/// The text half of `read`.
|
|
153
|
+
#[cfg(test)]
|
|
140
154
|
pub fn parse(backend: String, text: &str) -> Result<Self> {
|
|
155
|
+
let snapshot=Self::table(backend,text)?;
|
|
156
|
+
snapshot.path()?;
|
|
157
|
+
Ok(snapshot)
|
|
158
|
+
}
|
|
159
|
+
/// One complete saved filter table: its chains and rules, in order.
|
|
160
|
+
pub(super) fn table(backend: String, text: &str) -> Result<Self> {
|
|
141
161
|
if text.len()>MAX_OUTPUT {return Err(Error::Protocol);}
|
|
142
162
|
let mut table=false;let mut committed=false;let mut chains=BTreeMap::new();let mut rows=Vec::new();
|
|
143
163
|
for line in text.lines().filter(|line|!line.starts_with('#')&&!line.is_empty()) {
|
|
@@ -152,15 +172,33 @@ impl Snapshot {
|
|
|
152
172
|
rows.push(row);
|
|
153
173
|
}
|
|
154
174
|
}
|
|
155
|
-
if !committed
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
175
|
+
if !committed {return Err(Error::Conflict);}
|
|
176
|
+
let user=rows.iter().filter(|r|r[1]=="DOCKER-USER").cloned().collect();
|
|
177
|
+
Ok(Self {backend,chains,rows,user,raw:None})
|
|
178
|
+
}
|
|
179
|
+
/// FORWARD has policy DROP and dispatches first to DOCKER-USER, then to
|
|
180
|
+
/// DOCKER-FORWARD.
|
|
181
|
+
fn path(&self) -> Result<()> {
|
|
182
|
+
if self.chains.get("FORWARD").map(String::as_str)!=Some("DROP")
|
|
183
|
+
|| self.chains.get("DOCKER-USER").map(String::as_str)!=Some("-")
|
|
184
|
+
|| self.chains.get("DOCKER-FORWARD").map(String::as_str)!=Some("-") {return Err(Error::Conflict);}
|
|
185
|
+
let forward:Vec<_>=self.rows.iter().filter(|r|r[1]=="FORWARD").collect();
|
|
159
186
|
for (index,target) in ["DOCKER-USER","DOCKER-FORWARD"].iter().enumerate() {
|
|
160
187
|
if forward.get(index).map(|r|r.iter().map(String::as_str).collect::<Vec<_>>())!=Some(vec!["-A","FORWARD","-j",target]) {return Err(Error::Conflict);}
|
|
161
188
|
}
|
|
162
|
-
|
|
163
|
-
|
|
189
|
+
Ok(())
|
|
190
|
+
}
|
|
191
|
+
/// No rule of this owner exists in the filter table: neither saved text in
|
|
192
|
+
/// any chain nor native DOCKER-USER/FORWARD attributes carry its comment
|
|
193
|
+
/// prefix. Rules of other owners are not this owner's to prove.
|
|
194
|
+
pub fn owns_nothing(&self, options: &Options) -> Result<bool> {
|
|
195
|
+
let prefix=format!("snftd1:{}:",options.owner_id);
|
|
196
|
+
let raw=self.raw.as_ref().ok_or(Error::Conflict)?;
|
|
197
|
+
Ok(!self.mentions(&prefix) && !raw.mentions(prefix.as_bytes()))
|
|
198
|
+
}
|
|
199
|
+
/// Whether any saved rule, in any chain, carries a token starting with `prefix`.
|
|
200
|
+
pub(super) fn mentions(&self, prefix: &str) -> bool {
|
|
201
|
+
self.rows.iter().any(|r|r.iter().any(|t|t.starts_with(prefix)))
|
|
164
202
|
}
|
|
165
203
|
/// Text proves placement/count and logical policy; -C additionally checks
|
|
166
204
|
/// match-extension bytes hidden by save output. Deletion uses the same spec.
|
package/rust/src/docker.graph.rs
CHANGED
|
@@ -9,17 +9,36 @@ fn payload(base: u32, offset: u32, length: u32) -> Attr {
|
|
|
9
9
|
expr("payload",vec![Attr::u32(1,1),Attr::u32(2,base),Attr::u32(3,offset),Attr::u32(4,length)])
|
|
10
10
|
}
|
|
11
11
|
|
|
12
|
+
fn span(value: &str) -> Result<(u16, u16)> {
|
|
13
|
+
let (first,last)=value.split_once(':').unwrap_or((value,value));
|
|
14
|
+
Ok((first.parse::<u16>().map_err(|_|Error::Invalid)?,last.parse::<u16>().map_err(|_|Error::Invalid)?))
|
|
15
|
+
}
|
|
16
|
+
/// The one key of a rule among alternatives, with its value.
|
|
17
|
+
fn either<'a>(rule: &'a Rule, keys: [&'static str; 2]) -> Result<(usize, &'a str)> {
|
|
18
|
+
let mut found=keys.iter().enumerate().filter_map(|(index,key)|value(rule,key).ok().map(|value|(index,value)));
|
|
19
|
+
let result=found.next().ok_or(Error::Invalid)?;
|
|
20
|
+
if found.next().is_some() {return Err(Error::Invalid);}
|
|
21
|
+
Ok(result)
|
|
22
|
+
}
|
|
23
|
+
|
|
12
24
|
/// Fixed Linux UAPI / xtables 1.8.10 and 1.8.11 constructors, not a raw rule API.
|
|
13
25
|
pub(super) fn expected(rule: &Rule) -> Result<Vec<Attr>> {
|
|
14
26
|
let reply=value(rule,"--ctdir")?=="REPLY";
|
|
15
27
|
let new=value(rule,"--ctstate")?=="NEW";
|
|
16
28
|
let protocol=match value(rule,"-p")? {"tcp"=>6_u16,"udp"=>17,_=>return Err(Error::Invalid)};
|
|
17
|
-
|
|
18
|
-
let
|
|
19
|
-
let (
|
|
20
|
-
let
|
|
21
|
-
|
|
22
|
-
let
|
|
29
|
+
// The packet's own source (0) or destination (1) address and port.
|
|
30
|
+
let (side,live)=either(rule,["-s","-d"])?;
|
|
31
|
+
let live=live.strip_suffix("/32").ok_or(Error::Invalid)?.parse::<std::net::Ipv4Addr>().map_err(|_|Error::Invalid)?.octets();
|
|
32
|
+
let (port_side,ports)=either(rule,["--sport","--dport"])?;
|
|
33
|
+
if port_side!=side {return Err(Error::Invalid);}
|
|
34
|
+
let (first,last)=span(ports)?;
|
|
35
|
+
// The originally tracked source (0) or destination (1) address and port.
|
|
36
|
+
let (tracked_side,tracked)=either(rule,["--ctorigsrc","--ctorigdst"])?;
|
|
37
|
+
let tracked=tracked.parse::<std::net::Ipv4Addr>().map_err(|_|Error::Invalid)?.octets();
|
|
38
|
+
let (tracked_port_side,tracked_ports)=either(rule,["--ctorigsrcport","--ctorigdstport"])?;
|
|
39
|
+
if tracked_port_side!=tracked_side {return Err(Error::Invalid);}
|
|
40
|
+
let (tracked_first,tracked_last)=span(tracked_ports)?;
|
|
41
|
+
let mut result=vec![payload(1,if side==0 {12} else {16},4),compare(live.to_vec())];
|
|
23
42
|
for (key,argument) in [(6,"-i"),(7,"-o")] {
|
|
24
43
|
result.extend(meta(key,[value(rule,argument)?.as_bytes(),&[0]].concat()));
|
|
25
44
|
}
|
|
@@ -30,7 +49,7 @@ pub(super) fn expected(rule: &Rule) -> Result<Vec<Attr>> {
|
|
|
30
49
|
Attr::nested(4,vec![Attr::bytes(1,vec![0x17])]),Attr::nested(5,vec![Attr::bytes(1,vec![0])]),
|
|
31
50
|
]),compare(vec![2])]);
|
|
32
51
|
}
|
|
33
|
-
result.push(payload(2,if
|
|
52
|
+
result.push(payload(2,if side==0 {0} else {2},2));
|
|
34
53
|
result.push(if first==last {compare(first.to_be_bytes().to_vec())} else {
|
|
35
54
|
expr("range",vec![Attr::u32(1,1),Attr::u32(2,0),
|
|
36
55
|
Attr::nested(3,vec![Attr::bytes(1,first.to_be_bytes().to_vec())]),
|
|
@@ -38,11 +57,17 @@ pub(super) fn expected(rule: &Rule) -> Result<Vec<Attr>> {
|
|
|
38
57
|
});
|
|
39
58
|
// xt_conntrack_mtinfo3 is 164 bytes, XT_ALIGN'ed to 168 on both shipped
|
|
40
59
|
// 64-bit Linux targets. All fields and padding are initialized explicitly.
|
|
60
|
+
// The original source address/mask are at 0/16, its ports at 138/154; the
|
|
61
|
+
// original destination's at 32/48 and 140/156.
|
|
41
62
|
let mut conntrack=vec![0;168];
|
|
42
|
-
|
|
43
|
-
conntrack[
|
|
44
|
-
|
|
45
|
-
|
|
63
|
+
let address=32*tracked_side;
|
|
64
|
+
conntrack[address..address+4].copy_from_slice(&tracked);
|
|
65
|
+
conntrack[address+16..address+32].fill(0xff); // canonical bare-host nf_inet_addr mask
|
|
66
|
+
// STATE|PROTO|DIRECTION with ORIGSRC|ORIGSRC_PORT or ORIGDST|ORIGDST_PORT.
|
|
67
|
+
let flags=0x1003|if tracked_side==0 {0x0104} else {0x0208};
|
|
68
|
+
let port=2*tracked_side;
|
|
69
|
+
for (offset,number) in [(136,protocol),(138+port,tracked_first),(146,flags),
|
|
70
|
+
(148,if reply {0x1000} else {0}),(150,if new {8} else {2}),(154+port,tracked_last)] {
|
|
46
71
|
conntrack[offset..offset+2].copy_from_slice(&number.to_ne_bytes());
|
|
47
72
|
}
|
|
48
73
|
result.push(expr("match",vec![Attr::string(1,"conntrack"),Attr::u32(2,3),Attr::bytes(3,conntrack)]));
|
|
@@ -105,19 +130,33 @@ fn matches_graph(row: &[Attr], chain: &str, expected: &[Attr]) -> Result<bool> {
|
|
|
105
130
|
equivalent(&expressions,expected,true)
|
|
106
131
|
}
|
|
107
132
|
|
|
108
|
-
|
|
133
|
+
/// Whether any rule carries `marker` anywhere in its native attributes. An owner
|
|
134
|
+
/// comment is stored verbatim, whether the frontend can render the rule or not.
|
|
135
|
+
pub(super) fn mentions<'a>(rows: impl IntoIterator<Item=&'a Vec<Attr>>, marker: &[u8]) -> bool {
|
|
136
|
+
rows.into_iter().any(|row|wire::encode_attrs(row).windows(marker.len()).any(|window|window==marker))
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
pub struct Snapshot { generation: u32, rows: Vec<Vec<Attr>>, forward: Vec<Vec<Attr>> }
|
|
109
140
|
impl Snapshot {
|
|
141
|
+
/// The saved text and the native DOCKER-USER and FORWARD rules under one
|
|
142
|
+
/// unchanged ruleset generation. A missing table or chain dumps no rules.
|
|
110
143
|
pub fn read(save: impl FnOnce()->Result<String>) -> Result<(Self,String)> {
|
|
111
144
|
let mut socket=Socket::open()?;let generation=socket.generation()?;
|
|
112
145
|
let text=save()?;
|
|
113
146
|
let rows=socket.query_ipv4(7,vec![Attr::string(1,"filter"),Attr::string(2,"DOCKER-USER")],true)?;
|
|
114
147
|
let forward=socket.query_ipv4(7,vec![Attr::string(1,"filter"),Attr::string(2,"FORWARD")],true)?;
|
|
115
148
|
if socket.generation()!=Ok(generation) {return Err(Error::Conflict);}
|
|
149
|
+
Ok((Self {generation,rows,forward},text))
|
|
150
|
+
}
|
|
151
|
+
/// Docker's forwarding path: FORWARD's first two rules are its exact
|
|
152
|
+
/// unconditional jumps to DOCKER-USER and then DOCKER-FORWARD.
|
|
153
|
+
pub fn path(&self) -> Result<()> {
|
|
116
154
|
for (index,target) in ["DOCKER-USER","DOCKER-FORWARD"].iter().enumerate() {
|
|
117
|
-
if !matches_forward(forward.get(index).ok_or(Error::Conflict)?,target)? {return Err(Error::Conflict);}
|
|
155
|
+
if !matches_forward(self.forward.get(index).ok_or(Error::Conflict)?,target)? {return Err(Error::Conflict);}
|
|
118
156
|
}
|
|
119
|
-
Ok((
|
|
157
|
+
Ok(())
|
|
120
158
|
}
|
|
159
|
+
pub fn mentions(&self, marker: &[u8]) -> bool {mentions(self.rows.iter().chain(&self.forward),marker)}
|
|
121
160
|
pub fn matches(&self, expected: &[Rule]) -> Result<bool> {
|
|
122
161
|
if self.rows.len()<expected.len() {return Ok(false);}
|
|
123
162
|
for (row,rule) in self.rows.iter().zip(expected) {
|
|
@@ -99,46 +99,105 @@ pub struct Rule {
|
|
|
99
99
|
fn port_range(first: u16, last: u16) -> String {
|
|
100
100
|
if first == last { first.to_string() } else { format!("{first}:{last}") }
|
|
101
101
|
}
|
|
102
|
+
/// One side of a packet or of its originally tracked tuple: an exact address
|
|
103
|
+
/// and an inclusive port span.
|
|
104
|
+
#[derive(Clone, Copy)]
|
|
105
|
+
struct Tuple<'a> {
|
|
106
|
+
address: &'a str,
|
|
107
|
+
first: u16,
|
|
108
|
+
last: u16,
|
|
109
|
+
}
|
|
110
|
+
/// One flow class of the barrier's forward chain, in its original direction:
|
|
111
|
+
/// it enters on `input`, leaves on `output`, carries `live` as its source
|
|
112
|
+
/// (`source`) or destination, and conntrack tracked it with `tracked` as its
|
|
113
|
+
/// original source (`source`) or destination. A leased or symmetric flow is
|
|
114
|
+
/// keyed on its own source, before outer source NAT; an inbound publication on
|
|
115
|
+
/// its translated inside destination and its published outside destination.
|
|
116
|
+
struct Flow<'a> {
|
|
117
|
+
input: &'a str,
|
|
118
|
+
output: &'a str,
|
|
119
|
+
protocol: &'a str,
|
|
120
|
+
source: bool,
|
|
121
|
+
live: Tuple<'a>,
|
|
122
|
+
tracked: Tuple<'a>,
|
|
123
|
+
}
|
|
124
|
+
impl Flow<'_> {
|
|
125
|
+
/// The ESTABLISHED original direction, a NEW opening (TCP only with SYN and
|
|
126
|
+
/// FIN/RST/ACK clear) and the ESTABLISHED reply, as the barrier admits them.
|
|
127
|
+
fn rules(&self, mut push: impl FnMut(Vec<String>) -> Result<()>) -> Result<()> {
|
|
128
|
+
for state in ["NEW", "ESTABLISHED", "REPLY"] {
|
|
129
|
+
let reply = state == "REPLY";
|
|
130
|
+
// The reply swaps the interfaces and the packet's tuple sides; the
|
|
131
|
+
// originally tracked tuple stays the same.
|
|
132
|
+
let source = self.source != reply;
|
|
133
|
+
let mut args: Vec<String> = vec![
|
|
134
|
+
"-i".into(), if reply { self.output } else { self.input }.into(),
|
|
135
|
+
"-o".into(), if reply { self.input } else { self.output }.into(),
|
|
136
|
+
if source { "-s" } else { "-d" }.into(), format!("{}/32", self.live.address),
|
|
137
|
+
"-p".into(), self.protocol.into(), "-m".into(), self.protocol.into(),
|
|
138
|
+
if source { "--sport" } else { "--dport" }.into(), port_range(self.live.first, self.live.last),
|
|
139
|
+
];
|
|
140
|
+
if state == "NEW" && self.protocol == "tcp" {
|
|
141
|
+
args.extend(["--tcp-flags","FIN,SYN,RST,ACK","SYN"].map(String::from));
|
|
142
|
+
}
|
|
143
|
+
let (address, port) = if self.source { ("--ctorigsrc", "--ctorigsrcport") } else { ("--ctorigdst", "--ctorigdstport") };
|
|
144
|
+
args.extend(["-m","conntrack","--ctstate",if reply {"ESTABLISHED"} else {state},"--ctproto",if self.protocol == "tcp" {"6"} else {"17"},address].map(String::from));
|
|
145
|
+
// Canonical xtables host form: explicit /32 has different unused
|
|
146
|
+
// conntrack mask bytes and does not survive a save/restore round trip.
|
|
147
|
+
args.push(self.tracked.address.into());
|
|
148
|
+
args.push(port.into());args.push(port_range(self.tracked.first,self.tracked.last));
|
|
149
|
+
args.extend(["--ctdir",if reply {"REPLY"} else {"ORIGINAL"}].map(String::from));
|
|
150
|
+
push(args)?;
|
|
151
|
+
}
|
|
152
|
+
Ok(())
|
|
153
|
+
}
|
|
154
|
+
}
|
|
102
155
|
impl Prepared {
|
|
103
156
|
pub fn validate(&self) -> Result<()> {
|
|
104
157
|
if self.policy.clone().prepare(self.backend.clone())? != *self { return Err(Error::Invalid); }
|
|
105
158
|
Ok(())
|
|
106
159
|
}
|
|
160
|
+
/// Every forwarded flow the barrier admits, from the scope's one forwarding
|
|
161
|
+
/// enumeration: leased ranges first, then each publication's inbound flows
|
|
162
|
+
/// and, when symmetric, the flows it opens from its target ports.
|
|
163
|
+
fn flows(&self) -> Result<Vec<Flow<'_>>> {
|
|
164
|
+
let scope = self.policy.scope()?;
|
|
165
|
+
let forwarding = scope.forwarding()?;
|
|
166
|
+
let uplink = scope.uplink.interface_name.as_str();
|
|
167
|
+
let mut flows = Vec::new();
|
|
168
|
+
for leased in &forwarding.leased {
|
|
169
|
+
let tuple = Tuple { address:&leased.allocation.transit_source_address, first:leased.range.first, last:leased.range.last };
|
|
170
|
+
flows.push(Flow { input:&leased.handoff.link.interface_name, output:uplink, protocol:&leased.range.protocol,
|
|
171
|
+
source:true, live:tuple, tracked:tuple });
|
|
172
|
+
}
|
|
173
|
+
for published in &forwarding.published {
|
|
174
|
+
let port = published.port;
|
|
175
|
+
let inside = Tuple { address:&port.target_address, first:port.target_port, last:port.target_last() };
|
|
176
|
+
let outside = Tuple { address:&port.host_ip, first:port.host_port, last:port.host_last() };
|
|
177
|
+
let link = published.link.interface_name.as_str();
|
|
178
|
+
flows.push(Flow { input:uplink, output:link, protocol:&port.protocol, source:false, live:inside, tracked:outside });
|
|
179
|
+
if port.symmetric {
|
|
180
|
+
flows.push(Flow { input:link, output:uplink, protocol:&port.protocol, source:true, live:inside, tracked:inside });
|
|
181
|
+
}
|
|
182
|
+
}
|
|
183
|
+
Ok(flows)
|
|
184
|
+
}
|
|
107
185
|
pub fn rules(&self, options: &Options) -> Result<Vec<Rule>> {
|
|
108
186
|
options.validate()?;
|
|
109
|
-
let
|
|
187
|
+
let digest = self.digest.strip_prefix("sha256:").ok_or(Error::Invalid)?;
|
|
110
188
|
let mut result = Vec::new();
|
|
111
189
|
let mut bytes = 0;
|
|
112
|
-
for
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
];
|
|
124
|
-
if state == "NEW" && range.protocol == "tcp" {
|
|
125
|
-
args.extend(["--tcp-flags","FIN,SYN,RST,ACK","SYN"].map(String::from));
|
|
126
|
-
}
|
|
127
|
-
args.extend(["-m","conntrack","--ctstate",if reply {"ESTABLISHED"} else {state},"--ctproto",if range.protocol == "tcp" {"6"} else {"17"},"--ctorigsrc"].map(String::from));
|
|
128
|
-
// Canonical xtables host form: explicit /32 has different unused
|
|
129
|
-
// conntrack mask bytes and does not survive a save/restore round trip.
|
|
130
|
-
args.push(allocation.transit_source_address.clone());
|
|
131
|
-
args.push("--ctorigsrcport".into());args.push(port_range(range.first,range.last));
|
|
132
|
-
args.extend(["--ctdir",if reply {"REPLY"} else {"ORIGINAL"},"-m","comment","--comment"].map(String::from));
|
|
133
|
-
let comment = format!("snftd1:{}:{}:{}:{}",options.owner_id,options.instance_id,self.digest.strip_prefix("sha256:").ok_or(Error::Invalid)?,result.len());
|
|
134
|
-
args.push(comment.clone());args.extend(["-j","ACCEPT"].map(String::from));
|
|
135
|
-
bytes += args.iter().map(|arg|arg.len()+1).sum::<usize>()+32;
|
|
136
|
-
crate::capacity("restoreRules", 192, result.len() + 1)?;
|
|
137
|
-
crate::capacity("restoreBytes", 100_000, bytes)?;
|
|
138
|
-
result.push(Rule {comment,args});
|
|
139
|
-
}
|
|
140
|
-
}
|
|
141
|
-
}
|
|
190
|
+
for flow in self.flows()? {
|
|
191
|
+
flow.rules(|mut args| {
|
|
192
|
+
let comment = format!("snftd1:{}:{}:{}:{}",options.owner_id,options.instance_id,digest,result.len());
|
|
193
|
+
args.extend(["-m","comment","--comment"].map(String::from));
|
|
194
|
+
args.push(comment.clone());args.extend(["-j","ACCEPT"].map(String::from));
|
|
195
|
+
bytes += args.iter().map(|arg|arg.len()+1).sum::<usize>()+32;
|
|
196
|
+
crate::capacity("restoreRules", 192, result.len() + 1)?;
|
|
197
|
+
crate::capacity("restoreBytes", 100_000, bytes)?;
|
|
198
|
+
result.push(Rule {comment,args});
|
|
199
|
+
Ok(())
|
|
200
|
+
})?;
|
|
142
201
|
}
|
|
143
202
|
Ok(result)
|
|
144
203
|
}
|
package/rust/src/docker.rs
CHANGED
|
@@ -71,13 +71,38 @@ impl Owner {
|
|
|
71
71
|
|| expected.receipt.backend!=expected.prepared.backend {return Err(Error::Conflict);}
|
|
72
72
|
Ok(())
|
|
73
73
|
}
|
|
74
|
-
fn
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
74
|
+
fn handoffs(prepared: &Prepared) -> Result<Vec<&crate::egress::LocalLink>> {
|
|
75
|
+
Ok(prepared.policy.scope()?.handoffs.iter().map(|handoff|&handoff.link).collect())
|
|
76
|
+
}
|
|
77
|
+
fn bound(link: &crate::egress::LocalLink) -> wire::BoundInterface<'_> {
|
|
78
|
+
wire::BoundInterface {index:link.interface_index,name:&link.interface_name,kind:&link.interface_kind,
|
|
79
|
+
mac:link.mac_address.as_deref(),link_index:link.interface_link_index,addresses:&link.required_ipv4_addresses}
|
|
80
|
+
}
|
|
81
|
+
/// The handoff links of a transition: those both generations bind exactly,
|
|
82
|
+
/// and those it adds or drops.
|
|
83
|
+
fn fenced<'a>(previous: Option<&'a Prepared>, target: &'a Prepared) -> Result<(Vec<&'a crate::egress::LocalLink>, Vec<&'a crate::egress::LocalLink>)> {
|
|
84
|
+
let previous=previous.map(Self::handoffs).transpose()?.unwrap_or_default();
|
|
85
|
+
let target=Self::handoffs(target)?;
|
|
86
|
+
let kept=target.iter().copied().filter(|link|previous.contains(link)).collect();
|
|
87
|
+
let changed=previous.iter().copied().filter(|link|!target.contains(link))
|
|
88
|
+
.chain(target.iter().copied().filter(|link|!previous.contains(link))).collect();
|
|
89
|
+
Ok((kept,changed))
|
|
90
|
+
}
|
|
91
|
+
/// Reconcile fence, before and after the single restore transaction. A handoff
|
|
92
|
+
/// link both generations bind exactly may carry packets: the transaction swaps
|
|
93
|
+
/// the owned block atomically, so a flow both contributions admit passes
|
|
94
|
+
/// throughout, and the referenced barrier already is the target. Every link
|
|
95
|
+
/// the transition adds or drops must be down.
|
|
96
|
+
fn fence(previous: Option<&Prepared>, target: &Prepared) -> Result<()> {
|
|
97
|
+
let (kept,changed)=Self::fenced(previous,target)?;
|
|
98
|
+
wire::verify_bound_interfaces(kept.into_iter().map(Self::bound))?;
|
|
99
|
+
wire::verify_bound_interfaces_down(changed.into_iter().map(Self::bound))
|
|
100
|
+
}
|
|
101
|
+
/// Cleanup fence: deletion only withdraws admission. A handoff link that no
|
|
102
|
+
/// longer exists, such as a veth that left with its router namespace, counts
|
|
103
|
+
/// as down; one that exists must still be the exact bound link, and down.
|
|
104
|
+
fn cleanup_fence(prepared: &Prepared) -> Result<()> {
|
|
105
|
+
wire::verify_bound_interfaces_down_or_absent(Self::handoffs(prepared)?.into_iter().map(Self::bound))
|
|
81
106
|
}
|
|
82
107
|
fn applied(&self, prepared: Prepared, backend: String) -> Applied {
|
|
83
108
|
Applied {receipt:Receipt {identity:self.identity.clone(),backend,revision:prepared.policy.revision,digest:prepared.digest.clone()},prepared}
|
|
@@ -102,15 +127,14 @@ impl Owner {
|
|
|
102
127
|
return Ok(self.applied(transition.target.clone(),snapshot.backend));
|
|
103
128
|
}
|
|
104
129
|
if !snapshot.verified_matches(&self.options,&previous_rules,deadline)? {return Err(Error::Conflict);}
|
|
105
|
-
|
|
106
|
-
Self::fence(
|
|
130
|
+
let previous=transition.previous.as_ref().map(|v|&v.prepared);
|
|
131
|
+
Self::fence(previous,&transition.target)?;
|
|
107
132
|
frontend::replace(&previous_rules,&target_rules,deadline)?;
|
|
108
133
|
self.context()?;
|
|
109
134
|
let current=frontend::Snapshot::read(deadline)?;
|
|
110
135
|
if current.backend!=snapshot.backend || !current.verified_matches(&self.options,&target_rules,deadline)? {return Err(Error::Conflict);}
|
|
111
136
|
owner::Owner::verify_external(&transition.target.policy.barrier)?;
|
|
112
|
-
|
|
113
|
-
Self::fence(&transition.target)?;
|
|
137
|
+
Self::fence(previous,&transition.target)?;
|
|
114
138
|
Ok(self.applied(transition.target.clone(),current.backend))
|
|
115
139
|
})();
|
|
116
140
|
match result {
|
|
@@ -141,18 +165,25 @@ impl Owner {
|
|
|
141
165
|
if current.verified_matches(&self.options,&[],deadline)? {return Ok(());}
|
|
142
166
|
let rules=expected.prepared.rules(&self.options)?;
|
|
143
167
|
if !current.verified_matches(&self.options,&rules,deadline)? {return Err(Error::Conflict);}
|
|
144
|
-
Self::
|
|
168
|
+
Self::cleanup_fence(&expected.prepared)?;
|
|
145
169
|
frontend::replace(&rules,&[],deadline)?;
|
|
146
170
|
self.context()?;
|
|
147
171
|
let after=frontend::Snapshot::read(deadline)?;
|
|
148
172
|
if after.backend!=current.backend || !after.verified_matches(&self.options,&[],deadline)? {return Err(Error::Conflict);}
|
|
149
|
-
Self::
|
|
173
|
+
Self::cleanup_fence(&expected.prepared)
|
|
150
174
|
})();
|
|
151
175
|
match result {Ok(())=>{self.applied=Some(expected);self.error=None;self.released=true;Ok(())},Err(error)=>{self.error=Some(error);Err(error)}}
|
|
152
176
|
}
|
|
153
177
|
fn close(&mut self, deadline: std::time::Instant) -> Result<()> {
|
|
178
|
+
self.context()?;
|
|
179
|
+
// Holding no rule is complete cleanup, whatever was refused before: the
|
|
180
|
+
// same generation-consistent read proves it without Docker's forwarding
|
|
181
|
+
// path. A read that fails is that read's error, never cleanup.
|
|
182
|
+
if frontend::Snapshot::observe(deadline)?.owns_nothing(&self.options)? {
|
|
183
|
+
self.pending=None;self.released=true;self.error=None;
|
|
184
|
+
return Ok(());
|
|
185
|
+
}
|
|
154
186
|
if let Some(pending)=self.pending.clone() {
|
|
155
|
-
self.context()?;
|
|
156
187
|
let current=frontend::Snapshot::read(deadline)?;
|
|
157
188
|
if current.backend!=pending.target.backend {return Err(Error::Conflict);}
|
|
158
189
|
let known=if current.verified_matches(&self.options,&pending.target.rules(&self.options)?,deadline)? {
|
|
@@ -163,12 +194,9 @@ impl Owner {
|
|
|
163
194
|
} else if current.verified_matches(&self.options,&[],deadline)? {None} else {return Err(Error::Conflict);};
|
|
164
195
|
self.applied=known;self.pending=None;
|
|
165
196
|
}
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
self.released=true;self.error=None;
|
|
170
|
-
}
|
|
171
|
-
Ok(())
|
|
197
|
+
// Rules of this owner without a receipt for them are never deleted.
|
|
198
|
+
let applied=self.applied.clone().ok_or(Error::Conflict)?;
|
|
199
|
+
self.release(applied,deadline)
|
|
172
200
|
}
|
|
173
201
|
}
|
|
174
202
|
pub fn dispatch(owner: &mut Option<Owner>, request: Request) -> Result<serde_json::Value> {
|