fast-unionfind 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fast_unionfind-1.0.0/LICENSE +29 -0
- fast_unionfind-1.0.0/MANIFEST.in +16 -0
- fast_unionfind-1.0.0/PKG-INFO +203 -0
- fast_unionfind-1.0.0/README.md +179 -0
- fast_unionfind-1.0.0/pyproject.toml +103 -0
- fast_unionfind-1.0.0/setup.cfg +4 -0
- fast_unionfind-1.0.0/setup.py +34 -0
- fast_unionfind-1.0.0/src/fast_unionfind/__init__.pxd +5 -0
- fast_unionfind-1.0.0/src/fast_unionfind/__init__.py +65 -0
- fast_unionfind-1.0.0/src/fast_unionfind/core.pxd +121 -0
- fast_unionfind-1.0.0/src/fast_unionfind/core.pyx +339 -0
- fast_unionfind-1.0.0/src/fast_unionfind.egg-info/PKG-INFO +203 -0
- fast_unionfind-1.0.0/src/fast_unionfind.egg-info/SOURCES.txt +18 -0
- fast_unionfind-1.0.0/src/fast_unionfind.egg-info/dependency_links.txt +1 -0
- fast_unionfind-1.0.0/src/fast_unionfind.egg-info/requires.txt +5 -0
- fast_unionfind-1.0.0/src/fast_unionfind.egg-info/top_level.txt +1 -0
- fast_unionfind-1.0.0/tests/test_cimport.py +106 -0
- fast_unionfind-1.0.0/tests/test_threads.py +46 -0
- fast_unionfind-1.0.0/tests/test_unionfind.py +208 -0
- fast_unionfind-1.0.0/tests/unionfind_testlib.py +79 -0
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
Copyright © 2017-2018 Symantec Corporation. All Rights Reserved.
|
|
2
|
+
Copyright © 2026 Matteo Dell'Amico. All Rights Reserved.
|
|
3
|
+
|
|
4
|
+
Redistribution and use in source and binary forms, with or without
|
|
5
|
+
modification, are permitted provided that the following conditions are
|
|
6
|
+
met:
|
|
7
|
+
|
|
8
|
+
1. Redistributions of source code must retain the above copyright
|
|
9
|
+
notice, this list of conditions and the following disclaimer.
|
|
10
|
+
|
|
11
|
+
2. Redistributions in binary form must reproduce the above copyright
|
|
12
|
+
notice, this list of conditions and the following disclaimer in the
|
|
13
|
+
documentation and/or other materials provided with the distribution.
|
|
14
|
+
|
|
15
|
+
3. Neither the name of the copyright holder nor the names of its
|
|
16
|
+
contributors may be used to endorse or promote products derived from
|
|
17
|
+
this software without specific prior written permission.
|
|
18
|
+
|
|
19
|
+
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
|
|
20
|
+
"AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
|
|
21
|
+
LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR
|
|
22
|
+
A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT
|
|
23
|
+
HOLDER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
|
|
24
|
+
SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT
|
|
25
|
+
LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
|
|
26
|
+
DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
|
|
27
|
+
THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
|
28
|
+
(INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
|
29
|
+
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
# What an sdist must carry. setuptools' defaults ship .py but not .pyx or
|
|
2
|
+
# .pxd, and setup.py cythonizes core.pyx -- so without this line a source
|
|
3
|
+
# distribution has no sources to build from. The .pxd matters twice over
|
|
4
|
+
# here: it is not merely a declaration file but where the algorithm itself
|
|
5
|
+
# lives, so a consumer cimporting this package needs it installed.
|
|
6
|
+
recursive-include src/fast_unionfind *.pyx *.pxd
|
|
7
|
+
|
|
8
|
+
# The generated C is a build artefact: gitignored, and regenerated at build
|
|
9
|
+
# time because Cython is a build requirement (see pyproject's build-system).
|
|
10
|
+
# Excluded so that an sdist cannot pick up a stale copy left in the working
|
|
11
|
+
# tree and build from that instead.
|
|
12
|
+
recursive-exclude src/fast_unionfind *.c
|
|
13
|
+
|
|
14
|
+
# tests/ ships, so ship the helper every test module imports; without it the
|
|
15
|
+
# shipped tests cannot even be collected.
|
|
16
|
+
include tests/unionfind_testlib.py
|
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: fast-unionfind
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Disjoint sets (union-find) at C speed, with no dependencies
|
|
5
|
+
Author-email: Matteo Dell'Amico <della@linux.it>
|
|
6
|
+
License-Expression: BSD-3-Clause
|
|
7
|
+
Project-URL: Homepage, https://gitlab.com/bobtables/fast-unionfind
|
|
8
|
+
Keywords: union-find,disjoint-set,dsu,kruskal,connected-components,cython,free-threading
|
|
9
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
10
|
+
Classifier: Intended Audience :: Developers
|
|
11
|
+
Classifier: Intended Audience :: Science/Research
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Programming Language :: Cython
|
|
14
|
+
Classifier: Topic :: Scientific/Engineering
|
|
15
|
+
Classifier: Topic :: Software Development :: Libraries
|
|
16
|
+
Requires-Python: >=3.10
|
|
17
|
+
Description-Content-Type: text/markdown
|
|
18
|
+
License-File: LICENSE
|
|
19
|
+
Provides-Extra: test
|
|
20
|
+
Requires-Dist: pytest; extra == "test"
|
|
21
|
+
Requires-Dist: Cython>=3.0; extra == "test"
|
|
22
|
+
Requires-Dist: setuptools; extra == "test"
|
|
23
|
+
Dynamic: license-file
|
|
24
|
+
|
|
25
|
+
# fast-unionfind
|
|
26
|
+
|
|
27
|
+
Disjoint sets (union-find) for Python at C speed, with no dependencies.
|
|
28
|
+
|
|
29
|
+
```python
|
|
30
|
+
from fast_unionfind import unionfind
|
|
31
|
+
|
|
32
|
+
uf = unionfind(1_000_000)
|
|
33
|
+
uf.unite(3, 7) # True: two sets became one
|
|
34
|
+
uf.unite(7, 3) # False: already the same set
|
|
35
|
+
uf.connected(3, 7) # True
|
|
36
|
+
uf[7] == uf.find(3) # True
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
It is link-by-rank with path halving, so any sequence of m operations over
|
|
40
|
+
n elements costs O(m α(n)) — effectively constant per operation. It has no
|
|
41
|
+
runtime dependencies at all (not even numpy), it is compiled with Cython, and
|
|
42
|
+
it is safe to query from many threads at once, including on free-threaded
|
|
43
|
+
Python.
|
|
44
|
+
|
|
45
|
+
## Why another union-find
|
|
46
|
+
|
|
47
|
+
Two reasons.
|
|
48
|
+
|
|
49
|
+
**There was no C-speed union-find available as a stand-alone package.** The
|
|
50
|
+
fast implementations live inside larger libraries, as internals you cannot
|
|
51
|
+
depend on.
|
|
52
|
+
|
|
53
|
+
**Building one efficiently is less trivial than it looks.** The union-finds
|
|
54
|
+
inside two widely used clustering libraries, `hdbscan` and scikit-learn, had
|
|
55
|
+
subtle performance bugs that made single-linkage labelling quadratic while
|
|
56
|
+
every result stayed correct. We found and reported both:
|
|
57
|
+
[hdbscan#731](https://github.com/scikit-learn-contrib/hdbscan/issues/731) and
|
|
58
|
+
[scikit-learn#34626](https://github.com/scikit-learn/scikit-learn/issues/34626)
|
|
59
|
+
(a fix is in review as
|
|
60
|
+
[#34872](https://github.com/scikit-learn/scikit-learn/pull/34872)).
|
|
61
|
+
|
|
62
|
+
## Install
|
|
63
|
+
|
|
64
|
+
```
|
|
65
|
+
pip install fast-unionfind
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
Building from source needs a C compiler and Cython.
|
|
69
|
+
|
|
70
|
+
## Python API
|
|
71
|
+
|
|
72
|
+
`unionfind(n)` builds a union-find over the elements `0 … n-1` and picks the
|
|
73
|
+
narrowest index width that fits: a `UnionFind32` (4 bytes per element) up to
|
|
74
|
+
2³²−1 elements, a `UnionFind64` (8 bytes) beyond that. Either class can also be
|
|
75
|
+
built directly. `UnionFind` is their common base, for annotations and
|
|
76
|
+
`isinstance` checks, and cannot be instantiated itself.
|
|
77
|
+
|
|
78
|
+
| | |
|
|
79
|
+
|---|---|
|
|
80
|
+
| `uf.unite(x, y)` | Merge the sets holding x and y. `True` if they were distinct. |
|
|
81
|
+
| `uf.union(x, y)` | Alias of `unite`, under the name the literature uses. |
|
|
82
|
+
| `uf.find(x)`, `uf[x]` | The representative of x's set. |
|
|
83
|
+
| `uf.connected(x, y)` | Whether x and y are in the same set. |
|
|
84
|
+
| `len(uf)` | The number of elements, not of sets. |
|
|
85
|
+
| `uf.labels()` | Every element's representative, as an `array.array`. |
|
|
86
|
+
| `uf.width` | 32 or 64. |
|
|
87
|
+
| `width_for(n)` | The width `unionfind(n)` would choose, without allocating. |
|
|
88
|
+
|
|
89
|
+
Elements are ids, not sequence positions. Anything outside `range(n)` raises
|
|
90
|
+
`IndexError`, negative values included, so `uf[-1]` is an error rather than
|
|
91
|
+
the last element.
|
|
92
|
+
|
|
93
|
+
Representatives change as sets merge, so compare them rather than store them.
|
|
94
|
+
|
|
95
|
+
There is no set count or grouping method, because both are one visible line on
|
|
96
|
+
top of `labels()`, and writing them out keeps their O(n) cost in plain sight:
|
|
97
|
+
|
|
98
|
+
```python
|
|
99
|
+
n_sets = len(set(uf.labels()))
|
|
100
|
+
|
|
101
|
+
groups = collections.defaultdict(list)
|
|
102
|
+
for element, root in enumerate(uf.labels()):
|
|
103
|
+
groups[root].append(element)
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
`labels()` returns a buffer, so `numpy.frombuffer(uf.labels(), dtype=...)` wraps
|
|
107
|
+
it without copying if you do have numpy.
|
|
108
|
+
|
|
109
|
+
Union-finds pickle exactly, ranks included, so a restored one links later unions
|
|
110
|
+
the same way the original would have.
|
|
111
|
+
|
|
112
|
+
## Cython API
|
|
113
|
+
|
|
114
|
+
The union-find itself is not a method. It is a pair of fused `cdef inline`
|
|
115
|
+
functions **defined in the `.pxd`**, so a module that cimports them compiles its
|
|
116
|
+
own copy, specialised for its index width, and the C compiler inlines them into
|
|
117
|
+
your loop:
|
|
118
|
+
|
|
119
|
+
```cython
|
|
120
|
+
from libc.stdint cimport uint32_t
|
|
121
|
+
from fast_unionfind.core cimport UnionFind32, uf_find, uf_unite
|
|
122
|
+
|
|
123
|
+
def kruskal(Py_ssize_t n, uint32_t[::1] ei, uint32_t[::1] ej):
|
|
124
|
+
cdef UnionFind32 uf = UnionFind32(n) # owns the memory
|
|
125
|
+
cdef uint32_t* parents = uf.parents # plain struct reads:
|
|
126
|
+
cdef unsigned char* ranks = uf.ranks # the class is final
|
|
127
|
+
cdef Py_ssize_t k, merged = 0
|
|
128
|
+
with nogil:
|
|
129
|
+
for k in range(ei.shape[0]):
|
|
130
|
+
if uf_unite(parents, ranks, ei[k], ej[k]):
|
|
131
|
+
merged += 1
|
|
132
|
+
return merged
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
| | |
|
|
136
|
+
|---|---|
|
|
137
|
+
| `uf_find(parents, x)` | Root of x, halving the path. `noexcept nogil`, unchecked. |
|
|
138
|
+
| `uf_unite(parents, ranks, x, y)` | Link by rank. `True` if it merged. `noexcept nogil`, unchecked. |
|
|
139
|
+
|
|
140
|
+
Both are fused over `uint32_t` and `uint64_t`. Use `UnionFind64`'s `parents` for
|
|
141
|
+
the wide one. They do no bounds checking, which is the point of this layer. Keep
|
|
142
|
+
the owning `UnionFind32`/`UnionFind64` alive for as long as you use its pointers.
|
|
143
|
+
|
|
144
|
+
The `.pxd` files ship in the wheel, so an installed package is enough for
|
|
145
|
+
Cython to resolve the cimport. To be explicit anyway, add
|
|
146
|
+
`fast_unionfind.get_include()` to your extension's `include_dirs`.
|
|
147
|
+
|
|
148
|
+
Since the core compiles into your module, a consumer does not even import
|
|
149
|
+
`fast_unionfind` at runtime for these calls, only to construct the object that
|
|
150
|
+
owns the arrays.
|
|
151
|
+
|
|
152
|
+
## Thread safety
|
|
153
|
+
|
|
154
|
+
Concurrent `find`s are safe as long as no union is running at the same time.
|
|
155
|
+
They do race, on path-compression stores, but harmlessly. Every such store
|
|
156
|
+
points a node at one of its own ancestors, so the root stays reachable and
|
|
157
|
+
unchanged. A lost write can only undo some compression, never send a walk
|
|
158
|
+
somewhere wrong, and the aligned integer slots mean no store is ever
|
|
159
|
+
half-written. This is what makes a lock-free parallel "are these already
|
|
160
|
+
connected?" filter over a Kruskal edge list sound.
|
|
161
|
+
|
|
162
|
+
Unions must not run concurrently with each other or with finds. Serialise
|
|
163
|
+
them.
|
|
164
|
+
|
|
165
|
+
The extension is declared `freethreading_compatible`, and nothing in it needs
|
|
166
|
+
the GIL.
|
|
167
|
+
|
|
168
|
+
## Performance
|
|
169
|
+
|
|
170
|
+
Two workloads over m = 4n random edges, written as Cython loops. The first,
|
|
171
|
+
`unite_all`, is Kruskal's scan. The second, `filter_scan`, unites half the
|
|
172
|
+
edges and then asks "already connected?" of all of them, which is two finds
|
|
173
|
+
per edge. The comparison is the same algorithm written the usual way, as
|
|
174
|
+
methods on a `cdef class` cimported from another extension module, where
|
|
175
|
+
Cython must call through the imported vtable and nothing can inline.
|
|
176
|
+
|
|
177
|
+
| n | workload | inlined core | cimported class | speedup |
|
|
178
|
+
|---:|---|---:|---:|---:|
|
|
179
|
+
| 100 000 | `unite_all` | 2 ms | 3 ms | 1.51× |
|
|
180
|
+
| 100 000 | `filter_scan` | 2 ms | 4 ms | 1.64× |
|
|
181
|
+
| 300 000 | `unite_all` | 7 ms | 9 ms | 1.36× |
|
|
182
|
+
| 300 000 | `filter_scan` | 9 ms | 13 ms | 1.49× |
|
|
183
|
+
| 1 000 000 | `unite_all` | 24 ms | 35 ms | 1.45× |
|
|
184
|
+
| 1 000 000 | `filter_scan` | 32 ms | 51 ms | 1.57× |
|
|
185
|
+
| 10 000 000 | `unite_all` | 1.17 s | 1.53 s | 1.30× |
|
|
186
|
+
| 10 000 000 | `filter_scan` | 1.59 s | 2.17 s | 1.37× |
|
|
187
|
+
|
|
188
|
+
Best of 50 runs at n ≤ 300 000 and best of 5 above that, on x86-64 with
|
|
189
|
+
GCC -O3, Python 3.14 and Cython 3.3. The gain narrows at 10 million elements,
|
|
190
|
+
where the 40 MB parent array no longer fits in cache and memory traffic
|
|
191
|
+
dominates the call cost.
|
|
192
|
+
|
|
193
|
+
From Python, method calls cost about 42 ns each, which is almost all
|
|
194
|
+
interpreter overhead and the same as calling a class method directly.
|
|
195
|
+
|
|
196
|
+
The class-based column was measured before that implementation was retired, so
|
|
197
|
+
it cannot be re-run from this repository. `python benchmarks/bench.py` times
|
|
198
|
+
this package's workloads and Python-level calls, compiling the workloads on
|
|
199
|
+
first use.
|
|
200
|
+
|
|
201
|
+
## License
|
|
202
|
+
|
|
203
|
+
BSD 3-clause; see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
# fast-unionfind
|
|
2
|
+
|
|
3
|
+
Disjoint sets (union-find) for Python at C speed, with no dependencies.
|
|
4
|
+
|
|
5
|
+
```python
|
|
6
|
+
from fast_unionfind import unionfind
|
|
7
|
+
|
|
8
|
+
uf = unionfind(1_000_000)
|
|
9
|
+
uf.unite(3, 7) # True: two sets became one
|
|
10
|
+
uf.unite(7, 3) # False: already the same set
|
|
11
|
+
uf.connected(3, 7) # True
|
|
12
|
+
uf[7] == uf.find(3) # True
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
It is link-by-rank with path halving, so any sequence of m operations over
|
|
16
|
+
n elements costs O(m α(n)) — effectively constant per operation. It has no
|
|
17
|
+
runtime dependencies at all (not even numpy), it is compiled with Cython, and
|
|
18
|
+
it is safe to query from many threads at once, including on free-threaded
|
|
19
|
+
Python.
|
|
20
|
+
|
|
21
|
+
## Why another union-find
|
|
22
|
+
|
|
23
|
+
Two reasons.
|
|
24
|
+
|
|
25
|
+
**There was no C-speed union-find available as a stand-alone package.** The
|
|
26
|
+
fast implementations live inside larger libraries, as internals you cannot
|
|
27
|
+
depend on.
|
|
28
|
+
|
|
29
|
+
**Building one efficiently is less trivial than it looks.** The union-finds
|
|
30
|
+
inside two widely used clustering libraries, `hdbscan` and scikit-learn, had
|
|
31
|
+
subtle performance bugs that made single-linkage labelling quadratic while
|
|
32
|
+
every result stayed correct. We found and reported both:
|
|
33
|
+
[hdbscan#731](https://github.com/scikit-learn-contrib/hdbscan/issues/731) and
|
|
34
|
+
[scikit-learn#34626](https://github.com/scikit-learn/scikit-learn/issues/34626)
|
|
35
|
+
(a fix is in review as
|
|
36
|
+
[#34872](https://github.com/scikit-learn/scikit-learn/pull/34872)).
|
|
37
|
+
|
|
38
|
+
## Install
|
|
39
|
+
|
|
40
|
+
```
|
|
41
|
+
pip install fast-unionfind
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
Building from source needs a C compiler and Cython.
|
|
45
|
+
|
|
46
|
+
## Python API
|
|
47
|
+
|
|
48
|
+
`unionfind(n)` builds a union-find over the elements `0 … n-1` and picks the
|
|
49
|
+
narrowest index width that fits: a `UnionFind32` (4 bytes per element) up to
|
|
50
|
+
2³²−1 elements, a `UnionFind64` (8 bytes) beyond that. Either class can also be
|
|
51
|
+
built directly. `UnionFind` is their common base, for annotations and
|
|
52
|
+
`isinstance` checks, and cannot be instantiated itself.
|
|
53
|
+
|
|
54
|
+
| | |
|
|
55
|
+
|---|---|
|
|
56
|
+
| `uf.unite(x, y)` | Merge the sets holding x and y. `True` if they were distinct. |
|
|
57
|
+
| `uf.union(x, y)` | Alias of `unite`, under the name the literature uses. |
|
|
58
|
+
| `uf.find(x)`, `uf[x]` | The representative of x's set. |
|
|
59
|
+
| `uf.connected(x, y)` | Whether x and y are in the same set. |
|
|
60
|
+
| `len(uf)` | The number of elements, not of sets. |
|
|
61
|
+
| `uf.labels()` | Every element's representative, as an `array.array`. |
|
|
62
|
+
| `uf.width` | 32 or 64. |
|
|
63
|
+
| `width_for(n)` | The width `unionfind(n)` would choose, without allocating. |
|
|
64
|
+
|
|
65
|
+
Elements are ids, not sequence positions. Anything outside `range(n)` raises
|
|
66
|
+
`IndexError`, negative values included, so `uf[-1]` is an error rather than
|
|
67
|
+
the last element.
|
|
68
|
+
|
|
69
|
+
Representatives change as sets merge, so compare them rather than store them.
|
|
70
|
+
|
|
71
|
+
There is no set count or grouping method, because both are one visible line on
|
|
72
|
+
top of `labels()`, and writing them out keeps their O(n) cost in plain sight:
|
|
73
|
+
|
|
74
|
+
```python
|
|
75
|
+
n_sets = len(set(uf.labels()))
|
|
76
|
+
|
|
77
|
+
groups = collections.defaultdict(list)
|
|
78
|
+
for element, root in enumerate(uf.labels()):
|
|
79
|
+
groups[root].append(element)
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
`labels()` returns a buffer, so `numpy.frombuffer(uf.labels(), dtype=...)` wraps
|
|
83
|
+
it without copying if you do have numpy.
|
|
84
|
+
|
|
85
|
+
Union-finds pickle exactly, ranks included, so a restored one links later unions
|
|
86
|
+
the same way the original would have.
|
|
87
|
+
|
|
88
|
+
## Cython API
|
|
89
|
+
|
|
90
|
+
The union-find itself is not a method. It is a pair of fused `cdef inline`
|
|
91
|
+
functions **defined in the `.pxd`**, so a module that cimports them compiles its
|
|
92
|
+
own copy, specialised for its index width, and the C compiler inlines them into
|
|
93
|
+
your loop:
|
|
94
|
+
|
|
95
|
+
```cython
|
|
96
|
+
from libc.stdint cimport uint32_t
|
|
97
|
+
from fast_unionfind.core cimport UnionFind32, uf_find, uf_unite
|
|
98
|
+
|
|
99
|
+
def kruskal(Py_ssize_t n, uint32_t[::1] ei, uint32_t[::1] ej):
|
|
100
|
+
cdef UnionFind32 uf = UnionFind32(n) # owns the memory
|
|
101
|
+
cdef uint32_t* parents = uf.parents # plain struct reads:
|
|
102
|
+
cdef unsigned char* ranks = uf.ranks # the class is final
|
|
103
|
+
cdef Py_ssize_t k, merged = 0
|
|
104
|
+
with nogil:
|
|
105
|
+
for k in range(ei.shape[0]):
|
|
106
|
+
if uf_unite(parents, ranks, ei[k], ej[k]):
|
|
107
|
+
merged += 1
|
|
108
|
+
return merged
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
| | |
|
|
112
|
+
|---|---|
|
|
113
|
+
| `uf_find(parents, x)` | Root of x, halving the path. `noexcept nogil`, unchecked. |
|
|
114
|
+
| `uf_unite(parents, ranks, x, y)` | Link by rank. `True` if it merged. `noexcept nogil`, unchecked. |
|
|
115
|
+
|
|
116
|
+
Both are fused over `uint32_t` and `uint64_t`. Use `UnionFind64`'s `parents` for
|
|
117
|
+
the wide one. They do no bounds checking, which is the point of this layer. Keep
|
|
118
|
+
the owning `UnionFind32`/`UnionFind64` alive for as long as you use its pointers.
|
|
119
|
+
|
|
120
|
+
The `.pxd` files ship in the wheel, so an installed package is enough for
|
|
121
|
+
Cython to resolve the cimport. To be explicit anyway, add
|
|
122
|
+
`fast_unionfind.get_include()` to your extension's `include_dirs`.
|
|
123
|
+
|
|
124
|
+
Since the core compiles into your module, a consumer does not even import
|
|
125
|
+
`fast_unionfind` at runtime for these calls, only to construct the object that
|
|
126
|
+
owns the arrays.
|
|
127
|
+
|
|
128
|
+
## Thread safety
|
|
129
|
+
|
|
130
|
+
Concurrent `find`s are safe as long as no union is running at the same time.
|
|
131
|
+
They do race, on path-compression stores, but harmlessly. Every such store
|
|
132
|
+
points a node at one of its own ancestors, so the root stays reachable and
|
|
133
|
+
unchanged. A lost write can only undo some compression, never send a walk
|
|
134
|
+
somewhere wrong, and the aligned integer slots mean no store is ever
|
|
135
|
+
half-written. This is what makes a lock-free parallel "are these already
|
|
136
|
+
connected?" filter over a Kruskal edge list sound.
|
|
137
|
+
|
|
138
|
+
Unions must not run concurrently with each other or with finds. Serialise
|
|
139
|
+
them.
|
|
140
|
+
|
|
141
|
+
The extension is declared `freethreading_compatible`, and nothing in it needs
|
|
142
|
+
the GIL.
|
|
143
|
+
|
|
144
|
+
## Performance
|
|
145
|
+
|
|
146
|
+
Two workloads over m = 4n random edges, written as Cython loops. The first,
|
|
147
|
+
`unite_all`, is Kruskal's scan. The second, `filter_scan`, unites half the
|
|
148
|
+
edges and then asks "already connected?" of all of them, which is two finds
|
|
149
|
+
per edge. The comparison is the same algorithm written the usual way, as
|
|
150
|
+
methods on a `cdef class` cimported from another extension module, where
|
|
151
|
+
Cython must call through the imported vtable and nothing can inline.
|
|
152
|
+
|
|
153
|
+
| n | workload | inlined core | cimported class | speedup |
|
|
154
|
+
|---:|---|---:|---:|---:|
|
|
155
|
+
| 100 000 | `unite_all` | 2 ms | 3 ms | 1.51× |
|
|
156
|
+
| 100 000 | `filter_scan` | 2 ms | 4 ms | 1.64× |
|
|
157
|
+
| 300 000 | `unite_all` | 7 ms | 9 ms | 1.36× |
|
|
158
|
+
| 300 000 | `filter_scan` | 9 ms | 13 ms | 1.49× |
|
|
159
|
+
| 1 000 000 | `unite_all` | 24 ms | 35 ms | 1.45× |
|
|
160
|
+
| 1 000 000 | `filter_scan` | 32 ms | 51 ms | 1.57× |
|
|
161
|
+
| 10 000 000 | `unite_all` | 1.17 s | 1.53 s | 1.30× |
|
|
162
|
+
| 10 000 000 | `filter_scan` | 1.59 s | 2.17 s | 1.37× |
|
|
163
|
+
|
|
164
|
+
Best of 50 runs at n ≤ 300 000 and best of 5 above that, on x86-64 with
|
|
165
|
+
GCC -O3, Python 3.14 and Cython 3.3. The gain narrows at 10 million elements,
|
|
166
|
+
where the 40 MB parent array no longer fits in cache and memory traffic
|
|
167
|
+
dominates the call cost.
|
|
168
|
+
|
|
169
|
+
From Python, method calls cost about 42 ns each, which is almost all
|
|
170
|
+
interpreter overhead and the same as calling a class method directly.
|
|
171
|
+
|
|
172
|
+
The class-based column was measured before that implementation was retired, so
|
|
173
|
+
it cannot be re-run from this repository. `python benchmarks/bench.py` times
|
|
174
|
+
this package's workloads and Python-level calls, compiling the workloads on
|
|
175
|
+
first use.
|
|
176
|
+
|
|
177
|
+
## License
|
|
178
|
+
|
|
179
|
+
BSD 3-clause; see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77", "Cython>=3.0"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "fast-unionfind"
|
|
7
|
+
version = "1.0.0"
|
|
8
|
+
description = "Disjoint sets (union-find) at C speed, with no dependencies"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "BSD-3-Clause"
|
|
11
|
+
license-files = ["LICENSE"]
|
|
12
|
+
authors = [{ name = "Matteo Dell'Amico", email = "della@linux.it" }]
|
|
13
|
+
requires-python = ">=3.10"
|
|
14
|
+
# No runtime dependencies at all, and that is a feature rather than an
|
|
15
|
+
# accident: the structure needs two raw arrays, so pulling in numpy for them
|
|
16
|
+
# would cost every install on an interpreter the scientific stack has not
|
|
17
|
+
# built wheels for yet -- free-threaded builds in particular, where that lag
|
|
18
|
+
# is routinely months.
|
|
19
|
+
dependencies = []
|
|
20
|
+
keywords = ["union-find", "disjoint-set", "dsu", "kruskal",
|
|
21
|
+
"connected-components", "cython", "free-threading"]
|
|
22
|
+
classifiers = [
|
|
23
|
+
"Development Status :: 5 - Production/Stable",
|
|
24
|
+
"Intended Audience :: Developers",
|
|
25
|
+
"Intended Audience :: Science/Research",
|
|
26
|
+
"Programming Language :: Python :: 3",
|
|
27
|
+
"Programming Language :: Cython",
|
|
28
|
+
"Topic :: Scientific/Engineering",
|
|
29
|
+
"Topic :: Software Development :: Libraries",
|
|
30
|
+
]
|
|
31
|
+
|
|
32
|
+
[project.urls]
|
|
33
|
+
Homepage = "https://gitlab.com/bobtables/fast-unionfind"
|
|
34
|
+
|
|
35
|
+
[project.optional-dependencies]
|
|
36
|
+
# The Cython-consumer test compiles a module against the installed .pxd,
|
|
37
|
+
# which is the only thing here that needs a toolchain at test time.
|
|
38
|
+
test = ["pytest", "Cython>=3.0", "setuptools"]
|
|
39
|
+
|
|
40
|
+
[tool.setuptools]
|
|
41
|
+
packages = ["fast_unionfind"]
|
|
42
|
+
# src/ layout: keeps the repo root free of a bare `fast_unionfind/` directory,
|
|
43
|
+
# which would otherwise shadow the installed package whenever the root lands
|
|
44
|
+
# on sys.path (running pytest from here, `python -c "import fast_unionfind"`,
|
|
45
|
+
# ...) and silently import a namespace package with no extension module in it.
|
|
46
|
+
package-dir = { "" = "src" }
|
|
47
|
+
|
|
48
|
+
[tool.setuptools.package-data]
|
|
49
|
+
# The .pxd is not a build artefact here, it is the algorithm: a consumer that
|
|
50
|
+
# cimports this package compiles these bodies into its own object file. An
|
|
51
|
+
# install without them still runs Python callers and silently breaks Cython
|
|
52
|
+
# ones, so they ship with the wheel, not just the sdist.
|
|
53
|
+
fast_unionfind = ["*.pxd"]
|
|
54
|
+
|
|
55
|
+
[tool.cibuildwheel]
|
|
56
|
+
# 64-bit only: Windows' "auto" still adds 32-bit x86, which this package has
|
|
57
|
+
# never been built or tested on. PyPy and GraalPy are opt-in in cibuildwheel,
|
|
58
|
+
# so this is CPython, free-threaded builds included.
|
|
59
|
+
skip = "*-win32"
|
|
60
|
+
# The whole suite against every wheel; it takes seconds. The test extra
|
|
61
|
+
# brings Cython and setuptools, so the cimport test compiles a consumer
|
|
62
|
+
# against the installed .pxd -- the part a broken wheel would break.
|
|
63
|
+
test-extras = ["test"]
|
|
64
|
+
test-command = "pytest {project}/tests"
|
|
65
|
+
|
|
66
|
+
[tool.pytest.ini_options]
|
|
67
|
+
# importlib mode does not prepend rootdir/test dirs to sys.path, so tests
|
|
68
|
+
# always resolve `fast_unionfind` to the *installed* package rather than to
|
|
69
|
+
# whatever directory happens to sit next to them. Belt and braces with the
|
|
70
|
+
# src/ layout above: either alone fixes the shadowing, together they make it
|
|
71
|
+
# impossible.
|
|
72
|
+
addopts = "--import-mode=importlib"
|
|
73
|
+
testpaths = ["tests"]
|
|
74
|
+
# importlib mode deliberately does not put the test directory on sys.path, so
|
|
75
|
+
# the shared scaffolding in tests/unionfind_testlib.py needs saying
|
|
76
|
+
# explicitly. The module is named for this package rather than something like
|
|
77
|
+
# `helpers`, since what goes on sys.path here could otherwise shadow someone
|
|
78
|
+
# else's module.
|
|
79
|
+
pythonpath = ["tests"]
|
|
80
|
+
|
|
81
|
+
[tool.ruff]
|
|
82
|
+
# 79, matching the prose in this project's docstrings and comments, which are
|
|
83
|
+
# hand-wrapped to it. The formatter leaves comment and docstring *text* alone,
|
|
84
|
+
# so it will not undo that wrapping -- but a wider limit would let new code
|
|
85
|
+
# drift away from it.
|
|
86
|
+
line-length = 79
|
|
87
|
+
extend-exclude = ["src/*/*.c", "build"]
|
|
88
|
+
|
|
89
|
+
[tool.ruff.lint]
|
|
90
|
+
# F is the useful half: undefined names, unused imports, star-import damage.
|
|
91
|
+
# E501 keeps the line length honest. Deliberately not the whole default set --
|
|
92
|
+
# this is a tidiness tool here, and the bugs worth catching in this codebase
|
|
93
|
+
# have all been the kind no linter sees.
|
|
94
|
+
select = ["F", "E501"]
|
|
95
|
+
|
|
96
|
+
[tool.ruff.lint.pycodestyle]
|
|
97
|
+
# The formatter targets 79 and reflows code to it, but it deliberately leaves
|
|
98
|
+
# comment and docstring *text* alone -- so a sentence that lands on 80 cannot
|
|
99
|
+
# be fixed by the tool, only by hand, and would nag forever. E501 is therefore
|
|
100
|
+
# a backstop against genuinely runaway lines rather than a second opinion on
|
|
101
|
+
# prose: wrap at 79 by hand as the rest of this codebase does, and let the
|
|
102
|
+
# linter object only past 88.
|
|
103
|
+
max-line-length = 88
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
"""Build script for the Cython extension; all metadata is in pyproject.toml."""
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
import sys
|
|
5
|
+
|
|
6
|
+
from Cython.Build import cythonize
|
|
7
|
+
from setuptools import Extension, setup
|
|
8
|
+
|
|
9
|
+
if sys.platform == "win32":
|
|
10
|
+
extra_compile_args = ["/O2"]
|
|
11
|
+
else:
|
|
12
|
+
# -O3 so the parent-array init loop vectorises and the inlined find/unite
|
|
13
|
+
# in a caller's scan stay tight. No -ffast-math family here and no
|
|
14
|
+
# -fopenmp: there is no floating point in this package and no parallel
|
|
15
|
+
# loop of its own -- callers parallelise over it, which needs nothing
|
|
16
|
+
# from this build.
|
|
17
|
+
extra_compile_args = ["-O3"]
|
|
18
|
+
# -march=native produces a faster but machine-specific binary; opt in
|
|
19
|
+
# when building for the machine you'll run on, not for a portable wheel.
|
|
20
|
+
if os.environ.get("FAST_UNIONFIND_NATIVE"):
|
|
21
|
+
extra_compile_args.append("-march=native")
|
|
22
|
+
|
|
23
|
+
setup(
|
|
24
|
+
ext_modules=cythonize(
|
|
25
|
+
[
|
|
26
|
+
Extension(
|
|
27
|
+
"fast_unionfind.core",
|
|
28
|
+
["src/fast_unionfind/core.pyx"],
|
|
29
|
+
extra_compile_args=extra_compile_args,
|
|
30
|
+
),
|
|
31
|
+
],
|
|
32
|
+
language_level=3,
|
|
33
|
+
),
|
|
34
|
+
)
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
# Marks fast_unionfind as a cimportable package, so that an installed copy
|
|
2
|
+
# resolves `from fast_unionfind.core cimport uf_find, UnionFind32` off
|
|
3
|
+
# sys.path. numpy ships the same empty marker for the same reason; without it
|
|
4
|
+
# Cython finds the directory but not the package, and the cimport fails with
|
|
5
|
+
# "cannot find module".
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
# Copyright (c) 2017-2018 Symantec Corporation. All Rights Reserved.
|
|
2
|
+
# Copyright (c) 2026 Matteo Dell'Amico. All Rights Reserved.
|
|
3
|
+
# Redistribution and use governed by the BSD 3-clause license in LICENSE.
|
|
4
|
+
|
|
5
|
+
"""Disjoint sets (union-find) at C speed, with no dependencies.
|
|
6
|
+
|
|
7
|
+
`unionfind(n)` builds one over the elements `0..n-1`, picking `UnionFind32`
|
|
8
|
+
or `UnionFind64` by size; `UnionFind` is the base type to name in annotations
|
|
9
|
+
and `isinstance` checks. Merge with `unite(x, y)`, which returns True only
|
|
10
|
+
when it actually merged something, and query with `find(x)` or `uf[x]`.
|
|
11
|
+
|
|
12
|
+
>>> from fast_unionfind import unionfind
|
|
13
|
+
>>> uf = unionfind(5)
|
|
14
|
+
>>> uf.unite(0, 1), uf.unite(1, 2), uf.unite(0, 2)
|
|
15
|
+
(True, True, False)
|
|
16
|
+
>>> uf.connected(0, 2)
|
|
17
|
+
True
|
|
18
|
+
|
|
19
|
+
Link by rank and path halving, so a sequence of m operations over n elements
|
|
20
|
+
costs O(m alpha(n)) -- effectively constant per operation. Path compression is
|
|
21
|
+
easy to lose without a single wrong answer to show for it: the union-finds in
|
|
22
|
+
`hdbscan` and scikit-learn lost theirs that way, and their labelling went
|
|
23
|
+
quadratic. See the README.
|
|
24
|
+
|
|
25
|
+
Cython callers should `cimport` the core rather than call these methods:
|
|
26
|
+
|
|
27
|
+
from fast_unionfind.core cimport UnionFind32, uf_find, uf_unite
|
|
28
|
+
|
|
29
|
+
`uf_find` and `uf_unite` are fused `cdef inline` functions defined in the
|
|
30
|
+
.pxd, so they inline into your loop -- no call, no GIL, no numpy. The
|
|
31
|
+
extension type is only there to own the allocation.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
import os
|
|
35
|
+
|
|
36
|
+
from .core import (
|
|
37
|
+
UnionFind,
|
|
38
|
+
UnionFind32,
|
|
39
|
+
UnionFind64,
|
|
40
|
+
unionfind,
|
|
41
|
+
width_for,
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
__version__ = "1.0.0"
|
|
45
|
+
|
|
46
|
+
__all__ = [
|
|
47
|
+
"UnionFind",
|
|
48
|
+
"UnionFind32",
|
|
49
|
+
"UnionFind64",
|
|
50
|
+
"get_include",
|
|
51
|
+
"unionfind",
|
|
52
|
+
"width_for",
|
|
53
|
+
]
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def get_include():
|
|
57
|
+
"""Directory holding this package's .pxd files.
|
|
58
|
+
|
|
59
|
+
Cython finds them on sys.path unaided, thanks to the `__init__.pxd`
|
|
60
|
+
marker, so this is belt and braces for a build that would rather be
|
|
61
|
+
explicit:
|
|
62
|
+
|
|
63
|
+
Extension(..., include_dirs=[fast_unionfind.get_include()])
|
|
64
|
+
"""
|
|
65
|
+
return os.path.dirname(__file__)
|