fuzzgpu 0.1.5__cp310-abi3-win_amd64.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. LICENSE +21 -0
  2. README.md +529 -0
  3. assets/logo.png +0 -0
  4. crates/fuzzgpu-core/Cargo.toml +32 -0
  5. crates/fuzzgpu-core/benches/bench.rs +306 -0
  6. crates/fuzzgpu-core/src/damerau.rs +787 -0
  7. crates/fuzzgpu-core/src/fuzz.rs +322 -0
  8. crates/fuzzgpu-core/src/gpu.rs +693 -0
  9. crates/fuzzgpu-core/src/jaro.rs +1322 -0
  10. crates/fuzzgpu-core/src/levenshtein.rs +1825 -0
  11. crates/fuzzgpu-core/src/lib.rs +91 -0
  12. crates/fuzzgpu-core/src/needleman.rs +1030 -0
  13. crates/fuzzgpu-core/src/shaders/damerau.wgsl +132 -0
  14. crates/fuzzgpu-core/src/shaders/damerau_matrix.wgsl +110 -0
  15. crates/fuzzgpu-core/src/shaders/jaro.wgsl +155 -0
  16. crates/fuzzgpu-core/src/shaders/jaro_matrix.wgsl +126 -0
  17. crates/fuzzgpu-core/src/shaders/levenshtein.wgsl +72 -0
  18. crates/fuzzgpu-core/src/shaders/levenshtein_cdist_myers.wgsl +145 -0
  19. crates/fuzzgpu-core/src/shaders/levenshtein_matrix.wgsl +82 -0
  20. crates/fuzzgpu-core/src/shaders/levenshtein_myers.wgsl +145 -0
  21. crates/fuzzgpu-core/src/shaders/levenshtein_short.wgsl +78 -0
  22. crates/fuzzgpu-core/src/shaders/needleman_affine.wgsl +112 -0
  23. crates/fuzzgpu-core/src/shaders/needleman_wavefront.wgsl +150 -0
  24. crates/fuzzgpu-core/src/simd.rs +1551 -0
  25. crates/fuzzgpu-core/tests/differential.proptest-regressions +7 -0
  26. crates/fuzzgpu-core/tests/differential.rs +1039 -0
  27. crates/fuzzgpu-core/tests/fixtures/broken.wgsl +9 -0
  28. crates/fuzzgpu-core/tests/kernel_registration.rs +81 -0
  29. crates/fuzzgpu-python/Cargo.toml +24 -0
  30. crates/fuzzgpu-python/src/lib.rs +946 -0
  31. crates/fuzzgpu-wasm/Cargo.toml +22 -0
  32. crates/fuzzgpu-wasm/src/lib.rs +155 -0
  33. crates/fuzzgpu-wasm/tests/differential_harness.js +237 -0
  34. crates/fuzzgpu-wasm/tests/js_api.test.cjs +63 -0
  35. fuzzgpu/__init__.py +178 -0
  36. fuzzgpu/__init__.pyi +107 -0
  37. fuzzgpu/distance/DamerauLevenshtein.py +43 -0
  38. fuzzgpu/distance/Hamming.py +37 -0
  39. fuzzgpu/distance/Indel.py +42 -0
  40. fuzzgpu/distance/Jaro.py +24 -0
  41. fuzzgpu/distance/JaroWinkler.py +26 -0
  42. fuzzgpu/distance/Levenshtein.py +111 -0
  43. fuzzgpu/distance/OSA.py +48 -0
  44. fuzzgpu/distance/__init__.py +5 -0
  45. fuzzgpu/distance/_common.py +6 -0
  46. fuzzgpu/fuzz.py +105 -0
  47. fuzzgpu/fuzz.pyi +36 -0
  48. fuzzgpu/fuzzgpu.pyd +0 -0
  49. fuzzgpu/process.py +232 -0
  50. fuzzgpu/process.pyi +55 -0
  51. fuzzgpu-0.1.5.dist-info/METADATA +559 -0
  52. fuzzgpu-0.1.5.dist-info/RECORD +64 -0
  53. fuzzgpu-0.1.5.dist-info/WHEEL +4 -0
  54. fuzzgpu-0.1.5.dist-info/licenses/LICENSE +21 -0
  55. fuzzgpu-0.1.5.dist-info/sboms/fuzzgpu-python.cyclonedx.json +4090 -0
  56. tests/test_api_signatures.py +171 -0
  57. tests/test_basic.py +329 -0
  58. tests/test_concurrency.py +62 -0
  59. tests/test_edge_cases.py +353 -0
  60. tests/test_invariants.py +169 -0
  61. tests/test_out_buffers.py +160 -0
  62. tests/test_rapidfuzz_compat.py +43 -0
  63. tests/test_stress.py +123 -0
  64. tests/wasm_python_differential.py +171 -0
LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2025-2026 Flaxmbot
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
README.md ADDED
@@ -0,0 +1,529 @@
1
+ <div align="center">
2
+
3
+ <img src="https://raw.githubusercontent.com/kuntal-devrat/fuzzgpu/main/assets/logo.png" alt="fuzzgpu logo" width="140" height="140" />
4
+
5
+ # fuzzgpu
6
+
7
+ **Hardware-Accelerated Fuzzy String Matching & Sequence Alignment**
8
+
9
+ *Cross-platform GPU compute via WebGPU (`wgpu`) & Multi-Core CPU parallelism with Rayon. Zero CUDA dependencies.*
10
+
11
+ [![PyPI Version](https://img.shields.io/badge/pypi-v0.1.4-blue.svg?style=flat-square)](https://pypi.org/project/fuzzgpu/)
12
+ [![License: MIT](https://img.shields.io/badge/License-MIT-green.svg?style=flat-square)](https://opensource.org/licenses/MIT)
13
+ [![Rust](https://img.shields.io/badge/rust-1.87+-orange.svg?style=flat-square)](https://www.rust-lang.org)
14
+ [![Cross Platform](https://img.shields.io/badge/platform-Windows%20%7C%20macOS%20%7C%20Linux%20%7C%20WASM-lightgrey.svg?style=flat-square)](https://github.com/kuntal-devrat/fuzzgpu)
15
+ [![Release wasm package](https://github.com/kuntal-devrat/fuzzgpu/actions/workflows/wasm-release.yml/badge.svg)](https://github.com/kuntal-devrat/fuzzgpu/actions/workflows/wasm-release.yml)
16
+
17
+ </div>
18
+
19
+ ---
20
+
21
+ ## Overview
22
+
23
+ `fuzzgpu` is a high-throughput string distance and sequence alignment engine written in **Rust** with native **Python** and **WebAssembly** bindings. It leverages GPU compute shaders (`wgpu` / WGSL) and Rayon multi-threading to accelerate large-scale batch queries and distance matrix computations across:
24
+
25
+ - **Apple Silicon (Metal)**
26
+ - **Linux (Vulkan)**
27
+ - **Windows (DirectX 12 / Vulkan)**
28
+ - **Integrated GPUs (Intel Iris Xe, AMD Radeon)**
29
+ - **WebAssembly (In-browser execution)**
30
+
31
+ No NVIDIA CUDA drivers or complex toolkits required.
32
+
33
+ ---
34
+
35
+ ## Benchmark Results
36
+
37
+ *Hardware: Intel(R) Iris(R) Xe Graphics (Vulkan) + Intel Core i7 (Rayon uses all cores)*
38
+ *Versions: fuzzgpu 0.1.4 (release) · rapidfuzz 3.14.5 · python-Levenshtein 0.27.4*
39
+ *Method: median of 7 runs after a warmup call, one full library call per measurement.
40
+ Reproduce with `python benchmarks/bench_compare.py`.*
41
+
42
+ > **How to read these tables.** `fuzzgpu`'s CPU path is multi-threaded (Rayon)
43
+ > and uses the **Myers (1999) bit-vector** for ASCII pairs with a ≤ 64-char
44
+ > pattern (only the *pattern* must be short — the text can be any length),
45
+ > amplified by width-aware SIMD kernels — **AVX512** (8 texts/vector),
46
+ > **AVX2** (4), **NEON** (2), portable fallback — for Levenshtein *and*
47
+ > Jaro (bit-parallel matching-window pass). This is why the Levenshtein and
48
+ > Jaro-Winkler CPU numbers below *beat* rapidfuzz's C++/SIMD at scale, and
49
+ > Damerau (unrestricted Lowrance-Wagner) crushes it. The Python bindings are
50
+ > **zero-copy** (`abi3-py310` + pyo3 `Bound<str>` views — no `Vec<String>`
51
+ > copies per call).
52
+ >
53
+ > **GPU kernels exist for every metric.** Levenshtein uses the Myers bit-vector
54
+ > shader (shared Peq per workgroup, two-u32 bit-vector so no `SHADER_INT64`
55
+ > is needed) plus a **row-wise Myers cdist kernel** (~10× faster than the old
56
+ > DP matrix shader). Jaro-Winkler has a **bitmap-matching shader** (128-bit
57
+ > bitmaps in registers, transposed pair-major char layout for coalesced
58
+ > loads). Damerau has a **Lowrance-Wagner shader** that keeps each pair's full
59
+ > DP matrix in workgroup shared memory (bit-exact with the CPU reference,
60
+ > including non-adjacent transpositions like `ca`/`abc` = 2).
61
+ >
62
+ > **Routing is metric-aware and backend-aware.** The GPU carries a per-dispatch
63
+ > sync round-trip (~1 ms on an iGPU), and the Jaro/Damerau kernels are
64
+ > heavier per pair than Myers — measured on Iris Xe they lose to the SIMD CPU
65
+ > path at *every* scale (Jaro ~2×, Damerau ~6× at 50k pairs). Auto-routing
66
+ > therefore never sends them to an integrated GPU (the "GPU" columns below
67
+ > are the auto-routed result, i.e. the fast CPU path); on discrete GPUs they
68
+ > dispatch above a scaled threshold (Jaro ≥ 1,024 pairs, Damerau ≥ 2,048).
69
+ > Levenshtein's Myers kernel is cheap enough per pair to win on iGPUs at
70
+ > scale and is routed normally. `hardware_info()` shows the adapter class,
71
+ > auto threshold, and how many pairs actually went to the GPU;
72
+ > `set_gpu_threshold(n)` overrides any of it. These tables replace the
73
+ > v0.1.0 numbers, which were not reproducible: they compared against
74
+ > pure-Python loops and timed single un-warmed calls.
75
+
76
+ ### 1. Levenshtein Batch (1 query × N candidates, 10-char strings)
77
+ | Batch Size | `fuzzgpu` (GPU) | `fuzzgpu` (CPU) | `rapidfuzz` | vs RF (GPU) | vs RF (CPU) |
78
+ | :--- | :---: | :---: | :---: | :---: | :---: |
79
+ | **100** | 0.12 ms | 0.01 ms | 0.03 ms | 0.25× | 2.21× |
80
+ | **1,000** | 0.83 ms | 0.09 ms | 0.15 ms | 0.18× | 1.67× |
81
+ | **10,000** | 2.60 ms | 0.87 ms | 1.47 ms | 0.57× | 1.68× |
82
+ | **50,000** | 9.92 ms | 5.04 ms | 6.38 ms | 0.64× | 1.27× |
83
+
84
+ ### 2. Damerau-Levenshtein Batch
85
+ *Unrestricted Lowrance-Wagner (non-adjacent transpositions included, unlike
86
+ rapidfuzz's optimal-string-alignment). The GPU kernel exists and is
87
+ bit-exact, but on this iGPU auto-routing sends it to the CPU path (the GPU
88
+ column is the auto-routed result).*
89
+ | Batch Size | `fuzzgpu` (GPU) | `fuzzgpu` (CPU) | `rapidfuzz` | vs RF (GPU) | vs RF (CPU) |
90
+ | :--- | :---: | :---: | :---: | :---: | :---: |
91
+ | **100** | 0.10 ms | 0.06 ms | 0.14 ms | 1.43× | 2.18× |
92
+ | **1,000** | 0.58 ms | 0.40 ms | 1.63 ms | 2.81× | 4.06× |
93
+ | **10,000** | 2.95 ms | 2.15 ms | 20.86 ms | 7.06× | 9.70× |
94
+ | **50,000** | 17.49 ms | 12.08 ms | 120.25 ms | 6.88× | 9.95× |
95
+
96
+ ### 3. Jaro-Winkler Batch (p = 0.1)
97
+ *GPU bitmap-matching kernel exists; on this iGPU auto-routing sends it to the
98
+ SIMD CPU path (the GPU column is the auto-routed result).*
99
+ | Batch Size | `fuzzgpu` (GPU) | `fuzzgpu` (CPU) | `rapidfuzz` | vs RF (GPU) | vs RF (CPU) |
100
+ | :--- | :---: | :---: | :---: | :---: | :---: |
101
+ | **100** | 0.01 ms | 0.01 ms | 0.01 ms | 1.06× | 1.13× |
102
+ | **1,000** | 0.11 ms | 0.10 ms | 0.10 ms | 0.95× | 0.99× |
103
+ | **10,000** | 1.50 ms | 1.68 ms | 1.72 ms | 1.15× | 1.02× |
104
+ | **50,000** | 10.96 ms | 14.34 ms | 15.62 ms | 1.43× | 1.09× |
105
+
106
+ ### 4. Needleman-Wunsch Batch (affine, match=1, mismatch=-1, gap_open=-2, gap_extend=-1)
107
+ *rapidfuzz has no affine-gap Needleman-Wunsch scorer, so no comparison column.*
108
+ | Batch Size | `fuzzgpu` (GPU) | `fuzzgpu` (CPU) |
109
+ | :--- | :---: | :---: |
110
+ | **100** | 0.30 ms | 0.21 ms |
111
+ | **1,000** | 1.02 ms | 1.24 ms |
112
+ | **10,000** | 7.16 ms | 4.39 ms |
113
+ | **50,000** | 20.82 ms | 19.09 ms |
114
+
115
+ ### 5. Levenshtein Cross-Product Matrix (`cdist`)
116
+ | Matrix Size | Total Pairs | `fuzzgpu` (GPU) | `fuzzgpu` (CPU) | `rapidfuzz` | python-Levenshtein | vs RF (GPU) | vs RF (CPU) |
117
+ | :--- | :---: | :---: | :---: | :---: | :---: | :---: | :---: |
118
+ | **10 × 10** | 100 | 0.06 ms | 0.03 ms | 0.01 ms | 0.05 ms | 0.12× | 0.22× |
119
+ | **50 × 50** | 2,500 | 0.61 ms | 0.11 ms | 0.04 ms | 1.20 ms | 0.07× | 0.38× |
120
+ | **100 × 100** | 10,000 | 0.81 ms | 0.28 ms | 0.25 ms | 6.69 ms | 0.31× | 0.90× |
121
+ | **200 × 200** | 40,000 | 1.56 ms | 1.56 ms | 1.13 ms | 41.35 ms | 0.73× | 0.73× |
122
+
123
+ ---
124
+
125
+ ## Installation
126
+
127
+ ### Python
128
+ ```bash
129
+ pip install fuzzgpu
130
+ ```
131
+
132
+ ### Rust (Cargo.toml)
133
+ ```toml
134
+ [dependencies]
135
+ fuzzgpu-core = "0.1.4"
136
+ ```
137
+
138
+ ---
139
+
140
+ ## Quickstart
141
+
142
+ ```python
143
+ import fuzzgpu
144
+ from fuzzgpu.fuzz import ratio, partial_ratio, token_sort_ratio, token_set_ratio, extract, extractOne
145
+
146
+ # 1. Classical Distance Metrics
147
+ lev = fuzzgpu.levenshtein_distance("kitten", "sitting") # 3
148
+ dam = fuzzgpu.damerau_levenshtein_distance("ab", "ba") # 1 (transposition-aware)
149
+ jw = fuzzgpu.jaro_winkler_similarity("MARTHA", "MARHTA", 0.1) # 0.9611
150
+
151
+ # 2. High-Throughput Batch Processing (Auto-dispatched to GPU/CPU)
152
+ candidates = ["hallo", "hullo", "jello", "yellow", "hello world"] * 10_000
153
+ distances = fuzzgpu.levenshtein_batch("hello", candidates)
154
+ jw_scores = fuzzgpu.jaro_winkler_batch("hello", candidates, prefix_weight=0.1)
155
+
156
+ # 3. 2D Cross-Product Distance Matrix (Dedicated 2D Grid Shader)
157
+ matrix = fuzzgpu.levenshtein_cdist(["abc", "def", "xyz"], ["abd", "axy", "def"])
158
+
159
+ # 3b. Zero-Allocation Outputs (write into preallocated numpy buffers — the
160
+ # rapidfuzz binding model: no per-call Python int boxing, no GC churn)
161
+ import numpy as np
162
+ out_u32 = np.empty(len(candidates), dtype=np.uint32)
163
+ out_f64 = np.empty(len(candidates), dtype=np.float64)
164
+ mat_u32 = np.empty((3, 3), dtype=np.uint32)
165
+ fuzzgpu.levenshtein_batch_into("hello", candidates, out_u32) # fills in place
166
+ fuzzgpu.jaro_winkler_batch_into("hello", candidates, out_f64) # jaro/winkler: float64
167
+ fuzzgpu.levenshtein_cdist_into(["abc", "def", "xyz"], ["abd", "axy", "def"], mat_u32)
168
+ # Supported: levenshtein/damerau/jaro batch + cdist. `out` must be a numpy
169
+ # array of the exact shape/dtype (uint32 for distances, float64 for Jaro);
170
+ # it is validated (length, dtype, writable, contiguous) before any compute,
171
+ # and left untouched if validation fails.
172
+
173
+ # 4. Global Sequence Alignment (Gotoh 1982 Linear & Affine Gap Penalties)
174
+ score_linear = fuzzgpu.needleman_wunsch_score("AGTACGCA", "TATGC", match=2, mismatch=-1, gap=-2)
175
+ score_affine = fuzzgpu.needleman_wunsch_affine("AGTACGCA", "TATGC", match=2, mismatch=-1, gap_open=-3, gap_extend=-1)
176
+
177
+ # 5. RapidFuzz-Compatible Scorer & Search API
178
+ score = ratio("fuzzy was a bear", "fuzzy was a bear") # 100.0
179
+ part = partial_ratio("hello", "oh hello there") # 100.0
180
+ tsr = token_sort_ratio("new york mets", "mets new york") # 100.0
181
+ tset = token_set_ratio("fuzzy was a bear", "fuzzy bear") # 100.0
182
+
183
+ # 6. Top-K Best Match Search
184
+ best = extractOne("hellp", ["hello", "world", "help"], score_cutoff=50.0)
185
+ # Output: ("hello", 80.0, 0)
186
+
187
+ top_3 = extract("apple", ["apply", "ape", "banana", "applesauce"], score_cutoff=50.0, limit=3)
188
+
189
+ # 7. Hardware Diagnostics
190
+ print(fuzzgpu.gpu_info())
191
+ # Output: Intel(R) Iris(R) Xe Graphics (Vulkan) / Apple M2 (Metal)
192
+
193
+ # Full routing diagnostics: adapter class, auto threshold, last routing.
194
+ print(fuzzgpu.hardware_info())
195
+ # Output: GPU: Intel(R) Iris(R) Xe Graphics (Vulkan, IntegratedGpu) |
196
+ # auto threshold: 500 | override: auto | last routing: 50000 GPU / 0 CPU pairs | ...
197
+
198
+ # GPU/CPU routing is backend-aware by default (discrete GPUs route earlier;
199
+ # integrated and software GPUs are conservative). Override it explicitly:
200
+ fuzzgpu.set_gpu_threshold(100) # force GPU dispatch for batches >= 100 pairs
201
+ fuzzgpu.set_gpu_threshold(None) # restore auto-selection from the adapter
202
+ ```
203
+
204
+ ---
205
+
206
+ ## Rust API (fuzzgpu-core)
207
+
208
+ Add the dependency — the `gpu` feature (WebGPU via `wgpu`) is on by default with an automatic Rayon CPU fallback; set `default-features = false` for a pure-CPU build:
209
+
210
+ ```toml
211
+ [dependencies]
212
+ fuzzgpu-core = "0.1.4" # GPU + CPU fallback
213
+ # fuzzgpu-core = { version = "0.1.4", default-features = false } # CPU-only
214
+ ```
215
+
216
+ ### Batch compute & cross-product matrix (GPU, with CPU fallback)
217
+
218
+ The GPU kernel lazily initializes the wgpu device and auto-routes every workload: empty/identical pairs short-circuit, strings over 256 chars and batches under 500 pairs run on Rayon CPU, and everything else dispatches to the compute shader in chunks sized to the adapter's buffer limits:
219
+
220
+ ```rust
221
+ use fuzzgpu_core::levenshtein::gpu_ext::GpuLevenshteinKernel;
222
+
223
+ fn main() -> fuzzgpu_core::Result<()> {
224
+ let kernel = GpuLevenshteinKernel::get()?; // lazy wgpu device + pipeline setup
225
+
226
+ // Batch: one query vs N candidates.
227
+ let pairs: Vec<(&str, &str)> = vec![
228
+ ("kitten", "sitting"),
229
+ ("kitten", "kittens"),
230
+ ("hello", "hullo"),
231
+ ];
232
+ let distances = kernel.compute(&pairs)?;
233
+ assert_eq!(distances, vec![3, 1, 1]);
234
+
235
+ // Cross-product N×M matrix via the dedicated 2D-grid shader.
236
+ let list_a = ["kitten", "sitting"];
237
+ let list_b = ["kitten", "mittens", "sitting"];
238
+ let matrix = kernel.compute_matrix(&list_a, &list_b)?;
239
+ assert_eq!(matrix, vec![vec![0, 2, 3], vec![3, 3, 0]]);
240
+ Ok(())
241
+ }
242
+ ```
243
+
244
+ Jaro-Winkler (`fuzzgpu_core::jaro::gpu_ext::GpuJaroKernel`) and affine Needleman-Wunsch (`fuzzgpu_core::needleman::gpu_ext::GpuNeedlemanAffineKernel`) kernels follow the same `get()` / batch / matrix pattern.
245
+
246
+ ### CPU-only build (`default-features = false`)
247
+
248
+ ```rust
249
+ use fuzzgpu_core::LevenshteinKernel;
250
+ use fuzzgpu_core::levenshtein::levenshtein_cdist_cpu;
251
+
252
+ let kernel = LevenshteinKernel;
253
+ let distances = kernel.compute(&pairs)?; // Rayon-parallel batch
254
+ let matrix = levenshtein_cdist_cpu(&list_a, &list_b); // Rayon-parallel matrix
255
+ ```
256
+
257
+ ### Single-value API
258
+
259
+ The crate root also exposes the scalar functions behind the Python and wasm APIs:
260
+
261
+ ```rust
262
+ use fuzzgpu_core::{
263
+ damerau_levenshtein_distance, extract, jaro_winkler, levenshtein_distance_raw,
264
+ needleman_wunsch, needleman_wunsch_affine, partial_ratio, ratio,
265
+ };
266
+
267
+ assert_eq!(levenshtein_distance_raw("kitten", "sitting"), 3);
268
+ assert_eq!(damerau_levenshtein_distance("ab", "ba"), 1);
269
+ assert_eq!(jaro_winkler("MARTHA", "MARHTA", 0.1), 0.9611111111111111);
270
+ assert_eq!(ratio("fuzzy was a bear", "fuzzy was a bear"), 100.0);
271
+ assert_eq!(needleman_wunsch("AGTACGCA", "TATGC", 2, -1, -2), 1);
272
+ ```
273
+
274
+ > With the `gpu` feature, fallible APIs return `fuzzgpu_core::Result<T>` (`FuzzGpuError`); without it, the same alias is `Result<T, String>`.
275
+
276
+ ---
277
+
278
+ ## WebAssembly (JavaScript API)
279
+
280
+ Build the wasm module for your target and import it like any ES module:
281
+
282
+ ```bash
283
+ cd crates/fuzzgpu-wasm
284
+ wasm-pack build --target web --release # browser (ESM)
285
+ wasm-pack build --target nodejs --release # Node.js (CommonJS)
286
+ ```
287
+
288
+ ```js
289
+ // Browser (ESM) — `init` is generated for --target web
290
+ import init, { levenshtein_distance, jaro_winkler, needleman_wunsch, ratio } from './pkg/fuzzgpu_wasm.js';
291
+ await init();
292
+
293
+ // Node.js (CommonJS): const fg = require('./pkg/fuzzgpu_wasm.js'); // no init needed
294
+
295
+ // Classic distance & similarity metrics return plain JS numbers
296
+ levenshtein_distance('kitten', 'sitting'); // 3
297
+ jaro_winkler('MARTHA', 'MARHTA', 0.1); // 0.9611...
298
+ ratio('fuzzy was a bear', 'fuzzy was a bear'); // 100.0
299
+
300
+ // Batch & search helpers (returns match objects)
301
+ extract('apple', ['apply', 'ape', 'banana'], 50.0, 3);
302
+ ```
303
+
304
+ ### Generated package layout
305
+
306
+ `bash build-wasm.sh` (or `wasm-pack build --target web --release` — the script just wraps it with a release profile and `--out-dir ../../pkg`) produces the browser package at the repo root `pkg/`:
307
+
308
+ | File | Purpose |
309
+ | :--- | :--- |
310
+ | `fuzzgpu_wasm.js` | **ES module entry** — exports the whole API plus a default `init()`. This is the file you import. |
311
+ | `fuzzgpu_wasm_bg.wasm` | The compiled **WebAssembly module** (~140 KB). It is *not* inlined — `init()` fetches it at runtime, so it must ship alongside the glue. |
312
+ | `fuzzgpu_wasm.d.ts` | **TypeScript declarations** for every export, including the `bigint` (i64) Needleman-Wunsch signatures. |
313
+ | `fuzzgpu_wasm_bg.wasm.d.ts` | wasm-level type declaration picked up by some TS tooling. |
314
+ | `package.json` | npm manifest: `"type": "module"`, `"main": "fuzzgpu_wasm.js"`, `"types": "fuzzgpu_wasm.d.ts"`, plus a `files` list that controls exactly what ships when you publish the package. |
315
+
316
+ > The `.js` glue and the `.wasm` are **two separate files that must be served together** — the browser fetches `fuzzgpu_wasm_bg.wasm` by URL after the glue module loads. Bundlers that tree-shake or inline assets need the explicit import wiring below.
317
+
318
+ ### Wiring into a bundler (Vite / webpack)
319
+
320
+ The generated `init()` resolves the `.wasm` with `new URL('fuzzgpu_wasm_bg.wasm', import.meta.url)`, which **Vite and webpack 5 both understand natively** — the default pattern needs zero bundler config:
321
+
322
+ ```js
323
+ // main.js — default pattern (Vite and webpack 5)
324
+ import init, { levenshtein_distance, ratio, needleman_wunsch } from './pkg/fuzzgpu_wasm.js';
325
+
326
+ // init() fetches + instantiates fuzzgpu_wasm_bg.wasm (resolved relative to
327
+ // this module's URL) and returns a promise. All exports throw until it resolves.
328
+ await init();
329
+
330
+ console.log(levenshtein_distance('kitten', 'sitting')); // 3
331
+ ```
332
+
333
+ - **Await `init()` exactly once at startup.** Top-level `await` works in both bundlers; otherwise wrap it in your app's async bootstrap. Calling any export before `init()` resolves throws.
334
+ - **TypeScript:** with `"moduleResolution": "bundler"` the adjacent `fuzzgpu_wasm.d.ts` is picked up automatically — BigInt score parameters are typed `bigint`.
335
+
336
+ If the `.wasm` lives somewhere non-default (CDN, custom base path, or a bundler that doesn't follow `new URL`), pass the URL explicitly — `init()` accepts a string / `URL` / `Response`, or an object `{ module_or_path }`:
337
+
338
+ ```js
339
+ // Vite — explicit asset URL via the ?url suffix
340
+ import wasmUrl from './pkg/fuzzgpu_wasm_bg.wasm?url';
341
+ await init(wasmUrl);
342
+
343
+ // webpack 5 — asset/resource emits the .wasm as a URL string
344
+ // (module.rules: { test: /\.wasm$/, type: 'asset/resource' })
345
+ import wasmUrl from './pkg/fuzzgpu_wasm_bg.wasm';
346
+ await init(wasmUrl);
347
+ ```
348
+
349
+ To consume the package from npm instead of a local path, publish the `pkg/` directory and import by name — `fuzzgpu_wasm.js` is `"main"` and the `files` array keeps the `.wasm` + `.d.ts` in the published tarball:
350
+
351
+ ```js
352
+ import init, { levenshtein_distance } from 'fuzzgpu-wasm';
353
+ await init();
354
+ ```
355
+
356
+ ### BigInt Needleman-Wunsch scores
357
+
358
+ Needleman-Wunsch alignment scores are 64-bit integers. wasm-bindgen maps `i64` to JavaScript **`BigInt`** — not `Number` — so scores beyond the 32-bit range are never truncated:
359
+
360
+ ```js
361
+ // Linear gap penalty — pass BigInt arguments, receive a BigInt back
362
+ const s = needleman_wunsch('AGTACGCA', 'TATGC', 2n, -1n, -2n);
363
+ // s === 1n
364
+
365
+ // Affine (Gotoh) gap penalties
366
+ const a = needleman_wunsch_affine('AGTACGCA', 'TATGC', 2n, -1n, -3n, -1n);
367
+ // a === -2n
368
+
369
+ // Scores far beyond i32::MAX (~2.1e9) survive exactly:
370
+ const long = 'A'.repeat(100);
371
+ needleman_wunsch(long, long, 30_000_000n, -1n, -2n);
372
+ // 3000000000n — exact BigInt, not wrapped/truncated
373
+ ```
374
+
375
+ > **Note:** The score parameters are `i64`, so they must be passed as `BigInt` literals (`2n`) — passing a plain `Number` throws a `TypeError` (verified by the test suite). `BigInt` requires a modern runtime: all current browsers, Node ≥ 10.4.
376
+
377
+ ---
378
+
379
+ ## Technical Architecture
380
+
381
+ `fuzzgpu` combines a tiered execution pipeline to balance low-latency single queries and high-throughput batch workloads:
382
+
383
+ ```
384
+ ┌──────────────────────────┐
385
+ │ User Query / API │
386
+ └─────────────┬────────────┘
387
+ │
388
+ Batch Size / Dataset Assessment
389
+ │
390
+ ┌───────────────────────┴───────────────────────┐
391
+ ▼ ▼
392
+ Small Workloads (< 500) Large Batches (≥ 500)
393
+ │ │
394
+ ┌───────────────────────────┐ ┌───────────────────────────┐
395
+ │ Rayon Multi-Threaded │ │ wgpu WebGPU Compute │
396
+ │ CPU Parallelism │ │ Shaders (Metal/Vulkan) │
397
+ │ - Myers 1999 Bit-Vector │ │ - 2D Workgroup Grids │
398
+ │ - Zero PCIe Latency │ │ - Streaming Chunking │
399
+ └───────────────────────────┘ └───────────────────────────┘
400
+ ```
401
+
402
+ ### Key Architectural Optimizations
403
+
404
+ 1. **GPU kernels for every metric**: Levenshtein runs the Myers bit-vector
405
+ shader (`levenshtein_myers.wgsl`, two-u32 bit-vector, no `SHADER_INT64`)
406
+ plus a row-wise Myers cdist kernel (`levenshtein_cdist_myers.wgsl`);
407
+ Jaro-Winkler runs a bitmap-matching shader (`jaro.wgsl` / `jaro_matrix.wgsl`
408
+ — 128-bit register bitmaps, transposed pair-major char layout for
409
+ coalesced loads, no per-thread arrays); Damerau runs a Lowrance-Wagner
410
+ shader (`damerau.wgsl` / `damerau_matrix.wgsl`) that keeps each pair's full
411
+ DP matrix in workgroup shared memory, bit-exact with the CPU reference
412
+ including non-adjacent transpositions (`ca`/`abc` = 2). Matrix shaders
413
+ upload List A and List B once ($O(N + M)$ bandwidth) instead of
414
+ duplicating pairs across PCIe.
415
+ 2. **Myers (1999) Bit-Parallel CPU Engine**:
416
+ For strings $\le 64$ characters, computes Levenshtein edit distance using bit-vector operations with zero inner dynamic programming loops ($O(N)$ execution).
417
+ 3. **Lowrance & Wagner (1975) Unrestricted Damerau-Levenshtein**:
418
+ Full support for character insertions, deletions, substitutions, and arbitrary transpositions.
419
+ 4. **Gotoh (1982) Affine Gap Sequence Alignment**:
420
+ Memory-efficient 3-state recurrence ($O(N)$ auxiliary space) for bioinformatics and long-sequence alignment.
421
+ 5. **Streaming Chunk Partitioner**:
422
+ Datasets exceeding GPU buffer limits (>128MB or >500,000 pairs) are automatically streamed in chunks to prevent VRAM overflow.
423
+ 6. **Metric-Aware Backend Routing**:
424
+ The Myers kernel wins on iGPUs at scale and is routed at the auto threshold; the heavier Jaro/Damerau kernels are auto-routed to CPU on integrated GPUs (measured 2–6× slower there at every scale) and to GPU on discrete GPUs above a scaled threshold. `hardware_info()` reports every routing decision; `set_gpu_threshold(n)` overrides.
425
+ 7. **ISA-Aware SIMD Kernels (Levenshtein Myers & Jaro)**:
426
+ The bit-parallel kernels dispatch at runtime to the widest available instruction set — **AVX512** (8 texts per 512-bit vector), **AVX2** (4 texts per 256-bit vector), **NEON** on aarch64 (2 texts per 128-bit vector), or a portable scalar fallback. Every kernel is differentially tested against the portable reference (AVX512/AVX2 on x86 CI, NEON on a native arm64 CI runner). To pin a specific ISA (e.g. to work around a 512-bit downclocking part, or for benchmarking) set `FUZZGPU_SIMD=portable|neon|avx2|avx512`. The GPU shaders are backend-agnostic WGSL (no `u64`, no adapter features) and run on Vulkan, Metal, DX12, and WebGPU.
427
+
428
+ ---
429
+
430
+ ## Project Structure
431
+
432
+ ```
433
+ fuzzgpu/
434
+ ├── assets/
435
+ │ └── logo.svg # Vector brand asset
436
+ ├── crates/
437
+ │ ├── fuzzgpu-core/ # Core Rust engine & compute shaders
438
+ │ │ ├── src/
439
+ │ │ │ ├── gpu.rs # wgpu instance and device singleton
440
+ │ │ │ ├── levenshtein.rs # Levenshtein kernel & 2D matrix dispatch
441
+ │ │ │ ├── damerau.rs # Lowrance-Wagner Damerau-Levenshtein
442
+ │ │ │ ├── needleman.rs # Needleman-Wunsch (Linear & Affine)
443
+ │ │ │ ├── jaro.rs # Jaro / Jaro-Winkler GPU & CPU kernels
444
+ │ │ │ ├── fuzz.rs # Fuzzy ratio, token sort/set, extract
445
+ │ │ │ ├── simd.rs # Myers bit-vector algorithms
446
+ │ │ │ └── shaders/ # WGSL compute shaders (1D & 2D)
447
+ │ ├── fuzzgpu-python/ # PyO3 CPython C-extension module
448
+ │ └── fuzzgpu-wasm/ # wasm-bindgen WebAssembly module
449
+ ├── python/
450
+ │ └── fuzzgpu/ # Python package wrapper & typing
451
+ ├── tests/
452
+ │ └── test_basic.py # Comprehensive test suite (50 tests)
453
+ └── benchmarks/
454
+ └── bench_compare.py # Comparative benchmarking harness
455
+ ```
456
+
457
+ ---
458
+
459
+ ## Building from Source
460
+
461
+ ### Prerequisites
462
+ - [Rust Toolchain (1.87+)](https://rustup.rs/) (wgpu 30 MSRV)
463
+ - Python 3.10+ & `pip install maturin`
464
+
465
+ ### Build Python Extension
466
+ ```bash
467
+ # Clone the repository
468
+ git clone https://github.com/Flaxmbot/fuzzgpu.git
469
+ cd fuzzgpu
470
+
471
+ # Build and install into current virtual environment
472
+ maturin develop --release
473
+ ```
474
+
475
+ ### Run Tests & Benchmarks
476
+ ```bash
477
+ # Run pytest verification suite
478
+ pytest tests/ -v
479
+
480
+ # Run comparative benchmark harness
481
+ python benchmarks/bench_compare.py
482
+ ```
483
+
484
+ ### Fuzz Testing
485
+
486
+ The `fuzz/` crate holds libFuzzer targets (nightly + `cargo fuzz run <target>`)
487
+ for Levenshtein, Jaro, Needleman-Wunsch, and the fuzzy ratios, each asserting
488
+ its fast path against a naive oracle. The same drivers run on **stable** via a
489
+ self-harness — `cargo test --manifest-path fuzz/Cargo.toml --lib --release` —
490
+ which is wired into CI, so the differential fuzz checks execute on every push
491
+ without a nightly toolchain.
492
+
493
+ GPU test writers: see [docs/GPU_TESTING.md](docs/GPU_TESTING.md) for the fault-injection hooks (timeout / buffer / shader-error branches) and the dispatch-lock, skip, and CI conventions every GPU test must follow.
494
+
495
+ ### Build WebAssembly (Browser Target)
496
+ ```bash
497
+ cd crates/fuzzgpu-wasm
498
+ wasm-pack build --target web --release
499
+ ```
500
+
501
+ See [WebAssembly (JavaScript API)](#webassembly-javascript-api) for the JS usage patterns and the BigInt scoring API.
502
+
503
+ ---
504
+
505
+ ## Releasing
506
+
507
+ ### WebAssembly (`wasm-v*`)
508
+
509
+ The wasm package is cut with the **Release wasm package** workflow — manually triggered from the Actions tab with the new semver version (e.g. `0.2.0`):
510
+
511
+ 1. **Run the workflow**: *Actions → Release wasm package → Run workflow*, enter the new version. It validates the semver and refuses to re-release the current version.
512
+ 2. **Bump**: rewrites `version` in `crates/fuzzgpu-wasm/Cargo.toml` (the wasm package's version — it is its own workspace, independent of the core/Python versions).
513
+ 3. **Gate**: runs the `#[wasm_bindgen_test]` suite (BigInt i64 signatures, >2⁵³ scores) before anything ships.
514
+ 4. **Build & verify**: builds with the exact user-facing `bash build-wasm.sh` and checks `pkg/package.json` carries the new version plus the `.wasm` magic bytes.
515
+ 5. **Release**: commits the bump, creates the `wasm-v<version>` tag and a GitHub release attaching `fuzzgpu_wasm_bg.wasm`, the JS glue, `.d.ts`, `package.json`, and a zip of the full `pkg/`.
516
+
517
+ ### Python / PyPI (`v*`)
518
+
519
+ The Python wheels take a different path — the **Release & Publish to PyPI** workflow, triggered by pushing a `v*` tag (e.g. `v0.2.0`): it builds Linux/macOS/Windows wheels with maturin, publishes them to PyPI, and attaches the wheels to the tag's GitHub release.
520
+
521
+ **Why two flows?** The wasm package is a browser artifact (ESM glue + `.wasm` + TypeScript types) whose version lives in `crates/fuzzgpu-wasm/Cargo.toml`; the Python package is a native extension published to PyPI and versioned with the core crate. Keeping them on separate tag namespaces (`wasm-v*` vs `v*`) lets each cut independently without colliding, and each workflow is self-contained (bump → test → build → attach) so nothing ships untested.
522
+
523
+ **Version parity is enforced.** fuzzgpu releases core/python/wasm in lockstep: the wasm workflow refuses to cut a version that doesn't equal the core crate's current version (`crates/fuzzgpu-core/Cargo.toml`), so the browser artifact can never drift from the Python package. To release a new version, bump core first (e.g. via the normal `v*` release process), then cut the wasm release at the same version — the parity check passes, and the wasm bump (`wasm-v<version>`) is exactly the core version.
524
+
525
+ ---
526
+
527
+ ## License
528
+
529
+ This project is licensed under the [MIT License](LICENSE).
assets/logo.png ADDED
Binary file
@@ -0,0 +1,32 @@
1
+ [package]
2
+ name = "fuzzgpu-core"
3
+ version = "0.1.5"
4
+ edition = "2021"
5
+ description = "GPU-accelerated fuzzy string matching engine"
6
+ license = "MIT"
7
+ repository = "https://github.com/kuntal-devrat/fuzzgpu"
8
+ homepage = "https://github.com/kuntal-devrat/fuzzgpu"
9
+ authors = ["Devrat Kuntal"]
10
+
11
+ [features]
12
+ default = ["gpu"]
13
+ gpu = ["wgpu", "pollster", "bytemuck"]
14
+
15
+ [dependencies]
16
+ wgpu = { workspace = true, optional = true }
17
+ pollster = { workspace = true, optional = true }
18
+ bytemuck = { workspace = true, optional = true }
19
+ thiserror = "2"
20
+ log = "0.4"
21
+ rayon = "1.10"
22
+
23
+ [dev-dependencies]
24
+ proptest = "1.11.0"
25
+ criterion = "0.5"
26
+ # Integration tests construct wgpu objects (e.g. pipeline layouts) to drive the
27
+ # public kernel-registration API; same version as the gpu feature's dep.
28
+ wgpu = { workspace = true }
29
+
30
+ [[bench]]
31
+ name = "bench"
32
+ harness = false