hashcodecs 1.2.1__tar.gz → 1.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/BENCHMARK.md +49 -6
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/CHANGELOG.md +39 -1
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/CITATION.cff +2 -2
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/Cargo.lock +16 -5
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/Cargo.toml +7 -19
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/PKG-INFO +27 -17
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/README.md +26 -16
- hashcodecs-1.3.0/benches/Cargo.toml +37 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/benches/base64.rs +18 -12
- hashcodecs-1.3.0/benches/crossover.rs +90 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/benches/murmur3.rs +1 -5
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/benches/support/mod.rs +2 -0
- hashcodecs-1.3.0/benches/xxhash.rs +175 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/build.rs +1 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/ARCHITECTURE.md +61 -62
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/api/base64.md +2 -2
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/api/murmur3.md +1 -1
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/api/xxh3.md +1 -1
- hashcodecs-1.3.0/docs/benchmarks/base64-python-batch-memoryview.svg +238 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/benchmarks/base64-python-batch-reusable.svg +46 -46
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/benchmarks/base64-python-batch.svg +44 -44
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/benchmarks/base64-python.svg +40 -40
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/benchmarks/murmur3-python.svg +44 -44
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/benchmarks/performance-at-a-glance.svg +17 -17
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/benchmarks/results.csv +154 -94
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/benchmarks/xxh3-python.svg +106 -106
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/performance.md +4 -4
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/hashcodecs/__init__.py +2 -2
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/hashcodecs/__init__.pyi +1 -1
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/hashcodecs/_hashcodecs.pyi +106 -85
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/hashcodecs/base64.py +2 -2
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/hashcodecs/base64.pyi +1 -1
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/hashcodecs/murmur3.py +2 -2
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/hashcodecs/murmur3.pyi +1 -1
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/hashcodecs/xxhash.py +2 -2
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/hashcodecs/xxhash.pyi +1 -1
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/hatch_build.py +58 -6
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/pyproject.toml +3 -1
- hashcodecs-1.3.0/src/backend.rs +178 -0
- hashcodecs-1.3.0/src/base64/backend.rs +102 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/base64/decode/aarch64.rs +25 -20
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/base64/decode/avx2.rs +38 -40
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/base64/decode/avx512.rs +8 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/base64/decode/sse41.rs +8 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/base64/decode/ssse3.rs +18 -29
- hashcodecs-1.3.0/src/base64/decode/tables.rs +33 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/base64/decode.rs +66 -63
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/base64/encode/aarch64.rs +9 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/base64/encode/avx2.rs +84 -36
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/base64/encode/avx512.rs +2 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/base64/encode/ssse3.rs +1 -1
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/base64/encode.rs +61 -40
- hashcodecs-1.3.0/src/base64/output_buffer.rs +19 -0
- hashcodecs-1.2.1/src/base64/dispatch.rs → hashcodecs-1.3.0/src/base64/runtime_dispatch.rs +54 -31
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/base64/tests.rs +82 -49
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/base64.rs +8 -10
- hashcodecs-1.3.0/src/bindings/base64/batch.rs +236 -0
- hashcodecs-1.3.0/src/bindings/base64/callbacks.rs +287 -0
- hashcodecs-1.3.0/src/bindings/base64/decode/batch.rs +125 -0
- hashcodecs-1.3.0/src/bindings/base64/decode/fallback.rs +146 -0
- hashcodecs-1.3.0/src/bindings/base64/decode/native/advanced/config.rs +136 -0
- hashcodecs-1.3.0/src/bindings/base64/decode/native/advanced/scanner.rs +354 -0
- hashcodecs-1.3.0/src/bindings/base64/decode/native/advanced/specials.rs +73 -0
- hashcodecs-1.3.0/src/bindings/base64/decode/native/advanced/staging.rs +152 -0
- hashcodecs-1.3.0/src/bindings/base64/decode/native/advanced.rs +183 -0
- hashcodecs-1.3.0/src/bindings/base64/decode/native/advanced_tests.rs +574 -0
- hashcodecs-1.3.0/src/bindings/base64/decode/native/lenient/compat.rs +18 -0
- hashcodecs-1.3.0/src/bindings/base64/decode/native/lenient/helpers/aarch64.rs +69 -0
- hashcodecs-1.3.0/src/bindings/base64/decode/native/lenient/helpers/mod.rs +130 -0
- hashcodecs-1.3.0/src/bindings/base64/decode/native/lenient/helpers/scalar.rs +41 -0
- hashcodecs-1.3.0/src/bindings/base64/decode/native/lenient/helpers/x86.rs +237 -0
- hashcodecs-1.3.0/src/bindings/base64/decode/native/lenient/mod.rs +131 -0
- hashcodecs-1.3.0/src/bindings/base64/decode/native/lenient/state_machine.rs +241 -0
- hashcodecs-1.3.0/src/bindings/base64/decode/native/strict.rs +444 -0
- hashcodecs-1.3.0/src/bindings/base64/decode/native.rs +23 -0
- hashcodecs-1.3.0/src/bindings/base64/decode/output.rs +424 -0
- hashcodecs-1.3.0/src/bindings/base64/decode/plan.rs +371 -0
- hashcodecs-1.3.0/src/bindings/base64/decode.rs +88 -0
- hashcodecs-1.3.0/src/bindings/base64/encode/batch.rs +140 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/bindings/base64/encode.rs +65 -26
- hashcodecs-1.3.0/src/bindings/base64/methods.rs +20 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/bindings/base64/mod.rs +11 -217
- hashcodecs-1.3.0/src/bindings/base64/schema.rs +223 -0
- hashcodecs-1.3.0/src/bindings/base64/schema_generated.rs +1247 -0
- hashcodecs-1.3.0/src/bindings/buffer.rs +714 -0
- hashcodecs-1.3.0/src/bindings/murmur3/incremental.rs +154 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/bindings/murmur3/methods.rs +12 -7
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/bindings/objects.rs +49 -17
- hashcodecs-1.3.0/src/bindings/xxhash/batch.rs +374 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/bindings/xxhash/methods.rs +17 -13
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/lib.rs +4 -5
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/murmur3/block_buffer.rs +7 -3
- hashcodecs-1.3.0/src/murmur3/dispatch.rs +53 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/murmur3/primitives.rs +2 -3
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/murmur3/tests.rs +93 -63
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/murmur3/x64_128/x86.rs +29 -30
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/murmur3/x64_128.rs +54 -43
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/murmur3/x86_128/x86.rs +26 -28
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/murmur3/x86_128.rs +42 -36
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/murmur3/x86_32/x86.rs +25 -27
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/murmur3/x86_32.rs +44 -38
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/murmur3.rs +2 -3
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/xxhash/batch.rs +178 -27
- {hashcodecs-1.2.1/src/xxhash/long → hashcodecs-1.3.0/src/xxhash/long_inputs}/aarch64.rs +46 -16
- hashcodecs-1.3.0/src/xxhash/long_inputs/scalar.rs +66 -0
- {hashcodecs-1.2.1/src/xxhash/long → hashcodecs-1.3.0/src/xxhash/long_inputs}/x86/avx2.rs +20 -15
- {hashcodecs-1.2.1/src/xxhash/long → hashcodecs-1.3.0/src/xxhash/long_inputs}/x86/avx2_batch.rs +11 -11
- {hashcodecs-1.2.1/src/xxhash/long → hashcodecs-1.3.0/src/xxhash/long_inputs}/x86/avx512.rs +13 -8
- {hashcodecs-1.2.1/src/xxhash/long → hashcodecs-1.3.0/src/xxhash/long_inputs}/x86/mod.rs +2 -0
- {hashcodecs-1.2.1/src/xxhash/long → hashcodecs-1.3.0/src/xxhash/long_inputs}/x86/ssse3.rs +13 -8
- hashcodecs-1.2.1/src/xxhash/long.rs → hashcodecs-1.3.0/src/xxhash/long_inputs.rs +110 -46
- hashcodecs-1.3.0/src/xxhash/one_shot.rs +73 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/xxhash/primitives.rs +17 -10
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/xxhash/proofs.rs +5 -5
- hashcodecs-1.3.0/src/xxhash/short_inputs.rs +188 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/xxhash/tests.rs +66 -39
- hashcodecs-1.3.0/src/xxhash.rs +23 -0
- hashcodecs-1.3.0/tools/generate_api_metadata.py +489 -0
- hashcodecs-1.3.0/tools/verify_sdist.py +126 -0
- hashcodecs-1.2.1/benches/xxhash.rs +0 -141
- hashcodecs-1.2.1/src/backend.rs +0 -241
- hashcodecs-1.2.1/src/base64/backend.rs +0 -94
- hashcodecs-1.2.1/src/base64/output.rs +0 -17
- hashcodecs-1.2.1/src/bindings/base64/callbacks.rs +0 -755
- hashcodecs-1.2.1/src/bindings/base64/decode/plan.rs +0 -176
- hashcodecs-1.2.1/src/bindings/base64/decode.rs +0 -1686
- hashcodecs-1.2.1/src/bindings/base64/methods.rs +0 -678
- hashcodecs-1.2.1/src/bindings/buffer.rs +0 -400
- hashcodecs-1.2.1/src/bindings/murmur3/incremental.rs +0 -383
- hashcodecs-1.2.1/src/bindings/xxhash/batch.rs +0 -270
- hashcodecs-1.2.1/src/murmur3/dispatch.rs +0 -51
- hashcodecs-1.2.1/src/xxhash/hash.rs +0 -71
- hashcodecs-1.2.1/src/xxhash/long/scalar.rs +0 -53
- hashcodecs-1.2.1/src/xxhash/short.rs +0 -177
- hashcodecs-1.2.1/src/xxhash.rs +0 -28
- hashcodecs-1.2.1/tools/generate_api_metadata.py +0 -235
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/.gitignore +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/LICENSE +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/LICENSE-MIT +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/SAFETY.md +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/SECURITY.md +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/base64.md +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/benchmarks/base64-python-batch-large.svg +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/benchmarks/base64-python-lenient.svg +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/benchmarks/base64-python-memoryview.svg +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/benchmarks/base64-python-mutable.svg +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/benchmarks/base64-python-reusable.svg +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/benchmarks/base64-rust.svg +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/benchmarks/murmur3-python-mutable.svg +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/benchmarks/murmur3-rust.svg +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/benchmarks/xxh3-rust-batch-remainders.svg +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/benchmarks/xxh3-rust.svg +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/compatibility.md +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/index.md +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/murmur3.md +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/requirements.txt +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/docs/xxh3.md +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/hashcodecs/py.typed +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/base64/alphabet.rs +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/base64/decode/x86_contracts.rs +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/base64/encode/cache.rs +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/base64/error.rs +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/base64/miri_tests.rs +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/base64/proofs.rs +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/base64/tests/aarch64.rs +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/bindings/arguments.rs +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/bindings/mod.rs +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/bindings/murmur3/digest.rs +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/bindings/murmur3/mod.rs +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/bindings/murmur3/one_shot.rs +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/bindings/runtime.rs +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/bindings/xxhash/mod.rs +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/murmur3/miri_tests.rs +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/murmur3/proofs.rs +0 -0
- {hashcodecs-1.2.1 → hashcodecs-1.3.0}/src/xxhash/miri_tests.rs +0 -0
|
@@ -6,7 +6,7 @@ Pin one logical CPU. Run each case in one thread. Collect 50 Rust samples and 15
|
|
|
6
6
|
baseline with AVX2, the backend that hashcodecs selects on this host. Higher throughput wins.
|
|
7
7
|
|
|
8
8
|
Build the Python wheel with CPython 3.12 and the full C API. Keep competitor values from the latest comparison run.
|
|
9
|
-
Use `uv run python benchmarks/render_charts.py` to render the charts. Read exact values in
|
|
9
|
+
Use `uv run --python 3.12 --no-project python benchmarks/render_charts.py` to render the charts. Read exact values in
|
|
10
10
|
[docs/benchmarks/results.csv](docs/benchmarks/results.csv).
|
|
11
11
|
|
|
12
12
|
## Timing Controls
|
|
@@ -16,12 +16,16 @@ sampling time per case is at least their product, plus calibration; use lower va
|
|
|
16
16
|
hashcodecs-only pass, use `--hashcodecs-only --samples 3 --minimum-sample-seconds 0.05` with each Python benchmark
|
|
17
17
|
script.
|
|
18
18
|
|
|
19
|
+
Each Rust Criterion harness collects 50 samples per case. Pass Criterion's `--sample-size` option for an exploratory
|
|
20
|
+
run with a different count.
|
|
21
|
+
|
|
19
22
|
## Python Call Costs
|
|
20
23
|
|
|
21
24
|
Run `python benchmarks/python_calls.py` to measure positional calls from 0 through 256 bytes in nanoseconds per
|
|
22
25
|
call. Use `--keywords` for positional and keyword calls at 64 bytes, or `--thresholds` for latency around the
|
|
23
26
|
GIL-detachment cutoffs. The `--thread-scaling` mode measures aggregate throughput with one, two, and four threads;
|
|
24
|
-
it does not pin the process to one logical CPU.
|
|
27
|
+
it does not pin the process to one logical CPU. Use `--buffer-inputs` to compare 64-byte and 4 KiB XXH3-64 calls
|
|
28
|
+
across bytes, full and sliced memoryviews, writable and non-contiguous views, and `array('B')`.
|
|
25
29
|
|
|
26
30
|
## XXH3
|
|
27
31
|
|
|
@@ -30,10 +34,15 @@ For Python, run the upstream `xxhash` extension beside hashcodecs. Pass 32 equal
|
|
|
30
34
|
The Rust remainder cases pass two or three equal-size long inputs. Run Python remainder cases with
|
|
31
35
|
`python benchmarks/python_xxhash.py --batch-counts 2 3`.
|
|
32
36
|
|
|
37
|
+
The Rust mixed benchmarks use `[1024, 1024, 4096, 4096]`, `[240, 240, 241, 241]`, and the reverse boundary order.
|
|
38
|
+
The 1024/4096 case measures adjacent two-item long runs. The 240/241 cases measure both orders across the
|
|
39
|
+
short/long dispatch boundary.
|
|
40
|
+
|
|
33
41
|
Use the focused one-shot run to cover the AVX2 four-chain boundaries:
|
|
34
42
|
|
|
35
43
|
```sh
|
|
36
|
-
cargo bench --bench xxhash --
|
|
44
|
+
cargo bench --manifest-path benches/Cargo.toml --bench xxhash -- "xxh3_(64|128)/(240|241|512|768|1024|1536|2048|4096)/hashcodecs"
|
|
45
|
+
cargo bench --manifest-path benches/Cargo.toml --bench xxhash -- "xxh3_batch/mixed/.*/hashcodecs_(64|128)"
|
|
37
46
|
```
|
|
38
47
|
|
|
39
48
|
[](docs/benchmarks/xxh3-rust.svg)
|
|
@@ -57,9 +66,10 @@ noisy cases insert `!` at the same boundaries. Both cases measure returned bytes
|
|
|
57
66
|
|
|
58
67
|
## Python Memoryview Inputs
|
|
59
68
|
|
|
60
|
-
Use `--memoryview-input` for full immutable views and `--sliced-memoryview-input` for equal-length views
|
|
61
|
-
nonzero starting offset.
|
|
62
|
-
|
|
69
|
+
Use `--memoryview-input` for full immutable views and `--sliced-memoryview-input` for equal-length contiguous views
|
|
70
|
+
with a nonzero starting offset. Full views can recover their exact immutable owner at detachment sizes; slices cover
|
|
71
|
+
offset-buffer handling, which borrows under the GIL and stabilizes the input in free-threaded builds. The encoded data
|
|
72
|
+
remains identical.
|
|
63
73
|
|
|
64
74
|
[](docs/benchmarks/base64-python-memoryview.svg)
|
|
65
75
|
|
|
@@ -76,6 +86,15 @@ into a decode investigation:
|
|
|
76
86
|
python benchmarks/python_base64_batch.py --item-sizes 4096 --batch-sizes 512 768 1024 1280 2048 --decode-only
|
|
77
87
|
```
|
|
78
88
|
|
|
89
|
+
Add `--memoryview-input` to wrap every matrix input in an exact memoryview. This mode compares independent views
|
|
90
|
+
against the matching one-item loops and reusable-output paths.
|
|
91
|
+
|
|
92
|
+
[](docs/benchmarks/base64-python-batch-memoryview.svg)
|
|
93
|
+
|
|
94
|
+
```sh
|
|
95
|
+
python benchmarks/python_base64_batch.py --item-sizes 1048576 --batch-sizes 8 --memoryview-input --decode-only
|
|
96
|
+
```
|
|
97
|
+
|
|
79
98
|
Use a single operation when recording a sampling profile, or compare traced allocations without a sampler:
|
|
80
99
|
|
|
81
100
|
```sh
|
|
@@ -118,3 +137,27 @@ Pass `bytearray` inputs to the Base64 API.
|
|
|
118
137
|
Pass `bytearray` inputs to the MurmurHash3 API.
|
|
119
138
|
|
|
120
139
|
[](docs/benchmarks/murmur3-python-mutable.svg)
|
|
140
|
+
|
|
141
|
+
## Reproduction
|
|
142
|
+
|
|
143
|
+
Run the benchmark
|
|
144
|
+
|
|
145
|
+
```
|
|
146
|
+
uv sync --python 3.12 --frozen --group benchmark --no-install-project
|
|
147
|
+
|
|
148
|
+
uv run --python 3.12 --refresh-package hashcodecs --no-project --with . --with mmh3==5.2.1 --with pybase64==1.4.3 --with xxhash==3.8.1 python benchmarks/python_base64.py --hashcodecs-only
|
|
149
|
+
|
|
150
|
+
uv run --python 3.12 --refresh-package hashcodecs --no-project --with . --with mmh3==5.2.1 --with pybase64==1.4.3 --with xxhash==3.8.1 python benchmarks/python_base64_batch.py --hashcodecs-only
|
|
151
|
+
|
|
152
|
+
uv run --python 3.12 --refresh-package hashcodecs --no-project --with . --with mmh3==5.2.1 --with pybase64==1.4.3 --with xxhash==3.8.1 python benchmarks/python_murmur3.py --hashcodecs-only
|
|
153
|
+
|
|
154
|
+
uv run --python 3.12 --refresh-package hashcodecs --no-project --with . --with mmh3==5.2.1 --with pybase64==1.4.3 --with xxhash==3.8.1 python benchmarks/python_murmur3.py --hashcodecs-only --incremental
|
|
155
|
+
|
|
156
|
+
uv run --python 3.12 --refresh-package hashcodecs --no-project --with . --with mmh3==5.2.1 --with pybase64==1.4.3 --with xxhash==3.8.1 python benchmarks/python_xxhash.py --hashcodecs-only
|
|
157
|
+
```
|
|
158
|
+
|
|
159
|
+
Update the documentation
|
|
160
|
+
|
|
161
|
+
```
|
|
162
|
+
uv run --python 3.12 --no-project python benchmarks/render_charts.py
|
|
163
|
+
```
|
|
@@ -4,6 +4,43 @@ This file records notable user-facing changes to `hashcodecs`. Version 1.0.0 sta
|
|
|
4
4
|
|
|
5
5
|
## [Unreleased]
|
|
6
6
|
|
|
7
|
+
## [1.3.0] - 2026-09-04
|
|
8
|
+
|
|
9
|
+
### What's Changed
|
|
10
|
+
* chore: expand XXH3 benchmark coverage by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/61
|
|
11
|
+
* refactor: split Base64 decoder bindings by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/62
|
|
12
|
+
* refactor: declare Base64 binding schema by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/63
|
|
13
|
+
* perf: inspect exact CPython memoryviews directly by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/64
|
|
14
|
+
* refactor: avoid copying Base64 fallback inputs by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/65
|
|
15
|
+
* chore: strengthen runtime coverage and benchmarks by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/66
|
|
16
|
+
* fix: avoid unnecessary batch snapshots and scalar grouping by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/67
|
|
17
|
+
* refactor: generate Base64 binding metadata by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/68
|
|
18
|
+
* perf: remove advanced decode and batch allocations by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/69
|
|
19
|
+
* refactor: finish native codec cleanup by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/70
|
|
20
|
+
* chore: verify source distributions in CI by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/71
|
|
21
|
+
* fix: stabilize aliased Base64 decode inputs by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/72
|
|
22
|
+
* perf: remove Python batch input copies by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/73
|
|
23
|
+
* fix: use strict fast paths for decode-into by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/74
|
|
24
|
+
* refactor: streamline native bindings and benchmarks by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/75
|
|
25
|
+
* fix: restore XXH fast paths and wheel compatibility by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/76
|
|
26
|
+
* refactor: use four-lane AArch64 XXH3 accumulation by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/77
|
|
27
|
+
* refactor: clarify internal names and technical prose by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/78
|
|
28
|
+
* fix: restore Base64 batch encode throughput by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/79
|
|
29
|
+
* refactor: clarify internal naming by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/80
|
|
30
|
+
* refactor: simplify internal decode and dispatch paths by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/81
|
|
31
|
+
* fix: harden hashers and optimize codec paths by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/82
|
|
32
|
+
* refactor: centralize Python Base64 decode routing by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/83
|
|
33
|
+
* refactor: standardize Rust callback and input names by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/84
|
|
34
|
+
* refactor: clarify codec routing and CPU capabilities by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/85
|
|
35
|
+
* refactor: consolidate base64 batch ownership by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/86
|
|
36
|
+
* refactor: reduce XXH3 dispatch overhead by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/87
|
|
37
|
+
* Fix Base64 batch alias stabilization and AVX2 streaming stores by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/88
|
|
38
|
+
* fix: restore Base64 encode fast paths by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/89
|
|
39
|
+
* fix: restore Base64 memoryview batch throughput by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/90
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
**Full Changelog**: https://github.com/kozistr/hashcodecs-rs/compare/v1.2.1...v1.3.0
|
|
43
|
+
|
|
7
44
|
## [1.2.1] - 2026-08-26
|
|
8
45
|
|
|
9
46
|
### What's Changed
|
|
@@ -127,7 +164,8 @@ This file records notable user-facing changes to `hashcodecs`. Version 1.0.0 sta
|
|
|
127
164
|
- Initial Python and Rust APIs for Base64 and MurmurHash3.
|
|
128
165
|
- Runtime SIMD dispatch and platform-specific CPython wheels.
|
|
129
166
|
|
|
130
|
-
[Unreleased]: https://github.com/kozistr/hashcodecs-rs/compare/v1.
|
|
167
|
+
[Unreleased]: https://github.com/kozistr/hashcodecs-rs/compare/v1.3.0...HEAD
|
|
168
|
+
[1.3.0]: https://github.com/kozistr/hashcodecs-rs/compare/v1.2.1...v1.3.0
|
|
131
169
|
[1.2.1]: https://github.com/kozistr/hashcodecs-rs/compare/v1.2.0...v1.2.1
|
|
132
170
|
[1.2.0]: https://github.com/kozistr/hashcodecs-rs/compare/v1.1.0...v1.2.0
|
|
133
171
|
[1.1.0]: https://github.com/kozistr/hashcodecs-rs/compare/v1.0.0...v1.1.0
|
|
@@ -5,8 +5,8 @@ authors:
|
|
|
5
5
|
given-names: Hyeongchan
|
|
6
6
|
orcid: https://orcid.org/0000-0002-1729-0580
|
|
7
7
|
title: "hashcodecs: SIMD-accelerated Base64, MurmurHash3, and XXH3 for Python and Rust"
|
|
8
|
-
version: 1.
|
|
9
|
-
date-released: 2026-
|
|
8
|
+
version: 1.3.0
|
|
9
|
+
date-released: 2026-09-04
|
|
10
10
|
license: "MIT OR Apache-2.0"
|
|
11
11
|
repository-code: "https://github.com/kozistr/hashcodecs-rs"
|
|
12
12
|
url: "https://github.com/kozistr/hashcodecs-rs"
|
|
@@ -194,21 +194,32 @@ dependencies = [
|
|
|
194
194
|
|
|
195
195
|
[[package]]
|
|
196
196
|
name = "hashcodecs"
|
|
197
|
-
version = "1.
|
|
197
|
+
version = "1.3.0"
|
|
198
|
+
dependencies = [
|
|
199
|
+
"base64",
|
|
200
|
+
"memchr",
|
|
201
|
+
"mimalloc",
|
|
202
|
+
"murmur3",
|
|
203
|
+
"pyo3",
|
|
204
|
+
"pyo3-build-config",
|
|
205
|
+
"xxhash-c-sys",
|
|
206
|
+
"xxhash-rust",
|
|
207
|
+
]
|
|
208
|
+
|
|
209
|
+
[[package]]
|
|
210
|
+
name = "hashcodecs-benchmarks"
|
|
211
|
+
version = "0.0.0"
|
|
198
212
|
dependencies = [
|
|
199
213
|
"base64",
|
|
200
214
|
"base64-turbo",
|
|
201
215
|
"criterion",
|
|
202
216
|
"fastmurmur3",
|
|
203
|
-
"
|
|
217
|
+
"hashcodecs",
|
|
204
218
|
"mimalloc",
|
|
205
219
|
"mm3h",
|
|
206
220
|
"murmur3",
|
|
207
221
|
"murmurs",
|
|
208
|
-
"pyo3",
|
|
209
|
-
"pyo3-build-config",
|
|
210
222
|
"xxhash-c-sys",
|
|
211
|
-
"xxhash-rust",
|
|
212
223
|
]
|
|
213
224
|
|
|
214
225
|
[[package]]
|
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
[package]
|
|
2
2
|
name = "hashcodecs"
|
|
3
|
-
version = "1.
|
|
3
|
+
version = "1.3.0"
|
|
4
4
|
edition = "2024"
|
|
5
5
|
rust-version = "1.89"
|
|
6
|
+
autobenches = false
|
|
6
7
|
description = "SIMD-accelerated Base64 codecs and fast MurmurHash3 and xxHash implementations"
|
|
7
8
|
license = "MIT OR Apache-2.0"
|
|
8
9
|
repository = "https://github.com/kozistr/hashcodecs-rs"
|
|
@@ -19,6 +20,11 @@ include = [
|
|
|
19
20
|
"/LICENSE-MIT",
|
|
20
21
|
]
|
|
21
22
|
|
|
23
|
+
[workspace]
|
|
24
|
+
members = [".", "benches"]
|
|
25
|
+
default-members = ["."]
|
|
26
|
+
resolver = "3"
|
|
27
|
+
|
|
22
28
|
[lib]
|
|
23
29
|
name = "hashcodecs"
|
|
24
30
|
crate-type = ["rlib", "cdylib"]
|
|
@@ -37,28 +43,10 @@ pyo3-build-config = { version = "0.29.2", optional = true }
|
|
|
37
43
|
|
|
38
44
|
[dev-dependencies]
|
|
39
45
|
base64 = "=0.23.1"
|
|
40
|
-
base64-turbo = "=0.3.0"
|
|
41
|
-
criterion = { version = "=0.8.2", default-features = false, features = ["cargo_bench_support"] }
|
|
42
|
-
fastmurmur3 = "=0.2.0"
|
|
43
|
-
mimalloc = "=0.1.52"
|
|
44
|
-
mm3h = "=0.1.3"
|
|
45
46
|
murmur3 = "=0.5.2"
|
|
46
|
-
murmurs = "=1.0.5"
|
|
47
47
|
xxhash-c-sys = "=0.8.7"
|
|
48
48
|
xxhash-rust = { version = "=0.8.18", features = ["xxh3"] }
|
|
49
49
|
|
|
50
|
-
[[bench]]
|
|
51
|
-
name = "base64"
|
|
52
|
-
harness = false
|
|
53
|
-
|
|
54
|
-
[[bench]]
|
|
55
|
-
name = "murmur3"
|
|
56
|
-
harness = false
|
|
57
|
-
|
|
58
|
-
[[bench]]
|
|
59
|
-
name = "xxhash"
|
|
60
|
-
harness = false
|
|
61
|
-
|
|
62
50
|
[[test]]
|
|
63
51
|
name = "sanitizers"
|
|
64
52
|
path = "tests/sanitizers.rs"
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: hashcodecs
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.3.0
|
|
4
4
|
Summary: SIMD-accelerated Base64, MurmurHash3, and xxHash codecs
|
|
5
5
|
Project-URL: Documentation, https://hashcodecs-rs.readthedocs.io/
|
|
6
6
|
Project-URL: Repository, https://github.com/kozistr/hashcodecs-rs
|
|
@@ -43,20 +43,20 @@ Description-Content-Type: text/markdown
|
|
|
43
43
|
|
|
44
44
|
<p align="center">
|
|
45
45
|
<a href="BENCHMARK.md">
|
|
46
|
-
<img src="docs/benchmarks/performance-at-a-glance.svg" alt="CPython 3.12
|
|
46
|
+
<img src="docs/benchmarks/performance-at-a-glance.svg" alt="CPython 3.12 standard Base64 encoding and decoding benchmark">
|
|
47
47
|
</a>
|
|
48
48
|
</p>
|
|
49
49
|
|
|
50
50
|
SIMD-accelerated Base64, MurmurHash3, and XXH3 for Python and Rust.
|
|
51
51
|
|
|
52
52
|
Move byte-heavy work into Rust without changing your Python inputs. `hashcodecs` accepts `bytes`, `bytearray`, and
|
|
53
|
-
`memoryview`, selects the
|
|
53
|
+
`memoryview`, selects the highest-priority supported SIMD backend, and exposes batch and reusable-buffer APIs.
|
|
54
54
|
|
|
55
55
|
## Features
|
|
56
56
|
|
|
57
57
|
- Base64 encode and decode with standard, URL-safe, padded, unpadded, wrapped, and canonical modes.
|
|
58
58
|
- MurmurHash3 x86-32, x86-128, and x64-128 with one-shot and incremental APIs.
|
|
59
|
-
- Bit-for-bit compatible XXH3-64 and XXH3-128 with
|
|
59
|
+
- Bit-for-bit compatible XXH3-64 and XXH3-128 with allocating and allocation-free native batch APIs.
|
|
60
60
|
- Caller-managed `*_into` outputs for allocation-sensitive workloads.
|
|
61
61
|
- Runtime dispatch across AVX-512, AVX2, SSE4.1, SSSE3, NEON, and scalar implementations where applicable.
|
|
62
62
|
- Direct CPython buffer handling for `bytes`, `bytearray`, and `memoryview` inputs.
|
|
@@ -146,17 +146,25 @@ assert_eq!(
|
|
|
146
146
|
hashcodecs::xxhash::xxh3_64(b"", 0),
|
|
147
147
|
0x2d06_8005_38d3_94c2
|
|
148
148
|
);
|
|
149
|
+
|
|
150
|
+
let inputs: &[&[u8]] = &[b"hello", b"world"];
|
|
151
|
+
let mut hashes = [0_u64; 2];
|
|
152
|
+
let mut index = 0;
|
|
153
|
+
hashcodecs::xxhash::xxh3_64_batch_for_each(inputs, 0, |hash| {
|
|
154
|
+
hashes[index] = hash;
|
|
155
|
+
index += 1;
|
|
156
|
+
});
|
|
157
|
+
assert_eq!(index, inputs.len());
|
|
149
158
|
```
|
|
150
159
|
|
|
151
160
|
## Architecture
|
|
152
161
|
|
|
153
|
-
The Rust core owns algorithm behavior and SIMD dispatch.
|
|
154
|
-
|
|
155
|
-
adding per-call wrappers.
|
|
162
|
+
The Rust core owns algorithm behavior and SIMD dispatch. The CPython layer handles argument parsing, buffers,
|
|
163
|
+
reusable outputs, and GIL decisions. Root-level Python modules provide typed exports without per-call wrappers.
|
|
156
164
|
|
|
157
|
-
Each Rust algorithm exposes a small public
|
|
165
|
+
Each Rust algorithm exposes a small public module. Base64 groups internals by encode and decode operation and
|
|
158
166
|
places ISA kernels such as `encode/avx2.rs` and `decode/ssse3.rs` under their operation. MurmurHash3 groups code by
|
|
159
|
-
canonical variant. XXH3 uses processing-stage modules
|
|
167
|
+
canonical variant. XXH3 uses processing-stage modules. Long-input ISA kernels are under `xxhash/long_inputs/`.
|
|
160
168
|
|
|
161
169
|
See [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md) for the module layout, dispatch model, algorithm data flows,
|
|
162
170
|
CPython boundary, and safety invariants.
|
|
@@ -206,9 +214,10 @@ Read the focused cases, commands, and values in [BENCHMARK.md](BENCHMARK.md). Re
|
|
|
206
214
|
Comparison crates and Python packages are development-only dependencies and are not included in consumer builds.
|
|
207
215
|
|
|
208
216
|
```sh
|
|
209
|
-
cargo bench --bench base64
|
|
210
|
-
cargo bench --bench murmur3
|
|
211
|
-
cargo bench --bench xxhash
|
|
217
|
+
cargo bench --manifest-path benches/Cargo.toml --bench base64
|
|
218
|
+
cargo bench --manifest-path benches/Cargo.toml --bench murmur3
|
|
219
|
+
cargo bench --manifest-path benches/Cargo.toml --bench xxhash
|
|
220
|
+
cargo bench --manifest-path benches/Cargo.toml --bench crossover
|
|
212
221
|
|
|
213
222
|
uv sync --group benchmark --no-install-project
|
|
214
223
|
uv run --no-project --with . python benchmarks/python_base64.py
|
|
@@ -219,21 +228,22 @@ uv run --no-project --with . python benchmarks/python_xxhash.py
|
|
|
219
228
|
```
|
|
220
229
|
|
|
221
230
|
The Python benchmarks expose focused modes such as `--into`, `--lenient`, `--bytearray-input`, `--memoryview-input`,
|
|
222
|
-
`--sliced-memoryview-input`, `--incremental`, `--large`, and `--hashcodecs-only`. All scripts also
|
|
223
|
-
`--minimum-sample-seconds`; use `--help` on a benchmark script for its supported modes and
|
|
231
|
+
`--sliced-memoryview-input`, `--buffer-inputs`, `--incremental`, `--large`, and `--hashcodecs-only`. All scripts also
|
|
232
|
+
accept `--samples` and `--minimum-sample-seconds`; use `--help` on a benchmark script for its supported modes and
|
|
233
|
+
defaults.
|
|
224
234
|
|
|
225
235
|
For the same-ISA Windows XXH3 comparison shown above, rebuild the C baseline with:
|
|
226
236
|
|
|
227
237
|
```powershell
|
|
228
238
|
$env:CFLAGS='/O2 /arch:AVX2'
|
|
229
239
|
cargo clean -p xxhash-c-sys
|
|
230
|
-
cargo bench --bench xxhash
|
|
240
|
+
cargo bench --manifest-path benches/Cargo.toml --bench xxhash
|
|
231
241
|
```
|
|
232
242
|
|
|
233
243
|
## Performance snapshot
|
|
234
244
|
|
|
235
|
-
In the full 2026-
|
|
236
|
-
|
|
245
|
+
In the full 2026-09-02 hashcodecs-only run on the benchmark host, `hashcodecs.xxh3_64` processes a 1 MiB input at
|
|
246
|
+
90.07 GiB/s. With 256 B items in batches of 64, the Base64 batch API reaches 11.84 GiB/s for encode and 6.23 GiB/s for
|
|
237
247
|
decode. The run pins one logical CPU and uses 15 samples with a 0.2-second minimum per sample. Read the
|
|
238
248
|
[benchmark details](BENCHMARK.md) and [raw comparison results](docs/benchmarks/results.csv).
|
|
239
249
|
|
|
@@ -12,20 +12,20 @@
|
|
|
12
12
|
|
|
13
13
|
<p align="center">
|
|
14
14
|
<a href="BENCHMARK.md">
|
|
15
|
-
<img src="docs/benchmarks/performance-at-a-glance.svg" alt="CPython 3.12
|
|
15
|
+
<img src="docs/benchmarks/performance-at-a-glance.svg" alt="CPython 3.12 standard Base64 encoding and decoding benchmark">
|
|
16
16
|
</a>
|
|
17
17
|
</p>
|
|
18
18
|
|
|
19
19
|
SIMD-accelerated Base64, MurmurHash3, and XXH3 for Python and Rust.
|
|
20
20
|
|
|
21
21
|
Move byte-heavy work into Rust without changing your Python inputs. `hashcodecs` accepts `bytes`, `bytearray`, and
|
|
22
|
-
`memoryview`, selects the
|
|
22
|
+
`memoryview`, selects the highest-priority supported SIMD backend, and exposes batch and reusable-buffer APIs.
|
|
23
23
|
|
|
24
24
|
## Features
|
|
25
25
|
|
|
26
26
|
- Base64 encode and decode with standard, URL-safe, padded, unpadded, wrapped, and canonical modes.
|
|
27
27
|
- MurmurHash3 x86-32, x86-128, and x64-128 with one-shot and incremental APIs.
|
|
28
|
-
- Bit-for-bit compatible XXH3-64 and XXH3-128 with
|
|
28
|
+
- Bit-for-bit compatible XXH3-64 and XXH3-128 with allocating and allocation-free native batch APIs.
|
|
29
29
|
- Caller-managed `*_into` outputs for allocation-sensitive workloads.
|
|
30
30
|
- Runtime dispatch across AVX-512, AVX2, SSE4.1, SSSE3, NEON, and scalar implementations where applicable.
|
|
31
31
|
- Direct CPython buffer handling for `bytes`, `bytearray`, and `memoryview` inputs.
|
|
@@ -115,17 +115,25 @@ assert_eq!(
|
|
|
115
115
|
hashcodecs::xxhash::xxh3_64(b"", 0),
|
|
116
116
|
0x2d06_8005_38d3_94c2
|
|
117
117
|
);
|
|
118
|
+
|
|
119
|
+
let inputs: &[&[u8]] = &[b"hello", b"world"];
|
|
120
|
+
let mut hashes = [0_u64; 2];
|
|
121
|
+
let mut index = 0;
|
|
122
|
+
hashcodecs::xxhash::xxh3_64_batch_for_each(inputs, 0, |hash| {
|
|
123
|
+
hashes[index] = hash;
|
|
124
|
+
index += 1;
|
|
125
|
+
});
|
|
126
|
+
assert_eq!(index, inputs.len());
|
|
118
127
|
```
|
|
119
128
|
|
|
120
129
|
## Architecture
|
|
121
130
|
|
|
122
|
-
The Rust core owns algorithm behavior and SIMD dispatch.
|
|
123
|
-
|
|
124
|
-
adding per-call wrappers.
|
|
131
|
+
The Rust core owns algorithm behavior and SIMD dispatch. The CPython layer handles argument parsing, buffers,
|
|
132
|
+
reusable outputs, and GIL decisions. Root-level Python modules provide typed exports without per-call wrappers.
|
|
125
133
|
|
|
126
|
-
Each Rust algorithm exposes a small public
|
|
134
|
+
Each Rust algorithm exposes a small public module. Base64 groups internals by encode and decode operation and
|
|
127
135
|
places ISA kernels such as `encode/avx2.rs` and `decode/ssse3.rs` under their operation. MurmurHash3 groups code by
|
|
128
|
-
canonical variant. XXH3 uses processing-stage modules
|
|
136
|
+
canonical variant. XXH3 uses processing-stage modules. Long-input ISA kernels are under `xxhash/long_inputs/`.
|
|
129
137
|
|
|
130
138
|
See [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md) for the module layout, dispatch model, algorithm data flows,
|
|
131
139
|
CPython boundary, and safety invariants.
|
|
@@ -175,9 +183,10 @@ Read the focused cases, commands, and values in [BENCHMARK.md](BENCHMARK.md). Re
|
|
|
175
183
|
Comparison crates and Python packages are development-only dependencies and are not included in consumer builds.
|
|
176
184
|
|
|
177
185
|
```sh
|
|
178
|
-
cargo bench --bench base64
|
|
179
|
-
cargo bench --bench murmur3
|
|
180
|
-
cargo bench --bench xxhash
|
|
186
|
+
cargo bench --manifest-path benches/Cargo.toml --bench base64
|
|
187
|
+
cargo bench --manifest-path benches/Cargo.toml --bench murmur3
|
|
188
|
+
cargo bench --manifest-path benches/Cargo.toml --bench xxhash
|
|
189
|
+
cargo bench --manifest-path benches/Cargo.toml --bench crossover
|
|
181
190
|
|
|
182
191
|
uv sync --group benchmark --no-install-project
|
|
183
192
|
uv run --no-project --with . python benchmarks/python_base64.py
|
|
@@ -188,21 +197,22 @@ uv run --no-project --with . python benchmarks/python_xxhash.py
|
|
|
188
197
|
```
|
|
189
198
|
|
|
190
199
|
The Python benchmarks expose focused modes such as `--into`, `--lenient`, `--bytearray-input`, `--memoryview-input`,
|
|
191
|
-
`--sliced-memoryview-input`, `--incremental`, `--large`, and `--hashcodecs-only`. All scripts also
|
|
192
|
-
`--minimum-sample-seconds`; use `--help` on a benchmark script for its supported modes and
|
|
200
|
+
`--sliced-memoryview-input`, `--buffer-inputs`, `--incremental`, `--large`, and `--hashcodecs-only`. All scripts also
|
|
201
|
+
accept `--samples` and `--minimum-sample-seconds`; use `--help` on a benchmark script for its supported modes and
|
|
202
|
+
defaults.
|
|
193
203
|
|
|
194
204
|
For the same-ISA Windows XXH3 comparison shown above, rebuild the C baseline with:
|
|
195
205
|
|
|
196
206
|
```powershell
|
|
197
207
|
$env:CFLAGS='/O2 /arch:AVX2'
|
|
198
208
|
cargo clean -p xxhash-c-sys
|
|
199
|
-
cargo bench --bench xxhash
|
|
209
|
+
cargo bench --manifest-path benches/Cargo.toml --bench xxhash
|
|
200
210
|
```
|
|
201
211
|
|
|
202
212
|
## Performance snapshot
|
|
203
213
|
|
|
204
|
-
In the full 2026-
|
|
205
|
-
|
|
214
|
+
In the full 2026-09-02 hashcodecs-only run on the benchmark host, `hashcodecs.xxh3_64` processes a 1 MiB input at
|
|
215
|
+
90.07 GiB/s. With 256 B items in batches of 64, the Base64 batch API reaches 11.84 GiB/s for encode and 6.23 GiB/s for
|
|
206
216
|
decode. The run pins one logical CPU and uses 15 samples with a 0.2-second minimum per sample. Read the
|
|
207
217
|
[benchmark details](BENCHMARK.md) and [raw comparison results](docs/benchmarks/results.csv).
|
|
208
218
|
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
[package]
|
|
2
|
+
name = "hashcodecs-benchmarks"
|
|
3
|
+
version = "0.0.0"
|
|
4
|
+
edition = "2024"
|
|
5
|
+
publish = false
|
|
6
|
+
|
|
7
|
+
[dev-dependencies]
|
|
8
|
+
base64 = "=0.23.1"
|
|
9
|
+
base64-turbo = "=0.3.0"
|
|
10
|
+
criterion = { version = "=0.8.2", default-features = false, features = ["cargo_bench_support"] }
|
|
11
|
+
fastmurmur3 = "=0.2.0"
|
|
12
|
+
hashcodecs = { path = ".." }
|
|
13
|
+
mimalloc = "=0.1.52"
|
|
14
|
+
mm3h = "=0.1.3"
|
|
15
|
+
murmur3 = "=0.5.2"
|
|
16
|
+
murmurs = "=1.0.5"
|
|
17
|
+
xxhash-c-sys = "=0.8.7"
|
|
18
|
+
|
|
19
|
+
[[bench]]
|
|
20
|
+
name = "base64"
|
|
21
|
+
path = "base64.rs"
|
|
22
|
+
harness = false
|
|
23
|
+
|
|
24
|
+
[[bench]]
|
|
25
|
+
name = "murmur3"
|
|
26
|
+
path = "murmur3.rs"
|
|
27
|
+
harness = false
|
|
28
|
+
|
|
29
|
+
[[bench]]
|
|
30
|
+
name = "xxhash"
|
|
31
|
+
path = "xxhash.rs"
|
|
32
|
+
harness = false
|
|
33
|
+
|
|
34
|
+
[[bench]]
|
|
35
|
+
name = "crossover"
|
|
36
|
+
path = "crossover.rs"
|
|
37
|
+
harness = false
|
|
@@ -7,7 +7,6 @@ use criterion::{BenchmarkId, Criterion, Throughput, criterion_group, criterion_m
|
|
|
7
7
|
mod support;
|
|
8
8
|
|
|
9
9
|
const SIZES: [usize; 4] = [1024, 4 * 1024, 1024 * 1024, 8 * 1024 * 1024];
|
|
10
|
-
const SAMPLE_SIZE: usize = 50;
|
|
11
10
|
|
|
12
11
|
fn data(size: usize) -> Vec<u8> {
|
|
13
12
|
(0..size)
|
|
@@ -23,6 +22,17 @@ macro_rules! benchmark {
|
|
|
23
22
|
};
|
|
24
23
|
}
|
|
25
24
|
|
|
25
|
+
macro_rules! benchmark_encode {
|
|
26
|
+
($group:expr, $size:expr, $input:expr, $name:literal, $function:expr) => {
|
|
27
|
+
$group.bench_with_input(BenchmarkId::new($name, $size), $input, |bench, input| {
|
|
28
|
+
bench.iter(|| {
|
|
29
|
+
let output = black_box(($function)(black_box(input)));
|
|
30
|
+
black_box(output.bytes().fold(0_u8, u8::wrapping_add))
|
|
31
|
+
});
|
|
32
|
+
});
|
|
33
|
+
};
|
|
34
|
+
}
|
|
35
|
+
|
|
26
36
|
fn base64(c: &mut Criterion) {
|
|
27
37
|
support::pin_to_one_cpu();
|
|
28
38
|
standard_encode(c);
|
|
@@ -42,19 +52,18 @@ fn standard_encode(c: &mut Criterion) {
|
|
|
42
52
|
);
|
|
43
53
|
assert_eq!(base64_turbo::STANDARD.encode(&input), expected);
|
|
44
54
|
|
|
45
|
-
group.sample_size(SAMPLE_SIZE);
|
|
46
55
|
group.throughput(Throughput::Bytes(size as u64));
|
|
47
|
-
|
|
56
|
+
benchmark_encode!(
|
|
48
57
|
group,
|
|
49
58
|
size,
|
|
50
59
|
&input,
|
|
51
60
|
"hashcodecs",
|
|
52
61
|
hashcodecs::base64::b64encode
|
|
53
62
|
);
|
|
54
|
-
|
|
63
|
+
benchmark_encode!(group, size, &input, "base64", |input: &[u8]| {
|
|
55
64
|
base64::engine::general_purpose::STANDARD.encode(input)
|
|
56
65
|
});
|
|
57
|
-
|
|
66
|
+
benchmark_encode!(group, size, &input, "base64-turbo", |input: &[u8]| {
|
|
58
67
|
base64_turbo::STANDARD.encode(input)
|
|
59
68
|
});
|
|
60
69
|
}
|
|
@@ -72,19 +81,18 @@ fn urlsafe_encode(c: &mut Criterion) {
|
|
|
72
81
|
);
|
|
73
82
|
assert_eq!(base64_turbo::URL_SAFE.encode(&input), expected);
|
|
74
83
|
|
|
75
|
-
group.sample_size(SAMPLE_SIZE);
|
|
76
84
|
group.throughput(Throughput::Bytes(size as u64));
|
|
77
|
-
|
|
85
|
+
benchmark_encode!(
|
|
78
86
|
group,
|
|
79
87
|
size,
|
|
80
88
|
&input,
|
|
81
89
|
"hashcodecs",
|
|
82
90
|
hashcodecs::base64::b64encode_urlsafe
|
|
83
91
|
);
|
|
84
|
-
|
|
92
|
+
benchmark_encode!(group, size, &input, "base64", |input: &[u8]| {
|
|
85
93
|
base64::engine::general_purpose::URL_SAFE.encode(input)
|
|
86
94
|
});
|
|
87
|
-
|
|
95
|
+
benchmark_encode!(group, size, &input, "base64-turbo", |input: &[u8]| {
|
|
88
96
|
base64_turbo::URL_SAFE.encode(input)
|
|
89
97
|
});
|
|
90
98
|
}
|
|
@@ -104,7 +112,6 @@ fn standard_decode(c: &mut Criterion) {
|
|
|
104
112
|
);
|
|
105
113
|
assert_eq!(base64_turbo::STANDARD.decode(&input).unwrap(), expected);
|
|
106
114
|
|
|
107
|
-
group.sample_size(SAMPLE_SIZE);
|
|
108
115
|
group.throughput(Throughput::Bytes(size as u64));
|
|
109
116
|
benchmark!(
|
|
110
117
|
group,
|
|
@@ -142,7 +149,6 @@ fn urlsafe_decode(c: &mut Criterion) {
|
|
|
142
149
|
);
|
|
143
150
|
assert_eq!(base64_turbo::URL_SAFE.decode(&input).unwrap(), expected);
|
|
144
151
|
|
|
145
|
-
group.sample_size(SAMPLE_SIZE);
|
|
146
152
|
group.throughput(Throughput::Bytes(size as u64));
|
|
147
153
|
benchmark!(
|
|
148
154
|
group,
|
|
@@ -171,7 +177,7 @@ criterion_group! {
|
|
|
171
177
|
name = benches;
|
|
172
178
|
config = Criterion::default()
|
|
173
179
|
.measurement_time(Duration::from_secs(1))
|
|
174
|
-
.sample_size(
|
|
180
|
+
.sample_size(support::SAMPLE_SIZE)
|
|
175
181
|
.warm_up_time(Duration::from_millis(500));
|
|
176
182
|
targets = base64
|
|
177
183
|
}
|