hashcodecs 1.3.0__tar.gz → 1.4.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/.gitignore +3 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/BENCHMARK.md +125 -6
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/CHANGELOG.md +50 -1
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/CITATION.cff +2 -2
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/Cargo.lock +1 -1
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/Cargo.toml +2 -1
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/PKG-INFO +18 -12
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/README.md +17 -11
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/SAFETY.md +17 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/benches/crossover.rs +9 -5
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/benches/xxhash.rs +74 -3
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/docs/ARCHITECTURE.md +30 -9
- hashcodecs-1.4.1/docs/requirements.txt +2 -0
- hashcodecs-1.3.0/src/bindings/base64/schema_generated.rs → hashcodecs-1.4.1/generated/rust/binding_schema.rs +563 -51
- hashcodecs-1.4.1/generated/rust/murmur3_classes.rs +34 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/hashcodecs/_hashcodecs.pyi +20 -17
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/pyproject.toml +21 -11
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/alphabet.rs +14 -8
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/backend.rs +1 -0
- hashcodecs-1.4.1/src/base64/decode/aarch64.rs +354 -0
- hashcodecs-1.4.1/src/base64/decode/avx2.rs +333 -0
- hashcodecs-1.4.1/src/base64/decode/avx512.rs +355 -0
- hashcodecs-1.4.1/src/base64/decode/sse41.rs +129 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/decode/ssse3.rs +99 -9
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/decode/tables.rs +26 -1
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/decode/x86_contracts.rs +37 -4
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/decode.rs +198 -21
- hashcodecs-1.4.1/src/base64/encode/aarch64.rs +274 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/encode/avx2.rs +260 -101
- hashcodecs-1.4.1/src/base64/encode/avx512.rs +309 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/encode/cache.rs +6 -0
- hashcodecs-1.4.1/src/base64/encode/ssse3.rs +165 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/encode.rs +290 -5
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/miri_tests.rs +2 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/output_buffer.rs +2 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/proofs.rs +2 -0
- hashcodecs-1.4.1/src/base64/runtime_dispatch.rs +918 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/tests/aarch64.rs +45 -7
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/tests.rs +630 -18
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64.rs +11 -5
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/bindings/arguments.rs +11 -75
- hashcodecs-1.3.0/src/bindings/base64/callbacks.rs → hashcodecs-1.4.1/src/bindings/base64/api.rs +69 -54
- hashcodecs-1.4.1/src/bindings/base64/batch.rs +454 -0
- hashcodecs-1.4.1/src/bindings/base64/configured.rs +861 -0
- hashcodecs-1.4.1/src/bindings/base64/configured_tests.rs +915 -0
- hashcodecs-1.4.1/src/bindings/base64/decode.rs +1034 -0
- hashcodecs-1.4.1/src/bindings/base64/encode.rs +714 -0
- hashcodecs-1.4.1/src/bindings/base64/lenient.rs +622 -0
- hashcodecs-1.4.1/src/bindings/base64/policy.rs +508 -0
- {hashcodecs-1.3.0/src/bindings/base64/decode/native/lenient/helpers → hashcodecs-1.4.1/src/bindings/base64/scan}/aarch64.rs +44 -1
- hashcodecs-1.4.1/src/bindings/base64/scan/scalar.rs +60 -0
- {hashcodecs-1.3.0/src/bindings/base64/decode/native/lenient/helpers → hashcodecs-1.4.1/src/bindings/base64/scan}/x86.rs +86 -27
- hashcodecs-1.4.1/src/bindings/base64/scan.rs +182 -0
- hashcodecs-1.4.1/src/bindings/base64/staging.rs +380 -0
- hashcodecs-1.4.1/src/bindings/base64/strict.rs +277 -0
- hashcodecs-1.4.1/src/bindings/base64.rs +14 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/bindings/buffer.rs +288 -16
- hashcodecs-1.4.1/src/bindings/compatibility.rs +113 -0
- hashcodecs-1.4.1/src/bindings/murmur3/callbacks.rs +85 -0
- hashcodecs-1.4.1/src/bindings/murmur3/digest.rs +37 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/bindings/murmur3/incremental.rs +6 -37
- hashcodecs-1.4.1/src/bindings/murmur3/methods.rs +20 -0
- hashcodecs-1.3.0/src/bindings/murmur3/mod.rs → hashcodecs-1.4.1/src/bindings/murmur3.rs +1 -1
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/bindings/objects.rs +41 -54
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/bindings/runtime.rs +3 -1
- {hashcodecs-1.3.0/src/bindings/base64 → hashcodecs-1.4.1/src/bindings}/schema.rs +61 -19
- hashcodecs-1.4.1/src/bindings/xxhash/batch.rs +578 -0
- hashcodecs-1.4.1/src/bindings/xxhash/callbacks.rs +122 -0
- {hashcodecs-1.3.0/src/bindings/base64 → hashcodecs-1.4.1/src/bindings/xxhash}/methods.rs +2 -2
- hashcodecs-1.4.1/src/bindings/xxhash.rs +5 -0
- hashcodecs-1.3.0/src/bindings/mod.rs → hashcodecs-1.4.1/src/bindings.rs +2 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3/dispatch.rs +4 -2
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3/tests.rs +74 -21
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3/x64_128.rs +1 -7
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3/x86_128/x86.rs +22 -12
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3/x86_128.rs +1 -7
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3/x86_32.rs +1 -7
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/batch.rs +161 -51
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/long_inputs/aarch64.rs +3 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/long_inputs/scalar.rs +9 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/long_inputs/x86/avx2.rs +22 -5
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/long_inputs/x86/avx2_batch.rs +10 -4
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/long_inputs/x86/avx512.rs +7 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/long_inputs/x86/ssse3.rs +8 -0
- hashcodecs-1.3.0/src/xxhash/long_inputs/x86/mod.rs → hashcodecs-1.4.1/src/xxhash/long_inputs/x86.rs +0 -2
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/long_inputs.rs +72 -108
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/one_shot.rs +5 -2
- hashcodecs-1.4.1/src/xxhash/prepared.rs +136 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/primitives.rs +37 -9
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/proofs.rs +8 -2
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/short_inputs.rs +110 -19
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/tests.rs +127 -16
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash.rs +2 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/tools/generate_api_metadata.py +76 -153
- hashcodecs-1.4.1/tools/install_local_wheel.py +30 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/tools/verify_sdist.py +1 -0
- hashcodecs-1.3.0/docs/benchmarks/base64-python-batch-large.svg +0 -85
- hashcodecs-1.3.0/docs/benchmarks/base64-python-batch-memoryview.svg +0 -238
- hashcodecs-1.3.0/docs/benchmarks/base64-python-batch-reusable.svg +0 -106
- hashcodecs-1.3.0/docs/benchmarks/base64-python-batch.svg +0 -313
- hashcodecs-1.3.0/docs/benchmarks/base64-python-lenient.svg +0 -127
- hashcodecs-1.3.0/docs/benchmarks/base64-python-memoryview.svg +0 -159
- hashcodecs-1.3.0/docs/benchmarks/base64-python-mutable.svg +0 -83
- hashcodecs-1.3.0/docs/benchmarks/base64-python-reusable.svg +0 -115
- hashcodecs-1.3.0/docs/benchmarks/base64-python.svg +0 -203
- hashcodecs-1.3.0/docs/benchmarks/base64-rust.svg +0 -203
- hashcodecs-1.3.0/docs/benchmarks/murmur3-python-mutable.svg +0 -169
- hashcodecs-1.3.0/docs/benchmarks/murmur3-python.svg +0 -235
- hashcodecs-1.3.0/docs/benchmarks/murmur3-rust.svg +0 -193
- hashcodecs-1.3.0/docs/benchmarks/performance-at-a-glance.svg +0 -37
- hashcodecs-1.3.0/docs/benchmarks/results.csv +0 -587
- hashcodecs-1.3.0/docs/benchmarks/xxh3-python.svg +0 -181
- hashcodecs-1.3.0/docs/benchmarks/xxh3-rust-batch-remainders.svg +0 -139
- hashcodecs-1.3.0/docs/benchmarks/xxh3-rust.svg +0 -169
- hashcodecs-1.3.0/docs/requirements.txt +0 -2
- hashcodecs-1.3.0/src/base64/decode/aarch64.rs +0 -230
- hashcodecs-1.3.0/src/base64/decode/avx2.rs +0 -180
- hashcodecs-1.3.0/src/base64/decode/avx512.rs +0 -125
- hashcodecs-1.3.0/src/base64/decode/sse41.rs +0 -59
- hashcodecs-1.3.0/src/base64/encode/aarch64.rs +0 -122
- hashcodecs-1.3.0/src/base64/encode/avx512.rs +0 -134
- hashcodecs-1.3.0/src/base64/encode/ssse3.rs +0 -86
- hashcodecs-1.3.0/src/base64/runtime_dispatch.rs +0 -369
- hashcodecs-1.3.0/src/bindings/base64/batch.rs +0 -236
- hashcodecs-1.3.0/src/bindings/base64/decode/batch.rs +0 -125
- hashcodecs-1.3.0/src/bindings/base64/decode/fallback.rs +0 -146
- hashcodecs-1.3.0/src/bindings/base64/decode/native/advanced/config.rs +0 -136
- hashcodecs-1.3.0/src/bindings/base64/decode/native/advanced/scanner.rs +0 -354
- hashcodecs-1.3.0/src/bindings/base64/decode/native/advanced/specials.rs +0 -73
- hashcodecs-1.3.0/src/bindings/base64/decode/native/advanced/staging.rs +0 -152
- hashcodecs-1.3.0/src/bindings/base64/decode/native/advanced.rs +0 -183
- hashcodecs-1.3.0/src/bindings/base64/decode/native/advanced_tests.rs +0 -574
- hashcodecs-1.3.0/src/bindings/base64/decode/native/lenient/compat.rs +0 -18
- hashcodecs-1.3.0/src/bindings/base64/decode/native/lenient/helpers/mod.rs +0 -130
- hashcodecs-1.3.0/src/bindings/base64/decode/native/lenient/helpers/scalar.rs +0 -41
- hashcodecs-1.3.0/src/bindings/base64/decode/native/lenient/mod.rs +0 -131
- hashcodecs-1.3.0/src/bindings/base64/decode/native/lenient/state_machine.rs +0 -241
- hashcodecs-1.3.0/src/bindings/base64/decode/native/strict.rs +0 -444
- hashcodecs-1.3.0/src/bindings/base64/decode/native.rs +0 -23
- hashcodecs-1.3.0/src/bindings/base64/decode/output.rs +0 -424
- hashcodecs-1.3.0/src/bindings/base64/decode/plan.rs +0 -371
- hashcodecs-1.3.0/src/bindings/base64/decode.rs +0 -88
- hashcodecs-1.3.0/src/bindings/base64/encode/batch.rs +0 -140
- hashcodecs-1.3.0/src/bindings/base64/encode.rs +0 -260
- hashcodecs-1.3.0/src/bindings/base64/mod.rs +0 -293
- hashcodecs-1.3.0/src/bindings/murmur3/digest.rs +0 -25
- hashcodecs-1.3.0/src/bindings/murmur3/methods.rs +0 -108
- hashcodecs-1.3.0/src/bindings/murmur3/one_shot.rs +0 -86
- hashcodecs-1.3.0/src/bindings/xxhash/batch.rs +0 -374
- hashcodecs-1.3.0/src/bindings/xxhash/methods.rs +0 -205
- hashcodecs-1.3.0/src/bindings/xxhash/mod.rs +0 -195
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/LICENSE +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/LICENSE-MIT +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/SECURITY.md +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/benches/Cargo.toml +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/benches/base64.rs +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/benches/murmur3.rs +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/benches/support/mod.rs +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/build.rs +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/docs/api/base64.md +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/docs/api/murmur3.md +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/docs/api/xxh3.md +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/docs/base64.md +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/docs/compatibility.md +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/docs/index.md +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/docs/murmur3.md +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/docs/performance.md +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/docs/xxh3.md +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/hashcodecs/__init__.py +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/hashcodecs/__init__.pyi +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/hashcodecs/base64.py +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/hashcodecs/base64.pyi +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/hashcodecs/murmur3.py +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/hashcodecs/murmur3.pyi +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/hashcodecs/py.typed +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/hashcodecs/xxhash.py +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/hashcodecs/xxhash.pyi +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/hatch_build.py +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/backend.rs +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/error.rs +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/lib.rs +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3/block_buffer.rs +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3/miri_tests.rs +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3/primitives.rs +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3/proofs.rs +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3/x64_128/x86.rs +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3/x86_32/x86.rs +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3.rs +0 -0
- {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/miri_tests.rs +0 -0
|
@@ -9,6 +9,9 @@ Build the Python wheel with CPython 3.12 and the full C API. Keep competitor val
|
|
|
9
9
|
Use `uv run --python 3.12 --no-project python benchmarks/render_charts.py` to render the charts. Read exact values in
|
|
10
10
|
[docs/benchmarks/results.csv](docs/benchmarks/results.csv).
|
|
11
11
|
|
|
12
|
+
Python values use CPython 3.12.10 and report the median of 15 samples lasting at least 0.2 seconds each, with one
|
|
13
|
+
logical CPU pinned. Focused runs refresh only the affected series; other values retain their previous measurements.
|
|
14
|
+
|
|
12
15
|
## Timing Controls
|
|
13
16
|
|
|
14
17
|
Every Python benchmark accepts `--samples` (default: 15) and `--minimum-sample-seconds` (default: 0.2). Their
|
|
@@ -27,6 +30,43 @@ GIL-detachment cutoffs. The `--thread-scaling` mode measures aggregate throughpu
|
|
|
27
30
|
it does not pin the process to one logical CPU. Use `--buffer-inputs` to compare 64-byte and 4 KiB XXH3-64 calls
|
|
28
31
|
across bytes, full and sliced memoryviews, writable and non-contiguous views, and `array('B')`.
|
|
29
32
|
|
|
33
|
+
## Rust MurmurHash3 x64 Dispatch
|
|
34
|
+
|
|
35
|
+
Use scalar below 512 bytes of full blocks, then AVX2 when available. The SSE4.1 fallback starts at 512 bytes
|
|
36
|
+
and retains its 8 MiB upper limit. Incremental updates apply these thresholds to each batch of full blocks.
|
|
37
|
+
|
|
38
|
+
Forced-backend measurements with complete finalization on the Core Ultra 7 265K put scalar and AVX2 near parity
|
|
39
|
+
at 384 bytes (39.99 and 40.20 ns/hash), with AVX2 ahead at 512 bytes (53.98 and 52.76 ns/hash). Use 512 bytes as
|
|
40
|
+
a crossover candidate for this host. The shared minimum also avoids the measured SSE4.1 overhead on small inputs;
|
|
41
|
+
it does not establish an SSE4.1 crossover or an optimum across Intel and AMD CPUs.
|
|
42
|
+
|
|
43
|
+
The following public Rust API measurements include runtime dispatch and finalization. On 2026-09-13, we collected
|
|
44
|
+
50 Criterion samples per case with one logical CPU pinned, a 300 ms warmup, and a 1 s measurement target. These
|
|
45
|
+
values report Criterion's mean estimate with seed 42; they are separate from the forced-backend measurements.
|
|
46
|
+
|
|
47
|
+
| Input | ns/hash |
|
|
48
|
+
| --- | ---: |
|
|
49
|
+
| 15 B | 5.32 |
|
|
50
|
+
| 16 B | 5.47 |
|
|
51
|
+
| 31 B | 5.89 |
|
|
52
|
+
| 32 B | 6.03 |
|
|
53
|
+
| 64 B | 7.80 |
|
|
54
|
+
| 255 B | 24.42 |
|
|
55
|
+
| 256 B | 25.04 |
|
|
56
|
+
| 384 B | 37.93 |
|
|
57
|
+
| 511 B | 50.35 |
|
|
58
|
+
| 512 B | 49.79 |
|
|
59
|
+
| 513 B | 50.43 |
|
|
60
|
+
| 1 KiB | 96.89 |
|
|
61
|
+
|
|
62
|
+
We refreshed the x64 hashcodecs series in the Rust MurmurHash3 throughput chart on the same date, retaining the
|
|
63
|
+
x86 and competitor measurements from the previous comparison run.
|
|
64
|
+
|
|
65
|
+
```sh
|
|
66
|
+
cargo bench --manifest-path benches/Cargo.toml --bench crossover -- murmur_x64_128_crossover
|
|
67
|
+
cargo bench --manifest-path benches/Cargo.toml --bench murmur3 -- x64_128/hashcodecs
|
|
68
|
+
```
|
|
69
|
+
|
|
30
70
|
## XXH3
|
|
31
71
|
|
|
32
72
|
For the Rust comparison, link hashcodecs with xxHash 0.8.3 through `xxhash-c-sys`. Build the C baseline with AVX2.
|
|
@@ -34,15 +74,85 @@ For Python, run the upstream `xxhash` extension beside hashcodecs. Pass 32 equal
|
|
|
34
74
|
The Rust remainder cases pass two or three equal-size long inputs. Run Python remainder cases with
|
|
35
75
|
`python benchmarks/python_xxhash.py --batch-counts 2 3`.
|
|
36
76
|
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
77
|
+
To measure a branch against its parent, build both extensions with the same Python and Rust toolchains, then run:
|
|
78
|
+
|
|
79
|
+
```sh
|
|
80
|
+
uv run --frozen --no-sync python benchmarks/compare_xxh3_batches.py path/to/parent/_hashcodecs.pyd path/to/branch/_hashcodecs.pyd
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
Use the corresponding `.so` paths on Linux or macOS. The comparison loads both builds into one interpreter, pins
|
|
84
|
+
one CPU, and alternates their timing order across 15 samples. It checks matching digests and covers bytes,
|
|
85
|
+
bytearrays, and writable memoryviews. Use `--batch-counts 2 9 32 33 --sizes 64` to inspect small batches and the
|
|
86
|
+
32-result stack boundary. Positive `change_percent` values mean higher branch throughput.
|
|
87
|
+
|
|
88
|
+
The [32-item parent comparison](docs/benchmarks/xxh3-batch-parent-comparison.csv) records CPython 3.12.10 results
|
|
89
|
+
against parent commit `f17ab86`, measured on 2026-09-05. These paired measurements also cover the 256 KiB
|
|
90
|
+
GIL-detachment threshold and 1 MiB items. The Python XXH3 batch panels report CPython 3.12.10 measurements
|
|
91
|
+
from 2026-09-13, using 15 samples of at least 0.2 seconds; one-shot and upstream values retain their prior measurements.
|
|
92
|
+
|
|
93
|
+
Across the 30 paired 32-item cases, branch throughput ranges from 2.59% lower to 2.70% higher than the parent.
|
|
94
|
+
The [stack-boundary comparison](docs/benchmarks/xxh3-batch-boundary-comparison.csv) covers counts 2 and 33 with
|
|
95
|
+
64-byte items: two-item bytearray batches lose 5.40–6.05%, and 33-item bytes batches lose 6.06–7.05%. These
|
|
96
|
+
measurements show residual overhead for some small-input batches; they do not establish zero regression.
|
|
97
|
+
|
|
98
|
+
### Packed-batch detachment
|
|
99
|
+
|
|
100
|
+
XXH3 batches release the GIL at 1 MiB of total input or 16,384 items, provided the input buffers permit detachment.
|
|
101
|
+
The item limit accounts for per-item hashing and output work, including empty inputs. One-shot XXH3 keeps its
|
|
102
|
+
256 KiB threshold. Mutable inputs retain their existing synchronization rules.
|
|
103
|
+
|
|
104
|
+
The detached packed path retains input owners, borrows 64 inputs at a time on the stack, and stages results until
|
|
105
|
+
it reacquires the GIL and rechecks the output size. This removes one allocation and 16 bytes of temporary input
|
|
106
|
+
descriptors per item on 64-bit hosts. Little-endian hosts copy staged results in one operation.
|
|
107
|
+
|
|
108
|
+
The [candidate measurements](docs/benchmarks/xxh3-packed-candidates.csv) compare 256 KiB, 512 KiB, and 1 MiB
|
|
109
|
+
byte thresholds with the same allocation reduction and 16,384-item limit. Each exploratory value uses five
|
|
110
|
+
samples of at least 0.03 seconds. The 1 MiB policy keeps both 4,096- and 8,192-item batches of 64-byte inputs on
|
|
111
|
+
the direct-output path. This trades longer GIL holds for lower latency; it does not establish an optimal threshold
|
|
112
|
+
for other processors or contended workloads.
|
|
113
|
+
|
|
114
|
+
The [selected-policy measurements](docs/benchmarks/xxh3-packed-thresholds.csv) use CPython 3.15.0b4 on the
|
|
115
|
+
Intel Core Ultra 7 265K, with one logical CPU pinned, independent input allocations, and 15 samples of at least
|
|
116
|
+
0.2 seconds each, measured on 2026-09-13. They cover both digest widths and item sizes of 64 bytes, 1 KiB, and 64 KiB.
|
|
117
|
+
For 64-byte inputs:
|
|
118
|
+
|
|
119
|
+
| Items | XXH3-64 packed latency | XXH3-128 packed latency |
|
|
120
|
+
| ---: | ---: | ---: |
|
|
121
|
+
| 4,095 | 13.92 µs | 27.29 µs |
|
|
122
|
+
| 4,096 | 14.48 µs | 27.33 µs |
|
|
123
|
+
| 4,097 | 13.95 µs | 27.44 µs |
|
|
124
|
+
| 8,191 | 28.95 µs | 54.72 µs |
|
|
125
|
+
| 8,192 | 29.14 µs | 54.61 µs |
|
|
126
|
+
| 8,193 | 29.46 µs | 54.57 µs |
|
|
127
|
+
| 16,383 | 58.49 µs | 109.12 µs |
|
|
128
|
+
| 16,384 | 92.78 µs | 144.04 µs |
|
|
129
|
+
| 16,385 | 94.52 µs | 145.62 µs |
|
|
130
|
+
|
|
131
|
+
A discontinuity remains at the new detachment boundary because retaining owners and staging output still cost work.
|
|
132
|
+
The longest measured attached call near these boundaries takes about 109 microseconds. This is a latency
|
|
133
|
+
measurement on this host, not a bound on thread waiting time; the tests also verify GIL progress for large inputs
|
|
134
|
+
and for 16,384-item batches of empty, one-byte, and 64-byte inputs.
|
|
135
|
+
|
|
136
|
+
```sh
|
|
137
|
+
uv run --frozen --no-sync python benchmarks/python_xxhash_thresholds.py --output docs/benchmarks/xxh3-packed-thresholds.csv
|
|
138
|
+
uv run --frozen --no-sync python benchmarks/python_xxhash.py --batches-only --hashcodecs-only
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
`--thresholds-kib` selects benchmark input sizes around each candidate; it does not change the extension's
|
|
142
|
+
compiled policy. To compare candidates, rebuild the wheel after changing `BATCH_DETACH_BYTES` in
|
|
143
|
+
`src/bindings/xxhash/batch.rs`.
|
|
144
|
+
|
|
145
|
+
The Rust mixed benchmarks use `[1024, 1024, 4096, 4096]`, `[257, 258, 259, 260]`, `[240, 240, 241, 241]`, and the
|
|
146
|
+
reverse boundary order. The 1024/4096 case measures adjacent two-item long runs. The 257–260 case measures a
|
|
147
|
+
four-item run with one shared stripe count and distinct final stripes. The 240/241 cases measure both orders across
|
|
148
|
+
the short/long dispatch boundary.
|
|
40
149
|
|
|
41
150
|
Use the focused one-shot run to cover the AVX2 four-chain boundaries:
|
|
42
151
|
|
|
43
152
|
```sh
|
|
44
153
|
cargo bench --manifest-path benches/Cargo.toml --bench xxhash -- "xxh3_(64|128)/(240|241|512|768|1024|1536|2048|4096)/hashcodecs"
|
|
45
154
|
cargo bench --manifest-path benches/Cargo.toml --bench xxhash -- "xxh3_batch/mixed/.*/hashcodecs_(64|128)"
|
|
155
|
+
cargo bench --manifest-path benches/Cargo.toml --bench xxhash -- "xxh3_prepared"
|
|
46
156
|
```
|
|
47
157
|
|
|
48
158
|
[](docs/benchmarks/xxh3-rust.svg)
|
|
@@ -62,14 +172,23 @@ Pass one reusable `bytearray` to each `*_into` call.
|
|
|
62
172
|
Run `python benchmarks/python_base64.py --lenient`. The MIME cases insert CRLF after each 76-character line. The
|
|
63
173
|
noisy cases insert `!` at the same boundaries. Both cases measure returned bytes and reusable output buffers.
|
|
64
174
|
|
|
175
|
+
Run `uv run --no-project --python 3.12 python benchmarks/python_base64.py --custom-lenient` to measure clean and noisy
|
|
176
|
+
inputs with `@#` altchars. For a 1 MiB decoded payload, the custom clean case reaches 3.44 GiB/s with returned bytes
|
|
177
|
+
and 13.05 GiB/s with a reusable output buffer. The decoder translates complete symbol runs in a 4 KiB staging buffer
|
|
178
|
+
for SIMD decoding and handles padding and ignored characters through the lenient state machine.
|
|
179
|
+
|
|
65
180
|
[](docs/benchmarks/base64-python-lenient.svg)
|
|
66
181
|
|
|
182
|
+
## Wrapped Python Base64
|
|
183
|
+
|
|
184
|
+
Run `python benchmarks/python_base64.py --wrapped` with CPython 3.15 or newer. The benchmark inserts newlines after
|
|
185
|
+
76 output characters and measures returned bytes and a reusable `bytearray`.
|
|
186
|
+
|
|
67
187
|
## Python Memoryview Inputs
|
|
68
188
|
|
|
69
189
|
Use `--memoryview-input` for full immutable views and `--sliced-memoryview-input` for equal-length contiguous views
|
|
70
|
-
with a nonzero starting offset.
|
|
71
|
-
|
|
72
|
-
remains identical.
|
|
190
|
+
with a nonzero starting offset. At detachment sizes, both retain the immutable bytes owner without copying the input.
|
|
191
|
+
Callback-capable decode arguments keep their required snapshots and conversion order. Both modes use identical data.
|
|
73
192
|
|
|
74
193
|
[](docs/benchmarks/base64-python-memoryview.svg)
|
|
75
194
|
|
|
@@ -4,6 +4,53 @@ This file records notable user-facing changes to `hashcodecs`. Version 1.0.0 sta
|
|
|
4
4
|
|
|
5
5
|
## [Unreleased]
|
|
6
6
|
|
|
7
|
+
## [1.4.1] - 2026-09-13
|
|
8
|
+
|
|
9
|
+
### What's Changed
|
|
10
|
+
* fix: match CPython Base64 and XXH3 buffer semantics by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/114
|
|
11
|
+
* fix: preserve Base64 callback buffer semantics by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/115
|
|
12
|
+
* fix: preserve CPython Base64 observable behavior by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/116
|
|
13
|
+
* fix: resolve repository type diagnostics by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/117
|
|
14
|
+
* fix: release Base64 input buffers before writing output by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/118
|
|
15
|
+
* refactor: centralize buffer callback policy and clarify finalizers by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/119
|
|
16
|
+
* fix: accelerate custom-alphabet lenient Base64 decoding by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/120
|
|
17
|
+
* fix: reduce XXH3 packed batch detachment overhead by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/121
|
|
18
|
+
* fix: delay MurmurHash3 x64 SIMD dispatch until 512 bytes by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/122
|
|
19
|
+
* fix: restore Python buffer and batch throughput by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/123
|
|
20
|
+
* fix: typo in the performance viz image by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/124
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
**Full Changelog**: https://github.com/kozistr/hashcodecs-rs/compare/v1.4.0...v1.4.1
|
|
24
|
+
|
|
25
|
+
## [1.4.0] - 2026-09-10
|
|
26
|
+
|
|
27
|
+
### What's Changed
|
|
28
|
+
* refactor: flatten internal module layout by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/91
|
|
29
|
+
* refactor: prepare Base64 decoder policies once by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/92
|
|
30
|
+
* refactor: consolidate binding and dispatch cleanup by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/93
|
|
31
|
+
* refactor: simplify project module layout by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/94
|
|
32
|
+
* refactor: prepare codec policies and unify bindings by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/95
|
|
33
|
+
* refactor: accelerate Base64 codec hot paths by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/96
|
|
34
|
+
* fix: protect XXH3 batches from GC reentrancy by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/97
|
|
35
|
+
* refactor: unify Base64 decode execution by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/98
|
|
36
|
+
* refactor: flatten Base64 binding ownership by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/99
|
|
37
|
+
* refactor: improve XXH3 loads and internal naming by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/100
|
|
38
|
+
* refactor: reduce Base64 decoder setup and retry costs by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/101
|
|
39
|
+
* fix: preserve Base64 decode boundaries and restore ARM coverage by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/102
|
|
40
|
+
* refactor: keep XXH3 AVX2 tail chains in registers by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/103
|
|
41
|
+
* refactor: store AVX2 Base64 vectors in smaller groups by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/104
|
|
42
|
+
* fix: reduce Python overhead and align canonical decoding by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/106
|
|
43
|
+
* refactor: reduce cached AVX2 encoder setup by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/107
|
|
44
|
+
* feat: optimize XXH3 seeded and batch hashing by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/108
|
|
45
|
+
* Optimize AVX2 Base64 decoding by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/109
|
|
46
|
+
* refactor: optimize Python batch orchestration by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/110
|
|
47
|
+
* feat: optimize XXH3 batches and wrapped Base64 by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/111
|
|
48
|
+
* Fuse Base64 SIMD lookup paths and Murmur3 hex output by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/112
|
|
49
|
+
* fix: preserve custom alphabets and batch typing by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/113
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
**Full Changelog**: https://github.com/kozistr/hashcodecs-rs/compare/v1.3.0...v1.4.0
|
|
53
|
+
|
|
7
54
|
## [1.3.0] - 2026-09-04
|
|
8
55
|
|
|
9
56
|
### What's Changed
|
|
@@ -164,7 +211,9 @@ This file records notable user-facing changes to `hashcodecs`. Version 1.0.0 sta
|
|
|
164
211
|
- Initial Python and Rust APIs for Base64 and MurmurHash3.
|
|
165
212
|
- Runtime SIMD dispatch and platform-specific CPython wheels.
|
|
166
213
|
|
|
167
|
-
[Unreleased]: https://github.com/kozistr/hashcodecs-rs/compare/v1.
|
|
214
|
+
[Unreleased]: https://github.com/kozistr/hashcodecs-rs/compare/v1.4.1...HEAD
|
|
215
|
+
[1.4.1]: https://github.com/kozistr/hashcodecs-rs/compare/v1.4.0...v1.4.1
|
|
216
|
+
[1.4.0]: https://github.com/kozistr/hashcodecs-rs/compare/v1.3.0...v1.4.0
|
|
168
217
|
[1.3.0]: https://github.com/kozistr/hashcodecs-rs/compare/v1.2.1...v1.3.0
|
|
169
218
|
[1.2.1]: https://github.com/kozistr/hashcodecs-rs/compare/v1.2.0...v1.2.1
|
|
170
219
|
[1.2.0]: https://github.com/kozistr/hashcodecs-rs/compare/v1.1.0...v1.2.0
|
|
@@ -5,8 +5,8 @@ authors:
|
|
|
5
5
|
given-names: Hyeongchan
|
|
6
6
|
orcid: https://orcid.org/0000-0002-1729-0580
|
|
7
7
|
title: "hashcodecs: SIMD-accelerated Base64, MurmurHash3, and XXH3 for Python and Rust"
|
|
8
|
-
version: 1.
|
|
9
|
-
date-released: 2026-09-
|
|
8
|
+
version: 1.4.1
|
|
9
|
+
date-released: 2026-09-13
|
|
10
10
|
license: "MIT OR Apache-2.0"
|
|
11
11
|
repository-code: "https://github.com/kozistr/hashcodecs-rs"
|
|
12
12
|
url: "https://github.com/kozistr/hashcodecs-rs"
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[package]
|
|
2
2
|
name = "hashcodecs"
|
|
3
|
-
version = "1.
|
|
3
|
+
version = "1.4.1"
|
|
4
4
|
edition = "2024"
|
|
5
5
|
rust-version = "1.89"
|
|
6
6
|
autobenches = false
|
|
@@ -11,6 +11,7 @@ keywords = ["base64", "simd", "murmur3", "xxhash", "python"]
|
|
|
11
11
|
categories = ["encoding", "algorithms"]
|
|
12
12
|
readme = "README.md"
|
|
13
13
|
include = [
|
|
14
|
+
"/generated/**",
|
|
14
15
|
"/src/**",
|
|
15
16
|
"/benches/**",
|
|
16
17
|
"/tests/sanitizers.rs",
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: hashcodecs
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.4.1
|
|
4
4
|
Summary: SIMD-accelerated Base64, MurmurHash3, and xxHash codecs
|
|
5
5
|
Project-URL: Documentation, https://hashcodecs-rs.readthedocs.io/
|
|
6
6
|
Project-URL: Repository, https://github.com/kozistr/hashcodecs-rs
|
|
@@ -56,7 +56,7 @@ Move byte-heavy work into Rust without changing your Python inputs. `hashcodecs`
|
|
|
56
56
|
|
|
57
57
|
- Base64 encode and decode with standard, URL-safe, padded, unpadded, wrapped, and canonical modes.
|
|
58
58
|
- MurmurHash3 x86-32, x86-128, and x64-128 with one-shot and incremental APIs.
|
|
59
|
-
- Bit-for-bit compatible XXH3-64 and XXH3-128 with allocating and allocation-free native batch APIs.
|
|
59
|
+
- Bit-for-bit compatible XXH3-64 and XXH3-128 with prepared seeds and allocating and allocation-free native batch APIs.
|
|
60
60
|
- Caller-managed `*_into` outputs for allocation-sensitive workloads.
|
|
61
61
|
- Runtime dispatch across AVX-512, AVX2, SSE4.1, SSSE3, NEON, and scalar implementations where applicable.
|
|
62
62
|
- Direct CPython buffer handling for `bytes`, `bytearray`, and `memoryview` inputs.
|
|
@@ -147,7 +147,18 @@ assert_eq!(
|
|
|
147
147
|
0x2d06_8005_38d3_94c2
|
|
148
148
|
);
|
|
149
149
|
|
|
150
|
+
let seeded_input = vec![7; 1024];
|
|
151
|
+
let prepared = hashcodecs::xxhash::PreparedXxh3::new(42);
|
|
152
|
+
assert_eq!(
|
|
153
|
+
prepared.hash_64(&seeded_input),
|
|
154
|
+
hashcodecs::xxhash::xxh3_64(&seeded_input, 42)
|
|
155
|
+
);
|
|
156
|
+
|
|
150
157
|
let inputs: &[&[u8]] = &[b"hello", b"world"];
|
|
158
|
+
assert_eq!(
|
|
159
|
+
prepared.hash_64_batch(inputs),
|
|
160
|
+
hashcodecs::xxhash::xxh3_64_batch(inputs, 42)
|
|
161
|
+
);
|
|
151
162
|
let mut hashes = [0_u64; 2];
|
|
152
163
|
let mut index = 0;
|
|
153
164
|
hashcodecs::xxhash::xxh3_64_batch_for_each(inputs, 0, |hash| {
|
|
@@ -242,10 +253,7 @@ cargo bench --manifest-path benches/Cargo.toml --bench xxhash
|
|
|
242
253
|
|
|
243
254
|
## Performance snapshot
|
|
244
255
|
|
|
245
|
-
|
|
246
|
-
90.07 GiB/s. With 256 B items in batches of 64, the Base64 batch API reaches 11.84 GiB/s for encode and 6.23 GiB/s for
|
|
247
|
-
decode. The run pins one logical CPU and uses 15 samples with a 0.2-second minimum per sample. Read the
|
|
248
|
-
[benchmark details](BENCHMARK.md) and [raw comparison results](docs/benchmarks/results.csv).
|
|
256
|
+
On the benchmark host, `hashcodecs.xxh3_64` processes a 1 MiB input at 91.62 GiB/s. The Base64 batch API reaches 11.56 GiB/s for encode and 8.28 GiB/s for decode with 256 B items in batches of 64. Each run pins one logical CPU and uses 15 samples with a 0.2-second minimum per sample. Read the [benchmark details](BENCHMARK.md) and [raw comparison results](docs/benchmarks/results.csv).
|
|
249
257
|
|
|
250
258
|
## Development
|
|
251
259
|
|
|
@@ -258,14 +266,12 @@ uv build
|
|
|
258
266
|
Run the primary local checks:
|
|
259
267
|
|
|
260
268
|
```sh
|
|
261
|
-
|
|
262
|
-
cargo clippy --all-targets --features python -- -D warnings
|
|
263
|
-
cargo test --features python
|
|
264
|
-
uv run --frozen --no-sync ruff check . --no-cache
|
|
265
|
-
uv run --frozen --no-sync ruff format --check .
|
|
266
|
-
uv run --frozen --no-sync pytest tests --cov=hashcodecs --cov-branch --cov-fail-under=100
|
|
269
|
+
just check
|
|
267
270
|
```
|
|
268
271
|
|
|
272
|
+
`just check` builds and reinstalls the current wheel before running Python tests. Run `just full-check` before
|
|
273
|
+
committing to also check release builds, Rust core coverage, and the extracted source distribution.
|
|
274
|
+
|
|
269
275
|
Optimized paths are also checked with differential fuzzing, Kani, strict-provenance Miri, AddressSanitizer, and
|
|
270
276
|
MemorySanitizer in CI.
|
|
271
277
|
|
|
@@ -25,7 +25,7 @@ Move byte-heavy work into Rust without changing your Python inputs. `hashcodecs`
|
|
|
25
25
|
|
|
26
26
|
- Base64 encode and decode with standard, URL-safe, padded, unpadded, wrapped, and canonical modes.
|
|
27
27
|
- MurmurHash3 x86-32, x86-128, and x64-128 with one-shot and incremental APIs.
|
|
28
|
-
- Bit-for-bit compatible XXH3-64 and XXH3-128 with allocating and allocation-free native batch APIs.
|
|
28
|
+
- Bit-for-bit compatible XXH3-64 and XXH3-128 with prepared seeds and allocating and allocation-free native batch APIs.
|
|
29
29
|
- Caller-managed `*_into` outputs for allocation-sensitive workloads.
|
|
30
30
|
- Runtime dispatch across AVX-512, AVX2, SSE4.1, SSSE3, NEON, and scalar implementations where applicable.
|
|
31
31
|
- Direct CPython buffer handling for `bytes`, `bytearray`, and `memoryview` inputs.
|
|
@@ -116,7 +116,18 @@ assert_eq!(
|
|
|
116
116
|
0x2d06_8005_38d3_94c2
|
|
117
117
|
);
|
|
118
118
|
|
|
119
|
+
let seeded_input = vec![7; 1024];
|
|
120
|
+
let prepared = hashcodecs::xxhash::PreparedXxh3::new(42);
|
|
121
|
+
assert_eq!(
|
|
122
|
+
prepared.hash_64(&seeded_input),
|
|
123
|
+
hashcodecs::xxhash::xxh3_64(&seeded_input, 42)
|
|
124
|
+
);
|
|
125
|
+
|
|
119
126
|
let inputs: &[&[u8]] = &[b"hello", b"world"];
|
|
127
|
+
assert_eq!(
|
|
128
|
+
prepared.hash_64_batch(inputs),
|
|
129
|
+
hashcodecs::xxhash::xxh3_64_batch(inputs, 42)
|
|
130
|
+
);
|
|
120
131
|
let mut hashes = [0_u64; 2];
|
|
121
132
|
let mut index = 0;
|
|
122
133
|
hashcodecs::xxhash::xxh3_64_batch_for_each(inputs, 0, |hash| {
|
|
@@ -211,10 +222,7 @@ cargo bench --manifest-path benches/Cargo.toml --bench xxhash
|
|
|
211
222
|
|
|
212
223
|
## Performance snapshot
|
|
213
224
|
|
|
214
|
-
|
|
215
|
-
90.07 GiB/s. With 256 B items in batches of 64, the Base64 batch API reaches 11.84 GiB/s for encode and 6.23 GiB/s for
|
|
216
|
-
decode. The run pins one logical CPU and uses 15 samples with a 0.2-second minimum per sample. Read the
|
|
217
|
-
[benchmark details](BENCHMARK.md) and [raw comparison results](docs/benchmarks/results.csv).
|
|
225
|
+
On the benchmark host, `hashcodecs.xxh3_64` processes a 1 MiB input at 91.62 GiB/s. The Base64 batch API reaches 11.56 GiB/s for encode and 8.28 GiB/s for decode with 256 B items in batches of 64. Each run pins one logical CPU and uses 15 samples with a 0.2-second minimum per sample. Read the [benchmark details](BENCHMARK.md) and [raw comparison results](docs/benchmarks/results.csv).
|
|
218
226
|
|
|
219
227
|
## Development
|
|
220
228
|
|
|
@@ -227,14 +235,12 @@ uv build
|
|
|
227
235
|
Run the primary local checks:
|
|
228
236
|
|
|
229
237
|
```sh
|
|
230
|
-
|
|
231
|
-
cargo clippy --all-targets --features python -- -D warnings
|
|
232
|
-
cargo test --features python
|
|
233
|
-
uv run --frozen --no-sync ruff check . --no-cache
|
|
234
|
-
uv run --frozen --no-sync ruff format --check .
|
|
235
|
-
uv run --frozen --no-sync pytest tests --cov=hashcodecs --cov-branch --cov-fail-under=100
|
|
238
|
+
just check
|
|
236
239
|
```
|
|
237
240
|
|
|
241
|
+
`just check` builds and reinstalls the current wheel before running Python tests. Run `just full-check` before
|
|
242
|
+
committing to also check release builds, Rust core coverage, and the extracted source distribution.
|
|
243
|
+
|
|
238
244
|
Optimized paths are also checked with differential fuzzing, Kani, strict-provenance Miri, AddressSanitizer, and
|
|
239
245
|
MemorySanitizer in CI.
|
|
240
246
|
|
|
@@ -17,6 +17,23 @@ The checks are deliberately complementary:
|
|
|
17
17
|
exact Base64 buffers, incremental MurmurHash3 states, and XXH3 batch kernels.
|
|
18
18
|
- libFuzzer runs under sanitizers. Base64 is compared with `base64` 0.23.1,
|
|
19
19
|
MurmurHash3 with `murmur3` 0.5, and XXH3 with `xxhash-rust` 0.8.18.
|
|
20
|
+
XXH3 cases use independent equal-length and heterogeneous buffers, with
|
|
21
|
+
distinct lane contents and batch counts through nine to cover SIMD groups and tails.
|
|
22
|
+
|
|
23
|
+
The Python XXH3 batch bindings finish all input reads before allocating Python
|
|
24
|
+
result containers: a GC finalizer can clear the input list or resize a bytearray
|
|
25
|
+
even while the GIL is held. Up to 32 native results fit on the stack; larger
|
|
26
|
+
batches use a fallible vector. Large immutable inputs keep their owners across
|
|
27
|
+
GIL detachment. Subprocess tests on CPython 3.10/3.11 trigger finalizers during
|
|
28
|
+
allocation and reuse freed storage, covering both sides of the stack boundary.
|
|
29
|
+
|
|
30
|
+
Packed XXH3 batches retain immutable input owners before detaching, then borrow
|
|
31
|
+
64 inputs at a time into stack storage. They stage hashes in native memory;
|
|
32
|
+
after reattaching, they recheck the destination length before copying results.
|
|
33
|
+
The little-endian bulk copy relies on the padding-free layout contract of the
|
|
34
|
+
private `PackedDigest` trait. Big-endian hosts serialize each word. Tests clear
|
|
35
|
+
the source list and shrink the output from another Python thread before hashing
|
|
36
|
+
the retained inputs and checking the output size again.
|
|
20
37
|
|
|
21
38
|
Run the same checks locally on Linux:
|
|
22
39
|
|
|
@@ -8,6 +8,7 @@ mod support;
|
|
|
8
8
|
const BASE64_ENCODE_SIZES: [usize; 10] = [15, 16, 31, 32, 47, 48, 51, 52, 103, 104];
|
|
9
9
|
const BASE64_DECODE_SIZES: [usize; 8] = [12, 16, 28, 32, 60, 64, 124, 128];
|
|
10
10
|
const MURMUR_SIZES: [usize; 6] = [15, 16, 31, 32, 255, 256];
|
|
11
|
+
const MURMUR_X64_SIZES: [usize; 12] = [15, 16, 31, 32, 64, 255, 256, 384, 511, 512, 513, 1024];
|
|
11
12
|
|
|
12
13
|
fn data(size: usize) -> Vec<u8> {
|
|
13
14
|
(0..size)
|
|
@@ -41,9 +42,9 @@ fn base64_decode(c: &mut Criterion) {
|
|
|
41
42
|
}
|
|
42
43
|
|
|
43
44
|
macro_rules! murmur_group {
|
|
44
|
-
($criterion:expr, $name:literal, $function:path) => {{
|
|
45
|
+
($criterion:expr, $name:literal, $function:path, $sizes:expr) => {{
|
|
45
46
|
let mut group = $criterion.benchmark_group($name);
|
|
46
|
-
for size in
|
|
47
|
+
for size in $sizes {
|
|
47
48
|
let input = data(size);
|
|
48
49
|
group.throughput(Throughput::Bytes(size as u64));
|
|
49
50
|
group.bench_with_input(BenchmarkId::from_parameter(size), &input, |bench, input| {
|
|
@@ -58,17 +59,20 @@ fn murmur3(c: &mut Criterion) {
|
|
|
58
59
|
murmur_group!(
|
|
59
60
|
c,
|
|
60
61
|
"murmur_x86_32_crossover",
|
|
61
|
-
hashcodecs::murmur3::murmur3_x86_32
|
|
62
|
+
hashcodecs::murmur3::murmur3_x86_32,
|
|
63
|
+
MURMUR_SIZES
|
|
62
64
|
);
|
|
63
65
|
murmur_group!(
|
|
64
66
|
c,
|
|
65
67
|
"murmur_x86_128_crossover",
|
|
66
|
-
hashcodecs::murmur3::murmur3_x86_128
|
|
68
|
+
hashcodecs::murmur3::murmur3_x86_128,
|
|
69
|
+
MURMUR_SIZES
|
|
67
70
|
);
|
|
68
71
|
murmur_group!(
|
|
69
72
|
c,
|
|
70
73
|
"murmur_x64_128_crossover",
|
|
71
|
-
hashcodecs::murmur3::murmur3_x64_128
|
|
74
|
+
hashcodecs::murmur3::murmur3_x64_128,
|
|
75
|
+
MURMUR_X64_SIZES
|
|
72
76
|
);
|
|
73
77
|
}
|
|
74
78
|
|
|
@@ -6,9 +6,10 @@ use criterion::{BenchmarkId, Criterion, Throughput, criterion_group, criterion_m
|
|
|
6
6
|
|
|
7
7
|
mod support;
|
|
8
8
|
|
|
9
|
-
const SIZES: [usize;
|
|
9
|
+
const SIZES: [usize; 16] = [
|
|
10
10
|
16,
|
|
11
11
|
17,
|
|
12
|
+
32,
|
|
12
13
|
64,
|
|
13
14
|
128,
|
|
14
15
|
129,
|
|
@@ -23,8 +24,9 @@ const SIZES: [usize; 15] = [
|
|
|
23
24
|
1024 * 1024,
|
|
24
25
|
8 * 1024 * 1024,
|
|
25
26
|
];
|
|
26
|
-
const MIXED_BATCHES: [(&str, [usize; 4]);
|
|
27
|
+
const MIXED_BATCHES: [(&str, [usize; 4]); 4] = [
|
|
27
28
|
("two_long_runs", [1024, 1024, 4 * 1024, 4 * 1024]),
|
|
29
|
+
("nearby_long_lengths", [257, 258, 259, 260]),
|
|
28
30
|
("short_then_long_boundary", [240, 240, 241, 241]),
|
|
29
31
|
("long_then_short_boundary", [241, 241, 240, 240]),
|
|
30
32
|
];
|
|
@@ -140,7 +142,7 @@ fn benchmark_batch(c: &mut Criterion, group_name: &str, owned: &[Vec<u8>]) {
|
|
|
140
142
|
|
|
141
143
|
fn batch(c: &mut Criterion) {
|
|
142
144
|
for items in [2, 3, 32] {
|
|
143
|
-
for size in [64, 1024, 4 * 1024, 1024 * 1024] {
|
|
145
|
+
for size in [64, 241, 1024, 4 * 1024, 1024 * 1024] {
|
|
144
146
|
let owned = (0..items)
|
|
145
147
|
.map(|index| data(size, index as u8))
|
|
146
148
|
.collect::<Vec<_>>();
|
|
@@ -158,10 +160,79 @@ fn batch(c: &mut Criterion) {
|
|
|
158
160
|
}
|
|
159
161
|
}
|
|
160
162
|
|
|
163
|
+
fn prepared_seed(c: &mut Criterion) {
|
|
164
|
+
for size in [241, 1024] {
|
|
165
|
+
let input = data(size, 17);
|
|
166
|
+
let prepared = hashcodecs::xxhash::PreparedXxh3::new(42);
|
|
167
|
+
assert_eq!(
|
|
168
|
+
prepared.hash_64(&input),
|
|
169
|
+
hashcodecs::xxhash::xxh3_64(&input, 42)
|
|
170
|
+
);
|
|
171
|
+
assert_eq!(
|
|
172
|
+
prepared.hash_128(&input),
|
|
173
|
+
hashcodecs::xxhash::xxh3_128(&input, 42)
|
|
174
|
+
);
|
|
175
|
+
|
|
176
|
+
let mut group = c.benchmark_group(format!("xxh3_prepared/{size}"));
|
|
177
|
+
group.throughput(Throughput::Bytes(size as u64));
|
|
178
|
+
group.bench_function("one_shot_64", |bench| {
|
|
179
|
+
bench.iter(|| hashcodecs::xxhash::xxh3_64(black_box(&input), 42))
|
|
180
|
+
});
|
|
181
|
+
group.bench_function("prepared_64", |bench| {
|
|
182
|
+
bench.iter(|| prepared.hash_64(black_box(&input)))
|
|
183
|
+
});
|
|
184
|
+
group.bench_function("one_shot_128", |bench| {
|
|
185
|
+
bench.iter(|| hashcodecs::xxhash::xxh3_128(black_box(&input), 42))
|
|
186
|
+
});
|
|
187
|
+
group.bench_function("prepared_128", |bench| {
|
|
188
|
+
bench.iter(|| prepared.hash_128(black_box(&input)))
|
|
189
|
+
});
|
|
190
|
+
group.finish();
|
|
191
|
+
}
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
fn prepared_batch(c: &mut Criterion) {
|
|
195
|
+
for items in [2, 4] {
|
|
196
|
+
for size in [241, 1024] {
|
|
197
|
+
let owned = (0..items)
|
|
198
|
+
.map(|index| data(size, index as u8))
|
|
199
|
+
.collect::<Vec<_>>();
|
|
200
|
+
let inputs = owned.iter().map(Vec::as_slice).collect::<Vec<_>>();
|
|
201
|
+
let prepared = hashcodecs::xxhash::PreparedXxh3::new(42);
|
|
202
|
+
assert_eq!(
|
|
203
|
+
prepared.hash_64_batch(&inputs),
|
|
204
|
+
hashcodecs::xxhash::xxh3_64_batch(&inputs, 42)
|
|
205
|
+
);
|
|
206
|
+
assert_eq!(
|
|
207
|
+
prepared.hash_128_batch(&inputs),
|
|
208
|
+
hashcodecs::xxhash::xxh3_128_batch(&inputs, 42)
|
|
209
|
+
);
|
|
210
|
+
|
|
211
|
+
let mut group = c.benchmark_group(format!("xxh3_prepared_batch/{items}_items/{size}"));
|
|
212
|
+
group.throughput(Throughput::Bytes((items * size) as u64));
|
|
213
|
+
group.bench_function("batch_64", |bench| {
|
|
214
|
+
bench.iter(|| hashcodecs::xxhash::xxh3_64_batch(black_box(&inputs), 42))
|
|
215
|
+
});
|
|
216
|
+
group.bench_function("prepared_64", |bench| {
|
|
217
|
+
bench.iter(|| prepared.hash_64_batch(black_box(&inputs)))
|
|
218
|
+
});
|
|
219
|
+
group.bench_function("batch_128", |bench| {
|
|
220
|
+
bench.iter(|| hashcodecs::xxhash::xxh3_128_batch(black_box(&inputs), 42))
|
|
221
|
+
});
|
|
222
|
+
group.bench_function("prepared_128", |bench| {
|
|
223
|
+
bench.iter(|| prepared.hash_128_batch(black_box(&inputs)))
|
|
224
|
+
});
|
|
225
|
+
group.finish();
|
|
226
|
+
}
|
|
227
|
+
}
|
|
228
|
+
}
|
|
229
|
+
|
|
161
230
|
fn xxhash(c: &mut Criterion) {
|
|
162
231
|
support::pin_to_one_cpu();
|
|
163
232
|
one_shot(c);
|
|
164
233
|
batch(c);
|
|
234
|
+
prepared_seed(c);
|
|
235
|
+
prepared_batch(c);
|
|
165
236
|
}
|
|
166
237
|
|
|
167
238
|
criterion_group! {
|