hashcodecs 1.1.0__tar.gz → 1.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/BENCHMARK.md +50 -1
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/CHANGELOG.md +31 -1
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/CITATION.cff +3 -3
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/Cargo.lock +1 -1
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/Cargo.toml +11 -11
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/PKG-INFO +11 -4
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/README.md +10 -3
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/benches/xxhash.rs +60 -46
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/docs/ARCHITECTURE.md +32 -10
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/docs/api/base64.md +14 -14
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/docs/api/murmur3.md +3 -3
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/docs/api/xxh3.md +2 -2
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/docs/base64.md +3 -1
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/docs/benchmarks/base64-python-batch-large.svg +50 -50
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/docs/benchmarks/base64-python-batch-reusable.svg +39 -39
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/docs/benchmarks/base64-python-batch.svg +109 -109
- hashcodecs-1.2.1/docs/benchmarks/base64-python-lenient.svg +127 -0
- hashcodecs-1.2.1/docs/benchmarks/base64-python-memoryview.svg +159 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/docs/benchmarks/base64-python.svg +18 -18
- hashcodecs-1.2.1/docs/benchmarks/performance-at-a-glance.svg +37 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/docs/benchmarks/results.csv +204 -124
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/docs/benchmarks/xxh3-python.svg +100 -78
- hashcodecs-1.2.1/docs/benchmarks/xxh3-rust-batch-remainders.svg +139 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/docs/benchmarks/xxh3-rust.svg +52 -52
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/docs/performance.md +5 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/hashcodecs/__init__.py +2 -1
- hashcodecs-1.2.1/hashcodecs/__init__.pyi +76 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/hashcodecs/_hashcodecs.pyi +54 -25
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/hashcodecs/base64.py +12 -4
- hashcodecs-1.2.1/hashcodecs/base64.pyi +52 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/hashcodecs/murmur3.py +12 -0
- hashcodecs-1.2.1/hashcodecs/murmur3.pyi +16 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/hashcodecs/xxhash.py +12 -0
- hashcodecs-1.2.1/hashcodecs/xxhash.pyi +16 -0
- hashcodecs-1.2.1/hatch_build.py +86 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/pyproject.toml +5 -6
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/backend.rs +26 -12
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/base64/decode/avx512.rs +2 -2
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/base64/decode.rs +32 -9
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/base64/encode/avx512.rs +2 -2
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/base64/tests/aarch64.rs +3 -3
- hashcodecs-1.2.1/src/base64/tests.rs +1110 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/base64.rs +5 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/bindings/base64/callbacks.rs +50 -48
- hashcodecs-1.2.1/src/bindings/base64/decode/plan.rs +176 -0
- hashcodecs-1.2.1/src/bindings/base64/decode.rs +1686 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/bindings/base64/encode.rs +54 -3
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/bindings/base64/methods.rs +183 -143
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/bindings/base64/mod.rs +51 -13
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/bindings/buffer.rs +105 -7
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/bindings/murmur3/incremental.rs +11 -11
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/bindings/murmur3/methods.rs +10 -9
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/bindings/murmur3/one_shot.rs +14 -10
- hashcodecs-1.2.1/src/bindings/objects.rs +147 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/bindings/runtime.rs +44 -3
- hashcodecs-1.2.1/src/bindings/xxhash/batch.rs +270 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/bindings/xxhash/methods.rs +29 -22
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/bindings/xxhash/mod.rs +21 -15
- hashcodecs-1.2.1/src/murmur3/block_buffer.rs +73 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/murmur3/dispatch.rs +2 -2
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/murmur3/primitives.rs +18 -6
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/murmur3/proofs.rs +4 -3
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/murmur3/tests.rs +60 -14
- hashcodecs-1.2.1/src/murmur3/x64_128/x86.rs +190 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/murmur3/x64_128.rs +34 -18
- hashcodecs-1.2.1/src/murmur3/x86_128/x86.rs +206 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/murmur3/x86_128.rs +30 -12
- hashcodecs-1.2.1/src/murmur3/x86_32/x86.rs +139 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/murmur3/x86_32.rs +33 -17
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/murmur3.rs +1 -3
- hashcodecs-1.2.1/src/xxhash/batch.rs +282 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/xxhash/hash.rs +3 -36
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/xxhash/long/aarch64.rs +11 -9
- hashcodecs-1.2.1/src/xxhash/long/scalar.rs +53 -0
- hashcodecs-1.2.1/src/xxhash/long/x86/avx2.rs +286 -0
- hashcodecs-1.2.1/src/xxhash/long/x86/avx2_batch.rs +89 -0
- {hashcodecs-1.1.0/src/xxhash/long → hashcodecs-1.2.1/src/xxhash/long/x86}/avx512.rs +13 -8
- hashcodecs-1.2.1/src/xxhash/long/x86/mod.rs +4 -0
- {hashcodecs-1.1.0/src/xxhash/long → hashcodecs-1.2.1/src/xxhash/long/x86}/ssse3.rs +14 -10
- hashcodecs-1.2.1/src/xxhash/long.rs +485 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/xxhash/primitives.rs +13 -6
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/xxhash/proofs.rs +10 -8
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/xxhash/tests.rs +67 -38
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/xxhash.rs +2 -0
- hashcodecs-1.2.1/tools/generate_api_metadata.py +235 -0
- hashcodecs-1.1.0/docs/benchmarks/base64-python-memoryview.svg +0 -83
- hashcodecs-1.1.0/hashcodecs/base64.pyi +0 -219
- hashcodecs-1.1.0/hatch_build.py +0 -56
- hashcodecs-1.1.0/src/bindings/base64/decode.rs +0 -1003
- hashcodecs-1.1.0/src/bindings/objects.rs +0 -53
- hashcodecs-1.1.0/src/bindings/xxhash/batch.rs +0 -188
- hashcodecs-1.1.0/src/murmur3/incremental.rs +0 -42
- hashcodecs-1.1.0/src/murmur3/x86.rs +0 -515
- hashcodecs-1.1.0/src/xxhash/batch.rs +0 -132
- hashcodecs-1.1.0/src/xxhash/long/avx2.rs +0 -264
- hashcodecs-1.1.0/src/xxhash/long.rs +0 -203
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/.gitignore +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/LICENSE +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/LICENSE-MIT +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/SAFETY.md +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/SECURITY.md +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/benches/base64.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/benches/murmur3.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/benches/support/mod.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/build.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/docs/benchmarks/base64-python-mutable.svg +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/docs/benchmarks/base64-python-reusable.svg +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/docs/benchmarks/base64-rust.svg +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/docs/benchmarks/murmur3-python-mutable.svg +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/docs/benchmarks/murmur3-python.svg +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/docs/benchmarks/murmur3-rust.svg +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/docs/compatibility.md +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/docs/index.md +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/docs/murmur3.md +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/docs/requirements.txt +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/docs/xxh3.md +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/hashcodecs/py.typed +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/base64/alphabet.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/base64/backend.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/base64/decode/aarch64.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/base64/decode/avx2.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/base64/decode/sse41.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/base64/decode/ssse3.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/base64/decode/x86_contracts.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/base64/dispatch.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/base64/encode/aarch64.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/base64/encode/avx2.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/base64/encode/cache.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/base64/encode/ssse3.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/base64/encode.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/base64/error.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/base64/miri_tests.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/base64/output.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/base64/proofs.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/bindings/arguments.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/bindings/mod.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/bindings/murmur3/digest.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/bindings/murmur3/mod.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/lib.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/murmur3/miri_tests.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/xxhash/miri_tests.rs +0 -0
- {hashcodecs-1.1.0 → hashcodecs-1.2.1}/src/xxhash/short.rs +0 -0
|
@@ -16,14 +16,30 @@ sampling time per case is at least their product, plus calibration; use lower va
|
|
|
16
16
|
hashcodecs-only pass, use `--hashcodecs-only --samples 3 --minimum-sample-seconds 0.05` with each Python benchmark
|
|
17
17
|
script.
|
|
18
18
|
|
|
19
|
+
## Python Call Costs
|
|
20
|
+
|
|
21
|
+
Run `python benchmarks/python_calls.py` to measure positional calls from 0 through 256 bytes in nanoseconds per
|
|
22
|
+
call. Use `--keywords` for positional and keyword calls at 64 bytes, or `--thresholds` for latency around the
|
|
23
|
+
GIL-detachment cutoffs. The `--thread-scaling` mode measures aggregate throughput with one, two, and four threads;
|
|
24
|
+
it does not pin the process to one logical CPU.
|
|
19
25
|
|
|
20
26
|
## XXH3
|
|
21
27
|
|
|
22
28
|
For the Rust comparison, link hashcodecs with xxHash 0.8.3 through `xxhash-c-sys`. Build the C baseline with AVX2.
|
|
23
29
|
For Python, run the upstream `xxhash` extension beside hashcodecs. Pass 32 equal-size inputs to each batch case.
|
|
30
|
+
The Rust remainder cases pass two or three equal-size long inputs. Run Python remainder cases with
|
|
31
|
+
`python benchmarks/python_xxhash.py --batch-counts 2 3`.
|
|
32
|
+
|
|
33
|
+
Use the focused one-shot run to cover the AVX2 four-chain boundaries:
|
|
34
|
+
|
|
35
|
+
```sh
|
|
36
|
+
cargo bench --bench xxhash -- --sample-size 50 "xxh3_(64|128)/(241|512|768|1024|1536|2048|4096)/hashcodecs"
|
|
37
|
+
```
|
|
24
38
|
|
|
25
39
|
[](docs/benchmarks/xxh3-rust.svg)
|
|
26
40
|
|
|
41
|
+
[](docs/benchmarks/xxh3-rust-batch-remainders.svg)
|
|
42
|
+
|
|
27
43
|
[](docs/benchmarks/xxh3-python.svg)
|
|
28
44
|
|
|
29
45
|
## Reusable Python Buffers
|
|
@@ -32,9 +48,18 @@ Pass one reusable `bytearray` to each `*_into` call.
|
|
|
32
48
|
|
|
33
49
|
[](docs/benchmarks/base64-python-reusable.svg)
|
|
34
50
|
|
|
51
|
+
## Lenient Python Base64
|
|
52
|
+
|
|
53
|
+
Run `python benchmarks/python_base64.py --lenient`. The MIME cases insert CRLF after each 76-character line. The
|
|
54
|
+
noisy cases insert `!` at the same boundaries. Both cases measure returned bytes and reusable output buffers.
|
|
55
|
+
|
|
56
|
+
[](docs/benchmarks/base64-python-lenient.svg)
|
|
57
|
+
|
|
35
58
|
## Python Memoryview Inputs
|
|
36
59
|
|
|
37
|
-
Use full immutable
|
|
60
|
+
Use `--memoryview-input` for full immutable views and `--sliced-memoryview-input` for equal-length views with a
|
|
61
|
+
nonzero starting offset. The latter covers the copy/stabilization path used by slices while keeping the encoded data
|
|
62
|
+
identical.
|
|
38
63
|
|
|
39
64
|
[](docs/benchmarks/base64-python-memoryview.svg)
|
|
40
65
|
|
|
@@ -44,6 +69,30 @@ Set the horizontal axis to batch size. Read total input throughput on the vertic
|
|
|
44
69
|
|
|
45
70
|
[](docs/benchmarks/base64-python-batch.svg)
|
|
46
71
|
|
|
72
|
+
For focused runs, override the item and batch sizes directly. `--decode-only` avoids carrying encode allocator state
|
|
73
|
+
into a decode investigation:
|
|
74
|
+
|
|
75
|
+
```sh
|
|
76
|
+
python benchmarks/python_base64_batch.py --item-sizes 4096 --batch-sizes 512 768 1024 1280 2048 --decode-only
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
Use a single operation when recording a sampling profile, or compare traced allocations without a sampler:
|
|
80
|
+
|
|
81
|
+
```sh
|
|
82
|
+
python benchmarks/python_base64_batch.py --item-sizes 4096 --batch-sizes 1024 --profile-operation returned
|
|
83
|
+
python benchmarks/python_base64_batch.py --item-sizes 4096 --batch-sizes 1024 --allocation-profile
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
Add `--profile-direction encode` for returned encoding. By default, the profiling loop assigns the next result
|
|
87
|
+
before releasing the previous one. Add `--discard-profile-result` to release each result before the next call. Use
|
|
88
|
+
`b64encode_batch_into` for that workload.
|
|
89
|
+
|
|
90
|
+
```sh
|
|
91
|
+
python benchmarks/python_base64_batch.py --item-sizes 4096 --batch-sizes 1024 --profile-direction encode --profile-operation returned
|
|
92
|
+
python benchmarks/python_base64_batch.py --item-sizes 4096 --batch-sizes 1024 --profile-direction encode --profile-operation returned --discard-profile-result
|
|
93
|
+
python benchmarks/python_base64_batch.py --item-sizes 4096 --batch-sizes 1024 --profile-direction encode --allocation-profile
|
|
94
|
+
```
|
|
95
|
+
|
|
47
96
|
## Reusable Python Base64 Batch Buffers
|
|
48
97
|
|
|
49
98
|
Pass one reusable `bytearray` to each item in the batch. Use the `*_batch_into` APIs.
|
|
@@ -4,6 +4,34 @@ This file records notable user-facing changes to `hashcodecs`. Version 1.0.0 sta
|
|
|
4
4
|
|
|
5
5
|
## [Unreleased]
|
|
6
6
|
|
|
7
|
+
## [1.2.1] - 2026-08-26
|
|
8
|
+
|
|
9
|
+
### What's Changed
|
|
10
|
+
* fix: harden buffer and lenient decode paths by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/58
|
|
11
|
+
* Refactor Murmur SIMD and centralize XXH3 long dispatch by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/59
|
|
12
|
+
* Repair sdist verification and roll back unpublished 1.2.1 by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/60
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
**Full Changelog**: https://github.com/kozistr/hashcodecs-rs/compare/v1.2.0...v1.2.1
|
|
16
|
+
|
|
17
|
+
## [1.2.0] - 2026-08-25
|
|
18
|
+
|
|
19
|
+
### What's Changed
|
|
20
|
+
* fix: harden free-threaded bindings and API correctness by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/47
|
|
21
|
+
* feat: accelerate Python batch outputs by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/48
|
|
22
|
+
* feat: add native lenient Base64 decoding by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/49
|
|
23
|
+
* update: tune Python detach thresholds by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/50
|
|
24
|
+
* fix: harden Base64 binding edge cases by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/51
|
|
25
|
+
* fix: stabilize Base64 GIL release test by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/52
|
|
26
|
+
* feat: accelerate XXH3 batch remainders by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/53
|
|
27
|
+
* perf: refresh Base64 batch benchmarks by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/54
|
|
28
|
+
* perf: accelerate XXH3 long inputs by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/55
|
|
29
|
+
* perf: profile Base64 batch allocations by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/56
|
|
30
|
+
* refactor: organize Python binding infrastructure by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/57
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
**Full Changelog**: https://github.com/kozistr/hashcodecs-rs/compare/v1.1.0...v1.2.0
|
|
34
|
+
|
|
7
35
|
## [1.1.0] - 2026-08-23
|
|
8
36
|
|
|
9
37
|
### What's Changed
|
|
@@ -99,7 +127,9 @@ This file records notable user-facing changes to `hashcodecs`. Version 1.0.0 sta
|
|
|
99
127
|
- Initial Python and Rust APIs for Base64 and MurmurHash3.
|
|
100
128
|
- Runtime SIMD dispatch and platform-specific CPython wheels.
|
|
101
129
|
|
|
102
|
-
[Unreleased]: https://github.com/kozistr/hashcodecs-rs/compare/v1.1
|
|
130
|
+
[Unreleased]: https://github.com/kozistr/hashcodecs-rs/compare/v1.2.1...HEAD
|
|
131
|
+
[1.2.1]: https://github.com/kozistr/hashcodecs-rs/compare/v1.2.0...v1.2.1
|
|
132
|
+
[1.2.0]: https://github.com/kozistr/hashcodecs-rs/compare/v1.1.0...v1.2.0
|
|
103
133
|
[1.1.0]: https://github.com/kozistr/hashcodecs-rs/compare/v1.0.0...v1.1.0
|
|
104
134
|
[1.0.0]: https://github.com/kozistr/hashcodecs-rs/compare/v0.6.1...v1.0.0
|
|
105
135
|
[0.6.1]: https://github.com/kozistr/hashcodecs-rs/compare/v0.6.0...v0.6.1
|
|
@@ -5,8 +5,8 @@ authors:
|
|
|
5
5
|
given-names: Hyeongchan
|
|
6
6
|
orcid: https://orcid.org/0000-0002-1729-0580
|
|
7
7
|
title: "hashcodecs: SIMD-accelerated Base64, MurmurHash3, and XXH3 for Python and Rust"
|
|
8
|
-
version: 1.1
|
|
9
|
-
date-released: 2026-08-
|
|
8
|
+
version: 1.2.1
|
|
9
|
+
date-released: 2026-08-26
|
|
10
10
|
license: "MIT OR Apache-2.0"
|
|
11
11
|
repository-code: "https://github.com/kozistr/hashcodecs-rs"
|
|
12
|
-
url: "https://github.com/kozistr/hashcodecs-rs"
|
|
12
|
+
url: "https://github.com/kozistr/hashcodecs-rs"
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[package]
|
|
2
2
|
name = "hashcodecs"
|
|
3
|
-
version = "1.1
|
|
3
|
+
version = "1.2.1"
|
|
4
4
|
edition = "2024"
|
|
5
5
|
rust-version = "1.89"
|
|
6
6
|
description = "SIMD-accelerated Base64 codecs and fast MurmurHash3 and xxHash implementations"
|
|
@@ -36,16 +36,16 @@ pyo3 = { version = "0.29.2", optional = true }
|
|
|
36
36
|
pyo3-build-config = { version = "0.29.2", optional = true }
|
|
37
37
|
|
|
38
38
|
[dev-dependencies]
|
|
39
|
-
base64 = "0.23.1"
|
|
40
|
-
base64-turbo = "0.3.0"
|
|
41
|
-
criterion = { version = "0.8.2", default-features = false, features = ["cargo_bench_support"] }
|
|
42
|
-
fastmurmur3 = "0.2.0"
|
|
43
|
-
mimalloc = "0.1.52"
|
|
44
|
-
mm3h = "0.1.3"
|
|
45
|
-
murmur3 = "0.5"
|
|
46
|
-
murmurs = "1.0.5"
|
|
47
|
-
xxhash-c-sys = "0.8.7"
|
|
48
|
-
xxhash-rust = { version = "0.8.18", features = ["xxh3"] }
|
|
39
|
+
base64 = "=0.23.1"
|
|
40
|
+
base64-turbo = "=0.3.0"
|
|
41
|
+
criterion = { version = "=0.8.2", default-features = false, features = ["cargo_bench_support"] }
|
|
42
|
+
fastmurmur3 = "=0.2.0"
|
|
43
|
+
mimalloc = "=0.1.52"
|
|
44
|
+
mm3h = "=0.1.3"
|
|
45
|
+
murmur3 = "=0.5.2"
|
|
46
|
+
murmurs = "=1.0.5"
|
|
47
|
+
xxhash-c-sys = "=0.8.7"
|
|
48
|
+
xxhash-rust = { version = "=0.8.18", features = ["xxh3"] }
|
|
49
49
|
|
|
50
50
|
[[bench]]
|
|
51
51
|
name = "base64"
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: hashcodecs
|
|
3
|
-
Version: 1.1
|
|
3
|
+
Version: 1.2.1
|
|
4
4
|
Summary: SIMD-accelerated Base64, MurmurHash3, and xxHash codecs
|
|
5
5
|
Project-URL: Documentation, https://hashcodecs-rs.readthedocs.io/
|
|
6
6
|
Project-URL: Repository, https://github.com/kozistr/hashcodecs-rs
|
|
@@ -41,6 +41,12 @@ Description-Content-Type: text/markdown
|
|
|
41
41
|

|
|
42
42
|

|
|
43
43
|
|
|
44
|
+
<p align="center">
|
|
45
|
+
<a href="BENCHMARK.md">
|
|
46
|
+
<img src="docs/benchmarks/performance-at-a-glance.svg" alt="CPython 3.12 URL-safe Base64 encoding and decoding benchmark">
|
|
47
|
+
</a>
|
|
48
|
+
</p>
|
|
49
|
+
|
|
44
50
|
SIMD-accelerated Base64, MurmurHash3, and XXH3 for Python and Rust.
|
|
45
51
|
|
|
46
52
|
Move byte-heavy work into Rust without changing your Python inputs. `hashcodecs` accepts `bytes`, `bytearray`, and
|
|
@@ -144,7 +150,7 @@ assert_eq!(
|
|
|
144
150
|
|
|
145
151
|
## Architecture
|
|
146
152
|
|
|
147
|
-
The Rust core owns algorithm behavior and SIMD dispatch. A
|
|
153
|
+
The Rust core owns algorithm behavior and SIMD dispatch. A substantial CPython layer handles argument parsing, buffer
|
|
148
154
|
ownership, reusable outputs, and GIL decisions; root-level Python modules provide typed public exports without
|
|
149
155
|
adding per-call wrappers.
|
|
150
156
|
|
|
@@ -207,12 +213,13 @@ cargo bench --bench xxhash
|
|
|
207
213
|
uv sync --group benchmark --no-install-project
|
|
208
214
|
uv run --no-project --with . python benchmarks/python_base64.py
|
|
209
215
|
uv run --no-project --with . python benchmarks/python_base64_batch.py
|
|
216
|
+
uv run --no-project --with . python benchmarks/python_calls.py
|
|
210
217
|
uv run --no-project --with . python benchmarks/python_murmur3.py
|
|
211
218
|
uv run --no-project --with . python benchmarks/python_xxhash.py
|
|
212
219
|
```
|
|
213
220
|
|
|
214
|
-
The Python benchmarks expose focused modes such as `--into`, `--bytearray-input`, `--memoryview-input`,
|
|
215
|
-
`--incremental`, `--large`, and `--hashcodecs-only`. All scripts also accept `--samples` and
|
|
221
|
+
The Python benchmarks expose focused modes such as `--into`, `--lenient`, `--bytearray-input`, `--memoryview-input`,
|
|
222
|
+
`--sliced-memoryview-input`, `--incremental`, `--large`, and `--hashcodecs-only`. All scripts also accept `--samples` and
|
|
216
223
|
`--minimum-sample-seconds`; use `--help` on a benchmark script for its supported modes and defaults.
|
|
217
224
|
|
|
218
225
|
For the same-ISA Windows XXH3 comparison shown above, rebuild the C baseline with:
|
|
@@ -10,6 +10,12 @@
|
|
|
10
10
|

|
|
11
11
|

|
|
12
12
|
|
|
13
|
+
<p align="center">
|
|
14
|
+
<a href="BENCHMARK.md">
|
|
15
|
+
<img src="docs/benchmarks/performance-at-a-glance.svg" alt="CPython 3.12 URL-safe Base64 encoding and decoding benchmark">
|
|
16
|
+
</a>
|
|
17
|
+
</p>
|
|
18
|
+
|
|
13
19
|
SIMD-accelerated Base64, MurmurHash3, and XXH3 for Python and Rust.
|
|
14
20
|
|
|
15
21
|
Move byte-heavy work into Rust without changing your Python inputs. `hashcodecs` accepts `bytes`, `bytearray`, and
|
|
@@ -113,7 +119,7 @@ assert_eq!(
|
|
|
113
119
|
|
|
114
120
|
## Architecture
|
|
115
121
|
|
|
116
|
-
The Rust core owns algorithm behavior and SIMD dispatch. A
|
|
122
|
+
The Rust core owns algorithm behavior and SIMD dispatch. A substantial CPython layer handles argument parsing, buffer
|
|
117
123
|
ownership, reusable outputs, and GIL decisions; root-level Python modules provide typed public exports without
|
|
118
124
|
adding per-call wrappers.
|
|
119
125
|
|
|
@@ -176,12 +182,13 @@ cargo bench --bench xxhash
|
|
|
176
182
|
uv sync --group benchmark --no-install-project
|
|
177
183
|
uv run --no-project --with . python benchmarks/python_base64.py
|
|
178
184
|
uv run --no-project --with . python benchmarks/python_base64_batch.py
|
|
185
|
+
uv run --no-project --with . python benchmarks/python_calls.py
|
|
179
186
|
uv run --no-project --with . python benchmarks/python_murmur3.py
|
|
180
187
|
uv run --no-project --with . python benchmarks/python_xxhash.py
|
|
181
188
|
```
|
|
182
189
|
|
|
183
|
-
The Python benchmarks expose focused modes such as `--into`, `--bytearray-input`, `--memoryview-input`,
|
|
184
|
-
`--incremental`, `--large`, and `--hashcodecs-only`. All scripts also accept `--samples` and
|
|
190
|
+
The Python benchmarks expose focused modes such as `--into`, `--lenient`, `--bytearray-input`, `--memoryview-input`,
|
|
191
|
+
`--sliced-memoryview-input`, `--incremental`, `--large`, and `--hashcodecs-only`. All scripts also accept `--samples` and
|
|
185
192
|
`--minimum-sample-seconds`; use `--help` on a benchmark script for its supported modes and defaults.
|
|
186
193
|
|
|
187
194
|
For the same-ISA Windows XXH3 comparison shown above, rebuild the C baseline with:
|
|
@@ -6,7 +6,18 @@ use criterion::{BenchmarkId, Criterion, Throughput, criterion_group, criterion_m
|
|
|
6
6
|
|
|
7
7
|
mod support;
|
|
8
8
|
|
|
9
|
-
const SIZES: [usize;
|
|
9
|
+
const SIZES: [usize; 10] = [
|
|
10
|
+
64,
|
|
11
|
+
241,
|
|
12
|
+
512,
|
|
13
|
+
768,
|
|
14
|
+
1024,
|
|
15
|
+
1536,
|
|
16
|
+
2048,
|
|
17
|
+
4 * 1024,
|
|
18
|
+
1024 * 1024,
|
|
19
|
+
8 * 1024 * 1024,
|
|
20
|
+
];
|
|
10
21
|
|
|
11
22
|
fn data(size: usize, salt: u8) -> Vec<u8> {
|
|
12
23
|
(0..size)
|
|
@@ -61,52 +72,55 @@ fn one_shot(c: &mut Criterion) {
|
|
|
61
72
|
}
|
|
62
73
|
|
|
63
74
|
fn batch(c: &mut Criterion) {
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
75
|
+
for items in [2, 3, 32] {
|
|
76
|
+
for size in [64, 1024, 4 * 1024, 1024 * 1024] {
|
|
77
|
+
let owned = (0..items)
|
|
78
|
+
.map(|index| data(size, index as u8))
|
|
79
|
+
.collect::<Vec<_>>();
|
|
80
|
+
let inputs = owned.iter().map(Vec::as_slice).collect::<Vec<_>>();
|
|
81
|
+
let mut group = c.benchmark_group(format!("xxh3_batch/{items}_items/{size}"));
|
|
82
|
+
group.throughput(Throughput::Bytes((size * items) as u64));
|
|
72
83
|
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
84
|
+
group.bench_with_input(
|
|
85
|
+
BenchmarkId::new("hashcodecs_64", items),
|
|
86
|
+
&inputs,
|
|
87
|
+
|bench, inputs| {
|
|
88
|
+
bench.iter(|| hashcodecs::xxhash::xxh3_64_batch(black_box(inputs), 42))
|
|
89
|
+
},
|
|
90
|
+
);
|
|
91
|
+
group.bench_with_input(
|
|
92
|
+
BenchmarkId::new("upstream_c_64", items),
|
|
93
|
+
&inputs,
|
|
94
|
+
|bench, inputs| {
|
|
95
|
+
bench.iter(|| {
|
|
96
|
+
inputs
|
|
97
|
+
.iter()
|
|
98
|
+
.map(|input| c_xxh3_64(black_box(input), 42))
|
|
99
|
+
.collect::<Vec<_>>()
|
|
100
|
+
})
|
|
101
|
+
},
|
|
102
|
+
);
|
|
103
|
+
group.bench_with_input(
|
|
104
|
+
BenchmarkId::new("hashcodecs_128", items),
|
|
105
|
+
&inputs,
|
|
106
|
+
|bench, inputs| {
|
|
107
|
+
bench.iter(|| hashcodecs::xxhash::xxh3_128_batch(black_box(inputs), 42))
|
|
108
|
+
},
|
|
109
|
+
);
|
|
110
|
+
group.bench_with_input(
|
|
111
|
+
BenchmarkId::new("upstream_c_128", items),
|
|
112
|
+
&inputs,
|
|
113
|
+
|bench, inputs| {
|
|
114
|
+
bench.iter(|| {
|
|
115
|
+
inputs
|
|
116
|
+
.iter()
|
|
117
|
+
.map(|input| c_xxh3_128(black_box(input), 42))
|
|
118
|
+
.collect::<Vec<_>>()
|
|
119
|
+
})
|
|
120
|
+
},
|
|
121
|
+
);
|
|
122
|
+
group.finish();
|
|
123
|
+
}
|
|
110
124
|
}
|
|
111
125
|
}
|
|
112
126
|
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
# Architecture
|
|
2
2
|
|
|
3
|
-
`hashcodecs` is a Rust library with a
|
|
3
|
+
`hashcodecs` is a Rust library with a substantial CPython compatibility layer and a small Python facade. The
|
|
4
|
+
design keeps
|
|
4
5
|
algorithm correctness, CPU-specific execution, and language bindings separate so each layer can evolve without
|
|
5
6
|
changing the others.
|
|
6
7
|
|
|
@@ -95,7 +96,14 @@ The API has three output models:
|
|
|
95
96
|
|
|
96
97
|
- allocating functions return a new byte string or vector;
|
|
97
98
|
- `*_into` functions write into caller-managed storage;
|
|
98
|
-
- batch functions
|
|
99
|
+
- allocating batch functions discard partial result lists on failure;
|
|
100
|
+
- Base64 reusable-output batches are intentionally fail-fast and non-transactional;
|
|
101
|
+
- the XXH3 binding validates and stabilizes every packed-batch input before it mutates the destination.
|
|
102
|
+
|
|
103
|
+
The Python Base64 binding sends strict input to the SIMD core without constructing a discarded Python exception.
|
|
104
|
+
For lenient input, it keeps the MIME whitespace path on normalized SIMD input and decodes other ignored bytes into
|
|
105
|
+
the final Python object or reusable buffer. The native state machine follows the padding behavior of each supported
|
|
106
|
+
CPython patch series. It calls `binascii` only when malformed input needs CPython's exact exception.
|
|
99
107
|
|
|
100
108
|
Large aligned x86 encoding may use non-temporal stores after the input exceeds the detected private-cache working
|
|
101
109
|
set. Smaller work stays on ordinary cached stores.
|
|
@@ -113,15 +121,22 @@ secret initialization, scheduling, accumulation, and merging. XXH3-64 and XXH3-1
|
|
|
113
121
|
`long/{aarch64,avx2,avx512,ssse3}.rs` contains the ISA kernels beside the scalar long-input flow and its CPU
|
|
114
122
|
selection. Only long inputs enter those kernels.
|
|
115
123
|
|
|
116
|
-
|
|
117
|
-
|
|
124
|
+
The AVX2 one-shot kernel splits each full 1,024-byte block across four accumulator chains. It also splits tails
|
|
125
|
+
that contain at least four stripes, including the final overlapping stripe. The kernel reduces the chains before
|
|
126
|
+
each block scramble and before the final merge.
|
|
127
|
+
|
|
128
|
+
Native batches reuse the initialized secret and process inputs in groups of up to four. Two to four equal-size
|
|
129
|
+
inputs longer than 240 bytes use an AVX2 batch accumulator when available; single items and mixed sizes use the
|
|
130
|
+
regular paths.
|
|
118
131
|
Python exposes two result models:
|
|
119
132
|
|
|
120
133
|
- `xxh3_*_batch` returns ergonomic `list[int]` results;
|
|
121
134
|
- `xxh3_*_batch_into` writes packed little-endian digests into one reusable `bytearray` and returns bytes written.
|
|
122
135
|
|
|
123
|
-
The
|
|
124
|
-
|
|
136
|
+
The binding validates capacity and stabilizes every input before it mutates the destination. For small stable
|
|
137
|
+
batches, it writes each infallible hash result to the packed output. For detached large batches and arbitrary
|
|
138
|
+
exporters, it retains temporary results until hashing finishes; this fallback lets callers use the output bytearray
|
|
139
|
+
as an input.
|
|
125
140
|
|
|
126
141
|
## CPython Boundary
|
|
127
142
|
|
|
@@ -137,7 +152,8 @@ Buffer handling follows ownership rather than treating every buffer alike:
|
|
|
137
152
|
- exact `bytearray` values are borrowed but never used in detached work;
|
|
138
153
|
- small or sliced memoryviews are copied;
|
|
139
154
|
- full contiguous bytes or bytearray owners can be retained for large memoryviews;
|
|
140
|
-
- immutable operations at or above
|
|
155
|
+
- immutable Base64 and XXH3 operations at or above 256 KiB may detach from the GIL;
|
|
156
|
+
- immutable MurmurHash3 operations at or above 64 KiB may detach from the GIL;
|
|
141
157
|
- mutable storage never crosses a detached region.
|
|
142
158
|
|
|
143
159
|
Allocating outputs are initialized directly in CPython-owned memory. Reusable-output APIs validate capacity before
|
|
@@ -145,13 +161,19 @@ writing and preserve bytes beyond the returned length.
|
|
|
145
161
|
|
|
146
162
|
## Python Package
|
|
147
163
|
|
|
148
|
-
The
|
|
149
|
-
|
|
150
|
-
`py
|
|
164
|
+
The typed `_hashcodecs.pyi` declaration is the canonical Python API description. It drives the public modules,
|
|
165
|
+
package exports, module stubs, native text signatures and docstrings, and API-reference member lists through
|
|
166
|
+
`tools/generate_api_metadata.py`. The generated `base64.py`, `murmur3.py`, and `xxhash.py` modules organize exports
|
|
167
|
+
without adding per-call wrappers. `py.typed` makes the declarations visible to type checkers.
|
|
151
168
|
|
|
152
169
|
Wheel tests execute the installed package, while coverage paths map that installed location back to the root source
|
|
153
170
|
package.
|
|
154
171
|
|
|
172
|
+
The 100% Rust line-coverage gate continues to measure the Rust core without default features. Separate Linux
|
|
173
|
+
coverage jobs build instrumented CPython extensions for Python 3.12 and free-threaded Python 3.15, run the Python
|
|
174
|
+
suite through them, and merge the binding-layer Rust coverage under a `rust-bindings` flag. The feature-gated
|
|
175
|
+
binding layer remains outside the core percentage, the sanitizer jobs, Miri interpretation, and Kani proofs.
|
|
176
|
+
|
|
155
177
|
## Correctness and Safety
|
|
156
178
|
|
|
157
179
|
Optimized kernels must remain interchangeable with the scalar baseline. The repository enforces that through:
|
|
@@ -1,33 +1,33 @@
|
|
|
1
1
|
# Base64 API Reference
|
|
2
2
|
|
|
3
|
-
The functions below are rendered from the
|
|
4
|
-
behavior exposed by `hashcodecs.base64`.
|
|
3
|
+
The functions below are rendered from the authoritative typed API declaration. They describe the exact call signatures
|
|
4
|
+
and behavior exposed by `hashcodecs.base64`.
|
|
5
5
|
|
|
6
6
|
::: hashcodecs._hashcodecs
|
|
7
7
|
options:
|
|
8
8
|
members:
|
|
9
|
-
- b64encode
|
|
10
|
-
- b64encode_batch
|
|
11
|
-
- b64encode_batch_into
|
|
12
|
-
- b64encode_into
|
|
13
9
|
- b64decode
|
|
14
10
|
- b64decode_batch
|
|
15
11
|
- b64decode_batch_into
|
|
16
12
|
- b64decode_into
|
|
17
|
-
-
|
|
18
|
-
-
|
|
13
|
+
- b64encode
|
|
14
|
+
- b64encode_batch
|
|
15
|
+
- b64encode_batch_into
|
|
16
|
+
- b64encode_into
|
|
19
17
|
- standard_b64decode
|
|
18
|
+
- standard_b64decode_batch
|
|
19
|
+
- standard_b64decode_batch_into
|
|
20
20
|
- standard_b64decode_into
|
|
21
|
+
- standard_b64encode
|
|
21
22
|
- standard_b64encode_batch
|
|
22
23
|
- standard_b64encode_batch_into
|
|
23
|
-
-
|
|
24
|
-
- standard_b64decode_batch_into
|
|
25
|
-
- urlsafe_b64encode
|
|
26
|
-
- urlsafe_b64encode_into
|
|
24
|
+
- standard_b64encode_into
|
|
27
25
|
- urlsafe_b64decode
|
|
26
|
+
- urlsafe_b64decode_batch
|
|
27
|
+
- urlsafe_b64decode_batch_into
|
|
28
28
|
- urlsafe_b64decode_into
|
|
29
|
+
- urlsafe_b64encode
|
|
29
30
|
- urlsafe_b64encode_batch
|
|
30
31
|
- urlsafe_b64encode_batch_into
|
|
31
|
-
-
|
|
32
|
-
- urlsafe_b64decode_batch_into
|
|
32
|
+
- urlsafe_b64encode_into
|
|
33
33
|
show_root_heading: false
|
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
# MurmurHash3 API Reference
|
|
2
2
|
|
|
3
|
-
The functions and incremental hashers below are rendered from the
|
|
3
|
+
The functions and incremental hashers below are rendered from the authoritative typed API declaration.
|
|
4
4
|
|
|
5
5
|
::: hashcodecs._hashcodecs
|
|
6
6
|
options:
|
|
7
7
|
members:
|
|
8
8
|
- murmur3_32
|
|
9
|
-
-
|
|
9
|
+
- murmur3_x64_128
|
|
10
10
|
- murmur3_x64_128_digest
|
|
11
11
|
- murmur3_x86_32
|
|
12
12
|
- murmur3_x86_128
|
|
13
|
-
-
|
|
13
|
+
- murmur3_x86_128_digest
|
|
14
14
|
show_root_heading: false
|
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
# XXH3 API Reference
|
|
2
2
|
|
|
3
|
-
The functions below are rendered from the
|
|
3
|
+
The functions below are rendered from the authoritative typed API declaration.
|
|
4
4
|
|
|
5
5
|
::: hashcodecs._hashcodecs
|
|
6
6
|
options:
|
|
7
7
|
members:
|
|
8
8
|
- xxh3_64
|
|
9
|
-
- xxh3_128
|
|
10
9
|
- xxh3_64_batch
|
|
11
10
|
- xxh3_64_batch_into
|
|
11
|
+
- xxh3_128
|
|
12
12
|
- xxh3_128_batch
|
|
13
13
|
- xxh3_128_batch_into
|
|
14
14
|
show_root_heading: false
|
|
@@ -30,10 +30,12 @@ The `standard_b64encode` and `urlsafe_b64encode` helpers select the respective a
|
|
|
30
30
|
|
|
31
31
|
## Decoding and Validation
|
|
32
32
|
|
|
33
|
-
`b64decode(s, altchars=None, validate
|
|
33
|
+
`b64decode(s, altchars=None, validate=..., *, padded=True, ignorechars=..., canonical=False)` decodes an ASCII
|
|
34
34
|
string or bytes-like value.
|
|
35
35
|
|
|
36
36
|
- `validate=True` rejects characters outside the chosen alphabet instead of discarding them.
|
|
37
|
+
- If you supply `ignorechars` without `validate`, hashcodecs uses strict mode. Pass `validate=False` to keep lenient
|
|
38
|
+
mode; hashcodecs then limits discarded bytes to `ignorechars`.
|
|
37
39
|
- `padded=False` accepts an unpadded final Base64 quantum.
|
|
38
40
|
- `ignorechars` specifies which non-alphabet bytes lenient decoding may ignore. Its default preserves the standard
|
|
39
41
|
library's lenient behavior.
|