hashcodecs 1.4.0__tar.gz → 1.4.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/BENCHMARK.md +95 -8
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/CHANGELOG.md +20 -1
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/CITATION.cff +2 -2
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/Cargo.lock +1 -1
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/Cargo.toml +1 -1
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/PKG-INFO +2 -6
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/README.md +1 -5
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/SAFETY.md +8 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/benches/crossover.rs +9 -5
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/pyproject.toml +1 -1
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/base64/decode/avx2.rs +3 -8
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/arguments.rs +5 -2
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/base64/api.rs +16 -22
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/base64/batch.rs +32 -28
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/base64/configured.rs +70 -9
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/base64/configured_tests.rs +4 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/base64/decode.rs +450 -64
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/base64/encode.rs +101 -66
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/base64/lenient.rs +88 -15
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/base64/policy.rs +49 -4
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/base64/strict.rs +20 -17
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/buffer.rs +244 -8
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/objects.rs +2 -2
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/schema.rs +41 -10
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/xxhash/batch.rs +127 -21
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/murmur3/dispatch.rs +4 -2
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/murmur3/tests.rs +74 -21
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/xxhash/long_inputs.rs +1 -1
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/xxhash/primitives.rs +2 -2
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/xxhash/short_inputs.rs +16 -16
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/.gitignore +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/LICENSE +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/LICENSE-MIT +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/SECURITY.md +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/benches/Cargo.toml +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/benches/base64.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/benches/murmur3.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/benches/support/mod.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/benches/xxhash.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/build.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/docs/ARCHITECTURE.md +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/docs/api/base64.md +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/docs/api/murmur3.md +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/docs/api/xxh3.md +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/docs/base64.md +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/docs/compatibility.md +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/docs/index.md +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/docs/murmur3.md +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/docs/performance.md +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/docs/requirements.txt +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/docs/xxh3.md +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/generated/rust/binding_schema.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/generated/rust/murmur3_classes.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/hashcodecs/__init__.py +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/hashcodecs/__init__.pyi +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/hashcodecs/_hashcodecs.pyi +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/hashcodecs/base64.py +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/hashcodecs/base64.pyi +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/hashcodecs/murmur3.py +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/hashcodecs/murmur3.pyi +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/hashcodecs/py.typed +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/hashcodecs/xxhash.py +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/hashcodecs/xxhash.pyi +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/hatch_build.py +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/backend.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/base64/alphabet.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/base64/backend.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/base64/decode/aarch64.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/base64/decode/avx512.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/base64/decode/sse41.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/base64/decode/ssse3.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/base64/decode/tables.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/base64/decode/x86_contracts.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/base64/decode.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/base64/encode/aarch64.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/base64/encode/avx2.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/base64/encode/avx512.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/base64/encode/cache.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/base64/encode/ssse3.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/base64/encode.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/base64/error.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/base64/miri_tests.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/base64/output_buffer.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/base64/proofs.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/base64/runtime_dispatch.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/base64/tests/aarch64.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/base64/tests.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/base64.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/base64/scan/aarch64.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/base64/scan/scalar.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/base64/scan/x86.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/base64/scan.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/base64/staging.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/base64.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/compatibility.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/murmur3/callbacks.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/murmur3/digest.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/murmur3/incremental.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/murmur3/methods.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/murmur3.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/runtime.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/xxhash/callbacks.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/xxhash/methods.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings/xxhash.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/bindings.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/lib.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/murmur3/block_buffer.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/murmur3/miri_tests.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/murmur3/primitives.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/murmur3/proofs.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/murmur3/x64_128/x86.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/murmur3/x64_128.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/murmur3/x86_128/x86.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/murmur3/x86_128.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/murmur3/x86_32/x86.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/murmur3/x86_32.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/murmur3.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/xxhash/batch.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/xxhash/long_inputs/aarch64.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/xxhash/long_inputs/scalar.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/xxhash/long_inputs/x86/avx2.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/xxhash/long_inputs/x86/avx2_batch.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/xxhash/long_inputs/x86/avx512.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/xxhash/long_inputs/x86/ssse3.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/xxhash/long_inputs/x86.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/xxhash/miri_tests.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/xxhash/one_shot.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/xxhash/prepared.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/xxhash/proofs.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/xxhash/tests.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/src/xxhash.rs +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/tools/generate_api_metadata.py +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/tools/install_local_wheel.py +0 -0
- {hashcodecs-1.4.0 → hashcodecs-1.4.1}/tools/verify_sdist.py +0 -0
|
@@ -9,9 +9,8 @@ Build the Python wheel with CPython 3.12 and the full C API. Keep competitor val
|
|
|
9
9
|
Use `uv run --python 3.12 --no-project python benchmarks/render_charts.py` to render the charts. Read exact values in
|
|
10
10
|
[docs/benchmarks/results.csv](docs/benchmarks/results.csv).
|
|
11
11
|
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
seconds each, with one logical CPU pinned.
|
|
12
|
+
Python values use CPython 3.12.10 and report the median of 15 samples lasting at least 0.2 seconds each, with one
|
|
13
|
+
logical CPU pinned. Focused runs refresh only the affected series; other values retain their previous measurements.
|
|
15
14
|
|
|
16
15
|
## Timing Controls
|
|
17
16
|
|
|
@@ -31,6 +30,43 @@ GIL-detachment cutoffs. The `--thread-scaling` mode measures aggregate throughpu
|
|
|
31
30
|
it does not pin the process to one logical CPU. Use `--buffer-inputs` to compare 64-byte and 4 KiB XXH3-64 calls
|
|
32
31
|
across bytes, full and sliced memoryviews, writable and non-contiguous views, and `array('B')`.
|
|
33
32
|
|
|
33
|
+
## Rust MurmurHash3 x64 Dispatch
|
|
34
|
+
|
|
35
|
+
Use scalar below 512 bytes of full blocks, then AVX2 when available. The SSE4.1 fallback starts at 512 bytes
|
|
36
|
+
and retains its 8 MiB upper limit. Incremental updates apply these thresholds to each batch of full blocks.
|
|
37
|
+
|
|
38
|
+
Forced-backend measurements with complete finalization on the Core Ultra 7 265K put scalar and AVX2 near parity
|
|
39
|
+
at 384 bytes (39.99 and 40.20 ns/hash), with AVX2 ahead at 512 bytes (53.98 and 52.76 ns/hash). Use 512 bytes as
|
|
40
|
+
a crossover candidate for this host. The shared minimum also avoids the measured SSE4.1 overhead on small inputs;
|
|
41
|
+
it does not establish an SSE4.1 crossover or an optimum across Intel and AMD CPUs.
|
|
42
|
+
|
|
43
|
+
The following public Rust API measurements include runtime dispatch and finalization. On 2026-09-13, we collected
|
|
44
|
+
50 Criterion samples per case with one logical CPU pinned, a 300 ms warmup, and a 1 s measurement target. These
|
|
45
|
+
values report Criterion's mean estimate with seed 42; they are separate from the forced-backend measurements.
|
|
46
|
+
|
|
47
|
+
| Input | ns/hash |
|
|
48
|
+
| --- | ---: |
|
|
49
|
+
| 15 B | 5.32 |
|
|
50
|
+
| 16 B | 5.47 |
|
|
51
|
+
| 31 B | 5.89 |
|
|
52
|
+
| 32 B | 6.03 |
|
|
53
|
+
| 64 B | 7.80 |
|
|
54
|
+
| 255 B | 24.42 |
|
|
55
|
+
| 256 B | 25.04 |
|
|
56
|
+
| 384 B | 37.93 |
|
|
57
|
+
| 511 B | 50.35 |
|
|
58
|
+
| 512 B | 49.79 |
|
|
59
|
+
| 513 B | 50.43 |
|
|
60
|
+
| 1 KiB | 96.89 |
|
|
61
|
+
|
|
62
|
+
We refreshed the x64 hashcodecs series in the Rust MurmurHash3 throughput chart on the same date, retaining the
|
|
63
|
+
x86 and competitor measurements from the previous comparison run.
|
|
64
|
+
|
|
65
|
+
```sh
|
|
66
|
+
cargo bench --manifest-path benches/Cargo.toml --bench crossover -- murmur_x64_128_crossover
|
|
67
|
+
cargo bench --manifest-path benches/Cargo.toml --bench murmur3 -- x64_128/hashcodecs
|
|
68
|
+
```
|
|
69
|
+
|
|
34
70
|
## XXH3
|
|
35
71
|
|
|
36
72
|
For the Rust comparison, link hashcodecs with xxHash 0.8.3 through `xxhash-c-sys`. Build the C baseline with AVX2.
|
|
@@ -51,14 +87,61 @@ bytearrays, and writable memoryviews. Use `--batch-counts 2 9 32 33 --sizes 64`
|
|
|
51
87
|
|
|
52
88
|
The [32-item parent comparison](docs/benchmarks/xxh3-batch-parent-comparison.csv) records CPython 3.12.10 results
|
|
53
89
|
against parent commit `f17ab86`, measured on 2026-09-05. These paired measurements also cover the 256 KiB
|
|
54
|
-
GIL-detachment threshold and 1 MiB items. The Python XXH3
|
|
55
|
-
|
|
90
|
+
GIL-detachment threshold and 1 MiB items. The Python XXH3 batch panels report CPython 3.12.10 measurements
|
|
91
|
+
from 2026-09-13, using 15 samples of at least 0.2 seconds; one-shot and upstream values retain their prior measurements.
|
|
56
92
|
|
|
57
93
|
Across the 30 paired 32-item cases, branch throughput ranges from 2.59% lower to 2.70% higher than the parent.
|
|
58
94
|
The [stack-boundary comparison](docs/benchmarks/xxh3-batch-boundary-comparison.csv) covers counts 2 and 33 with
|
|
59
95
|
64-byte items: two-item bytearray batches lose 5.40–6.05%, and 33-item bytes batches lose 6.06–7.05%. These
|
|
60
96
|
measurements show residual overhead for some small-input batches; they do not establish zero regression.
|
|
61
97
|
|
|
98
|
+
### Packed-batch detachment
|
|
99
|
+
|
|
100
|
+
XXH3 batches release the GIL at 1 MiB of total input or 16,384 items, provided the input buffers permit detachment.
|
|
101
|
+
The item limit accounts for per-item hashing and output work, including empty inputs. One-shot XXH3 keeps its
|
|
102
|
+
256 KiB threshold. Mutable inputs retain their existing synchronization rules.
|
|
103
|
+
|
|
104
|
+
The detached packed path retains input owners, borrows 64 inputs at a time on the stack, and stages results until
|
|
105
|
+
it reacquires the GIL and rechecks the output size. This removes one allocation and 16 bytes of temporary input
|
|
106
|
+
descriptors per item on 64-bit hosts. Little-endian hosts copy staged results in one operation.
|
|
107
|
+
|
|
108
|
+
The [candidate measurements](docs/benchmarks/xxh3-packed-candidates.csv) compare 256 KiB, 512 KiB, and 1 MiB
|
|
109
|
+
byte thresholds with the same allocation reduction and 16,384-item limit. Each exploratory value uses five
|
|
110
|
+
samples of at least 0.03 seconds. The 1 MiB policy keeps both 4,096- and 8,192-item batches of 64-byte inputs on
|
|
111
|
+
the direct-output path. This trades longer GIL holds for lower latency; it does not establish an optimal threshold
|
|
112
|
+
for other processors or contended workloads.
|
|
113
|
+
|
|
114
|
+
The [selected-policy measurements](docs/benchmarks/xxh3-packed-thresholds.csv) use CPython 3.15.0b4 on the
|
|
115
|
+
Intel Core Ultra 7 265K, with one logical CPU pinned, independent input allocations, and 15 samples of at least
|
|
116
|
+
0.2 seconds each, measured on 2026-09-13. They cover both digest widths and item sizes of 64 bytes, 1 KiB, and 64 KiB.
|
|
117
|
+
For 64-byte inputs:
|
|
118
|
+
|
|
119
|
+
| Items | XXH3-64 packed latency | XXH3-128 packed latency |
|
|
120
|
+
| ---: | ---: | ---: |
|
|
121
|
+
| 4,095 | 13.92 µs | 27.29 µs |
|
|
122
|
+
| 4,096 | 14.48 µs | 27.33 µs |
|
|
123
|
+
| 4,097 | 13.95 µs | 27.44 µs |
|
|
124
|
+
| 8,191 | 28.95 µs | 54.72 µs |
|
|
125
|
+
| 8,192 | 29.14 µs | 54.61 µs |
|
|
126
|
+
| 8,193 | 29.46 µs | 54.57 µs |
|
|
127
|
+
| 16,383 | 58.49 µs | 109.12 µs |
|
|
128
|
+
| 16,384 | 92.78 µs | 144.04 µs |
|
|
129
|
+
| 16,385 | 94.52 µs | 145.62 µs |
|
|
130
|
+
|
|
131
|
+
A discontinuity remains at the new detachment boundary because retaining owners and staging output still cost work.
|
|
132
|
+
The longest measured attached call near these boundaries takes about 109 microseconds. This is a latency
|
|
133
|
+
measurement on this host, not a bound on thread waiting time; the tests also verify GIL progress for large inputs
|
|
134
|
+
and for 16,384-item batches of empty, one-byte, and 64-byte inputs.
|
|
135
|
+
|
|
136
|
+
```sh
|
|
137
|
+
uv run --frozen --no-sync python benchmarks/python_xxhash_thresholds.py --output docs/benchmarks/xxh3-packed-thresholds.csv
|
|
138
|
+
uv run --frozen --no-sync python benchmarks/python_xxhash.py --batches-only --hashcodecs-only
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
`--thresholds-kib` selects benchmark input sizes around each candidate; it does not change the extension's
|
|
142
|
+
compiled policy. To compare candidates, rebuild the wheel after changing `BATCH_DETACH_BYTES` in
|
|
143
|
+
`src/bindings/xxhash/batch.rs`.
|
|
144
|
+
|
|
62
145
|
The Rust mixed benchmarks use `[1024, 1024, 4096, 4096]`, `[257, 258, 259, 260]`, `[240, 240, 241, 241]`, and the
|
|
63
146
|
reverse boundary order. The 1024/4096 case measures adjacent two-item long runs. The 257–260 case measures a
|
|
64
147
|
four-item run with one shared stripe count and distinct final stripes. The 240/241 cases measure both orders across
|
|
@@ -89,6 +172,11 @@ Pass one reusable `bytearray` to each `*_into` call.
|
|
|
89
172
|
Run `python benchmarks/python_base64.py --lenient`. The MIME cases insert CRLF after each 76-character line. The
|
|
90
173
|
noisy cases insert `!` at the same boundaries. Both cases measure returned bytes and reusable output buffers.
|
|
91
174
|
|
|
175
|
+
Run `uv run --no-project --python 3.12 python benchmarks/python_base64.py --custom-lenient` to measure clean and noisy
|
|
176
|
+
inputs with `@#` altchars. For a 1 MiB decoded payload, the custom clean case reaches 3.44 GiB/s with returned bytes
|
|
177
|
+
and 13.05 GiB/s with a reusable output buffer. The decoder translates complete symbol runs in a 4 KiB staging buffer
|
|
178
|
+
for SIMD decoding and handles padding and ignored characters through the lenient state machine.
|
|
179
|
+
|
|
92
180
|
[](docs/benchmarks/base64-python-lenient.svg)
|
|
93
181
|
|
|
94
182
|
## Wrapped Python Base64
|
|
@@ -99,9 +187,8 @@ Run `python benchmarks/python_base64.py --wrapped` with CPython 3.15 or newer. T
|
|
|
99
187
|
## Python Memoryview Inputs
|
|
100
188
|
|
|
101
189
|
Use `--memoryview-input` for full immutable views and `--sliced-memoryview-input` for equal-length contiguous views
|
|
102
|
-
with a nonzero starting offset.
|
|
103
|
-
|
|
104
|
-
remains identical.
|
|
190
|
+
with a nonzero starting offset. At detachment sizes, both retain the immutable bytes owner without copying the input.
|
|
191
|
+
Callback-capable decode arguments keep their required snapshots and conversion order. Both modes use identical data.
|
|
105
192
|
|
|
106
193
|
[](docs/benchmarks/base64-python-memoryview.svg)
|
|
107
194
|
|
|
@@ -4,6 +4,24 @@ This file records notable user-facing changes to `hashcodecs`. Version 1.0.0 sta
|
|
|
4
4
|
|
|
5
5
|
## [Unreleased]
|
|
6
6
|
|
|
7
|
+
## [1.4.1] - 2026-09-13
|
|
8
|
+
|
|
9
|
+
### What's Changed
|
|
10
|
+
* fix: match CPython Base64 and XXH3 buffer semantics by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/114
|
|
11
|
+
* fix: preserve Base64 callback buffer semantics by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/115
|
|
12
|
+
* fix: preserve CPython Base64 observable behavior by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/116
|
|
13
|
+
* fix: resolve repository type diagnostics by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/117
|
|
14
|
+
* fix: release Base64 input buffers before writing output by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/118
|
|
15
|
+
* refactor: centralize buffer callback policy and clarify finalizers by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/119
|
|
16
|
+
* fix: accelerate custom-alphabet lenient Base64 decoding by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/120
|
|
17
|
+
* fix: reduce XXH3 packed batch detachment overhead by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/121
|
|
18
|
+
* fix: delay MurmurHash3 x64 SIMD dispatch until 512 bytes by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/122
|
|
19
|
+
* fix: restore Python buffer and batch throughput by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/123
|
|
20
|
+
* fix: typo in the performance viz image by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/124
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
**Full Changelog**: https://github.com/kozistr/hashcodecs-rs/compare/v1.4.0...v1.4.1
|
|
24
|
+
|
|
7
25
|
## [1.4.0] - 2026-09-10
|
|
8
26
|
|
|
9
27
|
### What's Changed
|
|
@@ -193,7 +211,8 @@ This file records notable user-facing changes to `hashcodecs`. Version 1.0.0 sta
|
|
|
193
211
|
- Initial Python and Rust APIs for Base64 and MurmurHash3.
|
|
194
212
|
- Runtime SIMD dispatch and platform-specific CPython wheels.
|
|
195
213
|
|
|
196
|
-
[Unreleased]: https://github.com/kozistr/hashcodecs-rs/compare/v1.4.
|
|
214
|
+
[Unreleased]: https://github.com/kozistr/hashcodecs-rs/compare/v1.4.1...HEAD
|
|
215
|
+
[1.4.1]: https://github.com/kozistr/hashcodecs-rs/compare/v1.4.0...v1.4.1
|
|
197
216
|
[1.4.0]: https://github.com/kozistr/hashcodecs-rs/compare/v1.3.0...v1.4.0
|
|
198
217
|
[1.3.0]: https://github.com/kozistr/hashcodecs-rs/compare/v1.2.1...v1.3.0
|
|
199
218
|
[1.2.1]: https://github.com/kozistr/hashcodecs-rs/compare/v1.2.0...v1.2.1
|
|
@@ -5,8 +5,8 @@ authors:
|
|
|
5
5
|
given-names: Hyeongchan
|
|
6
6
|
orcid: https://orcid.org/0000-0002-1729-0580
|
|
7
7
|
title: "hashcodecs: SIMD-accelerated Base64, MurmurHash3, and XXH3 for Python and Rust"
|
|
8
|
-
version: 1.4.
|
|
9
|
-
date-released: 2026-09-
|
|
8
|
+
version: 1.4.1
|
|
9
|
+
date-released: 2026-09-13
|
|
10
10
|
license: "MIT OR Apache-2.0"
|
|
11
11
|
repository-code: "https://github.com/kozistr/hashcodecs-rs"
|
|
12
12
|
url: "https://github.com/kozistr/hashcodecs-rs"
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: hashcodecs
|
|
3
|
-
Version: 1.4.
|
|
3
|
+
Version: 1.4.1
|
|
4
4
|
Summary: SIMD-accelerated Base64, MurmurHash3, and xxHash codecs
|
|
5
5
|
Project-URL: Documentation, https://hashcodecs-rs.readthedocs.io/
|
|
6
6
|
Project-URL: Repository, https://github.com/kozistr/hashcodecs-rs
|
|
@@ -253,11 +253,7 @@ cargo bench --manifest-path benches/Cargo.toml --bench xxhash
|
|
|
253
253
|
|
|
254
254
|
## Performance snapshot
|
|
255
255
|
|
|
256
|
-
|
|
257
|
-
at 91.27 GiB/s. In the full 2026-09-02 run, the Base64 batch API reaches 11.84 GiB/s for encode and 6.23 GiB/s for
|
|
258
|
-
decode with 256 B items in batches of 64. Each run pins one logical CPU and uses 15 samples with a 0.2-second
|
|
259
|
-
minimum per sample. Read the
|
|
260
|
-
[benchmark details](BENCHMARK.md) and [raw comparison results](docs/benchmarks/results.csv).
|
|
256
|
+
On the benchmark host, `hashcodecs.xxh3_64` processes a 1 MiB input at 91.62 GiB/s. The Base64 batch API reaches 11.56 GiB/s for encode and 8.28 GiB/s for decode with 256 B items in batches of 64. Each run pins one logical CPU and uses 15 samples with a 0.2-second minimum per sample. Read the [benchmark details](BENCHMARK.md) and [raw comparison results](docs/benchmarks/results.csv).
|
|
261
257
|
|
|
262
258
|
## Development
|
|
263
259
|
|
|
@@ -222,11 +222,7 @@ cargo bench --manifest-path benches/Cargo.toml --bench xxhash
|
|
|
222
222
|
|
|
223
223
|
## Performance snapshot
|
|
224
224
|
|
|
225
|
-
|
|
226
|
-
at 91.27 GiB/s. In the full 2026-09-02 run, the Base64 batch API reaches 11.84 GiB/s for encode and 6.23 GiB/s for
|
|
227
|
-
decode with 256 B items in batches of 64. Each run pins one logical CPU and uses 15 samples with a 0.2-second
|
|
228
|
-
minimum per sample. Read the
|
|
229
|
-
[benchmark details](BENCHMARK.md) and [raw comparison results](docs/benchmarks/results.csv).
|
|
225
|
+
On the benchmark host, `hashcodecs.xxh3_64` processes a 1 MiB input at 91.62 GiB/s. The Base64 batch API reaches 11.56 GiB/s for encode and 8.28 GiB/s for decode with 256 B items in batches of 64. Each run pins one logical CPU and uses 15 samples with a 0.2-second minimum per sample. Read the [benchmark details](BENCHMARK.md) and [raw comparison results](docs/benchmarks/results.csv).
|
|
230
226
|
|
|
231
227
|
## Development
|
|
232
228
|
|
|
@@ -27,6 +27,14 @@ batches use a fallible vector. Large immutable inputs keep their owners across
|
|
|
27
27
|
GIL detachment. Subprocess tests on CPython 3.10/3.11 trigger finalizers during
|
|
28
28
|
allocation and reuse freed storage, covering both sides of the stack boundary.
|
|
29
29
|
|
|
30
|
+
Packed XXH3 batches retain immutable input owners before detaching, then borrow
|
|
31
|
+
64 inputs at a time into stack storage. They stage hashes in native memory;
|
|
32
|
+
after reattaching, they recheck the destination length before copying results.
|
|
33
|
+
The little-endian bulk copy relies on the padding-free layout contract of the
|
|
34
|
+
private `PackedDigest` trait. Big-endian hosts serialize each word. Tests clear
|
|
35
|
+
the source list and shrink the output from another Python thread before hashing
|
|
36
|
+
the retained inputs and checking the output size again.
|
|
37
|
+
|
|
30
38
|
Run the same checks locally on Linux:
|
|
31
39
|
|
|
32
40
|
```sh
|
|
@@ -8,6 +8,7 @@ mod support;
|
|
|
8
8
|
const BASE64_ENCODE_SIZES: [usize; 10] = [15, 16, 31, 32, 47, 48, 51, 52, 103, 104];
|
|
9
9
|
const BASE64_DECODE_SIZES: [usize; 8] = [12, 16, 28, 32, 60, 64, 124, 128];
|
|
10
10
|
const MURMUR_SIZES: [usize; 6] = [15, 16, 31, 32, 255, 256];
|
|
11
|
+
const MURMUR_X64_SIZES: [usize; 12] = [15, 16, 31, 32, 64, 255, 256, 384, 511, 512, 513, 1024];
|
|
11
12
|
|
|
12
13
|
fn data(size: usize) -> Vec<u8> {
|
|
13
14
|
(0..size)
|
|
@@ -41,9 +42,9 @@ fn base64_decode(c: &mut Criterion) {
|
|
|
41
42
|
}
|
|
42
43
|
|
|
43
44
|
macro_rules! murmur_group {
|
|
44
|
-
($criterion:expr, $name:literal, $function:path) => {{
|
|
45
|
+
($criterion:expr, $name:literal, $function:path, $sizes:expr) => {{
|
|
45
46
|
let mut group = $criterion.benchmark_group($name);
|
|
46
|
-
for size in
|
|
47
|
+
for size in $sizes {
|
|
47
48
|
let input = data(size);
|
|
48
49
|
group.throughput(Throughput::Bytes(size as u64));
|
|
49
50
|
group.bench_with_input(BenchmarkId::from_parameter(size), &input, |bench, input| {
|
|
@@ -58,17 +59,20 @@ fn murmur3(c: &mut Criterion) {
|
|
|
58
59
|
murmur_group!(
|
|
59
60
|
c,
|
|
60
61
|
"murmur_x86_32_crossover",
|
|
61
|
-
hashcodecs::murmur3::murmur3_x86_32
|
|
62
|
+
hashcodecs::murmur3::murmur3_x86_32,
|
|
63
|
+
MURMUR_SIZES
|
|
62
64
|
);
|
|
63
65
|
murmur_group!(
|
|
64
66
|
c,
|
|
65
67
|
"murmur_x86_128_crossover",
|
|
66
|
-
hashcodecs::murmur3::murmur3_x86_128
|
|
68
|
+
hashcodecs::murmur3::murmur3_x86_128,
|
|
69
|
+
MURMUR_SIZES
|
|
67
70
|
);
|
|
68
71
|
murmur_group!(
|
|
69
72
|
c,
|
|
70
73
|
"murmur_x64_128_crossover",
|
|
71
|
-
hashcodecs::murmur3::murmur3_x64_128
|
|
74
|
+
hashcodecs::murmur3::murmur3_x64_128,
|
|
75
|
+
MURMUR_X64_SIZES
|
|
72
76
|
);
|
|
73
77
|
}
|
|
74
78
|
|
|
@@ -14,14 +14,9 @@ use super::tables::{
|
|
|
14
14
|
};
|
|
15
15
|
use super::x86_contracts::{Decoder, Store};
|
|
16
16
|
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
/// 2. Cache input_ptr / input_len.
|
|
21
|
-
/// 3. Inspect assembly for spills caused by the 4× unroll.
|
|
22
|
-
/// 4. Benchmark 2× versus 4× unrolling.
|
|
23
|
-
/// 5. Leave the combined error reduction alone unless profiling says otherwise.
|
|
24
|
-
/// I'd inspect cargo asm/Compiler Explorer output for spills and benchmark 2× vs 4×.
|
|
17
|
+
// Optimization candidate: compare two-block and four-block unrolling after
|
|
18
|
+
// checking generated assembly for register spills. Preserve the combined error
|
|
19
|
+
// reduction and the Store policy's exact output boundary for the final block.
|
|
25
20
|
#[target_feature(enable = "avx2")]
|
|
26
21
|
pub(crate) unsafe fn decode_avx2<A: Decoder, S: Store>(
|
|
27
22
|
input: &[u8],
|
|
@@ -3,6 +3,7 @@ use std::ptr;
|
|
|
3
3
|
|
|
4
4
|
use pyo3::ffi;
|
|
5
5
|
|
|
6
|
+
#[inline]
|
|
6
7
|
pub(super) unsafe fn parse_raw_arguments<const N: usize>(
|
|
7
8
|
args: *const *mut ffi::PyObject,
|
|
8
9
|
nargs: isize,
|
|
@@ -40,8 +41,10 @@ pub(super) unsafe fn parse_raw_arguments<const N: usize>(
|
|
|
40
41
|
for keyword_index in 0..keyword_count {
|
|
41
42
|
let keyword = unsafe { tuple_item(keywords, keyword_index) };
|
|
42
43
|
let value = unsafe { *args.add(nargs + keyword_index) };
|
|
43
|
-
|
|
44
|
-
|
|
44
|
+
// Valid keywords follow the positional arguments. Search those slots
|
|
45
|
+
// first, then check the filled slots to preserve duplicate diagnostics.
|
|
46
|
+
let parameter_index = (nargs..N).chain(0..nargs).find(|&index| {
|
|
47
|
+
(unsafe { ffi::PyUnicode_CompareWithASCIIString(keyword, parameter_names[index]) }) == 0
|
|
45
48
|
});
|
|
46
49
|
|
|
47
50
|
let Some(parameter_index) = parameter_index else {
|
|
@@ -63,9 +63,7 @@ callback! {
|
|
|
63
63
|
|
|
64
64
|
callback! {
|
|
65
65
|
urlsafe_b64encode, |py; s, padded| {
|
|
66
|
-
let result = padded
|
|
67
|
-
.truthy(py)
|
|
68
|
-
.and_then(|padded| encode::urlsafe_b64encode(py, s.raw(py), padded));
|
|
66
|
+
let result = encode::urlsafe_b64encode(py, s.raw(py), padded);
|
|
69
67
|
return_bound(py, result)
|
|
70
68
|
}
|
|
71
69
|
}
|
|
@@ -74,7 +72,7 @@ callback! {
|
|
|
74
72
|
urlsafe_b64encode_into, |py; s, output, padded| {
|
|
75
73
|
let result = (|| {
|
|
76
74
|
let output = output.raw(py).cast::<PyByteArray>()?;
|
|
77
|
-
encode::urlsafe_b64encode_into(s.raw(py), output, padded
|
|
75
|
+
encode::urlsafe_b64encode_into(py, s.raw(py), output, padded)
|
|
78
76
|
})();
|
|
79
77
|
return_usize(py, result)
|
|
80
78
|
}
|
|
@@ -169,17 +167,15 @@ callback! {
|
|
|
169
167
|
|
|
170
168
|
callback! {
|
|
171
169
|
b64decode, |py; s, altchars, validate, padded, ignorechars, canonical| {
|
|
172
|
-
let result = (
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
)
|
|
182
|
-
})();
|
|
170
|
+
let result = decode::b64decode(
|
|
171
|
+
py,
|
|
172
|
+
s.raw(py),
|
|
173
|
+
altchars.optional(py),
|
|
174
|
+
validate,
|
|
175
|
+
padded,
|
|
176
|
+
ignorechars.provided(py),
|
|
177
|
+
canonical,
|
|
178
|
+
);
|
|
183
179
|
return_bound(py, result)
|
|
184
180
|
}
|
|
185
181
|
}
|
|
@@ -278,10 +274,10 @@ callback! {
|
|
|
278
274
|
s.raw(py),
|
|
279
275
|
output,
|
|
280
276
|
altchars.optional(py),
|
|
281
|
-
validate
|
|
282
|
-
padded
|
|
277
|
+
validate,
|
|
278
|
+
padded,
|
|
283
279
|
ignorechars.provided(py),
|
|
284
|
-
canonical
|
|
280
|
+
canonical,
|
|
285
281
|
)
|
|
286
282
|
})();
|
|
287
283
|
return_usize(py, result)
|
|
@@ -290,9 +286,7 @@ callback! {
|
|
|
290
286
|
|
|
291
287
|
callback! {
|
|
292
288
|
urlsafe_b64decode, |py; s, padded| {
|
|
293
|
-
let result = padded
|
|
294
|
-
.truthy(py)
|
|
295
|
-
.and_then(|padded| decode::urlsafe_b64decode(py, s.raw(py), padded));
|
|
289
|
+
let result = decode::urlsafe_b64decode(py, s.raw(py), padded);
|
|
296
290
|
return_bound(py, result)
|
|
297
291
|
}
|
|
298
292
|
}
|
|
@@ -301,7 +295,7 @@ callback! {
|
|
|
301
295
|
urlsafe_b64decode_into, |py; s, output, padded| {
|
|
302
296
|
let result = (|| {
|
|
303
297
|
let output = output.raw(py).cast::<PyByteArray>()?;
|
|
304
|
-
decode::urlsafe_b64decode_into(py, s.raw(py), output, padded
|
|
298
|
+
decode::urlsafe_b64decode_into(py, s.raw(py), output, padded)
|
|
305
299
|
})();
|
|
306
300
|
return_usize(py, result)
|
|
307
301
|
}
|
|
@@ -10,8 +10,8 @@ use pyo3::prelude::*;
|
|
|
10
10
|
use pyo3::types::{PyAny, PyByteArray, PyBytes, PyInt, PyList, PyMemoryView, PyString};
|
|
11
11
|
|
|
12
12
|
#[cfg(not(Py_GIL_DISABLED))]
|
|
13
|
-
use super::encode::
|
|
14
|
-
use super::encode::{
|
|
13
|
+
use super::encode::encode_small_padded;
|
|
14
|
+
use super::encode::{PreparedEncoder, encode_into, encode_with_prepared};
|
|
15
15
|
use super::policy::{DecodePolicy, PreparedDecoder};
|
|
16
16
|
use crate::bindings::buffer::{
|
|
17
17
|
BufferRange, BytesLike, ascii_or_bytes, ascii_or_bytes_owned, contiguous_bytes_like,
|
|
@@ -232,10 +232,7 @@ fn prepare_batch_inputs<'py>(
|
|
|
232
232
|
stage_snapshots(&mut prepared, |input, policy| {
|
|
233
233
|
// ReleaseBeforeWrite also covers a temporary exact memoryview returned by
|
|
234
234
|
// a string subclass. The items list does not retain that memoryview.
|
|
235
|
-
|
|
236
|
-
|| (policy == SnapshotPolicy::ReleaseBeforeWrite && input.has_borrowed_buffer());
|
|
237
|
-
|
|
238
|
-
input.snapshot_if(needed)
|
|
235
|
+
input.snapshot_before_output_write(policy == SnapshotPolicy::ReleaseBeforeWrite)
|
|
239
236
|
})?;
|
|
240
237
|
|
|
241
238
|
// Capture destination ranges after every reentrant release, then stage all
|
|
@@ -269,39 +266,34 @@ pub(super) fn b64encode_batch_parsed<'py>(
|
|
|
269
266
|
) -> PyResult<Bound<'py, PyList>> {
|
|
270
267
|
#[cfg(not(Py_GIL_DISABLED))]
|
|
271
268
|
let (items, exact_bytes_fast_path) = list_items_and_all(items, |item| {
|
|
272
|
-
|
|
273
|
-
|
|
269
|
+
// Only the length is needed while retaining the list; avoid acquiring
|
|
270
|
+
// the bytes pointer through the C API for every tiny item.
|
|
271
|
+
unsafe {
|
|
272
|
+
ffi::PyBytes_CheckExact(item.as_ptr()) != 0
|
|
273
|
+
&& ffi::Py_SIZE(item.as_ptr()) as usize <= EXACT_BYTES_BATCH_MAX
|
|
274
|
+
}
|
|
274
275
|
})?;
|
|
275
276
|
#[cfg(Py_GIL_DISABLED)]
|
|
276
277
|
let items = list_items(items)?;
|
|
277
278
|
|
|
279
|
+
let encoder = PreparedEncoder::new(altchars, true, None);
|
|
280
|
+
|
|
278
281
|
#[cfg(not(Py_GIL_DISABLED))]
|
|
279
282
|
if exact_bytes_fast_path {
|
|
280
283
|
// Validation retains every input before allocating the output list.
|
|
281
284
|
// Creating a GC-tracked Python object can run finalizers which mutate
|
|
282
285
|
// the original list.
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
let item = unsafe {
|
|
287
|
-
items
|
|
288
|
-
.next()
|
|
289
|
-
.expect("batch item count is exact")
|
|
290
|
-
.cast_into_unchecked::<PyBytes>()
|
|
291
|
-
};
|
|
292
|
-
encode_exact(py, item.as_bytes(), altchars, true, None)
|
|
286
|
+
return list_from_fn(py, items.len(), |index| {
|
|
287
|
+
let input = BytesLike::Bytes(unsafe { items[index].cast_unchecked::<PyBytes>() });
|
|
288
|
+
unsafe { input.with_bytes(|input| encode_small_padded(py, input, &encoder)) }
|
|
293
289
|
});
|
|
294
290
|
}
|
|
295
291
|
let length = items.len();
|
|
296
292
|
let mut items = items.into_iter();
|
|
297
293
|
list_from_fn(py, length, |_| {
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
altchars,
|
|
302
|
-
true,
|
|
303
|
-
None,
|
|
304
|
-
)
|
|
294
|
+
let item = items.next().expect("batch item count is exact");
|
|
295
|
+
let input = crate::bindings::buffer::binascii_contiguous_bytes_like(&item)?;
|
|
296
|
+
encode_with_prepared(py, &input, &encoder)
|
|
305
297
|
})
|
|
306
298
|
}
|
|
307
299
|
|
|
@@ -342,14 +334,14 @@ pub(super) fn b64encode_batch_into_parsed<'py>(
|
|
|
342
334
|
{
|
|
343
335
|
Some(Ok(input)) => Ok(PyInt::new(
|
|
344
336
|
py,
|
|
345
|
-
|
|
337
|
+
encode_into(&input, output, altchars, true, None)?,
|
|
346
338
|
)),
|
|
347
339
|
Some(Err(error)) => Err(error),
|
|
348
340
|
None => {
|
|
349
341
|
let input = contiguous_bytes_like(&items[index], "s")?;
|
|
350
342
|
Ok(PyInt::new(
|
|
351
343
|
py,
|
|
352
|
-
|
|
344
|
+
encode_into(&input, output, altchars, true, None)?,
|
|
353
345
|
))
|
|
354
346
|
}
|
|
355
347
|
}
|
|
@@ -378,9 +370,11 @@ pub(super) fn b64decode_batch_parsed<'py>(
|
|
|
378
370
|
altchars: Option<[u8; 2]>,
|
|
379
371
|
validate: bool,
|
|
380
372
|
) -> PyResult<Bound<'py, PyList>> {
|
|
373
|
+
#[cfg(not(Py_GIL_DISABLED))]
|
|
374
|
+
let (items, exact_bytes_fast_path) = list_items_and_all(items, PyBytes::is_exact_type_of)?;
|
|
375
|
+
#[cfg(Py_GIL_DISABLED)]
|
|
381
376
|
let items = list_items(items)?;
|
|
382
377
|
let length = items.len();
|
|
383
|
-
let mut items = items.into_iter();
|
|
384
378
|
|
|
385
379
|
let decoder = PreparedDecoder::new_for_batch(
|
|
386
380
|
py,
|
|
@@ -388,6 +382,16 @@ pub(super) fn b64decode_batch_parsed<'py>(
|
|
|
388
382
|
length,
|
|
389
383
|
)?;
|
|
390
384
|
|
|
385
|
+
#[cfg(not(Py_GIL_DISABLED))]
|
|
386
|
+
if exact_bytes_fast_path {
|
|
387
|
+
// Retain the validated inputs through output-list allocation and borrow
|
|
388
|
+
// them throughout decoding, releasing the references after the batch.
|
|
389
|
+
return list_from_fn(py, length, |index| {
|
|
390
|
+
let input = BytesLike::Bytes(unsafe { items[index].cast_unchecked::<PyBytes>() });
|
|
391
|
+
decoder.decode_allocating(py, &input)
|
|
392
|
+
});
|
|
393
|
+
}
|
|
394
|
+
let mut items = items.into_iter();
|
|
391
395
|
list_from_fn(py, length, |_| {
|
|
392
396
|
let item = items.next().expect("batch item count is exact");
|
|
393
397
|
let input = ascii_or_bytes(py, &item, "s")?;
|