hashcodecs 1.3.0__tar.gz → 1.4.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (189) hide show
  1. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/.gitignore +3 -0
  2. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/BENCHMARK.md +125 -6
  3. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/CHANGELOG.md +50 -1
  4. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/CITATION.cff +2 -2
  5. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/Cargo.lock +1 -1
  6. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/Cargo.toml +2 -1
  7. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/PKG-INFO +18 -12
  8. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/README.md +17 -11
  9. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/SAFETY.md +17 -0
  10. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/benches/crossover.rs +9 -5
  11. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/benches/xxhash.rs +74 -3
  12. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/docs/ARCHITECTURE.md +30 -9
  13. hashcodecs-1.4.1/docs/requirements.txt +2 -0
  14. hashcodecs-1.3.0/src/bindings/base64/schema_generated.rs → hashcodecs-1.4.1/generated/rust/binding_schema.rs +563 -51
  15. hashcodecs-1.4.1/generated/rust/murmur3_classes.rs +34 -0
  16. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/hashcodecs/_hashcodecs.pyi +20 -17
  17. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/pyproject.toml +21 -11
  18. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/alphabet.rs +14 -8
  19. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/backend.rs +1 -0
  20. hashcodecs-1.4.1/src/base64/decode/aarch64.rs +354 -0
  21. hashcodecs-1.4.1/src/base64/decode/avx2.rs +333 -0
  22. hashcodecs-1.4.1/src/base64/decode/avx512.rs +355 -0
  23. hashcodecs-1.4.1/src/base64/decode/sse41.rs +129 -0
  24. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/decode/ssse3.rs +99 -9
  25. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/decode/tables.rs +26 -1
  26. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/decode/x86_contracts.rs +37 -4
  27. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/decode.rs +198 -21
  28. hashcodecs-1.4.1/src/base64/encode/aarch64.rs +274 -0
  29. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/encode/avx2.rs +260 -101
  30. hashcodecs-1.4.1/src/base64/encode/avx512.rs +309 -0
  31. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/encode/cache.rs +6 -0
  32. hashcodecs-1.4.1/src/base64/encode/ssse3.rs +165 -0
  33. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/encode.rs +290 -5
  34. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/miri_tests.rs +2 -0
  35. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/output_buffer.rs +2 -0
  36. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/proofs.rs +2 -0
  37. hashcodecs-1.4.1/src/base64/runtime_dispatch.rs +918 -0
  38. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/tests/aarch64.rs +45 -7
  39. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/tests.rs +630 -18
  40. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64.rs +11 -5
  41. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/bindings/arguments.rs +11 -75
  42. hashcodecs-1.3.0/src/bindings/base64/callbacks.rs → hashcodecs-1.4.1/src/bindings/base64/api.rs +69 -54
  43. hashcodecs-1.4.1/src/bindings/base64/batch.rs +454 -0
  44. hashcodecs-1.4.1/src/bindings/base64/configured.rs +861 -0
  45. hashcodecs-1.4.1/src/bindings/base64/configured_tests.rs +915 -0
  46. hashcodecs-1.4.1/src/bindings/base64/decode.rs +1034 -0
  47. hashcodecs-1.4.1/src/bindings/base64/encode.rs +714 -0
  48. hashcodecs-1.4.1/src/bindings/base64/lenient.rs +622 -0
  49. hashcodecs-1.4.1/src/bindings/base64/policy.rs +508 -0
  50. {hashcodecs-1.3.0/src/bindings/base64/decode/native/lenient/helpers → hashcodecs-1.4.1/src/bindings/base64/scan}/aarch64.rs +44 -1
  51. hashcodecs-1.4.1/src/bindings/base64/scan/scalar.rs +60 -0
  52. {hashcodecs-1.3.0/src/bindings/base64/decode/native/lenient/helpers → hashcodecs-1.4.1/src/bindings/base64/scan}/x86.rs +86 -27
  53. hashcodecs-1.4.1/src/bindings/base64/scan.rs +182 -0
  54. hashcodecs-1.4.1/src/bindings/base64/staging.rs +380 -0
  55. hashcodecs-1.4.1/src/bindings/base64/strict.rs +277 -0
  56. hashcodecs-1.4.1/src/bindings/base64.rs +14 -0
  57. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/bindings/buffer.rs +288 -16
  58. hashcodecs-1.4.1/src/bindings/compatibility.rs +113 -0
  59. hashcodecs-1.4.1/src/bindings/murmur3/callbacks.rs +85 -0
  60. hashcodecs-1.4.1/src/bindings/murmur3/digest.rs +37 -0
  61. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/bindings/murmur3/incremental.rs +6 -37
  62. hashcodecs-1.4.1/src/bindings/murmur3/methods.rs +20 -0
  63. hashcodecs-1.3.0/src/bindings/murmur3/mod.rs → hashcodecs-1.4.1/src/bindings/murmur3.rs +1 -1
  64. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/bindings/objects.rs +41 -54
  65. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/bindings/runtime.rs +3 -1
  66. {hashcodecs-1.3.0/src/bindings/base64 → hashcodecs-1.4.1/src/bindings}/schema.rs +61 -19
  67. hashcodecs-1.4.1/src/bindings/xxhash/batch.rs +578 -0
  68. hashcodecs-1.4.1/src/bindings/xxhash/callbacks.rs +122 -0
  69. {hashcodecs-1.3.0/src/bindings/base64 → hashcodecs-1.4.1/src/bindings/xxhash}/methods.rs +2 -2
  70. hashcodecs-1.4.1/src/bindings/xxhash.rs +5 -0
  71. hashcodecs-1.3.0/src/bindings/mod.rs → hashcodecs-1.4.1/src/bindings.rs +2 -0
  72. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3/dispatch.rs +4 -2
  73. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3/tests.rs +74 -21
  74. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3/x64_128.rs +1 -7
  75. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3/x86_128/x86.rs +22 -12
  76. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3/x86_128.rs +1 -7
  77. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3/x86_32.rs +1 -7
  78. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/batch.rs +161 -51
  79. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/long_inputs/aarch64.rs +3 -0
  80. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/long_inputs/scalar.rs +9 -0
  81. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/long_inputs/x86/avx2.rs +22 -5
  82. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/long_inputs/x86/avx2_batch.rs +10 -4
  83. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/long_inputs/x86/avx512.rs +7 -0
  84. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/long_inputs/x86/ssse3.rs +8 -0
  85. hashcodecs-1.3.0/src/xxhash/long_inputs/x86/mod.rs → hashcodecs-1.4.1/src/xxhash/long_inputs/x86.rs +0 -2
  86. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/long_inputs.rs +72 -108
  87. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/one_shot.rs +5 -2
  88. hashcodecs-1.4.1/src/xxhash/prepared.rs +136 -0
  89. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/primitives.rs +37 -9
  90. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/proofs.rs +8 -2
  91. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/short_inputs.rs +110 -19
  92. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/tests.rs +127 -16
  93. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash.rs +2 -0
  94. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/tools/generate_api_metadata.py +76 -153
  95. hashcodecs-1.4.1/tools/install_local_wheel.py +30 -0
  96. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/tools/verify_sdist.py +1 -0
  97. hashcodecs-1.3.0/docs/benchmarks/base64-python-batch-large.svg +0 -85
  98. hashcodecs-1.3.0/docs/benchmarks/base64-python-batch-memoryview.svg +0 -238
  99. hashcodecs-1.3.0/docs/benchmarks/base64-python-batch-reusable.svg +0 -106
  100. hashcodecs-1.3.0/docs/benchmarks/base64-python-batch.svg +0 -313
  101. hashcodecs-1.3.0/docs/benchmarks/base64-python-lenient.svg +0 -127
  102. hashcodecs-1.3.0/docs/benchmarks/base64-python-memoryview.svg +0 -159
  103. hashcodecs-1.3.0/docs/benchmarks/base64-python-mutable.svg +0 -83
  104. hashcodecs-1.3.0/docs/benchmarks/base64-python-reusable.svg +0 -115
  105. hashcodecs-1.3.0/docs/benchmarks/base64-python.svg +0 -203
  106. hashcodecs-1.3.0/docs/benchmarks/base64-rust.svg +0 -203
  107. hashcodecs-1.3.0/docs/benchmarks/murmur3-python-mutable.svg +0 -169
  108. hashcodecs-1.3.0/docs/benchmarks/murmur3-python.svg +0 -235
  109. hashcodecs-1.3.0/docs/benchmarks/murmur3-rust.svg +0 -193
  110. hashcodecs-1.3.0/docs/benchmarks/performance-at-a-glance.svg +0 -37
  111. hashcodecs-1.3.0/docs/benchmarks/results.csv +0 -587
  112. hashcodecs-1.3.0/docs/benchmarks/xxh3-python.svg +0 -181
  113. hashcodecs-1.3.0/docs/benchmarks/xxh3-rust-batch-remainders.svg +0 -139
  114. hashcodecs-1.3.0/docs/benchmarks/xxh3-rust.svg +0 -169
  115. hashcodecs-1.3.0/docs/requirements.txt +0 -2
  116. hashcodecs-1.3.0/src/base64/decode/aarch64.rs +0 -230
  117. hashcodecs-1.3.0/src/base64/decode/avx2.rs +0 -180
  118. hashcodecs-1.3.0/src/base64/decode/avx512.rs +0 -125
  119. hashcodecs-1.3.0/src/base64/decode/sse41.rs +0 -59
  120. hashcodecs-1.3.0/src/base64/encode/aarch64.rs +0 -122
  121. hashcodecs-1.3.0/src/base64/encode/avx512.rs +0 -134
  122. hashcodecs-1.3.0/src/base64/encode/ssse3.rs +0 -86
  123. hashcodecs-1.3.0/src/base64/runtime_dispatch.rs +0 -369
  124. hashcodecs-1.3.0/src/bindings/base64/batch.rs +0 -236
  125. hashcodecs-1.3.0/src/bindings/base64/decode/batch.rs +0 -125
  126. hashcodecs-1.3.0/src/bindings/base64/decode/fallback.rs +0 -146
  127. hashcodecs-1.3.0/src/bindings/base64/decode/native/advanced/config.rs +0 -136
  128. hashcodecs-1.3.0/src/bindings/base64/decode/native/advanced/scanner.rs +0 -354
  129. hashcodecs-1.3.0/src/bindings/base64/decode/native/advanced/specials.rs +0 -73
  130. hashcodecs-1.3.0/src/bindings/base64/decode/native/advanced/staging.rs +0 -152
  131. hashcodecs-1.3.0/src/bindings/base64/decode/native/advanced.rs +0 -183
  132. hashcodecs-1.3.0/src/bindings/base64/decode/native/advanced_tests.rs +0 -574
  133. hashcodecs-1.3.0/src/bindings/base64/decode/native/lenient/compat.rs +0 -18
  134. hashcodecs-1.3.0/src/bindings/base64/decode/native/lenient/helpers/mod.rs +0 -130
  135. hashcodecs-1.3.0/src/bindings/base64/decode/native/lenient/helpers/scalar.rs +0 -41
  136. hashcodecs-1.3.0/src/bindings/base64/decode/native/lenient/mod.rs +0 -131
  137. hashcodecs-1.3.0/src/bindings/base64/decode/native/lenient/state_machine.rs +0 -241
  138. hashcodecs-1.3.0/src/bindings/base64/decode/native/strict.rs +0 -444
  139. hashcodecs-1.3.0/src/bindings/base64/decode/native.rs +0 -23
  140. hashcodecs-1.3.0/src/bindings/base64/decode/output.rs +0 -424
  141. hashcodecs-1.3.0/src/bindings/base64/decode/plan.rs +0 -371
  142. hashcodecs-1.3.0/src/bindings/base64/decode.rs +0 -88
  143. hashcodecs-1.3.0/src/bindings/base64/encode/batch.rs +0 -140
  144. hashcodecs-1.3.0/src/bindings/base64/encode.rs +0 -260
  145. hashcodecs-1.3.0/src/bindings/base64/mod.rs +0 -293
  146. hashcodecs-1.3.0/src/bindings/murmur3/digest.rs +0 -25
  147. hashcodecs-1.3.0/src/bindings/murmur3/methods.rs +0 -108
  148. hashcodecs-1.3.0/src/bindings/murmur3/one_shot.rs +0 -86
  149. hashcodecs-1.3.0/src/bindings/xxhash/batch.rs +0 -374
  150. hashcodecs-1.3.0/src/bindings/xxhash/methods.rs +0 -205
  151. hashcodecs-1.3.0/src/bindings/xxhash/mod.rs +0 -195
  152. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/LICENSE +0 -0
  153. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/LICENSE-MIT +0 -0
  154. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/SECURITY.md +0 -0
  155. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/benches/Cargo.toml +0 -0
  156. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/benches/base64.rs +0 -0
  157. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/benches/murmur3.rs +0 -0
  158. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/benches/support/mod.rs +0 -0
  159. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/build.rs +0 -0
  160. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/docs/api/base64.md +0 -0
  161. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/docs/api/murmur3.md +0 -0
  162. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/docs/api/xxh3.md +0 -0
  163. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/docs/base64.md +0 -0
  164. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/docs/compatibility.md +0 -0
  165. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/docs/index.md +0 -0
  166. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/docs/murmur3.md +0 -0
  167. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/docs/performance.md +0 -0
  168. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/docs/xxh3.md +0 -0
  169. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/hashcodecs/__init__.py +0 -0
  170. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/hashcodecs/__init__.pyi +0 -0
  171. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/hashcodecs/base64.py +0 -0
  172. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/hashcodecs/base64.pyi +0 -0
  173. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/hashcodecs/murmur3.py +0 -0
  174. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/hashcodecs/murmur3.pyi +0 -0
  175. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/hashcodecs/py.typed +0 -0
  176. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/hashcodecs/xxhash.py +0 -0
  177. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/hashcodecs/xxhash.pyi +0 -0
  178. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/hatch_build.py +0 -0
  179. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/backend.rs +0 -0
  180. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/base64/error.rs +0 -0
  181. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/lib.rs +0 -0
  182. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3/block_buffer.rs +0 -0
  183. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3/miri_tests.rs +0 -0
  184. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3/primitives.rs +0 -0
  185. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3/proofs.rs +0 -0
  186. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3/x64_128/x86.rs +0 -0
  187. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3/x86_32/x86.rs +0 -0
  188. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/murmur3.rs +0 -0
  189. {hashcodecs-1.3.0 → hashcodecs-1.4.1}/src/xxhash/miri_tests.rs +0 -0
@@ -10,6 +10,9 @@ site/
10
10
  .venv*/
11
11
  .coverage
12
12
  coverage.xml
13
+ *.profraw
14
+ *.lcov
15
+ *.log
13
16
 
14
17
  # These are backup files generated by rustfmt
15
18
  **/*.rs.bk
@@ -9,6 +9,9 @@ Build the Python wheel with CPython 3.12 and the full C API. Keep competitor val
9
9
  Use `uv run --python 3.12 --no-project python benchmarks/render_charts.py` to render the charts. Read exact values in
10
10
  [docs/benchmarks/results.csv](docs/benchmarks/results.csv).
11
11
 
12
+ Python values use CPython 3.12.10 and report the median of 15 samples lasting at least 0.2 seconds each, with one
13
+ logical CPU pinned. Focused runs refresh only the affected series; other values retain their previous measurements.
14
+
12
15
  ## Timing Controls
13
16
 
14
17
  Every Python benchmark accepts `--samples` (default: 15) and `--minimum-sample-seconds` (default: 0.2). Their
@@ -27,6 +30,43 @@ GIL-detachment cutoffs. The `--thread-scaling` mode measures aggregate throughpu
27
30
  it does not pin the process to one logical CPU. Use `--buffer-inputs` to compare 64-byte and 4 KiB XXH3-64 calls
28
31
  across bytes, full and sliced memoryviews, writable and non-contiguous views, and `array('B')`.
29
32
 
33
+ ## Rust MurmurHash3 x64 Dispatch
34
+
35
+ Use scalar below 512 bytes of full blocks, then AVX2 when available. The SSE4.1 fallback starts at 512 bytes
36
+ and retains its 8 MiB upper limit. Incremental updates apply these thresholds to each batch of full blocks.
37
+
38
+ Forced-backend measurements with complete finalization on the Core Ultra 7 265K put scalar and AVX2 near parity
39
+ at 384 bytes (39.99 and 40.20 ns/hash), with AVX2 ahead at 512 bytes (53.98 and 52.76 ns/hash). Use 512 bytes as
40
+ a crossover candidate for this host. The shared minimum also avoids the measured SSE4.1 overhead on small inputs;
41
+ it does not establish an SSE4.1 crossover or an optimum across Intel and AMD CPUs.
42
+
43
+ The following public Rust API measurements include runtime dispatch and finalization. On 2026-09-13, we collected
44
+ 50 Criterion samples per case with one logical CPU pinned, a 300 ms warmup, and a 1 s measurement target. These
45
+ values report Criterion's mean estimate with seed 42; they are separate from the forced-backend measurements.
46
+
47
+ | Input | ns/hash |
48
+ | --- | ---: |
49
+ | 15 B | 5.32 |
50
+ | 16 B | 5.47 |
51
+ | 31 B | 5.89 |
52
+ | 32 B | 6.03 |
53
+ | 64 B | 7.80 |
54
+ | 255 B | 24.42 |
55
+ | 256 B | 25.04 |
56
+ | 384 B | 37.93 |
57
+ | 511 B | 50.35 |
58
+ | 512 B | 49.79 |
59
+ | 513 B | 50.43 |
60
+ | 1 KiB | 96.89 |
61
+
62
+ We refreshed the x64 hashcodecs series in the Rust MurmurHash3 throughput chart on the same date, retaining the
63
+ x86 and competitor measurements from the previous comparison run.
64
+
65
+ ```sh
66
+ cargo bench --manifest-path benches/Cargo.toml --bench crossover -- murmur_x64_128_crossover
67
+ cargo bench --manifest-path benches/Cargo.toml --bench murmur3 -- x64_128/hashcodecs
68
+ ```
69
+
30
70
  ## XXH3
31
71
 
32
72
  For the Rust comparison, link hashcodecs with xxHash 0.8.3 through `xxhash-c-sys`. Build the C baseline with AVX2.
@@ -34,15 +74,85 @@ For Python, run the upstream `xxhash` extension beside hashcodecs. Pass 32 equal
34
74
  The Rust remainder cases pass two or three equal-size long inputs. Run Python remainder cases with
35
75
  `python benchmarks/python_xxhash.py --batch-counts 2 3`.
36
76
 
37
- The Rust mixed benchmarks use `[1024, 1024, 4096, 4096]`, `[240, 240, 241, 241]`, and the reverse boundary order.
38
- The 1024/4096 case measures adjacent two-item long runs. The 240/241 cases measure both orders across the
39
- short/long dispatch boundary.
77
+ To measure a branch against its parent, build both extensions with the same Python and Rust toolchains, then run:
78
+
79
+ ```sh
80
+ uv run --frozen --no-sync python benchmarks/compare_xxh3_batches.py path/to/parent/_hashcodecs.pyd path/to/branch/_hashcodecs.pyd
81
+ ```
82
+
83
+ Use the corresponding `.so` paths on Linux or macOS. The comparison loads both builds into one interpreter, pins
84
+ one CPU, and alternates their timing order across 15 samples. It checks matching digests and covers bytes,
85
+ bytearrays, and writable memoryviews. Use `--batch-counts 2 9 32 33 --sizes 64` to inspect small batches and the
86
+ 32-result stack boundary. Positive `change_percent` values mean higher branch throughput.
87
+
88
+ The [32-item parent comparison](docs/benchmarks/xxh3-batch-parent-comparison.csv) records CPython 3.12.10 results
89
+ against parent commit `f17ab86`, measured on 2026-09-05. These paired measurements also cover the 256 KiB
90
+ GIL-detachment threshold and 1 MiB items. The Python XXH3 batch panels report CPython 3.12.10 measurements
91
+ from 2026-09-13, using 15 samples of at least 0.2 seconds; one-shot and upstream values retain their prior measurements.
92
+
93
+ Across the 30 paired 32-item cases, branch throughput ranges from 2.59% lower to 2.70% higher than the parent.
94
+ The [stack-boundary comparison](docs/benchmarks/xxh3-batch-boundary-comparison.csv) covers counts 2 and 33 with
95
+ 64-byte items: two-item bytearray batches lose 5.40–6.05%, and 33-item bytes batches lose 6.06–7.05%. These
96
+ measurements show residual overhead for some small-input batches; they do not establish zero regression.
97
+
98
+ ### Packed-batch detachment
99
+
100
+ XXH3 batches release the GIL at 1 MiB of total input or 16,384 items, provided the input buffers permit detachment.
101
+ The item limit accounts for per-item hashing and output work, including empty inputs. One-shot XXH3 keeps its
102
+ 256 KiB threshold. Mutable inputs retain their existing synchronization rules.
103
+
104
+ The detached packed path retains input owners, borrows 64 inputs at a time on the stack, and stages results until
105
+ it reacquires the GIL and rechecks the output size. This removes one allocation and 16 bytes of temporary input
106
+ descriptors per item on 64-bit hosts. Little-endian hosts copy staged results in one operation.
107
+
108
+ The [candidate measurements](docs/benchmarks/xxh3-packed-candidates.csv) compare 256 KiB, 512 KiB, and 1 MiB
109
+ byte thresholds with the same allocation reduction and 16,384-item limit. Each exploratory value uses five
110
+ samples of at least 0.03 seconds. The 1 MiB policy keeps both 4,096- and 8,192-item batches of 64-byte inputs on
111
+ the direct-output path. This trades longer GIL holds for lower latency; it does not establish an optimal threshold
112
+ for other processors or contended workloads.
113
+
114
+ The [selected-policy measurements](docs/benchmarks/xxh3-packed-thresholds.csv) use CPython 3.15.0b4 on the
115
+ Intel Core Ultra 7 265K, with one logical CPU pinned, independent input allocations, and 15 samples of at least
116
+ 0.2 seconds each, measured on 2026-09-13. They cover both digest widths and item sizes of 64 bytes, 1 KiB, and 64 KiB.
117
+ For 64-byte inputs:
118
+
119
+ | Items | XXH3-64 packed latency | XXH3-128 packed latency |
120
+ | ---: | ---: | ---: |
121
+ | 4,095 | 13.92 µs | 27.29 µs |
122
+ | 4,096 | 14.48 µs | 27.33 µs |
123
+ | 4,097 | 13.95 µs | 27.44 µs |
124
+ | 8,191 | 28.95 µs | 54.72 µs |
125
+ | 8,192 | 29.14 µs | 54.61 µs |
126
+ | 8,193 | 29.46 µs | 54.57 µs |
127
+ | 16,383 | 58.49 µs | 109.12 µs |
128
+ | 16,384 | 92.78 µs | 144.04 µs |
129
+ | 16,385 | 94.52 µs | 145.62 µs |
130
+
131
+ A discontinuity remains at the new detachment boundary because retaining owners and staging output still cost work.
132
+ The longest measured attached call near these boundaries takes about 109 microseconds. This is a latency
133
+ measurement on this host, not a bound on thread waiting time; the tests also verify GIL progress for large inputs
134
+ and for 16,384-item batches of empty, one-byte, and 64-byte inputs.
135
+
136
+ ```sh
137
+ uv run --frozen --no-sync python benchmarks/python_xxhash_thresholds.py --output docs/benchmarks/xxh3-packed-thresholds.csv
138
+ uv run --frozen --no-sync python benchmarks/python_xxhash.py --batches-only --hashcodecs-only
139
+ ```
140
+
141
+ `--thresholds-kib` selects benchmark input sizes around each candidate; it does not change the extension's
142
+ compiled policy. To compare candidates, rebuild the wheel after changing `BATCH_DETACH_BYTES` in
143
+ `src/bindings/xxhash/batch.rs`.
144
+
145
+ The Rust mixed benchmarks use `[1024, 1024, 4096, 4096]`, `[257, 258, 259, 260]`, `[240, 240, 241, 241]`, and the
146
+ reverse boundary order. The 1024/4096 case measures adjacent two-item long runs. The 257–260 case measures a
147
+ four-item run with one shared stripe count and distinct final stripes. The 240/241 cases measure both orders across
148
+ the short/long dispatch boundary.
40
149
 
41
150
  Use the focused one-shot run to cover the AVX2 four-chain boundaries:
42
151
 
43
152
  ```sh
44
153
  cargo bench --manifest-path benches/Cargo.toml --bench xxhash -- "xxh3_(64|128)/(240|241|512|768|1024|1536|2048|4096)/hashcodecs"
45
154
  cargo bench --manifest-path benches/Cargo.toml --bench xxhash -- "xxh3_batch/mixed/.*/hashcodecs_(64|128)"
155
+ cargo bench --manifest-path benches/Cargo.toml --bench xxhash -- "xxh3_prepared"
46
156
  ```
47
157
 
48
158
  [![Rust XXH3 throughput](docs/benchmarks/xxh3-rust.svg)](docs/benchmarks/xxh3-rust.svg)
@@ -62,14 +172,23 @@ Pass one reusable `bytearray` to each `*_into` call.
62
172
  Run `python benchmarks/python_base64.py --lenient`. The MIME cases insert CRLF after each 76-character line. The
63
173
  noisy cases insert `!` at the same boundaries. Both cases measure returned bytes and reusable output buffers.
64
174
 
175
+ Run `uv run --no-project --python 3.12 python benchmarks/python_base64.py --custom-lenient` to measure clean and noisy
176
+ inputs with `@#` altchars. For a 1 MiB decoded payload, the custom clean case reaches 3.44 GiB/s with returned bytes
177
+ and 13.05 GiB/s with a reusable output buffer. The decoder translates complete symbol runs in a 4 KiB staging buffer
178
+ for SIMD decoding and handles padding and ignored characters through the lenient state machine.
179
+
65
180
  [![Lenient Python Base64 throughput](docs/benchmarks/base64-python-lenient.svg)](docs/benchmarks/base64-python-lenient.svg)
66
181
 
182
+ ## Wrapped Python Base64
183
+
184
+ Run `python benchmarks/python_base64.py --wrapped` with CPython 3.15 or newer. The benchmark inserts newlines after
185
+ 76 output characters and measures returned bytes and a reusable `bytearray`.
186
+
67
187
  ## Python Memoryview Inputs
68
188
 
69
189
  Use `--memoryview-input` for full immutable views and `--sliced-memoryview-input` for equal-length contiguous views
70
- with a nonzero starting offset. Full views can recover their exact immutable owner at detachment sizes; slices cover
71
- offset-buffer handling, which borrows under the GIL and stabilizes the input in free-threaded builds. The encoded data
72
- remains identical.
190
+ with a nonzero starting offset. At detachment sizes, both retain the immutable bytes owner without copying the input.
191
+ Callback-capable decode arguments keep their required snapshots and conversion order. Both modes use identical data.
73
192
 
74
193
  [![Python Base64 memoryview inputs](docs/benchmarks/base64-python-memoryview.svg)](docs/benchmarks/base64-python-memoryview.svg)
75
194
 
@@ -4,6 +4,53 @@ This file records notable user-facing changes to `hashcodecs`. Version 1.0.0 sta
4
4
 
5
5
  ## [Unreleased]
6
6
 
7
+ ## [1.4.1] - 2026-09-13
8
+
9
+ ### What's Changed
10
+ * fix: match CPython Base64 and XXH3 buffer semantics by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/114
11
+ * fix: preserve Base64 callback buffer semantics by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/115
12
+ * fix: preserve CPython Base64 observable behavior by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/116
13
+ * fix: resolve repository type diagnostics by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/117
14
+ * fix: release Base64 input buffers before writing output by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/118
15
+ * refactor: centralize buffer callback policy and clarify finalizers by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/119
16
+ * fix: accelerate custom-alphabet lenient Base64 decoding by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/120
17
+ * fix: reduce XXH3 packed batch detachment overhead by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/121
18
+ * fix: delay MurmurHash3 x64 SIMD dispatch until 512 bytes by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/122
19
+ * fix: restore Python buffer and batch throughput by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/123
20
+ * fix: typo in the performance viz image by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/124
21
+
22
+
23
+ **Full Changelog**: https://github.com/kozistr/hashcodecs-rs/compare/v1.4.0...v1.4.1
24
+
25
+ ## [1.4.0] - 2026-09-10
26
+
27
+ ### What's Changed
28
+ * refactor: flatten internal module layout by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/91
29
+ * refactor: prepare Base64 decoder policies once by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/92
30
+ * refactor: consolidate binding and dispatch cleanup by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/93
31
+ * refactor: simplify project module layout by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/94
32
+ * refactor: prepare codec policies and unify bindings by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/95
33
+ * refactor: accelerate Base64 codec hot paths by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/96
34
+ * fix: protect XXH3 batches from GC reentrancy by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/97
35
+ * refactor: unify Base64 decode execution by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/98
36
+ * refactor: flatten Base64 binding ownership by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/99
37
+ * refactor: improve XXH3 loads and internal naming by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/100
38
+ * refactor: reduce Base64 decoder setup and retry costs by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/101
39
+ * fix: preserve Base64 decode boundaries and restore ARM coverage by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/102
40
+ * refactor: keep XXH3 AVX2 tail chains in registers by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/103
41
+ * refactor: store AVX2 Base64 vectors in smaller groups by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/104
42
+ * fix: reduce Python overhead and align canonical decoding by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/106
43
+ * refactor: reduce cached AVX2 encoder setup by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/107
44
+ * feat: optimize XXH3 seeded and batch hashing by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/108
45
+ * Optimize AVX2 Base64 decoding by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/109
46
+ * refactor: optimize Python batch orchestration by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/110
47
+ * feat: optimize XXH3 batches and wrapped Base64 by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/111
48
+ * Fuse Base64 SIMD lookup paths and Murmur3 hex output by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/112
49
+ * fix: preserve custom alphabets and batch typing by @kozistr in https://github.com/kozistr/hashcodecs-rs/pull/113
50
+
51
+
52
+ **Full Changelog**: https://github.com/kozistr/hashcodecs-rs/compare/v1.3.0...v1.4.0
53
+
7
54
  ## [1.3.0] - 2026-09-04
8
55
 
9
56
  ### What's Changed
@@ -164,7 +211,9 @@ This file records notable user-facing changes to `hashcodecs`. Version 1.0.0 sta
164
211
  - Initial Python and Rust APIs for Base64 and MurmurHash3.
165
212
  - Runtime SIMD dispatch and platform-specific CPython wheels.
166
213
 
167
- [Unreleased]: https://github.com/kozistr/hashcodecs-rs/compare/v1.3.0...HEAD
214
+ [Unreleased]: https://github.com/kozistr/hashcodecs-rs/compare/v1.4.1...HEAD
215
+ [1.4.1]: https://github.com/kozistr/hashcodecs-rs/compare/v1.4.0...v1.4.1
216
+ [1.4.0]: https://github.com/kozistr/hashcodecs-rs/compare/v1.3.0...v1.4.0
168
217
  [1.3.0]: https://github.com/kozistr/hashcodecs-rs/compare/v1.2.1...v1.3.0
169
218
  [1.2.1]: https://github.com/kozistr/hashcodecs-rs/compare/v1.2.0...v1.2.1
170
219
  [1.2.0]: https://github.com/kozistr/hashcodecs-rs/compare/v1.1.0...v1.2.0
@@ -5,8 +5,8 @@ authors:
5
5
  given-names: Hyeongchan
6
6
  orcid: https://orcid.org/0000-0002-1729-0580
7
7
  title: "hashcodecs: SIMD-accelerated Base64, MurmurHash3, and XXH3 for Python and Rust"
8
- version: 1.3.0
9
- date-released: 2026-09-04
8
+ version: 1.4.1
9
+ date-released: 2026-09-13
10
10
  license: "MIT OR Apache-2.0"
11
11
  repository-code: "https://github.com/kozistr/hashcodecs-rs"
12
12
  url: "https://github.com/kozistr/hashcodecs-rs"
@@ -194,7 +194,7 @@ dependencies = [
194
194
 
195
195
  [[package]]
196
196
  name = "hashcodecs"
197
- version = "1.3.0"
197
+ version = "1.4.1"
198
198
  dependencies = [
199
199
  "base64",
200
200
  "memchr",
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "hashcodecs"
3
- version = "1.3.0"
3
+ version = "1.4.1"
4
4
  edition = "2024"
5
5
  rust-version = "1.89"
6
6
  autobenches = false
@@ -11,6 +11,7 @@ keywords = ["base64", "simd", "murmur3", "xxhash", "python"]
11
11
  categories = ["encoding", "algorithms"]
12
12
  readme = "README.md"
13
13
  include = [
14
+ "/generated/**",
14
15
  "/src/**",
15
16
  "/benches/**",
16
17
  "/tests/sanitizers.rs",
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: hashcodecs
3
- Version: 1.3.0
3
+ Version: 1.4.1
4
4
  Summary: SIMD-accelerated Base64, MurmurHash3, and xxHash codecs
5
5
  Project-URL: Documentation, https://hashcodecs-rs.readthedocs.io/
6
6
  Project-URL: Repository, https://github.com/kozistr/hashcodecs-rs
@@ -56,7 +56,7 @@ Move byte-heavy work into Rust without changing your Python inputs. `hashcodecs`
56
56
 
57
57
  - Base64 encode and decode with standard, URL-safe, padded, unpadded, wrapped, and canonical modes.
58
58
  - MurmurHash3 x86-32, x86-128, and x64-128 with one-shot and incremental APIs.
59
- - Bit-for-bit compatible XXH3-64 and XXH3-128 with allocating and allocation-free native batch APIs.
59
+ - Bit-for-bit compatible XXH3-64 and XXH3-128 with prepared seeds and allocating and allocation-free native batch APIs.
60
60
  - Caller-managed `*_into` outputs for allocation-sensitive workloads.
61
61
  - Runtime dispatch across AVX-512, AVX2, SSE4.1, SSSE3, NEON, and scalar implementations where applicable.
62
62
  - Direct CPython buffer handling for `bytes`, `bytearray`, and `memoryview` inputs.
@@ -147,7 +147,18 @@ assert_eq!(
147
147
  0x2d06_8005_38d3_94c2
148
148
  );
149
149
 
150
+ let seeded_input = vec![7; 1024];
151
+ let prepared = hashcodecs::xxhash::PreparedXxh3::new(42);
152
+ assert_eq!(
153
+ prepared.hash_64(&seeded_input),
154
+ hashcodecs::xxhash::xxh3_64(&seeded_input, 42)
155
+ );
156
+
150
157
  let inputs: &[&[u8]] = &[b"hello", b"world"];
158
+ assert_eq!(
159
+ prepared.hash_64_batch(inputs),
160
+ hashcodecs::xxhash::xxh3_64_batch(inputs, 42)
161
+ );
151
162
  let mut hashes = [0_u64; 2];
152
163
  let mut index = 0;
153
164
  hashcodecs::xxhash::xxh3_64_batch_for_each(inputs, 0, |hash| {
@@ -242,10 +253,7 @@ cargo bench --manifest-path benches/Cargo.toml --bench xxhash
242
253
 
243
254
  ## Performance snapshot
244
255
 
245
- In the full 2026-09-02 hashcodecs-only run on the benchmark host, `hashcodecs.xxh3_64` processes a 1 MiB input at
246
- 90.07 GiB/s. With 256 B items in batches of 64, the Base64 batch API reaches 11.84 GiB/s for encode and 6.23 GiB/s for
247
- decode. The run pins one logical CPU and uses 15 samples with a 0.2-second minimum per sample. Read the
248
- [benchmark details](BENCHMARK.md) and [raw comparison results](docs/benchmarks/results.csv).
256
+ On the benchmark host, `hashcodecs.xxh3_64` processes a 1 MiB input at 91.62 GiB/s. The Base64 batch API reaches 11.56 GiB/s for encode and 8.28 GiB/s for decode with 256 B items in batches of 64. Each run pins one logical CPU and uses 15 samples with a 0.2-second minimum per sample. Read the [benchmark details](BENCHMARK.md) and [raw comparison results](docs/benchmarks/results.csv).
249
257
 
250
258
  ## Development
251
259
 
@@ -258,14 +266,12 @@ uv build
258
266
  Run the primary local checks:
259
267
 
260
268
  ```sh
261
- cargo fmt --check
262
- cargo clippy --all-targets --features python -- -D warnings
263
- cargo test --features python
264
- uv run --frozen --no-sync ruff check . --no-cache
265
- uv run --frozen --no-sync ruff format --check .
266
- uv run --frozen --no-sync pytest tests --cov=hashcodecs --cov-branch --cov-fail-under=100
269
+ just check
267
270
  ```
268
271
 
272
+ `just check` builds and reinstalls the current wheel before running Python tests. Run `just full-check` before
273
+ committing to also check release builds, Rust core coverage, and the extracted source distribution.
274
+
269
275
  Optimized paths are also checked with differential fuzzing, Kani, strict-provenance Miri, AddressSanitizer, and
270
276
  MemorySanitizer in CI.
271
277
 
@@ -25,7 +25,7 @@ Move byte-heavy work into Rust without changing your Python inputs. `hashcodecs`
25
25
 
26
26
  - Base64 encode and decode with standard, URL-safe, padded, unpadded, wrapped, and canonical modes.
27
27
  - MurmurHash3 x86-32, x86-128, and x64-128 with one-shot and incremental APIs.
28
- - Bit-for-bit compatible XXH3-64 and XXH3-128 with allocating and allocation-free native batch APIs.
28
+ - Bit-for-bit compatible XXH3-64 and XXH3-128 with prepared seeds and allocating and allocation-free native batch APIs.
29
29
  - Caller-managed `*_into` outputs for allocation-sensitive workloads.
30
30
  - Runtime dispatch across AVX-512, AVX2, SSE4.1, SSSE3, NEON, and scalar implementations where applicable.
31
31
  - Direct CPython buffer handling for `bytes`, `bytearray`, and `memoryview` inputs.
@@ -116,7 +116,18 @@ assert_eq!(
116
116
  0x2d06_8005_38d3_94c2
117
117
  );
118
118
 
119
+ let seeded_input = vec![7; 1024];
120
+ let prepared = hashcodecs::xxhash::PreparedXxh3::new(42);
121
+ assert_eq!(
122
+ prepared.hash_64(&seeded_input),
123
+ hashcodecs::xxhash::xxh3_64(&seeded_input, 42)
124
+ );
125
+
119
126
  let inputs: &[&[u8]] = &[b"hello", b"world"];
127
+ assert_eq!(
128
+ prepared.hash_64_batch(inputs),
129
+ hashcodecs::xxhash::xxh3_64_batch(inputs, 42)
130
+ );
120
131
  let mut hashes = [0_u64; 2];
121
132
  let mut index = 0;
122
133
  hashcodecs::xxhash::xxh3_64_batch_for_each(inputs, 0, |hash| {
@@ -211,10 +222,7 @@ cargo bench --manifest-path benches/Cargo.toml --bench xxhash
211
222
 
212
223
  ## Performance snapshot
213
224
 
214
- In the full 2026-09-02 hashcodecs-only run on the benchmark host, `hashcodecs.xxh3_64` processes a 1 MiB input at
215
- 90.07 GiB/s. With 256 B items in batches of 64, the Base64 batch API reaches 11.84 GiB/s for encode and 6.23 GiB/s for
216
- decode. The run pins one logical CPU and uses 15 samples with a 0.2-second minimum per sample. Read the
217
- [benchmark details](BENCHMARK.md) and [raw comparison results](docs/benchmarks/results.csv).
225
+ On the benchmark host, `hashcodecs.xxh3_64` processes a 1 MiB input at 91.62 GiB/s. The Base64 batch API reaches 11.56 GiB/s for encode and 8.28 GiB/s for decode with 256 B items in batches of 64. Each run pins one logical CPU and uses 15 samples with a 0.2-second minimum per sample. Read the [benchmark details](BENCHMARK.md) and [raw comparison results](docs/benchmarks/results.csv).
218
226
 
219
227
  ## Development
220
228
 
@@ -227,14 +235,12 @@ uv build
227
235
  Run the primary local checks:
228
236
 
229
237
  ```sh
230
- cargo fmt --check
231
- cargo clippy --all-targets --features python -- -D warnings
232
- cargo test --features python
233
- uv run --frozen --no-sync ruff check . --no-cache
234
- uv run --frozen --no-sync ruff format --check .
235
- uv run --frozen --no-sync pytest tests --cov=hashcodecs --cov-branch --cov-fail-under=100
238
+ just check
236
239
  ```
237
240
 
241
+ `just check` builds and reinstalls the current wheel before running Python tests. Run `just full-check` before
242
+ committing to also check release builds, Rust core coverage, and the extracted source distribution.
243
+
238
244
  Optimized paths are also checked with differential fuzzing, Kani, strict-provenance Miri, AddressSanitizer, and
239
245
  MemorySanitizer in CI.
240
246
 
@@ -17,6 +17,23 @@ The checks are deliberately complementary:
17
17
  exact Base64 buffers, incremental MurmurHash3 states, and XXH3 batch kernels.
18
18
  - libFuzzer runs under sanitizers. Base64 is compared with `base64` 0.23.1,
19
19
  MurmurHash3 with `murmur3` 0.5, and XXH3 with `xxhash-rust` 0.8.18.
20
+ XXH3 cases use independent equal-length and heterogeneous buffers, with
21
+ distinct lane contents and batch counts through nine to cover SIMD groups and tails.
22
+
23
+ The Python XXH3 batch bindings finish all input reads before allocating Python
24
+ result containers: a GC finalizer can clear the input list or resize a bytearray
25
+ even while the GIL is held. Up to 32 native results fit on the stack; larger
26
+ batches use a fallible vector. Large immutable inputs keep their owners across
27
+ GIL detachment. Subprocess tests on CPython 3.10/3.11 trigger finalizers during
28
+ allocation and reuse freed storage, covering both sides of the stack boundary.
29
+
30
+ Packed XXH3 batches retain immutable input owners before detaching, then borrow
31
+ 64 inputs at a time into stack storage. They stage hashes in native memory;
32
+ after reattaching, they recheck the destination length before copying results.
33
+ The little-endian bulk copy relies on the padding-free layout contract of the
34
+ private `PackedDigest` trait. Big-endian hosts serialize each word. Tests clear
35
+ the source list and shrink the output from another Python thread before hashing
36
+ the retained inputs and checking the output size again.
20
37
 
21
38
  Run the same checks locally on Linux:
22
39
 
@@ -8,6 +8,7 @@ mod support;
8
8
  const BASE64_ENCODE_SIZES: [usize; 10] = [15, 16, 31, 32, 47, 48, 51, 52, 103, 104];
9
9
  const BASE64_DECODE_SIZES: [usize; 8] = [12, 16, 28, 32, 60, 64, 124, 128];
10
10
  const MURMUR_SIZES: [usize; 6] = [15, 16, 31, 32, 255, 256];
11
+ const MURMUR_X64_SIZES: [usize; 12] = [15, 16, 31, 32, 64, 255, 256, 384, 511, 512, 513, 1024];
11
12
 
12
13
  fn data(size: usize) -> Vec<u8> {
13
14
  (0..size)
@@ -41,9 +42,9 @@ fn base64_decode(c: &mut Criterion) {
41
42
  }
42
43
 
43
44
  macro_rules! murmur_group {
44
- ($criterion:expr, $name:literal, $function:path) => {{
45
+ ($criterion:expr, $name:literal, $function:path, $sizes:expr) => {{
45
46
  let mut group = $criterion.benchmark_group($name);
46
- for size in MURMUR_SIZES {
47
+ for size in $sizes {
47
48
  let input = data(size);
48
49
  group.throughput(Throughput::Bytes(size as u64));
49
50
  group.bench_with_input(BenchmarkId::from_parameter(size), &input, |bench, input| {
@@ -58,17 +59,20 @@ fn murmur3(c: &mut Criterion) {
58
59
  murmur_group!(
59
60
  c,
60
61
  "murmur_x86_32_crossover",
61
- hashcodecs::murmur3::murmur3_x86_32
62
+ hashcodecs::murmur3::murmur3_x86_32,
63
+ MURMUR_SIZES
62
64
  );
63
65
  murmur_group!(
64
66
  c,
65
67
  "murmur_x86_128_crossover",
66
- hashcodecs::murmur3::murmur3_x86_128
68
+ hashcodecs::murmur3::murmur3_x86_128,
69
+ MURMUR_SIZES
67
70
  );
68
71
  murmur_group!(
69
72
  c,
70
73
  "murmur_x64_128_crossover",
71
- hashcodecs::murmur3::murmur3_x64_128
74
+ hashcodecs::murmur3::murmur3_x64_128,
75
+ MURMUR_X64_SIZES
72
76
  );
73
77
  }
74
78
 
@@ -6,9 +6,10 @@ use criterion::{BenchmarkId, Criterion, Throughput, criterion_group, criterion_m
6
6
 
7
7
  mod support;
8
8
 
9
- const SIZES: [usize; 15] = [
9
+ const SIZES: [usize; 16] = [
10
10
  16,
11
11
  17,
12
+ 32,
12
13
  64,
13
14
  128,
14
15
  129,
@@ -23,8 +24,9 @@ const SIZES: [usize; 15] = [
23
24
  1024 * 1024,
24
25
  8 * 1024 * 1024,
25
26
  ];
26
- const MIXED_BATCHES: [(&str, [usize; 4]); 3] = [
27
+ const MIXED_BATCHES: [(&str, [usize; 4]); 4] = [
27
28
  ("two_long_runs", [1024, 1024, 4 * 1024, 4 * 1024]),
29
+ ("nearby_long_lengths", [257, 258, 259, 260]),
28
30
  ("short_then_long_boundary", [240, 240, 241, 241]),
29
31
  ("long_then_short_boundary", [241, 241, 240, 240]),
30
32
  ];
@@ -140,7 +142,7 @@ fn benchmark_batch(c: &mut Criterion, group_name: &str, owned: &[Vec<u8>]) {
140
142
 
141
143
  fn batch(c: &mut Criterion) {
142
144
  for items in [2, 3, 32] {
143
- for size in [64, 1024, 4 * 1024, 1024 * 1024] {
145
+ for size in [64, 241, 1024, 4 * 1024, 1024 * 1024] {
144
146
  let owned = (0..items)
145
147
  .map(|index| data(size, index as u8))
146
148
  .collect::<Vec<_>>();
@@ -158,10 +160,79 @@ fn batch(c: &mut Criterion) {
158
160
  }
159
161
  }
160
162
 
163
+ fn prepared_seed(c: &mut Criterion) {
164
+ for size in [241, 1024] {
165
+ let input = data(size, 17);
166
+ let prepared = hashcodecs::xxhash::PreparedXxh3::new(42);
167
+ assert_eq!(
168
+ prepared.hash_64(&input),
169
+ hashcodecs::xxhash::xxh3_64(&input, 42)
170
+ );
171
+ assert_eq!(
172
+ prepared.hash_128(&input),
173
+ hashcodecs::xxhash::xxh3_128(&input, 42)
174
+ );
175
+
176
+ let mut group = c.benchmark_group(format!("xxh3_prepared/{size}"));
177
+ group.throughput(Throughput::Bytes(size as u64));
178
+ group.bench_function("one_shot_64", |bench| {
179
+ bench.iter(|| hashcodecs::xxhash::xxh3_64(black_box(&input), 42))
180
+ });
181
+ group.bench_function("prepared_64", |bench| {
182
+ bench.iter(|| prepared.hash_64(black_box(&input)))
183
+ });
184
+ group.bench_function("one_shot_128", |bench| {
185
+ bench.iter(|| hashcodecs::xxhash::xxh3_128(black_box(&input), 42))
186
+ });
187
+ group.bench_function("prepared_128", |bench| {
188
+ bench.iter(|| prepared.hash_128(black_box(&input)))
189
+ });
190
+ group.finish();
191
+ }
192
+ }
193
+
194
+ fn prepared_batch(c: &mut Criterion) {
195
+ for items in [2, 4] {
196
+ for size in [241, 1024] {
197
+ let owned = (0..items)
198
+ .map(|index| data(size, index as u8))
199
+ .collect::<Vec<_>>();
200
+ let inputs = owned.iter().map(Vec::as_slice).collect::<Vec<_>>();
201
+ let prepared = hashcodecs::xxhash::PreparedXxh3::new(42);
202
+ assert_eq!(
203
+ prepared.hash_64_batch(&inputs),
204
+ hashcodecs::xxhash::xxh3_64_batch(&inputs, 42)
205
+ );
206
+ assert_eq!(
207
+ prepared.hash_128_batch(&inputs),
208
+ hashcodecs::xxhash::xxh3_128_batch(&inputs, 42)
209
+ );
210
+
211
+ let mut group = c.benchmark_group(format!("xxh3_prepared_batch/{items}_items/{size}"));
212
+ group.throughput(Throughput::Bytes((items * size) as u64));
213
+ group.bench_function("batch_64", |bench| {
214
+ bench.iter(|| hashcodecs::xxhash::xxh3_64_batch(black_box(&inputs), 42))
215
+ });
216
+ group.bench_function("prepared_64", |bench| {
217
+ bench.iter(|| prepared.hash_64_batch(black_box(&inputs)))
218
+ });
219
+ group.bench_function("batch_128", |bench| {
220
+ bench.iter(|| hashcodecs::xxhash::xxh3_128_batch(black_box(&inputs), 42))
221
+ });
222
+ group.bench_function("prepared_128", |bench| {
223
+ bench.iter(|| prepared.hash_128_batch(black_box(&inputs)))
224
+ });
225
+ group.finish();
226
+ }
227
+ }
228
+ }
229
+
161
230
  fn xxhash(c: &mut Criterion) {
162
231
  support::pin_to_one_cpu();
163
232
  one_shot(c);
164
233
  batch(c);
234
+ prepared_seed(c);
235
+ prepared_batch(c);
165
236
  }
166
237
 
167
238
  criterion_group! {