loaderx 2.0.0__tar.gz → 2.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. {loaderx-2.0.0/loaderx.egg-info → loaderx-2.0.1}/PKG-INFO +98 -75
  2. {loaderx-2.0.0 → loaderx-2.0.1}/README.md +97 -74
  3. {loaderx-2.0.0 → loaderx-2.0.1}/build.zig.zon +1 -1
  4. {loaderx-2.0.0 → loaderx-2.0.1}/loaderx/__init__.py +1 -1
  5. {loaderx-2.0.0 → loaderx-2.0.1}/loaderx/zrecord.py +6 -9
  6. {loaderx-2.0.0 → loaderx-2.0.1/loaderx.egg-info}/PKG-INFO +98 -75
  7. {loaderx-2.0.0 → loaderx-2.0.1}/scripts/_bench_common.py +142 -116
  8. {loaderx-2.0.0 → loaderx-2.0.1}/scripts/bench_ragged.py +131 -137
  9. {loaderx-2.0.0 → loaderx-2.0.1}/scripts/test_loaderx.py +7 -7
  10. {loaderx-2.0.0 → loaderx-2.0.1}/LICENSE +0 -0
  11. {loaderx-2.0.0 → loaderx-2.0.1}/MANIFEST.in +0 -0
  12. {loaderx-2.0.0 → loaderx-2.0.1}/build.zig +0 -0
  13. {loaderx-2.0.0 → loaderx-2.0.1}/loaderx/_lib.py +0 -0
  14. {loaderx-2.0.0 → loaderx-2.0.1}/loaderx/_store.py +0 -0
  15. {loaderx-2.0.0 → loaderx-2.0.1}/loaderx/dataloader.py +0 -0
  16. {loaderx-2.0.0 → loaderx-2.0.1}/loaderx/utils.py +0 -0
  17. {loaderx-2.0.0 → loaderx-2.0.1}/loaderx/zsampler.py +0 -0
  18. {loaderx-2.0.0 → loaderx-2.0.1}/loaderx.egg-info/SOURCES.txt +0 -0
  19. {loaderx-2.0.0 → loaderx-2.0.1}/loaderx.egg-info/dependency_links.txt +0 -0
  20. {loaderx-2.0.0 → loaderx-2.0.1}/loaderx.egg-info/requires.txt +0 -0
  21. {loaderx-2.0.0 → loaderx-2.0.1}/loaderx.egg-info/top_level.txt +0 -0
  22. {loaderx-2.0.0 → loaderx-2.0.1}/pyproject.toml +0 -0
  23. {loaderx-2.0.0 → loaderx-2.0.1}/scripts/bench.py +0 -0
  24. {loaderx-2.0.0 → loaderx-2.0.1}/scripts/bench_dense.py +0 -0
  25. {loaderx-2.0.0 → loaderx-2.0.1}/scripts/build_wheels.py +0 -0
  26. {loaderx-2.0.0 → loaderx-2.0.1}/scripts/prepare_tokens.py +0 -0
  27. {loaderx-2.0.0 → loaderx-2.0.1}/scripts/requirements-bench.txt +0 -0
  28. {loaderx-2.0.0 → loaderx-2.0.1}/setup.cfg +0 -0
  29. {loaderx-2.0.0 → loaderx-2.0.1}/setup.py +0 -0
  30. {loaderx-2.0.0 → loaderx-2.0.1}/src/record/engine.zig +0 -0
  31. {loaderx-2.0.0 → loaderx-2.0.1}/src/record/executor.zig +0 -0
  32. {loaderx-2.0.0 → loaderx-2.0.1}/src/record/storage.zig +0 -0
  33. {loaderx-2.0.0 → loaderx-2.0.1}/src/store.zig +0 -0
  34. {loaderx-2.0.0 → loaderx-2.0.1}/src/zsampler.zig +0 -0
  35. {loaderx-2.0.0 → loaderx-2.0.1}/src/zstd/c.zig +0 -0
  36. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/COPYING +0 -0
  37. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/LICENSE +0 -0
  38. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/common/allocations.h +0 -0
  39. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/common/bits.h +0 -0
  40. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/common/bitstream.h +0 -0
  41. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/common/compiler.h +0 -0
  42. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/common/cpu.h +0 -0
  43. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/common/debug.c +0 -0
  44. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/common/debug.h +0 -0
  45. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/common/entropy_common.c +0 -0
  46. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/common/error_private.c +0 -0
  47. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/common/error_private.h +0 -0
  48. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/common/fse.h +0 -0
  49. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/common/fse_decompress.c +0 -0
  50. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/common/huf.h +0 -0
  51. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/common/mem.h +0 -0
  52. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/common/pool.c +0 -0
  53. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/common/pool.h +0 -0
  54. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/common/portability_macros.h +0 -0
  55. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/common/threading.c +0 -0
  56. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/common/threading.h +0 -0
  57. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/common/xxhash.c +0 -0
  58. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/common/xxhash.h +0 -0
  59. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/common/zstd_common.c +0 -0
  60. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/common/zstd_deps.h +0 -0
  61. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/common/zstd_internal.h +0 -0
  62. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/common/zstd_trace.h +0 -0
  63. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/clevels.h +0 -0
  64. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/fse_compress.c +0 -0
  65. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/hist.c +0 -0
  66. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/hist.h +0 -0
  67. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/huf_compress.c +0 -0
  68. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/zstd_compress.c +0 -0
  69. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/zstd_compress_internal.h +0 -0
  70. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/zstd_compress_literals.c +0 -0
  71. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/zstd_compress_literals.h +0 -0
  72. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/zstd_compress_sequences.c +0 -0
  73. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/zstd_compress_sequences.h +0 -0
  74. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/zstd_compress_superblock.c +0 -0
  75. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/zstd_compress_superblock.h +0 -0
  76. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/zstd_cwksp.h +0 -0
  77. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/zstd_double_fast.c +0 -0
  78. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/zstd_double_fast.h +0 -0
  79. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/zstd_fast.c +0 -0
  80. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/zstd_fast.h +0 -0
  81. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/zstd_lazy.c +0 -0
  82. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/zstd_lazy.h +0 -0
  83. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/zstd_ldm.c +0 -0
  84. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/zstd_ldm.h +0 -0
  85. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/zstd_ldm_geartab.h +0 -0
  86. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/zstd_opt.c +0 -0
  87. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/zstd_opt.h +0 -0
  88. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/zstdmt_compress.c +0 -0
  89. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/compress/zstdmt_compress.h +0 -0
  90. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/decompress/huf_decompress.c +0 -0
  91. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/decompress/zstd_ddict.c +0 -0
  92. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/decompress/zstd_ddict.h +0 -0
  93. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/decompress/zstd_decompress.c +0 -0
  94. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/decompress/zstd_decompress_block.c +0 -0
  95. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/decompress/zstd_decompress_block.h +0 -0
  96. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/decompress/zstd_decompress_internal.h +0 -0
  97. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/dictBuilder/cover.c +0 -0
  98. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/dictBuilder/cover.h +0 -0
  99. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/dictBuilder/divsufsort.c +0 -0
  100. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/dictBuilder/divsufsort.h +0 -0
  101. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/dictBuilder/fastcover.c +0 -0
  102. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/dictBuilder/zdict.c +0 -0
  103. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/zdict.h +0 -0
  104. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/zstd.h +0 -0
  105. {loaderx-2.0.0 → loaderx-2.0.1}/vendor/zstd/lib/zstd_errors.h +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: loaderx
3
- Version: 2.0.0
3
+ Version: 2.0.1
4
4
  Summary: Rebuildable high-performance record containers
5
5
  Author-email: Ben0i0d <ben0i0d@foxmail.com>
6
6
  License-Expression: MIT
@@ -121,8 +121,9 @@ Python defines the exact record schema: ``dtype`` plus the dense ``item_shape``,
121
121
  or only ``dtype`` for ragged stores. The MsgPack bytes live opaquely in the
122
122
  static page at the front of ``meta.zr``; Zig persists them but never interprets
123
123
  them. The schema accepts no user metadata. Python selects the geometry and gives
124
- the private native engine only the runtime record boundaries it needs. Each Ragged record
125
- carries its shape in an inline little-endian u64 prefix. One native physical
124
+ the private native engine only the runtime record boundaries it needs. Each
125
+ Ragged record carries a u8 rank followed by inline little-endian u64 dimensions.
126
+ One native physical
126
127
  engine consumes the trusted Dense stride or Ragged offsets. Append inputs are
127
128
  strictly NumPy arrays: Dense takes one batched ndarray and Ragged takes an
128
129
  iterable of ndarrays. Raw bytes and pre-encoded images are made explicit with
@@ -461,29 +462,50 @@ so CFFI, NumPy allocation, and Ragged list/shape reconstruction are timed.
461
462
 
462
463
  ### Methodology
463
464
 
464
- The results below are one complete qualitative pass from the same checkout on a
465
- warm page cache. They are not three-run medians: an unexpected result is traced
466
- separately instead of being hidden by repeated aggregation. Within each store workload every backend
467
- receives identical source records and random index plans. Loader backends receive
465
+ The results below are one complete run from the same checkout on a warm page
466
+ cache. They are not three-run medians: an unexpected result is traced separately
467
+ instead of being hidden by repeated aggregation. Within each store workload every
468
+ backend receives identical source records. Correctness and timing use independent
469
+ deterministic `zsampler` IID streams. Loader backends receive
468
470
  the same source and seed but use their own shipped samplers, so their exact
469
471
  permutations differ. Before gather timing, the store benchmark sweeps every record and
470
- validates every planned result for exact
472
+ validates a fixed number of IID batches for exact
471
473
  dtype, shape, order, and values. `logical write` and `logical gather` divide
472
474
  uncompressed NumPy payload bytes by elapsed time; they measure bytes accepted or
473
475
  returned by the public API, not physical storage bandwidth. Writable containers
474
- are created before timing. `logical write` times only append/write/assignment
475
- calls and stops when the final call returns; close, commit, format finalization
476
- and Zrecord Header publication happen afterward. No backend requests `fsync`.
477
- Reusable source preparation is outside that timer; in particular, ``zstd_dict``
478
- trains its standalone dictionary first and byte-record adapters encode input
479
- before their timed write calls.
476
+ are created before timing. `logical write` starts when the already-generated
477
+ source enters the backend, includes byte encoding, packing, key construction and
478
+ Arrow array construction, and ends after logical commit/finalization returns. No
479
+ backend requests `fsync`, LMDB `env.sync()`, or another stable-media durability
480
+ operation. Reusable source preparation is outside that timer; in particular,
481
+ ``zstd_dict`` trains its standalone dictionary first.
482
+
483
+ The measured Store paths are explicit:
484
+
485
+ ```text
486
+ write: make_data returns -> writer setup [untimed]
487
+ -> start -> encode/pack -> append/write -> logical finalize -> return -> stop
488
+ -> resource-only cleanup [untimed]
489
+
490
+ read: open -> full warm sweep -> IID correctness stream(seed) [untimed]
491
+ -> IID timing stream(seed + 1): sampler.next() [untimed]
492
+ -> start one gather -> public return -> stop
493
+ -> destroy returned batch [untimed]
494
+ -> repeat timed calls until their accumulated time is at least 2 seconds
495
+ ```
496
+
497
+ Logical finalize means Zrecord Header publication, LMDB transaction commit, or
498
+ an Arrow/Parquet footer; none of these paths requests stable-media sync.
480
499
  Disk size is allocated blocks, not sparse apparent size. `krecords/s` is gather
481
500
  record throughput and `p95` is the 95th-percentile latency of one random gather
482
501
  batch. Results are comparable within one workload table, not across payload
483
502
  distributions or geometries.
484
- The finalized output is opened read-only before timing. Every random gather plan
485
- is timed exactly once; output allocation, reads, decompression and reconstruction
486
- are included, while open and close are not.
503
+ The finalized output is opened read-only before timing. After the warm sweep and
504
+ correctness stream, an independent IID stream draws fresh indices until timed
505
+ gather calls accumulate at least two seconds. IID sampling is uniform with
506
+ replacement, and sampler time is excluded. Output allocation, reads, decompression and
507
+ reconstruction are included, while open, close and destruction after return are
508
+ not.
487
509
  Every backend name states its actual codec; the full default set is required
488
510
  rather than silently skipped when a package is missing.
489
511
  Dense and Ragged use the same CHW RGB image generator, record count, batch plan
@@ -519,24 +541,24 @@ Fixed-resolution vision records — 147 KiB per record, 36.8 MiB per batch:
519
541
 
520
542
  | backend | logical write | logical gather | krecords/s | p95 | disk | ratio |
521
543
  |---|---:|---:|---:|---:|---:|---:|
522
- | zrecord-zstd | 7206 MiB/s | 7395 MiB/s | 51.5 | 5.82 ms | 24.4 MiB | 14.69x |
523
- | zrecord-zstdict | 30 MiB/s | 7138 MiB/s | 49.7 | 6.05 ms | 15.8 MiB | 22.74x |
524
- | zrecord-raw | 3117 MiB/s | 9590 MiB/s | 66.8 | 4.70 ms | 358.9 MiB | 1.00x |
525
- | npy-mmap-raw | 2314 MiB/s | 4152 MiB/s | 28.9 | 10.74 ms | 358.9 MiB | 1.00x |
526
- | hdf5-raw | 2499 MiB/s | 1810 MiB/s | 12.6 | 25.49 ms | 359.0 MiB | 1.00x |
527
- | hdf5-gzip | 257 MiB/s | 592 MiB/s | 4.1 | 68.37 ms | 26.1 MiB | 13.73x |
528
- | lmdb-raw | 4291 MiB/s | 3800 MiB/s | 26.5 | 10.44 ms | 361.4 MiB | 0.99x |
529
- | arrow-ipc-raw | 3070 MiB/s | 2965 MiB/s | 20.7 | 16.28 ms | 358.9 MiB | 1.00x |
530
- | arrow-ipc-zstd | 700 MiB/s | 171 MiB/s | 1.2 | 254.38 ms | 22.6 MiB | 15.86x |
531
- | parquet-raw | 1700 MiB/s | 470 MiB/s | 3.3 | 86.56 ms | 358.9 MiB | 1.00x |
532
- | parquet-zstd | 613 MiB/s | 150 MiB/s | 1.0 | 289.02 ms | 22.6 MiB | 15.86x |
533
- | arrayrecord-raw | 2012 MiB/s | 1855 MiB/s | 12.9 | 24.12 ms | 359.2 MiB | 1.00x |
534
- | arrayrecord-zstd | 1024 MiB/s | 1531 MiB/s | 10.7 | 34.26 ms | 25.1 MiB | 14.32x |
535
- | tiledb-raw | 692 MiB/s | 668 MiB/s | 4.7 | 59.75 ms | 359.0 MiB | 1.00x |
536
- | tiledb-zstd | 1253 MiB/s | 1478 MiB/s | 10.3 | 27.34 ms | 26.9 MiB | 13.36x |
537
-
538
- At 147 KiB per record, Zrecord-raw reaches 9.4 GiB/s and is 2.3x npy-mmap-raw;
539
- plain zstd gathers at 7.2 GiB/s while reducing the corpus 14.69x. LMDB and Arrow
544
+ | zrecord-zstd | 7091 MiB/s | 11770 MiB/s | 82.0 | 3.97 ms | 24.4 MiB | 14.69x |
545
+ | zrecord-zstdict | 34 MiB/s | 11801 MiB/s | 82.2 | 4.12 ms | 15.8 MiB | 22.74x |
546
+ | zrecord-raw | 2708 MiB/s | 15883 MiB/s | 110.6 | 2.65 ms | 358.9 MiB | 1.00x |
547
+ | npy-mmap-raw | 2055 MiB/s | 4738 MiB/s | 33.0 | 11.00 ms | 358.9 MiB | 1.00x |
548
+ | hdf5-raw | 2360 MiB/s | 1948 MiB/s | 13.6 | 27.31 ms | 359.0 MiB | 1.00x |
549
+ | hdf5-gzip | 281 MiB/s | 641 MiB/s | 4.5 | 65.88 ms | 26.1 MiB | 13.73x |
550
+ | lmdb-raw | 1714 MiB/s | 4375 MiB/s | 30.5 | 10.83 ms | 361.4 MiB | 0.99x |
551
+ | arrow-ipc-raw | 1817 MiB/s | 3480 MiB/s | 24.2 | 14.26 ms | 358.9 MiB | 1.00x |
552
+ | arrow-ipc-zstd | 611 MiB/s | 177 MiB/s | 1.2 | 228.63 ms | 22.6 MiB | 15.86x |
553
+ | parquet-raw | 1341 MiB/s | 487 MiB/s | 3.4 | 87.83 ms | 358.9 MiB | 1.00x |
554
+ | parquet-zstd | 644 MiB/s | 162 MiB/s | 1.1 | 246.21 ms | 22.6 MiB | 15.86x |
555
+ | arrayrecord-raw | 1543 MiB/s | 2107 MiB/s | 14.7 | 21.69 ms | 359.2 MiB | 1.00x |
556
+ | arrayrecord-zstd | 917 MiB/s | 1586 MiB/s | 11.0 | 30.72 ms | 25.1 MiB | 14.32x |
557
+ | tiledb-raw | 712 MiB/s | 688 MiB/s | 4.8 | 58.83 ms | 359.0 MiB | 1.00x |
558
+ | tiledb-zstd | 1572 MiB/s | 1525 MiB/s | 10.6 | 26.80 ms | 26.9 MiB | 13.36x |
559
+
560
+ At 147 KiB per record, Zrecord-raw reaches 15.5 GiB/s and is 3.4x npy-mmap-raw;
561
+ plain zstd gathers at 11.5 GiB/s while reducing the corpus 14.69x. LMDB and Arrow
540
562
  IPC are competitive raw record
541
563
  stores, while codecs tied to whole IPC batches or Parquet row groups pay read
542
564
  amplification on random gathers. Dense demonstrates that
@@ -554,23 +576,23 @@ list or a one-dimensional variable-length abstraction is not enough.
554
576
 
555
577
  | backend | logical write | logical gather | krecords/s | p95 | disk | ratio |
556
578
  |---|---:|---:|---:|---:|---:|---:|
557
- | zrecord-zstd | 2220 MiB/s | 5666 MiB/s | 34.2 | 8.41 ms | 26.9 MiB | 15.52x |
558
- | zrecord-zstdict | 28 MiB/s | 5681 MiB/s | 34.3 | 8.55 ms | 17.7 MiB | 23.58x |
559
- | zrecord-raw | 1460 MiB/s | 6806 MiB/s | 41.1 | 7.58 ms | 417.2 MiB | 1.00x |
560
- | hdf5-raw | 1610 MiB/s | 1108 MiB/s | 6.7 | 45.80 ms | 418.0 MiB | 1.00x |
561
- | hdf5-gzip | 229 MiB/s | 114 MiB/s | 0.7 | 415.95 ms | 29.9 MiB | 13.94x |
562
- | lmdb-raw | 4126 MiB/s | 6837 MiB/s | 41.2 | 8.14 ms | 422.2 MiB | 0.99x |
563
- | arrow-ipc-raw | 3213 MiB/s | 4251 MiB/s | 25.6 | 12.90 ms | 417.2 MiB | 1.00x |
564
- | arrow-ipc-zstd | 706 MiB/s | 177 MiB/s | 1.1 | 266.37 ms | 25.9 MiB | 16.10x |
565
- | parquet-raw | 2219 MiB/s | 505 MiB/s | 3.0 | 92.79 ms | 417.2 MiB | 1.00x |
566
- | parquet-zstd | 728 MiB/s | 158 MiB/s | 1.0 | 296.60 ms | 25.9 MiB | 16.10x |
567
- | arrayrecord-raw | 1650 MiB/s | 2571 MiB/s | 15.5 | 18.34 ms | 417.6 MiB | 1.00x |
568
- | arrayrecord-zstd | 941 MiB/s | 1420 MiB/s | 8.6 | 37.14 ms | 27.2 MiB | 15.31x |
569
- | tiledb-raw | 443 MiB/s | 49 MiB/s | 0.3 | 920.09 ms | 417.2 MiB | 1.00x |
570
- | tiledb-zstd | 727 MiB/s | 124 MiB/s | 0.7 | 375.98 ms | 27.0 MiB | 15.46x |
571
-
572
- Zrecord-raw and LMDB are effectively tied in this pass; Arrow IPC is
573
- the strongest raw typed-file alternative. Zrecord-zstd delivers 5.5 GiB/s of
579
+ | zrecord-zstd | 2502 MiB/s | 10071 MiB/s | 60.5 | 5.06 ms | 26.9 MiB | 15.52x |
580
+ | zrecord-zstdict | 36 MiB/s | 10930 MiB/s | 65.6 | 4.62 ms | 17.7 MiB | 23.58x |
581
+ | zrecord-raw | 1661 MiB/s | 13056 MiB/s | 78.4 | 3.84 ms | 417.2 MiB | 1.00x |
582
+ | hdf5-raw | 1578 MiB/s | 1359 MiB/s | 8.2 | 37.26 ms | 418.0 MiB | 1.00x |
583
+ | hdf5-gzip | 243 MiB/s | 130 MiB/s | 0.8 | 356.26 ms | 29.9 MiB | 13.94x |
584
+ | lmdb-raw | 1790 MiB/s | 7412 MiB/s | 44.5 | 7.48 ms | 422.2 MiB | 0.99x |
585
+ | arrow-ipc-raw | 1572 MiB/s | 4070 MiB/s | 24.4 | 14.56 ms | 417.2 MiB | 1.00x |
586
+ | arrow-ipc-zstd | 620 MiB/s | 208 MiB/s | 1.2 | 232.74 ms | 25.9 MiB | 16.10x |
587
+ | parquet-raw | 889 MiB/s | 481 MiB/s | 2.9 | 93.87 ms | 417.2 MiB | 1.00x |
588
+ | parquet-zstd | 498 MiB/s | 184 MiB/s | 1.1 | 251.14 ms | 25.9 MiB | 16.10x |
589
+ | arrayrecord-raw | 1644 MiB/s | 2642 MiB/s | 15.9 | 18.22 ms | 417.6 MiB | 1.00x |
590
+ | arrayrecord-zstd | 829 MiB/s | 1556 MiB/s | 9.3 | 37.43 ms | 27.2 MiB | 15.31x |
591
+ | tiledb-raw | 411 MiB/s | 55 MiB/s | 0.3 | 806.95 ms | 417.2 MiB | 1.00x |
592
+ | tiledb-zstd | 713 MiB/s | 151 MiB/s | 0.9 | 307.58 ms | 27.0 MiB | 15.46x |
593
+
594
+ Zrecord-raw is 1.8x LMDB and 3.2x Arrow IPC in logical gather. Zrecord-zstd
595
+ delivers 9.8 GiB/s of
574
596
  logical payload while reducing the corpus to 26.9 MiB. HDF5, Arrow IPC, Parquet,
575
597
  ArrayRecord and TileDB
576
598
  show the same framework/codec tradeoffs in both tables; compressed batch, chunk
@@ -593,44 +615,45 @@ combines fragments shorter than 16 tokens, and splits records at 512 tokens into
593
615
 
594
616
  The Dense workload ignores text boundaries and packs the stream into 200,000
595
617
  fixed `int32[512]` records: 2 KiB per record and 390.6 MiB logical payload.
596
- Each of 100 random batches gathers 256 records.
618
+ Each IID batch gathers 256 records; fresh draws continue until timed gathers
619
+ accumulate at least two seconds.
597
620
 
598
621
  | backend | logical write | logical gather | krecords/s | p95 | disk | ratio |
599
622
  |---|---:|---:|---:|---:|---:|---:|
600
- | zrecord-zstd | 346 MiB/s | 1681 MiB/s | 860.9 | 0.38 ms | 193.2 MiB | 2.02x |
601
- | zrecord-zstdict | 41 MiB/s | 1892 MiB/s | 968.8 | 0.33 ms | 153.8 MiB | 2.54x |
602
- | zrecord-raw | 2477 MiB/s | 6261 MiB/s | 3205.5 | 0.11 ms | 393.7 MiB | 0.99x |
603
- | npy-mmap-raw | 2024 MiB/s | 7897 MiB/s | 4043.3 | 0.08 ms | 390.6 MiB | 1.00x |
604
- | lmdb-raw | 1022 MiB/s | 1066 MiB/s | 545.9 | 0.53 ms | 786.3 MiB | 0.50x |
605
- | arrow-ipc-raw | 2744 MiB/s | 207 MiB/s | 105.8 | 2.60 ms | 390.8 MiB | 1.00x |
606
- | arrayrecord-raw | 905 MiB/s | 149 MiB/s | 76.4 | 6.33 ms | 401.4 MiB | 0.97x |
607
- | arrayrecord-zstd | 114 MiB/s | 160 MiB/s | 82.0 | 4.83 ms | 201.3 MiB | 1.94x |
623
+ | zrecord-zstd | 456 MiB/s | 2047 MiB/s | 1047.9 | 0.32 ms | 193.2 MiB | 2.02x |
624
+ | zrecord-zstdict | 54 MiB/s | 2393 MiB/s | 1225.3 | 0.28 ms | 153.8 MiB | 2.54x |
625
+ | zrecord-raw | 2550 MiB/s | 7688 MiB/s | 3936.3 | 0.08 ms | 393.7 MiB | 0.99x |
626
+ | npy-mmap-raw | 2139 MiB/s | 10709 MiB/s | 5482.9 | 0.07 ms | 390.6 MiB | 1.00x |
627
+ | lmdb-raw | 721 MiB/s | 1166 MiB/s | 597.0 | 0.55 ms | 786.3 MiB | 0.50x |
628
+ | arrow-ipc-raw | 2335 MiB/s | 231 MiB/s | 118.2 | 2.53 ms | 390.8 MiB | 1.00x |
629
+ | arrayrecord-raw | 847 MiB/s | 162 MiB/s | 83.0 | 4.98 ms | 401.4 MiB | 0.97x |
630
+ | arrayrecord-zstd | 124 MiB/s | 133 MiB/s | 68.3 | 4.68 ms | 201.3 MiB | 1.94x |
608
631
 
609
632
  The contiguous NumPy baseline is strongest when the whole corpus is one fixed
610
- typed matrix. Zrecord-raw reaches 3.21 Mrecords/s while retaining independent
633
+ typed matrix. Zrecord-raw reaches 3.94 Mrecords/s while retaining independent
611
634
  record semantics; the per-record zstd codecs halve disk and still return
612
- 0.86–0.97 Mrecords/s. LMDB's B-tree/page overhead is visible in both throughput
635
+ 1.05–1.23 Mrecords/s. LMDB's B-tree/page overhead is visible in both throughput
613
636
  and disk.
614
637
 
615
638
  #### Variable Token Sequences
616
639
 
617
640
  The Ragged workload keeps 200,000 real text records of 16..512 tokens: p10 28,
618
641
  median 129, mean 138.8, p90 254, totaling 105.9 MiB. It uses the same 256-record,
619
- 100-batch random plan and every backend must return ordered `list[np.ndarray]`
642
+ random plan and every backend must return ordered `list[np.ndarray]`
620
643
  with exact `int32` values and original one-dimensional shapes.
621
644
 
622
645
  | backend | logical write | logical gather | krecords/s | p95 | disk | ratio |
623
646
  |---|---:|---:|---:|---:|---:|---:|
624
- | zrecord-zstd | 105 MiB/s | 225 MiB/s | 426.5 | 0.76 ms | 67.8 MiB | 1.56x |
625
- | zrecord-zstdict | 37 MiB/s | 277 MiB/s | 526.3 | 0.63 ms | 49.4 MiB | 2.14x |
626
- | zrecord-raw | 457 MiB/s | 325 MiB/s | 616.1 | 0.46 ms | 112.0 MiB | 0.95x |
627
- | lmdb-raw | 887 MiB/s | 106 MiB/s | 200.7 | 1.34 ms | 153.0 MiB | 0.69x |
628
- | arrow-ipc-raw | 2230 MiB/s | 39 MiB/s | 74.1 | 4.48 ms | 109.9 MiB | 0.96x |
629
- | arrayrecord-raw | 399 MiB/s | 24 MiB/s | 45.7 | 8.34 ms | 118.8 MiB | 0.89x |
630
- | arrayrecord-zstd | 59 MiB/s | 27 MiB/s | 51.1 | 7.79 ms | 76.2 MiB | 1.39x |
647
+ | zrecord-zstd | 114 MiB/s | 291 MiB/s | 550.1 | 0.59 ms | 67.5 MiB | 1.57x |
648
+ | zrecord-zstdict | 47 MiB/s | 325 MiB/s | 614.3 | 0.53 ms | 49.2 MiB | 2.15x |
649
+ | zrecord-raw | 571 MiB/s | 352 MiB/s | 665.5 | 0.50 ms | 110.7 MiB | 0.96x |
650
+ | lmdb-raw | 341 MiB/s | 116 MiB/s | 218.9 | 1.32 ms | 153.0 MiB | 0.69x |
651
+ | arrow-ipc-raw | 567 MiB/s | 44 MiB/s | 84.1 | 3.66 ms | 109.9 MiB | 0.96x |
652
+ | arrayrecord-raw | 290 MiB/s | 28 MiB/s | 53.5 | 7.22 ms | 118.8 MiB | 0.89x |
653
+ | arrayrecord-zstd | 61 MiB/s | 34 MiB/s | 64.1 | 5.12 ms | 76.2 MiB | 1.39x |
631
654
 
632
655
  Here the record contract, not bulk byte bandwidth, is the useful scale.
633
- Zrecord's three codecs return 427–616 krecords/s with less than 0.8 ms p95;
656
+ Zrecord's three codecs return 550–666 krecords/s with less than 0.6 ms p95;
634
657
  the dictionary gives the best disk ratio and is slightly ahead of plain zstd in
635
658
  this pass.
636
659
 
@@ -773,13 +796,13 @@ explicit solutions rather than silently changing the default transform.
773
796
  **What these numbers do not claim.** Everything runs with a warm page cache: this
774
797
  measures the access path, not cold storage or disk. `disk` is allocated
775
798
  blocks, and zrecord files grow to their written frontier. `logical write`
776
- measures append/write calls after writer setup and before close/finalization; it
799
+ measures source adaptation through logical finalization after writer setup; it
777
800
  does not benchmark durability, stronger transactional guarantees, or reusable
778
801
  dictionary training. Arrow IPC and Parquet use 256-record groups, so random
779
802
  batches pay their real group-level read amplification. Ragged Zrecord arrays are
780
803
  views into one batch allocation, while most byte-store adapters return
781
- independent copies. This is a single
782
- qualitative pass —
804
+ independent copies. This is a single benchmark run, while each Store read path
805
+ accumulates at least two timed seconds —
783
806
  the µs-scale sampler timings and the loader `batches/s` fluctuate with box load
784
807
  (on these 12 cores the compressed loader trails the raw one by about 25%), so
785
808
  treat the absolute numbers as ballpark and the cross-backend margins as the
@@ -96,8 +96,9 @@ Python defines the exact record schema: ``dtype`` plus the dense ``item_shape``,
96
96
  or only ``dtype`` for ragged stores. The MsgPack bytes live opaquely in the
97
97
  static page at the front of ``meta.zr``; Zig persists them but never interprets
98
98
  them. The schema accepts no user metadata. Python selects the geometry and gives
99
- the private native engine only the runtime record boundaries it needs. Each Ragged record
100
- carries its shape in an inline little-endian u64 prefix. One native physical
99
+ the private native engine only the runtime record boundaries it needs. Each
100
+ Ragged record carries a u8 rank followed by inline little-endian u64 dimensions.
101
+ One native physical
101
102
  engine consumes the trusted Dense stride or Ragged offsets. Append inputs are
102
103
  strictly NumPy arrays: Dense takes one batched ndarray and Ragged takes an
103
104
  iterable of ndarrays. Raw bytes and pre-encoded images are made explicit with
@@ -436,29 +437,50 @@ so CFFI, NumPy allocation, and Ragged list/shape reconstruction are timed.
436
437
 
437
438
  ### Methodology
438
439
 
439
- The results below are one complete qualitative pass from the same checkout on a
440
- warm page cache. They are not three-run medians: an unexpected result is traced
441
- separately instead of being hidden by repeated aggregation. Within each store workload every backend
442
- receives identical source records and random index plans. Loader backends receive
440
+ The results below are one complete run from the same checkout on a warm page
441
+ cache. They are not three-run medians: an unexpected result is traced separately
442
+ instead of being hidden by repeated aggregation. Within each store workload every
443
+ backend receives identical source records. Correctness and timing use independent
444
+ deterministic `zsampler` IID streams. Loader backends receive
443
445
  the same source and seed but use their own shipped samplers, so their exact
444
446
  permutations differ. Before gather timing, the store benchmark sweeps every record and
445
- validates every planned result for exact
447
+ validates a fixed number of IID batches for exact
446
448
  dtype, shape, order, and values. `logical write` and `logical gather` divide
447
449
  uncompressed NumPy payload bytes by elapsed time; they measure bytes accepted or
448
450
  returned by the public API, not physical storage bandwidth. Writable containers
449
- are created before timing. `logical write` times only append/write/assignment
450
- calls and stops when the final call returns; close, commit, format finalization
451
- and Zrecord Header publication happen afterward. No backend requests `fsync`.
452
- Reusable source preparation is outside that timer; in particular, ``zstd_dict``
453
- trains its standalone dictionary first and byte-record adapters encode input
454
- before their timed write calls.
451
+ are created before timing. `logical write` starts when the already-generated
452
+ source enters the backend, includes byte encoding, packing, key construction and
453
+ Arrow array construction, and ends after logical commit/finalization returns. No
454
+ backend requests `fsync`, LMDB `env.sync()`, or another stable-media durability
455
+ operation. Reusable source preparation is outside that timer; in particular,
456
+ ``zstd_dict`` trains its standalone dictionary first.
457
+
458
+ The measured Store paths are explicit:
459
+
460
+ ```text
461
+ write: make_data returns -> writer setup [untimed]
462
+ -> start -> encode/pack -> append/write -> logical finalize -> return -> stop
463
+ -> resource-only cleanup [untimed]
464
+
465
+ read: open -> full warm sweep -> IID correctness stream(seed) [untimed]
466
+ -> IID timing stream(seed + 1): sampler.next() [untimed]
467
+ -> start one gather -> public return -> stop
468
+ -> destroy returned batch [untimed]
469
+ -> repeat timed calls until their accumulated time is at least 2 seconds
470
+ ```
471
+
472
+ Logical finalize means Zrecord Header publication, LMDB transaction commit, or
473
+ an Arrow/Parquet footer; none of these paths requests stable-media sync.
455
474
  Disk size is allocated blocks, not sparse apparent size. `krecords/s` is gather
456
475
  record throughput and `p95` is the 95th-percentile latency of one random gather
457
476
  batch. Results are comparable within one workload table, not across payload
458
477
  distributions or geometries.
459
- The finalized output is opened read-only before timing. Every random gather plan
460
- is timed exactly once; output allocation, reads, decompression and reconstruction
461
- are included, while open and close are not.
478
+ The finalized output is opened read-only before timing. After the warm sweep and
479
+ correctness stream, an independent IID stream draws fresh indices until timed
480
+ gather calls accumulate at least two seconds. IID sampling is uniform with
481
+ replacement, and sampler time is excluded. Output allocation, reads, decompression and
482
+ reconstruction are included, while open, close and destruction after return are
483
+ not.
462
484
  Every backend name states its actual codec; the full default set is required
463
485
  rather than silently skipped when a package is missing.
464
486
  Dense and Ragged use the same CHW RGB image generator, record count, batch plan
@@ -494,24 +516,24 @@ Fixed-resolution vision records — 147 KiB per record, 36.8 MiB per batch:
494
516
 
495
517
  | backend | logical write | logical gather | krecords/s | p95 | disk | ratio |
496
518
  |---|---:|---:|---:|---:|---:|---:|
497
- | zrecord-zstd | 7206 MiB/s | 7395 MiB/s | 51.5 | 5.82 ms | 24.4 MiB | 14.69x |
498
- | zrecord-zstdict | 30 MiB/s | 7138 MiB/s | 49.7 | 6.05 ms | 15.8 MiB | 22.74x |
499
- | zrecord-raw | 3117 MiB/s | 9590 MiB/s | 66.8 | 4.70 ms | 358.9 MiB | 1.00x |
500
- | npy-mmap-raw | 2314 MiB/s | 4152 MiB/s | 28.9 | 10.74 ms | 358.9 MiB | 1.00x |
501
- | hdf5-raw | 2499 MiB/s | 1810 MiB/s | 12.6 | 25.49 ms | 359.0 MiB | 1.00x |
502
- | hdf5-gzip | 257 MiB/s | 592 MiB/s | 4.1 | 68.37 ms | 26.1 MiB | 13.73x |
503
- | lmdb-raw | 4291 MiB/s | 3800 MiB/s | 26.5 | 10.44 ms | 361.4 MiB | 0.99x |
504
- | arrow-ipc-raw | 3070 MiB/s | 2965 MiB/s | 20.7 | 16.28 ms | 358.9 MiB | 1.00x |
505
- | arrow-ipc-zstd | 700 MiB/s | 171 MiB/s | 1.2 | 254.38 ms | 22.6 MiB | 15.86x |
506
- | parquet-raw | 1700 MiB/s | 470 MiB/s | 3.3 | 86.56 ms | 358.9 MiB | 1.00x |
507
- | parquet-zstd | 613 MiB/s | 150 MiB/s | 1.0 | 289.02 ms | 22.6 MiB | 15.86x |
508
- | arrayrecord-raw | 2012 MiB/s | 1855 MiB/s | 12.9 | 24.12 ms | 359.2 MiB | 1.00x |
509
- | arrayrecord-zstd | 1024 MiB/s | 1531 MiB/s | 10.7 | 34.26 ms | 25.1 MiB | 14.32x |
510
- | tiledb-raw | 692 MiB/s | 668 MiB/s | 4.7 | 59.75 ms | 359.0 MiB | 1.00x |
511
- | tiledb-zstd | 1253 MiB/s | 1478 MiB/s | 10.3 | 27.34 ms | 26.9 MiB | 13.36x |
512
-
513
- At 147 KiB per record, Zrecord-raw reaches 9.4 GiB/s and is 2.3x npy-mmap-raw;
514
- plain zstd gathers at 7.2 GiB/s while reducing the corpus 14.69x. LMDB and Arrow
519
+ | zrecord-zstd | 7091 MiB/s | 11770 MiB/s | 82.0 | 3.97 ms | 24.4 MiB | 14.69x |
520
+ | zrecord-zstdict | 34 MiB/s | 11801 MiB/s | 82.2 | 4.12 ms | 15.8 MiB | 22.74x |
521
+ | zrecord-raw | 2708 MiB/s | 15883 MiB/s | 110.6 | 2.65 ms | 358.9 MiB | 1.00x |
522
+ | npy-mmap-raw | 2055 MiB/s | 4738 MiB/s | 33.0 | 11.00 ms | 358.9 MiB | 1.00x |
523
+ | hdf5-raw | 2360 MiB/s | 1948 MiB/s | 13.6 | 27.31 ms | 359.0 MiB | 1.00x |
524
+ | hdf5-gzip | 281 MiB/s | 641 MiB/s | 4.5 | 65.88 ms | 26.1 MiB | 13.73x |
525
+ | lmdb-raw | 1714 MiB/s | 4375 MiB/s | 30.5 | 10.83 ms | 361.4 MiB | 0.99x |
526
+ | arrow-ipc-raw | 1817 MiB/s | 3480 MiB/s | 24.2 | 14.26 ms | 358.9 MiB | 1.00x |
527
+ | arrow-ipc-zstd | 611 MiB/s | 177 MiB/s | 1.2 | 228.63 ms | 22.6 MiB | 15.86x |
528
+ | parquet-raw | 1341 MiB/s | 487 MiB/s | 3.4 | 87.83 ms | 358.9 MiB | 1.00x |
529
+ | parquet-zstd | 644 MiB/s | 162 MiB/s | 1.1 | 246.21 ms | 22.6 MiB | 15.86x |
530
+ | arrayrecord-raw | 1543 MiB/s | 2107 MiB/s | 14.7 | 21.69 ms | 359.2 MiB | 1.00x |
531
+ | arrayrecord-zstd | 917 MiB/s | 1586 MiB/s | 11.0 | 30.72 ms | 25.1 MiB | 14.32x |
532
+ | tiledb-raw | 712 MiB/s | 688 MiB/s | 4.8 | 58.83 ms | 359.0 MiB | 1.00x |
533
+ | tiledb-zstd | 1572 MiB/s | 1525 MiB/s | 10.6 | 26.80 ms | 26.9 MiB | 13.36x |
534
+
535
+ At 147 KiB per record, Zrecord-raw reaches 15.5 GiB/s and is 3.4x npy-mmap-raw;
536
+ plain zstd gathers at 11.5 GiB/s while reducing the corpus 14.69x. LMDB and Arrow
515
537
  IPC are competitive raw record
516
538
  stores, while codecs tied to whole IPC batches or Parquet row groups pay read
517
539
  amplification on random gathers. Dense demonstrates that
@@ -529,23 +551,23 @@ list or a one-dimensional variable-length abstraction is not enough.
529
551
 
530
552
  | backend | logical write | logical gather | krecords/s | p95 | disk | ratio |
531
553
  |---|---:|---:|---:|---:|---:|---:|
532
- | zrecord-zstd | 2220 MiB/s | 5666 MiB/s | 34.2 | 8.41 ms | 26.9 MiB | 15.52x |
533
- | zrecord-zstdict | 28 MiB/s | 5681 MiB/s | 34.3 | 8.55 ms | 17.7 MiB | 23.58x |
534
- | zrecord-raw | 1460 MiB/s | 6806 MiB/s | 41.1 | 7.58 ms | 417.2 MiB | 1.00x |
535
- | hdf5-raw | 1610 MiB/s | 1108 MiB/s | 6.7 | 45.80 ms | 418.0 MiB | 1.00x |
536
- | hdf5-gzip | 229 MiB/s | 114 MiB/s | 0.7 | 415.95 ms | 29.9 MiB | 13.94x |
537
- | lmdb-raw | 4126 MiB/s | 6837 MiB/s | 41.2 | 8.14 ms | 422.2 MiB | 0.99x |
538
- | arrow-ipc-raw | 3213 MiB/s | 4251 MiB/s | 25.6 | 12.90 ms | 417.2 MiB | 1.00x |
539
- | arrow-ipc-zstd | 706 MiB/s | 177 MiB/s | 1.1 | 266.37 ms | 25.9 MiB | 16.10x |
540
- | parquet-raw | 2219 MiB/s | 505 MiB/s | 3.0 | 92.79 ms | 417.2 MiB | 1.00x |
541
- | parquet-zstd | 728 MiB/s | 158 MiB/s | 1.0 | 296.60 ms | 25.9 MiB | 16.10x |
542
- | arrayrecord-raw | 1650 MiB/s | 2571 MiB/s | 15.5 | 18.34 ms | 417.6 MiB | 1.00x |
543
- | arrayrecord-zstd | 941 MiB/s | 1420 MiB/s | 8.6 | 37.14 ms | 27.2 MiB | 15.31x |
544
- | tiledb-raw | 443 MiB/s | 49 MiB/s | 0.3 | 920.09 ms | 417.2 MiB | 1.00x |
545
- | tiledb-zstd | 727 MiB/s | 124 MiB/s | 0.7 | 375.98 ms | 27.0 MiB | 15.46x |
546
-
547
- Zrecord-raw and LMDB are effectively tied in this pass; Arrow IPC is
548
- the strongest raw typed-file alternative. Zrecord-zstd delivers 5.5 GiB/s of
554
+ | zrecord-zstd | 2502 MiB/s | 10071 MiB/s | 60.5 | 5.06 ms | 26.9 MiB | 15.52x |
555
+ | zrecord-zstdict | 36 MiB/s | 10930 MiB/s | 65.6 | 4.62 ms | 17.7 MiB | 23.58x |
556
+ | zrecord-raw | 1661 MiB/s | 13056 MiB/s | 78.4 | 3.84 ms | 417.2 MiB | 1.00x |
557
+ | hdf5-raw | 1578 MiB/s | 1359 MiB/s | 8.2 | 37.26 ms | 418.0 MiB | 1.00x |
558
+ | hdf5-gzip | 243 MiB/s | 130 MiB/s | 0.8 | 356.26 ms | 29.9 MiB | 13.94x |
559
+ | lmdb-raw | 1790 MiB/s | 7412 MiB/s | 44.5 | 7.48 ms | 422.2 MiB | 0.99x |
560
+ | arrow-ipc-raw | 1572 MiB/s | 4070 MiB/s | 24.4 | 14.56 ms | 417.2 MiB | 1.00x |
561
+ | arrow-ipc-zstd | 620 MiB/s | 208 MiB/s | 1.2 | 232.74 ms | 25.9 MiB | 16.10x |
562
+ | parquet-raw | 889 MiB/s | 481 MiB/s | 2.9 | 93.87 ms | 417.2 MiB | 1.00x |
563
+ | parquet-zstd | 498 MiB/s | 184 MiB/s | 1.1 | 251.14 ms | 25.9 MiB | 16.10x |
564
+ | arrayrecord-raw | 1644 MiB/s | 2642 MiB/s | 15.9 | 18.22 ms | 417.6 MiB | 1.00x |
565
+ | arrayrecord-zstd | 829 MiB/s | 1556 MiB/s | 9.3 | 37.43 ms | 27.2 MiB | 15.31x |
566
+ | tiledb-raw | 411 MiB/s | 55 MiB/s | 0.3 | 806.95 ms | 417.2 MiB | 1.00x |
567
+ | tiledb-zstd | 713 MiB/s | 151 MiB/s | 0.9 | 307.58 ms | 27.0 MiB | 15.46x |
568
+
569
+ Zrecord-raw is 1.8x LMDB and 3.2x Arrow IPC in logical gather. Zrecord-zstd
570
+ delivers 9.8 GiB/s of
549
571
  logical payload while reducing the corpus to 26.9 MiB. HDF5, Arrow IPC, Parquet,
550
572
  ArrayRecord and TileDB
551
573
  show the same framework/codec tradeoffs in both tables; compressed batch, chunk
@@ -568,44 +590,45 @@ combines fragments shorter than 16 tokens, and splits records at 512 tokens into
568
590
 
569
591
  The Dense workload ignores text boundaries and packs the stream into 200,000
570
592
  fixed `int32[512]` records: 2 KiB per record and 390.6 MiB logical payload.
571
- Each of 100 random batches gathers 256 records.
593
+ Each IID batch gathers 256 records; fresh draws continue until timed gathers
594
+ accumulate at least two seconds.
572
595
 
573
596
  | backend | logical write | logical gather | krecords/s | p95 | disk | ratio |
574
597
  |---|---:|---:|---:|---:|---:|---:|
575
- | zrecord-zstd | 346 MiB/s | 1681 MiB/s | 860.9 | 0.38 ms | 193.2 MiB | 2.02x |
576
- | zrecord-zstdict | 41 MiB/s | 1892 MiB/s | 968.8 | 0.33 ms | 153.8 MiB | 2.54x |
577
- | zrecord-raw | 2477 MiB/s | 6261 MiB/s | 3205.5 | 0.11 ms | 393.7 MiB | 0.99x |
578
- | npy-mmap-raw | 2024 MiB/s | 7897 MiB/s | 4043.3 | 0.08 ms | 390.6 MiB | 1.00x |
579
- | lmdb-raw | 1022 MiB/s | 1066 MiB/s | 545.9 | 0.53 ms | 786.3 MiB | 0.50x |
580
- | arrow-ipc-raw | 2744 MiB/s | 207 MiB/s | 105.8 | 2.60 ms | 390.8 MiB | 1.00x |
581
- | arrayrecord-raw | 905 MiB/s | 149 MiB/s | 76.4 | 6.33 ms | 401.4 MiB | 0.97x |
582
- | arrayrecord-zstd | 114 MiB/s | 160 MiB/s | 82.0 | 4.83 ms | 201.3 MiB | 1.94x |
598
+ | zrecord-zstd | 456 MiB/s | 2047 MiB/s | 1047.9 | 0.32 ms | 193.2 MiB | 2.02x |
599
+ | zrecord-zstdict | 54 MiB/s | 2393 MiB/s | 1225.3 | 0.28 ms | 153.8 MiB | 2.54x |
600
+ | zrecord-raw | 2550 MiB/s | 7688 MiB/s | 3936.3 | 0.08 ms | 393.7 MiB | 0.99x |
601
+ | npy-mmap-raw | 2139 MiB/s | 10709 MiB/s | 5482.9 | 0.07 ms | 390.6 MiB | 1.00x |
602
+ | lmdb-raw | 721 MiB/s | 1166 MiB/s | 597.0 | 0.55 ms | 786.3 MiB | 0.50x |
603
+ | arrow-ipc-raw | 2335 MiB/s | 231 MiB/s | 118.2 | 2.53 ms | 390.8 MiB | 1.00x |
604
+ | arrayrecord-raw | 847 MiB/s | 162 MiB/s | 83.0 | 4.98 ms | 401.4 MiB | 0.97x |
605
+ | arrayrecord-zstd | 124 MiB/s | 133 MiB/s | 68.3 | 4.68 ms | 201.3 MiB | 1.94x |
583
606
 
584
607
  The contiguous NumPy baseline is strongest when the whole corpus is one fixed
585
- typed matrix. Zrecord-raw reaches 3.21 Mrecords/s while retaining independent
608
+ typed matrix. Zrecord-raw reaches 3.94 Mrecords/s while retaining independent
586
609
  record semantics; the per-record zstd codecs halve disk and still return
587
- 0.86–0.97 Mrecords/s. LMDB's B-tree/page overhead is visible in both throughput
610
+ 1.05–1.23 Mrecords/s. LMDB's B-tree/page overhead is visible in both throughput
588
611
  and disk.
589
612
 
590
613
  #### Variable Token Sequences
591
614
 
592
615
  The Ragged workload keeps 200,000 real text records of 16..512 tokens: p10 28,
593
616
  median 129, mean 138.8, p90 254, totaling 105.9 MiB. It uses the same 256-record,
594
- 100-batch random plan and every backend must return ordered `list[np.ndarray]`
617
+ random plan and every backend must return ordered `list[np.ndarray]`
595
618
  with exact `int32` values and original one-dimensional shapes.
596
619
 
597
620
  | backend | logical write | logical gather | krecords/s | p95 | disk | ratio |
598
621
  |---|---:|---:|---:|---:|---:|---:|
599
- | zrecord-zstd | 105 MiB/s | 225 MiB/s | 426.5 | 0.76 ms | 67.8 MiB | 1.56x |
600
- | zrecord-zstdict | 37 MiB/s | 277 MiB/s | 526.3 | 0.63 ms | 49.4 MiB | 2.14x |
601
- | zrecord-raw | 457 MiB/s | 325 MiB/s | 616.1 | 0.46 ms | 112.0 MiB | 0.95x |
602
- | lmdb-raw | 887 MiB/s | 106 MiB/s | 200.7 | 1.34 ms | 153.0 MiB | 0.69x |
603
- | arrow-ipc-raw | 2230 MiB/s | 39 MiB/s | 74.1 | 4.48 ms | 109.9 MiB | 0.96x |
604
- | arrayrecord-raw | 399 MiB/s | 24 MiB/s | 45.7 | 8.34 ms | 118.8 MiB | 0.89x |
605
- | arrayrecord-zstd | 59 MiB/s | 27 MiB/s | 51.1 | 7.79 ms | 76.2 MiB | 1.39x |
622
+ | zrecord-zstd | 114 MiB/s | 291 MiB/s | 550.1 | 0.59 ms | 67.5 MiB | 1.57x |
623
+ | zrecord-zstdict | 47 MiB/s | 325 MiB/s | 614.3 | 0.53 ms | 49.2 MiB | 2.15x |
624
+ | zrecord-raw | 571 MiB/s | 352 MiB/s | 665.5 | 0.50 ms | 110.7 MiB | 0.96x |
625
+ | lmdb-raw | 341 MiB/s | 116 MiB/s | 218.9 | 1.32 ms | 153.0 MiB | 0.69x |
626
+ | arrow-ipc-raw | 567 MiB/s | 44 MiB/s | 84.1 | 3.66 ms | 109.9 MiB | 0.96x |
627
+ | arrayrecord-raw | 290 MiB/s | 28 MiB/s | 53.5 | 7.22 ms | 118.8 MiB | 0.89x |
628
+ | arrayrecord-zstd | 61 MiB/s | 34 MiB/s | 64.1 | 5.12 ms | 76.2 MiB | 1.39x |
606
629
 
607
630
  Here the record contract, not bulk byte bandwidth, is the useful scale.
608
- Zrecord's three codecs return 427–616 krecords/s with less than 0.8 ms p95;
631
+ Zrecord's three codecs return 550–666 krecords/s with less than 0.6 ms p95;
609
632
  the dictionary gives the best disk ratio and is slightly ahead of plain zstd in
610
633
  this pass.
611
634
 
@@ -748,13 +771,13 @@ explicit solutions rather than silently changing the default transform.
748
771
  **What these numbers do not claim.** Everything runs with a warm page cache: this
749
772
  measures the access path, not cold storage or disk. `disk` is allocated
750
773
  blocks, and zrecord files grow to their written frontier. `logical write`
751
- measures append/write calls after writer setup and before close/finalization; it
774
+ measures source adaptation through logical finalization after writer setup; it
752
775
  does not benchmark durability, stronger transactional guarantees, or reusable
753
776
  dictionary training. Arrow IPC and Parquet use 256-record groups, so random
754
777
  batches pay their real group-level read amplification. Ragged Zrecord arrays are
755
778
  views into one batch allocation, while most byte-store adapters return
756
- independent copies. This is a single
757
- qualitative pass —
779
+ independent copies. This is a single benchmark run, while each Store read path
780
+ accumulates at least two timed seconds —
758
781
  the µs-scale sampler timings and the loader `batches/s` fluctuate with box load
759
782
  (on these 12 cores the compressed loader trails the raw one by about 25%), so
760
783
  treat the absolute numbers as ballpark and the cross-backend margins as the
@@ -1,6 +1,6 @@
1
1
  .{
2
2
  .name = .loaderx,
3
- .version = "2.0.0",
3
+ .version = "2.0.1",
4
4
  .fingerprint = 0x350591788e45d1d3,
5
5
  .dependencies = .{},
6
6
  .paths = .{
@@ -15,4 +15,4 @@ boundaries stay visible:
15
15
  step-based sampler + prefetch pipeline over them.
16
16
  """
17
17
 
18
- __version__ = "2.0.0"
18
+ __version__ = "2.0.1"
@@ -41,12 +41,9 @@ from ._store import MAX_RECORD_LEN, MAX_SCHEMA_LEN, ZrecordError, _Store
41
41
 
42
42
  __all__ = ["Dense", "Ragged", "ZrecordError", "DICT_TIERS"]
43
43
 
44
- _RANK_UNPACKER = struct.Struct("<Q")
45
44
  _SHAPE_UNPACKERS = tuple(
46
45
  struct.Struct("<" + "Q" * rank) for rank in range(9)
47
46
  )
48
- # The on-disk rank is u64; this is the active NumPy runtime's ndarray limit,
49
- # not a Zrecord format limit.
50
47
  _NUMPY_MAX_NDIMS = 64 if int(np.__version__.split(".", 1)[0]) >= 2 else 32
51
48
  _MAX_UINT64 = np.iinfo(np.uint64).max
52
49
  _MAX_INTP = np.iinfo(np.intp).max
@@ -320,7 +317,7 @@ class Ragged(_Base):
320
317
  """Variable-length records over a zrecord container.
321
318
 
322
319
  ``dtype`` is unified across the whole store (the static schema); each record
323
- keeps its own shape in an inline little-endian u64 prefix. Reads
320
+ keeps its own shape as a u8 rank followed by little-endian u64 dimensions. Reads
324
321
  restore every record to exactly the shape it was written with, so a batch
325
322
  is a list of the original arrays — ``ds[i]`` is one record.
326
323
 
@@ -372,7 +369,7 @@ class Ragged(_Base):
372
369
  rank = len(shape)
373
370
  packer = packers.get(rank)
374
371
  if packer is None:
375
- packer = packers[rank] = struct.Struct("<" + "Q" * (rank + 1))
372
+ packer = packers[rank] = struct.Struct("<B" + "Q" * rank)
376
373
  encoded_bytes = packer.size + payload_bytes
377
374
  if encoded_bytes > MAX_RECORD_LEN:
378
375
  raise ValueError(
@@ -408,17 +405,17 @@ class Ragged(_Base):
408
405
  for offset, length in zip(offsets, lengths):
409
406
  start = int(offset)
410
407
  length = int(length)
411
- if length <= _RANK_UNPACKER.size:
408
+ if length <= 1:
412
409
  raise ValueError("corrupt ragged record shape header")
413
- rank = _RANK_UNPACKER.unpack_from(buf, start)[0]
414
- header_bytes = 8 * (rank + 1)
410
+ rank = int(buf[start])
411
+ header_bytes = 1 + 8 * rank
415
412
  if rank > _NUMPY_MAX_NDIMS or header_bytes >= length:
416
413
  raise ValueError("corrupt ragged record shape header")
417
414
  if rank < len(_SHAPE_UNPACKERS):
418
415
  unpacker = _SHAPE_UNPACKERS[rank]
419
416
  else:
420
417
  unpacker = struct.Struct("<" + "Q" * rank)
421
- shape = unpacker.unpack_from(buf, start + 8)
418
+ shape = unpacker.unpack_from(buf, start + 1)
422
419
  payload_bytes = length - header_bytes
423
420
  itemsize = self.dtype.itemsize
424
421
  if not itemsize or payload_bytes % itemsize: