loaderx 2.0.1__tar.gz → 2.0.7__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {loaderx-2.0.1/loaderx.egg-info → loaderx-2.0.7}/PKG-INFO +139 -102
- {loaderx-2.0.1 → loaderx-2.0.7}/README.md +137 -100
- {loaderx-2.0.1 → loaderx-2.0.7}/build.zig.zon +1 -1
- {loaderx-2.0.1 → loaderx-2.0.7}/loaderx/__init__.py +5 -4
- {loaderx-2.0.1 → loaderx-2.0.7}/loaderx/_store.py +9 -7
- {loaderx-2.0.1 → loaderx-2.0.7}/loaderx/dataloader.py +26 -21
- {loaderx-2.0.1 → loaderx-2.0.7}/loaderx/utils.py +4 -3
- {loaderx-2.0.1 → loaderx-2.0.7}/loaderx/zrecord.py +40 -27
- {loaderx-2.0.1 → loaderx-2.0.7}/loaderx/zsampler.py +21 -8
- {loaderx-2.0.1 → loaderx-2.0.7/loaderx.egg-info}/PKG-INFO +139 -102
- {loaderx-2.0.1 → loaderx-2.0.7}/pyproject.toml +1 -1
- {loaderx-2.0.1 → loaderx-2.0.7}/scripts/test_loaderx.py +89 -139
- {loaderx-2.0.1 → loaderx-2.0.7}/src/record/engine.zig +340 -263
- loaderx-2.0.7/src/record/executor.zig +171 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/src/record/storage.zig +58 -29
- {loaderx-2.0.1 → loaderx-2.0.7}/src/store.zig +26 -15
- {loaderx-2.0.1 → loaderx-2.0.7}/src/zsampler.zig +28 -21
- loaderx-2.0.7/src/zstd/c.zig +35 -0
- loaderx-2.0.1/src/record/executor.zig +0 -60
- loaderx-2.0.1/src/zstd/c.zig +0 -735
- {loaderx-2.0.1 → loaderx-2.0.7}/LICENSE +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/MANIFEST.in +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/build.zig +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/loaderx/_lib.py +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/loaderx.egg-info/SOURCES.txt +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/loaderx.egg-info/dependency_links.txt +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/loaderx.egg-info/requires.txt +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/loaderx.egg-info/top_level.txt +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/scripts/_bench_common.py +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/scripts/bench.py +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/scripts/bench_dense.py +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/scripts/bench_ragged.py +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/scripts/build_wheels.py +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/scripts/prepare_tokens.py +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/scripts/requirements-bench.txt +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/setup.cfg +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/setup.py +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/COPYING +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/LICENSE +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/common/allocations.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/common/bits.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/common/bitstream.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/common/compiler.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/common/cpu.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/common/debug.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/common/debug.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/common/entropy_common.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/common/error_private.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/common/error_private.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/common/fse.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/common/fse_decompress.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/common/huf.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/common/mem.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/common/pool.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/common/pool.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/common/portability_macros.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/common/threading.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/common/threading.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/common/xxhash.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/common/xxhash.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/common/zstd_common.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/common/zstd_deps.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/common/zstd_internal.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/common/zstd_trace.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/clevels.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/fse_compress.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/hist.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/hist.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/huf_compress.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/zstd_compress.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/zstd_compress_internal.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/zstd_compress_literals.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/zstd_compress_literals.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/zstd_compress_sequences.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/zstd_compress_sequences.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/zstd_compress_superblock.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/zstd_compress_superblock.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/zstd_cwksp.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/zstd_double_fast.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/zstd_double_fast.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/zstd_fast.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/zstd_fast.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/zstd_lazy.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/zstd_lazy.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/zstd_ldm.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/zstd_ldm.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/zstd_ldm_geartab.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/zstd_opt.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/zstd_opt.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/zstdmt_compress.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/compress/zstdmt_compress.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/decompress/huf_decompress.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/decompress/zstd_ddict.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/decompress/zstd_ddict.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/decompress/zstd_decompress.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/decompress/zstd_decompress_block.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/decompress/zstd_decompress_block.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/decompress/zstd_decompress_internal.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/dictBuilder/cover.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/dictBuilder/cover.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/dictBuilder/divsufsort.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/dictBuilder/divsufsort.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/dictBuilder/fastcover.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/dictBuilder/zdict.c +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/zdict.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/zstd.h +0 -0
- {loaderx-2.0.1 → loaderx-2.0.7}/vendor/zstd/lib/zstd_errors.h +0 -0
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: loaderx
|
|
3
|
-
Version: 2.0.
|
|
4
|
-
Summary: Rebuildable high-performance record containers
|
|
3
|
+
Version: 2.0.7
|
|
4
|
+
Summary: Rebuildable high-performance ordered record containers
|
|
5
5
|
Author-email: Ben0i0d <ben0i0d@foxmail.com>
|
|
6
6
|
License-Expression: MIT
|
|
7
7
|
Project-URL: Homepage, https://codeberg.org/eoelab/loaderx
|
|
@@ -24,10 +24,11 @@ Requires-Dist: msgpack
|
|
|
24
24
|
Dynamic: license-file
|
|
25
25
|
|
|
26
26
|
# Loaderx
|
|
27
|
-
Zrecord is a rebuildable
|
|
28
|
-
|
|
29
|
-
publishes
|
|
30
|
-
|
|
27
|
+
Zrecord is a rebuildable, typed, ordered record sequence built from authoritative
|
|
28
|
+
source data and scripts. A creator consumes records in append order, ``close``
|
|
29
|
+
publishes one immutable container, and readers support both sequential slicing
|
|
30
|
+
and indexed gather without changing record identity. To change content or order,
|
|
31
|
+
rebuild it at a new path.
|
|
31
32
|
|
|
32
33
|
Zrecord is the typed on-disk container; Loaderx is the sampler and prefetch
|
|
33
34
|
loader that consumes Zrecord streams. They currently ship together while both
|
|
@@ -54,6 +55,9 @@ loaderx is built around several core principles:
|
|
|
54
55
|
dense stream stacks into one array per batch; a ragged one comes back as a
|
|
55
56
|
list. Neither is padded, and equal length is never treated as a special case
|
|
56
57
|
of variable length.
|
|
58
|
+
6. **Logical IDs are stable sequence positions.** Native chunks may complete in
|
|
59
|
+
any physical order, but append input order defines ``0..N-1`` and a published
|
|
60
|
+
container never deletes, compacts, updates, or renumbers those records.
|
|
57
61
|
|
|
58
62
|
## 设计文档
|
|
59
63
|
|
|
@@ -149,9 +153,10 @@ ds[0, 5, 2] # (3, 2, 4) — shape from the persisted sc
|
|
|
149
153
|
ds.close()
|
|
150
154
|
```
|
|
151
155
|
|
|
152
|
-
A store is
|
|
153
|
-
|
|
154
|
-
in ``0..len(ds)-1``. The ragged example below reads
|
|
156
|
+
A store is an ordered sequence of records, not an ndarray, so ``ds[0, 5, 2]``
|
|
157
|
+
selects sequence positions 0, 5 and 2 — never ``ds[0][5][2]``. A scalar selects
|
|
158
|
+
one record; indices must be in ``0..len(ds)-1``. The ragged example below reads
|
|
159
|
+
the same way.
|
|
155
160
|
|
|
156
161
|
``Ragged`` is the **ragged** contract for variable-length records. It is a
|
|
157
162
|
separate contract: :class:`Ragged` hands back a list of arrays, so a
|
|
@@ -179,6 +184,34 @@ for i, r in enumerate(records):
|
|
|
179
184
|
padded[i, :len(r)] = r # (B, max_len) — your policy, your loop
|
|
180
185
|
```
|
|
181
186
|
|
|
187
|
+
### Ordered sequences and build streams
|
|
188
|
+
|
|
189
|
+
Logical ID is the stable sequence position. One ``append`` preserves every
|
|
190
|
+
record in its input order; successive calls from one producer extend that
|
|
191
|
+
sequence. Native lanes may reserve and write payload chunks in a different
|
|
192
|
+
physical completion order, but each ``RecordLoc`` is installed in its original
|
|
193
|
+
logical slot, so physical scheduling never changes ``ds[i]``. After ``close``,
|
|
194
|
+
the sequence is immutable: there is no delete, compact, update, or reopen-append
|
|
195
|
+
operation that can renumber it.
|
|
196
|
+
|
|
197
|
+
This makes a creator a finite **build stream** and its published result an
|
|
198
|
+
immutable sequence. It is not a live log: readers do not tail a writer, and an
|
|
199
|
+
unbounded producer must choose a finite publication boundary. A large sequence
|
|
200
|
+
can be consumed in bounded ordered batches with ordinary slices:
|
|
201
|
+
|
|
202
|
+
```python
|
|
203
|
+
with Dense.open("events") as events:
|
|
204
|
+
for start in range(0, len(events), 1024):
|
|
205
|
+
batch = events[start:start + 1024]
|
|
206
|
+
consume(batch)
|
|
207
|
+
```
|
|
208
|
+
|
|
209
|
+
Ragged creators may consume a one-pass iterable, while Dense creators take
|
|
210
|
+
explicit ndarray batches. Concurrent append calls are serialized by the native
|
|
211
|
+
writer lock, but their batch order is the lock-acquisition order rather than
|
|
212
|
+
Python call-start order; sequence builders that require an external temporal
|
|
213
|
+
order should use one producer or order batches before append.
|
|
214
|
+
|
|
182
215
|
A ``DataLoader`` dynamically composes a dict of dense and ragged streams.
|
|
183
216
|
Collation is the ``transform`` — a batch dict in, a batch dict out:
|
|
184
217
|
|
|
@@ -225,8 +258,9 @@ warmup belongs outside build or loader benchmark timing.
|
|
|
225
258
|
|
|
226
259
|
### Creating containers
|
|
227
260
|
|
|
228
|
-
``Dense.create`` and ``Ragged.create`` return append-only
|
|
229
|
-
explicit — one
|
|
261
|
+
``Dense.create`` and ``Ragged.create`` return append-only ordered-sequence
|
|
262
|
+
builders. ``append`` is explicit — one input batch extends the logical sequence
|
|
263
|
+
without exposing native physical completion order.
|
|
230
264
|
Dense append is synchronous and borrows an already-contiguous ndarray without a
|
|
231
265
|
snapshot copy. Ragged append consumes its iterable once into one owned packed
|
|
232
266
|
buffer, then completes the native append before returning. Nothing is inferred.
|
|
@@ -541,24 +575,24 @@ Fixed-resolution vision records — 147 KiB per record, 36.8 MiB per batch:
|
|
|
541
575
|
|
|
542
576
|
| backend | logical write | logical gather | krecords/s | p95 | disk | ratio |
|
|
543
577
|
|---|---:|---:|---:|---:|---:|---:|
|
|
544
|
-
| zrecord-zstd |
|
|
545
|
-
| zrecord-zstdict |
|
|
546
|
-
| zrecord-raw |
|
|
547
|
-
| npy-mmap-raw |
|
|
548
|
-
| hdf5-raw |
|
|
549
|
-
| hdf5-gzip |
|
|
550
|
-
| lmdb-raw |
|
|
551
|
-
| arrow-ipc-raw |
|
|
552
|
-
| arrow-ipc-zstd |
|
|
553
|
-
| parquet-raw |
|
|
554
|
-
| parquet-zstd |
|
|
555
|
-
| arrayrecord-raw |
|
|
556
|
-
| arrayrecord-zstd |
|
|
557
|
-
| tiledb-raw |
|
|
558
|
-
| tiledb-zstd |
|
|
559
|
-
|
|
560
|
-
At 147 KiB per record, Zrecord-raw reaches
|
|
561
|
-
plain zstd gathers at
|
|
578
|
+
| zrecord-zstd | 7998 MiB/s | 10937 MiB/s | 76.2 | 4.37 ms | 24.4 MiB | 14.69x |
|
|
579
|
+
| zrecord-zstdict | 35 MiB/s | 12173 MiB/s | 84.8 | 4.06 ms | 15.8 MiB | 22.74x |
|
|
580
|
+
| zrecord-raw | 2523 MiB/s | 14329 MiB/s | 99.8 | 3.12 ms | 358.9 MiB | 1.00x |
|
|
581
|
+
| npy-mmap-raw | 2004 MiB/s | 4749 MiB/s | 33.1 | 11.08 ms | 358.9 MiB | 1.00x |
|
|
582
|
+
| hdf5-raw | 2350 MiB/s | 1836 MiB/s | 12.8 | 29.68 ms | 359.0 MiB | 1.00x |
|
|
583
|
+
| hdf5-gzip | 277 MiB/s | 657 MiB/s | 4.6 | 62.53 ms | 26.1 MiB | 13.73x |
|
|
584
|
+
| lmdb-raw | 1546 MiB/s | 4034 MiB/s | 28.1 | 11.38 ms | 361.4 MiB | 0.99x |
|
|
585
|
+
| arrow-ipc-raw | 1766 MiB/s | 3413 MiB/s | 23.8 | 15.10 ms | 358.9 MiB | 1.00x |
|
|
586
|
+
| arrow-ipc-zstd | 656 MiB/s | 174 MiB/s | 1.2 | 239.23 ms | 22.6 MiB | 15.86x |
|
|
587
|
+
| parquet-raw | 1174 MiB/s | 435 MiB/s | 3.0 | 95.45 ms | 358.9 MiB | 1.00x |
|
|
588
|
+
| parquet-zstd | 592 MiB/s | 159 MiB/s | 1.1 | 251.37 ms | 22.6 MiB | 15.86x |
|
|
589
|
+
| arrayrecord-raw | 1704 MiB/s | 2066 MiB/s | 14.4 | 20.36 ms | 359.2 MiB | 1.00x |
|
|
590
|
+
| arrayrecord-zstd | 804 MiB/s | 1294 MiB/s | 9.0 | 36.89 ms | 25.1 MiB | 14.32x |
|
|
591
|
+
| tiledb-raw | 743 MiB/s | 630 MiB/s | 4.4 | 66.01 ms | 359.0 MiB | 1.00x |
|
|
592
|
+
| tiledb-zstd | 1498 MiB/s | 1503 MiB/s | 10.5 | 27.82 ms | 26.9 MiB | 13.36x |
|
|
593
|
+
|
|
594
|
+
At 147 KiB per record, Zrecord-raw reaches 14.0 GiB/s and is 3.0x npy-mmap-raw;
|
|
595
|
+
plain zstd gathers at 10.7 GiB/s while reducing the corpus 14.69x. LMDB and Arrow
|
|
562
596
|
IPC are competitive raw record
|
|
563
597
|
stores, while codecs tied to whole IPC batches or Parquet row groups pay read
|
|
564
598
|
amplification on random gathers. Dense demonstrates that
|
|
@@ -576,23 +610,23 @@ list or a one-dimensional variable-length abstraction is not enough.
|
|
|
576
610
|
|
|
577
611
|
| backend | logical write | logical gather | krecords/s | p95 | disk | ratio |
|
|
578
612
|
|---|---:|---:|---:|---:|---:|---:|
|
|
579
|
-
| zrecord-zstd |
|
|
580
|
-
| zrecord-zstdict |
|
|
581
|
-
| zrecord-raw |
|
|
582
|
-
| hdf5-raw |
|
|
583
|
-
| hdf5-gzip |
|
|
584
|
-
| lmdb-raw |
|
|
585
|
-
| arrow-ipc-raw |
|
|
586
|
-
| arrow-ipc-zstd |
|
|
587
|
-
| parquet-raw |
|
|
588
|
-
| parquet-zstd |
|
|
589
|
-
| arrayrecord-raw |
|
|
590
|
-
| arrayrecord-zstd |
|
|
591
|
-
| tiledb-raw |
|
|
592
|
-
| tiledb-zstd |
|
|
613
|
+
| zrecord-zstd | 2407 MiB/s | 9927 MiB/s | 59.6 | 5.44 ms | 26.9 MiB | 15.52x |
|
|
614
|
+
| zrecord-zstdict | 37 MiB/s | 10814 MiB/s | 65.0 | 4.86 ms | 17.7 MiB | 23.58x |
|
|
615
|
+
| zrecord-raw | 1591 MiB/s | 13032 MiB/s | 78.2 | 3.87 ms | 417.2 MiB | 1.00x |
|
|
616
|
+
| hdf5-raw | 1451 MiB/s | 1059 MiB/s | 6.4 | 44.91 ms | 418.0 MiB | 1.00x |
|
|
617
|
+
| hdf5-gzip | 250 MiB/s | 126 MiB/s | 0.7 | 359.23 ms | 29.9 MiB | 13.94x |
|
|
618
|
+
| lmdb-raw | 1717 MiB/s | 7269 MiB/s | 43.6 | 7.93 ms | 422.2 MiB | 0.99x |
|
|
619
|
+
| arrow-ipc-raw | 1363 MiB/s | 4015 MiB/s | 24.1 | 14.97 ms | 417.2 MiB | 1.00x |
|
|
620
|
+
| arrow-ipc-zstd | 570 MiB/s | 198 MiB/s | 1.2 | 238.70 ms | 25.9 MiB | 16.10x |
|
|
621
|
+
| parquet-raw | 927 MiB/s | 501 MiB/s | 3.0 | 94.62 ms | 417.2 MiB | 1.00x |
|
|
622
|
+
| parquet-zstd | 507 MiB/s | 173 MiB/s | 1.0 | 268.56 ms | 25.9 MiB | 16.10x |
|
|
623
|
+
| arrayrecord-raw | 1650 MiB/s | 2661 MiB/s | 16.0 | 18.19 ms | 417.6 MiB | 1.00x |
|
|
624
|
+
| arrayrecord-zstd | 758 MiB/s | 1499 MiB/s | 9.0 | 39.16 ms | 27.2 MiB | 15.31x |
|
|
625
|
+
| tiledb-raw | 392 MiB/s | 48 MiB/s | 0.3 | 941.09 ms | 417.2 MiB | 1.00x |
|
|
626
|
+
| tiledb-zstd | 675 MiB/s | 138 MiB/s | 0.8 | 328.05 ms | 27.0 MiB | 15.46x |
|
|
593
627
|
|
|
594
628
|
Zrecord-raw is 1.8x LMDB and 3.2x Arrow IPC in logical gather. Zrecord-zstd
|
|
595
|
-
delivers 9.
|
|
629
|
+
delivers 9.7 GiB/s of
|
|
596
630
|
logical payload while reducing the corpus to 26.9 MiB. HDF5, Arrow IPC, Parquet,
|
|
597
631
|
ArrayRecord and TileDB
|
|
598
632
|
show the same framework/codec tradeoffs in both tables; compressed batch, chunk
|
|
@@ -620,19 +654,19 @@ accumulate at least two seconds.
|
|
|
620
654
|
|
|
621
655
|
| backend | logical write | logical gather | krecords/s | p95 | disk | ratio |
|
|
622
656
|
|---|---:|---:|---:|---:|---:|---:|
|
|
623
|
-
| zrecord-zstd |
|
|
624
|
-
| zrecord-zstdict |
|
|
625
|
-
| zrecord-raw |
|
|
626
|
-
| npy-mmap-raw |
|
|
627
|
-
| lmdb-raw |
|
|
628
|
-
| arrow-ipc-raw |
|
|
629
|
-
| arrayrecord-raw |
|
|
630
|
-
| arrayrecord-zstd |
|
|
657
|
+
| zrecord-zstd | 1019 MiB/s | 1959 MiB/s | 1002.8 | 0.35 ms | 193.2 MiB | 2.02x |
|
|
658
|
+
| zrecord-zstdict | 52 MiB/s | 2238 MiB/s | 1145.9 | 0.31 ms | 153.8 MiB | 2.54x |
|
|
659
|
+
| zrecord-raw | 2101 MiB/s | 7441 MiB/s | 3810.0 | 0.09 ms | 393.7 MiB | 0.99x |
|
|
660
|
+
| npy-mmap-raw | 1778 MiB/s | 11454 MiB/s | 5864.4 | 0.06 ms | 390.6 MiB | 1.00x |
|
|
661
|
+
| lmdb-raw | 639 MiB/s | 1104 MiB/s | 565.0 | 0.71 ms | 786.3 MiB | 0.50x |
|
|
662
|
+
| arrow-ipc-raw | 2068 MiB/s | 225 MiB/s | 115.0 | 2.81 ms | 390.8 MiB | 1.00x |
|
|
663
|
+
| arrayrecord-raw | 807 MiB/s | 150 MiB/s | 76.7 | 4.99 ms | 401.4 MiB | 0.97x |
|
|
664
|
+
| arrayrecord-zstd | 127 MiB/s | 135 MiB/s | 69.4 | 4.45 ms | 201.3 MiB | 1.94x |
|
|
631
665
|
|
|
632
666
|
The contiguous NumPy baseline is strongest when the whole corpus is one fixed
|
|
633
|
-
typed matrix. Zrecord-raw reaches 3.
|
|
667
|
+
typed matrix. Zrecord-raw reaches 3.81 Mrecords/s while retaining independent
|
|
634
668
|
record semantics; the per-record zstd codecs halve disk and still return
|
|
635
|
-
1.
|
|
669
|
+
1.00–1.15 Mrecords/s. LMDB's B-tree/page overhead is visible in both throughput
|
|
636
670
|
and disk.
|
|
637
671
|
|
|
638
672
|
#### Variable Token Sequences
|
|
@@ -644,16 +678,16 @@ with exact `int32` values and original one-dimensional shapes.
|
|
|
644
678
|
|
|
645
679
|
| backend | logical write | logical gather | krecords/s | p95 | disk | ratio |
|
|
646
680
|
|---|---:|---:|---:|---:|---:|---:|
|
|
647
|
-
| zrecord-zstd |
|
|
648
|
-
| zrecord-zstdict |
|
|
649
|
-
| zrecord-raw |
|
|
650
|
-
| lmdb-raw |
|
|
651
|
-
| arrow-ipc-raw |
|
|
652
|
-
| arrayrecord-raw |
|
|
653
|
-
| arrayrecord-zstd |
|
|
681
|
+
| zrecord-zstd | 236 MiB/s | 282 MiB/s | 533.7 | 0.66 ms | 67.5 MiB | 1.57x |
|
|
682
|
+
| zrecord-zstdict | 43 MiB/s | 305 MiB/s | 576.9 | 0.61 ms | 49.2 MiB | 2.15x |
|
|
683
|
+
| zrecord-raw | 486 MiB/s | 359 MiB/s | 679.0 | 0.52 ms | 110.7 MiB | 0.96x |
|
|
684
|
+
| lmdb-raw | 318 MiB/s | 113 MiB/s | 214.3 | 1.48 ms | 153.0 MiB | 0.69x |
|
|
685
|
+
| arrow-ipc-raw | 562 MiB/s | 44 MiB/s | 83.1 | 4.00 ms | 109.9 MiB | 0.96x |
|
|
686
|
+
| arrayrecord-raw | 288 MiB/s | 26 MiB/s | 49.3 | 7.39 ms | 118.8 MiB | 0.89x |
|
|
687
|
+
| arrayrecord-zstd | 62 MiB/s | 33 MiB/s | 61.5 | 5.58 ms | 76.2 MiB | 1.39x |
|
|
654
688
|
|
|
655
689
|
Here the record contract, not bulk byte bandwidth, is the useful scale.
|
|
656
|
-
Zrecord's three codecs return
|
|
690
|
+
Zrecord's three codecs return 534–679 krecords/s with 0.52–0.66 ms p95;
|
|
657
691
|
the dictionary gives the best disk ratio and is slightly ahead of plain zstd in
|
|
658
692
|
this pass.
|
|
659
693
|
|
|
@@ -661,16 +695,16 @@ this pass.
|
|
|
661
695
|
|
|
662
696
|
Index generation on its own, IID (with replacement), 1M index space, against
|
|
663
697
|
NumPy's modern API. The µs-scale figures fluctuate with box load; this pass shows
|
|
664
|
-
|
|
698
|
+
a 1.87–7.96x margin across batch sizes.
|
|
665
699
|
|
|
666
700
|
| sampler | batch | per batch | vs default_rng |
|
|
667
701
|
|---|---:|---:|---:|
|
|
668
|
-
| numpy default_rng | 256 |
|
|
669
|
-
| **zsampler** | 256 |
|
|
670
|
-
| numpy default_rng | 1024 |
|
|
671
|
-
| **zsampler** | 1024 |
|
|
672
|
-
| numpy default_rng | 8192 |
|
|
673
|
-
| **zsampler** | 8192 | 9.
|
|
702
|
+
| numpy default_rng | 256 | 5.3 µs | 1.00x |
|
|
703
|
+
| **zsampler** | 256 | 0.7 µs | **7.96x** |
|
|
704
|
+
| numpy default_rng | 1024 | 4.7 µs | 1.00x |
|
|
705
|
+
| **zsampler** | 1024 | 1.3 µs | **3.64x** |
|
|
706
|
+
| numpy default_rng | 8192 | 18.0 µs | 1.00x |
|
|
707
|
+
| **zsampler** | 8192 | 9.6 µs | **1.87x** |
|
|
674
708
|
|
|
675
709
|
### End-to-End DataLoader
|
|
676
710
|
|
|
@@ -700,19 +734,19 @@ and Grain reads ArrayRecord.
|
|
|
700
734
|
|
|
701
735
|
| loader | model | storage | batches/s | p95 | steady PSS | peak PSS | peak RSS |
|
|
702
736
|
|---|---|---|---:|---:|---:|---:|---:|
|
|
703
|
-
| **loaderx** | threads | zrecord-zstd |
|
|
704
|
-
| loaderx-raw | threads | zrecord-raw |
|
|
705
|
-
| torch | fork | npy-mmap-raw |
|
|
706
|
-
| torch-spawn | spawn | npy-mmap-raw |
|
|
707
|
-
| grain | processes | arrayrecord-zstd |
|
|
737
|
+
| **loaderx** | threads | zrecord-zstd | 167.1 | 14.63 ms | 986 MiB | 986 MiB | 989 MiB |
|
|
738
|
+
| loaderx-raw | threads | zrecord-raw | 215.3 | 13.77 ms | 987 MiB | 987 MiB | 991 MiB |
|
|
739
|
+
| torch | fork | npy-mmap-raw | 109.4 | 32.70 ms | 1783 MiB | 1889 MiB | 6396 MiB |
|
|
740
|
+
| torch-spawn | spawn | npy-mmap-raw | 111.6 | 30.69 ms | 2835 MiB | 2913 MiB | 4880 MiB |
|
|
741
|
+
| grain | processes | arrayrecord-zstd | 46.6 | 93.48 ms | 1847 MiB | 1946 MiB | 2065 MiB |
|
|
708
742
|
|
|
709
743
|
At 36.8 MiB per batch the per-batch gather dominates the tiny sampler cost, and
|
|
710
744
|
the transform threads overlap Python-side collation with the next gather. The
|
|
711
745
|
memory is the source, Zrecord container and bounded in-flight batches. loaderx prefetches in
|
|
712
746
|
threads inside one process, so workers share one interpreter, one NumPy runtime
|
|
713
747
|
and one set of gather buffers. With source geometry and entropy held constant,
|
|
714
|
-
raw is 1.
|
|
715
|
-
Torch spawn and 3.
|
|
748
|
+
raw is 1.29x compressed loaderx; compressed loaderx is 1.53x Torch fork, 1.50x
|
|
749
|
+
Torch spawn and 3.59x Grain, while raw is 1.97x, 1.93x and 4.62x faster.
|
|
716
750
|
Torch's aggregate RSS is high because
|
|
717
751
|
Linux fork mappings are counted repeatedly; it is not a total-memory ratio
|
|
718
752
|
against Zrecord's unaccounted page cache. The explicit `torch-spawn` row removes
|
|
@@ -721,19 +755,19 @@ not copy the full corpus into every worker; a Windows Dataset holding Python
|
|
|
721
755
|
lists or in-memory arrays would be a different, deliberately harsher workload.
|
|
722
756
|
|
|
723
757
|
The Torch-only worker sweep runs each count once. Worker 0 is an in-process
|
|
724
|
-
baseline (
|
|
758
|
+
baseline (154.5 and 163.5 batches/s with 1484/1486 MiB peak PSS in the two equivalent
|
|
725
759
|
rows), so the process-context comparison starts at one worker:
|
|
726
760
|
|
|
727
761
|
| workers | fork batches/s | fork peak PSS | fork peak RSS | spawn batches/s | spawn peak PSS | spawn peak RSS |
|
|
728
762
|
|---:|---:|---:|---:|---:|---:|---:|
|
|
729
|
-
| 1 | 42.5 |
|
|
730
|
-
| 2 |
|
|
731
|
-
| 4 |
|
|
732
|
-
| 8 |
|
|
733
|
-
|
|
734
|
-
Spawn peak PSS grows from
|
|
735
|
-
while fork grows from
|
|
736
|
-
workers spawn uses 2.
|
|
763
|
+
| 1 | 42.5 | 1647 MiB | 2483 MiB | 46.3 | 1890 MiB | 2128 MiB |
|
|
764
|
+
| 2 | 68.9 | 1725 MiB | 3796 MiB | 72.3 | 2241 MiB | 3066 MiB |
|
|
765
|
+
| 4 | 106.0 | 1849 MiB | 6385 MiB | 102.7 | 2878 MiB | 4859 MiB |
|
|
766
|
+
| 8 | 114.3 | 2021 MiB | 11334 MiB | 107.8 | 4195 MiB | 8402 MiB |
|
|
767
|
+
|
|
768
|
+
Spawn peak PSS grows from 1890 to 4195 MiB as workers rise from one to eight,
|
|
769
|
+
while fork grows from 1647 to 2021 MiB because it retains COW sharing. At eight
|
|
770
|
+
workers spawn uses 2.08x fork's peak PSS and throughput has already flattened.
|
|
737
771
|
This demonstrates no-fork memory pressure; it is not labeled OOM because this
|
|
738
772
|
31 GiB machine completed the run. Fork aggregate RSS grows faster because Linux
|
|
739
773
|
counts shared/COW mappings in every process, so RSS is diagnostic rather than
|
|
@@ -749,11 +783,11 @@ no separate no-GIL Store implementation; the Store table above applies to both.
|
|
|
749
783
|
|
|
750
784
|
| loader | GIL Python | GIL + Numba nogil | free-threaded Python | Numba gain | free-threaded gain |
|
|
751
785
|
|---|---:|---:|---:|---:|---:|
|
|
752
|
-
| **loaderx** |
|
|
753
|
-
| loaderx-raw |
|
|
786
|
+
| **loaderx** | 48.8 batches/s | 107.2 batches/s | 88.0 batches/s | **2.20x** | **1.80x** |
|
|
787
|
+
| loaderx-raw | 51.2 batches/s | 113.1 batches/s | 94.1 batches/s | **2.21x** | **1.84x** |
|
|
754
788
|
|
|
755
|
-
Peak PSS for compressed/raw was
|
|
756
|
-
|
|
789
|
+
Peak PSS for compressed/raw was 1193/1202 MiB with GIL Python,
|
|
790
|
+
1291/1314 MiB with Numba, and 1190/1191 MiB with free-threaded Python. These are
|
|
757
791
|
two deployment solutions to the transform bottleneck, not claims that Store
|
|
758
792
|
itself was optimized for either runtime.
|
|
759
793
|
|
|
@@ -766,11 +800,13 @@ prefix and reconstructs the exact arrays. The speedup is not
|
|
|
766
800
|
bought with sampling shortcuts: the IID draw is unbiased like NumPy's (Lemire
|
|
767
801
|
with rejection, so uniformity costs nothing over a real index space).
|
|
768
802
|
|
|
769
|
-
**The layouts match what a training loader does.** Zrecord
|
|
770
|
-
record access: Dense gathers fixed-width
|
|
771
|
-
Ragged restores independently shaped records
|
|
772
|
-
|
|
773
|
-
|
|
803
|
+
**The layouts match what a training loader does.** Zrecord preserves an ordered
|
|
804
|
+
record sequence while supporting indexed access: Dense gathers fixed-width
|
|
805
|
+
records directly into one ndarray; Ragged restores independently shaped records
|
|
806
|
+
from inline shape/payload entries. The benchmark deliberately stresses random
|
|
807
|
+
gather rather than claiming that ordering is absent. Array stores are built
|
|
808
|
+
primarily for contiguous scans, so a scattered batch fights their layout. A
|
|
809
|
+
dense raw gather fans out across the shared
|
|
774
810
|
Executor budget where NumPy fancy indexing is one thread; Ragged instead trades some raw
|
|
775
811
|
specialization for compression and a complete variable-shape persistence model.
|
|
776
812
|
|
|
@@ -784,8 +820,8 @@ format: in the current Dense structured-vision workload, plain zstd reaches
|
|
|
784
820
|
**Loader results combine architecture and storage.** loaderx uses threads and
|
|
785
821
|
never ends an epoch, so a step pays no IPC and never waits on an epoch boundary;
|
|
786
822
|
torch uses finite shuffled epochs, worker processes and shared-memory handoff.
|
|
787
|
-
Here compressed loaderx is 1.
|
|
788
|
-
raw loaderx is
|
|
823
|
+
Here compressed loaderx is 1.53x Torch fork, 1.50x Torch spawn and 3.59x Grain;
|
|
824
|
+
raw loaderx is 1.97x, 1.93x and 4.62x faster, respectively.
|
|
789
825
|
Storage also differs per loader — each reads from what it was
|
|
790
826
|
built for — so the loader table is a different comparison from either store
|
|
791
827
|
table, not a rerun.
|
|
@@ -804,7 +840,7 @@ views into one batch allocation, while most byte-store adapters return
|
|
|
804
840
|
independent copies. This is a single benchmark run, while each Store read path
|
|
805
841
|
accumulates at least two timed seconds —
|
|
806
842
|
the µs-scale sampler timings and the loader `batches/s` fluctuate with box load
|
|
807
|
-
(on these 12 cores the compressed loader trails the raw one by about
|
|
843
|
+
(on these 12 cores the compressed loader trails the raw one by about 22%), so
|
|
808
844
|
treat the absolute numbers as ballpark and the cross-backend margins as the
|
|
809
845
|
signal.
|
|
810
846
|
|
|
@@ -1011,11 +1047,12 @@ license to trust storage or the operating system: native code still validates
|
|
|
1011
1047
|
normal I/O behavior, basic malformed-store rejection and native memory safety; checks that only
|
|
1012
1048
|
defend against bypassing the public Python API do not belong in zrecord.
|
|
1013
1049
|
|
|
1014
|
-
1. `RecordEngine`
|
|
1015
|
-
|
|
1016
|
-
|
|
1017
|
-
|
|
1018
|
-
|
|
1050
|
+
1. `RecordEngine` stores N logically ordered records. Parallel chunks may finish
|
|
1051
|
+
and occupy `data.zr` in a different physical order, but logical ID is the
|
|
1052
|
+
stable append position and `RecordLoc[ID]` preserves it. Index and slice
|
|
1053
|
+
operations are implemented as ordered gathers over those positions.
|
|
1054
|
+
2. It hands the container layer a dense sequence space: records are exactly
|
|
1055
|
+
`0..N-1`. Named streams are composed dynamically by a plain Python dict;
|
|
1019
1056
|
`DataLoader` validates that the independent containers have equal lengths.
|
|
1020
1057
|
3. The engine reads and writes byte ranges. Python owns the record schema and
|
|
1021
1058
|
selects Dense or Ragged geometry; the private ABI receives only the runtime
|