litdata 0.2.68__tar.gz → 0.2.70__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {litdata-0.2.68/src/litdata.egg-info → litdata-0.2.70}/PKG-INFO +445 -94
- {litdata-0.2.68 → litdata-0.2.70}/README.md +444 -93
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/__about__.py +1 -1
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/__init__.py +23 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/constants.py +9 -1
- litdata-0.2.70/src/litdata/processing/complete.py +54 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/processing/data_processor.py +961 -244
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/processing/functions.py +24 -18
- litdata-0.2.70/src/litdata/processing/media_folder.py +117 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/processing/readers.py +69 -28
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/streaming/async_prefetch.py +9 -0
- litdata-0.2.70/src/litdata/streaming/collate.py +47 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/streaming/config.py +11 -3
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/streaming/dataloader.py +50 -5
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/streaming/dataset.py +252 -38
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/streaming/downloader.py +51 -0
- litdata-0.2.70/src/litdata/streaming/elastic.py +243 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/streaming/item_loader.py +57 -25
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/streaming/reader.py +21 -7
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/streaming/resolver.py +1 -1
- litdata-0.2.70/src/litdata/streaming/serializers.py +1668 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/streaming/writer.py +53 -27
- litdata-0.2.70/src/litdata/types.py +230 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/utilities/dataset_utilities.py +42 -4
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/utilities/keys_index.py +5 -3
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/utilities/train_test_split.py +54 -0
- {litdata-0.2.68 → litdata-0.2.70/src/litdata.egg-info}/PKG-INFO +445 -94
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata.egg-info/SOURCES.txt +5 -0
- litdata-0.2.68/src/litdata/streaming/serializers.py +0 -595
- {litdata-0.2.68 → litdata-0.2.70}/CONTRIBUTING.md +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/LICENSE +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/MANIFEST.in +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/requirements.txt +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/setup.cfg +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/setup.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/__main__.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/cli/__init__.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/cli/commands.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/cli/handler/__init__.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/cli/handler/cache.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/cli/handler/optimize.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/cli/parser.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/debugger.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/exceptions.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/helpers.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/imports.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/processing/__init__.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/processing/utilities.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/raw/__init__.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/raw/dataset.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/raw/indexer.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/raw/types.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/requirements.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/streaming/__init__.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/streaming/cache.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/streaming/client.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/streaming/combined.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/streaming/compression.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/streaming/dataset_update.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/streaming/fs_provider.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/streaming/parallel.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/streaming/posix_fast.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/streaming/sampler.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/streaming/shuffle.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/streaming/timing.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/utilities/__init__.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/utilities/_pytree.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/utilities/base.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/utilities/breakpoint.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/utilities/broadcast.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/utilities/encryption.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/utilities/env.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/utilities/format.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/utilities/hf_dataset.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/utilities/packing.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/utilities/parquet.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/utilities/shuffle.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/utilities/subsample.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata/utilities/torch_utils.py +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata.egg-info/dependency_links.txt +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata.egg-info/entry_points.txt +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata.egg-info/not-zip-safe +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata.egg-info/requires.txt +0 -0
- {litdata-0.2.68 → litdata-0.2.70}/src/litdata.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: litdata
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.70
|
|
4
4
|
Summary: The Deep Learning framework to train, deploy, and ship AI products Lightning fast.
|
|
5
5
|
Home-page: https://github.com/Lightning-AI/litdata
|
|
6
6
|
Download-URL: https://github.com/Lightning-AI/litdata
|
|
@@ -71,15 +71,31 @@ Dynamic: summary
|
|
|
71
71
|
|
|
72
72
|
|
|
73
73
|
|
|
74
|
-
<
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
✅
|
|
81
|
-
|
|
82
|
-
|
|
74
|
+
<table>
|
|
75
|
+
<tr>
|
|
76
|
+
<td valign="top" align="left">
|
|
77
|
+
|
|
78
|
+
**Transform**
|
|
79
|
+
|
|
80
|
+
✅ Parallelize data processing
|
|
81
|
+
✅ Create vector embeddings
|
|
82
|
+
✅ Run distributed inference
|
|
83
|
+
✅ Scrape websites at scale
|
|
84
|
+
|
|
85
|
+
</td>
|
|
86
|
+
<td valign="top" align="left">
|
|
87
|
+
|
|
88
|
+
**Optimize / Stream**
|
|
89
|
+
|
|
90
|
+
✅ Stream raw files with no prep
|
|
91
|
+
✅ Stream large cloud datasets
|
|
92
|
+
✅ Accelerate training by 20x
|
|
93
|
+
✅ Pause and resume data streaming
|
|
94
|
+
✅ Use remote data without local loading
|
|
95
|
+
|
|
96
|
+
</td>
|
|
97
|
+
</tr>
|
|
98
|
+
</table>
|
|
83
99
|
|
|
84
100
|
---
|
|
85
101
|
|
|
@@ -93,11 +109,14 @@ Transform Optimize / Stream
|
|
|
93
109
|
<a href="#quick-start">Quick start</a> •
|
|
94
110
|
<a href="#speed-up-model-training">Optimize data</a> •
|
|
95
111
|
<a href="#transform-datasets">Transform data</a> •
|
|
112
|
+
<a href="#modality">Modality</a> •
|
|
96
113
|
<a href="#key-features">Features</a> •
|
|
97
114
|
<a href="#stream-raw">Stream raw files</a> •
|
|
98
115
|
<a href="#resolve-paths">Paths & cloud URLs</a> •
|
|
99
116
|
<a href="#benchmarks">Benchmarks</a> •
|
|
100
117
|
<a href="#start-from-a-template">Templates</a> •
|
|
118
|
+
<a href="#used-by">Used by</a> •
|
|
119
|
+
<a href="#skills">Skills</a> •
|
|
101
120
|
<a href="#community">Community</a>
|
|
102
121
|
</p>
|
|
103
122
|
|
|
@@ -156,7 +175,7 @@ On Linux/macOS, `[extras]` includes optional `uvloop` for a faster asyncio event
|
|
|
156
175
|
<details>
|
|
157
176
|
<summary>AI agent skill (Cursor, Claude Code, …)</summary>
|
|
158
177
|
|
|
159
|
-
Install the LitData expert skill so coding agents know the full API, path resolver, optimize/stream recipes, and internals
|
|
178
|
+
Install the LitData expert skill so coding agents know the full API, path resolver, optimize/stream recipes, and internals. Full file map → [Skills](#skills).
|
|
160
179
|
|
|
161
180
|
```bash
|
|
162
181
|
npx skills add Lightning-AI/litData
|
|
@@ -215,24 +234,19 @@ Transform raw data into optimized chunks for maximum streaming speed.
|
|
|
215
234
|
This step formats the dataset for fast loading by writing data in an efficient chunked binary format.
|
|
216
235
|
|
|
217
236
|
```python
|
|
218
|
-
import io
|
|
219
237
|
import numpy as np
|
|
220
|
-
from PIL import Image
|
|
221
238
|
import litdata as ld
|
|
222
239
|
|
|
223
240
|
def random_images(index):
|
|
224
|
-
# Replace with your
|
|
225
|
-
#
|
|
226
|
-
#
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
# Keys/types must stay stable across samples; list lengths/types fixed
|
|
235
|
-
return {"index": index, "image": jpeg_image, "class": fake_labels}
|
|
241
|
+
# Replace with your files: Image(path="photo.jpg") or Image(bytes=...).
|
|
242
|
+
# Wrappers pick the serializer (a caption string is not an image).
|
|
243
|
+
# quality/format encode JPEG — not uncompressed PIL RAW.
|
|
244
|
+
array = np.random.randint(0, 256, (32, 32, 3), dtype=np.uint8)
|
|
245
|
+
return {
|
|
246
|
+
"index": index,
|
|
247
|
+
"image": ld.Image(array=array, quality=95, format="jpeg"),
|
|
248
|
+
"class": np.random.randint(10),
|
|
249
|
+
}
|
|
236
250
|
|
|
237
251
|
if __name__ == "__main__":
|
|
238
252
|
# Exactly one of chunk_bytes or chunk_size
|
|
@@ -350,6 +364,128 @@ ld.map(
|
|
|
350
364
|
|
|
351
365
|
----
|
|
352
366
|
|
|
367
|
+
# Modality <a id="media-types"></a>
|
|
368
|
+
|
|
369
|
+
Wrap each file so a caption is not treated as a path: Text(path=...), Image(path=...), Audio(path=...). Path and raw bytes are stored as-is; array / image / mesh encode.
|
|
370
|
+
|
|
371
|
+
<table width="100%">
|
|
372
|
+
<tr>
|
|
373
|
+
<th align="left">Type</th>
|
|
374
|
+
<th align="left">Write</th>
|
|
375
|
+
<th align="left">Stream</th>
|
|
376
|
+
</tr>
|
|
377
|
+
<tr>
|
|
378
|
+
<td colspan="3"><strong>Text</strong></td>
|
|
379
|
+
</tr>
|
|
380
|
+
<tr>
|
|
381
|
+
<td valign="top"><a href="examples/modality/text.py">Text</a></td>
|
|
382
|
+
<td>Text(path="a.txt")<br>Text(bytes=utf8)<br>Text(text="a caption")</td>
|
|
383
|
+
<td>text # str</td>
|
|
384
|
+
</tr>
|
|
385
|
+
<tr>
|
|
386
|
+
<td valign="top"><a href="examples/modality/text.py">Tokens</a></td>
|
|
387
|
+
<td>Tensor(array=token_ids)<br>optimize(..., item_loader=TokensLoader())</td>
|
|
388
|
+
<td>tokens # Tensor, length block_size — <a href="#llm-training">LLM training</a></td>
|
|
389
|
+
</tr>
|
|
390
|
+
<tr>
|
|
391
|
+
<td colspan="3"><strong>Image</strong></td>
|
|
392
|
+
</tr>
|
|
393
|
+
<tr>
|
|
394
|
+
<td valign="top"><a href="examples/modality/image.py">Image</a></td>
|
|
395
|
+
<td>Image(path="a.jpg")<br>Image(bytes=jpeg)<br>Image(array=hwc, quality=95, format="jpeg")</td>
|
|
396
|
+
<td>image.shape # Tensor CHW</td>
|
|
397
|
+
</tr>
|
|
398
|
+
<tr>
|
|
399
|
+
<td valign="top"><a href="examples/modality/jpeg.py">Jpeg</a></td>
|
|
400
|
+
<td>Jpeg(path="a.jpg")<br>Jpeg(array=hwc, quality=95)</td>
|
|
401
|
+
<td>image.shape # Tensor CHW</td>
|
|
402
|
+
</tr>
|
|
403
|
+
<tr>
|
|
404
|
+
<td valign="top"><a href="examples/modality/jpeg_array.py">JpegArray</a></td>
|
|
405
|
+
<td>JpegArray(images=[Jpeg(path=p) for p in frames])</td>
|
|
406
|
+
<td>images[0].shape # Tensor CHW</td>
|
|
407
|
+
</tr>
|
|
408
|
+
<tr>
|
|
409
|
+
<td valign="top"><a href="examples/modality/pil.py">Pil</a></td>
|
|
410
|
+
<td>Pil(path="a.png")<br>Pil(image=pil_img, mode="RGB")</td>
|
|
411
|
+
<td>pil_img.size # PIL.Image</td>
|
|
412
|
+
</tr>
|
|
413
|
+
<tr>
|
|
414
|
+
<td valign="top"><a href="examples/modality/tiff.py">Tiff</a></td>
|
|
415
|
+
<td>Tiff(path="a.tif")<br>Tiff(array=hw)</td>
|
|
416
|
+
<td>array.shape # NumPy</td>
|
|
417
|
+
</tr>
|
|
418
|
+
<tr>
|
|
419
|
+
<td colspan="3"><strong>Audio and Video</strong></td>
|
|
420
|
+
</tr>
|
|
421
|
+
<tr>
|
|
422
|
+
<td valign="top"><a href="examples/modality/audio.py">Audio</a></td>
|
|
423
|
+
<td>Audio(path="a.wav")<br>Audio(bytes=wav)<br>Audio(array=wave, sampling_rate=16000)</td>
|
|
424
|
+
<td>audio["array"]<br>audio["sampling_rate"]</td>
|
|
425
|
+
</tr>
|
|
426
|
+
<tr>
|
|
427
|
+
<td valign="top"><a href="examples/modality/video.py">Video</a></td>
|
|
428
|
+
<td>Video(path="c.mp4")<br>Video(bytes=mp4)<br>Video(array=frames, fps=25)</td>
|
|
429
|
+
<td>video.get_frames_at(0)<br>video.get_frames_in_range(0, 8)</td>
|
|
430
|
+
</tr>
|
|
431
|
+
<tr>
|
|
432
|
+
<td colspan="3"><strong>File</strong></td>
|
|
433
|
+
</tr>
|
|
434
|
+
<tr>
|
|
435
|
+
<td valign="top"><a href="examples/modality/file.py">File</a></td>
|
|
436
|
+
<td>File(path="doc.bin")<br>File(bytes=blob)</td>
|
|
437
|
+
<td>sidecar # raw bytes</td>
|
|
438
|
+
</tr>
|
|
439
|
+
<tr>
|
|
440
|
+
<td valign="top"><a href="examples/modality/pdf.py">Pdf</a></td>
|
|
441
|
+
<td>Pdf(path="p.pdf")<br>Pdf(pdf=pdfplumber_doc)</td>
|
|
442
|
+
<td>pdf.pages[0] # Pdfplumber</td>
|
|
443
|
+
</tr>
|
|
444
|
+
<tr>
|
|
445
|
+
<td colspan="3"><strong>3D and volume</strong></td>
|
|
446
|
+
</tr>
|
|
447
|
+
<tr>
|
|
448
|
+
<td valign="top"><a href="examples/modality/mesh.py">Mesh</a></td>
|
|
449
|
+
<td>Mesh(path="m.glb")<br>Mesh(mesh=trimesh_obj, file_type="glb")</td>
|
|
450
|
+
<td>mesh.vertices # Trimesh</td>
|
|
451
|
+
</tr>
|
|
452
|
+
<tr>
|
|
453
|
+
<td valign="top"><a href="examples/modality/nifti.py">Nifti</a></td>
|
|
454
|
+
<td>Nifti(path="v.nii.gz")<br>Nifti(array=vol, affine=np.eye(4))</td>
|
|
455
|
+
<td>nifti.get_fdata() # Nibabel</td>
|
|
456
|
+
</tr>
|
|
457
|
+
<tr>
|
|
458
|
+
<td colspan="3"><strong>Array and Graph</strong></td>
|
|
459
|
+
</tr>
|
|
460
|
+
<tr>
|
|
461
|
+
<td valign="top"><a href="examples/modality/numpy_array.py">Numpy</a></td>
|
|
462
|
+
<td>np.load("a.npy")<br>np.zeros((3, 4, 4))</td>
|
|
463
|
+
<td>array # NumPy</td>
|
|
464
|
+
</tr>
|
|
465
|
+
<tr>
|
|
466
|
+
<td valign="top"><a href="examples/modality/tensor.py">Tensor</a></td>
|
|
467
|
+
<td>Tensor(array=torch.randn(3, 4, 4))</td>
|
|
468
|
+
<td>feat # Tensor — 1-D token ids use TokensLoader under Text</td>
|
|
469
|
+
</tr>
|
|
470
|
+
<tr>
|
|
471
|
+
<td valign="top"><a href="examples/modality/graph.py">Graph</a></td>
|
|
472
|
+
<td>Data(x=…, edge_index=…, y=…)<br>Graph(x=…, edge_index=…, y=…)<br>Graph(data=pyg_data)</td>
|
|
473
|
+
<td>graph.x, graph.edge_index # PyG Data or Graph — <a href="#pyg-graphs">PyG graphs</a></td>
|
|
474
|
+
</tr>
|
|
475
|
+
<tr>
|
|
476
|
+
<td colspan="3"><strong>Parquet</strong></td>
|
|
477
|
+
</tr>
|
|
478
|
+
<tr>
|
|
479
|
+
<td valign="top"><a href="examples/modality/parquet.py">Parquet</a></td>
|
|
480
|
+
<td>folder of .parquet files<br>StreamingDataset(..., item_loader=ParquetLoader())</td>
|
|
481
|
+
<td>row["col"] # dict of columns — <a href="#stream-parquet">stream parquet</a></td>
|
|
482
|
+
</tr>
|
|
483
|
+
</table>
|
|
484
|
+
|
|
485
|
+
Examples (path on disk → optimize → batch): [examples/modality](examples/modality).
|
|
486
|
+
|
|
487
|
+
----
|
|
488
|
+
|
|
353
489
|
# Key Features
|
|
354
490
|
|
|
355
491
|
## Features for optimizing and streaming datasets for model training
|
|
@@ -565,24 +701,18 @@ How you return images from `optimize` controls storage size and streaming speed.
|
|
|
565
701
|
|
|
566
702
|
| What you return | Serializer | Result |
|
|
567
703
|
|-----------------|------------|--------|
|
|
568
|
-
| `
|
|
569
|
-
|
|
|
704
|
+
| `litdata.Image(path=...)` / `Image(array=..., quality=95, format="jpeg")` | `image` | Compressed bytes — **preferred** |
|
|
705
|
+
| `litdata.Jpeg(path=...)` / `Jpeg(array=..., quality=95)` | `jpeg` | JPEG bytes |
|
|
706
|
+
| `PIL.JpegImageFile` (e.g. `PIL.Image.open("x.jpg")`) | `jpeg` | Compressed bytes |
|
|
707
|
+
| Plain `PIL.Image` / `Image.fromarray(...)` | `pil` | Uncompressed pixels — often **10×+ larger** |
|
|
570
708
|
|
|
571
|
-
**Best practice:**
|
|
709
|
+
**Best practice:** wrap with `Image` / `Jpeg` at **quality ≈ 95**, or keep existing `.jpg` files via `Image(path=...)`. Resize when helpful.
|
|
572
710
|
|
|
573
711
|
```python
|
|
574
|
-
import io
|
|
575
|
-
from PIL import Image
|
|
576
712
|
import litdata as ld
|
|
577
713
|
|
|
578
714
|
def load_image(path):
|
|
579
|
-
|
|
580
|
-
if not str(path).lower().endswith((".jpg", ".jpeg")):
|
|
581
|
-
buf = io.BytesIO()
|
|
582
|
-
img.convert("RGB").save(buf, format="JPEG", quality=95)
|
|
583
|
-
buf.seek(0)
|
|
584
|
-
img = Image.open(buf) # JpegImageFile
|
|
585
|
-
return {"image": img, "path": path}
|
|
715
|
+
return {"image": ld.Image(path=path, quality=95, format="jpeg"), "id": path}
|
|
586
716
|
|
|
587
717
|
if __name__ == "__main__":
|
|
588
718
|
ld.optimize(fn=load_image, inputs=list_of_paths, output_dir="fast_data", chunk_bytes="64MB", num_workers=8)
|
|
@@ -596,9 +726,9 @@ Ready-made ImageNet optimize/stream scripts: `benchmarks/litdata/` (`--write_mod
|
|
|
596
726
|
<summary> ✅ Custom serializers <a id="serializers" href="#serializers">🔗</a> </summary>
|
|
597
727
|
|
|
598
728
|
|
|
599
|
-
LitData serializes each leaf
|
|
729
|
+
LitData serializes each **pytree leaf** with a pluggable registry. Built-ins (tried in order) include: `str`, `bool`, `int`, `float`, `video`, `audio`, `image`, `nifti`, `mesh`, `pdf`, `tifffile`, `file`, `pil`, `jpeg`, `jpeg_array`, `bytes`, `numpy` / `tensor` (and no-header variants), `graph`, and `pickle` (fallback).
|
|
600
730
|
|
|
601
|
-
For images,
|
|
731
|
+
Prefer [typed media wrappers](#media-types) (`Audio`, `Video`, `Image`, `Graph`, …) so a filepath is not confused with a caption. For images, `Image(..., quality=95, format="jpeg")` or a `JpegImageFile` stores JPEG; a plain `PIL.Image` selects **`pil`** (raw pixels). See [Optimize images as JPEG](#optimize-jpeg).
|
|
602
732
|
|
|
603
733
|
Pass custom serializers when **streaming** (and when using the lower-level `Cache` writer):
|
|
604
734
|
|
|
@@ -622,7 +752,133 @@ dataset = StreamingDataset(
|
|
|
622
752
|
)
|
|
623
753
|
```
|
|
624
754
|
|
|
625
|
-
Keys you pass are tried before the defaults (so they win over `pickle`). `optimize()` uses the built-in registry based on the Python types your `fn` returns — prefer JPEG / numpy / tensor leaves for best results.
|
|
755
|
+
Keys you pass are tried before the defaults (so they win over `pickle`). `optimize()` uses the built-in registry based on the Python types your `fn` returns — prefer typed wrappers / JPEG / numpy / tensor leaves for best results.
|
|
756
|
+
|
|
757
|
+
</details>
|
|
758
|
+
|
|
759
|
+
<details>
|
|
760
|
+
<summary> ✅ Stream PyG graphs <a id="pyg-graphs" href="#pyg-graphs">🔗</a> </summary>
|
|
761
|
+
|
|
762
|
+
|
|
763
|
+
Store [PyTorch Geometric](https://pytorch-geometric.readthedocs.io/) `Data` / `HeteroData` as packed tensors (`to_dict()`), not `torch.save` / pickle. On read, LitData reconstructs with `from_dict` when `torch-geometric` is installed. `optimize` needs a **top-level** function (spawn).
|
|
764
|
+
|
|
765
|
+
`StreamingDataLoader` uses `litdata_collate` by default: graph samples become a `DataBatch` (`Batch.from_data_list`); everything else uses PyTorch `default_collate`. For `follow_batch` / `exclude_keys`, use `torch_geometric.loader.DataLoader`. Without PyG, graph batches stay a list of `Graph`.
|
|
766
|
+
|
|
767
|
+
### Homogeneous `Data` + GCN
|
|
768
|
+
|
|
769
|
+
```python
|
|
770
|
+
import torch
|
|
771
|
+
import torch.nn.functional as F
|
|
772
|
+
from torch_geometric.data import Data
|
|
773
|
+
from torch_geometric.nn import GCNConv, global_mean_pool
|
|
774
|
+
|
|
775
|
+
from litdata import StreamingDataLoader, StreamingDataset, optimize
|
|
776
|
+
|
|
777
|
+
def make_graph(i: int) -> Data:
|
|
778
|
+
n = 8 + i % 5
|
|
779
|
+
src = torch.randint(0, n, (12,), dtype=torch.long)
|
|
780
|
+
dst = torch.randint(0, n, (12,), dtype=torch.long)
|
|
781
|
+
return Data(
|
|
782
|
+
x=torch.randn(n, 8),
|
|
783
|
+
edge_index=torch.stack([src, dst], 0),
|
|
784
|
+
y=torch.tensor(i % 3),
|
|
785
|
+
train_mask=torch.ones(n, dtype=torch.bool),
|
|
786
|
+
num_nodes=n,
|
|
787
|
+
)
|
|
788
|
+
|
|
789
|
+
optimize(make_graph, inputs=list(range(1024)), output_dir="graphs", chunk_size=64)
|
|
790
|
+
|
|
791
|
+
dataset = StreamingDataset("graphs")
|
|
792
|
+
sample = dataset[0] # Data when PyG is installed, else Graph
|
|
793
|
+
loader = StreamingDataLoader(dataset, batch_size=32, shuffle=True)
|
|
794
|
+
batch = next(iter(loader)) # DataBatch
|
|
795
|
+
|
|
796
|
+
class Net(torch.nn.Module):
|
|
797
|
+
def __init__(self):
|
|
798
|
+
super().__init__()
|
|
799
|
+
self.conv = GCNConv(8, 16)
|
|
800
|
+
self.lin = torch.nn.Linear(16, 3)
|
|
801
|
+
|
|
802
|
+
def forward(self, data):
|
|
803
|
+
x = F.relu(self.conv(data.x, data.edge_index))
|
|
804
|
+
return self.lin(global_mean_pool(x, data.batch))
|
|
805
|
+
```
|
|
806
|
+
|
|
807
|
+
### `Graph` wrapper (no PyG at write time)
|
|
808
|
+
|
|
809
|
+
```python
|
|
810
|
+
from litdata import Graph, optimize
|
|
811
|
+
|
|
812
|
+
def make_graph(i: int) -> Graph:
|
|
813
|
+
n = 6
|
|
814
|
+
return Graph(
|
|
815
|
+
x=torch.randn(n, 4),
|
|
816
|
+
edge_index=torch.tensor([[0, 1, 2], [1, 2, 0]], dtype=torch.long),
|
|
817
|
+
y=torch.tensor(i % 2),
|
|
818
|
+
data={"num_nodes": n}, # extra tensors/scalars; field kwargs override data=
|
|
819
|
+
)
|
|
820
|
+
|
|
821
|
+
optimize(make_graph, inputs=list(range(256)), output_dir="graphs")
|
|
822
|
+
# later: sample.to_pyg() if the stream returned Graph
|
|
823
|
+
```
|
|
824
|
+
|
|
825
|
+
`Graph(data=pyg_data)` uses `pyg_data.to_dict()`. Do not mix tensor fields with an opaque NetworkX `data=`.
|
|
826
|
+
|
|
827
|
+
### Heterogeneous `HeteroData`
|
|
828
|
+
|
|
829
|
+
```python
|
|
830
|
+
from torch_geometric.data import HeteroData
|
|
831
|
+
|
|
832
|
+
from litdata import StreamingDataLoader, StreamingDataset, optimize
|
|
833
|
+
|
|
834
|
+
def make_hetero(i: int) -> HeteroData:
|
|
835
|
+
data = HeteroData()
|
|
836
|
+
data["paper"].x = torch.randn(8, 16)
|
|
837
|
+
data["author"].x = torch.randn(4, 8)
|
|
838
|
+
data["author", "writes", "paper"].edge_index = torch.tensor(
|
|
839
|
+
[[0, 1, 2, 3], [0, 2, 4, 6]], dtype=torch.long
|
|
840
|
+
)
|
|
841
|
+
data.y = torch.tensor(i % 3)
|
|
842
|
+
return data
|
|
843
|
+
|
|
844
|
+
optimize(make_hetero, inputs=list(range(512)), output_dir="hetero", chunk_size=32)
|
|
845
|
+
|
|
846
|
+
dataset = StreamingDataset("hetero")
|
|
847
|
+
sample = dataset[0] # HeteroData
|
|
848
|
+
print(sample["paper"].x.shape, sample["author", "writes", "paper"].edge_index.shape)
|
|
849
|
+
|
|
850
|
+
loader = StreamingDataLoader(dataset, batch_size=16)
|
|
851
|
+
batch = next(iter(loader)) # HeteroDataBatch
|
|
852
|
+
# batch["paper"].x, batch["paper"].batch, batch["author", "writes", "paper"].edge_index
|
|
853
|
+
```
|
|
854
|
+
|
|
855
|
+
### Graph plus metadata in one sample
|
|
856
|
+
|
|
857
|
+
```python
|
|
858
|
+
def make_row(i: int) -> dict:
|
|
859
|
+
return {"id": i, "graph": make_graph(i)}
|
|
860
|
+
|
|
861
|
+
optimize(make_row, inputs=list(range(1024)), output_dir="rows")
|
|
862
|
+
loader = StreamingDataLoader(StreamingDataset("rows"), batch_size=8)
|
|
863
|
+
batch = next(iter(loader))
|
|
864
|
+
# batch["id"] is a tensor; batch["graph"] is a DataBatch
|
|
865
|
+
```
|
|
866
|
+
|
|
867
|
+
### Sample subgraphs first, then stream
|
|
868
|
+
|
|
869
|
+
`NeighborLoader` needs one in-memory graph. To stream, run the sampler in `optimize` and store each subgraph as a `Data`:
|
|
870
|
+
|
|
871
|
+
```python
|
|
872
|
+
# sampler = NeighborSampler(big_graph, num_neighbors=[10, 10])
|
|
873
|
+
|
|
874
|
+
def sample_seed(seed: int) -> Data:
|
|
875
|
+
out = sampler.sample_from_nodes(torch.tensor([seed]))
|
|
876
|
+
return Data(x=out.x, edge_index=out.edge_index, y=out.y)
|
|
877
|
+
|
|
878
|
+
optimize(sample_seed, inputs=train_seeds.tolist(), output_dir="subgraphs")
|
|
879
|
+
```
|
|
880
|
+
|
|
881
|
+
NetworkX (or any non-tensor object) uses `Graph(data=nx_graph)` → `graph:pickle`. Do not `torch.save` a graph into the sample.
|
|
626
882
|
|
|
627
883
|
</details>
|
|
628
884
|
|
|
@@ -989,6 +1245,12 @@ for batch_idx, batch in enumerate(dataloader):
|
|
|
989
1245
|
torch.save(dataloader.state_dict(), "dataloader_state.pt")
|
|
990
1246
|
```
|
|
991
1247
|
|
|
1248
|
+
Same `seed` and `shuffle` are required. For a **`StreamingDataset`**, **`num_workers` and `world_size` may change**: LitData drops a global `sample_in_epoch` prefix and restripes the rest (never duplicates remaining IDs). For a matching loss curve keep **global batch size** (`world_size * batch_size`) constant and DDP ranks in lockstep. `num_canonical_nodes` (default: first-run `world_size`) is frozen in the checkpoint. POSIX `WindowShuffle` resumes whole remaining chunks. **`CombinedStreamingDataset` and `ParallelStreamingDataset` resume only with the same `world_size`, `num_workers`, and `batch_size`.**
|
|
1249
|
+
|
|
1250
|
+
```python
|
|
1251
|
+
dataset = StreamingDataset("s3://my-bucket/my-data", shuffle=True, num_canonical_nodes=8)
|
|
1252
|
+
```
|
|
1253
|
+
|
|
992
1254
|
</details>
|
|
993
1255
|
|
|
994
1256
|
|
|
@@ -996,24 +1258,21 @@ for batch_idx, batch in enumerate(dataloader):
|
|
|
996
1258
|
<summary> ✅ Use shared queue for Optimizing <a id="shared-queue" href="#shared-queue">🔗</a> </summary>
|
|
997
1259
|
|
|
998
1260
|
|
|
999
|
-
|
|
1000
|
-
|
|
1001
|
-
This is especially useful when optimizing large datasets in parallel, where some workers may be slower than others.
|
|
1261
|
+
`optimize` / `map` default to a **shared per-node queue** (`keep_data_ordered=False`). Work is packed per node, then every worker on that node pulls the next item, so a slow worker does not leave others idle. Set `keep_data_ordered=True` to keep a static per-worker slice (required for `use_checkpoint` and `align_chunking`).
|
|
1002
1262
|
|
|
1003
|
-
|
|
1263
|
+
Local `output_dir` writes chunks in place. Remote inputs and outputs use the streaming downloader (`adownload_file` / `aupload_file`, obstore when available).
|
|
1004
1264
|
|
|
1005
1265
|
```python
|
|
1006
1266
|
import numpy as np
|
|
1007
|
-
from PIL import Image
|
|
1008
1267
|
import litdata as ld
|
|
1009
1268
|
|
|
1010
1269
|
def random_images(index):
|
|
1011
|
-
|
|
1012
|
-
|
|
1013
|
-
|
|
1014
|
-
|
|
1015
|
-
|
|
1016
|
-
|
|
1270
|
+
array = np.random.randint(0, 256, (32, 32, 3), dtype=np.uint8)
|
|
1271
|
+
return {
|
|
1272
|
+
"index": index,
|
|
1273
|
+
"image": ld.Image(array=array, quality=95, format="jpeg"),
|
|
1274
|
+
"class": np.random.randint(10),
|
|
1275
|
+
}
|
|
1017
1276
|
|
|
1018
1277
|
if __name__ == "__main__":
|
|
1019
1278
|
# The optimize function writes data in an optimized format.
|
|
@@ -1023,28 +1282,36 @@ if __name__ == "__main__":
|
|
|
1023
1282
|
output_dir="fast_data", # optimized data is stored here
|
|
1024
1283
|
num_workers=4, # The number of workers on the same machine
|
|
1025
1284
|
chunk_bytes="64MB" , # size of each chunk
|
|
1026
|
-
keep_data_ordered=False, #
|
|
1285
|
+
keep_data_ordered=False, # default: shared queue (set True to keep input order)
|
|
1027
1286
|
)
|
|
1028
1287
|
```
|
|
1029
1288
|
|
|
1030
1289
|
|
|
1031
|
-
###
|
|
1290
|
+
### Shared queue vs ordered (skewed local files)
|
|
1032
1291
|
|
|
1033
|
-
|
|
1292
|
+
`scripts/bench/bench_node_queue.py --files 4000 --workers 8` (first 500 files are 1 MiB). On `main`, unordered optimize sat on a 200s empty-queue timeout after work finished.
|
|
1034
1293
|
|
|
1035
|
-
|
|
|
1036
|
-
|
|
1037
|
-
|
|
|
1038
|
-
|
|
|
1294
|
+
| Tree | Mode | Time | Throughput |
|
|
1295
|
+
|------|------|-----:|-----------:|
|
|
1296
|
+
| `main` (old default) | `keep_data_ordered=True` | 23.7s | 169 files/s |
|
|
1297
|
+
| `main` | `keep_data_ordered=False` | 223.6s | 18 files/s |
|
|
1298
|
+
| this tree | `keep_data_ordered=True` | 22.9s | 175 files/s |
|
|
1299
|
+
| this tree (**new default**) | `keep_data_ordered=False` | **18.8s** | 213 files/s |
|
|
1039
1300
|
|
|
1040
|
-
|
|
1041
|
-
|
|
1042
|
-
|
|
1043
|
-
|
|
1301
|
+
Shared-queue **before → after: ~12×**. New default vs old ordered default: **1.22×**.
|
|
1302
|
+
|
|
1303
|
+
### Local / remote input × output
|
|
1304
|
+
|
|
1305
|
+
`python scripts/bench/bench_node_queue.py --files 200 --workers 4 --io-matrix` (first 50 files are 1 MiB).
|
|
1044
1306
|
|
|
1045
|
-
|
|
1307
|
+
| Topology | Ordered | Shared | Speedup |
|
|
1308
|
+
|----------|--------:|-------:|--------:|
|
|
1309
|
+
| local → local | 6.55s | **2.96s** | 2.21× |
|
|
1310
|
+
| remote → local | 7.44s | **3.48s** | 2.14× |
|
|
1311
|
+
| local → remote | 10.98s | **6.51s** | 1.69× |
|
|
1312
|
+
| remote → remote | 11.48s | **8.55s** | 1.34× |
|
|
1046
1313
|
|
|
1047
|
-
|
|
1314
|
+
Shared queue balances uneven workers. It does not change later `StreamingDataset` throughput.
|
|
1048
1315
|
|
|
1049
1316
|
</details>
|
|
1050
1317
|
|
|
@@ -1390,15 +1657,15 @@ Merge multiple optimized datasets into one.
|
|
|
1390
1657
|
|
|
1391
1658
|
```python
|
|
1392
1659
|
import numpy as np
|
|
1393
|
-
from PIL import Image
|
|
1394
1660
|
|
|
1395
|
-
from litdata import StreamingDataset, merge_datasets, optimize
|
|
1661
|
+
from litdata import Image, StreamingDataset, merge_datasets, optimize
|
|
1396
1662
|
|
|
1397
1663
|
|
|
1398
1664
|
def random_images(index):
|
|
1665
|
+
array = np.random.randint(0, 256, (32, 32, 3), dtype=np.uint8)
|
|
1399
1666
|
return {
|
|
1400
1667
|
"index": index,
|
|
1401
|
-
"image": Image
|
|
1668
|
+
"image": Image(array=array, quality=95, format="jpeg"),
|
|
1402
1669
|
"class": np.random.randint(10),
|
|
1403
1670
|
}
|
|
1404
1671
|
|
|
@@ -1415,6 +1682,16 @@ if __name__ == "__main__":
|
|
|
1415
1682
|
print(len(dataset))
|
|
1416
1683
|
# out: 1000
|
|
1417
1684
|
```
|
|
1685
|
+
|
|
1686
|
+
If you wrote chunks yourself (`Cache` / `BinaryWriter`) and only have `{rank}.index.json` shards, finish the dataset:
|
|
1687
|
+
|
|
1688
|
+
```python
|
|
1689
|
+
from litdata import complete_dataset, StreamingDataset
|
|
1690
|
+
|
|
1691
|
+
complete_dataset("my_chunks") # no-op if index.json already exists
|
|
1692
|
+
StreamingDataset("my_chunks") # also tries this automatically
|
|
1693
|
+
```
|
|
1694
|
+
|
|
1418
1695
|
</details>
|
|
1419
1696
|
|
|
1420
1697
|
<details>
|
|
@@ -1500,6 +1777,13 @@ print(test_dataset)
|
|
|
1500
1777
|
# out: 50,000
|
|
1501
1778
|
```
|
|
1502
1779
|
|
|
1780
|
+
Or pick exact indices with `StreamingDataset.subset`:
|
|
1781
|
+
|
|
1782
|
+
```python
|
|
1783
|
+
train = dataset.subset(range(0, 30_000))
|
|
1784
|
+
# dataset.subset(slice(0, 1000))
|
|
1785
|
+
```
|
|
1786
|
+
|
|
1503
1787
|
</details>
|
|
1504
1788
|
|
|
1505
1789
|
<details>
|
|
@@ -1516,6 +1800,9 @@ dataset = StreamingDataset("s3://my-bucket/my-data", subsample=0.01) # data are
|
|
|
1516
1800
|
|
|
1517
1801
|
print(len(dataset)) # display the length of your data
|
|
1518
1802
|
# out: 1000
|
|
1803
|
+
|
|
1804
|
+
# or a list / slice of global indices
|
|
1805
|
+
small = dataset.subset([0, 10, 20])
|
|
1519
1806
|
```
|
|
1520
1807
|
|
|
1521
1808
|
</details>
|
|
@@ -1632,7 +1919,7 @@ ld.index_parquet_dataset(
|
|
|
1632
1919
|
|
|
1633
1920
|
### Stream with `ParquetLoader`
|
|
1634
1921
|
|
|
1635
|
-
|
|
1922
|
+
If the folder looks like parquet and has no `index.json`, `StreamingDataset` now **builds the index automatically**. You still need `ParquetLoader` for local/S3/GCS (it must match `index.json`). `hf://` already auto-indexes and selects the loader.
|
|
1636
1923
|
|
|
1637
1924
|
```python
|
|
1638
1925
|
import litdata as ld
|
|
@@ -1938,7 +2225,8 @@ export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
|
|
|
1938
2225
|
| `LITDATA_OBSTORE_STREAM_MIN_CHUNK_MIB` | `8` | S3 obstore stream chunk size (MiB) |
|
|
1939
2226
|
| `MAX_WAIT_TIME` | `120` | Seconds to wait for a chunk before error |
|
|
1940
2227
|
| `FORCE_DOWNLOAD_TIME` | `30` | Seconds before force re-download of a missing chunk |
|
|
1941
|
-
| `
|
|
2228
|
+
| `LITDATA_CHECK_UPDATES` | unset | `1` enables the PyPI upgrade tip (off by default) |
|
|
2229
|
+
| `LITDATA_DISABLE_VERSION_CHECK` | on unless updates enabled | `1` skips the upgrade tip |
|
|
1942
2230
|
| `HF_TOKEN` | — | Gated Hugging Face datasets |
|
|
1943
2231
|
| `DEBUG_LITDATA` / `PRINT_DEBUG_LOGS` | `0` | Internal debug / stdout logs |
|
|
1944
2232
|
| `LITDATA_LOG_FILE` | `litdata_debug.log` | `enable_tracer()` output path |
|
|
@@ -2344,7 +2632,7 @@ if __name__ == "__main__":
|
|
|
2344
2632
|
| `start_method` | spawn† | Multiprocessing start method (†spawn unless IPython) |
|
|
2345
2633
|
| `optimize_dns` | `None` | Optimized DNS (Studio / cloud) |
|
|
2346
2634
|
| `storage_options` | `{}` | Cloud credentials / endpoints |
|
|
2347
|
-
| `keep_data_ordered` | `
|
|
2635
|
+
| `keep_data_ordered` | `False` | Shared work queue (faster for uneven workers). `True` keeps a static per-worker slice. Forced `True` with `use_checkpoint` / `align_chunking`. |
|
|
2348
2636
|
|
|
2349
2637
|
</details>
|
|
2350
2638
|
|
|
@@ -2380,7 +2668,7 @@ Full knob list for `litdata.optimize` (see Quick start for the minimal recipe).
|
|
|
2380
2668
|
| `start_method` | spawn† | Multiprocessing start method |
|
|
2381
2669
|
| `optimize_dns` | `None` | Optimized DNS |
|
|
2382
2670
|
| `storage_options` | `{}` | Cloud credentials / endpoints |
|
|
2383
|
-
| `keep_data_ordered` | `
|
|
2671
|
+
| `keep_data_ordered` | `False` | Shared queue among workers. `True` keeps input order. Forced `True` with `use_checkpoint` / `align_chunking`. |
|
|
2384
2672
|
| `verbose` | `True` | Progress logging |
|
|
2385
2673
|
|
|
2386
2674
|
Related features: [shared queue](#shared-queue), [queue input](#queue-input), [append/overwrite](#modify-datasets), [compression](#compression), [TokensLoader / LLM](#llm-training), [filter](#filter-data).
|
|
@@ -2445,26 +2733,12 @@ Speed to stream Imagenet 1.2M from other cloud storage providers:
|
|
|
2445
2733
|
|---|---|---|---|
|
|
2446
2734
|
| Cloudflare R2 | LitData | **5335** | **5630** |
|
|
2447
2735
|
|
|
2448
|
-
Speed to stream
|
|
2449
|
-
| Framework | Dataset Mode | Dataset Size @ 256px | Images / sec 1st Epoch (float32) | Images / sec 2nd Epoch (float32) |
|
|
2450
|
-
|---|---|---|---|---|
|
|
2451
|
-
| LitData | PIL RAW | 168 GB | 6647 | 6398 |
|
|
2452
|
-
| LitData | JPEG 90% | 12 GB | 6553 | 6537 |
|
|
2453
|
-
| ffcv (os_cache=True) | RAW | 170 GB | 7263 | 6698 |
|
|
2454
|
-
| ffcv (os_cache=False) | RAW | 170 GB | 7556 | 8169 |
|
|
2455
|
-
| ffcv(os_cache=True) | JPEG 90% | 20 GB | 7653 | 8051 |
|
|
2456
|
-
| ffcv(os_cache=False) | JPEG 90% | 20 GB | 8149 | 8607 |
|
|
2736
|
+
Speed to stream a synthetic ImageNet-scale set from **Vast NFS** with POSIX-fast (mmap in place, decode only, no transforms):
|
|
2457
2737
|
|
|
2458
|
-
|
|
2459
|
-
|
|
2460
|
-
|
|
|
2461
|
-
|
|
2462
|
-
| Copy into local cache (`LITDATA_POSIX_FAST=0`) | 48 | **16.7k** (2-epoch avg) |
|
|
2463
|
-
| POSIX-fast (this default on local/Vast paths) | 48 | **18.2k** |
|
|
2464
|
-
| POSIX-fast + README ImageNet augs (crop 224, flip, float32) | 48 | **12.9k** |
|
|
2465
|
-
| POSIX-fast, all CPU cores | **208** | **35.8k** |
|
|
2466
|
-
|
|
2467
|
-
Notes: 208 workers need enough **MemAvailable**. This host had **928×1 GiB hugepages** reserved and idle (~900 GiB locked); after `nr_hugepages=0`, 208 workers stayed healthy. If `num_workers=os.cpu_count()` would crowd RAM, LitData **clamps** workers (`LITDATA_POSIX_MAX_WORKERS=0` disables) and skips `WILLNEED` prefetch. Real ImageNet JPEG 90% is much smaller (~12 GiB) and usually decodes faster than this q95 noise set.
|
|
2738
|
+
| Workers | Images / sec |
|
|
2739
|
+
|---|---|
|
|
2740
|
+
| 48 | **18.2k** |
|
|
2741
|
+
| 208 | **35.8k** |
|
|
2468
2742
|
|
|
2469
2743
|
### Raw Dataset
|
|
2470
2744
|
|
|
@@ -2572,6 +2846,80 @@ Below are templates for real-world applications of LitData at scale.
|
|
|
2572
2846
|
|
|
2573
2847
|
----
|
|
2574
2848
|
|
|
2849
|
+
# Used by
|
|
2850
|
+
|
|
2851
|
+
<table width="100%">
|
|
2852
|
+
<tr>
|
|
2853
|
+
<th align="left" width="18%">Project</th>
|
|
2854
|
+
<th align="left">Description</th>
|
|
2855
|
+
</tr>
|
|
2856
|
+
<tr>
|
|
2857
|
+
<td valign="top"><a href="https://github.com/sunlabuiuc/PyHealth">PyHealth</a></td>
|
|
2858
|
+
<td>Deep-learning toolkit for clinical prediction (MIMIC, eICU, OMOP, sleep, CXR). <code>set_task()</code> writes processed samples with LitData; <code>SampleDataset</code> subclasses <code>StreamingDataset</code> so training streams chunked EHR tensors instead of holding the cohort in RAM.</td>
|
|
2859
|
+
</tr>
|
|
2860
|
+
<tr>
|
|
2861
|
+
<td valign="top"><a href="https://github.com/prescient-design/lobster">LBSTER</a></td>
|
|
2862
|
+
<td>Protein and biological-sequence language models from Prescient Design (Genentech). Pre-training and concept-bottleneck models (fitness, embeddings, guided generation) stream large sequence corpora through LitData.</td>
|
|
2863
|
+
</tr>
|
|
2864
|
+
<tr>
|
|
2865
|
+
<td valign="top"><a href="https://github.com/OpenSynth-energy/OpenSynth">OpenSynth</a></td>
|
|
2866
|
+
<td>Open toolkit for synthetic smart-meter / energy time series. Generated or historical meter traces are optimized and streamed for model training.</td>
|
|
2867
|
+
</tr>
|
|
2868
|
+
<tr>
|
|
2869
|
+
<td valign="top"><a href="https://github.com/BiomedSciAI/biomed-multi-view">biomed-multi-view</a></td>
|
|
2870
|
+
<td>IBM BiomedSciAI multi-view biomedical models. LitData is used to cache and stream paired modalities during training.</td>
|
|
2871
|
+
</tr>
|
|
2872
|
+
<tr>
|
|
2873
|
+
<td valign="top"><a href="https://github.com/cma2015/DEM">DEM</a></td>
|
|
2874
|
+
<td>Phenotype and gene-mining pipeline (<code>biodem</code>). Large genomic / trait tables are packed into LitData chunks for repeated training passes.</td>
|
|
2875
|
+
</tr>
|
|
2876
|
+
<tr>
|
|
2877
|
+
<td valign="top"><a href="https://pypi.org/project/deeptan/">deeptan</a></td>
|
|
2878
|
+
<td>Graph multi-task models for multi-omics trait-associated networks. Guide graphs and expression tables are converted to LitData chunks before GNN training.</td>
|
|
2879
|
+
</tr>
|
|
2880
|
+
<tr>
|
|
2881
|
+
<td valign="top"><a href="https://github.com/avitai/datarax">datarax</a></td>
|
|
2882
|
+
<td>Data tooling with an optional cloud-streaming extra that uses LitData to read remote datasets without a full local copy.</td>
|
|
2883
|
+
</tr>
|
|
2884
|
+
<tr>
|
|
2885
|
+
<td valign="top"><a href="https://pypi.org/project/fasr/">fasr</a></td>
|
|
2886
|
+
<td>Speech ASR framework. The LitData extra streams audio and transcripts for training instead of random-access file lists.</td>
|
|
2887
|
+
</tr>
|
|
2888
|
+
</table>
|
|
2889
|
+
|
|
2890
|
+
# Skills <a id="skills"></a>
|
|
2891
|
+
|
|
2892
|
+
Coding agents (Cursor, Claude Code, and others) should load the LitData skill instead of guessing the API.
|
|
2893
|
+
|
|
2894
|
+
```bash
|
|
2895
|
+
npx skills add Lightning-AI/litData
|
|
2896
|
+
```
|
|
2897
|
+
|
|
2898
|
+
Useful options: `-g` (user-global), `-a cursor` (Cursor only), `-y` (non-interactive). In this repo the skill already lives at [`.claude/skills/litdata/`](.claude/skills/litdata/). Installer: [skills CLI](https://github.com/vercel-labs/skills).
|
|
2899
|
+
|
|
2900
|
+
Start at [`SKILL.md`](.claude/skills/litdata/SKILL.md), then load [`reference/using-litdata.md`](.claude/skills/litdata/reference/using-litdata.md) before writing examples.
|
|
2901
|
+
|
|
2902
|
+
| File | When to load |
|
|
2903
|
+
| --- | --- |
|
|
2904
|
+
| [SKILL.md](.claude/skills/litdata/SKILL.md) | Triggers, public API, traps |
|
|
2905
|
+
| [using-litdata.md](.claude/skills/litdata/reference/using-litdata.md) | Optimize / stream / raw / modality cookbook |
|
|
2906
|
+
| [streaming.md](.claude/skills/litdata/reference/streaming.md) | Read path, shuffle, resume, serializers |
|
|
2907
|
+
| [processing.md](.claude/skills/litdata/reference/processing.md) | optimize / map orchestration |
|
|
2908
|
+
| [data-movement.md](.claude/skills/litdata/reference/data-movement.md) | Download / upload / FUSE vs direct I/O |
|
|
2909
|
+
| [multi-node.md](.claude/skills/litdata/reference/multi-node.md) | Studio num_nodes jobs |
|
|
2910
|
+
| [resolver.md](.claude/skills/litdata/reference/resolver.md) | Paths, URLs, Studio mounts |
|
|
2911
|
+
| [storage-format.md](.claude/skills/litdata/reference/storage-format.md) | Chunks, `index.json`, writer / reader |
|
|
2912
|
+
| [cache-and-chunk-lifecycle.md](.claude/skills/litdata/reference/cache-and-chunk-lifecycle.md) | Prefetch and eviction |
|
|
2913
|
+
| [env-vars.md](.claude/skills/litdata/reference/env-vars.md) | LITDATA_* and DATA_OPTIMIZER_* |
|
|
2914
|
+
| [keyed-lookup.md](.claude/skills/litdata/reference/keyed-lookup.md) | key_fn, dataset_update |
|
|
2915
|
+
| [debugging.md](.claude/skills/litdata/reference/debugging.md) | enable_tracer, Litracer |
|
|
2916
|
+
| [benchmarking.md](.claude/skills/litdata/reference/benchmarking.md) | Fair benches |
|
|
2917
|
+
| [lightning-studio.md](.claude/skills/litdata/reference/lightning-studio.md) | Studio env and credentials |
|
|
2918
|
+
| [testing.md](.claude/skills/litdata/reference/testing.md) | Pytest / CI |
|
|
2919
|
+
| [contributing.md](.claude/skills/litdata/reference/contributing.md) | PR / lint path |
|
|
2920
|
+
|
|
2921
|
+
Offline streaming what-if (not the Python package): [simulator/](simulator/) (litsim).
|
|
2922
|
+
|
|
2575
2923
|
# Community
|
|
2576
2924
|
LitData is a community project accepting contributions - Let's make the world's most advanced AI data processing framework.
|
|
2577
2925
|
|
|
@@ -2588,8 +2936,7 @@ LitData is a community project accepting contributions - Let's make the world's
|
|
|
2588
2936
|
author = {Thomas Chaton and Lightning AI},
|
|
2589
2937
|
title = {LitData: Transform datasets at scale. Optimize datasets for fast AI model training.},
|
|
2590
2938
|
year = {2023},
|
|
2591
|
-
howpublished = {\url{https://github.com/Lightning-AI/litdata}}
|
|
2592
|
-
note = {Accessed: 2025-04-09}
|
|
2939
|
+
howpublished = {\url{https://github.com/Lightning-AI/litdata}}
|
|
2593
2940
|
}
|
|
2594
2941
|
```
|
|
2595
2942
|
|
|
@@ -2597,8 +2944,12 @@ LitData is a community project accepting contributions - Let's make the world's
|
|
|
2597
2944
|
|
|
2598
2945
|
## Papers with LitData
|
|
2599
2946
|
|
|
2600
|
-
*
|
|
2601
|
-
|
|
2947
|
+
Papers that train or stream with LitData (`optimize` / `StreamingDataset`). Scholar hits for “litdata streaming” are often weather *lightning* data, or mention LitData only as example source code.
|
|
2948
|
+
|
|
2949
|
+
| Paper | Venue | How LitData is used |
|
|
2950
|
+
|---|---|---|
|
|
2951
|
+
| [Towards Interpretable Protein Structure Prediction with Sparse Autoencoders](https://arxiv.org/abs/2503.08764) ([code](https://github.com/johnyang101/reticular-sae)) | ICLR 2025 GEM | `optimize` shards ESM-2 embeddings; `StreamingDataset` streams from S3 for multi-GPU SAE training |
|
|
2952
|
+
| [TinyLlama: An Open-Source Small Language Model](https://arxiv.org/abs/2401.02385) ([code](https://github.com/jzhang38/TinyLlama)) | arXiv 2024 | 1.1B pretrain on SlimPajama + StarCoder via Lit-GPT’s `lightning.data` stack (now LitData): `CombinedStreamingDataset` + `TokensLoader` |
|
|
2602
2953
|
|
|
2603
2954
|
----
|
|
2604
2955
|
|