litdata 0.2.69__tar.gz → 0.2.71__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {litdata-0.2.69/src/litdata.egg-info → litdata-0.2.71}/PKG-INFO +453 -80
- {litdata-0.2.69 → litdata-0.2.71}/README.md +452 -79
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/__about__.py +1 -1
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/__init__.py +23 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/constants.py +9 -1
- litdata-0.2.71/src/litdata/processing/complete.py +54 -0
- litdata-0.2.71/src/litdata/processing/media_folder.py +117 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/processing/readers.py +69 -28
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/streaming/async_prefetch.py +28 -4
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/streaming/cache.py +5 -4
- litdata-0.2.71/src/litdata/streaming/collate.py +47 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/streaming/compression.py +18 -13
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/streaming/config.py +60 -10
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/streaming/dataloader.py +188 -51
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/streaming/dataset.py +45 -24
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/streaming/downloader.py +37 -9
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/streaming/item_loader.py +91 -38
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/streaming/reader.py +80 -21
- litdata-0.2.71/src/litdata/streaming/serializers.py +1670 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/streaming/shuffle.py +57 -29
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/streaming/writer.py +79 -21
- litdata-0.2.71/src/litdata/types.py +230 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/utilities/_pytree.py +79 -19
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/utilities/dataset_utilities.py +42 -4
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/utilities/env.py +12 -0
- litdata-0.2.71/src/litdata/utilities/format.py +168 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/utilities/keys_index.py +2 -2
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/utilities/shuffle.py +64 -12
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/utilities/train_test_split.py +54 -0
- {litdata-0.2.69 → litdata-0.2.71/src/litdata.egg-info}/PKG-INFO +453 -80
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata.egg-info/SOURCES.txt +4 -0
- litdata-0.2.69/src/litdata/streaming/serializers.py +0 -595
- litdata-0.2.69/src/litdata/utilities/format.py +0 -57
- {litdata-0.2.69 → litdata-0.2.71}/CONTRIBUTING.md +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/LICENSE +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/MANIFEST.in +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/requirements.txt +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/setup.cfg +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/setup.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/__main__.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/cli/__init__.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/cli/commands.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/cli/handler/__init__.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/cli/handler/cache.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/cli/handler/optimize.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/cli/parser.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/debugger.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/exceptions.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/helpers.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/imports.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/processing/__init__.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/processing/data_processor.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/processing/functions.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/processing/utilities.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/raw/__init__.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/raw/dataset.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/raw/indexer.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/raw/types.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/requirements.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/streaming/__init__.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/streaming/client.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/streaming/combined.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/streaming/dataset_update.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/streaming/elastic.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/streaming/fs_provider.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/streaming/parallel.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/streaming/posix_fast.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/streaming/resolver.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/streaming/sampler.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/streaming/timing.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/utilities/__init__.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/utilities/base.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/utilities/breakpoint.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/utilities/broadcast.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/utilities/encryption.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/utilities/hf_dataset.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/utilities/packing.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/utilities/parquet.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/utilities/subsample.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata/utilities/torch_utils.py +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata.egg-info/dependency_links.txt +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata.egg-info/entry_points.txt +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata.egg-info/not-zip-safe +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata.egg-info/requires.txt +0 -0
- {litdata-0.2.69 → litdata-0.2.71}/src/litdata.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: litdata
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.71
|
|
4
4
|
Summary: The Deep Learning framework to train, deploy, and ship AI products Lightning fast.
|
|
5
5
|
Home-page: https://github.com/Lightning-AI/litdata
|
|
6
6
|
Download-URL: https://github.com/Lightning-AI/litdata
|
|
@@ -71,15 +71,31 @@ Dynamic: summary
|
|
|
71
71
|
|
|
72
72
|
|
|
73
73
|
|
|
74
|
-
<
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
✅
|
|
81
|
-
|
|
82
|
-
|
|
74
|
+
<table>
|
|
75
|
+
<tr>
|
|
76
|
+
<td valign="top" align="left">
|
|
77
|
+
|
|
78
|
+
**Transform**
|
|
79
|
+
|
|
80
|
+
✅ Parallelize data processing
|
|
81
|
+
✅ Create vector embeddings
|
|
82
|
+
✅ Run distributed inference
|
|
83
|
+
✅ Scrape websites at scale
|
|
84
|
+
|
|
85
|
+
</td>
|
|
86
|
+
<td valign="top" align="left">
|
|
87
|
+
|
|
88
|
+
**Optimize / Stream**
|
|
89
|
+
|
|
90
|
+
✅ Stream raw files with no prep
|
|
91
|
+
✅ Stream large cloud datasets
|
|
92
|
+
✅ Accelerate training by 20x
|
|
93
|
+
✅ Pause and resume data streaming
|
|
94
|
+
✅ Use remote data without local loading
|
|
95
|
+
|
|
96
|
+
</td>
|
|
97
|
+
</tr>
|
|
98
|
+
</table>
|
|
83
99
|
|
|
84
100
|
---
|
|
85
101
|
|
|
@@ -93,11 +109,14 @@ Transform Optimize / Stream
|
|
|
93
109
|
<a href="#quick-start">Quick start</a> •
|
|
94
110
|
<a href="#speed-up-model-training">Optimize data</a> •
|
|
95
111
|
<a href="#transform-datasets">Transform data</a> •
|
|
112
|
+
<a href="#modality">Modality</a> •
|
|
96
113
|
<a href="#key-features">Features</a> •
|
|
97
114
|
<a href="#stream-raw">Stream raw files</a> •
|
|
98
115
|
<a href="#resolve-paths">Paths & cloud URLs</a> •
|
|
99
116
|
<a href="#benchmarks">Benchmarks</a> •
|
|
100
117
|
<a href="#start-from-a-template">Templates</a> •
|
|
118
|
+
<a href="#used-by">Used by</a> •
|
|
119
|
+
<a href="#skills">Skills</a> •
|
|
101
120
|
<a href="#community">Community</a>
|
|
102
121
|
</p>
|
|
103
122
|
|
|
@@ -156,7 +175,7 @@ On Linux/macOS, `[extras]` includes optional `uvloop` for a faster asyncio event
|
|
|
156
175
|
<details>
|
|
157
176
|
<summary>AI agent skill (Cursor, Claude Code, …)</summary>
|
|
158
177
|
|
|
159
|
-
Install the LitData expert skill so coding agents know the full API, path resolver, optimize/stream recipes, and internals
|
|
178
|
+
Install the LitData expert skill so coding agents know the full API, path resolver, optimize/stream recipes, and internals. Full file map → [Skills](#skills).
|
|
160
179
|
|
|
161
180
|
```bash
|
|
162
181
|
npx skills add Lightning-AI/litData
|
|
@@ -215,24 +234,19 @@ Transform raw data into optimized chunks for maximum streaming speed.
|
|
|
215
234
|
This step formats the dataset for fast loading by writing data in an efficient chunked binary format.
|
|
216
235
|
|
|
217
236
|
```python
|
|
218
|
-
import io
|
|
219
237
|
import numpy as np
|
|
220
|
-
from PIL import Image
|
|
221
238
|
import litdata as ld
|
|
222
239
|
|
|
223
240
|
def random_images(index):
|
|
224
|
-
# Replace with your
|
|
225
|
-
#
|
|
226
|
-
#
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
# Keys/types must stay stable across samples; list lengths/types fixed
|
|
235
|
-
return {"index": index, "image": jpeg_image, "class": fake_labels}
|
|
241
|
+
# Replace with your files: Image(path="photo.jpg") or Image(bytes=...).
|
|
242
|
+
# Wrappers pick the serializer (a caption string is not an image).
|
|
243
|
+
# quality/format encode JPEG — not uncompressed PIL RAW.
|
|
244
|
+
array = np.random.randint(0, 256, (32, 32, 3), dtype=np.uint8)
|
|
245
|
+
return {
|
|
246
|
+
"index": index,
|
|
247
|
+
"image": ld.Image(array=array, quality=95, format="jpeg"),
|
|
248
|
+
"class": np.random.randint(10),
|
|
249
|
+
}
|
|
236
250
|
|
|
237
251
|
if __name__ == "__main__":
|
|
238
252
|
# Exactly one of chunk_bytes or chunk_size
|
|
@@ -350,6 +364,128 @@ ld.map(
|
|
|
350
364
|
|
|
351
365
|
----
|
|
352
366
|
|
|
367
|
+
# Modality <a id="media-types"></a>
|
|
368
|
+
|
|
369
|
+
Wrap each file so a caption is not treated as a path: Text(path=...), Image(path=...), Audio(path=...). Path and raw bytes are stored as-is; array / image / mesh encode.
|
|
370
|
+
|
|
371
|
+
<table width="100%">
|
|
372
|
+
<tr>
|
|
373
|
+
<th align="left">Type</th>
|
|
374
|
+
<th align="left">Write</th>
|
|
375
|
+
<th align="left">Stream</th>
|
|
376
|
+
</tr>
|
|
377
|
+
<tr>
|
|
378
|
+
<td colspan="3"><strong>Text</strong></td>
|
|
379
|
+
</tr>
|
|
380
|
+
<tr>
|
|
381
|
+
<td valign="top"><a href="examples/modality/text.py">Text</a></td>
|
|
382
|
+
<td>Text(path="a.txt")<br>Text(bytes=utf8)<br>Text(text="a caption")</td>
|
|
383
|
+
<td>text # str</td>
|
|
384
|
+
</tr>
|
|
385
|
+
<tr>
|
|
386
|
+
<td valign="top"><a href="examples/modality/text.py">Tokens</a></td>
|
|
387
|
+
<td>Tensor(array=token_ids)<br>optimize(..., item_loader=TokensLoader())</td>
|
|
388
|
+
<td>tokens # Tensor, length block_size — <a href="#llm-training">LLM training</a></td>
|
|
389
|
+
</tr>
|
|
390
|
+
<tr>
|
|
391
|
+
<td colspan="3"><strong>Image</strong></td>
|
|
392
|
+
</tr>
|
|
393
|
+
<tr>
|
|
394
|
+
<td valign="top"><a href="examples/modality/image.py">Image</a></td>
|
|
395
|
+
<td>Image(path="a.jpg")<br>Image(bytes=jpeg)<br>Image(array=hwc, quality=95, format="jpeg")</td>
|
|
396
|
+
<td>image.shape # Tensor CHW</td>
|
|
397
|
+
</tr>
|
|
398
|
+
<tr>
|
|
399
|
+
<td valign="top"><a href="examples/modality/jpeg.py">Jpeg</a></td>
|
|
400
|
+
<td>Jpeg(path="a.jpg")<br>Jpeg(array=hwc, quality=95)</td>
|
|
401
|
+
<td>image.shape # Tensor CHW</td>
|
|
402
|
+
</tr>
|
|
403
|
+
<tr>
|
|
404
|
+
<td valign="top"><a href="examples/modality/jpeg_array.py">JpegArray</a></td>
|
|
405
|
+
<td>JpegArray(images=[Jpeg(path=p) for p in frames])</td>
|
|
406
|
+
<td>images[0].shape # Tensor CHW</td>
|
|
407
|
+
</tr>
|
|
408
|
+
<tr>
|
|
409
|
+
<td valign="top"><a href="examples/modality/pil.py">Pil</a></td>
|
|
410
|
+
<td>Pil(path="a.png")<br>Pil(image=pil_img, mode="RGB")</td>
|
|
411
|
+
<td>pil_img.size # PIL.Image</td>
|
|
412
|
+
</tr>
|
|
413
|
+
<tr>
|
|
414
|
+
<td valign="top"><a href="examples/modality/tiff.py">Tiff</a></td>
|
|
415
|
+
<td>Tiff(path="a.tif")<br>Tiff(array=hw)</td>
|
|
416
|
+
<td>array.shape # NumPy</td>
|
|
417
|
+
</tr>
|
|
418
|
+
<tr>
|
|
419
|
+
<td colspan="3"><strong>Audio and Video</strong></td>
|
|
420
|
+
</tr>
|
|
421
|
+
<tr>
|
|
422
|
+
<td valign="top"><a href="examples/modality/audio.py">Audio</a></td>
|
|
423
|
+
<td>Audio(path="a.wav")<br>Audio(bytes=wav)<br>Audio(array=wave, sampling_rate=16000)</td>
|
|
424
|
+
<td>audio["array"]<br>audio["sampling_rate"]</td>
|
|
425
|
+
</tr>
|
|
426
|
+
<tr>
|
|
427
|
+
<td valign="top"><a href="examples/modality/video.py">Video</a></td>
|
|
428
|
+
<td>Video(path="c.mp4")<br>Video(bytes=mp4)<br>Video(array=frames, fps=25)</td>
|
|
429
|
+
<td>video.get_frames_at(0)<br>video.get_frames_in_range(0, 8)</td>
|
|
430
|
+
</tr>
|
|
431
|
+
<tr>
|
|
432
|
+
<td colspan="3"><strong>File</strong></td>
|
|
433
|
+
</tr>
|
|
434
|
+
<tr>
|
|
435
|
+
<td valign="top"><a href="examples/modality/file.py">File</a></td>
|
|
436
|
+
<td>File(path="doc.bin")<br>File(bytes=blob)</td>
|
|
437
|
+
<td>sidecar # raw bytes</td>
|
|
438
|
+
</tr>
|
|
439
|
+
<tr>
|
|
440
|
+
<td valign="top"><a href="examples/modality/pdf.py">Pdf</a></td>
|
|
441
|
+
<td>Pdf(path="p.pdf")<br>Pdf(pdf=pdfplumber_doc)</td>
|
|
442
|
+
<td>pdf.pages[0] # Pdfplumber</td>
|
|
443
|
+
</tr>
|
|
444
|
+
<tr>
|
|
445
|
+
<td colspan="3"><strong>3D and volume</strong></td>
|
|
446
|
+
</tr>
|
|
447
|
+
<tr>
|
|
448
|
+
<td valign="top"><a href="examples/modality/mesh.py">Mesh</a></td>
|
|
449
|
+
<td>Mesh(path="m.glb")<br>Mesh(mesh=trimesh_obj, file_type="glb")</td>
|
|
450
|
+
<td>mesh.vertices # Trimesh</td>
|
|
451
|
+
</tr>
|
|
452
|
+
<tr>
|
|
453
|
+
<td valign="top"><a href="examples/modality/nifti.py">Nifti</a></td>
|
|
454
|
+
<td>Nifti(path="v.nii.gz")<br>Nifti(array=vol, affine=np.eye(4))</td>
|
|
455
|
+
<td>nifti.get_fdata() # Nibabel</td>
|
|
456
|
+
</tr>
|
|
457
|
+
<tr>
|
|
458
|
+
<td colspan="3"><strong>Array and Graph</strong></td>
|
|
459
|
+
</tr>
|
|
460
|
+
<tr>
|
|
461
|
+
<td valign="top"><a href="examples/modality/numpy_array.py">Numpy</a></td>
|
|
462
|
+
<td>np.load("a.npy")<br>np.zeros((3, 4, 4))</td>
|
|
463
|
+
<td>array # NumPy</td>
|
|
464
|
+
</tr>
|
|
465
|
+
<tr>
|
|
466
|
+
<td valign="top"><a href="examples/modality/tensor.py">Tensor</a></td>
|
|
467
|
+
<td>Tensor(array=torch.randn(3, 4, 4))</td>
|
|
468
|
+
<td>feat # Tensor — 1-D token ids use TokensLoader under Text</td>
|
|
469
|
+
</tr>
|
|
470
|
+
<tr>
|
|
471
|
+
<td valign="top"><a href="examples/modality/graph.py">Graph</a></td>
|
|
472
|
+
<td>Data(x=…, edge_index=…, y=…)<br>Graph(x=…, edge_index=…, y=…)<br>Graph(data=pyg_data)</td>
|
|
473
|
+
<td>graph.x, graph.edge_index # PyG Data or Graph — <a href="#pyg-graphs">PyG graphs</a></td>
|
|
474
|
+
</tr>
|
|
475
|
+
<tr>
|
|
476
|
+
<td colspan="3"><strong>Parquet</strong></td>
|
|
477
|
+
</tr>
|
|
478
|
+
<tr>
|
|
479
|
+
<td valign="top"><a href="examples/modality/parquet.py">Parquet</a></td>
|
|
480
|
+
<td>folder of .parquet files<br>StreamingDataset(..., item_loader=ParquetLoader())</td>
|
|
481
|
+
<td>row["col"] # dict of columns — <a href="#stream-parquet">stream parquet</a></td>
|
|
482
|
+
</tr>
|
|
483
|
+
</table>
|
|
484
|
+
|
|
485
|
+
Examples (path on disk → optimize → batch): [examples/modality](examples/modality).
|
|
486
|
+
|
|
487
|
+
----
|
|
488
|
+
|
|
353
489
|
# Key Features
|
|
354
490
|
|
|
355
491
|
## Features for optimizing and streaming datasets for model training
|
|
@@ -565,24 +701,18 @@ How you return images from `optimize` controls storage size and streaming speed.
|
|
|
565
701
|
|
|
566
702
|
| What you return | Serializer | Result |
|
|
567
703
|
|-----------------|------------|--------|
|
|
568
|
-
| `
|
|
569
|
-
|
|
|
704
|
+
| `litdata.Image(path=...)` / `Image(array=..., quality=95, format="jpeg")` | `image` | Compressed bytes — **preferred** |
|
|
705
|
+
| `litdata.Jpeg(path=...)` / `Jpeg(array=..., quality=95)` | `jpeg` | JPEG bytes |
|
|
706
|
+
| `PIL.JpegImageFile` (e.g. `PIL.Image.open("x.jpg")`) | `jpeg` | Compressed bytes |
|
|
707
|
+
| Plain `PIL.Image` / `Image.fromarray(...)` | `pil` | Uncompressed pixels — often **10×+ larger** |
|
|
570
708
|
|
|
571
|
-
**Best practice:**
|
|
709
|
+
**Best practice:** wrap with `Image` / `Jpeg` at **quality ≈ 95**, or keep existing `.jpg` files via `Image(path=...)`. Resize when helpful.
|
|
572
710
|
|
|
573
711
|
```python
|
|
574
|
-
import io
|
|
575
|
-
from PIL import Image
|
|
576
712
|
import litdata as ld
|
|
577
713
|
|
|
578
714
|
def load_image(path):
|
|
579
|
-
|
|
580
|
-
if not str(path).lower().endswith((".jpg", ".jpeg")):
|
|
581
|
-
buf = io.BytesIO()
|
|
582
|
-
img.convert("RGB").save(buf, format="JPEG", quality=95)
|
|
583
|
-
buf.seek(0)
|
|
584
|
-
img = Image.open(buf) # JpegImageFile
|
|
585
|
-
return {"image": img, "path": path}
|
|
715
|
+
return {"image": ld.Image(path=path, quality=95, format="jpeg"), "id": path}
|
|
586
716
|
|
|
587
717
|
if __name__ == "__main__":
|
|
588
718
|
ld.optimize(fn=load_image, inputs=list_of_paths, output_dir="fast_data", chunk_bytes="64MB", num_workers=8)
|
|
@@ -596,9 +726,9 @@ Ready-made ImageNet optimize/stream scripts: `benchmarks/litdata/` (`--write_mod
|
|
|
596
726
|
<summary> ✅ Custom serializers <a id="serializers" href="#serializers">🔗</a> </summary>
|
|
597
727
|
|
|
598
728
|
|
|
599
|
-
LitData serializes each leaf
|
|
729
|
+
LitData serializes each **pytree leaf** with a pluggable registry. Built-ins (tried in order) include: `str`, `bool`, `int`, `float`, `video`, `audio`, `image`, `nifti`, `mesh`, `pdf`, `tifffile`, `file`, `pil`, `jpeg`, `jpeg_array`, `bytes`, `numpy` / `tensor` (and no-header variants), `graph`, and `pickle` (fallback).
|
|
600
730
|
|
|
601
|
-
For images,
|
|
731
|
+
Prefer [typed media wrappers](#media-types) (`Audio`, `Video`, `Image`, `Graph`, …) so a filepath is not confused with a caption. For images, `Image(..., quality=95, format="jpeg")` or a `JpegImageFile` stores JPEG; a plain `PIL.Image` selects **`pil`** (raw pixels). See [Optimize images as JPEG](#optimize-jpeg).
|
|
602
732
|
|
|
603
733
|
Pass custom serializers when **streaming** (and when using the lower-level `Cache` writer):
|
|
604
734
|
|
|
@@ -622,7 +752,133 @@ dataset = StreamingDataset(
|
|
|
622
752
|
)
|
|
623
753
|
```
|
|
624
754
|
|
|
625
|
-
Keys you pass are tried before the defaults (so they win over `pickle`). `optimize()` uses the built-in registry based on the Python types your `fn` returns — prefer JPEG / numpy / tensor leaves for best results.
|
|
755
|
+
Keys you pass are tried before the defaults (so they win over `pickle`). `optimize()` uses the built-in registry based on the Python types your `fn` returns — prefer typed wrappers / JPEG / numpy / tensor leaves for best results.
|
|
756
|
+
|
|
757
|
+
</details>
|
|
758
|
+
|
|
759
|
+
<details>
|
|
760
|
+
<summary> ✅ Stream PyG graphs <a id="pyg-graphs" href="#pyg-graphs">🔗</a> </summary>
|
|
761
|
+
|
|
762
|
+
|
|
763
|
+
Store [PyTorch Geometric](https://pytorch-geometric.readthedocs.io/) `Data` / `HeteroData` as packed tensors (`to_dict()`), not `torch.save` / pickle. On read, LitData reconstructs with `from_dict` when `torch-geometric` is installed. `optimize` needs a **top-level** function (spawn).
|
|
764
|
+
|
|
765
|
+
`StreamingDataLoader` uses `litdata_collate` by default: graph samples become a `DataBatch` (`Batch.from_data_list`); everything else uses PyTorch `default_collate`. For `follow_batch` / `exclude_keys`, use `torch_geometric.loader.DataLoader`. Without PyG, graph batches stay a list of `Graph`.
|
|
766
|
+
|
|
767
|
+
### Homogeneous `Data` + GCN
|
|
768
|
+
|
|
769
|
+
```python
|
|
770
|
+
import torch
|
|
771
|
+
import torch.nn.functional as F
|
|
772
|
+
from torch_geometric.data import Data
|
|
773
|
+
from torch_geometric.nn import GCNConv, global_mean_pool
|
|
774
|
+
|
|
775
|
+
from litdata import StreamingDataLoader, StreamingDataset, optimize
|
|
776
|
+
|
|
777
|
+
def make_graph(i: int) -> Data:
|
|
778
|
+
n = 8 + i % 5
|
|
779
|
+
src = torch.randint(0, n, (12,), dtype=torch.long)
|
|
780
|
+
dst = torch.randint(0, n, (12,), dtype=torch.long)
|
|
781
|
+
return Data(
|
|
782
|
+
x=torch.randn(n, 8),
|
|
783
|
+
edge_index=torch.stack([src, dst], 0),
|
|
784
|
+
y=torch.tensor(i % 3),
|
|
785
|
+
train_mask=torch.ones(n, dtype=torch.bool),
|
|
786
|
+
num_nodes=n,
|
|
787
|
+
)
|
|
788
|
+
|
|
789
|
+
optimize(make_graph, inputs=list(range(1024)), output_dir="graphs", chunk_size=64)
|
|
790
|
+
|
|
791
|
+
dataset = StreamingDataset("graphs")
|
|
792
|
+
sample = dataset[0] # Data when PyG is installed, else Graph
|
|
793
|
+
loader = StreamingDataLoader(dataset, batch_size=32, shuffle=True)
|
|
794
|
+
batch = next(iter(loader)) # DataBatch
|
|
795
|
+
|
|
796
|
+
class Net(torch.nn.Module):
|
|
797
|
+
def __init__(self):
|
|
798
|
+
super().__init__()
|
|
799
|
+
self.conv = GCNConv(8, 16)
|
|
800
|
+
self.lin = torch.nn.Linear(16, 3)
|
|
801
|
+
|
|
802
|
+
def forward(self, data):
|
|
803
|
+
x = F.relu(self.conv(data.x, data.edge_index))
|
|
804
|
+
return self.lin(global_mean_pool(x, data.batch))
|
|
805
|
+
```
|
|
806
|
+
|
|
807
|
+
### `Graph` wrapper (no PyG at write time)
|
|
808
|
+
|
|
809
|
+
```python
|
|
810
|
+
from litdata import Graph, optimize
|
|
811
|
+
|
|
812
|
+
def make_graph(i: int) -> Graph:
|
|
813
|
+
n = 6
|
|
814
|
+
return Graph(
|
|
815
|
+
x=torch.randn(n, 4),
|
|
816
|
+
edge_index=torch.tensor([[0, 1, 2], [1, 2, 0]], dtype=torch.long),
|
|
817
|
+
y=torch.tensor(i % 2),
|
|
818
|
+
data={"num_nodes": n}, # extra tensors/scalars; field kwargs override data=
|
|
819
|
+
)
|
|
820
|
+
|
|
821
|
+
optimize(make_graph, inputs=list(range(256)), output_dir="graphs")
|
|
822
|
+
# later: sample.to_pyg() if the stream returned Graph
|
|
823
|
+
```
|
|
824
|
+
|
|
825
|
+
`Graph(data=pyg_data)` uses `pyg_data.to_dict()`. Do not mix tensor fields with an opaque NetworkX `data=`.
|
|
826
|
+
|
|
827
|
+
### Heterogeneous `HeteroData`
|
|
828
|
+
|
|
829
|
+
```python
|
|
830
|
+
from torch_geometric.data import HeteroData
|
|
831
|
+
|
|
832
|
+
from litdata import StreamingDataLoader, StreamingDataset, optimize
|
|
833
|
+
|
|
834
|
+
def make_hetero(i: int) -> HeteroData:
|
|
835
|
+
data = HeteroData()
|
|
836
|
+
data["paper"].x = torch.randn(8, 16)
|
|
837
|
+
data["author"].x = torch.randn(4, 8)
|
|
838
|
+
data["author", "writes", "paper"].edge_index = torch.tensor(
|
|
839
|
+
[[0, 1, 2, 3], [0, 2, 4, 6]], dtype=torch.long
|
|
840
|
+
)
|
|
841
|
+
data.y = torch.tensor(i % 3)
|
|
842
|
+
return data
|
|
843
|
+
|
|
844
|
+
optimize(make_hetero, inputs=list(range(512)), output_dir="hetero", chunk_size=32)
|
|
845
|
+
|
|
846
|
+
dataset = StreamingDataset("hetero")
|
|
847
|
+
sample = dataset[0] # HeteroData
|
|
848
|
+
print(sample["paper"].x.shape, sample["author", "writes", "paper"].edge_index.shape)
|
|
849
|
+
|
|
850
|
+
loader = StreamingDataLoader(dataset, batch_size=16)
|
|
851
|
+
batch = next(iter(loader)) # HeteroDataBatch
|
|
852
|
+
# batch["paper"].x, batch["paper"].batch, batch["author", "writes", "paper"].edge_index
|
|
853
|
+
```
|
|
854
|
+
|
|
855
|
+
### Graph plus metadata in one sample
|
|
856
|
+
|
|
857
|
+
```python
|
|
858
|
+
def make_row(i: int) -> dict:
|
|
859
|
+
return {"id": i, "graph": make_graph(i)}
|
|
860
|
+
|
|
861
|
+
optimize(make_row, inputs=list(range(1024)), output_dir="rows")
|
|
862
|
+
loader = StreamingDataLoader(StreamingDataset("rows"), batch_size=8)
|
|
863
|
+
batch = next(iter(loader))
|
|
864
|
+
# batch["id"] is a tensor; batch["graph"] is a DataBatch
|
|
865
|
+
```
|
|
866
|
+
|
|
867
|
+
### Sample subgraphs first, then stream
|
|
868
|
+
|
|
869
|
+
`NeighborLoader` needs one in-memory graph. To stream, run the sampler in `optimize` and store each subgraph as a `Data`:
|
|
870
|
+
|
|
871
|
+
```python
|
|
872
|
+
# sampler = NeighborSampler(big_graph, num_neighbors=[10, 10])
|
|
873
|
+
|
|
874
|
+
def sample_seed(seed: int) -> Data:
|
|
875
|
+
out = sampler.sample_from_nodes(torch.tensor([seed]))
|
|
876
|
+
return Data(x=out.x, edge_index=out.edge_index, y=out.y)
|
|
877
|
+
|
|
878
|
+
optimize(sample_seed, inputs=train_seeds.tolist(), output_dir="subgraphs")
|
|
879
|
+
```
|
|
880
|
+
|
|
881
|
+
NetworkX (or any non-tensor object) uses `Graph(data=nx_graph)` → `graph:pickle`. Do not `torch.save` a graph into the sample.
|
|
626
882
|
|
|
627
883
|
</details>
|
|
628
884
|
|
|
@@ -872,7 +1128,7 @@ Rough ImageNet order-of-magnitude on a Studio (not hard guarantees; right tuning
|
|
|
872
1128
|
| `drop_last` | `True` if distributed else `False` | Equal length across ranks |
|
|
873
1129
|
| `seed` | `42` | Shuffle / subsample RNG |
|
|
874
1130
|
| `serializers` | built-ins | Custom serialize/deserialize map |
|
|
875
|
-
| `max_cache_size` | `
|
|
1131
|
+
| `max_cache_size` | `None` | Evict consumed chunks beyond this size. Default: 75% of free disk, leaving ≥50GB when possible. Pin with `"100G"` / `"50GB"`, a fraction (`0.90`), or `MAX_CACHE_SIZE`. |
|
|
876
1132
|
| `max_pre_download` | `2` | Chunks each worker may prefetch (raise for throughput; watch disk / RAM) |
|
|
877
1133
|
| `subsample` | `1.0` | Fraction of data (`0.01`) or upsample (`2.5`) |
|
|
878
1134
|
| `encryption` | `None` | `FernetEncryption` / `RSAEncryption` / custom |
|
|
@@ -893,7 +1149,8 @@ On **Vast / NFS / local disk**, POSIX-fast is on by default (`LITDATA_POSIX_FAST
|
|
|
893
1149
|
| All usual `torch.utils.data.DataLoader` kwargs | `batch_size`, `num_workers`, `collate_fn`, `pin_memory`, … |
|
|
894
1150
|
| `shuffle` / `drop_last` | Forwarded to the streaming dataset |
|
|
895
1151
|
| `profile_batches` | `int` / `True` / `False` — viztracer worker trace (see [Profile data loading](#profile-loading)) |
|
|
896
|
-
| `
|
|
1152
|
+
| `profile_cprofile` | `True` — stdlib cProfile of the main process + worker 0 (see [Profile data loading](#profile-loading)) |
|
|
1153
|
+
| `profile_skip_batches` / `profile_dir` | Warm-up skip count; output dir for viztracer / cProfile files |
|
|
897
1154
|
| `multiprocessing_context` | Use **`"spawn"`** (or `"forkserver"`) with `ParquetLoader` + `num_workers>0` on Linux |
|
|
898
1155
|
|
|
899
1156
|
Prefer `StreamingDataLoader` over a plain PyTorch `DataLoader` for optimized / combined / parallel datasets (resume + correct batch metadata).
|
|
@@ -1008,16 +1265,15 @@ Local `output_dir` writes chunks in place. Remote inputs and outputs use the str
|
|
|
1008
1265
|
|
|
1009
1266
|
```python
|
|
1010
1267
|
import numpy as np
|
|
1011
|
-
from PIL import Image
|
|
1012
1268
|
import litdata as ld
|
|
1013
1269
|
|
|
1014
1270
|
def random_images(index):
|
|
1015
|
-
|
|
1016
|
-
|
|
1017
|
-
|
|
1018
|
-
|
|
1019
|
-
|
|
1020
|
-
|
|
1271
|
+
array = np.random.randint(0, 256, (32, 32, 3), dtype=np.uint8)
|
|
1272
|
+
return {
|
|
1273
|
+
"index": index,
|
|
1274
|
+
"image": ld.Image(array=array, quality=95, format="jpeg"),
|
|
1275
|
+
"class": np.random.randint(10),
|
|
1276
|
+
}
|
|
1021
1277
|
|
|
1022
1278
|
if __name__ == "__main__":
|
|
1023
1279
|
# The optimize function writes data in an optimized format.
|
|
@@ -1402,15 +1658,15 @@ Merge multiple optimized datasets into one.
|
|
|
1402
1658
|
|
|
1403
1659
|
```python
|
|
1404
1660
|
import numpy as np
|
|
1405
|
-
from PIL import Image
|
|
1406
1661
|
|
|
1407
|
-
from litdata import StreamingDataset, merge_datasets, optimize
|
|
1662
|
+
from litdata import Image, StreamingDataset, merge_datasets, optimize
|
|
1408
1663
|
|
|
1409
1664
|
|
|
1410
1665
|
def random_images(index):
|
|
1666
|
+
array = np.random.randint(0, 256, (32, 32, 3), dtype=np.uint8)
|
|
1411
1667
|
return {
|
|
1412
1668
|
"index": index,
|
|
1413
|
-
"image": Image
|
|
1669
|
+
"image": Image(array=array, quality=95, format="jpeg"),
|
|
1414
1670
|
"class": np.random.randint(10),
|
|
1415
1671
|
}
|
|
1416
1672
|
|
|
@@ -1427,6 +1683,16 @@ if __name__ == "__main__":
|
|
|
1427
1683
|
print(len(dataset))
|
|
1428
1684
|
# out: 1000
|
|
1429
1685
|
```
|
|
1686
|
+
|
|
1687
|
+
If you wrote chunks yourself (`Cache` / `BinaryWriter`) and only have `{rank}.index.json` shards, finish the dataset:
|
|
1688
|
+
|
|
1689
|
+
```python
|
|
1690
|
+
from litdata import complete_dataset, StreamingDataset
|
|
1691
|
+
|
|
1692
|
+
complete_dataset("my_chunks") # no-op if index.json already exists
|
|
1693
|
+
StreamingDataset("my_chunks") # also tries this automatically
|
|
1694
|
+
```
|
|
1695
|
+
|
|
1430
1696
|
</details>
|
|
1431
1697
|
|
|
1432
1698
|
<details>
|
|
@@ -1512,6 +1778,13 @@ print(test_dataset)
|
|
|
1512
1778
|
# out: 50,000
|
|
1513
1779
|
```
|
|
1514
1780
|
|
|
1781
|
+
Or pick exact indices with `StreamingDataset.subset`:
|
|
1782
|
+
|
|
1783
|
+
```python
|
|
1784
|
+
train = dataset.subset(range(0, 30_000))
|
|
1785
|
+
# dataset.subset(slice(0, 1000))
|
|
1786
|
+
```
|
|
1787
|
+
|
|
1515
1788
|
</details>
|
|
1516
1789
|
|
|
1517
1790
|
<details>
|
|
@@ -1528,6 +1801,9 @@ dataset = StreamingDataset("s3://my-bucket/my-data", subsample=0.01) # data are
|
|
|
1528
1801
|
|
|
1529
1802
|
print(len(dataset)) # display the length of your data
|
|
1530
1803
|
# out: 1000
|
|
1804
|
+
|
|
1805
|
+
# or a list / slice of global indices
|
|
1806
|
+
small = dataset.subset([0, 10, 20])
|
|
1531
1807
|
```
|
|
1532
1808
|
|
|
1533
1809
|
</details>
|
|
@@ -1644,7 +1920,7 @@ ld.index_parquet_dataset(
|
|
|
1644
1920
|
|
|
1645
1921
|
### Stream with `ParquetLoader`
|
|
1646
1922
|
|
|
1647
|
-
|
|
1923
|
+
If the folder looks like parquet and has no `index.json`, `StreamingDataset` now **builds the index automatically**. You still need `ParquetLoader` for local/S3/GCS (it must match `index.json`). `hf://` already auto-indexes and selects the loader.
|
|
1648
1924
|
|
|
1649
1925
|
```python
|
|
1650
1926
|
import litdata as ld
|
|
@@ -1840,7 +2116,32 @@ Only **worker 0** is instrumented. When an `int` is used, the tracer wraps `fetc
|
|
|
1840
2116
|
|
|
1841
2117
|
- Delete or change `profile_dir` between runs — LitData removes an existing `result.json` before starting.
|
|
1842
2118
|
- Pair with a wiped chunk cache if you care about **cold** epoch behavior (`litdata cache clear`).
|
|
1843
|
-
|
|
2119
|
+
### cProfile (main + worker 0)
|
|
2120
|
+
|
|
2121
|
+
Stdlib statistical profile of **the parent process** (queue wait, unpickle, collate handoff) and **worker 0** (fetch, decode, transforms). No extra package.
|
|
2122
|
+
|
|
2123
|
+
```python
|
|
2124
|
+
loader = StreamingDataLoader(
|
|
2125
|
+
dataset,
|
|
2126
|
+
batch_size=64,
|
|
2127
|
+
num_workers=4,
|
|
2128
|
+
profile_cprofile=True,
|
|
2129
|
+
profile_dir="./profiles",
|
|
2130
|
+
)
|
|
2131
|
+
|
|
2132
|
+
for batch in loader:
|
|
2133
|
+
train_step(batch)
|
|
2134
|
+
# writes profiles/cprofile_main.prof + cprofile_worker0.prof (and .txt summaries)
|
|
2135
|
+
```
|
|
2136
|
+
|
|
2137
|
+
```bash
|
|
2138
|
+
python -m pstats profiles/cprofile_worker0.prof
|
|
2139
|
+
# then: sort tottime stats 30
|
|
2140
|
+
```
|
|
2141
|
+
|
|
2142
|
+
Do not set `profile_cprofile` and `profile_batches` together — both install a `sys.setprofile` hook. With `num_workers=0` only the main file is written. The parent profiler starts **after** workers spawn so fork does not inherit an active cProfile.
|
|
2143
|
+
|
|
2144
|
+
- For deeper LitData internals (download / read / delete timeline), use `enable_tracer()` + [Litracer](https://github.com/Lightning-AI/litracer) instead — see [Debug & Profile LitData](#debug-profile). That path is complementary: viztracer = DataLoader worker CPU timeline; cProfile = function self/cum time; Litracer = LitData pipeline events.
|
|
1844
2145
|
|
|
1845
2146
|
</details>
|
|
1846
2147
|
|
|
@@ -1890,7 +2191,9 @@ outputs = optimize(
|
|
|
1890
2191
|
|
|
1891
2192
|
Control how much disk the local chunk cache may use. Downloaded chunks are deleted after use once the cache exceeds the limit.
|
|
1892
2193
|
|
|
1893
|
-
Default `max_cache_size` is **`
|
|
2194
|
+
Default `max_cache_size` is **`None`**: LitData uses **75% of currently free disk** and still leaves **≥50GB** free when the volume is large enough for checkpoints. Pass `"100G"` / `"50GB"` for a fixed budget, or a float (`0.90`) for that fraction of currently free space. `MAX_CACHE_SIZE` overrides the constructor (size or fraction).
|
|
2195
|
+
|
|
2196
|
+
Peak disk in flight is roughly:
|
|
1894
2197
|
|
|
1895
2198
|
```
|
|
1896
2199
|
num_workers × max_pre_download × mean_chunk_size
|
|
@@ -1930,7 +2233,9 @@ for batch in StreamingDataLoader(dataset, batch_size=64, num_workers=8):
|
|
|
1930
2233
|
| `LITDATA_ASYNC_CHUNK_PREFETCH=1` | Force on |
|
|
1931
2234
|
| `LITDATA_ASYNC_CHUNK_PREFETCH=0` | Force off |
|
|
1932
2235
|
|
|
1933
|
-
When async is on, LitData raises `max_pre_download` to at least **4** so `asyncio.gather` has enough in-flight downloads (override with `LITDATA_ASYNC_MIN_PRE_DOWNLOAD`; set `0` to disable the floor). Peak disk ≈ `num_workers × max_pre_download × chunk_size` — size `max_cache_size` accordingly.
|
|
2236
|
+
When async is on, LitData raises `max_pre_download` to at least **4** so `asyncio.gather` has enough in-flight downloads (override with `LITDATA_ASYNC_MIN_PRE_DOWNLOAD`; set `0` to disable the floor). Concurrent GETs are capped by `LITDATA_ASYNC_DOWNLOAD_CONCURRENCY` (default **8**). The prepare thread drains the prefetch queue up to that gather width whenever a cache slot is free. Peak disk ≈ `num_workers × max_pre_download × chunk_size` — size `max_cache_size` accordingly.
|
|
2237
|
+
|
|
2238
|
+
The reader does **not** poll the cache directory for each chunk. After a download finishes, the downloader atomically `os.replace`s the temp file and sets an in-process `Event`. Compressed chunks set that Event only after decompress publishes the readable `.bin`. Other DataLoader workers still fall back to a short filesystem check (Events are per process).
|
|
1934
2239
|
|
|
1935
2240
|
```bash
|
|
1936
2241
|
# Debugging download/delete races — force synchronous downloads
|
|
@@ -1938,6 +2243,9 @@ export LITDATA_ASYNC_CHUNK_PREFETCH=0
|
|
|
1938
2243
|
|
|
1939
2244
|
# Keep max_pre_download=2 even with async enabled
|
|
1940
2245
|
export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
|
|
2246
|
+
|
|
2247
|
+
# Cap overlapping remote GETs (default 8)
|
|
2248
|
+
export LITDATA_ASYNC_DOWNLOAD_CONCURRENCY=4
|
|
1941
2249
|
```
|
|
1942
2250
|
|
|
1943
2251
|
### Common environment variables
|
|
@@ -1947,10 +2255,12 @@ export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
|
|
|
1947
2255
|
| `LITDATA_CACHE_DIR` | `~/.lightning/chunks` | Default chunk cache directory |
|
|
1948
2256
|
| `LITDATA_ASYNC_CHUNK_PREFETCH` | on for remote | `0`/`1` force async chunk download overlap |
|
|
1949
2257
|
| `LITDATA_ASYNC_MIN_PRE_DOWNLOAD` | `4` | Floor for `max_pre_download` when async is on (`0` = no floor) |
|
|
2258
|
+
| `LITDATA_ASYNC_DOWNLOAD_CONCURRENCY` | `8` | Max in-flight chunk GETs per gather |
|
|
1950
2259
|
| `LITDATA_OBSTORE_STREAM_MIN_CHUNK_MIB` | `8` | S3 obstore stream chunk size (MiB) |
|
|
1951
2260
|
| `MAX_WAIT_TIME` | `120` | Seconds to wait for a chunk before error |
|
|
1952
2261
|
| `FORCE_DOWNLOAD_TIME` | `30` | Seconds before force re-download of a missing chunk |
|
|
1953
|
-
| `
|
|
2262
|
+
| `LITDATA_CHECK_UPDATES` | unset | `1` enables the PyPI upgrade tip (off by default) |
|
|
2263
|
+
| `LITDATA_DISABLE_VERSION_CHECK` | on unless updates enabled | `1` skips the upgrade tip |
|
|
1954
2264
|
| `HF_TOKEN` | — | Gated Hugging Face datasets |
|
|
1955
2265
|
| `DEBUG_LITDATA` / `PRINT_DEBUG_LOGS` | `0` | Internal debug / stdout logs |
|
|
1956
2266
|
| `LITDATA_LOG_FILE` | `litdata_debug.log` | `enable_tracer()` output path |
|
|
@@ -2457,26 +2767,12 @@ Speed to stream Imagenet 1.2M from other cloud storage providers:
|
|
|
2457
2767
|
|---|---|---|---|
|
|
2458
2768
|
| Cloudflare R2 | LitData | **5335** | **5630** |
|
|
2459
2769
|
|
|
2460
|
-
Speed to stream
|
|
2461
|
-
| Framework | Dataset Mode | Dataset Size @ 256px | Images / sec 1st Epoch (float32) | Images / sec 2nd Epoch (float32) |
|
|
2462
|
-
|---|---|---|---|---|
|
|
2463
|
-
| LitData | PIL RAW | 168 GB | 6647 | 6398 |
|
|
2464
|
-
| LitData | JPEG 90% | 12 GB | 6553 | 6537 |
|
|
2465
|
-
| ffcv (os_cache=True) | RAW | 170 GB | 7263 | 6698 |
|
|
2466
|
-
| ffcv (os_cache=False) | RAW | 170 GB | 7556 | 8169 |
|
|
2467
|
-
| ffcv(os_cache=True) | JPEG 90% | 20 GB | 7653 | 8051 |
|
|
2468
|
-
| ffcv(os_cache=False) | JPEG 90% | 20 GB | 8149 | 8607 |
|
|
2469
|
-
|
|
2470
|
-
Speed to stream a **synthetic ImageNet-scale set from Vast NFS** (NFSv3 `nconnect=32`, 208-CPU host, ~1 TiB RAM). Dataset: **1.08M** JPEG q95 256×256 (~160 GiB, 64 MiB chunks). `StreamingDataLoader`, batch **256**, `shuffle=True`, `drop_last=True`, decode only unless noted. POSIX-fast mmaps chunks **in place** (no copy into `~/.lightning/chunks`).
|
|
2471
|
-
|
|
2472
|
-
| Setup | Workers | Images / sec |
|
|
2473
|
-
|---|---|---|
|
|
2474
|
-
| Copy into local cache (`LITDATA_POSIX_FAST=0`) | 48 | **16.7k** (2-epoch avg) |
|
|
2475
|
-
| POSIX-fast (this default on local/Vast paths) | 48 | **18.2k** |
|
|
2476
|
-
| POSIX-fast + README ImageNet augs (crop 224, flip, float32) | 48 | **12.9k** |
|
|
2477
|
-
| POSIX-fast, all CPU cores | **208** | **35.8k** |
|
|
2770
|
+
Speed to stream a synthetic ImageNet-scale set from **Vast NFS** with POSIX-fast (mmap in place, decode only, no transforms):
|
|
2478
2771
|
|
|
2479
|
-
|
|
2772
|
+
| Workers | Images / sec |
|
|
2773
|
+
|---|---|
|
|
2774
|
+
| 48 | **18.2k** |
|
|
2775
|
+
| 208 | **35.8k** |
|
|
2480
2776
|
|
|
2481
2777
|
### Raw Dataset
|
|
2482
2778
|
|
|
@@ -2584,6 +2880,80 @@ Below are templates for real-world applications of LitData at scale.
|
|
|
2584
2880
|
|
|
2585
2881
|
----
|
|
2586
2882
|
|
|
2883
|
+
# Used by
|
|
2884
|
+
|
|
2885
|
+
<table width="100%">
|
|
2886
|
+
<tr>
|
|
2887
|
+
<th align="left" width="18%">Project</th>
|
|
2888
|
+
<th align="left">Description</th>
|
|
2889
|
+
</tr>
|
|
2890
|
+
<tr>
|
|
2891
|
+
<td valign="top"><a href="https://github.com/sunlabuiuc/PyHealth">PyHealth</a></td>
|
|
2892
|
+
<td>Deep-learning toolkit for clinical prediction (MIMIC, eICU, OMOP, sleep, CXR). <code>set_task()</code> writes processed samples with LitData; <code>SampleDataset</code> subclasses <code>StreamingDataset</code> so training streams chunked EHR tensors instead of holding the cohort in RAM.</td>
|
|
2893
|
+
</tr>
|
|
2894
|
+
<tr>
|
|
2895
|
+
<td valign="top"><a href="https://github.com/prescient-design/lobster">LBSTER</a></td>
|
|
2896
|
+
<td>Protein and biological-sequence language models from Prescient Design (Genentech). Pre-training and concept-bottleneck models (fitness, embeddings, guided generation) stream large sequence corpora through LitData.</td>
|
|
2897
|
+
</tr>
|
|
2898
|
+
<tr>
|
|
2899
|
+
<td valign="top"><a href="https://github.com/OpenSynth-energy/OpenSynth">OpenSynth</a></td>
|
|
2900
|
+
<td>Open toolkit for synthetic smart-meter / energy time series. Generated or historical meter traces are optimized and streamed for model training.</td>
|
|
2901
|
+
</tr>
|
|
2902
|
+
<tr>
|
|
2903
|
+
<td valign="top"><a href="https://github.com/BiomedSciAI/biomed-multi-view">biomed-multi-view</a></td>
|
|
2904
|
+
<td>IBM BiomedSciAI multi-view biomedical models. LitData is used to cache and stream paired modalities during training.</td>
|
|
2905
|
+
</tr>
|
|
2906
|
+
<tr>
|
|
2907
|
+
<td valign="top"><a href="https://github.com/cma2015/DEM">DEM</a></td>
|
|
2908
|
+
<td>Phenotype and gene-mining pipeline (<code>biodem</code>). Large genomic / trait tables are packed into LitData chunks for repeated training passes.</td>
|
|
2909
|
+
</tr>
|
|
2910
|
+
<tr>
|
|
2911
|
+
<td valign="top"><a href="https://pypi.org/project/deeptan/">deeptan</a></td>
|
|
2912
|
+
<td>Graph multi-task models for multi-omics trait-associated networks. Guide graphs and expression tables are converted to LitData chunks before GNN training.</td>
|
|
2913
|
+
</tr>
|
|
2914
|
+
<tr>
|
|
2915
|
+
<td valign="top"><a href="https://github.com/avitai/datarax">datarax</a></td>
|
|
2916
|
+
<td>Data tooling with an optional cloud-streaming extra that uses LitData to read remote datasets without a full local copy.</td>
|
|
2917
|
+
</tr>
|
|
2918
|
+
<tr>
|
|
2919
|
+
<td valign="top"><a href="https://pypi.org/project/fasr/">fasr</a></td>
|
|
2920
|
+
<td>Speech ASR framework. The LitData extra streams audio and transcripts for training instead of random-access file lists.</td>
|
|
2921
|
+
</tr>
|
|
2922
|
+
</table>
|
|
2923
|
+
|
|
2924
|
+
# Skills <a id="skills"></a>
|
|
2925
|
+
|
|
2926
|
+
Coding agents (Cursor, Claude Code, and others) should load the LitData skill instead of guessing the API.
|
|
2927
|
+
|
|
2928
|
+
```bash
|
|
2929
|
+
npx skills add Lightning-AI/litData
|
|
2930
|
+
```
|
|
2931
|
+
|
|
2932
|
+
Useful options: `-g` (user-global), `-a cursor` (Cursor only), `-y` (non-interactive). In this repo the skill already lives at [`.claude/skills/litdata/`](.claude/skills/litdata/). Installer: [skills CLI](https://github.com/vercel-labs/skills).
|
|
2933
|
+
|
|
2934
|
+
Start at [`SKILL.md`](.claude/skills/litdata/SKILL.md), then load [`reference/using-litdata.md`](.claude/skills/litdata/reference/using-litdata.md) before writing examples.
|
|
2935
|
+
|
|
2936
|
+
| File | When to load |
|
|
2937
|
+
| --- | --- |
|
|
2938
|
+
| [SKILL.md](.claude/skills/litdata/SKILL.md) | Triggers, public API, traps |
|
|
2939
|
+
| [using-litdata.md](.claude/skills/litdata/reference/using-litdata.md) | Optimize / stream / raw / modality cookbook |
|
|
2940
|
+
| [streaming.md](.claude/skills/litdata/reference/streaming.md) | Read path, shuffle, resume, serializers |
|
|
2941
|
+
| [processing.md](.claude/skills/litdata/reference/processing.md) | optimize / map orchestration |
|
|
2942
|
+
| [data-movement.md](.claude/skills/litdata/reference/data-movement.md) | Download / upload / FUSE vs direct I/O |
|
|
2943
|
+
| [multi-node.md](.claude/skills/litdata/reference/multi-node.md) | Studio num_nodes jobs |
|
|
2944
|
+
| [resolver.md](.claude/skills/litdata/reference/resolver.md) | Paths, URLs, Studio mounts |
|
|
2945
|
+
| [storage-format.md](.claude/skills/litdata/reference/storage-format.md) | Chunks, `index.json`, writer / reader |
|
|
2946
|
+
| [cache-and-chunk-lifecycle.md](.claude/skills/litdata/reference/cache-and-chunk-lifecycle.md) | Prefetch and eviction |
|
|
2947
|
+
| [env-vars.md](.claude/skills/litdata/reference/env-vars.md) | LITDATA_* and DATA_OPTIMIZER_* |
|
|
2948
|
+
| [keyed-lookup.md](.claude/skills/litdata/reference/keyed-lookup.md) | key_fn, dataset_update |
|
|
2949
|
+
| [debugging.md](.claude/skills/litdata/reference/debugging.md) | enable_tracer, Litracer |
|
|
2950
|
+
| [benchmarking.md](.claude/skills/litdata/reference/benchmarking.md) | Fair benches |
|
|
2951
|
+
| [lightning-studio.md](.claude/skills/litdata/reference/lightning-studio.md) | Studio env and credentials |
|
|
2952
|
+
| [testing.md](.claude/skills/litdata/reference/testing.md) | Pytest / CI |
|
|
2953
|
+
| [contributing.md](.claude/skills/litdata/reference/contributing.md) | PR / lint path |
|
|
2954
|
+
|
|
2955
|
+
Offline streaming what-if (not the Python package): [simulator/](simulator/) (litsim).
|
|
2956
|
+
|
|
2587
2957
|
# Community
|
|
2588
2958
|
LitData is a community project accepting contributions - Let's make the world's most advanced AI data processing framework.
|
|
2589
2959
|
|
|
@@ -2600,8 +2970,7 @@ LitData is a community project accepting contributions - Let's make the world's
|
|
|
2600
2970
|
author = {Thomas Chaton and Lightning AI},
|
|
2601
2971
|
title = {LitData: Transform datasets at scale. Optimize datasets for fast AI model training.},
|
|
2602
2972
|
year = {2023},
|
|
2603
|
-
howpublished = {\url{https://github.com/Lightning-AI/litdata}}
|
|
2604
|
-
note = {Accessed: 2025-04-09}
|
|
2973
|
+
howpublished = {\url{https://github.com/Lightning-AI/litdata}}
|
|
2605
2974
|
}
|
|
2606
2975
|
```
|
|
2607
2976
|
|
|
@@ -2609,8 +2978,12 @@ LitData is a community project accepting contributions - Let's make the world's
|
|
|
2609
2978
|
|
|
2610
2979
|
## Papers with LitData
|
|
2611
2980
|
|
|
2612
|
-
*
|
|
2613
|
-
|
|
2981
|
+
Papers that train or stream with LitData (`optimize` / `StreamingDataset`). Scholar hits for “litdata streaming” are often weather *lightning* data, or mention LitData only as example source code.
|
|
2982
|
+
|
|
2983
|
+
| Paper | Venue | How LitData is used |
|
|
2984
|
+
|---|---|---|
|
|
2985
|
+
| [Towards Interpretable Protein Structure Prediction with Sparse Autoencoders](https://arxiv.org/abs/2503.08764) ([code](https://github.com/johnyang101/reticular-sae)) | ICLR 2025 GEM | `optimize` shards ESM-2 embeddings; `StreamingDataset` streams from S3 for multi-GPU SAE training |
|
|
2986
|
+
| [TinyLlama: An Open-Source Small Language Model](https://arxiv.org/abs/2401.02385) ([code](https://github.com/jzhang38/TinyLlama)) | arXiv 2024 | 1.1B pretrain on SlimPajama + StarCoder via Lit-GPT’s `lightning.data` stack (now LitData): `CombinedStreamingDataset` + `TokensLoader` |
|
|
2614
2987
|
|
|
2615
2988
|
----
|
|
2616
2989
|
|