litdata 0.2.69__tar.gz → 0.2.70__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. {litdata-0.2.69/src/litdata.egg-info → litdata-0.2.70}/PKG-INFO +414 -75
  2. {litdata-0.2.69 → litdata-0.2.70}/README.md +413 -74
  3. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/__about__.py +1 -1
  4. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/__init__.py +23 -0
  5. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/constants.py +9 -1
  6. litdata-0.2.70/src/litdata/processing/complete.py +54 -0
  7. litdata-0.2.70/src/litdata/processing/media_folder.py +117 -0
  8. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/processing/readers.py +69 -28
  9. litdata-0.2.70/src/litdata/streaming/collate.py +47 -0
  10. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/streaming/config.py +11 -3
  11. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/streaming/dataloader.py +4 -2
  12. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/streaming/dataset.py +7 -1
  13. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/streaming/item_loader.py +57 -25
  14. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/streaming/reader.py +21 -7
  15. litdata-0.2.70/src/litdata/streaming/serializers.py +1668 -0
  16. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/streaming/writer.py +30 -16
  17. litdata-0.2.70/src/litdata/types.py +230 -0
  18. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/utilities/dataset_utilities.py +42 -4
  19. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/utilities/keys_index.py +2 -2
  20. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/utilities/train_test_split.py +54 -0
  21. {litdata-0.2.69 → litdata-0.2.70/src/litdata.egg-info}/PKG-INFO +414 -75
  22. {litdata-0.2.69 → litdata-0.2.70}/src/litdata.egg-info/SOURCES.txt +4 -0
  23. litdata-0.2.69/src/litdata/streaming/serializers.py +0 -595
  24. {litdata-0.2.69 → litdata-0.2.70}/CONTRIBUTING.md +0 -0
  25. {litdata-0.2.69 → litdata-0.2.70}/LICENSE +0 -0
  26. {litdata-0.2.69 → litdata-0.2.70}/MANIFEST.in +0 -0
  27. {litdata-0.2.69 → litdata-0.2.70}/requirements.txt +0 -0
  28. {litdata-0.2.69 → litdata-0.2.70}/setup.cfg +0 -0
  29. {litdata-0.2.69 → litdata-0.2.70}/setup.py +0 -0
  30. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/__main__.py +0 -0
  31. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/cli/__init__.py +0 -0
  32. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/cli/commands.py +0 -0
  33. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/cli/handler/__init__.py +0 -0
  34. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/cli/handler/cache.py +0 -0
  35. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/cli/handler/optimize.py +0 -0
  36. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/cli/parser.py +0 -0
  37. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/debugger.py +0 -0
  38. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/exceptions.py +0 -0
  39. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/helpers.py +0 -0
  40. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/imports.py +0 -0
  41. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/processing/__init__.py +0 -0
  42. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/processing/data_processor.py +0 -0
  43. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/processing/functions.py +0 -0
  44. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/processing/utilities.py +0 -0
  45. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/raw/__init__.py +0 -0
  46. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/raw/dataset.py +0 -0
  47. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/raw/indexer.py +0 -0
  48. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/raw/types.py +0 -0
  49. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/requirements.py +0 -0
  50. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/streaming/__init__.py +0 -0
  51. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/streaming/async_prefetch.py +0 -0
  52. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/streaming/cache.py +0 -0
  53. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/streaming/client.py +0 -0
  54. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/streaming/combined.py +0 -0
  55. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/streaming/compression.py +0 -0
  56. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/streaming/dataset_update.py +0 -0
  57. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/streaming/downloader.py +0 -0
  58. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/streaming/elastic.py +0 -0
  59. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/streaming/fs_provider.py +0 -0
  60. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/streaming/parallel.py +0 -0
  61. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/streaming/posix_fast.py +0 -0
  62. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/streaming/resolver.py +0 -0
  63. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/streaming/sampler.py +0 -0
  64. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/streaming/shuffle.py +0 -0
  65. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/streaming/timing.py +0 -0
  66. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/utilities/__init__.py +0 -0
  67. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/utilities/_pytree.py +0 -0
  68. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/utilities/base.py +0 -0
  69. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/utilities/breakpoint.py +0 -0
  70. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/utilities/broadcast.py +0 -0
  71. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/utilities/encryption.py +0 -0
  72. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/utilities/env.py +0 -0
  73. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/utilities/format.py +0 -0
  74. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/utilities/hf_dataset.py +0 -0
  75. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/utilities/packing.py +0 -0
  76. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/utilities/parquet.py +0 -0
  77. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/utilities/shuffle.py +0 -0
  78. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/utilities/subsample.py +0 -0
  79. {litdata-0.2.69 → litdata-0.2.70}/src/litdata/utilities/torch_utils.py +0 -0
  80. {litdata-0.2.69 → litdata-0.2.70}/src/litdata.egg-info/dependency_links.txt +0 -0
  81. {litdata-0.2.69 → litdata-0.2.70}/src/litdata.egg-info/entry_points.txt +0 -0
  82. {litdata-0.2.69 → litdata-0.2.70}/src/litdata.egg-info/not-zip-safe +0 -0
  83. {litdata-0.2.69 → litdata-0.2.70}/src/litdata.egg-info/requires.txt +0 -0
  84. {litdata-0.2.69 → litdata-0.2.70}/src/litdata.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: litdata
3
- Version: 0.2.69
3
+ Version: 0.2.70
4
4
  Summary: The Deep Learning framework to train, deploy, and ship AI products Lightning fast.
5
5
  Home-page: https://github.com/Lightning-AI/litdata
6
6
  Download-URL: https://github.com/Lightning-AI/litdata
@@ -71,15 +71,31 @@ Dynamic: summary
71
71
   
72
72
   
73
73
 
74
- <pre>
75
- Transform Optimize / Stream
76
-
77
- ✅ Parallelize data processing ✅ Stream raw files with no prep
78
- ✅ Create vector embeddings ✅ Stream large cloud datasets
79
- ✅ Run distributed inference ✅ Accelerate training by 20x
80
- ✅ Scrape websites at scale ✅ Pause and resume data streaming
81
- ✅ Use remote data without local loading
82
- </pre>
74
+ <table>
75
+ <tr>
76
+ <td valign="top" align="left">
77
+
78
+ **Transform**
79
+
80
+ ✅ Parallelize data processing
81
+ ✅ Create vector embeddings
82
+ ✅ Run distributed inference
83
+ ✅ Scrape websites at scale
84
+
85
+ </td>
86
+ <td valign="top" align="left">
87
+
88
+ **Optimize / Stream**
89
+
90
+ ✅ Stream raw files with no prep
91
+ ✅ Stream large cloud datasets
92
+ ✅ Accelerate training by 20x
93
+ ✅ Pause and resume data streaming
94
+ ✅ Use remote data without local loading
95
+
96
+ </td>
97
+ </tr>
98
+ </table>
83
99
 
84
100
  ---
85
101
 
@@ -93,11 +109,14 @@ Transform Optimize / Stream
93
109
  <a href="#quick-start">Quick start</a> •
94
110
  <a href="#speed-up-model-training">Optimize data</a> •
95
111
  <a href="#transform-datasets">Transform data</a> •
112
+ <a href="#modality">Modality</a> •
96
113
  <a href="#key-features">Features</a> •
97
114
  <a href="#stream-raw">Stream raw files</a> •
98
115
  <a href="#resolve-paths">Paths & cloud URLs</a> •
99
116
  <a href="#benchmarks">Benchmarks</a> •
100
117
  <a href="#start-from-a-template">Templates</a> •
118
+ <a href="#used-by">Used by</a> •
119
+ <a href="#skills">Skills</a> •
101
120
  <a href="#community">Community</a>
102
121
  </p>
103
122
 
@@ -156,7 +175,7 @@ On Linux/macOS, `[extras]` includes optional `uvloop` for a faster asyncio event
156
175
  <details>
157
176
  <summary>AI agent skill (Cursor, Claude Code, …)</summary>
158
177
 
159
- Install the LitData expert skill so coding agents know the full API, path resolver, optimize/stream recipes, and internals:
178
+ Install the LitData expert skill so coding agents know the full API, path resolver, optimize/stream recipes, and internals. Full file map → [Skills](#skills).
160
179
 
161
180
  ```bash
162
181
  npx skills add Lightning-AI/litData
@@ -215,24 +234,19 @@ Transform raw data into optimized chunks for maximum streaming speed.
215
234
  This step formats the dataset for fast loading by writing data in an efficient chunked binary format.
216
235
 
217
236
  ```python
218
- import io
219
237
  import numpy as np
220
- from PIL import Image
221
238
  import litdata as ld
222
239
 
223
240
  def random_images(index):
224
- # Replace with your actual image loading (e.g. Image.open("photo.jpg")).
225
- # Prefer JPEG: return a JpegImageFile, or re-encode at quality≈95. Plain
226
- # Image.fromarray(...) stores uncompressed PIL RAW and can be 10×+ larger.
227
- img = Image.fromarray(np.random.randint(0, 256, (32, 32, 3), dtype=np.uint8))
228
- buf = io.BytesIO()
229
- img.convert("RGB").save(buf, format="JPEG", quality=95)
230
- buf.seek(0)
231
- jpeg_image = Image.open(buf) # JpegImageFile → compressed bytes in the chunk
232
- fake_labels = np.random.randint(10)
233
-
234
- # Keys/types must stay stable across samples; list lengths/types fixed
235
- return {"index": index, "image": jpeg_image, "class": fake_labels}
241
+ # Replace with your files: Image(path="photo.jpg") or Image(bytes=...).
242
+ # Wrappers pick the serializer (a caption string is not an image).
243
+ # quality/format encode JPEG — not uncompressed PIL RAW.
244
+ array = np.random.randint(0, 256, (32, 32, 3), dtype=np.uint8)
245
+ return {
246
+ "index": index,
247
+ "image": ld.Image(array=array, quality=95, format="jpeg"),
248
+ "class": np.random.randint(10),
249
+ }
236
250
 
237
251
  if __name__ == "__main__":
238
252
  # Exactly one of chunk_bytes or chunk_size
@@ -350,6 +364,128 @@ ld.map(
350
364
 
351
365
  ----
352
366
 
367
+ # Modality <a id="media-types"></a>
368
+
369
+ Wrap each file so a caption is not treated as a path: Text(path=...), Image(path=...), Audio(path=...). Path and raw bytes are stored as-is; array / image / mesh encode.
370
+
371
+ <table width="100%">
372
+ <tr>
373
+ <th align="left">Type</th>
374
+ <th align="left">Write</th>
375
+ <th align="left">Stream</th>
376
+ </tr>
377
+ <tr>
378
+ <td colspan="3"><strong>Text</strong></td>
379
+ </tr>
380
+ <tr>
381
+ <td valign="top"><a href="examples/modality/text.py">Text</a></td>
382
+ <td>Text(path="a.txt")<br>Text(bytes=utf8)<br>Text(text="a caption")</td>
383
+ <td>text # str</td>
384
+ </tr>
385
+ <tr>
386
+ <td valign="top"><a href="examples/modality/text.py">Tokens</a></td>
387
+ <td>Tensor(array=token_ids)<br>optimize(..., item_loader=TokensLoader())</td>
388
+ <td>tokens # Tensor, length block_size — <a href="#llm-training">LLM training</a></td>
389
+ </tr>
390
+ <tr>
391
+ <td colspan="3"><strong>Image</strong></td>
392
+ </tr>
393
+ <tr>
394
+ <td valign="top"><a href="examples/modality/image.py">Image</a></td>
395
+ <td>Image(path="a.jpg")<br>Image(bytes=jpeg)<br>Image(array=hwc, quality=95, format="jpeg")</td>
396
+ <td>image.shape # Tensor CHW</td>
397
+ </tr>
398
+ <tr>
399
+ <td valign="top"><a href="examples/modality/jpeg.py">Jpeg</a></td>
400
+ <td>Jpeg(path="a.jpg")<br>Jpeg(array=hwc, quality=95)</td>
401
+ <td>image.shape # Tensor CHW</td>
402
+ </tr>
403
+ <tr>
404
+ <td valign="top"><a href="examples/modality/jpeg_array.py">JpegArray</a></td>
405
+ <td>JpegArray(images=[Jpeg(path=p) for p in frames])</td>
406
+ <td>images[0].shape # Tensor CHW</td>
407
+ </tr>
408
+ <tr>
409
+ <td valign="top"><a href="examples/modality/pil.py">Pil</a></td>
410
+ <td>Pil(path="a.png")<br>Pil(image=pil_img, mode="RGB")</td>
411
+ <td>pil_img.size # PIL.Image</td>
412
+ </tr>
413
+ <tr>
414
+ <td valign="top"><a href="examples/modality/tiff.py">Tiff</a></td>
415
+ <td>Tiff(path="a.tif")<br>Tiff(array=hw)</td>
416
+ <td>array.shape # NumPy</td>
417
+ </tr>
418
+ <tr>
419
+ <td colspan="3"><strong>Audio and Video</strong></td>
420
+ </tr>
421
+ <tr>
422
+ <td valign="top"><a href="examples/modality/audio.py">Audio</a></td>
423
+ <td>Audio(path="a.wav")<br>Audio(bytes=wav)<br>Audio(array=wave, sampling_rate=16000)</td>
424
+ <td>audio["array"]<br>audio["sampling_rate"]</td>
425
+ </tr>
426
+ <tr>
427
+ <td valign="top"><a href="examples/modality/video.py">Video</a></td>
428
+ <td>Video(path="c.mp4")<br>Video(bytes=mp4)<br>Video(array=frames, fps=25)</td>
429
+ <td>video.get_frames_at(0)<br>video.get_frames_in_range(0, 8)</td>
430
+ </tr>
431
+ <tr>
432
+ <td colspan="3"><strong>File</strong></td>
433
+ </tr>
434
+ <tr>
435
+ <td valign="top"><a href="examples/modality/file.py">File</a></td>
436
+ <td>File(path="doc.bin")<br>File(bytes=blob)</td>
437
+ <td>sidecar # raw bytes</td>
438
+ </tr>
439
+ <tr>
440
+ <td valign="top"><a href="examples/modality/pdf.py">Pdf</a></td>
441
+ <td>Pdf(path="p.pdf")<br>Pdf(pdf=pdfplumber_doc)</td>
442
+ <td>pdf.pages[0] # Pdfplumber</td>
443
+ </tr>
444
+ <tr>
445
+ <td colspan="3"><strong>3D and volume</strong></td>
446
+ </tr>
447
+ <tr>
448
+ <td valign="top"><a href="examples/modality/mesh.py">Mesh</a></td>
449
+ <td>Mesh(path="m.glb")<br>Mesh(mesh=trimesh_obj, file_type="glb")</td>
450
+ <td>mesh.vertices # Trimesh</td>
451
+ </tr>
452
+ <tr>
453
+ <td valign="top"><a href="examples/modality/nifti.py">Nifti</a></td>
454
+ <td>Nifti(path="v.nii.gz")<br>Nifti(array=vol, affine=np.eye(4))</td>
455
+ <td>nifti.get_fdata() # Nibabel</td>
456
+ </tr>
457
+ <tr>
458
+ <td colspan="3"><strong>Array and Graph</strong></td>
459
+ </tr>
460
+ <tr>
461
+ <td valign="top"><a href="examples/modality/numpy_array.py">Numpy</a></td>
462
+ <td>np.load("a.npy")<br>np.zeros((3, 4, 4))</td>
463
+ <td>array # NumPy</td>
464
+ </tr>
465
+ <tr>
466
+ <td valign="top"><a href="examples/modality/tensor.py">Tensor</a></td>
467
+ <td>Tensor(array=torch.randn(3, 4, 4))</td>
468
+ <td>feat # Tensor — 1-D token ids use TokensLoader under Text</td>
469
+ </tr>
470
+ <tr>
471
+ <td valign="top"><a href="examples/modality/graph.py">Graph</a></td>
472
+ <td>Data(x=…, edge_index=…, y=…)<br>Graph(x=…, edge_index=…, y=…)<br>Graph(data=pyg_data)</td>
473
+ <td>graph.x, graph.edge_index # PyG Data or Graph — <a href="#pyg-graphs">PyG graphs</a></td>
474
+ </tr>
475
+ <tr>
476
+ <td colspan="3"><strong>Parquet</strong></td>
477
+ </tr>
478
+ <tr>
479
+ <td valign="top"><a href="examples/modality/parquet.py">Parquet</a></td>
480
+ <td>folder of .parquet files<br>StreamingDataset(..., item_loader=ParquetLoader())</td>
481
+ <td>row["col"] # dict of columns — <a href="#stream-parquet">stream parquet</a></td>
482
+ </tr>
483
+ </table>
484
+
485
+ Examples (path on disk → optimize → batch): [examples/modality](examples/modality).
486
+
487
+ ----
488
+
353
489
  # Key Features
354
490
 
355
491
  ## Features for optimizing and streaming datasets for model training
@@ -565,24 +701,18 @@ How you return images from `optimize` controls storage size and streaming speed.
565
701
 
566
702
  | What you return | Serializer | Result |
567
703
  |-----------------|------------|--------|
568
- | `PIL.JpegImageFile` (e.g. `Image.open("x.jpg")`) | JPEG | Compressed bytes — **preferred** |
569
- | Plain `PIL.Image` / `Image.fromarray(...)` | PIL RAW | Uncompressed pixels — often **10×+ larger** |
704
+ | `litdata.Image(path=...)` / `Image(array=..., quality=95, format="jpeg")` | `image` | Compressed bytes — **preferred** |
705
+ | `litdata.Jpeg(path=...)` / `Jpeg(array=..., quality=95)` | `jpeg` | JPEG bytes |
706
+ | `PIL.JpegImageFile` (e.g. `PIL.Image.open("x.jpg")`) | `jpeg` | Compressed bytes |
707
+ | Plain `PIL.Image` / `Image.fromarray(...)` | `pil` | Uncompressed pixels — often **10×+ larger** |
570
708
 
571
- **Best practice:** store JPEG at **quality ≈ 95** (or keep existing `.jpg` files). Resize when helpful.
709
+ **Best practice:** wrap with `Image` / `Jpeg` at **quality ≈ 95**, or keep existing `.jpg` files via `Image(path=...)`. Resize when helpful.
572
710
 
573
711
  ```python
574
- import io
575
- from PIL import Image
576
712
  import litdata as ld
577
713
 
578
714
  def load_image(path):
579
- img = Image.open(path)
580
- if not str(path).lower().endswith((".jpg", ".jpeg")):
581
- buf = io.BytesIO()
582
- img.convert("RGB").save(buf, format="JPEG", quality=95)
583
- buf.seek(0)
584
- img = Image.open(buf) # JpegImageFile
585
- return {"image": img, "path": path}
715
+ return {"image": ld.Image(path=path, quality=95, format="jpeg"), "id": path}
586
716
 
587
717
  if __name__ == "__main__":
588
718
  ld.optimize(fn=load_image, inputs=list_of_paths, output_dir="fast_data", chunk_bytes="64MB", num_workers=8)
@@ -596,9 +726,9 @@ Ready-made ImageNet optimize/stream scripts: `benchmarks/litdata/` (`--write_mod
596
726
  <summary> ✅ Custom serializers <a id="serializers" href="#serializers">🔗</a> </summary>
597
727
  &nbsp;
598
728
 
599
- LitData serializes each leaf of your sample with a pluggable registry. Built-ins (tried in order) include: `str`, `bool`, `int`, `float`, `video`, `tifffile`, `pil`, `jpeg`, `jpeg_array`, `bytes`, `numpy` / `tensor` (and no-header variants), and `pickle` (fallback).
729
+ LitData serializes each **pytree leaf** with a pluggable registry. Built-ins (tried in order) include: `str`, `bool`, `int`, `float`, `video`, `audio`, `image`, `nifti`, `mesh`, `pdf`, `tifffile`, `file`, `pil`, `jpeg`, `jpeg_array`, `bytes`, `numpy` / `tensor` (and no-header variants), `graph`, and `pickle` (fallback).
600
730
 
601
- For images, returning a `JpegImageFile` selects **`jpeg`**; a plain `PIL.Image` selects **`pil`** (raw pixels). See [Optimize images as JPEG](#optimize-jpeg).
731
+ Prefer [typed media wrappers](#media-types) (`Audio`, `Video`, `Image`, `Graph`, …) so a filepath is not confused with a caption. For images, `Image(..., quality=95, format="jpeg")` or a `JpegImageFile` stores JPEG; a plain `PIL.Image` selects **`pil`** (raw pixels). See [Optimize images as JPEG](#optimize-jpeg).
602
732
 
603
733
  Pass custom serializers when **streaming** (and when using the lower-level `Cache` writer):
604
734
 
@@ -622,7 +752,133 @@ dataset = StreamingDataset(
622
752
  )
623
753
  ```
624
754
 
625
- Keys you pass are tried before the defaults (so they win over `pickle`). `optimize()` uses the built-in registry based on the Python types your `fn` returns — prefer JPEG / numpy / tensor leaves for best results.
755
+ Keys you pass are tried before the defaults (so they win over `pickle`). `optimize()` uses the built-in registry based on the Python types your `fn` returns — prefer typed wrappers / JPEG / numpy / tensor leaves for best results.
756
+
757
+ </details>
758
+
759
+ <details>
760
+ <summary> ✅ Stream PyG graphs <a id="pyg-graphs" href="#pyg-graphs">🔗</a> </summary>
761
+ &nbsp;
762
+
763
+ Store [PyTorch Geometric](https://pytorch-geometric.readthedocs.io/) `Data` / `HeteroData` as packed tensors (`to_dict()`), not `torch.save` / pickle. On read, LitData reconstructs with `from_dict` when `torch-geometric` is installed. `optimize` needs a **top-level** function (spawn).
764
+
765
+ `StreamingDataLoader` uses `litdata_collate` by default: graph samples become a `DataBatch` (`Batch.from_data_list`); everything else uses PyTorch `default_collate`. For `follow_batch` / `exclude_keys`, use `torch_geometric.loader.DataLoader`. Without PyG, graph batches stay a list of `Graph`.
766
+
767
+ ### Homogeneous `Data` + GCN
768
+
769
+ ```python
770
+ import torch
771
+ import torch.nn.functional as F
772
+ from torch_geometric.data import Data
773
+ from torch_geometric.nn import GCNConv, global_mean_pool
774
+
775
+ from litdata import StreamingDataLoader, StreamingDataset, optimize
776
+
777
+ def make_graph(i: int) -> Data:
778
+ n = 8 + i % 5
779
+ src = torch.randint(0, n, (12,), dtype=torch.long)
780
+ dst = torch.randint(0, n, (12,), dtype=torch.long)
781
+ return Data(
782
+ x=torch.randn(n, 8),
783
+ edge_index=torch.stack([src, dst], 0),
784
+ y=torch.tensor(i % 3),
785
+ train_mask=torch.ones(n, dtype=torch.bool),
786
+ num_nodes=n,
787
+ )
788
+
789
+ optimize(make_graph, inputs=list(range(1024)), output_dir="graphs", chunk_size=64)
790
+
791
+ dataset = StreamingDataset("graphs")
792
+ sample = dataset[0] # Data when PyG is installed, else Graph
793
+ loader = StreamingDataLoader(dataset, batch_size=32, shuffle=True)
794
+ batch = next(iter(loader)) # DataBatch
795
+
796
+ class Net(torch.nn.Module):
797
+ def __init__(self):
798
+ super().__init__()
799
+ self.conv = GCNConv(8, 16)
800
+ self.lin = torch.nn.Linear(16, 3)
801
+
802
+ def forward(self, data):
803
+ x = F.relu(self.conv(data.x, data.edge_index))
804
+ return self.lin(global_mean_pool(x, data.batch))
805
+ ```
806
+
807
+ ### `Graph` wrapper (no PyG at write time)
808
+
809
+ ```python
810
+ from litdata import Graph, optimize
811
+
812
+ def make_graph(i: int) -> Graph:
813
+ n = 6
814
+ return Graph(
815
+ x=torch.randn(n, 4),
816
+ edge_index=torch.tensor([[0, 1, 2], [1, 2, 0]], dtype=torch.long),
817
+ y=torch.tensor(i % 2),
818
+ data={"num_nodes": n}, # extra tensors/scalars; field kwargs override data=
819
+ )
820
+
821
+ optimize(make_graph, inputs=list(range(256)), output_dir="graphs")
822
+ # later: sample.to_pyg() if the stream returned Graph
823
+ ```
824
+
825
+ `Graph(data=pyg_data)` uses `pyg_data.to_dict()`. Do not mix tensor fields with an opaque NetworkX `data=`.
826
+
827
+ ### Heterogeneous `HeteroData`
828
+
829
+ ```python
830
+ from torch_geometric.data import HeteroData
831
+
832
+ from litdata import StreamingDataLoader, StreamingDataset, optimize
833
+
834
+ def make_hetero(i: int) -> HeteroData:
835
+ data = HeteroData()
836
+ data["paper"].x = torch.randn(8, 16)
837
+ data["author"].x = torch.randn(4, 8)
838
+ data["author", "writes", "paper"].edge_index = torch.tensor(
839
+ [[0, 1, 2, 3], [0, 2, 4, 6]], dtype=torch.long
840
+ )
841
+ data.y = torch.tensor(i % 3)
842
+ return data
843
+
844
+ optimize(make_hetero, inputs=list(range(512)), output_dir="hetero", chunk_size=32)
845
+
846
+ dataset = StreamingDataset("hetero")
847
+ sample = dataset[0] # HeteroData
848
+ print(sample["paper"].x.shape, sample["author", "writes", "paper"].edge_index.shape)
849
+
850
+ loader = StreamingDataLoader(dataset, batch_size=16)
851
+ batch = next(iter(loader)) # HeteroDataBatch
852
+ # batch["paper"].x, batch["paper"].batch, batch["author", "writes", "paper"].edge_index
853
+ ```
854
+
855
+ ### Graph plus metadata in one sample
856
+
857
+ ```python
858
+ def make_row(i: int) -> dict:
859
+ return {"id": i, "graph": make_graph(i)}
860
+
861
+ optimize(make_row, inputs=list(range(1024)), output_dir="rows")
862
+ loader = StreamingDataLoader(StreamingDataset("rows"), batch_size=8)
863
+ batch = next(iter(loader))
864
+ # batch["id"] is a tensor; batch["graph"] is a DataBatch
865
+ ```
866
+
867
+ ### Sample subgraphs first, then stream
868
+
869
+ `NeighborLoader` needs one in-memory graph. To stream, run the sampler in `optimize` and store each subgraph as a `Data`:
870
+
871
+ ```python
872
+ # sampler = NeighborSampler(big_graph, num_neighbors=[10, 10])
873
+
874
+ def sample_seed(seed: int) -> Data:
875
+ out = sampler.sample_from_nodes(torch.tensor([seed]))
876
+ return Data(x=out.x, edge_index=out.edge_index, y=out.y)
877
+
878
+ optimize(sample_seed, inputs=train_seeds.tolist(), output_dir="subgraphs")
879
+ ```
880
+
881
+ NetworkX (or any non-tensor object) uses `Graph(data=nx_graph)` → `graph:pickle`. Do not `torch.save` a graph into the sample.
626
882
 
627
883
  </details>
628
884
 
@@ -1008,16 +1264,15 @@ Local `output_dir` writes chunks in place. Remote inputs and outputs use the str
1008
1264
 
1009
1265
  ```python
1010
1266
  import numpy as np
1011
- from PIL import Image
1012
1267
  import litdata as ld
1013
1268
 
1014
1269
  def random_images(index):
1015
- fake_images = Image.fromarray(np.random.randint(0, 256, (32, 32, 3), dtype=np.uint8))
1016
- fake_labels = np.random.randint(10)
1017
-
1018
- data = {"index": index, "image": fake_images, "class": fake_labels}
1019
-
1020
- return data
1270
+ array = np.random.randint(0, 256, (32, 32, 3), dtype=np.uint8)
1271
+ return {
1272
+ "index": index,
1273
+ "image": ld.Image(array=array, quality=95, format="jpeg"),
1274
+ "class": np.random.randint(10),
1275
+ }
1021
1276
 
1022
1277
  if __name__ == "__main__":
1023
1278
  # The optimize function writes data in an optimized format.
@@ -1402,15 +1657,15 @@ Merge multiple optimized datasets into one.
1402
1657
 
1403
1658
  ```python
1404
1659
  import numpy as np
1405
- from PIL import Image
1406
1660
 
1407
- from litdata import StreamingDataset, merge_datasets, optimize
1661
+ from litdata import Image, StreamingDataset, merge_datasets, optimize
1408
1662
 
1409
1663
 
1410
1664
  def random_images(index):
1665
+ array = np.random.randint(0, 256, (32, 32, 3), dtype=np.uint8)
1411
1666
  return {
1412
1667
  "index": index,
1413
- "image": Image.fromarray(np.random.randint(0, 256, (32, 32, 3), dtype=np.uint8)),
1668
+ "image": Image(array=array, quality=95, format="jpeg"),
1414
1669
  "class": np.random.randint(10),
1415
1670
  }
1416
1671
 
@@ -1427,6 +1682,16 @@ if __name__ == "__main__":
1427
1682
  print(len(dataset))
1428
1683
  # out: 1000
1429
1684
  ```
1685
+
1686
+ If you wrote chunks yourself (`Cache` / `BinaryWriter`) and only have `{rank}.index.json` shards, finish the dataset:
1687
+
1688
+ ```python
1689
+ from litdata import complete_dataset, StreamingDataset
1690
+
1691
+ complete_dataset("my_chunks") # no-op if index.json already exists
1692
+ StreamingDataset("my_chunks") # also tries this automatically
1693
+ ```
1694
+
1430
1695
  </details>
1431
1696
 
1432
1697
  <details>
@@ -1512,6 +1777,13 @@ print(test_dataset)
1512
1777
  # out: 50,000
1513
1778
  ```
1514
1779
 
1780
+ Or pick exact indices with `StreamingDataset.subset`:
1781
+
1782
+ ```python
1783
+ train = dataset.subset(range(0, 30_000))
1784
+ # dataset.subset(slice(0, 1000))
1785
+ ```
1786
+
1515
1787
  </details>
1516
1788
 
1517
1789
  <details>
@@ -1528,6 +1800,9 @@ dataset = StreamingDataset("s3://my-bucket/my-data", subsample=0.01) # data are
1528
1800
 
1529
1801
  print(len(dataset)) # display the length of your data
1530
1802
  # out: 1000
1803
+
1804
+ # or a list / slice of global indices
1805
+ small = dataset.subset([0, 10, 20])
1531
1806
  ```
1532
1807
 
1533
1808
  </details>
@@ -1644,7 +1919,7 @@ ld.index_parquet_dataset(
1644
1919
 
1645
1920
  ### Stream with `ParquetLoader`
1646
1921
 
1647
- Unlike `hf://`, local/S3/GCS parquet **does not** auto-select the loader — pass `ParquetLoader` explicitly (it must match `index.json`).
1922
+ If the folder looks like parquet and has no `index.json`, `StreamingDataset` now **builds the index automatically**. You still need `ParquetLoader` for local/S3/GCS (it must match `index.json`). `hf://` already auto-indexes and selects the loader.
1648
1923
 
1649
1924
  ```python
1650
1925
  import litdata as ld
@@ -1950,7 +2225,8 @@ export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
1950
2225
  | `LITDATA_OBSTORE_STREAM_MIN_CHUNK_MIB` | `8` | S3 obstore stream chunk size (MiB) |
1951
2226
  | `MAX_WAIT_TIME` | `120` | Seconds to wait for a chunk before error |
1952
2227
  | `FORCE_DOWNLOAD_TIME` | `30` | Seconds before force re-download of a missing chunk |
1953
- | `LITDATA_DISABLE_VERSION_CHECK` | `0` | `1` skips the upgrade tip |
2228
+ | `LITDATA_CHECK_UPDATES` | unset | `1` enables the PyPI upgrade tip (off by default) |
2229
+ | `LITDATA_DISABLE_VERSION_CHECK` | on unless updates enabled | `1` skips the upgrade tip |
1954
2230
  | `HF_TOKEN` | — | Gated Hugging Face datasets |
1955
2231
  | `DEBUG_LITDATA` / `PRINT_DEBUG_LOGS` | `0` | Internal debug / stdout logs |
1956
2232
  | `LITDATA_LOG_FILE` | `litdata_debug.log` | `enable_tracer()` output path |
@@ -2457,26 +2733,12 @@ Speed to stream Imagenet 1.2M from other cloud storage providers:
2457
2733
  |---|---|---|---|
2458
2734
  | Cloudflare R2 | LitData | **5335** | **5630** |
2459
2735
 
2460
- Speed to stream Imagenet 1.2M from local disk with ffcv vs LitData:
2461
- | Framework | Dataset Mode | Dataset Size @ 256px | Images / sec 1st Epoch (float32) | Images / sec 2nd Epoch (float32) |
2462
- |---|---|---|---|---|
2463
- | LitData | PIL RAW | 168 GB | 6647 | 6398 |
2464
- | LitData | JPEG 90% | 12 GB | 6553 | 6537 |
2465
- | ffcv (os_cache=True) | RAW | 170 GB | 7263 | 6698 |
2466
- | ffcv (os_cache=False) | RAW | 170 GB | 7556 | 8169 |
2467
- | ffcv(os_cache=True) | JPEG 90% | 20 GB | 7653 | 8051 |
2468
- | ffcv(os_cache=False) | JPEG 90% | 20 GB | 8149 | 8607 |
2469
-
2470
- Speed to stream a **synthetic ImageNet-scale set from Vast NFS** (NFSv3 `nconnect=32`, 208-CPU host, ~1 TiB RAM). Dataset: **1.08M** JPEG q95 256×256 (~160 GiB, 64 MiB chunks). `StreamingDataLoader`, batch **256**, `shuffle=True`, `drop_last=True`, decode only unless noted. POSIX-fast mmaps chunks **in place** (no copy into `~/.lightning/chunks`).
2736
+ Speed to stream a synthetic ImageNet-scale set from **Vast NFS** with POSIX-fast (mmap in place, decode only, no transforms):
2471
2737
 
2472
- | Setup | Workers | Images / sec |
2473
- |---|---|---|
2474
- | Copy into local cache (`LITDATA_POSIX_FAST=0`) | 48 | **16.7k** (2-epoch avg) |
2475
- | POSIX-fast (this default on local/Vast paths) | 48 | **18.2k** |
2476
- | POSIX-fast + README ImageNet augs (crop 224, flip, float32) | 48 | **12.9k** |
2477
- | POSIX-fast, all CPU cores | **208** | **35.8k** |
2478
-
2479
- Notes: 208 workers need enough **MemAvailable**. This host had **928×1 GiB hugepages** reserved and idle (~900 GiB locked); after `nr_hugepages=0`, 208 workers stayed healthy. If `num_workers=os.cpu_count()` would crowd RAM, LitData **clamps** workers (`LITDATA_POSIX_MAX_WORKERS=0` disables) and skips `WILLNEED` prefetch. Real ImageNet JPEG 90% is much smaller (~12 GiB) and usually decodes faster than this q95 noise set.
2738
+ | Workers | Images / sec |
2739
+ |---|---|
2740
+ | 48 | **18.2k** |
2741
+ | 208 | **35.8k** |
2480
2742
 
2481
2743
  ### Raw Dataset
2482
2744
 
@@ -2584,6 +2846,80 @@ Below are templates for real-world applications of LitData at scale.
2584
2846
 
2585
2847
  ----
2586
2848
 
2849
+ # Used by
2850
+
2851
+ <table width="100%">
2852
+ <tr>
2853
+ <th align="left" width="18%">Project</th>
2854
+ <th align="left">Description</th>
2855
+ </tr>
2856
+ <tr>
2857
+ <td valign="top"><a href="https://github.com/sunlabuiuc/PyHealth">PyHealth</a></td>
2858
+ <td>Deep-learning toolkit for clinical prediction (MIMIC, eICU, OMOP, sleep, CXR). <code>set_task()</code> writes processed samples with LitData; <code>SampleDataset</code> subclasses <code>StreamingDataset</code> so training streams chunked EHR tensors instead of holding the cohort in RAM.</td>
2859
+ </tr>
2860
+ <tr>
2861
+ <td valign="top"><a href="https://github.com/prescient-design/lobster">LBSTER</a></td>
2862
+ <td>Protein and biological-sequence language models from Prescient Design (Genentech). Pre-training and concept-bottleneck models (fitness, embeddings, guided generation) stream large sequence corpora through LitData.</td>
2863
+ </tr>
2864
+ <tr>
2865
+ <td valign="top"><a href="https://github.com/OpenSynth-energy/OpenSynth">OpenSynth</a></td>
2866
+ <td>Open toolkit for synthetic smart-meter / energy time series. Generated or historical meter traces are optimized and streamed for model training.</td>
2867
+ </tr>
2868
+ <tr>
2869
+ <td valign="top"><a href="https://github.com/BiomedSciAI/biomed-multi-view">biomed-multi-view</a></td>
2870
+ <td>IBM BiomedSciAI multi-view biomedical models. LitData is used to cache and stream paired modalities during training.</td>
2871
+ </tr>
2872
+ <tr>
2873
+ <td valign="top"><a href="https://github.com/cma2015/DEM">DEM</a></td>
2874
+ <td>Phenotype and gene-mining pipeline (<code>biodem</code>). Large genomic / trait tables are packed into LitData chunks for repeated training passes.</td>
2875
+ </tr>
2876
+ <tr>
2877
+ <td valign="top"><a href="https://pypi.org/project/deeptan/">deeptan</a></td>
2878
+ <td>Graph multi-task models for multi-omics trait-associated networks. Guide graphs and expression tables are converted to LitData chunks before GNN training.</td>
2879
+ </tr>
2880
+ <tr>
2881
+ <td valign="top"><a href="https://github.com/avitai/datarax">datarax</a></td>
2882
+ <td>Data tooling with an optional cloud-streaming extra that uses LitData to read remote datasets without a full local copy.</td>
2883
+ </tr>
2884
+ <tr>
2885
+ <td valign="top"><a href="https://pypi.org/project/fasr/">fasr</a></td>
2886
+ <td>Speech ASR framework. The LitData extra streams audio and transcripts for training instead of random-access file lists.</td>
2887
+ </tr>
2888
+ </table>
2889
+
2890
+ # Skills <a id="skills"></a>
2891
+
2892
+ Coding agents (Cursor, Claude Code, and others) should load the LitData skill instead of guessing the API.
2893
+
2894
+ ```bash
2895
+ npx skills add Lightning-AI/litData
2896
+ ```
2897
+
2898
+ Useful options: `-g` (user-global), `-a cursor` (Cursor only), `-y` (non-interactive). In this repo the skill already lives at [`.claude/skills/litdata/`](.claude/skills/litdata/). Installer: [skills CLI](https://github.com/vercel-labs/skills).
2899
+
2900
+ Start at [`SKILL.md`](.claude/skills/litdata/SKILL.md), then load [`reference/using-litdata.md`](.claude/skills/litdata/reference/using-litdata.md) before writing examples.
2901
+
2902
+ | File | When to load |
2903
+ | --- | --- |
2904
+ | [SKILL.md](.claude/skills/litdata/SKILL.md) | Triggers, public API, traps |
2905
+ | [using-litdata.md](.claude/skills/litdata/reference/using-litdata.md) | Optimize / stream / raw / modality cookbook |
2906
+ | [streaming.md](.claude/skills/litdata/reference/streaming.md) | Read path, shuffle, resume, serializers |
2907
+ | [processing.md](.claude/skills/litdata/reference/processing.md) | optimize / map orchestration |
2908
+ | [data-movement.md](.claude/skills/litdata/reference/data-movement.md) | Download / upload / FUSE vs direct I/O |
2909
+ | [multi-node.md](.claude/skills/litdata/reference/multi-node.md) | Studio num_nodes jobs |
2910
+ | [resolver.md](.claude/skills/litdata/reference/resolver.md) | Paths, URLs, Studio mounts |
2911
+ | [storage-format.md](.claude/skills/litdata/reference/storage-format.md) | Chunks, `index.json`, writer / reader |
2912
+ | [cache-and-chunk-lifecycle.md](.claude/skills/litdata/reference/cache-and-chunk-lifecycle.md) | Prefetch and eviction |
2913
+ | [env-vars.md](.claude/skills/litdata/reference/env-vars.md) | LITDATA_* and DATA_OPTIMIZER_* |
2914
+ | [keyed-lookup.md](.claude/skills/litdata/reference/keyed-lookup.md) | key_fn, dataset_update |
2915
+ | [debugging.md](.claude/skills/litdata/reference/debugging.md) | enable_tracer, Litracer |
2916
+ | [benchmarking.md](.claude/skills/litdata/reference/benchmarking.md) | Fair benches |
2917
+ | [lightning-studio.md](.claude/skills/litdata/reference/lightning-studio.md) | Studio env and credentials |
2918
+ | [testing.md](.claude/skills/litdata/reference/testing.md) | Pytest / CI |
2919
+ | [contributing.md](.claude/skills/litdata/reference/contributing.md) | PR / lint path |
2920
+
2921
+ Offline streaming what-if (not the Python package): [simulator/](simulator/) (litsim).
2922
+
2587
2923
  # Community
2588
2924
  LitData is a community project accepting contributions - Let's make the world's most advanced AI data processing framework.
2589
2925
 
@@ -2600,8 +2936,7 @@ LitData is a community project accepting contributions - Let's make the world's
2600
2936
  author = {Thomas Chaton and Lightning AI},
2601
2937
  title = {LitData: Transform datasets at scale. Optimize datasets for fast AI model training.},
2602
2938
  year = {2023},
2603
- howpublished = {\url{https://github.com/Lightning-AI/litdata}},
2604
- note = {Accessed: 2025-04-09}
2939
+ howpublished = {\url{https://github.com/Lightning-AI/litdata}}
2605
2940
  }
2606
2941
  ```
2607
2942
 
@@ -2609,8 +2944,12 @@ LitData is a community project accepting contributions - Let's make the world's
2609
2944
 
2610
2945
  ## Papers with LitData
2611
2946
 
2612
- * [Towards Interpretable Protein Structure
2613
- Prediction with Sparse Autoencoders](https://arxiv.org/pdf/2503.08764) | [Github](https://github.com/johnyang101/reticular-sae) | (Nithin Parsan, David J. Yang and John J. Yang)
2947
+ Papers that train or stream with LitData (`optimize` / `StreamingDataset`). Scholar hits for “litdata streaming” are often weather *lightning* data, or mention LitData only as example source code.
2948
+
2949
+ | Paper | Venue | How LitData is used |
2950
+ |---|---|---|
2951
+ | [Towards Interpretable Protein Structure Prediction with Sparse Autoencoders](https://arxiv.org/abs/2503.08764) ([code](https://github.com/johnyang101/reticular-sae)) | ICLR 2025 GEM | `optimize` shards ESM-2 embeddings; `StreamingDataset` streams from S3 for multi-GPU SAE training |
2952
+ | [TinyLlama: An Open-Source Small Language Model](https://arxiv.org/abs/2401.02385) ([code](https://github.com/jzhang38/TinyLlama)) | arXiv 2024 | 1.1B pretrain on SlimPajama + StarCoder via Lit-GPT’s `lightning.data` stack (now LitData): `CombinedStreamingDataset` + `TokensLoader` |
2614
2953
 
2615
2954
  ----
2616
2955