litdata 0.2.67__tar.gz → 0.2.69__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. {litdata-0.2.67 → litdata-0.2.69}/CONTRIBUTING.md +1 -1
  2. {litdata-0.2.67/src/litdata.egg-info → litdata-0.2.69}/PKG-INFO +106 -56
  3. {litdata-0.2.67 → litdata-0.2.69}/README.md +105 -55
  4. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/__about__.py +1 -1
  5. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/__init__.py +6 -0
  6. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/constants.py +6 -0
  7. litdata-0.2.69/src/litdata/debugger.py +397 -0
  8. litdata-0.2.69/src/litdata/exceptions.py +36 -0
  9. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/processing/data_processor.py +1146 -238
  10. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/processing/functions.py +59 -23
  11. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/streaming/async_prefetch.py +19 -2
  12. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/streaming/cache.py +2 -2
  13. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/streaming/client.py +11 -0
  14. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/streaming/combined.py +3 -12
  15. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/streaming/compression.py +28 -8
  16. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/streaming/config.py +9 -17
  17. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/streaming/dataloader.py +70 -6
  18. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/streaming/dataset.py +354 -47
  19. litdata-0.2.69/src/litdata/streaming/dataset_update.py +299 -0
  20. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/streaming/downloader.py +280 -104
  21. litdata-0.2.69/src/litdata/streaming/elastic.py +243 -0
  22. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/streaming/item_loader.py +317 -118
  23. litdata-0.2.69/src/litdata/streaming/posix_fast.py +396 -0
  24. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/streaming/reader.py +117 -36
  25. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/streaming/resolver.py +1 -1
  26. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/streaming/shuffle.py +52 -0
  27. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/streaming/writer.py +49 -24
  28. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/utilities/format.py +1 -1
  29. litdata-0.2.69/src/litdata/utilities/keys_index.py +870 -0
  30. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/utilities/shuffle.py +114 -0
  31. {litdata-0.2.67 → litdata-0.2.69/src/litdata.egg-info}/PKG-INFO +106 -56
  32. {litdata-0.2.67 → litdata-0.2.69}/src/litdata.egg-info/SOURCES.txt +5 -0
  33. litdata-0.2.67/src/litdata/debugger.py +0 -205
  34. {litdata-0.2.67 → litdata-0.2.69}/LICENSE +0 -0
  35. {litdata-0.2.67 → litdata-0.2.69}/MANIFEST.in +0 -0
  36. {litdata-0.2.67 → litdata-0.2.69}/requirements.txt +0 -0
  37. {litdata-0.2.67 → litdata-0.2.69}/setup.cfg +0 -0
  38. {litdata-0.2.67 → litdata-0.2.69}/setup.py +0 -0
  39. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/__main__.py +0 -0
  40. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/cli/__init__.py +0 -0
  41. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/cli/commands.py +0 -0
  42. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/cli/handler/__init__.py +0 -0
  43. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/cli/handler/cache.py +0 -0
  44. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/cli/handler/optimize.py +0 -0
  45. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/cli/parser.py +0 -0
  46. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/helpers.py +0 -0
  47. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/imports.py +0 -0
  48. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/processing/__init__.py +0 -0
  49. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/processing/readers.py +0 -0
  50. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/processing/utilities.py +0 -0
  51. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/raw/__init__.py +0 -0
  52. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/raw/dataset.py +0 -0
  53. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/raw/indexer.py +0 -0
  54. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/raw/types.py +0 -0
  55. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/requirements.py +0 -0
  56. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/streaming/__init__.py +0 -0
  57. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/streaming/fs_provider.py +0 -0
  58. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/streaming/parallel.py +0 -0
  59. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/streaming/sampler.py +0 -0
  60. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/streaming/serializers.py +0 -0
  61. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/streaming/timing.py +0 -0
  62. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/utilities/__init__.py +0 -0
  63. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/utilities/_pytree.py +0 -0
  64. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/utilities/base.py +0 -0
  65. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/utilities/breakpoint.py +0 -0
  66. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/utilities/broadcast.py +0 -0
  67. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/utilities/dataset_utilities.py +0 -0
  68. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/utilities/encryption.py +0 -0
  69. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/utilities/env.py +0 -0
  70. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/utilities/hf_dataset.py +0 -0
  71. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/utilities/packing.py +0 -0
  72. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/utilities/parquet.py +0 -0
  73. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/utilities/subsample.py +0 -0
  74. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/utilities/torch_utils.py +0 -0
  75. {litdata-0.2.67 → litdata-0.2.69}/src/litdata/utilities/train_test_split.py +0 -0
  76. {litdata-0.2.67 → litdata-0.2.69}/src/litdata.egg-info/dependency_links.txt +0 -0
  77. {litdata-0.2.67 → litdata-0.2.69}/src/litdata.egg-info/entry_points.txt +0 -0
  78. {litdata-0.2.67 → litdata-0.2.69}/src/litdata.egg-info/not-zip-safe +0 -0
  79. {litdata-0.2.67 → litdata-0.2.69}/src/litdata.egg-info/requires.txt +0 -0
  80. {litdata-0.2.67 → litdata-0.2.69}/src/litdata.egg-info/top_level.txt +0 -0
@@ -2,7 +2,7 @@
2
2
 
3
3
  Welcome to the PyTorch Lightning community! We're building the most advanced research platform on the planet to implement the latest, best practices and integrations that the amazing PyTorch team and other research organization rolls out!
4
4
 
5
- If you are new to open source, check out [this blog to get started with your first Open Source contribution](https://devblog.pytorchlightning.ai/quick-contribution-guide-86d977171b3a).
5
+ If you are new to open source, check out [GitHub's guide to making your first contribution](https://docs.github.com/en/get-started/quickstart/contributing-to-projects).
6
6
 
7
7
  ## Main Core Value: One less thing to remember
8
8
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: litdata
3
- Version: 0.2.67
3
+ Version: 0.2.69
4
4
  Summary: The Deep Learning framework to train, deploy, and ship AI products Lightning fast.
5
5
  Home-page: https://github.com/Lightning-AI/litdata
6
6
  Download-URL: https://github.com/Lightning-AI/litdata
@@ -279,6 +279,22 @@ for sample in dataloader:
279
279
  img, cls = sample["image"], sample["class"]
280
280
  ```
281
281
 
282
+ **Keyed lookup and in-place patches** (needs `polars` and `optimize(..., key_fn=...)` or `build_keys_index`):
283
+
284
+ ```python
285
+ ld.optimize(fn=fn, inputs=inputs, output_dir="fast_data", chunk_bytes="64MB", key_fn=lambda s: s["id"])
286
+
287
+ ds = ld.StreamingDataset("fast_data")
288
+ sample = ds["entity-id"] # str keys
289
+ sample = ds.get_by_key(42) # int entity keys; ds[42] is still positional
290
+
291
+ with ld.dataset_update("fast_data") as update: # local directory only
292
+ update["entity-id"] = {"id": "entity-id", "x": 1}
293
+ update.commit()
294
+ ```
295
+
296
+ `mode="append"` continues chunk numbering. `use_checkpoint=True` tries to resume an interrupted optimize. They are not the same.
297
+
282
298
  **Key benefits:**
283
299
 
284
300
  ✅ **Accelerate training:** Optimized datasets load 20x faster.
@@ -785,6 +801,10 @@ Shuffling is **deterministic** and designed for distributed training:
785
801
 
786
802
  The permutation depends on `seed`, the epoch, and chunk metadata — the same settings always yield the same order (required for resumable `state_dict`).
787
803
 
804
+ **Object storage (`s3://`, `gs://`, …)** globally permutes chunks (`FullShuffle`). Random chunk order is cheap once files are already copied into the local cache.
805
+
806
+ **POSIX-fast** (automatic for any local path) mmaps chunks in place. **Vast / NFS / Lustre / GPFS** (and `LITDATA_POSIX_FAST=1`) use `WindowShuffle`: each worker gets **whole chunks** in a sequential stripe, then shuffles only inside a sliding window (default **16**, `LITDATA_POSIX_SHUFFLE_WINDOW`) for both chunk order and in-chunk items. Local disks (ext4/xfs) keep global `FullShuffle`. Object URLs stay on `FullShuffle`. `LITDATA_POSIX_FAST=0` disables in-place mmap.
807
+
788
808
  ```python
789
809
  from litdata import StreamingDataset, StreamingDataLoader
790
810
 
@@ -853,7 +873,7 @@ Rough ImageNet order-of-magnitude on a Studio (not hard guarantees; right tuning
853
873
  | `seed` | `42` | Shuffle / subsample RNG |
854
874
  | `serializers` | built-ins | Custom serialize/deserialize map |
855
875
  | `max_cache_size` | `"100GB"` | Evict consumed chunks beyond this size |
856
- | `max_pre_download` | `2` | Chunks each worker may prefetch (raise for throughput; watch disk) |
876
+ | `max_pre_download` | `2` | Chunks each worker may prefetch (raise for throughput; watch disk / RAM) |
857
877
  | `subsample` | `1.0` | Fraction of data (`0.01`) or upsample (`2.5`) |
858
878
  | `encryption` | `None` | `FernetEncryption` / `RSAEncryption` / custom |
859
879
  | `storage_options` | `{}` | Cloud client options |
@@ -864,6 +884,8 @@ Rough ImageNet order-of-magnitude on a Studio (not hard guarantees; right tuning
864
884
 
865
885
  Peak disk ≈ `num_workers × max_pre_download × mean_chunk_size`.
866
886
 
887
+ On **Vast / NFS / local disk**, POSIX-fast is on by default (`LITDATA_POSIX_FAST=0` to disable). `WILLNEED` prefetch and `num_workers` are capped when they would exceed about half of `MemAvailable`. Idle **hugepages** (common on GPU nodes) do not count as available RAM — drop unused `nr_hugepages` if `MemAvailable` looks tiny next to `MemTotal`.
888
+
867
889
  **`StreamingDataLoader`**
868
890
 
869
891
  | Argument | Description |
@@ -967,6 +989,12 @@ for batch_idx, batch in enumerate(dataloader):
967
989
  torch.save(dataloader.state_dict(), "dataloader_state.pt")
968
990
  ```
969
991
 
992
+ Same `seed` and `shuffle` are required. For a **`StreamingDataset`**, **`num_workers` and `world_size` may change**: LitData drops a global `sample_in_epoch` prefix and restripes the rest (never duplicates remaining IDs). For a matching loss curve keep **global batch size** (`world_size * batch_size`) constant and DDP ranks in lockstep. `num_canonical_nodes` (default: first-run `world_size`) is frozen in the checkpoint. POSIX `WindowShuffle` resumes whole remaining chunks. **`CombinedStreamingDataset` and `ParallelStreamingDataset` resume only with the same `world_size`, `num_workers`, and `batch_size`.**
993
+
994
+ ```python
995
+ dataset = StreamingDataset("s3://my-bucket/my-data", shuffle=True, num_canonical_nodes=8)
996
+ ```
997
+
970
998
  </details>
971
999
 
972
1000
 
@@ -974,11 +1002,9 @@ for batch_idx, batch in enumerate(dataloader):
974
1002
  <summary> ✅ Use shared queue for Optimizing <a id="shared-queue" href="#shared-queue">🔗</a> </summary>
975
1003
  &nbsp;
976
1004
 
977
- If you are using multiple workers to optimize your dataset, you can use a shared queue to speed up the process.
1005
+ `optimize` / `map` default to a **shared per-node queue** (`keep_data_ordered=False`). Work is packed per node, then every worker on that node pulls the next item, so a slow worker does not leave others idle. Set `keep_data_ordered=True` to keep a static per-worker slice (required for `use_checkpoint` and `align_chunking`).
978
1006
 
979
- This is especially useful when optimizing large datasets in parallel, where some workers may be slower than others.
980
-
981
- It can also improve fault tolerance when workers fail due to out-of-memory (OOM) errors.
1007
+ Local `output_dir` writes chunks in place. Remote inputs and outputs use the streaming downloader (`adownload_file` / `aupload_file`, obstore when available).
982
1008
 
983
1009
  ```python
984
1010
  import numpy as np
@@ -1001,28 +1027,36 @@ if __name__ == "__main__":
1001
1027
  output_dir="fast_data", # optimized data is stored here
1002
1028
  num_workers=4, # The number of workers on the same machine
1003
1029
  chunk_bytes="64MB" , # size of each chunk
1004
- keep_data_ordered=False, # Use a shared queue to speed up the process
1030
+ keep_data_ordered=False, # default: shared queue (set True to keep input order)
1005
1031
  )
1006
1032
  ```
1007
1033
 
1008
1034
 
1009
- ### Performance Difference between using a shared queue and not using it:
1035
+ ### Shared queue vs ordered (skewed local files)
1010
1036
 
1011
- **Note**: The following benchmarks were collected using the ImageNet dataset on an A10G machine with 16 workers.
1037
+ `scripts/bench/bench_node_queue.py --files 4000 --workers 8` (first 500 files are 1 MiB). On `main`, unordered optimize sat on a 200s empty-queue timeout after work finished.
1012
1038
 
1013
- | Configuration | Optimize Time (sec) | Stream 1 (img/sec) | Stream 2 (img/sec) |
1014
- |------------------|---------------------|---------------------|---------------------|
1015
- | shared_queue (`keep_data_ordered=False`) | 1281 | 5392 | 5732 |
1016
- | no shared_queue (`keep_data_ordered=True (default)`) | 1187 | 5257 | 5746 |
1039
+ | Tree | Mode | Time | Throughput |
1040
+ |------|------|-----:|-----------:|
1041
+ | `main` (old default) | `keep_data_ordered=True` | 23.7s | 169 files/s |
1042
+ | `main` | `keep_data_ordered=False` | 223.6s | 18 files/s |
1043
+ | this tree | `keep_data_ordered=True` | 22.9s | 175 files/s |
1044
+ | this tree (**new default**) | `keep_data_ordered=False` | **18.8s** | 213 files/s |
1017
1045
 
1018
- 📌 Note: The **shared_queue** option impacts optimization time, not streaming speed.
1019
- > While the streaming numbers may appear slightly different, this variation is incidental and not caused by shared_queue.
1020
- >
1021
- > Streaming happens after optimization and does not involve inter-process communication where shared_queue plays a role.
1046
+ Shared-queue **before → after: ~12×**. New default vs old ordered default: **1.22×**.
1022
1047
 
1023
- - 📄 Using a shared queue helps balance the load across workers, though it may slightly increase optimization time due to the overhead of pickling items sent between processes.
1048
+ ### Local / remote input × output
1024
1049
 
1025
- - ⚡ However, it can significantly improve optimizing performance — especially when some workers are slower than others.
1050
+ `python scripts/bench/bench_node_queue.py --files 200 --workers 4 --io-matrix` (first 50 files are 1 MiB).
1051
+
1052
+ | Topology | Ordered | Shared | Speedup |
1053
+ |----------|--------:|-------:|--------:|
1054
+ | local → local | 6.55s | **2.96s** | 2.21× |
1055
+ | remote → local | 7.44s | **3.48s** | 2.14× |
1056
+ | local → remote | 10.98s | **6.51s** | 1.69× |
1057
+ | remote → remote | 11.48s | **8.55s** | 1.34× |
1058
+
1059
+ Shared queue balances uneven workers. It does not change later `StreamingDataset` throughput.
1026
1060
 
1027
1061
  </details>
1028
1062
 
@@ -1806,7 +1840,7 @@ Only **worker 0** is instrumented. When an `int` is used, the tracer wraps `fetc
1806
1840
 
1807
1841
  - Delete or change `profile_dir` between runs — LitData removes an existing `result.json` before starting.
1808
1842
  - Pair with a wiped chunk cache if you care about **cold** epoch behavior (`litdata cache clear`).
1809
- - For deeper LitData internals (download / lock / delete timeline), use `enable_tracer()` + [Litracer](https://github.com/deependujha/litracer) instead — see [Debug & Profile LitData](#debug-profile). That path is complementary: viztracer = DataLoader worker CPU timeline; Litracer = LitData pipeline events.
1843
+ - For deeper LitData internals (download / read / delete timeline), use `enable_tracer()` + [Litracer](https://github.com/Lightning-AI/litracer) instead — see [Debug & Profile LitData](#debug-profile). That path is complementary: viztracer = DataLoader worker CPU timeline; Litracer = LitData pipeline events.
1810
1844
 
1811
1845
  </details>
1812
1846
 
@@ -1919,6 +1953,9 @@ export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
1919
1953
  | `LITDATA_DISABLE_VERSION_CHECK` | `0` | `1` skips the upgrade tip |
1920
1954
  | `HF_TOKEN` | — | Gated Hugging Face datasets |
1921
1955
  | `DEBUG_LITDATA` / `PRINT_DEBUG_LOGS` | `0` | Internal debug / stdout logs |
1956
+ | `LITDATA_LOG_FILE` | `litdata_debug.log` | `enable_tracer()` output path |
1957
+ | `LITDATA_TRACE_LEVEL` | unset | `batch` / `chunk` / `sample` / `debug` / `off` (see [Debug & Profile](#debug-profile)) |
1958
+ | `LITDATA_TRACE_CATEGORIES` | from level | Comma-separated cats, e.g. `download,read,delete` |
1922
1959
 
1923
1960
  Multi-node `optimize`/`map` on Studios also uses `DATA_OPTIMIZER_*` (set by the platform). Full catalog (debug logs, Studio injects, torchrun): see the LitData skill `reference/env-vars.md` when using agent skills, or the source modules `constants.py` / `async_prefetch.py`.
1924
1961
 
@@ -2084,66 +2121,68 @@ ds = StreamingDataset(input_dir=data_dir, encryption=rsa)
2084
2121
 
2085
2122
  &nbsp;
2086
2123
 
2087
- LitData comes with built-in logging and profiling capabilities to help you debug and profile your data streaming workloads.
2124
+ `enable_tracer()` records the streaming pipeline (download vs read vs delete vs batch) as one-line events. [Litracer](https://github.com/Lightning-AI/litracer) converts that log into a Chrome / [Perfetto](https://ui.perfetto.dev) trace.
2088
2125
 
2089
- <img width="1439" alt="431247797-0e955e71-2f9a-4aad-b7c1-a8218fed2e2e" src="https://github.com/user-attachments/assets/4e40676c-ba0b-49af-acac-975977173669" />
2126
+ This is complementary to [`profile_batches`](#profile-loading) (viztracer = DataLoader worker **CPU**; Litracer = LitData **pipeline** events).
2090
2127
 
2091
- - e.g., with LitData Streaming
2128
+ <img width="1439" alt="431247797-0e955e71-2f9a-4aad-b7c1-a8218fed2e2e" src="https://github.com/user-attachments/assets/4e40676c-ba0b-49af-acac-975977173669" />
2092
2129
 
2093
2130
  ```python
2094
2131
  import litdata as ld
2095
2132
  from litdata.debugger import enable_tracer
2096
2133
 
2097
- # WARNING: Remove existing trace `litdata_debug.log` file if it exists before re-tracing
2098
- enable_tracer()
2134
+ # Call once per process, before the DataLoader. Delete an existing log before re-tracing (append).
2135
+ enable_tracer(level="chunk", log_file="litdata_debug.log")
2136
+ # level="batch" | "chunk" (default) | "sample" | "debug" | "off"
2137
+ # enable_tracer(categories=["download", "read", "delete"])
2099
2138
 
2100
2139
  if __name__ == "__main__":
2101
2140
  dataset = ld.StreamingDataset("s3://my-bucket/my-data", shuffle=True)
2102
- dataloader = ld.StreamingDataLoader(dataset, batch_size=64)
2103
-
2104
- for batch in dataloader:
2105
- print(batch) # Replace with your data processing logic
2141
+ for batch in ld.StreamingDataLoader(dataset, batch_size=64, num_workers=8):
2142
+ ...
2106
2143
  ```
2107
2144
 
2108
- 1. Generate Debug Log:
2145
+ | Level | Events |
2146
+ | ----- | ------ |
2147
+ | `batch` | Epoch + per-batch spans, plus crashes |
2148
+ | `chunk` (default) | + `download`, `read`, `delete`, `decompress`, `prefetch` |
2149
+ | `sample` | + per-item `__getitem__` (high volume) |
2150
+ | `debug` | + `.cnt` lock refcount spans |
2151
+ | `off` | Disable |
2109
2152
 
2110
- - Run your Python program and it'll create a log file containing detailed debug information.
2153
+ Event **names** are stable (`download`, `read`, `delete`, `batch`, `sample`, `crash`). Chunk / sample indexes live in args so Perfetto groups all downloads together. Each line is `key: value;` pairs with Chrome **microsecond** timestamps. Crashes are a one-line instant (`ph: I`, `name: crash`); the Python traceback is printed to **stderr**, not the log file (a multi-line `logger.exception` would break Litracer).
2111
2154
 
2112
- ```bash
2113
- python main.py
2114
- ```
2155
+ Env overrides: `LITDATA_LOG_FILE`, `LITDATA_TRACE_LEVEL`, `LITDATA_TRACE_CATEGORIES` (comma-separated). Tracer calls are no-ops when tracing is off.
2115
2156
 
2116
- 2. Install [Litracer](https://github.com/deependujha/litracer/):
2157
+ 1. Generate the log:
2117
2158
 
2118
- - Option 1: Using Go (recommended)
2119
- - Install Go on your system.
2120
- - Run the following command to install Litracer:
2159
+ ```bash
2160
+ python train.py # writes litdata_debug.log
2161
+ ```
2121
2162
 
2122
- ```bash
2123
- go install github.com/deependujha/litracer@latest
2124
- ```
2163
+ 2. Install [Litracer](https://github.com/Lightning-AI/litracer) (Go 1.23+):
2125
2164
 
2126
- - Option 2: Download Binary
2127
- - Visit the [LitRacer GitHub Releases](https://github.com/deependujha/litracer/releases) page.
2128
- - Download the appropriate binary for your operating system and follow the installation instructions.
2165
+ ```bash
2166
+ git clone https://github.com/Lightning-AI/litracer.git
2167
+ cd litracer && go build -o litracer .
2168
+ ```
2129
2169
 
2130
- 3. Convert Debug Log to trace JSON:
2170
+ Or `go install github.com/deependujha/litracer@latest` (published Go module path). Until `go.mod` is renamed, `go install github.com/Lightning-AI/litracer@latest` does not work. Release binaries: [GitHub Releases](https://github.com/Lightning-AI/litracer/releases).
2131
2171
 
2132
- - Use litracer to convert the generated log file into a trace JSON file. This command uses 100 workers for conversion:
2172
+ 3. Convert and open in Perfetto:
2133
2173
 
2134
2174
  ```bash
2135
- litracer litdata_debug.log -o litdata_trace.json -w 100
2175
+ litracer --quiet --validate -o litdata_trace.json.gz litdata_debug.log
2176
+ litracer --quiet --cat download,read,delete -o io.json.gz litdata_debug.log
2177
+ # open the .json.gz at https://ui.perfetto.dev (preferred) or chrome://tracing
2136
2178
  ```
2137
2179
 
2138
- 4. Visualize the trace:
2139
-
2140
- - Use either `chrome://tracing` in the Chrome browser or `ui.perfetto.dev` to view the `litdata_trace.json` file for in-depth performance insights. You can also use `SQL queries` to analyze the logs.
2141
- - `Perfetto` is recommended over `chrome://tracing` for visualization & analyzing.
2180
+ `--quiet` prints a one-line summary (per-category durations, unmatched B/E, crashes). `--cat` keeps only those categories. Matched B/E pairs become complete (`ph: X`) spans unless `--no-complete`. Default output is gzip Chrome JSON (`.json.gz`) — both Perfetto and `chrome://tracing` open it; pass `-o file.json` for uncompressed.
2142
2181
 
2143
- - Key Points:
2182
+ - For trace files `> 2GB`, see [Perfetto large traces](https://perfetto.dev/docs/visualization/large-traces).
2183
+ - If you connect Perfetto to the RPC server, prefer Chrome over Brave (Brave often does not autodetect the RPC server).
2144
2184
 
2145
- - For very large trace.json files (`> 2GB`), refer to the [Perfetto documentation](https://perfetto.dev/docs/visualization/large-traces) for using native accelerators.
2146
- - If you are trying to connect Perfetto to the RPC server, it is recommended to use Chrome over Brave, as it has been observed that Perfetto in Brave does not autodetect the RPC server.
2185
+ **Multi-worker `s3://` `FileNotFoundError` after ~120s:** `num_workers=0` working while `num_workers>0` fails usually means the DataLoader parent started obstore (tokio) before fork and worker GETs hung. Current LitData fetches `index.json` with boto3 so workers can lazy-init obstore; they fall back to boto3 if the parent already started the runtime. On Studio R2 / `lightning_storage`, the same symptom can be a prefetch-thread crash (`data_connection_id` / `endpoint_url` into `boto3.Session`) — look for `[litdata] PrepareChunksThread CRASHED` on stderr and a `crash` instant in the trace.
2147
2186
 
2148
2187
  </details>
2149
2188
 
@@ -2317,7 +2356,7 @@ if __name__ == "__main__":
2317
2356
  | `start_method` | spawn† | Multiprocessing start method (†spawn unless IPython) |
2318
2357
  | `optimize_dns` | `None` | Optimized DNS (Studio / cloud) |
2319
2358
  | `storage_options` | `{}` | Cloud credentials / endpoints |
2320
- | `keep_data_ordered` | `True` | `False` = shared work queue (better for uneven/slow workers) |
2359
+ | `keep_data_ordered` | `False` | Shared work queue (faster for uneven workers). `True` keeps a static per-worker slice. Forced `True` with `use_checkpoint` / `align_chunking`. |
2321
2360
 
2322
2361
  </details>
2323
2362
 
@@ -2353,7 +2392,7 @@ Full knob list for `litdata.optimize` (see Quick start for the minimal recipe).
2353
2392
  | `start_method` | spawn† | Multiprocessing start method |
2354
2393
  | `optimize_dns` | `None` | Optimized DNS |
2355
2394
  | `storage_options` | `{}` | Cloud credentials / endpoints |
2356
- | `keep_data_ordered` | `True` | `False` = shared queue among workers |
2395
+ | `keep_data_ordered` | `False` | Shared queue among workers. `True` keeps input order. Forced `True` with `use_checkpoint` / `align_chunking`. |
2357
2396
  | `verbose` | `True` | Progress logging |
2358
2397
 
2359
2398
  Related features: [shared queue](#shared-queue), [queue input](#queue-input), [append/overwrite](#modify-datasets), [compression](#compression), [TokensLoader / LLM](#llm-training), [filter](#filter-data).
@@ -2428,6 +2467,17 @@ Speed to stream Imagenet 1.2M from local disk with ffcv vs LitData:
2428
2467
  | ffcv(os_cache=True) | JPEG 90% | 20 GB | 7653 | 8051 |
2429
2468
  | ffcv(os_cache=False) | JPEG 90% | 20 GB | 8149 | 8607 |
2430
2469
 
2470
+ Speed to stream a **synthetic ImageNet-scale set from Vast NFS** (NFSv3 `nconnect=32`, 208-CPU host, ~1 TiB RAM). Dataset: **1.08M** JPEG q95 256×256 (~160 GiB, 64 MiB chunks). `StreamingDataLoader`, batch **256**, `shuffle=True`, `drop_last=True`, decode only unless noted. POSIX-fast mmaps chunks **in place** (no copy into `~/.lightning/chunks`).
2471
+
2472
+ | Setup | Workers | Images / sec |
2473
+ |---|---|---|
2474
+ | Copy into local cache (`LITDATA_POSIX_FAST=0`) | 48 | **16.7k** (2-epoch avg) |
2475
+ | POSIX-fast (this default on local/Vast paths) | 48 | **18.2k** |
2476
+ | POSIX-fast + README ImageNet augs (crop 224, flip, float32) | 48 | **12.9k** |
2477
+ | POSIX-fast, all CPU cores | **208** | **35.8k** |
2478
+
2479
+ Notes: 208 workers need enough **MemAvailable**. This host had **928×1 GiB hugepages** reserved and idle (~900 GiB locked); after `nr_hugepages=0`, 208 workers stayed healthy. If `num_workers=os.cpu_count()` would crowd RAM, LitData **clamps** workers (`LITDATA_POSIX_MAX_WORKERS=0` disables) and skips `WILLNEED` prefetch. Real ImageNet JPEG 90% is much smaller (~12 GiB) and usually decodes faster than this q95 noise set.
2480
+
2431
2481
  ### Raw Dataset
2432
2482
 
2433
2483
  Speed to stream raw Imagenet 1.2M from different cloud storage providers:
@@ -215,6 +215,22 @@ for sample in dataloader:
215
215
  img, cls = sample["image"], sample["class"]
216
216
  ```
217
217
 
218
+ **Keyed lookup and in-place patches** (needs `polars` and `optimize(..., key_fn=...)` or `build_keys_index`):
219
+
220
+ ```python
221
+ ld.optimize(fn=fn, inputs=inputs, output_dir="fast_data", chunk_bytes="64MB", key_fn=lambda s: s["id"])
222
+
223
+ ds = ld.StreamingDataset("fast_data")
224
+ sample = ds["entity-id"] # str keys
225
+ sample = ds.get_by_key(42) # int entity keys; ds[42] is still positional
226
+
227
+ with ld.dataset_update("fast_data") as update: # local directory only
228
+ update["entity-id"] = {"id": "entity-id", "x": 1}
229
+ update.commit()
230
+ ```
231
+
232
+ `mode="append"` continues chunk numbering. `use_checkpoint=True` tries to resume an interrupted optimize. They are not the same.
233
+
218
234
  **Key benefits:**
219
235
 
220
236
  ✅ **Accelerate training:** Optimized datasets load 20x faster.
@@ -721,6 +737,10 @@ Shuffling is **deterministic** and designed for distributed training:
721
737
 
722
738
  The permutation depends on `seed`, the epoch, and chunk metadata — the same settings always yield the same order (required for resumable `state_dict`).
723
739
 
740
+ **Object storage (`s3://`, `gs://`, …)** globally permutes chunks (`FullShuffle`). Random chunk order is cheap once files are already copied into the local cache.
741
+
742
+ **POSIX-fast** (automatic for any local path) mmaps chunks in place. **Vast / NFS / Lustre / GPFS** (and `LITDATA_POSIX_FAST=1`) use `WindowShuffle`: each worker gets **whole chunks** in a sequential stripe, then shuffles only inside a sliding window (default **16**, `LITDATA_POSIX_SHUFFLE_WINDOW`) for both chunk order and in-chunk items. Local disks (ext4/xfs) keep global `FullShuffle`. Object URLs stay on `FullShuffle`. `LITDATA_POSIX_FAST=0` disables in-place mmap.
743
+
724
744
  ```python
725
745
  from litdata import StreamingDataset, StreamingDataLoader
726
746
 
@@ -789,7 +809,7 @@ Rough ImageNet order-of-magnitude on a Studio (not hard guarantees; right tuning
789
809
  | `seed` | `42` | Shuffle / subsample RNG |
790
810
  | `serializers` | built-ins | Custom serialize/deserialize map |
791
811
  | `max_cache_size` | `"100GB"` | Evict consumed chunks beyond this size |
792
- | `max_pre_download` | `2` | Chunks each worker may prefetch (raise for throughput; watch disk) |
812
+ | `max_pre_download` | `2` | Chunks each worker may prefetch (raise for throughput; watch disk / RAM) |
793
813
  | `subsample` | `1.0` | Fraction of data (`0.01`) or upsample (`2.5`) |
794
814
  | `encryption` | `None` | `FernetEncryption` / `RSAEncryption` / custom |
795
815
  | `storage_options` | `{}` | Cloud client options |
@@ -800,6 +820,8 @@ Rough ImageNet order-of-magnitude on a Studio (not hard guarantees; right tuning
800
820
 
801
821
  Peak disk ≈ `num_workers × max_pre_download × mean_chunk_size`.
802
822
 
823
+ On **Vast / NFS / local disk**, POSIX-fast is on by default (`LITDATA_POSIX_FAST=0` to disable). `WILLNEED` prefetch and `num_workers` are capped when they would exceed about half of `MemAvailable`. Idle **hugepages** (common on GPU nodes) do not count as available RAM — drop unused `nr_hugepages` if `MemAvailable` looks tiny next to `MemTotal`.
824
+
803
825
  **`StreamingDataLoader`**
804
826
 
805
827
  | Argument | Description |
@@ -903,6 +925,12 @@ for batch_idx, batch in enumerate(dataloader):
903
925
  torch.save(dataloader.state_dict(), "dataloader_state.pt")
904
926
  ```
905
927
 
928
+ Same `seed` and `shuffle` are required. For a **`StreamingDataset`**, **`num_workers` and `world_size` may change**: LitData drops a global `sample_in_epoch` prefix and restripes the rest (never duplicates remaining IDs). For a matching loss curve keep **global batch size** (`world_size * batch_size`) constant and DDP ranks in lockstep. `num_canonical_nodes` (default: first-run `world_size`) is frozen in the checkpoint. POSIX `WindowShuffle` resumes whole remaining chunks. **`CombinedStreamingDataset` and `ParallelStreamingDataset` resume only with the same `world_size`, `num_workers`, and `batch_size`.**
929
+
930
+ ```python
931
+ dataset = StreamingDataset("s3://my-bucket/my-data", shuffle=True, num_canonical_nodes=8)
932
+ ```
933
+
906
934
  </details>
907
935
 
908
936
 
@@ -910,11 +938,9 @@ for batch_idx, batch in enumerate(dataloader):
910
938
  <summary> ✅ Use shared queue for Optimizing <a id="shared-queue" href="#shared-queue">🔗</a> </summary>
911
939
  &nbsp;
912
940
 
913
- If you are using multiple workers to optimize your dataset, you can use a shared queue to speed up the process.
941
+ `optimize` / `map` default to a **shared per-node queue** (`keep_data_ordered=False`). Work is packed per node, then every worker on that node pulls the next item, so a slow worker does not leave others idle. Set `keep_data_ordered=True` to keep a static per-worker slice (required for `use_checkpoint` and `align_chunking`).
914
942
 
915
- This is especially useful when optimizing large datasets in parallel, where some workers may be slower than others.
916
-
917
- It can also improve fault tolerance when workers fail due to out-of-memory (OOM) errors.
943
+ Local `output_dir` writes chunks in place. Remote inputs and outputs use the streaming downloader (`adownload_file` / `aupload_file`, obstore when available).
918
944
 
919
945
  ```python
920
946
  import numpy as np
@@ -937,28 +963,36 @@ if __name__ == "__main__":
937
963
  output_dir="fast_data", # optimized data is stored here
938
964
  num_workers=4, # The number of workers on the same machine
939
965
  chunk_bytes="64MB" , # size of each chunk
940
- keep_data_ordered=False, # Use a shared queue to speed up the process
966
+ keep_data_ordered=False, # default: shared queue (set True to keep input order)
941
967
  )
942
968
  ```
943
969
 
944
970
 
945
- ### Performance Difference between using a shared queue and not using it:
971
+ ### Shared queue vs ordered (skewed local files)
946
972
 
947
- **Note**: The following benchmarks were collected using the ImageNet dataset on an A10G machine with 16 workers.
973
+ `scripts/bench/bench_node_queue.py --files 4000 --workers 8` (first 500 files are 1 MiB). On `main`, unordered optimize sat on a 200s empty-queue timeout after work finished.
948
974
 
949
- | Configuration | Optimize Time (sec) | Stream 1 (img/sec) | Stream 2 (img/sec) |
950
- |------------------|---------------------|---------------------|---------------------|
951
- | shared_queue (`keep_data_ordered=False`) | 1281 | 5392 | 5732 |
952
- | no shared_queue (`keep_data_ordered=True (default)`) | 1187 | 5257 | 5746 |
975
+ | Tree | Mode | Time | Throughput |
976
+ |------|------|-----:|-----------:|
977
+ | `main` (old default) | `keep_data_ordered=True` | 23.7s | 169 files/s |
978
+ | `main` | `keep_data_ordered=False` | 223.6s | 18 files/s |
979
+ | this tree | `keep_data_ordered=True` | 22.9s | 175 files/s |
980
+ | this tree (**new default**) | `keep_data_ordered=False` | **18.8s** | 213 files/s |
953
981
 
954
- 📌 Note: The **shared_queue** option impacts optimization time, not streaming speed.
955
- > While the streaming numbers may appear slightly different, this variation is incidental and not caused by shared_queue.
956
- >
957
- > Streaming happens after optimization and does not involve inter-process communication where shared_queue plays a role.
982
+ Shared-queue **before → after: ~12×**. New default vs old ordered default: **1.22×**.
958
983
 
959
- - 📄 Using a shared queue helps balance the load across workers, though it may slightly increase optimization time due to the overhead of pickling items sent between processes.
984
+ ### Local / remote input × output
960
985
 
961
- - ⚡ However, it can significantly improve optimizing performance — especially when some workers are slower than others.
986
+ `python scripts/bench/bench_node_queue.py --files 200 --workers 4 --io-matrix` (first 50 files are 1 MiB).
987
+
988
+ | Topology | Ordered | Shared | Speedup |
989
+ |----------|--------:|-------:|--------:|
990
+ | local → local | 6.55s | **2.96s** | 2.21× |
991
+ | remote → local | 7.44s | **3.48s** | 2.14× |
992
+ | local → remote | 10.98s | **6.51s** | 1.69× |
993
+ | remote → remote | 11.48s | **8.55s** | 1.34× |
994
+
995
+ Shared queue balances uneven workers. It does not change later `StreamingDataset` throughput.
962
996
 
963
997
  </details>
964
998
 
@@ -1742,7 +1776,7 @@ Only **worker 0** is instrumented. When an `int` is used, the tracer wraps `fetc
1742
1776
 
1743
1777
  - Delete or change `profile_dir` between runs — LitData removes an existing `result.json` before starting.
1744
1778
  - Pair with a wiped chunk cache if you care about **cold** epoch behavior (`litdata cache clear`).
1745
- - For deeper LitData internals (download / lock / delete timeline), use `enable_tracer()` + [Litracer](https://github.com/deependujha/litracer) instead — see [Debug & Profile LitData](#debug-profile). That path is complementary: viztracer = DataLoader worker CPU timeline; Litracer = LitData pipeline events.
1779
+ - For deeper LitData internals (download / read / delete timeline), use `enable_tracer()` + [Litracer](https://github.com/Lightning-AI/litracer) instead — see [Debug & Profile LitData](#debug-profile). That path is complementary: viztracer = DataLoader worker CPU timeline; Litracer = LitData pipeline events.
1746
1780
 
1747
1781
  </details>
1748
1782
 
@@ -1855,6 +1889,9 @@ export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
1855
1889
  | `LITDATA_DISABLE_VERSION_CHECK` | `0` | `1` skips the upgrade tip |
1856
1890
  | `HF_TOKEN` | — | Gated Hugging Face datasets |
1857
1891
  | `DEBUG_LITDATA` / `PRINT_DEBUG_LOGS` | `0` | Internal debug / stdout logs |
1892
+ | `LITDATA_LOG_FILE` | `litdata_debug.log` | `enable_tracer()` output path |
1893
+ | `LITDATA_TRACE_LEVEL` | unset | `batch` / `chunk` / `sample` / `debug` / `off` (see [Debug & Profile](#debug-profile)) |
1894
+ | `LITDATA_TRACE_CATEGORIES` | from level | Comma-separated cats, e.g. `download,read,delete` |
1858
1895
 
1859
1896
  Multi-node `optimize`/`map` on Studios also uses `DATA_OPTIMIZER_*` (set by the platform). Full catalog (debug logs, Studio injects, torchrun): see the LitData skill `reference/env-vars.md` when using agent skills, or the source modules `constants.py` / `async_prefetch.py`.
1860
1897
 
@@ -2020,66 +2057,68 @@ ds = StreamingDataset(input_dir=data_dir, encryption=rsa)
2020
2057
 
2021
2058
  &nbsp;
2022
2059
 
2023
- LitData comes with built-in logging and profiling capabilities to help you debug and profile your data streaming workloads.
2060
+ `enable_tracer()` records the streaming pipeline (download vs read vs delete vs batch) as one-line events. [Litracer](https://github.com/Lightning-AI/litracer) converts that log into a Chrome / [Perfetto](https://ui.perfetto.dev) trace.
2024
2061
 
2025
- <img width="1439" alt="431247797-0e955e71-2f9a-4aad-b7c1-a8218fed2e2e" src="https://github.com/user-attachments/assets/4e40676c-ba0b-49af-acac-975977173669" />
2062
+ This is complementary to [`profile_batches`](#profile-loading) (viztracer = DataLoader worker **CPU**; Litracer = LitData **pipeline** events).
2026
2063
 
2027
- - e.g., with LitData Streaming
2064
+ <img width="1439" alt="431247797-0e955e71-2f9a-4aad-b7c1-a8218fed2e2e" src="https://github.com/user-attachments/assets/4e40676c-ba0b-49af-acac-975977173669" />
2028
2065
 
2029
2066
  ```python
2030
2067
  import litdata as ld
2031
2068
  from litdata.debugger import enable_tracer
2032
2069
 
2033
- # WARNING: Remove existing trace `litdata_debug.log` file if it exists before re-tracing
2034
- enable_tracer()
2070
+ # Call once per process, before the DataLoader. Delete an existing log before re-tracing (append).
2071
+ enable_tracer(level="chunk", log_file="litdata_debug.log")
2072
+ # level="batch" | "chunk" (default) | "sample" | "debug" | "off"
2073
+ # enable_tracer(categories=["download", "read", "delete"])
2035
2074
 
2036
2075
  if __name__ == "__main__":
2037
2076
  dataset = ld.StreamingDataset("s3://my-bucket/my-data", shuffle=True)
2038
- dataloader = ld.StreamingDataLoader(dataset, batch_size=64)
2039
-
2040
- for batch in dataloader:
2041
- print(batch) # Replace with your data processing logic
2077
+ for batch in ld.StreamingDataLoader(dataset, batch_size=64, num_workers=8):
2078
+ ...
2042
2079
  ```
2043
2080
 
2044
- 1. Generate Debug Log:
2081
+ | Level | Events |
2082
+ | ----- | ------ |
2083
+ | `batch` | Epoch + per-batch spans, plus crashes |
2084
+ | `chunk` (default) | + `download`, `read`, `delete`, `decompress`, `prefetch` |
2085
+ | `sample` | + per-item `__getitem__` (high volume) |
2086
+ | `debug` | + `.cnt` lock refcount spans |
2087
+ | `off` | Disable |
2045
2088
 
2046
- - Run your Python program and it'll create a log file containing detailed debug information.
2089
+ Event **names** are stable (`download`, `read`, `delete`, `batch`, `sample`, `crash`). Chunk / sample indexes live in args so Perfetto groups all downloads together. Each line is `key: value;` pairs with Chrome **microsecond** timestamps. Crashes are a one-line instant (`ph: I`, `name: crash`); the Python traceback is printed to **stderr**, not the log file (a multi-line `logger.exception` would break Litracer).
2047
2090
 
2048
- ```bash
2049
- python main.py
2050
- ```
2091
+ Env overrides: `LITDATA_LOG_FILE`, `LITDATA_TRACE_LEVEL`, `LITDATA_TRACE_CATEGORIES` (comma-separated). Tracer calls are no-ops when tracing is off.
2051
2092
 
2052
- 2. Install [Litracer](https://github.com/deependujha/litracer/):
2093
+ 1. Generate the log:
2053
2094
 
2054
- - Option 1: Using Go (recommended)
2055
- - Install Go on your system.
2056
- - Run the following command to install Litracer:
2095
+ ```bash
2096
+ python train.py # writes litdata_debug.log
2097
+ ```
2057
2098
 
2058
- ```bash
2059
- go install github.com/deependujha/litracer@latest
2060
- ```
2099
+ 2. Install [Litracer](https://github.com/Lightning-AI/litracer) (Go 1.23+):
2061
2100
 
2062
- - Option 2: Download Binary
2063
- - Visit the [LitRacer GitHub Releases](https://github.com/deependujha/litracer/releases) page.
2064
- - Download the appropriate binary for your operating system and follow the installation instructions.
2101
+ ```bash
2102
+ git clone https://github.com/Lightning-AI/litracer.git
2103
+ cd litracer && go build -o litracer .
2104
+ ```
2065
2105
 
2066
- 3. Convert Debug Log to trace JSON:
2106
+ Or `go install github.com/deependujha/litracer@latest` (published Go module path). Until `go.mod` is renamed, `go install github.com/Lightning-AI/litracer@latest` does not work. Release binaries: [GitHub Releases](https://github.com/Lightning-AI/litracer/releases).
2067
2107
 
2068
- - Use litracer to convert the generated log file into a trace JSON file. This command uses 100 workers for conversion:
2108
+ 3. Convert and open in Perfetto:
2069
2109
 
2070
2110
  ```bash
2071
- litracer litdata_debug.log -o litdata_trace.json -w 100
2111
+ litracer --quiet --validate -o litdata_trace.json.gz litdata_debug.log
2112
+ litracer --quiet --cat download,read,delete -o io.json.gz litdata_debug.log
2113
+ # open the .json.gz at https://ui.perfetto.dev (preferred) or chrome://tracing
2072
2114
  ```
2073
2115
 
2074
- 4. Visualize the trace:
2075
-
2076
- - Use either `chrome://tracing` in the Chrome browser or `ui.perfetto.dev` to view the `litdata_trace.json` file for in-depth performance insights. You can also use `SQL queries` to analyze the logs.
2077
- - `Perfetto` is recommended over `chrome://tracing` for visualization & analyzing.
2116
+ `--quiet` prints a one-line summary (per-category durations, unmatched B/E, crashes). `--cat` keeps only those categories. Matched B/E pairs become complete (`ph: X`) spans unless `--no-complete`. Default output is gzip Chrome JSON (`.json.gz`) — both Perfetto and `chrome://tracing` open it; pass `-o file.json` for uncompressed.
2078
2117
 
2079
- - Key Points:
2118
+ - For trace files `> 2GB`, see [Perfetto large traces](https://perfetto.dev/docs/visualization/large-traces).
2119
+ - If you connect Perfetto to the RPC server, prefer Chrome over Brave (Brave often does not autodetect the RPC server).
2080
2120
 
2081
- - For very large trace.json files (`> 2GB`), refer to the [Perfetto documentation](https://perfetto.dev/docs/visualization/large-traces) for using native accelerators.
2082
- - If you are trying to connect Perfetto to the RPC server, it is recommended to use Chrome over Brave, as it has been observed that Perfetto in Brave does not autodetect the RPC server.
2121
+ **Multi-worker `s3://` `FileNotFoundError` after ~120s:** `num_workers=0` working while `num_workers>0` fails usually means the DataLoader parent started obstore (tokio) before fork and worker GETs hung. Current LitData fetches `index.json` with boto3 so workers can lazy-init obstore; they fall back to boto3 if the parent already started the runtime. On Studio R2 / `lightning_storage`, the same symptom can be a prefetch-thread crash (`data_connection_id` / `endpoint_url` into `boto3.Session`) — look for `[litdata] PrepareChunksThread CRASHED` on stderr and a `crash` instant in the trace.
2083
2122
 
2084
2123
  </details>
2085
2124
 
@@ -2253,7 +2292,7 @@ if __name__ == "__main__":
2253
2292
  | `start_method` | spawn† | Multiprocessing start method (†spawn unless IPython) |
2254
2293
  | `optimize_dns` | `None` | Optimized DNS (Studio / cloud) |
2255
2294
  | `storage_options` | `{}` | Cloud credentials / endpoints |
2256
- | `keep_data_ordered` | `True` | `False` = shared work queue (better for uneven/slow workers) |
2295
+ | `keep_data_ordered` | `False` | Shared work queue (faster for uneven workers). `True` keeps a static per-worker slice. Forced `True` with `use_checkpoint` / `align_chunking`. |
2257
2296
 
2258
2297
  </details>
2259
2298
 
@@ -2289,7 +2328,7 @@ Full knob list for `litdata.optimize` (see Quick start for the minimal recipe).
2289
2328
  | `start_method` | spawn† | Multiprocessing start method |
2290
2329
  | `optimize_dns` | `None` | Optimized DNS |
2291
2330
  | `storage_options` | `{}` | Cloud credentials / endpoints |
2292
- | `keep_data_ordered` | `True` | `False` = shared queue among workers |
2331
+ | `keep_data_ordered` | `False` | Shared queue among workers. `True` keeps input order. Forced `True` with `use_checkpoint` / `align_chunking`. |
2293
2332
  | `verbose` | `True` | Progress logging |
2294
2333
 
2295
2334
  Related features: [shared queue](#shared-queue), [queue input](#queue-input), [append/overwrite](#modify-datasets), [compression](#compression), [TokensLoader / LLM](#llm-training), [filter](#filter-data).
@@ -2364,6 +2403,17 @@ Speed to stream Imagenet 1.2M from local disk with ffcv vs LitData:
2364
2403
  | ffcv(os_cache=True) | JPEG 90% | 20 GB | 7653 | 8051 |
2365
2404
  | ffcv(os_cache=False) | JPEG 90% | 20 GB | 8149 | 8607 |
2366
2405
 
2406
+ Speed to stream a **synthetic ImageNet-scale set from Vast NFS** (NFSv3 `nconnect=32`, 208-CPU host, ~1 TiB RAM). Dataset: **1.08M** JPEG q95 256×256 (~160 GiB, 64 MiB chunks). `StreamingDataLoader`, batch **256**, `shuffle=True`, `drop_last=True`, decode only unless noted. POSIX-fast mmaps chunks **in place** (no copy into `~/.lightning/chunks`).
2407
+
2408
+ | Setup | Workers | Images / sec |
2409
+ |---|---|---|
2410
+ | Copy into local cache (`LITDATA_POSIX_FAST=0`) | 48 | **16.7k** (2-epoch avg) |
2411
+ | POSIX-fast (this default on local/Vast paths) | 48 | **18.2k** |
2412
+ | POSIX-fast + README ImageNet augs (crop 224, flip, float32) | 48 | **12.9k** |
2413
+ | POSIX-fast, all CPU cores | **208** | **35.8k** |
2414
+
2415
+ Notes: 208 workers need enough **MemAvailable**. This host had **928×1 GiB hugepages** reserved and idle (~900 GiB locked); after `nr_hugepages=0`, 208 workers stayed healthy. If `num_workers=os.cpu_count()` would crowd RAM, LitData **clamps** workers (`LITDATA_POSIX_MAX_WORKERS=0` disables) and skips `WILLNEED` prefetch. Real ImageNet JPEG 90% is much smaller (~12 GiB) and usually decodes faster than this q95 noise set.
2416
+
2367
2417
  ### Raw Dataset
2368
2418
 
2369
2419
  Speed to stream raw Imagenet 1.2M from different cloud storage providers:
@@ -14,7 +14,7 @@
14
14
 
15
15
  import time
16
16
 
17
- __version__ = "0.2.67"
17
+ __version__ = "0.2.69"
18
18
  __author__ = "Lightning AI et al."
19
19
  __author_email__ = "pytorch@lightning.ai"
20
20
  __license__ = "Apache-2.0"