litdata 0.2.67__tar.gz → 0.2.68__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. {litdata-0.2.67 → litdata-0.2.68}/CONTRIBUTING.md +1 -1
  2. {litdata-0.2.67/src/litdata.egg-info → litdata-0.2.68}/PKG-INFO +75 -37
  3. {litdata-0.2.67 → litdata-0.2.68}/README.md +74 -36
  4. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/__about__.py +1 -1
  5. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/__init__.py +6 -0
  6. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/constants.py +6 -0
  7. litdata-0.2.68/src/litdata/debugger.py +397 -0
  8. litdata-0.2.68/src/litdata/exceptions.py +36 -0
  9. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/processing/data_processor.py +207 -16
  10. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/processing/functions.py +35 -5
  11. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/async_prefetch.py +10 -2
  12. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/cache.py +2 -2
  13. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/client.py +11 -0
  14. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/combined.py +3 -12
  15. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/compression.py +28 -8
  16. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/config.py +9 -17
  17. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/dataloader.py +24 -3
  18. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/dataset.py +108 -9
  19. litdata-0.2.68/src/litdata/streaming/dataset_update.py +299 -0
  20. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/downloader.py +230 -105
  21. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/item_loader.py +317 -118
  22. litdata-0.2.68/src/litdata/streaming/posix_fast.py +396 -0
  23. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/reader.py +117 -36
  24. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/shuffle.py +52 -0
  25. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/writer.py +26 -13
  26. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/format.py +1 -1
  27. litdata-0.2.68/src/litdata/utilities/keys_index.py +868 -0
  28. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/shuffle.py +114 -0
  29. {litdata-0.2.67 → litdata-0.2.68/src/litdata.egg-info}/PKG-INFO +75 -37
  30. {litdata-0.2.67 → litdata-0.2.68}/src/litdata.egg-info/SOURCES.txt +4 -0
  31. litdata-0.2.67/src/litdata/debugger.py +0 -205
  32. {litdata-0.2.67 → litdata-0.2.68}/LICENSE +0 -0
  33. {litdata-0.2.67 → litdata-0.2.68}/MANIFEST.in +0 -0
  34. {litdata-0.2.67 → litdata-0.2.68}/requirements.txt +0 -0
  35. {litdata-0.2.67 → litdata-0.2.68}/setup.cfg +0 -0
  36. {litdata-0.2.67 → litdata-0.2.68}/setup.py +0 -0
  37. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/__main__.py +0 -0
  38. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/cli/__init__.py +0 -0
  39. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/cli/commands.py +0 -0
  40. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/cli/handler/__init__.py +0 -0
  41. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/cli/handler/cache.py +0 -0
  42. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/cli/handler/optimize.py +0 -0
  43. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/cli/parser.py +0 -0
  44. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/helpers.py +0 -0
  45. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/imports.py +0 -0
  46. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/processing/__init__.py +0 -0
  47. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/processing/readers.py +0 -0
  48. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/processing/utilities.py +0 -0
  49. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/raw/__init__.py +0 -0
  50. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/raw/dataset.py +0 -0
  51. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/raw/indexer.py +0 -0
  52. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/raw/types.py +0 -0
  53. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/requirements.py +0 -0
  54. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/__init__.py +0 -0
  55. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/fs_provider.py +0 -0
  56. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/parallel.py +0 -0
  57. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/resolver.py +0 -0
  58. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/sampler.py +0 -0
  59. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/serializers.py +0 -0
  60. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/timing.py +0 -0
  61. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/__init__.py +0 -0
  62. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/_pytree.py +0 -0
  63. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/base.py +0 -0
  64. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/breakpoint.py +0 -0
  65. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/broadcast.py +0 -0
  66. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/dataset_utilities.py +0 -0
  67. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/encryption.py +0 -0
  68. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/env.py +0 -0
  69. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/hf_dataset.py +0 -0
  70. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/packing.py +0 -0
  71. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/parquet.py +0 -0
  72. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/subsample.py +0 -0
  73. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/torch_utils.py +0 -0
  74. {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/train_test_split.py +0 -0
  75. {litdata-0.2.67 → litdata-0.2.68}/src/litdata.egg-info/dependency_links.txt +0 -0
  76. {litdata-0.2.67 → litdata-0.2.68}/src/litdata.egg-info/entry_points.txt +0 -0
  77. {litdata-0.2.67 → litdata-0.2.68}/src/litdata.egg-info/not-zip-safe +0 -0
  78. {litdata-0.2.67 → litdata-0.2.68}/src/litdata.egg-info/requires.txt +0 -0
  79. {litdata-0.2.67 → litdata-0.2.68}/src/litdata.egg-info/top_level.txt +0 -0
@@ -2,7 +2,7 @@
2
2
 
3
3
  Welcome to the PyTorch Lightning community! We're building the most advanced research platform on the planet to implement the latest, best practices and integrations that the amazing PyTorch team and other research organization rolls out!
4
4
 
5
- If you are new to open source, check out [this blog to get started with your first Open Source contribution](https://devblog.pytorchlightning.ai/quick-contribution-guide-86d977171b3a).
5
+ If you are new to open source, check out [GitHub's guide to making your first contribution](https://docs.github.com/en/get-started/quickstart/contributing-to-projects).
6
6
 
7
7
  ## Main Core Value: One less thing to remember
8
8
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: litdata
3
- Version: 0.2.67
3
+ Version: 0.2.68
4
4
  Summary: The Deep Learning framework to train, deploy, and ship AI products Lightning fast.
5
5
  Home-page: https://github.com/Lightning-AI/litdata
6
6
  Download-URL: https://github.com/Lightning-AI/litdata
@@ -279,6 +279,22 @@ for sample in dataloader:
279
279
  img, cls = sample["image"], sample["class"]
280
280
  ```
281
281
 
282
+ **Keyed lookup and in-place patches** (needs `polars` and `optimize(..., key_fn=...)` or `build_keys_index`):
283
+
284
+ ```python
285
+ ld.optimize(fn=fn, inputs=inputs, output_dir="fast_data", chunk_bytes="64MB", key_fn=lambda s: s["id"])
286
+
287
+ ds = ld.StreamingDataset("fast_data")
288
+ sample = ds["entity-id"] # str keys
289
+ sample = ds.get_by_key(42) # int entity keys; ds[42] is still positional
290
+
291
+ with ld.dataset_update("fast_data") as update: # local directory only
292
+ update["entity-id"] = {"id": "entity-id", "x": 1}
293
+ update.commit()
294
+ ```
295
+
296
+ `mode="append"` continues chunk numbering. `use_checkpoint=True` tries to resume an interrupted optimize. They are not the same.
297
+
282
298
  **Key benefits:**
283
299
 
284
300
  ✅ **Accelerate training:** Optimized datasets load 20x faster.
@@ -785,6 +801,10 @@ Shuffling is **deterministic** and designed for distributed training:
785
801
 
786
802
  The permutation depends on `seed`, the epoch, and chunk metadata — the same settings always yield the same order (required for resumable `state_dict`).
787
803
 
804
+ **Object storage (`s3://`, `gs://`, …)** globally permutes chunks (`FullShuffle`). Random chunk order is cheap once files are already copied into the local cache.
805
+
806
+ **POSIX-fast** (automatic for any local path) mmaps chunks in place. **Vast / NFS / Lustre / GPFS** (and `LITDATA_POSIX_FAST=1`) use `WindowShuffle`: each worker gets **whole chunks** in a sequential stripe, then shuffles only inside a sliding window (default **16**, `LITDATA_POSIX_SHUFFLE_WINDOW`) for both chunk order and in-chunk items. Local disks (ext4/xfs) keep global `FullShuffle`. Object URLs stay on `FullShuffle`. `LITDATA_POSIX_FAST=0` disables in-place mmap.
807
+
788
808
  ```python
789
809
  from litdata import StreamingDataset, StreamingDataLoader
790
810
 
@@ -853,7 +873,7 @@ Rough ImageNet order-of-magnitude on a Studio (not hard guarantees; right tuning
853
873
  | `seed` | `42` | Shuffle / subsample RNG |
854
874
  | `serializers` | built-ins | Custom serialize/deserialize map |
855
875
  | `max_cache_size` | `"100GB"` | Evict consumed chunks beyond this size |
856
- | `max_pre_download` | `2` | Chunks each worker may prefetch (raise for throughput; watch disk) |
876
+ | `max_pre_download` | `2` | Chunks each worker may prefetch (raise for throughput; watch disk / RAM) |
857
877
  | `subsample` | `1.0` | Fraction of data (`0.01`) or upsample (`2.5`) |
858
878
  | `encryption` | `None` | `FernetEncryption` / `RSAEncryption` / custom |
859
879
  | `storage_options` | `{}` | Cloud client options |
@@ -864,6 +884,8 @@ Rough ImageNet order-of-magnitude on a Studio (not hard guarantees; right tuning
864
884
 
865
885
  Peak disk ≈ `num_workers × max_pre_download × mean_chunk_size`.
866
886
 
887
+ On **Vast / NFS / local disk**, POSIX-fast is on by default (`LITDATA_POSIX_FAST=0` to disable). `WILLNEED` prefetch and `num_workers` are capped when they would exceed about half of `MemAvailable`. Idle **hugepages** (common on GPU nodes) do not count as available RAM — drop unused `nr_hugepages` if `MemAvailable` looks tiny next to `MemTotal`.
888
+
867
889
  **`StreamingDataLoader`**
868
890
 
869
891
  | Argument | Description |
@@ -1806,7 +1828,7 @@ Only **worker 0** is instrumented. When an `int` is used, the tracer wraps `fetc
1806
1828
 
1807
1829
  - Delete or change `profile_dir` between runs — LitData removes an existing `result.json` before starting.
1808
1830
  - Pair with a wiped chunk cache if you care about **cold** epoch behavior (`litdata cache clear`).
1809
- - For deeper LitData internals (download / lock / delete timeline), use `enable_tracer()` + [Litracer](https://github.com/deependujha/litracer) instead — see [Debug & Profile LitData](#debug-profile). That path is complementary: viztracer = DataLoader worker CPU timeline; Litracer = LitData pipeline events.
1831
+ - For deeper LitData internals (download / read / delete timeline), use `enable_tracer()` + [Litracer](https://github.com/Lightning-AI/litracer) instead — see [Debug & Profile LitData](#debug-profile). That path is complementary: viztracer = DataLoader worker CPU timeline; Litracer = LitData pipeline events.
1810
1832
 
1811
1833
  </details>
1812
1834
 
@@ -1919,6 +1941,9 @@ export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
1919
1941
  | `LITDATA_DISABLE_VERSION_CHECK` | `0` | `1` skips the upgrade tip |
1920
1942
  | `HF_TOKEN` | — | Gated Hugging Face datasets |
1921
1943
  | `DEBUG_LITDATA` / `PRINT_DEBUG_LOGS` | `0` | Internal debug / stdout logs |
1944
+ | `LITDATA_LOG_FILE` | `litdata_debug.log` | `enable_tracer()` output path |
1945
+ | `LITDATA_TRACE_LEVEL` | unset | `batch` / `chunk` / `sample` / `debug` / `off` (see [Debug & Profile](#debug-profile)) |
1946
+ | `LITDATA_TRACE_CATEGORIES` | from level | Comma-separated cats, e.g. `download,read,delete` |
1922
1947
 
1923
1948
  Multi-node `optimize`/`map` on Studios also uses `DATA_OPTIMIZER_*` (set by the platform). Full catalog (debug logs, Studio injects, torchrun): see the LitData skill `reference/env-vars.md` when using agent skills, or the source modules `constants.py` / `async_prefetch.py`.
1924
1949
 
@@ -2084,66 +2109,68 @@ ds = StreamingDataset(input_dir=data_dir, encryption=rsa)
2084
2109
 
2085
2110
  &nbsp;
2086
2111
 
2087
- LitData comes with built-in logging and profiling capabilities to help you debug and profile your data streaming workloads.
2112
+ `enable_tracer()` records the streaming pipeline (download vs read vs delete vs batch) as one-line events. [Litracer](https://github.com/Lightning-AI/litracer) converts that log into a Chrome / [Perfetto](https://ui.perfetto.dev) trace.
2088
2113
 
2089
- <img width="1439" alt="431247797-0e955e71-2f9a-4aad-b7c1-a8218fed2e2e" src="https://github.com/user-attachments/assets/4e40676c-ba0b-49af-acac-975977173669" />
2114
+ This is complementary to [`profile_batches`](#profile-loading) (viztracer = DataLoader worker **CPU**; Litracer = LitData **pipeline** events).
2090
2115
 
2091
- - e.g., with LitData Streaming
2116
+ <img width="1439" alt="431247797-0e955e71-2f9a-4aad-b7c1-a8218fed2e2e" src="https://github.com/user-attachments/assets/4e40676c-ba0b-49af-acac-975977173669" />
2092
2117
 
2093
2118
  ```python
2094
2119
  import litdata as ld
2095
2120
  from litdata.debugger import enable_tracer
2096
2121
 
2097
- # WARNING: Remove existing trace `litdata_debug.log` file if it exists before re-tracing
2098
- enable_tracer()
2122
+ # Call once per process, before the DataLoader. Delete an existing log before re-tracing (append).
2123
+ enable_tracer(level="chunk", log_file="litdata_debug.log")
2124
+ # level="batch" | "chunk" (default) | "sample" | "debug" | "off"
2125
+ # enable_tracer(categories=["download", "read", "delete"])
2099
2126
 
2100
2127
  if __name__ == "__main__":
2101
2128
  dataset = ld.StreamingDataset("s3://my-bucket/my-data", shuffle=True)
2102
- dataloader = ld.StreamingDataLoader(dataset, batch_size=64)
2103
-
2104
- for batch in dataloader:
2105
- print(batch) # Replace with your data processing logic
2129
+ for batch in ld.StreamingDataLoader(dataset, batch_size=64, num_workers=8):
2130
+ ...
2106
2131
  ```
2107
2132
 
2108
- 1. Generate Debug Log:
2133
+ | Level | Events |
2134
+ | ----- | ------ |
2135
+ | `batch` | Epoch + per-batch spans, plus crashes |
2136
+ | `chunk` (default) | + `download`, `read`, `delete`, `decompress`, `prefetch` |
2137
+ | `sample` | + per-item `__getitem__` (high volume) |
2138
+ | `debug` | + `.cnt` lock refcount spans |
2139
+ | `off` | Disable |
2109
2140
 
2110
- - Run your Python program and it'll create a log file containing detailed debug information.
2141
+ Event **names** are stable (`download`, `read`, `delete`, `batch`, `sample`, `crash`). Chunk / sample indexes live in args so Perfetto groups all downloads together. Each line is `key: value;` pairs with Chrome **microsecond** timestamps. Crashes are a one-line instant (`ph: I`, `name: crash`); the Python traceback is printed to **stderr**, not the log file (a multi-line `logger.exception` would break Litracer).
2111
2142
 
2112
- ```bash
2113
- python main.py
2114
- ```
2143
+ Env overrides: `LITDATA_LOG_FILE`, `LITDATA_TRACE_LEVEL`, `LITDATA_TRACE_CATEGORIES` (comma-separated). Tracer calls are no-ops when tracing is off.
2115
2144
 
2116
- 2. Install [Litracer](https://github.com/deependujha/litracer/):
2145
+ 1. Generate the log:
2117
2146
 
2118
- - Option 1: Using Go (recommended)
2119
- - Install Go on your system.
2120
- - Run the following command to install Litracer:
2147
+ ```bash
2148
+ python train.py # writes litdata_debug.log
2149
+ ```
2121
2150
 
2122
- ```bash
2123
- go install github.com/deependujha/litracer@latest
2124
- ```
2151
+ 2. Install [Litracer](https://github.com/Lightning-AI/litracer) (Go 1.23+):
2125
2152
 
2126
- - Option 2: Download Binary
2127
- - Visit the [LitRacer GitHub Releases](https://github.com/deependujha/litracer/releases) page.
2128
- - Download the appropriate binary for your operating system and follow the installation instructions.
2153
+ ```bash
2154
+ git clone https://github.com/Lightning-AI/litracer.git
2155
+ cd litracer && go build -o litracer .
2156
+ ```
2129
2157
 
2130
- 3. Convert Debug Log to trace JSON:
2158
+ Or `go install github.com/deependujha/litracer@latest` (published Go module path). Until `go.mod` is renamed, `go install github.com/Lightning-AI/litracer@latest` does not work. Release binaries: [GitHub Releases](https://github.com/Lightning-AI/litracer/releases).
2131
2159
 
2132
- - Use litracer to convert the generated log file into a trace JSON file. This command uses 100 workers for conversion:
2160
+ 3. Convert and open in Perfetto:
2133
2161
 
2134
2162
  ```bash
2135
- litracer litdata_debug.log -o litdata_trace.json -w 100
2163
+ litracer --quiet --validate -o litdata_trace.json.gz litdata_debug.log
2164
+ litracer --quiet --cat download,read,delete -o io.json.gz litdata_debug.log
2165
+ # open the .json.gz at https://ui.perfetto.dev (preferred) or chrome://tracing
2136
2166
  ```
2137
2167
 
2138
- 4. Visualize the trace:
2139
-
2140
- - Use either `chrome://tracing` in the Chrome browser or `ui.perfetto.dev` to view the `litdata_trace.json` file for in-depth performance insights. You can also use `SQL queries` to analyze the logs.
2141
- - `Perfetto` is recommended over `chrome://tracing` for visualization & analyzing.
2168
+ `--quiet` prints a one-line summary (per-category durations, unmatched B/E, crashes). `--cat` keeps only those categories. Matched B/E pairs become complete (`ph: X`) spans unless `--no-complete`. Default output is gzip Chrome JSON (`.json.gz`) — both Perfetto and `chrome://tracing` open it; pass `-o file.json` for uncompressed.
2142
2169
 
2143
- - Key Points:
2170
+ - For trace files `> 2GB`, see [Perfetto large traces](https://perfetto.dev/docs/visualization/large-traces).
2171
+ - If you connect Perfetto to the RPC server, prefer Chrome over Brave (Brave often does not autodetect the RPC server).
2144
2172
 
2145
- - For very large trace.json files (`> 2GB`), refer to the [Perfetto documentation](https://perfetto.dev/docs/visualization/large-traces) for using native accelerators.
2146
- - If you are trying to connect Perfetto to the RPC server, it is recommended to use Chrome over Brave, as it has been observed that Perfetto in Brave does not autodetect the RPC server.
2173
+ **Multi-worker `s3://` `FileNotFoundError` after ~120s:** `num_workers=0` working while `num_workers>0` fails usually means the DataLoader parent started obstore (tokio) before fork and worker GETs hung. Current LitData fetches `index.json` with boto3 so workers can lazy-init obstore; they fall back to boto3 if the parent already started the runtime. On Studio R2 / `lightning_storage`, the same symptom can be a prefetch-thread crash (`data_connection_id` / `endpoint_url` into `boto3.Session`) — look for `[litdata] PrepareChunksThread CRASHED` on stderr and a `crash` instant in the trace.
2147
2174
 
2148
2175
  </details>
2149
2176
 
@@ -2428,6 +2455,17 @@ Speed to stream Imagenet 1.2M from local disk with ffcv vs LitData:
2428
2455
  | ffcv(os_cache=True) | JPEG 90% | 20 GB | 7653 | 8051 |
2429
2456
  | ffcv(os_cache=False) | JPEG 90% | 20 GB | 8149 | 8607 |
2430
2457
 
2458
+ Speed to stream a **synthetic ImageNet-scale set from Vast NFS** (NFSv3 `nconnect=32`, 208-CPU host, ~1 TiB RAM). Dataset: **1.08M** JPEG q95 256×256 (~160 GiB, 64 MiB chunks). `StreamingDataLoader`, batch **256**, `shuffle=True`, `drop_last=True`, decode only unless noted. POSIX-fast mmaps chunks **in place** (no copy into `~/.lightning/chunks`).
2459
+
2460
+ | Setup | Workers | Images / sec |
2461
+ |---|---|---|
2462
+ | Copy into local cache (`LITDATA_POSIX_FAST=0`) | 48 | **16.7k** (2-epoch avg) |
2463
+ | POSIX-fast (this default on local/Vast paths) | 48 | **18.2k** |
2464
+ | POSIX-fast + README ImageNet augs (crop 224, flip, float32) | 48 | **12.9k** |
2465
+ | POSIX-fast, all CPU cores | **208** | **35.8k** |
2466
+
2467
+ Notes: 208 workers need enough **MemAvailable**. This host had **928×1 GiB hugepages** reserved and idle (~900 GiB locked); after `nr_hugepages=0`, 208 workers stayed healthy. If `num_workers=os.cpu_count()` would crowd RAM, LitData **clamps** workers (`LITDATA_POSIX_MAX_WORKERS=0` disables) and skips `WILLNEED` prefetch. Real ImageNet JPEG 90% is much smaller (~12 GiB) and usually decodes faster than this q95 noise set.
2468
+
2431
2469
  ### Raw Dataset
2432
2470
 
2433
2471
  Speed to stream raw Imagenet 1.2M from different cloud storage providers:
@@ -215,6 +215,22 @@ for sample in dataloader:
215
215
  img, cls = sample["image"], sample["class"]
216
216
  ```
217
217
 
218
+ **Keyed lookup and in-place patches** (needs `polars` and `optimize(..., key_fn=...)` or `build_keys_index`):
219
+
220
+ ```python
221
+ ld.optimize(fn=fn, inputs=inputs, output_dir="fast_data", chunk_bytes="64MB", key_fn=lambda s: s["id"])
222
+
223
+ ds = ld.StreamingDataset("fast_data")
224
+ sample = ds["entity-id"] # str keys
225
+ sample = ds.get_by_key(42) # int entity keys; ds[42] is still positional
226
+
227
+ with ld.dataset_update("fast_data") as update: # local directory only
228
+ update["entity-id"] = {"id": "entity-id", "x": 1}
229
+ update.commit()
230
+ ```
231
+
232
+ `mode="append"` continues chunk numbering. `use_checkpoint=True` tries to resume an interrupted optimize. They are not the same.
233
+
218
234
  **Key benefits:**
219
235
 
220
236
  ✅ **Accelerate training:** Optimized datasets load 20x faster.
@@ -721,6 +737,10 @@ Shuffling is **deterministic** and designed for distributed training:
721
737
 
722
738
  The permutation depends on `seed`, the epoch, and chunk metadata — the same settings always yield the same order (required for resumable `state_dict`).
723
739
 
740
+ **Object storage (`s3://`, `gs://`, …)** globally permutes chunks (`FullShuffle`). Random chunk order is cheap once files are already copied into the local cache.
741
+
742
+ **POSIX-fast** (automatic for any local path) mmaps chunks in place. **Vast / NFS / Lustre / GPFS** (and `LITDATA_POSIX_FAST=1`) use `WindowShuffle`: each worker gets **whole chunks** in a sequential stripe, then shuffles only inside a sliding window (default **16**, `LITDATA_POSIX_SHUFFLE_WINDOW`) for both chunk order and in-chunk items. Local disks (ext4/xfs) keep global `FullShuffle`. Object URLs stay on `FullShuffle`. `LITDATA_POSIX_FAST=0` disables in-place mmap.
743
+
724
744
  ```python
725
745
  from litdata import StreamingDataset, StreamingDataLoader
726
746
 
@@ -789,7 +809,7 @@ Rough ImageNet order-of-magnitude on a Studio (not hard guarantees; right tuning
789
809
  | `seed` | `42` | Shuffle / subsample RNG |
790
810
  | `serializers` | built-ins | Custom serialize/deserialize map |
791
811
  | `max_cache_size` | `"100GB"` | Evict consumed chunks beyond this size |
792
- | `max_pre_download` | `2` | Chunks each worker may prefetch (raise for throughput; watch disk) |
812
+ | `max_pre_download` | `2` | Chunks each worker may prefetch (raise for throughput; watch disk / RAM) |
793
813
  | `subsample` | `1.0` | Fraction of data (`0.01`) or upsample (`2.5`) |
794
814
  | `encryption` | `None` | `FernetEncryption` / `RSAEncryption` / custom |
795
815
  | `storage_options` | `{}` | Cloud client options |
@@ -800,6 +820,8 @@ Rough ImageNet order-of-magnitude on a Studio (not hard guarantees; right tuning
800
820
 
801
821
  Peak disk ≈ `num_workers × max_pre_download × mean_chunk_size`.
802
822
 
823
+ On **Vast / NFS / local disk**, POSIX-fast is on by default (`LITDATA_POSIX_FAST=0` to disable). `WILLNEED` prefetch and `num_workers` are capped when they would exceed about half of `MemAvailable`. Idle **hugepages** (common on GPU nodes) do not count as available RAM — drop unused `nr_hugepages` if `MemAvailable` looks tiny next to `MemTotal`.
824
+
803
825
  **`StreamingDataLoader`**
804
826
 
805
827
  | Argument | Description |
@@ -1742,7 +1764,7 @@ Only **worker 0** is instrumented. When an `int` is used, the tracer wraps `fetc
1742
1764
 
1743
1765
  - Delete or change `profile_dir` between runs — LitData removes an existing `result.json` before starting.
1744
1766
  - Pair with a wiped chunk cache if you care about **cold** epoch behavior (`litdata cache clear`).
1745
- - For deeper LitData internals (download / lock / delete timeline), use `enable_tracer()` + [Litracer](https://github.com/deependujha/litracer) instead — see [Debug & Profile LitData](#debug-profile). That path is complementary: viztracer = DataLoader worker CPU timeline; Litracer = LitData pipeline events.
1767
+ - For deeper LitData internals (download / read / delete timeline), use `enable_tracer()` + [Litracer](https://github.com/Lightning-AI/litracer) instead — see [Debug & Profile LitData](#debug-profile). That path is complementary: viztracer = DataLoader worker CPU timeline; Litracer = LitData pipeline events.
1746
1768
 
1747
1769
  </details>
1748
1770
 
@@ -1855,6 +1877,9 @@ export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
1855
1877
  | `LITDATA_DISABLE_VERSION_CHECK` | `0` | `1` skips the upgrade tip |
1856
1878
  | `HF_TOKEN` | — | Gated Hugging Face datasets |
1857
1879
  | `DEBUG_LITDATA` / `PRINT_DEBUG_LOGS` | `0` | Internal debug / stdout logs |
1880
+ | `LITDATA_LOG_FILE` | `litdata_debug.log` | `enable_tracer()` output path |
1881
+ | `LITDATA_TRACE_LEVEL` | unset | `batch` / `chunk` / `sample` / `debug` / `off` (see [Debug & Profile](#debug-profile)) |
1882
+ | `LITDATA_TRACE_CATEGORIES` | from level | Comma-separated cats, e.g. `download,read,delete` |
1858
1883
 
1859
1884
  Multi-node `optimize`/`map` on Studios also uses `DATA_OPTIMIZER_*` (set by the platform). Full catalog (debug logs, Studio injects, torchrun): see the LitData skill `reference/env-vars.md` when using agent skills, or the source modules `constants.py` / `async_prefetch.py`.
1860
1885
 
@@ -2020,66 +2045,68 @@ ds = StreamingDataset(input_dir=data_dir, encryption=rsa)
2020
2045
 
2021
2046
  &nbsp;
2022
2047
 
2023
- LitData comes with built-in logging and profiling capabilities to help you debug and profile your data streaming workloads.
2048
+ `enable_tracer()` records the streaming pipeline (download vs read vs delete vs batch) as one-line events. [Litracer](https://github.com/Lightning-AI/litracer) converts that log into a Chrome / [Perfetto](https://ui.perfetto.dev) trace.
2024
2049
 
2025
- <img width="1439" alt="431247797-0e955e71-2f9a-4aad-b7c1-a8218fed2e2e" src="https://github.com/user-attachments/assets/4e40676c-ba0b-49af-acac-975977173669" />
2050
+ This is complementary to [`profile_batches`](#profile-loading) (viztracer = DataLoader worker **CPU**; Litracer = LitData **pipeline** events).
2026
2051
 
2027
- - e.g., with LitData Streaming
2052
+ <img width="1439" alt="431247797-0e955e71-2f9a-4aad-b7c1-a8218fed2e2e" src="https://github.com/user-attachments/assets/4e40676c-ba0b-49af-acac-975977173669" />
2028
2053
 
2029
2054
  ```python
2030
2055
  import litdata as ld
2031
2056
  from litdata.debugger import enable_tracer
2032
2057
 
2033
- # WARNING: Remove existing trace `litdata_debug.log` file if it exists before re-tracing
2034
- enable_tracer()
2058
+ # Call once per process, before the DataLoader. Delete an existing log before re-tracing (append).
2059
+ enable_tracer(level="chunk", log_file="litdata_debug.log")
2060
+ # level="batch" | "chunk" (default) | "sample" | "debug" | "off"
2061
+ # enable_tracer(categories=["download", "read", "delete"])
2035
2062
 
2036
2063
  if __name__ == "__main__":
2037
2064
  dataset = ld.StreamingDataset("s3://my-bucket/my-data", shuffle=True)
2038
- dataloader = ld.StreamingDataLoader(dataset, batch_size=64)
2039
-
2040
- for batch in dataloader:
2041
- print(batch) # Replace with your data processing logic
2065
+ for batch in ld.StreamingDataLoader(dataset, batch_size=64, num_workers=8):
2066
+ ...
2042
2067
  ```
2043
2068
 
2044
- 1. Generate Debug Log:
2069
+ | Level | Events |
2070
+ | ----- | ------ |
2071
+ | `batch` | Epoch + per-batch spans, plus crashes |
2072
+ | `chunk` (default) | + `download`, `read`, `delete`, `decompress`, `prefetch` |
2073
+ | `sample` | + per-item `__getitem__` (high volume) |
2074
+ | `debug` | + `.cnt` lock refcount spans |
2075
+ | `off` | Disable |
2045
2076
 
2046
- - Run your Python program and it'll create a log file containing detailed debug information.
2077
+ Event **names** are stable (`download`, `read`, `delete`, `batch`, `sample`, `crash`). Chunk / sample indexes live in args so Perfetto groups all downloads together. Each line is `key: value;` pairs with Chrome **microsecond** timestamps. Crashes are a one-line instant (`ph: I`, `name: crash`); the Python traceback is printed to **stderr**, not the log file (a multi-line `logger.exception` would break Litracer).
2047
2078
 
2048
- ```bash
2049
- python main.py
2050
- ```
2079
+ Env overrides: `LITDATA_LOG_FILE`, `LITDATA_TRACE_LEVEL`, `LITDATA_TRACE_CATEGORIES` (comma-separated). Tracer calls are no-ops when tracing is off.
2051
2080
 
2052
- 2. Install [Litracer](https://github.com/deependujha/litracer/):
2081
+ 1. Generate the log:
2053
2082
 
2054
- - Option 1: Using Go (recommended)
2055
- - Install Go on your system.
2056
- - Run the following command to install Litracer:
2083
+ ```bash
2084
+ python train.py # writes litdata_debug.log
2085
+ ```
2057
2086
 
2058
- ```bash
2059
- go install github.com/deependujha/litracer@latest
2060
- ```
2087
+ 2. Install [Litracer](https://github.com/Lightning-AI/litracer) (Go 1.23+):
2061
2088
 
2062
- - Option 2: Download Binary
2063
- - Visit the [LitRacer GitHub Releases](https://github.com/deependujha/litracer/releases) page.
2064
- - Download the appropriate binary for your operating system and follow the installation instructions.
2089
+ ```bash
2090
+ git clone https://github.com/Lightning-AI/litracer.git
2091
+ cd litracer && go build -o litracer .
2092
+ ```
2065
2093
 
2066
- 3. Convert Debug Log to trace JSON:
2094
+ Or `go install github.com/deependujha/litracer@latest` (published Go module path). Until `go.mod` is renamed, `go install github.com/Lightning-AI/litracer@latest` does not work. Release binaries: [GitHub Releases](https://github.com/Lightning-AI/litracer/releases).
2067
2095
 
2068
- - Use litracer to convert the generated log file into a trace JSON file. This command uses 100 workers for conversion:
2096
+ 3. Convert and open in Perfetto:
2069
2097
 
2070
2098
  ```bash
2071
- litracer litdata_debug.log -o litdata_trace.json -w 100
2099
+ litracer --quiet --validate -o litdata_trace.json.gz litdata_debug.log
2100
+ litracer --quiet --cat download,read,delete -o io.json.gz litdata_debug.log
2101
+ # open the .json.gz at https://ui.perfetto.dev (preferred) or chrome://tracing
2072
2102
  ```
2073
2103
 
2074
- 4. Visualize the trace:
2075
-
2076
- - Use either `chrome://tracing` in the Chrome browser or `ui.perfetto.dev` to view the `litdata_trace.json` file for in-depth performance insights. You can also use `SQL queries` to analyze the logs.
2077
- - `Perfetto` is recommended over `chrome://tracing` for visualization & analyzing.
2104
+ `--quiet` prints a one-line summary (per-category durations, unmatched B/E, crashes). `--cat` keeps only those categories. Matched B/E pairs become complete (`ph: X`) spans unless `--no-complete`. Default output is gzip Chrome JSON (`.json.gz`) — both Perfetto and `chrome://tracing` open it; pass `-o file.json` for uncompressed.
2078
2105
 
2079
- - Key Points:
2106
+ - For trace files `> 2GB`, see [Perfetto large traces](https://perfetto.dev/docs/visualization/large-traces).
2107
+ - If you connect Perfetto to the RPC server, prefer Chrome over Brave (Brave often does not autodetect the RPC server).
2080
2108
 
2081
- - For very large trace.json files (`> 2GB`), refer to the [Perfetto documentation](https://perfetto.dev/docs/visualization/large-traces) for using native accelerators.
2082
- - If you are trying to connect Perfetto to the RPC server, it is recommended to use Chrome over Brave, as it has been observed that Perfetto in Brave does not autodetect the RPC server.
2109
+ **Multi-worker `s3://` `FileNotFoundError` after ~120s:** `num_workers=0` working while `num_workers>0` fails usually means the DataLoader parent started obstore (tokio) before fork and worker GETs hung. Current LitData fetches `index.json` with boto3 so workers can lazy-init obstore; they fall back to boto3 if the parent already started the runtime. On Studio R2 / `lightning_storage`, the same symptom can be a prefetch-thread crash (`data_connection_id` / `endpoint_url` into `boto3.Session`) — look for `[litdata] PrepareChunksThread CRASHED` on stderr and a `crash` instant in the trace.
2083
2110
 
2084
2111
  </details>
2085
2112
 
@@ -2364,6 +2391,17 @@ Speed to stream Imagenet 1.2M from local disk with ffcv vs LitData:
2364
2391
  | ffcv(os_cache=True) | JPEG 90% | 20 GB | 7653 | 8051 |
2365
2392
  | ffcv(os_cache=False) | JPEG 90% | 20 GB | 8149 | 8607 |
2366
2393
 
2394
+ Speed to stream a **synthetic ImageNet-scale set from Vast NFS** (NFSv3 `nconnect=32`, 208-CPU host, ~1 TiB RAM). Dataset: **1.08M** JPEG q95 256×256 (~160 GiB, 64 MiB chunks). `StreamingDataLoader`, batch **256**, `shuffle=True`, `drop_last=True`, decode only unless noted. POSIX-fast mmaps chunks **in place** (no copy into `~/.lightning/chunks`).
2395
+
2396
+ | Setup | Workers | Images / sec |
2397
+ |---|---|---|
2398
+ | Copy into local cache (`LITDATA_POSIX_FAST=0`) | 48 | **16.7k** (2-epoch avg) |
2399
+ | POSIX-fast (this default on local/Vast paths) | 48 | **18.2k** |
2400
+ | POSIX-fast + README ImageNet augs (crop 224, flip, float32) | 48 | **12.9k** |
2401
+ | POSIX-fast, all CPU cores | **208** | **35.8k** |
2402
+
2403
+ Notes: 208 workers need enough **MemAvailable**. This host had **928×1 GiB hugepages** reserved and idle (~900 GiB locked); after `nr_hugepages=0`, 208 workers stayed healthy. If `num_workers=os.cpu_count()` would crowd RAM, LitData **clamps** workers (`LITDATA_POSIX_MAX_WORKERS=0` disables) and skips `WILLNEED` prefetch. Real ImageNet JPEG 90% is much smaller (~12 GiB) and usually decodes faster than this q95 noise set.
2404
+
2367
2405
  ### Raw Dataset
2368
2406
 
2369
2407
  Speed to stream raw Imagenet 1.2M from different cloud storage providers:
@@ -14,7 +14,7 @@
14
14
 
15
15
  import time
16
16
 
17
- __version__ = "0.2.67"
17
+ __version__ = "0.2.68"
18
18
  __author__ = "Lightning AI et al."
19
19
  __author_email__ = "pytorch@lightning.ai"
20
20
  __license__ = "Apache-2.0"
@@ -14,16 +14,19 @@ import warnings
14
14
 
15
15
  from litdata.__about__ import * # noqa: F403
16
16
  from litdata.constants import _LIGHTNING_SDK_AVAILABLE
17
+ from litdata.exceptions import ChunkWaitTimeoutError
17
18
  from litdata.processing.functions import map, merge_datasets, optimize, walk
18
19
  from litdata.raw.dataset import StreamingRawDataset
19
20
  from litdata.streaming.combined import CombinedStreamingDataset
20
21
  from litdata.streaming.dataloader import StreamingDataLoader
21
22
  from litdata.streaming.dataset import StreamingDataset
23
+ from litdata.streaming.dataset_update import dataset_update
22
24
  from litdata.streaming.item_loader import TokensLoader
23
25
  from litdata.streaming.parallel import ParallelStreamingDataset
24
26
  from litdata.streaming.writer import index_parquet_dataset
25
27
  from litdata.utilities.breakpoint import breakpoint
26
28
  from litdata.utilities.hf_dataset import index_hf_dataset
29
+ from litdata.utilities.keys_index import build_keys_index
27
30
  from litdata.utilities.train_test_split import train_test_split
28
31
 
29
32
  warnings.filterwarnings(
@@ -41,12 +44,15 @@ __all__ = [
41
44
  "ParallelStreamingDataset",
42
45
  "map",
43
46
  "optimize",
47
+ "dataset_update",
48
+ "build_keys_index",
44
49
  "walk",
45
50
  "train_test_split",
46
51
  "merge_datasets",
47
52
  "index_parquet_dataset",
48
53
  "index_hf_dataset",
49
54
  "breakpoint",
55
+ "ChunkWaitTimeoutError",
50
56
  ]
51
57
 
52
58
  if _LIGHTNING_SDK_AVAILABLE:
@@ -20,6 +20,12 @@ import torch
20
20
  from lightning_utilities.core.imports import RequirementCache
21
21
 
22
22
  _INDEX_FILENAME = "index.json"
23
+ _KEYS_DIRNAME = "keys"
24
+ _KEYS_SHARD_TEMPLATE = "shard-{:05d}.parquet"
25
+ # Legacy single-file sidecar (still read if present).
26
+ _KEYS_FILENAME = "keys.parquet"
27
+ _RANK_KEYS_SUFFIX = ".keys.parquet"
28
+ _DEFAULT_KEYS_NUM_SHARDS = 1
23
29
  _DEFAULT_CHUNK_BYTES = 1 << 26 # 64M B
24
30
  _DEFAULT_FAST_DEV_RUN_ITEMS = 10
25
31
  _DEFAULT_CACHE_DIR = os.path.join(Path.home(), ".lightning", "chunks")