litdata 0.2.67__tar.gz → 0.2.68__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {litdata-0.2.67 → litdata-0.2.68}/CONTRIBUTING.md +1 -1
- {litdata-0.2.67/src/litdata.egg-info → litdata-0.2.68}/PKG-INFO +75 -37
- {litdata-0.2.67 → litdata-0.2.68}/README.md +74 -36
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/__about__.py +1 -1
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/__init__.py +6 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/constants.py +6 -0
- litdata-0.2.68/src/litdata/debugger.py +397 -0
- litdata-0.2.68/src/litdata/exceptions.py +36 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/processing/data_processor.py +207 -16
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/processing/functions.py +35 -5
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/async_prefetch.py +10 -2
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/cache.py +2 -2
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/client.py +11 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/combined.py +3 -12
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/compression.py +28 -8
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/config.py +9 -17
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/dataloader.py +24 -3
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/dataset.py +108 -9
- litdata-0.2.68/src/litdata/streaming/dataset_update.py +299 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/downloader.py +230 -105
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/item_loader.py +317 -118
- litdata-0.2.68/src/litdata/streaming/posix_fast.py +396 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/reader.py +117 -36
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/shuffle.py +52 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/writer.py +26 -13
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/format.py +1 -1
- litdata-0.2.68/src/litdata/utilities/keys_index.py +868 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/shuffle.py +114 -0
- {litdata-0.2.67 → litdata-0.2.68/src/litdata.egg-info}/PKG-INFO +75 -37
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata.egg-info/SOURCES.txt +4 -0
- litdata-0.2.67/src/litdata/debugger.py +0 -205
- {litdata-0.2.67 → litdata-0.2.68}/LICENSE +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/MANIFEST.in +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/requirements.txt +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/setup.cfg +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/setup.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/__main__.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/cli/__init__.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/cli/commands.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/cli/handler/__init__.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/cli/handler/cache.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/cli/handler/optimize.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/cli/parser.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/helpers.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/imports.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/processing/__init__.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/processing/readers.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/processing/utilities.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/raw/__init__.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/raw/dataset.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/raw/indexer.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/raw/types.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/requirements.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/__init__.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/fs_provider.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/parallel.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/resolver.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/sampler.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/serializers.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/streaming/timing.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/__init__.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/_pytree.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/base.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/breakpoint.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/broadcast.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/dataset_utilities.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/encryption.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/env.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/hf_dataset.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/packing.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/parquet.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/subsample.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/torch_utils.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata/utilities/train_test_split.py +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata.egg-info/dependency_links.txt +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata.egg-info/entry_points.txt +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata.egg-info/not-zip-safe +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata.egg-info/requires.txt +0 -0
- {litdata-0.2.67 → litdata-0.2.68}/src/litdata.egg-info/top_level.txt +0 -0
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
Welcome to the PyTorch Lightning community! We're building the most advanced research platform on the planet to implement the latest, best practices and integrations that the amazing PyTorch team and other research organization rolls out!
|
|
4
4
|
|
|
5
|
-
If you are new to open source, check out [
|
|
5
|
+
If you are new to open source, check out [GitHub's guide to making your first contribution](https://docs.github.com/en/get-started/quickstart/contributing-to-projects).
|
|
6
6
|
|
|
7
7
|
## Main Core Value: One less thing to remember
|
|
8
8
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: litdata
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.68
|
|
4
4
|
Summary: The Deep Learning framework to train, deploy, and ship AI products Lightning fast.
|
|
5
5
|
Home-page: https://github.com/Lightning-AI/litdata
|
|
6
6
|
Download-URL: https://github.com/Lightning-AI/litdata
|
|
@@ -279,6 +279,22 @@ for sample in dataloader:
|
|
|
279
279
|
img, cls = sample["image"], sample["class"]
|
|
280
280
|
```
|
|
281
281
|
|
|
282
|
+
**Keyed lookup and in-place patches** (needs `polars` and `optimize(..., key_fn=...)` or `build_keys_index`):
|
|
283
|
+
|
|
284
|
+
```python
|
|
285
|
+
ld.optimize(fn=fn, inputs=inputs, output_dir="fast_data", chunk_bytes="64MB", key_fn=lambda s: s["id"])
|
|
286
|
+
|
|
287
|
+
ds = ld.StreamingDataset("fast_data")
|
|
288
|
+
sample = ds["entity-id"] # str keys
|
|
289
|
+
sample = ds.get_by_key(42) # int entity keys; ds[42] is still positional
|
|
290
|
+
|
|
291
|
+
with ld.dataset_update("fast_data") as update: # local directory only
|
|
292
|
+
update["entity-id"] = {"id": "entity-id", "x": 1}
|
|
293
|
+
update.commit()
|
|
294
|
+
```
|
|
295
|
+
|
|
296
|
+
`mode="append"` continues chunk numbering. `use_checkpoint=True` tries to resume an interrupted optimize. They are not the same.
|
|
297
|
+
|
|
282
298
|
**Key benefits:**
|
|
283
299
|
|
|
284
300
|
✅ **Accelerate training:** Optimized datasets load 20x faster.
|
|
@@ -785,6 +801,10 @@ Shuffling is **deterministic** and designed for distributed training:
|
|
|
785
801
|
|
|
786
802
|
The permutation depends on `seed`, the epoch, and chunk metadata — the same settings always yield the same order (required for resumable `state_dict`).
|
|
787
803
|
|
|
804
|
+
**Object storage (`s3://`, `gs://`, …)** globally permutes chunks (`FullShuffle`). Random chunk order is cheap once files are already copied into the local cache.
|
|
805
|
+
|
|
806
|
+
**POSIX-fast** (automatic for any local path) mmaps chunks in place. **Vast / NFS / Lustre / GPFS** (and `LITDATA_POSIX_FAST=1`) use `WindowShuffle`: each worker gets **whole chunks** in a sequential stripe, then shuffles only inside a sliding window (default **16**, `LITDATA_POSIX_SHUFFLE_WINDOW`) for both chunk order and in-chunk items. Local disks (ext4/xfs) keep global `FullShuffle`. Object URLs stay on `FullShuffle`. `LITDATA_POSIX_FAST=0` disables in-place mmap.
|
|
807
|
+
|
|
788
808
|
```python
|
|
789
809
|
from litdata import StreamingDataset, StreamingDataLoader
|
|
790
810
|
|
|
@@ -853,7 +873,7 @@ Rough ImageNet order-of-magnitude on a Studio (not hard guarantees; right tuning
|
|
|
853
873
|
| `seed` | `42` | Shuffle / subsample RNG |
|
|
854
874
|
| `serializers` | built-ins | Custom serialize/deserialize map |
|
|
855
875
|
| `max_cache_size` | `"100GB"` | Evict consumed chunks beyond this size |
|
|
856
|
-
| `max_pre_download` | `2` | Chunks each worker may prefetch (raise for throughput; watch disk) |
|
|
876
|
+
| `max_pre_download` | `2` | Chunks each worker may prefetch (raise for throughput; watch disk / RAM) |
|
|
857
877
|
| `subsample` | `1.0` | Fraction of data (`0.01`) or upsample (`2.5`) |
|
|
858
878
|
| `encryption` | `None` | `FernetEncryption` / `RSAEncryption` / custom |
|
|
859
879
|
| `storage_options` | `{}` | Cloud client options |
|
|
@@ -864,6 +884,8 @@ Rough ImageNet order-of-magnitude on a Studio (not hard guarantees; right tuning
|
|
|
864
884
|
|
|
865
885
|
Peak disk ≈ `num_workers × max_pre_download × mean_chunk_size`.
|
|
866
886
|
|
|
887
|
+
On **Vast / NFS / local disk**, POSIX-fast is on by default (`LITDATA_POSIX_FAST=0` to disable). `WILLNEED` prefetch and `num_workers` are capped when they would exceed about half of `MemAvailable`. Idle **hugepages** (common on GPU nodes) do not count as available RAM — drop unused `nr_hugepages` if `MemAvailable` looks tiny next to `MemTotal`.
|
|
888
|
+
|
|
867
889
|
**`StreamingDataLoader`**
|
|
868
890
|
|
|
869
891
|
| Argument | Description |
|
|
@@ -1806,7 +1828,7 @@ Only **worker 0** is instrumented. When an `int` is used, the tracer wraps `fetc
|
|
|
1806
1828
|
|
|
1807
1829
|
- Delete or change `profile_dir` between runs — LitData removes an existing `result.json` before starting.
|
|
1808
1830
|
- Pair with a wiped chunk cache if you care about **cold** epoch behavior (`litdata cache clear`).
|
|
1809
|
-
- For deeper LitData internals (download /
|
|
1831
|
+
- For deeper LitData internals (download / read / delete timeline), use `enable_tracer()` + [Litracer](https://github.com/Lightning-AI/litracer) instead — see [Debug & Profile LitData](#debug-profile). That path is complementary: viztracer = DataLoader worker CPU timeline; Litracer = LitData pipeline events.
|
|
1810
1832
|
|
|
1811
1833
|
</details>
|
|
1812
1834
|
|
|
@@ -1919,6 +1941,9 @@ export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
|
|
|
1919
1941
|
| `LITDATA_DISABLE_VERSION_CHECK` | `0` | `1` skips the upgrade tip |
|
|
1920
1942
|
| `HF_TOKEN` | — | Gated Hugging Face datasets |
|
|
1921
1943
|
| `DEBUG_LITDATA` / `PRINT_DEBUG_LOGS` | `0` | Internal debug / stdout logs |
|
|
1944
|
+
| `LITDATA_LOG_FILE` | `litdata_debug.log` | `enable_tracer()` output path |
|
|
1945
|
+
| `LITDATA_TRACE_LEVEL` | unset | `batch` / `chunk` / `sample` / `debug` / `off` (see [Debug & Profile](#debug-profile)) |
|
|
1946
|
+
| `LITDATA_TRACE_CATEGORIES` | from level | Comma-separated cats, e.g. `download,read,delete` |
|
|
1922
1947
|
|
|
1923
1948
|
Multi-node `optimize`/`map` on Studios also uses `DATA_OPTIMIZER_*` (set by the platform). Full catalog (debug logs, Studio injects, torchrun): see the LitData skill `reference/env-vars.md` when using agent skills, or the source modules `constants.py` / `async_prefetch.py`.
|
|
1924
1949
|
|
|
@@ -2084,66 +2109,68 @@ ds = StreamingDataset(input_dir=data_dir, encryption=rsa)
|
|
|
2084
2109
|
|
|
2085
2110
|
|
|
2086
2111
|
|
|
2087
|
-
|
|
2112
|
+
`enable_tracer()` records the streaming pipeline (download vs read vs delete vs batch) as one-line events. [Litracer](https://github.com/Lightning-AI/litracer) converts that log into a Chrome / [Perfetto](https://ui.perfetto.dev) trace.
|
|
2088
2113
|
|
|
2089
|
-
|
|
2114
|
+
This is complementary to [`profile_batches`](#profile-loading) (viztracer = DataLoader worker **CPU**; Litracer = LitData **pipeline** events).
|
|
2090
2115
|
|
|
2091
|
-
-
|
|
2116
|
+
<img width="1439" alt="431247797-0e955e71-2f9a-4aad-b7c1-a8218fed2e2e" src="https://github.com/user-attachments/assets/4e40676c-ba0b-49af-acac-975977173669" />
|
|
2092
2117
|
|
|
2093
2118
|
```python
|
|
2094
2119
|
import litdata as ld
|
|
2095
2120
|
from litdata.debugger import enable_tracer
|
|
2096
2121
|
|
|
2097
|
-
#
|
|
2098
|
-
enable_tracer()
|
|
2122
|
+
# Call once per process, before the DataLoader. Delete an existing log before re-tracing (append).
|
|
2123
|
+
enable_tracer(level="chunk", log_file="litdata_debug.log")
|
|
2124
|
+
# level="batch" | "chunk" (default) | "sample" | "debug" | "off"
|
|
2125
|
+
# enable_tracer(categories=["download", "read", "delete"])
|
|
2099
2126
|
|
|
2100
2127
|
if __name__ == "__main__":
|
|
2101
2128
|
dataset = ld.StreamingDataset("s3://my-bucket/my-data", shuffle=True)
|
|
2102
|
-
|
|
2103
|
-
|
|
2104
|
-
for batch in dataloader:
|
|
2105
|
-
print(batch) # Replace with your data processing logic
|
|
2129
|
+
for batch in ld.StreamingDataLoader(dataset, batch_size=64, num_workers=8):
|
|
2130
|
+
...
|
|
2106
2131
|
```
|
|
2107
2132
|
|
|
2108
|
-
|
|
2133
|
+
| Level | Events |
|
|
2134
|
+
| ----- | ------ |
|
|
2135
|
+
| `batch` | Epoch + per-batch spans, plus crashes |
|
|
2136
|
+
| `chunk` (default) | + `download`, `read`, `delete`, `decompress`, `prefetch` |
|
|
2137
|
+
| `sample` | + per-item `__getitem__` (high volume) |
|
|
2138
|
+
| `debug` | + `.cnt` lock refcount spans |
|
|
2139
|
+
| `off` | Disable |
|
|
2109
2140
|
|
|
2110
|
-
|
|
2141
|
+
Event **names** are stable (`download`, `read`, `delete`, `batch`, `sample`, `crash`). Chunk / sample indexes live in args so Perfetto groups all downloads together. Each line is `key: value;` pairs with Chrome **microsecond** timestamps. Crashes are a one-line instant (`ph: I`, `name: crash`); the Python traceback is printed to **stderr**, not the log file (a multi-line `logger.exception` would break Litracer).
|
|
2111
2142
|
|
|
2112
|
-
|
|
2113
|
-
python main.py
|
|
2114
|
-
```
|
|
2143
|
+
Env overrides: `LITDATA_LOG_FILE`, `LITDATA_TRACE_LEVEL`, `LITDATA_TRACE_CATEGORIES` (comma-separated). Tracer calls are no-ops when tracing is off.
|
|
2115
2144
|
|
|
2116
|
-
|
|
2145
|
+
1. Generate the log:
|
|
2117
2146
|
|
|
2118
|
-
|
|
2119
|
-
|
|
2120
|
-
|
|
2147
|
+
```bash
|
|
2148
|
+
python train.py # writes litdata_debug.log
|
|
2149
|
+
```
|
|
2121
2150
|
|
|
2122
|
-
|
|
2123
|
-
go install github.com/deependujha/litracer@latest
|
|
2124
|
-
```
|
|
2151
|
+
2. Install [Litracer](https://github.com/Lightning-AI/litracer) (Go 1.23+):
|
|
2125
2152
|
|
|
2126
|
-
|
|
2127
|
-
|
|
2128
|
-
|
|
2153
|
+
```bash
|
|
2154
|
+
git clone https://github.com/Lightning-AI/litracer.git
|
|
2155
|
+
cd litracer && go build -o litracer .
|
|
2156
|
+
```
|
|
2129
2157
|
|
|
2130
|
-
|
|
2158
|
+
Or `go install github.com/deependujha/litracer@latest` (published Go module path). Until `go.mod` is renamed, `go install github.com/Lightning-AI/litracer@latest` does not work. Release binaries: [GitHub Releases](https://github.com/Lightning-AI/litracer/releases).
|
|
2131
2159
|
|
|
2132
|
-
|
|
2160
|
+
3. Convert and open in Perfetto:
|
|
2133
2161
|
|
|
2134
2162
|
```bash
|
|
2135
|
-
|
|
2163
|
+
litracer --quiet --validate -o litdata_trace.json.gz litdata_debug.log
|
|
2164
|
+
litracer --quiet --cat download,read,delete -o io.json.gz litdata_debug.log
|
|
2165
|
+
# open the .json.gz at https://ui.perfetto.dev (preferred) or chrome://tracing
|
|
2136
2166
|
```
|
|
2137
2167
|
|
|
2138
|
-
|
|
2139
|
-
|
|
2140
|
-
- Use either `chrome://tracing` in the Chrome browser or `ui.perfetto.dev` to view the `litdata_trace.json` file for in-depth performance insights. You can also use `SQL queries` to analyze the logs.
|
|
2141
|
-
- `Perfetto` is recommended over `chrome://tracing` for visualization & analyzing.
|
|
2168
|
+
`--quiet` prints a one-line summary (per-category durations, unmatched B/E, crashes). `--cat` keeps only those categories. Matched B/E pairs become complete (`ph: X`) spans unless `--no-complete`. Default output is gzip Chrome JSON (`.json.gz`) — both Perfetto and `chrome://tracing` open it; pass `-o file.json` for uncompressed.
|
|
2142
2169
|
|
|
2143
|
-
-
|
|
2170
|
+
- For trace files `> 2GB`, see [Perfetto large traces](https://perfetto.dev/docs/visualization/large-traces).
|
|
2171
|
+
- If you connect Perfetto to the RPC server, prefer Chrome over Brave (Brave often does not autodetect the RPC server).
|
|
2144
2172
|
|
|
2145
|
-
|
|
2146
|
-
- If you are trying to connect Perfetto to the RPC server, it is recommended to use Chrome over Brave, as it has been observed that Perfetto in Brave does not autodetect the RPC server.
|
|
2173
|
+
**Multi-worker `s3://` `FileNotFoundError` after ~120s:** `num_workers=0` working while `num_workers>0` fails usually means the DataLoader parent started obstore (tokio) before fork and worker GETs hung. Current LitData fetches `index.json` with boto3 so workers can lazy-init obstore; they fall back to boto3 if the parent already started the runtime. On Studio R2 / `lightning_storage`, the same symptom can be a prefetch-thread crash (`data_connection_id` / `endpoint_url` into `boto3.Session`) — look for `[litdata] PrepareChunksThread CRASHED` on stderr and a `crash` instant in the trace.
|
|
2147
2174
|
|
|
2148
2175
|
</details>
|
|
2149
2176
|
|
|
@@ -2428,6 +2455,17 @@ Speed to stream Imagenet 1.2M from local disk with ffcv vs LitData:
|
|
|
2428
2455
|
| ffcv(os_cache=True) | JPEG 90% | 20 GB | 7653 | 8051 |
|
|
2429
2456
|
| ffcv(os_cache=False) | JPEG 90% | 20 GB | 8149 | 8607 |
|
|
2430
2457
|
|
|
2458
|
+
Speed to stream a **synthetic ImageNet-scale set from Vast NFS** (NFSv3 `nconnect=32`, 208-CPU host, ~1 TiB RAM). Dataset: **1.08M** JPEG q95 256×256 (~160 GiB, 64 MiB chunks). `StreamingDataLoader`, batch **256**, `shuffle=True`, `drop_last=True`, decode only unless noted. POSIX-fast mmaps chunks **in place** (no copy into `~/.lightning/chunks`).
|
|
2459
|
+
|
|
2460
|
+
| Setup | Workers | Images / sec |
|
|
2461
|
+
|---|---|---|
|
|
2462
|
+
| Copy into local cache (`LITDATA_POSIX_FAST=0`) | 48 | **16.7k** (2-epoch avg) |
|
|
2463
|
+
| POSIX-fast (this default on local/Vast paths) | 48 | **18.2k** |
|
|
2464
|
+
| POSIX-fast + README ImageNet augs (crop 224, flip, float32) | 48 | **12.9k** |
|
|
2465
|
+
| POSIX-fast, all CPU cores | **208** | **35.8k** |
|
|
2466
|
+
|
|
2467
|
+
Notes: 208 workers need enough **MemAvailable**. This host had **928×1 GiB hugepages** reserved and idle (~900 GiB locked); after `nr_hugepages=0`, 208 workers stayed healthy. If `num_workers=os.cpu_count()` would crowd RAM, LitData **clamps** workers (`LITDATA_POSIX_MAX_WORKERS=0` disables) and skips `WILLNEED` prefetch. Real ImageNet JPEG 90% is much smaller (~12 GiB) and usually decodes faster than this q95 noise set.
|
|
2468
|
+
|
|
2431
2469
|
### Raw Dataset
|
|
2432
2470
|
|
|
2433
2471
|
Speed to stream raw Imagenet 1.2M from different cloud storage providers:
|
|
@@ -215,6 +215,22 @@ for sample in dataloader:
|
|
|
215
215
|
img, cls = sample["image"], sample["class"]
|
|
216
216
|
```
|
|
217
217
|
|
|
218
|
+
**Keyed lookup and in-place patches** (needs `polars` and `optimize(..., key_fn=...)` or `build_keys_index`):
|
|
219
|
+
|
|
220
|
+
```python
|
|
221
|
+
ld.optimize(fn=fn, inputs=inputs, output_dir="fast_data", chunk_bytes="64MB", key_fn=lambda s: s["id"])
|
|
222
|
+
|
|
223
|
+
ds = ld.StreamingDataset("fast_data")
|
|
224
|
+
sample = ds["entity-id"] # str keys
|
|
225
|
+
sample = ds.get_by_key(42) # int entity keys; ds[42] is still positional
|
|
226
|
+
|
|
227
|
+
with ld.dataset_update("fast_data") as update: # local directory only
|
|
228
|
+
update["entity-id"] = {"id": "entity-id", "x": 1}
|
|
229
|
+
update.commit()
|
|
230
|
+
```
|
|
231
|
+
|
|
232
|
+
`mode="append"` continues chunk numbering. `use_checkpoint=True` tries to resume an interrupted optimize. They are not the same.
|
|
233
|
+
|
|
218
234
|
**Key benefits:**
|
|
219
235
|
|
|
220
236
|
✅ **Accelerate training:** Optimized datasets load 20x faster.
|
|
@@ -721,6 +737,10 @@ Shuffling is **deterministic** and designed for distributed training:
|
|
|
721
737
|
|
|
722
738
|
The permutation depends on `seed`, the epoch, and chunk metadata — the same settings always yield the same order (required for resumable `state_dict`).
|
|
723
739
|
|
|
740
|
+
**Object storage (`s3://`, `gs://`, …)** globally permutes chunks (`FullShuffle`). Random chunk order is cheap once files are already copied into the local cache.
|
|
741
|
+
|
|
742
|
+
**POSIX-fast** (automatic for any local path) mmaps chunks in place. **Vast / NFS / Lustre / GPFS** (and `LITDATA_POSIX_FAST=1`) use `WindowShuffle`: each worker gets **whole chunks** in a sequential stripe, then shuffles only inside a sliding window (default **16**, `LITDATA_POSIX_SHUFFLE_WINDOW`) for both chunk order and in-chunk items. Local disks (ext4/xfs) keep global `FullShuffle`. Object URLs stay on `FullShuffle`. `LITDATA_POSIX_FAST=0` disables in-place mmap.
|
|
743
|
+
|
|
724
744
|
```python
|
|
725
745
|
from litdata import StreamingDataset, StreamingDataLoader
|
|
726
746
|
|
|
@@ -789,7 +809,7 @@ Rough ImageNet order-of-magnitude on a Studio (not hard guarantees; right tuning
|
|
|
789
809
|
| `seed` | `42` | Shuffle / subsample RNG |
|
|
790
810
|
| `serializers` | built-ins | Custom serialize/deserialize map |
|
|
791
811
|
| `max_cache_size` | `"100GB"` | Evict consumed chunks beyond this size |
|
|
792
|
-
| `max_pre_download` | `2` | Chunks each worker may prefetch (raise for throughput; watch disk) |
|
|
812
|
+
| `max_pre_download` | `2` | Chunks each worker may prefetch (raise for throughput; watch disk / RAM) |
|
|
793
813
|
| `subsample` | `1.0` | Fraction of data (`0.01`) or upsample (`2.5`) |
|
|
794
814
|
| `encryption` | `None` | `FernetEncryption` / `RSAEncryption` / custom |
|
|
795
815
|
| `storage_options` | `{}` | Cloud client options |
|
|
@@ -800,6 +820,8 @@ Rough ImageNet order-of-magnitude on a Studio (not hard guarantees; right tuning
|
|
|
800
820
|
|
|
801
821
|
Peak disk ≈ `num_workers × max_pre_download × mean_chunk_size`.
|
|
802
822
|
|
|
823
|
+
On **Vast / NFS / local disk**, POSIX-fast is on by default (`LITDATA_POSIX_FAST=0` to disable). `WILLNEED` prefetch and `num_workers` are capped when they would exceed about half of `MemAvailable`. Idle **hugepages** (common on GPU nodes) do not count as available RAM — drop unused `nr_hugepages` if `MemAvailable` looks tiny next to `MemTotal`.
|
|
824
|
+
|
|
803
825
|
**`StreamingDataLoader`**
|
|
804
826
|
|
|
805
827
|
| Argument | Description |
|
|
@@ -1742,7 +1764,7 @@ Only **worker 0** is instrumented. When an `int` is used, the tracer wraps `fetc
|
|
|
1742
1764
|
|
|
1743
1765
|
- Delete or change `profile_dir` between runs — LitData removes an existing `result.json` before starting.
|
|
1744
1766
|
- Pair with a wiped chunk cache if you care about **cold** epoch behavior (`litdata cache clear`).
|
|
1745
|
-
- For deeper LitData internals (download /
|
|
1767
|
+
- For deeper LitData internals (download / read / delete timeline), use `enable_tracer()` + [Litracer](https://github.com/Lightning-AI/litracer) instead — see [Debug & Profile LitData](#debug-profile). That path is complementary: viztracer = DataLoader worker CPU timeline; Litracer = LitData pipeline events.
|
|
1746
1768
|
|
|
1747
1769
|
</details>
|
|
1748
1770
|
|
|
@@ -1855,6 +1877,9 @@ export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
|
|
|
1855
1877
|
| `LITDATA_DISABLE_VERSION_CHECK` | `0` | `1` skips the upgrade tip |
|
|
1856
1878
|
| `HF_TOKEN` | — | Gated Hugging Face datasets |
|
|
1857
1879
|
| `DEBUG_LITDATA` / `PRINT_DEBUG_LOGS` | `0` | Internal debug / stdout logs |
|
|
1880
|
+
| `LITDATA_LOG_FILE` | `litdata_debug.log` | `enable_tracer()` output path |
|
|
1881
|
+
| `LITDATA_TRACE_LEVEL` | unset | `batch` / `chunk` / `sample` / `debug` / `off` (see [Debug & Profile](#debug-profile)) |
|
|
1882
|
+
| `LITDATA_TRACE_CATEGORIES` | from level | Comma-separated cats, e.g. `download,read,delete` |
|
|
1858
1883
|
|
|
1859
1884
|
Multi-node `optimize`/`map` on Studios also uses `DATA_OPTIMIZER_*` (set by the platform). Full catalog (debug logs, Studio injects, torchrun): see the LitData skill `reference/env-vars.md` when using agent skills, or the source modules `constants.py` / `async_prefetch.py`.
|
|
1860
1885
|
|
|
@@ -2020,66 +2045,68 @@ ds = StreamingDataset(input_dir=data_dir, encryption=rsa)
|
|
|
2020
2045
|
|
|
2021
2046
|
|
|
2022
2047
|
|
|
2023
|
-
|
|
2048
|
+
`enable_tracer()` records the streaming pipeline (download vs read vs delete vs batch) as one-line events. [Litracer](https://github.com/Lightning-AI/litracer) converts that log into a Chrome / [Perfetto](https://ui.perfetto.dev) trace.
|
|
2024
2049
|
|
|
2025
|
-
|
|
2050
|
+
This is complementary to [`profile_batches`](#profile-loading) (viztracer = DataLoader worker **CPU**; Litracer = LitData **pipeline** events).
|
|
2026
2051
|
|
|
2027
|
-
-
|
|
2052
|
+
<img width="1439" alt="431247797-0e955e71-2f9a-4aad-b7c1-a8218fed2e2e" src="https://github.com/user-attachments/assets/4e40676c-ba0b-49af-acac-975977173669" />
|
|
2028
2053
|
|
|
2029
2054
|
```python
|
|
2030
2055
|
import litdata as ld
|
|
2031
2056
|
from litdata.debugger import enable_tracer
|
|
2032
2057
|
|
|
2033
|
-
#
|
|
2034
|
-
enable_tracer()
|
|
2058
|
+
# Call once per process, before the DataLoader. Delete an existing log before re-tracing (append).
|
|
2059
|
+
enable_tracer(level="chunk", log_file="litdata_debug.log")
|
|
2060
|
+
# level="batch" | "chunk" (default) | "sample" | "debug" | "off"
|
|
2061
|
+
# enable_tracer(categories=["download", "read", "delete"])
|
|
2035
2062
|
|
|
2036
2063
|
if __name__ == "__main__":
|
|
2037
2064
|
dataset = ld.StreamingDataset("s3://my-bucket/my-data", shuffle=True)
|
|
2038
|
-
|
|
2039
|
-
|
|
2040
|
-
for batch in dataloader:
|
|
2041
|
-
print(batch) # Replace with your data processing logic
|
|
2065
|
+
for batch in ld.StreamingDataLoader(dataset, batch_size=64, num_workers=8):
|
|
2066
|
+
...
|
|
2042
2067
|
```
|
|
2043
2068
|
|
|
2044
|
-
|
|
2069
|
+
| Level | Events |
|
|
2070
|
+
| ----- | ------ |
|
|
2071
|
+
| `batch` | Epoch + per-batch spans, plus crashes |
|
|
2072
|
+
| `chunk` (default) | + `download`, `read`, `delete`, `decompress`, `prefetch` |
|
|
2073
|
+
| `sample` | + per-item `__getitem__` (high volume) |
|
|
2074
|
+
| `debug` | + `.cnt` lock refcount spans |
|
|
2075
|
+
| `off` | Disable |
|
|
2045
2076
|
|
|
2046
|
-
|
|
2077
|
+
Event **names** are stable (`download`, `read`, `delete`, `batch`, `sample`, `crash`). Chunk / sample indexes live in args so Perfetto groups all downloads together. Each line is `key: value;` pairs with Chrome **microsecond** timestamps. Crashes are a one-line instant (`ph: I`, `name: crash`); the Python traceback is printed to **stderr**, not the log file (a multi-line `logger.exception` would break Litracer).
|
|
2047
2078
|
|
|
2048
|
-
|
|
2049
|
-
python main.py
|
|
2050
|
-
```
|
|
2079
|
+
Env overrides: `LITDATA_LOG_FILE`, `LITDATA_TRACE_LEVEL`, `LITDATA_TRACE_CATEGORIES` (comma-separated). Tracer calls are no-ops when tracing is off.
|
|
2051
2080
|
|
|
2052
|
-
|
|
2081
|
+
1. Generate the log:
|
|
2053
2082
|
|
|
2054
|
-
|
|
2055
|
-
|
|
2056
|
-
|
|
2083
|
+
```bash
|
|
2084
|
+
python train.py # writes litdata_debug.log
|
|
2085
|
+
```
|
|
2057
2086
|
|
|
2058
|
-
|
|
2059
|
-
go install github.com/deependujha/litracer@latest
|
|
2060
|
-
```
|
|
2087
|
+
2. Install [Litracer](https://github.com/Lightning-AI/litracer) (Go 1.23+):
|
|
2061
2088
|
|
|
2062
|
-
|
|
2063
|
-
|
|
2064
|
-
|
|
2089
|
+
```bash
|
|
2090
|
+
git clone https://github.com/Lightning-AI/litracer.git
|
|
2091
|
+
cd litracer && go build -o litracer .
|
|
2092
|
+
```
|
|
2065
2093
|
|
|
2066
|
-
|
|
2094
|
+
Or `go install github.com/deependujha/litracer@latest` (published Go module path). Until `go.mod` is renamed, `go install github.com/Lightning-AI/litracer@latest` does not work. Release binaries: [GitHub Releases](https://github.com/Lightning-AI/litracer/releases).
|
|
2067
2095
|
|
|
2068
|
-
|
|
2096
|
+
3. Convert and open in Perfetto:
|
|
2069
2097
|
|
|
2070
2098
|
```bash
|
|
2071
|
-
|
|
2099
|
+
litracer --quiet --validate -o litdata_trace.json.gz litdata_debug.log
|
|
2100
|
+
litracer --quiet --cat download,read,delete -o io.json.gz litdata_debug.log
|
|
2101
|
+
# open the .json.gz at https://ui.perfetto.dev (preferred) or chrome://tracing
|
|
2072
2102
|
```
|
|
2073
2103
|
|
|
2074
|
-
|
|
2075
|
-
|
|
2076
|
-
- Use either `chrome://tracing` in the Chrome browser or `ui.perfetto.dev` to view the `litdata_trace.json` file for in-depth performance insights. You can also use `SQL queries` to analyze the logs.
|
|
2077
|
-
- `Perfetto` is recommended over `chrome://tracing` for visualization & analyzing.
|
|
2104
|
+
`--quiet` prints a one-line summary (per-category durations, unmatched B/E, crashes). `--cat` keeps only those categories. Matched B/E pairs become complete (`ph: X`) spans unless `--no-complete`. Default output is gzip Chrome JSON (`.json.gz`) — both Perfetto and `chrome://tracing` open it; pass `-o file.json` for uncompressed.
|
|
2078
2105
|
|
|
2079
|
-
-
|
|
2106
|
+
- For trace files `> 2GB`, see [Perfetto large traces](https://perfetto.dev/docs/visualization/large-traces).
|
|
2107
|
+
- If you connect Perfetto to the RPC server, prefer Chrome over Brave (Brave often does not autodetect the RPC server).
|
|
2080
2108
|
|
|
2081
|
-
|
|
2082
|
-
- If you are trying to connect Perfetto to the RPC server, it is recommended to use Chrome over Brave, as it has been observed that Perfetto in Brave does not autodetect the RPC server.
|
|
2109
|
+
**Multi-worker `s3://` `FileNotFoundError` after ~120s:** `num_workers=0` working while `num_workers>0` fails usually means the DataLoader parent started obstore (tokio) before fork and worker GETs hung. Current LitData fetches `index.json` with boto3 so workers can lazy-init obstore; they fall back to boto3 if the parent already started the runtime. On Studio R2 / `lightning_storage`, the same symptom can be a prefetch-thread crash (`data_connection_id` / `endpoint_url` into `boto3.Session`) — look for `[litdata] PrepareChunksThread CRASHED` on stderr and a `crash` instant in the trace.
|
|
2083
2110
|
|
|
2084
2111
|
</details>
|
|
2085
2112
|
|
|
@@ -2364,6 +2391,17 @@ Speed to stream Imagenet 1.2M from local disk with ffcv vs LitData:
|
|
|
2364
2391
|
| ffcv(os_cache=True) | JPEG 90% | 20 GB | 7653 | 8051 |
|
|
2365
2392
|
| ffcv(os_cache=False) | JPEG 90% | 20 GB | 8149 | 8607 |
|
|
2366
2393
|
|
|
2394
|
+
Speed to stream a **synthetic ImageNet-scale set from Vast NFS** (NFSv3 `nconnect=32`, 208-CPU host, ~1 TiB RAM). Dataset: **1.08M** JPEG q95 256×256 (~160 GiB, 64 MiB chunks). `StreamingDataLoader`, batch **256**, `shuffle=True`, `drop_last=True`, decode only unless noted. POSIX-fast mmaps chunks **in place** (no copy into `~/.lightning/chunks`).
|
|
2395
|
+
|
|
2396
|
+
| Setup | Workers | Images / sec |
|
|
2397
|
+
|---|---|---|
|
|
2398
|
+
| Copy into local cache (`LITDATA_POSIX_FAST=0`) | 48 | **16.7k** (2-epoch avg) |
|
|
2399
|
+
| POSIX-fast (this default on local/Vast paths) | 48 | **18.2k** |
|
|
2400
|
+
| POSIX-fast + README ImageNet augs (crop 224, flip, float32) | 48 | **12.9k** |
|
|
2401
|
+
| POSIX-fast, all CPU cores | **208** | **35.8k** |
|
|
2402
|
+
|
|
2403
|
+
Notes: 208 workers need enough **MemAvailable**. This host had **928×1 GiB hugepages** reserved and idle (~900 GiB locked); after `nr_hugepages=0`, 208 workers stayed healthy. If `num_workers=os.cpu_count()` would crowd RAM, LitData **clamps** workers (`LITDATA_POSIX_MAX_WORKERS=0` disables) and skips `WILLNEED` prefetch. Real ImageNet JPEG 90% is much smaller (~12 GiB) and usually decodes faster than this q95 noise set.
|
|
2404
|
+
|
|
2367
2405
|
### Raw Dataset
|
|
2368
2406
|
|
|
2369
2407
|
Speed to stream raw Imagenet 1.2M from different cloud storage providers:
|
|
@@ -14,16 +14,19 @@ import warnings
|
|
|
14
14
|
|
|
15
15
|
from litdata.__about__ import * # noqa: F403
|
|
16
16
|
from litdata.constants import _LIGHTNING_SDK_AVAILABLE
|
|
17
|
+
from litdata.exceptions import ChunkWaitTimeoutError
|
|
17
18
|
from litdata.processing.functions import map, merge_datasets, optimize, walk
|
|
18
19
|
from litdata.raw.dataset import StreamingRawDataset
|
|
19
20
|
from litdata.streaming.combined import CombinedStreamingDataset
|
|
20
21
|
from litdata.streaming.dataloader import StreamingDataLoader
|
|
21
22
|
from litdata.streaming.dataset import StreamingDataset
|
|
23
|
+
from litdata.streaming.dataset_update import dataset_update
|
|
22
24
|
from litdata.streaming.item_loader import TokensLoader
|
|
23
25
|
from litdata.streaming.parallel import ParallelStreamingDataset
|
|
24
26
|
from litdata.streaming.writer import index_parquet_dataset
|
|
25
27
|
from litdata.utilities.breakpoint import breakpoint
|
|
26
28
|
from litdata.utilities.hf_dataset import index_hf_dataset
|
|
29
|
+
from litdata.utilities.keys_index import build_keys_index
|
|
27
30
|
from litdata.utilities.train_test_split import train_test_split
|
|
28
31
|
|
|
29
32
|
warnings.filterwarnings(
|
|
@@ -41,12 +44,15 @@ __all__ = [
|
|
|
41
44
|
"ParallelStreamingDataset",
|
|
42
45
|
"map",
|
|
43
46
|
"optimize",
|
|
47
|
+
"dataset_update",
|
|
48
|
+
"build_keys_index",
|
|
44
49
|
"walk",
|
|
45
50
|
"train_test_split",
|
|
46
51
|
"merge_datasets",
|
|
47
52
|
"index_parquet_dataset",
|
|
48
53
|
"index_hf_dataset",
|
|
49
54
|
"breakpoint",
|
|
55
|
+
"ChunkWaitTimeoutError",
|
|
50
56
|
]
|
|
51
57
|
|
|
52
58
|
if _LIGHTNING_SDK_AVAILABLE:
|
|
@@ -20,6 +20,12 @@ import torch
|
|
|
20
20
|
from lightning_utilities.core.imports import RequirementCache
|
|
21
21
|
|
|
22
22
|
_INDEX_FILENAME = "index.json"
|
|
23
|
+
_KEYS_DIRNAME = "keys"
|
|
24
|
+
_KEYS_SHARD_TEMPLATE = "shard-{:05d}.parquet"
|
|
25
|
+
# Legacy single-file sidecar (still read if present).
|
|
26
|
+
_KEYS_FILENAME = "keys.parquet"
|
|
27
|
+
_RANK_KEYS_SUFFIX = ".keys.parquet"
|
|
28
|
+
_DEFAULT_KEYS_NUM_SHARDS = 1
|
|
23
29
|
_DEFAULT_CHUNK_BYTES = 1 << 26 # 64M B
|
|
24
30
|
_DEFAULT_FAST_DEV_RUN_ITEMS = 10
|
|
25
31
|
_DEFAULT_CACHE_DIR = os.path.join(Path.home(), ".lightning", "chunks")
|