litdata 0.2.70__tar.gz → 0.2.71__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {litdata-0.2.70/src/litdata.egg-info → litdata-0.2.71}/PKG-INFO +40 -6
- {litdata-0.2.70 → litdata-0.2.71}/README.md +39 -5
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/__about__.py +1 -1
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/async_prefetch.py +28 -4
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/cache.py +5 -4
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/compression.py +18 -13
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/config.py +49 -7
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/dataloader.py +184 -49
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/dataset.py +38 -23
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/downloader.py +37 -9
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/item_loader.py +35 -14
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/reader.py +59 -14
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/serializers.py +3 -1
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/shuffle.py +57 -29
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/writer.py +52 -8
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/_pytree.py +79 -19
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/env.py +12 -0
- litdata-0.2.71/src/litdata/utilities/format.py +168 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/shuffle.py +64 -12
- {litdata-0.2.70 → litdata-0.2.71/src/litdata.egg-info}/PKG-INFO +40 -6
- litdata-0.2.70/src/litdata/utilities/format.py +0 -57
- {litdata-0.2.70 → litdata-0.2.71}/CONTRIBUTING.md +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/LICENSE +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/MANIFEST.in +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/requirements.txt +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/setup.cfg +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/setup.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/__init__.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/__main__.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/cli/__init__.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/cli/commands.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/cli/handler/__init__.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/cli/handler/cache.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/cli/handler/optimize.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/cli/parser.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/constants.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/debugger.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/exceptions.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/helpers.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/imports.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/processing/__init__.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/processing/complete.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/processing/data_processor.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/processing/functions.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/processing/media_folder.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/processing/readers.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/processing/utilities.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/raw/__init__.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/raw/dataset.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/raw/indexer.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/raw/types.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/requirements.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/__init__.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/client.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/collate.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/combined.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/dataset_update.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/elastic.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/fs_provider.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/parallel.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/posix_fast.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/resolver.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/sampler.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/timing.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/types.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/__init__.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/base.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/breakpoint.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/broadcast.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/dataset_utilities.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/encryption.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/hf_dataset.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/keys_index.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/packing.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/parquet.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/subsample.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/torch_utils.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/train_test_split.py +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata.egg-info/SOURCES.txt +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata.egg-info/dependency_links.txt +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata.egg-info/entry_points.txt +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata.egg-info/not-zip-safe +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata.egg-info/requires.txt +0 -0
- {litdata-0.2.70 → litdata-0.2.71}/src/litdata.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: litdata
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.71
|
|
4
4
|
Summary: The Deep Learning framework to train, deploy, and ship AI products Lightning fast.
|
|
5
5
|
Home-page: https://github.com/Lightning-AI/litdata
|
|
6
6
|
Download-URL: https://github.com/Lightning-AI/litdata
|
|
@@ -1128,7 +1128,7 @@ Rough ImageNet order-of-magnitude on a Studio (not hard guarantees; right tuning
|
|
|
1128
1128
|
| `drop_last` | `True` if distributed else `False` | Equal length across ranks |
|
|
1129
1129
|
| `seed` | `42` | Shuffle / subsample RNG |
|
|
1130
1130
|
| `serializers` | built-ins | Custom serialize/deserialize map |
|
|
1131
|
-
| `max_cache_size` | `
|
|
1131
|
+
| `max_cache_size` | `None` | Evict consumed chunks beyond this size. Default: 75% of free disk, leaving ≥50GB when possible. Pin with `"100G"` / `"50GB"`, a fraction (`0.90`), or `MAX_CACHE_SIZE`. |
|
|
1132
1132
|
| `max_pre_download` | `2` | Chunks each worker may prefetch (raise for throughput; watch disk / RAM) |
|
|
1133
1133
|
| `subsample` | `1.0` | Fraction of data (`0.01`) or upsample (`2.5`) |
|
|
1134
1134
|
| `encryption` | `None` | `FernetEncryption` / `RSAEncryption` / custom |
|
|
@@ -1149,7 +1149,8 @@ On **Vast / NFS / local disk**, POSIX-fast is on by default (`LITDATA_POSIX_FAST
|
|
|
1149
1149
|
| All usual `torch.utils.data.DataLoader` kwargs | `batch_size`, `num_workers`, `collate_fn`, `pin_memory`, … |
|
|
1150
1150
|
| `shuffle` / `drop_last` | Forwarded to the streaming dataset |
|
|
1151
1151
|
| `profile_batches` | `int` / `True` / `False` — viztracer worker trace (see [Profile data loading](#profile-loading)) |
|
|
1152
|
-
| `
|
|
1152
|
+
| `profile_cprofile` | `True` — stdlib cProfile of the main process + worker 0 (see [Profile data loading](#profile-loading)) |
|
|
1153
|
+
| `profile_skip_batches` / `profile_dir` | Warm-up skip count; output dir for viztracer / cProfile files |
|
|
1153
1154
|
| `multiprocessing_context` | Use **`"spawn"`** (or `"forkserver"`) with `ParquetLoader` + `num_workers>0` on Linux |
|
|
1154
1155
|
|
|
1155
1156
|
Prefer `StreamingDataLoader` over a plain PyTorch `DataLoader` for optimized / combined / parallel datasets (resume + correct batch metadata).
|
|
@@ -2115,7 +2116,32 @@ Only **worker 0** is instrumented. When an `int` is used, the tracer wraps `fetc
|
|
|
2115
2116
|
|
|
2116
2117
|
- Delete or change `profile_dir` between runs — LitData removes an existing `result.json` before starting.
|
|
2117
2118
|
- Pair with a wiped chunk cache if you care about **cold** epoch behavior (`litdata cache clear`).
|
|
2118
|
-
|
|
2119
|
+
### cProfile (main + worker 0)
|
|
2120
|
+
|
|
2121
|
+
Stdlib statistical profile of **the parent process** (queue wait, unpickle, collate handoff) and **worker 0** (fetch, decode, transforms). No extra package.
|
|
2122
|
+
|
|
2123
|
+
```python
|
|
2124
|
+
loader = StreamingDataLoader(
|
|
2125
|
+
dataset,
|
|
2126
|
+
batch_size=64,
|
|
2127
|
+
num_workers=4,
|
|
2128
|
+
profile_cprofile=True,
|
|
2129
|
+
profile_dir="./profiles",
|
|
2130
|
+
)
|
|
2131
|
+
|
|
2132
|
+
for batch in loader:
|
|
2133
|
+
train_step(batch)
|
|
2134
|
+
# writes profiles/cprofile_main.prof + cprofile_worker0.prof (and .txt summaries)
|
|
2135
|
+
```
|
|
2136
|
+
|
|
2137
|
+
```bash
|
|
2138
|
+
python -m pstats profiles/cprofile_worker0.prof
|
|
2139
|
+
# then: sort tottime stats 30
|
|
2140
|
+
```
|
|
2141
|
+
|
|
2142
|
+
Do not set `profile_cprofile` and `profile_batches` together — both install a `sys.setprofile` hook. With `num_workers=0` only the main file is written. The parent profiler starts **after** workers spawn so fork does not inherit an active cProfile.
|
|
2143
|
+
|
|
2144
|
+
- For deeper LitData internals (download / read / delete timeline), use `enable_tracer()` + [Litracer](https://github.com/Lightning-AI/litracer) instead — see [Debug & Profile LitData](#debug-profile). That path is complementary: viztracer = DataLoader worker CPU timeline; cProfile = function self/cum time; Litracer = LitData pipeline events.
|
|
2119
2145
|
|
|
2120
2146
|
</details>
|
|
2121
2147
|
|
|
@@ -2165,7 +2191,9 @@ outputs = optimize(
|
|
|
2165
2191
|
|
|
2166
2192
|
Control how much disk the local chunk cache may use. Downloaded chunks are deleted after use once the cache exceeds the limit.
|
|
2167
2193
|
|
|
2168
|
-
Default `max_cache_size` is **`
|
|
2194
|
+
Default `max_cache_size` is **`None`**: LitData uses **75% of currently free disk** and still leaves **≥50GB** free when the volume is large enough for checkpoints. Pass `"100G"` / `"50GB"` for a fixed budget, or a float (`0.90`) for that fraction of currently free space. `MAX_CACHE_SIZE` overrides the constructor (size or fraction).
|
|
2195
|
+
|
|
2196
|
+
Peak disk in flight is roughly:
|
|
2169
2197
|
|
|
2170
2198
|
```
|
|
2171
2199
|
num_workers × max_pre_download × mean_chunk_size
|
|
@@ -2205,7 +2233,9 @@ for batch in StreamingDataLoader(dataset, batch_size=64, num_workers=8):
|
|
|
2205
2233
|
| `LITDATA_ASYNC_CHUNK_PREFETCH=1` | Force on |
|
|
2206
2234
|
| `LITDATA_ASYNC_CHUNK_PREFETCH=0` | Force off |
|
|
2207
2235
|
|
|
2208
|
-
When async is on, LitData raises `max_pre_download` to at least **4** so `asyncio.gather` has enough in-flight downloads (override with `LITDATA_ASYNC_MIN_PRE_DOWNLOAD`; set `0` to disable the floor). Peak disk ≈ `num_workers × max_pre_download × chunk_size` — size `max_cache_size` accordingly.
|
|
2236
|
+
When async is on, LitData raises `max_pre_download` to at least **4** so `asyncio.gather` has enough in-flight downloads (override with `LITDATA_ASYNC_MIN_PRE_DOWNLOAD`; set `0` to disable the floor). Concurrent GETs are capped by `LITDATA_ASYNC_DOWNLOAD_CONCURRENCY` (default **8**). The prepare thread drains the prefetch queue up to that gather width whenever a cache slot is free. Peak disk ≈ `num_workers × max_pre_download × chunk_size` — size `max_cache_size` accordingly.
|
|
2237
|
+
|
|
2238
|
+
The reader does **not** poll the cache directory for each chunk. After a download finishes, the downloader atomically `os.replace`s the temp file and sets an in-process `Event`. Compressed chunks set that Event only after decompress publishes the readable `.bin`. Other DataLoader workers still fall back to a short filesystem check (Events are per process).
|
|
2209
2239
|
|
|
2210
2240
|
```bash
|
|
2211
2241
|
# Debugging download/delete races — force synchronous downloads
|
|
@@ -2213,6 +2243,9 @@ export LITDATA_ASYNC_CHUNK_PREFETCH=0
|
|
|
2213
2243
|
|
|
2214
2244
|
# Keep max_pre_download=2 even with async enabled
|
|
2215
2245
|
export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
|
|
2246
|
+
|
|
2247
|
+
# Cap overlapping remote GETs (default 8)
|
|
2248
|
+
export LITDATA_ASYNC_DOWNLOAD_CONCURRENCY=4
|
|
2216
2249
|
```
|
|
2217
2250
|
|
|
2218
2251
|
### Common environment variables
|
|
@@ -2222,6 +2255,7 @@ export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
|
|
|
2222
2255
|
| `LITDATA_CACHE_DIR` | `~/.lightning/chunks` | Default chunk cache directory |
|
|
2223
2256
|
| `LITDATA_ASYNC_CHUNK_PREFETCH` | on for remote | `0`/`1` force async chunk download overlap |
|
|
2224
2257
|
| `LITDATA_ASYNC_MIN_PRE_DOWNLOAD` | `4` | Floor for `max_pre_download` when async is on (`0` = no floor) |
|
|
2258
|
+
| `LITDATA_ASYNC_DOWNLOAD_CONCURRENCY` | `8` | Max in-flight chunk GETs per gather |
|
|
2225
2259
|
| `LITDATA_OBSTORE_STREAM_MIN_CHUNK_MIB` | `8` | S3 obstore stream chunk size (MiB) |
|
|
2226
2260
|
| `MAX_WAIT_TIME` | `120` | Seconds to wait for a chunk before error |
|
|
2227
2261
|
| `FORCE_DOWNLOAD_TIME` | `30` | Seconds before force re-download of a missing chunk |
|
|
@@ -1064,7 +1064,7 @@ Rough ImageNet order-of-magnitude on a Studio (not hard guarantees; right tuning
|
|
|
1064
1064
|
| `drop_last` | `True` if distributed else `False` | Equal length across ranks |
|
|
1065
1065
|
| `seed` | `42` | Shuffle / subsample RNG |
|
|
1066
1066
|
| `serializers` | built-ins | Custom serialize/deserialize map |
|
|
1067
|
-
| `max_cache_size` | `
|
|
1067
|
+
| `max_cache_size` | `None` | Evict consumed chunks beyond this size. Default: 75% of free disk, leaving ≥50GB when possible. Pin with `"100G"` / `"50GB"`, a fraction (`0.90`), or `MAX_CACHE_SIZE`. |
|
|
1068
1068
|
| `max_pre_download` | `2` | Chunks each worker may prefetch (raise for throughput; watch disk / RAM) |
|
|
1069
1069
|
| `subsample` | `1.0` | Fraction of data (`0.01`) or upsample (`2.5`) |
|
|
1070
1070
|
| `encryption` | `None` | `FernetEncryption` / `RSAEncryption` / custom |
|
|
@@ -1085,7 +1085,8 @@ On **Vast / NFS / local disk**, POSIX-fast is on by default (`LITDATA_POSIX_FAST
|
|
|
1085
1085
|
| All usual `torch.utils.data.DataLoader` kwargs | `batch_size`, `num_workers`, `collate_fn`, `pin_memory`, … |
|
|
1086
1086
|
| `shuffle` / `drop_last` | Forwarded to the streaming dataset |
|
|
1087
1087
|
| `profile_batches` | `int` / `True` / `False` — viztracer worker trace (see [Profile data loading](#profile-loading)) |
|
|
1088
|
-
| `
|
|
1088
|
+
| `profile_cprofile` | `True` — stdlib cProfile of the main process + worker 0 (see [Profile data loading](#profile-loading)) |
|
|
1089
|
+
| `profile_skip_batches` / `profile_dir` | Warm-up skip count; output dir for viztracer / cProfile files |
|
|
1089
1090
|
| `multiprocessing_context` | Use **`"spawn"`** (or `"forkserver"`) with `ParquetLoader` + `num_workers>0` on Linux |
|
|
1090
1091
|
|
|
1091
1092
|
Prefer `StreamingDataLoader` over a plain PyTorch `DataLoader` for optimized / combined / parallel datasets (resume + correct batch metadata).
|
|
@@ -2051,7 +2052,32 @@ Only **worker 0** is instrumented. When an `int` is used, the tracer wraps `fetc
|
|
|
2051
2052
|
|
|
2052
2053
|
- Delete or change `profile_dir` between runs — LitData removes an existing `result.json` before starting.
|
|
2053
2054
|
- Pair with a wiped chunk cache if you care about **cold** epoch behavior (`litdata cache clear`).
|
|
2054
|
-
|
|
2055
|
+
### cProfile (main + worker 0)
|
|
2056
|
+
|
|
2057
|
+
Stdlib statistical profile of **the parent process** (queue wait, unpickle, collate handoff) and **worker 0** (fetch, decode, transforms). No extra package.
|
|
2058
|
+
|
|
2059
|
+
```python
|
|
2060
|
+
loader = StreamingDataLoader(
|
|
2061
|
+
dataset,
|
|
2062
|
+
batch_size=64,
|
|
2063
|
+
num_workers=4,
|
|
2064
|
+
profile_cprofile=True,
|
|
2065
|
+
profile_dir="./profiles",
|
|
2066
|
+
)
|
|
2067
|
+
|
|
2068
|
+
for batch in loader:
|
|
2069
|
+
train_step(batch)
|
|
2070
|
+
# writes profiles/cprofile_main.prof + cprofile_worker0.prof (and .txt summaries)
|
|
2071
|
+
```
|
|
2072
|
+
|
|
2073
|
+
```bash
|
|
2074
|
+
python -m pstats profiles/cprofile_worker0.prof
|
|
2075
|
+
# then: sort tottime stats 30
|
|
2076
|
+
```
|
|
2077
|
+
|
|
2078
|
+
Do not set `profile_cprofile` and `profile_batches` together — both install a `sys.setprofile` hook. With `num_workers=0` only the main file is written. The parent profiler starts **after** workers spawn so fork does not inherit an active cProfile.
|
|
2079
|
+
|
|
2080
|
+
- For deeper LitData internals (download / read / delete timeline), use `enable_tracer()` + [Litracer](https://github.com/Lightning-AI/litracer) instead — see [Debug & Profile LitData](#debug-profile). That path is complementary: viztracer = DataLoader worker CPU timeline; cProfile = function self/cum time; Litracer = LitData pipeline events.
|
|
2055
2081
|
|
|
2056
2082
|
</details>
|
|
2057
2083
|
|
|
@@ -2101,7 +2127,9 @@ outputs = optimize(
|
|
|
2101
2127
|
|
|
2102
2128
|
Control how much disk the local chunk cache may use. Downloaded chunks are deleted after use once the cache exceeds the limit.
|
|
2103
2129
|
|
|
2104
|
-
Default `max_cache_size` is **`
|
|
2130
|
+
Default `max_cache_size` is **`None`**: LitData uses **75% of currently free disk** and still leaves **≥50GB** free when the volume is large enough for checkpoints. Pass `"100G"` / `"50GB"` for a fixed budget, or a float (`0.90`) for that fraction of currently free space. `MAX_CACHE_SIZE` overrides the constructor (size or fraction).
|
|
2131
|
+
|
|
2132
|
+
Peak disk in flight is roughly:
|
|
2105
2133
|
|
|
2106
2134
|
```
|
|
2107
2135
|
num_workers × max_pre_download × mean_chunk_size
|
|
@@ -2141,7 +2169,9 @@ for batch in StreamingDataLoader(dataset, batch_size=64, num_workers=8):
|
|
|
2141
2169
|
| `LITDATA_ASYNC_CHUNK_PREFETCH=1` | Force on |
|
|
2142
2170
|
| `LITDATA_ASYNC_CHUNK_PREFETCH=0` | Force off |
|
|
2143
2171
|
|
|
2144
|
-
When async is on, LitData raises `max_pre_download` to at least **4** so `asyncio.gather` has enough in-flight downloads (override with `LITDATA_ASYNC_MIN_PRE_DOWNLOAD`; set `0` to disable the floor). Peak disk ≈ `num_workers × max_pre_download × chunk_size` — size `max_cache_size` accordingly.
|
|
2172
|
+
When async is on, LitData raises `max_pre_download` to at least **4** so `asyncio.gather` has enough in-flight downloads (override with `LITDATA_ASYNC_MIN_PRE_DOWNLOAD`; set `0` to disable the floor). Concurrent GETs are capped by `LITDATA_ASYNC_DOWNLOAD_CONCURRENCY` (default **8**). The prepare thread drains the prefetch queue up to that gather width whenever a cache slot is free. Peak disk ≈ `num_workers × max_pre_download × chunk_size` — size `max_cache_size` accordingly.
|
|
2173
|
+
|
|
2174
|
+
The reader does **not** poll the cache directory for each chunk. After a download finishes, the downloader atomically `os.replace`s the temp file and sets an in-process `Event`. Compressed chunks set that Event only after decompress publishes the readable `.bin`. Other DataLoader workers still fall back to a short filesystem check (Events are per process).
|
|
2145
2175
|
|
|
2146
2176
|
```bash
|
|
2147
2177
|
# Debugging download/delete races — force synchronous downloads
|
|
@@ -2149,6 +2179,9 @@ export LITDATA_ASYNC_CHUNK_PREFETCH=0
|
|
|
2149
2179
|
|
|
2150
2180
|
# Keep max_pre_download=2 even with async enabled
|
|
2151
2181
|
export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
|
|
2182
|
+
|
|
2183
|
+
# Cap overlapping remote GETs (default 8)
|
|
2184
|
+
export LITDATA_ASYNC_DOWNLOAD_CONCURRENCY=4
|
|
2152
2185
|
```
|
|
2153
2186
|
|
|
2154
2187
|
### Common environment variables
|
|
@@ -2158,6 +2191,7 @@ export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
|
|
|
2158
2191
|
| `LITDATA_CACHE_DIR` | `~/.lightning/chunks` | Default chunk cache directory |
|
|
2159
2192
|
| `LITDATA_ASYNC_CHUNK_PREFETCH` | on for remote | `0`/`1` force async chunk download overlap |
|
|
2160
2193
|
| `LITDATA_ASYNC_MIN_PRE_DOWNLOAD` | `4` | Floor for `max_pre_download` when async is on (`0` = no floor) |
|
|
2194
|
+
| `LITDATA_ASYNC_DOWNLOAD_CONCURRENCY` | `8` | Max in-flight chunk GETs per gather |
|
|
2161
2195
|
| `LITDATA_OBSTORE_STREAM_MIN_CHUNK_MIB` | `8` | S3 obstore stream chunk size (MiB) |
|
|
2162
2196
|
| `MAX_WAIT_TIME` | `120` | Seconds to wait for a chunk before error |
|
|
2163
2197
|
| `FORCE_DOWNLOAD_TIME` | `30` | Seconds before force re-download of a missing chunk |
|
|
@@ -54,6 +54,8 @@ _THREAD_LOOPS = threading.local()
|
|
|
54
54
|
# Empirically, async gather on real S3 is bottlenecked when max_pre_download==2
|
|
55
55
|
# (only 1–2 in-flight). Floor to 4 when the feature is enabled unless overridden.
|
|
56
56
|
_DEFAULT_ASYNC_MIN_PRE_DOWNLOAD = 4
|
|
57
|
+
# Cap in-flight GETs so a large drain batch does not open one connection per slot.
|
|
58
|
+
_DEFAULT_ASYNC_DOWNLOAD_CONCURRENCY = 8
|
|
57
59
|
|
|
58
60
|
|
|
59
61
|
def async_chunk_prefetch_enabled(remote_dir: str | None = None) -> bool:
|
|
@@ -81,6 +83,17 @@ def async_prefetch_min_pre_download() -> int:
|
|
|
81
83
|
return max(0, int(raw))
|
|
82
84
|
|
|
83
85
|
|
|
86
|
+
def async_download_concurrency(n_chunks: int) -> int:
|
|
87
|
+
"""How many remote chunk GETs to run at once inside ``asyncio.gather``.
|
|
88
|
+
|
|
89
|
+
Override with ``LITDATA_ASYNC_DOWNLOAD_CONCURRENCY`` (default 8). Always at
|
|
90
|
+
least 1 and at most ``n_chunks``.
|
|
91
|
+
"""
|
|
92
|
+
raw = os.getenv("LITDATA_ASYNC_DOWNLOAD_CONCURRENCY")
|
|
93
|
+
limit = _DEFAULT_ASYNC_DOWNLOAD_CONCURRENCY if raw is None else max(1, int(raw))
|
|
94
|
+
return max(1, min(limit, max(1, n_chunks)))
|
|
95
|
+
|
|
96
|
+
|
|
84
97
|
def apply_async_pre_download_floor(max_pre_download: int, remote_dir: str | None = None) -> int:
|
|
85
98
|
"""Raise ``max_pre_download`` when async prefetch needs gather width."""
|
|
86
99
|
if not async_chunk_prefetch_enabled(remote_dir):
|
|
@@ -133,6 +146,7 @@ def _remote_join(remote_dir: str, filename: str) -> str:
|
|
|
133
146
|
async def _adownload_file_to_path(downloader: Downloader, remote_filepath: str, local_filepath: str) -> None:
|
|
134
147
|
"""Fetch ``remote_filepath`` asynchronously and publish atomically."""
|
|
135
148
|
if os.path.exists(local_filepath):
|
|
149
|
+
downloader._notify_published(local_filepath)
|
|
136
150
|
return
|
|
137
151
|
# Prefer streaming-to-disk when the backend overrides adownload_file.
|
|
138
152
|
if type(downloader).adownload_file is not Downloader.adownload_file:
|
|
@@ -149,7 +163,7 @@ async def _adownload_file_to_path(downloader: Downloader, remote_filepath: str,
|
|
|
149
163
|
os.makedirs(os.path.dirname(local_filepath) or ".", exist_ok=True)
|
|
150
164
|
with open(tmp_path, "wb") as f:
|
|
151
165
|
f.write(data)
|
|
152
|
-
downloader.
|
|
166
|
+
downloader._publish_file(tmp_path, local_filepath)
|
|
153
167
|
except Exception:
|
|
154
168
|
with contextlib.suppress(FileNotFoundError, PermissionError):
|
|
155
169
|
os.remove(tmp_path)
|
|
@@ -172,7 +186,7 @@ async def _adownload_chunk_index(config: ChunksConfig, chunk_index: int) -> None
|
|
|
172
186
|
)
|
|
173
187
|
|
|
174
188
|
if os.path.exists(local_chunkpath):
|
|
175
|
-
config.try_decompress
|
|
189
|
+
await asyncio.to_thread(config.try_decompress, local_chunkpath)
|
|
176
190
|
if lazily_ref_counted:
|
|
177
191
|
downloader._increment_local_lock(lock_path, chunk_index)
|
|
178
192
|
return
|
|
@@ -186,7 +200,7 @@ async def _adownload_chunk_index(config: ChunksConfig, chunk_index: int) -> None
|
|
|
186
200
|
# Overlap blocking SDK calls across threads when native async is unavailable.
|
|
187
201
|
await asyncio.to_thread(downloader.download_chunk_from_index, chunk_index)
|
|
188
202
|
|
|
189
|
-
config.try_decompress
|
|
203
|
+
await asyncio.to_thread(config.try_decompress, local_chunkpath)
|
|
190
204
|
|
|
191
205
|
|
|
192
206
|
async def adownload_chunk_indexes(config: ChunksConfig, chunk_indexes: list[int]) -> None:
|
|
@@ -196,7 +210,17 @@ async def adownload_chunk_indexes(config: ChunksConfig, chunk_indexes: list[int]
|
|
|
196
210
|
if len(chunk_indexes) == 1:
|
|
197
211
|
await _adownload_chunk_index(config, chunk_indexes[0])
|
|
198
212
|
return
|
|
199
|
-
|
|
213
|
+
limit = async_download_concurrency(len(chunk_indexes))
|
|
214
|
+
if limit >= len(chunk_indexes):
|
|
215
|
+
await asyncio.gather(*[_adownload_chunk_index(config, idx) for idx in chunk_indexes])
|
|
216
|
+
return
|
|
217
|
+
sem = asyncio.Semaphore(limit)
|
|
218
|
+
|
|
219
|
+
async def _one(idx: int) -> None:
|
|
220
|
+
async with sem:
|
|
221
|
+
await _adownload_chunk_index(config, idx)
|
|
222
|
+
|
|
223
|
+
await asyncio.gather(*[_one(idx) for idx in chunk_indexes])
|
|
200
224
|
|
|
201
225
|
|
|
202
226
|
def _thread_event_loop() -> asyncio.AbstractEventLoop:
|
|
@@ -27,7 +27,7 @@ from litdata.streaming.serializers import Serializer
|
|
|
27
27
|
from litdata.streaming.writer import BinaryWriter
|
|
28
28
|
from litdata.utilities.encryption import Encryption
|
|
29
29
|
from litdata.utilities.env import _DistributedEnv, _WorkerEnv
|
|
30
|
-
from litdata.utilities.format import
|
|
30
|
+
from litdata.utilities.format import _resolve_max_cache_size
|
|
31
31
|
|
|
32
32
|
logger = logging.Logger(__name__)
|
|
33
33
|
|
|
@@ -43,7 +43,7 @@ class Cache:
|
|
|
43
43
|
chunk_size: int | None = None,
|
|
44
44
|
chunk_bytes: int | str | None = None,
|
|
45
45
|
item_loader: BaseItemLoader | None = None,
|
|
46
|
-
max_cache_size: int | str =
|
|
46
|
+
max_cache_size: int | float | str | None = None,
|
|
47
47
|
serializers: dict[str, Serializer] | None = None,
|
|
48
48
|
writer_chunk_index: int | None = None,
|
|
49
49
|
storage_options: dict | None = {},
|
|
@@ -64,7 +64,8 @@ class Cache:
|
|
|
64
64
|
chunk_bytes: The maximum number of bytes within a chunk.
|
|
65
65
|
chunk_size: The maximum number of items within a chunk.
|
|
66
66
|
item_loader: The object responsible to generate the chunk intervals and load an item froma chunk.
|
|
67
|
-
max_cache_size:
|
|
67
|
+
max_cache_size: Cache budget. ``None`` uses 75% of free disk (see ``StreamingDataset``).
|
|
68
|
+
A float such as ``0.90`` is that fraction of currently free space.
|
|
68
69
|
serializers: Provide your own serializers.
|
|
69
70
|
writer_chunk_index: The index of the chunk to start from when writing.
|
|
70
71
|
storage_options: Additional connection options for accessing storage services.
|
|
@@ -93,7 +94,7 @@ class Cache:
|
|
|
93
94
|
self._cache_dir,
|
|
94
95
|
subsampled_files=subsampled_files,
|
|
95
96
|
region_of_interest=region_of_interest,
|
|
96
|
-
max_cache_size=
|
|
97
|
+
max_cache_size=_resolve_max_cache_size(max_cache_size, self._cache_dir),
|
|
97
98
|
remote_input_dir=input_dir.url,
|
|
98
99
|
compression=compression,
|
|
99
100
|
encryption=encryption,
|
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
|
|
14
14
|
import shutil
|
|
15
15
|
from abc import ABC, abstractmethod
|
|
16
|
-
from typing import TypeVar
|
|
16
|
+
from typing import Any, TypeVar
|
|
17
17
|
|
|
18
18
|
from litdata.constants import _PYTHON_GREATER_EQUAL_3_14, _ZSTD_AVAILABLE
|
|
19
19
|
from litdata.debugger import CAT_DECOMPRESS, trace_span
|
|
@@ -54,27 +54,32 @@ class ZSTDCompressor(Compressor):
|
|
|
54
54
|
raise ModuleNotFoundError(str(_ZSTD_AVAILABLE))
|
|
55
55
|
self.level = level
|
|
56
56
|
self.extension = "zstd"
|
|
57
|
+
self._zstd: Any | None = None
|
|
58
|
+
|
|
59
|
+
def __getstate__(self) -> dict[str, Any]:
|
|
60
|
+
state = self.__dict__.copy()
|
|
61
|
+
state["_zstd"] = None
|
|
62
|
+
return state
|
|
63
|
+
|
|
64
|
+
def _zstd_mod(self) -> Any:
|
|
65
|
+
if self._zstd is None:
|
|
66
|
+
if _PYTHON_GREATER_EQUAL_3_14:
|
|
67
|
+
from compression import zstd as mod
|
|
68
|
+
else:
|
|
69
|
+
import zstd as mod
|
|
70
|
+
self._zstd = mod
|
|
71
|
+
return self._zstd
|
|
57
72
|
|
|
58
73
|
@property
|
|
59
74
|
def name(self) -> str:
|
|
60
75
|
return f"{self.extension}:{self.level}"
|
|
61
76
|
|
|
62
77
|
def compress(self, data: bytes) -> bytes:
|
|
63
|
-
|
|
64
|
-
from compression import zstd
|
|
65
|
-
else:
|
|
66
|
-
import zstd
|
|
67
|
-
|
|
68
|
-
return zstd.compress(data, self.level)
|
|
78
|
+
return self._zstd_mod().compress(data, self.level)
|
|
69
79
|
|
|
70
80
|
def decompress(self, data: bytes) -> bytes:
|
|
71
|
-
if _PYTHON_GREATER_EQUAL_3_14:
|
|
72
|
-
from compression import zstd
|
|
73
|
-
else:
|
|
74
|
-
import zstd
|
|
75
|
-
|
|
76
81
|
with trace_span("decompress", CAT_DECOMPRESS):
|
|
77
|
-
return
|
|
82
|
+
return self._zstd_mod().decompress(data)
|
|
78
83
|
|
|
79
84
|
def decompress_file(self, src: str, dst: str) -> None:
|
|
80
85
|
if _PYTHON_GREATER_EQUAL_3_14:
|
|
@@ -14,9 +14,10 @@
|
|
|
14
14
|
import contextlib
|
|
15
15
|
import logging
|
|
16
16
|
import os
|
|
17
|
+
import threading
|
|
17
18
|
from collections import defaultdict
|
|
18
19
|
from contextlib import suppress
|
|
19
|
-
from time import
|
|
20
|
+
from time import time
|
|
20
21
|
from typing import Any, Optional
|
|
21
22
|
|
|
22
23
|
from filelock import FileLock, Timeout
|
|
@@ -92,10 +93,15 @@ class ChunksConfig:
|
|
|
92
93
|
self._length = self._intervals[-1][-1] if len(self._intervals) > 0 else 0
|
|
93
94
|
self._downloader = None
|
|
94
95
|
|
|
96
|
+
# In-process "file is visible" Events, set after atomic publish (download or decompress).
|
|
97
|
+
self._file_published: dict[str, threading.Event] = {}
|
|
98
|
+
self._file_published_lock = threading.Lock()
|
|
99
|
+
|
|
95
100
|
if remote_dir:
|
|
96
101
|
self._downloader = get_downloader(
|
|
97
102
|
remote_dir, cache_dir, self._chunks, self._storage_options, self._session_options
|
|
98
103
|
)
|
|
104
|
+
self._downloader._on_file_published = self.notify_file_published
|
|
99
105
|
|
|
100
106
|
self._compressor_name = self._config["compression"]
|
|
101
107
|
self._compressor: Compressor | None = None
|
|
@@ -256,6 +262,28 @@ class ChunksConfig:
|
|
|
256
262
|
|
|
257
263
|
return self._downloader.download_chunk_bytes_from_index(chunk_index, offset, length)
|
|
258
264
|
|
|
265
|
+
def notify_file_published(self, local_filepath: str) -> None:
|
|
266
|
+
"""Mark ``local_filepath`` as atomically published and, if it is a readable chunk, signal ready."""
|
|
267
|
+
path = os.path.abspath(local_filepath)
|
|
268
|
+
with self._file_published_lock:
|
|
269
|
+
event = self._file_published.get(path)
|
|
270
|
+
if event is None:
|
|
271
|
+
event = threading.Event()
|
|
272
|
+
self._file_published[path] = event
|
|
273
|
+
event.set()
|
|
274
|
+
|
|
275
|
+
def wait_file_published(self, local_filepath: str, timeout: float) -> bool:
|
|
276
|
+
"""Wait for an in-process publish Event; fall back to ``exists`` for other workers."""
|
|
277
|
+
path = os.path.abspath(local_filepath)
|
|
278
|
+
with self._file_published_lock:
|
|
279
|
+
event = self._file_published.get(path)
|
|
280
|
+
if event is None:
|
|
281
|
+
event = threading.Event()
|
|
282
|
+
self._file_published[path] = event
|
|
283
|
+
if event.wait(timeout=timeout):
|
|
284
|
+
return True
|
|
285
|
+
return os.path.exists(path)
|
|
286
|
+
|
|
259
287
|
def _remove_compressed_source(self, local_chunkpath: str, target_local_chunkpath: str) -> None:
|
|
260
288
|
"""Drop the compressed download once the decompressed chunk is on disk."""
|
|
261
289
|
if local_chunkpath == target_local_chunkpath:
|
|
@@ -273,13 +301,12 @@ class ChunksConfig:
|
|
|
273
301
|
self._remove_compressed_source(local_chunkpath, target_local_chunkpath)
|
|
274
302
|
return
|
|
275
303
|
|
|
276
|
-
# Wait until
|
|
277
|
-
#
|
|
278
|
-
# existence of that path means the download is complete — do NOT use chunk_size (item
|
|
279
|
-
# count) as a byte threshold.
|
|
304
|
+
# Wait until the downloader publishes the compressed file, or another worker
|
|
305
|
+
# publishes the decompressed target. Cross-process downloads still fall back to exists.
|
|
280
306
|
start_time = time()
|
|
281
|
-
while not os.path.exists(
|
|
282
|
-
|
|
307
|
+
while not os.path.exists(target_local_chunkpath) and not os.path.exists(local_chunkpath):
|
|
308
|
+
self.wait_file_published(local_chunkpath, 0.05)
|
|
309
|
+
self.wait_file_published(target_local_chunkpath, 0.0)
|
|
283
310
|
if (time() - start_time) > _MAX_WAIT_TIME:
|
|
284
311
|
raise ChunkWaitTimeoutError(local_chunkpath, time() - start_time)
|
|
285
312
|
|
|
@@ -307,6 +334,7 @@ class ChunksConfig:
|
|
|
307
334
|
f"({os.stat(tmp_path).st_size} < {expected_bytes})."
|
|
308
335
|
)
|
|
309
336
|
os.replace(tmp_path, target_local_chunkpath)
|
|
337
|
+
self.notify_file_published(target_local_chunkpath)
|
|
310
338
|
except Exception:
|
|
311
339
|
with contextlib.suppress(FileNotFoundError, PermissionError):
|
|
312
340
|
os.remove(tmp_path)
|
|
@@ -464,6 +492,20 @@ class ChunksConfig:
|
|
|
464
492
|
session_options,
|
|
465
493
|
)
|
|
466
494
|
|
|
495
|
+
def __getstate__(self) -> dict[str, Any]:
|
|
496
|
+
state = self.__dict__.copy()
|
|
497
|
+
# threading.Lock / Event are not picklable (DataLoader spawn / deepcopy).
|
|
498
|
+
state["_file_published"] = {}
|
|
499
|
+
state["_file_published_lock"] = None
|
|
500
|
+
return state
|
|
501
|
+
|
|
502
|
+
def __setstate__(self, state: dict[str, Any]) -> None:
|
|
503
|
+
self.__dict__.update(state)
|
|
504
|
+
self._file_published = {}
|
|
505
|
+
self._file_published_lock = threading.Lock()
|
|
506
|
+
if self._downloader is not None:
|
|
507
|
+
self._downloader._on_file_published = self.notify_file_published
|
|
508
|
+
|
|
467
509
|
def __len__(self) -> int:
|
|
468
510
|
return self._length
|
|
469
511
|
|