litdata 0.2.70__tar.gz → 0.2.72__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {litdata-0.2.70/src/litdata.egg-info → litdata-0.2.72}/PKG-INFO +41 -6
- {litdata-0.2.70 → litdata-0.2.72}/README.md +39 -5
- {litdata-0.2.70 → litdata-0.2.72}/requirements.txt +1 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/__about__.py +1 -1
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/async_prefetch.py +28 -4
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/cache.py +5 -4
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/client.py +190 -35
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/compression.py +18 -13
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/config.py +49 -7
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/dataloader.py +184 -49
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/dataset.py +38 -23
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/downloader.py +37 -9
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/item_loader.py +35 -14
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/reader.py +59 -14
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/serializers.py +5 -3
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/shuffle.py +57 -29
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/writer.py +52 -8
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/_pytree.py +79 -19
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/env.py +12 -0
- litdata-0.2.72/src/litdata/utilities/format.py +168 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/shuffle.py +64 -12
- {litdata-0.2.70 → litdata-0.2.72/src/litdata.egg-info}/PKG-INFO +41 -6
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata.egg-info/requires.txt +1 -0
- litdata-0.2.70/src/litdata/utilities/format.py +0 -57
- {litdata-0.2.70 → litdata-0.2.72}/CONTRIBUTING.md +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/LICENSE +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/MANIFEST.in +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/setup.cfg +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/setup.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/__init__.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/__main__.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/cli/__init__.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/cli/commands.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/cli/handler/__init__.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/cli/handler/cache.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/cli/handler/optimize.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/cli/parser.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/constants.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/debugger.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/exceptions.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/helpers.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/imports.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/processing/__init__.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/processing/complete.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/processing/data_processor.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/processing/functions.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/processing/media_folder.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/processing/readers.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/processing/utilities.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/raw/__init__.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/raw/dataset.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/raw/indexer.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/raw/types.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/requirements.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/__init__.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/collate.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/combined.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/dataset_update.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/elastic.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/fs_provider.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/parallel.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/posix_fast.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/resolver.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/sampler.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/timing.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/types.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/__init__.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/base.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/breakpoint.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/broadcast.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/dataset_utilities.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/encryption.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/hf_dataset.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/keys_index.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/packing.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/parquet.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/subsample.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/torch_utils.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/train_test_split.py +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata.egg-info/SOURCES.txt +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata.egg-info/dependency_links.txt +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata.egg-info/entry_points.txt +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata.egg-info/not-zip-safe +0 -0
- {litdata-0.2.70 → litdata-0.2.72}/src/litdata.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: litdata
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.72
|
|
4
4
|
Summary: The Deep Learning framework to train, deploy, and ship AI products Lightning fast.
|
|
5
5
|
Home-page: https://github.com/Lightning-AI/litdata
|
|
6
6
|
Download-URL: https://github.com/Lightning-AI/litdata
|
|
@@ -34,6 +34,7 @@ Requires-Dist: filelock
|
|
|
34
34
|
Requires-Dist: numpy
|
|
35
35
|
Requires-Dist: boto3
|
|
36
36
|
Requires-Dist: requests
|
|
37
|
+
Requires-Dist: urllib3>=1.26
|
|
37
38
|
Requires-Dist: tifffile
|
|
38
39
|
Requires-Dist: obstore
|
|
39
40
|
Provides-Extra: extras
|
|
@@ -1128,7 +1129,7 @@ Rough ImageNet order-of-magnitude on a Studio (not hard guarantees; right tuning
|
|
|
1128
1129
|
| `drop_last` | `True` if distributed else `False` | Equal length across ranks |
|
|
1129
1130
|
| `seed` | `42` | Shuffle / subsample RNG |
|
|
1130
1131
|
| `serializers` | built-ins | Custom serialize/deserialize map |
|
|
1131
|
-
| `max_cache_size` | `
|
|
1132
|
+
| `max_cache_size` | `None` | Evict consumed chunks beyond this size. Default: 75% of free disk, leaving ≥50GB when possible. Pin with `"100G"` / `"50GB"`, a fraction (`0.90`), or `MAX_CACHE_SIZE`. |
|
|
1132
1133
|
| `max_pre_download` | `2` | Chunks each worker may prefetch (raise for throughput; watch disk / RAM) |
|
|
1133
1134
|
| `subsample` | `1.0` | Fraction of data (`0.01`) or upsample (`2.5`) |
|
|
1134
1135
|
| `encryption` | `None` | `FernetEncryption` / `RSAEncryption` / custom |
|
|
@@ -1149,7 +1150,8 @@ On **Vast / NFS / local disk**, POSIX-fast is on by default (`LITDATA_POSIX_FAST
|
|
|
1149
1150
|
| All usual `torch.utils.data.DataLoader` kwargs | `batch_size`, `num_workers`, `collate_fn`, `pin_memory`, … |
|
|
1150
1151
|
| `shuffle` / `drop_last` | Forwarded to the streaming dataset |
|
|
1151
1152
|
| `profile_batches` | `int` / `True` / `False` — viztracer worker trace (see [Profile data loading](#profile-loading)) |
|
|
1152
|
-
| `
|
|
1153
|
+
| `profile_cprofile` | `True` — stdlib cProfile of the main process + worker 0 (see [Profile data loading](#profile-loading)) |
|
|
1154
|
+
| `profile_skip_batches` / `profile_dir` | Warm-up skip count; output dir for viztracer / cProfile files |
|
|
1153
1155
|
| `multiprocessing_context` | Use **`"spawn"`** (or `"forkserver"`) with `ParquetLoader` + `num_workers>0` on Linux |
|
|
1154
1156
|
|
|
1155
1157
|
Prefer `StreamingDataLoader` over a plain PyTorch `DataLoader` for optimized / combined / parallel datasets (resume + correct batch metadata).
|
|
@@ -2115,7 +2117,32 @@ Only **worker 0** is instrumented. When an `int` is used, the tracer wraps `fetc
|
|
|
2115
2117
|
|
|
2116
2118
|
- Delete or change `profile_dir` between runs — LitData removes an existing `result.json` before starting.
|
|
2117
2119
|
- Pair with a wiped chunk cache if you care about **cold** epoch behavior (`litdata cache clear`).
|
|
2118
|
-
|
|
2120
|
+
### cProfile (main + worker 0)
|
|
2121
|
+
|
|
2122
|
+
Stdlib statistical profile of **the parent process** (queue wait, unpickle, collate handoff) and **worker 0** (fetch, decode, transforms). No extra package.
|
|
2123
|
+
|
|
2124
|
+
```python
|
|
2125
|
+
loader = StreamingDataLoader(
|
|
2126
|
+
dataset,
|
|
2127
|
+
batch_size=64,
|
|
2128
|
+
num_workers=4,
|
|
2129
|
+
profile_cprofile=True,
|
|
2130
|
+
profile_dir="./profiles",
|
|
2131
|
+
)
|
|
2132
|
+
|
|
2133
|
+
for batch in loader:
|
|
2134
|
+
train_step(batch)
|
|
2135
|
+
# writes profiles/cprofile_main.prof + cprofile_worker0.prof (and .txt summaries)
|
|
2136
|
+
```
|
|
2137
|
+
|
|
2138
|
+
```bash
|
|
2139
|
+
python -m pstats profiles/cprofile_worker0.prof
|
|
2140
|
+
# then: sort tottime stats 30
|
|
2141
|
+
```
|
|
2142
|
+
|
|
2143
|
+
Do not set `profile_cprofile` and `profile_batches` together — both install a `sys.setprofile` hook. With `num_workers=0` only the main file is written. The parent profiler starts **after** workers spawn so fork does not inherit an active cProfile.
|
|
2144
|
+
|
|
2145
|
+
- For deeper LitData internals (download / read / delete timeline), use `enable_tracer()` + [Litracer](https://github.com/Lightning-AI/litracer) instead — see [Debug & Profile LitData](#debug-profile). That path is complementary: viztracer = DataLoader worker CPU timeline; cProfile = function self/cum time; Litracer = LitData pipeline events.
|
|
2119
2146
|
|
|
2120
2147
|
</details>
|
|
2121
2148
|
|
|
@@ -2165,7 +2192,9 @@ outputs = optimize(
|
|
|
2165
2192
|
|
|
2166
2193
|
Control how much disk the local chunk cache may use. Downloaded chunks are deleted after use once the cache exceeds the limit.
|
|
2167
2194
|
|
|
2168
|
-
Default `max_cache_size` is **`
|
|
2195
|
+
Default `max_cache_size` is **`None`**: LitData uses **75% of currently free disk** and still leaves **≥50GB** free when the volume is large enough for checkpoints. Pass `"100G"` / `"50GB"` for a fixed budget, or a float (`0.90`) for that fraction of currently free space. `MAX_CACHE_SIZE` overrides the constructor (size or fraction).
|
|
2196
|
+
|
|
2197
|
+
Peak disk in flight is roughly:
|
|
2169
2198
|
|
|
2170
2199
|
```
|
|
2171
2200
|
num_workers × max_pre_download × mean_chunk_size
|
|
@@ -2205,7 +2234,9 @@ for batch in StreamingDataLoader(dataset, batch_size=64, num_workers=8):
|
|
|
2205
2234
|
| `LITDATA_ASYNC_CHUNK_PREFETCH=1` | Force on |
|
|
2206
2235
|
| `LITDATA_ASYNC_CHUNK_PREFETCH=0` | Force off |
|
|
2207
2236
|
|
|
2208
|
-
When async is on, LitData raises `max_pre_download` to at least **4** so `asyncio.gather` has enough in-flight downloads (override with `LITDATA_ASYNC_MIN_PRE_DOWNLOAD`; set `0` to disable the floor). Peak disk ≈ `num_workers × max_pre_download × chunk_size` — size `max_cache_size` accordingly.
|
|
2237
|
+
When async is on, LitData raises `max_pre_download` to at least **4** so `asyncio.gather` has enough in-flight downloads (override with `LITDATA_ASYNC_MIN_PRE_DOWNLOAD`; set `0` to disable the floor). Concurrent GETs are capped by `LITDATA_ASYNC_DOWNLOAD_CONCURRENCY` (default **8**). The prepare thread drains the prefetch queue up to that gather width whenever a cache slot is free. Peak disk ≈ `num_workers × max_pre_download × chunk_size` — size `max_cache_size` accordingly.
|
|
2238
|
+
|
|
2239
|
+
The reader does **not** poll the cache directory for each chunk. After a download finishes, the downloader atomically `os.replace`s the temp file and sets an in-process `Event`. Compressed chunks set that Event only after decompress publishes the readable `.bin`. Other DataLoader workers still fall back to a short filesystem check (Events are per process).
|
|
2209
2240
|
|
|
2210
2241
|
```bash
|
|
2211
2242
|
# Debugging download/delete races — force synchronous downloads
|
|
@@ -2213,6 +2244,9 @@ export LITDATA_ASYNC_CHUNK_PREFETCH=0
|
|
|
2213
2244
|
|
|
2214
2245
|
# Keep max_pre_download=2 even with async enabled
|
|
2215
2246
|
export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
|
|
2247
|
+
|
|
2248
|
+
# Cap overlapping remote GETs (default 8)
|
|
2249
|
+
export LITDATA_ASYNC_DOWNLOAD_CONCURRENCY=4
|
|
2216
2250
|
```
|
|
2217
2251
|
|
|
2218
2252
|
### Common environment variables
|
|
@@ -2222,6 +2256,7 @@ export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
|
|
|
2222
2256
|
| `LITDATA_CACHE_DIR` | `~/.lightning/chunks` | Default chunk cache directory |
|
|
2223
2257
|
| `LITDATA_ASYNC_CHUNK_PREFETCH` | on for remote | `0`/`1` force async chunk download overlap |
|
|
2224
2258
|
| `LITDATA_ASYNC_MIN_PRE_DOWNLOAD` | `4` | Floor for `max_pre_download` when async is on (`0` = no floor) |
|
|
2259
|
+
| `LITDATA_ASYNC_DOWNLOAD_CONCURRENCY` | `8` | Max in-flight chunk GETs per gather |
|
|
2225
2260
|
| `LITDATA_OBSTORE_STREAM_MIN_CHUNK_MIB` | `8` | S3 obstore stream chunk size (MiB) |
|
|
2226
2261
|
| `MAX_WAIT_TIME` | `120` | Seconds to wait for a chunk before error |
|
|
2227
2262
|
| `FORCE_DOWNLOAD_TIME` | `30` | Seconds before force re-download of a missing chunk |
|
|
@@ -1064,7 +1064,7 @@ Rough ImageNet order-of-magnitude on a Studio (not hard guarantees; right tuning
|
|
|
1064
1064
|
| `drop_last` | `True` if distributed else `False` | Equal length across ranks |
|
|
1065
1065
|
| `seed` | `42` | Shuffle / subsample RNG |
|
|
1066
1066
|
| `serializers` | built-ins | Custom serialize/deserialize map |
|
|
1067
|
-
| `max_cache_size` | `
|
|
1067
|
+
| `max_cache_size` | `None` | Evict consumed chunks beyond this size. Default: 75% of free disk, leaving ≥50GB when possible. Pin with `"100G"` / `"50GB"`, a fraction (`0.90`), or `MAX_CACHE_SIZE`. |
|
|
1068
1068
|
| `max_pre_download` | `2` | Chunks each worker may prefetch (raise for throughput; watch disk / RAM) |
|
|
1069
1069
|
| `subsample` | `1.0` | Fraction of data (`0.01`) or upsample (`2.5`) |
|
|
1070
1070
|
| `encryption` | `None` | `FernetEncryption` / `RSAEncryption` / custom |
|
|
@@ -1085,7 +1085,8 @@ On **Vast / NFS / local disk**, POSIX-fast is on by default (`LITDATA_POSIX_FAST
|
|
|
1085
1085
|
| All usual `torch.utils.data.DataLoader` kwargs | `batch_size`, `num_workers`, `collate_fn`, `pin_memory`, … |
|
|
1086
1086
|
| `shuffle` / `drop_last` | Forwarded to the streaming dataset |
|
|
1087
1087
|
| `profile_batches` | `int` / `True` / `False` — viztracer worker trace (see [Profile data loading](#profile-loading)) |
|
|
1088
|
-
| `
|
|
1088
|
+
| `profile_cprofile` | `True` — stdlib cProfile of the main process + worker 0 (see [Profile data loading](#profile-loading)) |
|
|
1089
|
+
| `profile_skip_batches` / `profile_dir` | Warm-up skip count; output dir for viztracer / cProfile files |
|
|
1089
1090
|
| `multiprocessing_context` | Use **`"spawn"`** (or `"forkserver"`) with `ParquetLoader` + `num_workers>0` on Linux |
|
|
1090
1091
|
|
|
1091
1092
|
Prefer `StreamingDataLoader` over a plain PyTorch `DataLoader` for optimized / combined / parallel datasets (resume + correct batch metadata).
|
|
@@ -2051,7 +2052,32 @@ Only **worker 0** is instrumented. When an `int` is used, the tracer wraps `fetc
|
|
|
2051
2052
|
|
|
2052
2053
|
- Delete or change `profile_dir` between runs — LitData removes an existing `result.json` before starting.
|
|
2053
2054
|
- Pair with a wiped chunk cache if you care about **cold** epoch behavior (`litdata cache clear`).
|
|
2054
|
-
|
|
2055
|
+
### cProfile (main + worker 0)
|
|
2056
|
+
|
|
2057
|
+
Stdlib statistical profile of **the parent process** (queue wait, unpickle, collate handoff) and **worker 0** (fetch, decode, transforms). No extra package.
|
|
2058
|
+
|
|
2059
|
+
```python
|
|
2060
|
+
loader = StreamingDataLoader(
|
|
2061
|
+
dataset,
|
|
2062
|
+
batch_size=64,
|
|
2063
|
+
num_workers=4,
|
|
2064
|
+
profile_cprofile=True,
|
|
2065
|
+
profile_dir="./profiles",
|
|
2066
|
+
)
|
|
2067
|
+
|
|
2068
|
+
for batch in loader:
|
|
2069
|
+
train_step(batch)
|
|
2070
|
+
# writes profiles/cprofile_main.prof + cprofile_worker0.prof (and .txt summaries)
|
|
2071
|
+
```
|
|
2072
|
+
|
|
2073
|
+
```bash
|
|
2074
|
+
python -m pstats profiles/cprofile_worker0.prof
|
|
2075
|
+
# then: sort tottime stats 30
|
|
2076
|
+
```
|
|
2077
|
+
|
|
2078
|
+
Do not set `profile_cprofile` and `profile_batches` together — both install a `sys.setprofile` hook. With `num_workers=0` only the main file is written. The parent profiler starts **after** workers spawn so fork does not inherit an active cProfile.
|
|
2079
|
+
|
|
2080
|
+
- For deeper LitData internals (download / read / delete timeline), use `enable_tracer()` + [Litracer](https://github.com/Lightning-AI/litracer) instead — see [Debug & Profile LitData](#debug-profile). That path is complementary: viztracer = DataLoader worker CPU timeline; cProfile = function self/cum time; Litracer = LitData pipeline events.
|
|
2055
2081
|
|
|
2056
2082
|
</details>
|
|
2057
2083
|
|
|
@@ -2101,7 +2127,9 @@ outputs = optimize(
|
|
|
2101
2127
|
|
|
2102
2128
|
Control how much disk the local chunk cache may use. Downloaded chunks are deleted after use once the cache exceeds the limit.
|
|
2103
2129
|
|
|
2104
|
-
Default `max_cache_size` is **`
|
|
2130
|
+
Default `max_cache_size` is **`None`**: LitData uses **75% of currently free disk** and still leaves **≥50GB** free when the volume is large enough for checkpoints. Pass `"100G"` / `"50GB"` for a fixed budget, or a float (`0.90`) for that fraction of currently free space. `MAX_CACHE_SIZE` overrides the constructor (size or fraction).
|
|
2131
|
+
|
|
2132
|
+
Peak disk in flight is roughly:
|
|
2105
2133
|
|
|
2106
2134
|
```
|
|
2107
2135
|
num_workers × max_pre_download × mean_chunk_size
|
|
@@ -2141,7 +2169,9 @@ for batch in StreamingDataLoader(dataset, batch_size=64, num_workers=8):
|
|
|
2141
2169
|
| `LITDATA_ASYNC_CHUNK_PREFETCH=1` | Force on |
|
|
2142
2170
|
| `LITDATA_ASYNC_CHUNK_PREFETCH=0` | Force off |
|
|
2143
2171
|
|
|
2144
|
-
When async is on, LitData raises `max_pre_download` to at least **4** so `asyncio.gather` has enough in-flight downloads (override with `LITDATA_ASYNC_MIN_PRE_DOWNLOAD`; set `0` to disable the floor). Peak disk ≈ `num_workers × max_pre_download × chunk_size` — size `max_cache_size` accordingly.
|
|
2172
|
+
When async is on, LitData raises `max_pre_download` to at least **4** so `asyncio.gather` has enough in-flight downloads (override with `LITDATA_ASYNC_MIN_PRE_DOWNLOAD`; set `0` to disable the floor). Concurrent GETs are capped by `LITDATA_ASYNC_DOWNLOAD_CONCURRENCY` (default **8**). The prepare thread drains the prefetch queue up to that gather width whenever a cache slot is free. Peak disk ≈ `num_workers × max_pre_download × chunk_size` — size `max_cache_size` accordingly.
|
|
2173
|
+
|
|
2174
|
+
The reader does **not** poll the cache directory for each chunk. After a download finishes, the downloader atomically `os.replace`s the temp file and sets an in-process `Event`. Compressed chunks set that Event only after decompress publishes the readable `.bin`. Other DataLoader workers still fall back to a short filesystem check (Events are per process).
|
|
2145
2175
|
|
|
2146
2176
|
```bash
|
|
2147
2177
|
# Debugging download/delete races — force synchronous downloads
|
|
@@ -2149,6 +2179,9 @@ export LITDATA_ASYNC_CHUNK_PREFETCH=0
|
|
|
2149
2179
|
|
|
2150
2180
|
# Keep max_pre_download=2 even with async enabled
|
|
2151
2181
|
export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
|
|
2182
|
+
|
|
2183
|
+
# Cap overlapping remote GETs (default 8)
|
|
2184
|
+
export LITDATA_ASYNC_DOWNLOAD_CONCURRENCY=4
|
|
2152
2185
|
```
|
|
2153
2186
|
|
|
2154
2187
|
### Common environment variables
|
|
@@ -2158,6 +2191,7 @@ export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
|
|
|
2158
2191
|
| `LITDATA_CACHE_DIR` | `~/.lightning/chunks` | Default chunk cache directory |
|
|
2159
2192
|
| `LITDATA_ASYNC_CHUNK_PREFETCH` | on for remote | `0`/`1` force async chunk download overlap |
|
|
2160
2193
|
| `LITDATA_ASYNC_MIN_PRE_DOWNLOAD` | `4` | Floor for `max_pre_download` when async is on (`0` = no floor) |
|
|
2194
|
+
| `LITDATA_ASYNC_DOWNLOAD_CONCURRENCY` | `8` | Max in-flight chunk GETs per gather |
|
|
2161
2195
|
| `LITDATA_OBSTORE_STREAM_MIN_CHUNK_MIB` | `8` | S3 obstore stream chunk size (MiB) |
|
|
2162
2196
|
| `MAX_WAIT_TIME` | `120` | Seconds to wait for a chunk before error |
|
|
2163
2197
|
| `FORCE_DOWNLOAD_TIME` | `30` | Seconds before force re-download of a missing chunk |
|
|
@@ -54,6 +54,8 @@ _THREAD_LOOPS = threading.local()
|
|
|
54
54
|
# Empirically, async gather on real S3 is bottlenecked when max_pre_download==2
|
|
55
55
|
# (only 1–2 in-flight). Floor to 4 when the feature is enabled unless overridden.
|
|
56
56
|
_DEFAULT_ASYNC_MIN_PRE_DOWNLOAD = 4
|
|
57
|
+
# Cap in-flight GETs so a large drain batch does not open one connection per slot.
|
|
58
|
+
_DEFAULT_ASYNC_DOWNLOAD_CONCURRENCY = 8
|
|
57
59
|
|
|
58
60
|
|
|
59
61
|
def async_chunk_prefetch_enabled(remote_dir: str | None = None) -> bool:
|
|
@@ -81,6 +83,17 @@ def async_prefetch_min_pre_download() -> int:
|
|
|
81
83
|
return max(0, int(raw))
|
|
82
84
|
|
|
83
85
|
|
|
86
|
+
def async_download_concurrency(n_chunks: int) -> int:
|
|
87
|
+
"""How many remote chunk GETs to run at once inside ``asyncio.gather``.
|
|
88
|
+
|
|
89
|
+
Override with ``LITDATA_ASYNC_DOWNLOAD_CONCURRENCY`` (default 8). Always at
|
|
90
|
+
least 1 and at most ``n_chunks``.
|
|
91
|
+
"""
|
|
92
|
+
raw = os.getenv("LITDATA_ASYNC_DOWNLOAD_CONCURRENCY")
|
|
93
|
+
limit = _DEFAULT_ASYNC_DOWNLOAD_CONCURRENCY if raw is None else max(1, int(raw))
|
|
94
|
+
return max(1, min(limit, max(1, n_chunks)))
|
|
95
|
+
|
|
96
|
+
|
|
84
97
|
def apply_async_pre_download_floor(max_pre_download: int, remote_dir: str | None = None) -> int:
|
|
85
98
|
"""Raise ``max_pre_download`` when async prefetch needs gather width."""
|
|
86
99
|
if not async_chunk_prefetch_enabled(remote_dir):
|
|
@@ -133,6 +146,7 @@ def _remote_join(remote_dir: str, filename: str) -> str:
|
|
|
133
146
|
async def _adownload_file_to_path(downloader: Downloader, remote_filepath: str, local_filepath: str) -> None:
|
|
134
147
|
"""Fetch ``remote_filepath`` asynchronously and publish atomically."""
|
|
135
148
|
if os.path.exists(local_filepath):
|
|
149
|
+
downloader._notify_published(local_filepath)
|
|
136
150
|
return
|
|
137
151
|
# Prefer streaming-to-disk when the backend overrides adownload_file.
|
|
138
152
|
if type(downloader).adownload_file is not Downloader.adownload_file:
|
|
@@ -149,7 +163,7 @@ async def _adownload_file_to_path(downloader: Downloader, remote_filepath: str,
|
|
|
149
163
|
os.makedirs(os.path.dirname(local_filepath) or ".", exist_ok=True)
|
|
150
164
|
with open(tmp_path, "wb") as f:
|
|
151
165
|
f.write(data)
|
|
152
|
-
downloader.
|
|
166
|
+
downloader._publish_file(tmp_path, local_filepath)
|
|
153
167
|
except Exception:
|
|
154
168
|
with contextlib.suppress(FileNotFoundError, PermissionError):
|
|
155
169
|
os.remove(tmp_path)
|
|
@@ -172,7 +186,7 @@ async def _adownload_chunk_index(config: ChunksConfig, chunk_index: int) -> None
|
|
|
172
186
|
)
|
|
173
187
|
|
|
174
188
|
if os.path.exists(local_chunkpath):
|
|
175
|
-
config.try_decompress
|
|
189
|
+
await asyncio.to_thread(config.try_decompress, local_chunkpath)
|
|
176
190
|
if lazily_ref_counted:
|
|
177
191
|
downloader._increment_local_lock(lock_path, chunk_index)
|
|
178
192
|
return
|
|
@@ -186,7 +200,7 @@ async def _adownload_chunk_index(config: ChunksConfig, chunk_index: int) -> None
|
|
|
186
200
|
# Overlap blocking SDK calls across threads when native async is unavailable.
|
|
187
201
|
await asyncio.to_thread(downloader.download_chunk_from_index, chunk_index)
|
|
188
202
|
|
|
189
|
-
config.try_decompress
|
|
203
|
+
await asyncio.to_thread(config.try_decompress, local_chunkpath)
|
|
190
204
|
|
|
191
205
|
|
|
192
206
|
async def adownload_chunk_indexes(config: ChunksConfig, chunk_indexes: list[int]) -> None:
|
|
@@ -196,7 +210,17 @@ async def adownload_chunk_indexes(config: ChunksConfig, chunk_indexes: list[int]
|
|
|
196
210
|
if len(chunk_indexes) == 1:
|
|
197
211
|
await _adownload_chunk_index(config, chunk_indexes[0])
|
|
198
212
|
return
|
|
199
|
-
|
|
213
|
+
limit = async_download_concurrency(len(chunk_indexes))
|
|
214
|
+
if limit >= len(chunk_indexes):
|
|
215
|
+
await asyncio.gather(*[_adownload_chunk_index(config, idx) for idx in chunk_indexes])
|
|
216
|
+
return
|
|
217
|
+
sem = asyncio.Semaphore(limit)
|
|
218
|
+
|
|
219
|
+
async def _one(idx: int) -> None:
|
|
220
|
+
async with sem:
|
|
221
|
+
await _adownload_chunk_index(config, idx)
|
|
222
|
+
|
|
223
|
+
await asyncio.gather(*[_one(idx) for idx in chunk_indexes])
|
|
200
224
|
|
|
201
225
|
|
|
202
226
|
def _thread_event_loop() -> asyncio.AbstractEventLoop:
|
|
@@ -27,7 +27,7 @@ from litdata.streaming.serializers import Serializer
|
|
|
27
27
|
from litdata.streaming.writer import BinaryWriter
|
|
28
28
|
from litdata.utilities.encryption import Encryption
|
|
29
29
|
from litdata.utilities.env import _DistributedEnv, _WorkerEnv
|
|
30
|
-
from litdata.utilities.format import
|
|
30
|
+
from litdata.utilities.format import _resolve_max_cache_size
|
|
31
31
|
|
|
32
32
|
logger = logging.Logger(__name__)
|
|
33
33
|
|
|
@@ -43,7 +43,7 @@ class Cache:
|
|
|
43
43
|
chunk_size: int | None = None,
|
|
44
44
|
chunk_bytes: int | str | None = None,
|
|
45
45
|
item_loader: BaseItemLoader | None = None,
|
|
46
|
-
max_cache_size: int | str =
|
|
46
|
+
max_cache_size: int | float | str | None = None,
|
|
47
47
|
serializers: dict[str, Serializer] | None = None,
|
|
48
48
|
writer_chunk_index: int | None = None,
|
|
49
49
|
storage_options: dict | None = {},
|
|
@@ -64,7 +64,8 @@ class Cache:
|
|
|
64
64
|
chunk_bytes: The maximum number of bytes within a chunk.
|
|
65
65
|
chunk_size: The maximum number of items within a chunk.
|
|
66
66
|
item_loader: The object responsible to generate the chunk intervals and load an item froma chunk.
|
|
67
|
-
max_cache_size:
|
|
67
|
+
max_cache_size: Cache budget. ``None`` uses 75% of free disk (see ``StreamingDataset``).
|
|
68
|
+
A float such as ``0.90`` is that fraction of currently free space.
|
|
68
69
|
serializers: Provide your own serializers.
|
|
69
70
|
writer_chunk_index: The index of the chunk to start from when writing.
|
|
70
71
|
storage_options: Additional connection options for accessing storage services.
|
|
@@ -93,7 +94,7 @@ class Cache:
|
|
|
93
94
|
self._cache_dir,
|
|
94
95
|
subsampled_files=subsampled_files,
|
|
95
96
|
region_of_interest=region_of_interest,
|
|
96
|
-
max_cache_size=
|
|
97
|
+
max_cache_size=_resolve_max_cache_size(max_cache_size, self._cache_dir),
|
|
97
98
|
remote_input_dir=input_dir.url,
|
|
98
99
|
compression=compression,
|
|
99
100
|
encryption=encryption,
|