litdata 0.2.70__tar.gz → 0.2.71__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. {litdata-0.2.70/src/litdata.egg-info → litdata-0.2.71}/PKG-INFO +40 -6
  2. {litdata-0.2.70 → litdata-0.2.71}/README.md +39 -5
  3. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/__about__.py +1 -1
  4. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/async_prefetch.py +28 -4
  5. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/cache.py +5 -4
  6. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/compression.py +18 -13
  7. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/config.py +49 -7
  8. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/dataloader.py +184 -49
  9. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/dataset.py +38 -23
  10. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/downloader.py +37 -9
  11. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/item_loader.py +35 -14
  12. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/reader.py +59 -14
  13. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/serializers.py +3 -1
  14. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/shuffle.py +57 -29
  15. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/writer.py +52 -8
  16. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/_pytree.py +79 -19
  17. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/env.py +12 -0
  18. litdata-0.2.71/src/litdata/utilities/format.py +168 -0
  19. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/shuffle.py +64 -12
  20. {litdata-0.2.70 → litdata-0.2.71/src/litdata.egg-info}/PKG-INFO +40 -6
  21. litdata-0.2.70/src/litdata/utilities/format.py +0 -57
  22. {litdata-0.2.70 → litdata-0.2.71}/CONTRIBUTING.md +0 -0
  23. {litdata-0.2.70 → litdata-0.2.71}/LICENSE +0 -0
  24. {litdata-0.2.70 → litdata-0.2.71}/MANIFEST.in +0 -0
  25. {litdata-0.2.70 → litdata-0.2.71}/requirements.txt +0 -0
  26. {litdata-0.2.70 → litdata-0.2.71}/setup.cfg +0 -0
  27. {litdata-0.2.70 → litdata-0.2.71}/setup.py +0 -0
  28. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/__init__.py +0 -0
  29. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/__main__.py +0 -0
  30. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/cli/__init__.py +0 -0
  31. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/cli/commands.py +0 -0
  32. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/cli/handler/__init__.py +0 -0
  33. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/cli/handler/cache.py +0 -0
  34. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/cli/handler/optimize.py +0 -0
  35. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/cli/parser.py +0 -0
  36. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/constants.py +0 -0
  37. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/debugger.py +0 -0
  38. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/exceptions.py +0 -0
  39. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/helpers.py +0 -0
  40. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/imports.py +0 -0
  41. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/processing/__init__.py +0 -0
  42. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/processing/complete.py +0 -0
  43. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/processing/data_processor.py +0 -0
  44. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/processing/functions.py +0 -0
  45. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/processing/media_folder.py +0 -0
  46. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/processing/readers.py +0 -0
  47. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/processing/utilities.py +0 -0
  48. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/raw/__init__.py +0 -0
  49. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/raw/dataset.py +0 -0
  50. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/raw/indexer.py +0 -0
  51. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/raw/types.py +0 -0
  52. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/requirements.py +0 -0
  53. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/__init__.py +0 -0
  54. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/client.py +0 -0
  55. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/collate.py +0 -0
  56. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/combined.py +0 -0
  57. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/dataset_update.py +0 -0
  58. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/elastic.py +0 -0
  59. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/fs_provider.py +0 -0
  60. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/parallel.py +0 -0
  61. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/posix_fast.py +0 -0
  62. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/resolver.py +0 -0
  63. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/sampler.py +0 -0
  64. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/streaming/timing.py +0 -0
  65. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/types.py +0 -0
  66. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/__init__.py +0 -0
  67. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/base.py +0 -0
  68. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/breakpoint.py +0 -0
  69. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/broadcast.py +0 -0
  70. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/dataset_utilities.py +0 -0
  71. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/encryption.py +0 -0
  72. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/hf_dataset.py +0 -0
  73. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/keys_index.py +0 -0
  74. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/packing.py +0 -0
  75. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/parquet.py +0 -0
  76. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/subsample.py +0 -0
  77. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/torch_utils.py +0 -0
  78. {litdata-0.2.70 → litdata-0.2.71}/src/litdata/utilities/train_test_split.py +0 -0
  79. {litdata-0.2.70 → litdata-0.2.71}/src/litdata.egg-info/SOURCES.txt +0 -0
  80. {litdata-0.2.70 → litdata-0.2.71}/src/litdata.egg-info/dependency_links.txt +0 -0
  81. {litdata-0.2.70 → litdata-0.2.71}/src/litdata.egg-info/entry_points.txt +0 -0
  82. {litdata-0.2.70 → litdata-0.2.71}/src/litdata.egg-info/not-zip-safe +0 -0
  83. {litdata-0.2.70 → litdata-0.2.71}/src/litdata.egg-info/requires.txt +0 -0
  84. {litdata-0.2.70 → litdata-0.2.71}/src/litdata.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: litdata
3
- Version: 0.2.70
3
+ Version: 0.2.71
4
4
  Summary: The Deep Learning framework to train, deploy, and ship AI products Lightning fast.
5
5
  Home-page: https://github.com/Lightning-AI/litdata
6
6
  Download-URL: https://github.com/Lightning-AI/litdata
@@ -1128,7 +1128,7 @@ Rough ImageNet order-of-magnitude on a Studio (not hard guarantees; right tuning
1128
1128
  | `drop_last` | `True` if distributed else `False` | Equal length across ranks |
1129
1129
  | `seed` | `42` | Shuffle / subsample RNG |
1130
1130
  | `serializers` | built-ins | Custom serialize/deserialize map |
1131
- | `max_cache_size` | `"100GB"` | Evict consumed chunks beyond this size |
1131
+ | `max_cache_size` | `None` | Evict consumed chunks beyond this size. Default: 75% of free disk, leaving ≥50GB when possible. Pin with `"100G"` / `"50GB"`, a fraction (`0.90`), or `MAX_CACHE_SIZE`. |
1132
1132
  | `max_pre_download` | `2` | Chunks each worker may prefetch (raise for throughput; watch disk / RAM) |
1133
1133
  | `subsample` | `1.0` | Fraction of data (`0.01`) or upsample (`2.5`) |
1134
1134
  | `encryption` | `None` | `FernetEncryption` / `RSAEncryption` / custom |
@@ -1149,7 +1149,8 @@ On **Vast / NFS / local disk**, POSIX-fast is on by default (`LITDATA_POSIX_FAST
1149
1149
  | All usual `torch.utils.data.DataLoader` kwargs | `batch_size`, `num_workers`, `collate_fn`, `pin_memory`, … |
1150
1150
  | `shuffle` / `drop_last` | Forwarded to the streaming dataset |
1151
1151
  | `profile_batches` | `int` / `True` / `False` — viztracer worker trace (see [Profile data loading](#profile-loading)) |
1152
- | `profile_skip_batches` / `profile_dir` | Warm-up skip count; output dir for `result.json` |
1152
+ | `profile_cprofile` | `True` — stdlib cProfile of the main process + worker 0 (see [Profile data loading](#profile-loading)) |
1153
+ | `profile_skip_batches` / `profile_dir` | Warm-up skip count; output dir for viztracer / cProfile files |
1153
1154
  | `multiprocessing_context` | Use **`"spawn"`** (or `"forkserver"`) with `ParquetLoader` + `num_workers>0` on Linux |
1154
1155
 
1155
1156
  Prefer `StreamingDataLoader` over a plain PyTorch `DataLoader` for optimized / combined / parallel datasets (resume + correct batch metadata).
@@ -2115,7 +2116,32 @@ Only **worker 0** is instrumented. When an `int` is used, the tracer wraps `fetc
2115
2116
 
2116
2117
  - Delete or change `profile_dir` between runs — LitData removes an existing `result.json` before starting.
2117
2118
  - Pair with a wiped chunk cache if you care about **cold** epoch behavior (`litdata cache clear`).
2118
- - For deeper LitData internals (download / read / delete timeline), use `enable_tracer()` + [Litracer](https://github.com/Lightning-AI/litracer) instead — see [Debug & Profile LitData](#debug-profile). That path is complementary: viztracer = DataLoader worker CPU timeline; Litracer = LitData pipeline events.
2119
+ ### cProfile (main + worker 0)
2120
+
2121
+ Stdlib statistical profile of **the parent process** (queue wait, unpickle, collate handoff) and **worker 0** (fetch, decode, transforms). No extra package.
2122
+
2123
+ ```python
2124
+ loader = StreamingDataLoader(
2125
+ dataset,
2126
+ batch_size=64,
2127
+ num_workers=4,
2128
+ profile_cprofile=True,
2129
+ profile_dir="./profiles",
2130
+ )
2131
+
2132
+ for batch in loader:
2133
+ train_step(batch)
2134
+ # writes profiles/cprofile_main.prof + cprofile_worker0.prof (and .txt summaries)
2135
+ ```
2136
+
2137
+ ```bash
2138
+ python -m pstats profiles/cprofile_worker0.prof
2139
+ # then: sort tottime stats 30
2140
+ ```
2141
+
2142
+ Do not set `profile_cprofile` and `profile_batches` together — both install a `sys.setprofile` hook. With `num_workers=0` only the main file is written. The parent profiler starts **after** workers spawn so fork does not inherit an active cProfile.
2143
+
2144
+ - For deeper LitData internals (download / read / delete timeline), use `enable_tracer()` + [Litracer](https://github.com/Lightning-AI/litracer) instead — see [Debug & Profile LitData](#debug-profile). That path is complementary: viztracer = DataLoader worker CPU timeline; cProfile = function self/cum time; Litracer = LitData pipeline events.
2119
2145
 
2120
2146
  </details>
2121
2147
 
@@ -2165,7 +2191,9 @@ outputs = optimize(
2165
2191
 
2166
2192
  Control how much disk the local chunk cache may use. Downloaded chunks are deleted after use once the cache exceeds the limit.
2167
2193
 
2168
- Default `max_cache_size` is **`100GB`**. Peak disk in flight is roughly:
2194
+ Default `max_cache_size` is **`None`**: LitData uses **75% of currently free disk** and still leaves **≥50GB** free when the volume is large enough for checkpoints. Pass `"100G"` / `"50GB"` for a fixed budget, or a float (`0.90`) for that fraction of currently free space. `MAX_CACHE_SIZE` overrides the constructor (size or fraction).
2195
+
2196
+ Peak disk in flight is roughly:
2169
2197
 
2170
2198
  ```
2171
2199
  num_workers × max_pre_download × mean_chunk_size
@@ -2205,7 +2233,9 @@ for batch in StreamingDataLoader(dataset, batch_size=64, num_workers=8):
2205
2233
  | `LITDATA_ASYNC_CHUNK_PREFETCH=1` | Force on |
2206
2234
  | `LITDATA_ASYNC_CHUNK_PREFETCH=0` | Force off |
2207
2235
 
2208
- When async is on, LitData raises `max_pre_download` to at least **4** so `asyncio.gather` has enough in-flight downloads (override with `LITDATA_ASYNC_MIN_PRE_DOWNLOAD`; set `0` to disable the floor). Peak disk ≈ `num_workers × max_pre_download × chunk_size` — size `max_cache_size` accordingly.
2236
+ When async is on, LitData raises `max_pre_download` to at least **4** so `asyncio.gather` has enough in-flight downloads (override with `LITDATA_ASYNC_MIN_PRE_DOWNLOAD`; set `0` to disable the floor). Concurrent GETs are capped by `LITDATA_ASYNC_DOWNLOAD_CONCURRENCY` (default **8**). The prepare thread drains the prefetch queue up to that gather width whenever a cache slot is free. Peak disk ≈ `num_workers × max_pre_download × chunk_size` — size `max_cache_size` accordingly.
2237
+
2238
+ The reader does **not** poll the cache directory for each chunk. After a download finishes, the downloader atomically `os.replace`s the temp file and sets an in-process `Event`. Compressed chunks set that Event only after decompress publishes the readable `.bin`. Other DataLoader workers still fall back to a short filesystem check (Events are per process).
2209
2239
 
2210
2240
  ```bash
2211
2241
  # Debugging download/delete races — force synchronous downloads
@@ -2213,6 +2243,9 @@ export LITDATA_ASYNC_CHUNK_PREFETCH=0
2213
2243
 
2214
2244
  # Keep max_pre_download=2 even with async enabled
2215
2245
  export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
2246
+
2247
+ # Cap overlapping remote GETs (default 8)
2248
+ export LITDATA_ASYNC_DOWNLOAD_CONCURRENCY=4
2216
2249
  ```
2217
2250
 
2218
2251
  ### Common environment variables
@@ -2222,6 +2255,7 @@ export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
2222
2255
  | `LITDATA_CACHE_DIR` | `~/.lightning/chunks` | Default chunk cache directory |
2223
2256
  | `LITDATA_ASYNC_CHUNK_PREFETCH` | on for remote | `0`/`1` force async chunk download overlap |
2224
2257
  | `LITDATA_ASYNC_MIN_PRE_DOWNLOAD` | `4` | Floor for `max_pre_download` when async is on (`0` = no floor) |
2258
+ | `LITDATA_ASYNC_DOWNLOAD_CONCURRENCY` | `8` | Max in-flight chunk GETs per gather |
2225
2259
  | `LITDATA_OBSTORE_STREAM_MIN_CHUNK_MIB` | `8` | S3 obstore stream chunk size (MiB) |
2226
2260
  | `MAX_WAIT_TIME` | `120` | Seconds to wait for a chunk before error |
2227
2261
  | `FORCE_DOWNLOAD_TIME` | `30` | Seconds before force re-download of a missing chunk |
@@ -1064,7 +1064,7 @@ Rough ImageNet order-of-magnitude on a Studio (not hard guarantees; right tuning
1064
1064
  | `drop_last` | `True` if distributed else `False` | Equal length across ranks |
1065
1065
  | `seed` | `42` | Shuffle / subsample RNG |
1066
1066
  | `serializers` | built-ins | Custom serialize/deserialize map |
1067
- | `max_cache_size` | `"100GB"` | Evict consumed chunks beyond this size |
1067
+ | `max_cache_size` | `None` | Evict consumed chunks beyond this size. Default: 75% of free disk, leaving ≥50GB when possible. Pin with `"100G"` / `"50GB"`, a fraction (`0.90`), or `MAX_CACHE_SIZE`. |
1068
1068
  | `max_pre_download` | `2` | Chunks each worker may prefetch (raise for throughput; watch disk / RAM) |
1069
1069
  | `subsample` | `1.0` | Fraction of data (`0.01`) or upsample (`2.5`) |
1070
1070
  | `encryption` | `None` | `FernetEncryption` / `RSAEncryption` / custom |
@@ -1085,7 +1085,8 @@ On **Vast / NFS / local disk**, POSIX-fast is on by default (`LITDATA_POSIX_FAST
1085
1085
  | All usual `torch.utils.data.DataLoader` kwargs | `batch_size`, `num_workers`, `collate_fn`, `pin_memory`, … |
1086
1086
  | `shuffle` / `drop_last` | Forwarded to the streaming dataset |
1087
1087
  | `profile_batches` | `int` / `True` / `False` — viztracer worker trace (see [Profile data loading](#profile-loading)) |
1088
- | `profile_skip_batches` / `profile_dir` | Warm-up skip count; output dir for `result.json` |
1088
+ | `profile_cprofile` | `True` — stdlib cProfile of the main process + worker 0 (see [Profile data loading](#profile-loading)) |
1089
+ | `profile_skip_batches` / `profile_dir` | Warm-up skip count; output dir for viztracer / cProfile files |
1089
1090
  | `multiprocessing_context` | Use **`"spawn"`** (or `"forkserver"`) with `ParquetLoader` + `num_workers>0` on Linux |
1090
1091
 
1091
1092
  Prefer `StreamingDataLoader` over a plain PyTorch `DataLoader` for optimized / combined / parallel datasets (resume + correct batch metadata).
@@ -2051,7 +2052,32 @@ Only **worker 0** is instrumented. When an `int` is used, the tracer wraps `fetc
2051
2052
 
2052
2053
  - Delete or change `profile_dir` between runs — LitData removes an existing `result.json` before starting.
2053
2054
  - Pair with a wiped chunk cache if you care about **cold** epoch behavior (`litdata cache clear`).
2054
- - For deeper LitData internals (download / read / delete timeline), use `enable_tracer()` + [Litracer](https://github.com/Lightning-AI/litracer) instead — see [Debug & Profile LitData](#debug-profile). That path is complementary: viztracer = DataLoader worker CPU timeline; Litracer = LitData pipeline events.
2055
+ ### cProfile (main + worker 0)
2056
+
2057
+ Stdlib statistical profile of **the parent process** (queue wait, unpickle, collate handoff) and **worker 0** (fetch, decode, transforms). No extra package.
2058
+
2059
+ ```python
2060
+ loader = StreamingDataLoader(
2061
+ dataset,
2062
+ batch_size=64,
2063
+ num_workers=4,
2064
+ profile_cprofile=True,
2065
+ profile_dir="./profiles",
2066
+ )
2067
+
2068
+ for batch in loader:
2069
+ train_step(batch)
2070
+ # writes profiles/cprofile_main.prof + cprofile_worker0.prof (and .txt summaries)
2071
+ ```
2072
+
2073
+ ```bash
2074
+ python -m pstats profiles/cprofile_worker0.prof
2075
+ # then: sort tottime stats 30
2076
+ ```
2077
+
2078
+ Do not set `profile_cprofile` and `profile_batches` together — both install a `sys.setprofile` hook. With `num_workers=0` only the main file is written. The parent profiler starts **after** workers spawn so fork does not inherit an active cProfile.
2079
+
2080
+ - For deeper LitData internals (download / read / delete timeline), use `enable_tracer()` + [Litracer](https://github.com/Lightning-AI/litracer) instead — see [Debug & Profile LitData](#debug-profile). That path is complementary: viztracer = DataLoader worker CPU timeline; cProfile = function self/cum time; Litracer = LitData pipeline events.
2055
2081
 
2056
2082
  </details>
2057
2083
 
@@ -2101,7 +2127,9 @@ outputs = optimize(
2101
2127
 
2102
2128
  Control how much disk the local chunk cache may use. Downloaded chunks are deleted after use once the cache exceeds the limit.
2103
2129
 
2104
- Default `max_cache_size` is **`100GB`**. Peak disk in flight is roughly:
2130
+ Default `max_cache_size` is **`None`**: LitData uses **75% of currently free disk** and still leaves **≥50GB** free when the volume is large enough for checkpoints. Pass `"100G"` / `"50GB"` for a fixed budget, or a float (`0.90`) for that fraction of currently free space. `MAX_CACHE_SIZE` overrides the constructor (size or fraction).
2131
+
2132
+ Peak disk in flight is roughly:
2105
2133
 
2106
2134
  ```
2107
2135
  num_workers × max_pre_download × mean_chunk_size
@@ -2141,7 +2169,9 @@ for batch in StreamingDataLoader(dataset, batch_size=64, num_workers=8):
2141
2169
  | `LITDATA_ASYNC_CHUNK_PREFETCH=1` | Force on |
2142
2170
  | `LITDATA_ASYNC_CHUNK_PREFETCH=0` | Force off |
2143
2171
 
2144
- When async is on, LitData raises `max_pre_download` to at least **4** so `asyncio.gather` has enough in-flight downloads (override with `LITDATA_ASYNC_MIN_PRE_DOWNLOAD`; set `0` to disable the floor). Peak disk ≈ `num_workers × max_pre_download × chunk_size` — size `max_cache_size` accordingly.
2172
+ When async is on, LitData raises `max_pre_download` to at least **4** so `asyncio.gather` has enough in-flight downloads (override with `LITDATA_ASYNC_MIN_PRE_DOWNLOAD`; set `0` to disable the floor). Concurrent GETs are capped by `LITDATA_ASYNC_DOWNLOAD_CONCURRENCY` (default **8**). The prepare thread drains the prefetch queue up to that gather width whenever a cache slot is free. Peak disk ≈ `num_workers × max_pre_download × chunk_size` — size `max_cache_size` accordingly.
2173
+
2174
+ The reader does **not** poll the cache directory for each chunk. After a download finishes, the downloader atomically `os.replace`s the temp file and sets an in-process `Event`. Compressed chunks set that Event only after decompress publishes the readable `.bin`. Other DataLoader workers still fall back to a short filesystem check (Events are per process).
2145
2175
 
2146
2176
  ```bash
2147
2177
  # Debugging download/delete races — force synchronous downloads
@@ -2149,6 +2179,9 @@ export LITDATA_ASYNC_CHUNK_PREFETCH=0
2149
2179
 
2150
2180
  # Keep max_pre_download=2 even with async enabled
2151
2181
  export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
2182
+
2183
+ # Cap overlapping remote GETs (default 8)
2184
+ export LITDATA_ASYNC_DOWNLOAD_CONCURRENCY=4
2152
2185
  ```
2153
2186
 
2154
2187
  ### Common environment variables
@@ -2158,6 +2191,7 @@ export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
2158
2191
  | `LITDATA_CACHE_DIR` | `~/.lightning/chunks` | Default chunk cache directory |
2159
2192
  | `LITDATA_ASYNC_CHUNK_PREFETCH` | on for remote | `0`/`1` force async chunk download overlap |
2160
2193
  | `LITDATA_ASYNC_MIN_PRE_DOWNLOAD` | `4` | Floor for `max_pre_download` when async is on (`0` = no floor) |
2194
+ | `LITDATA_ASYNC_DOWNLOAD_CONCURRENCY` | `8` | Max in-flight chunk GETs per gather |
2161
2195
  | `LITDATA_OBSTORE_STREAM_MIN_CHUNK_MIB` | `8` | S3 obstore stream chunk size (MiB) |
2162
2196
  | `MAX_WAIT_TIME` | `120` | Seconds to wait for a chunk before error |
2163
2197
  | `FORCE_DOWNLOAD_TIME` | `30` | Seconds before force re-download of a missing chunk |
@@ -14,7 +14,7 @@
14
14
 
15
15
  import time
16
16
 
17
- __version__ = "0.2.70"
17
+ __version__ = "0.2.71"
18
18
  __author__ = "Lightning AI et al."
19
19
  __author_email__ = "pytorch@lightning.ai"
20
20
  __license__ = "Apache-2.0"
@@ -54,6 +54,8 @@ _THREAD_LOOPS = threading.local()
54
54
  # Empirically, async gather on real S3 is bottlenecked when max_pre_download==2
55
55
  # (only 1–2 in-flight). Floor to 4 when the feature is enabled unless overridden.
56
56
  _DEFAULT_ASYNC_MIN_PRE_DOWNLOAD = 4
57
+ # Cap in-flight GETs so a large drain batch does not open one connection per slot.
58
+ _DEFAULT_ASYNC_DOWNLOAD_CONCURRENCY = 8
57
59
 
58
60
 
59
61
  def async_chunk_prefetch_enabled(remote_dir: str | None = None) -> bool:
@@ -81,6 +83,17 @@ def async_prefetch_min_pre_download() -> int:
81
83
  return max(0, int(raw))
82
84
 
83
85
 
86
+ def async_download_concurrency(n_chunks: int) -> int:
87
+ """How many remote chunk GETs to run at once inside ``asyncio.gather``.
88
+
89
+ Override with ``LITDATA_ASYNC_DOWNLOAD_CONCURRENCY`` (default 8). Always at
90
+ least 1 and at most ``n_chunks``.
91
+ """
92
+ raw = os.getenv("LITDATA_ASYNC_DOWNLOAD_CONCURRENCY")
93
+ limit = _DEFAULT_ASYNC_DOWNLOAD_CONCURRENCY if raw is None else max(1, int(raw))
94
+ return max(1, min(limit, max(1, n_chunks)))
95
+
96
+
84
97
  def apply_async_pre_download_floor(max_pre_download: int, remote_dir: str | None = None) -> int:
85
98
  """Raise ``max_pre_download`` when async prefetch needs gather width."""
86
99
  if not async_chunk_prefetch_enabled(remote_dir):
@@ -133,6 +146,7 @@ def _remote_join(remote_dir: str, filename: str) -> str:
133
146
  async def _adownload_file_to_path(downloader: Downloader, remote_filepath: str, local_filepath: str) -> None:
134
147
  """Fetch ``remote_filepath`` asynchronously and publish atomically."""
135
148
  if os.path.exists(local_filepath):
149
+ downloader._notify_published(local_filepath)
136
150
  return
137
151
  # Prefer streaming-to-disk when the backend overrides adownload_file.
138
152
  if type(downloader).adownload_file is not Downloader.adownload_file:
@@ -149,7 +163,7 @@ async def _adownload_file_to_path(downloader: Downloader, remote_filepath: str,
149
163
  os.makedirs(os.path.dirname(local_filepath) or ".", exist_ok=True)
150
164
  with open(tmp_path, "wb") as f:
151
165
  f.write(data)
152
- downloader._atomic_replace(tmp_path, local_filepath)
166
+ downloader._publish_file(tmp_path, local_filepath)
153
167
  except Exception:
154
168
  with contextlib.suppress(FileNotFoundError, PermissionError):
155
169
  os.remove(tmp_path)
@@ -172,7 +186,7 @@ async def _adownload_chunk_index(config: ChunksConfig, chunk_index: int) -> None
172
186
  )
173
187
 
174
188
  if os.path.exists(local_chunkpath):
175
- config.try_decompress(local_chunkpath)
189
+ await asyncio.to_thread(config.try_decompress, local_chunkpath)
176
190
  if lazily_ref_counted:
177
191
  downloader._increment_local_lock(lock_path, chunk_index)
178
192
  return
@@ -186,7 +200,7 @@ async def _adownload_chunk_index(config: ChunksConfig, chunk_index: int) -> None
186
200
  # Overlap blocking SDK calls across threads when native async is unavailable.
187
201
  await asyncio.to_thread(downloader.download_chunk_from_index, chunk_index)
188
202
 
189
- config.try_decompress(local_chunkpath)
203
+ await asyncio.to_thread(config.try_decompress, local_chunkpath)
190
204
 
191
205
 
192
206
  async def adownload_chunk_indexes(config: ChunksConfig, chunk_indexes: list[int]) -> None:
@@ -196,7 +210,17 @@ async def adownload_chunk_indexes(config: ChunksConfig, chunk_indexes: list[int]
196
210
  if len(chunk_indexes) == 1:
197
211
  await _adownload_chunk_index(config, chunk_indexes[0])
198
212
  return
199
- await asyncio.gather(*[_adownload_chunk_index(config, idx) for idx in chunk_indexes])
213
+ limit = async_download_concurrency(len(chunk_indexes))
214
+ if limit >= len(chunk_indexes):
215
+ await asyncio.gather(*[_adownload_chunk_index(config, idx) for idx in chunk_indexes])
216
+ return
217
+ sem = asyncio.Semaphore(limit)
218
+
219
+ async def _one(idx: int) -> None:
220
+ async with sem:
221
+ await _adownload_chunk_index(config, idx)
222
+
223
+ await asyncio.gather(*[_one(idx) for idx in chunk_indexes])
200
224
 
201
225
 
202
226
  def _thread_event_loop() -> asyncio.AbstractEventLoop:
@@ -27,7 +27,7 @@ from litdata.streaming.serializers import Serializer
27
27
  from litdata.streaming.writer import BinaryWriter
28
28
  from litdata.utilities.encryption import Encryption
29
29
  from litdata.utilities.env import _DistributedEnv, _WorkerEnv
30
- from litdata.utilities.format import _convert_bytes_to_int
30
+ from litdata.utilities.format import _resolve_max_cache_size
31
31
 
32
32
  logger = logging.Logger(__name__)
33
33
 
@@ -43,7 +43,7 @@ class Cache:
43
43
  chunk_size: int | None = None,
44
44
  chunk_bytes: int | str | None = None,
45
45
  item_loader: BaseItemLoader | None = None,
46
- max_cache_size: int | str = "100GB",
46
+ max_cache_size: int | float | str | None = None,
47
47
  serializers: dict[str, Serializer] | None = None,
48
48
  writer_chunk_index: int | None = None,
49
49
  storage_options: dict | None = {},
@@ -64,7 +64,8 @@ class Cache:
64
64
  chunk_bytes: The maximum number of bytes within a chunk.
65
65
  chunk_size: The maximum number of items within a chunk.
66
66
  item_loader: The object responsible to generate the chunk intervals and load an item froma chunk.
67
- max_cache_size: The maximum cache size used by the reader when fetching the chunks.
67
+ max_cache_size: Cache budget. ``None`` uses 75% of free disk (see ``StreamingDataset``).
68
+ A float such as ``0.90`` is that fraction of currently free space.
68
69
  serializers: Provide your own serializers.
69
70
  writer_chunk_index: The index of the chunk to start from when writing.
70
71
  storage_options: Additional connection options for accessing storage services.
@@ -93,7 +94,7 @@ class Cache:
93
94
  self._cache_dir,
94
95
  subsampled_files=subsampled_files,
95
96
  region_of_interest=region_of_interest,
96
- max_cache_size=_convert_bytes_to_int(max_cache_size) if isinstance(max_cache_size, str) else max_cache_size,
97
+ max_cache_size=_resolve_max_cache_size(max_cache_size, self._cache_dir),
97
98
  remote_input_dir=input_dir.url,
98
99
  compression=compression,
99
100
  encryption=encryption,
@@ -13,7 +13,7 @@
13
13
 
14
14
  import shutil
15
15
  from abc import ABC, abstractmethod
16
- from typing import TypeVar
16
+ from typing import Any, TypeVar
17
17
 
18
18
  from litdata.constants import _PYTHON_GREATER_EQUAL_3_14, _ZSTD_AVAILABLE
19
19
  from litdata.debugger import CAT_DECOMPRESS, trace_span
@@ -54,27 +54,32 @@ class ZSTDCompressor(Compressor):
54
54
  raise ModuleNotFoundError(str(_ZSTD_AVAILABLE))
55
55
  self.level = level
56
56
  self.extension = "zstd"
57
+ self._zstd: Any | None = None
58
+
59
+ def __getstate__(self) -> dict[str, Any]:
60
+ state = self.__dict__.copy()
61
+ state["_zstd"] = None
62
+ return state
63
+
64
+ def _zstd_mod(self) -> Any:
65
+ if self._zstd is None:
66
+ if _PYTHON_GREATER_EQUAL_3_14:
67
+ from compression import zstd as mod
68
+ else:
69
+ import zstd as mod
70
+ self._zstd = mod
71
+ return self._zstd
57
72
 
58
73
  @property
59
74
  def name(self) -> str:
60
75
  return f"{self.extension}:{self.level}"
61
76
 
62
77
  def compress(self, data: bytes) -> bytes:
63
- if _PYTHON_GREATER_EQUAL_3_14:
64
- from compression import zstd
65
- else:
66
- import zstd
67
-
68
- return zstd.compress(data, self.level)
78
+ return self._zstd_mod().compress(data, self.level)
69
79
 
70
80
  def decompress(self, data: bytes) -> bytes:
71
- if _PYTHON_GREATER_EQUAL_3_14:
72
- from compression import zstd
73
- else:
74
- import zstd
75
-
76
81
  with trace_span("decompress", CAT_DECOMPRESS):
77
- return zstd.decompress(data)
82
+ return self._zstd_mod().decompress(data)
78
83
 
79
84
  def decompress_file(self, src: str, dst: str) -> None:
80
85
  if _PYTHON_GREATER_EQUAL_3_14:
@@ -14,9 +14,10 @@
14
14
  import contextlib
15
15
  import logging
16
16
  import os
17
+ import threading
17
18
  from collections import defaultdict
18
19
  from contextlib import suppress
19
- from time import sleep, time
20
+ from time import time
20
21
  from typing import Any, Optional
21
22
 
22
23
  from filelock import FileLock, Timeout
@@ -92,10 +93,15 @@ class ChunksConfig:
92
93
  self._length = self._intervals[-1][-1] if len(self._intervals) > 0 else 0
93
94
  self._downloader = None
94
95
 
96
+ # In-process "file is visible" Events, set after atomic publish (download or decompress).
97
+ self._file_published: dict[str, threading.Event] = {}
98
+ self._file_published_lock = threading.Lock()
99
+
95
100
  if remote_dir:
96
101
  self._downloader = get_downloader(
97
102
  remote_dir, cache_dir, self._chunks, self._storage_options, self._session_options
98
103
  )
104
+ self._downloader._on_file_published = self.notify_file_published
99
105
 
100
106
  self._compressor_name = self._config["compression"]
101
107
  self._compressor: Compressor | None = None
@@ -256,6 +262,28 @@ class ChunksConfig:
256
262
 
257
263
  return self._downloader.download_chunk_bytes_from_index(chunk_index, offset, length)
258
264
 
265
+ def notify_file_published(self, local_filepath: str) -> None:
266
+ """Mark ``local_filepath`` as atomically published and, if it is a readable chunk, signal ready."""
267
+ path = os.path.abspath(local_filepath)
268
+ with self._file_published_lock:
269
+ event = self._file_published.get(path)
270
+ if event is None:
271
+ event = threading.Event()
272
+ self._file_published[path] = event
273
+ event.set()
274
+
275
+ def wait_file_published(self, local_filepath: str, timeout: float) -> bool:
276
+ """Wait for an in-process publish Event; fall back to ``exists`` for other workers."""
277
+ path = os.path.abspath(local_filepath)
278
+ with self._file_published_lock:
279
+ event = self._file_published.get(path)
280
+ if event is None:
281
+ event = threading.Event()
282
+ self._file_published[path] = event
283
+ if event.wait(timeout=timeout):
284
+ return True
285
+ return os.path.exists(path)
286
+
259
287
  def _remove_compressed_source(self, local_chunkpath: str, target_local_chunkpath: str) -> None:
260
288
  """Drop the compressed download once the decompressed chunk is on disk."""
261
289
  if local_chunkpath == target_local_chunkpath:
@@ -273,13 +301,12 @@ class ChunksConfig:
273
301
  self._remove_compressed_source(local_chunkpath, target_local_chunkpath)
274
302
  return
275
303
 
276
- # Wait until either the decompressed target appears (another worker finished) or the
277
- # compressed source exists. Cloud downloaders publish the compressed path atomically, so
278
- # existence of that path means the download is complete — do NOT use chunk_size (item
279
- # count) as a byte threshold.
304
+ # Wait until the downloader publishes the compressed file, or another worker
305
+ # publishes the decompressed target. Cross-process downloads still fall back to exists.
280
306
  start_time = time()
281
- while not os.path.exists(local_chunkpath) and not os.path.exists(target_local_chunkpath):
282
- sleep(0.1)
307
+ while not os.path.exists(target_local_chunkpath) and not os.path.exists(local_chunkpath):
308
+ self.wait_file_published(local_chunkpath, 0.05)
309
+ self.wait_file_published(target_local_chunkpath, 0.0)
283
310
  if (time() - start_time) > _MAX_WAIT_TIME:
284
311
  raise ChunkWaitTimeoutError(local_chunkpath, time() - start_time)
285
312
 
@@ -307,6 +334,7 @@ class ChunksConfig:
307
334
  f"({os.stat(tmp_path).st_size} < {expected_bytes})."
308
335
  )
309
336
  os.replace(tmp_path, target_local_chunkpath)
337
+ self.notify_file_published(target_local_chunkpath)
310
338
  except Exception:
311
339
  with contextlib.suppress(FileNotFoundError, PermissionError):
312
340
  os.remove(tmp_path)
@@ -464,6 +492,20 @@ class ChunksConfig:
464
492
  session_options,
465
493
  )
466
494
 
495
+ def __getstate__(self) -> dict[str, Any]:
496
+ state = self.__dict__.copy()
497
+ # threading.Lock / Event are not picklable (DataLoader spawn / deepcopy).
498
+ state["_file_published"] = {}
499
+ state["_file_published_lock"] = None
500
+ return state
501
+
502
+ def __setstate__(self, state: dict[str, Any]) -> None:
503
+ self.__dict__.update(state)
504
+ self._file_published = {}
505
+ self._file_published_lock = threading.Lock()
506
+ if self._downloader is not None:
507
+ self._downloader._on_file_published = self.notify_file_published
508
+
467
509
  def __len__(self) -> int:
468
510
  return self._length
469
511