litdata 0.2.70__tar.gz → 0.2.72__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. {litdata-0.2.70/src/litdata.egg-info → litdata-0.2.72}/PKG-INFO +41 -6
  2. {litdata-0.2.70 → litdata-0.2.72}/README.md +39 -5
  3. {litdata-0.2.70 → litdata-0.2.72}/requirements.txt +1 -0
  4. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/__about__.py +1 -1
  5. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/async_prefetch.py +28 -4
  6. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/cache.py +5 -4
  7. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/client.py +190 -35
  8. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/compression.py +18 -13
  9. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/config.py +49 -7
  10. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/dataloader.py +184 -49
  11. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/dataset.py +38 -23
  12. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/downloader.py +37 -9
  13. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/item_loader.py +35 -14
  14. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/reader.py +59 -14
  15. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/serializers.py +5 -3
  16. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/shuffle.py +57 -29
  17. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/writer.py +52 -8
  18. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/_pytree.py +79 -19
  19. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/env.py +12 -0
  20. litdata-0.2.72/src/litdata/utilities/format.py +168 -0
  21. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/shuffle.py +64 -12
  22. {litdata-0.2.70 → litdata-0.2.72/src/litdata.egg-info}/PKG-INFO +41 -6
  23. {litdata-0.2.70 → litdata-0.2.72}/src/litdata.egg-info/requires.txt +1 -0
  24. litdata-0.2.70/src/litdata/utilities/format.py +0 -57
  25. {litdata-0.2.70 → litdata-0.2.72}/CONTRIBUTING.md +0 -0
  26. {litdata-0.2.70 → litdata-0.2.72}/LICENSE +0 -0
  27. {litdata-0.2.70 → litdata-0.2.72}/MANIFEST.in +0 -0
  28. {litdata-0.2.70 → litdata-0.2.72}/setup.cfg +0 -0
  29. {litdata-0.2.70 → litdata-0.2.72}/setup.py +0 -0
  30. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/__init__.py +0 -0
  31. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/__main__.py +0 -0
  32. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/cli/__init__.py +0 -0
  33. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/cli/commands.py +0 -0
  34. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/cli/handler/__init__.py +0 -0
  35. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/cli/handler/cache.py +0 -0
  36. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/cli/handler/optimize.py +0 -0
  37. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/cli/parser.py +0 -0
  38. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/constants.py +0 -0
  39. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/debugger.py +0 -0
  40. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/exceptions.py +0 -0
  41. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/helpers.py +0 -0
  42. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/imports.py +0 -0
  43. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/processing/__init__.py +0 -0
  44. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/processing/complete.py +0 -0
  45. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/processing/data_processor.py +0 -0
  46. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/processing/functions.py +0 -0
  47. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/processing/media_folder.py +0 -0
  48. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/processing/readers.py +0 -0
  49. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/processing/utilities.py +0 -0
  50. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/raw/__init__.py +0 -0
  51. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/raw/dataset.py +0 -0
  52. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/raw/indexer.py +0 -0
  53. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/raw/types.py +0 -0
  54. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/requirements.py +0 -0
  55. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/__init__.py +0 -0
  56. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/collate.py +0 -0
  57. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/combined.py +0 -0
  58. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/dataset_update.py +0 -0
  59. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/elastic.py +0 -0
  60. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/fs_provider.py +0 -0
  61. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/parallel.py +0 -0
  62. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/posix_fast.py +0 -0
  63. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/resolver.py +0 -0
  64. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/sampler.py +0 -0
  65. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/streaming/timing.py +0 -0
  66. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/types.py +0 -0
  67. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/__init__.py +0 -0
  68. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/base.py +0 -0
  69. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/breakpoint.py +0 -0
  70. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/broadcast.py +0 -0
  71. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/dataset_utilities.py +0 -0
  72. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/encryption.py +0 -0
  73. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/hf_dataset.py +0 -0
  74. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/keys_index.py +0 -0
  75. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/packing.py +0 -0
  76. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/parquet.py +0 -0
  77. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/subsample.py +0 -0
  78. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/torch_utils.py +0 -0
  79. {litdata-0.2.70 → litdata-0.2.72}/src/litdata/utilities/train_test_split.py +0 -0
  80. {litdata-0.2.70 → litdata-0.2.72}/src/litdata.egg-info/SOURCES.txt +0 -0
  81. {litdata-0.2.70 → litdata-0.2.72}/src/litdata.egg-info/dependency_links.txt +0 -0
  82. {litdata-0.2.70 → litdata-0.2.72}/src/litdata.egg-info/entry_points.txt +0 -0
  83. {litdata-0.2.70 → litdata-0.2.72}/src/litdata.egg-info/not-zip-safe +0 -0
  84. {litdata-0.2.70 → litdata-0.2.72}/src/litdata.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: litdata
3
- Version: 0.2.70
3
+ Version: 0.2.72
4
4
  Summary: The Deep Learning framework to train, deploy, and ship AI products Lightning fast.
5
5
  Home-page: https://github.com/Lightning-AI/litdata
6
6
  Download-URL: https://github.com/Lightning-AI/litdata
@@ -34,6 +34,7 @@ Requires-Dist: filelock
34
34
  Requires-Dist: numpy
35
35
  Requires-Dist: boto3
36
36
  Requires-Dist: requests
37
+ Requires-Dist: urllib3>=1.26
37
38
  Requires-Dist: tifffile
38
39
  Requires-Dist: obstore
39
40
  Provides-Extra: extras
@@ -1128,7 +1129,7 @@ Rough ImageNet order-of-magnitude on a Studio (not hard guarantees; right tuning
1128
1129
  | `drop_last` | `True` if distributed else `False` | Equal length across ranks |
1129
1130
  | `seed` | `42` | Shuffle / subsample RNG |
1130
1131
  | `serializers` | built-ins | Custom serialize/deserialize map |
1131
- | `max_cache_size` | `"100GB"` | Evict consumed chunks beyond this size |
1132
+ | `max_cache_size` | `None` | Evict consumed chunks beyond this size. Default: 75% of free disk, leaving ≥50GB when possible. Pin with `"100G"` / `"50GB"`, a fraction (`0.90`), or `MAX_CACHE_SIZE`. |
1132
1133
  | `max_pre_download` | `2` | Chunks each worker may prefetch (raise for throughput; watch disk / RAM) |
1133
1134
  | `subsample` | `1.0` | Fraction of data (`0.01`) or upsample (`2.5`) |
1134
1135
  | `encryption` | `None` | `FernetEncryption` / `RSAEncryption` / custom |
@@ -1149,7 +1150,8 @@ On **Vast / NFS / local disk**, POSIX-fast is on by default (`LITDATA_POSIX_FAST
1149
1150
  | All usual `torch.utils.data.DataLoader` kwargs | `batch_size`, `num_workers`, `collate_fn`, `pin_memory`, … |
1150
1151
  | `shuffle` / `drop_last` | Forwarded to the streaming dataset |
1151
1152
  | `profile_batches` | `int` / `True` / `False` — viztracer worker trace (see [Profile data loading](#profile-loading)) |
1152
- | `profile_skip_batches` / `profile_dir` | Warm-up skip count; output dir for `result.json` |
1153
+ | `profile_cprofile` | `True` — stdlib cProfile of the main process + worker 0 (see [Profile data loading](#profile-loading)) |
1154
+ | `profile_skip_batches` / `profile_dir` | Warm-up skip count; output dir for viztracer / cProfile files |
1153
1155
  | `multiprocessing_context` | Use **`"spawn"`** (or `"forkserver"`) with `ParquetLoader` + `num_workers>0` on Linux |
1154
1156
 
1155
1157
  Prefer `StreamingDataLoader` over a plain PyTorch `DataLoader` for optimized / combined / parallel datasets (resume + correct batch metadata).
@@ -2115,7 +2117,32 @@ Only **worker 0** is instrumented. When an `int` is used, the tracer wraps `fetc
2115
2117
 
2116
2118
  - Delete or change `profile_dir` between runs — LitData removes an existing `result.json` before starting.
2117
2119
  - Pair with a wiped chunk cache if you care about **cold** epoch behavior (`litdata cache clear`).
2118
- - For deeper LitData internals (download / read / delete timeline), use `enable_tracer()` + [Litracer](https://github.com/Lightning-AI/litracer) instead — see [Debug & Profile LitData](#debug-profile). That path is complementary: viztracer = DataLoader worker CPU timeline; Litracer = LitData pipeline events.
2120
+ ### cProfile (main + worker 0)
2121
+
2122
+ Stdlib statistical profile of **the parent process** (queue wait, unpickle, collate handoff) and **worker 0** (fetch, decode, transforms). No extra package.
2123
+
2124
+ ```python
2125
+ loader = StreamingDataLoader(
2126
+ dataset,
2127
+ batch_size=64,
2128
+ num_workers=4,
2129
+ profile_cprofile=True,
2130
+ profile_dir="./profiles",
2131
+ )
2132
+
2133
+ for batch in loader:
2134
+ train_step(batch)
2135
+ # writes profiles/cprofile_main.prof + cprofile_worker0.prof (and .txt summaries)
2136
+ ```
2137
+
2138
+ ```bash
2139
+ python -m pstats profiles/cprofile_worker0.prof
2140
+ # then: sort tottime stats 30
2141
+ ```
2142
+
2143
+ Do not set `profile_cprofile` and `profile_batches` together — both install a `sys.setprofile` hook. With `num_workers=0` only the main file is written. The parent profiler starts **after** workers spawn so fork does not inherit an active cProfile.
2144
+
2145
+ - For deeper LitData internals (download / read / delete timeline), use `enable_tracer()` + [Litracer](https://github.com/Lightning-AI/litracer) instead — see [Debug & Profile LitData](#debug-profile). That path is complementary: viztracer = DataLoader worker CPU timeline; cProfile = function self/cum time; Litracer = LitData pipeline events.
2119
2146
 
2120
2147
  </details>
2121
2148
 
@@ -2165,7 +2192,9 @@ outputs = optimize(
2165
2192
 
2166
2193
  Control how much disk the local chunk cache may use. Downloaded chunks are deleted after use once the cache exceeds the limit.
2167
2194
 
2168
- Default `max_cache_size` is **`100GB`**. Peak disk in flight is roughly:
2195
+ Default `max_cache_size` is **`None`**: LitData uses **75% of currently free disk** and still leaves **≥50GB** free when the volume is large enough for checkpoints. Pass `"100G"` / `"50GB"` for a fixed budget, or a float (`0.90`) for that fraction of currently free space. `MAX_CACHE_SIZE` overrides the constructor (size or fraction).
2196
+
2197
+ Peak disk in flight is roughly:
2169
2198
 
2170
2199
  ```
2171
2200
  num_workers × max_pre_download × mean_chunk_size
@@ -2205,7 +2234,9 @@ for batch in StreamingDataLoader(dataset, batch_size=64, num_workers=8):
2205
2234
  | `LITDATA_ASYNC_CHUNK_PREFETCH=1` | Force on |
2206
2235
  | `LITDATA_ASYNC_CHUNK_PREFETCH=0` | Force off |
2207
2236
 
2208
- When async is on, LitData raises `max_pre_download` to at least **4** so `asyncio.gather` has enough in-flight downloads (override with `LITDATA_ASYNC_MIN_PRE_DOWNLOAD`; set `0` to disable the floor). Peak disk ≈ `num_workers × max_pre_download × chunk_size` — size `max_cache_size` accordingly.
2237
+ When async is on, LitData raises `max_pre_download` to at least **4** so `asyncio.gather` has enough in-flight downloads (override with `LITDATA_ASYNC_MIN_PRE_DOWNLOAD`; set `0` to disable the floor). Concurrent GETs are capped by `LITDATA_ASYNC_DOWNLOAD_CONCURRENCY` (default **8**). The prepare thread drains the prefetch queue up to that gather width whenever a cache slot is free. Peak disk ≈ `num_workers × max_pre_download × chunk_size` — size `max_cache_size` accordingly.
2238
+
2239
+ The reader does **not** poll the cache directory for each chunk. After a download finishes, the downloader atomically `os.replace`s the temp file and sets an in-process `Event`. Compressed chunks set that Event only after decompress publishes the readable `.bin`. Other DataLoader workers still fall back to a short filesystem check (Events are per process).
2209
2240
 
2210
2241
  ```bash
2211
2242
  # Debugging download/delete races — force synchronous downloads
@@ -2213,6 +2244,9 @@ export LITDATA_ASYNC_CHUNK_PREFETCH=0
2213
2244
 
2214
2245
  # Keep max_pre_download=2 even with async enabled
2215
2246
  export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
2247
+
2248
+ # Cap overlapping remote GETs (default 8)
2249
+ export LITDATA_ASYNC_DOWNLOAD_CONCURRENCY=4
2216
2250
  ```
2217
2251
 
2218
2252
  ### Common environment variables
@@ -2222,6 +2256,7 @@ export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
2222
2256
  | `LITDATA_CACHE_DIR` | `~/.lightning/chunks` | Default chunk cache directory |
2223
2257
  | `LITDATA_ASYNC_CHUNK_PREFETCH` | on for remote | `0`/`1` force async chunk download overlap |
2224
2258
  | `LITDATA_ASYNC_MIN_PRE_DOWNLOAD` | `4` | Floor for `max_pre_download` when async is on (`0` = no floor) |
2259
+ | `LITDATA_ASYNC_DOWNLOAD_CONCURRENCY` | `8` | Max in-flight chunk GETs per gather |
2225
2260
  | `LITDATA_OBSTORE_STREAM_MIN_CHUNK_MIB` | `8` | S3 obstore stream chunk size (MiB) |
2226
2261
  | `MAX_WAIT_TIME` | `120` | Seconds to wait for a chunk before error |
2227
2262
  | `FORCE_DOWNLOAD_TIME` | `30` | Seconds before force re-download of a missing chunk |
@@ -1064,7 +1064,7 @@ Rough ImageNet order-of-magnitude on a Studio (not hard guarantees; right tuning
1064
1064
  | `drop_last` | `True` if distributed else `False` | Equal length across ranks |
1065
1065
  | `seed` | `42` | Shuffle / subsample RNG |
1066
1066
  | `serializers` | built-ins | Custom serialize/deserialize map |
1067
- | `max_cache_size` | `"100GB"` | Evict consumed chunks beyond this size |
1067
+ | `max_cache_size` | `None` | Evict consumed chunks beyond this size. Default: 75% of free disk, leaving ≥50GB when possible. Pin with `"100G"` / `"50GB"`, a fraction (`0.90`), or `MAX_CACHE_SIZE`. |
1068
1068
  | `max_pre_download` | `2` | Chunks each worker may prefetch (raise for throughput; watch disk / RAM) |
1069
1069
  | `subsample` | `1.0` | Fraction of data (`0.01`) or upsample (`2.5`) |
1070
1070
  | `encryption` | `None` | `FernetEncryption` / `RSAEncryption` / custom |
@@ -1085,7 +1085,8 @@ On **Vast / NFS / local disk**, POSIX-fast is on by default (`LITDATA_POSIX_FAST
1085
1085
  | All usual `torch.utils.data.DataLoader` kwargs | `batch_size`, `num_workers`, `collate_fn`, `pin_memory`, … |
1086
1086
  | `shuffle` / `drop_last` | Forwarded to the streaming dataset |
1087
1087
  | `profile_batches` | `int` / `True` / `False` — viztracer worker trace (see [Profile data loading](#profile-loading)) |
1088
- | `profile_skip_batches` / `profile_dir` | Warm-up skip count; output dir for `result.json` |
1088
+ | `profile_cprofile` | `True` — stdlib cProfile of the main process + worker 0 (see [Profile data loading](#profile-loading)) |
1089
+ | `profile_skip_batches` / `profile_dir` | Warm-up skip count; output dir for viztracer / cProfile files |
1089
1090
  | `multiprocessing_context` | Use **`"spawn"`** (or `"forkserver"`) with `ParquetLoader` + `num_workers>0` on Linux |
1090
1091
 
1091
1092
  Prefer `StreamingDataLoader` over a plain PyTorch `DataLoader` for optimized / combined / parallel datasets (resume + correct batch metadata).
@@ -2051,7 +2052,32 @@ Only **worker 0** is instrumented. When an `int` is used, the tracer wraps `fetc
2051
2052
 
2052
2053
  - Delete or change `profile_dir` between runs — LitData removes an existing `result.json` before starting.
2053
2054
  - Pair with a wiped chunk cache if you care about **cold** epoch behavior (`litdata cache clear`).
2054
- - For deeper LitData internals (download / read / delete timeline), use `enable_tracer()` + [Litracer](https://github.com/Lightning-AI/litracer) instead — see [Debug & Profile LitData](#debug-profile). That path is complementary: viztracer = DataLoader worker CPU timeline; Litracer = LitData pipeline events.
2055
+ ### cProfile (main + worker 0)
2056
+
2057
+ Stdlib statistical profile of **the parent process** (queue wait, unpickle, collate handoff) and **worker 0** (fetch, decode, transforms). No extra package.
2058
+
2059
+ ```python
2060
+ loader = StreamingDataLoader(
2061
+ dataset,
2062
+ batch_size=64,
2063
+ num_workers=4,
2064
+ profile_cprofile=True,
2065
+ profile_dir="./profiles",
2066
+ )
2067
+
2068
+ for batch in loader:
2069
+ train_step(batch)
2070
+ # writes profiles/cprofile_main.prof + cprofile_worker0.prof (and .txt summaries)
2071
+ ```
2072
+
2073
+ ```bash
2074
+ python -m pstats profiles/cprofile_worker0.prof
2075
+ # then: sort tottime stats 30
2076
+ ```
2077
+
2078
+ Do not set `profile_cprofile` and `profile_batches` together — both install a `sys.setprofile` hook. With `num_workers=0` only the main file is written. The parent profiler starts **after** workers spawn so fork does not inherit an active cProfile.
2079
+
2080
+ - For deeper LitData internals (download / read / delete timeline), use `enable_tracer()` + [Litracer](https://github.com/Lightning-AI/litracer) instead — see [Debug & Profile LitData](#debug-profile). That path is complementary: viztracer = DataLoader worker CPU timeline; cProfile = function self/cum time; Litracer = LitData pipeline events.
2055
2081
 
2056
2082
  </details>
2057
2083
 
@@ -2101,7 +2127,9 @@ outputs = optimize(
2101
2127
 
2102
2128
  Control how much disk the local chunk cache may use. Downloaded chunks are deleted after use once the cache exceeds the limit.
2103
2129
 
2104
- Default `max_cache_size` is **`100GB`**. Peak disk in flight is roughly:
2130
+ Default `max_cache_size` is **`None`**: LitData uses **75% of currently free disk** and still leaves **≥50GB** free when the volume is large enough for checkpoints. Pass `"100G"` / `"50GB"` for a fixed budget, or a float (`0.90`) for that fraction of currently free space. `MAX_CACHE_SIZE` overrides the constructor (size or fraction).
2131
+
2132
+ Peak disk in flight is roughly:
2105
2133
 
2106
2134
  ```
2107
2135
  num_workers × max_pre_download × mean_chunk_size
@@ -2141,7 +2169,9 @@ for batch in StreamingDataLoader(dataset, batch_size=64, num_workers=8):
2141
2169
  | `LITDATA_ASYNC_CHUNK_PREFETCH=1` | Force on |
2142
2170
  | `LITDATA_ASYNC_CHUNK_PREFETCH=0` | Force off |
2143
2171
 
2144
- When async is on, LitData raises `max_pre_download` to at least **4** so `asyncio.gather` has enough in-flight downloads (override with `LITDATA_ASYNC_MIN_PRE_DOWNLOAD`; set `0` to disable the floor). Peak disk ≈ `num_workers × max_pre_download × chunk_size` — size `max_cache_size` accordingly.
2172
+ When async is on, LitData raises `max_pre_download` to at least **4** so `asyncio.gather` has enough in-flight downloads (override with `LITDATA_ASYNC_MIN_PRE_DOWNLOAD`; set `0` to disable the floor). Concurrent GETs are capped by `LITDATA_ASYNC_DOWNLOAD_CONCURRENCY` (default **8**). The prepare thread drains the prefetch queue up to that gather width whenever a cache slot is free. Peak disk ≈ `num_workers × max_pre_download × chunk_size` — size `max_cache_size` accordingly.
2173
+
2174
+ The reader does **not** poll the cache directory for each chunk. After a download finishes, the downloader atomically `os.replace`s the temp file and sets an in-process `Event`. Compressed chunks set that Event only after decompress publishes the readable `.bin`. Other DataLoader workers still fall back to a short filesystem check (Events are per process).
2145
2175
 
2146
2176
  ```bash
2147
2177
  # Debugging download/delete races — force synchronous downloads
@@ -2149,6 +2179,9 @@ export LITDATA_ASYNC_CHUNK_PREFETCH=0
2149
2179
 
2150
2180
  # Keep max_pre_download=2 even with async enabled
2151
2181
  export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
2182
+
2183
+ # Cap overlapping remote GETs (default 8)
2184
+ export LITDATA_ASYNC_DOWNLOAD_CONCURRENCY=4
2152
2185
  ```
2153
2186
 
2154
2187
  ### Common environment variables
@@ -2158,6 +2191,7 @@ export LITDATA_ASYNC_MIN_PRE_DOWNLOAD=0
2158
2191
  | `LITDATA_CACHE_DIR` | `~/.lightning/chunks` | Default chunk cache directory |
2159
2192
  | `LITDATA_ASYNC_CHUNK_PREFETCH` | on for remote | `0`/`1` force async chunk download overlap |
2160
2193
  | `LITDATA_ASYNC_MIN_PRE_DOWNLOAD` | `4` | Floor for `max_pre_download` when async is on (`0` = no floor) |
2194
+ | `LITDATA_ASYNC_DOWNLOAD_CONCURRENCY` | `8` | Max in-flight chunk GETs per gather |
2161
2195
  | `LITDATA_OBSTORE_STREAM_MIN_CHUNK_MIB` | `8` | S3 obstore stream chunk size (MiB) |
2162
2196
  | `MAX_WAIT_TIME` | `120` | Seconds to wait for a chunk before error |
2163
2197
  | `FORCE_DOWNLOAD_TIME` | `30` | Seconds before force re-download of a missing chunk |
@@ -5,5 +5,6 @@ filelock
5
5
  numpy
6
6
  boto3
7
7
  requests
8
+ urllib3 >=1.26 # Retry(allowed_methods=...) in streaming/client.py
8
9
  tifffile
9
10
  obstore
@@ -14,7 +14,7 @@
14
14
 
15
15
  import time
16
16
 
17
- __version__ = "0.2.70"
17
+ __version__ = "0.2.72"
18
18
  __author__ = "Lightning AI et al."
19
19
  __author_email__ = "pytorch@lightning.ai"
20
20
  __license__ = "Apache-2.0"
@@ -54,6 +54,8 @@ _THREAD_LOOPS = threading.local()
54
54
  # Empirically, async gather on real S3 is bottlenecked when max_pre_download==2
55
55
  # (only 1–2 in-flight). Floor to 4 when the feature is enabled unless overridden.
56
56
  _DEFAULT_ASYNC_MIN_PRE_DOWNLOAD = 4
57
+ # Cap in-flight GETs so a large drain batch does not open one connection per slot.
58
+ _DEFAULT_ASYNC_DOWNLOAD_CONCURRENCY = 8
57
59
 
58
60
 
59
61
  def async_chunk_prefetch_enabled(remote_dir: str | None = None) -> bool:
@@ -81,6 +83,17 @@ def async_prefetch_min_pre_download() -> int:
81
83
  return max(0, int(raw))
82
84
 
83
85
 
86
+ def async_download_concurrency(n_chunks: int) -> int:
87
+ """How many remote chunk GETs to run at once inside ``asyncio.gather``.
88
+
89
+ Override with ``LITDATA_ASYNC_DOWNLOAD_CONCURRENCY`` (default 8). Always at
90
+ least 1 and at most ``n_chunks``.
91
+ """
92
+ raw = os.getenv("LITDATA_ASYNC_DOWNLOAD_CONCURRENCY")
93
+ limit = _DEFAULT_ASYNC_DOWNLOAD_CONCURRENCY if raw is None else max(1, int(raw))
94
+ return max(1, min(limit, max(1, n_chunks)))
95
+
96
+
84
97
  def apply_async_pre_download_floor(max_pre_download: int, remote_dir: str | None = None) -> int:
85
98
  """Raise ``max_pre_download`` when async prefetch needs gather width."""
86
99
  if not async_chunk_prefetch_enabled(remote_dir):
@@ -133,6 +146,7 @@ def _remote_join(remote_dir: str, filename: str) -> str:
133
146
  async def _adownload_file_to_path(downloader: Downloader, remote_filepath: str, local_filepath: str) -> None:
134
147
  """Fetch ``remote_filepath`` asynchronously and publish atomically."""
135
148
  if os.path.exists(local_filepath):
149
+ downloader._notify_published(local_filepath)
136
150
  return
137
151
  # Prefer streaming-to-disk when the backend overrides adownload_file.
138
152
  if type(downloader).adownload_file is not Downloader.adownload_file:
@@ -149,7 +163,7 @@ async def _adownload_file_to_path(downloader: Downloader, remote_filepath: str,
149
163
  os.makedirs(os.path.dirname(local_filepath) or ".", exist_ok=True)
150
164
  with open(tmp_path, "wb") as f:
151
165
  f.write(data)
152
- downloader._atomic_replace(tmp_path, local_filepath)
166
+ downloader._publish_file(tmp_path, local_filepath)
153
167
  except Exception:
154
168
  with contextlib.suppress(FileNotFoundError, PermissionError):
155
169
  os.remove(tmp_path)
@@ -172,7 +186,7 @@ async def _adownload_chunk_index(config: ChunksConfig, chunk_index: int) -> None
172
186
  )
173
187
 
174
188
  if os.path.exists(local_chunkpath):
175
- config.try_decompress(local_chunkpath)
189
+ await asyncio.to_thread(config.try_decompress, local_chunkpath)
176
190
  if lazily_ref_counted:
177
191
  downloader._increment_local_lock(lock_path, chunk_index)
178
192
  return
@@ -186,7 +200,7 @@ async def _adownload_chunk_index(config: ChunksConfig, chunk_index: int) -> None
186
200
  # Overlap blocking SDK calls across threads when native async is unavailable.
187
201
  await asyncio.to_thread(downloader.download_chunk_from_index, chunk_index)
188
202
 
189
- config.try_decompress(local_chunkpath)
203
+ await asyncio.to_thread(config.try_decompress, local_chunkpath)
190
204
 
191
205
 
192
206
  async def adownload_chunk_indexes(config: ChunksConfig, chunk_indexes: list[int]) -> None:
@@ -196,7 +210,17 @@ async def adownload_chunk_indexes(config: ChunksConfig, chunk_indexes: list[int]
196
210
  if len(chunk_indexes) == 1:
197
211
  await _adownload_chunk_index(config, chunk_indexes[0])
198
212
  return
199
- await asyncio.gather(*[_adownload_chunk_index(config, idx) for idx in chunk_indexes])
213
+ limit = async_download_concurrency(len(chunk_indexes))
214
+ if limit >= len(chunk_indexes):
215
+ await asyncio.gather(*[_adownload_chunk_index(config, idx) for idx in chunk_indexes])
216
+ return
217
+ sem = asyncio.Semaphore(limit)
218
+
219
+ async def _one(idx: int) -> None:
220
+ async with sem:
221
+ await _adownload_chunk_index(config, idx)
222
+
223
+ await asyncio.gather(*[_one(idx) for idx in chunk_indexes])
200
224
 
201
225
 
202
226
  def _thread_event_loop() -> asyncio.AbstractEventLoop:
@@ -27,7 +27,7 @@ from litdata.streaming.serializers import Serializer
27
27
  from litdata.streaming.writer import BinaryWriter
28
28
  from litdata.utilities.encryption import Encryption
29
29
  from litdata.utilities.env import _DistributedEnv, _WorkerEnv
30
- from litdata.utilities.format import _convert_bytes_to_int
30
+ from litdata.utilities.format import _resolve_max_cache_size
31
31
 
32
32
  logger = logging.Logger(__name__)
33
33
 
@@ -43,7 +43,7 @@ class Cache:
43
43
  chunk_size: int | None = None,
44
44
  chunk_bytes: int | str | None = None,
45
45
  item_loader: BaseItemLoader | None = None,
46
- max_cache_size: int | str = "100GB",
46
+ max_cache_size: int | float | str | None = None,
47
47
  serializers: dict[str, Serializer] | None = None,
48
48
  writer_chunk_index: int | None = None,
49
49
  storage_options: dict | None = {},
@@ -64,7 +64,8 @@ class Cache:
64
64
  chunk_bytes: The maximum number of bytes within a chunk.
65
65
  chunk_size: The maximum number of items within a chunk.
66
66
  item_loader: The object responsible to generate the chunk intervals and load an item froma chunk.
67
- max_cache_size: The maximum cache size used by the reader when fetching the chunks.
67
+ max_cache_size: Cache budget. ``None`` uses 75% of free disk (see ``StreamingDataset``).
68
+ A float such as ``0.90`` is that fraction of currently free space.
68
69
  serializers: Provide your own serializers.
69
70
  writer_chunk_index: The index of the chunk to start from when writing.
70
71
  storage_options: Additional connection options for accessing storage services.
@@ -93,7 +94,7 @@ class Cache:
93
94
  self._cache_dir,
94
95
  subsampled_files=subsampled_files,
95
96
  region_of_interest=region_of_interest,
96
- max_cache_size=_convert_bytes_to_int(max_cache_size) if isinstance(max_cache_size, str) else max_cache_size,
97
+ max_cache_size=_resolve_max_cache_size(max_cache_size, self._cache_dir),
97
98
  remote_input_dir=input_dir.url,
98
99
  compression=compression,
99
100
  encryption=encryption,