litdata 0.2.68__tar.gz → 0.2.69__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. {litdata-0.2.68/src/litdata.egg-info → litdata-0.2.69}/PKG-INFO +32 -20
  2. {litdata-0.2.68 → litdata-0.2.69}/README.md +31 -19
  3. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/__about__.py +1 -1
  4. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/processing/data_processor.py +961 -244
  5. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/processing/functions.py +24 -18
  6. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/streaming/async_prefetch.py +9 -0
  7. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/streaming/dataloader.py +46 -3
  8. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/streaming/dataset.py +245 -37
  9. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/streaming/downloader.py +51 -0
  10. litdata-0.2.69/src/litdata/streaming/elastic.py +243 -0
  11. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/streaming/resolver.py +1 -1
  12. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/streaming/writer.py +23 -11
  13. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/utilities/keys_index.py +3 -1
  14. {litdata-0.2.68 → litdata-0.2.69/src/litdata.egg-info}/PKG-INFO +32 -20
  15. {litdata-0.2.68 → litdata-0.2.69}/src/litdata.egg-info/SOURCES.txt +1 -0
  16. {litdata-0.2.68 → litdata-0.2.69}/CONTRIBUTING.md +0 -0
  17. {litdata-0.2.68 → litdata-0.2.69}/LICENSE +0 -0
  18. {litdata-0.2.68 → litdata-0.2.69}/MANIFEST.in +0 -0
  19. {litdata-0.2.68 → litdata-0.2.69}/requirements.txt +0 -0
  20. {litdata-0.2.68 → litdata-0.2.69}/setup.cfg +0 -0
  21. {litdata-0.2.68 → litdata-0.2.69}/setup.py +0 -0
  22. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/__init__.py +0 -0
  23. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/__main__.py +0 -0
  24. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/cli/__init__.py +0 -0
  25. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/cli/commands.py +0 -0
  26. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/cli/handler/__init__.py +0 -0
  27. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/cli/handler/cache.py +0 -0
  28. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/cli/handler/optimize.py +0 -0
  29. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/cli/parser.py +0 -0
  30. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/constants.py +0 -0
  31. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/debugger.py +0 -0
  32. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/exceptions.py +0 -0
  33. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/helpers.py +0 -0
  34. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/imports.py +0 -0
  35. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/processing/__init__.py +0 -0
  36. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/processing/readers.py +0 -0
  37. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/processing/utilities.py +0 -0
  38. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/raw/__init__.py +0 -0
  39. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/raw/dataset.py +0 -0
  40. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/raw/indexer.py +0 -0
  41. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/raw/types.py +0 -0
  42. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/requirements.py +0 -0
  43. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/streaming/__init__.py +0 -0
  44. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/streaming/cache.py +0 -0
  45. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/streaming/client.py +0 -0
  46. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/streaming/combined.py +0 -0
  47. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/streaming/compression.py +0 -0
  48. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/streaming/config.py +0 -0
  49. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/streaming/dataset_update.py +0 -0
  50. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/streaming/fs_provider.py +0 -0
  51. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/streaming/item_loader.py +0 -0
  52. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/streaming/parallel.py +0 -0
  53. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/streaming/posix_fast.py +0 -0
  54. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/streaming/reader.py +0 -0
  55. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/streaming/sampler.py +0 -0
  56. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/streaming/serializers.py +0 -0
  57. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/streaming/shuffle.py +0 -0
  58. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/streaming/timing.py +0 -0
  59. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/utilities/__init__.py +0 -0
  60. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/utilities/_pytree.py +0 -0
  61. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/utilities/base.py +0 -0
  62. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/utilities/breakpoint.py +0 -0
  63. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/utilities/broadcast.py +0 -0
  64. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/utilities/dataset_utilities.py +0 -0
  65. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/utilities/encryption.py +0 -0
  66. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/utilities/env.py +0 -0
  67. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/utilities/format.py +0 -0
  68. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/utilities/hf_dataset.py +0 -0
  69. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/utilities/packing.py +0 -0
  70. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/utilities/parquet.py +0 -0
  71. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/utilities/shuffle.py +0 -0
  72. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/utilities/subsample.py +0 -0
  73. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/utilities/torch_utils.py +0 -0
  74. {litdata-0.2.68 → litdata-0.2.69}/src/litdata/utilities/train_test_split.py +0 -0
  75. {litdata-0.2.68 → litdata-0.2.69}/src/litdata.egg-info/dependency_links.txt +0 -0
  76. {litdata-0.2.68 → litdata-0.2.69}/src/litdata.egg-info/entry_points.txt +0 -0
  77. {litdata-0.2.68 → litdata-0.2.69}/src/litdata.egg-info/not-zip-safe +0 -0
  78. {litdata-0.2.68 → litdata-0.2.69}/src/litdata.egg-info/requires.txt +0 -0
  79. {litdata-0.2.68 → litdata-0.2.69}/src/litdata.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: litdata
3
- Version: 0.2.68
3
+ Version: 0.2.69
4
4
  Summary: The Deep Learning framework to train, deploy, and ship AI products Lightning fast.
5
5
  Home-page: https://github.com/Lightning-AI/litdata
6
6
  Download-URL: https://github.com/Lightning-AI/litdata
@@ -989,6 +989,12 @@ for batch_idx, batch in enumerate(dataloader):
989
989
  torch.save(dataloader.state_dict(), "dataloader_state.pt")
990
990
  ```
991
991
 
992
+ Same `seed` and `shuffle` are required. For a **`StreamingDataset`**, **`num_workers` and `world_size` may change**: LitData drops a global `sample_in_epoch` prefix and restripes the rest (never duplicates remaining IDs). For a matching loss curve keep **global batch size** (`world_size * batch_size`) constant and DDP ranks in lockstep. `num_canonical_nodes` (default: first-run `world_size`) is frozen in the checkpoint. POSIX `WindowShuffle` resumes whole remaining chunks. **`CombinedStreamingDataset` and `ParallelStreamingDataset` resume only with the same `world_size`, `num_workers`, and `batch_size`.**
993
+
994
+ ```python
995
+ dataset = StreamingDataset("s3://my-bucket/my-data", shuffle=True, num_canonical_nodes=8)
996
+ ```
997
+
992
998
  </details>
993
999
 
994
1000
 
@@ -996,11 +1002,9 @@ for batch_idx, batch in enumerate(dataloader):
996
1002
  <summary> ✅ Use shared queue for Optimizing <a id="shared-queue" href="#shared-queue">🔗</a> </summary>
997
1003
  &nbsp;
998
1004
 
999
- If you are using multiple workers to optimize your dataset, you can use a shared queue to speed up the process.
1005
+ `optimize` / `map` default to a **shared per-node queue** (`keep_data_ordered=False`). Work is packed per node, then every worker on that node pulls the next item, so a slow worker does not leave others idle. Set `keep_data_ordered=True` to keep a static per-worker slice (required for `use_checkpoint` and `align_chunking`).
1000
1006
 
1001
- This is especially useful when optimizing large datasets in parallel, where some workers may be slower than others.
1002
-
1003
- It can also improve fault tolerance when workers fail due to out-of-memory (OOM) errors.
1007
+ Local `output_dir` writes chunks in place. Remote inputs and outputs use the streaming downloader (`adownload_file` / `aupload_file`, obstore when available).
1004
1008
 
1005
1009
  ```python
1006
1010
  import numpy as np
@@ -1023,28 +1027,36 @@ if __name__ == "__main__":
1023
1027
  output_dir="fast_data", # optimized data is stored here
1024
1028
  num_workers=4, # The number of workers on the same machine
1025
1029
  chunk_bytes="64MB" , # size of each chunk
1026
- keep_data_ordered=False, # Use a shared queue to speed up the process
1030
+ keep_data_ordered=False, # default: shared queue (set True to keep input order)
1027
1031
  )
1028
1032
  ```
1029
1033
 
1030
1034
 
1031
- ### Performance Difference between using a shared queue and not using it:
1035
+ ### Shared queue vs ordered (skewed local files)
1032
1036
 
1033
- **Note**: The following benchmarks were collected using the ImageNet dataset on an A10G machine with 16 workers.
1037
+ `scripts/bench/bench_node_queue.py --files 4000 --workers 8` (first 500 files are 1 MiB). On `main`, unordered optimize sat on a 200s empty-queue timeout after work finished.
1034
1038
 
1035
- | Configuration | Optimize Time (sec) | Stream 1 (img/sec) | Stream 2 (img/sec) |
1036
- |------------------|---------------------|---------------------|---------------------|
1037
- | shared_queue (`keep_data_ordered=False`) | 1281 | 5392 | 5732 |
1038
- | no shared_queue (`keep_data_ordered=True (default)`) | 1187 | 5257 | 5746 |
1039
+ | Tree | Mode | Time | Throughput |
1040
+ |------|------|-----:|-----------:|
1041
+ | `main` (old default) | `keep_data_ordered=True` | 23.7s | 169 files/s |
1042
+ | `main` | `keep_data_ordered=False` | 223.6s | 18 files/s |
1043
+ | this tree | `keep_data_ordered=True` | 22.9s | 175 files/s |
1044
+ | this tree (**new default**) | `keep_data_ordered=False` | **18.8s** | 213 files/s |
1039
1045
 
1040
- 📌 Note: The **shared_queue** option impacts optimization time, not streaming speed.
1041
- > While the streaming numbers may appear slightly different, this variation is incidental and not caused by shared_queue.
1042
- >
1043
- > Streaming happens after optimization and does not involve inter-process communication where shared_queue plays a role.
1046
+ Shared-queue **before → after: ~12×**. New default vs old ordered default: **1.22×**.
1047
+
1048
+ ### Local / remote input × output
1049
+
1050
+ `python scripts/bench/bench_node_queue.py --files 200 --workers 4 --io-matrix` (first 50 files are 1 MiB).
1044
1051
 
1045
- - 📄 Using a shared queue helps balance the load across workers, though it may slightly increase optimization time due to the overhead of pickling items sent between processes.
1052
+ | Topology | Ordered | Shared | Speedup |
1053
+ |----------|--------:|-------:|--------:|
1054
+ | local → local | 6.55s | **2.96s** | 2.21× |
1055
+ | remote → local | 7.44s | **3.48s** | 2.14× |
1056
+ | local → remote | 10.98s | **6.51s** | 1.69× |
1057
+ | remote → remote | 11.48s | **8.55s** | 1.34× |
1046
1058
 
1047
- - ⚡ However, it can significantly improve optimizing performance — especially when some workers are slower than others.
1059
+ Shared queue balances uneven workers. It does not change later `StreamingDataset` throughput.
1048
1060
 
1049
1061
  </details>
1050
1062
 
@@ -2344,7 +2356,7 @@ if __name__ == "__main__":
2344
2356
  | `start_method` | spawn† | Multiprocessing start method (†spawn unless IPython) |
2345
2357
  | `optimize_dns` | `None` | Optimized DNS (Studio / cloud) |
2346
2358
  | `storage_options` | `{}` | Cloud credentials / endpoints |
2347
- | `keep_data_ordered` | `True` | `False` = shared work queue (better for uneven/slow workers) |
2359
+ | `keep_data_ordered` | `False` | Shared work queue (faster for uneven workers). `True` keeps a static per-worker slice. Forced `True` with `use_checkpoint` / `align_chunking`. |
2348
2360
 
2349
2361
  </details>
2350
2362
 
@@ -2380,7 +2392,7 @@ Full knob list for `litdata.optimize` (see Quick start for the minimal recipe).
2380
2392
  | `start_method` | spawn† | Multiprocessing start method |
2381
2393
  | `optimize_dns` | `None` | Optimized DNS |
2382
2394
  | `storage_options` | `{}` | Cloud credentials / endpoints |
2383
- | `keep_data_ordered` | `True` | `False` = shared queue among workers |
2395
+ | `keep_data_ordered` | `False` | Shared queue among workers. `True` keeps input order. Forced `True` with `use_checkpoint` / `align_chunking`. |
2384
2396
  | `verbose` | `True` | Progress logging |
2385
2397
 
2386
2398
  Related features: [shared queue](#shared-queue), [queue input](#queue-input), [append/overwrite](#modify-datasets), [compression](#compression), [TokensLoader / LLM](#llm-training), [filter](#filter-data).
@@ -925,6 +925,12 @@ for batch_idx, batch in enumerate(dataloader):
925
925
  torch.save(dataloader.state_dict(), "dataloader_state.pt")
926
926
  ```
927
927
 
928
+ Same `seed` and `shuffle` are required. For a **`StreamingDataset`**, **`num_workers` and `world_size` may change**: LitData drops a global `sample_in_epoch` prefix and restripes the rest (never duplicates remaining IDs). For a matching loss curve keep **global batch size** (`world_size * batch_size`) constant and DDP ranks in lockstep. `num_canonical_nodes` (default: first-run `world_size`) is frozen in the checkpoint. POSIX `WindowShuffle` resumes whole remaining chunks. **`CombinedStreamingDataset` and `ParallelStreamingDataset` resume only with the same `world_size`, `num_workers`, and `batch_size`.**
929
+
930
+ ```python
931
+ dataset = StreamingDataset("s3://my-bucket/my-data", shuffle=True, num_canonical_nodes=8)
932
+ ```
933
+
928
934
  </details>
929
935
 
930
936
 
@@ -932,11 +938,9 @@ for batch_idx, batch in enumerate(dataloader):
932
938
  <summary> ✅ Use shared queue for Optimizing <a id="shared-queue" href="#shared-queue">🔗</a> </summary>
933
939
  &nbsp;
934
940
 
935
- If you are using multiple workers to optimize your dataset, you can use a shared queue to speed up the process.
941
+ `optimize` / `map` default to a **shared per-node queue** (`keep_data_ordered=False`). Work is packed per node, then every worker on that node pulls the next item, so a slow worker does not leave others idle. Set `keep_data_ordered=True` to keep a static per-worker slice (required for `use_checkpoint` and `align_chunking`).
936
942
 
937
- This is especially useful when optimizing large datasets in parallel, where some workers may be slower than others.
938
-
939
- It can also improve fault tolerance when workers fail due to out-of-memory (OOM) errors.
943
+ Local `output_dir` writes chunks in place. Remote inputs and outputs use the streaming downloader (`adownload_file` / `aupload_file`, obstore when available).
940
944
 
941
945
  ```python
942
946
  import numpy as np
@@ -959,28 +963,36 @@ if __name__ == "__main__":
959
963
  output_dir="fast_data", # optimized data is stored here
960
964
  num_workers=4, # The number of workers on the same machine
961
965
  chunk_bytes="64MB" , # size of each chunk
962
- keep_data_ordered=False, # Use a shared queue to speed up the process
966
+ keep_data_ordered=False, # default: shared queue (set True to keep input order)
963
967
  )
964
968
  ```
965
969
 
966
970
 
967
- ### Performance Difference between using a shared queue and not using it:
971
+ ### Shared queue vs ordered (skewed local files)
968
972
 
969
- **Note**: The following benchmarks were collected using the ImageNet dataset on an A10G machine with 16 workers.
973
+ `scripts/bench/bench_node_queue.py --files 4000 --workers 8` (first 500 files are 1 MiB). On `main`, unordered optimize sat on a 200s empty-queue timeout after work finished.
970
974
 
971
- | Configuration | Optimize Time (sec) | Stream 1 (img/sec) | Stream 2 (img/sec) |
972
- |------------------|---------------------|---------------------|---------------------|
973
- | shared_queue (`keep_data_ordered=False`) | 1281 | 5392 | 5732 |
974
- | no shared_queue (`keep_data_ordered=True (default)`) | 1187 | 5257 | 5746 |
975
+ | Tree | Mode | Time | Throughput |
976
+ |------|------|-----:|-----------:|
977
+ | `main` (old default) | `keep_data_ordered=True` | 23.7s | 169 files/s |
978
+ | `main` | `keep_data_ordered=False` | 223.6s | 18 files/s |
979
+ | this tree | `keep_data_ordered=True` | 22.9s | 175 files/s |
980
+ | this tree (**new default**) | `keep_data_ordered=False` | **18.8s** | 213 files/s |
975
981
 
976
- 📌 Note: The **shared_queue** option impacts optimization time, not streaming speed.
977
- > While the streaming numbers may appear slightly different, this variation is incidental and not caused by shared_queue.
978
- >
979
- > Streaming happens after optimization and does not involve inter-process communication where shared_queue plays a role.
982
+ Shared-queue **before → after: ~12×**. New default vs old ordered default: **1.22×**.
983
+
984
+ ### Local / remote input × output
985
+
986
+ `python scripts/bench/bench_node_queue.py --files 200 --workers 4 --io-matrix` (first 50 files are 1 MiB).
980
987
 
981
- - 📄 Using a shared queue helps balance the load across workers, though it may slightly increase optimization time due to the overhead of pickling items sent between processes.
988
+ | Topology | Ordered | Shared | Speedup |
989
+ |----------|--------:|-------:|--------:|
990
+ | local → local | 6.55s | **2.96s** | 2.21× |
991
+ | remote → local | 7.44s | **3.48s** | 2.14× |
992
+ | local → remote | 10.98s | **6.51s** | 1.69× |
993
+ | remote → remote | 11.48s | **8.55s** | 1.34× |
982
994
 
983
- - ⚡ However, it can significantly improve optimizing performance — especially when some workers are slower than others.
995
+ Shared queue balances uneven workers. It does not change later `StreamingDataset` throughput.
984
996
 
985
997
  </details>
986
998
 
@@ -2280,7 +2292,7 @@ if __name__ == "__main__":
2280
2292
  | `start_method` | spawn† | Multiprocessing start method (†spawn unless IPython) |
2281
2293
  | `optimize_dns` | `None` | Optimized DNS (Studio / cloud) |
2282
2294
  | `storage_options` | `{}` | Cloud credentials / endpoints |
2283
- | `keep_data_ordered` | `True` | `False` = shared work queue (better for uneven/slow workers) |
2295
+ | `keep_data_ordered` | `False` | Shared work queue (faster for uneven workers). `True` keeps a static per-worker slice. Forced `True` with `use_checkpoint` / `align_chunking`. |
2284
2296
 
2285
2297
  </details>
2286
2298
 
@@ -2316,7 +2328,7 @@ Full knob list for `litdata.optimize` (see Quick start for the minimal recipe).
2316
2328
  | `start_method` | spawn† | Multiprocessing start method |
2317
2329
  | `optimize_dns` | `None` | Optimized DNS |
2318
2330
  | `storage_options` | `{}` | Cloud credentials / endpoints |
2319
- | `keep_data_ordered` | `True` | `False` = shared queue among workers |
2331
+ | `keep_data_ordered` | `False` | Shared queue among workers. `True` keeps input order. Forced `True` with `use_checkpoint` / `align_chunking`. |
2320
2332
  | `verbose` | `True` | Progress logging |
2321
2333
 
2322
2334
  Related features: [shared queue](#shared-queue), [queue input](#queue-input), [append/overwrite](#modify-datasets), [compression](#compression), [TokensLoader / LLM](#llm-training), [filter](#filter-data).
@@ -14,7 +14,7 @@
14
14
 
15
15
  import time
16
16
 
17
- __version__ = "0.2.68"
17
+ __version__ = "0.2.69"
18
18
  __author__ = "Lightning AI et al."
19
19
  __author_email__ = "pytorch@lightning.ai"
20
20
  __license__ = "Apache-2.0"