litdata 0.2.72__tar.gz → 0.2.73__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. {litdata-0.2.72/src/litdata.egg-info → litdata-0.2.73}/PKG-INFO +1 -1
  2. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/__about__.py +1 -1
  3. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/processing/data_processor.py +13 -2
  4. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/processing/functions.py +22 -0
  5. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/streaming/dataloader.py +10 -10
  6. {litdata-0.2.72 → litdata-0.2.73/src/litdata.egg-info}/PKG-INFO +1 -1
  7. {litdata-0.2.72 → litdata-0.2.73}/CONTRIBUTING.md +0 -0
  8. {litdata-0.2.72 → litdata-0.2.73}/LICENSE +0 -0
  9. {litdata-0.2.72 → litdata-0.2.73}/MANIFEST.in +0 -0
  10. {litdata-0.2.72 → litdata-0.2.73}/README.md +0 -0
  11. {litdata-0.2.72 → litdata-0.2.73}/requirements.txt +0 -0
  12. {litdata-0.2.72 → litdata-0.2.73}/setup.cfg +0 -0
  13. {litdata-0.2.72 → litdata-0.2.73}/setup.py +0 -0
  14. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/__init__.py +0 -0
  15. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/__main__.py +0 -0
  16. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/cli/__init__.py +0 -0
  17. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/cli/commands.py +0 -0
  18. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/cli/handler/__init__.py +0 -0
  19. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/cli/handler/cache.py +0 -0
  20. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/cli/handler/optimize.py +0 -0
  21. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/cli/parser.py +0 -0
  22. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/constants.py +0 -0
  23. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/debugger.py +0 -0
  24. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/exceptions.py +0 -0
  25. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/helpers.py +0 -0
  26. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/imports.py +0 -0
  27. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/processing/__init__.py +0 -0
  28. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/processing/complete.py +0 -0
  29. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/processing/media_folder.py +0 -0
  30. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/processing/readers.py +0 -0
  31. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/processing/utilities.py +0 -0
  32. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/raw/__init__.py +0 -0
  33. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/raw/dataset.py +0 -0
  34. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/raw/indexer.py +0 -0
  35. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/raw/types.py +0 -0
  36. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/requirements.py +0 -0
  37. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/streaming/__init__.py +0 -0
  38. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/streaming/async_prefetch.py +0 -0
  39. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/streaming/cache.py +0 -0
  40. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/streaming/client.py +0 -0
  41. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/streaming/collate.py +0 -0
  42. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/streaming/combined.py +0 -0
  43. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/streaming/compression.py +0 -0
  44. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/streaming/config.py +0 -0
  45. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/streaming/dataset.py +0 -0
  46. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/streaming/dataset_update.py +0 -0
  47. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/streaming/downloader.py +0 -0
  48. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/streaming/elastic.py +0 -0
  49. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/streaming/fs_provider.py +0 -0
  50. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/streaming/item_loader.py +0 -0
  51. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/streaming/parallel.py +0 -0
  52. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/streaming/posix_fast.py +0 -0
  53. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/streaming/reader.py +0 -0
  54. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/streaming/resolver.py +0 -0
  55. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/streaming/sampler.py +0 -0
  56. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/streaming/serializers.py +0 -0
  57. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/streaming/shuffle.py +0 -0
  58. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/streaming/timing.py +0 -0
  59. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/streaming/writer.py +0 -0
  60. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/types.py +0 -0
  61. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/utilities/__init__.py +0 -0
  62. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/utilities/_pytree.py +0 -0
  63. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/utilities/base.py +0 -0
  64. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/utilities/breakpoint.py +0 -0
  65. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/utilities/broadcast.py +0 -0
  66. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/utilities/dataset_utilities.py +0 -0
  67. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/utilities/encryption.py +0 -0
  68. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/utilities/env.py +0 -0
  69. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/utilities/format.py +0 -0
  70. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/utilities/hf_dataset.py +0 -0
  71. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/utilities/keys_index.py +0 -0
  72. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/utilities/packing.py +0 -0
  73. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/utilities/parquet.py +0 -0
  74. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/utilities/shuffle.py +0 -0
  75. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/utilities/subsample.py +0 -0
  76. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/utilities/torch_utils.py +0 -0
  77. {litdata-0.2.72 → litdata-0.2.73}/src/litdata/utilities/train_test_split.py +0 -0
  78. {litdata-0.2.72 → litdata-0.2.73}/src/litdata.egg-info/SOURCES.txt +0 -0
  79. {litdata-0.2.72 → litdata-0.2.73}/src/litdata.egg-info/dependency_links.txt +0 -0
  80. {litdata-0.2.72 → litdata-0.2.73}/src/litdata.egg-info/entry_points.txt +0 -0
  81. {litdata-0.2.72 → litdata-0.2.73}/src/litdata.egg-info/not-zip-safe +0 -0
  82. {litdata-0.2.72 → litdata-0.2.73}/src/litdata.egg-info/requires.txt +0 -0
  83. {litdata-0.2.72 → litdata-0.2.73}/src/litdata.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: litdata
3
- Version: 0.2.72
3
+ Version: 0.2.73
4
4
  Summary: The Deep Learning framework to train, deploy, and ship AI products Lightning fast.
5
5
  Home-page: https://github.com/Lightning-AI/litdata
6
6
  Download-URL: https://github.com/Lightning-AI/litdata
@@ -14,7 +14,7 @@
14
14
 
15
15
  import time
16
16
 
17
- __version__ = "0.2.72"
17
+ __version__ = "0.2.73"
18
18
  __author__ = "Lightning AI et al."
19
19
  __author_email__ = "pytorch@lightning.ai"
20
20
  __license__ = "Apache-2.0"
@@ -1523,7 +1523,16 @@ class DataChunkRecipe(DataRecipe):
1523
1523
 
1524
1524
  merge_cache = Cache(cache_dir, chunk_bytes=1)
1525
1525
  node_rank = _get_node_rank()
1526
- merge_cache._merge_no_wait(node_rank if num_nodes > 1 else None, getattr(self, "existing_index", None))
1526
+ # With more than one node, every node would otherwise fold the existing
1527
+ # index into its own {node_rank}-index.json, and the final cross-node
1528
+ # merge below then repeats those chunks once per node. Add the existing
1529
+ # index only once, either here for the single-node case or in the final
1530
+ # merge for the multi-node case.
1531
+ existing_index = getattr(self, "existing_index", None)
1532
+ merge_cache._merge_no_wait(
1533
+ node_rank if num_nodes > 1 else None,
1534
+ existing_index if num_nodes == 1 else None,
1535
+ )
1527
1536
 
1528
1537
  self._merge_and_upload_keys(output_dir, cache_dir, num_nodes, node_rank)
1529
1538
  self._upload_index(output_dir, cache_dir, num_nodes, node_rank)
@@ -1734,7 +1743,9 @@ class DataChunkRecipe(DataRecipe):
1734
1743
  shutil.copyfile(remote_filepath, node_index_filepath)
1735
1744
 
1736
1745
  merge_cache = Cache(merge_dir, chunk_bytes=1)
1737
- merge_cache._merge_no_wait()
1746
+ # The per-node index files hold only their own new chunks now, so the
1747
+ # existing index is folded in here, once, as the node files are merged.
1748
+ merge_cache._merge_no_wait(existing_index=getattr(self, "existing_index", None))
1738
1749
  self._upload_index(output_dir, merge_dir, 1, None)
1739
1750
 
1740
1751
 
@@ -147,6 +147,17 @@ class LambdaMapRecipe(MapRecipe):
147
147
  self._contains_device = "device" in params
148
148
  self._contains_is_last = "is_last" in params
149
149
 
150
+ def __getstate__(self) -> dict[str, Any]:
151
+ """Drop the full input sequence when pickling into spawn workers.
152
+
153
+ The parent already shards items onto per-worker queues. Workers only need
154
+ ``prepare_item``; keeping ``_inputs`` in the pickle would duplicate the list
155
+ once per process.
156
+ """
157
+ state = self.__dict__.copy()
158
+ state["_inputs"] = None
159
+ return state
160
+
150
161
  def prepare_structure(self, _: str | None) -> Any:
151
162
  return self._inputs
152
163
 
@@ -210,6 +221,17 @@ class LambdaDataChunkRecipe(DataChunkRecipe):
210
221
 
211
222
  self.prepare_item = self._prepare_item_generator if self.is_generator else self._prepare_item # type: ignore
212
223
 
224
+ def __getstate__(self) -> dict[str, Any]:
225
+ """Drop the full input sequence when pickling into spawn workers.
226
+
227
+ The parent already shards items onto per-worker queues. Workers only need
228
+ ``prepare_item``; keeping ``_inputs`` in the pickle would duplicate the list
229
+ once per process.
230
+ """
231
+ state = self.__dict__.copy()
232
+ state["_inputs"] = None
233
+ return state
234
+
213
235
  def check_fn(self) -> None:
214
236
  if (
215
237
  isinstance(self._fn, (partial, FunctionType))
@@ -558,25 +558,30 @@ class _StreamingMultiProcessingDataLoaderIter(_MultiProcessingDataLoaderIter):
558
558
  )
559
559
  self._num_workers = loader.num_workers
560
560
  self._cprofile: cProfile.Profile | None = None
561
- self._orig_worker_loop: Any = None
562
561
 
563
562
  distributed_env = _DistributedEnv.detect()
564
563
  profile_cprofile = bool(getattr(self._loader, "_profile_cprofile", False))
564
+ from torch.utils.data._utils import worker
565
565
 
566
- if distributed_env.global_rank == 0:
567
- from torch.utils.data._utils import worker
566
+ original_worker_loop: Any = None
568
567
 
568
+ if distributed_env.global_rank == 0:
569
569
  if self._loader._profile_batches and _VIZ_TRACKER_AVAILABLE:
570
+ original_worker_loop = worker._worker_loop
570
571
  worker._worker_loop = _ProfileWorkerLoop(
571
572
  self._loader._profile_batches, self._loader._profile_skip_batches, self._loader._profile_dir
572
573
  )
573
574
  elif profile_cprofile:
574
- self._orig_worker_loop = worker._worker_loop
575
+ original_worker_loop = worker._worker_loop
575
576
  worker._worker_loop = _CProfileWorkerLoop(_cprofile_output_dir(self._loader._profile_dir))
576
577
 
577
578
  # Workers fork/spawn here. Enable the parent profiler only after that so
578
579
  # the child does not inherit an active cProfile (one profiler per process).
579
- super().__init__(loader)
580
+ try:
581
+ super().__init__(loader)
582
+ finally:
583
+ if original_worker_loop is not None:
584
+ worker._worker_loop = original_worker_loop
580
585
 
581
586
  if profile_cprofile and distributed_env.global_rank == 0:
582
587
  self._cprofile = cProfile.Profile()
@@ -587,11 +592,6 @@ class _StreamingMultiProcessingDataLoaderIter(_MultiProcessingDataLoaderIter):
587
592
  stem = os.path.join(_cprofile_output_dir(self._loader._profile_dir), _CPROFILE_MAIN_STEM)
588
593
  _dump_cprofile(self._cprofile, stem)
589
594
  self._cprofile = None
590
- if self._orig_worker_loop is not None:
591
- from torch.utils.data._utils import worker
592
-
593
- worker._worker_loop = self._orig_worker_loop
594
- self._orig_worker_loop = None
595
595
  super()._shutdown_workers()
596
596
 
597
597
  def _try_put_index(self) -> None:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: litdata
3
- Version: 0.2.72
3
+ Version: 0.2.73
4
4
  Summary: The Deep Learning framework to train, deploy, and ship AI products Lightning fast.
5
5
  Home-page: https://github.com/Lightning-AI/litdata
6
6
  Download-URL: https://github.com/Lightning-AI/litdata
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes