lazynwb 1.0.0.dev2__tar.gz → 1.0.0.dev5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {lazynwb-1.0.0.dev2/src/lazynwb.egg-info → lazynwb-1.0.0.dev5}/PKG-INFO +6 -4
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/pyproject.toml +6 -4
- lazynwb-1.0.0.dev5/src/lazynwb/_zarr/chunk_planner.py +440 -0
- lazynwb-1.0.0.dev5/src/lazynwb/_zarr/chunk_transfer.py +468 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_zarr/reader.py +571 -2
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/base.py +102 -100
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/conversion.py +203 -129
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/file_io.py +297 -14
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/lazyframe.py +146 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/table_metadata.py +18 -2
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/tables.py +560 -53
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/timeseries.py +165 -2
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5/src/lazynwb.egg-info}/PKG-INFO +6 -4
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb.egg-info/SOURCES.txt +5 -1
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb.egg-info/requires.txt +9 -3
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/tests/test_catalog_backend.py +168 -14
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/tests/test_file_io.py +119 -27
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/tests/test_lazyframe.py +37 -2
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/tests/test_tables.py +222 -25
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/tests/test_timeseries.py +84 -0
- lazynwb-1.0.0.dev5/tests/test_zarr_chunk_planner.py +216 -0
- lazynwb-1.0.0.dev5/tests/test_zarr_chunk_transfer.py +270 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/LICENSE +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/README.md +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/setup.cfg +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/__init__.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_cache/__init__.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_cache/sqlite.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_catalog/__init__.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_catalog/_schema.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_catalog/accessor.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_catalog/backend.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_catalog/models.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_catalog/polars.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_cli/__init__.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_cli/_config.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_cli/_context.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_cli/_errors.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_cli/_formatting.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_cli/_main.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_cli/_preview.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_cli/_query.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_cli/_schema.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_cli/_sources.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_cli/_tables.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_config.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_hdf5/__init__.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_hdf5/parser.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_hdf5/range_reader.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_hdf5/reader.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_storage_options.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/_zarr/__init__.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/attrs.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/dandi.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/exceptions.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/py.typed +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/types_.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb/utils.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb.egg-info/dependency_links.txt +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb.egg-info/entry_points.txt +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/src/lazynwb.egg-info/top_level.txt +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/tests/test_attrs.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/tests/test_cache.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/tests/test_catalog_schema.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/tests/test_cli.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/tests/test_conversion.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/tests/test_core.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/tests/test_dandi.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/tests/test_dandi_metadata.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/tests/test_dandi_tables.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/tests/test_dandi_timeseries.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/tests/test_dandi_unit.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/tests/test_hdf5_backend.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/tests/test_range_reader.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/tests/test_remote_schema_budget.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/tests/test_speed.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/tests/test_table_metadata.py +0 -0
- {lazynwb-1.0.0.dev2 → lazynwb-1.0.0.dev5}/tests/test_utils.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: lazynwb
|
|
3
|
-
Version: 1.0.0.
|
|
3
|
+
Version: 1.0.0.dev5
|
|
4
4
|
Summary: An attempt to speed-up access to large NWB (Neurodata Without Borders) files stored in the cloud.
|
|
5
5
|
Author-email: Ben Hardcastle <ben.hardcastle@alleninstitue.org>
|
|
6
6
|
License: MIT
|
|
@@ -20,8 +20,10 @@ Requires-Python: >=3.10
|
|
|
20
20
|
Description-Content-Type: text/markdown
|
|
21
21
|
License-File: LICENSE
|
|
22
22
|
Requires-Dist: h5py>=3.10.0
|
|
23
|
-
Requires-Dist: zarr<3.0,>=2.17.0
|
|
24
|
-
Requires-Dist:
|
|
23
|
+
Requires-Dist: zarr<3.0,>=2.17.0; python_version < "3.11"
|
|
24
|
+
Requires-Dist: zarr<4.0,>=3.1.0; python_version >= "3.11"
|
|
25
|
+
Requires-Dist: numcodecs<0.16; python_version < "3.11"
|
|
26
|
+
Requires-Dist: numcodecs>=0.15.1; python_version >= "3.11"
|
|
25
27
|
Requires-Dist: remfile>=0.1.10
|
|
26
28
|
Requires-Dist: tqdm>=4.66.2
|
|
27
29
|
Requires-Dist: pandas>=2.0.0
|
|
@@ -36,7 +38,7 @@ Requires-Dist: pydantic-settings>=2.11.0
|
|
|
36
38
|
Requires-Dist: aiosqlite>=0.20.0
|
|
37
39
|
Requires-Dist: tomli>=2.0.0; python_version < "3.11"
|
|
38
40
|
Provides-Extra: pynwb
|
|
39
|
-
Requires-Dist: hdmf-zarr>=0.12.0; extra == "pynwb"
|
|
41
|
+
Requires-Dist: hdmf-zarr>=0.12.0; python_version < "3.11" and extra == "pynwb"
|
|
40
42
|
Requires-Dist: pynwb>=3.1.3; extra == "pynwb"
|
|
41
43
|
Dynamic: license-file
|
|
42
44
|
|
|
@@ -8,8 +8,10 @@ readme = "README.md"
|
|
|
8
8
|
requires-python = ">=3.10"
|
|
9
9
|
dependencies = [
|
|
10
10
|
"h5py>=3.10.0",
|
|
11
|
-
"zarr>=2.17.0,<3.0",
|
|
12
|
-
"
|
|
11
|
+
"zarr>=2.17.0,<3.0; python_version < '3.11'",
|
|
12
|
+
"zarr>=3.1.0,<4.0; python_version >= '3.11'",
|
|
13
|
+
"numcodecs<0.16; python_version < '3.11'", # for zarr<3.0 https://github.com/zarr-developers/numcodecs/issues/721
|
|
14
|
+
"numcodecs>=0.15.1; python_version >= '3.11'",
|
|
13
15
|
"remfile>=0.1.10",
|
|
14
16
|
"tqdm>=4.66.2",
|
|
15
17
|
"pandas>=2.0.0",
|
|
@@ -24,7 +26,7 @@ dependencies = [
|
|
|
24
26
|
"aiosqlite>=0.20.0",
|
|
25
27
|
"tomli>=2.0.0; python_version < '3.11'",
|
|
26
28
|
]
|
|
27
|
-
version = "1.0.0.
|
|
29
|
+
version = "1.0.0.dev5"
|
|
28
30
|
classifiers = [
|
|
29
31
|
"Development Status :: 3 - Alpha", # https://pypi.org/classifiers/
|
|
30
32
|
"Programming Language :: Python :: 3",
|
|
@@ -55,7 +57,7 @@ task = "poethepoet:main"
|
|
|
55
57
|
|
|
56
58
|
[project.optional-dependencies]
|
|
57
59
|
pynwb = [
|
|
58
|
-
"hdmf-zarr>=0.12.0",
|
|
60
|
+
"hdmf-zarr>=0.12.0; python_version < '3.11'",
|
|
59
61
|
"pynwb>=3.1.3",
|
|
60
62
|
]
|
|
61
63
|
|
|
@@ -0,0 +1,440 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import dataclasses
|
|
4
|
+
import itertools
|
|
5
|
+
import logging
|
|
6
|
+
from collections.abc import Container, Iterator, Sequence
|
|
7
|
+
|
|
8
|
+
logger = logging.getLogger(__name__)
|
|
9
|
+
|
|
10
|
+
_SUPPORTED_RANKS = frozenset({1, 2})
|
|
11
|
+
_SUPPORTED_DIMENSION_SEPARATORS = frozenset({".", "/"})
|
|
12
|
+
_IndexSelection = int | slice
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclasses.dataclass(frozen=True, slots=True)
|
|
16
|
+
class _ChunkReadPlan:
|
|
17
|
+
"""Pure plan for one logical Zarr v2 chunk read."""
|
|
18
|
+
|
|
19
|
+
chunk_coords: tuple[int, ...]
|
|
20
|
+
chunk_key: str
|
|
21
|
+
input_selection: tuple[_IndexSelection, ...]
|
|
22
|
+
output_selection: tuple[slice, ...]
|
|
23
|
+
array_selection: tuple[_IndexSelection, ...]
|
|
24
|
+
is_missing: bool
|
|
25
|
+
fill_value: object | None
|
|
26
|
+
|
|
27
|
+
@property
|
|
28
|
+
def requires_fill(self) -> bool:
|
|
29
|
+
return self.is_missing
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclasses.dataclass(frozen=True, slots=True)
|
|
33
|
+
class _ArrayChunkPlan:
|
|
34
|
+
"""Pure exact-array chunk plan independent of transport and decoding."""
|
|
35
|
+
|
|
36
|
+
array_path: str
|
|
37
|
+
shape: tuple[int, ...]
|
|
38
|
+
chunk_shape: tuple[int, ...]
|
|
39
|
+
selection: tuple[_IndexSelection, ...]
|
|
40
|
+
dimension_separator: str
|
|
41
|
+
output_shape: tuple[int, ...]
|
|
42
|
+
chunk_reads: tuple[_ChunkReadPlan, ...]
|
|
43
|
+
fill_value: object | None
|
|
44
|
+
missing_chunk_count: int
|
|
45
|
+
|
|
46
|
+
@property
|
|
47
|
+
def chunk_count(self) -> int:
|
|
48
|
+
return len(self.chunk_reads)
|
|
49
|
+
|
|
50
|
+
@property
|
|
51
|
+
def has_missing_chunks(self) -> bool:
|
|
52
|
+
return self.missing_chunk_count > 0
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@dataclasses.dataclass(frozen=True, slots=True)
|
|
56
|
+
class _AxisSelection:
|
|
57
|
+
start: int
|
|
58
|
+
stop: int
|
|
59
|
+
index: int | None
|
|
60
|
+
output_axis: int | None
|
|
61
|
+
|
|
62
|
+
@property
|
|
63
|
+
def is_integer(self) -> bool:
|
|
64
|
+
return self.index is not None
|
|
65
|
+
|
|
66
|
+
@property
|
|
67
|
+
def output_size(self) -> int:
|
|
68
|
+
if self.is_integer:
|
|
69
|
+
return 0
|
|
70
|
+
return max(0, self.stop - self.start)
|
|
71
|
+
|
|
72
|
+
def as_index_selection(self) -> _IndexSelection:
|
|
73
|
+
if self.index is not None:
|
|
74
|
+
return self.index
|
|
75
|
+
return slice(self.start, self.stop)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _plan_array_chunks(
|
|
79
|
+
*,
|
|
80
|
+
array_path: str,
|
|
81
|
+
shape: Sequence[int],
|
|
82
|
+
chunks: Sequence[int],
|
|
83
|
+
selection: object = None,
|
|
84
|
+
dimension_separator: str = ".",
|
|
85
|
+
fill_value: object | None = None,
|
|
86
|
+
available_chunk_keys: Container[str] | None = None,
|
|
87
|
+
missing_chunk_keys: Container[str] | None = None,
|
|
88
|
+
) -> _ArrayChunkPlan:
|
|
89
|
+
"""Plan Zarr v2 chunk reads for a simple exact-array selection.
|
|
90
|
+
|
|
91
|
+
Supports 1D/2D arrays, unit-step slices, and integer indices. Missing chunk
|
|
92
|
+
metadata is only derived from caller-provided key containers; when neither
|
|
93
|
+
container is provided, all planned chunks are treated as present.
|
|
94
|
+
"""
|
|
95
|
+
|
|
96
|
+
normalized_shape = _normalize_shape(shape)
|
|
97
|
+
chunk_shape = _normalize_chunk_shape(chunks, len(normalized_shape))
|
|
98
|
+
normalized_dimension_separator = _normalize_dimension_separator(dimension_separator)
|
|
99
|
+
normalized_array_path = _normalize_array_path(array_path)
|
|
100
|
+
axis_selections = _normalize_selection(selection, normalized_shape)
|
|
101
|
+
output_shape = tuple(
|
|
102
|
+
axis.output_size for axis in axis_selections if axis.output_axis is not None
|
|
103
|
+
)
|
|
104
|
+
normalized_selection = tuple(axis.as_index_selection() for axis in axis_selections)
|
|
105
|
+
_validate_missing_chunk_inputs(available_chunk_keys, missing_chunk_keys)
|
|
106
|
+
|
|
107
|
+
logger.debug(
|
|
108
|
+
"planning Zarr v2 chunks for %r shape=%s chunks=%s selection=%r "
|
|
109
|
+
"dimension_separator=%r",
|
|
110
|
+
normalized_array_path,
|
|
111
|
+
normalized_shape,
|
|
112
|
+
chunk_shape,
|
|
113
|
+
normalized_selection,
|
|
114
|
+
normalized_dimension_separator,
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
chunk_reads = tuple(
|
|
118
|
+
_plan_chunk_read(
|
|
119
|
+
array_path=normalized_array_path,
|
|
120
|
+
shape=normalized_shape,
|
|
121
|
+
chunk_shape=chunk_shape,
|
|
122
|
+
axis_selections=axis_selections,
|
|
123
|
+
chunk_coords=chunk_coords,
|
|
124
|
+
dimension_separator=normalized_dimension_separator,
|
|
125
|
+
fill_value=fill_value,
|
|
126
|
+
available_chunk_keys=available_chunk_keys,
|
|
127
|
+
missing_chunk_keys=missing_chunk_keys,
|
|
128
|
+
)
|
|
129
|
+
for chunk_coords in _iter_chunk_coords(axis_selections, chunk_shape)
|
|
130
|
+
)
|
|
131
|
+
missing_chunk_count = sum(1 for chunk_read in chunk_reads if chunk_read.is_missing)
|
|
132
|
+
plan = _ArrayChunkPlan(
|
|
133
|
+
array_path=normalized_array_path,
|
|
134
|
+
shape=normalized_shape,
|
|
135
|
+
chunk_shape=chunk_shape,
|
|
136
|
+
selection=normalized_selection,
|
|
137
|
+
dimension_separator=normalized_dimension_separator,
|
|
138
|
+
output_shape=output_shape,
|
|
139
|
+
chunk_reads=chunk_reads,
|
|
140
|
+
fill_value=fill_value,
|
|
141
|
+
missing_chunk_count=missing_chunk_count,
|
|
142
|
+
)
|
|
143
|
+
_log_plan(plan)
|
|
144
|
+
return plan
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _normalize_shape(shape: Sequence[int]) -> tuple[int, ...]:
|
|
148
|
+
normalized = tuple(int(axis_size) for axis_size in shape)
|
|
149
|
+
if len(normalized) not in _SUPPORTED_RANKS:
|
|
150
|
+
msg = f"Zarr exact-array chunk planning supports 1D/2D arrays, got rank {len(normalized)}"
|
|
151
|
+
raise ValueError(msg)
|
|
152
|
+
if any(axis_size < 0 for axis_size in normalized):
|
|
153
|
+
msg = f"array shape values must be non-negative, got {normalized!r}"
|
|
154
|
+
raise ValueError(msg)
|
|
155
|
+
return normalized
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _normalize_chunk_shape(chunks: Sequence[int], rank: int) -> tuple[int, ...]:
|
|
159
|
+
chunk_shape = tuple(int(chunk_size) for chunk_size in chunks)
|
|
160
|
+
if len(chunk_shape) != rank:
|
|
161
|
+
msg = f"chunk rank {len(chunk_shape)} does not match array rank {rank}"
|
|
162
|
+
raise ValueError(msg)
|
|
163
|
+
if any(chunk_size <= 0 for chunk_size in chunk_shape):
|
|
164
|
+
msg = f"chunk sizes must be positive, got {chunk_shape!r}"
|
|
165
|
+
raise ValueError(msg)
|
|
166
|
+
return chunk_shape
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def _normalize_dimension_separator(dimension_separator: str) -> str:
|
|
170
|
+
if dimension_separator not in _SUPPORTED_DIMENSION_SEPARATORS:
|
|
171
|
+
msg = (
|
|
172
|
+
"Zarr v2 dimension_separator must be '.' or '/', "
|
|
173
|
+
f"got {dimension_separator!r}"
|
|
174
|
+
)
|
|
175
|
+
raise ValueError(msg)
|
|
176
|
+
return dimension_separator
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def _normalize_array_path(array_path: str) -> str:
|
|
180
|
+
if array_path in {"", "/"}:
|
|
181
|
+
return ""
|
|
182
|
+
return "/".join(part for part in str(array_path).split("/") if part)
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def _normalize_selection(
|
|
186
|
+
selection: object,
|
|
187
|
+
shape: tuple[int, ...],
|
|
188
|
+
) -> tuple[_AxisSelection, ...]:
|
|
189
|
+
raw_selection = _selection_tuple(selection, len(shape))
|
|
190
|
+
output_axis = 0
|
|
191
|
+
normalized_axes: list[_AxisSelection] = []
|
|
192
|
+
for axis_size, axis_selection in zip(shape, raw_selection):
|
|
193
|
+
normalized_axis = _normalize_axis_selection(
|
|
194
|
+
axis_selection,
|
|
195
|
+
axis_size=axis_size,
|
|
196
|
+
output_axis=output_axis,
|
|
197
|
+
)
|
|
198
|
+
normalized_axes.append(normalized_axis)
|
|
199
|
+
if normalized_axis.output_axis is not None:
|
|
200
|
+
output_axis += 1
|
|
201
|
+
return tuple(normalized_axes)
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def _selection_tuple(selection: object, rank: int) -> tuple[object, ...]:
|
|
205
|
+
if selection is None or selection is Ellipsis:
|
|
206
|
+
return tuple(slice(None) for _ in range(rank))
|
|
207
|
+
if not isinstance(selection, tuple):
|
|
208
|
+
return _pad_selection((selection,), rank)
|
|
209
|
+
if selection.count(Ellipsis) > 1:
|
|
210
|
+
msg = "selection can contain at most one ellipsis"
|
|
211
|
+
raise IndexError(msg)
|
|
212
|
+
if Ellipsis in selection:
|
|
213
|
+
ellipsis_index = selection.index(Ellipsis)
|
|
214
|
+
explicit_count = len(selection) - 1
|
|
215
|
+
if explicit_count > rank:
|
|
216
|
+
msg = f"too many indices for {rank}D array"
|
|
217
|
+
raise IndexError(msg)
|
|
218
|
+
fill_count = rank - explicit_count
|
|
219
|
+
expanded = (
|
|
220
|
+
*selection[:ellipsis_index],
|
|
221
|
+
*(slice(None) for _ in range(fill_count)),
|
|
222
|
+
*selection[ellipsis_index + 1 :],
|
|
223
|
+
)
|
|
224
|
+
return _pad_selection(expanded, rank)
|
|
225
|
+
return _pad_selection(selection, rank)
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def _pad_selection(selection: tuple[object, ...], rank: int) -> tuple[object, ...]:
|
|
229
|
+
if len(selection) > rank:
|
|
230
|
+
msg = f"too many indices for {rank}D array"
|
|
231
|
+
raise IndexError(msg)
|
|
232
|
+
return (*selection, *(slice(None) for _ in range(rank - len(selection))))
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def _normalize_axis_selection(
|
|
236
|
+
axis_selection: object,
|
|
237
|
+
*,
|
|
238
|
+
axis_size: int,
|
|
239
|
+
output_axis: int,
|
|
240
|
+
) -> _AxisSelection:
|
|
241
|
+
if isinstance(axis_selection, slice):
|
|
242
|
+
start, stop, step = axis_selection.indices(axis_size)
|
|
243
|
+
if step != 1:
|
|
244
|
+
msg = "only unit-step slices are supported for Zarr chunk planning"
|
|
245
|
+
raise ValueError(msg)
|
|
246
|
+
return _AxisSelection(
|
|
247
|
+
start=start,
|
|
248
|
+
stop=stop,
|
|
249
|
+
index=None,
|
|
250
|
+
output_axis=output_axis,
|
|
251
|
+
)
|
|
252
|
+
if isinstance(axis_selection, int):
|
|
253
|
+
index = _normalize_integer_index(axis_selection, axis_size)
|
|
254
|
+
return _AxisSelection(
|
|
255
|
+
start=index,
|
|
256
|
+
stop=index + 1,
|
|
257
|
+
index=index,
|
|
258
|
+
output_axis=None,
|
|
259
|
+
)
|
|
260
|
+
msg = f"unsupported Zarr chunk selection item {axis_selection!r}"
|
|
261
|
+
raise TypeError(msg)
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def _normalize_integer_index(index: int, axis_size: int) -> int:
|
|
265
|
+
normalized_index = index + axis_size if index < 0 else index
|
|
266
|
+
if normalized_index < 0 or normalized_index >= axis_size:
|
|
267
|
+
msg = f"index {index} is out of bounds for axis with size {axis_size}"
|
|
268
|
+
raise IndexError(msg)
|
|
269
|
+
return normalized_index
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def _iter_chunk_coords(
|
|
273
|
+
axis_selections: tuple[_AxisSelection, ...],
|
|
274
|
+
chunk_shape: tuple[int, ...],
|
|
275
|
+
) -> Iterator[tuple[int, ...]]:
|
|
276
|
+
per_axis_coords = tuple(
|
|
277
|
+
_axis_chunk_coords(axis_selection, chunk_size)
|
|
278
|
+
for axis_selection, chunk_size in zip(axis_selections, chunk_shape)
|
|
279
|
+
)
|
|
280
|
+
return itertools.product(*per_axis_coords)
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def _axis_chunk_coords(
|
|
284
|
+
axis_selection: _AxisSelection,
|
|
285
|
+
chunk_size: int,
|
|
286
|
+
) -> range:
|
|
287
|
+
if axis_selection.index is not None:
|
|
288
|
+
chunk_coord = axis_selection.index // chunk_size
|
|
289
|
+
return range(chunk_coord, chunk_coord + 1)
|
|
290
|
+
if axis_selection.start >= axis_selection.stop:
|
|
291
|
+
return range(0)
|
|
292
|
+
first_chunk = axis_selection.start // chunk_size
|
|
293
|
+
last_chunk = (axis_selection.stop - 1) // chunk_size
|
|
294
|
+
return range(first_chunk, last_chunk + 1)
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
def _plan_chunk_read(
|
|
298
|
+
*,
|
|
299
|
+
array_path: str,
|
|
300
|
+
shape: tuple[int, ...],
|
|
301
|
+
chunk_shape: tuple[int, ...],
|
|
302
|
+
axis_selections: tuple[_AxisSelection, ...],
|
|
303
|
+
chunk_coords: tuple[int, ...],
|
|
304
|
+
dimension_separator: str,
|
|
305
|
+
fill_value: object | None,
|
|
306
|
+
available_chunk_keys: Container[str] | None,
|
|
307
|
+
missing_chunk_keys: Container[str] | None,
|
|
308
|
+
) -> _ChunkReadPlan:
|
|
309
|
+
chunk_key = _chunk_key(array_path, chunk_coords, dimension_separator)
|
|
310
|
+
input_selection, output_selection, array_selection = _chunk_selections(
|
|
311
|
+
shape=shape,
|
|
312
|
+
chunk_shape=chunk_shape,
|
|
313
|
+
axis_selections=axis_selections,
|
|
314
|
+
chunk_coords=chunk_coords,
|
|
315
|
+
)
|
|
316
|
+
return _ChunkReadPlan(
|
|
317
|
+
chunk_coords=chunk_coords,
|
|
318
|
+
chunk_key=chunk_key,
|
|
319
|
+
input_selection=input_selection,
|
|
320
|
+
output_selection=output_selection,
|
|
321
|
+
array_selection=array_selection,
|
|
322
|
+
is_missing=_is_missing_chunk(
|
|
323
|
+
chunk_key,
|
|
324
|
+
available_chunk_keys=available_chunk_keys,
|
|
325
|
+
missing_chunk_keys=missing_chunk_keys,
|
|
326
|
+
),
|
|
327
|
+
fill_value=fill_value,
|
|
328
|
+
)
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
def _chunk_key(
|
|
332
|
+
array_path: str,
|
|
333
|
+
chunk_coords: tuple[int, ...],
|
|
334
|
+
dimension_separator: str,
|
|
335
|
+
) -> str:
|
|
336
|
+
chunk_name = dimension_separator.join(str(coord) for coord in chunk_coords)
|
|
337
|
+
if not array_path:
|
|
338
|
+
return chunk_name
|
|
339
|
+
return f"{array_path}/{chunk_name}"
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
def _chunk_selections(
|
|
343
|
+
*,
|
|
344
|
+
shape: tuple[int, ...],
|
|
345
|
+
chunk_shape: tuple[int, ...],
|
|
346
|
+
axis_selections: tuple[_AxisSelection, ...],
|
|
347
|
+
chunk_coords: tuple[int, ...],
|
|
348
|
+
) -> tuple[tuple[_IndexSelection, ...], tuple[slice, ...], tuple[_IndexSelection, ...]]:
|
|
349
|
+
input_selection: list[_IndexSelection] = []
|
|
350
|
+
output_selection: list[slice] = []
|
|
351
|
+
array_selection: list[_IndexSelection] = []
|
|
352
|
+
|
|
353
|
+
for axis_size, chunk_size, axis_selection, chunk_coord in zip(
|
|
354
|
+
shape,
|
|
355
|
+
chunk_shape,
|
|
356
|
+
axis_selections,
|
|
357
|
+
chunk_coords,
|
|
358
|
+
):
|
|
359
|
+
chunk_start = chunk_coord * chunk_size
|
|
360
|
+
chunk_stop = min(chunk_start + chunk_size, axis_size)
|
|
361
|
+
_append_axis_selections(
|
|
362
|
+
input_selection=input_selection,
|
|
363
|
+
output_selection=output_selection,
|
|
364
|
+
array_selection=array_selection,
|
|
365
|
+
axis_selection=axis_selection,
|
|
366
|
+
chunk_start=chunk_start,
|
|
367
|
+
chunk_stop=chunk_stop,
|
|
368
|
+
)
|
|
369
|
+
|
|
370
|
+
return tuple(input_selection), tuple(output_selection), tuple(array_selection)
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
def _append_axis_selections(
|
|
374
|
+
*,
|
|
375
|
+
input_selection: list[_IndexSelection],
|
|
376
|
+
output_selection: list[slice],
|
|
377
|
+
array_selection: list[_IndexSelection],
|
|
378
|
+
axis_selection: _AxisSelection,
|
|
379
|
+
chunk_start: int,
|
|
380
|
+
chunk_stop: int,
|
|
381
|
+
) -> None:
|
|
382
|
+
if axis_selection.index is not None:
|
|
383
|
+
input_selection.append(axis_selection.index - chunk_start)
|
|
384
|
+
array_selection.append(axis_selection.index)
|
|
385
|
+
return
|
|
386
|
+
|
|
387
|
+
read_start = max(axis_selection.start, chunk_start)
|
|
388
|
+
read_stop = min(axis_selection.stop, chunk_stop)
|
|
389
|
+
input_selection.append(slice(read_start - chunk_start, read_stop - chunk_start))
|
|
390
|
+
output_selection.append(
|
|
391
|
+
slice(read_start - axis_selection.start, read_stop - axis_selection.start)
|
|
392
|
+
)
|
|
393
|
+
array_selection.append(slice(read_start, read_stop))
|
|
394
|
+
|
|
395
|
+
|
|
396
|
+
def _validate_missing_chunk_inputs(
|
|
397
|
+
available_chunk_keys: Container[str] | None,
|
|
398
|
+
missing_chunk_keys: Container[str] | None,
|
|
399
|
+
) -> None:
|
|
400
|
+
if available_chunk_keys is not None and missing_chunk_keys is not None:
|
|
401
|
+
msg = "provide either available_chunk_keys or missing_chunk_keys, not both"
|
|
402
|
+
raise ValueError(msg)
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
def _is_missing_chunk(
|
|
406
|
+
chunk_key: str,
|
|
407
|
+
*,
|
|
408
|
+
available_chunk_keys: Container[str] | None,
|
|
409
|
+
missing_chunk_keys: Container[str] | None,
|
|
410
|
+
) -> bool:
|
|
411
|
+
if available_chunk_keys is not None:
|
|
412
|
+
return chunk_key not in available_chunk_keys
|
|
413
|
+
if missing_chunk_keys is not None:
|
|
414
|
+
return chunk_key in missing_chunk_keys
|
|
415
|
+
return False
|
|
416
|
+
|
|
417
|
+
|
|
418
|
+
def _log_plan(plan: _ArrayChunkPlan) -> None:
|
|
419
|
+
logger.debug(
|
|
420
|
+
"planned Zarr v2 chunks for %r: output_shape=%s chunk_count=%d "
|
|
421
|
+
"missing_chunk_count=%d fill_value=%r",
|
|
422
|
+
plan.array_path,
|
|
423
|
+
plan.output_shape,
|
|
424
|
+
plan.chunk_count,
|
|
425
|
+
plan.missing_chunk_count,
|
|
426
|
+
plan.fill_value,
|
|
427
|
+
)
|
|
428
|
+
if not logger.isEnabledFor(logging.DEBUG):
|
|
429
|
+
return
|
|
430
|
+
for chunk_read in plan.chunk_reads:
|
|
431
|
+
logger.debug(
|
|
432
|
+
"planned Zarr v2 chunk key=%r coords=%s input=%r output=%r "
|
|
433
|
+
"array=%r missing=%s",
|
|
434
|
+
chunk_read.chunk_key,
|
|
435
|
+
chunk_read.chunk_coords,
|
|
436
|
+
chunk_read.input_selection,
|
|
437
|
+
chunk_read.output_selection,
|
|
438
|
+
chunk_read.array_selection,
|
|
439
|
+
chunk_read.is_missing,
|
|
440
|
+
)
|