xarray-binfile 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,22 @@
1
+ """Read and write raw binary files through xarray."""
2
+
3
+ from xarray_binfile._version import __version__, __version_tuple__
4
+ from xarray_binfile.read.entrypoint import RawBinaryEntrypoint
5
+ from xarray_binfile.read.file_metadata import ReadSpecs, ReadSpecsGetterProtocol
6
+
7
+ # The write accessor classes are xarray plugins: importing their module
8
+ # registers ``.binary_engine`` on Dataset and DataArray, which is the only
9
+ # supported way to reach them. The read backend, in contrast, is registered
10
+ # automatically through the ``xarray.backends`` entry point.
11
+ from xarray_binfile.write import accessor as _accessor
12
+ from xarray_binfile.write.file_metadata import WriteSpecs, WriteSpecsGetterProtocol
13
+
14
+ __all__ = [
15
+ "RawBinaryEntrypoint",
16
+ "ReadSpecs",
17
+ "ReadSpecsGetterProtocol",
18
+ "WriteSpecs",
19
+ "WriteSpecsGetterProtocol",
20
+ "__version__",
21
+ "__version_tuple__",
22
+ ]
@@ -0,0 +1,24 @@
1
+ # file generated by vcs-versioning
2
+ # don't change, don't track in version control
3
+ from __future__ import annotations
4
+
5
+ __all__ = [
6
+ "__version__",
7
+ "__version_tuple__",
8
+ "version",
9
+ "version_tuple",
10
+ "__commit_id__",
11
+ "commit_id",
12
+ ]
13
+
14
+ version: str
15
+ __version__: str
16
+ __version_tuple__: tuple[int | str, ...]
17
+ version_tuple: tuple[int | str, ...]
18
+ commit_id: str | None
19
+ __commit_id__: str | None
20
+
21
+ __version__ = version = '0.1.0'
22
+ __version_tuple__ = version_tuple = (0, 1, 0)
23
+
24
+ __commit_id__ = commit_id = None
File without changes
@@ -0,0 +1,2 @@
1
+ from xarray_binfile.read.entrypoint import RawBinaryEntrypoint
2
+ from xarray_binfile.read.file_metadata import ReadSpecs, ReadSpecsGetterProtocol
@@ -0,0 +1,201 @@
1
+ """
2
+ Defines a backend array for reading binary files in Xarray.
3
+ """
4
+
5
+ import numpy as np
6
+ import xarray as xr
7
+ from xarray.backends import BackendArray
8
+ from xarray.core import indexing
9
+
10
+ from xarray_binfile.read.file_metadata import ReadSpecs
11
+
12
+
13
+ def _is_coord_sliced(size: int, slice_spec: slice | int) -> bool:
14
+ """
15
+ Check if a coordinate is sliced.
16
+ Args:
17
+ size: Size of the coordinate.
18
+ slice_spec: Slice specification or integer index.
19
+ Returns:
20
+ True if the coordinate is sliced, False otherwise.
21
+ """
22
+ if not isinstance(slice_spec, slice):
23
+ return True
24
+ return any(
25
+ (
26
+ (slice_spec.start or 0) != 0,
27
+ (slice_spec.stop or size) != size,
28
+ (slice_spec.step or 1) != 1,
29
+ ),
30
+ )
31
+
32
+
33
+ class BinaryEngineBackendArray(BackendArray):
34
+ """
35
+ Backend array for reading binary files in Xarray.
36
+
37
+ Attributes:
38
+ metadata: Metadata describing the binary file.
39
+ dtype: Data type of the array.
40
+ shape: Shape of the array.
41
+ """
42
+
43
+ def __init__(self, metadata: ReadSpecs):
44
+ """
45
+ Initializes the backend array.
46
+
47
+ Args:
48
+ metadata: Metadata describing the binary file.
49
+
50
+ Raises:
51
+ ValueError: If the file size does not match the size expected
52
+ from the metadata shape and dtype.
53
+ FileNotFoundError: If the binary file does not exist.
54
+ """
55
+ self.metadata = metadata
56
+
57
+ # Attributes required by BackendArray
58
+ self.dtype = np.dtype(self.metadata.dtype)
59
+ self.shape = self.metadata.shape
60
+
61
+ self._validate_file_size()
62
+
63
+ def _validate_file_size(self) -> None:
64
+ """
65
+ Ensure the file size matches the metadata shape and dtype.
66
+
67
+ Raw binary files carry no metadata, so a size mismatch is the earliest
68
+ possible signal that the declared shape, dtype, or byte order is wrong.
69
+ Failing fast here avoids confusing downstream errors or silently
70
+ misread data.
71
+
72
+ Raises:
73
+ ValueError: If the file size does not match the expected size.
74
+ """
75
+ expected_bytes = int(np.prod(self.shape, dtype=np.int64)) * self.dtype.itemsize
76
+ actual_bytes = self.metadata.filepath.stat().st_size
77
+ if actual_bytes != expected_bytes:
78
+ error_message = (
79
+ f"Size mismatch for {self.metadata.filepath}: file has "
80
+ f"{actual_bytes} bytes, but the read specs expect "
81
+ f"{expected_bytes} bytes (shape {self.shape}, dtype {self.dtype}). "
82
+ "Check the shape, dtype, and byte order declared by the read specs getter."
83
+ )
84
+ raise ValueError(error_message)
85
+
86
+ def __getitem__(self, key: indexing.ExplicitIndexer) -> np.typing.ArrayLike:
87
+ """
88
+ Retrieves data from the array using explicit indexing.
89
+
90
+ Args:
91
+ key: Indexing key specifying the data to retrieve.
92
+
93
+ Returns:
94
+ The retrieved data.
95
+ """
96
+ return indexing.explicit_indexing_adapter(
97
+ key=key,
98
+ shape=self.metadata.shape,
99
+ indexing_support=indexing.IndexingSupport.BASIC,
100
+ raw_indexing_method=self._raw_indexing_method,
101
+ )
102
+
103
+ def _raw_indexing_method(self, key: tuple[slice | int, ...]) -> np.typing.ArrayLike:
104
+ """
105
+ Performs raw indexing on the binary file.
106
+
107
+ Args:
108
+ key: Tuple of slices or integers specifying the indices to read.
109
+
110
+ Returns:
111
+ The data read from the binary file.
112
+ """
113
+ with open(self.metadata.filepath, "rb") as file:
114
+ return self._read_binary_at_slices(file, key)
115
+
116
+ def _is_sliced(self, key: tuple[slice | int, ...]) -> bool:
117
+ """
118
+ Checks if the key is a slice of the original array.
119
+
120
+ Args:
121
+ key: Tuple of slices or integers specifying the indices to check.
122
+
123
+ Returns:
124
+ True if the key is a slice, False otherwise.
125
+ """
126
+
127
+ return any(
128
+ _is_coord_sliced(s, k)
129
+ for k, s in zip(key, self.metadata.shape, strict=True)
130
+ )
131
+
132
+ def _wrap_numpy_fromfile(self, file) -> np.typing.NDArray:
133
+ """
134
+ Reads the entire binary file into a NumPy array.
135
+
136
+ Args:
137
+ file: The binary file to read.
138
+
139
+ Returns:
140
+ The data read from the file.
141
+ """
142
+ return np.fromfile(
143
+ file, dtype=self.metadata.dtype, count=np.prod(self.metadata.shape)
144
+ ).reshape(self.metadata.shape)
145
+
146
+ def _wrap_numpy_memmap(
147
+ self, file, key: tuple[slice | int, ...]
148
+ ) -> np.typing.NDArray:
149
+ """
150
+ Reads a portion of the binary file using memory mapping.
151
+
152
+ Args:
153
+ file: The binary file to read.
154
+ key: Tuple of slices or integers specifying the indices to read.
155
+
156
+ Returns:
157
+ The data read from the file.
158
+ """
159
+ memory_map = np.memmap(
160
+ file,
161
+ dtype=self.metadata.dtype,
162
+ mode="r",
163
+ shape=self.metadata.shape,
164
+ order="C",
165
+ )
166
+ return np.asarray(memory_map[key]) # ensure we actually read the data
167
+
168
+ def _read_binary_at_slices(
169
+ self, file, key: tuple[slice | int, ...]
170
+ ) -> np.typing.NDArray:
171
+ """
172
+ Reads a binary file at specific locations based on the key tuple of slices.
173
+
174
+ Args:
175
+ file: The binary file to read.
176
+ key: Tuple of slices or integers specifying the indices to read.
177
+
178
+ Returns:
179
+ The array data read from the file at the specified slices.
180
+ """
181
+ if self._is_sliced(key):
182
+ return self._wrap_numpy_memmap(file, key)
183
+ return self._wrap_numpy_fromfile(file)
184
+
185
+ def get_xarray_dataset(self) -> xr.Dataset:
186
+ """
187
+ Converts the backend array to an Xarray Dataset.
188
+
189
+ Returns:
190
+ The Xarray Dataset representation of the backend array.
191
+ """
192
+ return xr.Dataset(
193
+ data_vars={
194
+ self.metadata.name: (
195
+ self.metadata.dims,
196
+ indexing.LazilyIndexedArray(self),
197
+ )
198
+ },
199
+ coords=self.metadata.coords,
200
+ attrs=self.metadata.attrs,
201
+ )
@@ -0,0 +1,74 @@
1
+ """
2
+ Backend for reading binary files in Xarray.
3
+
4
+ References:
5
+ * https://docs.xarray.dev/en/latest/internals/how-to-add-new-backend.html
6
+ * https://github.com/pydata/xarray/discussions/6406
7
+ """
8
+
9
+ import os
10
+ from collections.abc import Iterable
11
+ from pathlib import Path
12
+ from typing import Any
13
+
14
+ from xarray import Dataset
15
+ from xarray.backends import BackendEntrypoint
16
+
17
+ from xarray_binfile.read.array import BinaryEngineBackendArray
18
+ from xarray_binfile.read.file_metadata import ReadSpecsGetterProtocol
19
+
20
+
21
+ class RawBinaryEntrypoint(BackendEntrypoint):
22
+ """
23
+ Backend entry point for reading binary files in Xarray.
24
+
25
+ Attributes:
26
+ open_dataset_parameters: Parameters accepted by the `open_dataset` method.
27
+ description: Description of the backend.
28
+ url: URL to the backend documentation.
29
+ """
30
+
31
+ open_dataset_parameters = ("filename_or_obj", "read_specs_getter", "drop_variables")
32
+ description = "Read and write raw binary files using the familiar interface from the Xarray library."
33
+ url = "https://docs.fschuch.com/xarray-binfile/"
34
+
35
+ def open_dataset( # type: ignore[override]
36
+ self,
37
+ filename_or_obj: str | os.PathLike[Any],
38
+ *,
39
+ read_specs_getter: ReadSpecsGetterProtocol,
40
+ drop_variables: str | Iterable[str] | None = None,
41
+ ) -> Dataset:
42
+ """
43
+ Open a dataset from a binary file.
44
+
45
+ Args:
46
+ filename_or_obj: Path to the binary file or a file-like object.
47
+ read_specs_getter: A callable that generates read specifications for the binary file.
48
+ drop_variables: Variables to drop from the dataset. Defaults to None.
49
+
50
+ Returns:
51
+ The opened Xarray dataset.
52
+
53
+ Raises:
54
+ ValueError: If `filename_or_obj` is not a valid file path.
55
+ ValueError: If there is an error reading the metadata from the file path.
56
+ """
57
+ try:
58
+ file_path = Path(filename_or_obj)
59
+ except TypeError as err:
60
+ error_message = f"Expected a file path or file-like object, but got: {filename_or_obj!r}"
61
+ raise ValueError(error_message) from err
62
+ try:
63
+ file_metadata = read_specs_getter(path=file_path)
64
+ except Exception as err:
65
+ error_message = f"Error reading metadata from {file_path}: {err}"
66
+ raise ValueError(error_message) from err
67
+ if (
68
+ isinstance(drop_variables, str) and file_metadata.name == drop_variables
69
+ ) or (
70
+ isinstance(drop_variables, Iterable)
71
+ and file_metadata.name in drop_variables
72
+ ):
73
+ return Dataset()
74
+ return BinaryEngineBackendArray(metadata=file_metadata).get_xarray_dataset()
@@ -0,0 +1,86 @@
1
+ """
2
+ Defines metadata structures and protocols for reading binary files.
3
+ """
4
+
5
+ from dataclasses import dataclass
6
+ from functools import cached_property
7
+ from pathlib import Path
8
+ from typing import Protocol
9
+
10
+ from xarray_binfile.typing import AttributesLike, CoordsLike, DTypeLike
11
+
12
+
13
+ @dataclass(frozen=True)
14
+ class ReadSpecs:
15
+ """
16
+ Immutable metadata contract for interpreting one binary file.
17
+
18
+ A raw binary file does not carry enough semantic information to be decoded
19
+ safely by itself. This object provides that missing context so the backend
20
+ can expose the data as a labeled xarray variable.
21
+
22
+ Notes:
23
+ - Dimension order is inferred from ``coords`` key order.
24
+ - ``shape`` is inferred from coordinate lengths.
25
+ - ``dtype`` should be explicit about byte order when portability matters
26
+ (for example ``"<f4"`` for little-endian float32).
27
+
28
+ Attributes:
29
+ filepath: Path to the binary file.
30
+ dtype: Data type of the binary file.
31
+ coords: Coordinates of the data in the binary file.
32
+ name: Name of the dataset or variable.
33
+ attrs: Additional attributes for the dataset or variable.
34
+ """
35
+
36
+ filepath: Path
37
+ dtype: DTypeLike
38
+ coords: CoordsLike
39
+ name: str
40
+ attrs: AttributesLike | None = None
41
+
42
+ @cached_property
43
+ def shape(self) -> tuple[int, ...]:
44
+ """
45
+ Gets the shape of the data based on the coordinates.
46
+
47
+ Returns:
48
+ Shape of the data.
49
+ """
50
+ return tuple(len(i) for i in self.coords.values())
51
+
52
+ @cached_property
53
+ def dims(self) -> tuple[str, ...]:
54
+ """
55
+ Gets the dimension names of the data.
56
+
57
+ Returns:
58
+ Dimension names.
59
+ """
60
+ return tuple(self.coords.keys())
61
+
62
+
63
+ class ReadSpecsGetterProtocol(Protocol):
64
+ """
65
+ Structural protocol for read spec getter implementations.
66
+
67
+ Any callable matching ``(path: Path) -> ReadSpecs`` can be used as a read
68
+ specs getter. No inheritance is required.
69
+
70
+ Typical responsibilities:
71
+ - Parse filename conventions.
72
+ - Resolve variable identity (for example variable name, step index).
73
+ - Provide explicit dtype, coordinates, and optional attrs.
74
+ """
75
+
76
+ def __call__(self, path: Path) -> ReadSpecs:
77
+ """
78
+ Generate read specifications for a binary file.
79
+
80
+ Args:
81
+ path: Path to the binary file.
82
+
83
+ Returns:
84
+ The metadata required by the binary backend to decode ``path``.
85
+ """
86
+ ...
@@ -0,0 +1,2 @@
1
+ from xarray_binfile.tutorial.dataset_generator import DatasetGenerator
2
+ from xarray_binfile.tutorial.file_metadata import FileSpecsGetter
@@ -0,0 +1,87 @@
1
+ """
2
+ Provides a utility for generating xarray Datasets with random data.
3
+ """
4
+
5
+ from collections.abc import Iterator
6
+ from dataclasses import dataclass
7
+ from pathlib import Path
8
+
9
+ import numpy as np
10
+ import xarray as xr
11
+
12
+ from xarray_binfile.read.file_metadata import ReadSpecs, ReadSpecsGetterProtocol
13
+
14
+
15
+ @dataclass(frozen=True)
16
+ class DatasetGenerator:
17
+ """
18
+ Utility that generates synthetic datasets from read specs metadata.
19
+
20
+ This helper is intended for tutorials and tests where you want deterministic
21
+ mock data that follows the same metadata conventions expected by the backend.
22
+
23
+ Attributes:
24
+ read_specs_getter: A callable that generates read specifications for binary files.
25
+ random_generator: A random number generator for creating random data.
26
+ """
27
+
28
+ read_specs_getter: ReadSpecsGetterProtocol
29
+ random_generator = np.random.Generator(np.random.PCG64(1234))
30
+
31
+ def _get_numpy_array(self, metadata: ReadSpecs) -> np.ndarray:
32
+ """
33
+ Generate a random NumPy array matching ``metadata`` shape and dtype.
34
+
35
+ Args:
36
+ metadata: Metadata describing the array.
37
+
38
+ Returns:
39
+ A random NumPy array.
40
+ """
41
+ return self.random_generator.random(size=metadata.shape, dtype=metadata.dtype)
42
+
43
+ def _get_xarray_array(self, metadata: ReadSpecs) -> xr.DataArray:
44
+ """
45
+ Generate a random xarray DataArray matching ``metadata``.
46
+
47
+ Args:
48
+ metadata: Metadata describing the array.
49
+
50
+ Returns:
51
+ A random xarray DataArray.
52
+ """
53
+ return xr.DataArray(
54
+ data=self._get_numpy_array(metadata),
55
+ coords=metadata.coords,
56
+ attrs=metadata.attrs,
57
+ )
58
+
59
+ def _get_dataset(self, metadata: ReadSpecs) -> xr.Dataset:
60
+ """
61
+ Wrap the generated DataArray into a single-variable Dataset.
62
+
63
+ Args:
64
+ metadata: Metadata describing the dataset.
65
+
66
+ Returns:
67
+ A random xarray Dataset.
68
+ """
69
+ return self._get_xarray_array(metadata).to_dataset(name=metadata.name)
70
+
71
+ def __call__(self, iter_filepath: Iterator[Path]) -> xr.Dataset:
72
+ """
73
+ Generate and merge datasets for the provided file paths.
74
+
75
+ Each path is converted to ``ReadSpecs`` using ``read_specs_getter``.
76
+ A synthetic variable is generated for each metadata object, and all
77
+ variables are merged into one Dataset.
78
+
79
+ Args:
80
+ iter_filepath: An iterator over file paths.
81
+
82
+ Returns:
83
+ A merged xarray Dataset.
84
+ """
85
+ metadata = map(self.read_specs_getter, iter_filepath)
86
+ datasets = map(self._get_dataset, metadata)
87
+ return xr.merge(datasets, join="outer", compat="no_conflicts")
@@ -0,0 +1,114 @@
1
+ """
2
+ Defines utilities for generating file metadata for reading and writing binary files.
3
+ """
4
+
5
+ import re
6
+ from collections.abc import Iterator
7
+ from dataclasses import dataclass
8
+ from functools import cached_property
9
+ from pathlib import Path
10
+
11
+ import numpy as np
12
+ from xarray import DataArray
13
+
14
+ from xarray_binfile.read.file_metadata import ReadSpecs
15
+ from xarray_binfile.typing import ArrayLike, DTypeLike
16
+ from xarray_binfile.write.file_metadata import WriteSpecs
17
+
18
+
19
+ @dataclass(frozen=True)
20
+ class FileSpecsGetter:
21
+ """
22
+ Reference implementation of both read and write spec getter protocols.
23
+
24
+ This helper is intended for tutorials, tests, and simple projects with a
25
+ filename convention that encodes variable name and step index.
26
+
27
+ It is intentionally narrow: the default regex expects time-indexed files
28
+ (for example ``ux-0001.bin``). For mixed layouts that include static files
29
+ (for example ``epsi.bin``), implement a custom getter that still follows
30
+ the same read/write protocol contracts.
31
+
32
+ It implements:
33
+ - ``reader(path) -> ReadSpecs``
34
+ - ``writer(data_array) -> Iterator[WriteSpecs]``
35
+
36
+ Attributes:
37
+ base_coords: Base coordinates for the data.
38
+ dtype: Data type of the binary file. Defaults to np.float64.
39
+ filename_template: Template for generating filenames.
40
+ filename_regex: Regular expression for parsing filenames.
41
+ """
42
+
43
+ base_coords: dict[str, ArrayLike]
44
+ dtype: DTypeLike = np.float64
45
+ filename_template: str = "{name}-{digits:04}.bin"
46
+ filename_regex: re.Pattern = re.compile(r"(?P<name>\w+)-(?P<digits>\d{4})\.bin")
47
+
48
+ def reader(self, path: Path) -> ReadSpecs:
49
+ """
50
+ Build ``ReadSpecs`` from a file path.
51
+
52
+ The implementation parses ``path.name`` using ``filename_regex`` and
53
+ appends a single-value ``time`` coordinate based on the extracted step
54
+ number.
55
+
56
+ Args:
57
+ path: Path to the binary file.
58
+
59
+ Returns:
60
+ ReadSpecs: Metadata for reading the binary file.
61
+
62
+ Raises:
63
+ ValueError: If the filename does not match the expected pattern.
64
+ """
65
+ match = self.filename_regex.match(path.name)
66
+ if not match:
67
+ error_message = f"Invalid filename: {path.name}"
68
+ raise ValueError(error_message)
69
+
70
+ name, digits = match.groups()
71
+ time = np.array([int(digits)], dtype=np.int64)
72
+
73
+ return ReadSpecs(
74
+ filepath=path.resolve(),
75
+ dtype=self.dtype,
76
+ coords=self.base_coords | {"time": time},
77
+ name=name,
78
+ )
79
+
80
+ def writer(self, data_array: DataArray) -> Iterator[WriteSpecs]:
81
+ """
82
+ Yield ``WriteSpecs`` for one DataArray.
83
+
84
+ The default behavior writes one file per time coordinate value using
85
+ ``filename_template`` and transposes each slice to ``base_coords``
86
+ dimension order before serialization. Each slice is cast to ``dtype``
87
+ on write, so files always round-trip with :meth:`reader`.
88
+
89
+ Args:
90
+ data_array: The data array to generate write specifications for.
91
+
92
+ Returns:
93
+ An iterator over write specifications.
94
+ """
95
+ for time in data_array.coords["time"]:
96
+ yield WriteSpecs(
97
+ filename=self.filename_template.format(
98
+ name=data_array.name, digits=int(time)
99
+ ),
100
+ sub_array=data_array.sel(time=time).transpose(
101
+ *self._base_dims, missing_dims="raise"
102
+ ),
103
+ dtype=self.dtype,
104
+ )
105
+
106
+ @cached_property
107
+ def _base_dims(self) -> tuple[str, ...]:
108
+ """
109
+ Return the canonical base dimension order.
110
+
111
+ Returns:
112
+ A tuple of base dimension names.
113
+ """
114
+ return tuple(self.base_coords.keys())
@@ -0,0 +1,12 @@
1
+ """Type hints shared across the xarray_binfile package."""
2
+
3
+ import typing
4
+
5
+ import numpy.typing
6
+
7
+ # TODO: Type aliases could look better on the docs https://github.com/sphinx-doc/sphinx/issues/10785#issuecomment-1897551241
8
+
9
+ ArrayLike = numpy.typing.ArrayLike
10
+ DTypeLike = numpy.typing.DTypeLike
11
+ AttributesLike: typing.TypeAlias = typing.Mapping[typing.Any, typing.Any]
12
+ CoordsLike: typing.TypeAlias = typing.Mapping[str, numpy.typing.ArrayLike]
@@ -0,0 +1,2 @@
1
+ from xarray_binfile.write.accessor import BinaryEngineDataArray, BinaryEngineDataset
2
+ from xarray_binfile.write.file_metadata import WriteSpecs, WriteSpecsGetterProtocol
@@ -0,0 +1,117 @@
1
+ """
2
+ Provides accessors for writing xarray Dataset and DataArray objects to binary files.
3
+ """
4
+
5
+ import os
6
+ import tempfile
7
+ from pathlib import Path
8
+
9
+ import xarray as xr
10
+
11
+ from xarray_binfile.write.file_metadata import WriteSpecsGetterProtocol
12
+
13
+
14
+ @xr.register_dataset_accessor("binary_engine")
15
+ class BinaryEngineDataset:
16
+ """
17
+ An accessor with extra utilities for xarray.Dataset.
18
+ """
19
+
20
+ def __init__(self, data_set: xr.Dataset):
21
+ """
22
+ Initializes the BinaryEngineDataset accessor.
23
+
24
+ Args:
25
+ data_set: The dataset to attach the accessor to.
26
+ """
27
+ self._data_set = data_set
28
+
29
+ def to_file(
30
+ self,
31
+ write_specs_getter: WriteSpecsGetterProtocol,
32
+ directory: Path | None = None,
33
+ ) -> None:
34
+ """
35
+ Writes the dataset to binary files.
36
+
37
+ Every data variable is delegated to
38
+ :meth:`BinaryEngineDataArray.to_file`, so the same eager, atomic,
39
+ whole-file write semantics apply: each output file is fully
40
+ materialized in memory (triggering a Dask compute for lazy variables),
41
+ written in a single pass, and moved into place only once complete.
42
+ See that method for guidance on sizing files and on alternatives when
43
+ streaming or partial writes are needed.
44
+
45
+ Args:
46
+ write_specs_getter: A callable that generates write specifications for the data arrays.
47
+ directory: The directory where the binary files will be written. Defaults to the current working directory.
48
+ """
49
+ for data_array in self._data_set.data_vars.values():
50
+ data_array.binary_engine.to_file(write_specs_getter, directory)
51
+
52
+
53
+ @xr.register_dataarray_accessor("binary_engine")
54
+ class BinaryEngineDataArray:
55
+ """
56
+ An accessor with extra utilities for xarray.DataArray.
57
+ """
58
+
59
+ def __init__(self, data_array: xr.DataArray):
60
+ """
61
+ Initializes the BinaryEngineDataArray accessor.
62
+
63
+ Args:
64
+ data_array: The data array to attach the accessor to.
65
+ """
66
+ self._data_array = data_array
67
+
68
+ def to_file(
69
+ self,
70
+ write_specs_getter: WriteSpecsGetterProtocol,
71
+ directory: Path | None = None,
72
+ ) -> None:
73
+ """
74
+ Writes the data array to binary files.
75
+
76
+ Writes are eager and whole-file only. For each write specification,
77
+ the entire ``sub_array`` is loaded into memory (triggering a Dask
78
+ compute for lazy data) and the target file is written in full, in a
79
+ single pass. There is no partial, appending, or resuming write mode:
80
+ re-writing a file always replaces its whole content instead of trying
81
+ to guess or patch existing bytes, which avoids leaving files in a
82
+ partially-updated, corrupted state.
83
+
84
+ Writes are also atomic per file: the bytes are first serialized into
85
+ a unique temporary directory created inside the destination directory
86
+ (so the final move stays on the same filesystem), and each file is
87
+ moved to its final path with :func:`os.replace` only once it is
88
+ complete. An interrupted write never leaves a truncated file at the
89
+ destination, and the temporary directory is removed automatically.
90
+
91
+ Plan the write specifications so that every individual output file
92
+ fits comfortably in memory, for example by splitting the array into
93
+ one file per time step. If you need streaming, incremental, or
94
+ partial writes, prefer one of the other file formats supported by
95
+ xarray, such as NetCDF or Zarr.
96
+
97
+ Each file is written with the in-memory dtype and native byte order,
98
+ unless the write specification sets ``dtype``, in which case the
99
+ values are cast right before serialization.
100
+
101
+ Args:
102
+ write_specs_getter: A callable that generates write specifications for the data array.
103
+ directory: The directory where the binary files will be written. Defaults to the current working directory.
104
+ """
105
+ _directory = directory or Path.cwd()
106
+ with tempfile.TemporaryDirectory(
107
+ dir=_directory, prefix=".binary_engine-"
108
+ ) as temporary_directory:
109
+ for details in write_specs_getter(self._data_array):
110
+ new_type = (
111
+ details.dtype
112
+ if details.dtype is not None
113
+ else details.sub_array.dtype
114
+ )
115
+ temporary_file = Path(temporary_directory) / details.filename
116
+ details.sub_array.values.astype(new_type).tofile(temporary_file)
117
+ os.replace(temporary_file, _directory / details.filename)
@@ -0,0 +1,62 @@
1
+ """
2
+ Defines metadata structures and protocols for writing binary files.
3
+ """
4
+
5
+ from collections.abc import Iterator
6
+ from typing import NamedTuple, Protocol
7
+
8
+ import xarray as xr
9
+
10
+ from xarray_binfile.typing import DTypeLike
11
+
12
+
13
+ class WriteSpecs(NamedTuple):
14
+ """
15
+ Immutable write instruction for one output binary file.
16
+
17
+ Each item defines both the output filename and the exact DataArray slice to
18
+ serialize into that file.
19
+
20
+ Attributes:
21
+ filename: The name of the binary file.
22
+ sub_array: The portion of the DataArray to be written.
23
+ dtype: Optional on-disk data type. When set, the sub-array is cast to
24
+ this dtype (including byte order, for example ``"<f4"``) right
25
+ before serialization. When ``None``, the in-memory dtype and
26
+ native byte order are written as-is, so make sure they match what
27
+ your read specs getter declares.
28
+ """
29
+
30
+ filename: str
31
+ sub_array: xr.DataArray
32
+ dtype: DTypeLike | None = None
33
+
34
+
35
+ class WriteSpecsGetterProtocol(Protocol):
36
+ """
37
+ Structural protocol for write spec getter implementations.
38
+
39
+ Any callable matching
40
+ ``(data_array: xr.DataArray) -> Iterator[WriteSpecs]`` can be used as a
41
+ write specs getter. No inheritance is required.
42
+
43
+ Typical responsibilities:
44
+ - Define filename conventions.
45
+ - Split a DataArray into one or more per-file slices.
46
+ - Ensure each slice is transposed to the expected on-disk dimension order.
47
+ - Keep each per-file slice small enough to fit in memory, since every
48
+ ``sub_array`` is fully materialized (Dask compute) before its file
49
+ is written in a single pass.
50
+ """
51
+
52
+ def __call__(self, data_array: xr.DataArray) -> Iterator[WriteSpecs]:
53
+ """
54
+ Generate write instructions for a DataArray.
55
+
56
+ Args:
57
+ data_array: The data array for which to generate write specifications.
58
+
59
+ Returns:
60
+ An iterator over per-file write instructions.
61
+ """
62
+ ...
@@ -0,0 +1,101 @@
1
+ Metadata-Version: 2.4
2
+ Name: xarray-binfile
3
+ Version: 0.1.0
4
+ Summary: Read and write raw binary files using the familiar interface from the Xarray library
5
+ Project-URL: Source, https://github.com/fschuch/xarray-binfile
6
+ Project-URL: Tracker, https://github.com/fschuch/xarray-binfile/issues
7
+ Project-URL: Changelog, https://docs.fschuch.com/xarray-binfile/references/what-is-new.html
8
+ Project-URL: Documentation, https://docs.fschuch.com/xarray-binfile/
9
+ Author-email: fschuch <me@fschuch.com>
10
+ License-Expression: MIT
11
+ License-File: LICENSE
12
+ Classifier: Development Status :: 5 - Production/Stable
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Programming Language :: Python :: 3.14
22
+ Classifier: Programming Language :: Python :: 3.15
23
+ Requires-Python: >=3.10
24
+ Requires-Dist: dask>=2024.0.0
25
+ Requires-Dist: numpy>=1.24
26
+ Requires-Dist: xarray>=2024.1.0
27
+ Provides-Extra: benchmarks
28
+ Requires-Dist: pytest-memray>=1.7.0; (sys_platform != 'win32') and extra == 'benchmarks'
29
+ Provides-Extra: docs
30
+ Requires-Dist: graphviz>=0.20.3; extra == 'docs'
31
+ Requires-Dist: jupyter-book==1.0.4.post1; extra == 'docs'
32
+ Requires-Dist: matplotlib>=3.9.0; extra == 'docs'
33
+ Requires-Dist: sphinx-autobuild==2024.10.3; extra == 'docs'
34
+ Requires-Dist: sphinx-github-changelog==1.7.2; extra == 'docs'
35
+ Requires-Dist: sphinx==7.4.7; extra == 'docs'
36
+ Provides-Extra: tests
37
+ Requires-Dist: coverage[toml]>=7.5.3; extra == 'tests'
38
+ Requires-Dist: hypothesis>=6.130.4; extra == 'tests'
39
+ Requires-Dist: pre-commit>=3.5.0; extra == 'tests'
40
+ Requires-Dist: pytest-benchmark>=5.1.0; extra == 'tests'
41
+ Requires-Dist: pytest-cov>=5.0.0; extra == 'tests'
42
+ Requires-Dist: pytest>=8.2.2; extra == 'tests'
43
+ Provides-Extra: tests-extra
44
+ Requires-Dist: pytest-randomly==4.1.0; extra == 'tests-extra'
45
+ Requires-Dist: pytest-rerunfailures==16.3; extra == 'tests-extra'
46
+ Requires-Dist: pytest-xdist==3.8.0; extra == 'tests-extra'
47
+ Description-Content-Type: text/markdown
48
+
49
+ # Xarray-binfile
50
+
51
+ <p align="center">
52
+ <a href="https://github.com/fschuch/xarray-binfile"><img src="https://raw.githubusercontent.com/fschuch/xarray-binfile/refs/heads/main/docs/logo.png" alt="Xarray-binfile logo" width="320"></a>
53
+ </p>
54
+ <p align="center">
55
+ <em>Custom Xarray file engine to handle raw binary files</em>
56
+ </p>
57
+
58
+ ______________________________________________________________________
59
+
60
+ - QA:
61
+ [![CI](https://github.com/fschuch/xarray-binfile/actions/workflows/ci.yaml/badge.svg?branch=main)](https://github.com/fschuch/xarray-binfile/actions/workflows/ci.yaml)
62
+ [![CodeQL](https://github.com/fschuch/xarray-binfile/actions/workflows/github-code-scanning/codeql/badge.svg)](https://github.com/fschuch/xarray-binfile/actions/workflows/github-code-scanning/codeql)
63
+ [![pre-commit.ci status](https://results.pre-commit.ci/badge/github/fschuch/xarray-binfile/main.svg)](https://results.pre-commit.ci/latest/github/fschuch/xarray-binfile/main)
64
+ [![Quality Gate Status](https://sonarcloud.io/api/project_badges/measure?project=fschuch_xarray-binfile&metric=alert_status)](https://sonarcloud.io/summary/new_code?id=fschuch_xarray-binfile)
65
+ [![Coverage](https://sonarcloud.io/api/project_badges/measure?project=fschuch_xarray-binfile&metric=coverage)](https://sonarcloud.io/summary/new_code?id=fschuch_xarray-binfile)
66
+ [![CodeFactor](https://www.codefactor.io/repository/github/fschuch/xarray-binfile/badge)](https://www.codefactor.io/repository/github/fschuch/xarray-binfile)
67
+
68
+ - Docs:
69
+ [![Docs](https://github.com/fschuch/xarray-binfile/actions/workflows/docs.yaml/badge.svg?branch=main)](https://docs.fschuch.com/xarray-binfile)
70
+
71
+ - Package:
72
+ [![PyPI - Version](https://img.shields.io/pypi/v/xarray-binfile.svg?logo=pypi&label=PyPI)](https://pypi.org/project/xarray-binfile/)
73
+ [![PyPI - Python Version](https://img.shields.io/pypi/pyversions/xarray-binfile.svg?logo=python&label=Python)](https://pypi.org/project/xarray-binfile/)
74
+
75
+ - Meta:
76
+ [![Wizard Template](https://img.shields.io/badge/Wizard-Template-%23447CAA)](https://github.com/fschuch/wizard-template)
77
+ [![Checked with mypy](https://www.mypy-lang.org/static/mypy_badge.svg)](https://mypy-lang.org/)
78
+ [![Hatch project](https://img.shields.io/badge/%F0%9F%A5%9A-Hatch-4051b5.svg)](https://github.com/pypa/hatch)
79
+ [![Ruff](https://img.shields.io/endpoint?url=https://raw.githubusercontent.com/astral-sh/ruff/main/assets/badge/v2.json)](https://github.com/astral-sh/ruff)
80
+ ![GitHub License](https://img.shields.io/github/license/fschuch/xarray-binfile?color=blue)
81
+ [![EffVer Versioning](https://img.shields.io/badge/version_scheme-EffVer-0097a7)](https://jacobtomlinson.dev/effver)
82
+
83
+ ______________________________________________________________________
84
+
85
+ ## Overview
86
+
87
+ Xarray-binfile is a Python package that integrates raw binary files with xarray. You provide the metadata that describes each binary file, and the package exposes those files through xarray datasets and data arrays with optional lazy chunking through Dask.
88
+
89
+ The package was first designed around binary outputs from the Fortran framework [2DECOMP&FFT](https://github.com/2decomp-fft/2decomp-fft) and the CFD solver [Xcompact3d](https://github.com/xcompact3d/Incompact3d), which is built on top of it. It is not limited to those ecosystems and can be used with any raw binary convention once the corresponding metadata mapping is provided.
90
+
91
+ It is aimed at workflows where raw binary files are part of an existing convention, such as simulation outputs or NumPy `.tofile()` dumps. Once opened, the data can be indexed, reduced, plotted, and computed with the usual xarray and Dask APIs.
92
+
93
+ See the documentation for end-to-end examples covering lazy loading, Dask-backed computations, task-graph visualization, writing derived results back to `.bin` files, parallel and larger-than-memory workflows, benchmarking on your own data, batch dtype conversion, probe time-series export to CSV, and quick matplotlib plots.
94
+
95
+ > [!CAUTION]
96
+ > Raw binary files are machine- and convention-dependent. Differences such as little-endian vs big-endian byte order, word size, record layout, and memory ordering can change how bytes should be interpreted. Always verify dtype and endianness when sharing files across machines or toolchains.
97
+
98
+ ## Copyright and License
99
+
100
+ © 2025 [Felipe N. Schuch](https://github.com/fschuch).
101
+ All content is under [MIT License](https://github.com/fschuch/xarray-binfile/blob/main/LICENSE).
@@ -0,0 +1,19 @@
1
+ xarray_binfile/__init__.py,sha256=w4kVH6JRQGV653J8o43o3wPA93Fp5tYbIHiLC4XzE-s,874
2
+ xarray_binfile/_version.py,sha256=n_5vdJsPNu7wZ57LGuRL585uvll-hiuvZUBWzdG0RQU,520
3
+ xarray_binfile/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
4
+ xarray_binfile/typing.py,sha256=oSm8QBMfBzXSfdViJixNYfMYEqARJB2N1FjejU_BXM8,441
5
+ xarray_binfile/read/__init__.py,sha256=Rmvab90D3K9OcJSp6jJUulSEcg_0QbKQaUUVWM20wqE,144
6
+ xarray_binfile/read/array.py,sha256=I2YRptyQoZhZRdZwqz9kd4zGqUyaWRMoKCb3hlnRcDg,6300
7
+ xarray_binfile/read/entrypoint.py,sha256=Kz_8nIzmbasJJ25BU26_eD6bGhMn-OYMq9lDMgR_Yyc,2719
8
+ xarray_binfile/read/file_metadata.py,sha256=2mCWxEo0cCSuXiJPCzXlb7052KwcWuXmXS1ycyuaHec,2512
9
+ xarray_binfile/tutorial/__init__.py,sha256=1XMAc-gELAxmiHBDaSX9cwnhRm3OEISvF-_vSviiBzM,137
10
+ xarray_binfile/tutorial/dataset_generator.py,sha256=efDd8XifxA0DMu6mjGq7qcjl86pbwGhrG70fMGM6K6c,2750
11
+ xarray_binfile/tutorial/file_metadata.py,sha256=fIeZicVPuRsSopmLl5dw4THnq-788v4VKTF5bFszkls,3843
12
+ xarray_binfile/write/__init__.py,sha256=zckfWb6NXP8OE7r38Iz2r3crr1_rL0OOM5kpQ-RmD-Q,169
13
+ xarray_binfile/write/accessor.py,sha256=KpM6oEj_i0gtC2qHYEweFy2oNdcLvOiRsRlgzzDWcAY,4665
14
+ xarray_binfile/write/file_metadata.py,sha256=XrB82ZzWBBDIrzS59BguBQbjmHmfhCjmQ1LXhuE_dSA,2022
15
+ xarray_binfile-0.1.0.dist-info/METADATA,sha256=IplVKBOkNgoJi9NgN9XmcGfoCbp9WeTzcWxSsv8bYvg,6685
16
+ xarray_binfile-0.1.0.dist-info/WHEEL,sha256=mffPy8wBnZQn2VnJUU5jE99KsxaSfiyMHV9Yt0aLVxs,87
17
+ xarray_binfile-0.1.0.dist-info/entry_points.txt,sha256=SPu6_MRtBw578qcwsBcjsaQeW-P-sutg3HfHeKFzA0k,79
18
+ xarray_binfile-0.1.0.dist-info/licenses/LICENSE,sha256=hsXQT5P2j7Kz8HdmUWgf7uiJ1FCY_EI-3Gxht-DVITU,1073
19
+ xarray_binfile-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.30.1
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,2 @@
1
+ [xarray.backends]
2
+ binfile = xarray_binfile.read.entrypoint:RawBinaryEntrypoint
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2025 Felipe N. Schuch
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.