xarray-binfile 0.2.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/PKG-INFO +1 -1
  2. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/docs/_toc.yml +1 -0
  3. xarray_binfile-0.3.0/docs/tutorials/stacked-variables.ipynb +311 -0
  4. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/docs/usage/spec-getter-protocols.md +52 -7
  5. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/src/xarray_binfile/_version.py +2 -2
  6. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/src/xarray_binfile/conventions/__init__.py +12 -1
  7. xarray_binfile-0.3.0/src/xarray_binfile/conventions/base.py +211 -0
  8. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/src/xarray_binfile/conventions/filename_pattern.py +92 -15
  9. xarray_binfile-0.3.0/src/xarray_binfile/conventions/folders.py +395 -0
  10. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/src/xarray_binfile/conventions/getters.py +219 -88
  11. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/src/xarray_binfile/conventions/layout.py +15 -2
  12. xarray_binfile-0.3.0/src/xarray_binfile/conventions/stacking.py +319 -0
  13. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/src/xarray_binfile/read/array.py +7 -3
  14. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/src/xarray_binfile/read/file_metadata.py +14 -1
  15. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/src/xarray_binfile/typing.py +3 -0
  16. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/src/xarray_binfile/write/accessor.py +34 -6
  17. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/src/xarray_binfile/write/file_metadata.py +6 -1
  18. xarray_binfile-0.3.0/tests/integration/test_xcompact3d_like.py +111 -0
  19. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/tests/unit/conventions/test_filename_pattern.py +91 -0
  20. xarray_binfile-0.3.0/tests/unit/conventions/test_folders.py +399 -0
  21. xarray_binfile-0.3.0/tests/unit/conventions/test_getters.py +501 -0
  22. xarray_binfile-0.3.0/tests/unit/conventions/test_stacking.py +214 -0
  23. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/tests/unit/read/test_array.py +82 -0
  24. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/tests/unit/write/test_accessor.py +31 -0
  25. xarray_binfile-0.2.0/src/xarray_binfile/conventions/folders.py +0 -129
  26. xarray_binfile-0.2.0/tests/unit/conventions/test_folders.py +0 -90
  27. xarray_binfile-0.2.0/tests/unit/conventions/test_getters.py +0 -206
  28. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/.github/dependabot.yml +0 -0
  29. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/.github/release.yml +0 -0
  30. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/.github/workflows/check-links.yaml +0 -0
  31. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/.github/workflows/ci.yaml +0 -0
  32. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/.github/workflows/docs.yaml +0 -0
  33. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/.gitignore +0 -0
  34. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/.pre-commit-config.yaml +0 -0
  35. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/.vscode/extensions.json +0 -0
  36. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/.vscode/settings.json +0 -0
  37. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/CITATION.cff +0 -0
  38. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/CODE_OF_CONDUCT.md +0 -0
  39. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/CONTRIBUTING.md +0 -0
  40. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/LICENSE +0 -0
  41. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/README.md +0 -0
  42. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/docs/_config.yml +0 -0
  43. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/docs/getting-started/about.md +0 -0
  44. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/docs/getting-started/installation.md +0 -0
  45. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/docs/intro.md +0 -0
  46. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/docs/logo.png +0 -0
  47. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/docs/references/api-reference.rst +0 -0
  48. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/docs/references/further-reading.md +0 -0
  49. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/docs/references/how-to-contribute.md +0 -0
  50. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/docs/references/what-is-new.md +0 -0
  51. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/docs/references.bib +0 -0
  52. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/docs/tutorials/benchmarking.ipynb +0 -0
  53. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/docs/tutorials/examples.ipynb +0 -0
  54. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/docs/tutorials/parallel-and-larger-than-memory.ipynb +0 -0
  55. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/docs/tutorials/read.ipynb +0 -0
  56. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/docs/tutorials/write.ipynb +0 -0
  57. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/pyproject.toml +0 -0
  58. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/src/xarray_binfile/__init__.py +0 -0
  59. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/src/xarray_binfile/conventions/protocol.py +0 -0
  60. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/src/xarray_binfile/py.typed +0 -0
  61. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/src/xarray_binfile/read/__init__.py +0 -0
  62. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/src/xarray_binfile/read/entrypoint.py +0 -0
  63. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/src/xarray_binfile/tutorial/__init__.py +0 -0
  64. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/src/xarray_binfile/tutorial/dataset_generator.py +0 -0
  65. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/src/xarray_binfile/tutorial/file_metadata.py +0 -0
  66. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/src/xarray_binfile/write/__init__.py +0 -0
  67. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/tests/__init__.py +0 -0
  68. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/tests/integration/__init__.py +0 -0
  69. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/tests/integration/test_conventions_roundtrip.py +0 -0
  70. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/tests/integration/test_indexing.py +0 -0
  71. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/tests/integration/test_load_dataset.py +0 -0
  72. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/tests/integration/test_roundtrip.py +0 -0
  73. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/tests/unit/__init__.py +0 -0
  74. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/tests/unit/conventions/__init__.py +0 -0
  75. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/tests/unit/conventions/test_layout.py +0 -0
  76. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/tests/unit/read/__init__.py +0 -0
  77. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/tests/unit/read/test_entrypoint.py +0 -0
  78. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/tests/unit/tutorial/__init__.py +0 -0
  79. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/tests/unit/tutorial/test_file_metadata.py +0 -0
  80. {xarray_binfile-0.2.0 → xarray_binfile-0.3.0}/tests/unit/write/__init__.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: xarray-binfile
3
- Version: 0.2.0
3
+ Version: 0.3.0
4
4
  Summary: Read and write raw binary files using the familiar interface from the Xarray library
5
5
  Project-URL: Source, https://github.com/fschuch/xarray-binfile
6
6
  Project-URL: Tracker, https://github.com/fschuch/xarray-binfile/issues
@@ -13,6 +13,7 @@ parts:
13
13
  - file: usage/spec-getter-protocols.md
14
14
  - file: tutorials/read.ipynb
15
15
  - file: tutorials/write.ipynb
16
+ - file: tutorials/stacked-variables.ipynb
16
17
  - file: tutorials/parallel-and-larger-than-memory.ipynb
17
18
  - file: tutorials/benchmarking.ipynb
18
19
  - file: tutorials/examples.ipynb
@@ -0,0 +1,311 @@
1
+ {
2
+ "cells": [
3
+ {
4
+ "cell_type": "markdown",
5
+ "id": "0",
6
+ "metadata": {},
7
+ "source": [
8
+ "# Stack And Split Variables Across Files\n",
9
+ "\n",
10
+ "Many solvers store a vector field as one file per component (`ux-0001.bin`, `uy-0001.bin`, `uz-0001.bin`) and a set of scalar fractions as numbered variables (`phi1-0001.bin`, `phi2-0001.bin`). In memory, you would rather work with one array `u` carrying a component dimension `i`, and one array `phi` carrying a fraction dimension `n`.\n",
11
+ "\n",
12
+ "xarray-binfile describes that encoding once, with a `VariableStack`, and uses it in both directions:\n",
13
+ "\n",
14
+ "- **split** on write: `u` with `i = [\"x\", \"y\", \"z\"]` becomes the variables `ux`, `uy` and `uz`, each written to its own files\n",
15
+ "- **stack** on read: `ux`, `uy` and `uz` found on disk are concatenated back into `u`\n",
16
+ "\n",
17
+ "Nothing about `u`, `x`, `y`, `z` or `phi` is hard-coded in the library: the dimension name, the name template and the allowed values all come from the declaration."
18
+ ]
19
+ },
20
+ {
21
+ "cell_type": "code",
22
+ "id": "1",
23
+ "metadata": {},
24
+ "execution_count": null,
25
+ "outputs": [],
26
+ "source": [
27
+ "import pathlib\n",
28
+ "import tempfile\n",
29
+ "\n",
30
+ "import numpy as np\n",
31
+ "import xarray as xr\n",
32
+ "\n",
33
+ "import xarray_binfile # noqa: F401 (registers the .binary_engine accessors)\n",
34
+ "from xarray_binfile.conventions import Layout, StepIndexedFiles, VariableStack\n",
35
+ "\n",
36
+ "directory_holder = tempfile.TemporaryDirectory()\n",
37
+ "directory = pathlib.Path(directory_holder.name)"
38
+ ]
39
+ },
40
+ {
41
+ "cell_type": "markdown",
42
+ "id": "2",
43
+ "metadata": {},
44
+ "source": [
45
+ "## Declare the stacks\n",
46
+ "\n",
47
+ "A `VariableStack` has three parts:\n",
48
+ "\n",
49
+ "- `dim`: the dimension that lives in the variable names rather than in the files\n",
50
+ "- `template`: how a variable name is built from the base `{name}` and the value of `dim`\n",
51
+ "- `values`: the values the dimension may take, in the order the stacked coordinate should follow\n",
52
+ "\n",
53
+ "The template is parsed with `FilenamePattern`, so integer values can carry zero-fill (`\"{name}{n:02d}\"`) and are converted back to `int` on read. `values` is required because templates without a separator would otherwise match almost any name: `pp` fits `\"{name}{i}\"` as `p` + `p`."
54
+ ]
55
+ },
56
+ {
57
+ "cell_type": "code",
58
+ "id": "3",
59
+ "metadata": {},
60
+ "execution_count": null,
61
+ "outputs": [],
62
+ "source": [
63
+ "velocity = VariableStack(\"i\", \"{name}{i}\", values=(\"x\", \"y\", \"z\"))\n",
64
+ "scalars = VariableStack(\"n\", \"{name}{n:d}\", values=range(1, 10), names=(\"phi\",))\n",
65
+ "velocity, scalars"
66
+ ]
67
+ },
68
+ {
69
+ "cell_type": "markdown",
70
+ "id": "4",
71
+ "metadata": {},
72
+ "source": [
73
+ "The optional `names` restricts stacking to given base names. Without it, a vorticity component written as `w3` would be picked up by `scalars` as `w` with `n = [3]`, which is harmless but probably not what you want.\n",
74
+ "\n",
75
+ "## Attach them to a convention\n",
76
+ "\n",
77
+ "Every shipped convention accepts `stacks`. On write, arrays are split in the order the stacks are listed, before the per-file split along `time`; on read, `open` rebuilds the stacked arrays by default."
78
+ ]
79
+ },
80
+ {
81
+ "cell_type": "code",
82
+ "id": "5",
83
+ "metadata": {},
84
+ "execution_count": null,
85
+ "outputs": [],
86
+ "source": [
87
+ "layout = Layout(\n",
88
+ " {\n",
89
+ " \"x\": np.linspace(0.0, 1.0, 8, dtype=np.float32),\n",
90
+ " \"y\": np.linspace(0.0, 2.0, 6, dtype=np.float32),\n",
91
+ " \"z\": np.linspace(0.0, 0.5, 4, dtype=np.float32),\n",
92
+ " },\n",
93
+ " dtype=np.float32,\n",
94
+ ")\n",
95
+ "convention = StepIndexedFiles(\n",
96
+ " layout,\n",
97
+ " pattern=\"{name}-{step:03d}.bin\",\n",
98
+ " time_step=0.25,\n",
99
+ " stacks=[velocity, scalars],\n",
100
+ ")\n",
101
+ "convention"
102
+ ]
103
+ },
104
+ {
105
+ "cell_type": "markdown",
106
+ "id": "6",
107
+ "metadata": {},
108
+ "source": [
109
+ "## Build a dataset with stacked arrays\n",
110
+ "\n",
111
+ "`u` carries `i`, `phi` carries `n`, and `pp` carries neither. All of them share the mesh and the `time` dimension."
112
+ ]
113
+ },
114
+ {
115
+ "cell_type": "code",
116
+ "id": "7",
117
+ "metadata": {},
118
+ "execution_count": null,
119
+ "outputs": [],
120
+ "source": [
121
+ "rng = np.random.default_rng(0)\n",
122
+ "time = np.arange(3) * 0.25\n",
123
+ "\n",
124
+ "\n",
125
+ "def field(**extra):\n",
126
+ " coords = {**extra, **layout.coords, \"time\": time}\n",
127
+ " shape = tuple(len(values) for values in coords.values())\n",
128
+ " return xr.DataArray(rng.random(shape, dtype=np.float32), coords=coords)\n",
129
+ "\n",
130
+ "\n",
131
+ "dataset = xr.Dataset(\n",
132
+ " {\n",
133
+ " \"u\": field(i=[\"x\", \"y\", \"z\"]),\n",
134
+ " \"phi\": field(n=[1, 2]),\n",
135
+ " \"pp\": field(),\n",
136
+ " }\n",
137
+ ")\n",
138
+ "dataset"
139
+ ]
140
+ },
141
+ {
142
+ "cell_type": "markdown",
143
+ "id": "8",
144
+ "metadata": {},
145
+ "source": [
146
+ "## Split on write\n",
147
+ "\n",
148
+ "Writing the dataset through the convention produces one file per component, per fraction and per step. `u` never appears on disk; `pp` is written as is."
149
+ ]
150
+ },
151
+ {
152
+ "cell_type": "code",
153
+ "id": "9",
154
+ "metadata": {},
155
+ "execution_count": null,
156
+ "outputs": [],
157
+ "source": [
158
+ "dataset.binary_engine.to_file(convention.writer, directory)\n",
159
+ "sorted(path.name for path in directory.glob(\"*.bin\"))"
160
+ ]
161
+ },
162
+ {
163
+ "cell_type": "markdown",
164
+ "id": "10",
165
+ "metadata": {},
166
+ "source": [
167
+ "The split happens before the file split, so the per-file slices are plain `(x, y, z)` arrays. You can inspect the plan without writing anything by iterating the writer:"
168
+ ]
169
+ },
170
+ {
171
+ "cell_type": "code",
172
+ "id": "11",
173
+ "metadata": {},
174
+ "execution_count": null,
175
+ "outputs": [],
176
+ "source": [
177
+ "[(spec.filename, spec.sub_array.dims) for spec in convention.writer(dataset[\"u\"])][:4]"
178
+ ]
179
+ },
180
+ {
181
+ "cell_type": "markdown",
182
+ "id": "12",
183
+ "metadata": {},
184
+ "source": [
185
+ "## Stack on read\n",
186
+ "\n",
187
+ "`open` lists the files that follow the pattern, opens them lazily with `xarray.open_mfdataset`, and applies the stacks. The result has the same variables as the original dataset, with `i` and `n` as proper coordinates."
188
+ ]
189
+ },
190
+ {
191
+ "cell_type": "code",
192
+ "id": "13",
193
+ "metadata": {},
194
+ "execution_count": null,
195
+ "outputs": [],
196
+ "source": [
197
+ "opened = convention.open(directory, chunks={\"time\": 1})\n",
198
+ "opened"
199
+ ]
200
+ },
201
+ {
202
+ "cell_type": "code",
203
+ "id": "14",
204
+ "metadata": {},
205
+ "execution_count": null,
206
+ "outputs": [],
207
+ "source": [
208
+ "for name in dataset.data_vars:\n",
209
+ " xr.testing.assert_allclose(\n",
210
+ " opened[name].transpose(*dataset[name].dims).load(), dataset[name]\n",
211
+ " )\n",
212
+ "print(\"round trip matches\")"
213
+ ]
214
+ },
215
+ {
216
+ "cell_type": "markdown",
217
+ "id": "15",
218
+ "metadata": {},
219
+ "source": [
220
+ "Stacking is lazy, so a Dask-backed dataset stays Dask-backed, and `i` and `n` can be used like any other coordinate:"
221
+ ]
222
+ },
223
+ {
224
+ "cell_type": "code",
225
+ "id": "16",
226
+ "metadata": {},
227
+ "execution_count": null,
228
+ "outputs": [],
229
+ "source": [
230
+ "speed = np.sqrt((opened[\"u\"] ** 2).sum(\"i\")).rename(\"speed\")\n",
231
+ "speed.isel(time=0, z=0).compute()"
232
+ ]
233
+ },
234
+ {
235
+ "cell_type": "markdown",
236
+ "id": "17",
237
+ "metadata": {},
238
+ "source": [
239
+ "## Opt out, or stack by hand\n",
240
+ "\n",
241
+ "Pass `stack=False` to see the variables exactly as they are on disk. The same declarations are available as functions for datasets opened another way: `VariableStack.stack` for one dimension, `stack_variables` for a sequence, and `split_variables` for the inverse."
242
+ ]
243
+ },
244
+ {
245
+ "cell_type": "code",
246
+ "id": "18",
247
+ "metadata": {},
248
+ "execution_count": null,
249
+ "outputs": [],
250
+ "source": [
251
+ "raw = convention.open(directory, stack=False)\n",
252
+ "sorted(raw.data_vars)"
253
+ ]
254
+ },
255
+ {
256
+ "cell_type": "code",
257
+ "id": "19",
258
+ "metadata": {},
259
+ "execution_count": null,
260
+ "outputs": [],
261
+ "source": [
262
+ "from xarray_binfile.conventions import split_variables, stack_variables\n",
263
+ "\n",
264
+ "restacked = stack_variables(raw, [velocity, scalars])\n",
265
+ "pieces = [piece.name for piece in split_variables(dataset[\"u\"], [velocity])]\n",
266
+ "sorted(restacked.data_vars), pieces"
267
+ ]
268
+ },
269
+ {
270
+ "cell_type": "markdown",
271
+ "id": "20",
272
+ "metadata": {},
273
+ "source": [
274
+ "## Guard rails\n",
275
+ "\n",
276
+ "A value that is not declared cannot be written, because the reader would have no way to stack it back:"
277
+ ]
278
+ },
279
+ {
280
+ "cell_type": "code",
281
+ "id": "21",
282
+ "metadata": {},
283
+ "execution_count": null,
284
+ "outputs": [],
285
+ "source": [
286
+ "bad = dataset[\"u\"].assign_coords(i=[\"x\", \"y\", \"w\"])\n",
287
+ "try:\n",
288
+ " list(convention.writer(bad))\n",
289
+ "except ValueError as error:\n",
290
+ " print(error)"
291
+ ]
292
+ },
293
+ {
294
+ "cell_type": "code",
295
+ "id": "22",
296
+ "metadata": {},
297
+ "execution_count": null,
298
+ "outputs": [],
299
+ "source": [
300
+ "directory_holder.cleanup()"
301
+ ]
302
+ }
303
+ ],
304
+ "metadata": {
305
+ "language_info": {
306
+ "name": "python"
307
+ }
308
+ },
309
+ "nbformat": 4,
310
+ "nbformat_minor": 5
311
+ }
@@ -33,6 +33,8 @@ Contract:
33
33
  - `coords`: mapping of dimension name to coordinate array, in on-disk order
34
34
  - `name`: variable name
35
35
  - `attrs`: optional attributes
36
+ - `order`: memory layout of the file, `"C"` (default, last dimension varies fastest, as NumPy's `tofile` and C programs write) or `"F"` (first dimension varies fastest, as Fortran programs such as 2DECOMP&FFT and Xcompact3d write). Both describe the same bytes: a Fortran file with dims `(x, y, z)` is also a C file with dims `(z, y, x)`. Pick the one that keeps your dimension names in their natural order
37
+ - `coord_attrs`: optional attributes per coordinate, attached on read (for example `{"x": {"units": "m"}}`)
36
38
 
37
39
  The backend validates the file size against `coords` and `dtype` before reading anything, so a wrong shape or byte order fails early instead of producing garbage.
38
40
 
@@ -50,6 +52,9 @@ Contract:
50
52
  - `filename`: output path. A relative path is resolved against the directory passed to `to_file`, which must exist, and may include sub-folders, which are created on demand, but it must stay inside that directory (`..` escaping it is rejected; symbolic links are not followed for this check). An absolute path is used as is, so one call can target several locations
51
53
  - `sub_array`: array slice to write into that file, already transposed to on-disk order
52
54
  - `dtype`: optional on-disk data type; when set, the sub-array is cast right before serialization (including byte order, for example `"<f4"`). When omitted, the in-memory dtype and native byte order are written as-is, so make sure they match what your read specs getter declares.
55
+ - `order`: memory layout of the file, `"C"` (default) or `"F"`, with the same meaning as on `ReadSpecs`. `tofile` always serializes in C order, so Fortran order is obtained by flattening the array first.
56
+
57
+ `to_file` also accepts a `progress` wrapper applied to the iterator of write specs, for example `progress=tqdm` or `progress=functools.partial(tqdm, desc="ux")`, to report progress without the library depending on a progress-bar package.
53
58
 
54
59
  ```{warning}
55
60
  Writes are eager and whole-file only. Each `sub_array` is fully loaded into memory (a Dask compute for lazy data) and its file is written in a single pass — the backend never appends to, patches, or resumes a file. Files are staged in a temporary file next to their destination and atomically moved into place with `os.replace` once complete, so an interrupted write never leaves a truncated file behind. That protects against partially-updated, corrupted files, but it means every `sub_array` you yield must fit in memory. Split large arrays into more, smaller files, and reach for xarray-supported formats such as NetCDF or Zarr when you need streaming or partial writes.
@@ -61,8 +66,18 @@ See the [API reference](../references/api-reference.rst) for the authoritative s
61
66
 
62
67
  The module `xarray_binfile.conventions` provides small, ready-to-use implementations of both protocols. Each convention exposes a `reader` (the read specs getter) and a `writer` (the write specs getter) built from the same two ingredients, so files written by one can always be read back by the other:
63
68
 
64
- - `FilenamePattern`: one `str.format`-style template such as `"{name}-{step:04d}.bin"`. The regular expression used for reading is derived from the template, so the two cannot drift apart. Matching is anchored to the whole filename, and a zero-fill width is a minimum width (step `12345` written with `04d` still reads back).
65
- - `Layout`: the dimension order, coordinate values and dtype of one file on disk. Readers attach these coordinates to every file; writers refuse arrays that do not match them.
69
+ - `FilenamePattern`: one `str.format`-style template such as `"{name}-{step:04d}.bin"`. The regular expression used for reading is derived from the template, so the two cannot drift apart. Matching is anchored to the whole filename, and a zero-fill width is a minimum width (step `12345` written with `04d` still reads back). Templates without a separator between a name ending in a digit and the number are ambiguous under that rule (`phi1000` reads as `phi` + `1000`); pass `exact_width=True` to match exactly the declared width instead (`phi1` + `000`), which also refuses wider steps at write time. `glob()` turns the template into a pattern for `pathlib.Path.glob` (`"*-*.bin"`, or `"ux-*.bin"` with `glob(name="ux")`), and a `name` value may start with folder segments (`"geometry/epsi"`), which are kept in front of the formatted filename.
70
+ - `Layout`: the dimension order, coordinate values, dtype and memory `order` of one file on disk, plus optional `coord_attrs`. Readers attach these coordinates to every file; writers refuse arrays that do not match them.
71
+
72
+ Every shipped convention inherits `Convention`, the base class that turns a `reader`/`writer` pair into a full convention: `name_of_file(path)`, `accepts(path)`, `files(directory)`, `stack(dataset)` and `open(directory, ...)`, all derived from `reader` by default and overridden where something cheaper exists. Subclass it for your own conventions; a bare object with only `reader` and `writer` is wrapped through `Convention.adapt` when it joins a `FolderConventions` or `PatternConventions`, so it gets the same defaults.
73
+
74
+ Every shipped convention also offers:
75
+
76
+ - `files(directory)`: the files in a folder that follow its pattern, and `open(directory, ...)`: those files as one lazy dataset through `xarray.open_mfdataset`, with `variables=[...]` to open a subset and any other keyword (`chunks`, `parallel`) forwarded. Discovery is driven by the anchored pattern only, so unrelated files in the data folder (an XDMF index, notes, backups, hidden files, the temporary files of an interrupted write) are never opened
77
+ - `names`: the variable names the convention accepts. When the pattern alone is too permissive, for example a bare `"{name}"` template for extension-less files that would also match `README` or `Makefile`, `names=("epsilon",)` restricts reading, discovery and writing to the listed variables, and lets a `PatternConventions` move on to the next member
78
+ - `stacks`: dimensions encoded in the variable names, see [Stacked variables](#stacked-variables)
79
+ - `name_of`: a hook returning the name to write an array under, in place of `data_array.name` (for example `lambda da: da.attrs["file_name"]`)
80
+ - `time_dtype` (time-series conventions): the dtype of the `time` coordinate on read, for example `np.float32` to match single-precision data
66
81
 
67
82
  ### Time-series numbered by step
68
83
 
@@ -118,6 +133,37 @@ from xarray_binfile.conventions import StaticFiles
118
133
  static = StaticFiles(layout, pattern="{name}.bin")
119
134
  ```
120
135
 
136
+ ### Stacked variables
137
+
138
+ Solvers often store a vector field as one file per component (`ux`, `uy`, `uz`) and a set of scalar fractions as numbered variables (`phi1`, `phi2`). A `VariableStack` declares that encoding once: the dimension it stands for, the template that builds a variable name from `{name}` and the value, and the values it may take. Conventions split arrays along those dimensions on write and stack the variables back on read:
139
+
140
+ ```python
141
+ from xarray_binfile.conventions import VariableStack
142
+
143
+ velocity = VariableStack("i", "{name}{i}", values=("x", "y", "z"))
144
+ scalars = VariableStack("n", "{name}{n:d}", values=range(1, 10), names=("phi",))
145
+ convention = StepIndexedFiles(layout, stacks=[velocity, scalars])
146
+
147
+ # u with i = ["x", "y", "z"] is written as ux-0001.bin, uy-0001.bin, uz-0001.bin
148
+ dataset.binary_engine.to_file(convention.writer, output_dir)
149
+
150
+ # ux, uy, uz found on disk come back as u with the coordinate i
151
+ opened = convention.open(output_dir) # stacks by default
152
+ raw = convention.open(output_dir, stack=False) # ux, uy, uz as on disk
153
+ ```
154
+
155
+ `values` is required because templates without a separator match almost any name (`pp` fits `"{name}{i}"` as `p` + `p`); it also fixes the order of the stacked coordinate, and writing a coordinate value that is not listed raises. On read each value is matched literally, so `vortx` is `vort` + `x`, and `names` (any collection, kept as a `frozenset`; a bare string is rejected) says which base names to reassemble; without it, a group needs `min_components` (two by default) components, so a lone `vorticity` is not read as `vorticit` + `y`. Stacking refuses to overwrite a variable that already carries the base name. Writing always splits an array carrying the dimension, whatever its name. The same declarations work on their own through `VariableStack.stack`, `stack_variables` and `split_variables`. See the [stacked variables tutorial](../tutorials/stacked-variables.ipynb).
156
+
157
+ ### Mixed conventions in one folder
158
+
159
+ When files following different conventions share a folder, for example `ux-0001.bin` next to `epsi.bin`, `PatternConventions` tries its members in order on read and keeps the first whose pattern accepts the filename, so list the most specific pattern first. On write it offers the array to every member and requires exactly one to accept it, based on dimensions and coordinates:
160
+
161
+ ```python
162
+ from xarray_binfile.conventions import PatternConventions
163
+
164
+ conventions = PatternConventions([StepIndexedFiles(layout), StaticFiles(layout)])
165
+ ```
166
+
121
167
  ### Shapes that depend on the folder
122
168
 
123
169
  Projects often keep arrays of different shapes in different folders:
@@ -132,7 +178,7 @@ caseC/
132
178
  epsi.bin
133
179
  ```
134
180
 
135
- `FolderConventions` dispatches to one convention per folder. Reading picks the convention whose folder matches the end of the file's parent path, preferring the most specific folder when several match (`snapshots/3d` over `3d`). Writing offers the array to every convention and requires exactly one to accept it, based on the array dimensions and coordinates, then writes into that folder. If no convention or more than one accepts the array, a `LayoutMismatchError` is raised: the backend never guesses.
181
+ `FolderConventions` dispatches to one convention per folder. Reading picks the convention whose folder matches the end of the file's parent path, preferring the most specific folder when several match (`snapshots/3d` over `3d`). The dataset root is registered as `"."` and matches any file not claimed by a more specific folder. Writing offers the array to every convention and requires exactly one to accept it, based on the array dimensions and coordinates, then writes into that folder. If no convention or more than one accepts the array, a `LayoutMismatchError` is raised: the backend never guesses. An array whose name starts with a registered folder (`"static/epsi"`) is sent to that folder's convention directly, under the remaining name, which resolves the ambiguity between folders sharing one layout. `files(directory)` and `open(directory)` walk every registered folder; `PatternConventions` can be a member, for example as the root convention.
136
182
 
137
183
  ```python
138
184
  from xarray_binfile.conventions import FolderConventions
@@ -146,9 +192,7 @@ conventions = FolderConventions(
146
192
  }
147
193
  )
148
194
 
149
- dataset = xr.open_mfdataset(
150
- sorted(case_dir.rglob("*.bin")), engine="binfile", read_specs_getter=conventions.reader
151
- )
195
+ dataset = conventions.open(case_dir) # or xr.open_mfdataset(conventions.files(case_dir), engine="binfile", read_specs_getter=conventions.reader)
152
196
  dataset.binary_engine.to_file(conventions.writer, output_dir)
153
197
  ```
154
198
 
@@ -189,7 +233,7 @@ Write specs getter:
189
233
 
190
234
  - Derive the read and write filename rules from one definition.
191
235
  - Anchor filename matching to the whole name so stale or backup files are rejected.
192
- - Keep dimension order explicit and consistent.
236
+ - Keep dimension order explicit and consistent, and declare the memory `order` (`"C"` or `"F"`) that matches the program that wrote the files.
193
237
  - Make dtype and endianness explicit.
194
238
  - Fail loudly on layout mismatch instead of broadcasting or squeezing.
195
239
  - Test read/write round-trips (`xarray.testing.assert_identical`).
@@ -202,4 +246,5 @@ Raw binaries are machine-specific unless conventions are explicit. Endianness (l
202
246
 
203
247
  - [Built-in tutorial: reading](../tutorials/read.ipynb)
204
248
  - [Built-in tutorial: writing](../tutorials/write.ipynb)
249
+ - [Built-in tutorial: stacked variables](../tutorials/stacked-variables.ipynb)
205
250
  - [API reference](../references/api-reference.rst)
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
18
18
  commit_id: str | None
19
19
  __commit_id__: str | None
20
20
 
21
- __version__ = version = '0.2.0'
22
- __version_tuple__ = version_tuple = (0, 2, 0)
21
+ __version__ = version = '0.3.0'
22
+ __version_tuple__ = version_tuple = (0, 3, 0)
23
23
 
24
24
  __commit_id__ = commit_id = None
@@ -5,8 +5,9 @@ Start from the convention closest to your project's naming scheme, or copy one
5
5
  and adapt it when none fits.
6
6
  """
7
7
 
8
+ from xarray_binfile.conventions.base import Convention
8
9
  from xarray_binfile.conventions.filename_pattern import FilenamePattern
9
- from xarray_binfile.conventions.folders import FolderConventions
10
+ from xarray_binfile.conventions.folders import FolderConventions, PatternConventions
10
11
  from xarray_binfile.conventions.getters import (
11
12
  StaticFiles,
12
13
  StepIndexedFiles,
@@ -14,14 +15,24 @@ from xarray_binfile.conventions.getters import (
14
15
  )
15
16
  from xarray_binfile.conventions.layout import Layout, LayoutMismatchError
16
17
  from xarray_binfile.conventions.protocol import ConventionProtocol
18
+ from xarray_binfile.conventions.stacking import (
19
+ VariableStack,
20
+ split_variables,
21
+ stack_variables,
22
+ )
17
23
 
18
24
  __all__ = [
25
+ "Convention",
19
26
  "ConventionProtocol",
20
27
  "FilenamePattern",
21
28
  "FolderConventions",
22
29
  "Layout",
23
30
  "LayoutMismatchError",
31
+ "PatternConventions",
24
32
  "StaticFiles",
25
33
  "StepIndexedFiles",
26
34
  "TimeStampedFiles",
35
+ "VariableStack",
36
+ "split_variables",
37
+ "stack_variables",
27
38
  ]