chronozarr 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- chronozarr/__init__.py +21 -0
- chronozarr/append.py +642 -0
- chronozarr/backend.py +191 -0
- chronozarr/cli.py +663 -0
- chronozarr/convert.py +1716 -0
- chronozarr/decode.py +403 -0
- chronozarr/doctor.py +573 -0
- chronozarr/encode.py +1371 -0
- chronozarr/export.py +221 -0
- chronozarr/schema.py +1158 -0
- chronozarr/stac.py +309 -0
- chronozarr/view.py +215 -0
- chronozarr-0.2.0.dist-info/METADATA +255 -0
- chronozarr-0.2.0.dist-info/RECORD +17 -0
- chronozarr-0.2.0.dist-info/WHEEL +4 -0
- chronozarr-0.2.0.dist-info/entry_points.txt +5 -0
- chronozarr-0.2.0.dist-info/licenses/LICENSE +202 -0
chronozarr/__init__.py
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""chronozarr: Zarr v3 convention and reader for raster time series (optional star-delta)."""
|
|
2
|
+
|
|
3
|
+
from chronozarr.append import AppendReport, append
|
|
4
|
+
from chronozarr.decode import ChronoStore, HttpStore, open_store
|
|
5
|
+
from chronozarr.encode import EncodeReport, encode
|
|
6
|
+
from chronozarr.schema import Band, SchemaError, validate
|
|
7
|
+
from chronozarr.view import view
|
|
8
|
+
|
|
9
|
+
__all__ = [
|
|
10
|
+
"AppendReport",
|
|
11
|
+
"Band",
|
|
12
|
+
"ChronoStore",
|
|
13
|
+
"EncodeReport",
|
|
14
|
+
"HttpStore",
|
|
15
|
+
"SchemaError",
|
|
16
|
+
"append",
|
|
17
|
+
"encode",
|
|
18
|
+
"open_store",
|
|
19
|
+
"validate",
|
|
20
|
+
"view",
|
|
21
|
+
]
|
chronozarr/append.py
ADDED
|
@@ -0,0 +1,642 @@
|
|
|
1
|
+
"""chronozarr append: add timesteps to the end of an existing store.
|
|
2
|
+
|
|
3
|
+
An append resizes the time axis of every level, writes only the objects that gain data (the
|
|
4
|
+
shards or chunks holding the new timesteps), and rewrites the metadata that describes the
|
|
5
|
+
longer axis. Every existing chunk keeps its bytes: star-delta references already recorded are
|
|
6
|
+
never changed (spec 4.2), so what a published chunk decodes to cannot change. A shard that
|
|
7
|
+
already holds earlier timesteps is rewritten with those chunks at the same offsets, followed
|
|
8
|
+
by the new chunk and a new index.
|
|
9
|
+
|
|
10
|
+
The new timesteps go through the same cell-by-cell pyramid as `encode`, so levels 1 and up are
|
|
11
|
+
block means of the true values, exactly as a fresh encode would write them. A new timestep of
|
|
12
|
+
a star-delta store references the nearest anchor that exists after the append (spec 4.2); an
|
|
13
|
+
anchor from an earlier append is read back from the store, one chunk per cell and level.
|
|
14
|
+
|
|
15
|
+
Appending is not transactional. Inputs are validated, and a source that is an iterable is
|
|
16
|
+
spilled to disk, before the store is touched; a failure after that leaves the store partly
|
|
17
|
+
modified. Run it on a working copy and publish after `chronozarr validate`.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import json
|
|
23
|
+
import os
|
|
24
|
+
import shutil
|
|
25
|
+
import tempfile
|
|
26
|
+
import time
|
|
27
|
+
import warnings
|
|
28
|
+
from collections.abc import Iterable, Mapping, Sequence
|
|
29
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
30
|
+
from dataclasses import dataclass, replace
|
|
31
|
+
from pathlib import Path
|
|
32
|
+
from typing import Any
|
|
33
|
+
|
|
34
|
+
import numpy as np
|
|
35
|
+
import xarray as xr
|
|
36
|
+
import zarr
|
|
37
|
+
from zarr.errors import ZarrUserWarning
|
|
38
|
+
|
|
39
|
+
from chronozarr import schema
|
|
40
|
+
from chronozarr.decode import ChronoStore, open_store
|
|
41
|
+
from chronozarr.encode import (
|
|
42
|
+
DEFAULT_CELLS_IN_FLIGHT,
|
|
43
|
+
VOLATILITY_SCALE,
|
|
44
|
+
Block,
|
|
45
|
+
_ArraySource,
|
|
46
|
+
_CellWriter,
|
|
47
|
+
_Input,
|
|
48
|
+
_iso_times,
|
|
49
|
+
_LevelArrays,
|
|
50
|
+
_prepare_input,
|
|
51
|
+
_Pyramid,
|
|
52
|
+
_resolve_bands,
|
|
53
|
+
_resolve_transform,
|
|
54
|
+
_shard_bytes,
|
|
55
|
+
_Source,
|
|
56
|
+
_spill_timesteps,
|
|
57
|
+
_write_time_coord,
|
|
58
|
+
)
|
|
59
|
+
from chronozarr.schema import Band, Chronozarr, LevelRef, Temporal, Transform
|
|
60
|
+
|
|
61
|
+
# A `none` store does not record the nominal schedule its volatility was computed against
|
|
62
|
+
# (spec 5); appended timesteps use the writer default.
|
|
63
|
+
NOMINAL_ANCHOR_INTERVAL = 6
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
@dataclass(frozen=True)
|
|
67
|
+
class AppendReport:
|
|
68
|
+
n_appended: int
|
|
69
|
+
n_time: int # timesteps in the store after the append
|
|
70
|
+
objects_written: int # files created or rewritten, metadata included
|
|
71
|
+
bytes_written: int # their total size
|
|
72
|
+
seconds: float
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def is_store(path: str | Path) -> bool:
|
|
76
|
+
"""True when `path` is a local directory whose root group carries a `chronozarr` block."""
|
|
77
|
+
manifest = Path(path) / "zarr.json"
|
|
78
|
+
if not manifest.is_file():
|
|
79
|
+
return False
|
|
80
|
+
try:
|
|
81
|
+
attributes = json.loads(manifest.read_text()).get("attributes", {})
|
|
82
|
+
except (OSError, ValueError):
|
|
83
|
+
return False
|
|
84
|
+
return isinstance(attributes, dict) and "chronozarr" in attributes
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
# --- The store being appended to ---------------------------------------------------------------
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
@dataclass
|
|
91
|
+
class _Target:
|
|
92
|
+
path: Path
|
|
93
|
+
root: zarr.Group
|
|
94
|
+
meta: Chronozarr
|
|
95
|
+
datasets: tuple[LevelRef, ...]
|
|
96
|
+
groups: list[zarr.Group]
|
|
97
|
+
arrays: list[_LevelArrays]
|
|
98
|
+
|
|
99
|
+
@property
|
|
100
|
+
def n_time(self) -> int:
|
|
101
|
+
return len(self.meta.times)
|
|
102
|
+
|
|
103
|
+
@property
|
|
104
|
+
def transform(self) -> Transform:
|
|
105
|
+
return schema.parse_level_attrs(self.groups[0].attrs.asdict(), "level 0").transform
|
|
106
|
+
|
|
107
|
+
@property
|
|
108
|
+
def chunk_size(self) -> int:
|
|
109
|
+
return schema.cell_size(self.arrays[0].data, "level 0/data")
|
|
110
|
+
|
|
111
|
+
@property
|
|
112
|
+
def shapes(self) -> list[tuple[int, int]]:
|
|
113
|
+
return [(a.data.shape[2], a.data.shape[3]) for a in self.arrays]
|
|
114
|
+
|
|
115
|
+
@property
|
|
116
|
+
def star_delta(self) -> bool:
|
|
117
|
+
return self.meta.temporal.encoding == schema.STAR_DELTA
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _open_target(store: str | Path) -> _Target:
|
|
121
|
+
path = Path(store)
|
|
122
|
+
if not path.is_dir():
|
|
123
|
+
raise ValueError(f"{store} is not a local directory; append works on a local store")
|
|
124
|
+
problems = schema.validate(path)
|
|
125
|
+
if problems:
|
|
126
|
+
shown = "\n ".join(problems[:5])
|
|
127
|
+
more = f"\n ... and {len(problems) - 5} more" if len(problems) > 5 else ""
|
|
128
|
+
raise ValueError(
|
|
129
|
+
f"{store} does not validate, so nothing was appended:\n {shown}{more}\n"
|
|
130
|
+
"If an earlier append was interrupted, restore the store from its published copy."
|
|
131
|
+
)
|
|
132
|
+
root = zarr.open_group(path, mode="r+", zarr_format=3, use_consolidated=False)
|
|
133
|
+
parsed = schema.parse_root_attrs(root.attrs.asdict())
|
|
134
|
+
meta = parsed.chronozarr
|
|
135
|
+
if not meta.spec_version.startswith("0.2."):
|
|
136
|
+
raise ValueError(
|
|
137
|
+
f"{store} is chronozarr {meta.spec_version}; append needs a 0.2 store "
|
|
138
|
+
"(re-encode it with this version first)"
|
|
139
|
+
)
|
|
140
|
+
groups = [schema.get_group(root, d.path, "store") for d in parsed.datasets]
|
|
141
|
+
arrays = []
|
|
142
|
+
for group in groups:
|
|
143
|
+
data = schema.get_array(group, meta.variable, "store")
|
|
144
|
+
mask = (
|
|
145
|
+
None
|
|
146
|
+
if meta.mask_variable is None
|
|
147
|
+
else schema.get_array(group, meta.mask_variable, "store")
|
|
148
|
+
)
|
|
149
|
+
coverage = (
|
|
150
|
+
None
|
|
151
|
+
if meta.coverage_variable is None
|
|
152
|
+
else schema.get_array(group, meta.coverage_variable, "store")
|
|
153
|
+
)
|
|
154
|
+
arrays.append(_LevelArrays(data, mask, coverage))
|
|
155
|
+
return _Target(path, root, meta, parsed.datasets, groups, arrays)
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
# --- The new timesteps --------------------------------------------------------------------------
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
@dataclass
|
|
162
|
+
class _Request:
|
|
163
|
+
"""What the caller passed, reduced to one description for `_prepare_input` and the checks."""
|
|
164
|
+
|
|
165
|
+
data: Any
|
|
166
|
+
times: Any
|
|
167
|
+
mask: Any
|
|
168
|
+
coverage: Any
|
|
169
|
+
crs: str | None
|
|
170
|
+
transform: Sequence[float] | None
|
|
171
|
+
bands: Sequence[str | Band | Mapping] | None
|
|
172
|
+
strict_bands: bool # compare band metadata (scale, offset, units), not just names
|
|
173
|
+
nodata: Any = "unchecked"
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _request_from_store(source: str | Path | ChronoStore) -> _Request:
|
|
177
|
+
"""Every timestep of a chronozarr store (for example one written by `convert`), level 0."""
|
|
178
|
+
store = source if isinstance(source, ChronoStore) else open_store(source)
|
|
179
|
+
count = len(store.times)
|
|
180
|
+
level0 = store.levels[0]
|
|
181
|
+
return _Request(
|
|
182
|
+
data=(store.read(t) for t in range(count)),
|
|
183
|
+
times=store.times,
|
|
184
|
+
mask=None if level0.mask is None else (store.read_mask(t) for t in range(count)),
|
|
185
|
+
coverage=None
|
|
186
|
+
if level0.coverage is None
|
|
187
|
+
else (store.read_coverage(t) for t in range(count)),
|
|
188
|
+
crs=store.attrs.crs,
|
|
189
|
+
transform=level0.transform,
|
|
190
|
+
bands=store.attrs.bands,
|
|
191
|
+
strict_bands=True,
|
|
192
|
+
nodata=store.nodata,
|
|
193
|
+
)
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _mismatches(
|
|
197
|
+
target: _Target, request: _Request, prepared: _Input, times_ms: np.ndarray
|
|
198
|
+
) -> list[str]:
|
|
199
|
+
"""Every way the new data differs from the store; empty when it can be appended."""
|
|
200
|
+
meta = target.meta
|
|
201
|
+
found: list[str] = []
|
|
202
|
+
height, width = target.shapes[0]
|
|
203
|
+
if (prepared.height, prepared.width) != (height, width):
|
|
204
|
+
found.append(
|
|
205
|
+
f"grid: the input is {prepared.height} x {prepared.width} pixels, the store's "
|
|
206
|
+
f"level 0 is {height} x {width}"
|
|
207
|
+
)
|
|
208
|
+
if prepared.dtype != target.arrays[0].data.dtype:
|
|
209
|
+
found.append(
|
|
210
|
+
f"dtype: the input is {prepared.dtype}, the store is {target.arrays[0].data.dtype} "
|
|
211
|
+
"(chronozarr never converts dtype)"
|
|
212
|
+
)
|
|
213
|
+
crs = request.crs if request.crs is not None else prepared.attrs.get("crs")
|
|
214
|
+
if crs is not None and str(crs) != meta.crs:
|
|
215
|
+
found.append(f"crs: the input is {str(crs)!r}, the store is {meta.crs!r}")
|
|
216
|
+
|
|
217
|
+
georeferenced = request.transform is not None or (
|
|
218
|
+
prepared.da is not None
|
|
219
|
+
and (
|
|
220
|
+
"transform" in prepared.da.attrs
|
|
221
|
+
or ("x" in prepared.da.coords and "y" in prepared.da.coords)
|
|
222
|
+
)
|
|
223
|
+
)
|
|
224
|
+
if georeferenced:
|
|
225
|
+
given: Transform = _resolve_transform(request.transform, prepared.da)
|
|
226
|
+
if not schema.same_numbers(given, target.transform):
|
|
227
|
+
found.append(
|
|
228
|
+
f"transform: the input's is {list(given)}, the store's is {list(target.transform)}"
|
|
229
|
+
)
|
|
230
|
+
|
|
231
|
+
if prepared.n_band != len(meta.bands):
|
|
232
|
+
found.append(f"bands: the input has {prepared.n_band}, the store has {len(meta.bands)}")
|
|
233
|
+
elif request.bands is not None or prepared.band_coords is not None:
|
|
234
|
+
given_bands = _resolve_bands(request.bands, prepared.band_coords, prepared.n_band)
|
|
235
|
+
for new, old in zip(given_bands, meta.bands, strict=True):
|
|
236
|
+
if _band_conflict(new, old, strict=request.strict_bands):
|
|
237
|
+
found.append(
|
|
238
|
+
f"bands: the input has {_band_text(new)}, the store has {_band_text(old)}"
|
|
239
|
+
)
|
|
240
|
+
break
|
|
241
|
+
|
|
242
|
+
if (prepared.mask is None) != (meta.mask_variable is None):
|
|
243
|
+
found.append(
|
|
244
|
+
"mask: the store has a mask, so the input needs one validity plane per timestep"
|
|
245
|
+
if meta.mask_variable
|
|
246
|
+
else "mask: the store has no mask and append cannot add one; leave the mask out"
|
|
247
|
+
)
|
|
248
|
+
if (prepared.coverage is None) != (meta.coverage_variable is None):
|
|
249
|
+
found.append(
|
|
250
|
+
"coverage: the store has coverage, so the input needs one plane per timestep"
|
|
251
|
+
if meta.coverage_variable
|
|
252
|
+
else "coverage: the store has no coverage and append cannot add it; leave it out"
|
|
253
|
+
)
|
|
254
|
+
if not isinstance(request.nodata, str) and request.nodata != meta.nodata:
|
|
255
|
+
found.append(f"nodata: the input has {request.nodata!r}, the store has {meta.nodata!r}")
|
|
256
|
+
|
|
257
|
+
last = schema.parse_time(meta.times[-1])
|
|
258
|
+
first_new = times_ms[0].astype("datetime64[ms]")
|
|
259
|
+
if first_new <= last:
|
|
260
|
+
found.append(
|
|
261
|
+
f"times: the input starts at {np.datetime_as_string(first_new, unit='ms')}Z, "
|
|
262
|
+
f"which is not after the store's last time {meta.times[-1]}"
|
|
263
|
+
)
|
|
264
|
+
return found
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
def _band_conflict(new: Band, old: Band, *, strict: bool) -> bool:
|
|
268
|
+
"""Whether `new` cannot continue the band `old`.
|
|
269
|
+
|
|
270
|
+
Names must match. When `new` carries metadata (always, if `strict`), the physical meaning must
|
|
271
|
+
agree: scale and offset after their defaults (1 and 0), and common name and units when both
|
|
272
|
+
sides give one.
|
|
273
|
+
"""
|
|
274
|
+
if new.name != old.name:
|
|
275
|
+
return True
|
|
276
|
+
if not strict and new == Band(new.name):
|
|
277
|
+
return False
|
|
278
|
+
scale = (1.0 if new.scale is None else new.scale, 1.0 if old.scale is None else old.scale)
|
|
279
|
+
offset = (0.0 if new.offset is None else new.offset, 0.0 if old.offset is None else old.offset)
|
|
280
|
+
return (
|
|
281
|
+
scale[0] != scale[1]
|
|
282
|
+
or offset[0] != offset[1]
|
|
283
|
+
or (None not in (new.units, old.units) and new.units != old.units)
|
|
284
|
+
or (None not in (new.common_name, old.common_name) and new.common_name != old.common_name)
|
|
285
|
+
)
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
def _band_text(band: Band) -> str:
|
|
289
|
+
extras = {k: v for k, v in band.to_attrs().items() if k != "name"}
|
|
290
|
+
return f"{band.name!r}" + (f" {extras}" if extras else "")
|
|
291
|
+
|
|
292
|
+
|
|
293
|
+
# --- Schedule ------------------------------------------------------------------------------------
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
@dataclass(frozen=True)
|
|
297
|
+
class _Plan:
|
|
298
|
+
"""What the new timesteps reference, and how the volatility moves."""
|
|
299
|
+
|
|
300
|
+
old_n: int
|
|
301
|
+
new_n: int
|
|
302
|
+
stored_refs: Mapping[int, int] # new non-anchor timestep -> anchor; empty for plain stores
|
|
303
|
+
volatility_refs: Mapping[int, int] # same, or the nominal schedule for plain stores
|
|
304
|
+
n_deltas_before: int # non-anchor timesteps counted by the volatility before the append
|
|
305
|
+
|
|
306
|
+
@property
|
|
307
|
+
def n_deltas_after(self) -> int:
|
|
308
|
+
return self.n_deltas_before + len(self.volatility_refs)
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
def _plan(target: _Target, n_new: int) -> _Plan:
|
|
312
|
+
old_n, new_n = target.n_time, target.n_time + n_new
|
|
313
|
+
temporal = target.meta.temporal
|
|
314
|
+
if target.star_delta:
|
|
315
|
+
anchors, schedule = schema.compute_anchor_schedule(new_n, temporal.anchor_interval)
|
|
316
|
+
if tuple(anchors[: len(temporal.anchor_indices)]) != temporal.anchor_indices:
|
|
317
|
+
raise AssertionError("anchor positions changed on append")
|
|
318
|
+
refs = {t: a for t, a in schedule.items() if t >= old_n}
|
|
319
|
+
return _Plan(old_n, new_n, refs, refs, len(temporal.delta_reference))
|
|
320
|
+
_, nominal_after = schema.compute_anchor_schedule(new_n, NOMINAL_ANCHOR_INTERVAL)
|
|
321
|
+
_, nominal_before = schema.compute_anchor_schedule(old_n, NOMINAL_ANCHOR_INTERVAL)
|
|
322
|
+
refs = {t: a for t, a in nominal_after.items() if t >= old_n}
|
|
323
|
+
return _Plan(old_n, new_n, {}, refs, len(nominal_before))
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
# --- One cell ------------------------------------------------------------------------------------
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
def _abs_diff_sum(a: np.ndarray, b: np.ndarray) -> float:
|
|
330
|
+
wide = np.float64 if a.dtype.kind == "f" else np.int32
|
|
331
|
+
return float(np.abs(a.astype(wide) - b.astype(wide)).sum(dtype=np.float64))
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
@dataclass(frozen=True)
|
|
335
|
+
class _CellDone:
|
|
336
|
+
seconds: float
|
|
337
|
+
abs_delta_gain: float # sum of |value - anchor| over the new delta timesteps (level 0 only)
|
|
338
|
+
|
|
339
|
+
|
|
340
|
+
def _append_cell(
|
|
341
|
+
block: Block,
|
|
342
|
+
arrays: _LevelArrays,
|
|
343
|
+
ys: slice,
|
|
344
|
+
xs: slice,
|
|
345
|
+
plan: _Plan,
|
|
346
|
+
*,
|
|
347
|
+
want_volatility: bool,
|
|
348
|
+
) -> _CellDone:
|
|
349
|
+
"""Write the new timesteps of one cell: star-delta applied, mask and coverage alongside."""
|
|
350
|
+
started = time.perf_counter()
|
|
351
|
+
stored_anchors: dict[int, np.ndarray] = {}
|
|
352
|
+
|
|
353
|
+
def truth(t: int) -> np.ndarray:
|
|
354
|
+
"""True values of timestep t: from the new block, or an anchor read back from the store."""
|
|
355
|
+
if t >= plan.old_n:
|
|
356
|
+
return block.data[t - plan.old_n]
|
|
357
|
+
if t not in stored_anchors:
|
|
358
|
+
stored_anchors[t] = np.asarray(arrays.data[t, :, ys, xs])
|
|
359
|
+
return stored_anchors[t]
|
|
360
|
+
|
|
361
|
+
stored = np.empty_like(block.data)
|
|
362
|
+
for j in range(block.data.shape[0]):
|
|
363
|
+
t = plan.old_n + j
|
|
364
|
+
anchor = plan.stored_refs.get(t)
|
|
365
|
+
if anchor is None:
|
|
366
|
+
stored[j] = block.data[j]
|
|
367
|
+
else:
|
|
368
|
+
np.subtract(block.data[j], truth(anchor), out=stored[j]) # unsigned wraps silently
|
|
369
|
+
arrays.data[plan.old_n : plan.new_n, :, ys, xs] = stored
|
|
370
|
+
if arrays.mask is not None and block.mask is not None:
|
|
371
|
+
arrays.mask[plan.old_n : plan.new_n, ys, xs] = block.mask
|
|
372
|
+
if arrays.coverage is not None and block.coverage is not None:
|
|
373
|
+
arrays.coverage[plan.old_n : plan.new_n, ys, xs] = block.coverage
|
|
374
|
+
|
|
375
|
+
gain = 0.0
|
|
376
|
+
if want_volatility:
|
|
377
|
+
gain = sum(_abs_diff_sum(truth(t), truth(a)) for t, a in plan.volatility_refs.items())
|
|
378
|
+
return _CellDone(time.perf_counter() - started, gain)
|
|
379
|
+
|
|
380
|
+
|
|
381
|
+
# --- Metadata ------------------------------------------------------------------------------------
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
def _update_volatility(target: _Target, plan: _Plan, gains: np.ndarray) -> None:
|
|
385
|
+
"""Fold the new deltas into the per-cell mean |value - anchor| the volatility records.
|
|
386
|
+
|
|
387
|
+
The stored value is clip(mean / 10000), so the old sum is recovered as value * 10000 * count.
|
|
388
|
+
The result equals what the full computation over the recorded references gives, up to
|
|
389
|
+
float32 rounding; a cell already clipped at 1.0 stays at 1.0.
|
|
390
|
+
"""
|
|
391
|
+
array = schema.get_array(target.root, target.meta.volatility_path, "store")
|
|
392
|
+
rows, cols = array.shape
|
|
393
|
+
cs, (height, width) = target.chunk_size, target.shapes[0]
|
|
394
|
+
n_band = len(target.meta.bands)
|
|
395
|
+
pixels = np.empty((rows, cols), dtype=np.float64)
|
|
396
|
+
for row in range(rows):
|
|
397
|
+
for col in range(cols):
|
|
398
|
+
cell_h = min(cs, height - row * cs)
|
|
399
|
+
cell_w = min(cs, width - col * cs)
|
|
400
|
+
pixels[row, col] = cell_h * cell_w * n_band
|
|
401
|
+
before = np.asarray(array[:], dtype=np.float64) * VOLATILITY_SCALE * plan.n_deltas_before
|
|
402
|
+
total = before * pixels + gains
|
|
403
|
+
if plan.n_deltas_after == 0:
|
|
404
|
+
updated = np.zeros((rows, cols))
|
|
405
|
+
else:
|
|
406
|
+
updated = total / (plan.n_deltas_after * pixels * VOLATILITY_SCALE)
|
|
407
|
+
new_values = np.clip(updated, 0.0, 1.0).astype(np.float32)
|
|
408
|
+
if not np.array_equal(new_values, np.asarray(array[:])): # an anchor adds no deltas
|
|
409
|
+
array[:] = new_values
|
|
410
|
+
|
|
411
|
+
|
|
412
|
+
def _commit_metadata(
|
|
413
|
+
target: _Target, plan: _Plan, times_iso: list[str], times_ms: np.ndarray
|
|
414
|
+
) -> None:
|
|
415
|
+
"""Time arrays, root attributes and consolidated metadata for the longer axis."""
|
|
416
|
+
meta = target.meta
|
|
417
|
+
all_ms = np.concatenate(
|
|
418
|
+
[
|
|
419
|
+
np.array([schema.parse_time(t) for t in meta.times], dtype="datetime64[ms]").astype(
|
|
420
|
+
np.int64
|
|
421
|
+
),
|
|
422
|
+
times_ms,
|
|
423
|
+
]
|
|
424
|
+
)
|
|
425
|
+
for group in target.groups:
|
|
426
|
+
_write_time_coord(group, all_ms, overwrite=True)
|
|
427
|
+
|
|
428
|
+
temporal = meta.temporal
|
|
429
|
+
if target.star_delta:
|
|
430
|
+
reference = {**temporal.delta_reference, **plan.stored_refs}
|
|
431
|
+
anchors, _ = schema.compute_anchor_schedule(plan.new_n, temporal.anchor_interval)
|
|
432
|
+
temporal = Temporal(
|
|
433
|
+
temporal.anchor_interval,
|
|
434
|
+
tuple(anchors),
|
|
435
|
+
reference,
|
|
436
|
+
schema.STAR_DELTA,
|
|
437
|
+
temporal.selection,
|
|
438
|
+
)
|
|
439
|
+
else:
|
|
440
|
+
temporal = Temporal.plain(plan.new_n, temporal.selection)
|
|
441
|
+
levels = (
|
|
442
|
+
None
|
|
443
|
+
if meta.levels is None
|
|
444
|
+
else tuple(replace(lv, shape=(plan.new_n, *lv.shape[1:])) for lv in meta.levels)
|
|
445
|
+
)
|
|
446
|
+
shard_bytes = (
|
|
447
|
+
None if meta.shard_bytes is None else _shard_bytes(target.path, len(target.datasets))
|
|
448
|
+
)
|
|
449
|
+
updated = replace(
|
|
450
|
+
meta,
|
|
451
|
+
times=(*meta.times, *times_iso),
|
|
452
|
+
temporal=temporal,
|
|
453
|
+
levels=levels,
|
|
454
|
+
shard_bytes=shard_bytes,
|
|
455
|
+
)
|
|
456
|
+
# `multiscales` is not rewritten: it stays byte-identical (spec 14), so a store published
|
|
457
|
+
# with `pixels_per_tile` keeps it.
|
|
458
|
+
target.root.attrs.update({"chronozarr": updated.to_attrs()})
|
|
459
|
+
with warnings.catch_warnings():
|
|
460
|
+
# Consolidated metadata is deliberate (spec 3.1); see encode._write_store.
|
|
461
|
+
warnings.simplefilter("ignore", ZarrUserWarning)
|
|
462
|
+
zarr.consolidate_metadata(str(target.path))
|
|
463
|
+
|
|
464
|
+
|
|
465
|
+
# --- Entry point ---------------------------------------------------------------------------------
|
|
466
|
+
|
|
467
|
+
|
|
468
|
+
def _snapshot(path: Path) -> dict[str, tuple[int, int]]:
|
|
469
|
+
"""(size, mtime_ns) of every file under `path`, by relative path."""
|
|
470
|
+
entries = {}
|
|
471
|
+
for root, _, files in os.walk(path):
|
|
472
|
+
for name in files:
|
|
473
|
+
full = Path(root, name)
|
|
474
|
+
stat = full.stat()
|
|
475
|
+
entries[str(full.relative_to(path))] = (stat.st_size, stat.st_mtime_ns)
|
|
476
|
+
return entries
|
|
477
|
+
|
|
478
|
+
|
|
479
|
+
def append(
|
|
480
|
+
store: str | Path,
|
|
481
|
+
data: xr.DataArray | Iterable[np.ndarray] | str | Path | ChronoStore,
|
|
482
|
+
*,
|
|
483
|
+
times: Sequence | np.ndarray | None = None,
|
|
484
|
+
crs: str | None = None,
|
|
485
|
+
transform: Sequence[float] | None = None,
|
|
486
|
+
bands: Sequence[str | Band | Mapping] | None = None,
|
|
487
|
+
mask: object = None,
|
|
488
|
+
coverage: object = None,
|
|
489
|
+
workers: int | None = None,
|
|
490
|
+
spill_dir: str | Path | None = None,
|
|
491
|
+
) -> AppendReport:
|
|
492
|
+
"""Append timesteps to the end of the chronozarr v0.2 store at `store`, in place.
|
|
493
|
+
|
|
494
|
+
Args:
|
|
495
|
+
store: A local store directory that validates.
|
|
496
|
+
data: Either a DataArray with dims (time, band, y, x) like `encode` takes, an iterable of
|
|
497
|
+
per-timestep (band, y, x) arrays (needs `times`), or a chronozarr store (a path or a
|
|
498
|
+
`ChronoStore`, for example one month written by `convert`), whose every timestep is
|
|
499
|
+
appended. A store carries its own times, mask, coverage, CRS, grid and bands, so
|
|
500
|
+
none of the other arguments may be given with it.
|
|
501
|
+
times: datetime64 timestamps, strictly increasing and after the store's last time
|
|
502
|
+
(iterable input only).
|
|
503
|
+
crs, transform, bands: Optional, and compared with the store when given or carried by a
|
|
504
|
+
DataArray (attributes, coordinates, band coordinate). The store's own are used.
|
|
505
|
+
mask, coverage: Required exactly when the store has them: one plane per timestep, shaped
|
|
506
|
+
as for `encode`.
|
|
507
|
+
workers: Cells written concurrently (default 4).
|
|
508
|
+
spill_dir: Directory for the temp files of iterable input. Default: next to `store`.
|
|
509
|
+
|
|
510
|
+
The input must match the store's grid, bands, dtype, CRS and nodata; a mismatch raises a
|
|
511
|
+
ValueError listing every difference and leaves the store untouched. Only the shards (or
|
|
512
|
+
chunks of an unsharded store) that gain a timestep are written; the other objects keep their
|
|
513
|
+
bytes. See the module docstring for what happens when the append fails midway.
|
|
514
|
+
"""
|
|
515
|
+
started = time.perf_counter()
|
|
516
|
+
cells_in_flight = workers if workers is not None else DEFAULT_CELLS_IN_FLIGHT
|
|
517
|
+
if cells_in_flight < 1:
|
|
518
|
+
raise ValueError(f"workers must be >= 1, got {cells_in_flight}")
|
|
519
|
+
target = _open_target(store)
|
|
520
|
+
|
|
521
|
+
if isinstance(data, str | Path | ChronoStore):
|
|
522
|
+
if any(given is not None for given in (times, crs, transform, bands, mask, coverage)):
|
|
523
|
+
raise ValueError(
|
|
524
|
+
"a chronozarr store carries its own times, grid, bands, mask and coverage; "
|
|
525
|
+
"pass only the store"
|
|
526
|
+
)
|
|
527
|
+
request = _request_from_store(data)
|
|
528
|
+
else:
|
|
529
|
+
request = _Request(data, times, mask, coverage, crs, transform, bands, strict_bands=False)
|
|
530
|
+
|
|
531
|
+
prepared = _prepare_input(request.data, request.times, request.mask, request.coverage)
|
|
532
|
+
times_iso, times_ms = _iso_times(prepared.times)
|
|
533
|
+
if len(times_iso) != prepared.n_time:
|
|
534
|
+
raise ValueError(f"{len(times_iso)} times for {prepared.n_time} timesteps")
|
|
535
|
+
problems = _mismatches(target, request, prepared, times_ms)
|
|
536
|
+
if problems:
|
|
537
|
+
listed = "\n - ".join(problems)
|
|
538
|
+
raise ValueError(f"cannot append to {store}: the input does not match it:\n - {listed}")
|
|
539
|
+
|
|
540
|
+
plan = _plan(target, prepared.n_time)
|
|
541
|
+
spill: Path | None = None
|
|
542
|
+
try:
|
|
543
|
+
if prepared.da is not None:
|
|
544
|
+
source: _Source = _ArraySource(
|
|
545
|
+
prepared.da,
|
|
546
|
+
prepared.mask if isinstance(prepared.mask, xr.DataArray) else None,
|
|
547
|
+
prepared.coverage if isinstance(prepared.coverage, xr.DataArray) else None,
|
|
548
|
+
target.chunk_size,
|
|
549
|
+
)
|
|
550
|
+
else:
|
|
551
|
+
spill = Path(
|
|
552
|
+
tempfile.mkdtemp(
|
|
553
|
+
prefix=f".{target.path.name}-spill-",
|
|
554
|
+
dir=str(spill_dir) if spill_dir else target.path.parent,
|
|
555
|
+
)
|
|
556
|
+
)
|
|
557
|
+
source = _spill_timesteps(
|
|
558
|
+
prepared,
|
|
559
|
+
spill,
|
|
560
|
+
chunk_size=target.chunk_size,
|
|
561
|
+
has_mask=target.meta.mask_variable is not None,
|
|
562
|
+
has_coverage=target.meta.coverage_variable is not None,
|
|
563
|
+
)
|
|
564
|
+
before = _snapshot(target.path)
|
|
565
|
+
try:
|
|
566
|
+
_write(target, plan, prepared, source, times_iso, times_ms, cells_in_flight)
|
|
567
|
+
except BaseException as exc:
|
|
568
|
+
exc.add_note(
|
|
569
|
+
f"{target.path} may be partly modified; restore it from its published copy "
|
|
570
|
+
"before appending again"
|
|
571
|
+
)
|
|
572
|
+
raise
|
|
573
|
+
finally:
|
|
574
|
+
if spill is not None:
|
|
575
|
+
shutil.rmtree(spill, ignore_errors=True)
|
|
576
|
+
|
|
577
|
+
after = _snapshot(target.path)
|
|
578
|
+
written = [key for key, state in after.items() if before.get(key) != state]
|
|
579
|
+
return AppendReport(
|
|
580
|
+
n_appended=prepared.n_time,
|
|
581
|
+
n_time=plan.new_n,
|
|
582
|
+
objects_written=len(written),
|
|
583
|
+
bytes_written=sum(after[key][0] for key in written),
|
|
584
|
+
seconds=time.perf_counter() - started,
|
|
585
|
+
)
|
|
586
|
+
|
|
587
|
+
|
|
588
|
+
def _write(
|
|
589
|
+
target: _Target,
|
|
590
|
+
plan: _Plan,
|
|
591
|
+
prepared: _Input,
|
|
592
|
+
source: _Source,
|
|
593
|
+
times_iso: list[str],
|
|
594
|
+
times_ms: np.ndarray,
|
|
595
|
+
cells_in_flight: int,
|
|
596
|
+
) -> None:
|
|
597
|
+
for level in target.arrays:
|
|
598
|
+
for array in (level.data, level.mask, level.coverage):
|
|
599
|
+
if array is not None:
|
|
600
|
+
array.resize((plan.new_n, *array.shape[1:]))
|
|
601
|
+
|
|
602
|
+
cs = target.chunk_size
|
|
603
|
+
grid0 = schema.grid_shape(*target.shapes[0], cs)
|
|
604
|
+
gains = np.zeros(grid0, dtype=np.float64)
|
|
605
|
+
writer = _CellWriter[_CellDone](cells_in_flight)
|
|
606
|
+
compute = ThreadPoolExecutor(max_workers=os.cpu_count() or 1)
|
|
607
|
+
pyramid = _Pyramid(
|
|
608
|
+
target.shapes,
|
|
609
|
+
cs,
|
|
610
|
+
n_time=prepared.n_time,
|
|
611
|
+
n_band=prepared.n_band,
|
|
612
|
+
dtype=prepared.dtype,
|
|
613
|
+
nodata=target.meta.nodata,
|
|
614
|
+
source=source,
|
|
615
|
+
compute=compute,
|
|
616
|
+
)
|
|
617
|
+
|
|
618
|
+
def submit(k: int, row: int, col: int, block: Block) -> None:
|
|
619
|
+
ys = slice(row * cs, row * cs + block.height)
|
|
620
|
+
xs = slice(col * cs, col * cs + block.width)
|
|
621
|
+
writer.submit(
|
|
622
|
+
k,
|
|
623
|
+
row,
|
|
624
|
+
col,
|
|
625
|
+
lambda: _append_cell(block, target.arrays[k], ys, xs, plan, want_volatility=k == 0),
|
|
626
|
+
)
|
|
627
|
+
|
|
628
|
+
try:
|
|
629
|
+
pyramid.walk(submit)
|
|
630
|
+
results = writer.results()
|
|
631
|
+
except BaseException:
|
|
632
|
+
writer.shutdown()
|
|
633
|
+
raise
|
|
634
|
+
finally:
|
|
635
|
+
compute.shutdown(wait=True)
|
|
636
|
+
writer.shutdown()
|
|
637
|
+
|
|
638
|
+
for k, row, col, done in results:
|
|
639
|
+
if k == 0:
|
|
640
|
+
gains[row, col] = done.abs_delta_gain
|
|
641
|
+
_update_volatility(target, plan, gains)
|
|
642
|
+
_commit_metadata(target, plan, times_iso, times_ms)
|