h5t 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- h5t/__init__.py +20 -0
- h5t/_cli.py +66 -0
- h5t/_compile.py +840 -0
- h5t/_errors.py +24 -0
- h5t/_spec.py +91 -0
- h5t/py.typed +0 -0
- h5t-0.2.0.dist-info/METADATA +162 -0
- h5t-0.2.0.dist-info/RECORD +11 -0
- h5t-0.2.0.dist-info/WHEEL +4 -0
- h5t-0.2.0.dist-info/entry_points.txt +3 -0
- h5t-0.2.0.dist-info/licenses/LICENSE +21 -0
h5t/__init__.py
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
"""Detached, typed records for read-only HDF5 access."""
|
|
2
|
+
|
|
3
|
+
from h5t._compile import Dataset, Group
|
|
4
|
+
from h5t._errors import ConversionError, H5TError, SchemaError, ValidationError
|
|
5
|
+
from h5t._spec import Attr, Eager, Name
|
|
6
|
+
|
|
7
|
+
__version__ = "0.2.0"
|
|
8
|
+
|
|
9
|
+
__all__ = [
|
|
10
|
+
"Attr",
|
|
11
|
+
"ConversionError",
|
|
12
|
+
"Dataset",
|
|
13
|
+
"Eager",
|
|
14
|
+
"Group",
|
|
15
|
+
"H5TError",
|
|
16
|
+
"Name",
|
|
17
|
+
"SchemaError",
|
|
18
|
+
"ValidationError",
|
|
19
|
+
"__version__",
|
|
20
|
+
]
|
h5t/_cli.py
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
"""The ``h5t check`` command."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import importlib
|
|
7
|
+
import sys
|
|
8
|
+
from typing import NoReturn
|
|
9
|
+
|
|
10
|
+
from h5t._compile import Group
|
|
11
|
+
from h5t._errors import SchemaError, ValidationError
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _fail(message: str) -> NoReturn:
|
|
15
|
+
print(f"error: {message}", file=sys.stderr)
|
|
16
|
+
raise SystemExit(2)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _load_schema(ref: str) -> type[Group]:
|
|
20
|
+
module_name, separator, class_name = ref.partition(":")
|
|
21
|
+
if not separator or not module_name or not class_name:
|
|
22
|
+
_fail(f"--schema must look like pkg.module:ClassName, got {ref!r}")
|
|
23
|
+
try:
|
|
24
|
+
module = importlib.import_module(module_name)
|
|
25
|
+
except Exception as exc:
|
|
26
|
+
_fail(f"could not import module {module_name!r}: {exc}")
|
|
27
|
+
try:
|
|
28
|
+
schema = getattr(module, class_name)
|
|
29
|
+
except AttributeError:
|
|
30
|
+
_fail(f"module {module_name!r} has no attribute {class_name!r}")
|
|
31
|
+
if not (isinstance(schema, type) and issubclass(schema, Group)):
|
|
32
|
+
_fail(f"{ref!r} is not an h5t.Group schema class")
|
|
33
|
+
return schema
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _cmd_check(args: argparse.Namespace) -> int:
|
|
37
|
+
schema = _load_schema(args.schema)
|
|
38
|
+
try:
|
|
39
|
+
schema.from_file(args.file, root=args.root)
|
|
40
|
+
except ValidationError as exc:
|
|
41
|
+
print(f"invalid: {args.file} against {schema.__name__}: {exc}")
|
|
42
|
+
return 1
|
|
43
|
+
except (OSError, ValueError, TypeError, SchemaError) as exc:
|
|
44
|
+
_fail(f"could not check {args.file!r}: {exc}")
|
|
45
|
+
print(f"ok: {args.file} validates against {schema.__name__} at {args.root}")
|
|
46
|
+
return 0
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def main(argv: list[str] | None = None) -> int:
|
|
50
|
+
"""Run the command line interface and return its process exit code."""
|
|
51
|
+
parser = argparse.ArgumentParser(
|
|
52
|
+
prog="h5t",
|
|
53
|
+
description="Validate and materialize an HDF5 group schema.",
|
|
54
|
+
)
|
|
55
|
+
subparsers = parser.add_subparsers(dest="command", required=True)
|
|
56
|
+
check = subparsers.add_parser("check", help="check an HDF5 file against a Group schema")
|
|
57
|
+
check.add_argument("file", help="path to the HDF5 file")
|
|
58
|
+
check.add_argument("--schema", required=True, help="schema reference, e.g. pkg.schemas:Result")
|
|
59
|
+
check.add_argument("--root", default="/", help="absolute HDF5 group path (default: /)")
|
|
60
|
+
check.set_defaults(func=_cmd_check)
|
|
61
|
+
args = parser.parse_args(argv)
|
|
62
|
+
return int(args.func(args))
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
if __name__ == "__main__":
|
|
66
|
+
sys.exit(main())
|
h5t/_compile.py
ADDED
|
@@ -0,0 +1,840 @@
|
|
|
1
|
+
"""Schema compilation and detached HDF5 loading."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import inspect
|
|
6
|
+
import os
|
|
7
|
+
import posixpath
|
|
8
|
+
import reprlib
|
|
9
|
+
import sys
|
|
10
|
+
import types
|
|
11
|
+
import typing
|
|
12
|
+
import weakref
|
|
13
|
+
from collections.abc import Collection, Iterator, Mapping
|
|
14
|
+
from contextlib import contextmanager
|
|
15
|
+
from enum import Enum
|
|
16
|
+
from types import MappingProxyType
|
|
17
|
+
from typing import Annotated, Any, ClassVar, Literal, Self
|
|
18
|
+
|
|
19
|
+
import h5py
|
|
20
|
+
import numpy as np
|
|
21
|
+
from pydantic import ConfigDict, TypeAdapter
|
|
22
|
+
from pydantic import ValidationError as PydanticValidationError
|
|
23
|
+
|
|
24
|
+
from h5t._errors import ConversionError, SchemaError, ValidationError
|
|
25
|
+
from h5t._spec import (
|
|
26
|
+
_NO_DEFAULT,
|
|
27
|
+
Attr,
|
|
28
|
+
ClassSpec,
|
|
29
|
+
Eager,
|
|
30
|
+
Extras,
|
|
31
|
+
FieldSpec,
|
|
32
|
+
MemberKind,
|
|
33
|
+
Name,
|
|
34
|
+
attr_path,
|
|
35
|
+
child_path,
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
_SCALAR_TYPES = (str, int, float, bool, bytes, complex)
|
|
39
|
+
_DATA_NOT_LOADED = object()
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _schema_error(owner: type, field: str, message: str) -> SchemaError:
|
|
43
|
+
return SchemaError(f"{owner.__name__}.{field}: {message}")
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
_ANNOTATION_FILE = "<h5t annotation>"
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class _UnresolvedAnnotation(Exception):
|
|
50
|
+
"""An annotation names something not defined yet, so compilation must defer.
|
|
51
|
+
|
|
52
|
+
Private to this module: raised while resolving annotations and turned into the
|
|
53
|
+
user-visible ``SchemaError`` by ``_ensure_compiled`` if the name is still missing
|
|
54
|
+
when the spec is finally read.
|
|
55
|
+
"""
|
|
56
|
+
|
|
57
|
+
def __init__(self, owner: type, name: str | None, message: str) -> None:
|
|
58
|
+
super().__init__(message)
|
|
59
|
+
self.owner = owner
|
|
60
|
+
self.name = name
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _raised_by_annotation(exc: NameError, cls: type) -> bool:
|
|
64
|
+
"""Report whether ``exc`` came from the annotation expression itself, not a callee.
|
|
65
|
+
|
|
66
|
+
Only an unbound name in the expression is a forward reference worth deferring for.
|
|
67
|
+
A ``NameError`` escaping a function the annotation calls is a bug in that function,
|
|
68
|
+
so it must take the ``SchemaError`` path instead. The frame that raised identifies
|
|
69
|
+
the two: a PEP 563 string annotation is compiled under ``_ANNOTATION_FILE``, and a
|
|
70
|
+
PEP 649 lazy annotation is evaluated by ``cls``'s own ``__annotate__`` function.
|
|
71
|
+
"""
|
|
72
|
+
tb = exc.__traceback__
|
|
73
|
+
while tb is not None and tb.tb_next is not None:
|
|
74
|
+
tb = tb.tb_next
|
|
75
|
+
if tb is None:
|
|
76
|
+
return False
|
|
77
|
+
code = tb.tb_frame.f_code
|
|
78
|
+
if code.co_filename == _ANNOTATION_FILE:
|
|
79
|
+
return True
|
|
80
|
+
# A class with no annotations of its own has ``__annotate__ is None``.
|
|
81
|
+
return code is getattr(getattr(cls, "__annotate__", None), "__code__", None)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _code_names(code: types.CodeType) -> set[str]:
|
|
85
|
+
"""Collect every name ``code`` and the code objects nested in it could load.
|
|
86
|
+
|
|
87
|
+
``co_names`` over-approximates -- it also holds attribute names -- which is the
|
|
88
|
+
safe direction, since a name the snapshot drops is one resolution cannot find.
|
|
89
|
+
Nested code objects (a lambda inside an annotation) load their names the same way.
|
|
90
|
+
|
|
91
|
+
``co_freevars`` is deliberately not collected. A class ``__annotate__`` has free
|
|
92
|
+
variables only when an enclosing function scope exists, which is exactly when its
|
|
93
|
+
closure keeps those cells reachable on its own -- so for that set the snapshot is
|
|
94
|
+
redundant, and adding it would only widen what ``_weak_scope`` holds strongly.
|
|
95
|
+
"""
|
|
96
|
+
names: set[str] = set()
|
|
97
|
+
pending = [code]
|
|
98
|
+
while pending:
|
|
99
|
+
current = pending.pop()
|
|
100
|
+
names.update(current.co_names)
|
|
101
|
+
pending.extend(c for c in current.co_consts if isinstance(c, types.CodeType))
|
|
102
|
+
return names
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _names_in_source(source: str) -> set[str]:
|
|
106
|
+
"""Collect every name the annotation expression ``source`` could load."""
|
|
107
|
+
try:
|
|
108
|
+
code = compile(source, _ANNOTATION_FILE, "eval")
|
|
109
|
+
except SyntaxError:
|
|
110
|
+
# Unparseable: reading the spec raises SchemaError, so nothing is needed.
|
|
111
|
+
return set()
|
|
112
|
+
return _code_names(code)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _referenced_names(annotations: Mapping[str, Any]) -> set[str]:
|
|
116
|
+
"""Collect every name the string annotations in ``annotations`` could load."""
|
|
117
|
+
names: set[str] = set()
|
|
118
|
+
for annotation in annotations.values():
|
|
119
|
+
if isinstance(annotation, str):
|
|
120
|
+
names.update(_names_in_source(annotation))
|
|
121
|
+
return names
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _quoted_names(code: types.CodeType) -> set[str]:
|
|
125
|
+
"""Collect every name the string constants in ``code`` could load.
|
|
126
|
+
|
|
127
|
+
A quoted annotation is a plain string constant in ``__annotate__``: the names it
|
|
128
|
+
mentions never reach ``co_names``, and the compiler makes no closure cell for them
|
|
129
|
+
either, so unlike a bare annotation it has nothing else to resolve through.
|
|
130
|
+
Reading them back out of the constants is what keeps a quoted forward reference
|
|
131
|
+
resolvable in a class that some *other* annotation forced onto the deferred path.
|
|
132
|
+
"""
|
|
133
|
+
names: set[str] = set()
|
|
134
|
+
pending = [code]
|
|
135
|
+
while pending:
|
|
136
|
+
current = pending.pop()
|
|
137
|
+
for const in current.co_consts:
|
|
138
|
+
if isinstance(const, types.CodeType):
|
|
139
|
+
pending.append(const)
|
|
140
|
+
elif isinstance(const, str):
|
|
141
|
+
names.update(_names_in_source(const))
|
|
142
|
+
return names
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def _annotation_names(cls: type) -> set[str]:
|
|
146
|
+
"""Collect every name ``cls``'s own annotations could load, however they failed.
|
|
147
|
+
|
|
148
|
+
Reading the annotations yields PEP 563 strings, which name what they mention. On
|
|
149
|
+
3.14 the read is itself the evaluation (PEP 649), so an annotation that could not
|
|
150
|
+
resolve has no value left to inspect: the names come straight off the
|
|
151
|
+
``__annotate__`` code object instead, without re-running an expression that
|
|
152
|
+
already raised. ``annotationlib``'s ``STRING`` and ``FORWARDREF`` formats look
|
|
153
|
+
like the obvious source here and are not: both re-execute the annotation with the
|
|
154
|
+
real globals first and swallow whatever it raises, which would run a user's
|
|
155
|
+
annotation helper -- and hide its bug -- from inside an exception handler.
|
|
156
|
+
"""
|
|
157
|
+
try:
|
|
158
|
+
annotations = inspect.get_annotations(cls, eval_str=False)
|
|
159
|
+
except Exception:
|
|
160
|
+
annotate = getattr(cls, "__annotate__", None)
|
|
161
|
+
code = getattr(annotate, "__code__", None)
|
|
162
|
+
if code is None:
|
|
163
|
+
return set()
|
|
164
|
+
# _weak_scope keeps only names the scope holds, so the compiler's own
|
|
165
|
+
# __classdict__ freevar drops out.
|
|
166
|
+
return _code_names(code) | _quoted_names(code)
|
|
167
|
+
return _referenced_names(annotations)
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _weak_scope(scope: Mapping[str, Any], names: Collection[str]) -> dict[str, Any]:
|
|
171
|
+
"""Snapshot the ``names`` entries of ``scope``, weakly wherever the value allows.
|
|
172
|
+
|
|
173
|
+
Scopes are only retained by the deferred path, where the snapshot may outlive the
|
|
174
|
+
frame it came from, so it keeps just the names the unresolved annotations mention.
|
|
175
|
+
Values that reject ``weakref.ref`` (``int``, ``str``, tuples, dicts) are stored
|
|
176
|
+
directly, and an unfiltered snapshot would let one of those containers pin an
|
|
177
|
+
unrelated local object graph for as long as the schema class lives.
|
|
178
|
+
"""
|
|
179
|
+
snapshot: dict[str, Any] = {}
|
|
180
|
+
for name in names:
|
|
181
|
+
if name not in scope:
|
|
182
|
+
continue
|
|
183
|
+
value = scope[name]
|
|
184
|
+
try:
|
|
185
|
+
snapshot[name] = weakref.ref(value)
|
|
186
|
+
except TypeError:
|
|
187
|
+
snapshot[name] = value
|
|
188
|
+
return snapshot
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def _unpack_scope(snapshot: Mapping[str, Any]) -> dict[str, Any]:
|
|
192
|
+
"""Reverse ``_weak_scope``, dropping entries whose referent was collected."""
|
|
193
|
+
scope: dict[str, Any] = {}
|
|
194
|
+
for name, value in snapshot.items():
|
|
195
|
+
if isinstance(value, weakref.ref):
|
|
196
|
+
referent = value()
|
|
197
|
+
if referent is not None:
|
|
198
|
+
scope[name] = referent
|
|
199
|
+
else:
|
|
200
|
+
scope[name] = value
|
|
201
|
+
return scope
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def _split_annotation(owner: type, py_name: str, annotation: Any) -> tuple[Any, list[Any], bool]:
|
|
205
|
+
"""Extract h5t/constraint metadata and optionality from an annotation."""
|
|
206
|
+
metadata: list[Any] = []
|
|
207
|
+
optional = False
|
|
208
|
+
core = annotation
|
|
209
|
+
while True:
|
|
210
|
+
if typing.get_origin(core) is Annotated:
|
|
211
|
+
args = typing.get_args(core)
|
|
212
|
+
core = args[0]
|
|
213
|
+
metadata.extend(args[1:])
|
|
214
|
+
continue
|
|
215
|
+
origin = typing.get_origin(core)
|
|
216
|
+
if origin in (typing.Union, types.UnionType):
|
|
217
|
+
args = typing.get_args(core)
|
|
218
|
+
without_none = tuple(arg for arg in args if arg is not type(None))
|
|
219
|
+
if len(without_none) != len(args):
|
|
220
|
+
optional = True
|
|
221
|
+
if len(without_none) != 1:
|
|
222
|
+
raise _schema_error(
|
|
223
|
+
owner,
|
|
224
|
+
py_name,
|
|
225
|
+
"only optional unions of the form 'T | None' are supported",
|
|
226
|
+
)
|
|
227
|
+
core = without_none[0]
|
|
228
|
+
continue
|
|
229
|
+
return core, metadata, optional
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def _adapter_annotation(core: Any, metadata: list[Any], optional: bool) -> Any:
|
|
233
|
+
foreign = tuple(item for item in metadata if not isinstance(item, (Name, Attr, Eager)))
|
|
234
|
+
annotation = Annotated[core, *foreign] if foreign else core
|
|
235
|
+
return annotation | None if optional else annotation
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def _build_adapter(owner: type, py_name: str, annotation: Any) -> TypeAdapter[Any]:
|
|
239
|
+
try:
|
|
240
|
+
return TypeAdapter(annotation, config=ConfigDict(arbitrary_types_allowed=True))
|
|
241
|
+
except Exception as exc:
|
|
242
|
+
raise _schema_error(owner, py_name, f"cannot construct a validator: {exc}") from exc
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def _marker(metadata: list[Any], marker_type: type, owner: type, py_name: str) -> Any | None:
|
|
246
|
+
found = [item for item in metadata if isinstance(item, marker_type)]
|
|
247
|
+
if len(found) > 1:
|
|
248
|
+
raise _schema_error(owner, py_name, f"{marker_type.__name__} may appear only once")
|
|
249
|
+
return found[0] if found else None
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
# PEP 695 aliases are 3.12+; on 3.11 the empty tuple makes every isinstance False.
|
|
253
|
+
_alias_type = getattr(typing, "TypeAliasType", None)
|
|
254
|
+
_TYPE_ALIAS_TYPES: tuple[type, ...] = (_alias_type,) if isinstance(_alias_type, type) else ()
|
|
255
|
+
_MAX_ALIAS_DEPTH = 16
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def _is_ndarray_annotation(core: Any) -> bool:
|
|
259
|
+
"""Report whether ``core`` denotes ``np.ndarray``, however many aliases deep.
|
|
260
|
+
|
|
261
|
+
numpy >= 2.5 defines ``npt.NDArray`` as a PEP 695 ``TypeAliasType``, which defers
|
|
262
|
+
its right-hand side -- deferring is what lets an alias recurse -- so
|
|
263
|
+
``typing.get_origin`` stops at the alias itself. numpy <= 2.4 built the same name
|
|
264
|
+
out of eager substitution, leaving ``np.ndarray`` directly visible as the origin.
|
|
265
|
+
Expanding ``__value__`` sees through either shape, and through a user's own alias
|
|
266
|
+
of one. The walk is depth-bounded because a PEP 695 alias may be self-referential.
|
|
267
|
+
"""
|
|
268
|
+
for _ in range(_MAX_ALIAS_DEPTH):
|
|
269
|
+
if core is np.ndarray:
|
|
270
|
+
return True
|
|
271
|
+
origin = typing.get_origin(core)
|
|
272
|
+
if origin is np.ndarray:
|
|
273
|
+
return True
|
|
274
|
+
if isinstance(core, _TYPE_ALIAS_TYPES):
|
|
275
|
+
core = core.__value__
|
|
276
|
+
elif isinstance(origin, _TYPE_ALIAS_TYPES):
|
|
277
|
+
# A subscripted alias: __value__ carries the parameter, which is dropped
|
|
278
|
+
# here along with the dtype the classification already ignores.
|
|
279
|
+
core = origin.__value__
|
|
280
|
+
else:
|
|
281
|
+
return False
|
|
282
|
+
return False
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
def _is_scalar_annotation(core: Any) -> bool:
|
|
286
|
+
if core in _SCALAR_TYPES or typing.get_origin(core) is Literal:
|
|
287
|
+
return True
|
|
288
|
+
return isinstance(core, type) and (issubclass(core, Enum) or issubclass(core, np.generic))
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def _field_spec(
|
|
292
|
+
owner: type,
|
|
293
|
+
py_name: str,
|
|
294
|
+
annotation: Any,
|
|
295
|
+
default: Any,
|
|
296
|
+
*,
|
|
297
|
+
dataset_owner: bool,
|
|
298
|
+
) -> FieldSpec:
|
|
299
|
+
core, metadata, optional = _split_annotation(owner, py_name, annotation)
|
|
300
|
+
name_marker = _marker(metadata, Name, owner, py_name)
|
|
301
|
+
attr_marker = _marker(metadata, Attr, owner, py_name)
|
|
302
|
+
eager_marker = _marker(metadata, Eager, owner, py_name)
|
|
303
|
+
|
|
304
|
+
h5_name = name_marker.name if name_marker is not None else py_name
|
|
305
|
+
if not isinstance(h5_name, str) or not h5_name:
|
|
306
|
+
raise _schema_error(owner, py_name, "Name requires a non-empty string")
|
|
307
|
+
if attr_marker is not None and attr_marker.converter is not None:
|
|
308
|
+
if not callable(attr_marker.converter):
|
|
309
|
+
raise _schema_error(owner, py_name, "Attr.converter must be callable")
|
|
310
|
+
if attr_marker is not None and eager_marker is not None:
|
|
311
|
+
raise _schema_error(owner, py_name, "Attr and Eager cannot be combined")
|
|
312
|
+
|
|
313
|
+
is_dataset = isinstance(core, type) and issubclass(core, Dataset)
|
|
314
|
+
is_group = isinstance(core, type) and issubclass(core, Group)
|
|
315
|
+
|
|
316
|
+
if attr_marker is not None:
|
|
317
|
+
if is_dataset or is_group:
|
|
318
|
+
raise _schema_error(owner, py_name, "Attr cannot annotate a Group or Dataset field")
|
|
319
|
+
kind = MemberKind.ATTRIBUTE
|
|
320
|
+
member_type = None
|
|
321
|
+
elif is_dataset:
|
|
322
|
+
kind = MemberKind.DATASET
|
|
323
|
+
member_type = core
|
|
324
|
+
elif is_group:
|
|
325
|
+
kind = MemberKind.GROUP
|
|
326
|
+
member_type = core
|
|
327
|
+
elif _is_ndarray_annotation(core):
|
|
328
|
+
# Accept npt.NDArray[...] aliases. The dtype parameter is not validated
|
|
329
|
+
# (the adapter degrades to an isinstance check), so normalize to the
|
|
330
|
+
# plain type to keep declaration equivalence dtype-agnostic.
|
|
331
|
+
kind = MemberKind.ARRAY
|
|
332
|
+
member_type = None
|
|
333
|
+
core = np.ndarray
|
|
334
|
+
elif _is_scalar_annotation(core):
|
|
335
|
+
kind = MemberKind.ATTRIBUTE
|
|
336
|
+
member_type = None
|
|
337
|
+
else:
|
|
338
|
+
raise _schema_error(
|
|
339
|
+
owner,
|
|
340
|
+
py_name,
|
|
341
|
+
f"unsupported annotation {core!r}; containers require Attr(), and dynamic "
|
|
342
|
+
"collections are not supported",
|
|
343
|
+
)
|
|
344
|
+
|
|
345
|
+
if eager_marker is not None and kind is not MemberKind.DATASET:
|
|
346
|
+
raise _schema_error(owner, py_name, "Eager applies only to Dataset fields")
|
|
347
|
+
if dataset_owner and kind is not MemberKind.ATTRIBUTE:
|
|
348
|
+
raise _schema_error(owner, py_name, "a Dataset subclass may declare only attributes")
|
|
349
|
+
if kind is not MemberKind.ATTRIBUTE and "/" in h5_name:
|
|
350
|
+
raise _schema_error(owner, py_name, "a child Name cannot contain '/'")
|
|
351
|
+
|
|
352
|
+
adapter_ann = _adapter_annotation(core, metadata, optional)
|
|
353
|
+
return FieldSpec(
|
|
354
|
+
py_name=py_name,
|
|
355
|
+
h5_name=h5_name,
|
|
356
|
+
kind=kind,
|
|
357
|
+
annotation=adapter_ann,
|
|
358
|
+
adapter=_build_adapter(owner, py_name, adapter_ann),
|
|
359
|
+
optional=optional,
|
|
360
|
+
default=default,
|
|
361
|
+
converter=attr_marker.converter if attr_marker is not None else None,
|
|
362
|
+
eager=eager_marker is not None,
|
|
363
|
+
member_type=member_type,
|
|
364
|
+
)
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
def _annotation_failure(cls: type, exc: Exception) -> Exception:
|
|
368
|
+
"""Classify a failure to resolve ``cls``'s annotations."""
|
|
369
|
+
if isinstance(exc, NameError) and _raised_by_annotation(exc, cls):
|
|
370
|
+
return _UnresolvedAnnotation(cls, exc.name, str(exc))
|
|
371
|
+
return SchemaError(f"{cls.__name__}: could not resolve annotations: {exc}")
|
|
372
|
+
|
|
373
|
+
|
|
374
|
+
def _resolved_annotations(cls: type, scope: Mapping[str, Any] | None = None) -> dict[str, Any]:
|
|
375
|
+
"""Return ``cls``'s own annotations, evaluating PEP 563 strings.
|
|
376
|
+
|
|
377
|
+
Names resolve against the defining module, then ``scope`` -- the live locals of
|
|
378
|
+
the defining frame while the class is being created, or the weak snapshot kept
|
|
379
|
+
by the deferred path -- then the class body.
|
|
380
|
+
"""
|
|
381
|
+
try:
|
|
382
|
+
annotations = inspect.get_annotations(cls, eval_str=False)
|
|
383
|
+
except Exception as exc:
|
|
384
|
+
# PEP 649 (3.14+): this call evaluates lazy annotations, so it fails here
|
|
385
|
+
# where older versions failed at the ``class`` statement itself.
|
|
386
|
+
raise _annotation_failure(cls, exc) from exc
|
|
387
|
+
if not any(isinstance(annotation, str) for annotation in annotations.values()):
|
|
388
|
+
return annotations
|
|
389
|
+
if scope is None:
|
|
390
|
+
scope = _unpack_scope(cls.__dict__.get("_h5t_scope", {}))
|
|
391
|
+
module = sys.modules.get(cls.__module__)
|
|
392
|
+
namespace: dict[str, Any] = dict(vars(module)) if module is not None else {}
|
|
393
|
+
namespace.update(scope)
|
|
394
|
+
namespace.update(vars(cls))
|
|
395
|
+
try:
|
|
396
|
+
return {
|
|
397
|
+
name: eval(compile(annotation, _ANNOTATION_FILE, "eval"), namespace)
|
|
398
|
+
if isinstance(annotation, str)
|
|
399
|
+
else annotation
|
|
400
|
+
for name, annotation in annotations.items()
|
|
401
|
+
}
|
|
402
|
+
except Exception as exc:
|
|
403
|
+
raise _annotation_failure(cls, exc) from exc
|
|
404
|
+
|
|
405
|
+
|
|
406
|
+
def _own_fields(
|
|
407
|
+
cls: type, *, dataset_owner: bool, scope: Mapping[str, Any] | None = None
|
|
408
|
+
) -> dict[str, FieldSpec]:
|
|
409
|
+
annotations = _resolved_annotations(cls, scope)
|
|
410
|
+
fields: dict[str, FieldSpec] = {}
|
|
411
|
+
for py_name, annotation in annotations.items():
|
|
412
|
+
if py_name.startswith("_") or typing.get_origin(annotation) is ClassVar:
|
|
413
|
+
continue
|
|
414
|
+
default = cls.__dict__.get(py_name, _NO_DEFAULT)
|
|
415
|
+
fields[py_name] = _field_spec(
|
|
416
|
+
cls,
|
|
417
|
+
py_name,
|
|
418
|
+
annotation,
|
|
419
|
+
default,
|
|
420
|
+
dataset_owner=dataset_owner,
|
|
421
|
+
)
|
|
422
|
+
return fields
|
|
423
|
+
|
|
424
|
+
|
|
425
|
+
def _merged_fields(cls: type) -> dict[str, FieldSpec]:
|
|
426
|
+
merged: dict[str, tuple[type, FieldSpec]] = {}
|
|
427
|
+
for candidate in reversed(cls.__mro__):
|
|
428
|
+
own = candidate.__dict__.get("_h5t_own")
|
|
429
|
+
if not own:
|
|
430
|
+
continue
|
|
431
|
+
for name, field in own.items():
|
|
432
|
+
previous = merged.get(name)
|
|
433
|
+
if previous is None:
|
|
434
|
+
merged[name] = (candidate, field)
|
|
435
|
+
continue
|
|
436
|
+
previous_owner, previous_field = previous
|
|
437
|
+
if issubclass(candidate, previous_owner):
|
|
438
|
+
merged[name] = (candidate, field)
|
|
439
|
+
elif not _fields_equivalent(field, previous_field):
|
|
440
|
+
raise SchemaError(
|
|
441
|
+
f"{cls.__name__}.{name}: conflicting declarations in bases "
|
|
442
|
+
f"{previous_owner.__name__} and {candidate.__name__}"
|
|
443
|
+
)
|
|
444
|
+
return {name: field for name, (_, field) in merged.items()}
|
|
445
|
+
|
|
446
|
+
|
|
447
|
+
def _fields_equivalent(left: FieldSpec, right: FieldSpec) -> bool:
|
|
448
|
+
"""Compare declarations without adapter identity or array-valued equality."""
|
|
449
|
+
return (
|
|
450
|
+
left.py_name == right.py_name
|
|
451
|
+
and left.h5_name == right.h5_name
|
|
452
|
+
and left.kind is right.kind
|
|
453
|
+
and repr(left.annotation) == repr(right.annotation)
|
|
454
|
+
and left.optional == right.optional
|
|
455
|
+
and repr(left.default) == repr(right.default)
|
|
456
|
+
and left.converter is right.converter
|
|
457
|
+
and left.eager == right.eager
|
|
458
|
+
and left.member_type is right.member_type
|
|
459
|
+
)
|
|
460
|
+
|
|
461
|
+
|
|
462
|
+
def _resolve_extras(cls: type) -> Extras:
|
|
463
|
+
requested = cls.__dict__.get("_h5t_extras")
|
|
464
|
+
if requested is not None:
|
|
465
|
+
try:
|
|
466
|
+
return Extras(requested)
|
|
467
|
+
except ValueError:
|
|
468
|
+
raise SchemaError(
|
|
469
|
+
f"{cls.__name__}: extras must be 'ignore' or 'forbid', got {requested!r}"
|
|
470
|
+
) from None
|
|
471
|
+
for base in cls.__mro__[1:]:
|
|
472
|
+
spec = base.__dict__.get("_h5t_spec")
|
|
473
|
+
if spec is not None:
|
|
474
|
+
return spec.extras
|
|
475
|
+
return Extras.IGNORE
|
|
476
|
+
|
|
477
|
+
|
|
478
|
+
def _reserved_names(base: type) -> frozenset[str]:
|
|
479
|
+
return frozenset(name for name in dir(base) if not name.startswith("_"))
|
|
480
|
+
|
|
481
|
+
|
|
482
|
+
def _check_fields(cls: type, own: Mapping[str, FieldSpec], fields: Mapping[str, FieldSpec]) -> None:
|
|
483
|
+
base = Dataset if issubclass(cls, Dataset) else Group
|
|
484
|
+
reserved = _reserved_names(base)
|
|
485
|
+
for name in own:
|
|
486
|
+
if name in reserved:
|
|
487
|
+
raise _schema_error(
|
|
488
|
+
cls,
|
|
489
|
+
name,
|
|
490
|
+
f"field shadows the h5t API; rename it and use Name({name!r})",
|
|
491
|
+
)
|
|
492
|
+
|
|
493
|
+
attr_names: dict[str, str] = {}
|
|
494
|
+
child_names: dict[str, str] = {}
|
|
495
|
+
for py_name, field in fields.items():
|
|
496
|
+
namespace = attr_names if field.kind is MemberKind.ATTRIBUTE else child_names
|
|
497
|
+
if field.h5_name in namespace:
|
|
498
|
+
raise SchemaError(
|
|
499
|
+
f"{cls.__name__}: duplicate HDF5 name {field.h5_name!r} for fields "
|
|
500
|
+
f"{namespace[field.h5_name]!r} and {py_name!r}"
|
|
501
|
+
)
|
|
502
|
+
namespace[field.h5_name] = py_name
|
|
503
|
+
|
|
504
|
+
|
|
505
|
+
def _compile_class(cls: SchemaMeta, scope: Mapping[str, Any] | None = None) -> None:
|
|
506
|
+
"""Compile ``cls`` into its own fields and its flattened schema."""
|
|
507
|
+
for base in cls.__mro__[1:]:
|
|
508
|
+
if isinstance(base, SchemaMeta) and "_h5t_spec" not in base.__dict__:
|
|
509
|
+
# _merged_fields and _resolve_extras read compiled state off the bases.
|
|
510
|
+
# A base that is still waiting on a forward reference defers ``cls`` too.
|
|
511
|
+
_compile_class(base)
|
|
512
|
+
own = _own_fields(cls, dataset_owner=issubclass(cls, Dataset), scope=scope)
|
|
513
|
+
cls._h5t_own = own
|
|
514
|
+
fields = _merged_fields(cls)
|
|
515
|
+
_check_fields(cls, own, fields)
|
|
516
|
+
cls._h5t_spec = ClassSpec(tuple(fields.values()), _resolve_extras(cls))
|
|
517
|
+
|
|
518
|
+
|
|
519
|
+
def _ensure_compiled(cls: SchemaMeta) -> None:
|
|
520
|
+
if "_h5t_spec" in cls.__dict__:
|
|
521
|
+
return
|
|
522
|
+
try:
|
|
523
|
+
_compile_class(cls)
|
|
524
|
+
except _UnresolvedAnnotation as exc:
|
|
525
|
+
raise SchemaError(f"{exc.owner.__name__}: could not resolve annotations: {exc}") from exc
|
|
526
|
+
|
|
527
|
+
|
|
528
|
+
class SchemaMeta(type):
|
|
529
|
+
"""Metaclass exposing a schema class's compiled spec.
|
|
530
|
+
|
|
531
|
+
Schemas compile at the ``class`` statement, so a bad declaration raises there.
|
|
532
|
+
This property is the fallback route for the classes that could not: one whose
|
|
533
|
+
annotations name something defined later compiles on the first read instead.
|
|
534
|
+
"""
|
|
535
|
+
|
|
536
|
+
_h5t_own: dict[str, FieldSpec]
|
|
537
|
+
_h5t_spec: ClassSpec
|
|
538
|
+
|
|
539
|
+
@property
|
|
540
|
+
def __h5spec__(cls) -> ClassSpec:
|
|
541
|
+
"""The flattened schema of this class, compiled on first access."""
|
|
542
|
+
_ensure_compiled(cls)
|
|
543
|
+
return cls._h5t_spec
|
|
544
|
+
|
|
545
|
+
|
|
546
|
+
def _short_pydantic_error(exc: PydanticValidationError) -> str:
|
|
547
|
+
errors = exc.errors(include_url=False, include_context=False, include_input=False)
|
|
548
|
+
if not errors:
|
|
549
|
+
return str(exc).splitlines()[0]
|
|
550
|
+
first = errors[0]
|
|
551
|
+
location = ".".join(str(part) for part in first.get("loc", ()))
|
|
552
|
+
message = str(first.get("msg", "invalid value"))
|
|
553
|
+
return f"{location}: {message}" if location else message
|
|
554
|
+
|
|
555
|
+
|
|
556
|
+
def _validate_value(field: FieldSpec, value: Any, path: str) -> Any:
|
|
557
|
+
try:
|
|
558
|
+
return field.adapter.validate_python(value)
|
|
559
|
+
except PydanticValidationError as exc:
|
|
560
|
+
raise ValidationError(path, _short_pydantic_error(exc)) from exc
|
|
561
|
+
|
|
562
|
+
|
|
563
|
+
def _missing_value(field: FieldSpec, path: str) -> Any:
|
|
564
|
+
if field.has_default:
|
|
565
|
+
return _validate_value(field, field.default, path)
|
|
566
|
+
if field.optional:
|
|
567
|
+
return None
|
|
568
|
+
kind = field.kind.value
|
|
569
|
+
raise ValidationError(path, f"required {kind} {field.h5_name!r} is missing")
|
|
570
|
+
|
|
571
|
+
|
|
572
|
+
def _raw_attrs(node: h5py.Group | h5py.Dataset) -> dict[str, Any]:
|
|
573
|
+
return {str(name): node.attrs[name] for name in node.attrs}
|
|
574
|
+
|
|
575
|
+
|
|
576
|
+
def _check_group_extras(spec: ClassSpec, group: h5py.Group, path: str) -> None:
|
|
577
|
+
if spec.extras is Extras.IGNORE:
|
|
578
|
+
return
|
|
579
|
+
attrs = {field.h5_name for field in spec.fields if field.kind is MemberKind.ATTRIBUTE}
|
|
580
|
+
children = {field.h5_name for field in spec.fields if field.kind is not MemberKind.ATTRIBUTE}
|
|
581
|
+
for name in group.attrs:
|
|
582
|
+
rendered = str(name)
|
|
583
|
+
if rendered not in attrs:
|
|
584
|
+
raise ValidationError(attr_path(path, rendered), "unexpected attribute")
|
|
585
|
+
for name in group.keys():
|
|
586
|
+
rendered = str(name)
|
|
587
|
+
if rendered not in children:
|
|
588
|
+
raise ValidationError(child_path(path, rendered), "unexpected child node")
|
|
589
|
+
|
|
590
|
+
|
|
591
|
+
def _check_dataset_extras(spec: ClassSpec, dataset: h5py.Dataset, path: str) -> None:
|
|
592
|
+
if spec.extras is Extras.IGNORE:
|
|
593
|
+
return
|
|
594
|
+
attrs = {field.h5_name for field in spec.fields}
|
|
595
|
+
for name in dataset.attrs:
|
|
596
|
+
rendered = str(name)
|
|
597
|
+
if rendered not in attrs:
|
|
598
|
+
raise ValidationError(attr_path(path, rendered), "unexpected attribute")
|
|
599
|
+
|
|
600
|
+
|
|
601
|
+
def _load_attribute(
|
|
602
|
+
field: FieldSpec,
|
|
603
|
+
raw: Mapping[str, Any],
|
|
604
|
+
attrs: dict[str, Any],
|
|
605
|
+
parent_path: str,
|
|
606
|
+
) -> Any:
|
|
607
|
+
path = attr_path(parent_path, field.h5_name)
|
|
608
|
+
if field.h5_name not in raw:
|
|
609
|
+
value = _missing_value(field, path)
|
|
610
|
+
else:
|
|
611
|
+
value = raw[field.h5_name]
|
|
612
|
+
if field.converter is not None:
|
|
613
|
+
try:
|
|
614
|
+
value = field.converter(value)
|
|
615
|
+
except Exception as exc:
|
|
616
|
+
raise ConversionError(path, f"attribute converter failed: {exc}") from exc
|
|
617
|
+
value = _validate_value(field, value, path)
|
|
618
|
+
attrs[field.h5_name] = value
|
|
619
|
+
return value
|
|
620
|
+
|
|
621
|
+
|
|
622
|
+
def _load_dataset(
|
|
623
|
+
dataset_type: type[Dataset],
|
|
624
|
+
dataset: h5py.Dataset,
|
|
625
|
+
filename: str,
|
|
626
|
+
path: str,
|
|
627
|
+
) -> Dataset:
|
|
628
|
+
spec = dataset_type.__h5spec__
|
|
629
|
+
_check_dataset_extras(spec, dataset, path)
|
|
630
|
+
raw = _raw_attrs(dataset)
|
|
631
|
+
attrs = dict(raw)
|
|
632
|
+
values: dict[str, Any] = {}
|
|
633
|
+
for field in spec.fields:
|
|
634
|
+
values[field.py_name] = _load_attribute(field, raw, attrs, path)
|
|
635
|
+
|
|
636
|
+
instance = object.__new__(dataset_type)
|
|
637
|
+
object.__setattr__(instance, "_h5t_filename", filename)
|
|
638
|
+
object.__setattr__(instance, "_h5t_path", path)
|
|
639
|
+
object.__setattr__(instance, "_h5t_shape", tuple(dataset.shape))
|
|
640
|
+
object.__setattr__(instance, "_h5t_dtype", np.dtype(dataset.dtype))
|
|
641
|
+
object.__setattr__(instance, "_h5t_attrs", MappingProxyType(attrs))
|
|
642
|
+
object.__setattr__(instance, "_h5t_data", _DATA_NOT_LOADED)
|
|
643
|
+
for name, value in values.items():
|
|
644
|
+
object.__setattr__(instance, name, value)
|
|
645
|
+
return instance
|
|
646
|
+
|
|
647
|
+
|
|
648
|
+
def _load_group(group_type: type[Group], group: h5py.Group, filename: str, path: str) -> Group:
|
|
649
|
+
spec = group_type.__h5spec__
|
|
650
|
+
_check_group_extras(spec, group, path)
|
|
651
|
+
raw = _raw_attrs(group)
|
|
652
|
+
attrs = dict(raw)
|
|
653
|
+
values: dict[str, Any] = {}
|
|
654
|
+
|
|
655
|
+
for field in spec.fields:
|
|
656
|
+
if field.kind is MemberKind.ATTRIBUTE:
|
|
657
|
+
values[field.py_name] = _load_attribute(field, raw, attrs, path)
|
|
658
|
+
continue
|
|
659
|
+
|
|
660
|
+
member_path = child_path(path, field.h5_name)
|
|
661
|
+
if field.h5_name not in group:
|
|
662
|
+
values[field.py_name] = _missing_value(field, member_path)
|
|
663
|
+
continue
|
|
664
|
+
node = group[field.h5_name]
|
|
665
|
+
if field.kind is MemberKind.GROUP:
|
|
666
|
+
if not isinstance(node, h5py.Group):
|
|
667
|
+
raise ValidationError(member_path, "expected a group, found a dataset")
|
|
668
|
+
assert field.member_type is not None
|
|
669
|
+
nested_group_type = typing.cast(type[Group], field.member_type)
|
|
670
|
+
value = _load_group(nested_group_type, node, filename, member_path)
|
|
671
|
+
elif field.kind is MemberKind.DATASET:
|
|
672
|
+
if not isinstance(node, h5py.Dataset):
|
|
673
|
+
raise ValidationError(member_path, "expected a dataset, found a group")
|
|
674
|
+
assert field.member_type is not None
|
|
675
|
+
dataset_type = typing.cast(type[Dataset], field.member_type)
|
|
676
|
+
value = _load_dataset(dataset_type, node, filename, member_path)
|
|
677
|
+
if field.eager:
|
|
678
|
+
object.__setattr__(value, "_h5t_data", np.asarray(node[()]))
|
|
679
|
+
else:
|
|
680
|
+
if not isinstance(node, h5py.Dataset):
|
|
681
|
+
raise ValidationError(member_path, "expected a dataset, found a group")
|
|
682
|
+
value = np.asarray(node[()])
|
|
683
|
+
values[field.py_name] = _validate_value(field, value, member_path)
|
|
684
|
+
|
|
685
|
+
instance = object.__new__(group_type)
|
|
686
|
+
object.__setattr__(instance, "_h5t_attrs", MappingProxyType(attrs))
|
|
687
|
+
for name, value in values.items():
|
|
688
|
+
object.__setattr__(instance, name, value)
|
|
689
|
+
return instance
|
|
690
|
+
|
|
691
|
+
|
|
692
|
+
class _Record(metaclass=SchemaMeta):
|
|
693
|
+
"""Shared declaration handling and repr for detached schema records."""
|
|
694
|
+
|
|
695
|
+
_h5t_extras: ClassVar[str | None] = None
|
|
696
|
+
_h5t_scope: ClassVar[Mapping[str, Any]] = {}
|
|
697
|
+
|
|
698
|
+
def __init_subclass__(
|
|
699
|
+
cls,
|
|
700
|
+
*,
|
|
701
|
+
extras: Literal["ignore", "forbid"] | None = None,
|
|
702
|
+
**kwargs: Any,
|
|
703
|
+
) -> None:
|
|
704
|
+
super().__init_subclass__(**kwargs)
|
|
705
|
+
cls._h5t_extras = extras
|
|
706
|
+
if cls.__module__ == __name__:
|
|
707
|
+
# Group and Dataset themselves: the module globals _compile_class reads
|
|
708
|
+
# (issubclass(cls, Dataset)) are not bound yet. They compile on first use.
|
|
709
|
+
return
|
|
710
|
+
# The defining frame is still live, so its locals resolve annotations naming
|
|
711
|
+
# function-local classes without retaining anything. Only a genuine forward
|
|
712
|
+
# reference defers, and only that path keeps a (weak) snapshot of the scope.
|
|
713
|
+
frame = inspect.currentframe()
|
|
714
|
+
scope: Mapping[str, Any] = (
|
|
715
|
+
frame.f_back.f_locals if frame is not None and frame.f_back else {}
|
|
716
|
+
)
|
|
717
|
+
try:
|
|
718
|
+
_compile_class(cls, scope)
|
|
719
|
+
except _UnresolvedAnnotation:
|
|
720
|
+
cls._h5t_scope = _weak_scope(scope, _annotation_names(cls))
|
|
721
|
+
|
|
722
|
+
def _h5t_repr_fields(self) -> list[tuple[str, Any]]:
|
|
723
|
+
spec = type(self).__h5spec__
|
|
724
|
+
return [(field.py_name, getattr(self, field.py_name)) for field in spec.fields]
|
|
725
|
+
|
|
726
|
+
def __repr__(self) -> str:
|
|
727
|
+
fields = ", ".join(
|
|
728
|
+
f"{name}={reprlib.repr(value)}" for name, value in self._h5t_repr_fields()
|
|
729
|
+
)
|
|
730
|
+
return f"{type(self).__name__}({fields})"
|
|
731
|
+
|
|
732
|
+
|
|
733
|
+
class Dataset(_Record):
|
|
734
|
+
"""A detached HDF5 dataset with snapshotted metadata and lazy payload data."""
|
|
735
|
+
|
|
736
|
+
_h5t_filename: str
|
|
737
|
+
_h5t_path: str
|
|
738
|
+
_h5t_shape: tuple[int, ...]
|
|
739
|
+
_h5t_dtype: np.dtype[Any]
|
|
740
|
+
_h5t_attrs: Mapping[str, Any]
|
|
741
|
+
_h5t_data: object
|
|
742
|
+
|
|
743
|
+
def __init__(self, *args: Any, **kwargs: Any) -> None:
|
|
744
|
+
raise TypeError(f"{type(self).__name__} objects are created by Group.from_file()")
|
|
745
|
+
|
|
746
|
+
@property
|
|
747
|
+
def attrs(self) -> Mapping[str, Any]:
|
|
748
|
+
"""Immutable snapshot of all dataset attributes."""
|
|
749
|
+
return self._h5t_attrs
|
|
750
|
+
|
|
751
|
+
@property
|
|
752
|
+
def path(self) -> str:
|
|
753
|
+
"""Absolute HDF5 path of this dataset."""
|
|
754
|
+
return self._h5t_path
|
|
755
|
+
|
|
756
|
+
@property
|
|
757
|
+
def shape(self) -> tuple[int, ...]:
|
|
758
|
+
"""Dataset shape captured while the model was loaded."""
|
|
759
|
+
return self._h5t_shape
|
|
760
|
+
|
|
761
|
+
@property
|
|
762
|
+
def dtype(self) -> np.dtype[Any]:
|
|
763
|
+
"""Dataset dtype captured while the model was loaded."""
|
|
764
|
+
return self._h5t_dtype
|
|
765
|
+
|
|
766
|
+
@property
|
|
767
|
+
def ndim(self) -> int:
|
|
768
|
+
"""Number of dimensions in the captured shape."""
|
|
769
|
+
return len(self._h5t_shape)
|
|
770
|
+
|
|
771
|
+
@property
|
|
772
|
+
def data(self) -> np.ndarray:
|
|
773
|
+
"""Read and cache the complete current payload as a NumPy array."""
|
|
774
|
+
if self._h5t_data is _DATA_NOT_LOADED:
|
|
775
|
+
with self.open() as dataset:
|
|
776
|
+
value = np.asarray(dataset[()])
|
|
777
|
+
object.__setattr__(self, "_h5t_data", value)
|
|
778
|
+
return typing.cast(np.ndarray, self._h5t_data)
|
|
779
|
+
|
|
780
|
+
def read(self) -> np.ndarray:
|
|
781
|
+
"""Return the same cached complete payload as ``data``."""
|
|
782
|
+
return self.data
|
|
783
|
+
|
|
784
|
+
@contextmanager
|
|
785
|
+
def open(self) -> Iterator[h5py.Dataset]:
|
|
786
|
+
"""Open the current source file and yield this dataset for live access."""
|
|
787
|
+
with h5py.File(self._h5t_filename, mode="r") as h5file:
|
|
788
|
+
node = h5file.get(self._h5t_path)
|
|
789
|
+
if node is None:
|
|
790
|
+
raise ValidationError(self._h5t_path, "dataset no longer exists")
|
|
791
|
+
if not isinstance(node, h5py.Dataset):
|
|
792
|
+
raise ValidationError(self._h5t_path, "expected a dataset, found a group")
|
|
793
|
+
yield node
|
|
794
|
+
|
|
795
|
+
def _h5t_repr_fields(self) -> list[tuple[str, Any]]:
|
|
796
|
+
fields: list[tuple[str, Any]] = [
|
|
797
|
+
("path", self.path),
|
|
798
|
+
("shape", self.shape),
|
|
799
|
+
("dtype", self.dtype),
|
|
800
|
+
]
|
|
801
|
+
fields.extend(super()._h5t_repr_fields())
|
|
802
|
+
return fields
|
|
803
|
+
|
|
804
|
+
|
|
805
|
+
class Group(_Record):
|
|
806
|
+
"""Base class for detached, typed HDF5 group records."""
|
|
807
|
+
|
|
808
|
+
_h5t_attrs: Mapping[str, Any]
|
|
809
|
+
|
|
810
|
+
def __init__(self, *args: Any, **kwargs: Any) -> None:
|
|
811
|
+
raise TypeError(
|
|
812
|
+
f"{type(self).__name__} objects are created by {type(self).__name__}.from_file()"
|
|
813
|
+
)
|
|
814
|
+
|
|
815
|
+
@property
|
|
816
|
+
def attrs(self) -> Mapping[str, Any]:
|
|
817
|
+
"""Immutable snapshot of all group attributes."""
|
|
818
|
+
return self._h5t_attrs
|
|
819
|
+
|
|
820
|
+
@classmethod
|
|
821
|
+
def from_file(cls, path: str | os.PathLike[str], root: str = "/") -> Self:
|
|
822
|
+
"""Load this group schema from ``root`` and close the HDF5 file."""
|
|
823
|
+
_ensure_compiled(cls) # compile a deferred schema before touching the filesystem
|
|
824
|
+
try:
|
|
825
|
+
filesystem_path = os.fsdecode(os.fspath(path))
|
|
826
|
+
except TypeError as exc:
|
|
827
|
+
raise TypeError("path must be a filesystem path") from exc
|
|
828
|
+
if not isinstance(root, str) or not posixpath.isabs(root):
|
|
829
|
+
raise ValueError("root must be an absolute HDF5 group path")
|
|
830
|
+
normalized_root = posixpath.normpath(root)
|
|
831
|
+
if normalized_root.startswith("//"):
|
|
832
|
+
normalized_root = "/" + normalized_root.lstrip("/")
|
|
833
|
+
filename = os.path.abspath(filesystem_path)
|
|
834
|
+
with h5py.File(filename, mode="r") as h5file:
|
|
835
|
+
node = h5file.get(normalized_root)
|
|
836
|
+
if node is None:
|
|
837
|
+
raise ValidationError(normalized_root, "root group does not exist")
|
|
838
|
+
if not isinstance(node, h5py.Group):
|
|
839
|
+
raise ValidationError(normalized_root, "expected a group, found a dataset")
|
|
840
|
+
return typing.cast(Self, _load_group(cls, node, filename, normalized_root))
|
h5t/_errors.py
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
"""Public exception types for schema compilation and detached loading."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class H5TError(Exception):
|
|
7
|
+
"""Base class for h5t errors."""
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class SchemaError(H5TError):
|
|
11
|
+
"""A schema class declaration is incoherent or unsupported."""
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class ValidationError(H5TError):
|
|
15
|
+
"""An HDF5 node does not conform to a compiled schema."""
|
|
16
|
+
|
|
17
|
+
def __init__(self, path: str, message: str) -> None:
|
|
18
|
+
super().__init__(f"{path}: {message}")
|
|
19
|
+
self.path = path
|
|
20
|
+
self.message = message
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class ConversionError(ValidationError):
|
|
24
|
+
"""An explicit ``Attr`` converter failed."""
|
h5t/_spec.py
ADDED
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
"""Compiled schema records and public annotation markers."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Callable
|
|
6
|
+
from dataclasses import dataclass, field
|
|
7
|
+
from enum import Enum
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
from pydantic import TypeAdapter
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass(frozen=True)
|
|
14
|
+
class Name:
|
|
15
|
+
"""Use ``name`` for this member in the HDF5 file."""
|
|
16
|
+
|
|
17
|
+
name: str
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass(frozen=True)
|
|
21
|
+
class Attr:
|
|
22
|
+
"""Declare a field as an HDF5 attribute.
|
|
23
|
+
|
|
24
|
+
``converter`` is applied to the value returned by h5py before Pydantic
|
|
25
|
+
validation. It is useful for serialized attributes such as JSON strings.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
converter: Callable[[Any], Any] | None = None
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass(frozen=True)
|
|
32
|
+
class Eager:
|
|
33
|
+
"""Load a detached dataset's complete payload during ``from_file``."""
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class Extras(Enum):
|
|
37
|
+
"""Policy for undeclared immediate HDF5 members."""
|
|
38
|
+
|
|
39
|
+
IGNORE = "ignore"
|
|
40
|
+
FORBID = "forbid"
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class MemberKind(Enum):
|
|
44
|
+
"""How a declared field is represented in HDF5."""
|
|
45
|
+
|
|
46
|
+
ATTRIBUTE = "attribute"
|
|
47
|
+
ARRAY = "array"
|
|
48
|
+
DATASET = "dataset"
|
|
49
|
+
GROUP = "group"
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
_NO_DEFAULT = object()
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@dataclass(frozen=True)
|
|
56
|
+
class FieldSpec:
|
|
57
|
+
"""The compiled loading instructions for one annotated field."""
|
|
58
|
+
|
|
59
|
+
py_name: str
|
|
60
|
+
h5_name: str
|
|
61
|
+
kind: MemberKind
|
|
62
|
+
annotation: Any
|
|
63
|
+
adapter: TypeAdapter[Any] = field(compare=False, repr=False)
|
|
64
|
+
optional: bool = False
|
|
65
|
+
default: Any = _NO_DEFAULT
|
|
66
|
+
converter: Callable[[Any], Any] | None = None
|
|
67
|
+
eager: bool = False
|
|
68
|
+
member_type: type | None = None
|
|
69
|
+
|
|
70
|
+
@property
|
|
71
|
+
def has_default(self) -> bool:
|
|
72
|
+
"""Whether the class body supplied a default."""
|
|
73
|
+
return self.default is not _NO_DEFAULT
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
@dataclass(frozen=True)
|
|
77
|
+
class ClassSpec:
|
|
78
|
+
"""The flattened schema compiled for a ``Group`` or ``Dataset`` class."""
|
|
79
|
+
|
|
80
|
+
fields: tuple[FieldSpec, ...] = ()
|
|
81
|
+
extras: Extras = Extras.IGNORE
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def child_path(parent: str, name: str) -> str:
|
|
85
|
+
"""Join an absolute group path and one immediate child name."""
|
|
86
|
+
return f"/{name}" if parent == "/" else f"{parent.rstrip('/')}/{name}"
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def attr_path(parent: str, name: str) -> str:
|
|
90
|
+
"""Return a display path for an HDF5 attribute."""
|
|
91
|
+
return f"{parent.rstrip('/') or '/'}@{name}"
|
h5t/py.typed
ADDED
|
File without changes
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: h5t
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Load HDF5 groups as detached, typed Python records.
|
|
5
|
+
Keywords: hdf5,h5py,schema,pydantic,typing,validation
|
|
6
|
+
Author: binado
|
|
7
|
+
Author-email: binado <bernardopveronese@gmail.com>
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Intended Audience :: Developers
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: Operating System :: OS Independent
|
|
14
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
19
|
+
Classifier: Topic :: Scientific/Engineering
|
|
20
|
+
Classifier: Topic :: Software Development :: Libraries
|
|
21
|
+
Classifier: Typing :: Typed
|
|
22
|
+
Requires-Dist: h5py>=3.10
|
|
23
|
+
Requires-Dist: numpy>=1.26
|
|
24
|
+
Requires-Dist: pydantic>=2.4,<3
|
|
25
|
+
Requires-Python: >=3.11
|
|
26
|
+
Project-URL: Homepage, https://github.com/binado/h5t
|
|
27
|
+
Project-URL: Repository, https://github.com/binado/h5t
|
|
28
|
+
Project-URL: Issues, https://github.com/binado/h5t/issues
|
|
29
|
+
Project-URL: Changelog, https://github.com/binado/h5t/blob/main/CHANGELOG.md
|
|
30
|
+
Description-Content-Type: text/markdown
|
|
31
|
+
|
|
32
|
+
# h5t
|
|
33
|
+
|
|
34
|
+
[](https://pypi.org/project/h5t/)
|
|
35
|
+
[](https://pypi.org/project/h5t/)
|
|
36
|
+
[](https://github.com/binado/h5t/actions/workflows/ci.yml)
|
|
37
|
+
[](https://github.com/binado/h5t/blob/main/LICENSE)
|
|
38
|
+
|
|
39
|
+
`h5t` loads HDF5 groups into detached, typed Python records. Group attributes and
|
|
40
|
+
ordinary NumPy-array fields are materialized while the file is open. Typed datasets
|
|
41
|
+
keep snapshot metadata and can read their payload lazily without retaining an open
|
|
42
|
+
file descriptor.
|
|
43
|
+
|
|
44
|
+
## Installation
|
|
45
|
+
|
|
46
|
+
```bash
|
|
47
|
+
pip install h5t # or: uv add h5t
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
Requires Python 3.11+. `h5py`, `numpy`, and Pydantic v2 are installed automatically.
|
|
51
|
+
|
|
52
|
+
## Example
|
|
53
|
+
|
|
54
|
+
```python
|
|
55
|
+
import json
|
|
56
|
+
from pathlib import Path
|
|
57
|
+
from typing import Annotated, Any
|
|
58
|
+
|
|
59
|
+
import numpy as np
|
|
60
|
+
|
|
61
|
+
import h5t
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class Measurement(h5t.Dataset, extras="forbid"):
|
|
65
|
+
unit: str
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
class Nested(h5t.Group):
|
|
69
|
+
label: str
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
class Result(h5t.Group, extras="ignore"):
|
|
73
|
+
version: int # implicit HDF5 attribute
|
|
74
|
+
title: Annotated[str, h5t.Name("name")] # renamed attribute
|
|
75
|
+
config: Annotated[
|
|
76
|
+
dict[str, Any],
|
|
77
|
+
h5t.Attr(converter=json.loads),
|
|
78
|
+
]
|
|
79
|
+
array_attr: Annotated[np.ndarray, h5t.Attr()]
|
|
80
|
+
values: np.ndarray # eager dataset payload
|
|
81
|
+
measurement: Measurement # lazy detached dataset
|
|
82
|
+
eager_measurement: Annotated[Measurement, h5t.Eager()]
|
|
83
|
+
nested: Nested # recursively loaded group
|
|
84
|
+
note: str | None # absent becomes None
|
|
85
|
+
revision: int = 1 # absent uses a validated default
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
result = Result.from_file(Path("result.h5"), root="/")
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
`result.attrs` and `result.measurement.attrs` are immutable mappings keyed by their
|
|
92
|
+
on-disk HDF5 names. Declared attributes contain their Pydantic-processed values;
|
|
93
|
+
undeclared attributes retained under `extras="ignore"` contain the raw h5py values.
|
|
94
|
+
|
|
95
|
+
Typed datasets expose snapshot metadata and explicit data access:
|
|
96
|
+
|
|
97
|
+
```python
|
|
98
|
+
dataset = result.measurement
|
|
99
|
+
dataset.path, dataset.shape, dataset.dtype, dataset.ndim
|
|
100
|
+
|
|
101
|
+
complete = dataset.data # first access reads and caches an ndarray
|
|
102
|
+
assert dataset.read() is complete
|
|
103
|
+
|
|
104
|
+
with dataset.open() as live:
|
|
105
|
+
first_hundred = live[:100] # fresh file view, useful for slices
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
Both `from_file()` and `Dataset.open()` close every handle on normal and exceptional
|
|
109
|
+
exits. Schema objects cannot be directly constructed, written, or serialized by h5t.
|
|
110
|
+
|
|
111
|
+
## Field rules
|
|
112
|
+
|
|
113
|
+
| Annotation | HDF5 representation | Loading behavior |
|
|
114
|
+
| --- | --- | --- |
|
|
115
|
+
| `Group` subclass | child group | recursively materialized |
|
|
116
|
+
| `Dataset` or subclass | child dataset | metadata/attrs snapshot, payload lazy |
|
|
117
|
+
| `Annotated[DatasetSubtype, Eager()]` | child dataset | complete payload cached during loading |
|
|
118
|
+
| `np.ndarray` | child dataset | complete payload loaded as an ndarray |
|
|
119
|
+
| `Annotated[T, Attr(...)]` | attribute | converter, then Pydantic validation |
|
|
120
|
+
| scalar or `Literal[...]` | attribute | Pydantic validation |
|
|
121
|
+
|
|
122
|
+
`Name("stored-name")` renames any field kind. `Attr` is valid only for attributes and
|
|
123
|
+
`Eager` only for typed datasets. A parameterized alias such as
|
|
124
|
+
`numpy.typing.NDArray[np.floating]` is accepted wherever `np.ndarray` is; the dtype
|
|
125
|
+
parameter is not validated. Unsupported collection-shaped child annotations raise
|
|
126
|
+
`SchemaError`; dynamic collections are not yet supported. Declarations are compiled at
|
|
127
|
+
the `class` statement, so a `SchemaError` surfaces there. A schema class that names a
|
|
128
|
+
class defined later in its module instead compiles on first use.
|
|
129
|
+
|
|
130
|
+
Each `Group` and `Dataset` subclass accepts `extras="ignore"` (the default) or
|
|
131
|
+
`extras="forbid"`. A group policy applies to its immediate child and attribute
|
|
132
|
+
names. A typed dataset policy applies to its attributes. Nested schemas keep their own
|
|
133
|
+
policy, while a plain `np.ndarray` field never checks the dataset's attributes.
|
|
134
|
+
|
|
135
|
+
## Snapshot consistency
|
|
136
|
+
|
|
137
|
+
A loaded model is a snapshot, with one deliberate exception:
|
|
138
|
+
|
|
139
|
+
- Group and dataset attributes, dataset shape/dtype/path, eager datasets, and plain
|
|
140
|
+
arrays reflect the file during `from_file()`.
|
|
141
|
+
- A lazy dataset's first `.data`/`.read()` observes the file at that later moment and
|
|
142
|
+
caches the resulting array permanently.
|
|
143
|
+
- `.open()` always opens the current file and current dataset. It may therefore observe
|
|
144
|
+
replacements or fail if the source was changed or deleted.
|
|
145
|
+
|
|
146
|
+
## CLI
|
|
147
|
+
|
|
148
|
+
```bash
|
|
149
|
+
h5t check result.h5 --schema mypackage.schemas:Result --root /results/latest
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
Exit status is 0 on success, 1 for the first file/schema mismatch, and 2 for import,
|
|
153
|
+
declaration, usage, or I/O errors.
|
|
154
|
+
|
|
155
|
+
## Development
|
|
156
|
+
|
|
157
|
+
```bash
|
|
158
|
+
uv sync
|
|
159
|
+
uv run pytest
|
|
160
|
+
uv run ty check
|
|
161
|
+
uv run ruff check .
|
|
162
|
+
```
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
h5t/__init__.py,sha256=DcO56_5dRsigvAw6Wq_RFgOJCG4T-uyx27donmFN84A,421
|
|
2
|
+
h5t/_cli.py,sha256=4sVwhS038-u7wIodHEQDu3RMDNkkqXmhvkwrLaZ5bD4,2356
|
|
3
|
+
h5t/_compile.py,sha256=yY67PE3xQSVdTKKsdsJOSLBAA4764uZeiuDk8Pzg9Bw,32925
|
|
4
|
+
h5t/_errors.py,sha256=Pl-MbGtd1UECA66opjqjo0CBb6LVsyPhW8SBRMyfZDs,623
|
|
5
|
+
h5t/_spec.py,sha256=3JTJwx65eYw7Wf7bEZEhxij8XUeLmb-HYfpxyF7ujkg,2211
|
|
6
|
+
h5t/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
7
|
+
h5t-0.2.0.dist-info/licenses/LICENSE,sha256=qNMhJr4DHkvGr-u5gkjTT4rHf6Jx2fvMXQdmQgr_vNM,1063
|
|
8
|
+
h5t-0.2.0.dist-info/WHEEL,sha256=WKk67IZPzp9uSkBO-iWoZzrqp9-AwH7FE9EB8EoJbf4,81
|
|
9
|
+
h5t-0.2.0.dist-info/entry_points.txt,sha256=my7HaIw3FZOKSZS2vaE4G0BHPUMHPCS1g1vR338bA2s,39
|
|
10
|
+
h5t-0.2.0.dist-info/METADATA,sha256=hwqG_eX_H5Nhn1HpaFbisziCvaw_o2sqgDfDlDxjrVk,6159
|
|
11
|
+
h5t-0.2.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 binado
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|