hdfmap 0.4__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hdfmap/__init__.py +76 -0
- hdfmap/eval_functions.py +172 -0
- hdfmap/file_functions.py +210 -0
- hdfmap/hdfmap_class.py +656 -0
- hdfmap/logging.py +40 -0
- hdfmap/nexus.py +240 -0
- hdfmap/reloader_class.py +140 -0
- hdfmap-0.4.dist-info/LICENSE +201 -0
- hdfmap-0.4.dist-info/METADATA +476 -0
- hdfmap-0.4.dist-info/RECORD +12 -0
- hdfmap-0.4.dist-info/WHEEL +5 -0
- hdfmap-0.4.dist-info/top_level.txt +1 -0
hdfmap/hdfmap_class.py
ADDED
|
@@ -0,0 +1,656 @@
|
|
|
1
|
+
"""
|
|
2
|
+
hdfmap class definition
|
|
3
|
+
"""
|
|
4
|
+
import typing
|
|
5
|
+
from collections import defaultdict
|
|
6
|
+
from types import SimpleNamespace
|
|
7
|
+
|
|
8
|
+
import numpy as np
|
|
9
|
+
import h5py
|
|
10
|
+
|
|
11
|
+
from .logging import create_logger
|
|
12
|
+
from .eval_functions import expression_safe_name, extra_hdf_data, eval_hdf, format_hdf, dataset2data
|
|
13
|
+
|
|
14
|
+
try:
|
|
15
|
+
import hdf5plugin # required for compressed data
|
|
16
|
+
except ImportError:
|
|
17
|
+
print('Warning: hdf5plugin not available.')
|
|
18
|
+
|
|
19
|
+
# parameters
|
|
20
|
+
SEP = '/' # HDF path separator
|
|
21
|
+
LOCAL_NAME = 'local_name' # dataset attribute name for alt_name
|
|
22
|
+
OMIT = '/value' # omit this name in paths when determining identifier
|
|
23
|
+
|
|
24
|
+
# logger
|
|
25
|
+
logger = create_logger(__name__)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class Group(typing.NamedTuple):
|
|
29
|
+
nx_class: str
|
|
30
|
+
name: str
|
|
31
|
+
attrs: dict
|
|
32
|
+
datasets: list[str]
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class Dataset(typing.NamedTuple):
|
|
36
|
+
name: str
|
|
37
|
+
names: list[str]
|
|
38
|
+
size: int
|
|
39
|
+
shape: tuple[int]
|
|
40
|
+
attrs: dict
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def generate_identifier(hdf_path: str | bytes) -> str:
|
|
44
|
+
"""
|
|
45
|
+
Generate a valid python identifier from a hdf dataset path or other string
|
|
46
|
+
- Decodes to ascii
|
|
47
|
+
- omits '/value'
|
|
48
|
+
- splits by path separator (/) and takes final element
|
|
49
|
+
- converts special characters to '_'
|
|
50
|
+
- removes replication of strings separated by '_'
|
|
51
|
+
E.G.
|
|
52
|
+
/entry/group/motor1 >> "motor1"
|
|
53
|
+
/entry/group/motor/value >> "motor"
|
|
54
|
+
/entry/group/subgroup.motor >> "subgroup_motor"
|
|
55
|
+
motor.motor >> "motor"
|
|
56
|
+
:param hdf_path: str hdf path address
|
|
57
|
+
:return: str expression safe name
|
|
58
|
+
"""
|
|
59
|
+
if hasattr(hdf_path, 'decode'): # Byte string
|
|
60
|
+
hdf_path = hdf_path.decode('ascii')
|
|
61
|
+
if hdf_path.endswith(OMIT):
|
|
62
|
+
hdf_path = hdf_path[:-len(OMIT)] # omit 'value'
|
|
63
|
+
substrings = hdf_path.split(SEP)
|
|
64
|
+
name = expression_safe_name(substrings[-1])
|
|
65
|
+
# remove replication (handles local_names 'name.name' convention)
|
|
66
|
+
return '_'.join(dict.fromkeys(name.split('_')))
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def build_hdf_path(*args: str | bytes) -> str:
|
|
70
|
+
"""
|
|
71
|
+
Build path from string or bytes arguments
|
|
72
|
+
'/entry/measurement' = build_hdf_path(b'entry', 'measurement')
|
|
73
|
+
:param args: str or bytes arguments
|
|
74
|
+
:return: str hdf path
|
|
75
|
+
"""
|
|
76
|
+
return SEP + SEP.join((arg.decode() if isinstance(arg, bytes) else arg).strip(SEP) for arg in args)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def disp_dict(mydict: dict, indent: int = 10) -> str:
|
|
80
|
+
return '\n'.join([f"{key:>{indent}}: {value}" for key, value in mydict.items()])
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
class DataHolder(SimpleNamespace):
|
|
84
|
+
"""
|
|
85
|
+
Convert dict to class that looks like a class object with key names as attributes
|
|
86
|
+
Replicates slightly the old scisoftpy.dictutils.DataHolder class, also known as DLS dat format.
|
|
87
|
+
obj = DataHolder(**{'item1': 'value1'})
|
|
88
|
+
obj['item1'] -> 'value1'
|
|
89
|
+
obj.item1 -> 'value1'
|
|
90
|
+
"""
|
|
91
|
+
|
|
92
|
+
def __getitem__(self, item):
|
|
93
|
+
return self.__dict__.__getitem__(item)
|
|
94
|
+
|
|
95
|
+
def __iter__(self):
|
|
96
|
+
return self.__dict__.__iter__()
|
|
97
|
+
|
|
98
|
+
def keys(self):
|
|
99
|
+
return self.__dict__.keys()
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
class HdfMap:
|
|
103
|
+
"""
|
|
104
|
+
HdfMap object, container for paths of different objects in an HDF file
|
|
105
|
+
|
|
106
|
+
with h5py.File('file.hdf') as hdf:
|
|
107
|
+
map = HdfMap(hdf)
|
|
108
|
+
|
|
109
|
+
map.get_path('data') -> '/entry/measurement/data'
|
|
110
|
+
map['data'] -> '/entry/measurement/data'
|
|
111
|
+
|
|
112
|
+
with h5py.File('another_file.hdf') as hdf:
|
|
113
|
+
data = map.get_data(hdf, 'data')
|
|
114
|
+
array = map.get_scannables_array(hdf)
|
|
115
|
+
metadata = map.get_metadata(hdf)
|
|
116
|
+
out = map.eval(hdf, 'data / 10')
|
|
117
|
+
outstr = map.format(hdf, 'the data looks like: {data}')
|
|
118
|
+
|
|
119
|
+
Objects within the HDF file are separated into Groups and Datasets. Each object has a
|
|
120
|
+
defined 'path' and 'name' paramater, as well as other attributes
|
|
121
|
+
path -> '/entry/measurement/data' -> the location of an object within the file
|
|
122
|
+
name -> 'data' -> an path expressed as a simple variable name
|
|
123
|
+
Paths are unique location within the file but can be used to identify similar objects in other files
|
|
124
|
+
Names may not be unique within a file and are generated from the final element of the hdf path.
|
|
125
|
+
- When multiple paths produce the same name, the name is overwritten each time, so the last path in the
|
|
126
|
+
file has priority.
|
|
127
|
+
- Names are also stored using the 'local_name' attribute, if it exists
|
|
128
|
+
|
|
129
|
+
Names of different types of datasets are stored for arrays (size > 0) and values (size 0)
|
|
130
|
+
Names for scannables relate to all arrays of a particular size
|
|
131
|
+
A combined list of names is provided where scannables > arrays > values
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
Attributes:
|
|
136
|
+
map.groups stores attributes of each group by path
|
|
137
|
+
map.classes stores list of group paths by nx_class
|
|
138
|
+
map.datasets stores attributes of each dataset by path
|
|
139
|
+
map.arrays stores array dataset paths by name
|
|
140
|
+
map.values stores value dataset paths by name
|
|
141
|
+
map.scannables stores array dataset paths with given size, by name
|
|
142
|
+
map.combined stores array and value paths (arrays overwrite values)
|
|
143
|
+
map.image_data stores dataset paths of image data
|
|
144
|
+
E.G.
|
|
145
|
+
map.groups = {'/hdf/group': ('class', 'name', {attrs}, [datasets])}
|
|
146
|
+
map.classes = {'class_name': ['/hdf/group1', '/hdf/group2']}
|
|
147
|
+
map.datasets = {'/hdf/group/dataset': ('name', size, shape, {attrs})}
|
|
148
|
+
map.arrays = {'name': '/hdf/group/dataset'}
|
|
149
|
+
map.values = {'name': '/hdf/group/dataset'}
|
|
150
|
+
map.scannables = {'name': '/hdf/group/dataset'}
|
|
151
|
+
map.image_data = {'name': '/hdf/group/dataset'}
|
|
152
|
+
|
|
153
|
+
Methods:
|
|
154
|
+
map.populate(h5py.File) -> populates the dictionaries using the given file
|
|
155
|
+
map.generate_scannables(array_size) -> populates scannables namespace with arrays of same size
|
|
156
|
+
map.most_common_size -> returns the most common dataset size > 1
|
|
157
|
+
map.get_size('name_or_path') -> return dataset size
|
|
158
|
+
map.get_shape('name_or_path') -> return dataset size
|
|
159
|
+
map.get_attr('name_or_path', 'attr') -> return value of dataset attribute
|
|
160
|
+
map.get_path('name_or_group_or_class') -> returns path of object with name
|
|
161
|
+
map.get_image_path() -> returns default path of detector dataset (or largest dataset)
|
|
162
|
+
map.get_group_path('name_or_path_or_class') -> return path of group with class
|
|
163
|
+
map.get_group_datasets('name_or_path_or_class') -> return list of dataset pathes in class
|
|
164
|
+
File Methods:
|
|
165
|
+
map.get_metadata(h5py.File) -> returns dict of value datasets
|
|
166
|
+
map.get_scannables(h5py.File) -> returns dict of scannable datasets
|
|
167
|
+
map.get_scannalbes_array(h5py.File) -> returns numpy array of scannable datasets
|
|
168
|
+
map.get_data_block(h5py.File) -> returns dict like object with metadata and scannables
|
|
169
|
+
map.get_image(h5py.File, index) -> returns image data
|
|
170
|
+
map.get_data(h5py.File, 'name') -> returns data from dataset
|
|
171
|
+
map.eval(h5py.File, 'expression') -> returns output of expression
|
|
172
|
+
map.format(h5py.File, 'string {name}') -> returns output of str expression
|
|
173
|
+
"""
|
|
174
|
+
|
|
175
|
+
def __init__(self, file: h5py.File | None = None):
|
|
176
|
+
self.filename = ''
|
|
177
|
+
self.all_paths = []
|
|
178
|
+
self.groups = {} # stores attributes of each group by path
|
|
179
|
+
self.datasets = {} # stores attributes of each dataset by path
|
|
180
|
+
self.classes = defaultdict(list) # stores lists of group paths by nx_class
|
|
181
|
+
self.arrays = {} # stores array dataset paths by name, altname + group_name
|
|
182
|
+
self.values = {} # stores value dataset paths by name, altname + group_name
|
|
183
|
+
self.metadata = {} # stores value dataset path by altname only
|
|
184
|
+
self.scannables = {} # stores array dataset paths with given size, by name
|
|
185
|
+
self.combined = {} # stores array and value paths (arrays overwrite values)
|
|
186
|
+
self.image_data = {} # stores dataset paths of image data
|
|
187
|
+
self._default_image_path = None
|
|
188
|
+
|
|
189
|
+
if isinstance(file, h5py.File):
|
|
190
|
+
self.populate(file)
|
|
191
|
+
|
|
192
|
+
def __getitem__(self, item):
|
|
193
|
+
return self.combined[item]
|
|
194
|
+
|
|
195
|
+
def __iter__(self):
|
|
196
|
+
return iter(self.combined)
|
|
197
|
+
|
|
198
|
+
def __contains__(self, item):
|
|
199
|
+
return item in self.combined or item in self.datasets
|
|
200
|
+
|
|
201
|
+
def __repr__(self):
|
|
202
|
+
return f"HdfMap based on '{self.filename}'"
|
|
203
|
+
|
|
204
|
+
def __str__(self):
|
|
205
|
+
return f"{repr(self)}\n{self.info_names()}\n{self.info_scannables()}"
|
|
206
|
+
|
|
207
|
+
def info_groups(self) -> str:
|
|
208
|
+
"""Return str info on groups"""
|
|
209
|
+
out = f"{repr(self)}\n"
|
|
210
|
+
out += "Groups:\n"
|
|
211
|
+
out += disp_dict(self.groups, 20)
|
|
212
|
+
out += '\n\nClasses:\n'
|
|
213
|
+
out += disp_dict(self.classes, 20)
|
|
214
|
+
return out
|
|
215
|
+
|
|
216
|
+
def info_datasets(self) -> str:
|
|
217
|
+
"""Return str info on datasets"""
|
|
218
|
+
out = f"{repr(self)}\n"
|
|
219
|
+
out += "Datasets:\n"
|
|
220
|
+
out += disp_dict(self.datasets, 20)
|
|
221
|
+
return out
|
|
222
|
+
|
|
223
|
+
def info_dataset_types(self) -> str:
|
|
224
|
+
"""Return str info on dataset types"""
|
|
225
|
+
out = "Values:\n"
|
|
226
|
+
out += disp_dict(self.values, 20)
|
|
227
|
+
out += "Arrays:\n"
|
|
228
|
+
out += '\n'.join([
|
|
229
|
+
f"{name:>30}: {str(self.datasets[path].shape):10} : {path:60}"
|
|
230
|
+
for name, path in self.arrays.items()
|
|
231
|
+
])
|
|
232
|
+
out += "Images:\n"
|
|
233
|
+
out += '\n'.join([
|
|
234
|
+
f"{name:>30}: {str(self.datasets[path].shape):10} : {path:60}"
|
|
235
|
+
for name, path in self.image_data.items()
|
|
236
|
+
])
|
|
237
|
+
return out
|
|
238
|
+
|
|
239
|
+
def info_names(self) -> str:
|
|
240
|
+
"""Return str info on combined namespace"""
|
|
241
|
+
out = "Combined Namespace:\n"
|
|
242
|
+
out += '\n'.join([
|
|
243
|
+
f"{name:>30}: {str(self.datasets[path].shape):10} : {path:60}"
|
|
244
|
+
for name, path in self.combined.items()
|
|
245
|
+
])
|
|
246
|
+
return out
|
|
247
|
+
|
|
248
|
+
def info_scannables(self) -> str:
|
|
249
|
+
"""Return str info on scannables namespace"""
|
|
250
|
+
out = "Scannables Namespace:\n"
|
|
251
|
+
out += '\n'.join([
|
|
252
|
+
f"{name:>30}: {str(self.datasets[path].shape):10} : {path:60}"
|
|
253
|
+
for name, path in self.scannables.items()
|
|
254
|
+
])
|
|
255
|
+
return out
|
|
256
|
+
|
|
257
|
+
def _store_group(self, hdf_group: h5py.Group, path: str, name: str):
|
|
258
|
+
|
|
259
|
+
nx_class = hdf_group.attrs.get('NX_class', default='Group')
|
|
260
|
+
if hasattr(nx_class, 'decode'):
|
|
261
|
+
nx_class = nx_class.decode()
|
|
262
|
+
self.groups[path] = Group(
|
|
263
|
+
nx_class,
|
|
264
|
+
name,
|
|
265
|
+
dict(hdf_group.attrs),
|
|
266
|
+
[key for key, item in hdf_group.items() if isinstance(item, h5py.Dataset)]
|
|
267
|
+
)
|
|
268
|
+
self.classes[name].append(path)
|
|
269
|
+
self.classes[nx_class].append(path)
|
|
270
|
+
logger.debug(f"{path} HDFGroup: {nx_class}")
|
|
271
|
+
return nx_class
|
|
272
|
+
|
|
273
|
+
def _store_dataset(self, hdf_dataset: h5py.Dataset, hdf_path: str, name: str):
|
|
274
|
+
# New: add group_name to namespace as standard, helps with names like s5/x + s4/x
|
|
275
|
+
# this significantly increases the number of names in namespaces
|
|
276
|
+
group_name = generate_identifier(f"{hdf_path.split(SEP)[-2]}_{name}")
|
|
277
|
+
alt_name = generate_identifier(hdf_dataset.attrs[LOCAL_NAME]) if LOCAL_NAME in hdf_dataset.attrs else None
|
|
278
|
+
names = {n: hdf_path for n in {name, group_name, alt_name} if n}
|
|
279
|
+
self.datasets[hdf_path] = Dataset(
|
|
280
|
+
name=name,
|
|
281
|
+
names=list(names),
|
|
282
|
+
size=hdf_dataset.size,
|
|
283
|
+
shape=hdf_dataset.shape,
|
|
284
|
+
attrs=dict(hdf_dataset.attrs),
|
|
285
|
+
)
|
|
286
|
+
if hdf_dataset.ndim >= 3:
|
|
287
|
+
self.image_data[name] = hdf_path
|
|
288
|
+
self.image_data[group_name] = hdf_path
|
|
289
|
+
self.arrays.update(names)
|
|
290
|
+
logger.debug(f"{hdf_path} HDFDataset: image_data & array {name, hdf_dataset.size, hdf_dataset.shape}")
|
|
291
|
+
elif hdf_dataset.ndim > 0:
|
|
292
|
+
self.arrays.update(names)
|
|
293
|
+
logger.debug(f"{hdf_path} HDFDataset: array {name, hdf_dataset.size, hdf_dataset.shape}")
|
|
294
|
+
else:
|
|
295
|
+
self.values.update(names)
|
|
296
|
+
if alt_name:
|
|
297
|
+
self.metadata[alt_name] = hdf_path
|
|
298
|
+
logger.debug(f"{hdf_path} HDFDataset: value")
|
|
299
|
+
|
|
300
|
+
def _populate(self, hdf_group: h5py.Group, root: str = '',
|
|
301
|
+
recursive: bool = True, groups: list[str] = None):
|
|
302
|
+
"""
|
|
303
|
+
populate HdfMap dictionary's using recursive method
|
|
304
|
+
:param hdf_group: HDF group object, from HDF File
|
|
305
|
+
:param root: str path of hdf Group, used to build dataset paths
|
|
306
|
+
:param recursive: if True, will recursively search through subgroups
|
|
307
|
+
:param groups: if not None, will only search subgroups named in list, e.g. ['entry','NX_DATA']
|
|
308
|
+
:return: None
|
|
309
|
+
"""
|
|
310
|
+
logger.info(f"{repr(self)}._populate root='{root}'")
|
|
311
|
+
for key in hdf_group:
|
|
312
|
+
obj = hdf_group.get(key)
|
|
313
|
+
link = hdf_group.get(key, getlink=True)
|
|
314
|
+
logger.debug(f"{key}: {repr(obj)} : {repr(link)}")
|
|
315
|
+
if obj is None:
|
|
316
|
+
continue # dataset may be missing due to a broken link
|
|
317
|
+
hdf_path = root + SEP + key # build hdf path - a cross-file unique identifier
|
|
318
|
+
# New: store all paths in file, useful for checking if anything was missed, but might be slow
|
|
319
|
+
self.all_paths.append(hdf_path)
|
|
320
|
+
name = generate_identifier(hdf_path)
|
|
321
|
+
logger.info(f"{hdf_path}: {name}, link={repr(link)}")
|
|
322
|
+
|
|
323
|
+
# Group
|
|
324
|
+
if isinstance(obj, h5py.Group):
|
|
325
|
+
nx_class = self._store_group(obj, hdf_path, name)
|
|
326
|
+
if recursive and (key in groups or nx_class in groups if groups else True):
|
|
327
|
+
self._populate(obj, hdf_path, recursive)
|
|
328
|
+
|
|
329
|
+
# Dataset
|
|
330
|
+
elif isinstance(obj, h5py.Dataset) and not isinstance(link, h5py.SoftLink):
|
|
331
|
+
self._store_dataset(obj, hdf_path, name)
|
|
332
|
+
|
|
333
|
+
def populate(self, hdf_file: h5py.File):
|
|
334
|
+
"""Populate all datasets from file"""
|
|
335
|
+
self.filename = hdf_file.filename
|
|
336
|
+
self._populate(hdf_file)
|
|
337
|
+
size = self.most_common_size()
|
|
338
|
+
self.generate_scannables(size)
|
|
339
|
+
|
|
340
|
+
def generate_combined(self):
|
|
341
|
+
self.combined = {**self.values, **self.arrays, **self.scannables}
|
|
342
|
+
|
|
343
|
+
def all_attrs(self) -> dict:
|
|
344
|
+
"""Return dict of all attributes in self.datasets and self.groups"""
|
|
345
|
+
ds_attrs = {k: v for path, ds in self.datasets.items() for k, v in ds.attrs.items()}
|
|
346
|
+
grp_attrs = {k: v for path, grp in self.groups.items() for k, v in grp.attrs.items()}
|
|
347
|
+
return {**grp_attrs, **ds_attrs}
|
|
348
|
+
|
|
349
|
+
def most_common_size(self) -> int:
|
|
350
|
+
"""Return most common array size > 1"""
|
|
351
|
+
array_sizes = [size for name, path in self.arrays.items() if (size := self.datasets[path].size) > 1]
|
|
352
|
+
return max(set(array_sizes), key=array_sizes.count)
|
|
353
|
+
|
|
354
|
+
def most_common_shape(self) -> tuple:
|
|
355
|
+
"""Return most common non-singular array shape"""
|
|
356
|
+
array_shapes = [shape for name, path in self.arrays.items() if len(shape := self.datasets[path].shape) > 0]
|
|
357
|
+
return max(set(array_shapes), key=array_shapes.count)
|
|
358
|
+
|
|
359
|
+
def scannables_length(self) -> int:
|
|
360
|
+
if not self.scannables:
|
|
361
|
+
return 0
|
|
362
|
+
path = next(iter(self.scannables.values()))
|
|
363
|
+
shape = self.datasets[path].shape
|
|
364
|
+
return shape[0]
|
|
365
|
+
|
|
366
|
+
def generate_scannables(self, array_size):
|
|
367
|
+
"""Populate self.scannables field with datasets size that match array_size"""
|
|
368
|
+
self.scannables = {k: v for k, v in self.arrays.items() if self.datasets[v].size == array_size}
|
|
369
|
+
# create combined dict, scannables and arrays overwrite values with same name
|
|
370
|
+
self.generate_combined()
|
|
371
|
+
|
|
372
|
+
def generate_scannables_from_group(self, hdf_group: h5py.Group, group_path: str = None):
|
|
373
|
+
"""
|
|
374
|
+
Generate scannables list from a specific group, using the first item to define array size
|
|
375
|
+
:param hdf_group: h5py.Group
|
|
376
|
+
:param group_path: str path of group hdf_group if hdf_group.name is incorrect
|
|
377
|
+
"""
|
|
378
|
+
first_dataset = hdf_group[next(iter(hdf_group))]
|
|
379
|
+
array_size = first_dataset.size
|
|
380
|
+
# watch out - hdf_group.name may not point to a location in the file!
|
|
381
|
+
hdf_path = hdf_group.name if group_path is None else group_path
|
|
382
|
+
self._populate(hdf_group, root=hdf_path, recursive=False)
|
|
383
|
+
self.scannables = {
|
|
384
|
+
k: build_hdf_path(hdf_path, k)
|
|
385
|
+
for k in hdf_group if isinstance(hdf_group[k], h5py.Dataset) and hdf_group[k].size == array_size
|
|
386
|
+
}
|
|
387
|
+
logger.debug(f"Scannables from group: {list(self.scannables.keys())}")
|
|
388
|
+
self.generate_combined()
|
|
389
|
+
|
|
390
|
+
def generate_scannables_from_names(self, names: list[str]):
|
|
391
|
+
"""Generate scannables list from a set of dataset names, using the first item to define array size"""
|
|
392
|
+
# concert names or paths to name (to match alt_name)
|
|
393
|
+
array_names = [n for name in names if (n := generate_identifier(name)) in self.arrays]
|
|
394
|
+
logger.debug(f"Scannables from names: {array_names}")
|
|
395
|
+
array_size = self.datasets[self.arrays[array_names[0]]].size
|
|
396
|
+
self.scannables = {
|
|
397
|
+
name: self.arrays[name] for name in array_names if self.datasets[self.arrays[name]].size == array_size
|
|
398
|
+
}
|
|
399
|
+
self.generate_combined()
|
|
400
|
+
|
|
401
|
+
def get_path(self, name_or_path):
|
|
402
|
+
"""Return hdf path of object in HdfMap"""
|
|
403
|
+
if name_or_path in self.datasets or name_or_path in self.groups:
|
|
404
|
+
return name_or_path
|
|
405
|
+
if name_or_path in self.combined:
|
|
406
|
+
return self.combined[name_or_path]
|
|
407
|
+
if name_or_path in self.image_data:
|
|
408
|
+
return self.image_data[name_or_path]
|
|
409
|
+
if name_or_path in self.classes:
|
|
410
|
+
return self.classes[name_or_path][0] # return first path in list
|
|
411
|
+
|
|
412
|
+
def get_group_path(self, name_or_path):
|
|
413
|
+
"""Return group path of object in HdfMap"""
|
|
414
|
+
hdf_path = self.get_path(name_or_path)
|
|
415
|
+
while hdf_path and hdf_path not in self.groups:
|
|
416
|
+
hdf_path = SEP.join(hdf_path.split(SEP)[:-1])
|
|
417
|
+
if not hdf_path:
|
|
418
|
+
return SEP
|
|
419
|
+
return hdf_path
|
|
420
|
+
|
|
421
|
+
def find_paths(self, string: str, name_only=True) -> list[str]:
|
|
422
|
+
"""
|
|
423
|
+
Find any dataset paths that contain the given string argument
|
|
424
|
+
:param string: str to find in list of datasets
|
|
425
|
+
:param name_only: if True, search only the name of the dataset, not the full path
|
|
426
|
+
:return: list of hdf paths
|
|
427
|
+
"""
|
|
428
|
+
# find string in combined
|
|
429
|
+
combined_paths = [path for name, path in self.combined.items() if string in name]
|
|
430
|
+
if name_only:
|
|
431
|
+
return [
|
|
432
|
+
path for path, dataset in self.datasets.items()
|
|
433
|
+
if string in dataset.name and path not in combined_paths
|
|
434
|
+
] + combined_paths
|
|
435
|
+
return [
|
|
436
|
+
path for path in self.datasets if string in path and path not in combined_paths
|
|
437
|
+
] + combined_paths
|
|
438
|
+
|
|
439
|
+
def find_names(self, string: str) -> list[str]:
|
|
440
|
+
"""
|
|
441
|
+
Find any dataset names that contain the given string argument, searching names in self.combined
|
|
442
|
+
:param string: str to find in list of datasets
|
|
443
|
+
:return: list of names
|
|
444
|
+
"""
|
|
445
|
+
return [name for name in self.combined if string in name]
|
|
446
|
+
|
|
447
|
+
def find_attr(self, attr_name: str) -> list[str]:
|
|
448
|
+
"""
|
|
449
|
+
Find any dataset or group path with an attribute that contains attr_name.
|
|
450
|
+
:param attr_name: str name of hdfobj.attr
|
|
451
|
+
:return: list of hdf paths
|
|
452
|
+
"""
|
|
453
|
+
return [
|
|
454
|
+
path for path, ds in self.datasets.items() if attr_name in ds.attrs
|
|
455
|
+
] + [
|
|
456
|
+
path for path, grp in self.groups.items() if attr_name in grp.attrs
|
|
457
|
+
]
|
|
458
|
+
|
|
459
|
+
def get_attrs(self, name_or_path: str) -> dict | None:
|
|
460
|
+
"""Return attributes of dataset or group"""
|
|
461
|
+
if name_or_path in self.datasets:
|
|
462
|
+
return self.datasets[name_or_path].attrs
|
|
463
|
+
if name_or_path in self.groups:
|
|
464
|
+
return self.groups[name_or_path].attrs
|
|
465
|
+
if name_or_path in self.combined:
|
|
466
|
+
return self.datasets[self.combined[name_or_path]].attrs
|
|
467
|
+
if name_or_path in self.classes:
|
|
468
|
+
return self.groups[self.classes[name_or_path][0]].attrs
|
|
469
|
+
|
|
470
|
+
def get_attr(self, name_or_path: str, attr_label: str, default: str | typing.Any = '') -> str | None:
|
|
471
|
+
"""Return named attribute from dataset or group, or default"""
|
|
472
|
+
attrs = self.get_attrs(name_or_path)
|
|
473
|
+
if attrs and attr_label in attrs:
|
|
474
|
+
return attr.decode() if hasattr(attr := attrs[attr_label], 'decode') else attr
|
|
475
|
+
return default
|
|
476
|
+
|
|
477
|
+
def set_image_path(self, name_or_path: str):
|
|
478
|
+
"""Set the default image path, used by get_image"""
|
|
479
|
+
if name_or_path is None:
|
|
480
|
+
self._default_image_path = None
|
|
481
|
+
else:
|
|
482
|
+
path = self.get_path(name_or_path)
|
|
483
|
+
if path:
|
|
484
|
+
self._default_image_path = path
|
|
485
|
+
logger.info(f"Default image path: {self._default_image_path}")
|
|
486
|
+
|
|
487
|
+
def get_image_path(self) -> str | None:
|
|
488
|
+
"""Return HDF path of first dataset in self.image_data"""
|
|
489
|
+
if self._default_image_path:
|
|
490
|
+
return self._default_image_path
|
|
491
|
+
if self.image_data:
|
|
492
|
+
return next(iter(self.image_data.values()))
|
|
493
|
+
|
|
494
|
+
def get_group_datasets(self, name_or_path: str) -> list[str] | None:
|
|
495
|
+
"""Find the path associate with the given name and return all datasets in that group"""
|
|
496
|
+
group_path = self.get_group_path(name_or_path)
|
|
497
|
+
if group_path:
|
|
498
|
+
return self.groups[group_path].datasets
|
|
499
|
+
|
|
500
|
+
"--------------------------------------------------------"
|
|
501
|
+
"---------------------- FILE READERS --------------------"
|
|
502
|
+
"--------------------------------------------------------"
|
|
503
|
+
|
|
504
|
+
def get_data(self, hdf_file: h5py.File, name_or_path: str, index=(), default=None, direct_load=False):
|
|
505
|
+
"""
|
|
506
|
+
Return data from dataset in file, converted into either datetime, str or squeezed numpy.array objects
|
|
507
|
+
See hdfmap.eval_functions.dataset2data for more information.
|
|
508
|
+
:param hdf_file: hdf file object
|
|
509
|
+
:param name_or_path: str name or path pointing to dataset in hdf file
|
|
510
|
+
:param index: index or slice of data in hdf file
|
|
511
|
+
:param default: value to return if name not found in hdf file
|
|
512
|
+
:param direct_load: return str, datetime or squeezed array if False, otherwise load data directly
|
|
513
|
+
:return: dataset2data(dataset) -> datetime, str or squeezed array as required.
|
|
514
|
+
"""
|
|
515
|
+
path = self.get_path(name_or_path)
|
|
516
|
+
if path and path in hdf_file:
|
|
517
|
+
return dataset2data(hdf_file[path], index, direct_load)
|
|
518
|
+
return default
|
|
519
|
+
|
|
520
|
+
def get_metadata(self, hdf_file: h5py.File, default=None, direct_load=False, name_list: list = None) -> dict:
|
|
521
|
+
"""
|
|
522
|
+
Return metadata dict from file, loading data for each item in the metadata list
|
|
523
|
+
The metadata list is taken from name_list, otherwise self.metadata or self.values
|
|
524
|
+
:param hdf_file: hdf file object
|
|
525
|
+
:param default: Value to return for names not associated with a dataset
|
|
526
|
+
:param direct_load: if True, loads data from hdf file directory, without conversion
|
|
527
|
+
:param name_list: if available, uses this list of dataset names to generate the metadata list
|
|
528
|
+
:return:
|
|
529
|
+
"""
|
|
530
|
+
extra = extra_hdf_data(hdf_file)
|
|
531
|
+
if name_list:
|
|
532
|
+
metadata_paths = {name: self.combined.get(name, '') for name in name_list}
|
|
533
|
+
else:
|
|
534
|
+
metadata_paths = self.metadata if len(self.metadata) > 0 else self.values
|
|
535
|
+
metadata = {
|
|
536
|
+
name: dataset2data(hdf_file[path], direct_load=direct_load) if path in hdf_file else default
|
|
537
|
+
for name, path in metadata_paths.items()
|
|
538
|
+
}
|
|
539
|
+
return {**extra, **metadata}
|
|
540
|
+
|
|
541
|
+
def get_scannables(self, hdf_file: h5py.File) -> dict:
|
|
542
|
+
"""Return scannables from file (values associated with hdfmap.scannables)"""
|
|
543
|
+
return {
|
|
544
|
+
name: hdf_file[path][()] for name, path in self.scannables.items()
|
|
545
|
+
if path in hdf_file
|
|
546
|
+
}
|
|
547
|
+
|
|
548
|
+
def get_image(self, hdf_file: h5py.File, index: slice = None) -> np.ndarray | None:
|
|
549
|
+
"""
|
|
550
|
+
Get image data from file, using default image path
|
|
551
|
+
:param hdf_file: hdf file object
|
|
552
|
+
:param index: (slice,) or None to take the middle image
|
|
553
|
+
:return: numpy array of image
|
|
554
|
+
"""
|
|
555
|
+
if index is None:
|
|
556
|
+
index = self.scannables_length() // 2
|
|
557
|
+
image_path = self.get_image_path()
|
|
558
|
+
logger.debug(f"image path: {image_path}")
|
|
559
|
+
if image_path and image_path in hdf_file:
|
|
560
|
+
return hdf_file[image_path][index].squeeze() # remove trailing dimensions
|
|
561
|
+
|
|
562
|
+
def _get_numeric_scannables(self, hdf_file: h5py.File) -> list[tuple[str, str]]:
|
|
563
|
+
"""Return numeric scannables available in file"""
|
|
564
|
+
return [
|
|
565
|
+
(name, path) for name, path in self.scannables.items()
|
|
566
|
+
if hdf_file.get(path) and np.issubdtype(hdf_file.get(path).dtype, np.number)
|
|
567
|
+
]
|
|
568
|
+
|
|
569
|
+
def get_scannables_array(self, hdf_file: h5py.File) -> np.ndarray:
|
|
570
|
+
"""Return 2D array of all scannables in file"""
|
|
571
|
+
_scannables = self._get_numeric_scannables(hdf_file)
|
|
572
|
+
dtypes = np.dtype([
|
|
573
|
+
(name, hdf_file[path].dtype) for name, path in _scannables
|
|
574
|
+
])
|
|
575
|
+
return np.array([hdf_file.get(path)[()] for name, path in _scannables], dtype=dtypes)
|
|
576
|
+
|
|
577
|
+
def create_scannables_table(self, hdf_file: h5py.File, delimiter=', ',
|
|
578
|
+
string_spec='', format_spec='f', default_decimals=8) -> str:
|
|
579
|
+
"""
|
|
580
|
+
Return str representation of scannables as a table
|
|
581
|
+
The table starts with a header row given by names of the scannables.
|
|
582
|
+
Each row contains the numeric values for each scannable, formated by the given string spec:
|
|
583
|
+
{value: "string_spec.decimals format_spec"}
|
|
584
|
+
e.g. {value: "5.8f"}
|
|
585
|
+
decimals is taken from each scannables "decimals" attribute if it exits, otherwise uses default
|
|
586
|
+
:param hdf_file: h5py.File object
|
|
587
|
+
:param delimiter: str seperator between each column
|
|
588
|
+
:param string_spec: str first element of float format specifier - length of string
|
|
589
|
+
:param format_spec: str type element of format specifier - 'f'=float, 'e'=exponential, 'g'=general
|
|
590
|
+
:param default_decimals: int default number of decimals given
|
|
591
|
+
:return: str
|
|
592
|
+
"""
|
|
593
|
+
_scannables = self._get_numeric_scannables(hdf_file)
|
|
594
|
+
fmt = string_spec + '.%d' + format_spec
|
|
595
|
+
formats = [
|
|
596
|
+
'{:' + fmt % self.get_attr(path, 'decimals', default=default_decimals) + '}'
|
|
597
|
+
for name, path in _scannables
|
|
598
|
+
]
|
|
599
|
+
|
|
600
|
+
length = self.scannables_length()
|
|
601
|
+
out = delimiter.join([name for name, _ in _scannables]) + '\n'
|
|
602
|
+
out += '\n'.join([
|
|
603
|
+
delimiter.join([
|
|
604
|
+
fmt.format(hdf_file.get(path)[n])
|
|
605
|
+
for (_, path), fmt in zip(_scannables, formats)
|
|
606
|
+
])
|
|
607
|
+
for n in range(length)
|
|
608
|
+
])
|
|
609
|
+
return out
|
|
610
|
+
|
|
611
|
+
def get_dataholder(self, hdf_file: h5py.File) -> DataHolder:
|
|
612
|
+
"""
|
|
613
|
+
Return DataHolder object - a simple replication of scisoftpy.dictutils.DataHolder
|
|
614
|
+
Also known as DLS dat format.
|
|
615
|
+
dataholder.scannable -> array
|
|
616
|
+
dataholder.metadata.value -> metadata
|
|
617
|
+
dataholder['scannable'] -> array
|
|
618
|
+
dataholder.metadata['value'] -> metadata
|
|
619
|
+
:param hdf_file: h5py.File object
|
|
620
|
+
:return: data_object (similar to dict)
|
|
621
|
+
"""
|
|
622
|
+
metadata = self.get_metadata(hdf_file)
|
|
623
|
+
scannables = self.get_scannables(hdf_file)
|
|
624
|
+
scannables['metadata'] = DataHolder(**metadata)
|
|
625
|
+
return DataHolder(**scannables)
|
|
626
|
+
|
|
627
|
+
def eval(self, hdf_file: h5py.File, expression: str):
|
|
628
|
+
"""
|
|
629
|
+
Evaluate an expression using the namespace of the hdf file
|
|
630
|
+
:param hdf_file: h5py.File object
|
|
631
|
+
:param expression: str expression to be evaluated
|
|
632
|
+
:return: eval(expression)
|
|
633
|
+
"""
|
|
634
|
+
return eval_hdf(hdf_file, expression, self.combined)
|
|
635
|
+
|
|
636
|
+
def format_hdf(self, hdf_file: h5py.File, expression: str) -> str:
|
|
637
|
+
"""
|
|
638
|
+
Evaluate a formatted string expression using the namespace of the hdf file
|
|
639
|
+
:param hdf_file: h5py.File object
|
|
640
|
+
:param expression: str expression using {name} format specifiers
|
|
641
|
+
:return: eval_hdf(f"expression")
|
|
642
|
+
"""
|
|
643
|
+
return format_hdf(hdf_file, expression, self.combined)
|
|
644
|
+
|
|
645
|
+
def info_data(self, hdf_file):
|
|
646
|
+
"""Return string showing metadata values associated with names"""
|
|
647
|
+
out = repr(self) + '\n'
|
|
648
|
+
out += "Combined Namespace:\n"
|
|
649
|
+
out += '\n'.join([
|
|
650
|
+
f"{name:>30}: " +
|
|
651
|
+
f"{str(data if np.size(data := dataset2data(hdf_file[path])) <= 1 else self.datasets[path].shape):20}" +
|
|
652
|
+
f": {path:60}"
|
|
653
|
+
for name, path in self.combined.items()
|
|
654
|
+
])
|
|
655
|
+
out += f"\n{self.info_scannables()}"
|
|
656
|
+
return out
|