hdfmap 0.4__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
hdfmap/hdfmap_class.py ADDED
@@ -0,0 +1,656 @@
1
+ """
2
+ hdfmap class definition
3
+ """
4
+ import typing
5
+ from collections import defaultdict
6
+ from types import SimpleNamespace
7
+
8
+ import numpy as np
9
+ import h5py
10
+
11
+ from .logging import create_logger
12
+ from .eval_functions import expression_safe_name, extra_hdf_data, eval_hdf, format_hdf, dataset2data
13
+
14
+ try:
15
+ import hdf5plugin # required for compressed data
16
+ except ImportError:
17
+ print('Warning: hdf5plugin not available.')
18
+
19
+ # parameters
20
+ SEP = '/' # HDF path separator
21
+ LOCAL_NAME = 'local_name' # dataset attribute name for alt_name
22
+ OMIT = '/value' # omit this name in paths when determining identifier
23
+
24
+ # logger
25
+ logger = create_logger(__name__)
26
+
27
+
28
+ class Group(typing.NamedTuple):
29
+ nx_class: str
30
+ name: str
31
+ attrs: dict
32
+ datasets: list[str]
33
+
34
+
35
+ class Dataset(typing.NamedTuple):
36
+ name: str
37
+ names: list[str]
38
+ size: int
39
+ shape: tuple[int]
40
+ attrs: dict
41
+
42
+
43
+ def generate_identifier(hdf_path: str | bytes) -> str:
44
+ """
45
+ Generate a valid python identifier from a hdf dataset path or other string
46
+ - Decodes to ascii
47
+ - omits '/value'
48
+ - splits by path separator (/) and takes final element
49
+ - converts special characters to '_'
50
+ - removes replication of strings separated by '_'
51
+ E.G.
52
+ /entry/group/motor1 >> "motor1"
53
+ /entry/group/motor/value >> "motor"
54
+ /entry/group/subgroup.motor >> "subgroup_motor"
55
+ motor.motor >> "motor"
56
+ :param hdf_path: str hdf path address
57
+ :return: str expression safe name
58
+ """
59
+ if hasattr(hdf_path, 'decode'): # Byte string
60
+ hdf_path = hdf_path.decode('ascii')
61
+ if hdf_path.endswith(OMIT):
62
+ hdf_path = hdf_path[:-len(OMIT)] # omit 'value'
63
+ substrings = hdf_path.split(SEP)
64
+ name = expression_safe_name(substrings[-1])
65
+ # remove replication (handles local_names 'name.name' convention)
66
+ return '_'.join(dict.fromkeys(name.split('_')))
67
+
68
+
69
+ def build_hdf_path(*args: str | bytes) -> str:
70
+ """
71
+ Build path from string or bytes arguments
72
+ '/entry/measurement' = build_hdf_path(b'entry', 'measurement')
73
+ :param args: str or bytes arguments
74
+ :return: str hdf path
75
+ """
76
+ return SEP + SEP.join((arg.decode() if isinstance(arg, bytes) else arg).strip(SEP) for arg in args)
77
+
78
+
79
+ def disp_dict(mydict: dict, indent: int = 10) -> str:
80
+ return '\n'.join([f"{key:>{indent}}: {value}" for key, value in mydict.items()])
81
+
82
+
83
+ class DataHolder(SimpleNamespace):
84
+ """
85
+ Convert dict to class that looks like a class object with key names as attributes
86
+ Replicates slightly the old scisoftpy.dictutils.DataHolder class, also known as DLS dat format.
87
+ obj = DataHolder(**{'item1': 'value1'})
88
+ obj['item1'] -> 'value1'
89
+ obj.item1 -> 'value1'
90
+ """
91
+
92
+ def __getitem__(self, item):
93
+ return self.__dict__.__getitem__(item)
94
+
95
+ def __iter__(self):
96
+ return self.__dict__.__iter__()
97
+
98
+ def keys(self):
99
+ return self.__dict__.keys()
100
+
101
+
102
+ class HdfMap:
103
+ """
104
+ HdfMap object, container for paths of different objects in an HDF file
105
+
106
+ with h5py.File('file.hdf') as hdf:
107
+ map = HdfMap(hdf)
108
+
109
+ map.get_path('data') -> '/entry/measurement/data'
110
+ map['data'] -> '/entry/measurement/data'
111
+
112
+ with h5py.File('another_file.hdf') as hdf:
113
+ data = map.get_data(hdf, 'data')
114
+ array = map.get_scannables_array(hdf)
115
+ metadata = map.get_metadata(hdf)
116
+ out = map.eval(hdf, 'data / 10')
117
+ outstr = map.format(hdf, 'the data looks like: {data}')
118
+
119
+ Objects within the HDF file are separated into Groups and Datasets. Each object has a
120
+ defined 'path' and 'name' paramater, as well as other attributes
121
+ path -> '/entry/measurement/data' -> the location of an object within the file
122
+ name -> 'data' -> an path expressed as a simple variable name
123
+ Paths are unique location within the file but can be used to identify similar objects in other files
124
+ Names may not be unique within a file and are generated from the final element of the hdf path.
125
+ - When multiple paths produce the same name, the name is overwritten each time, so the last path in the
126
+ file has priority.
127
+ - Names are also stored using the 'local_name' attribute, if it exists
128
+
129
+ Names of different types of datasets are stored for arrays (size > 0) and values (size 0)
130
+ Names for scannables relate to all arrays of a particular size
131
+ A combined list of names is provided where scannables > arrays > values
132
+
133
+
134
+
135
+ Attributes:
136
+ map.groups stores attributes of each group by path
137
+ map.classes stores list of group paths by nx_class
138
+ map.datasets stores attributes of each dataset by path
139
+ map.arrays stores array dataset paths by name
140
+ map.values stores value dataset paths by name
141
+ map.scannables stores array dataset paths with given size, by name
142
+ map.combined stores array and value paths (arrays overwrite values)
143
+ map.image_data stores dataset paths of image data
144
+ E.G.
145
+ map.groups = {'/hdf/group': ('class', 'name', {attrs}, [datasets])}
146
+ map.classes = {'class_name': ['/hdf/group1', '/hdf/group2']}
147
+ map.datasets = {'/hdf/group/dataset': ('name', size, shape, {attrs})}
148
+ map.arrays = {'name': '/hdf/group/dataset'}
149
+ map.values = {'name': '/hdf/group/dataset'}
150
+ map.scannables = {'name': '/hdf/group/dataset'}
151
+ map.image_data = {'name': '/hdf/group/dataset'}
152
+
153
+ Methods:
154
+ map.populate(h5py.File) -> populates the dictionaries using the given file
155
+ map.generate_scannables(array_size) -> populates scannables namespace with arrays of same size
156
+ map.most_common_size -> returns the most common dataset size > 1
157
+ map.get_size('name_or_path') -> return dataset size
158
+ map.get_shape('name_or_path') -> return dataset size
159
+ map.get_attr('name_or_path', 'attr') -> return value of dataset attribute
160
+ map.get_path('name_or_group_or_class') -> returns path of object with name
161
+ map.get_image_path() -> returns default path of detector dataset (or largest dataset)
162
+ map.get_group_path('name_or_path_or_class') -> return path of group with class
163
+ map.get_group_datasets('name_or_path_or_class') -> return list of dataset pathes in class
164
+ File Methods:
165
+ map.get_metadata(h5py.File) -> returns dict of value datasets
166
+ map.get_scannables(h5py.File) -> returns dict of scannable datasets
167
+ map.get_scannalbes_array(h5py.File) -> returns numpy array of scannable datasets
168
+ map.get_data_block(h5py.File) -> returns dict like object with metadata and scannables
169
+ map.get_image(h5py.File, index) -> returns image data
170
+ map.get_data(h5py.File, 'name') -> returns data from dataset
171
+ map.eval(h5py.File, 'expression') -> returns output of expression
172
+ map.format(h5py.File, 'string {name}') -> returns output of str expression
173
+ """
174
+
175
+ def __init__(self, file: h5py.File | None = None):
176
+ self.filename = ''
177
+ self.all_paths = []
178
+ self.groups = {} # stores attributes of each group by path
179
+ self.datasets = {} # stores attributes of each dataset by path
180
+ self.classes = defaultdict(list) # stores lists of group paths by nx_class
181
+ self.arrays = {} # stores array dataset paths by name, altname + group_name
182
+ self.values = {} # stores value dataset paths by name, altname + group_name
183
+ self.metadata = {} # stores value dataset path by altname only
184
+ self.scannables = {} # stores array dataset paths with given size, by name
185
+ self.combined = {} # stores array and value paths (arrays overwrite values)
186
+ self.image_data = {} # stores dataset paths of image data
187
+ self._default_image_path = None
188
+
189
+ if isinstance(file, h5py.File):
190
+ self.populate(file)
191
+
192
+ def __getitem__(self, item):
193
+ return self.combined[item]
194
+
195
+ def __iter__(self):
196
+ return iter(self.combined)
197
+
198
+ def __contains__(self, item):
199
+ return item in self.combined or item in self.datasets
200
+
201
+ def __repr__(self):
202
+ return f"HdfMap based on '{self.filename}'"
203
+
204
+ def __str__(self):
205
+ return f"{repr(self)}\n{self.info_names()}\n{self.info_scannables()}"
206
+
207
+ def info_groups(self) -> str:
208
+ """Return str info on groups"""
209
+ out = f"{repr(self)}\n"
210
+ out += "Groups:\n"
211
+ out += disp_dict(self.groups, 20)
212
+ out += '\n\nClasses:\n'
213
+ out += disp_dict(self.classes, 20)
214
+ return out
215
+
216
+ def info_datasets(self) -> str:
217
+ """Return str info on datasets"""
218
+ out = f"{repr(self)}\n"
219
+ out += "Datasets:\n"
220
+ out += disp_dict(self.datasets, 20)
221
+ return out
222
+
223
+ def info_dataset_types(self) -> str:
224
+ """Return str info on dataset types"""
225
+ out = "Values:\n"
226
+ out += disp_dict(self.values, 20)
227
+ out += "Arrays:\n"
228
+ out += '\n'.join([
229
+ f"{name:>30}: {str(self.datasets[path].shape):10} : {path:60}"
230
+ for name, path in self.arrays.items()
231
+ ])
232
+ out += "Images:\n"
233
+ out += '\n'.join([
234
+ f"{name:>30}: {str(self.datasets[path].shape):10} : {path:60}"
235
+ for name, path in self.image_data.items()
236
+ ])
237
+ return out
238
+
239
+ def info_names(self) -> str:
240
+ """Return str info on combined namespace"""
241
+ out = "Combined Namespace:\n"
242
+ out += '\n'.join([
243
+ f"{name:>30}: {str(self.datasets[path].shape):10} : {path:60}"
244
+ for name, path in self.combined.items()
245
+ ])
246
+ return out
247
+
248
+ def info_scannables(self) -> str:
249
+ """Return str info on scannables namespace"""
250
+ out = "Scannables Namespace:\n"
251
+ out += '\n'.join([
252
+ f"{name:>30}: {str(self.datasets[path].shape):10} : {path:60}"
253
+ for name, path in self.scannables.items()
254
+ ])
255
+ return out
256
+
257
+ def _store_group(self, hdf_group: h5py.Group, path: str, name: str):
258
+
259
+ nx_class = hdf_group.attrs.get('NX_class', default='Group')
260
+ if hasattr(nx_class, 'decode'):
261
+ nx_class = nx_class.decode()
262
+ self.groups[path] = Group(
263
+ nx_class,
264
+ name,
265
+ dict(hdf_group.attrs),
266
+ [key for key, item in hdf_group.items() if isinstance(item, h5py.Dataset)]
267
+ )
268
+ self.classes[name].append(path)
269
+ self.classes[nx_class].append(path)
270
+ logger.debug(f"{path} HDFGroup: {nx_class}")
271
+ return nx_class
272
+
273
+ def _store_dataset(self, hdf_dataset: h5py.Dataset, hdf_path: str, name: str):
274
+ # New: add group_name to namespace as standard, helps with names like s5/x + s4/x
275
+ # this significantly increases the number of names in namespaces
276
+ group_name = generate_identifier(f"{hdf_path.split(SEP)[-2]}_{name}")
277
+ alt_name = generate_identifier(hdf_dataset.attrs[LOCAL_NAME]) if LOCAL_NAME in hdf_dataset.attrs else None
278
+ names = {n: hdf_path for n in {name, group_name, alt_name} if n}
279
+ self.datasets[hdf_path] = Dataset(
280
+ name=name,
281
+ names=list(names),
282
+ size=hdf_dataset.size,
283
+ shape=hdf_dataset.shape,
284
+ attrs=dict(hdf_dataset.attrs),
285
+ )
286
+ if hdf_dataset.ndim >= 3:
287
+ self.image_data[name] = hdf_path
288
+ self.image_data[group_name] = hdf_path
289
+ self.arrays.update(names)
290
+ logger.debug(f"{hdf_path} HDFDataset: image_data & array {name, hdf_dataset.size, hdf_dataset.shape}")
291
+ elif hdf_dataset.ndim > 0:
292
+ self.arrays.update(names)
293
+ logger.debug(f"{hdf_path} HDFDataset: array {name, hdf_dataset.size, hdf_dataset.shape}")
294
+ else:
295
+ self.values.update(names)
296
+ if alt_name:
297
+ self.metadata[alt_name] = hdf_path
298
+ logger.debug(f"{hdf_path} HDFDataset: value")
299
+
300
+ def _populate(self, hdf_group: h5py.Group, root: str = '',
301
+ recursive: bool = True, groups: list[str] = None):
302
+ """
303
+ populate HdfMap dictionary's using recursive method
304
+ :param hdf_group: HDF group object, from HDF File
305
+ :param root: str path of hdf Group, used to build dataset paths
306
+ :param recursive: if True, will recursively search through subgroups
307
+ :param groups: if not None, will only search subgroups named in list, e.g. ['entry','NX_DATA']
308
+ :return: None
309
+ """
310
+ logger.info(f"{repr(self)}._populate root='{root}'")
311
+ for key in hdf_group:
312
+ obj = hdf_group.get(key)
313
+ link = hdf_group.get(key, getlink=True)
314
+ logger.debug(f"{key}: {repr(obj)} : {repr(link)}")
315
+ if obj is None:
316
+ continue # dataset may be missing due to a broken link
317
+ hdf_path = root + SEP + key # build hdf path - a cross-file unique identifier
318
+ # New: store all paths in file, useful for checking if anything was missed, but might be slow
319
+ self.all_paths.append(hdf_path)
320
+ name = generate_identifier(hdf_path)
321
+ logger.info(f"{hdf_path}: {name}, link={repr(link)}")
322
+
323
+ # Group
324
+ if isinstance(obj, h5py.Group):
325
+ nx_class = self._store_group(obj, hdf_path, name)
326
+ if recursive and (key in groups or nx_class in groups if groups else True):
327
+ self._populate(obj, hdf_path, recursive)
328
+
329
+ # Dataset
330
+ elif isinstance(obj, h5py.Dataset) and not isinstance(link, h5py.SoftLink):
331
+ self._store_dataset(obj, hdf_path, name)
332
+
333
+ def populate(self, hdf_file: h5py.File):
334
+ """Populate all datasets from file"""
335
+ self.filename = hdf_file.filename
336
+ self._populate(hdf_file)
337
+ size = self.most_common_size()
338
+ self.generate_scannables(size)
339
+
340
+ def generate_combined(self):
341
+ self.combined = {**self.values, **self.arrays, **self.scannables}
342
+
343
+ def all_attrs(self) -> dict:
344
+ """Return dict of all attributes in self.datasets and self.groups"""
345
+ ds_attrs = {k: v for path, ds in self.datasets.items() for k, v in ds.attrs.items()}
346
+ grp_attrs = {k: v for path, grp in self.groups.items() for k, v in grp.attrs.items()}
347
+ return {**grp_attrs, **ds_attrs}
348
+
349
+ def most_common_size(self) -> int:
350
+ """Return most common array size > 1"""
351
+ array_sizes = [size for name, path in self.arrays.items() if (size := self.datasets[path].size) > 1]
352
+ return max(set(array_sizes), key=array_sizes.count)
353
+
354
+ def most_common_shape(self) -> tuple:
355
+ """Return most common non-singular array shape"""
356
+ array_shapes = [shape for name, path in self.arrays.items() if len(shape := self.datasets[path].shape) > 0]
357
+ return max(set(array_shapes), key=array_shapes.count)
358
+
359
+ def scannables_length(self) -> int:
360
+ if not self.scannables:
361
+ return 0
362
+ path = next(iter(self.scannables.values()))
363
+ shape = self.datasets[path].shape
364
+ return shape[0]
365
+
366
+ def generate_scannables(self, array_size):
367
+ """Populate self.scannables field with datasets size that match array_size"""
368
+ self.scannables = {k: v for k, v in self.arrays.items() if self.datasets[v].size == array_size}
369
+ # create combined dict, scannables and arrays overwrite values with same name
370
+ self.generate_combined()
371
+
372
+ def generate_scannables_from_group(self, hdf_group: h5py.Group, group_path: str = None):
373
+ """
374
+ Generate scannables list from a specific group, using the first item to define array size
375
+ :param hdf_group: h5py.Group
376
+ :param group_path: str path of group hdf_group if hdf_group.name is incorrect
377
+ """
378
+ first_dataset = hdf_group[next(iter(hdf_group))]
379
+ array_size = first_dataset.size
380
+ # watch out - hdf_group.name may not point to a location in the file!
381
+ hdf_path = hdf_group.name if group_path is None else group_path
382
+ self._populate(hdf_group, root=hdf_path, recursive=False)
383
+ self.scannables = {
384
+ k: build_hdf_path(hdf_path, k)
385
+ for k in hdf_group if isinstance(hdf_group[k], h5py.Dataset) and hdf_group[k].size == array_size
386
+ }
387
+ logger.debug(f"Scannables from group: {list(self.scannables.keys())}")
388
+ self.generate_combined()
389
+
390
+ def generate_scannables_from_names(self, names: list[str]):
391
+ """Generate scannables list from a set of dataset names, using the first item to define array size"""
392
+ # concert names or paths to name (to match alt_name)
393
+ array_names = [n for name in names if (n := generate_identifier(name)) in self.arrays]
394
+ logger.debug(f"Scannables from names: {array_names}")
395
+ array_size = self.datasets[self.arrays[array_names[0]]].size
396
+ self.scannables = {
397
+ name: self.arrays[name] for name in array_names if self.datasets[self.arrays[name]].size == array_size
398
+ }
399
+ self.generate_combined()
400
+
401
+ def get_path(self, name_or_path):
402
+ """Return hdf path of object in HdfMap"""
403
+ if name_or_path in self.datasets or name_or_path in self.groups:
404
+ return name_or_path
405
+ if name_or_path in self.combined:
406
+ return self.combined[name_or_path]
407
+ if name_or_path in self.image_data:
408
+ return self.image_data[name_or_path]
409
+ if name_or_path in self.classes:
410
+ return self.classes[name_or_path][0] # return first path in list
411
+
412
+ def get_group_path(self, name_or_path):
413
+ """Return group path of object in HdfMap"""
414
+ hdf_path = self.get_path(name_or_path)
415
+ while hdf_path and hdf_path not in self.groups:
416
+ hdf_path = SEP.join(hdf_path.split(SEP)[:-1])
417
+ if not hdf_path:
418
+ return SEP
419
+ return hdf_path
420
+
421
+ def find_paths(self, string: str, name_only=True) -> list[str]:
422
+ """
423
+ Find any dataset paths that contain the given string argument
424
+ :param string: str to find in list of datasets
425
+ :param name_only: if True, search only the name of the dataset, not the full path
426
+ :return: list of hdf paths
427
+ """
428
+ # find string in combined
429
+ combined_paths = [path for name, path in self.combined.items() if string in name]
430
+ if name_only:
431
+ return [
432
+ path for path, dataset in self.datasets.items()
433
+ if string in dataset.name and path not in combined_paths
434
+ ] + combined_paths
435
+ return [
436
+ path for path in self.datasets if string in path and path not in combined_paths
437
+ ] + combined_paths
438
+
439
+ def find_names(self, string: str) -> list[str]:
440
+ """
441
+ Find any dataset names that contain the given string argument, searching names in self.combined
442
+ :param string: str to find in list of datasets
443
+ :return: list of names
444
+ """
445
+ return [name for name in self.combined if string in name]
446
+
447
+ def find_attr(self, attr_name: str) -> list[str]:
448
+ """
449
+ Find any dataset or group path with an attribute that contains attr_name.
450
+ :param attr_name: str name of hdfobj.attr
451
+ :return: list of hdf paths
452
+ """
453
+ return [
454
+ path for path, ds in self.datasets.items() if attr_name in ds.attrs
455
+ ] + [
456
+ path for path, grp in self.groups.items() if attr_name in grp.attrs
457
+ ]
458
+
459
+ def get_attrs(self, name_or_path: str) -> dict | None:
460
+ """Return attributes of dataset or group"""
461
+ if name_or_path in self.datasets:
462
+ return self.datasets[name_or_path].attrs
463
+ if name_or_path in self.groups:
464
+ return self.groups[name_or_path].attrs
465
+ if name_or_path in self.combined:
466
+ return self.datasets[self.combined[name_or_path]].attrs
467
+ if name_or_path in self.classes:
468
+ return self.groups[self.classes[name_or_path][0]].attrs
469
+
470
+ def get_attr(self, name_or_path: str, attr_label: str, default: str | typing.Any = '') -> str | None:
471
+ """Return named attribute from dataset or group, or default"""
472
+ attrs = self.get_attrs(name_or_path)
473
+ if attrs and attr_label in attrs:
474
+ return attr.decode() if hasattr(attr := attrs[attr_label], 'decode') else attr
475
+ return default
476
+
477
+ def set_image_path(self, name_or_path: str):
478
+ """Set the default image path, used by get_image"""
479
+ if name_or_path is None:
480
+ self._default_image_path = None
481
+ else:
482
+ path = self.get_path(name_or_path)
483
+ if path:
484
+ self._default_image_path = path
485
+ logger.info(f"Default image path: {self._default_image_path}")
486
+
487
+ def get_image_path(self) -> str | None:
488
+ """Return HDF path of first dataset in self.image_data"""
489
+ if self._default_image_path:
490
+ return self._default_image_path
491
+ if self.image_data:
492
+ return next(iter(self.image_data.values()))
493
+
494
+ def get_group_datasets(self, name_or_path: str) -> list[str] | None:
495
+ """Find the path associate with the given name and return all datasets in that group"""
496
+ group_path = self.get_group_path(name_or_path)
497
+ if group_path:
498
+ return self.groups[group_path].datasets
499
+
500
+ "--------------------------------------------------------"
501
+ "---------------------- FILE READERS --------------------"
502
+ "--------------------------------------------------------"
503
+
504
+ def get_data(self, hdf_file: h5py.File, name_or_path: str, index=(), default=None, direct_load=False):
505
+ """
506
+ Return data from dataset in file, converted into either datetime, str or squeezed numpy.array objects
507
+ See hdfmap.eval_functions.dataset2data for more information.
508
+ :param hdf_file: hdf file object
509
+ :param name_or_path: str name or path pointing to dataset in hdf file
510
+ :param index: index or slice of data in hdf file
511
+ :param default: value to return if name not found in hdf file
512
+ :param direct_load: return str, datetime or squeezed array if False, otherwise load data directly
513
+ :return: dataset2data(dataset) -> datetime, str or squeezed array as required.
514
+ """
515
+ path = self.get_path(name_or_path)
516
+ if path and path in hdf_file:
517
+ return dataset2data(hdf_file[path], index, direct_load)
518
+ return default
519
+
520
+ def get_metadata(self, hdf_file: h5py.File, default=None, direct_load=False, name_list: list = None) -> dict:
521
+ """
522
+ Return metadata dict from file, loading data for each item in the metadata list
523
+ The metadata list is taken from name_list, otherwise self.metadata or self.values
524
+ :param hdf_file: hdf file object
525
+ :param default: Value to return for names not associated with a dataset
526
+ :param direct_load: if True, loads data from hdf file directory, without conversion
527
+ :param name_list: if available, uses this list of dataset names to generate the metadata list
528
+ :return:
529
+ """
530
+ extra = extra_hdf_data(hdf_file)
531
+ if name_list:
532
+ metadata_paths = {name: self.combined.get(name, '') for name in name_list}
533
+ else:
534
+ metadata_paths = self.metadata if len(self.metadata) > 0 else self.values
535
+ metadata = {
536
+ name: dataset2data(hdf_file[path], direct_load=direct_load) if path in hdf_file else default
537
+ for name, path in metadata_paths.items()
538
+ }
539
+ return {**extra, **metadata}
540
+
541
+ def get_scannables(self, hdf_file: h5py.File) -> dict:
542
+ """Return scannables from file (values associated with hdfmap.scannables)"""
543
+ return {
544
+ name: hdf_file[path][()] for name, path in self.scannables.items()
545
+ if path in hdf_file
546
+ }
547
+
548
+ def get_image(self, hdf_file: h5py.File, index: slice = None) -> np.ndarray | None:
549
+ """
550
+ Get image data from file, using default image path
551
+ :param hdf_file: hdf file object
552
+ :param index: (slice,) or None to take the middle image
553
+ :return: numpy array of image
554
+ """
555
+ if index is None:
556
+ index = self.scannables_length() // 2
557
+ image_path = self.get_image_path()
558
+ logger.debug(f"image path: {image_path}")
559
+ if image_path and image_path in hdf_file:
560
+ return hdf_file[image_path][index].squeeze() # remove trailing dimensions
561
+
562
+ def _get_numeric_scannables(self, hdf_file: h5py.File) -> list[tuple[str, str]]:
563
+ """Return numeric scannables available in file"""
564
+ return [
565
+ (name, path) for name, path in self.scannables.items()
566
+ if hdf_file.get(path) and np.issubdtype(hdf_file.get(path).dtype, np.number)
567
+ ]
568
+
569
+ def get_scannables_array(self, hdf_file: h5py.File) -> np.ndarray:
570
+ """Return 2D array of all scannables in file"""
571
+ _scannables = self._get_numeric_scannables(hdf_file)
572
+ dtypes = np.dtype([
573
+ (name, hdf_file[path].dtype) for name, path in _scannables
574
+ ])
575
+ return np.array([hdf_file.get(path)[()] for name, path in _scannables], dtype=dtypes)
576
+
577
+ def create_scannables_table(self, hdf_file: h5py.File, delimiter=', ',
578
+ string_spec='', format_spec='f', default_decimals=8) -> str:
579
+ """
580
+ Return str representation of scannables as a table
581
+ The table starts with a header row given by names of the scannables.
582
+ Each row contains the numeric values for each scannable, formated by the given string spec:
583
+ {value: "string_spec.decimals format_spec"}
584
+ e.g. {value: "5.8f"}
585
+ decimals is taken from each scannables "decimals" attribute if it exits, otherwise uses default
586
+ :param hdf_file: h5py.File object
587
+ :param delimiter: str seperator between each column
588
+ :param string_spec: str first element of float format specifier - length of string
589
+ :param format_spec: str type element of format specifier - 'f'=float, 'e'=exponential, 'g'=general
590
+ :param default_decimals: int default number of decimals given
591
+ :return: str
592
+ """
593
+ _scannables = self._get_numeric_scannables(hdf_file)
594
+ fmt = string_spec + '.%d' + format_spec
595
+ formats = [
596
+ '{:' + fmt % self.get_attr(path, 'decimals', default=default_decimals) + '}'
597
+ for name, path in _scannables
598
+ ]
599
+
600
+ length = self.scannables_length()
601
+ out = delimiter.join([name for name, _ in _scannables]) + '\n'
602
+ out += '\n'.join([
603
+ delimiter.join([
604
+ fmt.format(hdf_file.get(path)[n])
605
+ for (_, path), fmt in zip(_scannables, formats)
606
+ ])
607
+ for n in range(length)
608
+ ])
609
+ return out
610
+
611
+ def get_dataholder(self, hdf_file: h5py.File) -> DataHolder:
612
+ """
613
+ Return DataHolder object - a simple replication of scisoftpy.dictutils.DataHolder
614
+ Also known as DLS dat format.
615
+ dataholder.scannable -> array
616
+ dataholder.metadata.value -> metadata
617
+ dataholder['scannable'] -> array
618
+ dataholder.metadata['value'] -> metadata
619
+ :param hdf_file: h5py.File object
620
+ :return: data_object (similar to dict)
621
+ """
622
+ metadata = self.get_metadata(hdf_file)
623
+ scannables = self.get_scannables(hdf_file)
624
+ scannables['metadata'] = DataHolder(**metadata)
625
+ return DataHolder(**scannables)
626
+
627
+ def eval(self, hdf_file: h5py.File, expression: str):
628
+ """
629
+ Evaluate an expression using the namespace of the hdf file
630
+ :param hdf_file: h5py.File object
631
+ :param expression: str expression to be evaluated
632
+ :return: eval(expression)
633
+ """
634
+ return eval_hdf(hdf_file, expression, self.combined)
635
+
636
+ def format_hdf(self, hdf_file: h5py.File, expression: str) -> str:
637
+ """
638
+ Evaluate a formatted string expression using the namespace of the hdf file
639
+ :param hdf_file: h5py.File object
640
+ :param expression: str expression using {name} format specifiers
641
+ :return: eval_hdf(f"expression")
642
+ """
643
+ return format_hdf(hdf_file, expression, self.combined)
644
+
645
+ def info_data(self, hdf_file):
646
+ """Return string showing metadata values associated with names"""
647
+ out = repr(self) + '\n'
648
+ out += "Combined Namespace:\n"
649
+ out += '\n'.join([
650
+ f"{name:>30}: " +
651
+ f"{str(data if np.size(data := dataset2data(hdf_file[path])) <= 1 else self.datasets[path].shape):20}" +
652
+ f": {path:60}"
653
+ for name, path in self.combined.items()
654
+ ])
655
+ out += f"\n{self.info_scannables()}"
656
+ return out