hdfmap 0.4__tar.gz → 0.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: hdfmap
3
- Version: 0.4
3
+ Version: 0.5
4
4
  Summary: Map objects within a HDF file and create a dataset namespace
5
5
  Author-email: Dan Porter <dan.porter@diamond.ac.uk>
6
6
  Maintainer-email: Dan Porter <dan.porter@diamond.ac.uk>
@@ -206,11 +206,11 @@ License: Apache License
206
206
  See the License for the specific language governing permissions and
207
207
  limitations under the License.
208
208
 
209
- Project-URL: Homepage, https://github.com/DanPorter/hdfmap
210
- Project-URL: Documentation, https://github.com/DanPorter/hdfmap
211
- Project-URL: Repository, https://github.com/DanPorter/hdfmap
212
- Project-URL: Bug Tracker, https://github.com/DanPorter/hdfmap
213
- Project-URL: Changelog, https://github.com/DanPorter/hdfmap/blob/master/README.md
209
+ Project-URL: Homepage, https://github.com/DiamondLightSource/hdfmap
210
+ Project-URL: Documentation, https://diamondlightsource.github.io/hdfmap/
211
+ Project-URL: Repository, https://github.com/DiamondLightSource/hdfmap
212
+ Project-URL: Bug Tracker, https://github.com/DiamondLightSource/hdfmap
213
+ Project-URL: Changelog, https://github.com/DiamondLightSource/hdfmap/blob/master/README.md
214
214
  Keywords: nexus
215
215
  Classifier: Programming Language :: Python :: 3.10
216
216
  Classifier: Intended Audience :: Science/Research
@@ -231,7 +231,7 @@ Map objects within an HDF file and create a dataset namespace.
231
231
  [![License](https://img.shields.io/badge/License-Apache%202.0-blue.svg)](https://opensource.org/licenses/Apache-2.0)
232
232
  [![](https://img.shields.io/github/forks/DiamondLightSource/hdfmap?label=GitHub%20Repo&style=social)](https://github.com/DiamondLightSource/hdfmap)
233
233
 
234
- **Version 0.4**
234
+ **Version 0.5**
235
235
 
236
236
  | By Dan Porter |
237
237
  |----------------------|
@@ -349,7 +349,7 @@ map.image_data = {'name': '/hdf/group/dataset'}
349
349
  | `map.get_metadata(h5py.File)` | returns dict of value datasets |
350
350
  | `map.get_scannables(h5py.File)` | returns dict of scannable datasets |
351
351
  | `map.get_scannalbes_array(h5py.File)` | returns numpy array of scannable datasets |
352
- | `map.get_data_block(h5py.File)` | returns dict like object with metadata and scannables |
352
+ | `map.get_dataholder(h5py.File)` | returns dict like object with metadata and scannables |
353
353
  | `map.get_image(h5py.File, index)` | returns image data |
354
354
  | `map.get_data(h5py.File, 'name')` | returns data from dataset |
355
355
  | `map.eval(h5py.File, 'expression')` | returns output of expression using dataset names |
@@ -5,7 +5,7 @@ Map objects within an HDF file and create a dataset namespace.
5
5
  [![License](https://img.shields.io/badge/License-Apache%202.0-blue.svg)](https://opensource.org/licenses/Apache-2.0)
6
6
  [![](https://img.shields.io/github/forks/DiamondLightSource/hdfmap?label=GitHub%20Repo&style=social)](https://github.com/DiamondLightSource/hdfmap)
7
7
 
8
- **Version 0.4**
8
+ **Version 0.5**
9
9
 
10
10
  | By Dan Porter |
11
11
  |----------------------|
@@ -123,7 +123,7 @@ map.image_data = {'name': '/hdf/group/dataset'}
123
123
  | `map.get_metadata(h5py.File)` | returns dict of value datasets |
124
124
  | `map.get_scannables(h5py.File)` | returns dict of scannable datasets |
125
125
  | `map.get_scannalbes_array(h5py.File)` | returns numpy array of scannable datasets |
126
- | `map.get_data_block(h5py.File)` | returns dict like object with metadata and scannables |
126
+ | `map.get_dataholder(h5py.File)` | returns dict like object with metadata and scannables |
127
127
  | `map.get_image(h5py.File, index)` | returns image data |
128
128
  | `map.get_data(h5py.File, 'name')` | returns data from dataset |
129
129
  | `map.eval(h5py.File, 'expression')` | returns output of expression using dataset names |
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "hdfmap"
7
- version = "0.4"
7
+ version = "0.5"
8
8
  dependencies = [
9
9
  "numpy",
10
10
  "h5py",
@@ -32,14 +32,9 @@ classifiers = [
32
32
  ]
33
33
 
34
34
  [project.urls]
35
- Homepage = "https://github.com/DanPorter/hdfmap"
36
- Documentation = "https://github.com/DanPorter/hdfmap"
37
- Repository = "https://github.com/DanPorter/hdfmap"
38
- "Bug Tracker" = "https://github.com/DanPorter/hdfmap"
39
- Changelog = "https://github.com/DanPorter/hdfmap/blob/master/README.md"
35
+ Homepage = "https://github.com/DiamondLightSource/hdfmap"
36
+ Documentation = "https://diamondlightsource.github.io/hdfmap/"
37
+ Repository = "https://github.com/DiamondLightSource/hdfmap"
38
+ "Bug Tracker" = "https://github.com/DiamondLightSource/hdfmap"
39
+ Changelog = "https://github.com/DiamondLightSource/hdfmap/blob/master/README.md"
40
40
 
41
-
42
- [tool.pytest.ini_options]
43
- pythonpath = [
44
- "src",
45
- ]
@@ -2,29 +2,29 @@
2
2
  hdfmap
3
3
  Map objects within an HDF file and create a dataset namespace.
4
4
 
5
- --- Usage ---
6
- # HdfMap from NeXus file:
7
- from hdfmap import create_nexus_map, load_hdf
8
- hmap = create_nexus_map('file.nxs')
9
- with load_hdf('file.nxs') as nxs:
10
- address = hmap.get_address('energy')
11
- energy = nxs[address][()]
12
- string = hmap.format_hdf(nxs, "the energy is {energy:.2f} keV")
13
- d = hmap.get_data_block(nxs) # classic data table, d.scannable, d.metadata
5
+ # Usage
6
+ ### HdfMap from NeXus file
7
+ from hdfmap import create_nexus_map, load_hdf
8
+ hmap = create_nexus_map('file.nxs')
9
+ with load_hdf('file.nxs') as nxs:
10
+ address = hmap.get_address('energy')
11
+ energy = nxs[address][()]
12
+ string = hmap.format_hdf(nxs, "the energy is {energy:.2f} keV")
13
+ d = hmap.get_dataholder(nxs) # classic data table, d.scannable, d.metadata
14
14
 
15
- # Shortcuts - single file reloading class
16
- from hdfmap import NexusLoader
17
- scan = NexusLoader('file.nxs')
18
- [data1, data2] = scan.get_data(['dataset_name_1', 'dataset_name_2'])
19
- data = scan.eval('dataset_name_1 * 100 + 2')
20
- string = scan.format('my data is {dataset_name_1:.2f}')
15
+ ### Shortcuts - single file reloading class
16
+ from hdfmap import NexusLoader
17
+ scan = NexusLoader('file.nxs')
18
+ [data1, data2] = scan.get_data(['dataset_name_1', 'dataset_name_2'])
19
+ data = scan.eval('dataset_name_1 * 100 + 2')
20
+ string = scan.format('my data is {dataset_name_1:.2f}')
21
21
 
22
- # Shortcuts - multifile load data
23
- from hdfmap import hdf_data, hdf_eval, hdf_format, hdf_image
24
- all_data = hdf_data([f"file{n}.nxs" for n in range(100)], 'dataset_name')
25
- normalised_data = hdf_eval(filenames, 'total / Transmission / (rc / 300.)')
26
- descriptions = hdf_eval(filenames, 'Energy: {en:5.3f} keV')
27
- image = hdf_image(filenames, index=31)
22
+ ### Shortcuts - multifile load data
23
+ from hdfmap import hdf_data, hdf_eval, hdf_format, hdf_image
24
+ all_data = hdf_data([f"file{n}.nxs" for n in range(100)], 'dataset_name')
25
+ normalised_data = hdf_eval(filenames, 'total / Transmission / (rc / 300.)')
26
+ descriptions = hdf_eval(filenames, 'Energy: {en:5.3f} keV')
27
+ image = hdf_image(filenames, index=31)
28
28
 
29
29
 
30
30
  By Dr Dan Porter
@@ -33,9 +33,10 @@ Diamond Light Source Ltd
33
33
  """
34
34
 
35
35
  from .logging import set_all_logging_level
36
+ from .hdf_loader import load_hdf
36
37
  from .hdfmap_class import HdfMap
37
38
  from .nexus import NexusMap
38
- from .file_functions import list_files, load_hdf, create_hdf_map, create_nexus_map
39
+ from .file_functions import list_files, create_hdf_map, create_nexus_map
39
40
  from .file_functions import hdf_data, hdf_image, hdf_eval, hdf_format, nexus_data_block
40
41
  from .reloader_class import HdfLoader, NexusLoader
41
42
 
@@ -46,8 +47,8 @@ __all__ = [
46
47
  set_all_logging_level
47
48
  ]
48
49
 
49
- __version__ = "0.4.0"
50
- __date__ = "2024/08/16"
50
+ __version__ = "0.5.0"
51
+ __date__ = "2024/09/25"
51
52
 
52
53
 
53
54
  def version_info() -> str:
@@ -57,6 +57,11 @@ def dataset2data(dataset: h5py.Dataset, index: int | slice = (), direct_load=Fal
57
57
  return dataset[index]
58
58
  if np.issubdtype(dataset, np.number):
59
59
  return np.squeeze(dataset[index]) # numeric np.ndarray
60
+ try:
61
+ # str integers will be cast as timestamps (years), capture as int
62
+ return np.squeeze(dataset[index]).astype(int)
63
+ except ValueError:
64
+ pass
60
65
  try:
61
66
  # timestamp -> datetime64 -> datetime
62
67
  timestamp = np.squeeze(dataset[index]).astype(np.datetime64).astype(datetime.datetime)
@@ -72,6 +77,40 @@ def dataset2data(dataset: h5py.Dataset, index: int | slice = (), direct_load=Fal
72
77
  return np.squeeze(dataset[index]) # other np.ndarray
73
78
 
74
79
 
80
+ def dataset2str(dataset: h5py.Dataset, index: int | slice = ()) -> str:
81
+ """
82
+ Read the data from a h5py Dataset and convert to a representative string
83
+ Strings are given with quotes
84
+ numbers are shorted by attribute 'decimals'
85
+ numeric arrays are summarised as "dtype (shape)"
86
+ string arrays are summarised as "['str0', ...]
87
+
88
+ :param dataset: h5py.Dataset containing data
89
+ :param index: index of array (not used if dataset is string/ bytes type)
90
+ :return str: string representation of data
91
+ """
92
+ if np.issubdtype(dataset, np.number):
93
+ if dataset.size > 1:
94
+ return f"{dataset.dtype} {dataset.shape}"
95
+ value = np.squeeze(dataset[index]) # numeric np.ndarray
96
+ if 'decimals' in dataset.attrs:
97
+ value = value.round(dataset.attrs['decimals'])
98
+ return str(value)
99
+ try:
100
+ # timestamp -> datetime64 -> datetime
101
+ timestamp = np.squeeze(dataset[index]).astype(np.datetime64).astype(datetime.datetime)
102
+ # single datetime obj vs array of datetime obj
103
+ return f"'{timestamp[()]}'" if timestamp.ndim == 0 else f"['{timestamp[0]}', ...({len(timestamp)})]"
104
+ except ValueError:
105
+ try:
106
+ string_dataset = dataset.asstr()[()]
107
+ if dataset.ndim == 0:
108
+ return f"'{round_string_floats(string_dataset)}'" # bytes or str -> str
109
+ return f"['{string_dataset[0]}', ...({len(string_dataset)})]" # str array
110
+ except ValueError:
111
+ return str(np.squeeze(dataset[index])) # other np.ndarray
112
+
113
+
75
114
  def check_unsafe_eval(eval_str: str) -> None:
76
115
  """
77
116
  Check str for naughty eval arguments such as sys, os or import
@@ -1,23 +1,15 @@
1
1
  import os
2
- import h5py
3
2
  import numpy as np
4
3
 
4
+ from . import load_hdf, HdfMap, NexusMap
5
5
  from .logging import create_logger
6
- from .hdfmap_class import HdfMap
7
- from .nexus import NexusMap
8
6
 
9
7
 
10
8
  EXTENSIONS = ['.nxs', '.hdf', '.hdf5', '.h5']
11
9
  DEFAULT_EXTENSION = EXTENSIONS[0]
12
- DEFAULT_HDF_PATH = "entry1/scan_command"
13
10
  logger = create_logger(__name__)
14
11
 
15
12
 
16
- def load_hdf(hdf_filename: str) -> h5py.File:
17
- """Load hdf file, return h5py.File object"""
18
- return h5py.File(hdf_filename, 'r')
19
-
20
-
21
13
  def list_files(folder_directory: str, extension=DEFAULT_EXTENSION) -> list[str]:
22
14
  """Return list of files in directory with extension, returning list of full file paths"""
23
15
  try:
@@ -207,4 +199,3 @@ def nexus_data_block(filenames: str | list[str], hdf_map: HdfMap = None, fixed_o
207
199
  if not fixed_output and len(filenames) == 1:
208
200
  return out[0]
209
201
  return out
210
-
@@ -0,0 +1,12 @@
1
+
2
+ import h5py
3
+
4
+ try:
5
+ import hdf5plugin # required for compressed data
6
+ except ImportError:
7
+ print('Warning: hdf5plugin not available.')
8
+
9
+
10
+ def load_hdf(hdf_filename: str) -> h5py.File:
11
+ """Load hdf file, return h5py.File object"""
12
+ return h5py.File(hdf_filename, 'r')
@@ -8,13 +8,10 @@ from types import SimpleNamespace
8
8
  import numpy as np
9
9
  import h5py
10
10
 
11
+ from . import load_hdf
11
12
  from .logging import create_logger
12
- from .eval_functions import expression_safe_name, extra_hdf_data, eval_hdf, format_hdf, dataset2data
13
+ from .eval_functions import expression_safe_name, extra_hdf_data, eval_hdf, format_hdf, dataset2data, dataset2str
13
14
 
14
- try:
15
- import hdf5plugin # required for compressed data
16
- except ImportError:
17
- print('Warning: hdf5plugin not available.')
18
15
 
19
16
  # parameters
20
17
  SEP = '/' # HDF path separator
@@ -66,6 +63,15 @@ def generate_identifier(hdf_path: str | bytes) -> str:
66
63
  return '_'.join(dict.fromkeys(name.split('_')))
67
64
 
68
65
 
66
+ def generate_alt_name(hdf_dataset: h5py.Dataset) -> str:
67
+ """Generate alt_name of dataset if 'local_name' in attributes"""
68
+ if LOCAL_NAME in hdf_dataset.attrs:
69
+ alt_name = hdf_dataset.attrs[LOCAL_NAME]
70
+ if hasattr(alt_name, 'decode'):
71
+ alt_name = alt_name.decode()
72
+ return expression_safe_name(alt_name.split('.')[-1])
73
+
74
+
69
75
  def build_hdf_path(*args: str | bytes) -> str:
70
76
  """
71
77
  Build path from string or bytes arguments
@@ -117,59 +123,62 @@ class HdfMap:
117
123
  outstr = map.format(hdf, 'the data looks like: {data}')
118
124
 
119
125
  Objects within the HDF file are separated into Groups and Datasets. Each object has a
120
- defined 'path' and 'name' paramater, as well as other attributes
121
- path -> '/entry/measurement/data' -> the location of an object within the file
122
- name -> 'data' -> an path expressed as a simple variable name
126
+ defined 'path' and 'name' paramater, as well as other attribute:
127
+
128
+ - path -> '/entry/measurement/data' -> the location of an object within the file
129
+ - name -> 'data' -> a path expressed as a simple variable name
130
+
123
131
  Paths are unique location within the file but can be used to identify similar objects in other files
124
132
  Names may not be unique within a file and are generated from the final element of the hdf path.
125
- - When multiple paths produce the same name, the name is overwritten each time, so the last path in the
133
+
134
+ - When multiple paths produce the same name, the name is overwritten each time, so the last path in the
126
135
  file has priority.
127
- - Names are also stored using the 'local_name' attribute, if it exists
136
+ - Names are also stored using the 'local_name' attribute, if it exists
128
137
 
129
138
  Names of different types of datasets are stored for arrays (size > 0) and values (size 0)
130
139
  Names for scannables relate to all arrays of a particular size
131
140
  A combined list of names is provided where scannables > arrays > values
132
141
 
133
-
134
-
135
- Attributes:
136
- map.groups stores attributes of each group by path
137
- map.classes stores list of group paths by nx_class
138
- map.datasets stores attributes of each dataset by path
139
- map.arrays stores array dataset paths by name
140
- map.values stores value dataset paths by name
141
- map.scannables stores array dataset paths with given size, by name
142
- map.combined stores array and value paths (arrays overwrite values)
143
- map.image_data stores dataset paths of image data
144
- E.G.
145
- map.groups = {'/hdf/group': ('class', 'name', {attrs}, [datasets])}
146
- map.classes = {'class_name': ['/hdf/group1', '/hdf/group2']}
147
- map.datasets = {'/hdf/group/dataset': ('name', size, shape, {attrs})}
148
- map.arrays = {'name': '/hdf/group/dataset'}
149
- map.values = {'name': '/hdf/group/dataset'}
150
- map.scannables = {'name': '/hdf/group/dataset'}
151
- map.image_data = {'name': '/hdf/group/dataset'}
152
-
153
- Methods:
154
- map.populate(h5py.File) -> populates the dictionaries using the given file
155
- map.generate_scannables(array_size) -> populates scannables namespace with arrays of same size
156
- map.most_common_size -> returns the most common dataset size > 1
157
- map.get_size('name_or_path') -> return dataset size
158
- map.get_shape('name_or_path') -> return dataset size
159
- map.get_attr('name_or_path', 'attr') -> return value of dataset attribute
160
- map.get_path('name_or_group_or_class') -> returns path of object with name
161
- map.get_image_path() -> returns default path of detector dataset (or largest dataset)
162
- map.get_group_path('name_or_path_or_class') -> return path of group with class
163
- map.get_group_datasets('name_or_path_or_class') -> return list of dataset pathes in class
164
- File Methods:
165
- map.get_metadata(h5py.File) -> returns dict of value datasets
166
- map.get_scannables(h5py.File) -> returns dict of scannable datasets
167
- map.get_scannalbes_array(h5py.File) -> returns numpy array of scannable datasets
168
- map.get_data_block(h5py.File) -> returns dict like object with metadata and scannables
169
- map.get_image(h5py.File, index) -> returns image data
170
- map.get_data(h5py.File, 'name') -> returns data from dataset
171
- map.eval(h5py.File, 'expression') -> returns output of expression
172
- map.format(h5py.File, 'string {name}') -> returns output of str expression
142
+ ### Attributes
143
+ - map.groups stores attributes of each group by path
144
+ - map.classes stores list of group paths by nx_class
145
+ - map.datasets stores attributes of each dataset by path
146
+ - map.arrays stores array dataset paths by name
147
+ - map.values stores value dataset paths by name
148
+ - map.metadata stores value dataset path by altname only
149
+ - map.scannables stores array dataset paths with given size, by name
150
+ - map.combined stores array and value paths (arrays overwrite values)
151
+ - map.image_data stores dataset paths of image data
152
+ #### E.G.
153
+ - map.groups = {'/hdf/group': ('class', 'name', {attrs}, [datasets])}
154
+ - map.classes = {'class_name': ['/hdf/group1', '/hdf/group2']}
155
+ - map.datasets = {'/hdf/group/dataset': ('name', size, shape, {attrs})}
156
+ - map.arrays = {'name': '/hdf/group/dataset'}
157
+ - map.values = {'name': '/hdf/group/dataset'}
158
+ - map.scannables = {'name': '/hdf/group/dataset'}
159
+ - map.image_data = {'name': '/hdf/group/dataset'}
160
+
161
+ ### Methods
162
+ - map.populate(h5py.File) -> populates the dictionaries using the given file
163
+ - map.generate_scannables(array_size) -> populates scannables namespace with arrays of same size
164
+ - map.most_common_size -> returns the most common dataset size > 1
165
+ - map.get_size('name_or_path') -> return dataset size
166
+ - map.get_shape('name_or_path') -> return dataset size
167
+ - map.get_attr('name_or_path', 'attr') -> return value of dataset attribute
168
+ - map.get_path('name_or_group_or_class') -> returns path of object with name
169
+ - map.get_image_path() -> returns default path of detector dataset (or largest dataset)
170
+ - map.get_group_path('name_or_path_or_class') -> return path of group with class
171
+ - map.get_group_datasets('name_or_path_or_class') -> return list of dataset pathes in class
172
+ ### File Methods
173
+ - map.get_metadata(h5py.File) -> returns dict of value datasets
174
+ - map.get_scannables(h5py.File) -> returns dict of scannable datasets
175
+ - map.get_scannables_array(h5py.File) -> returns numpy array of scannable datasets
176
+ - map.get_dataholder(h5py.File) -> returns dict like object with metadata and scannables
177
+ - map.get_image(h5py.File, index) -> returns image data
178
+ - map.get_data(h5py.File, 'name') -> returns data from dataset
179
+ - map.get_string(h5py.File, 'name') -> returns string summary of dataset
180
+ - map.eval(h5py.File, 'expression') -> returns output of expression
181
+ - map.format(h5py.File, 'string {name}') -> returns output of str expression
173
182
  """
174
183
 
175
184
  def __init__(self, file: h5py.File | None = None):
@@ -208,8 +217,21 @@ class HdfMap:
208
217
  """Return str info on groups"""
209
218
  out = f"{repr(self)}\n"
210
219
  out += "Groups:\n"
211
- out += disp_dict(self.groups, 20)
212
- out += '\n\nClasses:\n'
220
+ for path, group in self.groups.items():
221
+ out += f"{path} [{group.nx_class}: '{group.name}']\n"
222
+ out += '\n'.join(f" @{attr}: {self.get_attr(path, attr)}" for attr in group.attrs)
223
+ out += '\n'
224
+ for dataset_name in group.datasets:
225
+ dataset_path = build_hdf_path(path, dataset_name)
226
+ if dataset_path in self.datasets:
227
+ dataset = self.datasets[dataset_path]
228
+ out += f" {dataset_name}: {dataset.shape}\n"
229
+ return out
230
+
231
+ def info_classes(self) -> str:
232
+ """Return str info on group class names"""
233
+ out = f"{repr(self)}\n"
234
+ out += 'Classes:\n'
213
235
  out += disp_dict(self.classes, 20)
214
236
  return out
215
237
 
@@ -274,7 +296,8 @@ class HdfMap:
274
296
  # New: add group_name to namespace as standard, helps with names like s5/x + s4/x
275
297
  # this significantly increases the number of names in namespaces
276
298
  group_name = generate_identifier(f"{hdf_path.split(SEP)[-2]}_{name}")
277
- alt_name = generate_identifier(hdf_dataset.attrs[LOCAL_NAME]) if LOCAL_NAME in hdf_dataset.attrs else None
299
+ # alt_name = generate_identifier(hdf_dataset.attrs[LOCAL_NAME]) if LOCAL_NAME in hdf_dataset.attrs else None
300
+ alt_name = generate_alt_name(hdf_dataset)
278
301
  names = {n: hdf_path for n in {name, group_name, alt_name} if n}
279
302
  self.datasets[hdf_path] = Dataset(
280
303
  name=name,
@@ -357,11 +380,11 @@ class HdfMap:
357
380
  return max(set(array_shapes), key=array_shapes.count)
358
381
 
359
382
  def scannables_length(self) -> int:
383
+ """Return the length of the first axis of scannables array"""
360
384
  if not self.scannables:
361
385
  return 0
362
386
  path = next(iter(self.scannables.values()))
363
- shape = self.datasets[path].shape
364
- return shape[0]
387
+ return self.datasets[path].size
365
388
 
366
389
  def generate_scannables(self, array_size):
367
390
  """Populate self.scannables field with datasets size that match array_size"""
@@ -418,13 +441,30 @@ class HdfMap:
418
441
  return SEP
419
442
  return hdf_path
420
443
 
421
- def find_paths(self, string: str, name_only=True) -> list[str]:
444
+ def get_group_dataset_path(self, group_name, dataset_name) -> str | None:
445
+ """Return path of dataset defined by group and dataset name/attribute"""
446
+ if group_name in self.groups:
447
+ group_paths = [group_name]
448
+ else:
449
+ group_paths = self.classes[group_name]
450
+ for group_path in group_paths:
451
+ group = self.groups[group_path]
452
+ for name in group.datasets:
453
+ dataset_path = build_hdf_path(group_path, name)
454
+ dataset = self.datasets[dataset_path]
455
+ if dataset_name in dataset.names:
456
+ return dataset_path
457
+
458
+ def find_paths(self, string: str, name_only=True, whole_word=False) -> list[str]:
422
459
  """
423
460
  Find any dataset paths that contain the given string argument
424
461
  :param string: str to find in list of datasets
425
462
  :param name_only: if True, search only the name of the dataset, not the full path
463
+ :param whole_word: if True, search only for whole-word names (case in-sensitive)
426
464
  :return: list of hdf paths
427
465
  """
466
+ if whole_word:
467
+ return [path for name, path in self.combined.items() if string.lower() == name.lower()]
428
468
  # find string in combined
429
469
  combined_paths = [path for name, path in self.combined.items() if string in name]
430
470
  if name_only:
@@ -491,6 +531,17 @@ class HdfMap:
491
531
  if self.image_data:
492
532
  return next(iter(self.image_data.values()))
493
533
 
534
+ def get_image_shape(self) -> tuple:
535
+ """Return the scan shape of the detector dataset"""
536
+ path = self.get_image_path()
537
+ if path:
538
+ return self.datasets[path].shape
539
+ return 0, 0
540
+
541
+ def get_image_index(self, index: int) -> tuple:
542
+ """Return image slice index for index along total scan size"""
543
+ return np.unravel_index(index, self.get_image_shape()[:-2])
544
+
494
545
  def get_group_datasets(self, name_or_path: str) -> list[str] | None:
495
546
  """Find the path associate with the given name and return all datasets in that group"""
496
547
  group_path = self.get_group_path(name_or_path)
@@ -501,6 +552,19 @@ class HdfMap:
501
552
  "---------------------- FILE READERS --------------------"
502
553
  "--------------------------------------------------------"
503
554
 
555
+ def load_hdf(self, filename: str | None = None, name_or_path: str = None) -> h5py.File | h5py.Dataset:
556
+ """
557
+ Load hdf file or hdf dataset in open state
558
+ :param filename: str filename of hdf file, or None to use self.filename
559
+ :param name_or_path: if given, returns the dataset
560
+ :return: h5py.File object or h5py.dataset object if dataset name given
561
+ """
562
+ if filename is None:
563
+ filename = self.filename
564
+ if name_or_path is None:
565
+ return load_hdf(filename)
566
+ return load_hdf(filename).get(self.get_path(name_or_path))
567
+
504
568
  def get_data(self, hdf_file: h5py.File, name_or_path: str, index=(), default=None, direct_load=False):
505
569
  """
506
570
  Return data from dataset in file, converted into either datetime, str or squeezed numpy.array objects
@@ -517,7 +581,23 @@ class HdfMap:
517
581
  return dataset2data(hdf_file[path], index, direct_load)
518
582
  return default
519
583
 
520
- def get_metadata(self, hdf_file: h5py.File, default=None, direct_load=False, name_list: list = None) -> dict:
584
+ def get_string(self, hdf_file: h5py.File, name_or_path: str, index=(), default='') -> str:
585
+ """
586
+ Return data from dataset in file, converted into string summary of data
587
+ See hdfmap.eval_functions.dataset2str for more information.
588
+ :param hdf_file: hdf file object
589
+ :param name_or_path: str name or path pointing to dataset in hdf file
590
+ :param index: index or slice of data in hdf file
591
+ :param default: value to return if name not found in hdf file
592
+ :return: dataset2str(dataset) -> str
593
+ """
594
+ path = self.get_path(name_or_path)
595
+ if path and path in hdf_file:
596
+ return dataset2str(hdf_file[path], index)
597
+ return default
598
+
599
+ def get_metadata(self, hdf_file: h5py.File, default=None, direct_load=False,
600
+ name_list: list = None, string_output=False) -> dict:
521
601
  """
522
602
  Return metadata dict from file, loading data for each item in the metadata list
523
603
  The metadata list is taken from name_list, otherwise self.metadata or self.values
@@ -525,27 +605,53 @@ class HdfMap:
525
605
  :param default: Value to return for names not associated with a dataset
526
606
  :param direct_load: if True, loads data from hdf file directory, without conversion
527
607
  :param name_list: if available, uses this list of dataset names to generate the metadata list
528
- :return:
608
+ :param string_output: if True, returns string summary of each value
609
+ :return: {name: value}
529
610
  """
530
611
  extra = extra_hdf_data(hdf_file)
531
612
  if name_list:
532
613
  metadata_paths = {name: self.combined.get(name, '') for name in name_list}
533
614
  else:
534
- metadata_paths = self.metadata if len(self.metadata) > 0 else self.values
535
- metadata = {
536
- name: dataset2data(hdf_file[path], direct_load=direct_load) if path in hdf_file else default
537
- for name, path in metadata_paths.items()
538
- }
615
+ metadata_paths = self.metadata if self.metadata else self.values
616
+ if string_output:
617
+ extra = {key: f"'{val}'" for key, val in extra.items()}
618
+ metadata = {
619
+ name: dataset2str(hdf_file[path]) if path in hdf_file else str(default)
620
+ for name, path in metadata_paths.items()
621
+ }
622
+ else:
623
+ metadata = {
624
+ name: dataset2data(hdf_file[path], direct_load=direct_load) if path in hdf_file else default
625
+ for name, path in metadata_paths.items()
626
+ }
539
627
  return {**extra, **metadata}
540
628
 
541
- def get_scannables(self, hdf_file: h5py.File) -> dict:
629
+ def create_metadata_list(self, hdf_file: h5py.File, default=None, name_list: list = None,
630
+ line_separator: str = '\n', value_separator: str = '=') -> str:
631
+ """
632
+ Return a metadata string, using self.get_metadata
633
+ :param hdf_file: hdf file object
634
+ :param default: Value to return for names not associated with a dataset
635
+ :param name_list: if available, uses this list of dataset names to generate the metadata list
636
+ :param line_separator: str separating each metadata parameter
637
+ :param value_separator: str separating name from value
638
+ :return: multi-line string
639
+ """
640
+ return line_separator.join(
641
+ f"{name}{value_separator}{value}"
642
+ for name, value in self.get_metadata(hdf_file, default=default,
643
+ name_list=name_list, string_output=True).items()
644
+ )
645
+
646
+ def get_scannables(self, hdf_file: h5py.File, flatten: bool = False) -> dict:
542
647
  """Return scannables from file (values associated with hdfmap.scannables)"""
543
648
  return {
544
- name: hdf_file[path][()] for name, path in self.scannables.items()
649
+ name: hdf_file[path][()].flatten() if flatten else hdf_file[path][()]
650
+ for name, path in self.scannables.items()
545
651
  if path in hdf_file
546
652
  }
547
653
 
548
- def get_image(self, hdf_file: h5py.File, index: slice = None) -> np.ndarray | None:
654
+ def get_image(self, hdf_file: h5py.File, index: int | tuple | slice = None) -> np.ndarray | None:
549
655
  """
550
656
  Get image data from file, using default image path
551
657
  :param hdf_file: hdf file object
@@ -553,26 +659,28 @@ class HdfMap:
553
659
  :return: numpy array of image
554
660
  """
555
661
  if index is None:
556
- index = self.scannables_length() // 2
662
+ index = self.get_image_index(self.scannables_length() // 2)
663
+ if isinstance(index, int):
664
+ index = self.get_image_index(index)
557
665
  image_path = self.get_image_path()
558
666
  logger.debug(f"image path: {image_path}")
559
667
  if image_path and image_path in hdf_file:
560
668
  return hdf_file[image_path][index].squeeze() # remove trailing dimensions
561
669
 
562
- def _get_numeric_scannables(self, hdf_file: h5py.File) -> list[tuple[str, str]]:
670
+ def _get_numeric_scannables(self, hdf_file: h5py.File) -> list[tuple[str, str, np.ndarray]]:
563
671
  """Return numeric scannables available in file"""
564
672
  return [
565
- (name, path) for name, path in self.scannables.items()
566
- if hdf_file.get(path) and np.issubdtype(hdf_file.get(path).dtype, np.number)
673
+ (name, path, dataset[()].flatten()) for name, path in self.scannables.items()
674
+ if (dataset := hdf_file.get(path)) and np.issubdtype(dataset.dtype, np.number)
567
675
  ]
568
676
 
569
677
  def get_scannables_array(self, hdf_file: h5py.File) -> np.ndarray:
570
- """Return 2D array of all scannables in file"""
678
+ """Return 2D array of all numeric scannables in file"""
571
679
  _scannables = self._get_numeric_scannables(hdf_file)
572
680
  dtypes = np.dtype([
573
- (name, hdf_file[path].dtype) for name, path in _scannables
681
+ (name, hdf_file[path].dtype) for name, path, array in _scannables
574
682
  ])
575
- return np.array([hdf_file.get(path)[()] for name, path in _scannables], dtype=dtypes)
683
+ return np.array([array for name, path, array in _scannables], dtype=dtypes)
576
684
 
577
685
  def create_scannables_table(self, hdf_file: h5py.File, delimiter=', ',
578
686
  string_spec='', format_spec='f', default_decimals=8) -> str:
@@ -594,21 +702,21 @@ class HdfMap:
594
702
  fmt = string_spec + '.%d' + format_spec
595
703
  formats = [
596
704
  '{:' + fmt % self.get_attr(path, 'decimals', default=default_decimals) + '}'
597
- for name, path in _scannables
705
+ for name, path, array in _scannables
598
706
  ]
599
707
 
600
708
  length = self.scannables_length()
601
- out = delimiter.join([name for name, _ in _scannables]) + '\n'
709
+ out = delimiter.join([name for name, _, _ in _scannables]) + '\n'
602
710
  out += '\n'.join([
603
711
  delimiter.join([
604
- fmt.format(hdf_file.get(path)[n])
605
- for (_, path), fmt in zip(_scannables, formats)
712
+ fmt.format(array[n])
713
+ for (_, path, array), fmt in zip(_scannables, formats)
606
714
  ])
607
715
  for n in range(length)
608
716
  ])
609
717
  return out
610
718
 
611
- def get_dataholder(self, hdf_file: h5py.File) -> DataHolder:
719
+ def get_dataholder(self, hdf_file: h5py.File, flatten_scannables: bool = False) -> DataHolder:
612
720
  """
613
721
  Return DataHolder object - a simple replication of scisoftpy.dictutils.DataHolder
614
722
  Also known as DLS dat format.
@@ -617,10 +725,11 @@ class HdfMap:
617
725
  dataholder['scannable'] -> array
618
726
  dataholder.metadata['value'] -> metadata
619
727
  :param hdf_file: h5py.File object
728
+ :param flatten_scannables: bool, it True the scannables will be flattened arrays
620
729
  :return: data_object (similar to dict)
621
730
  """
622
731
  metadata = self.get_metadata(hdf_file)
623
- scannables = self.get_scannables(hdf_file)
732
+ scannables = self.get_scannables(hdf_file, flatten=flatten_scannables)
624
733
  scannables['metadata'] = DataHolder(**metadata)
625
734
  return DataHolder(**scannables)
626
735
 
@@ -642,13 +751,17 @@ class HdfMap:
642
751
  """
643
752
  return format_hdf(hdf_file, expression, self.combined)
644
753
 
645
- def info_data(self, hdf_file):
754
+ def create_dataset_summary(self, hdf_file: h5py.File) -> str:
755
+ """Create summary of all datasets in file"""
756
+ return '\n'.join(f"{path:60}: {self.get_string(hdf_file, path)}" for path in self.datasets)
757
+
758
+ def info_data(self, hdf_file: h5py.File) -> str:
646
759
  """Return string showing metadata values associated with names"""
647
760
  out = repr(self) + '\n'
648
761
  out += "Combined Namespace:\n"
649
762
  out += '\n'.join([
650
763
  f"{name:>30}: " +
651
- f"{str(data if np.size(data := dataset2data(hdf_file[path])) <= 1 else self.datasets[path].shape):20}" +
764
+ f"{dataset2str(hdf_file[path]):20}" +
652
765
  f": {path:60}"
653
766
  for name, path in self.combined.items()
654
767
  ])
@@ -30,7 +30,8 @@ def set_all_logging_level(level: str | int):
30
30
  """
31
31
  try:
32
32
  level = level.upper()
33
- level = logging.getLevelNamesMapping()[level]
33
+ # level = logging.getLevelNamesMapping()[level] # Python >3.11
34
+ level = logging._nameToLevel[level]
34
35
  except AttributeError:
35
36
  level = int(level)
36
37
 
@@ -147,17 +147,21 @@ class NexusMap(HdfMap):
147
147
  out += disp_dict({k: v for k, v in self.classes.items() if k in nx_classes}, 20)
148
148
  out += '\nDefaults:\n'
149
149
  out += f" @{NX_DEFAULT}: {self.find_attr(NX_DEFAULT)}\n"
150
- out += f" @{NX_AXES}: {self.find_attr(NX_AXES)}\n"
151
- out += f" @{NX_SIGNAL}: {self.find_attr(NX_SIGNAL)}\n"
150
+ out += f" @{NX_AXES}: {self.get_path(NX_AXES)}\n"
151
+ out += f" @{NX_SIGNAL}: {self.get_path(NX_SIGNAL)}\n"
152
152
  return out
153
153
 
154
154
  def _default_nexus_paths(self, hdf_file):
155
155
  """Load Nexus default axes and signal"""
156
156
  try:
157
157
  axes_paths, signal_path = find_nexus_data(hdf_file)
158
- # TODO: add method of including multiple axes, e.g. axes1, axes2, ..., or self.get_axes
159
158
  if axes_paths and axes_paths[0] in hdf_file:
160
159
  self.arrays[NX_AXES] = axes_paths[0]
160
+ n = 0
161
+ for axes_path in axes_paths:
162
+ if axes_path in hdf_file and isinstance(hdf_file[axes_path], h5py.Dataset):
163
+ self.arrays[f"{NX_AXES}{n}"] = axes_path
164
+ n += 1
161
165
  logger.info(f"DEFAULT axes: {axes_paths}")
162
166
  if signal_path in hdf_file:
163
167
  self.arrays[NX_SIGNAL] = signal_path
@@ -165,6 +169,12 @@ class NexusMap(HdfMap):
165
169
  except KeyError:
166
170
  pass
167
171
 
172
+ def nexus_defaults(self):
173
+ """Return default axes and signal paths"""
174
+ axes_paths = [self.arrays[axes] for n in range(10) if (axes := f"{NX_AXES}{n}") in self.arrays]
175
+ signal_path = self.arrays[NX_SIGNAL]
176
+ return axes_paths, signal_path
177
+
168
178
  def _scannables_from_scan_fields_or_nxdata(self, hdf_file: h5py.File):
169
179
  """Generate scannables from scan_field names or default NXdata"""
170
180
  # find 'scan_fields' to generate scannables list
@@ -5,25 +5,25 @@ Reloader class
5
5
  import h5py
6
6
  import numpy as np
7
7
 
8
- from .hdfmap_class import HdfMap
9
- from .nexus import NexusMap
10
- from .file_functions import load_hdf, create_hdf_map, create_nexus_map
8
+ from . import load_hdf, HdfMap, NexusMap
9
+ from .file_functions import create_hdf_map, create_nexus_map
11
10
 
12
11
 
13
12
  class HdfLoader:
14
13
  """
15
- HDF Loader
16
- contains the filename and hdfmap for a HDF file, the hdfmap contains all the dataset paths and a
14
+ HDF Loader contains the filename and hdfmap for a HDF file, the hdfmap contains all the dataset paths and a
17
15
  namespace, allowing data to be called from the file using variable names, loading only the required datasets
18
16
  for each operation.
19
- E.G.
17
+
18
+ ### E.G.
20
19
  hdf = HdfLoader('file.hdf')
21
- [data1, data2] = hdf.get_data(['dataset_name_1', 'dataset_name_2'])
20
+ [data1, data2] = hdf.get_data(*['dataset_name_1', 'dataset_name_2'])
22
21
  data = hdf.eval('dataset_name_1 * 100 + 2')
23
22
  string = hdf.format('my data is {dataset_name_1:.2f}')
23
+ print(hdf.summary())
24
24
  """
25
25
 
26
- def __init__(self, hdf_filename: str, hdf_map: HdfMap | None = None):
26
+ def __init__(self, hdf_filename: str, hdf_map: HdfMap | NexusMap | None = None):
27
27
  self.filename = hdf_filename
28
28
  if hdf_map is None:
29
29
  self.map = create_hdf_map(hdf_filename)
@@ -51,14 +51,15 @@ class HdfLoader:
51
51
  """Return hdf path of object in HdfMap"""
52
52
  return self.map.get_path(name_or_path)
53
53
 
54
- def find_hdf_paths(self, string: str, name_only: bool = True) -> list[str]:
54
+ def find_hdf_paths(self, string: str, name_only: bool = True, whole_word: bool = False) -> list[str]:
55
55
  """
56
56
  Find any dataset paths that contain the given string argument
57
57
  :param string: str to find in list of datasets
58
58
  :param name_only: if True, search only the name of the dataset, not the full path
59
+ :param whole_word: if True, search only for case in-sensitive name
59
60
  :return: list of hdf paths
60
61
  """
61
- return self.map.find_paths(string, name_only)
62
+ return self.map.find_paths(string, name_only, whole_word)
62
63
 
63
64
  def find_names(self, string: str) -> list[str]:
64
65
  """
@@ -84,6 +85,21 @@ class HdfLoader:
84
85
  return out[0]
85
86
  return out
86
87
 
88
+ def get_string(self, *name_or_path, index: slice = (), default=''):
89
+ """
90
+ Return data from dataset in file, converted into summary string
91
+ See hdfmap.eval_functions.dataset2data for more information.
92
+ :param name_or_path: str name or path pointing to dataset in hdf file
93
+ :param index: index or slice of data in hdf file
94
+ :param default: value to return if name not found in hdf file
95
+ :return: dataset2str(dataset) -> str
96
+ """
97
+ with self._load() as hdf:
98
+ out = [self.map.get_string(hdf, name, index, default) for name in name_or_path]
99
+ if len(name_or_path) == 1:
100
+ return out[0]
101
+ return out
102
+
87
103
  def get_image(self, index: slice = None) -> np.ndarray:
88
104
  """
89
105
  Get image data from file, using default image path
@@ -102,6 +118,11 @@ class HdfLoader:
102
118
  with self._load() as hdf:
103
119
  return self.map.get_scannables(hdf)
104
120
 
121
+ def summary(self) -> str:
122
+ """Return string summary of datasets"""
123
+ with self._load() as hdf:
124
+ return self.map.create_dataset_summary(hdf)
125
+
105
126
  def eval(self, expression: str):
106
127
  """
107
128
  Evaluate an expression using the namespace of the hdf file
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: hdfmap
3
- Version: 0.4
3
+ Version: 0.5
4
4
  Summary: Map objects within a HDF file and create a dataset namespace
5
5
  Author-email: Dan Porter <dan.porter@diamond.ac.uk>
6
6
  Maintainer-email: Dan Porter <dan.porter@diamond.ac.uk>
@@ -206,11 +206,11 @@ License: Apache License
206
206
  See the License for the specific language governing permissions and
207
207
  limitations under the License.
208
208
 
209
- Project-URL: Homepage, https://github.com/DanPorter/hdfmap
210
- Project-URL: Documentation, https://github.com/DanPorter/hdfmap
211
- Project-URL: Repository, https://github.com/DanPorter/hdfmap
212
- Project-URL: Bug Tracker, https://github.com/DanPorter/hdfmap
213
- Project-URL: Changelog, https://github.com/DanPorter/hdfmap/blob/master/README.md
209
+ Project-URL: Homepage, https://github.com/DiamondLightSource/hdfmap
210
+ Project-URL: Documentation, https://diamondlightsource.github.io/hdfmap/
211
+ Project-URL: Repository, https://github.com/DiamondLightSource/hdfmap
212
+ Project-URL: Bug Tracker, https://github.com/DiamondLightSource/hdfmap
213
+ Project-URL: Changelog, https://github.com/DiamondLightSource/hdfmap/blob/master/README.md
214
214
  Keywords: nexus
215
215
  Classifier: Programming Language :: Python :: 3.10
216
216
  Classifier: Intended Audience :: Science/Research
@@ -231,7 +231,7 @@ Map objects within an HDF file and create a dataset namespace.
231
231
  [![License](https://img.shields.io/badge/License-Apache%202.0-blue.svg)](https://opensource.org/licenses/Apache-2.0)
232
232
  [![](https://img.shields.io/github/forks/DiamondLightSource/hdfmap?label=GitHub%20Repo&style=social)](https://github.com/DiamondLightSource/hdfmap)
233
233
 
234
- **Version 0.4**
234
+ **Version 0.5**
235
235
 
236
236
  | By Dan Porter |
237
237
  |----------------------|
@@ -349,7 +349,7 @@ map.image_data = {'name': '/hdf/group/dataset'}
349
349
  | `map.get_metadata(h5py.File)` | returns dict of value datasets |
350
350
  | `map.get_scannables(h5py.File)` | returns dict of scannable datasets |
351
351
  | `map.get_scannalbes_array(h5py.File)` | returns numpy array of scannable datasets |
352
- | `map.get_data_block(h5py.File)` | returns dict like object with metadata and scannables |
352
+ | `map.get_dataholder(h5py.File)` | returns dict like object with metadata and scannables |
353
353
  | `map.get_image(h5py.File, index)` | returns image data |
354
354
  | `map.get_data(h5py.File, 'name')` | returns data from dataset |
355
355
  | `map.eval(h5py.File, 'expression')` | returns output of expression using dataset names |
@@ -4,6 +4,7 @@ pyproject.toml
4
4
  src/hdfmap/__init__.py
5
5
  src/hdfmap/eval_functions.py
6
6
  src/hdfmap/file_functions.py
7
+ src/hdfmap/hdf_loader.py
7
8
  src/hdfmap/hdfmap_class.py
8
9
  src/hdfmap/logging.py
9
10
  src/hdfmap/nexus.py
@@ -1,7 +1,9 @@
1
1
  from os import path
2
2
  import json
3
3
  import hdfmap
4
+ import hdfmap.hdf_loader
4
5
 
6
+ from . import only_dls_file_system
5
7
 
6
8
  # Edge case files, create this list from create_test_files.py
7
9
  TEST_FILES = path.join(path.dirname(__file__), 'data', 'test_files.json')
@@ -9,6 +11,7 @@ with open(TEST_FILES, 'r') as f:
9
11
  CHECK_FILES = json.load(f)
10
12
 
11
13
 
14
+ @only_dls_file_system
12
15
  def test_edge_cases():
13
16
  n = 0
14
17
  for chk in CHECK_FILES:
@@ -26,19 +29,23 @@ def test_edge_cases():
26
29
  print(f"Completed {n} edge case files")
27
30
 
28
31
 
32
+ @only_dls_file_system
29
33
  def test_old_i16_file():
30
34
  filename = '/dls/science/groups/das/ExampleData/hdfmap_tests/i16/1040311.nxs'
35
+ assert path.isfile(filename) is True, f"{filename} doesn't exist"
31
36
  mymap = hdfmap.create_nexus_map(filename)
32
- with hdfmap.load_hdf(filename) as hdf:
37
+ with hdfmap.hdf_loader.load_hdf(filename) as hdf:
33
38
  value, address = mymap.eval(hdf, 'np.sum(sum), _sum')
34
39
  assert abs(value + 407) < 0.01, 'expression "np.sum(sum)" gives wrong result'
35
40
  assert address == '/entry1/measurement/sum', 'expression "_sum" returns wrong address'
36
41
 
37
42
 
43
+ @only_dls_file_system
38
44
  def test_new_i16_file():
39
45
  filename = '/dls/science/groups/das/ExampleData/hdfmap_tests/i16/1040323.nxs'
46
+ assert path.isfile(filename) is True, f"{filename} doesn't exist"
40
47
  mymap = hdfmap.create_nexus_map(filename)
41
- with hdfmap.load_hdf(filename) as hdf:
48
+ with hdfmap.hdf_loader.load_hdf(filename) as hdf:
42
49
  h, k, l, hkl, _h, fname = mymap.eval(hdf, 'h, k, l, hkl, _h, filename')
43
50
  assert h.shape == (21,), 'expression "h" has wrong shape'
44
51
  assert hkl == '--', 'default for expression "hkl" is incorrect'
@@ -1,13 +1,14 @@
1
1
  import pytest
2
2
  import os
3
3
  import hdfmap.file_functions as ff
4
+ import hdfmap.hdf_loader
4
5
 
5
6
  DATA_FOLDER = os.path.join(os.path.dirname(__file__), 'data')
6
7
 
7
8
 
8
9
  @pytest.fixture
9
10
  def files():
10
- files = ff.list_files(DATA_FOLDER)
11
+ files = ff.list_files(DATA_FOLDER, extension='.nxs')
11
12
  yield files
12
13
 
13
14
 
@@ -30,7 +31,7 @@ def test_hdf_eval(files):
30
31
  file = files[0]
31
32
  mymap = ff.create_hdf_map(file)
32
33
  expr = "int(total[0] / Transmission)"
33
- with ff.load_hdf(file) as hdf:
34
+ with hdfmap.hdf_loader.load_hdf(file) as hdf:
34
35
  out = mymap.eval(hdf, expr)
35
36
  assert ff.hdf_eval(file, expr) == out, "expression output doesn't match"
36
37
 
@@ -39,7 +40,7 @@ def test_hdf_format(files):
39
40
  file = files[0]
40
41
  mymap = ff.create_hdf_map(file)
41
42
  expr = "energy is {en:.2f} keV"
42
- with ff.load_hdf(file) as hdf:
43
+ with hdfmap.hdf_loader.load_hdf(file) as hdf:
43
44
  out = mymap.format_hdf(hdf, expr)
44
45
  assert ff.hdf_format(file, expr) == out, "expression output doesn't match"
45
46
 
@@ -47,6 +48,6 @@ def test_hdf_format(files):
47
48
  def test_hdf_image(files):
48
49
  file = files[0]
49
50
  mymap = ff.create_hdf_map(file)
50
- with ff.load_hdf(file) as hdf:
51
+ with hdfmap.hdf_loader.load_hdf(file) as hdf:
51
52
  image = mymap.get_image(hdf)
52
53
  assert ff.hdf_image(file).shape == image.shape, "image doesn't match"
@@ -1,6 +1,7 @@
1
1
  import pytest
2
2
  import os
3
3
  import hdfmap
4
+ import hdfmap.hdf_loader
4
5
 
5
6
  DATA_FOLDER = os.path.join(os.path.dirname(__file__), 'data')
6
7
  FILE_HKL = DATA_FOLDER + "/1049598.nxs" # hkl scan, pilatus
@@ -8,7 +9,7 @@ FILE_HKL = DATA_FOLDER + "/1049598.nxs" # hkl scan, pilatus
8
9
 
9
10
  @pytest.fixture
10
11
  def hdf_map():
11
- with hdfmap.load_hdf(FILE_HKL) as hdf:
12
+ with hdfmap.hdf_loader.load_hdf(FILE_HKL) as hdf:
12
13
  hdf_map = hdfmap.HdfMap(hdf)
13
14
  yield hdf_map
14
15
 
@@ -70,22 +71,31 @@ def test_get_group_datasets(hdf_map):
70
71
 
71
72
 
72
73
  def test_get_data(hdf_map):
73
- with hdfmap.load_hdf(FILE_HKL) as hdf:
74
+ with hdfmap.hdf_loader.load_hdf(FILE_HKL) as hdf:
74
75
  en = hdf['/entry1/before_scan/mono/en'][()]
75
76
  h = hdf['/entry1/measurement/h'][()]
76
77
  cmd = hdf['/entry1/scan_command'].asstr()[()]
78
+ scanno = int(hdf['/entry1/entry_identifier'][()])
77
79
  assert hdf_map.get_data(hdf, 'en') == en, "'en' produces wrong result"
78
80
  assert (hdf_map.get_data(hdf, 'h') == h).all(), "'h' produces wrong result"
79
81
  assert hdf_map.get_data(hdf, 'scan_command')[:8] == cmd[:8], "'cmd' produces wrong result"
82
+ assert hdf_map.get_data(hdf, 'entry_identifier') == scanno, "'entry_identifier' gives wrong result"
83
+
84
+
85
+ def test_get_string(hdf_map):
86
+ with hdfmap.hdf_loader.load_hdf(FILE_HKL) as hdf:
87
+ assert hdf_map.get_string(hdf, 'en') == '3.5800002233729673'
88
+ assert hdf_map.get_string(hdf, 'h') == 'float64 (101,)'
89
+ assert hdf_map.get_string(hdf, 'start_time') == "'2024-05-17 14:13:27.025000'"
80
90
 
81
91
 
82
92
  def test_get_image(hdf_map):
83
- with hdfmap.load_hdf(FILE_HKL) as hdf:
93
+ with hdfmap.hdf_loader.load_hdf(FILE_HKL) as hdf:
84
94
  assert hdf_map.get_image(hdf).shape == (195, 487)
85
95
 
86
96
 
87
97
  def test_get_dataholder(hdf_map):
88
- with hdfmap.load_hdf(FILE_HKL) as hdf:
98
+ with hdfmap.hdf_loader.load_hdf(FILE_HKL) as hdf:
89
99
  d = hdf_map.get_dataholder(hdf)
90
100
  assert d.metadata.filepath == FILE_HKL, "Filename not included in data object metadata"
91
101
  assert int(100 * d.metadata.en) == 358, "metadata energy is wrong"
@@ -93,39 +103,49 @@ def test_get_dataholder(hdf_map):
93
103
 
94
104
 
95
105
  def test_get_metadata(hdf_map):
96
- with hdfmap.load_hdf(FILE_HKL) as hdf:
106
+ with hdfmap.hdf_loader.load_hdf(FILE_HKL) as hdf:
97
107
  meta = hdf_map.get_metadata(hdf)
98
108
  meta_small = hdf_map.get_metadata(hdf, name_list=['scan_command', 'incident_energy'])
109
+ meta_string = hdf_map.get_metadata(hdf, string_output=True)
99
110
  assert len(meta) == 423, "Length of metadata wrong"
100
111
  assert meta['filename'] == '1049598.nxs', "filename is wrong"
101
112
  assert abs(meta_small['incident_energy'] - 3.58) < 0.01, "Energy is wrong"
113
+ cmd = "'scan hkl [-0.05, -7.878e-16, 0.933] [0.05, -7.878e-16, 0.933] [0.001, 0, 0] BeamOK pil3_100k 1 roi2 roi1'"
114
+ assert meta_string['scan_command'] == cmd
115
+ assert meta_string['ppchi'] == '-44.999994057'
116
+
117
+
118
+ def test_create_metadata_list(hdf_map):
119
+ with hdfmap.hdf_loader.load_hdf(FILE_HKL) as hdf:
120
+ meta = hdf_map.create_metadata_list(hdf)
121
+ assert len(meta) == 11391, "Length of metadata list wrong"
102
122
 
103
123
 
104
124
  def test_get_scannables(hdf_map):
105
- with hdfmap.load_hdf(FILE_HKL) as hdf:
125
+ with hdfmap.hdf_loader.load_hdf(FILE_HKL) as hdf:
106
126
  scannables = hdf_map.get_scannables(hdf)
107
127
  assert len(scannables) == 131, "Length of scannables is wrong"
108
128
 
109
129
 
110
130
  def test_get_scannables_array(hdf_map):
111
- with hdfmap.load_hdf(FILE_HKL) as hdf:
131
+ with hdfmap.hdf_loader.load_hdf(FILE_HKL) as hdf:
112
132
  scannables = hdf_map.get_scannables_array(hdf)
113
133
  assert scannables.shape == (129, 101), "scannables array is wrong shape"
114
134
 
115
135
 
116
136
  def test_create_scannables_table(hdf_map):
117
- with hdfmap.load_hdf(FILE_HKL) as hdf:
137
+ with hdfmap.hdf_loader.load_hdf(FILE_HKL) as hdf:
118
138
  scannables = hdf_map.create_scannables_table(hdf, '\t')
119
139
  assert len(scannables) == 165703, "scannables str is wrong length"
120
140
 
121
141
 
122
142
  def test_eval(hdf_map):
123
- with hdfmap.load_hdf(FILE_HKL) as hdf:
143
+ with hdfmap.hdf_loader.load_hdf(FILE_HKL) as hdf:
124
144
  out = hdf_map.eval(hdf, 'int(np.max(sum / Transmission / count_time))')
125
145
  assert out == 6533183, "Expression output gives wrong result"
126
146
 
127
147
 
128
148
  def test_format_hdf(hdf_map):
129
- with hdfmap.load_hdf(FILE_HKL) as hdf:
149
+ with hdfmap.hdf_loader.load_hdf(FILE_HKL) as hdf:
130
150
  out = hdf_map.format_hdf(hdf, 'The energy is {en:.3} keV')
131
151
  assert out == 'The energy is 3.58 keV', "Expression output gives wrong result"
@@ -1,7 +1,9 @@
1
1
  import os
2
2
  from time import perf_counter
3
3
  import hdfmap
4
+ import hdfmap.hdf_loader
4
5
 
6
+ from . import only_dls_file_system
5
7
 
6
8
  # Folder with over 1000 files
7
9
  THOUSAND_FILES = '/dls/science/groups/das/ExampleData/hdfmap_tests/i16/cm37262-1'
@@ -10,13 +12,15 @@ NFILES = 1332 # number of files to test (max 1332)
10
12
  FORMAT_STRING = '#{entry_identifier}: {start_time} : E={incident_energy:.3f} keV : {scan_command}'
11
13
 
12
14
 
15
+ @only_dls_file_system
13
16
  def test_compare_time_for_many_files():
14
17
  files = hdfmap.list_files(THOUSAND_FILES)[:NFILES]
18
+ assert len(files) > 1, "Files not found"
15
19
  # time to read single entry from each files
16
20
  start = perf_counter()
17
21
  output1 = []
18
22
  for file in files:
19
- with hdfmap.load_hdf(file) as hdf:
23
+ with hdfmap.hdf_loader.load_hdf(file) as hdf:
20
24
  output1.append((
21
25
  hdf['/entry1/scan_command'][()],
22
26
  hdf['/entry1/entry_identifier'][()],
@@ -36,7 +40,7 @@ def test_compare_time_for_many_files():
36
40
  start = perf_counter()
37
41
  output1 = []
38
42
  for file in files:
39
- with hdfmap.load_hdf(file) as hdf:
43
+ with hdfmap.hdf_loader.load_hdf(file) as hdf:
40
44
  output1.append((
41
45
  hdf['/entry1/scan_command'][()],
42
46
  hdf['/entry1/entry_identifier'][()],
@@ -4,11 +4,11 @@ import hdfmap
4
4
 
5
5
  DATA_FOLDER = os.path.join(os.path.dirname(__file__), 'data')
6
6
  FILE_NEW_NEXUS = DATA_FOLDER + '/1040323.nxs' # new nexus format
7
+ FILE_3D_NEXUS = DATA_FOLDER + '/i06-353130.nxs' # new nexus format
7
8
 
8
9
  hdfmap.set_all_logging_level('debug')
9
10
 
10
11
 
11
-
12
12
  @pytest.fixture
13
13
  def hdf_map():
14
14
  hdf_map = hdfmap.NexusMap()
@@ -19,7 +19,7 @@ def hdf_map():
19
19
 
20
20
  def test_populate(hdf_map):
21
21
  assert len(hdf_map.datasets) == 431, "Wrong number of datasets"
22
- assert len(hdf_map.combined) == 634, "Wrong number of names in map.combined"
22
+ assert len(hdf_map.combined) == 633, "Wrong number of names in map.combined"
23
23
  assert hdf_map.scannables_length() == 21, "Wrong length for scannables"
24
24
  assert hdf_map['axes'] == '/entry/measurement/h', "Wrong path for default axes"
25
25
  assert hdf_map.get_image_path() == '/entry/instrument/pil3_100k/data', "Wrong image path"
@@ -32,11 +32,28 @@ def test_dataset_names(hdf_map):
32
32
 
33
33
 
34
34
  def test_nexus_eval(hdf_map):
35
- with hdfmap.load_hdf(FILE_NEW_NEXUS) as hdf:
35
+ with hdfmap.hdf_loader.load_hdf(FILE_NEW_NEXUS) as hdf:
36
36
  out = hdf_map.eval(hdf, 'int(np.max(total / Transmission / count_time))')
37
37
  assert out == 70, "Expression output gives wrong result"
38
38
  path = hdf_map.eval(hdf, '_axes')
39
39
  assert path == '/entry/measurement/h', "Wrong axes path"
40
40
  title = hdf_map.format_hdf(hdf, '{filename}: {scan_command}')
41
41
  correct = '1040323.nxs: scan hkl [0.97, 0.022, 0.112] [0.97, 0.022, 0.132] [0, 0, 0.001] MapperProc pil3_100k 1'
42
- assert title == correct, "Expression output gives wrong result"
42
+ assert title == correct, "Expression output gives wrong result"
43
+
44
+
45
+ def test_3d_scan(hdf_map):
46
+ hdf_map = hdfmap.create_nexus_map(FILE_3D_NEXUS)
47
+ assert hdf_map.scannables_length() == 80, "Scannables have the wrong length"
48
+ axes, signal = hdf_map.nexus_defaults()
49
+ assert len(axes) == 3, "Number of default axes is wrong"
50
+ assert signal == '/entry/medipix/data', "Incorrect default signal"
51
+ with hdf_map.load_hdf() as hdf:
52
+ table = hdf_map.create_scannables_table(hdf)
53
+ assert table.count('\n') == 80, "table has the wrong length"
54
+ assert len(table) == 4085, "wrong number of characters in table"
55
+
56
+ image = hdf_map.get_image(hdf, index=None)
57
+ assert image.shape == (512, 512), "image shape is wrong"
58
+
59
+
File without changes
File without changes