pytesprocess 0.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. pytesprocess/__init__.py +9 -0
  2. pytesprocess/_version.py +2 -0
  3. pytesprocess/cli/__init__.py +1 -0
  4. pytesprocess/cli/commands/__init__.py +5 -0
  5. pytesprocess/cli/commands/event.py +66 -0
  6. pytesprocess/cli/commands/filter.py +17 -0
  7. pytesprocess/cli/commands/ivsweep.py +29 -0
  8. pytesprocess/cli/common.py +86 -0
  9. pytesprocess/cli/main.py +81 -0
  10. pytesprocess/config/__init__.py +4 -0
  11. pytesprocess/config/loader.py +94 -0
  12. pytesprocess/config/manager.py +297 -0
  13. pytesprocess/config/resolvers/__init__.py +5 -0
  14. pytesprocess/config/resolvers/common.py +56 -0
  15. pytesprocess/config/resolvers/feature.py +293 -0
  16. pytesprocess/config/resolvers/salting.py +86 -0
  17. pytesprocess/config/resolvers/trigger.py +84 -0
  18. pytesprocess/config/selectors.py +108 -0
  19. pytesprocess/config/validation.py +314 -0
  20. pytesprocess/config/warnings.py +2 -0
  21. pytesprocess/core/__init__.py +10 -0
  22. pytesprocess/core/algorithms.py +1455 -0
  23. pytesprocess/core/didv.py +1648 -0
  24. pytesprocess/core/eventbuilder.py +495 -0
  25. pytesprocess/core/filterbuilder.py +81 -0
  26. pytesprocess/core/filterdata.py +1849 -0
  27. pytesprocess/core/ivsweep.py +2072 -0
  28. pytesprocess/core/noise.py +923 -0
  29. pytesprocess/core/noisemodel.py +1408 -0
  30. pytesprocess/core/oftrigger.py +1035 -0
  31. pytesprocess/core/template.py +450 -0
  32. pytesprocess/process/__init__.py +6 -0
  33. pytesprocess/process/data_source.py +185 -0
  34. pytesprocess/process/event_context.py +35 -0
  35. pytesprocess/process/feature_plan.py +186 -0
  36. pytesprocess/process/feature_resources.py +267 -0
  37. pytesprocess/process/features.py +1024 -0
  38. pytesprocess/process/filterprocess.py +1176 -0
  39. pytesprocess/process/ivprocess.py +1380 -0
  40. pytesprocess/process/processing_data.py +967 -0
  41. pytesprocess/process/randoms.py +921 -0
  42. pytesprocess/process/triggers.py +1011 -0
  43. pytesprocess/salting/__init__.py +7 -0
  44. pytesprocess/salting/generator.py +364 -0
  45. pytesprocess/salting/injector.py +329 -0
  46. pytesprocess/salting/sampling.py +84 -0
  47. pytesprocess/utils/__init__.py +5 -0
  48. pytesprocess/utils/arg_utils.py +122 -0
  49. pytesprocess/utils/dataframe_output.py +120 -0
  50. pytesprocess/utils/filter_hdf5.py +594 -0
  51. pytesprocess/utils/utils.py +701 -0
  52. pytesprocess/workflows/__init__.py +3 -0
  53. pytesprocess/workflows/processing.py +317 -0
  54. pytesprocess/workflows/salting.py +133 -0
  55. pytesprocess-0.1.1.dist-info/METADATA +211 -0
  56. pytesprocess-0.1.1.dist-info/RECORD +60 -0
  57. pytesprocess-0.1.1.dist-info/WHEEL +5 -0
  58. pytesprocess-0.1.1.dist-info/entry_points.txt +2 -0
  59. pytesprocess-0.1.1.dist-info/licenses/LICENSE +21 -0
  60. pytesprocess-0.1.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,923 @@
1
+ import copy
2
+ from glob import glob
3
+ from pathlib import Path
4
+ import os
5
+
6
+ import numpy as np
7
+ import qetpy as qp
8
+ import vaex as vx
9
+ from pytesdaqx.io import AcquisitionCatalog, StreamReader
10
+
11
+ from pytesprocess.core.filterdata import FilterData
12
+ from pytesprocess.process.randoms import Randoms
13
+ from pytesprocess.utils import (
14
+ convert_length_msec_to_samples,
15
+ extract_stream_id,
16
+ extract_stream_num,
17
+ utils,
18
+ )
19
+
20
+
21
+ class Noise(FilterData):
22
+ """
23
+ Class to manage noise calculation from
24
+ randoms for multiple channels
25
+ """
26
+
27
+ def __init__(self, verbose=True, filter_data=None):
28
+ super().__init__(verbose=verbose, filter_data=filter_data)
29
+
30
+ self._raw_data_files = None
31
+ self._acquisition_name = None
32
+ self._raw_base_path = None
33
+ self._stream_list = None
34
+ self._detector_config = None
35
+ self._dataframe = None
36
+ self._record_list = None
37
+ self._fs = None
38
+ self._offset = dict()
39
+ self._offset_err = dict()
40
+ self._available_channels = None
41
+ self._catalog = None
42
+
43
+
44
+ def get_detector_config(self, channel):
45
+ """
46
+ get detector config
47
+ """
48
+ if self._detector_config is None:
49
+ print('WARNING: No data has been set yet! '
50
+ 'Returning None ')
51
+ return None
52
+ elif channel not in self._detector_config.keys():
53
+ print(f'WARNING: No channel {channel} found! '
54
+ f'Returning None ')
55
+ return None
56
+ return self._detector_config[channel]
57
+
58
+ def get_sample_rate(self):
59
+ """
60
+ Get sample rate in Hz ("calc_psd" needs to be
61
+ called before)
62
+ """
63
+
64
+ return self._fs
65
+
66
+
67
+ def get_offset(self, channel):
68
+ """
69
+ Get offset, return None if no offset
70
+ available
71
+ """
72
+ offset = None
73
+ if channel in self._offset:
74
+ offset = self._offset[channel]
75
+ else:
76
+ print(f'WARNING: No offset available for channel '
77
+ f'{channel}. You need to calculate psd first! '
78
+ f'Returning None. ')
79
+
80
+ return offset
81
+
82
+ def get_offset_error(self, channel):
83
+ """Get the statistical uncertainty on the stored baseline offset."""
84
+ if channel in self._offset_err:
85
+ return self._offset_err[channel]
86
+ if self._verbose:
87
+ print(
88
+ f"WARNING: No offset uncertainty available for channel {channel}. "
89
+ "Calculate an offset or PSD first. Returning None."
90
+ )
91
+ return None
92
+
93
+ def clear_randoms(self):
94
+ """
95
+ Clear internal data, however
96
+ keep self._filter_data
97
+ """
98
+
99
+ # clear data
100
+ self._dataframe = None
101
+ self._record_list = None
102
+ self._raw_data_files = None
103
+ self._acquisition_name = None
104
+ self._raw_base_path = None
105
+ self._stream_list = None
106
+ self._detector_config = None
107
+ self._fs = None
108
+ self._offset = dict()
109
+ self._offset_err = dict()
110
+ self._available_channels = None
111
+ self._catalog = None
112
+
113
+ def set_randoms(self, data_path,
114
+ streams=None,
115
+ dataframe=None,
116
+ record_list=None,
117
+ restricted=False,
118
+ data_type=None):
119
+ """
120
+ Set randoms events
121
+ - Select randoms using datafame (if not None) or record_list
122
+ - OR use full trace (if datafame is None)
123
+ """
124
+
125
+ # initialize data
126
+ self.clear_randoms()
127
+
128
+ # check arguments
129
+ if (dataframe is not None
130
+ and record_list is not None):
131
+ raise ValueError('ERROR: choose between "dataframe" and '
132
+ '"record_list", not both')
133
+
134
+ # data catalog
135
+ catalog_full = AcquisitionCatalog(data_path, verbose=True)
136
+ self._catalog = catalog_full.filter(
137
+ measurement_types=data_type,
138
+ streams=streams,
139
+ restricted=restricted
140
+ )
141
+
142
+ # get file dictionary
143
+ data_dict = self._catalog.select_files_by_stream(
144
+ stream_key='stream_id'
145
+ )
146
+
147
+ if not data_dict:
148
+ raise ValueError(f'No files with data type "{data_type}" '
149
+ f'were found! Check raw data path...')
150
+
151
+ # storage format
152
+ self._storage_format = self._catalog.storage_format
153
+
154
+
155
+ # save info
156
+ self._raw_data_files = copy.deepcopy(data_dict)
157
+ acquisition_path = self._catalog.acquisition_path
158
+ path_obj = Path(acquisition_path)
159
+ self._acquisition_name = str(path_obj.name)
160
+ self._raw_base_path = str(path_obj.parent)
161
+ self._stream_list = list(self._raw_data_files.keys())
162
+ self._detector_config = dict()
163
+ self._available_channels = self._catalog.entries[0]['record_channels']
164
+
165
+
166
+ # check dataframe
167
+ if dataframe is not None:
168
+
169
+ if isinstance(dataframe, vx.dataframe.DataFrame):
170
+ if len(dataframe)<1:
171
+ raise ValueError('ERROR: No event found in the datafame!')
172
+ elif isinstance(dataframe, str):
173
+ dataframe = self._load_dataframe(dataframe)
174
+
175
+ self._dataframe = dataframe
176
+
177
+ elif record_list is not None:
178
+ self._record_list = record_list
179
+
180
+ # check filter data
181
+ if self._filter_data:
182
+ print('WARNING: Some noise data have been previously saved. '
183
+ 'Use "describe()" to check. If needed clear data '
184
+ 'using "clear_data(channels=None, tag=None)" function!')
185
+
186
+
187
+ def generate_randoms(self, data_path,
188
+ streams=None,
189
+ random_rate=None, nrandoms=None,
190
+ min_separation_msec=100,
191
+ edge_exclusion_msec=50,
192
+ partition_target_duration_s=None,
193
+ random_seed=None,
194
+ restricted=False,
195
+ data_type='background'):
196
+ """
197
+ Generate randoms from continuous data
198
+ """
199
+ # initialize data
200
+ self.clear_randoms()
201
+
202
+ # data catalog
203
+ catalog_full = AcquisitionCatalog(data_path, verbose=True)
204
+ self._catalog = catalog_full.filter(
205
+ measurement_types=data_type,
206
+ streams=streams,
207
+ restricted=restricted
208
+ )
209
+
210
+ # get file dictionary
211
+ data_dict = self._catalog.select_files_by_stream(
212
+ stream_key='stream_id'
213
+ )
214
+
215
+ if not data_dict:
216
+ raise ValueError(f'No files with data type "{data_type}" '
217
+ f'were found! Check raw data path...')
218
+
219
+ # raw data format
220
+ self._storage_format = self._catalog.storage_format
221
+
222
+ # save info
223
+ self._raw_data_files = copy.deepcopy(data_dict)
224
+ acquisition_path = self._catalog.acquisition_path
225
+ path_obj = Path(acquisition_path)
226
+ self._acquisition_name = str(path_obj.name)
227
+ self._raw_base_path = str(path_obj.parent)
228
+ self._stream_list = list(self._raw_data_files.keys())
229
+ self._detector_config = dict()
230
+ self._available_channels = self._catalog.entries[0]['record_channels']
231
+
232
+ # generate randoms
233
+ rand_inst = Randoms(data_path, streams=streams,
234
+ verbose=self._verbose,
235
+ restricted=restricted,
236
+ data_type=data_type)
237
+
238
+
239
+ self._dataframe = rand_inst.process(
240
+ random_rate=random_rate,
241
+ nrandoms=nrandoms,
242
+ min_separation_msec=min_separation_msec,
243
+ edge_exclusion_msec=edge_exclusion_msec,
244
+ partition_target_duration_s=partition_target_duration_s,
245
+ random_seed=random_seed,
246
+ lgc_save=False,
247
+ lgc_output=True
248
+ )
249
+
250
+ # check filter data
251
+ if self._filter_data:
252
+ print('WARNING: Some noise data have been previously saved. '
253
+ 'Use "describe()" to check. If needed clear data '
254
+ 'using "clear_data(channels=None, tag=None)" function!')
255
+
256
+
257
+ def calc_offset(self, channels=None,
258
+ stream=None,
259
+ trace_length_msec=None,
260
+ trace_length_samples=None,
261
+ pretrigger_length_msec=None,
262
+ pretrigger_length_samples=None,
263
+ nrandoms=None):
264
+ """Calculate baseline offsets from the currently selected randoms.
265
+
266
+ The same random-event selection used by :meth:`calc_psd` is used here,
267
+ but no spectrum is calculated. This is useful for per-cycle operating
268
+ point calibration where the background baseline is needed before the
269
+ dIdV/noise processing chain.
270
+ """
271
+ if nrandoms is None:
272
+ raise ValueError('ERROR: Maximum number of randoms required! Add "nrandoms" argument.')
273
+ if self._raw_data_files is None:
274
+ raise ValueError('ERROR: No raw data available. Use set_randoms() or generate_randoms() first!')
275
+
276
+ if channels is None:
277
+ channels = self._available_channels
278
+ if isinstance(channels, str):
279
+ channels = [channels]
280
+
281
+ results = {}
282
+ for channel in channels:
283
+ channel_list, separator = utils.split_channel_name(
284
+ channel, self._available_channels
285
+ )
286
+ if separator is not None or len(channel_list) != 1:
287
+ raise ValueError(
288
+ 'ERROR: calc_offset() requires individual channels; '
289
+ f'got {channel!r}.'
290
+ )
291
+
292
+ traces, metadata = self._get_traces(
293
+ channel_list,
294
+ nrandoms=nrandoms,
295
+ trace_length_msec=trace_length_msec,
296
+ trace_length_samples=trace_length_samples,
297
+ pretrigger_length_msec=pretrigger_length_msec,
298
+ pretrigger_length_samples=pretrigger_length_samples,
299
+ stream=stream,
300
+ )
301
+ fs = float(metadata['sample_rate_hz'])
302
+ traces = traces[:, 0, :] if traces.ndim == 3 else traces
303
+ cut = qp.autocuts_noise(traces, fs=fs)
304
+ if np.sum(cut) == 0:
305
+ raise ValueError(
306
+ f'ERROR: No events selected after noise autocut for channel {channel}!'
307
+ )
308
+
309
+ offset, offset_err = qp.utils.calc_offset(traces[cut], fs=fs)
310
+ self._fs = fs
311
+ self._offset[channel] = offset
312
+ self._offset_err[channel] = offset_err
313
+ results[channel] = {
314
+ 'offset': offset,
315
+ 'offset_err': offset_err,
316
+ 'cut_efficiency': float(np.sum(cut)) / len(cut) * 100.0,
317
+ 'sample_rate_hz': fs,
318
+ }
319
+
320
+ return results
321
+
322
+
323
+ def calc_psd(self, channels=None,
324
+ stream=None,
325
+ trace_length_msec=None,
326
+ trace_length_samples=None,
327
+ pretrigger_length_msec=None,
328
+ pretrigger_length_samples=None,
329
+ nrandoms=None,
330
+ weights=None,
331
+ tag='default'):
332
+ """
333
+ Calculate two-sided and folded-over PSD in Amps^2/Hz
334
+ A specific stream can be specified
335
+
336
+ """
337
+
338
+ # --------------------------------
339
+ # Check arguments
340
+ # --------------------------------
341
+
342
+ if channels is None:
343
+ channels = self._available_channels
344
+ if isinstance(channels, str):
345
+ channels = [channels]
346
+
347
+ if nrandoms is None:
348
+ raise ValueError('ERROR: Maximum number of randoms required!'
349
+ ' Add "nrandoms" argument.')
350
+
351
+ # check raw data has been loaded
352
+ if self._raw_data_files is None:
353
+ raise ValueError('ERROR: No raw data available. Use '
354
+ + '"set_randoms()" or "generate_randoms() '
355
+ + 'function first!')
356
+
357
+ # check trace length
358
+ if (trace_length_msec is not None
359
+ and trace_length_samples is not None):
360
+ raise ValueError('ERROR: Trace length need to be '
361
+ 'in msec OR samples, nto both')
362
+
363
+ if (pretrigger_length_msec is not None
364
+ and pretrigger_length_samples is not None):
365
+ raise ValueError('ERROR: Pretrigger length need to be '
366
+ 'in msec OR samples, nto both')
367
+
368
+
369
+ # --------------------------------
370
+ # Loop channels and calculate PSD
371
+ # --------------------------------
372
+ for chan in channels:
373
+
374
+ if self._verbose:
375
+ if stream is None:
376
+ print('INFO: Processing PSD for channel '
377
+ + chan)
378
+ else:
379
+ print('INFO: Processing PSD for channel '
380
+ + chan + ' using stream '
381
+ + str(stream))
382
+
383
+
384
+ # let's check if sum of pulses
385
+ separator = None
386
+ chan_list, separator = utils.split_channel_name(
387
+ chan, self._available_channels
388
+ )
389
+
390
+
391
+ # weights
392
+ weights_array = None
393
+ if weights is not None:
394
+
395
+ weights_array = np.ones(len(chan_list))
396
+
397
+ for ichan, chan_split in enumerate(chan_list):
398
+ if chan_split in weights:
399
+ weights_array[ichan] = weights[chan_split]
400
+
401
+ # let's do first overall all PSD
402
+ traces, traces_metadata = self._get_traces(
403
+ chan_list,
404
+ nrandoms=nrandoms,
405
+ trace_length_msec=trace_length_msec,
406
+ trace_length_samples=trace_length_samples,
407
+ pretrigger_length_msec=pretrigger_length_msec,
408
+ pretrigger_length_samples=pretrigger_length_samples,
409
+ stream=stream
410
+ )
411
+
412
+ self._fs = traces_metadata['sample_rate_hz']
413
+
414
+ if separator == '+':
415
+ if weights_array is not None:
416
+ weights_array = weights_array[np.newaxis, :, np.newaxis]
417
+ traces = traces * weights_array
418
+ traces = np.sum(traces, axis=1)
419
+
420
+ elif separator == '-':
421
+ if weights_array is not None:
422
+ traces = (traces[:,0,:]*weights_array[0]
423
+ - traces[:,1,:]*weights_array[1])
424
+ else:
425
+ traces = traces[:,0,:] - traces[:,1,:]
426
+
427
+ elif separator is not None:
428
+ raise ValueError('ERROR: PSD can only be calculated '
429
+ 'on single channels!')
430
+
431
+ if traces.ndim==3:
432
+ if traces.shape[1] != 1:
433
+ raise ValueError('ERROR: Multiple channels. Expecting '
434
+ 'only one. Something went wrong!')
435
+ traces = traces[:,0,:]
436
+
437
+ # autocut_noise
438
+ cut = qp.autocuts_noise(traces, fs=self._fs)
439
+
440
+ if np.sum(cut)==0:
441
+ raise ValueError('ERROR: No events selected after noise autocut! '
442
+ + 'Unable to calculate PSD')
443
+
444
+ cut_eff = np.sum(cut)/len(cut)*100
445
+ if self._verbose:
446
+ print('INFO: Number of events after cuts = '
447
+ '{}, efficiency = '
448
+ '{:0.2f}%'.format(np.sum(cut), cut_eff))
449
+
450
+ # calc PSD
451
+ freqs, psd = qp.calc_psd(traces[cut],
452
+ fs=self._fs,
453
+ folded_over=False)
454
+
455
+ # calc baseline offset and uncertainty
456
+ offset, offset_err = qp.utils.calc_offset(traces[cut], fs=self._fs)
457
+ self._offset[chan] = offset
458
+ self._offset_err[chan] = offset_err
459
+
460
+ # metadata
461
+ traces_metadata['cut_efficiency'] = cut_eff
462
+
463
+ if weights is not None:
464
+ for wchan,wval in weights.items():
465
+ param = f'weights_{wchan}'
466
+ traces_metadata[param] = wval
467
+
468
+ self.set_psd(
469
+ chan,
470
+ psd,
471
+ freqs,
472
+ sample_rate=self._fs,
473
+ pretrigger_length_samples=traces_metadata.get(
474
+ 'nb_pretrigger_samples'
475
+ ),
476
+ metadata=traces_metadata,
477
+ tag=tag,
478
+ )
479
+
480
+
481
+
482
+ def calc_csd(self, channels,
483
+ stream=None,
484
+ trace_length_msec=None,
485
+ trace_length_samples=None,
486
+ pretrigger_length_msec=None,
487
+ pretrigger_length_samples=None,
488
+ nrandoms=None,
489
+ tag='default',
490
+ use_hann_window=False):
491
+ """
492
+ Calculate two-sided and folded CSD.
493
+
494
+ """
495
+
496
+ # --------------------------------
497
+ # Check arguments
498
+ # --------------------------------
499
+
500
+ if nrandoms is None:
501
+ raise ValueError('ERROR: Maximum number of randoms required!'
502
+ ' Add "nrandoms" argument.')
503
+
504
+ # check raw data has been loaded
505
+ if self._raw_data_files is None:
506
+ raise ValueError('ERROR: No raw data available. Use '
507
+ + '"set_randoms()" function first!')
508
+
509
+ # check trace length
510
+ if (trace_length_msec is not None
511
+ and trace_length_samples is not None):
512
+ raise ValueError('ERROR: Trace length need to be '
513
+ 'in msec OR samples, nto both')
514
+
515
+ if (pretrigger_length_msec is not None
516
+ and pretrigger_length_samples is not None):
517
+ raise ValueError('ERROR: Pretrigger length need to be '
518
+ 'in msec OR samples, nto both')
519
+
520
+
521
+ # channels
522
+ if isinstance(channels, str):
523
+
524
+ if '|' in channels:
525
+ channels = channels.replace(' ','')
526
+ channels = channels.split('|')
527
+ else:
528
+ raise ValueError(
529
+ 'ERROR: At least 2 channels required to calculate csd'
530
+ )
531
+
532
+ # -------------
533
+ # Get data and
534
+ # apply cuts
535
+ # -------------
536
+
537
+ traces, traces_metadata = self._get_traces(
538
+ channels,
539
+ nrandoms=nrandoms,
540
+ trace_length_msec=trace_length_msec,
541
+ trace_length_samples=trace_length_samples,
542
+ pretrigger_length_msec=pretrigger_length_msec,
543
+ pretrigger_length_samples=pretrigger_length_samples,
544
+ stream=stream
545
+ )
546
+
547
+ # check shape
548
+ if traces.shape[1] != len(channels):
549
+ raise ValueError('ERROR: No all channels found in raw data. ')
550
+
551
+ self._fs = traces_metadata['sample_rate_hz']
552
+
553
+ # apply pileup cut
554
+ cut = np.ones(traces.shape[0], dtype=bool)
555
+ for ichan in range(len(channels)):
556
+
557
+ traces_chan = traces[:, ichan,:]
558
+
559
+ # autocut_noise
560
+ cut_chan = qp.autocuts_noise(traces_chan, fs=self._fs)
561
+
562
+ if np.sum(cut_chan) == 0:
563
+ raise ValueError(f'ERROR: No events selected after pileup autocut '
564
+ f'for channel {channels[ichan]} ')
565
+ cut &= cut_chan
566
+
567
+ # check efficiency total cut
568
+ if np.sum(cut) == 0:
569
+ raise ValueError(f'ERROR: No events selected after pileup cut!')
570
+
571
+ cut_eff = np.sum(cut)/len(cut)*100
572
+ if self._verbose:
573
+ print('INFO: Number of events after cuts = '
574
+ '{}, efficiency = '
575
+ '{:0.2f}%'.format(np.sum(cut), cut_eff))
576
+
577
+ # calc CSD two-sided
578
+ freqs, csd = qp.calc_csd(traces[cut],
579
+ fs=self._fs,
580
+ folded_over=False,
581
+ use_hann_window=use_hann_window)
582
+ # metadata
583
+ traces_metadata['cut_efficiency'] = cut_eff
584
+
585
+ self.set_csd(
586
+ channels,
587
+ csd,
588
+ freqs,
589
+ sample_rate=self._fs,
590
+ pretrigger_length_samples=traces_metadata.get(
591
+ 'nb_pretrigger_samples'
592
+ ),
593
+ metadata=traces_metadata,
594
+ tag=tag,
595
+ )
596
+
597
+
598
+ def _get_traces(self, channels,
599
+ stream=None,
600
+ trace_length_msec=None,
601
+ trace_length_samples=None,
602
+ pretrigger_length_msec=None,
603
+ pretrigger_length_samples=None,
604
+ nrandoms=5000):
605
+ """
606
+ Get raw data traces
607
+ """
608
+
609
+ # channels
610
+ if isinstance(channels, str):
611
+ channels = [channels]
612
+ nb_channels = len(channels)
613
+
614
+ # filter catalog with stream
615
+ catalog = self._catalog
616
+
617
+ # filter based on stream
618
+ first_stream = self._stream_list[0]
619
+ stream_num = None
620
+ stream_id = None
621
+ if stream is not None:
622
+ stream_num = extract_stream_num(stream)
623
+ stream_id = extract_stream_id(stream)
624
+
625
+ # check if stream in list
626
+ if stream_id not in self._raw_data_files.keys():
627
+ raise ValueError(
628
+ f'ERROR: stream {stream} not found! '
629
+ f'Check raw data path input.')
630
+
631
+ catalog = catalog.filter(streams=stream_id)
632
+
633
+ # sample rate
634
+ fs = catalog.sample_rate_hz
635
+
636
+ # trace length
637
+ nb_samples = None
638
+ if trace_length_samples is not None:
639
+ nb_samples = trace_length_samples
640
+ elif trace_length_msec is not None:
641
+ nb_samples = convert_length_msec_to_samples(
642
+ trace_length_msec, fs)
643
+ else:
644
+ if self._dataframe is not None:
645
+ raise ValueError('ERROR: number of samples required!')
646
+
647
+ # pretrigger length
648
+ nb_pretrigger_samples = None
649
+ if pretrigger_length_samples is not None:
650
+ nb_pretrigger_samples = pretrigger_length_samples
651
+ elif pretrigger_length_msec is not None:
652
+ nb_pretrigger_samples = convert_length_msec_to_samples(
653
+ pretrigger_length_msec, fs)
654
+
655
+ if (nb_samples is not None
656
+ and nb_pretrigger_samples is None):
657
+ nb_pretrigger_samples = nb_samples//2
658
+
659
+
660
+ # get (filtered) event list
661
+ record_list = None
662
+ stream_ids = None
663
+ if (self._record_list is not None
664
+ or self._dataframe is not None):
665
+
666
+ record_list = self._get_filtered_record_list(
667
+ stream_num=stream_num,
668
+ nrandoms=nrandoms,
669
+ nb_samples=nb_samples,
670
+ nb_pretrigger_samples=nb_pretrigger_samples)
671
+
672
+ if len(record_list) == 0:
673
+ raise ValueError('ERROR: No events selected! Something '
674
+ + 'went wrong!')
675
+
676
+ if len(record_list)>nrandoms:
677
+ record_list = record_list[:nrandoms]
678
+
679
+ nrandoms = len(record_list)
680
+
681
+ else:
682
+ # case raw data directly -> get stream list
683
+
684
+ # list of stream
685
+ stream_ids = list()
686
+ if stream_id is not None:
687
+ stream_ids.append(stream_id)
688
+ else:
689
+ for stream_id in self._raw_data_files.keys():
690
+ stream_ids.append(stream_id)
691
+
692
+ # instantiate StreamReader
693
+ raw_path = self._raw_base_path + '/' + self._acquisition_name
694
+ reader = StreamReader(
695
+ raw_path,
696
+ streams=stream_ids,
697
+ )
698
+
699
+
700
+ # detector config
701
+ self._detector_config = reader.get_detector_settings()
702
+
703
+
704
+ trace_array = reader.read_records(
705
+ record_list=record_list,
706
+ n_records=nrandoms,
707
+ channels=channels,
708
+ trace_length_samples=nb_samples,
709
+ pretrigger_length_samples=nb_pretrigger_samples,
710
+ include_metadata=False,
711
+ units='amps',
712
+ stack=True)
713
+
714
+
715
+ if self._verbose:
716
+ chan_string = str(channels)
717
+ if len(channels)==1:
718
+ chan_string = channels[0]
719
+ print('INFO: ' + str(trace_array.shape[0])
720
+ + ' events found in raw data'
721
+ + ' for channel(s) '
722
+ + str(chan_string))
723
+
724
+ # metadata
725
+ if nb_samples is None: # full trace
726
+ nb_samples = trace_array.shape[-1]
727
+ nb_pretrigger_samples = nb_samples//2
728
+
729
+ trace_metadata = dict()
730
+ trace_metadata['sample_rate_hz'] = fs
731
+ trace_metadata['nb_samples'] = nb_samples
732
+ trace_metadata['nb_pretrigger_samples'] = nb_pretrigger_samples
733
+ trace_metadata['nb_randoms'] = trace_array.shape[0]
734
+ trace_metadata['fridge_run'] = catalog.summary()['fridge_run']
735
+
736
+ return trace_array, trace_metadata
737
+
738
+
739
+ def _get_filtered_record_list(self,
740
+ stream_num=None,
741
+ nrandoms=None,
742
+ nb_samples=None,
743
+ nb_pretrigger_samples=None):
744
+ """
745
+ Get event list from dataframe or unfiltered
746
+ event list
747
+ """
748
+
749
+ # check if data available
750
+ if (self._record_list is None
751
+ and self._dataframe is None):
752
+ return record_list
753
+
754
+
755
+ # convert dataframe into an event list
756
+ record_list = list()
757
+ if self._dataframe is not None:
758
+
759
+ # case dataframe input
760
+ dataframe = self._dataframe.copy()
761
+
762
+ # only take randoms
763
+ cut = dataframe.trigger_type == 3
764
+
765
+ # check stream number
766
+ if stream_num is not None:
767
+ cut = cut & (dataframe.stream_number == stream_num)
768
+
769
+ # filter
770
+ dataframe = dataframe.filter(cut)
771
+
772
+ # loop dataframe ans build event list
773
+ for idx in range(len(dataframe)):
774
+
775
+ # even record from dataframe
776
+ record = dataframe.to_records(idx)
777
+
778
+ # extract event parameters and stored in
779
+ # dictionary
780
+ stored_record = dict()
781
+ stored_record['acquisition_num'] = str(
782
+ record['acquisition_number'])
783
+
784
+ stored_record['stream_num'] = int(
785
+ record['stream_number'])
786
+
787
+
788
+ if self._storage_format == 'hdf5':
789
+ stored_record['global_segment_num'] = int(
790
+ record['global_segment_number']
791
+ )
792
+
793
+ if 'segment_trigger_index' in record:
794
+ stored_record['segment_trigger_index'] = int(
795
+ record['segment_trigger_index']
796
+ )
797
+ stored_record['segment_length_samples'] = int(
798
+ record['segment_length_samples']
799
+ )
800
+ else:
801
+ if 'stream_trigger_index' in record:
802
+ stored_record['stream_trigger_index'] = (
803
+ record['stream_trigger_index']
804
+ )
805
+ stored_record['stream_length_samples'] = int(
806
+ record['stream_length_samples']
807
+ )
808
+
809
+ elif ('partition_trigger_index' in record
810
+ and 'partition_start_index' in record):
811
+ stored_record['partition_trigger_index'] = (
812
+ record['partition_trigger_index']
813
+ )
814
+ stored_record['partition_start_index'] = (
815
+ record['partition_start_index']
816
+ )
817
+
818
+ stored_record['partition_length_samples'] = (
819
+ record['partition_length_samples']
820
+ )
821
+
822
+ # append to list
823
+ record_list.append(stored_record)
824
+
825
+ else:
826
+ # loop list of dictionaries and check stream num
827
+ for irecord, record in enumerate(self._record_list):
828
+
829
+ event_stream_num = record.get('stream_number')
830
+ if event_stream_num is None:
831
+ raise ValueError('ERROR: Wrong input event list format! It should '
832
+ 'contain dictionaries with '
833
+ '"stream_number"!')
834
+
835
+ event_stream_num = int(event_stream_num)
836
+
837
+ # check if stream if resquested
838
+ if (stream_num is not None
839
+ and event_stream_num != stream_num):
840
+ continue
841
+
842
+ record['stream_num'] = event_stream_num
843
+ record_list.append(copy.deepcopy(record))
844
+
845
+
846
+
847
+ # cut edge ot be able to extract trace
848
+ output_record_list = list()
849
+ if nb_samples is not None:
850
+
851
+ for irecord, record in enumerate(record_list):
852
+
853
+ min_index = 0
854
+ max_index = 0
855
+ full_length_samples = None
856
+
857
+ if self._storage_format == 'hdf5':
858
+ if 'segment_trigger_index':
859
+ segment_trigger_index = record['segment_trigger_index']
860
+ full_length_samples = record['segment_length_samples']
861
+ min_index = segment_trigger_index - nb_pretrigger_samples
862
+ max_index = min_index + nb_samples -1
863
+ else:
864
+
865
+ if 'stream_trigger_index' in record:
866
+ stream_trigger_index = record['stream_trigger_index']
867
+ full_length_samples = record['stream_length_samples']
868
+ min_index = stream_trigger_index - nb_pretrigger_samples
869
+ max_index = min_index + nb_samples -1
870
+ elif 'partition_trigger_index' in record:
871
+ partition_trigger_index = record['partition_trigger_index']
872
+ full_length_samples = record['partition_length_samples']
873
+ min_index = partition_trigger_index - nb_pretrigger_samples
874
+ max_index = min_index + nb_samples -1
875
+
876
+ if full_length_samples is not None:
877
+ if (min_index < 0
878
+ or max_index >= full_length_samples):
879
+ continue # skip
880
+
881
+ output_record_list.append(record)
882
+ else:
883
+ output_record_list = record_list
884
+
885
+
886
+ # check number of randoms
887
+ if (nrandoms is not None
888
+ and len(output_record_list)>nrandoms):
889
+ output_record_list = output_record_list[:nrandoms]
890
+
891
+ return output_record_list
892
+
893
+
894
+
895
+ def _load_dataframe(self, dataframe_path):
896
+ """
897
+ Load vaex dataframe (hdf5), path can be a file
898
+ or directory
899
+ """
900
+
901
+ file_list = []
902
+
903
+ if os.path.isdir(dataframe_path):
904
+ file_list = glob(f'{dataframe_path}/*.hdf5')
905
+ if len(file_list) == 0:
906
+ raise ValueError('ERROR: No vaex file found. Check path!')
907
+
908
+ elif os.path.isfile(dataframe_path):
909
+ if dataframe_path.find('.hdf5') != -1:
910
+ raise ValueError(
911
+ f'ERROR: dataframe file {dataframe_path} '
912
+ f'not recognized!'
913
+ )
914
+ file_list = [dataframe_path]
915
+
916
+
917
+ # get dataframe
918
+ dataframe = vx.open_many(file_list)
919
+ return dataframe
920
+
921
+
922
+
923
+