pytesprocess 0.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. pytesprocess/__init__.py +9 -0
  2. pytesprocess/_version.py +2 -0
  3. pytesprocess/cli/__init__.py +1 -0
  4. pytesprocess/cli/commands/__init__.py +5 -0
  5. pytesprocess/cli/commands/event.py +66 -0
  6. pytesprocess/cli/commands/filter.py +17 -0
  7. pytesprocess/cli/commands/ivsweep.py +29 -0
  8. pytesprocess/cli/common.py +86 -0
  9. pytesprocess/cli/main.py +81 -0
  10. pytesprocess/config/__init__.py +4 -0
  11. pytesprocess/config/loader.py +94 -0
  12. pytesprocess/config/manager.py +297 -0
  13. pytesprocess/config/resolvers/__init__.py +5 -0
  14. pytesprocess/config/resolvers/common.py +56 -0
  15. pytesprocess/config/resolvers/feature.py +293 -0
  16. pytesprocess/config/resolvers/salting.py +86 -0
  17. pytesprocess/config/resolvers/trigger.py +84 -0
  18. pytesprocess/config/selectors.py +108 -0
  19. pytesprocess/config/validation.py +314 -0
  20. pytesprocess/config/warnings.py +2 -0
  21. pytesprocess/core/__init__.py +10 -0
  22. pytesprocess/core/algorithms.py +1455 -0
  23. pytesprocess/core/didv.py +1648 -0
  24. pytesprocess/core/eventbuilder.py +495 -0
  25. pytesprocess/core/filterbuilder.py +81 -0
  26. pytesprocess/core/filterdata.py +1849 -0
  27. pytesprocess/core/ivsweep.py +2072 -0
  28. pytesprocess/core/noise.py +923 -0
  29. pytesprocess/core/noisemodel.py +1408 -0
  30. pytesprocess/core/oftrigger.py +1035 -0
  31. pytesprocess/core/template.py +450 -0
  32. pytesprocess/process/__init__.py +6 -0
  33. pytesprocess/process/data_source.py +185 -0
  34. pytesprocess/process/event_context.py +35 -0
  35. pytesprocess/process/feature_plan.py +186 -0
  36. pytesprocess/process/feature_resources.py +267 -0
  37. pytesprocess/process/features.py +1024 -0
  38. pytesprocess/process/filterprocess.py +1176 -0
  39. pytesprocess/process/ivprocess.py +1380 -0
  40. pytesprocess/process/processing_data.py +967 -0
  41. pytesprocess/process/randoms.py +921 -0
  42. pytesprocess/process/triggers.py +1011 -0
  43. pytesprocess/salting/__init__.py +7 -0
  44. pytesprocess/salting/generator.py +364 -0
  45. pytesprocess/salting/injector.py +329 -0
  46. pytesprocess/salting/sampling.py +84 -0
  47. pytesprocess/utils/__init__.py +5 -0
  48. pytesprocess/utils/arg_utils.py +122 -0
  49. pytesprocess/utils/dataframe_output.py +120 -0
  50. pytesprocess/utils/filter_hdf5.py +594 -0
  51. pytesprocess/utils/utils.py +701 -0
  52. pytesprocess/workflows/__init__.py +3 -0
  53. pytesprocess/workflows/processing.py +317 -0
  54. pytesprocess/workflows/salting.py +133 -0
  55. pytesprocess-0.1.1.dist-info/METADATA +211 -0
  56. pytesprocess-0.1.1.dist-info/RECORD +60 -0
  57. pytesprocess-0.1.1.dist-info/WHEEL +5 -0
  58. pytesprocess-0.1.1.dist-info/entry_points.txt +2 -0
  59. pytesprocess-0.1.1.dist-info/licenses/LICENSE +21 -0
  60. pytesprocess-0.1.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1011 @@
1
+ import copy
2
+ from itertools import repeat
3
+ import os
4
+ from pathlib import Path
5
+ import warnings
6
+
7
+ from humanfriendly import parse_size
8
+ import numpy as np
9
+ import pyarrow as pa
10
+ import vaex as vx
11
+ import yaml
12
+ from pytesdaqx.io import AcquisitionCatalog
13
+
14
+ from pytesprocess.config import ProcessingConfig
15
+ from pytesprocess.core.eventbuilder import EventBuilder
16
+ from pytesprocess.core.oftrigger import OptimumFilterTrigger
17
+ from pytesprocess.process.processing_data import ProcessingData
18
+ from pytesprocess.utils import (
19
+ convert_length_msec_to_samples,
20
+ create_dataframe_group,
21
+ )
22
+
23
+ warnings.filterwarnings("ignore")
24
+
25
+ vx.settings.main.thread_count = 1
26
+ vx.settings.main.thread_count_io = 1
27
+ pa.set_cpu_count(1)
28
+
29
+ os.environ["OMP_NUM_THREADS"] = "1"
30
+ os.environ["MKL_NUM_THREADS"] = "1"
31
+ os.environ["NUMEXPR_NUM_THREADS"] = "1"
32
+ os.environ["OPENBLAS_NUM_THREADS"] = "1"
33
+
34
+
35
+ __all__ = [
36
+ 'TriggerProcessing'
37
+ ]
38
+
39
+
40
+ class TriggerProcessing:
41
+ """
42
+ Class to manage trigger processing and
43
+ extract features, dataframe can be saved
44
+ in hdf5 using vaex framework
45
+
46
+ Multiple nodes can be used if data splitted in
47
+ different streams
48
+
49
+ """
50
+
51
+ def __init__(self, data_path, config_data,
52
+ streams=None,
53
+ processing_label=None,
54
+ restricted=False,
55
+ data_type='background',
56
+ salting_dataframe=None,
57
+ verbose=True):
58
+ """
59
+ Intialize data processing
60
+
61
+ Parameters
62
+ ---------
63
+
64
+ data_path : str
65
+ raw data directory
66
+
67
+ config_data : str or ProcessingConfig object
68
+ Full path and file name to the YAML settings for the
69
+ processing or ProcessingConfig object
70
+
71
+ streams : str or list of str, optional
72
+ Streams to process.
73
+
74
+ processing_label : str, optional
75
+ an optional processing name. This is used to be build
76
+ output subdirectory name and is saved as a feature in DetaFrame
77
+ so it can then be used later during
78
+ analysis to make a cut on a specific processing when mutliple
79
+ datasets/processing are added together.
80
+
81
+ restricted : boolean
82
+ if True, use restricted data
83
+ if False (default), exclude restricted data
84
+
85
+ data_type : str, optional
86
+ measurement type, default='background'
87
+
88
+ salting_dataframe : str or vaex dataframe
89
+ str if path to vaex hdf5 file or directly a dataframe
90
+
91
+ verbose : bool, optional
92
+ if True, display info
93
+
94
+
95
+
96
+ Return
97
+ ------
98
+ None
99
+ """
100
+
101
+ # display
102
+ self._verbose = verbose
103
+
104
+ # processing id
105
+ self._processing_label = processing_label
106
+
107
+ # restricted
108
+ self._restricted = restricted
109
+ if data_type != 'background':
110
+ self._restricted = False
111
+
112
+ # extract input file list
113
+ # data type
114
+ self._data_type = data_type
115
+
116
+ # streams
117
+ self._streams = streams
118
+
119
+ # catalog
120
+ catalog_full = AcquisitionCatalog(data_path, verbose=verbose)
121
+
122
+ # filter based on user input
123
+ self._catalog = catalog_full.filter(
124
+ measurement_types=self._data_type,
125
+ streams=self._streams,
126
+ restricted=self._restricted)
127
+
128
+ self._acquisition_name = self._catalog.acquisition_name
129
+ self._acquisition_path = self._catalog.acquisition_path
130
+ path_obj = Path(self._acquisition_path)
131
+ self._acquisition_base_path = str(path_obj.parent)
132
+ self._available_channels = self._catalog.record_channels
133
+ self._sample_rate = self._catalog.sample_rate_hz
134
+
135
+
136
+ rawdata_files = copy.deepcopy(
137
+ self._catalog.select_files_by_stream(
138
+ stream_key='stream_id',
139
+ )
140
+ )
141
+
142
+ if not rawdata_files:
143
+ raise ValueError('ERROR: No files were found! Check configuration...')
144
+
145
+ # list of streams
146
+ self._stream_list = list(rawdata_files.keys())
147
+
148
+
149
+ # config file
150
+ config_dict = {}
151
+ if isinstance(config_data, str):
152
+
153
+ if not os.path.isfile(config_data):
154
+ raise ValueError(f'ERROR: argument "{config_data}" '
155
+ f'should be a file or ProcessingConfig object!')
156
+
157
+ yaml = ProcessingConfig(config_data, self._available_channels, sample_rate=self._sample_rate)
158
+ config_dict = yaml.get_config('trigger')
159
+
160
+ else:
161
+
162
+ if not isinstance(config_data, ProcessingConfig):
163
+ raise ValueError(
164
+ 'ERROR: raw data argument should be either '
165
+ 'a directory or ProcessingConfig object'
166
+ )
167
+
168
+ config_dict = config_data.get_config('trigger')
169
+
170
+ self._trigger_config = copy.deepcopy(config_dict['channels'])
171
+ self._evtbuilder_config = copy.deepcopy(config_dict['overall'])
172
+ self._trigger_channels = copy.deepcopy(config_dict['channel_list'])
173
+
174
+ if not 'filter_file' in config_dict['overall']:
175
+ raise ValueError('ERROR: Filter file missing in yaml file!')
176
+
177
+ # check channels to be processed
178
+ if not self._trigger_channels:
179
+ raise ValueError('No trigger channels to be processed! ' +
180
+ 'Check configuration...')
181
+
182
+ # initialize output path
183
+ self._output_group_path = None
184
+
185
+ # instantiate processing data
186
+ self._processing_data_inst = ProcessingData(
187
+ self._acquisition_base_path,
188
+ rawdata_files,
189
+ acquisition_name=self._acquisition_name,
190
+ filter_file=config_dict['overall']['filter_file'],
191
+ salting_dataframe=salting_dataframe,
192
+ available_channels=self._available_channels,
193
+ verbose=verbose,
194
+ catalog=self._catalog,
195
+ )
196
+
197
+
198
+ def get_output_path(self):
199
+ """
200
+ Get output group path
201
+ """
202
+ return self._output_group_path
203
+
204
+
205
+ def process(self, ntriggers=-1,
206
+ lgc_save=False,
207
+ lgc_output=False,
208
+ save_path=None,
209
+ output_group_name=None,
210
+ ncores=1,
211
+ edge_exclusion_msec=None,
212
+ livetime=None,
213
+ partition_target_duration_s=10.0,
214
+ memory_limit='2GB'):
215
+
216
+ """
217
+ Process data
218
+
219
+ Parameters
220
+ ---------
221
+
222
+ ntriggers : int, optional
223
+ number of events to be processed
224
+ if not all events, requires ncores = 1
225
+ Default: all available (=-1)
226
+
227
+ lgc_save : bool, optional
228
+ if True, save dataframe in hdf5 files
229
+ (dataframe not returned)
230
+ if False, return dataframe (memory limit applies
231
+ so not all events may be processed)
232
+ Default: True
233
+
234
+ output_group_path : str, optional
235
+ base directory where output group will be saved
236
+ default: same base path as input data
237
+
238
+ ncores: int, optional
239
+ number of cores that will be used for processing
240
+ default: 1
241
+
242
+ partition_target_duration_s : float, optional
243
+ Core partition target duration for continuous Zarr triggering.
244
+ Adjacent reads overlap automatically so internal partition edges do
245
+ not create trigger deadtime. Default: 10 seconds.
246
+
247
+ memory_limit : str or float, optional
248
+ memory safety limit when returning a dataframe in memory.
249
+ Saved output is not split by this value; each worker writes one F#### shard.
250
+
251
+ """
252
+
253
+
254
+ # check input
255
+ if (ncores>1 and ntriggers>-1):
256
+ raise ValueError('ERROR: Multi cores processing only allowed when '
257
+ + 'processing ALL events!')
258
+
259
+ # check number cores allowed
260
+ if ncores>len(self._stream_list):
261
+ ncores = len(self._stream_list)
262
+ if self._verbose:
263
+ print('INFO: Changing number cores to '
264
+ + str(ncores) + ' (maximum allowed)')
265
+
266
+ # edge exclusion and livetime
267
+ self._edge_exclusion_msec = edge_exclusion_msec
268
+ self._livetime = livetime
269
+ self._partition_target_duration_s = float(partition_target_duration_s)
270
+ if self._partition_target_duration_s <= 0:
271
+ raise ValueError('ERROR: partition_target_duration_s must be positive!')
272
+
273
+ # Create one dataframe-group identity for the whole processing call.
274
+ # F#### is a task/shard index, not a second processing ID.
275
+ output_group_path = None
276
+ dataframe_group = None
277
+
278
+ if lgc_save:
279
+ if save_path is None:
280
+ save_path = self._acquisition_base_path + '/processed'
281
+ if '/raw/processed' in save_path:
282
+ save_path = save_path.replace('/raw/processed', '/processed')
283
+
284
+ if self._acquisition_name not in save_path:
285
+ save_path = save_path + '/' + self._acquisition_name
286
+
287
+ dataframe_group = create_dataframe_group(
288
+ save_path,
289
+ dataframe_type='trigger',
290
+ facility=self._processing_data_inst.get_facility(),
291
+ processing_label=self._processing_label,
292
+ restricted=self._restricted,
293
+ data_type=self._data_type,
294
+ output_group_name=output_group_name,
295
+ )
296
+ output_group_path = dataframe_group.path
297
+ if self._verbose:
298
+ print(f'INFO: Processing output group path: {output_group_path}')
299
+
300
+ self._output_group_path = output_group_path
301
+ self._dataframe_group = dataframe_group
302
+
303
+
304
+ # convert memory usage in bytes
305
+ if isinstance(memory_limit, str):
306
+ memory_limit = parse_size(memory_limit)
307
+
308
+ # initialize output
309
+ output_df = None
310
+
311
+ # case only 1 node used for processing
312
+ if ncores == 1:
313
+ output_df = self._process(1,
314
+ self._stream_list,
315
+ ntriggers,
316
+ lgc_save,
317
+ lgc_output,
318
+ dataframe_group,
319
+ output_group_path,
320
+ memory_limit)
321
+
322
+ else:
323
+
324
+
325
+ # disable vaex multi-threading
326
+ vx.settings.main.thread_count = 1
327
+ vx.settings.main.thread_count_io = 1
328
+ pa.set_cpu_count(1)
329
+
330
+ # split data
331
+ stream_list_split = self._split_streams(ncores)
332
+
333
+ # for multi-core processing, we need to decrease the
334
+ # max memory so it fits in RAM
335
+ memory_limit /= ncores
336
+
337
+ # lauch pool processing
338
+ if self._verbose:
339
+ print(f'INFO: Processing with be split between {ncores} cores!')
340
+
341
+ node_nums = list(range(ncores+1))[1:]
342
+ pool = Pool(processes=ncores)
343
+ output_df_list = pool.starmap(self._process,
344
+ zip(node_nums,
345
+ stream_list_split,
346
+ repeat(ntriggers),
347
+ repeat(lgc_save),
348
+ repeat(lgc_output),
349
+ repeat(dataframe_group),
350
+ repeat(output_group_path),
351
+ repeat(memory_limit)))
352
+ pool.close()
353
+ pool.join()
354
+
355
+ # concatenate output
356
+ if lgc_output:
357
+ df_list = list()
358
+ for df in output_df_list:
359
+ if df is not None:
360
+ df_list.append(df)
361
+ if df_list:
362
+ output_df = vx.concat(df_list)
363
+
364
+ # processing done
365
+ if self._verbose:
366
+ print('INFO: Trigger processing done!')
367
+
368
+ if lgc_output:
369
+ return output_df
370
+
371
+
372
+ def _process(self, node_num,
373
+ stream_list, ntriggers,
374
+ lgc_save, lgc_output,
375
+ dataframe_group,
376
+ output_group_path,
377
+ memory_limit):
378
+ """
379
+ Process data
380
+
381
+ Parameters
382
+ ---------
383
+
384
+ node_num : int
385
+ node id number, used for display
386
+
387
+ stream_list : str
388
+ list of streams name to be processed
389
+
390
+ ntriggers : int, optional
391
+ number of events to be processed
392
+ if not all events, requires ncores = 1
393
+ Default: all available (=-1)
394
+
395
+ lgc_save : bool, optional
396
+ if True, save dataframe in hdf5 files
397
+ (dataframe not returned)
398
+ if False, return dataframe (memory limit applies
399
+ so not all events may be processed)
400
+ Default: True
401
+
402
+ output_path : str, optional
403
+ base directory where output feature file will be saved
404
+ default: same base path as input data
405
+
406
+ ncores: int, optional
407
+ number of cores that will be used for processing
408
+ default: 1
409
+
410
+ memory_limit : float, optionl
411
+ memory limit per file in bytes
412
+ (and/or if return_df=True, max dataframe size)
413
+
414
+ """
415
+
416
+
417
+ # disable vaex multi-threading
418
+ vx.settings.main.thread_count = 1
419
+ vx.settings.main.thread_count_io = 1
420
+ pa.set_cpu_count(1)
421
+
422
+ # check argument
423
+ if lgc_output and lgc_save:
424
+ raise ValueError('ERROR: Unable to save and output datafame '
425
+ + 'at the same time. Set either lgc_output '
426
+ + 'or lgc_save to False.')
427
+
428
+ # node string (for display)
429
+ node_num_str = str()
430
+ if node_num>-1:
431
+ node_num_str = ' node #' + str(node_num)
432
+
433
+
434
+ # salting dataframe
435
+ self._processing_data_inst.load_salting_dataframe()
436
+
437
+ # instantiate event builder
438
+ evtbuilder_inst = EventBuilder()
439
+
440
+ # instantiate OF trigger and add to EventBuilder'
441
+ trigger_config = copy.deepcopy(self._trigger_config)
442
+ nb_trigger_chans = len(list(trigger_config.keys()))
443
+ overlap_left_samples = 0
444
+ overlap_right_samples = 0
445
+ max_pileup_samples = 0
446
+ for trig_chan, trig_data in trigger_config.items():
447
+
448
+ # channel name
449
+ channel_name = trig_data['channel_name']
450
+
451
+ # get template
452
+ template_tag = 'default'
453
+ if 'template_tag' in trig_data:
454
+ template_tag = trig_data['template_tag']
455
+
456
+ template, template_metadata = (
457
+ self._processing_data_inst.get_template(
458
+ channel_name,
459
+ tag=template_tag)
460
+ )
461
+
462
+ nb_pretrigger_samples = None
463
+ if 'nb_pretrigger_samples' in template_metadata.keys():
464
+ nb_pretrigger_samples = (
465
+ template_metadata['nb_pretrigger_samples']
466
+ )
467
+ else:
468
+ # back compatibility
469
+ if 'pretrigger_length_samples' in template_metadata.keys():
470
+ nb_pretrigger_samples = (
471
+ template_metadata['pretrigger_length_samples']
472
+ )
473
+ elif 'pretrigger_samples' in template_metadata.keys():
474
+ nb_pretrigger_samples = (
475
+ template_metadata['pretrigger_samples']
476
+ )
477
+ if nb_pretrigger_samples is None:
478
+ raise ValueError('ERROR: Template metadata needs to contain '
479
+ '"nb_pretrigger_samples" value')
480
+
481
+ # Continuous-Zarr partitions are read with overlap. The current
482
+ # OptimumFilterTrigger zeros one template length at each input edge;
483
+ # include that padding plus the trigger-index shift so the logical
484
+ # partition core has no artificial internal-edge deadtime.
485
+ nb_template_samples = int(template.shape[-1])
486
+ trigger_shift = int(nb_pretrigger_samples) - nb_template_samples // 2
487
+ overlap_left_samples = max(
488
+ overlap_left_samples,
489
+ nb_template_samples + max(trigger_shift, 0),
490
+ )
491
+ overlap_right_samples = max(
492
+ overlap_right_samples,
493
+ nb_template_samples + max(-trigger_shift, 0),
494
+ )
495
+
496
+ if trig_data.get('pileup_window_samples') is not None:
497
+ max_pileup_samples = max(
498
+ max_pileup_samples, int(trig_data['pileup_window_samples'])
499
+ )
500
+ elif trig_data.get('pileup_window_msec') is not None:
501
+ max_pileup_samples = max(
502
+ max_pileup_samples,
503
+ convert_length_msec_to_samples(
504
+ trig_data['pileup_window_msec'], self._sample_rate
505
+ ),
506
+ )
507
+
508
+ # Get noise spectrum (CSD/PSD)
509
+ csd_tag = 'default'
510
+ if 'csd_tag' in trig_data:
511
+ csd_tag = trig_data['csd_tag']
512
+
513
+ csd, csd_freqs, csd_metadata = (
514
+ self._processing_data_inst.get_noise(
515
+ channel_name,
516
+ tag=csd_tag)
517
+ )
518
+
519
+ # ignored frequency peaks
520
+ frequency_peaks = None
521
+ ignore_harmonics = False
522
+ if 'ignored_frequency_peaks' in trig_data:
523
+ frequency_peaks = trig_data['ignored_frequency_peaks']
524
+ if not isinstance(frequency_peaks, list):
525
+ frequency_peaks = [frequency_peaks]
526
+ if 'ignore_harmonics' in trig_data:
527
+ ignore_harmonics = trig_data['ignore_harmonics']
528
+
529
+ # sample rate
530
+ fs = self._processing_data_inst.get_sample_rate()
531
+
532
+ # instantiate optimal filter trigger
533
+ oftrigger_inst = OptimumFilterTrigger(
534
+ channel_name, fs, template, csd,
535
+ nb_pretrigger_samples,
536
+ ignored_frequency_peaks=frequency_peaks,
537
+ ignore_harmonics=ignore_harmonics,
538
+ trigger_name=trig_chan
539
+ )
540
+
541
+ # add in EventBuilder
542
+ evtbuilder_inst.add_trigger_object(
543
+ trig_chan, oftrigger_inst)
544
+
545
+ edge_samples = 0
546
+ if self._edge_exclusion_msec is not None:
547
+ edge_samples = convert_length_msec_to_samples(
548
+ self._edge_exclusion_msec, fs
549
+ )
550
+ coincident_samples = 0
551
+ if self._evtbuilder_config.get('coincident_window_samples') is not None:
552
+ coincident_samples = int(
553
+ self._evtbuilder_config['coincident_window_samples']
554
+ )
555
+ elif self._evtbuilder_config.get('coincident_window_msec') is not None:
556
+ coincident_samples = convert_length_msec_to_samples(
557
+ self._evtbuilder_config['coincident_window_msec'], fs
558
+ )
559
+ # The physical stream-edge deadtime is set by OF validity/padding and
560
+ # any explicit edge exclusion. Pileup/coincidence context is additional
561
+ # overlap needed only at *internal* partition boundaries and therefore
562
+ # must not be counted as lost livetime.
563
+ stream_edge_left_samples = max(overlap_left_samples, edge_samples)
564
+ stream_edge_right_samples = max(overlap_right_samples, edge_samples)
565
+ merge_context_samples = max(max_pileup_samples, coincident_samples)
566
+ overlap_left_samples = stream_edge_left_samples + merge_context_samples
567
+ overlap_right_samples = stream_edge_right_samples + merge_context_samples
568
+ self._processing_data_inst.configure_trigger_partitions(
569
+ partition_target_duration_s=self._partition_target_duration_s,
570
+ overlap_left_samples=overlap_left_samples,
571
+ overlap_right_samples=overlap_right_samples,
572
+ )
573
+
574
+ trigger_livetime = self._livetime
575
+ if self._catalog.storage_format == 'zarr':
576
+ # With overlapping partitions, internal partition boundaries no
577
+ # longer contribute deadtime. Only the physical beginning/end of
578
+ # each continuous stream remain unavailable with the current OF
579
+ # padding behavior.
580
+ trigger_livetime = 0.0
581
+ for stream_id in self._stream_list:
582
+ stream_catalog = self._catalog.filter(streams=stream_id)
583
+ n_samples = stream_catalog.n_samples
584
+ if n_samples is None:
585
+ continue
586
+ usable = max(
587
+ 0,
588
+ int(n_samples)
589
+ - stream_edge_left_samples
590
+ - stream_edge_right_samples,
591
+ )
592
+ trigger_livetime += usable / fs
593
+ if self._verbose and node_num == 1:
594
+ print(
595
+ 'INFO: Continuous-Zarr trigger livetime with overlapping '
596
+ f'partitions = {trigger_livetime/60:.3f} minutes'
597
+ )
598
+
599
+ output_file = None
600
+ dataframe_metadata = {}
601
+ if lgc_save:
602
+ output_file = dataframe_group.file_path(node_num)
603
+ dataframe_metadata = dataframe_group.row_metadata(node_num)
604
+
605
+ trigger_counter = 0
606
+
607
+ # intialize output dataframe
608
+ process_df = None
609
+
610
+ # loop streams assigned to this task/worker
611
+ for stream_index, stream in enumerate(stream_list):
612
+ final_stream = (stream_index == len(stream_list) - 1)
613
+
614
+ if self._verbose:
615
+ print('INFO' + node_num_str
616
+ + ': starting processing stream '
617
+ + stream)
618
+
619
+ # set file list
620
+ self._processing_data_inst.set_stream(stream)
621
+
622
+ # loop events
623
+ do_stop = False
624
+ while (not do_stop):
625
+
626
+ # -----------------------
627
+ # Check number events
628
+ # and memory usage
629
+ # -----------------------
630
+ ntriggers_limit_reached = (ntriggers>0
631
+ and trigger_counter>=ntriggers)
632
+
633
+ # flag memory limit reached
634
+ memory_usage = 0
635
+ memory_limit_reached = False
636
+ if process_df is not None:
637
+ memory_usage = process_df.shape[0]*process_df.shape[1]*8
638
+ memory_limit_reached = (not lgc_save and memory_usage >= memory_limit)
639
+
640
+ # display
641
+ if self._verbose:
642
+ if (trigger_counter%100==0 and trigger_counter!=0):
643
+ print('INFO' + node_num_str
644
+ + ': Local number of events = '
645
+ + str(trigger_counter)
646
+ + ' (memory = ' + str(memory_usage/1e6) + ' MB)')
647
+
648
+ # -----------------------
649
+ # Read next event
650
+ # -----------------------
651
+ success = self._processing_data_inst.read_next_event(
652
+ channels=self._trigger_channels
653
+ )
654
+
655
+ # end of file or raw data issue
656
+ if not success:
657
+ print('INFO' + node_num_str
658
+ + ': '
659
+ + str(trigger_counter)
660
+ + ' events counted, triggering processing done')
661
+ do_stop = True
662
+
663
+ # -----------------------
664
+ # Handle stop or
665
+ # nb trigger/memory limit
666
+ # reached
667
+ # -----------------------
668
+
669
+
670
+ # let's handle case we need to stop
671
+ # or memory/nb events limit reached
672
+ if ((do_stop and (final_stream or not lgc_save))
673
+ or ntriggers_limit_reached
674
+ or memory_limit_reached):
675
+
676
+
677
+ # case nb triggers reached
678
+ if ntriggers_limit_reached:
679
+ nextra = trigger_counter-ntriggers
680
+ nkeep = len(process_df)-nextra
681
+ if nkeep>0:
682
+ process_df = process_df[0:nkeep]
683
+
684
+
685
+ # save file if needed
686
+ if lgc_save and process_df is not None:
687
+ # One processing task writes exactly one F#### shard.
688
+ try:
689
+ process_df.export_hdf5(output_file, mode='w')
690
+ process_df.close()
691
+ except Exception as e:
692
+ print('WARNING: Export failed with error: ', e)
693
+ print('Will try again...')
694
+ df_pandas = process_df.to_pandas_df()
695
+ process_df = vx.from_pandas(df_pandas, copy_index=False)
696
+ process_df.export_hdf5(output_file, mode='w')
697
+ process_df.close()
698
+ del process_df
699
+ process_df = None
700
+
701
+
702
+ # case maximum number of events reached
703
+ # -> processing done!
704
+ if ntriggers_limit_reached:
705
+ if self._verbose:
706
+ print('INFO' + node_num_str
707
+ + ': Requested nb events reached. '
708
+ + 'Stopping processing!')
709
+
710
+ return process_df
711
+
712
+ # case memory limit reached
713
+ # -> processing needs to stop!
714
+ if lgc_output and memory_limit_reached:
715
+ raise ValueError(
716
+ 'ERROR: memory limit reached! '
717
+ + 'Change memory limit or only save hdf5 files '
718
+ +'(lgc_save=True AND lgc_output=False) '
719
+ )
720
+
721
+
722
+ # check if stop
723
+ if do_stop:
724
+ break
725
+
726
+
727
+ # -----------------------
728
+ # process triggers
729
+ # -----------------------
730
+
731
+ # clear event
732
+ evtbuilder_inst.clear_event()
733
+
734
+
735
+ # loop trigger channels
736
+ for trig_chan, trig_data in trigger_config.items():
737
+
738
+ # channel
739
+ channel_name = trig_data['channel_name']
740
+
741
+ # get threshold
742
+ threshold = None
743
+ if 'threshold_sigma' in trig_data.keys():
744
+ threshold = float(trig_data['threshold_sigma'])
745
+ elif 'threshold' in trig_data.keys():
746
+ threshold = float(trig_data['threshold'])
747
+ else:
748
+ raise ValueError(
749
+ 'ERROR: "treshold_sigma" missing in '
750
+ + 'yaml configuration file')
751
+
752
+ # pileup window
753
+ pileup_window_msec = None
754
+ if 'pileup_window_msec' in trig_data.keys():
755
+ pileup_window_msec = float(
756
+ trig_data['pileup_window_msec']
757
+ )
758
+
759
+ pileup_window_samples = None
760
+ if 'pileup_window_samples' in trig_data.keys():
761
+ pileup_window_samples = int(
762
+ trig_data['pileup_window_samples'])
763
+
764
+ # residual triggering
765
+ run_residual = False
766
+ sat_amps_50kHz = None
767
+ if 'run_residual' in trig_data.keys():
768
+ run_residual = bool(
769
+ trig_data['run_residual'])
770
+ if 'sat_amps_50kHz' in trig_data.keys():
771
+ sat_amps_50kHz = trig_data['sat_amps_50kHz']
772
+ sat_amps_50kHz = [float(amp) for amp in sat_amps_50kHz]
773
+
774
+ # positive pulse
775
+ positive_pulses = True
776
+ if 'positive_pulses' in trig_data.keys():
777
+ positive_pulses = trig_data['positive_pulses']
778
+
779
+ # get trace (If multiple channels, need to follow order)
780
+ trace = self._processing_data_inst.get_channel_trace(
781
+ channel_name
782
+ )
783
+
784
+ # acquire trigger
785
+ evtbuilder_inst.acquire_triggers(
786
+ trig_chan,
787
+ trace,
788
+ threshold,
789
+ pileup_window_msec=pileup_window_msec,
790
+ pileup_window_samples=pileup_window_samples,
791
+ positive_pulses=positive_pulses,
792
+ run_residual=run_residual,
793
+ sat_amps_50kHz=sat_amps_50kHz,
794
+ edge_exclusion_msec=self._edge_exclusion_msec,
795
+ livetime=trigger_livetime
796
+ )
797
+
798
+
799
+ # -----------------------
800
+ # build event
801
+ # merge coincident triggers
802
+ # -----------------------
803
+
804
+ coincident_window_msec = None
805
+ if 'coincident_window_msec' in self._evtbuilder_config.keys():
806
+ coincident_window_msec = (
807
+ self._evtbuilder_config['coincident_window_msec'])
808
+
809
+ coincident_window_samples = None
810
+ if 'coincident_window_samples' in self._evtbuilder_config.keys():
811
+ coincident_window_samples = (
812
+ self._evtbuilder_config['coincident_window_samples'])
813
+
814
+ # get event metadata
815
+ event_info = self._processing_data_inst.get_event_admin()
816
+
817
+ # build event
818
+ evtbuilder_inst.build_event(
819
+ event_info,
820
+ fs=fs,
821
+ coincident_window_msec=coincident_window_msec,
822
+ coincident_window_samples=coincident_window_samples,
823
+ nb_trigger_channels=nb_trigger_chans
824
+ )
825
+
826
+
827
+ # get trigger data
828
+ event_df = evtbuilder_inst.get_event_df()
829
+
830
+ # For continuous Zarr, trigger indices produced by the OF are
831
+ # relative to the expanded read window. Keep only triggers
832
+ # belonging to this non-overlapping logical core, then expose
833
+ # canonical stream/partition coordinates. The legacy
834
+ # ``trigger_index`` output is retained as a partition-local
835
+ # compatibility alias.
836
+ partition_context = self._processing_data_inst.get_partition_context()
837
+ if event_df is not None and partition_context is None and len(event_df):
838
+ # For native/legacy HDF5 records the trigger index is
839
+ # segment-local. Expose that coordinate explicitly rather
840
+ # than requiring downstream code to infer it from the
841
+ # generic trigger_index field.
842
+ if event_info.get('global_segment_number') is not None:
843
+ event_df['segment_trigger_index'] = np.asarray(
844
+ event_df['trigger_index'].values, dtype=np.int64
845
+ )
846
+ if event_df is not None and partition_context is not None:
847
+ local_indices = np.asarray(event_df['trigger_index'].values, dtype=np.int64)
848
+ stream_indices = (
849
+ local_indices
850
+ + int(partition_context['partition_read_start_index'])
851
+ )
852
+ core_start = int(partition_context['partition_start_index'])
853
+ core_stop = core_start + int(
854
+ partition_context['partition_length_samples']
855
+ )
856
+ keep = (stream_indices >= core_start) & (stream_indices < core_stop)
857
+ if not np.all(keep):
858
+ event_pd = event_df.to_pandas_df()
859
+ event_pd = event_pd.loc[keep].reset_index(drop=True)
860
+ event_df = vx.from_pandas(event_pd, copy_index=False)
861
+ stream_indices = stream_indices[keep]
862
+ if len(event_df):
863
+ partition_indices = stream_indices - core_start
864
+ event_df['stream_trigger_index'] = stream_indices
865
+ event_df['partition_trigger_index'] = partition_indices
866
+ event_df['partition_start_index'] = np.full(
867
+ len(event_df), core_start, dtype=np.int64
868
+ )
869
+ event_df['partition_length_samples'] = np.full(
870
+ len(event_df),
871
+ int(partition_context['partition_length_samples']),
872
+ dtype=np.int64,
873
+ )
874
+ event_df['stream_length_samples'] = np.full(
875
+ len(event_df),
876
+ int(partition_context['stream_length_samples']),
877
+ dtype=np.int64,
878
+ )
879
+ event_df['trigger_index'] = partition_indices
880
+ event_df['trigger_time'] = partition_indices / fs
881
+
882
+ # EventBuilder assumes non-overlapping input records
883
+ # when constructing event times. Recompute these legacy
884
+ # timing fields from the canonical stream-global index
885
+ # for overlapped Zarr partitions.
886
+ stream_start = event_info.get('stream_start_time')
887
+ if stream_start is not None and np.isfinite(float(stream_start)):
888
+ event_times_float = (
889
+ float(stream_start) + stream_indices / fs
890
+ )
891
+ event_df['event_time'] = np.rint(
892
+ event_times_float
893
+ ).astype(np.int64)
894
+ event_df['time_since_stream_start_s'] = (
895
+ stream_indices / fs
896
+ )
897
+ acquisition_start = event_info.get(
898
+ 'acquisition_start_time'
899
+ )
900
+ if (acquisition_start is not None
901
+ and np.isfinite(float(acquisition_start))):
902
+ event_df['time_since_acquisition_start_s'] = (
903
+ event_times_float - float(acquisition_start)
904
+ )
905
+ fridge_start = event_info.get('fridge_run_start_time')
906
+ if (fridge_start is not None
907
+ and np.isfinite(float(fridge_start))):
908
+ event_df['time_since_fridge_run_start_s'] = (
909
+ event_times_float - float(fridge_start)
910
+ )
911
+
912
+ # check if triggers
913
+ if (event_df is None or len(event_df)==0):
914
+ continue
915
+
916
+
917
+ # increment counter
918
+ nb_triggers = len(event_df)
919
+
920
+ trigger_counter += nb_triggers
921
+
922
+
923
+ # -----------------------
924
+ # Add metadata
925
+ # -----------------------
926
+
927
+ # add acquisition/stream provenance needed by feature
928
+ # processing. Keep legacy EventBuilder columns untouched.
929
+ event_df['acquisition_name'] = np.array(
930
+ [self._acquisition_name] * nb_triggers
931
+ )
932
+ if 'stream_id' in event_info:
933
+ event_df['stream_id'] = np.array(
934
+ [str(event_info['stream_id'])] * nb_triggers
935
+ )
936
+ if 'stream_number' in event_info:
937
+ event_df['stream_number'] = np.array(
938
+ [np.int64(event_info['stream_number'])] * nb_triggers
939
+ )
940
+
941
+ # Canonical processed-dataframe identity. The same group ID is
942
+ # shared by every worker; only dataframe_file_index differs.
943
+ for key, value in dataframe_metadata.items():
944
+ event_df[key] = np.array([value] * nb_triggers)
945
+
946
+ # done processing event!
947
+ # append event dictionary to dataframe
948
+ if process_df is None:
949
+ process_df = event_df
950
+ else:
951
+ process_df = vx.concat([process_df, event_df])
952
+
953
+
954
+
955
+ # cleanup
956
+ del evtbuilder_inst
957
+
958
+ # return features
959
+ return process_df
960
+
961
+
962
+ def _create_output_directory(self, base_path, facility,
963
+ output_group_name=None,
964
+ restricted=False,
965
+ data_type='background'):
966
+ """Compatibility wrapper around the common dataframe-group helper."""
967
+ group = create_dataframe_group(
968
+ base_path, dataframe_type='trigger', facility=facility,
969
+ processing_label=self._processing_label, restricted=restricted,
970
+ data_type=data_type, output_group_name=output_group_name,
971
+ )
972
+ return group.path, group.group_number
973
+
974
+
975
+ def _split_streams(self, ncores):
976
+ """
977
+ Split raw streams/tasks between workers
978
+
979
+
980
+ Parameters
981
+ ----------
982
+
983
+ ncores : int
984
+ number of cores
985
+
986
+ Return
987
+ ------
988
+
989
+ output_list : list
990
+ list of dictionaries (length=ncores) containing
991
+ data
992
+
993
+
994
+ """
995
+
996
+ output_list = list()
997
+
998
+ # split streams
999
+ stream_split = np.array_split(self._stream_list, ncores)
1000
+
1001
+ # remove empty array
1002
+ for stream_sublist in stream_split:
1003
+ if stream_sublist.size == 0:
1004
+ continue
1005
+ output_list.append(list(stream_sublist))
1006
+
1007
+
1008
+ return output_list
1009
+
1010
+
1011
+