pytesprocess 0.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pytesprocess/__init__.py +9 -0
- pytesprocess/_version.py +2 -0
- pytesprocess/cli/__init__.py +1 -0
- pytesprocess/cli/commands/__init__.py +5 -0
- pytesprocess/cli/commands/event.py +66 -0
- pytesprocess/cli/commands/filter.py +17 -0
- pytesprocess/cli/commands/ivsweep.py +29 -0
- pytesprocess/cli/common.py +86 -0
- pytesprocess/cli/main.py +81 -0
- pytesprocess/config/__init__.py +4 -0
- pytesprocess/config/loader.py +94 -0
- pytesprocess/config/manager.py +297 -0
- pytesprocess/config/resolvers/__init__.py +5 -0
- pytesprocess/config/resolvers/common.py +56 -0
- pytesprocess/config/resolvers/feature.py +293 -0
- pytesprocess/config/resolvers/salting.py +86 -0
- pytesprocess/config/resolvers/trigger.py +84 -0
- pytesprocess/config/selectors.py +108 -0
- pytesprocess/config/validation.py +314 -0
- pytesprocess/config/warnings.py +2 -0
- pytesprocess/core/__init__.py +10 -0
- pytesprocess/core/algorithms.py +1455 -0
- pytesprocess/core/didv.py +1648 -0
- pytesprocess/core/eventbuilder.py +495 -0
- pytesprocess/core/filterbuilder.py +81 -0
- pytesprocess/core/filterdata.py +1849 -0
- pytesprocess/core/ivsweep.py +2072 -0
- pytesprocess/core/noise.py +923 -0
- pytesprocess/core/noisemodel.py +1408 -0
- pytesprocess/core/oftrigger.py +1035 -0
- pytesprocess/core/template.py +450 -0
- pytesprocess/process/__init__.py +6 -0
- pytesprocess/process/data_source.py +185 -0
- pytesprocess/process/event_context.py +35 -0
- pytesprocess/process/feature_plan.py +186 -0
- pytesprocess/process/feature_resources.py +267 -0
- pytesprocess/process/features.py +1024 -0
- pytesprocess/process/filterprocess.py +1176 -0
- pytesprocess/process/ivprocess.py +1380 -0
- pytesprocess/process/processing_data.py +967 -0
- pytesprocess/process/randoms.py +921 -0
- pytesprocess/process/triggers.py +1011 -0
- pytesprocess/salting/__init__.py +7 -0
- pytesprocess/salting/generator.py +364 -0
- pytesprocess/salting/injector.py +329 -0
- pytesprocess/salting/sampling.py +84 -0
- pytesprocess/utils/__init__.py +5 -0
- pytesprocess/utils/arg_utils.py +122 -0
- pytesprocess/utils/dataframe_output.py +120 -0
- pytesprocess/utils/filter_hdf5.py +594 -0
- pytesprocess/utils/utils.py +701 -0
- pytesprocess/workflows/__init__.py +3 -0
- pytesprocess/workflows/processing.py +317 -0
- pytesprocess/workflows/salting.py +133 -0
- pytesprocess-0.1.1.dist-info/METADATA +211 -0
- pytesprocess-0.1.1.dist-info/RECORD +60 -0
- pytesprocess-0.1.1.dist-info/WHEEL +5 -0
- pytesprocess-0.1.1.dist-info/entry_points.txt +2 -0
- pytesprocess-0.1.1.dist-info/licenses/LICENSE +21 -0
- pytesprocess-0.1.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1011 @@
|
|
|
1
|
+
import copy
|
|
2
|
+
from itertools import repeat
|
|
3
|
+
import os
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
import warnings
|
|
6
|
+
|
|
7
|
+
from humanfriendly import parse_size
|
|
8
|
+
import numpy as np
|
|
9
|
+
import pyarrow as pa
|
|
10
|
+
import vaex as vx
|
|
11
|
+
import yaml
|
|
12
|
+
from pytesdaqx.io import AcquisitionCatalog
|
|
13
|
+
|
|
14
|
+
from pytesprocess.config import ProcessingConfig
|
|
15
|
+
from pytesprocess.core.eventbuilder import EventBuilder
|
|
16
|
+
from pytesprocess.core.oftrigger import OptimumFilterTrigger
|
|
17
|
+
from pytesprocess.process.processing_data import ProcessingData
|
|
18
|
+
from pytesprocess.utils import (
|
|
19
|
+
convert_length_msec_to_samples,
|
|
20
|
+
create_dataframe_group,
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
warnings.filterwarnings("ignore")
|
|
24
|
+
|
|
25
|
+
vx.settings.main.thread_count = 1
|
|
26
|
+
vx.settings.main.thread_count_io = 1
|
|
27
|
+
pa.set_cpu_count(1)
|
|
28
|
+
|
|
29
|
+
os.environ["OMP_NUM_THREADS"] = "1"
|
|
30
|
+
os.environ["MKL_NUM_THREADS"] = "1"
|
|
31
|
+
os.environ["NUMEXPR_NUM_THREADS"] = "1"
|
|
32
|
+
os.environ["OPENBLAS_NUM_THREADS"] = "1"
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
__all__ = [
|
|
36
|
+
'TriggerProcessing'
|
|
37
|
+
]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class TriggerProcessing:
|
|
41
|
+
"""
|
|
42
|
+
Class to manage trigger processing and
|
|
43
|
+
extract features, dataframe can be saved
|
|
44
|
+
in hdf5 using vaex framework
|
|
45
|
+
|
|
46
|
+
Multiple nodes can be used if data splitted in
|
|
47
|
+
different streams
|
|
48
|
+
|
|
49
|
+
"""
|
|
50
|
+
|
|
51
|
+
def __init__(self, data_path, config_data,
|
|
52
|
+
streams=None,
|
|
53
|
+
processing_label=None,
|
|
54
|
+
restricted=False,
|
|
55
|
+
data_type='background',
|
|
56
|
+
salting_dataframe=None,
|
|
57
|
+
verbose=True):
|
|
58
|
+
"""
|
|
59
|
+
Intialize data processing
|
|
60
|
+
|
|
61
|
+
Parameters
|
|
62
|
+
---------
|
|
63
|
+
|
|
64
|
+
data_path : str
|
|
65
|
+
raw data directory
|
|
66
|
+
|
|
67
|
+
config_data : str or ProcessingConfig object
|
|
68
|
+
Full path and file name to the YAML settings for the
|
|
69
|
+
processing or ProcessingConfig object
|
|
70
|
+
|
|
71
|
+
streams : str or list of str, optional
|
|
72
|
+
Streams to process.
|
|
73
|
+
|
|
74
|
+
processing_label : str, optional
|
|
75
|
+
an optional processing name. This is used to be build
|
|
76
|
+
output subdirectory name and is saved as a feature in DetaFrame
|
|
77
|
+
so it can then be used later during
|
|
78
|
+
analysis to make a cut on a specific processing when mutliple
|
|
79
|
+
datasets/processing are added together.
|
|
80
|
+
|
|
81
|
+
restricted : boolean
|
|
82
|
+
if True, use restricted data
|
|
83
|
+
if False (default), exclude restricted data
|
|
84
|
+
|
|
85
|
+
data_type : str, optional
|
|
86
|
+
measurement type, default='background'
|
|
87
|
+
|
|
88
|
+
salting_dataframe : str or vaex dataframe
|
|
89
|
+
str if path to vaex hdf5 file or directly a dataframe
|
|
90
|
+
|
|
91
|
+
verbose : bool, optional
|
|
92
|
+
if True, display info
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
Return
|
|
97
|
+
------
|
|
98
|
+
None
|
|
99
|
+
"""
|
|
100
|
+
|
|
101
|
+
# display
|
|
102
|
+
self._verbose = verbose
|
|
103
|
+
|
|
104
|
+
# processing id
|
|
105
|
+
self._processing_label = processing_label
|
|
106
|
+
|
|
107
|
+
# restricted
|
|
108
|
+
self._restricted = restricted
|
|
109
|
+
if data_type != 'background':
|
|
110
|
+
self._restricted = False
|
|
111
|
+
|
|
112
|
+
# extract input file list
|
|
113
|
+
# data type
|
|
114
|
+
self._data_type = data_type
|
|
115
|
+
|
|
116
|
+
# streams
|
|
117
|
+
self._streams = streams
|
|
118
|
+
|
|
119
|
+
# catalog
|
|
120
|
+
catalog_full = AcquisitionCatalog(data_path, verbose=verbose)
|
|
121
|
+
|
|
122
|
+
# filter based on user input
|
|
123
|
+
self._catalog = catalog_full.filter(
|
|
124
|
+
measurement_types=self._data_type,
|
|
125
|
+
streams=self._streams,
|
|
126
|
+
restricted=self._restricted)
|
|
127
|
+
|
|
128
|
+
self._acquisition_name = self._catalog.acquisition_name
|
|
129
|
+
self._acquisition_path = self._catalog.acquisition_path
|
|
130
|
+
path_obj = Path(self._acquisition_path)
|
|
131
|
+
self._acquisition_base_path = str(path_obj.parent)
|
|
132
|
+
self._available_channels = self._catalog.record_channels
|
|
133
|
+
self._sample_rate = self._catalog.sample_rate_hz
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
rawdata_files = copy.deepcopy(
|
|
137
|
+
self._catalog.select_files_by_stream(
|
|
138
|
+
stream_key='stream_id',
|
|
139
|
+
)
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
if not rawdata_files:
|
|
143
|
+
raise ValueError('ERROR: No files were found! Check configuration...')
|
|
144
|
+
|
|
145
|
+
# list of streams
|
|
146
|
+
self._stream_list = list(rawdata_files.keys())
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
# config file
|
|
150
|
+
config_dict = {}
|
|
151
|
+
if isinstance(config_data, str):
|
|
152
|
+
|
|
153
|
+
if not os.path.isfile(config_data):
|
|
154
|
+
raise ValueError(f'ERROR: argument "{config_data}" '
|
|
155
|
+
f'should be a file or ProcessingConfig object!')
|
|
156
|
+
|
|
157
|
+
yaml = ProcessingConfig(config_data, self._available_channels, sample_rate=self._sample_rate)
|
|
158
|
+
config_dict = yaml.get_config('trigger')
|
|
159
|
+
|
|
160
|
+
else:
|
|
161
|
+
|
|
162
|
+
if not isinstance(config_data, ProcessingConfig):
|
|
163
|
+
raise ValueError(
|
|
164
|
+
'ERROR: raw data argument should be either '
|
|
165
|
+
'a directory or ProcessingConfig object'
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
config_dict = config_data.get_config('trigger')
|
|
169
|
+
|
|
170
|
+
self._trigger_config = copy.deepcopy(config_dict['channels'])
|
|
171
|
+
self._evtbuilder_config = copy.deepcopy(config_dict['overall'])
|
|
172
|
+
self._trigger_channels = copy.deepcopy(config_dict['channel_list'])
|
|
173
|
+
|
|
174
|
+
if not 'filter_file' in config_dict['overall']:
|
|
175
|
+
raise ValueError('ERROR: Filter file missing in yaml file!')
|
|
176
|
+
|
|
177
|
+
# check channels to be processed
|
|
178
|
+
if not self._trigger_channels:
|
|
179
|
+
raise ValueError('No trigger channels to be processed! ' +
|
|
180
|
+
'Check configuration...')
|
|
181
|
+
|
|
182
|
+
# initialize output path
|
|
183
|
+
self._output_group_path = None
|
|
184
|
+
|
|
185
|
+
# instantiate processing data
|
|
186
|
+
self._processing_data_inst = ProcessingData(
|
|
187
|
+
self._acquisition_base_path,
|
|
188
|
+
rawdata_files,
|
|
189
|
+
acquisition_name=self._acquisition_name,
|
|
190
|
+
filter_file=config_dict['overall']['filter_file'],
|
|
191
|
+
salting_dataframe=salting_dataframe,
|
|
192
|
+
available_channels=self._available_channels,
|
|
193
|
+
verbose=verbose,
|
|
194
|
+
catalog=self._catalog,
|
|
195
|
+
)
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def get_output_path(self):
|
|
199
|
+
"""
|
|
200
|
+
Get output group path
|
|
201
|
+
"""
|
|
202
|
+
return self._output_group_path
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def process(self, ntriggers=-1,
|
|
206
|
+
lgc_save=False,
|
|
207
|
+
lgc_output=False,
|
|
208
|
+
save_path=None,
|
|
209
|
+
output_group_name=None,
|
|
210
|
+
ncores=1,
|
|
211
|
+
edge_exclusion_msec=None,
|
|
212
|
+
livetime=None,
|
|
213
|
+
partition_target_duration_s=10.0,
|
|
214
|
+
memory_limit='2GB'):
|
|
215
|
+
|
|
216
|
+
"""
|
|
217
|
+
Process data
|
|
218
|
+
|
|
219
|
+
Parameters
|
|
220
|
+
---------
|
|
221
|
+
|
|
222
|
+
ntriggers : int, optional
|
|
223
|
+
number of events to be processed
|
|
224
|
+
if not all events, requires ncores = 1
|
|
225
|
+
Default: all available (=-1)
|
|
226
|
+
|
|
227
|
+
lgc_save : bool, optional
|
|
228
|
+
if True, save dataframe in hdf5 files
|
|
229
|
+
(dataframe not returned)
|
|
230
|
+
if False, return dataframe (memory limit applies
|
|
231
|
+
so not all events may be processed)
|
|
232
|
+
Default: True
|
|
233
|
+
|
|
234
|
+
output_group_path : str, optional
|
|
235
|
+
base directory where output group will be saved
|
|
236
|
+
default: same base path as input data
|
|
237
|
+
|
|
238
|
+
ncores: int, optional
|
|
239
|
+
number of cores that will be used for processing
|
|
240
|
+
default: 1
|
|
241
|
+
|
|
242
|
+
partition_target_duration_s : float, optional
|
|
243
|
+
Core partition target duration for continuous Zarr triggering.
|
|
244
|
+
Adjacent reads overlap automatically so internal partition edges do
|
|
245
|
+
not create trigger deadtime. Default: 10 seconds.
|
|
246
|
+
|
|
247
|
+
memory_limit : str or float, optional
|
|
248
|
+
memory safety limit when returning a dataframe in memory.
|
|
249
|
+
Saved output is not split by this value; each worker writes one F#### shard.
|
|
250
|
+
|
|
251
|
+
"""
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
# check input
|
|
255
|
+
if (ncores>1 and ntriggers>-1):
|
|
256
|
+
raise ValueError('ERROR: Multi cores processing only allowed when '
|
|
257
|
+
+ 'processing ALL events!')
|
|
258
|
+
|
|
259
|
+
# check number cores allowed
|
|
260
|
+
if ncores>len(self._stream_list):
|
|
261
|
+
ncores = len(self._stream_list)
|
|
262
|
+
if self._verbose:
|
|
263
|
+
print('INFO: Changing number cores to '
|
|
264
|
+
+ str(ncores) + ' (maximum allowed)')
|
|
265
|
+
|
|
266
|
+
# edge exclusion and livetime
|
|
267
|
+
self._edge_exclusion_msec = edge_exclusion_msec
|
|
268
|
+
self._livetime = livetime
|
|
269
|
+
self._partition_target_duration_s = float(partition_target_duration_s)
|
|
270
|
+
if self._partition_target_duration_s <= 0:
|
|
271
|
+
raise ValueError('ERROR: partition_target_duration_s must be positive!')
|
|
272
|
+
|
|
273
|
+
# Create one dataframe-group identity for the whole processing call.
|
|
274
|
+
# F#### is a task/shard index, not a second processing ID.
|
|
275
|
+
output_group_path = None
|
|
276
|
+
dataframe_group = None
|
|
277
|
+
|
|
278
|
+
if lgc_save:
|
|
279
|
+
if save_path is None:
|
|
280
|
+
save_path = self._acquisition_base_path + '/processed'
|
|
281
|
+
if '/raw/processed' in save_path:
|
|
282
|
+
save_path = save_path.replace('/raw/processed', '/processed')
|
|
283
|
+
|
|
284
|
+
if self._acquisition_name not in save_path:
|
|
285
|
+
save_path = save_path + '/' + self._acquisition_name
|
|
286
|
+
|
|
287
|
+
dataframe_group = create_dataframe_group(
|
|
288
|
+
save_path,
|
|
289
|
+
dataframe_type='trigger',
|
|
290
|
+
facility=self._processing_data_inst.get_facility(),
|
|
291
|
+
processing_label=self._processing_label,
|
|
292
|
+
restricted=self._restricted,
|
|
293
|
+
data_type=self._data_type,
|
|
294
|
+
output_group_name=output_group_name,
|
|
295
|
+
)
|
|
296
|
+
output_group_path = dataframe_group.path
|
|
297
|
+
if self._verbose:
|
|
298
|
+
print(f'INFO: Processing output group path: {output_group_path}')
|
|
299
|
+
|
|
300
|
+
self._output_group_path = output_group_path
|
|
301
|
+
self._dataframe_group = dataframe_group
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
# convert memory usage in bytes
|
|
305
|
+
if isinstance(memory_limit, str):
|
|
306
|
+
memory_limit = parse_size(memory_limit)
|
|
307
|
+
|
|
308
|
+
# initialize output
|
|
309
|
+
output_df = None
|
|
310
|
+
|
|
311
|
+
# case only 1 node used for processing
|
|
312
|
+
if ncores == 1:
|
|
313
|
+
output_df = self._process(1,
|
|
314
|
+
self._stream_list,
|
|
315
|
+
ntriggers,
|
|
316
|
+
lgc_save,
|
|
317
|
+
lgc_output,
|
|
318
|
+
dataframe_group,
|
|
319
|
+
output_group_path,
|
|
320
|
+
memory_limit)
|
|
321
|
+
|
|
322
|
+
else:
|
|
323
|
+
|
|
324
|
+
|
|
325
|
+
# disable vaex multi-threading
|
|
326
|
+
vx.settings.main.thread_count = 1
|
|
327
|
+
vx.settings.main.thread_count_io = 1
|
|
328
|
+
pa.set_cpu_count(1)
|
|
329
|
+
|
|
330
|
+
# split data
|
|
331
|
+
stream_list_split = self._split_streams(ncores)
|
|
332
|
+
|
|
333
|
+
# for multi-core processing, we need to decrease the
|
|
334
|
+
# max memory so it fits in RAM
|
|
335
|
+
memory_limit /= ncores
|
|
336
|
+
|
|
337
|
+
# lauch pool processing
|
|
338
|
+
if self._verbose:
|
|
339
|
+
print(f'INFO: Processing with be split between {ncores} cores!')
|
|
340
|
+
|
|
341
|
+
node_nums = list(range(ncores+1))[1:]
|
|
342
|
+
pool = Pool(processes=ncores)
|
|
343
|
+
output_df_list = pool.starmap(self._process,
|
|
344
|
+
zip(node_nums,
|
|
345
|
+
stream_list_split,
|
|
346
|
+
repeat(ntriggers),
|
|
347
|
+
repeat(lgc_save),
|
|
348
|
+
repeat(lgc_output),
|
|
349
|
+
repeat(dataframe_group),
|
|
350
|
+
repeat(output_group_path),
|
|
351
|
+
repeat(memory_limit)))
|
|
352
|
+
pool.close()
|
|
353
|
+
pool.join()
|
|
354
|
+
|
|
355
|
+
# concatenate output
|
|
356
|
+
if lgc_output:
|
|
357
|
+
df_list = list()
|
|
358
|
+
for df in output_df_list:
|
|
359
|
+
if df is not None:
|
|
360
|
+
df_list.append(df)
|
|
361
|
+
if df_list:
|
|
362
|
+
output_df = vx.concat(df_list)
|
|
363
|
+
|
|
364
|
+
# processing done
|
|
365
|
+
if self._verbose:
|
|
366
|
+
print('INFO: Trigger processing done!')
|
|
367
|
+
|
|
368
|
+
if lgc_output:
|
|
369
|
+
return output_df
|
|
370
|
+
|
|
371
|
+
|
|
372
|
+
def _process(self, node_num,
|
|
373
|
+
stream_list, ntriggers,
|
|
374
|
+
lgc_save, lgc_output,
|
|
375
|
+
dataframe_group,
|
|
376
|
+
output_group_path,
|
|
377
|
+
memory_limit):
|
|
378
|
+
"""
|
|
379
|
+
Process data
|
|
380
|
+
|
|
381
|
+
Parameters
|
|
382
|
+
---------
|
|
383
|
+
|
|
384
|
+
node_num : int
|
|
385
|
+
node id number, used for display
|
|
386
|
+
|
|
387
|
+
stream_list : str
|
|
388
|
+
list of streams name to be processed
|
|
389
|
+
|
|
390
|
+
ntriggers : int, optional
|
|
391
|
+
number of events to be processed
|
|
392
|
+
if not all events, requires ncores = 1
|
|
393
|
+
Default: all available (=-1)
|
|
394
|
+
|
|
395
|
+
lgc_save : bool, optional
|
|
396
|
+
if True, save dataframe in hdf5 files
|
|
397
|
+
(dataframe not returned)
|
|
398
|
+
if False, return dataframe (memory limit applies
|
|
399
|
+
so not all events may be processed)
|
|
400
|
+
Default: True
|
|
401
|
+
|
|
402
|
+
output_path : str, optional
|
|
403
|
+
base directory where output feature file will be saved
|
|
404
|
+
default: same base path as input data
|
|
405
|
+
|
|
406
|
+
ncores: int, optional
|
|
407
|
+
number of cores that will be used for processing
|
|
408
|
+
default: 1
|
|
409
|
+
|
|
410
|
+
memory_limit : float, optionl
|
|
411
|
+
memory limit per file in bytes
|
|
412
|
+
(and/or if return_df=True, max dataframe size)
|
|
413
|
+
|
|
414
|
+
"""
|
|
415
|
+
|
|
416
|
+
|
|
417
|
+
# disable vaex multi-threading
|
|
418
|
+
vx.settings.main.thread_count = 1
|
|
419
|
+
vx.settings.main.thread_count_io = 1
|
|
420
|
+
pa.set_cpu_count(1)
|
|
421
|
+
|
|
422
|
+
# check argument
|
|
423
|
+
if lgc_output and lgc_save:
|
|
424
|
+
raise ValueError('ERROR: Unable to save and output datafame '
|
|
425
|
+
+ 'at the same time. Set either lgc_output '
|
|
426
|
+
+ 'or lgc_save to False.')
|
|
427
|
+
|
|
428
|
+
# node string (for display)
|
|
429
|
+
node_num_str = str()
|
|
430
|
+
if node_num>-1:
|
|
431
|
+
node_num_str = ' node #' + str(node_num)
|
|
432
|
+
|
|
433
|
+
|
|
434
|
+
# salting dataframe
|
|
435
|
+
self._processing_data_inst.load_salting_dataframe()
|
|
436
|
+
|
|
437
|
+
# instantiate event builder
|
|
438
|
+
evtbuilder_inst = EventBuilder()
|
|
439
|
+
|
|
440
|
+
# instantiate OF trigger and add to EventBuilder'
|
|
441
|
+
trigger_config = copy.deepcopy(self._trigger_config)
|
|
442
|
+
nb_trigger_chans = len(list(trigger_config.keys()))
|
|
443
|
+
overlap_left_samples = 0
|
|
444
|
+
overlap_right_samples = 0
|
|
445
|
+
max_pileup_samples = 0
|
|
446
|
+
for trig_chan, trig_data in trigger_config.items():
|
|
447
|
+
|
|
448
|
+
# channel name
|
|
449
|
+
channel_name = trig_data['channel_name']
|
|
450
|
+
|
|
451
|
+
# get template
|
|
452
|
+
template_tag = 'default'
|
|
453
|
+
if 'template_tag' in trig_data:
|
|
454
|
+
template_tag = trig_data['template_tag']
|
|
455
|
+
|
|
456
|
+
template, template_metadata = (
|
|
457
|
+
self._processing_data_inst.get_template(
|
|
458
|
+
channel_name,
|
|
459
|
+
tag=template_tag)
|
|
460
|
+
)
|
|
461
|
+
|
|
462
|
+
nb_pretrigger_samples = None
|
|
463
|
+
if 'nb_pretrigger_samples' in template_metadata.keys():
|
|
464
|
+
nb_pretrigger_samples = (
|
|
465
|
+
template_metadata['nb_pretrigger_samples']
|
|
466
|
+
)
|
|
467
|
+
else:
|
|
468
|
+
# back compatibility
|
|
469
|
+
if 'pretrigger_length_samples' in template_metadata.keys():
|
|
470
|
+
nb_pretrigger_samples = (
|
|
471
|
+
template_metadata['pretrigger_length_samples']
|
|
472
|
+
)
|
|
473
|
+
elif 'pretrigger_samples' in template_metadata.keys():
|
|
474
|
+
nb_pretrigger_samples = (
|
|
475
|
+
template_metadata['pretrigger_samples']
|
|
476
|
+
)
|
|
477
|
+
if nb_pretrigger_samples is None:
|
|
478
|
+
raise ValueError('ERROR: Template metadata needs to contain '
|
|
479
|
+
'"nb_pretrigger_samples" value')
|
|
480
|
+
|
|
481
|
+
# Continuous-Zarr partitions are read with overlap. The current
|
|
482
|
+
# OptimumFilterTrigger zeros one template length at each input edge;
|
|
483
|
+
# include that padding plus the trigger-index shift so the logical
|
|
484
|
+
# partition core has no artificial internal-edge deadtime.
|
|
485
|
+
nb_template_samples = int(template.shape[-1])
|
|
486
|
+
trigger_shift = int(nb_pretrigger_samples) - nb_template_samples // 2
|
|
487
|
+
overlap_left_samples = max(
|
|
488
|
+
overlap_left_samples,
|
|
489
|
+
nb_template_samples + max(trigger_shift, 0),
|
|
490
|
+
)
|
|
491
|
+
overlap_right_samples = max(
|
|
492
|
+
overlap_right_samples,
|
|
493
|
+
nb_template_samples + max(-trigger_shift, 0),
|
|
494
|
+
)
|
|
495
|
+
|
|
496
|
+
if trig_data.get('pileup_window_samples') is not None:
|
|
497
|
+
max_pileup_samples = max(
|
|
498
|
+
max_pileup_samples, int(trig_data['pileup_window_samples'])
|
|
499
|
+
)
|
|
500
|
+
elif trig_data.get('pileup_window_msec') is not None:
|
|
501
|
+
max_pileup_samples = max(
|
|
502
|
+
max_pileup_samples,
|
|
503
|
+
convert_length_msec_to_samples(
|
|
504
|
+
trig_data['pileup_window_msec'], self._sample_rate
|
|
505
|
+
),
|
|
506
|
+
)
|
|
507
|
+
|
|
508
|
+
# Get noise spectrum (CSD/PSD)
|
|
509
|
+
csd_tag = 'default'
|
|
510
|
+
if 'csd_tag' in trig_data:
|
|
511
|
+
csd_tag = trig_data['csd_tag']
|
|
512
|
+
|
|
513
|
+
csd, csd_freqs, csd_metadata = (
|
|
514
|
+
self._processing_data_inst.get_noise(
|
|
515
|
+
channel_name,
|
|
516
|
+
tag=csd_tag)
|
|
517
|
+
)
|
|
518
|
+
|
|
519
|
+
# ignored frequency peaks
|
|
520
|
+
frequency_peaks = None
|
|
521
|
+
ignore_harmonics = False
|
|
522
|
+
if 'ignored_frequency_peaks' in trig_data:
|
|
523
|
+
frequency_peaks = trig_data['ignored_frequency_peaks']
|
|
524
|
+
if not isinstance(frequency_peaks, list):
|
|
525
|
+
frequency_peaks = [frequency_peaks]
|
|
526
|
+
if 'ignore_harmonics' in trig_data:
|
|
527
|
+
ignore_harmonics = trig_data['ignore_harmonics']
|
|
528
|
+
|
|
529
|
+
# sample rate
|
|
530
|
+
fs = self._processing_data_inst.get_sample_rate()
|
|
531
|
+
|
|
532
|
+
# instantiate optimal filter trigger
|
|
533
|
+
oftrigger_inst = OptimumFilterTrigger(
|
|
534
|
+
channel_name, fs, template, csd,
|
|
535
|
+
nb_pretrigger_samples,
|
|
536
|
+
ignored_frequency_peaks=frequency_peaks,
|
|
537
|
+
ignore_harmonics=ignore_harmonics,
|
|
538
|
+
trigger_name=trig_chan
|
|
539
|
+
)
|
|
540
|
+
|
|
541
|
+
# add in EventBuilder
|
|
542
|
+
evtbuilder_inst.add_trigger_object(
|
|
543
|
+
trig_chan, oftrigger_inst)
|
|
544
|
+
|
|
545
|
+
edge_samples = 0
|
|
546
|
+
if self._edge_exclusion_msec is not None:
|
|
547
|
+
edge_samples = convert_length_msec_to_samples(
|
|
548
|
+
self._edge_exclusion_msec, fs
|
|
549
|
+
)
|
|
550
|
+
coincident_samples = 0
|
|
551
|
+
if self._evtbuilder_config.get('coincident_window_samples') is not None:
|
|
552
|
+
coincident_samples = int(
|
|
553
|
+
self._evtbuilder_config['coincident_window_samples']
|
|
554
|
+
)
|
|
555
|
+
elif self._evtbuilder_config.get('coincident_window_msec') is not None:
|
|
556
|
+
coincident_samples = convert_length_msec_to_samples(
|
|
557
|
+
self._evtbuilder_config['coincident_window_msec'], fs
|
|
558
|
+
)
|
|
559
|
+
# The physical stream-edge deadtime is set by OF validity/padding and
|
|
560
|
+
# any explicit edge exclusion. Pileup/coincidence context is additional
|
|
561
|
+
# overlap needed only at *internal* partition boundaries and therefore
|
|
562
|
+
# must not be counted as lost livetime.
|
|
563
|
+
stream_edge_left_samples = max(overlap_left_samples, edge_samples)
|
|
564
|
+
stream_edge_right_samples = max(overlap_right_samples, edge_samples)
|
|
565
|
+
merge_context_samples = max(max_pileup_samples, coincident_samples)
|
|
566
|
+
overlap_left_samples = stream_edge_left_samples + merge_context_samples
|
|
567
|
+
overlap_right_samples = stream_edge_right_samples + merge_context_samples
|
|
568
|
+
self._processing_data_inst.configure_trigger_partitions(
|
|
569
|
+
partition_target_duration_s=self._partition_target_duration_s,
|
|
570
|
+
overlap_left_samples=overlap_left_samples,
|
|
571
|
+
overlap_right_samples=overlap_right_samples,
|
|
572
|
+
)
|
|
573
|
+
|
|
574
|
+
trigger_livetime = self._livetime
|
|
575
|
+
if self._catalog.storage_format == 'zarr':
|
|
576
|
+
# With overlapping partitions, internal partition boundaries no
|
|
577
|
+
# longer contribute deadtime. Only the physical beginning/end of
|
|
578
|
+
# each continuous stream remain unavailable with the current OF
|
|
579
|
+
# padding behavior.
|
|
580
|
+
trigger_livetime = 0.0
|
|
581
|
+
for stream_id in self._stream_list:
|
|
582
|
+
stream_catalog = self._catalog.filter(streams=stream_id)
|
|
583
|
+
n_samples = stream_catalog.n_samples
|
|
584
|
+
if n_samples is None:
|
|
585
|
+
continue
|
|
586
|
+
usable = max(
|
|
587
|
+
0,
|
|
588
|
+
int(n_samples)
|
|
589
|
+
- stream_edge_left_samples
|
|
590
|
+
- stream_edge_right_samples,
|
|
591
|
+
)
|
|
592
|
+
trigger_livetime += usable / fs
|
|
593
|
+
if self._verbose and node_num == 1:
|
|
594
|
+
print(
|
|
595
|
+
'INFO: Continuous-Zarr trigger livetime with overlapping '
|
|
596
|
+
f'partitions = {trigger_livetime/60:.3f} minutes'
|
|
597
|
+
)
|
|
598
|
+
|
|
599
|
+
output_file = None
|
|
600
|
+
dataframe_metadata = {}
|
|
601
|
+
if lgc_save:
|
|
602
|
+
output_file = dataframe_group.file_path(node_num)
|
|
603
|
+
dataframe_metadata = dataframe_group.row_metadata(node_num)
|
|
604
|
+
|
|
605
|
+
trigger_counter = 0
|
|
606
|
+
|
|
607
|
+
# intialize output dataframe
|
|
608
|
+
process_df = None
|
|
609
|
+
|
|
610
|
+
# loop streams assigned to this task/worker
|
|
611
|
+
for stream_index, stream in enumerate(stream_list):
|
|
612
|
+
final_stream = (stream_index == len(stream_list) - 1)
|
|
613
|
+
|
|
614
|
+
if self._verbose:
|
|
615
|
+
print('INFO' + node_num_str
|
|
616
|
+
+ ': starting processing stream '
|
|
617
|
+
+ stream)
|
|
618
|
+
|
|
619
|
+
# set file list
|
|
620
|
+
self._processing_data_inst.set_stream(stream)
|
|
621
|
+
|
|
622
|
+
# loop events
|
|
623
|
+
do_stop = False
|
|
624
|
+
while (not do_stop):
|
|
625
|
+
|
|
626
|
+
# -----------------------
|
|
627
|
+
# Check number events
|
|
628
|
+
# and memory usage
|
|
629
|
+
# -----------------------
|
|
630
|
+
ntriggers_limit_reached = (ntriggers>0
|
|
631
|
+
and trigger_counter>=ntriggers)
|
|
632
|
+
|
|
633
|
+
# flag memory limit reached
|
|
634
|
+
memory_usage = 0
|
|
635
|
+
memory_limit_reached = False
|
|
636
|
+
if process_df is not None:
|
|
637
|
+
memory_usage = process_df.shape[0]*process_df.shape[1]*8
|
|
638
|
+
memory_limit_reached = (not lgc_save and memory_usage >= memory_limit)
|
|
639
|
+
|
|
640
|
+
# display
|
|
641
|
+
if self._verbose:
|
|
642
|
+
if (trigger_counter%100==0 and trigger_counter!=0):
|
|
643
|
+
print('INFO' + node_num_str
|
|
644
|
+
+ ': Local number of events = '
|
|
645
|
+
+ str(trigger_counter)
|
|
646
|
+
+ ' (memory = ' + str(memory_usage/1e6) + ' MB)')
|
|
647
|
+
|
|
648
|
+
# -----------------------
|
|
649
|
+
# Read next event
|
|
650
|
+
# -----------------------
|
|
651
|
+
success = self._processing_data_inst.read_next_event(
|
|
652
|
+
channels=self._trigger_channels
|
|
653
|
+
)
|
|
654
|
+
|
|
655
|
+
# end of file or raw data issue
|
|
656
|
+
if not success:
|
|
657
|
+
print('INFO' + node_num_str
|
|
658
|
+
+ ': '
|
|
659
|
+
+ str(trigger_counter)
|
|
660
|
+
+ ' events counted, triggering processing done')
|
|
661
|
+
do_stop = True
|
|
662
|
+
|
|
663
|
+
# -----------------------
|
|
664
|
+
# Handle stop or
|
|
665
|
+
# nb trigger/memory limit
|
|
666
|
+
# reached
|
|
667
|
+
# -----------------------
|
|
668
|
+
|
|
669
|
+
|
|
670
|
+
# let's handle case we need to stop
|
|
671
|
+
# or memory/nb events limit reached
|
|
672
|
+
if ((do_stop and (final_stream or not lgc_save))
|
|
673
|
+
or ntriggers_limit_reached
|
|
674
|
+
or memory_limit_reached):
|
|
675
|
+
|
|
676
|
+
|
|
677
|
+
# case nb triggers reached
|
|
678
|
+
if ntriggers_limit_reached:
|
|
679
|
+
nextra = trigger_counter-ntriggers
|
|
680
|
+
nkeep = len(process_df)-nextra
|
|
681
|
+
if nkeep>0:
|
|
682
|
+
process_df = process_df[0:nkeep]
|
|
683
|
+
|
|
684
|
+
|
|
685
|
+
# save file if needed
|
|
686
|
+
if lgc_save and process_df is not None:
|
|
687
|
+
# One processing task writes exactly one F#### shard.
|
|
688
|
+
try:
|
|
689
|
+
process_df.export_hdf5(output_file, mode='w')
|
|
690
|
+
process_df.close()
|
|
691
|
+
except Exception as e:
|
|
692
|
+
print('WARNING: Export failed with error: ', e)
|
|
693
|
+
print('Will try again...')
|
|
694
|
+
df_pandas = process_df.to_pandas_df()
|
|
695
|
+
process_df = vx.from_pandas(df_pandas, copy_index=False)
|
|
696
|
+
process_df.export_hdf5(output_file, mode='w')
|
|
697
|
+
process_df.close()
|
|
698
|
+
del process_df
|
|
699
|
+
process_df = None
|
|
700
|
+
|
|
701
|
+
|
|
702
|
+
# case maximum number of events reached
|
|
703
|
+
# -> processing done!
|
|
704
|
+
if ntriggers_limit_reached:
|
|
705
|
+
if self._verbose:
|
|
706
|
+
print('INFO' + node_num_str
|
|
707
|
+
+ ': Requested nb events reached. '
|
|
708
|
+
+ 'Stopping processing!')
|
|
709
|
+
|
|
710
|
+
return process_df
|
|
711
|
+
|
|
712
|
+
# case memory limit reached
|
|
713
|
+
# -> processing needs to stop!
|
|
714
|
+
if lgc_output and memory_limit_reached:
|
|
715
|
+
raise ValueError(
|
|
716
|
+
'ERROR: memory limit reached! '
|
|
717
|
+
+ 'Change memory limit or only save hdf5 files '
|
|
718
|
+
+'(lgc_save=True AND lgc_output=False) '
|
|
719
|
+
)
|
|
720
|
+
|
|
721
|
+
|
|
722
|
+
# check if stop
|
|
723
|
+
if do_stop:
|
|
724
|
+
break
|
|
725
|
+
|
|
726
|
+
|
|
727
|
+
# -----------------------
|
|
728
|
+
# process triggers
|
|
729
|
+
# -----------------------
|
|
730
|
+
|
|
731
|
+
# clear event
|
|
732
|
+
evtbuilder_inst.clear_event()
|
|
733
|
+
|
|
734
|
+
|
|
735
|
+
# loop trigger channels
|
|
736
|
+
for trig_chan, trig_data in trigger_config.items():
|
|
737
|
+
|
|
738
|
+
# channel
|
|
739
|
+
channel_name = trig_data['channel_name']
|
|
740
|
+
|
|
741
|
+
# get threshold
|
|
742
|
+
threshold = None
|
|
743
|
+
if 'threshold_sigma' in trig_data.keys():
|
|
744
|
+
threshold = float(trig_data['threshold_sigma'])
|
|
745
|
+
elif 'threshold' in trig_data.keys():
|
|
746
|
+
threshold = float(trig_data['threshold'])
|
|
747
|
+
else:
|
|
748
|
+
raise ValueError(
|
|
749
|
+
'ERROR: "treshold_sigma" missing in '
|
|
750
|
+
+ 'yaml configuration file')
|
|
751
|
+
|
|
752
|
+
# pileup window
|
|
753
|
+
pileup_window_msec = None
|
|
754
|
+
if 'pileup_window_msec' in trig_data.keys():
|
|
755
|
+
pileup_window_msec = float(
|
|
756
|
+
trig_data['pileup_window_msec']
|
|
757
|
+
)
|
|
758
|
+
|
|
759
|
+
pileup_window_samples = None
|
|
760
|
+
if 'pileup_window_samples' in trig_data.keys():
|
|
761
|
+
pileup_window_samples = int(
|
|
762
|
+
trig_data['pileup_window_samples'])
|
|
763
|
+
|
|
764
|
+
# residual triggering
|
|
765
|
+
run_residual = False
|
|
766
|
+
sat_amps_50kHz = None
|
|
767
|
+
if 'run_residual' in trig_data.keys():
|
|
768
|
+
run_residual = bool(
|
|
769
|
+
trig_data['run_residual'])
|
|
770
|
+
if 'sat_amps_50kHz' in trig_data.keys():
|
|
771
|
+
sat_amps_50kHz = trig_data['sat_amps_50kHz']
|
|
772
|
+
sat_amps_50kHz = [float(amp) for amp in sat_amps_50kHz]
|
|
773
|
+
|
|
774
|
+
# positive pulse
|
|
775
|
+
positive_pulses = True
|
|
776
|
+
if 'positive_pulses' in trig_data.keys():
|
|
777
|
+
positive_pulses = trig_data['positive_pulses']
|
|
778
|
+
|
|
779
|
+
# get trace (If multiple channels, need to follow order)
|
|
780
|
+
trace = self._processing_data_inst.get_channel_trace(
|
|
781
|
+
channel_name
|
|
782
|
+
)
|
|
783
|
+
|
|
784
|
+
# acquire trigger
|
|
785
|
+
evtbuilder_inst.acquire_triggers(
|
|
786
|
+
trig_chan,
|
|
787
|
+
trace,
|
|
788
|
+
threshold,
|
|
789
|
+
pileup_window_msec=pileup_window_msec,
|
|
790
|
+
pileup_window_samples=pileup_window_samples,
|
|
791
|
+
positive_pulses=positive_pulses,
|
|
792
|
+
run_residual=run_residual,
|
|
793
|
+
sat_amps_50kHz=sat_amps_50kHz,
|
|
794
|
+
edge_exclusion_msec=self._edge_exclusion_msec,
|
|
795
|
+
livetime=trigger_livetime
|
|
796
|
+
)
|
|
797
|
+
|
|
798
|
+
|
|
799
|
+
# -----------------------
|
|
800
|
+
# build event
|
|
801
|
+
# merge coincident triggers
|
|
802
|
+
# -----------------------
|
|
803
|
+
|
|
804
|
+
coincident_window_msec = None
|
|
805
|
+
if 'coincident_window_msec' in self._evtbuilder_config.keys():
|
|
806
|
+
coincident_window_msec = (
|
|
807
|
+
self._evtbuilder_config['coincident_window_msec'])
|
|
808
|
+
|
|
809
|
+
coincident_window_samples = None
|
|
810
|
+
if 'coincident_window_samples' in self._evtbuilder_config.keys():
|
|
811
|
+
coincident_window_samples = (
|
|
812
|
+
self._evtbuilder_config['coincident_window_samples'])
|
|
813
|
+
|
|
814
|
+
# get event metadata
|
|
815
|
+
event_info = self._processing_data_inst.get_event_admin()
|
|
816
|
+
|
|
817
|
+
# build event
|
|
818
|
+
evtbuilder_inst.build_event(
|
|
819
|
+
event_info,
|
|
820
|
+
fs=fs,
|
|
821
|
+
coincident_window_msec=coincident_window_msec,
|
|
822
|
+
coincident_window_samples=coincident_window_samples,
|
|
823
|
+
nb_trigger_channels=nb_trigger_chans
|
|
824
|
+
)
|
|
825
|
+
|
|
826
|
+
|
|
827
|
+
# get trigger data
|
|
828
|
+
event_df = evtbuilder_inst.get_event_df()
|
|
829
|
+
|
|
830
|
+
# For continuous Zarr, trigger indices produced by the OF are
|
|
831
|
+
# relative to the expanded read window. Keep only triggers
|
|
832
|
+
# belonging to this non-overlapping logical core, then expose
|
|
833
|
+
# canonical stream/partition coordinates. The legacy
|
|
834
|
+
# ``trigger_index`` output is retained as a partition-local
|
|
835
|
+
# compatibility alias.
|
|
836
|
+
partition_context = self._processing_data_inst.get_partition_context()
|
|
837
|
+
if event_df is not None and partition_context is None and len(event_df):
|
|
838
|
+
# For native/legacy HDF5 records the trigger index is
|
|
839
|
+
# segment-local. Expose that coordinate explicitly rather
|
|
840
|
+
# than requiring downstream code to infer it from the
|
|
841
|
+
# generic trigger_index field.
|
|
842
|
+
if event_info.get('global_segment_number') is not None:
|
|
843
|
+
event_df['segment_trigger_index'] = np.asarray(
|
|
844
|
+
event_df['trigger_index'].values, dtype=np.int64
|
|
845
|
+
)
|
|
846
|
+
if event_df is not None and partition_context is not None:
|
|
847
|
+
local_indices = np.asarray(event_df['trigger_index'].values, dtype=np.int64)
|
|
848
|
+
stream_indices = (
|
|
849
|
+
local_indices
|
|
850
|
+
+ int(partition_context['partition_read_start_index'])
|
|
851
|
+
)
|
|
852
|
+
core_start = int(partition_context['partition_start_index'])
|
|
853
|
+
core_stop = core_start + int(
|
|
854
|
+
partition_context['partition_length_samples']
|
|
855
|
+
)
|
|
856
|
+
keep = (stream_indices >= core_start) & (stream_indices < core_stop)
|
|
857
|
+
if not np.all(keep):
|
|
858
|
+
event_pd = event_df.to_pandas_df()
|
|
859
|
+
event_pd = event_pd.loc[keep].reset_index(drop=True)
|
|
860
|
+
event_df = vx.from_pandas(event_pd, copy_index=False)
|
|
861
|
+
stream_indices = stream_indices[keep]
|
|
862
|
+
if len(event_df):
|
|
863
|
+
partition_indices = stream_indices - core_start
|
|
864
|
+
event_df['stream_trigger_index'] = stream_indices
|
|
865
|
+
event_df['partition_trigger_index'] = partition_indices
|
|
866
|
+
event_df['partition_start_index'] = np.full(
|
|
867
|
+
len(event_df), core_start, dtype=np.int64
|
|
868
|
+
)
|
|
869
|
+
event_df['partition_length_samples'] = np.full(
|
|
870
|
+
len(event_df),
|
|
871
|
+
int(partition_context['partition_length_samples']),
|
|
872
|
+
dtype=np.int64,
|
|
873
|
+
)
|
|
874
|
+
event_df['stream_length_samples'] = np.full(
|
|
875
|
+
len(event_df),
|
|
876
|
+
int(partition_context['stream_length_samples']),
|
|
877
|
+
dtype=np.int64,
|
|
878
|
+
)
|
|
879
|
+
event_df['trigger_index'] = partition_indices
|
|
880
|
+
event_df['trigger_time'] = partition_indices / fs
|
|
881
|
+
|
|
882
|
+
# EventBuilder assumes non-overlapping input records
|
|
883
|
+
# when constructing event times. Recompute these legacy
|
|
884
|
+
# timing fields from the canonical stream-global index
|
|
885
|
+
# for overlapped Zarr partitions.
|
|
886
|
+
stream_start = event_info.get('stream_start_time')
|
|
887
|
+
if stream_start is not None and np.isfinite(float(stream_start)):
|
|
888
|
+
event_times_float = (
|
|
889
|
+
float(stream_start) + stream_indices / fs
|
|
890
|
+
)
|
|
891
|
+
event_df['event_time'] = np.rint(
|
|
892
|
+
event_times_float
|
|
893
|
+
).astype(np.int64)
|
|
894
|
+
event_df['time_since_stream_start_s'] = (
|
|
895
|
+
stream_indices / fs
|
|
896
|
+
)
|
|
897
|
+
acquisition_start = event_info.get(
|
|
898
|
+
'acquisition_start_time'
|
|
899
|
+
)
|
|
900
|
+
if (acquisition_start is not None
|
|
901
|
+
and np.isfinite(float(acquisition_start))):
|
|
902
|
+
event_df['time_since_acquisition_start_s'] = (
|
|
903
|
+
event_times_float - float(acquisition_start)
|
|
904
|
+
)
|
|
905
|
+
fridge_start = event_info.get('fridge_run_start_time')
|
|
906
|
+
if (fridge_start is not None
|
|
907
|
+
and np.isfinite(float(fridge_start))):
|
|
908
|
+
event_df['time_since_fridge_run_start_s'] = (
|
|
909
|
+
event_times_float - float(fridge_start)
|
|
910
|
+
)
|
|
911
|
+
|
|
912
|
+
# check if triggers
|
|
913
|
+
if (event_df is None or len(event_df)==0):
|
|
914
|
+
continue
|
|
915
|
+
|
|
916
|
+
|
|
917
|
+
# increment counter
|
|
918
|
+
nb_triggers = len(event_df)
|
|
919
|
+
|
|
920
|
+
trigger_counter += nb_triggers
|
|
921
|
+
|
|
922
|
+
|
|
923
|
+
# -----------------------
|
|
924
|
+
# Add metadata
|
|
925
|
+
# -----------------------
|
|
926
|
+
|
|
927
|
+
# add acquisition/stream provenance needed by feature
|
|
928
|
+
# processing. Keep legacy EventBuilder columns untouched.
|
|
929
|
+
event_df['acquisition_name'] = np.array(
|
|
930
|
+
[self._acquisition_name] * nb_triggers
|
|
931
|
+
)
|
|
932
|
+
if 'stream_id' in event_info:
|
|
933
|
+
event_df['stream_id'] = np.array(
|
|
934
|
+
[str(event_info['stream_id'])] * nb_triggers
|
|
935
|
+
)
|
|
936
|
+
if 'stream_number' in event_info:
|
|
937
|
+
event_df['stream_number'] = np.array(
|
|
938
|
+
[np.int64(event_info['stream_number'])] * nb_triggers
|
|
939
|
+
)
|
|
940
|
+
|
|
941
|
+
# Canonical processed-dataframe identity. The same group ID is
|
|
942
|
+
# shared by every worker; only dataframe_file_index differs.
|
|
943
|
+
for key, value in dataframe_metadata.items():
|
|
944
|
+
event_df[key] = np.array([value] * nb_triggers)
|
|
945
|
+
|
|
946
|
+
# done processing event!
|
|
947
|
+
# append event dictionary to dataframe
|
|
948
|
+
if process_df is None:
|
|
949
|
+
process_df = event_df
|
|
950
|
+
else:
|
|
951
|
+
process_df = vx.concat([process_df, event_df])
|
|
952
|
+
|
|
953
|
+
|
|
954
|
+
|
|
955
|
+
# cleanup
|
|
956
|
+
del evtbuilder_inst
|
|
957
|
+
|
|
958
|
+
# return features
|
|
959
|
+
return process_df
|
|
960
|
+
|
|
961
|
+
|
|
962
|
+
def _create_output_directory(self, base_path, facility,
|
|
963
|
+
output_group_name=None,
|
|
964
|
+
restricted=False,
|
|
965
|
+
data_type='background'):
|
|
966
|
+
"""Compatibility wrapper around the common dataframe-group helper."""
|
|
967
|
+
group = create_dataframe_group(
|
|
968
|
+
base_path, dataframe_type='trigger', facility=facility,
|
|
969
|
+
processing_label=self._processing_label, restricted=restricted,
|
|
970
|
+
data_type=data_type, output_group_name=output_group_name,
|
|
971
|
+
)
|
|
972
|
+
return group.path, group.group_number
|
|
973
|
+
|
|
974
|
+
|
|
975
|
+
def _split_streams(self, ncores):
|
|
976
|
+
"""
|
|
977
|
+
Split raw streams/tasks between workers
|
|
978
|
+
|
|
979
|
+
|
|
980
|
+
Parameters
|
|
981
|
+
----------
|
|
982
|
+
|
|
983
|
+
ncores : int
|
|
984
|
+
number of cores
|
|
985
|
+
|
|
986
|
+
Return
|
|
987
|
+
------
|
|
988
|
+
|
|
989
|
+
output_list : list
|
|
990
|
+
list of dictionaries (length=ncores) containing
|
|
991
|
+
data
|
|
992
|
+
|
|
993
|
+
|
|
994
|
+
"""
|
|
995
|
+
|
|
996
|
+
output_list = list()
|
|
997
|
+
|
|
998
|
+
# split streams
|
|
999
|
+
stream_split = np.array_split(self._stream_list, ncores)
|
|
1000
|
+
|
|
1001
|
+
# remove empty array
|
|
1002
|
+
for stream_sublist in stream_split:
|
|
1003
|
+
if stream_sublist.size == 0:
|
|
1004
|
+
continue
|
|
1005
|
+
output_list.append(list(stream_sublist))
|
|
1006
|
+
|
|
1007
|
+
|
|
1008
|
+
return output_list
|
|
1009
|
+
|
|
1010
|
+
|
|
1011
|
+
|