pytesprocess 0.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. pytesprocess/__init__.py +9 -0
  2. pytesprocess/_version.py +2 -0
  3. pytesprocess/cli/__init__.py +1 -0
  4. pytesprocess/cli/commands/__init__.py +5 -0
  5. pytesprocess/cli/commands/event.py +66 -0
  6. pytesprocess/cli/commands/filter.py +17 -0
  7. pytesprocess/cli/commands/ivsweep.py +29 -0
  8. pytesprocess/cli/common.py +86 -0
  9. pytesprocess/cli/main.py +81 -0
  10. pytesprocess/config/__init__.py +4 -0
  11. pytesprocess/config/loader.py +94 -0
  12. pytesprocess/config/manager.py +297 -0
  13. pytesprocess/config/resolvers/__init__.py +5 -0
  14. pytesprocess/config/resolvers/common.py +56 -0
  15. pytesprocess/config/resolvers/feature.py +293 -0
  16. pytesprocess/config/resolvers/salting.py +86 -0
  17. pytesprocess/config/resolvers/trigger.py +84 -0
  18. pytesprocess/config/selectors.py +108 -0
  19. pytesprocess/config/validation.py +314 -0
  20. pytesprocess/config/warnings.py +2 -0
  21. pytesprocess/core/__init__.py +10 -0
  22. pytesprocess/core/algorithms.py +1455 -0
  23. pytesprocess/core/didv.py +1648 -0
  24. pytesprocess/core/eventbuilder.py +495 -0
  25. pytesprocess/core/filterbuilder.py +81 -0
  26. pytesprocess/core/filterdata.py +1849 -0
  27. pytesprocess/core/ivsweep.py +2072 -0
  28. pytesprocess/core/noise.py +923 -0
  29. pytesprocess/core/noisemodel.py +1408 -0
  30. pytesprocess/core/oftrigger.py +1035 -0
  31. pytesprocess/core/template.py +450 -0
  32. pytesprocess/process/__init__.py +6 -0
  33. pytesprocess/process/data_source.py +185 -0
  34. pytesprocess/process/event_context.py +35 -0
  35. pytesprocess/process/feature_plan.py +186 -0
  36. pytesprocess/process/feature_resources.py +267 -0
  37. pytesprocess/process/features.py +1024 -0
  38. pytesprocess/process/filterprocess.py +1176 -0
  39. pytesprocess/process/ivprocess.py +1380 -0
  40. pytesprocess/process/processing_data.py +967 -0
  41. pytesprocess/process/randoms.py +921 -0
  42. pytesprocess/process/triggers.py +1011 -0
  43. pytesprocess/salting/__init__.py +7 -0
  44. pytesprocess/salting/generator.py +364 -0
  45. pytesprocess/salting/injector.py +329 -0
  46. pytesprocess/salting/sampling.py +84 -0
  47. pytesprocess/utils/__init__.py +5 -0
  48. pytesprocess/utils/arg_utils.py +122 -0
  49. pytesprocess/utils/dataframe_output.py +120 -0
  50. pytesprocess/utils/filter_hdf5.py +594 -0
  51. pytesprocess/utils/utils.py +701 -0
  52. pytesprocess/workflows/__init__.py +3 -0
  53. pytesprocess/workflows/processing.py +317 -0
  54. pytesprocess/workflows/salting.py +133 -0
  55. pytesprocess-0.1.1.dist-info/METADATA +211 -0
  56. pytesprocess-0.1.1.dist-info/RECORD +60 -0
  57. pytesprocess-0.1.1.dist-info/WHEEL +5 -0
  58. pytesprocess-0.1.1.dist-info/entry_points.txt +2 -0
  59. pytesprocess-0.1.1.dist-info/licenses/LICENSE +21 -0
  60. pytesprocess-0.1.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,921 @@
1
+ """Random-trigger generation for continuous/segmented raw data.
2
+
3
+ This refactor keeps the public Randoms API close to the historical version,
4
+ while separating storage-region discovery, count allocation, and random index
5
+ sampling.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import copy
11
+ from dataclasses import dataclass
12
+ from math import ceil
13
+ from pathlib import Path
14
+ import random
15
+ import re
16
+
17
+ import numpy as np
18
+ import vaex as vx
19
+
20
+ from pytesdaqx.io import AcquisitionCatalog
21
+ from pytesprocess.utils import create_dataframe_group
22
+
23
+
24
+ __all__ = ["Randoms"]
25
+
26
+
27
+ @dataclass(frozen=True)
28
+ class _SelectionRegion:
29
+ """One independent interval from which random triggers may be selected."""
30
+
31
+ region_id: int
32
+ stream_id: str
33
+ start_index: int
34
+ length_samples: int
35
+ usable_start_index: int
36
+ usable_stop_index: int # exclusive, local to the region
37
+ usable_samples: int
38
+ capacity: int
39
+
40
+ # HDF5 segment metadata
41
+ dump_num: int | None = None
42
+ global_segment_num: int | None = None
43
+
44
+ # Zarr processing-partition metadata
45
+ partition_index: int | None = None
46
+ partition_start_index: int | None = None
47
+ partition_length_samples: int | None = None
48
+
49
+
50
+ class Randoms:
51
+ """Generate random-trigger metadata from raw-data streams.
52
+
53
+ HDF5 data are sampled independently within each stored segment.
54
+
55
+ Zarr data support two modes:
56
+ * ``partition_target_duration_s`` provided: sample independently within
57
+ logical processing partitions, preserving the historical segment-like
58
+ behavior.
59
+ * ``partition_target_duration_s=None``: sample across each full
60
+ continuous stream, so edge exclusion is applied only at the beginning
61
+ and end of the stream.
62
+ """
63
+
64
+ def __init__(
65
+ self,
66
+ data_path,
67
+ streams=None,
68
+ processing_label=None,
69
+ data_type="background",
70
+ restricted=False,
71
+ verbose=True,
72
+ ):
73
+ self._verbose = verbose
74
+
75
+ self._processing_label = processing_label
76
+ self._restricted = restricted
77
+ self._data_type = data_type
78
+
79
+ catalog_full = AcquisitionCatalog(data_path, verbose=verbose)
80
+ self._catalog = catalog_full.filter(
81
+ streams=streams,
82
+ measurement_types=data_type,
83
+ restricted=restricted,
84
+ )
85
+
86
+ selected_data = self._catalog.select_files_by_stream(
87
+ stream_key="stream_id"
88
+ )
89
+ if not selected_data:
90
+ raise ValueError("No files were found! Check configuration...")
91
+
92
+ self._files_by_stream = copy.deepcopy(selected_data)
93
+ self._streams = list(selected_data.keys())
94
+
95
+ acquisition_path = Path(self._catalog.acquisition_path)
96
+ self._acquisition_name = self._catalog.acquisition_name
97
+ self._input_base_path = str(acquisition_path.parent)
98
+
99
+ match = re.search(r"_I(\d+)", self._acquisition_name)
100
+ if not match:
101
+ raise ValueError(
102
+ "ERROR: No facility found from acquisition name!"
103
+ )
104
+ self._facility = int(match.group(1))
105
+
106
+ self._duration_s = self._catalog.get_duration()
107
+ if self._verbose:
108
+ print(
109
+ "INFO: Total raw data duration = "
110
+ f"{self._duration_s / 60} minutes"
111
+ )
112
+
113
+ self._storage_format = self._catalog.storage_format
114
+ self._sample_rate_hz = self._catalog.sample_rate_hz
115
+
116
+ self._output_path = None
117
+ self._dataframe_group = None
118
+ self._nrandoms = None
119
+ self._randoms_livetime_s = 0.0
120
+
121
+ @property
122
+ def verbose(self):
123
+ return self._verbose
124
+
125
+ @verbose.setter
126
+ def verbose(self, value):
127
+ self._verbose = value
128
+
129
+ def get_base_path(self):
130
+ return self._input_base_path
131
+
132
+ def get_acquisition_name(self):
133
+ return self._acquisition_name
134
+
135
+ def get_output_path(self):
136
+ """Get output acquisition path."""
137
+ return self._output_path
138
+
139
+ def process(
140
+ self,
141
+ random_rate=None,
142
+ nrandoms=None,
143
+ min_separation_msec=0,
144
+ edge_exclusion_msec=0,
145
+ edge_exclusion_samples=None,
146
+ partition_target_duration_s=None,
147
+ random_seed=None,
148
+ lgc_save=False,
149
+ lgc_output=False,
150
+ save_path=None,
151
+ output_dir_name=None,
152
+ ):
153
+ """Generate random triggers.
154
+
155
+ Exactly one of ``random_rate`` or ``nrandoms`` must be provided.
156
+
157
+ Parameters
158
+ ----------
159
+ random_rate : float, optional
160
+ Requested average random-trigger rate in Hz. The total generated
161
+ count is ``round(random_rate * eligible_livetime)``.
162
+ nrandoms : int, optional
163
+ Exact number of random triggers to generate. If the requested
164
+ number cannot satisfy ``min_separation_msec`` within the eligible
165
+ regions, a ValueError is raised.
166
+ min_separation_msec : float
167
+ Minimum separation between triggers selected from the same region.
168
+ For HDF5 a region is one segment. For partitioned Zarr it is one
169
+ logical partition. For whole-stream Zarr it is one stream.
170
+ edge_exclusion_msec : float
171
+ Exclusion applied to both edges of each selection region.
172
+ edge_exclusion_samples : int, optional
173
+ Same as ``edge_exclusion_msec`` but specified directly in samples.
174
+ Takes precedence when provided.
175
+ partition_target_duration_s : float, optional
176
+ For Zarr, use logical processing partitions near this duration.
177
+ If None, select across each entire continuous Zarr stream. HDF5
178
+ always uses stored segments.
179
+ random_seed : int, optional
180
+ Seed for reproducible count allocation and trigger selection.
181
+ lgc_save : bool
182
+ If True, save the generated random-trigger dataframe.
183
+ lgc_output : bool
184
+ If True, return the generated Vaex dataframe.
185
+ save_path : str, optional
186
+ Base output path used when ``lgc_save=True``.
187
+ output_dir_name : str, optional
188
+ Optional output directory name used when ``lgc_save=True``.
189
+ """
190
+
191
+ self._validate_selection_arguments(
192
+ random_rate=random_rate,
193
+ nrandoms=nrandoms,
194
+ min_separation_msec=min_separation_msec,
195
+ edge_exclusion_msec=edge_exclusion_msec,
196
+ edge_exclusion_samples=edge_exclusion_samples,
197
+ )
198
+
199
+ if edge_exclusion_samples is not None:
200
+ edge_exclusion_samples = int(edge_exclusion_samples)
201
+ else:
202
+ edge_exclusion_samples = int(
203
+ ceil(
204
+ float(edge_exclusion_msec)
205
+ * self._sample_rate_hz
206
+ / 1000.0
207
+ )
208
+ )
209
+
210
+ min_separation_samples = int(
211
+ ceil(
212
+ float(min_separation_msec)
213
+ * self._sample_rate_hz
214
+ / 1000.0
215
+ )
216
+ )
217
+
218
+ regions_by_stream = self._build_selection_regions(
219
+ edge_exclusion_samples=edge_exclusion_samples,
220
+ min_separation_samples=min_separation_samples,
221
+ partition_target_duration_s=partition_target_duration_s,
222
+ )
223
+ all_regions = [
224
+ region
225
+ for stream_regions in regions_by_stream.values()
226
+ for region in stream_regions
227
+ ]
228
+
229
+ total_usable_samples = sum(
230
+ region.usable_samples for region in all_regions
231
+ )
232
+ self._randoms_livetime_s = (
233
+ total_usable_samples / float(self._sample_rate_hz)
234
+ )
235
+ if self._verbose:
236
+ print(
237
+ "INFO: Livetime for randoms = "
238
+ f"{self._randoms_livetime_s / 60} minutes"
239
+ )
240
+
241
+ if total_usable_samples <= 0:
242
+ raise ValueError(
243
+ "ERROR: No eligible samples remain after edge exclusion."
244
+ )
245
+
246
+ if nrandoms is not None:
247
+ target_nrandoms = int(nrandoms)
248
+ else:
249
+ target_nrandoms = int(
250
+ round(float(random_rate) * self._randoms_livetime_s)
251
+ )
252
+
253
+ total_capacity = sum(region.capacity for region in all_regions)
254
+ if target_nrandoms > total_capacity:
255
+ max_rate = total_capacity / self._randoms_livetime_s
256
+ raise ValueError(
257
+ "ERROR: Requested random selection is not possible with the "
258
+ "specified edge exclusion and minimum separation. "
259
+ f"Requested={target_nrandoms}, maximum={total_capacity}, "
260
+ f"eligible_livetime_s={self._randoms_livetime_s:.6g}, "
261
+ f"maximum_average_rate_hz={max_rate:.6g}."
262
+ )
263
+
264
+ self._nrandoms = target_nrandoms
265
+
266
+ rng = random.Random(random_seed)
267
+ region_counts = self._allocate_exact_counts(
268
+ all_regions,
269
+ target_nrandoms,
270
+ rng=rng,
271
+ )
272
+
273
+ output_path = None
274
+ dataframe_group = None
275
+ if lgc_save:
276
+ if save_path is None:
277
+ save_path = str(Path(self._input_base_path) / "processed")
278
+ save_path = save_path.replace("/raw/processed", "/processed")
279
+
280
+ if self._acquisition_name not in str(save_path):
281
+ save_path = str(Path(save_path) / self._acquisition_name)
282
+
283
+ dataframe_group = create_dataframe_group(
284
+ save_path, dataframe_type="randoms", facility=self._facility,
285
+ processing_label=self._processing_label,
286
+ restricted=self._restricted, data_type=self._data_type,
287
+ output_group_name=output_dir_name,
288
+ )
289
+ output_path = dataframe_group.path
290
+ if self._verbose:
291
+ print(
292
+ "INFO: Processing output acquisition path: "
293
+ f"{output_path}"
294
+ )
295
+
296
+ self._output_path = output_path
297
+ self._dataframe_group = dataframe_group
298
+
299
+ output_df = self._generate_dataframe(
300
+ regions_by_stream=regions_by_stream,
301
+ region_counts=region_counts,
302
+ min_separation_samples=min_separation_samples,
303
+ edge_exclusion_samples=edge_exclusion_samples,
304
+ partition_target_duration_s=partition_target_duration_s,
305
+ rng=rng,
306
+ dataframe_group=dataframe_group,
307
+ )
308
+
309
+ actual = len(output_df)
310
+ if actual != target_nrandoms:
311
+ raise RuntimeError(
312
+ "ERROR: Internal random-selection count mismatch: "
313
+ f"expected {target_nrandoms}, generated {actual}."
314
+ )
315
+
316
+ if lgc_save:
317
+ self._save_dataframe(output_df, dataframe_group=dataframe_group)
318
+
319
+ if self._verbose:
320
+ print(
321
+ "INFO: Randoms acquisition done! "
322
+ f"Generated {target_nrandoms} randoms."
323
+ )
324
+
325
+ if lgc_output:
326
+ return output_df
327
+ return None
328
+
329
+ def _build_selection_regions(
330
+ self,
331
+ *,
332
+ edge_exclusion_samples,
333
+ min_separation_samples,
334
+ partition_target_duration_s,
335
+ ):
336
+ """Build HDF5-segment or Zarr stream/partition selection regions."""
337
+
338
+ regions_by_stream = {}
339
+ self._stream_length_samples_by_stream = {}
340
+ region_id = 0
341
+
342
+ for stream_id in self._streams:
343
+ catalog_stream = self._catalog.filter(streams=stream_id)
344
+ sample_rate_hz = catalog_stream.sample_rate_hz
345
+ if not np.isclose(sample_rate_hz, self._sample_rate_hz):
346
+ raise ValueError(
347
+ "ERROR: Randoms currently requires one common sample rate "
348
+ "across selected streams."
349
+ )
350
+
351
+ stream_regions = []
352
+
353
+ if self._storage_format == "hdf5":
354
+ stream_start_index = 0
355
+ entries = sorted(
356
+ catalog_stream.entries,
357
+ key=lambda entry: int(entry.get("dump_num", 0)),
358
+ )
359
+
360
+ for entry in entries:
361
+ dump_num = int(entry["dump_num"])
362
+ n_segments = int(entry.get("n_segments") or 0)
363
+ segment_length_samples = entry.get("segment_length_samples")
364
+ if segment_length_samples is None:
365
+ segment_duration_s = entry.get("segment_duration_s")
366
+ if segment_duration_s is None:
367
+ raise ValueError(
368
+ "ERROR: Unable to determine HDF5 segment length "
369
+ f"for dump {dump_num}."
370
+ )
371
+ segment_length_samples = int(
372
+ round(float(segment_duration_s) * sample_rate_hz)
373
+ )
374
+ segment_length_samples = int(segment_length_samples)
375
+
376
+ for segment_num in range(1, n_segments + 1):
377
+ usable_start, usable_stop, usable_samples, capacity = (
378
+ self._region_limits(
379
+ segment_length_samples,
380
+ edge_exclusion_samples,
381
+ min_separation_samples,
382
+ )
383
+ )
384
+ stream_regions.append(
385
+ _SelectionRegion(
386
+ region_id=region_id,
387
+ stream_id=stream_id,
388
+ start_index=stream_start_index,
389
+ length_samples=segment_length_samples,
390
+ usable_start_index=usable_start,
391
+ usable_stop_index=usable_stop,
392
+ usable_samples=usable_samples,
393
+ capacity=capacity,
394
+ dump_num=dump_num,
395
+ global_segment_num=(
396
+ dump_num * 100000 + segment_num
397
+ ),
398
+ )
399
+ )
400
+ region_id += 1
401
+ stream_start_index += segment_length_samples
402
+
403
+ elif self._storage_format == "zarr":
404
+ stream_length_samples = int(catalog_stream.n_samples)
405
+
406
+ if partition_target_duration_s is None:
407
+ usable_start, usable_stop, usable_samples, capacity = (
408
+ self._region_limits(
409
+ stream_length_samples,
410
+ edge_exclusion_samples,
411
+ min_separation_samples,
412
+ )
413
+ )
414
+ stream_regions.append(
415
+ _SelectionRegion(
416
+ region_id=region_id,
417
+ stream_id=stream_id,
418
+ start_index=0,
419
+ length_samples=stream_length_samples,
420
+ usable_start_index=usable_start,
421
+ usable_stop_index=usable_stop,
422
+ usable_samples=usable_samples,
423
+ capacity=capacity,
424
+ )
425
+ )
426
+ region_id += 1
427
+ else:
428
+ stream_plan = catalog_stream.plan_partitions(
429
+ partition_target_duration_s
430
+ )[0]
431
+ partition_length_samples = int(
432
+ stream_plan["partition_length_samples"]
433
+ )
434
+
435
+ for partition in stream_plan["partitions"]:
436
+ partition_index = int(partition["partition_index"])
437
+ partition_start_index = int(
438
+ partition["partition_start_index"]
439
+ )
440
+ (
441
+ usable_start,
442
+ usable_stop,
443
+ usable_samples,
444
+ capacity,
445
+ ) = self._region_limits(
446
+ partition_length_samples,
447
+ edge_exclusion_samples,
448
+ min_separation_samples,
449
+ )
450
+ stream_regions.append(
451
+ _SelectionRegion(
452
+ region_id=region_id,
453
+ stream_id=stream_id,
454
+ start_index=partition_start_index,
455
+ length_samples=partition_length_samples,
456
+ usable_start_index=usable_start,
457
+ usable_stop_index=usable_stop,
458
+ usable_samples=usable_samples,
459
+ capacity=capacity,
460
+ partition_index=partition_index,
461
+ partition_start_index=partition_start_index,
462
+ partition_length_samples=(
463
+ partition_length_samples
464
+ ),
465
+ )
466
+ )
467
+ region_id += 1
468
+ else:
469
+ raise ValueError(
470
+ f'ERROR: Unsupported storage format "{self._storage_format}".'
471
+ )
472
+
473
+ regions_by_stream[stream_id] = stream_regions
474
+ if self._storage_format == "hdf5":
475
+ self._stream_length_samples_by_stream[stream_id] = (
476
+ stream_start_index
477
+ )
478
+ else:
479
+ self._stream_length_samples_by_stream[stream_id] = (
480
+ stream_length_samples
481
+ )
482
+
483
+ return regions_by_stream
484
+
485
+ @staticmethod
486
+ def _region_limits(
487
+ length_samples,
488
+ edge_exclusion_samples,
489
+ min_separation_samples,
490
+ ):
491
+ """Return usable interval and maximum trigger capacity for a region."""
492
+
493
+ length_samples = int(length_samples)
494
+ edge = int(edge_exclusion_samples)
495
+ min_sep = int(min_separation_samples)
496
+
497
+ usable_start = edge
498
+ usable_stop = length_samples - edge
499
+ usable_samples = max(0, usable_stop - usable_start)
500
+
501
+ if usable_samples == 0:
502
+ capacity = 0
503
+ else:
504
+ spacing = max(1, min_sep)
505
+ capacity = 1 + (usable_samples - 1) // spacing
506
+
507
+ return usable_start, usable_stop, usable_samples, capacity
508
+
509
+ @staticmethod
510
+ def _allocate_exact_counts(regions, total_count, *, rng):
511
+ """Allocate exactly ``total_count`` among regions.
512
+
513
+ Allocation is proportional to eligible samples, but each region is
514
+ capped by its minimum-separation capacity. Fractional ties are broken
515
+ randomly so equal-sized partitions are not systematically favored.
516
+ """
517
+
518
+ total_count = int(total_count)
519
+ if total_count < 0:
520
+ raise ValueError("nrandoms must be non-negative")
521
+
522
+ counts = {region.region_id: 0 for region in regions}
523
+ if total_count == 0:
524
+ return counts
525
+
526
+ if total_count > sum(region.capacity for region in regions):
527
+ raise ValueError("Requested count exceeds total region capacity")
528
+
529
+ active = [region for region in regions if region.capacity > 0]
530
+ remaining = total_count
531
+
532
+ while remaining > 0:
533
+ active = [
534
+ region
535
+ for region in active
536
+ if counts[region.region_id] < region.capacity
537
+ ]
538
+ if not active:
539
+ raise RuntimeError("Unable to allocate requested random count")
540
+
541
+ total_weight = sum(region.usable_samples for region in active)
542
+ if total_weight <= 0:
543
+ raise RuntimeError("No eligible samples available for allocation")
544
+
545
+ quotas = {
546
+ region.region_id: (
547
+ remaining * region.usable_samples / total_weight
548
+ )
549
+ for region in active
550
+ }
551
+
552
+ allocated_this_round = 0
553
+ for region in active:
554
+ room = region.capacity - counts[region.region_id]
555
+ add = min(room, int(np.floor(quotas[region.region_id])))
556
+ if add > 0:
557
+ counts[region.region_id] += add
558
+ allocated_this_round += add
559
+
560
+ remaining -= allocated_this_round
561
+ if remaining == 0:
562
+ break
563
+
564
+ active = [
565
+ region
566
+ for region in active
567
+ if counts[region.region_id] < region.capacity
568
+ ]
569
+ if not active:
570
+ raise RuntimeError("Unable to allocate requested random count")
571
+
572
+ # Largest remainder, with random tie breaking for equal regions.
573
+ ranked = sorted(
574
+ active,
575
+ key=lambda region: (
576
+ quotas.get(region.region_id, 0.0)
577
+ - np.floor(quotas.get(region.region_id, 0.0)),
578
+ rng.random(),
579
+ ),
580
+ reverse=True,
581
+ )
582
+
583
+ for region in ranked:
584
+ if remaining == 0:
585
+ break
586
+ if counts[region.region_id] >= region.capacity:
587
+ continue
588
+ counts[region.region_id] += 1
589
+ remaining -= 1
590
+
591
+ return counts
592
+
593
+ @staticmethod
594
+ def _sample_trigger_indices(region, count, min_separation_samples, *, rng):
595
+ """Sample exact local trigger indices with minimum separation."""
596
+
597
+ count = int(count)
598
+ if count == 0:
599
+ return np.empty(0, dtype=np.int64)
600
+ if count > region.capacity:
601
+ raise ValueError(
602
+ f"Requested {count} triggers from region capacity "
603
+ f"{region.capacity}."
604
+ )
605
+
606
+ spacing = max(1, int(min_separation_samples))
607
+ usable_count = int(region.usable_samples)
608
+
609
+ # Compress away the required gaps, sample distinct compressed
610
+ # positions, then reinsert the gaps. This avoids building a potentially
611
+ # enormous list(range(...)) for long continuous streams.
612
+ compressed_count = usable_count - (count - 1) * (spacing - 1)
613
+ if compressed_count < count:
614
+ raise ValueError(
615
+ "Requested random count cannot satisfy minimum separation."
616
+ )
617
+
618
+ reduced = sorted(rng.sample(range(compressed_count), count))
619
+ trigger_indices = np.asarray(reduced, dtype=np.int64)
620
+ trigger_indices += np.arange(count, dtype=np.int64) * (spacing - 1)
621
+ trigger_indices += int(region.usable_start_index)
622
+
623
+ return trigger_indices
624
+
625
+ def _generate_dataframe(
626
+ self,
627
+ *,
628
+ regions_by_stream,
629
+ region_counts,
630
+ min_separation_samples,
631
+ edge_exclusion_samples,
632
+ partition_target_duration_s,
633
+ rng,
634
+ dataframe_group,
635
+ ):
636
+ """Generate one dataframe for all selected streams."""
637
+
638
+ dataframe_metadata = {}
639
+ if dataframe_group is not None:
640
+ dataframe_metadata = dataframe_group.row_metadata(1)
641
+
642
+ feature_dict = self._initialize_feature_dict(
643
+ include_hdf5=(self._storage_format == "hdf5"),
644
+ include_partition=(
645
+ self._storage_format == "zarr"
646
+ and partition_target_duration_s is not None
647
+ ),
648
+ )
649
+
650
+ for stream_id in self._streams:
651
+ stream_regions = regions_by_stream[stream_id]
652
+ n_stream_randoms = sum(
653
+ region_counts[region.region_id]
654
+ for region in stream_regions
655
+ )
656
+ if n_stream_randoms == 0:
657
+ continue
658
+
659
+ if self._verbose:
660
+ print(
661
+ f"INFO: Acquiring {n_stream_randoms} randoms "
662
+ f"for stream {stream_id}"
663
+ )
664
+
665
+ catalog_stream = self._catalog.filter(streams=stream_id)
666
+ sample_rate_hz = float(catalog_stream.sample_rate_hz)
667
+ catalog_entries = catalog_stream.entries
668
+ stream_length_samples = int(
669
+ self._stream_length_samples_by_stream[stream_id]
670
+ )
671
+
672
+ first_entry = catalog_entries[0]
673
+ acquisition_num = first_entry["acquisition_num"]
674
+ acquisition_start = first_entry.get("acquisition_start")
675
+ fridge_run_start = first_entry.get("fridge_run_start")
676
+ fridge_run = first_entry.get("fridge_run")
677
+ stream_start = first_entry.get("stream_start")
678
+ stream_num = first_entry["stream_num"]
679
+
680
+ stream_trigger_id = 0
681
+ for region in stream_regions:
682
+ count = region_counts[region.region_id]
683
+ if count == 0:
684
+ continue
685
+
686
+ trigger_indices = self._sample_trigger_indices(
687
+ region,
688
+ count,
689
+ min_separation_samples,
690
+ rng=rng,
691
+ )
692
+
693
+ for local_trigger_index in trigger_indices:
694
+ local_trigger_index = int(local_trigger_index)
695
+ stream_trigger_id += 1
696
+ stream_trigger_index = (
697
+ region.start_index + local_trigger_index
698
+ )
699
+
700
+ self._append_feature_row(
701
+ feature_dict,
702
+ acquisition_num=acquisition_num,
703
+ stream_num=stream_num,
704
+ fridge_run=fridge_run,
705
+ stream_start=stream_start,
706
+ acquisition_start=acquisition_start,
707
+ fridge_run_start=fridge_run_start,
708
+ stream_trigger_index=stream_trigger_index,
709
+ stream_trigger_id=stream_trigger_id,
710
+ stream_length_samples=stream_length_samples,
711
+ sample_rate_hz=sample_rate_hz,
712
+ min_separation_samples=min_separation_samples,
713
+ edge_exclusion_samples=edge_exclusion_samples,
714
+ dataframe_metadata=dataframe_metadata,
715
+ region=region,
716
+ local_trigger_index=local_trigger_index,
717
+ )
718
+
719
+ return vx.from_dict(feature_dict)
720
+
721
+ def _save_dataframe(self, dataframe, *, dataframe_group):
722
+ """Save the generated random-trigger dataframe as one F0001 shard."""
723
+ dataframe.export_hdf5(dataframe_group.file_path(1), mode="w")
724
+
725
+ def _initialize_feature_dict(self, *, include_hdf5, include_partition):
726
+ feature_dict = {
727
+ "acquisition_number": [],
728
+ "storage_format": [],
729
+ "stream_id": [],
730
+ "stream_number": [],
731
+ "fridge_run_number": [],
732
+ "data_type": [],
733
+ "stream_start": [],
734
+ "acquisition_start": [],
735
+ "fridge_run_start": [],
736
+ "stream_trigger_index": [],
737
+ "stream_trigger_time_s": [],
738
+ "stream_length_samples": [],
739
+ "stream_trigger_id": [],
740
+ "trigger_type": [],
741
+ "randoms_min_separation_time_s": [],
742
+ "randoms_edge_exclusion_time_s": [],
743
+ "randoms_livetime_s": [],
744
+ "dataframe_type": [],
745
+ "dataframe_group_name": [],
746
+ "dataframe_group_id": [],
747
+ "dataframe_group_number": [],
748
+ "dataframe_file_index": [],
749
+ "processing_label": [],
750
+ }
751
+
752
+ if include_hdf5:
753
+ feature_dict.update(
754
+ {
755
+ "dump_number": [],
756
+ "global_segment_number": [],
757
+ "segment_trigger_index": [],
758
+ "segment_trigger_time_s": [],
759
+ "segment_length_samples": [],
760
+ }
761
+ )
762
+
763
+ if include_partition:
764
+ feature_dict.update(
765
+ {
766
+ "partition_start_time_s": [],
767
+ "partition_start_index": [],
768
+ "partition_trigger_time_s": [],
769
+ "partition_trigger_index": [],
770
+ "partition_length_samples": [],
771
+ }
772
+ )
773
+
774
+ return feature_dict
775
+
776
+ def _append_feature_row(
777
+ self,
778
+ feature_dict,
779
+ *,
780
+ acquisition_num,
781
+ stream_num,
782
+ fridge_run,
783
+ stream_start,
784
+ acquisition_start,
785
+ fridge_run_start,
786
+ stream_trigger_index,
787
+ stream_trigger_id,
788
+ stream_length_samples,
789
+ sample_rate_hz,
790
+ min_separation_samples,
791
+ edge_exclusion_samples,
792
+ dataframe_metadata,
793
+ region,
794
+ local_trigger_index,
795
+ ):
796
+ feature_dict["acquisition_number"].append(acquisition_num)
797
+ feature_dict["storage_format"].append(self._storage_format)
798
+ feature_dict["stream_id"].append(region.stream_id)
799
+ feature_dict["stream_number"].append(stream_num)
800
+ feature_dict["fridge_run_number"].append(fridge_run)
801
+ feature_dict["data_type"].append(self._data_type)
802
+ feature_dict["stream_start"].append(stream_start)
803
+ feature_dict["acquisition_start"].append(acquisition_start)
804
+ feature_dict["fridge_run_start"].append(fridge_run_start)
805
+ feature_dict["stream_trigger_index"].append(stream_trigger_index)
806
+ feature_dict["stream_trigger_time_s"].append(
807
+ stream_trigger_index / sample_rate_hz
808
+ )
809
+ feature_dict["stream_length_samples"].append(stream_length_samples)
810
+ feature_dict["stream_trigger_id"].append(stream_trigger_id)
811
+ feature_dict["trigger_type"].append(3)
812
+ feature_dict["randoms_min_separation_time_s"].append(
813
+ min_separation_samples / sample_rate_hz
814
+ )
815
+ feature_dict["randoms_edge_exclusion_time_s"].append(
816
+ edge_exclusion_samples / sample_rate_hz
817
+ )
818
+ feature_dict["randoms_livetime_s"].append(
819
+ self._randoms_livetime_s
820
+ )
821
+ feature_dict["dataframe_type"].append(
822
+ dataframe_metadata.get("dataframe_type", np.nan)
823
+ )
824
+ feature_dict["dataframe_group_name"].append(
825
+ dataframe_metadata.get("dataframe_group_name", np.nan)
826
+ )
827
+ feature_dict["dataframe_group_id"].append(
828
+ dataframe_metadata.get("dataframe_group_id", np.nan)
829
+ )
830
+ feature_dict["dataframe_group_number"].append(
831
+ dataframe_metadata.get("dataframe_group_number", np.nan)
832
+ )
833
+ feature_dict["dataframe_file_index"].append(
834
+ dataframe_metadata.get("dataframe_file_index", np.nan)
835
+ )
836
+ feature_dict["processing_label"].append(
837
+ dataframe_metadata.get("processing_label", None)
838
+ )
839
+
840
+ if self._storage_format == "hdf5":
841
+ feature_dict["dump_number"].append(region.dump_num)
842
+ feature_dict["global_segment_number"].append(
843
+ region.global_segment_num
844
+ )
845
+ feature_dict["segment_trigger_index"].append(
846
+ local_trigger_index
847
+ )
848
+ feature_dict["segment_trigger_time_s"].append(
849
+ local_trigger_index / sample_rate_hz
850
+ )
851
+ feature_dict["segment_length_samples"].append(
852
+ region.length_samples
853
+ )
854
+
855
+ elif region.partition_start_index is not None:
856
+ feature_dict["partition_start_time_s"].append(
857
+ region.partition_start_index / sample_rate_hz
858
+ )
859
+ feature_dict["partition_start_index"].append(
860
+ region.partition_start_index
861
+ )
862
+ feature_dict["partition_trigger_index"].append(
863
+ local_trigger_index
864
+ )
865
+ feature_dict["partition_trigger_time_s"].append(
866
+ local_trigger_index / sample_rate_hz
867
+ )
868
+ feature_dict["partition_length_samples"].append(
869
+ region.partition_length_samples
870
+ )
871
+
872
+ @staticmethod
873
+ def _validate_selection_arguments(
874
+ *,
875
+ random_rate,
876
+ nrandoms,
877
+ min_separation_msec,
878
+ edge_exclusion_msec,
879
+ edge_exclusion_samples,
880
+ ):
881
+ if (random_rate is None) == (nrandoms is None):
882
+ raise ValueError(
883
+ 'ERROR: Use exactly one of "random_rate" or "nrandoms".'
884
+ )
885
+
886
+ if random_rate is not None and float(random_rate) <= 0:
887
+ raise ValueError("ERROR: random_rate must be positive.")
888
+
889
+ if nrandoms is not None:
890
+ if int(nrandoms) != nrandoms or int(nrandoms) < 0:
891
+ raise ValueError(
892
+ "ERROR: nrandoms must be a non-negative integer."
893
+ )
894
+
895
+ if float(min_separation_msec) < 0:
896
+ raise ValueError(
897
+ "ERROR: min_separation_msec must be non-negative."
898
+ )
899
+ if float(edge_exclusion_msec) < 0:
900
+ raise ValueError(
901
+ "ERROR: edge_exclusion_msec must be non-negative."
902
+ )
903
+ if (
904
+ edge_exclusion_samples is not None
905
+ and int(edge_exclusion_samples) < 0
906
+ ):
907
+ raise ValueError(
908
+ "ERROR: edge_exclusion_samples must be non-negative."
909
+ )
910
+
911
+ def _create_output_directory(
912
+ self, base_path, facility, output_dir_name=None, restricted=False,
913
+ data_type="background",
914
+ ):
915
+ """Compatibility wrapper around the common dataframe-group helper."""
916
+ group = create_dataframe_group(
917
+ base_path, dataframe_type="randoms", facility=facility,
918
+ processing_label=self._processing_label, restricted=restricted,
919
+ data_type=data_type, output_group_name=output_dir_name,
920
+ )
921
+ return group.path, group.group_number