pytesprocess 0.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. pytesprocess/__init__.py +9 -0
  2. pytesprocess/_version.py +2 -0
  3. pytesprocess/cli/__init__.py +1 -0
  4. pytesprocess/cli/commands/__init__.py +5 -0
  5. pytesprocess/cli/commands/event.py +66 -0
  6. pytesprocess/cli/commands/filter.py +17 -0
  7. pytesprocess/cli/commands/ivsweep.py +29 -0
  8. pytesprocess/cli/common.py +86 -0
  9. pytesprocess/cli/main.py +81 -0
  10. pytesprocess/config/__init__.py +4 -0
  11. pytesprocess/config/loader.py +94 -0
  12. pytesprocess/config/manager.py +297 -0
  13. pytesprocess/config/resolvers/__init__.py +5 -0
  14. pytesprocess/config/resolvers/common.py +56 -0
  15. pytesprocess/config/resolvers/feature.py +293 -0
  16. pytesprocess/config/resolvers/salting.py +86 -0
  17. pytesprocess/config/resolvers/trigger.py +84 -0
  18. pytesprocess/config/selectors.py +108 -0
  19. pytesprocess/config/validation.py +314 -0
  20. pytesprocess/config/warnings.py +2 -0
  21. pytesprocess/core/__init__.py +10 -0
  22. pytesprocess/core/algorithms.py +1455 -0
  23. pytesprocess/core/didv.py +1648 -0
  24. pytesprocess/core/eventbuilder.py +495 -0
  25. pytesprocess/core/filterbuilder.py +81 -0
  26. pytesprocess/core/filterdata.py +1849 -0
  27. pytesprocess/core/ivsweep.py +2072 -0
  28. pytesprocess/core/noise.py +923 -0
  29. pytesprocess/core/noisemodel.py +1408 -0
  30. pytesprocess/core/oftrigger.py +1035 -0
  31. pytesprocess/core/template.py +450 -0
  32. pytesprocess/process/__init__.py +6 -0
  33. pytesprocess/process/data_source.py +185 -0
  34. pytesprocess/process/event_context.py +35 -0
  35. pytesprocess/process/feature_plan.py +186 -0
  36. pytesprocess/process/feature_resources.py +267 -0
  37. pytesprocess/process/features.py +1024 -0
  38. pytesprocess/process/filterprocess.py +1176 -0
  39. pytesprocess/process/ivprocess.py +1380 -0
  40. pytesprocess/process/processing_data.py +967 -0
  41. pytesprocess/process/randoms.py +921 -0
  42. pytesprocess/process/triggers.py +1011 -0
  43. pytesprocess/salting/__init__.py +7 -0
  44. pytesprocess/salting/generator.py +364 -0
  45. pytesprocess/salting/injector.py +329 -0
  46. pytesprocess/salting/sampling.py +84 -0
  47. pytesprocess/utils/__init__.py +5 -0
  48. pytesprocess/utils/arg_utils.py +122 -0
  49. pytesprocess/utils/dataframe_output.py +120 -0
  50. pytesprocess/utils/filter_hdf5.py +594 -0
  51. pytesprocess/utils/utils.py +701 -0
  52. pytesprocess/workflows/__init__.py +3 -0
  53. pytesprocess/workflows/processing.py +317 -0
  54. pytesprocess/workflows/salting.py +133 -0
  55. pytesprocess-0.1.1.dist-info/METADATA +211 -0
  56. pytesprocess-0.1.1.dist-info/RECORD +60 -0
  57. pytesprocess-0.1.1.dist-info/WHEEL +5 -0
  58. pytesprocess-0.1.1.dist-info/entry_points.txt +2 -0
  59. pytesprocess-0.1.1.dist-info/licenses/LICENSE +21 -0
  60. pytesprocess-0.1.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1380 @@
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import dataclass
4
+ from datetime import datetime
5
+ import hashlib
6
+ from itertools import repeat
7
+ from multiprocessing import Pool
8
+ from pathlib import Path
9
+ import numpy as np
10
+ import pandas as pd
11
+ import qetpy as qp
12
+
13
+ from pytesdaqx.io import AcquisitionCatalog, StreamReader
14
+ from pytesprocess.core import FilterData
15
+ from pytesprocess.process.randoms import Randoms
16
+ from pytesprocess.utils import find_linear_segment
17
+
18
+
19
+ __all__ = ["IVSweepProcessing"]
20
+
21
+
22
+ @dataclass(frozen=True, slots=True)
23
+ class _SweepPoint:
24
+ """Storage-independent description of one IV or dIdV sweep stream."""
25
+
26
+ acquisition_path: str
27
+ acquisition_name: str
28
+ measurement_type: str
29
+ stream_id: str
30
+ stream_num: int | None
31
+ stream_name: str | None
32
+ storage_format: str
33
+ sample_rate_hz: float
34
+ bias_ua: float
35
+ scan_index: int | None
36
+ sequence_index: int | None
37
+ sequence_repeat_index: int | None
38
+ adc_mode: str | None
39
+ raw_shape_model: str | None
40
+ duration_s: float
41
+ n_samples: int | None
42
+ hdf5_dump_segments: tuple[tuple[int, int, int], ...] = ()
43
+ trace_length_samples: int | None = None
44
+ n_traces: int | None = None
45
+
46
+
47
+ @dataclass(frozen=True, slots=True)
48
+ class _SweepPair:
49
+ iv: _SweepPoint
50
+ didv: _SweepPoint
51
+ bias_delta_ua: float
52
+ match_method: str
53
+
54
+
55
+ class IVSweepProcessing:
56
+ """Process IV/dIdV sweeps from HDF5 and Zarr acquisitions.
57
+
58
+ Sweep discovery is catalog-driven. New acquisitions use explicit
59
+ ``scan`` metadata when available; legacy acquisitions fall back to
60
+ identifying detector channels whose TES DC bias changes across streams.
61
+
62
+ IV trace handling is storage independent:
63
+ * legacy HDF5 with native records equal to the requested IV trace length
64
+ is read directly;
65
+ * longer HDF5 segments are sampled with :class:`Randoms` and read with
66
+ triggered ``StreamReader.read_records`` calls;
67
+ * continuous Zarr streams are sampled across the complete stream (no
68
+ processing partition required) and read the same way.
69
+
70
+ dIdV processing continues to use native finite records.
71
+ """
72
+
73
+ def __init__(
74
+ self,
75
+ data_paths,
76
+ processing_label=None,
77
+ bias_tolerance_percent=0.1,
78
+ bias_tolerance_ua=0.001,
79
+ verbose=True,
80
+ ):
81
+ self._processing_label = processing_label
82
+ self._verbose = bool(verbose)
83
+ self._bias_tolerance_percent = float(bias_tolerance_percent)
84
+ self._bias_tolerance_ua = float(bias_tolerance_ua)
85
+
86
+ discovery = self._discover_sweep_data(data_paths)
87
+ self._raw_data_dict = discovery["data"]
88
+ self._acquisition_name_iv = discovery["acquisition_name_iv"]
89
+ self._acquisition_name_didv = discovery["acquisition_name_didv"]
90
+ self._base_path_iv = discovery["base_path_iv"]
91
+ self._base_path_didv = discovery["base_path_didv"]
92
+
93
+ self.describe()
94
+ self._filter_data = FilterData()
95
+
96
+ def describe(self):
97
+ """Describe available IV/dIdV sweep points."""
98
+
99
+ print("\nIV/dIdV sweep available data:")
100
+ for channel, channel_data in self._raw_data_dict.items():
101
+ print(f"\n{channel}:")
102
+
103
+ iv_points = channel_data.get("IV") or []
104
+ didv_points = channel_data.get("dIdV") or []
105
+ common_points = channel_data.get("IV_dIdV") or []
106
+
107
+ if iv_points:
108
+ print(f" -IV: {len(iv_points)} bias points")
109
+ if didv_points:
110
+ print(f" -dIdV: {len(didv_points)} bias points")
111
+ if common_points:
112
+ print(f" -Common IV-dIdV: {len(common_points)} bias points")
113
+ elif iv_points and didv_points:
114
+ print(" -Common IV-dIdV: No bias points")
115
+
116
+ def process(
117
+ self,
118
+ channels=None,
119
+ enable_iv=True,
120
+ enable_didv=True,
121
+ trace_length_iv_msec=100,
122
+ nrandoms_iv=None,
123
+ min_separation_iv_msec=None,
124
+ random_seed=None,
125
+ lgc_output=True,
126
+ lgc_save=False,
127
+ save_path=None,
128
+ ncores=1,
129
+ ):
130
+ """Process selected IV/dIdV sweep channels.
131
+
132
+ Parameters
133
+ ----------
134
+ channels : str or list[str], optional
135
+ Detector channels. ``None`` processes every discovered sweep
136
+ channel.
137
+ enable_iv, enable_didv : bool
138
+ Enable IV and/or dIdV processing.
139
+ trace_length_iv_msec : float
140
+ IV noise trace length. The default is 100 ms, matching the legacy
141
+ HDF5 IV segment length.
142
+ nrandoms_iv : int, optional
143
+ Number of randomly selected IV traces per continuous bias-point
144
+ stream. ``None`` uses the maximum count allowed by the edge and
145
+ minimum-separation constraints, which approximates using all
146
+ available data without overlap.
147
+ min_separation_iv_msec : float, optional
148
+ Minimum trigger separation for randomly selected continuous IV
149
+ traces. ``None`` defaults to ``trace_length_iv_msec`` so selected
150
+ traces do not overlap.
151
+ random_seed : int, optional
152
+ Base seed for reproducible IV random selection. A stable per-stream
153
+ seed is derived from this value so results do not depend on the
154
+ multiprocessing split.
155
+ lgc_save : bool
156
+ Save the resulting FilterData HDF5 file.
157
+ lgc_output : bool
158
+ Return ``dict[channel, pandas.DataFrame]``.
159
+ save_path : str, optional
160
+ Output directory base.
161
+ ncores : int
162
+ Number of processes used across bias points for each channel.
163
+ """
164
+
165
+ if not enable_iv and not enable_didv:
166
+ raise ValueError("ERROR: You need to enable IV or dIdV!")
167
+ if float(trace_length_iv_msec) <= 0:
168
+ raise ValueError("trace_length_iv_msec must be positive")
169
+ if int(ncores) <= 0:
170
+ raise ValueError("ncores must be a positive integer")
171
+ if nrandoms_iv is not None and int(nrandoms_iv) <= 0:
172
+ raise ValueError("nrandoms_iv must be positive when provided")
173
+ if min_separation_iv_msec is None:
174
+ min_separation_iv_msec = float(trace_length_iv_msec)
175
+ if float(min_separation_iv_msec) < float(trace_length_iv_msec):
176
+ if self._verbose:
177
+ print(
178
+ "WARNING: min_separation_iv_msec is shorter than the IV "
179
+ "trace length, so randomly selected IV traces may overlap."
180
+ )
181
+
182
+ if channels is None:
183
+ channels = list(self._raw_data_dict)
184
+ if not channels:
185
+ raise ValueError("ERROR: No channels available!")
186
+ elif isinstance(channels, str):
187
+ channels = [channels]
188
+ else:
189
+ channels = list(channels)
190
+
191
+ for channel in channels:
192
+ if channel not in self._raw_data_dict:
193
+ raise ValueError(f"ERROR: channel {channel} not available!")
194
+
195
+ output_dict = {}
196
+
197
+ for channel in channels:
198
+ if self._verbose:
199
+ print(f"INFO: Channel {channel} IV and/or dIdV processing")
200
+
201
+ channel_data = self._raw_data_dict[channel]
202
+ channel_enable_iv = bool(enable_iv and channel_data.get("IV"))
203
+ channel_enable_didv = bool(enable_didv and channel_data.get("dIdV"))
204
+
205
+ if not channel_enable_iv and not channel_enable_didv:
206
+ raise ValueError(
207
+ f"ERROR: No requested IV or dIdV data found for channel {channel}."
208
+ )
209
+
210
+ if channel_enable_iv and channel_enable_didv:
211
+ work_items = list(channel_data.get("IV_dIdV") or [])
212
+ processing_type = "IV_dIdV"
213
+ if not work_items:
214
+ raise ValueError(
215
+ f"ERROR: Unable to process both IV and dIdV for channel "
216
+ f"{channel}. No common bias points were found. Adjust "
217
+ "bias_tolerance_percent/bias_tolerance_ua if appropriate."
218
+ )
219
+ elif channel_enable_iv:
220
+ work_items = list(channel_data["IV"])
221
+ processing_type = "IV"
222
+ else:
223
+ work_items = list(channel_data["dIdV"])
224
+ processing_type = "dIdV"
225
+
226
+ if not work_items:
227
+ raise ValueError(f"ERROR: No sweep points found for channel {channel}.")
228
+
229
+ n_workers = max(1, min(int(ncores), len(work_items)))
230
+ chunks = self._split_work_items(work_items, n_workers)
231
+
232
+ worker_args = zip(
233
+ chunks,
234
+ repeat(channel),
235
+ repeat(processing_type),
236
+ repeat(float(trace_length_iv_msec)),
237
+ repeat(None if nrandoms_iv is None else int(nrandoms_iv)),
238
+ repeat(float(min_separation_iv_msec)),
239
+ repeat(random_seed),
240
+ )
241
+
242
+ if n_workers == 1:
243
+ output_channel_df = self._process_work_items(*next(worker_args))
244
+ else:
245
+ if self._verbose:
246
+ print(
247
+ f"INFO: Processing {len(work_items)} bias points with "
248
+ f"{n_workers} processes"
249
+ )
250
+ with Pool(processes=n_workers) as pool:
251
+ df_list = pool.starmap(self._process_work_items, worker_args)
252
+ output_channel_df = pd.concat(df_list, ignore_index=True)
253
+
254
+ output_channel_df = output_channel_df.sort_values(
255
+ "tes_bias", ascending=False, key=np.abs, ignore_index=True
256
+ )
257
+ self._assign_sweep_state(output_channel_df)
258
+ output_dict[channel] = output_channel_df
259
+
260
+ if self._verbose:
261
+ counts = output_channel_df["state"].value_counts(dropna=False)
262
+ n_normal = int(counts.get("normal", 0))
263
+ n_sc = int(counts.get("sc", 0))
264
+ print(f"INFO: IV/dIdV processing done for channel {channel}!")
265
+ if n_sc:
266
+ print(f"INFO: Found {n_sc} SC points based on linearity")
267
+ else:
268
+ print("INFO: Unable to estimate SC points based on linearity")
269
+ if n_normal:
270
+ print(f"INFO: Found {n_normal} normal points based on linearity")
271
+ else:
272
+ print("INFO: Unable to estimate normal points based on linearity")
273
+
274
+ self._filter_data.set_ivsweep_data_from_dict(output_dict)
275
+
276
+ if lgc_save:
277
+ processed_iv = any("offset_iv" in df.columns for df in output_dict.values())
278
+ file_name = self._save_filter_data(
279
+ save_path, prefer_iv=processed_iv
280
+ )
281
+ print(f"INFO: Saving dataframe in {file_name}")
282
+
283
+ if self._verbose:
284
+ print("INFO: IV/dIdV processing done!")
285
+
286
+ if lgc_output:
287
+ return output_dict
288
+ return None
289
+
290
+ def plot_ivsweep_offset(self, channel, tag="default"):
291
+ self._filter_data.plot_ivsweep_offset(channel=channel, tag=tag)
292
+
293
+ def _process_work_items(
294
+ self,
295
+ work_items,
296
+ channel,
297
+ processing_type,
298
+ trace_length_iv_msec,
299
+ nrandoms_iv,
300
+ min_separation_iv_msec,
301
+ random_seed,
302
+ ):
303
+ rows = []
304
+
305
+ for item in work_items:
306
+ if processing_type == "IV_dIdV":
307
+ iv_point = item.iv
308
+ didv_point = item.didv
309
+ pair = item
310
+ elif processing_type == "IV":
311
+ iv_point = item
312
+ didv_point = None
313
+ pair = None
314
+ else:
315
+ iv_point = None
316
+ didv_point = item
317
+ pair = None
318
+
319
+ display_bias = iv_point.bias_ua if iv_point is not None else didv_point.bias_ua
320
+ if self._verbose:
321
+ print(f"INFO: processing channel {channel}, bias point {display_bias} uA")
322
+
323
+ row = {
324
+ "channel": channel,
325
+ "processing_type": processing_type,
326
+ "processing_label": self._processing_label,
327
+ "tes_bias_ua": float(display_bias),
328
+ }
329
+
330
+ first_result = None
331
+ if iv_point is not None:
332
+ iv_result = self._process_measurement_point(
333
+ iv_point,
334
+ channel=channel,
335
+ measurement_type="iv",
336
+ trace_length_iv_msec=trace_length_iv_msec,
337
+ nrandoms_iv=nrandoms_iv,
338
+ min_separation_iv_msec=min_separation_iv_msec,
339
+ random_seed=random_seed,
340
+ )
341
+ row.update(iv_result)
342
+ first_result = iv_result
343
+
344
+ if didv_point is not None:
345
+ didv_result = self._process_measurement_point(
346
+ didv_point,
347
+ channel=channel,
348
+ measurement_type="didv",
349
+ trace_length_iv_msec=trace_length_iv_msec,
350
+ nrandoms_iv=nrandoms_iv,
351
+ min_separation_iv_msec=min_separation_iv_msec,
352
+ random_seed=random_seed,
353
+ )
354
+ row.update(didv_result)
355
+ if first_result is None:
356
+ first_result = didv_result
357
+
358
+ row["tes_bias"] = float(first_result["_tes_bias"])
359
+ row["rshunt"] = float(first_result["_rshunt"])
360
+ row["rp"] = first_result["_rp"]
361
+ row["temperature_mc"] = first_result["_temperature_mc"]
362
+ row["temperature_cp"] = first_result["_temperature_cp"]
363
+ row["temperature_still"] = first_result["_temperature_still"]
364
+
365
+ if iv_point is not None:
366
+ row["tes_bias_point_iv_ua"] = float(iv_point.bias_ua)
367
+ if didv_point is not None:
368
+ row["tes_bias_point_didv_ua"] = float(didv_point.bias_ua)
369
+ if pair is not None:
370
+ row["bias_match_delta_ua"] = float(pair.bias_delta_ua)
371
+ row["bias_match_method"] = pair.match_method
372
+
373
+ for key in list(row):
374
+ if key.startswith("_"):
375
+ row.pop(key)
376
+ rows.append(row)
377
+
378
+ return pd.DataFrame(rows)
379
+
380
+ def _process_measurement_point(
381
+ self,
382
+ point,
383
+ *,
384
+ channel,
385
+ measurement_type,
386
+ trace_length_iv_msec,
387
+ nrandoms_iv,
388
+ min_separation_iv_msec,
389
+ random_seed,
390
+ ):
391
+ traces, infos, detector_settings, trace_selection = self._read_point_traces(
392
+ point,
393
+ channel=channel,
394
+ measurement_type=measurement_type,
395
+ trace_length_iv_msec=trace_length_iv_msec,
396
+ nrandoms_iv=nrandoms_iv,
397
+ min_separation_iv_msec=min_separation_iv_msec,
398
+ random_seed=random_seed,
399
+ )
400
+
401
+ traces = np.asarray(traces)
402
+ if traces.ndim != 3 or traces.shape[1] != 1:
403
+ raise ValueError(
404
+ f"Expected traces with shape (n_traces, 1, n_samples) for "
405
+ f"{channel}, got {traces.shape}."
406
+ )
407
+ traces = traces[:, 0, :]
408
+
409
+ if traces.shape[0] == 0:
410
+ raise ValueError(
411
+ f"No {measurement_type} traces available for channel {channel}, "
412
+ f"stream {point.stream_id}."
413
+ )
414
+
415
+ fs = float(infos[0].get("sample_rate_hz", point.sample_rate_hz)) if infos else float(point.sample_rate_hz)
416
+ settings = detector_settings[channel]
417
+
418
+ tes_bias = self._finite_float(settings.get("tes_bias_dc_amps"), "tes_bias_dc_amps", channel)
419
+ output_gain = self._float_or_nan(settings.get("output_gain"))
420
+ close_loop_norm = self._float_or_nan(settings.get("close_loop_norm"))
421
+ output_offset = self._float_or_nan(settings.get("output_offset_vdc"))
422
+ rshunt = self._finite_float(settings.get("shunt_resistance_ohm"), "shunt_resistance_ohm", channel)
423
+ rp = self._float_or_nan(
424
+ settings.get("parasitic_resistance_ohm", settings.get("rp"))
425
+ )
426
+ sgamp = self._float_or_nan(settings.get("tes_bias_ac_amplitude_amps"))
427
+ sgfreq = self._float_or_nan(settings.get("tes_bias_ac_frequency_hz"))
428
+ dutycycle = self._float_or_default(
429
+ settings.get(
430
+ "dutycycle",
431
+ settings.get("duty_cycle", settings.get("tes_bias_ac_dutycycle")),
432
+ ),
433
+ 0.5,
434
+ )
435
+
436
+ acquisition_name = (
437
+ infos[0].get("acquisition_name") if infos else None
438
+ ) or point.acquisition_name
439
+ stream_name = (
440
+ infos[0].get("stream_name") if infos else None
441
+ ) or point.stream_name or point.stream_id
442
+
443
+ # Reject only traces that are entirely zero. The previous implementation
444
+ # accidentally rejected every trace containing even a single exact zero.
445
+ nonzero = ~np.all(traces == 0, axis=1)
446
+ traces = traces[nonzero]
447
+ if traces.shape[0] == 0:
448
+ raise ValueError(
449
+ f"All {measurement_type} traces are zero for channel {channel}, "
450
+ f"stream {point.stream_id}."
451
+ )
452
+
453
+ result = {
454
+ f"stream_id_{measurement_type}": str(point.stream_id),
455
+ f"stream_name_{measurement_type}": str(stream_name),
456
+ f"acquisition_name_{measurement_type}": str(acquisition_name),
457
+ f"fs_{measurement_type}": fs,
458
+ f"output_variable_gain_{measurement_type}": output_gain,
459
+ f"output_variable_offset_{measurement_type}": output_offset,
460
+ f"close_loop_norm_{measurement_type}": close_loop_norm,
461
+ f"rshunt_{measurement_type}": rshunt,
462
+ f"rp_{measurement_type}": rp,
463
+ f"tes_bias_{measurement_type}": tes_bias,
464
+ f"ntraces_{measurement_type}": int(traces.shape[0]),
465
+ f"trace_length_samples_{measurement_type}": int(traces.shape[-1]),
466
+ f"trace_selection_{measurement_type}": trace_selection,
467
+ "_tes_bias": tes_bias,
468
+ "_rshunt": rshunt,
469
+ "_rp": rp,
470
+ "_temperature_mc": self._temperature_value(settings, "mc"),
471
+ "_temperature_cp": self._temperature_value(settings, "cp"),
472
+ "_temperature_still": self._temperature_value(settings, "still"),
473
+ }
474
+
475
+ if measurement_type == "iv":
476
+ cut = np.asarray(qp.autocuts_noise(traces, fs=fs), dtype=bool)
477
+ else:
478
+ cut = np.asarray(qp.autocuts_didv(traces, fs=fs), dtype=bool)
479
+
480
+ if cut.size != traces.shape[0]:
481
+ raise ValueError(
482
+ f"Unexpected autocut size {cut.size} for {traces.shape[0]} traces."
483
+ )
484
+ n_pass = int(np.sum(cut))
485
+ if n_pass == 0:
486
+ raise ValueError(
487
+ f"No {measurement_type} traces survive autocuts for channel "
488
+ f"{channel}, stream {point.stream_id}."
489
+ )
490
+
491
+ cut_eff = n_pass / len(cut)
492
+ selected = traces[cut]
493
+
494
+ if measurement_type == "iv":
495
+ psd_freq, psd = qp.calc_psd(selected, fs=fs, folded_over=False)
496
+ offset, offset_err = qp.utils.calc_offset(selected, fs=fs)
497
+ avgtrace = np.mean(selected, axis=0)
498
+ result.update({"psd": psd, "psd_freq": psd_freq})
499
+ else:
500
+ if not np.isfinite(sgfreq) or sgfreq <= 0:
501
+ raise ValueError(
502
+ f"Invalid dIdV frequency {sgfreq!r} for channel {channel}, "
503
+ f"stream {point.stream_id}."
504
+ )
505
+ if not np.isfinite(sgamp):
506
+ raise ValueError(
507
+ f"Invalid dIdV amplitude {sgamp!r} for channel {channel}, "
508
+ f"stream {point.stream_id}."
509
+ )
510
+
511
+ offset, offset_err = qp.utils.calc_offset(
512
+ selected, fs=fs, sgfreq=sgfreq, is_didv=True
513
+ )
514
+ avgtrace = np.mean(selected, axis=0)
515
+
516
+ avgtrace_lp = qp.utils.lowpassfilter(
517
+ avgtrace, cut_off_freq=1.5e3, fs=fs, order=1
518
+ )
519
+ nb_bins = len(avgtrace_lp)
520
+ nb_cycles = (nb_bins / fs) * sgfreq
521
+ start_bin = round(nb_bins * 0.1)
522
+ end_bin = round(nb_bins * 0.9)
523
+ if nb_cycles > 1.5:
524
+ start_bin = round(fs / sgfreq * 0.25)
525
+ end_bin = min(nb_bins, 4 * start_bin)
526
+
527
+ calc_values = avgtrace_lp[start_bin:end_bin]
528
+ if calc_values.size == 0:
529
+ rtes_estimate = np.nan
530
+ else:
531
+ delta_i = float(calc_values.max() - calc_values.min())
532
+ delta_v = sgamp * rshunt
533
+ rtes_estimate = np.nan if delta_i == 0 else delta_v / delta_i * 1e3
534
+
535
+ didvobj = qp.DIDV(
536
+ selected,
537
+ fs,
538
+ sgfreq,
539
+ sgamp,
540
+ rshunt,
541
+ autoresample=False,
542
+ dutycycle=dutycycle,
543
+ )
544
+ didvobj.processtraces()
545
+
546
+ result.update(
547
+ {
548
+ "sgamp": sgamp,
549
+ "sgfreq": sgfreq,
550
+ "didvmean": didvobj._didvmean,
551
+ "didvstd": didvobj._didvstd,
552
+ "dutycycle": dutycycle,
553
+ "rtes_estimate": rtes_estimate,
554
+ }
555
+ )
556
+
557
+ result.update(
558
+ {
559
+ f"offset_{measurement_type}": offset,
560
+ f"offset_err_{measurement_type}": offset_err,
561
+ f"cut_eff_{measurement_type}": cut_eff,
562
+ f"cut_{measurement_type}": cut,
563
+ f"cut_pass_{measurement_type}": True,
564
+ f"avgtrace_{measurement_type}": avgtrace,
565
+ }
566
+ )
567
+ return result
568
+
569
+ def _read_point_traces(
570
+ self,
571
+ point,
572
+ *,
573
+ channel,
574
+ measurement_type,
575
+ trace_length_iv_msec,
576
+ nrandoms_iv,
577
+ min_separation_iv_msec,
578
+ random_seed,
579
+ ):
580
+ """Read native or randomly selected records for one sweep point."""
581
+
582
+ with StreamReader(
583
+ point.acquisition_path,
584
+ streams=point.stream_id,
585
+ measurement_types=measurement_type,
586
+ verbose=False,
587
+ ) as reader:
588
+ detector_settings = reader.get_detector_settings()
589
+ if channel not in detector_settings:
590
+ raise ValueError(
591
+ f"Channel {channel} is not available in stream {point.stream_id}."
592
+ )
593
+
594
+ if measurement_type == "didv":
595
+ # HDF5 dIdV is historically stored as finite segments even when
596
+ # adc_mode metadata says "continuous". Only native Zarr
597
+ # channel-sample streams are truly continuous here.
598
+ if point.storage_format == "zarr" and reader.is_continuous_stream:
599
+ raise ValueError(
600
+ f"Continuous dIdV stream {point.stream_id} is not supported "
601
+ "by IVSweepProcessing yet; dIdV requires finite native records."
602
+ )
603
+ traces, infos = reader.read_records(
604
+ channels=channel,
605
+ units="amps",
606
+ include_metadata=True,
607
+ stack=True,
608
+ )
609
+ return traces, infos, detector_settings, "native_records"
610
+
611
+ fs = float(reader.sample_rate_hz)
612
+ trace_length_samples = int(
613
+ round(float(trace_length_iv_msec) * fs / 1000.0)
614
+ )
615
+ if trace_length_samples <= 0:
616
+ raise ValueError("Requested IV trace length converts to zero samples")
617
+ pretrigger_samples = trace_length_samples // 2
618
+ edge_exclusion_msec = pretrigger_samples / fs * 1000.0
619
+ min_separation_samples = int(
620
+ np.ceil(float(min_separation_iv_msec) * fs / 1000.0)
621
+ )
622
+
623
+ if point.storage_format == "hdf5":
624
+ native_lengths = {
625
+ int(length)
626
+ for _, _, length in point.hdf5_dump_segments
627
+ if int(length) > 0
628
+ }
629
+ if not native_lengths:
630
+ raise ValueError(
631
+ f"Unable to determine HDF5 segment length for stream {point.stream_id}."
632
+ )
633
+ if len(native_lengths) != 1:
634
+ raise ValueError(
635
+ f"HDF5 stream {point.stream_id} contains mixed segment lengths "
636
+ f"{sorted(native_lengths)}; IVSweepProcessing expects one native "
637
+ "segment length per sweep point."
638
+ )
639
+ native_length = next(iter(native_lengths))
640
+
641
+ if native_length == trace_length_samples:
642
+ record_list = None
643
+ n_records = None
644
+ selection = "native_hdf5_segments"
645
+ if nrandoms_iv is not None:
646
+ record_list = self._random_native_hdf5_records(
647
+ point,
648
+ int(nrandoms_iv),
649
+ seed=self._point_seed(
650
+ random_seed, channel, measurement_type, point
651
+ ),
652
+ )
653
+ selection = "random_native_hdf5_segments"
654
+ traces, infos = reader.read_records(
655
+ record_list=record_list,
656
+ n_records=n_records,
657
+ channels=channel,
658
+ units="amps",
659
+ include_metadata=True,
660
+ stack=True,
661
+ )
662
+ return traces, infos, detector_settings, selection
663
+
664
+ if native_length < trace_length_samples:
665
+ raise ValueError(
666
+ f"Requested IV trace length is {trace_length_samples} samples, "
667
+ f"but HDF5 stream {point.stream_id} has {native_length}-sample "
668
+ "segments. Triggered records cannot span HDF5 segment boundaries."
669
+ )
670
+
671
+ selection = "random_within_hdf5_segments"
672
+
673
+ elif point.storage_format == "zarr":
674
+ if not reader.is_continuous_stream:
675
+ native_length = point.trace_length_samples
676
+ if native_length is not None and int(native_length) != trace_length_samples:
677
+ raise ValueError(
678
+ f"Finite Zarr IV stream {point.stream_id} has native trace "
679
+ f"length {native_length}, requested {trace_length_samples}."
680
+ )
681
+ record_list = None
682
+ if nrandoms_iv is not None:
683
+ record_list = self._random_native_zarr_records(
684
+ point,
685
+ int(nrandoms_iv),
686
+ seed=self._point_seed(
687
+ random_seed, channel, measurement_type, point
688
+ ),
689
+ )
690
+ traces, infos = reader.read_records(
691
+ record_list=record_list,
692
+ channels=channel,
693
+ units="amps",
694
+ include_metadata=True,
695
+ stack=True,
696
+ )
697
+ return traces, infos, detector_settings, "native_zarr_traces"
698
+ selection = "random_across_zarr_stream"
699
+ else:
700
+ raise ValueError(
701
+ f"Unsupported storage format {point.storage_format!r}."
702
+ )
703
+
704
+ target_nrandoms = nrandoms_iv
705
+ if target_nrandoms is None:
706
+ target_nrandoms = self._maximum_nonoverlap_randoms(
707
+ point,
708
+ trace_length_samples=trace_length_samples,
709
+ pretrigger_samples=pretrigger_samples,
710
+ min_separation_samples=min_separation_samples,
711
+ )
712
+ if target_nrandoms <= 0:
713
+ raise ValueError(
714
+ f"No complete {trace_length_iv_msec:g} ms IV traces fit in "
715
+ f"stream {point.stream_id} with the requested edge exclusion."
716
+ )
717
+
718
+ seed = self._point_seed(
719
+ random_seed, channel, measurement_type, point
720
+ )
721
+ randoms = Randoms(
722
+ point.acquisition_path,
723
+ streams=point.stream_id,
724
+ data_type="iv",
725
+ verbose=False,
726
+ )
727
+ random_df = randoms.process(
728
+ nrandoms=int(target_nrandoms),
729
+ min_separation_msec=float(min_separation_iv_msec),
730
+ edge_exclusion_msec=float(edge_exclusion_msec),
731
+ partition_target_duration_s=None,
732
+ random_seed=seed,
733
+ lgc_save=False,
734
+ lgc_output=True,
735
+ )
736
+ record_list = self._random_dataframe_to_record_list(
737
+ random_df, point.storage_format
738
+ )
739
+
740
+ traces, infos = reader.read_records(
741
+ record_list=record_list,
742
+ channels=channel,
743
+ trace_length_samples=trace_length_samples,
744
+ pretrigger_length_samples=pretrigger_samples,
745
+ units="amps",
746
+ include_metadata=True,
747
+ stack=True,
748
+ )
749
+ return traces, infos, detector_settings, selection
750
+
751
+ def _discover_sweep_data(self, data_paths):
752
+ if self._verbose:
753
+ print("INFO: Checking sweep data")
754
+
755
+ if isinstance(data_paths, (str, Path)):
756
+ data_paths = [data_paths]
757
+ else:
758
+ data_paths = list(data_paths)
759
+ if not data_paths:
760
+ raise ValueError("No IV/dIdV acquisition path provided")
761
+
762
+ sources = {"iv": None, "didv": None}
763
+ for path in data_paths:
764
+ catalog = AcquisitionCatalog(path, verbose=self._verbose)
765
+ for measurement_type in ("iv", "didv"):
766
+ view = catalog.filter(measurement_types=measurement_type)
767
+ if not view.entries:
768
+ continue
769
+ if sources[measurement_type] is not None:
770
+ previous = sources[measurement_type]["acquisition_path"]
771
+ if Path(previous).resolve() != Path(catalog.acquisition_path).resolve():
772
+ raise ValueError(
773
+ f"ERROR: {measurement_type} data should be in a single "
774
+ "acquisition directory."
775
+ )
776
+ sources[measurement_type] = {
777
+ "catalog": catalog,
778
+ "view": view,
779
+ "acquisition_path": str(catalog.acquisition_path),
780
+ "acquisition_name": catalog.acquisition_name,
781
+ "base_path": str(catalog.base_path),
782
+ }
783
+
784
+ if sources["iv"] is None and sources["didv"] is None:
785
+ raise ValueError("No IV or dIdV data were found in the supplied acquisition paths")
786
+
787
+ points_by_type = {}
788
+ for measurement_type in ("iv", "didv"):
789
+ source = sources[measurement_type]
790
+ if source is None:
791
+ points_by_type[measurement_type] = {}
792
+ continue
793
+ points_by_type[measurement_type] = self._discover_measurement_points(
794
+ source["view"], measurement_type
795
+ )
796
+
797
+ channels = sorted(
798
+ set(points_by_type["iv"]).union(points_by_type["didv"])
799
+ )
800
+ output = {}
801
+ for channel in channels:
802
+ iv_points = points_by_type["iv"].get(channel, [])
803
+ didv_points = points_by_type["didv"].get(channel, [])
804
+ pairs = self._match_iv_didv_points(iv_points, didv_points)
805
+ output[channel] = {
806
+ "IV": iv_points or None,
807
+ "dIdV": didv_points or None,
808
+ "IV_dIdV": pairs or None,
809
+ }
810
+
811
+ iv_source = sources["iv"]
812
+ didv_source = sources["didv"]
813
+ return {
814
+ "data": output,
815
+ "acquisition_name_iv": None if iv_source is None else iv_source["acquisition_name"],
816
+ "acquisition_name_didv": None if didv_source is None else didv_source["acquisition_name"],
817
+ "base_path_iv": None if iv_source is None else iv_source["base_path"],
818
+ "base_path_didv": None if didv_source is None else didv_source["base_path"],
819
+ }
820
+
821
+ def _discover_measurement_points(self, catalog_view, measurement_type):
822
+ entries_by_stream = {}
823
+ for entry in catalog_view.entries:
824
+ stream_id = self._entry_stream_id(entry)
825
+ entries_by_stream.setdefault(stream_id, []).append(entry)
826
+
827
+ stream_info = []
828
+ explicit_scan_channels = set()
829
+
830
+ for stream_id, entries in sorted(
831
+ entries_by_stream.items(), key=lambda item: self._stream_sort_key(item[1][0])
832
+ ):
833
+ entries = sorted(entries, key=self._resource_sort_key)
834
+ first_entry = entries[0]
835
+ resource = catalog_view.resource_path(first_entry)
836
+
837
+ # StreamReader provides the common HDF5/Zarr metadata interface.
838
+ # ``resource`` is the selected stream resource from the catalog.
839
+ with StreamReader(resource, verbose=False) as reader:
840
+ detector_settings = reader.get_detector_settings()
841
+ metadata = reader.get_metadata()
842
+
843
+ scan = metadata.get("scan") or first_entry.get("scan")
844
+ explicit_scan_channels.update(self._tes_bias_scan_channels(scan))
845
+
846
+ record_channels = set()
847
+ for entry in entries:
848
+ channels = entry.get("record_channels") or []
849
+ if isinstance(channels, str):
850
+ channels = [channels]
851
+ record_channels.update(str(name) for name in channels)
852
+
853
+ stream_info.append(
854
+ {
855
+ "stream_id": stream_id,
856
+ "entries": entries,
857
+ "settings": detector_settings,
858
+ "scan": scan,
859
+ "record_channels": record_channels,
860
+ }
861
+ )
862
+
863
+ if explicit_scan_channels:
864
+ sweep_channels = explicit_scan_channels
865
+ else:
866
+ sweep_channels = self._infer_legacy_sweep_channels(stream_info)
867
+ if self._verbose and sweep_channels:
868
+ print(
869
+ f"INFO: {measurement_type} acquisition has no explicit TES-bias "
870
+ "scan-channel metadata; inferred channels from bias changes: "
871
+ + ", ".join(sorted(sweep_channels))
872
+ )
873
+
874
+ output = {channel: [] for channel in sweep_channels}
875
+
876
+ for info in stream_info:
877
+ entries = info["entries"]
878
+ settings = info["settings"]
879
+ first_entry = entries[0]
880
+ scan = info["scan"]
881
+ scan_index = self._scan_index(scan)
882
+
883
+ for channel in sweep_channels:
884
+ record_channels = info.get("record_channels") or set()
885
+ if record_channels and channel not in record_channels:
886
+ continue
887
+ channel_settings = settings.get(channel)
888
+ if channel_settings is None:
889
+ continue
890
+ bias_amp = self._float_or_nan(channel_settings.get("tes_bias_dc_amps"))
891
+ if not np.isfinite(bias_amp):
892
+ continue
893
+
894
+ point = self._make_sweep_point(
895
+ catalog_view,
896
+ entries,
897
+ measurement_type=measurement_type,
898
+ stream_id=info["stream_id"],
899
+ bias_ua=bias_amp * 1e6,
900
+ scan_index=scan_index,
901
+ )
902
+ output[channel].append(point)
903
+
904
+ for channel in list(output):
905
+ output[channel] = sorted(output[channel], key=self._point_sort_key)
906
+ if not output[channel]:
907
+ output.pop(channel)
908
+ return output
909
+
910
+ def _make_sweep_point(
911
+ self,
912
+ catalog_view,
913
+ entries,
914
+ *,
915
+ measurement_type,
916
+ stream_id,
917
+ bias_ua,
918
+ scan_index,
919
+ ):
920
+ first = entries[0]
921
+ storage_format = str(first.get("storage_format") or catalog_view.storage_format)
922
+ sample_rate = float(first.get("sample_rate_hz") or catalog_view.sample_rate_hz)
923
+
924
+ hdf5_dump_segments = []
925
+ if storage_format == "hdf5":
926
+ for entry in entries:
927
+ dump_num = int(entry.get("dump_num") or 0)
928
+ n_segments = int(entry.get("n_segments") or 0)
929
+ segment_length = entry.get("segment_length_samples")
930
+ if segment_length is None and entry.get("segment_duration_s") is not None:
931
+ segment_length = int(round(float(entry["segment_duration_s"]) * sample_rate))
932
+ if segment_length is None:
933
+ continue
934
+ hdf5_dump_segments.append(
935
+ (dump_num, n_segments, int(segment_length))
936
+ )
937
+
938
+ trace_lengths = {
939
+ int(entry["trace_length_samples"])
940
+ for entry in entries
941
+ if entry.get("trace_length_samples") is not None
942
+ }
943
+ trace_length_samples = next(iter(trace_lengths)) if len(trace_lengths) == 1 else None
944
+ n_traces_values = [int(entry.get("n_traces") or 0) for entry in entries]
945
+ n_traces = sum(n_traces_values) if any(n_traces_values) else None
946
+
947
+ n_samples = None
948
+ if storage_format == "zarr" and str(first.get("raw_shape_model")) == "channel_sample":
949
+ values = [entry.get("n_samples") for entry in entries if entry.get("n_samples") is not None]
950
+ if values:
951
+ # Native Zarr normally has exactly one resource per stream.
952
+ n_samples = int(sum(int(value) for value in values))
953
+ else:
954
+ try:
955
+ stream_view = catalog_view.filter(streams=stream_id)
956
+ n_samples = int(stream_view.n_samples)
957
+ except Exception:
958
+ n_samples = None
959
+
960
+ duration_s = float(sum(self._entry_duration_s(entry) for entry in entries))
961
+ return _SweepPoint(
962
+ acquisition_path=str(catalog_view.acquisition_path),
963
+ acquisition_name=str(catalog_view.acquisition_name),
964
+ measurement_type=measurement_type,
965
+ stream_id=str(stream_id),
966
+ stream_num=self._optional_int(first.get("stream_num")),
967
+ stream_name=first.get("stream_name"),
968
+ storage_format=storage_format,
969
+ sample_rate_hz=sample_rate,
970
+ bias_ua=float(bias_ua),
971
+ scan_index=scan_index,
972
+ sequence_index=self._optional_int(first.get("sequence_index")),
973
+ sequence_repeat_index=self._optional_int(first.get("sequence_repeat_index")),
974
+ adc_mode=first.get("adc_mode"),
975
+ raw_shape_model=first.get("raw_shape_model"),
976
+ duration_s=duration_s,
977
+ n_samples=n_samples,
978
+ hdf5_dump_segments=tuple(hdf5_dump_segments),
979
+ trace_length_samples=trace_length_samples,
980
+ n_traces=n_traces,
981
+ )
982
+
983
+ def _match_iv_didv_points(self, iv_points, didv_points):
984
+ if not iv_points or not didv_points:
985
+ return []
986
+
987
+ pairs = []
988
+ used_iv = set()
989
+ used_didv = set()
990
+
991
+ # Modern same-acquisition data: scan index + repeat is the strongest
992
+ # identity. Bias is still retained in the output as a consistency check.
993
+ for i, iv in enumerate(iv_points):
994
+ if iv.scan_index is None:
995
+ continue
996
+ for j, didv in enumerate(didv_points):
997
+ if j in used_didv or didv.scan_index is None:
998
+ continue
999
+ if Path(iv.acquisition_path).resolve() != Path(didv.acquisition_path).resolve():
1000
+ continue
1001
+ if iv.scan_index != didv.scan_index:
1002
+ continue
1003
+ if (
1004
+ iv.sequence_repeat_index is not None
1005
+ and didv.sequence_repeat_index is not None
1006
+ and iv.sequence_repeat_index != didv.sequence_repeat_index
1007
+ ):
1008
+ continue
1009
+ pairs.append(
1010
+ _SweepPair(
1011
+ iv=iv,
1012
+ didv=didv,
1013
+ bias_delta_ua=float(didv.bias_ua - iv.bias_ua),
1014
+ match_method="scan_index",
1015
+ )
1016
+ )
1017
+ used_iv.add(i)
1018
+ used_didv.add(j)
1019
+ break
1020
+
1021
+ candidates = []
1022
+ for i, iv in enumerate(iv_points):
1023
+ if i in used_iv:
1024
+ continue
1025
+ for j, didv in enumerate(didv_points):
1026
+ if j in used_didv:
1027
+ continue
1028
+ if self._biases_match(iv.bias_ua, didv.bias_ua):
1029
+ candidates.append((abs(iv.bias_ua - didv.bias_ua), i, j))
1030
+
1031
+ for _, i, j in sorted(candidates):
1032
+ if i in used_iv or j in used_didv:
1033
+ continue
1034
+ iv = iv_points[i]
1035
+ didv = didv_points[j]
1036
+ pairs.append(
1037
+ _SweepPair(
1038
+ iv=iv,
1039
+ didv=didv,
1040
+ bias_delta_ua=float(didv.bias_ua - iv.bias_ua),
1041
+ match_method="bias",
1042
+ )
1043
+ )
1044
+ used_iv.add(i)
1045
+ used_didv.add(j)
1046
+
1047
+ return sorted(pairs, key=lambda pair: self._point_sort_key(pair.iv))
1048
+
1049
+ def _assign_sweep_state(self, dataframe):
1050
+ dataframe["state"] = pd.Series(
1051
+ np.full(len(dataframe), np.nan, dtype=object),
1052
+ index=dataframe.index,
1053
+ dtype=object,
1054
+ )
1055
+ if dataframe.empty:
1056
+ return
1057
+
1058
+ tes_bias = dataframe["tes_bias"].to_numpy()
1059
+ if "offset_iv" in dataframe.columns:
1060
+ offset = dataframe["offset_iv"].to_numpy()
1061
+ elif "offset_didv" in dataframe.columns:
1062
+ offset = dataframe["offset_didv"].to_numpy()
1063
+ else:
1064
+ return
1065
+
1066
+ normal_indices = np.asarray(find_linear_segment(tes_bias, offset), dtype=int)
1067
+ if normal_indices.size:
1068
+ dataframe.loc[normal_indices, "state"] = "normal"
1069
+
1070
+ reversed_bias = tes_bias[::-1].copy()
1071
+ reversed_offset = offset[::-1].copy()
1072
+ sc_reversed = np.asarray(
1073
+ find_linear_segment(reversed_bias, reversed_offset), dtype=int
1074
+ )
1075
+ if sc_reversed.size:
1076
+ sc_indices = len(tes_bias) - sc_reversed - 1
1077
+ dataframe.loc[sc_indices, "state"] = "sc"
1078
+
1079
+ def _maximum_nonoverlap_randoms(
1080
+ self,
1081
+ point,
1082
+ *,
1083
+ trace_length_samples,
1084
+ pretrigger_samples,
1085
+ min_separation_samples,
1086
+ ):
1087
+ edge = int(pretrigger_samples)
1088
+ spacing = max(1, int(min_separation_samples))
1089
+
1090
+ if point.storage_format == "hdf5":
1091
+ total = 0
1092
+ for _, n_segments, segment_length in point.hdf5_dump_segments:
1093
+ usable = int(segment_length) - 2 * edge
1094
+ if usable <= 0:
1095
+ continue
1096
+ capacity = 1 + (usable - 1) // spacing
1097
+ total += int(n_segments) * capacity
1098
+ return total
1099
+
1100
+ if point.storage_format == "zarr":
1101
+ n_samples = point.n_samples
1102
+ if n_samples is None:
1103
+ n_samples = int(round(point.duration_s * point.sample_rate_hz))
1104
+ usable = int(n_samples) - 2 * edge
1105
+ if usable <= 0:
1106
+ return 0
1107
+ return 1 + (usable - 1) // spacing
1108
+
1109
+ return 0
1110
+
1111
+ @staticmethod
1112
+ def _random_dataframe_to_record_list(dataframe, storage_format):
1113
+ if isinstance(dataframe, pd.DataFrame):
1114
+ pdf = dataframe
1115
+ elif hasattr(dataframe, "to_pandas_df"):
1116
+ pdf = dataframe.to_pandas_df()
1117
+ elif hasattr(dataframe, "to_pandas_dataframe"):
1118
+ pdf = dataframe.to_pandas_dataframe()
1119
+ else:
1120
+ raise TypeError(
1121
+ "Randoms output must be a Vaex or pandas dataframe to build a record list"
1122
+ )
1123
+
1124
+ records = []
1125
+ for row in pdf.to_dict(orient="records"):
1126
+ record = {}
1127
+ if storage_format == "hdf5":
1128
+ global_segment = row.get(
1129
+ "global_segment_num", row.get("global_segment_number")
1130
+ )
1131
+ if global_segment is None:
1132
+ raise ValueError("HDF5 random row is missing global segment identity")
1133
+ record["global_segment_num"] = int(global_segment)
1134
+ record["segment_trigger_index"] = int(row["segment_trigger_index"])
1135
+ else:
1136
+ record["stream_trigger_index"] = int(row["stream_trigger_index"])
1137
+ records.append(record)
1138
+ return records
1139
+
1140
+ def _random_native_hdf5_records(self, point, n_records, *, seed):
1141
+ all_records = []
1142
+ for dump_num, n_segments, _ in point.hdf5_dump_segments:
1143
+ for segment_num in range(1, n_segments + 1):
1144
+ all_records.append(
1145
+ {"global_segment_num": dump_num * 100000 + segment_num}
1146
+ )
1147
+ if n_records > len(all_records):
1148
+ raise ValueError(
1149
+ f"Requested nrandoms_iv={n_records}, but HDF5 stream "
1150
+ f"{point.stream_id} contains only {len(all_records)} native segments."
1151
+ )
1152
+ if n_records == len(all_records):
1153
+ return all_records
1154
+ rng = np.random.default_rng(seed)
1155
+ selected = np.sort(rng.choice(len(all_records), size=n_records, replace=False))
1156
+ return [all_records[int(index)] for index in selected]
1157
+
1158
+ def _random_native_zarr_records(self, point, n_records, *, seed):
1159
+ if point.n_traces is None:
1160
+ raise ValueError(
1161
+ f"Unable to determine number of finite Zarr traces for {point.stream_id}."
1162
+ )
1163
+ if n_records > point.n_traces:
1164
+ raise ValueError(
1165
+ f"Requested nrandoms_iv={n_records}, but finite Zarr stream "
1166
+ f"{point.stream_id} contains only {point.n_traces} traces."
1167
+ )
1168
+ rng = np.random.default_rng(seed)
1169
+ selected = np.sort(rng.choice(point.n_traces, size=n_records, replace=False))
1170
+ return [{"trace_index": int(index)} for index in selected]
1171
+
1172
+ @staticmethod
1173
+ def _split_work_items(work_items, n_workers):
1174
+ chunks = np.array_split(np.asarray(work_items, dtype=object), n_workers)
1175
+ return [list(chunk) for chunk in chunks if len(chunk)]
1176
+
1177
+ def _save_filter_data(self, save_path, *, prefer_iv=True):
1178
+ if prefer_iv and self._base_path_iv is not None:
1179
+ base_path = self._base_path_iv
1180
+ group_name = self._acquisition_name_iv
1181
+ else:
1182
+ base_path = self._base_path_didv or self._base_path_iv
1183
+ group_name = self._acquisition_name_didv or self._acquisition_name_iv
1184
+ if save_path is None:
1185
+ save_path = Path(base_path) / "filterdata"
1186
+ save_path = Path(str(save_path).replace("/raw/filterdata", "/filterdata"))
1187
+ else:
1188
+ save_path = Path(save_path)
1189
+
1190
+ if group_name and group_name not in str(save_path):
1191
+ save_path = save_path / group_name
1192
+ save_path.mkdir(parents=True, exist_ok=True)
1193
+
1194
+ now = datetime.now()
1195
+ timestamp_id = now.strftime("D%Y%m%d_T%H%M%S")
1196
+ if self._processing_label is not None:
1197
+ file_name = save_path / f"{self._processing_label}_{timestamp_id}.hdf5"
1198
+ else:
1199
+ file_name = save_path / f"ivsweep_processing_{timestamp_id}.hdf5"
1200
+ self._filter_data.save_hdf5(str(file_name))
1201
+ return str(file_name)
1202
+
1203
+ def _infer_legacy_sweep_channels(self, stream_info):
1204
+ values = {}
1205
+ for info in stream_info:
1206
+ for channel, settings in info["settings"].items():
1207
+ bias = self._float_or_nan(settings.get("tes_bias_dc_amps"))
1208
+ if np.isfinite(bias):
1209
+ values.setdefault(channel, []).append(bias)
1210
+
1211
+ sweep_channels = set()
1212
+ for channel, biases in values.items():
1213
+ if len(biases) < 2:
1214
+ continue
1215
+ biases = np.asarray(biases, dtype=float)
1216
+ spread = float(np.ptp(biases))
1217
+ scale = max(float(np.max(np.abs(biases))), 1e-15)
1218
+ if spread > max(1e-15, 1e-3 * scale):
1219
+ sweep_channels.add(channel)
1220
+ return sweep_channels
1221
+
1222
+ @staticmethod
1223
+ def _tes_bias_scan_channels(scan):
1224
+ channels = set()
1225
+ if not isinstance(scan, dict):
1226
+ return channels
1227
+ values = scan.get("values")
1228
+ if not isinstance(values, dict):
1229
+ return channels
1230
+ for entry in values.values():
1231
+ if not isinstance(entry, dict):
1232
+ continue
1233
+ target = str(entry.get("target") or "").strip().lower()
1234
+ if target not in {
1235
+ "tes_bias.dc",
1236
+ "tes_bias_dc",
1237
+ "tes_bias.dc_amps",
1238
+ "tes_bias_dc_amps",
1239
+ }:
1240
+ continue
1241
+ entry_channels = entry.get("channels") or []
1242
+ if isinstance(entry_channels, str):
1243
+ entry_channels = [entry_channels]
1244
+ channels.update(str(channel) for channel in entry_channels)
1245
+ return channels
1246
+
1247
+ @staticmethod
1248
+ def _scan_index(scan):
1249
+ if not isinstance(scan, dict) or scan.get("index") is None:
1250
+ return None
1251
+ try:
1252
+ return int(scan["index"])
1253
+ except (TypeError, ValueError):
1254
+ return None
1255
+
1256
+ def _biases_match(self, first, second):
1257
+ return bool(
1258
+ np.isclose(
1259
+ float(first),
1260
+ float(second),
1261
+ rtol=self._bias_tolerance_percent / 100.0,
1262
+ atol=self._bias_tolerance_ua,
1263
+ )
1264
+ )
1265
+
1266
+ @staticmethod
1267
+ def _point_seed(base_seed, channel, measurement_type, point):
1268
+ if base_seed is None:
1269
+ return None
1270
+ text = "|".join(
1271
+ [
1272
+ str(int(base_seed)),
1273
+ str(channel),
1274
+ str(measurement_type),
1275
+ str(point.acquisition_name),
1276
+ str(point.stream_id),
1277
+ ]
1278
+ )
1279
+ digest = hashlib.sha256(text.encode("utf-8")).digest()
1280
+ return int.from_bytes(digest[:8], "big") % (2**32)
1281
+
1282
+ @staticmethod
1283
+ def _entry_stream_id(entry):
1284
+ value = entry.get("stream_id")
1285
+ if value is not None:
1286
+ return str(value)
1287
+ value = entry.get("stream_name")
1288
+ if value is not None:
1289
+ return str(value)
1290
+ value = entry.get("stream_num")
1291
+ if value is not None:
1292
+ return str(value)
1293
+ raise ValueError(f"Catalog entry is missing stream identity: {entry}")
1294
+
1295
+ @staticmethod
1296
+ def _stream_sort_key(entry):
1297
+ value = entry.get("stream_num")
1298
+ if value is not None:
1299
+ return (0, int(value))
1300
+ return (1, str(entry.get("stream_id") or entry.get("stream_name") or ""))
1301
+
1302
+ @staticmethod
1303
+ def _resource_sort_key(entry):
1304
+ dump = entry.get("dump_num")
1305
+ if dump is not None:
1306
+ return (0, int(dump))
1307
+ return (1, str(entry.get("resource_path") or ""))
1308
+
1309
+ @staticmethod
1310
+ def _point_sort_key(point):
1311
+ return (
1312
+ point.scan_index is None,
1313
+ point.scan_index if point.scan_index is not None else 0,
1314
+ point.sequence_repeat_index if point.sequence_repeat_index is not None else 0,
1315
+ point.stream_num if point.stream_num is not None else 0,
1316
+ )
1317
+
1318
+ @staticmethod
1319
+ def _entry_duration_s(entry):
1320
+ if entry.get("duration_s") is not None:
1321
+ return float(entry["duration_s"])
1322
+ sample_rate = entry.get("sample_rate_hz")
1323
+ if sample_rate is None or float(sample_rate) <= 0:
1324
+ return 0.0
1325
+ if entry.get("n_samples") is not None:
1326
+ return float(entry["n_samples"]) / float(sample_rate)
1327
+ if entry.get("n_segments") is not None and entry.get("segment_length_samples") is not None:
1328
+ return (
1329
+ float(entry["n_segments"])
1330
+ * float(entry["segment_length_samples"])
1331
+ / float(sample_rate)
1332
+ )
1333
+ return 0.0
1334
+
1335
+ @staticmethod
1336
+ def _temperature_value(settings, name):
1337
+ for key in (
1338
+ f"temperature_{name}",
1339
+ f"temperature_{name}_k",
1340
+ ):
1341
+ if key in settings:
1342
+ try:
1343
+ return float(settings[key])
1344
+ except (TypeError, ValueError):
1345
+ return np.nan
1346
+ return np.nan
1347
+
1348
+ @staticmethod
1349
+ def _optional_int(value):
1350
+ if value is None:
1351
+ return None
1352
+ try:
1353
+ return int(value)
1354
+ except (TypeError, ValueError):
1355
+ return None
1356
+
1357
+ @staticmethod
1358
+ def _float_or_nan(value):
1359
+ try:
1360
+ return float(value)
1361
+ except (TypeError, ValueError):
1362
+ return np.nan
1363
+
1364
+ @staticmethod
1365
+ def _float_or_default(value, default):
1366
+ try:
1367
+ value = float(value)
1368
+ except (TypeError, ValueError):
1369
+ return float(default)
1370
+ return value if np.isfinite(value) else float(default)
1371
+
1372
+ @staticmethod
1373
+ def _finite_float(value, name, channel):
1374
+ try:
1375
+ value = float(value)
1376
+ except (TypeError, ValueError) as exc:
1377
+ raise ValueError(f"Invalid {name} for channel {channel}: {value!r}") from exc
1378
+ if not np.isfinite(value):
1379
+ raise ValueError(f"Invalid {name} for channel {channel}: {value!r}")
1380
+ return value