pytesprocess 0.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. pytesprocess/__init__.py +9 -0
  2. pytesprocess/_version.py +2 -0
  3. pytesprocess/cli/__init__.py +1 -0
  4. pytesprocess/cli/commands/__init__.py +5 -0
  5. pytesprocess/cli/commands/event.py +66 -0
  6. pytesprocess/cli/commands/filter.py +17 -0
  7. pytesprocess/cli/commands/ivsweep.py +29 -0
  8. pytesprocess/cli/common.py +86 -0
  9. pytesprocess/cli/main.py +81 -0
  10. pytesprocess/config/__init__.py +4 -0
  11. pytesprocess/config/loader.py +94 -0
  12. pytesprocess/config/manager.py +297 -0
  13. pytesprocess/config/resolvers/__init__.py +5 -0
  14. pytesprocess/config/resolvers/common.py +56 -0
  15. pytesprocess/config/resolvers/feature.py +293 -0
  16. pytesprocess/config/resolvers/salting.py +86 -0
  17. pytesprocess/config/resolvers/trigger.py +84 -0
  18. pytesprocess/config/selectors.py +108 -0
  19. pytesprocess/config/validation.py +314 -0
  20. pytesprocess/config/warnings.py +2 -0
  21. pytesprocess/core/__init__.py +10 -0
  22. pytesprocess/core/algorithms.py +1455 -0
  23. pytesprocess/core/didv.py +1648 -0
  24. pytesprocess/core/eventbuilder.py +495 -0
  25. pytesprocess/core/filterbuilder.py +81 -0
  26. pytesprocess/core/filterdata.py +1849 -0
  27. pytesprocess/core/ivsweep.py +2072 -0
  28. pytesprocess/core/noise.py +923 -0
  29. pytesprocess/core/noisemodel.py +1408 -0
  30. pytesprocess/core/oftrigger.py +1035 -0
  31. pytesprocess/core/template.py +450 -0
  32. pytesprocess/process/__init__.py +6 -0
  33. pytesprocess/process/data_source.py +185 -0
  34. pytesprocess/process/event_context.py +35 -0
  35. pytesprocess/process/feature_plan.py +186 -0
  36. pytesprocess/process/feature_resources.py +267 -0
  37. pytesprocess/process/features.py +1024 -0
  38. pytesprocess/process/filterprocess.py +1176 -0
  39. pytesprocess/process/ivprocess.py +1380 -0
  40. pytesprocess/process/processing_data.py +967 -0
  41. pytesprocess/process/randoms.py +921 -0
  42. pytesprocess/process/triggers.py +1011 -0
  43. pytesprocess/salting/__init__.py +7 -0
  44. pytesprocess/salting/generator.py +364 -0
  45. pytesprocess/salting/injector.py +329 -0
  46. pytesprocess/salting/sampling.py +84 -0
  47. pytesprocess/utils/__init__.py +5 -0
  48. pytesprocess/utils/arg_utils.py +122 -0
  49. pytesprocess/utils/dataframe_output.py +120 -0
  50. pytesprocess/utils/filter_hdf5.py +594 -0
  51. pytesprocess/utils/utils.py +701 -0
  52. pytesprocess/workflows/__init__.py +3 -0
  53. pytesprocess/workflows/processing.py +317 -0
  54. pytesprocess/workflows/salting.py +133 -0
  55. pytesprocess-0.1.1.dist-info/METADATA +211 -0
  56. pytesprocess-0.1.1.dist-info/RECORD +60 -0
  57. pytesprocess-0.1.1.dist-info/WHEEL +5 -0
  58. pytesprocess-0.1.1.dist-info/entry_points.txt +2 -0
  59. pytesprocess-0.1.1.dist-info/licenses/LICENSE +21 -0
  60. pytesprocess-0.1.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,84 @@
1
+ """Sampling helpers for salting metadata generation."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import cloudpickle
6
+ import numpy as np
7
+
8
+
9
+ def sample_pdf(function, xrange, nsamples=1000, npoints=10000,
10
+ normalize_cdf=True, rng=None):
11
+ """Sample a one-dimensional PDF with inverse-transform sampling."""
12
+ if not callable(function):
13
+ raise TypeError("Input PDF must be callable.")
14
+ if len(xrange) != 2 or float(xrange[1]) <= float(xrange[0]):
15
+ raise ValueError("xrange must contain [xmin, xmax] with xmax > xmin.")
16
+ if int(nsamples) < 0:
17
+ raise ValueError("nsamples must be non-negative.")
18
+ if int(npoints) < 2:
19
+ raise ValueError("npoints must be at least 2.")
20
+
21
+ x = np.linspace(float(xrange[0]), float(xrange[1]), num=int(npoints))
22
+ pdf = np.asarray(function(x), dtype=float)
23
+ if pdf.shape != x.shape:
24
+ raise ValueError("PDF function must return one value per input point.")
25
+ if np.any(~np.isfinite(pdf)) or np.any(pdf < 0):
26
+ raise ValueError("PDF values must be finite and non-negative.")
27
+
28
+ # Cumulative trapezoidal integral without requiring scipy. This works on
29
+ # NumPy 1.26 and current NumPy releases.
30
+ dx = np.diff(x)
31
+ areas = 0.5 * (pdf[:-1] + pdf[1:]) * dx
32
+ cdf = np.concatenate(([0.0], np.cumsum(areas)))
33
+ total = float(cdf[-1])
34
+ if total <= 0:
35
+ raise ValueError("PDF has zero integral over the requested range.")
36
+ if normalize_cdf:
37
+ cdf = cdf / total
38
+
39
+ # np.interp requires monotonically increasing sample points. Flat regions
40
+ # of a PDF create duplicate CDF values, so retain the first occurrence.
41
+ cdf_unique, unique_inds = np.unique(cdf, return_index=True)
42
+ x_unique = x[unique_inds]
43
+ if len(cdf_unique) < 2:
44
+ raise ValueError("Unable to construct an invertible CDF from the PDF.")
45
+
46
+ if rng is None:
47
+ rng = np.random.default_rng()
48
+ high = 1.0 if normalize_cdf else total
49
+ samples = rng.random(int(nsamples)) * high
50
+ return np.interp(samples, cdf_unique, x_unique)
51
+
52
+
53
+ def sample_dm_distributions(pdf_file, nsamples_per_mass, *,
54
+ energy_unit_to_ev=1.0e3,
55
+ energy_range=(1.0e-5, 1.0),
56
+ random_seed=None):
57
+ """Sample the historical cloudpickle DM distribution file.
58
+
59
+ Returns
60
+ -------
61
+ energies_ev : ndarray
62
+ Recoil energies in eV.
63
+ masses_mev : ndarray
64
+ DM mass corresponding to each sampled recoil energy.
65
+ """
66
+ rng = np.random.default_rng(random_seed)
67
+ with open(pdf_file, "rb") as handle:
68
+ distributions = cloudpickle.load(handle)
69
+
70
+ energies = []
71
+ masses = []
72
+ for mass, data in distributions.items():
73
+ if not isinstance(data, dict) or "dmrate" not in data:
74
+ raise ValueError(
75
+ f'DM distribution entry for mass {mass!r} is missing "dmrate".'
76
+ )
77
+ sampled = sample_pdf(
78
+ data["dmrate"], energy_range,
79
+ nsamples=int(nsamples_per_mass), rng=rng,
80
+ )
81
+ energies.extend(np.asarray(sampled) * float(energy_unit_to_ev))
82
+ masses.extend([mass] * len(sampled))
83
+
84
+ return np.asarray(energies, dtype=float), np.asarray(masses)
@@ -0,0 +1,5 @@
1
+ from .utils import *
2
+ from .arg_utils import *
3
+ from .filter_hdf5 import *
4
+
5
+ from .dataframe_output import *
@@ -0,0 +1,122 @@
1
+ import re
2
+
3
+ import numpy as np
4
+
5
+
6
+ def build_range_str(data_list):
7
+ """Build a compact range string such as ``1-3_7_9-11``."""
8
+ data_list.sort()
9
+ units = []
10
+ prev_val = data_list[0]
11
+
12
+ for val in data_list:
13
+ if val == prev_val + 1:
14
+ units[-1].append(val)
15
+ else:
16
+ units.append([val])
17
+ prev_val = val
18
+
19
+ return "_".join(
20
+ f"{unit[0]}-{unit[-1]}" if len(unit) > 1 else str(unit[0])
21
+ for unit in units
22
+ )
23
+
24
+
25
+ def hyphen_range(data_str):
26
+ """Expand strings such as ``1-3,7,9-11`` into a sorted integer list."""
27
+ data_str = "".join(data_str.split())
28
+ data_str = data_str.replace(",", "_")
29
+ values = set()
30
+
31
+ for item in data_str.split("_"):
32
+ bounds = item.split("-")
33
+ if len(bounds) not in (1, 2):
34
+ raise SyntaxError(
35
+ "hyphen_range received an incorrectly formatted argument: "
36
+ + data_str.replace("_", ",")
37
+ )
38
+ if len(bounds) == 1:
39
+ values.add(int(bounds[0]))
40
+ else:
41
+ values.update(range(int(bounds[0]), int(bounds[1]) + 1))
42
+
43
+ return sorted(values)
44
+
45
+
46
+ def extract_list(arg_list):
47
+ """Extract a list separated by whitespace, comma, or semicolon."""
48
+ if not isinstance(arg_list, list):
49
+ arg_list = [arg_list]
50
+
51
+ output_list = []
52
+ for item in arg_list:
53
+ output_list.extend(
54
+ token for token in re.split(r"[\s,;]+", item.strip()) if token
55
+ )
56
+ return output_list
57
+
58
+
59
+ def convert_to_seconds(par):
60
+ """Convert a duration string to seconds."""
61
+ par = str(par)
62
+ if par.isdigit():
63
+ return float(par)
64
+
65
+ seconds_per_unit = {
66
+ "us": 1e-6,
67
+ "ms": 1e-3,
68
+ "s": 1,
69
+ "m": 60,
70
+ "h": 3600,
71
+ "d": 86400,
72
+ "w": 604800,
73
+ }
74
+
75
+ suffix_length = 2 if "us" in par or "ms" in par else 1
76
+ unit = par[-suffix_length:]
77
+ if unit not in seconds_per_unit:
78
+ raise ValueError(f'ERROR: time unit "{unit}" unknown!')
79
+
80
+ return float(par[:-suffix_length]) * seconds_per_unit[unit]
81
+
82
+
83
+ def extract_stream_id(stream):
84
+ if isinstance(stream, str):
85
+ if stream.startswith("I"):
86
+ return stream
87
+ if not stream.isdigit():
88
+ stream = extract_stream_num(stream)
89
+
90
+ stream_str = str(stream)
91
+ return (
92
+ "I"
93
+ + stream_str[:-14]
94
+ + "_D"
95
+ + stream_str[-14:-6]
96
+ + "_T"
97
+ + stream_str[-6:]
98
+ )
99
+
100
+
101
+ def extract_stream_num(stream):
102
+ if isinstance(stream, int):
103
+ return stream
104
+ if not isinstance(stream, str):
105
+ raise ValueError("ERROR in extract_stream_num: stream should be a string")
106
+
107
+ stream_split = stream.split("_")
108
+ if (
109
+ len(stream_split) < 3
110
+ or not stream_split[-1].startswith("T")
111
+ or not stream_split[-2].startswith("D")
112
+ or not stream_split[-3].startswith("I")
113
+ ):
114
+ raise ValueError("ERROR in extract_stream_num: unknown stream name format")
115
+
116
+ return np.uint64(
117
+ float(
118
+ stream_split[-3][1:]
119
+ + stream_split[-2][1:]
120
+ + stream_split[-1][1:]
121
+ )
122
+ )
@@ -0,0 +1,120 @@
1
+ """Common naming and identity helpers for processed dataframe products."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ from datetime import datetime
7
+ from pathlib import Path
8
+ import os
9
+ import stat
10
+
11
+ from .arg_utils import extract_stream_id, extract_stream_num
12
+
13
+
14
+ @dataclass(frozen=True)
15
+ class DataframeGroup:
16
+ """Identity and filesystem location of one processed dataframe product."""
17
+
18
+ dataframe_type: str
19
+ name: str
20
+ group_id: str
21
+ group_number: int
22
+ path: str
23
+ processing_label: str | None = None
24
+
25
+ def file_path(self, file_index: int) -> str:
26
+ index = int(file_index)
27
+ if index < 1:
28
+ raise ValueError("dataframe_file_index must be >= 1")
29
+ return str(Path(self.path) / f"{self.name}_F{index:04d}.hdf5")
30
+
31
+ def row_metadata(self, file_index: int) -> dict:
32
+ return {
33
+ "dataframe_type": self.dataframe_type,
34
+ "dataframe_group_name": self.name,
35
+ "dataframe_group_id": self.group_id,
36
+ "dataframe_group_number": self.group_number,
37
+ "dataframe_file_index": int(file_index),
38
+ "processing_label": self.processing_label,
39
+ }
40
+
41
+
42
+ def create_dataframe_group(
43
+ base_path,
44
+ *,
45
+ dataframe_type,
46
+ facility,
47
+ processing_label=None,
48
+ restricted=False,
49
+ data_type="background",
50
+ output_group_name=None,
51
+ name_prefix=None,
52
+ now=None,
53
+ ):
54
+ """Create one processed-dataframe group with one canonical ``I#_D#_T#`` ID.
55
+
56
+ ``F####`` is intentionally not part of the group identity. It is a shard
57
+ index assigned to an independent processing task/worker.
58
+ """
59
+
60
+ dataframe_type = str(dataframe_type).strip().lower()
61
+ if not dataframe_type:
62
+ raise ValueError("dataframe_type is required")
63
+
64
+ base_path = Path(base_path)
65
+ now = datetime.now() if now is None else now
66
+
67
+ if output_group_name is None:
68
+ group_id = f"I{int(facility)}_D{now:%Y%m%d}_T{now:%H%M%S}"
69
+ prefix = str(name_prefix).strip() if name_prefix else dataframe_type
70
+ if processing_label:
71
+ prefix = f"{processing_label}_{prefix}"
72
+ if restricted:
73
+ prefix += "_restricted"
74
+ elif data_type == "calibration":
75
+ prefix += "_calibration"
76
+ group_name = f"{prefix}_{group_id}"
77
+ output_dir = base_path / group_name
78
+ else:
79
+ group_name = str(output_group_name)
80
+ try:
81
+ group_id = extract_stream_id(group_name)
82
+ except Exception as exc:
83
+ raise ValueError(
84
+ "output_group_name must contain a canonical I#_Dyyyymmdd_Thhmmss ID"
85
+ ) from exc
86
+ output_dir = base_path if base_path.name == group_name else base_path / group_name
87
+
88
+ group_number = int(extract_stream_num(group_id))
89
+ output_dir.mkdir(parents=True, exist_ok=True)
90
+ try:
91
+ os.chmod(
92
+ output_dir,
93
+ stat.S_IRWXG | stat.S_IRWXU | stat.S_IROTH | stat.S_IXOTH,
94
+ )
95
+ except OSError:
96
+ # Directory creation is the important operation; chmod can legitimately
97
+ # fail on some shared/object-backed filesystems.
98
+ pass
99
+
100
+ return DataframeGroup(
101
+ dataframe_type=dataframe_type,
102
+ name=group_name,
103
+ group_id=group_id,
104
+ group_number=group_number,
105
+ path=str(output_dir),
106
+ processing_label=processing_label,
107
+ )
108
+
109
+
110
+ def add_dataframe_group_columns(dataframe, group: DataframeGroup, file_index: int):
111
+ """Attach canonical dataframe-group identity columns to a Vaex dataframe."""
112
+ import numpy as np
113
+
114
+ count = len(dataframe)
115
+ for key, value in group.row_metadata(file_index).items():
116
+ if value is None:
117
+ dataframe[key] = np.asarray([None] * count, dtype=object)
118
+ else:
119
+ dataframe[key] = np.asarray([value] * count)
120
+ return dataframe