pytesprocess 0.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pytesprocess/__init__.py +9 -0
- pytesprocess/_version.py +2 -0
- pytesprocess/cli/__init__.py +1 -0
- pytesprocess/cli/commands/__init__.py +5 -0
- pytesprocess/cli/commands/event.py +66 -0
- pytesprocess/cli/commands/filter.py +17 -0
- pytesprocess/cli/commands/ivsweep.py +29 -0
- pytesprocess/cli/common.py +86 -0
- pytesprocess/cli/main.py +81 -0
- pytesprocess/config/__init__.py +4 -0
- pytesprocess/config/loader.py +94 -0
- pytesprocess/config/manager.py +297 -0
- pytesprocess/config/resolvers/__init__.py +5 -0
- pytesprocess/config/resolvers/common.py +56 -0
- pytesprocess/config/resolvers/feature.py +293 -0
- pytesprocess/config/resolvers/salting.py +86 -0
- pytesprocess/config/resolvers/trigger.py +84 -0
- pytesprocess/config/selectors.py +108 -0
- pytesprocess/config/validation.py +314 -0
- pytesprocess/config/warnings.py +2 -0
- pytesprocess/core/__init__.py +10 -0
- pytesprocess/core/algorithms.py +1455 -0
- pytesprocess/core/didv.py +1648 -0
- pytesprocess/core/eventbuilder.py +495 -0
- pytesprocess/core/filterbuilder.py +81 -0
- pytesprocess/core/filterdata.py +1849 -0
- pytesprocess/core/ivsweep.py +2072 -0
- pytesprocess/core/noise.py +923 -0
- pytesprocess/core/noisemodel.py +1408 -0
- pytesprocess/core/oftrigger.py +1035 -0
- pytesprocess/core/template.py +450 -0
- pytesprocess/process/__init__.py +6 -0
- pytesprocess/process/data_source.py +185 -0
- pytesprocess/process/event_context.py +35 -0
- pytesprocess/process/feature_plan.py +186 -0
- pytesprocess/process/feature_resources.py +267 -0
- pytesprocess/process/features.py +1024 -0
- pytesprocess/process/filterprocess.py +1176 -0
- pytesprocess/process/ivprocess.py +1380 -0
- pytesprocess/process/processing_data.py +967 -0
- pytesprocess/process/randoms.py +921 -0
- pytesprocess/process/triggers.py +1011 -0
- pytesprocess/salting/__init__.py +7 -0
- pytesprocess/salting/generator.py +364 -0
- pytesprocess/salting/injector.py +329 -0
- pytesprocess/salting/sampling.py +84 -0
- pytesprocess/utils/__init__.py +5 -0
- pytesprocess/utils/arg_utils.py +122 -0
- pytesprocess/utils/dataframe_output.py +120 -0
- pytesprocess/utils/filter_hdf5.py +594 -0
- pytesprocess/utils/utils.py +701 -0
- pytesprocess/workflows/__init__.py +3 -0
- pytesprocess/workflows/processing.py +317 -0
- pytesprocess/workflows/salting.py +133 -0
- pytesprocess-0.1.1.dist-info/METADATA +211 -0
- pytesprocess-0.1.1.dist-info/RECORD +60 -0
- pytesprocess-0.1.1.dist-info/WHEEL +5 -0
- pytesprocess-0.1.1.dist-info/entry_points.txt +2 -0
- pytesprocess-0.1.1.dist-info/licenses/LICENSE +21 -0
- pytesprocess-0.1.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
"""Sampling helpers for salting metadata generation."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import cloudpickle
|
|
6
|
+
import numpy as np
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def sample_pdf(function, xrange, nsamples=1000, npoints=10000,
|
|
10
|
+
normalize_cdf=True, rng=None):
|
|
11
|
+
"""Sample a one-dimensional PDF with inverse-transform sampling."""
|
|
12
|
+
if not callable(function):
|
|
13
|
+
raise TypeError("Input PDF must be callable.")
|
|
14
|
+
if len(xrange) != 2 or float(xrange[1]) <= float(xrange[0]):
|
|
15
|
+
raise ValueError("xrange must contain [xmin, xmax] with xmax > xmin.")
|
|
16
|
+
if int(nsamples) < 0:
|
|
17
|
+
raise ValueError("nsamples must be non-negative.")
|
|
18
|
+
if int(npoints) < 2:
|
|
19
|
+
raise ValueError("npoints must be at least 2.")
|
|
20
|
+
|
|
21
|
+
x = np.linspace(float(xrange[0]), float(xrange[1]), num=int(npoints))
|
|
22
|
+
pdf = np.asarray(function(x), dtype=float)
|
|
23
|
+
if pdf.shape != x.shape:
|
|
24
|
+
raise ValueError("PDF function must return one value per input point.")
|
|
25
|
+
if np.any(~np.isfinite(pdf)) or np.any(pdf < 0):
|
|
26
|
+
raise ValueError("PDF values must be finite and non-negative.")
|
|
27
|
+
|
|
28
|
+
# Cumulative trapezoidal integral without requiring scipy. This works on
|
|
29
|
+
# NumPy 1.26 and current NumPy releases.
|
|
30
|
+
dx = np.diff(x)
|
|
31
|
+
areas = 0.5 * (pdf[:-1] + pdf[1:]) * dx
|
|
32
|
+
cdf = np.concatenate(([0.0], np.cumsum(areas)))
|
|
33
|
+
total = float(cdf[-1])
|
|
34
|
+
if total <= 0:
|
|
35
|
+
raise ValueError("PDF has zero integral over the requested range.")
|
|
36
|
+
if normalize_cdf:
|
|
37
|
+
cdf = cdf / total
|
|
38
|
+
|
|
39
|
+
# np.interp requires monotonically increasing sample points. Flat regions
|
|
40
|
+
# of a PDF create duplicate CDF values, so retain the first occurrence.
|
|
41
|
+
cdf_unique, unique_inds = np.unique(cdf, return_index=True)
|
|
42
|
+
x_unique = x[unique_inds]
|
|
43
|
+
if len(cdf_unique) < 2:
|
|
44
|
+
raise ValueError("Unable to construct an invertible CDF from the PDF.")
|
|
45
|
+
|
|
46
|
+
if rng is None:
|
|
47
|
+
rng = np.random.default_rng()
|
|
48
|
+
high = 1.0 if normalize_cdf else total
|
|
49
|
+
samples = rng.random(int(nsamples)) * high
|
|
50
|
+
return np.interp(samples, cdf_unique, x_unique)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def sample_dm_distributions(pdf_file, nsamples_per_mass, *,
|
|
54
|
+
energy_unit_to_ev=1.0e3,
|
|
55
|
+
energy_range=(1.0e-5, 1.0),
|
|
56
|
+
random_seed=None):
|
|
57
|
+
"""Sample the historical cloudpickle DM distribution file.
|
|
58
|
+
|
|
59
|
+
Returns
|
|
60
|
+
-------
|
|
61
|
+
energies_ev : ndarray
|
|
62
|
+
Recoil energies in eV.
|
|
63
|
+
masses_mev : ndarray
|
|
64
|
+
DM mass corresponding to each sampled recoil energy.
|
|
65
|
+
"""
|
|
66
|
+
rng = np.random.default_rng(random_seed)
|
|
67
|
+
with open(pdf_file, "rb") as handle:
|
|
68
|
+
distributions = cloudpickle.load(handle)
|
|
69
|
+
|
|
70
|
+
energies = []
|
|
71
|
+
masses = []
|
|
72
|
+
for mass, data in distributions.items():
|
|
73
|
+
if not isinstance(data, dict) or "dmrate" not in data:
|
|
74
|
+
raise ValueError(
|
|
75
|
+
f'DM distribution entry for mass {mass!r} is missing "dmrate".'
|
|
76
|
+
)
|
|
77
|
+
sampled = sample_pdf(
|
|
78
|
+
data["dmrate"], energy_range,
|
|
79
|
+
nsamples=int(nsamples_per_mass), rng=rng,
|
|
80
|
+
)
|
|
81
|
+
energies.extend(np.asarray(sampled) * float(energy_unit_to_ev))
|
|
82
|
+
masses.extend([mass] * len(sampled))
|
|
83
|
+
|
|
84
|
+
return np.asarray(energies, dtype=float), np.asarray(masses)
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
import re
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def build_range_str(data_list):
|
|
7
|
+
"""Build a compact range string such as ``1-3_7_9-11``."""
|
|
8
|
+
data_list.sort()
|
|
9
|
+
units = []
|
|
10
|
+
prev_val = data_list[0]
|
|
11
|
+
|
|
12
|
+
for val in data_list:
|
|
13
|
+
if val == prev_val + 1:
|
|
14
|
+
units[-1].append(val)
|
|
15
|
+
else:
|
|
16
|
+
units.append([val])
|
|
17
|
+
prev_val = val
|
|
18
|
+
|
|
19
|
+
return "_".join(
|
|
20
|
+
f"{unit[0]}-{unit[-1]}" if len(unit) > 1 else str(unit[0])
|
|
21
|
+
for unit in units
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def hyphen_range(data_str):
|
|
26
|
+
"""Expand strings such as ``1-3,7,9-11`` into a sorted integer list."""
|
|
27
|
+
data_str = "".join(data_str.split())
|
|
28
|
+
data_str = data_str.replace(",", "_")
|
|
29
|
+
values = set()
|
|
30
|
+
|
|
31
|
+
for item in data_str.split("_"):
|
|
32
|
+
bounds = item.split("-")
|
|
33
|
+
if len(bounds) not in (1, 2):
|
|
34
|
+
raise SyntaxError(
|
|
35
|
+
"hyphen_range received an incorrectly formatted argument: "
|
|
36
|
+
+ data_str.replace("_", ",")
|
|
37
|
+
)
|
|
38
|
+
if len(bounds) == 1:
|
|
39
|
+
values.add(int(bounds[0]))
|
|
40
|
+
else:
|
|
41
|
+
values.update(range(int(bounds[0]), int(bounds[1]) + 1))
|
|
42
|
+
|
|
43
|
+
return sorted(values)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def extract_list(arg_list):
|
|
47
|
+
"""Extract a list separated by whitespace, comma, or semicolon."""
|
|
48
|
+
if not isinstance(arg_list, list):
|
|
49
|
+
arg_list = [arg_list]
|
|
50
|
+
|
|
51
|
+
output_list = []
|
|
52
|
+
for item in arg_list:
|
|
53
|
+
output_list.extend(
|
|
54
|
+
token for token in re.split(r"[\s,;]+", item.strip()) if token
|
|
55
|
+
)
|
|
56
|
+
return output_list
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def convert_to_seconds(par):
|
|
60
|
+
"""Convert a duration string to seconds."""
|
|
61
|
+
par = str(par)
|
|
62
|
+
if par.isdigit():
|
|
63
|
+
return float(par)
|
|
64
|
+
|
|
65
|
+
seconds_per_unit = {
|
|
66
|
+
"us": 1e-6,
|
|
67
|
+
"ms": 1e-3,
|
|
68
|
+
"s": 1,
|
|
69
|
+
"m": 60,
|
|
70
|
+
"h": 3600,
|
|
71
|
+
"d": 86400,
|
|
72
|
+
"w": 604800,
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
suffix_length = 2 if "us" in par or "ms" in par else 1
|
|
76
|
+
unit = par[-suffix_length:]
|
|
77
|
+
if unit not in seconds_per_unit:
|
|
78
|
+
raise ValueError(f'ERROR: time unit "{unit}" unknown!')
|
|
79
|
+
|
|
80
|
+
return float(par[:-suffix_length]) * seconds_per_unit[unit]
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def extract_stream_id(stream):
|
|
84
|
+
if isinstance(stream, str):
|
|
85
|
+
if stream.startswith("I"):
|
|
86
|
+
return stream
|
|
87
|
+
if not stream.isdigit():
|
|
88
|
+
stream = extract_stream_num(stream)
|
|
89
|
+
|
|
90
|
+
stream_str = str(stream)
|
|
91
|
+
return (
|
|
92
|
+
"I"
|
|
93
|
+
+ stream_str[:-14]
|
|
94
|
+
+ "_D"
|
|
95
|
+
+ stream_str[-14:-6]
|
|
96
|
+
+ "_T"
|
|
97
|
+
+ stream_str[-6:]
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def extract_stream_num(stream):
|
|
102
|
+
if isinstance(stream, int):
|
|
103
|
+
return stream
|
|
104
|
+
if not isinstance(stream, str):
|
|
105
|
+
raise ValueError("ERROR in extract_stream_num: stream should be a string")
|
|
106
|
+
|
|
107
|
+
stream_split = stream.split("_")
|
|
108
|
+
if (
|
|
109
|
+
len(stream_split) < 3
|
|
110
|
+
or not stream_split[-1].startswith("T")
|
|
111
|
+
or not stream_split[-2].startswith("D")
|
|
112
|
+
or not stream_split[-3].startswith("I")
|
|
113
|
+
):
|
|
114
|
+
raise ValueError("ERROR in extract_stream_num: unknown stream name format")
|
|
115
|
+
|
|
116
|
+
return np.uint64(
|
|
117
|
+
float(
|
|
118
|
+
stream_split[-3][1:]
|
|
119
|
+
+ stream_split[-2][1:]
|
|
120
|
+
+ stream_split[-1][1:]
|
|
121
|
+
)
|
|
122
|
+
)
|
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
"""Common naming and identity helpers for processed dataframe products."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from datetime import datetime
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
import os
|
|
9
|
+
import stat
|
|
10
|
+
|
|
11
|
+
from .arg_utils import extract_stream_id, extract_stream_num
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass(frozen=True)
|
|
15
|
+
class DataframeGroup:
|
|
16
|
+
"""Identity and filesystem location of one processed dataframe product."""
|
|
17
|
+
|
|
18
|
+
dataframe_type: str
|
|
19
|
+
name: str
|
|
20
|
+
group_id: str
|
|
21
|
+
group_number: int
|
|
22
|
+
path: str
|
|
23
|
+
processing_label: str | None = None
|
|
24
|
+
|
|
25
|
+
def file_path(self, file_index: int) -> str:
|
|
26
|
+
index = int(file_index)
|
|
27
|
+
if index < 1:
|
|
28
|
+
raise ValueError("dataframe_file_index must be >= 1")
|
|
29
|
+
return str(Path(self.path) / f"{self.name}_F{index:04d}.hdf5")
|
|
30
|
+
|
|
31
|
+
def row_metadata(self, file_index: int) -> dict:
|
|
32
|
+
return {
|
|
33
|
+
"dataframe_type": self.dataframe_type,
|
|
34
|
+
"dataframe_group_name": self.name,
|
|
35
|
+
"dataframe_group_id": self.group_id,
|
|
36
|
+
"dataframe_group_number": self.group_number,
|
|
37
|
+
"dataframe_file_index": int(file_index),
|
|
38
|
+
"processing_label": self.processing_label,
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def create_dataframe_group(
|
|
43
|
+
base_path,
|
|
44
|
+
*,
|
|
45
|
+
dataframe_type,
|
|
46
|
+
facility,
|
|
47
|
+
processing_label=None,
|
|
48
|
+
restricted=False,
|
|
49
|
+
data_type="background",
|
|
50
|
+
output_group_name=None,
|
|
51
|
+
name_prefix=None,
|
|
52
|
+
now=None,
|
|
53
|
+
):
|
|
54
|
+
"""Create one processed-dataframe group with one canonical ``I#_D#_T#`` ID.
|
|
55
|
+
|
|
56
|
+
``F####`` is intentionally not part of the group identity. It is a shard
|
|
57
|
+
index assigned to an independent processing task/worker.
|
|
58
|
+
"""
|
|
59
|
+
|
|
60
|
+
dataframe_type = str(dataframe_type).strip().lower()
|
|
61
|
+
if not dataframe_type:
|
|
62
|
+
raise ValueError("dataframe_type is required")
|
|
63
|
+
|
|
64
|
+
base_path = Path(base_path)
|
|
65
|
+
now = datetime.now() if now is None else now
|
|
66
|
+
|
|
67
|
+
if output_group_name is None:
|
|
68
|
+
group_id = f"I{int(facility)}_D{now:%Y%m%d}_T{now:%H%M%S}"
|
|
69
|
+
prefix = str(name_prefix).strip() if name_prefix else dataframe_type
|
|
70
|
+
if processing_label:
|
|
71
|
+
prefix = f"{processing_label}_{prefix}"
|
|
72
|
+
if restricted:
|
|
73
|
+
prefix += "_restricted"
|
|
74
|
+
elif data_type == "calibration":
|
|
75
|
+
prefix += "_calibration"
|
|
76
|
+
group_name = f"{prefix}_{group_id}"
|
|
77
|
+
output_dir = base_path / group_name
|
|
78
|
+
else:
|
|
79
|
+
group_name = str(output_group_name)
|
|
80
|
+
try:
|
|
81
|
+
group_id = extract_stream_id(group_name)
|
|
82
|
+
except Exception as exc:
|
|
83
|
+
raise ValueError(
|
|
84
|
+
"output_group_name must contain a canonical I#_Dyyyymmdd_Thhmmss ID"
|
|
85
|
+
) from exc
|
|
86
|
+
output_dir = base_path if base_path.name == group_name else base_path / group_name
|
|
87
|
+
|
|
88
|
+
group_number = int(extract_stream_num(group_id))
|
|
89
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
90
|
+
try:
|
|
91
|
+
os.chmod(
|
|
92
|
+
output_dir,
|
|
93
|
+
stat.S_IRWXG | stat.S_IRWXU | stat.S_IROTH | stat.S_IXOTH,
|
|
94
|
+
)
|
|
95
|
+
except OSError:
|
|
96
|
+
# Directory creation is the important operation; chmod can legitimately
|
|
97
|
+
# fail on some shared/object-backed filesystems.
|
|
98
|
+
pass
|
|
99
|
+
|
|
100
|
+
return DataframeGroup(
|
|
101
|
+
dataframe_type=dataframe_type,
|
|
102
|
+
name=group_name,
|
|
103
|
+
group_id=group_id,
|
|
104
|
+
group_number=group_number,
|
|
105
|
+
path=str(output_dir),
|
|
106
|
+
processing_label=processing_label,
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def add_dataframe_group_columns(dataframe, group: DataframeGroup, file_index: int):
|
|
111
|
+
"""Attach canonical dataframe-group identity columns to a Vaex dataframe."""
|
|
112
|
+
import numpy as np
|
|
113
|
+
|
|
114
|
+
count = len(dataframe)
|
|
115
|
+
for key, value in group.row_metadata(file_index).items():
|
|
116
|
+
if value is None:
|
|
117
|
+
dataframe[key] = np.asarray([None] * count, dtype=object)
|
|
118
|
+
else:
|
|
119
|
+
dataframe[key] = np.asarray([value] * count)
|
|
120
|
+
return dataframe
|