labmcp-ms-data 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1 @@
1
+ """LabMCP server for Mass Spectrometry Data (mzML, Bruker TDF, vendor conversion)."""
@@ -0,0 +1,148 @@
1
+ """Numerical helpers: chromatogram downsampling, XIC peak integration, spectrum peak lists (numpy only)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import csv
6
+ from dataclasses import dataclass
7
+ from pathlib import Path
8
+
9
+ import numpy as np
10
+
11
+ PROTON_MASS = 1.007276
12
+
13
+
14
+ def downsample_max_indices(y: np.ndarray, max_points: int) -> np.ndarray:
15
+ """Indices kept by :func:`downsample_max` (the highest point of each of ``max_points`` bins)."""
16
+ y = np.asarray(y, dtype=float)
17
+ if y.size <= max_points:
18
+ return np.arange(y.size)
19
+ bins = np.array_split(np.arange(y.size), max_points)
20
+ return np.array([b[int(np.argmax(y[b]))] for b in bins if b.size], dtype=np.int64)
21
+
22
+
23
+ def downsample_max(x: np.ndarray, y: np.ndarray, max_points: int) -> tuple[list[float], list[float]]:
24
+ """Split into ``max_points`` contiguous bins and keep, per bin, the point with the highest ``y``.
25
+
26
+ Keeping the maximum (not the mean) preserves chromatographic peak heights and apex positions.
27
+ """
28
+ x = np.asarray(x, dtype=float)
29
+ y = np.asarray(y, dtype=float)
30
+ idx = downsample_max_indices(y, max_points)
31
+ return x[idx].tolist(), y[idx].tolist()
32
+
33
+
34
+ def trapezoid(y: np.ndarray, x: np.ndarray) -> float:
35
+ y = np.asarray(y, dtype=float)
36
+ x = np.asarray(x, dtype=float)
37
+ if y.size < 2:
38
+ return 0.0
39
+ return float(np.sum((y[1:] + y[:-1]) * np.diff(x)) / 2.0)
40
+
41
+
42
+ @dataclass
43
+ class ChromPeak:
44
+ apex_rt_min: float
45
+ apex_intensity: float
46
+ start_rt_min: float
47
+ end_rt_min: float
48
+ area: float # intensity x minutes, baseline not subtracted
49
+ fwhm_s: float | None
50
+ points_across_peak: int
51
+
52
+
53
+ def _half_crossing(
54
+ x: np.ndarray, y: np.ndarray, i: int, level: float, step: int, lo: int, hi: int
55
+ ) -> float | None:
56
+ """RT where the trace first falls below ``level`` walking from ``i``, within ``lo..hi`` only."""
57
+ j = i
58
+ while lo <= j + step <= hi:
59
+ k = j + step
60
+ if y[k] < level:
61
+ frac = (y[j] - level) / (y[j] - y[k]) if y[j] != y[k] else 0.0
62
+ return float(x[j] + frac * (x[k] - x[j]))
63
+ j = k
64
+ return None
65
+
66
+
67
+ def integrate_apex_peak(
68
+ rt_min: np.ndarray, y: np.ndarray, *, stop_fraction: float = 0.01
69
+ ) -> ChromPeak | None:
70
+ """Find the most intense point and integrate the peak around it.
71
+
72
+ The peak extends from the apex on each side until the trace falls below ``stop_fraction`` (1 %) of the
73
+ apex or reaches a valley (a point lower than the next 2 points on that side). Area is the
74
+ trapezoidal integral of intensity over retention time in minutes, without baseline subtraction.
75
+ """
76
+ rt = np.asarray(rt_min, dtype=float)
77
+ y = np.asarray(y, dtype=float)
78
+ if y.size == 0 or not np.any(y > 0):
79
+ return None
80
+ i = int(np.argmax(y))
81
+ apex = float(y[i])
82
+ stop = stop_fraction * apex
83
+
84
+ def walk(step: int) -> int:
85
+ j = i
86
+ while 0 <= j + step < y.size:
87
+ k = j + step
88
+ if y[k] <= stop:
89
+ return k
90
+ k2 = k + step
91
+ k3 = k2 + step
92
+ if 0 <= k3 < y.size and y[k2] > y[k] and y[k3] > y[k]:
93
+ return k # valley
94
+ j = k
95
+ return j
96
+
97
+ lo, hi = walk(-1), walk(+1)
98
+ # FWHM of the apex peak only: a co-eluting neighbour beyond a valley must not widen it.
99
+ left = _half_crossing(rt, y, i, apex / 2, -1, lo, hi)
100
+ right = _half_crossing(rt, y, i, apex / 2, +1, lo, hi)
101
+ fwhm = (right - left) * 60.0 if left is not None and right is not None else None
102
+ return ChromPeak(
103
+ apex_rt_min=float(rt[i]),
104
+ apex_intensity=apex,
105
+ start_rt_min=float(rt[lo]),
106
+ end_rt_min=float(rt[hi]),
107
+ area=trapezoid(y[lo : hi + 1], rt[lo : hi + 1]),
108
+ fwhm_s=fwhm,
109
+ points_across_peak=int(hi - lo + 1),
110
+ )
111
+
112
+
113
+ def tolerance_da(mz: float, tol: float, unit: str) -> float:
114
+ return mz * tol * 1e-6 if unit == "ppm" else tol
115
+
116
+
117
+ def local_maxima(mz: np.ndarray, intensity: np.ndarray) -> tuple[np.ndarray, np.ndarray]:
118
+ """Crude centroiding of profile data: keep points higher than both neighbours."""
119
+ if intensity.size < 3:
120
+ return mz, intensity
121
+ y = intensity
122
+ keep = np.zeros(y.size, dtype=bool)
123
+ keep[1:-1] = (y[1:-1] > y[:-2]) & (y[1:-1] >= y[2:])
124
+ keep[0] = y[0] > y[1]
125
+ keep[-1] = y[-1] > y[-2]
126
+ return mz[keep], y[keep]
127
+
128
+
129
+ def write_csv(path: Path, header: list[str], columns: list[np.ndarray | list], *, overwrite: bool = False) -> str:
130
+ """Write equal-length columns to CSV and return the absolute path.
131
+
132
+ Without ``overwrite`` the file is created exclusively (``FileExistsError`` if it appeared meanwhile).
133
+ """
134
+ path.parent.mkdir(parents=True, exist_ok=True)
135
+ with path.open("w" if overwrite else "x", newline="", encoding="utf-8") as fh:
136
+ writer = csv.writer(fh)
137
+ writer.writerow(header)
138
+ for row in zip(*columns, strict=True):
139
+ out = []
140
+ for v in row:
141
+ if v is None or (isinstance(v, float) and not np.isfinite(v)):
142
+ out.append("")
143
+ elif isinstance(v, (float, np.floating)):
144
+ out.append(f"{float(v):.8g}")
145
+ else:
146
+ out.append(str(v))
147
+ writer.writerow(out)
148
+ return str(path)
@@ -0,0 +1,277 @@
1
+ """Vendor-file conversion to mzML with a converter the USER has installed.
2
+
3
+ No vendor libraries are bundled or downloaded (their licences do not allow redistribution).
4
+ Supported converters (command lines checked against the official documentation):
5
+
6
+ * **ProteoWizard msconvert** (https://proteowizard.sourceforge.io/tools/msconvert.html), native
7
+ on Windows with the vendor readers::
8
+
9
+ msconvert <input> -o <outdir> --mzML --zlib [--gzip] [--filter "peakPicking vendor msLevel=1-"]
10
+
11
+ (``peakPicking [<PickerType> [snr=] [peakSpace=] [msLevel=<ms_levels>]]``; it must be the first
12
+ filter to use the vendor centroiding.)
13
+ * **msconvert in Docker** on Linux/macOS, using the image
14
+ ``proteowizard/pwiz-skyline-i-agree-to-the-vendor-licenses`` (by pulling it the user accepts
15
+ the vendor licences; https://hub.docker.com/r/proteowizard/pwiz-skyline-i-agree-to-the-vendor-licenses)::
16
+
17
+ docker run --rm -e WINEDEBUG=-all -v <input dir>:/data:ro -v <output dir>:/out <image> \
18
+ wine msconvert /data/<file> -o /out ...
19
+
20
+ Only the input's folder (read-only) and the output folder are mounted.
21
+
22
+ * **ThermoRawFileParser** (https://github.com/compomics/ThermoRawFileParser), cross-platform .NET,
23
+ Thermo ``.raw`` only. Options only work in ``-option=value`` form::
24
+
25
+ ThermoRawFileParser -i=<input.raw> -o=<outdir> -f=2 [-p] [-g]
26
+
27
+ (``-f=2`` indexed mzML; ``-p`` disables the native Thermo peak picking; ``-g`` gzips the output.)
28
+
29
+ Argument lists are built as lists and run without a shell, with a timeout. On timeout the whole
30
+ process tree is killed (converters are often started through wrapper scripts), and a Docker
31
+ container is killed by name.
32
+ """
33
+
34
+ from __future__ import annotations
35
+
36
+ import os
37
+ import platform
38
+ import shutil
39
+ import signal
40
+ import subprocess
41
+ import time
42
+ import uuid
43
+ from dataclasses import dataclass, field
44
+ from pathlib import Path
45
+
46
+ from labmcp import InstrumentProtocolError
47
+
48
+ from labmcp_ms_data.files import FORMATS, run_stem
49
+
50
+ DOCKER_IMAGE = "proteowizard/pwiz-skyline-i-agree-to-the-vendor-licenses"
51
+ CONVERTERS = ("auto", "msconvert", "docker", "thermorawfileparser")
52
+ CONVERTIBLE = {"thermo_raw", "waters_raw", "agilent_d", "bruker_baf", "bruker_tdf", "sciex_wiff",
53
+ "shimadzu_lcd", "mzxml"} # fmt: skip
54
+ _TRFP_NAMES = ("ThermoRawFileParser", "ThermoRawFileParser.sh", "ThermoRawFileParser.exe")
55
+ _MSCONVERT_NAMES = ("msconvert", "msconvert.exe")
56
+
57
+
58
+ @dataclass
59
+ class ConversionPlan:
60
+ converter: str
61
+ argv: list[str]
62
+ output: Path
63
+ notes: list[str] = field(default_factory=list)
64
+ container_name: str | None = None
65
+
66
+
67
+ @dataclass
68
+ class ConversionResult:
69
+ returncode: int | None
70
+ duration_s: float
71
+ stdout_tail: str
72
+ stderr_tail: str
73
+ timed_out: bool
74
+
75
+
76
+ def _which(names: tuple[str, ...]) -> str | None:
77
+ for n in names:
78
+ found = shutil.which(n)
79
+ if found:
80
+ return found
81
+ return None
82
+
83
+
84
+ def _launcher(path: str) -> list[str]:
85
+ """How to start a .NET tool given as a .dll / .exe path (ThermoRawFileParser releases)."""
86
+ low = path.lower()
87
+ if low.endswith(".dll"):
88
+ return ["dotnet", path]
89
+ if low.endswith(".exe") and os.name != "nt":
90
+ return ["mono", path]
91
+ return [path]
92
+
93
+
94
+ def choose_converter(fmt: str, requested: str, converter_path: str | None) -> tuple[str, str | None]:
95
+ """Return (converter, executable) for ``fmt``; raise with install hints if none is usable."""
96
+ requested = (requested or "auto").lower()
97
+ if requested not in CONVERTERS:
98
+ raise InstrumentProtocolError(
99
+ f"--option converter must be one of {', '.join(CONVERTERS)} (got {requested!r})."
100
+ )
101
+ if requested == "thermorawfileparser" and fmt != "thermo_raw":
102
+ raise InstrumentProtocolError(
103
+ "ThermoRawFileParser only reads Thermo .raw files. Use converter=msconvert or converter=docker."
104
+ )
105
+ if requested != "auto":
106
+ exe = converter_path
107
+ if exe is None:
108
+ exe = {
109
+ "msconvert": _which(_MSCONVERT_NAMES),
110
+ "docker": _which(("docker",)),
111
+ "thermorawfileparser": _which(_TRFP_NAMES),
112
+ }[requested]
113
+ return requested, exe
114
+ if converter_path: # auto, but the user named the program: tell which one it is from its name
115
+ name = Path(converter_path).name.lower()
116
+ for conv, key in (("thermorawfileparser", "thermorawfileparser"), ("docker", "docker"),
117
+ ("msconvert", "msconvert")): # fmt: skip
118
+ if key in name:
119
+ return choose_converter(fmt, conv, converter_path)
120
+ raise InstrumentProtocolError(
121
+ f"Cannot tell which converter {converter_path!r} is. Add --option converter=msconvert, "
122
+ "docker or thermorawfileparser."
123
+ )
124
+ order = ["thermorawfileparser", "msconvert", "docker"] if fmt == "thermo_raw" else ["msconvert", "docker"]
125
+ for conv in order:
126
+ exe = {
127
+ "msconvert": _which(_MSCONVERT_NAMES),
128
+ "docker": _which(("docker",)),
129
+ "thermorawfileparser": _which(_TRFP_NAMES),
130
+ }[conv]
131
+ if exe:
132
+ return conv, exe
133
+ return "none", None
134
+
135
+
136
+ INSTALL_HELP = (
137
+ "No converter found. Install one and restart the server with --option converter=...: "
138
+ "ThermoRawFileParser for Thermo .raw (https://github.com/compomics/ThermoRawFileParser, any OS); "
139
+ "ProteoWizard msconvert on Windows (https://proteowizard.sourceforge.io/download.html); or Docker with "
140
+ f"`docker pull {DOCKER_IMAGE}` on Linux/macOS (pulling it means you accept the vendor licences). "
141
+ "Use --option converter_path=<path> if the program is not on PATH."
142
+ )
143
+
144
+
145
+ def plan_conversion(
146
+ input_path: Path,
147
+ fmt: str,
148
+ out_dir: Path,
149
+ *,
150
+ converter: str = "auto",
151
+ converter_path: str | None = None,
152
+ docker_image: str | None = None,
153
+ peak_picking: bool = True,
154
+ gzip: bool = False,
155
+ ) -> ConversionPlan:
156
+ if fmt not in CONVERTIBLE:
157
+ raise InstrumentProtocolError(
158
+ f"{input_path.name} is {FORMATS.get(fmt, ('', fmt, False))[1]}; it can be read directly, no conversion needed."
159
+ if fmt in ("mzml", "mzmlb")
160
+ else f"{input_path.name} is not a convertible vendor format."
161
+ )
162
+ conv, exe = choose_converter(fmt, converter, converter_path)
163
+ if conv == "none" or not exe:
164
+ raise InstrumentProtocolError(
165
+ INSTALL_HELP if conv == "none" else f"{conv} was not found on PATH. {INSTALL_HELP}"
166
+ )
167
+ ext = ".mzML.gz" if gzip else ".mzML"
168
+ output = out_dir / (run_stem(input_path) + ext)
169
+ notes: list[str] = []
170
+ if conv == "thermorawfileparser":
171
+ argv = _launcher(exe) + [f"-i={input_path}", f"-o={out_dir}", "-f=2"]
172
+ if not peak_picking:
173
+ argv.append("-p")
174
+ if gzip:
175
+ argv.append("-g")
176
+ return ConversionPlan(conv, argv, output, notes)
177
+ flags = ["--mzML", "--zlib"]
178
+ if gzip:
179
+ flags.append("--gzip")
180
+ if peak_picking:
181
+ flags += ["--filter", "peakPicking vendor msLevel=1-"]
182
+ if conv == "msconvert":
183
+ if os.name != "nt":
184
+ notes.append(
185
+ "Native msconvert reads vendor formats only on Windows; on Linux/macOS use converter=docker."
186
+ )
187
+ return ConversionPlan(conv, [exe, str(input_path), "-o", str(out_dir), *flags], output, notes)
188
+ # docker + wine
189
+ image = docker_image or DOCKER_IMAGE
190
+ name = f"labmcp-msconvert-{uuid.uuid4().hex[:12]}"
191
+ in_dir = input_path.parent # the whole folder: SCIEX .wiff needs its .wiff.scan sidecar
192
+ for d in (in_dir, out_dir):
193
+ if not _docker_mountable(d):
194
+ raise InstrumentProtocolError(
195
+ f"Docker cannot mount {str(d)!r}: the path contains ':' or ','. Rename the folder, or use "
196
+ "another converter."
197
+ )
198
+ argv = [exe, "run", "--rm", "--name", name, "-e", "WINEDEBUG=-all"]
199
+ if platform.machine().lower() in ("arm64", "aarch64"):
200
+ argv += ["--platform", "linux/amd64"] # the image is x86-64 only (runs under emulation)
201
+ notes.append("Apple Silicon / ARM: the x86-64 image runs under emulation and is slow.")
202
+ # Input folder read-only; only the output folder is writable.
203
+ argv += ["-v", f"{in_dir}:/data:ro", "-v", f"{out_dir}:/out"]
204
+ argv += [image, "wine", "msconvert", f"/data/{input_path.name}", "-o", "/out", *flags]
205
+ if os.name == "posix" and platform.system() == "Linux":
206
+ notes.append("On Linux the output file is owned by root (written inside the container).")
207
+ return ConversionPlan(conv, argv, output, notes, container_name=name)
208
+
209
+
210
+ def _docker_mountable(path: Path) -> bool:
211
+ """False if ``docker -v <path>:...`` would split the path (':' separates fields; ',' breaks --mount)."""
212
+ text = str(path)
213
+ if os.name == "nt" and len(text) >= 2 and text[1] == ":":
214
+ text = text[2:] # the drive letter
215
+ return ":" not in text and "," not in text
216
+
217
+
218
+ def _tail(text: str | bytes | None, n: int = 1500) -> str:
219
+ if text is None:
220
+ return ""
221
+ if isinstance(text, bytes):
222
+ text = text.decode(errors="replace")
223
+ return text[-n:]
224
+
225
+
226
+ def _kill_tree(proc: subprocess.Popen[str]) -> None:
227
+ """Kill the converter and everything it started (wrapper scripts start mono/dotnet/wine)."""
228
+ try:
229
+ if os.name == "nt":
230
+ subprocess.run( # noqa: S603, S607
231
+ ["taskkill", "/T", "/F", "/PID", str(proc.pid)], capture_output=True, timeout=30, check=False
232
+ )
233
+ else:
234
+ os.killpg(proc.pid, signal.SIGKILL) # the child leads its own session / process group
235
+ except (OSError, subprocess.SubprocessError):
236
+ pass
237
+ try:
238
+ proc.kill()
239
+ except OSError:
240
+ pass
241
+
242
+
243
+ def run_conversion(plan: ConversionPlan, timeout_s: float) -> ConversionResult:
244
+ t0 = time.monotonic()
245
+ group: dict[str, object] = (
246
+ {"creationflags": subprocess.CREATE_NEW_PROCESS_GROUP} if os.name == "nt" else {"start_new_session": True}
247
+ )
248
+ try:
249
+ proc = subprocess.Popen( # noqa: S603 - argv list, no shell
250
+ plan.argv,
251
+ stdin=subprocess.DEVNULL,
252
+ stdout=subprocess.PIPE,
253
+ stderr=subprocess.PIPE,
254
+ text=True,
255
+ errors="replace",
256
+ shell=False,
257
+ **group, # type: ignore[arg-type]
258
+ )
259
+ except OSError as exc:
260
+ raise InstrumentProtocolError(f"Could not start {plan.argv[0]!r}: {exc}. {INSTALL_HELP}") from exc
261
+ try:
262
+ out, err = proc.communicate(timeout=timeout_s)
263
+ except subprocess.TimeoutExpired as exc:
264
+ _kill_tree(proc)
265
+ if plan.container_name: # killing the docker client does not stop the container
266
+ try:
267
+ subprocess.run( # noqa: S603
268
+ [plan.argv[0], "kill", plan.container_name], capture_output=True, timeout=30, check=False
269
+ )
270
+ except (OSError, subprocess.SubprocessError):
271
+ pass
272
+ try:
273
+ out, err = proc.communicate(timeout=10)
274
+ except subprocess.TimeoutExpired: # something still holds the pipes; give up on the log
275
+ out, err = exc.stdout, exc.stderr
276
+ return ConversionResult(None, time.monotonic() - t0, _tail(out), _tail(err), True)
277
+ return ConversionResult(proc.returncode, time.monotonic() - t0, _tail(out), _tail(err), False)