prideQC 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
prideqc/__init__.py ADDED
@@ -0,0 +1,13 @@
1
+ """QC and technical annotations from one pass over mass spectrometry data."""
2
+
3
+ try:
4
+ from ._version import __version__
5
+ except ImportError: # pragma: no cover - source tree before the Hatch VCS hook runs
6
+ from importlib.metadata import PackageNotFoundError, version
7
+
8
+ try:
9
+ __version__ = version("prideQC")
10
+ except PackageNotFoundError:
11
+ __version__ = "0+unknown"
12
+
13
+ __all__ = ["__version__"]
prideqc/__main__.py ADDED
@@ -0,0 +1,5 @@
1
+ """Support ``python -m prideqc``."""
2
+
3
+ from prideqc.cli import main
4
+
5
+ raise SystemExit(main())
prideqc/_version.py ADDED
@@ -0,0 +1,24 @@
1
+ # file generated by vcs-versioning
2
+ # don't change, don't track in version control
3
+ from __future__ import annotations
4
+
5
+ __all__ = [
6
+ "__version__",
7
+ "__version_tuple__",
8
+ "version",
9
+ "version_tuple",
10
+ "__commit_id__",
11
+ "commit_id",
12
+ ]
13
+
14
+ version: str
15
+ __version__: str
16
+ __version_tuple__: tuple[int | str, ...]
17
+ version_tuple: tuple[int | str, ...]
18
+ commit_id: str | None
19
+ __commit_id__: str | None
20
+
21
+ __version__ = version = '0.2.0'
22
+ __version_tuple__ = version_tuple = (0, 2, 0)
23
+
24
+ __commit_id__ = commit_id = None
prideqc/annotations.py ADDED
@@ -0,0 +1,217 @@
1
+ """Technical evidence and conservative, auditable SDRF suggestions."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import math
6
+ from collections import Counter
7
+
8
+ import numpy as np
9
+
10
+ from prideqc.metrics import RunSummary
11
+ from prideqc.models import Annotation, CVTerm, EvidenceKind, RunMetadata, Spectrum
12
+
13
+ OBSERVED = EvidenceKind.OBSERVED
14
+ INFERRED = EvidenceKind.INFERRED
15
+ UNAVAILABLE = EvidenceKind.UNAVAILABLE
16
+
17
+
18
+ class TechnicalAnnotator:
19
+ """Separate measured header/scan facts from acquisition-mode heuristics."""
20
+
21
+ def annotate(self, metadata: RunMetadata, run: RunSummary) -> list[Annotation]:
22
+ annotations = []
23
+ for name, terms, column in (
24
+ ("instrument", metadata.instruments, "comment[instrument]"),
25
+ ("mass_analyzers", metadata.analyzers, None),
26
+ ("ionization", metadata.ionization, None),
27
+ ):
28
+ annotations.append(Annotation(
29
+ name, [{"accession": t.accession, "name": t.name} for t in terms] or None,
30
+ OBSERVED if terms else UNAVAILABLE, "mzML instrumentConfiguration CV terms",
31
+ "All configurations are reported; only an unambiguous instrument is proposed.",
32
+ sdrf_column=column,
33
+ sdrf_value=terms[0].sdrf_value() if len(terms) == 1 and column else None,
34
+ ))
35
+ annotations.append(Annotation(
36
+ "instrument_details",
37
+ metadata.instrument_details or None,
38
+ OBSERVED if metadata.instrument_details else UNAVAILABLE,
39
+ "OpenMS ExperimentalSettings instrument fields",
40
+ "Plain instrument fields are preserved as observed metadata; no CV accession is inferred.",
41
+ ))
42
+ annotations.append(Annotation("instrument_serial_numbers", metadata.serial_numbers or None,
43
+ OBSERVED if metadata.serial_numbers else UNAVAILABLE,
44
+ "mzML instrument serial number"))
45
+ for level, summary in sorted(run.levels.items()):
46
+ if level < 2:
47
+ continue
48
+ methods = [
49
+ {"accession": key[0], "name": key[1], "count": count}
50
+ for key, count in sorted(summary.activation.items())
51
+ ]
52
+ only = next(iter(summary.activation)) if len(summary.activation) == 1 else None
53
+ complete = summary.missing_activation == 0
54
+ annotations.append(Annotation(
55
+ f"dissociation_ms{level}", methods or None, OBSERVED if methods else UNAVAILABLE,
56
+ "mzML precursor activation; scan counts per method",
57
+ "Co-occurring methods are preserved; missing activation prevents automatic fill.",
58
+ support=summary.count - summary.missing_activation, total=summary.count,
59
+ sdrf_column="comment[dissociation method]" if level == 2 else None,
60
+ sdrf_value=CVTerm(*only).sdrf_value() if only and complete and level == 2 else None,
61
+ ))
62
+ energies = [f"{value:g} {unit}" for value, unit in sorted(summary.collision_energy)]
63
+ annotations.append(Annotation(
64
+ f"collision_energy_ms{level}", energies or None,
65
+ OBSERVED if energies else UNAVAILABLE, "mzML activation energy with preserved units",
66
+ "Multiple recorded values do not establish stepped energy within each scan.",
67
+ support=summary.count - summary.missing_collision_energy, total=summary.count,
68
+ sdrf_column="comment[collision energy]" if level == 2 else None,
69
+ sdrf_value=";".join(
70
+ energies,
71
+ ) if energies and level == 2 and summary.missing_collision_energy == 0 else None,
72
+ ))
73
+ annotations.extend(self._acquisition(run))
74
+ for name, column in (
75
+ ("precursor_mass_tolerance", "comment[precursor mass tolerance]"),
76
+ ("fragment_mass_tolerance", "comment[fragment mass tolerance]"),
77
+ ):
78
+ annotations.append(Annotation(
79
+ name, None, UNAVAILABLE, "No calibrated mass-error estimator configured",
80
+ "Search tolerances cannot be measured from instrument class or isolation width.",
81
+ sdrf_column=column,
82
+ ))
83
+ return annotations
84
+
85
+ def _acquisition(self, run: RunSummary) -> list[Annotation]:
86
+ summary = run.levels.get(2)
87
+ widths = np.asarray(summary.isolation_widths) if summary else np.array([])
88
+ total = summary.count if summary else 0
89
+ label, accession, support, support_total = None, None, 0, total
90
+ method = (
91
+ "acquisition heuristic v2: >=100 MS2, >=90% width coverage; fixed-cycle DIA "
92
+ "requires >=20 cycles, >=8 MS2/cycle, >=90% cycle/set/order consensus, target-grid "
93
+ "agreement, and median width >=5 Th; narrow DDA requires >=500 unique targets"
94
+ )
95
+ detail = (
96
+ "High-specificity inference only. Wide-window consensus supports DIA. Stable repeated "
97
+ "MS1-delimited cycles with non-narrow isolation also support DIA. Narrow isolation is "
98
+ "called DDA only when many distinct precursor targets are observed; small fixed-target "
99
+ "narrow runs abstain because PRM and narrow-window DIA are ambiguous. Width unit is Th (m/z)."
100
+ )
101
+
102
+ if summary is not None and widths.size >= 100 and total and widths.size / total >= 0.9:
103
+ narrow = int(np.count_nonzero(widths <= 3))
104
+ wide = int(np.count_nonzero(widths >= 15))
105
+ narrow_fraction = narrow / widths.size
106
+ wide_fraction = wide / widths.size
107
+ median_width = float(np.median(widths))
108
+
109
+ cycle = run.acquisition_cycles.metrics()
110
+ cycle_count = int(cycle["AcquisitionCycle_Count"] or 0)
111
+ cycle_mode = int(cycle["AcquisitionCycle_MS2Count_Mode"] or 0)
112
+ cycle_modal_fraction = cycle["AcquisitionCycle_MS2Count_ModalFraction"]
113
+ target_coverage = cycle["AcquisitionCycle_TargetCoverageFraction"]
114
+ target_eligible_fraction = cycle["AcquisitionCycle_TargetEligibleFraction"]
115
+ set_modal_fraction = cycle["AcquisitionCycle_TargetSetModalFraction"]
116
+ order_modal_fraction = cycle["AcquisitionCycle_TargetOrderModalFraction"]
117
+
118
+ rounded_targets = np.round(np.asarray(summary.isolation_precursor_mz), 1)
119
+ unique_targets = int(np.unique(rounded_targets).size) if rounded_targets.size else 0
120
+ target_grid_tolerance = max(1, int(round(cycle_mode * 0.1)))
121
+
122
+ stable_cycle = (
123
+ cycle_count >= 20
124
+ and cycle_mode >= 8
125
+ and 8 <= unique_targets <= 200
126
+ and abs(unique_targets - cycle_mode) <= target_grid_tolerance
127
+ and cycle_modal_fraction is not None and cycle_modal_fraction >= 0.9
128
+ and target_coverage is not None and target_coverage >= 0.9
129
+ and target_eligible_fraction is not None and target_eligible_fraction >= 0.9
130
+ and set_modal_fraction is not None and set_modal_fraction >= 0.9
131
+ and order_modal_fraction is not None and order_modal_fraction >= 0.9
132
+ )
133
+
134
+ if wide_fraction >= 0.9:
135
+ label, accession, support = "Data-independent acquisition", "PRIDE:0000450", wide
136
+ elif stable_cycle and median_width >= 5:
137
+ label, accession = "Data-independent acquisition", "PRIDE:0000450"
138
+ modal_cycles = int(round(order_modal_fraction * cycle_count))
139
+ support_total = total
140
+ support = min(total, modal_cycles * cycle_mode)
141
+ elif narrow_fraction >= 0.9 and unique_targets >= 500:
142
+ label, accession, support = "Data-dependent acquisition", "PRIDE:0000627", narrow
143
+
144
+ return [Annotation(
145
+ "acquisition_method", label, INFERRED if label else UNAVAILABLE,
146
+ method,
147
+ detail,
148
+ support=support, total=support_total,
149
+ sdrf_column="comment[proteomics data acquisition method]",
150
+ sdrf_value=CVTerm(accession, label).sdrf_value() if accession and label else None,
151
+ )]
152
+
153
+
154
+ class DiagnosticIonCollector:
155
+ """Optional lightweight signature screen; never asserts a PTM or label plex.
156
+
157
+ Count centroid MS2/MS3 spectra with several reporter/diagnostic ions. No
158
+ database search, mass-shift pairing, random null model, or stale dependency.
159
+ """
160
+
161
+ TARGETS = {
162
+ "TMT_family": (126.127726, 127.131081, 128.134436, 129.137790, 130.141145, 131.138180),
163
+ "iTRAQ_family": (114.1112, 115.1083, 116.1116, 117.1150),
164
+ "glycan_oxonium": (204.0867, 366.1395),
165
+ }
166
+
167
+ def __init__(self, ppm: float = 20.0, min_relative_intensity: float = 0.01) -> None:
168
+ if not math.isfinite(ppm) or ppm <= 0:
169
+ raise ValueError("Diagnostic tolerance must be finite and positive.")
170
+ if not 0 < min_relative_intensity <= 1:
171
+ raise ValueError("Relative intensity threshold must be in (0, 1].")
172
+ self.ppm = ppm
173
+ self.min_relative_intensity = min_relative_intensity
174
+ self.counts: Counter[str] = Counter()
175
+ self.total = 0
176
+ self.skipped_profile = 0
177
+
178
+ def consume_spectrum(self, spectrum: Spectrum) -> None:
179
+ if spectrum.ms_level not in (2, 3):
180
+ return
181
+ if spectrum.representation != "centroid":
182
+ self.skipped_profile += 1
183
+ return
184
+ self.total += 1
185
+ mz, intensity = spectrum.mz, spectrum.intensity
186
+ valid = np.isfinite(mz) & (mz > 0) & np.isfinite(intensity) & (intensity > 0)
187
+ mz, intensity = mz[valid], intensity[valid]
188
+ if not mz.size:
189
+ return
190
+ mz = np.sort(mz[intensity >= intensity.max() * self.min_relative_intensity])
191
+ for name, targets in self.TARGETS.items():
192
+ masses = np.array(targets)
193
+ delta = masses * self.ppm / 1e6
194
+ matches = np.searchsorted(
195
+ mz,
196
+ masses + delta,
197
+ side="right",
198
+ ) > np.searchsorted(
199
+ mz,
200
+ masses - delta,
201
+ side="left",
202
+ )
203
+ required = 2 if name == "glycan_oxonium" else 3
204
+ self.counts[name] += int(np.count_nonzero(matches) >= required)
205
+
206
+ def annotations(self) -> list[Annotation]:
207
+ return [Annotation(
208
+ f"signature_{name}", {"matching_spectra": self.counts[name], "eligible_spectra": self.total,
209
+ "excluded_profile_or_unknown": self.skipped_profile},
210
+ INFERRED if self.counts[name] else UNAVAILABLE,
211
+ f"diagnostic ion screen v1; {self.ppm:g} ppm; relative intensity >= {self.min_relative_intensity:g}",
212
+ (
213
+ "Candidate evidence only; not proof of a modification, enrichment, labeling "
214
+ "channel, or plex. Never written into SDRF automatically."
215
+ ),
216
+ support=self.counts[name], total=self.total,
217
+ ) for name in self.TARGETS]