prideQC 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- prideqc/__init__.py +13 -0
- prideqc/__main__.py +5 -0
- prideqc/_version.py +24 -0
- prideqc/annotations.py +217 -0
- prideqc/cli.py +503 -0
- prideqc/cohort.py +781 -0
- prideqc/conversion.py +48 -0
- prideqc/data/llm-adjudication-request-v1.schema.json +280 -0
- prideqc/data/llm-refinement-decision-v1.schema.json +71 -0
- prideqc/data/llm-refinement-packet-v1.schema.json +344 -0
- prideqc/io.py +45 -0
- prideqc/local_llm.py +556 -0
- prideqc/mass_error.py +1171 -0
- prideqc/mass_shift.py +1143 -0
- prideqc/metrics.py +572 -0
- prideqc/models.py +176 -0
- prideqc/mzqc.py +519 -0
- prideqc/pipeline.py +1102 -0
- prideqc/pride.py +187 -0
- prideqc/py.typed +0 -0
- prideqc/readers.py +532 -0
- prideqc/refinement_adjudication.py +341 -0
- prideqc/refinement_model.py +645 -0
- prideqc/refinement_packet.py +916 -0
- prideqc/refinement_policy.py +649 -0
- prideqc/sdrf.py +340 -0
- prideqc/submission.py +661 -0
- prideqc/validation.py +72 -0
- prideqc-0.2.0.dist-info/METADATA +118 -0
- prideqc-0.2.0.dist-info/RECORD +35 -0
- prideqc-0.2.0.dist-info/WHEEL +4 -0
- prideqc-0.2.0.dist-info/entry_points.txt +2 -0
- prideqc-0.2.0.dist-info/licenses/LICENSE +201 -0
- prideqc-0.2.0.dist-info/licenses/NOTICE +34 -0
- prideqc-0.2.0.dist-info/licenses/licenses/rawQC-MIT.txt +21 -0
prideqc/__init__.py
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
"""QC and technical annotations from one pass over mass spectrometry data."""
|
|
2
|
+
|
|
3
|
+
try:
|
|
4
|
+
from ._version import __version__
|
|
5
|
+
except ImportError: # pragma: no cover - source tree before the Hatch VCS hook runs
|
|
6
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
7
|
+
|
|
8
|
+
try:
|
|
9
|
+
__version__ = version("prideQC")
|
|
10
|
+
except PackageNotFoundError:
|
|
11
|
+
__version__ = "0+unknown"
|
|
12
|
+
|
|
13
|
+
__all__ = ["__version__"]
|
prideqc/__main__.py
ADDED
prideqc/_version.py
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# file generated by vcs-versioning
|
|
2
|
+
# don't change, don't track in version control
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
__all__ = [
|
|
6
|
+
"__version__",
|
|
7
|
+
"__version_tuple__",
|
|
8
|
+
"version",
|
|
9
|
+
"version_tuple",
|
|
10
|
+
"__commit_id__",
|
|
11
|
+
"commit_id",
|
|
12
|
+
]
|
|
13
|
+
|
|
14
|
+
version: str
|
|
15
|
+
__version__: str
|
|
16
|
+
__version_tuple__: tuple[int | str, ...]
|
|
17
|
+
version_tuple: tuple[int | str, ...]
|
|
18
|
+
commit_id: str | None
|
|
19
|
+
__commit_id__: str | None
|
|
20
|
+
|
|
21
|
+
__version__ = version = '0.2.0'
|
|
22
|
+
__version_tuple__ = version_tuple = (0, 2, 0)
|
|
23
|
+
|
|
24
|
+
__commit_id__ = commit_id = None
|
prideqc/annotations.py
ADDED
|
@@ -0,0 +1,217 @@
|
|
|
1
|
+
"""Technical evidence and conservative, auditable SDRF suggestions."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import math
|
|
6
|
+
from collections import Counter
|
|
7
|
+
|
|
8
|
+
import numpy as np
|
|
9
|
+
|
|
10
|
+
from prideqc.metrics import RunSummary
|
|
11
|
+
from prideqc.models import Annotation, CVTerm, EvidenceKind, RunMetadata, Spectrum
|
|
12
|
+
|
|
13
|
+
OBSERVED = EvidenceKind.OBSERVED
|
|
14
|
+
INFERRED = EvidenceKind.INFERRED
|
|
15
|
+
UNAVAILABLE = EvidenceKind.UNAVAILABLE
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class TechnicalAnnotator:
|
|
19
|
+
"""Separate measured header/scan facts from acquisition-mode heuristics."""
|
|
20
|
+
|
|
21
|
+
def annotate(self, metadata: RunMetadata, run: RunSummary) -> list[Annotation]:
|
|
22
|
+
annotations = []
|
|
23
|
+
for name, terms, column in (
|
|
24
|
+
("instrument", metadata.instruments, "comment[instrument]"),
|
|
25
|
+
("mass_analyzers", metadata.analyzers, None),
|
|
26
|
+
("ionization", metadata.ionization, None),
|
|
27
|
+
):
|
|
28
|
+
annotations.append(Annotation(
|
|
29
|
+
name, [{"accession": t.accession, "name": t.name} for t in terms] or None,
|
|
30
|
+
OBSERVED if terms else UNAVAILABLE, "mzML instrumentConfiguration CV terms",
|
|
31
|
+
"All configurations are reported; only an unambiguous instrument is proposed.",
|
|
32
|
+
sdrf_column=column,
|
|
33
|
+
sdrf_value=terms[0].sdrf_value() if len(terms) == 1 and column else None,
|
|
34
|
+
))
|
|
35
|
+
annotations.append(Annotation(
|
|
36
|
+
"instrument_details",
|
|
37
|
+
metadata.instrument_details or None,
|
|
38
|
+
OBSERVED if metadata.instrument_details else UNAVAILABLE,
|
|
39
|
+
"OpenMS ExperimentalSettings instrument fields",
|
|
40
|
+
"Plain instrument fields are preserved as observed metadata; no CV accession is inferred.",
|
|
41
|
+
))
|
|
42
|
+
annotations.append(Annotation("instrument_serial_numbers", metadata.serial_numbers or None,
|
|
43
|
+
OBSERVED if metadata.serial_numbers else UNAVAILABLE,
|
|
44
|
+
"mzML instrument serial number"))
|
|
45
|
+
for level, summary in sorted(run.levels.items()):
|
|
46
|
+
if level < 2:
|
|
47
|
+
continue
|
|
48
|
+
methods = [
|
|
49
|
+
{"accession": key[0], "name": key[1], "count": count}
|
|
50
|
+
for key, count in sorted(summary.activation.items())
|
|
51
|
+
]
|
|
52
|
+
only = next(iter(summary.activation)) if len(summary.activation) == 1 else None
|
|
53
|
+
complete = summary.missing_activation == 0
|
|
54
|
+
annotations.append(Annotation(
|
|
55
|
+
f"dissociation_ms{level}", methods or None, OBSERVED if methods else UNAVAILABLE,
|
|
56
|
+
"mzML precursor activation; scan counts per method",
|
|
57
|
+
"Co-occurring methods are preserved; missing activation prevents automatic fill.",
|
|
58
|
+
support=summary.count - summary.missing_activation, total=summary.count,
|
|
59
|
+
sdrf_column="comment[dissociation method]" if level == 2 else None,
|
|
60
|
+
sdrf_value=CVTerm(*only).sdrf_value() if only and complete and level == 2 else None,
|
|
61
|
+
))
|
|
62
|
+
energies = [f"{value:g} {unit}" for value, unit in sorted(summary.collision_energy)]
|
|
63
|
+
annotations.append(Annotation(
|
|
64
|
+
f"collision_energy_ms{level}", energies or None,
|
|
65
|
+
OBSERVED if energies else UNAVAILABLE, "mzML activation energy with preserved units",
|
|
66
|
+
"Multiple recorded values do not establish stepped energy within each scan.",
|
|
67
|
+
support=summary.count - summary.missing_collision_energy, total=summary.count,
|
|
68
|
+
sdrf_column="comment[collision energy]" if level == 2 else None,
|
|
69
|
+
sdrf_value=";".join(
|
|
70
|
+
energies,
|
|
71
|
+
) if energies and level == 2 and summary.missing_collision_energy == 0 else None,
|
|
72
|
+
))
|
|
73
|
+
annotations.extend(self._acquisition(run))
|
|
74
|
+
for name, column in (
|
|
75
|
+
("precursor_mass_tolerance", "comment[precursor mass tolerance]"),
|
|
76
|
+
("fragment_mass_tolerance", "comment[fragment mass tolerance]"),
|
|
77
|
+
):
|
|
78
|
+
annotations.append(Annotation(
|
|
79
|
+
name, None, UNAVAILABLE, "No calibrated mass-error estimator configured",
|
|
80
|
+
"Search tolerances cannot be measured from instrument class or isolation width.",
|
|
81
|
+
sdrf_column=column,
|
|
82
|
+
))
|
|
83
|
+
return annotations
|
|
84
|
+
|
|
85
|
+
def _acquisition(self, run: RunSummary) -> list[Annotation]:
|
|
86
|
+
summary = run.levels.get(2)
|
|
87
|
+
widths = np.asarray(summary.isolation_widths) if summary else np.array([])
|
|
88
|
+
total = summary.count if summary else 0
|
|
89
|
+
label, accession, support, support_total = None, None, 0, total
|
|
90
|
+
method = (
|
|
91
|
+
"acquisition heuristic v2: >=100 MS2, >=90% width coverage; fixed-cycle DIA "
|
|
92
|
+
"requires >=20 cycles, >=8 MS2/cycle, >=90% cycle/set/order consensus, target-grid "
|
|
93
|
+
"agreement, and median width >=5 Th; narrow DDA requires >=500 unique targets"
|
|
94
|
+
)
|
|
95
|
+
detail = (
|
|
96
|
+
"High-specificity inference only. Wide-window consensus supports DIA. Stable repeated "
|
|
97
|
+
"MS1-delimited cycles with non-narrow isolation also support DIA. Narrow isolation is "
|
|
98
|
+
"called DDA only when many distinct precursor targets are observed; small fixed-target "
|
|
99
|
+
"narrow runs abstain because PRM and narrow-window DIA are ambiguous. Width unit is Th (m/z)."
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
if summary is not None and widths.size >= 100 and total and widths.size / total >= 0.9:
|
|
103
|
+
narrow = int(np.count_nonzero(widths <= 3))
|
|
104
|
+
wide = int(np.count_nonzero(widths >= 15))
|
|
105
|
+
narrow_fraction = narrow / widths.size
|
|
106
|
+
wide_fraction = wide / widths.size
|
|
107
|
+
median_width = float(np.median(widths))
|
|
108
|
+
|
|
109
|
+
cycle = run.acquisition_cycles.metrics()
|
|
110
|
+
cycle_count = int(cycle["AcquisitionCycle_Count"] or 0)
|
|
111
|
+
cycle_mode = int(cycle["AcquisitionCycle_MS2Count_Mode"] or 0)
|
|
112
|
+
cycle_modal_fraction = cycle["AcquisitionCycle_MS2Count_ModalFraction"]
|
|
113
|
+
target_coverage = cycle["AcquisitionCycle_TargetCoverageFraction"]
|
|
114
|
+
target_eligible_fraction = cycle["AcquisitionCycle_TargetEligibleFraction"]
|
|
115
|
+
set_modal_fraction = cycle["AcquisitionCycle_TargetSetModalFraction"]
|
|
116
|
+
order_modal_fraction = cycle["AcquisitionCycle_TargetOrderModalFraction"]
|
|
117
|
+
|
|
118
|
+
rounded_targets = np.round(np.asarray(summary.isolation_precursor_mz), 1)
|
|
119
|
+
unique_targets = int(np.unique(rounded_targets).size) if rounded_targets.size else 0
|
|
120
|
+
target_grid_tolerance = max(1, int(round(cycle_mode * 0.1)))
|
|
121
|
+
|
|
122
|
+
stable_cycle = (
|
|
123
|
+
cycle_count >= 20
|
|
124
|
+
and cycle_mode >= 8
|
|
125
|
+
and 8 <= unique_targets <= 200
|
|
126
|
+
and abs(unique_targets - cycle_mode) <= target_grid_tolerance
|
|
127
|
+
and cycle_modal_fraction is not None and cycle_modal_fraction >= 0.9
|
|
128
|
+
and target_coverage is not None and target_coverage >= 0.9
|
|
129
|
+
and target_eligible_fraction is not None and target_eligible_fraction >= 0.9
|
|
130
|
+
and set_modal_fraction is not None and set_modal_fraction >= 0.9
|
|
131
|
+
and order_modal_fraction is not None and order_modal_fraction >= 0.9
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
if wide_fraction >= 0.9:
|
|
135
|
+
label, accession, support = "Data-independent acquisition", "PRIDE:0000450", wide
|
|
136
|
+
elif stable_cycle and median_width >= 5:
|
|
137
|
+
label, accession = "Data-independent acquisition", "PRIDE:0000450"
|
|
138
|
+
modal_cycles = int(round(order_modal_fraction * cycle_count))
|
|
139
|
+
support_total = total
|
|
140
|
+
support = min(total, modal_cycles * cycle_mode)
|
|
141
|
+
elif narrow_fraction >= 0.9 and unique_targets >= 500:
|
|
142
|
+
label, accession, support = "Data-dependent acquisition", "PRIDE:0000627", narrow
|
|
143
|
+
|
|
144
|
+
return [Annotation(
|
|
145
|
+
"acquisition_method", label, INFERRED if label else UNAVAILABLE,
|
|
146
|
+
method,
|
|
147
|
+
detail,
|
|
148
|
+
support=support, total=support_total,
|
|
149
|
+
sdrf_column="comment[proteomics data acquisition method]",
|
|
150
|
+
sdrf_value=CVTerm(accession, label).sdrf_value() if accession and label else None,
|
|
151
|
+
)]
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
class DiagnosticIonCollector:
|
|
155
|
+
"""Optional lightweight signature screen; never asserts a PTM or label plex.
|
|
156
|
+
|
|
157
|
+
Count centroid MS2/MS3 spectra with several reporter/diagnostic ions. No
|
|
158
|
+
database search, mass-shift pairing, random null model, or stale dependency.
|
|
159
|
+
"""
|
|
160
|
+
|
|
161
|
+
TARGETS = {
|
|
162
|
+
"TMT_family": (126.127726, 127.131081, 128.134436, 129.137790, 130.141145, 131.138180),
|
|
163
|
+
"iTRAQ_family": (114.1112, 115.1083, 116.1116, 117.1150),
|
|
164
|
+
"glycan_oxonium": (204.0867, 366.1395),
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
def __init__(self, ppm: float = 20.0, min_relative_intensity: float = 0.01) -> None:
|
|
168
|
+
if not math.isfinite(ppm) or ppm <= 0:
|
|
169
|
+
raise ValueError("Diagnostic tolerance must be finite and positive.")
|
|
170
|
+
if not 0 < min_relative_intensity <= 1:
|
|
171
|
+
raise ValueError("Relative intensity threshold must be in (0, 1].")
|
|
172
|
+
self.ppm = ppm
|
|
173
|
+
self.min_relative_intensity = min_relative_intensity
|
|
174
|
+
self.counts: Counter[str] = Counter()
|
|
175
|
+
self.total = 0
|
|
176
|
+
self.skipped_profile = 0
|
|
177
|
+
|
|
178
|
+
def consume_spectrum(self, spectrum: Spectrum) -> None:
|
|
179
|
+
if spectrum.ms_level not in (2, 3):
|
|
180
|
+
return
|
|
181
|
+
if spectrum.representation != "centroid":
|
|
182
|
+
self.skipped_profile += 1
|
|
183
|
+
return
|
|
184
|
+
self.total += 1
|
|
185
|
+
mz, intensity = spectrum.mz, spectrum.intensity
|
|
186
|
+
valid = np.isfinite(mz) & (mz > 0) & np.isfinite(intensity) & (intensity > 0)
|
|
187
|
+
mz, intensity = mz[valid], intensity[valid]
|
|
188
|
+
if not mz.size:
|
|
189
|
+
return
|
|
190
|
+
mz = np.sort(mz[intensity >= intensity.max() * self.min_relative_intensity])
|
|
191
|
+
for name, targets in self.TARGETS.items():
|
|
192
|
+
masses = np.array(targets)
|
|
193
|
+
delta = masses * self.ppm / 1e6
|
|
194
|
+
matches = np.searchsorted(
|
|
195
|
+
mz,
|
|
196
|
+
masses + delta,
|
|
197
|
+
side="right",
|
|
198
|
+
) > np.searchsorted(
|
|
199
|
+
mz,
|
|
200
|
+
masses - delta,
|
|
201
|
+
side="left",
|
|
202
|
+
)
|
|
203
|
+
required = 2 if name == "glycan_oxonium" else 3
|
|
204
|
+
self.counts[name] += int(np.count_nonzero(matches) >= required)
|
|
205
|
+
|
|
206
|
+
def annotations(self) -> list[Annotation]:
|
|
207
|
+
return [Annotation(
|
|
208
|
+
f"signature_{name}", {"matching_spectra": self.counts[name], "eligible_spectra": self.total,
|
|
209
|
+
"excluded_profile_or_unknown": self.skipped_profile},
|
|
210
|
+
INFERRED if self.counts[name] else UNAVAILABLE,
|
|
211
|
+
f"diagnostic ion screen v1; {self.ppm:g} ppm; relative intensity >= {self.min_relative_intensity:g}",
|
|
212
|
+
(
|
|
213
|
+
"Candidate evidence only; not proof of a modification, enrichment, labeling "
|
|
214
|
+
"channel, or plex. Never written into SDRF automatically."
|
|
215
|
+
),
|
|
216
|
+
support=self.counts[name], total=self.total,
|
|
217
|
+
) for name in self.TARGETS]
|