spectrseqtools 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- spectrseqtools/__init__.py +0 -0
- spectrseqtools/assets/element_masses.tsv +6 -0
- spectrseqtools/assets/elemental_composition.tsv +5 -0
- spectrseqtools/assets/masses.tsv +145 -0
- spectrseqtools/cli.py +259 -0
- spectrseqtools/common.py +93 -0
- spectrseqtools/deconvolution.py +411 -0
- spectrseqtools/fragment_classification.py +139 -0
- spectrseqtools/linear_program.py +325 -0
- spectrseqtools/mass_explanation.py +320 -0
- spectrseqtools/mass_table.py +487 -0
- spectrseqtools/masses.py +160 -0
- spectrseqtools/plotting.py +157 -0
- spectrseqtools/prediction.py +322 -0
- spectrseqtools/preprocessing.py +130 -0
- spectrseqtools/singleton_identification.py +225 -0
- spectrseqtools/skeleton_building.py +516 -0
- spectrseqtools/utils.py +68 -0
- spectrseqtools-0.1.0.dist-info/METADATA +26 -0
- spectrseqtools-0.1.0.dist-info/RECORD +23 -0
- spectrseqtools-0.1.0.dist-info/WHEEL +4 -0
- spectrseqtools-0.1.0.dist-info/entry_points.txt +2 -0
- spectrseqtools-0.1.0.dist-info/licenses/LICENSE +674 -0
|
File without changes
|
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
nucleoside canonical_name monoisotopic_mass modification_rate
|
|
2
|
+
A A 267.09675 1.0
|
|
3
|
+
C C 243.08552 1.0
|
|
4
|
+
G G 283.09167 1.0
|
|
5
|
+
U U 244.06954 1.0
|
|
6
|
+
0A Am 281.1124 1.0
|
|
7
|
+
00A Ar(p) 479.1053 1.0
|
|
8
|
+
01A m1Am 295.1281 1.0
|
|
9
|
+
019A m1Im 296.1121 1.0
|
|
10
|
+
06A m6Am 295.1281 1.0
|
|
11
|
+
066A m6,6Am 309.1437 1.0
|
|
12
|
+
09A Im 282.0964 1.0
|
|
13
|
+
1A m1A 281.1124 1.0
|
|
14
|
+
19A m1I 282.0964 1.0
|
|
15
|
+
2A m2A 281.1124 1.0
|
|
16
|
+
21161A msms2i6A 427.1347 1.0
|
|
17
|
+
2160A ms2io6A 397.142 1.0
|
|
18
|
+
2161A ms2i6A 381.1471 1.0
|
|
19
|
+
2162A ms2t6A 458.122 1.0
|
|
20
|
+
2163A ms2hn6A 472.1376 1.0
|
|
21
|
+
2164A ms2ct6A 440.1114 1.0
|
|
22
|
+
2165A ht6A 428.1292 1.0
|
|
23
|
+
28A m2,8A 295.1281 1.0
|
|
24
|
+
6A m6A 281.1124 1.0
|
|
25
|
+
60A io6A 351.1543 1.0
|
|
26
|
+
61A i6A 335.1594 1.0
|
|
27
|
+
62A t6A 412.1343 1.0
|
|
28
|
+
621A ms2m6A 327.1001 1.0
|
|
29
|
+
63A hn6A 426.1499 1.0
|
|
30
|
+
64A ac6A 309.1073 1.0
|
|
31
|
+
65A g6A 368.108 1.0
|
|
32
|
+
66A m6,6A 295.1281 1.0
|
|
33
|
+
662A m6t6A 426.1499 1.0
|
|
34
|
+
67A f6A 295.0917 1.0
|
|
35
|
+
68A hm6A 297.1073 1.0
|
|
36
|
+
69A ct6A 394.1237 1.0
|
|
37
|
+
8A m8A 281.1124 1.0
|
|
38
|
+
9A I 268.0808 1.0
|
|
39
|
+
0C Cm 257.10117055 1.0
|
|
40
|
+
04C m4Cm 271.11682061 1.0
|
|
41
|
+
042C ac4Cm 299.11173523 1.0
|
|
42
|
+
044C m4,4Cm 285.13247067 1.0
|
|
43
|
+
05C m5Cm 271.11682061 1.0
|
|
44
|
+
051C hm5Cm 287.1117 1.0
|
|
45
|
+
071C f5Cm 285.09608517 1.0
|
|
46
|
+
2C s2C 259.06267687 1.0
|
|
47
|
+
20C C+ 355.1965 1.0
|
|
48
|
+
21C k2C 371.18048347 1.0
|
|
49
|
+
3C m3C 257.10117055 1.0
|
|
50
|
+
4C m4C 257.10117055 1.0
|
|
51
|
+
42C ac4C 285.09608517 1.0
|
|
52
|
+
44C m4,4C 271.11682061 1.0
|
|
53
|
+
5C m5C 257.10117055 1.0
|
|
54
|
+
50C ho5C 259.08043511 1.0
|
|
55
|
+
51C hm5C 273.09608517 1.0
|
|
56
|
+
71C f5C 271.08043511 1.0
|
|
57
|
+
0G Gm 297.1073 1.0
|
|
58
|
+
00G Gr(p) 495.1003 1.0
|
|
59
|
+
01G m1Gm 311.123 1.0
|
|
60
|
+
02G m2Gm 311.123 1.0
|
|
61
|
+
022G m2,2Gm 325.1386 1.0
|
|
62
|
+
027G m2,7Gm 327.1543 1.0
|
|
63
|
+
1G m1G 297.1073 1.0
|
|
64
|
+
10G Q 409.1597 1.0
|
|
65
|
+
100G preQ0 307.0917 1.0
|
|
66
|
+
101G preQ1 311.123 1.0
|
|
67
|
+
102G oQ 425.1547 1.0
|
|
68
|
+
103G G+ 324.1182 1.0
|
|
69
|
+
104G galQ 571.2126 1.0
|
|
70
|
+
105G gluQ 538.2023 1.0
|
|
71
|
+
106G manQ 571.2126 1.0
|
|
72
|
+
2G m2G 297.1073 1.0
|
|
73
|
+
22G m2,2G 311.123 1.0
|
|
74
|
+
227G m2,2,7G 327.1543 1.0
|
|
75
|
+
27G m2,7G 313.1386 1.0
|
|
76
|
+
34G imG 335.123 1.0
|
|
77
|
+
342G mimG 349.1386 1.0
|
|
78
|
+
347G yW-72 436.1706 1.0
|
|
79
|
+
3470G OHyWx 452.1655 1.0
|
|
80
|
+
348G yW-58 450.1863 1.0
|
|
81
|
+
3480G OHyWy 466.1812 1.0
|
|
82
|
+
3483G yW 508.1918 1.0
|
|
83
|
+
34830G OHyW 524.1867 1.0
|
|
84
|
+
34832G o2yW 540.1816 1.0
|
|
85
|
+
4G imG-14 321.1073 1.0
|
|
86
|
+
42G imG2 335.123 1.0
|
|
87
|
+
47G yW-86 422.155 1.0
|
|
88
|
+
7G m7G 299.123 1.0
|
|
89
|
+
0U Um 258.08518614 1.0
|
|
90
|
+
02U s2Um 274.06234252 1.0
|
|
91
|
+
03U m3Um 272.1008362 1.0
|
|
92
|
+
05U m5Um 272.1008362 1.0
|
|
93
|
+
0503U mcmo5Um 346.10123012 1.0
|
|
94
|
+
051U cmnm5Um 345.11721453 1.0
|
|
95
|
+
0521U mcm5Um 330.1063155 1.0
|
|
96
|
+
0522U mchm5Um 346.10123012 1.0
|
|
97
|
+
053U ncm5Um 315.10664985 1.0
|
|
98
|
+
0583U inm5Um 355.17433547 1.0
|
|
99
|
+
09U Ym 258.08518614 1.0
|
|
100
|
+
1309U m1acp3Y 359.13286459 1.0
|
|
101
|
+
19U m1Y 258.08518614 1.0
|
|
102
|
+
2U s2U 260.04669246 1.0
|
|
103
|
+
20U se2U 307.9911 1.0
|
|
104
|
+
2051U cmnm5se2U 395.0232 1.0
|
|
105
|
+
20510U nm5se2U 337.0177 1.0
|
|
106
|
+
20511U mnm5se2U 365.049 1.0
|
|
107
|
+
21U ges2U 396.17189294 1.0
|
|
108
|
+
2151U cmnm5ges2U 483.20392133 1.0
|
|
109
|
+
21510U nm5ges2U 425.19844 1.0
|
|
110
|
+
21511U mnm5ges2U 439.21409209 1.0
|
|
111
|
+
25U m5s2U 274.0623 1.0
|
|
112
|
+
251U cmnm5s2U 347.07872085 1.0
|
|
113
|
+
2510U nm5s2U 289.07324155 1.0
|
|
114
|
+
2511U mnm5s2U 303.08889161 1.0
|
|
115
|
+
2521U mcm5s2U 332.06782182 1.0
|
|
116
|
+
253U ncm5s2U 317.06815617 1.0
|
|
117
|
+
254U tm5s2U 397.06135653 1.0
|
|
118
|
+
2540U cm5s2U 318.05217176 1.0
|
|
119
|
+
2583U inm5s2U 357.13584179 1.0
|
|
120
|
+
3U m3U 258.08518614 1.0
|
|
121
|
+
30U acp3U 345.11721453 1.0
|
|
122
|
+
308U acp3D 347.13286459 1.0
|
|
123
|
+
309U acp3Y 345.11721453 1.0
|
|
124
|
+
39U m3Y 258.08518614 1.0
|
|
125
|
+
5U m5U 258.08518614 1.0
|
|
126
|
+
50U ho5U 260.0644507 1.0
|
|
127
|
+
501U mo5U 274.08010076 1.0
|
|
128
|
+
502U cmo5U 318.06993 1.0
|
|
129
|
+
503U mcmo5U 332.08558006 1.0
|
|
130
|
+
51U cmnm5U 331.10156447 1.0
|
|
131
|
+
510U nm5U 273.09608517 1.0
|
|
132
|
+
511U mnm5U 287.11173523 1.0
|
|
133
|
+
52U cm5U 302.07501538 1.0
|
|
134
|
+
520U chm5U 318.06993 1.0
|
|
135
|
+
521U mcm5U 316.09066544 1.0
|
|
136
|
+
522U mchm5U 332.08558006 1.0
|
|
137
|
+
53U ncm5U 301.09099979 1.0
|
|
138
|
+
531U nchm5U 317.08591441 1.0
|
|
139
|
+
54U tm5U 381.08420015 1.0
|
|
140
|
+
55U cnm5U 283.08043511 1.0
|
|
141
|
+
58U m5D 260.1008362 1.0
|
|
142
|
+
583U inm5U 341.15868541 1.0
|
|
143
|
+
74U s4U 260.04669246 1.0
|
|
144
|
+
8U D 246.08518614 1.0
|
|
145
|
+
9U Y 244.06953608 1.0
|
spectrseqtools/cli.py
ADDED
|
@@ -0,0 +1,259 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import polars as pl
|
|
3
|
+
import yaml
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from tap import Tap
|
|
6
|
+
from typing import List, Literal
|
|
7
|
+
|
|
8
|
+
from spectrseqtools.fragment_classification import classify_fragments
|
|
9
|
+
from spectrseqtools.mass_table import DynamicProgrammingTable, SequenceInformation
|
|
10
|
+
from spectrseqtools.masses import (
|
|
11
|
+
COMPRESSION_RATE,
|
|
12
|
+
DEFAULT_INTENSITY_CUTOFF,
|
|
13
|
+
EXPLANATION_MASSES,
|
|
14
|
+
MATCHING_THRESHOLD,
|
|
15
|
+
NUC_REPS,
|
|
16
|
+
TOLERANCE,
|
|
17
|
+
UNMODIFIED_BASES,
|
|
18
|
+
build_breakage_dict,
|
|
19
|
+
)
|
|
20
|
+
from spectrseqtools.prediction import Predictor
|
|
21
|
+
from spectrseqtools.preprocessing import preprocess
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class Settings(Tap):
|
|
25
|
+
fragments: Path # Path to TSV table or RAW data of observed fragments to use for prediction
|
|
26
|
+
meta: Path # Path to YAML with meta information to use for prediction
|
|
27
|
+
fragment_predictions: (
|
|
28
|
+
Path # Path to TSV table that shall contain the per fragment predictions
|
|
29
|
+
)
|
|
30
|
+
sequence_prediction: (
|
|
31
|
+
Path # Path to FASTA file that shall contain the predicted sequence
|
|
32
|
+
)
|
|
33
|
+
output_dir: Path = None # Output directory (default: input directory)
|
|
34
|
+
sequence_name: str
|
|
35
|
+
modification_rate: float = 0.5 # Maximum percentage of modification in sequence
|
|
36
|
+
solver: Literal["gurobi", "cbc"] = (
|
|
37
|
+
"gurobi" # Solver to use for the optimization problem
|
|
38
|
+
)
|
|
39
|
+
lp_timeout_short: int = 5 # Time-out for shorter solving of LP instances
|
|
40
|
+
lp_timeout_long: int = 60 # Time-out for longer solving of LP instances
|
|
41
|
+
cutoff_percentile: int = 75 # Intensity percentile used as cutoff
|
|
42
|
+
threads: int = 1 # Number of threads to use for the optimization problem
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def main():
|
|
46
|
+
settings = Settings(underscores_to_dashes=True).parse_args()
|
|
47
|
+
|
|
48
|
+
# Set parameters for LP solver
|
|
49
|
+
solver_params = {
|
|
50
|
+
"fixed": {
|
|
51
|
+
"solver": select_solver(settings.solver),
|
|
52
|
+
"threads": settings.threads,
|
|
53
|
+
"msg": False,
|
|
54
|
+
},
|
|
55
|
+
"timeLimit(short)": settings.lp_timeout_short,
|
|
56
|
+
"timeLimit(long)": settings.lp_timeout_long,
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
settings.fragments = settings.fragments.resolve()
|
|
60
|
+
fragment_dir = (
|
|
61
|
+
settings.fragments.parent
|
|
62
|
+
if settings.output_dir is None
|
|
63
|
+
else settings.output_dir
|
|
64
|
+
)
|
|
65
|
+
file_prefix = settings.fragments.stem
|
|
66
|
+
with open(settings.meta, "r") as f:
|
|
67
|
+
meta = yaml.safe_load(f)
|
|
68
|
+
|
|
69
|
+
# Preprocess data if necessary
|
|
70
|
+
match settings.fragments.suffix:
|
|
71
|
+
case ".raw":
|
|
72
|
+
print("RAW file found. Preprocessing raw data...")
|
|
73
|
+
# Preprocess raw data
|
|
74
|
+
fragments, singletons, meta = preprocess(
|
|
75
|
+
file_path=settings.fragments,
|
|
76
|
+
deconvolution_params={},
|
|
77
|
+
meta_params=meta,
|
|
78
|
+
cutoff_percentile=settings.cutoff_percentile,
|
|
79
|
+
)
|
|
80
|
+
# Save preprocessed fragments
|
|
81
|
+
fragments.write_csv(fragment_dir / f"{file_prefix}.tsv", separator="\t")
|
|
82
|
+
|
|
83
|
+
# Save singletons detected from raw data
|
|
84
|
+
singletons.write_csv(
|
|
85
|
+
fragment_dir / f"{file_prefix}.singletons.tsv", separator="\t"
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
# Save updated meta data
|
|
89
|
+
with open(fragment_dir / f"{file_prefix}.preprocessed.meta.yaml", "w") as f:
|
|
90
|
+
yaml.dump(meta, f)
|
|
91
|
+
|
|
92
|
+
print("Preprocessing completed!\n")
|
|
93
|
+
case ".tsv":
|
|
94
|
+
print("TSV file found. Proceeding without preprocessing.")
|
|
95
|
+
# Read already preprocessed fragments
|
|
96
|
+
fragments = pl.read_csv(settings.fragments, separator="\t")
|
|
97
|
+
|
|
98
|
+
# Read singletons if given
|
|
99
|
+
singletons = None
|
|
100
|
+
if os.path.isfile(fragment_dir / f"{file_prefix}.singletons.tsv"):
|
|
101
|
+
singletons = pl.read_csv(
|
|
102
|
+
fragment_dir / f"{file_prefix}.singletons.tsv", separator="\t"
|
|
103
|
+
)
|
|
104
|
+
case _:
|
|
105
|
+
raise NotImplementedError(
|
|
106
|
+
"Support is currently only given for TSV or RAW files."
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
print("Singletons identified during preprocessing:", singletons)
|
|
110
|
+
print()
|
|
111
|
+
|
|
112
|
+
explanation_masses = EXPLANATION_MASSES
|
|
113
|
+
|
|
114
|
+
# Filter by singletons
|
|
115
|
+
if singletons is not None:
|
|
116
|
+
# Map singletons to their mass representative
|
|
117
|
+
singletons = singletons.with_columns(
|
|
118
|
+
pl.col("nucleoside").replace_strict(NUC_REPS).alias("nucleoside")
|
|
119
|
+
)
|
|
120
|
+
|
|
121
|
+
# Select only bases found in singletons
|
|
122
|
+
explanation_masses = explanation_masses.with_columns(
|
|
123
|
+
pl.when(
|
|
124
|
+
pl.col("nucleoside").is_in(
|
|
125
|
+
singletons.get_column("nucleoside").to_list()
|
|
126
|
+
)
|
|
127
|
+
)
|
|
128
|
+
.then(pl.col("modification_rate"))
|
|
129
|
+
.otherwise(pl.lit(0.0))
|
|
130
|
+
.alias("modification_rate")
|
|
131
|
+
)
|
|
132
|
+
|
|
133
|
+
# Ensure modification rates of unmodified bases are set to 1
|
|
134
|
+
explanation_masses = explanation_masses.with_columns(
|
|
135
|
+
pl.when(~pl.col("nucleoside").is_in(UNMODIFIED_BASES))
|
|
136
|
+
.then(pl.col("modification_rate"))
|
|
137
|
+
.otherwise(pl.lit(1.0))
|
|
138
|
+
.alias("modification_rate")
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
# Read additional parameter from meta file
|
|
142
|
+
intensity_cutoff = meta.setdefault("intensity_cutoff", DEFAULT_INTENSITY_CUTOFF)
|
|
143
|
+
start_tag = meta.setdefault("label_mass_5T", 555.1294)
|
|
144
|
+
end_tag = meta.setdefault("label_mass_3T", 455.1491)
|
|
145
|
+
|
|
146
|
+
# Build breakage dict
|
|
147
|
+
breakage_dict = build_breakage_dict(mass_5_prime=start_tag, mass_3_prime=end_tag)
|
|
148
|
+
|
|
149
|
+
# Standardize sequence mass (remove START_END breakage to gain SU mass)
|
|
150
|
+
seq_mass_obs = meta["sequence_mass"]
|
|
151
|
+
seq_mass_su = (
|
|
152
|
+
seq_mass_obs
|
|
153
|
+
- [
|
|
154
|
+
mass * TOLERANCE
|
|
155
|
+
for mass in breakage_dict
|
|
156
|
+
if "START_END" in breakage_dict[mass]
|
|
157
|
+
][0]
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
# Initialize SequenceInformation class
|
|
161
|
+
seq_info = SequenceInformation(
|
|
162
|
+
max_len=int(
|
|
163
|
+
seq_mass_su
|
|
164
|
+
/ TOLERANCE
|
|
165
|
+
/ min(
|
|
166
|
+
pl.Series(
|
|
167
|
+
explanation_masses.filter(pl.col("modification_rate") > 0.0).select(
|
|
168
|
+
"tolerated_integer_masses"
|
|
169
|
+
)
|
|
170
|
+
).to_list()
|
|
171
|
+
)
|
|
172
|
+
),
|
|
173
|
+
su_mass=seq_mass_su,
|
|
174
|
+
obs_mass=seq_mass_obs,
|
|
175
|
+
modification_rate=settings.modification_rate,
|
|
176
|
+
)
|
|
177
|
+
|
|
178
|
+
# Initialize DynamicProgrammingTable class
|
|
179
|
+
dp_table = DynamicProgrammingTable(
|
|
180
|
+
nucleotide_df=explanation_masses,
|
|
181
|
+
compression_rate=int(COMPRESSION_RATE),
|
|
182
|
+
tolerance=MATCHING_THRESHOLD,
|
|
183
|
+
precision=TOLERANCE,
|
|
184
|
+
seq=seq_info,
|
|
185
|
+
)
|
|
186
|
+
|
|
187
|
+
print("Alphabet after singleton reduction:")
|
|
188
|
+
dp_table.print_masses()
|
|
189
|
+
print()
|
|
190
|
+
|
|
191
|
+
# Classify preprocessed fragments
|
|
192
|
+
fragments = classify_fragments(
|
|
193
|
+
fragment_masses=fragments,
|
|
194
|
+
dp_table=dp_table,
|
|
195
|
+
breakage_dict=breakage_dict,
|
|
196
|
+
output_file_path=fragment_dir / f"{file_prefix}.standard_unit_fragments.tsv",
|
|
197
|
+
intensity_cutoff=intensity_cutoff,
|
|
198
|
+
)
|
|
199
|
+
|
|
200
|
+
# Predict sequence
|
|
201
|
+
prediction = Predictor(
|
|
202
|
+
dp_table=dp_table,
|
|
203
|
+
explanation_masses=explanation_masses,
|
|
204
|
+
).predict(
|
|
205
|
+
fragments=fragments,
|
|
206
|
+
solver_params=solver_params,
|
|
207
|
+
)
|
|
208
|
+
|
|
209
|
+
print("Predicted sequence =\t", prediction.sequence)
|
|
210
|
+
|
|
211
|
+
# Save fragment predictions
|
|
212
|
+
prediction.fragments.write_csv(settings.fragment_predictions, separator="\t")
|
|
213
|
+
|
|
214
|
+
# Save predicted sequence
|
|
215
|
+
with open(settings.sequence_prediction, "w") as f:
|
|
216
|
+
print(f">{settings.sequence_name}", file=f)
|
|
217
|
+
print("".join(prediction.sequence), file=f)
|
|
218
|
+
print(f">{settings.sequence_name}_full", file=f)
|
|
219
|
+
print(format_sequence_to_full_version(seq=prediction.sequence), file=f)
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def format_sequence_to_full_version(seq: List[str]) -> str:
|
|
223
|
+
"""
|
|
224
|
+
Format a sequence to its full version (i.e. include alternate nucleotides).
|
|
225
|
+
|
|
226
|
+
Parameters
|
|
227
|
+
----------
|
|
228
|
+
seq: List[str]
|
|
229
|
+
Given predicted sequence.
|
|
230
|
+
|
|
231
|
+
Returns
|
|
232
|
+
-------
|
|
233
|
+
str
|
|
234
|
+
Sequence with all alternate nucleotides.
|
|
235
|
+
|
|
236
|
+
"""
|
|
237
|
+
output = ""
|
|
238
|
+
for nuc in seq:
|
|
239
|
+
alt_nucs = (
|
|
240
|
+
EXPLANATION_MASSES.filter(pl.col("nucleoside") == nuc)
|
|
241
|
+
.select("nucleoside_list")
|
|
242
|
+
.item()
|
|
243
|
+
.to_list()
|
|
244
|
+
)
|
|
245
|
+
if len(alt_nucs) == 1:
|
|
246
|
+
output += nuc
|
|
247
|
+
else:
|
|
248
|
+
output += "[" + "|".join(alt_nucs) + "]"
|
|
249
|
+
return output
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def select_solver(solver: str):
|
|
253
|
+
match solver:
|
|
254
|
+
case "gurobi":
|
|
255
|
+
return "GUROBI_CMD"
|
|
256
|
+
case "cbc":
|
|
257
|
+
return "PULP_CBC_CMD"
|
|
258
|
+
case _:
|
|
259
|
+
raise NotImplementedError(f"Support for '{solver}' is currently not given.")
|
spectrseqtools/common.py
ADDED
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
import ms_deisotope as ms_ditp
|
|
2
|
+
import re
|
|
3
|
+
|
|
4
|
+
from clr_loader import get_mono
|
|
5
|
+
from typing import List
|
|
6
|
+
|
|
7
|
+
from spectrseqtools.mass_explanation import explain_mass_with_table
|
|
8
|
+
from spectrseqtools.mass_table import DynamicProgrammingTable
|
|
9
|
+
|
|
10
|
+
rt = get_mono()
|
|
11
|
+
|
|
12
|
+
ERROR_METHOD = "l1_norm"
|
|
13
|
+
_NUCLEOSIDE_RE = re.compile(r"\d*[ACGU]")
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def parse_nucleosides(sequence: str):
|
|
17
|
+
return _NUCLEOSIDE_RE.findall(sequence)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class Explanation:
|
|
21
|
+
def __init__(self, *nucleosides):
|
|
22
|
+
self.nucleosides = tuple(sorted(nucleosides))
|
|
23
|
+
|
|
24
|
+
def __iter__(self):
|
|
25
|
+
yield from self.nucleosides
|
|
26
|
+
|
|
27
|
+
def __len__(self):
|
|
28
|
+
return len(self.nucleosides)
|
|
29
|
+
|
|
30
|
+
def __repr__(self):
|
|
31
|
+
return f"{{{','.join(self.nucleosides)}}}"
|
|
32
|
+
|
|
33
|
+
def __eq__(self, other):
|
|
34
|
+
return self.nucleosides == other
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def calculate_error_threshold(mass1: float, mass2: float, threshold: float) -> float:
|
|
38
|
+
match ERROR_METHOD:
|
|
39
|
+
case "l1_norm":
|
|
40
|
+
return threshold * (mass1 + mass2)
|
|
41
|
+
case "l2_norm":
|
|
42
|
+
return threshold * ((mass1**2 + mass2**2) ** 0.5)
|
|
43
|
+
case _:
|
|
44
|
+
raise NotImplementedError("This error method is not implemented.")
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def calculate_explanations(
|
|
48
|
+
diff: float,
|
|
49
|
+
threshold: float,
|
|
50
|
+
dp_table: DynamicProgrammingTable,
|
|
51
|
+
) -> List[Explanation]:
|
|
52
|
+
explanation_list = explain_mass_with_table(
|
|
53
|
+
diff,
|
|
54
|
+
dp_table=dp_table,
|
|
55
|
+
max_modifications=round(dp_table.seq.modification_rate * dp_table.seq.max_len),
|
|
56
|
+
threshold=threshold,
|
|
57
|
+
).explanations
|
|
58
|
+
|
|
59
|
+
# Return None if no explanation was found
|
|
60
|
+
if explanation_list is None:
|
|
61
|
+
return None
|
|
62
|
+
|
|
63
|
+
# Return all found explanations
|
|
64
|
+
explanation_list = list(explanation_list)
|
|
65
|
+
return [Explanation(*explanation_list[i]) for i in range(len(explanation_list))]
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def initialize_raw_file_iterator(
|
|
69
|
+
file_path: str,
|
|
70
|
+
) -> ms_ditp.data_source.thermo_raw_net.ThermoRawLoader:
|
|
71
|
+
"""
|
|
72
|
+
Initialize iterator over scans in ThermoFisher RAW file format.
|
|
73
|
+
|
|
74
|
+
Parameters
|
|
75
|
+
----------
|
|
76
|
+
file_path : str
|
|
77
|
+
Path of RAW file from ThermoFisher.
|
|
78
|
+
|
|
79
|
+
Returns
|
|
80
|
+
-------
|
|
81
|
+
raw_file : ms_deisotope.data_source.thermo_raw_net.ThermoRawLoader
|
|
82
|
+
Iterator over scans from RAW file.
|
|
83
|
+
|
|
84
|
+
"""
|
|
85
|
+
# Read data from file
|
|
86
|
+
raw_file = ms_ditp.data_source.thermo_raw_net.ThermoRawLoader(
|
|
87
|
+
file_path, _load_metadata=True
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
# Initialize an iterator while ungrouping MS1 from MS2 scans
|
|
91
|
+
raw_file.make_iterator(grouped=False)
|
|
92
|
+
|
|
93
|
+
return raw_file
|