spectrseqtools 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
File without changes
@@ -0,0 +1,6 @@
1
+ symbol mass
2
+ H+ 1.007276
3
+ H 1.00783
4
+ C 12.00000
5
+ O 15.99493
6
+ P 30.97376
@@ -0,0 +1,5 @@
1
+ base C H N O P S
2
+ adenosine 10 13 5 4 0 0
3
+ cytidine 9 13 3 5 0 0
4
+ guanosine 10 13 5 5 0 0
5
+ uridine 9 12 2 6 0 0
@@ -0,0 +1,145 @@
1
+ nucleoside canonical_name monoisotopic_mass modification_rate
2
+ A A 267.09675 1.0
3
+ C C 243.08552 1.0
4
+ G G 283.09167 1.0
5
+ U U 244.06954 1.0
6
+ 0A Am 281.1124 1.0
7
+ 00A Ar(p) 479.1053 1.0
8
+ 01A m1Am 295.1281 1.0
9
+ 019A m1Im 296.1121 1.0
10
+ 06A m6Am 295.1281 1.0
11
+ 066A m6,6Am 309.1437 1.0
12
+ 09A Im 282.0964 1.0
13
+ 1A m1A 281.1124 1.0
14
+ 19A m1I 282.0964 1.0
15
+ 2A m2A 281.1124 1.0
16
+ 21161A msms2i6A 427.1347 1.0
17
+ 2160A ms2io6A 397.142 1.0
18
+ 2161A ms2i6A 381.1471 1.0
19
+ 2162A ms2t6A 458.122 1.0
20
+ 2163A ms2hn6A 472.1376 1.0
21
+ 2164A ms2ct6A 440.1114 1.0
22
+ 2165A ht6A 428.1292 1.0
23
+ 28A m2,8A 295.1281 1.0
24
+ 6A m6A 281.1124 1.0
25
+ 60A io6A 351.1543 1.0
26
+ 61A i6A 335.1594 1.0
27
+ 62A t6A 412.1343 1.0
28
+ 621A ms2m6A 327.1001 1.0
29
+ 63A hn6A 426.1499 1.0
30
+ 64A ac6A 309.1073 1.0
31
+ 65A g6A 368.108 1.0
32
+ 66A m6,6A 295.1281 1.0
33
+ 662A m6t6A 426.1499 1.0
34
+ 67A f6A 295.0917 1.0
35
+ 68A hm6A 297.1073 1.0
36
+ 69A ct6A 394.1237 1.0
37
+ 8A m8A 281.1124 1.0
38
+ 9A I 268.0808 1.0
39
+ 0C Cm 257.10117055 1.0
40
+ 04C m4Cm 271.11682061 1.0
41
+ 042C ac4Cm 299.11173523 1.0
42
+ 044C m4,4Cm 285.13247067 1.0
43
+ 05C m5Cm 271.11682061 1.0
44
+ 051C hm5Cm 287.1117 1.0
45
+ 071C f5Cm 285.09608517 1.0
46
+ 2C s2C 259.06267687 1.0
47
+ 20C C+ 355.1965 1.0
48
+ 21C k2C 371.18048347 1.0
49
+ 3C m3C 257.10117055 1.0
50
+ 4C m4C 257.10117055 1.0
51
+ 42C ac4C 285.09608517 1.0
52
+ 44C m4,4C 271.11682061 1.0
53
+ 5C m5C 257.10117055 1.0
54
+ 50C ho5C 259.08043511 1.0
55
+ 51C hm5C 273.09608517 1.0
56
+ 71C f5C 271.08043511 1.0
57
+ 0G Gm 297.1073 1.0
58
+ 00G Gr(p) 495.1003 1.0
59
+ 01G m1Gm 311.123 1.0
60
+ 02G m2Gm 311.123 1.0
61
+ 022G m2,2Gm 325.1386 1.0
62
+ 027G m2,7Gm 327.1543 1.0
63
+ 1G m1G 297.1073 1.0
64
+ 10G Q 409.1597 1.0
65
+ 100G preQ0 307.0917 1.0
66
+ 101G preQ1 311.123 1.0
67
+ 102G oQ 425.1547 1.0
68
+ 103G G+ 324.1182 1.0
69
+ 104G galQ 571.2126 1.0
70
+ 105G gluQ 538.2023 1.0
71
+ 106G manQ 571.2126 1.0
72
+ 2G m2G 297.1073 1.0
73
+ 22G m2,2G 311.123 1.0
74
+ 227G m2,2,7G 327.1543 1.0
75
+ 27G m2,7G 313.1386 1.0
76
+ 34G imG 335.123 1.0
77
+ 342G mimG 349.1386 1.0
78
+ 347G yW-72 436.1706 1.0
79
+ 3470G OHyWx 452.1655 1.0
80
+ 348G yW-58 450.1863 1.0
81
+ 3480G OHyWy 466.1812 1.0
82
+ 3483G yW 508.1918 1.0
83
+ 34830G OHyW 524.1867 1.0
84
+ 34832G o2yW 540.1816 1.0
85
+ 4G imG-14 321.1073 1.0
86
+ 42G imG2 335.123 1.0
87
+ 47G yW-86 422.155 1.0
88
+ 7G m7G 299.123 1.0
89
+ 0U Um 258.08518614 1.0
90
+ 02U s2Um 274.06234252 1.0
91
+ 03U m3Um 272.1008362 1.0
92
+ 05U m5Um 272.1008362 1.0
93
+ 0503U mcmo5Um 346.10123012 1.0
94
+ 051U cmnm5Um 345.11721453 1.0
95
+ 0521U mcm5Um 330.1063155 1.0
96
+ 0522U mchm5Um 346.10123012 1.0
97
+ 053U ncm5Um 315.10664985 1.0
98
+ 0583U inm5Um 355.17433547 1.0
99
+ 09U Ym 258.08518614 1.0
100
+ 1309U m1acp3Y 359.13286459 1.0
101
+ 19U m1Y 258.08518614 1.0
102
+ 2U s2U 260.04669246 1.0
103
+ 20U se2U 307.9911 1.0
104
+ 2051U cmnm5se2U 395.0232 1.0
105
+ 20510U nm5se2U 337.0177 1.0
106
+ 20511U mnm5se2U 365.049 1.0
107
+ 21U ges2U 396.17189294 1.0
108
+ 2151U cmnm5ges2U 483.20392133 1.0
109
+ 21510U nm5ges2U 425.19844 1.0
110
+ 21511U mnm5ges2U 439.21409209 1.0
111
+ 25U m5s2U 274.0623 1.0
112
+ 251U cmnm5s2U 347.07872085 1.0
113
+ 2510U nm5s2U 289.07324155 1.0
114
+ 2511U mnm5s2U 303.08889161 1.0
115
+ 2521U mcm5s2U 332.06782182 1.0
116
+ 253U ncm5s2U 317.06815617 1.0
117
+ 254U tm5s2U 397.06135653 1.0
118
+ 2540U cm5s2U 318.05217176 1.0
119
+ 2583U inm5s2U 357.13584179 1.0
120
+ 3U m3U 258.08518614 1.0
121
+ 30U acp3U 345.11721453 1.0
122
+ 308U acp3D 347.13286459 1.0
123
+ 309U acp3Y 345.11721453 1.0
124
+ 39U m3Y 258.08518614 1.0
125
+ 5U m5U 258.08518614 1.0
126
+ 50U ho5U 260.0644507 1.0
127
+ 501U mo5U 274.08010076 1.0
128
+ 502U cmo5U 318.06993 1.0
129
+ 503U mcmo5U 332.08558006 1.0
130
+ 51U cmnm5U 331.10156447 1.0
131
+ 510U nm5U 273.09608517 1.0
132
+ 511U mnm5U 287.11173523 1.0
133
+ 52U cm5U 302.07501538 1.0
134
+ 520U chm5U 318.06993 1.0
135
+ 521U mcm5U 316.09066544 1.0
136
+ 522U mchm5U 332.08558006 1.0
137
+ 53U ncm5U 301.09099979 1.0
138
+ 531U nchm5U 317.08591441 1.0
139
+ 54U tm5U 381.08420015 1.0
140
+ 55U cnm5U 283.08043511 1.0
141
+ 58U m5D 260.1008362 1.0
142
+ 583U inm5U 341.15868541 1.0
143
+ 74U s4U 260.04669246 1.0
144
+ 8U D 246.08518614 1.0
145
+ 9U Y 244.06953608 1.0
spectrseqtools/cli.py ADDED
@@ -0,0 +1,259 @@
1
+ import os
2
+ import polars as pl
3
+ import yaml
4
+ from pathlib import Path
5
+ from tap import Tap
6
+ from typing import List, Literal
7
+
8
+ from spectrseqtools.fragment_classification import classify_fragments
9
+ from spectrseqtools.mass_table import DynamicProgrammingTable, SequenceInformation
10
+ from spectrseqtools.masses import (
11
+ COMPRESSION_RATE,
12
+ DEFAULT_INTENSITY_CUTOFF,
13
+ EXPLANATION_MASSES,
14
+ MATCHING_THRESHOLD,
15
+ NUC_REPS,
16
+ TOLERANCE,
17
+ UNMODIFIED_BASES,
18
+ build_breakage_dict,
19
+ )
20
+ from spectrseqtools.prediction import Predictor
21
+ from spectrseqtools.preprocessing import preprocess
22
+
23
+
24
+ class Settings(Tap):
25
+ fragments: Path # Path to TSV table or RAW data of observed fragments to use for prediction
26
+ meta: Path # Path to YAML with meta information to use for prediction
27
+ fragment_predictions: (
28
+ Path # Path to TSV table that shall contain the per fragment predictions
29
+ )
30
+ sequence_prediction: (
31
+ Path # Path to FASTA file that shall contain the predicted sequence
32
+ )
33
+ output_dir: Path = None # Output directory (default: input directory)
34
+ sequence_name: str
35
+ modification_rate: float = 0.5 # Maximum percentage of modification in sequence
36
+ solver: Literal["gurobi", "cbc"] = (
37
+ "gurobi" # Solver to use for the optimization problem
38
+ )
39
+ lp_timeout_short: int = 5 # Time-out for shorter solving of LP instances
40
+ lp_timeout_long: int = 60 # Time-out for longer solving of LP instances
41
+ cutoff_percentile: int = 75 # Intensity percentile used as cutoff
42
+ threads: int = 1 # Number of threads to use for the optimization problem
43
+
44
+
45
+ def main():
46
+ settings = Settings(underscores_to_dashes=True).parse_args()
47
+
48
+ # Set parameters for LP solver
49
+ solver_params = {
50
+ "fixed": {
51
+ "solver": select_solver(settings.solver),
52
+ "threads": settings.threads,
53
+ "msg": False,
54
+ },
55
+ "timeLimit(short)": settings.lp_timeout_short,
56
+ "timeLimit(long)": settings.lp_timeout_long,
57
+ }
58
+
59
+ settings.fragments = settings.fragments.resolve()
60
+ fragment_dir = (
61
+ settings.fragments.parent
62
+ if settings.output_dir is None
63
+ else settings.output_dir
64
+ )
65
+ file_prefix = settings.fragments.stem
66
+ with open(settings.meta, "r") as f:
67
+ meta = yaml.safe_load(f)
68
+
69
+ # Preprocess data if necessary
70
+ match settings.fragments.suffix:
71
+ case ".raw":
72
+ print("RAW file found. Preprocessing raw data...")
73
+ # Preprocess raw data
74
+ fragments, singletons, meta = preprocess(
75
+ file_path=settings.fragments,
76
+ deconvolution_params={},
77
+ meta_params=meta,
78
+ cutoff_percentile=settings.cutoff_percentile,
79
+ )
80
+ # Save preprocessed fragments
81
+ fragments.write_csv(fragment_dir / f"{file_prefix}.tsv", separator="\t")
82
+
83
+ # Save singletons detected from raw data
84
+ singletons.write_csv(
85
+ fragment_dir / f"{file_prefix}.singletons.tsv", separator="\t"
86
+ )
87
+
88
+ # Save updated meta data
89
+ with open(fragment_dir / f"{file_prefix}.preprocessed.meta.yaml", "w") as f:
90
+ yaml.dump(meta, f)
91
+
92
+ print("Preprocessing completed!\n")
93
+ case ".tsv":
94
+ print("TSV file found. Proceeding without preprocessing.")
95
+ # Read already preprocessed fragments
96
+ fragments = pl.read_csv(settings.fragments, separator="\t")
97
+
98
+ # Read singletons if given
99
+ singletons = None
100
+ if os.path.isfile(fragment_dir / f"{file_prefix}.singletons.tsv"):
101
+ singletons = pl.read_csv(
102
+ fragment_dir / f"{file_prefix}.singletons.tsv", separator="\t"
103
+ )
104
+ case _:
105
+ raise NotImplementedError(
106
+ "Support is currently only given for TSV or RAW files."
107
+ )
108
+
109
+ print("Singletons identified during preprocessing:", singletons)
110
+ print()
111
+
112
+ explanation_masses = EXPLANATION_MASSES
113
+
114
+ # Filter by singletons
115
+ if singletons is not None:
116
+ # Map singletons to their mass representative
117
+ singletons = singletons.with_columns(
118
+ pl.col("nucleoside").replace_strict(NUC_REPS).alias("nucleoside")
119
+ )
120
+
121
+ # Select only bases found in singletons
122
+ explanation_masses = explanation_masses.with_columns(
123
+ pl.when(
124
+ pl.col("nucleoside").is_in(
125
+ singletons.get_column("nucleoside").to_list()
126
+ )
127
+ )
128
+ .then(pl.col("modification_rate"))
129
+ .otherwise(pl.lit(0.0))
130
+ .alias("modification_rate")
131
+ )
132
+
133
+ # Ensure modification rates of unmodified bases are set to 1
134
+ explanation_masses = explanation_masses.with_columns(
135
+ pl.when(~pl.col("nucleoside").is_in(UNMODIFIED_BASES))
136
+ .then(pl.col("modification_rate"))
137
+ .otherwise(pl.lit(1.0))
138
+ .alias("modification_rate")
139
+ )
140
+
141
+ # Read additional parameter from meta file
142
+ intensity_cutoff = meta.setdefault("intensity_cutoff", DEFAULT_INTENSITY_CUTOFF)
143
+ start_tag = meta.setdefault("label_mass_5T", 555.1294)
144
+ end_tag = meta.setdefault("label_mass_3T", 455.1491)
145
+
146
+ # Build breakage dict
147
+ breakage_dict = build_breakage_dict(mass_5_prime=start_tag, mass_3_prime=end_tag)
148
+
149
+ # Standardize sequence mass (remove START_END breakage to gain SU mass)
150
+ seq_mass_obs = meta["sequence_mass"]
151
+ seq_mass_su = (
152
+ seq_mass_obs
153
+ - [
154
+ mass * TOLERANCE
155
+ for mass in breakage_dict
156
+ if "START_END" in breakage_dict[mass]
157
+ ][0]
158
+ )
159
+
160
+ # Initialize SequenceInformation class
161
+ seq_info = SequenceInformation(
162
+ max_len=int(
163
+ seq_mass_su
164
+ / TOLERANCE
165
+ / min(
166
+ pl.Series(
167
+ explanation_masses.filter(pl.col("modification_rate") > 0.0).select(
168
+ "tolerated_integer_masses"
169
+ )
170
+ ).to_list()
171
+ )
172
+ ),
173
+ su_mass=seq_mass_su,
174
+ obs_mass=seq_mass_obs,
175
+ modification_rate=settings.modification_rate,
176
+ )
177
+
178
+ # Initialize DynamicProgrammingTable class
179
+ dp_table = DynamicProgrammingTable(
180
+ nucleotide_df=explanation_masses,
181
+ compression_rate=int(COMPRESSION_RATE),
182
+ tolerance=MATCHING_THRESHOLD,
183
+ precision=TOLERANCE,
184
+ seq=seq_info,
185
+ )
186
+
187
+ print("Alphabet after singleton reduction:")
188
+ dp_table.print_masses()
189
+ print()
190
+
191
+ # Classify preprocessed fragments
192
+ fragments = classify_fragments(
193
+ fragment_masses=fragments,
194
+ dp_table=dp_table,
195
+ breakage_dict=breakage_dict,
196
+ output_file_path=fragment_dir / f"{file_prefix}.standard_unit_fragments.tsv",
197
+ intensity_cutoff=intensity_cutoff,
198
+ )
199
+
200
+ # Predict sequence
201
+ prediction = Predictor(
202
+ dp_table=dp_table,
203
+ explanation_masses=explanation_masses,
204
+ ).predict(
205
+ fragments=fragments,
206
+ solver_params=solver_params,
207
+ )
208
+
209
+ print("Predicted sequence =\t", prediction.sequence)
210
+
211
+ # Save fragment predictions
212
+ prediction.fragments.write_csv(settings.fragment_predictions, separator="\t")
213
+
214
+ # Save predicted sequence
215
+ with open(settings.sequence_prediction, "w") as f:
216
+ print(f">{settings.sequence_name}", file=f)
217
+ print("".join(prediction.sequence), file=f)
218
+ print(f">{settings.sequence_name}_full", file=f)
219
+ print(format_sequence_to_full_version(seq=prediction.sequence), file=f)
220
+
221
+
222
+ def format_sequence_to_full_version(seq: List[str]) -> str:
223
+ """
224
+ Format a sequence to its full version (i.e. include alternate nucleotides).
225
+
226
+ Parameters
227
+ ----------
228
+ seq: List[str]
229
+ Given predicted sequence.
230
+
231
+ Returns
232
+ -------
233
+ str
234
+ Sequence with all alternate nucleotides.
235
+
236
+ """
237
+ output = ""
238
+ for nuc in seq:
239
+ alt_nucs = (
240
+ EXPLANATION_MASSES.filter(pl.col("nucleoside") == nuc)
241
+ .select("nucleoside_list")
242
+ .item()
243
+ .to_list()
244
+ )
245
+ if len(alt_nucs) == 1:
246
+ output += nuc
247
+ else:
248
+ output += "[" + "|".join(alt_nucs) + "]"
249
+ return output
250
+
251
+
252
+ def select_solver(solver: str):
253
+ match solver:
254
+ case "gurobi":
255
+ return "GUROBI_CMD"
256
+ case "cbc":
257
+ return "PULP_CBC_CMD"
258
+ case _:
259
+ raise NotImplementedError(f"Support for '{solver}' is currently not given.")
@@ -0,0 +1,93 @@
1
+ import ms_deisotope as ms_ditp
2
+ import re
3
+
4
+ from clr_loader import get_mono
5
+ from typing import List
6
+
7
+ from spectrseqtools.mass_explanation import explain_mass_with_table
8
+ from spectrseqtools.mass_table import DynamicProgrammingTable
9
+
10
+ rt = get_mono()
11
+
12
+ ERROR_METHOD = "l1_norm"
13
+ _NUCLEOSIDE_RE = re.compile(r"\d*[ACGU]")
14
+
15
+
16
+ def parse_nucleosides(sequence: str):
17
+ return _NUCLEOSIDE_RE.findall(sequence)
18
+
19
+
20
+ class Explanation:
21
+ def __init__(self, *nucleosides):
22
+ self.nucleosides = tuple(sorted(nucleosides))
23
+
24
+ def __iter__(self):
25
+ yield from self.nucleosides
26
+
27
+ def __len__(self):
28
+ return len(self.nucleosides)
29
+
30
+ def __repr__(self):
31
+ return f"{{{','.join(self.nucleosides)}}}"
32
+
33
+ def __eq__(self, other):
34
+ return self.nucleosides == other
35
+
36
+
37
+ def calculate_error_threshold(mass1: float, mass2: float, threshold: float) -> float:
38
+ match ERROR_METHOD:
39
+ case "l1_norm":
40
+ return threshold * (mass1 + mass2)
41
+ case "l2_norm":
42
+ return threshold * ((mass1**2 + mass2**2) ** 0.5)
43
+ case _:
44
+ raise NotImplementedError("This error method is not implemented.")
45
+
46
+
47
+ def calculate_explanations(
48
+ diff: float,
49
+ threshold: float,
50
+ dp_table: DynamicProgrammingTable,
51
+ ) -> List[Explanation]:
52
+ explanation_list = explain_mass_with_table(
53
+ diff,
54
+ dp_table=dp_table,
55
+ max_modifications=round(dp_table.seq.modification_rate * dp_table.seq.max_len),
56
+ threshold=threshold,
57
+ ).explanations
58
+
59
+ # Return None if no explanation was found
60
+ if explanation_list is None:
61
+ return None
62
+
63
+ # Return all found explanations
64
+ explanation_list = list(explanation_list)
65
+ return [Explanation(*explanation_list[i]) for i in range(len(explanation_list))]
66
+
67
+
68
+ def initialize_raw_file_iterator(
69
+ file_path: str,
70
+ ) -> ms_ditp.data_source.thermo_raw_net.ThermoRawLoader:
71
+ """
72
+ Initialize iterator over scans in ThermoFisher RAW file format.
73
+
74
+ Parameters
75
+ ----------
76
+ file_path : str
77
+ Path of RAW file from ThermoFisher.
78
+
79
+ Returns
80
+ -------
81
+ raw_file : ms_deisotope.data_source.thermo_raw_net.ThermoRawLoader
82
+ Iterator over scans from RAW file.
83
+
84
+ """
85
+ # Read data from file
86
+ raw_file = ms_ditp.data_source.thermo_raw_net.ThermoRawLoader(
87
+ file_path, _load_metadata=True
88
+ )
89
+
90
+ # Initialize an iterator while ungrouping MS1 from MS2 scans
91
+ raw_file.make_iterator(grouped=False)
92
+
93
+ return raw_file