cutseq 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
cutseq/run.py ADDED
@@ -0,0 +1,399 @@
1
+ #!/usr/bin/env python
2
+ # -*- coding: utf-8 -*-
3
+ #
4
+ # Copyright © 2024 Ye Chang yech1990@gmail.com
5
+ # Distributed under terms of the GNU license.
6
+ #
7
+ # Created: 2024-04-19 18:57
8
+
9
+ import argparse
10
+ import logging
11
+ import re
12
+ import subprocess
13
+ import sys
14
+
15
+ logging.basicConfig(
16
+ level=logging.INFO,
17
+ format="%(asctime)s - %(levelname)s - %(message)s",
18
+ )
19
+
20
+
21
+ def reverse_complement(b):
22
+ return "".join(
23
+ [dict(zip("ATGCNatgcn", "TACGNtacgn"))[x] for x in b[::-1] if x in "ATGCNatgcn"]
24
+ )
25
+
26
+
27
+ def remove_fq_suffix(f):
28
+ suffixes = [
29
+ "_R1_001.fastq",
30
+ "_R2_001.fastq",
31
+ "_R1.fastq",
32
+ "_R2.fastq",
33
+ ".fastq",
34
+ ".fq",
35
+ ]
36
+ suffixes = [s + ".gz" for s in suffixes] + suffixes
37
+ for suffix in suffixes:
38
+ if f.endswith(suffix):
39
+ return f.removesuffix(suffix)
40
+ return f
41
+
42
+
43
+ class BarcodeConfig:
44
+ def __init__(self, adapter=None):
45
+ self.strand = None
46
+ self.p5_fw = ""
47
+ self.p5_rc = ""
48
+ self.p7_fw = ""
49
+ self.p7_rc = ""
50
+ self.inline5_fw = ""
51
+ self.inline5_rc = ""
52
+ self.inline5 = 0
53
+ self.inline3_fw = ""
54
+ self.inline3_rc = ""
55
+ self.inline3 = 0
56
+ self.umi5 = 0
57
+ self.umi3 = 0
58
+ self.mask5 = 0
59
+ self.mask3 = 0
60
+ if adapter is not None:
61
+ self._parse_barcode(adapter)
62
+
63
+ def _parse_barcode(self, b):
64
+ m = re.match(
65
+ r"(?P<p5>[ATGCatgc]+)(\((?P<inline5>[ATGCatgc]+)\))?(?P<umi5>N*)(?P<mask5>X*)(?P<strand>-|>|<)(?P<mask3>X*)(?P<umi3>N*)(\((?P<inline3>[ATGCatgc]+)\))?(?P<p7>[ATGCatgc]+)",
66
+ b,
67
+ )
68
+ if m is None:
69
+ logging.error(f"barcode {b} is not valid")
70
+ sys.exit(1)
71
+ d = m.groupdict()
72
+ if d["inline5"] is None:
73
+ d["inline5"] = ""
74
+ if d["inline3"] is None:
75
+ d["inline3"] = ""
76
+ self.strand = "+" if d["strand"] == ">" else "-" if d["strand"] == "<" else None
77
+ self.p5_fw = d["p5"]
78
+ self.p5_rc = reverse_complement(d["p5"])
79
+ self.p7_fw = d["p7"]
80
+ self.p7_rc = reverse_complement(d["p7"])
81
+ self.inline5_fw = d["inline5"] if d["inline5"] else ""
82
+ self.inline5_rc = reverse_complement(d["inline5"]) if d["inline5"] else ""
83
+ self.inline5 = len(d["inline5"])
84
+ self.inline3_fw = d["inline3"] if d["inline3"] else ""
85
+ self.inline3_rc = reverse_complement(d["inline3"]) if d["inline3"] else ""
86
+ self.inline3 = len(d["inline3"])
87
+ self.umi5 = len(d["umi5"])
88
+ self.umi3 = len(d["umi3"])
89
+ self.mask5 = len(d["mask5"])
90
+ self.mask3 = len(d["mask3"])
91
+
92
+
93
+ class CutadaptConfig:
94
+ def __init__(self):
95
+ self.rname_suffix = False
96
+ self.discarded_untrimmed = False
97
+ self.trim_polyA = False
98
+ self.min_length = 20
99
+ self.min_quality = 20
100
+ self.dry_run = False
101
+ self.threads = 1
102
+
103
+
104
+ def run_cutadapt_PE(
105
+ input1, input2, output1, output2, discard1, discard2, barcode, settings
106
+ ):
107
+ cutadapt = f"cutadapt -j {settings.threads}"
108
+ steps = []
109
+ # step 1: remove adapter on the 5' end, artifact of template switching
110
+ if settings.rname_suffix:
111
+ config_rname = " --strip-suffix '/1' --strip-suffix '/2' --strip-suffix '.1' --strip-suffix '.2'"
112
+ else:
113
+ config_rname = ""
114
+ steps.append(
115
+ f"{cutadapt}{config_rname} -e 0.25 -n 2 -O 10 -g '{barcode.p5_fw};rightmost' -G '{barcode.p7_rc};rightmost' --interleaved {input1} {input2}"
116
+ )
117
+ # step 2: remove adapter on the 3' end, read though in the sequencing
118
+ steps.append(
119
+ f"{cutadapt} -e 0.2 -n 2 -O 3 -a '{barcode.p7_fw}' -A '{barcode.p5_rc}' --interleaved -"
120
+ )
121
+ # step 3: trim inline barcode
122
+ config_inline_args = []
123
+ if barcode.inline5 > 0:
124
+ config_inline_args.append(f"-g ^{barcode.inline5_fw} -U -{barcode.inline5}")
125
+ if barcode.inline3 > 0:
126
+ config_inline_args.append(f"-G ^{barcode.inline3_rc} -u -{barcode.inline3}")
127
+ if barcode.inline5 + barcode.inline3 > 0:
128
+ if settings.discarded_untrimmed:
129
+ config_inline_args.append(
130
+ f"--untrimmed-output={discard1} --untrimmed-paired-output={discard2}"
131
+ )
132
+ config_inline = " ".join(config_inline_args)
133
+ steps.append(f"{cutadapt} {config_inline} --interleaved -")
134
+ # step 4: extract UMI
135
+ if barcode.umi5 + barcode.umi3 > 0:
136
+ steps.append(
137
+ f"{cutadapt} -u {barcode.umi5} -u -{barcode.umi3} -U {barcode.umi3} -U -{barcode.umi5} --rename='{{id}}_{{r1.cut_prefix}}{{r2.cut_prefix}}' --interleaved -"
138
+ )
139
+ else:
140
+ steps.append(f"{cutadapt} --rename='{{id}}' --interleaved -")
141
+ # step 5: mask tail in the RNA, which might be artifact of RT
142
+ if barcode.mask5 + barcode.mask3 > 0:
143
+ steps.append(
144
+ f"{cutadapt} -u {barcode.mask5} -u -{barcode.mask3} -U {barcode.mask3} -U -{barcode.mask5} --interleaved -"
145
+ )
146
+ # step 6: trim polyA
147
+ if settings.trim_polyA:
148
+ if barcode.strand == "+":
149
+ steps.append(
150
+ f"{cutadapt} -O 6 -e 0.15 -a 'A{{100}}' -G 'T{{100}}' --interleaved -"
151
+ )
152
+ elif barcode.strand == "-":
153
+ steps.append(
154
+ f"{cutadapt} -O 6 -e 0.15 -g 'T{{100}}' -A 'A{{100}}' --interleaved -"
155
+ )
156
+ else:
157
+ logging.info("No strand information provided, skip polyA trimming.")
158
+ # step 7: quality control, remove short reads
159
+ steps.append(
160
+ f"{cutadapt} -q {settings.min_quality} --max-n=0 -m {settings.min_length} --too-short-output={discard1} --too-short-paired-output={discard2} -o {output1} -p {output2} --interleaved -"
161
+ )
162
+
163
+ if settings.dry_run:
164
+ print(
165
+ " |\\\n".join([(" " + s if i > 0 else s) for i, s in enumerate(steps)])
166
+ )
167
+ process = subprocess.run("true", shell=True, capture_output=True)
168
+ else:
169
+ cmd = " | ".join(steps)
170
+ process = subprocess.run(cmd, shell=True, capture_output=True)
171
+ return process.stdout.decode(), process.stderr.decode()
172
+
173
+
174
+ def run_cutadapt_SE(input1, output1, discard1, barcode, settings):
175
+ cutadapt = f"cutadapt -j {settings.threads}"
176
+ steps = []
177
+ # step 1: remove adapter on the 5' end, artifact of template switching
178
+ if settings.rname_suffix:
179
+ config_rname = " --strip-suffix '/1' --strip-suffix '.1'"
180
+ else:
181
+ config_rname = ""
182
+ steps.append(
183
+ f"{cutadapt}{config_rname} -e 0.25 -n 2 -O 10 -g '{barcode.p5_fw};rightmost' {input1}"
184
+ )
185
+ # step 2: remove adapter on the 3' end, read though in the sequencing
186
+ steps.append(f"{cutadapt} -e 0.2 -n 2 -O 3 -a '{barcode.p7_fw}' -")
187
+ # step 3: trim inline barcode
188
+ config_inline_args = []
189
+ if barcode.inline3 > 0:
190
+ config_inline_args.append(f"-a {barcode.inline3_fw}$")
191
+ if barcode.inline5 > 0:
192
+ config_inline_args.append(f"-g ^{barcode.inline5_fw}")
193
+ if barcode.inline5 + barcode.inline3 > 0:
194
+ if settings.discarded_untrimmed:
195
+ config_inline_args.append(f"--untrimmed-output={discard1}")
196
+
197
+ config_inline = " ".join(config_inline_args)
198
+ steps.append(f"{cutadapt} {config_inline} -")
199
+ # step 4: extract UMI
200
+ if barcode.umi5 + barcode.umi3 > 0:
201
+ steps.append(
202
+ f"{cutadapt} -u {barcode.umi5} -u -{barcode.umi3} --rename='{{id}}_{{cut_prefix}}{{cut_suffix}}' -"
203
+ )
204
+ # step 5: mask tail in the RNA, which might be artifact of RT
205
+ steps.append(f"{cutadapt} -u {barcode.mask5} -u -{barcode.mask3} -")
206
+ # step 6: trim polyA
207
+ if settings.trim_polyA:
208
+ if barcode.strand == "+":
209
+ steps.append(f"{cutadapt} -O 6 -e 0.15 -a 'A{{100}}' -")
210
+ elif barcode.strand == "-":
211
+ steps.append(f"{cutadapt} -O 6 -e 0.15 -g 'T{{100}}' -")
212
+ else:
213
+ logging.info("No strand information provided, skip polyA trimming.")
214
+ # step 7: quality control, remove short reads
215
+ steps.append(
216
+ f"{cutadapt} -q {settings.min_quality} --max-n=0 -m {settings.min_length} --too-short-output={discard1} -o {output1} -"
217
+ )
218
+
219
+ if settings.dry_run:
220
+ print(
221
+ " |\\\n".join([(" " + s if i > 0 else s) for i, s in enumerate(steps)])
222
+ )
223
+ process = subprocess.run("true", shell=True, capture_output=True)
224
+ else:
225
+ cmd = " | ".join(steps)
226
+ process = subprocess.run(cmd, shell=True, capture_output=True)
227
+ return process.stdout.decode(), process.stderr.decode()
228
+
229
+
230
+ def run_cutseq(args):
231
+ barcode_config = BarcodeConfig(args.adapter)
232
+ settings = CutadaptConfig()
233
+ if args.with_rname_suffix:
234
+ settings.rname_suffix = True
235
+ if args.discarded_untrimmed:
236
+ settings.discarded_untrimmed = True
237
+ if args.trim_polyA:
238
+ settings.trim_polyA = True
239
+ settings.threads = args.threads
240
+ settings.min_length = args.min_length
241
+ settings.dry_run = args.dry_run
242
+ # Example command setup, you'll need to expand this based on your actual requirements
243
+ if len(args.input_file) == 1:
244
+ stdout, stderr = run_cutadapt_SE(
245
+ args.input_file[0],
246
+ args.output_file[0],
247
+ args.discard_file[0],
248
+ barcode_config,
249
+ settings,
250
+ )
251
+ else:
252
+ stdout, stderr = run_cutadapt_PE(
253
+ args.input_file[0],
254
+ args.input_file[1],
255
+ args.output_file[0],
256
+ args.output_file[1],
257
+ args.discard_file[0],
258
+ args.discard_file[1],
259
+ barcode_config,
260
+ settings,
261
+ )
262
+ print(stdout)
263
+
264
+
265
+ def main():
266
+ parser = argparse.ArgumentParser(
267
+ description="Trim sequencing adapters from NGS data automatically."
268
+ )
269
+ # input file can be one or two for single or paired-end reads, but can not be more than two
270
+ parser.add_argument(
271
+ "input_file",
272
+ type=str,
273
+ nargs="+",
274
+ help="Input file path for NGS data, one or two files.",
275
+ )
276
+ # output file can be number of files matching the input files, if not provided it will generate based on the output suffix,
277
+ # if no output suffix provided it will generate based on the input file name
278
+ parser.add_argument(
279
+ "-a",
280
+ "--adapter",
281
+ type=str,
282
+ required=True,
283
+ help="Adapter sequence configuration.",
284
+ )
285
+ parser.add_argument(
286
+ "-O",
287
+ "--output-suffix",
288
+ type=str,
289
+ help="Output file suffix for keep trimmed data.",
290
+ )
291
+ parser.add_argument(
292
+ "-o",
293
+ "--output-file",
294
+ type=str,
295
+ nargs="+",
296
+ help="Output file path for keep trimmed data.",
297
+ )
298
+
299
+ # discard short reads
300
+ parser.add_argument(
301
+ "-D",
302
+ "--discard-file",
303
+ type=str,
304
+ nargs="+",
305
+ help="Output file path for discarded trimmed data.",
306
+ )
307
+ parser.add_argument(
308
+ "-m",
309
+ "--min-length",
310
+ type=int,
311
+ default=20,
312
+ help="Minimum length of the reads to keep.",
313
+ )
314
+ parser.add_argument(
315
+ "-q",
316
+ "--min-quality",
317
+ type=int,
318
+ default=20,
319
+ help="Minimum quality of the read tails in the reads to keep..",
320
+ )
321
+
322
+ parser.add_argument(
323
+ "--with-rname-suffix",
324
+ action="store_true",
325
+ help="R1 and R2 suffix cotains suffix. MGI platform.",
326
+ )
327
+ parser.add_argument(
328
+ "--discarded-untrimmed",
329
+ action="store_true",
330
+ help="Discard untrimmed reads (without inline barcode matching).",
331
+ )
332
+ parser.add_argument("--trim-polyA", action="store_true", help="Trim polyA tail.")
333
+
334
+ parser.add_argument(
335
+ "-t",
336
+ "--threads",
337
+ type=int,
338
+ default=1,
339
+ help="Number of threads to use for trimming.",
340
+ )
341
+ parser.add_argument(
342
+ "-n",
343
+ "--dry-run",
344
+ action="store_true",
345
+ help="Print command instead of running it.",
346
+ )
347
+ args = parser.parse_args()
348
+
349
+ if len(args.input_file) > 2:
350
+ raise ValueError("Input file can not be more than two.")
351
+
352
+ if args.output_file:
353
+ if len(args.output_file) != len(args.input_file):
354
+ raise ValueError("Output file should be same as input file.")
355
+ elif args.output_suffix:
356
+ if len(args.input_file) == 1:
357
+ args.output_file = [args.output_suffix + "_trimmed_R1.fastq.gz"]
358
+ else:
359
+ args.output_file = [
360
+ args.output_suffix + "_trimmed_R1.fastq.gz",
361
+ args.output_suffix + "_trimmed_R2.fastq.gz",
362
+ ]
363
+ else:
364
+ if len(args.input_file) == 1:
365
+ args.output_file = [
366
+ remove_fq_suffix(args.input_file[0]) + "_trimmed_R1.fastq.gz",
367
+ ]
368
+ else:
369
+ args.output_file = [
370
+ remove_fq_suffix(args.input_file[0]) + "_trimmed_R1.fastq.gz",
371
+ remove_fq_suffix(args.input_file[1]) + "_trimmed_R2.fastq.gz",
372
+ ]
373
+
374
+ if args.discard_file:
375
+ if len(args.discard_file) != len(args.input_file):
376
+ raise ValueError("Discard file should be same as input file.")
377
+ elif args.output_suffix:
378
+ if len(args.input_file) == 1:
379
+ args.discard_file = [args.output_suffix + "_discarded_R1.fastq.gz"]
380
+ else:
381
+ args.discard_file = [
382
+ args.output_suffix + "_discarded_R1.fastq.gz",
383
+ args.output_suffix + "_discarded_R2.fastq.gz",
384
+ ]
385
+ else:
386
+ if len(args.input_file) == 1:
387
+ args.discard_file = [
388
+ remove_fq_suffix(args.input_file[0]) + "_discarded_R1.fastq.gz",
389
+ ]
390
+ else:
391
+ args.discard_file = [
392
+ remove_fq_suffix(args.input_file[0]) + "_discarded_R1.fastq.gz",
393
+ remove_fq_suffix(args.input_file[1]) + "_discarded_R2.fastq.gz",
394
+ ]
395
+ run_cutseq(args)
396
+
397
+
398
+ if __name__ == "__main__":
399
+ main()
@@ -0,0 +1,23 @@
1
+ Metadata-Version: 2.1
2
+ Name: cutseq
3
+ Version: 0.0.1
4
+ Summary: Automatic cutadapter and barcode process for NGS data
5
+ Home-page: https://github.com/y9c/cutseq
6
+ License: MIT
7
+ Keywords: bioinformatics,NGS,adapter,barcode,UMI
8
+ Author: Ye Chang
9
+ Author-email: yech1990@gmail.com
10
+ Requires-Python: >=3.8,<4.0
11
+ Classifier: License :: OSI Approved :: MIT License
12
+ Classifier: Programming Language :: Python :: 3
13
+ Classifier: Programming Language :: Python :: 3.8
14
+ Classifier: Programming Language :: Python :: 3.9
15
+ Classifier: Programming Language :: Python :: 3.10
16
+ Classifier: Programming Language :: Python :: 3.11
17
+ Classifier: Programming Language :: Python :: 3.12
18
+ Requires-Dist: cutadapt (>=4.8,<5.0)
19
+ Project-URL: Repository, https://github.com/y9c/cutseq
20
+ Description-Content-Type: text/markdown
21
+
22
+ # CutSeq
23
+
@@ -0,0 +1,5 @@
1
+ cutseq/run.py,sha256=WWQJKxd_tyc8Y4uQsjAo1fvHXd37skCLllqUEWJ-4p0,13807
2
+ cutseq-0.0.1.dist-info/METADATA,sha256=JdXExUzHOvGGdNWj0COyVqgVY8HFJ1U6_QnzvVnTDb4,789
3
+ cutseq-0.0.1.dist-info/WHEEL,sha256=sP946D7jFCHeNz5Iq4fL4Lu-PrWrFsgfLXbbkciIZwg,88
4
+ cutseq-0.0.1.dist-info/entry_points.txt,sha256=cz5WcGOzPAtTCE9mFfyDna3ofZv0NOeGtqE50rROBFU,42
5
+ cutseq-0.0.1.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: poetry-core 1.9.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,3 @@
1
+ [console_scripts]
2
+ cutseq=cutseq.run:main
3
+