codonyat 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- aa_caller/__init__.py +16 -0
- aa_caller/__main__.py +4 -0
- aa_caller/app.py +781 -0
- aa_caller/runner.py +133 -0
- codonyat-0.1.0.dist-info/METADATA +140 -0
- codonyat-0.1.0.dist-info/RECORD +10 -0
- codonyat-0.1.0.dist-info/WHEEL +5 -0
- codonyat-0.1.0.dist-info/entry_points.txt +3 -0
- codonyat-0.1.0.dist-info/licenses/LICENSE +21 -0
- codonyat-0.1.0.dist-info/top_level.txt +1 -0
aa_caller/__init__.py
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
"""Light wrapper to expose the RT variant caller as a package."""
|
|
2
|
+
|
|
3
|
+
from .app import main, FullReference, SamContainer, SamEntry, parse_amplicons
|
|
4
|
+
from .runner import VariantCallResult, call_variants, call_variants_from_args, runner_cli
|
|
5
|
+
|
|
6
|
+
__all__ = [
|
|
7
|
+
"call_variants",
|
|
8
|
+
"call_variants_from_args",
|
|
9
|
+
"runner_cli",
|
|
10
|
+
"VariantCallResult",
|
|
11
|
+
"main",
|
|
12
|
+
"FullReference",
|
|
13
|
+
"SamContainer",
|
|
14
|
+
"SamEntry",
|
|
15
|
+
"parse_amplicons",
|
|
16
|
+
]
|
aa_caller/__main__.py
ADDED
aa_caller/app.py
ADDED
|
@@ -0,0 +1,781 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Standalone RT amino acid variant caller translated from stand_alone_fullkkwr2.pl.
|
|
3
|
+
|
|
4
|
+
This module exposes the full object model replicated from the legacy Perl
|
|
5
|
+
implementation so it can be executed within modern Python environments with
|
|
6
|
+
identical TSV/XML diagnostics and ratio/entropy helpers that the downstream
|
|
7
|
+
pipeline relies on."""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import argparse
|
|
12
|
+
import csv
|
|
13
|
+
import logging
|
|
14
|
+
import math
|
|
15
|
+
import re
|
|
16
|
+
import xml.etree.ElementTree as ET
|
|
17
|
+
from dataclasses import dataclass, field
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
from typing import Dict, List, Optional
|
|
20
|
+
|
|
21
|
+
from Bio import SeqIO
|
|
22
|
+
|
|
23
|
+
logger = logging.getLogger(__name__)
|
|
24
|
+
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class Qual:
|
|
28
|
+
"""Convert SAM quality strings into floating-point scores for filtering."""
|
|
29
|
+
_SOLEXA13 = {
|
|
30
|
+
"@": 0.01,
|
|
31
|
+
"A": 1,
|
|
32
|
+
"B": 2,
|
|
33
|
+
"C": 3,
|
|
34
|
+
"D": 4,
|
|
35
|
+
"E": 5,
|
|
36
|
+
"F": 6,
|
|
37
|
+
"G": 7,
|
|
38
|
+
"H": 8,
|
|
39
|
+
"I": 9,
|
|
40
|
+
"J": 10,
|
|
41
|
+
"K": 11,
|
|
42
|
+
"L": 12,
|
|
43
|
+
"M": 13,
|
|
44
|
+
"N": 14,
|
|
45
|
+
"O": 15,
|
|
46
|
+
"P": 16,
|
|
47
|
+
"Q": 17,
|
|
48
|
+
"R": 18,
|
|
49
|
+
"S": 19,
|
|
50
|
+
"T": 20,
|
|
51
|
+
"U": 21,
|
|
52
|
+
"V": 22,
|
|
53
|
+
"W": 23,
|
|
54
|
+
"X": 24,
|
|
55
|
+
"Y": 25,
|
|
56
|
+
"Z": 26,
|
|
57
|
+
"[": 27,
|
|
58
|
+
"\\": 28,
|
|
59
|
+
"]": 29,
|
|
60
|
+
"^": 30,
|
|
61
|
+
"_": 31,
|
|
62
|
+
"`": 32,
|
|
63
|
+
"a": 33,
|
|
64
|
+
"b": 34,
|
|
65
|
+
"c": 35,
|
|
66
|
+
"d": 36,
|
|
67
|
+
"e": 37,
|
|
68
|
+
"f": 38,
|
|
69
|
+
"g": 39,
|
|
70
|
+
"h": 40,
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
_SOLEXA = {
|
|
74
|
+
";": 0.01,
|
|
75
|
+
"<": 1,
|
|
76
|
+
"=": 2,
|
|
77
|
+
">": 3,
|
|
78
|
+
"?": 4,
|
|
79
|
+
"@": 5,
|
|
80
|
+
"A": 6,
|
|
81
|
+
"B": 7,
|
|
82
|
+
"C": 8,
|
|
83
|
+
"D": 9,
|
|
84
|
+
"E": 10,
|
|
85
|
+
"F": 11,
|
|
86
|
+
"G": 12,
|
|
87
|
+
"H": 13,
|
|
88
|
+
"I": 14,
|
|
89
|
+
"J": 15,
|
|
90
|
+
"K": 16,
|
|
91
|
+
"L": 17,
|
|
92
|
+
"M": 18,
|
|
93
|
+
"N": 19,
|
|
94
|
+
"O": 20,
|
|
95
|
+
"P": 21,
|
|
96
|
+
"Q": 22,
|
|
97
|
+
"R": 23,
|
|
98
|
+
"S": 24,
|
|
99
|
+
"T": 25,
|
|
100
|
+
"U": 26,
|
|
101
|
+
"V": 27,
|
|
102
|
+
"W": 28,
|
|
103
|
+
"X": 29,
|
|
104
|
+
"Y": 30,
|
|
105
|
+
"Z": 31,
|
|
106
|
+
"[": 32,
|
|
107
|
+
"\\": 33,
|
|
108
|
+
"]": 34,
|
|
109
|
+
"^": 35,
|
|
110
|
+
"_": 36,
|
|
111
|
+
"`": 37,
|
|
112
|
+
"a": 38,
|
|
113
|
+
"b": 39,
|
|
114
|
+
"c": 40,
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
_SANGER = {
|
|
118
|
+
"!": 0.01,
|
|
119
|
+
'"': 1,
|
|
120
|
+
"#": 2,
|
|
121
|
+
"$": 3,
|
|
122
|
+
"%": 4,
|
|
123
|
+
"&": 5,
|
|
124
|
+
"'": 6,
|
|
125
|
+
"(": 7,
|
|
126
|
+
")": 8,
|
|
127
|
+
"*": 9,
|
|
128
|
+
"+": 10,
|
|
129
|
+
",": 11,
|
|
130
|
+
"-": 12,
|
|
131
|
+
".": 13,
|
|
132
|
+
"/": 14,
|
|
133
|
+
"0": 15,
|
|
134
|
+
"1": 16,
|
|
135
|
+
"2": 17,
|
|
136
|
+
"3": 18,
|
|
137
|
+
"4": 19,
|
|
138
|
+
"5": 20,
|
|
139
|
+
"6": 21,
|
|
140
|
+
"7": 22,
|
|
141
|
+
"8": 23,
|
|
142
|
+
"9": 24,
|
|
143
|
+
":": 25,
|
|
144
|
+
";": 26,
|
|
145
|
+
"<": 27,
|
|
146
|
+
"=": 28,
|
|
147
|
+
">": 29,
|
|
148
|
+
"?": 30,
|
|
149
|
+
"@": 31,
|
|
150
|
+
"A": 32,
|
|
151
|
+
"B": 33,
|
|
152
|
+
"C": 34,
|
|
153
|
+
"D": 35,
|
|
154
|
+
"E": 36,
|
|
155
|
+
"F": 37,
|
|
156
|
+
"G": 38,
|
|
157
|
+
"H": 39,
|
|
158
|
+
"I": 40,
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
def __init__(self, quality_string: str, qual_type: str = "sanger") -> None:
|
|
162
|
+
"""Build a numeric representation of the provided FASTQ-style score string."""
|
|
163
|
+
self.quality_string = quality_string
|
|
164
|
+
self.qual_type = qual_type.lower()
|
|
165
|
+
self.quality_array = self._quality_to_numerical_array()
|
|
166
|
+
|
|
167
|
+
def _quality_to_numerical_array(self) -> List[float]:
|
|
168
|
+
"""Map each character to the configured score table, defaulting to Sanger."""
|
|
169
|
+
mapping = {
|
|
170
|
+
"solexa13": self._SOLEXA13,
|
|
171
|
+
"solexa1.3": self._SOLEXA13,
|
|
172
|
+
"solexa": self._SOLEXA,
|
|
173
|
+
"sanger": self._SANGER,
|
|
174
|
+
}.get(self.qual_type, self._SANGER)
|
|
175
|
+
return [mapping.get(symbol, 0.0) for symbol in self.quality_string]
|
|
176
|
+
|
|
177
|
+
def return_quality_by_pos(self, pos: int) -> float:
|
|
178
|
+
"""Retrieve the pre-computed quality score for a single read position."""
|
|
179
|
+
return self.quality_array[pos]
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
@dataclass
|
|
183
|
+
class Protein:
|
|
184
|
+
"""Stores the coordinates and description of a protein annotated in the reference."""
|
|
185
|
+
name: str
|
|
186
|
+
description: str
|
|
187
|
+
start_coordinate: int
|
|
188
|
+
end_coordinate: int
|
|
189
|
+
|
|
190
|
+
@classmethod
|
|
191
|
+
def from_string(cls, text: str) -> "Protein":
|
|
192
|
+
"""Parse a single FASTA header fragment describing a protein interval."""
|
|
193
|
+
left, interval = text.split(":")
|
|
194
|
+
name, desc = left.split("(")
|
|
195
|
+
desc = desc.strip(")")
|
|
196
|
+
start, end = interval.split("-")
|
|
197
|
+
return cls(name=name, description=desc, start_coordinate=int(start), end_coordinate=int(end))
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
@dataclass
|
|
201
|
+
class Gene:
|
|
202
|
+
"""Associates a gene name with its protein interval and genomic span."""
|
|
203
|
+
name: str
|
|
204
|
+
start_coordinate: int
|
|
205
|
+
end_coordinate: int
|
|
206
|
+
protein: Protein
|
|
207
|
+
|
|
208
|
+
@classmethod
|
|
209
|
+
def from_annotation(cls, line: str) -> "Gene":
|
|
210
|
+
"""Build a Gene object from a TSV-style annotation line."""
|
|
211
|
+
parts = line.strip().split("\t")
|
|
212
|
+
name = parts[1]
|
|
213
|
+
start_coordinate = int(parts[2])
|
|
214
|
+
end_coordinate = int(parts[3])
|
|
215
|
+
protein_text = ":".join(parts[4:8])
|
|
216
|
+
protein = Protein.from_string(protein_text)
|
|
217
|
+
return cls(name=name, start_coordinate=start_coordinate, end_coordinate=end_coordinate, protein=protein)
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
@dataclass
|
|
221
|
+
class Amplicon:
|
|
222
|
+
"""Captures metadata for each configured amplicon used in the pipeline."""
|
|
223
|
+
label: str
|
|
224
|
+
protein: str
|
|
225
|
+
reference: str
|
|
226
|
+
_5prime_sequence: str
|
|
227
|
+
_3prime_sequence: str
|
|
228
|
+
start_coordinate: int
|
|
229
|
+
end_coordinate: int
|
|
230
|
+
|
|
231
|
+
@classmethod
|
|
232
|
+
def from_string(cls, line: str) -> "Amplicon":
|
|
233
|
+
"""Translate a CSV/TSV record into the in-memory amplicon definition."""
|
|
234
|
+
clean = line.strip().replace('"', "")
|
|
235
|
+
parts = re.split(r"[\t,\s]+", clean)
|
|
236
|
+
if len(parts) < 7:
|
|
237
|
+
raise ValueError("Amplicon definition requires at least 7 columns")
|
|
238
|
+
return cls(
|
|
239
|
+
label=parts[0],
|
|
240
|
+
protein=parts[1],
|
|
241
|
+
reference=parts[2],
|
|
242
|
+
_5prime_sequence=parts[3],
|
|
243
|
+
_3prime_sequence=parts[4],
|
|
244
|
+
start_coordinate=int(parts[5]),
|
|
245
|
+
end_coordinate=int(parts[6]),
|
|
246
|
+
)
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
class FullReference:
|
|
250
|
+
"""Loads the reference sequence and its annotated proteins from FASTA."""
|
|
251
|
+
def __init__(self, fasta_path: Path) -> None:
|
|
252
|
+
"""Initialize the reference by parsing a FASTA file with protein metadata."""
|
|
253
|
+
self.path = fasta_path
|
|
254
|
+
self.seq: str = ""
|
|
255
|
+
self.id: str = ""
|
|
256
|
+
self.proteins: Dict[str, Protein] = {}
|
|
257
|
+
self._load_reference()
|
|
258
|
+
|
|
259
|
+
def _load_reference(self) -> None:
|
|
260
|
+
"""Populate the sequence and protein dictionary from the FASTA header."""
|
|
261
|
+
record = next(SeqIO.parse(str(self.path), "fasta"))
|
|
262
|
+
self.seq = str(record.seq).upper()
|
|
263
|
+
self.id = record.id
|
|
264
|
+
header_parts = record.description.split(";")
|
|
265
|
+
for part in header_parts:
|
|
266
|
+
part = part.strip()
|
|
267
|
+
if not part:
|
|
268
|
+
continue
|
|
269
|
+
protein = Protein.from_string(part)
|
|
270
|
+
self.proteins[protein.name] = protein
|
|
271
|
+
|
|
272
|
+
def get_seq_at(self, position: int, length: int = 3) -> str:
|
|
273
|
+
"""Return a substring of the reference centered at the provided base coordinate."""
|
|
274
|
+
return self.seq[position - 1 : position - 1 + length]
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
GENETIC_CODE = {
|
|
278
|
+
"TCA": "S",
|
|
279
|
+
"TCC": "S",
|
|
280
|
+
"TCG": "S",
|
|
281
|
+
"TCT": "S",
|
|
282
|
+
"TTC": "F",
|
|
283
|
+
"TTT": "F",
|
|
284
|
+
"TTA": "L",
|
|
285
|
+
"TTG": "L",
|
|
286
|
+
"TAC": "Y",
|
|
287
|
+
"TAT": "Y",
|
|
288
|
+
"TAA": "_",
|
|
289
|
+
"TAG": "_",
|
|
290
|
+
"TGC": "C",
|
|
291
|
+
"TGT": "C",
|
|
292
|
+
"TGA": "_",
|
|
293
|
+
"TGG": "W",
|
|
294
|
+
"CTA": "L",
|
|
295
|
+
"CTC": "L",
|
|
296
|
+
"CTG": "L",
|
|
297
|
+
"CTT": "L",
|
|
298
|
+
"CCA": "P",
|
|
299
|
+
"CAT": "H",
|
|
300
|
+
"CAA": "Q",
|
|
301
|
+
"CAG": "Q",
|
|
302
|
+
"CGA": "R",
|
|
303
|
+
"CGC": "R",
|
|
304
|
+
"CGG": "R",
|
|
305
|
+
"CGT": "R",
|
|
306
|
+
"ATA": "I",
|
|
307
|
+
"ATC": "I",
|
|
308
|
+
"ATT": "I",
|
|
309
|
+
"ATG": "M",
|
|
310
|
+
"ACA": "T",
|
|
311
|
+
"ACC": "T",
|
|
312
|
+
"ACG": "T",
|
|
313
|
+
"ACT": "T",
|
|
314
|
+
"AAC": "N",
|
|
315
|
+
"AAT": "N",
|
|
316
|
+
"AAA": "K",
|
|
317
|
+
"AAG": "K",
|
|
318
|
+
"AGC": "S",
|
|
319
|
+
"AGT": "S",
|
|
320
|
+
"AGA": "R",
|
|
321
|
+
"AGG": "R",
|
|
322
|
+
"CCC": "P",
|
|
323
|
+
"CCG": "P",
|
|
324
|
+
"CCT": "P",
|
|
325
|
+
"CAC": "H",
|
|
326
|
+
"GTA": "V",
|
|
327
|
+
"GTC": "V",
|
|
328
|
+
"GTG": "V",
|
|
329
|
+
"GTT": "V",
|
|
330
|
+
"GCA": "A",
|
|
331
|
+
"GCC": "A",
|
|
332
|
+
"GCG": "A",
|
|
333
|
+
"GCT": "A",
|
|
334
|
+
"GAC": "D",
|
|
335
|
+
"GAT": "D",
|
|
336
|
+
"GAA": "E",
|
|
337
|
+
"GAG": "E",
|
|
338
|
+
"GGA": "G",
|
|
339
|
+
"GGC": "G",
|
|
340
|
+
"GGG": "G",
|
|
341
|
+
"GGT": "G",
|
|
342
|
+
"---": "del",
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
DEFAULT_RATIO_UPPER = 3.16227766
|
|
346
|
+
DEFAULT_RATIO_LOWER = 0.316227766
|
|
347
|
+
DEFAULT_ENTROPY_THRESHOLD = 0.0
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
def codon_to_aminoacid(codon: str) -> str:
|
|
351
|
+
"""Translate a three-base codon into its corresponding amino acid symbol."""
|
|
352
|
+
return GENETIC_CODE.get(codon.upper(), "X")
|
|
353
|
+
|
|
354
|
+
|
|
355
|
+
@dataclass
|
|
356
|
+
class Variant:
|
|
357
|
+
"""Tracks counts and strand-aware details for every codon observed at a position."""
|
|
358
|
+
codon: str
|
|
359
|
+
count: int = 0
|
|
360
|
+
fw_reads: int = 0
|
|
361
|
+
rv_reads: int = 0
|
|
362
|
+
balanced_fw: int = 0
|
|
363
|
+
balanced_rv: int = 0
|
|
364
|
+
fw_freq: float = 0.0
|
|
365
|
+
rv_freq: float = 0.0
|
|
366
|
+
ratio: float = 0.0
|
|
367
|
+
amplicons: Dict[str, Dict[str, int]] = field(default_factory=dict)
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
class SamEntry:
|
|
371
|
+
"""Wraps a SAM record with convenience helpers used during variant aggregation."""
|
|
372
|
+
|
|
373
|
+
def __init__(self, line: str) -> None:
|
|
374
|
+
"""Parse a single SAM line and cache derived attributes that the pipeline needs."""
|
|
375
|
+
self.raw = line.strip()
|
|
376
|
+
self.fields = self.raw.split("\t")
|
|
377
|
+
self.identifier = self.fields[0]
|
|
378
|
+
self.flag = int(self.fields[1])
|
|
379
|
+
self.reference = self.fields[2]
|
|
380
|
+
self.coordinate = int(self.fields[3])
|
|
381
|
+
self.map_quality = int(self.fields[4])
|
|
382
|
+
self.cigar = self.fields[5]
|
|
383
|
+
self.sequence = self.fields[9]
|
|
384
|
+
self.quality_string = self.fields[10]
|
|
385
|
+
self.options = self.fields[11:]
|
|
386
|
+
self.orientation = self._determine_orientation()
|
|
387
|
+
self.occurences = self._parse_occurences()
|
|
388
|
+
self._reference_covered_length = self._reference_covered_length()
|
|
389
|
+
self._quality = Qual(self.quality_string, "sanger")
|
|
390
|
+
self.amplicon = self._parse_amplicon()
|
|
391
|
+
|
|
392
|
+
def _determine_orientation(self) -> str:
|
|
393
|
+
"""Return whether the read is mapped to the forward or reverse strand."""
|
|
394
|
+
if self.flag & 16:
|
|
395
|
+
return "R"
|
|
396
|
+
return "F"
|
|
397
|
+
|
|
398
|
+
def _parse_occurences(self) -> int:
|
|
399
|
+
"""Detect if the identifier encodes a multiplicity suffix and return count."""
|
|
400
|
+
parts = self.identifier.split("_")
|
|
401
|
+
tail = parts[-1]
|
|
402
|
+
if tail.isdigit():
|
|
403
|
+
return max(1, int(tail))
|
|
404
|
+
return 1
|
|
405
|
+
|
|
406
|
+
def _reference_covered_length(self) -> int:
|
|
407
|
+
"""Sum the CIGAR operations that consume reference bases to get coverage."""
|
|
408
|
+
ops = re.findall(r"(\d+)([MIDNSHP=X])", self.cigar)
|
|
409
|
+
total = 0
|
|
410
|
+
for length, op in ops:
|
|
411
|
+
if op in "MD=X":
|
|
412
|
+
total += int(length)
|
|
413
|
+
return total
|
|
414
|
+
|
|
415
|
+
def _parse_amplicon(self) -> str:
|
|
416
|
+
"""Extract the amplicon tag embedded in the read identifier, if present."""
|
|
417
|
+
match = re.search(r"(Amp_[0-9]+)", self.identifier)
|
|
418
|
+
if match:
|
|
419
|
+
return match.group(1)
|
|
420
|
+
return "Amp_NONE"
|
|
421
|
+
|
|
422
|
+
def covers(self, position: int) -> bool:
|
|
423
|
+
"""Check whether this read spans the requested reference position."""
|
|
424
|
+
return self.coordinate <= position <= self.coordinate + self._reference_covered_length - 1
|
|
425
|
+
|
|
426
|
+
def codon_at(self, position: int) -> Optional[str]:
|
|
427
|
+
"""Return the codon that starts at the provided reference position if the read covers it."""
|
|
428
|
+
if not self.covers(position + 2):
|
|
429
|
+
return None
|
|
430
|
+
start = position - self.coordinate
|
|
431
|
+
if start < 0 or start + 3 > len(self.sequence):
|
|
432
|
+
return None
|
|
433
|
+
return self.sequence[start : start + 3]
|
|
434
|
+
|
|
435
|
+
def is_mapped(self) -> bool:
|
|
436
|
+
"""True when the read is not flagged as unmapped."""
|
|
437
|
+
return not (self.flag & 4)
|
|
438
|
+
|
|
439
|
+
|
|
440
|
+
class SamContainer:
|
|
441
|
+
"""Aggregates SAM entries and computes RT variant statistics for an amplicon set."""
|
|
442
|
+
|
|
443
|
+
def __init__(
|
|
444
|
+
self,
|
|
445
|
+
sam_path: Path,
|
|
446
|
+
ratio_upper: float,
|
|
447
|
+
ratio_lower: float,
|
|
448
|
+
entropy_threshold: float,
|
|
449
|
+
) -> None:
|
|
450
|
+
"""Initialize the container with the SAM path, ratio window, and entropy threshold."""
|
|
451
|
+
self.sam_path = sam_path
|
|
452
|
+
self.reads: List[SamEntry] = []
|
|
453
|
+
self.variants: Dict[int, Dict[str, Variant]] = {}
|
|
454
|
+
self.ratio_upper = ratio_upper
|
|
455
|
+
self.ratio_lower = ratio_lower
|
|
456
|
+
self.entropy_threshold = entropy_threshold
|
|
457
|
+
|
|
458
|
+
def load(self) -> None:
|
|
459
|
+
"""Load every mapped read from the SAM file so they can be aggregated."""
|
|
460
|
+
with open(self.sam_path, "r") as fh:
|
|
461
|
+
for line in fh:
|
|
462
|
+
if line.startswith("@"):
|
|
463
|
+
continue
|
|
464
|
+
entry = SamEntry(line)
|
|
465
|
+
if entry.is_mapped():
|
|
466
|
+
self.reads.append(entry)
|
|
467
|
+
|
|
468
|
+
def calculate_variant_frequencies(self, protein_name: str, reference: FullReference) -> None:
|
|
469
|
+
"""Walk every RT position of the requested protein and build variant histories."""
|
|
470
|
+
if protein_name not in reference.proteins:
|
|
471
|
+
raise ValueError(f"Protein {protein_name} not found in reference")
|
|
472
|
+
protein = reference.proteins[protein_name]
|
|
473
|
+
for pos in range(protein.start_coordinate, protein.end_coordinate, 3):
|
|
474
|
+
variants: Dict[str, Variant] = {}
|
|
475
|
+
fw_depth = 0
|
|
476
|
+
rv_depth = 0
|
|
477
|
+
amplicons_info: Dict[str, Dict[str, int]] = {}
|
|
478
|
+
for read in self.reads:
|
|
479
|
+
# only consider reads that fully span the queried codon
|
|
480
|
+
if not read.covers(pos):
|
|
481
|
+
continue
|
|
482
|
+
codon = read.codon_at(pos)
|
|
483
|
+
if not codon or len(codon) < 3:
|
|
484
|
+
continue
|
|
485
|
+
depth_increment = read.occurences
|
|
486
|
+
# track forward/reverse depth separately for ratio calculations
|
|
487
|
+
if read.orientation == "F":
|
|
488
|
+
fw_depth += depth_increment
|
|
489
|
+
else:
|
|
490
|
+
rv_depth += depth_increment
|
|
491
|
+
variant = variants.setdefault(codon, Variant(codon=codon))
|
|
492
|
+
variant.count += depth_increment
|
|
493
|
+
if read.orientation == "F":
|
|
494
|
+
variant.fw_reads += depth_increment
|
|
495
|
+
else:
|
|
496
|
+
variant.rv_reads += depth_increment
|
|
497
|
+
amplicon_label = read.amplicon or "Amp_NONE"
|
|
498
|
+
# record how each amplicon contributes to the variant counts
|
|
499
|
+
amp_variant = variant.amplicons.setdefault(amplicon_label, {"fw": 0, "rv": 0})
|
|
500
|
+
if read.orientation == "F":
|
|
501
|
+
amp_variant["fw"] += depth_increment
|
|
502
|
+
else:
|
|
503
|
+
amp_variant["rv"] += depth_increment
|
|
504
|
+
amp_position = amplicons_info.setdefault(amplicon_label, {"fw_depth": 0, "rv_depth": 0, "depth": 0})
|
|
505
|
+
# accumulate per-amplicon depth metrics for downstream reporting
|
|
506
|
+
if read.orientation == "F":
|
|
507
|
+
amp_position["fw_depth"] += depth_increment
|
|
508
|
+
else:
|
|
509
|
+
amp_position["rv_depth"] += depth_increment
|
|
510
|
+
amp_position["depth"] += depth_increment
|
|
511
|
+
entry = {
|
|
512
|
+
"variants": variants,
|
|
513
|
+
"fw_depth": fw_depth,
|
|
514
|
+
"rv_depth": rv_depth,
|
|
515
|
+
"depth": fw_depth + rv_depth,
|
|
516
|
+
"amplicons": amplicons_info,
|
|
517
|
+
}
|
|
518
|
+
# cache the computed metrics for this protein position
|
|
519
|
+
self.variants[pos] = entry
|
|
520
|
+
self.calculate_ratios_nucleotide(pos)
|
|
521
|
+
self.calculate_ratios_aminoacid(pos)
|
|
522
|
+
self.calculate_shannon_entropy_single_position(pos)
|
|
523
|
+
self.log_position_stats(pos)
|
|
524
|
+
|
|
525
|
+
def calculate_ratios_nucleotide(self, position: int) -> None:
|
|
526
|
+
"""Determine strand-specific ratios per variant and flag the balanced reads."""
|
|
527
|
+
entry = self.variants.get(position)
|
|
528
|
+
if not entry or entry["fw_depth"] == 0 or entry["rv_depth"] == 0:
|
|
529
|
+
return
|
|
530
|
+
for variant in entry["variants"].values():
|
|
531
|
+
# compute the per-strand frequency for each codon then derive the ratio
|
|
532
|
+
variant.fw_freq = variant.fw_reads / entry["fw_depth"] if entry["fw_depth"] else 0
|
|
533
|
+
variant.rv_freq = variant.rv_reads / entry["rv_depth"] if entry["rv_depth"] else 0
|
|
534
|
+
variant.ratio = (variant.fw_freq / variant.rv_freq) if variant.rv_freq else 0
|
|
535
|
+
# only mark variants within the balanced ratio window
|
|
536
|
+
if self.ratio_lower <= variant.ratio <= self.ratio_upper:
|
|
537
|
+
variant.balanced_fw += variant.fw_reads
|
|
538
|
+
variant.balanced_rv += variant.rv_reads
|
|
539
|
+
|
|
540
|
+
def has_balanced_variants(self, position: int) -> bool:
|
|
541
|
+
"""Return True when at least one variant falls within the configured strand-ratio window."""
|
|
542
|
+
entry = self.variants.get(position)
|
|
543
|
+
if not entry:
|
|
544
|
+
return False
|
|
545
|
+
return any(self.ratio_lower <= variant.ratio <= self.ratio_upper for variant in entry["variants"].values())
|
|
546
|
+
|
|
547
|
+
def calculate_ratios_aminoacid(self, position: int) -> None:
|
|
548
|
+
"""Aggregate amino-acid frequencies across codon variants and capture balance metadata."""
|
|
549
|
+
entry = self.variants.get(position)
|
|
550
|
+
if not entry:
|
|
551
|
+
return
|
|
552
|
+
aa_stats: Dict[str, Dict[str, float]] = {}
|
|
553
|
+
for variant in entry["variants"].values():
|
|
554
|
+
aa = codon_to_aminoacid(variant.codon)
|
|
555
|
+
stats = aa_stats.setdefault(aa, {"count": 0.0, "fw": 0.0, "rv": 0.0})
|
|
556
|
+
stats["count"] += variant.count
|
|
557
|
+
stats["fw"] += variant.fw_reads
|
|
558
|
+
stats["rv"] += variant.rv_reads
|
|
559
|
+
for aa, stats in aa_stats.items():
|
|
560
|
+
fw = stats["fw"]
|
|
561
|
+
rv = stats["rv"]
|
|
562
|
+
ratio = (fw / rv) if rv else 0
|
|
563
|
+
# flag balanced amino-acid counts using the same ratio window as nucleotides
|
|
564
|
+
stats["ratio"] = ratio
|
|
565
|
+
stats["balanced"] = (fw + rv) if self.ratio_lower <= ratio <= self.ratio_upper else 0
|
|
566
|
+
entry["aminoacid_stats"] = aa_stats
|
|
567
|
+
|
|
568
|
+
def calculate_shannon_entropy_single_position(self, position: int) -> float:
|
|
569
|
+
"""Compute the Shannon entropy for the observed codon mixture at a position."""
|
|
570
|
+
entry = self.variants.get(position)
|
|
571
|
+
if not entry:
|
|
572
|
+
return 0.0
|
|
573
|
+
depth = entry.get("depth", 0)
|
|
574
|
+
if depth == 0:
|
|
575
|
+
entry["shannon_entropy"] = 0.0
|
|
576
|
+
return 0.0
|
|
577
|
+
entropy = 0.0
|
|
578
|
+
for variant in entry["variants"].values():
|
|
579
|
+
if variant.count == 0:
|
|
580
|
+
continue
|
|
581
|
+
p = variant.count / depth
|
|
582
|
+
# standard Shannon formula: sum of -p log(p) for each variant
|
|
583
|
+
entropy -= p * math.log(p)
|
|
584
|
+
entry["shannon_entropy"] = entropy
|
|
585
|
+
entry["entropy_pass"] = entropy >= self.entropy_threshold
|
|
586
|
+
return entropy
|
|
587
|
+
|
|
588
|
+
def log_position_stats(self, position: int) -> None:
|
|
589
|
+
"""Log strand balance and entropy so downstream debugging is easier."""
|
|
590
|
+
entry = self.variants.get(position)
|
|
591
|
+
if not entry:
|
|
592
|
+
return
|
|
593
|
+
balanced = [v.codon for v in entry["variants"].values() if self.ratio_lower <= v.ratio <= self.ratio_upper]
|
|
594
|
+
entropy = entry.get("shannon_entropy", 0.0)
|
|
595
|
+
logger.debug(
|
|
596
|
+
"Position %s -- depth=%s, fw=%s, rv=%s, entropy=%.3f, balanced=%s",
|
|
597
|
+
position,
|
|
598
|
+
entry["depth"],
|
|
599
|
+
entry["fw_depth"],
|
|
600
|
+
entry["rv_depth"],
|
|
601
|
+
entropy,
|
|
602
|
+
balanced,
|
|
603
|
+
)
|
|
604
|
+
|
|
605
|
+
def write_csv(self, sample_name: str, reference: FullReference, output_path: Path) -> None:
|
|
606
|
+
"""Emit the variant summary as a TSV file that mirrors the Perl outputs."""
|
|
607
|
+
fieldnames = [
|
|
608
|
+
"FILE",
|
|
609
|
+
"REFERENCE",
|
|
610
|
+
"PROTEIN",
|
|
611
|
+
"VARIANT",
|
|
612
|
+
"POSITION",
|
|
613
|
+
"FREQ",
|
|
614
|
+
"FWCOV",
|
|
615
|
+
"RVCOV",
|
|
616
|
+
"TOTALCOV",
|
|
617
|
+
"RATIO",
|
|
618
|
+
]
|
|
619
|
+
with open(output_path, "w", newline="") as csvfile:
|
|
620
|
+
writer = csv.DictWriter(csvfile, fieldnames=fieldnames, delimiter="\t")
|
|
621
|
+
writer.writeheader()
|
|
622
|
+
for pos, entry in sorted(self.variants.items()):
|
|
623
|
+
depth = entry["depth"]
|
|
624
|
+
ratio = entry["fw_depth"] / entry["rv_depth"] if entry["rv_depth"] else 0
|
|
625
|
+
for variant in entry["variants"].values():
|
|
626
|
+
# emit each variant using the cached depths and ratios
|
|
627
|
+
writer.writerow(
|
|
628
|
+
{
|
|
629
|
+
"FILE": sample_name,
|
|
630
|
+
"REFERENCE": reference.id,
|
|
631
|
+
"PROTEIN": "RT",
|
|
632
|
+
"VARIANT": variant.codon,
|
|
633
|
+
"POSITION": pos,
|
|
634
|
+
"FREQ": round((variant.count / depth * 100) if depth else 0, 3),
|
|
635
|
+
"FWCOV": entry["fw_depth"],
|
|
636
|
+
"RVCOV": entry["rv_depth"],
|
|
637
|
+
"TOTALCOV": depth,
|
|
638
|
+
"RATIO": round(ratio, 3),
|
|
639
|
+
}
|
|
640
|
+
)
|
|
641
|
+
|
|
642
|
+
def write_xml(self, sample_name: str, reference: FullReference, output_path: Path) -> None:
|
|
643
|
+
"""Dump the internal state into XML for downstream diagnostics."""
|
|
644
|
+
root = ET.Element("SamContainer", sample=sample_name, reference=reference.id)
|
|
645
|
+
for pos, entry in sorted(self.variants.items()):
|
|
646
|
+
position_elem = ET.SubElement(root, "Position", index=str(pos))
|
|
647
|
+
ET.SubElement(position_elem, "Depth").text = str(entry["depth"])
|
|
648
|
+
ET.SubElement(position_elem, "FwCover").text = str(entry["fw_depth"])
|
|
649
|
+
ET.SubElement(position_elem, "RvCover").text = str(entry["rv_depth"])
|
|
650
|
+
variants_elem = ET.SubElement(position_elem, "Variants")
|
|
651
|
+
for variant in entry["variants"].values():
|
|
652
|
+
# describe each variant using XML attributes for diagnostics
|
|
653
|
+
var_elem = ET.SubElement(variants_elem, "Variant", codon=variant.codon)
|
|
654
|
+
var_elem.set("count", str(variant.count))
|
|
655
|
+
var_elem.set("fw_reads", str(variant.fw_reads))
|
|
656
|
+
var_elem.set("rv_reads", str(variant.rv_reads))
|
|
657
|
+
tree = ET.ElementTree(root)
|
|
658
|
+
tree.write(output_path, encoding="utf-8", xml_declaration=True)
|
|
659
|
+
|
|
660
|
+
|
|
661
|
+
def validate_sam_file(path: Path) -> None:
|
|
662
|
+
"""Perform a lightweight sanity check on the SAM file before parsing."""
|
|
663
|
+
with open(path, "r") as fh:
|
|
664
|
+
for line in fh:
|
|
665
|
+
if line.startswith("@"):
|
|
666
|
+
continue
|
|
667
|
+
fields = line.strip().split("\t")
|
|
668
|
+
if len(fields) < 11 or not fields[0] or not fields[3].isdigit():
|
|
669
|
+
raise ValueError(f"SAM file {path} looks malformed: {line.strip()}")
|
|
670
|
+
return
|
|
671
|
+
raise ValueError(f"SAM file {path} contains no alignment entries")
|
|
672
|
+
|
|
673
|
+
|
|
674
|
+
def validate_reference_file(path: Path) -> None:
|
|
675
|
+
"""Ensure the reference FASTA carries the protein metadata that RT parsing needs."""
|
|
676
|
+
record = next(SeqIO.parse(str(path), "fasta"), None)
|
|
677
|
+
if record is None:
|
|
678
|
+
raise ValueError(f"Reference file {path} is empty or not FASTA")
|
|
679
|
+
if "RT" not in record.description:
|
|
680
|
+
raise ValueError(f"Reference {path} header is missing an RT protein annotation")
|
|
681
|
+
|
|
682
|
+
|
|
683
|
+
def validate_amplicon_file(path: Path) -> None:
|
|
684
|
+
"""Check that the amplicon TSV/CSV has at least a header plus one valid row."""
|
|
685
|
+
with open(path, "r") as fh:
|
|
686
|
+
header = fh.readline()
|
|
687
|
+
if not header or len(header.strip().split("\t")) < 7:
|
|
688
|
+
raise ValueError(f"Amplicon file {path} does not contain the expected header")
|
|
689
|
+
for line in fh:
|
|
690
|
+
if not line.strip():
|
|
691
|
+
continue
|
|
692
|
+
parts = re.split(r"[\t,\s]+", line.strip().replace('"', ""))
|
|
693
|
+
if len(parts) < 7:
|
|
694
|
+
raise ValueError(f"Amplicon line needs 7 columns: {line.strip()}")
|
|
695
|
+
return
|
|
696
|
+
raise ValueError(f"Amplicon file {path} contains no definitions")
|
|
697
|
+
|
|
698
|
+
|
|
699
|
+
def parse_amplicons(path: Path) -> Dict[str, Amplicon]:
|
|
700
|
+
"""Read the amplicon configuration file and construct Amplicon objects."""
|
|
701
|
+
amplicons: Dict[str, Amplicon] = {}
|
|
702
|
+
with open(path, "r") as fh:
|
|
703
|
+
next(fh)
|
|
704
|
+
for line in fh:
|
|
705
|
+
if not line.strip():
|
|
706
|
+
continue
|
|
707
|
+
# parse each non-empty line into an Amplicon object
|
|
708
|
+
amplicon = Amplicon.from_string(line)
|
|
709
|
+
amplicons[amplicon.label] = amplicon
|
|
710
|
+
return amplicons
|
|
711
|
+
|
|
712
|
+
|
|
713
|
+
def main() -> None:
|
|
714
|
+
"""Command-line entry point that wires inputs/key outputs together."""
|
|
715
|
+
parser = argparse.ArgumentParser(description="Standalone RT variant collector")
|
|
716
|
+
parser.add_argument("sam_file", type=Path)
|
|
717
|
+
parser.add_argument("reference_file", type=Path)
|
|
718
|
+
parser.add_argument("amplicons_file", type=Path)
|
|
719
|
+
parser.add_argument(
|
|
720
|
+
"--ratio-upper",
|
|
721
|
+
type=float,
|
|
722
|
+
default=DEFAULT_RATIO_UPPER,
|
|
723
|
+
help="Upper bound for strand-ratio balancing (default mirrors Perl).",
|
|
724
|
+
)
|
|
725
|
+
parser.add_argument(
|
|
726
|
+
"--ratio-lower",
|
|
727
|
+
type=float,
|
|
728
|
+
default=DEFAULT_RATIO_LOWER,
|
|
729
|
+
help="Lower bound for strand-ratio balancing (default mirrors Perl).",
|
|
730
|
+
)
|
|
731
|
+
parser.add_argument(
|
|
732
|
+
"--entropy-threshold",
|
|
733
|
+
type=float,
|
|
734
|
+
default=DEFAULT_ENTROPY_THRESHOLD,
|
|
735
|
+
help="Minimum Shannon entropy required to mark a position as diverse.",
|
|
736
|
+
)
|
|
737
|
+
args = parser.parse_args()
|
|
738
|
+
|
|
739
|
+
if not args.sam_file.exists():
|
|
740
|
+
parser.error(f"SAM file {args.sam_file} does not exist")
|
|
741
|
+
if not args.reference_file.exists():
|
|
742
|
+
parser.error(f"Reference file {args.reference_file} does not exist")
|
|
743
|
+
if not args.amplicons_file.exists():
|
|
744
|
+
parser.error(f"Amplicon file {args.amplicons_file} does not exist")
|
|
745
|
+
|
|
746
|
+
logger.info("Validating input formats before parsing")
|
|
747
|
+
validate_sam_file(args.sam_file)
|
|
748
|
+
validate_reference_file(args.reference_file)
|
|
749
|
+
validate_amplicon_file(args.amplicons_file)
|
|
750
|
+
|
|
751
|
+
logger.info("Reading amplicon configuration")
|
|
752
|
+
parse_amplicons(args.amplicons_file)
|
|
753
|
+
|
|
754
|
+
logger.info("Loading reference")
|
|
755
|
+
# reference contains the RT protein coordinates used for variant calling
|
|
756
|
+
reference = FullReference(args.reference_file)
|
|
757
|
+
|
|
758
|
+
logger.info("Parsing SAM entries")
|
|
759
|
+
# load every mapped read from the SAM file into the working buffer
|
|
760
|
+
container = SamContainer(
|
|
761
|
+
args.sam_file,
|
|
762
|
+
ratio_upper=args.ratio_upper,
|
|
763
|
+
ratio_lower=args.ratio_lower,
|
|
764
|
+
entropy_threshold=args.entropy_threshold,
|
|
765
|
+
)
|
|
766
|
+
container.load()
|
|
767
|
+
|
|
768
|
+
logger.info("Calculating RT variant frequencies")
|
|
769
|
+
container.calculate_variant_frequencies("RT", reference)
|
|
770
|
+
|
|
771
|
+
csv_path = args.sam_file.with_suffix(args.sam_file.suffix + ".tsv")
|
|
772
|
+
xml_path = args.sam_file.with_suffix(args.sam_file.suffix + ".xml")
|
|
773
|
+
logger.info("Writing TSV results")
|
|
774
|
+
# final exports mirror the Perl output format consumed by downstream workflows
|
|
775
|
+
container.write_csv(str(args.sam_file), reference, csv_path)
|
|
776
|
+
logger.info("Writing XML dump")
|
|
777
|
+
container.write_xml(str(args.sam_file), reference, xml_path)
|
|
778
|
+
|
|
779
|
+
|
|
780
|
+
if __name__ == "__main__":
|
|
781
|
+
main()
|
aa_caller/runner.py
ADDED
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
from argparse import Namespace, ArgumentParser
|
|
2
|
+
from dataclasses import dataclass
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from typing import Any, Dict, Mapping
|
|
5
|
+
|
|
6
|
+
from .app import (
|
|
7
|
+
DEFAULT_ENTROPY_THRESHOLD,
|
|
8
|
+
DEFAULT_RATIO_LOWER,
|
|
9
|
+
DEFAULT_RATIO_UPPER,
|
|
10
|
+
Amplicon,
|
|
11
|
+
FullReference,
|
|
12
|
+
SamContainer,
|
|
13
|
+
parse_amplicons,
|
|
14
|
+
validate_amplicon_file,
|
|
15
|
+
validate_reference_file,
|
|
16
|
+
validate_sam_file,
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass
|
|
21
|
+
class VariantCallResult:
|
|
22
|
+
"""Encapsulates the runner outputs along with the container state."""
|
|
23
|
+
|
|
24
|
+
csv_path: Path
|
|
25
|
+
xml_path: Path
|
|
26
|
+
container: SamContainer
|
|
27
|
+
reference: FullReference
|
|
28
|
+
amplicons: Dict[str, Amplicon]
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def call_variants(
|
|
32
|
+
sam_path: Path | str,
|
|
33
|
+
reference_path: Path | str,
|
|
34
|
+
amplicons_path: Path | str,
|
|
35
|
+
*,
|
|
36
|
+
ratio_upper: float = DEFAULT_RATIO_UPPER,
|
|
37
|
+
ratio_lower: float = DEFAULT_RATIO_LOWER,
|
|
38
|
+
entropy_threshold: float = DEFAULT_ENTROPY_THRESHOLD,
|
|
39
|
+
csv_path: Path | str | None = None,
|
|
40
|
+
xml_path: Path | str | None = None,
|
|
41
|
+
) -> VariantCallResult:
|
|
42
|
+
"""Run the codon-level pipeline and return the generated artifacts."""
|
|
43
|
+
|
|
44
|
+
sam_path = Path(sam_path)
|
|
45
|
+
reference_path = Path(reference_path)
|
|
46
|
+
amplicons_path = Path(amplicons_path)
|
|
47
|
+
|
|
48
|
+
validate_sam_file(sam_path)
|
|
49
|
+
validate_reference_file(reference_path)
|
|
50
|
+
validate_amplicon_file(amplicons_path)
|
|
51
|
+
|
|
52
|
+
amplicons = parse_amplicons(amplicons_path)
|
|
53
|
+
reference = FullReference(reference_path)
|
|
54
|
+
container = SamContainer(
|
|
55
|
+
sam_path,
|
|
56
|
+
ratio_upper=ratio_upper,
|
|
57
|
+
ratio_lower=ratio_lower,
|
|
58
|
+
entropy_threshold=entropy_threshold,
|
|
59
|
+
)
|
|
60
|
+
container.load()
|
|
61
|
+
container.calculate_variant_frequencies("RT", reference)
|
|
62
|
+
|
|
63
|
+
csv_path = Path(csv_path) if csv_path else sam_path.with_suffix(sam_path.suffix + ".tsv")
|
|
64
|
+
xml_path = Path(xml_path) if xml_path else sam_path.with_suffix(sam_path.suffix + ".xml")
|
|
65
|
+
|
|
66
|
+
container.write_csv(str(sam_path), reference, csv_path)
|
|
67
|
+
container.write_xml(str(sam_path), reference, xml_path)
|
|
68
|
+
|
|
69
|
+
return VariantCallResult(
|
|
70
|
+
csv_path=csv_path,
|
|
71
|
+
xml_path=xml_path,
|
|
72
|
+
container=container,
|
|
73
|
+
reference=reference,
|
|
74
|
+
amplicons=amplicons,
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def call_variants_from_args(
|
|
79
|
+
args: Namespace | Mapping[str, Any], *, ratio_upper: float | None = None, ratio_lower: float | None = None, entropy_threshold: float | None = None,
|
|
80
|
+
csv_path: Path | str | None = None,
|
|
81
|
+
xml_path: Path | str | None = None,
|
|
82
|
+
) -> VariantCallResult:
|
|
83
|
+
"""Run the pipeline using parsed CLI arguments or a plain mapping."""
|
|
84
|
+
|
|
85
|
+
def _pick(key: str, default: Any = None) -> Any:
|
|
86
|
+
if isinstance(args, Namespace):
|
|
87
|
+
return getattr(args, key, default)
|
|
88
|
+
if isinstance(args, Mapping):
|
|
89
|
+
return args.get(key, default)
|
|
90
|
+
return default
|
|
91
|
+
|
|
92
|
+
sam = _pick("sam_file", _pick("sam_path"))
|
|
93
|
+
reference = _pick("reference_file", _pick("reference_path"))
|
|
94
|
+
amplicons = _pick("amplicons_file", _pick("amplicons_path"))
|
|
95
|
+
if not (sam and reference and amplicons):
|
|
96
|
+
raise ValueError("args must include sam_file/reference_file/amplicons_file")
|
|
97
|
+
|
|
98
|
+
effective_ratio_upper = ratio_upper if ratio_upper is not None else _pick("ratio_upper", DEFAULT_RATIO_UPPER)
|
|
99
|
+
effective_ratio_lower = ratio_lower if ratio_lower is not None else _pick("ratio_lower", DEFAULT_RATIO_LOWER)
|
|
100
|
+
effective_entropy_threshold = entropy_threshold if entropy_threshold is not None else _pick("entropy_threshold", DEFAULT_ENTROPY_THRESHOLD)
|
|
101
|
+
|
|
102
|
+
overriding_csv = csv_path if csv_path is not None else _pick("csv_path")
|
|
103
|
+
overriding_xml = xml_path if xml_path is not None else _pick("xml_path")
|
|
104
|
+
|
|
105
|
+
return call_variants(
|
|
106
|
+
sam_path=sam,
|
|
107
|
+
reference_path=reference,
|
|
108
|
+
amplicons_path=amplicons,
|
|
109
|
+
ratio_upper=effective_ratio_upper,
|
|
110
|
+
ratio_lower=effective_ratio_lower,
|
|
111
|
+
entropy_threshold=effective_entropy_threshold,
|
|
112
|
+
csv_path=overriding_csv,
|
|
113
|
+
xml_path=overriding_xml,
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def runner_cli() -> None:
|
|
118
|
+
"""Simple CLI that forwards parsed arguments to `call_variants_from_args`."""
|
|
119
|
+
|
|
120
|
+
parser = ArgumentParser(description="Run the codonyat variant caller from Python")
|
|
121
|
+
parser.add_argument("sam_file", type=Path, help="Path to the input SAM file")
|
|
122
|
+
parser.add_argument("reference_file", type=Path, help="FASTA reference with RT annotations")
|
|
123
|
+
parser.add_argument("amplicons_file", type=Path, help="Amplicon TSV/CSV describing primers")
|
|
124
|
+
parser.add_argument("--ratio-upper", type=float, default=DEFAULT_RATIO_UPPER, help="Upper bound for strand balance")
|
|
125
|
+
parser.add_argument("--ratio-lower", type=float, default=DEFAULT_RATIO_LOWER, help="Lower bound for strand balance")
|
|
126
|
+
parser.add_argument("--entropy-threshold", type=float, default=DEFAULT_ENTROPY_THRESHOLD, help="Minimum entropy to mark a position")
|
|
127
|
+
parser.add_argument("--csv-path", type=Path, help="Override the TSV output path")
|
|
128
|
+
parser.add_argument("--xml-path", type=Path, help="Override the XML output path")
|
|
129
|
+
|
|
130
|
+
args = parser.parse_args()
|
|
131
|
+
result = call_variants_from_args(args, csv_path=args.csv_path, xml_path=args.xml_path)
|
|
132
|
+
print(f"Wrote TSV: {result.csv_path}")
|
|
133
|
+
print(f"Wrote XML: {result.xml_path}")
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: codonyat
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: codon-yat — A codon-aware amino acid variant typer from SAM alignments of viral NGS data.
|
|
5
|
+
Author-email: Marc Noguera Julian <info@treetopunder.com>
|
|
6
|
+
Project-URL: Source, https://github.com/mnoguera/aa_caller
|
|
7
|
+
Classifier: Programming Language :: Python :: 3
|
|
8
|
+
Classifier: License :: Other/Proprietary License
|
|
9
|
+
Classifier: Operating System :: OS Independent
|
|
10
|
+
Requires-Python: >=3.10
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
12
|
+
License-File: LICENSE
|
|
13
|
+
Requires-Dist: biopython>=1.79
|
|
14
|
+
Provides-Extra: dev
|
|
15
|
+
Requires-Dist: pytest; extra == "dev"
|
|
16
|
+
Requires-Dist: ruff; extra == "dev"
|
|
17
|
+
Dynamic: license-file
|
|
18
|
+
|
|
19
|
+
# codonyat
|
|
20
|
+
|
|
21
|
+
codon-yat — A codon-aware amino acid variant typer from SAM alignments of viral NGS data.
|
|
22
|
+
|
|
23
|
+
A standalone Python package that performs amino acid variant calling, including:
|
|
24
|
+
|
|
25
|
+
- SAM parsing that honors amplicon labels and strand orientation.
|
|
26
|
+
- Ratio balancing, entropy tracking, and TSV/XML exporters mirroring the legacy Perl outputs.
|
|
27
|
+
- Configurable CLI flags for strand ratio bounds and entropy sensitivity.
|
|
28
|
+
|
|
29
|
+
- Works directly on viral genomic datasets generated by high-throughput NGS pipelines and relies on protein-level annotations so you can derive amino-acid variants without reimplementing parsing logic.
|
|
30
|
+
|
|
31
|
+
## Highlights
|
|
32
|
+
|
|
33
|
+
- Produces consistent TSV/XML diagnostics while adding Python objects (`SamContainer`, `FullReference`, etc.) that downstream tooling can import.
|
|
34
|
+
- Validates every input file before parsing to surface malformed SAM/FASTA/amplicon data early.
|
|
35
|
+
- Logs per-position entropy and strand balance for easier debugging in CI or local runs.
|
|
36
|
+
|
|
37
|
+
## Installation
|
|
38
|
+
|
|
39
|
+
```bash
|
|
40
|
+
pip install .
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
Or publish the package (e.g., via `twine`/PyPI) and install it like any other dependency.
|
|
44
|
+
|
|
45
|
+
## CLI usage
|
|
46
|
+
|
|
47
|
+
Once installed, the `codonyat` entry point is available:
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
codonyat /path/to/sample.sam /path/to/reference.fasta /path/to/amplicons.tsv
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
Supply `--ratio-upper`, `--ratio-lower`, and `--entropy-threshold` to tune the balancing heuristics.
|
|
54
|
+
|
|
55
|
+
The CLI writes `[sam-file].tsv` (columns: FILE, REFERENCE, PROTEIN, VARIANT, POSITION, FREQ, FWCOV, RVCOV, TOTALCOV, RATIO) and `[sam-file].xml` (per-position `<Depth>`, `<FwCover>`, `<RvCover>`, `<Variants>`).
|
|
56
|
+
|
|
57
|
+
## Package API
|
|
58
|
+
|
|
59
|
+
Import `aa_caller` to reuse the core objects:
|
|
60
|
+
|
|
61
|
+
```python
|
|
62
|
+
from aa_caller import SamContainer, FullReference, parse_amplicons
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
The `SamContainer` constructor still accepts `ratio_upper`, `ratio_lower`, and `entropy_threshold` so you can reuse the balancing logic in scripts.
|
|
66
|
+
|
|
67
|
+
### Python wrapper
|
|
68
|
+
|
|
69
|
+
Use `call_variants` to run the full pipeline from Python without touching the CLI. It validates the inputs, builds the `SamContainer`, and writes the TSV/XML artifacts while returning a `VariantCallResult` you can inspect.
|
|
70
|
+
|
|
71
|
+
```python
|
|
72
|
+
from pathlib import Path
|
|
73
|
+
|
|
74
|
+
from aa_caller import call_variants
|
|
75
|
+
|
|
76
|
+
result = call_variants(
|
|
77
|
+
sam_path=Path("reads.sam"),
|
|
78
|
+
reference_path=Path("reference.fasta"),
|
|
79
|
+
amplicons_path=Path("amps.tsv"),
|
|
80
|
+
ratio_upper=3.2,
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
print(result.csv_path, result.xml_path)
|
|
84
|
+
print(result.container.variants.keys())
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
If you already have an `argparse.Namespace` or mapping of the CLI arguments, `call_variants_from_args` adapts them directly.
|
|
88
|
+
|
|
89
|
+
```python
|
|
90
|
+
from argparse import Namespace
|
|
91
|
+
|
|
92
|
+
args = Namespace(
|
|
93
|
+
sam_file="reads.sam",
|
|
94
|
+
reference_file="reference.fasta",
|
|
95
|
+
amplicons_file="amps.tsv",
|
|
96
|
+
ratio_upper=3.2,
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
call_variants_from_args(args)
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
### CLI wrapper
|
|
103
|
+
|
|
104
|
+
This repository installs a lightweight runner at `codonyat-runner` that exposes the same entry arguments as `call_variants_from_args`. Use it when you prefer a small CLI shim over the full `codonyat` entry point:
|
|
105
|
+
|
|
106
|
+
```bash
|
|
107
|
+
codonyat-runner reads.sam reference.fasta amps.tsv --ratio-upper 3.1 --csv-path results.tsv
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
## Development
|
|
111
|
+
|
|
112
|
+
Install the repository with the optional dev tooling so your local environment matches CI:
|
|
113
|
+
|
|
114
|
+
```bash
|
|
115
|
+
pip install --upgrade pip
|
|
116
|
+
pip install -e .[dev]
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
Now you can run the same checks that land in [.github/workflows/python-tests.yml](.github/workflows/python-tests.yml#L1-L27):
|
|
120
|
+
|
|
121
|
+
```bash
|
|
122
|
+
ruff check .
|
|
123
|
+
python -m pytest
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
The workflow installs `pytest` and `ruff`, runs the linter, and then executes the pytest suite on every push/PR against `main`.
|
|
127
|
+
|
|
128
|
+
## Testing
|
|
129
|
+
|
|
130
|
+
Run the upstream validation helpers with `pytest`:
|
|
131
|
+
|
|
132
|
+
```bash
|
|
133
|
+
python -m pytest
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
Keep the code tidy with `ruff` before committing:
|
|
137
|
+
|
|
138
|
+
```bash
|
|
139
|
+
ruff check .
|
|
140
|
+
```
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
aa_caller/__init__.py,sha256=DiyW2NABViudIoNWfIUzH3Itj2U9fmbozqKTqdIP5No,410
|
|
2
|
+
aa_caller/__main__.py,sha256=876xZTqhPeDBD2GKKr8pKX33Af8PhFTosAI8bYQsFgQ,61
|
|
3
|
+
aa_caller/app.py,sha256=2cbbaKso1ekwZyjxKFtqOxsDDsFk_LY4eV_ak-l1PFI,27112
|
|
4
|
+
aa_caller/runner.py,sha256=0UsIQWvUbFz9nzg2Ds279ARMg6tw7CwrGfSIFM21V-o,5160
|
|
5
|
+
codonyat-0.1.0.dist-info/licenses/LICENSE,sha256=KeMCiBK7OWzZYdhzm1OYzN0cpXc3OE-2uFB2KtVV6dM,1078
|
|
6
|
+
codonyat-0.1.0.dist-info/METADATA,sha256=q7B_8U4fHAApN3UvUS4PBWBl87DWl2zVisdyAHoVDDU,4379
|
|
7
|
+
codonyat-0.1.0.dist-info/WHEEL,sha256=aeYiig01lYGDzBgS8HxWXOg3uV61G9ijOsup-k9o1sk,91
|
|
8
|
+
codonyat-0.1.0.dist-info/entry_points.txt,sha256=vyWs-tzjcSH-OBUPTSvsCR0mCJT71g69X7zAdcB6jEA,94
|
|
9
|
+
codonyat-0.1.0.dist-info/top_level.txt,sha256=egBbinrv_BSVldRA-j7gJsvarAqMsJ3tLBq98KXfEvw,10
|
|
10
|
+
codonyat-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2024 codonyat contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
aa_caller
|