codonyat 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
aa_caller/__init__.py ADDED
@@ -0,0 +1,16 @@
1
+ """Light wrapper to expose the RT variant caller as a package."""
2
+
3
+ from .app import main, FullReference, SamContainer, SamEntry, parse_amplicons
4
+ from .runner import VariantCallResult, call_variants, call_variants_from_args, runner_cli
5
+
6
+ __all__ = [
7
+ "call_variants",
8
+ "call_variants_from_args",
9
+ "runner_cli",
10
+ "VariantCallResult",
11
+ "main",
12
+ "FullReference",
13
+ "SamContainer",
14
+ "SamEntry",
15
+ "parse_amplicons",
16
+ ]
aa_caller/__main__.py ADDED
@@ -0,0 +1,4 @@
1
+ from .app import main
2
+
3
+ if __name__ == "__main__":
4
+ main()
aa_caller/app.py ADDED
@@ -0,0 +1,781 @@
1
+ #!/usr/bin/env python3
2
+ """Standalone RT amino acid variant caller translated from stand_alone_fullkkwr2.pl.
3
+
4
+ This module exposes the full object model replicated from the legacy Perl
5
+ implementation so it can be executed within modern Python environments with
6
+ identical TSV/XML diagnostics and ratio/entropy helpers that the downstream
7
+ pipeline relies on."""
8
+
9
+ from __future__ import annotations
10
+
11
+ import argparse
12
+ import csv
13
+ import logging
14
+ import math
15
+ import re
16
+ import xml.etree.ElementTree as ET
17
+ from dataclasses import dataclass, field
18
+ from pathlib import Path
19
+ from typing import Dict, List, Optional
20
+
21
+ from Bio import SeqIO
22
+
23
+ logger = logging.getLogger(__name__)
24
+ logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
25
+
26
+
27
+ class Qual:
28
+ """Convert SAM quality strings into floating-point scores for filtering."""
29
+ _SOLEXA13 = {
30
+ "@": 0.01,
31
+ "A": 1,
32
+ "B": 2,
33
+ "C": 3,
34
+ "D": 4,
35
+ "E": 5,
36
+ "F": 6,
37
+ "G": 7,
38
+ "H": 8,
39
+ "I": 9,
40
+ "J": 10,
41
+ "K": 11,
42
+ "L": 12,
43
+ "M": 13,
44
+ "N": 14,
45
+ "O": 15,
46
+ "P": 16,
47
+ "Q": 17,
48
+ "R": 18,
49
+ "S": 19,
50
+ "T": 20,
51
+ "U": 21,
52
+ "V": 22,
53
+ "W": 23,
54
+ "X": 24,
55
+ "Y": 25,
56
+ "Z": 26,
57
+ "[": 27,
58
+ "\\": 28,
59
+ "]": 29,
60
+ "^": 30,
61
+ "_": 31,
62
+ "`": 32,
63
+ "a": 33,
64
+ "b": 34,
65
+ "c": 35,
66
+ "d": 36,
67
+ "e": 37,
68
+ "f": 38,
69
+ "g": 39,
70
+ "h": 40,
71
+ }
72
+
73
+ _SOLEXA = {
74
+ ";": 0.01,
75
+ "<": 1,
76
+ "=": 2,
77
+ ">": 3,
78
+ "?": 4,
79
+ "@": 5,
80
+ "A": 6,
81
+ "B": 7,
82
+ "C": 8,
83
+ "D": 9,
84
+ "E": 10,
85
+ "F": 11,
86
+ "G": 12,
87
+ "H": 13,
88
+ "I": 14,
89
+ "J": 15,
90
+ "K": 16,
91
+ "L": 17,
92
+ "M": 18,
93
+ "N": 19,
94
+ "O": 20,
95
+ "P": 21,
96
+ "Q": 22,
97
+ "R": 23,
98
+ "S": 24,
99
+ "T": 25,
100
+ "U": 26,
101
+ "V": 27,
102
+ "W": 28,
103
+ "X": 29,
104
+ "Y": 30,
105
+ "Z": 31,
106
+ "[": 32,
107
+ "\\": 33,
108
+ "]": 34,
109
+ "^": 35,
110
+ "_": 36,
111
+ "`": 37,
112
+ "a": 38,
113
+ "b": 39,
114
+ "c": 40,
115
+ }
116
+
117
+ _SANGER = {
118
+ "!": 0.01,
119
+ '"': 1,
120
+ "#": 2,
121
+ "$": 3,
122
+ "%": 4,
123
+ "&": 5,
124
+ "'": 6,
125
+ "(": 7,
126
+ ")": 8,
127
+ "*": 9,
128
+ "+": 10,
129
+ ",": 11,
130
+ "-": 12,
131
+ ".": 13,
132
+ "/": 14,
133
+ "0": 15,
134
+ "1": 16,
135
+ "2": 17,
136
+ "3": 18,
137
+ "4": 19,
138
+ "5": 20,
139
+ "6": 21,
140
+ "7": 22,
141
+ "8": 23,
142
+ "9": 24,
143
+ ":": 25,
144
+ ";": 26,
145
+ "<": 27,
146
+ "=": 28,
147
+ ">": 29,
148
+ "?": 30,
149
+ "@": 31,
150
+ "A": 32,
151
+ "B": 33,
152
+ "C": 34,
153
+ "D": 35,
154
+ "E": 36,
155
+ "F": 37,
156
+ "G": 38,
157
+ "H": 39,
158
+ "I": 40,
159
+ }
160
+
161
+ def __init__(self, quality_string: str, qual_type: str = "sanger") -> None:
162
+ """Build a numeric representation of the provided FASTQ-style score string."""
163
+ self.quality_string = quality_string
164
+ self.qual_type = qual_type.lower()
165
+ self.quality_array = self._quality_to_numerical_array()
166
+
167
+ def _quality_to_numerical_array(self) -> List[float]:
168
+ """Map each character to the configured score table, defaulting to Sanger."""
169
+ mapping = {
170
+ "solexa13": self._SOLEXA13,
171
+ "solexa1.3": self._SOLEXA13,
172
+ "solexa": self._SOLEXA,
173
+ "sanger": self._SANGER,
174
+ }.get(self.qual_type, self._SANGER)
175
+ return [mapping.get(symbol, 0.0) for symbol in self.quality_string]
176
+
177
+ def return_quality_by_pos(self, pos: int) -> float:
178
+ """Retrieve the pre-computed quality score for a single read position."""
179
+ return self.quality_array[pos]
180
+
181
+
182
+ @dataclass
183
+ class Protein:
184
+ """Stores the coordinates and description of a protein annotated in the reference."""
185
+ name: str
186
+ description: str
187
+ start_coordinate: int
188
+ end_coordinate: int
189
+
190
+ @classmethod
191
+ def from_string(cls, text: str) -> "Protein":
192
+ """Parse a single FASTA header fragment describing a protein interval."""
193
+ left, interval = text.split(":")
194
+ name, desc = left.split("(")
195
+ desc = desc.strip(")")
196
+ start, end = interval.split("-")
197
+ return cls(name=name, description=desc, start_coordinate=int(start), end_coordinate=int(end))
198
+
199
+
200
+ @dataclass
201
+ class Gene:
202
+ """Associates a gene name with its protein interval and genomic span."""
203
+ name: str
204
+ start_coordinate: int
205
+ end_coordinate: int
206
+ protein: Protein
207
+
208
+ @classmethod
209
+ def from_annotation(cls, line: str) -> "Gene":
210
+ """Build a Gene object from a TSV-style annotation line."""
211
+ parts = line.strip().split("\t")
212
+ name = parts[1]
213
+ start_coordinate = int(parts[2])
214
+ end_coordinate = int(parts[3])
215
+ protein_text = ":".join(parts[4:8])
216
+ protein = Protein.from_string(protein_text)
217
+ return cls(name=name, start_coordinate=start_coordinate, end_coordinate=end_coordinate, protein=protein)
218
+
219
+
220
+ @dataclass
221
+ class Amplicon:
222
+ """Captures metadata for each configured amplicon used in the pipeline."""
223
+ label: str
224
+ protein: str
225
+ reference: str
226
+ _5prime_sequence: str
227
+ _3prime_sequence: str
228
+ start_coordinate: int
229
+ end_coordinate: int
230
+
231
+ @classmethod
232
+ def from_string(cls, line: str) -> "Amplicon":
233
+ """Translate a CSV/TSV record into the in-memory amplicon definition."""
234
+ clean = line.strip().replace('"', "")
235
+ parts = re.split(r"[\t,\s]+", clean)
236
+ if len(parts) < 7:
237
+ raise ValueError("Amplicon definition requires at least 7 columns")
238
+ return cls(
239
+ label=parts[0],
240
+ protein=parts[1],
241
+ reference=parts[2],
242
+ _5prime_sequence=parts[3],
243
+ _3prime_sequence=parts[4],
244
+ start_coordinate=int(parts[5]),
245
+ end_coordinate=int(parts[6]),
246
+ )
247
+
248
+
249
+ class FullReference:
250
+ """Loads the reference sequence and its annotated proteins from FASTA."""
251
+ def __init__(self, fasta_path: Path) -> None:
252
+ """Initialize the reference by parsing a FASTA file with protein metadata."""
253
+ self.path = fasta_path
254
+ self.seq: str = ""
255
+ self.id: str = ""
256
+ self.proteins: Dict[str, Protein] = {}
257
+ self._load_reference()
258
+
259
+ def _load_reference(self) -> None:
260
+ """Populate the sequence and protein dictionary from the FASTA header."""
261
+ record = next(SeqIO.parse(str(self.path), "fasta"))
262
+ self.seq = str(record.seq).upper()
263
+ self.id = record.id
264
+ header_parts = record.description.split(";")
265
+ for part in header_parts:
266
+ part = part.strip()
267
+ if not part:
268
+ continue
269
+ protein = Protein.from_string(part)
270
+ self.proteins[protein.name] = protein
271
+
272
+ def get_seq_at(self, position: int, length: int = 3) -> str:
273
+ """Return a substring of the reference centered at the provided base coordinate."""
274
+ return self.seq[position - 1 : position - 1 + length]
275
+
276
+
277
+ GENETIC_CODE = {
278
+ "TCA": "S",
279
+ "TCC": "S",
280
+ "TCG": "S",
281
+ "TCT": "S",
282
+ "TTC": "F",
283
+ "TTT": "F",
284
+ "TTA": "L",
285
+ "TTG": "L",
286
+ "TAC": "Y",
287
+ "TAT": "Y",
288
+ "TAA": "_",
289
+ "TAG": "_",
290
+ "TGC": "C",
291
+ "TGT": "C",
292
+ "TGA": "_",
293
+ "TGG": "W",
294
+ "CTA": "L",
295
+ "CTC": "L",
296
+ "CTG": "L",
297
+ "CTT": "L",
298
+ "CCA": "P",
299
+ "CAT": "H",
300
+ "CAA": "Q",
301
+ "CAG": "Q",
302
+ "CGA": "R",
303
+ "CGC": "R",
304
+ "CGG": "R",
305
+ "CGT": "R",
306
+ "ATA": "I",
307
+ "ATC": "I",
308
+ "ATT": "I",
309
+ "ATG": "M",
310
+ "ACA": "T",
311
+ "ACC": "T",
312
+ "ACG": "T",
313
+ "ACT": "T",
314
+ "AAC": "N",
315
+ "AAT": "N",
316
+ "AAA": "K",
317
+ "AAG": "K",
318
+ "AGC": "S",
319
+ "AGT": "S",
320
+ "AGA": "R",
321
+ "AGG": "R",
322
+ "CCC": "P",
323
+ "CCG": "P",
324
+ "CCT": "P",
325
+ "CAC": "H",
326
+ "GTA": "V",
327
+ "GTC": "V",
328
+ "GTG": "V",
329
+ "GTT": "V",
330
+ "GCA": "A",
331
+ "GCC": "A",
332
+ "GCG": "A",
333
+ "GCT": "A",
334
+ "GAC": "D",
335
+ "GAT": "D",
336
+ "GAA": "E",
337
+ "GAG": "E",
338
+ "GGA": "G",
339
+ "GGC": "G",
340
+ "GGG": "G",
341
+ "GGT": "G",
342
+ "---": "del",
343
+ }
344
+
345
+ DEFAULT_RATIO_UPPER = 3.16227766
346
+ DEFAULT_RATIO_LOWER = 0.316227766
347
+ DEFAULT_ENTROPY_THRESHOLD = 0.0
348
+
349
+
350
+ def codon_to_aminoacid(codon: str) -> str:
351
+ """Translate a three-base codon into its corresponding amino acid symbol."""
352
+ return GENETIC_CODE.get(codon.upper(), "X")
353
+
354
+
355
+ @dataclass
356
+ class Variant:
357
+ """Tracks counts and strand-aware details for every codon observed at a position."""
358
+ codon: str
359
+ count: int = 0
360
+ fw_reads: int = 0
361
+ rv_reads: int = 0
362
+ balanced_fw: int = 0
363
+ balanced_rv: int = 0
364
+ fw_freq: float = 0.0
365
+ rv_freq: float = 0.0
366
+ ratio: float = 0.0
367
+ amplicons: Dict[str, Dict[str, int]] = field(default_factory=dict)
368
+
369
+
370
+ class SamEntry:
371
+ """Wraps a SAM record with convenience helpers used during variant aggregation."""
372
+
373
+ def __init__(self, line: str) -> None:
374
+ """Parse a single SAM line and cache derived attributes that the pipeline needs."""
375
+ self.raw = line.strip()
376
+ self.fields = self.raw.split("\t")
377
+ self.identifier = self.fields[0]
378
+ self.flag = int(self.fields[1])
379
+ self.reference = self.fields[2]
380
+ self.coordinate = int(self.fields[3])
381
+ self.map_quality = int(self.fields[4])
382
+ self.cigar = self.fields[5]
383
+ self.sequence = self.fields[9]
384
+ self.quality_string = self.fields[10]
385
+ self.options = self.fields[11:]
386
+ self.orientation = self._determine_orientation()
387
+ self.occurences = self._parse_occurences()
388
+ self._reference_covered_length = self._reference_covered_length()
389
+ self._quality = Qual(self.quality_string, "sanger")
390
+ self.amplicon = self._parse_amplicon()
391
+
392
+ def _determine_orientation(self) -> str:
393
+ """Return whether the read is mapped to the forward or reverse strand."""
394
+ if self.flag & 16:
395
+ return "R"
396
+ return "F"
397
+
398
+ def _parse_occurences(self) -> int:
399
+ """Detect if the identifier encodes a multiplicity suffix and return count."""
400
+ parts = self.identifier.split("_")
401
+ tail = parts[-1]
402
+ if tail.isdigit():
403
+ return max(1, int(tail))
404
+ return 1
405
+
406
+ def _reference_covered_length(self) -> int:
407
+ """Sum the CIGAR operations that consume reference bases to get coverage."""
408
+ ops = re.findall(r"(\d+)([MIDNSHP=X])", self.cigar)
409
+ total = 0
410
+ for length, op in ops:
411
+ if op in "MD=X":
412
+ total += int(length)
413
+ return total
414
+
415
+ def _parse_amplicon(self) -> str:
416
+ """Extract the amplicon tag embedded in the read identifier, if present."""
417
+ match = re.search(r"(Amp_[0-9]+)", self.identifier)
418
+ if match:
419
+ return match.group(1)
420
+ return "Amp_NONE"
421
+
422
+ def covers(self, position: int) -> bool:
423
+ """Check whether this read spans the requested reference position."""
424
+ return self.coordinate <= position <= self.coordinate + self._reference_covered_length - 1
425
+
426
+ def codon_at(self, position: int) -> Optional[str]:
427
+ """Return the codon that starts at the provided reference position if the read covers it."""
428
+ if not self.covers(position + 2):
429
+ return None
430
+ start = position - self.coordinate
431
+ if start < 0 or start + 3 > len(self.sequence):
432
+ return None
433
+ return self.sequence[start : start + 3]
434
+
435
+ def is_mapped(self) -> bool:
436
+ """True when the read is not flagged as unmapped."""
437
+ return not (self.flag & 4)
438
+
439
+
440
+ class SamContainer:
441
+ """Aggregates SAM entries and computes RT variant statistics for an amplicon set."""
442
+
443
+ def __init__(
444
+ self,
445
+ sam_path: Path,
446
+ ratio_upper: float,
447
+ ratio_lower: float,
448
+ entropy_threshold: float,
449
+ ) -> None:
450
+ """Initialize the container with the SAM path, ratio window, and entropy threshold."""
451
+ self.sam_path = sam_path
452
+ self.reads: List[SamEntry] = []
453
+ self.variants: Dict[int, Dict[str, Variant]] = {}
454
+ self.ratio_upper = ratio_upper
455
+ self.ratio_lower = ratio_lower
456
+ self.entropy_threshold = entropy_threshold
457
+
458
+ def load(self) -> None:
459
+ """Load every mapped read from the SAM file so they can be aggregated."""
460
+ with open(self.sam_path, "r") as fh:
461
+ for line in fh:
462
+ if line.startswith("@"):
463
+ continue
464
+ entry = SamEntry(line)
465
+ if entry.is_mapped():
466
+ self.reads.append(entry)
467
+
468
+ def calculate_variant_frequencies(self, protein_name: str, reference: FullReference) -> None:
469
+ """Walk every RT position of the requested protein and build variant histories."""
470
+ if protein_name not in reference.proteins:
471
+ raise ValueError(f"Protein {protein_name} not found in reference")
472
+ protein = reference.proteins[protein_name]
473
+ for pos in range(protein.start_coordinate, protein.end_coordinate, 3):
474
+ variants: Dict[str, Variant] = {}
475
+ fw_depth = 0
476
+ rv_depth = 0
477
+ amplicons_info: Dict[str, Dict[str, int]] = {}
478
+ for read in self.reads:
479
+ # only consider reads that fully span the queried codon
480
+ if not read.covers(pos):
481
+ continue
482
+ codon = read.codon_at(pos)
483
+ if not codon or len(codon) < 3:
484
+ continue
485
+ depth_increment = read.occurences
486
+ # track forward/reverse depth separately for ratio calculations
487
+ if read.orientation == "F":
488
+ fw_depth += depth_increment
489
+ else:
490
+ rv_depth += depth_increment
491
+ variant = variants.setdefault(codon, Variant(codon=codon))
492
+ variant.count += depth_increment
493
+ if read.orientation == "F":
494
+ variant.fw_reads += depth_increment
495
+ else:
496
+ variant.rv_reads += depth_increment
497
+ amplicon_label = read.amplicon or "Amp_NONE"
498
+ # record how each amplicon contributes to the variant counts
499
+ amp_variant = variant.amplicons.setdefault(amplicon_label, {"fw": 0, "rv": 0})
500
+ if read.orientation == "F":
501
+ amp_variant["fw"] += depth_increment
502
+ else:
503
+ amp_variant["rv"] += depth_increment
504
+ amp_position = amplicons_info.setdefault(amplicon_label, {"fw_depth": 0, "rv_depth": 0, "depth": 0})
505
+ # accumulate per-amplicon depth metrics for downstream reporting
506
+ if read.orientation == "F":
507
+ amp_position["fw_depth"] += depth_increment
508
+ else:
509
+ amp_position["rv_depth"] += depth_increment
510
+ amp_position["depth"] += depth_increment
511
+ entry = {
512
+ "variants": variants,
513
+ "fw_depth": fw_depth,
514
+ "rv_depth": rv_depth,
515
+ "depth": fw_depth + rv_depth,
516
+ "amplicons": amplicons_info,
517
+ }
518
+ # cache the computed metrics for this protein position
519
+ self.variants[pos] = entry
520
+ self.calculate_ratios_nucleotide(pos)
521
+ self.calculate_ratios_aminoacid(pos)
522
+ self.calculate_shannon_entropy_single_position(pos)
523
+ self.log_position_stats(pos)
524
+
525
+ def calculate_ratios_nucleotide(self, position: int) -> None:
526
+ """Determine strand-specific ratios per variant and flag the balanced reads."""
527
+ entry = self.variants.get(position)
528
+ if not entry or entry["fw_depth"] == 0 or entry["rv_depth"] == 0:
529
+ return
530
+ for variant in entry["variants"].values():
531
+ # compute the per-strand frequency for each codon then derive the ratio
532
+ variant.fw_freq = variant.fw_reads / entry["fw_depth"] if entry["fw_depth"] else 0
533
+ variant.rv_freq = variant.rv_reads / entry["rv_depth"] if entry["rv_depth"] else 0
534
+ variant.ratio = (variant.fw_freq / variant.rv_freq) if variant.rv_freq else 0
535
+ # only mark variants within the balanced ratio window
536
+ if self.ratio_lower <= variant.ratio <= self.ratio_upper:
537
+ variant.balanced_fw += variant.fw_reads
538
+ variant.balanced_rv += variant.rv_reads
539
+
540
+ def has_balanced_variants(self, position: int) -> bool:
541
+ """Return True when at least one variant falls within the configured strand-ratio window."""
542
+ entry = self.variants.get(position)
543
+ if not entry:
544
+ return False
545
+ return any(self.ratio_lower <= variant.ratio <= self.ratio_upper for variant in entry["variants"].values())
546
+
547
+ def calculate_ratios_aminoacid(self, position: int) -> None:
548
+ """Aggregate amino-acid frequencies across codon variants and capture balance metadata."""
549
+ entry = self.variants.get(position)
550
+ if not entry:
551
+ return
552
+ aa_stats: Dict[str, Dict[str, float]] = {}
553
+ for variant in entry["variants"].values():
554
+ aa = codon_to_aminoacid(variant.codon)
555
+ stats = aa_stats.setdefault(aa, {"count": 0.0, "fw": 0.0, "rv": 0.0})
556
+ stats["count"] += variant.count
557
+ stats["fw"] += variant.fw_reads
558
+ stats["rv"] += variant.rv_reads
559
+ for aa, stats in aa_stats.items():
560
+ fw = stats["fw"]
561
+ rv = stats["rv"]
562
+ ratio = (fw / rv) if rv else 0
563
+ # flag balanced amino-acid counts using the same ratio window as nucleotides
564
+ stats["ratio"] = ratio
565
+ stats["balanced"] = (fw + rv) if self.ratio_lower <= ratio <= self.ratio_upper else 0
566
+ entry["aminoacid_stats"] = aa_stats
567
+
568
+ def calculate_shannon_entropy_single_position(self, position: int) -> float:
569
+ """Compute the Shannon entropy for the observed codon mixture at a position."""
570
+ entry = self.variants.get(position)
571
+ if not entry:
572
+ return 0.0
573
+ depth = entry.get("depth", 0)
574
+ if depth == 0:
575
+ entry["shannon_entropy"] = 0.0
576
+ return 0.0
577
+ entropy = 0.0
578
+ for variant in entry["variants"].values():
579
+ if variant.count == 0:
580
+ continue
581
+ p = variant.count / depth
582
+ # standard Shannon formula: sum of -p log(p) for each variant
583
+ entropy -= p * math.log(p)
584
+ entry["shannon_entropy"] = entropy
585
+ entry["entropy_pass"] = entropy >= self.entropy_threshold
586
+ return entropy
587
+
588
+ def log_position_stats(self, position: int) -> None:
589
+ """Log strand balance and entropy so downstream debugging is easier."""
590
+ entry = self.variants.get(position)
591
+ if not entry:
592
+ return
593
+ balanced = [v.codon for v in entry["variants"].values() if self.ratio_lower <= v.ratio <= self.ratio_upper]
594
+ entropy = entry.get("shannon_entropy", 0.0)
595
+ logger.debug(
596
+ "Position %s -- depth=%s, fw=%s, rv=%s, entropy=%.3f, balanced=%s",
597
+ position,
598
+ entry["depth"],
599
+ entry["fw_depth"],
600
+ entry["rv_depth"],
601
+ entropy,
602
+ balanced,
603
+ )
604
+
605
+ def write_csv(self, sample_name: str, reference: FullReference, output_path: Path) -> None:
606
+ """Emit the variant summary as a TSV file that mirrors the Perl outputs."""
607
+ fieldnames = [
608
+ "FILE",
609
+ "REFERENCE",
610
+ "PROTEIN",
611
+ "VARIANT",
612
+ "POSITION",
613
+ "FREQ",
614
+ "FWCOV",
615
+ "RVCOV",
616
+ "TOTALCOV",
617
+ "RATIO",
618
+ ]
619
+ with open(output_path, "w", newline="") as csvfile:
620
+ writer = csv.DictWriter(csvfile, fieldnames=fieldnames, delimiter="\t")
621
+ writer.writeheader()
622
+ for pos, entry in sorted(self.variants.items()):
623
+ depth = entry["depth"]
624
+ ratio = entry["fw_depth"] / entry["rv_depth"] if entry["rv_depth"] else 0
625
+ for variant in entry["variants"].values():
626
+ # emit each variant using the cached depths and ratios
627
+ writer.writerow(
628
+ {
629
+ "FILE": sample_name,
630
+ "REFERENCE": reference.id,
631
+ "PROTEIN": "RT",
632
+ "VARIANT": variant.codon,
633
+ "POSITION": pos,
634
+ "FREQ": round((variant.count / depth * 100) if depth else 0, 3),
635
+ "FWCOV": entry["fw_depth"],
636
+ "RVCOV": entry["rv_depth"],
637
+ "TOTALCOV": depth,
638
+ "RATIO": round(ratio, 3),
639
+ }
640
+ )
641
+
642
+ def write_xml(self, sample_name: str, reference: FullReference, output_path: Path) -> None:
643
+ """Dump the internal state into XML for downstream diagnostics."""
644
+ root = ET.Element("SamContainer", sample=sample_name, reference=reference.id)
645
+ for pos, entry in sorted(self.variants.items()):
646
+ position_elem = ET.SubElement(root, "Position", index=str(pos))
647
+ ET.SubElement(position_elem, "Depth").text = str(entry["depth"])
648
+ ET.SubElement(position_elem, "FwCover").text = str(entry["fw_depth"])
649
+ ET.SubElement(position_elem, "RvCover").text = str(entry["rv_depth"])
650
+ variants_elem = ET.SubElement(position_elem, "Variants")
651
+ for variant in entry["variants"].values():
652
+ # describe each variant using XML attributes for diagnostics
653
+ var_elem = ET.SubElement(variants_elem, "Variant", codon=variant.codon)
654
+ var_elem.set("count", str(variant.count))
655
+ var_elem.set("fw_reads", str(variant.fw_reads))
656
+ var_elem.set("rv_reads", str(variant.rv_reads))
657
+ tree = ET.ElementTree(root)
658
+ tree.write(output_path, encoding="utf-8", xml_declaration=True)
659
+
660
+
661
+ def validate_sam_file(path: Path) -> None:
662
+ """Perform a lightweight sanity check on the SAM file before parsing."""
663
+ with open(path, "r") as fh:
664
+ for line in fh:
665
+ if line.startswith("@"):
666
+ continue
667
+ fields = line.strip().split("\t")
668
+ if len(fields) < 11 or not fields[0] or not fields[3].isdigit():
669
+ raise ValueError(f"SAM file {path} looks malformed: {line.strip()}")
670
+ return
671
+ raise ValueError(f"SAM file {path} contains no alignment entries")
672
+
673
+
674
+ def validate_reference_file(path: Path) -> None:
675
+ """Ensure the reference FASTA carries the protein metadata that RT parsing needs."""
676
+ record = next(SeqIO.parse(str(path), "fasta"), None)
677
+ if record is None:
678
+ raise ValueError(f"Reference file {path} is empty or not FASTA")
679
+ if "RT" not in record.description:
680
+ raise ValueError(f"Reference {path} header is missing an RT protein annotation")
681
+
682
+
683
+ def validate_amplicon_file(path: Path) -> None:
684
+ """Check that the amplicon TSV/CSV has at least a header plus one valid row."""
685
+ with open(path, "r") as fh:
686
+ header = fh.readline()
687
+ if not header or len(header.strip().split("\t")) < 7:
688
+ raise ValueError(f"Amplicon file {path} does not contain the expected header")
689
+ for line in fh:
690
+ if not line.strip():
691
+ continue
692
+ parts = re.split(r"[\t,\s]+", line.strip().replace('"', ""))
693
+ if len(parts) < 7:
694
+ raise ValueError(f"Amplicon line needs 7 columns: {line.strip()}")
695
+ return
696
+ raise ValueError(f"Amplicon file {path} contains no definitions")
697
+
698
+
699
+ def parse_amplicons(path: Path) -> Dict[str, Amplicon]:
700
+ """Read the amplicon configuration file and construct Amplicon objects."""
701
+ amplicons: Dict[str, Amplicon] = {}
702
+ with open(path, "r") as fh:
703
+ next(fh)
704
+ for line in fh:
705
+ if not line.strip():
706
+ continue
707
+ # parse each non-empty line into an Amplicon object
708
+ amplicon = Amplicon.from_string(line)
709
+ amplicons[amplicon.label] = amplicon
710
+ return amplicons
711
+
712
+
713
+ def main() -> None:
714
+ """Command-line entry point that wires inputs/key outputs together."""
715
+ parser = argparse.ArgumentParser(description="Standalone RT variant collector")
716
+ parser.add_argument("sam_file", type=Path)
717
+ parser.add_argument("reference_file", type=Path)
718
+ parser.add_argument("amplicons_file", type=Path)
719
+ parser.add_argument(
720
+ "--ratio-upper",
721
+ type=float,
722
+ default=DEFAULT_RATIO_UPPER,
723
+ help="Upper bound for strand-ratio balancing (default mirrors Perl).",
724
+ )
725
+ parser.add_argument(
726
+ "--ratio-lower",
727
+ type=float,
728
+ default=DEFAULT_RATIO_LOWER,
729
+ help="Lower bound for strand-ratio balancing (default mirrors Perl).",
730
+ )
731
+ parser.add_argument(
732
+ "--entropy-threshold",
733
+ type=float,
734
+ default=DEFAULT_ENTROPY_THRESHOLD,
735
+ help="Minimum Shannon entropy required to mark a position as diverse.",
736
+ )
737
+ args = parser.parse_args()
738
+
739
+ if not args.sam_file.exists():
740
+ parser.error(f"SAM file {args.sam_file} does not exist")
741
+ if not args.reference_file.exists():
742
+ parser.error(f"Reference file {args.reference_file} does not exist")
743
+ if not args.amplicons_file.exists():
744
+ parser.error(f"Amplicon file {args.amplicons_file} does not exist")
745
+
746
+ logger.info("Validating input formats before parsing")
747
+ validate_sam_file(args.sam_file)
748
+ validate_reference_file(args.reference_file)
749
+ validate_amplicon_file(args.amplicons_file)
750
+
751
+ logger.info("Reading amplicon configuration")
752
+ parse_amplicons(args.amplicons_file)
753
+
754
+ logger.info("Loading reference")
755
+ # reference contains the RT protein coordinates used for variant calling
756
+ reference = FullReference(args.reference_file)
757
+
758
+ logger.info("Parsing SAM entries")
759
+ # load every mapped read from the SAM file into the working buffer
760
+ container = SamContainer(
761
+ args.sam_file,
762
+ ratio_upper=args.ratio_upper,
763
+ ratio_lower=args.ratio_lower,
764
+ entropy_threshold=args.entropy_threshold,
765
+ )
766
+ container.load()
767
+
768
+ logger.info("Calculating RT variant frequencies")
769
+ container.calculate_variant_frequencies("RT", reference)
770
+
771
+ csv_path = args.sam_file.with_suffix(args.sam_file.suffix + ".tsv")
772
+ xml_path = args.sam_file.with_suffix(args.sam_file.suffix + ".xml")
773
+ logger.info("Writing TSV results")
774
+ # final exports mirror the Perl output format consumed by downstream workflows
775
+ container.write_csv(str(args.sam_file), reference, csv_path)
776
+ logger.info("Writing XML dump")
777
+ container.write_xml(str(args.sam_file), reference, xml_path)
778
+
779
+
780
+ if __name__ == "__main__":
781
+ main()
aa_caller/runner.py ADDED
@@ -0,0 +1,133 @@
1
+ from argparse import Namespace, ArgumentParser
2
+ from dataclasses import dataclass
3
+ from pathlib import Path
4
+ from typing import Any, Dict, Mapping
5
+
6
+ from .app import (
7
+ DEFAULT_ENTROPY_THRESHOLD,
8
+ DEFAULT_RATIO_LOWER,
9
+ DEFAULT_RATIO_UPPER,
10
+ Amplicon,
11
+ FullReference,
12
+ SamContainer,
13
+ parse_amplicons,
14
+ validate_amplicon_file,
15
+ validate_reference_file,
16
+ validate_sam_file,
17
+ )
18
+
19
+
20
+ @dataclass
21
+ class VariantCallResult:
22
+ """Encapsulates the runner outputs along with the container state."""
23
+
24
+ csv_path: Path
25
+ xml_path: Path
26
+ container: SamContainer
27
+ reference: FullReference
28
+ amplicons: Dict[str, Amplicon]
29
+
30
+
31
+ def call_variants(
32
+ sam_path: Path | str,
33
+ reference_path: Path | str,
34
+ amplicons_path: Path | str,
35
+ *,
36
+ ratio_upper: float = DEFAULT_RATIO_UPPER,
37
+ ratio_lower: float = DEFAULT_RATIO_LOWER,
38
+ entropy_threshold: float = DEFAULT_ENTROPY_THRESHOLD,
39
+ csv_path: Path | str | None = None,
40
+ xml_path: Path | str | None = None,
41
+ ) -> VariantCallResult:
42
+ """Run the codon-level pipeline and return the generated artifacts."""
43
+
44
+ sam_path = Path(sam_path)
45
+ reference_path = Path(reference_path)
46
+ amplicons_path = Path(amplicons_path)
47
+
48
+ validate_sam_file(sam_path)
49
+ validate_reference_file(reference_path)
50
+ validate_amplicon_file(amplicons_path)
51
+
52
+ amplicons = parse_amplicons(amplicons_path)
53
+ reference = FullReference(reference_path)
54
+ container = SamContainer(
55
+ sam_path,
56
+ ratio_upper=ratio_upper,
57
+ ratio_lower=ratio_lower,
58
+ entropy_threshold=entropy_threshold,
59
+ )
60
+ container.load()
61
+ container.calculate_variant_frequencies("RT", reference)
62
+
63
+ csv_path = Path(csv_path) if csv_path else sam_path.with_suffix(sam_path.suffix + ".tsv")
64
+ xml_path = Path(xml_path) if xml_path else sam_path.with_suffix(sam_path.suffix + ".xml")
65
+
66
+ container.write_csv(str(sam_path), reference, csv_path)
67
+ container.write_xml(str(sam_path), reference, xml_path)
68
+
69
+ return VariantCallResult(
70
+ csv_path=csv_path,
71
+ xml_path=xml_path,
72
+ container=container,
73
+ reference=reference,
74
+ amplicons=amplicons,
75
+ )
76
+
77
+
78
+ def call_variants_from_args(
79
+ args: Namespace | Mapping[str, Any], *, ratio_upper: float | None = None, ratio_lower: float | None = None, entropy_threshold: float | None = None,
80
+ csv_path: Path | str | None = None,
81
+ xml_path: Path | str | None = None,
82
+ ) -> VariantCallResult:
83
+ """Run the pipeline using parsed CLI arguments or a plain mapping."""
84
+
85
+ def _pick(key: str, default: Any = None) -> Any:
86
+ if isinstance(args, Namespace):
87
+ return getattr(args, key, default)
88
+ if isinstance(args, Mapping):
89
+ return args.get(key, default)
90
+ return default
91
+
92
+ sam = _pick("sam_file", _pick("sam_path"))
93
+ reference = _pick("reference_file", _pick("reference_path"))
94
+ amplicons = _pick("amplicons_file", _pick("amplicons_path"))
95
+ if not (sam and reference and amplicons):
96
+ raise ValueError("args must include sam_file/reference_file/amplicons_file")
97
+
98
+ effective_ratio_upper = ratio_upper if ratio_upper is not None else _pick("ratio_upper", DEFAULT_RATIO_UPPER)
99
+ effective_ratio_lower = ratio_lower if ratio_lower is not None else _pick("ratio_lower", DEFAULT_RATIO_LOWER)
100
+ effective_entropy_threshold = entropy_threshold if entropy_threshold is not None else _pick("entropy_threshold", DEFAULT_ENTROPY_THRESHOLD)
101
+
102
+ overriding_csv = csv_path if csv_path is not None else _pick("csv_path")
103
+ overriding_xml = xml_path if xml_path is not None else _pick("xml_path")
104
+
105
+ return call_variants(
106
+ sam_path=sam,
107
+ reference_path=reference,
108
+ amplicons_path=amplicons,
109
+ ratio_upper=effective_ratio_upper,
110
+ ratio_lower=effective_ratio_lower,
111
+ entropy_threshold=effective_entropy_threshold,
112
+ csv_path=overriding_csv,
113
+ xml_path=overriding_xml,
114
+ )
115
+
116
+
117
+ def runner_cli() -> None:
118
+ """Simple CLI that forwards parsed arguments to `call_variants_from_args`."""
119
+
120
+ parser = ArgumentParser(description="Run the codonyat variant caller from Python")
121
+ parser.add_argument("sam_file", type=Path, help="Path to the input SAM file")
122
+ parser.add_argument("reference_file", type=Path, help="FASTA reference with RT annotations")
123
+ parser.add_argument("amplicons_file", type=Path, help="Amplicon TSV/CSV describing primers")
124
+ parser.add_argument("--ratio-upper", type=float, default=DEFAULT_RATIO_UPPER, help="Upper bound for strand balance")
125
+ parser.add_argument("--ratio-lower", type=float, default=DEFAULT_RATIO_LOWER, help="Lower bound for strand balance")
126
+ parser.add_argument("--entropy-threshold", type=float, default=DEFAULT_ENTROPY_THRESHOLD, help="Minimum entropy to mark a position")
127
+ parser.add_argument("--csv-path", type=Path, help="Override the TSV output path")
128
+ parser.add_argument("--xml-path", type=Path, help="Override the XML output path")
129
+
130
+ args = parser.parse_args()
131
+ result = call_variants_from_args(args, csv_path=args.csv_path, xml_path=args.xml_path)
132
+ print(f"Wrote TSV: {result.csv_path}")
133
+ print(f"Wrote XML: {result.xml_path}")
@@ -0,0 +1,140 @@
1
+ Metadata-Version: 2.4
2
+ Name: codonyat
3
+ Version: 0.1.0
4
+ Summary: codon-yat — A codon-aware amino acid variant typer from SAM alignments of viral NGS data.
5
+ Author-email: Marc Noguera Julian <info@treetopunder.com>
6
+ Project-URL: Source, https://github.com/mnoguera/aa_caller
7
+ Classifier: Programming Language :: Python :: 3
8
+ Classifier: License :: Other/Proprietary License
9
+ Classifier: Operating System :: OS Independent
10
+ Requires-Python: >=3.10
11
+ Description-Content-Type: text/markdown
12
+ License-File: LICENSE
13
+ Requires-Dist: biopython>=1.79
14
+ Provides-Extra: dev
15
+ Requires-Dist: pytest; extra == "dev"
16
+ Requires-Dist: ruff; extra == "dev"
17
+ Dynamic: license-file
18
+
19
+ # codonyat
20
+
21
+ codon-yat — A codon-aware amino acid variant typer from SAM alignments of viral NGS data.
22
+
23
+ A standalone Python package that performs amino acid variant calling, including:
24
+
25
+ - SAM parsing that honors amplicon labels and strand orientation.
26
+ - Ratio balancing, entropy tracking, and TSV/XML exporters mirroring the legacy Perl outputs.
27
+ - Configurable CLI flags for strand ratio bounds and entropy sensitivity.
28
+
29
+ - Works directly on viral genomic datasets generated by high-throughput NGS pipelines and relies on protein-level annotations so you can derive amino-acid variants without reimplementing parsing logic.
30
+
31
+ ## Highlights
32
+
33
+ - Produces consistent TSV/XML diagnostics while adding Python objects (`SamContainer`, `FullReference`, etc.) that downstream tooling can import.
34
+ - Validates every input file before parsing to surface malformed SAM/FASTA/amplicon data early.
35
+ - Logs per-position entropy and strand balance for easier debugging in CI or local runs.
36
+
37
+ ## Installation
38
+
39
+ ```bash
40
+ pip install .
41
+ ```
42
+
43
+ Or publish the package (e.g., via `twine`/PyPI) and install it like any other dependency.
44
+
45
+ ## CLI usage
46
+
47
+ Once installed, the `codonyat` entry point is available:
48
+
49
+ ```bash
50
+ codonyat /path/to/sample.sam /path/to/reference.fasta /path/to/amplicons.tsv
51
+ ```
52
+
53
+ Supply `--ratio-upper`, `--ratio-lower`, and `--entropy-threshold` to tune the balancing heuristics.
54
+
55
+ The CLI writes `[sam-file].tsv` (columns: FILE, REFERENCE, PROTEIN, VARIANT, POSITION, FREQ, FWCOV, RVCOV, TOTALCOV, RATIO) and `[sam-file].xml` (per-position `<Depth>`, `<FwCover>`, `<RvCover>`, `<Variants>`).
56
+
57
+ ## Package API
58
+
59
+ Import `aa_caller` to reuse the core objects:
60
+
61
+ ```python
62
+ from aa_caller import SamContainer, FullReference, parse_amplicons
63
+ ```
64
+
65
+ The `SamContainer` constructor still accepts `ratio_upper`, `ratio_lower`, and `entropy_threshold` so you can reuse the balancing logic in scripts.
66
+
67
+ ### Python wrapper
68
+
69
+ Use `call_variants` to run the full pipeline from Python without touching the CLI. It validates the inputs, builds the `SamContainer`, and writes the TSV/XML artifacts while returning a `VariantCallResult` you can inspect.
70
+
71
+ ```python
72
+ from pathlib import Path
73
+
74
+ from aa_caller import call_variants
75
+
76
+ result = call_variants(
77
+ sam_path=Path("reads.sam"),
78
+ reference_path=Path("reference.fasta"),
79
+ amplicons_path=Path("amps.tsv"),
80
+ ratio_upper=3.2,
81
+ )
82
+
83
+ print(result.csv_path, result.xml_path)
84
+ print(result.container.variants.keys())
85
+ ```
86
+
87
+ If you already have an `argparse.Namespace` or mapping of the CLI arguments, `call_variants_from_args` adapts them directly.
88
+
89
+ ```python
90
+ from argparse import Namespace
91
+
92
+ args = Namespace(
93
+ sam_file="reads.sam",
94
+ reference_file="reference.fasta",
95
+ amplicons_file="amps.tsv",
96
+ ratio_upper=3.2,
97
+ )
98
+
99
+ call_variants_from_args(args)
100
+ ```
101
+
102
+ ### CLI wrapper
103
+
104
+ This repository installs a lightweight runner at `codonyat-runner` that exposes the same entry arguments as `call_variants_from_args`. Use it when you prefer a small CLI shim over the full `codonyat` entry point:
105
+
106
+ ```bash
107
+ codonyat-runner reads.sam reference.fasta amps.tsv --ratio-upper 3.1 --csv-path results.tsv
108
+ ```
109
+
110
+ ## Development
111
+
112
+ Install the repository with the optional dev tooling so your local environment matches CI:
113
+
114
+ ```bash
115
+ pip install --upgrade pip
116
+ pip install -e .[dev]
117
+ ```
118
+
119
+ Now you can run the same checks that land in [.github/workflows/python-tests.yml](.github/workflows/python-tests.yml#L1-L27):
120
+
121
+ ```bash
122
+ ruff check .
123
+ python -m pytest
124
+ ```
125
+
126
+ The workflow installs `pytest` and `ruff`, runs the linter, and then executes the pytest suite on every push/PR against `main`.
127
+
128
+ ## Testing
129
+
130
+ Run the upstream validation helpers with `pytest`:
131
+
132
+ ```bash
133
+ python -m pytest
134
+ ```
135
+
136
+ Keep the code tidy with `ruff` before committing:
137
+
138
+ ```bash
139
+ ruff check .
140
+ ```
@@ -0,0 +1,10 @@
1
+ aa_caller/__init__.py,sha256=DiyW2NABViudIoNWfIUzH3Itj2U9fmbozqKTqdIP5No,410
2
+ aa_caller/__main__.py,sha256=876xZTqhPeDBD2GKKr8pKX33Af8PhFTosAI8bYQsFgQ,61
3
+ aa_caller/app.py,sha256=2cbbaKso1ekwZyjxKFtqOxsDDsFk_LY4eV_ak-l1PFI,27112
4
+ aa_caller/runner.py,sha256=0UsIQWvUbFz9nzg2Ds279ARMg6tw7CwrGfSIFM21V-o,5160
5
+ codonyat-0.1.0.dist-info/licenses/LICENSE,sha256=KeMCiBK7OWzZYdhzm1OYzN0cpXc3OE-2uFB2KtVV6dM,1078
6
+ codonyat-0.1.0.dist-info/METADATA,sha256=q7B_8U4fHAApN3UvUS4PBWBl87DWl2zVisdyAHoVDDU,4379
7
+ codonyat-0.1.0.dist-info/WHEEL,sha256=aeYiig01lYGDzBgS8HxWXOg3uV61G9ijOsup-k9o1sk,91
8
+ codonyat-0.1.0.dist-info/entry_points.txt,sha256=vyWs-tzjcSH-OBUPTSvsCR0mCJT71g69X7zAdcB6jEA,94
9
+ codonyat-0.1.0.dist-info/top_level.txt,sha256=egBbinrv_BSVldRA-j7gJsvarAqMsJ3tLBq98KXfEvw,10
10
+ codonyat-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (82.0.1)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,3 @@
1
+ [console_scripts]
2
+ codonyat = aa_caller.app:main
3
+ codonyat-runner = aa_caller.runner:runner_cli
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2024 codonyat contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1 @@
1
+ aa_caller