chronaeon 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
chronaeon/__init__.py ADDED
@@ -0,0 +1,65 @@
1
+ """
2
+ ChronAeon: Ultra-Fast Molecular Clock Dating, Phylodynamics, and Genomic Surveillance.
3
+ """
4
+
5
+ from .dating import (
6
+ run_mrca_dating,
7
+ run_ols_dating,
8
+ run_pgls_dating,
9
+ run_restricted_spline_clock_dating,
10
+ run_powerlaw_clock_dating,
11
+ verify_coding_alignment,
12
+ parse_sample_dates,
13
+ generate_consensus_sequence,
14
+ generate_time_decay_consensus_sequence,
15
+ )
16
+ from .geo import (
17
+ run_phylogeography_analysis,
18
+ estimate_spatial_pgls_epicenter,
19
+ parse_geo_metadata,
20
+ )
21
+ from .r0 import (
22
+ run_r0_analysis,
23
+ compute_reproduction_numbers,
24
+ plot_r0_diagnostics,
25
+ PATHOGEN_PRESETS,
26
+ )
27
+ from .autoclock import (
28
+ AutoClockDeconvolution,
29
+ run_autoclock_deconvolution,
30
+ HierarchicalAutoClock,
31
+ run_hierarchical_autoclock,
32
+ fit_clock,
33
+ recursive_spectral_autoclock,
34
+ classify_community,
35
+ classify_leaf_community,
36
+ )
37
+ from .triage import ChronAeonSieve
38
+ from .sketch import (
39
+ CanonicalMinHashSketcher,
40
+ AlignmentFreeBinner,
41
+ AlignmentFreeCentrifuge,
42
+ )
43
+ from .alignment import (
44
+ ReferenceCodonAligner,
45
+ ReferenceGuidedCodonThreader,
46
+ )
47
+
48
+ __version__ = "0.1.0"
49
+ __all__ = [
50
+ "run_mrca_dating", "run_ols_dating", "run_pgls_dating",
51
+ "run_restricted_spline_clock_dating", "run_powerlaw_clock_dating",
52
+ "verify_coding_alignment", "parse_sample_dates",
53
+ "generate_consensus_sequence", "generate_time_decay_consensus_sequence",
54
+ "run_phylogeography_analysis", "estimate_spatial_pgls_epicenter",
55
+ "parse_geo_metadata",
56
+ "run_r0_analysis", "compute_reproduction_numbers",
57
+ "plot_r0_diagnostics", "PATHOGEN_PRESETS",
58
+ "AutoClockDeconvolution", "run_autoclock_deconvolution",
59
+ "HierarchicalAutoClock", "run_hierarchical_autoclock",
60
+ "fit_clock", "recursive_spectral_autoclock",
61
+ "classify_community", "classify_leaf_community",
62
+ "ChronAeonSieve",
63
+ "CanonicalMinHashSketcher", "AlignmentFreeBinner", "AlignmentFreeCentrifuge",
64
+ "ReferenceCodonAligner", "ReferenceGuidedCodonThreader",
65
+ ]
chronaeon/alignment.py ADDED
@@ -0,0 +1,200 @@
1
+ """
2
+ ChronAeon Reference-Guided Codon-Aware Threader
3
+ ==============================================
4
+ Provides linear-time O(N · L) frame-locked codon alignment against
5
+ canonical structural references.
6
+
7
+ Key Capabilities:
8
+ 1. Automatic reading frame & strand orientation detection.
9
+ 2. C-accelerated affine pairwise amino acid alignment against reference fold.
10
+ 3. Reverse-codon threading: guarantees uniform length L_ref across all
11
+ sequences, eliminates frame-shifts, and pads deletions with '---'.
12
+ 4. Directly outputs dense aligned codon tensors compatible with
13
+ PhyloAxialTransformer and ChronAeon.
14
+
15
+ Author: Sergei L. Kosakovsky Pond & DeepMind Antigravity Pair Programmer
16
+ """
17
+
18
+ import os
19
+ import sys
20
+ import time
21
+ from typing import Dict, List, Tuple, Optional, Any
22
+ from concurrent.futures import ThreadPoolExecutor
23
+ import numpy as np
24
+ from Bio import Align
25
+ from Bio.Seq import Seq
26
+
27
+
28
+ class ReferenceCodonAligner:
29
+ """
30
+ Aligns raw, unaligned coding sequences against a canonical reference CDS
31
+ by threading codons through amino acid alignment coordinates.
32
+ """
33
+
34
+ def __init__(
35
+ self,
36
+ ref_seq: str,
37
+ ref_name: str = "reference",
38
+ open_gap_score: float = -10.0,
39
+ extend_gap_score: float = -1.0,
40
+ match_score: float = 2.0,
41
+ mismatch_score: float = -1.0,
42
+ check_reverse_strand: bool = True,
43
+ ):
44
+ self.ref_name = ref_name
45
+ self.ref_dna = ref_seq.upper().replace("-", "").replace(".", "").strip()
46
+ if len(self.ref_dna) % 3 != 0:
47
+ rem = len(self.ref_dna) % 3
48
+ self.ref_dna = self.ref_dna[:-rem]
49
+
50
+ self.ref_aa = str(Seq(self.ref_dna).translate())
51
+ self.l_ref_codons = len(self.ref_dna) // 3
52
+ self.l_ref_nt = len(self.ref_dna)
53
+
54
+ self.check_reverse_strand = check_reverse_strand
55
+
56
+ # Configure C-accelerated Biopython PairwiseAligner
57
+ self.aligner = Align.PairwiseAligner()
58
+ self.aligner.mode = 'global'
59
+ self.aligner.open_gap_score = open_gap_score
60
+ self.aligner.extend_gap_score = extend_gap_score
61
+ self.aligner.match_score = match_score
62
+ self.aligner.mismatch_score = mismatch_score
63
+
64
+ def align_single(self, query_dna: str) -> Tuple[Optional[str], float, str]:
65
+ """
66
+ Aligns a single unaligned nucleotide query to the reference.
67
+
68
+ Returns:
69
+ (aligned_dna_str, alignment_score, status_message)
70
+ """
71
+ raw_q = query_dna.upper().replace("-", "").replace(".", "").strip()
72
+ if len(raw_q) < 30:
73
+ return None, 0.0, "too_short (<30 bp)"
74
+
75
+ candidates = [("forward", raw_q)]
76
+ if self.check_reverse_strand:
77
+ tr = str.maketrans("ACGTURYKMSWBDHVN", "TGCAAYRMKSWVHDBN")
78
+ rc_q = raw_q.translate(tr)[::-1]
79
+ candidates.append(("reverse", rc_q))
80
+
81
+ best_score = -1e9
82
+ best_aln = None
83
+ best_codons = None
84
+ best_strand = "forward"
85
+ best_frame = 0
86
+
87
+ for strand, q_seq in candidates:
88
+ for f in range(3):
89
+ sub = q_seq[f:]
90
+ trim_l = len(sub) - (len(sub) % 3)
91
+ sub = sub[:trim_l]
92
+ if len(sub) < 30:
93
+ continue
94
+
95
+ q_aa = str(Seq(sub).translate())
96
+ # Penalize severe internal stops
97
+ stop_count = q_aa[:-1].count("*")
98
+ if stop_count > 3:
99
+ continue
100
+
101
+ alns = self.aligner.align(self.ref_aa, q_aa)
102
+ if alns:
103
+ score = alns[0].score - (stop_count * 15.0)
104
+ if score > best_score:
105
+ best_score = score
106
+ best_aln = alns[0]
107
+ best_codons = [sub[i*3:i*3+3] for i in range(len(sub)//3)]
108
+ best_strand = strand
109
+ best_frame = f
110
+
111
+ if best_aln is None or best_score < 0:
112
+ return None, float(best_score), "alignment_failed_or_low_score"
113
+
114
+ # Reverse-thread codons into reference coordinates
115
+ ref_aln = best_aln[0]
116
+ qry_aln = best_aln[1]
117
+
118
+ threaded_codons = []
119
+ q_idx = 0
120
+ n_inserted_codons = 0
121
+
122
+ for col in range(len(ref_aln)):
123
+ r_char = ref_aln[col]
124
+ q_char = qry_aln[col]
125
+
126
+ if r_char != '-':
127
+ if q_char != '-':
128
+ codon = best_codons[q_idx]
129
+ threaded_codons.append(codon)
130
+ q_idx += 1
131
+ else:
132
+ threaded_codons.append('---')
133
+ else:
134
+ # Insertion relative to reference
135
+ if q_char != '-':
136
+ q_idx += 1
137
+ n_inserted_codons += 1
138
+
139
+ aligned_str = ''.join(threaded_codons)
140
+ if len(aligned_str) != self.l_ref_nt:
141
+ return None, float(best_score), f"length_mismatch ({len(aligned_str)} != {self.l_ref_nt})"
142
+
143
+ status = f"ok (strand={best_strand}, frame={best_frame}, score={best_score:.1f})"
144
+ return aligned_str, float(best_score), status
145
+
146
+ def align_batch(
147
+ self,
148
+ seq_dict: Dict[str, str],
149
+ max_workers: int = 4,
150
+ quiet: bool = False,
151
+ ) -> Dict[str, Any]:
152
+ """
153
+ Aligns a batch of sequences in parallel.
154
+
155
+ Returns:
156
+ Dict containing:
157
+ - 'aligned_seqs': {taxon: aligned_sequence}
158
+ - 'failed': {taxon: reason}
159
+ - 'scores': {taxon: score}
160
+ - 'elapsed_seconds': float
161
+ """
162
+ t0 = time.time()
163
+ n_total = len(seq_dict)
164
+ if not quiet:
165
+ print(f"[*] Reference-guided codon alignment of {n_total:,} sequences against '{self.ref_name}' ({self.l_ref_nt} bp)...", flush=True)
166
+
167
+ aligned_seqs: Dict[str, str] = {}
168
+ failed: Dict[str, str] = {}
169
+ scores: Dict[str, float] = {}
170
+
171
+ taxa = list(seq_dict.keys())
172
+
173
+ def _worker(t: str):
174
+ res_dna, score, status = self.align_single(seq_dict[t])
175
+ return t, res_dna, score, status
176
+
177
+ with ThreadPoolExecutor(max_workers=max_workers) as executor:
178
+ for t, res_dna, score, status in executor.map(_worker, taxa):
179
+ if res_dna is not None:
180
+ aligned_seqs[t] = res_dna
181
+ scores[t] = score
182
+ else:
183
+ failed[t] = status
184
+
185
+ t_elapsed = time.time() - t0
186
+ if not quiet:
187
+ rate = n_total / max(0.01, t_elapsed)
188
+ print(f"[✓] Aligned {len(aligned_seqs):,}/{n_total:,} sequences in {t_elapsed:.2f}s ({rate:.1f} seqs/s). Discarded {len(failed):,}.", flush=True)
189
+
190
+ return {
191
+ "aligned_seqs": aligned_seqs,
192
+ "failed": failed,
193
+ "scores": scores,
194
+ "l_ref_nt": self.l_ref_nt,
195
+ "elapsed_seconds": t_elapsed,
196
+ }
197
+
198
+
199
+ # Aliases for convenience
200
+ ReferenceGuidedCodonThreader = ReferenceCodonAligner