chronaeon 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- chronaeon/__init__.py +65 -0
- chronaeon/alignment.py +200 -0
- chronaeon/autoclock.py +2600 -0
- chronaeon/beast_export.py +508 -0
- chronaeon/cli.py +991 -0
- chronaeon/dating.py +60 -0
- chronaeon/dating_divergence.py +440 -0
- chronaeon/dating_io.py +518 -0
- chronaeon/dating_kernels.py +215 -0
- chronaeon/dating_models.py +1288 -0
- chronaeon/dating_pipeline.py +1021 -0
- chronaeon/dating_plots.py +472 -0
- chronaeon/dudas.py +346 -0
- chronaeon/geo.py +921 -0
- chronaeon/r0.py +940 -0
- chronaeon/sketch.py +302 -0
- chronaeon/triage.py +803 -0
- chronaeon-0.1.0.dist-info/METADATA +246 -0
- chronaeon-0.1.0.dist-info/RECORD +23 -0
- chronaeon-0.1.0.dist-info/WHEEL +5 -0
- chronaeon-0.1.0.dist-info/entry_points.txt +2 -0
- chronaeon-0.1.0.dist-info/licenses/LICENSE +21 -0
- chronaeon-0.1.0.dist-info/top_level.txt +1 -0
chronaeon/__init__.py
ADDED
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
"""
|
|
2
|
+
ChronAeon: Ultra-Fast Molecular Clock Dating, Phylodynamics, and Genomic Surveillance.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
from .dating import (
|
|
6
|
+
run_mrca_dating,
|
|
7
|
+
run_ols_dating,
|
|
8
|
+
run_pgls_dating,
|
|
9
|
+
run_restricted_spline_clock_dating,
|
|
10
|
+
run_powerlaw_clock_dating,
|
|
11
|
+
verify_coding_alignment,
|
|
12
|
+
parse_sample_dates,
|
|
13
|
+
generate_consensus_sequence,
|
|
14
|
+
generate_time_decay_consensus_sequence,
|
|
15
|
+
)
|
|
16
|
+
from .geo import (
|
|
17
|
+
run_phylogeography_analysis,
|
|
18
|
+
estimate_spatial_pgls_epicenter,
|
|
19
|
+
parse_geo_metadata,
|
|
20
|
+
)
|
|
21
|
+
from .r0 import (
|
|
22
|
+
run_r0_analysis,
|
|
23
|
+
compute_reproduction_numbers,
|
|
24
|
+
plot_r0_diagnostics,
|
|
25
|
+
PATHOGEN_PRESETS,
|
|
26
|
+
)
|
|
27
|
+
from .autoclock import (
|
|
28
|
+
AutoClockDeconvolution,
|
|
29
|
+
run_autoclock_deconvolution,
|
|
30
|
+
HierarchicalAutoClock,
|
|
31
|
+
run_hierarchical_autoclock,
|
|
32
|
+
fit_clock,
|
|
33
|
+
recursive_spectral_autoclock,
|
|
34
|
+
classify_community,
|
|
35
|
+
classify_leaf_community,
|
|
36
|
+
)
|
|
37
|
+
from .triage import ChronAeonSieve
|
|
38
|
+
from .sketch import (
|
|
39
|
+
CanonicalMinHashSketcher,
|
|
40
|
+
AlignmentFreeBinner,
|
|
41
|
+
AlignmentFreeCentrifuge,
|
|
42
|
+
)
|
|
43
|
+
from .alignment import (
|
|
44
|
+
ReferenceCodonAligner,
|
|
45
|
+
ReferenceGuidedCodonThreader,
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
__version__ = "0.1.0"
|
|
49
|
+
__all__ = [
|
|
50
|
+
"run_mrca_dating", "run_ols_dating", "run_pgls_dating",
|
|
51
|
+
"run_restricted_spline_clock_dating", "run_powerlaw_clock_dating",
|
|
52
|
+
"verify_coding_alignment", "parse_sample_dates",
|
|
53
|
+
"generate_consensus_sequence", "generate_time_decay_consensus_sequence",
|
|
54
|
+
"run_phylogeography_analysis", "estimate_spatial_pgls_epicenter",
|
|
55
|
+
"parse_geo_metadata",
|
|
56
|
+
"run_r0_analysis", "compute_reproduction_numbers",
|
|
57
|
+
"plot_r0_diagnostics", "PATHOGEN_PRESETS",
|
|
58
|
+
"AutoClockDeconvolution", "run_autoclock_deconvolution",
|
|
59
|
+
"HierarchicalAutoClock", "run_hierarchical_autoclock",
|
|
60
|
+
"fit_clock", "recursive_spectral_autoclock",
|
|
61
|
+
"classify_community", "classify_leaf_community",
|
|
62
|
+
"ChronAeonSieve",
|
|
63
|
+
"CanonicalMinHashSketcher", "AlignmentFreeBinner", "AlignmentFreeCentrifuge",
|
|
64
|
+
"ReferenceCodonAligner", "ReferenceGuidedCodonThreader",
|
|
65
|
+
]
|
chronaeon/alignment.py
ADDED
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
"""
|
|
2
|
+
ChronAeon Reference-Guided Codon-Aware Threader
|
|
3
|
+
==============================================
|
|
4
|
+
Provides linear-time O(N · L) frame-locked codon alignment against
|
|
5
|
+
canonical structural references.
|
|
6
|
+
|
|
7
|
+
Key Capabilities:
|
|
8
|
+
1. Automatic reading frame & strand orientation detection.
|
|
9
|
+
2. C-accelerated affine pairwise amino acid alignment against reference fold.
|
|
10
|
+
3. Reverse-codon threading: guarantees uniform length L_ref across all
|
|
11
|
+
sequences, eliminates frame-shifts, and pads deletions with '---'.
|
|
12
|
+
4. Directly outputs dense aligned codon tensors compatible with
|
|
13
|
+
PhyloAxialTransformer and ChronAeon.
|
|
14
|
+
|
|
15
|
+
Author: Sergei L. Kosakovsky Pond & DeepMind Antigravity Pair Programmer
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
import os
|
|
19
|
+
import sys
|
|
20
|
+
import time
|
|
21
|
+
from typing import Dict, List, Tuple, Optional, Any
|
|
22
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
23
|
+
import numpy as np
|
|
24
|
+
from Bio import Align
|
|
25
|
+
from Bio.Seq import Seq
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class ReferenceCodonAligner:
|
|
29
|
+
"""
|
|
30
|
+
Aligns raw, unaligned coding sequences against a canonical reference CDS
|
|
31
|
+
by threading codons through amino acid alignment coordinates.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
def __init__(
|
|
35
|
+
self,
|
|
36
|
+
ref_seq: str,
|
|
37
|
+
ref_name: str = "reference",
|
|
38
|
+
open_gap_score: float = -10.0,
|
|
39
|
+
extend_gap_score: float = -1.0,
|
|
40
|
+
match_score: float = 2.0,
|
|
41
|
+
mismatch_score: float = -1.0,
|
|
42
|
+
check_reverse_strand: bool = True,
|
|
43
|
+
):
|
|
44
|
+
self.ref_name = ref_name
|
|
45
|
+
self.ref_dna = ref_seq.upper().replace("-", "").replace(".", "").strip()
|
|
46
|
+
if len(self.ref_dna) % 3 != 0:
|
|
47
|
+
rem = len(self.ref_dna) % 3
|
|
48
|
+
self.ref_dna = self.ref_dna[:-rem]
|
|
49
|
+
|
|
50
|
+
self.ref_aa = str(Seq(self.ref_dna).translate())
|
|
51
|
+
self.l_ref_codons = len(self.ref_dna) // 3
|
|
52
|
+
self.l_ref_nt = len(self.ref_dna)
|
|
53
|
+
|
|
54
|
+
self.check_reverse_strand = check_reverse_strand
|
|
55
|
+
|
|
56
|
+
# Configure C-accelerated Biopython PairwiseAligner
|
|
57
|
+
self.aligner = Align.PairwiseAligner()
|
|
58
|
+
self.aligner.mode = 'global'
|
|
59
|
+
self.aligner.open_gap_score = open_gap_score
|
|
60
|
+
self.aligner.extend_gap_score = extend_gap_score
|
|
61
|
+
self.aligner.match_score = match_score
|
|
62
|
+
self.aligner.mismatch_score = mismatch_score
|
|
63
|
+
|
|
64
|
+
def align_single(self, query_dna: str) -> Tuple[Optional[str], float, str]:
|
|
65
|
+
"""
|
|
66
|
+
Aligns a single unaligned nucleotide query to the reference.
|
|
67
|
+
|
|
68
|
+
Returns:
|
|
69
|
+
(aligned_dna_str, alignment_score, status_message)
|
|
70
|
+
"""
|
|
71
|
+
raw_q = query_dna.upper().replace("-", "").replace(".", "").strip()
|
|
72
|
+
if len(raw_q) < 30:
|
|
73
|
+
return None, 0.0, "too_short (<30 bp)"
|
|
74
|
+
|
|
75
|
+
candidates = [("forward", raw_q)]
|
|
76
|
+
if self.check_reverse_strand:
|
|
77
|
+
tr = str.maketrans("ACGTURYKMSWBDHVN", "TGCAAYRMKSWVHDBN")
|
|
78
|
+
rc_q = raw_q.translate(tr)[::-1]
|
|
79
|
+
candidates.append(("reverse", rc_q))
|
|
80
|
+
|
|
81
|
+
best_score = -1e9
|
|
82
|
+
best_aln = None
|
|
83
|
+
best_codons = None
|
|
84
|
+
best_strand = "forward"
|
|
85
|
+
best_frame = 0
|
|
86
|
+
|
|
87
|
+
for strand, q_seq in candidates:
|
|
88
|
+
for f in range(3):
|
|
89
|
+
sub = q_seq[f:]
|
|
90
|
+
trim_l = len(sub) - (len(sub) % 3)
|
|
91
|
+
sub = sub[:trim_l]
|
|
92
|
+
if len(sub) < 30:
|
|
93
|
+
continue
|
|
94
|
+
|
|
95
|
+
q_aa = str(Seq(sub).translate())
|
|
96
|
+
# Penalize severe internal stops
|
|
97
|
+
stop_count = q_aa[:-1].count("*")
|
|
98
|
+
if stop_count > 3:
|
|
99
|
+
continue
|
|
100
|
+
|
|
101
|
+
alns = self.aligner.align(self.ref_aa, q_aa)
|
|
102
|
+
if alns:
|
|
103
|
+
score = alns[0].score - (stop_count * 15.0)
|
|
104
|
+
if score > best_score:
|
|
105
|
+
best_score = score
|
|
106
|
+
best_aln = alns[0]
|
|
107
|
+
best_codons = [sub[i*3:i*3+3] for i in range(len(sub)//3)]
|
|
108
|
+
best_strand = strand
|
|
109
|
+
best_frame = f
|
|
110
|
+
|
|
111
|
+
if best_aln is None or best_score < 0:
|
|
112
|
+
return None, float(best_score), "alignment_failed_or_low_score"
|
|
113
|
+
|
|
114
|
+
# Reverse-thread codons into reference coordinates
|
|
115
|
+
ref_aln = best_aln[0]
|
|
116
|
+
qry_aln = best_aln[1]
|
|
117
|
+
|
|
118
|
+
threaded_codons = []
|
|
119
|
+
q_idx = 0
|
|
120
|
+
n_inserted_codons = 0
|
|
121
|
+
|
|
122
|
+
for col in range(len(ref_aln)):
|
|
123
|
+
r_char = ref_aln[col]
|
|
124
|
+
q_char = qry_aln[col]
|
|
125
|
+
|
|
126
|
+
if r_char != '-':
|
|
127
|
+
if q_char != '-':
|
|
128
|
+
codon = best_codons[q_idx]
|
|
129
|
+
threaded_codons.append(codon)
|
|
130
|
+
q_idx += 1
|
|
131
|
+
else:
|
|
132
|
+
threaded_codons.append('---')
|
|
133
|
+
else:
|
|
134
|
+
# Insertion relative to reference
|
|
135
|
+
if q_char != '-':
|
|
136
|
+
q_idx += 1
|
|
137
|
+
n_inserted_codons += 1
|
|
138
|
+
|
|
139
|
+
aligned_str = ''.join(threaded_codons)
|
|
140
|
+
if len(aligned_str) != self.l_ref_nt:
|
|
141
|
+
return None, float(best_score), f"length_mismatch ({len(aligned_str)} != {self.l_ref_nt})"
|
|
142
|
+
|
|
143
|
+
status = f"ok (strand={best_strand}, frame={best_frame}, score={best_score:.1f})"
|
|
144
|
+
return aligned_str, float(best_score), status
|
|
145
|
+
|
|
146
|
+
def align_batch(
|
|
147
|
+
self,
|
|
148
|
+
seq_dict: Dict[str, str],
|
|
149
|
+
max_workers: int = 4,
|
|
150
|
+
quiet: bool = False,
|
|
151
|
+
) -> Dict[str, Any]:
|
|
152
|
+
"""
|
|
153
|
+
Aligns a batch of sequences in parallel.
|
|
154
|
+
|
|
155
|
+
Returns:
|
|
156
|
+
Dict containing:
|
|
157
|
+
- 'aligned_seqs': {taxon: aligned_sequence}
|
|
158
|
+
- 'failed': {taxon: reason}
|
|
159
|
+
- 'scores': {taxon: score}
|
|
160
|
+
- 'elapsed_seconds': float
|
|
161
|
+
"""
|
|
162
|
+
t0 = time.time()
|
|
163
|
+
n_total = len(seq_dict)
|
|
164
|
+
if not quiet:
|
|
165
|
+
print(f"[*] Reference-guided codon alignment of {n_total:,} sequences against '{self.ref_name}' ({self.l_ref_nt} bp)...", flush=True)
|
|
166
|
+
|
|
167
|
+
aligned_seqs: Dict[str, str] = {}
|
|
168
|
+
failed: Dict[str, str] = {}
|
|
169
|
+
scores: Dict[str, float] = {}
|
|
170
|
+
|
|
171
|
+
taxa = list(seq_dict.keys())
|
|
172
|
+
|
|
173
|
+
def _worker(t: str):
|
|
174
|
+
res_dna, score, status = self.align_single(seq_dict[t])
|
|
175
|
+
return t, res_dna, score, status
|
|
176
|
+
|
|
177
|
+
with ThreadPoolExecutor(max_workers=max_workers) as executor:
|
|
178
|
+
for t, res_dna, score, status in executor.map(_worker, taxa):
|
|
179
|
+
if res_dna is not None:
|
|
180
|
+
aligned_seqs[t] = res_dna
|
|
181
|
+
scores[t] = score
|
|
182
|
+
else:
|
|
183
|
+
failed[t] = status
|
|
184
|
+
|
|
185
|
+
t_elapsed = time.time() - t0
|
|
186
|
+
if not quiet:
|
|
187
|
+
rate = n_total / max(0.01, t_elapsed)
|
|
188
|
+
print(f"[✓] Aligned {len(aligned_seqs):,}/{n_total:,} sequences in {t_elapsed:.2f}s ({rate:.1f} seqs/s). Discarded {len(failed):,}.", flush=True)
|
|
189
|
+
|
|
190
|
+
return {
|
|
191
|
+
"aligned_seqs": aligned_seqs,
|
|
192
|
+
"failed": failed,
|
|
193
|
+
"scores": scores,
|
|
194
|
+
"l_ref_nt": self.l_ref_nt,
|
|
195
|
+
"elapsed_seconds": t_elapsed,
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
# Aliases for convenience
|
|
200
|
+
ReferenceGuidedCodonThreader = ReferenceCodonAligner
|