rust-annovar 0.1.0.beta.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
data/src/gene.rs ADDED
@@ -0,0 +1,520 @@
1
+ use crate::model::{Annotation, AnnotationKind, Variant, normalize_chrom};
2
+ use anyhow::{Context, Result, bail};
3
+ use std::collections::HashMap;
4
+ use std::io::BufRead;
5
+ use std::path::Path;
6
+
7
+ #[derive(Debug, Clone)]
8
+ pub struct Transcript {
9
+ pub id: String,
10
+ pub gene: String,
11
+ pub chrom: String,
12
+ pub strand: char,
13
+ pub tx_start: u64,
14
+ pub tx_end: u64,
15
+ pub cds_start: u64,
16
+ pub cds_end: u64,
17
+ pub exons: Vec<(u64, u64)>,
18
+ }
19
+
20
+ #[derive(Debug)]
21
+ pub struct GeneDatabase {
22
+ by_chrom: HashMap<String, Vec<Transcript>>,
23
+ sequences: HashMap<String, Vec<u8>>,
24
+ }
25
+
26
+ impl GeneDatabase {
27
+ pub fn load(model_path: &Path, fasta_path: Option<&Path>) -> Result<Self> {
28
+ let mut by_chrom: HashMap<String, Vec<Transcript>> = HashMap::new();
29
+ for (line_no, line) in crate::io::open_reader(model_path)?.lines().enumerate() {
30
+ let line = line?;
31
+ if line.trim().is_empty() || line.starts_with('#') {
32
+ continue;
33
+ }
34
+ let fields: Vec<&str> = line.split('\t').collect();
35
+ let offset = usize::from(fields.first().is_some_and(|v| v.parse::<u64>().is_ok()));
36
+ if fields.len() < offset + 10 {
37
+ bail!(
38
+ "{}:{}: unsupported gene model",
39
+ model_path.display(),
40
+ line_no + 1
41
+ );
42
+ }
43
+ let exon_count: usize = fields[offset + 7].parse()?;
44
+ let starts = parse_positions(fields[offset + 8])?;
45
+ let ends = parse_positions(fields[offset + 9])?;
46
+ if starts.len() != exon_count || ends.len() != exon_count {
47
+ bail!("exon count mismatch at line {}", line_no + 1);
48
+ }
49
+ let transcript = Transcript {
50
+ id: fields[offset].to_string(),
51
+ chrom: normalize_chrom(fields[offset + 1]),
52
+ strand: fields[offset + 2]
53
+ .chars()
54
+ .next()
55
+ .context("missing strand")?,
56
+ tx_start: fields[offset + 3].parse()?,
57
+ tx_end: fields[offset + 4].parse()?,
58
+ cds_start: fields[offset + 5].parse()?,
59
+ cds_end: fields[offset + 6].parse()?,
60
+ exons: starts.into_iter().zip(ends).collect(),
61
+ gene: fields
62
+ .get(offset + 11)
63
+ .unwrap_or(&fields[offset])
64
+ .to_string(),
65
+ };
66
+ by_chrom
67
+ .entry(transcript.chrom.clone())
68
+ .or_default()
69
+ .push(transcript);
70
+ }
71
+ for transcripts in by_chrom.values_mut() {
72
+ transcripts.sort_by_key(|tx| tx.tx_start);
73
+ }
74
+ let sequences = fasta_path.map(read_fasta).transpose()?.unwrap_or_default();
75
+ Ok(Self {
76
+ by_chrom,
77
+ sequences,
78
+ })
79
+ }
80
+
81
+ pub fn annotate(
82
+ &self,
83
+ variant: &Variant,
84
+ protocol: &str,
85
+ splice: u64,
86
+ flank: u64,
87
+ ) -> Annotation {
88
+ let mut hits = Vec::new();
89
+ if let Some(transcripts) = self.by_chrom.get(&variant.chrom) {
90
+ for transcript in transcripts {
91
+ if transcript.tx_start > variant.end.saturating_add(flank) {
92
+ break;
93
+ }
94
+ if transcript.tx_end.saturating_add(flank) < variant.start {
95
+ continue;
96
+ }
97
+ if let Some(hit) = classify(
98
+ variant,
99
+ transcript,
100
+ self.sequences.get(&transcript.id),
101
+ splice,
102
+ flank,
103
+ ) {
104
+ hits.push(hit);
105
+ }
106
+ }
107
+ }
108
+ if hits.is_empty() {
109
+ hits.push(self.nearest_intergenic(variant));
110
+ }
111
+ hits.sort_by_key(|hit| hit.rank);
112
+ let best_rank = hits[0].rank;
113
+ let best: Vec<_> = hits.iter().filter(|hit| hit.rank == best_rank).collect();
114
+ Annotation {
115
+ protocol: protocol.to_string(),
116
+ kind: AnnotationKind::Gene,
117
+ values: vec![
118
+ join_unique(best.iter().map(|hit| hit.function.as_str())),
119
+ join_unique(best.iter().map(|hit| hit.gene.as_str())),
120
+ join_unique(best.iter().map(|hit| hit.detail.as_str())),
121
+ join_unique(best.iter().map(|hit| hit.exonic_function.as_str())),
122
+ join_unique(best.iter().map(|hit| hit.aa_change.as_str())),
123
+ ],
124
+ }
125
+ }
126
+
127
+ fn nearest_intergenic(&self, variant: &Variant) -> GeneHit {
128
+ let Some(transcripts) = self.by_chrom.get(&variant.chrom) else {
129
+ return GeneHit::intergenic();
130
+ };
131
+ let mut left: Option<(&str, u64)> = None;
132
+ let mut right: Option<(&str, u64)> = None;
133
+ for transcript in transcripts {
134
+ if transcript.tx_end <= variant.start {
135
+ let distance =
136
+ variant.start - transcript.tx_end + u64::from(!variant.reference.is_empty());
137
+ if left.is_none_or(|(_, best)| distance < best) {
138
+ left = Some((&transcript.gene, distance));
139
+ }
140
+ } else if transcript.tx_start >= variant.end {
141
+ let distance = transcript.tx_start - variant.end + 1;
142
+ if right.is_none_or(|(_, best)| distance < best) {
143
+ right = Some((&transcript.gene, distance));
144
+ }
145
+ }
146
+ }
147
+ let nearest = [left, right].into_iter().flatten().collect::<Vec<_>>();
148
+ if nearest.is_empty() {
149
+ return GeneHit::intergenic();
150
+ }
151
+ GeneHit {
152
+ rank: 8,
153
+ function: "intergenic".into(),
154
+ gene: nearest
155
+ .iter()
156
+ .map(|(gene, _)| *gene)
157
+ .collect::<Vec<_>>()
158
+ .join(","),
159
+ detail: nearest
160
+ .iter()
161
+ .map(|(_, distance)| format!("dist={distance}"))
162
+ .collect::<Vec<_>>()
163
+ .join(";"),
164
+ exonic_function: ".".into(),
165
+ aa_change: ".".into(),
166
+ }
167
+ }
168
+ }
169
+
170
+ #[derive(Debug)]
171
+ struct GeneHit {
172
+ rank: u8,
173
+ function: String,
174
+ gene: String,
175
+ detail: String,
176
+ exonic_function: String,
177
+ aa_change: String,
178
+ }
179
+
180
+ impl GeneHit {
181
+ fn intergenic() -> Self {
182
+ Self {
183
+ rank: 8,
184
+ function: "intergenic".into(),
185
+ gene: ".".into(),
186
+ detail: ".".into(),
187
+ exonic_function: ".".into(),
188
+ aa_change: ".".into(),
189
+ }
190
+ }
191
+ }
192
+
193
+ fn classify(
194
+ variant: &Variant,
195
+ tx: &Transcript,
196
+ sequence: Option<&Vec<u8>>,
197
+ splice: u64,
198
+ flank: u64,
199
+ ) -> Option<GeneHit> {
200
+ let pos = variant.start;
201
+ if variant.end <= tx.tx_start {
202
+ let function = if tx.strand == '+' {
203
+ "upstream"
204
+ } else {
205
+ "downstream"
206
+ };
207
+ let distance = tx.tx_start - variant.end + 1;
208
+ return (distance <= flank).then(|| flank_hit(function, tx, distance));
209
+ }
210
+ if pos >= tx.tx_end {
211
+ let function = if tx.strand == '+' {
212
+ "downstream"
213
+ } else {
214
+ "upstream"
215
+ };
216
+ let distance = pos - tx.tx_end + u64::from(!variant.reference.is_empty());
217
+ return (distance <= flank).then(|| flank_hit(function, tx, distance));
218
+ }
219
+ let exon_index = tx
220
+ .exons
221
+ .iter()
222
+ .position(|(start, end)| variant.overlaps(*start, *end));
223
+ if exon_index.is_none() {
224
+ let near_boundary = tx
225
+ .exons
226
+ .iter()
227
+ .any(|(start, end)| pos.abs_diff(*start) <= splice || pos.abs_diff(*end) <= splice);
228
+ return Some(basic_hit(
229
+ if near_boundary { 1 } else { 5 },
230
+ if near_boundary {
231
+ "splicing"
232
+ } else {
233
+ "intronic"
234
+ },
235
+ tx,
236
+ ));
237
+ }
238
+ let exon_index = exon_index.unwrap();
239
+ if tx.cds_start == tx.cds_end {
240
+ return Some(basic_hit(2, "ncRNA_exonic", tx));
241
+ }
242
+ if variant.end <= tx.cds_start {
243
+ return Some(basic_hit(
244
+ 3,
245
+ if tx.strand == '+' { "UTR5" } else { "UTR3" },
246
+ tx,
247
+ ));
248
+ }
249
+ if variant.start >= tx.cds_end {
250
+ return Some(basic_hit(
251
+ 3,
252
+ if tx.strand == '+' { "UTR3" } else { "UTR5" },
253
+ tx,
254
+ ));
255
+ }
256
+ let mut hit = basic_hit(0, "exonic", tx);
257
+ hit.detail = ".".into();
258
+ let reference_len = if variant.reference == "0" {
259
+ variant.end.saturating_sub(variant.start) as usize
260
+ } else {
261
+ variant.reference.len()
262
+ };
263
+ let delta = variant.alternate.len() as isize - reference_len as isize;
264
+ if delta != 0 {
265
+ let frame = if delta.unsigned_abs() % 3 == 0 {
266
+ "nonframeshift"
267
+ } else {
268
+ "frameshift"
269
+ };
270
+ let event = if variant.reference.is_empty() {
271
+ "insertion"
272
+ } else if variant.alternate.is_empty() {
273
+ "deletion"
274
+ } else {
275
+ "substitution"
276
+ };
277
+ hit.exonic_function = format!("{frame} {event}");
278
+ hit.aa_change = format!(
279
+ "{}:{}:exon{}:c.?",
280
+ tx.gene,
281
+ tx.id,
282
+ transcript_exon_number(tx, exon_index)
283
+ );
284
+ } else if variant.reference.len() == 1 && variant.alternate.len() == 1 {
285
+ annotate_snv(&mut hit, variant, tx, sequence);
286
+ } else {
287
+ hit.exonic_function = "nonsynonymous block substitution".into();
288
+ }
289
+ Some(hit)
290
+ }
291
+
292
+ fn annotate_snv(hit: &mut GeneHit, variant: &Variant, tx: &Transcript, sequence: Option<&Vec<u8>>) {
293
+ let Some(sequence) = sequence else {
294
+ hit.exonic_function = "unknown".into();
295
+ return;
296
+ };
297
+ let Some(cdna_pos) = genomic_to_cdna(tx, variant.start) else {
298
+ hit.exonic_function = "unknown".into();
299
+ return;
300
+ };
301
+ let Some(cds_start) = genomic_to_cdna(
302
+ tx,
303
+ if tx.strand == '+' {
304
+ tx.cds_start
305
+ } else {
306
+ tx.cds_end - 1
307
+ },
308
+ ) else {
309
+ hit.exonic_function = "unknown".into();
310
+ return;
311
+ };
312
+ let coding_pos = cdna_pos.abs_diff(cds_start);
313
+ let codon_start = cds_start + (coding_pos / 3) * 3;
314
+ if codon_start + 3 > sequence.len() {
315
+ hit.exonic_function = "unknown".into();
316
+ return;
317
+ }
318
+ let old = &sequence[codon_start..codon_start + 3];
319
+ let mut new = old.to_vec();
320
+ let offset = coding_pos % 3;
321
+ let alt = if tx.strand == '+' {
322
+ variant.alternate.as_bytes()[0]
323
+ } else {
324
+ complement(variant.alternate.as_bytes()[0])
325
+ };
326
+ new[offset] = alt;
327
+ let old_aa = translate(old);
328
+ let new_aa = translate(&new);
329
+ hit.exonic_function = match (old_aa, new_aa) {
330
+ (a, b) if a == b => "synonymous SNV",
331
+ (_, b'*') => "stopgain",
332
+ (b'*', _) => "stoploss",
333
+ _ => "nonsynonymous SNV",
334
+ }
335
+ .into();
336
+ let cdna_number = coding_pos + 1;
337
+ let protein_number = coding_pos / 3 + 1;
338
+ let reference = if tx.strand == '+' {
339
+ variant.reference.as_bytes()[0]
340
+ } else {
341
+ complement(variant.reference.as_bytes()[0])
342
+ } as char;
343
+ let alternate = alt as char;
344
+ let exon = tx
345
+ .exons
346
+ .iter()
347
+ .position(|(start, end)| variant.overlaps(*start, *end))
348
+ .map(|index| transcript_exon_number(tx, index))
349
+ .unwrap_or(0);
350
+ hit.aa_change = format!(
351
+ "{}:{}:exon{}:c.{}{}{}:p.{}{}{}",
352
+ tx.gene,
353
+ tx.id,
354
+ exon,
355
+ reference,
356
+ cdna_number,
357
+ alternate,
358
+ aa_name(old_aa),
359
+ protein_number,
360
+ aa_name(new_aa)
361
+ );
362
+ }
363
+
364
+ fn basic_hit(rank: u8, function: &str, tx: &Transcript) -> GeneHit {
365
+ GeneHit {
366
+ rank,
367
+ function: function.into(),
368
+ gene: tx.gene.clone(),
369
+ detail: if matches!(function, "splicing" | "UTR5" | "UTR3") {
370
+ tx.id.clone()
371
+ } else {
372
+ ".".into()
373
+ },
374
+ exonic_function: ".".into(),
375
+ aa_change: ".".into(),
376
+ }
377
+ }
378
+
379
+ fn flank_hit(function: &str, tx: &Transcript, distance: u64) -> GeneHit {
380
+ let mut hit = basic_hit(6, function, tx);
381
+ hit.detail = format!("dist={distance}");
382
+ hit
383
+ }
384
+
385
+ fn genomic_to_cdna(tx: &Transcript, genomic: u64) -> Option<usize> {
386
+ let mut offset = 0usize;
387
+ let iter: Box<dyn Iterator<Item = &(u64, u64)>> = if tx.strand == '+' {
388
+ Box::new(tx.exons.iter())
389
+ } else {
390
+ Box::new(tx.exons.iter().rev())
391
+ };
392
+ for (start, end) in iter {
393
+ if genomic >= *start && genomic < *end {
394
+ return Some(
395
+ offset
396
+ + if tx.strand == '+' {
397
+ (genomic - start) as usize
398
+ } else {
399
+ (end - 1 - genomic) as usize
400
+ },
401
+ );
402
+ }
403
+ offset += (end - start) as usize;
404
+ }
405
+ None
406
+ }
407
+
408
+ fn transcript_exon_number(tx: &Transcript, genomic_index: usize) -> usize {
409
+ if tx.strand == '+' {
410
+ genomic_index + 1
411
+ } else {
412
+ tx.exons.len() - genomic_index
413
+ }
414
+ }
415
+ fn parse_positions(value: &str) -> Result<Vec<u64>> {
416
+ value
417
+ .trim_end_matches(',')
418
+ .split(',')
419
+ .filter(|v| !v.is_empty())
420
+ .map(|v| v.parse().map_err(Into::into))
421
+ .collect()
422
+ }
423
+
424
+ fn read_fasta(path: &Path) -> Result<HashMap<String, Vec<u8>>> {
425
+ let mut result = HashMap::new();
426
+ let mut id = None::<String>;
427
+ let mut sequence = Vec::new();
428
+ for line in crate::io::open_reader(path)?.lines() {
429
+ let line = line?;
430
+ if let Some(header) = line.strip_prefix('>') {
431
+ if let Some(previous) = id.replace(
432
+ header
433
+ .split_whitespace()
434
+ .next()
435
+ .context("empty FASTA header")?
436
+ .to_string(),
437
+ ) {
438
+ result.insert(previous, std::mem::take(&mut sequence));
439
+ }
440
+ } else {
441
+ sequence.extend(line.trim().as_bytes().iter().map(u8::to_ascii_uppercase));
442
+ }
443
+ }
444
+ if let Some(id) = id {
445
+ result.insert(id, sequence);
446
+ }
447
+ Ok(result)
448
+ }
449
+
450
+ fn join_unique<'a>(values: impl Iterator<Item = &'a str>) -> String {
451
+ let mut result: Vec<&str> = Vec::new();
452
+ for value in values {
453
+ if !result.contains(&value) {
454
+ result.push(value);
455
+ }
456
+ }
457
+ result.join(",")
458
+ }
459
+
460
+ fn complement(base: u8) -> u8 {
461
+ match base.to_ascii_uppercase() {
462
+ b'A' => b'T',
463
+ b'T' => b'A',
464
+ b'C' => b'G',
465
+ b'G' => b'C',
466
+ other => other,
467
+ }
468
+ }
469
+ fn translate(codon: &[u8]) -> u8 {
470
+ match codon {
471
+ b"TTT" | b"TTC" => b'F',
472
+ b"TTA" | b"TTG" | b"CTT" | b"CTC" | b"CTA" | b"CTG" => b'L',
473
+ b"ATT" | b"ATC" | b"ATA" => b'I',
474
+ b"ATG" => b'M',
475
+ b"GTT" | b"GTC" | b"GTA" | b"GTG" => b'V',
476
+ b"TCT" | b"TCC" | b"TCA" | b"TCG" | b"AGT" | b"AGC" => b'S',
477
+ b"CCT" | b"CCC" | b"CCA" | b"CCG" => b'P',
478
+ b"ACT" | b"ACC" | b"ACA" | b"ACG" => b'T',
479
+ b"GCT" | b"GCC" | b"GCA" | b"GCG" => b'A',
480
+ b"TAT" | b"TAC" => b'Y',
481
+ b"TAA" | b"TAG" | b"TGA" => b'*',
482
+ b"CAT" | b"CAC" => b'H',
483
+ b"CAA" | b"CAG" => b'Q',
484
+ b"AAT" | b"AAC" => b'N',
485
+ b"AAA" | b"AAG" => b'K',
486
+ b"GAT" | b"GAC" => b'D',
487
+ b"GAA" | b"GAG" => b'E',
488
+ b"TGT" | b"TGC" => b'C',
489
+ b"TGG" => b'W',
490
+ b"CGT" | b"CGC" | b"CGA" | b"CGG" | b"AGA" | b"AGG" => b'R',
491
+ b"GGT" | b"GGC" | b"GGA" | b"GGG" => b'G',
492
+ _ => b'X',
493
+ }
494
+ }
495
+ fn aa_name(aa: u8) -> &'static str {
496
+ match aa {
497
+ b'A' => "A",
498
+ b'R' => "R",
499
+ b'N' => "N",
500
+ b'D' => "D",
501
+ b'C' => "C",
502
+ b'Q' => "Q",
503
+ b'E' => "E",
504
+ b'G' => "G",
505
+ b'H' => "H",
506
+ b'I' => "I",
507
+ b'L' => "L",
508
+ b'K' => "K",
509
+ b'M' => "M",
510
+ b'F' => "F",
511
+ b'P' => "P",
512
+ b'S' => "S",
513
+ b'T' => "T",
514
+ b'W' => "W",
515
+ b'Y' => "Y",
516
+ b'V' => "V",
517
+ b'*' => "X",
518
+ _ => "X",
519
+ }
520
+ }