countess-variant-caller 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,26 @@
1
+ Copyright (C) 2022- CountESS Developers.
2
+
3
+ Redistribution and use in source and binary forms, with or without
4
+ modification, are permitted provided that the following conditions are met:
5
+
6
+ 1. Redistributions of source code must retain the above copyright notice, this
7
+ list of conditions and the following disclaimer.
8
+
9
+ 2. Redistributions in binary form must reproduce the above copyright notice,
10
+ this list of conditions and the following disclaimer in the documentation
11
+ and/or other materials provided with the distribution.
12
+
13
+ 3. Neither the name of the copyright holder nor the names of its contributors
14
+ may be used to endorse or promote products derived from this software without
15
+ specific prior written permission.
16
+
17
+ THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
18
+ ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
19
+ WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
20
+ DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
21
+ FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
22
+ DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
23
+ SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
24
+ CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
25
+ OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
26
+ OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
@@ -0,0 +1,27 @@
1
+ Metadata-Version: 2.4
2
+ Name: countess-variant-caller
3
+ Version: 0.1.0
4
+ Summary: An efficient HGVS variant caller
5
+ Author-email: Nick Moore <nick@zoic.org>
6
+ Maintainer-email: Nick Moore <nick@zoic.org>
7
+ Classifier: Development Status :: 4 - Beta
8
+ Classifier: Intended Audience :: Science/Research
9
+ Classifier: Operating System :: OS Independent
10
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
11
+ Requires-Python: >=3.10
12
+ Description-Content-Type: text/markdown
13
+ License-File: LICENSE.txt
14
+ Requires-Dist: fqfa~=1.3.1
15
+ Requires-Dist: rapidfuzz~=3.14.5
16
+ Provides-Extra: dev
17
+ Requires-Dist: pytest~=8.4.2; extra == "dev"
18
+ Dynamic: license-file
19
+
20
+ # Countess-Variant-Caller
21
+
22
+ This is a variant caller which makes HGVS variant strings from DNA sequences.
23
+ It it part of the [CountESS Project](https://github.com/CountESS-Project/)
24
+ but may be used separately.
25
+
26
+
27
+
@@ -0,0 +1,8 @@
1
+ # Countess-Variant-Caller
2
+
3
+ This is a variant caller which makes HGVS variant strings from DNA sequences.
4
+ It it part of the [CountESS Project](https://github.com/CountESS-Project/)
5
+ but may be used separately.
6
+
7
+
8
+
@@ -0,0 +1,42 @@
1
+ [project]
2
+ name = 'countess-variant-caller'
3
+ dynamic = ["version"]
4
+ readme = "README.md"
5
+ authors = [
6
+ { name = "Nick Moore", email="nick@zoic.org" },
7
+ ]
8
+ maintainers = [
9
+ { name = "Nick Moore", email="nick@zoic.org" },
10
+ ]
11
+ description = "An efficient HGVS variant caller"
12
+ requires-python = ">=3.10"
13
+ classifiers = [
14
+ 'Development Status :: 4 - Beta',
15
+ 'Intended Audience :: Science/Research',
16
+ 'Operating System :: OS Independent',
17
+ 'Topic :: Scientific/Engineering :: Bio-Informatics',
18
+ ]
19
+ dependencies = [
20
+ 'fqfa~=1.3.1',
21
+ 'rapidfuzz~=3.14.5',
22
+ ]
23
+
24
+ [project.optional-dependencies]
25
+ dev = [
26
+ 'pytest~=8.4.2',
27
+ ]
28
+
29
+ [tool.setuptools.dynamic]
30
+ version = { attr = "countess_variant_caller.VERSION" }
31
+ readme = { file = "README.md", content-type="text/markdown" }
32
+
33
+ [tool.pylint]
34
+ max-line-length = 120
35
+
36
+ [tool.black]
37
+ line-length = 120
38
+
39
+ [tool.pytest.ini_options]
40
+ addopts = "--doctest-modules"
41
+ testpaths = "src"
42
+
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,27 @@
1
+ Metadata-Version: 2.4
2
+ Name: countess-variant-caller
3
+ Version: 0.1.0
4
+ Summary: An efficient HGVS variant caller
5
+ Author-email: Nick Moore <nick@zoic.org>
6
+ Maintainer-email: Nick Moore <nick@zoic.org>
7
+ Classifier: Development Status :: 4 - Beta
8
+ Classifier: Intended Audience :: Science/Research
9
+ Classifier: Operating System :: OS Independent
10
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
11
+ Requires-Python: >=3.10
12
+ Description-Content-Type: text/markdown
13
+ License-File: LICENSE.txt
14
+ Requires-Dist: fqfa~=1.3.1
15
+ Requires-Dist: rapidfuzz~=3.14.5
16
+ Provides-Extra: dev
17
+ Requires-Dist: pytest~=8.4.2; extra == "dev"
18
+ Dynamic: license-file
19
+
20
+ # Countess-Variant-Caller
21
+
22
+ This is a variant caller which makes HGVS variant strings from DNA sequences.
23
+ It it part of the [CountESS Project](https://github.com/CountESS-Project/)
24
+ but may be used separately.
25
+
26
+
27
+
@@ -0,0 +1,9 @@
1
+ LICENSE.txt
2
+ README.md
3
+ pyproject.toml
4
+ src/countess_variant_caller.py
5
+ src/countess_variant_caller.egg-info/PKG-INFO
6
+ src/countess_variant_caller.egg-info/SOURCES.txt
7
+ src/countess_variant_caller.egg-info/dependency_links.txt
8
+ src/countess_variant_caller.egg-info/requires.txt
9
+ src/countess_variant_caller.egg-info/top_level.txt
@@ -0,0 +1,5 @@
1
+ fqfa~=1.3.1
2
+ rapidfuzz~=3.14.5
3
+
4
+ [dev]
5
+ pytest~=8.4.2
@@ -0,0 +1,597 @@
1
+ """ countess_variant_caller library
2
+
3
+ Efficiently call HGVS variants from DNA sequences"""
4
+
5
+ import re
6
+ from typing import Iterable, Optional
7
+
8
+ from fqfa.constants.iupac.protein import AA_CODES # type: ignore
9
+ from fqfa.util.nucleotide import reverse_complement # type: ignore
10
+ from fqfa.util.translate import translate_dna # type: ignore
11
+ from rapidfuzz.distance.Levenshtein import opcodes as levenshtein_opcodes
12
+
13
+ VERSION = "0.1.0"
14
+
15
+ # Insertions shorter than this won't be searched for, just included.
16
+ MIN_SEARCH_LENGTH = 10
17
+
18
+
19
+ class TooManyVariationsException(ValueError):
20
+ pass
21
+
22
+
23
+ def translate_aa(aa_seq: str) -> str:
24
+ """translate a sequence of single-letter amino acid codes
25
+ to a sequence of three-letter amindo acid codes
26
+
27
+ >>> translate_aa("HYPERSENSITIVITIES")
28
+ 'HisTyrProGluArgSerGluAsnSerIleThrIleValIleThrIleGluSer'
29
+
30
+ >>> translate_aa("SYZYGY")
31
+ Traceback (most recent call last):
32
+ ...
33
+ ValueError: Invalid AA Sequence
34
+ """
35
+
36
+ try:
37
+ return "".join(AA_CODES[x] for x in aa_seq)
38
+ except KeyError as exc:
39
+ raise ValueError("Invalid AA Sequence") from exc
40
+
41
+
42
+ def search_for_sequence(ref_seq: str, var_seq: str, min_search_length: int = MIN_SEARCH_LENGTH) -> str:
43
+ """look for a copy of `var_seq` within `ref_seq` and if found return a reference
44
+ in the form "offset1_offset2" otherwise return the `var_seq` itself. Don't bother
45
+ searching if the length of var_seq is less than MIN_SEARCH_LENGTH.
46
+
47
+ >>> search_for_sequence("GGGGGGG", "CAT", 1)
48
+ 'CAT'
49
+
50
+ >>> search_for_sequence("GGCATGG", "CAT", 1)
51
+ '3_5'
52
+
53
+ >>> search_for_sequence("GGATGAA", "CAT", 1)
54
+ '3_5inv'
55
+
56
+ >>> search_for_sequence("CATCATC", "CAT", 4)
57
+ 'CAT'
58
+ """
59
+
60
+ # XXX this could be extended to allow for inverted insertions like `850_900inv`,
61
+ # repeated insertions like `A[26]` and/or complex insertion formats such as
62
+ # `T;450_470;AGGG` but actually finding that kind of match is not simple.
63
+
64
+ # XXX consider using something like re.match(r"(.+?)\1+$", var_seq) to search
65
+ # for repeated sequences, but check that it doesn't go O(N!) or whatever
66
+ # (so far on cpython 3.10 this seems fine)
67
+
68
+ if len(var_seq) < min_search_length:
69
+ return var_seq
70
+
71
+ idx = ref_seq.rfind(var_seq)
72
+ if idx >= 0:
73
+ return f"{idx+1}_{idx+len(var_seq)}"
74
+
75
+ inv_seq = reverse_complement(var_seq)
76
+ idx = ref_seq.rfind(inv_seq)
77
+ if idx >= 0:
78
+ return f"{idx+1}_{idx+len(var_seq)}inv"
79
+
80
+ return var_seq
81
+
82
+
83
+ def find_variant_dna(ref_seq: str, var_seq: str, offset: int = 0) -> Iterable[str]:
84
+ """ finds HGVS variants between DNA sequences in ref_seq and var_seq.
85
+ https://varnomen.hgvs.org/recommendations/DNA/variant/insertion/
86
+ Doesn't look for things like complex insertions (yet)
87
+
88
+ https://varnomen.hgvs.org/bg-material/numbering/ is pretty clear
89
+ that numbering of nucleotides starts at `1`.
90
+
91
+ SUBSTITUTION OF SINGLE NUCLEOTIDES (checked against Enrich2)
92
+
93
+ >>> list(find_variant_dna("AGAAGTAGAGG", "TGAAGTAGAGG"))
94
+ ['1A>T']
95
+ >>> list(find_variant_dna("AGAAGTAGAGG", "AGAAGTTGTGG"))
96
+ ['7A>T', '9A>T']
97
+ >>> list(find_variant_dna("AGAAGTAGAGG", "ATAAGAAGAGG"))
98
+ ['2G>T', '6T>A']
99
+
100
+ DUPLICATION OF NUCLEOTIDES
101
+
102
+ (examples from https://varnomen.hgvs.org/recommendations/DNA/variant/duplication/ )
103
+
104
+ >>> list(find_variant_dna(\
105
+ "ATGCTTTGGTGGGAAGAAGTAGAGGACTGTTATGAAAGAGAAGATGTTCAAAAGAA", \
106
+ "ATGCTTTGGTGGGAAGAAGTTAGAGGACTGTTATGAAAGAGAAGATGTTCAAAAGAA"))
107
+ ['20dup']
108
+ >>> list(find_variant_dna(\
109
+ "ATGCTTTGGTGGGAAGAAGTAGAGGACTGTTATGAAAGAGAAGATGTTCAAAAGAA", \
110
+ "ATGCTTTGGTGGGAAGAAGTAGATAGAGGACTGTTATGAAAGAGAAGATGTTCAAAAGAA"))
111
+ ['20_23dup']
112
+
113
+ (checked with Enrich2 ... it comes up with `g.6dupT` for the first
114
+ and g.12dupA for the second, but that trailing symbol is not how it
115
+ is in the HGVS docs. The multi-nucleotide ones it gets very different
116
+ answers for)
117
+
118
+ >>> list(find_variant_dna("AGAAGTAGAGG", "AGAAGTTAGAGG"))
119
+ ['6dup']
120
+ >>> list(find_variant_dna("ATTGAAAAAAAATTAG", "ATTGAAAAAAAAATTAG"))
121
+ ['12dup']
122
+ >>> list(find_variant_dna("AGAAGTAGAGG", "AGAAGTAGATAGAGG"))
123
+ ['6_9dup']
124
+ >>> list(find_variant_dna("AAAACTAAAA", "AAAACTCTAAAA"))
125
+ ['5_6dup']
126
+ >>> list(find_variant_dna("AAAACTGAAAA", "AAAACTGCTGAAAA"))
127
+ ['5_7dup']
128
+ >>> list(find_variant_dna("AAAACTGAAAA", "AAAACTGTGAAAA"))
129
+ ['6_7dup']
130
+
131
+ It should always pick the 3'-most copy to identify as
132
+ a duplicate
133
+
134
+ >>> list(find_variant_dna("ATGGCCGCCAGCCAA", "ATGGCCGCCGCCAGCCAA"))
135
+ ['7_9dup']
136
+
137
+
138
+ INSERTION OF NUCLEOTIDES
139
+
140
+ >>> list(find_variant_dna("AGAAGTAGAGG", "AGAAGTCAGAGG"))
141
+ ['6_7insC']
142
+ >>> list(find_variant_dna("AGAAGTAGAGG", "AGAAGTCATAGAGG"))
143
+ ['6_7insCAT']
144
+
145
+ 0 1
146
+ 12345678901234
147
+ AGAAGTAGAGG
148
+ AGAAGTGAAAGAGG
149
+ ^^^ ^^^
150
+
151
+ GAA appears earlier in the sequence, but it's too short to bother
152
+ including as a reference.
153
+
154
+ >>> list(find_variant_dna("AGAAGTAGAGG", "AGAAGTGAAAGAGG"))
155
+ ['6_7insGAA']
156
+
157
+ the sequence CTTTTTTTTT is long enough to use a reference for.
158
+
159
+ >>> list(find_variant_dna("ACTTTTTTTTTAA", "ACTTTTTTTTTACTTTTTTTTTA"))
160
+ ['12_13ins2_11']
161
+
162
+ the most 3'ward copy of the sequence should be the one which is
163
+ referred to.
164
+
165
+ >>> list(find_variant_dna(\
166
+ "GGCATCATCATCATGGCATCATCATCATGGGG", \
167
+ "GGCATCATCATCATGGCATCATCATCATGGCATCATCATCATGG"))
168
+ ['30_31ins17_28']
169
+
170
+ >>> list(find_variant_dna(\
171
+ "GGCATCATCATCATGGCATCATCATCATGGGGCATCATCATCATGG", \
172
+ "GGCATCATCATCATGGCATCATCATCATGGCATCATCATCATGGCATCATCATCATGG"))
173
+ ['30_31ins33_44']
174
+
175
+ (example from Q&A on https://varnomen.hgvs.org/recommendations/DNA/variant/insertion/ )
176
+ "How should I describe the change ATCGATCGATCGATCGAGGGTCCC to
177
+ ATCGATCGATCGATCGAATCGATCGATCGGGTCCC? The fact that the inserted sequence
178
+ (ATCGATCGATCG) is present in the original sequence suggests it derives
179
+ from a duplicative event."
180
+ "The variant should be described as an insertion; g.17_18ins5_16."
181
+
182
+ 0 1 2 3
183
+ 1234567890123456789012345678901234
184
+ ATCGATCGATCGATCGAGGGTCCC
185
+ ATCGATCGATCGATCGAATCGATCGATCGGGTCCC
186
+ ^^^^^^^^^^^ ^^^^^^^^^^^
187
+
188
+ right so the duplicated part is 5_15 and its inserted between 17 and 18.
189
+ I think the example in the Q&A is wrong.
190
+
191
+ >>> list(find_variant_dna(\
192
+ "ATCGATCGATCGATCGAGGGTCCC", \
193
+ "ATCGATCGATCGATCGAATCGATCGATCGGGTCCC"))
194
+ ['17_18ins5_15']
195
+
196
+ >>> list(find_variant_dna("AAAAAAAAAACCCGGGGGGGGGGTTT", "AAAAAAAAAACCCAAAAAAAAAATTT"))
197
+ ['14_23delins1_10']
198
+
199
+ DELETION OF NUCLEOTIDES (checked against Enrich2)
200
+ enrich2 has g.5_5del for the first one which isn't correct though
201
+
202
+ >>> list(find_variant_dna("AGAAGTAGAGG", "AGAATAGAGG"))
203
+ ['5del']
204
+ >>> list(find_variant_dna("AGAAGTAGAGG", "AGAAAGAGG"))
205
+ ['5_6del']
206
+
207
+ INVERSIONS
208
+
209
+ >>> list(find_variant_dna("AAACCCTTT", "AAAGGGTTT"))
210
+ ['4_6inv']
211
+
212
+ OFFSETS
213
+
214
+ >>> list(find_variant_dna("AGAAGTAGAGG", "ATAAGAAGAGG", 100))
215
+ ['102G>T', '106T>A']
216
+
217
+ >>> list(find_variant_dna("AGAAGTAGAGG", "ATAAGAAGAGG", -200))
218
+ ['198G>T', '194T>A']
219
+ """
220
+
221
+ ref_seq = ref_seq.strip().upper()
222
+ var_seq = var_seq.strip().upper()
223
+
224
+ if not re.match("[AGTCN]+$", ref_seq):
225
+ raise ValueError("Invalid reference sequence")
226
+
227
+ if not re.match("[AGTC]+$", var_seq):
228
+ raise ValueError("Invalid variant sequence")
229
+
230
+ # Levenshtein algorithm finds the overlapping parts of our reference and
231
+ # variant sequences.
232
+ #
233
+ # each element is a text substitution operation on the string of symbols.
234
+ # offsets are python-style whereas HGVS offsets are 1-based and inclusive.
235
+ #
236
+ # 'delete' => delete symbols src_start:src_end
237
+ # 'insert' => insert symbols dest_start:dest_end at src_start
238
+ # 'replace' => replace symbols src_start:src_end with symbols dest_start:dest_end
239
+
240
+ opcodes = list(levenshtein_opcodes(ref_seq, var_seq))
241
+
242
+ # Sometimes, Levenshtein tries a little too hard to
243
+ # find an "equal" operation between inserts, so this
244
+ # looks for that specific case and fixes it.
245
+ #
246
+ # example: find_variant_string("g.", "ATGGTTGGTTCG", "ATGGTTGGTGGTTC")"
247
+ # before: "g.[9_10insGG;10dup;12del]"
248
+ # after: "g.[7_9dup;12del]"
249
+ #
250
+ # this code recognizes that if there's an "equal" sequence
251
+ # followed by an "insert" of the same sequence, then we
252
+ # can swap them, and *if* there's an "insert" before them
253
+ # then we can reduce the complexity of the output by
254
+ # swapping them and merging the two inserts.
255
+ #
256
+ # XXX can do something similar to turn aligned replace/equal/replace
257
+ # sequences into a single whole-codon replace (delins) as per
258
+ # https://varnomen.hgvs.org/recommendations/DNA/variant/substitution/
259
+ # "two variants separated by one nucleotide, together affecting one
260
+ # amino acid, should be described as a “delins”"
261
+ #
262
+ # example: find_variant_string("c.", "ATGTACAAA", "ATGGATAAA")
263
+ # before: "c.[4T>G;6C>T]"
264
+ # after: "c.[4_6delinsGAT]"
265
+
266
+ for n in range(0, len(opcodes) - 2):
267
+ op0, op1, op2 = opcodes[n : n + 3]
268
+ if op0.tag == "insert" and op1.tag == "equal" and op2.tag == "insert":
269
+ seq1 = var_seq[op1.dest_start : op1.dest_end]
270
+ seq2 = var_seq[op2.dest_start : op2.dest_end]
271
+ if seq1 == seq2:
272
+ # extend the first insert and remove the
273
+ # second insert (the following code ignores
274
+ # 'equal's, so it's effectively a NOP)
275
+ op0.dest_end = op1.dest_end
276
+ op2.tag = "equal"
277
+
278
+ for opcode in opcodes:
279
+ src_seq = ref_seq[opcode.src_start : opcode.src_end]
280
+ dest_seq = var_seq[opcode.dest_start : opcode.dest_end]
281
+ start, end = opcode.src_start + offset, opcode.src_end + offset
282
+
283
+ if opcode.tag == "delete":
284
+ assert dest_seq == ""
285
+ # 'delete' opcode maps to HGVS 'del' operation
286
+ if len(src_seq) == 1:
287
+ yield f"{abs(start+1)}del"
288
+ else:
289
+ yield f"{abs(start+1)}_{abs(end)}del"
290
+
291
+ elif opcode.tag == "insert":
292
+ assert src_seq == ""
293
+ # 'insert' opcode maps to either an HGVS 'dup' or 'ins' operation
294
+
295
+ if ref_seq[opcode.src_start - len(dest_seq) : opcode.src_start] == dest_seq:
296
+ # This is a duplication of one or more symbols immediately
297
+ # preceding this point.
298
+ if len(dest_seq) == 1:
299
+ yield f"{abs(start)}dup"
300
+ else:
301
+ yield f"{abs(start - len(dest_seq) + 1)}_{abs(start)}dup"
302
+ else:
303
+ inserted_sequence = search_for_sequence(ref_seq, dest_seq)
304
+ yield f"{abs(start)}_{abs(start+1)}ins{inserted_sequence}"
305
+
306
+ elif opcode.tag == "replace":
307
+ # 'replace' opcode maps to either an HGVS '>' (single substitution) or
308
+ # 'inv' (inversion) or 'delins' (delete+insert) operation.
309
+
310
+ # XXX does not support "exception: two variants separated by one nucleotide,
311
+ # together affecting one amino acid, should be described as a “delins”",
312
+ # as this code has no concept of amino acid alignment.
313
+
314
+ if len(src_seq) == 1 and len(dest_seq) == 1:
315
+ yield f"{abs(start+1)}{src_seq}>{dest_seq}"
316
+ elif len(src_seq) == len(dest_seq) and dest_seq == reverse_complement(src_seq):
317
+ yield f"{abs(start+1)}_{abs(end)}inv"
318
+ else:
319
+ inserted_sequence = search_for_sequence(ref_seq, dest_seq)
320
+ yield f"{abs(start+1)}_{abs(end)}delins{inserted_sequence}"
321
+
322
+
323
+ def find_variant_protein(ref_seq: str, var_seq: str, offset: int = 0):
324
+ """Find changes between two DNA sequences, expressed
325
+ as amino acid changes per HGVS standard.
326
+
327
+ identical sequences:
328
+
329
+ >>> list(find_variant_protein("ATGGTTGGTTCA", "ATGGTTGGTTCA"))
330
+ []
331
+
332
+ a single AA substitution:
333
+
334
+ >>> list(find_variant_protein("ATGGTTGGTTCA", "ATGGTTCCATCA"))
335
+ ['Gly3Pro']
336
+
337
+ a single AA deletion:
338
+
339
+ >>> list(find_variant_protein("ATGGTTGGTTCA", "ATGGGTTCA"))
340
+ ['Val2del']
341
+
342
+ a double AA deletion:
343
+
344
+ >>> list(find_variant_protein("ATGGTTGGTTCAGGC", "ATGTCAGGC"))
345
+ ['Val2_Gly3del']
346
+
347
+ a single AA duplication:
348
+
349
+ >>> list(find_variant_protein("ATGGTTGGTTCA", "ATGGTTGGTGGTTCA"))
350
+ ['Gly3dup']
351
+
352
+ a double AA duplication:
353
+
354
+ >>> list(find_variant_protein("ATGGTTGGTTCA", "ATGGTTGGTGTTGGTTCA"))
355
+ ['Val2_Gly3dup']
356
+
357
+ a single AA insertion
358
+ >>> list(find_variant_protein("ATGGTTGGTTCA", "ATGGTTGGTAAATCA"))
359
+ ['Gly3_Ser4insLys']
360
+
361
+ two substitutions are coded as a delins:
362
+
363
+ >>> list(find_variant_protein("ATGGTTGGTTCA", "ATGGCTGCTTCA"))
364
+ ['Val2_Gly3delinsAlaAla']
365
+
366
+ TODO: this isn't quite correct according to
367
+ https://hgvs-nomenclature.org/stable/recommendations/protein/extension/
368
+
369
+ >>> list(find_variant_protein("ATGGTTGGTTCA", "ATGGTTGGTTCAAAACAG"))
370
+ ['Ser4extLysGln']
371
+
372
+ Protein calling should stop at the first Ter encountered:
373
+
374
+ >>> list(find_variant_protein("ATGGTTGGTTCA", "ATGGTTTAGACA"))
375
+ ['Gly3Ter']
376
+
377
+ >>> list(find_variant_protein("ATGGTTTGGTAG", "ATGGTTTAGTAG"))
378
+ ['Trp3Ter']
379
+
380
+ Call specific synonyms:
381
+
382
+ >>> list(find_variant_protein("ATGGCCTAA", "ATGGCGTAA"))
383
+ ['Ala2=']
384
+
385
+ >>> list(find_variant_protein("ATGGCCAAACCCTAA", "ATGGCGAAGCCATAA"))
386
+ ['Ala2_Pro4=']
387
+
388
+ >>> list(find_variant_protein("ATGGCCAAACCCTAA", "ATGGCGAATCCATAA"))
389
+ ['Ala2=', 'Lys3Asn', 'Pro4=']
390
+
391
+ >>> list(find_variant_protein("ATGGCCCCCAAATAA", "ATGGCGCCAAATTAA"))
392
+ ['Ala2_Pro3=', 'Lys4Asn']
393
+ """
394
+
395
+ ref_seq = ref_seq.strip().upper()
396
+ var_seq = var_seq.strip().upper()
397
+
398
+ if not re.match("[AGTC]+$", ref_seq):
399
+ raise ValueError("Invalid reference sequence") # pragma: no cover
400
+
401
+ if not re.match("[AGTC]+$", var_seq):
402
+ raise ValueError("Invalid variant sequence") # pragma: no cover
403
+
404
+ frame = (3 - offset) % 3
405
+ ref_pro = translate_dna(ref_seq[frame:])[0]
406
+ var_pro = translate_dna(var_seq[frame:])[0]
407
+ offset = (offset + 2) // 3
408
+
409
+ # cut protein translations off at first '*' (terminator)
410
+ if "*" in ref_pro:
411
+ ref_pro = ref_pro[: ref_pro.find("*") + 1]
412
+ if "*" in var_pro:
413
+ var_pro = var_pro[: var_pro.find("*") + 1]
414
+
415
+ def _ref(pos):
416
+ return f"{AA_CODES[ref_pro[pos]]}{pos+1+offset}"
417
+
418
+ opcodes = list(levenshtein_opcodes(ref_pro, var_pro))
419
+
420
+ for opcode in opcodes:
421
+ start, end = opcode.src_start, opcode.src_end
422
+ dest_pro = var_pro[opcode.dest_start : opcode.dest_end]
423
+
424
+ if opcode.tag == "delete":
425
+ assert dest_pro == ""
426
+ if len(ref_pro) > end and ref_pro[end] == "*":
427
+ # if the codon just after this deletion is a terminator,
428
+ # consider this an early termination.
429
+ yield f"{_ref(start)}Ter"
430
+ return
431
+ if end - start == 1:
432
+ yield f"{_ref(start)}del"
433
+ else:
434
+ yield f"{_ref(start)}_{_ref(end-1)}del"
435
+
436
+ elif opcode.tag == "insert":
437
+ assert start == end
438
+
439
+ if ref_pro[start - len(dest_pro) : start] == dest_pro:
440
+ if len(dest_pro) == 1:
441
+ yield f"{_ref(start-1)}dup"
442
+ else:
443
+ yield f"{_ref(start-len(dest_pro))}_{_ref(start-1)}dup"
444
+ elif start == len(ref_pro):
445
+ # 'extension', not quite standards compliant
446
+ yield f"{_ref(start-1)}ext{translate_aa(dest_pro)}"
447
+ else:
448
+ yield f"{_ref(start-1)}_{_ref(end)}ins{translate_aa(dest_pro)}"
449
+
450
+ elif opcode.tag == "replace":
451
+ # XXX handle extension if src_pro[-1] == '*'
452
+
453
+ if end - start == 1 and len(dest_pro) == 1:
454
+ yield f"{_ref(start)}{translate_aa(dest_pro)}"
455
+ else:
456
+ yield f"{_ref(start)}_{_ref(end-1)}delins{translate_aa(dest_pro)}"
457
+
458
+ # If the variant protein terminated, stop translating now:
459
+ if dest_pro[-1] == "*":
460
+ return
461
+
462
+ elif opcode.tag == "equal":
463
+ # Handle calling synonymous changes
464
+ assert end - start == opcode.dest_end - opcode.dest_start
465
+ start_ofs = None
466
+ for ofs in range(0, end - start):
467
+ src_dna = ref_seq[(start + ofs) * 3 + frame :][:3]
468
+ dest_dna = var_seq[(opcode.dest_start + ofs) * 3 + frame :][:3]
469
+ if src_dna == dest_dna:
470
+ if start_ofs is not None:
471
+ if start_ofs == ofs - 1:
472
+ yield f"{_ref(start+start_ofs)}="
473
+ else:
474
+ yield f"{_ref(start+start_ofs)}_{_ref(start+ofs-1)}="
475
+ start_ofs = None
476
+ elif start_ofs is None:
477
+ start_ofs = ofs
478
+ if start_ofs is not None:
479
+ if start_ofs == ofs:
480
+ yield f"{_ref(start+start_ofs)}="
481
+ else:
482
+ yield f"{_ref(start+start_ofs)}_{_ref(start+ofs)}="
483
+
484
+
485
+ def find_variant_string(
486
+ prefix: str,
487
+ ref_seq: str,
488
+ var_seq: str,
489
+ max_mutations: Optional[int] = None,
490
+ offset: int = 0,
491
+ minus_strand: bool = False,
492
+ ) -> str:
493
+ """As above, but returns a single string instead of a generator
494
+
495
+ MULTIPLE VARIATIONS
496
+
497
+ >>> find_variant_string("g.", "GATTACA", "GATTACA")
498
+ 'g.='
499
+ >>> find_variant_string("g.", "GATTACA", "GTTTACA")
500
+ 'g.2A>T'
501
+ >>> find_variant_string("g.", "GATTACA", "GTTTAGA")
502
+ 'g.[2A>T;6C>G]'
503
+ >>> find_variant_string("g.", "GATTACA", "GTTCAGA")
504
+ 'g.[2A>T;4T>C;6C>G]'
505
+
506
+ >>> find_variant_string("g.", "ATGGTTGGTTC", "ATGGTTGGTGGTTC")
507
+ 'g.7_9dup'
508
+ >>> find_variant_string("g.", "ATGGTTGGTTCG", "ATGGTTGGTGGTTC")
509
+ 'g.[7_9dup;12del]'
510
+ >>> find_variant_string("g.", "ATGGTTGGTTC", "ATGGTTGGTGGTTCG")
511
+ 'g.[7_9dup;11_12insG]'
512
+
513
+ PROTEINS
514
+
515
+ >>> find_variant_string("p.", "ATGGTTGGTTCA", "ATGGTTGGTTCA")
516
+ 'p.='
517
+
518
+ >>> find_variant_string("p.", "ATGGTTGGTTCA", "ATGGTTCCATCA")
519
+ 'p.Gly3Pro'
520
+
521
+ >>> find_variant_string("p.", "ATGGTTGGTTCA", "ATGGGTTCA")
522
+ 'p.Val2del'
523
+
524
+ >>> find_variant_string("p.", "ATGGTTGGTTCA", "ATGGTTGGTGGTTCA")
525
+ 'p.Gly3dup'
526
+
527
+ >>> find_variant_string("p.", "ATGGTTGGTTCA", "ATGGCTGCTTCA")
528
+ 'p.Val2_Gly3delinsAlaAla'
529
+
530
+ MINUS STRAND
531
+
532
+ this example is actually comparing TGTAATC and TCTGAAC ...
533
+
534
+ >>> find_variant_string("g.", "GATTACA", "GTTCAGA", minus_strand=True)
535
+ 'g.[2G>C;3_4insG;6del]'
536
+
537
+ CHECK FOR INVALID INPUTS
538
+
539
+ >>> find_variant_string("g.", "HELLO", "CAT")
540
+ Traceback (most recent call last):
541
+ ...
542
+ ValueError: Invalid reference sequence
543
+
544
+ >>> find_variant_string("g.", "CAT", "HELLO")
545
+ Traceback (most recent call last):
546
+ ...
547
+ ValueError: Invalid variant sequence
548
+
549
+ CHECK FOR MAX MUTATIONS
550
+
551
+ >>> find_variant_string("g.", "ATTACC", "GATTACA",1)
552
+ Traceback (most recent call last):
553
+ ...
554
+ countess_variant_caller.TooManyVariationsException: Too many variations (2) in GATTACA
555
+ """
556
+
557
+ if minus_strand:
558
+ ref_seq = reverse_complement(ref_seq)
559
+ var_seq = reverse_complement(var_seq)
560
+
561
+ if prefix.endswith("p."):
562
+ variations = list(find_variant_protein(ref_seq, var_seq, offset))
563
+ else:
564
+ variations = list(find_variant_dna(ref_seq, var_seq, offset))
565
+
566
+ if len(variations) == 0:
567
+ return prefix + "="
568
+
569
+ if max_mutations is not None and len(variations) > max_mutations:
570
+ raise TooManyVariationsException("Too many variations (%d) in %s" % (len(variations), var_seq))
571
+
572
+ if len(variations) == 1:
573
+ return prefix + variations[0]
574
+ else:
575
+ return prefix + "[" + ";".join(variations) + "]"
576
+
577
+
578
+ def classify_protein_variant(pv: str) -> str:
579
+ """Classify an HGVS protein variant as either wild type 'W',
580
+ synonymous 'S', deletion 'D', nonsense 'N', insertion 'I' or
581
+ missense 'M' or if none of the above '?'. Not particularly
582
+ strict as it is just used to classify the output of the
583
+ protein variant caller."""
584
+
585
+ if pv == "p.=":
586
+ return "W"
587
+ if re.match(r"p.\w+\d+=$", pv):
588
+ return "S"
589
+ if re.match(r"p.\w+\d+(_\w+\d+)?del$", pv):
590
+ return "D"
591
+ if re.match(r"p.\w+\d+Ter$", pv):
592
+ return "N"
593
+ if re.match(r"p.\w+\d+(_\w+\d+)?(dup|ins\w+)$", pv):
594
+ return "I"
595
+ if re.match(r"p.\w+\d+\w+$", pv):
596
+ return "M"
597
+ return "?"