countess-variant-caller 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- countess_variant_caller-0.1.0/LICENSE.txt +26 -0
- countess_variant_caller-0.1.0/PKG-INFO +27 -0
- countess_variant_caller-0.1.0/README.md +8 -0
- countess_variant_caller-0.1.0/pyproject.toml +42 -0
- countess_variant_caller-0.1.0/setup.cfg +4 -0
- countess_variant_caller-0.1.0/src/countess_variant_caller.egg-info/PKG-INFO +27 -0
- countess_variant_caller-0.1.0/src/countess_variant_caller.egg-info/SOURCES.txt +9 -0
- countess_variant_caller-0.1.0/src/countess_variant_caller.egg-info/dependency_links.txt +1 -0
- countess_variant_caller-0.1.0/src/countess_variant_caller.egg-info/requires.txt +5 -0
- countess_variant_caller-0.1.0/src/countess_variant_caller.egg-info/top_level.txt +1 -0
- countess_variant_caller-0.1.0/src/countess_variant_caller.py +597 -0
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
Copyright (C) 2022- CountESS Developers.
|
|
2
|
+
|
|
3
|
+
Redistribution and use in source and binary forms, with or without
|
|
4
|
+
modification, are permitted provided that the following conditions are met:
|
|
5
|
+
|
|
6
|
+
1. Redistributions of source code must retain the above copyright notice, this
|
|
7
|
+
list of conditions and the following disclaimer.
|
|
8
|
+
|
|
9
|
+
2. Redistributions in binary form must reproduce the above copyright notice,
|
|
10
|
+
this list of conditions and the following disclaimer in the documentation
|
|
11
|
+
and/or other materials provided with the distribution.
|
|
12
|
+
|
|
13
|
+
3. Neither the name of the copyright holder nor the names of its contributors
|
|
14
|
+
may be used to endorse or promote products derived from this software without
|
|
15
|
+
specific prior written permission.
|
|
16
|
+
|
|
17
|
+
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
|
|
18
|
+
ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
|
|
19
|
+
WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
|
20
|
+
DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
|
21
|
+
FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
|
22
|
+
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
|
23
|
+
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
|
24
|
+
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
|
25
|
+
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
|
26
|
+
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: countess-variant-caller
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: An efficient HGVS variant caller
|
|
5
|
+
Author-email: Nick Moore <nick@zoic.org>
|
|
6
|
+
Maintainer-email: Nick Moore <nick@zoic.org>
|
|
7
|
+
Classifier: Development Status :: 4 - Beta
|
|
8
|
+
Classifier: Intended Audience :: Science/Research
|
|
9
|
+
Classifier: Operating System :: OS Independent
|
|
10
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
11
|
+
Requires-Python: >=3.10
|
|
12
|
+
Description-Content-Type: text/markdown
|
|
13
|
+
License-File: LICENSE.txt
|
|
14
|
+
Requires-Dist: fqfa~=1.3.1
|
|
15
|
+
Requires-Dist: rapidfuzz~=3.14.5
|
|
16
|
+
Provides-Extra: dev
|
|
17
|
+
Requires-Dist: pytest~=8.4.2; extra == "dev"
|
|
18
|
+
Dynamic: license-file
|
|
19
|
+
|
|
20
|
+
# Countess-Variant-Caller
|
|
21
|
+
|
|
22
|
+
This is a variant caller which makes HGVS variant strings from DNA sequences.
|
|
23
|
+
It it part of the [CountESS Project](https://github.com/CountESS-Project/)
|
|
24
|
+
but may be used separately.
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = 'countess-variant-caller'
|
|
3
|
+
dynamic = ["version"]
|
|
4
|
+
readme = "README.md"
|
|
5
|
+
authors = [
|
|
6
|
+
{ name = "Nick Moore", email="nick@zoic.org" },
|
|
7
|
+
]
|
|
8
|
+
maintainers = [
|
|
9
|
+
{ name = "Nick Moore", email="nick@zoic.org" },
|
|
10
|
+
]
|
|
11
|
+
description = "An efficient HGVS variant caller"
|
|
12
|
+
requires-python = ">=3.10"
|
|
13
|
+
classifiers = [
|
|
14
|
+
'Development Status :: 4 - Beta',
|
|
15
|
+
'Intended Audience :: Science/Research',
|
|
16
|
+
'Operating System :: OS Independent',
|
|
17
|
+
'Topic :: Scientific/Engineering :: Bio-Informatics',
|
|
18
|
+
]
|
|
19
|
+
dependencies = [
|
|
20
|
+
'fqfa~=1.3.1',
|
|
21
|
+
'rapidfuzz~=3.14.5',
|
|
22
|
+
]
|
|
23
|
+
|
|
24
|
+
[project.optional-dependencies]
|
|
25
|
+
dev = [
|
|
26
|
+
'pytest~=8.4.2',
|
|
27
|
+
]
|
|
28
|
+
|
|
29
|
+
[tool.setuptools.dynamic]
|
|
30
|
+
version = { attr = "countess_variant_caller.VERSION" }
|
|
31
|
+
readme = { file = "README.md", content-type="text/markdown" }
|
|
32
|
+
|
|
33
|
+
[tool.pylint]
|
|
34
|
+
max-line-length = 120
|
|
35
|
+
|
|
36
|
+
[tool.black]
|
|
37
|
+
line-length = 120
|
|
38
|
+
|
|
39
|
+
[tool.pytest.ini_options]
|
|
40
|
+
addopts = "--doctest-modules"
|
|
41
|
+
testpaths = "src"
|
|
42
|
+
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: countess-variant-caller
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: An efficient HGVS variant caller
|
|
5
|
+
Author-email: Nick Moore <nick@zoic.org>
|
|
6
|
+
Maintainer-email: Nick Moore <nick@zoic.org>
|
|
7
|
+
Classifier: Development Status :: 4 - Beta
|
|
8
|
+
Classifier: Intended Audience :: Science/Research
|
|
9
|
+
Classifier: Operating System :: OS Independent
|
|
10
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
11
|
+
Requires-Python: >=3.10
|
|
12
|
+
Description-Content-Type: text/markdown
|
|
13
|
+
License-File: LICENSE.txt
|
|
14
|
+
Requires-Dist: fqfa~=1.3.1
|
|
15
|
+
Requires-Dist: rapidfuzz~=3.14.5
|
|
16
|
+
Provides-Extra: dev
|
|
17
|
+
Requires-Dist: pytest~=8.4.2; extra == "dev"
|
|
18
|
+
Dynamic: license-file
|
|
19
|
+
|
|
20
|
+
# Countess-Variant-Caller
|
|
21
|
+
|
|
22
|
+
This is a variant caller which makes HGVS variant strings from DNA sequences.
|
|
23
|
+
It it part of the [CountESS Project](https://github.com/CountESS-Project/)
|
|
24
|
+
but may be used separately.
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
LICENSE.txt
|
|
2
|
+
README.md
|
|
3
|
+
pyproject.toml
|
|
4
|
+
src/countess_variant_caller.py
|
|
5
|
+
src/countess_variant_caller.egg-info/PKG-INFO
|
|
6
|
+
src/countess_variant_caller.egg-info/SOURCES.txt
|
|
7
|
+
src/countess_variant_caller.egg-info/dependency_links.txt
|
|
8
|
+
src/countess_variant_caller.egg-info/requires.txt
|
|
9
|
+
src/countess_variant_caller.egg-info/top_level.txt
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
countess_variant_caller
|
|
@@ -0,0 +1,597 @@
|
|
|
1
|
+
""" countess_variant_caller library
|
|
2
|
+
|
|
3
|
+
Efficiently call HGVS variants from DNA sequences"""
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from typing import Iterable, Optional
|
|
7
|
+
|
|
8
|
+
from fqfa.constants.iupac.protein import AA_CODES # type: ignore
|
|
9
|
+
from fqfa.util.nucleotide import reverse_complement # type: ignore
|
|
10
|
+
from fqfa.util.translate import translate_dna # type: ignore
|
|
11
|
+
from rapidfuzz.distance.Levenshtein import opcodes as levenshtein_opcodes
|
|
12
|
+
|
|
13
|
+
VERSION = "0.1.0"
|
|
14
|
+
|
|
15
|
+
# Insertions shorter than this won't be searched for, just included.
|
|
16
|
+
MIN_SEARCH_LENGTH = 10
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class TooManyVariationsException(ValueError):
|
|
20
|
+
pass
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def translate_aa(aa_seq: str) -> str:
|
|
24
|
+
"""translate a sequence of single-letter amino acid codes
|
|
25
|
+
to a sequence of three-letter amindo acid codes
|
|
26
|
+
|
|
27
|
+
>>> translate_aa("HYPERSENSITIVITIES")
|
|
28
|
+
'HisTyrProGluArgSerGluAsnSerIleThrIleValIleThrIleGluSer'
|
|
29
|
+
|
|
30
|
+
>>> translate_aa("SYZYGY")
|
|
31
|
+
Traceback (most recent call last):
|
|
32
|
+
...
|
|
33
|
+
ValueError: Invalid AA Sequence
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
try:
|
|
37
|
+
return "".join(AA_CODES[x] for x in aa_seq)
|
|
38
|
+
except KeyError as exc:
|
|
39
|
+
raise ValueError("Invalid AA Sequence") from exc
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def search_for_sequence(ref_seq: str, var_seq: str, min_search_length: int = MIN_SEARCH_LENGTH) -> str:
|
|
43
|
+
"""look for a copy of `var_seq` within `ref_seq` and if found return a reference
|
|
44
|
+
in the form "offset1_offset2" otherwise return the `var_seq` itself. Don't bother
|
|
45
|
+
searching if the length of var_seq is less than MIN_SEARCH_LENGTH.
|
|
46
|
+
|
|
47
|
+
>>> search_for_sequence("GGGGGGG", "CAT", 1)
|
|
48
|
+
'CAT'
|
|
49
|
+
|
|
50
|
+
>>> search_for_sequence("GGCATGG", "CAT", 1)
|
|
51
|
+
'3_5'
|
|
52
|
+
|
|
53
|
+
>>> search_for_sequence("GGATGAA", "CAT", 1)
|
|
54
|
+
'3_5inv'
|
|
55
|
+
|
|
56
|
+
>>> search_for_sequence("CATCATC", "CAT", 4)
|
|
57
|
+
'CAT'
|
|
58
|
+
"""
|
|
59
|
+
|
|
60
|
+
# XXX this could be extended to allow for inverted insertions like `850_900inv`,
|
|
61
|
+
# repeated insertions like `A[26]` and/or complex insertion formats such as
|
|
62
|
+
# `T;450_470;AGGG` but actually finding that kind of match is not simple.
|
|
63
|
+
|
|
64
|
+
# XXX consider using something like re.match(r"(.+?)\1+$", var_seq) to search
|
|
65
|
+
# for repeated sequences, but check that it doesn't go O(N!) or whatever
|
|
66
|
+
# (so far on cpython 3.10 this seems fine)
|
|
67
|
+
|
|
68
|
+
if len(var_seq) < min_search_length:
|
|
69
|
+
return var_seq
|
|
70
|
+
|
|
71
|
+
idx = ref_seq.rfind(var_seq)
|
|
72
|
+
if idx >= 0:
|
|
73
|
+
return f"{idx+1}_{idx+len(var_seq)}"
|
|
74
|
+
|
|
75
|
+
inv_seq = reverse_complement(var_seq)
|
|
76
|
+
idx = ref_seq.rfind(inv_seq)
|
|
77
|
+
if idx >= 0:
|
|
78
|
+
return f"{idx+1}_{idx+len(var_seq)}inv"
|
|
79
|
+
|
|
80
|
+
return var_seq
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def find_variant_dna(ref_seq: str, var_seq: str, offset: int = 0) -> Iterable[str]:
|
|
84
|
+
""" finds HGVS variants between DNA sequences in ref_seq and var_seq.
|
|
85
|
+
https://varnomen.hgvs.org/recommendations/DNA/variant/insertion/
|
|
86
|
+
Doesn't look for things like complex insertions (yet)
|
|
87
|
+
|
|
88
|
+
https://varnomen.hgvs.org/bg-material/numbering/ is pretty clear
|
|
89
|
+
that numbering of nucleotides starts at `1`.
|
|
90
|
+
|
|
91
|
+
SUBSTITUTION OF SINGLE NUCLEOTIDES (checked against Enrich2)
|
|
92
|
+
|
|
93
|
+
>>> list(find_variant_dna("AGAAGTAGAGG", "TGAAGTAGAGG"))
|
|
94
|
+
['1A>T']
|
|
95
|
+
>>> list(find_variant_dna("AGAAGTAGAGG", "AGAAGTTGTGG"))
|
|
96
|
+
['7A>T', '9A>T']
|
|
97
|
+
>>> list(find_variant_dna("AGAAGTAGAGG", "ATAAGAAGAGG"))
|
|
98
|
+
['2G>T', '6T>A']
|
|
99
|
+
|
|
100
|
+
DUPLICATION OF NUCLEOTIDES
|
|
101
|
+
|
|
102
|
+
(examples from https://varnomen.hgvs.org/recommendations/DNA/variant/duplication/ )
|
|
103
|
+
|
|
104
|
+
>>> list(find_variant_dna(\
|
|
105
|
+
"ATGCTTTGGTGGGAAGAAGTAGAGGACTGTTATGAAAGAGAAGATGTTCAAAAGAA", \
|
|
106
|
+
"ATGCTTTGGTGGGAAGAAGTTAGAGGACTGTTATGAAAGAGAAGATGTTCAAAAGAA"))
|
|
107
|
+
['20dup']
|
|
108
|
+
>>> list(find_variant_dna(\
|
|
109
|
+
"ATGCTTTGGTGGGAAGAAGTAGAGGACTGTTATGAAAGAGAAGATGTTCAAAAGAA", \
|
|
110
|
+
"ATGCTTTGGTGGGAAGAAGTAGATAGAGGACTGTTATGAAAGAGAAGATGTTCAAAAGAA"))
|
|
111
|
+
['20_23dup']
|
|
112
|
+
|
|
113
|
+
(checked with Enrich2 ... it comes up with `g.6dupT` for the first
|
|
114
|
+
and g.12dupA for the second, but that trailing symbol is not how it
|
|
115
|
+
is in the HGVS docs. The multi-nucleotide ones it gets very different
|
|
116
|
+
answers for)
|
|
117
|
+
|
|
118
|
+
>>> list(find_variant_dna("AGAAGTAGAGG", "AGAAGTTAGAGG"))
|
|
119
|
+
['6dup']
|
|
120
|
+
>>> list(find_variant_dna("ATTGAAAAAAAATTAG", "ATTGAAAAAAAAATTAG"))
|
|
121
|
+
['12dup']
|
|
122
|
+
>>> list(find_variant_dna("AGAAGTAGAGG", "AGAAGTAGATAGAGG"))
|
|
123
|
+
['6_9dup']
|
|
124
|
+
>>> list(find_variant_dna("AAAACTAAAA", "AAAACTCTAAAA"))
|
|
125
|
+
['5_6dup']
|
|
126
|
+
>>> list(find_variant_dna("AAAACTGAAAA", "AAAACTGCTGAAAA"))
|
|
127
|
+
['5_7dup']
|
|
128
|
+
>>> list(find_variant_dna("AAAACTGAAAA", "AAAACTGTGAAAA"))
|
|
129
|
+
['6_7dup']
|
|
130
|
+
|
|
131
|
+
It should always pick the 3'-most copy to identify as
|
|
132
|
+
a duplicate
|
|
133
|
+
|
|
134
|
+
>>> list(find_variant_dna("ATGGCCGCCAGCCAA", "ATGGCCGCCGCCAGCCAA"))
|
|
135
|
+
['7_9dup']
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
INSERTION OF NUCLEOTIDES
|
|
139
|
+
|
|
140
|
+
>>> list(find_variant_dna("AGAAGTAGAGG", "AGAAGTCAGAGG"))
|
|
141
|
+
['6_7insC']
|
|
142
|
+
>>> list(find_variant_dna("AGAAGTAGAGG", "AGAAGTCATAGAGG"))
|
|
143
|
+
['6_7insCAT']
|
|
144
|
+
|
|
145
|
+
0 1
|
|
146
|
+
12345678901234
|
|
147
|
+
AGAAGTAGAGG
|
|
148
|
+
AGAAGTGAAAGAGG
|
|
149
|
+
^^^ ^^^
|
|
150
|
+
|
|
151
|
+
GAA appears earlier in the sequence, but it's too short to bother
|
|
152
|
+
including as a reference.
|
|
153
|
+
|
|
154
|
+
>>> list(find_variant_dna("AGAAGTAGAGG", "AGAAGTGAAAGAGG"))
|
|
155
|
+
['6_7insGAA']
|
|
156
|
+
|
|
157
|
+
the sequence CTTTTTTTTT is long enough to use a reference for.
|
|
158
|
+
|
|
159
|
+
>>> list(find_variant_dna("ACTTTTTTTTTAA", "ACTTTTTTTTTACTTTTTTTTTA"))
|
|
160
|
+
['12_13ins2_11']
|
|
161
|
+
|
|
162
|
+
the most 3'ward copy of the sequence should be the one which is
|
|
163
|
+
referred to.
|
|
164
|
+
|
|
165
|
+
>>> list(find_variant_dna(\
|
|
166
|
+
"GGCATCATCATCATGGCATCATCATCATGGGG", \
|
|
167
|
+
"GGCATCATCATCATGGCATCATCATCATGGCATCATCATCATGG"))
|
|
168
|
+
['30_31ins17_28']
|
|
169
|
+
|
|
170
|
+
>>> list(find_variant_dna(\
|
|
171
|
+
"GGCATCATCATCATGGCATCATCATCATGGGGCATCATCATCATGG", \
|
|
172
|
+
"GGCATCATCATCATGGCATCATCATCATGGCATCATCATCATGGCATCATCATCATGG"))
|
|
173
|
+
['30_31ins33_44']
|
|
174
|
+
|
|
175
|
+
(example from Q&A on https://varnomen.hgvs.org/recommendations/DNA/variant/insertion/ )
|
|
176
|
+
"How should I describe the change ATCGATCGATCGATCGAGGGTCCC to
|
|
177
|
+
ATCGATCGATCGATCGAATCGATCGATCGGGTCCC? The fact that the inserted sequence
|
|
178
|
+
(ATCGATCGATCG) is present in the original sequence suggests it derives
|
|
179
|
+
from a duplicative event."
|
|
180
|
+
"The variant should be described as an insertion; g.17_18ins5_16."
|
|
181
|
+
|
|
182
|
+
0 1 2 3
|
|
183
|
+
1234567890123456789012345678901234
|
|
184
|
+
ATCGATCGATCGATCGAGGGTCCC
|
|
185
|
+
ATCGATCGATCGATCGAATCGATCGATCGGGTCCC
|
|
186
|
+
^^^^^^^^^^^ ^^^^^^^^^^^
|
|
187
|
+
|
|
188
|
+
right so the duplicated part is 5_15 and its inserted between 17 and 18.
|
|
189
|
+
I think the example in the Q&A is wrong.
|
|
190
|
+
|
|
191
|
+
>>> list(find_variant_dna(\
|
|
192
|
+
"ATCGATCGATCGATCGAGGGTCCC", \
|
|
193
|
+
"ATCGATCGATCGATCGAATCGATCGATCGGGTCCC"))
|
|
194
|
+
['17_18ins5_15']
|
|
195
|
+
|
|
196
|
+
>>> list(find_variant_dna("AAAAAAAAAACCCGGGGGGGGGGTTT", "AAAAAAAAAACCCAAAAAAAAAATTT"))
|
|
197
|
+
['14_23delins1_10']
|
|
198
|
+
|
|
199
|
+
DELETION OF NUCLEOTIDES (checked against Enrich2)
|
|
200
|
+
enrich2 has g.5_5del for the first one which isn't correct though
|
|
201
|
+
|
|
202
|
+
>>> list(find_variant_dna("AGAAGTAGAGG", "AGAATAGAGG"))
|
|
203
|
+
['5del']
|
|
204
|
+
>>> list(find_variant_dna("AGAAGTAGAGG", "AGAAAGAGG"))
|
|
205
|
+
['5_6del']
|
|
206
|
+
|
|
207
|
+
INVERSIONS
|
|
208
|
+
|
|
209
|
+
>>> list(find_variant_dna("AAACCCTTT", "AAAGGGTTT"))
|
|
210
|
+
['4_6inv']
|
|
211
|
+
|
|
212
|
+
OFFSETS
|
|
213
|
+
|
|
214
|
+
>>> list(find_variant_dna("AGAAGTAGAGG", "ATAAGAAGAGG", 100))
|
|
215
|
+
['102G>T', '106T>A']
|
|
216
|
+
|
|
217
|
+
>>> list(find_variant_dna("AGAAGTAGAGG", "ATAAGAAGAGG", -200))
|
|
218
|
+
['198G>T', '194T>A']
|
|
219
|
+
"""
|
|
220
|
+
|
|
221
|
+
ref_seq = ref_seq.strip().upper()
|
|
222
|
+
var_seq = var_seq.strip().upper()
|
|
223
|
+
|
|
224
|
+
if not re.match("[AGTCN]+$", ref_seq):
|
|
225
|
+
raise ValueError("Invalid reference sequence")
|
|
226
|
+
|
|
227
|
+
if not re.match("[AGTC]+$", var_seq):
|
|
228
|
+
raise ValueError("Invalid variant sequence")
|
|
229
|
+
|
|
230
|
+
# Levenshtein algorithm finds the overlapping parts of our reference and
|
|
231
|
+
# variant sequences.
|
|
232
|
+
#
|
|
233
|
+
# each element is a text substitution operation on the string of symbols.
|
|
234
|
+
# offsets are python-style whereas HGVS offsets are 1-based and inclusive.
|
|
235
|
+
#
|
|
236
|
+
# 'delete' => delete symbols src_start:src_end
|
|
237
|
+
# 'insert' => insert symbols dest_start:dest_end at src_start
|
|
238
|
+
# 'replace' => replace symbols src_start:src_end with symbols dest_start:dest_end
|
|
239
|
+
|
|
240
|
+
opcodes = list(levenshtein_opcodes(ref_seq, var_seq))
|
|
241
|
+
|
|
242
|
+
# Sometimes, Levenshtein tries a little too hard to
|
|
243
|
+
# find an "equal" operation between inserts, so this
|
|
244
|
+
# looks for that specific case and fixes it.
|
|
245
|
+
#
|
|
246
|
+
# example: find_variant_string("g.", "ATGGTTGGTTCG", "ATGGTTGGTGGTTC")"
|
|
247
|
+
# before: "g.[9_10insGG;10dup;12del]"
|
|
248
|
+
# after: "g.[7_9dup;12del]"
|
|
249
|
+
#
|
|
250
|
+
# this code recognizes that if there's an "equal" sequence
|
|
251
|
+
# followed by an "insert" of the same sequence, then we
|
|
252
|
+
# can swap them, and *if* there's an "insert" before them
|
|
253
|
+
# then we can reduce the complexity of the output by
|
|
254
|
+
# swapping them and merging the two inserts.
|
|
255
|
+
#
|
|
256
|
+
# XXX can do something similar to turn aligned replace/equal/replace
|
|
257
|
+
# sequences into a single whole-codon replace (delins) as per
|
|
258
|
+
# https://varnomen.hgvs.org/recommendations/DNA/variant/substitution/
|
|
259
|
+
# "two variants separated by one nucleotide, together affecting one
|
|
260
|
+
# amino acid, should be described as a “delins”"
|
|
261
|
+
#
|
|
262
|
+
# example: find_variant_string("c.", "ATGTACAAA", "ATGGATAAA")
|
|
263
|
+
# before: "c.[4T>G;6C>T]"
|
|
264
|
+
# after: "c.[4_6delinsGAT]"
|
|
265
|
+
|
|
266
|
+
for n in range(0, len(opcodes) - 2):
|
|
267
|
+
op0, op1, op2 = opcodes[n : n + 3]
|
|
268
|
+
if op0.tag == "insert" and op1.tag == "equal" and op2.tag == "insert":
|
|
269
|
+
seq1 = var_seq[op1.dest_start : op1.dest_end]
|
|
270
|
+
seq2 = var_seq[op2.dest_start : op2.dest_end]
|
|
271
|
+
if seq1 == seq2:
|
|
272
|
+
# extend the first insert and remove the
|
|
273
|
+
# second insert (the following code ignores
|
|
274
|
+
# 'equal's, so it's effectively a NOP)
|
|
275
|
+
op0.dest_end = op1.dest_end
|
|
276
|
+
op2.tag = "equal"
|
|
277
|
+
|
|
278
|
+
for opcode in opcodes:
|
|
279
|
+
src_seq = ref_seq[opcode.src_start : opcode.src_end]
|
|
280
|
+
dest_seq = var_seq[opcode.dest_start : opcode.dest_end]
|
|
281
|
+
start, end = opcode.src_start + offset, opcode.src_end + offset
|
|
282
|
+
|
|
283
|
+
if opcode.tag == "delete":
|
|
284
|
+
assert dest_seq == ""
|
|
285
|
+
# 'delete' opcode maps to HGVS 'del' operation
|
|
286
|
+
if len(src_seq) == 1:
|
|
287
|
+
yield f"{abs(start+1)}del"
|
|
288
|
+
else:
|
|
289
|
+
yield f"{abs(start+1)}_{abs(end)}del"
|
|
290
|
+
|
|
291
|
+
elif opcode.tag == "insert":
|
|
292
|
+
assert src_seq == ""
|
|
293
|
+
# 'insert' opcode maps to either an HGVS 'dup' or 'ins' operation
|
|
294
|
+
|
|
295
|
+
if ref_seq[opcode.src_start - len(dest_seq) : opcode.src_start] == dest_seq:
|
|
296
|
+
# This is a duplication of one or more symbols immediately
|
|
297
|
+
# preceding this point.
|
|
298
|
+
if len(dest_seq) == 1:
|
|
299
|
+
yield f"{abs(start)}dup"
|
|
300
|
+
else:
|
|
301
|
+
yield f"{abs(start - len(dest_seq) + 1)}_{abs(start)}dup"
|
|
302
|
+
else:
|
|
303
|
+
inserted_sequence = search_for_sequence(ref_seq, dest_seq)
|
|
304
|
+
yield f"{abs(start)}_{abs(start+1)}ins{inserted_sequence}"
|
|
305
|
+
|
|
306
|
+
elif opcode.tag == "replace":
|
|
307
|
+
# 'replace' opcode maps to either an HGVS '>' (single substitution) or
|
|
308
|
+
# 'inv' (inversion) or 'delins' (delete+insert) operation.
|
|
309
|
+
|
|
310
|
+
# XXX does not support "exception: two variants separated by one nucleotide,
|
|
311
|
+
# together affecting one amino acid, should be described as a “delins”",
|
|
312
|
+
# as this code has no concept of amino acid alignment.
|
|
313
|
+
|
|
314
|
+
if len(src_seq) == 1 and len(dest_seq) == 1:
|
|
315
|
+
yield f"{abs(start+1)}{src_seq}>{dest_seq}"
|
|
316
|
+
elif len(src_seq) == len(dest_seq) and dest_seq == reverse_complement(src_seq):
|
|
317
|
+
yield f"{abs(start+1)}_{abs(end)}inv"
|
|
318
|
+
else:
|
|
319
|
+
inserted_sequence = search_for_sequence(ref_seq, dest_seq)
|
|
320
|
+
yield f"{abs(start+1)}_{abs(end)}delins{inserted_sequence}"
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
def find_variant_protein(ref_seq: str, var_seq: str, offset: int = 0):
|
|
324
|
+
"""Find changes between two DNA sequences, expressed
|
|
325
|
+
as amino acid changes per HGVS standard.
|
|
326
|
+
|
|
327
|
+
identical sequences:
|
|
328
|
+
|
|
329
|
+
>>> list(find_variant_protein("ATGGTTGGTTCA", "ATGGTTGGTTCA"))
|
|
330
|
+
[]
|
|
331
|
+
|
|
332
|
+
a single AA substitution:
|
|
333
|
+
|
|
334
|
+
>>> list(find_variant_protein("ATGGTTGGTTCA", "ATGGTTCCATCA"))
|
|
335
|
+
['Gly3Pro']
|
|
336
|
+
|
|
337
|
+
a single AA deletion:
|
|
338
|
+
|
|
339
|
+
>>> list(find_variant_protein("ATGGTTGGTTCA", "ATGGGTTCA"))
|
|
340
|
+
['Val2del']
|
|
341
|
+
|
|
342
|
+
a double AA deletion:
|
|
343
|
+
|
|
344
|
+
>>> list(find_variant_protein("ATGGTTGGTTCAGGC", "ATGTCAGGC"))
|
|
345
|
+
['Val2_Gly3del']
|
|
346
|
+
|
|
347
|
+
a single AA duplication:
|
|
348
|
+
|
|
349
|
+
>>> list(find_variant_protein("ATGGTTGGTTCA", "ATGGTTGGTGGTTCA"))
|
|
350
|
+
['Gly3dup']
|
|
351
|
+
|
|
352
|
+
a double AA duplication:
|
|
353
|
+
|
|
354
|
+
>>> list(find_variant_protein("ATGGTTGGTTCA", "ATGGTTGGTGTTGGTTCA"))
|
|
355
|
+
['Val2_Gly3dup']
|
|
356
|
+
|
|
357
|
+
a single AA insertion
|
|
358
|
+
>>> list(find_variant_protein("ATGGTTGGTTCA", "ATGGTTGGTAAATCA"))
|
|
359
|
+
['Gly3_Ser4insLys']
|
|
360
|
+
|
|
361
|
+
two substitutions are coded as a delins:
|
|
362
|
+
|
|
363
|
+
>>> list(find_variant_protein("ATGGTTGGTTCA", "ATGGCTGCTTCA"))
|
|
364
|
+
['Val2_Gly3delinsAlaAla']
|
|
365
|
+
|
|
366
|
+
TODO: this isn't quite correct according to
|
|
367
|
+
https://hgvs-nomenclature.org/stable/recommendations/protein/extension/
|
|
368
|
+
|
|
369
|
+
>>> list(find_variant_protein("ATGGTTGGTTCA", "ATGGTTGGTTCAAAACAG"))
|
|
370
|
+
['Ser4extLysGln']
|
|
371
|
+
|
|
372
|
+
Protein calling should stop at the first Ter encountered:
|
|
373
|
+
|
|
374
|
+
>>> list(find_variant_protein("ATGGTTGGTTCA", "ATGGTTTAGACA"))
|
|
375
|
+
['Gly3Ter']
|
|
376
|
+
|
|
377
|
+
>>> list(find_variant_protein("ATGGTTTGGTAG", "ATGGTTTAGTAG"))
|
|
378
|
+
['Trp3Ter']
|
|
379
|
+
|
|
380
|
+
Call specific synonyms:
|
|
381
|
+
|
|
382
|
+
>>> list(find_variant_protein("ATGGCCTAA", "ATGGCGTAA"))
|
|
383
|
+
['Ala2=']
|
|
384
|
+
|
|
385
|
+
>>> list(find_variant_protein("ATGGCCAAACCCTAA", "ATGGCGAAGCCATAA"))
|
|
386
|
+
['Ala2_Pro4=']
|
|
387
|
+
|
|
388
|
+
>>> list(find_variant_protein("ATGGCCAAACCCTAA", "ATGGCGAATCCATAA"))
|
|
389
|
+
['Ala2=', 'Lys3Asn', 'Pro4=']
|
|
390
|
+
|
|
391
|
+
>>> list(find_variant_protein("ATGGCCCCCAAATAA", "ATGGCGCCAAATTAA"))
|
|
392
|
+
['Ala2_Pro3=', 'Lys4Asn']
|
|
393
|
+
"""
|
|
394
|
+
|
|
395
|
+
ref_seq = ref_seq.strip().upper()
|
|
396
|
+
var_seq = var_seq.strip().upper()
|
|
397
|
+
|
|
398
|
+
if not re.match("[AGTC]+$", ref_seq):
|
|
399
|
+
raise ValueError("Invalid reference sequence") # pragma: no cover
|
|
400
|
+
|
|
401
|
+
if not re.match("[AGTC]+$", var_seq):
|
|
402
|
+
raise ValueError("Invalid variant sequence") # pragma: no cover
|
|
403
|
+
|
|
404
|
+
frame = (3 - offset) % 3
|
|
405
|
+
ref_pro = translate_dna(ref_seq[frame:])[0]
|
|
406
|
+
var_pro = translate_dna(var_seq[frame:])[0]
|
|
407
|
+
offset = (offset + 2) // 3
|
|
408
|
+
|
|
409
|
+
# cut protein translations off at first '*' (terminator)
|
|
410
|
+
if "*" in ref_pro:
|
|
411
|
+
ref_pro = ref_pro[: ref_pro.find("*") + 1]
|
|
412
|
+
if "*" in var_pro:
|
|
413
|
+
var_pro = var_pro[: var_pro.find("*") + 1]
|
|
414
|
+
|
|
415
|
+
def _ref(pos):
|
|
416
|
+
return f"{AA_CODES[ref_pro[pos]]}{pos+1+offset}"
|
|
417
|
+
|
|
418
|
+
opcodes = list(levenshtein_opcodes(ref_pro, var_pro))
|
|
419
|
+
|
|
420
|
+
for opcode in opcodes:
|
|
421
|
+
start, end = opcode.src_start, opcode.src_end
|
|
422
|
+
dest_pro = var_pro[opcode.dest_start : opcode.dest_end]
|
|
423
|
+
|
|
424
|
+
if opcode.tag == "delete":
|
|
425
|
+
assert dest_pro == ""
|
|
426
|
+
if len(ref_pro) > end and ref_pro[end] == "*":
|
|
427
|
+
# if the codon just after this deletion is a terminator,
|
|
428
|
+
# consider this an early termination.
|
|
429
|
+
yield f"{_ref(start)}Ter"
|
|
430
|
+
return
|
|
431
|
+
if end - start == 1:
|
|
432
|
+
yield f"{_ref(start)}del"
|
|
433
|
+
else:
|
|
434
|
+
yield f"{_ref(start)}_{_ref(end-1)}del"
|
|
435
|
+
|
|
436
|
+
elif opcode.tag == "insert":
|
|
437
|
+
assert start == end
|
|
438
|
+
|
|
439
|
+
if ref_pro[start - len(dest_pro) : start] == dest_pro:
|
|
440
|
+
if len(dest_pro) == 1:
|
|
441
|
+
yield f"{_ref(start-1)}dup"
|
|
442
|
+
else:
|
|
443
|
+
yield f"{_ref(start-len(dest_pro))}_{_ref(start-1)}dup"
|
|
444
|
+
elif start == len(ref_pro):
|
|
445
|
+
# 'extension', not quite standards compliant
|
|
446
|
+
yield f"{_ref(start-1)}ext{translate_aa(dest_pro)}"
|
|
447
|
+
else:
|
|
448
|
+
yield f"{_ref(start-1)}_{_ref(end)}ins{translate_aa(dest_pro)}"
|
|
449
|
+
|
|
450
|
+
elif opcode.tag == "replace":
|
|
451
|
+
# XXX handle extension if src_pro[-1] == '*'
|
|
452
|
+
|
|
453
|
+
if end - start == 1 and len(dest_pro) == 1:
|
|
454
|
+
yield f"{_ref(start)}{translate_aa(dest_pro)}"
|
|
455
|
+
else:
|
|
456
|
+
yield f"{_ref(start)}_{_ref(end-1)}delins{translate_aa(dest_pro)}"
|
|
457
|
+
|
|
458
|
+
# If the variant protein terminated, stop translating now:
|
|
459
|
+
if dest_pro[-1] == "*":
|
|
460
|
+
return
|
|
461
|
+
|
|
462
|
+
elif opcode.tag == "equal":
|
|
463
|
+
# Handle calling synonymous changes
|
|
464
|
+
assert end - start == opcode.dest_end - opcode.dest_start
|
|
465
|
+
start_ofs = None
|
|
466
|
+
for ofs in range(0, end - start):
|
|
467
|
+
src_dna = ref_seq[(start + ofs) * 3 + frame :][:3]
|
|
468
|
+
dest_dna = var_seq[(opcode.dest_start + ofs) * 3 + frame :][:3]
|
|
469
|
+
if src_dna == dest_dna:
|
|
470
|
+
if start_ofs is not None:
|
|
471
|
+
if start_ofs == ofs - 1:
|
|
472
|
+
yield f"{_ref(start+start_ofs)}="
|
|
473
|
+
else:
|
|
474
|
+
yield f"{_ref(start+start_ofs)}_{_ref(start+ofs-1)}="
|
|
475
|
+
start_ofs = None
|
|
476
|
+
elif start_ofs is None:
|
|
477
|
+
start_ofs = ofs
|
|
478
|
+
if start_ofs is not None:
|
|
479
|
+
if start_ofs == ofs:
|
|
480
|
+
yield f"{_ref(start+start_ofs)}="
|
|
481
|
+
else:
|
|
482
|
+
yield f"{_ref(start+start_ofs)}_{_ref(start+ofs)}="
|
|
483
|
+
|
|
484
|
+
|
|
485
|
+
def find_variant_string(
|
|
486
|
+
prefix: str,
|
|
487
|
+
ref_seq: str,
|
|
488
|
+
var_seq: str,
|
|
489
|
+
max_mutations: Optional[int] = None,
|
|
490
|
+
offset: int = 0,
|
|
491
|
+
minus_strand: bool = False,
|
|
492
|
+
) -> str:
|
|
493
|
+
"""As above, but returns a single string instead of a generator
|
|
494
|
+
|
|
495
|
+
MULTIPLE VARIATIONS
|
|
496
|
+
|
|
497
|
+
>>> find_variant_string("g.", "GATTACA", "GATTACA")
|
|
498
|
+
'g.='
|
|
499
|
+
>>> find_variant_string("g.", "GATTACA", "GTTTACA")
|
|
500
|
+
'g.2A>T'
|
|
501
|
+
>>> find_variant_string("g.", "GATTACA", "GTTTAGA")
|
|
502
|
+
'g.[2A>T;6C>G]'
|
|
503
|
+
>>> find_variant_string("g.", "GATTACA", "GTTCAGA")
|
|
504
|
+
'g.[2A>T;4T>C;6C>G]'
|
|
505
|
+
|
|
506
|
+
>>> find_variant_string("g.", "ATGGTTGGTTC", "ATGGTTGGTGGTTC")
|
|
507
|
+
'g.7_9dup'
|
|
508
|
+
>>> find_variant_string("g.", "ATGGTTGGTTCG", "ATGGTTGGTGGTTC")
|
|
509
|
+
'g.[7_9dup;12del]'
|
|
510
|
+
>>> find_variant_string("g.", "ATGGTTGGTTC", "ATGGTTGGTGGTTCG")
|
|
511
|
+
'g.[7_9dup;11_12insG]'
|
|
512
|
+
|
|
513
|
+
PROTEINS
|
|
514
|
+
|
|
515
|
+
>>> find_variant_string("p.", "ATGGTTGGTTCA", "ATGGTTGGTTCA")
|
|
516
|
+
'p.='
|
|
517
|
+
|
|
518
|
+
>>> find_variant_string("p.", "ATGGTTGGTTCA", "ATGGTTCCATCA")
|
|
519
|
+
'p.Gly3Pro'
|
|
520
|
+
|
|
521
|
+
>>> find_variant_string("p.", "ATGGTTGGTTCA", "ATGGGTTCA")
|
|
522
|
+
'p.Val2del'
|
|
523
|
+
|
|
524
|
+
>>> find_variant_string("p.", "ATGGTTGGTTCA", "ATGGTTGGTGGTTCA")
|
|
525
|
+
'p.Gly3dup'
|
|
526
|
+
|
|
527
|
+
>>> find_variant_string("p.", "ATGGTTGGTTCA", "ATGGCTGCTTCA")
|
|
528
|
+
'p.Val2_Gly3delinsAlaAla'
|
|
529
|
+
|
|
530
|
+
MINUS STRAND
|
|
531
|
+
|
|
532
|
+
this example is actually comparing TGTAATC and TCTGAAC ...
|
|
533
|
+
|
|
534
|
+
>>> find_variant_string("g.", "GATTACA", "GTTCAGA", minus_strand=True)
|
|
535
|
+
'g.[2G>C;3_4insG;6del]'
|
|
536
|
+
|
|
537
|
+
CHECK FOR INVALID INPUTS
|
|
538
|
+
|
|
539
|
+
>>> find_variant_string("g.", "HELLO", "CAT")
|
|
540
|
+
Traceback (most recent call last):
|
|
541
|
+
...
|
|
542
|
+
ValueError: Invalid reference sequence
|
|
543
|
+
|
|
544
|
+
>>> find_variant_string("g.", "CAT", "HELLO")
|
|
545
|
+
Traceback (most recent call last):
|
|
546
|
+
...
|
|
547
|
+
ValueError: Invalid variant sequence
|
|
548
|
+
|
|
549
|
+
CHECK FOR MAX MUTATIONS
|
|
550
|
+
|
|
551
|
+
>>> find_variant_string("g.", "ATTACC", "GATTACA",1)
|
|
552
|
+
Traceback (most recent call last):
|
|
553
|
+
...
|
|
554
|
+
countess_variant_caller.TooManyVariationsException: Too many variations (2) in GATTACA
|
|
555
|
+
"""
|
|
556
|
+
|
|
557
|
+
if minus_strand:
|
|
558
|
+
ref_seq = reverse_complement(ref_seq)
|
|
559
|
+
var_seq = reverse_complement(var_seq)
|
|
560
|
+
|
|
561
|
+
if prefix.endswith("p."):
|
|
562
|
+
variations = list(find_variant_protein(ref_seq, var_seq, offset))
|
|
563
|
+
else:
|
|
564
|
+
variations = list(find_variant_dna(ref_seq, var_seq, offset))
|
|
565
|
+
|
|
566
|
+
if len(variations) == 0:
|
|
567
|
+
return prefix + "="
|
|
568
|
+
|
|
569
|
+
if max_mutations is not None and len(variations) > max_mutations:
|
|
570
|
+
raise TooManyVariationsException("Too many variations (%d) in %s" % (len(variations), var_seq))
|
|
571
|
+
|
|
572
|
+
if len(variations) == 1:
|
|
573
|
+
return prefix + variations[0]
|
|
574
|
+
else:
|
|
575
|
+
return prefix + "[" + ";".join(variations) + "]"
|
|
576
|
+
|
|
577
|
+
|
|
578
|
+
def classify_protein_variant(pv: str) -> str:
|
|
579
|
+
"""Classify an HGVS protein variant as either wild type 'W',
|
|
580
|
+
synonymous 'S', deletion 'D', nonsense 'N', insertion 'I' or
|
|
581
|
+
missense 'M' or if none of the above '?'. Not particularly
|
|
582
|
+
strict as it is just used to classify the output of the
|
|
583
|
+
protein variant caller."""
|
|
584
|
+
|
|
585
|
+
if pv == "p.=":
|
|
586
|
+
return "W"
|
|
587
|
+
if re.match(r"p.\w+\d+=$", pv):
|
|
588
|
+
return "S"
|
|
589
|
+
if re.match(r"p.\w+\d+(_\w+\d+)?del$", pv):
|
|
590
|
+
return "D"
|
|
591
|
+
if re.match(r"p.\w+\d+Ter$", pv):
|
|
592
|
+
return "N"
|
|
593
|
+
if re.match(r"p.\w+\d+(_\w+\d+)?(dup|ins\w+)$", pv):
|
|
594
|
+
return "I"
|
|
595
|
+
if re.match(r"p.\w+\d+\w+$", pv):
|
|
596
|
+
return "M"
|
|
597
|
+
return "?"
|