geneimpacts 0.3.8__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- geneimpacts/__init__.py +3 -0
- geneimpacts/effect.py +647 -0
- geneimpacts/tests/__init__.py +1 -0
- geneimpacts/tests/bcfts.txt.gz +0 -0
- geneimpacts/tests/snpeff-anns.txt.gz +0 -0
- geneimpacts/tests/test_impacts.py +281 -0
- geneimpacts/tests/vep-csqs.txt.gz +0 -0
- geneimpacts-0.3.8.dist-info/METADATA +60 -0
- geneimpacts-0.3.8.dist-info/RECORD +12 -0
- geneimpacts-0.3.8.dist-info/WHEEL +5 -0
- geneimpacts-0.3.8.dist-info/licenses/LICENSE +22 -0
- geneimpacts-0.3.8.dist-info/top_level.txt +1 -0
geneimpacts/__init__.py
ADDED
geneimpacts/effect.py
ADDED
|
@@ -0,0 +1,647 @@
|
|
|
1
|
+
from __future__ import print_function
|
|
2
|
+
import sys
|
|
3
|
+
|
|
4
|
+
from functools import total_ordering
|
|
5
|
+
import re
|
|
6
|
+
import itertools as it
|
|
7
|
+
try:
|
|
8
|
+
izip = it.izip
|
|
9
|
+
except AttributeError:
|
|
10
|
+
izip = zip
|
|
11
|
+
basestring = str
|
|
12
|
+
|
|
13
|
+
old_snpeff_effect_so = {'CDS': 'coding_sequence_variant',
|
|
14
|
+
'CODON_CHANGE': 'coding_sequence_variant',
|
|
15
|
+
'CODON_CHANGE_PLUS_CODON_DELETION': 'disruptive_inframe_deletion',
|
|
16
|
+
'CODON_CHANGE_PLUS_CODON_INSERTION': 'disruptive_inframe_insertion',
|
|
17
|
+
'CODON_DELETION': 'inframe_deletion',
|
|
18
|
+
'CODON_INSERTION': 'inframe_insertion',
|
|
19
|
+
'DOWNSTREAM': 'downstream_gene_variant',
|
|
20
|
+
'EXON': 'exon_variant',
|
|
21
|
+
'EXON_DELETED': 'exon_loss_variant',
|
|
22
|
+
'FRAME_SHIFT': 'frameshift_variant',
|
|
23
|
+
'GENE': 'gene_variant',
|
|
24
|
+
'INTERGENIC': 'intergenic_variant',
|
|
25
|
+
'INTERGENIC_REGION': 'intergenic_region',
|
|
26
|
+
'INTERGENIC_CONSERVED': 'conserved_intergenic_variant',
|
|
27
|
+
'INTRAGENIC': 'intragenic_variant',
|
|
28
|
+
'INTRON': 'intron_variant',
|
|
29
|
+
'INTRON_CONSERVED': 'conserved_intron_variant',
|
|
30
|
+
'NON_SYNONYMOUS_CODING': 'missense_variant',
|
|
31
|
+
'RARE_AMINO_ACID': 'rare_amino_acid_variant',
|
|
32
|
+
'SPLICE_SITE_ACCEPTOR': 'splice_acceptor_variant',
|
|
33
|
+
'SPLICE_SITE_DONOR': 'splice_donor_variant',
|
|
34
|
+
'SPLICE_SITE_REGION': 'splice_region_variant',
|
|
35
|
+
#'START_GAINED': '5_prime_UTR_premature_start_codon_gain_variant',
|
|
36
|
+
'START_GAINED': '5_prime_UTR_premature_start_codon_variant',
|
|
37
|
+
'START_LOST': 'start_lost',
|
|
38
|
+
'STOP_GAINED': 'stop_gained',
|
|
39
|
+
'STOP_LOST': 'stop_lost',
|
|
40
|
+
'SYNONYMOUS_CODING': 'synonymous_variant',
|
|
41
|
+
'SYNONYMOUS_START': 'start_retained_variant',
|
|
42
|
+
'SYNONYMOUS_STOP': 'stop_retained_variant',
|
|
43
|
+
'TRANSCRIPT': 'transcript_variant',
|
|
44
|
+
'UPSTREAM': 'upstream_gene_variant',
|
|
45
|
+
'UTR_3_DELETED': '3_prime_UTR_truncation_+_exon_loss_variant',
|
|
46
|
+
'UTR_3_PRIME': '3_prime_UTR_variant',
|
|
47
|
+
'UTR_5_DELETED': '5_prime_UTR_truncation_+_exon_loss_variant',
|
|
48
|
+
'UTR_5_PRIME': '5_prime_UTR_variant',
|
|
49
|
+
'NON_SYNONYMOUS_START': 'initiator_codon_variant',
|
|
50
|
+
'NONE': 'None',
|
|
51
|
+
'CHROMOSOME_LARGE_DELETION': 'chromosomal_deletion'}
|
|
52
|
+
|
|
53
|
+
old_snpeff_lookup = {'CDS': 'LOW',
|
|
54
|
+
'CHROMOSOME_LARGE_DELETION': 'HIGH',
|
|
55
|
+
'CODON_CHANGE': 'MED',
|
|
56
|
+
'CODON_CHANGE_PLUS_CODON_DELETION': 'MED',
|
|
57
|
+
'CODON_CHANGE_PLUS_CODON_INSERTION': 'MED',
|
|
58
|
+
'CODON_DELETION': 'MED',
|
|
59
|
+
'CODON_INSERTION': 'MED',
|
|
60
|
+
'DOWNSTREAM': 'LOW',
|
|
61
|
+
'EXON': 'LOW',
|
|
62
|
+
'EXON_DELETED': 'HIGH',
|
|
63
|
+
'FRAME_SHIFT': 'HIGH',
|
|
64
|
+
'GENE': 'LOW',
|
|
65
|
+
'INTERGENIC': 'LOW',
|
|
66
|
+
'INTERGENIC_CONSERVED': 'LOW',
|
|
67
|
+
'INTRAGENIC': 'LOW',
|
|
68
|
+
'INTRON': 'LOW',
|
|
69
|
+
'INTRON_CONSERVED': 'LOW',
|
|
70
|
+
'NONE': 'LOW',
|
|
71
|
+
'NON_SYNONYMOUS_CODING': 'MED',
|
|
72
|
+
'NON_SYNONYMOUS_START': 'HIGH',
|
|
73
|
+
'RARE_AMINO_ACID': 'HIGH',
|
|
74
|
+
'SPLICE_SITE_ACCEPTOR': 'HIGH',
|
|
75
|
+
'SPLICE_SITE_DONOR': 'HIGH',
|
|
76
|
+
'SPLICE_SITE_REGION': 'MED',
|
|
77
|
+
'START_GAINED': 'LOW',
|
|
78
|
+
'START_LOST': 'HIGH',
|
|
79
|
+
'STOP_GAINED': 'HIGH',
|
|
80
|
+
'STOP_LOST': 'HIGH',
|
|
81
|
+
'SYNONYMOUS_CODING': 'LOW',
|
|
82
|
+
'SYNONYMOUS_START': 'LOW',
|
|
83
|
+
'SYNONYMOUS_STOP': 'LOW',
|
|
84
|
+
'TRANSCRIPT': 'LOW',
|
|
85
|
+
'UPSTREAM': 'LOW',
|
|
86
|
+
'UTR_3_DELETED': 'MED',
|
|
87
|
+
'UTR_3_PRIME': 'LOW',
|
|
88
|
+
'UTR_5_DELETED': 'MED',
|
|
89
|
+
'UTR_5_PRIME': 'LOW'}
|
|
90
|
+
# http://uswest.ensembl.org/info/genome/variation/predicted_data.html#consequences
|
|
91
|
+
IMPACT_SEVERITY = [
|
|
92
|
+
('chromosome_number_variation', 'HIGH'), # snpEff
|
|
93
|
+
('transcript_ablation', 'HIGH'), # VEP
|
|
94
|
+
('exon_loss_variant', 'HIGH'), # snpEff
|
|
95
|
+
('exon_loss', 'HIGH'), # snpEff
|
|
96
|
+
('rare_amino_acid_variant', 'HIGH'),
|
|
97
|
+
('protein_protein_contact', 'HIGH'), # snpEff
|
|
98
|
+
('structural_interaction_variant', 'HIGH'), #snpEff
|
|
99
|
+
('feature_fusion', 'HIGH'), #snpEff
|
|
100
|
+
('bidirectional_gene_fusion', 'HIGH'), #snpEff
|
|
101
|
+
('gene_fusion', 'HIGH'), #snpEff
|
|
102
|
+
('feature_ablation', 'HIGH'), #snpEff, structural varint
|
|
103
|
+
('splice_acceptor_variant', 'HIGH'), # VEP
|
|
104
|
+
('splice_donor_variant', 'HIGH'), # VEP
|
|
105
|
+
('start_retained_variant', 'HIGH'), # new VEP
|
|
106
|
+
('stop_gained', 'HIGH'), # VEP
|
|
107
|
+
('frameshift_variant', 'HIGH'), # VEP
|
|
108
|
+
('stop_lost', 'HIGH'), # VEP
|
|
109
|
+
('start_lost', 'HIGH'), # VEP
|
|
110
|
+
('transcript_amplification', 'HIGH'), # VEP
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
('disruptive_inframe_deletion', 'MED'), #snpEff
|
|
114
|
+
('conservative_inframe_deletion', 'MED'), #snpEff
|
|
115
|
+
('disruptive_inframe_insertion', 'MED'), #snpEff
|
|
116
|
+
('conservative_inframe_insertion', 'MED'), #snpEff
|
|
117
|
+
('duplication', 'MED'), # snpEff, structural variant
|
|
118
|
+
('inversion', 'MED'), # snpEff, structural variant
|
|
119
|
+
('exon_region', 'MED'), # snpEff, structural variant
|
|
120
|
+
('inframe_insertion', 'MED'), # VEP
|
|
121
|
+
('inframe_deletion', 'MED'), # VEP
|
|
122
|
+
('missense_variant', 'MED'), # VEP
|
|
123
|
+
('protein_altering_variant', 'MED'), # VEP
|
|
124
|
+
('initiator_codon_variant', 'MED'), # snpEff
|
|
125
|
+
('regulatory_region_ablation', 'MED'), # VEP
|
|
126
|
+
|
|
127
|
+
('5_prime_UTR_truncation', 'MED'), # found in snpEff
|
|
128
|
+
('splice_donor_5th_base_variant', 'MED'), # VEP changed to have medium priority
|
|
129
|
+
('splice_region_variant', 'MED'), # VEP changed to have medium priority
|
|
130
|
+
('splice_donor_region_variant', 'MED'), # VEP changed to have medium priority
|
|
131
|
+
('splice_polypyrimidine_tract_variant', 'MED'), # VEP changed to have medium priority
|
|
132
|
+
|
|
133
|
+
('3_prime_UTR_truncation', 'LOW'), # found in snpEff
|
|
134
|
+
('non_canonical_start_codon', 'LOW'), # found in snpEff
|
|
135
|
+
|
|
136
|
+
('synonymous_variant', 'LOW'), # VEP
|
|
137
|
+
('coding_sequence_variant', 'LOW'), # VEP
|
|
138
|
+
('incomplete_terminal_codon_variant', 'LOW'), # VEP
|
|
139
|
+
('stop_retained_variant', 'LOW'), # VEP
|
|
140
|
+
('mature_miRNA_variant', 'LOW'), # VEP
|
|
141
|
+
('5_prime_UTR_premature_start_codon_variant', 'LOW'), # snpEff
|
|
142
|
+
('5_prime_UTR_premature_start_codon_gain_variant', 'LOW'), #snpEff
|
|
143
|
+
('5_prime_UTR_variant', 'LOW'), # VEP
|
|
144
|
+
('3_prime_UTR_variant', 'LOW'), # VEP
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
('non_coding_transcript_exon_variant', 'LOW'), # VEP
|
|
148
|
+
('conserved_intron_variant', 'LOW'), # snpEff
|
|
149
|
+
('intron_variant', 'LOW'), # VEP
|
|
150
|
+
('exon_variant', 'LOW'), # snpEff
|
|
151
|
+
('gene_variant', 'LOW'), # snpEff
|
|
152
|
+
('NMD_transcript_variant', 'LOW'), # VEP
|
|
153
|
+
('non_coding_transcript_variant', 'LOW'), # VEP
|
|
154
|
+
('coding_transcript_variant', 'LOW'), # VEP
|
|
155
|
+
('upstream_gene_variant', 'LOW'), # VEP
|
|
156
|
+
('downstream_gene_variant', 'LOW'), # VEP
|
|
157
|
+
('TFBS_ablation', 'LOW'), # VEP
|
|
158
|
+
('TFBS_amplification', 'LOW'), # VEP
|
|
159
|
+
('TF_binding_site_variant', 'LOW'), # VEP
|
|
160
|
+
('regulatory_region_amplification', 'LOW'), # VEP
|
|
161
|
+
('feature_elongation', 'LOW'), # VEP
|
|
162
|
+
('sequence_variant', 'LOW'), # VEP
|
|
163
|
+
('miRNA', 'LOW'), # snpEff
|
|
164
|
+
('transcript_variant', 'LOW'), # snpEff
|
|
165
|
+
('start_retained', 'LOW'), # snpEff
|
|
166
|
+
('regulatory_region_variant', 'LOW'), # VEP
|
|
167
|
+
('feature_truncation', 'LOW'), # VEP
|
|
168
|
+
('non_coding_exon_variant', 'LOW'),
|
|
169
|
+
('nc_transcript_variant', 'LOW'),
|
|
170
|
+
('conserved_intergenic_variant', 'LOW'), # snpEff
|
|
171
|
+
('intergenic_variant', 'LOW'), # VEP
|
|
172
|
+
('intergenic_region', 'LOW'), # snpEff
|
|
173
|
+
('intragenic_variant', 'LOW'), # snpEff
|
|
174
|
+
('non_coding_transcript_exon_variant', 'LOW'), # snpEff
|
|
175
|
+
('non_coding_transcript_variant', 'LOW'), # snpEff
|
|
176
|
+
('transcript', 'LOW'), # ? snpEff older
|
|
177
|
+
('sequence_feature', 'LOW'), # snpEff older
|
|
178
|
+
('non_coding', 'LOW'), # BCSQ
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
('?', 'UNKNOWN'), # some VEP annotations have '?'
|
|
182
|
+
('', 'UNKNOWN'), # some VEP annotations have ''
|
|
183
|
+
('UNKNOWN', 'UNKNOWN'), # some snpEFF annotations have 'unknown'
|
|
184
|
+
]
|
|
185
|
+
|
|
186
|
+
# bcftools doesn't add _variant on the end.
|
|
187
|
+
for (csq, imp) in list(IMPACT_SEVERITY[::-1]):
|
|
188
|
+
if csq.endswith('_variant'):
|
|
189
|
+
for i, (a, b) in enumerate(IMPACT_SEVERITY):
|
|
190
|
+
if (a, b) == (csq, imp):
|
|
191
|
+
IMPACT_SEVERITY.insert(i, (csq[:-8].lower(), imp))
|
|
192
|
+
break
|
|
193
|
+
|
|
194
|
+
IMPACT_SEVERITY_ORDER = dict((x[0], i) for i, x in enumerate(IMPACT_SEVERITY[::-1]))
|
|
195
|
+
IMPACT_SEVERITY = dict(IMPACT_SEVERITY)
|
|
196
|
+
|
|
197
|
+
EXONIC_IMPACTS = set(["stop_gained",
|
|
198
|
+
"exon_variant",
|
|
199
|
+
"stop_lost",
|
|
200
|
+
"frameshift_variant",
|
|
201
|
+
"initiator_codon_variant",
|
|
202
|
+
"inframe_deletion",
|
|
203
|
+
"inframe_insertion",
|
|
204
|
+
"missense_variant",
|
|
205
|
+
"protein_altering_variant",
|
|
206
|
+
"incomplete_terminal_codon_variant",
|
|
207
|
+
"stop_retained_variant",
|
|
208
|
+
"5_prime_UTR_premature_start_codon_variant",
|
|
209
|
+
"synonymous_variant",
|
|
210
|
+
"coding_sequence_variant",
|
|
211
|
+
"5_prime_UTR_variant",
|
|
212
|
+
"3_prime_UTR_variant",
|
|
213
|
+
"transcript_ablation",
|
|
214
|
+
"transcript_amplification",
|
|
215
|
+
"feature_elongation",
|
|
216
|
+
"feature_truncation"])
|
|
217
|
+
|
|
218
|
+
for im in list(EXONIC_IMPACTS):
|
|
219
|
+
if im.endswith("_variant"):
|
|
220
|
+
EXONIC_IMPACTS.add(im[:-8])
|
|
221
|
+
EXONIC_IMPACTS = frozenset(EXONIC_IMPACTS)
|
|
222
|
+
|
|
223
|
+
def snpeff_aa_length(self):
|
|
224
|
+
try:
|
|
225
|
+
v = self.effects['AA.pos / AA.length']
|
|
226
|
+
if v.strip():
|
|
227
|
+
return int(v.split("/")[1].strip())
|
|
228
|
+
except:
|
|
229
|
+
try:
|
|
230
|
+
return int(self.effects['Amino_Acid_length'])
|
|
231
|
+
except:
|
|
232
|
+
return None
|
|
233
|
+
|
|
234
|
+
def vep_aa_length(self):
|
|
235
|
+
if not 'Protein_position' in self.effects:
|
|
236
|
+
return None
|
|
237
|
+
try:
|
|
238
|
+
return int(self.effects['Protein_position'])
|
|
239
|
+
except ValueError:
|
|
240
|
+
try:
|
|
241
|
+
return self.effects['Protein_position']
|
|
242
|
+
except KeyError:
|
|
243
|
+
return None
|
|
244
|
+
|
|
245
|
+
def vep_polyphen_pred(self):
|
|
246
|
+
try:
|
|
247
|
+
return self.effects['PolyPhen'].split('(')[0]
|
|
248
|
+
except (KeyError, IndexError):
|
|
249
|
+
return None
|
|
250
|
+
|
|
251
|
+
def vep_polyphen_score(self):
|
|
252
|
+
try:
|
|
253
|
+
return float(self.effects['PolyPhen'].split('(')[1][:-1])
|
|
254
|
+
except (KeyError, IndexError):
|
|
255
|
+
return None
|
|
256
|
+
|
|
257
|
+
def vep_sift_score(self):
|
|
258
|
+
try:
|
|
259
|
+
return float(self.effects['SIFT'].split("(")[1][:-1])
|
|
260
|
+
except (IndexError, KeyError):
|
|
261
|
+
return None
|
|
262
|
+
|
|
263
|
+
def vep_sift_pred(self):
|
|
264
|
+
try:
|
|
265
|
+
return self.effects['SIFT'].split("(")[0]
|
|
266
|
+
except (IndexError, KeyError):
|
|
267
|
+
return None
|
|
268
|
+
|
|
269
|
+
snpeff_lookup = {
|
|
270
|
+
'transcript': ['Feature_ID', 'Transcript_ID', 'Transcript'],
|
|
271
|
+
'gene': 'Gene_Name',
|
|
272
|
+
'exon': ['Rank', 'Exon', 'Exon_Rank'],
|
|
273
|
+
'codon_change': ['HGVS.c', 'Codon_Change'],
|
|
274
|
+
'aa_change': ['HGVS.p', 'Amino_Acid_Change', 'Amino_Acid_change'],
|
|
275
|
+
'aa_length': snpeff_aa_length,
|
|
276
|
+
'biotype': ['Transcript_BioType', 'Gene_BioType'],
|
|
277
|
+
'alt': 'Allele',
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
bcft_lookup = {}
|
|
281
|
+
|
|
282
|
+
vep_lookup = {
|
|
283
|
+
'transcript': 'Feature',
|
|
284
|
+
'gene': ['SYMBOL', 'HGNC', 'Gene'],
|
|
285
|
+
'ensembl_gene_id': 'Gene',
|
|
286
|
+
'exon': 'EXON',
|
|
287
|
+
'codon_change': 'Codons',
|
|
288
|
+
'aa_change': 'Amino_acids',
|
|
289
|
+
'aa_length': vep_aa_length,
|
|
290
|
+
'biotype': 'BIOTYPE',
|
|
291
|
+
'polyphen_pred': vep_polyphen_pred,
|
|
292
|
+
'polyphen_score': vep_polyphen_score,
|
|
293
|
+
'sift_pred': vep_sift_pred,
|
|
294
|
+
'sift_score': vep_sift_score,
|
|
295
|
+
'alt': 'ALLELE',
|
|
296
|
+
}
|
|
297
|
+
|
|
298
|
+
# lookup here instead of returning ''.
|
|
299
|
+
defaults = {'gene': None}
|
|
300
|
+
|
|
301
|
+
@total_ordering
|
|
302
|
+
class Effect(object):
|
|
303
|
+
_top_consequence = None
|
|
304
|
+
lookup = None
|
|
305
|
+
|
|
306
|
+
def __init__(self, key, effect_dict, keys, prioritize_canonical):
|
|
307
|
+
raise NotImplemented
|
|
308
|
+
|
|
309
|
+
@classmethod
|
|
310
|
+
def new(self, key, effect_dict, keys):
|
|
311
|
+
lookup = {"CSQ": VEP, "ANN": SnpEff, "EFF": OldSnpEff, "BCSQ": BCFT}
|
|
312
|
+
assert key in lookup
|
|
313
|
+
return lookup[key](effect_dict, keys)
|
|
314
|
+
|
|
315
|
+
@property
|
|
316
|
+
def is_exonic(self):
|
|
317
|
+
return self.top_consequence in EXONIC_IMPACTS
|
|
318
|
+
|
|
319
|
+
def unused(self):
|
|
320
|
+
return []
|
|
321
|
+
|
|
322
|
+
@property
|
|
323
|
+
def top_consequence(self):
|
|
324
|
+
# sort by order and return the top
|
|
325
|
+
if self._top_consequence is None:
|
|
326
|
+
self._top_consequence = sorted([(IMPACT_SEVERITY_ORDER.get(c, 0), c) for c in
|
|
327
|
+
self.consequences], reverse=True)[0][1]
|
|
328
|
+
return self._top_consequence
|
|
329
|
+
|
|
330
|
+
@property
|
|
331
|
+
def so(self):
|
|
332
|
+
return self.top_consequence
|
|
333
|
+
|
|
334
|
+
@property
|
|
335
|
+
def is_coding(self):
|
|
336
|
+
return self.biotype == "protein_coding" and self.is_exonic and ("_UTR_" not in self.top_consequence)
|
|
337
|
+
|
|
338
|
+
@property
|
|
339
|
+
def is_splicing(self):
|
|
340
|
+
return "splice" in self.top_consequence
|
|
341
|
+
|
|
342
|
+
@property
|
|
343
|
+
def is_lof(self):
|
|
344
|
+
return self.biotype == "protein_coding" and self.impact_severity == "HIGH"
|
|
345
|
+
|
|
346
|
+
def __le__(self, other):
|
|
347
|
+
# we sort so that the effects with the highest impacts come last
|
|
348
|
+
# (highest) and so, we:
|
|
349
|
+
# + return true if self has lower impact than other.
|
|
350
|
+
# + return false if self has higher impact than other.
|
|
351
|
+
self_has_lower_impact = True
|
|
352
|
+
self_has_higher_impact = False
|
|
353
|
+
|
|
354
|
+
if self.prioritize_canonical:
|
|
355
|
+
scanon, ocanon = self.is_canonical, other.is_canonical
|
|
356
|
+
if scanon and not ocanon:
|
|
357
|
+
return self_has_higher_impact
|
|
358
|
+
elif ocanon and not scanon:
|
|
359
|
+
return self_has_lower_impact
|
|
360
|
+
|
|
361
|
+
spg = self.is_pseudogene
|
|
362
|
+
opg = other.is_pseudogene
|
|
363
|
+
if spg and not opg:
|
|
364
|
+
return self_has_lower_impact
|
|
365
|
+
elif opg and not spg:
|
|
366
|
+
return self_has_higher_impact
|
|
367
|
+
|
|
368
|
+
sc, oc = self.coding, other.coding
|
|
369
|
+
if sc and not oc:
|
|
370
|
+
# other is not coding. is is splicing?
|
|
371
|
+
# if other is splicing, we have lower impact.
|
|
372
|
+
if not (self.is_splicing or other.is_splicing):
|
|
373
|
+
return self_has_higher_impact
|
|
374
|
+
elif oc and not sc:
|
|
375
|
+
# self. is not coding. is it splicing?
|
|
376
|
+
# if self is splicing it has higher impact
|
|
377
|
+
if not (self.is_splicing or other.is_splicing):
|
|
378
|
+
return self_has_lower_impact
|
|
379
|
+
|
|
380
|
+
if self.severity != other.severity:
|
|
381
|
+
return self.severity <= other.severity
|
|
382
|
+
|
|
383
|
+
if self.biotype == "protein_coding" and not other.biotype == "protein_coding":
|
|
384
|
+
return False
|
|
385
|
+
elif other.biotype == "protein_coding" and not self.biotype == "protein_coding":
|
|
386
|
+
return True
|
|
387
|
+
|
|
388
|
+
if self.biotype == "processed_transcript" and not other.biotype == "processed_transcript":
|
|
389
|
+
return False
|
|
390
|
+
elif other.biotype == "processed_transcript" and not self.biotype == "processed_transcript":
|
|
391
|
+
return True
|
|
392
|
+
|
|
393
|
+
# sift higher == more damaing
|
|
394
|
+
if (self.sift_value or 10000) < (other.sift_value or 10000):
|
|
395
|
+
return True
|
|
396
|
+
|
|
397
|
+
# polyphen, lower == more damaging
|
|
398
|
+
if (self.polyphen_value or -10000) > (other.polyphen_value or -10000):
|
|
399
|
+
return True
|
|
400
|
+
|
|
401
|
+
return max(IMPACT_SEVERITY_ORDER.get(c, 0) for c in self.consequences) <= \
|
|
402
|
+
max(IMPACT_SEVERITY_ORDER.get(co, 0) for co in other.consequences)
|
|
403
|
+
|
|
404
|
+
@classmethod
|
|
405
|
+
def top_severity(cls, effects):
|
|
406
|
+
for i, e in enumerate(effects):
|
|
407
|
+
if isinstance(e, basestring):
|
|
408
|
+
|
|
409
|
+
effects[i] = cls(e)
|
|
410
|
+
|
|
411
|
+
if len(effects) == 0:
|
|
412
|
+
return None
|
|
413
|
+
if len(effects) == 1:
|
|
414
|
+
return effects[0]
|
|
415
|
+
effects = sorted(effects)
|
|
416
|
+
if effects[-1] > effects[-2]:
|
|
417
|
+
return effects[-1]
|
|
418
|
+
ret = [effects[-1], effects[-2]]
|
|
419
|
+
for i in range(-3, -(len(effects) - 1), -1):
|
|
420
|
+
if effects[-1] > effects[i]: break
|
|
421
|
+
ret.append(effects[i])
|
|
422
|
+
return ret
|
|
423
|
+
|
|
424
|
+
def __getitem__(self, key):
|
|
425
|
+
return self.effects[key]
|
|
426
|
+
|
|
427
|
+
def __eq__(self, other):
|
|
428
|
+
if not isinstance(other, Effect): return False
|
|
429
|
+
return self.effect_string == other.effect_string
|
|
430
|
+
|
|
431
|
+
def __str__(self):
|
|
432
|
+
return repr(self)
|
|
433
|
+
|
|
434
|
+
def __repr__(self):
|
|
435
|
+
return "%s(%s-%s, %s)" % (self.__class__.__name__, self.gene,
|
|
436
|
+
self.consequence, self.impact_severity)
|
|
437
|
+
|
|
438
|
+
@property
|
|
439
|
+
def effect_severity(self):
|
|
440
|
+
return self.impact_severity
|
|
441
|
+
|
|
442
|
+
@property
|
|
443
|
+
def lof(self):
|
|
444
|
+
return self.biotype == "protein_coding" and self.impact_severity == "HIGH"
|
|
445
|
+
|
|
446
|
+
@property
|
|
447
|
+
def severity(self, lookup={'HIGH': 3, 'MED': 2, 'LOW': 1, 'UNKNOWN': 0}, sev=IMPACT_SEVERITY):
|
|
448
|
+
# higher is more severe. used for ordering.
|
|
449
|
+
try:
|
|
450
|
+
v = max(lookup[sev[csq]] for csq in self.consequences)
|
|
451
|
+
except KeyError:
|
|
452
|
+
v = 0
|
|
453
|
+
if v == 0:
|
|
454
|
+
excl = []
|
|
455
|
+
for i, c in [(i, c) for i, c in enumerate(self.consequences) if not c in sev]:
|
|
456
|
+
sys.stderr.write("WARNING: unknown severity for '%s' with effect '%s'\n" % (self.effect_string, c))
|
|
457
|
+
sys.stderr.write("Please report this on github with the effect-string above\n")
|
|
458
|
+
excl.append(i)
|
|
459
|
+
if len(excl) == len(self.consequences):
|
|
460
|
+
v = 1
|
|
461
|
+
else:
|
|
462
|
+
v = max(lookup[sev[csq]] for i, csq in enumerate(self.consequences) if not i in excl)
|
|
463
|
+
return max(v, 1)
|
|
464
|
+
|
|
465
|
+
@property
|
|
466
|
+
def impact_severity(self):
|
|
467
|
+
return ['xxx', 'LOW', 'MED', 'HIGH'][self.severity]
|
|
468
|
+
|
|
469
|
+
@property
|
|
470
|
+
def consequence(self):
|
|
471
|
+
return self.top_consequence
|
|
472
|
+
|
|
473
|
+
@property
|
|
474
|
+
def is_pseudogene(self): #bool
|
|
475
|
+
return self.biotype is not None and 'pseudogene' in self.biotype
|
|
476
|
+
|
|
477
|
+
|
|
478
|
+
def __getattr__(self, k):
|
|
479
|
+
v = self.lookup.get(k)
|
|
480
|
+
if v is None: return v
|
|
481
|
+
if isinstance(v, basestring):
|
|
482
|
+
ret = self.effects.get(v)
|
|
483
|
+
# if we didn't get value, there may be a column
|
|
484
|
+
# specific value stored in defaults so we look import
|
|
485
|
+
# up.
|
|
486
|
+
if not ret and ret is not False:
|
|
487
|
+
return defaults.get(k, '')
|
|
488
|
+
return ret
|
|
489
|
+
elif isinstance(v, list):
|
|
490
|
+
for key in v:
|
|
491
|
+
try:
|
|
492
|
+
return self.effects[key]
|
|
493
|
+
except KeyError:
|
|
494
|
+
continue
|
|
495
|
+
return defaults.get(k, '')
|
|
496
|
+
return v(self)
|
|
497
|
+
|
|
498
|
+
class BCFT(Effect):
|
|
499
|
+
__slots__ = ('effect_string', 'effects', 'biotype', 'gene', 'transcript', 'aa_change', 'dna_change')
|
|
500
|
+
keys = "consequence,gene,transcript,biotype,strand,amino_acid_change,dna_change".split(",")
|
|
501
|
+
lookup = bcft_lookup
|
|
502
|
+
|
|
503
|
+
def __init__(self, effect_string, keys=None, prioritize_canonical=False):
|
|
504
|
+
if keys is not None: self.keys = keys
|
|
505
|
+
self.effect_string = effect_string
|
|
506
|
+
self.effects = dict(izip(self.keys, (x.strip().replace(' ', '_') for x in effect_string.split("|"))))
|
|
507
|
+
self.biotype = self.effects.get('biotype', None)
|
|
508
|
+
self.transcript = self.effects.get('transcript', None)
|
|
509
|
+
self.gene = self.effects.get('gene', None)
|
|
510
|
+
self.aa_change = self.effects.get('amino_acid_change', None)
|
|
511
|
+
self.consequences = self.effects[self.keys[0]].split('&')
|
|
512
|
+
|
|
513
|
+
def unused(self, used=frozenset("csq|gene|transcript|biotype|strand|aa_change|dna_change".lower().split("|"))):
|
|
514
|
+
"""Return fields that were in the VCF but weren't utilized as part of the standard fields supported here."""
|
|
515
|
+
return [k for k in self.keys if not k.lower() in used]
|
|
516
|
+
|
|
517
|
+
@property
|
|
518
|
+
def exonic(self):
|
|
519
|
+
return self.biotype == "protein_coding" and any(csq in EXONIC_IMPACTS for csq in self.consequences)
|
|
520
|
+
|
|
521
|
+
@property
|
|
522
|
+
def coding(self):
|
|
523
|
+
# what about start/stop_gained?
|
|
524
|
+
return self.exonic and any(csq[1:] != "_prime_utr" for csq in self.consequences)
|
|
525
|
+
|
|
526
|
+
|
|
527
|
+
class VEP(Effect):
|
|
528
|
+
__slots__ = ('effect_string', 'effects', 'biotype')
|
|
529
|
+
keys = "Consequence|Codons|Amino_acids|Gene|SYMBOL|Feature|EXON|PolyPhen|SIFT|Protein_position|BIOTYPE|CANONICAL".split("|")
|
|
530
|
+
lookup = vep_lookup
|
|
531
|
+
|
|
532
|
+
def __init__(self, effect_string, keys=None, checks=True, prioritize_canonical=False):
|
|
533
|
+
if checks:
|
|
534
|
+
assert not "," in effect_string
|
|
535
|
+
assert not "=" in effect_string
|
|
536
|
+
self.effect_string = effect_string
|
|
537
|
+
if keys is not None: self.keys = keys
|
|
538
|
+
|
|
539
|
+
self.effect_string = effect_string
|
|
540
|
+
self.effects = dict(izip(self.keys, (x.strip() for x in effect_string.split("|"))))
|
|
541
|
+
self.biotype = self.effects.get('BIOTYPE', None)
|
|
542
|
+
self.prioritize_canonical = prioritize_canonical
|
|
543
|
+
|
|
544
|
+
@property
|
|
545
|
+
def consequences(self, _cache={}):
|
|
546
|
+
try:
|
|
547
|
+
# this is a bottleneck so we keep a cache
|
|
548
|
+
return _cache[self.effects['Consequence']]
|
|
549
|
+
except KeyError:
|
|
550
|
+
res = _cache[self.effects['Consequence']] = list(it.chain.from_iterable(x.split("+") for x in self.effects['Consequence'].split('&')))
|
|
551
|
+
return res
|
|
552
|
+
|
|
553
|
+
def unused(self, used=frozenset("Consequence|Codons|Amino_acids|Gene|SYMBOL|Feature|EXON|PolyPhen|SIFT|Protein_position|BIOTYPE|CANONICAL".lower().split("|"))):
|
|
554
|
+
"""Return fields that were in the VCF but weren't utilized as part of the standard fields supported here."""
|
|
555
|
+
return [k for k in self.keys if not k.lower() in used]
|
|
556
|
+
|
|
557
|
+
@property
|
|
558
|
+
def coding(self):
|
|
559
|
+
# what about start/stop_gained?
|
|
560
|
+
return self.exonic and any(csq[1:] != "_prime_UTR_variant" for csq in self.consequences)
|
|
561
|
+
|
|
562
|
+
@property
|
|
563
|
+
def exonic(self):
|
|
564
|
+
return self.biotype == "protein_coding" and any(csq in EXONIC_IMPACTS for csq in self.consequences)
|
|
565
|
+
|
|
566
|
+
@property
|
|
567
|
+
def is_canonical(self):
|
|
568
|
+
return self.effects.get("CANONICAL", "") != ""
|
|
569
|
+
|
|
570
|
+
class SnpEff(Effect):
|
|
571
|
+
lookup = snpeff_lookup
|
|
572
|
+
|
|
573
|
+
__slots__ = ('effects', 'effect_string', 'biotype')
|
|
574
|
+
|
|
575
|
+
keys = [x.strip() for x in 'Allele | Annotation | Annotation_Impact | Gene_Name | Gene_ID | Feature_Type | Feature_ID | Transcript_BioType | Rank | HGVS.c | HGVS.p | cDNA.pos / cDNA.length | CDS.pos / CDS.length | AA.pos / AA.length | Distance | ERRORS / WARNINGS / INFO'.split("|")]
|
|
576
|
+
|
|
577
|
+
def __init__(self, effect_string, keys=None, prioritize_canonical=False):
|
|
578
|
+
assert not "," in effect_string
|
|
579
|
+
assert not "=" == effect_string[3]
|
|
580
|
+
self.effect_string = effect_string
|
|
581
|
+
if keys is not None:
|
|
582
|
+
self.keys = keys
|
|
583
|
+
self.effects = dict(izip(self.keys, (x.strip() for x in effect_string.split("|", len(self.keys)))))
|
|
584
|
+
self.biotype = self.effects['Transcript_BioType']
|
|
585
|
+
|
|
586
|
+
@property
|
|
587
|
+
def consequences(self):
|
|
588
|
+
return list(it.chain.from_iterable(x.split("+") for x in self.effects['Annotation'].split('&')))
|
|
589
|
+
|
|
590
|
+
@property
|
|
591
|
+
def coding(self):
|
|
592
|
+
# TODO: check start_gained and utr
|
|
593
|
+
return self.exonic and not "utr" in self.consequence and not "start_gained" in self.consequence
|
|
594
|
+
|
|
595
|
+
|
|
596
|
+
@property
|
|
597
|
+
def exonic(self):
|
|
598
|
+
csqs = self.consequence
|
|
599
|
+
if isinstance(csqs, basestring):
|
|
600
|
+
csqs = [csqs]
|
|
601
|
+
return any(csq in EXONIC_IMPACTS for csq in csqs) and self.effects['Transcript_BioType'] == 'protein_coding'
|
|
602
|
+
|
|
603
|
+
|
|
604
|
+
class OldSnpEff(SnpEff):
|
|
605
|
+
|
|
606
|
+
keys = [x.strip() for x in "Effect | Effect_Impact | Functional_Class | Codon_Change | Amino_Acid_change| Amino_Acid_length | Gene_Name | Gene_BioType | Coding | Transcript | Exon | ERRORS | WARNINGS".split("|")]
|
|
607
|
+
|
|
608
|
+
def __init__(self, effect_string, keys=None, _patt=re.compile(r"\||\("),
|
|
609
|
+
prioritize_canonical=False):
|
|
610
|
+
assert not "," in effect_string
|
|
611
|
+
assert not "=" in effect_string
|
|
612
|
+
effect_string = effect_string.rstrip(")")
|
|
613
|
+
self.effect_string = effect_string
|
|
614
|
+
if keys is not None:
|
|
615
|
+
self.keys = keys
|
|
616
|
+
self.effects = dict(izip(self.keys, (x.strip() for x in _patt.split(effect_string))))
|
|
617
|
+
|
|
618
|
+
@property
|
|
619
|
+
def consequence(self):
|
|
620
|
+
if '&' in self.effects['Effect']:
|
|
621
|
+
return self.effects['Effect'].split('&')
|
|
622
|
+
return self.effects['Effect']
|
|
623
|
+
|
|
624
|
+
@property
|
|
625
|
+
def consequences(self):
|
|
626
|
+
try:
|
|
627
|
+
return [old_snpeff_effect_so.get(c, old_snpeff_effect_so[c.upper()]) for c in it.chain.from_iterable(x.split("+") for x in
|
|
628
|
+
self.effects['Effect'].split('&'))]
|
|
629
|
+
except KeyError:
|
|
630
|
+
return list(it.chain.from_iterable(x.split("+") for x in self.effects['Effect'].split('&')))
|
|
631
|
+
|
|
632
|
+
@property
|
|
633
|
+
def severity(self, lookup={'HIGH': 3, 'MED': 2, 'LOW': 1}):
|
|
634
|
+
# higher is more severe. used for ordering.
|
|
635
|
+
try:
|
|
636
|
+
return max(lookup[old_snpeff_lookup[csq]] for csq in self.consequences)
|
|
637
|
+
except KeyError:
|
|
638
|
+
try:
|
|
639
|
+
#in between
|
|
640
|
+
sevs = [IMPACT_SEVERITY.get(csq, "LOW") for csq in self.consequences]
|
|
641
|
+
return max(lookup[s] for s in sevs)
|
|
642
|
+
except KeyError:
|
|
643
|
+
return Effect.severity.fget(self)
|
|
644
|
+
|
|
645
|
+
@property
|
|
646
|
+
def is_lof(self):
|
|
647
|
+
return self.biotype == "protein_coding" and self.impact_severity == "HIGH"
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
#
|
|
Binary file
|
|
Binary file
|
|
@@ -0,0 +1,281 @@
|
|
|
1
|
+
import sys
|
|
2
|
+
import os
|
|
3
|
+
import gzip
|
|
4
|
+
from geneimpacts import SnpEff, VEP, Effect, OldSnpEff, BCFT
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
HERE = os.path.dirname(__file__)
|
|
8
|
+
|
|
9
|
+
def test_bug():
|
|
10
|
+
e = sorted([VEP('missense_variant|tTt/tGt|F/C|ENSG00000186092|OR4F5|ENST00000335137|1/1|possibly_damaging(0.568)|deleterious(0)|113/305|protein_coding'),
|
|
11
|
+
VEP("splice_region_variant&non_coding_exon_variant&nc_transcript_variant|||ENSG00000223972|DDX11L1|ENST00000456328|2/3||||processed_transcript")])
|
|
12
|
+
assert e[-1].so == 'missense_variant', e[-1].so
|
|
13
|
+
|
|
14
|
+
def test_snpeff():
|
|
15
|
+
|
|
16
|
+
ann = SnpEff("C|splice_donor_variant&splice_region_variant&splice_region_variant&intron_variant|HIGH|DDX11L1|ENSG00000223972|transcript|ENST00000518655|transcribed_unprocessed_pseudogene|3/3|n.734+2_734+3delAG||||||")
|
|
17
|
+
|
|
18
|
+
assert ann.gene == "DDX11L1"
|
|
19
|
+
assert ann.transcript == "ENST00000518655"
|
|
20
|
+
assert ann.biotype == "transcribed_unprocessed_pseudogene", ann.biotype
|
|
21
|
+
assert ann.consequences == 'splice_donor_variant&splice_region_variant&splice_region_variant&intron_variant'.split('&')
|
|
22
|
+
assert ann.severity == 3
|
|
23
|
+
assert ann.impact_severity == "HIGH"
|
|
24
|
+
assert ann.aa_change == ""
|
|
25
|
+
assert ann.exon == '3/3', ann.exon
|
|
26
|
+
assert not ann.coding
|
|
27
|
+
assert ann.is_pseudogene
|
|
28
|
+
|
|
29
|
+
def test_unused():
|
|
30
|
+
extra = ['YYY']
|
|
31
|
+
keys = VEP.keys + extra
|
|
32
|
+
ann = VEP('missense_variant|tTt/tGt|F/C|ENSG00000186092|OR4F5|ENST00000335137|1/1|possibly_damaging(0.568)|deleterious(0)|113/305|protein_coding|xval|yval', keys=keys)
|
|
33
|
+
assert ann.unused() == extra, ann.unused()
|
|
34
|
+
|
|
35
|
+
assert ann.effects['YYY'] == 'yval'
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def test_vep():
|
|
40
|
+
|
|
41
|
+
ann = VEP('missense_variant|tTt/tGt|F/C|ENSG00000186092|OR4F5|ENST00000335137|1/1|possibly_damaging(0.568)|deleterious(0)|113/305|protein_coding|')
|
|
42
|
+
assert ann.gene == 'OR4F5'
|
|
43
|
+
assert ann.transcript == 'ENST00000335137'
|
|
44
|
+
assert ann.aa_change == "F/C", ann.aa_change
|
|
45
|
+
assert ann.consequences == ['missense_variant']
|
|
46
|
+
assert ann.coding
|
|
47
|
+
assert ann.biotype == "protein_coding"
|
|
48
|
+
assert ann.severity == 2
|
|
49
|
+
assert ann.impact_severity == "MED", ann.impact_severity
|
|
50
|
+
assert not ann.is_pseudogene
|
|
51
|
+
assert ann.polyphen_score == 0.568, ann.polyphen
|
|
52
|
+
assert ann.polyphen_pred == "possibly_damaging", ann.polyphen
|
|
53
|
+
assert ann.sift_score == 0.0, ann.sift
|
|
54
|
+
assert ann.sift_pred == "deleterious", ann.sift
|
|
55
|
+
assert not ann.canonical
|
|
56
|
+
|
|
57
|
+
def test_vep_canonical():
|
|
58
|
+
|
|
59
|
+
ann = VEP('missense_variant|tTt/tGt|F/C|ENSG00000186092|OR4F5|ENST00000335137|1/1|possibly_damaging(0.568)|deleterious(0)|113/305|protein_coding|*', prioritize_canonical=True)
|
|
60
|
+
assert ann.gene == 'OR4F5'
|
|
61
|
+
assert ann.transcript == 'ENST00000335137'
|
|
62
|
+
assert ann.aa_change == "F/C", ann.aa_change
|
|
63
|
+
assert ann.consequences == ['missense_variant']
|
|
64
|
+
assert ann.coding
|
|
65
|
+
assert ann.biotype == "protein_coding"
|
|
66
|
+
assert ann.severity == 2
|
|
67
|
+
assert ann.impact_severity == "MED", ann.impact_severity
|
|
68
|
+
assert not ann.is_pseudogene
|
|
69
|
+
assert ann.polyphen_score == 0.568, ann.polyphen
|
|
70
|
+
assert ann.polyphen_pred == "possibly_damaging", ann.polyphen
|
|
71
|
+
assert ann.sift_score == 0.0, ann.sift
|
|
72
|
+
assert ann.sift_pred == "deleterious", ann.sift
|
|
73
|
+
assert ann.is_canonical
|
|
74
|
+
|
|
75
|
+
def test_bcfts():
|
|
76
|
+
f = os.path.join(HERE, "bcfts.txt.gz")
|
|
77
|
+
with gzip.open(f, "rt") as fh:
|
|
78
|
+
for csq in (BCFT(l.rstrip()) for l in fh):
|
|
79
|
+
assert csq.severity in (1, 2, 3)
|
|
80
|
+
assert csq.is_pseudogene in (True, False)
|
|
81
|
+
assert csq.coding in (True, False), (csq.coding, csq)
|
|
82
|
+
assert csq.is_exonic in (True, False)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def test_veps():
|
|
86
|
+
|
|
87
|
+
f = os.path.join(HERE, "vep-csqs.txt.gz")
|
|
88
|
+
with gzip.open(f, "rt") as veps:
|
|
89
|
+
for csq in (VEP(l.strip()) for l in veps):
|
|
90
|
+
assert csq.severity in (1, 2, 3)
|
|
91
|
+
assert csq.is_pseudogene in (True, False)
|
|
92
|
+
assert csq.coding in (True, False)
|
|
93
|
+
assert isinstance(csq.polyphen_value, float) or csq.polyphen_value is None
|
|
94
|
+
csq.gene
|
|
95
|
+
assert isinstance(csq.sift_value, float) or csq.sift_value is None
|
|
96
|
+
|
|
97
|
+
def test_snpeffs():
|
|
98
|
+
f = os.path.join(HERE, "snpeff-anns.txt.gz")
|
|
99
|
+
with gzip.open(f, "rt") as anns:
|
|
100
|
+
for csq in (SnpEff(l.strip()) for l in anns):
|
|
101
|
+
assert csq.severity in (1, 2, 3)
|
|
102
|
+
assert csq.is_pseudogene in (True, False)
|
|
103
|
+
assert csq.coding in (True, False)
|
|
104
|
+
assert csq.polyphen_value is None
|
|
105
|
+
|
|
106
|
+
EFFECTS = [VEP("upstream_gene_variant|||ENSG00000223972|DDX11L1|ENST00000456328|||||processed_transcript"),
|
|
107
|
+
VEP("downstream_gene_variant|||ENSG00000227232|WASH7P|ENST00000488147|||||unprocessed_pseudogene"),
|
|
108
|
+
VEP("non_coding_exon_variant&nc_transcript_variant|||ENSG00000223972|DDX11L1|ENST00000456328|2/3||||processed_transcript"),
|
|
109
|
+
VEP("non_coding_exon_variant&nc_transcript_variant|||ENSG00000223972|DDX11L1|ENST00000456328|2/3||||processed_transcript"),
|
|
110
|
+
VEP("splice_region_variant&non_coding_exon_variant&nc_transcript_variant|||ENSG00000223972|DDX11L1|ENST00000456328|2/3||||processed_transcript"),
|
|
111
|
+
VEP("splice_region_variant&non_coding_exon_variant&nc_transcript_variant|||ENSG00000223972|DDX11L1|ENST00000456328|2/3||||processed_transcript"),
|
|
112
|
+
VEP("splice_region_variant&non_coding_exon_variant&nc_transcript_variant|||ENSG00000223972|DDX11L1|ENST00000456328|2/3||||processed_transcript"),
|
|
113
|
+
VEP("intron_variant&nc_transcript_variant|||ENSG00000223972|DDX11L1|ENST00000450305|||||transcribed_unprocessed_pseudogene"),
|
|
114
|
+
VEP("intron_variant&nc_transcript_variant|||ENSG00000223972|DDX11L1|ENST00000450305|||||transcribed_unprocessed_pseudogene"),
|
|
115
|
+
VEP('missense_variant|tTt/tGt|F/C|ENSG00000186092|OR4F5|ENST00000335137|1/1|possibly_damaging(0.568)|deleterious(0)|113/305|protein_coding'),
|
|
116
|
+
VEP("non_coding_exon_variant&nc_transcript_variant&feature_elongation|||ENSG00000223972|DDX11L1|ENST00000456328|3/3||||processed_transcript"),
|
|
117
|
+
]
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def test_order():
|
|
122
|
+
|
|
123
|
+
effects = sorted(EFFECTS)
|
|
124
|
+
assert effects[-1].impact_severity == "MED"
|
|
125
|
+
assert effects[0].impact_severity == "LOW"
|
|
126
|
+
|
|
127
|
+
def test_canonical_order():
|
|
128
|
+
effects = EFFECTS[:]
|
|
129
|
+
effects.append(VEP("intron_variant&nc_transcript_variant|||ENSG00000223972|DDX11L1|ENST00000450305|||||transcribed_unprocessed_pseudogene|*", prioritize_canonical=True))
|
|
130
|
+
effects = sorted(effects)
|
|
131
|
+
assert effects[-1].is_canonical
|
|
132
|
+
assert effects[0].impact_severity == "LOW"
|
|
133
|
+
assert not effects[0].is_canonical
|
|
134
|
+
|
|
135
|
+
def test_o2():
|
|
136
|
+
|
|
137
|
+
keys = [x.strip() for x in "Effect | Effect_Impact | Functional_Class | Codon_Change | Amino_Acid_change| Amino_Acid_length | Gene_Name | Gene_BioType | Coding | Transcript | Exon | ERRORS | WARNINGS".split("|")]
|
|
138
|
+
|
|
139
|
+
effects = [OldSnpEff(v, keys) for v in "DOWNSTREAM(MODIFIER|||||RP5-902P8.10|processed_transcript|NON_CODING|ENST00000434139|),DOWNSTREAM(MODIFIER|||||RP5-902P8.10|processed_transcript|NON_CODING|ENST00000453732|),INTRON(MODIFIER||||138|SCNN1D|protein_coding|CODING|ENST00000470022|3),INTRON(MODIFIER||||638|SCNN1D|protein_coding|CODING|ENST00000338555|3),INTRON(MODIFIER||||638|SCNN1D|protein_coding|CODING|ENST00000400928|2),INTRON(MODIFIER||||669|SCNN1D|protein_coding|CODING|ENST00000379110|6),INTRON(MODIFIER||||704|SCNN1D|protein_coding|CODING|ENST00000325425|2),INTRON(MODIFIER||||802|SCNN1D|protein_coding|CODING|ENST00000379116|5),INTRON(MODIFIER|||||SCNN1D|nonsense_mediated_decay|CODING|ENST00000379101|5),INTRON(MODIFIER|||||SCNN1D|processed_transcript|CODING|ENST00000467651|3)".split(",")]
|
|
140
|
+
|
|
141
|
+
effects = sorted(effects)
|
|
142
|
+
assert effects[-1].gene == "SCNN1D", effects[-1].gene
|
|
143
|
+
|
|
144
|
+
effects = sorted([OldSnpEff(v, keys) for v in "DOWNSTREAM(MODIFIER||||85|FAM138A|protein_coding|CODING|ENST00000417324|),DOWNSTREAM(MODIFIER|||||FAM138A|processed_transcript|CODING|ENST00000461467|),DOWNSTREAM(MODIFIER|||||MIR1302-10|miRNA|NON_CODING|ENST00000408384|),EXON(MODIFIER|||||MIR1302-10|antisense|NON_CODING|ENST00000469289|1),INTRON(MODIFIER|||||MIR1302-10|antisense|NON_CODING|ENST00000473358|1),UPSTREAM(MODIFIER|||||WASH7P|unprocessed_pseudogene|NON_CODING|ENST00000423562|),UPSTREAM(MODIFIER|||||WASH7P|unprocessed_pseudogene|NON_CODING|ENST00000430492|),UPSTREAM(MODIFIER|||||WASH7P|unprocessed_pseudogene|NON_CODING|ENST00000438504|),UPSTREAM(MODIFIER|||||WASH7P|unprocessed_pseudogene|NON_CODING|ENST00000488147|),UPSTREAM(MODIFIER|||||WASH7P|unprocessed_pseudogene|NON_CODING|ENST00000538476|)".split(",")])
|
|
145
|
+
s = "\n".join(e.effect_string for e in effects[::-1])
|
|
146
|
+
|
|
147
|
+
# reversed so that most significant is first
|
|
148
|
+
assert s == """\
|
|
149
|
+
DOWNSTREAM(MODIFIER||||85|FAM138A|protein_coding|CODING|ENST00000417324|
|
|
150
|
+
DOWNSTREAM(MODIFIER|||||FAM138A|processed_transcript|CODING|ENST00000461467|
|
|
151
|
+
INTRON(MODIFIER|||||MIR1302-10|antisense|NON_CODING|ENST00000473358|1
|
|
152
|
+
EXON(MODIFIER|||||MIR1302-10|antisense|NON_CODING|ENST00000469289|1
|
|
153
|
+
DOWNSTREAM(MODIFIER|||||MIR1302-10|miRNA|NON_CODING|ENST00000408384|
|
|
154
|
+
UPSTREAM(MODIFIER|||||WASH7P|unprocessed_pseudogene|NON_CODING|ENST00000423562|
|
|
155
|
+
UPSTREAM(MODIFIER|||||WASH7P|unprocessed_pseudogene|NON_CODING|ENST00000430492|
|
|
156
|
+
UPSTREAM(MODIFIER|||||WASH7P|unprocessed_pseudogene|NON_CODING|ENST00000438504|
|
|
157
|
+
UPSTREAM(MODIFIER|||||WASH7P|unprocessed_pseudogene|NON_CODING|ENST00000488147|
|
|
158
|
+
UPSTREAM(MODIFIER|||||WASH7P|unprocessed_pseudogene|NON_CODING|ENST00000538476|"""
|
|
159
|
+
|
|
160
|
+
def test_highest():
|
|
161
|
+
effects = sorted(EFFECTS)
|
|
162
|
+
|
|
163
|
+
top = Effect.top_severity(effects)
|
|
164
|
+
assert top.impact_severity == "MED"
|
|
165
|
+
assert top.so == "missense_variant"
|
|
166
|
+
#assert top[0].
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
effects.append(effects[-1])
|
|
170
|
+
|
|
171
|
+
top = Effect.top_severity(effects)
|
|
172
|
+
assert isinstance(top, list)
|
|
173
|
+
assert top[0].impact_severity == "MED"
|
|
174
|
+
|
|
175
|
+
def test_splice():
|
|
176
|
+
|
|
177
|
+
e = VEP('splice_acceptor_variant&intron_variant&feature_truncation|||ENSG00000221978|CCNL2|ENST00000408918||||-/226|protein_coding|1')
|
|
178
|
+
assert (e.is_coding, e.is_exonic, e.is_splicing) == (False, False, True)
|
|
179
|
+
|
|
180
|
+
e = VEP('intron_variant&feature_elongation|||ENSG00000187634|SAMD11|ENST00000341065||||-/589|protein_coding|1')
|
|
181
|
+
assert (e.is_coding, e.is_exonic, e.is_splicing) == (False, False, False)
|
|
182
|
+
|
|
183
|
+
def test_eff_splice():
|
|
184
|
+
|
|
185
|
+
keys = [x.strip() for x in "Effect | Effect_Impact | Functional_Class | Codon_Change | Amino_Acid_change| Amino_Acid_length | Gene_Name | Gene_BioType | Coding | Transcript | Exon | ERRORS | WARNINGS".split("|")]
|
|
186
|
+
e = OldSnpEff("SPLICE_SITE_REGION+SYNONYMOUS_CODING(LOW|SILENT|acG/acA|T245|1134|ANKS1A|protein_coding|CODING|ENST00000360359|5|A)", keys)
|
|
187
|
+
assert e.aa_change == "T245"
|
|
188
|
+
# note that we choose splice_site_region over synonymous coding
|
|
189
|
+
assert e.is_splicing, e.is_splicing
|
|
190
|
+
|
|
191
|
+
assert not e.is_coding
|
|
192
|
+
|
|
193
|
+
e = OldSnpEff("intergenic_region(MODIFIER|||n.null_nulldelAAGGAAGG|||||||A",
|
|
194
|
+
keys)
|
|
195
|
+
assert e.consequences != []
|
|
196
|
+
|
|
197
|
+
def test_regr():
|
|
198
|
+
keys = [x.strip() for x in 'Effect | Effect_Impact | Functional_Class | Codon_Change | Amino_Acid_change| Amino_Acid_length | Gene_Name | Transcript_BioType | Gene_Coding | Transcript_ID | Exon_Rank | Genotype_Number | ERRORS | WARNINGS'.split("|")]
|
|
199
|
+
v = OldSnpEff('SPLICE_SITE_REGION+SYNONYMOUS_CODING(LOW|SILENT|acG/acA|T245|1134|ANKS1A|protein_coding|CODING|ENST00000360359|5|A)', keys)
|
|
200
|
+
assert v.consequences == ['splice_region_variant', 'synonymous_variant'], v.consequences
|
|
201
|
+
assert v.severity == 2, v.severity
|
|
202
|
+
assert v.aa_change == 'T245'
|
|
203
|
+
v = OldSnpEff('UPSTREAM(MODIFIER||2771|||PSMB1|processed_transcript|CODING|ENST00000462957||C)', keys)
|
|
204
|
+
assert v.consequences == ['upstream_gene_variant'], v.consequences
|
|
205
|
+
assert v.severity == 1, v.severity
|
|
206
|
+
|
|
207
|
+
v = OldSnpEff('NEXT_PROT[maturation_peptide](LOW||||241|PSMB1|protein_coding|CODING|||C)', keys)
|
|
208
|
+
assert v.consequences == ['NEXT_PROT[maturation_peptide]'], v.consequences
|
|
209
|
+
assert v.severity == 1, v.severity
|
|
210
|
+
|
|
211
|
+
assert v <= v
|
|
212
|
+
|
|
213
|
+
def test_aa_change():
|
|
214
|
+
|
|
215
|
+
eff = OldSnpEff('NON_SYNONYMOUS_CODING(MODERATE|MISSENSE|Agc/Ggc|S418G|696|C1orf170|protein_coding|CODING|ENST00000433179|3|C)')
|
|
216
|
+
assert eff.aa_change == 'S418G'
|
|
217
|
+
ann = SnpEff('C|missense_variant|MODERATE|C1orf170|ENSG00000187642|transcript|ENST00000433179|protein_coding|3/5|c.1252A>G|p.Ser418Gly|1252/3064|1252/2091|418/696||')
|
|
218
|
+
assert ann.aa_change == 'p.Ser418Gly'
|
|
219
|
+
|
|
220
|
+
def test_old():
|
|
221
|
+
keys = [x.strip() for x in 'Effect | Effect_Impact | Functional_Class | Codon_Change | Amino_Acid_change| Amino_Acid_length | Gene_Name | Transcript_BioType | Gene_Coding | Transcript_ID | Exon_Rank | Genotype_Number | ERRORS | WARNINGS'.split("|")]
|
|
222
|
+
v = OldSnpEff('SPLICE_SITE_REGION+SYNONYMOUS_CODING(LOW|SILENT|acG/acA|T245|1134|ANKS1A|protein_coding|CODING|ENST00000360359|5|A)', keys)
|
|
223
|
+
assert v.so == "splice_region_variant", v.so
|
|
224
|
+
v = OldSnpEff('SYNONYMOUS_CODING+SPLICE_SITE_REGION(LOW|SILENT|acG/acA|T245|1134|ANKS1A|protein_coding|CODING|ENST00000360359|5|A)', keys)
|
|
225
|
+
assert v.so == "splice_region_variant", v.so
|
|
226
|
+
assert v.aa_length == 1134, v.aa_length
|
|
227
|
+
assert v.exon == "5", v.exon
|
|
228
|
+
assert v.codon_change == "acG/acA", v.codon_change
|
|
229
|
+
assert v.transcript == "ENST00000360359", v.transcript
|
|
230
|
+
|
|
231
|
+
def test_old2():
|
|
232
|
+
keys = [x.strip() for x in 'Effect | Effect_Impact | Functional_Class | Codon_Change | Amino_Acid_change| Amino_Acid_length | Gene_Name | Transcript_BioType | Gene_Coding | Transcript_ID | Exon_Rank | Genotype_Number | ERRORS | WARNINGS'.split("|")]
|
|
233
|
+
v = OldSnpEff('SPLICE_SITE_REGION+NON_SYNONYMOUS_CODING(LOW|SILENT|acG/acA|T245|1134|ANKS1A|protein_coding|CODING|ENST00000360359|5|A)', keys)
|
|
234
|
+
assert v.so == "missense_variant", v.so
|
|
235
|
+
|
|
236
|
+
def test_weird_vep():
|
|
237
|
+
keys = "Consequence|Codons|Amino_acids|Gene|SYMBOL|Feature|EXON|PolyPhen|SIFT|Protein_position|BIOTYPE|CANONICAL|CCDS|RadialSVM_score|RadialSVM_pred|LR_score|LR_pred|CADD_raw|CADD_phred|Reliability_index".split("|")
|
|
238
|
+
|
|
239
|
+
csqs = ["?|||117581|TWIST2|NM_001271893.1|1/1||||protein_coding|YES||||||||,non_coding_transcript_exon_variant&non_coding_transcript_variant|||117581|TWIST2|NM_001271893.1_dupl8|1/1||||mRNA|||||||||",
|
|
240
|
+
"non_coding_transcript_exon_variant&non_coding_transcript_variant|||117581|TWIST2|NM_001271893.1_dupl8|1/1||||mRNA|||||||||,?|||117581|TWIST2|NM_001271893.1|1/1||||protein_coding|YES||||||||",
|
|
241
|
+
"?|||115286|SLC25A26|NM_173471.3|1/1||||protein_coding|YES||||||||",
|
|
242
|
+
|
|
243
|
+
"|||ENSG00000138190|EXOC6|ENST00000260762||||-/804|protein_coding,|||ENSG00000138190|EXOC6|ENST00000371547||||-/820|protein_coding,|||ENSG00000138190|EXOC6|ENST00000443748||||-/701|protein_coding,NMD_transcript_variant|||ENSG00000138190|EXOC6|ENST00000495132||||-/404|nonsense_mediated_decay,|||ENSG00000138190|EXOC6|ENST00000371552||||-/799|protein_coding",
|
|
244
|
+
"|||ENSG00000013503|POLR3B|ENST00000539066||||-/1075|protein_coding,nc_transcript_variant|||ENSG00000013503|POLR3B|ENST00000549195|||||processed_transcript,|||ENSG00000013503|POLR3B|ENST00000549569||||-/170|protein_coding,|||ENSG00000013503|POLR3B|ENST00000228347||||-/1133|",
|
|
245
|
+
"|||ENSG00000147202|DIAPH2|ENST00000373054||||-/1097|protein_coding,|||ENSG00000147202|DIAPH2|ENST00000355827||||-/1096|protein_coding,|||ENSG00000147202|DIAPH2|ENST00000324765||||-/1101|protein_coding,|||ENSG00000147202|DIAPH2|ENST00000373049||||-/1096|protein_coding,|||ENSG00000147202|DIAPH2|ENST00000373061||||-/1101|protein_coding",
|
|
246
|
+
|
|
247
|
+
]
|
|
248
|
+
import sys
|
|
249
|
+
for cs in csqs:
|
|
250
|
+
for c in cs.split(","):
|
|
251
|
+
v = VEP(c, keys)
|
|
252
|
+
assert v.impact_severity in ('LOW', 'MEDIUM', 'HIGH')
|
|
253
|
+
|
|
254
|
+
def test_empty_snpeff():
|
|
255
|
+
|
|
256
|
+
keys = [x.strip() for x in 'Effect | Effect_Impact | Functional_Class | Codon_Change | Amino_Acid_change| Amino_Acid_length | Gene_Name | Transcript_BioType | Gene_Coding | Transcript_ID | Exon_Rank | Genotype_Number | ERRORS | WARNINGS'.split("|")]
|
|
257
|
+
|
|
258
|
+
eff = "(MODIFIER||||||||||A|ERROR_CHROMOSOME_NOT_FOUND)"
|
|
259
|
+
v = OldSnpEff(eff, keys)
|
|
260
|
+
assert v.impact_severity == "LOW", v.impact_severity
|
|
261
|
+
|
|
262
|
+
def test_protein_contact():
|
|
263
|
+
ann = SnpEff('C|protein_protein_contact|HIGH|C1orf170|ENSG00000187642|transcript|ENST00000433179|protein_coding|3/5|c.1252A>G|p.Ser418Gly|1252/3064|1252/2091|418/696||')
|
|
264
|
+
assert ann.impact_severity == "HIGH"
|
|
265
|
+
|
|
266
|
+
def test_gemini_issue812():
|
|
267
|
+
ann = VEP('protein_altering_variant|caGCAGCAGCAGCAGCAACAGCAG/caA|QQQQQQQQ/Q|ENSG00000204842|ATXN2|ENST00000608853|1/25|||14-21/1153|protein_coding|', keys="Consequence|Codons|Amino_acids|Gene|SYMBOL|Feature|EXON|PolyPhen|SIFT|Protein_position|BIOTYPE|CANONICAL".split("|"))
|
|
268
|
+
assert ann.is_coding
|
|
269
|
+
|
|
270
|
+
def test_bug_vcf2db_21():
|
|
271
|
+
ann = VEP('synonymous_variant|tcA/tcG|S|ENSG00000186092|OR4F5|ENST00000335137|1/1|||60/305|protein_coding||Low_complexity_(Seg):seg&Transmembrane_helices:TMhelix&Prints_domain:PR00237&Superfamily_domains:SSF81321&Gene3D:1.20.1070.10&hmmpanther:PTHR26451&hmmpanther:PTHR26451:SF72&PROSITE_profiles:PS50262||||ENST00000335137.3:c.180A>G|ENST00000335137.3:c.180A>G(p.%3D)|||-0.817044|0.039', keys="Consequence|Codons|Amino_acids|Gene|SYMBOL|Feature|EXON|PolyPhen|SIFT|Protein_position|BIOTYPE|CANONICAL|DOMAINS|CLIN_SIG".split("|"))
|
|
272
|
+
|
|
273
|
+
assert ann.codon_change == "tcA/tcG", ann.codon_change
|
|
274
|
+
|
|
275
|
+
def test_32():
|
|
276
|
+
keys = "Allele|Consequence|IMPACT|SYMBOL|Gene|Feature_type|Feature|BIOTYPE|EXON|INTRON|HGVSc|HGVSp|cDNA_position|CDS_position|Protein_position|Amino_acids|Codons|Existing_variation|DISTANCE|STRAND|FLAGS|VARIANT_CLASS|SYMBOL_SOURCE|HGNC_ID|CANONICAL|TSL|APPRIS|CCDS|ENSP|SWISSPROT|TREMBL|UNIPARC|REFSEQ_MATCH|SOURCE|GIVEN_REF|USED_REF|GENE_PHENO|SIFT|PolyPhen|DOMAINS|HGVS_OFFSET|AF|AFR_AF|AMR_AF|EAS_AF|EUR_AF|SAS_AF|AA_AF|EA_AF|gnomAD_AF|gnomAD_AFR_AF|gnomAD_AMR_AF|gnomAD_ASJ_AF|gnomAD_EAS_AF|gnomAD_FIN_AF|gnomAD_NFE_AF|gnomAD_OTH_AF|gnomAD_SAS_AF|MAX_AF|MAX_AF_POPS|CLIN_SIG|SOMATIC|PHENO|PUBMED|MOTIF_NAME|MOTIF_POS|HIGH_INF_POS|MOTIF_SCORE_CHANGE|MaxEntScan_alt|MaxEntScan_diff|MaxEntScan_ref|SpliceRegion".split("|")
|
|
277
|
+
s = "-|frameshift_variant&start_lost&start_retained_variant|HIGH|HRNR|ENSG00000197915|Transcript|ENST00000368801|protein_coding|2/3||ENST00000368801.2:c.1del|ENSP00000357791.2:p.Met1?|77/9623|1/8553|1/2850|M/X|Atg/tg|rs34061715&COSM111478||-1||deletion|HGNC|HGNC:20846|YES|1|P1|CCDS30859.1|ENSP00000357791|Q86YZ3||UPI00001D7CAD||Ensembl|T|T||||||0.874|0.7337|0.8818|0.9544|0.9592|0.8875|||0.9028|0.7227|0.8276|0.9554|0.9063|0.9541|0.9411|0.9142|0.9069|0.9592|EUR||0&1|0&1|||||||||,-|intron_variant&non_coding_transcript_variant|MODIFIER|FLG-AS1|ENSG00000237975|Transcript|ENST00000420707|antisense_RNA||1/8|ENST00000420707.5:n.159-25632del|||||||rs34061715&COSM111478||1||deletion|HGNC|HGNC:27913||5||||||||Ensembl|T|T|||||10|0.874|0.7337|0.8818|0.9544|0.9592|0.8875|||0.9028|0.7227|0.8276|0.9554|0.9063|0.9541|0.9411|0.9142|0.9069|0.9592|EUR||0&1|0&1|||||||||,-|intron_variant&non_coding_transcript_variant|MODIFIER|FLG-AS1|ENSG00000237975|Transcript|ENST00000593011|antisense_RNA||1/3|ENST00000593011.5:n.296+54843del|||||||rs34061715&COSM111478||1||deletion|HGNC|HGNC:27913||4||||||||Ensembl|T|T|||||10|0.874|0.7337|0.8818|0.9544|0.9592|0.8875|||0.9028|0.7227|0.8276|0.9554|0.9063|0.9541|0.9411|0.9142|0.9069|0.9592|EUR||0&1|0&1|||||||||,-|frameshift_variant&start_lost&start_retained_variant|HIGH|HRNR|388697|Transcript|NM_001009931.2|protein_coding|2/3||NM_001009931.2:c.1del|NP_001009931.1:p.Met1?|80/9632|1/8553|1/2850|M/X|Atg/tg|rs34061715&COSM111478||-1||deletion|EntrezGene|HGNC:20846|YES||||NP_001009931.1||||rseq_mrna_match|RefSeq|T|T||||||0.874|0.7337|0.8818|0.9544|0.9592|0.8875|||0.9028|0.7227|0.8276|0.9554|0.9063|0.9541|0.9411|0.9142|0.9069|0.9592|EUR||0&1|0&1|||||||||".split(",")
|
|
278
|
+
for e in s:
|
|
279
|
+
eff = VEP(e, keys=keys)
|
|
280
|
+
if not "intron" in e.lower():
|
|
281
|
+
assert eff.impact_severity == "HIGH", (eff.impact_severity, e)
|
|
Binary file
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: geneimpacts
|
|
3
|
+
Version: 0.3.8
|
|
4
|
+
Summary: normalize effects from variant annotation tools (snpEff, VEP)
|
|
5
|
+
Author: Brent Pedersen
|
|
6
|
+
Author-email: bpederse@gmail.com
|
|
7
|
+
Classifier: Development Status :: 4 - Beta
|
|
8
|
+
Classifier: Intended Audience :: Science/Research
|
|
9
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
10
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
12
|
+
License-File: LICENSE
|
|
13
|
+
Dynamic: author
|
|
14
|
+
Dynamic: author-email
|
|
15
|
+
Dynamic: classifier
|
|
16
|
+
Dynamic: description
|
|
17
|
+
Dynamic: description-content-type
|
|
18
|
+
Dynamic: license-file
|
|
19
|
+
Dynamic: summary
|
|
20
|
+
|
|
21
|
+
Given multiple snpEff or VEP or BCFTools consequence annotations for a single variant, get an orderable python object for each annotation.
|
|
22
|
+
|
|
23
|
+
[](https://travis-ci.org/brentp/geneimpacts)
|
|
24
|
+
|
|
25
|
+
This is to provide a consistent interface to
|
|
26
|
+
different variant annotations such as from [snpEff ANN field](http://snpeff.sourceforge.net/) and the [VEP CSQ field](http://www.ensembl.org/info/docs/tools/vep/index.html).
|
|
27
|
+
and the [BCFTools consequence field](http://biorxiv.org/content/early/2016/12/01/090811)
|
|
28
|
+
|
|
29
|
+
This will be used in [gemini](http://gemini.rtfd.org/) but should also be of
|
|
30
|
+
general utility.
|
|
31
|
+
|
|
32
|
+
Design
|
|
33
|
+
======
|
|
34
|
+
|
|
35
|
+
There is an effect base-class and then a sub-class for `snpEff`, one for `VEP`, and one for `BCFT`
|
|
36
|
+
|
|
37
|
+
`Effect` objects are orderable (via \_\_le\_\_ ) and should have an \_\_eq\_\_ method so that we can use [functools.total_ordering](https://docs.python.org/2/library/functools.html#functools.total_ordering) to provide the other comparison operators.
|
|
38
|
+
|
|
39
|
+
Given 2 effects objects, `a` and `b`: `a < b == True` iff the *severity* of `b` is greater than `a`.
|
|
40
|
+
|
|
41
|
+
We will have a classmethod: `Effect.top_severity([eff1, ... effn]) that will return the single highest severity if that exists or
|
|
42
|
+
a list of the ties for highest
|
|
43
|
+
|
|
44
|
+
Rules for severity:
|
|
45
|
+
===================
|
|
46
|
+
|
|
47
|
+
Given 2 annotations, *a* and *b*
|
|
48
|
+
*a* is more severe than *b* if:
|
|
49
|
+
|
|
50
|
+
1. *b* is a pseudogene and *a* is not
|
|
51
|
+
2. *a* is coding and *b* is not
|
|
52
|
+
3. *a* has higher severity than *b* ( see below)
|
|
53
|
+
4. polyphen, then sift
|
|
54
|
+
5. ??? transcript length? (we dont have access to this).
|
|
55
|
+
|
|
56
|
+
severity
|
|
57
|
+
--------
|
|
58
|
+
|
|
59
|
+
Severity is based on the [impacts from VEP](http://uswest.ensembl.org/info/docs/tools/vep/script/vep_other.html#pick)
|
|
60
|
+
and the [impacts from snpEff](http://snpeff.sourceforge.net/VCFannotationformat_v1.0.pdf). We reduce from the 4 categories HIGH, MODERATE, LOW, MODIFIER to 3 by renaming MODERATE to MED and renaming MODIFIER to LOW.
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
geneimpacts/__init__.py,sha256=roiMgKQm079MwmQ1fx0SmxBjGEFVnUMAWqYE_ozMil4,80
|
|
2
|
+
geneimpacts/effect.py,sha256=-Iqq2LLrGLse8sSr1NeuK1LB3Mciwmul8NuA8ud-aWY,24669
|
|
3
|
+
geneimpacts/tests/__init__.py,sha256=MsSFjiLMLJZ7QhUPpVBWKiyDnCzryquRyr329NoCACI,2
|
|
4
|
+
geneimpacts/tests/bcfts.txt.gz,sha256=0OiOw1XWu_s5KlD3Xy6QQnfXIGnITcWE-7S5WXTyMoM,6684
|
|
5
|
+
geneimpacts/tests/snpeff-anns.txt.gz,sha256=Z0vmJLAfqwK_DHEE2wbAmE0DGpUR_P_yOYg2dxggoR4,6930
|
|
6
|
+
geneimpacts/tests/test_impacts.py,sha256=XbCuVPqNGBJF8BT5pc255w3vP55D2AcQaXLmPSNh0d8,19622
|
|
7
|
+
geneimpacts/tests/vep-csqs.txt.gz,sha256=r7Cbf6xK3vKJGbOWq1ci-4_37fQKavFU0VlIxUixI2Q,6220
|
|
8
|
+
geneimpacts-0.3.8.dist-info/licenses/LICENSE,sha256=tgqe_a6fIQcB41fHR_3DxJL6Y3xJ7UYqc-JsQKxo400,1099
|
|
9
|
+
geneimpacts-0.3.8.dist-info/METADATA,sha256=KUOrCg-Pz-fjO_2HCwxnubG-dis0KJAxsbk-apYFe-w,2559
|
|
10
|
+
geneimpacts-0.3.8.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
11
|
+
geneimpacts-0.3.8.dist-info/top_level.txt,sha256=zEvuIfEuzZBG93GQ7w7gHnbzBxF38FyDc0WQ7_lBgVI,12
|
|
12
|
+
geneimpacts-0.3.8.dist-info/RECORD,,
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
The MIT License (MIT)
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2015 Brent Pedersen - Bioinformatics
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
22
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
geneimpacts
|