geneimpacts 0.3.8__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,3 @@
1
+ from .effect import Effect, SnpEff, VEP, OldSnpEff, BCFT
2
+
3
+ __version__ = "0.3.8"
geneimpacts/effect.py ADDED
@@ -0,0 +1,647 @@
1
+ from __future__ import print_function
2
+ import sys
3
+
4
+ from functools import total_ordering
5
+ import re
6
+ import itertools as it
7
+ try:
8
+ izip = it.izip
9
+ except AttributeError:
10
+ izip = zip
11
+ basestring = str
12
+
13
+ old_snpeff_effect_so = {'CDS': 'coding_sequence_variant',
14
+ 'CODON_CHANGE': 'coding_sequence_variant',
15
+ 'CODON_CHANGE_PLUS_CODON_DELETION': 'disruptive_inframe_deletion',
16
+ 'CODON_CHANGE_PLUS_CODON_INSERTION': 'disruptive_inframe_insertion',
17
+ 'CODON_DELETION': 'inframe_deletion',
18
+ 'CODON_INSERTION': 'inframe_insertion',
19
+ 'DOWNSTREAM': 'downstream_gene_variant',
20
+ 'EXON': 'exon_variant',
21
+ 'EXON_DELETED': 'exon_loss_variant',
22
+ 'FRAME_SHIFT': 'frameshift_variant',
23
+ 'GENE': 'gene_variant',
24
+ 'INTERGENIC': 'intergenic_variant',
25
+ 'INTERGENIC_REGION': 'intergenic_region',
26
+ 'INTERGENIC_CONSERVED': 'conserved_intergenic_variant',
27
+ 'INTRAGENIC': 'intragenic_variant',
28
+ 'INTRON': 'intron_variant',
29
+ 'INTRON_CONSERVED': 'conserved_intron_variant',
30
+ 'NON_SYNONYMOUS_CODING': 'missense_variant',
31
+ 'RARE_AMINO_ACID': 'rare_amino_acid_variant',
32
+ 'SPLICE_SITE_ACCEPTOR': 'splice_acceptor_variant',
33
+ 'SPLICE_SITE_DONOR': 'splice_donor_variant',
34
+ 'SPLICE_SITE_REGION': 'splice_region_variant',
35
+ #'START_GAINED': '5_prime_UTR_premature_start_codon_gain_variant',
36
+ 'START_GAINED': '5_prime_UTR_premature_start_codon_variant',
37
+ 'START_LOST': 'start_lost',
38
+ 'STOP_GAINED': 'stop_gained',
39
+ 'STOP_LOST': 'stop_lost',
40
+ 'SYNONYMOUS_CODING': 'synonymous_variant',
41
+ 'SYNONYMOUS_START': 'start_retained_variant',
42
+ 'SYNONYMOUS_STOP': 'stop_retained_variant',
43
+ 'TRANSCRIPT': 'transcript_variant',
44
+ 'UPSTREAM': 'upstream_gene_variant',
45
+ 'UTR_3_DELETED': '3_prime_UTR_truncation_+_exon_loss_variant',
46
+ 'UTR_3_PRIME': '3_prime_UTR_variant',
47
+ 'UTR_5_DELETED': '5_prime_UTR_truncation_+_exon_loss_variant',
48
+ 'UTR_5_PRIME': '5_prime_UTR_variant',
49
+ 'NON_SYNONYMOUS_START': 'initiator_codon_variant',
50
+ 'NONE': 'None',
51
+ 'CHROMOSOME_LARGE_DELETION': 'chromosomal_deletion'}
52
+
53
+ old_snpeff_lookup = {'CDS': 'LOW',
54
+ 'CHROMOSOME_LARGE_DELETION': 'HIGH',
55
+ 'CODON_CHANGE': 'MED',
56
+ 'CODON_CHANGE_PLUS_CODON_DELETION': 'MED',
57
+ 'CODON_CHANGE_PLUS_CODON_INSERTION': 'MED',
58
+ 'CODON_DELETION': 'MED',
59
+ 'CODON_INSERTION': 'MED',
60
+ 'DOWNSTREAM': 'LOW',
61
+ 'EXON': 'LOW',
62
+ 'EXON_DELETED': 'HIGH',
63
+ 'FRAME_SHIFT': 'HIGH',
64
+ 'GENE': 'LOW',
65
+ 'INTERGENIC': 'LOW',
66
+ 'INTERGENIC_CONSERVED': 'LOW',
67
+ 'INTRAGENIC': 'LOW',
68
+ 'INTRON': 'LOW',
69
+ 'INTRON_CONSERVED': 'LOW',
70
+ 'NONE': 'LOW',
71
+ 'NON_SYNONYMOUS_CODING': 'MED',
72
+ 'NON_SYNONYMOUS_START': 'HIGH',
73
+ 'RARE_AMINO_ACID': 'HIGH',
74
+ 'SPLICE_SITE_ACCEPTOR': 'HIGH',
75
+ 'SPLICE_SITE_DONOR': 'HIGH',
76
+ 'SPLICE_SITE_REGION': 'MED',
77
+ 'START_GAINED': 'LOW',
78
+ 'START_LOST': 'HIGH',
79
+ 'STOP_GAINED': 'HIGH',
80
+ 'STOP_LOST': 'HIGH',
81
+ 'SYNONYMOUS_CODING': 'LOW',
82
+ 'SYNONYMOUS_START': 'LOW',
83
+ 'SYNONYMOUS_STOP': 'LOW',
84
+ 'TRANSCRIPT': 'LOW',
85
+ 'UPSTREAM': 'LOW',
86
+ 'UTR_3_DELETED': 'MED',
87
+ 'UTR_3_PRIME': 'LOW',
88
+ 'UTR_5_DELETED': 'MED',
89
+ 'UTR_5_PRIME': 'LOW'}
90
+ # http://uswest.ensembl.org/info/genome/variation/predicted_data.html#consequences
91
+ IMPACT_SEVERITY = [
92
+ ('chromosome_number_variation', 'HIGH'), # snpEff
93
+ ('transcript_ablation', 'HIGH'), # VEP
94
+ ('exon_loss_variant', 'HIGH'), # snpEff
95
+ ('exon_loss', 'HIGH'), # snpEff
96
+ ('rare_amino_acid_variant', 'HIGH'),
97
+ ('protein_protein_contact', 'HIGH'), # snpEff
98
+ ('structural_interaction_variant', 'HIGH'), #snpEff
99
+ ('feature_fusion', 'HIGH'), #snpEff
100
+ ('bidirectional_gene_fusion', 'HIGH'), #snpEff
101
+ ('gene_fusion', 'HIGH'), #snpEff
102
+ ('feature_ablation', 'HIGH'), #snpEff, structural varint
103
+ ('splice_acceptor_variant', 'HIGH'), # VEP
104
+ ('splice_donor_variant', 'HIGH'), # VEP
105
+ ('start_retained_variant', 'HIGH'), # new VEP
106
+ ('stop_gained', 'HIGH'), # VEP
107
+ ('frameshift_variant', 'HIGH'), # VEP
108
+ ('stop_lost', 'HIGH'), # VEP
109
+ ('start_lost', 'HIGH'), # VEP
110
+ ('transcript_amplification', 'HIGH'), # VEP
111
+
112
+
113
+ ('disruptive_inframe_deletion', 'MED'), #snpEff
114
+ ('conservative_inframe_deletion', 'MED'), #snpEff
115
+ ('disruptive_inframe_insertion', 'MED'), #snpEff
116
+ ('conservative_inframe_insertion', 'MED'), #snpEff
117
+ ('duplication', 'MED'), # snpEff, structural variant
118
+ ('inversion', 'MED'), # snpEff, structural variant
119
+ ('exon_region', 'MED'), # snpEff, structural variant
120
+ ('inframe_insertion', 'MED'), # VEP
121
+ ('inframe_deletion', 'MED'), # VEP
122
+ ('missense_variant', 'MED'), # VEP
123
+ ('protein_altering_variant', 'MED'), # VEP
124
+ ('initiator_codon_variant', 'MED'), # snpEff
125
+ ('regulatory_region_ablation', 'MED'), # VEP
126
+
127
+ ('5_prime_UTR_truncation', 'MED'), # found in snpEff
128
+ ('splice_donor_5th_base_variant', 'MED'), # VEP changed to have medium priority
129
+ ('splice_region_variant', 'MED'), # VEP changed to have medium priority
130
+ ('splice_donor_region_variant', 'MED'), # VEP changed to have medium priority
131
+ ('splice_polypyrimidine_tract_variant', 'MED'), # VEP changed to have medium priority
132
+
133
+ ('3_prime_UTR_truncation', 'LOW'), # found in snpEff
134
+ ('non_canonical_start_codon', 'LOW'), # found in snpEff
135
+
136
+ ('synonymous_variant', 'LOW'), # VEP
137
+ ('coding_sequence_variant', 'LOW'), # VEP
138
+ ('incomplete_terminal_codon_variant', 'LOW'), # VEP
139
+ ('stop_retained_variant', 'LOW'), # VEP
140
+ ('mature_miRNA_variant', 'LOW'), # VEP
141
+ ('5_prime_UTR_premature_start_codon_variant', 'LOW'), # snpEff
142
+ ('5_prime_UTR_premature_start_codon_gain_variant', 'LOW'), #snpEff
143
+ ('5_prime_UTR_variant', 'LOW'), # VEP
144
+ ('3_prime_UTR_variant', 'LOW'), # VEP
145
+
146
+
147
+ ('non_coding_transcript_exon_variant', 'LOW'), # VEP
148
+ ('conserved_intron_variant', 'LOW'), # snpEff
149
+ ('intron_variant', 'LOW'), # VEP
150
+ ('exon_variant', 'LOW'), # snpEff
151
+ ('gene_variant', 'LOW'), # snpEff
152
+ ('NMD_transcript_variant', 'LOW'), # VEP
153
+ ('non_coding_transcript_variant', 'LOW'), # VEP
154
+ ('coding_transcript_variant', 'LOW'), # VEP
155
+ ('upstream_gene_variant', 'LOW'), # VEP
156
+ ('downstream_gene_variant', 'LOW'), # VEP
157
+ ('TFBS_ablation', 'LOW'), # VEP
158
+ ('TFBS_amplification', 'LOW'), # VEP
159
+ ('TF_binding_site_variant', 'LOW'), # VEP
160
+ ('regulatory_region_amplification', 'LOW'), # VEP
161
+ ('feature_elongation', 'LOW'), # VEP
162
+ ('sequence_variant', 'LOW'), # VEP
163
+ ('miRNA', 'LOW'), # snpEff
164
+ ('transcript_variant', 'LOW'), # snpEff
165
+ ('start_retained', 'LOW'), # snpEff
166
+ ('regulatory_region_variant', 'LOW'), # VEP
167
+ ('feature_truncation', 'LOW'), # VEP
168
+ ('non_coding_exon_variant', 'LOW'),
169
+ ('nc_transcript_variant', 'LOW'),
170
+ ('conserved_intergenic_variant', 'LOW'), # snpEff
171
+ ('intergenic_variant', 'LOW'), # VEP
172
+ ('intergenic_region', 'LOW'), # snpEff
173
+ ('intragenic_variant', 'LOW'), # snpEff
174
+ ('non_coding_transcript_exon_variant', 'LOW'), # snpEff
175
+ ('non_coding_transcript_variant', 'LOW'), # snpEff
176
+ ('transcript', 'LOW'), # ? snpEff older
177
+ ('sequence_feature', 'LOW'), # snpEff older
178
+ ('non_coding', 'LOW'), # BCSQ
179
+
180
+
181
+ ('?', 'UNKNOWN'), # some VEP annotations have '?'
182
+ ('', 'UNKNOWN'), # some VEP annotations have ''
183
+ ('UNKNOWN', 'UNKNOWN'), # some snpEFF annotations have 'unknown'
184
+ ]
185
+
186
+ # bcftools doesn't add _variant on the end.
187
+ for (csq, imp) in list(IMPACT_SEVERITY[::-1]):
188
+ if csq.endswith('_variant'):
189
+ for i, (a, b) in enumerate(IMPACT_SEVERITY):
190
+ if (a, b) == (csq, imp):
191
+ IMPACT_SEVERITY.insert(i, (csq[:-8].lower(), imp))
192
+ break
193
+
194
+ IMPACT_SEVERITY_ORDER = dict((x[0], i) for i, x in enumerate(IMPACT_SEVERITY[::-1]))
195
+ IMPACT_SEVERITY = dict(IMPACT_SEVERITY)
196
+
197
+ EXONIC_IMPACTS = set(["stop_gained",
198
+ "exon_variant",
199
+ "stop_lost",
200
+ "frameshift_variant",
201
+ "initiator_codon_variant",
202
+ "inframe_deletion",
203
+ "inframe_insertion",
204
+ "missense_variant",
205
+ "protein_altering_variant",
206
+ "incomplete_terminal_codon_variant",
207
+ "stop_retained_variant",
208
+ "5_prime_UTR_premature_start_codon_variant",
209
+ "synonymous_variant",
210
+ "coding_sequence_variant",
211
+ "5_prime_UTR_variant",
212
+ "3_prime_UTR_variant",
213
+ "transcript_ablation",
214
+ "transcript_amplification",
215
+ "feature_elongation",
216
+ "feature_truncation"])
217
+
218
+ for im in list(EXONIC_IMPACTS):
219
+ if im.endswith("_variant"):
220
+ EXONIC_IMPACTS.add(im[:-8])
221
+ EXONIC_IMPACTS = frozenset(EXONIC_IMPACTS)
222
+
223
+ def snpeff_aa_length(self):
224
+ try:
225
+ v = self.effects['AA.pos / AA.length']
226
+ if v.strip():
227
+ return int(v.split("/")[1].strip())
228
+ except:
229
+ try:
230
+ return int(self.effects['Amino_Acid_length'])
231
+ except:
232
+ return None
233
+
234
+ def vep_aa_length(self):
235
+ if not 'Protein_position' in self.effects:
236
+ return None
237
+ try:
238
+ return int(self.effects['Protein_position'])
239
+ except ValueError:
240
+ try:
241
+ return self.effects['Protein_position']
242
+ except KeyError:
243
+ return None
244
+
245
+ def vep_polyphen_pred(self):
246
+ try:
247
+ return self.effects['PolyPhen'].split('(')[0]
248
+ except (KeyError, IndexError):
249
+ return None
250
+
251
+ def vep_polyphen_score(self):
252
+ try:
253
+ return float(self.effects['PolyPhen'].split('(')[1][:-1])
254
+ except (KeyError, IndexError):
255
+ return None
256
+
257
+ def vep_sift_score(self):
258
+ try:
259
+ return float(self.effects['SIFT'].split("(")[1][:-1])
260
+ except (IndexError, KeyError):
261
+ return None
262
+
263
+ def vep_sift_pred(self):
264
+ try:
265
+ return self.effects['SIFT'].split("(")[0]
266
+ except (IndexError, KeyError):
267
+ return None
268
+
269
+ snpeff_lookup = {
270
+ 'transcript': ['Feature_ID', 'Transcript_ID', 'Transcript'],
271
+ 'gene': 'Gene_Name',
272
+ 'exon': ['Rank', 'Exon', 'Exon_Rank'],
273
+ 'codon_change': ['HGVS.c', 'Codon_Change'],
274
+ 'aa_change': ['HGVS.p', 'Amino_Acid_Change', 'Amino_Acid_change'],
275
+ 'aa_length': snpeff_aa_length,
276
+ 'biotype': ['Transcript_BioType', 'Gene_BioType'],
277
+ 'alt': 'Allele',
278
+ }
279
+
280
+ bcft_lookup = {}
281
+
282
+ vep_lookup = {
283
+ 'transcript': 'Feature',
284
+ 'gene': ['SYMBOL', 'HGNC', 'Gene'],
285
+ 'ensembl_gene_id': 'Gene',
286
+ 'exon': 'EXON',
287
+ 'codon_change': 'Codons',
288
+ 'aa_change': 'Amino_acids',
289
+ 'aa_length': vep_aa_length,
290
+ 'biotype': 'BIOTYPE',
291
+ 'polyphen_pred': vep_polyphen_pred,
292
+ 'polyphen_score': vep_polyphen_score,
293
+ 'sift_pred': vep_sift_pred,
294
+ 'sift_score': vep_sift_score,
295
+ 'alt': 'ALLELE',
296
+ }
297
+
298
+ # lookup here instead of returning ''.
299
+ defaults = {'gene': None}
300
+
301
+ @total_ordering
302
+ class Effect(object):
303
+ _top_consequence = None
304
+ lookup = None
305
+
306
+ def __init__(self, key, effect_dict, keys, prioritize_canonical):
307
+ raise NotImplemented
308
+
309
+ @classmethod
310
+ def new(self, key, effect_dict, keys):
311
+ lookup = {"CSQ": VEP, "ANN": SnpEff, "EFF": OldSnpEff, "BCSQ": BCFT}
312
+ assert key in lookup
313
+ return lookup[key](effect_dict, keys)
314
+
315
+ @property
316
+ def is_exonic(self):
317
+ return self.top_consequence in EXONIC_IMPACTS
318
+
319
+ def unused(self):
320
+ return []
321
+
322
+ @property
323
+ def top_consequence(self):
324
+ # sort by order and return the top
325
+ if self._top_consequence is None:
326
+ self._top_consequence = sorted([(IMPACT_SEVERITY_ORDER.get(c, 0), c) for c in
327
+ self.consequences], reverse=True)[0][1]
328
+ return self._top_consequence
329
+
330
+ @property
331
+ def so(self):
332
+ return self.top_consequence
333
+
334
+ @property
335
+ def is_coding(self):
336
+ return self.biotype == "protein_coding" and self.is_exonic and ("_UTR_" not in self.top_consequence)
337
+
338
+ @property
339
+ def is_splicing(self):
340
+ return "splice" in self.top_consequence
341
+
342
+ @property
343
+ def is_lof(self):
344
+ return self.biotype == "protein_coding" and self.impact_severity == "HIGH"
345
+
346
+ def __le__(self, other):
347
+ # we sort so that the effects with the highest impacts come last
348
+ # (highest) and so, we:
349
+ # + return true if self has lower impact than other.
350
+ # + return false if self has higher impact than other.
351
+ self_has_lower_impact = True
352
+ self_has_higher_impact = False
353
+
354
+ if self.prioritize_canonical:
355
+ scanon, ocanon = self.is_canonical, other.is_canonical
356
+ if scanon and not ocanon:
357
+ return self_has_higher_impact
358
+ elif ocanon and not scanon:
359
+ return self_has_lower_impact
360
+
361
+ spg = self.is_pseudogene
362
+ opg = other.is_pseudogene
363
+ if spg and not opg:
364
+ return self_has_lower_impact
365
+ elif opg and not spg:
366
+ return self_has_higher_impact
367
+
368
+ sc, oc = self.coding, other.coding
369
+ if sc and not oc:
370
+ # other is not coding. is is splicing?
371
+ # if other is splicing, we have lower impact.
372
+ if not (self.is_splicing or other.is_splicing):
373
+ return self_has_higher_impact
374
+ elif oc and not sc:
375
+ # self. is not coding. is it splicing?
376
+ # if self is splicing it has higher impact
377
+ if not (self.is_splicing or other.is_splicing):
378
+ return self_has_lower_impact
379
+
380
+ if self.severity != other.severity:
381
+ return self.severity <= other.severity
382
+
383
+ if self.biotype == "protein_coding" and not other.biotype == "protein_coding":
384
+ return False
385
+ elif other.biotype == "protein_coding" and not self.biotype == "protein_coding":
386
+ return True
387
+
388
+ if self.biotype == "processed_transcript" and not other.biotype == "processed_transcript":
389
+ return False
390
+ elif other.biotype == "processed_transcript" and not self.biotype == "processed_transcript":
391
+ return True
392
+
393
+ # sift higher == more damaing
394
+ if (self.sift_value or 10000) < (other.sift_value or 10000):
395
+ return True
396
+
397
+ # polyphen, lower == more damaging
398
+ if (self.polyphen_value or -10000) > (other.polyphen_value or -10000):
399
+ return True
400
+
401
+ return max(IMPACT_SEVERITY_ORDER.get(c, 0) for c in self.consequences) <= \
402
+ max(IMPACT_SEVERITY_ORDER.get(co, 0) for co in other.consequences)
403
+
404
+ @classmethod
405
+ def top_severity(cls, effects):
406
+ for i, e in enumerate(effects):
407
+ if isinstance(e, basestring):
408
+
409
+ effects[i] = cls(e)
410
+
411
+ if len(effects) == 0:
412
+ return None
413
+ if len(effects) == 1:
414
+ return effects[0]
415
+ effects = sorted(effects)
416
+ if effects[-1] > effects[-2]:
417
+ return effects[-1]
418
+ ret = [effects[-1], effects[-2]]
419
+ for i in range(-3, -(len(effects) - 1), -1):
420
+ if effects[-1] > effects[i]: break
421
+ ret.append(effects[i])
422
+ return ret
423
+
424
+ def __getitem__(self, key):
425
+ return self.effects[key]
426
+
427
+ def __eq__(self, other):
428
+ if not isinstance(other, Effect): return False
429
+ return self.effect_string == other.effect_string
430
+
431
+ def __str__(self):
432
+ return repr(self)
433
+
434
+ def __repr__(self):
435
+ return "%s(%s-%s, %s)" % (self.__class__.__name__, self.gene,
436
+ self.consequence, self.impact_severity)
437
+
438
+ @property
439
+ def effect_severity(self):
440
+ return self.impact_severity
441
+
442
+ @property
443
+ def lof(self):
444
+ return self.biotype == "protein_coding" and self.impact_severity == "HIGH"
445
+
446
+ @property
447
+ def severity(self, lookup={'HIGH': 3, 'MED': 2, 'LOW': 1, 'UNKNOWN': 0}, sev=IMPACT_SEVERITY):
448
+ # higher is more severe. used for ordering.
449
+ try:
450
+ v = max(lookup[sev[csq]] for csq in self.consequences)
451
+ except KeyError:
452
+ v = 0
453
+ if v == 0:
454
+ excl = []
455
+ for i, c in [(i, c) for i, c in enumerate(self.consequences) if not c in sev]:
456
+ sys.stderr.write("WARNING: unknown severity for '%s' with effect '%s'\n" % (self.effect_string, c))
457
+ sys.stderr.write("Please report this on github with the effect-string above\n")
458
+ excl.append(i)
459
+ if len(excl) == len(self.consequences):
460
+ v = 1
461
+ else:
462
+ v = max(lookup[sev[csq]] for i, csq in enumerate(self.consequences) if not i in excl)
463
+ return max(v, 1)
464
+
465
+ @property
466
+ def impact_severity(self):
467
+ return ['xxx', 'LOW', 'MED', 'HIGH'][self.severity]
468
+
469
+ @property
470
+ def consequence(self):
471
+ return self.top_consequence
472
+
473
+ @property
474
+ def is_pseudogene(self): #bool
475
+ return self.biotype is not None and 'pseudogene' in self.biotype
476
+
477
+
478
+ def __getattr__(self, k):
479
+ v = self.lookup.get(k)
480
+ if v is None: return v
481
+ if isinstance(v, basestring):
482
+ ret = self.effects.get(v)
483
+ # if we didn't get value, there may be a column
484
+ # specific value stored in defaults so we look import
485
+ # up.
486
+ if not ret and ret is not False:
487
+ return defaults.get(k, '')
488
+ return ret
489
+ elif isinstance(v, list):
490
+ for key in v:
491
+ try:
492
+ return self.effects[key]
493
+ except KeyError:
494
+ continue
495
+ return defaults.get(k, '')
496
+ return v(self)
497
+
498
+ class BCFT(Effect):
499
+ __slots__ = ('effect_string', 'effects', 'biotype', 'gene', 'transcript', 'aa_change', 'dna_change')
500
+ keys = "consequence,gene,transcript,biotype,strand,amino_acid_change,dna_change".split(",")
501
+ lookup = bcft_lookup
502
+
503
+ def __init__(self, effect_string, keys=None, prioritize_canonical=False):
504
+ if keys is not None: self.keys = keys
505
+ self.effect_string = effect_string
506
+ self.effects = dict(izip(self.keys, (x.strip().replace(' ', '_') for x in effect_string.split("|"))))
507
+ self.biotype = self.effects.get('biotype', None)
508
+ self.transcript = self.effects.get('transcript', None)
509
+ self.gene = self.effects.get('gene', None)
510
+ self.aa_change = self.effects.get('amino_acid_change', None)
511
+ self.consequences = self.effects[self.keys[0]].split('&')
512
+
513
+ def unused(self, used=frozenset("csq|gene|transcript|biotype|strand|aa_change|dna_change".lower().split("|"))):
514
+ """Return fields that were in the VCF but weren't utilized as part of the standard fields supported here."""
515
+ return [k for k in self.keys if not k.lower() in used]
516
+
517
+ @property
518
+ def exonic(self):
519
+ return self.biotype == "protein_coding" and any(csq in EXONIC_IMPACTS for csq in self.consequences)
520
+
521
+ @property
522
+ def coding(self):
523
+ # what about start/stop_gained?
524
+ return self.exonic and any(csq[1:] != "_prime_utr" for csq in self.consequences)
525
+
526
+
527
+ class VEP(Effect):
528
+ __slots__ = ('effect_string', 'effects', 'biotype')
529
+ keys = "Consequence|Codons|Amino_acids|Gene|SYMBOL|Feature|EXON|PolyPhen|SIFT|Protein_position|BIOTYPE|CANONICAL".split("|")
530
+ lookup = vep_lookup
531
+
532
+ def __init__(self, effect_string, keys=None, checks=True, prioritize_canonical=False):
533
+ if checks:
534
+ assert not "," in effect_string
535
+ assert not "=" in effect_string
536
+ self.effect_string = effect_string
537
+ if keys is not None: self.keys = keys
538
+
539
+ self.effect_string = effect_string
540
+ self.effects = dict(izip(self.keys, (x.strip() for x in effect_string.split("|"))))
541
+ self.biotype = self.effects.get('BIOTYPE', None)
542
+ self.prioritize_canonical = prioritize_canonical
543
+
544
+ @property
545
+ def consequences(self, _cache={}):
546
+ try:
547
+ # this is a bottleneck so we keep a cache
548
+ return _cache[self.effects['Consequence']]
549
+ except KeyError:
550
+ res = _cache[self.effects['Consequence']] = list(it.chain.from_iterable(x.split("+") for x in self.effects['Consequence'].split('&')))
551
+ return res
552
+
553
+ def unused(self, used=frozenset("Consequence|Codons|Amino_acids|Gene|SYMBOL|Feature|EXON|PolyPhen|SIFT|Protein_position|BIOTYPE|CANONICAL".lower().split("|"))):
554
+ """Return fields that were in the VCF but weren't utilized as part of the standard fields supported here."""
555
+ return [k for k in self.keys if not k.lower() in used]
556
+
557
+ @property
558
+ def coding(self):
559
+ # what about start/stop_gained?
560
+ return self.exonic and any(csq[1:] != "_prime_UTR_variant" for csq in self.consequences)
561
+
562
+ @property
563
+ def exonic(self):
564
+ return self.biotype == "protein_coding" and any(csq in EXONIC_IMPACTS for csq in self.consequences)
565
+
566
+ @property
567
+ def is_canonical(self):
568
+ return self.effects.get("CANONICAL", "") != ""
569
+
570
+ class SnpEff(Effect):
571
+ lookup = snpeff_lookup
572
+
573
+ __slots__ = ('effects', 'effect_string', 'biotype')
574
+
575
+ keys = [x.strip() for x in 'Allele | Annotation | Annotation_Impact | Gene_Name | Gene_ID | Feature_Type | Feature_ID | Transcript_BioType | Rank | HGVS.c | HGVS.p | cDNA.pos / cDNA.length | CDS.pos / CDS.length | AA.pos / AA.length | Distance | ERRORS / WARNINGS / INFO'.split("|")]
576
+
577
+ def __init__(self, effect_string, keys=None, prioritize_canonical=False):
578
+ assert not "," in effect_string
579
+ assert not "=" == effect_string[3]
580
+ self.effect_string = effect_string
581
+ if keys is not None:
582
+ self.keys = keys
583
+ self.effects = dict(izip(self.keys, (x.strip() for x in effect_string.split("|", len(self.keys)))))
584
+ self.biotype = self.effects['Transcript_BioType']
585
+
586
+ @property
587
+ def consequences(self):
588
+ return list(it.chain.from_iterable(x.split("+") for x in self.effects['Annotation'].split('&')))
589
+
590
+ @property
591
+ def coding(self):
592
+ # TODO: check start_gained and utr
593
+ return self.exonic and not "utr" in self.consequence and not "start_gained" in self.consequence
594
+
595
+
596
+ @property
597
+ def exonic(self):
598
+ csqs = self.consequence
599
+ if isinstance(csqs, basestring):
600
+ csqs = [csqs]
601
+ return any(csq in EXONIC_IMPACTS for csq in csqs) and self.effects['Transcript_BioType'] == 'protein_coding'
602
+
603
+
604
+ class OldSnpEff(SnpEff):
605
+
606
+ keys = [x.strip() for x in "Effect | Effect_Impact | Functional_Class | Codon_Change | Amino_Acid_change| Amino_Acid_length | Gene_Name | Gene_BioType | Coding | Transcript | Exon | ERRORS | WARNINGS".split("|")]
607
+
608
+ def __init__(self, effect_string, keys=None, _patt=re.compile(r"\||\("),
609
+ prioritize_canonical=False):
610
+ assert not "," in effect_string
611
+ assert not "=" in effect_string
612
+ effect_string = effect_string.rstrip(")")
613
+ self.effect_string = effect_string
614
+ if keys is not None:
615
+ self.keys = keys
616
+ self.effects = dict(izip(self.keys, (x.strip() for x in _patt.split(effect_string))))
617
+
618
+ @property
619
+ def consequence(self):
620
+ if '&' in self.effects['Effect']:
621
+ return self.effects['Effect'].split('&')
622
+ return self.effects['Effect']
623
+
624
+ @property
625
+ def consequences(self):
626
+ try:
627
+ return [old_snpeff_effect_so.get(c, old_snpeff_effect_so[c.upper()]) for c in it.chain.from_iterable(x.split("+") for x in
628
+ self.effects['Effect'].split('&'))]
629
+ except KeyError:
630
+ return list(it.chain.from_iterable(x.split("+") for x in self.effects['Effect'].split('&')))
631
+
632
+ @property
633
+ def severity(self, lookup={'HIGH': 3, 'MED': 2, 'LOW': 1}):
634
+ # higher is more severe. used for ordering.
635
+ try:
636
+ return max(lookup[old_snpeff_lookup[csq]] for csq in self.consequences)
637
+ except KeyError:
638
+ try:
639
+ #in between
640
+ sevs = [IMPACT_SEVERITY.get(csq, "LOW") for csq in self.consequences]
641
+ return max(lookup[s] for s in sevs)
642
+ except KeyError:
643
+ return Effect.severity.fget(self)
644
+
645
+ @property
646
+ def is_lof(self):
647
+ return self.biotype == "protein_coding" and self.impact_severity == "HIGH"
@@ -0,0 +1 @@
1
+ #
Binary file
Binary file
@@ -0,0 +1,281 @@
1
+ import sys
2
+ import os
3
+ import gzip
4
+ from geneimpacts import SnpEff, VEP, Effect, OldSnpEff, BCFT
5
+
6
+
7
+ HERE = os.path.dirname(__file__)
8
+
9
+ def test_bug():
10
+ e = sorted([VEP('missense_variant|tTt/tGt|F/C|ENSG00000186092|OR4F5|ENST00000335137|1/1|possibly_damaging(0.568)|deleterious(0)|113/305|protein_coding'),
11
+ VEP("splice_region_variant&non_coding_exon_variant&nc_transcript_variant|||ENSG00000223972|DDX11L1|ENST00000456328|2/3||||processed_transcript")])
12
+ assert e[-1].so == 'missense_variant', e[-1].so
13
+
14
+ def test_snpeff():
15
+
16
+ ann = SnpEff("C|splice_donor_variant&splice_region_variant&splice_region_variant&intron_variant|HIGH|DDX11L1|ENSG00000223972|transcript|ENST00000518655|transcribed_unprocessed_pseudogene|3/3|n.734+2_734+3delAG||||||")
17
+
18
+ assert ann.gene == "DDX11L1"
19
+ assert ann.transcript == "ENST00000518655"
20
+ assert ann.biotype == "transcribed_unprocessed_pseudogene", ann.biotype
21
+ assert ann.consequences == 'splice_donor_variant&splice_region_variant&splice_region_variant&intron_variant'.split('&')
22
+ assert ann.severity == 3
23
+ assert ann.impact_severity == "HIGH"
24
+ assert ann.aa_change == ""
25
+ assert ann.exon == '3/3', ann.exon
26
+ assert not ann.coding
27
+ assert ann.is_pseudogene
28
+
29
+ def test_unused():
30
+ extra = ['YYY']
31
+ keys = VEP.keys + extra
32
+ ann = VEP('missense_variant|tTt/tGt|F/C|ENSG00000186092|OR4F5|ENST00000335137|1/1|possibly_damaging(0.568)|deleterious(0)|113/305|protein_coding|xval|yval', keys=keys)
33
+ assert ann.unused() == extra, ann.unused()
34
+
35
+ assert ann.effects['YYY'] == 'yval'
36
+
37
+
38
+
39
+ def test_vep():
40
+
41
+ ann = VEP('missense_variant|tTt/tGt|F/C|ENSG00000186092|OR4F5|ENST00000335137|1/1|possibly_damaging(0.568)|deleterious(0)|113/305|protein_coding|')
42
+ assert ann.gene == 'OR4F5'
43
+ assert ann.transcript == 'ENST00000335137'
44
+ assert ann.aa_change == "F/C", ann.aa_change
45
+ assert ann.consequences == ['missense_variant']
46
+ assert ann.coding
47
+ assert ann.biotype == "protein_coding"
48
+ assert ann.severity == 2
49
+ assert ann.impact_severity == "MED", ann.impact_severity
50
+ assert not ann.is_pseudogene
51
+ assert ann.polyphen_score == 0.568, ann.polyphen
52
+ assert ann.polyphen_pred == "possibly_damaging", ann.polyphen
53
+ assert ann.sift_score == 0.0, ann.sift
54
+ assert ann.sift_pred == "deleterious", ann.sift
55
+ assert not ann.canonical
56
+
57
+ def test_vep_canonical():
58
+
59
+ ann = VEP('missense_variant|tTt/tGt|F/C|ENSG00000186092|OR4F5|ENST00000335137|1/1|possibly_damaging(0.568)|deleterious(0)|113/305|protein_coding|*', prioritize_canonical=True)
60
+ assert ann.gene == 'OR4F5'
61
+ assert ann.transcript == 'ENST00000335137'
62
+ assert ann.aa_change == "F/C", ann.aa_change
63
+ assert ann.consequences == ['missense_variant']
64
+ assert ann.coding
65
+ assert ann.biotype == "protein_coding"
66
+ assert ann.severity == 2
67
+ assert ann.impact_severity == "MED", ann.impact_severity
68
+ assert not ann.is_pseudogene
69
+ assert ann.polyphen_score == 0.568, ann.polyphen
70
+ assert ann.polyphen_pred == "possibly_damaging", ann.polyphen
71
+ assert ann.sift_score == 0.0, ann.sift
72
+ assert ann.sift_pred == "deleterious", ann.sift
73
+ assert ann.is_canonical
74
+
75
+ def test_bcfts():
76
+ f = os.path.join(HERE, "bcfts.txt.gz")
77
+ with gzip.open(f, "rt") as fh:
78
+ for csq in (BCFT(l.rstrip()) for l in fh):
79
+ assert csq.severity in (1, 2, 3)
80
+ assert csq.is_pseudogene in (True, False)
81
+ assert csq.coding in (True, False), (csq.coding, csq)
82
+ assert csq.is_exonic in (True, False)
83
+
84
+
85
+ def test_veps():
86
+
87
+ f = os.path.join(HERE, "vep-csqs.txt.gz")
88
+ with gzip.open(f, "rt") as veps:
89
+ for csq in (VEP(l.strip()) for l in veps):
90
+ assert csq.severity in (1, 2, 3)
91
+ assert csq.is_pseudogene in (True, False)
92
+ assert csq.coding in (True, False)
93
+ assert isinstance(csq.polyphen_value, float) or csq.polyphen_value is None
94
+ csq.gene
95
+ assert isinstance(csq.sift_value, float) or csq.sift_value is None
96
+
97
+ def test_snpeffs():
98
+ f = os.path.join(HERE, "snpeff-anns.txt.gz")
99
+ with gzip.open(f, "rt") as anns:
100
+ for csq in (SnpEff(l.strip()) for l in anns):
101
+ assert csq.severity in (1, 2, 3)
102
+ assert csq.is_pseudogene in (True, False)
103
+ assert csq.coding in (True, False)
104
+ assert csq.polyphen_value is None
105
+
106
+ EFFECTS = [VEP("upstream_gene_variant|||ENSG00000223972|DDX11L1|ENST00000456328|||||processed_transcript"),
107
+ VEP("downstream_gene_variant|||ENSG00000227232|WASH7P|ENST00000488147|||||unprocessed_pseudogene"),
108
+ VEP("non_coding_exon_variant&nc_transcript_variant|||ENSG00000223972|DDX11L1|ENST00000456328|2/3||||processed_transcript"),
109
+ VEP("non_coding_exon_variant&nc_transcript_variant|||ENSG00000223972|DDX11L1|ENST00000456328|2/3||||processed_transcript"),
110
+ VEP("splice_region_variant&non_coding_exon_variant&nc_transcript_variant|||ENSG00000223972|DDX11L1|ENST00000456328|2/3||||processed_transcript"),
111
+ VEP("splice_region_variant&non_coding_exon_variant&nc_transcript_variant|||ENSG00000223972|DDX11L1|ENST00000456328|2/3||||processed_transcript"),
112
+ VEP("splice_region_variant&non_coding_exon_variant&nc_transcript_variant|||ENSG00000223972|DDX11L1|ENST00000456328|2/3||||processed_transcript"),
113
+ VEP("intron_variant&nc_transcript_variant|||ENSG00000223972|DDX11L1|ENST00000450305|||||transcribed_unprocessed_pseudogene"),
114
+ VEP("intron_variant&nc_transcript_variant|||ENSG00000223972|DDX11L1|ENST00000450305|||||transcribed_unprocessed_pseudogene"),
115
+ VEP('missense_variant|tTt/tGt|F/C|ENSG00000186092|OR4F5|ENST00000335137|1/1|possibly_damaging(0.568)|deleterious(0)|113/305|protein_coding'),
116
+ VEP("non_coding_exon_variant&nc_transcript_variant&feature_elongation|||ENSG00000223972|DDX11L1|ENST00000456328|3/3||||processed_transcript"),
117
+ ]
118
+
119
+
120
+
121
+ def test_order():
122
+
123
+ effects = sorted(EFFECTS)
124
+ assert effects[-1].impact_severity == "MED"
125
+ assert effects[0].impact_severity == "LOW"
126
+
127
+ def test_canonical_order():
128
+ effects = EFFECTS[:]
129
+ effects.append(VEP("intron_variant&nc_transcript_variant|||ENSG00000223972|DDX11L1|ENST00000450305|||||transcribed_unprocessed_pseudogene|*", prioritize_canonical=True))
130
+ effects = sorted(effects)
131
+ assert effects[-1].is_canonical
132
+ assert effects[0].impact_severity == "LOW"
133
+ assert not effects[0].is_canonical
134
+
135
+ def test_o2():
136
+
137
+ keys = [x.strip() for x in "Effect | Effect_Impact | Functional_Class | Codon_Change | Amino_Acid_change| Amino_Acid_length | Gene_Name | Gene_BioType | Coding | Transcript | Exon | ERRORS | WARNINGS".split("|")]
138
+
139
+ effects = [OldSnpEff(v, keys) for v in "DOWNSTREAM(MODIFIER|||||RP5-902P8.10|processed_transcript|NON_CODING|ENST00000434139|),DOWNSTREAM(MODIFIER|||||RP5-902P8.10|processed_transcript|NON_CODING|ENST00000453732|),INTRON(MODIFIER||||138|SCNN1D|protein_coding|CODING|ENST00000470022|3),INTRON(MODIFIER||||638|SCNN1D|protein_coding|CODING|ENST00000338555|3),INTRON(MODIFIER||||638|SCNN1D|protein_coding|CODING|ENST00000400928|2),INTRON(MODIFIER||||669|SCNN1D|protein_coding|CODING|ENST00000379110|6),INTRON(MODIFIER||||704|SCNN1D|protein_coding|CODING|ENST00000325425|2),INTRON(MODIFIER||||802|SCNN1D|protein_coding|CODING|ENST00000379116|5),INTRON(MODIFIER|||||SCNN1D|nonsense_mediated_decay|CODING|ENST00000379101|5),INTRON(MODIFIER|||||SCNN1D|processed_transcript|CODING|ENST00000467651|3)".split(",")]
140
+
141
+ effects = sorted(effects)
142
+ assert effects[-1].gene == "SCNN1D", effects[-1].gene
143
+
144
+ effects = sorted([OldSnpEff(v, keys) for v in "DOWNSTREAM(MODIFIER||||85|FAM138A|protein_coding|CODING|ENST00000417324|),DOWNSTREAM(MODIFIER|||||FAM138A|processed_transcript|CODING|ENST00000461467|),DOWNSTREAM(MODIFIER|||||MIR1302-10|miRNA|NON_CODING|ENST00000408384|),EXON(MODIFIER|||||MIR1302-10|antisense|NON_CODING|ENST00000469289|1),INTRON(MODIFIER|||||MIR1302-10|antisense|NON_CODING|ENST00000473358|1),UPSTREAM(MODIFIER|||||WASH7P|unprocessed_pseudogene|NON_CODING|ENST00000423562|),UPSTREAM(MODIFIER|||||WASH7P|unprocessed_pseudogene|NON_CODING|ENST00000430492|),UPSTREAM(MODIFIER|||||WASH7P|unprocessed_pseudogene|NON_CODING|ENST00000438504|),UPSTREAM(MODIFIER|||||WASH7P|unprocessed_pseudogene|NON_CODING|ENST00000488147|),UPSTREAM(MODIFIER|||||WASH7P|unprocessed_pseudogene|NON_CODING|ENST00000538476|)".split(",")])
145
+ s = "\n".join(e.effect_string for e in effects[::-1])
146
+
147
+ # reversed so that most significant is first
148
+ assert s == """\
149
+ DOWNSTREAM(MODIFIER||||85|FAM138A|protein_coding|CODING|ENST00000417324|
150
+ DOWNSTREAM(MODIFIER|||||FAM138A|processed_transcript|CODING|ENST00000461467|
151
+ INTRON(MODIFIER|||||MIR1302-10|antisense|NON_CODING|ENST00000473358|1
152
+ EXON(MODIFIER|||||MIR1302-10|antisense|NON_CODING|ENST00000469289|1
153
+ DOWNSTREAM(MODIFIER|||||MIR1302-10|miRNA|NON_CODING|ENST00000408384|
154
+ UPSTREAM(MODIFIER|||||WASH7P|unprocessed_pseudogene|NON_CODING|ENST00000423562|
155
+ UPSTREAM(MODIFIER|||||WASH7P|unprocessed_pseudogene|NON_CODING|ENST00000430492|
156
+ UPSTREAM(MODIFIER|||||WASH7P|unprocessed_pseudogene|NON_CODING|ENST00000438504|
157
+ UPSTREAM(MODIFIER|||||WASH7P|unprocessed_pseudogene|NON_CODING|ENST00000488147|
158
+ UPSTREAM(MODIFIER|||||WASH7P|unprocessed_pseudogene|NON_CODING|ENST00000538476|"""
159
+
160
+ def test_highest():
161
+ effects = sorted(EFFECTS)
162
+
163
+ top = Effect.top_severity(effects)
164
+ assert top.impact_severity == "MED"
165
+ assert top.so == "missense_variant"
166
+ #assert top[0].
167
+
168
+
169
+ effects.append(effects[-1])
170
+
171
+ top = Effect.top_severity(effects)
172
+ assert isinstance(top, list)
173
+ assert top[0].impact_severity == "MED"
174
+
175
+ def test_splice():
176
+
177
+ e = VEP('splice_acceptor_variant&intron_variant&feature_truncation|||ENSG00000221978|CCNL2|ENST00000408918||||-/226|protein_coding|1')
178
+ assert (e.is_coding, e.is_exonic, e.is_splicing) == (False, False, True)
179
+
180
+ e = VEP('intron_variant&feature_elongation|||ENSG00000187634|SAMD11|ENST00000341065||||-/589|protein_coding|1')
181
+ assert (e.is_coding, e.is_exonic, e.is_splicing) == (False, False, False)
182
+
183
+ def test_eff_splice():
184
+
185
+ keys = [x.strip() for x in "Effect | Effect_Impact | Functional_Class | Codon_Change | Amino_Acid_change| Amino_Acid_length | Gene_Name | Gene_BioType | Coding | Transcript | Exon | ERRORS | WARNINGS".split("|")]
186
+ e = OldSnpEff("SPLICE_SITE_REGION+SYNONYMOUS_CODING(LOW|SILENT|acG/acA|T245|1134|ANKS1A|protein_coding|CODING|ENST00000360359|5|A)", keys)
187
+ assert e.aa_change == "T245"
188
+ # note that we choose splice_site_region over synonymous coding
189
+ assert e.is_splicing, e.is_splicing
190
+
191
+ assert not e.is_coding
192
+
193
+ e = OldSnpEff("intergenic_region(MODIFIER|||n.null_nulldelAAGGAAGG|||||||A",
194
+ keys)
195
+ assert e.consequences != []
196
+
197
+ def test_regr():
198
+ keys = [x.strip() for x in 'Effect | Effect_Impact | Functional_Class | Codon_Change | Amino_Acid_change| Amino_Acid_length | Gene_Name | Transcript_BioType | Gene_Coding | Transcript_ID | Exon_Rank | Genotype_Number | ERRORS | WARNINGS'.split("|")]
199
+ v = OldSnpEff('SPLICE_SITE_REGION+SYNONYMOUS_CODING(LOW|SILENT|acG/acA|T245|1134|ANKS1A|protein_coding|CODING|ENST00000360359|5|A)', keys)
200
+ assert v.consequences == ['splice_region_variant', 'synonymous_variant'], v.consequences
201
+ assert v.severity == 2, v.severity
202
+ assert v.aa_change == 'T245'
203
+ v = OldSnpEff('UPSTREAM(MODIFIER||2771|||PSMB1|processed_transcript|CODING|ENST00000462957||C)', keys)
204
+ assert v.consequences == ['upstream_gene_variant'], v.consequences
205
+ assert v.severity == 1, v.severity
206
+
207
+ v = OldSnpEff('NEXT_PROT[maturation_peptide](LOW||||241|PSMB1|protein_coding|CODING|||C)', keys)
208
+ assert v.consequences == ['NEXT_PROT[maturation_peptide]'], v.consequences
209
+ assert v.severity == 1, v.severity
210
+
211
+ assert v <= v
212
+
213
+ def test_aa_change():
214
+
215
+ eff = OldSnpEff('NON_SYNONYMOUS_CODING(MODERATE|MISSENSE|Agc/Ggc|S418G|696|C1orf170|protein_coding|CODING|ENST00000433179|3|C)')
216
+ assert eff.aa_change == 'S418G'
217
+ ann = SnpEff('C|missense_variant|MODERATE|C1orf170|ENSG00000187642|transcript|ENST00000433179|protein_coding|3/5|c.1252A>G|p.Ser418Gly|1252/3064|1252/2091|418/696||')
218
+ assert ann.aa_change == 'p.Ser418Gly'
219
+
220
+ def test_old():
221
+ keys = [x.strip() for x in 'Effect | Effect_Impact | Functional_Class | Codon_Change | Amino_Acid_change| Amino_Acid_length | Gene_Name | Transcript_BioType | Gene_Coding | Transcript_ID | Exon_Rank | Genotype_Number | ERRORS | WARNINGS'.split("|")]
222
+ v = OldSnpEff('SPLICE_SITE_REGION+SYNONYMOUS_CODING(LOW|SILENT|acG/acA|T245|1134|ANKS1A|protein_coding|CODING|ENST00000360359|5|A)', keys)
223
+ assert v.so == "splice_region_variant", v.so
224
+ v = OldSnpEff('SYNONYMOUS_CODING+SPLICE_SITE_REGION(LOW|SILENT|acG/acA|T245|1134|ANKS1A|protein_coding|CODING|ENST00000360359|5|A)', keys)
225
+ assert v.so == "splice_region_variant", v.so
226
+ assert v.aa_length == 1134, v.aa_length
227
+ assert v.exon == "5", v.exon
228
+ assert v.codon_change == "acG/acA", v.codon_change
229
+ assert v.transcript == "ENST00000360359", v.transcript
230
+
231
+ def test_old2():
232
+ keys = [x.strip() for x in 'Effect | Effect_Impact | Functional_Class | Codon_Change | Amino_Acid_change| Amino_Acid_length | Gene_Name | Transcript_BioType | Gene_Coding | Transcript_ID | Exon_Rank | Genotype_Number | ERRORS | WARNINGS'.split("|")]
233
+ v = OldSnpEff('SPLICE_SITE_REGION+NON_SYNONYMOUS_CODING(LOW|SILENT|acG/acA|T245|1134|ANKS1A|protein_coding|CODING|ENST00000360359|5|A)', keys)
234
+ assert v.so == "missense_variant", v.so
235
+
236
+ def test_weird_vep():
237
+ keys = "Consequence|Codons|Amino_acids|Gene|SYMBOL|Feature|EXON|PolyPhen|SIFT|Protein_position|BIOTYPE|CANONICAL|CCDS|RadialSVM_score|RadialSVM_pred|LR_score|LR_pred|CADD_raw|CADD_phred|Reliability_index".split("|")
238
+
239
+ csqs = ["?|||117581|TWIST2|NM_001271893.1|1/1||||protein_coding|YES||||||||,non_coding_transcript_exon_variant&non_coding_transcript_variant|||117581|TWIST2|NM_001271893.1_dupl8|1/1||||mRNA|||||||||",
240
+ "non_coding_transcript_exon_variant&non_coding_transcript_variant|||117581|TWIST2|NM_001271893.1_dupl8|1/1||||mRNA|||||||||,?|||117581|TWIST2|NM_001271893.1|1/1||||protein_coding|YES||||||||",
241
+ "?|||115286|SLC25A26|NM_173471.3|1/1||||protein_coding|YES||||||||",
242
+
243
+ "|||ENSG00000138190|EXOC6|ENST00000260762||||-/804|protein_coding,|||ENSG00000138190|EXOC6|ENST00000371547||||-/820|protein_coding,|||ENSG00000138190|EXOC6|ENST00000443748||||-/701|protein_coding,NMD_transcript_variant|||ENSG00000138190|EXOC6|ENST00000495132||||-/404|nonsense_mediated_decay,|||ENSG00000138190|EXOC6|ENST00000371552||||-/799|protein_coding",
244
+ "|||ENSG00000013503|POLR3B|ENST00000539066||||-/1075|protein_coding,nc_transcript_variant|||ENSG00000013503|POLR3B|ENST00000549195|||||processed_transcript,|||ENSG00000013503|POLR3B|ENST00000549569||||-/170|protein_coding,|||ENSG00000013503|POLR3B|ENST00000228347||||-/1133|",
245
+ "|||ENSG00000147202|DIAPH2|ENST00000373054||||-/1097|protein_coding,|||ENSG00000147202|DIAPH2|ENST00000355827||||-/1096|protein_coding,|||ENSG00000147202|DIAPH2|ENST00000324765||||-/1101|protein_coding,|||ENSG00000147202|DIAPH2|ENST00000373049||||-/1096|protein_coding,|||ENSG00000147202|DIAPH2|ENST00000373061||||-/1101|protein_coding",
246
+
247
+ ]
248
+ import sys
249
+ for cs in csqs:
250
+ for c in cs.split(","):
251
+ v = VEP(c, keys)
252
+ assert v.impact_severity in ('LOW', 'MEDIUM', 'HIGH')
253
+
254
+ def test_empty_snpeff():
255
+
256
+ keys = [x.strip() for x in 'Effect | Effect_Impact | Functional_Class | Codon_Change | Amino_Acid_change| Amino_Acid_length | Gene_Name | Transcript_BioType | Gene_Coding | Transcript_ID | Exon_Rank | Genotype_Number | ERRORS | WARNINGS'.split("|")]
257
+
258
+ eff = "(MODIFIER||||||||||A|ERROR_CHROMOSOME_NOT_FOUND)"
259
+ v = OldSnpEff(eff, keys)
260
+ assert v.impact_severity == "LOW", v.impact_severity
261
+
262
+ def test_protein_contact():
263
+ ann = SnpEff('C|protein_protein_contact|HIGH|C1orf170|ENSG00000187642|transcript|ENST00000433179|protein_coding|3/5|c.1252A>G|p.Ser418Gly|1252/3064|1252/2091|418/696||')
264
+ assert ann.impact_severity == "HIGH"
265
+
266
+ def test_gemini_issue812():
267
+ ann = VEP('protein_altering_variant|caGCAGCAGCAGCAGCAACAGCAG/caA|QQQQQQQQ/Q|ENSG00000204842|ATXN2|ENST00000608853|1/25|||14-21/1153|protein_coding|', keys="Consequence|Codons|Amino_acids|Gene|SYMBOL|Feature|EXON|PolyPhen|SIFT|Protein_position|BIOTYPE|CANONICAL".split("|"))
268
+ assert ann.is_coding
269
+
270
+ def test_bug_vcf2db_21():
271
+ ann = VEP('synonymous_variant|tcA/tcG|S|ENSG00000186092|OR4F5|ENST00000335137|1/1|||60/305|protein_coding||Low_complexity_(Seg):seg&Transmembrane_helices:TMhelix&Prints_domain:PR00237&Superfamily_domains:SSF81321&Gene3D:1.20.1070.10&hmmpanther:PTHR26451&hmmpanther:PTHR26451:SF72&PROSITE_profiles:PS50262||||ENST00000335137.3:c.180A>G|ENST00000335137.3:c.180A>G(p.%3D)|||-0.817044|0.039', keys="Consequence|Codons|Amino_acids|Gene|SYMBOL|Feature|EXON|PolyPhen|SIFT|Protein_position|BIOTYPE|CANONICAL|DOMAINS|CLIN_SIG".split("|"))
272
+
273
+ assert ann.codon_change == "tcA/tcG", ann.codon_change
274
+
275
+ def test_32():
276
+ keys = "Allele|Consequence|IMPACT|SYMBOL|Gene|Feature_type|Feature|BIOTYPE|EXON|INTRON|HGVSc|HGVSp|cDNA_position|CDS_position|Protein_position|Amino_acids|Codons|Existing_variation|DISTANCE|STRAND|FLAGS|VARIANT_CLASS|SYMBOL_SOURCE|HGNC_ID|CANONICAL|TSL|APPRIS|CCDS|ENSP|SWISSPROT|TREMBL|UNIPARC|REFSEQ_MATCH|SOURCE|GIVEN_REF|USED_REF|GENE_PHENO|SIFT|PolyPhen|DOMAINS|HGVS_OFFSET|AF|AFR_AF|AMR_AF|EAS_AF|EUR_AF|SAS_AF|AA_AF|EA_AF|gnomAD_AF|gnomAD_AFR_AF|gnomAD_AMR_AF|gnomAD_ASJ_AF|gnomAD_EAS_AF|gnomAD_FIN_AF|gnomAD_NFE_AF|gnomAD_OTH_AF|gnomAD_SAS_AF|MAX_AF|MAX_AF_POPS|CLIN_SIG|SOMATIC|PHENO|PUBMED|MOTIF_NAME|MOTIF_POS|HIGH_INF_POS|MOTIF_SCORE_CHANGE|MaxEntScan_alt|MaxEntScan_diff|MaxEntScan_ref|SpliceRegion".split("|")
277
+ s = "-|frameshift_variant&start_lost&start_retained_variant|HIGH|HRNR|ENSG00000197915|Transcript|ENST00000368801|protein_coding|2/3||ENST00000368801.2:c.1del|ENSP00000357791.2:p.Met1?|77/9623|1/8553|1/2850|M/X|Atg/tg|rs34061715&COSM111478||-1||deletion|HGNC|HGNC:20846|YES|1|P1|CCDS30859.1|ENSP00000357791|Q86YZ3||UPI00001D7CAD||Ensembl|T|T||||||0.874|0.7337|0.8818|0.9544|0.9592|0.8875|||0.9028|0.7227|0.8276|0.9554|0.9063|0.9541|0.9411|0.9142|0.9069|0.9592|EUR||0&1|0&1|||||||||,-|intron_variant&non_coding_transcript_variant|MODIFIER|FLG-AS1|ENSG00000237975|Transcript|ENST00000420707|antisense_RNA||1/8|ENST00000420707.5:n.159-25632del|||||||rs34061715&COSM111478||1||deletion|HGNC|HGNC:27913||5||||||||Ensembl|T|T|||||10|0.874|0.7337|0.8818|0.9544|0.9592|0.8875|||0.9028|0.7227|0.8276|0.9554|0.9063|0.9541|0.9411|0.9142|0.9069|0.9592|EUR||0&1|0&1|||||||||,-|intron_variant&non_coding_transcript_variant|MODIFIER|FLG-AS1|ENSG00000237975|Transcript|ENST00000593011|antisense_RNA||1/3|ENST00000593011.5:n.296+54843del|||||||rs34061715&COSM111478||1||deletion|HGNC|HGNC:27913||4||||||||Ensembl|T|T|||||10|0.874|0.7337|0.8818|0.9544|0.9592|0.8875|||0.9028|0.7227|0.8276|0.9554|0.9063|0.9541|0.9411|0.9142|0.9069|0.9592|EUR||0&1|0&1|||||||||,-|frameshift_variant&start_lost&start_retained_variant|HIGH|HRNR|388697|Transcript|NM_001009931.2|protein_coding|2/3||NM_001009931.2:c.1del|NP_001009931.1:p.Met1?|80/9632|1/8553|1/2850|M/X|Atg/tg|rs34061715&COSM111478||-1||deletion|EntrezGene|HGNC:20846|YES||||NP_001009931.1||||rseq_mrna_match|RefSeq|T|T||||||0.874|0.7337|0.8818|0.9544|0.9592|0.8875|||0.9028|0.7227|0.8276|0.9554|0.9063|0.9541|0.9411|0.9142|0.9069|0.9592|EUR||0&1|0&1|||||||||".split(",")
278
+ for e in s:
279
+ eff = VEP(e, keys=keys)
280
+ if not "intron" in e.lower():
281
+ assert eff.impact_severity == "HIGH", (eff.impact_severity, e)
Binary file
@@ -0,0 +1,60 @@
1
+ Metadata-Version: 2.4
2
+ Name: geneimpacts
3
+ Version: 0.3.8
4
+ Summary: normalize effects from variant annotation tools (snpEff, VEP)
5
+ Author: Brent Pedersen
6
+ Author-email: bpederse@gmail.com
7
+ Classifier: Development Status :: 4 - Beta
8
+ Classifier: Intended Audience :: Science/Research
9
+ Classifier: License :: OSI Approved :: MIT License
10
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
11
+ Description-Content-Type: text/markdown
12
+ License-File: LICENSE
13
+ Dynamic: author
14
+ Dynamic: author-email
15
+ Dynamic: classifier
16
+ Dynamic: description
17
+ Dynamic: description-content-type
18
+ Dynamic: license-file
19
+ Dynamic: summary
20
+
21
+ Given multiple snpEff or VEP or BCFTools consequence annotations for a single variant, get an orderable python object for each annotation.
22
+
23
+ [![Build Status](https://travis-ci.org/brentp/geneimpacts.svg?branch=master)](https://travis-ci.org/brentp/geneimpacts)
24
+
25
+ This is to provide a consistent interface to
26
+ different variant annotations such as from [snpEff ANN field](http://snpeff.sourceforge.net/) and the [VEP CSQ field](http://www.ensembl.org/info/docs/tools/vep/index.html).
27
+ and the [BCFTools consequence field](http://biorxiv.org/content/early/2016/12/01/090811)
28
+
29
+ This will be used in [gemini](http://gemini.rtfd.org/) but should also be of
30
+ general utility.
31
+
32
+ Design
33
+ ======
34
+
35
+ There is an effect base-class and then a sub-class for `snpEff`, one for `VEP`, and one for `BCFT`
36
+
37
+ `Effect` objects are orderable (via \_\_le\_\_ ) and should have an \_\_eq\_\_ method so that we can use [functools.total_ordering](https://docs.python.org/2/library/functools.html#functools.total_ordering) to provide the other comparison operators.
38
+
39
+ Given 2 effects objects, `a` and `b`: `a < b == True` iff the *severity* of `b` is greater than `a`.
40
+
41
+ We will have a classmethod: `Effect.top_severity([eff1, ... effn]) that will return the single highest severity if that exists or
42
+ a list of the ties for highest
43
+
44
+ Rules for severity:
45
+ ===================
46
+
47
+ Given 2 annotations, *a* and *b*
48
+ *a* is more severe than *b* if:
49
+
50
+ 1. *b* is a pseudogene and *a* is not
51
+ 2. *a* is coding and *b* is not
52
+ 3. *a* has higher severity than *b* ( see below)
53
+ 4. polyphen, then sift
54
+ 5. ??? transcript length? (we dont have access to this).
55
+
56
+ severity
57
+ --------
58
+
59
+ Severity is based on the [impacts from VEP](http://uswest.ensembl.org/info/docs/tools/vep/script/vep_other.html#pick)
60
+ and the [impacts from snpEff](http://snpeff.sourceforge.net/VCFannotationformat_v1.0.pdf). We reduce from the 4 categories HIGH, MODERATE, LOW, MODIFIER to 3 by renaming MODERATE to MED and renaming MODIFIER to LOW.
@@ -0,0 +1,12 @@
1
+ geneimpacts/__init__.py,sha256=roiMgKQm079MwmQ1fx0SmxBjGEFVnUMAWqYE_ozMil4,80
2
+ geneimpacts/effect.py,sha256=-Iqq2LLrGLse8sSr1NeuK1LB3Mciwmul8NuA8ud-aWY,24669
3
+ geneimpacts/tests/__init__.py,sha256=MsSFjiLMLJZ7QhUPpVBWKiyDnCzryquRyr329NoCACI,2
4
+ geneimpacts/tests/bcfts.txt.gz,sha256=0OiOw1XWu_s5KlD3Xy6QQnfXIGnITcWE-7S5WXTyMoM,6684
5
+ geneimpacts/tests/snpeff-anns.txt.gz,sha256=Z0vmJLAfqwK_DHEE2wbAmE0DGpUR_P_yOYg2dxggoR4,6930
6
+ geneimpacts/tests/test_impacts.py,sha256=XbCuVPqNGBJF8BT5pc255w3vP55D2AcQaXLmPSNh0d8,19622
7
+ geneimpacts/tests/vep-csqs.txt.gz,sha256=r7Cbf6xK3vKJGbOWq1ci-4_37fQKavFU0VlIxUixI2Q,6220
8
+ geneimpacts-0.3.8.dist-info/licenses/LICENSE,sha256=tgqe_a6fIQcB41fHR_3DxJL6Y3xJ7UYqc-JsQKxo400,1099
9
+ geneimpacts-0.3.8.dist-info/METADATA,sha256=KUOrCg-Pz-fjO_2HCwxnubG-dis0KJAxsbk-apYFe-w,2559
10
+ geneimpacts-0.3.8.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
11
+ geneimpacts-0.3.8.dist-info/top_level.txt,sha256=zEvuIfEuzZBG93GQ7w7gHnbzBxF38FyDc0WQ7_lBgVI,12
12
+ geneimpacts-0.3.8.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,22 @@
1
+ The MIT License (MIT)
2
+
3
+ Copyright (c) 2015 Brent Pedersen - Bioinformatics
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
22
+
@@ -0,0 +1 @@
1
+ geneimpacts