gencodegenes 1.0.0__tar.gz → 1.0.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gencodegenes-1.0.2/MANIFEST.in +16 -0
- {gencodegenes-1.0.0 → gencodegenes-1.0.2}/PKG-INFO +1 -1
- {gencodegenes-1.0.0 → gencodegenes-1.0.2}/setup.py +1 -1
- gencodegenes-1.0.2/src/gencodegenes/gencode.pyx +436 -0
- gencodegenes-1.0.2/src/gencodegenes/transcript.pxd +90 -0
- gencodegenes-1.0.2/src/gencodegenes/transcript.pyx +378 -0
- {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gencodegenes.egg-info/PKG-INFO +1 -1
- {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gencodegenes.egg-info/SOURCES.txt +3 -0
- gencodegenes-1.0.0/MANIFEST.in +0 -15
- {gencodegenes-1.0.0 → gencodegenes-1.0.2}/LICENSE.txt +0 -0
- {gencodegenes-1.0.0 → gencodegenes-1.0.2}/README.md +0 -0
- {gencodegenes-1.0.0 → gencodegenes-1.0.2}/pyproject.toml +0 -0
- {gencodegenes-1.0.0 → gencodegenes-1.0.2}/setup.cfg +0 -0
- {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gencode.cpp +0 -0
- {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gencode.h +0 -0
- {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gencodegenes/__init__.py +0 -0
- {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gencodegenes/gencode.cpp +0 -0
- {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gencodegenes/transcript.cpp +0 -0
- {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gencodegenes.egg-info/dependency_links.txt +0 -0
- {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gencodegenes.egg-info/requires.txt +0 -0
- {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gencodegenes.egg-info/top_level.txt +0 -0
- {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gtf.cpp +0 -0
- {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gtf.h +0 -0
- {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gzstream/gzstream.C +0 -0
- {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gzstream/gzstream.h +0 -0
- {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/tx.cpp +0 -0
- {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/tx.h +0 -0
- {gencodegenes-1.0.0 → gencodegenes-1.0.2}/tests/__init__.py +0 -0
- {gencodegenes-1.0.0 → gencodegenes-1.0.2}/tests/test_gencode.py +0 -0
- {gencodegenes-1.0.0 → gencodegenes-1.0.2}/tests/test_transcript.py +0 -0
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
# gencodegenes code
|
|
2
|
+
include MANIFEST.in
|
|
3
|
+
include pyproject.toml
|
|
4
|
+
include src/gencodegenes/*.cpp
|
|
5
|
+
include src/gencodegenes/*.py
|
|
6
|
+
include src/gencodegenes/*.pyx
|
|
7
|
+
include src/gencodegenes/*.pxd
|
|
8
|
+
include src/gencodegenes/data/rates.txt
|
|
9
|
+
include src/*.h
|
|
10
|
+
include src/*.cpp
|
|
11
|
+
include src/gzstream/gzstream.C
|
|
12
|
+
include src/gzstream/gzstream.h
|
|
13
|
+
include data/*.txt
|
|
14
|
+
|
|
15
|
+
# gencodegenes tests
|
|
16
|
+
include tests/*.py
|
|
@@ -107,7 +107,7 @@ setup(name="gencodegenes",
|
|
|
107
107
|
description='Package to load genes from GENCODE GTF files',
|
|
108
108
|
long_description=io.open('README.md', encoding='utf-8').read(),
|
|
109
109
|
long_description_content_type='text/markdown',
|
|
110
|
-
version="1.0.
|
|
110
|
+
version="1.0.2",
|
|
111
111
|
author="Jeremy McRae",
|
|
112
112
|
author_email="jeremy.mcrae@gmail.com",
|
|
113
113
|
license="MIT",
|
|
@@ -0,0 +1,436 @@
|
|
|
1
|
+
# cython: language_level=3, boundscheck=False, emit_linenums=True
|
|
2
|
+
|
|
3
|
+
import bisect
|
|
4
|
+
import logging
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
from libcpp.algorithm cimport lower_bound, upper_bound
|
|
8
|
+
from libcpp.vector cimport vector
|
|
9
|
+
from libcpp.string cimport string
|
|
10
|
+
from libcpp cimport bool
|
|
11
|
+
from libcpp.map cimport map
|
|
12
|
+
|
|
13
|
+
from pyfaidx import Fasta
|
|
14
|
+
|
|
15
|
+
from gencodegenes.transcript cimport Tx, Region, CDS_coords
|
|
16
|
+
from gencodegenes.transcript import Transcript
|
|
17
|
+
|
|
18
|
+
cdef extern from "gtf.h" namespace "gencode":
|
|
19
|
+
cdef struct GTFLine:
|
|
20
|
+
string chrom
|
|
21
|
+
string feature
|
|
22
|
+
int start
|
|
23
|
+
int end
|
|
24
|
+
string strand
|
|
25
|
+
string symbol
|
|
26
|
+
string tx_id
|
|
27
|
+
string transcript_type
|
|
28
|
+
int is_canonical
|
|
29
|
+
|
|
30
|
+
GTFLine parse_gtfline(string line)
|
|
31
|
+
|
|
32
|
+
cdef extern from "gencode.h" namespace "gencode":
|
|
33
|
+
cdef struct NamedTx:
|
|
34
|
+
string symbol
|
|
35
|
+
Tx tx
|
|
36
|
+
int is_canonical
|
|
37
|
+
|
|
38
|
+
cdef struct GenePoint:
|
|
39
|
+
int pos
|
|
40
|
+
string symbol
|
|
41
|
+
|
|
42
|
+
vector[NamedTx] open_gencode(string, bool)
|
|
43
|
+
bool CompFunc(const GenePoint &l, const GenePoint &r)
|
|
44
|
+
vector[string] _in_region(string chrom, int start, int end,
|
|
45
|
+
map[string, vector[GenePoint]] & starts, map[string, vector[GenePoint]] & ends,
|
|
46
|
+
int max_window) except+
|
|
47
|
+
|
|
48
|
+
cpdef _parse_gtfline(string line):
|
|
49
|
+
''' python function for unit testing GTF parsing
|
|
50
|
+
'''
|
|
51
|
+
return parse_gtfline(line)
|
|
52
|
+
|
|
53
|
+
cdef _convert_exons(vector[Region] exons):
|
|
54
|
+
''' convert vector of exon Regions to list of lists
|
|
55
|
+
|
|
56
|
+
We need exons and CDS as lists of lists for constructing the python
|
|
57
|
+
Transcript object.
|
|
58
|
+
'''
|
|
59
|
+
return [[y.start, y.end] for y in exons]
|
|
60
|
+
|
|
61
|
+
cpdef _open_gencode(gtf_path, coding_only=True):
|
|
62
|
+
''' python function for unit testing loading transcripts from GTF
|
|
63
|
+
'''
|
|
64
|
+
cdef vector[NamedTx] _transcripts = open_gencode(gtf_path.encode('utf8'), coding_only)
|
|
65
|
+
|
|
66
|
+
transcripts = []
|
|
67
|
+
for x in _transcripts:
|
|
68
|
+
tx = x.tx
|
|
69
|
+
chrom = tx.get_chrom().decode('utf8')
|
|
70
|
+
start = tx.get_start()
|
|
71
|
+
end = tx.get_end()
|
|
72
|
+
exons = _convert_exons(tx.get_exons())
|
|
73
|
+
cds = _convert_exons(tx.get_cds())
|
|
74
|
+
strand = chr(tx.get_strand())
|
|
75
|
+
tx_id = tx.get_name().decode('utf8')
|
|
76
|
+
transcript_type = tx.get_type().decode('utf8')
|
|
77
|
+
transcript = Transcript(tx_id, chrom, start, end, strand, transcript_type, exons, cds, offset=0)
|
|
78
|
+
transcripts.append((x.symbol.decode('utf8'), transcript, x.is_canonical))
|
|
79
|
+
return transcripts
|
|
80
|
+
|
|
81
|
+
__genome_ = None
|
|
82
|
+
|
|
83
|
+
cdef class Gene:
|
|
84
|
+
cdef string _symbol
|
|
85
|
+
cdef vector[Tx] _transcripts
|
|
86
|
+
cdef vector[int] _canonical
|
|
87
|
+
cdef str _chrom
|
|
88
|
+
cdef int _start, _end
|
|
89
|
+
def __cinit__(self, symbol):
|
|
90
|
+
if isinstance(symbol, str):
|
|
91
|
+
symbol = symbol.encode('utf8')
|
|
92
|
+
self._symbol = symbol
|
|
93
|
+
self.start = 999999999
|
|
94
|
+
self.end = -999999999
|
|
95
|
+
|
|
96
|
+
cdef add_tx(self, Tx tx, int is_canonical):
|
|
97
|
+
self._transcripts.push_back(tx)
|
|
98
|
+
self._canonical.push_back(is_canonical)
|
|
99
|
+
self.chrom = tx.get_chrom().decode('utf8')
|
|
100
|
+
self.start = min(self.start, tx.get_start())
|
|
101
|
+
self.end = max(self.end, tx.get_end())
|
|
102
|
+
|
|
103
|
+
def add_transcript(self, _tx):
|
|
104
|
+
''' add a Transcript to the gene object
|
|
105
|
+
|
|
106
|
+
This ends up coping the data from the Transcript object, rather than
|
|
107
|
+
reusing the Tx object contained in the Trnascript, but it's not too much
|
|
108
|
+
time wasted, so long as we don't do this millions of times.
|
|
109
|
+
'''
|
|
110
|
+
assert isinstance(_tx, Transcript)
|
|
111
|
+
# construct a new Tx obect by copying out the relevant data
|
|
112
|
+
cdef string tx_id = _tx.get_name().encode('utf8')
|
|
113
|
+
cdef string chrom = _tx.get_chrom().encode('utf8')
|
|
114
|
+
cdef int start = _tx.get_start()
|
|
115
|
+
cdef int end = _tx.get_end()
|
|
116
|
+
cdef char strand = ord(_tx.get_strand())
|
|
117
|
+
cdef string tx_type = _tx.get_type().encode('utf8')
|
|
118
|
+
cdef vector[vector[int]] exons
|
|
119
|
+
cdef vector[vector[int]] cds
|
|
120
|
+
cdef vector[int] exon
|
|
121
|
+
for x in _tx.get_exons():
|
|
122
|
+
exon = [x['start'], x['end']]
|
|
123
|
+
exons.push_back(exon)
|
|
124
|
+
for x in _tx.get_cds():
|
|
125
|
+
exon = [x['start'], x['end']]
|
|
126
|
+
cds.push_back(exon)
|
|
127
|
+
|
|
128
|
+
cdef Tx tx = Tx(tx_id, chrom, start, end, strand, tx_type)
|
|
129
|
+
tx.set_exons(exons, cds)
|
|
130
|
+
tx.set_cds(cds)
|
|
131
|
+
|
|
132
|
+
cdef string seq = _tx.get_genomic_sequence().encode('utf8')
|
|
133
|
+
cdef int offset = _tx.get_genomic_offset()
|
|
134
|
+
if len(seq) > 0:
|
|
135
|
+
if chr(strand) == '-':
|
|
136
|
+
_tx.reverse_complement(seq)
|
|
137
|
+
seq = _tx.reverse_complement(seq).encode('utf8')
|
|
138
|
+
tx.add_genomic_sequence(seq, offset)
|
|
139
|
+
self.add_tx(tx, False)
|
|
140
|
+
|
|
141
|
+
def __repr__(self):
|
|
142
|
+
chrom = self.chrom
|
|
143
|
+
return f'Gene("{self.symbol}", {chrom}:{self.start}-{self.end})'
|
|
144
|
+
|
|
145
|
+
@property
|
|
146
|
+
def symbol(self):
|
|
147
|
+
return self._symbol.decode('utf8')
|
|
148
|
+
|
|
149
|
+
@property
|
|
150
|
+
def chrom(self):
|
|
151
|
+
return self._chrom
|
|
152
|
+
@chrom.setter
|
|
153
|
+
def chrom(self, value):
|
|
154
|
+
self._chrom = value
|
|
155
|
+
|
|
156
|
+
@property
|
|
157
|
+
def start(self):
|
|
158
|
+
return self._start
|
|
159
|
+
@start.setter
|
|
160
|
+
def start(self, value):
|
|
161
|
+
self._start = value
|
|
162
|
+
|
|
163
|
+
@property
|
|
164
|
+
def end(self):
|
|
165
|
+
return self._end
|
|
166
|
+
@end.setter
|
|
167
|
+
def end(self, value):
|
|
168
|
+
self._end = value
|
|
169
|
+
|
|
170
|
+
@property
|
|
171
|
+
def strand(self):
|
|
172
|
+
if self._transcripts.size() > 0:
|
|
173
|
+
return chr(self._transcripts[0].get_strand())
|
|
174
|
+
raise IndexError('no transcripts in gene yet')
|
|
175
|
+
|
|
176
|
+
cdef _convert_exons(self, vector[Region] exons):
|
|
177
|
+
''' convert vector of exon Regions to list of lists
|
|
178
|
+
|
|
179
|
+
We need exons and CDS as lists of lists for constructing the python
|
|
180
|
+
Transcript object.
|
|
181
|
+
'''
|
|
182
|
+
return [[y.start, y.end] for y in exons]
|
|
183
|
+
|
|
184
|
+
cdef _to_Transcript(self, Tx tx):
|
|
185
|
+
''' construct Transcript (python object) from Tx (c++ object)
|
|
186
|
+
'''
|
|
187
|
+
offset = 5 if tx.get_genomic_offset() == 0 else tx.get_genomic_offset()
|
|
188
|
+
chrom = tx.get_chrom().decode('utf8')
|
|
189
|
+
start = tx.get_start()
|
|
190
|
+
end = tx.get_end()
|
|
191
|
+
exons = self._convert_exons(tx.get_exons())
|
|
192
|
+
cds = self._convert_exons(tx.get_cds())
|
|
193
|
+
seq = tx.get_genomic_sequence().decode('utf8')
|
|
194
|
+
if seq == '':
|
|
195
|
+
seq = None
|
|
196
|
+
if seq is None and __genome_ is not None:
|
|
197
|
+
seq = __genome_[chrom][start-1-offset:end-1+offset].seq.upper()
|
|
198
|
+
strand = chr(tx.get_strand())
|
|
199
|
+
if strand == '-' and seq is not None:
|
|
200
|
+
seq = tx.reverse_complement(seq.encode('utf8')).decode('utf8')
|
|
201
|
+
tx_id = tx.get_name().decode('utf8')
|
|
202
|
+
tx_type = tx.get_type().decode('utf8')
|
|
203
|
+
return Transcript(tx_id, chrom, start, end, strand, tx_type, exons, cds, seq, offset=offset)
|
|
204
|
+
|
|
205
|
+
@property
|
|
206
|
+
def transcripts(self):
|
|
207
|
+
''' get list of Transcripts for gene, with genomic DNA included
|
|
208
|
+
'''
|
|
209
|
+
return [self._to_Transcript(x) for x in self._transcripts]
|
|
210
|
+
|
|
211
|
+
cdef int _cds_len(self, Tx tx):
|
|
212
|
+
''' get length of coding sequence for a Tx object based transcript
|
|
213
|
+
'''
|
|
214
|
+
cdef CDS_coords coords = tx.get_coding_distance(tx.get_cds_end())
|
|
215
|
+
return coords.position + 1
|
|
216
|
+
|
|
217
|
+
cdef Tx _max_by_cds(self, vector[Tx] transcripts) except *:
|
|
218
|
+
''' get longest transcript by CDS length
|
|
219
|
+
'''
|
|
220
|
+
cdef Tx max_tx
|
|
221
|
+
length = 0
|
|
222
|
+
for tx in transcripts:
|
|
223
|
+
curr_len = self._cds_len(tx)
|
|
224
|
+
if curr_len > length:
|
|
225
|
+
length = curr_len
|
|
226
|
+
max_tx = tx
|
|
227
|
+
if length == 0:
|
|
228
|
+
raise ValueError('no coding transcripts')
|
|
229
|
+
return max_tx
|
|
230
|
+
|
|
231
|
+
cdef int _exonic_len(self, Tx tx):
|
|
232
|
+
''' get length of exonic sequence for a Tx object based transcript
|
|
233
|
+
'''
|
|
234
|
+
length = 0
|
|
235
|
+
for start, end in _convert_exons(tx.get_exons()):
|
|
236
|
+
length += abs(end - start) + 1
|
|
237
|
+
return length
|
|
238
|
+
# return sum(abs(x.end - x.start) + 1 for x in tx.get_exons())
|
|
239
|
+
|
|
240
|
+
cdef Tx _max_by_exonic(self, vector[Tx] transcripts) except *:
|
|
241
|
+
''' get longest transcript by CDS length
|
|
242
|
+
'''
|
|
243
|
+
cdef Tx max_tx
|
|
244
|
+
length = 0
|
|
245
|
+
for tx in transcripts:
|
|
246
|
+
curr_len = self._exonic_len(tx)
|
|
247
|
+
if curr_len > length:
|
|
248
|
+
length = curr_len
|
|
249
|
+
max_tx = tx
|
|
250
|
+
if length == 0:
|
|
251
|
+
raise ValueError('no exonic transcripts')
|
|
252
|
+
return max_tx
|
|
253
|
+
|
|
254
|
+
@property
|
|
255
|
+
def canonical(self):
|
|
256
|
+
''' find the canonical transcript for a gene.
|
|
257
|
+
|
|
258
|
+
Canonical is defined as:
|
|
259
|
+
- transcript with Ensembl_canonical tag (peferred)
|
|
260
|
+
- transcript with longest CDS tagged with appris_principal in the GTF
|
|
261
|
+
- if no appris_principal tags for any tx, use the tx with longest CDS
|
|
262
|
+
# TODO: for the last case, maybe check the protein coding subset first
|
|
263
|
+
|
|
264
|
+
Occasionally there are multiple transcripts tagged as appris_principal
|
|
265
|
+
and with the same longest CDS, we use the first one of those.
|
|
266
|
+
'''
|
|
267
|
+
cdef vector[Tx] canonical
|
|
268
|
+
for i in range(self._transcripts.size()):
|
|
269
|
+
max_score = max(self._canonical)
|
|
270
|
+
if self._canonical[i] == max_score:
|
|
271
|
+
canonical.push_back(self._transcripts[i])
|
|
272
|
+
|
|
273
|
+
if canonical.size() == 0:
|
|
274
|
+
canonical = self._transcripts
|
|
275
|
+
|
|
276
|
+
cdef Tx max_tx
|
|
277
|
+
try:
|
|
278
|
+
max_tx = self._max_by_cds(canonical)
|
|
279
|
+
except ValueError:
|
|
280
|
+
max_tx = self._max_by_exonic(canonical)
|
|
281
|
+
|
|
282
|
+
return self._to_Transcript(max_tx)
|
|
283
|
+
|
|
284
|
+
def in_any_tx_cds(self, pos):
|
|
285
|
+
''' find if a pos is in coding region of any transcript of a gene
|
|
286
|
+
'''
|
|
287
|
+
return any(tx.in_coding_region(pos) for tx in self._transcripts)
|
|
288
|
+
|
|
289
|
+
def distance(self, chrom, pos):
|
|
290
|
+
''' get distance to nearest boundary of a gene
|
|
291
|
+
'''
|
|
292
|
+
# sanatize the chromosome first
|
|
293
|
+
if self.chrom.startswith('chr') and not chrom.startswith('chr'):
|
|
294
|
+
chrom = f'chr{chrom}'
|
|
295
|
+
elif not self.chrom.startswith('chr') and chrom.startswith('chr'):
|
|
296
|
+
chrom = chrom[3:]
|
|
297
|
+
|
|
298
|
+
if self.chrom != chrom:
|
|
299
|
+
return None
|
|
300
|
+
if self.start <= pos <= self.end:
|
|
301
|
+
return 0
|
|
302
|
+
return min(abs(self.start - pos), abs(self.end - pos))
|
|
303
|
+
|
|
304
|
+
cdef class Gencode:
|
|
305
|
+
cdef dict genes
|
|
306
|
+
cdef map[string, vector[GenePoint]] starts, ends
|
|
307
|
+
def __cinit__(self, gencode=None, fasta=None, coding_only=True):
|
|
308
|
+
''' initialise Gencode
|
|
309
|
+
|
|
310
|
+
Args:
|
|
311
|
+
gencode: path to gencode annotations file
|
|
312
|
+
fasta: path to fasta for genome matching annotations build
|
|
313
|
+
coding: restrict to protein_coding only by default
|
|
314
|
+
'''
|
|
315
|
+
if gencode is not None and not Path(gencode).exists():
|
|
316
|
+
raise ValueError(f'cannot find gencode at: {gencode}')
|
|
317
|
+
if fasta is not None and not Path(fasta).exists():
|
|
318
|
+
raise ValueError(f'cannot find fasta at: {fasta}')
|
|
319
|
+
self.genes = {}
|
|
320
|
+
if fasta:
|
|
321
|
+
logging.info(f'opening genome fasta: {fasta}')
|
|
322
|
+
global __genome_
|
|
323
|
+
__genome_ = Fasta(str(fasta))
|
|
324
|
+
logging.info(f'opening gencode annotations: {gencode}')
|
|
325
|
+
cdef vector[NamedTx] transcripts
|
|
326
|
+
cdef Gene curr
|
|
327
|
+
if gencode is not None:
|
|
328
|
+
transcripts = open_gencode(str(gencode).encode('utf8'), coding_only)
|
|
329
|
+
for x in transcripts:
|
|
330
|
+
symbol = x.symbol.decode('utf8')
|
|
331
|
+
if symbol not in self.genes:
|
|
332
|
+
self.genes[symbol] = Gene(symbol.encode('utf8'))
|
|
333
|
+
curr = self.genes[symbol]
|
|
334
|
+
curr.add_tx(x.tx, x.is_canonical)
|
|
335
|
+
self.genes[symbol] = curr
|
|
336
|
+
self._sort()
|
|
337
|
+
|
|
338
|
+
def _sort(self):
|
|
339
|
+
''' index by starts and ends, to speed finding genes in a region
|
|
340
|
+
'''
|
|
341
|
+
for symbol in self.genes:
|
|
342
|
+
gene = self.genes[symbol]
|
|
343
|
+
chrom = gene.chrom.encode('utf8')
|
|
344
|
+
symbol = symbol.encode('utf8')
|
|
345
|
+
|
|
346
|
+
# ensure the chromosome is present
|
|
347
|
+
if self.starts.count(chrom) == 0:
|
|
348
|
+
self.starts[chrom] = []
|
|
349
|
+
if self.ends.count(chrom) == 0:
|
|
350
|
+
self.ends[chrom] = []
|
|
351
|
+
|
|
352
|
+
self.starts[chrom].push_back(GenePoint(gene.start, symbol))
|
|
353
|
+
self.ends[chrom].push_back(GenePoint(gene.end, symbol))
|
|
354
|
+
|
|
355
|
+
# sort start and end coords by position
|
|
356
|
+
for x, values in self.starts:
|
|
357
|
+
self.starts[x] = sorted(values, key=lambda x: x['pos'])
|
|
358
|
+
for x, values in self.ends:
|
|
359
|
+
self.ends[x] = sorted(values, key=lambda x: x['pos'])
|
|
360
|
+
|
|
361
|
+
def __repr__(self):
|
|
362
|
+
return f'Gencode(n_genes={len(self)})'
|
|
363
|
+
def __len__(self):
|
|
364
|
+
return len(self.genes)
|
|
365
|
+
def __getitem__(self, symbol):
|
|
366
|
+
return self.genes[symbol]
|
|
367
|
+
def __iter__(self):
|
|
368
|
+
for x in self.genes:
|
|
369
|
+
yield x
|
|
370
|
+
|
|
371
|
+
def add_gene(self, gene):
|
|
372
|
+
''' add another gene to the Gencode object
|
|
373
|
+
'''
|
|
374
|
+
if gene.symbol not in self.genes:
|
|
375
|
+
self.genes[gene.symbol] = gene
|
|
376
|
+
self._sort()
|
|
377
|
+
|
|
378
|
+
def nearest(self, str chrom, int pos):
|
|
379
|
+
''' find the nearest gene to a genomic chrom, pos coordinate
|
|
380
|
+
'''
|
|
381
|
+
chrom = f'chr{chrom}' if not chrom.startswith('chr') else chrom
|
|
382
|
+
_chrom = chrom.encode('utf8')
|
|
383
|
+
|
|
384
|
+
# first, account for any overlapping genespython
|
|
385
|
+
overlaps = self.in_region(chrom, pos-1, pos+1) # NOTE: possibly fix?
|
|
386
|
+
if len(overlaps) > 0:
|
|
387
|
+
# if we have > 0 prioritise if the position is in the CDS
|
|
388
|
+
cds_overlaps = [x for x in overlaps if x.in_any_tx_cds(pos)]
|
|
389
|
+
if len(cds_overlaps) > 0:
|
|
390
|
+
overlaps = cds_overlaps
|
|
391
|
+
# prioritise the gene with longest CDS (in the canonical tx)
|
|
392
|
+
txs = [x.canonical for x in overlaps]
|
|
393
|
+
lengths = [x.get_coding_distance(x.get_cds_end())['pos'] for x in txs]
|
|
394
|
+
idx = lengths.index(max(lengths))
|
|
395
|
+
return overlaps[idx]
|
|
396
|
+
|
|
397
|
+
# no overlaps observed, look for the nearest upstream or downstream gene
|
|
398
|
+
cdef GenePoint site = GenePoint(pos, b'A');
|
|
399
|
+
cdef unsigned int i = lower_bound(self.starts[_chrom].begin(), self.starts[_chrom].end(), site, &CompFunc) - self.starts[_chrom].begin()
|
|
400
|
+
cdef unsigned int j = upper_bound(self.ends[_chrom].begin(), self.ends[_chrom].begin(), site, &CompFunc) - self.ends[_chrom].begin()
|
|
401
|
+
|
|
402
|
+
i = min(i, self.starts[_chrom].size() - 1)
|
|
403
|
+
j = min(j, self.starts[_chrom].size() - 1)
|
|
404
|
+
|
|
405
|
+
downstream = self[self.starts[_chrom][i].symbol.decode('utf8')]
|
|
406
|
+
upstream = self[self.starts[_chrom][j].symbol.decode('utf8')]
|
|
407
|
+
|
|
408
|
+
if upstream.distance(chrom, pos) <= downstream.distance(chrom, pos):
|
|
409
|
+
return upstream
|
|
410
|
+
else:
|
|
411
|
+
return downstream
|
|
412
|
+
|
|
413
|
+
def in_region(self, str _chrom, int start, int end, int max_window=2500000):
|
|
414
|
+
''' find genes within a genomic region
|
|
415
|
+
|
|
416
|
+
Args:
|
|
417
|
+
chrom: chromosome to search on
|
|
418
|
+
start: start position of region
|
|
419
|
+
end: end position of region
|
|
420
|
+
max_window: some genes encapsulate the region, which means we have
|
|
421
|
+
to account for gene lengths of up to 2.3 Mb in the human genome.
|
|
422
|
+
This permits extra search space in other organisms.
|
|
423
|
+
|
|
424
|
+
Returns:
|
|
425
|
+
list of Gene objects
|
|
426
|
+
'''
|
|
427
|
+
symbols = _in_region(_chrom.encode('utf8'), start, end, self.starts,
|
|
428
|
+
self.ends, max_window)
|
|
429
|
+
return [self[x.decode('utf8')] for x in symbols]
|
|
430
|
+
|
|
431
|
+
def __exit__(self):
|
|
432
|
+
''' cleanup at exit
|
|
433
|
+
'''
|
|
434
|
+
global __genome_
|
|
435
|
+
__genome_.close()
|
|
436
|
+
__genome_ = None
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
# cython: language_level=3, boundscheck=False
|
|
2
|
+
'''
|
|
3
|
+
Copyright (c) 2015 Genome Research Ltd.
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy of
|
|
6
|
+
this software and associated documentation files (the "Software"), to deal in
|
|
7
|
+
the Software without restriction, including without limitation the rights to
|
|
8
|
+
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies
|
|
9
|
+
of the Software, and to permit persons to whom the Software is furnished to do
|
|
10
|
+
so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
|
17
|
+
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
|
18
|
+
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
|
|
19
|
+
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
|
20
|
+
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
|
21
|
+
'''
|
|
22
|
+
|
|
23
|
+
from libcpp.vector cimport vector
|
|
24
|
+
from libcpp.string cimport string
|
|
25
|
+
from libcpp cimport bool
|
|
26
|
+
|
|
27
|
+
cdef extern from "tx.h":
|
|
28
|
+
cdef cppclass Tx:
|
|
29
|
+
Tx(string, string, int, int, char, string) except +
|
|
30
|
+
Tx() except +
|
|
31
|
+
|
|
32
|
+
void set_exons(vector[vector[int]], vector[vector[int]]) except +
|
|
33
|
+
void set_cds(vector[vector[int]]) except +
|
|
34
|
+
Region fix_cds_boundary(int) except +
|
|
35
|
+
|
|
36
|
+
vector[Region] get_exons()
|
|
37
|
+
vector[Region] get_cds()
|
|
38
|
+
string get_name()
|
|
39
|
+
string get_chrom()
|
|
40
|
+
int get_start()
|
|
41
|
+
int get_end()
|
|
42
|
+
char get_strand()
|
|
43
|
+
string get_type()
|
|
44
|
+
int get_cds_start()
|
|
45
|
+
int get_cds_end()
|
|
46
|
+
|
|
47
|
+
bool is_exonic(int)
|
|
48
|
+
int closest_exon_num(int)
|
|
49
|
+
Region get_closest_exon(int)
|
|
50
|
+
bool in_coding_region(int)
|
|
51
|
+
CDS_coords to_closest_exon(int, Region)
|
|
52
|
+
CDS_coords get_coding_distance(int) except +
|
|
53
|
+
|
|
54
|
+
int get_position_on_chrom(int, int) except +
|
|
55
|
+
int get_codon_number_for_cds_position(int)
|
|
56
|
+
int get_position_within_codon(int)
|
|
57
|
+
void add_cds_sequence(string)
|
|
58
|
+
void add_genomic_sequence(string, int) except +
|
|
59
|
+
string get_cds_sequence()
|
|
60
|
+
string get_genomic_sequence()
|
|
61
|
+
int get_genomic_offset()
|
|
62
|
+
|
|
63
|
+
string reverse_complement(string)
|
|
64
|
+
string get_centered_sequence(int, int) except +
|
|
65
|
+
string get_codon_sequence(int) except +
|
|
66
|
+
string get_seq_in_region(int, int) except +
|
|
67
|
+
string translate(string) except +
|
|
68
|
+
|
|
69
|
+
Codon get_codon_info(int) except +
|
|
70
|
+
int get_boundary_distance(int) except +
|
|
71
|
+
string consequence(int, string, string) except +
|
|
72
|
+
|
|
73
|
+
cdef struct CDS_coords:
|
|
74
|
+
int position
|
|
75
|
+
int offset
|
|
76
|
+
|
|
77
|
+
cdef struct Region:
|
|
78
|
+
int start
|
|
79
|
+
int end
|
|
80
|
+
|
|
81
|
+
cdef struct Codon:
|
|
82
|
+
int cds_pos
|
|
83
|
+
string codon_seq
|
|
84
|
+
int intra_codon
|
|
85
|
+
int codon_number
|
|
86
|
+
char initial_aa
|
|
87
|
+
int offset
|
|
88
|
+
|
|
89
|
+
cdef class Transcript:
|
|
90
|
+
cdef Tx *thisptr # hold a C++ instance which we're wrapping
|
|
@@ -0,0 +1,378 @@
|
|
|
1
|
+
# cython: language_level=3, boundscheck=False
|
|
2
|
+
'''
|
|
3
|
+
Copyright (c) 2015 Genome Research Ltd.
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy of
|
|
6
|
+
this software and associated documentation files (the "Software"), to deal in
|
|
7
|
+
the Software without restriction, including without limitation the rights to
|
|
8
|
+
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies
|
|
9
|
+
of the Software, and to permit persons to whom the Software is furnished to do
|
|
10
|
+
so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
|
17
|
+
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
|
18
|
+
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
|
|
19
|
+
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
|
20
|
+
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
|
21
|
+
'''
|
|
22
|
+
|
|
23
|
+
from itertools import combinations
|
|
24
|
+
|
|
25
|
+
cdef class Transcript:
|
|
26
|
+
def __cinit__(self, name, chrom, start, end, strand,
|
|
27
|
+
transcript_type='protein_coding', exons=None, cds=None, sequence=None,
|
|
28
|
+
offset=0):
|
|
29
|
+
''' construct a Transcript object
|
|
30
|
+
|
|
31
|
+
Args:
|
|
32
|
+
name: ID of the transcript
|
|
33
|
+
start: position in bp at 5' edge of transcript (on + strand)
|
|
34
|
+
end: position in bp at 3' edge of transcript (on + strand)
|
|
35
|
+
exons: list of tuples defining start and end positions of exons
|
|
36
|
+
cds: list of tuples defining start and end positions of CDS regions
|
|
37
|
+
sequence: DNA sequence of genome region of the transcript.
|
|
38
|
+
offset: how many base pairs the DNA sequence extends outwards
|
|
39
|
+
'''
|
|
40
|
+
|
|
41
|
+
name = name.encode('utf8')
|
|
42
|
+
chrom = chrom.encode('utf8')
|
|
43
|
+
transcript_type = transcript_type.encode('utf8')
|
|
44
|
+
self.thisptr = new Tx(name, chrom, start, end, ord(strand), transcript_type)
|
|
45
|
+
|
|
46
|
+
if exons is not None and cds is not None:
|
|
47
|
+
self.set_exons(exons, cds)
|
|
48
|
+
self.set_cds(cds)
|
|
49
|
+
|
|
50
|
+
if sequence is not None:
|
|
51
|
+
self.add_genomic_sequence(sequence, offset)
|
|
52
|
+
|
|
53
|
+
def __dealloc__(self):
|
|
54
|
+
del self.thisptr
|
|
55
|
+
|
|
56
|
+
def __repr__(self):
|
|
57
|
+
|
|
58
|
+
exons = [ (x['start'], x['end']) for x in self.get_exons() ]
|
|
59
|
+
cds = [ (x['start'], x['end']) for x in self.get_cds() ]
|
|
60
|
+
seq = self.get_genomic_sequence()
|
|
61
|
+
|
|
62
|
+
if len(seq) > 40:
|
|
63
|
+
seq = seq[:20] + '...[{} bp]...'.format(len(seq) - 40) + seq[-20:]
|
|
64
|
+
|
|
65
|
+
if exons == []:
|
|
66
|
+
exons = None
|
|
67
|
+
|
|
68
|
+
if cds == []:
|
|
69
|
+
cds = None
|
|
70
|
+
|
|
71
|
+
if seq == '':
|
|
72
|
+
seq = None
|
|
73
|
+
else:
|
|
74
|
+
seq = '"' + seq + '"'
|
|
75
|
+
|
|
76
|
+
return f'Transcript(name="{self.get_name()}", chrom="{self.get_chrom()}", ' \
|
|
77
|
+
f'start={self.get_start()}, end={self.get_end()}, strand="{self.get_strand()}", ' \
|
|
78
|
+
f'transcript_type="{self.get_type()}", exons={exons}, cds={cds}, ' \
|
|
79
|
+
f'sequence={seq}, offset={self.get_genomic_offset()})'
|
|
80
|
+
|
|
81
|
+
def __str__(self):
|
|
82
|
+
return self.__repr__()
|
|
83
|
+
|
|
84
|
+
def __hash__(self):
|
|
85
|
+
return hash((self.thisptr.get_chrom(), self.thisptr.get_start(),
|
|
86
|
+
self.thisptr.get_end()))
|
|
87
|
+
|
|
88
|
+
def __richcmp__(self, other, op):
|
|
89
|
+
|
|
90
|
+
if op == 2:
|
|
91
|
+
return self.__hash__() == other.__hash__()
|
|
92
|
+
else:
|
|
93
|
+
err_msg = "op {0} isn't implemented yet".format(op)
|
|
94
|
+
raise NotImplementedError(err_msg)
|
|
95
|
+
|
|
96
|
+
def set_exons(self, exon_ranges, cds_ranges):
|
|
97
|
+
''' add exon ranges
|
|
98
|
+
|
|
99
|
+
Args:
|
|
100
|
+
exon_ranges: a CDS position of the selected base.
|
|
101
|
+
'''
|
|
102
|
+
self.thisptr.set_exons(exon_ranges, cds_ranges)
|
|
103
|
+
|
|
104
|
+
def get_overlaps(self, exon, regions):
|
|
105
|
+
''' find all regions which overlap a given region
|
|
106
|
+
'''
|
|
107
|
+
return [ i for i, x in enumerate(regions) if
|
|
108
|
+
exon['start'] <= x['end'] and exon['end'] >= x['start'] ]
|
|
109
|
+
|
|
110
|
+
def insert_region(self, coords, region):
|
|
111
|
+
''' include a region into a list of regions
|
|
112
|
+
|
|
113
|
+
To include a region, we have to check which pre-existing regions the new
|
|
114
|
+
region overlaps, so any overlaps can be merged into a single region.
|
|
115
|
+
|
|
116
|
+
Args:
|
|
117
|
+
coords: list of {start: X, end: Y} dictionaries
|
|
118
|
+
region: dict of {'start': X, 'end': Y} positions
|
|
119
|
+
'''
|
|
120
|
+
indices = self.get_overlaps(region, coords)
|
|
121
|
+
overlaps = [ coords[i] for i in indices ]
|
|
122
|
+
start = min( x['start'] for x in overlaps + [region] )
|
|
123
|
+
end = max( x['end'] for x in overlaps + [region] )
|
|
124
|
+
|
|
125
|
+
for i in sorted(indices, reverse=True):
|
|
126
|
+
del coords[i]
|
|
127
|
+
|
|
128
|
+
return coords + [{'start': start, 'end': end}]
|
|
129
|
+
|
|
130
|
+
def merge_coordinates(self, first, second):
|
|
131
|
+
''' merge two sets of coordinates, to get the union of regions
|
|
132
|
+
|
|
133
|
+
This uses an inefficient approach, looping over and over, but we won't
|
|
134
|
+
need to perform this often.
|
|
135
|
+
|
|
136
|
+
Args:
|
|
137
|
+
first: list of {'start': x, 'end': y} dictionaries for first transcript
|
|
138
|
+
second: list of {'start': x, 'end': y} dictionaries for second transcript
|
|
139
|
+
|
|
140
|
+
Returns:
|
|
141
|
+
list of [start, end] lists, sorted by position.
|
|
142
|
+
'''
|
|
143
|
+
coords = []
|
|
144
|
+
for a, b in combinations(first + second, 2):
|
|
145
|
+
if a['start'] <= b['end'] and a['end'] >= b['start']:
|
|
146
|
+
region = {'start': min(a['start'], b['start']),
|
|
147
|
+
'end': max(a['end'], b['end'])}
|
|
148
|
+
a, b = region, region
|
|
149
|
+
|
|
150
|
+
coords = self.insert_region(coords, a)
|
|
151
|
+
coords = self.insert_region(coords, b)
|
|
152
|
+
|
|
153
|
+
return [ (x['start'], x['end']) for x in sorted(coords, key=lambda x: x['start']) ]
|
|
154
|
+
|
|
155
|
+
def merge_genomic_seq(self, other):
|
|
156
|
+
''' merge the genomic sequence from two transcripts
|
|
157
|
+
|
|
158
|
+
We have two transcripts A, and B. We need to get the contiguous sequence
|
|
159
|
+
from the start of the first transcript on the chromosome to the end of
|
|
160
|
+
the second transcript. The transcripts may or may not overlap. There are
|
|
161
|
+
three scenarios we need to account for:
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
overlap without A =================
|
|
165
|
+
enveloping ================= B
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
overlap A ==============================
|
|
169
|
+
and envelop ================== B
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
no overlap A ===============
|
|
173
|
+
=============== B
|
|
174
|
+
|
|
175
|
+
I've called the transcript whose sequence is first along the chromosome
|
|
176
|
+
as 'lead', and the transcript whose sequence is last as 'lag', and the
|
|
177
|
+
converse as 'not_lead', and 'not_lag'. Note that in the envelope case,
|
|
178
|
+
the lead transcript can also be the lag transcript.
|
|
179
|
+
'''
|
|
180
|
+
|
|
181
|
+
# make sure that the surrounding sequence is the same length in both
|
|
182
|
+
# transcripts.
|
|
183
|
+
# TODO: this could be worked around, by figuring the minimum offset length,
|
|
184
|
+
# TODO: then trimming the respective DNA offset sequences to that length.
|
|
185
|
+
assert self.get_genomic_offset() == other.get_genomic_offset()
|
|
186
|
+
|
|
187
|
+
# figure out which transcripts hold the leading and lagging sections
|
|
188
|
+
lead, not_lead = self, other
|
|
189
|
+
if self.get_start() > other.get_start():
|
|
190
|
+
lead, not_lead = other, self
|
|
191
|
+
|
|
192
|
+
lag, not_lag = self, other
|
|
193
|
+
if self.get_end() < other.get_end():
|
|
194
|
+
lag, not_lag = other, self
|
|
195
|
+
|
|
196
|
+
lead_offset = lead.get_genomic_offset()
|
|
197
|
+
lead_gdna = lead.get_genomic_sequence()
|
|
198
|
+
initial = lead_gdna[:not_lead.get_start() - lead.get_start() + lead_offset]
|
|
199
|
+
|
|
200
|
+
if self.get_start() <= other.get_end() and self.get_end() >= other.get_start():
|
|
201
|
+
intersect_start = not_lead.get_start() - lead.get_start() + lead_offset
|
|
202
|
+
intersect_end = not_lag.get_end() - lead.get_start() + lead_offset
|
|
203
|
+
intersect = lead_gdna[intersect_start:intersect_end]
|
|
204
|
+
else:
|
|
205
|
+
intersect = 'N' * (lag.get_start() - lead.get_end() - lead_offset * 2)
|
|
206
|
+
|
|
207
|
+
lag_offset = lag.get_genomic_offset()
|
|
208
|
+
lag_gdna = lag.get_genomic_sequence()
|
|
209
|
+
|
|
210
|
+
# some transcripts overlap, but some do not. We need to find the position
|
|
211
|
+
# where the lagging transcript takes over, which is either at the end of
|
|
212
|
+
# not lagging transcript, or the start of the lagging transcript,
|
|
213
|
+
# whichever is higher
|
|
214
|
+
pos = max(not_lag.get_end(), lag.get_start())
|
|
215
|
+
final = lag_gdna[pos - lag.get_start() + lead_offset:]
|
|
216
|
+
|
|
217
|
+
return initial + intersect + final
|
|
218
|
+
|
|
219
|
+
def __add__(self, other):
|
|
220
|
+
""" combine the coding sequences of two Transcript objects
|
|
221
|
+
|
|
222
|
+
When we determine the sites for sampling, occasioally we want to
|
|
223
|
+
use sites from multiple alternative transcripts. We determine the sites
|
|
224
|
+
for each transcript in turn, but mask the sites that have been collected
|
|
225
|
+
in the preceeding transcripts. In order to be able to mask all previous
|
|
226
|
+
trabnscripts, we need to combine the coding sequence of the transcripts
|
|
227
|
+
as we move through them. This function performs the union of coding
|
|
228
|
+
sequence regions between different transcripts.
|
|
229
|
+
|
|
230
|
+
We do this outside of the c++ class, so as to be able to set up a
|
|
231
|
+
Transcript object correctly.
|
|
232
|
+
|
|
233
|
+
Args:
|
|
234
|
+
other: a transcript to be combined with the current object.
|
|
235
|
+
|
|
236
|
+
Returns:
|
|
237
|
+
an altered instance of the class, where the coding sequence regions
|
|
238
|
+
are the union of the coding regions of two Transcript objects. This
|
|
239
|
+
disrupts the ability to get meaningingful sequence from the object,
|
|
240
|
+
so don't try to extract sequence from the returned object.
|
|
241
|
+
"""
|
|
242
|
+
|
|
243
|
+
# if we try transcript + None or None + transcript, return the original
|
|
244
|
+
# transcript, rather than raising an error.
|
|
245
|
+
if other is None:
|
|
246
|
+
return self
|
|
247
|
+
if self is None:
|
|
248
|
+
return other
|
|
249
|
+
|
|
250
|
+
altered = Transcript('{}:{}'.format(self.get_name(), other.get_name()),
|
|
251
|
+
self.get_chrom(), min(self.get_start(), other.get_start()),
|
|
252
|
+
max(self.get_end(), other.get_end()), self.get_strand(), self.get_type())
|
|
253
|
+
|
|
254
|
+
exons = self.merge_coordinates(self.get_exons(), other.get_exons())
|
|
255
|
+
cds = self.merge_coordinates(self.get_cds(), other.get_cds())
|
|
256
|
+
|
|
257
|
+
altered.set_exons(exons, cds)
|
|
258
|
+
altered.set_cds(cds)
|
|
259
|
+
|
|
260
|
+
if self.get_genomic_sequence() != "":
|
|
261
|
+
altered.add_genomic_sequence(self.merge_genomic_seq(other), self.get_genomic_offset())
|
|
262
|
+
|
|
263
|
+
return altered
|
|
264
|
+
|
|
265
|
+
def set_cds(self, cds_ranges):
|
|
266
|
+
''' set CDS ranges
|
|
267
|
+
|
|
268
|
+
Args:
|
|
269
|
+
cds_ranges: a CDS position of the selected base.
|
|
270
|
+
'''
|
|
271
|
+
|
|
272
|
+
self.thisptr.set_cds(cds_ranges)
|
|
273
|
+
|
|
274
|
+
def get_genomic_offset(self):
|
|
275
|
+
return self.thisptr.get_genomic_offset()
|
|
276
|
+
def get_exons(self):
|
|
277
|
+
return self.thisptr.get_exons()
|
|
278
|
+
def get_cds(self):
|
|
279
|
+
return self.thisptr.get_cds()
|
|
280
|
+
def get_name(self):
|
|
281
|
+
return self.thisptr.get_name().decode('utf8')
|
|
282
|
+
def get_chrom(self):
|
|
283
|
+
return self.thisptr.get_chrom().decode('utf8')
|
|
284
|
+
def get_type(self):
|
|
285
|
+
return self.thisptr.get_type().decode('utf8')
|
|
286
|
+
def get_start(self):
|
|
287
|
+
return self.thisptr.get_start()
|
|
288
|
+
def get_end(self):
|
|
289
|
+
return self.thisptr.get_end()
|
|
290
|
+
def get_strand(self):
|
|
291
|
+
return chr(self.thisptr.get_strand())
|
|
292
|
+
def get_cds_start(self):
|
|
293
|
+
return self.thisptr.get_cds_start()
|
|
294
|
+
def get_cds_end(self):
|
|
295
|
+
return self.thisptr.get_cds_end()
|
|
296
|
+
def fix_cds_boundary(self, pos):
|
|
297
|
+
return self.thisptr.fix_cds_boundary(pos)
|
|
298
|
+
|
|
299
|
+
def in_exons(self, position):
|
|
300
|
+
''' check if a site lies within the exon ranges
|
|
301
|
+
|
|
302
|
+
Args:
|
|
303
|
+
position: an integer-based chromosome position.
|
|
304
|
+
'''
|
|
305
|
+
|
|
306
|
+
return self.thisptr.is_exonic(position)
|
|
307
|
+
|
|
308
|
+
def get_closest_exon(self, position):
|
|
309
|
+
''' finds the positions of the exon closest to a position
|
|
310
|
+
'''
|
|
311
|
+
return self.thisptr.get_closest_exon(position)
|
|
312
|
+
|
|
313
|
+
def in_coding_region(self, position):
|
|
314
|
+
return self.thisptr.in_coding_region(position)
|
|
315
|
+
|
|
316
|
+
def get_coding_distance(self, pos):
|
|
317
|
+
''' get distance to CDS start (and intronic offset)
|
|
318
|
+
'''
|
|
319
|
+
coords = self.thisptr.get_coding_distance(pos)
|
|
320
|
+
|
|
321
|
+
return {'pos': coords.position, 'offset': coords.offset}
|
|
322
|
+
|
|
323
|
+
def get_position_on_chrom(self, pos, offset=0):
|
|
324
|
+
return self.thisptr.get_position_on_chrom(pos, offset)
|
|
325
|
+
|
|
326
|
+
def get_codon_number_for_cds_position(self, pos):
|
|
327
|
+
return self.thisptr.get_codon_number_for_cds_position(pos)
|
|
328
|
+
|
|
329
|
+
def get_position_within_codon(self, pos):
|
|
330
|
+
return self.thisptr.get_position_within_codon(pos)
|
|
331
|
+
|
|
332
|
+
def add_cds_sequence(self, text):
|
|
333
|
+
self.thisptr.add_cds_sequence(text.encode('utf8'))
|
|
334
|
+
|
|
335
|
+
def get_cds_sequence(self):
|
|
336
|
+
return self.thisptr.get_cds_sequence().decode('utf8')
|
|
337
|
+
|
|
338
|
+
def add_genomic_sequence(self, text, offset=0):
|
|
339
|
+
self.thisptr.add_genomic_sequence(text.encode('utf8'), offset)
|
|
340
|
+
|
|
341
|
+
def get_genomic_sequence(self):
|
|
342
|
+
return self.thisptr.get_genomic_sequence().decode('utf8')
|
|
343
|
+
|
|
344
|
+
def reverse_complement(self, text):
|
|
345
|
+
return self.thisptr.reverse_complement(text).decode('utf8')
|
|
346
|
+
|
|
347
|
+
def get_centered_sequence(self, pos, length=3):
|
|
348
|
+
return self.thisptr.get_centered_sequence(pos, length).decode('utf8')
|
|
349
|
+
|
|
350
|
+
def get_codon_sequence(self, pos):
|
|
351
|
+
return self.thisptr.get_codon_sequence(pos).decode('utf8')
|
|
352
|
+
|
|
353
|
+
def translate(self, text):
|
|
354
|
+
return self.thisptr.translate(text.encode('utf8')).decode('utf8')
|
|
355
|
+
|
|
356
|
+
def get_codon_info(self, pos):
|
|
357
|
+
codon = dict(self.thisptr.get_codon_info(pos))
|
|
358
|
+
|
|
359
|
+
if codon['codon_number'] == -9999999:
|
|
360
|
+
codon['codon_number'] = None
|
|
361
|
+
codon['intra_codon'] = None
|
|
362
|
+
codon['codon_seq'] = None
|
|
363
|
+
codon['initial_aa'] = None
|
|
364
|
+
|
|
365
|
+
if codon['codon_seq'] is not None:
|
|
366
|
+
codon['codon_seq'] = codon['codon_seq'].decode('utf8')
|
|
367
|
+
|
|
368
|
+
if codon['initial_aa'] is not None:
|
|
369
|
+
codon['initial_aa'] = chr(codon['initial_aa'])
|
|
370
|
+
|
|
371
|
+
return codon
|
|
372
|
+
|
|
373
|
+
def get_boundary_distance(self, pos):
|
|
374
|
+
return self.thisptr.get_boundary_distance(pos)
|
|
375
|
+
|
|
376
|
+
def consequence(self, pos, ref, alt):
|
|
377
|
+
cq = self.thisptr.consequence(pos, ref.encode('utf8'), alt.encode('utf8'))
|
|
378
|
+
return cq.decode('utf8')
|
|
@@ -11,7 +11,10 @@ src/tx.cpp
|
|
|
11
11
|
src/tx.h
|
|
12
12
|
src/gencodegenes/__init__.py
|
|
13
13
|
src/gencodegenes/gencode.cpp
|
|
14
|
+
src/gencodegenes/gencode.pyx
|
|
14
15
|
src/gencodegenes/transcript.cpp
|
|
16
|
+
src/gencodegenes/transcript.pxd
|
|
17
|
+
src/gencodegenes/transcript.pyx
|
|
15
18
|
src/gencodegenes.egg-info/PKG-INFO
|
|
16
19
|
src/gencodegenes.egg-info/SOURCES.txt
|
|
17
20
|
src/gencodegenes.egg-info/dependency_links.txt
|
gencodegenes-1.0.0/MANIFEST.in
DELETED
|
@@ -1,15 +0,0 @@
|
|
|
1
|
-
# genecodegenes code
|
|
2
|
-
include MANIFEST.in
|
|
3
|
-
include pyproject.toml
|
|
4
|
-
include genecodegenes/*.cpp
|
|
5
|
-
include genecodegenes/*.py
|
|
6
|
-
include genecodegenes/*.pyx
|
|
7
|
-
include genecodegenes/*.pxd
|
|
8
|
-
include src/*.h
|
|
9
|
-
include src/*.cpp
|
|
10
|
-
include src/gzstream/gzstream.C
|
|
11
|
-
include src/gzstream/gzstream.h
|
|
12
|
-
include data/*.txt
|
|
13
|
-
|
|
14
|
-
# genecodegenes tests
|
|
15
|
-
include tests/*.py
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|