gencodegenes 1.0.0__tar.gz → 1.0.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. gencodegenes-1.0.2/MANIFEST.in +16 -0
  2. {gencodegenes-1.0.0 → gencodegenes-1.0.2}/PKG-INFO +1 -1
  3. {gencodegenes-1.0.0 → gencodegenes-1.0.2}/setup.py +1 -1
  4. gencodegenes-1.0.2/src/gencodegenes/gencode.pyx +436 -0
  5. gencodegenes-1.0.2/src/gencodegenes/transcript.pxd +90 -0
  6. gencodegenes-1.0.2/src/gencodegenes/transcript.pyx +378 -0
  7. {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gencodegenes.egg-info/PKG-INFO +1 -1
  8. {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gencodegenes.egg-info/SOURCES.txt +3 -0
  9. gencodegenes-1.0.0/MANIFEST.in +0 -15
  10. {gencodegenes-1.0.0 → gencodegenes-1.0.2}/LICENSE.txt +0 -0
  11. {gencodegenes-1.0.0 → gencodegenes-1.0.2}/README.md +0 -0
  12. {gencodegenes-1.0.0 → gencodegenes-1.0.2}/pyproject.toml +0 -0
  13. {gencodegenes-1.0.0 → gencodegenes-1.0.2}/setup.cfg +0 -0
  14. {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gencode.cpp +0 -0
  15. {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gencode.h +0 -0
  16. {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gencodegenes/__init__.py +0 -0
  17. {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gencodegenes/gencode.cpp +0 -0
  18. {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gencodegenes/transcript.cpp +0 -0
  19. {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gencodegenes.egg-info/dependency_links.txt +0 -0
  20. {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gencodegenes.egg-info/requires.txt +0 -0
  21. {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gencodegenes.egg-info/top_level.txt +0 -0
  22. {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gtf.cpp +0 -0
  23. {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gtf.h +0 -0
  24. {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gzstream/gzstream.C +0 -0
  25. {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/gzstream/gzstream.h +0 -0
  26. {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/tx.cpp +0 -0
  27. {gencodegenes-1.0.0 → gencodegenes-1.0.2}/src/tx.h +0 -0
  28. {gencodegenes-1.0.0 → gencodegenes-1.0.2}/tests/__init__.py +0 -0
  29. {gencodegenes-1.0.0 → gencodegenes-1.0.2}/tests/test_gencode.py +0 -0
  30. {gencodegenes-1.0.0 → gencodegenes-1.0.2}/tests/test_transcript.py +0 -0
@@ -0,0 +1,16 @@
1
+ # gencodegenes code
2
+ include MANIFEST.in
3
+ include pyproject.toml
4
+ include src/gencodegenes/*.cpp
5
+ include src/gencodegenes/*.py
6
+ include src/gencodegenes/*.pyx
7
+ include src/gencodegenes/*.pxd
8
+ include src/gencodegenes/data/rates.txt
9
+ include src/*.h
10
+ include src/*.cpp
11
+ include src/gzstream/gzstream.C
12
+ include src/gzstream/gzstream.h
13
+ include data/*.txt
14
+
15
+ # gencodegenes tests
16
+ include tests/*.py
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: gencodegenes
3
- Version: 1.0.0
3
+ Version: 1.0.2
4
4
  Summary: Package to load genes from GENCODE GTF files
5
5
  Home-page: https://github.com/jeremymcrae/gencodegenes
6
6
  Author: Jeremy McRae
@@ -107,7 +107,7 @@ setup(name="gencodegenes",
107
107
  description='Package to load genes from GENCODE GTF files',
108
108
  long_description=io.open('README.md', encoding='utf-8').read(),
109
109
  long_description_content_type='text/markdown',
110
- version="1.0.0",
110
+ version="1.0.2",
111
111
  author="Jeremy McRae",
112
112
  author_email="jeremy.mcrae@gmail.com",
113
113
  license="MIT",
@@ -0,0 +1,436 @@
1
+ # cython: language_level=3, boundscheck=False, emit_linenums=True
2
+
3
+ import bisect
4
+ import logging
5
+ from pathlib import Path
6
+
7
+ from libcpp.algorithm cimport lower_bound, upper_bound
8
+ from libcpp.vector cimport vector
9
+ from libcpp.string cimport string
10
+ from libcpp cimport bool
11
+ from libcpp.map cimport map
12
+
13
+ from pyfaidx import Fasta
14
+
15
+ from gencodegenes.transcript cimport Tx, Region, CDS_coords
16
+ from gencodegenes.transcript import Transcript
17
+
18
+ cdef extern from "gtf.h" namespace "gencode":
19
+ cdef struct GTFLine:
20
+ string chrom
21
+ string feature
22
+ int start
23
+ int end
24
+ string strand
25
+ string symbol
26
+ string tx_id
27
+ string transcript_type
28
+ int is_canonical
29
+
30
+ GTFLine parse_gtfline(string line)
31
+
32
+ cdef extern from "gencode.h" namespace "gencode":
33
+ cdef struct NamedTx:
34
+ string symbol
35
+ Tx tx
36
+ int is_canonical
37
+
38
+ cdef struct GenePoint:
39
+ int pos
40
+ string symbol
41
+
42
+ vector[NamedTx] open_gencode(string, bool)
43
+ bool CompFunc(const GenePoint &l, const GenePoint &r)
44
+ vector[string] _in_region(string chrom, int start, int end,
45
+ map[string, vector[GenePoint]] & starts, map[string, vector[GenePoint]] & ends,
46
+ int max_window) except+
47
+
48
+ cpdef _parse_gtfline(string line):
49
+ ''' python function for unit testing GTF parsing
50
+ '''
51
+ return parse_gtfline(line)
52
+
53
+ cdef _convert_exons(vector[Region] exons):
54
+ ''' convert vector of exon Regions to list of lists
55
+
56
+ We need exons and CDS as lists of lists for constructing the python
57
+ Transcript object.
58
+ '''
59
+ return [[y.start, y.end] for y in exons]
60
+
61
+ cpdef _open_gencode(gtf_path, coding_only=True):
62
+ ''' python function for unit testing loading transcripts from GTF
63
+ '''
64
+ cdef vector[NamedTx] _transcripts = open_gencode(gtf_path.encode('utf8'), coding_only)
65
+
66
+ transcripts = []
67
+ for x in _transcripts:
68
+ tx = x.tx
69
+ chrom = tx.get_chrom().decode('utf8')
70
+ start = tx.get_start()
71
+ end = tx.get_end()
72
+ exons = _convert_exons(tx.get_exons())
73
+ cds = _convert_exons(tx.get_cds())
74
+ strand = chr(tx.get_strand())
75
+ tx_id = tx.get_name().decode('utf8')
76
+ transcript_type = tx.get_type().decode('utf8')
77
+ transcript = Transcript(tx_id, chrom, start, end, strand, transcript_type, exons, cds, offset=0)
78
+ transcripts.append((x.symbol.decode('utf8'), transcript, x.is_canonical))
79
+ return transcripts
80
+
81
+ __genome_ = None
82
+
83
+ cdef class Gene:
84
+ cdef string _symbol
85
+ cdef vector[Tx] _transcripts
86
+ cdef vector[int] _canonical
87
+ cdef str _chrom
88
+ cdef int _start, _end
89
+ def __cinit__(self, symbol):
90
+ if isinstance(symbol, str):
91
+ symbol = symbol.encode('utf8')
92
+ self._symbol = symbol
93
+ self.start = 999999999
94
+ self.end = -999999999
95
+
96
+ cdef add_tx(self, Tx tx, int is_canonical):
97
+ self._transcripts.push_back(tx)
98
+ self._canonical.push_back(is_canonical)
99
+ self.chrom = tx.get_chrom().decode('utf8')
100
+ self.start = min(self.start, tx.get_start())
101
+ self.end = max(self.end, tx.get_end())
102
+
103
+ def add_transcript(self, _tx):
104
+ ''' add a Transcript to the gene object
105
+
106
+ This ends up coping the data from the Transcript object, rather than
107
+ reusing the Tx object contained in the Trnascript, but it's not too much
108
+ time wasted, so long as we don't do this millions of times.
109
+ '''
110
+ assert isinstance(_tx, Transcript)
111
+ # construct a new Tx obect by copying out the relevant data
112
+ cdef string tx_id = _tx.get_name().encode('utf8')
113
+ cdef string chrom = _tx.get_chrom().encode('utf8')
114
+ cdef int start = _tx.get_start()
115
+ cdef int end = _tx.get_end()
116
+ cdef char strand = ord(_tx.get_strand())
117
+ cdef string tx_type = _tx.get_type().encode('utf8')
118
+ cdef vector[vector[int]] exons
119
+ cdef vector[vector[int]] cds
120
+ cdef vector[int] exon
121
+ for x in _tx.get_exons():
122
+ exon = [x['start'], x['end']]
123
+ exons.push_back(exon)
124
+ for x in _tx.get_cds():
125
+ exon = [x['start'], x['end']]
126
+ cds.push_back(exon)
127
+
128
+ cdef Tx tx = Tx(tx_id, chrom, start, end, strand, tx_type)
129
+ tx.set_exons(exons, cds)
130
+ tx.set_cds(cds)
131
+
132
+ cdef string seq = _tx.get_genomic_sequence().encode('utf8')
133
+ cdef int offset = _tx.get_genomic_offset()
134
+ if len(seq) > 0:
135
+ if chr(strand) == '-':
136
+ _tx.reverse_complement(seq)
137
+ seq = _tx.reverse_complement(seq).encode('utf8')
138
+ tx.add_genomic_sequence(seq, offset)
139
+ self.add_tx(tx, False)
140
+
141
+ def __repr__(self):
142
+ chrom = self.chrom
143
+ return f'Gene("{self.symbol}", {chrom}:{self.start}-{self.end})'
144
+
145
+ @property
146
+ def symbol(self):
147
+ return self._symbol.decode('utf8')
148
+
149
+ @property
150
+ def chrom(self):
151
+ return self._chrom
152
+ @chrom.setter
153
+ def chrom(self, value):
154
+ self._chrom = value
155
+
156
+ @property
157
+ def start(self):
158
+ return self._start
159
+ @start.setter
160
+ def start(self, value):
161
+ self._start = value
162
+
163
+ @property
164
+ def end(self):
165
+ return self._end
166
+ @end.setter
167
+ def end(self, value):
168
+ self._end = value
169
+
170
+ @property
171
+ def strand(self):
172
+ if self._transcripts.size() > 0:
173
+ return chr(self._transcripts[0].get_strand())
174
+ raise IndexError('no transcripts in gene yet')
175
+
176
+ cdef _convert_exons(self, vector[Region] exons):
177
+ ''' convert vector of exon Regions to list of lists
178
+
179
+ We need exons and CDS as lists of lists for constructing the python
180
+ Transcript object.
181
+ '''
182
+ return [[y.start, y.end] for y in exons]
183
+
184
+ cdef _to_Transcript(self, Tx tx):
185
+ ''' construct Transcript (python object) from Tx (c++ object)
186
+ '''
187
+ offset = 5 if tx.get_genomic_offset() == 0 else tx.get_genomic_offset()
188
+ chrom = tx.get_chrom().decode('utf8')
189
+ start = tx.get_start()
190
+ end = tx.get_end()
191
+ exons = self._convert_exons(tx.get_exons())
192
+ cds = self._convert_exons(tx.get_cds())
193
+ seq = tx.get_genomic_sequence().decode('utf8')
194
+ if seq == '':
195
+ seq = None
196
+ if seq is None and __genome_ is not None:
197
+ seq = __genome_[chrom][start-1-offset:end-1+offset].seq.upper()
198
+ strand = chr(tx.get_strand())
199
+ if strand == '-' and seq is not None:
200
+ seq = tx.reverse_complement(seq.encode('utf8')).decode('utf8')
201
+ tx_id = tx.get_name().decode('utf8')
202
+ tx_type = tx.get_type().decode('utf8')
203
+ return Transcript(tx_id, chrom, start, end, strand, tx_type, exons, cds, seq, offset=offset)
204
+
205
+ @property
206
+ def transcripts(self):
207
+ ''' get list of Transcripts for gene, with genomic DNA included
208
+ '''
209
+ return [self._to_Transcript(x) for x in self._transcripts]
210
+
211
+ cdef int _cds_len(self, Tx tx):
212
+ ''' get length of coding sequence for a Tx object based transcript
213
+ '''
214
+ cdef CDS_coords coords = tx.get_coding_distance(tx.get_cds_end())
215
+ return coords.position + 1
216
+
217
+ cdef Tx _max_by_cds(self, vector[Tx] transcripts) except *:
218
+ ''' get longest transcript by CDS length
219
+ '''
220
+ cdef Tx max_tx
221
+ length = 0
222
+ for tx in transcripts:
223
+ curr_len = self._cds_len(tx)
224
+ if curr_len > length:
225
+ length = curr_len
226
+ max_tx = tx
227
+ if length == 0:
228
+ raise ValueError('no coding transcripts')
229
+ return max_tx
230
+
231
+ cdef int _exonic_len(self, Tx tx):
232
+ ''' get length of exonic sequence for a Tx object based transcript
233
+ '''
234
+ length = 0
235
+ for start, end in _convert_exons(tx.get_exons()):
236
+ length += abs(end - start) + 1
237
+ return length
238
+ # return sum(abs(x.end - x.start) + 1 for x in tx.get_exons())
239
+
240
+ cdef Tx _max_by_exonic(self, vector[Tx] transcripts) except *:
241
+ ''' get longest transcript by CDS length
242
+ '''
243
+ cdef Tx max_tx
244
+ length = 0
245
+ for tx in transcripts:
246
+ curr_len = self._exonic_len(tx)
247
+ if curr_len > length:
248
+ length = curr_len
249
+ max_tx = tx
250
+ if length == 0:
251
+ raise ValueError('no exonic transcripts')
252
+ return max_tx
253
+
254
+ @property
255
+ def canonical(self):
256
+ ''' find the canonical transcript for a gene.
257
+
258
+ Canonical is defined as:
259
+ - transcript with Ensembl_canonical tag (peferred)
260
+ - transcript with longest CDS tagged with appris_principal in the GTF
261
+ - if no appris_principal tags for any tx, use the tx with longest CDS
262
+ # TODO: for the last case, maybe check the protein coding subset first
263
+
264
+ Occasionally there are multiple transcripts tagged as appris_principal
265
+ and with the same longest CDS, we use the first one of those.
266
+ '''
267
+ cdef vector[Tx] canonical
268
+ for i in range(self._transcripts.size()):
269
+ max_score = max(self._canonical)
270
+ if self._canonical[i] == max_score:
271
+ canonical.push_back(self._transcripts[i])
272
+
273
+ if canonical.size() == 0:
274
+ canonical = self._transcripts
275
+
276
+ cdef Tx max_tx
277
+ try:
278
+ max_tx = self._max_by_cds(canonical)
279
+ except ValueError:
280
+ max_tx = self._max_by_exonic(canonical)
281
+
282
+ return self._to_Transcript(max_tx)
283
+
284
+ def in_any_tx_cds(self, pos):
285
+ ''' find if a pos is in coding region of any transcript of a gene
286
+ '''
287
+ return any(tx.in_coding_region(pos) for tx in self._transcripts)
288
+
289
+ def distance(self, chrom, pos):
290
+ ''' get distance to nearest boundary of a gene
291
+ '''
292
+ # sanatize the chromosome first
293
+ if self.chrom.startswith('chr') and not chrom.startswith('chr'):
294
+ chrom = f'chr{chrom}'
295
+ elif not self.chrom.startswith('chr') and chrom.startswith('chr'):
296
+ chrom = chrom[3:]
297
+
298
+ if self.chrom != chrom:
299
+ return None
300
+ if self.start <= pos <= self.end:
301
+ return 0
302
+ return min(abs(self.start - pos), abs(self.end - pos))
303
+
304
+ cdef class Gencode:
305
+ cdef dict genes
306
+ cdef map[string, vector[GenePoint]] starts, ends
307
+ def __cinit__(self, gencode=None, fasta=None, coding_only=True):
308
+ ''' initialise Gencode
309
+
310
+ Args:
311
+ gencode: path to gencode annotations file
312
+ fasta: path to fasta for genome matching annotations build
313
+ coding: restrict to protein_coding only by default
314
+ '''
315
+ if gencode is not None and not Path(gencode).exists():
316
+ raise ValueError(f'cannot find gencode at: {gencode}')
317
+ if fasta is not None and not Path(fasta).exists():
318
+ raise ValueError(f'cannot find fasta at: {fasta}')
319
+ self.genes = {}
320
+ if fasta:
321
+ logging.info(f'opening genome fasta: {fasta}')
322
+ global __genome_
323
+ __genome_ = Fasta(str(fasta))
324
+ logging.info(f'opening gencode annotations: {gencode}')
325
+ cdef vector[NamedTx] transcripts
326
+ cdef Gene curr
327
+ if gencode is not None:
328
+ transcripts = open_gencode(str(gencode).encode('utf8'), coding_only)
329
+ for x in transcripts:
330
+ symbol = x.symbol.decode('utf8')
331
+ if symbol not in self.genes:
332
+ self.genes[symbol] = Gene(symbol.encode('utf8'))
333
+ curr = self.genes[symbol]
334
+ curr.add_tx(x.tx, x.is_canonical)
335
+ self.genes[symbol] = curr
336
+ self._sort()
337
+
338
+ def _sort(self):
339
+ ''' index by starts and ends, to speed finding genes in a region
340
+ '''
341
+ for symbol in self.genes:
342
+ gene = self.genes[symbol]
343
+ chrom = gene.chrom.encode('utf8')
344
+ symbol = symbol.encode('utf8')
345
+
346
+ # ensure the chromosome is present
347
+ if self.starts.count(chrom) == 0:
348
+ self.starts[chrom] = []
349
+ if self.ends.count(chrom) == 0:
350
+ self.ends[chrom] = []
351
+
352
+ self.starts[chrom].push_back(GenePoint(gene.start, symbol))
353
+ self.ends[chrom].push_back(GenePoint(gene.end, symbol))
354
+
355
+ # sort start and end coords by position
356
+ for x, values in self.starts:
357
+ self.starts[x] = sorted(values, key=lambda x: x['pos'])
358
+ for x, values in self.ends:
359
+ self.ends[x] = sorted(values, key=lambda x: x['pos'])
360
+
361
+ def __repr__(self):
362
+ return f'Gencode(n_genes={len(self)})'
363
+ def __len__(self):
364
+ return len(self.genes)
365
+ def __getitem__(self, symbol):
366
+ return self.genes[symbol]
367
+ def __iter__(self):
368
+ for x in self.genes:
369
+ yield x
370
+
371
+ def add_gene(self, gene):
372
+ ''' add another gene to the Gencode object
373
+ '''
374
+ if gene.symbol not in self.genes:
375
+ self.genes[gene.symbol] = gene
376
+ self._sort()
377
+
378
+ def nearest(self, str chrom, int pos):
379
+ ''' find the nearest gene to a genomic chrom, pos coordinate
380
+ '''
381
+ chrom = f'chr{chrom}' if not chrom.startswith('chr') else chrom
382
+ _chrom = chrom.encode('utf8')
383
+
384
+ # first, account for any overlapping genespython
385
+ overlaps = self.in_region(chrom, pos-1, pos+1) # NOTE: possibly fix?
386
+ if len(overlaps) > 0:
387
+ # if we have > 0 prioritise if the position is in the CDS
388
+ cds_overlaps = [x for x in overlaps if x.in_any_tx_cds(pos)]
389
+ if len(cds_overlaps) > 0:
390
+ overlaps = cds_overlaps
391
+ # prioritise the gene with longest CDS (in the canonical tx)
392
+ txs = [x.canonical for x in overlaps]
393
+ lengths = [x.get_coding_distance(x.get_cds_end())['pos'] for x in txs]
394
+ idx = lengths.index(max(lengths))
395
+ return overlaps[idx]
396
+
397
+ # no overlaps observed, look for the nearest upstream or downstream gene
398
+ cdef GenePoint site = GenePoint(pos, b'A');
399
+ cdef unsigned int i = lower_bound(self.starts[_chrom].begin(), self.starts[_chrom].end(), site, &CompFunc) - self.starts[_chrom].begin()
400
+ cdef unsigned int j = upper_bound(self.ends[_chrom].begin(), self.ends[_chrom].begin(), site, &CompFunc) - self.ends[_chrom].begin()
401
+
402
+ i = min(i, self.starts[_chrom].size() - 1)
403
+ j = min(j, self.starts[_chrom].size() - 1)
404
+
405
+ downstream = self[self.starts[_chrom][i].symbol.decode('utf8')]
406
+ upstream = self[self.starts[_chrom][j].symbol.decode('utf8')]
407
+
408
+ if upstream.distance(chrom, pos) <= downstream.distance(chrom, pos):
409
+ return upstream
410
+ else:
411
+ return downstream
412
+
413
+ def in_region(self, str _chrom, int start, int end, int max_window=2500000):
414
+ ''' find genes within a genomic region
415
+
416
+ Args:
417
+ chrom: chromosome to search on
418
+ start: start position of region
419
+ end: end position of region
420
+ max_window: some genes encapsulate the region, which means we have
421
+ to account for gene lengths of up to 2.3 Mb in the human genome.
422
+ This permits extra search space in other organisms.
423
+
424
+ Returns:
425
+ list of Gene objects
426
+ '''
427
+ symbols = _in_region(_chrom.encode('utf8'), start, end, self.starts,
428
+ self.ends, max_window)
429
+ return [self[x.decode('utf8')] for x in symbols]
430
+
431
+ def __exit__(self):
432
+ ''' cleanup at exit
433
+ '''
434
+ global __genome_
435
+ __genome_.close()
436
+ __genome_ = None
@@ -0,0 +1,90 @@
1
+ # cython: language_level=3, boundscheck=False
2
+ '''
3
+ Copyright (c) 2015 Genome Research Ltd.
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy of
6
+ this software and associated documentation files (the "Software"), to deal in
7
+ the Software without restriction, including without limitation the rights to
8
+ use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies
9
+ of the Software, and to permit persons to whom the Software is furnished to do
10
+ so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
17
+ FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
18
+ COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
19
+ IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
20
+ CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
21
+ '''
22
+
23
+ from libcpp.vector cimport vector
24
+ from libcpp.string cimport string
25
+ from libcpp cimport bool
26
+
27
+ cdef extern from "tx.h":
28
+ cdef cppclass Tx:
29
+ Tx(string, string, int, int, char, string) except +
30
+ Tx() except +
31
+
32
+ void set_exons(vector[vector[int]], vector[vector[int]]) except +
33
+ void set_cds(vector[vector[int]]) except +
34
+ Region fix_cds_boundary(int) except +
35
+
36
+ vector[Region] get_exons()
37
+ vector[Region] get_cds()
38
+ string get_name()
39
+ string get_chrom()
40
+ int get_start()
41
+ int get_end()
42
+ char get_strand()
43
+ string get_type()
44
+ int get_cds_start()
45
+ int get_cds_end()
46
+
47
+ bool is_exonic(int)
48
+ int closest_exon_num(int)
49
+ Region get_closest_exon(int)
50
+ bool in_coding_region(int)
51
+ CDS_coords to_closest_exon(int, Region)
52
+ CDS_coords get_coding_distance(int) except +
53
+
54
+ int get_position_on_chrom(int, int) except +
55
+ int get_codon_number_for_cds_position(int)
56
+ int get_position_within_codon(int)
57
+ void add_cds_sequence(string)
58
+ void add_genomic_sequence(string, int) except +
59
+ string get_cds_sequence()
60
+ string get_genomic_sequence()
61
+ int get_genomic_offset()
62
+
63
+ string reverse_complement(string)
64
+ string get_centered_sequence(int, int) except +
65
+ string get_codon_sequence(int) except +
66
+ string get_seq_in_region(int, int) except +
67
+ string translate(string) except +
68
+
69
+ Codon get_codon_info(int) except +
70
+ int get_boundary_distance(int) except +
71
+ string consequence(int, string, string) except +
72
+
73
+ cdef struct CDS_coords:
74
+ int position
75
+ int offset
76
+
77
+ cdef struct Region:
78
+ int start
79
+ int end
80
+
81
+ cdef struct Codon:
82
+ int cds_pos
83
+ string codon_seq
84
+ int intra_codon
85
+ int codon_number
86
+ char initial_aa
87
+ int offset
88
+
89
+ cdef class Transcript:
90
+ cdef Tx *thisptr # hold a C++ instance which we're wrapping
@@ -0,0 +1,378 @@
1
+ # cython: language_level=3, boundscheck=False
2
+ '''
3
+ Copyright (c) 2015 Genome Research Ltd.
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy of
6
+ this software and associated documentation files (the "Software"), to deal in
7
+ the Software without restriction, including without limitation the rights to
8
+ use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies
9
+ of the Software, and to permit persons to whom the Software is furnished to do
10
+ so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
17
+ FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
18
+ COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
19
+ IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
20
+ CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
21
+ '''
22
+
23
+ from itertools import combinations
24
+
25
+ cdef class Transcript:
26
+ def __cinit__(self, name, chrom, start, end, strand,
27
+ transcript_type='protein_coding', exons=None, cds=None, sequence=None,
28
+ offset=0):
29
+ ''' construct a Transcript object
30
+
31
+ Args:
32
+ name: ID of the transcript
33
+ start: position in bp at 5' edge of transcript (on + strand)
34
+ end: position in bp at 3' edge of transcript (on + strand)
35
+ exons: list of tuples defining start and end positions of exons
36
+ cds: list of tuples defining start and end positions of CDS regions
37
+ sequence: DNA sequence of genome region of the transcript.
38
+ offset: how many base pairs the DNA sequence extends outwards
39
+ '''
40
+
41
+ name = name.encode('utf8')
42
+ chrom = chrom.encode('utf8')
43
+ transcript_type = transcript_type.encode('utf8')
44
+ self.thisptr = new Tx(name, chrom, start, end, ord(strand), transcript_type)
45
+
46
+ if exons is not None and cds is not None:
47
+ self.set_exons(exons, cds)
48
+ self.set_cds(cds)
49
+
50
+ if sequence is not None:
51
+ self.add_genomic_sequence(sequence, offset)
52
+
53
+ def __dealloc__(self):
54
+ del self.thisptr
55
+
56
+ def __repr__(self):
57
+
58
+ exons = [ (x['start'], x['end']) for x in self.get_exons() ]
59
+ cds = [ (x['start'], x['end']) for x in self.get_cds() ]
60
+ seq = self.get_genomic_sequence()
61
+
62
+ if len(seq) > 40:
63
+ seq = seq[:20] + '...[{} bp]...'.format(len(seq) - 40) + seq[-20:]
64
+
65
+ if exons == []:
66
+ exons = None
67
+
68
+ if cds == []:
69
+ cds = None
70
+
71
+ if seq == '':
72
+ seq = None
73
+ else:
74
+ seq = '"' + seq + '"'
75
+
76
+ return f'Transcript(name="{self.get_name()}", chrom="{self.get_chrom()}", ' \
77
+ f'start={self.get_start()}, end={self.get_end()}, strand="{self.get_strand()}", ' \
78
+ f'transcript_type="{self.get_type()}", exons={exons}, cds={cds}, ' \
79
+ f'sequence={seq}, offset={self.get_genomic_offset()})'
80
+
81
+ def __str__(self):
82
+ return self.__repr__()
83
+
84
+ def __hash__(self):
85
+ return hash((self.thisptr.get_chrom(), self.thisptr.get_start(),
86
+ self.thisptr.get_end()))
87
+
88
+ def __richcmp__(self, other, op):
89
+
90
+ if op == 2:
91
+ return self.__hash__() == other.__hash__()
92
+ else:
93
+ err_msg = "op {0} isn't implemented yet".format(op)
94
+ raise NotImplementedError(err_msg)
95
+
96
+ def set_exons(self, exon_ranges, cds_ranges):
97
+ ''' add exon ranges
98
+
99
+ Args:
100
+ exon_ranges: a CDS position of the selected base.
101
+ '''
102
+ self.thisptr.set_exons(exon_ranges, cds_ranges)
103
+
104
+ def get_overlaps(self, exon, regions):
105
+ ''' find all regions which overlap a given region
106
+ '''
107
+ return [ i for i, x in enumerate(regions) if
108
+ exon['start'] <= x['end'] and exon['end'] >= x['start'] ]
109
+
110
+ def insert_region(self, coords, region):
111
+ ''' include a region into a list of regions
112
+
113
+ To include a region, we have to check which pre-existing regions the new
114
+ region overlaps, so any overlaps can be merged into a single region.
115
+
116
+ Args:
117
+ coords: list of {start: X, end: Y} dictionaries
118
+ region: dict of {'start': X, 'end': Y} positions
119
+ '''
120
+ indices = self.get_overlaps(region, coords)
121
+ overlaps = [ coords[i] for i in indices ]
122
+ start = min( x['start'] for x in overlaps + [region] )
123
+ end = max( x['end'] for x in overlaps + [region] )
124
+
125
+ for i in sorted(indices, reverse=True):
126
+ del coords[i]
127
+
128
+ return coords + [{'start': start, 'end': end}]
129
+
130
+ def merge_coordinates(self, first, second):
131
+ ''' merge two sets of coordinates, to get the union of regions
132
+
133
+ This uses an inefficient approach, looping over and over, but we won't
134
+ need to perform this often.
135
+
136
+ Args:
137
+ first: list of {'start': x, 'end': y} dictionaries for first transcript
138
+ second: list of {'start': x, 'end': y} dictionaries for second transcript
139
+
140
+ Returns:
141
+ list of [start, end] lists, sorted by position.
142
+ '''
143
+ coords = []
144
+ for a, b in combinations(first + second, 2):
145
+ if a['start'] <= b['end'] and a['end'] >= b['start']:
146
+ region = {'start': min(a['start'], b['start']),
147
+ 'end': max(a['end'], b['end'])}
148
+ a, b = region, region
149
+
150
+ coords = self.insert_region(coords, a)
151
+ coords = self.insert_region(coords, b)
152
+
153
+ return [ (x['start'], x['end']) for x in sorted(coords, key=lambda x: x['start']) ]
154
+
155
+ def merge_genomic_seq(self, other):
156
+ ''' merge the genomic sequence from two transcripts
157
+
158
+ We have two transcripts A, and B. We need to get the contiguous sequence
159
+ from the start of the first transcript on the chromosome to the end of
160
+ the second transcript. The transcripts may or may not overlap. There are
161
+ three scenarios we need to account for:
162
+
163
+
164
+ overlap without A =================
165
+ enveloping ================= B
166
+
167
+
168
+ overlap A ==============================
169
+ and envelop ================== B
170
+
171
+
172
+ no overlap A ===============
173
+ =============== B
174
+
175
+ I've called the transcript whose sequence is first along the chromosome
176
+ as 'lead', and the transcript whose sequence is last as 'lag', and the
177
+ converse as 'not_lead', and 'not_lag'. Note that in the envelope case,
178
+ the lead transcript can also be the lag transcript.
179
+ '''
180
+
181
+ # make sure that the surrounding sequence is the same length in both
182
+ # transcripts.
183
+ # TODO: this could be worked around, by figuring the minimum offset length,
184
+ # TODO: then trimming the respective DNA offset sequences to that length.
185
+ assert self.get_genomic_offset() == other.get_genomic_offset()
186
+
187
+ # figure out which transcripts hold the leading and lagging sections
188
+ lead, not_lead = self, other
189
+ if self.get_start() > other.get_start():
190
+ lead, not_lead = other, self
191
+
192
+ lag, not_lag = self, other
193
+ if self.get_end() < other.get_end():
194
+ lag, not_lag = other, self
195
+
196
+ lead_offset = lead.get_genomic_offset()
197
+ lead_gdna = lead.get_genomic_sequence()
198
+ initial = lead_gdna[:not_lead.get_start() - lead.get_start() + lead_offset]
199
+
200
+ if self.get_start() <= other.get_end() and self.get_end() >= other.get_start():
201
+ intersect_start = not_lead.get_start() - lead.get_start() + lead_offset
202
+ intersect_end = not_lag.get_end() - lead.get_start() + lead_offset
203
+ intersect = lead_gdna[intersect_start:intersect_end]
204
+ else:
205
+ intersect = 'N' * (lag.get_start() - lead.get_end() - lead_offset * 2)
206
+
207
+ lag_offset = lag.get_genomic_offset()
208
+ lag_gdna = lag.get_genomic_sequence()
209
+
210
+ # some transcripts overlap, but some do not. We need to find the position
211
+ # where the lagging transcript takes over, which is either at the end of
212
+ # not lagging transcript, or the start of the lagging transcript,
213
+ # whichever is higher
214
+ pos = max(not_lag.get_end(), lag.get_start())
215
+ final = lag_gdna[pos - lag.get_start() + lead_offset:]
216
+
217
+ return initial + intersect + final
218
+
219
+ def __add__(self, other):
220
+ """ combine the coding sequences of two Transcript objects
221
+
222
+ When we determine the sites for sampling, occasioally we want to
223
+ use sites from multiple alternative transcripts. We determine the sites
224
+ for each transcript in turn, but mask the sites that have been collected
225
+ in the preceeding transcripts. In order to be able to mask all previous
226
+ trabnscripts, we need to combine the coding sequence of the transcripts
227
+ as we move through them. This function performs the union of coding
228
+ sequence regions between different transcripts.
229
+
230
+ We do this outside of the c++ class, so as to be able to set up a
231
+ Transcript object correctly.
232
+
233
+ Args:
234
+ other: a transcript to be combined with the current object.
235
+
236
+ Returns:
237
+ an altered instance of the class, where the coding sequence regions
238
+ are the union of the coding regions of two Transcript objects. This
239
+ disrupts the ability to get meaningingful sequence from the object,
240
+ so don't try to extract sequence from the returned object.
241
+ """
242
+
243
+ # if we try transcript + None or None + transcript, return the original
244
+ # transcript, rather than raising an error.
245
+ if other is None:
246
+ return self
247
+ if self is None:
248
+ return other
249
+
250
+ altered = Transcript('{}:{}'.format(self.get_name(), other.get_name()),
251
+ self.get_chrom(), min(self.get_start(), other.get_start()),
252
+ max(self.get_end(), other.get_end()), self.get_strand(), self.get_type())
253
+
254
+ exons = self.merge_coordinates(self.get_exons(), other.get_exons())
255
+ cds = self.merge_coordinates(self.get_cds(), other.get_cds())
256
+
257
+ altered.set_exons(exons, cds)
258
+ altered.set_cds(cds)
259
+
260
+ if self.get_genomic_sequence() != "":
261
+ altered.add_genomic_sequence(self.merge_genomic_seq(other), self.get_genomic_offset())
262
+
263
+ return altered
264
+
265
+ def set_cds(self, cds_ranges):
266
+ ''' set CDS ranges
267
+
268
+ Args:
269
+ cds_ranges: a CDS position of the selected base.
270
+ '''
271
+
272
+ self.thisptr.set_cds(cds_ranges)
273
+
274
+ def get_genomic_offset(self):
275
+ return self.thisptr.get_genomic_offset()
276
+ def get_exons(self):
277
+ return self.thisptr.get_exons()
278
+ def get_cds(self):
279
+ return self.thisptr.get_cds()
280
+ def get_name(self):
281
+ return self.thisptr.get_name().decode('utf8')
282
+ def get_chrom(self):
283
+ return self.thisptr.get_chrom().decode('utf8')
284
+ def get_type(self):
285
+ return self.thisptr.get_type().decode('utf8')
286
+ def get_start(self):
287
+ return self.thisptr.get_start()
288
+ def get_end(self):
289
+ return self.thisptr.get_end()
290
+ def get_strand(self):
291
+ return chr(self.thisptr.get_strand())
292
+ def get_cds_start(self):
293
+ return self.thisptr.get_cds_start()
294
+ def get_cds_end(self):
295
+ return self.thisptr.get_cds_end()
296
+ def fix_cds_boundary(self, pos):
297
+ return self.thisptr.fix_cds_boundary(pos)
298
+
299
+ def in_exons(self, position):
300
+ ''' check if a site lies within the exon ranges
301
+
302
+ Args:
303
+ position: an integer-based chromosome position.
304
+ '''
305
+
306
+ return self.thisptr.is_exonic(position)
307
+
308
+ def get_closest_exon(self, position):
309
+ ''' finds the positions of the exon closest to a position
310
+ '''
311
+ return self.thisptr.get_closest_exon(position)
312
+
313
+ def in_coding_region(self, position):
314
+ return self.thisptr.in_coding_region(position)
315
+
316
+ def get_coding_distance(self, pos):
317
+ ''' get distance to CDS start (and intronic offset)
318
+ '''
319
+ coords = self.thisptr.get_coding_distance(pos)
320
+
321
+ return {'pos': coords.position, 'offset': coords.offset}
322
+
323
+ def get_position_on_chrom(self, pos, offset=0):
324
+ return self.thisptr.get_position_on_chrom(pos, offset)
325
+
326
+ def get_codon_number_for_cds_position(self, pos):
327
+ return self.thisptr.get_codon_number_for_cds_position(pos)
328
+
329
+ def get_position_within_codon(self, pos):
330
+ return self.thisptr.get_position_within_codon(pos)
331
+
332
+ def add_cds_sequence(self, text):
333
+ self.thisptr.add_cds_sequence(text.encode('utf8'))
334
+
335
+ def get_cds_sequence(self):
336
+ return self.thisptr.get_cds_sequence().decode('utf8')
337
+
338
+ def add_genomic_sequence(self, text, offset=0):
339
+ self.thisptr.add_genomic_sequence(text.encode('utf8'), offset)
340
+
341
+ def get_genomic_sequence(self):
342
+ return self.thisptr.get_genomic_sequence().decode('utf8')
343
+
344
+ def reverse_complement(self, text):
345
+ return self.thisptr.reverse_complement(text).decode('utf8')
346
+
347
+ def get_centered_sequence(self, pos, length=3):
348
+ return self.thisptr.get_centered_sequence(pos, length).decode('utf8')
349
+
350
+ def get_codon_sequence(self, pos):
351
+ return self.thisptr.get_codon_sequence(pos).decode('utf8')
352
+
353
+ def translate(self, text):
354
+ return self.thisptr.translate(text.encode('utf8')).decode('utf8')
355
+
356
+ def get_codon_info(self, pos):
357
+ codon = dict(self.thisptr.get_codon_info(pos))
358
+
359
+ if codon['codon_number'] == -9999999:
360
+ codon['codon_number'] = None
361
+ codon['intra_codon'] = None
362
+ codon['codon_seq'] = None
363
+ codon['initial_aa'] = None
364
+
365
+ if codon['codon_seq'] is not None:
366
+ codon['codon_seq'] = codon['codon_seq'].decode('utf8')
367
+
368
+ if codon['initial_aa'] is not None:
369
+ codon['initial_aa'] = chr(codon['initial_aa'])
370
+
371
+ return codon
372
+
373
+ def get_boundary_distance(self, pos):
374
+ return self.thisptr.get_boundary_distance(pos)
375
+
376
+ def consequence(self, pos, ref, alt):
377
+ cq = self.thisptr.consequence(pos, ref.encode('utf8'), alt.encode('utf8'))
378
+ return cq.decode('utf8')
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: gencodegenes
3
- Version: 1.0.0
3
+ Version: 1.0.2
4
4
  Summary: Package to load genes from GENCODE GTF files
5
5
  Home-page: https://github.com/jeremymcrae/gencodegenes
6
6
  Author: Jeremy McRae
@@ -11,7 +11,10 @@ src/tx.cpp
11
11
  src/tx.h
12
12
  src/gencodegenes/__init__.py
13
13
  src/gencodegenes/gencode.cpp
14
+ src/gencodegenes/gencode.pyx
14
15
  src/gencodegenes/transcript.cpp
16
+ src/gencodegenes/transcript.pxd
17
+ src/gencodegenes/transcript.pyx
15
18
  src/gencodegenes.egg-info/PKG-INFO
16
19
  src/gencodegenes.egg-info/SOURCES.txt
17
20
  src/gencodegenes.egg-info/dependency_links.txt
@@ -1,15 +0,0 @@
1
- # genecodegenes code
2
- include MANIFEST.in
3
- include pyproject.toml
4
- include genecodegenes/*.cpp
5
- include genecodegenes/*.py
6
- include genecodegenes/*.pyx
7
- include genecodegenes/*.pxd
8
- include src/*.h
9
- include src/*.cpp
10
- include src/gzstream/gzstream.C
11
- include src/gzstream/gzstream.h
12
- include data/*.txt
13
-
14
- # genecodegenes tests
15
- include tests/*.py
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes