gencodegenes 1.1.7__tar.gz → 1.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/MANIFEST.in +0 -2
- {gencodegenes-1.1.7/src/gencodegenes.egg-info → gencodegenes-1.2.0}/PKG-INFO +26 -16
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/README.md +23 -13
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/pyproject.toml +7 -4
- gencodegenes-1.2.0/setup.py +69 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/gencode.cpp +108 -58
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/gencode.h +2 -2
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/gencodegenes/gencode.cpp +15058 -9079
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/gencodegenes/gencode.pyi +13 -4
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/gencodegenes/gencode.pyx +95 -47
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/gencodegenes/transcript.cpp +9451 -4707
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/gencodegenes/transcript.pxd +2 -2
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/gencodegenes/transcript.pyx +16 -9
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/gencodegenes/tx.cpp +96 -53
- {gencodegenes-1.1.7/src → gencodegenes-1.2.0/src/gencodegenes}/tx.h +4 -4
- {gencodegenes-1.1.7 → gencodegenes-1.2.0/src/gencodegenes.egg-info}/PKG-INFO +26 -16
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/gencodegenes.egg-info/SOURCES.txt +0 -2
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/gtf.cpp +98 -50
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/gtf.h +21 -9
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/tx.cpp +96 -53
- {gencodegenes-1.1.7/src/gencodegenes → gencodegenes-1.2.0/src}/tx.h +4 -4
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/tests/test_gencode.py +362 -1
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/tests/test_sequence_methods.py +38 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/tests/test_transcript.py +89 -0
- gencodegenes-1.1.7/setup.py +0 -117
- gencodegenes-1.1.7/src/gzstream/gzstream.C +0 -165
- gencodegenes-1.1.7/src/gzstream/gzstream.h +0 -121
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/LICENSE.txt +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/setup.cfg +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/gencodegenes/__init__.py +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/gencodegenes/py.typed +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/gencodegenes/transcript.pyi +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/gencodegenes.egg-info/dependency_links.txt +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/gencodegenes.egg-info/requires.txt +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/gencodegenes.egg-info/top_level.txt +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/adler32.c +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/compress.c +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/crc32.c +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/crc32.h +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/deflate.c +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/deflate.h +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/gzclose.c +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/gzguts.h +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/gzlib.c +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/gzread.c +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/gzwrite.c +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/infback.c +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/inffast.c +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/inffast.h +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/inffixed.h +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/inflate.c +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/inflate.h +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/inftrees.c +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/inftrees.h +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/trees.c +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/trees.h +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/uncompr.c +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/zconf.h +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/zlib.h +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/zutil.c +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/src/zlib/zutil.h +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/tests/__init__.py +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/tests/data/example.grch38.fa +0 -0
- {gencodegenes-1.1.7 → gencodegenes-1.2.0}/tests/data/example.grch38.gtf +0 -0
|
@@ -1,13 +1,13 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gencodegenes
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.2.0
|
|
4
4
|
Summary: Package to load genes from GENCODE GTF files
|
|
5
5
|
Author-email: Jeremy McRae <jeremy.mcrae@gmail.com>
|
|
6
|
+
License-Expression: MIT
|
|
6
7
|
Project-URL: homepage, https://github.com/jeremymcrae/gencodegenes
|
|
7
|
-
Classifier: License :: OSI Approved :: MIT License
|
|
8
8
|
Classifier: Development Status :: 4 - Beta
|
|
9
9
|
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
10
|
-
Requires-Python: >=3.
|
|
10
|
+
Requires-Python: >=3.10
|
|
11
11
|
Description-Content-Type: text/markdown
|
|
12
12
|
License-File: LICENSE.txt
|
|
13
13
|
Requires-Dist: pyfaidx>=0.5.8
|
|
@@ -16,7 +16,7 @@ Dynamic: license-file
|
|
|
16
16
|
|
|
17
17
|
### GENCODEGenes
|
|
18
18
|
|
|
19
|
-
This package loads genes from GENCODE GTF
|
|
19
|
+
This package loads genes from GENCODE GTF files, groups transcripts by gene,
|
|
20
20
|
and provides methods for transcripts, so you can find exon coordinates, CDS
|
|
21
21
|
distances and sequences.
|
|
22
22
|
|
|
@@ -31,22 +31,30 @@ pip install gencodegenes
|
|
|
31
31
|
from gencodegenes import Gencode
|
|
32
32
|
|
|
33
33
|
gencode = Gencode(GTF_PATH)
|
|
34
|
-
# full function arguments are Gencode(
|
|
35
|
-
# -
|
|
34
|
+
# full function arguments are Gencode(gencode, fasta=None, coding_only=True)
|
|
35
|
+
# - gencode: path to GTF file (plain or gzipped)
|
|
36
|
+
# - fasta: pass in path to fasta file to get gene transcripts with sequence
|
|
36
37
|
# - coding_only: pass in False to include all transcripts, not just protein coding
|
|
37
38
|
|
|
39
|
+
# use as a context manager to close the fasta when done
|
|
40
|
+
with Gencode(GTF_PATH, fasta=FASTA_PATH) as gencode:
|
|
41
|
+
...
|
|
42
|
+
|
|
38
43
|
# get gene by HGNC symbol
|
|
39
44
|
gene = gencode['OR5A1']
|
|
40
45
|
transcripts = gene.transcripts
|
|
41
|
-
canonical = gene.canonical # picks
|
|
42
|
-
#
|
|
43
|
-
#
|
|
44
|
-
#
|
|
45
|
-
# picks the longest cDNA
|
|
46
|
+
canonical = gene.canonical # picks the transcript tagged Ensembl_canonical (the
|
|
47
|
+
# MANE Select transcript in human, where one exists), if
|
|
48
|
+
# none tagged, picks from those tagged appris_principal,
|
|
49
|
+
# if none tagged, picks from all transcripts. Within
|
|
50
|
+
# these, picks the longest CDS, or the longest cDNA if
|
|
51
|
+
# none are protein coding
|
|
46
52
|
gene.start, gene.end, gene.chrom, gene.strand, gene.symbol # other attributes available
|
|
53
|
+
gene.alternate_ids # gene_id and hgnc_id from the GTF
|
|
47
54
|
|
|
48
55
|
|
|
49
|
-
# find gene nearest a genomic position, or overlapping a genomic region
|
|
56
|
+
# find gene nearest a genomic position, or overlapping a genomic region.
|
|
57
|
+
# Chromosomes match with or without the 'chr' prefix
|
|
50
58
|
gencode.nearest('chr1', 1000000)
|
|
51
59
|
gencode.in_region('chr1', 1000000, 2000000)
|
|
52
60
|
|
|
@@ -58,14 +66,15 @@ tx.get_coding_distance(pos) # get distance in CDS to CDS start
|
|
|
58
66
|
tx.get_closest_exon(pos) # find exon closest to position
|
|
59
67
|
tx.get_position_on_chrom(cds_pos) # convert CDS pos to genomic pos
|
|
60
68
|
tx.get_codon_info(pos) # get info about codon for a site
|
|
61
|
-
tx.
|
|
62
|
-
tx.translate(seq) # translate DNA to AA
|
|
69
|
+
tx.get_codon_number_for_cds_position(cds_pos) # convert CDS pos to codon number
|
|
70
|
+
tx.translate(seq) # translate DNA to AA
|
|
71
|
+
tx.consequence(pos, ref, alt) # get variant consequence (if opened with fasta)
|
|
63
72
|
|
|
64
73
|
# the transcript also has associated data fields
|
|
65
74
|
tx.name # transcript ID
|
|
66
75
|
tx.chrom # transcript chromosome
|
|
67
|
-
tx.start # transcript start (
|
|
68
|
-
tx.end # transcript end
|
|
76
|
+
tx.start # transcript start (lowest position on chromosome)
|
|
77
|
+
tx.end # transcript end (highest position on chromosome)
|
|
69
78
|
tx.cds_start # CDS start position
|
|
70
79
|
tx.cds_end # CDS end position
|
|
71
80
|
tx.type # transcript type e.g. protein_coding
|
|
@@ -73,5 +82,6 @@ tx.strand # strand (+ or -)
|
|
|
73
82
|
tx.exons # list of exon coordinates
|
|
74
83
|
tx.cds # list of CDS coordinates
|
|
75
84
|
tx.cds_sequence # get cDNA sequence (if Gencode was opened with fasta)
|
|
85
|
+
tx.attributes # dict-like view of the GTF attributes e.g. tx.attributes['gene_id']
|
|
76
86
|
|
|
77
87
|
```
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
|
|
2
2
|
### GENCODEGenes
|
|
3
3
|
|
|
4
|
-
This package loads genes from GENCODE GTF
|
|
4
|
+
This package loads genes from GENCODE GTF files, groups transcripts by gene,
|
|
5
5
|
and provides methods for transcripts, so you can find exon coordinates, CDS
|
|
6
6
|
distances and sequences.
|
|
7
7
|
|
|
@@ -16,22 +16,30 @@ pip install gencodegenes
|
|
|
16
16
|
from gencodegenes import Gencode
|
|
17
17
|
|
|
18
18
|
gencode = Gencode(GTF_PATH)
|
|
19
|
-
# full function arguments are Gencode(
|
|
20
|
-
# -
|
|
19
|
+
# full function arguments are Gencode(gencode, fasta=None, coding_only=True)
|
|
20
|
+
# - gencode: path to GTF file (plain or gzipped)
|
|
21
|
+
# - fasta: pass in path to fasta file to get gene transcripts with sequence
|
|
21
22
|
# - coding_only: pass in False to include all transcripts, not just protein coding
|
|
22
23
|
|
|
24
|
+
# use as a context manager to close the fasta when done
|
|
25
|
+
with Gencode(GTF_PATH, fasta=FASTA_PATH) as gencode:
|
|
26
|
+
...
|
|
27
|
+
|
|
23
28
|
# get gene by HGNC symbol
|
|
24
29
|
gene = gencode['OR5A1']
|
|
25
30
|
transcripts = gene.transcripts
|
|
26
|
-
canonical = gene.canonical # picks
|
|
27
|
-
#
|
|
28
|
-
#
|
|
29
|
-
#
|
|
30
|
-
# picks the longest cDNA
|
|
31
|
+
canonical = gene.canonical # picks the transcript tagged Ensembl_canonical (the
|
|
32
|
+
# MANE Select transcript in human, where one exists), if
|
|
33
|
+
# none tagged, picks from those tagged appris_principal,
|
|
34
|
+
# if none tagged, picks from all transcripts. Within
|
|
35
|
+
# these, picks the longest CDS, or the longest cDNA if
|
|
36
|
+
# none are protein coding
|
|
31
37
|
gene.start, gene.end, gene.chrom, gene.strand, gene.symbol # other attributes available
|
|
38
|
+
gene.alternate_ids # gene_id and hgnc_id from the GTF
|
|
32
39
|
|
|
33
40
|
|
|
34
|
-
# find gene nearest a genomic position, or overlapping a genomic region
|
|
41
|
+
# find gene nearest a genomic position, or overlapping a genomic region.
|
|
42
|
+
# Chromosomes match with or without the 'chr' prefix
|
|
35
43
|
gencode.nearest('chr1', 1000000)
|
|
36
44
|
gencode.in_region('chr1', 1000000, 2000000)
|
|
37
45
|
|
|
@@ -43,14 +51,15 @@ tx.get_coding_distance(pos) # get distance in CDS to CDS start
|
|
|
43
51
|
tx.get_closest_exon(pos) # find exon closest to position
|
|
44
52
|
tx.get_position_on_chrom(cds_pos) # convert CDS pos to genomic pos
|
|
45
53
|
tx.get_codon_info(pos) # get info about codon for a site
|
|
46
|
-
tx.
|
|
47
|
-
tx.translate(seq) # translate DNA to AA
|
|
54
|
+
tx.get_codon_number_for_cds_position(cds_pos) # convert CDS pos to codon number
|
|
55
|
+
tx.translate(seq) # translate DNA to AA
|
|
56
|
+
tx.consequence(pos, ref, alt) # get variant consequence (if opened with fasta)
|
|
48
57
|
|
|
49
58
|
# the transcript also has associated data fields
|
|
50
59
|
tx.name # transcript ID
|
|
51
60
|
tx.chrom # transcript chromosome
|
|
52
|
-
tx.start # transcript start (
|
|
53
|
-
tx.end # transcript end
|
|
61
|
+
tx.start # transcript start (lowest position on chromosome)
|
|
62
|
+
tx.end # transcript end (highest position on chromosome)
|
|
54
63
|
tx.cds_start # CDS start position
|
|
55
64
|
tx.cds_end # CDS end position
|
|
56
65
|
tx.type # transcript type e.g. protein_coding
|
|
@@ -58,5 +67,6 @@ tx.strand # strand (+ or -)
|
|
|
58
67
|
tx.exons # list of exon coordinates
|
|
59
68
|
tx.cds # list of CDS coordinates
|
|
60
69
|
tx.cds_sequence # get cDNA sequence (if Gencode was opened with fasta)
|
|
70
|
+
tx.attributes # dict-like view of the GTF attributes e.g. tx.attributes['gene_id']
|
|
61
71
|
|
|
62
72
|
```
|
|
@@ -1,13 +1,14 @@
|
|
|
1
1
|
[build-system]
|
|
2
|
-
requires = ["cython", "setuptools >=
|
|
2
|
+
requires = ["cython", "setuptools >= 77.0.0"]
|
|
3
3
|
build-backend = "setuptools.build_meta"
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = 'gencodegenes'
|
|
7
|
-
version = '1.
|
|
7
|
+
version = '1.2.0'
|
|
8
8
|
description = 'Package to load genes from GENCODE GTF files'
|
|
9
9
|
readme = 'README.md'
|
|
10
|
-
|
|
10
|
+
license = "MIT"
|
|
11
|
+
requires-python = ">=3.10"
|
|
11
12
|
authors = [
|
|
12
13
|
{name = 'Jeremy McRae', email = 'jeremy.mcrae@gmail.com'}
|
|
13
14
|
]
|
|
@@ -17,7 +18,6 @@ dependencies = [
|
|
|
17
18
|
]
|
|
18
19
|
|
|
19
20
|
classifiers = [
|
|
20
|
-
"License :: OSI Approved :: MIT License",
|
|
21
21
|
"Development Status :: 4 - Beta",
|
|
22
22
|
"Topic :: Scientific/Engineering :: Bio-Informatics",
|
|
23
23
|
]
|
|
@@ -27,5 +27,8 @@ packages = [
|
|
|
27
27
|
"gencodegenes"
|
|
28
28
|
]
|
|
29
29
|
|
|
30
|
+
[tool.setuptools.package-data]
|
|
31
|
+
gencodegenes = ['transcript.pxd', 'tx.h', 'tx.cpp', 'py.typed', '*.pyi']
|
|
32
|
+
|
|
30
33
|
[project.urls]
|
|
31
34
|
homepage = 'https://github.com/jeremymcrae/gencodegenes'
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
|
|
2
|
+
import glob
|
|
3
|
+
import os
|
|
4
|
+
import sys
|
|
5
|
+
import shutil
|
|
6
|
+
|
|
7
|
+
from setuptools import setup, Extension
|
|
8
|
+
from Cython.Build import cythonize
|
|
9
|
+
|
|
10
|
+
EXTRA_COMPILE_ARGS = []
|
|
11
|
+
EXTRA_LINK_ARGS = []
|
|
12
|
+
if sys.platform == "win32":
|
|
13
|
+
EXTRA_COMPILE_ARGS += ['/std:c++14']
|
|
14
|
+
else:
|
|
15
|
+
EXTRA_COMPILE_ARGS += ['-std=c++11']
|
|
16
|
+
if sys.platform == "darwin":
|
|
17
|
+
EXTRA_COMPILE_ARGS += ["-stdlib=libc++"]
|
|
18
|
+
EXTRA_LINK_ARGS += ["-stdlib=libc++"]
|
|
19
|
+
sdk_base = "/Library/Developer/CommandLineTools/SDKs/MacOSX.sdk"
|
|
20
|
+
if os.path.exists(sdk_base):
|
|
21
|
+
EXTRA_COMPILE_ARGS += [
|
|
22
|
+
f"-I{sdk_base}/usr/include/c++/v1",
|
|
23
|
+
f"-I{sdk_base}/usr/include",
|
|
24
|
+
]
|
|
25
|
+
EXTRA_LINK_ARGS += [
|
|
26
|
+
f"-L{sdk_base}/usr/lib",
|
|
27
|
+
]
|
|
28
|
+
|
|
29
|
+
gencode_sources = [
|
|
30
|
+
"src/gencodegenes/gencode.pyx",
|
|
31
|
+
"src/gencode.cpp",
|
|
32
|
+
"src/gtf.cpp",
|
|
33
|
+
"src/tx.cpp",
|
|
34
|
+
]
|
|
35
|
+
|
|
36
|
+
libs = ['z']
|
|
37
|
+
include_dirs = ['src/']
|
|
38
|
+
|
|
39
|
+
if sys.platform == 'win32':
|
|
40
|
+
gencode_sources += glob.glob('src/zlib/*.c')
|
|
41
|
+
include_dirs.append('src/zlib/')
|
|
42
|
+
libs = []
|
|
43
|
+
|
|
44
|
+
extensions = [
|
|
45
|
+
Extension("gencodegenes.transcript",
|
|
46
|
+
extra_compile_args=EXTRA_COMPILE_ARGS,
|
|
47
|
+
extra_link_args=EXTRA_LINK_ARGS,
|
|
48
|
+
sources=[
|
|
49
|
+
"src/gencodegenes/transcript.pyx",
|
|
50
|
+
"src/tx.cpp"],
|
|
51
|
+
include_dirs=["src/"],
|
|
52
|
+
language="c++"),
|
|
53
|
+
Extension("gencodegenes.gencode",
|
|
54
|
+
extra_compile_args=EXTRA_COMPILE_ARGS,
|
|
55
|
+
extra_link_args=EXTRA_LINK_ARGS,
|
|
56
|
+
sources=gencode_sources,
|
|
57
|
+
include_dirs=include_dirs,
|
|
58
|
+
libraries=libs,
|
|
59
|
+
language="c++"),
|
|
60
|
+
]
|
|
61
|
+
|
|
62
|
+
# include tx.h in the package, for downstream usage
|
|
63
|
+
shutil.copy("src/tx.h", "src/gencodegenes/tx.h")
|
|
64
|
+
shutil.copy("src/tx.cpp", "src/gencodegenes/tx.cpp")
|
|
65
|
+
|
|
66
|
+
setup(
|
|
67
|
+
package_dir={'': 'src'},
|
|
68
|
+
ext_modules=cythonize(extensions),
|
|
69
|
+
)
|
|
@@ -1,10 +1,12 @@
|
|
|
1
1
|
|
|
2
2
|
#include <algorithm>
|
|
3
3
|
#include <cstdint>
|
|
4
|
+
#include <deque>
|
|
4
5
|
#include <string>
|
|
5
6
|
#include <map>
|
|
6
7
|
#include <set>
|
|
7
8
|
#include <stdexcept>
|
|
9
|
+
#include <unordered_map>
|
|
8
10
|
#include <vector>
|
|
9
11
|
|
|
10
12
|
#include <iostream>
|
|
@@ -72,68 +74,127 @@ static void include_end_codons(std::map<std::string, int> cds_range, TxInfo & in
|
|
|
72
74
|
}
|
|
73
75
|
}
|
|
74
76
|
|
|
77
|
+
// set the transcript span from its exons and CDS, for GTFs without transcript lines
|
|
78
|
+
static void set_span(TxInfo & info) {
|
|
79
|
+
if (info.start != 0 || info.end != 0) {
|
|
80
|
+
return;
|
|
81
|
+
}
|
|
82
|
+
bool first = true;
|
|
83
|
+
for (auto regions : {&info.exons, &info.cds}) {
|
|
84
|
+
for (auto & x : *regions) {
|
|
85
|
+
info.start = first ? x[0] : std::min(info.start, x[0]);
|
|
86
|
+
info.end = first ? x[1] : std::max(info.end, x[1]);
|
|
87
|
+
first = false;
|
|
88
|
+
}
|
|
89
|
+
}
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
// construct a Tx from the features collected for a transcript
|
|
93
|
+
//
|
|
94
|
+
// Transcripts with inconsistent coordinates (e.g. a stop codon outside the
|
|
95
|
+
// exons) are skipped with a warning, so one malformed transcript doesn't stop
|
|
96
|
+
// the rest of the GTF from loading.
|
|
97
|
+
static void add_transcript(std::vector<NamedTx> & transcripts, TxInfo & info,
|
|
98
|
+
std::map<std::string, int> & cds_range, std::string & symbol,
|
|
99
|
+
std::vector<std::string> & alt_ids) {
|
|
100
|
+
try {
|
|
101
|
+
// adjust CDS for start and stop codon coords
|
|
102
|
+
include_end_codons(cds_range, info);
|
|
103
|
+
set_span(info);
|
|
104
|
+
Tx tx = Tx(info.name, info.chrom, info.start, info.end, info.strand[0],
|
|
105
|
+
info.transcript_type, info.attributes);
|
|
106
|
+
tx.set_exons(info.exons);
|
|
107
|
+
tx.set_cds(info.cds);
|
|
108
|
+
transcripts.push_back({symbol, alt_ids, tx, info.is_canonical});
|
|
109
|
+
} catch (const std::invalid_argument & e) {
|
|
110
|
+
std::cerr << "skipping transcript " << info.name << ": " << e.what() << std::endl;
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
// features collected for a transcript, while loading the GTF
|
|
115
|
+
struct TxEntry {
|
|
116
|
+
TxInfo info;
|
|
117
|
+
std::map<std::string, int> cds_range = {{"max", 0}, {"min", 999999999}};
|
|
118
|
+
std::string symbol;
|
|
119
|
+
std::vector<std::string> alt_ids;
|
|
120
|
+
};
|
|
121
|
+
|
|
122
|
+
// build transcripts in the order they first appeared, freeing each entry once
|
|
123
|
+
// its transcript is built
|
|
124
|
+
//
|
|
125
|
+
// @param pos start of the current GTF line. Transcripts are only built once this
|
|
126
|
+
// passes their end, as no further lines for them can follow in a
|
|
127
|
+
// position-sorted GTF. Use -1 to build all transcripts.
|
|
128
|
+
static void build_transcripts(std::vector<NamedTx> & transcripts,
|
|
129
|
+
std::deque<TxEntry> & entries, std::unordered_map<std::string, TxEntry *> & index,
|
|
130
|
+
int pos=-1) {
|
|
131
|
+
while (!entries.empty()) {
|
|
132
|
+
TxEntry & x = entries.front();
|
|
133
|
+
if (pos != -1 && (x.info.end == 0 || pos <= x.info.end)) {
|
|
134
|
+
break;
|
|
135
|
+
}
|
|
136
|
+
index.erase(x.info.name);
|
|
137
|
+
add_transcript(transcripts, x.info, x.cds_range, x.symbol, x.alt_ids);
|
|
138
|
+
entries.pop_front();
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
|
|
75
142
|
// collect all features for a transcript into a single object
|
|
76
143
|
//
|
|
77
144
|
// When we load lines from gencode GTF files, each line represents a single exon
|
|
78
|
-
// or CDS, and we need to combine these based on transcript ID
|
|
145
|
+
// or CDS, and we need to combine these based on transcript ID. Lines for a
|
|
146
|
+
// transcript are usually contiguous, but position-sorted GTFs interleave
|
|
147
|
+
// transcripts, so features are collected by transcript ID until the chromosome
|
|
148
|
+
// changes (GTFs are grouped by chromosome), or the GTF moves past the transcript.
|
|
79
149
|
static void load_transcripts(std::vector<NamedTx> & transcripts, GTF >f_file, bool coding=true) {
|
|
80
150
|
std::set<std::string> permit = {"exon", "CDS", "UTR", "transcript",
|
|
81
151
|
"stop_codon", "start_codon"};
|
|
82
|
-
std::
|
|
83
|
-
std::string
|
|
84
|
-
|
|
85
|
-
std::vector<std::string> alt_ids;
|
|
86
|
-
std::string current;
|
|
87
|
-
TxInfo info;
|
|
152
|
+
std::deque<TxEntry> entries;
|
|
153
|
+
std::unordered_map<std::string, TxEntry *> index;
|
|
154
|
+
TxEntry * entry = nullptr;
|
|
88
155
|
|
|
89
156
|
GTFLine gtf;
|
|
90
157
|
|
|
91
|
-
while (
|
|
92
|
-
try {
|
|
93
|
-
gtf = gtf_file.next();
|
|
94
|
-
} catch (const std::out_of_range& e) {
|
|
95
|
-
break;
|
|
96
|
-
}
|
|
97
|
-
|
|
98
|
-
|
|
158
|
+
while (gtf_file.next(gtf)) {
|
|
99
159
|
if (permit.count(gtf.feature) == 0) {
|
|
100
160
|
continue;
|
|
101
161
|
} else if (coding && (gtf.transcript_type != "protein_coding")) {
|
|
102
162
|
continue;
|
|
103
163
|
}
|
|
104
164
|
|
|
105
|
-
|
|
106
|
-
if (
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
info.transcript_type = gtf.transcript_type;
|
|
165
|
+
// only look up the transcript when it differs from the previous line's
|
|
166
|
+
if (entry == nullptr || gtf.tx_id != entry->info.name || gtf.chrom != entry->info.chrom) {
|
|
167
|
+
int pos = (entry != nullptr && gtf.chrom != entry->info.chrom) ? -1 : gtf.start;
|
|
168
|
+
build_transcripts(transcripts, entries, index, pos);
|
|
169
|
+
auto it = index.find(gtf.tx_id);
|
|
170
|
+
if (it != index.end()) {
|
|
171
|
+
entry = it->second;
|
|
172
|
+
} else {
|
|
173
|
+
entries.emplace_back();
|
|
174
|
+
entry = &entries.back();
|
|
175
|
+
index[gtf.tx_id] = entry;
|
|
176
|
+
|
|
177
|
+
TxInfo & info = entry->info;
|
|
178
|
+
info.name = gtf.tx_id;
|
|
179
|
+
info.chrom = gtf.chrom;
|
|
180
|
+
info.strand = gtf.strand;
|
|
181
|
+
info.is_canonical = gtf.is_canonical;
|
|
182
|
+
info.transcript_type = gtf.transcript_type;
|
|
183
|
+
entry->symbol = gtf.symbol;
|
|
184
|
+
entry->alt_ids = gtf.alternate_ids;
|
|
185
|
+
if (gtf.feature != "transcript") {
|
|
186
|
+
// without a transcript line, use the first line's attributes,
|
|
187
|
+
// minus the fields specific to that feature
|
|
188
|
+
info.attributes = gtf.attributes;
|
|
189
|
+
for (auto field : {"exon_number", "exon_id", "exon_version"}) {
|
|
190
|
+
info.attributes.erase(field);
|
|
191
|
+
}
|
|
192
|
+
}
|
|
193
|
+
}
|
|
135
194
|
}
|
|
136
195
|
|
|
196
|
+
TxInfo & info = entry->info;
|
|
197
|
+
std::map<std::string, int> & cds_range = entry->cds_range;
|
|
137
198
|
if (gtf.feature == "transcript") {
|
|
138
199
|
info.start = gtf.start;
|
|
139
200
|
info.end = gtf.end;
|
|
@@ -150,14 +211,7 @@ static void load_transcripts(std::vector<NamedTx> & transcripts, GTF >f_file,
|
|
|
150
211
|
}
|
|
151
212
|
}
|
|
152
213
|
|
|
153
|
-
|
|
154
|
-
if (info.name != "") {
|
|
155
|
-
include_end_codons(cds_range, info);
|
|
156
|
-
Tx tx = Tx(info.name, info.chrom, info.start, info.end, info.strand[0], info.transcript_type, info.attributes);
|
|
157
|
-
tx.set_exons(info.exons);
|
|
158
|
-
tx.set_cds(info.cds);
|
|
159
|
-
transcripts.push_back({symbol, alt_ids, tx, info.is_canonical});
|
|
160
|
-
}
|
|
214
|
+
build_transcripts(transcripts, entries, index);
|
|
161
215
|
}
|
|
162
216
|
|
|
163
217
|
std::vector<NamedTx> open_gencode(std::string path, bool coding) {
|
|
@@ -177,10 +231,6 @@ std::vector<std::string> _in_region(std::string chrom, int start, int end,
|
|
|
177
231
|
std::map<std::string, std::vector<GenePoint>> & ends,
|
|
178
232
|
int max_window=2500000) {
|
|
179
233
|
|
|
180
|
-
if (chrom.size() < 3 || chrom.substr(0, 3) != "chr") {
|
|
181
|
-
chrom.insert(0, "chr");
|
|
182
|
-
}
|
|
183
|
-
|
|
184
234
|
if (starts.count(chrom) == 0) {
|
|
185
235
|
throw std::invalid_argument("unknown_chrom: " + chrom);
|
|
186
236
|
}
|
|
@@ -257,7 +307,7 @@ std::vector<std::string> _in_region(std::string chrom, int start, int end,
|
|
|
257
307
|
// gencode::open_gencode(path);
|
|
258
308
|
// }
|
|
259
309
|
//
|
|
260
|
-
// g++ -std=c++11 gencode.cpp gtf.cpp tx.cpp
|
|
310
|
+
// g++ -std=c++11 gencode.cpp gtf.cpp tx.cpp -lz
|
|
261
311
|
|
|
262
312
|
|
|
263
313
|
|