gencodegenes 1.1.6__tar.gz → 1.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/MANIFEST.in +2 -2
- {gencodegenes-1.1.6/src/gencodegenes.egg-info → gencodegenes-1.2.0}/PKG-INFO +26 -16
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/README.md +23 -13
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/pyproject.toml +7 -4
- gencodegenes-1.2.0/setup.py +69 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencode.cpp +109 -58
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencode.h +3 -2
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencodegenes/gencode.cpp +15445 -10225
- gencodegenes-1.2.0/src/gencodegenes/gencode.pyi +98 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencodegenes/gencode.pyx +122 -115
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencodegenes/transcript.cpp +15955 -8069
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencodegenes/transcript.pxd +11 -3
- gencodegenes-1.2.0/src/gencodegenes/transcript.pyi +137 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencodegenes/transcript.pyx +107 -12
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencodegenes/tx.cpp +100 -55
- {gencodegenes-1.1.6/src → gencodegenes-1.2.0/src/gencodegenes}/tx.h +13 -5
- {gencodegenes-1.1.6 → gencodegenes-1.2.0/src/gencodegenes.egg-info}/PKG-INFO +26 -16
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencodegenes.egg-info/SOURCES.txt +3 -2
- gencodegenes-1.2.0/src/gtf.cpp +300 -0
- gencodegenes-1.2.0/src/gtf.h +58 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/tx.cpp +100 -55
- {gencodegenes-1.1.6/src/gencodegenes → gencodegenes-1.2.0/src}/tx.h +13 -5
- gencodegenes-1.2.0/tests/__init__.py +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/tests/test_gencode.py +436 -14
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/tests/test_sequence_methods.py +38 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/tests/test_transcript.py +153 -0
- gencodegenes-1.1.6/setup.py +0 -116
- gencodegenes-1.1.6/src/gtf.cpp +0 -180
- gencodegenes-1.1.6/src/gtf.h +0 -44
- gencodegenes-1.1.6/src/gzstream/gzstream.C +0 -165
- gencodegenes-1.1.6/src/gzstream/gzstream.h +0 -121
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/LICENSE.txt +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/setup.cfg +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencodegenes/__init__.py +0 -0
- /gencodegenes-1.1.6/tests/__init__.py → /gencodegenes-1.2.0/src/gencodegenes/py.typed +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencodegenes.egg-info/dependency_links.txt +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencodegenes.egg-info/requires.txt +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencodegenes.egg-info/top_level.txt +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/adler32.c +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/compress.c +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/crc32.c +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/crc32.h +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/deflate.c +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/deflate.h +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/gzclose.c +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/gzguts.h +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/gzlib.c +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/gzread.c +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/gzwrite.c +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/infback.c +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/inffast.c +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/inffast.h +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/inffixed.h +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/inflate.c +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/inflate.h +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/inftrees.c +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/inftrees.h +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/trees.c +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/trees.h +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/uncompr.c +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/zconf.h +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/zlib.h +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/zutil.c +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/zutil.h +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/tests/data/example.grch38.fa +0 -0
- {gencodegenes-1.1.6 → gencodegenes-1.2.0}/tests/data/example.grch38.gtf +0 -0
|
@@ -5,10 +5,10 @@ include src/gencodegenes/*.cpp
|
|
|
5
5
|
include src/gencodegenes/*.py
|
|
6
6
|
include src/gencodegenes/*.pyx
|
|
7
7
|
include src/gencodegenes/*.pxd
|
|
8
|
+
include src/gencodegenes/*.pyi
|
|
9
|
+
include src/gencodegenes/py.typed
|
|
8
10
|
include src/*.h
|
|
9
11
|
include src/*.cpp
|
|
10
|
-
include src/gzstream/gzstream.C
|
|
11
|
-
include src/gzstream/gzstream.h
|
|
12
12
|
|
|
13
13
|
include src/zlib/*.c
|
|
14
14
|
include src/zlib/*.h
|
|
@@ -1,13 +1,13 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gencodegenes
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.2.0
|
|
4
4
|
Summary: Package to load genes from GENCODE GTF files
|
|
5
5
|
Author-email: Jeremy McRae <jeremy.mcrae@gmail.com>
|
|
6
|
+
License-Expression: MIT
|
|
6
7
|
Project-URL: homepage, https://github.com/jeremymcrae/gencodegenes
|
|
7
|
-
Classifier: License :: OSI Approved :: MIT License
|
|
8
8
|
Classifier: Development Status :: 4 - Beta
|
|
9
9
|
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
10
|
-
Requires-Python: >=3.
|
|
10
|
+
Requires-Python: >=3.10
|
|
11
11
|
Description-Content-Type: text/markdown
|
|
12
12
|
License-File: LICENSE.txt
|
|
13
13
|
Requires-Dist: pyfaidx>=0.5.8
|
|
@@ -16,7 +16,7 @@ Dynamic: license-file
|
|
|
16
16
|
|
|
17
17
|
### GENCODEGenes
|
|
18
18
|
|
|
19
|
-
This package loads genes from GENCODE GTF
|
|
19
|
+
This package loads genes from GENCODE GTF files, groups transcripts by gene,
|
|
20
20
|
and provides methods for transcripts, so you can find exon coordinates, CDS
|
|
21
21
|
distances and sequences.
|
|
22
22
|
|
|
@@ -31,22 +31,30 @@ pip install gencodegenes
|
|
|
31
31
|
from gencodegenes import Gencode
|
|
32
32
|
|
|
33
33
|
gencode = Gencode(GTF_PATH)
|
|
34
|
-
# full function arguments are Gencode(
|
|
35
|
-
# -
|
|
34
|
+
# full function arguments are Gencode(gencode, fasta=None, coding_only=True)
|
|
35
|
+
# - gencode: path to GTF file (plain or gzipped)
|
|
36
|
+
# - fasta: pass in path to fasta file to get gene transcripts with sequence
|
|
36
37
|
# - coding_only: pass in False to include all transcripts, not just protein coding
|
|
37
38
|
|
|
39
|
+
# use as a context manager to close the fasta when done
|
|
40
|
+
with Gencode(GTF_PATH, fasta=FASTA_PATH) as gencode:
|
|
41
|
+
...
|
|
42
|
+
|
|
38
43
|
# get gene by HGNC symbol
|
|
39
44
|
gene = gencode['OR5A1']
|
|
40
45
|
transcripts = gene.transcripts
|
|
41
|
-
canonical = gene.canonical # picks
|
|
42
|
-
#
|
|
43
|
-
#
|
|
44
|
-
#
|
|
45
|
-
# picks the longest cDNA
|
|
46
|
+
canonical = gene.canonical # picks the transcript tagged Ensembl_canonical (the
|
|
47
|
+
# MANE Select transcript in human, where one exists), if
|
|
48
|
+
# none tagged, picks from those tagged appris_principal,
|
|
49
|
+
# if none tagged, picks from all transcripts. Within
|
|
50
|
+
# these, picks the longest CDS, or the longest cDNA if
|
|
51
|
+
# none are protein coding
|
|
46
52
|
gene.start, gene.end, gene.chrom, gene.strand, gene.symbol # other attributes available
|
|
53
|
+
gene.alternate_ids # gene_id and hgnc_id from the GTF
|
|
47
54
|
|
|
48
55
|
|
|
49
|
-
# find gene nearest a genomic position, or overlapping a genomic region
|
|
56
|
+
# find gene nearest a genomic position, or overlapping a genomic region.
|
|
57
|
+
# Chromosomes match with or without the 'chr' prefix
|
|
50
58
|
gencode.nearest('chr1', 1000000)
|
|
51
59
|
gencode.in_region('chr1', 1000000, 2000000)
|
|
52
60
|
|
|
@@ -58,14 +66,15 @@ tx.get_coding_distance(pos) # get distance in CDS to CDS start
|
|
|
58
66
|
tx.get_closest_exon(pos) # find exon closest to position
|
|
59
67
|
tx.get_position_on_chrom(cds_pos) # convert CDS pos to genomic pos
|
|
60
68
|
tx.get_codon_info(pos) # get info about codon for a site
|
|
61
|
-
tx.
|
|
62
|
-
tx.translate(seq) # translate DNA to AA
|
|
69
|
+
tx.get_codon_number_for_cds_position(cds_pos) # convert CDS pos to codon number
|
|
70
|
+
tx.translate(seq) # translate DNA to AA
|
|
71
|
+
tx.consequence(pos, ref, alt) # get variant consequence (if opened with fasta)
|
|
63
72
|
|
|
64
73
|
# the transcript also has associated data fields
|
|
65
74
|
tx.name # transcript ID
|
|
66
75
|
tx.chrom # transcript chromosome
|
|
67
|
-
tx.start # transcript start (
|
|
68
|
-
tx.end # transcript end
|
|
76
|
+
tx.start # transcript start (lowest position on chromosome)
|
|
77
|
+
tx.end # transcript end (highest position on chromosome)
|
|
69
78
|
tx.cds_start # CDS start position
|
|
70
79
|
tx.cds_end # CDS end position
|
|
71
80
|
tx.type # transcript type e.g. protein_coding
|
|
@@ -73,5 +82,6 @@ tx.strand # strand (+ or -)
|
|
|
73
82
|
tx.exons # list of exon coordinates
|
|
74
83
|
tx.cds # list of CDS coordinates
|
|
75
84
|
tx.cds_sequence # get cDNA sequence (if Gencode was opened with fasta)
|
|
85
|
+
tx.attributes # dict-like view of the GTF attributes e.g. tx.attributes['gene_id']
|
|
76
86
|
|
|
77
87
|
```
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
|
|
2
2
|
### GENCODEGenes
|
|
3
3
|
|
|
4
|
-
This package loads genes from GENCODE GTF
|
|
4
|
+
This package loads genes from GENCODE GTF files, groups transcripts by gene,
|
|
5
5
|
and provides methods for transcripts, so you can find exon coordinates, CDS
|
|
6
6
|
distances and sequences.
|
|
7
7
|
|
|
@@ -16,22 +16,30 @@ pip install gencodegenes
|
|
|
16
16
|
from gencodegenes import Gencode
|
|
17
17
|
|
|
18
18
|
gencode = Gencode(GTF_PATH)
|
|
19
|
-
# full function arguments are Gencode(
|
|
20
|
-
# -
|
|
19
|
+
# full function arguments are Gencode(gencode, fasta=None, coding_only=True)
|
|
20
|
+
# - gencode: path to GTF file (plain or gzipped)
|
|
21
|
+
# - fasta: pass in path to fasta file to get gene transcripts with sequence
|
|
21
22
|
# - coding_only: pass in False to include all transcripts, not just protein coding
|
|
22
23
|
|
|
24
|
+
# use as a context manager to close the fasta when done
|
|
25
|
+
with Gencode(GTF_PATH, fasta=FASTA_PATH) as gencode:
|
|
26
|
+
...
|
|
27
|
+
|
|
23
28
|
# get gene by HGNC symbol
|
|
24
29
|
gene = gencode['OR5A1']
|
|
25
30
|
transcripts = gene.transcripts
|
|
26
|
-
canonical = gene.canonical # picks
|
|
27
|
-
#
|
|
28
|
-
#
|
|
29
|
-
#
|
|
30
|
-
# picks the longest cDNA
|
|
31
|
+
canonical = gene.canonical # picks the transcript tagged Ensembl_canonical (the
|
|
32
|
+
# MANE Select transcript in human, where one exists), if
|
|
33
|
+
# none tagged, picks from those tagged appris_principal,
|
|
34
|
+
# if none tagged, picks from all transcripts. Within
|
|
35
|
+
# these, picks the longest CDS, or the longest cDNA if
|
|
36
|
+
# none are protein coding
|
|
31
37
|
gene.start, gene.end, gene.chrom, gene.strand, gene.symbol # other attributes available
|
|
38
|
+
gene.alternate_ids # gene_id and hgnc_id from the GTF
|
|
32
39
|
|
|
33
40
|
|
|
34
|
-
# find gene nearest a genomic position, or overlapping a genomic region
|
|
41
|
+
# find gene nearest a genomic position, or overlapping a genomic region.
|
|
42
|
+
# Chromosomes match with or without the 'chr' prefix
|
|
35
43
|
gencode.nearest('chr1', 1000000)
|
|
36
44
|
gencode.in_region('chr1', 1000000, 2000000)
|
|
37
45
|
|
|
@@ -43,14 +51,15 @@ tx.get_coding_distance(pos) # get distance in CDS to CDS start
|
|
|
43
51
|
tx.get_closest_exon(pos) # find exon closest to position
|
|
44
52
|
tx.get_position_on_chrom(cds_pos) # convert CDS pos to genomic pos
|
|
45
53
|
tx.get_codon_info(pos) # get info about codon for a site
|
|
46
|
-
tx.
|
|
47
|
-
tx.translate(seq) # translate DNA to AA
|
|
54
|
+
tx.get_codon_number_for_cds_position(cds_pos) # convert CDS pos to codon number
|
|
55
|
+
tx.translate(seq) # translate DNA to AA
|
|
56
|
+
tx.consequence(pos, ref, alt) # get variant consequence (if opened with fasta)
|
|
48
57
|
|
|
49
58
|
# the transcript also has associated data fields
|
|
50
59
|
tx.name # transcript ID
|
|
51
60
|
tx.chrom # transcript chromosome
|
|
52
|
-
tx.start # transcript start (
|
|
53
|
-
tx.end # transcript end
|
|
61
|
+
tx.start # transcript start (lowest position on chromosome)
|
|
62
|
+
tx.end # transcript end (highest position on chromosome)
|
|
54
63
|
tx.cds_start # CDS start position
|
|
55
64
|
tx.cds_end # CDS end position
|
|
56
65
|
tx.type # transcript type e.g. protein_coding
|
|
@@ -58,5 +67,6 @@ tx.strand # strand (+ or -)
|
|
|
58
67
|
tx.exons # list of exon coordinates
|
|
59
68
|
tx.cds # list of CDS coordinates
|
|
60
69
|
tx.cds_sequence # get cDNA sequence (if Gencode was opened with fasta)
|
|
70
|
+
tx.attributes # dict-like view of the GTF attributes e.g. tx.attributes['gene_id']
|
|
61
71
|
|
|
62
72
|
```
|
|
@@ -1,13 +1,14 @@
|
|
|
1
1
|
[build-system]
|
|
2
|
-
requires = ["cython", "setuptools >=
|
|
2
|
+
requires = ["cython", "setuptools >= 77.0.0"]
|
|
3
3
|
build-backend = "setuptools.build_meta"
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = 'gencodegenes'
|
|
7
|
-
version = '1.
|
|
7
|
+
version = '1.2.0'
|
|
8
8
|
description = 'Package to load genes from GENCODE GTF files'
|
|
9
9
|
readme = 'README.md'
|
|
10
|
-
|
|
10
|
+
license = "MIT"
|
|
11
|
+
requires-python = ">=3.10"
|
|
11
12
|
authors = [
|
|
12
13
|
{name = 'Jeremy McRae', email = 'jeremy.mcrae@gmail.com'}
|
|
13
14
|
]
|
|
@@ -17,7 +18,6 @@ dependencies = [
|
|
|
17
18
|
]
|
|
18
19
|
|
|
19
20
|
classifiers = [
|
|
20
|
-
"License :: OSI Approved :: MIT License",
|
|
21
21
|
"Development Status :: 4 - Beta",
|
|
22
22
|
"Topic :: Scientific/Engineering :: Bio-Informatics",
|
|
23
23
|
]
|
|
@@ -27,5 +27,8 @@ packages = [
|
|
|
27
27
|
"gencodegenes"
|
|
28
28
|
]
|
|
29
29
|
|
|
30
|
+
[tool.setuptools.package-data]
|
|
31
|
+
gencodegenes = ['transcript.pxd', 'tx.h', 'tx.cpp', 'py.typed', '*.pyi']
|
|
32
|
+
|
|
30
33
|
[project.urls]
|
|
31
34
|
homepage = 'https://github.com/jeremymcrae/gencodegenes'
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
|
|
2
|
+
import glob
|
|
3
|
+
import os
|
|
4
|
+
import sys
|
|
5
|
+
import shutil
|
|
6
|
+
|
|
7
|
+
from setuptools import setup, Extension
|
|
8
|
+
from Cython.Build import cythonize
|
|
9
|
+
|
|
10
|
+
EXTRA_COMPILE_ARGS = []
|
|
11
|
+
EXTRA_LINK_ARGS = []
|
|
12
|
+
if sys.platform == "win32":
|
|
13
|
+
EXTRA_COMPILE_ARGS += ['/std:c++14']
|
|
14
|
+
else:
|
|
15
|
+
EXTRA_COMPILE_ARGS += ['-std=c++11']
|
|
16
|
+
if sys.platform == "darwin":
|
|
17
|
+
EXTRA_COMPILE_ARGS += ["-stdlib=libc++"]
|
|
18
|
+
EXTRA_LINK_ARGS += ["-stdlib=libc++"]
|
|
19
|
+
sdk_base = "/Library/Developer/CommandLineTools/SDKs/MacOSX.sdk"
|
|
20
|
+
if os.path.exists(sdk_base):
|
|
21
|
+
EXTRA_COMPILE_ARGS += [
|
|
22
|
+
f"-I{sdk_base}/usr/include/c++/v1",
|
|
23
|
+
f"-I{sdk_base}/usr/include",
|
|
24
|
+
]
|
|
25
|
+
EXTRA_LINK_ARGS += [
|
|
26
|
+
f"-L{sdk_base}/usr/lib",
|
|
27
|
+
]
|
|
28
|
+
|
|
29
|
+
gencode_sources = [
|
|
30
|
+
"src/gencodegenes/gencode.pyx",
|
|
31
|
+
"src/gencode.cpp",
|
|
32
|
+
"src/gtf.cpp",
|
|
33
|
+
"src/tx.cpp",
|
|
34
|
+
]
|
|
35
|
+
|
|
36
|
+
libs = ['z']
|
|
37
|
+
include_dirs = ['src/']
|
|
38
|
+
|
|
39
|
+
if sys.platform == 'win32':
|
|
40
|
+
gencode_sources += glob.glob('src/zlib/*.c')
|
|
41
|
+
include_dirs.append('src/zlib/')
|
|
42
|
+
libs = []
|
|
43
|
+
|
|
44
|
+
extensions = [
|
|
45
|
+
Extension("gencodegenes.transcript",
|
|
46
|
+
extra_compile_args=EXTRA_COMPILE_ARGS,
|
|
47
|
+
extra_link_args=EXTRA_LINK_ARGS,
|
|
48
|
+
sources=[
|
|
49
|
+
"src/gencodegenes/transcript.pyx",
|
|
50
|
+
"src/tx.cpp"],
|
|
51
|
+
include_dirs=["src/"],
|
|
52
|
+
language="c++"),
|
|
53
|
+
Extension("gencodegenes.gencode",
|
|
54
|
+
extra_compile_args=EXTRA_COMPILE_ARGS,
|
|
55
|
+
extra_link_args=EXTRA_LINK_ARGS,
|
|
56
|
+
sources=gencode_sources,
|
|
57
|
+
include_dirs=include_dirs,
|
|
58
|
+
libraries=libs,
|
|
59
|
+
language="c++"),
|
|
60
|
+
]
|
|
61
|
+
|
|
62
|
+
# include tx.h in the package, for downstream usage
|
|
63
|
+
shutil.copy("src/tx.h", "src/gencodegenes/tx.h")
|
|
64
|
+
shutil.copy("src/tx.cpp", "src/gencodegenes/tx.cpp")
|
|
65
|
+
|
|
66
|
+
setup(
|
|
67
|
+
package_dir={'': 'src'},
|
|
68
|
+
ext_modules=cythonize(extensions),
|
|
69
|
+
)
|
|
@@ -1,10 +1,12 @@
|
|
|
1
1
|
|
|
2
2
|
#include <algorithm>
|
|
3
3
|
#include <cstdint>
|
|
4
|
+
#include <deque>
|
|
4
5
|
#include <string>
|
|
5
6
|
#include <map>
|
|
6
7
|
#include <set>
|
|
7
8
|
#include <stdexcept>
|
|
9
|
+
#include <unordered_map>
|
|
8
10
|
#include <vector>
|
|
9
11
|
|
|
10
12
|
#include <iostream>
|
|
@@ -72,71 +74,131 @@ static void include_end_codons(std::map<std::string, int> cds_range, TxInfo & in
|
|
|
72
74
|
}
|
|
73
75
|
}
|
|
74
76
|
|
|
77
|
+
// set the transcript span from its exons and CDS, for GTFs without transcript lines
|
|
78
|
+
static void set_span(TxInfo & info) {
|
|
79
|
+
if (info.start != 0 || info.end != 0) {
|
|
80
|
+
return;
|
|
81
|
+
}
|
|
82
|
+
bool first = true;
|
|
83
|
+
for (auto regions : {&info.exons, &info.cds}) {
|
|
84
|
+
for (auto & x : *regions) {
|
|
85
|
+
info.start = first ? x[0] : std::min(info.start, x[0]);
|
|
86
|
+
info.end = first ? x[1] : std::max(info.end, x[1]);
|
|
87
|
+
first = false;
|
|
88
|
+
}
|
|
89
|
+
}
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
// construct a Tx from the features collected for a transcript
|
|
93
|
+
//
|
|
94
|
+
// Transcripts with inconsistent coordinates (e.g. a stop codon outside the
|
|
95
|
+
// exons) are skipped with a warning, so one malformed transcript doesn't stop
|
|
96
|
+
// the rest of the GTF from loading.
|
|
97
|
+
static void add_transcript(std::vector<NamedTx> & transcripts, TxInfo & info,
|
|
98
|
+
std::map<std::string, int> & cds_range, std::string & symbol,
|
|
99
|
+
std::vector<std::string> & alt_ids) {
|
|
100
|
+
try {
|
|
101
|
+
// adjust CDS for start and stop codon coords
|
|
102
|
+
include_end_codons(cds_range, info);
|
|
103
|
+
set_span(info);
|
|
104
|
+
Tx tx = Tx(info.name, info.chrom, info.start, info.end, info.strand[0],
|
|
105
|
+
info.transcript_type, info.attributes);
|
|
106
|
+
tx.set_exons(info.exons);
|
|
107
|
+
tx.set_cds(info.cds);
|
|
108
|
+
transcripts.push_back({symbol, alt_ids, tx, info.is_canonical});
|
|
109
|
+
} catch (const std::invalid_argument & e) {
|
|
110
|
+
std::cerr << "skipping transcript " << info.name << ": " << e.what() << std::endl;
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
// features collected for a transcript, while loading the GTF
|
|
115
|
+
struct TxEntry {
|
|
116
|
+
TxInfo info;
|
|
117
|
+
std::map<std::string, int> cds_range = {{"max", 0}, {"min", 999999999}};
|
|
118
|
+
std::string symbol;
|
|
119
|
+
std::vector<std::string> alt_ids;
|
|
120
|
+
};
|
|
121
|
+
|
|
122
|
+
// build transcripts in the order they first appeared, freeing each entry once
|
|
123
|
+
// its transcript is built
|
|
124
|
+
//
|
|
125
|
+
// @param pos start of the current GTF line. Transcripts are only built once this
|
|
126
|
+
// passes their end, as no further lines for them can follow in a
|
|
127
|
+
// position-sorted GTF. Use -1 to build all transcripts.
|
|
128
|
+
static void build_transcripts(std::vector<NamedTx> & transcripts,
|
|
129
|
+
std::deque<TxEntry> & entries, std::unordered_map<std::string, TxEntry *> & index,
|
|
130
|
+
int pos=-1) {
|
|
131
|
+
while (!entries.empty()) {
|
|
132
|
+
TxEntry & x = entries.front();
|
|
133
|
+
if (pos != -1 && (x.info.end == 0 || pos <= x.info.end)) {
|
|
134
|
+
break;
|
|
135
|
+
}
|
|
136
|
+
index.erase(x.info.name);
|
|
137
|
+
add_transcript(transcripts, x.info, x.cds_range, x.symbol, x.alt_ids);
|
|
138
|
+
entries.pop_front();
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
|
|
75
142
|
// collect all features for a transcript into a single object
|
|
76
143
|
//
|
|
77
144
|
// When we load lines from gencode GTF files, each line represents a single exon
|
|
78
|
-
// or CDS, and we need to combine these based on transcript ID
|
|
145
|
+
// or CDS, and we need to combine these based on transcript ID. Lines for a
|
|
146
|
+
// transcript are usually contiguous, but position-sorted GTFs interleave
|
|
147
|
+
// transcripts, so features are collected by transcript ID until the chromosome
|
|
148
|
+
// changes (GTFs are grouped by chromosome), or the GTF moves past the transcript.
|
|
79
149
|
static void load_transcripts(std::vector<NamedTx> & transcripts, GTF >f_file, bool coding=true) {
|
|
80
150
|
std::set<std::string> permit = {"exon", "CDS", "UTR", "transcript",
|
|
81
151
|
"stop_codon", "start_codon"};
|
|
82
|
-
std::
|
|
83
|
-
std::string
|
|
84
|
-
|
|
85
|
-
std::vector<std::string> alt_ids;
|
|
86
|
-
std::string current;
|
|
87
|
-
TxInfo info;
|
|
152
|
+
std::deque<TxEntry> entries;
|
|
153
|
+
std::unordered_map<std::string, TxEntry *> index;
|
|
154
|
+
TxEntry * entry = nullptr;
|
|
88
155
|
|
|
89
156
|
GTFLine gtf;
|
|
90
157
|
|
|
91
|
-
while (
|
|
92
|
-
try {
|
|
93
|
-
gtf = gtf_file.next();
|
|
94
|
-
} catch (const std::out_of_range& e) {
|
|
95
|
-
break;
|
|
96
|
-
}
|
|
97
|
-
|
|
98
|
-
|
|
158
|
+
while (gtf_file.next(gtf)) {
|
|
99
159
|
if (permit.count(gtf.feature) == 0) {
|
|
100
160
|
continue;
|
|
101
161
|
} else if (coding && (gtf.transcript_type != "protein_coding")) {
|
|
102
162
|
continue;
|
|
103
163
|
}
|
|
104
164
|
|
|
105
|
-
|
|
106
|
-
if (
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
info.transcript_type = gtf.transcript_type;
|
|
165
|
+
// only look up the transcript when it differs from the previous line's
|
|
166
|
+
if (entry == nullptr || gtf.tx_id != entry->info.name || gtf.chrom != entry->info.chrom) {
|
|
167
|
+
int pos = (entry != nullptr && gtf.chrom != entry->info.chrom) ? -1 : gtf.start;
|
|
168
|
+
build_transcripts(transcripts, entries, index, pos);
|
|
169
|
+
auto it = index.find(gtf.tx_id);
|
|
170
|
+
if (it != index.end()) {
|
|
171
|
+
entry = it->second;
|
|
172
|
+
} else {
|
|
173
|
+
entries.emplace_back();
|
|
174
|
+
entry = &entries.back();
|
|
175
|
+
index[gtf.tx_id] = entry;
|
|
176
|
+
|
|
177
|
+
TxInfo & info = entry->info;
|
|
178
|
+
info.name = gtf.tx_id;
|
|
179
|
+
info.chrom = gtf.chrom;
|
|
180
|
+
info.strand = gtf.strand;
|
|
181
|
+
info.is_canonical = gtf.is_canonical;
|
|
182
|
+
info.transcript_type = gtf.transcript_type;
|
|
183
|
+
entry->symbol = gtf.symbol;
|
|
184
|
+
entry->alt_ids = gtf.alternate_ids;
|
|
185
|
+
if (gtf.feature != "transcript") {
|
|
186
|
+
// without a transcript line, use the first line's attributes,
|
|
187
|
+
// minus the fields specific to that feature
|
|
188
|
+
info.attributes = gtf.attributes;
|
|
189
|
+
for (auto field : {"exon_number", "exon_id", "exon_version"}) {
|
|
190
|
+
info.attributes.erase(field);
|
|
191
|
+
}
|
|
192
|
+
}
|
|
193
|
+
}
|
|
135
194
|
}
|
|
136
195
|
|
|
196
|
+
TxInfo & info = entry->info;
|
|
197
|
+
std::map<std::string, int> & cds_range = entry->cds_range;
|
|
137
198
|
if (gtf.feature == "transcript") {
|
|
138
199
|
info.start = gtf.start;
|
|
139
200
|
info.end = gtf.end;
|
|
201
|
+
info.attributes = std::move(gtf.attributes);
|
|
140
202
|
} else if (gtf.feature == "CDS") {
|
|
141
203
|
info.cds.push_back(std::vector<int> {gtf.start, gtf.end});
|
|
142
204
|
cds_range["max"] = std::max(std::max(cds_range["max"], gtf.start), gtf.end);
|
|
@@ -149,14 +211,7 @@ static void load_transcripts(std::vector<NamedTx> & transcripts, GTF >f_file,
|
|
|
149
211
|
}
|
|
150
212
|
}
|
|
151
213
|
|
|
152
|
-
|
|
153
|
-
if (info.name != "") {
|
|
154
|
-
include_end_codons(cds_range, info);
|
|
155
|
-
Tx tx = Tx(info.name, info.chrom, info.start, info.end, info.strand[0], info.transcript_type);
|
|
156
|
-
tx.set_exons(info.exons);
|
|
157
|
-
tx.set_cds(info.cds);
|
|
158
|
-
transcripts.push_back({symbol, alt_ids, tx, info.is_canonical});
|
|
159
|
-
}
|
|
214
|
+
build_transcripts(transcripts, entries, index);
|
|
160
215
|
}
|
|
161
216
|
|
|
162
217
|
std::vector<NamedTx> open_gencode(std::string path, bool coding) {
|
|
@@ -176,10 +231,6 @@ std::vector<std::string> _in_region(std::string chrom, int start, int end,
|
|
|
176
231
|
std::map<std::string, std::vector<GenePoint>> & ends,
|
|
177
232
|
int max_window=2500000) {
|
|
178
233
|
|
|
179
|
-
if (chrom.size() < 3 || chrom.substr(0, 3) != "chr") {
|
|
180
|
-
chrom.insert(0, "chr");
|
|
181
|
-
}
|
|
182
|
-
|
|
183
234
|
if (starts.count(chrom) == 0) {
|
|
184
235
|
throw std::invalid_argument("unknown_chrom: " + chrom);
|
|
185
236
|
}
|
|
@@ -256,7 +307,7 @@ std::vector<std::string> _in_region(std::string chrom, int start, int end,
|
|
|
256
307
|
// gencode::open_gencode(path);
|
|
257
308
|
// }
|
|
258
309
|
//
|
|
259
|
-
// g++ -std=c++11 gencode.cpp gtf.cpp tx.cpp
|
|
310
|
+
// g++ -std=c++11 gencode.cpp gtf.cpp tx.cpp -lz
|
|
260
311
|
|
|
261
312
|
|
|
262
313
|
|
|
@@ -17,14 +17,15 @@ namespace gencode {
|
|
|
17
17
|
struct TxInfo {
|
|
18
18
|
std::string name = "";
|
|
19
19
|
std::string chrom;
|
|
20
|
-
int start;
|
|
21
|
-
int end;
|
|
20
|
+
int start = 0;
|
|
21
|
+
int end = 0;
|
|
22
22
|
std::string strand;
|
|
23
23
|
std::string transcript_type;
|
|
24
24
|
std::vector<std::vector<int> > exons;
|
|
25
25
|
std::vector<std::vector<int> > cds;
|
|
26
26
|
int offset = 0;
|
|
27
27
|
int is_canonical = 0;
|
|
28
|
+
std::map<std::string, std::string> attributes;
|
|
28
29
|
};
|
|
29
30
|
|
|
30
31
|
// stores HGNC symbol with the transcript, so we can collect transcripts by gene
|