gencodegenes 1.1.6__tar.gz → 1.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/MANIFEST.in +2 -2
  2. {gencodegenes-1.1.6/src/gencodegenes.egg-info → gencodegenes-1.2.0}/PKG-INFO +26 -16
  3. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/README.md +23 -13
  4. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/pyproject.toml +7 -4
  5. gencodegenes-1.2.0/setup.py +69 -0
  6. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencode.cpp +109 -58
  7. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencode.h +3 -2
  8. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencodegenes/gencode.cpp +15445 -10225
  9. gencodegenes-1.2.0/src/gencodegenes/gencode.pyi +98 -0
  10. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencodegenes/gencode.pyx +122 -115
  11. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencodegenes/transcript.cpp +15955 -8069
  12. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencodegenes/transcript.pxd +11 -3
  13. gencodegenes-1.2.0/src/gencodegenes/transcript.pyi +137 -0
  14. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencodegenes/transcript.pyx +107 -12
  15. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencodegenes/tx.cpp +100 -55
  16. {gencodegenes-1.1.6/src → gencodegenes-1.2.0/src/gencodegenes}/tx.h +13 -5
  17. {gencodegenes-1.1.6 → gencodegenes-1.2.0/src/gencodegenes.egg-info}/PKG-INFO +26 -16
  18. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencodegenes.egg-info/SOURCES.txt +3 -2
  19. gencodegenes-1.2.0/src/gtf.cpp +300 -0
  20. gencodegenes-1.2.0/src/gtf.h +58 -0
  21. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/tx.cpp +100 -55
  22. {gencodegenes-1.1.6/src/gencodegenes → gencodegenes-1.2.0/src}/tx.h +13 -5
  23. gencodegenes-1.2.0/tests/__init__.py +0 -0
  24. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/tests/test_gencode.py +436 -14
  25. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/tests/test_sequence_methods.py +38 -0
  26. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/tests/test_transcript.py +153 -0
  27. gencodegenes-1.1.6/setup.py +0 -116
  28. gencodegenes-1.1.6/src/gtf.cpp +0 -180
  29. gencodegenes-1.1.6/src/gtf.h +0 -44
  30. gencodegenes-1.1.6/src/gzstream/gzstream.C +0 -165
  31. gencodegenes-1.1.6/src/gzstream/gzstream.h +0 -121
  32. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/LICENSE.txt +0 -0
  33. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/setup.cfg +0 -0
  34. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencodegenes/__init__.py +0 -0
  35. /gencodegenes-1.1.6/tests/__init__.py → /gencodegenes-1.2.0/src/gencodegenes/py.typed +0 -0
  36. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencodegenes.egg-info/dependency_links.txt +0 -0
  37. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencodegenes.egg-info/requires.txt +0 -0
  38. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/gencodegenes.egg-info/top_level.txt +0 -0
  39. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/adler32.c +0 -0
  40. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/compress.c +0 -0
  41. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/crc32.c +0 -0
  42. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/crc32.h +0 -0
  43. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/deflate.c +0 -0
  44. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/deflate.h +0 -0
  45. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/gzclose.c +0 -0
  46. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/gzguts.h +0 -0
  47. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/gzlib.c +0 -0
  48. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/gzread.c +0 -0
  49. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/gzwrite.c +0 -0
  50. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/infback.c +0 -0
  51. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/inffast.c +0 -0
  52. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/inffast.h +0 -0
  53. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/inffixed.h +0 -0
  54. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/inflate.c +0 -0
  55. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/inflate.h +0 -0
  56. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/inftrees.c +0 -0
  57. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/inftrees.h +0 -0
  58. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/trees.c +0 -0
  59. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/trees.h +0 -0
  60. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/uncompr.c +0 -0
  61. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/zconf.h +0 -0
  62. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/zlib.h +0 -0
  63. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/zutil.c +0 -0
  64. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/src/zlib/zutil.h +0 -0
  65. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/tests/data/example.grch38.fa +0 -0
  66. {gencodegenes-1.1.6 → gencodegenes-1.2.0}/tests/data/example.grch38.gtf +0 -0
@@ -5,10 +5,10 @@ include src/gencodegenes/*.cpp
5
5
  include src/gencodegenes/*.py
6
6
  include src/gencodegenes/*.pyx
7
7
  include src/gencodegenes/*.pxd
8
+ include src/gencodegenes/*.pyi
9
+ include src/gencodegenes/py.typed
8
10
  include src/*.h
9
11
  include src/*.cpp
10
- include src/gzstream/gzstream.C
11
- include src/gzstream/gzstream.h
12
12
 
13
13
  include src/zlib/*.c
14
14
  include src/zlib/*.h
@@ -1,13 +1,13 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gencodegenes
3
- Version: 1.1.6
3
+ Version: 1.2.0
4
4
  Summary: Package to load genes from GENCODE GTF files
5
5
  Author-email: Jeremy McRae <jeremy.mcrae@gmail.com>
6
+ License-Expression: MIT
6
7
  Project-URL: homepage, https://github.com/jeremymcrae/gencodegenes
7
- Classifier: License :: OSI Approved :: MIT License
8
8
  Classifier: Development Status :: 4 - Beta
9
9
  Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
10
- Requires-Python: >=3.8
10
+ Requires-Python: >=3.10
11
11
  Description-Content-Type: text/markdown
12
12
  License-File: LICENSE.txt
13
13
  Requires-Dist: pyfaidx>=0.5.8
@@ -16,7 +16,7 @@ Dynamic: license-file
16
16
 
17
17
  ### GENCODEGenes
18
18
 
19
- This package loads genes from GENCODE GTF/GFF files, groups transcripts by gene,
19
+ This package loads genes from GENCODE GTF files, groups transcripts by gene,
20
20
  and provides methods for transcripts, so you can find exon coordinates, CDS
21
21
  distances and sequences.
22
22
 
@@ -31,22 +31,30 @@ pip install gencodegenes
31
31
  from gencodegenes import Gencode
32
32
 
33
33
  gencode = Gencode(GTF_PATH)
34
- # full function arguments are Gencode(gtf_path, fasta_path=None, coding_only=True)
35
- # - fasta_path: pass in path to fasta file to get gene transcripts with sequence
34
+ # full function arguments are Gencode(gencode, fasta=None, coding_only=True)
35
+ # - gencode: path to GTF file (plain or gzipped)
36
+ # - fasta: pass in path to fasta file to get gene transcripts with sequence
36
37
  # - coding_only: pass in False to include all transcripts, not just protein coding
37
38
 
39
+ # use as a context manager to close the fasta when done
40
+ with Gencode(GTF_PATH, fasta=FASTA_PATH) as gencode:
41
+ ...
42
+
38
43
  # get gene by HGNC symbol
39
44
  gene = gencode['OR5A1']
40
45
  transcripts = gene.transcripts
41
- canonical = gene.canonical # picks MANE transcript if available, if none named
42
- # as MANE, picks the one tagged as appris_principal
43
- # (or longest CDS if multiple), if none tagged, picks
44
- # the longest protein coding, if none protein coding,
45
- # picks the longest cDNA
46
+ canonical = gene.canonical # picks the transcript tagged Ensembl_canonical (the
47
+ # MANE Select transcript in human, where one exists), if
48
+ # none tagged, picks from those tagged appris_principal,
49
+ # if none tagged, picks from all transcripts. Within
50
+ # these, picks the longest CDS, or the longest cDNA if
51
+ # none are protein coding
46
52
  gene.start, gene.end, gene.chrom, gene.strand, gene.symbol # other attributes available
53
+ gene.alternate_ids # gene_id and hgnc_id from the GTF
47
54
 
48
55
 
49
- # find gene nearest a genomic position, or overlapping a genomic region
56
+ # find gene nearest a genomic position, or overlapping a genomic region.
57
+ # Chromosomes match with or without the 'chr' prefix
50
58
  gencode.nearest('chr1', 1000000)
51
59
  gencode.in_region('chr1', 1000000, 2000000)
52
60
 
@@ -58,14 +66,15 @@ tx.get_coding_distance(pos) # get distance in CDS to CDS start
58
66
  tx.get_closest_exon(pos) # find exon closest to position
59
67
  tx.get_position_on_chrom(cds_pos) # convert CDS pos to genomic pos
60
68
  tx.get_codon_info(pos) # get info about codon for a site
61
- tx.get_codon_number_for_cds_pos(cds_pos) # convert CDS pos to codon number
62
- tx.translate(seq) # translate DNA to AA (if opened with Fasta)
69
+ tx.get_codon_number_for_cds_position(cds_pos) # convert CDS pos to codon number
70
+ tx.translate(seq) # translate DNA to AA
71
+ tx.consequence(pos, ref, alt) # get variant consequence (if opened with fasta)
63
72
 
64
73
  # the transcript also has associated data fields
65
74
  tx.name # transcript ID
66
75
  tx.chrom # transcript chromosome
67
- tx.start # transcript start (TSS)
68
- tx.end # transcript end
76
+ tx.start # transcript start (lowest position on chromosome)
77
+ tx.end # transcript end (highest position on chromosome)
69
78
  tx.cds_start # CDS start position
70
79
  tx.cds_end # CDS end position
71
80
  tx.type # transcript type e.g. protein_coding
@@ -73,5 +82,6 @@ tx.strand # strand (+ or -)
73
82
  tx.exons # list of exon coordinates
74
83
  tx.cds # list of CDS coordinates
75
84
  tx.cds_sequence # get cDNA sequence (if Gencode was opened with fasta)
85
+ tx.attributes # dict-like view of the GTF attributes e.g. tx.attributes['gene_id']
76
86
 
77
87
  ```
@@ -1,7 +1,7 @@
1
1
 
2
2
  ### GENCODEGenes
3
3
 
4
- This package loads genes from GENCODE GTF/GFF files, groups transcripts by gene,
4
+ This package loads genes from GENCODE GTF files, groups transcripts by gene,
5
5
  and provides methods for transcripts, so you can find exon coordinates, CDS
6
6
  distances and sequences.
7
7
 
@@ -16,22 +16,30 @@ pip install gencodegenes
16
16
  from gencodegenes import Gencode
17
17
 
18
18
  gencode = Gencode(GTF_PATH)
19
- # full function arguments are Gencode(gtf_path, fasta_path=None, coding_only=True)
20
- # - fasta_path: pass in path to fasta file to get gene transcripts with sequence
19
+ # full function arguments are Gencode(gencode, fasta=None, coding_only=True)
20
+ # - gencode: path to GTF file (plain or gzipped)
21
+ # - fasta: pass in path to fasta file to get gene transcripts with sequence
21
22
  # - coding_only: pass in False to include all transcripts, not just protein coding
22
23
 
24
+ # use as a context manager to close the fasta when done
25
+ with Gencode(GTF_PATH, fasta=FASTA_PATH) as gencode:
26
+ ...
27
+
23
28
  # get gene by HGNC symbol
24
29
  gene = gencode['OR5A1']
25
30
  transcripts = gene.transcripts
26
- canonical = gene.canonical # picks MANE transcript if available, if none named
27
- # as MANE, picks the one tagged as appris_principal
28
- # (or longest CDS if multiple), if none tagged, picks
29
- # the longest protein coding, if none protein coding,
30
- # picks the longest cDNA
31
+ canonical = gene.canonical # picks the transcript tagged Ensembl_canonical (the
32
+ # MANE Select transcript in human, where one exists), if
33
+ # none tagged, picks from those tagged appris_principal,
34
+ # if none tagged, picks from all transcripts. Within
35
+ # these, picks the longest CDS, or the longest cDNA if
36
+ # none are protein coding
31
37
  gene.start, gene.end, gene.chrom, gene.strand, gene.symbol # other attributes available
38
+ gene.alternate_ids # gene_id and hgnc_id from the GTF
32
39
 
33
40
 
34
- # find gene nearest a genomic position, or overlapping a genomic region
41
+ # find gene nearest a genomic position, or overlapping a genomic region.
42
+ # Chromosomes match with or without the 'chr' prefix
35
43
  gencode.nearest('chr1', 1000000)
36
44
  gencode.in_region('chr1', 1000000, 2000000)
37
45
 
@@ -43,14 +51,15 @@ tx.get_coding_distance(pos) # get distance in CDS to CDS start
43
51
  tx.get_closest_exon(pos) # find exon closest to position
44
52
  tx.get_position_on_chrom(cds_pos) # convert CDS pos to genomic pos
45
53
  tx.get_codon_info(pos) # get info about codon for a site
46
- tx.get_codon_number_for_cds_pos(cds_pos) # convert CDS pos to codon number
47
- tx.translate(seq) # translate DNA to AA (if opened with Fasta)
54
+ tx.get_codon_number_for_cds_position(cds_pos) # convert CDS pos to codon number
55
+ tx.translate(seq) # translate DNA to AA
56
+ tx.consequence(pos, ref, alt) # get variant consequence (if opened with fasta)
48
57
 
49
58
  # the transcript also has associated data fields
50
59
  tx.name # transcript ID
51
60
  tx.chrom # transcript chromosome
52
- tx.start # transcript start (TSS)
53
- tx.end # transcript end
61
+ tx.start # transcript start (lowest position on chromosome)
62
+ tx.end # transcript end (highest position on chromosome)
54
63
  tx.cds_start # CDS start position
55
64
  tx.cds_end # CDS end position
56
65
  tx.type # transcript type e.g. protein_coding
@@ -58,5 +67,6 @@ tx.strand # strand (+ or -)
58
67
  tx.exons # list of exon coordinates
59
68
  tx.cds # list of CDS coordinates
60
69
  tx.cds_sequence # get cDNA sequence (if Gencode was opened with fasta)
70
+ tx.attributes # dict-like view of the GTF attributes e.g. tx.attributes['gene_id']
61
71
 
62
72
  ```
@@ -1,13 +1,14 @@
1
1
  [build-system]
2
- requires = ["cython", "setuptools >= 40.6.0", "wheel"]
2
+ requires = ["cython", "setuptools >= 77.0.0"]
3
3
  build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = 'gencodegenes'
7
- version = '1.1.6'
7
+ version = '1.2.0'
8
8
  description = 'Package to load genes from GENCODE GTF files'
9
9
  readme = 'README.md'
10
- requires-python = ">=3.8"
10
+ license = "MIT"
11
+ requires-python = ">=3.10"
11
12
  authors = [
12
13
  {name = 'Jeremy McRae', email = 'jeremy.mcrae@gmail.com'}
13
14
  ]
@@ -17,7 +18,6 @@ dependencies = [
17
18
  ]
18
19
 
19
20
  classifiers = [
20
- "License :: OSI Approved :: MIT License",
21
21
  "Development Status :: 4 - Beta",
22
22
  "Topic :: Scientific/Engineering :: Bio-Informatics",
23
23
  ]
@@ -27,5 +27,8 @@ packages = [
27
27
  "gencodegenes"
28
28
  ]
29
29
 
30
+ [tool.setuptools.package-data]
31
+ gencodegenes = ['transcript.pxd', 'tx.h', 'tx.cpp', 'py.typed', '*.pyi']
32
+
30
33
  [project.urls]
31
34
  homepage = 'https://github.com/jeremymcrae/gencodegenes'
@@ -0,0 +1,69 @@
1
+
2
+ import glob
3
+ import os
4
+ import sys
5
+ import shutil
6
+
7
+ from setuptools import setup, Extension
8
+ from Cython.Build import cythonize
9
+
10
+ EXTRA_COMPILE_ARGS = []
11
+ EXTRA_LINK_ARGS = []
12
+ if sys.platform == "win32":
13
+ EXTRA_COMPILE_ARGS += ['/std:c++14']
14
+ else:
15
+ EXTRA_COMPILE_ARGS += ['-std=c++11']
16
+ if sys.platform == "darwin":
17
+ EXTRA_COMPILE_ARGS += ["-stdlib=libc++"]
18
+ EXTRA_LINK_ARGS += ["-stdlib=libc++"]
19
+ sdk_base = "/Library/Developer/CommandLineTools/SDKs/MacOSX.sdk"
20
+ if os.path.exists(sdk_base):
21
+ EXTRA_COMPILE_ARGS += [
22
+ f"-I{sdk_base}/usr/include/c++/v1",
23
+ f"-I{sdk_base}/usr/include",
24
+ ]
25
+ EXTRA_LINK_ARGS += [
26
+ f"-L{sdk_base}/usr/lib",
27
+ ]
28
+
29
+ gencode_sources = [
30
+ "src/gencodegenes/gencode.pyx",
31
+ "src/gencode.cpp",
32
+ "src/gtf.cpp",
33
+ "src/tx.cpp",
34
+ ]
35
+
36
+ libs = ['z']
37
+ include_dirs = ['src/']
38
+
39
+ if sys.platform == 'win32':
40
+ gencode_sources += glob.glob('src/zlib/*.c')
41
+ include_dirs.append('src/zlib/')
42
+ libs = []
43
+
44
+ extensions = [
45
+ Extension("gencodegenes.transcript",
46
+ extra_compile_args=EXTRA_COMPILE_ARGS,
47
+ extra_link_args=EXTRA_LINK_ARGS,
48
+ sources=[
49
+ "src/gencodegenes/transcript.pyx",
50
+ "src/tx.cpp"],
51
+ include_dirs=["src/"],
52
+ language="c++"),
53
+ Extension("gencodegenes.gencode",
54
+ extra_compile_args=EXTRA_COMPILE_ARGS,
55
+ extra_link_args=EXTRA_LINK_ARGS,
56
+ sources=gencode_sources,
57
+ include_dirs=include_dirs,
58
+ libraries=libs,
59
+ language="c++"),
60
+ ]
61
+
62
+ # include tx.h in the package, for downstream usage
63
+ shutil.copy("src/tx.h", "src/gencodegenes/tx.h")
64
+ shutil.copy("src/tx.cpp", "src/gencodegenes/tx.cpp")
65
+
66
+ setup(
67
+ package_dir={'': 'src'},
68
+ ext_modules=cythonize(extensions),
69
+ )
@@ -1,10 +1,12 @@
1
1
 
2
2
  #include <algorithm>
3
3
  #include <cstdint>
4
+ #include <deque>
4
5
  #include <string>
5
6
  #include <map>
6
7
  #include <set>
7
8
  #include <stdexcept>
9
+ #include <unordered_map>
8
10
  #include <vector>
9
11
 
10
12
  #include <iostream>
@@ -72,71 +74,131 @@ static void include_end_codons(std::map<std::string, int> cds_range, TxInfo & in
72
74
  }
73
75
  }
74
76
 
77
+ // set the transcript span from its exons and CDS, for GTFs without transcript lines
78
+ static void set_span(TxInfo & info) {
79
+ if (info.start != 0 || info.end != 0) {
80
+ return;
81
+ }
82
+ bool first = true;
83
+ for (auto regions : {&info.exons, &info.cds}) {
84
+ for (auto & x : *regions) {
85
+ info.start = first ? x[0] : std::min(info.start, x[0]);
86
+ info.end = first ? x[1] : std::max(info.end, x[1]);
87
+ first = false;
88
+ }
89
+ }
90
+ }
91
+
92
+ // construct a Tx from the features collected for a transcript
93
+ //
94
+ // Transcripts with inconsistent coordinates (e.g. a stop codon outside the
95
+ // exons) are skipped with a warning, so one malformed transcript doesn't stop
96
+ // the rest of the GTF from loading.
97
+ static void add_transcript(std::vector<NamedTx> & transcripts, TxInfo & info,
98
+ std::map<std::string, int> & cds_range, std::string & symbol,
99
+ std::vector<std::string> & alt_ids) {
100
+ try {
101
+ // adjust CDS for start and stop codon coords
102
+ include_end_codons(cds_range, info);
103
+ set_span(info);
104
+ Tx tx = Tx(info.name, info.chrom, info.start, info.end, info.strand[0],
105
+ info.transcript_type, info.attributes);
106
+ tx.set_exons(info.exons);
107
+ tx.set_cds(info.cds);
108
+ transcripts.push_back({symbol, alt_ids, tx, info.is_canonical});
109
+ } catch (const std::invalid_argument & e) {
110
+ std::cerr << "skipping transcript " << info.name << ": " << e.what() << std::endl;
111
+ }
112
+ }
113
+
114
+ // features collected for a transcript, while loading the GTF
115
+ struct TxEntry {
116
+ TxInfo info;
117
+ std::map<std::string, int> cds_range = {{"max", 0}, {"min", 999999999}};
118
+ std::string symbol;
119
+ std::vector<std::string> alt_ids;
120
+ };
121
+
122
+ // build transcripts in the order they first appeared, freeing each entry once
123
+ // its transcript is built
124
+ //
125
+ // @param pos start of the current GTF line. Transcripts are only built once this
126
+ // passes their end, as no further lines for them can follow in a
127
+ // position-sorted GTF. Use -1 to build all transcripts.
128
+ static void build_transcripts(std::vector<NamedTx> & transcripts,
129
+ std::deque<TxEntry> & entries, std::unordered_map<std::string, TxEntry *> & index,
130
+ int pos=-1) {
131
+ while (!entries.empty()) {
132
+ TxEntry & x = entries.front();
133
+ if (pos != -1 && (x.info.end == 0 || pos <= x.info.end)) {
134
+ break;
135
+ }
136
+ index.erase(x.info.name);
137
+ add_transcript(transcripts, x.info, x.cds_range, x.symbol, x.alt_ids);
138
+ entries.pop_front();
139
+ }
140
+ }
141
+
75
142
  // collect all features for a transcript into a single object
76
143
  //
77
144
  // When we load lines from gencode GTF files, each line represents a single exon
78
- // or CDS, and we need to combine these based on transcript ID
145
+ // or CDS, and we need to combine these based on transcript ID. Lines for a
146
+ // transcript are usually contiguous, but position-sorted GTFs interleave
147
+ // transcripts, so features are collected by transcript ID until the chromosome
148
+ // changes (GTFs are grouped by chromosome), or the GTF moves past the transcript.
79
149
  static void load_transcripts(std::vector<NamedTx> & transcripts, GTF &gtf_file, bool coding=true) {
80
150
  std::set<std::string> permit = {"exon", "CDS", "UTR", "transcript",
81
151
  "stop_codon", "start_codon"};
82
- std::map<std::string, int> cds_range = {{"max", 0}, {"min", 999999999}};
83
- std::string tx_id = "";
84
- std::string symbol = "";
85
- std::vector<std::string> alt_ids;
86
- std::string current;
87
- TxInfo info;
152
+ std::deque<TxEntry> entries;
153
+ std::unordered_map<std::string, TxEntry *> index;
154
+ TxEntry * entry = nullptr;
88
155
 
89
156
  GTFLine gtf;
90
157
 
91
- while (true) {
92
- try {
93
- gtf = gtf_file.next();
94
- } catch (const std::out_of_range& e) {
95
- break;
96
- }
97
-
98
-
158
+ while (gtf_file.next(gtf)) {
99
159
  if (permit.count(gtf.feature) == 0) {
100
160
  continue;
101
161
  } else if (coding && (gtf.transcript_type != "protein_coding")) {
102
162
  continue;
103
163
  }
104
164
 
105
- current = gtf.tx_id;
106
- if (tx_id == "") {
107
- tx_id = current;
108
- symbol = gtf.symbol;
109
- alt_ids = gtf.alternate_ids;
110
- }
111
-
112
- if (tx_id != current) {
113
- // adjust CDS for start and stop codon coords
114
- include_end_codons(cds_range, info);
115
- Tx tx = Tx(info.name, info.chrom, info.start, info.end, info.strand[0],
116
- info.transcript_type);
117
- tx.set_exons(info.exons);
118
- tx.set_cds(info.cds);
119
- transcripts.push_back({symbol, alt_ids, tx, info.is_canonical});
120
- info = {};
121
- tx_id = current;
122
- cds_range["max"] = 0;
123
- cds_range["min"] = 999999999;
124
- symbol = gtf.symbol;
125
- alt_ids = gtf.alternate_ids;
126
- info.is_canonical = false;
127
- }
128
-
129
- if (info.name == "") {
130
- info.name = tx_id;
131
- info.chrom = gtf.chrom;
132
- info.strand = gtf.strand;
133
- info.is_canonical = gtf.is_canonical;
134
- info.transcript_type = gtf.transcript_type;
165
+ // only look up the transcript when it differs from the previous line's
166
+ if (entry == nullptr || gtf.tx_id != entry->info.name || gtf.chrom != entry->info.chrom) {
167
+ int pos = (entry != nullptr && gtf.chrom != entry->info.chrom) ? -1 : gtf.start;
168
+ build_transcripts(transcripts, entries, index, pos);
169
+ auto it = index.find(gtf.tx_id);
170
+ if (it != index.end()) {
171
+ entry = it->second;
172
+ } else {
173
+ entries.emplace_back();
174
+ entry = &entries.back();
175
+ index[gtf.tx_id] = entry;
176
+
177
+ TxInfo & info = entry->info;
178
+ info.name = gtf.tx_id;
179
+ info.chrom = gtf.chrom;
180
+ info.strand = gtf.strand;
181
+ info.is_canonical = gtf.is_canonical;
182
+ info.transcript_type = gtf.transcript_type;
183
+ entry->symbol = gtf.symbol;
184
+ entry->alt_ids = gtf.alternate_ids;
185
+ if (gtf.feature != "transcript") {
186
+ // without a transcript line, use the first line's attributes,
187
+ // minus the fields specific to that feature
188
+ info.attributes = gtf.attributes;
189
+ for (auto field : {"exon_number", "exon_id", "exon_version"}) {
190
+ info.attributes.erase(field);
191
+ }
192
+ }
193
+ }
135
194
  }
136
195
 
196
+ TxInfo & info = entry->info;
197
+ std::map<std::string, int> & cds_range = entry->cds_range;
137
198
  if (gtf.feature == "transcript") {
138
199
  info.start = gtf.start;
139
200
  info.end = gtf.end;
201
+ info.attributes = std::move(gtf.attributes);
140
202
  } else if (gtf.feature == "CDS") {
141
203
  info.cds.push_back(std::vector<int> {gtf.start, gtf.end});
142
204
  cds_range["max"] = std::max(std::max(cds_range["max"], gtf.start), gtf.end);
@@ -149,14 +211,7 @@ static void load_transcripts(std::vector<NamedTx> & transcripts, GTF &gtf_file,
149
211
  }
150
212
  }
151
213
 
152
- // also include the final transcript (if it transcript exists)
153
- if (info.name != "") {
154
- include_end_codons(cds_range, info);
155
- Tx tx = Tx(info.name, info.chrom, info.start, info.end, info.strand[0], info.transcript_type);
156
- tx.set_exons(info.exons);
157
- tx.set_cds(info.cds);
158
- transcripts.push_back({symbol, alt_ids, tx, info.is_canonical});
159
- }
214
+ build_transcripts(transcripts, entries, index);
160
215
  }
161
216
 
162
217
  std::vector<NamedTx> open_gencode(std::string path, bool coding) {
@@ -176,10 +231,6 @@ std::vector<std::string> _in_region(std::string chrom, int start, int end,
176
231
  std::map<std::string, std::vector<GenePoint>> & ends,
177
232
  int max_window=2500000) {
178
233
 
179
- if (chrom.size() < 3 || chrom.substr(0, 3) != "chr") {
180
- chrom.insert(0, "chr");
181
- }
182
-
183
234
  if (starts.count(chrom) == 0) {
184
235
  throw std::invalid_argument("unknown_chrom: " + chrom);
185
236
  }
@@ -256,7 +307,7 @@ std::vector<std::string> _in_region(std::string chrom, int start, int end,
256
307
  // gencode::open_gencode(path);
257
308
  // }
258
309
  //
259
- // g++ -std=c++11 gencode.cpp gtf.cpp tx.cpp gzstream/gzstream.C -Igzstream -lz
310
+ // g++ -std=c++11 gencode.cpp gtf.cpp tx.cpp -lz
260
311
 
261
312
 
262
313
 
@@ -17,14 +17,15 @@ namespace gencode {
17
17
  struct TxInfo {
18
18
  std::string name = "";
19
19
  std::string chrom;
20
- int start;
21
- int end;
20
+ int start = 0;
21
+ int end = 0;
22
22
  std::string strand;
23
23
  std::string transcript_type;
24
24
  std::vector<std::vector<int> > exons;
25
25
  std::vector<std::vector<int> > cds;
26
26
  int offset = 0;
27
27
  int is_canonical = 0;
28
+ std::map<std::string, std::string> attributes;
28
29
  };
29
30
 
30
31
  // stores HGNC symbol with the transcript, so we can collect transcripts by gene