gencodegenes 1.1.0__tar.gz → 1.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. {gencodegenes-1.1.0/src/gencodegenes.egg-info → gencodegenes-1.1.1}/PKG-INFO +5 -2
  2. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/README.md +4 -1
  3. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/setup.py +1 -1
  4. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencodegenes/tx.cpp +10 -0
  5. {gencodegenes-1.1.0 → gencodegenes-1.1.1/src/gencodegenes.egg-info}/PKG-INFO +5 -2
  6. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gtf.cpp +31 -20
  7. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gtf.h +1 -1
  8. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/tx.cpp +10 -0
  9. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/tests/test_gencode.py +58 -33
  10. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/LICENSE.txt +0 -0
  11. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/MANIFEST.in +0 -0
  12. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/pyproject.toml +0 -0
  13. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/setup.cfg +0 -0
  14. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencode.cpp +0 -0
  15. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencode.h +0 -0
  16. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencodegenes/__init__.py +0 -0
  17. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencodegenes/gencode.cpp +0 -0
  18. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencodegenes/gencode.pyx +0 -0
  19. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencodegenes/transcript.cpp +0 -0
  20. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencodegenes/transcript.pxd +0 -0
  21. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencodegenes/transcript.pyx +0 -0
  22. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencodegenes/tx.h +0 -0
  23. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencodegenes.egg-info/SOURCES.txt +0 -0
  24. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencodegenes.egg-info/dependency_links.txt +0 -0
  25. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencodegenes.egg-info/requires.txt +0 -0
  26. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencodegenes.egg-info/top_level.txt +0 -0
  27. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gzstream/gzstream.C +0 -0
  28. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gzstream/gzstream.h +0 -0
  29. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/tx.h +0 -0
  30. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/tests/__init__.py +0 -0
  31. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/tests/data/example.grch38.fa +0 -0
  32. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/tests/data/example.grch38.gtf +0 -0
  33. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/tests/test_sequence_methods.py +0 -0
  34. {gencodegenes-1.1.0 → gencodegenes-1.1.1}/tests/test_transcript.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: gencodegenes
3
- Version: 1.1.0
3
+ Version: 1.1.1
4
4
  Summary: Package to load genes from GENCODE GTF files
5
5
  Home-page: https://github.com/jeremymcrae/gencodegenes
6
6
  Author: Jeremy McRae
@@ -29,7 +29,10 @@ pip install gencodegenes
29
29
  ```py
30
30
  from gencodegenes import Gencode
31
31
 
32
- gencode = Gencode(GTF_PATH) # or Gencode(GTF, FASTA_PATH) to give transcripts DNA sequence
32
+ gencode = Gencode(GTF_PATH)
33
+ # full function arguments are Gencode(gtf_path, fasta_path=None, coding_only=True)
34
+ # - fasta_path: pass in path to fasta file to get gene transcripts with sequence
35
+ # - coding_only: pass in False to include all transcripts, not just protein coding
33
36
 
34
37
  # get gene by HGNC symbol
35
38
  gene = gencode['OR5A1']
@@ -14,7 +14,10 @@ pip install gencodegenes
14
14
  ```py
15
15
  from gencodegenes import Gencode
16
16
 
17
- gencode = Gencode(GTF_PATH) # or Gencode(GTF, FASTA_PATH) to give transcripts DNA sequence
17
+ gencode = Gencode(GTF_PATH)
18
+ # full function arguments are Gencode(gtf_path, fasta_path=None, coding_only=True)
19
+ # - fasta_path: pass in path to fasta file to get gene transcripts with sequence
20
+ # - coding_only: pass in False to include all transcripts, not just protein coding
18
21
 
19
22
  # get gene by HGNC symbol
20
23
  gene = gencode['OR5A1']
@@ -112,7 +112,7 @@ setup(name="gencodegenes",
112
112
  description='Package to load genes from GENCODE GTF files',
113
113
  long_description=io.open('README.md', encoding='utf-8').read(),
114
114
  long_description_content_type='text/markdown',
115
- version="1.1.0",
115
+ version="1.1.1",
116
116
  author="Jeremy McRae",
117
117
  author_email="jeremy.mcrae@gmail.com",
118
118
  license="MIT",
@@ -66,6 +66,10 @@ void Tx::set_exons(std::vector<std::vector<int>> exon_ranges) {
66
66
  void Tx::set_cds(std::vector<std::vector<int>> cds_ranges) {
67
67
  cds.clear();
68
68
 
69
+ if (cds_ranges.size() == 0) {
70
+ return;
71
+ }
72
+
69
73
  // If the transcript lacks exon coordinates and only has a single CDS
70
74
  // region, then if the CDS region fits within the gene range, make a
71
75
  // single exon, using the transcript start and end. This prevents issues
@@ -247,6 +251,9 @@ int Tx::closest_exon_num(int pos, std::vector<Region> & group) {
247
251
  //
248
252
  // @param position integer chromosome position e.g. 10000000
249
253
  Region Tx::get_closest_exon(int pos) {
254
+ if (exons.size() == 0) {
255
+ throw std::invalid_argument("no exons assigned to this transcript");
256
+ }
250
257
  int idx = closest_exon_num(pos);
251
258
  return exons[idx];
252
259
  }
@@ -255,6 +262,9 @@ Region Tx::get_closest_exon(int pos) {
255
262
  //
256
263
  // @param position integer chromosome position e.g. 10000000
257
264
  bool Tx::in_coding_region(int pos) {
265
+ if (cds.size() == 0) {
266
+ return false;
267
+ }
258
268
  int idx = closest_exon_num(pos, cds);
259
269
  Region region = cds[idx];
260
270
  return (pos >= region.start) && (pos <= region.end);
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: gencodegenes
3
- Version: 1.1.0
3
+ Version: 1.1.1
4
4
  Summary: Package to load genes from GENCODE GTF files
5
5
  Home-page: https://github.com/jeremymcrae/gencodegenes
6
6
  Author: Jeremy McRae
@@ -29,7 +29,10 @@ pip install gencodegenes
29
29
  ```py
30
30
  from gencodegenes import Gencode
31
31
 
32
- gencode = Gencode(GTF_PATH) # or Gencode(GTF, FASTA_PATH) to give transcripts DNA sequence
32
+ gencode = Gencode(GTF_PATH)
33
+ # full function arguments are Gencode(gtf_path, fasta_path=None, coding_only=True)
34
+ # - fasta_path: pass in path to fasta file to get gene transcripts with sequence
35
+ # - coding_only: pass in False to include all transcripts, not just protein coding
33
36
 
34
37
  # get gene by HGNC symbol
35
38
  gene = gencode['OR5A1']
@@ -14,18 +14,25 @@
14
14
 
15
15
  namespace gencode {
16
16
 
17
- // parse the required firleds from the attributes field
17
+ std::string trim(const std::string &s, const std::string &vals) {
18
+ size_t start = s.find_first_not_of(vals);
19
+ if (start == std::string::npos) {
20
+ return "";
21
+ }
22
+ size_t end = s.find_last_not_of(vals);
23
+ return s.substr(start, (end + 1) - start);
24
+ }
25
+
26
+ // parse the required fields from the attributes field
18
27
  static void get_attributes_fields(GTFLine &info, std::string &line, int offset) {
19
- // we could check for each field individually, but since we know the order
20
- // of the fields, it's much quicker to just search the remaining substring
21
- const std::string tx_id_key = "transcript_id \"";
22
- const std::string gene_id_key = "gene_id \"";
23
- const std::string gene_name_key = "gene_name \"";
24
- std::string type_key = "transcript_type \"";
25
- const std::string hgnc_id_key = "hgnc_id \"";
28
+ const std::string tx_id_key = "transcript_id ";
29
+ const std::string gene_id_key = "gene_id ";
30
+ const std::string gene_name_key = "gene_name ";
31
+ std::string type_key = "transcript_type ";
32
+ const std::string hgnc_id_key = "hgnc_id ";
26
33
 
27
34
  size_t tx_start = line.find(tx_id_key, offset) + tx_id_key.size();
28
- size_t tx_end = line.find("\"", tx_start);
35
+ size_t tx_end = line.find(";", tx_start);
29
36
 
30
37
  if (tx_start - tx_id_key.size() == std::string::npos) {
31
38
  // handle if the string was not found
@@ -34,7 +41,7 @@ static void get_attributes_fields(GTFLine &info, std::string &line, int offset)
34
41
  }
35
42
 
36
43
  size_t gene_id_start = line.find(gene_id_key, offset) + gene_id_key.size();
37
- size_t gene_id_end = line.find("\"", gene_id_start);
44
+ size_t gene_id_end = line.find(";", gene_id_start);
38
45
  if (gene_id_start - gene_id_key.size() == std::string::npos) {
39
46
  // handle if the string was not found
40
47
  gene_id_start = offset;
@@ -42,7 +49,7 @@ static void get_attributes_fields(GTFLine &info, std::string &line, int offset)
42
49
  }
43
50
 
44
51
  size_t gene_start = line.find(gene_name_key, offset) + gene_name_key.size();
45
- size_t gene_end = line.find("\"", gene_start);
52
+ size_t gene_end = line.find(";", gene_start);
46
53
 
47
54
  if (gene_start - gene_name_key.size() == std::string::npos) {
48
55
  // handle if the string was not found
@@ -53,10 +60,10 @@ static void get_attributes_fields(GTFLine &info, std::string &line, int offset)
53
60
  size_t type_start = line.find(type_key, offset) + type_key.size();
54
61
  if (type_start - type_key.size() == std::string::npos) {
55
62
  // allow for alternate transcript_type key, as found in non-gencode GTF files
56
- type_key = "transcript_biotype \"";
63
+ type_key = "transcript_biotype ";
57
64
  type_start = line.find(type_key, offset) + type_key.size();
58
65
  }
59
- size_t type_end = line.find("\"", type_start);
66
+ size_t type_end = line.find(";", type_start);
60
67
 
61
68
  if (type_start - type_key.size() == std::string::npos) {
62
69
  // handle if the string was not found
@@ -65,25 +72,29 @@ static void get_attributes_fields(GTFLine &info, std::string &line, int offset)
65
72
  }
66
73
 
67
74
  size_t hgnc_id_start = line.find(hgnc_id_key, offset) + hgnc_id_key.size();
68
- size_t hgnc_id_end = line.find("\"", hgnc_id_start);
75
+ size_t hgnc_id_end = line.find(";", hgnc_id_start);
69
76
  if (hgnc_id_start - hgnc_id_key.size() == std::string::npos) {
70
77
  // handle if the string was not found
71
78
  hgnc_id_start = offset;
72
79
  hgnc_id_end = offset;
73
80
  }
74
81
 
75
- info.symbol = line.substr(gene_start, gene_end - gene_start);
76
- info.tx_id = line.substr(tx_start, tx_end - tx_start);
77
- info.transcript_type = line.substr(type_start, type_end - type_start);
82
+ info.symbol = trim(line.substr(gene_start, gene_end - gene_start), " \"");
83
+ info.tx_id = trim(line.substr(tx_start, tx_end - tx_start), " \"");
84
+ info.transcript_type = trim(line.substr(type_start, type_end - type_start), " \"");
78
85
 
79
86
  if (gene_id_start != gene_id_end) {
80
- info.alternate_ids.push_back(line.substr(gene_id_start, gene_id_end - gene_id_start));
87
+ std::string gene_id = trim(line.substr(gene_id_start, gene_id_end - gene_id_start), " \"");
88
+ if (info.symbol.size() == 0) {
89
+ info.symbol = gene_id;
90
+ } else {
91
+ info.alternate_ids.push_back(gene_id);
92
+ }
81
93
  }
82
94
  if (hgnc_id_start != hgnc_id_end) {
83
- info.alternate_ids.push_back(line.substr(hgnc_id_start, hgnc_id_end - hgnc_id_start));
95
+ info.alternate_ids.push_back(trim(line.substr(hgnc_id_start, hgnc_id_end - hgnc_id_start), " \""));
84
96
  }
85
97
 
86
- info.is_canonical = 0;
87
98
  if (info.feature == "transcript") {
88
99
  if (line.find("appris_principal", offset) != std::string::npos) {
89
100
  info.is_canonical = 5;
@@ -21,7 +21,7 @@ struct GTFLine {
21
21
  std::vector<std::string> alternate_ids;
22
22
  std::string tx_id;
23
23
  std::string transcript_type;
24
- int is_canonical;
24
+ int is_canonical = 0;
25
25
  };
26
26
 
27
27
  GTFLine parse_gtfline(std::string &line);
@@ -66,6 +66,10 @@ void Tx::set_exons(std::vector<std::vector<int>> exon_ranges) {
66
66
  void Tx::set_cds(std::vector<std::vector<int>> cds_ranges) {
67
67
  cds.clear();
68
68
 
69
+ if (cds_ranges.size() == 0) {
70
+ return;
71
+ }
72
+
69
73
  // If the transcript lacks exon coordinates and only has a single CDS
70
74
  // region, then if the CDS region fits within the gene range, make a
71
75
  // single exon, using the transcript start and end. This prevents issues
@@ -247,6 +251,9 @@ int Tx::closest_exon_num(int pos, std::vector<Region> & group) {
247
251
  //
248
252
  // @param position integer chromosome position e.g. 10000000
249
253
  Region Tx::get_closest_exon(int pos) {
254
+ if (exons.size() == 0) {
255
+ throw std::invalid_argument("no exons assigned to this transcript");
256
+ }
250
257
  int idx = closest_exon_num(pos);
251
258
  return exons[idx];
252
259
  }
@@ -255,6 +262,9 @@ Region Tx::get_closest_exon(int pos) {
255
262
  //
256
263
  // @param position integer chromosome position e.g. 10000000
257
264
  bool Tx::in_coding_region(int pos) {
265
+ if (cds.size() == 0) {
266
+ return false;
267
+ }
258
268
  int idx = closest_exon_num(pos, cds);
259
269
  Region region = cds[idx];
260
270
  return (pos >= region.start) && (pos <= region.end);
@@ -55,12 +55,12 @@ class TestGencode(unittest.TestCase):
55
55
  lines = '##format: gtf\n' \
56
56
  'chr1\tHAVANA\tgene\t10\t20\t.\t-\t.\tgene_name "TEST1";\n' \
57
57
  'chr1\tHAVANA\ttranscript\t10\t20\t.\t-\t.\ttranscript_id "ENST_A";gene_name "TEST1"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
58
- 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n' \
59
- 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n' \
58
+ 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n' \
59
+ 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n' \
60
60
  'chr1\tHAVANA\tgene\t100\t110\t.\t-\t.\tgene_name "TEST2";\n' \
61
61
  'chr1\tHAVANA\ttranscript\t100\t110\t.\t-\t.\ttranscript_id "ENST_B";gene_name "TEST2"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
62
- 'chr1\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST2"; transcript_type "protein_coding"\n' \
63
- 'chr1\tHAVANA\tCDS\t105\t110\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST2"; transcript_type "protein_coding"\n'\
62
+ 'chr1\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding;"\n' \
63
+ 'chr1\tHAVANA\tCDS\t105\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding;"\n'\
64
64
  'chr2\tHAVANA\tgene\t100\t110\t.\t-\t.\tgene_name "TEST3";\n' \
65
65
  'chr2\tHAVANA\ttranscript\t100\t110\t.\t-\t.\ttranscript_id "ENST_C";gene_name "TEST3"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
66
66
  'chr2\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_C" gene_name "TEST3"; transcript_type "protein_coding"\n' \
@@ -112,16 +112,16 @@ class TestGencode(unittest.TestCase):
112
112
  lines = ['##format: gtf\n',
113
113
  'chr1\tHAVANA\tgene\t10\t20\t.\t-\t.\tgene_name "TEST1";\n',
114
114
  'chr1\tHAVANA\ttranscript\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding"; tag "appris_principal_1";\n',
115
- 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n',
116
- 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n',
115
+ 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n',
116
+ 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n',
117
117
  'chr1\tHAVANA\tgene\t100\t110\t.\t-\t.\tgene_name "TEST2";\n',
118
118
  'chr1\tHAVANA\ttranscript\t100\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding"; tag "appris_principal_1";\n',
119
- 'chr1\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST2"; transcript_type "protein_coding"\n',
120
- 'chr1\tHAVANA\tCDS\t105\t110\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST2"; transcript_type "protein_coding"\n',
119
+ 'chr1\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding;"\n',
120
+ 'chr1\tHAVANA\tCDS\t105\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding;"\n',
121
121
  'chr2\tHAVANA\tgene\t100\t110\t.\t-\t.\tgene_name "TEST3";\n',
122
122
  'chr2\tHAVANA\ttranscript\t100\t110\t.\t-\t.\ttranscript_id "ENST_C"; gene_name "TEST3"; transcript_type "protein_coding"; tag "appris_principal_1";\n',
123
- 'chr2\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_C" gene_name "TEST3"; transcript_type "protein_coding"\n',
124
- 'chr2\tHAVANA\tCDS\t105\t110\t.\t-\t.\ttranscript_id "ENST_C" gene_name "TEST3"; transcript_type "protein_coding"\n']
123
+ 'chr2\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_C"; gene_name "TEST3"; transcript_type "protein_coding;"\n',
124
+ 'chr2\tHAVANA\tCDS\t105\t110\t.\t-\t.\ttranscript_id "ENST_C"; gene_name "TEST3"; transcript_type "protein_coding;"\n']
125
125
 
126
126
  write_gtf(self.temp_gtf_path, lines)
127
127
  make_fasta(self.temp_fasta_path, ['chr1', 'chr2'])
@@ -150,12 +150,12 @@ class TestGencode(unittest.TestCase):
150
150
  lines = ['##format: gtf\n',
151
151
  'chr1\tHAVANA\tgene\t20\t30\t.\t-\t.\tgene_name "TEST1";\n',
152
152
  'chr1\tHAVANA\ttranscript\t20\t30\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding"; tag "appris_principal_1";\n',
153
- 'chr1\tHAVANA\texon\t20\t30\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n',
154
- 'chr1\tHAVANA\tCDS\t25\t30\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n',
153
+ 'chr1\tHAVANA\texon\t20\t30\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n',
154
+ 'chr1\tHAVANA\tCDS\t25\t30\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n',
155
155
  'chr1\tHAVANA\tgene\t100\t110\t.\t-\t.\tgene_name "TEST1";\n',
156
156
  'chr1\tHAVANA\ttranscript\t100\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST1"; transcript_type "protein_coding";\n',
157
- 'chr1\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST1"; transcript_type "protein_coding"\n',
158
- 'chr1\tHAVANA\tCDS\t110\t100\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST1"; transcript_type "protein_coding"\n',
157
+ 'chr1\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST1"; transcript_type "protein_coding;"\n',
158
+ 'chr1\tHAVANA\tCDS\t110\t100\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST1"; transcript_type "protein_coding;"\n',
159
159
  ]
160
160
 
161
161
  write_gtf(self.temp_gtf_path, lines)
@@ -231,7 +231,7 @@ class TestGencode(unittest.TestCase):
231
231
  'gene_status "KNOWN"; gene_name "OR4F5"; transcript_type "protein_coding"; ' \
232
232
  'transcript_status "KNOWN"; transcript_name "OR4F5-001"; level 2; ' \
233
233
  'protein_id "ENSP00000334393.3"; tag "basic"; transcript_support_level "NA"; ' \
234
- 'hgnc_id "HGNC:14825", tag "appris_principal_1"; tag "CCDS"; ccdsid "CCDS30547.1"; ' \
234
+ 'hgnc_id "HGNC:14825"; tag "appris_principal_1"; tag "CCDS"; ccdsid "CCDS30547.1"; ' \
235
235
  'havana_gene "OTTHUMG00000001094.2"; havana_transcript "OTTHUMT00000003223.2";\n'
236
236
  obj = _parse_gtfline(line.encode('utf8'))
237
237
  expected = {'chrom': b'chr1',
@@ -251,7 +251,7 @@ class TestGencode(unittest.TestCase):
251
251
  line = 'chr1\tHAVANA\ttranscript\t69091\t70008\t.\t+\t.\t '\
252
252
  'transcript_id "ENST00000335137.3"; gene_type "protein_coding"; ' \
253
253
  'gene_status "KNOWN"; gene_name "OR4F5"; transcript_type "protein_coding";' \
254
- 'hgnc_id "HGNC:14825", tag "appris_principal_1"; tag "CCDS"; ccdsid "CCDS30547.1"; '
254
+ 'hgnc_id "HGNC:14825"; tag "appris_principal_1"; tag "CCDS"; ccdsid "CCDS30547.1"; '
255
255
  obj = _parse_gtfline(line.encode('utf8'))
256
256
  expected = {'chrom': b'chr1',
257
257
  'feature': b'transcript',
@@ -431,12 +431,12 @@ class TestGencode(unittest.TestCase):
431
431
  lines = '##format: gtf\n' \
432
432
  'chr1\tHAVANA\tgene\t10\t20\t.\t-\t.\tgene_name "TEST1";\n' \
433
433
  'chr1\tHAVANA\ttranscript\t10\t20\t.\t-\t.\ttranscript_id "ENST_A";gene_name "TEST1"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
434
- 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n' \
435
- 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n' \
434
+ 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n' \
435
+ 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n' \
436
436
  'chr1\tHAVANA\tgene\t10\t30\t.\t-\t.\tgene_name "TEST2";\n' \
437
437
  'chr1\tHAVANA\ttranscript\t10\t30\t.\t-\t.\ttranscript_id "ENST_B";gene_name "TEST2"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
438
- 'chr1\tHAVANA\texon\t10\t30\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST2"; transcript_type "protein_coding"\n' \
439
- 'chr1\tHAVANA\tCDS\t15\t30\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST2"; transcript_type "protein_coding"\n'
438
+ 'chr1\tHAVANA\texon\t10\t30\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding;"\n' \
439
+ 'chr1\tHAVANA\tCDS\t15\t30\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding;"\n'
440
440
 
441
441
  write_gtf(self.temp_gtf_path, lines)
442
442
  data = _open_gencode(self.temp_gtf_path)
@@ -452,8 +452,8 @@ class TestGencode(unittest.TestCase):
452
452
  lines = '##format: gtf\n' \
453
453
  'chr1\tHAVANA\tgene\t10\t20\t.\t-\t.\tgene_name "TEST";\n' \
454
454
  'chr1\tHAVANA\ttranscript\t10\t20\t.\t-\t.\ttranscript_id "ENST_A";gene_name "TEST"; transcript_type "processed_transcript"; tag "appris_principal_1";\n' \
455
- 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "processed_transcript"\n' \
456
- 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "processed_transcript"\n' \
455
+ 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "processed_transcript;"\n' \
456
+ 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "processed_transcript;"\n' \
457
457
 
458
458
  write_gtf(self.temp_gtf_path, lines)
459
459
  data = _open_gencode(self.temp_gtf_path)
@@ -472,12 +472,12 @@ class TestGencode(unittest.TestCase):
472
472
  lines = '##format: gtf\n' \
473
473
  'chr1\tHAVANA\tgene\t10\t20\t.\t-\t.\tgene_name "TEST";\n' \
474
474
  'chr1\tHAVANA\ttranscript\t10\t20\t.\t-\t.\ttranscript_id "ENST_A";gene_name "TEST"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
475
- 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
476
- 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
475
+ 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
476
+ 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
477
477
  'chr1\tHAVANA\tgene\t10\t30\t.\t-\t.\tgene_name "TEST";\n' \
478
478
  'chr1\tHAVANA\ttranscript\t10\t30\t.\t-\t.\ttranscript_id "ENST_B";gene_name "TEST"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
479
- 'chr1\tHAVANA\texon\t10\t30\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST"; transcript_type "protein_coding"\n' \
480
- 'chr1\tHAVANA\tCDS\t15\t30\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST"; transcript_type "protein_coding"\n'
479
+ 'chr1\tHAVANA\texon\t10\t30\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
480
+ 'chr1\tHAVANA\tCDS\t15\t30\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST"; transcript_type "protein_coding;"\n'
481
481
 
482
482
  write_gtf(self.temp_gtf_path, lines)
483
483
  data = _open_gencode(self.temp_gtf_path)
@@ -497,13 +497,38 @@ class TestGencode(unittest.TestCase):
497
497
  lines = '##format: gtf\n' \
498
498
  'chr1\tHAVANA\tgene\t10\t100\t.\t-\t.\tgene_name "TEST";\n' \
499
499
  'chr1\tHAVANA\ttranscript\t10\t100\t.\t-\t.\ttranscript_id "ENST_A";gene_name "TEST"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
500
- 'chr1\tHAVANA\tUTR\t10\t15\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding";\n' \
501
- 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
502
- 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
503
- 'chr1\tHAVANA\texon\t30\t40\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
504
- 'chr1\tHAVANA\tCDS\t30\t40\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
505
- 'chr1\tHAVANA\texon\t90\t100\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
506
- 'chr1\tHAVANA\tUTR\t90\t100\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n'
500
+ 'chr1\tHAVANA\tUTR\t10\t15\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding";\n' \
501
+ 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
502
+ 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
503
+ 'chr1\tHAVANA\texon\t30\t40\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
504
+ 'chr1\tHAVANA\tCDS\t30\t40\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
505
+ 'chr1\tHAVANA\texon\t90\t100\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
506
+ 'chr1\tHAVANA\tUTR\t90\t100\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n'
507
+
508
+ write_gtf(self.temp_gtf_path, lines)
509
+ data = _open_gencode(self.temp_gtf_path)
510
+
511
+ self.assertEqual(len(data), 1)
512
+ symbol, tx, is_principal = data[0]
513
+ self.assertEqual(symbol, 'TEST')
514
+ self.assertEqual(tx.name, 'ENST_A')
515
+ self.assertEqual(tx.strand, '-')
516
+ self.assertEqual(tx.exons, [{'start': 10, 'end': 20}, {'start': 30, 'end': 40}, {'start': 90, 'end': 100}])
517
+ self.assertEqual(tx.cds, [{'start': 15, 'end': 20}, {'start': 30, 'end': 40}])
518
+
519
+ def test__open_gencode_unquoted(self):
520
+ '''test we can parse a GTF without quoted attributes
521
+ '''
522
+ lines = '##format: gtf\n' \
523
+ 'chr1\tHAVANA\tgene\t10\t100\t.\t-\t.\tgene_name "TEST";\n' \
524
+ 'chr1\tHAVANA\ttranscript\t10\t100\t.\t-\t.\ttranscript_id ENST_A;gene_name TEST; transcript_type protein_coding; tag appris_principal_1;\n' \
525
+ 'chr1\tHAVANA\tUTR\t10\t15\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n' \
526
+ 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n' \
527
+ 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n' \
528
+ 'chr1\tHAVANA\texon\t30\t40\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n' \
529
+ 'chr1\tHAVANA\tCDS\t30\t40\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n' \
530
+ 'chr1\tHAVANA\texon\t90\t100\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n' \
531
+ 'chr1\tHAVANA\tUTR\t90\t100\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n'
507
532
 
508
533
  write_gtf(self.temp_gtf_path, lines)
509
534
  data = _open_gencode(self.temp_gtf_path)
File without changes
File without changes
File without changes
File without changes
File without changes