gencodegenes 1.1.0__tar.gz → 1.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gencodegenes-1.1.0/src/gencodegenes.egg-info → gencodegenes-1.1.1}/PKG-INFO +5 -2
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/README.md +4 -1
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/setup.py +1 -1
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencodegenes/tx.cpp +10 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1/src/gencodegenes.egg-info}/PKG-INFO +5 -2
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gtf.cpp +31 -20
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gtf.h +1 -1
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/tx.cpp +10 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/tests/test_gencode.py +58 -33
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/LICENSE.txt +0 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/MANIFEST.in +0 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/pyproject.toml +0 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/setup.cfg +0 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencode.cpp +0 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencode.h +0 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencodegenes/__init__.py +0 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencodegenes/gencode.cpp +0 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencodegenes/gencode.pyx +0 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencodegenes/transcript.cpp +0 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencodegenes/transcript.pxd +0 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencodegenes/transcript.pyx +0 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencodegenes/tx.h +0 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencodegenes.egg-info/SOURCES.txt +0 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencodegenes.egg-info/dependency_links.txt +0 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencodegenes.egg-info/requires.txt +0 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gencodegenes.egg-info/top_level.txt +0 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gzstream/gzstream.C +0 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/gzstream/gzstream.h +0 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/src/tx.h +0 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/tests/__init__.py +0 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/tests/data/example.grch38.fa +0 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/tests/data/example.grch38.gtf +0 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/tests/test_sequence_methods.py +0 -0
- {gencodegenes-1.1.0 → gencodegenes-1.1.1}/tests/test_transcript.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: gencodegenes
|
|
3
|
-
Version: 1.1.
|
|
3
|
+
Version: 1.1.1
|
|
4
4
|
Summary: Package to load genes from GENCODE GTF files
|
|
5
5
|
Home-page: https://github.com/jeremymcrae/gencodegenes
|
|
6
6
|
Author: Jeremy McRae
|
|
@@ -29,7 +29,10 @@ pip install gencodegenes
|
|
|
29
29
|
```py
|
|
30
30
|
from gencodegenes import Gencode
|
|
31
31
|
|
|
32
|
-
gencode = Gencode(GTF_PATH)
|
|
32
|
+
gencode = Gencode(GTF_PATH)
|
|
33
|
+
# full function arguments are Gencode(gtf_path, fasta_path=None, coding_only=True)
|
|
34
|
+
# - fasta_path: pass in path to fasta file to get gene transcripts with sequence
|
|
35
|
+
# - coding_only: pass in False to include all transcripts, not just protein coding
|
|
33
36
|
|
|
34
37
|
# get gene by HGNC symbol
|
|
35
38
|
gene = gencode['OR5A1']
|
|
@@ -14,7 +14,10 @@ pip install gencodegenes
|
|
|
14
14
|
```py
|
|
15
15
|
from gencodegenes import Gencode
|
|
16
16
|
|
|
17
|
-
gencode = Gencode(GTF_PATH)
|
|
17
|
+
gencode = Gencode(GTF_PATH)
|
|
18
|
+
# full function arguments are Gencode(gtf_path, fasta_path=None, coding_only=True)
|
|
19
|
+
# - fasta_path: pass in path to fasta file to get gene transcripts with sequence
|
|
20
|
+
# - coding_only: pass in False to include all transcripts, not just protein coding
|
|
18
21
|
|
|
19
22
|
# get gene by HGNC symbol
|
|
20
23
|
gene = gencode['OR5A1']
|
|
@@ -112,7 +112,7 @@ setup(name="gencodegenes",
|
|
|
112
112
|
description='Package to load genes from GENCODE GTF files',
|
|
113
113
|
long_description=io.open('README.md', encoding='utf-8').read(),
|
|
114
114
|
long_description_content_type='text/markdown',
|
|
115
|
-
version="1.1.
|
|
115
|
+
version="1.1.1",
|
|
116
116
|
author="Jeremy McRae",
|
|
117
117
|
author_email="jeremy.mcrae@gmail.com",
|
|
118
118
|
license="MIT",
|
|
@@ -66,6 +66,10 @@ void Tx::set_exons(std::vector<std::vector<int>> exon_ranges) {
|
|
|
66
66
|
void Tx::set_cds(std::vector<std::vector<int>> cds_ranges) {
|
|
67
67
|
cds.clear();
|
|
68
68
|
|
|
69
|
+
if (cds_ranges.size() == 0) {
|
|
70
|
+
return;
|
|
71
|
+
}
|
|
72
|
+
|
|
69
73
|
// If the transcript lacks exon coordinates and only has a single CDS
|
|
70
74
|
// region, then if the CDS region fits within the gene range, make a
|
|
71
75
|
// single exon, using the transcript start and end. This prevents issues
|
|
@@ -247,6 +251,9 @@ int Tx::closest_exon_num(int pos, std::vector<Region> & group) {
|
|
|
247
251
|
//
|
|
248
252
|
// @param position integer chromosome position e.g. 10000000
|
|
249
253
|
Region Tx::get_closest_exon(int pos) {
|
|
254
|
+
if (exons.size() == 0) {
|
|
255
|
+
throw std::invalid_argument("no exons assigned to this transcript");
|
|
256
|
+
}
|
|
250
257
|
int idx = closest_exon_num(pos);
|
|
251
258
|
return exons[idx];
|
|
252
259
|
}
|
|
@@ -255,6 +262,9 @@ Region Tx::get_closest_exon(int pos) {
|
|
|
255
262
|
//
|
|
256
263
|
// @param position integer chromosome position e.g. 10000000
|
|
257
264
|
bool Tx::in_coding_region(int pos) {
|
|
265
|
+
if (cds.size() == 0) {
|
|
266
|
+
return false;
|
|
267
|
+
}
|
|
258
268
|
int idx = closest_exon_num(pos, cds);
|
|
259
269
|
Region region = cds[idx];
|
|
260
270
|
return (pos >= region.start) && (pos <= region.end);
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: gencodegenes
|
|
3
|
-
Version: 1.1.
|
|
3
|
+
Version: 1.1.1
|
|
4
4
|
Summary: Package to load genes from GENCODE GTF files
|
|
5
5
|
Home-page: https://github.com/jeremymcrae/gencodegenes
|
|
6
6
|
Author: Jeremy McRae
|
|
@@ -29,7 +29,10 @@ pip install gencodegenes
|
|
|
29
29
|
```py
|
|
30
30
|
from gencodegenes import Gencode
|
|
31
31
|
|
|
32
|
-
gencode = Gencode(GTF_PATH)
|
|
32
|
+
gencode = Gencode(GTF_PATH)
|
|
33
|
+
# full function arguments are Gencode(gtf_path, fasta_path=None, coding_only=True)
|
|
34
|
+
# - fasta_path: pass in path to fasta file to get gene transcripts with sequence
|
|
35
|
+
# - coding_only: pass in False to include all transcripts, not just protein coding
|
|
33
36
|
|
|
34
37
|
# get gene by HGNC symbol
|
|
35
38
|
gene = gencode['OR5A1']
|
|
@@ -14,18 +14,25 @@
|
|
|
14
14
|
|
|
15
15
|
namespace gencode {
|
|
16
16
|
|
|
17
|
-
|
|
17
|
+
std::string trim(const std::string &s, const std::string &vals) {
|
|
18
|
+
size_t start = s.find_first_not_of(vals);
|
|
19
|
+
if (start == std::string::npos) {
|
|
20
|
+
return "";
|
|
21
|
+
}
|
|
22
|
+
size_t end = s.find_last_not_of(vals);
|
|
23
|
+
return s.substr(start, (end + 1) - start);
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
// parse the required fields from the attributes field
|
|
18
27
|
static void get_attributes_fields(GTFLine &info, std::string &line, int offset) {
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
const std::string
|
|
22
|
-
|
|
23
|
-
const std::string
|
|
24
|
-
std::string type_key = "transcript_type \"";
|
|
25
|
-
const std::string hgnc_id_key = "hgnc_id \"";
|
|
28
|
+
const std::string tx_id_key = "transcript_id ";
|
|
29
|
+
const std::string gene_id_key = "gene_id ";
|
|
30
|
+
const std::string gene_name_key = "gene_name ";
|
|
31
|
+
std::string type_key = "transcript_type ";
|
|
32
|
+
const std::string hgnc_id_key = "hgnc_id ";
|
|
26
33
|
|
|
27
34
|
size_t tx_start = line.find(tx_id_key, offset) + tx_id_key.size();
|
|
28
|
-
size_t tx_end = line.find("
|
|
35
|
+
size_t tx_end = line.find(";", tx_start);
|
|
29
36
|
|
|
30
37
|
if (tx_start - tx_id_key.size() == std::string::npos) {
|
|
31
38
|
// handle if the string was not found
|
|
@@ -34,7 +41,7 @@ static void get_attributes_fields(GTFLine &info, std::string &line, int offset)
|
|
|
34
41
|
}
|
|
35
42
|
|
|
36
43
|
size_t gene_id_start = line.find(gene_id_key, offset) + gene_id_key.size();
|
|
37
|
-
size_t gene_id_end = line.find("
|
|
44
|
+
size_t gene_id_end = line.find(";", gene_id_start);
|
|
38
45
|
if (gene_id_start - gene_id_key.size() == std::string::npos) {
|
|
39
46
|
// handle if the string was not found
|
|
40
47
|
gene_id_start = offset;
|
|
@@ -42,7 +49,7 @@ static void get_attributes_fields(GTFLine &info, std::string &line, int offset)
|
|
|
42
49
|
}
|
|
43
50
|
|
|
44
51
|
size_t gene_start = line.find(gene_name_key, offset) + gene_name_key.size();
|
|
45
|
-
size_t gene_end = line.find("
|
|
52
|
+
size_t gene_end = line.find(";", gene_start);
|
|
46
53
|
|
|
47
54
|
if (gene_start - gene_name_key.size() == std::string::npos) {
|
|
48
55
|
// handle if the string was not found
|
|
@@ -53,10 +60,10 @@ static void get_attributes_fields(GTFLine &info, std::string &line, int offset)
|
|
|
53
60
|
size_t type_start = line.find(type_key, offset) + type_key.size();
|
|
54
61
|
if (type_start - type_key.size() == std::string::npos) {
|
|
55
62
|
// allow for alternate transcript_type key, as found in non-gencode GTF files
|
|
56
|
-
type_key = "transcript_biotype
|
|
63
|
+
type_key = "transcript_biotype ";
|
|
57
64
|
type_start = line.find(type_key, offset) + type_key.size();
|
|
58
65
|
}
|
|
59
|
-
size_t type_end = line.find("
|
|
66
|
+
size_t type_end = line.find(";", type_start);
|
|
60
67
|
|
|
61
68
|
if (type_start - type_key.size() == std::string::npos) {
|
|
62
69
|
// handle if the string was not found
|
|
@@ -65,25 +72,29 @@ static void get_attributes_fields(GTFLine &info, std::string &line, int offset)
|
|
|
65
72
|
}
|
|
66
73
|
|
|
67
74
|
size_t hgnc_id_start = line.find(hgnc_id_key, offset) + hgnc_id_key.size();
|
|
68
|
-
size_t hgnc_id_end = line.find("
|
|
75
|
+
size_t hgnc_id_end = line.find(";", hgnc_id_start);
|
|
69
76
|
if (hgnc_id_start - hgnc_id_key.size() == std::string::npos) {
|
|
70
77
|
// handle if the string was not found
|
|
71
78
|
hgnc_id_start = offset;
|
|
72
79
|
hgnc_id_end = offset;
|
|
73
80
|
}
|
|
74
81
|
|
|
75
|
-
info.symbol = line.substr(gene_start, gene_end - gene_start);
|
|
76
|
-
info.tx_id = line.substr(tx_start, tx_end - tx_start);
|
|
77
|
-
info.transcript_type = line.substr(type_start, type_end - type_start);
|
|
82
|
+
info.symbol = trim(line.substr(gene_start, gene_end - gene_start), " \"");
|
|
83
|
+
info.tx_id = trim(line.substr(tx_start, tx_end - tx_start), " \"");
|
|
84
|
+
info.transcript_type = trim(line.substr(type_start, type_end - type_start), " \"");
|
|
78
85
|
|
|
79
86
|
if (gene_id_start != gene_id_end) {
|
|
80
|
-
|
|
87
|
+
std::string gene_id = trim(line.substr(gene_id_start, gene_id_end - gene_id_start), " \"");
|
|
88
|
+
if (info.symbol.size() == 0) {
|
|
89
|
+
info.symbol = gene_id;
|
|
90
|
+
} else {
|
|
91
|
+
info.alternate_ids.push_back(gene_id);
|
|
92
|
+
}
|
|
81
93
|
}
|
|
82
94
|
if (hgnc_id_start != hgnc_id_end) {
|
|
83
|
-
info.alternate_ids.push_back(line.substr(hgnc_id_start, hgnc_id_end - hgnc_id_start));
|
|
95
|
+
info.alternate_ids.push_back(trim(line.substr(hgnc_id_start, hgnc_id_end - hgnc_id_start), " \""));
|
|
84
96
|
}
|
|
85
97
|
|
|
86
|
-
info.is_canonical = 0;
|
|
87
98
|
if (info.feature == "transcript") {
|
|
88
99
|
if (line.find("appris_principal", offset) != std::string::npos) {
|
|
89
100
|
info.is_canonical = 5;
|
|
@@ -66,6 +66,10 @@ void Tx::set_exons(std::vector<std::vector<int>> exon_ranges) {
|
|
|
66
66
|
void Tx::set_cds(std::vector<std::vector<int>> cds_ranges) {
|
|
67
67
|
cds.clear();
|
|
68
68
|
|
|
69
|
+
if (cds_ranges.size() == 0) {
|
|
70
|
+
return;
|
|
71
|
+
}
|
|
72
|
+
|
|
69
73
|
// If the transcript lacks exon coordinates and only has a single CDS
|
|
70
74
|
// region, then if the CDS region fits within the gene range, make a
|
|
71
75
|
// single exon, using the transcript start and end. This prevents issues
|
|
@@ -247,6 +251,9 @@ int Tx::closest_exon_num(int pos, std::vector<Region> & group) {
|
|
|
247
251
|
//
|
|
248
252
|
// @param position integer chromosome position e.g. 10000000
|
|
249
253
|
Region Tx::get_closest_exon(int pos) {
|
|
254
|
+
if (exons.size() == 0) {
|
|
255
|
+
throw std::invalid_argument("no exons assigned to this transcript");
|
|
256
|
+
}
|
|
250
257
|
int idx = closest_exon_num(pos);
|
|
251
258
|
return exons[idx];
|
|
252
259
|
}
|
|
@@ -255,6 +262,9 @@ Region Tx::get_closest_exon(int pos) {
|
|
|
255
262
|
//
|
|
256
263
|
// @param position integer chromosome position e.g. 10000000
|
|
257
264
|
bool Tx::in_coding_region(int pos) {
|
|
265
|
+
if (cds.size() == 0) {
|
|
266
|
+
return false;
|
|
267
|
+
}
|
|
258
268
|
int idx = closest_exon_num(pos, cds);
|
|
259
269
|
Region region = cds[idx];
|
|
260
270
|
return (pos >= region.start) && (pos <= region.end);
|
|
@@ -55,12 +55,12 @@ class TestGencode(unittest.TestCase):
|
|
|
55
55
|
lines = '##format: gtf\n' \
|
|
56
56
|
'chr1\tHAVANA\tgene\t10\t20\t.\t-\t.\tgene_name "TEST1";\n' \
|
|
57
57
|
'chr1\tHAVANA\ttranscript\t10\t20\t.\t-\t.\ttranscript_id "ENST_A";gene_name "TEST1"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
|
|
58
|
-
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n' \
|
|
59
|
-
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n' \
|
|
58
|
+
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n' \
|
|
59
|
+
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n' \
|
|
60
60
|
'chr1\tHAVANA\tgene\t100\t110\t.\t-\t.\tgene_name "TEST2";\n' \
|
|
61
61
|
'chr1\tHAVANA\ttranscript\t100\t110\t.\t-\t.\ttranscript_id "ENST_B";gene_name "TEST2"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
|
|
62
|
-
'chr1\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST2"; transcript_type "protein_coding"\n' \
|
|
63
|
-
'chr1\tHAVANA\tCDS\t105\t110\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST2"; transcript_type "protein_coding"\n'\
|
|
62
|
+
'chr1\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding;"\n' \
|
|
63
|
+
'chr1\tHAVANA\tCDS\t105\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding;"\n'\
|
|
64
64
|
'chr2\tHAVANA\tgene\t100\t110\t.\t-\t.\tgene_name "TEST3";\n' \
|
|
65
65
|
'chr2\tHAVANA\ttranscript\t100\t110\t.\t-\t.\ttranscript_id "ENST_C";gene_name "TEST3"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
|
|
66
66
|
'chr2\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_C" gene_name "TEST3"; transcript_type "protein_coding"\n' \
|
|
@@ -112,16 +112,16 @@ class TestGencode(unittest.TestCase):
|
|
|
112
112
|
lines = ['##format: gtf\n',
|
|
113
113
|
'chr1\tHAVANA\tgene\t10\t20\t.\t-\t.\tgene_name "TEST1";\n',
|
|
114
114
|
'chr1\tHAVANA\ttranscript\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding"; tag "appris_principal_1";\n',
|
|
115
|
-
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n',
|
|
116
|
-
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n',
|
|
115
|
+
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n',
|
|
116
|
+
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n',
|
|
117
117
|
'chr1\tHAVANA\tgene\t100\t110\t.\t-\t.\tgene_name "TEST2";\n',
|
|
118
118
|
'chr1\tHAVANA\ttranscript\t100\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding"; tag "appris_principal_1";\n',
|
|
119
|
-
'chr1\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST2"; transcript_type "protein_coding"\n',
|
|
120
|
-
'chr1\tHAVANA\tCDS\t105\t110\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST2"; transcript_type "protein_coding"\n',
|
|
119
|
+
'chr1\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding;"\n',
|
|
120
|
+
'chr1\tHAVANA\tCDS\t105\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding;"\n',
|
|
121
121
|
'chr2\tHAVANA\tgene\t100\t110\t.\t-\t.\tgene_name "TEST3";\n',
|
|
122
122
|
'chr2\tHAVANA\ttranscript\t100\t110\t.\t-\t.\ttranscript_id "ENST_C"; gene_name "TEST3"; transcript_type "protein_coding"; tag "appris_principal_1";\n',
|
|
123
|
-
'chr2\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_C" gene_name "TEST3"; transcript_type "protein_coding"\n',
|
|
124
|
-
'chr2\tHAVANA\tCDS\t105\t110\t.\t-\t.\ttranscript_id "ENST_C" gene_name "TEST3"; transcript_type "protein_coding"\n']
|
|
123
|
+
'chr2\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_C"; gene_name "TEST3"; transcript_type "protein_coding;"\n',
|
|
124
|
+
'chr2\tHAVANA\tCDS\t105\t110\t.\t-\t.\ttranscript_id "ENST_C"; gene_name "TEST3"; transcript_type "protein_coding;"\n']
|
|
125
125
|
|
|
126
126
|
write_gtf(self.temp_gtf_path, lines)
|
|
127
127
|
make_fasta(self.temp_fasta_path, ['chr1', 'chr2'])
|
|
@@ -150,12 +150,12 @@ class TestGencode(unittest.TestCase):
|
|
|
150
150
|
lines = ['##format: gtf\n',
|
|
151
151
|
'chr1\tHAVANA\tgene\t20\t30\t.\t-\t.\tgene_name "TEST1";\n',
|
|
152
152
|
'chr1\tHAVANA\ttranscript\t20\t30\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding"; tag "appris_principal_1";\n',
|
|
153
|
-
'chr1\tHAVANA\texon\t20\t30\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n',
|
|
154
|
-
'chr1\tHAVANA\tCDS\t25\t30\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n',
|
|
153
|
+
'chr1\tHAVANA\texon\t20\t30\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n',
|
|
154
|
+
'chr1\tHAVANA\tCDS\t25\t30\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n',
|
|
155
155
|
'chr1\tHAVANA\tgene\t100\t110\t.\t-\t.\tgene_name "TEST1";\n',
|
|
156
156
|
'chr1\tHAVANA\ttranscript\t100\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST1"; transcript_type "protein_coding";\n',
|
|
157
|
-
'chr1\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST1"; transcript_type "protein_coding"\n',
|
|
158
|
-
'chr1\tHAVANA\tCDS\t110\t100\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST1"; transcript_type "protein_coding"\n',
|
|
157
|
+
'chr1\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST1"; transcript_type "protein_coding;"\n',
|
|
158
|
+
'chr1\tHAVANA\tCDS\t110\t100\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST1"; transcript_type "protein_coding;"\n',
|
|
159
159
|
]
|
|
160
160
|
|
|
161
161
|
write_gtf(self.temp_gtf_path, lines)
|
|
@@ -231,7 +231,7 @@ class TestGencode(unittest.TestCase):
|
|
|
231
231
|
'gene_status "KNOWN"; gene_name "OR4F5"; transcript_type "protein_coding"; ' \
|
|
232
232
|
'transcript_status "KNOWN"; transcript_name "OR4F5-001"; level 2; ' \
|
|
233
233
|
'protein_id "ENSP00000334393.3"; tag "basic"; transcript_support_level "NA"; ' \
|
|
234
|
-
'hgnc_id "HGNC:14825"
|
|
234
|
+
'hgnc_id "HGNC:14825"; tag "appris_principal_1"; tag "CCDS"; ccdsid "CCDS30547.1"; ' \
|
|
235
235
|
'havana_gene "OTTHUMG00000001094.2"; havana_transcript "OTTHUMT00000003223.2";\n'
|
|
236
236
|
obj = _parse_gtfline(line.encode('utf8'))
|
|
237
237
|
expected = {'chrom': b'chr1',
|
|
@@ -251,7 +251,7 @@ class TestGencode(unittest.TestCase):
|
|
|
251
251
|
line = 'chr1\tHAVANA\ttranscript\t69091\t70008\t.\t+\t.\t '\
|
|
252
252
|
'transcript_id "ENST00000335137.3"; gene_type "protein_coding"; ' \
|
|
253
253
|
'gene_status "KNOWN"; gene_name "OR4F5"; transcript_type "protein_coding";' \
|
|
254
|
-
'hgnc_id "HGNC:14825"
|
|
254
|
+
'hgnc_id "HGNC:14825"; tag "appris_principal_1"; tag "CCDS"; ccdsid "CCDS30547.1"; '
|
|
255
255
|
obj = _parse_gtfline(line.encode('utf8'))
|
|
256
256
|
expected = {'chrom': b'chr1',
|
|
257
257
|
'feature': b'transcript',
|
|
@@ -431,12 +431,12 @@ class TestGencode(unittest.TestCase):
|
|
|
431
431
|
lines = '##format: gtf\n' \
|
|
432
432
|
'chr1\tHAVANA\tgene\t10\t20\t.\t-\t.\tgene_name "TEST1";\n' \
|
|
433
433
|
'chr1\tHAVANA\ttranscript\t10\t20\t.\t-\t.\ttranscript_id "ENST_A";gene_name "TEST1"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
|
|
434
|
-
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n' \
|
|
435
|
-
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n' \
|
|
434
|
+
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n' \
|
|
435
|
+
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n' \
|
|
436
436
|
'chr1\tHAVANA\tgene\t10\t30\t.\t-\t.\tgene_name "TEST2";\n' \
|
|
437
437
|
'chr1\tHAVANA\ttranscript\t10\t30\t.\t-\t.\ttranscript_id "ENST_B";gene_name "TEST2"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
|
|
438
|
-
'chr1\tHAVANA\texon\t10\t30\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST2"; transcript_type "protein_coding"\n' \
|
|
439
|
-
'chr1\tHAVANA\tCDS\t15\t30\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST2"; transcript_type "protein_coding"\n'
|
|
438
|
+
'chr1\tHAVANA\texon\t10\t30\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding;"\n' \
|
|
439
|
+
'chr1\tHAVANA\tCDS\t15\t30\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding;"\n'
|
|
440
440
|
|
|
441
441
|
write_gtf(self.temp_gtf_path, lines)
|
|
442
442
|
data = _open_gencode(self.temp_gtf_path)
|
|
@@ -452,8 +452,8 @@ class TestGencode(unittest.TestCase):
|
|
|
452
452
|
lines = '##format: gtf\n' \
|
|
453
453
|
'chr1\tHAVANA\tgene\t10\t20\t.\t-\t.\tgene_name "TEST";\n' \
|
|
454
454
|
'chr1\tHAVANA\ttranscript\t10\t20\t.\t-\t.\ttranscript_id "ENST_A";gene_name "TEST"; transcript_type "processed_transcript"; tag "appris_principal_1";\n' \
|
|
455
|
-
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "processed_transcript"\n' \
|
|
456
|
-
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "processed_transcript"\n' \
|
|
455
|
+
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "processed_transcript;"\n' \
|
|
456
|
+
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "processed_transcript;"\n' \
|
|
457
457
|
|
|
458
458
|
write_gtf(self.temp_gtf_path, lines)
|
|
459
459
|
data = _open_gencode(self.temp_gtf_path)
|
|
@@ -472,12 +472,12 @@ class TestGencode(unittest.TestCase):
|
|
|
472
472
|
lines = '##format: gtf\n' \
|
|
473
473
|
'chr1\tHAVANA\tgene\t10\t20\t.\t-\t.\tgene_name "TEST";\n' \
|
|
474
474
|
'chr1\tHAVANA\ttranscript\t10\t20\t.\t-\t.\ttranscript_id "ENST_A";gene_name "TEST"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
|
|
475
|
-
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
|
|
476
|
-
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
|
|
475
|
+
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
|
|
476
|
+
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
|
|
477
477
|
'chr1\tHAVANA\tgene\t10\t30\t.\t-\t.\tgene_name "TEST";\n' \
|
|
478
478
|
'chr1\tHAVANA\ttranscript\t10\t30\t.\t-\t.\ttranscript_id "ENST_B";gene_name "TEST"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
|
|
479
|
-
'chr1\tHAVANA\texon\t10\t30\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST"; transcript_type "protein_coding"\n' \
|
|
480
|
-
'chr1\tHAVANA\tCDS\t15\t30\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST"; transcript_type "protein_coding"\n'
|
|
479
|
+
'chr1\tHAVANA\texon\t10\t30\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
|
|
480
|
+
'chr1\tHAVANA\tCDS\t15\t30\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST"; transcript_type "protein_coding;"\n'
|
|
481
481
|
|
|
482
482
|
write_gtf(self.temp_gtf_path, lines)
|
|
483
483
|
data = _open_gencode(self.temp_gtf_path)
|
|
@@ -497,13 +497,38 @@ class TestGencode(unittest.TestCase):
|
|
|
497
497
|
lines = '##format: gtf\n' \
|
|
498
498
|
'chr1\tHAVANA\tgene\t10\t100\t.\t-\t.\tgene_name "TEST";\n' \
|
|
499
499
|
'chr1\tHAVANA\ttranscript\t10\t100\t.\t-\t.\ttranscript_id "ENST_A";gene_name "TEST"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
|
|
500
|
-
'chr1\tHAVANA\tUTR\t10\t15\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding";\n' \
|
|
501
|
-
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
|
|
502
|
-
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
|
|
503
|
-
'chr1\tHAVANA\texon\t30\t40\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
|
|
504
|
-
'chr1\tHAVANA\tCDS\t30\t40\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
|
|
505
|
-
'chr1\tHAVANA\texon\t90\t100\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
|
|
506
|
-
'chr1\tHAVANA\tUTR\t90\t100\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n'
|
|
500
|
+
'chr1\tHAVANA\tUTR\t10\t15\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding";\n' \
|
|
501
|
+
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
|
|
502
|
+
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
|
|
503
|
+
'chr1\tHAVANA\texon\t30\t40\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
|
|
504
|
+
'chr1\tHAVANA\tCDS\t30\t40\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
|
|
505
|
+
'chr1\tHAVANA\texon\t90\t100\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
|
|
506
|
+
'chr1\tHAVANA\tUTR\t90\t100\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n'
|
|
507
|
+
|
|
508
|
+
write_gtf(self.temp_gtf_path, lines)
|
|
509
|
+
data = _open_gencode(self.temp_gtf_path)
|
|
510
|
+
|
|
511
|
+
self.assertEqual(len(data), 1)
|
|
512
|
+
symbol, tx, is_principal = data[0]
|
|
513
|
+
self.assertEqual(symbol, 'TEST')
|
|
514
|
+
self.assertEqual(tx.name, 'ENST_A')
|
|
515
|
+
self.assertEqual(tx.strand, '-')
|
|
516
|
+
self.assertEqual(tx.exons, [{'start': 10, 'end': 20}, {'start': 30, 'end': 40}, {'start': 90, 'end': 100}])
|
|
517
|
+
self.assertEqual(tx.cds, [{'start': 15, 'end': 20}, {'start': 30, 'end': 40}])
|
|
518
|
+
|
|
519
|
+
def test__open_gencode_unquoted(self):
|
|
520
|
+
'''test we can parse a GTF without quoted attributes
|
|
521
|
+
'''
|
|
522
|
+
lines = '##format: gtf\n' \
|
|
523
|
+
'chr1\tHAVANA\tgene\t10\t100\t.\t-\t.\tgene_name "TEST";\n' \
|
|
524
|
+
'chr1\tHAVANA\ttranscript\t10\t100\t.\t-\t.\ttranscript_id ENST_A;gene_name TEST; transcript_type protein_coding; tag appris_principal_1;\n' \
|
|
525
|
+
'chr1\tHAVANA\tUTR\t10\t15\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n' \
|
|
526
|
+
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n' \
|
|
527
|
+
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n' \
|
|
528
|
+
'chr1\tHAVANA\texon\t30\t40\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n' \
|
|
529
|
+
'chr1\tHAVANA\tCDS\t30\t40\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n' \
|
|
530
|
+
'chr1\tHAVANA\texon\t90\t100\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n' \
|
|
531
|
+
'chr1\tHAVANA\tUTR\t90\t100\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n'
|
|
507
532
|
|
|
508
533
|
write_gtf(self.temp_gtf_path, lines)
|
|
509
534
|
data = _open_gencode(self.temp_gtf_path)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|