gencodegenes 1.0.10__tar.gz → 1.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gencodegenes-1.0.10/src/gencodegenes.egg-info → gencodegenes-1.1.1}/PKG-INFO +5 -2
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/README.md +4 -1
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/setup.py +1 -1
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencodegenes/gencode.cpp +55 -41
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencodegenes/gencode.pyx +1 -1
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencodegenes/tx.cpp +10 -0
- {gencodegenes-1.0.10 → gencodegenes-1.1.1/src/gencodegenes.egg-info}/PKG-INFO +5 -2
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gtf.cpp +49 -28
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gtf.h +1 -1
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/tx.cpp +10 -0
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/tests/test_gencode.py +58 -33
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/LICENSE.txt +0 -0
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/MANIFEST.in +0 -0
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/pyproject.toml +0 -0
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/setup.cfg +0 -0
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencode.cpp +0 -0
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencode.h +0 -0
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencodegenes/__init__.py +0 -0
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencodegenes/transcript.cpp +0 -0
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencodegenes/transcript.pxd +0 -0
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencodegenes/transcript.pyx +0 -0
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencodegenes/tx.h +0 -0
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencodegenes.egg-info/SOURCES.txt +0 -0
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencodegenes.egg-info/dependency_links.txt +0 -0
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencodegenes.egg-info/requires.txt +0 -0
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencodegenes.egg-info/top_level.txt +0 -0
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gzstream/gzstream.C +0 -0
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gzstream/gzstream.h +0 -0
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/tx.h +0 -0
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/tests/__init__.py +0 -0
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/tests/data/example.grch38.fa +0 -0
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/tests/data/example.grch38.gtf +0 -0
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/tests/test_sequence_methods.py +0 -0
- {gencodegenes-1.0.10 → gencodegenes-1.1.1}/tests/test_transcript.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: gencodegenes
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.1.1
|
|
4
4
|
Summary: Package to load genes from GENCODE GTF files
|
|
5
5
|
Home-page: https://github.com/jeremymcrae/gencodegenes
|
|
6
6
|
Author: Jeremy McRae
|
|
@@ -29,7 +29,10 @@ pip install gencodegenes
|
|
|
29
29
|
```py
|
|
30
30
|
from gencodegenes import Gencode
|
|
31
31
|
|
|
32
|
-
gencode = Gencode(GTF_PATH)
|
|
32
|
+
gencode = Gencode(GTF_PATH)
|
|
33
|
+
# full function arguments are Gencode(gtf_path, fasta_path=None, coding_only=True)
|
|
34
|
+
# - fasta_path: pass in path to fasta file to get gene transcripts with sequence
|
|
35
|
+
# - coding_only: pass in False to include all transcripts, not just protein coding
|
|
33
36
|
|
|
34
37
|
# get gene by HGNC symbol
|
|
35
38
|
gene = gencode['OR5A1']
|
|
@@ -14,7 +14,10 @@ pip install gencodegenes
|
|
|
14
14
|
```py
|
|
15
15
|
from gencodegenes import Gencode
|
|
16
16
|
|
|
17
|
-
gencode = Gencode(GTF_PATH)
|
|
17
|
+
gencode = Gencode(GTF_PATH)
|
|
18
|
+
# full function arguments are Gencode(gtf_path, fasta_path=None, coding_only=True)
|
|
19
|
+
# - fasta_path: pass in path to fasta file to get gene transcripts with sequence
|
|
20
|
+
# - coding_only: pass in False to include all transcripts, not just protein coding
|
|
18
21
|
|
|
19
22
|
# get gene by HGNC symbol
|
|
20
23
|
gene = gencode['OR5A1']
|
|
@@ -112,7 +112,7 @@ setup(name="gencodegenes",
|
|
|
112
112
|
description='Package to load genes from GENCODE GTF files',
|
|
113
113
|
long_description=io.open('README.md', encoding='utf-8').read(),
|
|
114
114
|
long_description_content_type='text/markdown',
|
|
115
|
-
version="1.
|
|
115
|
+
version="1.1.1",
|
|
116
116
|
author="Jeremy McRae",
|
|
117
117
|
author_email="jeremy.mcrae@gmail.com",
|
|
118
118
|
license="MIT",
|
|
@@ -5755,11 +5755,12 @@ static PyObject *__pyx_f_12gencodegenes_7gencode__open_gencode(PyObject *__pyx_v
|
|
|
5755
5755
|
int __pyx_t_4;
|
|
5756
5756
|
std::string __pyx_t_5;
|
|
5757
5757
|
bool __pyx_t_6;
|
|
5758
|
-
std::vector<struct gencode::NamedTx>
|
|
5759
|
-
struct gencode::NamedTx __pyx_t_8;
|
|
5760
|
-
gencode::
|
|
5761
|
-
|
|
5762
|
-
|
|
5758
|
+
std::vector<struct gencode::NamedTx> __pyx_t_7;
|
|
5759
|
+
std::vector<struct gencode::NamedTx> ::iterator __pyx_t_8;
|
|
5760
|
+
struct gencode::NamedTx __pyx_t_9;
|
|
5761
|
+
gencode::Tx __pyx_t_10;
|
|
5762
|
+
PyObject *__pyx_t_11 = NULL;
|
|
5763
|
+
int __pyx_t_12;
|
|
5763
5764
|
int __pyx_lineno = 0;
|
|
5764
5765
|
const char *__pyx_filename = NULL;
|
|
5765
5766
|
int __pyx_clineno = 0;
|
|
@@ -5804,7 +5805,13 @@ static PyObject *__pyx_f_12gencodegenes_7gencode__open_gencode(PyObject *__pyx_v
|
|
|
5804
5805
|
__pyx_t_5 = __pyx_convert_string_from_py_6libcpp_6string_std__in_string(__pyx_t_1); if (unlikely(PyErr_Occurred())) __PYX_ERR(0, 66, __pyx_L1_error)
|
|
5805
5806
|
__Pyx_DECREF(__pyx_t_1); __pyx_t_1 = 0;
|
|
5806
5807
|
__pyx_t_6 = __Pyx_PyObject_IsTrue(__pyx_v_coding_only); if (unlikely((__pyx_t_6 == ((bool)-1)) && PyErr_Occurred())) __PYX_ERR(0, 66, __pyx_L1_error)
|
|
5807
|
-
|
|
5808
|
+
try {
|
|
5809
|
+
__pyx_t_7 = gencode::open_gencode(__PYX_STD_MOVE_IF_SUPPORTED(__pyx_t_5), __pyx_t_6);
|
|
5810
|
+
} catch(...) {
|
|
5811
|
+
__Pyx_CppExn2PyErr();
|
|
5812
|
+
__PYX_ERR(0, 66, __pyx_L1_error)
|
|
5813
|
+
}
|
|
5814
|
+
__pyx_v__transcripts = __PYX_STD_MOVE_IF_SUPPORTED(__pyx_t_7);
|
|
5808
5815
|
|
|
5809
5816
|
/* "gencodegenes/gencode.pyx":68
|
|
5810
5817
|
* cdef vector[NamedTx] _transcripts = open_gencode(gtf_path.encode('utf8'), coding_only)
|
|
@@ -5825,12 +5832,12 @@ static PyObject *__pyx_f_12gencodegenes_7gencode__open_gencode(PyObject *__pyx_v
|
|
|
5825
5832
|
* tx = x.tx
|
|
5826
5833
|
* chrom = tx.get_chrom().decode('utf8')
|
|
5827
5834
|
*/
|
|
5828
|
-
|
|
5835
|
+
__pyx_t_8 = __pyx_v__transcripts.begin();
|
|
5829
5836
|
for (;;) {
|
|
5830
|
-
if (!(
|
|
5831
|
-
|
|
5832
|
-
++
|
|
5833
|
-
__pyx_v_x =
|
|
5837
|
+
if (!(__pyx_t_8 != __pyx_v__transcripts.end())) break;
|
|
5838
|
+
__pyx_t_9 = *__pyx_t_8;
|
|
5839
|
+
++__pyx_t_8;
|
|
5840
|
+
__pyx_v_x = __pyx_t_9;
|
|
5834
5841
|
|
|
5835
5842
|
/* "gencodegenes/gencode.pyx":70
|
|
5836
5843
|
* transcripts = []
|
|
@@ -5839,8 +5846,8 @@ static PyObject *__pyx_f_12gencodegenes_7gencode__open_gencode(PyObject *__pyx_v
|
|
|
5839
5846
|
* chrom = tx.get_chrom().decode('utf8')
|
|
5840
5847
|
* start = tx.get_start()
|
|
5841
5848
|
*/
|
|
5842
|
-
|
|
5843
|
-
__pyx_v_tx = __PYX_STD_MOVE_IF_SUPPORTED(
|
|
5849
|
+
__pyx_t_10 = __pyx_v_x.tx;
|
|
5850
|
+
__pyx_v_tx = __PYX_STD_MOVE_IF_SUPPORTED(__pyx_t_10);
|
|
5844
5851
|
|
|
5845
5852
|
/* "gencodegenes/gencode.pyx":71
|
|
5846
5853
|
* for x in _transcripts:
|
|
@@ -5948,39 +5955,39 @@ static PyObject *__pyx_f_12gencodegenes_7gencode__open_gencode(PyObject *__pyx_v
|
|
|
5948
5955
|
__Pyx_GOTREF(__pyx_t_1);
|
|
5949
5956
|
__pyx_t_3 = __Pyx_PyInt_From_int(__pyx_v_end); if (unlikely(!__pyx_t_3)) __PYX_ERR(0, 79, __pyx_L1_error)
|
|
5950
5957
|
__Pyx_GOTREF(__pyx_t_3);
|
|
5951
|
-
|
|
5952
|
-
__Pyx_GOTREF(
|
|
5958
|
+
__pyx_t_11 = PyTuple_New(8); if (unlikely(!__pyx_t_11)) __PYX_ERR(0, 79, __pyx_L1_error)
|
|
5959
|
+
__Pyx_GOTREF(__pyx_t_11);
|
|
5953
5960
|
__Pyx_INCREF(__pyx_v_tx_id);
|
|
5954
5961
|
__Pyx_GIVEREF(__pyx_v_tx_id);
|
|
5955
|
-
if (__Pyx_PyTuple_SET_ITEM(
|
|
5962
|
+
if (__Pyx_PyTuple_SET_ITEM(__pyx_t_11, 0, __pyx_v_tx_id)) __PYX_ERR(0, 79, __pyx_L1_error);
|
|
5956
5963
|
__Pyx_INCREF(__pyx_v_chrom);
|
|
5957
5964
|
__Pyx_GIVEREF(__pyx_v_chrom);
|
|
5958
|
-
if (__Pyx_PyTuple_SET_ITEM(
|
|
5965
|
+
if (__Pyx_PyTuple_SET_ITEM(__pyx_t_11, 1, __pyx_v_chrom)) __PYX_ERR(0, 79, __pyx_L1_error);
|
|
5959
5966
|
__Pyx_GIVEREF(__pyx_t_1);
|
|
5960
|
-
if (__Pyx_PyTuple_SET_ITEM(
|
|
5967
|
+
if (__Pyx_PyTuple_SET_ITEM(__pyx_t_11, 2, __pyx_t_1)) __PYX_ERR(0, 79, __pyx_L1_error);
|
|
5961
5968
|
__Pyx_GIVEREF(__pyx_t_3);
|
|
5962
|
-
if (__Pyx_PyTuple_SET_ITEM(
|
|
5969
|
+
if (__Pyx_PyTuple_SET_ITEM(__pyx_t_11, 3, __pyx_t_3)) __PYX_ERR(0, 79, __pyx_L1_error);
|
|
5963
5970
|
__Pyx_INCREF(__pyx_v_strand);
|
|
5964
5971
|
__Pyx_GIVEREF(__pyx_v_strand);
|
|
5965
|
-
if (__Pyx_PyTuple_SET_ITEM(
|
|
5972
|
+
if (__Pyx_PyTuple_SET_ITEM(__pyx_t_11, 4, __pyx_v_strand)) __PYX_ERR(0, 79, __pyx_L1_error);
|
|
5966
5973
|
__Pyx_INCREF(__pyx_v_transcript_type);
|
|
5967
5974
|
__Pyx_GIVEREF(__pyx_v_transcript_type);
|
|
5968
|
-
if (__Pyx_PyTuple_SET_ITEM(
|
|
5975
|
+
if (__Pyx_PyTuple_SET_ITEM(__pyx_t_11, 5, __pyx_v_transcript_type)) __PYX_ERR(0, 79, __pyx_L1_error);
|
|
5969
5976
|
__Pyx_INCREF(__pyx_v_exons);
|
|
5970
5977
|
__Pyx_GIVEREF(__pyx_v_exons);
|
|
5971
|
-
if (__Pyx_PyTuple_SET_ITEM(
|
|
5978
|
+
if (__Pyx_PyTuple_SET_ITEM(__pyx_t_11, 6, __pyx_v_exons)) __PYX_ERR(0, 79, __pyx_L1_error);
|
|
5972
5979
|
__Pyx_INCREF(__pyx_v_cds);
|
|
5973
5980
|
__Pyx_GIVEREF(__pyx_v_cds);
|
|
5974
|
-
if (__Pyx_PyTuple_SET_ITEM(
|
|
5981
|
+
if (__Pyx_PyTuple_SET_ITEM(__pyx_t_11, 7, __pyx_v_cds)) __PYX_ERR(0, 79, __pyx_L1_error);
|
|
5975
5982
|
__pyx_t_1 = 0;
|
|
5976
5983
|
__pyx_t_3 = 0;
|
|
5977
5984
|
__pyx_t_3 = __Pyx_PyDict_NewPresized(1); if (unlikely(!__pyx_t_3)) __PYX_ERR(0, 79, __pyx_L1_error)
|
|
5978
5985
|
__Pyx_GOTREF(__pyx_t_3);
|
|
5979
5986
|
if (PyDict_SetItem(__pyx_t_3, __pyx_n_s_offset, __pyx_int_0) < 0) __PYX_ERR(0, 79, __pyx_L1_error)
|
|
5980
|
-
__pyx_t_1 = __Pyx_PyObject_Call(__pyx_t_2,
|
|
5987
|
+
__pyx_t_1 = __Pyx_PyObject_Call(__pyx_t_2, __pyx_t_11, __pyx_t_3); if (unlikely(!__pyx_t_1)) __PYX_ERR(0, 79, __pyx_L1_error)
|
|
5981
5988
|
__Pyx_GOTREF(__pyx_t_1);
|
|
5982
5989
|
__Pyx_DECREF(__pyx_t_2); __pyx_t_2 = 0;
|
|
5983
|
-
__Pyx_DECREF(
|
|
5990
|
+
__Pyx_DECREF(__pyx_t_11); __pyx_t_11 = 0;
|
|
5984
5991
|
__Pyx_DECREF(__pyx_t_3); __pyx_t_3 = 0;
|
|
5985
5992
|
__Pyx_XDECREF_SET(__pyx_v_transcript, __pyx_t_1);
|
|
5986
5993
|
__pyx_t_1 = 0;
|
|
@@ -5996,19 +6003,19 @@ static PyObject *__pyx_f_12gencodegenes_7gencode__open_gencode(PyObject *__pyx_v
|
|
|
5996
6003
|
__Pyx_GOTREF(__pyx_t_1);
|
|
5997
6004
|
__pyx_t_3 = __Pyx_PyInt_From_int(__pyx_v_x.is_canonical); if (unlikely(!__pyx_t_3)) __PYX_ERR(0, 80, __pyx_L1_error)
|
|
5998
6005
|
__Pyx_GOTREF(__pyx_t_3);
|
|
5999
|
-
|
|
6000
|
-
__Pyx_GOTREF(
|
|
6006
|
+
__pyx_t_11 = PyTuple_New(3); if (unlikely(!__pyx_t_11)) __PYX_ERR(0, 80, __pyx_L1_error)
|
|
6007
|
+
__Pyx_GOTREF(__pyx_t_11);
|
|
6001
6008
|
__Pyx_GIVEREF(__pyx_t_1);
|
|
6002
|
-
if (__Pyx_PyTuple_SET_ITEM(
|
|
6009
|
+
if (__Pyx_PyTuple_SET_ITEM(__pyx_t_11, 0, __pyx_t_1)) __PYX_ERR(0, 80, __pyx_L1_error);
|
|
6003
6010
|
__Pyx_INCREF(__pyx_v_transcript);
|
|
6004
6011
|
__Pyx_GIVEREF(__pyx_v_transcript);
|
|
6005
|
-
if (__Pyx_PyTuple_SET_ITEM(
|
|
6012
|
+
if (__Pyx_PyTuple_SET_ITEM(__pyx_t_11, 1, __pyx_v_transcript)) __PYX_ERR(0, 80, __pyx_L1_error);
|
|
6006
6013
|
__Pyx_GIVEREF(__pyx_t_3);
|
|
6007
|
-
if (__Pyx_PyTuple_SET_ITEM(
|
|
6014
|
+
if (__Pyx_PyTuple_SET_ITEM(__pyx_t_11, 2, __pyx_t_3)) __PYX_ERR(0, 80, __pyx_L1_error);
|
|
6008
6015
|
__pyx_t_1 = 0;
|
|
6009
6016
|
__pyx_t_3 = 0;
|
|
6010
|
-
|
|
6011
|
-
__Pyx_DECREF(
|
|
6017
|
+
__pyx_t_12 = __Pyx_PyList_Append(__pyx_v_transcripts, __pyx_t_11); if (unlikely(__pyx_t_12 == ((int)-1))) __PYX_ERR(0, 80, __pyx_L1_error)
|
|
6018
|
+
__Pyx_DECREF(__pyx_t_11); __pyx_t_11 = 0;
|
|
6012
6019
|
|
|
6013
6020
|
/* "gencodegenes/gencode.pyx":69
|
|
6014
6021
|
*
|
|
@@ -6044,7 +6051,7 @@ static PyObject *__pyx_f_12gencodegenes_7gencode__open_gencode(PyObject *__pyx_v
|
|
|
6044
6051
|
__Pyx_XDECREF(__pyx_t_1);
|
|
6045
6052
|
__Pyx_XDECREF(__pyx_t_2);
|
|
6046
6053
|
__Pyx_XDECREF(__pyx_t_3);
|
|
6047
|
-
__Pyx_XDECREF(
|
|
6054
|
+
__Pyx_XDECREF(__pyx_t_11);
|
|
6048
6055
|
__Pyx_AddTraceback("gencodegenes.gencode._open_gencode", __pyx_clineno, __pyx_lineno, __pyx_filename);
|
|
6049
6056
|
__pyx_r = 0;
|
|
6050
6057
|
__pyx_L0:;
|
|
@@ -11146,8 +11153,9 @@ static int __pyx_pf_12gencodegenes_7gencode_7Gencode___cinit__(struct __pyx_obj_
|
|
|
11146
11153
|
int __pyx_t_8;
|
|
11147
11154
|
std::string __pyx_t_9;
|
|
11148
11155
|
bool __pyx_t_10;
|
|
11149
|
-
std::vector<struct gencode::NamedTx>
|
|
11150
|
-
struct gencode::NamedTx __pyx_t_12;
|
|
11156
|
+
std::vector<struct gencode::NamedTx> __pyx_t_11;
|
|
11157
|
+
std::vector<struct gencode::NamedTx> ::iterator __pyx_t_12;
|
|
11158
|
+
struct gencode::NamedTx __pyx_t_13;
|
|
11151
11159
|
int __pyx_lineno = 0;
|
|
11152
11160
|
const char *__pyx_filename = NULL;
|
|
11153
11161
|
int __pyx_clineno = 0;
|
|
@@ -11549,7 +11557,13 @@ static int __pyx_pf_12gencodegenes_7gencode_7Gencode___cinit__(struct __pyx_obj_
|
|
|
11549
11557
|
__pyx_t_9 = __pyx_convert_string_from_py_6libcpp_6string_std__in_string(__pyx_t_3); if (unlikely(PyErr_Occurred())) __PYX_ERR(0, 349, __pyx_L1_error)
|
|
11550
11558
|
__Pyx_DECREF(__pyx_t_3); __pyx_t_3 = 0;
|
|
11551
11559
|
__pyx_t_10 = __Pyx_PyObject_IsTrue(__pyx_v_coding_only); if (unlikely((__pyx_t_10 == ((bool)-1)) && PyErr_Occurred())) __PYX_ERR(0, 349, __pyx_L1_error)
|
|
11552
|
-
|
|
11560
|
+
try {
|
|
11561
|
+
__pyx_t_11 = gencode::open_gencode(__PYX_STD_MOVE_IF_SUPPORTED(__pyx_t_9), __pyx_t_10);
|
|
11562
|
+
} catch(...) {
|
|
11563
|
+
__Pyx_CppExn2PyErr();
|
|
11564
|
+
__PYX_ERR(0, 349, __pyx_L1_error)
|
|
11565
|
+
}
|
|
11566
|
+
__pyx_v_transcripts = __PYX_STD_MOVE_IF_SUPPORTED(__pyx_t_11);
|
|
11553
11567
|
|
|
11554
11568
|
/* "gencodegenes/gencode.pyx":350
|
|
11555
11569
|
* if gencode is not None:
|
|
@@ -11558,12 +11572,12 @@ static int __pyx_pf_12gencodegenes_7gencode_7Gencode___cinit__(struct __pyx_obj_
|
|
|
11558
11572
|
* symbol = x.symbol.decode('utf8')
|
|
11559
11573
|
* if symbol not in self.genes:
|
|
11560
11574
|
*/
|
|
11561
|
-
|
|
11575
|
+
__pyx_t_12 = __pyx_v_transcripts.begin();
|
|
11562
11576
|
for (;;) {
|
|
11563
|
-
if (!(
|
|
11564
|
-
|
|
11565
|
-
++
|
|
11566
|
-
__pyx_v_x =
|
|
11577
|
+
if (!(__pyx_t_12 != __pyx_v_transcripts.end())) break;
|
|
11578
|
+
__pyx_t_13 = *__pyx_t_12;
|
|
11579
|
+
++__pyx_t_12;
|
|
11580
|
+
__pyx_v_x = __pyx_t_13;
|
|
11567
11581
|
|
|
11568
11582
|
/* "gencodegenes/gencode.pyx":351
|
|
11569
11583
|
* transcripts = open_gencode(str(gencode).encode('utf8'), coding_only)
|
|
@@ -41,7 +41,7 @@ cdef extern from "gencode.h" namespace "gencode":
|
|
|
41
41
|
int pos
|
|
42
42
|
string symbol
|
|
43
43
|
|
|
44
|
-
vector[NamedTx] open_gencode(string, bool)
|
|
44
|
+
vector[NamedTx] open_gencode(string, bool) except +
|
|
45
45
|
bool CompFunc(const GenePoint &l, const GenePoint &r)
|
|
46
46
|
vector[string] _in_region(string chrom, int start, int end,
|
|
47
47
|
map[string, vector[GenePoint]] & starts, map[string, vector[GenePoint]] & ends,
|
|
@@ -66,6 +66,10 @@ void Tx::set_exons(std::vector<std::vector<int>> exon_ranges) {
|
|
|
66
66
|
void Tx::set_cds(std::vector<std::vector<int>> cds_ranges) {
|
|
67
67
|
cds.clear();
|
|
68
68
|
|
|
69
|
+
if (cds_ranges.size() == 0) {
|
|
70
|
+
return;
|
|
71
|
+
}
|
|
72
|
+
|
|
69
73
|
// If the transcript lacks exon coordinates and only has a single CDS
|
|
70
74
|
// region, then if the CDS region fits within the gene range, make a
|
|
71
75
|
// single exon, using the transcript start and end. This prevents issues
|
|
@@ -247,6 +251,9 @@ int Tx::closest_exon_num(int pos, std::vector<Region> & group) {
|
|
|
247
251
|
//
|
|
248
252
|
// @param position integer chromosome position e.g. 10000000
|
|
249
253
|
Region Tx::get_closest_exon(int pos) {
|
|
254
|
+
if (exons.size() == 0) {
|
|
255
|
+
throw std::invalid_argument("no exons assigned to this transcript");
|
|
256
|
+
}
|
|
250
257
|
int idx = closest_exon_num(pos);
|
|
251
258
|
return exons[idx];
|
|
252
259
|
}
|
|
@@ -255,6 +262,9 @@ Region Tx::get_closest_exon(int pos) {
|
|
|
255
262
|
//
|
|
256
263
|
// @param position integer chromosome position e.g. 10000000
|
|
257
264
|
bool Tx::in_coding_region(int pos) {
|
|
265
|
+
if (cds.size() == 0) {
|
|
266
|
+
return false;
|
|
267
|
+
}
|
|
258
268
|
int idx = closest_exon_num(pos, cds);
|
|
259
269
|
Region region = cds[idx];
|
|
260
270
|
return (pos >= region.start) && (pos <= region.end);
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: gencodegenes
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.1.1
|
|
4
4
|
Summary: Package to load genes from GENCODE GTF files
|
|
5
5
|
Home-page: https://github.com/jeremymcrae/gencodegenes
|
|
6
6
|
Author: Jeremy McRae
|
|
@@ -29,7 +29,10 @@ pip install gencodegenes
|
|
|
29
29
|
```py
|
|
30
30
|
from gencodegenes import Gencode
|
|
31
31
|
|
|
32
|
-
gencode = Gencode(GTF_PATH)
|
|
32
|
+
gencode = Gencode(GTF_PATH)
|
|
33
|
+
# full function arguments are Gencode(gtf_path, fasta_path=None, coding_only=True)
|
|
34
|
+
# - fasta_path: pass in path to fasta file to get gene transcripts with sequence
|
|
35
|
+
# - coding_only: pass in False to include all transcripts, not just protein coding
|
|
33
36
|
|
|
34
37
|
# get gene by HGNC symbol
|
|
35
38
|
gene = gencode['OR5A1']
|
|
@@ -14,70 +14,91 @@
|
|
|
14
14
|
|
|
15
15
|
namespace gencode {
|
|
16
16
|
|
|
17
|
-
|
|
17
|
+
std::string trim(const std::string &s, const std::string &vals) {
|
|
18
|
+
size_t start = s.find_first_not_of(vals);
|
|
19
|
+
if (start == std::string::npos) {
|
|
20
|
+
return "";
|
|
21
|
+
}
|
|
22
|
+
size_t end = s.find_last_not_of(vals);
|
|
23
|
+
return s.substr(start, (end + 1) - start);
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
// parse the required fields from the attributes field
|
|
18
27
|
static void get_attributes_fields(GTFLine &info, std::string &line, int offset) {
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
28
|
+
const std::string tx_id_key = "transcript_id ";
|
|
29
|
+
const std::string gene_id_key = "gene_id ";
|
|
30
|
+
const std::string gene_name_key = "gene_name ";
|
|
31
|
+
std::string type_key = "transcript_type ";
|
|
32
|
+
const std::string hgnc_id_key = "hgnc_id ";
|
|
33
|
+
|
|
34
|
+
size_t tx_start = line.find(tx_id_key, offset) + tx_id_key.size();
|
|
35
|
+
size_t tx_end = line.find(";", tx_start);
|
|
23
36
|
|
|
24
|
-
if (tx_start -
|
|
37
|
+
if (tx_start - tx_id_key.size() == std::string::npos) {
|
|
25
38
|
// handle if the string was not found
|
|
26
39
|
tx_start = offset;
|
|
27
40
|
tx_end = offset;
|
|
28
41
|
}
|
|
29
42
|
|
|
30
|
-
size_t gene_id_start = line.find(
|
|
31
|
-
size_t gene_id_end = line.find("
|
|
32
|
-
if (gene_id_start -
|
|
43
|
+
size_t gene_id_start = line.find(gene_id_key, offset) + gene_id_key.size();
|
|
44
|
+
size_t gene_id_end = line.find(";", gene_id_start);
|
|
45
|
+
if (gene_id_start - gene_id_key.size() == std::string::npos) {
|
|
33
46
|
// handle if the string was not found
|
|
34
47
|
gene_id_start = offset;
|
|
35
48
|
gene_id_end = offset;
|
|
36
49
|
}
|
|
37
50
|
|
|
51
|
+
size_t gene_start = line.find(gene_name_key, offset) + gene_name_key.size();
|
|
52
|
+
size_t gene_end = line.find(";", gene_start);
|
|
38
53
|
|
|
39
|
-
|
|
40
|
-
size_t gene_end = line.find("\"", gene_start);
|
|
41
|
-
|
|
42
|
-
if (gene_start - 11 == std::string::npos) {
|
|
54
|
+
if (gene_start - gene_name_key.size() == std::string::npos) {
|
|
43
55
|
// handle if the string was not found
|
|
44
56
|
gene_start = tx_end;
|
|
45
57
|
gene_end = tx_end;
|
|
46
58
|
}
|
|
47
59
|
|
|
48
|
-
size_t type_start = line.find(
|
|
49
|
-
|
|
60
|
+
size_t type_start = line.find(type_key, offset) + type_key.size();
|
|
61
|
+
if (type_start - type_key.size() == std::string::npos) {
|
|
62
|
+
// allow for alternate transcript_type key, as found in non-gencode GTF files
|
|
63
|
+
type_key = "transcript_biotype ";
|
|
64
|
+
type_start = line.find(type_key, offset) + type_key.size();
|
|
65
|
+
}
|
|
66
|
+
size_t type_end = line.find(";", type_start);
|
|
50
67
|
|
|
51
|
-
if (type_start -
|
|
68
|
+
if (type_start - type_key.size() == std::string::npos) {
|
|
52
69
|
// handle if the string was not found
|
|
53
70
|
type_start = gene_end;
|
|
54
71
|
type_end = gene_end;
|
|
55
72
|
}
|
|
56
73
|
|
|
57
|
-
size_t hgnc_id_start = line.find(
|
|
58
|
-
size_t hgnc_id_end = line.find("
|
|
59
|
-
if (hgnc_id_start -
|
|
74
|
+
size_t hgnc_id_start = line.find(hgnc_id_key, offset) + hgnc_id_key.size();
|
|
75
|
+
size_t hgnc_id_end = line.find(";", hgnc_id_start);
|
|
76
|
+
if (hgnc_id_start - hgnc_id_key.size() == std::string::npos) {
|
|
60
77
|
// handle if the string was not found
|
|
61
78
|
hgnc_id_start = offset;
|
|
62
79
|
hgnc_id_end = offset;
|
|
63
80
|
}
|
|
64
81
|
|
|
65
|
-
info.symbol = line.substr(gene_start, gene_end - gene_start);
|
|
66
|
-
info.tx_id = line.substr(tx_start, tx_end - tx_start);
|
|
67
|
-
info.transcript_type = line.substr(type_start, type_end - type_start);
|
|
82
|
+
info.symbol = trim(line.substr(gene_start, gene_end - gene_start), " \"");
|
|
83
|
+
info.tx_id = trim(line.substr(tx_start, tx_end - tx_start), " \"");
|
|
84
|
+
info.transcript_type = trim(line.substr(type_start, type_end - type_start), " \"");
|
|
68
85
|
|
|
69
86
|
if (gene_id_start != gene_id_end) {
|
|
70
|
-
|
|
87
|
+
std::string gene_id = trim(line.substr(gene_id_start, gene_id_end - gene_id_start), " \"");
|
|
88
|
+
if (info.symbol.size() == 0) {
|
|
89
|
+
info.symbol = gene_id;
|
|
90
|
+
} else {
|
|
91
|
+
info.alternate_ids.push_back(gene_id);
|
|
92
|
+
}
|
|
71
93
|
}
|
|
72
94
|
if (hgnc_id_start != hgnc_id_end) {
|
|
73
|
-
info.alternate_ids.push_back(line.substr(hgnc_id_start, hgnc_id_end - hgnc_id_start));
|
|
95
|
+
info.alternate_ids.push_back(trim(line.substr(hgnc_id_start, hgnc_id_end - hgnc_id_start), " \""));
|
|
74
96
|
}
|
|
75
97
|
|
|
76
|
-
info.is_canonical = 0;
|
|
77
98
|
if (info.feature == "transcript") {
|
|
78
|
-
if (line.find("appris_principal",
|
|
99
|
+
if (line.find("appris_principal", offset) != std::string::npos) {
|
|
79
100
|
info.is_canonical = 5;
|
|
80
|
-
} else if (line.find("Ensembl_canonical",
|
|
101
|
+
} else if (line.find("Ensembl_canonical", offset) != std::string::npos) {
|
|
81
102
|
info.is_canonical = 10;
|
|
82
103
|
}
|
|
83
104
|
}
|
|
@@ -96,7 +117,7 @@ GTFLine parse_gtfline(std::string & line) {
|
|
|
96
117
|
// tab along the line, then extract the substring to get the required fields.
|
|
97
118
|
// getline() with tab delimiter was 2X slower.
|
|
98
119
|
int chr_idx = 0;
|
|
99
|
-
int source_idx = line.find("\t", chr_idx
|
|
120
|
+
int source_idx = line.find("\t", chr_idx);
|
|
100
121
|
int feature_idx = line.find("\t", source_idx + 6);
|
|
101
122
|
int start_idx = line.find("\t", feature_idx + 3);
|
|
102
123
|
int end_idx = line.find("\t", start_idx + 2);
|
|
@@ -66,6 +66,10 @@ void Tx::set_exons(std::vector<std::vector<int>> exon_ranges) {
|
|
|
66
66
|
void Tx::set_cds(std::vector<std::vector<int>> cds_ranges) {
|
|
67
67
|
cds.clear();
|
|
68
68
|
|
|
69
|
+
if (cds_ranges.size() == 0) {
|
|
70
|
+
return;
|
|
71
|
+
}
|
|
72
|
+
|
|
69
73
|
// If the transcript lacks exon coordinates and only has a single CDS
|
|
70
74
|
// region, then if the CDS region fits within the gene range, make a
|
|
71
75
|
// single exon, using the transcript start and end. This prevents issues
|
|
@@ -247,6 +251,9 @@ int Tx::closest_exon_num(int pos, std::vector<Region> & group) {
|
|
|
247
251
|
//
|
|
248
252
|
// @param position integer chromosome position e.g. 10000000
|
|
249
253
|
Region Tx::get_closest_exon(int pos) {
|
|
254
|
+
if (exons.size() == 0) {
|
|
255
|
+
throw std::invalid_argument("no exons assigned to this transcript");
|
|
256
|
+
}
|
|
250
257
|
int idx = closest_exon_num(pos);
|
|
251
258
|
return exons[idx];
|
|
252
259
|
}
|
|
@@ -255,6 +262,9 @@ Region Tx::get_closest_exon(int pos) {
|
|
|
255
262
|
//
|
|
256
263
|
// @param position integer chromosome position e.g. 10000000
|
|
257
264
|
bool Tx::in_coding_region(int pos) {
|
|
265
|
+
if (cds.size() == 0) {
|
|
266
|
+
return false;
|
|
267
|
+
}
|
|
258
268
|
int idx = closest_exon_num(pos, cds);
|
|
259
269
|
Region region = cds[idx];
|
|
260
270
|
return (pos >= region.start) && (pos <= region.end);
|
|
@@ -55,12 +55,12 @@ class TestGencode(unittest.TestCase):
|
|
|
55
55
|
lines = '##format: gtf\n' \
|
|
56
56
|
'chr1\tHAVANA\tgene\t10\t20\t.\t-\t.\tgene_name "TEST1";\n' \
|
|
57
57
|
'chr1\tHAVANA\ttranscript\t10\t20\t.\t-\t.\ttranscript_id "ENST_A";gene_name "TEST1"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
|
|
58
|
-
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n' \
|
|
59
|
-
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n' \
|
|
58
|
+
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n' \
|
|
59
|
+
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n' \
|
|
60
60
|
'chr1\tHAVANA\tgene\t100\t110\t.\t-\t.\tgene_name "TEST2";\n' \
|
|
61
61
|
'chr1\tHAVANA\ttranscript\t100\t110\t.\t-\t.\ttranscript_id "ENST_B";gene_name "TEST2"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
|
|
62
|
-
'chr1\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST2"; transcript_type "protein_coding"\n' \
|
|
63
|
-
'chr1\tHAVANA\tCDS\t105\t110\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST2"; transcript_type "protein_coding"\n'\
|
|
62
|
+
'chr1\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding;"\n' \
|
|
63
|
+
'chr1\tHAVANA\tCDS\t105\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding;"\n'\
|
|
64
64
|
'chr2\tHAVANA\tgene\t100\t110\t.\t-\t.\tgene_name "TEST3";\n' \
|
|
65
65
|
'chr2\tHAVANA\ttranscript\t100\t110\t.\t-\t.\ttranscript_id "ENST_C";gene_name "TEST3"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
|
|
66
66
|
'chr2\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_C" gene_name "TEST3"; transcript_type "protein_coding"\n' \
|
|
@@ -112,16 +112,16 @@ class TestGencode(unittest.TestCase):
|
|
|
112
112
|
lines = ['##format: gtf\n',
|
|
113
113
|
'chr1\tHAVANA\tgene\t10\t20\t.\t-\t.\tgene_name "TEST1";\n',
|
|
114
114
|
'chr1\tHAVANA\ttranscript\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding"; tag "appris_principal_1";\n',
|
|
115
|
-
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n',
|
|
116
|
-
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n',
|
|
115
|
+
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n',
|
|
116
|
+
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n',
|
|
117
117
|
'chr1\tHAVANA\tgene\t100\t110\t.\t-\t.\tgene_name "TEST2";\n',
|
|
118
118
|
'chr1\tHAVANA\ttranscript\t100\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding"; tag "appris_principal_1";\n',
|
|
119
|
-
'chr1\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST2"; transcript_type "protein_coding"\n',
|
|
120
|
-
'chr1\tHAVANA\tCDS\t105\t110\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST2"; transcript_type "protein_coding"\n',
|
|
119
|
+
'chr1\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding;"\n',
|
|
120
|
+
'chr1\tHAVANA\tCDS\t105\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding;"\n',
|
|
121
121
|
'chr2\tHAVANA\tgene\t100\t110\t.\t-\t.\tgene_name "TEST3";\n',
|
|
122
122
|
'chr2\tHAVANA\ttranscript\t100\t110\t.\t-\t.\ttranscript_id "ENST_C"; gene_name "TEST3"; transcript_type "protein_coding"; tag "appris_principal_1";\n',
|
|
123
|
-
'chr2\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_C" gene_name "TEST3"; transcript_type "protein_coding"\n',
|
|
124
|
-
'chr2\tHAVANA\tCDS\t105\t110\t.\t-\t.\ttranscript_id "ENST_C" gene_name "TEST3"; transcript_type "protein_coding"\n']
|
|
123
|
+
'chr2\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_C"; gene_name "TEST3"; transcript_type "protein_coding;"\n',
|
|
124
|
+
'chr2\tHAVANA\tCDS\t105\t110\t.\t-\t.\ttranscript_id "ENST_C"; gene_name "TEST3"; transcript_type "protein_coding;"\n']
|
|
125
125
|
|
|
126
126
|
write_gtf(self.temp_gtf_path, lines)
|
|
127
127
|
make_fasta(self.temp_fasta_path, ['chr1', 'chr2'])
|
|
@@ -150,12 +150,12 @@ class TestGencode(unittest.TestCase):
|
|
|
150
150
|
lines = ['##format: gtf\n',
|
|
151
151
|
'chr1\tHAVANA\tgene\t20\t30\t.\t-\t.\tgene_name "TEST1";\n',
|
|
152
152
|
'chr1\tHAVANA\ttranscript\t20\t30\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding"; tag "appris_principal_1";\n',
|
|
153
|
-
'chr1\tHAVANA\texon\t20\t30\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n',
|
|
154
|
-
'chr1\tHAVANA\tCDS\t25\t30\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n',
|
|
153
|
+
'chr1\tHAVANA\texon\t20\t30\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n',
|
|
154
|
+
'chr1\tHAVANA\tCDS\t25\t30\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n',
|
|
155
155
|
'chr1\tHAVANA\tgene\t100\t110\t.\t-\t.\tgene_name "TEST1";\n',
|
|
156
156
|
'chr1\tHAVANA\ttranscript\t100\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST1"; transcript_type "protein_coding";\n',
|
|
157
|
-
'chr1\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST1"; transcript_type "protein_coding"\n',
|
|
158
|
-
'chr1\tHAVANA\tCDS\t110\t100\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST1"; transcript_type "protein_coding"\n',
|
|
157
|
+
'chr1\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST1"; transcript_type "protein_coding;"\n',
|
|
158
|
+
'chr1\tHAVANA\tCDS\t110\t100\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST1"; transcript_type "protein_coding;"\n',
|
|
159
159
|
]
|
|
160
160
|
|
|
161
161
|
write_gtf(self.temp_gtf_path, lines)
|
|
@@ -231,7 +231,7 @@ class TestGencode(unittest.TestCase):
|
|
|
231
231
|
'gene_status "KNOWN"; gene_name "OR4F5"; transcript_type "protein_coding"; ' \
|
|
232
232
|
'transcript_status "KNOWN"; transcript_name "OR4F5-001"; level 2; ' \
|
|
233
233
|
'protein_id "ENSP00000334393.3"; tag "basic"; transcript_support_level "NA"; ' \
|
|
234
|
-
'hgnc_id "HGNC:14825"
|
|
234
|
+
'hgnc_id "HGNC:14825"; tag "appris_principal_1"; tag "CCDS"; ccdsid "CCDS30547.1"; ' \
|
|
235
235
|
'havana_gene "OTTHUMG00000001094.2"; havana_transcript "OTTHUMT00000003223.2";\n'
|
|
236
236
|
obj = _parse_gtfline(line.encode('utf8'))
|
|
237
237
|
expected = {'chrom': b'chr1',
|
|
@@ -251,7 +251,7 @@ class TestGencode(unittest.TestCase):
|
|
|
251
251
|
line = 'chr1\tHAVANA\ttranscript\t69091\t70008\t.\t+\t.\t '\
|
|
252
252
|
'transcript_id "ENST00000335137.3"; gene_type "protein_coding"; ' \
|
|
253
253
|
'gene_status "KNOWN"; gene_name "OR4F5"; transcript_type "protein_coding";' \
|
|
254
|
-
'hgnc_id "HGNC:14825"
|
|
254
|
+
'hgnc_id "HGNC:14825"; tag "appris_principal_1"; tag "CCDS"; ccdsid "CCDS30547.1"; '
|
|
255
255
|
obj = _parse_gtfline(line.encode('utf8'))
|
|
256
256
|
expected = {'chrom': b'chr1',
|
|
257
257
|
'feature': b'transcript',
|
|
@@ -431,12 +431,12 @@ class TestGencode(unittest.TestCase):
|
|
|
431
431
|
lines = '##format: gtf\n' \
|
|
432
432
|
'chr1\tHAVANA\tgene\t10\t20\t.\t-\t.\tgene_name "TEST1";\n' \
|
|
433
433
|
'chr1\tHAVANA\ttranscript\t10\t20\t.\t-\t.\ttranscript_id "ENST_A";gene_name "TEST1"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
|
|
434
|
-
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n' \
|
|
435
|
-
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n' \
|
|
434
|
+
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n' \
|
|
435
|
+
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n' \
|
|
436
436
|
'chr1\tHAVANA\tgene\t10\t30\t.\t-\t.\tgene_name "TEST2";\n' \
|
|
437
437
|
'chr1\tHAVANA\ttranscript\t10\t30\t.\t-\t.\ttranscript_id "ENST_B";gene_name "TEST2"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
|
|
438
|
-
'chr1\tHAVANA\texon\t10\t30\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST2"; transcript_type "protein_coding"\n' \
|
|
439
|
-
'chr1\tHAVANA\tCDS\t15\t30\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST2"; transcript_type "protein_coding"\n'
|
|
438
|
+
'chr1\tHAVANA\texon\t10\t30\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding;"\n' \
|
|
439
|
+
'chr1\tHAVANA\tCDS\t15\t30\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding;"\n'
|
|
440
440
|
|
|
441
441
|
write_gtf(self.temp_gtf_path, lines)
|
|
442
442
|
data = _open_gencode(self.temp_gtf_path)
|
|
@@ -452,8 +452,8 @@ class TestGencode(unittest.TestCase):
|
|
|
452
452
|
lines = '##format: gtf\n' \
|
|
453
453
|
'chr1\tHAVANA\tgene\t10\t20\t.\t-\t.\tgene_name "TEST";\n' \
|
|
454
454
|
'chr1\tHAVANA\ttranscript\t10\t20\t.\t-\t.\ttranscript_id "ENST_A";gene_name "TEST"; transcript_type "processed_transcript"; tag "appris_principal_1";\n' \
|
|
455
|
-
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "processed_transcript"\n' \
|
|
456
|
-
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "processed_transcript"\n' \
|
|
455
|
+
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "processed_transcript;"\n' \
|
|
456
|
+
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "processed_transcript;"\n' \
|
|
457
457
|
|
|
458
458
|
write_gtf(self.temp_gtf_path, lines)
|
|
459
459
|
data = _open_gencode(self.temp_gtf_path)
|
|
@@ -472,12 +472,12 @@ class TestGencode(unittest.TestCase):
|
|
|
472
472
|
lines = '##format: gtf\n' \
|
|
473
473
|
'chr1\tHAVANA\tgene\t10\t20\t.\t-\t.\tgene_name "TEST";\n' \
|
|
474
474
|
'chr1\tHAVANA\ttranscript\t10\t20\t.\t-\t.\ttranscript_id "ENST_A";gene_name "TEST"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
|
|
475
|
-
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
|
|
476
|
-
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
|
|
475
|
+
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
|
|
476
|
+
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
|
|
477
477
|
'chr1\tHAVANA\tgene\t10\t30\t.\t-\t.\tgene_name "TEST";\n' \
|
|
478
478
|
'chr1\tHAVANA\ttranscript\t10\t30\t.\t-\t.\ttranscript_id "ENST_B";gene_name "TEST"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
|
|
479
|
-
'chr1\tHAVANA\texon\t10\t30\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST"; transcript_type "protein_coding"\n' \
|
|
480
|
-
'chr1\tHAVANA\tCDS\t15\t30\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST"; transcript_type "protein_coding"\n'
|
|
479
|
+
'chr1\tHAVANA\texon\t10\t30\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
|
|
480
|
+
'chr1\tHAVANA\tCDS\t15\t30\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST"; transcript_type "protein_coding;"\n'
|
|
481
481
|
|
|
482
482
|
write_gtf(self.temp_gtf_path, lines)
|
|
483
483
|
data = _open_gencode(self.temp_gtf_path)
|
|
@@ -497,13 +497,38 @@ class TestGencode(unittest.TestCase):
|
|
|
497
497
|
lines = '##format: gtf\n' \
|
|
498
498
|
'chr1\tHAVANA\tgene\t10\t100\t.\t-\t.\tgene_name "TEST";\n' \
|
|
499
499
|
'chr1\tHAVANA\ttranscript\t10\t100\t.\t-\t.\ttranscript_id "ENST_A";gene_name "TEST"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
|
|
500
|
-
'chr1\tHAVANA\tUTR\t10\t15\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding";\n' \
|
|
501
|
-
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
|
|
502
|
-
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
|
|
503
|
-
'chr1\tHAVANA\texon\t30\t40\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
|
|
504
|
-
'chr1\tHAVANA\tCDS\t30\t40\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
|
|
505
|
-
'chr1\tHAVANA\texon\t90\t100\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
|
|
506
|
-
'chr1\tHAVANA\tUTR\t90\t100\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n'
|
|
500
|
+
'chr1\tHAVANA\tUTR\t10\t15\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding";\n' \
|
|
501
|
+
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
|
|
502
|
+
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
|
|
503
|
+
'chr1\tHAVANA\texon\t30\t40\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
|
|
504
|
+
'chr1\tHAVANA\tCDS\t30\t40\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
|
|
505
|
+
'chr1\tHAVANA\texon\t90\t100\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
|
|
506
|
+
'chr1\tHAVANA\tUTR\t90\t100\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n'
|
|
507
|
+
|
|
508
|
+
write_gtf(self.temp_gtf_path, lines)
|
|
509
|
+
data = _open_gencode(self.temp_gtf_path)
|
|
510
|
+
|
|
511
|
+
self.assertEqual(len(data), 1)
|
|
512
|
+
symbol, tx, is_principal = data[0]
|
|
513
|
+
self.assertEqual(symbol, 'TEST')
|
|
514
|
+
self.assertEqual(tx.name, 'ENST_A')
|
|
515
|
+
self.assertEqual(tx.strand, '-')
|
|
516
|
+
self.assertEqual(tx.exons, [{'start': 10, 'end': 20}, {'start': 30, 'end': 40}, {'start': 90, 'end': 100}])
|
|
517
|
+
self.assertEqual(tx.cds, [{'start': 15, 'end': 20}, {'start': 30, 'end': 40}])
|
|
518
|
+
|
|
519
|
+
def test__open_gencode_unquoted(self):
|
|
520
|
+
'''test we can parse a GTF without quoted attributes
|
|
521
|
+
'''
|
|
522
|
+
lines = '##format: gtf\n' \
|
|
523
|
+
'chr1\tHAVANA\tgene\t10\t100\t.\t-\t.\tgene_name "TEST";\n' \
|
|
524
|
+
'chr1\tHAVANA\ttranscript\t10\t100\t.\t-\t.\ttranscript_id ENST_A;gene_name TEST; transcript_type protein_coding; tag appris_principal_1;\n' \
|
|
525
|
+
'chr1\tHAVANA\tUTR\t10\t15\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n' \
|
|
526
|
+
'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n' \
|
|
527
|
+
'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n' \
|
|
528
|
+
'chr1\tHAVANA\texon\t30\t40\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n' \
|
|
529
|
+
'chr1\tHAVANA\tCDS\t30\t40\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n' \
|
|
530
|
+
'chr1\tHAVANA\texon\t90\t100\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n' \
|
|
531
|
+
'chr1\tHAVANA\tUTR\t90\t100\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n'
|
|
507
532
|
|
|
508
533
|
write_gtf(self.temp_gtf_path, lines)
|
|
509
534
|
data = _open_gencode(self.temp_gtf_path)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|