gencodegenes 1.0.10__tar.gz → 1.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. {gencodegenes-1.0.10/src/gencodegenes.egg-info → gencodegenes-1.1.1}/PKG-INFO +5 -2
  2. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/README.md +4 -1
  3. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/setup.py +1 -1
  4. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencodegenes/gencode.cpp +55 -41
  5. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencodegenes/gencode.pyx +1 -1
  6. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencodegenes/tx.cpp +10 -0
  7. {gencodegenes-1.0.10 → gencodegenes-1.1.1/src/gencodegenes.egg-info}/PKG-INFO +5 -2
  8. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gtf.cpp +49 -28
  9. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gtf.h +1 -1
  10. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/tx.cpp +10 -0
  11. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/tests/test_gencode.py +58 -33
  12. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/LICENSE.txt +0 -0
  13. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/MANIFEST.in +0 -0
  14. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/pyproject.toml +0 -0
  15. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/setup.cfg +0 -0
  16. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencode.cpp +0 -0
  17. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencode.h +0 -0
  18. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencodegenes/__init__.py +0 -0
  19. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencodegenes/transcript.cpp +0 -0
  20. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencodegenes/transcript.pxd +0 -0
  21. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencodegenes/transcript.pyx +0 -0
  22. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencodegenes/tx.h +0 -0
  23. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencodegenes.egg-info/SOURCES.txt +0 -0
  24. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencodegenes.egg-info/dependency_links.txt +0 -0
  25. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencodegenes.egg-info/requires.txt +0 -0
  26. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gencodegenes.egg-info/top_level.txt +0 -0
  27. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gzstream/gzstream.C +0 -0
  28. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/gzstream/gzstream.h +0 -0
  29. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/src/tx.h +0 -0
  30. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/tests/__init__.py +0 -0
  31. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/tests/data/example.grch38.fa +0 -0
  32. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/tests/data/example.grch38.gtf +0 -0
  33. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/tests/test_sequence_methods.py +0 -0
  34. {gencodegenes-1.0.10 → gencodegenes-1.1.1}/tests/test_transcript.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: gencodegenes
3
- Version: 1.0.10
3
+ Version: 1.1.1
4
4
  Summary: Package to load genes from GENCODE GTF files
5
5
  Home-page: https://github.com/jeremymcrae/gencodegenes
6
6
  Author: Jeremy McRae
@@ -29,7 +29,10 @@ pip install gencodegenes
29
29
  ```py
30
30
  from gencodegenes import Gencode
31
31
 
32
- gencode = Gencode(GTF_PATH) # or Gencode(GTF, FASTA_PATH) to give transcripts DNA sequence
32
+ gencode = Gencode(GTF_PATH)
33
+ # full function arguments are Gencode(gtf_path, fasta_path=None, coding_only=True)
34
+ # - fasta_path: pass in path to fasta file to get gene transcripts with sequence
35
+ # - coding_only: pass in False to include all transcripts, not just protein coding
33
36
 
34
37
  # get gene by HGNC symbol
35
38
  gene = gencode['OR5A1']
@@ -14,7 +14,10 @@ pip install gencodegenes
14
14
  ```py
15
15
  from gencodegenes import Gencode
16
16
 
17
- gencode = Gencode(GTF_PATH) # or Gencode(GTF, FASTA_PATH) to give transcripts DNA sequence
17
+ gencode = Gencode(GTF_PATH)
18
+ # full function arguments are Gencode(gtf_path, fasta_path=None, coding_only=True)
19
+ # - fasta_path: pass in path to fasta file to get gene transcripts with sequence
20
+ # - coding_only: pass in False to include all transcripts, not just protein coding
18
21
 
19
22
  # get gene by HGNC symbol
20
23
  gene = gencode['OR5A1']
@@ -112,7 +112,7 @@ setup(name="gencodegenes",
112
112
  description='Package to load genes from GENCODE GTF files',
113
113
  long_description=io.open('README.md', encoding='utf-8').read(),
114
114
  long_description_content_type='text/markdown',
115
- version="1.0.10",
115
+ version="1.1.1",
116
116
  author="Jeremy McRae",
117
117
  author_email="jeremy.mcrae@gmail.com",
118
118
  license="MIT",
@@ -5755,11 +5755,12 @@ static PyObject *__pyx_f_12gencodegenes_7gencode__open_gencode(PyObject *__pyx_v
5755
5755
  int __pyx_t_4;
5756
5756
  std::string __pyx_t_5;
5757
5757
  bool __pyx_t_6;
5758
- std::vector<struct gencode::NamedTx> ::iterator __pyx_t_7;
5759
- struct gencode::NamedTx __pyx_t_8;
5760
- gencode::Tx __pyx_t_9;
5761
- PyObject *__pyx_t_10 = NULL;
5762
- int __pyx_t_11;
5758
+ std::vector<struct gencode::NamedTx> __pyx_t_7;
5759
+ std::vector<struct gencode::NamedTx> ::iterator __pyx_t_8;
5760
+ struct gencode::NamedTx __pyx_t_9;
5761
+ gencode::Tx __pyx_t_10;
5762
+ PyObject *__pyx_t_11 = NULL;
5763
+ int __pyx_t_12;
5763
5764
  int __pyx_lineno = 0;
5764
5765
  const char *__pyx_filename = NULL;
5765
5766
  int __pyx_clineno = 0;
@@ -5804,7 +5805,13 @@ static PyObject *__pyx_f_12gencodegenes_7gencode__open_gencode(PyObject *__pyx_v
5804
5805
  __pyx_t_5 = __pyx_convert_string_from_py_6libcpp_6string_std__in_string(__pyx_t_1); if (unlikely(PyErr_Occurred())) __PYX_ERR(0, 66, __pyx_L1_error)
5805
5806
  __Pyx_DECREF(__pyx_t_1); __pyx_t_1 = 0;
5806
5807
  __pyx_t_6 = __Pyx_PyObject_IsTrue(__pyx_v_coding_only); if (unlikely((__pyx_t_6 == ((bool)-1)) && PyErr_Occurred())) __PYX_ERR(0, 66, __pyx_L1_error)
5807
- __pyx_v__transcripts = gencode::open_gencode(__PYX_STD_MOVE_IF_SUPPORTED(__pyx_t_5), __pyx_t_6);
5808
+ try {
5809
+ __pyx_t_7 = gencode::open_gencode(__PYX_STD_MOVE_IF_SUPPORTED(__pyx_t_5), __pyx_t_6);
5810
+ } catch(...) {
5811
+ __Pyx_CppExn2PyErr();
5812
+ __PYX_ERR(0, 66, __pyx_L1_error)
5813
+ }
5814
+ __pyx_v__transcripts = __PYX_STD_MOVE_IF_SUPPORTED(__pyx_t_7);
5808
5815
 
5809
5816
  /* "gencodegenes/gencode.pyx":68
5810
5817
  * cdef vector[NamedTx] _transcripts = open_gencode(gtf_path.encode('utf8'), coding_only)
@@ -5825,12 +5832,12 @@ static PyObject *__pyx_f_12gencodegenes_7gencode__open_gencode(PyObject *__pyx_v
5825
5832
  * tx = x.tx
5826
5833
  * chrom = tx.get_chrom().decode('utf8')
5827
5834
  */
5828
- __pyx_t_7 = __pyx_v__transcripts.begin();
5835
+ __pyx_t_8 = __pyx_v__transcripts.begin();
5829
5836
  for (;;) {
5830
- if (!(__pyx_t_7 != __pyx_v__transcripts.end())) break;
5831
- __pyx_t_8 = *__pyx_t_7;
5832
- ++__pyx_t_7;
5833
- __pyx_v_x = __pyx_t_8;
5837
+ if (!(__pyx_t_8 != __pyx_v__transcripts.end())) break;
5838
+ __pyx_t_9 = *__pyx_t_8;
5839
+ ++__pyx_t_8;
5840
+ __pyx_v_x = __pyx_t_9;
5834
5841
 
5835
5842
  /* "gencodegenes/gencode.pyx":70
5836
5843
  * transcripts = []
@@ -5839,8 +5846,8 @@ static PyObject *__pyx_f_12gencodegenes_7gencode__open_gencode(PyObject *__pyx_v
5839
5846
  * chrom = tx.get_chrom().decode('utf8')
5840
5847
  * start = tx.get_start()
5841
5848
  */
5842
- __pyx_t_9 = __pyx_v_x.tx;
5843
- __pyx_v_tx = __PYX_STD_MOVE_IF_SUPPORTED(__pyx_t_9);
5849
+ __pyx_t_10 = __pyx_v_x.tx;
5850
+ __pyx_v_tx = __PYX_STD_MOVE_IF_SUPPORTED(__pyx_t_10);
5844
5851
 
5845
5852
  /* "gencodegenes/gencode.pyx":71
5846
5853
  * for x in _transcripts:
@@ -5948,39 +5955,39 @@ static PyObject *__pyx_f_12gencodegenes_7gencode__open_gencode(PyObject *__pyx_v
5948
5955
  __Pyx_GOTREF(__pyx_t_1);
5949
5956
  __pyx_t_3 = __Pyx_PyInt_From_int(__pyx_v_end); if (unlikely(!__pyx_t_3)) __PYX_ERR(0, 79, __pyx_L1_error)
5950
5957
  __Pyx_GOTREF(__pyx_t_3);
5951
- __pyx_t_10 = PyTuple_New(8); if (unlikely(!__pyx_t_10)) __PYX_ERR(0, 79, __pyx_L1_error)
5952
- __Pyx_GOTREF(__pyx_t_10);
5958
+ __pyx_t_11 = PyTuple_New(8); if (unlikely(!__pyx_t_11)) __PYX_ERR(0, 79, __pyx_L1_error)
5959
+ __Pyx_GOTREF(__pyx_t_11);
5953
5960
  __Pyx_INCREF(__pyx_v_tx_id);
5954
5961
  __Pyx_GIVEREF(__pyx_v_tx_id);
5955
- if (__Pyx_PyTuple_SET_ITEM(__pyx_t_10, 0, __pyx_v_tx_id)) __PYX_ERR(0, 79, __pyx_L1_error);
5962
+ if (__Pyx_PyTuple_SET_ITEM(__pyx_t_11, 0, __pyx_v_tx_id)) __PYX_ERR(0, 79, __pyx_L1_error);
5956
5963
  __Pyx_INCREF(__pyx_v_chrom);
5957
5964
  __Pyx_GIVEREF(__pyx_v_chrom);
5958
- if (__Pyx_PyTuple_SET_ITEM(__pyx_t_10, 1, __pyx_v_chrom)) __PYX_ERR(0, 79, __pyx_L1_error);
5965
+ if (__Pyx_PyTuple_SET_ITEM(__pyx_t_11, 1, __pyx_v_chrom)) __PYX_ERR(0, 79, __pyx_L1_error);
5959
5966
  __Pyx_GIVEREF(__pyx_t_1);
5960
- if (__Pyx_PyTuple_SET_ITEM(__pyx_t_10, 2, __pyx_t_1)) __PYX_ERR(0, 79, __pyx_L1_error);
5967
+ if (__Pyx_PyTuple_SET_ITEM(__pyx_t_11, 2, __pyx_t_1)) __PYX_ERR(0, 79, __pyx_L1_error);
5961
5968
  __Pyx_GIVEREF(__pyx_t_3);
5962
- if (__Pyx_PyTuple_SET_ITEM(__pyx_t_10, 3, __pyx_t_3)) __PYX_ERR(0, 79, __pyx_L1_error);
5969
+ if (__Pyx_PyTuple_SET_ITEM(__pyx_t_11, 3, __pyx_t_3)) __PYX_ERR(0, 79, __pyx_L1_error);
5963
5970
  __Pyx_INCREF(__pyx_v_strand);
5964
5971
  __Pyx_GIVEREF(__pyx_v_strand);
5965
- if (__Pyx_PyTuple_SET_ITEM(__pyx_t_10, 4, __pyx_v_strand)) __PYX_ERR(0, 79, __pyx_L1_error);
5972
+ if (__Pyx_PyTuple_SET_ITEM(__pyx_t_11, 4, __pyx_v_strand)) __PYX_ERR(0, 79, __pyx_L1_error);
5966
5973
  __Pyx_INCREF(__pyx_v_transcript_type);
5967
5974
  __Pyx_GIVEREF(__pyx_v_transcript_type);
5968
- if (__Pyx_PyTuple_SET_ITEM(__pyx_t_10, 5, __pyx_v_transcript_type)) __PYX_ERR(0, 79, __pyx_L1_error);
5975
+ if (__Pyx_PyTuple_SET_ITEM(__pyx_t_11, 5, __pyx_v_transcript_type)) __PYX_ERR(0, 79, __pyx_L1_error);
5969
5976
  __Pyx_INCREF(__pyx_v_exons);
5970
5977
  __Pyx_GIVEREF(__pyx_v_exons);
5971
- if (__Pyx_PyTuple_SET_ITEM(__pyx_t_10, 6, __pyx_v_exons)) __PYX_ERR(0, 79, __pyx_L1_error);
5978
+ if (__Pyx_PyTuple_SET_ITEM(__pyx_t_11, 6, __pyx_v_exons)) __PYX_ERR(0, 79, __pyx_L1_error);
5972
5979
  __Pyx_INCREF(__pyx_v_cds);
5973
5980
  __Pyx_GIVEREF(__pyx_v_cds);
5974
- if (__Pyx_PyTuple_SET_ITEM(__pyx_t_10, 7, __pyx_v_cds)) __PYX_ERR(0, 79, __pyx_L1_error);
5981
+ if (__Pyx_PyTuple_SET_ITEM(__pyx_t_11, 7, __pyx_v_cds)) __PYX_ERR(0, 79, __pyx_L1_error);
5975
5982
  __pyx_t_1 = 0;
5976
5983
  __pyx_t_3 = 0;
5977
5984
  __pyx_t_3 = __Pyx_PyDict_NewPresized(1); if (unlikely(!__pyx_t_3)) __PYX_ERR(0, 79, __pyx_L1_error)
5978
5985
  __Pyx_GOTREF(__pyx_t_3);
5979
5986
  if (PyDict_SetItem(__pyx_t_3, __pyx_n_s_offset, __pyx_int_0) < 0) __PYX_ERR(0, 79, __pyx_L1_error)
5980
- __pyx_t_1 = __Pyx_PyObject_Call(__pyx_t_2, __pyx_t_10, __pyx_t_3); if (unlikely(!__pyx_t_1)) __PYX_ERR(0, 79, __pyx_L1_error)
5987
+ __pyx_t_1 = __Pyx_PyObject_Call(__pyx_t_2, __pyx_t_11, __pyx_t_3); if (unlikely(!__pyx_t_1)) __PYX_ERR(0, 79, __pyx_L1_error)
5981
5988
  __Pyx_GOTREF(__pyx_t_1);
5982
5989
  __Pyx_DECREF(__pyx_t_2); __pyx_t_2 = 0;
5983
- __Pyx_DECREF(__pyx_t_10); __pyx_t_10 = 0;
5990
+ __Pyx_DECREF(__pyx_t_11); __pyx_t_11 = 0;
5984
5991
  __Pyx_DECREF(__pyx_t_3); __pyx_t_3 = 0;
5985
5992
  __Pyx_XDECREF_SET(__pyx_v_transcript, __pyx_t_1);
5986
5993
  __pyx_t_1 = 0;
@@ -5996,19 +6003,19 @@ static PyObject *__pyx_f_12gencodegenes_7gencode__open_gencode(PyObject *__pyx_v
5996
6003
  __Pyx_GOTREF(__pyx_t_1);
5997
6004
  __pyx_t_3 = __Pyx_PyInt_From_int(__pyx_v_x.is_canonical); if (unlikely(!__pyx_t_3)) __PYX_ERR(0, 80, __pyx_L1_error)
5998
6005
  __Pyx_GOTREF(__pyx_t_3);
5999
- __pyx_t_10 = PyTuple_New(3); if (unlikely(!__pyx_t_10)) __PYX_ERR(0, 80, __pyx_L1_error)
6000
- __Pyx_GOTREF(__pyx_t_10);
6006
+ __pyx_t_11 = PyTuple_New(3); if (unlikely(!__pyx_t_11)) __PYX_ERR(0, 80, __pyx_L1_error)
6007
+ __Pyx_GOTREF(__pyx_t_11);
6001
6008
  __Pyx_GIVEREF(__pyx_t_1);
6002
- if (__Pyx_PyTuple_SET_ITEM(__pyx_t_10, 0, __pyx_t_1)) __PYX_ERR(0, 80, __pyx_L1_error);
6009
+ if (__Pyx_PyTuple_SET_ITEM(__pyx_t_11, 0, __pyx_t_1)) __PYX_ERR(0, 80, __pyx_L1_error);
6003
6010
  __Pyx_INCREF(__pyx_v_transcript);
6004
6011
  __Pyx_GIVEREF(__pyx_v_transcript);
6005
- if (__Pyx_PyTuple_SET_ITEM(__pyx_t_10, 1, __pyx_v_transcript)) __PYX_ERR(0, 80, __pyx_L1_error);
6012
+ if (__Pyx_PyTuple_SET_ITEM(__pyx_t_11, 1, __pyx_v_transcript)) __PYX_ERR(0, 80, __pyx_L1_error);
6006
6013
  __Pyx_GIVEREF(__pyx_t_3);
6007
- if (__Pyx_PyTuple_SET_ITEM(__pyx_t_10, 2, __pyx_t_3)) __PYX_ERR(0, 80, __pyx_L1_error);
6014
+ if (__Pyx_PyTuple_SET_ITEM(__pyx_t_11, 2, __pyx_t_3)) __PYX_ERR(0, 80, __pyx_L1_error);
6008
6015
  __pyx_t_1 = 0;
6009
6016
  __pyx_t_3 = 0;
6010
- __pyx_t_11 = __Pyx_PyList_Append(__pyx_v_transcripts, __pyx_t_10); if (unlikely(__pyx_t_11 == ((int)-1))) __PYX_ERR(0, 80, __pyx_L1_error)
6011
- __Pyx_DECREF(__pyx_t_10); __pyx_t_10 = 0;
6017
+ __pyx_t_12 = __Pyx_PyList_Append(__pyx_v_transcripts, __pyx_t_11); if (unlikely(__pyx_t_12 == ((int)-1))) __PYX_ERR(0, 80, __pyx_L1_error)
6018
+ __Pyx_DECREF(__pyx_t_11); __pyx_t_11 = 0;
6012
6019
 
6013
6020
  /* "gencodegenes/gencode.pyx":69
6014
6021
  *
@@ -6044,7 +6051,7 @@ static PyObject *__pyx_f_12gencodegenes_7gencode__open_gencode(PyObject *__pyx_v
6044
6051
  __Pyx_XDECREF(__pyx_t_1);
6045
6052
  __Pyx_XDECREF(__pyx_t_2);
6046
6053
  __Pyx_XDECREF(__pyx_t_3);
6047
- __Pyx_XDECREF(__pyx_t_10);
6054
+ __Pyx_XDECREF(__pyx_t_11);
6048
6055
  __Pyx_AddTraceback("gencodegenes.gencode._open_gencode", __pyx_clineno, __pyx_lineno, __pyx_filename);
6049
6056
  __pyx_r = 0;
6050
6057
  __pyx_L0:;
@@ -11146,8 +11153,9 @@ static int __pyx_pf_12gencodegenes_7gencode_7Gencode___cinit__(struct __pyx_obj_
11146
11153
  int __pyx_t_8;
11147
11154
  std::string __pyx_t_9;
11148
11155
  bool __pyx_t_10;
11149
- std::vector<struct gencode::NamedTx> ::iterator __pyx_t_11;
11150
- struct gencode::NamedTx __pyx_t_12;
11156
+ std::vector<struct gencode::NamedTx> __pyx_t_11;
11157
+ std::vector<struct gencode::NamedTx> ::iterator __pyx_t_12;
11158
+ struct gencode::NamedTx __pyx_t_13;
11151
11159
  int __pyx_lineno = 0;
11152
11160
  const char *__pyx_filename = NULL;
11153
11161
  int __pyx_clineno = 0;
@@ -11549,7 +11557,13 @@ static int __pyx_pf_12gencodegenes_7gencode_7Gencode___cinit__(struct __pyx_obj_
11549
11557
  __pyx_t_9 = __pyx_convert_string_from_py_6libcpp_6string_std__in_string(__pyx_t_3); if (unlikely(PyErr_Occurred())) __PYX_ERR(0, 349, __pyx_L1_error)
11550
11558
  __Pyx_DECREF(__pyx_t_3); __pyx_t_3 = 0;
11551
11559
  __pyx_t_10 = __Pyx_PyObject_IsTrue(__pyx_v_coding_only); if (unlikely((__pyx_t_10 == ((bool)-1)) && PyErr_Occurred())) __PYX_ERR(0, 349, __pyx_L1_error)
11552
- __pyx_v_transcripts = gencode::open_gencode(__PYX_STD_MOVE_IF_SUPPORTED(__pyx_t_9), __pyx_t_10);
11560
+ try {
11561
+ __pyx_t_11 = gencode::open_gencode(__PYX_STD_MOVE_IF_SUPPORTED(__pyx_t_9), __pyx_t_10);
11562
+ } catch(...) {
11563
+ __Pyx_CppExn2PyErr();
11564
+ __PYX_ERR(0, 349, __pyx_L1_error)
11565
+ }
11566
+ __pyx_v_transcripts = __PYX_STD_MOVE_IF_SUPPORTED(__pyx_t_11);
11553
11567
 
11554
11568
  /* "gencodegenes/gencode.pyx":350
11555
11569
  * if gencode is not None:
@@ -11558,12 +11572,12 @@ static int __pyx_pf_12gencodegenes_7gencode_7Gencode___cinit__(struct __pyx_obj_
11558
11572
  * symbol = x.symbol.decode('utf8')
11559
11573
  * if symbol not in self.genes:
11560
11574
  */
11561
- __pyx_t_11 = __pyx_v_transcripts.begin();
11575
+ __pyx_t_12 = __pyx_v_transcripts.begin();
11562
11576
  for (;;) {
11563
- if (!(__pyx_t_11 != __pyx_v_transcripts.end())) break;
11564
- __pyx_t_12 = *__pyx_t_11;
11565
- ++__pyx_t_11;
11566
- __pyx_v_x = __pyx_t_12;
11577
+ if (!(__pyx_t_12 != __pyx_v_transcripts.end())) break;
11578
+ __pyx_t_13 = *__pyx_t_12;
11579
+ ++__pyx_t_12;
11580
+ __pyx_v_x = __pyx_t_13;
11567
11581
 
11568
11582
  /* "gencodegenes/gencode.pyx":351
11569
11583
  * transcripts = open_gencode(str(gencode).encode('utf8'), coding_only)
@@ -41,7 +41,7 @@ cdef extern from "gencode.h" namespace "gencode":
41
41
  int pos
42
42
  string symbol
43
43
 
44
- vector[NamedTx] open_gencode(string, bool)
44
+ vector[NamedTx] open_gencode(string, bool) except +
45
45
  bool CompFunc(const GenePoint &l, const GenePoint &r)
46
46
  vector[string] _in_region(string chrom, int start, int end,
47
47
  map[string, vector[GenePoint]] & starts, map[string, vector[GenePoint]] & ends,
@@ -66,6 +66,10 @@ void Tx::set_exons(std::vector<std::vector<int>> exon_ranges) {
66
66
  void Tx::set_cds(std::vector<std::vector<int>> cds_ranges) {
67
67
  cds.clear();
68
68
 
69
+ if (cds_ranges.size() == 0) {
70
+ return;
71
+ }
72
+
69
73
  // If the transcript lacks exon coordinates and only has a single CDS
70
74
  // region, then if the CDS region fits within the gene range, make a
71
75
  // single exon, using the transcript start and end. This prevents issues
@@ -247,6 +251,9 @@ int Tx::closest_exon_num(int pos, std::vector<Region> & group) {
247
251
  //
248
252
  // @param position integer chromosome position e.g. 10000000
249
253
  Region Tx::get_closest_exon(int pos) {
254
+ if (exons.size() == 0) {
255
+ throw std::invalid_argument("no exons assigned to this transcript");
256
+ }
250
257
  int idx = closest_exon_num(pos);
251
258
  return exons[idx];
252
259
  }
@@ -255,6 +262,9 @@ Region Tx::get_closest_exon(int pos) {
255
262
  //
256
263
  // @param position integer chromosome position e.g. 10000000
257
264
  bool Tx::in_coding_region(int pos) {
265
+ if (cds.size() == 0) {
266
+ return false;
267
+ }
258
268
  int idx = closest_exon_num(pos, cds);
259
269
  Region region = cds[idx];
260
270
  return (pos >= region.start) && (pos <= region.end);
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: gencodegenes
3
- Version: 1.0.10
3
+ Version: 1.1.1
4
4
  Summary: Package to load genes from GENCODE GTF files
5
5
  Home-page: https://github.com/jeremymcrae/gencodegenes
6
6
  Author: Jeremy McRae
@@ -29,7 +29,10 @@ pip install gencodegenes
29
29
  ```py
30
30
  from gencodegenes import Gencode
31
31
 
32
- gencode = Gencode(GTF_PATH) # or Gencode(GTF, FASTA_PATH) to give transcripts DNA sequence
32
+ gencode = Gencode(GTF_PATH)
33
+ # full function arguments are Gencode(gtf_path, fasta_path=None, coding_only=True)
34
+ # - fasta_path: pass in path to fasta file to get gene transcripts with sequence
35
+ # - coding_only: pass in False to include all transcripts, not just protein coding
33
36
 
34
37
  # get gene by HGNC symbol
35
38
  gene = gencode['OR5A1']
@@ -14,70 +14,91 @@
14
14
 
15
15
  namespace gencode {
16
16
 
17
- // parse the required firleds from the attributes field
17
+ std::string trim(const std::string &s, const std::string &vals) {
18
+ size_t start = s.find_first_not_of(vals);
19
+ if (start == std::string::npos) {
20
+ return "";
21
+ }
22
+ size_t end = s.find_last_not_of(vals);
23
+ return s.substr(start, (end + 1) - start);
24
+ }
25
+
26
+ // parse the required fields from the attributes field
18
27
  static void get_attributes_fields(GTFLine &info, std::string &line, int offset) {
19
- // we could check for each field individually, but since we know the order
20
- // of the fields, it's much quicker to just search the remaining substring
21
- size_t tx_start = line.find("transcript_id", offset) + 15;
22
- size_t tx_end = line.find("\"", tx_start);
28
+ const std::string tx_id_key = "transcript_id ";
29
+ const std::string gene_id_key = "gene_id ";
30
+ const std::string gene_name_key = "gene_name ";
31
+ std::string type_key = "transcript_type ";
32
+ const std::string hgnc_id_key = "hgnc_id ";
33
+
34
+ size_t tx_start = line.find(tx_id_key, offset) + tx_id_key.size();
35
+ size_t tx_end = line.find(";", tx_start);
23
36
 
24
- if (tx_start - 15 == std::string::npos) {
37
+ if (tx_start - tx_id_key.size() == std::string::npos) {
25
38
  // handle if the string was not found
26
39
  tx_start = offset;
27
40
  tx_end = offset;
28
41
  }
29
42
 
30
- size_t gene_id_start = line.find("gene_id", offset) + 9;
31
- size_t gene_id_end = line.find("\"", gene_id_start);
32
- if (gene_id_start - 9 == std::string::npos) {
43
+ size_t gene_id_start = line.find(gene_id_key, offset) + gene_id_key.size();
44
+ size_t gene_id_end = line.find(";", gene_id_start);
45
+ if (gene_id_start - gene_id_key.size() == std::string::npos) {
33
46
  // handle if the string was not found
34
47
  gene_id_start = offset;
35
48
  gene_id_end = offset;
36
49
  }
37
50
 
51
+ size_t gene_start = line.find(gene_name_key, offset) + gene_name_key.size();
52
+ size_t gene_end = line.find(";", gene_start);
38
53
 
39
- size_t gene_start = line.find("gene_name", tx_end) + 11;
40
- size_t gene_end = line.find("\"", gene_start);
41
-
42
- if (gene_start - 11 == std::string::npos) {
54
+ if (gene_start - gene_name_key.size() == std::string::npos) {
43
55
  // handle if the string was not found
44
56
  gene_start = tx_end;
45
57
  gene_end = tx_end;
46
58
  }
47
59
 
48
- size_t type_start = line.find("transcript_type", gene_end) + 17;
49
- size_t type_end = line.find("\"", type_start);
60
+ size_t type_start = line.find(type_key, offset) + type_key.size();
61
+ if (type_start - type_key.size() == std::string::npos) {
62
+ // allow for alternate transcript_type key, as found in non-gencode GTF files
63
+ type_key = "transcript_biotype ";
64
+ type_start = line.find(type_key, offset) + type_key.size();
65
+ }
66
+ size_t type_end = line.find(";", type_start);
50
67
 
51
- if (type_start - 17 == std::string::npos) {
68
+ if (type_start - type_key.size() == std::string::npos) {
52
69
  // handle if the string was not found
53
70
  type_start = gene_end;
54
71
  type_end = gene_end;
55
72
  }
56
73
 
57
- size_t hgnc_id_start = line.find("hgnc_id", type_end) + 9;
58
- size_t hgnc_id_end = line.find("\"", hgnc_id_start);
59
- if (hgnc_id_start - 9 == std::string::npos) {
74
+ size_t hgnc_id_start = line.find(hgnc_id_key, offset) + hgnc_id_key.size();
75
+ size_t hgnc_id_end = line.find(";", hgnc_id_start);
76
+ if (hgnc_id_start - hgnc_id_key.size() == std::string::npos) {
60
77
  // handle if the string was not found
61
78
  hgnc_id_start = offset;
62
79
  hgnc_id_end = offset;
63
80
  }
64
81
 
65
- info.symbol = line.substr(gene_start, gene_end - gene_start);
66
- info.tx_id = line.substr(tx_start, tx_end - tx_start);
67
- info.transcript_type = line.substr(type_start, type_end - type_start);
82
+ info.symbol = trim(line.substr(gene_start, gene_end - gene_start), " \"");
83
+ info.tx_id = trim(line.substr(tx_start, tx_end - tx_start), " \"");
84
+ info.transcript_type = trim(line.substr(type_start, type_end - type_start), " \"");
68
85
 
69
86
  if (gene_id_start != gene_id_end) {
70
- info.alternate_ids.push_back(line.substr(gene_id_start, gene_id_end - gene_id_start));
87
+ std::string gene_id = trim(line.substr(gene_id_start, gene_id_end - gene_id_start), " \"");
88
+ if (info.symbol.size() == 0) {
89
+ info.symbol = gene_id;
90
+ } else {
91
+ info.alternate_ids.push_back(gene_id);
92
+ }
71
93
  }
72
94
  if (hgnc_id_start != hgnc_id_end) {
73
- info.alternate_ids.push_back(line.substr(hgnc_id_start, hgnc_id_end - hgnc_id_start));
95
+ info.alternate_ids.push_back(trim(line.substr(hgnc_id_start, hgnc_id_end - hgnc_id_start), " \""));
74
96
  }
75
97
 
76
- info.is_canonical = 0;
77
98
  if (info.feature == "transcript") {
78
- if (line.find("appris_principal", type_end) != std::string::npos) {
99
+ if (line.find("appris_principal", offset) != std::string::npos) {
79
100
  info.is_canonical = 5;
80
- } else if (line.find("Ensembl_canonical", type_end) != std::string::npos) {
101
+ } else if (line.find("Ensembl_canonical", offset) != std::string::npos) {
81
102
  info.is_canonical = 10;
82
103
  }
83
104
  }
@@ -96,7 +117,7 @@ GTFLine parse_gtfline(std::string & line) {
96
117
  // tab along the line, then extract the substring to get the required fields.
97
118
  // getline() with tab delimiter was 2X slower.
98
119
  int chr_idx = 0;
99
- int source_idx = line.find("\t", chr_idx + 4);
120
+ int source_idx = line.find("\t", chr_idx);
100
121
  int feature_idx = line.find("\t", source_idx + 6);
101
122
  int start_idx = line.find("\t", feature_idx + 3);
102
123
  int end_idx = line.find("\t", start_idx + 2);
@@ -21,7 +21,7 @@ struct GTFLine {
21
21
  std::vector<std::string> alternate_ids;
22
22
  std::string tx_id;
23
23
  std::string transcript_type;
24
- int is_canonical;
24
+ int is_canonical = 0;
25
25
  };
26
26
 
27
27
  GTFLine parse_gtfline(std::string &line);
@@ -66,6 +66,10 @@ void Tx::set_exons(std::vector<std::vector<int>> exon_ranges) {
66
66
  void Tx::set_cds(std::vector<std::vector<int>> cds_ranges) {
67
67
  cds.clear();
68
68
 
69
+ if (cds_ranges.size() == 0) {
70
+ return;
71
+ }
72
+
69
73
  // If the transcript lacks exon coordinates and only has a single CDS
70
74
  // region, then if the CDS region fits within the gene range, make a
71
75
  // single exon, using the transcript start and end. This prevents issues
@@ -247,6 +251,9 @@ int Tx::closest_exon_num(int pos, std::vector<Region> & group) {
247
251
  //
248
252
  // @param position integer chromosome position e.g. 10000000
249
253
  Region Tx::get_closest_exon(int pos) {
254
+ if (exons.size() == 0) {
255
+ throw std::invalid_argument("no exons assigned to this transcript");
256
+ }
250
257
  int idx = closest_exon_num(pos);
251
258
  return exons[idx];
252
259
  }
@@ -255,6 +262,9 @@ Region Tx::get_closest_exon(int pos) {
255
262
  //
256
263
  // @param position integer chromosome position e.g. 10000000
257
264
  bool Tx::in_coding_region(int pos) {
265
+ if (cds.size() == 0) {
266
+ return false;
267
+ }
258
268
  int idx = closest_exon_num(pos, cds);
259
269
  Region region = cds[idx];
260
270
  return (pos >= region.start) && (pos <= region.end);
@@ -55,12 +55,12 @@ class TestGencode(unittest.TestCase):
55
55
  lines = '##format: gtf\n' \
56
56
  'chr1\tHAVANA\tgene\t10\t20\t.\t-\t.\tgene_name "TEST1";\n' \
57
57
  'chr1\tHAVANA\ttranscript\t10\t20\t.\t-\t.\ttranscript_id "ENST_A";gene_name "TEST1"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
58
- 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n' \
59
- 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n' \
58
+ 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n' \
59
+ 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n' \
60
60
  'chr1\tHAVANA\tgene\t100\t110\t.\t-\t.\tgene_name "TEST2";\n' \
61
61
  'chr1\tHAVANA\ttranscript\t100\t110\t.\t-\t.\ttranscript_id "ENST_B";gene_name "TEST2"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
62
- 'chr1\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST2"; transcript_type "protein_coding"\n' \
63
- 'chr1\tHAVANA\tCDS\t105\t110\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST2"; transcript_type "protein_coding"\n'\
62
+ 'chr1\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding;"\n' \
63
+ 'chr1\tHAVANA\tCDS\t105\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding;"\n'\
64
64
  'chr2\tHAVANA\tgene\t100\t110\t.\t-\t.\tgene_name "TEST3";\n' \
65
65
  'chr2\tHAVANA\ttranscript\t100\t110\t.\t-\t.\ttranscript_id "ENST_C";gene_name "TEST3"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
66
66
  'chr2\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_C" gene_name "TEST3"; transcript_type "protein_coding"\n' \
@@ -112,16 +112,16 @@ class TestGencode(unittest.TestCase):
112
112
  lines = ['##format: gtf\n',
113
113
  'chr1\tHAVANA\tgene\t10\t20\t.\t-\t.\tgene_name "TEST1";\n',
114
114
  'chr1\tHAVANA\ttranscript\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding"; tag "appris_principal_1";\n',
115
- 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n',
116
- 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n',
115
+ 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n',
116
+ 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n',
117
117
  'chr1\tHAVANA\tgene\t100\t110\t.\t-\t.\tgene_name "TEST2";\n',
118
118
  'chr1\tHAVANA\ttranscript\t100\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding"; tag "appris_principal_1";\n',
119
- 'chr1\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST2"; transcript_type "protein_coding"\n',
120
- 'chr1\tHAVANA\tCDS\t105\t110\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST2"; transcript_type "protein_coding"\n',
119
+ 'chr1\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding;"\n',
120
+ 'chr1\tHAVANA\tCDS\t105\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding;"\n',
121
121
  'chr2\tHAVANA\tgene\t100\t110\t.\t-\t.\tgene_name "TEST3";\n',
122
122
  'chr2\tHAVANA\ttranscript\t100\t110\t.\t-\t.\ttranscript_id "ENST_C"; gene_name "TEST3"; transcript_type "protein_coding"; tag "appris_principal_1";\n',
123
- 'chr2\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_C" gene_name "TEST3"; transcript_type "protein_coding"\n',
124
- 'chr2\tHAVANA\tCDS\t105\t110\t.\t-\t.\ttranscript_id "ENST_C" gene_name "TEST3"; transcript_type "protein_coding"\n']
123
+ 'chr2\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_C"; gene_name "TEST3"; transcript_type "protein_coding;"\n',
124
+ 'chr2\tHAVANA\tCDS\t105\t110\t.\t-\t.\ttranscript_id "ENST_C"; gene_name "TEST3"; transcript_type "protein_coding;"\n']
125
125
 
126
126
  write_gtf(self.temp_gtf_path, lines)
127
127
  make_fasta(self.temp_fasta_path, ['chr1', 'chr2'])
@@ -150,12 +150,12 @@ class TestGencode(unittest.TestCase):
150
150
  lines = ['##format: gtf\n',
151
151
  'chr1\tHAVANA\tgene\t20\t30\t.\t-\t.\tgene_name "TEST1";\n',
152
152
  'chr1\tHAVANA\ttranscript\t20\t30\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding"; tag "appris_principal_1";\n',
153
- 'chr1\tHAVANA\texon\t20\t30\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n',
154
- 'chr1\tHAVANA\tCDS\t25\t30\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n',
153
+ 'chr1\tHAVANA\texon\t20\t30\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n',
154
+ 'chr1\tHAVANA\tCDS\t25\t30\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n',
155
155
  'chr1\tHAVANA\tgene\t100\t110\t.\t-\t.\tgene_name "TEST1";\n',
156
156
  'chr1\tHAVANA\ttranscript\t100\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST1"; transcript_type "protein_coding";\n',
157
- 'chr1\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST1"; transcript_type "protein_coding"\n',
158
- 'chr1\tHAVANA\tCDS\t110\t100\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST1"; transcript_type "protein_coding"\n',
157
+ 'chr1\tHAVANA\texon\t100\t110\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST1"; transcript_type "protein_coding;"\n',
158
+ 'chr1\tHAVANA\tCDS\t110\t100\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST1"; transcript_type "protein_coding;"\n',
159
159
  ]
160
160
 
161
161
  write_gtf(self.temp_gtf_path, lines)
@@ -231,7 +231,7 @@ class TestGencode(unittest.TestCase):
231
231
  'gene_status "KNOWN"; gene_name "OR4F5"; transcript_type "protein_coding"; ' \
232
232
  'transcript_status "KNOWN"; transcript_name "OR4F5-001"; level 2; ' \
233
233
  'protein_id "ENSP00000334393.3"; tag "basic"; transcript_support_level "NA"; ' \
234
- 'hgnc_id "HGNC:14825", tag "appris_principal_1"; tag "CCDS"; ccdsid "CCDS30547.1"; ' \
234
+ 'hgnc_id "HGNC:14825"; tag "appris_principal_1"; tag "CCDS"; ccdsid "CCDS30547.1"; ' \
235
235
  'havana_gene "OTTHUMG00000001094.2"; havana_transcript "OTTHUMT00000003223.2";\n'
236
236
  obj = _parse_gtfline(line.encode('utf8'))
237
237
  expected = {'chrom': b'chr1',
@@ -251,7 +251,7 @@ class TestGencode(unittest.TestCase):
251
251
  line = 'chr1\tHAVANA\ttranscript\t69091\t70008\t.\t+\t.\t '\
252
252
  'transcript_id "ENST00000335137.3"; gene_type "protein_coding"; ' \
253
253
  'gene_status "KNOWN"; gene_name "OR4F5"; transcript_type "protein_coding";' \
254
- 'hgnc_id "HGNC:14825", tag "appris_principal_1"; tag "CCDS"; ccdsid "CCDS30547.1"; '
254
+ 'hgnc_id "HGNC:14825"; tag "appris_principal_1"; tag "CCDS"; ccdsid "CCDS30547.1"; '
255
255
  obj = _parse_gtfline(line.encode('utf8'))
256
256
  expected = {'chrom': b'chr1',
257
257
  'feature': b'transcript',
@@ -431,12 +431,12 @@ class TestGencode(unittest.TestCase):
431
431
  lines = '##format: gtf\n' \
432
432
  'chr1\tHAVANA\tgene\t10\t20\t.\t-\t.\tgene_name "TEST1";\n' \
433
433
  'chr1\tHAVANA\ttranscript\t10\t20\t.\t-\t.\ttranscript_id "ENST_A";gene_name "TEST1"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
434
- 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n' \
435
- 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST1"; transcript_type "protein_coding"\n' \
434
+ 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n' \
435
+ 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST1"; transcript_type "protein_coding;"\n' \
436
436
  'chr1\tHAVANA\tgene\t10\t30\t.\t-\t.\tgene_name "TEST2";\n' \
437
437
  'chr1\tHAVANA\ttranscript\t10\t30\t.\t-\t.\ttranscript_id "ENST_B";gene_name "TEST2"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
438
- 'chr1\tHAVANA\texon\t10\t30\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST2"; transcript_type "protein_coding"\n' \
439
- 'chr1\tHAVANA\tCDS\t15\t30\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST2"; transcript_type "protein_coding"\n'
438
+ 'chr1\tHAVANA\texon\t10\t30\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding;"\n' \
439
+ 'chr1\tHAVANA\tCDS\t15\t30\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST2"; transcript_type "protein_coding;"\n'
440
440
 
441
441
  write_gtf(self.temp_gtf_path, lines)
442
442
  data = _open_gencode(self.temp_gtf_path)
@@ -452,8 +452,8 @@ class TestGencode(unittest.TestCase):
452
452
  lines = '##format: gtf\n' \
453
453
  'chr1\tHAVANA\tgene\t10\t20\t.\t-\t.\tgene_name "TEST";\n' \
454
454
  'chr1\tHAVANA\ttranscript\t10\t20\t.\t-\t.\ttranscript_id "ENST_A";gene_name "TEST"; transcript_type "processed_transcript"; tag "appris_principal_1";\n' \
455
- 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "processed_transcript"\n' \
456
- 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "processed_transcript"\n' \
455
+ 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "processed_transcript;"\n' \
456
+ 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "processed_transcript;"\n' \
457
457
 
458
458
  write_gtf(self.temp_gtf_path, lines)
459
459
  data = _open_gencode(self.temp_gtf_path)
@@ -472,12 +472,12 @@ class TestGencode(unittest.TestCase):
472
472
  lines = '##format: gtf\n' \
473
473
  'chr1\tHAVANA\tgene\t10\t20\t.\t-\t.\tgene_name "TEST";\n' \
474
474
  'chr1\tHAVANA\ttranscript\t10\t20\t.\t-\t.\ttranscript_id "ENST_A";gene_name "TEST"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
475
- 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
476
- 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
475
+ 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
476
+ 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
477
477
  'chr1\tHAVANA\tgene\t10\t30\t.\t-\t.\tgene_name "TEST";\n' \
478
478
  'chr1\tHAVANA\ttranscript\t10\t30\t.\t-\t.\ttranscript_id "ENST_B";gene_name "TEST"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
479
- 'chr1\tHAVANA\texon\t10\t30\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST"; transcript_type "protein_coding"\n' \
480
- 'chr1\tHAVANA\tCDS\t15\t30\t.\t-\t.\ttranscript_id "ENST_B" gene_name "TEST"; transcript_type "protein_coding"\n'
479
+ 'chr1\tHAVANA\texon\t10\t30\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
480
+ 'chr1\tHAVANA\tCDS\t15\t30\t.\t-\t.\ttranscript_id "ENST_B"; gene_name "TEST"; transcript_type "protein_coding;"\n'
481
481
 
482
482
  write_gtf(self.temp_gtf_path, lines)
483
483
  data = _open_gencode(self.temp_gtf_path)
@@ -497,13 +497,38 @@ class TestGencode(unittest.TestCase):
497
497
  lines = '##format: gtf\n' \
498
498
  'chr1\tHAVANA\tgene\t10\t100\t.\t-\t.\tgene_name "TEST";\n' \
499
499
  'chr1\tHAVANA\ttranscript\t10\t100\t.\t-\t.\ttranscript_id "ENST_A";gene_name "TEST"; transcript_type "protein_coding"; tag "appris_principal_1";\n' \
500
- 'chr1\tHAVANA\tUTR\t10\t15\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding";\n' \
501
- 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
502
- 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
503
- 'chr1\tHAVANA\texon\t30\t40\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
504
- 'chr1\tHAVANA\tCDS\t30\t40\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
505
- 'chr1\tHAVANA\texon\t90\t100\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n' \
506
- 'chr1\tHAVANA\tUTR\t90\t100\t.\t-\t.\ttranscript_id "ENST_A" gene_name "TEST"; transcript_type "protein_coding"\n'
500
+ 'chr1\tHAVANA\tUTR\t10\t15\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding";\n' \
501
+ 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
502
+ 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
503
+ 'chr1\tHAVANA\texon\t30\t40\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
504
+ 'chr1\tHAVANA\tCDS\t30\t40\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
505
+ 'chr1\tHAVANA\texon\t90\t100\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n' \
506
+ 'chr1\tHAVANA\tUTR\t90\t100\t.\t-\t.\ttranscript_id "ENST_A"; gene_name "TEST"; transcript_type "protein_coding;"\n'
507
+
508
+ write_gtf(self.temp_gtf_path, lines)
509
+ data = _open_gencode(self.temp_gtf_path)
510
+
511
+ self.assertEqual(len(data), 1)
512
+ symbol, tx, is_principal = data[0]
513
+ self.assertEqual(symbol, 'TEST')
514
+ self.assertEqual(tx.name, 'ENST_A')
515
+ self.assertEqual(tx.strand, '-')
516
+ self.assertEqual(tx.exons, [{'start': 10, 'end': 20}, {'start': 30, 'end': 40}, {'start': 90, 'end': 100}])
517
+ self.assertEqual(tx.cds, [{'start': 15, 'end': 20}, {'start': 30, 'end': 40}])
518
+
519
+ def test__open_gencode_unquoted(self):
520
+ '''test we can parse a GTF without quoted attributes
521
+ '''
522
+ lines = '##format: gtf\n' \
523
+ 'chr1\tHAVANA\tgene\t10\t100\t.\t-\t.\tgene_name "TEST";\n' \
524
+ 'chr1\tHAVANA\ttranscript\t10\t100\t.\t-\t.\ttranscript_id ENST_A;gene_name TEST; transcript_type protein_coding; tag appris_principal_1;\n' \
525
+ 'chr1\tHAVANA\tUTR\t10\t15\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n' \
526
+ 'chr1\tHAVANA\texon\t10\t20\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n' \
527
+ 'chr1\tHAVANA\tCDS\t15\t20\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n' \
528
+ 'chr1\tHAVANA\texon\t30\t40\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n' \
529
+ 'chr1\tHAVANA\tCDS\t30\t40\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n' \
530
+ 'chr1\tHAVANA\texon\t90\t100\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n' \
531
+ 'chr1\tHAVANA\tUTR\t90\t100\t.\t-\t.\ttranscript_id ENST_A; gene_name TEST; transcript_type protein_coding;\n'
507
532
 
508
533
  write_gtf(self.temp_gtf_path, lines)
509
534
  data = _open_gencode(self.temp_gtf_path)
File without changes
File without changes
File without changes
File without changes