gtfparse 1.3.0__tar.gz → 2.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gtfparse-1.3.0 → gtfparse-2.1.0}/PKG-INFO +8 -5
- {gtfparse-1.3.0 → gtfparse-2.1.0}/gtfparse/__init__.py +12 -3
- {gtfparse-1.3.0 → gtfparse-2.1.0}/gtfparse/attribute_parsing.py +17 -18
- {gtfparse-1.3.0 → gtfparse-2.1.0}/gtfparse/create_missing_features.py +1 -1
- gtfparse-2.1.0/gtfparse/read_gtf.py +297 -0
- {gtfparse-1.3.0 → gtfparse-2.1.0}/gtfparse.egg-info/PKG-INFO +8 -5
- {gtfparse-1.3.0 → gtfparse-2.1.0}/gtfparse.egg-info/SOURCES.txt +10 -10
- gtfparse-2.1.0/gtfparse.egg-info/requires.txt +2 -0
- gtfparse-2.1.0/pyproject.toml +27 -0
- gtfparse-2.1.0/requirements.txt +2 -0
- {gtfparse-1.3.0/test → gtfparse-2.1.0/tests}/test_create_missing_features.py +2 -1
- {gtfparse-1.3.0/test → gtfparse-2.1.0/tests}/test_ensembl_gtf.py +2 -2
- gtfparse-2.1.0/tests/test_expand_attributes.py +36 -0
- {gtfparse-1.3.0/test → gtfparse-2.1.0/tests}/test_multiple_values_for_tag_attribute.py +6 -8
- {gtfparse-1.3.0/test → gtfparse-2.1.0/tests}/test_parse_gtf_lines.py +21 -29
- gtfparse-1.3.0/gtfparse/read_gtf.py +0 -243
- gtfparse-1.3.0/gtfparse/required_columns.py +0 -62
- gtfparse-1.3.0/gtfparse/version.py +0 -1
- gtfparse-1.3.0/gtfparse.egg-info/requires.txt +0 -2
- gtfparse-1.3.0/setup.py +0 -61
- gtfparse-1.3.0/test/test_expand_attributes.py +0 -37
- {gtfparse-1.3.0 → gtfparse-2.1.0}/LICENSE +0 -0
- {gtfparse-1.3.0 → gtfparse-2.1.0}/README.md +0 -0
- {gtfparse-1.3.0 → gtfparse-2.1.0}/gtfparse/parsing_error.py +0 -0
- {gtfparse-1.3.0 → gtfparse-2.1.0}/gtfparse.egg-info/dependency_links.txt +0 -0
- {gtfparse-1.3.0 → gtfparse-2.1.0}/gtfparse.egg-info/top_level.txt +0 -0
- {gtfparse-1.3.0 → gtfparse-2.1.0}/setup.cfg +0 -0
- {gtfparse-1.3.0/test → gtfparse-2.1.0/tests}/test_read_stringtie_gtf.py +0 -0
- {gtfparse-1.3.0/test → gtfparse-2.1.0/tests}/test_refseq_gtf.py +0 -0
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: gtfparse
|
|
3
|
-
Version: 1.
|
|
4
|
-
Summary: GTF
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
3
|
+
Version: 2.1.0
|
|
4
|
+
Summary: Parsing library for extracting data frames of genomic features from GTF files
|
|
5
|
+
Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
|
|
6
|
+
Project-URL: Homepage, https://github.com/openvax/gtfparse
|
|
7
|
+
Project-URL: Bug Tracker, https://github.com/openvax/gtfparse
|
|
8
8
|
Classifier: Development Status :: 4 - Beta
|
|
9
9
|
Classifier: Environment :: Console
|
|
10
10
|
Classifier: Operating System :: OS Independent
|
|
@@ -12,8 +12,11 @@ Classifier: Intended Audience :: Science/Research
|
|
|
12
12
|
Classifier: License :: OSI Approved :: Apache Software License
|
|
13
13
|
Classifier: Programming Language :: Python
|
|
14
14
|
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
15
|
+
Requires-Python: >=3.7
|
|
15
16
|
Description-Content-Type: text/markdown
|
|
16
17
|
License-File: LICENSE
|
|
18
|
+
Requires-Dist: polars<0.21.0,>=0.20.2
|
|
19
|
+
Requires-Dist: pyarrow<14.1.0,>=14.0.2
|
|
17
20
|
|
|
18
21
|
[](https://travis-ci.org/openvax/gtfparse) [](https://coveralls.io/github/openvax/gtfparse?branch=master)
|
|
19
22
|
<a href="https://pypi.python.org/pypi/gtfparse/">
|
|
@@ -12,17 +12,26 @@
|
|
|
12
12
|
|
|
13
13
|
from .attribute_parsing import expand_attribute_strings
|
|
14
14
|
from .create_missing_features import create_missing_features
|
|
15
|
-
from .required_columns import REQUIRED_COLUMNS
|
|
16
15
|
from .parsing_error import ParsingError
|
|
17
|
-
from .read_gtf import
|
|
16
|
+
from .read_gtf import (
|
|
17
|
+
read_gtf,
|
|
18
|
+
parse_gtf,
|
|
19
|
+
parse_gtf_pandas,
|
|
20
|
+
parse_gtf_and_expand_attributes,
|
|
21
|
+
REQUIRED_COLUMNS,
|
|
22
|
+
)
|
|
18
23
|
|
|
24
|
+
__version__ = "2.1.0"
|
|
19
25
|
|
|
20
26
|
__all__ = [
|
|
27
|
+
"__version__",
|
|
21
28
|
"expand_attribute_strings",
|
|
22
29
|
"create_missing_features",
|
|
23
|
-
|
|
30
|
+
|
|
24
31
|
"parse_gtf_and_expand_attributes",
|
|
25
32
|
"REQUIRED_COLUMNS",
|
|
26
33
|
"ParsingError",
|
|
27
34
|
"read_gtf",
|
|
35
|
+
"parse_gtf",
|
|
36
|
+
"parse_gtf_pandas",
|
|
28
37
|
]
|
|
@@ -18,9 +18,10 @@ logging.basicConfig(level=logging.INFO)
|
|
|
18
18
|
logger = logging.getLogger(__name__)
|
|
19
19
|
|
|
20
20
|
|
|
21
|
+
|
|
21
22
|
def expand_attribute_strings(
|
|
22
23
|
attribute_strings,
|
|
23
|
-
quote_char='
|
|
24
|
+
quote_char="'",
|
|
24
25
|
missing_value="",
|
|
25
26
|
usecols=None):
|
|
26
27
|
"""
|
|
@@ -64,10 +65,11 @@ def expand_attribute_strings(
|
|
|
64
65
|
# using a local dictionary, hence the two dictionaries below
|
|
65
66
|
# and pair of try/except blocks in the loop.
|
|
66
67
|
column_interned_strings = {}
|
|
67
|
-
value_interned_strings = {}
|
|
68
68
|
|
|
69
|
-
for (i,
|
|
70
|
-
|
|
69
|
+
for (i, kv_strings) in enumerate(attribute_strings):
|
|
70
|
+
if type(kv_strings) is str:
|
|
71
|
+
kv_strings = kv_strings.split(";")
|
|
72
|
+
for kv in kv_strings:
|
|
71
73
|
# We're slicing the first two elements out of split() because
|
|
72
74
|
# Ensembl release 79 added values like:
|
|
73
75
|
# transcript_support_level "1 (assigned to previous version 5)";
|
|
@@ -88,28 +90,25 @@ def expand_attribute_strings(
|
|
|
88
90
|
if usecols is not None and column_name not in usecols:
|
|
89
91
|
continue
|
|
90
92
|
|
|
93
|
+
if value[0] == quote_char:
|
|
94
|
+
value = value.replace(quote_char, "")
|
|
95
|
+
|
|
91
96
|
try:
|
|
92
97
|
column = extra_columns[column_name]
|
|
98
|
+
# if an attribute is used repeatedly then
|
|
99
|
+
# keep track of all its values in a list
|
|
100
|
+
old_value = column[i]
|
|
101
|
+
if old_value is missing_value:
|
|
102
|
+
column[i] = value
|
|
103
|
+
else:
|
|
104
|
+
column[i] = "%s,%s" % (old_value, value)
|
|
93
105
|
except KeyError:
|
|
94
106
|
column = [missing_value] * n
|
|
107
|
+
column[i] = value
|
|
95
108
|
extra_columns[column_name] = column
|
|
96
109
|
column_order.append(column_name)
|
|
97
110
|
|
|
98
|
-
value = value.replace(quote_char, "") if value.startswith(quote_char) else value
|
|
99
|
-
|
|
100
|
-
try:
|
|
101
|
-
value = value_interned_strings[value]
|
|
102
|
-
except KeyError:
|
|
103
|
-
value = intern(str(value))
|
|
104
|
-
value_interned_strings[value] = value
|
|
105
111
|
|
|
106
|
-
# if an attribute is used repeatedly then
|
|
107
|
-
# keep track of all its values in a list
|
|
108
|
-
old_value = column[i]
|
|
109
|
-
if old_value is missing_value:
|
|
110
|
-
column[i] = value
|
|
111
|
-
else:
|
|
112
|
-
column[i] = "%s,%s" % (old_value, value)
|
|
113
112
|
|
|
114
113
|
logging.info("Extracted GTF attributes: %s" % column_order)
|
|
115
114
|
return OrderedDict(
|
|
@@ -55,7 +55,7 @@ def create_missing_features(
|
|
|
55
55
|
extra_dataframes = []
|
|
56
56
|
|
|
57
57
|
existing_features = set(dataframe["feature"])
|
|
58
|
-
existing_columns = set(dataframe.
|
|
58
|
+
existing_columns = set(dataframe.columns)
|
|
59
59
|
|
|
60
60
|
for (feature_name, groupby_key) in unique_keys.items():
|
|
61
61
|
if feature_name in existing_features:
|
|
@@ -0,0 +1,297 @@
|
|
|
1
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
2
|
+
# you may not use this file except in compliance with the License.
|
|
3
|
+
# You may obtain a copy of the License at
|
|
4
|
+
#
|
|
5
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
6
|
+
#
|
|
7
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
8
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
9
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
10
|
+
# See the License for the specific language governing permissions and
|
|
11
|
+
# limitations under the License.
|
|
12
|
+
|
|
13
|
+
import logging
|
|
14
|
+
from os.path import exists
|
|
15
|
+
from io import StringIO
|
|
16
|
+
import gzip
|
|
17
|
+
|
|
18
|
+
import polars
|
|
19
|
+
|
|
20
|
+
from .attribute_parsing import expand_attribute_strings
|
|
21
|
+
from .parsing_error import ParsingError
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
logging.basicConfig(level=logging.INFO)
|
|
25
|
+
logger = logging.getLogger(__name__)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
"""
|
|
29
|
+
Columns of a GTF file:
|
|
30
|
+
|
|
31
|
+
seqname - name of the chromosome or scaffold; chromosome names
|
|
32
|
+
without a 'chr' in Ensembl (but sometimes with a 'chr'
|
|
33
|
+
elsewhere)
|
|
34
|
+
source - name of the program that generated this feature, or
|
|
35
|
+
the data source (database or project name)
|
|
36
|
+
feature - feature type name.
|
|
37
|
+
Features currently in Ensembl GTFs:
|
|
38
|
+
gene
|
|
39
|
+
transcript
|
|
40
|
+
exon
|
|
41
|
+
CDS
|
|
42
|
+
Selenocysteine
|
|
43
|
+
start_codon
|
|
44
|
+
stop_codon
|
|
45
|
+
UTR
|
|
46
|
+
Older Ensembl releases may be missing some of these features.
|
|
47
|
+
start - start position of the feature, with sequence numbering
|
|
48
|
+
starting at 1.
|
|
49
|
+
end - end position of the feature, with sequence numbering
|
|
50
|
+
starting at 1.
|
|
51
|
+
score - a floating point value indiciating the score of a feature
|
|
52
|
+
strand - defined as + (forward) or - (reverse).
|
|
53
|
+
frame - one of '0', '1' or '2'. Frame indicates the number of base pairs
|
|
54
|
+
before you encounter a full codon. '0' indicates the feature
|
|
55
|
+
begins with a whole codon. '1' indicates there is an extra
|
|
56
|
+
base (the 3rd base of the prior codon) at the start of this feature.
|
|
57
|
+
'2' indicates there are two extra bases (2nd and 3rd base of the
|
|
58
|
+
prior exon) before the first codon. All values are given with
|
|
59
|
+
relation to the 5' end.
|
|
60
|
+
attribute - a semicolon-separated list of tag-value pairs (separated by a space),
|
|
61
|
+
providing additional information about each feature. A key can be
|
|
62
|
+
repeated multiple times.
|
|
63
|
+
|
|
64
|
+
(from ftp://ftp.ensembl.org/pub/release-75/gtf/homo_sapiens/README)
|
|
65
|
+
"""
|
|
66
|
+
|
|
67
|
+
REQUIRED_COLUMNS = [
|
|
68
|
+
"seqname",
|
|
69
|
+
"source",
|
|
70
|
+
"feature",
|
|
71
|
+
"start",
|
|
72
|
+
"end",
|
|
73
|
+
"score",
|
|
74
|
+
"strand",
|
|
75
|
+
"frame",
|
|
76
|
+
"attribute",
|
|
77
|
+
]
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def parse_with_polars_lazy(
|
|
81
|
+
filepath_or_buffer,
|
|
82
|
+
split_attributes=True,
|
|
83
|
+
features=None,
|
|
84
|
+
fix_quotes_columns=["attribute"]):
|
|
85
|
+
# use a global string cache so that all strings get intern'd into
|
|
86
|
+
# a single numbering system
|
|
87
|
+
polars.enable_string_cache()
|
|
88
|
+
kwargs = dict(
|
|
89
|
+
has_header=False,
|
|
90
|
+
separator="\t",
|
|
91
|
+
comment_prefix="#",
|
|
92
|
+
null_values=".",
|
|
93
|
+
dtypes={
|
|
94
|
+
"seqname": polars.Categorical,
|
|
95
|
+
"source": polars.Categorical,
|
|
96
|
+
|
|
97
|
+
"start": polars.Int64,
|
|
98
|
+
"end": polars.Int64,
|
|
99
|
+
"score": polars.Float32,
|
|
100
|
+
|
|
101
|
+
"feature": polars.Categorical,
|
|
102
|
+
"strand": polars.Categorical,
|
|
103
|
+
"frame": polars.UInt32,
|
|
104
|
+
})
|
|
105
|
+
try:
|
|
106
|
+
if type(filepath_or_buffer) is StringIO:
|
|
107
|
+
df = polars.read_csv(
|
|
108
|
+
filepath_or_buffer,
|
|
109
|
+
new_columns=REQUIRED_COLUMNS,
|
|
110
|
+
**kwargs).lazy()
|
|
111
|
+
elif filepath_or_buffer.endswith(".gz") or filepath_or_buffer.endswith(".gzip"):
|
|
112
|
+
with gzip.open(filepath_or_buffer) as f:
|
|
113
|
+
df = polars.read_csv(
|
|
114
|
+
f,
|
|
115
|
+
new_columns=REQUIRED_COLUMNS,
|
|
116
|
+
**kwargs).lazy()
|
|
117
|
+
else:
|
|
118
|
+
df = polars.scan_csv(
|
|
119
|
+
filepath_or_buffer,
|
|
120
|
+
with_column_names=lambda cols: REQUIRED_COLUMNS,
|
|
121
|
+
**kwargs).lazy()
|
|
122
|
+
except polars.ShapeError:
|
|
123
|
+
raise ParsingError("Wrong number of columns")
|
|
124
|
+
|
|
125
|
+
df = df.with_columns([
|
|
126
|
+
polars.col("frame").fill_null(0),
|
|
127
|
+
polars.col("attribute").str.replace_all('"', "'")
|
|
128
|
+
])
|
|
129
|
+
|
|
130
|
+
for fix_quotes_column in fix_quotes_columns:
|
|
131
|
+
# Catch mistaken semicolons by replacing "xyz;" with "xyz"
|
|
132
|
+
# Required to do this since the Ensembl GTF for Ensembl
|
|
133
|
+
# release 78 has mistakes such as:
|
|
134
|
+
# gene_name = "PRAMEF6;" transcript_name = "PRAMEF6;-201"
|
|
135
|
+
df = df.with_columns([
|
|
136
|
+
polars.col(fix_quotes_column).str.replace(';\"', '\"').str.replace(";-", "-")
|
|
137
|
+
])
|
|
138
|
+
|
|
139
|
+
if features is not None:
|
|
140
|
+
features = sorted(set(features))
|
|
141
|
+
df = df.filter(polars.col("feature").is_in(features))
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
if split_attributes:
|
|
145
|
+
df = df.with_columns([
|
|
146
|
+
polars.col("attribute").str.split(";").alias("attribute_split")
|
|
147
|
+
])
|
|
148
|
+
return df
|
|
149
|
+
|
|
150
|
+
def parse_gtf(
|
|
151
|
+
filepath_or_buffer,
|
|
152
|
+
split_attributes=True,
|
|
153
|
+
features=None,
|
|
154
|
+
fix_quotes_columns=["attribute"]):
|
|
155
|
+
df_lazy = parse_with_polars_lazy(
|
|
156
|
+
filepath_or_buffer=filepath_or_buffer,
|
|
157
|
+
split_attributes=split_attributes,
|
|
158
|
+
features=features,
|
|
159
|
+
fix_quotes_columns=fix_quotes_columns)
|
|
160
|
+
return df_lazy.collect()
|
|
161
|
+
|
|
162
|
+
def parse_gtf_pandas(*args, **kwargs):
|
|
163
|
+
return parse_gtf(*args, **kwargs).to_pandas()
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def parse_gtf_and_expand_attributes(
|
|
167
|
+
filepath_or_buffer,
|
|
168
|
+
restrict_attribute_columns=None,
|
|
169
|
+
features=None):
|
|
170
|
+
"""
|
|
171
|
+
Parse lines into column->values dictionary and then expand
|
|
172
|
+
the 'attribute' column into multiple columns. This expansion happens
|
|
173
|
+
by replacing strings of semi-colon separated key-value values in the
|
|
174
|
+
'attribute' column with one column per distinct key, with a list of
|
|
175
|
+
values for each row (using None for rows where key didn't occur).
|
|
176
|
+
|
|
177
|
+
Parameters
|
|
178
|
+
----------
|
|
179
|
+
filepath_or_buffer : str or buffer object
|
|
180
|
+
|
|
181
|
+
chunksize : int
|
|
182
|
+
|
|
183
|
+
restrict_attribute_columns : list/set of str or None
|
|
184
|
+
If given, then only use these attribute columns.
|
|
185
|
+
|
|
186
|
+
features : set or None
|
|
187
|
+
Ignore entries which don't correspond to one of the supplied features
|
|
188
|
+
"""
|
|
189
|
+
df = parse_gtf(
|
|
190
|
+
filepath_or_buffer=filepath_or_buffer,
|
|
191
|
+
features=features,
|
|
192
|
+
split_attributes=True)
|
|
193
|
+
if type(restrict_attribute_columns) is str:
|
|
194
|
+
restrict_attribute_columns = {restrict_attribute_columns}
|
|
195
|
+
elif restrict_attribute_columns:
|
|
196
|
+
restrict_attribute_columns = set(restrict_attribute_columns)
|
|
197
|
+
df.drop_in_place("attribute")
|
|
198
|
+
attribute_pairs = df.drop_in_place("attribute_split")
|
|
199
|
+
return df.with_columns([
|
|
200
|
+
polars.Series(k, vs)
|
|
201
|
+
for (k, vs) in
|
|
202
|
+
expand_attribute_strings(attribute_pairs).items()
|
|
203
|
+
if restrict_attribute_columns is None or k in restrict_attribute_columns
|
|
204
|
+
])
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def read_gtf(
|
|
208
|
+
filepath_or_buffer,
|
|
209
|
+
expand_attribute_column=True,
|
|
210
|
+
infer_biotype_column=False,
|
|
211
|
+
column_converters={},
|
|
212
|
+
usecols=None,
|
|
213
|
+
features=None,
|
|
214
|
+
result_type='polars'):
|
|
215
|
+
"""
|
|
216
|
+
Parse a GTF into a dictionary mapping column names to sequences of values.
|
|
217
|
+
|
|
218
|
+
Parameters
|
|
219
|
+
----------
|
|
220
|
+
filepath_or_buffer : str or buffer object
|
|
221
|
+
Path to GTF file (may be gzip compressed) or buffer object
|
|
222
|
+
such as StringIO
|
|
223
|
+
|
|
224
|
+
expand_attribute_column : bool
|
|
225
|
+
Replace strings of semi-colon separated key-value values in the
|
|
226
|
+
'attribute' column with one column per distinct key, with a list of
|
|
227
|
+
values for each row (using None for rows where key didn't occur).
|
|
228
|
+
|
|
229
|
+
infer_biotype_column : bool
|
|
230
|
+
Due to the annoying ambiguity of the second GTF column across multiple
|
|
231
|
+
Ensembl releases, figure out if an older GTF's source column is actually
|
|
232
|
+
the gene_biotype or transcript_biotype.
|
|
233
|
+
|
|
234
|
+
column_converters : dict, optional
|
|
235
|
+
Dictionary mapping column names to conversion functions. Will replace
|
|
236
|
+
empty strings with None and otherwise passes them to given conversion
|
|
237
|
+
function.
|
|
238
|
+
|
|
239
|
+
usecols : list of str or None
|
|
240
|
+
Restrict which columns are loaded to the give set. If None, then
|
|
241
|
+
load all columns.
|
|
242
|
+
|
|
243
|
+
features : set of str or None
|
|
244
|
+
Drop rows which aren't one of the features in the supplied set
|
|
245
|
+
|
|
246
|
+
result_type : One of 'polars', 'pandas', or 'dict'
|
|
247
|
+
Default behavior is to return a Polars DataFrame, but will convert to
|
|
248
|
+
Pandas DataFrame or dictionary if specified.
|
|
249
|
+
"""
|
|
250
|
+
if type(filepath_or_buffer) is str and not exists(filepath_or_buffer):
|
|
251
|
+
raise ValueError("GTF file does not exist: %s" % filepath_or_buffer)
|
|
252
|
+
|
|
253
|
+
if expand_attribute_column:
|
|
254
|
+
result_df = parse_gtf_and_expand_attributes(
|
|
255
|
+
filepath_or_buffer,
|
|
256
|
+
restrict_attribute_columns=usecols,
|
|
257
|
+
features=features)
|
|
258
|
+
else:
|
|
259
|
+
result_df = parse_gtf(result_df, features=features)
|
|
260
|
+
|
|
261
|
+
result_df = result_df.with_columns(
|
|
262
|
+
[
|
|
263
|
+
polars.col(column_name).map_elements(lambda x: column_type(x) if len(x) > 0 else None)
|
|
264
|
+
for column_name, column_type in column_converters.items()
|
|
265
|
+
]
|
|
266
|
+
)
|
|
267
|
+
|
|
268
|
+
# Hackishly infer whether the values in the 'source' column of this GTF
|
|
269
|
+
# are actually representing a biotype by checking for the most common
|
|
270
|
+
# gene_biotype and transcript_biotype value 'protein_coding'
|
|
271
|
+
if infer_biotype_column:
|
|
272
|
+
unique_source_values = set(result_df["source"])
|
|
273
|
+
if "protein_coding" in unique_source_values:
|
|
274
|
+
column_names = set(result_df.columns)
|
|
275
|
+
# Disambiguate between the two biotypes by checking if
|
|
276
|
+
# gene_biotype is already present in another column. If it is,
|
|
277
|
+
# the 2nd column is the transcript_biotype (otherwise, it's the
|
|
278
|
+
# gene_biotype)
|
|
279
|
+
if "gene_biotype" not in column_names:
|
|
280
|
+
logging.info("Using column 'source' to replace missing 'gene_biotype'")
|
|
281
|
+
result_df = result_df.with_column(polars.col("source").alias("gene_biotype"))
|
|
282
|
+
if "transcript_biotype" not in column_names:
|
|
283
|
+
logging.info("Using column 'source' to replace missing 'transcript_biotype'")
|
|
284
|
+
result_df = result_df.with_column(polars.col("source").alias("transcript_biotype"))
|
|
285
|
+
|
|
286
|
+
if usecols is not None:
|
|
287
|
+
column_names = set(result_df.columns)
|
|
288
|
+
valid_columns = [c for c in usecols if c in column_names]
|
|
289
|
+
result_df = result_df.select(valid_columns)
|
|
290
|
+
|
|
291
|
+
if result_type == "pandas":
|
|
292
|
+
result = result_df.to_pandas()
|
|
293
|
+
elif result_type == "polars":
|
|
294
|
+
result = result_df
|
|
295
|
+
elif result_type == "dict":
|
|
296
|
+
result = result_df.to_dict()
|
|
297
|
+
return result
|
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: gtfparse
|
|
3
|
-
Version: 1.
|
|
4
|
-
Summary: GTF
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
3
|
+
Version: 2.1.0
|
|
4
|
+
Summary: Parsing library for extracting data frames of genomic features from GTF files
|
|
5
|
+
Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
|
|
6
|
+
Project-URL: Homepage, https://github.com/openvax/gtfparse
|
|
7
|
+
Project-URL: Bug Tracker, https://github.com/openvax/gtfparse
|
|
8
8
|
Classifier: Development Status :: 4 - Beta
|
|
9
9
|
Classifier: Environment :: Console
|
|
10
10
|
Classifier: Operating System :: OS Independent
|
|
@@ -12,8 +12,11 @@ Classifier: Intended Audience :: Science/Research
|
|
|
12
12
|
Classifier: License :: OSI Approved :: Apache Software License
|
|
13
13
|
Classifier: Programming Language :: Python
|
|
14
14
|
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
15
|
+
Requires-Python: >=3.7
|
|
15
16
|
Description-Content-Type: text/markdown
|
|
16
17
|
License-File: LICENSE
|
|
18
|
+
Requires-Dist: polars<0.21.0,>=0.20.2
|
|
19
|
+
Requires-Dist: pyarrow<14.1.0,>=14.0.2
|
|
17
20
|
|
|
18
21
|
[](https://travis-ci.org/openvax/gtfparse) [](https://coveralls.io/github/openvax/gtfparse?branch=master)
|
|
19
22
|
<a href="https://pypi.python.org/pypi/gtfparse/">
|
|
@@ -1,22 +1,22 @@
|
|
|
1
1
|
LICENSE
|
|
2
2
|
README.md
|
|
3
|
-
|
|
3
|
+
pyproject.toml
|
|
4
|
+
requirements.txt
|
|
4
5
|
gtfparse/__init__.py
|
|
5
6
|
gtfparse/attribute_parsing.py
|
|
6
7
|
gtfparse/create_missing_features.py
|
|
7
8
|
gtfparse/parsing_error.py
|
|
8
9
|
gtfparse/read_gtf.py
|
|
9
|
-
gtfparse/required_columns.py
|
|
10
|
-
gtfparse/version.py
|
|
11
10
|
gtfparse.egg-info/PKG-INFO
|
|
12
11
|
gtfparse.egg-info/SOURCES.txt
|
|
13
12
|
gtfparse.egg-info/dependency_links.txt
|
|
14
13
|
gtfparse.egg-info/requires.txt
|
|
15
14
|
gtfparse.egg-info/top_level.txt
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
15
|
+
gtfparse/../requirements.txt
|
|
16
|
+
tests/test_create_missing_features.py
|
|
17
|
+
tests/test_ensembl_gtf.py
|
|
18
|
+
tests/test_expand_attributes.py
|
|
19
|
+
tests/test_multiple_values_for_tag_attribute.py
|
|
20
|
+
tests/test_parse_gtf_lines.py
|
|
21
|
+
tests/test_read_stringtie_gtf.py
|
|
22
|
+
tests/test_refseq_gtf.py
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "gtfparse"
|
|
3
|
+
requires-python = ">=3.7"
|
|
4
|
+
authors = [ {name="Alex Rubinsteyn", email="alex.rubinsteyn@unc.edu" } ]
|
|
5
|
+
description = "Parsing library for extracting data frames of genomic features from GTF files"
|
|
6
|
+
classifiers = [
|
|
7
|
+
'Development Status :: 4 - Beta',
|
|
8
|
+
'Environment :: Console',
|
|
9
|
+
'Operating System :: OS Independent',
|
|
10
|
+
'Intended Audience :: Science/Research',
|
|
11
|
+
'License :: OSI Approved :: Apache Software License',
|
|
12
|
+
'Programming Language :: Python',
|
|
13
|
+
'Topic :: Scientific/Engineering :: Bio-Informatics',
|
|
14
|
+
]
|
|
15
|
+
readme = "README.md"
|
|
16
|
+
dynamic = ["version", "dependencies"]
|
|
17
|
+
|
|
18
|
+
[tool.setuptools.dynamic]
|
|
19
|
+
version = {attr = "gtfparse.__version__"}
|
|
20
|
+
dependencies = {file = ["requirements.txt"]}
|
|
21
|
+
|
|
22
|
+
[tool.setuptools]
|
|
23
|
+
packages = ["gtfparse"]
|
|
24
|
+
|
|
25
|
+
[project.urls]
|
|
26
|
+
"Homepage" = "https://github.com/openvax/gtfparse"
|
|
27
|
+
"Bug Tracker" = "https://github.com/openvax/gtfparse"
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
from gtfparse import create_missing_features, parse_gtf_and_expand_attributes
|
|
2
|
-
from
|
|
2
|
+
from io import StringIO
|
|
3
3
|
|
|
4
4
|
# two lines from the Ensembl 54 human GTF containing only a stop_codon and
|
|
5
5
|
# exon features, but from which gene and transcript information could be
|
|
@@ -19,6 +19,7 @@ GTF_TEXT = "\n".join([
|
|
|
19
19
|
|
|
20
20
|
|
|
21
21
|
GTF_DATAFRAME = parse_gtf_and_expand_attributes(StringIO(GTF_TEXT))
|
|
22
|
+
GTF_DATAFRAME = GTF_DATAFRAME.to_pandas()
|
|
22
23
|
|
|
23
24
|
def test_create_missing_features_identity():
|
|
24
25
|
df_should_be_same = create_missing_features(GTF_DATAFRAME, {})
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
from data import data_path
|
|
2
2
|
from gtfparse import read_gtf
|
|
3
|
-
|
|
3
|
+
|
|
4
4
|
|
|
5
5
|
ENSEMBL_GTF_PATH = data_path("ensembl_grch37.head.gtf")
|
|
6
6
|
|
|
@@ -18,7 +18,7 @@ EXPECTED_FEATURES = set([
|
|
|
18
18
|
def test_ensembl_gtf_columns():
|
|
19
19
|
df = read_gtf(ENSEMBL_GTF_PATH)
|
|
20
20
|
features = set(df["feature"])
|
|
21
|
-
|
|
21
|
+
assert features == EXPECTED_FEATURES
|
|
22
22
|
|
|
23
23
|
# first 1000 lines of GTF only contained these genes
|
|
24
24
|
EXPECTED_GENE_NAMES = {
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
from gtfparse import expand_attribute_strings
|
|
2
|
+
|
|
3
|
+
def test_attributes_in_quotes():
|
|
4
|
+
attributes = [
|
|
5
|
+
"gene_id \"ENSG001\"; tag \"bogotron\"; version \"1\";",
|
|
6
|
+
"gene_id \"ENSG002\"; tag \"wolfpuppy\"; version \"2\";"
|
|
7
|
+
]
|
|
8
|
+
parsed_dict = expand_attribute_strings(attributes, quote_char='"')
|
|
9
|
+
assert list(sorted(parsed_dict.keys())), ["gene_id", "tag", "version"]
|
|
10
|
+
assert parsed_dict["gene_id"] == ["ENSG001", "ENSG002"]
|
|
11
|
+
assert parsed_dict["tag"] == ["bogotron", "wolfpuppy"]
|
|
12
|
+
assert parsed_dict["version"] == ["1", "2"]
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def test_attributes_without_quotes():
|
|
16
|
+
attributes = [
|
|
17
|
+
"gene_id ENSG001; tag bogotron; version 1;",
|
|
18
|
+
"gene_id ENSG002; tag wolfpuppy; version 2"
|
|
19
|
+
]
|
|
20
|
+
parsed_dict = expand_attribute_strings(attributes)
|
|
21
|
+
assert list(sorted(parsed_dict.keys())) == ["gene_id", "tag", "version"]
|
|
22
|
+
assert parsed_dict["gene_id"] == ["ENSG001", "ENSG002"]
|
|
23
|
+
assert parsed_dict["tag"] == ["bogotron", "wolfpuppy"]
|
|
24
|
+
assert parsed_dict["version"] == ["1", "2"]
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def test_optional_attributes():
|
|
28
|
+
attributes = [
|
|
29
|
+
"gene_id ENSG001; sometimes-present bogotron;",
|
|
30
|
+
"gene_id ENSG002;",
|
|
31
|
+
"gene_id ENSG003; sometimes-present wolfpuppy;",
|
|
32
|
+
]
|
|
33
|
+
parsed_dict = expand_attribute_strings(attributes)
|
|
34
|
+
assert list(sorted(parsed_dict.keys())) == ["gene_id", "sometimes-present"]
|
|
35
|
+
assert parsed_dict["gene_id"] == ["ENSG001", "ENSG002", "ENSG003"]
|
|
36
|
+
assert parsed_dict["sometimes-present"] == ["bogotron", "", "wolfpuppy"]
|
|
@@ -1,6 +1,5 @@
|
|
|
1
|
-
from
|
|
1
|
+
from io import StringIO
|
|
2
2
|
from gtfparse import parse_gtf_and_expand_attributes
|
|
3
|
-
from nose.tools import eq_
|
|
4
3
|
|
|
5
4
|
# failing example from https://github.com/openvax/gtfparse/issues/2
|
|
6
5
|
GTF_TEXT = (
|
|
@@ -15,23 +14,22 @@ GTF_TEXT = (
|
|
|
15
14
|
def test_parse_tag_attributes():
|
|
16
15
|
parsed = parse_gtf_and_expand_attributes(StringIO(GTF_TEXT))
|
|
17
16
|
tag_column = parsed["tag"]
|
|
18
|
-
|
|
17
|
+
assert len(tag_column) == 1
|
|
19
18
|
tags = tag_column[0]
|
|
20
|
-
|
|
19
|
+
assert tags == 'cds_end_NF,mRNA_end_NF'
|
|
21
20
|
|
|
22
21
|
def test_parse_tag_attributes_with_usecols():
|
|
23
22
|
parsed = parse_gtf_and_expand_attributes(
|
|
24
23
|
StringIO(GTF_TEXT),
|
|
25
24
|
restrict_attribute_columns=["tag"])
|
|
26
25
|
tag_column = parsed["tag"]
|
|
27
|
-
|
|
26
|
+
assert len(tag_column) == 1
|
|
28
27
|
tags = tag_column[0]
|
|
29
|
-
|
|
28
|
+
assert tags == 'cds_end_NF,mRNA_end_NF'
|
|
30
29
|
|
|
31
30
|
def test_parse_tag_attributes_with_usecols_other_column():
|
|
32
31
|
parsed = parse_gtf_and_expand_attributes(
|
|
33
32
|
StringIO(GTF_TEXT),
|
|
34
33
|
restrict_attribute_columns=["exon_id"])
|
|
35
|
-
tag_column = parsed.get("tag")
|
|
36
34
|
|
|
37
|
-
assert
|
|
35
|
+
assert "tag" not in parsed, "Expected 'tag' to get dropped but got %s" % (parsed,)
|
|
@@ -1,12 +1,11 @@
|
|
|
1
|
-
|
|
2
|
-
from nose.tools import eq_, assert_raises
|
|
1
|
+
from pytest import raises
|
|
3
2
|
from gtfparse import (
|
|
4
3
|
parse_gtf,
|
|
5
4
|
parse_gtf_and_expand_attributes,
|
|
6
5
|
REQUIRED_COLUMNS,
|
|
7
6
|
ParsingError
|
|
8
7
|
)
|
|
9
|
-
from
|
|
8
|
+
from io import StringIO
|
|
10
9
|
|
|
11
10
|
gtf_text = """
|
|
12
11
|
# sample GTF data copied from:
|
|
@@ -16,7 +15,9 @@ gtf_text = """
|
|
|
16
15
|
"""
|
|
17
16
|
|
|
18
17
|
def test_parse_gtf_lines_with_expand_attributes():
|
|
19
|
-
|
|
18
|
+
df = parse_gtf_and_expand_attributes(StringIO(gtf_text))
|
|
19
|
+
|
|
20
|
+
|
|
20
21
|
# excluding 'attribute' column from required names
|
|
21
22
|
expected_columns = REQUIRED_COLUMNS[:8] + [
|
|
22
23
|
"gene_id",
|
|
@@ -28,40 +29,31 @@ def test_parse_gtf_lines_with_expand_attributes():
|
|
|
28
29
|
"transcript_source",
|
|
29
30
|
]
|
|
30
31
|
# convert to list since Py3's dictionary keys are a distinct collection type
|
|
31
|
-
|
|
32
|
-
|
|
32
|
+
assert list(df.columns) == expected_columns
|
|
33
|
+
assert list(df["seqname"]) == ["1", "1"]
|
|
33
34
|
# convert to list for comparison since numerical columns may be NumPy arrays
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
scores
|
|
38
|
-
assert
|
|
39
|
-
|
|
40
|
-
eq_(list(parsed_dict["transcript_id"]), ["", "ENST00000456328"])
|
|
35
|
+
assert list(df["start"]) == [11869, 11869]
|
|
36
|
+
assert list(df["end"]) == [14409, 14409]
|
|
37
|
+
|
|
38
|
+
assert df["score"].is_null().all(), "Unexpected scores: %s" % (df["score"],)
|
|
39
|
+
assert list(df["gene_id"]) == ["ENSG00000223972", "ENSG00000223972"]
|
|
40
|
+
assert list(df["transcript_id"]) == ["", "ENST00000456328"]
|
|
41
41
|
|
|
42
42
|
|
|
43
43
|
def test_parse_gtf_lines_without_expand_attributes():
|
|
44
|
-
|
|
44
|
+
df = parse_gtf(StringIO(gtf_text), split_attributes=False)
|
|
45
45
|
|
|
46
46
|
# convert to list since Py3's dictionary keys are a distinct collection type
|
|
47
|
-
|
|
48
|
-
|
|
47
|
+
assert list(df.columns) == REQUIRED_COLUMNS
|
|
48
|
+
assert list(df["seqname"]) == ["1", "1"]
|
|
49
49
|
# convert to list for comparison since numerical columns may be NumPy arrays
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
assert np.isnan(scores).all(), "Unexpected scores: %s" % scores
|
|
55
|
-
assert len(parsed_dict["attribute"]) == 2
|
|
56
|
-
|
|
57
|
-
def test_parse_gtf_lines_error_too_many_fields():
|
|
58
|
-
bad_gtf_text = gtf_text.replace(" ", "\t")
|
|
59
|
-
# pylint: disable=no-value-for-parameter
|
|
60
|
-
with assert_raises(ParsingError):
|
|
61
|
-
parse_gtf(StringIO(bad_gtf_text))
|
|
50
|
+
assert list(df["start"]) == [11869, 11869]
|
|
51
|
+
assert list(df["end"]) == [14409, 14409]
|
|
52
|
+
assert df["score"].is_null().all(), "Unexpected scores: %s" % (df["score"],)
|
|
53
|
+
assert len(df["attribute"]) == 2
|
|
62
54
|
|
|
63
55
|
def test_parse_gtf_lines_error_too_few_fields():
|
|
64
56
|
bad_gtf_text = gtf_text.replace("\t", " ")
|
|
65
57
|
# pylint: disable=no-value-for-parameter
|
|
66
|
-
with
|
|
58
|
+
with raises(ParsingError):
|
|
67
59
|
parse_gtf(StringIO(bad_gtf_text))
|
|
@@ -1,243 +0,0 @@
|
|
|
1
|
-
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
2
|
-
# you may not use this file except in compliance with the License.
|
|
3
|
-
# You may obtain a copy of the License at
|
|
4
|
-
#
|
|
5
|
-
# http://www.apache.org/licenses/LICENSE-2.0
|
|
6
|
-
#
|
|
7
|
-
# Unless required by applicable law or agreed to in writing, software
|
|
8
|
-
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
9
|
-
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
10
|
-
# See the License for the specific language governing permissions and
|
|
11
|
-
# limitations under the License.
|
|
12
|
-
|
|
13
|
-
import logging
|
|
14
|
-
from os.path import exists
|
|
15
|
-
|
|
16
|
-
from sys import intern
|
|
17
|
-
import numpy as np
|
|
18
|
-
import pandas as pd
|
|
19
|
-
|
|
20
|
-
from .attribute_parsing import expand_attribute_strings
|
|
21
|
-
from .parsing_error import ParsingError
|
|
22
|
-
from .required_columns import REQUIRED_COLUMNS
|
|
23
|
-
|
|
24
|
-
logging.basicConfig(level=logging.INFO)
|
|
25
|
-
logger = logging.getLogger(__name__)
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
def parse_gtf(
|
|
29
|
-
filepath_or_buffer,
|
|
30
|
-
chunksize=1024 * 1024,
|
|
31
|
-
features=None,
|
|
32
|
-
intern_columns=["seqname", "source", "strand", "frame"],
|
|
33
|
-
fix_quotes_columns=["attribute"]):
|
|
34
|
-
"""
|
|
35
|
-
Parameters
|
|
36
|
-
----------
|
|
37
|
-
|
|
38
|
-
filepath_or_buffer : str or buffer object
|
|
39
|
-
|
|
40
|
-
chunksize : int
|
|
41
|
-
|
|
42
|
-
features : set or None
|
|
43
|
-
Drop entries which aren't one of these features
|
|
44
|
-
|
|
45
|
-
intern_columns : list
|
|
46
|
-
These columns are short strings which should be interned
|
|
47
|
-
|
|
48
|
-
fix_quotes_columns : list
|
|
49
|
-
Most commonly the 'attribute' column which had broken quotes on
|
|
50
|
-
some Ensembl release GTF files.
|
|
51
|
-
"""
|
|
52
|
-
if features is not None:
|
|
53
|
-
features = set(features)
|
|
54
|
-
|
|
55
|
-
dataframes = []
|
|
56
|
-
|
|
57
|
-
def parse_frame(s):
|
|
58
|
-
if s == ".":
|
|
59
|
-
return 0
|
|
60
|
-
else:
|
|
61
|
-
return int(s)
|
|
62
|
-
|
|
63
|
-
# GTF columns:
|
|
64
|
-
# 1) seqname: str ("1", "X", "chrX", etc...)
|
|
65
|
-
# 2) source : str
|
|
66
|
-
# Different versions of GTF use second column as of:
|
|
67
|
-
# (a) gene biotype
|
|
68
|
-
# (b) transcript biotype
|
|
69
|
-
# (c) the annotation source
|
|
70
|
-
# See: https://www.biostars.org/p/120306/#120321
|
|
71
|
-
# 3) feature : str ("gene", "transcript", &c)
|
|
72
|
-
# 4) start : int
|
|
73
|
-
# 5) end : int
|
|
74
|
-
# 6) score : float or "."
|
|
75
|
-
# 7) strand : "+", "-", or "."
|
|
76
|
-
# 8) frame : 0, 1, 2 or "."
|
|
77
|
-
# 9) attribute : key-value pairs separated by semicolons
|
|
78
|
-
# (see more complete description in docstring at top of file)
|
|
79
|
-
|
|
80
|
-
chunk_iterator = pd.read_csv(
|
|
81
|
-
filepath_or_buffer,
|
|
82
|
-
sep="\t",
|
|
83
|
-
comment="#",
|
|
84
|
-
names=REQUIRED_COLUMNS,
|
|
85
|
-
skipinitialspace=True,
|
|
86
|
-
skip_blank_lines=True,
|
|
87
|
-
on_bad_lines="error",
|
|
88
|
-
chunksize=chunksize,
|
|
89
|
-
engine="c",
|
|
90
|
-
dtype={
|
|
91
|
-
"start": np.int64,
|
|
92
|
-
"end": np.int64,
|
|
93
|
-
"score": np.float32,
|
|
94
|
-
"seqname": str,
|
|
95
|
-
},
|
|
96
|
-
na_values=".",
|
|
97
|
-
converters={"frame": parse_frame})
|
|
98
|
-
dataframes = []
|
|
99
|
-
try:
|
|
100
|
-
for df in chunk_iterator:
|
|
101
|
-
for intern_column in intern_columns:
|
|
102
|
-
df[intern_column] = [intern(str(s)) for s in df[intern_column]]
|
|
103
|
-
|
|
104
|
-
# compare feature strings after interning
|
|
105
|
-
if features is not None:
|
|
106
|
-
df = df[df["feature"].isin(features)]
|
|
107
|
-
|
|
108
|
-
for fix_quotes_column in fix_quotes_columns:
|
|
109
|
-
# Catch mistaken semicolons by replacing "xyz;" with "xyz"
|
|
110
|
-
# Required to do this since the Ensembl GTF for Ensembl
|
|
111
|
-
# release 78 has mistakes such as:
|
|
112
|
-
# gene_name = "PRAMEF6;" transcript_name = "PRAMEF6;-201"
|
|
113
|
-
df[fix_quotes_column] = [
|
|
114
|
-
s.replace(';\"', '\"').replace(";-", "-")
|
|
115
|
-
for s in df[fix_quotes_column]
|
|
116
|
-
]
|
|
117
|
-
dataframes.append(df)
|
|
118
|
-
except Exception as e:
|
|
119
|
-
raise ParsingError(str(e))
|
|
120
|
-
df = pd.concat(dataframes)
|
|
121
|
-
return df
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
def parse_gtf_and_expand_attributes(
|
|
125
|
-
filepath_or_buffer,
|
|
126
|
-
chunksize=1024 * 1024,
|
|
127
|
-
restrict_attribute_columns=None,
|
|
128
|
-
features=None):
|
|
129
|
-
"""
|
|
130
|
-
Parse lines into column->values dictionary and then expand
|
|
131
|
-
the 'attribute' column into multiple columns. This expansion happens
|
|
132
|
-
by replacing strings of semi-colon separated key-value values in the
|
|
133
|
-
'attribute' column with one column per distinct key, with a list of
|
|
134
|
-
values for each row (using None for rows where key didn't occur).
|
|
135
|
-
|
|
136
|
-
Parameters
|
|
137
|
-
----------
|
|
138
|
-
filepath_or_buffer : str or buffer object
|
|
139
|
-
|
|
140
|
-
chunksize : int
|
|
141
|
-
|
|
142
|
-
restrict_attribute_columns : list/set of str or None
|
|
143
|
-
If given, then only usese attribute columns.
|
|
144
|
-
|
|
145
|
-
features : set or None
|
|
146
|
-
Ignore entries which don't correspond to one of the supplied features
|
|
147
|
-
"""
|
|
148
|
-
result = parse_gtf(
|
|
149
|
-
filepath_or_buffer,
|
|
150
|
-
chunksize=chunksize,
|
|
151
|
-
features=features)
|
|
152
|
-
attribute_values = result["attribute"]
|
|
153
|
-
del result["attribute"]
|
|
154
|
-
for column_name, values in expand_attribute_strings(
|
|
155
|
-
attribute_values, usecols=restrict_attribute_columns).items():
|
|
156
|
-
result[column_name] = values
|
|
157
|
-
return result
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
def read_gtf(
|
|
161
|
-
filepath_or_buffer,
|
|
162
|
-
expand_attribute_column=True,
|
|
163
|
-
infer_biotype_column=False,
|
|
164
|
-
column_converters={},
|
|
165
|
-
usecols=None,
|
|
166
|
-
features=None,
|
|
167
|
-
chunksize=1024 * 1024):
|
|
168
|
-
"""
|
|
169
|
-
Parse a GTF into a dictionary mapping column names to sequences of values.
|
|
170
|
-
|
|
171
|
-
Parameters
|
|
172
|
-
----------
|
|
173
|
-
filepath_or_buffer : str or buffer object
|
|
174
|
-
Path to GTF file (may be gzip compressed) or buffer object
|
|
175
|
-
such as StringIO
|
|
176
|
-
|
|
177
|
-
expand_attribute_column : bool
|
|
178
|
-
Replace strings of semi-colon separated key-value values in the
|
|
179
|
-
'attribute' column with one column per distinct key, with a list of
|
|
180
|
-
values for each row (using None for rows where key didn't occur).
|
|
181
|
-
|
|
182
|
-
infer_biotype_column : bool
|
|
183
|
-
Due to the annoying ambiguity of the second GTF column across multiple
|
|
184
|
-
Ensembl releases, figure out if an older GTF's source column is actually
|
|
185
|
-
the gene_biotype or transcript_biotype.
|
|
186
|
-
|
|
187
|
-
column_converters : dict, optional
|
|
188
|
-
Dictionary mapping column names to conversion functions. Will replace
|
|
189
|
-
empty strings with None and otherwise passes them to given conversion
|
|
190
|
-
function.
|
|
191
|
-
|
|
192
|
-
usecols : list of str or None
|
|
193
|
-
Restrict which columns are loaded to the give set. If None, then
|
|
194
|
-
load all columns.
|
|
195
|
-
|
|
196
|
-
features : set of str or None
|
|
197
|
-
Drop rows which aren't one of the features in the supplied set
|
|
198
|
-
|
|
199
|
-
chunksize : int
|
|
200
|
-
"""
|
|
201
|
-
if type(filepath_or_buffer) is str and not exists(filepath_or_buffer):
|
|
202
|
-
raise ValueError("GTF file does not exist: %s" % filepath_or_buffer)
|
|
203
|
-
|
|
204
|
-
if expand_attribute_column:
|
|
205
|
-
result_df = parse_gtf_and_expand_attributes(
|
|
206
|
-
filepath_or_buffer,
|
|
207
|
-
chunksize=chunksize,
|
|
208
|
-
restrict_attribute_columns=usecols,
|
|
209
|
-
features=features)
|
|
210
|
-
else:
|
|
211
|
-
result_df = parse_gtf(result_df, features=features)
|
|
212
|
-
|
|
213
|
-
for column_name, column_type in list(column_converters.items()):
|
|
214
|
-
result_df[column_name] = [
|
|
215
|
-
column_type(string_value) if len(string_value) > 0 else None
|
|
216
|
-
for string_value
|
|
217
|
-
in result_df[column_name]
|
|
218
|
-
]
|
|
219
|
-
|
|
220
|
-
# Hackishly infer whether the values in the 'source' column of this GTF
|
|
221
|
-
# are actually representing a biotype by checking for the most common
|
|
222
|
-
# gene_biotype and transcript_biotype value 'protein_coding'
|
|
223
|
-
if infer_biotype_column:
|
|
224
|
-
unique_source_values = set(result_df["source"])
|
|
225
|
-
if "protein_coding" in unique_source_values:
|
|
226
|
-
column_names = set(result_df.columns)
|
|
227
|
-
# Disambiguate between the two biotypes by checking if
|
|
228
|
-
# gene_biotype is already present in another column. If it is,
|
|
229
|
-
# the 2nd column is the transcript_biotype (otherwise, it's the
|
|
230
|
-
# gene_biotype)
|
|
231
|
-
if "gene_biotype" not in column_names:
|
|
232
|
-
logging.info("Using column 'source' to replace missing 'gene_biotype'")
|
|
233
|
-
result_df["gene_biotype"] = result_df["source"]
|
|
234
|
-
if "transcript_biotype" not in column_names:
|
|
235
|
-
logging.info("Using column 'source' to replace missing 'transcript_biotype'")
|
|
236
|
-
result_df["transcript_biotype"] = result_df["source"]
|
|
237
|
-
|
|
238
|
-
if usecols is not None:
|
|
239
|
-
column_names = set(result_df.columns)
|
|
240
|
-
valid_columns = [c for c in usecols if c in column_names]
|
|
241
|
-
result_df = result_df[valid_columns]
|
|
242
|
-
|
|
243
|
-
return result_df
|
|
@@ -1,62 +0,0 @@
|
|
|
1
|
-
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
2
|
-
# you may not use this file except in compliance with the License.
|
|
3
|
-
# You may obtain a copy of the License at
|
|
4
|
-
#
|
|
5
|
-
# http://www.apache.org/licenses/LICENSE-2.0
|
|
6
|
-
#
|
|
7
|
-
# Unless required by applicable law or agreed to in writing, software
|
|
8
|
-
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
9
|
-
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
10
|
-
# See the License for the specific language governing permissions and
|
|
11
|
-
# limitations under the License.
|
|
12
|
-
|
|
13
|
-
"""
|
|
14
|
-
Columns of a GTF file:
|
|
15
|
-
|
|
16
|
-
seqname - name of the chromosome or scaffold; chromosome names
|
|
17
|
-
without a 'chr' in Ensembl (but sometimes with a 'chr'
|
|
18
|
-
elsewhere)
|
|
19
|
-
source - name of the program that generated this feature, or
|
|
20
|
-
the data source (database or project name)
|
|
21
|
-
feature - feature type name.
|
|
22
|
-
Features currently in Ensembl GTFs:
|
|
23
|
-
gene
|
|
24
|
-
transcript
|
|
25
|
-
exon
|
|
26
|
-
CDS
|
|
27
|
-
Selenocysteine
|
|
28
|
-
start_codon
|
|
29
|
-
stop_codon
|
|
30
|
-
UTR
|
|
31
|
-
Older Ensembl releases may be missing some of these features.
|
|
32
|
-
start - start position of the feature, with sequence numbering
|
|
33
|
-
starting at 1.
|
|
34
|
-
end - end position of the feature, with sequence numbering
|
|
35
|
-
starting at 1.
|
|
36
|
-
score - a floating point value indiciating the score of a feature
|
|
37
|
-
strand - defined as + (forward) or - (reverse).
|
|
38
|
-
frame - one of '0', '1' or '2'. Frame indicates the number of base pairs
|
|
39
|
-
before you encounter a full codon. '0' indicates the feature
|
|
40
|
-
begins with a whole codon. '1' indicates there is an extra
|
|
41
|
-
base (the 3rd base of the prior codon) at the start of this feature.
|
|
42
|
-
'2' indicates there are two extra bases (2nd and 3rd base of the
|
|
43
|
-
prior exon) before the first codon. All values are given with
|
|
44
|
-
relation to the 5' end.
|
|
45
|
-
attribute - a semicolon-separated list of tag-value pairs (separated by a space),
|
|
46
|
-
providing additional information about each feature. A key can be
|
|
47
|
-
repeated multiple times.
|
|
48
|
-
|
|
49
|
-
(from ftp://ftp.ensembl.org/pub/release-75/gtf/homo_sapiens/README)
|
|
50
|
-
"""
|
|
51
|
-
|
|
52
|
-
REQUIRED_COLUMNS = [
|
|
53
|
-
"seqname",
|
|
54
|
-
"source",
|
|
55
|
-
"feature",
|
|
56
|
-
"start",
|
|
57
|
-
"end",
|
|
58
|
-
"score",
|
|
59
|
-
"strand",
|
|
60
|
-
"frame",
|
|
61
|
-
"attribute",
|
|
62
|
-
]
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
__version__ = "1.3.0"
|
gtfparse-1.3.0/setup.py
DELETED
|
@@ -1,61 +0,0 @@
|
|
|
1
|
-
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
2
|
-
# you may not use this file except in compliance with the License.
|
|
3
|
-
# You may obtain a copy of the License at
|
|
4
|
-
#
|
|
5
|
-
# http://www.apache.org/licenses/LICENSE-2.0
|
|
6
|
-
#
|
|
7
|
-
# Unless required by applicable law or agreed to in writing, software
|
|
8
|
-
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
9
|
-
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
10
|
-
# See the License for the specific language governing permissions and
|
|
11
|
-
# limitations under the License.
|
|
12
|
-
|
|
13
|
-
from __future__ import print_function
|
|
14
|
-
import os
|
|
15
|
-
import re
|
|
16
|
-
|
|
17
|
-
from setuptools import setup, find_packages
|
|
18
|
-
|
|
19
|
-
readme_filename = "README.md"
|
|
20
|
-
current_directory = os.path.dirname(__file__)
|
|
21
|
-
readme_path = os.path.join(current_directory, readme_filename)
|
|
22
|
-
|
|
23
|
-
readme_markdown = ""
|
|
24
|
-
try:
|
|
25
|
-
with open(readme_path, 'r') as f:
|
|
26
|
-
readme_markdown = f.read()
|
|
27
|
-
except Exception as e:
|
|
28
|
-
print(e)
|
|
29
|
-
print("Failed to open %s" % readme_path)
|
|
30
|
-
|
|
31
|
-
with open('gtfparse/version.py', 'r') as f:
|
|
32
|
-
version = re.search(
|
|
33
|
-
r'^__version__\s*=\s*[\'"]([^\'"]*)[\'"]',
|
|
34
|
-
f.read(),
|
|
35
|
-
re.MULTILINE).group(1)
|
|
36
|
-
|
|
37
|
-
if __name__ == '__main__':
|
|
38
|
-
setup(
|
|
39
|
-
name='gtfparse',
|
|
40
|
-
packages=find_packages(),
|
|
41
|
-
version=version,
|
|
42
|
-
description="GTF Parsing",
|
|
43
|
-
long_description=readme_markdown,
|
|
44
|
-
long_description_content_type='text/markdown',
|
|
45
|
-
url="https://github.com/openvax/gtfparse",
|
|
46
|
-
author="Alex Rubinsteyn",
|
|
47
|
-
license="http://www.apache.org/licenses/LICENSE-2.0.html",
|
|
48
|
-
classifiers=[
|
|
49
|
-
'Development Status :: 4 - Beta',
|
|
50
|
-
'Environment :: Console',
|
|
51
|
-
'Operating System :: OS Independent',
|
|
52
|
-
'Intended Audience :: Science/Research',
|
|
53
|
-
'License :: OSI Approved :: Apache Software License',
|
|
54
|
-
'Programming Language :: Python',
|
|
55
|
-
'Topic :: Scientific/Engineering :: Bio-Informatics',
|
|
56
|
-
],
|
|
57
|
-
install_requires=[
|
|
58
|
-
'numpy>=1.7',
|
|
59
|
-
'pandas>=0.15',
|
|
60
|
-
],
|
|
61
|
-
)
|
|
@@ -1,37 +0,0 @@
|
|
|
1
|
-
from gtfparse import expand_attribute_strings
|
|
2
|
-
from nose.tools import eq_
|
|
3
|
-
|
|
4
|
-
def test_attributes_in_quotes():
|
|
5
|
-
attributes = [
|
|
6
|
-
"gene_id \"ENSG001\"; tag \"bogotron\"; version \"1\";",
|
|
7
|
-
"gene_id \"ENSG002\"; tag \"wolfpuppy\"; version \"2\";"
|
|
8
|
-
]
|
|
9
|
-
parsed_dict = expand_attribute_strings(attributes)
|
|
10
|
-
eq_(list(sorted(parsed_dict.keys())), ["gene_id", "tag", "version"])
|
|
11
|
-
eq_(parsed_dict["gene_id"], ["ENSG001", "ENSG002"])
|
|
12
|
-
eq_(parsed_dict["tag"], ["bogotron", "wolfpuppy"])
|
|
13
|
-
eq_(parsed_dict["version"], ["1", "2"])
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
def test_attributes_without_quotes():
|
|
17
|
-
attributes = [
|
|
18
|
-
"gene_id ENSG001; tag bogotron; version 1;",
|
|
19
|
-
"gene_id ENSG002; tag wolfpuppy; version 2"
|
|
20
|
-
]
|
|
21
|
-
parsed_dict = expand_attribute_strings(attributes)
|
|
22
|
-
eq_(list(sorted(parsed_dict.keys())), ["gene_id", "tag", "version"])
|
|
23
|
-
eq_(parsed_dict["gene_id"], ["ENSG001", "ENSG002"])
|
|
24
|
-
eq_(parsed_dict["tag"], ["bogotron", "wolfpuppy"])
|
|
25
|
-
eq_(parsed_dict["version"], ["1", "2"])
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
def test_optional_attributes():
|
|
29
|
-
attributes = [
|
|
30
|
-
"gene_id ENSG001; sometimes-present bogotron;",
|
|
31
|
-
"gene_id ENSG002;",
|
|
32
|
-
"gene_id ENSG003; sometimes-present wolfpuppy;",
|
|
33
|
-
]
|
|
34
|
-
parsed_dict = expand_attribute_strings(attributes)
|
|
35
|
-
eq_(list(sorted(parsed_dict.keys())), ["gene_id", "sometimes-present"])
|
|
36
|
-
eq_(parsed_dict["gene_id"], ["ENSG001", "ENSG002", "ENSG003"])
|
|
37
|
-
eq_(parsed_dict["sometimes-present"], ["bogotron", "", "wolfpuppy"])
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|