gtfparse 2.6.0__tar.gz → 2.6.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gtfparse-2.6.0 → gtfparse-2.6.3}/PKG-INFO +13 -3
- {gtfparse-2.6.0 → gtfparse-2.6.3}/gtfparse/__init__.py +10 -11
- {gtfparse-2.6.0 → gtfparse-2.6.3}/gtfparse/attribute_parsing.py +4 -13
- {gtfparse-2.6.0 → gtfparse-2.6.3}/gtfparse/create_missing_features.py +14 -22
- {gtfparse-2.6.0 → gtfparse-2.6.3}/gtfparse/parsing_error.py +1 -0
- {gtfparse-2.6.0 → gtfparse-2.6.3}/gtfparse/read_gtf.py +65 -77
- {gtfparse-2.6.0 → gtfparse-2.6.3}/gtfparse.egg-info/PKG-INFO +13 -3
- {gtfparse-2.6.0 → gtfparse-2.6.3}/gtfparse.egg-info/requires.txt +6 -0
- gtfparse-2.6.3/pyproject.toml +104 -0
- gtfparse-2.6.3/tests/test_create_missing_features.py +85 -0
- gtfparse-2.6.3/tests/test_ensembl_gtf.py +147 -0
- {gtfparse-2.6.0 → gtfparse-2.6.3}/tests/test_expand_attributes.py +9 -8
- {gtfparse-2.6.0 → gtfparse-2.6.3}/tests/test_multiple_values_for_tag_attribute.py +9 -7
- {gtfparse-2.6.0 → gtfparse-2.6.3}/tests/test_parse_gtf_lines.py +9 -10
- {gtfparse-2.6.0 → gtfparse-2.6.3}/tests/test_read_stringtie_gtf.py +18 -12
- {gtfparse-2.6.0 → gtfparse-2.6.3}/tests/test_refseq_gtf.py +10 -2
- gtfparse-2.6.0/pyproject.toml +0 -27
- gtfparse-2.6.0/tests/test_create_missing_features.py +0 -86
- gtfparse-2.6.0/tests/test_ensembl_gtf.py +0 -81
- {gtfparse-2.6.0 → gtfparse-2.6.3}/LICENSE +0 -0
- {gtfparse-2.6.0 → gtfparse-2.6.3}/README.md +0 -0
- {gtfparse-2.6.0 → gtfparse-2.6.3}/gtfparse.egg-info/SOURCES.txt +0 -0
- {gtfparse-2.6.0 → gtfparse-2.6.3}/gtfparse.egg-info/dependency_links.txt +0 -0
- {gtfparse-2.6.0 → gtfparse-2.6.3}/gtfparse.egg-info/top_level.txt +0 -0
- {gtfparse-2.6.0 → gtfparse-2.6.3}/requirements.txt +0 -0
- {gtfparse-2.6.0 → gtfparse-2.6.3}/setup.cfg +0 -0
|
@@ -1,23 +1,33 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gtfparse
|
|
3
|
-
Version: 2.6.
|
|
3
|
+
Version: 2.6.3
|
|
4
4
|
Summary: Parsing library for extracting data frames of genomic features from GTF files
|
|
5
5
|
Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
|
|
6
6
|
Project-URL: Homepage, https://github.com/openvax/gtfparse
|
|
7
|
-
Project-URL: Bug Tracker, https://github.com/openvax/gtfparse
|
|
7
|
+
Project-URL: Bug Tracker, https://github.com/openvax/gtfparse/issues
|
|
8
8
|
Classifier: Development Status :: 4 - Beta
|
|
9
9
|
Classifier: Environment :: Console
|
|
10
10
|
Classifier: Operating System :: OS Independent
|
|
11
11
|
Classifier: Intended Audience :: Science/Research
|
|
12
12
|
Classifier: License :: OSI Approved :: Apache Software License
|
|
13
13
|
Classifier: Programming Language :: Python
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
14
19
|
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
15
|
-
Requires-Python: >=3.
|
|
20
|
+
Requires-Python: >=3.9
|
|
16
21
|
Description-Content-Type: text/markdown
|
|
17
22
|
License-File: LICENSE
|
|
18
23
|
Requires-Dist: polars>=0.20.2
|
|
19
24
|
Requires-Dist: pyarrow>=18.0.0
|
|
20
25
|
Requires-Dist: pandas>=2.1.0
|
|
26
|
+
Provides-Extra: dev
|
|
27
|
+
Requires-Dist: pytest; extra == "dev"
|
|
28
|
+
Requires-Dist: pytest-cov; extra == "dev"
|
|
29
|
+
Requires-Dist: ruff; extra == "dev"
|
|
30
|
+
Requires-Dist: coveralls; extra == "dev"
|
|
21
31
|
Dynamic: license-file
|
|
22
32
|
|
|
23
33
|
[](https://github.com/openvax/gtfparse/actions/workflows/tests.yml)
|
|
@@ -14,24 +14,23 @@ from .attribute_parsing import expand_attribute_strings
|
|
|
14
14
|
from .create_missing_features import create_missing_features
|
|
15
15
|
from .parsing_error import ParsingError
|
|
16
16
|
from .read_gtf import (
|
|
17
|
-
read_gtf,
|
|
18
|
-
parse_gtf,
|
|
19
|
-
parse_gtf_pandas,
|
|
20
|
-
parse_gtf_and_expand_attributes,
|
|
21
17
|
REQUIRED_COLUMNS,
|
|
18
|
+
parse_gtf,
|
|
19
|
+
parse_gtf_and_expand_attributes,
|
|
20
|
+
parse_gtf_pandas,
|
|
21
|
+
read_gtf,
|
|
22
22
|
)
|
|
23
23
|
|
|
24
|
-
__version__ = "2.6.
|
|
24
|
+
__version__ = "2.6.3"
|
|
25
25
|
|
|
26
26
|
__all__ = [
|
|
27
|
-
"__version__",
|
|
28
|
-
"expand_attribute_strings",
|
|
29
|
-
"create_missing_features",
|
|
30
|
-
|
|
31
|
-
"parse_gtf_and_expand_attributes",
|
|
32
27
|
"REQUIRED_COLUMNS",
|
|
33
28
|
"ParsingError",
|
|
34
|
-
"
|
|
29
|
+
"__version__",
|
|
30
|
+
"create_missing_features",
|
|
31
|
+
"expand_attribute_strings",
|
|
35
32
|
"parse_gtf",
|
|
33
|
+
"parse_gtf_and_expand_attributes",
|
|
36
34
|
"parse_gtf_pandas",
|
|
35
|
+
"read_gtf",
|
|
37
36
|
]
|
|
@@ -18,12 +18,7 @@ logging.basicConfig(level=logging.INFO)
|
|
|
18
18
|
logger = logging.getLogger(__name__)
|
|
19
19
|
|
|
20
20
|
|
|
21
|
-
|
|
22
|
-
def expand_attribute_strings(
|
|
23
|
-
attribute_strings,
|
|
24
|
-
quote_char="'",
|
|
25
|
-
missing_value="",
|
|
26
|
-
usecols=None):
|
|
21
|
+
def expand_attribute_strings(attribute_strings, quote_char="'", missing_value="", usecols=None):
|
|
27
22
|
"""
|
|
28
23
|
The last column of GTF has a variable number of key value pairs
|
|
29
24
|
of the format: "key1 value1; key2 value2;"
|
|
@@ -66,7 +61,7 @@ def expand_attribute_strings(
|
|
|
66
61
|
# and pair of try/except blocks in the loop.
|
|
67
62
|
column_interned_strings = {}
|
|
68
63
|
|
|
69
|
-
for
|
|
64
|
+
for i, kv_strings in enumerate(attribute_strings):
|
|
70
65
|
if type(kv_strings) is str:
|
|
71
66
|
kv_strings = kv_strings.split(";")
|
|
72
67
|
for kv in kv_strings:
|
|
@@ -92,7 +87,7 @@ def expand_attribute_strings(
|
|
|
92
87
|
|
|
93
88
|
if value[0] == quote_char:
|
|
94
89
|
value = value.replace(quote_char, "")
|
|
95
|
-
|
|
90
|
+
|
|
96
91
|
try:
|
|
97
92
|
column = extra_columns[column_name]
|
|
98
93
|
# if an attribute is used repeatedly then
|
|
@@ -108,9 +103,5 @@ def expand_attribute_strings(
|
|
|
108
103
|
extra_columns[column_name] = column
|
|
109
104
|
column_order.append(column_name)
|
|
110
105
|
|
|
111
|
-
|
|
112
|
-
|
|
113
106
|
logging.info("Extracted GTF attributes: %s" % column_order)
|
|
114
|
-
return OrderedDict(
|
|
115
|
-
(column_name, extra_columns[column_name])
|
|
116
|
-
for column_name in column_order)
|
|
107
|
+
return OrderedDict((column_name, extra_columns[column_name]) for column_name in column_order)
|
|
@@ -19,11 +19,7 @@ logging.basicConfig(level=logging.INFO)
|
|
|
19
19
|
logger = logging.getLogger(__name__)
|
|
20
20
|
|
|
21
21
|
|
|
22
|
-
def create_missing_features(
|
|
23
|
-
dataframe,
|
|
24
|
-
unique_keys={},
|
|
25
|
-
extra_columns={},
|
|
26
|
-
missing_value=None):
|
|
22
|
+
def create_missing_features(dataframe, unique_keys={}, extra_columns={}, missing_value=None):
|
|
27
23
|
"""
|
|
28
24
|
Helper function used to construct a missing feature such as 'transcript'
|
|
29
25
|
or 'gene'. Some GTF files only have 'exon' and 'CDS' entries, but have
|
|
@@ -49,29 +45,25 @@ def create_missing_features(
|
|
|
49
45
|
missing_value : any
|
|
50
46
|
Which value to fill in for columns that we don't infer values for.
|
|
51
47
|
|
|
52
|
-
Returns original dataframe (converted to Pandas if necessary) along with all
|
|
48
|
+
Returns original dataframe (converted to Pandas if necessary) along with all
|
|
53
49
|
extra rows created for missing features.
|
|
54
50
|
"""
|
|
55
51
|
if hasattr(dataframe, "to_pandas"):
|
|
56
52
|
dataframe = dataframe.to_pandas()
|
|
57
|
-
|
|
53
|
+
|
|
58
54
|
extra_dataframes = []
|
|
59
55
|
|
|
60
56
|
existing_features = set(dataframe["feature"])
|
|
61
57
|
existing_columns = set(dataframe.columns)
|
|
62
|
-
|
|
63
|
-
for
|
|
64
|
-
|
|
58
|
+
|
|
59
|
+
for feature_name, groupby_key in unique_keys.items():
|
|
65
60
|
if feature_name in existing_features:
|
|
66
|
-
logging.info(
|
|
67
|
-
"Feature '%s' already exists in GTF data" % feature_name)
|
|
61
|
+
logging.info("Feature '%s' already exists in GTF data" % feature_name)
|
|
68
62
|
continue
|
|
69
63
|
logging.info("Creating rows for missing feature '%s'" % feature_name)
|
|
70
64
|
|
|
71
65
|
# don't include rows where the groupby key was missing
|
|
72
|
-
missing = pd.Series([
|
|
73
|
-
x is None or x == ""
|
|
74
|
-
for x in dataframe[groupby_key]])
|
|
66
|
+
missing = pd.Series([x is None or x == "" for x in dataframe[groupby_key]])
|
|
75
67
|
not_missing = ~missing
|
|
76
68
|
row_groups = dataframe[not_missing].groupby(groupby_key)
|
|
77
69
|
|
|
@@ -79,10 +71,9 @@ def create_missing_features(
|
|
|
79
71
|
# other columns may or may not be uniquely defined. Start off by
|
|
80
72
|
# assuming the values for every column are missing and fill them in
|
|
81
73
|
# where possible.
|
|
82
|
-
feature_values = OrderedDict(
|
|
83
|
-
(column_name, [missing_value] * row_groups.ngroups)
|
|
84
|
-
|
|
85
|
-
])
|
|
74
|
+
feature_values = OrderedDict(
|
|
75
|
+
[(column_name, [missing_value] * row_groups.ngroups) for column_name in dataframe]
|
|
76
|
+
)
|
|
86
77
|
|
|
87
78
|
# User specifies which non-required columns should we try to infer
|
|
88
79
|
# values for
|
|
@@ -111,8 +102,9 @@ def create_missing_features(
|
|
|
111
102
|
for column_name in feature_columns:
|
|
112
103
|
if column_name not in existing_columns:
|
|
113
104
|
raise ValueError(
|
|
114
|
-
"Column '%s' does not exist in GTF, columns = %s"
|
|
115
|
-
|
|
105
|
+
"Column '%s' does not exist in GTF, columns = %s"
|
|
106
|
+
% (column_name, existing_columns)
|
|
107
|
+
)
|
|
116
108
|
|
|
117
109
|
# expect that all entries related to a reconstructed feature
|
|
118
110
|
# are related and are thus within the same interval of
|
|
@@ -121,4 +113,4 @@ def create_missing_features(
|
|
|
121
113
|
if len(unique_values) == 1:
|
|
122
114
|
feature_values[column_name][i] = unique_values[0]
|
|
123
115
|
extra_dataframes.append(pd.DataFrame(feature_values))
|
|
124
|
-
return pd.concat([dataframe
|
|
116
|
+
return pd.concat([dataframe, *extra_dataframes], ignore_index=True)
|
|
@@ -13,12 +13,11 @@
|
|
|
13
13
|
import logging
|
|
14
14
|
from os.path import exists
|
|
15
15
|
|
|
16
|
-
import polars
|
|
16
|
+
import polars
|
|
17
17
|
|
|
18
18
|
from .attribute_parsing import expand_attribute_strings
|
|
19
19
|
from .parsing_error import ParsingError
|
|
20
20
|
|
|
21
|
-
|
|
22
21
|
logging.basicConfig(level=logging.INFO)
|
|
23
22
|
logger = logging.getLogger(__name__)
|
|
24
23
|
|
|
@@ -76,88 +75,79 @@ REQUIRED_COLUMNS = [
|
|
|
76
75
|
|
|
77
76
|
|
|
78
77
|
DEFAULT_COLUMN_DTYPES = {
|
|
79
|
-
"seqname": polars.Categorical,
|
|
80
|
-
"source": polars.Categorical,
|
|
81
|
-
|
|
78
|
+
"seqname": polars.Categorical,
|
|
79
|
+
"source": polars.Categorical,
|
|
82
80
|
"start": polars.Int64,
|
|
83
81
|
"end": polars.Int64,
|
|
84
82
|
"score": polars.Float32,
|
|
85
|
-
|
|
86
|
-
"
|
|
87
|
-
"strand": polars.Categorical,
|
|
83
|
+
"feature": polars.Categorical,
|
|
84
|
+
"strand": polars.Categorical,
|
|
88
85
|
"frame": polars.UInt32,
|
|
89
86
|
}
|
|
90
87
|
|
|
88
|
+
|
|
91
89
|
def parse_with_polars_lazy(
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
features=None,
|
|
95
|
-
fix_quotes_columns=["attribute"]):
|
|
90
|
+
filepath_or_buffer, split_attributes=True, features=None, fix_quotes_columns=["attribute"]
|
|
91
|
+
):
|
|
96
92
|
# use a global string cache so that all strings get intern'd into
|
|
97
93
|
# a single numbering system
|
|
98
94
|
polars.enable_string_cache()
|
|
99
|
-
kwargs =
|
|
100
|
-
has_header
|
|
101
|
-
separator
|
|
102
|
-
comment_prefix
|
|
103
|
-
null_values
|
|
104
|
-
schema_overrides
|
|
95
|
+
kwargs = {
|
|
96
|
+
"has_header": False,
|
|
97
|
+
"separator": "\t",
|
|
98
|
+
"comment_prefix": "#",
|
|
99
|
+
"null_values": ".",
|
|
100
|
+
"schema_overrides": DEFAULT_COLUMN_DTYPES,
|
|
101
|
+
}
|
|
105
102
|
try:
|
|
106
|
-
df = polars.read_csv(
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
**kwargs).lazy()
|
|
110
|
-
except polars.exceptions.ShapeError:
|
|
111
|
-
raise ParsingError("Wrong number of columns")
|
|
103
|
+
df = polars.read_csv(filepath_or_buffer, new_columns=REQUIRED_COLUMNS, **kwargs).lazy()
|
|
104
|
+
except polars.exceptions.ShapeError as err:
|
|
105
|
+
raise ParsingError("Wrong number of columns") from err
|
|
112
106
|
|
|
113
107
|
# Drop empty lines that may appear as all-null rows
|
|
114
108
|
df = df.filter(polars.col("seqname").is_not_null())
|
|
115
109
|
|
|
116
|
-
df = df.with_columns(
|
|
117
|
-
polars.col("frame").fill_null(0),
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
110
|
+
df = df.with_columns(
|
|
111
|
+
[polars.col("frame").fill_null(0), polars.col("attribute").str.replace_all('"', "'")]
|
|
112
|
+
)
|
|
113
|
+
|
|
121
114
|
for fix_quotes_column in fix_quotes_columns:
|
|
122
115
|
# Catch mistaken semicolons by replacing "xyz;" with "xyz"
|
|
123
116
|
# Required to do this since the Ensembl GTF for Ensembl
|
|
124
117
|
# release 78 has mistakes such as:
|
|
125
118
|
# gene_name = "PRAMEF6;" transcript_name = "PRAMEF6;-201"
|
|
126
|
-
df = df.with_columns(
|
|
127
|
-
polars.col(fix_quotes_column).str.replace('
|
|
128
|
-
|
|
119
|
+
df = df.with_columns(
|
|
120
|
+
[polars.col(fix_quotes_column).str.replace(';"', '"').str.replace(";-", "-")]
|
|
121
|
+
)
|
|
129
122
|
|
|
130
123
|
if features is not None:
|
|
131
124
|
features = sorted(set(features))
|
|
132
125
|
df = df.filter(polars.col("feature").is_in(features))
|
|
133
126
|
|
|
134
|
-
|
|
135
127
|
if split_attributes:
|
|
136
|
-
df = df.with_columns([
|
|
137
|
-
polars.col("attribute").str.split(";").alias("attribute_split")
|
|
138
|
-
])
|
|
128
|
+
df = df.with_columns([polars.col("attribute").str.split(";").alias("attribute_split")])
|
|
139
129
|
return df
|
|
140
130
|
|
|
131
|
+
|
|
141
132
|
def parse_gtf(
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
features=None,
|
|
145
|
-
fix_quotes_columns=["attribute"]):
|
|
133
|
+
filepath_or_buffer, split_attributes=True, features=None, fix_quotes_columns=["attribute"]
|
|
134
|
+
):
|
|
146
135
|
df_lazy = parse_with_polars_lazy(
|
|
147
136
|
filepath_or_buffer=filepath_or_buffer,
|
|
148
137
|
split_attributes=split_attributes,
|
|
149
138
|
features=features,
|
|
150
|
-
fix_quotes_columns=fix_quotes_columns
|
|
139
|
+
fix_quotes_columns=fix_quotes_columns,
|
|
140
|
+
)
|
|
151
141
|
return df_lazy.collect()
|
|
152
142
|
|
|
143
|
+
|
|
153
144
|
def parse_gtf_pandas(*args, **kwargs):
|
|
154
145
|
return parse_gtf(*args, **kwargs).to_pandas()
|
|
155
146
|
|
|
156
|
-
|
|
147
|
+
|
|
157
148
|
def parse_gtf_and_expand_attributes(
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
features=None):
|
|
149
|
+
filepath_or_buffer, restrict_attribute_columns=None, features=None
|
|
150
|
+
):
|
|
161
151
|
"""
|
|
162
152
|
Parse lines into column->values dictionary and then expand
|
|
163
153
|
the 'attribute' column into multiple columns. This expansion happens
|
|
@@ -177,33 +167,32 @@ def parse_gtf_and_expand_attributes(
|
|
|
177
167
|
features : set or None
|
|
178
168
|
Ignore entries which don't correspond to one of the supplied features
|
|
179
169
|
"""
|
|
180
|
-
df = parse_gtf(
|
|
181
|
-
filepath_or_buffer=filepath_or_buffer,
|
|
182
|
-
features=features,
|
|
183
|
-
split_attributes=True)
|
|
170
|
+
df = parse_gtf(filepath_or_buffer=filepath_or_buffer, features=features, split_attributes=True)
|
|
184
171
|
if type(restrict_attribute_columns) is str:
|
|
185
172
|
restrict_attribute_columns = {restrict_attribute_columns}
|
|
186
173
|
elif restrict_attribute_columns:
|
|
187
174
|
restrict_attribute_columns = set(restrict_attribute_columns)
|
|
188
175
|
df.drop_in_place("attribute")
|
|
189
176
|
attribute_pairs = df.drop_in_place("attribute_split")
|
|
190
|
-
return df.with_columns(
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
177
|
+
return df.with_columns(
|
|
178
|
+
[
|
|
179
|
+
polars.Series(k, vs)
|
|
180
|
+
for (k, vs) in expand_attribute_strings(attribute_pairs).items()
|
|
181
|
+
if restrict_attribute_columns is None or k in restrict_attribute_columns
|
|
182
|
+
]
|
|
183
|
+
)
|
|
184
|
+
|
|
197
185
|
|
|
198
186
|
def read_gtf(
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
187
|
+
filepath_or_buffer,
|
|
188
|
+
expand_attribute_column=True,
|
|
189
|
+
infer_biotype_column=False,
|
|
190
|
+
column_converters={},
|
|
191
|
+
column_cast_types={},
|
|
192
|
+
usecols=None,
|
|
193
|
+
features=None,
|
|
194
|
+
result_type="polars",
|
|
195
|
+
):
|
|
207
196
|
"""
|
|
208
197
|
Parse a GTF into a dictionary mapping column names to sequences of values.
|
|
209
198
|
|
|
@@ -231,7 +220,7 @@ def read_gtf(
|
|
|
231
220
|
column_cast_types : dict, optional
|
|
232
221
|
Dictionary mapping column names to dtypes. Will cast columns to given
|
|
233
222
|
Polars types.
|
|
234
|
-
|
|
223
|
+
|
|
235
224
|
usecols : list of str or None
|
|
236
225
|
Restrict which columns are loaded to the give set. If None, then
|
|
237
226
|
load all columns.
|
|
@@ -240,7 +229,7 @@ def read_gtf(
|
|
|
240
229
|
Drop rows which aren't one of the features in the supplied set
|
|
241
230
|
|
|
242
231
|
result_type : One of 'polars', 'pandas', or 'dict'
|
|
243
|
-
Default behavior is to return a Polars DataFrame, but will convert to
|
|
232
|
+
Default behavior is to return a Polars DataFrame, but will convert to
|
|
244
233
|
Pandas DataFrame or dictionary if specified.
|
|
245
234
|
"""
|
|
246
235
|
if type(filepath_or_buffer) is str and not exists(filepath_or_buffer):
|
|
@@ -248,9 +237,8 @@ def read_gtf(
|
|
|
248
237
|
|
|
249
238
|
if expand_attribute_column:
|
|
250
239
|
result_df = parse_gtf_and_expand_attributes(
|
|
251
|
-
filepath_or_buffer,
|
|
252
|
-
|
|
253
|
-
features=features)
|
|
240
|
+
filepath_or_buffer, restrict_attribute_columns=usecols, features=features
|
|
241
|
+
)
|
|
254
242
|
else:
|
|
255
243
|
result_df = parse_gtf(result_df, features=features)
|
|
256
244
|
|
|
@@ -259,26 +247,26 @@ def read_gtf(
|
|
|
259
247
|
# and are generally insane to chase down
|
|
260
248
|
result_df = result_df.to_pandas()
|
|
261
249
|
if column_converters or column_cast_types:
|
|
250
|
+
|
|
262
251
|
def wrap_to_always_accept_none(f):
|
|
263
252
|
def wrapped_fn(x):
|
|
264
253
|
if x is None or x == "":
|
|
265
254
|
return None
|
|
266
255
|
else:
|
|
267
256
|
return f(x)
|
|
257
|
+
|
|
268
258
|
return wrapped_fn
|
|
269
|
-
|
|
259
|
+
|
|
270
260
|
column_names = set(column_converters.keys()).union(column_cast_types.keys())
|
|
271
261
|
for column_name in column_names:
|
|
272
|
-
|
|
273
262
|
if column_name in column_converters:
|
|
274
|
-
column_fn = wrap_to_always_accept_none(
|
|
275
|
-
column_converters[column_name])
|
|
263
|
+
column_fn = wrap_to_always_accept_none(column_converters[column_name])
|
|
276
264
|
result_df[column_name] = result_df[column_name].apply(column_fn)
|
|
277
265
|
|
|
278
266
|
if column_name in column_cast_types:
|
|
279
267
|
column_type = column_cast_types[column_name]
|
|
280
268
|
result_df[column_name] = result_df[column_name].astype(column_type)
|
|
281
|
-
|
|
269
|
+
|
|
282
270
|
# Hackishly infer whether the values in the 'source' column of this GTF
|
|
283
271
|
# are actually representing a biotype by checking for the most common
|
|
284
272
|
# gene_biotype and transcript_biotype value 'protein_coding'
|
|
@@ -292,11 +280,11 @@ def read_gtf(
|
|
|
292
280
|
# gene_biotype)
|
|
293
281
|
if "gene_biotype" not in column_names:
|
|
294
282
|
logging.info("Using column 'source' to replace missing 'gene_biotype'")
|
|
295
|
-
result_df[
|
|
283
|
+
result_df["gene_biotype"] = result_df["source"]
|
|
296
284
|
if "transcript_biotype" not in column_names:
|
|
297
285
|
logging.info("Using column 'source' to replace missing 'transcript_biotype'")
|
|
298
|
-
result_df[
|
|
299
|
-
|
|
286
|
+
result_df["transcript_biotype"] = result_df["source"]
|
|
287
|
+
|
|
300
288
|
if usecols is not None:
|
|
301
289
|
column_names = set(result_df.columns)
|
|
302
290
|
valid_columns = [c for c in usecols if c in column_names]
|
|
@@ -1,23 +1,33 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gtfparse
|
|
3
|
-
Version: 2.6.
|
|
3
|
+
Version: 2.6.3
|
|
4
4
|
Summary: Parsing library for extracting data frames of genomic features from GTF files
|
|
5
5
|
Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
|
|
6
6
|
Project-URL: Homepage, https://github.com/openvax/gtfparse
|
|
7
|
-
Project-URL: Bug Tracker, https://github.com/openvax/gtfparse
|
|
7
|
+
Project-URL: Bug Tracker, https://github.com/openvax/gtfparse/issues
|
|
8
8
|
Classifier: Development Status :: 4 - Beta
|
|
9
9
|
Classifier: Environment :: Console
|
|
10
10
|
Classifier: Operating System :: OS Independent
|
|
11
11
|
Classifier: Intended Audience :: Science/Research
|
|
12
12
|
Classifier: License :: OSI Approved :: Apache Software License
|
|
13
13
|
Classifier: Programming Language :: Python
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
14
19
|
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
15
|
-
Requires-Python: >=3.
|
|
20
|
+
Requires-Python: >=3.9
|
|
16
21
|
Description-Content-Type: text/markdown
|
|
17
22
|
License-File: LICENSE
|
|
18
23
|
Requires-Dist: polars>=0.20.2
|
|
19
24
|
Requires-Dist: pyarrow>=18.0.0
|
|
20
25
|
Requires-Dist: pandas>=2.1.0
|
|
26
|
+
Provides-Extra: dev
|
|
27
|
+
Requires-Dist: pytest; extra == "dev"
|
|
28
|
+
Requires-Dist: pytest-cov; extra == "dev"
|
|
29
|
+
Requires-Dist: ruff; extra == "dev"
|
|
30
|
+
Requires-Dist: coveralls; extra == "dev"
|
|
21
31
|
Dynamic: license-file
|
|
22
32
|
|
|
23
33
|
[](https://github.com/openvax/gtfparse/actions/workflows/tests.yml)
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=61.0", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "gtfparse"
|
|
7
|
+
requires-python = ">=3.9"
|
|
8
|
+
authors = [ {name="Alex Rubinsteyn", email="alex.rubinsteyn@unc.edu" } ]
|
|
9
|
+
description = "Parsing library for extracting data frames of genomic features from GTF files"
|
|
10
|
+
classifiers = [
|
|
11
|
+
"Development Status :: 4 - Beta",
|
|
12
|
+
"Environment :: Console",
|
|
13
|
+
"Operating System :: OS Independent",
|
|
14
|
+
"Intended Audience :: Science/Research",
|
|
15
|
+
"License :: OSI Approved :: Apache Software License",
|
|
16
|
+
"Programming Language :: Python",
|
|
17
|
+
"Programming Language :: Python :: 3",
|
|
18
|
+
"Programming Language :: Python :: 3.9",
|
|
19
|
+
"Programming Language :: Python :: 3.10",
|
|
20
|
+
"Programming Language :: Python :: 3.11",
|
|
21
|
+
"Programming Language :: Python :: 3.12",
|
|
22
|
+
"Topic :: Scientific/Engineering :: Bio-Informatics",
|
|
23
|
+
]
|
|
24
|
+
readme = "README.md"
|
|
25
|
+
dynamic = ["version", "dependencies"]
|
|
26
|
+
|
|
27
|
+
[project.optional-dependencies]
|
|
28
|
+
dev = [
|
|
29
|
+
"pytest",
|
|
30
|
+
"pytest-cov",
|
|
31
|
+
"ruff",
|
|
32
|
+
"coveralls",
|
|
33
|
+
]
|
|
34
|
+
|
|
35
|
+
[project.urls]
|
|
36
|
+
"Homepage" = "https://github.com/openvax/gtfparse"
|
|
37
|
+
"Bug Tracker" = "https://github.com/openvax/gtfparse/issues"
|
|
38
|
+
|
|
39
|
+
[tool.setuptools.dynamic]
|
|
40
|
+
version = {attr = "gtfparse.__version__"}
|
|
41
|
+
dependencies = {file = ["requirements.txt"]}
|
|
42
|
+
|
|
43
|
+
[tool.setuptools]
|
|
44
|
+
packages = ["gtfparse"]
|
|
45
|
+
|
|
46
|
+
[tool.ruff]
|
|
47
|
+
target-version = "py39"
|
|
48
|
+
line-length = 100
|
|
49
|
+
src = ["gtfparse", "tests"]
|
|
50
|
+
exclude = [
|
|
51
|
+
".git",
|
|
52
|
+
".venv",
|
|
53
|
+
"__pycache__",
|
|
54
|
+
"build",
|
|
55
|
+
"dist",
|
|
56
|
+
"*.egg-info",
|
|
57
|
+
".eggs",
|
|
58
|
+
]
|
|
59
|
+
|
|
60
|
+
[tool.ruff.lint]
|
|
61
|
+
select = [
|
|
62
|
+
"E", # pycodestyle errors
|
|
63
|
+
"W", # pycodestyle warnings
|
|
64
|
+
"F", # Pyflakes
|
|
65
|
+
"I", # isort
|
|
66
|
+
"B", # flake8-bugbear
|
|
67
|
+
"C4", # flake8-comprehensions
|
|
68
|
+
"UP", # pyupgrade
|
|
69
|
+
"SIM", # flake8-simplify
|
|
70
|
+
"RUF", # Ruff-specific rules
|
|
71
|
+
]
|
|
72
|
+
ignore = [
|
|
73
|
+
"E501", # line too long (handled by formatter)
|
|
74
|
+
"E741", # ambiguous variable name (pre-existing in codebase)
|
|
75
|
+
"B006", # mutable default args (pre-existing; behavior-risky to change)
|
|
76
|
+
"B008", # do not perform function calls in argument defaults
|
|
77
|
+
"B905", # zip() without explicit strict
|
|
78
|
+
"SIM108", # use ternary operator instead of if-else
|
|
79
|
+
"UP007", # use X | Y for type unions (need Python 3.10+)
|
|
80
|
+
"UP031", # %-format strings (pre-existing; out of scope for config PR)
|
|
81
|
+
]
|
|
82
|
+
|
|
83
|
+
[tool.ruff.lint.per-file-ignores]
|
|
84
|
+
"__init__.py" = ["F401"]
|
|
85
|
+
"tests/*" = ["F401", "B011"]
|
|
86
|
+
|
|
87
|
+
[tool.ruff.lint.isort]
|
|
88
|
+
known-first-party = ["gtfparse"]
|
|
89
|
+
|
|
90
|
+
[tool.pytest.ini_options]
|
|
91
|
+
testpaths = ["tests"]
|
|
92
|
+
python_files = "test_*.py"
|
|
93
|
+
python_functions = "test_*"
|
|
94
|
+
addopts = "-v --tb=short"
|
|
95
|
+
|
|
96
|
+
[tool.coverage.run]
|
|
97
|
+
source = ["gtfparse"]
|
|
98
|
+
|
|
99
|
+
[tool.coverage.report]
|
|
100
|
+
exclude_lines = [
|
|
101
|
+
"pragma: no cover",
|
|
102
|
+
"if __name__ == .__main__.:",
|
|
103
|
+
"raise NotImplementedError",
|
|
104
|
+
]
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
from io import StringIO
|
|
2
|
+
|
|
3
|
+
from gtfparse import create_missing_features, parse_gtf_and_expand_attributes
|
|
4
|
+
|
|
5
|
+
# two lines from the Ensembl 54 human GTF containing only a stop_codon and
|
|
6
|
+
# exon features, but from which gene and transcript information could be
|
|
7
|
+
# inferred
|
|
8
|
+
GTF_TEXT = "\n".join(
|
|
9
|
+
[
|
|
10
|
+
"# seqname biotype feature start end score strand frame attribute",
|
|
11
|
+
"".join(
|
|
12
|
+
[
|
|
13
|
+
"""18\tprotein_coding\tstop_codon\t32630766\t32630768\t.\t-\t0\t""",
|
|
14
|
+
"""gene_id "ENSG00000134779"; transcript_id "ENST00000334295"; exon_number "7";"""
|
|
15
|
+
""" gene_name "C18orf10";""",
|
|
16
|
+
""" transcript_name "C18orf10-201";""",
|
|
17
|
+
]
|
|
18
|
+
),
|
|
19
|
+
"".join(
|
|
20
|
+
[
|
|
21
|
+
"""18\tprotein_coding\texon\t32663078\t32663157\t.\t+\t.\tgene_id "ENSG00000150477"; """,
|
|
22
|
+
"""transcript_id "ENST00000383055"; exon_number "1"; gene_name "KIAA1328"; """,
|
|
23
|
+
"""transcript_name "KIAA1328-202";""",
|
|
24
|
+
]
|
|
25
|
+
),
|
|
26
|
+
]
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
GTF_DATAFRAME = parse_gtf_and_expand_attributes(StringIO(GTF_TEXT))
|
|
31
|
+
GTF_DATAFRAME = GTF_DATAFRAME.to_pandas()
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def test_create_missing_features_identity():
|
|
35
|
+
df_should_be_same = create_missing_features(GTF_DATAFRAME, {})
|
|
36
|
+
assert len(GTF_DATAFRAME) == len(df_should_be_same), "GTF DataFrames should be same size"
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _check_expanded_dataframe(df):
|
|
40
|
+
assert "gene" in set(df["feature"]), "Extended GTF should contain gene feature"
|
|
41
|
+
assert "transcript" in set(df["feature"]), "Extended GTF should contain transcript feature"
|
|
42
|
+
|
|
43
|
+
C18orf10_201_transcript_mask = (df["feature"] == "transcript") & (
|
|
44
|
+
df["transcript_name"] == "C18orf10-201"
|
|
45
|
+
)
|
|
46
|
+
assert len(df[C18orf10_201_transcript_mask]) == 1, (
|
|
47
|
+
"Expected only 1 gene entry for C18orf10-201, got %s" % (df[C18orf10_201_transcript_mask],)
|
|
48
|
+
)
|
|
49
|
+
transcript_seqname = df[C18orf10_201_transcript_mask].seqname.iloc[0]
|
|
50
|
+
assert transcript_seqname == "18", "Wrong seqname for C18orf10-201: %s" % transcript_seqname
|
|
51
|
+
transcript_start = df[C18orf10_201_transcript_mask].start.iloc[0]
|
|
52
|
+
assert transcript_start == 32630766, "Wrong start for C18orf10-201: %s" % transcript_start
|
|
53
|
+
transcript_end = df[C18orf10_201_transcript_mask].end.iloc[0]
|
|
54
|
+
assert transcript_end == 32630768, "Wrong end for C18orf10-201: %s" % transcript_end
|
|
55
|
+
transcript_strand = df[C18orf10_201_transcript_mask].strand.iloc[0]
|
|
56
|
+
assert transcript_strand == "-", "Wrong strand for C18orf10-201: %s" % transcript_strand
|
|
57
|
+
|
|
58
|
+
KIAA1328_gene_mask = (df["feature"] == "gene") & (df["gene_name"] == "KIAA1328")
|
|
59
|
+
assert len(df[KIAA1328_gene_mask]) == 1, "Expected only 1 gene entry for KIAA1328"
|
|
60
|
+
gene_seqname = df[KIAA1328_gene_mask].seqname.iloc[0]
|
|
61
|
+
assert gene_seqname == "18", "Wrong seqname for KIAA1328: %s" % gene_seqname
|
|
62
|
+
gene_start = df[KIAA1328_gene_mask].start.iloc[0]
|
|
63
|
+
assert gene_start == 32663078, "Wrong start for KIAA1328: %s" % (gene_start,)
|
|
64
|
+
gene_end = df[KIAA1328_gene_mask].end.iloc[0]
|
|
65
|
+
assert gene_end == 32663157, "Wrong end for KIAA1328: %s" % (gene_end,)
|
|
66
|
+
gene_strand = df[KIAA1328_gene_mask].strand.iloc[0]
|
|
67
|
+
assert gene_strand == "+", "Wrong strand for KIAA1328: %s" % gene_strand
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def test_create_missing_features():
|
|
71
|
+
assert "gene" not in set(GTF_DATAFRAME["feature"]), (
|
|
72
|
+
"Original GTF should not contain gene feature"
|
|
73
|
+
)
|
|
74
|
+
assert "transcript" not in set(GTF_DATAFRAME["feature"]), (
|
|
75
|
+
"Original GTF should not contain transcript feature"
|
|
76
|
+
)
|
|
77
|
+
df_extra_features = create_missing_features(
|
|
78
|
+
GTF_DATAFRAME,
|
|
79
|
+
unique_keys={"gene": "gene_id", "transcript": "transcript_id"},
|
|
80
|
+
extra_columns={
|
|
81
|
+
"gene": {"gene_name"},
|
|
82
|
+
"transcript": {"gene_id", "gene_name", "transcript_name"},
|
|
83
|
+
},
|
|
84
|
+
)
|
|
85
|
+
_check_expanded_dataframe(df_extra_features)
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
from gtfparse import read_gtf
|
|
2
|
+
|
|
3
|
+
from .data import data_path
|
|
4
|
+
|
|
5
|
+
ENSEMBL_GTF_PATH = data_path("ensembl_grch37.head.gtf")
|
|
6
|
+
|
|
7
|
+
EXPECTED_FEATURES = {
|
|
8
|
+
"gene",
|
|
9
|
+
"transcript",
|
|
10
|
+
"exon",
|
|
11
|
+
"CDS",
|
|
12
|
+
"UTR",
|
|
13
|
+
"start_codon",
|
|
14
|
+
"stop_codon",
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def test_ensembl_gtf_columns():
|
|
19
|
+
df = read_gtf(ENSEMBL_GTF_PATH)
|
|
20
|
+
features = set(df["feature"])
|
|
21
|
+
assert features == EXPECTED_FEATURES
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
# first 1000 lines of GTF only contained these genes
|
|
25
|
+
EXPECTED_GENE_NAMES = {
|
|
26
|
+
"FAM41C",
|
|
27
|
+
"CICP27",
|
|
28
|
+
"RNU6-1100P",
|
|
29
|
+
"NOC2L",
|
|
30
|
+
"AP006222.1",
|
|
31
|
+
"LINC01128",
|
|
32
|
+
"RP4-669L17.1",
|
|
33
|
+
"RP11-206L10.2",
|
|
34
|
+
"PLEKHN1",
|
|
35
|
+
"WBP1LP7",
|
|
36
|
+
"RP5-857K21.1",
|
|
37
|
+
"RP5-857K21.5",
|
|
38
|
+
"RNU6-1199P",
|
|
39
|
+
"RP11-206L10.10",
|
|
40
|
+
"RP11-54O7.16",
|
|
41
|
+
"CICP7",
|
|
42
|
+
"AL627309.1",
|
|
43
|
+
"RP5-857K21.11",
|
|
44
|
+
"DDX11L1",
|
|
45
|
+
"RP5-857K21.3",
|
|
46
|
+
"RP11-34P13.7",
|
|
47
|
+
"AL669831.1",
|
|
48
|
+
"MTATP6P1",
|
|
49
|
+
"CICP3",
|
|
50
|
+
"WBP1LP6",
|
|
51
|
+
"LINC00115",
|
|
52
|
+
"hsa-mir-6723",
|
|
53
|
+
"RP5-857K21.7",
|
|
54
|
+
"SAMD11",
|
|
55
|
+
"RP11-206L10.5",
|
|
56
|
+
"RP11-34P13.8",
|
|
57
|
+
"RP11-206L10.9",
|
|
58
|
+
"RP11-34P13.15",
|
|
59
|
+
"TUBB8P11",
|
|
60
|
+
"MTATP8P1",
|
|
61
|
+
"RP4-669L17.8",
|
|
62
|
+
"RP11-206L10.1",
|
|
63
|
+
"RP11-34P13.13",
|
|
64
|
+
"RP11-206L10.3",
|
|
65
|
+
"RP11-206L10.4",
|
|
66
|
+
"RP11-54O7.3",
|
|
67
|
+
"RP5-857K21.2",
|
|
68
|
+
"OR4F5",
|
|
69
|
+
"MTND1P23",
|
|
70
|
+
"AL645608.1",
|
|
71
|
+
"RP11-34P13.16",
|
|
72
|
+
"RP11-34P13.14",
|
|
73
|
+
"AP006222.2",
|
|
74
|
+
"OR4F29",
|
|
75
|
+
"RP4-669L17.4",
|
|
76
|
+
"AL732372.1",
|
|
77
|
+
"OR4G4P",
|
|
78
|
+
"MTND2P28",
|
|
79
|
+
"OR4F16",
|
|
80
|
+
"KLHL17",
|
|
81
|
+
"FAM138A",
|
|
82
|
+
"OR4G11P",
|
|
83
|
+
"FAM87B",
|
|
84
|
+
"RP5-857K21.15",
|
|
85
|
+
"AL645608.2",
|
|
86
|
+
"RP11-206L10.8",
|
|
87
|
+
"RP5-857K21.4",
|
|
88
|
+
"MIR1302-10",
|
|
89
|
+
"RP11-54O7.2",
|
|
90
|
+
"RP4-669L17.10",
|
|
91
|
+
"RP11-54O7.1",
|
|
92
|
+
"RP11-34P13.9",
|
|
93
|
+
"WASH7P",
|
|
94
|
+
"RP4-669L17.2",
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def test_ensembl_gtf_gene_names():
|
|
99
|
+
df = read_gtf(ENSEMBL_GTF_PATH)
|
|
100
|
+
gene_names = set(df["gene_name"])
|
|
101
|
+
assert gene_names == EXPECTED_GENE_NAMES, (
|
|
102
|
+
"Wrong gene names: %s, missing %s and unexpected %s"
|
|
103
|
+
% (
|
|
104
|
+
gene_names,
|
|
105
|
+
EXPECTED_GENE_NAMES.difference(gene_names),
|
|
106
|
+
gene_names.difference(EXPECTED_GENE_NAMES),
|
|
107
|
+
)
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def test_ensembl_gtf_gene_names_with_usecols():
|
|
112
|
+
df = read_gtf(ENSEMBL_GTF_PATH, usecols=["gene_name"])
|
|
113
|
+
gene_names = set(df["gene_name"])
|
|
114
|
+
assert gene_names == EXPECTED_GENE_NAMES, (
|
|
115
|
+
"Wrong gene names: %s, missing %s and unexpected %s"
|
|
116
|
+
% (
|
|
117
|
+
gene_names,
|
|
118
|
+
EXPECTED_GENE_NAMES.difference(gene_names),
|
|
119
|
+
gene_names.difference(EXPECTED_GENE_NAMES),
|
|
120
|
+
)
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def test_ensembl_gtf_gene_names_zip():
|
|
125
|
+
df = read_gtf(ENSEMBL_GTF_PATH + ".gz")
|
|
126
|
+
gene_names = set(df["gene_name"])
|
|
127
|
+
assert gene_names == EXPECTED_GENE_NAMES, (
|
|
128
|
+
"Wrong gene names: %s, missing %s and unexpected %s"
|
|
129
|
+
% (
|
|
130
|
+
gene_names,
|
|
131
|
+
EXPECTED_GENE_NAMES.difference(gene_names),
|
|
132
|
+
gene_names.difference(EXPECTED_GENE_NAMES),
|
|
133
|
+
)
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def test_ensembl_gtf_gene_names_with_usecols_gzip():
|
|
138
|
+
df = read_gtf(ENSEMBL_GTF_PATH + ".gz", usecols=["gene_name"])
|
|
139
|
+
gene_names = set(df["gene_name"])
|
|
140
|
+
assert gene_names == EXPECTED_GENE_NAMES, (
|
|
141
|
+
"Wrong gene names: %s, missing %s and unexpected %s"
|
|
142
|
+
% (
|
|
143
|
+
gene_names,
|
|
144
|
+
EXPECTED_GENE_NAMES.difference(gene_names),
|
|
145
|
+
gene_names.difference(EXPECTED_GENE_NAMES),
|
|
146
|
+
)
|
|
147
|
+
)
|
|
@@ -1,12 +1,13 @@
|
|
|
1
1
|
from gtfparse import expand_attribute_strings
|
|
2
2
|
|
|
3
|
+
|
|
3
4
|
def test_attributes_in_quotes():
|
|
4
5
|
attributes = [
|
|
5
|
-
|
|
6
|
-
|
|
6
|
+
'gene_id "ENSG001"; tag "bogotron"; version "1";',
|
|
7
|
+
'gene_id "ENSG002"; tag "wolfpuppy"; version "2";',
|
|
7
8
|
]
|
|
8
9
|
parsed_dict = expand_attribute_strings(attributes, quote_char='"')
|
|
9
|
-
assert
|
|
10
|
+
assert sorted(parsed_dict.keys()), ["gene_id", "tag", "version"]
|
|
10
11
|
assert parsed_dict["gene_id"] == ["ENSG001", "ENSG002"]
|
|
11
12
|
assert parsed_dict["tag"] == ["bogotron", "wolfpuppy"]
|
|
12
13
|
assert parsed_dict["version"] == ["1", "2"]
|
|
@@ -15,10 +16,10 @@ def test_attributes_in_quotes():
|
|
|
15
16
|
def test_attributes_without_quotes():
|
|
16
17
|
attributes = [
|
|
17
18
|
"gene_id ENSG001; tag bogotron; version 1;",
|
|
18
|
-
"gene_id ENSG002; tag wolfpuppy; version 2"
|
|
19
|
+
"gene_id ENSG002; tag wolfpuppy; version 2",
|
|
19
20
|
]
|
|
20
21
|
parsed_dict = expand_attribute_strings(attributes)
|
|
21
|
-
assert
|
|
22
|
+
assert sorted(parsed_dict.keys()) == ["gene_id", "tag", "version"]
|
|
22
23
|
assert parsed_dict["gene_id"] == ["ENSG001", "ENSG002"]
|
|
23
24
|
assert parsed_dict["tag"] == ["bogotron", "wolfpuppy"]
|
|
24
25
|
assert parsed_dict["version"] == ["1", "2"]
|
|
@@ -31,6 +32,6 @@ def test_optional_attributes():
|
|
|
31
32
|
"gene_id ENSG003; sometimes-present wolfpuppy;",
|
|
32
33
|
]
|
|
33
34
|
parsed_dict = expand_attribute_strings(attributes)
|
|
34
|
-
assert
|
|
35
|
-
assert parsed_dict["gene_id"] ==
|
|
36
|
-
assert parsed_dict["sometimes-present"] ==
|
|
35
|
+
assert sorted(parsed_dict.keys()) == ["gene_id", "sometimes-present"]
|
|
36
|
+
assert parsed_dict["gene_id"] == ["ENSG001", "ENSG002", "ENSG003"]
|
|
37
|
+
assert parsed_dict["sometimes-present"] == ["bogotron", "", "wolfpuppy"]
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
from io import StringIO
|
|
2
|
+
|
|
2
3
|
from gtfparse import parse_gtf_and_expand_attributes
|
|
3
4
|
|
|
4
5
|
# failing example from https://github.com/openvax/gtfparse/issues/2
|
|
@@ -11,25 +12,26 @@ GTF_TEXT = (
|
|
|
11
12
|
"""tag "cds_end_NF"; tag "mRNA_end_NF"; """
|
|
12
13
|
)
|
|
13
14
|
|
|
15
|
+
|
|
14
16
|
def test_parse_tag_attributes():
|
|
15
17
|
parsed = parse_gtf_and_expand_attributes(StringIO(GTF_TEXT))
|
|
16
18
|
tag_column = parsed["tag"]
|
|
17
19
|
assert len(tag_column) == 1
|
|
18
20
|
tags = tag_column[0]
|
|
19
|
-
assert tags ==
|
|
21
|
+
assert tags == "cds_end_NF,mRNA_end_NF"
|
|
22
|
+
|
|
20
23
|
|
|
21
24
|
def test_parse_tag_attributes_with_usecols():
|
|
22
|
-
parsed = parse_gtf_and_expand_attributes(
|
|
23
|
-
StringIO(GTF_TEXT),
|
|
24
|
-
restrict_attribute_columns=["tag"])
|
|
25
|
+
parsed = parse_gtf_and_expand_attributes(StringIO(GTF_TEXT), restrict_attribute_columns=["tag"])
|
|
25
26
|
tag_column = parsed["tag"]
|
|
26
27
|
assert len(tag_column) == 1
|
|
27
28
|
tags = tag_column[0]
|
|
28
|
-
assert tags ==
|
|
29
|
+
assert tags == "cds_end_NF,mRNA_end_NF"
|
|
30
|
+
|
|
29
31
|
|
|
30
32
|
def test_parse_tag_attributes_with_usecols_other_column():
|
|
31
33
|
parsed = parse_gtf_and_expand_attributes(
|
|
32
|
-
StringIO(GTF_TEXT),
|
|
33
|
-
|
|
34
|
+
StringIO(GTF_TEXT), restrict_attribute_columns=["exon_id"]
|
|
35
|
+
)
|
|
34
36
|
|
|
35
37
|
assert "tag" not in parsed, "Expected 'tag' to get dropped but got %s" % (parsed,)
|
|
@@ -1,11 +1,8 @@
|
|
|
1
|
+
from io import StringIO
|
|
2
|
+
|
|
1
3
|
from pytest import raises
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
parse_gtf_and_expand_attributes,
|
|
5
|
-
REQUIRED_COLUMNS,
|
|
6
|
-
ParsingError
|
|
7
|
-
)
|
|
8
|
-
from io import StringIO
|
|
4
|
+
|
|
5
|
+
from gtfparse import REQUIRED_COLUMNS, ParsingError, parse_gtf, parse_gtf_and_expand_attributes
|
|
9
6
|
|
|
10
7
|
gtf_text = """
|
|
11
8
|
# sample GTF data copied from:
|
|
@@ -14,12 +11,13 @@ gtf_text = """
|
|
|
14
11
|
1\tprocessed_transcript\ttranscript\t11869\t14409\t.\t+\t.\tgene_id "ENSG00000223972"; transcript_id "ENST00000456328"; gene_name "DDX11L1"; gene_source "havana"; gene_biotype "transcribed_unprocessed_pseudogene"; transcript_name "DDX11L1-002"; transcript_source "havana";
|
|
15
12
|
"""
|
|
16
13
|
|
|
14
|
+
|
|
17
15
|
def test_parse_gtf_lines_with_expand_attributes():
|
|
18
16
|
df = parse_gtf_and_expand_attributes(StringIO(gtf_text))
|
|
19
17
|
|
|
20
|
-
|
|
21
18
|
# excluding 'attribute' column from required names
|
|
22
|
-
expected_columns =
|
|
19
|
+
expected_columns = [
|
|
20
|
+
*REQUIRED_COLUMNS[:8],
|
|
23
21
|
"gene_id",
|
|
24
22
|
"gene_name",
|
|
25
23
|
"gene_source",
|
|
@@ -29,7 +27,7 @@ def test_parse_gtf_lines_with_expand_attributes():
|
|
|
29
27
|
"transcript_source",
|
|
30
28
|
]
|
|
31
29
|
# convert to list since Py3's dictionary keys are a distinct collection type
|
|
32
|
-
assert list(df.columns) ==
|
|
30
|
+
assert list(df.columns) == expected_columns
|
|
33
31
|
assert list(df["seqname"]) == ["1", "1"]
|
|
34
32
|
# convert to list for comparison since numerical columns may be NumPy arrays
|
|
35
33
|
assert list(df["start"]) == [11869, 11869]
|
|
@@ -52,6 +50,7 @@ def test_parse_gtf_lines_without_expand_attributes():
|
|
|
52
50
|
assert df["score"].is_null().all(), "Unexpected scores: %s" % (df["score"],)
|
|
53
51
|
assert len(df["attribute"]) == 2
|
|
54
52
|
|
|
53
|
+
|
|
55
54
|
def test_parse_gtf_lines_error_too_few_fields():
|
|
56
55
|
bad_gtf_text = gtf_text.replace("\t", " ")
|
|
57
56
|
# pylint: disable=no-value-for-parameter
|
|
@@ -1,8 +1,10 @@
|
|
|
1
1
|
from gtfparse import read_gtf
|
|
2
|
+
|
|
2
3
|
from .data import data_path
|
|
3
4
|
|
|
4
5
|
B16_GTF_PATH = data_path("B16.stringtie.head.gtf")
|
|
5
6
|
|
|
7
|
+
|
|
6
8
|
def _check_required_columns(gtf_dict):
|
|
7
9
|
assert "feature" in gtf_dict, "Expected column named 'feature' in StringTie GTF"
|
|
8
10
|
assert "cov" in gtf_dict, "Expected column named 'cov' in StringTie GTF"
|
|
@@ -11,32 +13,37 @@ def _check_required_columns(gtf_dict):
|
|
|
11
13
|
assert "exon" in features, "No exons in GTF (available: %s)" % features
|
|
12
14
|
assert "transcript" in features, "No transcripts in GTF (available: %s)" % features
|
|
13
15
|
|
|
16
|
+
|
|
14
17
|
def _check_string_cov_and_FPKM(gtf_dict):
|
|
15
18
|
for i, feature_name in enumerate(gtf_dict["feature"]):
|
|
16
19
|
cov = gtf_dict["cov"][i]
|
|
17
20
|
fpkm = gtf_dict["FPKM"][i]
|
|
18
21
|
if feature_name == "exon":
|
|
19
|
-
assert len(fpkm) == 0,
|
|
20
|
-
|
|
21
|
-
assert len(cov) > 0 and float(cov) >= 0, \
|
|
22
|
+
assert len(fpkm) == 0, "Expected missing FPKM for exon, got %s" % (fpkm,)
|
|
23
|
+
assert len(cov) > 0 and float(cov) >= 0, (
|
|
22
24
|
"Expected non-negative cov for exon, got %s" % (cov,)
|
|
25
|
+
)
|
|
23
26
|
elif feature_name == "transcript":
|
|
24
|
-
assert len(cov) and float(cov) >= 0,
|
|
27
|
+
assert len(cov) and float(cov) >= 0, (
|
|
25
28
|
"Expected non-negative cov for transcript, got %s" % (cov,)
|
|
26
|
-
|
|
29
|
+
)
|
|
30
|
+
assert len(fpkm) > 0 and float(fpkm) >= 0, (
|
|
27
31
|
"Expected non-negative FPKM for transcript, got %s" % (fpkm,)
|
|
32
|
+
)
|
|
33
|
+
|
|
28
34
|
|
|
29
35
|
def _check_float_cov_and_FPKM(gtf_dict):
|
|
30
36
|
for i, feature_name in enumerate(gtf_dict["feature"]):
|
|
31
37
|
cov = gtf_dict["cov"][i]
|
|
32
38
|
fpkm = gtf_dict["FPKM"][i]
|
|
33
|
-
assert isinstance(cov, float),
|
|
34
|
-
"Expected cov to be float but got %s : %s" % (cov, type(cov))
|
|
39
|
+
assert isinstance(cov, float), "Expected cov to be float but got %s : %s" % (cov, type(cov))
|
|
35
40
|
if feature_name == "exon":
|
|
36
41
|
assert cov >= 0, "Expected non-negative cov for exon, got %s" % (cov,)
|
|
37
42
|
elif feature_name == "transcript":
|
|
38
|
-
assert isinstance(fpkm, float),
|
|
39
|
-
|
|
43
|
+
assert isinstance(fpkm, float), "Expected FPKM to be float but got %s : %s" % (
|
|
44
|
+
fpkm,
|
|
45
|
+
type(fpkm),
|
|
46
|
+
)
|
|
40
47
|
assert cov >= 0, "Expected non-negative cov for transcript, got %s" % (cov,)
|
|
41
48
|
assert fpkm >= 0, "Expected non-negative FPKM for transcript, got %s" % (fpkm,)
|
|
42
49
|
|
|
@@ -46,9 +53,8 @@ def test_read_stringtie_gtf_as_dataframe():
|
|
|
46
53
|
_check_required_columns(gtf_df)
|
|
47
54
|
_check_string_cov_and_FPKM(gtf_df)
|
|
48
55
|
|
|
56
|
+
|
|
49
57
|
def test_read_stringtie_gtf_as_dataframe_float_values():
|
|
50
|
-
gtf_df = read_gtf(
|
|
51
|
-
B16_GTF_PATH,
|
|
52
|
-
column_converters={"cov": float, "FPKM": float})
|
|
58
|
+
gtf_df = read_gtf(B16_GTF_PATH, column_converters={"cov": float, "FPKM": float})
|
|
53
59
|
_check_required_columns(gtf_df)
|
|
54
60
|
_check_float_cov_and_FPKM(gtf_df)
|
|
@@ -1,8 +1,10 @@
|
|
|
1
1
|
from gtfparse import read_gtf
|
|
2
|
+
|
|
2
3
|
from .data import data_path
|
|
3
4
|
|
|
4
5
|
REFSEQ_GTF_PATH = data_path("refseq.ucsc.small.gtf")
|
|
5
6
|
|
|
7
|
+
|
|
6
8
|
def _check_required_columns(gtf_dict):
|
|
7
9
|
assert "feature" in gtf_dict, "Expected column named 'feature' in RefSeq GTF"
|
|
8
10
|
assert "gene_id" in gtf_dict, "Expected column named 'gene_id' in RefSeq GTF"
|
|
@@ -11,10 +13,16 @@ def _check_required_columns(gtf_dict):
|
|
|
11
13
|
assert "exon" in features, "No exon features in GTF (available: %s)" % features
|
|
12
14
|
assert "CDS" in features, "No CDS features in GTF (available: %s)" % features
|
|
13
15
|
|
|
16
|
+
|
|
14
17
|
def test_read_refseq_gtf_as_dataframe():
|
|
15
18
|
gtf_df = read_gtf(REFSEQ_GTF_PATH)
|
|
16
19
|
_check_required_columns(gtf_df)
|
|
17
20
|
|
|
21
|
+
|
|
18
22
|
def test_read_refseq_and_transform_columns():
|
|
19
|
-
gtf_df = read_gtf(
|
|
20
|
-
|
|
23
|
+
gtf_df = read_gtf(
|
|
24
|
+
REFSEQ_GTF_PATH,
|
|
25
|
+
column_converters={"start": int, "end": int},
|
|
26
|
+
column_cast_types={"score": float},
|
|
27
|
+
)
|
|
28
|
+
print(gtf_df)
|
gtfparse-2.6.0/pyproject.toml
DELETED
|
@@ -1,27 +0,0 @@
|
|
|
1
|
-
[project]
|
|
2
|
-
name = "gtfparse"
|
|
3
|
-
requires-python = ">=3.7"
|
|
4
|
-
authors = [ {name="Alex Rubinsteyn", email="alex.rubinsteyn@unc.edu" } ]
|
|
5
|
-
description = "Parsing library for extracting data frames of genomic features from GTF files"
|
|
6
|
-
classifiers = [
|
|
7
|
-
'Development Status :: 4 - Beta',
|
|
8
|
-
'Environment :: Console',
|
|
9
|
-
'Operating System :: OS Independent',
|
|
10
|
-
'Intended Audience :: Science/Research',
|
|
11
|
-
'License :: OSI Approved :: Apache Software License',
|
|
12
|
-
'Programming Language :: Python',
|
|
13
|
-
'Topic :: Scientific/Engineering :: Bio-Informatics',
|
|
14
|
-
]
|
|
15
|
-
readme = "README.md"
|
|
16
|
-
dynamic = ["version", "dependencies"]
|
|
17
|
-
|
|
18
|
-
[tool.setuptools.dynamic]
|
|
19
|
-
version = {attr = "gtfparse.__version__"}
|
|
20
|
-
dependencies = {file = ["requirements.txt"]}
|
|
21
|
-
|
|
22
|
-
[tool.setuptools]
|
|
23
|
-
packages = ["gtfparse"]
|
|
24
|
-
|
|
25
|
-
[project.urls]
|
|
26
|
-
"Homepage" = "https://github.com/openvax/gtfparse"
|
|
27
|
-
"Bug Tracker" = "https://github.com/openvax/gtfparse"
|
|
@@ -1,86 +0,0 @@
|
|
|
1
|
-
from gtfparse import create_missing_features, parse_gtf_and_expand_attributes
|
|
2
|
-
from io import StringIO
|
|
3
|
-
|
|
4
|
-
# two lines from the Ensembl 54 human GTF containing only a stop_codon and
|
|
5
|
-
# exon features, but from which gene and transcript information could be
|
|
6
|
-
# inferred
|
|
7
|
-
GTF_TEXT = "\n".join([
|
|
8
|
-
"# seqname biotype feature start end score strand frame attribute",
|
|
9
|
-
"".join([
|
|
10
|
-
"""18\tprotein_coding\tstop_codon\t32630766\t32630768\t.\t-\t0\t""",
|
|
11
|
-
"""gene_id "ENSG00000134779"; transcript_id "ENST00000334295"; exon_number "7";"""
|
|
12
|
-
""" gene_name "C18orf10";""",
|
|
13
|
-
""" transcript_name "C18orf10-201";"""]),
|
|
14
|
-
"".join([
|
|
15
|
-
"""18\tprotein_coding\texon\t32663078\t32663157\t.\t+\t.\tgene_id "ENSG00000150477"; """,
|
|
16
|
-
"""transcript_id "ENST00000383055"; exon_number "1"; gene_name "KIAA1328"; """,
|
|
17
|
-
"""transcript_name "KIAA1328-202";"""]),
|
|
18
|
-
])
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
GTF_DATAFRAME = parse_gtf_and_expand_attributes(StringIO(GTF_TEXT))
|
|
22
|
-
GTF_DATAFRAME = GTF_DATAFRAME.to_pandas()
|
|
23
|
-
|
|
24
|
-
def test_create_missing_features_identity():
|
|
25
|
-
df_should_be_same = create_missing_features(GTF_DATAFRAME, {})
|
|
26
|
-
assert len(GTF_DATAFRAME) == len(df_should_be_same), \
|
|
27
|
-
"GTF DataFrames should be same size"
|
|
28
|
-
|
|
29
|
-
def _check_expanded_dataframe(df):
|
|
30
|
-
assert "gene" in set(df["feature"]), \
|
|
31
|
-
"Extended GTF should contain gene feature"
|
|
32
|
-
assert "transcript" in set(df["feature"]), \
|
|
33
|
-
"Extended GTF should contain transcript feature"
|
|
34
|
-
|
|
35
|
-
C18orf10_201_transcript_mask = (
|
|
36
|
-
(df["feature"] == "transcript") &
|
|
37
|
-
(df["transcript_name"] == "C18orf10-201"))
|
|
38
|
-
assert len(df[C18orf10_201_transcript_mask]) == 1, \
|
|
39
|
-
"Expected only 1 gene entry for C18orf10-201, got %s" % (
|
|
40
|
-
df[C18orf10_201_transcript_mask],)
|
|
41
|
-
transcript_seqname = df[C18orf10_201_transcript_mask].seqname.iloc[0]
|
|
42
|
-
assert (transcript_seqname == "18"), \
|
|
43
|
-
"Wrong seqname for C18orf10-201: %s" % transcript_seqname
|
|
44
|
-
transcript_start = df[C18orf10_201_transcript_mask].start.iloc[0]
|
|
45
|
-
assert (transcript_start == 32630766), \
|
|
46
|
-
"Wrong start for C18orf10-201: %s" % transcript_start
|
|
47
|
-
transcript_end = df[C18orf10_201_transcript_mask].end.iloc[0]
|
|
48
|
-
assert (transcript_end == 32630768), \
|
|
49
|
-
"Wrong end for C18orf10-201: %s" % transcript_end
|
|
50
|
-
transcript_strand = df[C18orf10_201_transcript_mask].strand.iloc[0]
|
|
51
|
-
assert (transcript_strand == "-"), \
|
|
52
|
-
"Wrong strand for C18orf10-201: %s" % transcript_strand
|
|
53
|
-
|
|
54
|
-
KIAA1328_gene_mask = (
|
|
55
|
-
(df["feature"] == "gene") &
|
|
56
|
-
(df["gene_name"] == "KIAA1328"))
|
|
57
|
-
assert len(df[KIAA1328_gene_mask]) == 1, "Expected only 1 gene entry for KIAA1328"
|
|
58
|
-
gene_seqname = df[KIAA1328_gene_mask].seqname.iloc[0]
|
|
59
|
-
assert (gene_seqname == "18"), \
|
|
60
|
-
"Wrong seqname for KIAA1328: %s" % gene_seqname
|
|
61
|
-
gene_start = df[KIAA1328_gene_mask].start.iloc[0]
|
|
62
|
-
assert (gene_start == 32663078), \
|
|
63
|
-
"Wrong start for KIAA1328: %s" % (gene_start,)
|
|
64
|
-
gene_end = df[KIAA1328_gene_mask].end.iloc[0]
|
|
65
|
-
assert (gene_end == 32663157), \
|
|
66
|
-
"Wrong end for KIAA1328: %s" % (gene_end,)
|
|
67
|
-
gene_strand = df[KIAA1328_gene_mask].strand.iloc[0]
|
|
68
|
-
assert (gene_strand == "+"), \
|
|
69
|
-
"Wrong strand for KIAA1328: %s" % gene_strand
|
|
70
|
-
|
|
71
|
-
def test_create_missing_features():
|
|
72
|
-
assert "gene" not in set(GTF_DATAFRAME["feature"]), \
|
|
73
|
-
"Original GTF should not contain gene feature"
|
|
74
|
-
assert "transcript" not in set(GTF_DATAFRAME["feature"]), \
|
|
75
|
-
"Original GTF should not contain transcript feature"
|
|
76
|
-
df_extra_features = create_missing_features(
|
|
77
|
-
GTF_DATAFRAME,
|
|
78
|
-
unique_keys={
|
|
79
|
-
"gene": "gene_id",
|
|
80
|
-
"transcript": "transcript_id"
|
|
81
|
-
},
|
|
82
|
-
extra_columns={
|
|
83
|
-
"gene": {"gene_name"},
|
|
84
|
-
"transcript": {"gene_id", "gene_name", "transcript_name"},
|
|
85
|
-
})
|
|
86
|
-
_check_expanded_dataframe(df_extra_features)
|
|
@@ -1,81 +0,0 @@
|
|
|
1
|
-
from gtfparse import read_gtf
|
|
2
|
-
|
|
3
|
-
from .data import data_path
|
|
4
|
-
|
|
5
|
-
ENSEMBL_GTF_PATH = data_path("ensembl_grch37.head.gtf")
|
|
6
|
-
|
|
7
|
-
EXPECTED_FEATURES = set([
|
|
8
|
-
"gene",
|
|
9
|
-
"transcript",
|
|
10
|
-
"exon",
|
|
11
|
-
"CDS",
|
|
12
|
-
"UTR",
|
|
13
|
-
"start_codon",
|
|
14
|
-
"stop_codon",
|
|
15
|
-
])
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
def test_ensembl_gtf_columns():
|
|
19
|
-
df = read_gtf(ENSEMBL_GTF_PATH)
|
|
20
|
-
features = set(df["feature"])
|
|
21
|
-
assert features == EXPECTED_FEATURES
|
|
22
|
-
|
|
23
|
-
# first 1000 lines of GTF only contained these genes
|
|
24
|
-
EXPECTED_GENE_NAMES = {
|
|
25
|
-
'FAM41C', 'CICP27', 'RNU6-1100P', 'NOC2L', 'AP006222.1',
|
|
26
|
-
'LINC01128', 'RP4-669L17.1', 'RP11-206L10.2', 'PLEKHN1',
|
|
27
|
-
'WBP1LP7', 'RP5-857K21.1', 'RP5-857K21.5', 'RNU6-1199P',
|
|
28
|
-
'RP11-206L10.10', 'RP11-54O7.16', 'CICP7', 'AL627309.1',
|
|
29
|
-
'RP5-857K21.11', 'DDX11L1', 'RP5-857K21.3', 'RP11-34P13.7',
|
|
30
|
-
'AL669831.1', 'MTATP6P1', 'CICP3', 'WBP1LP6', 'LINC00115',
|
|
31
|
-
'hsa-mir-6723', 'RP5-857K21.7', 'SAMD11', 'RP11-206L10.5',
|
|
32
|
-
'RP11-34P13.8', 'RP11-206L10.9', 'RP11-34P13.15', 'TUBB8P11',
|
|
33
|
-
'MTATP8P1', 'RP4-669L17.8', 'RP11-206L10.1', 'RP11-34P13.13',
|
|
34
|
-
'RP11-206L10.3', 'RP11-206L10.4', 'RP11-54O7.3', 'RP5-857K21.2',
|
|
35
|
-
'OR4F5', 'MTND1P23', 'AL645608.1', 'RP11-34P13.16', 'RP11-34P13.14',
|
|
36
|
-
'AP006222.2', 'OR4F29', 'RP4-669L17.4', 'AL732372.1', 'OR4G4P',
|
|
37
|
-
'MTND2P28', 'OR4F16', 'KLHL17', 'FAM138A', 'OR4G11P', 'FAM87B',
|
|
38
|
-
'RP5-857K21.15', 'AL645608.2', 'RP11-206L10.8', 'RP5-857K21.4',
|
|
39
|
-
'MIR1302-10', 'RP11-54O7.2', 'RP4-669L17.10', 'RP11-54O7.1',
|
|
40
|
-
'RP11-34P13.9', 'WASH7P', 'RP4-669L17.2'
|
|
41
|
-
}
|
|
42
|
-
|
|
43
|
-
def test_ensembl_gtf_gene_names():
|
|
44
|
-
df = read_gtf(ENSEMBL_GTF_PATH)
|
|
45
|
-
gene_names = set(df["gene_name"])
|
|
46
|
-
assert gene_names == EXPECTED_GENE_NAMES, \
|
|
47
|
-
"Wrong gene names: %s, missing %s and unexpected %s" % (
|
|
48
|
-
gene_names,
|
|
49
|
-
EXPECTED_GENE_NAMES.difference(gene_names),
|
|
50
|
-
gene_names.difference(EXPECTED_GENE_NAMES)
|
|
51
|
-
)
|
|
52
|
-
|
|
53
|
-
def test_ensembl_gtf_gene_names_with_usecols():
|
|
54
|
-
df = read_gtf(ENSEMBL_GTF_PATH, usecols=["gene_name"])
|
|
55
|
-
gene_names = set(df["gene_name"])
|
|
56
|
-
assert gene_names == EXPECTED_GENE_NAMES, \
|
|
57
|
-
"Wrong gene names: %s, missing %s and unexpected %s" % (
|
|
58
|
-
gene_names,
|
|
59
|
-
EXPECTED_GENE_NAMES.difference(gene_names),
|
|
60
|
-
gene_names.difference(EXPECTED_GENE_NAMES)
|
|
61
|
-
)
|
|
62
|
-
|
|
63
|
-
def test_ensembl_gtf_gene_names_zip():
|
|
64
|
-
df = read_gtf(ENSEMBL_GTF_PATH + ".gz")
|
|
65
|
-
gene_names = set(df["gene_name"])
|
|
66
|
-
assert gene_names == EXPECTED_GENE_NAMES, \
|
|
67
|
-
"Wrong gene names: %s, missing %s and unexpected %s" % (
|
|
68
|
-
gene_names,
|
|
69
|
-
EXPECTED_GENE_NAMES.difference(gene_names),
|
|
70
|
-
gene_names.difference(EXPECTED_GENE_NAMES)
|
|
71
|
-
)
|
|
72
|
-
|
|
73
|
-
def test_ensembl_gtf_gene_names_with_usecols_gzip():
|
|
74
|
-
df = read_gtf(ENSEMBL_GTF_PATH + ".gz", usecols=["gene_name"])
|
|
75
|
-
gene_names = set(df["gene_name"])
|
|
76
|
-
assert gene_names == EXPECTED_GENE_NAMES, \
|
|
77
|
-
"Wrong gene names: %s, missing %s and unexpected %s" % (
|
|
78
|
-
gene_names,
|
|
79
|
-
EXPECTED_GENE_NAMES.difference(gene_names),
|
|
80
|
-
gene_names.difference(EXPECTED_GENE_NAMES)
|
|
81
|
-
)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|