gtfparse 2.6.2__tar.gz → 2.7.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gtfparse-2.6.2 → gtfparse-2.7.0}/PKG-INFO +13 -3
- {gtfparse-2.6.2 → gtfparse-2.7.0}/gtfparse/__init__.py +13 -9
- {gtfparse-2.6.2 → gtfparse-2.7.0}/gtfparse/attribute_parsing.py +4 -13
- {gtfparse-2.6.2 → gtfparse-2.7.0}/gtfparse/create_missing_features.py +14 -22
- {gtfparse-2.6.2 → gtfparse-2.7.0}/gtfparse/parsing_error.py +1 -0
- {gtfparse-2.6.2 → gtfparse-2.7.0}/gtfparse/read_gtf.py +188 -76
- {gtfparse-2.6.2 → gtfparse-2.7.0}/gtfparse.egg-info/PKG-INFO +13 -3
- {gtfparse-2.6.2 → gtfparse-2.7.0}/gtfparse.egg-info/SOURCES.txt +1 -0
- {gtfparse-2.6.2 → gtfparse-2.7.0}/gtfparse.egg-info/requires.txt +6 -0
- gtfparse-2.7.0/pyproject.toml +104 -0
- gtfparse-2.7.0/tests/test_create_missing_features.py +85 -0
- gtfparse-2.7.0/tests/test_ensembl_gtf.py +147 -0
- {gtfparse-2.6.2 → gtfparse-2.7.0}/tests/test_expand_attributes.py +9 -8
- gtfparse-2.7.0/tests/test_gencode_gtf.py +563 -0
- {gtfparse-2.6.2 → gtfparse-2.7.0}/tests/test_multiple_values_for_tag_attribute.py +9 -7
- {gtfparse-2.6.2 → gtfparse-2.7.0}/tests/test_parse_gtf_lines.py +9 -10
- {gtfparse-2.6.2 → gtfparse-2.7.0}/tests/test_read_stringtie_gtf.py +18 -12
- {gtfparse-2.6.2 → gtfparse-2.7.0}/tests/test_refseq_gtf.py +10 -2
- gtfparse-2.6.2/pyproject.toml +0 -35
- gtfparse-2.6.2/tests/test_create_missing_features.py +0 -86
- gtfparse-2.6.2/tests/test_ensembl_gtf.py +0 -81
- {gtfparse-2.6.2 → gtfparse-2.7.0}/LICENSE +0 -0
- {gtfparse-2.6.2 → gtfparse-2.7.0}/README.md +0 -0
- {gtfparse-2.6.2 → gtfparse-2.7.0}/gtfparse.egg-info/dependency_links.txt +0 -0
- {gtfparse-2.6.2 → gtfparse-2.7.0}/gtfparse.egg-info/top_level.txt +0 -0
- {gtfparse-2.6.2 → gtfparse-2.7.0}/requirements.txt +0 -0
- {gtfparse-2.6.2 → gtfparse-2.7.0}/setup.cfg +0 -0
|
@@ -1,23 +1,33 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gtfparse
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.7.0
|
|
4
4
|
Summary: Parsing library for extracting data frames of genomic features from GTF files
|
|
5
5
|
Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
|
|
6
6
|
Project-URL: Homepage, https://github.com/openvax/gtfparse
|
|
7
|
-
Project-URL: Bug Tracker, https://github.com/openvax/gtfparse
|
|
7
|
+
Project-URL: Bug Tracker, https://github.com/openvax/gtfparse/issues
|
|
8
8
|
Classifier: Development Status :: 4 - Beta
|
|
9
9
|
Classifier: Environment :: Console
|
|
10
10
|
Classifier: Operating System :: OS Independent
|
|
11
11
|
Classifier: Intended Audience :: Science/Research
|
|
12
12
|
Classifier: License :: OSI Approved :: Apache Software License
|
|
13
13
|
Classifier: Programming Language :: Python
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
14
19
|
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
15
|
-
Requires-Python: >=3.
|
|
20
|
+
Requires-Python: >=3.9
|
|
16
21
|
Description-Content-Type: text/markdown
|
|
17
22
|
License-File: LICENSE
|
|
18
23
|
Requires-Dist: polars>=0.20.2
|
|
19
24
|
Requires-Dist: pyarrow>=18.0.0
|
|
20
25
|
Requires-Dist: pandas>=2.1.0
|
|
26
|
+
Provides-Extra: dev
|
|
27
|
+
Requires-Dist: pytest; extra == "dev"
|
|
28
|
+
Requires-Dist: pytest-cov; extra == "dev"
|
|
29
|
+
Requires-Dist: ruff; extra == "dev"
|
|
30
|
+
Requires-Dist: coveralls; extra == "dev"
|
|
21
31
|
Dynamic: license-file
|
|
22
32
|
|
|
23
33
|
[](https://github.com/openvax/gtfparse/actions/workflows/tests.yml)
|
|
@@ -14,23 +14,27 @@ from .attribute_parsing import expand_attribute_strings
|
|
|
14
14
|
from .create_missing_features import create_missing_features
|
|
15
15
|
from .parsing_error import ParsingError
|
|
16
16
|
from .read_gtf import (
|
|
17
|
-
|
|
17
|
+
GENCODE_BIOTYPE_ALIASES,
|
|
18
|
+
INTEGER_VERSION_COLUMNS,
|
|
19
|
+
REQUIRED_COLUMNS,
|
|
18
20
|
parse_gtf,
|
|
19
|
-
parse_gtf_pandas,
|
|
20
21
|
parse_gtf_and_expand_attributes,
|
|
21
|
-
|
|
22
|
+
parse_gtf_pandas,
|
|
23
|
+
read_gtf,
|
|
22
24
|
)
|
|
23
25
|
|
|
24
|
-
__version__ = "2.
|
|
26
|
+
__version__ = "2.7.0"
|
|
25
27
|
|
|
26
28
|
__all__ = [
|
|
27
|
-
"
|
|
28
|
-
"
|
|
29
|
-
"create_missing_features",
|
|
30
|
-
"parse_gtf_and_expand_attributes",
|
|
29
|
+
"GENCODE_BIOTYPE_ALIASES",
|
|
30
|
+
"INTEGER_VERSION_COLUMNS",
|
|
31
31
|
"REQUIRED_COLUMNS",
|
|
32
32
|
"ParsingError",
|
|
33
|
-
"
|
|
33
|
+
"__version__",
|
|
34
|
+
"create_missing_features",
|
|
35
|
+
"expand_attribute_strings",
|
|
34
36
|
"parse_gtf",
|
|
37
|
+
"parse_gtf_and_expand_attributes",
|
|
35
38
|
"parse_gtf_pandas",
|
|
39
|
+
"read_gtf",
|
|
36
40
|
]
|
|
@@ -18,12 +18,7 @@ logging.basicConfig(level=logging.INFO)
|
|
|
18
18
|
logger = logging.getLogger(__name__)
|
|
19
19
|
|
|
20
20
|
|
|
21
|
-
|
|
22
|
-
def expand_attribute_strings(
|
|
23
|
-
attribute_strings,
|
|
24
|
-
quote_char="'",
|
|
25
|
-
missing_value="",
|
|
26
|
-
usecols=None):
|
|
21
|
+
def expand_attribute_strings(attribute_strings, quote_char="'", missing_value="", usecols=None):
|
|
27
22
|
"""
|
|
28
23
|
The last column of GTF has a variable number of key value pairs
|
|
29
24
|
of the format: "key1 value1; key2 value2;"
|
|
@@ -66,7 +61,7 @@ def expand_attribute_strings(
|
|
|
66
61
|
# and pair of try/except blocks in the loop.
|
|
67
62
|
column_interned_strings = {}
|
|
68
63
|
|
|
69
|
-
for
|
|
64
|
+
for i, kv_strings in enumerate(attribute_strings):
|
|
70
65
|
if type(kv_strings) is str:
|
|
71
66
|
kv_strings = kv_strings.split(";")
|
|
72
67
|
for kv in kv_strings:
|
|
@@ -92,7 +87,7 @@ def expand_attribute_strings(
|
|
|
92
87
|
|
|
93
88
|
if value[0] == quote_char:
|
|
94
89
|
value = value.replace(quote_char, "")
|
|
95
|
-
|
|
90
|
+
|
|
96
91
|
try:
|
|
97
92
|
column = extra_columns[column_name]
|
|
98
93
|
# if an attribute is used repeatedly then
|
|
@@ -108,9 +103,5 @@ def expand_attribute_strings(
|
|
|
108
103
|
extra_columns[column_name] = column
|
|
109
104
|
column_order.append(column_name)
|
|
110
105
|
|
|
111
|
-
|
|
112
|
-
|
|
113
106
|
logging.info("Extracted GTF attributes: %s" % column_order)
|
|
114
|
-
return OrderedDict(
|
|
115
|
-
(column_name, extra_columns[column_name])
|
|
116
|
-
for column_name in column_order)
|
|
107
|
+
return OrderedDict((column_name, extra_columns[column_name]) for column_name in column_order)
|
|
@@ -19,11 +19,7 @@ logging.basicConfig(level=logging.INFO)
|
|
|
19
19
|
logger = logging.getLogger(__name__)
|
|
20
20
|
|
|
21
21
|
|
|
22
|
-
def create_missing_features(
|
|
23
|
-
dataframe,
|
|
24
|
-
unique_keys={},
|
|
25
|
-
extra_columns={},
|
|
26
|
-
missing_value=None):
|
|
22
|
+
def create_missing_features(dataframe, unique_keys={}, extra_columns={}, missing_value=None):
|
|
27
23
|
"""
|
|
28
24
|
Helper function used to construct a missing feature such as 'transcript'
|
|
29
25
|
or 'gene'. Some GTF files only have 'exon' and 'CDS' entries, but have
|
|
@@ -49,29 +45,25 @@ def create_missing_features(
|
|
|
49
45
|
missing_value : any
|
|
50
46
|
Which value to fill in for columns that we don't infer values for.
|
|
51
47
|
|
|
52
|
-
Returns original dataframe (converted to Pandas if necessary) along with all
|
|
48
|
+
Returns original dataframe (converted to Pandas if necessary) along with all
|
|
53
49
|
extra rows created for missing features.
|
|
54
50
|
"""
|
|
55
51
|
if hasattr(dataframe, "to_pandas"):
|
|
56
52
|
dataframe = dataframe.to_pandas()
|
|
57
|
-
|
|
53
|
+
|
|
58
54
|
extra_dataframes = []
|
|
59
55
|
|
|
60
56
|
existing_features = set(dataframe["feature"])
|
|
61
57
|
existing_columns = set(dataframe.columns)
|
|
62
|
-
|
|
63
|
-
for
|
|
64
|
-
|
|
58
|
+
|
|
59
|
+
for feature_name, groupby_key in unique_keys.items():
|
|
65
60
|
if feature_name in existing_features:
|
|
66
|
-
logging.info(
|
|
67
|
-
"Feature '%s' already exists in GTF data" % feature_name)
|
|
61
|
+
logging.info("Feature '%s' already exists in GTF data" % feature_name)
|
|
68
62
|
continue
|
|
69
63
|
logging.info("Creating rows for missing feature '%s'" % feature_name)
|
|
70
64
|
|
|
71
65
|
# don't include rows where the groupby key was missing
|
|
72
|
-
missing = pd.Series([
|
|
73
|
-
x is None or x == ""
|
|
74
|
-
for x in dataframe[groupby_key]])
|
|
66
|
+
missing = pd.Series([x is None or x == "" for x in dataframe[groupby_key]])
|
|
75
67
|
not_missing = ~missing
|
|
76
68
|
row_groups = dataframe[not_missing].groupby(groupby_key)
|
|
77
69
|
|
|
@@ -79,10 +71,9 @@ def create_missing_features(
|
|
|
79
71
|
# other columns may or may not be uniquely defined. Start off by
|
|
80
72
|
# assuming the values for every column are missing and fill them in
|
|
81
73
|
# where possible.
|
|
82
|
-
feature_values = OrderedDict(
|
|
83
|
-
(column_name, [missing_value] * row_groups.ngroups)
|
|
84
|
-
|
|
85
|
-
])
|
|
74
|
+
feature_values = OrderedDict(
|
|
75
|
+
[(column_name, [missing_value] * row_groups.ngroups) for column_name in dataframe]
|
|
76
|
+
)
|
|
86
77
|
|
|
87
78
|
# User specifies which non-required columns should we try to infer
|
|
88
79
|
# values for
|
|
@@ -111,8 +102,9 @@ def create_missing_features(
|
|
|
111
102
|
for column_name in feature_columns:
|
|
112
103
|
if column_name not in existing_columns:
|
|
113
104
|
raise ValueError(
|
|
114
|
-
"Column '%s' does not exist in GTF, columns = %s"
|
|
115
|
-
|
|
105
|
+
"Column '%s' does not exist in GTF, columns = %s"
|
|
106
|
+
% (column_name, existing_columns)
|
|
107
|
+
)
|
|
116
108
|
|
|
117
109
|
# expect that all entries related to a reconstructed feature
|
|
118
110
|
# are related and are thus within the same interval of
|
|
@@ -121,4 +113,4 @@ def create_missing_features(
|
|
|
121
113
|
if len(unique_values) == 1:
|
|
122
114
|
feature_values[column_name][i] = unique_values[0]
|
|
123
115
|
extra_dataframes.append(pd.DataFrame(feature_values))
|
|
124
|
-
return pd.concat([dataframe
|
|
116
|
+
return pd.concat([dataframe, *extra_dataframes], ignore_index=True)
|
|
@@ -13,16 +13,37 @@
|
|
|
13
13
|
import logging
|
|
14
14
|
from os.path import exists
|
|
15
15
|
|
|
16
|
-
import
|
|
16
|
+
import pandas as pd
|
|
17
|
+
import polars
|
|
17
18
|
|
|
18
19
|
from .attribute_parsing import expand_attribute_strings
|
|
19
20
|
from .parsing_error import ParsingError
|
|
20
21
|
|
|
21
|
-
|
|
22
22
|
logging.basicConfig(level=logging.INFO)
|
|
23
23
|
logger = logging.getLogger(__name__)
|
|
24
24
|
|
|
25
25
|
|
|
26
|
+
# GENCODE GTFs use *_type where Ensembl GTFs use *_biotype. Pass this
|
|
27
|
+
# (or a superset) as `attribute_aliases` to read_gtf to normalize a
|
|
28
|
+
# GENCODE-format GTF onto the Ensembl column names that downstream
|
|
29
|
+
# tools like pyensembl expect.
|
|
30
|
+
GENCODE_BIOTYPE_ALIASES = {
|
|
31
|
+
"gene_type": "gene_biotype",
|
|
32
|
+
"transcript_type": "transcript_biotype",
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
# Ensembl-style attribute columns that are always integer-valued when
|
|
37
|
+
# present. read_gtf casts these from string to pandas nullable Int64 by
|
|
38
|
+
# default; pass cast_version_columns=False to keep them as strings.
|
|
39
|
+
INTEGER_VERSION_COLUMNS = (
|
|
40
|
+
"gene_version",
|
|
41
|
+
"transcript_version",
|
|
42
|
+
"protein_version",
|
|
43
|
+
"exon_version",
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
|
|
26
47
|
"""
|
|
27
48
|
Columns of a GTF file:
|
|
28
49
|
|
|
@@ -76,88 +97,79 @@ REQUIRED_COLUMNS = [
|
|
|
76
97
|
|
|
77
98
|
|
|
78
99
|
DEFAULT_COLUMN_DTYPES = {
|
|
79
|
-
"seqname": polars.Categorical,
|
|
80
|
-
"source": polars.Categorical,
|
|
81
|
-
|
|
100
|
+
"seqname": polars.Categorical,
|
|
101
|
+
"source": polars.Categorical,
|
|
82
102
|
"start": polars.Int64,
|
|
83
103
|
"end": polars.Int64,
|
|
84
104
|
"score": polars.Float32,
|
|
85
|
-
|
|
86
|
-
"
|
|
87
|
-
"strand": polars.Categorical,
|
|
105
|
+
"feature": polars.Categorical,
|
|
106
|
+
"strand": polars.Categorical,
|
|
88
107
|
"frame": polars.UInt32,
|
|
89
108
|
}
|
|
90
109
|
|
|
110
|
+
|
|
91
111
|
def parse_with_polars_lazy(
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
features=None,
|
|
95
|
-
fix_quotes_columns=["attribute"]):
|
|
112
|
+
filepath_or_buffer, split_attributes=True, features=None, fix_quotes_columns=["attribute"]
|
|
113
|
+
):
|
|
96
114
|
# use a global string cache so that all strings get intern'd into
|
|
97
115
|
# a single numbering system
|
|
98
116
|
polars.enable_string_cache()
|
|
99
|
-
kwargs =
|
|
100
|
-
has_header
|
|
101
|
-
separator
|
|
102
|
-
comment_prefix
|
|
103
|
-
null_values
|
|
104
|
-
schema_overrides
|
|
117
|
+
kwargs = {
|
|
118
|
+
"has_header": False,
|
|
119
|
+
"separator": "\t",
|
|
120
|
+
"comment_prefix": "#",
|
|
121
|
+
"null_values": ".",
|
|
122
|
+
"schema_overrides": DEFAULT_COLUMN_DTYPES,
|
|
123
|
+
}
|
|
105
124
|
try:
|
|
106
|
-
df = polars.read_csv(
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
**kwargs).lazy()
|
|
110
|
-
except polars.exceptions.ShapeError:
|
|
111
|
-
raise ParsingError("Wrong number of columns")
|
|
125
|
+
df = polars.read_csv(filepath_or_buffer, new_columns=REQUIRED_COLUMNS, **kwargs).lazy()
|
|
126
|
+
except polars.exceptions.ShapeError as err:
|
|
127
|
+
raise ParsingError("Wrong number of columns") from err
|
|
112
128
|
|
|
113
129
|
# Drop empty lines that may appear as all-null rows
|
|
114
130
|
df = df.filter(polars.col("seqname").is_not_null())
|
|
115
131
|
|
|
116
|
-
df = df.with_columns(
|
|
117
|
-
polars.col("frame").fill_null(0),
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
132
|
+
df = df.with_columns(
|
|
133
|
+
[polars.col("frame").fill_null(0), polars.col("attribute").str.replace_all('"', "'")]
|
|
134
|
+
)
|
|
135
|
+
|
|
121
136
|
for fix_quotes_column in fix_quotes_columns:
|
|
122
137
|
# Catch mistaken semicolons by replacing "xyz;" with "xyz"
|
|
123
138
|
# Required to do this since the Ensembl GTF for Ensembl
|
|
124
139
|
# release 78 has mistakes such as:
|
|
125
140
|
# gene_name = "PRAMEF6;" transcript_name = "PRAMEF6;-201"
|
|
126
|
-
df = df.with_columns(
|
|
127
|
-
polars.col(fix_quotes_column).str.replace('
|
|
128
|
-
|
|
141
|
+
df = df.with_columns(
|
|
142
|
+
[polars.col(fix_quotes_column).str.replace(';"', '"').str.replace(";-", "-")]
|
|
143
|
+
)
|
|
129
144
|
|
|
130
145
|
if features is not None:
|
|
131
146
|
features = sorted(set(features))
|
|
132
147
|
df = df.filter(polars.col("feature").is_in(features))
|
|
133
148
|
|
|
134
|
-
|
|
135
149
|
if split_attributes:
|
|
136
|
-
df = df.with_columns([
|
|
137
|
-
polars.col("attribute").str.split(";").alias("attribute_split")
|
|
138
|
-
])
|
|
150
|
+
df = df.with_columns([polars.col("attribute").str.split(";").alias("attribute_split")])
|
|
139
151
|
return df
|
|
140
152
|
|
|
153
|
+
|
|
141
154
|
def parse_gtf(
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
features=None,
|
|
145
|
-
fix_quotes_columns=["attribute"]):
|
|
155
|
+
filepath_or_buffer, split_attributes=True, features=None, fix_quotes_columns=["attribute"]
|
|
156
|
+
):
|
|
146
157
|
df_lazy = parse_with_polars_lazy(
|
|
147
158
|
filepath_or_buffer=filepath_or_buffer,
|
|
148
159
|
split_attributes=split_attributes,
|
|
149
160
|
features=features,
|
|
150
|
-
fix_quotes_columns=fix_quotes_columns
|
|
161
|
+
fix_quotes_columns=fix_quotes_columns,
|
|
162
|
+
)
|
|
151
163
|
return df_lazy.collect()
|
|
152
164
|
|
|
165
|
+
|
|
153
166
|
def parse_gtf_pandas(*args, **kwargs):
|
|
154
167
|
return parse_gtf(*args, **kwargs).to_pandas()
|
|
155
168
|
|
|
156
|
-
|
|
169
|
+
|
|
157
170
|
def parse_gtf_and_expand_attributes(
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
features=None):
|
|
171
|
+
filepath_or_buffer, restrict_attribute_columns=None, features=None
|
|
172
|
+
):
|
|
161
173
|
"""
|
|
162
174
|
Parse lines into column->values dictionary and then expand
|
|
163
175
|
the 'attribute' column into multiple columns. This expansion happens
|
|
@@ -177,33 +189,93 @@ def parse_gtf_and_expand_attributes(
|
|
|
177
189
|
features : set or None
|
|
178
190
|
Ignore entries which don't correspond to one of the supplied features
|
|
179
191
|
"""
|
|
180
|
-
df = parse_gtf(
|
|
181
|
-
filepath_or_buffer=filepath_or_buffer,
|
|
182
|
-
features=features,
|
|
183
|
-
split_attributes=True)
|
|
192
|
+
df = parse_gtf(filepath_or_buffer=filepath_or_buffer, features=features, split_attributes=True)
|
|
184
193
|
if type(restrict_attribute_columns) is str:
|
|
185
194
|
restrict_attribute_columns = {restrict_attribute_columns}
|
|
186
195
|
elif restrict_attribute_columns:
|
|
187
196
|
restrict_attribute_columns = set(restrict_attribute_columns)
|
|
188
197
|
df.drop_in_place("attribute")
|
|
189
198
|
attribute_pairs = df.drop_in_place("attribute_split")
|
|
190
|
-
return df.with_columns(
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
199
|
+
return df.with_columns(
|
|
200
|
+
[
|
|
201
|
+
polars.Series(k, vs)
|
|
202
|
+
for (k, vs) in expand_attribute_strings(attribute_pairs).items()
|
|
203
|
+
if restrict_attribute_columns is None or k in restrict_attribute_columns
|
|
204
|
+
]
|
|
205
|
+
)
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def _apply_attribute_aliases(result_df, attribute_aliases):
|
|
209
|
+
"""
|
|
210
|
+
Rename alias attribute columns onto canonical names in-place.
|
|
211
|
+
|
|
212
|
+
For each (alias -> canonical) pair, in iteration order:
|
|
213
|
+
* if only the alias is present, rename it to the canonical name.
|
|
214
|
+
* if both are present, drop the alias and warn (canonical wins).
|
|
215
|
+
* if neither is present, do nothing.
|
|
216
|
+
|
|
217
|
+
When two aliases target the same canonical (e.g. both ``gene_type``
|
|
218
|
+
and a hypothetical ``gene_kind`` map to ``gene_biotype``), the first
|
|
219
|
+
rename in iteration order wins; subsequent aliases targeting an
|
|
220
|
+
already-renamed canonical are treated as collisions, dropped, and
|
|
221
|
+
warned about.
|
|
222
|
+
"""
|
|
223
|
+
if not attribute_aliases:
|
|
224
|
+
return result_df
|
|
225
|
+
columns_present = set(result_df.columns)
|
|
226
|
+
rename_map = {}
|
|
227
|
+
drop_aliases = []
|
|
228
|
+
for alias, canonical in attribute_aliases.items():
|
|
229
|
+
if alias not in columns_present:
|
|
230
|
+
continue
|
|
231
|
+
if canonical in columns_present:
|
|
232
|
+
logger.warning(
|
|
233
|
+
"Both alias column '%s' and canonical column '%s' are present; "
|
|
234
|
+
"dropping alias and keeping canonical values.",
|
|
235
|
+
alias,
|
|
236
|
+
canonical,
|
|
237
|
+
)
|
|
238
|
+
drop_aliases.append(alias)
|
|
239
|
+
else:
|
|
240
|
+
rename_map[alias] = canonical
|
|
241
|
+
# Reflect the rename in the running column set so a later
|
|
242
|
+
# alias mapping to the same canonical sees the collision
|
|
243
|
+
# instead of silently producing a duplicate-named column.
|
|
244
|
+
columns_present.discard(alias)
|
|
245
|
+
columns_present.add(canonical)
|
|
246
|
+
if drop_aliases:
|
|
247
|
+
result_df = result_df.drop(columns=drop_aliases)
|
|
248
|
+
if rename_map:
|
|
249
|
+
result_df = result_df.rename(columns=rename_map)
|
|
250
|
+
return result_df
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def _cast_version_columns(result_df, version_columns=INTEGER_VERSION_COLUMNS):
|
|
254
|
+
"""
|
|
255
|
+
Cast known Ensembl *_version attribute columns from strings to
|
|
256
|
+
pandas nullable Int64 in-place. Missing/empty values become pd.NA.
|
|
257
|
+
"""
|
|
258
|
+
for column_name in version_columns:
|
|
259
|
+
if column_name not in result_df.columns:
|
|
260
|
+
continue
|
|
261
|
+
result_df[column_name] = pd.to_numeric(
|
|
262
|
+
result_df[column_name].replace("", None), errors="coerce"
|
|
263
|
+
).astype("Int64")
|
|
264
|
+
return result_df
|
|
265
|
+
|
|
197
266
|
|
|
198
267
|
def read_gtf(
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
268
|
+
filepath_or_buffer,
|
|
269
|
+
expand_attribute_column=True,
|
|
270
|
+
infer_biotype_column=False,
|
|
271
|
+
column_converters={},
|
|
272
|
+
column_cast_types={},
|
|
273
|
+
usecols=None,
|
|
274
|
+
features=None,
|
|
275
|
+
result_type="polars",
|
|
276
|
+
attribute_aliases=None,
|
|
277
|
+
cast_version_columns=True,
|
|
278
|
+
):
|
|
207
279
|
"""
|
|
208
280
|
Parse a GTF into a dictionary mapping column names to sequences of values.
|
|
209
281
|
|
|
@@ -231,7 +303,7 @@ def read_gtf(
|
|
|
231
303
|
column_cast_types : dict, optional
|
|
232
304
|
Dictionary mapping column names to dtypes. Will cast columns to given
|
|
233
305
|
Polars types.
|
|
234
|
-
|
|
306
|
+
|
|
235
307
|
usecols : list of str or None
|
|
236
308
|
Restrict which columns are loaded to the give set. If None, then
|
|
237
309
|
load all columns.
|
|
@@ -240,17 +312,47 @@ def read_gtf(
|
|
|
240
312
|
Drop rows which aren't one of the features in the supplied set
|
|
241
313
|
|
|
242
314
|
result_type : One of 'polars', 'pandas', or 'dict'
|
|
243
|
-
Default behavior is to return a Polars DataFrame, but will convert to
|
|
315
|
+
Default behavior is to return a Polars DataFrame, but will convert to
|
|
244
316
|
Pandas DataFrame or dictionary if specified.
|
|
317
|
+
|
|
318
|
+
attribute_aliases : dict of str -> str, optional
|
|
319
|
+
Maps alias attribute names onto canonical ones. After attributes
|
|
320
|
+
are expanded into columns, each alias column is renamed to its
|
|
321
|
+
canonical name when the canonical column is absent. If both are
|
|
322
|
+
present the alias is dropped and a warning is logged. Pass
|
|
323
|
+
`GENCODE_BIOTYPE_ALIASES` to normalize a GENCODE GTF's
|
|
324
|
+
`gene_type`/`transcript_type` onto Ensembl's
|
|
325
|
+
`gene_biotype`/`transcript_biotype`.
|
|
326
|
+
|
|
327
|
+
cast_version_columns : bool
|
|
328
|
+
When True (default), cast the well-known integer version
|
|
329
|
+
attribute columns (`gene_version`, `transcript_version`,
|
|
330
|
+
`protein_version`, `exon_version`) from strings to pandas
|
|
331
|
+
nullable Int64 when present. Set to False to keep them as
|
|
332
|
+
strings.
|
|
245
333
|
"""
|
|
246
334
|
if type(filepath_or_buffer) is str and not exists(filepath_or_buffer):
|
|
247
335
|
raise ValueError("GTF file does not exist: %s" % filepath_or_buffer)
|
|
248
336
|
|
|
337
|
+
# If usecols asks for a canonical column that's only present in the
|
|
338
|
+
# GTF under an alias name, expand the parse-time column filter to
|
|
339
|
+
# also pull the alias through — otherwise it gets dropped at parse
|
|
340
|
+
# time before _apply_attribute_aliases can see it. The end-of-function
|
|
341
|
+
# usecols filter still narrows the result down to the canonical name.
|
|
342
|
+
parse_usecols = usecols
|
|
343
|
+
if usecols is not None and attribute_aliases:
|
|
344
|
+
usecols_set = set(usecols)
|
|
345
|
+
parse_usecols = set(usecols_set)
|
|
346
|
+
for alias, canonical in attribute_aliases.items():
|
|
347
|
+
if canonical in usecols_set:
|
|
348
|
+
parse_usecols.add(alias)
|
|
349
|
+
|
|
249
350
|
if expand_attribute_column:
|
|
250
351
|
result_df = parse_gtf_and_expand_attributes(
|
|
251
352
|
filepath_or_buffer,
|
|
252
|
-
restrict_attribute_columns=
|
|
253
|
-
features=features
|
|
353
|
+
restrict_attribute_columns=parse_usecols,
|
|
354
|
+
features=features,
|
|
355
|
+
)
|
|
254
356
|
else:
|
|
255
357
|
result_df = parse_gtf(result_df, features=features)
|
|
256
358
|
|
|
@@ -259,26 +361,36 @@ def read_gtf(
|
|
|
259
361
|
# and are generally insane to chase down
|
|
260
362
|
result_df = result_df.to_pandas()
|
|
261
363
|
if column_converters or column_cast_types:
|
|
364
|
+
|
|
262
365
|
def wrap_to_always_accept_none(f):
|
|
263
366
|
def wrapped_fn(x):
|
|
264
367
|
if x is None or x == "":
|
|
265
368
|
return None
|
|
266
369
|
else:
|
|
267
370
|
return f(x)
|
|
371
|
+
|
|
268
372
|
return wrapped_fn
|
|
269
|
-
|
|
373
|
+
|
|
270
374
|
column_names = set(column_converters.keys()).union(column_cast_types.keys())
|
|
271
375
|
for column_name in column_names:
|
|
272
|
-
|
|
273
376
|
if column_name in column_converters:
|
|
274
|
-
column_fn = wrap_to_always_accept_none(
|
|
275
|
-
column_converters[column_name])
|
|
377
|
+
column_fn = wrap_to_always_accept_none(column_converters[column_name])
|
|
276
378
|
result_df[column_name] = result_df[column_name].apply(column_fn)
|
|
277
379
|
|
|
278
380
|
if column_name in column_cast_types:
|
|
279
381
|
column_type = column_cast_types[column_name]
|
|
280
382
|
result_df[column_name] = result_df[column_name].astype(column_type)
|
|
281
|
-
|
|
383
|
+
|
|
384
|
+
# Rename alias attribute columns onto their canonical names. Done before
|
|
385
|
+
# infer_biotype_column so an aliased gene_biotype/transcript_biotype is
|
|
386
|
+
# visible to the inference logic.
|
|
387
|
+
result_df = _apply_attribute_aliases(result_df, attribute_aliases)
|
|
388
|
+
|
|
389
|
+
# Cast Ensembl *_version columns from strings to nullable integers so
|
|
390
|
+
# downstream consumers (e.g. pyensembl) don't have to int(...) themselves.
|
|
391
|
+
if cast_version_columns:
|
|
392
|
+
result_df = _cast_version_columns(result_df)
|
|
393
|
+
|
|
282
394
|
# Hackishly infer whether the values in the 'source' column of this GTF
|
|
283
395
|
# are actually representing a biotype by checking for the most common
|
|
284
396
|
# gene_biotype and transcript_biotype value 'protein_coding'
|
|
@@ -292,11 +404,11 @@ def read_gtf(
|
|
|
292
404
|
# gene_biotype)
|
|
293
405
|
if "gene_biotype" not in column_names:
|
|
294
406
|
logging.info("Using column 'source' to replace missing 'gene_biotype'")
|
|
295
|
-
result_df[
|
|
407
|
+
result_df["gene_biotype"] = result_df["source"]
|
|
296
408
|
if "transcript_biotype" not in column_names:
|
|
297
409
|
logging.info("Using column 'source' to replace missing 'transcript_biotype'")
|
|
298
|
-
result_df[
|
|
299
|
-
|
|
410
|
+
result_df["transcript_biotype"] = result_df["source"]
|
|
411
|
+
|
|
300
412
|
if usecols is not None:
|
|
301
413
|
column_names = set(result_df.columns)
|
|
302
414
|
valid_columns = [c for c in usecols if c in column_names]
|
|
@@ -1,23 +1,33 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gtfparse
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.7.0
|
|
4
4
|
Summary: Parsing library for extracting data frames of genomic features from GTF files
|
|
5
5
|
Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
|
|
6
6
|
Project-URL: Homepage, https://github.com/openvax/gtfparse
|
|
7
|
-
Project-URL: Bug Tracker, https://github.com/openvax/gtfparse
|
|
7
|
+
Project-URL: Bug Tracker, https://github.com/openvax/gtfparse/issues
|
|
8
8
|
Classifier: Development Status :: 4 - Beta
|
|
9
9
|
Classifier: Environment :: Console
|
|
10
10
|
Classifier: Operating System :: OS Independent
|
|
11
11
|
Classifier: Intended Audience :: Science/Research
|
|
12
12
|
Classifier: License :: OSI Approved :: Apache Software License
|
|
13
13
|
Classifier: Programming Language :: Python
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
14
19
|
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
15
|
-
Requires-Python: >=3.
|
|
20
|
+
Requires-Python: >=3.9
|
|
16
21
|
Description-Content-Type: text/markdown
|
|
17
22
|
License-File: LICENSE
|
|
18
23
|
Requires-Dist: polars>=0.20.2
|
|
19
24
|
Requires-Dist: pyarrow>=18.0.0
|
|
20
25
|
Requires-Dist: pandas>=2.1.0
|
|
26
|
+
Provides-Extra: dev
|
|
27
|
+
Requires-Dist: pytest; extra == "dev"
|
|
28
|
+
Requires-Dist: pytest-cov; extra == "dev"
|
|
29
|
+
Requires-Dist: ruff; extra == "dev"
|
|
30
|
+
Requires-Dist: coveralls; extra == "dev"
|
|
21
31
|
Dynamic: license-file
|
|
22
32
|
|
|
23
33
|
[](https://github.com/openvax/gtfparse/actions/workflows/tests.yml)
|
|
@@ -16,6 +16,7 @@ gtfparse/../requirements.txt
|
|
|
16
16
|
tests/test_create_missing_features.py
|
|
17
17
|
tests/test_ensembl_gtf.py
|
|
18
18
|
tests/test_expand_attributes.py
|
|
19
|
+
tests/test_gencode_gtf.py
|
|
19
20
|
tests/test_multiple_values_for_tag_attribute.py
|
|
20
21
|
tests/test_parse_gtf_lines.py
|
|
21
22
|
tests/test_read_stringtie_gtf.py
|