gtfparse 2.1.0__tar.gz → 2.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gtfparse-2.1.0 → gtfparse-2.3.0}/PKG-INFO +5 -2
- {gtfparse-2.1.0 → gtfparse-2.3.0}/README.md +4 -1
- {gtfparse-2.1.0 → gtfparse-2.3.0}/gtfparse/__init__.py +1 -1
- {gtfparse-2.1.0 → gtfparse-2.3.0}/gtfparse/create_missing_features.py +12 -6
- {gtfparse-2.1.0 → gtfparse-2.3.0}/gtfparse/read_gtf.py +44 -18
- {gtfparse-2.1.0 → gtfparse-2.3.0}/gtfparse.egg-info/PKG-INFO +5 -2
- {gtfparse-2.1.0 → gtfparse-2.3.0}/tests/test_refseq_gtf.py +4 -0
- {gtfparse-2.1.0 → gtfparse-2.3.0}/LICENSE +0 -0
- {gtfparse-2.1.0 → gtfparse-2.3.0}/gtfparse/attribute_parsing.py +0 -0
- {gtfparse-2.1.0 → gtfparse-2.3.0}/gtfparse/parsing_error.py +0 -0
- {gtfparse-2.1.0 → gtfparse-2.3.0}/gtfparse.egg-info/SOURCES.txt +0 -0
- {gtfparse-2.1.0 → gtfparse-2.3.0}/gtfparse.egg-info/dependency_links.txt +0 -0
- {gtfparse-2.1.0 → gtfparse-2.3.0}/gtfparse.egg-info/requires.txt +0 -0
- {gtfparse-2.1.0 → gtfparse-2.3.0}/gtfparse.egg-info/top_level.txt +0 -0
- {gtfparse-2.1.0 → gtfparse-2.3.0}/pyproject.toml +0 -0
- {gtfparse-2.1.0 → gtfparse-2.3.0}/requirements.txt +0 -0
- {gtfparse-2.1.0 → gtfparse-2.3.0}/setup.cfg +0 -0
- {gtfparse-2.1.0 → gtfparse-2.3.0}/tests/test_create_missing_features.py +0 -0
- {gtfparse-2.1.0 → gtfparse-2.3.0}/tests/test_ensembl_gtf.py +0 -0
- {gtfparse-2.1.0 → gtfparse-2.3.0}/tests/test_expand_attributes.py +0 -0
- {gtfparse-2.1.0 → gtfparse-2.3.0}/tests/test_multiple_values_for_tag_attribute.py +0 -0
- {gtfparse-2.1.0 → gtfparse-2.3.0}/tests/test_parse_gtf_lines.py +0 -0
- {gtfparse-2.1.0 → gtfparse-2.3.0}/tests/test_read_stringtie_gtf.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: gtfparse
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.3.0
|
|
4
4
|
Summary: Parsing library for extracting data frames of genomic features from GTF files
|
|
5
5
|
Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
|
|
6
6
|
Project-URL: Homepage, https://github.com/openvax/gtfparse
|
|
@@ -18,7 +18,10 @@ License-File: LICENSE
|
|
|
18
18
|
Requires-Dist: polars<0.21.0,>=0.20.2
|
|
19
19
|
Requires-Dist: pyarrow<14.1.0,>=14.0.2
|
|
20
20
|
|
|
21
|
-
|
|
21
|
+
<!--
|
|
22
|
+
[](https://travis-ci.org/openvax/gtfparse)
|
|
23
|
+
-->
|
|
24
|
+
[](https://coveralls.io/github/openvax/gtfparse?branch=master)
|
|
22
25
|
<a href="https://pypi.python.org/pypi/gtfparse/">
|
|
23
26
|
<img src="https://img.shields.io/pypi/v/gtfparse.svg?maxAge=1000" alt="PyPI" />
|
|
24
27
|
</a>
|
|
@@ -1,4 +1,7 @@
|
|
|
1
|
-
|
|
1
|
+
<!--
|
|
2
|
+
[](https://travis-ci.org/openvax/gtfparse)
|
|
3
|
+
-->
|
|
4
|
+
[](https://coveralls.io/github/openvax/gtfparse?branch=master)
|
|
2
5
|
<a href="https://pypi.python.org/pypi/gtfparse/">
|
|
3
6
|
<img src="https://img.shields.io/pypi/v/gtfparse.svg?maxAge=1000" alt="PyPI" />
|
|
4
7
|
</a>
|
|
@@ -49,15 +49,19 @@ def create_missing_features(
|
|
|
49
49
|
missing_value : any
|
|
50
50
|
Which value to fill in for columns that we don't infer values for.
|
|
51
51
|
|
|
52
|
-
Returns original dataframe
|
|
53
|
-
features.
|
|
52
|
+
Returns original dataframe (converted to Pandas if necessary) along with all
|
|
53
|
+
extra rows created for missing features.
|
|
54
54
|
"""
|
|
55
|
+
if hasattr(dataframe, "to_pandas"):
|
|
56
|
+
dataframe = dataframe.to_pandas()
|
|
57
|
+
|
|
55
58
|
extra_dataframes = []
|
|
56
59
|
|
|
57
60
|
existing_features = set(dataframe["feature"])
|
|
58
61
|
existing_columns = set(dataframe.columns)
|
|
59
|
-
|
|
62
|
+
|
|
60
63
|
for (feature_name, groupby_key) in unique_keys.items():
|
|
64
|
+
|
|
61
65
|
if feature_name in existing_features:
|
|
62
66
|
logging.info(
|
|
63
67
|
"Feature '%s' already exists in GTF data" % feature_name)
|
|
@@ -65,9 +69,11 @@ def create_missing_features(
|
|
|
65
69
|
logging.info("Creating rows for missing feature '%s'" % feature_name)
|
|
66
70
|
|
|
67
71
|
# don't include rows where the groupby key was missing
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
72
|
+
missing = pd.Series([
|
|
73
|
+
x is None or x == ""
|
|
74
|
+
for x in dataframe[groupby_key]])
|
|
75
|
+
not_missing = ~missing
|
|
76
|
+
row_groups = dataframe[not_missing].groupby(groupby_key)
|
|
71
77
|
|
|
72
78
|
# Each group corresponds to a unique feature entry for which the
|
|
73
79
|
# other columns may or may not be uniquely defined. Start off by
|
|
@@ -77,6 +77,19 @@ REQUIRED_COLUMNS = [
|
|
|
77
77
|
]
|
|
78
78
|
|
|
79
79
|
|
|
80
|
+
DEFAULT_COLUMN_DTYPES = {
|
|
81
|
+
"seqname": polars.Categorical,
|
|
82
|
+
"source": polars.Categorical,
|
|
83
|
+
|
|
84
|
+
"start": polars.Int64,
|
|
85
|
+
"end": polars.Int64,
|
|
86
|
+
"score": polars.Float32,
|
|
87
|
+
|
|
88
|
+
"feature": polars.Categorical,
|
|
89
|
+
"strand": polars.Categorical,
|
|
90
|
+
"frame": polars.UInt32,
|
|
91
|
+
}
|
|
92
|
+
|
|
80
93
|
def parse_with_polars_lazy(
|
|
81
94
|
filepath_or_buffer,
|
|
82
95
|
split_attributes=True,
|
|
@@ -90,18 +103,7 @@ def parse_with_polars_lazy(
|
|
|
90
103
|
separator="\t",
|
|
91
104
|
comment_prefix="#",
|
|
92
105
|
null_values=".",
|
|
93
|
-
dtypes=
|
|
94
|
-
"seqname": polars.Categorical,
|
|
95
|
-
"source": polars.Categorical,
|
|
96
|
-
|
|
97
|
-
"start": polars.Int64,
|
|
98
|
-
"end": polars.Int64,
|
|
99
|
-
"score": polars.Float32,
|
|
100
|
-
|
|
101
|
-
"feature": polars.Categorical,
|
|
102
|
-
"strand": polars.Categorical,
|
|
103
|
-
"frame": polars.UInt32,
|
|
104
|
-
})
|
|
106
|
+
dtypes=DEFAULT_COLUMN_DTYPES)
|
|
105
107
|
try:
|
|
106
108
|
if type(filepath_or_buffer) is StringIO:
|
|
107
109
|
df = polars.read_csv(
|
|
@@ -209,6 +211,7 @@ def read_gtf(
|
|
|
209
211
|
expand_attribute_column=True,
|
|
210
212
|
infer_biotype_column=False,
|
|
211
213
|
column_converters={},
|
|
214
|
+
column_cast_types={},
|
|
212
215
|
usecols=None,
|
|
213
216
|
features=None,
|
|
214
217
|
result_type='polars'):
|
|
@@ -236,6 +239,10 @@ def read_gtf(
|
|
|
236
239
|
empty strings with None and otherwise passes them to given conversion
|
|
237
240
|
function.
|
|
238
241
|
|
|
242
|
+
column_cast_types : dict, optional
|
|
243
|
+
Dictionary mapping column names to dtypes. Will cast columns to given
|
|
244
|
+
Polars types.
|
|
245
|
+
|
|
239
246
|
usecols : list of str or None
|
|
240
247
|
Restrict which columns are loaded to the give set. If None, then
|
|
241
248
|
load all columns.
|
|
@@ -258,12 +265,31 @@ def read_gtf(
|
|
|
258
265
|
else:
|
|
259
266
|
result_df = parse_gtf(result_df, features=features)
|
|
260
267
|
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
268
|
+
|
|
269
|
+
if column_converters or column_cast_types:
|
|
270
|
+
# transform columns with user-specified functions and/or cast them to user-specified types
|
|
271
|
+
polars_expressions = []
|
|
272
|
+
|
|
273
|
+
def wrap_to_always_accept_none(f):
|
|
274
|
+
def wrapped_fn(x):
|
|
275
|
+
if x is None or x == "":
|
|
276
|
+
return None
|
|
277
|
+
else:
|
|
278
|
+
return f(x)
|
|
279
|
+
return wrapped_fn
|
|
280
|
+
|
|
281
|
+
column_names = set(column_converters.keys()).union(column_cast_types.keys())
|
|
282
|
+
for column_name in column_names:
|
|
283
|
+
e = polars.col(column_name)
|
|
284
|
+
if column_name in column_converters:
|
|
285
|
+
column_fn = column_converters[column_name]
|
|
286
|
+
e = e.map_elements(wrap_to_always_accept_none(column_fn))
|
|
287
|
+
|
|
288
|
+
if column_name in column_cast_types:
|
|
289
|
+
column_type = column_cast_types[column_name]
|
|
290
|
+
e = e.cast(column_type)
|
|
291
|
+
polars_expressions.append(e)
|
|
292
|
+
result_df = result_df.with_columns(polars_expressions)
|
|
267
293
|
|
|
268
294
|
# Hackishly infer whether the values in the 'source' column of this GTF
|
|
269
295
|
# are actually representing a biotype by checking for the most common
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: gtfparse
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.3.0
|
|
4
4
|
Summary: Parsing library for extracting data frames of genomic features from GTF files
|
|
5
5
|
Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
|
|
6
6
|
Project-URL: Homepage, https://github.com/openvax/gtfparse
|
|
@@ -18,7 +18,10 @@ License-File: LICENSE
|
|
|
18
18
|
Requires-Dist: polars<0.21.0,>=0.20.2
|
|
19
19
|
Requires-Dist: pyarrow<14.1.0,>=14.0.2
|
|
20
20
|
|
|
21
|
-
|
|
21
|
+
<!--
|
|
22
|
+
[](https://travis-ci.org/openvax/gtfparse)
|
|
23
|
+
-->
|
|
24
|
+
[](https://coveralls.io/github/openvax/gtfparse?branch=master)
|
|
22
25
|
<a href="https://pypi.python.org/pypi/gtfparse/">
|
|
23
26
|
<img src="https://img.shields.io/pypi/v/gtfparse.svg?maxAge=1000" alt="PyPI" />
|
|
24
27
|
</a>
|
|
@@ -14,3 +14,7 @@ def _check_required_columns(gtf_dict):
|
|
|
14
14
|
def test_read_refseq_gtf_as_dataframe():
|
|
15
15
|
gtf_df = read_gtf(REFSEQ_GTF_PATH)
|
|
16
16
|
_check_required_columns(gtf_df)
|
|
17
|
+
|
|
18
|
+
def test_read_refseq_and_transform_columns():
|
|
19
|
+
gtf_df = read_gtf(REFSEQ_GTF_PATH, column_converters={"start": int, "end": int}, column_cast_types={"score": float})
|
|
20
|
+
print(gtf_df)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|