gtfparse 2.1.0__tar.gz → 2.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (23) hide show
  1. {gtfparse-2.1.0 → gtfparse-2.3.0}/PKG-INFO +5 -2
  2. {gtfparse-2.1.0 → gtfparse-2.3.0}/README.md +4 -1
  3. {gtfparse-2.1.0 → gtfparse-2.3.0}/gtfparse/__init__.py +1 -1
  4. {gtfparse-2.1.0 → gtfparse-2.3.0}/gtfparse/create_missing_features.py +12 -6
  5. {gtfparse-2.1.0 → gtfparse-2.3.0}/gtfparse/read_gtf.py +44 -18
  6. {gtfparse-2.1.0 → gtfparse-2.3.0}/gtfparse.egg-info/PKG-INFO +5 -2
  7. {gtfparse-2.1.0 → gtfparse-2.3.0}/tests/test_refseq_gtf.py +4 -0
  8. {gtfparse-2.1.0 → gtfparse-2.3.0}/LICENSE +0 -0
  9. {gtfparse-2.1.0 → gtfparse-2.3.0}/gtfparse/attribute_parsing.py +0 -0
  10. {gtfparse-2.1.0 → gtfparse-2.3.0}/gtfparse/parsing_error.py +0 -0
  11. {gtfparse-2.1.0 → gtfparse-2.3.0}/gtfparse.egg-info/SOURCES.txt +0 -0
  12. {gtfparse-2.1.0 → gtfparse-2.3.0}/gtfparse.egg-info/dependency_links.txt +0 -0
  13. {gtfparse-2.1.0 → gtfparse-2.3.0}/gtfparse.egg-info/requires.txt +0 -0
  14. {gtfparse-2.1.0 → gtfparse-2.3.0}/gtfparse.egg-info/top_level.txt +0 -0
  15. {gtfparse-2.1.0 → gtfparse-2.3.0}/pyproject.toml +0 -0
  16. {gtfparse-2.1.0 → gtfparse-2.3.0}/requirements.txt +0 -0
  17. {gtfparse-2.1.0 → gtfparse-2.3.0}/setup.cfg +0 -0
  18. {gtfparse-2.1.0 → gtfparse-2.3.0}/tests/test_create_missing_features.py +0 -0
  19. {gtfparse-2.1.0 → gtfparse-2.3.0}/tests/test_ensembl_gtf.py +0 -0
  20. {gtfparse-2.1.0 → gtfparse-2.3.0}/tests/test_expand_attributes.py +0 -0
  21. {gtfparse-2.1.0 → gtfparse-2.3.0}/tests/test_multiple_values_for_tag_attribute.py +0 -0
  22. {gtfparse-2.1.0 → gtfparse-2.3.0}/tests/test_parse_gtf_lines.py +0 -0
  23. {gtfparse-2.1.0 → gtfparse-2.3.0}/tests/test_read_stringtie_gtf.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: gtfparse
3
- Version: 2.1.0
3
+ Version: 2.3.0
4
4
  Summary: Parsing library for extracting data frames of genomic features from GTF files
5
5
  Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
6
6
  Project-URL: Homepage, https://github.com/openvax/gtfparse
@@ -18,7 +18,10 @@ License-File: LICENSE
18
18
  Requires-Dist: polars<0.21.0,>=0.20.2
19
19
  Requires-Dist: pyarrow<14.1.0,>=14.0.2
20
20
 
21
- [![Build Status](https://travis-ci.org/openvax/gtfparse.svg?branch=master)](https://travis-ci.org/openvax/gtfparse) [![Coverage Status](https://coveralls.io/repos/openvax/gtfparse/badge.svg?branch=master&service=github)](https://coveralls.io/github/openvax/gtfparse?branch=master)
21
+ <!--
22
+ [![Build Status](https://travis-ci.org/openvax/gtfparse.svg?branch=master)](https://travis-ci.org/openvax/gtfparse)
23
+ -->
24
+ [![Coverage Status](https://coveralls.io/repos/openvax/gtfparse/badge.svg?branch=master&service=github)](https://coveralls.io/github/openvax/gtfparse?branch=master)
22
25
  <a href="https://pypi.python.org/pypi/gtfparse/">
23
26
  <img src="https://img.shields.io/pypi/v/gtfparse.svg?maxAge=1000" alt="PyPI" />
24
27
  </a>
@@ -1,4 +1,7 @@
1
- [![Build Status](https://travis-ci.org/openvax/gtfparse.svg?branch=master)](https://travis-ci.org/openvax/gtfparse) [![Coverage Status](https://coveralls.io/repos/openvax/gtfparse/badge.svg?branch=master&service=github)](https://coveralls.io/github/openvax/gtfparse?branch=master)
1
+ <!--
2
+ [![Build Status](https://travis-ci.org/openvax/gtfparse.svg?branch=master)](https://travis-ci.org/openvax/gtfparse)
3
+ -->
4
+ [![Coverage Status](https://coveralls.io/repos/openvax/gtfparse/badge.svg?branch=master&service=github)](https://coveralls.io/github/openvax/gtfparse?branch=master)
2
5
  <a href="https://pypi.python.org/pypi/gtfparse/">
3
6
  <img src="https://img.shields.io/pypi/v/gtfparse.svg?maxAge=1000" alt="PyPI" />
4
7
  </a>
@@ -21,7 +21,7 @@ from .read_gtf import (
21
21
  REQUIRED_COLUMNS,
22
22
  )
23
23
 
24
- __version__ = "2.1.0"
24
+ __version__ = "2.3.0"
25
25
 
26
26
  __all__ = [
27
27
  "__version__",
@@ -49,15 +49,19 @@ def create_missing_features(
49
49
  missing_value : any
50
50
  Which value to fill in for columns that we don't infer values for.
51
51
 
52
- Returns original dataframe along with all extra rows created for missing
53
- features.
52
+ Returns original dataframe (converted to Pandas if necessary) along with all
53
+ extra rows created for missing features.
54
54
  """
55
+ if hasattr(dataframe, "to_pandas"):
56
+ dataframe = dataframe.to_pandas()
57
+
55
58
  extra_dataframes = []
56
59
 
57
60
  existing_features = set(dataframe["feature"])
58
61
  existing_columns = set(dataframe.columns)
59
-
62
+
60
63
  for (feature_name, groupby_key) in unique_keys.items():
64
+
61
65
  if feature_name in existing_features:
62
66
  logging.info(
63
67
  "Feature '%s' already exists in GTF data" % feature_name)
@@ -65,9 +69,11 @@ def create_missing_features(
65
69
  logging.info("Creating rows for missing feature '%s'" % feature_name)
66
70
 
67
71
  # don't include rows where the groupby key was missing
68
- empty_key_values = dataframe[groupby_key].map(
69
- lambda x: x == "" or x is None)
70
- row_groups = dataframe[~empty_key_values].groupby(groupby_key)
72
+ missing = pd.Series([
73
+ x is None or x == ""
74
+ for x in dataframe[groupby_key]])
75
+ not_missing = ~missing
76
+ row_groups = dataframe[not_missing].groupby(groupby_key)
71
77
 
72
78
  # Each group corresponds to a unique feature entry for which the
73
79
  # other columns may or may not be uniquely defined. Start off by
@@ -77,6 +77,19 @@ REQUIRED_COLUMNS = [
77
77
  ]
78
78
 
79
79
 
80
+ DEFAULT_COLUMN_DTYPES = {
81
+ "seqname": polars.Categorical,
82
+ "source": polars.Categorical,
83
+
84
+ "start": polars.Int64,
85
+ "end": polars.Int64,
86
+ "score": polars.Float32,
87
+
88
+ "feature": polars.Categorical,
89
+ "strand": polars.Categorical,
90
+ "frame": polars.UInt32,
91
+ }
92
+
80
93
  def parse_with_polars_lazy(
81
94
  filepath_or_buffer,
82
95
  split_attributes=True,
@@ -90,18 +103,7 @@ def parse_with_polars_lazy(
90
103
  separator="\t",
91
104
  comment_prefix="#",
92
105
  null_values=".",
93
- dtypes={
94
- "seqname": polars.Categorical,
95
- "source": polars.Categorical,
96
-
97
- "start": polars.Int64,
98
- "end": polars.Int64,
99
- "score": polars.Float32,
100
-
101
- "feature": polars.Categorical,
102
- "strand": polars.Categorical,
103
- "frame": polars.UInt32,
104
- })
106
+ dtypes=DEFAULT_COLUMN_DTYPES)
105
107
  try:
106
108
  if type(filepath_or_buffer) is StringIO:
107
109
  df = polars.read_csv(
@@ -209,6 +211,7 @@ def read_gtf(
209
211
  expand_attribute_column=True,
210
212
  infer_biotype_column=False,
211
213
  column_converters={},
214
+ column_cast_types={},
212
215
  usecols=None,
213
216
  features=None,
214
217
  result_type='polars'):
@@ -236,6 +239,10 @@ def read_gtf(
236
239
  empty strings with None and otherwise passes them to given conversion
237
240
  function.
238
241
 
242
+ column_cast_types : dict, optional
243
+ Dictionary mapping column names to dtypes. Will cast columns to given
244
+ Polars types.
245
+
239
246
  usecols : list of str or None
240
247
  Restrict which columns are loaded to the give set. If None, then
241
248
  load all columns.
@@ -258,12 +265,31 @@ def read_gtf(
258
265
  else:
259
266
  result_df = parse_gtf(result_df, features=features)
260
267
 
261
- result_df = result_df.with_columns(
262
- [
263
- polars.col(column_name).map_elements(lambda x: column_type(x) if len(x) > 0 else None)
264
- for column_name, column_type in column_converters.items()
265
- ]
266
- )
268
+
269
+ if column_converters or column_cast_types:
270
+ # transform columns with user-specified functions and/or cast them to user-specified types
271
+ polars_expressions = []
272
+
273
+ def wrap_to_always_accept_none(f):
274
+ def wrapped_fn(x):
275
+ if x is None or x == "":
276
+ return None
277
+ else:
278
+ return f(x)
279
+ return wrapped_fn
280
+
281
+ column_names = set(column_converters.keys()).union(column_cast_types.keys())
282
+ for column_name in column_names:
283
+ e = polars.col(column_name)
284
+ if column_name in column_converters:
285
+ column_fn = column_converters[column_name]
286
+ e = e.map_elements(wrap_to_always_accept_none(column_fn))
287
+
288
+ if column_name in column_cast_types:
289
+ column_type = column_cast_types[column_name]
290
+ e = e.cast(column_type)
291
+ polars_expressions.append(e)
292
+ result_df = result_df.with_columns(polars_expressions)
267
293
 
268
294
  # Hackishly infer whether the values in the 'source' column of this GTF
269
295
  # are actually representing a biotype by checking for the most common
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: gtfparse
3
- Version: 2.1.0
3
+ Version: 2.3.0
4
4
  Summary: Parsing library for extracting data frames of genomic features from GTF files
5
5
  Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
6
6
  Project-URL: Homepage, https://github.com/openvax/gtfparse
@@ -18,7 +18,10 @@ License-File: LICENSE
18
18
  Requires-Dist: polars<0.21.0,>=0.20.2
19
19
  Requires-Dist: pyarrow<14.1.0,>=14.0.2
20
20
 
21
- [![Build Status](https://travis-ci.org/openvax/gtfparse.svg?branch=master)](https://travis-ci.org/openvax/gtfparse) [![Coverage Status](https://coveralls.io/repos/openvax/gtfparse/badge.svg?branch=master&service=github)](https://coveralls.io/github/openvax/gtfparse?branch=master)
21
+ <!--
22
+ [![Build Status](https://travis-ci.org/openvax/gtfparse.svg?branch=master)](https://travis-ci.org/openvax/gtfparse)
23
+ -->
24
+ [![Coverage Status](https://coveralls.io/repos/openvax/gtfparse/badge.svg?branch=master&service=github)](https://coveralls.io/github/openvax/gtfparse?branch=master)
22
25
  <a href="https://pypi.python.org/pypi/gtfparse/">
23
26
  <img src="https://img.shields.io/pypi/v/gtfparse.svg?maxAge=1000" alt="PyPI" />
24
27
  </a>
@@ -14,3 +14,7 @@ def _check_required_columns(gtf_dict):
14
14
  def test_read_refseq_gtf_as_dataframe():
15
15
  gtf_df = read_gtf(REFSEQ_GTF_PATH)
16
16
  _check_required_columns(gtf_df)
17
+
18
+ def test_read_refseq_and_transform_columns():
19
+ gtf_df = read_gtf(REFSEQ_GTF_PATH, column_converters={"start": int, "end": int}, column_cast_types={"score": float})
20
+ print(gtf_df)
File without changes
File without changes
File without changes
File without changes