gtfparse 2.4.0__tar.gz → 2.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (23) hide show
  1. {gtfparse-2.4.0 → gtfparse-2.5.0}/PKG-INFO +1 -1
  2. {gtfparse-2.4.0 → gtfparse-2.5.0}/gtfparse/__init__.py +1 -1
  3. {gtfparse-2.4.0 → gtfparse-2.5.0}/gtfparse/read_gtf.py +16 -19
  4. {gtfparse-2.4.0 → gtfparse-2.5.0}/gtfparse.egg-info/PKG-INFO +1 -1
  5. {gtfparse-2.4.0 → gtfparse-2.5.0}/LICENSE +0 -0
  6. {gtfparse-2.4.0 → gtfparse-2.5.0}/README.md +0 -0
  7. {gtfparse-2.4.0 → gtfparse-2.5.0}/gtfparse/attribute_parsing.py +0 -0
  8. {gtfparse-2.4.0 → gtfparse-2.5.0}/gtfparse/create_missing_features.py +0 -0
  9. {gtfparse-2.4.0 → gtfparse-2.5.0}/gtfparse/parsing_error.py +0 -0
  10. {gtfparse-2.4.0 → gtfparse-2.5.0}/gtfparse.egg-info/SOURCES.txt +0 -0
  11. {gtfparse-2.4.0 → gtfparse-2.5.0}/gtfparse.egg-info/dependency_links.txt +0 -0
  12. {gtfparse-2.4.0 → gtfparse-2.5.0}/gtfparse.egg-info/requires.txt +0 -0
  13. {gtfparse-2.4.0 → gtfparse-2.5.0}/gtfparse.egg-info/top_level.txt +0 -0
  14. {gtfparse-2.4.0 → gtfparse-2.5.0}/pyproject.toml +0 -0
  15. {gtfparse-2.4.0 → gtfparse-2.5.0}/requirements.txt +0 -0
  16. {gtfparse-2.4.0 → gtfparse-2.5.0}/setup.cfg +0 -0
  17. {gtfparse-2.4.0 → gtfparse-2.5.0}/tests/test_create_missing_features.py +0 -0
  18. {gtfparse-2.4.0 → gtfparse-2.5.0}/tests/test_ensembl_gtf.py +0 -0
  19. {gtfparse-2.4.0 → gtfparse-2.5.0}/tests/test_expand_attributes.py +0 -0
  20. {gtfparse-2.4.0 → gtfparse-2.5.0}/tests/test_multiple_values_for_tag_attribute.py +0 -0
  21. {gtfparse-2.4.0 → gtfparse-2.5.0}/tests/test_parse_gtf_lines.py +0 -0
  22. {gtfparse-2.4.0 → gtfparse-2.5.0}/tests/test_read_stringtie_gtf.py +0 -0
  23. {gtfparse-2.4.0 → gtfparse-2.5.0}/tests/test_refseq_gtf.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: gtfparse
3
- Version: 2.4.0
3
+ Version: 2.5.0
4
4
  Summary: Parsing library for extracting data frames of genomic features from GTF files
5
5
  Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
6
6
  Project-URL: Homepage, https://github.com/openvax/gtfparse
@@ -21,7 +21,7 @@ from .read_gtf import (
21
21
  REQUIRED_COLUMNS,
22
22
  )
23
23
 
24
- __version__ = "2.4.0"
24
+ __version__ = "2.5.0"
25
25
 
26
26
  __all__ = [
27
27
  "__version__",
@@ -12,8 +12,6 @@
12
12
 
13
13
  import logging
14
14
  from os.path import exists
15
- from io import StringIO
16
- import gzip
17
15
 
18
16
  import polars
19
17
 
@@ -253,11 +251,11 @@ def read_gtf(
253
251
  else:
254
252
  result_df = parse_gtf(result_df, features=features)
255
253
 
256
-
254
+ # converting back to pandas here because Polars bugs manifest
255
+ # as `pyo3_runtime.PanicException: assertion `left == right` failed: impl error`
256
+ # and are generally insane to chase down
257
+ result_df = result_df.to_pandas()
257
258
  if column_converters or column_cast_types:
258
- # transform columns with user-specified functions and/or cast them to user-specified types
259
- polars_expressions = []
260
-
261
259
  def wrap_to_always_accept_none(f):
262
260
  def wrapped_fn(x):
263
261
  if x is None or x == "":
@@ -268,17 +266,16 @@ def read_gtf(
268
266
 
269
267
  column_names = set(column_converters.keys()).union(column_cast_types.keys())
270
268
  for column_name in column_names:
271
- e = polars.col(column_name)
269
+
272
270
  if column_name in column_converters:
273
- column_fn = column_converters[column_name]
274
- e = e.map_elements(wrap_to_always_accept_none(column_fn))
271
+ column_fn = wrap_to_always_accept_none(
272
+ column_converters[column_name])
273
+ result_df[column_name] = result_df[column_name].apply(column_fn)
275
274
 
276
275
  if column_name in column_cast_types:
277
276
  column_type = column_cast_types[column_name]
278
- e = e.cast(column_type)
279
- polars_expressions.append(e)
280
- result_df = result_df.with_columns(polars_expressions)
281
-
277
+ result_df[column_name] = result_df[column_name].astype(column_type)
278
+
282
279
  # Hackishly infer whether the values in the 'source' column of this GTF
283
280
  # are actually representing a biotype by checking for the most common
284
281
  # gene_biotype and transcript_biotype value 'protein_coding'
@@ -292,20 +289,20 @@ def read_gtf(
292
289
  # gene_biotype)
293
290
  if "gene_biotype" not in column_names:
294
291
  logging.info("Using column 'source' to replace missing 'gene_biotype'")
295
- result_df = result_df.with_column(polars.col("source").alias("gene_biotype"))
292
+ result_df['gene_biotype'] = result_df['source']
296
293
  if "transcript_biotype" not in column_names:
297
294
  logging.info("Using column 'source' to replace missing 'transcript_biotype'")
298
- result_df = result_df.with_column(polars.col("source").alias("transcript_biotype"))
299
-
295
+ result_df['transcript_biotype'] = result_df['source']
296
+
300
297
  if usecols is not None:
301
298
  column_names = set(result_df.columns)
302
299
  valid_columns = [c for c in usecols if c in column_names]
303
- result_df = result_df.select(valid_columns)
300
+ result_df = result_df[valid_columns]
304
301
 
305
302
  if result_type == "pandas":
306
- result = result_df.to_pandas()
307
- elif result_type == "polars":
308
303
  result = result_df
304
+ elif result_type == "polars":
305
+ result = polars.from_pandas(result_df)
309
306
  elif result_type == "dict":
310
307
  result = result_df.to_dict()
311
308
  return result
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: gtfparse
3
- Version: 2.4.0
3
+ Version: 2.5.0
4
4
  Summary: Parsing library for extracting data frames of genomic features from GTF files
5
5
  Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
6
6
  Project-URL: Homepage, https://github.com/openvax/gtfparse
File without changes
File without changes
File without changes
File without changes
File without changes