gtfparse 2.7.0__tar.gz → 2.7.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (25) hide show
  1. {gtfparse-2.7.0 → gtfparse-2.7.2}/PKG-INFO +1 -1
  2. {gtfparse-2.7.0 → gtfparse-2.7.2}/gtfparse/__init__.py +1 -1
  3. {gtfparse-2.7.0 → gtfparse-2.7.2}/gtfparse/attribute_parsing.py +1 -2
  4. {gtfparse-2.7.0 → gtfparse-2.7.2}/gtfparse/create_missing_features.py +2 -3
  5. {gtfparse-2.7.0 → gtfparse-2.7.2}/gtfparse/read_gtf.py +6 -4
  6. {gtfparse-2.7.0 → gtfparse-2.7.2}/gtfparse.egg-info/PKG-INFO +1 -1
  7. {gtfparse-2.7.0 → gtfparse-2.7.2}/gtfparse.egg-info/SOURCES.txt +1 -0
  8. gtfparse-2.7.2/tests/test_expand_attribute_column_false.py +74 -0
  9. {gtfparse-2.7.0 → gtfparse-2.7.2}/LICENSE +0 -0
  10. {gtfparse-2.7.0 → gtfparse-2.7.2}/README.md +0 -0
  11. {gtfparse-2.7.0 → gtfparse-2.7.2}/gtfparse/parsing_error.py +0 -0
  12. {gtfparse-2.7.0 → gtfparse-2.7.2}/gtfparse.egg-info/dependency_links.txt +0 -0
  13. {gtfparse-2.7.0 → gtfparse-2.7.2}/gtfparse.egg-info/requires.txt +0 -0
  14. {gtfparse-2.7.0 → gtfparse-2.7.2}/gtfparse.egg-info/top_level.txt +0 -0
  15. {gtfparse-2.7.0 → gtfparse-2.7.2}/pyproject.toml +0 -0
  16. {gtfparse-2.7.0 → gtfparse-2.7.2}/requirements.txt +0 -0
  17. {gtfparse-2.7.0 → gtfparse-2.7.2}/setup.cfg +0 -0
  18. {gtfparse-2.7.0 → gtfparse-2.7.2}/tests/test_create_missing_features.py +0 -0
  19. {gtfparse-2.7.0 → gtfparse-2.7.2}/tests/test_ensembl_gtf.py +0 -0
  20. {gtfparse-2.7.0 → gtfparse-2.7.2}/tests/test_expand_attributes.py +0 -0
  21. {gtfparse-2.7.0 → gtfparse-2.7.2}/tests/test_gencode_gtf.py +0 -0
  22. {gtfparse-2.7.0 → gtfparse-2.7.2}/tests/test_multiple_values_for_tag_attribute.py +0 -0
  23. {gtfparse-2.7.0 → gtfparse-2.7.2}/tests/test_parse_gtf_lines.py +0 -0
  24. {gtfparse-2.7.0 → gtfparse-2.7.2}/tests/test_read_stringtie_gtf.py +0 -0
  25. {gtfparse-2.7.0 → gtfparse-2.7.2}/tests/test_refseq_gtf.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gtfparse
3
- Version: 2.7.0
3
+ Version: 2.7.2
4
4
  Summary: Parsing library for extracting data frames of genomic features from GTF files
5
5
  Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
6
6
  Project-URL: Homepage, https://github.com/openvax/gtfparse
@@ -23,7 +23,7 @@ from .read_gtf import (
23
23
  read_gtf,
24
24
  )
25
25
 
26
- __version__ = "2.7.0"
26
+ __version__ = "2.7.2"
27
27
 
28
28
  __all__ = [
29
29
  "GENCODE_BIOTYPE_ALIASES",
@@ -14,7 +14,6 @@ import logging
14
14
  from collections import OrderedDict
15
15
  from sys import intern
16
16
 
17
- logging.basicConfig(level=logging.INFO)
18
17
  logger = logging.getLogger(__name__)
19
18
 
20
19
 
@@ -103,5 +102,5 @@ def expand_attribute_strings(attribute_strings, quote_char="'", missing_value=""
103
102
  extra_columns[column_name] = column
104
103
  column_order.append(column_name)
105
104
 
106
- logging.info("Extracted GTF attributes: %s" % column_order)
105
+ logger.info("Extracted GTF attributes: %s", column_order)
107
106
  return OrderedDict((column_name, extra_columns[column_name]) for column_name in column_order)
@@ -15,7 +15,6 @@ from collections import OrderedDict
15
15
 
16
16
  import pandas as pd
17
17
 
18
- logging.basicConfig(level=logging.INFO)
19
18
  logger = logging.getLogger(__name__)
20
19
 
21
20
 
@@ -58,9 +57,9 @@ def create_missing_features(dataframe, unique_keys={}, extra_columns={}, missing
58
57
 
59
58
  for feature_name, groupby_key in unique_keys.items():
60
59
  if feature_name in existing_features:
61
- logging.info("Feature '%s' already exists in GTF data" % feature_name)
60
+ logger.info("Feature '%s' already exists in GTF data", feature_name)
62
61
  continue
63
- logging.info("Creating rows for missing feature '%s'" % feature_name)
62
+ logger.info("Creating rows for missing feature '%s'", feature_name)
64
63
 
65
64
  # don't include rows where the groupby key was missing
66
65
  missing = pd.Series([x is None or x == "" for x in dataframe[groupby_key]])
@@ -19,7 +19,6 @@ import polars
19
19
  from .attribute_parsing import expand_attribute_strings
20
20
  from .parsing_error import ParsingError
21
21
 
22
- logging.basicConfig(level=logging.INFO)
23
22
  logger = logging.getLogger(__name__)
24
23
 
25
24
 
@@ -354,7 +353,10 @@ def read_gtf(
354
353
  features=features,
355
354
  )
356
355
  else:
357
- result_df = parse_gtf(result_df, features=features)
356
+ # When the caller opts out of attribute expansion they want the raw
357
+ # 'attribute' column verbatim — no need to also produce the
358
+ # 'attribute_split' helper that parse_gtf adds by default.
359
+ result_df = parse_gtf(filepath_or_buffer, features=features, split_attributes=False)
358
360
 
359
361
  # converting back to pandas here because Polars bugs manifest
360
362
  # as `pyo3_runtime.PanicException: assertion `left == right` failed: impl error`
@@ -403,10 +405,10 @@ def read_gtf(
403
405
  # the 2nd column is the transcript_biotype (otherwise, it's the
404
406
  # gene_biotype)
405
407
  if "gene_biotype" not in column_names:
406
- logging.info("Using column 'source' to replace missing 'gene_biotype'")
408
+ logger.info("Using column 'source' to replace missing 'gene_biotype'")
407
409
  result_df["gene_biotype"] = result_df["source"]
408
410
  if "transcript_biotype" not in column_names:
409
- logging.info("Using column 'source' to replace missing 'transcript_biotype'")
411
+ logger.info("Using column 'source' to replace missing 'transcript_biotype'")
410
412
  result_df["transcript_biotype"] = result_df["source"]
411
413
 
412
414
  if usecols is not None:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gtfparse
3
- Version: 2.7.0
3
+ Version: 2.7.2
4
4
  Summary: Parsing library for extracting data frames of genomic features from GTF files
5
5
  Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
6
6
  Project-URL: Homepage, https://github.com/openvax/gtfparse
@@ -15,6 +15,7 @@ gtfparse.egg-info/top_level.txt
15
15
  gtfparse/../requirements.txt
16
16
  tests/test_create_missing_features.py
17
17
  tests/test_ensembl_gtf.py
18
+ tests/test_expand_attribute_column_false.py
18
19
  tests/test_expand_attributes.py
19
20
  tests/test_gencode_gtf.py
20
21
  tests/test_multiple_values_for_tag_attribute.py
@@ -0,0 +1,74 @@
1
+ """Regression tests for #56: read_gtf(expand_attribute_column=False)
2
+ used to raise NameError because the else branch referenced `result_df`
3
+ before it had been assigned.
4
+ """
5
+
6
+ import pandas as pd
7
+
8
+ from gtfparse import read_gtf
9
+
10
+ from .data import data_path
11
+
12
+ GTF_PATH = data_path("ensembl_grch37.head.gtf")
13
+
14
+
15
+ def test_expand_attribute_column_false_returns_raw_attribute_pandas():
16
+ df = read_gtf(GTF_PATH, expand_attribute_column=False, result_type="pandas")
17
+ assert isinstance(df, pd.DataFrame)
18
+ # raw attribute column is preserved verbatim
19
+ assert "attribute" in df.columns
20
+ # none of the per-key attribute columns are produced
21
+ assert "gene_name" not in df.columns
22
+ assert "transcript_id" not in df.columns
23
+ # the helper 'attribute_split' column from parse_gtf is also suppressed
24
+ assert "attribute_split" not in df.columns
25
+ # core GTF columns are present and populated
26
+ for col in ("seqname", "source", "feature", "start", "end", "strand"):
27
+ assert col in df.columns
28
+ assert len(df) > 0
29
+ # spot-check that the raw attribute string carries the original key/value form
30
+ assert any("gene_id" in val for val in df["attribute"].astype(str))
31
+
32
+
33
+ def test_expand_attribute_column_false_returns_polars():
34
+ df = read_gtf(GTF_PATH, expand_attribute_column=False, result_type="polars")
35
+ # polars dataframe — has columns attribute but no per-key columns
36
+ assert "attribute" in df.columns
37
+ assert "gene_name" not in df.columns
38
+ assert "attribute_split" not in df.columns
39
+
40
+
41
+ def test_expand_attribute_column_false_returns_dict():
42
+ result = read_gtf(GTF_PATH, expand_attribute_column=False, result_type="dict")
43
+ assert isinstance(result, dict)
44
+ assert "attribute" in result
45
+ assert "gene_name" not in result
46
+
47
+
48
+ def test_expand_attribute_column_false_with_features_filter():
49
+ """The features filter must still apply when not expanding."""
50
+ df = read_gtf(
51
+ GTF_PATH,
52
+ expand_attribute_column=False,
53
+ features={"gene"},
54
+ result_type="pandas",
55
+ )
56
+ assert set(df["feature"]) == {"gene"}
57
+
58
+
59
+ def test_expand_attribute_column_false_skips_alias_and_version_logic():
60
+ """When attribute columns aren't expanded, attribute_aliases has
61
+ nothing to rename and cast_version_columns has nothing to cast.
62
+ Neither should raise — both must be graceful no-ops on the raw
63
+ 'attribute'-column-only frame."""
64
+ df = read_gtf(
65
+ GTF_PATH,
66
+ expand_attribute_column=False,
67
+ attribute_aliases={"gene_type": "gene_biotype"},
68
+ cast_version_columns=True,
69
+ result_type="pandas",
70
+ )
71
+ # alias source wasn't in columns → no rename happened → no canonical added
72
+ assert "gene_biotype" not in df.columns
73
+ # version columns weren't present → no cast → still nothing
74
+ assert "gene_version" not in df.columns
File without changes
File without changes
File without changes
File without changes
File without changes