gtfparse 2.7.0__tar.gz → 2.7.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gtfparse-2.7.0 → gtfparse-2.7.1}/PKG-INFO +1 -1
- {gtfparse-2.7.0 → gtfparse-2.7.1}/gtfparse/__init__.py +1 -1
- {gtfparse-2.7.0 → gtfparse-2.7.1}/gtfparse/read_gtf.py +4 -1
- {gtfparse-2.7.0 → gtfparse-2.7.1}/gtfparse.egg-info/PKG-INFO +1 -1
- {gtfparse-2.7.0 → gtfparse-2.7.1}/gtfparse.egg-info/SOURCES.txt +1 -0
- gtfparse-2.7.1/tests/test_expand_attribute_column_false.py +74 -0
- {gtfparse-2.7.0 → gtfparse-2.7.1}/LICENSE +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.1}/README.md +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.1}/gtfparse/attribute_parsing.py +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.1}/gtfparse/create_missing_features.py +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.1}/gtfparse/parsing_error.py +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.1}/gtfparse.egg-info/dependency_links.txt +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.1}/gtfparse.egg-info/requires.txt +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.1}/gtfparse.egg-info/top_level.txt +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.1}/pyproject.toml +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.1}/requirements.txt +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.1}/setup.cfg +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.1}/tests/test_create_missing_features.py +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.1}/tests/test_ensembl_gtf.py +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.1}/tests/test_expand_attributes.py +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.1}/tests/test_gencode_gtf.py +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.1}/tests/test_multiple_values_for_tag_attribute.py +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.1}/tests/test_parse_gtf_lines.py +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.1}/tests/test_read_stringtie_gtf.py +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.1}/tests/test_refseq_gtf.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gtfparse
|
|
3
|
-
Version: 2.7.
|
|
3
|
+
Version: 2.7.1
|
|
4
4
|
Summary: Parsing library for extracting data frames of genomic features from GTF files
|
|
5
5
|
Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
|
|
6
6
|
Project-URL: Homepage, https://github.com/openvax/gtfparse
|
|
@@ -354,7 +354,10 @@ def read_gtf(
|
|
|
354
354
|
features=features,
|
|
355
355
|
)
|
|
356
356
|
else:
|
|
357
|
-
|
|
357
|
+
# When the caller opts out of attribute expansion they want the raw
|
|
358
|
+
# 'attribute' column verbatim — no need to also produce the
|
|
359
|
+
# 'attribute_split' helper that parse_gtf adds by default.
|
|
360
|
+
result_df = parse_gtf(filepath_or_buffer, features=features, split_attributes=False)
|
|
358
361
|
|
|
359
362
|
# converting back to pandas here because Polars bugs manifest
|
|
360
363
|
# as `pyo3_runtime.PanicException: assertion `left == right` failed: impl error`
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gtfparse
|
|
3
|
-
Version: 2.7.
|
|
3
|
+
Version: 2.7.1
|
|
4
4
|
Summary: Parsing library for extracting data frames of genomic features from GTF files
|
|
5
5
|
Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
|
|
6
6
|
Project-URL: Homepage, https://github.com/openvax/gtfparse
|
|
@@ -15,6 +15,7 @@ gtfparse.egg-info/top_level.txt
|
|
|
15
15
|
gtfparse/../requirements.txt
|
|
16
16
|
tests/test_create_missing_features.py
|
|
17
17
|
tests/test_ensembl_gtf.py
|
|
18
|
+
tests/test_expand_attribute_column_false.py
|
|
18
19
|
tests/test_expand_attributes.py
|
|
19
20
|
tests/test_gencode_gtf.py
|
|
20
21
|
tests/test_multiple_values_for_tag_attribute.py
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
"""Regression tests for #56: read_gtf(expand_attribute_column=False)
|
|
2
|
+
used to raise NameError because the else branch referenced `result_df`
|
|
3
|
+
before it had been assigned.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import pandas as pd
|
|
7
|
+
|
|
8
|
+
from gtfparse import read_gtf
|
|
9
|
+
|
|
10
|
+
from .data import data_path
|
|
11
|
+
|
|
12
|
+
GTF_PATH = data_path("ensembl_grch37.head.gtf")
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def test_expand_attribute_column_false_returns_raw_attribute_pandas():
|
|
16
|
+
df = read_gtf(GTF_PATH, expand_attribute_column=False, result_type="pandas")
|
|
17
|
+
assert isinstance(df, pd.DataFrame)
|
|
18
|
+
# raw attribute column is preserved verbatim
|
|
19
|
+
assert "attribute" in df.columns
|
|
20
|
+
# none of the per-key attribute columns are produced
|
|
21
|
+
assert "gene_name" not in df.columns
|
|
22
|
+
assert "transcript_id" not in df.columns
|
|
23
|
+
# the helper 'attribute_split' column from parse_gtf is also suppressed
|
|
24
|
+
assert "attribute_split" not in df.columns
|
|
25
|
+
# core GTF columns are present and populated
|
|
26
|
+
for col in ("seqname", "source", "feature", "start", "end", "strand"):
|
|
27
|
+
assert col in df.columns
|
|
28
|
+
assert len(df) > 0
|
|
29
|
+
# spot-check that the raw attribute string carries the original key/value form
|
|
30
|
+
assert any("gene_id" in val for val in df["attribute"].astype(str))
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def test_expand_attribute_column_false_returns_polars():
|
|
34
|
+
df = read_gtf(GTF_PATH, expand_attribute_column=False, result_type="polars")
|
|
35
|
+
# polars dataframe — has columns attribute but no per-key columns
|
|
36
|
+
assert "attribute" in df.columns
|
|
37
|
+
assert "gene_name" not in df.columns
|
|
38
|
+
assert "attribute_split" not in df.columns
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def test_expand_attribute_column_false_returns_dict():
|
|
42
|
+
result = read_gtf(GTF_PATH, expand_attribute_column=False, result_type="dict")
|
|
43
|
+
assert isinstance(result, dict)
|
|
44
|
+
assert "attribute" in result
|
|
45
|
+
assert "gene_name" not in result
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def test_expand_attribute_column_false_with_features_filter():
|
|
49
|
+
"""The features filter must still apply when not expanding."""
|
|
50
|
+
df = read_gtf(
|
|
51
|
+
GTF_PATH,
|
|
52
|
+
expand_attribute_column=False,
|
|
53
|
+
features={"gene"},
|
|
54
|
+
result_type="pandas",
|
|
55
|
+
)
|
|
56
|
+
assert set(df["feature"]) == {"gene"}
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def test_expand_attribute_column_false_skips_alias_and_version_logic():
|
|
60
|
+
"""When attribute columns aren't expanded, attribute_aliases has
|
|
61
|
+
nothing to rename and cast_version_columns has nothing to cast.
|
|
62
|
+
Neither should raise — both must be graceful no-ops on the raw
|
|
63
|
+
'attribute'-column-only frame."""
|
|
64
|
+
df = read_gtf(
|
|
65
|
+
GTF_PATH,
|
|
66
|
+
expand_attribute_column=False,
|
|
67
|
+
attribute_aliases={"gene_type": "gene_biotype"},
|
|
68
|
+
cast_version_columns=True,
|
|
69
|
+
result_type="pandas",
|
|
70
|
+
)
|
|
71
|
+
# alias source wasn't in columns → no rename happened → no canonical added
|
|
72
|
+
assert "gene_biotype" not in df.columns
|
|
73
|
+
# version columns weren't present → no cast → still nothing
|
|
74
|
+
assert "gene_version" not in df.columns
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|