gtfparse 2.7.0__tar.gz → 2.7.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gtfparse-2.7.0 → gtfparse-2.7.2}/PKG-INFO +1 -1
- {gtfparse-2.7.0 → gtfparse-2.7.2}/gtfparse/__init__.py +1 -1
- {gtfparse-2.7.0 → gtfparse-2.7.2}/gtfparse/attribute_parsing.py +1 -2
- {gtfparse-2.7.0 → gtfparse-2.7.2}/gtfparse/create_missing_features.py +2 -3
- {gtfparse-2.7.0 → gtfparse-2.7.2}/gtfparse/read_gtf.py +6 -4
- {gtfparse-2.7.0 → gtfparse-2.7.2}/gtfparse.egg-info/PKG-INFO +1 -1
- {gtfparse-2.7.0 → gtfparse-2.7.2}/gtfparse.egg-info/SOURCES.txt +1 -0
- gtfparse-2.7.2/tests/test_expand_attribute_column_false.py +74 -0
- {gtfparse-2.7.0 → gtfparse-2.7.2}/LICENSE +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.2}/README.md +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.2}/gtfparse/parsing_error.py +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.2}/gtfparse.egg-info/dependency_links.txt +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.2}/gtfparse.egg-info/requires.txt +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.2}/gtfparse.egg-info/top_level.txt +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.2}/pyproject.toml +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.2}/requirements.txt +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.2}/setup.cfg +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.2}/tests/test_create_missing_features.py +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.2}/tests/test_ensembl_gtf.py +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.2}/tests/test_expand_attributes.py +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.2}/tests/test_gencode_gtf.py +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.2}/tests/test_multiple_values_for_tag_attribute.py +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.2}/tests/test_parse_gtf_lines.py +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.2}/tests/test_read_stringtie_gtf.py +0 -0
- {gtfparse-2.7.0 → gtfparse-2.7.2}/tests/test_refseq_gtf.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gtfparse
|
|
3
|
-
Version: 2.7.
|
|
3
|
+
Version: 2.7.2
|
|
4
4
|
Summary: Parsing library for extracting data frames of genomic features from GTF files
|
|
5
5
|
Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
|
|
6
6
|
Project-URL: Homepage, https://github.com/openvax/gtfparse
|
|
@@ -14,7 +14,6 @@ import logging
|
|
|
14
14
|
from collections import OrderedDict
|
|
15
15
|
from sys import intern
|
|
16
16
|
|
|
17
|
-
logging.basicConfig(level=logging.INFO)
|
|
18
17
|
logger = logging.getLogger(__name__)
|
|
19
18
|
|
|
20
19
|
|
|
@@ -103,5 +102,5 @@ def expand_attribute_strings(attribute_strings, quote_char="'", missing_value=""
|
|
|
103
102
|
extra_columns[column_name] = column
|
|
104
103
|
column_order.append(column_name)
|
|
105
104
|
|
|
106
|
-
|
|
105
|
+
logger.info("Extracted GTF attributes: %s", column_order)
|
|
107
106
|
return OrderedDict((column_name, extra_columns[column_name]) for column_name in column_order)
|
|
@@ -15,7 +15,6 @@ from collections import OrderedDict
|
|
|
15
15
|
|
|
16
16
|
import pandas as pd
|
|
17
17
|
|
|
18
|
-
logging.basicConfig(level=logging.INFO)
|
|
19
18
|
logger = logging.getLogger(__name__)
|
|
20
19
|
|
|
21
20
|
|
|
@@ -58,9 +57,9 @@ def create_missing_features(dataframe, unique_keys={}, extra_columns={}, missing
|
|
|
58
57
|
|
|
59
58
|
for feature_name, groupby_key in unique_keys.items():
|
|
60
59
|
if feature_name in existing_features:
|
|
61
|
-
|
|
60
|
+
logger.info("Feature '%s' already exists in GTF data", feature_name)
|
|
62
61
|
continue
|
|
63
|
-
|
|
62
|
+
logger.info("Creating rows for missing feature '%s'", feature_name)
|
|
64
63
|
|
|
65
64
|
# don't include rows where the groupby key was missing
|
|
66
65
|
missing = pd.Series([x is None or x == "" for x in dataframe[groupby_key]])
|
|
@@ -19,7 +19,6 @@ import polars
|
|
|
19
19
|
from .attribute_parsing import expand_attribute_strings
|
|
20
20
|
from .parsing_error import ParsingError
|
|
21
21
|
|
|
22
|
-
logging.basicConfig(level=logging.INFO)
|
|
23
22
|
logger = logging.getLogger(__name__)
|
|
24
23
|
|
|
25
24
|
|
|
@@ -354,7 +353,10 @@ def read_gtf(
|
|
|
354
353
|
features=features,
|
|
355
354
|
)
|
|
356
355
|
else:
|
|
357
|
-
|
|
356
|
+
# When the caller opts out of attribute expansion they want the raw
|
|
357
|
+
# 'attribute' column verbatim — no need to also produce the
|
|
358
|
+
# 'attribute_split' helper that parse_gtf adds by default.
|
|
359
|
+
result_df = parse_gtf(filepath_or_buffer, features=features, split_attributes=False)
|
|
358
360
|
|
|
359
361
|
# converting back to pandas here because Polars bugs manifest
|
|
360
362
|
# as `pyo3_runtime.PanicException: assertion `left == right` failed: impl error`
|
|
@@ -403,10 +405,10 @@ def read_gtf(
|
|
|
403
405
|
# the 2nd column is the transcript_biotype (otherwise, it's the
|
|
404
406
|
# gene_biotype)
|
|
405
407
|
if "gene_biotype" not in column_names:
|
|
406
|
-
|
|
408
|
+
logger.info("Using column 'source' to replace missing 'gene_biotype'")
|
|
407
409
|
result_df["gene_biotype"] = result_df["source"]
|
|
408
410
|
if "transcript_biotype" not in column_names:
|
|
409
|
-
|
|
411
|
+
logger.info("Using column 'source' to replace missing 'transcript_biotype'")
|
|
410
412
|
result_df["transcript_biotype"] = result_df["source"]
|
|
411
413
|
|
|
412
414
|
if usecols is not None:
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gtfparse
|
|
3
|
-
Version: 2.7.
|
|
3
|
+
Version: 2.7.2
|
|
4
4
|
Summary: Parsing library for extracting data frames of genomic features from GTF files
|
|
5
5
|
Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
|
|
6
6
|
Project-URL: Homepage, https://github.com/openvax/gtfparse
|
|
@@ -15,6 +15,7 @@ gtfparse.egg-info/top_level.txt
|
|
|
15
15
|
gtfparse/../requirements.txt
|
|
16
16
|
tests/test_create_missing_features.py
|
|
17
17
|
tests/test_ensembl_gtf.py
|
|
18
|
+
tests/test_expand_attribute_column_false.py
|
|
18
19
|
tests/test_expand_attributes.py
|
|
19
20
|
tests/test_gencode_gtf.py
|
|
20
21
|
tests/test_multiple_values_for_tag_attribute.py
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
"""Regression tests for #56: read_gtf(expand_attribute_column=False)
|
|
2
|
+
used to raise NameError because the else branch referenced `result_df`
|
|
3
|
+
before it had been assigned.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import pandas as pd
|
|
7
|
+
|
|
8
|
+
from gtfparse import read_gtf
|
|
9
|
+
|
|
10
|
+
from .data import data_path
|
|
11
|
+
|
|
12
|
+
GTF_PATH = data_path("ensembl_grch37.head.gtf")
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def test_expand_attribute_column_false_returns_raw_attribute_pandas():
|
|
16
|
+
df = read_gtf(GTF_PATH, expand_attribute_column=False, result_type="pandas")
|
|
17
|
+
assert isinstance(df, pd.DataFrame)
|
|
18
|
+
# raw attribute column is preserved verbatim
|
|
19
|
+
assert "attribute" in df.columns
|
|
20
|
+
# none of the per-key attribute columns are produced
|
|
21
|
+
assert "gene_name" not in df.columns
|
|
22
|
+
assert "transcript_id" not in df.columns
|
|
23
|
+
# the helper 'attribute_split' column from parse_gtf is also suppressed
|
|
24
|
+
assert "attribute_split" not in df.columns
|
|
25
|
+
# core GTF columns are present and populated
|
|
26
|
+
for col in ("seqname", "source", "feature", "start", "end", "strand"):
|
|
27
|
+
assert col in df.columns
|
|
28
|
+
assert len(df) > 0
|
|
29
|
+
# spot-check that the raw attribute string carries the original key/value form
|
|
30
|
+
assert any("gene_id" in val for val in df["attribute"].astype(str))
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def test_expand_attribute_column_false_returns_polars():
|
|
34
|
+
df = read_gtf(GTF_PATH, expand_attribute_column=False, result_type="polars")
|
|
35
|
+
# polars dataframe — has columns attribute but no per-key columns
|
|
36
|
+
assert "attribute" in df.columns
|
|
37
|
+
assert "gene_name" not in df.columns
|
|
38
|
+
assert "attribute_split" not in df.columns
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def test_expand_attribute_column_false_returns_dict():
|
|
42
|
+
result = read_gtf(GTF_PATH, expand_attribute_column=False, result_type="dict")
|
|
43
|
+
assert isinstance(result, dict)
|
|
44
|
+
assert "attribute" in result
|
|
45
|
+
assert "gene_name" not in result
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def test_expand_attribute_column_false_with_features_filter():
|
|
49
|
+
"""The features filter must still apply when not expanding."""
|
|
50
|
+
df = read_gtf(
|
|
51
|
+
GTF_PATH,
|
|
52
|
+
expand_attribute_column=False,
|
|
53
|
+
features={"gene"},
|
|
54
|
+
result_type="pandas",
|
|
55
|
+
)
|
|
56
|
+
assert set(df["feature"]) == {"gene"}
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def test_expand_attribute_column_false_skips_alias_and_version_logic():
|
|
60
|
+
"""When attribute columns aren't expanded, attribute_aliases has
|
|
61
|
+
nothing to rename and cast_version_columns has nothing to cast.
|
|
62
|
+
Neither should raise — both must be graceful no-ops on the raw
|
|
63
|
+
'attribute'-column-only frame."""
|
|
64
|
+
df = read_gtf(
|
|
65
|
+
GTF_PATH,
|
|
66
|
+
expand_attribute_column=False,
|
|
67
|
+
attribute_aliases={"gene_type": "gene_biotype"},
|
|
68
|
+
cast_version_columns=True,
|
|
69
|
+
result_type="pandas",
|
|
70
|
+
)
|
|
71
|
+
# alias source wasn't in columns → no rename happened → no canonical added
|
|
72
|
+
assert "gene_biotype" not in df.columns
|
|
73
|
+
# version columns weren't present → no cast → still nothing
|
|
74
|
+
assert "gene_version" not in df.columns
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|