gtfparse 2.7.2__tar.gz → 2.8.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gtfparse-2.7.2 → gtfparse-2.8.0}/PKG-INFO +1 -1
- {gtfparse-2.7.2 → gtfparse-2.8.0}/gtfparse/__init__.py +3 -1
- gtfparse-2.8.0/gtfparse/write_gtf.py +168 -0
- {gtfparse-2.7.2 → gtfparse-2.8.0}/gtfparse.egg-info/PKG-INFO +1 -1
- {gtfparse-2.7.2 → gtfparse-2.8.0}/gtfparse.egg-info/SOURCES.txt +3 -1
- gtfparse-2.8.0/tests/test_write_gtf.py +222 -0
- {gtfparse-2.7.2 → gtfparse-2.8.0}/LICENSE +0 -0
- {gtfparse-2.7.2 → gtfparse-2.8.0}/README.md +0 -0
- {gtfparse-2.7.2 → gtfparse-2.8.0}/gtfparse/attribute_parsing.py +0 -0
- {gtfparse-2.7.2 → gtfparse-2.8.0}/gtfparse/create_missing_features.py +0 -0
- {gtfparse-2.7.2 → gtfparse-2.8.0}/gtfparse/parsing_error.py +0 -0
- {gtfparse-2.7.2 → gtfparse-2.8.0}/gtfparse/read_gtf.py +0 -0
- {gtfparse-2.7.2 → gtfparse-2.8.0}/gtfparse.egg-info/dependency_links.txt +0 -0
- {gtfparse-2.7.2 → gtfparse-2.8.0}/gtfparse.egg-info/requires.txt +0 -0
- {gtfparse-2.7.2 → gtfparse-2.8.0}/gtfparse.egg-info/top_level.txt +0 -0
- {gtfparse-2.7.2 → gtfparse-2.8.0}/pyproject.toml +0 -0
- {gtfparse-2.7.2 → gtfparse-2.8.0}/requirements.txt +0 -0
- {gtfparse-2.7.2 → gtfparse-2.8.0}/setup.cfg +0 -0
- {gtfparse-2.7.2 → gtfparse-2.8.0}/tests/test_create_missing_features.py +0 -0
- {gtfparse-2.7.2 → gtfparse-2.8.0}/tests/test_ensembl_gtf.py +0 -0
- {gtfparse-2.7.2 → gtfparse-2.8.0}/tests/test_expand_attribute_column_false.py +0 -0
- {gtfparse-2.7.2 → gtfparse-2.8.0}/tests/test_expand_attributes.py +0 -0
- {gtfparse-2.7.2 → gtfparse-2.8.0}/tests/test_gencode_gtf.py +0 -0
- {gtfparse-2.7.2 → gtfparse-2.8.0}/tests/test_multiple_values_for_tag_attribute.py +0 -0
- {gtfparse-2.7.2 → gtfparse-2.8.0}/tests/test_parse_gtf_lines.py +0 -0
- {gtfparse-2.7.2 → gtfparse-2.8.0}/tests/test_read_stringtie_gtf.py +0 -0
- {gtfparse-2.7.2 → gtfparse-2.8.0}/tests/test_refseq_gtf.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gtfparse
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.8.0
|
|
4
4
|
Summary: Parsing library for extracting data frames of genomic features from GTF files
|
|
5
5
|
Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
|
|
6
6
|
Project-URL: Homepage, https://github.com/openvax/gtfparse
|
|
@@ -22,8 +22,9 @@ from .read_gtf import (
|
|
|
22
22
|
parse_gtf_pandas,
|
|
23
23
|
read_gtf,
|
|
24
24
|
)
|
|
25
|
+
from .write_gtf import write_gtf
|
|
25
26
|
|
|
26
|
-
__version__ = "2.
|
|
27
|
+
__version__ = "2.8.0"
|
|
27
28
|
|
|
28
29
|
__all__ = [
|
|
29
30
|
"GENCODE_BIOTYPE_ALIASES",
|
|
@@ -37,4 +38,5 @@ __all__ = [
|
|
|
37
38
|
"parse_gtf_and_expand_attributes",
|
|
38
39
|
"parse_gtf_pandas",
|
|
39
40
|
"read_gtf",
|
|
41
|
+
"write_gtf",
|
|
40
42
|
]
|
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
2
|
+
# you may not use this file except in compliance with the License.
|
|
3
|
+
# You may obtain a copy of the License at
|
|
4
|
+
#
|
|
5
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
6
|
+
#
|
|
7
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
8
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
9
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
10
|
+
# See the License for the specific language governing permissions and
|
|
11
|
+
# limitations under the License.
|
|
12
|
+
|
|
13
|
+
import gzip
|
|
14
|
+
import logging
|
|
15
|
+
from collections.abc import Iterable
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from typing import TYPE_CHECKING, Optional, Union
|
|
18
|
+
|
|
19
|
+
import polars
|
|
20
|
+
|
|
21
|
+
if TYPE_CHECKING:
|
|
22
|
+
import pandas
|
|
23
|
+
|
|
24
|
+
logger = logging.getLogger(__name__)
|
|
25
|
+
|
|
26
|
+
# The eight tab-separated columns that precede the attribute field of a GTF
|
|
27
|
+
# line, in the order they must be written. Any other column in a DataFrame is
|
|
28
|
+
# treated as an expanded attribute (see read_gtf's expand_attribute_column).
|
|
29
|
+
GTF_FIXED_COLUMNS = [
|
|
30
|
+
"seqname",
|
|
31
|
+
"source",
|
|
32
|
+
"feature",
|
|
33
|
+
"start",
|
|
34
|
+
"end",
|
|
35
|
+
"score",
|
|
36
|
+
"strand",
|
|
37
|
+
"frame",
|
|
38
|
+
]
|
|
39
|
+
|
|
40
|
+
# GTF uses a single dot to denote a missing value in the fixed columns.
|
|
41
|
+
MISSING_VALUE = "."
|
|
42
|
+
|
|
43
|
+
# Name of the raw, unexpanded attribute column produced by
|
|
44
|
+
# read_gtf(expand_attribute_column=False).
|
|
45
|
+
RAW_ATTRIBUTE_COLUMN = "attribute"
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _attribute_expr(attribute_columns: list[str]) -> polars.Expr:
|
|
49
|
+
"""
|
|
50
|
+
Build a polars expression that renders the GTF attribute field for each row
|
|
51
|
+
from a set of expanded attribute columns.
|
|
52
|
+
|
|
53
|
+
Each column ``key`` becomes ``key "value";`` and the per-row pairs are
|
|
54
|
+
joined with a single space, matching the format emitted by Ensembl/GENCODE
|
|
55
|
+
and parsed by read_gtf.
|
|
56
|
+
|
|
57
|
+
A pair is omitted for any row where the value is "absent". read_gtf marks an
|
|
58
|
+
absent key with null (numeric columns, e.g. the ``*_version`` fields) or the
|
|
59
|
+
empty string (string columns) and cannot distinguish absent from
|
|
60
|
+
present-but-empty, so we treat both null and "" as absent. Values that are
|
|
61
|
+
merely falsy but non-empty -- notably the string "0" -- are written and
|
|
62
|
+
survive a round trip (this is the regression that the naive ``if value:``
|
|
63
|
+
check in earlier drafts got wrong).
|
|
64
|
+
"""
|
|
65
|
+
if not attribute_columns:
|
|
66
|
+
return polars.lit("")
|
|
67
|
+
pairs = []
|
|
68
|
+
for name in attribute_columns:
|
|
69
|
+
value = polars.col(name).cast(polars.String)
|
|
70
|
+
pairs.append(
|
|
71
|
+
polars.when(value.is_null() | (value == ""))
|
|
72
|
+
.then(None)
|
|
73
|
+
.otherwise(polars.format('{} "{}";', polars.lit(name), value))
|
|
74
|
+
)
|
|
75
|
+
# ignore_nulls drops absent keys; fill_null covers the all-absent row so the
|
|
76
|
+
# surrounding line expression never collapses to null.
|
|
77
|
+
return polars.concat_str(pairs, separator=" ", ignore_nulls=True).fill_null("")
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _line_series(df: polars.DataFrame) -> polars.Series:
|
|
81
|
+
"""
|
|
82
|
+
Turn a DataFrame into a Series of fully-formatted GTF lines (without
|
|
83
|
+
trailing newlines), building the whole thing with vectorized polars
|
|
84
|
+
expressions rather than per-row Python.
|
|
85
|
+
"""
|
|
86
|
+
columns = df.columns
|
|
87
|
+
missing = [name for name in GTF_FIXED_COLUMNS if name not in columns]
|
|
88
|
+
if missing:
|
|
89
|
+
raise ValueError("DataFrame is missing required GTF column(s): %s" % ", ".join(missing))
|
|
90
|
+
|
|
91
|
+
# Fixed columns are always written in canonical order; nulls become '.'.
|
|
92
|
+
# Cast before fill_null so numeric columns accept the string sentinel.
|
|
93
|
+
fixed = [
|
|
94
|
+
polars.col(name).cast(polars.String).fill_null(MISSING_VALUE) for name in GTF_FIXED_COLUMNS
|
|
95
|
+
]
|
|
96
|
+
|
|
97
|
+
if RAW_ATTRIBUTE_COLUMN in columns:
|
|
98
|
+
# Unexpanded read: the attribute column is already a formatted string.
|
|
99
|
+
attribute = polars.col(RAW_ATTRIBUTE_COLUMN).cast(polars.String).fill_null("")
|
|
100
|
+
else:
|
|
101
|
+
attribute_columns = [name for name in columns if name not in GTF_FIXED_COLUMNS]
|
|
102
|
+
attribute = _attribute_expr(attribute_columns)
|
|
103
|
+
|
|
104
|
+
# Always keep the 9th (attribute) field, even when empty, so every line has
|
|
105
|
+
# the nine tab-separated columns read_gtf expects.
|
|
106
|
+
line = polars.concat_str([*fixed, attribute], separator="\t")
|
|
107
|
+
return df.select(line.alias("_gtf_line")).get_column("_gtf_line")
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def write_gtf(
|
|
111
|
+
df: Union[polars.DataFrame, "pandas.DataFrame"],
|
|
112
|
+
path: Union[str, Path],
|
|
113
|
+
header_lines: Optional[Iterable[str]] = None,
|
|
114
|
+
) -> None:
|
|
115
|
+
"""
|
|
116
|
+
Write a DataFrame of genomic features back out to a GTF file.
|
|
117
|
+
|
|
118
|
+
This is the inverse of :func:`read_gtf`. A DataFrame produced by
|
|
119
|
+
``read_gtf`` (whether the attribute column was expanded into one column per
|
|
120
|
+
key or left as a raw ``attribute`` string) can be written back out and
|
|
121
|
+
re-read to recover an equivalent DataFrame.
|
|
122
|
+
|
|
123
|
+
Parameters
|
|
124
|
+
----------
|
|
125
|
+
df : polars.DataFrame or pandas.DataFrame
|
|
126
|
+
Feature rows to write. Must contain the fixed GTF columns
|
|
127
|
+
(seqname, source, feature, start, end, score, strand, frame).
|
|
128
|
+
Any additional column is written as an attribute, except a column
|
|
129
|
+
literally named ``attribute``, which is treated as a pre-formatted
|
|
130
|
+
attribute string and emitted verbatim.
|
|
131
|
+
|
|
132
|
+
path : str or pathlib.Path
|
|
133
|
+
Destination file path. Any existing file is overwritten. If the path
|
|
134
|
+
ends in ``.gz`` the output is gzip-compressed (mirroring read_gtf,
|
|
135
|
+
which transparently reads gzip-compressed GTFs).
|
|
136
|
+
|
|
137
|
+
header_lines : iterable of str, optional
|
|
138
|
+
Lines to write at the top of the file before any feature rows, e.g.
|
|
139
|
+
``["##description: example", "##provider: GENCODE"]``. Each is written
|
|
140
|
+
verbatim on its own line, so include a leading ``#`` if you want it to
|
|
141
|
+
be parsed back as a comment.
|
|
142
|
+
|
|
143
|
+
Notes
|
|
144
|
+
-----
|
|
145
|
+
GTF has no escaping mechanism for the structural characters ``"`` and
|
|
146
|
+
``;`` inside attribute values (nor for the tab and newline that delimit
|
|
147
|
+
columns and rows), and read_gtf strips quotes and splits on ``;`` when
|
|
148
|
+
parsing. Values returned by read_gtf therefore never contain those
|
|
149
|
+
characters, so any DataFrame obtained from read_gtf round-trips exactly. A
|
|
150
|
+
DataFrame built by hand whose attribute values contain ``"``, ``;``, a tab,
|
|
151
|
+
or a newline cannot be represented losslessly and will not round-trip.
|
|
152
|
+
"""
|
|
153
|
+
# Accept a pandas DataFrame too, since read_gtf(result_type="pandas")
|
|
154
|
+
# returns one; convert to polars so the formatting below is uniform.
|
|
155
|
+
if not isinstance(df, polars.DataFrame):
|
|
156
|
+
df = polars.from_pandas(df)
|
|
157
|
+
|
|
158
|
+
lines = _line_series(df)
|
|
159
|
+
|
|
160
|
+
open_file = gzip.open if str(path).lower().endswith(".gz") else open
|
|
161
|
+
with open_file(path, "wt", encoding="utf-8", newline="\n") as output_file:
|
|
162
|
+
if header_lines is not None:
|
|
163
|
+
for header_line in header_lines:
|
|
164
|
+
output_file.write("%s\n" % header_line)
|
|
165
|
+
for line in lines:
|
|
166
|
+
output_file.write("%s\n" % line)
|
|
167
|
+
|
|
168
|
+
logger.info("Wrote %d GTF rows to %s", lines.len(), path)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gtfparse
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.8.0
|
|
4
4
|
Summary: Parsing library for extracting data frames of genomic features from GTF files
|
|
5
5
|
Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
|
|
6
6
|
Project-URL: Homepage, https://github.com/openvax/gtfparse
|
|
@@ -7,6 +7,7 @@ gtfparse/attribute_parsing.py
|
|
|
7
7
|
gtfparse/create_missing_features.py
|
|
8
8
|
gtfparse/parsing_error.py
|
|
9
9
|
gtfparse/read_gtf.py
|
|
10
|
+
gtfparse/write_gtf.py
|
|
10
11
|
gtfparse.egg-info/PKG-INFO
|
|
11
12
|
gtfparse.egg-info/SOURCES.txt
|
|
12
13
|
gtfparse.egg-info/dependency_links.txt
|
|
@@ -21,4 +22,5 @@ tests/test_gencode_gtf.py
|
|
|
21
22
|
tests/test_multiple_values_for_tag_attribute.py
|
|
22
23
|
tests/test_parse_gtf_lines.py
|
|
23
24
|
tests/test_read_stringtie_gtf.py
|
|
24
|
-
tests/test_refseq_gtf.py
|
|
25
|
+
tests/test_refseq_gtf.py
|
|
26
|
+
tests/test_write_gtf.py
|
|
@@ -0,0 +1,222 @@
|
|
|
1
|
+
import gzip
|
|
2
|
+
|
|
3
|
+
import polars
|
|
4
|
+
import pytest
|
|
5
|
+
from polars.testing import assert_frame_equal
|
|
6
|
+
|
|
7
|
+
from gtfparse import read_gtf, write_gtf
|
|
8
|
+
|
|
9
|
+
from .data import data_path
|
|
10
|
+
|
|
11
|
+
# A spread of real GTF flavors: RefSeq (minimal attrs), Ensembl (many attrs +
|
|
12
|
+
# null scores + *_version columns), GENCODE (mixed feature types so different
|
|
13
|
+
# rows carry different attribute sets), and StringTie.
|
|
14
|
+
FIXTURES = [
|
|
15
|
+
"refseq.ucsc.small.gtf",
|
|
16
|
+
"ensembl_grch37.head.gtf",
|
|
17
|
+
"gencode.head.gtf",
|
|
18
|
+
"gencode.real.head.gtf",
|
|
19
|
+
"B16.stringtie.head.gtf",
|
|
20
|
+
]
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _minimal_df(**extra_columns):
|
|
24
|
+
"""Build a one-row DataFrame with the fixed GTF columns plus extras.
|
|
25
|
+
|
|
26
|
+
Each extra column is passed as a single-element list, e.g.
|
|
27
|
+
``_minimal_df(gene_id=["G1"])``.
|
|
28
|
+
"""
|
|
29
|
+
row = {
|
|
30
|
+
"seqname": ["chr1"],
|
|
31
|
+
"source": ["test"],
|
|
32
|
+
"feature": ["gene"],
|
|
33
|
+
"start": [1],
|
|
34
|
+
"end": [100],
|
|
35
|
+
"score": [None],
|
|
36
|
+
"strand": ["+"],
|
|
37
|
+
"frame": [None],
|
|
38
|
+
}
|
|
39
|
+
row.update(extra_columns)
|
|
40
|
+
return polars.DataFrame(row)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
# ---------------------------------------------------------------------------
|
|
44
|
+
# write(read(...)): start from a GTF file, read it, write it back.
|
|
45
|
+
# ---------------------------------------------------------------------------
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
@pytest.mark.parametrize("fixture", FIXTURES)
|
|
49
|
+
@pytest.mark.parametrize("expand", [True, False])
|
|
50
|
+
def test_write_read_recovers_parsed_frame(fixture, expand, tmp_path):
|
|
51
|
+
"""read -> write -> read reproduces the parsed DataFrame exactly (column
|
|
52
|
+
order included), for both expanded and unexpanded attribute modes."""
|
|
53
|
+
original = read_gtf(data_path(fixture), expand_attribute_column=expand)
|
|
54
|
+
out_path = tmp_path / "out.gtf"
|
|
55
|
+
write_gtf(original, out_path)
|
|
56
|
+
recovered = read_gtf(str(out_path), expand_attribute_column=expand)
|
|
57
|
+
assert_frame_equal(original, recovered, categorical_as_str=True)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
@pytest.mark.parametrize("fixture", FIXTURES)
|
|
61
|
+
def test_written_output_is_idempotent(fixture, tmp_path):
|
|
62
|
+
"""write(read(...)) is a fixed point: writing what we read, reading it, and
|
|
63
|
+
writing it again yields byte-identical output."""
|
|
64
|
+
df = read_gtf(data_path(fixture))
|
|
65
|
+
first = tmp_path / "first.gtf"
|
|
66
|
+
second = tmp_path / "second.gtf"
|
|
67
|
+
write_gtf(df, first)
|
|
68
|
+
write_gtf(read_gtf(str(first)), second)
|
|
69
|
+
assert first.read_bytes() == second.read_bytes()
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
# ---------------------------------------------------------------------------
|
|
73
|
+
# read(write(...)): start from an in-memory DataFrame, write it, read it back.
|
|
74
|
+
# ---------------------------------------------------------------------------
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def test_read_write_recovers_dataframe(tmp_path):
|
|
78
|
+
"""A DataFrame written out and read back preserves its attribute values."""
|
|
79
|
+
df = _minimal_df(gene_id=["ENSG1"], gene_name=["DDX11L1"], gene_version=["5"])
|
|
80
|
+
out_path = tmp_path / "df.gtf"
|
|
81
|
+
write_gtf(df, out_path)
|
|
82
|
+
recovered = read_gtf(str(out_path))
|
|
83
|
+
assert recovered["gene_id"].to_list() == ["ENSG1"]
|
|
84
|
+
assert recovered["gene_name"].to_list() == ["DDX11L1"]
|
|
85
|
+
# writing the recovered frame is itself a fixed point
|
|
86
|
+
again = tmp_path / "df2.gtf"
|
|
87
|
+
write_gtf(recovered, again)
|
|
88
|
+
assert_frame_equal(recovered, read_gtf(str(again)), categorical_as_str=True)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def test_round_trip_from_pandas(tmp_path):
|
|
92
|
+
"""write_gtf accepts a pandas DataFrame (read_gtf(result_type='pandas'))."""
|
|
93
|
+
polars_df = read_gtf(data_path("ensembl_grch37.head.gtf"))
|
|
94
|
+
pandas_df = read_gtf(data_path("ensembl_grch37.head.gtf"), result_type="pandas")
|
|
95
|
+
out_path = tmp_path / "from_pandas.gtf"
|
|
96
|
+
write_gtf(pandas_df, out_path)
|
|
97
|
+
assert_frame_equal(polars_df, read_gtf(str(out_path)), categorical_as_str=True)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def test_gzip_output_round_trips(tmp_path):
|
|
101
|
+
"""A '.gz' path is gzip-compressed on disk and read back transparently."""
|
|
102
|
+
df = read_gtf(data_path("ensembl_grch37.head.gtf"))
|
|
103
|
+
out_path = tmp_path / "out.gtf.gz"
|
|
104
|
+
write_gtf(df, out_path)
|
|
105
|
+
# actually gzip-compressed on disk
|
|
106
|
+
assert out_path.read_bytes()[:2] == b"\x1f\x8b"
|
|
107
|
+
with gzip.open(out_path, "rt") as handle:
|
|
108
|
+
assert "\t" in handle.readline()
|
|
109
|
+
assert_frame_equal(df, read_gtf(str(out_path)), categorical_as_str=True)
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def test_gzip_detection_is_case_insensitive(tmp_path):
|
|
113
|
+
"""An uppercase '.GZ' suffix is still gzip-compressed."""
|
|
114
|
+
df = read_gtf(data_path("refseq.ucsc.small.gtf"))
|
|
115
|
+
out_path = tmp_path / "out.GZ"
|
|
116
|
+
write_gtf(df, out_path)
|
|
117
|
+
assert out_path.read_bytes()[:2] == b"\x1f\x8b"
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def test_empty_dataframe_writes_no_rows(tmp_path):
|
|
121
|
+
"""A zero-row DataFrame produces a file with only its header lines."""
|
|
122
|
+
empty = read_gtf(data_path("refseq.ucsc.small.gtf")).clear()
|
|
123
|
+
out_path = tmp_path / "empty.gtf"
|
|
124
|
+
write_gtf(empty, out_path, header_lines=["##empty"])
|
|
125
|
+
assert out_path.read_text() == "##empty\n"
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def test_fixed_columns_only(tmp_path):
|
|
129
|
+
"""A DataFrame with only the fixed columns writes a valid 9-field line
|
|
130
|
+
(empty attribute field) and reads back."""
|
|
131
|
+
fixed_only = polars.DataFrame(
|
|
132
|
+
{
|
|
133
|
+
"seqname": ["chr1"],
|
|
134
|
+
"source": ["test"],
|
|
135
|
+
"feature": ["gene"],
|
|
136
|
+
"start": [1],
|
|
137
|
+
"end": [100],
|
|
138
|
+
"score": [None],
|
|
139
|
+
"strand": ["+"],
|
|
140
|
+
"frame": [None],
|
|
141
|
+
}
|
|
142
|
+
)
|
|
143
|
+
out_path = tmp_path / "fixed.gtf"
|
|
144
|
+
write_gtf(fixed_only, out_path)
|
|
145
|
+
# nine tab-separated fields, the last (attribute) empty
|
|
146
|
+
assert out_path.read_text().strip("\n").split("\t") == [
|
|
147
|
+
"chr1",
|
|
148
|
+
"test",
|
|
149
|
+
"gene",
|
|
150
|
+
"1",
|
|
151
|
+
"100",
|
|
152
|
+
".",
|
|
153
|
+
"+",
|
|
154
|
+
".",
|
|
155
|
+
"",
|
|
156
|
+
]
|
|
157
|
+
# read_gtf accepts it (no attribute columns to expand)
|
|
158
|
+
recovered = read_gtf(str(out_path), expand_attribute_column=False)
|
|
159
|
+
assert recovered["seqname"].to_list() == ["chr1"]
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
# ---------------------------------------------------------------------------
|
|
163
|
+
# Attribute / missing-value semantics.
|
|
164
|
+
# ---------------------------------------------------------------------------
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def test_nonempty_value_written_empty_and_null_omitted(tmp_path):
|
|
168
|
+
"""A non-empty value such as '0' is written; empty string and null are
|
|
169
|
+
treated as absent and omitted (matching read_gtf's missing-value model)."""
|
|
170
|
+
df = _minimal_df(
|
|
171
|
+
gene_id=["G1"],
|
|
172
|
+
zero_attr=["0"],
|
|
173
|
+
empty_attr=[""],
|
|
174
|
+
missing_attr=[None],
|
|
175
|
+
)
|
|
176
|
+
out_path = tmp_path / "attrs.gtf"
|
|
177
|
+
write_gtf(df, out_path)
|
|
178
|
+
line = out_path.read_text().strip()
|
|
179
|
+
assert 'gene_id "G1";' in line
|
|
180
|
+
assert 'zero_attr "0";' in line # non-empty falsy value survives
|
|
181
|
+
assert "empty_attr" not in line # empty string omitted as absent
|
|
182
|
+
assert "missing_attr" not in line # null omitted as absent
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def test_missing_value_fixed_columns_use_dot(tmp_path):
|
|
186
|
+
"""None in the fixed columns is serialized as '.'."""
|
|
187
|
+
df = _minimal_df(gene_id=["G1"]) # score and frame are None
|
|
188
|
+
out_path = tmp_path / "dots.gtf"
|
|
189
|
+
write_gtf(df, out_path)
|
|
190
|
+
fields = out_path.read_text().strip().split("\t")
|
|
191
|
+
assert fields[5] == "." # score
|
|
192
|
+
assert fields[7] == "." # frame
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def test_structural_characters_are_not_round_trippable(tmp_path):
|
|
196
|
+
"""GTF has no escaping for '"' or ';'; read_gtf strips quotes and splits on
|
|
197
|
+
';'. Pin that such values cannot round-trip so a future change is noticed."""
|
|
198
|
+
df = _minimal_df(gene_id=["A;B"], note=['say "hi"'])
|
|
199
|
+
out_path = tmp_path / "special.gtf"
|
|
200
|
+
write_gtf(df, out_path)
|
|
201
|
+
recovered = read_gtf(str(out_path))
|
|
202
|
+
# the semicolon split the value apart
|
|
203
|
+
assert recovered["gene_id"].to_list() != ["A;B"]
|
|
204
|
+
# the double quotes were stripped from the value
|
|
205
|
+
assert '"' not in recovered["note"][0]
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def test_header_lines_are_written(tmp_path):
|
|
209
|
+
df = read_gtf(data_path("refseq.ucsc.small.gtf"))
|
|
210
|
+
out_path = tmp_path / "with_header.gtf"
|
|
211
|
+
write_gtf(df, out_path, header_lines=["##description: test", "##provider: gtfparse"])
|
|
212
|
+
lines = out_path.read_text().splitlines()
|
|
213
|
+
assert lines[0] == "##description: test"
|
|
214
|
+
assert lines[1] == "##provider: gtfparse"
|
|
215
|
+
# comment lines are ignored by read_gtf, so the data still round-trips
|
|
216
|
+
assert_frame_equal(df, read_gtf(str(out_path)), categorical_as_str=True)
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def test_missing_required_column_raises(tmp_path):
|
|
220
|
+
df = polars.DataFrame({"seqname": ["chr1"], "gene_id": ["G1"]})
|
|
221
|
+
with pytest.raises(ValueError, match="missing required GTF column"):
|
|
222
|
+
write_gtf(df, tmp_path / "bad.gtf")
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|