gtfparse 2.7.2__tar.gz → 2.8.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. {gtfparse-2.7.2 → gtfparse-2.8.0}/PKG-INFO +1 -1
  2. {gtfparse-2.7.2 → gtfparse-2.8.0}/gtfparse/__init__.py +3 -1
  3. gtfparse-2.8.0/gtfparse/write_gtf.py +168 -0
  4. {gtfparse-2.7.2 → gtfparse-2.8.0}/gtfparse.egg-info/PKG-INFO +1 -1
  5. {gtfparse-2.7.2 → gtfparse-2.8.0}/gtfparse.egg-info/SOURCES.txt +3 -1
  6. gtfparse-2.8.0/tests/test_write_gtf.py +222 -0
  7. {gtfparse-2.7.2 → gtfparse-2.8.0}/LICENSE +0 -0
  8. {gtfparse-2.7.2 → gtfparse-2.8.0}/README.md +0 -0
  9. {gtfparse-2.7.2 → gtfparse-2.8.0}/gtfparse/attribute_parsing.py +0 -0
  10. {gtfparse-2.7.2 → gtfparse-2.8.0}/gtfparse/create_missing_features.py +0 -0
  11. {gtfparse-2.7.2 → gtfparse-2.8.0}/gtfparse/parsing_error.py +0 -0
  12. {gtfparse-2.7.2 → gtfparse-2.8.0}/gtfparse/read_gtf.py +0 -0
  13. {gtfparse-2.7.2 → gtfparse-2.8.0}/gtfparse.egg-info/dependency_links.txt +0 -0
  14. {gtfparse-2.7.2 → gtfparse-2.8.0}/gtfparse.egg-info/requires.txt +0 -0
  15. {gtfparse-2.7.2 → gtfparse-2.8.0}/gtfparse.egg-info/top_level.txt +0 -0
  16. {gtfparse-2.7.2 → gtfparse-2.8.0}/pyproject.toml +0 -0
  17. {gtfparse-2.7.2 → gtfparse-2.8.0}/requirements.txt +0 -0
  18. {gtfparse-2.7.2 → gtfparse-2.8.0}/setup.cfg +0 -0
  19. {gtfparse-2.7.2 → gtfparse-2.8.0}/tests/test_create_missing_features.py +0 -0
  20. {gtfparse-2.7.2 → gtfparse-2.8.0}/tests/test_ensembl_gtf.py +0 -0
  21. {gtfparse-2.7.2 → gtfparse-2.8.0}/tests/test_expand_attribute_column_false.py +0 -0
  22. {gtfparse-2.7.2 → gtfparse-2.8.0}/tests/test_expand_attributes.py +0 -0
  23. {gtfparse-2.7.2 → gtfparse-2.8.0}/tests/test_gencode_gtf.py +0 -0
  24. {gtfparse-2.7.2 → gtfparse-2.8.0}/tests/test_multiple_values_for_tag_attribute.py +0 -0
  25. {gtfparse-2.7.2 → gtfparse-2.8.0}/tests/test_parse_gtf_lines.py +0 -0
  26. {gtfparse-2.7.2 → gtfparse-2.8.0}/tests/test_read_stringtie_gtf.py +0 -0
  27. {gtfparse-2.7.2 → gtfparse-2.8.0}/tests/test_refseq_gtf.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gtfparse
3
- Version: 2.7.2
3
+ Version: 2.8.0
4
4
  Summary: Parsing library for extracting data frames of genomic features from GTF files
5
5
  Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
6
6
  Project-URL: Homepage, https://github.com/openvax/gtfparse
@@ -22,8 +22,9 @@ from .read_gtf import (
22
22
  parse_gtf_pandas,
23
23
  read_gtf,
24
24
  )
25
+ from .write_gtf import write_gtf
25
26
 
26
- __version__ = "2.7.2"
27
+ __version__ = "2.8.0"
27
28
 
28
29
  __all__ = [
29
30
  "GENCODE_BIOTYPE_ALIASES",
@@ -37,4 +38,5 @@ __all__ = [
37
38
  "parse_gtf_and_expand_attributes",
38
39
  "parse_gtf_pandas",
39
40
  "read_gtf",
41
+ "write_gtf",
40
42
  ]
@@ -0,0 +1,168 @@
1
+ # Licensed under the Apache License, Version 2.0 (the "License");
2
+ # you may not use this file except in compliance with the License.
3
+ # You may obtain a copy of the License at
4
+ #
5
+ # http://www.apache.org/licenses/LICENSE-2.0
6
+ #
7
+ # Unless required by applicable law or agreed to in writing, software
8
+ # distributed under the License is distributed on an "AS IS" BASIS,
9
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
10
+ # See the License for the specific language governing permissions and
11
+ # limitations under the License.
12
+
13
+ import gzip
14
+ import logging
15
+ from collections.abc import Iterable
16
+ from pathlib import Path
17
+ from typing import TYPE_CHECKING, Optional, Union
18
+
19
+ import polars
20
+
21
+ if TYPE_CHECKING:
22
+ import pandas
23
+
24
+ logger = logging.getLogger(__name__)
25
+
26
+ # The eight tab-separated columns that precede the attribute field of a GTF
27
+ # line, in the order they must be written. Any other column in a DataFrame is
28
+ # treated as an expanded attribute (see read_gtf's expand_attribute_column).
29
+ GTF_FIXED_COLUMNS = [
30
+ "seqname",
31
+ "source",
32
+ "feature",
33
+ "start",
34
+ "end",
35
+ "score",
36
+ "strand",
37
+ "frame",
38
+ ]
39
+
40
+ # GTF uses a single dot to denote a missing value in the fixed columns.
41
+ MISSING_VALUE = "."
42
+
43
+ # Name of the raw, unexpanded attribute column produced by
44
+ # read_gtf(expand_attribute_column=False).
45
+ RAW_ATTRIBUTE_COLUMN = "attribute"
46
+
47
+
48
+ def _attribute_expr(attribute_columns: list[str]) -> polars.Expr:
49
+ """
50
+ Build a polars expression that renders the GTF attribute field for each row
51
+ from a set of expanded attribute columns.
52
+
53
+ Each column ``key`` becomes ``key "value";`` and the per-row pairs are
54
+ joined with a single space, matching the format emitted by Ensembl/GENCODE
55
+ and parsed by read_gtf.
56
+
57
+ A pair is omitted for any row where the value is "absent". read_gtf marks an
58
+ absent key with null (numeric columns, e.g. the ``*_version`` fields) or the
59
+ empty string (string columns) and cannot distinguish absent from
60
+ present-but-empty, so we treat both null and "" as absent. Values that are
61
+ merely falsy but non-empty -- notably the string "0" -- are written and
62
+ survive a round trip (this is the regression that the naive ``if value:``
63
+ check in earlier drafts got wrong).
64
+ """
65
+ if not attribute_columns:
66
+ return polars.lit("")
67
+ pairs = []
68
+ for name in attribute_columns:
69
+ value = polars.col(name).cast(polars.String)
70
+ pairs.append(
71
+ polars.when(value.is_null() | (value == ""))
72
+ .then(None)
73
+ .otherwise(polars.format('{} "{}";', polars.lit(name), value))
74
+ )
75
+ # ignore_nulls drops absent keys; fill_null covers the all-absent row so the
76
+ # surrounding line expression never collapses to null.
77
+ return polars.concat_str(pairs, separator=" ", ignore_nulls=True).fill_null("")
78
+
79
+
80
+ def _line_series(df: polars.DataFrame) -> polars.Series:
81
+ """
82
+ Turn a DataFrame into a Series of fully-formatted GTF lines (without
83
+ trailing newlines), building the whole thing with vectorized polars
84
+ expressions rather than per-row Python.
85
+ """
86
+ columns = df.columns
87
+ missing = [name for name in GTF_FIXED_COLUMNS if name not in columns]
88
+ if missing:
89
+ raise ValueError("DataFrame is missing required GTF column(s): %s" % ", ".join(missing))
90
+
91
+ # Fixed columns are always written in canonical order; nulls become '.'.
92
+ # Cast before fill_null so numeric columns accept the string sentinel.
93
+ fixed = [
94
+ polars.col(name).cast(polars.String).fill_null(MISSING_VALUE) for name in GTF_FIXED_COLUMNS
95
+ ]
96
+
97
+ if RAW_ATTRIBUTE_COLUMN in columns:
98
+ # Unexpanded read: the attribute column is already a formatted string.
99
+ attribute = polars.col(RAW_ATTRIBUTE_COLUMN).cast(polars.String).fill_null("")
100
+ else:
101
+ attribute_columns = [name for name in columns if name not in GTF_FIXED_COLUMNS]
102
+ attribute = _attribute_expr(attribute_columns)
103
+
104
+ # Always keep the 9th (attribute) field, even when empty, so every line has
105
+ # the nine tab-separated columns read_gtf expects.
106
+ line = polars.concat_str([*fixed, attribute], separator="\t")
107
+ return df.select(line.alias("_gtf_line")).get_column("_gtf_line")
108
+
109
+
110
+ def write_gtf(
111
+ df: Union[polars.DataFrame, "pandas.DataFrame"],
112
+ path: Union[str, Path],
113
+ header_lines: Optional[Iterable[str]] = None,
114
+ ) -> None:
115
+ """
116
+ Write a DataFrame of genomic features back out to a GTF file.
117
+
118
+ This is the inverse of :func:`read_gtf`. A DataFrame produced by
119
+ ``read_gtf`` (whether the attribute column was expanded into one column per
120
+ key or left as a raw ``attribute`` string) can be written back out and
121
+ re-read to recover an equivalent DataFrame.
122
+
123
+ Parameters
124
+ ----------
125
+ df : polars.DataFrame or pandas.DataFrame
126
+ Feature rows to write. Must contain the fixed GTF columns
127
+ (seqname, source, feature, start, end, score, strand, frame).
128
+ Any additional column is written as an attribute, except a column
129
+ literally named ``attribute``, which is treated as a pre-formatted
130
+ attribute string and emitted verbatim.
131
+
132
+ path : str or pathlib.Path
133
+ Destination file path. Any existing file is overwritten. If the path
134
+ ends in ``.gz`` the output is gzip-compressed (mirroring read_gtf,
135
+ which transparently reads gzip-compressed GTFs).
136
+
137
+ header_lines : iterable of str, optional
138
+ Lines to write at the top of the file before any feature rows, e.g.
139
+ ``["##description: example", "##provider: GENCODE"]``. Each is written
140
+ verbatim on its own line, so include a leading ``#`` if you want it to
141
+ be parsed back as a comment.
142
+
143
+ Notes
144
+ -----
145
+ GTF has no escaping mechanism for the structural characters ``"`` and
146
+ ``;`` inside attribute values (nor for the tab and newline that delimit
147
+ columns and rows), and read_gtf strips quotes and splits on ``;`` when
148
+ parsing. Values returned by read_gtf therefore never contain those
149
+ characters, so any DataFrame obtained from read_gtf round-trips exactly. A
150
+ DataFrame built by hand whose attribute values contain ``"``, ``;``, a tab,
151
+ or a newline cannot be represented losslessly and will not round-trip.
152
+ """
153
+ # Accept a pandas DataFrame too, since read_gtf(result_type="pandas")
154
+ # returns one; convert to polars so the formatting below is uniform.
155
+ if not isinstance(df, polars.DataFrame):
156
+ df = polars.from_pandas(df)
157
+
158
+ lines = _line_series(df)
159
+
160
+ open_file = gzip.open if str(path).lower().endswith(".gz") else open
161
+ with open_file(path, "wt", encoding="utf-8", newline="\n") as output_file:
162
+ if header_lines is not None:
163
+ for header_line in header_lines:
164
+ output_file.write("%s\n" % header_line)
165
+ for line in lines:
166
+ output_file.write("%s\n" % line)
167
+
168
+ logger.info("Wrote %d GTF rows to %s", lines.len(), path)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gtfparse
3
- Version: 2.7.2
3
+ Version: 2.8.0
4
4
  Summary: Parsing library for extracting data frames of genomic features from GTF files
5
5
  Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
6
6
  Project-URL: Homepage, https://github.com/openvax/gtfparse
@@ -7,6 +7,7 @@ gtfparse/attribute_parsing.py
7
7
  gtfparse/create_missing_features.py
8
8
  gtfparse/parsing_error.py
9
9
  gtfparse/read_gtf.py
10
+ gtfparse/write_gtf.py
10
11
  gtfparse.egg-info/PKG-INFO
11
12
  gtfparse.egg-info/SOURCES.txt
12
13
  gtfparse.egg-info/dependency_links.txt
@@ -21,4 +22,5 @@ tests/test_gencode_gtf.py
21
22
  tests/test_multiple_values_for_tag_attribute.py
22
23
  tests/test_parse_gtf_lines.py
23
24
  tests/test_read_stringtie_gtf.py
24
- tests/test_refseq_gtf.py
25
+ tests/test_refseq_gtf.py
26
+ tests/test_write_gtf.py
@@ -0,0 +1,222 @@
1
+ import gzip
2
+
3
+ import polars
4
+ import pytest
5
+ from polars.testing import assert_frame_equal
6
+
7
+ from gtfparse import read_gtf, write_gtf
8
+
9
+ from .data import data_path
10
+
11
+ # A spread of real GTF flavors: RefSeq (minimal attrs), Ensembl (many attrs +
12
+ # null scores + *_version columns), GENCODE (mixed feature types so different
13
+ # rows carry different attribute sets), and StringTie.
14
+ FIXTURES = [
15
+ "refseq.ucsc.small.gtf",
16
+ "ensembl_grch37.head.gtf",
17
+ "gencode.head.gtf",
18
+ "gencode.real.head.gtf",
19
+ "B16.stringtie.head.gtf",
20
+ ]
21
+
22
+
23
+ def _minimal_df(**extra_columns):
24
+ """Build a one-row DataFrame with the fixed GTF columns plus extras.
25
+
26
+ Each extra column is passed as a single-element list, e.g.
27
+ ``_minimal_df(gene_id=["G1"])``.
28
+ """
29
+ row = {
30
+ "seqname": ["chr1"],
31
+ "source": ["test"],
32
+ "feature": ["gene"],
33
+ "start": [1],
34
+ "end": [100],
35
+ "score": [None],
36
+ "strand": ["+"],
37
+ "frame": [None],
38
+ }
39
+ row.update(extra_columns)
40
+ return polars.DataFrame(row)
41
+
42
+
43
+ # ---------------------------------------------------------------------------
44
+ # write(read(...)): start from a GTF file, read it, write it back.
45
+ # ---------------------------------------------------------------------------
46
+
47
+
48
+ @pytest.mark.parametrize("fixture", FIXTURES)
49
+ @pytest.mark.parametrize("expand", [True, False])
50
+ def test_write_read_recovers_parsed_frame(fixture, expand, tmp_path):
51
+ """read -> write -> read reproduces the parsed DataFrame exactly (column
52
+ order included), for both expanded and unexpanded attribute modes."""
53
+ original = read_gtf(data_path(fixture), expand_attribute_column=expand)
54
+ out_path = tmp_path / "out.gtf"
55
+ write_gtf(original, out_path)
56
+ recovered = read_gtf(str(out_path), expand_attribute_column=expand)
57
+ assert_frame_equal(original, recovered, categorical_as_str=True)
58
+
59
+
60
+ @pytest.mark.parametrize("fixture", FIXTURES)
61
+ def test_written_output_is_idempotent(fixture, tmp_path):
62
+ """write(read(...)) is a fixed point: writing what we read, reading it, and
63
+ writing it again yields byte-identical output."""
64
+ df = read_gtf(data_path(fixture))
65
+ first = tmp_path / "first.gtf"
66
+ second = tmp_path / "second.gtf"
67
+ write_gtf(df, first)
68
+ write_gtf(read_gtf(str(first)), second)
69
+ assert first.read_bytes() == second.read_bytes()
70
+
71
+
72
+ # ---------------------------------------------------------------------------
73
+ # read(write(...)): start from an in-memory DataFrame, write it, read it back.
74
+ # ---------------------------------------------------------------------------
75
+
76
+
77
+ def test_read_write_recovers_dataframe(tmp_path):
78
+ """A DataFrame written out and read back preserves its attribute values."""
79
+ df = _minimal_df(gene_id=["ENSG1"], gene_name=["DDX11L1"], gene_version=["5"])
80
+ out_path = tmp_path / "df.gtf"
81
+ write_gtf(df, out_path)
82
+ recovered = read_gtf(str(out_path))
83
+ assert recovered["gene_id"].to_list() == ["ENSG1"]
84
+ assert recovered["gene_name"].to_list() == ["DDX11L1"]
85
+ # writing the recovered frame is itself a fixed point
86
+ again = tmp_path / "df2.gtf"
87
+ write_gtf(recovered, again)
88
+ assert_frame_equal(recovered, read_gtf(str(again)), categorical_as_str=True)
89
+
90
+
91
+ def test_round_trip_from_pandas(tmp_path):
92
+ """write_gtf accepts a pandas DataFrame (read_gtf(result_type='pandas'))."""
93
+ polars_df = read_gtf(data_path("ensembl_grch37.head.gtf"))
94
+ pandas_df = read_gtf(data_path("ensembl_grch37.head.gtf"), result_type="pandas")
95
+ out_path = tmp_path / "from_pandas.gtf"
96
+ write_gtf(pandas_df, out_path)
97
+ assert_frame_equal(polars_df, read_gtf(str(out_path)), categorical_as_str=True)
98
+
99
+
100
+ def test_gzip_output_round_trips(tmp_path):
101
+ """A '.gz' path is gzip-compressed on disk and read back transparently."""
102
+ df = read_gtf(data_path("ensembl_grch37.head.gtf"))
103
+ out_path = tmp_path / "out.gtf.gz"
104
+ write_gtf(df, out_path)
105
+ # actually gzip-compressed on disk
106
+ assert out_path.read_bytes()[:2] == b"\x1f\x8b"
107
+ with gzip.open(out_path, "rt") as handle:
108
+ assert "\t" in handle.readline()
109
+ assert_frame_equal(df, read_gtf(str(out_path)), categorical_as_str=True)
110
+
111
+
112
+ def test_gzip_detection_is_case_insensitive(tmp_path):
113
+ """An uppercase '.GZ' suffix is still gzip-compressed."""
114
+ df = read_gtf(data_path("refseq.ucsc.small.gtf"))
115
+ out_path = tmp_path / "out.GZ"
116
+ write_gtf(df, out_path)
117
+ assert out_path.read_bytes()[:2] == b"\x1f\x8b"
118
+
119
+
120
+ def test_empty_dataframe_writes_no_rows(tmp_path):
121
+ """A zero-row DataFrame produces a file with only its header lines."""
122
+ empty = read_gtf(data_path("refseq.ucsc.small.gtf")).clear()
123
+ out_path = tmp_path / "empty.gtf"
124
+ write_gtf(empty, out_path, header_lines=["##empty"])
125
+ assert out_path.read_text() == "##empty\n"
126
+
127
+
128
+ def test_fixed_columns_only(tmp_path):
129
+ """A DataFrame with only the fixed columns writes a valid 9-field line
130
+ (empty attribute field) and reads back."""
131
+ fixed_only = polars.DataFrame(
132
+ {
133
+ "seqname": ["chr1"],
134
+ "source": ["test"],
135
+ "feature": ["gene"],
136
+ "start": [1],
137
+ "end": [100],
138
+ "score": [None],
139
+ "strand": ["+"],
140
+ "frame": [None],
141
+ }
142
+ )
143
+ out_path = tmp_path / "fixed.gtf"
144
+ write_gtf(fixed_only, out_path)
145
+ # nine tab-separated fields, the last (attribute) empty
146
+ assert out_path.read_text().strip("\n").split("\t") == [
147
+ "chr1",
148
+ "test",
149
+ "gene",
150
+ "1",
151
+ "100",
152
+ ".",
153
+ "+",
154
+ ".",
155
+ "",
156
+ ]
157
+ # read_gtf accepts it (no attribute columns to expand)
158
+ recovered = read_gtf(str(out_path), expand_attribute_column=False)
159
+ assert recovered["seqname"].to_list() == ["chr1"]
160
+
161
+
162
+ # ---------------------------------------------------------------------------
163
+ # Attribute / missing-value semantics.
164
+ # ---------------------------------------------------------------------------
165
+
166
+
167
+ def test_nonempty_value_written_empty_and_null_omitted(tmp_path):
168
+ """A non-empty value such as '0' is written; empty string and null are
169
+ treated as absent and omitted (matching read_gtf's missing-value model)."""
170
+ df = _minimal_df(
171
+ gene_id=["G1"],
172
+ zero_attr=["0"],
173
+ empty_attr=[""],
174
+ missing_attr=[None],
175
+ )
176
+ out_path = tmp_path / "attrs.gtf"
177
+ write_gtf(df, out_path)
178
+ line = out_path.read_text().strip()
179
+ assert 'gene_id "G1";' in line
180
+ assert 'zero_attr "0";' in line # non-empty falsy value survives
181
+ assert "empty_attr" not in line # empty string omitted as absent
182
+ assert "missing_attr" not in line # null omitted as absent
183
+
184
+
185
+ def test_missing_value_fixed_columns_use_dot(tmp_path):
186
+ """None in the fixed columns is serialized as '.'."""
187
+ df = _minimal_df(gene_id=["G1"]) # score and frame are None
188
+ out_path = tmp_path / "dots.gtf"
189
+ write_gtf(df, out_path)
190
+ fields = out_path.read_text().strip().split("\t")
191
+ assert fields[5] == "." # score
192
+ assert fields[7] == "." # frame
193
+
194
+
195
+ def test_structural_characters_are_not_round_trippable(tmp_path):
196
+ """GTF has no escaping for '"' or ';'; read_gtf strips quotes and splits on
197
+ ';'. Pin that such values cannot round-trip so a future change is noticed."""
198
+ df = _minimal_df(gene_id=["A;B"], note=['say "hi"'])
199
+ out_path = tmp_path / "special.gtf"
200
+ write_gtf(df, out_path)
201
+ recovered = read_gtf(str(out_path))
202
+ # the semicolon split the value apart
203
+ assert recovered["gene_id"].to_list() != ["A;B"]
204
+ # the double quotes were stripped from the value
205
+ assert '"' not in recovered["note"][0]
206
+
207
+
208
+ def test_header_lines_are_written(tmp_path):
209
+ df = read_gtf(data_path("refseq.ucsc.small.gtf"))
210
+ out_path = tmp_path / "with_header.gtf"
211
+ write_gtf(df, out_path, header_lines=["##description: test", "##provider: gtfparse"])
212
+ lines = out_path.read_text().splitlines()
213
+ assert lines[0] == "##description: test"
214
+ assert lines[1] == "##provider: gtfparse"
215
+ # comment lines are ignored by read_gtf, so the data still round-trips
216
+ assert_frame_equal(df, read_gtf(str(out_path)), categorical_as_str=True)
217
+
218
+
219
+ def test_missing_required_column_raises(tmp_path):
220
+ df = polars.DataFrame({"seqname": ["chr1"], "gene_id": ["G1"]})
221
+ with pytest.raises(ValueError, match="missing required GTF column"):
222
+ write_gtf(df, tmp_path / "bad.gtf")
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes