gtfparse 2.6.3__tar.gz → 2.7.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (24) hide show
  1. {gtfparse-2.6.3 → gtfparse-2.7.0}/PKG-INFO +1 -1
  2. {gtfparse-2.6.3 → gtfparse-2.7.0}/gtfparse/__init__.py +5 -1
  3. {gtfparse-2.6.3 → gtfparse-2.7.0}/gtfparse/read_gtf.py +125 -1
  4. {gtfparse-2.6.3 → gtfparse-2.7.0}/gtfparse.egg-info/PKG-INFO +1 -1
  5. {gtfparse-2.6.3 → gtfparse-2.7.0}/gtfparse.egg-info/SOURCES.txt +1 -0
  6. gtfparse-2.7.0/tests/test_gencode_gtf.py +563 -0
  7. {gtfparse-2.6.3 → gtfparse-2.7.0}/LICENSE +0 -0
  8. {gtfparse-2.6.3 → gtfparse-2.7.0}/README.md +0 -0
  9. {gtfparse-2.6.3 → gtfparse-2.7.0}/gtfparse/attribute_parsing.py +0 -0
  10. {gtfparse-2.6.3 → gtfparse-2.7.0}/gtfparse/create_missing_features.py +0 -0
  11. {gtfparse-2.6.3 → gtfparse-2.7.0}/gtfparse/parsing_error.py +0 -0
  12. {gtfparse-2.6.3 → gtfparse-2.7.0}/gtfparse.egg-info/dependency_links.txt +0 -0
  13. {gtfparse-2.6.3 → gtfparse-2.7.0}/gtfparse.egg-info/requires.txt +0 -0
  14. {gtfparse-2.6.3 → gtfparse-2.7.0}/gtfparse.egg-info/top_level.txt +0 -0
  15. {gtfparse-2.6.3 → gtfparse-2.7.0}/pyproject.toml +0 -0
  16. {gtfparse-2.6.3 → gtfparse-2.7.0}/requirements.txt +0 -0
  17. {gtfparse-2.6.3 → gtfparse-2.7.0}/setup.cfg +0 -0
  18. {gtfparse-2.6.3 → gtfparse-2.7.0}/tests/test_create_missing_features.py +0 -0
  19. {gtfparse-2.6.3 → gtfparse-2.7.0}/tests/test_ensembl_gtf.py +0 -0
  20. {gtfparse-2.6.3 → gtfparse-2.7.0}/tests/test_expand_attributes.py +0 -0
  21. {gtfparse-2.6.3 → gtfparse-2.7.0}/tests/test_multiple_values_for_tag_attribute.py +0 -0
  22. {gtfparse-2.6.3 → gtfparse-2.7.0}/tests/test_parse_gtf_lines.py +0 -0
  23. {gtfparse-2.6.3 → gtfparse-2.7.0}/tests/test_read_stringtie_gtf.py +0 -0
  24. {gtfparse-2.6.3 → gtfparse-2.7.0}/tests/test_refseq_gtf.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gtfparse
3
- Version: 2.6.3
3
+ Version: 2.7.0
4
4
  Summary: Parsing library for extracting data frames of genomic features from GTF files
5
5
  Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
6
6
  Project-URL: Homepage, https://github.com/openvax/gtfparse
@@ -14,6 +14,8 @@ from .attribute_parsing import expand_attribute_strings
14
14
  from .create_missing_features import create_missing_features
15
15
  from .parsing_error import ParsingError
16
16
  from .read_gtf import (
17
+ GENCODE_BIOTYPE_ALIASES,
18
+ INTEGER_VERSION_COLUMNS,
17
19
  REQUIRED_COLUMNS,
18
20
  parse_gtf,
19
21
  parse_gtf_and_expand_attributes,
@@ -21,9 +23,11 @@ from .read_gtf import (
21
23
  read_gtf,
22
24
  )
23
25
 
24
- __version__ = "2.6.3"
26
+ __version__ = "2.7.0"
25
27
 
26
28
  __all__ = [
29
+ "GENCODE_BIOTYPE_ALIASES",
30
+ "INTEGER_VERSION_COLUMNS",
27
31
  "REQUIRED_COLUMNS",
28
32
  "ParsingError",
29
33
  "__version__",
@@ -13,6 +13,7 @@
13
13
  import logging
14
14
  from os.path import exists
15
15
 
16
+ import pandas as pd
16
17
  import polars
17
18
 
18
19
  from .attribute_parsing import expand_attribute_strings
@@ -22,6 +23,27 @@ logging.basicConfig(level=logging.INFO)
22
23
  logger = logging.getLogger(__name__)
23
24
 
24
25
 
26
+ # GENCODE GTFs use *_type where Ensembl GTFs use *_biotype. Pass this
27
+ # (or a superset) as `attribute_aliases` to read_gtf to normalize a
28
+ # GENCODE-format GTF onto the Ensembl column names that downstream
29
+ # tools like pyensembl expect.
30
+ GENCODE_BIOTYPE_ALIASES = {
31
+ "gene_type": "gene_biotype",
32
+ "transcript_type": "transcript_biotype",
33
+ }
34
+
35
+
36
+ # Ensembl-style attribute columns that are always integer-valued when
37
+ # present. read_gtf casts these from string to pandas nullable Int64 by
38
+ # default; pass cast_version_columns=False to keep them as strings.
39
+ INTEGER_VERSION_COLUMNS = (
40
+ "gene_version",
41
+ "transcript_version",
42
+ "protein_version",
43
+ "exon_version",
44
+ )
45
+
46
+
25
47
  """
26
48
  Columns of a GTF file:
27
49
 
@@ -183,6 +205,65 @@ def parse_gtf_and_expand_attributes(
183
205
  )
184
206
 
185
207
 
208
+ def _apply_attribute_aliases(result_df, attribute_aliases):
209
+ """
210
+ Rename alias attribute columns onto canonical names in-place.
211
+
212
+ For each (alias -> canonical) pair, in iteration order:
213
+ * if only the alias is present, rename it to the canonical name.
214
+ * if both are present, drop the alias and warn (canonical wins).
215
+ * if neither is present, do nothing.
216
+
217
+ When two aliases target the same canonical (e.g. both ``gene_type``
218
+ and a hypothetical ``gene_kind`` map to ``gene_biotype``), the first
219
+ rename in iteration order wins; subsequent aliases targeting an
220
+ already-renamed canonical are treated as collisions, dropped, and
221
+ warned about.
222
+ """
223
+ if not attribute_aliases:
224
+ return result_df
225
+ columns_present = set(result_df.columns)
226
+ rename_map = {}
227
+ drop_aliases = []
228
+ for alias, canonical in attribute_aliases.items():
229
+ if alias not in columns_present:
230
+ continue
231
+ if canonical in columns_present:
232
+ logger.warning(
233
+ "Both alias column '%s' and canonical column '%s' are present; "
234
+ "dropping alias and keeping canonical values.",
235
+ alias,
236
+ canonical,
237
+ )
238
+ drop_aliases.append(alias)
239
+ else:
240
+ rename_map[alias] = canonical
241
+ # Reflect the rename in the running column set so a later
242
+ # alias mapping to the same canonical sees the collision
243
+ # instead of silently producing a duplicate-named column.
244
+ columns_present.discard(alias)
245
+ columns_present.add(canonical)
246
+ if drop_aliases:
247
+ result_df = result_df.drop(columns=drop_aliases)
248
+ if rename_map:
249
+ result_df = result_df.rename(columns=rename_map)
250
+ return result_df
251
+
252
+
253
+ def _cast_version_columns(result_df, version_columns=INTEGER_VERSION_COLUMNS):
254
+ """
255
+ Cast known Ensembl *_version attribute columns from strings to
256
+ pandas nullable Int64 in-place. Missing/empty values become pd.NA.
257
+ """
258
+ for column_name in version_columns:
259
+ if column_name not in result_df.columns:
260
+ continue
261
+ result_df[column_name] = pd.to_numeric(
262
+ result_df[column_name].replace("", None), errors="coerce"
263
+ ).astype("Int64")
264
+ return result_df
265
+
266
+
186
267
  def read_gtf(
187
268
  filepath_or_buffer,
188
269
  expand_attribute_column=True,
@@ -192,6 +273,8 @@ def read_gtf(
192
273
  usecols=None,
193
274
  features=None,
194
275
  result_type="polars",
276
+ attribute_aliases=None,
277
+ cast_version_columns=True,
195
278
  ):
196
279
  """
197
280
  Parse a GTF into a dictionary mapping column names to sequences of values.
@@ -231,13 +314,44 @@ def read_gtf(
231
314
  result_type : One of 'polars', 'pandas', or 'dict'
232
315
  Default behavior is to return a Polars DataFrame, but will convert to
233
316
  Pandas DataFrame or dictionary if specified.
317
+
318
+ attribute_aliases : dict of str -> str, optional
319
+ Maps alias attribute names onto canonical ones. After attributes
320
+ are expanded into columns, each alias column is renamed to its
321
+ canonical name when the canonical column is absent. If both are
322
+ present the alias is dropped and a warning is logged. Pass
323
+ `GENCODE_BIOTYPE_ALIASES` to normalize a GENCODE GTF's
324
+ `gene_type`/`transcript_type` onto Ensembl's
325
+ `gene_biotype`/`transcript_biotype`.
326
+
327
+ cast_version_columns : bool
328
+ When True (default), cast the well-known integer version
329
+ attribute columns (`gene_version`, `transcript_version`,
330
+ `protein_version`, `exon_version`) from strings to pandas
331
+ nullable Int64 when present. Set to False to keep them as
332
+ strings.
234
333
  """
235
334
  if type(filepath_or_buffer) is str and not exists(filepath_or_buffer):
236
335
  raise ValueError("GTF file does not exist: %s" % filepath_or_buffer)
237
336
 
337
+ # If usecols asks for a canonical column that's only present in the
338
+ # GTF under an alias name, expand the parse-time column filter to
339
+ # also pull the alias through — otherwise it gets dropped at parse
340
+ # time before _apply_attribute_aliases can see it. The end-of-function
341
+ # usecols filter still narrows the result down to the canonical name.
342
+ parse_usecols = usecols
343
+ if usecols is not None and attribute_aliases:
344
+ usecols_set = set(usecols)
345
+ parse_usecols = set(usecols_set)
346
+ for alias, canonical in attribute_aliases.items():
347
+ if canonical in usecols_set:
348
+ parse_usecols.add(alias)
349
+
238
350
  if expand_attribute_column:
239
351
  result_df = parse_gtf_and_expand_attributes(
240
- filepath_or_buffer, restrict_attribute_columns=usecols, features=features
352
+ filepath_or_buffer,
353
+ restrict_attribute_columns=parse_usecols,
354
+ features=features,
241
355
  )
242
356
  else:
243
357
  result_df = parse_gtf(result_df, features=features)
@@ -267,6 +381,16 @@ def read_gtf(
267
381
  column_type = column_cast_types[column_name]
268
382
  result_df[column_name] = result_df[column_name].astype(column_type)
269
383
 
384
+ # Rename alias attribute columns onto their canonical names. Done before
385
+ # infer_biotype_column so an aliased gene_biotype/transcript_biotype is
386
+ # visible to the inference logic.
387
+ result_df = _apply_attribute_aliases(result_df, attribute_aliases)
388
+
389
+ # Cast Ensembl *_version columns from strings to nullable integers so
390
+ # downstream consumers (e.g. pyensembl) don't have to int(...) themselves.
391
+ if cast_version_columns:
392
+ result_df = _cast_version_columns(result_df)
393
+
270
394
  # Hackishly infer whether the values in the 'source' column of this GTF
271
395
  # are actually representing a biotype by checking for the most common
272
396
  # gene_biotype and transcript_biotype value 'protein_coding'
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gtfparse
3
- Version: 2.6.3
3
+ Version: 2.7.0
4
4
  Summary: Parsing library for extracting data frames of genomic features from GTF files
5
5
  Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
6
6
  Project-URL: Homepage, https://github.com/openvax/gtfparse
@@ -16,6 +16,7 @@ gtfparse/../requirements.txt
16
16
  tests/test_create_missing_features.py
17
17
  tests/test_ensembl_gtf.py
18
18
  tests/test_expand_attributes.py
19
+ tests/test_gencode_gtf.py
19
20
  tests/test_multiple_values_for_tag_attribute.py
20
21
  tests/test_parse_gtf_lines.py
21
22
  tests/test_read_stringtie_gtf.py
@@ -0,0 +1,563 @@
1
+ """Tests for GENCODE attribute aliases (#63) and version-column casting (#64)."""
2
+
3
+ import gzip
4
+ import logging
5
+ import os
6
+ import tempfile
7
+
8
+ from gtfparse import GENCODE_BIOTYPE_ALIASES, INTEGER_VERSION_COLUMNS, read_gtf
9
+
10
+ from .data import data_path
11
+
12
+ GENCODE_GTF_PATH = data_path("gencode.head.gtf")
13
+ GENCODE_REAL_GTF_PATH = data_path("gencode.real.head.gtf")
14
+
15
+
16
+ # -----------------------------
17
+ # attribute_aliases (#63)
18
+ # -----------------------------
19
+
20
+
21
+ def test_no_aliases_by_default_leaves_columns_alone():
22
+ df = read_gtf(GENCODE_GTF_PATH, result_type="pandas")
23
+ # gene_type / transcript_type come through as-is, gene_biotype / transcript_biotype absent
24
+ assert "gene_type" in df.columns
25
+ assert "transcript_type" in df.columns
26
+ assert "gene_biotype" not in df.columns
27
+ assert "transcript_biotype" not in df.columns
28
+
29
+
30
+ def test_attribute_aliases_renames_gencode_to_ensembl():
31
+ df = read_gtf(
32
+ GENCODE_GTF_PATH,
33
+ attribute_aliases=GENCODE_BIOTYPE_ALIASES,
34
+ result_type="pandas",
35
+ )
36
+ assert "gene_biotype" in df.columns
37
+ assert "transcript_biotype" in df.columns
38
+ assert "gene_type" not in df.columns
39
+ assert "transcript_type" not in df.columns
40
+ # values are preserved
41
+ gene_rows = df[df["feature"] == "gene"]
42
+ assert set(gene_rows["gene_biotype"]) == {
43
+ "transcribed_unprocessed_pseudogene",
44
+ "protein_coding",
45
+ }
46
+
47
+
48
+ def test_attribute_aliases_with_polars_result_type():
49
+ df = read_gtf(
50
+ GENCODE_GTF_PATH,
51
+ attribute_aliases=GENCODE_BIOTYPE_ALIASES,
52
+ result_type="polars",
53
+ )
54
+ assert "gene_biotype" in df.columns
55
+ assert "gene_type" not in df.columns
56
+
57
+
58
+ def test_attribute_aliases_when_canonical_already_present_prefers_canonical(caplog):
59
+ """If both the alias and canonical column exist, drop the alias and warn."""
60
+ # Hand-build a GTF where gene_biotype is already populated AND gene_type is also present
61
+ contents = (
62
+ "chr1\tHAVANA\tgene\t1\t100\t.\t+\t.\t"
63
+ 'gene_id "ENSGX"; gene_type "alias_value"; gene_biotype "canonical_value";\n'
64
+ )
65
+ fd, path = tempfile.mkstemp(suffix=".gtf")
66
+ try:
67
+ with os.fdopen(fd, "w") as fh:
68
+ fh.write(contents)
69
+ with caplog.at_level(logging.WARNING):
70
+ df = read_gtf(
71
+ path,
72
+ attribute_aliases={"gene_type": "gene_biotype"},
73
+ result_type="pandas",
74
+ )
75
+ assert "gene_biotype" in df.columns
76
+ assert "gene_type" not in df.columns
77
+ # canonical wins
78
+ assert list(df["gene_biotype"]) == ["canonical_value"]
79
+ # warning was logged
80
+ assert any("Both alias column" in rec.message for rec in caplog.records)
81
+ finally:
82
+ os.unlink(path)
83
+
84
+
85
+ def test_attribute_aliases_with_missing_alias_is_noop():
86
+ """Aliases that don't appear in the file are silently skipped."""
87
+ df = read_gtf(
88
+ GENCODE_GTF_PATH,
89
+ attribute_aliases={"never_in_file": "gene_biotype"},
90
+ result_type="pandas",
91
+ )
92
+ # gene_biotype did not exist and the alias did not exist -> still no gene_biotype
93
+ assert "gene_biotype" not in df.columns
94
+
95
+
96
+ def test_attribute_aliases_with_empty_dict_is_noop():
97
+ df = read_gtf(GENCODE_GTF_PATH, attribute_aliases={}, result_type="pandas")
98
+ assert "gene_type" in df.columns
99
+
100
+
101
+ # -----------------------------
102
+ # cast_version_columns (#64)
103
+ # -----------------------------
104
+
105
+
106
+ def test_version_columns_cast_to_int_by_default():
107
+ df = read_gtf(GENCODE_GTF_PATH, result_type="pandas")
108
+ # all four version columns should be present and Int64-typed (nullable)
109
+ for column_name in INTEGER_VERSION_COLUMNS:
110
+ assert column_name in df.columns, column_name
111
+ assert str(df[column_name].dtype) == "Int64", (
112
+ f"{column_name} dtype is {df[column_name].dtype}, expected Int64"
113
+ )
114
+ # spot-check actual values are integers, not strings
115
+ gene_rows = df[df["feature"] == "gene"].reset_index(drop=True)
116
+ assert gene_rows.loc[0, "gene_version"] == 5
117
+ assert gene_rows.loc[1, "gene_version"] == 6
118
+
119
+
120
+ def test_version_columns_handle_missing_values():
121
+ """Rows where a version attribute is absent should produce pd.NA, not raise."""
122
+ df = read_gtf(GENCODE_GTF_PATH, result_type="pandas")
123
+ # gene rows don't have transcript_version / protein_version / exon_version
124
+ gene_rows = df[df["feature"] == "gene"]
125
+ assert gene_rows["transcript_version"].isna().all()
126
+ assert gene_rows["protein_version"].isna().all()
127
+ assert gene_rows["exon_version"].isna().all()
128
+ # gene_version IS populated on gene rows
129
+ assert gene_rows["gene_version"].notna().all()
130
+
131
+
132
+ def test_cast_version_columns_false_keeps_strings():
133
+ import pandas as pd
134
+
135
+ df = read_gtf(GENCODE_GTF_PATH, cast_version_columns=False, result_type="pandas")
136
+ # column is still string-typed when opted out — accept either the legacy
137
+ # object dtype or pandas' newer StringDtype, depending on the installed
138
+ # pandas version.
139
+ assert pd.api.types.is_string_dtype(df["gene_version"])
140
+ gene_rows = df[df["feature"] == "gene"].reset_index(drop=True)
141
+ assert gene_rows.loc[0, "gene_version"] == "5"
142
+
143
+
144
+ def test_version_columns_present_in_polars_result():
145
+ df = read_gtf(GENCODE_GTF_PATH, result_type="polars")
146
+ for column_name in INTEGER_VERSION_COLUMNS:
147
+ assert column_name in df.columns
148
+
149
+
150
+ def test_version_columns_missing_when_attribute_absent():
151
+ """When the underlying GTF doesn't carry version attributes, the columns
152
+ simply don't exist — casting must not raise."""
153
+ # The classic Ensembl release-75 fixture has no *_version attributes
154
+ df = read_gtf(data_path("ensembl_grch37.head.gtf"), result_type="pandas")
155
+ for column_name in INTEGER_VERSION_COLUMNS:
156
+ assert column_name not in df.columns
157
+
158
+
159
+ # -----------------------------
160
+ # Interaction between #63 and #64
161
+ # -----------------------------
162
+
163
+
164
+ def test_aliases_and_version_casting_work_together():
165
+ df = read_gtf(
166
+ GENCODE_GTF_PATH,
167
+ attribute_aliases=GENCODE_BIOTYPE_ALIASES,
168
+ result_type="pandas",
169
+ )
170
+ assert "gene_biotype" in df.columns
171
+ assert "transcript_biotype" in df.columns
172
+ assert str(df["gene_version"].dtype) == "Int64"
173
+ transcript_rows = df[df["feature"] == "transcript"].reset_index(drop=True)
174
+ assert transcript_rows.loc[0, "transcript_biotype"] == "processed_transcript"
175
+ assert transcript_rows.loc[0, "transcript_version"] == 2
176
+
177
+
178
+ def test_aliases_visible_to_infer_biotype_column():
179
+ """attribute_aliases is applied before infer_biotype_column, so an aliased
180
+ gene_biotype must prevent the inference path from overwriting it."""
181
+ df = read_gtf(
182
+ GENCODE_GTF_PATH,
183
+ attribute_aliases=GENCODE_BIOTYPE_ALIASES,
184
+ infer_biotype_column=True,
185
+ result_type="pandas",
186
+ )
187
+ # Should keep the GENCODE-derived biotypes, not infer from "source"
188
+ gene_rows = df[df["feature"] == "gene"]
189
+ assert "protein_coding" in set(gene_rows["gene_biotype"])
190
+ assert "transcribed_unprocessed_pseudogene" in set(gene_rows["gene_biotype"])
191
+ # source column is HAVANA, definitely not a biotype value
192
+ assert "HAVANA" not in set(gene_rows["gene_biotype"])
193
+
194
+
195
+ # -----------------------------
196
+ # Sanity: constants are stable
197
+ # -----------------------------
198
+
199
+
200
+ def test_gencode_biotype_aliases_constant():
201
+ assert GENCODE_BIOTYPE_ALIASES == {
202
+ "gene_type": "gene_biotype",
203
+ "transcript_type": "transcript_biotype",
204
+ }
205
+
206
+
207
+ def test_integer_version_columns_constant():
208
+ assert set(INTEGER_VERSION_COLUMNS) == {
209
+ "gene_version",
210
+ "transcript_version",
211
+ "protein_version",
212
+ "exon_version",
213
+ }
214
+
215
+
216
+ # -----------------------------
217
+ # Explicit-None defaults
218
+ # -----------------------------
219
+
220
+
221
+ def test_attribute_aliases_none_is_explicit_noop():
222
+ """The kwarg default is None; passing it explicitly must match
223
+ the implicit behavior (no rename, no error)."""
224
+ df_default = read_gtf(GENCODE_GTF_PATH, result_type="pandas")
225
+ df_explicit = read_gtf(GENCODE_GTF_PATH, attribute_aliases=None, result_type="pandas")
226
+ assert list(df_default.columns) == list(df_explicit.columns)
227
+ assert "gene_type" in df_explicit.columns
228
+ assert "gene_biotype" not in df_explicit.columns
229
+
230
+
231
+ # -----------------------------
232
+ # result_type="dict" with new kwargs
233
+ # -----------------------------
234
+
235
+
236
+ def test_attribute_aliases_with_dict_result_type():
237
+ result = read_gtf(
238
+ GENCODE_GTF_PATH,
239
+ attribute_aliases=GENCODE_BIOTYPE_ALIASES,
240
+ result_type="dict",
241
+ )
242
+ assert isinstance(result, dict)
243
+ assert "gene_biotype" in result
244
+ assert "transcript_biotype" in result
245
+ assert "gene_type" not in result
246
+ assert "transcript_type" not in result
247
+
248
+
249
+ def test_version_columns_with_dict_result_type():
250
+ result = read_gtf(GENCODE_GTF_PATH, result_type="dict")
251
+ assert "gene_version" in result
252
+ # values come back as ints when present, pd.NA when missing
253
+ values = list(result["gene_version"].values())
254
+ non_null = [v for v in values if v is not None and not _is_na(v)]
255
+ assert non_null and all(isinstance(v, int) for v in non_null), non_null
256
+
257
+
258
+ def _is_na(v):
259
+ # pd.NA isn't equal-comparable; use bool() guarded form
260
+ try:
261
+ return v != v # NaN-style: pd.NA also returns NA from != itself
262
+ except TypeError:
263
+ return True
264
+
265
+
266
+ # -----------------------------
267
+ # Multi-alias and ordering edge cases
268
+ # -----------------------------
269
+
270
+
271
+ def test_multiple_aliases_mapping_to_same_canonical_first_wins(caplog):
272
+ """Two aliases that both target the same canonical: the first in
273
+ iteration order is renamed; the second is treated as a conflict
274
+ and dropped. Prevents pandas producing two columns with the same
275
+ name."""
276
+ contents = (
277
+ "chr1\tHAVANA\tgene\t1\t100\t.\t+\t.\t"
278
+ 'gene_id "ENSGX"; gene_type "alias_a"; gene_kind "alias_b";\n'
279
+ )
280
+ fd, path = tempfile.mkstemp(suffix=".gtf")
281
+ try:
282
+ with os.fdopen(fd, "w") as fh:
283
+ fh.write(contents)
284
+ with caplog.at_level(logging.WARNING):
285
+ df = read_gtf(
286
+ path,
287
+ # both aliases map to gene_biotype; gene_type comes first
288
+ attribute_aliases={
289
+ "gene_type": "gene_biotype",
290
+ "gene_kind": "gene_biotype",
291
+ },
292
+ result_type="pandas",
293
+ )
294
+ # exactly one gene_biotype column, and the first alias's value wins
295
+ assert list(df.columns).count("gene_biotype") == 1
296
+ assert "gene_type" not in df.columns
297
+ assert "gene_kind" not in df.columns
298
+ assert list(df["gene_biotype"]) == ["alias_a"]
299
+ # the collision was warned about
300
+ assert any("Both alias column" in rec.message for rec in caplog.records)
301
+ finally:
302
+ os.unlink(path)
303
+
304
+
305
+ def test_aliases_iteration_order_matters_for_chains():
306
+ """When canonical of one rule is the alias of another, behavior
307
+ depends on iteration order against the original column set. This
308
+ test pins the current behavior: each rule is decided against the
309
+ pre-rename state, so chains aren't followed. Documents the contract."""
310
+ contents = 'chr1\tHAVANA\tgene\t1\t100\t.\t+\t.\tgene_id "ENSGX"; a "1"; b "2";\n'
311
+ fd, path = tempfile.mkstemp(suffix=".gtf")
312
+ try:
313
+ with os.fdopen(fd, "w") as fh:
314
+ fh.write(contents)
315
+ # {"a": "b"} with b already in the file should drop a (both
316
+ # present, canonical wins). NOT chain to {"a": "b", "b": "c"}.
317
+ df = read_gtf(
318
+ path,
319
+ attribute_aliases={"a": "b"},
320
+ result_type="pandas",
321
+ )
322
+ assert "a" not in df.columns
323
+ assert "b" in df.columns
324
+ assert list(df["b"]) == ["2"]
325
+ finally:
326
+ os.unlink(path)
327
+
328
+
329
+ def test_aliases_preserves_other_attribute_columns():
330
+ """Renaming one attribute must not perturb the others."""
331
+ df = read_gtf(
332
+ GENCODE_GTF_PATH,
333
+ attribute_aliases=GENCODE_BIOTYPE_ALIASES,
334
+ result_type="pandas",
335
+ )
336
+ # spot-check several unrelated columns survive
337
+ for col in ("gene_id", "transcript_id", "gene_name", "transcript_name"):
338
+ assert col in df.columns, col
339
+
340
+
341
+ # -----------------------------
342
+ # usecols interaction
343
+ # -----------------------------
344
+
345
+
346
+ def test_aliases_with_usecols_filtering_canonical_column():
347
+ """usecols is applied AFTER aliasing, so requesting the canonical
348
+ column name returns the data even when the GTF only had the alias."""
349
+ df = read_gtf(
350
+ GENCODE_GTF_PATH,
351
+ attribute_aliases=GENCODE_BIOTYPE_ALIASES,
352
+ usecols=["gene_biotype"],
353
+ result_type="pandas",
354
+ )
355
+ assert list(df.columns) == ["gene_biotype"]
356
+ assert "protein_coding" in set(df["gene_biotype"])
357
+
358
+
359
+ # -----------------------------
360
+ # cast_version_columns edge cases
361
+ # -----------------------------
362
+
363
+
364
+ def test_version_zero_is_preserved_not_treated_as_missing():
365
+ """0 must round-trip as Int64(0), not become pd.NA — the
366
+ .replace("", None) call only targets empty strings."""
367
+ contents = 'chr1\tHAVANA\tgene\t1\t100\t.\t+\t.\tgene_id "ENSGX"; gene_version "0";\n'
368
+ fd, path = tempfile.mkstemp(suffix=".gtf")
369
+ try:
370
+ with os.fdopen(fd, "w") as fh:
371
+ fh.write(contents)
372
+ df = read_gtf(path, result_type="pandas")
373
+ assert df.loc[0, "gene_version"] == 0
374
+ assert not df["gene_version"].isna().iloc[0]
375
+ finally:
376
+ os.unlink(path)
377
+
378
+
379
+ def test_version_with_corrupted_non_integer_value_coerces_to_na(caplog):
380
+ """A malformed version string (e.g. 'v3' instead of '3') is
381
+ coerced to pd.NA rather than raising. Documents the lenient
382
+ behavior — corrupted-GTF data quality bugs become missing
383
+ values, not exceptions."""
384
+ contents = 'chr1\tHAVANA\tgene\t1\t100\t.\t+\t.\tgene_id "ENSGX"; gene_version "v3";\n'
385
+ fd, path = tempfile.mkstemp(suffix=".gtf")
386
+ try:
387
+ with os.fdopen(fd, "w") as fh:
388
+ fh.write(contents)
389
+ df = read_gtf(path, result_type="pandas")
390
+ assert df["gene_version"].isna().iloc[0]
391
+ finally:
392
+ os.unlink(path)
393
+
394
+
395
+ def test_version_cast_is_idempotent_via_double_read():
396
+ """Reading the same file twice with the cast on must produce the
397
+ same dtypes — sanity check that the cast doesn't accumulate
398
+ side effects via the global polars string cache or similar."""
399
+ df1 = read_gtf(GENCODE_GTF_PATH, result_type="pandas")
400
+ df2 = read_gtf(GENCODE_GTF_PATH, result_type="pandas")
401
+ for col in INTEGER_VERSION_COLUMNS:
402
+ assert str(df1[col].dtype) == str(df2[col].dtype) == "Int64"
403
+
404
+
405
+ # -----------------------------
406
+ # Real GENCODE-format fixture (versions in IDs, not in attributes)
407
+ # -----------------------------
408
+
409
+
410
+ def test_real_gencode_format_alias_rename():
411
+ """A GENCODE GTF in its real on-disk shape — versioned IDs
412
+ embedded in gene_id / transcript_id, GENCODE-only fields like
413
+ level / hgnc_id / havana_gene / tag — should normalize to
414
+ Ensembl-style biotype column names cleanly."""
415
+ df = read_gtf(
416
+ GENCODE_REAL_GTF_PATH,
417
+ attribute_aliases=GENCODE_BIOTYPE_ALIASES,
418
+ result_type="pandas",
419
+ )
420
+ assert "gene_biotype" in df.columns
421
+ assert "transcript_biotype" in df.columns
422
+ assert "gene_type" not in df.columns
423
+ gene_rows = df[df["feature"] == "gene"]
424
+ biotypes = set(gene_rows["gene_biotype"])
425
+ assert biotypes == {"transcribed_unprocessed_pseudogene", "protein_coding"}
426
+
427
+
428
+ def test_real_gencode_format_has_no_separate_version_attrs():
429
+ """Real GENCODE doesn't carry separate *_version attribute
430
+ fields — versions are baked into gene_id/transcript_id strings.
431
+ cast_version_columns must be a graceful no-op (not create empty
432
+ columns, not raise)."""
433
+ df = read_gtf(GENCODE_REAL_GTF_PATH, result_type="pandas")
434
+ for col in INTEGER_VERSION_COLUMNS:
435
+ assert col not in df.columns, (
436
+ f"Real GENCODE has no '{col}' attribute; "
437
+ f"version casting should not have synthesized one."
438
+ )
439
+
440
+
441
+ def test_real_gencode_format_preserves_versioned_id_strings():
442
+ """The version baked into gene_id / transcript_id / protein_id /
443
+ exon_id must come through verbatim — those strings ARE the
444
+ canonical identifier in GENCODE."""
445
+ df = read_gtf(GENCODE_REAL_GTF_PATH, result_type="pandas")
446
+ gene_ids = set(df["gene_id"])
447
+ assert "ENSG00000223972.5" in gene_ids
448
+ assert "ENSG00000186092.7" in gene_ids
449
+ assert "ENSG00000198888.2" in gene_ids
450
+ # protein_id and exon_id likewise carry .N
451
+ protein_rows = df[df["feature"] == "CDS"]
452
+ assert "ENSP00000493376.2" in set(protein_rows["protein_id"])
453
+
454
+
455
+ def test_real_gencode_format_preserves_gencode_only_fields():
456
+ """GENCODE-specific attribute fields (level, hgnc_id,
457
+ havana_gene, havana_transcript, tag, transcript_support_level)
458
+ must survive parsing — they're useful downstream signal."""
459
+ df = read_gtf(GENCODE_REAL_GTF_PATH, result_type="pandas")
460
+ for col in (
461
+ "level",
462
+ "hgnc_id",
463
+ "havana_gene",
464
+ "tag",
465
+ "transcript_support_level",
466
+ "havana_transcript",
467
+ ):
468
+ assert col in df.columns, col
469
+ # spot-check values
470
+ transcript_rows = df[df["feature"] == "transcript"].reset_index(drop=True)
471
+ assert "MANE_Select" in set(df["tag"])
472
+ assert "HGNC:14825" in set(df["hgnc_id"])
473
+ # transcript_support_level is "1" (string — we deliberately don't
474
+ # auto-cast it since GENCODE allows non-numeric values like "NA")
475
+ assert "1" in set(transcript_rows["transcript_support_level"])
476
+
477
+
478
+ def test_real_gencode_format_uses_chr_prefixed_seqnames():
479
+ """GENCODE uses 'chr1' / 'chrM' while Ensembl uses bare '1' /
480
+ 'MT'. Confirming the seqname comes through unchanged so
481
+ downstream tooling can normalize if needed."""
482
+ df = read_gtf(GENCODE_REAL_GTF_PATH, result_type="pandas")
483
+ seqnames = set(df["seqname"])
484
+ assert seqnames <= {"chr1", "chrM"}
485
+ assert "chr1" in seqnames
486
+ assert "chrM" in seqnames
487
+
488
+
489
+ def test_real_gencode_format_via_gzip_compressed_input():
490
+ """End-to-end: gzip a real-format GENCODE GTF, point read_gtf at
491
+ the .gz, confirm aliasing + parsing both still work."""
492
+ with open(GENCODE_REAL_GTF_PATH, "rb") as src:
493
+ raw = src.read()
494
+ fd, gz_path = tempfile.mkstemp(suffix=".gtf.gz")
495
+ os.close(fd)
496
+ try:
497
+ with gzip.open(gz_path, "wb") as out:
498
+ out.write(raw)
499
+ df = read_gtf(
500
+ gz_path,
501
+ attribute_aliases=GENCODE_BIOTYPE_ALIASES,
502
+ result_type="pandas",
503
+ )
504
+ assert "gene_biotype" in df.columns
505
+ assert "ENSG00000186092.7" in set(df["gene_id"])
506
+ finally:
507
+ os.unlink(gz_path)
508
+
509
+
510
+ def test_real_gencode_format_with_infer_biotype_column_is_safe():
511
+ """infer_biotype_column would normally fire if 'protein_coding'
512
+ appears in the 'source' column. Real GENCODE's source is
513
+ 'HAVANA' / 'ENSEMBL', not 'protein_coding', so inference
514
+ correctly stays silent. Regression guard against accidentally
515
+ triggering it on GENCODE input."""
516
+ df = read_gtf(
517
+ GENCODE_REAL_GTF_PATH,
518
+ attribute_aliases=GENCODE_BIOTYPE_ALIASES,
519
+ infer_biotype_column=True,
520
+ result_type="pandas",
521
+ )
522
+ # The aliased gene_biotype should be the authoritative one
523
+ gene_rows = df[df["feature"] == "gene"]
524
+ assert set(gene_rows["gene_biotype"]) == {
525
+ "transcribed_unprocessed_pseudogene",
526
+ "protein_coding",
527
+ }
528
+
529
+
530
+ # -----------------------------
531
+ # Ensembl regression checks
532
+ # -----------------------------
533
+
534
+
535
+ def test_old_ensembl_fixture_still_uses_source_for_biotype_inference():
536
+ """Ensembl release-75 puts biotype into the 'source' column. With
537
+ infer_biotype_column=True the fix should still populate
538
+ gene_biotype / transcript_biotype from source. Make sure the new
539
+ kwargs don't break that legacy path."""
540
+ df = read_gtf(
541
+ data_path("ensembl_grch37.head.gtf"),
542
+ infer_biotype_column=True,
543
+ result_type="pandas",
544
+ )
545
+ assert "gene_biotype" in df.columns
546
+ assert "transcript_biotype" in df.columns
547
+ # The release-75 fixture has pseudogenes and protein_coding genes
548
+ assert "protein_coding" in set(df["gene_biotype"])
549
+
550
+
551
+ def test_ensembl_attribute_aliases_no_op_when_aliases_absent():
552
+ """Passing GENCODE_BIOTYPE_ALIASES against a pure Ensembl file
553
+ that doesn't have gene_type / transcript_type must do nothing
554
+ — no spurious columns, no exceptions."""
555
+ df = read_gtf(
556
+ data_path("ensembl_grch37.head.gtf"),
557
+ attribute_aliases=GENCODE_BIOTYPE_ALIASES,
558
+ result_type="pandas",
559
+ )
560
+ # gene_biotype exists in this fixture; aliasing didn't fight with it
561
+ assert "gene_biotype" in df.columns
562
+ assert "gene_type" not in df.columns
563
+ assert "transcript_type" not in df.columns
File without changes
File without changes
File without changes
File without changes
File without changes