gtfparse 2.6.2__tar.gz → 2.7.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. {gtfparse-2.6.2 → gtfparse-2.7.0}/PKG-INFO +13 -3
  2. {gtfparse-2.6.2 → gtfparse-2.7.0}/gtfparse/__init__.py +13 -9
  3. {gtfparse-2.6.2 → gtfparse-2.7.0}/gtfparse/attribute_parsing.py +4 -13
  4. {gtfparse-2.6.2 → gtfparse-2.7.0}/gtfparse/create_missing_features.py +14 -22
  5. {gtfparse-2.6.2 → gtfparse-2.7.0}/gtfparse/parsing_error.py +1 -0
  6. {gtfparse-2.6.2 → gtfparse-2.7.0}/gtfparse/read_gtf.py +188 -76
  7. {gtfparse-2.6.2 → gtfparse-2.7.0}/gtfparse.egg-info/PKG-INFO +13 -3
  8. {gtfparse-2.6.2 → gtfparse-2.7.0}/gtfparse.egg-info/SOURCES.txt +1 -0
  9. {gtfparse-2.6.2 → gtfparse-2.7.0}/gtfparse.egg-info/requires.txt +6 -0
  10. gtfparse-2.7.0/pyproject.toml +104 -0
  11. gtfparse-2.7.0/tests/test_create_missing_features.py +85 -0
  12. gtfparse-2.7.0/tests/test_ensembl_gtf.py +147 -0
  13. {gtfparse-2.6.2 → gtfparse-2.7.0}/tests/test_expand_attributes.py +9 -8
  14. gtfparse-2.7.0/tests/test_gencode_gtf.py +563 -0
  15. {gtfparse-2.6.2 → gtfparse-2.7.0}/tests/test_multiple_values_for_tag_attribute.py +9 -7
  16. {gtfparse-2.6.2 → gtfparse-2.7.0}/tests/test_parse_gtf_lines.py +9 -10
  17. {gtfparse-2.6.2 → gtfparse-2.7.0}/tests/test_read_stringtie_gtf.py +18 -12
  18. {gtfparse-2.6.2 → gtfparse-2.7.0}/tests/test_refseq_gtf.py +10 -2
  19. gtfparse-2.6.2/pyproject.toml +0 -35
  20. gtfparse-2.6.2/tests/test_create_missing_features.py +0 -86
  21. gtfparse-2.6.2/tests/test_ensembl_gtf.py +0 -81
  22. {gtfparse-2.6.2 → gtfparse-2.7.0}/LICENSE +0 -0
  23. {gtfparse-2.6.2 → gtfparse-2.7.0}/README.md +0 -0
  24. {gtfparse-2.6.2 → gtfparse-2.7.0}/gtfparse.egg-info/dependency_links.txt +0 -0
  25. {gtfparse-2.6.2 → gtfparse-2.7.0}/gtfparse.egg-info/top_level.txt +0 -0
  26. {gtfparse-2.6.2 → gtfparse-2.7.0}/requirements.txt +0 -0
  27. {gtfparse-2.6.2 → gtfparse-2.7.0}/setup.cfg +0 -0
@@ -1,23 +1,33 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gtfparse
3
- Version: 2.6.2
3
+ Version: 2.7.0
4
4
  Summary: Parsing library for extracting data frames of genomic features from GTF files
5
5
  Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
6
6
  Project-URL: Homepage, https://github.com/openvax/gtfparse
7
- Project-URL: Bug Tracker, https://github.com/openvax/gtfparse
7
+ Project-URL: Bug Tracker, https://github.com/openvax/gtfparse/issues
8
8
  Classifier: Development Status :: 4 - Beta
9
9
  Classifier: Environment :: Console
10
10
  Classifier: Operating System :: OS Independent
11
11
  Classifier: Intended Audience :: Science/Research
12
12
  Classifier: License :: OSI Approved :: Apache Software License
13
13
  Classifier: Programming Language :: Python
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.9
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
14
19
  Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
15
- Requires-Python: >=3.7
20
+ Requires-Python: >=3.9
16
21
  Description-Content-Type: text/markdown
17
22
  License-File: LICENSE
18
23
  Requires-Dist: polars>=0.20.2
19
24
  Requires-Dist: pyarrow>=18.0.0
20
25
  Requires-Dist: pandas>=2.1.0
26
+ Provides-Extra: dev
27
+ Requires-Dist: pytest; extra == "dev"
28
+ Requires-Dist: pytest-cov; extra == "dev"
29
+ Requires-Dist: ruff; extra == "dev"
30
+ Requires-Dist: coveralls; extra == "dev"
21
31
  Dynamic: license-file
22
32
 
23
33
  [![Tests](https://github.com/openvax/gtfparse/actions/workflows/tests.yml/badge.svg)](https://github.com/openvax/gtfparse/actions/workflows/tests.yml)
@@ -14,23 +14,27 @@ from .attribute_parsing import expand_attribute_strings
14
14
  from .create_missing_features import create_missing_features
15
15
  from .parsing_error import ParsingError
16
16
  from .read_gtf import (
17
- read_gtf,
17
+ GENCODE_BIOTYPE_ALIASES,
18
+ INTEGER_VERSION_COLUMNS,
19
+ REQUIRED_COLUMNS,
18
20
  parse_gtf,
19
- parse_gtf_pandas,
20
21
  parse_gtf_and_expand_attributes,
21
- REQUIRED_COLUMNS,
22
+ parse_gtf_pandas,
23
+ read_gtf,
22
24
  )
23
25
 
24
- __version__ = "2.6.2"
26
+ __version__ = "2.7.0"
25
27
 
26
28
  __all__ = [
27
- "__version__",
28
- "expand_attribute_strings",
29
- "create_missing_features",
30
- "parse_gtf_and_expand_attributes",
29
+ "GENCODE_BIOTYPE_ALIASES",
30
+ "INTEGER_VERSION_COLUMNS",
31
31
  "REQUIRED_COLUMNS",
32
32
  "ParsingError",
33
- "read_gtf",
33
+ "__version__",
34
+ "create_missing_features",
35
+ "expand_attribute_strings",
34
36
  "parse_gtf",
37
+ "parse_gtf_and_expand_attributes",
35
38
  "parse_gtf_pandas",
39
+ "read_gtf",
36
40
  ]
@@ -18,12 +18,7 @@ logging.basicConfig(level=logging.INFO)
18
18
  logger = logging.getLogger(__name__)
19
19
 
20
20
 
21
-
22
- def expand_attribute_strings(
23
- attribute_strings,
24
- quote_char="'",
25
- missing_value="",
26
- usecols=None):
21
+ def expand_attribute_strings(attribute_strings, quote_char="'", missing_value="", usecols=None):
27
22
  """
28
23
  The last column of GTF has a variable number of key value pairs
29
24
  of the format: "key1 value1; key2 value2;"
@@ -66,7 +61,7 @@ def expand_attribute_strings(
66
61
  # and pair of try/except blocks in the loop.
67
62
  column_interned_strings = {}
68
63
 
69
- for (i, kv_strings) in enumerate(attribute_strings):
64
+ for i, kv_strings in enumerate(attribute_strings):
70
65
  if type(kv_strings) is str:
71
66
  kv_strings = kv_strings.split(";")
72
67
  for kv in kv_strings:
@@ -92,7 +87,7 @@ def expand_attribute_strings(
92
87
 
93
88
  if value[0] == quote_char:
94
89
  value = value.replace(quote_char, "")
95
-
90
+
96
91
  try:
97
92
  column = extra_columns[column_name]
98
93
  # if an attribute is used repeatedly then
@@ -108,9 +103,5 @@ def expand_attribute_strings(
108
103
  extra_columns[column_name] = column
109
104
  column_order.append(column_name)
110
105
 
111
-
112
-
113
106
  logging.info("Extracted GTF attributes: %s" % column_order)
114
- return OrderedDict(
115
- (column_name, extra_columns[column_name])
116
- for column_name in column_order)
107
+ return OrderedDict((column_name, extra_columns[column_name]) for column_name in column_order)
@@ -19,11 +19,7 @@ logging.basicConfig(level=logging.INFO)
19
19
  logger = logging.getLogger(__name__)
20
20
 
21
21
 
22
- def create_missing_features(
23
- dataframe,
24
- unique_keys={},
25
- extra_columns={},
26
- missing_value=None):
22
+ def create_missing_features(dataframe, unique_keys={}, extra_columns={}, missing_value=None):
27
23
  """
28
24
  Helper function used to construct a missing feature such as 'transcript'
29
25
  or 'gene'. Some GTF files only have 'exon' and 'CDS' entries, but have
@@ -49,29 +45,25 @@ def create_missing_features(
49
45
  missing_value : any
50
46
  Which value to fill in for columns that we don't infer values for.
51
47
 
52
- Returns original dataframe (converted to Pandas if necessary) along with all
48
+ Returns original dataframe (converted to Pandas if necessary) along with all
53
49
  extra rows created for missing features.
54
50
  """
55
51
  if hasattr(dataframe, "to_pandas"):
56
52
  dataframe = dataframe.to_pandas()
57
-
53
+
58
54
  extra_dataframes = []
59
55
 
60
56
  existing_features = set(dataframe["feature"])
61
57
  existing_columns = set(dataframe.columns)
62
-
63
- for (feature_name, groupby_key) in unique_keys.items():
64
-
58
+
59
+ for feature_name, groupby_key in unique_keys.items():
65
60
  if feature_name in existing_features:
66
- logging.info(
67
- "Feature '%s' already exists in GTF data" % feature_name)
61
+ logging.info("Feature '%s' already exists in GTF data" % feature_name)
68
62
  continue
69
63
  logging.info("Creating rows for missing feature '%s'" % feature_name)
70
64
 
71
65
  # don't include rows where the groupby key was missing
72
- missing = pd.Series([
73
- x is None or x == ""
74
- for x in dataframe[groupby_key]])
66
+ missing = pd.Series([x is None or x == "" for x in dataframe[groupby_key]])
75
67
  not_missing = ~missing
76
68
  row_groups = dataframe[not_missing].groupby(groupby_key)
77
69
 
@@ -79,10 +71,9 @@ def create_missing_features(
79
71
  # other columns may or may not be uniquely defined. Start off by
80
72
  # assuming the values for every column are missing and fill them in
81
73
  # where possible.
82
- feature_values = OrderedDict([
83
- (column_name, [missing_value] * row_groups.ngroups)
84
- for column_name in dataframe.keys()
85
- ])
74
+ feature_values = OrderedDict(
75
+ [(column_name, [missing_value] * row_groups.ngroups) for column_name in dataframe]
76
+ )
86
77
 
87
78
  # User specifies which non-required columns should we try to infer
88
79
  # values for
@@ -111,8 +102,9 @@ def create_missing_features(
111
102
  for column_name in feature_columns:
112
103
  if column_name not in existing_columns:
113
104
  raise ValueError(
114
- "Column '%s' does not exist in GTF, columns = %s" % (
115
- column_name, existing_columns))
105
+ "Column '%s' does not exist in GTF, columns = %s"
106
+ % (column_name, existing_columns)
107
+ )
116
108
 
117
109
  # expect that all entries related to a reconstructed feature
118
110
  # are related and are thus within the same interval of
@@ -121,4 +113,4 @@ def create_missing_features(
121
113
  if len(unique_values) == 1:
122
114
  feature_values[column_name][i] = unique_values[0]
123
115
  extra_dataframes.append(pd.DataFrame(feature_values))
124
- return pd.concat([dataframe] + extra_dataframes, ignore_index=True)
116
+ return pd.concat([dataframe, *extra_dataframes], ignore_index=True)
@@ -10,5 +10,6 @@
10
10
  # See the License for the specific language governing permissions and
11
11
  # limitations under the License.
12
12
 
13
+
13
14
  class ParsingError(Exception):
14
15
  pass
@@ -13,16 +13,37 @@
13
13
  import logging
14
14
  from os.path import exists
15
15
 
16
- import polars
16
+ import pandas as pd
17
+ import polars
17
18
 
18
19
  from .attribute_parsing import expand_attribute_strings
19
20
  from .parsing_error import ParsingError
20
21
 
21
-
22
22
  logging.basicConfig(level=logging.INFO)
23
23
  logger = logging.getLogger(__name__)
24
24
 
25
25
 
26
+ # GENCODE GTFs use *_type where Ensembl GTFs use *_biotype. Pass this
27
+ # (or a superset) as `attribute_aliases` to read_gtf to normalize a
28
+ # GENCODE-format GTF onto the Ensembl column names that downstream
29
+ # tools like pyensembl expect.
30
+ GENCODE_BIOTYPE_ALIASES = {
31
+ "gene_type": "gene_biotype",
32
+ "transcript_type": "transcript_biotype",
33
+ }
34
+
35
+
36
+ # Ensembl-style attribute columns that are always integer-valued when
37
+ # present. read_gtf casts these from string to pandas nullable Int64 by
38
+ # default; pass cast_version_columns=False to keep them as strings.
39
+ INTEGER_VERSION_COLUMNS = (
40
+ "gene_version",
41
+ "transcript_version",
42
+ "protein_version",
43
+ "exon_version",
44
+ )
45
+
46
+
26
47
  """
27
48
  Columns of a GTF file:
28
49
 
@@ -76,88 +97,79 @@ REQUIRED_COLUMNS = [
76
97
 
77
98
 
78
99
  DEFAULT_COLUMN_DTYPES = {
79
- "seqname": polars.Categorical,
80
- "source": polars.Categorical,
81
-
100
+ "seqname": polars.Categorical,
101
+ "source": polars.Categorical,
82
102
  "start": polars.Int64,
83
103
  "end": polars.Int64,
84
104
  "score": polars.Float32,
85
-
86
- "feature": polars.Categorical,
87
- "strand": polars.Categorical,
105
+ "feature": polars.Categorical,
106
+ "strand": polars.Categorical,
88
107
  "frame": polars.UInt32,
89
108
  }
90
109
 
110
+
91
111
  def parse_with_polars_lazy(
92
- filepath_or_buffer,
93
- split_attributes=True,
94
- features=None,
95
- fix_quotes_columns=["attribute"]):
112
+ filepath_or_buffer, split_attributes=True, features=None, fix_quotes_columns=["attribute"]
113
+ ):
96
114
  # use a global string cache so that all strings get intern'd into
97
115
  # a single numbering system
98
116
  polars.enable_string_cache()
99
- kwargs = dict(
100
- has_header=False,
101
- separator="\t",
102
- comment_prefix="#",
103
- null_values=".",
104
- schema_overrides=DEFAULT_COLUMN_DTYPES)
117
+ kwargs = {
118
+ "has_header": False,
119
+ "separator": "\t",
120
+ "comment_prefix": "#",
121
+ "null_values": ".",
122
+ "schema_overrides": DEFAULT_COLUMN_DTYPES,
123
+ }
105
124
  try:
106
- df = polars.read_csv(
107
- filepath_or_buffer,
108
- new_columns=REQUIRED_COLUMNS,
109
- **kwargs).lazy()
110
- except polars.exceptions.ShapeError:
111
- raise ParsingError("Wrong number of columns")
125
+ df = polars.read_csv(filepath_or_buffer, new_columns=REQUIRED_COLUMNS, **kwargs).lazy()
126
+ except polars.exceptions.ShapeError as err:
127
+ raise ParsingError("Wrong number of columns") from err
112
128
 
113
129
  # Drop empty lines that may appear as all-null rows
114
130
  df = df.filter(polars.col("seqname").is_not_null())
115
131
 
116
- df = df.with_columns([
117
- polars.col("frame").fill_null(0),
118
- polars.col("attribute").str.replace_all('"', "'")
119
- ])
120
-
132
+ df = df.with_columns(
133
+ [polars.col("frame").fill_null(0), polars.col("attribute").str.replace_all('"', "'")]
134
+ )
135
+
121
136
  for fix_quotes_column in fix_quotes_columns:
122
137
  # Catch mistaken semicolons by replacing "xyz;" with "xyz"
123
138
  # Required to do this since the Ensembl GTF for Ensembl
124
139
  # release 78 has mistakes such as:
125
140
  # gene_name = "PRAMEF6;" transcript_name = "PRAMEF6;-201"
126
- df = df.with_columns([
127
- polars.col(fix_quotes_column).str.replace(';\"', '\"').str.replace(";-", "-")
128
- ])
141
+ df = df.with_columns(
142
+ [polars.col(fix_quotes_column).str.replace(';"', '"').str.replace(";-", "-")]
143
+ )
129
144
 
130
145
  if features is not None:
131
146
  features = sorted(set(features))
132
147
  df = df.filter(polars.col("feature").is_in(features))
133
148
 
134
-
135
149
  if split_attributes:
136
- df = df.with_columns([
137
- polars.col("attribute").str.split(";").alias("attribute_split")
138
- ])
150
+ df = df.with_columns([polars.col("attribute").str.split(";").alias("attribute_split")])
139
151
  return df
140
152
 
153
+
141
154
  def parse_gtf(
142
- filepath_or_buffer,
143
- split_attributes=True,
144
- features=None,
145
- fix_quotes_columns=["attribute"]):
155
+ filepath_or_buffer, split_attributes=True, features=None, fix_quotes_columns=["attribute"]
156
+ ):
146
157
  df_lazy = parse_with_polars_lazy(
147
158
  filepath_or_buffer=filepath_or_buffer,
148
159
  split_attributes=split_attributes,
149
160
  features=features,
150
- fix_quotes_columns=fix_quotes_columns)
161
+ fix_quotes_columns=fix_quotes_columns,
162
+ )
151
163
  return df_lazy.collect()
152
164
 
165
+
153
166
  def parse_gtf_pandas(*args, **kwargs):
154
167
  return parse_gtf(*args, **kwargs).to_pandas()
155
168
 
156
-
169
+
157
170
  def parse_gtf_and_expand_attributes(
158
- filepath_or_buffer,
159
- restrict_attribute_columns=None,
160
- features=None):
171
+ filepath_or_buffer, restrict_attribute_columns=None, features=None
172
+ ):
161
173
  """
162
174
  Parse lines into column->values dictionary and then expand
163
175
  the 'attribute' column into multiple columns. This expansion happens
@@ -177,33 +189,93 @@ def parse_gtf_and_expand_attributes(
177
189
  features : set or None
178
190
  Ignore entries which don't correspond to one of the supplied features
179
191
  """
180
- df = parse_gtf(
181
- filepath_or_buffer=filepath_or_buffer,
182
- features=features,
183
- split_attributes=True)
192
+ df = parse_gtf(filepath_or_buffer=filepath_or_buffer, features=features, split_attributes=True)
184
193
  if type(restrict_attribute_columns) is str:
185
194
  restrict_attribute_columns = {restrict_attribute_columns}
186
195
  elif restrict_attribute_columns:
187
196
  restrict_attribute_columns = set(restrict_attribute_columns)
188
197
  df.drop_in_place("attribute")
189
198
  attribute_pairs = df.drop_in_place("attribute_split")
190
- return df.with_columns([
191
- polars.Series(k, vs)
192
- for (k, vs) in
193
- expand_attribute_strings(attribute_pairs).items()
194
- if restrict_attribute_columns is None or k in restrict_attribute_columns
195
- ])
196
-
199
+ return df.with_columns(
200
+ [
201
+ polars.Series(k, vs)
202
+ for (k, vs) in expand_attribute_strings(attribute_pairs).items()
203
+ if restrict_attribute_columns is None or k in restrict_attribute_columns
204
+ ]
205
+ )
206
+
207
+
208
+ def _apply_attribute_aliases(result_df, attribute_aliases):
209
+ """
210
+ Rename alias attribute columns onto canonical names in-place.
211
+
212
+ For each (alias -> canonical) pair, in iteration order:
213
+ * if only the alias is present, rename it to the canonical name.
214
+ * if both are present, drop the alias and warn (canonical wins).
215
+ * if neither is present, do nothing.
216
+
217
+ When two aliases target the same canonical (e.g. both ``gene_type``
218
+ and a hypothetical ``gene_kind`` map to ``gene_biotype``), the first
219
+ rename in iteration order wins; subsequent aliases targeting an
220
+ already-renamed canonical are treated as collisions, dropped, and
221
+ warned about.
222
+ """
223
+ if not attribute_aliases:
224
+ return result_df
225
+ columns_present = set(result_df.columns)
226
+ rename_map = {}
227
+ drop_aliases = []
228
+ for alias, canonical in attribute_aliases.items():
229
+ if alias not in columns_present:
230
+ continue
231
+ if canonical in columns_present:
232
+ logger.warning(
233
+ "Both alias column '%s' and canonical column '%s' are present; "
234
+ "dropping alias and keeping canonical values.",
235
+ alias,
236
+ canonical,
237
+ )
238
+ drop_aliases.append(alias)
239
+ else:
240
+ rename_map[alias] = canonical
241
+ # Reflect the rename in the running column set so a later
242
+ # alias mapping to the same canonical sees the collision
243
+ # instead of silently producing a duplicate-named column.
244
+ columns_present.discard(alias)
245
+ columns_present.add(canonical)
246
+ if drop_aliases:
247
+ result_df = result_df.drop(columns=drop_aliases)
248
+ if rename_map:
249
+ result_df = result_df.rename(columns=rename_map)
250
+ return result_df
251
+
252
+
253
+ def _cast_version_columns(result_df, version_columns=INTEGER_VERSION_COLUMNS):
254
+ """
255
+ Cast known Ensembl *_version attribute columns from strings to
256
+ pandas nullable Int64 in-place. Missing/empty values become pd.NA.
257
+ """
258
+ for column_name in version_columns:
259
+ if column_name not in result_df.columns:
260
+ continue
261
+ result_df[column_name] = pd.to_numeric(
262
+ result_df[column_name].replace("", None), errors="coerce"
263
+ ).astype("Int64")
264
+ return result_df
265
+
197
266
 
198
267
  def read_gtf(
199
- filepath_or_buffer,
200
- expand_attribute_column=True,
201
- infer_biotype_column=False,
202
- column_converters={},
203
- column_cast_types={},
204
- usecols=None,
205
- features=None,
206
- result_type='polars'):
268
+ filepath_or_buffer,
269
+ expand_attribute_column=True,
270
+ infer_biotype_column=False,
271
+ column_converters={},
272
+ column_cast_types={},
273
+ usecols=None,
274
+ features=None,
275
+ result_type="polars",
276
+ attribute_aliases=None,
277
+ cast_version_columns=True,
278
+ ):
207
279
  """
208
280
  Parse a GTF into a dictionary mapping column names to sequences of values.
209
281
 
@@ -231,7 +303,7 @@ def read_gtf(
231
303
  column_cast_types : dict, optional
232
304
  Dictionary mapping column names to dtypes. Will cast columns to given
233
305
  Polars types.
234
-
306
+
235
307
  usecols : list of str or None
236
308
  Restrict which columns are loaded to the give set. If None, then
237
309
  load all columns.
@@ -240,17 +312,47 @@ def read_gtf(
240
312
  Drop rows which aren't one of the features in the supplied set
241
313
 
242
314
  result_type : One of 'polars', 'pandas', or 'dict'
243
- Default behavior is to return a Polars DataFrame, but will convert to
315
+ Default behavior is to return a Polars DataFrame, but will convert to
244
316
  Pandas DataFrame or dictionary if specified.
317
+
318
+ attribute_aliases : dict of str -> str, optional
319
+ Maps alias attribute names onto canonical ones. After attributes
320
+ are expanded into columns, each alias column is renamed to its
321
+ canonical name when the canonical column is absent. If both are
322
+ present the alias is dropped and a warning is logged. Pass
323
+ `GENCODE_BIOTYPE_ALIASES` to normalize a GENCODE GTF's
324
+ `gene_type`/`transcript_type` onto Ensembl's
325
+ `gene_biotype`/`transcript_biotype`.
326
+
327
+ cast_version_columns : bool
328
+ When True (default), cast the well-known integer version
329
+ attribute columns (`gene_version`, `transcript_version`,
330
+ `protein_version`, `exon_version`) from strings to pandas
331
+ nullable Int64 when present. Set to False to keep them as
332
+ strings.
245
333
  """
246
334
  if type(filepath_or_buffer) is str and not exists(filepath_or_buffer):
247
335
  raise ValueError("GTF file does not exist: %s" % filepath_or_buffer)
248
336
 
337
+ # If usecols asks for a canonical column that's only present in the
338
+ # GTF under an alias name, expand the parse-time column filter to
339
+ # also pull the alias through — otherwise it gets dropped at parse
340
+ # time before _apply_attribute_aliases can see it. The end-of-function
341
+ # usecols filter still narrows the result down to the canonical name.
342
+ parse_usecols = usecols
343
+ if usecols is not None and attribute_aliases:
344
+ usecols_set = set(usecols)
345
+ parse_usecols = set(usecols_set)
346
+ for alias, canonical in attribute_aliases.items():
347
+ if canonical in usecols_set:
348
+ parse_usecols.add(alias)
349
+
249
350
  if expand_attribute_column:
250
351
  result_df = parse_gtf_and_expand_attributes(
251
352
  filepath_or_buffer,
252
- restrict_attribute_columns=usecols,
253
- features=features)
353
+ restrict_attribute_columns=parse_usecols,
354
+ features=features,
355
+ )
254
356
  else:
255
357
  result_df = parse_gtf(result_df, features=features)
256
358
 
@@ -259,26 +361,36 @@ def read_gtf(
259
361
  # and are generally insane to chase down
260
362
  result_df = result_df.to_pandas()
261
363
  if column_converters or column_cast_types:
364
+
262
365
  def wrap_to_always_accept_none(f):
263
366
  def wrapped_fn(x):
264
367
  if x is None or x == "":
265
368
  return None
266
369
  else:
267
370
  return f(x)
371
+
268
372
  return wrapped_fn
269
-
373
+
270
374
  column_names = set(column_converters.keys()).union(column_cast_types.keys())
271
375
  for column_name in column_names:
272
-
273
376
  if column_name in column_converters:
274
- column_fn = wrap_to_always_accept_none(
275
- column_converters[column_name])
377
+ column_fn = wrap_to_always_accept_none(column_converters[column_name])
276
378
  result_df[column_name] = result_df[column_name].apply(column_fn)
277
379
 
278
380
  if column_name in column_cast_types:
279
381
  column_type = column_cast_types[column_name]
280
382
  result_df[column_name] = result_df[column_name].astype(column_type)
281
-
383
+
384
+ # Rename alias attribute columns onto their canonical names. Done before
385
+ # infer_biotype_column so an aliased gene_biotype/transcript_biotype is
386
+ # visible to the inference logic.
387
+ result_df = _apply_attribute_aliases(result_df, attribute_aliases)
388
+
389
+ # Cast Ensembl *_version columns from strings to nullable integers so
390
+ # downstream consumers (e.g. pyensembl) don't have to int(...) themselves.
391
+ if cast_version_columns:
392
+ result_df = _cast_version_columns(result_df)
393
+
282
394
  # Hackishly infer whether the values in the 'source' column of this GTF
283
395
  # are actually representing a biotype by checking for the most common
284
396
  # gene_biotype and transcript_biotype value 'protein_coding'
@@ -292,11 +404,11 @@ def read_gtf(
292
404
  # gene_biotype)
293
405
  if "gene_biotype" not in column_names:
294
406
  logging.info("Using column 'source' to replace missing 'gene_biotype'")
295
- result_df['gene_biotype'] = result_df['source']
407
+ result_df["gene_biotype"] = result_df["source"]
296
408
  if "transcript_biotype" not in column_names:
297
409
  logging.info("Using column 'source' to replace missing 'transcript_biotype'")
298
- result_df['transcript_biotype'] = result_df['source']
299
-
410
+ result_df["transcript_biotype"] = result_df["source"]
411
+
300
412
  if usecols is not None:
301
413
  column_names = set(result_df.columns)
302
414
  valid_columns = [c for c in usecols if c in column_names]
@@ -1,23 +1,33 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gtfparse
3
- Version: 2.6.2
3
+ Version: 2.7.0
4
4
  Summary: Parsing library for extracting data frames of genomic features from GTF files
5
5
  Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
6
6
  Project-URL: Homepage, https://github.com/openvax/gtfparse
7
- Project-URL: Bug Tracker, https://github.com/openvax/gtfparse
7
+ Project-URL: Bug Tracker, https://github.com/openvax/gtfparse/issues
8
8
  Classifier: Development Status :: 4 - Beta
9
9
  Classifier: Environment :: Console
10
10
  Classifier: Operating System :: OS Independent
11
11
  Classifier: Intended Audience :: Science/Research
12
12
  Classifier: License :: OSI Approved :: Apache Software License
13
13
  Classifier: Programming Language :: Python
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.9
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
14
19
  Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
15
- Requires-Python: >=3.7
20
+ Requires-Python: >=3.9
16
21
  Description-Content-Type: text/markdown
17
22
  License-File: LICENSE
18
23
  Requires-Dist: polars>=0.20.2
19
24
  Requires-Dist: pyarrow>=18.0.0
20
25
  Requires-Dist: pandas>=2.1.0
26
+ Provides-Extra: dev
27
+ Requires-Dist: pytest; extra == "dev"
28
+ Requires-Dist: pytest-cov; extra == "dev"
29
+ Requires-Dist: ruff; extra == "dev"
30
+ Requires-Dist: coveralls; extra == "dev"
21
31
  Dynamic: license-file
22
32
 
23
33
  [![Tests](https://github.com/openvax/gtfparse/actions/workflows/tests.yml/badge.svg)](https://github.com/openvax/gtfparse/actions/workflows/tests.yml)
@@ -16,6 +16,7 @@ gtfparse/../requirements.txt
16
16
  tests/test_create_missing_features.py
17
17
  tests/test_ensembl_gtf.py
18
18
  tests/test_expand_attributes.py
19
+ tests/test_gencode_gtf.py
19
20
  tests/test_multiple_values_for_tag_attribute.py
20
21
  tests/test_parse_gtf_lines.py
21
22
  tests/test_read_stringtie_gtf.py
@@ -1,3 +1,9 @@
1
1
  polars>=0.20.2
2
2
  pyarrow>=18.0.0
3
3
  pandas>=2.1.0
4
+
5
+ [dev]
6
+ pytest
7
+ pytest-cov
8
+ ruff
9
+ coveralls