gtfparse 2.6.0__tar.gz → 2.6.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (26) hide show
  1. {gtfparse-2.6.0 → gtfparse-2.6.3}/PKG-INFO +13 -3
  2. {gtfparse-2.6.0 → gtfparse-2.6.3}/gtfparse/__init__.py +10 -11
  3. {gtfparse-2.6.0 → gtfparse-2.6.3}/gtfparse/attribute_parsing.py +4 -13
  4. {gtfparse-2.6.0 → gtfparse-2.6.3}/gtfparse/create_missing_features.py +14 -22
  5. {gtfparse-2.6.0 → gtfparse-2.6.3}/gtfparse/parsing_error.py +1 -0
  6. {gtfparse-2.6.0 → gtfparse-2.6.3}/gtfparse/read_gtf.py +65 -77
  7. {gtfparse-2.6.0 → gtfparse-2.6.3}/gtfparse.egg-info/PKG-INFO +13 -3
  8. {gtfparse-2.6.0 → gtfparse-2.6.3}/gtfparse.egg-info/requires.txt +6 -0
  9. gtfparse-2.6.3/pyproject.toml +104 -0
  10. gtfparse-2.6.3/tests/test_create_missing_features.py +85 -0
  11. gtfparse-2.6.3/tests/test_ensembl_gtf.py +147 -0
  12. {gtfparse-2.6.0 → gtfparse-2.6.3}/tests/test_expand_attributes.py +9 -8
  13. {gtfparse-2.6.0 → gtfparse-2.6.3}/tests/test_multiple_values_for_tag_attribute.py +9 -7
  14. {gtfparse-2.6.0 → gtfparse-2.6.3}/tests/test_parse_gtf_lines.py +9 -10
  15. {gtfparse-2.6.0 → gtfparse-2.6.3}/tests/test_read_stringtie_gtf.py +18 -12
  16. {gtfparse-2.6.0 → gtfparse-2.6.3}/tests/test_refseq_gtf.py +10 -2
  17. gtfparse-2.6.0/pyproject.toml +0 -27
  18. gtfparse-2.6.0/tests/test_create_missing_features.py +0 -86
  19. gtfparse-2.6.0/tests/test_ensembl_gtf.py +0 -81
  20. {gtfparse-2.6.0 → gtfparse-2.6.3}/LICENSE +0 -0
  21. {gtfparse-2.6.0 → gtfparse-2.6.3}/README.md +0 -0
  22. {gtfparse-2.6.0 → gtfparse-2.6.3}/gtfparse.egg-info/SOURCES.txt +0 -0
  23. {gtfparse-2.6.0 → gtfparse-2.6.3}/gtfparse.egg-info/dependency_links.txt +0 -0
  24. {gtfparse-2.6.0 → gtfparse-2.6.3}/gtfparse.egg-info/top_level.txt +0 -0
  25. {gtfparse-2.6.0 → gtfparse-2.6.3}/requirements.txt +0 -0
  26. {gtfparse-2.6.0 → gtfparse-2.6.3}/setup.cfg +0 -0
@@ -1,23 +1,33 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gtfparse
3
- Version: 2.6.0
3
+ Version: 2.6.3
4
4
  Summary: Parsing library for extracting data frames of genomic features from GTF files
5
5
  Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
6
6
  Project-URL: Homepage, https://github.com/openvax/gtfparse
7
- Project-URL: Bug Tracker, https://github.com/openvax/gtfparse
7
+ Project-URL: Bug Tracker, https://github.com/openvax/gtfparse/issues
8
8
  Classifier: Development Status :: 4 - Beta
9
9
  Classifier: Environment :: Console
10
10
  Classifier: Operating System :: OS Independent
11
11
  Classifier: Intended Audience :: Science/Research
12
12
  Classifier: License :: OSI Approved :: Apache Software License
13
13
  Classifier: Programming Language :: Python
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.9
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
14
19
  Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
15
- Requires-Python: >=3.7
20
+ Requires-Python: >=3.9
16
21
  Description-Content-Type: text/markdown
17
22
  License-File: LICENSE
18
23
  Requires-Dist: polars>=0.20.2
19
24
  Requires-Dist: pyarrow>=18.0.0
20
25
  Requires-Dist: pandas>=2.1.0
26
+ Provides-Extra: dev
27
+ Requires-Dist: pytest; extra == "dev"
28
+ Requires-Dist: pytest-cov; extra == "dev"
29
+ Requires-Dist: ruff; extra == "dev"
30
+ Requires-Dist: coveralls; extra == "dev"
21
31
  Dynamic: license-file
22
32
 
23
33
  [![Tests](https://github.com/openvax/gtfparse/actions/workflows/tests.yml/badge.svg)](https://github.com/openvax/gtfparse/actions/workflows/tests.yml)
@@ -14,24 +14,23 @@ from .attribute_parsing import expand_attribute_strings
14
14
  from .create_missing_features import create_missing_features
15
15
  from .parsing_error import ParsingError
16
16
  from .read_gtf import (
17
- read_gtf,
18
- parse_gtf,
19
- parse_gtf_pandas,
20
- parse_gtf_and_expand_attributes,
21
17
  REQUIRED_COLUMNS,
18
+ parse_gtf,
19
+ parse_gtf_and_expand_attributes,
20
+ parse_gtf_pandas,
21
+ read_gtf,
22
22
  )
23
23
 
24
- __version__ = "2.6.0"
24
+ __version__ = "2.6.3"
25
25
 
26
26
  __all__ = [
27
- "__version__",
28
- "expand_attribute_strings",
29
- "create_missing_features",
30
-
31
- "parse_gtf_and_expand_attributes",
32
27
  "REQUIRED_COLUMNS",
33
28
  "ParsingError",
34
- "read_gtf",
29
+ "__version__",
30
+ "create_missing_features",
31
+ "expand_attribute_strings",
35
32
  "parse_gtf",
33
+ "parse_gtf_and_expand_attributes",
36
34
  "parse_gtf_pandas",
35
+ "read_gtf",
37
36
  ]
@@ -18,12 +18,7 @@ logging.basicConfig(level=logging.INFO)
18
18
  logger = logging.getLogger(__name__)
19
19
 
20
20
 
21
-
22
- def expand_attribute_strings(
23
- attribute_strings,
24
- quote_char="'",
25
- missing_value="",
26
- usecols=None):
21
+ def expand_attribute_strings(attribute_strings, quote_char="'", missing_value="", usecols=None):
27
22
  """
28
23
  The last column of GTF has a variable number of key value pairs
29
24
  of the format: "key1 value1; key2 value2;"
@@ -66,7 +61,7 @@ def expand_attribute_strings(
66
61
  # and pair of try/except blocks in the loop.
67
62
  column_interned_strings = {}
68
63
 
69
- for (i, kv_strings) in enumerate(attribute_strings):
64
+ for i, kv_strings in enumerate(attribute_strings):
70
65
  if type(kv_strings) is str:
71
66
  kv_strings = kv_strings.split(";")
72
67
  for kv in kv_strings:
@@ -92,7 +87,7 @@ def expand_attribute_strings(
92
87
 
93
88
  if value[0] == quote_char:
94
89
  value = value.replace(quote_char, "")
95
-
90
+
96
91
  try:
97
92
  column = extra_columns[column_name]
98
93
  # if an attribute is used repeatedly then
@@ -108,9 +103,5 @@ def expand_attribute_strings(
108
103
  extra_columns[column_name] = column
109
104
  column_order.append(column_name)
110
105
 
111
-
112
-
113
106
  logging.info("Extracted GTF attributes: %s" % column_order)
114
- return OrderedDict(
115
- (column_name, extra_columns[column_name])
116
- for column_name in column_order)
107
+ return OrderedDict((column_name, extra_columns[column_name]) for column_name in column_order)
@@ -19,11 +19,7 @@ logging.basicConfig(level=logging.INFO)
19
19
  logger = logging.getLogger(__name__)
20
20
 
21
21
 
22
- def create_missing_features(
23
- dataframe,
24
- unique_keys={},
25
- extra_columns={},
26
- missing_value=None):
22
+ def create_missing_features(dataframe, unique_keys={}, extra_columns={}, missing_value=None):
27
23
  """
28
24
  Helper function used to construct a missing feature such as 'transcript'
29
25
  or 'gene'. Some GTF files only have 'exon' and 'CDS' entries, but have
@@ -49,29 +45,25 @@ def create_missing_features(
49
45
  missing_value : any
50
46
  Which value to fill in for columns that we don't infer values for.
51
47
 
52
- Returns original dataframe (converted to Pandas if necessary) along with all
48
+ Returns original dataframe (converted to Pandas if necessary) along with all
53
49
  extra rows created for missing features.
54
50
  """
55
51
  if hasattr(dataframe, "to_pandas"):
56
52
  dataframe = dataframe.to_pandas()
57
-
53
+
58
54
  extra_dataframes = []
59
55
 
60
56
  existing_features = set(dataframe["feature"])
61
57
  existing_columns = set(dataframe.columns)
62
-
63
- for (feature_name, groupby_key) in unique_keys.items():
64
-
58
+
59
+ for feature_name, groupby_key in unique_keys.items():
65
60
  if feature_name in existing_features:
66
- logging.info(
67
- "Feature '%s' already exists in GTF data" % feature_name)
61
+ logging.info("Feature '%s' already exists in GTF data" % feature_name)
68
62
  continue
69
63
  logging.info("Creating rows for missing feature '%s'" % feature_name)
70
64
 
71
65
  # don't include rows where the groupby key was missing
72
- missing = pd.Series([
73
- x is None or x == ""
74
- for x in dataframe[groupby_key]])
66
+ missing = pd.Series([x is None or x == "" for x in dataframe[groupby_key]])
75
67
  not_missing = ~missing
76
68
  row_groups = dataframe[not_missing].groupby(groupby_key)
77
69
 
@@ -79,10 +71,9 @@ def create_missing_features(
79
71
  # other columns may or may not be uniquely defined. Start off by
80
72
  # assuming the values for every column are missing and fill them in
81
73
  # where possible.
82
- feature_values = OrderedDict([
83
- (column_name, [missing_value] * row_groups.ngroups)
84
- for column_name in dataframe.keys()
85
- ])
74
+ feature_values = OrderedDict(
75
+ [(column_name, [missing_value] * row_groups.ngroups) for column_name in dataframe]
76
+ )
86
77
 
87
78
  # User specifies which non-required columns should we try to infer
88
79
  # values for
@@ -111,8 +102,9 @@ def create_missing_features(
111
102
  for column_name in feature_columns:
112
103
  if column_name not in existing_columns:
113
104
  raise ValueError(
114
- "Column '%s' does not exist in GTF, columns = %s" % (
115
- column_name, existing_columns))
105
+ "Column '%s' does not exist in GTF, columns = %s"
106
+ % (column_name, existing_columns)
107
+ )
116
108
 
117
109
  # expect that all entries related to a reconstructed feature
118
110
  # are related and are thus within the same interval of
@@ -121,4 +113,4 @@ def create_missing_features(
121
113
  if len(unique_values) == 1:
122
114
  feature_values[column_name][i] = unique_values[0]
123
115
  extra_dataframes.append(pd.DataFrame(feature_values))
124
- return pd.concat([dataframe] + extra_dataframes, ignore_index=True)
116
+ return pd.concat([dataframe, *extra_dataframes], ignore_index=True)
@@ -10,5 +10,6 @@
10
10
  # See the License for the specific language governing permissions and
11
11
  # limitations under the License.
12
12
 
13
+
13
14
  class ParsingError(Exception):
14
15
  pass
@@ -13,12 +13,11 @@
13
13
  import logging
14
14
  from os.path import exists
15
15
 
16
- import polars
16
+ import polars
17
17
 
18
18
  from .attribute_parsing import expand_attribute_strings
19
19
  from .parsing_error import ParsingError
20
20
 
21
-
22
21
  logging.basicConfig(level=logging.INFO)
23
22
  logger = logging.getLogger(__name__)
24
23
 
@@ -76,88 +75,79 @@ REQUIRED_COLUMNS = [
76
75
 
77
76
 
78
77
  DEFAULT_COLUMN_DTYPES = {
79
- "seqname": polars.Categorical,
80
- "source": polars.Categorical,
81
-
78
+ "seqname": polars.Categorical,
79
+ "source": polars.Categorical,
82
80
  "start": polars.Int64,
83
81
  "end": polars.Int64,
84
82
  "score": polars.Float32,
85
-
86
- "feature": polars.Categorical,
87
- "strand": polars.Categorical,
83
+ "feature": polars.Categorical,
84
+ "strand": polars.Categorical,
88
85
  "frame": polars.UInt32,
89
86
  }
90
87
 
88
+
91
89
  def parse_with_polars_lazy(
92
- filepath_or_buffer,
93
- split_attributes=True,
94
- features=None,
95
- fix_quotes_columns=["attribute"]):
90
+ filepath_or_buffer, split_attributes=True, features=None, fix_quotes_columns=["attribute"]
91
+ ):
96
92
  # use a global string cache so that all strings get intern'd into
97
93
  # a single numbering system
98
94
  polars.enable_string_cache()
99
- kwargs = dict(
100
- has_header=False,
101
- separator="\t",
102
- comment_prefix="#",
103
- null_values=".",
104
- schema_overrides=DEFAULT_COLUMN_DTYPES)
95
+ kwargs = {
96
+ "has_header": False,
97
+ "separator": "\t",
98
+ "comment_prefix": "#",
99
+ "null_values": ".",
100
+ "schema_overrides": DEFAULT_COLUMN_DTYPES,
101
+ }
105
102
  try:
106
- df = polars.read_csv(
107
- filepath_or_buffer,
108
- new_columns=REQUIRED_COLUMNS,
109
- **kwargs).lazy()
110
- except polars.exceptions.ShapeError:
111
- raise ParsingError("Wrong number of columns")
103
+ df = polars.read_csv(filepath_or_buffer, new_columns=REQUIRED_COLUMNS, **kwargs).lazy()
104
+ except polars.exceptions.ShapeError as err:
105
+ raise ParsingError("Wrong number of columns") from err
112
106
 
113
107
  # Drop empty lines that may appear as all-null rows
114
108
  df = df.filter(polars.col("seqname").is_not_null())
115
109
 
116
- df = df.with_columns([
117
- polars.col("frame").fill_null(0),
118
- polars.col("attribute").str.replace_all('"', "'")
119
- ])
120
-
110
+ df = df.with_columns(
111
+ [polars.col("frame").fill_null(0), polars.col("attribute").str.replace_all('"', "'")]
112
+ )
113
+
121
114
  for fix_quotes_column in fix_quotes_columns:
122
115
  # Catch mistaken semicolons by replacing "xyz;" with "xyz"
123
116
  # Required to do this since the Ensembl GTF for Ensembl
124
117
  # release 78 has mistakes such as:
125
118
  # gene_name = "PRAMEF6;" transcript_name = "PRAMEF6;-201"
126
- df = df.with_columns([
127
- polars.col(fix_quotes_column).str.replace(';\"', '\"').str.replace(";-", "-")
128
- ])
119
+ df = df.with_columns(
120
+ [polars.col(fix_quotes_column).str.replace(';"', '"').str.replace(";-", "-")]
121
+ )
129
122
 
130
123
  if features is not None:
131
124
  features = sorted(set(features))
132
125
  df = df.filter(polars.col("feature").is_in(features))
133
126
 
134
-
135
127
  if split_attributes:
136
- df = df.with_columns([
137
- polars.col("attribute").str.split(";").alias("attribute_split")
138
- ])
128
+ df = df.with_columns([polars.col("attribute").str.split(";").alias("attribute_split")])
139
129
  return df
140
130
 
131
+
141
132
  def parse_gtf(
142
- filepath_or_buffer,
143
- split_attributes=True,
144
- features=None,
145
- fix_quotes_columns=["attribute"]):
133
+ filepath_or_buffer, split_attributes=True, features=None, fix_quotes_columns=["attribute"]
134
+ ):
146
135
  df_lazy = parse_with_polars_lazy(
147
136
  filepath_or_buffer=filepath_or_buffer,
148
137
  split_attributes=split_attributes,
149
138
  features=features,
150
- fix_quotes_columns=fix_quotes_columns)
139
+ fix_quotes_columns=fix_quotes_columns,
140
+ )
151
141
  return df_lazy.collect()
152
142
 
143
+
153
144
  def parse_gtf_pandas(*args, **kwargs):
154
145
  return parse_gtf(*args, **kwargs).to_pandas()
155
146
 
156
-
147
+
157
148
  def parse_gtf_and_expand_attributes(
158
- filepath_or_buffer,
159
- restrict_attribute_columns=None,
160
- features=None):
149
+ filepath_or_buffer, restrict_attribute_columns=None, features=None
150
+ ):
161
151
  """
162
152
  Parse lines into column->values dictionary and then expand
163
153
  the 'attribute' column into multiple columns. This expansion happens
@@ -177,33 +167,32 @@ def parse_gtf_and_expand_attributes(
177
167
  features : set or None
178
168
  Ignore entries which don't correspond to one of the supplied features
179
169
  """
180
- df = parse_gtf(
181
- filepath_or_buffer=filepath_or_buffer,
182
- features=features,
183
- split_attributes=True)
170
+ df = parse_gtf(filepath_or_buffer=filepath_or_buffer, features=features, split_attributes=True)
184
171
  if type(restrict_attribute_columns) is str:
185
172
  restrict_attribute_columns = {restrict_attribute_columns}
186
173
  elif restrict_attribute_columns:
187
174
  restrict_attribute_columns = set(restrict_attribute_columns)
188
175
  df.drop_in_place("attribute")
189
176
  attribute_pairs = df.drop_in_place("attribute_split")
190
- return df.with_columns([
191
- polars.Series(k, vs)
192
- for (k, vs) in
193
- expand_attribute_strings(attribute_pairs).items()
194
- if restrict_attribute_columns is None or k in restrict_attribute_columns
195
- ])
196
-
177
+ return df.with_columns(
178
+ [
179
+ polars.Series(k, vs)
180
+ for (k, vs) in expand_attribute_strings(attribute_pairs).items()
181
+ if restrict_attribute_columns is None or k in restrict_attribute_columns
182
+ ]
183
+ )
184
+
197
185
 
198
186
  def read_gtf(
199
- filepath_or_buffer,
200
- expand_attribute_column=True,
201
- infer_biotype_column=False,
202
- column_converters={},
203
- column_cast_types={},
204
- usecols=None,
205
- features=None,
206
- result_type='polars'):
187
+ filepath_or_buffer,
188
+ expand_attribute_column=True,
189
+ infer_biotype_column=False,
190
+ column_converters={},
191
+ column_cast_types={},
192
+ usecols=None,
193
+ features=None,
194
+ result_type="polars",
195
+ ):
207
196
  """
208
197
  Parse a GTF into a dictionary mapping column names to sequences of values.
209
198
 
@@ -231,7 +220,7 @@ def read_gtf(
231
220
  column_cast_types : dict, optional
232
221
  Dictionary mapping column names to dtypes. Will cast columns to given
233
222
  Polars types.
234
-
223
+
235
224
  usecols : list of str or None
236
225
  Restrict which columns are loaded to the give set. If None, then
237
226
  load all columns.
@@ -240,7 +229,7 @@ def read_gtf(
240
229
  Drop rows which aren't one of the features in the supplied set
241
230
 
242
231
  result_type : One of 'polars', 'pandas', or 'dict'
243
- Default behavior is to return a Polars DataFrame, but will convert to
232
+ Default behavior is to return a Polars DataFrame, but will convert to
244
233
  Pandas DataFrame or dictionary if specified.
245
234
  """
246
235
  if type(filepath_or_buffer) is str and not exists(filepath_or_buffer):
@@ -248,9 +237,8 @@ def read_gtf(
248
237
 
249
238
  if expand_attribute_column:
250
239
  result_df = parse_gtf_and_expand_attributes(
251
- filepath_or_buffer,
252
- restrict_attribute_columns=usecols,
253
- features=features)
240
+ filepath_or_buffer, restrict_attribute_columns=usecols, features=features
241
+ )
254
242
  else:
255
243
  result_df = parse_gtf(result_df, features=features)
256
244
 
@@ -259,26 +247,26 @@ def read_gtf(
259
247
  # and are generally insane to chase down
260
248
  result_df = result_df.to_pandas()
261
249
  if column_converters or column_cast_types:
250
+
262
251
  def wrap_to_always_accept_none(f):
263
252
  def wrapped_fn(x):
264
253
  if x is None or x == "":
265
254
  return None
266
255
  else:
267
256
  return f(x)
257
+
268
258
  return wrapped_fn
269
-
259
+
270
260
  column_names = set(column_converters.keys()).union(column_cast_types.keys())
271
261
  for column_name in column_names:
272
-
273
262
  if column_name in column_converters:
274
- column_fn = wrap_to_always_accept_none(
275
- column_converters[column_name])
263
+ column_fn = wrap_to_always_accept_none(column_converters[column_name])
276
264
  result_df[column_name] = result_df[column_name].apply(column_fn)
277
265
 
278
266
  if column_name in column_cast_types:
279
267
  column_type = column_cast_types[column_name]
280
268
  result_df[column_name] = result_df[column_name].astype(column_type)
281
-
269
+
282
270
  # Hackishly infer whether the values in the 'source' column of this GTF
283
271
  # are actually representing a biotype by checking for the most common
284
272
  # gene_biotype and transcript_biotype value 'protein_coding'
@@ -292,11 +280,11 @@ def read_gtf(
292
280
  # gene_biotype)
293
281
  if "gene_biotype" not in column_names:
294
282
  logging.info("Using column 'source' to replace missing 'gene_biotype'")
295
- result_df['gene_biotype'] = result_df['source']
283
+ result_df["gene_biotype"] = result_df["source"]
296
284
  if "transcript_biotype" not in column_names:
297
285
  logging.info("Using column 'source' to replace missing 'transcript_biotype'")
298
- result_df['transcript_biotype'] = result_df['source']
299
-
286
+ result_df["transcript_biotype"] = result_df["source"]
287
+
300
288
  if usecols is not None:
301
289
  column_names = set(result_df.columns)
302
290
  valid_columns = [c for c in usecols if c in column_names]
@@ -1,23 +1,33 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gtfparse
3
- Version: 2.6.0
3
+ Version: 2.6.3
4
4
  Summary: Parsing library for extracting data frames of genomic features from GTF files
5
5
  Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
6
6
  Project-URL: Homepage, https://github.com/openvax/gtfparse
7
- Project-URL: Bug Tracker, https://github.com/openvax/gtfparse
7
+ Project-URL: Bug Tracker, https://github.com/openvax/gtfparse/issues
8
8
  Classifier: Development Status :: 4 - Beta
9
9
  Classifier: Environment :: Console
10
10
  Classifier: Operating System :: OS Independent
11
11
  Classifier: Intended Audience :: Science/Research
12
12
  Classifier: License :: OSI Approved :: Apache Software License
13
13
  Classifier: Programming Language :: Python
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.9
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
14
19
  Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
15
- Requires-Python: >=3.7
20
+ Requires-Python: >=3.9
16
21
  Description-Content-Type: text/markdown
17
22
  License-File: LICENSE
18
23
  Requires-Dist: polars>=0.20.2
19
24
  Requires-Dist: pyarrow>=18.0.0
20
25
  Requires-Dist: pandas>=2.1.0
26
+ Provides-Extra: dev
27
+ Requires-Dist: pytest; extra == "dev"
28
+ Requires-Dist: pytest-cov; extra == "dev"
29
+ Requires-Dist: ruff; extra == "dev"
30
+ Requires-Dist: coveralls; extra == "dev"
21
31
  Dynamic: license-file
22
32
 
23
33
  [![Tests](https://github.com/openvax/gtfparse/actions/workflows/tests.yml/badge.svg)](https://github.com/openvax/gtfparse/actions/workflows/tests.yml)
@@ -1,3 +1,9 @@
1
1
  polars>=0.20.2
2
2
  pyarrow>=18.0.0
3
3
  pandas>=2.1.0
4
+
5
+ [dev]
6
+ pytest
7
+ pytest-cov
8
+ ruff
9
+ coveralls
@@ -0,0 +1,104 @@
1
+ [build-system]
2
+ requires = ["setuptools>=61.0", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "gtfparse"
7
+ requires-python = ">=3.9"
8
+ authors = [ {name="Alex Rubinsteyn", email="alex.rubinsteyn@unc.edu" } ]
9
+ description = "Parsing library for extracting data frames of genomic features from GTF files"
10
+ classifiers = [
11
+ "Development Status :: 4 - Beta",
12
+ "Environment :: Console",
13
+ "Operating System :: OS Independent",
14
+ "Intended Audience :: Science/Research",
15
+ "License :: OSI Approved :: Apache Software License",
16
+ "Programming Language :: Python",
17
+ "Programming Language :: Python :: 3",
18
+ "Programming Language :: Python :: 3.9",
19
+ "Programming Language :: Python :: 3.10",
20
+ "Programming Language :: Python :: 3.11",
21
+ "Programming Language :: Python :: 3.12",
22
+ "Topic :: Scientific/Engineering :: Bio-Informatics",
23
+ ]
24
+ readme = "README.md"
25
+ dynamic = ["version", "dependencies"]
26
+
27
+ [project.optional-dependencies]
28
+ dev = [
29
+ "pytest",
30
+ "pytest-cov",
31
+ "ruff",
32
+ "coveralls",
33
+ ]
34
+
35
+ [project.urls]
36
+ "Homepage" = "https://github.com/openvax/gtfparse"
37
+ "Bug Tracker" = "https://github.com/openvax/gtfparse/issues"
38
+
39
+ [tool.setuptools.dynamic]
40
+ version = {attr = "gtfparse.__version__"}
41
+ dependencies = {file = ["requirements.txt"]}
42
+
43
+ [tool.setuptools]
44
+ packages = ["gtfparse"]
45
+
46
+ [tool.ruff]
47
+ target-version = "py39"
48
+ line-length = 100
49
+ src = ["gtfparse", "tests"]
50
+ exclude = [
51
+ ".git",
52
+ ".venv",
53
+ "__pycache__",
54
+ "build",
55
+ "dist",
56
+ "*.egg-info",
57
+ ".eggs",
58
+ ]
59
+
60
+ [tool.ruff.lint]
61
+ select = [
62
+ "E", # pycodestyle errors
63
+ "W", # pycodestyle warnings
64
+ "F", # Pyflakes
65
+ "I", # isort
66
+ "B", # flake8-bugbear
67
+ "C4", # flake8-comprehensions
68
+ "UP", # pyupgrade
69
+ "SIM", # flake8-simplify
70
+ "RUF", # Ruff-specific rules
71
+ ]
72
+ ignore = [
73
+ "E501", # line too long (handled by formatter)
74
+ "E741", # ambiguous variable name (pre-existing in codebase)
75
+ "B006", # mutable default args (pre-existing; behavior-risky to change)
76
+ "B008", # do not perform function calls in argument defaults
77
+ "B905", # zip() without explicit strict
78
+ "SIM108", # use ternary operator instead of if-else
79
+ "UP007", # use X | Y for type unions (need Python 3.10+)
80
+ "UP031", # %-format strings (pre-existing; out of scope for config PR)
81
+ ]
82
+
83
+ [tool.ruff.lint.per-file-ignores]
84
+ "__init__.py" = ["F401"]
85
+ "tests/*" = ["F401", "B011"]
86
+
87
+ [tool.ruff.lint.isort]
88
+ known-first-party = ["gtfparse"]
89
+
90
+ [tool.pytest.ini_options]
91
+ testpaths = ["tests"]
92
+ python_files = "test_*.py"
93
+ python_functions = "test_*"
94
+ addopts = "-v --tb=short"
95
+
96
+ [tool.coverage.run]
97
+ source = ["gtfparse"]
98
+
99
+ [tool.coverage.report]
100
+ exclude_lines = [
101
+ "pragma: no cover",
102
+ "if __name__ == .__main__.:",
103
+ "raise NotImplementedError",
104
+ ]
@@ -0,0 +1,85 @@
1
+ from io import StringIO
2
+
3
+ from gtfparse import create_missing_features, parse_gtf_and_expand_attributes
4
+
5
+ # two lines from the Ensembl 54 human GTF containing only a stop_codon and
6
+ # exon features, but from which gene and transcript information could be
7
+ # inferred
8
+ GTF_TEXT = "\n".join(
9
+ [
10
+ "# seqname biotype feature start end score strand frame attribute",
11
+ "".join(
12
+ [
13
+ """18\tprotein_coding\tstop_codon\t32630766\t32630768\t.\t-\t0\t""",
14
+ """gene_id "ENSG00000134779"; transcript_id "ENST00000334295"; exon_number "7";"""
15
+ """ gene_name "C18orf10";""",
16
+ """ transcript_name "C18orf10-201";""",
17
+ ]
18
+ ),
19
+ "".join(
20
+ [
21
+ """18\tprotein_coding\texon\t32663078\t32663157\t.\t+\t.\tgene_id "ENSG00000150477"; """,
22
+ """transcript_id "ENST00000383055"; exon_number "1"; gene_name "KIAA1328"; """,
23
+ """transcript_name "KIAA1328-202";""",
24
+ ]
25
+ ),
26
+ ]
27
+ )
28
+
29
+
30
+ GTF_DATAFRAME = parse_gtf_and_expand_attributes(StringIO(GTF_TEXT))
31
+ GTF_DATAFRAME = GTF_DATAFRAME.to_pandas()
32
+
33
+
34
+ def test_create_missing_features_identity():
35
+ df_should_be_same = create_missing_features(GTF_DATAFRAME, {})
36
+ assert len(GTF_DATAFRAME) == len(df_should_be_same), "GTF DataFrames should be same size"
37
+
38
+
39
+ def _check_expanded_dataframe(df):
40
+ assert "gene" in set(df["feature"]), "Extended GTF should contain gene feature"
41
+ assert "transcript" in set(df["feature"]), "Extended GTF should contain transcript feature"
42
+
43
+ C18orf10_201_transcript_mask = (df["feature"] == "transcript") & (
44
+ df["transcript_name"] == "C18orf10-201"
45
+ )
46
+ assert len(df[C18orf10_201_transcript_mask]) == 1, (
47
+ "Expected only 1 gene entry for C18orf10-201, got %s" % (df[C18orf10_201_transcript_mask],)
48
+ )
49
+ transcript_seqname = df[C18orf10_201_transcript_mask].seqname.iloc[0]
50
+ assert transcript_seqname == "18", "Wrong seqname for C18orf10-201: %s" % transcript_seqname
51
+ transcript_start = df[C18orf10_201_transcript_mask].start.iloc[0]
52
+ assert transcript_start == 32630766, "Wrong start for C18orf10-201: %s" % transcript_start
53
+ transcript_end = df[C18orf10_201_transcript_mask].end.iloc[0]
54
+ assert transcript_end == 32630768, "Wrong end for C18orf10-201: %s" % transcript_end
55
+ transcript_strand = df[C18orf10_201_transcript_mask].strand.iloc[0]
56
+ assert transcript_strand == "-", "Wrong strand for C18orf10-201: %s" % transcript_strand
57
+
58
+ KIAA1328_gene_mask = (df["feature"] == "gene") & (df["gene_name"] == "KIAA1328")
59
+ assert len(df[KIAA1328_gene_mask]) == 1, "Expected only 1 gene entry for KIAA1328"
60
+ gene_seqname = df[KIAA1328_gene_mask].seqname.iloc[0]
61
+ assert gene_seqname == "18", "Wrong seqname for KIAA1328: %s" % gene_seqname
62
+ gene_start = df[KIAA1328_gene_mask].start.iloc[0]
63
+ assert gene_start == 32663078, "Wrong start for KIAA1328: %s" % (gene_start,)
64
+ gene_end = df[KIAA1328_gene_mask].end.iloc[0]
65
+ assert gene_end == 32663157, "Wrong end for KIAA1328: %s" % (gene_end,)
66
+ gene_strand = df[KIAA1328_gene_mask].strand.iloc[0]
67
+ assert gene_strand == "+", "Wrong strand for KIAA1328: %s" % gene_strand
68
+
69
+
70
+ def test_create_missing_features():
71
+ assert "gene" not in set(GTF_DATAFRAME["feature"]), (
72
+ "Original GTF should not contain gene feature"
73
+ )
74
+ assert "transcript" not in set(GTF_DATAFRAME["feature"]), (
75
+ "Original GTF should not contain transcript feature"
76
+ )
77
+ df_extra_features = create_missing_features(
78
+ GTF_DATAFRAME,
79
+ unique_keys={"gene": "gene_id", "transcript": "transcript_id"},
80
+ extra_columns={
81
+ "gene": {"gene_name"},
82
+ "transcript": {"gene_id", "gene_name", "transcript_name"},
83
+ },
84
+ )
85
+ _check_expanded_dataframe(df_extra_features)
@@ -0,0 +1,147 @@
1
+ from gtfparse import read_gtf
2
+
3
+ from .data import data_path
4
+
5
+ ENSEMBL_GTF_PATH = data_path("ensembl_grch37.head.gtf")
6
+
7
+ EXPECTED_FEATURES = {
8
+ "gene",
9
+ "transcript",
10
+ "exon",
11
+ "CDS",
12
+ "UTR",
13
+ "start_codon",
14
+ "stop_codon",
15
+ }
16
+
17
+
18
+ def test_ensembl_gtf_columns():
19
+ df = read_gtf(ENSEMBL_GTF_PATH)
20
+ features = set(df["feature"])
21
+ assert features == EXPECTED_FEATURES
22
+
23
+
24
+ # first 1000 lines of GTF only contained these genes
25
+ EXPECTED_GENE_NAMES = {
26
+ "FAM41C",
27
+ "CICP27",
28
+ "RNU6-1100P",
29
+ "NOC2L",
30
+ "AP006222.1",
31
+ "LINC01128",
32
+ "RP4-669L17.1",
33
+ "RP11-206L10.2",
34
+ "PLEKHN1",
35
+ "WBP1LP7",
36
+ "RP5-857K21.1",
37
+ "RP5-857K21.5",
38
+ "RNU6-1199P",
39
+ "RP11-206L10.10",
40
+ "RP11-54O7.16",
41
+ "CICP7",
42
+ "AL627309.1",
43
+ "RP5-857K21.11",
44
+ "DDX11L1",
45
+ "RP5-857K21.3",
46
+ "RP11-34P13.7",
47
+ "AL669831.1",
48
+ "MTATP6P1",
49
+ "CICP3",
50
+ "WBP1LP6",
51
+ "LINC00115",
52
+ "hsa-mir-6723",
53
+ "RP5-857K21.7",
54
+ "SAMD11",
55
+ "RP11-206L10.5",
56
+ "RP11-34P13.8",
57
+ "RP11-206L10.9",
58
+ "RP11-34P13.15",
59
+ "TUBB8P11",
60
+ "MTATP8P1",
61
+ "RP4-669L17.8",
62
+ "RP11-206L10.1",
63
+ "RP11-34P13.13",
64
+ "RP11-206L10.3",
65
+ "RP11-206L10.4",
66
+ "RP11-54O7.3",
67
+ "RP5-857K21.2",
68
+ "OR4F5",
69
+ "MTND1P23",
70
+ "AL645608.1",
71
+ "RP11-34P13.16",
72
+ "RP11-34P13.14",
73
+ "AP006222.2",
74
+ "OR4F29",
75
+ "RP4-669L17.4",
76
+ "AL732372.1",
77
+ "OR4G4P",
78
+ "MTND2P28",
79
+ "OR4F16",
80
+ "KLHL17",
81
+ "FAM138A",
82
+ "OR4G11P",
83
+ "FAM87B",
84
+ "RP5-857K21.15",
85
+ "AL645608.2",
86
+ "RP11-206L10.8",
87
+ "RP5-857K21.4",
88
+ "MIR1302-10",
89
+ "RP11-54O7.2",
90
+ "RP4-669L17.10",
91
+ "RP11-54O7.1",
92
+ "RP11-34P13.9",
93
+ "WASH7P",
94
+ "RP4-669L17.2",
95
+ }
96
+
97
+
98
+ def test_ensembl_gtf_gene_names():
99
+ df = read_gtf(ENSEMBL_GTF_PATH)
100
+ gene_names = set(df["gene_name"])
101
+ assert gene_names == EXPECTED_GENE_NAMES, (
102
+ "Wrong gene names: %s, missing %s and unexpected %s"
103
+ % (
104
+ gene_names,
105
+ EXPECTED_GENE_NAMES.difference(gene_names),
106
+ gene_names.difference(EXPECTED_GENE_NAMES),
107
+ )
108
+ )
109
+
110
+
111
+ def test_ensembl_gtf_gene_names_with_usecols():
112
+ df = read_gtf(ENSEMBL_GTF_PATH, usecols=["gene_name"])
113
+ gene_names = set(df["gene_name"])
114
+ assert gene_names == EXPECTED_GENE_NAMES, (
115
+ "Wrong gene names: %s, missing %s and unexpected %s"
116
+ % (
117
+ gene_names,
118
+ EXPECTED_GENE_NAMES.difference(gene_names),
119
+ gene_names.difference(EXPECTED_GENE_NAMES),
120
+ )
121
+ )
122
+
123
+
124
+ def test_ensembl_gtf_gene_names_zip():
125
+ df = read_gtf(ENSEMBL_GTF_PATH + ".gz")
126
+ gene_names = set(df["gene_name"])
127
+ assert gene_names == EXPECTED_GENE_NAMES, (
128
+ "Wrong gene names: %s, missing %s and unexpected %s"
129
+ % (
130
+ gene_names,
131
+ EXPECTED_GENE_NAMES.difference(gene_names),
132
+ gene_names.difference(EXPECTED_GENE_NAMES),
133
+ )
134
+ )
135
+
136
+
137
+ def test_ensembl_gtf_gene_names_with_usecols_gzip():
138
+ df = read_gtf(ENSEMBL_GTF_PATH + ".gz", usecols=["gene_name"])
139
+ gene_names = set(df["gene_name"])
140
+ assert gene_names == EXPECTED_GENE_NAMES, (
141
+ "Wrong gene names: %s, missing %s and unexpected %s"
142
+ % (
143
+ gene_names,
144
+ EXPECTED_GENE_NAMES.difference(gene_names),
145
+ gene_names.difference(EXPECTED_GENE_NAMES),
146
+ )
147
+ )
@@ -1,12 +1,13 @@
1
1
  from gtfparse import expand_attribute_strings
2
2
 
3
+
3
4
  def test_attributes_in_quotes():
4
5
  attributes = [
5
- "gene_id \"ENSG001\"; tag \"bogotron\"; version \"1\";",
6
- "gene_id \"ENSG002\"; tag \"wolfpuppy\"; version \"2\";"
6
+ 'gene_id "ENSG001"; tag "bogotron"; version "1";',
7
+ 'gene_id "ENSG002"; tag "wolfpuppy"; version "2";',
7
8
  ]
8
9
  parsed_dict = expand_attribute_strings(attributes, quote_char='"')
9
- assert list(sorted(parsed_dict.keys())), ["gene_id", "tag", "version"]
10
+ assert sorted(parsed_dict.keys()), ["gene_id", "tag", "version"]
10
11
  assert parsed_dict["gene_id"] == ["ENSG001", "ENSG002"]
11
12
  assert parsed_dict["tag"] == ["bogotron", "wolfpuppy"]
12
13
  assert parsed_dict["version"] == ["1", "2"]
@@ -15,10 +16,10 @@ def test_attributes_in_quotes():
15
16
  def test_attributes_without_quotes():
16
17
  attributes = [
17
18
  "gene_id ENSG001; tag bogotron; version 1;",
18
- "gene_id ENSG002; tag wolfpuppy; version 2"
19
+ "gene_id ENSG002; tag wolfpuppy; version 2",
19
20
  ]
20
21
  parsed_dict = expand_attribute_strings(attributes)
21
- assert list(sorted(parsed_dict.keys())) == ["gene_id", "tag", "version"]
22
+ assert sorted(parsed_dict.keys()) == ["gene_id", "tag", "version"]
22
23
  assert parsed_dict["gene_id"] == ["ENSG001", "ENSG002"]
23
24
  assert parsed_dict["tag"] == ["bogotron", "wolfpuppy"]
24
25
  assert parsed_dict["version"] == ["1", "2"]
@@ -31,6 +32,6 @@ def test_optional_attributes():
31
32
  "gene_id ENSG003; sometimes-present wolfpuppy;",
32
33
  ]
33
34
  parsed_dict = expand_attribute_strings(attributes)
34
- assert list(sorted(parsed_dict.keys())) == ["gene_id", "sometimes-present"]
35
- assert parsed_dict["gene_id"] == ["ENSG001", "ENSG002", "ENSG003"]
36
- assert parsed_dict["sometimes-present"] == ["bogotron", "", "wolfpuppy"]
35
+ assert sorted(parsed_dict.keys()) == ["gene_id", "sometimes-present"]
36
+ assert parsed_dict["gene_id"] == ["ENSG001", "ENSG002", "ENSG003"]
37
+ assert parsed_dict["sometimes-present"] == ["bogotron", "", "wolfpuppy"]
@@ -1,4 +1,5 @@
1
1
  from io import StringIO
2
+
2
3
  from gtfparse import parse_gtf_and_expand_attributes
3
4
 
4
5
  # failing example from https://github.com/openvax/gtfparse/issues/2
@@ -11,25 +12,26 @@ GTF_TEXT = (
11
12
  """tag "cds_end_NF"; tag "mRNA_end_NF"; """
12
13
  )
13
14
 
15
+
14
16
  def test_parse_tag_attributes():
15
17
  parsed = parse_gtf_and_expand_attributes(StringIO(GTF_TEXT))
16
18
  tag_column = parsed["tag"]
17
19
  assert len(tag_column) == 1
18
20
  tags = tag_column[0]
19
- assert tags == 'cds_end_NF,mRNA_end_NF'
21
+ assert tags == "cds_end_NF,mRNA_end_NF"
22
+
20
23
 
21
24
  def test_parse_tag_attributes_with_usecols():
22
- parsed = parse_gtf_and_expand_attributes(
23
- StringIO(GTF_TEXT),
24
- restrict_attribute_columns=["tag"])
25
+ parsed = parse_gtf_and_expand_attributes(StringIO(GTF_TEXT), restrict_attribute_columns=["tag"])
25
26
  tag_column = parsed["tag"]
26
27
  assert len(tag_column) == 1
27
28
  tags = tag_column[0]
28
- assert tags == 'cds_end_NF,mRNA_end_NF'
29
+ assert tags == "cds_end_NF,mRNA_end_NF"
30
+
29
31
 
30
32
  def test_parse_tag_attributes_with_usecols_other_column():
31
33
  parsed = parse_gtf_and_expand_attributes(
32
- StringIO(GTF_TEXT),
33
- restrict_attribute_columns=["exon_id"])
34
+ StringIO(GTF_TEXT), restrict_attribute_columns=["exon_id"]
35
+ )
34
36
 
35
37
  assert "tag" not in parsed, "Expected 'tag' to get dropped but got %s" % (parsed,)
@@ -1,11 +1,8 @@
1
+ from io import StringIO
2
+
1
3
  from pytest import raises
2
- from gtfparse import (
3
- parse_gtf,
4
- parse_gtf_and_expand_attributes,
5
- REQUIRED_COLUMNS,
6
- ParsingError
7
- )
8
- from io import StringIO
4
+
5
+ from gtfparse import REQUIRED_COLUMNS, ParsingError, parse_gtf, parse_gtf_and_expand_attributes
9
6
 
10
7
  gtf_text = """
11
8
  # sample GTF data copied from:
@@ -14,12 +11,13 @@ gtf_text = """
14
11
  1\tprocessed_transcript\ttranscript\t11869\t14409\t.\t+\t.\tgene_id "ENSG00000223972"; transcript_id "ENST00000456328"; gene_name "DDX11L1"; gene_source "havana"; gene_biotype "transcribed_unprocessed_pseudogene"; transcript_name "DDX11L1-002"; transcript_source "havana";
15
12
  """
16
13
 
14
+
17
15
  def test_parse_gtf_lines_with_expand_attributes():
18
16
  df = parse_gtf_and_expand_attributes(StringIO(gtf_text))
19
17
 
20
-
21
18
  # excluding 'attribute' column from required names
22
- expected_columns = REQUIRED_COLUMNS[:8] + [
19
+ expected_columns = [
20
+ *REQUIRED_COLUMNS[:8],
23
21
  "gene_id",
24
22
  "gene_name",
25
23
  "gene_source",
@@ -29,7 +27,7 @@ def test_parse_gtf_lines_with_expand_attributes():
29
27
  "transcript_source",
30
28
  ]
31
29
  # convert to list since Py3's dictionary keys are a distinct collection type
32
- assert list(df.columns) == expected_columns
30
+ assert list(df.columns) == expected_columns
33
31
  assert list(df["seqname"]) == ["1", "1"]
34
32
  # convert to list for comparison since numerical columns may be NumPy arrays
35
33
  assert list(df["start"]) == [11869, 11869]
@@ -52,6 +50,7 @@ def test_parse_gtf_lines_without_expand_attributes():
52
50
  assert df["score"].is_null().all(), "Unexpected scores: %s" % (df["score"],)
53
51
  assert len(df["attribute"]) == 2
54
52
 
53
+
55
54
  def test_parse_gtf_lines_error_too_few_fields():
56
55
  bad_gtf_text = gtf_text.replace("\t", " ")
57
56
  # pylint: disable=no-value-for-parameter
@@ -1,8 +1,10 @@
1
1
  from gtfparse import read_gtf
2
+
2
3
  from .data import data_path
3
4
 
4
5
  B16_GTF_PATH = data_path("B16.stringtie.head.gtf")
5
6
 
7
+
6
8
  def _check_required_columns(gtf_dict):
7
9
  assert "feature" in gtf_dict, "Expected column named 'feature' in StringTie GTF"
8
10
  assert "cov" in gtf_dict, "Expected column named 'cov' in StringTie GTF"
@@ -11,32 +13,37 @@ def _check_required_columns(gtf_dict):
11
13
  assert "exon" in features, "No exons in GTF (available: %s)" % features
12
14
  assert "transcript" in features, "No transcripts in GTF (available: %s)" % features
13
15
 
16
+
14
17
  def _check_string_cov_and_FPKM(gtf_dict):
15
18
  for i, feature_name in enumerate(gtf_dict["feature"]):
16
19
  cov = gtf_dict["cov"][i]
17
20
  fpkm = gtf_dict["FPKM"][i]
18
21
  if feature_name == "exon":
19
- assert len(fpkm) == 0, \
20
- "Expected missing FPKM for exon, got %s" % (fpkm,)
21
- assert len(cov) > 0 and float(cov) >= 0, \
22
+ assert len(fpkm) == 0, "Expected missing FPKM for exon, got %s" % (fpkm,)
23
+ assert len(cov) > 0 and float(cov) >= 0, (
22
24
  "Expected non-negative cov for exon, got %s" % (cov,)
25
+ )
23
26
  elif feature_name == "transcript":
24
- assert len(cov) and float(cov) >= 0, \
27
+ assert len(cov) and float(cov) >= 0, (
25
28
  "Expected non-negative cov for transcript, got %s" % (cov,)
26
- assert len(fpkm) > 0 and float(fpkm) >= 0, \
29
+ )
30
+ assert len(fpkm) > 0 and float(fpkm) >= 0, (
27
31
  "Expected non-negative FPKM for transcript, got %s" % (fpkm,)
32
+ )
33
+
28
34
 
29
35
  def _check_float_cov_and_FPKM(gtf_dict):
30
36
  for i, feature_name in enumerate(gtf_dict["feature"]):
31
37
  cov = gtf_dict["cov"][i]
32
38
  fpkm = gtf_dict["FPKM"][i]
33
- assert isinstance(cov, float), \
34
- "Expected cov to be float but got %s : %s" % (cov, type(cov))
39
+ assert isinstance(cov, float), "Expected cov to be float but got %s : %s" % (cov, type(cov))
35
40
  if feature_name == "exon":
36
41
  assert cov >= 0, "Expected non-negative cov for exon, got %s" % (cov,)
37
42
  elif feature_name == "transcript":
38
- assert isinstance(fpkm, float), \
39
- "Expected FPKM to be float but got %s : %s" % (fpkm, type(fpkm))
43
+ assert isinstance(fpkm, float), "Expected FPKM to be float but got %s : %s" % (
44
+ fpkm,
45
+ type(fpkm),
46
+ )
40
47
  assert cov >= 0, "Expected non-negative cov for transcript, got %s" % (cov,)
41
48
  assert fpkm >= 0, "Expected non-negative FPKM for transcript, got %s" % (fpkm,)
42
49
 
@@ -46,9 +53,8 @@ def test_read_stringtie_gtf_as_dataframe():
46
53
  _check_required_columns(gtf_df)
47
54
  _check_string_cov_and_FPKM(gtf_df)
48
55
 
56
+
49
57
  def test_read_stringtie_gtf_as_dataframe_float_values():
50
- gtf_df = read_gtf(
51
- B16_GTF_PATH,
52
- column_converters={"cov": float, "FPKM": float})
58
+ gtf_df = read_gtf(B16_GTF_PATH, column_converters={"cov": float, "FPKM": float})
53
59
  _check_required_columns(gtf_df)
54
60
  _check_float_cov_and_FPKM(gtf_df)
@@ -1,8 +1,10 @@
1
1
  from gtfparse import read_gtf
2
+
2
3
  from .data import data_path
3
4
 
4
5
  REFSEQ_GTF_PATH = data_path("refseq.ucsc.small.gtf")
5
6
 
7
+
6
8
  def _check_required_columns(gtf_dict):
7
9
  assert "feature" in gtf_dict, "Expected column named 'feature' in RefSeq GTF"
8
10
  assert "gene_id" in gtf_dict, "Expected column named 'gene_id' in RefSeq GTF"
@@ -11,10 +13,16 @@ def _check_required_columns(gtf_dict):
11
13
  assert "exon" in features, "No exon features in GTF (available: %s)" % features
12
14
  assert "CDS" in features, "No CDS features in GTF (available: %s)" % features
13
15
 
16
+
14
17
  def test_read_refseq_gtf_as_dataframe():
15
18
  gtf_df = read_gtf(REFSEQ_GTF_PATH)
16
19
  _check_required_columns(gtf_df)
17
20
 
21
+
18
22
  def test_read_refseq_and_transform_columns():
19
- gtf_df = read_gtf(REFSEQ_GTF_PATH, column_converters={"start": int, "end": int}, column_cast_types={"score": float})
20
- print(gtf_df)
23
+ gtf_df = read_gtf(
24
+ REFSEQ_GTF_PATH,
25
+ column_converters={"start": int, "end": int},
26
+ column_cast_types={"score": float},
27
+ )
28
+ print(gtf_df)
@@ -1,27 +0,0 @@
1
- [project]
2
- name = "gtfparse"
3
- requires-python = ">=3.7"
4
- authors = [ {name="Alex Rubinsteyn", email="alex.rubinsteyn@unc.edu" } ]
5
- description = "Parsing library for extracting data frames of genomic features from GTF files"
6
- classifiers = [
7
- 'Development Status :: 4 - Beta',
8
- 'Environment :: Console',
9
- 'Operating System :: OS Independent',
10
- 'Intended Audience :: Science/Research',
11
- 'License :: OSI Approved :: Apache Software License',
12
- 'Programming Language :: Python',
13
- 'Topic :: Scientific/Engineering :: Bio-Informatics',
14
- ]
15
- readme = "README.md"
16
- dynamic = ["version", "dependencies"]
17
-
18
- [tool.setuptools.dynamic]
19
- version = {attr = "gtfparse.__version__"}
20
- dependencies = {file = ["requirements.txt"]}
21
-
22
- [tool.setuptools]
23
- packages = ["gtfparse"]
24
-
25
- [project.urls]
26
- "Homepage" = "https://github.com/openvax/gtfparse"
27
- "Bug Tracker" = "https://github.com/openvax/gtfparse"
@@ -1,86 +0,0 @@
1
- from gtfparse import create_missing_features, parse_gtf_and_expand_attributes
2
- from io import StringIO
3
-
4
- # two lines from the Ensembl 54 human GTF containing only a stop_codon and
5
- # exon features, but from which gene and transcript information could be
6
- # inferred
7
- GTF_TEXT = "\n".join([
8
- "# seqname biotype feature start end score strand frame attribute",
9
- "".join([
10
- """18\tprotein_coding\tstop_codon\t32630766\t32630768\t.\t-\t0\t""",
11
- """gene_id "ENSG00000134779"; transcript_id "ENST00000334295"; exon_number "7";"""
12
- """ gene_name "C18orf10";""",
13
- """ transcript_name "C18orf10-201";"""]),
14
- "".join([
15
- """18\tprotein_coding\texon\t32663078\t32663157\t.\t+\t.\tgene_id "ENSG00000150477"; """,
16
- """transcript_id "ENST00000383055"; exon_number "1"; gene_name "KIAA1328"; """,
17
- """transcript_name "KIAA1328-202";"""]),
18
- ])
19
-
20
-
21
- GTF_DATAFRAME = parse_gtf_and_expand_attributes(StringIO(GTF_TEXT))
22
- GTF_DATAFRAME = GTF_DATAFRAME.to_pandas()
23
-
24
- def test_create_missing_features_identity():
25
- df_should_be_same = create_missing_features(GTF_DATAFRAME, {})
26
- assert len(GTF_DATAFRAME) == len(df_should_be_same), \
27
- "GTF DataFrames should be same size"
28
-
29
- def _check_expanded_dataframe(df):
30
- assert "gene" in set(df["feature"]), \
31
- "Extended GTF should contain gene feature"
32
- assert "transcript" in set(df["feature"]), \
33
- "Extended GTF should contain transcript feature"
34
-
35
- C18orf10_201_transcript_mask = (
36
- (df["feature"] == "transcript") &
37
- (df["transcript_name"] == "C18orf10-201"))
38
- assert len(df[C18orf10_201_transcript_mask]) == 1, \
39
- "Expected only 1 gene entry for C18orf10-201, got %s" % (
40
- df[C18orf10_201_transcript_mask],)
41
- transcript_seqname = df[C18orf10_201_transcript_mask].seqname.iloc[0]
42
- assert (transcript_seqname == "18"), \
43
- "Wrong seqname for C18orf10-201: %s" % transcript_seqname
44
- transcript_start = df[C18orf10_201_transcript_mask].start.iloc[0]
45
- assert (transcript_start == 32630766), \
46
- "Wrong start for C18orf10-201: %s" % transcript_start
47
- transcript_end = df[C18orf10_201_transcript_mask].end.iloc[0]
48
- assert (transcript_end == 32630768), \
49
- "Wrong end for C18orf10-201: %s" % transcript_end
50
- transcript_strand = df[C18orf10_201_transcript_mask].strand.iloc[0]
51
- assert (transcript_strand == "-"), \
52
- "Wrong strand for C18orf10-201: %s" % transcript_strand
53
-
54
- KIAA1328_gene_mask = (
55
- (df["feature"] == "gene") &
56
- (df["gene_name"] == "KIAA1328"))
57
- assert len(df[KIAA1328_gene_mask]) == 1, "Expected only 1 gene entry for KIAA1328"
58
- gene_seqname = df[KIAA1328_gene_mask].seqname.iloc[0]
59
- assert (gene_seqname == "18"), \
60
- "Wrong seqname for KIAA1328: %s" % gene_seqname
61
- gene_start = df[KIAA1328_gene_mask].start.iloc[0]
62
- assert (gene_start == 32663078), \
63
- "Wrong start for KIAA1328: %s" % (gene_start,)
64
- gene_end = df[KIAA1328_gene_mask].end.iloc[0]
65
- assert (gene_end == 32663157), \
66
- "Wrong end for KIAA1328: %s" % (gene_end,)
67
- gene_strand = df[KIAA1328_gene_mask].strand.iloc[0]
68
- assert (gene_strand == "+"), \
69
- "Wrong strand for KIAA1328: %s" % gene_strand
70
-
71
- def test_create_missing_features():
72
- assert "gene" not in set(GTF_DATAFRAME["feature"]), \
73
- "Original GTF should not contain gene feature"
74
- assert "transcript" not in set(GTF_DATAFRAME["feature"]), \
75
- "Original GTF should not contain transcript feature"
76
- df_extra_features = create_missing_features(
77
- GTF_DATAFRAME,
78
- unique_keys={
79
- "gene": "gene_id",
80
- "transcript": "transcript_id"
81
- },
82
- extra_columns={
83
- "gene": {"gene_name"},
84
- "transcript": {"gene_id", "gene_name", "transcript_name"},
85
- })
86
- _check_expanded_dataframe(df_extra_features)
@@ -1,81 +0,0 @@
1
- from gtfparse import read_gtf
2
-
3
- from .data import data_path
4
-
5
- ENSEMBL_GTF_PATH = data_path("ensembl_grch37.head.gtf")
6
-
7
- EXPECTED_FEATURES = set([
8
- "gene",
9
- "transcript",
10
- "exon",
11
- "CDS",
12
- "UTR",
13
- "start_codon",
14
- "stop_codon",
15
- ])
16
-
17
-
18
- def test_ensembl_gtf_columns():
19
- df = read_gtf(ENSEMBL_GTF_PATH)
20
- features = set(df["feature"])
21
- assert features == EXPECTED_FEATURES
22
-
23
- # first 1000 lines of GTF only contained these genes
24
- EXPECTED_GENE_NAMES = {
25
- 'FAM41C', 'CICP27', 'RNU6-1100P', 'NOC2L', 'AP006222.1',
26
- 'LINC01128', 'RP4-669L17.1', 'RP11-206L10.2', 'PLEKHN1',
27
- 'WBP1LP7', 'RP5-857K21.1', 'RP5-857K21.5', 'RNU6-1199P',
28
- 'RP11-206L10.10', 'RP11-54O7.16', 'CICP7', 'AL627309.1',
29
- 'RP5-857K21.11', 'DDX11L1', 'RP5-857K21.3', 'RP11-34P13.7',
30
- 'AL669831.1', 'MTATP6P1', 'CICP3', 'WBP1LP6', 'LINC00115',
31
- 'hsa-mir-6723', 'RP5-857K21.7', 'SAMD11', 'RP11-206L10.5',
32
- 'RP11-34P13.8', 'RP11-206L10.9', 'RP11-34P13.15', 'TUBB8P11',
33
- 'MTATP8P1', 'RP4-669L17.8', 'RP11-206L10.1', 'RP11-34P13.13',
34
- 'RP11-206L10.3', 'RP11-206L10.4', 'RP11-54O7.3', 'RP5-857K21.2',
35
- 'OR4F5', 'MTND1P23', 'AL645608.1', 'RP11-34P13.16', 'RP11-34P13.14',
36
- 'AP006222.2', 'OR4F29', 'RP4-669L17.4', 'AL732372.1', 'OR4G4P',
37
- 'MTND2P28', 'OR4F16', 'KLHL17', 'FAM138A', 'OR4G11P', 'FAM87B',
38
- 'RP5-857K21.15', 'AL645608.2', 'RP11-206L10.8', 'RP5-857K21.4',
39
- 'MIR1302-10', 'RP11-54O7.2', 'RP4-669L17.10', 'RP11-54O7.1',
40
- 'RP11-34P13.9', 'WASH7P', 'RP4-669L17.2'
41
- }
42
-
43
- def test_ensembl_gtf_gene_names():
44
- df = read_gtf(ENSEMBL_GTF_PATH)
45
- gene_names = set(df["gene_name"])
46
- assert gene_names == EXPECTED_GENE_NAMES, \
47
- "Wrong gene names: %s, missing %s and unexpected %s" % (
48
- gene_names,
49
- EXPECTED_GENE_NAMES.difference(gene_names),
50
- gene_names.difference(EXPECTED_GENE_NAMES)
51
- )
52
-
53
- def test_ensembl_gtf_gene_names_with_usecols():
54
- df = read_gtf(ENSEMBL_GTF_PATH, usecols=["gene_name"])
55
- gene_names = set(df["gene_name"])
56
- assert gene_names == EXPECTED_GENE_NAMES, \
57
- "Wrong gene names: %s, missing %s and unexpected %s" % (
58
- gene_names,
59
- EXPECTED_GENE_NAMES.difference(gene_names),
60
- gene_names.difference(EXPECTED_GENE_NAMES)
61
- )
62
-
63
- def test_ensembl_gtf_gene_names_zip():
64
- df = read_gtf(ENSEMBL_GTF_PATH + ".gz")
65
- gene_names = set(df["gene_name"])
66
- assert gene_names == EXPECTED_GENE_NAMES, \
67
- "Wrong gene names: %s, missing %s and unexpected %s" % (
68
- gene_names,
69
- EXPECTED_GENE_NAMES.difference(gene_names),
70
- gene_names.difference(EXPECTED_GENE_NAMES)
71
- )
72
-
73
- def test_ensembl_gtf_gene_names_with_usecols_gzip():
74
- df = read_gtf(ENSEMBL_GTF_PATH + ".gz", usecols=["gene_name"])
75
- gene_names = set(df["gene_name"])
76
- assert gene_names == EXPECTED_GENE_NAMES, \
77
- "Wrong gene names: %s, missing %s and unexpected %s" % (
78
- gene_names,
79
- EXPECTED_GENE_NAMES.difference(gene_names),
80
- gene_names.difference(EXPECTED_GENE_NAMES)
81
- )
File without changes
File without changes
File without changes
File without changes