gtfparse 1.3.0__tar.gz → 2.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (29) hide show
  1. {gtfparse-1.3.0 → gtfparse-2.1.0}/PKG-INFO +8 -5
  2. {gtfparse-1.3.0 → gtfparse-2.1.0}/gtfparse/__init__.py +12 -3
  3. {gtfparse-1.3.0 → gtfparse-2.1.0}/gtfparse/attribute_parsing.py +17 -18
  4. {gtfparse-1.3.0 → gtfparse-2.1.0}/gtfparse/create_missing_features.py +1 -1
  5. gtfparse-2.1.0/gtfparse/read_gtf.py +297 -0
  6. {gtfparse-1.3.0 → gtfparse-2.1.0}/gtfparse.egg-info/PKG-INFO +8 -5
  7. {gtfparse-1.3.0 → gtfparse-2.1.0}/gtfparse.egg-info/SOURCES.txt +10 -10
  8. gtfparse-2.1.0/gtfparse.egg-info/requires.txt +2 -0
  9. gtfparse-2.1.0/pyproject.toml +27 -0
  10. gtfparse-2.1.0/requirements.txt +2 -0
  11. {gtfparse-1.3.0/test → gtfparse-2.1.0/tests}/test_create_missing_features.py +2 -1
  12. {gtfparse-1.3.0/test → gtfparse-2.1.0/tests}/test_ensembl_gtf.py +2 -2
  13. gtfparse-2.1.0/tests/test_expand_attributes.py +36 -0
  14. {gtfparse-1.3.0/test → gtfparse-2.1.0/tests}/test_multiple_values_for_tag_attribute.py +6 -8
  15. {gtfparse-1.3.0/test → gtfparse-2.1.0/tests}/test_parse_gtf_lines.py +21 -29
  16. gtfparse-1.3.0/gtfparse/read_gtf.py +0 -243
  17. gtfparse-1.3.0/gtfparse/required_columns.py +0 -62
  18. gtfparse-1.3.0/gtfparse/version.py +0 -1
  19. gtfparse-1.3.0/gtfparse.egg-info/requires.txt +0 -2
  20. gtfparse-1.3.0/setup.py +0 -61
  21. gtfparse-1.3.0/test/test_expand_attributes.py +0 -37
  22. {gtfparse-1.3.0 → gtfparse-2.1.0}/LICENSE +0 -0
  23. {gtfparse-1.3.0 → gtfparse-2.1.0}/README.md +0 -0
  24. {gtfparse-1.3.0 → gtfparse-2.1.0}/gtfparse/parsing_error.py +0 -0
  25. {gtfparse-1.3.0 → gtfparse-2.1.0}/gtfparse.egg-info/dependency_links.txt +0 -0
  26. {gtfparse-1.3.0 → gtfparse-2.1.0}/gtfparse.egg-info/top_level.txt +0 -0
  27. {gtfparse-1.3.0 → gtfparse-2.1.0}/setup.cfg +0 -0
  28. {gtfparse-1.3.0/test → gtfparse-2.1.0/tests}/test_read_stringtie_gtf.py +0 -0
  29. {gtfparse-1.3.0/test → gtfparse-2.1.0/tests}/test_refseq_gtf.py +0 -0
@@ -1,10 +1,10 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: gtfparse
3
- Version: 1.3.0
4
- Summary: GTF Parsing
5
- Home-page: https://github.com/openvax/gtfparse
6
- Author: Alex Rubinsteyn
7
- License: http://www.apache.org/licenses/LICENSE-2.0.html
3
+ Version: 2.1.0
4
+ Summary: Parsing library for extracting data frames of genomic features from GTF files
5
+ Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
6
+ Project-URL: Homepage, https://github.com/openvax/gtfparse
7
+ Project-URL: Bug Tracker, https://github.com/openvax/gtfparse
8
8
  Classifier: Development Status :: 4 - Beta
9
9
  Classifier: Environment :: Console
10
10
  Classifier: Operating System :: OS Independent
@@ -12,8 +12,11 @@ Classifier: Intended Audience :: Science/Research
12
12
  Classifier: License :: OSI Approved :: Apache Software License
13
13
  Classifier: Programming Language :: Python
14
14
  Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
15
+ Requires-Python: >=3.7
15
16
  Description-Content-Type: text/markdown
16
17
  License-File: LICENSE
18
+ Requires-Dist: polars<0.21.0,>=0.20.2
19
+ Requires-Dist: pyarrow<14.1.0,>=14.0.2
17
20
 
18
21
  [![Build Status](https://travis-ci.org/openvax/gtfparse.svg?branch=master)](https://travis-ci.org/openvax/gtfparse) [![Coverage Status](https://coveralls.io/repos/openvax/gtfparse/badge.svg?branch=master&service=github)](https://coveralls.io/github/openvax/gtfparse?branch=master)
19
22
  <a href="https://pypi.python.org/pypi/gtfparse/">
@@ -12,17 +12,26 @@
12
12
 
13
13
  from .attribute_parsing import expand_attribute_strings
14
14
  from .create_missing_features import create_missing_features
15
- from .required_columns import REQUIRED_COLUMNS
16
15
  from .parsing_error import ParsingError
17
- from .read_gtf import read_gtf, parse_gtf, parse_gtf_and_expand_attributes
16
+ from .read_gtf import (
17
+ read_gtf,
18
+ parse_gtf,
19
+ parse_gtf_pandas,
20
+ parse_gtf_and_expand_attributes,
21
+ REQUIRED_COLUMNS,
22
+ )
18
23
 
24
+ __version__ = "2.1.0"
19
25
 
20
26
  __all__ = [
27
+ "__version__",
21
28
  "expand_attribute_strings",
22
29
  "create_missing_features",
23
- "parse_gtf",
30
+
24
31
  "parse_gtf_and_expand_attributes",
25
32
  "REQUIRED_COLUMNS",
26
33
  "ParsingError",
27
34
  "read_gtf",
35
+ "parse_gtf",
36
+ "parse_gtf_pandas",
28
37
  ]
@@ -18,9 +18,10 @@ logging.basicConfig(level=logging.INFO)
18
18
  logger = logging.getLogger(__name__)
19
19
 
20
20
 
21
+
21
22
  def expand_attribute_strings(
22
23
  attribute_strings,
23
- quote_char='\"',
24
+ quote_char="'",
24
25
  missing_value="",
25
26
  usecols=None):
26
27
  """
@@ -64,10 +65,11 @@ def expand_attribute_strings(
64
65
  # using a local dictionary, hence the two dictionaries below
65
66
  # and pair of try/except blocks in the loop.
66
67
  column_interned_strings = {}
67
- value_interned_strings = {}
68
68
 
69
- for (i, attribute_string) in enumerate(attribute_strings):
70
- for kv in attribute_string.split(";"):
69
+ for (i, kv_strings) in enumerate(attribute_strings):
70
+ if type(kv_strings) is str:
71
+ kv_strings = kv_strings.split(";")
72
+ for kv in kv_strings:
71
73
  # We're slicing the first two elements out of split() because
72
74
  # Ensembl release 79 added values like:
73
75
  # transcript_support_level "1 (assigned to previous version 5)";
@@ -88,28 +90,25 @@ def expand_attribute_strings(
88
90
  if usecols is not None and column_name not in usecols:
89
91
  continue
90
92
 
93
+ if value[0] == quote_char:
94
+ value = value.replace(quote_char, "")
95
+
91
96
  try:
92
97
  column = extra_columns[column_name]
98
+ # if an attribute is used repeatedly then
99
+ # keep track of all its values in a list
100
+ old_value = column[i]
101
+ if old_value is missing_value:
102
+ column[i] = value
103
+ else:
104
+ column[i] = "%s,%s" % (old_value, value)
93
105
  except KeyError:
94
106
  column = [missing_value] * n
107
+ column[i] = value
95
108
  extra_columns[column_name] = column
96
109
  column_order.append(column_name)
97
110
 
98
- value = value.replace(quote_char, "") if value.startswith(quote_char) else value
99
-
100
- try:
101
- value = value_interned_strings[value]
102
- except KeyError:
103
- value = intern(str(value))
104
- value_interned_strings[value] = value
105
111
 
106
- # if an attribute is used repeatedly then
107
- # keep track of all its values in a list
108
- old_value = column[i]
109
- if old_value is missing_value:
110
- column[i] = value
111
- else:
112
- column[i] = "%s,%s" % (old_value, value)
113
112
 
114
113
  logging.info("Extracted GTF attributes: %s" % column_order)
115
114
  return OrderedDict(
@@ -55,7 +55,7 @@ def create_missing_features(
55
55
  extra_dataframes = []
56
56
 
57
57
  existing_features = set(dataframe["feature"])
58
- existing_columns = set(dataframe.keys())
58
+ existing_columns = set(dataframe.columns)
59
59
 
60
60
  for (feature_name, groupby_key) in unique_keys.items():
61
61
  if feature_name in existing_features:
@@ -0,0 +1,297 @@
1
+ # Licensed under the Apache License, Version 2.0 (the "License");
2
+ # you may not use this file except in compliance with the License.
3
+ # You may obtain a copy of the License at
4
+ #
5
+ # http://www.apache.org/licenses/LICENSE-2.0
6
+ #
7
+ # Unless required by applicable law or agreed to in writing, software
8
+ # distributed under the License is distributed on an "AS IS" BASIS,
9
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
10
+ # See the License for the specific language governing permissions and
11
+ # limitations under the License.
12
+
13
+ import logging
14
+ from os.path import exists
15
+ from io import StringIO
16
+ import gzip
17
+
18
+ import polars
19
+
20
+ from .attribute_parsing import expand_attribute_strings
21
+ from .parsing_error import ParsingError
22
+
23
+
24
+ logging.basicConfig(level=logging.INFO)
25
+ logger = logging.getLogger(__name__)
26
+
27
+
28
+ """
29
+ Columns of a GTF file:
30
+
31
+ seqname - name of the chromosome or scaffold; chromosome names
32
+ without a 'chr' in Ensembl (but sometimes with a 'chr'
33
+ elsewhere)
34
+ source - name of the program that generated this feature, or
35
+ the data source (database or project name)
36
+ feature - feature type name.
37
+ Features currently in Ensembl GTFs:
38
+ gene
39
+ transcript
40
+ exon
41
+ CDS
42
+ Selenocysteine
43
+ start_codon
44
+ stop_codon
45
+ UTR
46
+ Older Ensembl releases may be missing some of these features.
47
+ start - start position of the feature, with sequence numbering
48
+ starting at 1.
49
+ end - end position of the feature, with sequence numbering
50
+ starting at 1.
51
+ score - a floating point value indiciating the score of a feature
52
+ strand - defined as + (forward) or - (reverse).
53
+ frame - one of '0', '1' or '2'. Frame indicates the number of base pairs
54
+ before you encounter a full codon. '0' indicates the feature
55
+ begins with a whole codon. '1' indicates there is an extra
56
+ base (the 3rd base of the prior codon) at the start of this feature.
57
+ '2' indicates there are two extra bases (2nd and 3rd base of the
58
+ prior exon) before the first codon. All values are given with
59
+ relation to the 5' end.
60
+ attribute - a semicolon-separated list of tag-value pairs (separated by a space),
61
+ providing additional information about each feature. A key can be
62
+ repeated multiple times.
63
+
64
+ (from ftp://ftp.ensembl.org/pub/release-75/gtf/homo_sapiens/README)
65
+ """
66
+
67
+ REQUIRED_COLUMNS = [
68
+ "seqname",
69
+ "source",
70
+ "feature",
71
+ "start",
72
+ "end",
73
+ "score",
74
+ "strand",
75
+ "frame",
76
+ "attribute",
77
+ ]
78
+
79
+
80
+ def parse_with_polars_lazy(
81
+ filepath_or_buffer,
82
+ split_attributes=True,
83
+ features=None,
84
+ fix_quotes_columns=["attribute"]):
85
+ # use a global string cache so that all strings get intern'd into
86
+ # a single numbering system
87
+ polars.enable_string_cache()
88
+ kwargs = dict(
89
+ has_header=False,
90
+ separator="\t",
91
+ comment_prefix="#",
92
+ null_values=".",
93
+ dtypes={
94
+ "seqname": polars.Categorical,
95
+ "source": polars.Categorical,
96
+
97
+ "start": polars.Int64,
98
+ "end": polars.Int64,
99
+ "score": polars.Float32,
100
+
101
+ "feature": polars.Categorical,
102
+ "strand": polars.Categorical,
103
+ "frame": polars.UInt32,
104
+ })
105
+ try:
106
+ if type(filepath_or_buffer) is StringIO:
107
+ df = polars.read_csv(
108
+ filepath_or_buffer,
109
+ new_columns=REQUIRED_COLUMNS,
110
+ **kwargs).lazy()
111
+ elif filepath_or_buffer.endswith(".gz") or filepath_or_buffer.endswith(".gzip"):
112
+ with gzip.open(filepath_or_buffer) as f:
113
+ df = polars.read_csv(
114
+ f,
115
+ new_columns=REQUIRED_COLUMNS,
116
+ **kwargs).lazy()
117
+ else:
118
+ df = polars.scan_csv(
119
+ filepath_or_buffer,
120
+ with_column_names=lambda cols: REQUIRED_COLUMNS,
121
+ **kwargs).lazy()
122
+ except polars.ShapeError:
123
+ raise ParsingError("Wrong number of columns")
124
+
125
+ df = df.with_columns([
126
+ polars.col("frame").fill_null(0),
127
+ polars.col("attribute").str.replace_all('"', "'")
128
+ ])
129
+
130
+ for fix_quotes_column in fix_quotes_columns:
131
+ # Catch mistaken semicolons by replacing "xyz;" with "xyz"
132
+ # Required to do this since the Ensembl GTF for Ensembl
133
+ # release 78 has mistakes such as:
134
+ # gene_name = "PRAMEF6;" transcript_name = "PRAMEF6;-201"
135
+ df = df.with_columns([
136
+ polars.col(fix_quotes_column).str.replace(';\"', '\"').str.replace(";-", "-")
137
+ ])
138
+
139
+ if features is not None:
140
+ features = sorted(set(features))
141
+ df = df.filter(polars.col("feature").is_in(features))
142
+
143
+
144
+ if split_attributes:
145
+ df = df.with_columns([
146
+ polars.col("attribute").str.split(";").alias("attribute_split")
147
+ ])
148
+ return df
149
+
150
+ def parse_gtf(
151
+ filepath_or_buffer,
152
+ split_attributes=True,
153
+ features=None,
154
+ fix_quotes_columns=["attribute"]):
155
+ df_lazy = parse_with_polars_lazy(
156
+ filepath_or_buffer=filepath_or_buffer,
157
+ split_attributes=split_attributes,
158
+ features=features,
159
+ fix_quotes_columns=fix_quotes_columns)
160
+ return df_lazy.collect()
161
+
162
+ def parse_gtf_pandas(*args, **kwargs):
163
+ return parse_gtf(*args, **kwargs).to_pandas()
164
+
165
+
166
+ def parse_gtf_and_expand_attributes(
167
+ filepath_or_buffer,
168
+ restrict_attribute_columns=None,
169
+ features=None):
170
+ """
171
+ Parse lines into column->values dictionary and then expand
172
+ the 'attribute' column into multiple columns. This expansion happens
173
+ by replacing strings of semi-colon separated key-value values in the
174
+ 'attribute' column with one column per distinct key, with a list of
175
+ values for each row (using None for rows where key didn't occur).
176
+
177
+ Parameters
178
+ ----------
179
+ filepath_or_buffer : str or buffer object
180
+
181
+ chunksize : int
182
+
183
+ restrict_attribute_columns : list/set of str or None
184
+ If given, then only use these attribute columns.
185
+
186
+ features : set or None
187
+ Ignore entries which don't correspond to one of the supplied features
188
+ """
189
+ df = parse_gtf(
190
+ filepath_or_buffer=filepath_or_buffer,
191
+ features=features,
192
+ split_attributes=True)
193
+ if type(restrict_attribute_columns) is str:
194
+ restrict_attribute_columns = {restrict_attribute_columns}
195
+ elif restrict_attribute_columns:
196
+ restrict_attribute_columns = set(restrict_attribute_columns)
197
+ df.drop_in_place("attribute")
198
+ attribute_pairs = df.drop_in_place("attribute_split")
199
+ return df.with_columns([
200
+ polars.Series(k, vs)
201
+ for (k, vs) in
202
+ expand_attribute_strings(attribute_pairs).items()
203
+ if restrict_attribute_columns is None or k in restrict_attribute_columns
204
+ ])
205
+
206
+
207
+ def read_gtf(
208
+ filepath_or_buffer,
209
+ expand_attribute_column=True,
210
+ infer_biotype_column=False,
211
+ column_converters={},
212
+ usecols=None,
213
+ features=None,
214
+ result_type='polars'):
215
+ """
216
+ Parse a GTF into a dictionary mapping column names to sequences of values.
217
+
218
+ Parameters
219
+ ----------
220
+ filepath_or_buffer : str or buffer object
221
+ Path to GTF file (may be gzip compressed) or buffer object
222
+ such as StringIO
223
+
224
+ expand_attribute_column : bool
225
+ Replace strings of semi-colon separated key-value values in the
226
+ 'attribute' column with one column per distinct key, with a list of
227
+ values for each row (using None for rows where key didn't occur).
228
+
229
+ infer_biotype_column : bool
230
+ Due to the annoying ambiguity of the second GTF column across multiple
231
+ Ensembl releases, figure out if an older GTF's source column is actually
232
+ the gene_biotype or transcript_biotype.
233
+
234
+ column_converters : dict, optional
235
+ Dictionary mapping column names to conversion functions. Will replace
236
+ empty strings with None and otherwise passes them to given conversion
237
+ function.
238
+
239
+ usecols : list of str or None
240
+ Restrict which columns are loaded to the give set. If None, then
241
+ load all columns.
242
+
243
+ features : set of str or None
244
+ Drop rows which aren't one of the features in the supplied set
245
+
246
+ result_type : One of 'polars', 'pandas', or 'dict'
247
+ Default behavior is to return a Polars DataFrame, but will convert to
248
+ Pandas DataFrame or dictionary if specified.
249
+ """
250
+ if type(filepath_or_buffer) is str and not exists(filepath_or_buffer):
251
+ raise ValueError("GTF file does not exist: %s" % filepath_or_buffer)
252
+
253
+ if expand_attribute_column:
254
+ result_df = parse_gtf_and_expand_attributes(
255
+ filepath_or_buffer,
256
+ restrict_attribute_columns=usecols,
257
+ features=features)
258
+ else:
259
+ result_df = parse_gtf(result_df, features=features)
260
+
261
+ result_df = result_df.with_columns(
262
+ [
263
+ polars.col(column_name).map_elements(lambda x: column_type(x) if len(x) > 0 else None)
264
+ for column_name, column_type in column_converters.items()
265
+ ]
266
+ )
267
+
268
+ # Hackishly infer whether the values in the 'source' column of this GTF
269
+ # are actually representing a biotype by checking for the most common
270
+ # gene_biotype and transcript_biotype value 'protein_coding'
271
+ if infer_biotype_column:
272
+ unique_source_values = set(result_df["source"])
273
+ if "protein_coding" in unique_source_values:
274
+ column_names = set(result_df.columns)
275
+ # Disambiguate between the two biotypes by checking if
276
+ # gene_biotype is already present in another column. If it is,
277
+ # the 2nd column is the transcript_biotype (otherwise, it's the
278
+ # gene_biotype)
279
+ if "gene_biotype" not in column_names:
280
+ logging.info("Using column 'source' to replace missing 'gene_biotype'")
281
+ result_df = result_df.with_column(polars.col("source").alias("gene_biotype"))
282
+ if "transcript_biotype" not in column_names:
283
+ logging.info("Using column 'source' to replace missing 'transcript_biotype'")
284
+ result_df = result_df.with_column(polars.col("source").alias("transcript_biotype"))
285
+
286
+ if usecols is not None:
287
+ column_names = set(result_df.columns)
288
+ valid_columns = [c for c in usecols if c in column_names]
289
+ result_df = result_df.select(valid_columns)
290
+
291
+ if result_type == "pandas":
292
+ result = result_df.to_pandas()
293
+ elif result_type == "polars":
294
+ result = result_df
295
+ elif result_type == "dict":
296
+ result = result_df.to_dict()
297
+ return result
@@ -1,10 +1,10 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: gtfparse
3
- Version: 1.3.0
4
- Summary: GTF Parsing
5
- Home-page: https://github.com/openvax/gtfparse
6
- Author: Alex Rubinsteyn
7
- License: http://www.apache.org/licenses/LICENSE-2.0.html
3
+ Version: 2.1.0
4
+ Summary: Parsing library for extracting data frames of genomic features from GTF files
5
+ Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
6
+ Project-URL: Homepage, https://github.com/openvax/gtfparse
7
+ Project-URL: Bug Tracker, https://github.com/openvax/gtfparse
8
8
  Classifier: Development Status :: 4 - Beta
9
9
  Classifier: Environment :: Console
10
10
  Classifier: Operating System :: OS Independent
@@ -12,8 +12,11 @@ Classifier: Intended Audience :: Science/Research
12
12
  Classifier: License :: OSI Approved :: Apache Software License
13
13
  Classifier: Programming Language :: Python
14
14
  Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
15
+ Requires-Python: >=3.7
15
16
  Description-Content-Type: text/markdown
16
17
  License-File: LICENSE
18
+ Requires-Dist: polars<0.21.0,>=0.20.2
19
+ Requires-Dist: pyarrow<14.1.0,>=14.0.2
17
20
 
18
21
  [![Build Status](https://travis-ci.org/openvax/gtfparse.svg?branch=master)](https://travis-ci.org/openvax/gtfparse) [![Coverage Status](https://coveralls.io/repos/openvax/gtfparse/badge.svg?branch=master&service=github)](https://coveralls.io/github/openvax/gtfparse?branch=master)
19
22
  <a href="https://pypi.python.org/pypi/gtfparse/">
@@ -1,22 +1,22 @@
1
1
  LICENSE
2
2
  README.md
3
- setup.py
3
+ pyproject.toml
4
+ requirements.txt
4
5
  gtfparse/__init__.py
5
6
  gtfparse/attribute_parsing.py
6
7
  gtfparse/create_missing_features.py
7
8
  gtfparse/parsing_error.py
8
9
  gtfparse/read_gtf.py
9
- gtfparse/required_columns.py
10
- gtfparse/version.py
11
10
  gtfparse.egg-info/PKG-INFO
12
11
  gtfparse.egg-info/SOURCES.txt
13
12
  gtfparse.egg-info/dependency_links.txt
14
13
  gtfparse.egg-info/requires.txt
15
14
  gtfparse.egg-info/top_level.txt
16
- test/test_create_missing_features.py
17
- test/test_ensembl_gtf.py
18
- test/test_expand_attributes.py
19
- test/test_multiple_values_for_tag_attribute.py
20
- test/test_parse_gtf_lines.py
21
- test/test_read_stringtie_gtf.py
22
- test/test_refseq_gtf.py
15
+ gtfparse/../requirements.txt
16
+ tests/test_create_missing_features.py
17
+ tests/test_ensembl_gtf.py
18
+ tests/test_expand_attributes.py
19
+ tests/test_multiple_values_for_tag_attribute.py
20
+ tests/test_parse_gtf_lines.py
21
+ tests/test_read_stringtie_gtf.py
22
+ tests/test_refseq_gtf.py
@@ -0,0 +1,2 @@
1
+ polars<0.21.0,>=0.20.2
2
+ pyarrow<14.1.0,>=14.0.2
@@ -0,0 +1,27 @@
1
+ [project]
2
+ name = "gtfparse"
3
+ requires-python = ">=3.7"
4
+ authors = [ {name="Alex Rubinsteyn", email="alex.rubinsteyn@unc.edu" } ]
5
+ description = "Parsing library for extracting data frames of genomic features from GTF files"
6
+ classifiers = [
7
+ 'Development Status :: 4 - Beta',
8
+ 'Environment :: Console',
9
+ 'Operating System :: OS Independent',
10
+ 'Intended Audience :: Science/Research',
11
+ 'License :: OSI Approved :: Apache Software License',
12
+ 'Programming Language :: Python',
13
+ 'Topic :: Scientific/Engineering :: Bio-Informatics',
14
+ ]
15
+ readme = "README.md"
16
+ dynamic = ["version", "dependencies"]
17
+
18
+ [tool.setuptools.dynamic]
19
+ version = {attr = "gtfparse.__version__"}
20
+ dependencies = {file = ["requirements.txt"]}
21
+
22
+ [tool.setuptools]
23
+ packages = ["gtfparse"]
24
+
25
+ [project.urls]
26
+ "Homepage" = "https://github.com/openvax/gtfparse"
27
+ "Bug Tracker" = "https://github.com/openvax/gtfparse"
@@ -0,0 +1,2 @@
1
+ polars>=0.20.2,<0.21.0
2
+ pyarrow>=14.0.2,<14.1.0
@@ -1,5 +1,5 @@
1
1
  from gtfparse import create_missing_features, parse_gtf_and_expand_attributes
2
- from six import StringIO
2
+ from io import StringIO
3
3
 
4
4
  # two lines from the Ensembl 54 human GTF containing only a stop_codon and
5
5
  # exon features, but from which gene and transcript information could be
@@ -19,6 +19,7 @@ GTF_TEXT = "\n".join([
19
19
 
20
20
 
21
21
  GTF_DATAFRAME = parse_gtf_and_expand_attributes(StringIO(GTF_TEXT))
22
+ GTF_DATAFRAME = GTF_DATAFRAME.to_pandas()
22
23
 
23
24
  def test_create_missing_features_identity():
24
25
  df_should_be_same = create_missing_features(GTF_DATAFRAME, {})
@@ -1,6 +1,6 @@
1
1
  from data import data_path
2
2
  from gtfparse import read_gtf
3
- from nose.tools import eq_
3
+
4
4
 
5
5
  ENSEMBL_GTF_PATH = data_path("ensembl_grch37.head.gtf")
6
6
 
@@ -18,7 +18,7 @@ EXPECTED_FEATURES = set([
18
18
  def test_ensembl_gtf_columns():
19
19
  df = read_gtf(ENSEMBL_GTF_PATH)
20
20
  features = set(df["feature"])
21
- eq_(features, EXPECTED_FEATURES)
21
+ assert features == EXPECTED_FEATURES
22
22
 
23
23
  # first 1000 lines of GTF only contained these genes
24
24
  EXPECTED_GENE_NAMES = {
@@ -0,0 +1,36 @@
1
+ from gtfparse import expand_attribute_strings
2
+
3
+ def test_attributes_in_quotes():
4
+ attributes = [
5
+ "gene_id \"ENSG001\"; tag \"bogotron\"; version \"1\";",
6
+ "gene_id \"ENSG002\"; tag \"wolfpuppy\"; version \"2\";"
7
+ ]
8
+ parsed_dict = expand_attribute_strings(attributes, quote_char='"')
9
+ assert list(sorted(parsed_dict.keys())), ["gene_id", "tag", "version"]
10
+ assert parsed_dict["gene_id"] == ["ENSG001", "ENSG002"]
11
+ assert parsed_dict["tag"] == ["bogotron", "wolfpuppy"]
12
+ assert parsed_dict["version"] == ["1", "2"]
13
+
14
+
15
+ def test_attributes_without_quotes():
16
+ attributes = [
17
+ "gene_id ENSG001; tag bogotron; version 1;",
18
+ "gene_id ENSG002; tag wolfpuppy; version 2"
19
+ ]
20
+ parsed_dict = expand_attribute_strings(attributes)
21
+ assert list(sorted(parsed_dict.keys())) == ["gene_id", "tag", "version"]
22
+ assert parsed_dict["gene_id"] == ["ENSG001", "ENSG002"]
23
+ assert parsed_dict["tag"] == ["bogotron", "wolfpuppy"]
24
+ assert parsed_dict["version"] == ["1", "2"]
25
+
26
+
27
+ def test_optional_attributes():
28
+ attributes = [
29
+ "gene_id ENSG001; sometimes-present bogotron;",
30
+ "gene_id ENSG002;",
31
+ "gene_id ENSG003; sometimes-present wolfpuppy;",
32
+ ]
33
+ parsed_dict = expand_attribute_strings(attributes)
34
+ assert list(sorted(parsed_dict.keys())) == ["gene_id", "sometimes-present"]
35
+ assert parsed_dict["gene_id"] == ["ENSG001", "ENSG002", "ENSG003"]
36
+ assert parsed_dict["sometimes-present"] == ["bogotron", "", "wolfpuppy"]
@@ -1,6 +1,5 @@
1
- from six import StringIO
1
+ from io import StringIO
2
2
  from gtfparse import parse_gtf_and_expand_attributes
3
- from nose.tools import eq_
4
3
 
5
4
  # failing example from https://github.com/openvax/gtfparse/issues/2
6
5
  GTF_TEXT = (
@@ -15,23 +14,22 @@ GTF_TEXT = (
15
14
  def test_parse_tag_attributes():
16
15
  parsed = parse_gtf_and_expand_attributes(StringIO(GTF_TEXT))
17
16
  tag_column = parsed["tag"]
18
- eq_(len(tag_column), 1)
17
+ assert len(tag_column) == 1
19
18
  tags = tag_column[0]
20
- eq_(tags, 'cds_end_NF,mRNA_end_NF')
19
+ assert tags == 'cds_end_NF,mRNA_end_NF'
21
20
 
22
21
  def test_parse_tag_attributes_with_usecols():
23
22
  parsed = parse_gtf_and_expand_attributes(
24
23
  StringIO(GTF_TEXT),
25
24
  restrict_attribute_columns=["tag"])
26
25
  tag_column = parsed["tag"]
27
- eq_(len(tag_column), 1)
26
+ assert len(tag_column) == 1
28
27
  tags = tag_column[0]
29
- eq_(tags, 'cds_end_NF,mRNA_end_NF')
28
+ assert tags == 'cds_end_NF,mRNA_end_NF'
30
29
 
31
30
  def test_parse_tag_attributes_with_usecols_other_column():
32
31
  parsed = parse_gtf_and_expand_attributes(
33
32
  StringIO(GTF_TEXT),
34
33
  restrict_attribute_columns=["exon_id"])
35
- tag_column = parsed.get("tag")
36
34
 
37
- assert tag_column is None, "Expected 'tag' to get dropped but got %s" % (parsed,)
35
+ assert "tag" not in parsed, "Expected 'tag' to get dropped but got %s" % (parsed,)
@@ -1,12 +1,11 @@
1
- import numpy as np
2
- from nose.tools import eq_, assert_raises
1
+ from pytest import raises
3
2
  from gtfparse import (
4
3
  parse_gtf,
5
4
  parse_gtf_and_expand_attributes,
6
5
  REQUIRED_COLUMNS,
7
6
  ParsingError
8
7
  )
9
- from six import StringIO
8
+ from io import StringIO
10
9
 
11
10
  gtf_text = """
12
11
  # sample GTF data copied from:
@@ -16,7 +15,9 @@ gtf_text = """
16
15
  """
17
16
 
18
17
  def test_parse_gtf_lines_with_expand_attributes():
19
- parsed_dict = parse_gtf_and_expand_attributes(StringIO(gtf_text))
18
+ df = parse_gtf_and_expand_attributes(StringIO(gtf_text))
19
+
20
+
20
21
  # excluding 'attribute' column from required names
21
22
  expected_columns = REQUIRED_COLUMNS[:8] + [
22
23
  "gene_id",
@@ -28,40 +29,31 @@ def test_parse_gtf_lines_with_expand_attributes():
28
29
  "transcript_source",
29
30
  ]
30
31
  # convert to list since Py3's dictionary keys are a distinct collection type
31
- eq_(list(parsed_dict.keys()), expected_columns)
32
- eq_(list(parsed_dict["seqname"]), ["1", "1"])
32
+ assert list(df.columns) == expected_columns
33
+ assert list(df["seqname"]) == ["1", "1"]
33
34
  # convert to list for comparison since numerical columns may be NumPy arrays
34
- eq_(list(parsed_dict["start"]), [11869, 11869])
35
- eq_(list(parsed_dict["end"]), [14409, 14409])
36
- # can't compare NaN with equality
37
- scores = list(parsed_dict["score"])
38
- assert np.isnan(scores).all(), "Unexpected scores: %s" % scores
39
- eq_(list(parsed_dict["gene_id"]), ["ENSG00000223972", "ENSG00000223972"])
40
- eq_(list(parsed_dict["transcript_id"]), ["", "ENST00000456328"])
35
+ assert list(df["start"]) == [11869, 11869]
36
+ assert list(df["end"]) == [14409, 14409]
37
+
38
+ assert df["score"].is_null().all(), "Unexpected scores: %s" % (df["score"],)
39
+ assert list(df["gene_id"]) == ["ENSG00000223972", "ENSG00000223972"]
40
+ assert list(df["transcript_id"]) == ["", "ENST00000456328"]
41
41
 
42
42
 
43
43
  def test_parse_gtf_lines_without_expand_attributes():
44
- parsed_dict = parse_gtf(StringIO(gtf_text))
44
+ df = parse_gtf(StringIO(gtf_text), split_attributes=False)
45
45
 
46
46
  # convert to list since Py3's dictionary keys are a distinct collection type
47
- eq_(list(parsed_dict.keys()), REQUIRED_COLUMNS)
48
- eq_(list(parsed_dict["seqname"]), ["1", "1"])
47
+ assert list(df.columns) == REQUIRED_COLUMNS
48
+ assert list(df["seqname"]) == ["1", "1"]
49
49
  # convert to list for comparison since numerical columns may be NumPy arrays
50
- eq_(list(parsed_dict["start"]), [11869, 11869])
51
- eq_(list(parsed_dict["end"]), [14409, 14409])
52
- # can't compare NaN with equality
53
- scores = list(parsed_dict["score"])
54
- assert np.isnan(scores).all(), "Unexpected scores: %s" % scores
55
- assert len(parsed_dict["attribute"]) == 2
56
-
57
- def test_parse_gtf_lines_error_too_many_fields():
58
- bad_gtf_text = gtf_text.replace(" ", "\t")
59
- # pylint: disable=no-value-for-parameter
60
- with assert_raises(ParsingError):
61
- parse_gtf(StringIO(bad_gtf_text))
50
+ assert list(df["start"]) == [11869, 11869]
51
+ assert list(df["end"]) == [14409, 14409]
52
+ assert df["score"].is_null().all(), "Unexpected scores: %s" % (df["score"],)
53
+ assert len(df["attribute"]) == 2
62
54
 
63
55
  def test_parse_gtf_lines_error_too_few_fields():
64
56
  bad_gtf_text = gtf_text.replace("\t", " ")
65
57
  # pylint: disable=no-value-for-parameter
66
- with assert_raises(ParsingError):
58
+ with raises(ParsingError):
67
59
  parse_gtf(StringIO(bad_gtf_text))
@@ -1,243 +0,0 @@
1
- # Licensed under the Apache License, Version 2.0 (the "License");
2
- # you may not use this file except in compliance with the License.
3
- # You may obtain a copy of the License at
4
- #
5
- # http://www.apache.org/licenses/LICENSE-2.0
6
- #
7
- # Unless required by applicable law or agreed to in writing, software
8
- # distributed under the License is distributed on an "AS IS" BASIS,
9
- # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
10
- # See the License for the specific language governing permissions and
11
- # limitations under the License.
12
-
13
- import logging
14
- from os.path import exists
15
-
16
- from sys import intern
17
- import numpy as np
18
- import pandas as pd
19
-
20
- from .attribute_parsing import expand_attribute_strings
21
- from .parsing_error import ParsingError
22
- from .required_columns import REQUIRED_COLUMNS
23
-
24
- logging.basicConfig(level=logging.INFO)
25
- logger = logging.getLogger(__name__)
26
-
27
-
28
- def parse_gtf(
29
- filepath_or_buffer,
30
- chunksize=1024 * 1024,
31
- features=None,
32
- intern_columns=["seqname", "source", "strand", "frame"],
33
- fix_quotes_columns=["attribute"]):
34
- """
35
- Parameters
36
- ----------
37
-
38
- filepath_or_buffer : str or buffer object
39
-
40
- chunksize : int
41
-
42
- features : set or None
43
- Drop entries which aren't one of these features
44
-
45
- intern_columns : list
46
- These columns are short strings which should be interned
47
-
48
- fix_quotes_columns : list
49
- Most commonly the 'attribute' column which had broken quotes on
50
- some Ensembl release GTF files.
51
- """
52
- if features is not None:
53
- features = set(features)
54
-
55
- dataframes = []
56
-
57
- def parse_frame(s):
58
- if s == ".":
59
- return 0
60
- else:
61
- return int(s)
62
-
63
- # GTF columns:
64
- # 1) seqname: str ("1", "X", "chrX", etc...)
65
- # 2) source : str
66
- # Different versions of GTF use second column as of:
67
- # (a) gene biotype
68
- # (b) transcript biotype
69
- # (c) the annotation source
70
- # See: https://www.biostars.org/p/120306/#120321
71
- # 3) feature : str ("gene", "transcript", &c)
72
- # 4) start : int
73
- # 5) end : int
74
- # 6) score : float or "."
75
- # 7) strand : "+", "-", or "."
76
- # 8) frame : 0, 1, 2 or "."
77
- # 9) attribute : key-value pairs separated by semicolons
78
- # (see more complete description in docstring at top of file)
79
-
80
- chunk_iterator = pd.read_csv(
81
- filepath_or_buffer,
82
- sep="\t",
83
- comment="#",
84
- names=REQUIRED_COLUMNS,
85
- skipinitialspace=True,
86
- skip_blank_lines=True,
87
- on_bad_lines="error",
88
- chunksize=chunksize,
89
- engine="c",
90
- dtype={
91
- "start": np.int64,
92
- "end": np.int64,
93
- "score": np.float32,
94
- "seqname": str,
95
- },
96
- na_values=".",
97
- converters={"frame": parse_frame})
98
- dataframes = []
99
- try:
100
- for df in chunk_iterator:
101
- for intern_column in intern_columns:
102
- df[intern_column] = [intern(str(s)) for s in df[intern_column]]
103
-
104
- # compare feature strings after interning
105
- if features is not None:
106
- df = df[df["feature"].isin(features)]
107
-
108
- for fix_quotes_column in fix_quotes_columns:
109
- # Catch mistaken semicolons by replacing "xyz;" with "xyz"
110
- # Required to do this since the Ensembl GTF for Ensembl
111
- # release 78 has mistakes such as:
112
- # gene_name = "PRAMEF6;" transcript_name = "PRAMEF6;-201"
113
- df[fix_quotes_column] = [
114
- s.replace(';\"', '\"').replace(";-", "-")
115
- for s in df[fix_quotes_column]
116
- ]
117
- dataframes.append(df)
118
- except Exception as e:
119
- raise ParsingError(str(e))
120
- df = pd.concat(dataframes)
121
- return df
122
-
123
-
124
- def parse_gtf_and_expand_attributes(
125
- filepath_or_buffer,
126
- chunksize=1024 * 1024,
127
- restrict_attribute_columns=None,
128
- features=None):
129
- """
130
- Parse lines into column->values dictionary and then expand
131
- the 'attribute' column into multiple columns. This expansion happens
132
- by replacing strings of semi-colon separated key-value values in the
133
- 'attribute' column with one column per distinct key, with a list of
134
- values for each row (using None for rows where key didn't occur).
135
-
136
- Parameters
137
- ----------
138
- filepath_or_buffer : str or buffer object
139
-
140
- chunksize : int
141
-
142
- restrict_attribute_columns : list/set of str or None
143
- If given, then only usese attribute columns.
144
-
145
- features : set or None
146
- Ignore entries which don't correspond to one of the supplied features
147
- """
148
- result = parse_gtf(
149
- filepath_or_buffer,
150
- chunksize=chunksize,
151
- features=features)
152
- attribute_values = result["attribute"]
153
- del result["attribute"]
154
- for column_name, values in expand_attribute_strings(
155
- attribute_values, usecols=restrict_attribute_columns).items():
156
- result[column_name] = values
157
- return result
158
-
159
-
160
- def read_gtf(
161
- filepath_or_buffer,
162
- expand_attribute_column=True,
163
- infer_biotype_column=False,
164
- column_converters={},
165
- usecols=None,
166
- features=None,
167
- chunksize=1024 * 1024):
168
- """
169
- Parse a GTF into a dictionary mapping column names to sequences of values.
170
-
171
- Parameters
172
- ----------
173
- filepath_or_buffer : str or buffer object
174
- Path to GTF file (may be gzip compressed) or buffer object
175
- such as StringIO
176
-
177
- expand_attribute_column : bool
178
- Replace strings of semi-colon separated key-value values in the
179
- 'attribute' column with one column per distinct key, with a list of
180
- values for each row (using None for rows where key didn't occur).
181
-
182
- infer_biotype_column : bool
183
- Due to the annoying ambiguity of the second GTF column across multiple
184
- Ensembl releases, figure out if an older GTF's source column is actually
185
- the gene_biotype or transcript_biotype.
186
-
187
- column_converters : dict, optional
188
- Dictionary mapping column names to conversion functions. Will replace
189
- empty strings with None and otherwise passes them to given conversion
190
- function.
191
-
192
- usecols : list of str or None
193
- Restrict which columns are loaded to the give set. If None, then
194
- load all columns.
195
-
196
- features : set of str or None
197
- Drop rows which aren't one of the features in the supplied set
198
-
199
- chunksize : int
200
- """
201
- if type(filepath_or_buffer) is str and not exists(filepath_or_buffer):
202
- raise ValueError("GTF file does not exist: %s" % filepath_or_buffer)
203
-
204
- if expand_attribute_column:
205
- result_df = parse_gtf_and_expand_attributes(
206
- filepath_or_buffer,
207
- chunksize=chunksize,
208
- restrict_attribute_columns=usecols,
209
- features=features)
210
- else:
211
- result_df = parse_gtf(result_df, features=features)
212
-
213
- for column_name, column_type in list(column_converters.items()):
214
- result_df[column_name] = [
215
- column_type(string_value) if len(string_value) > 0 else None
216
- for string_value
217
- in result_df[column_name]
218
- ]
219
-
220
- # Hackishly infer whether the values in the 'source' column of this GTF
221
- # are actually representing a biotype by checking for the most common
222
- # gene_biotype and transcript_biotype value 'protein_coding'
223
- if infer_biotype_column:
224
- unique_source_values = set(result_df["source"])
225
- if "protein_coding" in unique_source_values:
226
- column_names = set(result_df.columns)
227
- # Disambiguate between the two biotypes by checking if
228
- # gene_biotype is already present in another column. If it is,
229
- # the 2nd column is the transcript_biotype (otherwise, it's the
230
- # gene_biotype)
231
- if "gene_biotype" not in column_names:
232
- logging.info("Using column 'source' to replace missing 'gene_biotype'")
233
- result_df["gene_biotype"] = result_df["source"]
234
- if "transcript_biotype" not in column_names:
235
- logging.info("Using column 'source' to replace missing 'transcript_biotype'")
236
- result_df["transcript_biotype"] = result_df["source"]
237
-
238
- if usecols is not None:
239
- column_names = set(result_df.columns)
240
- valid_columns = [c for c in usecols if c in column_names]
241
- result_df = result_df[valid_columns]
242
-
243
- return result_df
@@ -1,62 +0,0 @@
1
- # Licensed under the Apache License, Version 2.0 (the "License");
2
- # you may not use this file except in compliance with the License.
3
- # You may obtain a copy of the License at
4
- #
5
- # http://www.apache.org/licenses/LICENSE-2.0
6
- #
7
- # Unless required by applicable law or agreed to in writing, software
8
- # distributed under the License is distributed on an "AS IS" BASIS,
9
- # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
10
- # See the License for the specific language governing permissions and
11
- # limitations under the License.
12
-
13
- """
14
- Columns of a GTF file:
15
-
16
- seqname - name of the chromosome or scaffold; chromosome names
17
- without a 'chr' in Ensembl (but sometimes with a 'chr'
18
- elsewhere)
19
- source - name of the program that generated this feature, or
20
- the data source (database or project name)
21
- feature - feature type name.
22
- Features currently in Ensembl GTFs:
23
- gene
24
- transcript
25
- exon
26
- CDS
27
- Selenocysteine
28
- start_codon
29
- stop_codon
30
- UTR
31
- Older Ensembl releases may be missing some of these features.
32
- start - start position of the feature, with sequence numbering
33
- starting at 1.
34
- end - end position of the feature, with sequence numbering
35
- starting at 1.
36
- score - a floating point value indiciating the score of a feature
37
- strand - defined as + (forward) or - (reverse).
38
- frame - one of '0', '1' or '2'. Frame indicates the number of base pairs
39
- before you encounter a full codon. '0' indicates the feature
40
- begins with a whole codon. '1' indicates there is an extra
41
- base (the 3rd base of the prior codon) at the start of this feature.
42
- '2' indicates there are two extra bases (2nd and 3rd base of the
43
- prior exon) before the first codon. All values are given with
44
- relation to the 5' end.
45
- attribute - a semicolon-separated list of tag-value pairs (separated by a space),
46
- providing additional information about each feature. A key can be
47
- repeated multiple times.
48
-
49
- (from ftp://ftp.ensembl.org/pub/release-75/gtf/homo_sapiens/README)
50
- """
51
-
52
- REQUIRED_COLUMNS = [
53
- "seqname",
54
- "source",
55
- "feature",
56
- "start",
57
- "end",
58
- "score",
59
- "strand",
60
- "frame",
61
- "attribute",
62
- ]
@@ -1 +0,0 @@
1
- __version__ = "1.3.0"
@@ -1,2 +0,0 @@
1
- numpy>=1.7
2
- pandas>=0.15
gtfparse-1.3.0/setup.py DELETED
@@ -1,61 +0,0 @@
1
- # Licensed under the Apache License, Version 2.0 (the "License");
2
- # you may not use this file except in compliance with the License.
3
- # You may obtain a copy of the License at
4
- #
5
- # http://www.apache.org/licenses/LICENSE-2.0
6
- #
7
- # Unless required by applicable law or agreed to in writing, software
8
- # distributed under the License is distributed on an "AS IS" BASIS,
9
- # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
10
- # See the License for the specific language governing permissions and
11
- # limitations under the License.
12
-
13
- from __future__ import print_function
14
- import os
15
- import re
16
-
17
- from setuptools import setup, find_packages
18
-
19
- readme_filename = "README.md"
20
- current_directory = os.path.dirname(__file__)
21
- readme_path = os.path.join(current_directory, readme_filename)
22
-
23
- readme_markdown = ""
24
- try:
25
- with open(readme_path, 'r') as f:
26
- readme_markdown = f.read()
27
- except Exception as e:
28
- print(e)
29
- print("Failed to open %s" % readme_path)
30
-
31
- with open('gtfparse/version.py', 'r') as f:
32
- version = re.search(
33
- r'^__version__\s*=\s*[\'"]([^\'"]*)[\'"]',
34
- f.read(),
35
- re.MULTILINE).group(1)
36
-
37
- if __name__ == '__main__':
38
- setup(
39
- name='gtfparse',
40
- packages=find_packages(),
41
- version=version,
42
- description="GTF Parsing",
43
- long_description=readme_markdown,
44
- long_description_content_type='text/markdown',
45
- url="https://github.com/openvax/gtfparse",
46
- author="Alex Rubinsteyn",
47
- license="http://www.apache.org/licenses/LICENSE-2.0.html",
48
- classifiers=[
49
- 'Development Status :: 4 - Beta',
50
- 'Environment :: Console',
51
- 'Operating System :: OS Independent',
52
- 'Intended Audience :: Science/Research',
53
- 'License :: OSI Approved :: Apache Software License',
54
- 'Programming Language :: Python',
55
- 'Topic :: Scientific/Engineering :: Bio-Informatics',
56
- ],
57
- install_requires=[
58
- 'numpy>=1.7',
59
- 'pandas>=0.15',
60
- ],
61
- )
@@ -1,37 +0,0 @@
1
- from gtfparse import expand_attribute_strings
2
- from nose.tools import eq_
3
-
4
- def test_attributes_in_quotes():
5
- attributes = [
6
- "gene_id \"ENSG001\"; tag \"bogotron\"; version \"1\";",
7
- "gene_id \"ENSG002\"; tag \"wolfpuppy\"; version \"2\";"
8
- ]
9
- parsed_dict = expand_attribute_strings(attributes)
10
- eq_(list(sorted(parsed_dict.keys())), ["gene_id", "tag", "version"])
11
- eq_(parsed_dict["gene_id"], ["ENSG001", "ENSG002"])
12
- eq_(parsed_dict["tag"], ["bogotron", "wolfpuppy"])
13
- eq_(parsed_dict["version"], ["1", "2"])
14
-
15
-
16
- def test_attributes_without_quotes():
17
- attributes = [
18
- "gene_id ENSG001; tag bogotron; version 1;",
19
- "gene_id ENSG002; tag wolfpuppy; version 2"
20
- ]
21
- parsed_dict = expand_attribute_strings(attributes)
22
- eq_(list(sorted(parsed_dict.keys())), ["gene_id", "tag", "version"])
23
- eq_(parsed_dict["gene_id"], ["ENSG001", "ENSG002"])
24
- eq_(parsed_dict["tag"], ["bogotron", "wolfpuppy"])
25
- eq_(parsed_dict["version"], ["1", "2"])
26
-
27
-
28
- def test_optional_attributes():
29
- attributes = [
30
- "gene_id ENSG001; sometimes-present bogotron;",
31
- "gene_id ENSG002;",
32
- "gene_id ENSG003; sometimes-present wolfpuppy;",
33
- ]
34
- parsed_dict = expand_attribute_strings(attributes)
35
- eq_(list(sorted(parsed_dict.keys())), ["gene_id", "sometimes-present"])
36
- eq_(parsed_dict["gene_id"], ["ENSG001", "ENSG002", "ENSG003"])
37
- eq_(parsed_dict["sometimes-present"], ["bogotron", "", "wolfpuppy"])
File without changes
File without changes
File without changes