gtfreader 0.2.0__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gtfreader-0.3.0/PKG-INFO +57 -0
- gtfreader-0.3.0/README.md +38 -0
- {gtfreader-0.2.0 → gtfreader-0.3.0}/gtfreader/__init__.py +8 -0
- gtfreader-0.3.0/gtfreader/_parser.pyx +296 -0
- gtfreader-0.3.0/gtfreader/readers.py +570 -0
- gtfreader-0.3.0/gtfreader.egg-info/PKG-INFO +57 -0
- {gtfreader-0.2.0 → gtfreader-0.3.0}/gtfreader.egg-info/SOURCES.txt +2 -0
- gtfreader-0.3.0/gtfreader.egg-info/requires.txt +5 -0
- {gtfreader-0.2.0 → gtfreader-0.3.0}/pyproject.toml +22 -2
- gtfreader-0.3.0/tests/test_arrow_parity.py +366 -0
- gtfreader-0.3.0/tests/test_attribute_parsers.py +136 -0
- gtfreader-0.2.0/PKG-INFO +0 -35
- gtfreader-0.2.0/README.md +0 -19
- gtfreader-0.2.0/gtfreader/_parser.pyx +0 -197
- gtfreader-0.2.0/gtfreader/readers.py +0 -328
- gtfreader-0.2.0/gtfreader.egg-info/PKG-INFO +0 -35
- gtfreader-0.2.0/gtfreader.egg-info/requires.txt +0 -1
- {gtfreader-0.2.0 → gtfreader-0.3.0}/MANIFEST.in +0 -0
- {gtfreader-0.2.0 → gtfreader-0.3.0}/gtfreader.egg-info/dependency_links.txt +0 -0
- {gtfreader-0.2.0 → gtfreader-0.3.0}/gtfreader.egg-info/top_level.txt +0 -0
- {gtfreader-0.2.0 → gtfreader-0.3.0}/setup.cfg +0 -0
- {gtfreader-0.2.0 → gtfreader-0.3.0}/setup.py +0 -0
- {gtfreader-0.2.0 → gtfreader-0.3.0}/tests/test_parser.py +0 -0
- {gtfreader-0.2.0 → gtfreader-0.3.0}/tests/test_readers.py +0 -0
gtfreader-0.3.0/PKG-INFO
ADDED
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: gtfreader
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: Fast Cython-backed parsing for GTF attribute columns.
|
|
5
|
+
Author: Endre Bakken Stovner
|
|
6
|
+
Classifier: Programming Language :: Python :: 3
|
|
7
|
+
Classifier: Programming Language :: Cython
|
|
8
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
9
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
11
|
+
Classifier: Operating System :: OS Independent
|
|
12
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
13
|
+
Requires-Python: >=3.12
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
Requires-Dist: pandas>=2.0
|
|
16
|
+
Requires-Dist: numpy
|
|
17
|
+
Provides-Extra: fast-io
|
|
18
|
+
Requires-Dist: pyarrow; extra == "fast-io"
|
|
19
|
+
|
|
20
|
+
# gtfreader
|
|
21
|
+
|
|
22
|
+
Fast GTF reading into pandas DataFrames.
|
|
23
|
+
|
|
24
|
+
## Install
|
|
25
|
+
|
|
26
|
+
```bash
|
|
27
|
+
python -m pip install -e .
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
### Optional: parse on every core
|
|
31
|
+
|
|
32
|
+
pandas' CSV parser is single-threaded, and on a large GTF the nine fixed
|
|
33
|
+
columns are about a third of the cost of reading. With `pyarrow` installed that
|
|
34
|
+
part is parsed on every core instead:
|
|
35
|
+
|
|
36
|
+
```bash
|
|
37
|
+
python -m pip install -e ".[fast-io]"
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
`read_gtf` is 1.5x faster end to end at 10^6 rows on twelve cores. It is only
|
|
41
|
+
the tabular parse that speeds up; expanding the attribute column is the rest of
|
|
42
|
+
the work and is unchanged.
|
|
43
|
+
|
|
44
|
+
pandas remains the reference implementation and the fallback. The pyarrow path
|
|
45
|
+
is skipped when pyarrow is missing, when `nrows` is set, when a file is shaped
|
|
46
|
+
in a way it does not model, and whenever the two parsers would disagree -- the
|
|
47
|
+
result is the same frame either way.
|
|
48
|
+
|
|
49
|
+
## Example
|
|
50
|
+
|
|
51
|
+
```python
|
|
52
|
+
from gtfreader import read_gtf
|
|
53
|
+
|
|
54
|
+
df = read_gtf("annotation.gtf")
|
|
55
|
+
print(df.columns)
|
|
56
|
+
print(df.head())
|
|
57
|
+
```
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
# gtfreader
|
|
2
|
+
|
|
3
|
+
Fast GTF reading into pandas DataFrames.
|
|
4
|
+
|
|
5
|
+
## Install
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
python -m pip install -e .
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
### Optional: parse on every core
|
|
12
|
+
|
|
13
|
+
pandas' CSV parser is single-threaded, and on a large GTF the nine fixed
|
|
14
|
+
columns are about a third of the cost of reading. With `pyarrow` installed that
|
|
15
|
+
part is parsed on every core instead:
|
|
16
|
+
|
|
17
|
+
```bash
|
|
18
|
+
python -m pip install -e ".[fast-io]"
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
`read_gtf` is 1.5x faster end to end at 10^6 rows on twelve cores. It is only
|
|
22
|
+
the tabular parse that speeds up; expanding the attribute column is the rest of
|
|
23
|
+
the work and is unchanged.
|
|
24
|
+
|
|
25
|
+
pandas remains the reference implementation and the fallback. The pyarrow path
|
|
26
|
+
is skipped when pyarrow is missing, when `nrows` is set, when a file is shaped
|
|
27
|
+
in a way it does not model, and whenever the two parsers would disagree -- the
|
|
28
|
+
result is the same frame either way.
|
|
29
|
+
|
|
30
|
+
## Example
|
|
31
|
+
|
|
32
|
+
```python
|
|
33
|
+
from gtfreader import read_gtf
|
|
34
|
+
|
|
35
|
+
df = read_gtf("annotation.gtf")
|
|
36
|
+
print(df.columns)
|
|
37
|
+
print(df.head())
|
|
38
|
+
```
|
|
@@ -7,6 +7,13 @@ except ImportError:
|
|
|
7
7
|
msg = "gtfreader.parse_chunk_columns requires the compiled extension. Use read_gtf_python or read_gtf_full_python for the pure Python fallback."
|
|
8
8
|
raise ImportError(msg)
|
|
9
9
|
|
|
10
|
+
try:
|
|
11
|
+
from ._parser import parse_gff3_chunk_columns
|
|
12
|
+
except ImportError:
|
|
13
|
+
def parse_gff3_chunk_columns(*_args, **_kwargs):
|
|
14
|
+
msg = "gtfreader.parse_gff3_chunk_columns requires the compiled extension."
|
|
15
|
+
raise ImportError(msg)
|
|
16
|
+
|
|
10
17
|
from .readers import (
|
|
11
18
|
find_first_data_line_index,
|
|
12
19
|
parse_kv_fields,
|
|
@@ -21,6 +28,7 @@ from .readers import (
|
|
|
21
28
|
__all__ = [
|
|
22
29
|
"find_first_data_line_index",
|
|
23
30
|
"parse_chunk_columns",
|
|
31
|
+
"parse_gff3_chunk_columns",
|
|
24
32
|
"parse_kv_fields",
|
|
25
33
|
"read_gtf",
|
|
26
34
|
"read_gtf_full",
|
|
@@ -0,0 +1,296 @@
|
|
|
1
|
+
# cython: language_level=3
|
|
2
|
+
# cython: boundscheck=False
|
|
3
|
+
# cython: wraparound=False
|
|
4
|
+
# cython: initializedcheck=False
|
|
5
|
+
# cython: infer_types=True
|
|
6
|
+
# cython: nonecheck=False
|
|
7
|
+
|
|
8
|
+
"""Compiled attribute-column parsers for GTF and GFF3.
|
|
9
|
+
|
|
10
|
+
Both work on the UTF-8 bytes of each line rather than indexing the `str`.
|
|
11
|
+
Indexing a Python string goes through a width-aware read per character, which
|
|
12
|
+
measured 350-700 MB/s; the delimiters being looked for are all ASCII, so the
|
|
13
|
+
scan can run over the raw buffer instead and comparisons become `memcmp`.
|
|
14
|
+
CPython hands back the UTF-8 buffer of an ASCII string without copying, and
|
|
15
|
+
every GTF and GFF3 attribute in practice is ASCII; a line that is not still
|
|
16
|
+
works, because offsets stay byte offsets throughout and values are decoded back
|
|
17
|
+
with `PyUnicode_DecodeUTF8`.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from libc.string cimport memcmp
|
|
21
|
+
|
|
22
|
+
cdef extern from "Python.h":
|
|
23
|
+
const char* PyUnicode_AsUTF8AndSize(object unicode, Py_ssize_t *size) except NULL
|
|
24
|
+
object PyUnicode_DecodeUTF8(const char *s, Py_ssize_t size, const char *errors)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
cdef enum:
|
|
28
|
+
TAB = 9
|
|
29
|
+
SPACE = 32
|
|
30
|
+
QUOTE = 34
|
|
31
|
+
SEMICOLON = 59
|
|
32
|
+
EQUALS = 61
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
# Values repeat: an annotation names the same gene on every one of its exons,
|
|
36
|
+
# and columns like gene_type hold a handful of values across the whole file.
|
|
37
|
+
# Handing pandas the *same* object each time rather than an equal one is worth
|
|
38
|
+
# 1.9x on frame construction at gencode-like cardinality, and frame
|
|
39
|
+
# construction is half of attribute expansion.
|
|
40
|
+
#
|
|
41
|
+
# Two levels, because they catch different shapes. The run memo compares
|
|
42
|
+
# against the column's previous value and allocates nothing at all on a hit,
|
|
43
|
+
# which is the common case in a coordinate-sorted file. The dict catches values
|
|
44
|
+
# that recur without being adjacent -- `tag` cycling through five -- but has to
|
|
45
|
+
# build the string before it can look it up, so it is the weaker of the two. It
|
|
46
|
+
# is capped: a column with a distinct value per row, like exon_id, would
|
|
47
|
+
# otherwise fill it with a million entries it can never hit.
|
|
48
|
+
DEF VALUE_CACHE_CAP = 8192
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
cdef inline bint _matches(const char* s, Py_ssize_t a, Py_ssize_t b, object other):
|
|
52
|
+
"""Is s[a:b] equal to `other`, without building s[a:b]?"""
|
|
53
|
+
cdef Py_ssize_t n = b - a
|
|
54
|
+
cdef Py_ssize_t other_len
|
|
55
|
+
cdef const char* other_buf
|
|
56
|
+
|
|
57
|
+
if other is None:
|
|
58
|
+
return False
|
|
59
|
+
other_buf = PyUnicode_AsUTF8AndSize(other, &other_len)
|
|
60
|
+
if other_len != n:
|
|
61
|
+
return False
|
|
62
|
+
if n == 0:
|
|
63
|
+
return True
|
|
64
|
+
return memcmp(s + a, other_buf, n) == 0
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
cdef inline Py_ssize_t _key_index(
|
|
68
|
+
const char* s,
|
|
69
|
+
Py_ssize_t a,
|
|
70
|
+
Py_ssize_t b,
|
|
71
|
+
list keys,
|
|
72
|
+
Py_ssize_t ordinal,
|
|
73
|
+
):
|
|
74
|
+
"""Index of s[a:b] in `keys`, or -1. Does not build s[a:b].
|
|
75
|
+
|
|
76
|
+
A file has a handful of distinct attribute keys, repeats them on every row,
|
|
77
|
+
and -- this is the part worth exploiting -- lists them in the same order
|
|
78
|
+
every time. So the key in position `ordinal` of this row is almost always
|
|
79
|
+
the one that was in position `ordinal` of the last row: try that first and
|
|
80
|
+
the scan is one comparison, not half the table.
|
|
81
|
+
"""
|
|
82
|
+
cdef Py_ssize_t i
|
|
83
|
+
cdef Py_ssize_t n = len(keys)
|
|
84
|
+
|
|
85
|
+
if 0 <= ordinal < n and _matches(s, a, b, <str> keys[ordinal]):
|
|
86
|
+
return ordinal
|
|
87
|
+
for i in range(n):
|
|
88
|
+
if _matches(s, a, b, <str> keys[i]):
|
|
89
|
+
return i
|
|
90
|
+
return -1
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
cdef inline object _value_at(
|
|
94
|
+
const char* s,
|
|
95
|
+
Py_ssize_t a,
|
|
96
|
+
Py_ssize_t b,
|
|
97
|
+
list last_values,
|
|
98
|
+
list caches,
|
|
99
|
+
Py_ssize_t index,
|
|
100
|
+
):
|
|
101
|
+
"""s[a:b], reusing an equal object already held for this column."""
|
|
102
|
+
cdef object value, cached, last
|
|
103
|
+
cdef dict cache
|
|
104
|
+
|
|
105
|
+
last = last_values[index]
|
|
106
|
+
if _matches(s, a, b, last):
|
|
107
|
+
return last
|
|
108
|
+
|
|
109
|
+
value = PyUnicode_DecodeUTF8(s + a, b - a, NULL)
|
|
110
|
+
cache = <dict> caches[index]
|
|
111
|
+
cached = cache.get(value)
|
|
112
|
+
if cached is None:
|
|
113
|
+
if len(cache) < VALUE_CACHE_CAP:
|
|
114
|
+
cache[value] = value
|
|
115
|
+
else:
|
|
116
|
+
value = cached
|
|
117
|
+
|
|
118
|
+
last_values[index] = value
|
|
119
|
+
return value
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
cdef inline Py_ssize_t _add_column(
|
|
123
|
+
object key,
|
|
124
|
+
Py_ssize_t rows,
|
|
125
|
+
dict columns,
|
|
126
|
+
list keys,
|
|
127
|
+
list cols,
|
|
128
|
+
list last_values,
|
|
129
|
+
list caches,
|
|
130
|
+
):
|
|
131
|
+
cdef list col = [None] * rows
|
|
132
|
+
|
|
133
|
+
keys.append(key)
|
|
134
|
+
cols.append(col)
|
|
135
|
+
last_values.append(None)
|
|
136
|
+
caches.append({})
|
|
137
|
+
columns[key] = col
|
|
138
|
+
return len(keys) - 1
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def parse_chunk_columns(lines):
|
|
142
|
+
"""Expand GTF `key "value";` attribute strings into a dict of columns.
|
|
143
|
+
|
|
144
|
+
Column-oriented: one list per key, filled in place, so the caller can hand
|
|
145
|
+
the dict straight to pandas without building a dict per row first.
|
|
146
|
+
"""
|
|
147
|
+
cdef Py_ssize_t n, row_i, m
|
|
148
|
+
cdef Py_ssize_t pos, key_start, key_end, val_start, val_end, index, ordinal
|
|
149
|
+
cdef dict columns
|
|
150
|
+
cdef list keys, cols, last_values, caches, col
|
|
151
|
+
cdef str line
|
|
152
|
+
cdef const char* s
|
|
153
|
+
|
|
154
|
+
n = len(lines)
|
|
155
|
+
columns = {}
|
|
156
|
+
keys = []
|
|
157
|
+
cols = []
|
|
158
|
+
last_values = []
|
|
159
|
+
caches = []
|
|
160
|
+
|
|
161
|
+
for row_i in range(n):
|
|
162
|
+
line = lines[row_i]
|
|
163
|
+
s = PyUnicode_AsUTF8AndSize(line, &m)
|
|
164
|
+
pos = 0
|
|
165
|
+
ordinal = 0
|
|
166
|
+
|
|
167
|
+
while pos < m:
|
|
168
|
+
while pos < m and (s[pos] == SPACE or s[pos] == TAB or s[pos] == SEMICOLON):
|
|
169
|
+
pos += 1
|
|
170
|
+
if pos >= m:
|
|
171
|
+
break
|
|
172
|
+
|
|
173
|
+
key_start = pos
|
|
174
|
+
|
|
175
|
+
while pos < m and s[pos] != SPACE and s[pos] != TAB:
|
|
176
|
+
pos += 1
|
|
177
|
+
key_end = pos
|
|
178
|
+
|
|
179
|
+
while pos < m and (s[pos] == SPACE or s[pos] == TAB):
|
|
180
|
+
pos += 1
|
|
181
|
+
if pos >= m:
|
|
182
|
+
break
|
|
183
|
+
|
|
184
|
+
if s[pos] == QUOTE:
|
|
185
|
+
val_start = pos + 1
|
|
186
|
+
pos += 1
|
|
187
|
+
|
|
188
|
+
while pos < m and s[pos] != QUOTE:
|
|
189
|
+
pos += 1
|
|
190
|
+
if pos >= m:
|
|
191
|
+
break
|
|
192
|
+
|
|
193
|
+
val_end = pos
|
|
194
|
+
else:
|
|
195
|
+
val_start = pos
|
|
196
|
+
while pos < m and s[pos] != SEMICOLON:
|
|
197
|
+
pos += 1
|
|
198
|
+
val_end = pos
|
|
199
|
+
while val_end > val_start and (s[val_end - 1] == SPACE or s[val_end - 1] == TAB):
|
|
200
|
+
val_end -= 1
|
|
201
|
+
|
|
202
|
+
index = _key_index(s, key_start, key_end, keys, ordinal)
|
|
203
|
+
if index < 0:
|
|
204
|
+
index = _add_column(
|
|
205
|
+
PyUnicode_DecodeUTF8(s + key_start, key_end - key_start, NULL),
|
|
206
|
+
n,
|
|
207
|
+
columns,
|
|
208
|
+
keys,
|
|
209
|
+
cols,
|
|
210
|
+
last_values,
|
|
211
|
+
caches,
|
|
212
|
+
)
|
|
213
|
+
col = <list> cols[index]
|
|
214
|
+
|
|
215
|
+
col[row_i] = _value_at(s, val_start, val_end, last_values, caches, index)
|
|
216
|
+
|
|
217
|
+
ordinal += 1
|
|
218
|
+
pos += 1
|
|
219
|
+
|
|
220
|
+
return columns
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def parse_gff3_chunk_columns(lines):
|
|
224
|
+
"""Expand GFF3 `key=value;` attribute strings into a dict of columns.
|
|
225
|
+
|
|
226
|
+
Matches `pyranges1.readers.to_keys_and_values` exactly, quirks included:
|
|
227
|
+
|
|
228
|
+
- the string is right-stripped of `;` and spaces, then split on `;`
|
|
229
|
+
- a segment with no `=` is skipped, which is also what makes an empty
|
|
230
|
+
attribute an empty row rather than an error
|
|
231
|
+
- the split is at the *first* `=`, so a value may contain more
|
|
232
|
+
- no whitespace is trimmed around a key, so `ID=a; Name=b` really does
|
|
233
|
+
yield a key of `" Name"`. GFF3 does not put a space there, but files do
|
|
234
|
+
- a repeated key keeps the last value
|
|
235
|
+
- columns come out in order of first appearance across the chunk
|
|
236
|
+
|
|
237
|
+
Column-oriented on purpose. Building a dict per row and handing the lot to
|
|
238
|
+
`DataFrame.from_records` costs as much again as the parsing does; filling
|
|
239
|
+
one list per column skips that entirely.
|
|
240
|
+
"""
|
|
241
|
+
cdef Py_ssize_t n, row_i, m
|
|
242
|
+
cdef Py_ssize_t pos, end, seg_start, seg_end, eq, index, ordinal
|
|
243
|
+
cdef dict columns
|
|
244
|
+
cdef list keys, cols, last_values, caches, col
|
|
245
|
+
cdef str line
|
|
246
|
+
cdef const char* s
|
|
247
|
+
|
|
248
|
+
n = len(lines)
|
|
249
|
+
columns = {}
|
|
250
|
+
keys = []
|
|
251
|
+
cols = []
|
|
252
|
+
last_values = []
|
|
253
|
+
caches = []
|
|
254
|
+
|
|
255
|
+
for row_i in range(n):
|
|
256
|
+
line = lines[row_i]
|
|
257
|
+
s = PyUnicode_AsUTF8AndSize(line, &m)
|
|
258
|
+
|
|
259
|
+
# `line.rstrip("; ")`
|
|
260
|
+
end = m
|
|
261
|
+
while end > 0 and (s[end - 1] == SEMICOLON or s[end - 1] == SPACE):
|
|
262
|
+
end -= 1
|
|
263
|
+
|
|
264
|
+
pos = 0
|
|
265
|
+
ordinal = 0
|
|
266
|
+
while pos < end:
|
|
267
|
+
seg_start = pos
|
|
268
|
+
while pos < end and s[pos] != SEMICOLON:
|
|
269
|
+
pos += 1
|
|
270
|
+
seg_end = pos
|
|
271
|
+
pos += 1
|
|
272
|
+
|
|
273
|
+
eq = seg_start
|
|
274
|
+
while eq < seg_end and s[eq] != EQUALS:
|
|
275
|
+
eq += 1
|
|
276
|
+
if eq >= seg_end:
|
|
277
|
+
# No `=` in this segment: not a tag=value pair.
|
|
278
|
+
continue
|
|
279
|
+
|
|
280
|
+
index = _key_index(s, seg_start, eq, keys, ordinal)
|
|
281
|
+
if index < 0:
|
|
282
|
+
index = _add_column(
|
|
283
|
+
PyUnicode_DecodeUTF8(s + seg_start, eq - seg_start, NULL),
|
|
284
|
+
n,
|
|
285
|
+
columns,
|
|
286
|
+
keys,
|
|
287
|
+
cols,
|
|
288
|
+
last_values,
|
|
289
|
+
caches,
|
|
290
|
+
)
|
|
291
|
+
col = <list> cols[index]
|
|
292
|
+
|
|
293
|
+
col[row_i] = _value_at(s, eq + 1, seg_end, last_values, caches, index)
|
|
294
|
+
ordinal += 1
|
|
295
|
+
|
|
296
|
+
return columns
|