gtfreader 0.2.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,57 @@
1
+ Metadata-Version: 2.4
2
+ Name: gtfreader
3
+ Version: 0.3.0
4
+ Summary: Fast Cython-backed parsing for GTF attribute columns.
5
+ Author: Endre Bakken Stovner
6
+ Classifier: Programming Language :: Python :: 3
7
+ Classifier: Programming Language :: Cython
8
+ Classifier: Programming Language :: Python :: 3 :: Only
9
+ Classifier: Programming Language :: Python :: 3.12
10
+ Classifier: Programming Language :: Python :: 3.13
11
+ Classifier: Operating System :: OS Independent
12
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
13
+ Requires-Python: >=3.12
14
+ Description-Content-Type: text/markdown
15
+ Requires-Dist: pandas>=2.0
16
+ Requires-Dist: numpy
17
+ Provides-Extra: fast-io
18
+ Requires-Dist: pyarrow; extra == "fast-io"
19
+
20
+ # gtfreader
21
+
22
+ Fast GTF reading into pandas DataFrames.
23
+
24
+ ## Install
25
+
26
+ ```bash
27
+ python -m pip install -e .
28
+ ```
29
+
30
+ ### Optional: parse on every core
31
+
32
+ pandas' CSV parser is single-threaded, and on a large GTF the nine fixed
33
+ columns are about a third of the cost of reading. With `pyarrow` installed that
34
+ part is parsed on every core instead:
35
+
36
+ ```bash
37
+ python -m pip install -e ".[fast-io]"
38
+ ```
39
+
40
+ `read_gtf` is 1.5x faster end to end at 10^6 rows on twelve cores. It is only
41
+ the tabular parse that speeds up; expanding the attribute column is the rest of
42
+ the work and is unchanged.
43
+
44
+ pandas remains the reference implementation and the fallback. The pyarrow path
45
+ is skipped when pyarrow is missing, when `nrows` is set, when a file is shaped
46
+ in a way it does not model, and whenever the two parsers would disagree -- the
47
+ result is the same frame either way.
48
+
49
+ ## Example
50
+
51
+ ```python
52
+ from gtfreader import read_gtf
53
+
54
+ df = read_gtf("annotation.gtf")
55
+ print(df.columns)
56
+ print(df.head())
57
+ ```
@@ -0,0 +1,38 @@
1
+ # gtfreader
2
+
3
+ Fast GTF reading into pandas DataFrames.
4
+
5
+ ## Install
6
+
7
+ ```bash
8
+ python -m pip install -e .
9
+ ```
10
+
11
+ ### Optional: parse on every core
12
+
13
+ pandas' CSV parser is single-threaded, and on a large GTF the nine fixed
14
+ columns are about a third of the cost of reading. With `pyarrow` installed that
15
+ part is parsed on every core instead:
16
+
17
+ ```bash
18
+ python -m pip install -e ".[fast-io]"
19
+ ```
20
+
21
+ `read_gtf` is 1.5x faster end to end at 10^6 rows on twelve cores. It is only
22
+ the tabular parse that speeds up; expanding the attribute column is the rest of
23
+ the work and is unchanged.
24
+
25
+ pandas remains the reference implementation and the fallback. The pyarrow path
26
+ is skipped when pyarrow is missing, when `nrows` is set, when a file is shaped
27
+ in a way it does not model, and whenever the two parsers would disagree -- the
28
+ result is the same frame either way.
29
+
30
+ ## Example
31
+
32
+ ```python
33
+ from gtfreader import read_gtf
34
+
35
+ df = read_gtf("annotation.gtf")
36
+ print(df.columns)
37
+ print(df.head())
38
+ ```
@@ -7,6 +7,13 @@ except ImportError:
7
7
  msg = "gtfreader.parse_chunk_columns requires the compiled extension. Use read_gtf_python or read_gtf_full_python for the pure Python fallback."
8
8
  raise ImportError(msg)
9
9
 
10
+ try:
11
+ from ._parser import parse_gff3_chunk_columns
12
+ except ImportError:
13
+ def parse_gff3_chunk_columns(*_args, **_kwargs):
14
+ msg = "gtfreader.parse_gff3_chunk_columns requires the compiled extension."
15
+ raise ImportError(msg)
16
+
10
17
  from .readers import (
11
18
  find_first_data_line_index,
12
19
  parse_kv_fields,
@@ -21,6 +28,7 @@ from .readers import (
21
28
  __all__ = [
22
29
  "find_first_data_line_index",
23
30
  "parse_chunk_columns",
31
+ "parse_gff3_chunk_columns",
24
32
  "parse_kv_fields",
25
33
  "read_gtf",
26
34
  "read_gtf_full",
@@ -0,0 +1,296 @@
1
+ # cython: language_level=3
2
+ # cython: boundscheck=False
3
+ # cython: wraparound=False
4
+ # cython: initializedcheck=False
5
+ # cython: infer_types=True
6
+ # cython: nonecheck=False
7
+
8
+ """Compiled attribute-column parsers for GTF and GFF3.
9
+
10
+ Both work on the UTF-8 bytes of each line rather than indexing the `str`.
11
+ Indexing a Python string goes through a width-aware read per character, which
12
+ measured 350-700 MB/s; the delimiters being looked for are all ASCII, so the
13
+ scan can run over the raw buffer instead and comparisons become `memcmp`.
14
+ CPython hands back the UTF-8 buffer of an ASCII string without copying, and
15
+ every GTF and GFF3 attribute in practice is ASCII; a line that is not still
16
+ works, because offsets stay byte offsets throughout and values are decoded back
17
+ with `PyUnicode_DecodeUTF8`.
18
+ """
19
+
20
+ from libc.string cimport memcmp
21
+
22
+ cdef extern from "Python.h":
23
+ const char* PyUnicode_AsUTF8AndSize(object unicode, Py_ssize_t *size) except NULL
24
+ object PyUnicode_DecodeUTF8(const char *s, Py_ssize_t size, const char *errors)
25
+
26
+
27
+ cdef enum:
28
+ TAB = 9
29
+ SPACE = 32
30
+ QUOTE = 34
31
+ SEMICOLON = 59
32
+ EQUALS = 61
33
+
34
+
35
+ # Values repeat: an annotation names the same gene on every one of its exons,
36
+ # and columns like gene_type hold a handful of values across the whole file.
37
+ # Handing pandas the *same* object each time rather than an equal one is worth
38
+ # 1.9x on frame construction at gencode-like cardinality, and frame
39
+ # construction is half of attribute expansion.
40
+ #
41
+ # Two levels, because they catch different shapes. The run memo compares
42
+ # against the column's previous value and allocates nothing at all on a hit,
43
+ # which is the common case in a coordinate-sorted file. The dict catches values
44
+ # that recur without being adjacent -- `tag` cycling through five -- but has to
45
+ # build the string before it can look it up, so it is the weaker of the two. It
46
+ # is capped: a column with a distinct value per row, like exon_id, would
47
+ # otherwise fill it with a million entries it can never hit.
48
+ DEF VALUE_CACHE_CAP = 8192
49
+
50
+
51
+ cdef inline bint _matches(const char* s, Py_ssize_t a, Py_ssize_t b, object other):
52
+ """Is s[a:b] equal to `other`, without building s[a:b]?"""
53
+ cdef Py_ssize_t n = b - a
54
+ cdef Py_ssize_t other_len
55
+ cdef const char* other_buf
56
+
57
+ if other is None:
58
+ return False
59
+ other_buf = PyUnicode_AsUTF8AndSize(other, &other_len)
60
+ if other_len != n:
61
+ return False
62
+ if n == 0:
63
+ return True
64
+ return memcmp(s + a, other_buf, n) == 0
65
+
66
+
67
+ cdef inline Py_ssize_t _key_index(
68
+ const char* s,
69
+ Py_ssize_t a,
70
+ Py_ssize_t b,
71
+ list keys,
72
+ Py_ssize_t ordinal,
73
+ ):
74
+ """Index of s[a:b] in `keys`, or -1. Does not build s[a:b].
75
+
76
+ A file has a handful of distinct attribute keys, repeats them on every row,
77
+ and -- this is the part worth exploiting -- lists them in the same order
78
+ every time. So the key in position `ordinal` of this row is almost always
79
+ the one that was in position `ordinal` of the last row: try that first and
80
+ the scan is one comparison, not half the table.
81
+ """
82
+ cdef Py_ssize_t i
83
+ cdef Py_ssize_t n = len(keys)
84
+
85
+ if 0 <= ordinal < n and _matches(s, a, b, <str> keys[ordinal]):
86
+ return ordinal
87
+ for i in range(n):
88
+ if _matches(s, a, b, <str> keys[i]):
89
+ return i
90
+ return -1
91
+
92
+
93
+ cdef inline object _value_at(
94
+ const char* s,
95
+ Py_ssize_t a,
96
+ Py_ssize_t b,
97
+ list last_values,
98
+ list caches,
99
+ Py_ssize_t index,
100
+ ):
101
+ """s[a:b], reusing an equal object already held for this column."""
102
+ cdef object value, cached, last
103
+ cdef dict cache
104
+
105
+ last = last_values[index]
106
+ if _matches(s, a, b, last):
107
+ return last
108
+
109
+ value = PyUnicode_DecodeUTF8(s + a, b - a, NULL)
110
+ cache = <dict> caches[index]
111
+ cached = cache.get(value)
112
+ if cached is None:
113
+ if len(cache) < VALUE_CACHE_CAP:
114
+ cache[value] = value
115
+ else:
116
+ value = cached
117
+
118
+ last_values[index] = value
119
+ return value
120
+
121
+
122
+ cdef inline Py_ssize_t _add_column(
123
+ object key,
124
+ Py_ssize_t rows,
125
+ dict columns,
126
+ list keys,
127
+ list cols,
128
+ list last_values,
129
+ list caches,
130
+ ):
131
+ cdef list col = [None] * rows
132
+
133
+ keys.append(key)
134
+ cols.append(col)
135
+ last_values.append(None)
136
+ caches.append({})
137
+ columns[key] = col
138
+ return len(keys) - 1
139
+
140
+
141
+ def parse_chunk_columns(lines):
142
+ """Expand GTF `key "value";` attribute strings into a dict of columns.
143
+
144
+ Column-oriented: one list per key, filled in place, so the caller can hand
145
+ the dict straight to pandas without building a dict per row first.
146
+ """
147
+ cdef Py_ssize_t n, row_i, m
148
+ cdef Py_ssize_t pos, key_start, key_end, val_start, val_end, index, ordinal
149
+ cdef dict columns
150
+ cdef list keys, cols, last_values, caches, col
151
+ cdef str line
152
+ cdef const char* s
153
+
154
+ n = len(lines)
155
+ columns = {}
156
+ keys = []
157
+ cols = []
158
+ last_values = []
159
+ caches = []
160
+
161
+ for row_i in range(n):
162
+ line = lines[row_i]
163
+ s = PyUnicode_AsUTF8AndSize(line, &m)
164
+ pos = 0
165
+ ordinal = 0
166
+
167
+ while pos < m:
168
+ while pos < m and (s[pos] == SPACE or s[pos] == TAB or s[pos] == SEMICOLON):
169
+ pos += 1
170
+ if pos >= m:
171
+ break
172
+
173
+ key_start = pos
174
+
175
+ while pos < m and s[pos] != SPACE and s[pos] != TAB:
176
+ pos += 1
177
+ key_end = pos
178
+
179
+ while pos < m and (s[pos] == SPACE or s[pos] == TAB):
180
+ pos += 1
181
+ if pos >= m:
182
+ break
183
+
184
+ if s[pos] == QUOTE:
185
+ val_start = pos + 1
186
+ pos += 1
187
+
188
+ while pos < m and s[pos] != QUOTE:
189
+ pos += 1
190
+ if pos >= m:
191
+ break
192
+
193
+ val_end = pos
194
+ else:
195
+ val_start = pos
196
+ while pos < m and s[pos] != SEMICOLON:
197
+ pos += 1
198
+ val_end = pos
199
+ while val_end > val_start and (s[val_end - 1] == SPACE or s[val_end - 1] == TAB):
200
+ val_end -= 1
201
+
202
+ index = _key_index(s, key_start, key_end, keys, ordinal)
203
+ if index < 0:
204
+ index = _add_column(
205
+ PyUnicode_DecodeUTF8(s + key_start, key_end - key_start, NULL),
206
+ n,
207
+ columns,
208
+ keys,
209
+ cols,
210
+ last_values,
211
+ caches,
212
+ )
213
+ col = <list> cols[index]
214
+
215
+ col[row_i] = _value_at(s, val_start, val_end, last_values, caches, index)
216
+
217
+ ordinal += 1
218
+ pos += 1
219
+
220
+ return columns
221
+
222
+
223
+ def parse_gff3_chunk_columns(lines):
224
+ """Expand GFF3 `key=value;` attribute strings into a dict of columns.
225
+
226
+ Matches `pyranges1.readers.to_keys_and_values` exactly, quirks included:
227
+
228
+ - the string is right-stripped of `;` and spaces, then split on `;`
229
+ - a segment with no `=` is skipped, which is also what makes an empty
230
+ attribute an empty row rather than an error
231
+ - the split is at the *first* `=`, so a value may contain more
232
+ - no whitespace is trimmed around a key, so `ID=a; Name=b` really does
233
+ yield a key of `" Name"`. GFF3 does not put a space there, but files do
234
+ - a repeated key keeps the last value
235
+ - columns come out in order of first appearance across the chunk
236
+
237
+ Column-oriented on purpose. Building a dict per row and handing the lot to
238
+ `DataFrame.from_records` costs as much again as the parsing does; filling
239
+ one list per column skips that entirely.
240
+ """
241
+ cdef Py_ssize_t n, row_i, m
242
+ cdef Py_ssize_t pos, end, seg_start, seg_end, eq, index, ordinal
243
+ cdef dict columns
244
+ cdef list keys, cols, last_values, caches, col
245
+ cdef str line
246
+ cdef const char* s
247
+
248
+ n = len(lines)
249
+ columns = {}
250
+ keys = []
251
+ cols = []
252
+ last_values = []
253
+ caches = []
254
+
255
+ for row_i in range(n):
256
+ line = lines[row_i]
257
+ s = PyUnicode_AsUTF8AndSize(line, &m)
258
+
259
+ # `line.rstrip("; ")`
260
+ end = m
261
+ while end > 0 and (s[end - 1] == SEMICOLON or s[end - 1] == SPACE):
262
+ end -= 1
263
+
264
+ pos = 0
265
+ ordinal = 0
266
+ while pos < end:
267
+ seg_start = pos
268
+ while pos < end and s[pos] != SEMICOLON:
269
+ pos += 1
270
+ seg_end = pos
271
+ pos += 1
272
+
273
+ eq = seg_start
274
+ while eq < seg_end and s[eq] != EQUALS:
275
+ eq += 1
276
+ if eq >= seg_end:
277
+ # No `=` in this segment: not a tag=value pair.
278
+ continue
279
+
280
+ index = _key_index(s, seg_start, eq, keys, ordinal)
281
+ if index < 0:
282
+ index = _add_column(
283
+ PyUnicode_DecodeUTF8(s + seg_start, eq - seg_start, NULL),
284
+ n,
285
+ columns,
286
+ keys,
287
+ cols,
288
+ last_values,
289
+ caches,
290
+ )
291
+ col = <list> cols[index]
292
+
293
+ col[row_i] = _value_at(s, eq + 1, seg_end, last_values, caches, index)
294
+ ordinal += 1
295
+
296
+ return columns