gtfreader 0.1.2__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gtfreader-0.2.0/PKG-INFO +35 -0
- gtfreader-0.2.0/README.md +19 -0
- {gtfreader-0.1.2 → gtfreader-0.2.0}/gtfreader/__init__.py +0 -4
- gtfreader-0.2.0/gtfreader/_parser.pyx +197 -0
- {gtfreader-0.1.2 → gtfreader-0.2.0}/gtfreader/readers.py +14 -111
- gtfreader-0.2.0/gtfreader.egg-info/PKG-INFO +35 -0
- {gtfreader-0.1.2 → gtfreader-0.2.0}/pyproject.toml +1 -1
- {gtfreader-0.1.2 → gtfreader-0.2.0}/tests/test_parser.py +8 -2
- {gtfreader-0.1.2 → gtfreader-0.2.0}/tests/test_readers.py +6 -54
- gtfreader-0.1.2/PKG-INFO +0 -71
- gtfreader-0.1.2/README.md +0 -55
- gtfreader-0.1.2/gtfreader/_parser.pyx +0 -352
- gtfreader-0.1.2/gtfreader.egg-info/PKG-INFO +0 -71
- {gtfreader-0.1.2 → gtfreader-0.2.0}/MANIFEST.in +0 -0
- {gtfreader-0.1.2 → gtfreader-0.2.0}/gtfreader.egg-info/SOURCES.txt +0 -0
- {gtfreader-0.1.2 → gtfreader-0.2.0}/gtfreader.egg-info/dependency_links.txt +0 -0
- {gtfreader-0.1.2 → gtfreader-0.2.0}/gtfreader.egg-info/requires.txt +0 -0
- {gtfreader-0.1.2 → gtfreader-0.2.0}/gtfreader.egg-info/top_level.txt +0 -0
- {gtfreader-0.1.2 → gtfreader-0.2.0}/setup.cfg +0 -0
- {gtfreader-0.1.2 → gtfreader-0.2.0}/setup.py +0 -0
gtfreader-0.2.0/PKG-INFO
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: gtfreader
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Fast Cython-backed parsing for GTF attribute columns.
|
|
5
|
+
Author: Endre Bakken Stovner
|
|
6
|
+
Classifier: Programming Language :: Python :: 3
|
|
7
|
+
Classifier: Programming Language :: Cython
|
|
8
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
9
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
11
|
+
Classifier: Operating System :: OS Independent
|
|
12
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
13
|
+
Requires-Python: >=3.12
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
Requires-Dist: pandas>=2.0
|
|
16
|
+
|
|
17
|
+
# gtfreader
|
|
18
|
+
|
|
19
|
+
Fast GTF reading into pandas DataFrames.
|
|
20
|
+
|
|
21
|
+
## Install
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
python -m pip install -e .
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
## Example
|
|
28
|
+
|
|
29
|
+
```python
|
|
30
|
+
from gtfreader import read_gtf
|
|
31
|
+
|
|
32
|
+
df = read_gtf("annotation.gtf")
|
|
33
|
+
print(df.columns)
|
|
34
|
+
print(df.head())
|
|
35
|
+
```
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
# gtfreader
|
|
2
|
+
|
|
3
|
+
Fast GTF reading into pandas DataFrames.
|
|
4
|
+
|
|
5
|
+
## Install
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
python -m pip install -e .
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
## Example
|
|
12
|
+
|
|
13
|
+
```python
|
|
14
|
+
from gtfreader import read_gtf
|
|
15
|
+
|
|
16
|
+
df = read_gtf("annotation.gtf")
|
|
17
|
+
print(df.columns)
|
|
18
|
+
print(df.head())
|
|
19
|
+
```
|
|
@@ -14,8 +14,6 @@ from .readers import (
|
|
|
14
14
|
read_gtf_full,
|
|
15
15
|
read_gtf_full_python,
|
|
16
16
|
read_gtf_python,
|
|
17
|
-
read_gtf_restricted,
|
|
18
|
-
read_gtf_restricted_python,
|
|
19
17
|
to_rows,
|
|
20
18
|
to_rows_keep_duplicates,
|
|
21
19
|
)
|
|
@@ -28,8 +26,6 @@ __all__ = [
|
|
|
28
26
|
"read_gtf_full",
|
|
29
27
|
"read_gtf_full_python",
|
|
30
28
|
"read_gtf_python",
|
|
31
|
-
"read_gtf_restricted",
|
|
32
|
-
"read_gtf_restricted_python",
|
|
33
29
|
"to_rows",
|
|
34
30
|
"to_rows_keep_duplicates",
|
|
35
31
|
]
|
|
@@ -0,0 +1,197 @@
|
|
|
1
|
+
# cython: language_level=3
|
|
2
|
+
# cython: boundscheck=False
|
|
3
|
+
# cython: wraparound=False
|
|
4
|
+
# cython: initializedcheck=False
|
|
5
|
+
# cython: infer_types=True
|
|
6
|
+
# cython: nonecheck=False
|
|
7
|
+
|
|
8
|
+
cdef inline int _key_code(str s, Py_ssize_t a, Py_ssize_t b):
|
|
9
|
+
cdef Py_ssize_t n
|
|
10
|
+
n = b - a
|
|
11
|
+
|
|
12
|
+
if n == 7:
|
|
13
|
+
if (
|
|
14
|
+
s[a] == 'g' and s[a + 1] == 'e' and s[a + 2] == 'n' and
|
|
15
|
+
s[a + 3] == 'e' and s[a + 4] == '_' and s[a + 5] == 'i' and
|
|
16
|
+
s[a + 6] == 'd'
|
|
17
|
+
):
|
|
18
|
+
return 1
|
|
19
|
+
if (
|
|
20
|
+
s[a] == 'e' and s[a + 1] == 'x' and s[a + 2] == 'o' and
|
|
21
|
+
s[a + 3] == 'n' and s[a + 4] == '_' and s[a + 5] == 'i' and
|
|
22
|
+
s[a + 6] == 'd'
|
|
23
|
+
):
|
|
24
|
+
return 6
|
|
25
|
+
|
|
26
|
+
elif n == 9:
|
|
27
|
+
if (
|
|
28
|
+
s[a] == 'g' and s[a + 1] == 'e' and s[a + 2] == 'n' and
|
|
29
|
+
s[a + 3] == 'e' and s[a + 4] == '_' and s[a + 5] == 'n' and
|
|
30
|
+
s[a + 6] == 'a' and s[a + 7] == 'm' and s[a + 8] == 'e'
|
|
31
|
+
):
|
|
32
|
+
return 3
|
|
33
|
+
|
|
34
|
+
elif n == 11:
|
|
35
|
+
if (
|
|
36
|
+
s[a] == 'e' and s[a + 1] == 'x' and s[a + 2] == 'o' and
|
|
37
|
+
s[a + 3] == 'n' and s[a + 4] == '_' and s[a + 5] == 'n' and
|
|
38
|
+
s[a + 6] == 'u' and s[a + 7] == 'm' and s[a + 8] == 'b' and
|
|
39
|
+
s[a + 9] == 'e' and s[a + 10] == 'r'
|
|
40
|
+
):
|
|
41
|
+
return 5
|
|
42
|
+
|
|
43
|
+
elif n == 13:
|
|
44
|
+
if (
|
|
45
|
+
s[a] == 't' and s[a + 1] == 'r' and s[a + 2] == 'a' and
|
|
46
|
+
s[a + 3] == 'n' and s[a + 4] == 's' and s[a + 5] == 'c' and
|
|
47
|
+
s[a + 6] == 'r' and s[a + 7] == 'i' and s[a + 8] == 'p' and
|
|
48
|
+
s[a + 9] == 't' and s[a + 10] == '_' and s[a + 11] == 'i' and
|
|
49
|
+
s[a + 12] == 'd'
|
|
50
|
+
):
|
|
51
|
+
return 2
|
|
52
|
+
|
|
53
|
+
elif n == 15:
|
|
54
|
+
if (
|
|
55
|
+
s[a] == 't' and s[a + 1] == 'r' and s[a + 2] == 'a' and
|
|
56
|
+
s[a + 3] == 'n' and s[a + 4] == 's' and s[a + 5] == 'c' and
|
|
57
|
+
s[a + 6] == 'r' and s[a + 7] == 'i' and s[a + 8] == 'p' and
|
|
58
|
+
s[a + 9] == 't' and s[a + 10] == '_' and s[a + 11] == 'n' and
|
|
59
|
+
s[a + 12] == 'a' and s[a + 13] == 'm' and s[a + 14] == 'e'
|
|
60
|
+
):
|
|
61
|
+
return 4
|
|
62
|
+
|
|
63
|
+
return 0
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
cdef inline str _cached_str(dict cache, str value):
|
|
67
|
+
cdef object obj
|
|
68
|
+
obj = cache.get(value)
|
|
69
|
+
if obj is None:
|
|
70
|
+
cache[value] = value
|
|
71
|
+
return value
|
|
72
|
+
return <str> obj
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def parse_chunk_columns(lines):
|
|
76
|
+
cdef Py_ssize_t n, row_i
|
|
77
|
+
cdef Py_ssize_t pos, m, key_start, key_end, val_start, val_end
|
|
78
|
+
cdef dict columns
|
|
79
|
+
cdef dict gene_name_cache
|
|
80
|
+
|
|
81
|
+
cdef str s, key, value
|
|
82
|
+
cdef list col
|
|
83
|
+
cdef int code
|
|
84
|
+
|
|
85
|
+
cdef object obj_col
|
|
86
|
+
|
|
87
|
+
cdef list gene_id_col
|
|
88
|
+
cdef list transcript_id_col
|
|
89
|
+
cdef list gene_name_col
|
|
90
|
+
cdef list transcript_name_col
|
|
91
|
+
cdef list exon_number_col
|
|
92
|
+
cdef list exon_id_col
|
|
93
|
+
|
|
94
|
+
n = len(lines)
|
|
95
|
+
columns = {}
|
|
96
|
+
gene_name_cache = {}
|
|
97
|
+
|
|
98
|
+
gene_id_col = None
|
|
99
|
+
transcript_id_col = None
|
|
100
|
+
gene_name_col = None
|
|
101
|
+
transcript_name_col = None
|
|
102
|
+
exon_number_col = None
|
|
103
|
+
exon_id_col = None
|
|
104
|
+
|
|
105
|
+
for row_i in range(n):
|
|
106
|
+
s = <str> lines[row_i]
|
|
107
|
+
m = len(s)
|
|
108
|
+
pos = 0
|
|
109
|
+
|
|
110
|
+
while pos < m:
|
|
111
|
+
while pos < m and (s[pos] == ' ' or s[pos] == '\t' or s[pos] == ';'):
|
|
112
|
+
pos += 1
|
|
113
|
+
if pos >= m:
|
|
114
|
+
break
|
|
115
|
+
|
|
116
|
+
key_start = pos
|
|
117
|
+
|
|
118
|
+
while pos < m and s[pos] != ' ' and s[pos] != '\t':
|
|
119
|
+
pos += 1
|
|
120
|
+
key_end = pos
|
|
121
|
+
|
|
122
|
+
while pos < m and (s[pos] == ' ' or s[pos] == '\t'):
|
|
123
|
+
pos += 1
|
|
124
|
+
if pos >= m:
|
|
125
|
+
break
|
|
126
|
+
|
|
127
|
+
if s[pos] == '"':
|
|
128
|
+
val_start = pos + 1
|
|
129
|
+
pos += 1
|
|
130
|
+
|
|
131
|
+
while pos < m and s[pos] != '"':
|
|
132
|
+
pos += 1
|
|
133
|
+
if pos >= m:
|
|
134
|
+
break
|
|
135
|
+
|
|
136
|
+
val_end = pos
|
|
137
|
+
else:
|
|
138
|
+
val_start = pos
|
|
139
|
+
while pos < m and s[pos] != ';':
|
|
140
|
+
pos += 1
|
|
141
|
+
val_end = pos
|
|
142
|
+
while val_end > val_start and (s[val_end - 1] == ' ' or s[val_end - 1] == '\t'):
|
|
143
|
+
val_end -= 1
|
|
144
|
+
|
|
145
|
+
value = s[val_start:val_end]
|
|
146
|
+
|
|
147
|
+
code = _key_code(s, key_start, key_end)
|
|
148
|
+
|
|
149
|
+
if code == 1:
|
|
150
|
+
if gene_id_col is None:
|
|
151
|
+
gene_id_col = [None] * n
|
|
152
|
+
columns["gene_id"] = gene_id_col
|
|
153
|
+
gene_id_col[row_i] = value
|
|
154
|
+
|
|
155
|
+
elif code == 2:
|
|
156
|
+
if transcript_id_col is None:
|
|
157
|
+
transcript_id_col = [None] * n
|
|
158
|
+
columns["transcript_id"] = transcript_id_col
|
|
159
|
+
transcript_id_col[row_i] = value
|
|
160
|
+
|
|
161
|
+
elif code == 3:
|
|
162
|
+
if gene_name_col is None:
|
|
163
|
+
gene_name_col = [None] * n
|
|
164
|
+
columns["gene_name"] = gene_name_col
|
|
165
|
+
gene_name_col[row_i] = _cached_str(gene_name_cache, value)
|
|
166
|
+
|
|
167
|
+
elif code == 4:
|
|
168
|
+
if transcript_name_col is None:
|
|
169
|
+
transcript_name_col = [None] * n
|
|
170
|
+
columns["transcript_name"] = transcript_name_col
|
|
171
|
+
transcript_name_col[row_i] = value
|
|
172
|
+
|
|
173
|
+
elif code == 5:
|
|
174
|
+
if exon_number_col is None:
|
|
175
|
+
exon_number_col = [None] * n
|
|
176
|
+
columns["exon_number"] = exon_number_col
|
|
177
|
+
exon_number_col[row_i] = value
|
|
178
|
+
|
|
179
|
+
elif code == 6:
|
|
180
|
+
if exon_id_col is None:
|
|
181
|
+
exon_id_col = [None] * n
|
|
182
|
+
columns["exon_id"] = exon_id_col
|
|
183
|
+
exon_id_col[row_i] = value
|
|
184
|
+
|
|
185
|
+
else:
|
|
186
|
+
key = s[key_start:key_end]
|
|
187
|
+
obj_col = columns.get(key)
|
|
188
|
+
if obj_col is None:
|
|
189
|
+
col = [None] * n
|
|
190
|
+
columns[key] = col
|
|
191
|
+
else:
|
|
192
|
+
col = <list> obj_col
|
|
193
|
+
col[row_i] = value
|
|
194
|
+
|
|
195
|
+
pos += 1
|
|
196
|
+
|
|
197
|
+
return columns
|
|
@@ -23,7 +23,6 @@ GTF_DTYPES = {
|
|
|
23
23
|
"Frame": "category",
|
|
24
24
|
}
|
|
25
25
|
GTF_NAMES = ["Chromosome", "Source", "Feature", "Start", "End", "Score", "Strand", "Frame", "Attribute"]
|
|
26
|
-
RESTRICTED_ATTRIBUTE_COLUMNS = ["gene_id", "transcript_id", "exon_number", "exon_id"]
|
|
27
26
|
|
|
28
27
|
|
|
29
28
|
def find_first_data_line_index(file_path: str | Path) -> int:
|
|
@@ -205,22 +204,6 @@ def _parse_attributes_python(
|
|
|
205
204
|
return to_rows(attribute_column, ignore_bad=ignore_bad)
|
|
206
205
|
|
|
207
206
|
|
|
208
|
-
def _parse_restricted_attributes_compiled(attribute_column: pd.Series) -> pd.DataFrame:
|
|
209
|
-
if _parse_chunk_columns_compiled is None:
|
|
210
|
-
return _parse_restricted_attributes_python(attribute_column)
|
|
211
|
-
|
|
212
|
-
attribute_column = _normalize_attribute_series(attribute_column)
|
|
213
|
-
columns = _parse_chunk_columns_compiled(attribute_column.to_numpy(copy=False))
|
|
214
|
-
return pd.DataFrame(
|
|
215
|
-
{name: columns[name] for name in RESTRICTED_ATTRIBUTE_COLUMNS},
|
|
216
|
-
index=attribute_column.index,
|
|
217
|
-
)
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
def _parse_restricted_attributes_python(attribute_column: pd.Series) -> pd.DataFrame:
|
|
221
|
-
return to_rows(_normalize_attribute_series(attribute_column)).reindex(columns=RESTRICTED_ATTRIBUTE_COLUMNS)
|
|
222
|
-
|
|
223
|
-
|
|
224
207
|
def _read_gtf_full(
|
|
225
208
|
path: Path,
|
|
226
209
|
*,
|
|
@@ -243,55 +226,24 @@ def _read_gtf_full(
|
|
|
243
226
|
return _finalize_gtf_frame(pd.concat(dfs, sort=False))
|
|
244
227
|
|
|
245
228
|
|
|
246
|
-
def _read_gtf_restricted(
|
|
247
|
-
path: Path,
|
|
248
|
-
*,
|
|
249
|
-
skiprows: int,
|
|
250
|
-
nrows: int | None,
|
|
251
|
-
chunksize: int,
|
|
252
|
-
parse_attributes,
|
|
253
|
-
) -> pd.DataFrame:
|
|
254
|
-
dfs: list[pd.DataFrame] = []
|
|
255
|
-
with _open_gtf_reader(path, chunksize=chunksize, skiprows=skiprows, nrows=nrows) as df_iter:
|
|
256
|
-
for df in df_iter:
|
|
257
|
-
subset = parse_attributes(df["Attribute"])
|
|
258
|
-
dfs.append(
|
|
259
|
-
pd.concat(
|
|
260
|
-
[df[["Chromosome", "Source", "Feature", "Start", "End", "Score", "Strand", "Frame"]], subset],
|
|
261
|
-
axis=1,
|
|
262
|
-
sort=False,
|
|
263
|
-
)
|
|
264
|
-
)
|
|
265
|
-
|
|
266
|
-
if not dfs:
|
|
267
|
-
return pd.DataFrame(columns=GTF_NAMES[:-1] + RESTRICTED_ATTRIBUTE_COLUMNS)
|
|
268
|
-
|
|
269
|
-
return _finalize_gtf_frame(pd.concat(dfs, sort=False))
|
|
270
|
-
|
|
271
|
-
|
|
272
229
|
def read_gtf(
|
|
273
230
|
f: str | Path,
|
|
274
231
|
/,
|
|
275
232
|
*,
|
|
276
233
|
nrows: int | None = None,
|
|
277
|
-
full: bool = True,
|
|
278
234
|
duplicate_attr: bool = False,
|
|
279
235
|
ignore_bad: bool = False,
|
|
280
236
|
) -> pd.DataFrame:
|
|
281
237
|
"""Read a GTF file using the compiled parser path when available."""
|
|
282
238
|
path = Path(f)
|
|
283
239
|
skiprows = find_first_data_line_index(path)
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
ignore_bad=ignore_bad,
|
|
292
|
-
)
|
|
293
|
-
|
|
294
|
-
return read_gtf_restricted(path, skiprows=skiprows, nrows=nrows)
|
|
240
|
+
return read_gtf_full(
|
|
241
|
+
path,
|
|
242
|
+
nrows=nrows,
|
|
243
|
+
skiprows=skiprows,
|
|
244
|
+
duplicate_attr=duplicate_attr,
|
|
245
|
+
ignore_bad=ignore_bad,
|
|
246
|
+
)
|
|
295
247
|
|
|
296
248
|
|
|
297
249
|
def read_gtf_python(
|
|
@@ -299,24 +251,19 @@ def read_gtf_python(
|
|
|
299
251
|
/,
|
|
300
252
|
*,
|
|
301
253
|
nrows: int | None = None,
|
|
302
|
-
full: bool = True,
|
|
303
254
|
duplicate_attr: bool = False,
|
|
304
255
|
ignore_bad: bool = False,
|
|
305
256
|
) -> pd.DataFrame:
|
|
306
257
|
"""Read a GTF file using the pure Python attribute parser."""
|
|
307
258
|
path = Path(f)
|
|
308
259
|
skiprows = find_first_data_line_index(path)
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
ignore_bad=ignore_bad,
|
|
317
|
-
)
|
|
318
|
-
|
|
319
|
-
return read_gtf_restricted_python(path, skiprows=skiprows, nrows=nrows)
|
|
260
|
+
return read_gtf_full_python(
|
|
261
|
+
path,
|
|
262
|
+
nrows=nrows,
|
|
263
|
+
skiprows=skiprows,
|
|
264
|
+
duplicate_attr=duplicate_attr,
|
|
265
|
+
ignore_bad=ignore_bad,
|
|
266
|
+
)
|
|
320
267
|
|
|
321
268
|
|
|
322
269
|
def read_gtf_full(
|
|
@@ -369,48 +316,6 @@ def read_gtf_full_python(
|
|
|
369
316
|
)
|
|
370
317
|
|
|
371
318
|
|
|
372
|
-
def read_gtf_restricted(
|
|
373
|
-
f: str | Path,
|
|
374
|
-
/,
|
|
375
|
-
skiprows: int = 0,
|
|
376
|
-
nrows: int | None = None,
|
|
377
|
-
chunksize: int = int(1e5),
|
|
378
|
-
*,
|
|
379
|
-
chunk_size: int | None = None,
|
|
380
|
-
) -> pd.DataFrame:
|
|
381
|
-
"""Read core GTF columns plus a small compiled-parser attribute subset."""
|
|
382
|
-
path = Path(f)
|
|
383
|
-
chunksize = _resolve_chunksize(chunksize, chunk_size)
|
|
384
|
-
return _read_gtf_restricted(
|
|
385
|
-
path,
|
|
386
|
-
skiprows=skiprows,
|
|
387
|
-
nrows=nrows,
|
|
388
|
-
chunksize=chunksize,
|
|
389
|
-
parse_attributes=_parse_restricted_attributes_compiled,
|
|
390
|
-
)
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
def read_gtf_restricted_python(
|
|
394
|
-
f: str | Path,
|
|
395
|
-
/,
|
|
396
|
-
skiprows: int = 0,
|
|
397
|
-
nrows: int | None = None,
|
|
398
|
-
chunksize: int = int(1e5),
|
|
399
|
-
*,
|
|
400
|
-
chunk_size: int | None = None,
|
|
401
|
-
) -> pd.DataFrame:
|
|
402
|
-
"""Read core GTF columns plus a small pure-Python attribute subset."""
|
|
403
|
-
path = Path(f)
|
|
404
|
-
chunksize = _resolve_chunksize(chunksize, chunk_size)
|
|
405
|
-
return _read_gtf_restricted(
|
|
406
|
-
path,
|
|
407
|
-
skiprows=skiprows,
|
|
408
|
-
nrows=nrows,
|
|
409
|
-
chunksize=chunksize,
|
|
410
|
-
parse_attributes=_parse_restricted_attributes_python,
|
|
411
|
-
)
|
|
412
|
-
|
|
413
|
-
|
|
414
319
|
__all__ = [
|
|
415
320
|
"find_first_data_line_index",
|
|
416
321
|
"parse_kv_fields",
|
|
@@ -418,8 +323,6 @@ __all__ = [
|
|
|
418
323
|
"read_gtf_full",
|
|
419
324
|
"read_gtf_full_python",
|
|
420
325
|
"read_gtf_python",
|
|
421
|
-
"read_gtf_restricted",
|
|
422
|
-
"read_gtf_restricted_python",
|
|
423
326
|
"to_rows",
|
|
424
327
|
"to_rows_keep_duplicates",
|
|
425
328
|
]
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: gtfreader
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Fast Cython-backed parsing for GTF attribute columns.
|
|
5
|
+
Author: Endre Bakken Stovner
|
|
6
|
+
Classifier: Programming Language :: Python :: 3
|
|
7
|
+
Classifier: Programming Language :: Cython
|
|
8
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
9
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
11
|
+
Classifier: Operating System :: OS Independent
|
|
12
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
13
|
+
Requires-Python: >=3.12
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
Requires-Dist: pandas>=2.0
|
|
16
|
+
|
|
17
|
+
# gtfreader
|
|
18
|
+
|
|
19
|
+
Fast GTF reading into pandas DataFrames.
|
|
20
|
+
|
|
21
|
+
## Install
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
python -m pip install -e .
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
## Example
|
|
28
|
+
|
|
29
|
+
```python
|
|
30
|
+
from gtfreader import read_gtf
|
|
31
|
+
|
|
32
|
+
df = read_gtf("annotation.gtf")
|
|
33
|
+
print(df.columns)
|
|
34
|
+
print(df.head())
|
|
35
|
+
```
|
|
@@ -25,8 +25,14 @@ def test_known_and_dynamic_columns_are_parsed():
|
|
|
25
25
|
assert columns["exon_number"] == ["1", None]
|
|
26
26
|
assert columns["custom_key"] == ["custom", None]
|
|
27
27
|
assert columns["gene_name"][0] is columns["gene_name"][1]
|
|
28
|
-
assert columns["gene_type"]
|
|
29
|
-
assert columns["level"]
|
|
28
|
+
assert columns["gene_type"] == ["protein_coding", "protein_coding"]
|
|
29
|
+
assert columns["level"] == ["2", "2"]
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def test_compiled_parser_omits_unseen_fast_path_columns():
|
|
33
|
+
columns = parse_chunk_columns(['gene_id "GENE1"; transcript_id "TX1";'])
|
|
34
|
+
|
|
35
|
+
assert set(columns) == {"gene_id", "transcript_id"}
|
|
30
36
|
|
|
31
37
|
|
|
32
38
|
def test_compiled_parser_supports_semicolons_in_quotes_and_unquoted_values():
|
|
@@ -11,8 +11,6 @@ from gtfreader import (
|
|
|
11
11
|
read_gtf_full,
|
|
12
12
|
read_gtf_full_python,
|
|
13
13
|
read_gtf_python,
|
|
14
|
-
read_gtf_restricted,
|
|
15
|
-
read_gtf_restricted_python,
|
|
16
14
|
)
|
|
17
15
|
|
|
18
16
|
|
|
@@ -170,15 +168,14 @@ def test_read_gtf_full_multiple_chunks_match(tmp_path: Path):
|
|
|
170
168
|
assert_frame_equal(chunked, unchunked, check_dtype=False)
|
|
171
169
|
|
|
172
170
|
|
|
173
|
-
def
|
|
171
|
+
def test_read_gtf_omits_unseen_fast_path_columns(tmp_path: Path):
|
|
174
172
|
path = _write_temp_gtf(
|
|
175
173
|
tmp_path,
|
|
176
174
|
"# header\n"
|
|
177
|
-
'chr1\thavana\
|
|
175
|
+
'chr1\thavana\tgene\t12010\t12057\t.\t+\t.\tgene_id "ENSG1"; gene_name "DDX11L1";\n',
|
|
178
176
|
)
|
|
179
177
|
|
|
180
|
-
|
|
181
|
-
result = read_gtf_restricted(path, skiprows=skiprows)
|
|
178
|
+
result = read_gtf(path)
|
|
182
179
|
|
|
183
180
|
assert list(result.columns) == [
|
|
184
181
|
"Chromosome",
|
|
@@ -190,33 +187,10 @@ def test_read_gtf_restricted_returns_core_columns(tmp_path: Path):
|
|
|
190
187
|
"Strand",
|
|
191
188
|
"Frame",
|
|
192
189
|
"gene_id",
|
|
193
|
-
"
|
|
194
|
-
"exon_number",
|
|
195
|
-
"exon_id",
|
|
190
|
+
"gene_name",
|
|
196
191
|
]
|
|
197
192
|
assert result.iloc[0]["Start"] == 12009
|
|
198
|
-
assert result.iloc[0]["
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
@pytest.mark.parametrize(
|
|
202
|
-
"reader",
|
|
203
|
-
[read_gtf_restricted, read_gtf_restricted_python],
|
|
204
|
-
ids=["default", "python"],
|
|
205
|
-
)
|
|
206
|
-
def test_restricted_readers_support_unquoted_values(tmp_path: Path, reader):
|
|
207
|
-
path = _write_temp_gtf(
|
|
208
|
-
tmp_path,
|
|
209
|
-
"# header\n"
|
|
210
|
-
'chr1\tensembl_havana\texon\t3069203\t3069296\t.\t+\t.\tgene_id "G1"; transcript_id "T1"; exon_number 4; exon_id "EX4";\n',
|
|
211
|
-
)
|
|
212
|
-
|
|
213
|
-
skiprows = find_first_data_line_index(path)
|
|
214
|
-
result = _normalize_frame_for_compare(reader(path, skiprows=skiprows))
|
|
215
|
-
|
|
216
|
-
assert result.iloc[0]["gene_id"] == "G1"
|
|
217
|
-
assert result.iloc[0]["transcript_id"] == "T1"
|
|
218
|
-
assert result.iloc[0]["exon_number"] == "4"
|
|
219
|
-
assert result.iloc[0]["exon_id"] == "EX4"
|
|
193
|
+
assert result.iloc[0]["gene_name"] == "DDX11L1"
|
|
220
194
|
|
|
221
195
|
|
|
222
196
|
def test_read_gtf_python_matches_compiled_reader(tmp_path: Path):
|
|
@@ -234,12 +208,7 @@ def test_read_gtf_python_matches_compiled_reader(tmp_path: Path):
|
|
|
234
208
|
compiled = _normalize_frame_for_compare(compiled)
|
|
235
209
|
python = _normalize_frame_for_compare(python)
|
|
236
210
|
|
|
237
|
-
|
|
238
|
-
assert list(python["gene_name"]) == ["DDX11L1", None]
|
|
239
|
-
assert list(python["transcript_id"]) == [None, "ENST1"]
|
|
240
|
-
assert list(python["transcript_name"]) == [None, "DDX11L1-201"]
|
|
241
|
-
assert list(python["tag"]) == [None, "basic"]
|
|
242
|
-
assert list(python["Start"]) == list(compiled["Start"])
|
|
211
|
+
assert_frame_equal(compiled, python, check_dtype=False)
|
|
243
212
|
|
|
244
213
|
|
|
245
214
|
def test_read_gtf_python_duplicate_attr_keeps_all_values(tmp_path: Path):
|
|
@@ -254,20 +223,3 @@ def test_read_gtf_python_duplicate_attr_keeps_all_values(tmp_path: Path):
|
|
|
254
223
|
|
|
255
224
|
assert compiled.iloc[0]["tag"] == "CCDS,basic"
|
|
256
225
|
assert python.iloc[0]["tag"] == "CCDS,basic"
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
def test_read_gtf_restricted_python_matches_compiled(tmp_path: Path):
|
|
260
|
-
path = _write_temp_gtf(
|
|
261
|
-
tmp_path,
|
|
262
|
-
"# header\n"
|
|
263
|
-
'chr1\thavana\texon\t12010\t12057\t.\t+\t.\tgene_id "ENSG1"; transcript_id "ENST1"; exon_number "1"; exon_id "ENSE1";\n',
|
|
264
|
-
)
|
|
265
|
-
|
|
266
|
-
skiprows = find_first_data_line_index(path)
|
|
267
|
-
compiled = read_gtf_restricted(path, skiprows=skiprows)
|
|
268
|
-
python = read_gtf_restricted_python(path, skiprows=skiprows)
|
|
269
|
-
|
|
270
|
-
compiled = _normalize_frame_for_compare(compiled)
|
|
271
|
-
python = _normalize_frame_for_compare(python)
|
|
272
|
-
|
|
273
|
-
assert_frame_equal(compiled, python, check_dtype=False)
|
gtfreader-0.1.2/PKG-INFO
DELETED
|
@@ -1,71 +0,0 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: gtfreader
|
|
3
|
-
Version: 0.1.2
|
|
4
|
-
Summary: Fast Cython-backed parsing for GTF attribute columns.
|
|
5
|
-
Author: Endre Bakken Stovner
|
|
6
|
-
Classifier: Programming Language :: Python :: 3
|
|
7
|
-
Classifier: Programming Language :: Cython
|
|
8
|
-
Classifier: Programming Language :: Python :: 3 :: Only
|
|
9
|
-
Classifier: Programming Language :: Python :: 3.12
|
|
10
|
-
Classifier: Programming Language :: Python :: 3.13
|
|
11
|
-
Classifier: Operating System :: OS Independent
|
|
12
|
-
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
13
|
-
Requires-Python: >=3.12
|
|
14
|
-
Description-Content-Type: text/markdown
|
|
15
|
-
Requires-Dist: pandas>=2.0
|
|
16
|
-
|
|
17
|
-
# gtfreader
|
|
18
|
-
|
|
19
|
-
`gtfreader` is a small package for parsing and reading GTF files into pandas dataframes.
|
|
20
|
-
|
|
21
|
-
Requires Python 3.12 or newer.
|
|
22
|
-
|
|
23
|
-
## Install
|
|
24
|
-
|
|
25
|
-
```bash
|
|
26
|
-
python -m pip install -e .
|
|
27
|
-
```
|
|
28
|
-
|
|
29
|
-
## Usage
|
|
30
|
-
|
|
31
|
-
```python
|
|
32
|
-
from gtfreader import read_gtf, read_gtf_python
|
|
33
|
-
|
|
34
|
-
df = read_gtf("annotation.gtf")
|
|
35
|
-
df_python = read_gtf_python("annotation.gtf")
|
|
36
|
-
```
|
|
37
|
-
|
|
38
|
-
`read_gtf(...)` uses the compiled parser path when available. `read_gtf_python(...)` uses the high-level pure Python parser path used by the current `pyrunges` reader style.
|
|
39
|
-
|
|
40
|
-
If you want to use the compiled low-level parser directly, pass it raw attribute strings from column 9 of the GTF before they have been expanded:
|
|
41
|
-
|
|
42
|
-
```python
|
|
43
|
-
import pandas as pd
|
|
44
|
-
|
|
45
|
-
from gtfreader import find_first_data_line_index, parse_chunk_columns
|
|
46
|
-
|
|
47
|
-
skiprows = find_first_data_line_index("annotation.gtf")
|
|
48
|
-
attribute_lines = pd.read_csv(
|
|
49
|
-
"annotation.gtf",
|
|
50
|
-
sep="\t",
|
|
51
|
-
header=None,
|
|
52
|
-
usecols=[8],
|
|
53
|
-
names=["Attribute"],
|
|
54
|
-
comment="#",
|
|
55
|
-
skiprows=skiprows,
|
|
56
|
-
)["Attribute"].tolist()
|
|
57
|
-
|
|
58
|
-
compiled_columns = parse_chunk_columns(attribute_lines)
|
|
59
|
-
```
|
|
60
|
-
|
|
61
|
-
## Build
|
|
62
|
-
|
|
63
|
-
```bash
|
|
64
|
-
python -m build
|
|
65
|
-
```
|
|
66
|
-
|
|
67
|
-
## Test
|
|
68
|
-
|
|
69
|
-
```bash
|
|
70
|
-
python -m pytest -q
|
|
71
|
-
```
|
gtfreader-0.1.2/README.md
DELETED
|
@@ -1,55 +0,0 @@
|
|
|
1
|
-
# gtfreader
|
|
2
|
-
|
|
3
|
-
`gtfreader` is a small package for parsing and reading GTF files into pandas dataframes.
|
|
4
|
-
|
|
5
|
-
Requires Python 3.12 or newer.
|
|
6
|
-
|
|
7
|
-
## Install
|
|
8
|
-
|
|
9
|
-
```bash
|
|
10
|
-
python -m pip install -e .
|
|
11
|
-
```
|
|
12
|
-
|
|
13
|
-
## Usage
|
|
14
|
-
|
|
15
|
-
```python
|
|
16
|
-
from gtfreader import read_gtf, read_gtf_python
|
|
17
|
-
|
|
18
|
-
df = read_gtf("annotation.gtf")
|
|
19
|
-
df_python = read_gtf_python("annotation.gtf")
|
|
20
|
-
```
|
|
21
|
-
|
|
22
|
-
`read_gtf(...)` uses the compiled parser path when available. `read_gtf_python(...)` uses the high-level pure Python parser path used by the current `pyrunges` reader style.
|
|
23
|
-
|
|
24
|
-
If you want to use the compiled low-level parser directly, pass it raw attribute strings from column 9 of the GTF before they have been expanded:
|
|
25
|
-
|
|
26
|
-
```python
|
|
27
|
-
import pandas as pd
|
|
28
|
-
|
|
29
|
-
from gtfreader import find_first_data_line_index, parse_chunk_columns
|
|
30
|
-
|
|
31
|
-
skiprows = find_first_data_line_index("annotation.gtf")
|
|
32
|
-
attribute_lines = pd.read_csv(
|
|
33
|
-
"annotation.gtf",
|
|
34
|
-
sep="\t",
|
|
35
|
-
header=None,
|
|
36
|
-
usecols=[8],
|
|
37
|
-
names=["Attribute"],
|
|
38
|
-
comment="#",
|
|
39
|
-
skiprows=skiprows,
|
|
40
|
-
)["Attribute"].tolist()
|
|
41
|
-
|
|
42
|
-
compiled_columns = parse_chunk_columns(attribute_lines)
|
|
43
|
-
```
|
|
44
|
-
|
|
45
|
-
## Build
|
|
46
|
-
|
|
47
|
-
```bash
|
|
48
|
-
python -m build
|
|
49
|
-
```
|
|
50
|
-
|
|
51
|
-
## Test
|
|
52
|
-
|
|
53
|
-
```bash
|
|
54
|
-
python -m pytest -q
|
|
55
|
-
```
|
|
@@ -1,352 +0,0 @@
|
|
|
1
|
-
# cython: language_level=3
|
|
2
|
-
# cython: boundscheck=False
|
|
3
|
-
# cython: wraparound=False
|
|
4
|
-
# cython: initializedcheck=False
|
|
5
|
-
# cython: infer_types=True
|
|
6
|
-
# cython: nonecheck=False
|
|
7
|
-
|
|
8
|
-
cdef inline int _key_code(str s, Py_ssize_t a, Py_ssize_t b):
|
|
9
|
-
cdef Py_ssize_t n
|
|
10
|
-
n = b - a
|
|
11
|
-
|
|
12
|
-
if n == 3:
|
|
13
|
-
if s[a] == 't' and s[a + 1] == 'a' and s[a + 2] == 'g':
|
|
14
|
-
return 9
|
|
15
|
-
if s[a] == 'o' and s[a + 1] == 'n' and s[a + 2] == 't':
|
|
16
|
-
return 17
|
|
17
|
-
|
|
18
|
-
elif n == 5:
|
|
19
|
-
if (
|
|
20
|
-
s[a] == 'l' and s[a + 1] == 'e' and s[a + 2] == 'v' and
|
|
21
|
-
s[a + 3] == 'e' and s[a + 4] == 'l'
|
|
22
|
-
):
|
|
23
|
-
return 16
|
|
24
|
-
|
|
25
|
-
elif n == 6:
|
|
26
|
-
if (
|
|
27
|
-
s[a] == 'c' and s[a + 1] == 'c' and s[a + 2] == 'd' and
|
|
28
|
-
s[a + 3] == 's' and s[a + 4] == 'i' and s[a + 5] == 'd'
|
|
29
|
-
):
|
|
30
|
-
return 14
|
|
31
|
-
|
|
32
|
-
elif n == 7:
|
|
33
|
-
if (
|
|
34
|
-
s[a] == 'g' and s[a + 1] == 'e' and s[a + 2] == 'n' and
|
|
35
|
-
s[a + 3] == 'e' and s[a + 4] == '_' and s[a + 5] == 'i' and
|
|
36
|
-
s[a + 6] == 'd'
|
|
37
|
-
):
|
|
38
|
-
return 1
|
|
39
|
-
if (
|
|
40
|
-
s[a] == 'e' and s[a + 1] == 'x' and s[a + 2] == 'o' and
|
|
41
|
-
s[a + 3] == 'n' and s[a + 4] == '_' and s[a + 5] == 'i' and
|
|
42
|
-
s[a + 6] == 'd'
|
|
43
|
-
):
|
|
44
|
-
return 8
|
|
45
|
-
if (
|
|
46
|
-
s[a] == 'h' and s[a + 1] == 'g' and s[a + 2] == 'n' and
|
|
47
|
-
s[a + 3] == 'c' and s[a + 4] == '_' and s[a + 5] == 'i' and
|
|
48
|
-
s[a + 6] == 'd'
|
|
49
|
-
):
|
|
50
|
-
return 13
|
|
51
|
-
|
|
52
|
-
elif n == 9:
|
|
53
|
-
if (
|
|
54
|
-
s[a] == 'g' and s[a + 1] == 'e' and s[a + 2] == 'n' and
|
|
55
|
-
s[a + 3] == 'e' and s[a + 4] == '_'
|
|
56
|
-
):
|
|
57
|
-
if (
|
|
58
|
-
s[a + 5] == 'n' and s[a + 6] == 'a' and
|
|
59
|
-
s[a + 7] == 'm' and s[a + 8] == 'e'
|
|
60
|
-
):
|
|
61
|
-
return 3
|
|
62
|
-
if (
|
|
63
|
-
s[a + 5] == 't' and s[a + 6] == 'y' and
|
|
64
|
-
s[a + 7] == 'p' and s[a + 8] == 'e'
|
|
65
|
-
):
|
|
66
|
-
return 4
|
|
67
|
-
|
|
68
|
-
elif n == 10:
|
|
69
|
-
if (
|
|
70
|
-
s[a] == 'a' and s[a + 1] == 'r' and s[a + 2] == 't' and
|
|
71
|
-
s[a + 3] == 'i' and s[a + 4] == 'f' and s[a + 5] == '_' and
|
|
72
|
-
s[a + 6] == 'd' and s[a + 7] == 'u' and s[a + 8] == 'p' and
|
|
73
|
-
s[a + 9] == 'l'
|
|
74
|
-
):
|
|
75
|
-
return 15
|
|
76
|
-
|
|
77
|
-
elif n == 11:
|
|
78
|
-
if (
|
|
79
|
-
s[a] == 'e' and s[a + 1] == 'x' and s[a + 2] == 'o' and
|
|
80
|
-
s[a + 3] == 'n' and s[a + 4] == '_' and s[a + 5] == 'n' and
|
|
81
|
-
s[a + 6] == 'u' and s[a + 7] == 'm' and s[a + 8] == 'b' and
|
|
82
|
-
s[a + 9] == 'e' and s[a + 10] == 'r'
|
|
83
|
-
):
|
|
84
|
-
return 7
|
|
85
|
-
if (
|
|
86
|
-
s[a] == 'h' and s[a + 1] == 'a' and s[a + 2] == 'v' and
|
|
87
|
-
s[a + 3] == 'a' and s[a + 4] == 'n' and s[a + 5] == 'a' and
|
|
88
|
-
s[a + 6] == '_' and s[a + 7] == 'g' and s[a + 8] == 'e' and
|
|
89
|
-
s[a + 9] == 'n' and s[a + 10] == 'e'
|
|
90
|
-
):
|
|
91
|
-
return 11
|
|
92
|
-
|
|
93
|
-
elif n == 13:
|
|
94
|
-
if (
|
|
95
|
-
s[a] == 't' and s[a + 1] == 'r' and s[a + 2] == 'a' and
|
|
96
|
-
s[a + 3] == 'n' and s[a + 4] == 's' and s[a + 5] == 'c' and
|
|
97
|
-
s[a + 6] == 'r' and s[a + 7] == 'i' and s[a + 8] == 'p' and
|
|
98
|
-
s[a + 9] == 't' and s[a + 10] == '_' and s[a + 11] == 'i' and
|
|
99
|
-
s[a + 12] == 'd'
|
|
100
|
-
):
|
|
101
|
-
return 2
|
|
102
|
-
|
|
103
|
-
elif n == 15:
|
|
104
|
-
if (
|
|
105
|
-
s[a] == 't' and s[a + 1] == 'r' and s[a + 2] == 'a' and
|
|
106
|
-
s[a + 3] == 'n' and s[a + 4] == 's' and s[a + 5] == 'c' and
|
|
107
|
-
s[a + 6] == 'r' and s[a + 7] == 'i' and s[a + 8] == 'p' and
|
|
108
|
-
s[a + 9] == 't' and s[a + 10] == '_'
|
|
109
|
-
):
|
|
110
|
-
if (
|
|
111
|
-
s[a + 11] == 'n' and s[a + 12] == 'a' and
|
|
112
|
-
s[a + 13] == 'm' and s[a + 14] == 'e'
|
|
113
|
-
):
|
|
114
|
-
return 5
|
|
115
|
-
if (
|
|
116
|
-
s[a + 11] == 't' and s[a + 12] == 'y' and
|
|
117
|
-
s[a + 13] == 'p' and s[a + 14] == 'e'
|
|
118
|
-
):
|
|
119
|
-
return 6
|
|
120
|
-
|
|
121
|
-
elif n == 17:
|
|
122
|
-
if (
|
|
123
|
-
s[a] == 'h' and s[a + 1] == 'a' and s[a + 2] == 'v' and
|
|
124
|
-
s[a + 3] == 'a' and s[a + 4] == 'n' and s[a + 5] == 'a' and
|
|
125
|
-
s[a + 6] == '_' and s[a + 7] == 't' and s[a + 8] == 'r' and
|
|
126
|
-
s[a + 9] == 'a' and s[a + 10] == 'n' and s[a + 11] == 's' and
|
|
127
|
-
s[a + 12] == 'c' and s[a + 13] == 'r' and s[a + 14] == 'i' and
|
|
128
|
-
s[a + 15] == 'p' and s[a + 16] == 't'
|
|
129
|
-
):
|
|
130
|
-
return 10
|
|
131
|
-
|
|
132
|
-
elif n == 24:
|
|
133
|
-
if (
|
|
134
|
-
s[a] == 't' and s[a + 1] == 'r' and s[a + 2] == 'a' and
|
|
135
|
-
s[a + 3] == 'n' and s[a + 4] == 's' and s[a + 5] == 'c' and
|
|
136
|
-
s[a + 6] == 'r' and s[a + 7] == 'i' and s[a + 8] == 'p' and
|
|
137
|
-
s[a + 9] == 't' and s[a + 10] == '_' and s[a + 11] == 's' and
|
|
138
|
-
s[a + 12] == 'u' and s[a + 13] == 'p' and s[a + 14] == 'p' and
|
|
139
|
-
s[a + 15] == 'o' and s[a + 16] == 'r' and s[a + 17] == 't' and
|
|
140
|
-
s[a + 18] == '_' and s[a + 19] == 'l' and s[a + 20] == 'e' and
|
|
141
|
-
s[a + 21] == 'v' and s[a + 22] == 'e' and s[a + 23] == 'l'
|
|
142
|
-
):
|
|
143
|
-
return 12
|
|
144
|
-
|
|
145
|
-
return 0
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
cdef inline str _cached_str(dict cache, str value):
|
|
149
|
-
cdef object obj
|
|
150
|
-
obj = cache.get(value)
|
|
151
|
-
if obj is None:
|
|
152
|
-
cache[value] = value
|
|
153
|
-
return value
|
|
154
|
-
return <str> obj
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
def parse_chunk_columns(lines):
|
|
158
|
-
cdef Py_ssize_t n, row_i
|
|
159
|
-
cdef Py_ssize_t pos, m, key_start, key_end, val_start, val_end
|
|
160
|
-
cdef dict columns
|
|
161
|
-
|
|
162
|
-
cdef dict gene_name_cache
|
|
163
|
-
cdef dict gene_type_cache
|
|
164
|
-
cdef dict transcript_type_cache
|
|
165
|
-
cdef dict tag_cache
|
|
166
|
-
cdef dict level_cache
|
|
167
|
-
cdef dict ont_cache
|
|
168
|
-
cdef dict transcript_support_level_cache
|
|
169
|
-
cdef dict artif_dupl_cache
|
|
170
|
-
|
|
171
|
-
cdef str s, key, value
|
|
172
|
-
cdef list col
|
|
173
|
-
cdef int code
|
|
174
|
-
|
|
175
|
-
cdef object obj_col
|
|
176
|
-
|
|
177
|
-
cdef list gene_id_col
|
|
178
|
-
cdef list transcript_id_col
|
|
179
|
-
cdef list gene_name_col
|
|
180
|
-
cdef list gene_type_col
|
|
181
|
-
cdef list transcript_name_col
|
|
182
|
-
cdef list transcript_type_col
|
|
183
|
-
cdef list exon_number_col
|
|
184
|
-
cdef list exon_id_col
|
|
185
|
-
cdef list tag_col
|
|
186
|
-
cdef list havana_transcript_col
|
|
187
|
-
cdef list havana_gene_col
|
|
188
|
-
cdef list transcript_support_level_col
|
|
189
|
-
cdef list hgnc_id_col
|
|
190
|
-
cdef list ccdsid_col
|
|
191
|
-
cdef list artif_dupl_col
|
|
192
|
-
cdef list level_col
|
|
193
|
-
cdef list ont_col
|
|
194
|
-
|
|
195
|
-
n = len(lines)
|
|
196
|
-
columns = {}
|
|
197
|
-
|
|
198
|
-
gene_name_cache = {}
|
|
199
|
-
gene_type_cache = {}
|
|
200
|
-
transcript_type_cache = {}
|
|
201
|
-
tag_cache = {}
|
|
202
|
-
level_cache = {}
|
|
203
|
-
ont_cache = {}
|
|
204
|
-
transcript_support_level_cache = {}
|
|
205
|
-
artif_dupl_cache = {}
|
|
206
|
-
|
|
207
|
-
gene_id_col = [None] * n
|
|
208
|
-
transcript_id_col = [None] * n
|
|
209
|
-
gene_name_col = [None] * n
|
|
210
|
-
gene_type_col = [None] * n
|
|
211
|
-
transcript_name_col = [None] * n
|
|
212
|
-
transcript_type_col = [None] * n
|
|
213
|
-
exon_number_col = [None] * n
|
|
214
|
-
exon_id_col = [None] * n
|
|
215
|
-
tag_col = [None] * n
|
|
216
|
-
havana_transcript_col = [None] * n
|
|
217
|
-
havana_gene_col = [None] * n
|
|
218
|
-
transcript_support_level_col = [None] * n
|
|
219
|
-
hgnc_id_col = [None] * n
|
|
220
|
-
ccdsid_col = [None] * n
|
|
221
|
-
artif_dupl_col = [None] * n
|
|
222
|
-
level_col = [None] * n
|
|
223
|
-
ont_col = [None] * n
|
|
224
|
-
|
|
225
|
-
columns["gene_id"] = gene_id_col
|
|
226
|
-
columns["transcript_id"] = transcript_id_col
|
|
227
|
-
columns["gene_name"] = gene_name_col
|
|
228
|
-
columns["gene_type"] = gene_type_col
|
|
229
|
-
columns["transcript_name"] = transcript_name_col
|
|
230
|
-
columns["transcript_type"] = transcript_type_col
|
|
231
|
-
columns["exon_number"] = exon_number_col
|
|
232
|
-
columns["exon_id"] = exon_id_col
|
|
233
|
-
columns["tag"] = tag_col
|
|
234
|
-
columns["havana_transcript"] = havana_transcript_col
|
|
235
|
-
columns["havana_gene"] = havana_gene_col
|
|
236
|
-
columns["transcript_support_level"] = transcript_support_level_col
|
|
237
|
-
columns["hgnc_id"] = hgnc_id_col
|
|
238
|
-
columns["ccdsid"] = ccdsid_col
|
|
239
|
-
columns["artif_dupl"] = artif_dupl_col
|
|
240
|
-
columns["level"] = level_col
|
|
241
|
-
columns["ont"] = ont_col
|
|
242
|
-
|
|
243
|
-
for row_i in range(n):
|
|
244
|
-
s = <str> lines[row_i]
|
|
245
|
-
m = len(s)
|
|
246
|
-
pos = 0
|
|
247
|
-
|
|
248
|
-
while pos < m:
|
|
249
|
-
while pos < m and (s[pos] == ' ' or s[pos] == '\t' or s[pos] == ';'):
|
|
250
|
-
pos += 1
|
|
251
|
-
if pos >= m:
|
|
252
|
-
break
|
|
253
|
-
|
|
254
|
-
key_start = pos
|
|
255
|
-
|
|
256
|
-
while pos < m and s[pos] != ' ' and s[pos] != '\t':
|
|
257
|
-
pos += 1
|
|
258
|
-
key_end = pos
|
|
259
|
-
|
|
260
|
-
while pos < m and (s[pos] == ' ' or s[pos] == '\t'):
|
|
261
|
-
pos += 1
|
|
262
|
-
if pos >= m:
|
|
263
|
-
break
|
|
264
|
-
|
|
265
|
-
if s[pos] == '"':
|
|
266
|
-
val_start = pos + 1
|
|
267
|
-
pos += 1
|
|
268
|
-
|
|
269
|
-
while pos < m and s[pos] != '"':
|
|
270
|
-
pos += 1
|
|
271
|
-
if pos >= m:
|
|
272
|
-
break
|
|
273
|
-
|
|
274
|
-
val_end = pos
|
|
275
|
-
else:
|
|
276
|
-
val_start = pos
|
|
277
|
-
while pos < m and s[pos] != ';':
|
|
278
|
-
pos += 1
|
|
279
|
-
val_end = pos
|
|
280
|
-
while val_end > val_start and (s[val_end - 1] == ' ' or s[val_end - 1] == '\t'):
|
|
281
|
-
val_end -= 1
|
|
282
|
-
|
|
283
|
-
value = s[val_start:val_end]
|
|
284
|
-
|
|
285
|
-
code = _key_code(s, key_start, key_end)
|
|
286
|
-
|
|
287
|
-
if code == 1:
|
|
288
|
-
gene_id_col[row_i] = value
|
|
289
|
-
|
|
290
|
-
elif code == 2:
|
|
291
|
-
transcript_id_col[row_i] = value
|
|
292
|
-
|
|
293
|
-
elif code == 3:
|
|
294
|
-
gene_name_col[row_i] = _cached_str(gene_name_cache, value)
|
|
295
|
-
|
|
296
|
-
elif code == 4:
|
|
297
|
-
gene_type_col[row_i] = _cached_str(gene_type_cache, value)
|
|
298
|
-
|
|
299
|
-
elif code == 5:
|
|
300
|
-
transcript_name_col[row_i] = value
|
|
301
|
-
|
|
302
|
-
elif code == 6:
|
|
303
|
-
transcript_type_col[row_i] = _cached_str(transcript_type_cache, value)
|
|
304
|
-
|
|
305
|
-
elif code == 7:
|
|
306
|
-
exon_number_col[row_i] = value
|
|
307
|
-
|
|
308
|
-
elif code == 8:
|
|
309
|
-
exon_id_col[row_i] = value
|
|
310
|
-
|
|
311
|
-
elif code == 9:
|
|
312
|
-
tag_col[row_i] = _cached_str(tag_cache, value)
|
|
313
|
-
|
|
314
|
-
elif code == 10:
|
|
315
|
-
havana_transcript_col[row_i] = value
|
|
316
|
-
|
|
317
|
-
elif code == 11:
|
|
318
|
-
havana_gene_col[row_i] = value
|
|
319
|
-
|
|
320
|
-
elif code == 12:
|
|
321
|
-
transcript_support_level_col[row_i] = _cached_str(
|
|
322
|
-
transcript_support_level_cache, value
|
|
323
|
-
)
|
|
324
|
-
|
|
325
|
-
elif code == 13:
|
|
326
|
-
hgnc_id_col[row_i] = value
|
|
327
|
-
|
|
328
|
-
elif code == 14:
|
|
329
|
-
ccdsid_col[row_i] = value
|
|
330
|
-
|
|
331
|
-
elif code == 15:
|
|
332
|
-
artif_dupl_col[row_i] = _cached_str(artif_dupl_cache, value)
|
|
333
|
-
|
|
334
|
-
elif code == 16:
|
|
335
|
-
level_col[row_i] = _cached_str(level_cache, value)
|
|
336
|
-
|
|
337
|
-
elif code == 17:
|
|
338
|
-
ont_col[row_i] = _cached_str(ont_cache, value)
|
|
339
|
-
|
|
340
|
-
else:
|
|
341
|
-
key = s[key_start:key_end]
|
|
342
|
-
obj_col = columns.get(key)
|
|
343
|
-
if obj_col is None:
|
|
344
|
-
col = [None] * n
|
|
345
|
-
columns[key] = col
|
|
346
|
-
else:
|
|
347
|
-
col = <list> obj_col
|
|
348
|
-
col[row_i] = value
|
|
349
|
-
|
|
350
|
-
pos += 1
|
|
351
|
-
|
|
352
|
-
return columns
|
|
@@ -1,71 +0,0 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: gtfreader
|
|
3
|
-
Version: 0.1.2
|
|
4
|
-
Summary: Fast Cython-backed parsing for GTF attribute columns.
|
|
5
|
-
Author: Endre Bakken Stovner
|
|
6
|
-
Classifier: Programming Language :: Python :: 3
|
|
7
|
-
Classifier: Programming Language :: Cython
|
|
8
|
-
Classifier: Programming Language :: Python :: 3 :: Only
|
|
9
|
-
Classifier: Programming Language :: Python :: 3.12
|
|
10
|
-
Classifier: Programming Language :: Python :: 3.13
|
|
11
|
-
Classifier: Operating System :: OS Independent
|
|
12
|
-
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
13
|
-
Requires-Python: >=3.12
|
|
14
|
-
Description-Content-Type: text/markdown
|
|
15
|
-
Requires-Dist: pandas>=2.0
|
|
16
|
-
|
|
17
|
-
# gtfreader
|
|
18
|
-
|
|
19
|
-
`gtfreader` is a small package for parsing and reading GTF files into pandas dataframes.
|
|
20
|
-
|
|
21
|
-
Requires Python 3.12 or newer.
|
|
22
|
-
|
|
23
|
-
## Install
|
|
24
|
-
|
|
25
|
-
```bash
|
|
26
|
-
python -m pip install -e .
|
|
27
|
-
```
|
|
28
|
-
|
|
29
|
-
## Usage
|
|
30
|
-
|
|
31
|
-
```python
|
|
32
|
-
from gtfreader import read_gtf, read_gtf_python
|
|
33
|
-
|
|
34
|
-
df = read_gtf("annotation.gtf")
|
|
35
|
-
df_python = read_gtf_python("annotation.gtf")
|
|
36
|
-
```
|
|
37
|
-
|
|
38
|
-
`read_gtf(...)` uses the compiled parser path when available. `read_gtf_python(...)` uses the high-level pure Python parser path used by the current `pyrunges` reader style.
|
|
39
|
-
|
|
40
|
-
If you want to use the compiled low-level parser directly, pass it raw attribute strings from column 9 of the GTF before they have been expanded:
|
|
41
|
-
|
|
42
|
-
```python
|
|
43
|
-
import pandas as pd
|
|
44
|
-
|
|
45
|
-
from gtfreader import find_first_data_line_index, parse_chunk_columns
|
|
46
|
-
|
|
47
|
-
skiprows = find_first_data_line_index("annotation.gtf")
|
|
48
|
-
attribute_lines = pd.read_csv(
|
|
49
|
-
"annotation.gtf",
|
|
50
|
-
sep="\t",
|
|
51
|
-
header=None,
|
|
52
|
-
usecols=[8],
|
|
53
|
-
names=["Attribute"],
|
|
54
|
-
comment="#",
|
|
55
|
-
skiprows=skiprows,
|
|
56
|
-
)["Attribute"].tolist()
|
|
57
|
-
|
|
58
|
-
compiled_columns = parse_chunk_columns(attribute_lines)
|
|
59
|
-
```
|
|
60
|
-
|
|
61
|
-
## Build
|
|
62
|
-
|
|
63
|
-
```bash
|
|
64
|
-
python -m build
|
|
65
|
-
```
|
|
66
|
-
|
|
67
|
-
## Test
|
|
68
|
-
|
|
69
|
-
```bash
|
|
70
|
-
python -m pytest -q
|
|
71
|
-
```
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|