gtfreader 0.1.2__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,35 @@
1
+ Metadata-Version: 2.4
2
+ Name: gtfreader
3
+ Version: 0.2.0
4
+ Summary: Fast Cython-backed parsing for GTF attribute columns.
5
+ Author: Endre Bakken Stovner
6
+ Classifier: Programming Language :: Python :: 3
7
+ Classifier: Programming Language :: Cython
8
+ Classifier: Programming Language :: Python :: 3 :: Only
9
+ Classifier: Programming Language :: Python :: 3.12
10
+ Classifier: Programming Language :: Python :: 3.13
11
+ Classifier: Operating System :: OS Independent
12
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
13
+ Requires-Python: >=3.12
14
+ Description-Content-Type: text/markdown
15
+ Requires-Dist: pandas>=2.0
16
+
17
+ # gtfreader
18
+
19
+ Fast GTF reading into pandas DataFrames.
20
+
21
+ ## Install
22
+
23
+ ```bash
24
+ python -m pip install -e .
25
+ ```
26
+
27
+ ## Example
28
+
29
+ ```python
30
+ from gtfreader import read_gtf
31
+
32
+ df = read_gtf("annotation.gtf")
33
+ print(df.columns)
34
+ print(df.head())
35
+ ```
@@ -0,0 +1,19 @@
1
+ # gtfreader
2
+
3
+ Fast GTF reading into pandas DataFrames.
4
+
5
+ ## Install
6
+
7
+ ```bash
8
+ python -m pip install -e .
9
+ ```
10
+
11
+ ## Example
12
+
13
+ ```python
14
+ from gtfreader import read_gtf
15
+
16
+ df = read_gtf("annotation.gtf")
17
+ print(df.columns)
18
+ print(df.head())
19
+ ```
@@ -14,8 +14,6 @@ from .readers import (
14
14
  read_gtf_full,
15
15
  read_gtf_full_python,
16
16
  read_gtf_python,
17
- read_gtf_restricted,
18
- read_gtf_restricted_python,
19
17
  to_rows,
20
18
  to_rows_keep_duplicates,
21
19
  )
@@ -28,8 +26,6 @@ __all__ = [
28
26
  "read_gtf_full",
29
27
  "read_gtf_full_python",
30
28
  "read_gtf_python",
31
- "read_gtf_restricted",
32
- "read_gtf_restricted_python",
33
29
  "to_rows",
34
30
  "to_rows_keep_duplicates",
35
31
  ]
@@ -0,0 +1,197 @@
1
+ # cython: language_level=3
2
+ # cython: boundscheck=False
3
+ # cython: wraparound=False
4
+ # cython: initializedcheck=False
5
+ # cython: infer_types=True
6
+ # cython: nonecheck=False
7
+
8
+ cdef inline int _key_code(str s, Py_ssize_t a, Py_ssize_t b):
9
+ cdef Py_ssize_t n
10
+ n = b - a
11
+
12
+ if n == 7:
13
+ if (
14
+ s[a] == 'g' and s[a + 1] == 'e' and s[a + 2] == 'n' and
15
+ s[a + 3] == 'e' and s[a + 4] == '_' and s[a + 5] == 'i' and
16
+ s[a + 6] == 'd'
17
+ ):
18
+ return 1
19
+ if (
20
+ s[a] == 'e' and s[a + 1] == 'x' and s[a + 2] == 'o' and
21
+ s[a + 3] == 'n' and s[a + 4] == '_' and s[a + 5] == 'i' and
22
+ s[a + 6] == 'd'
23
+ ):
24
+ return 6
25
+
26
+ elif n == 9:
27
+ if (
28
+ s[a] == 'g' and s[a + 1] == 'e' and s[a + 2] == 'n' and
29
+ s[a + 3] == 'e' and s[a + 4] == '_' and s[a + 5] == 'n' and
30
+ s[a + 6] == 'a' and s[a + 7] == 'm' and s[a + 8] == 'e'
31
+ ):
32
+ return 3
33
+
34
+ elif n == 11:
35
+ if (
36
+ s[a] == 'e' and s[a + 1] == 'x' and s[a + 2] == 'o' and
37
+ s[a + 3] == 'n' and s[a + 4] == '_' and s[a + 5] == 'n' and
38
+ s[a + 6] == 'u' and s[a + 7] == 'm' and s[a + 8] == 'b' and
39
+ s[a + 9] == 'e' and s[a + 10] == 'r'
40
+ ):
41
+ return 5
42
+
43
+ elif n == 13:
44
+ if (
45
+ s[a] == 't' and s[a + 1] == 'r' and s[a + 2] == 'a' and
46
+ s[a + 3] == 'n' and s[a + 4] == 's' and s[a + 5] == 'c' and
47
+ s[a + 6] == 'r' and s[a + 7] == 'i' and s[a + 8] == 'p' and
48
+ s[a + 9] == 't' and s[a + 10] == '_' and s[a + 11] == 'i' and
49
+ s[a + 12] == 'd'
50
+ ):
51
+ return 2
52
+
53
+ elif n == 15:
54
+ if (
55
+ s[a] == 't' and s[a + 1] == 'r' and s[a + 2] == 'a' and
56
+ s[a + 3] == 'n' and s[a + 4] == 's' and s[a + 5] == 'c' and
57
+ s[a + 6] == 'r' and s[a + 7] == 'i' and s[a + 8] == 'p' and
58
+ s[a + 9] == 't' and s[a + 10] == '_' and s[a + 11] == 'n' and
59
+ s[a + 12] == 'a' and s[a + 13] == 'm' and s[a + 14] == 'e'
60
+ ):
61
+ return 4
62
+
63
+ return 0
64
+
65
+
66
+ cdef inline str _cached_str(dict cache, str value):
67
+ cdef object obj
68
+ obj = cache.get(value)
69
+ if obj is None:
70
+ cache[value] = value
71
+ return value
72
+ return <str> obj
73
+
74
+
75
+ def parse_chunk_columns(lines):
76
+ cdef Py_ssize_t n, row_i
77
+ cdef Py_ssize_t pos, m, key_start, key_end, val_start, val_end
78
+ cdef dict columns
79
+ cdef dict gene_name_cache
80
+
81
+ cdef str s, key, value
82
+ cdef list col
83
+ cdef int code
84
+
85
+ cdef object obj_col
86
+
87
+ cdef list gene_id_col
88
+ cdef list transcript_id_col
89
+ cdef list gene_name_col
90
+ cdef list transcript_name_col
91
+ cdef list exon_number_col
92
+ cdef list exon_id_col
93
+
94
+ n = len(lines)
95
+ columns = {}
96
+ gene_name_cache = {}
97
+
98
+ gene_id_col = None
99
+ transcript_id_col = None
100
+ gene_name_col = None
101
+ transcript_name_col = None
102
+ exon_number_col = None
103
+ exon_id_col = None
104
+
105
+ for row_i in range(n):
106
+ s = <str> lines[row_i]
107
+ m = len(s)
108
+ pos = 0
109
+
110
+ while pos < m:
111
+ while pos < m and (s[pos] == ' ' or s[pos] == '\t' or s[pos] == ';'):
112
+ pos += 1
113
+ if pos >= m:
114
+ break
115
+
116
+ key_start = pos
117
+
118
+ while pos < m and s[pos] != ' ' and s[pos] != '\t':
119
+ pos += 1
120
+ key_end = pos
121
+
122
+ while pos < m and (s[pos] == ' ' or s[pos] == '\t'):
123
+ pos += 1
124
+ if pos >= m:
125
+ break
126
+
127
+ if s[pos] == '"':
128
+ val_start = pos + 1
129
+ pos += 1
130
+
131
+ while pos < m and s[pos] != '"':
132
+ pos += 1
133
+ if pos >= m:
134
+ break
135
+
136
+ val_end = pos
137
+ else:
138
+ val_start = pos
139
+ while pos < m and s[pos] != ';':
140
+ pos += 1
141
+ val_end = pos
142
+ while val_end > val_start and (s[val_end - 1] == ' ' or s[val_end - 1] == '\t'):
143
+ val_end -= 1
144
+
145
+ value = s[val_start:val_end]
146
+
147
+ code = _key_code(s, key_start, key_end)
148
+
149
+ if code == 1:
150
+ if gene_id_col is None:
151
+ gene_id_col = [None] * n
152
+ columns["gene_id"] = gene_id_col
153
+ gene_id_col[row_i] = value
154
+
155
+ elif code == 2:
156
+ if transcript_id_col is None:
157
+ transcript_id_col = [None] * n
158
+ columns["transcript_id"] = transcript_id_col
159
+ transcript_id_col[row_i] = value
160
+
161
+ elif code == 3:
162
+ if gene_name_col is None:
163
+ gene_name_col = [None] * n
164
+ columns["gene_name"] = gene_name_col
165
+ gene_name_col[row_i] = _cached_str(gene_name_cache, value)
166
+
167
+ elif code == 4:
168
+ if transcript_name_col is None:
169
+ transcript_name_col = [None] * n
170
+ columns["transcript_name"] = transcript_name_col
171
+ transcript_name_col[row_i] = value
172
+
173
+ elif code == 5:
174
+ if exon_number_col is None:
175
+ exon_number_col = [None] * n
176
+ columns["exon_number"] = exon_number_col
177
+ exon_number_col[row_i] = value
178
+
179
+ elif code == 6:
180
+ if exon_id_col is None:
181
+ exon_id_col = [None] * n
182
+ columns["exon_id"] = exon_id_col
183
+ exon_id_col[row_i] = value
184
+
185
+ else:
186
+ key = s[key_start:key_end]
187
+ obj_col = columns.get(key)
188
+ if obj_col is None:
189
+ col = [None] * n
190
+ columns[key] = col
191
+ else:
192
+ col = <list> obj_col
193
+ col[row_i] = value
194
+
195
+ pos += 1
196
+
197
+ return columns
@@ -23,7 +23,6 @@ GTF_DTYPES = {
23
23
  "Frame": "category",
24
24
  }
25
25
  GTF_NAMES = ["Chromosome", "Source", "Feature", "Start", "End", "Score", "Strand", "Frame", "Attribute"]
26
- RESTRICTED_ATTRIBUTE_COLUMNS = ["gene_id", "transcript_id", "exon_number", "exon_id"]
27
26
 
28
27
 
29
28
  def find_first_data_line_index(file_path: str | Path) -> int:
@@ -205,22 +204,6 @@ def _parse_attributes_python(
205
204
  return to_rows(attribute_column, ignore_bad=ignore_bad)
206
205
 
207
206
 
208
- def _parse_restricted_attributes_compiled(attribute_column: pd.Series) -> pd.DataFrame:
209
- if _parse_chunk_columns_compiled is None:
210
- return _parse_restricted_attributes_python(attribute_column)
211
-
212
- attribute_column = _normalize_attribute_series(attribute_column)
213
- columns = _parse_chunk_columns_compiled(attribute_column.to_numpy(copy=False))
214
- return pd.DataFrame(
215
- {name: columns[name] for name in RESTRICTED_ATTRIBUTE_COLUMNS},
216
- index=attribute_column.index,
217
- )
218
-
219
-
220
- def _parse_restricted_attributes_python(attribute_column: pd.Series) -> pd.DataFrame:
221
- return to_rows(_normalize_attribute_series(attribute_column)).reindex(columns=RESTRICTED_ATTRIBUTE_COLUMNS)
222
-
223
-
224
207
  def _read_gtf_full(
225
208
  path: Path,
226
209
  *,
@@ -243,55 +226,24 @@ def _read_gtf_full(
243
226
  return _finalize_gtf_frame(pd.concat(dfs, sort=False))
244
227
 
245
228
 
246
- def _read_gtf_restricted(
247
- path: Path,
248
- *,
249
- skiprows: int,
250
- nrows: int | None,
251
- chunksize: int,
252
- parse_attributes,
253
- ) -> pd.DataFrame:
254
- dfs: list[pd.DataFrame] = []
255
- with _open_gtf_reader(path, chunksize=chunksize, skiprows=skiprows, nrows=nrows) as df_iter:
256
- for df in df_iter:
257
- subset = parse_attributes(df["Attribute"])
258
- dfs.append(
259
- pd.concat(
260
- [df[["Chromosome", "Source", "Feature", "Start", "End", "Score", "Strand", "Frame"]], subset],
261
- axis=1,
262
- sort=False,
263
- )
264
- )
265
-
266
- if not dfs:
267
- return pd.DataFrame(columns=GTF_NAMES[:-1] + RESTRICTED_ATTRIBUTE_COLUMNS)
268
-
269
- return _finalize_gtf_frame(pd.concat(dfs, sort=False))
270
-
271
-
272
229
  def read_gtf(
273
230
  f: str | Path,
274
231
  /,
275
232
  *,
276
233
  nrows: int | None = None,
277
- full: bool = True,
278
234
  duplicate_attr: bool = False,
279
235
  ignore_bad: bool = False,
280
236
  ) -> pd.DataFrame:
281
237
  """Read a GTF file using the compiled parser path when available."""
282
238
  path = Path(f)
283
239
  skiprows = find_first_data_line_index(path)
284
-
285
- if full:
286
- return read_gtf_full(
287
- path,
288
- nrows=nrows,
289
- skiprows=skiprows,
290
- duplicate_attr=duplicate_attr,
291
- ignore_bad=ignore_bad,
292
- )
293
-
294
- return read_gtf_restricted(path, skiprows=skiprows, nrows=nrows)
240
+ return read_gtf_full(
241
+ path,
242
+ nrows=nrows,
243
+ skiprows=skiprows,
244
+ duplicate_attr=duplicate_attr,
245
+ ignore_bad=ignore_bad,
246
+ )
295
247
 
296
248
 
297
249
  def read_gtf_python(
@@ -299,24 +251,19 @@ def read_gtf_python(
299
251
  /,
300
252
  *,
301
253
  nrows: int | None = None,
302
- full: bool = True,
303
254
  duplicate_attr: bool = False,
304
255
  ignore_bad: bool = False,
305
256
  ) -> pd.DataFrame:
306
257
  """Read a GTF file using the pure Python attribute parser."""
307
258
  path = Path(f)
308
259
  skiprows = find_first_data_line_index(path)
309
-
310
- if full:
311
- return read_gtf_full_python(
312
- path,
313
- nrows=nrows,
314
- skiprows=skiprows,
315
- duplicate_attr=duplicate_attr,
316
- ignore_bad=ignore_bad,
317
- )
318
-
319
- return read_gtf_restricted_python(path, skiprows=skiprows, nrows=nrows)
260
+ return read_gtf_full_python(
261
+ path,
262
+ nrows=nrows,
263
+ skiprows=skiprows,
264
+ duplicate_attr=duplicate_attr,
265
+ ignore_bad=ignore_bad,
266
+ )
320
267
 
321
268
 
322
269
  def read_gtf_full(
@@ -369,48 +316,6 @@ def read_gtf_full_python(
369
316
  )
370
317
 
371
318
 
372
- def read_gtf_restricted(
373
- f: str | Path,
374
- /,
375
- skiprows: int = 0,
376
- nrows: int | None = None,
377
- chunksize: int = int(1e5),
378
- *,
379
- chunk_size: int | None = None,
380
- ) -> pd.DataFrame:
381
- """Read core GTF columns plus a small compiled-parser attribute subset."""
382
- path = Path(f)
383
- chunksize = _resolve_chunksize(chunksize, chunk_size)
384
- return _read_gtf_restricted(
385
- path,
386
- skiprows=skiprows,
387
- nrows=nrows,
388
- chunksize=chunksize,
389
- parse_attributes=_parse_restricted_attributes_compiled,
390
- )
391
-
392
-
393
- def read_gtf_restricted_python(
394
- f: str | Path,
395
- /,
396
- skiprows: int = 0,
397
- nrows: int | None = None,
398
- chunksize: int = int(1e5),
399
- *,
400
- chunk_size: int | None = None,
401
- ) -> pd.DataFrame:
402
- """Read core GTF columns plus a small pure-Python attribute subset."""
403
- path = Path(f)
404
- chunksize = _resolve_chunksize(chunksize, chunk_size)
405
- return _read_gtf_restricted(
406
- path,
407
- skiprows=skiprows,
408
- nrows=nrows,
409
- chunksize=chunksize,
410
- parse_attributes=_parse_restricted_attributes_python,
411
- )
412
-
413
-
414
319
  __all__ = [
415
320
  "find_first_data_line_index",
416
321
  "parse_kv_fields",
@@ -418,8 +323,6 @@ __all__ = [
418
323
  "read_gtf_full",
419
324
  "read_gtf_full_python",
420
325
  "read_gtf_python",
421
- "read_gtf_restricted",
422
- "read_gtf_restricted_python",
423
326
  "to_rows",
424
327
  "to_rows_keep_duplicates",
425
328
  ]
@@ -0,0 +1,35 @@
1
+ Metadata-Version: 2.4
2
+ Name: gtfreader
3
+ Version: 0.2.0
4
+ Summary: Fast Cython-backed parsing for GTF attribute columns.
5
+ Author: Endre Bakken Stovner
6
+ Classifier: Programming Language :: Python :: 3
7
+ Classifier: Programming Language :: Cython
8
+ Classifier: Programming Language :: Python :: 3 :: Only
9
+ Classifier: Programming Language :: Python :: 3.12
10
+ Classifier: Programming Language :: Python :: 3.13
11
+ Classifier: Operating System :: OS Independent
12
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
13
+ Requires-Python: >=3.12
14
+ Description-Content-Type: text/markdown
15
+ Requires-Dist: pandas>=2.0
16
+
17
+ # gtfreader
18
+
19
+ Fast GTF reading into pandas DataFrames.
20
+
21
+ ## Install
22
+
23
+ ```bash
24
+ python -m pip install -e .
25
+ ```
26
+
27
+ ## Example
28
+
29
+ ```python
30
+ from gtfreader import read_gtf
31
+
32
+ df = read_gtf("annotation.gtf")
33
+ print(df.columns)
34
+ print(df.head())
35
+ ```
@@ -8,7 +8,7 @@ build-backend = "setuptools.build_meta"
8
8
 
9
9
  [project]
10
10
  name = "gtfreader"
11
- version = "0.1.2"
11
+ version = "0.2.0"
12
12
  description = "Fast Cython-backed parsing for GTF attribute columns."
13
13
  readme = "README.md"
14
14
  requires-python = ">=3.12"
@@ -25,8 +25,14 @@ def test_known_and_dynamic_columns_are_parsed():
25
25
  assert columns["exon_number"] == ["1", None]
26
26
  assert columns["custom_key"] == ["custom", None]
27
27
  assert columns["gene_name"][0] is columns["gene_name"][1]
28
- assert columns["gene_type"][0] is columns["gene_type"][1]
29
- assert columns["level"][0] is columns["level"][1]
28
+ assert columns["gene_type"] == ["protein_coding", "protein_coding"]
29
+ assert columns["level"] == ["2", "2"]
30
+
31
+
32
+ def test_compiled_parser_omits_unseen_fast_path_columns():
33
+ columns = parse_chunk_columns(['gene_id "GENE1"; transcript_id "TX1";'])
34
+
35
+ assert set(columns) == {"gene_id", "transcript_id"}
30
36
 
31
37
 
32
38
  def test_compiled_parser_supports_semicolons_in_quotes_and_unquoted_values():
@@ -11,8 +11,6 @@ from gtfreader import (
11
11
  read_gtf_full,
12
12
  read_gtf_full_python,
13
13
  read_gtf_python,
14
- read_gtf_restricted,
15
- read_gtf_restricted_python,
16
14
  )
17
15
 
18
16
 
@@ -170,15 +168,14 @@ def test_read_gtf_full_multiple_chunks_match(tmp_path: Path):
170
168
  assert_frame_equal(chunked, unchunked, check_dtype=False)
171
169
 
172
170
 
173
- def test_read_gtf_restricted_returns_core_columns(tmp_path: Path):
171
+ def test_read_gtf_omits_unseen_fast_path_columns(tmp_path: Path):
174
172
  path = _write_temp_gtf(
175
173
  tmp_path,
176
174
  "# header\n"
177
- 'chr1\thavana\texon\t12010\t12057\t.\t+\t.\tgene_id "ENSG1"; transcript_id "ENST1"; exon_number "1"; exon_id "ENSE1";\n',
175
+ 'chr1\thavana\tgene\t12010\t12057\t.\t+\t.\tgene_id "ENSG1"; gene_name "DDX11L1";\n',
178
176
  )
179
177
 
180
- skiprows = find_first_data_line_index(path)
181
- result = read_gtf_restricted(path, skiprows=skiprows)
178
+ result = read_gtf(path)
182
179
 
183
180
  assert list(result.columns) == [
184
181
  "Chromosome",
@@ -190,33 +187,10 @@ def test_read_gtf_restricted_returns_core_columns(tmp_path: Path):
190
187
  "Strand",
191
188
  "Frame",
192
189
  "gene_id",
193
- "transcript_id",
194
- "exon_number",
195
- "exon_id",
190
+ "gene_name",
196
191
  ]
197
192
  assert result.iloc[0]["Start"] == 12009
198
- assert result.iloc[0]["transcript_id"] == "ENST1"
199
-
200
-
201
- @pytest.mark.parametrize(
202
- "reader",
203
- [read_gtf_restricted, read_gtf_restricted_python],
204
- ids=["default", "python"],
205
- )
206
- def test_restricted_readers_support_unquoted_values(tmp_path: Path, reader):
207
- path = _write_temp_gtf(
208
- tmp_path,
209
- "# header\n"
210
- 'chr1\tensembl_havana\texon\t3069203\t3069296\t.\t+\t.\tgene_id "G1"; transcript_id "T1"; exon_number 4; exon_id "EX4";\n',
211
- )
212
-
213
- skiprows = find_first_data_line_index(path)
214
- result = _normalize_frame_for_compare(reader(path, skiprows=skiprows))
215
-
216
- assert result.iloc[0]["gene_id"] == "G1"
217
- assert result.iloc[0]["transcript_id"] == "T1"
218
- assert result.iloc[0]["exon_number"] == "4"
219
- assert result.iloc[0]["exon_id"] == "EX4"
193
+ assert result.iloc[0]["gene_name"] == "DDX11L1"
220
194
 
221
195
 
222
196
  def test_read_gtf_python_matches_compiled_reader(tmp_path: Path):
@@ -234,12 +208,7 @@ def test_read_gtf_python_matches_compiled_reader(tmp_path: Path):
234
208
  compiled = _normalize_frame_for_compare(compiled)
235
209
  python = _normalize_frame_for_compare(python)
236
210
 
237
- assert list(python["gene_id"]) == list(compiled["gene_id"])
238
- assert list(python["gene_name"]) == ["DDX11L1", None]
239
- assert list(python["transcript_id"]) == [None, "ENST1"]
240
- assert list(python["transcript_name"]) == [None, "DDX11L1-201"]
241
- assert list(python["tag"]) == [None, "basic"]
242
- assert list(python["Start"]) == list(compiled["Start"])
211
+ assert_frame_equal(compiled, python, check_dtype=False)
243
212
 
244
213
 
245
214
  def test_read_gtf_python_duplicate_attr_keeps_all_values(tmp_path: Path):
@@ -254,20 +223,3 @@ def test_read_gtf_python_duplicate_attr_keeps_all_values(tmp_path: Path):
254
223
 
255
224
  assert compiled.iloc[0]["tag"] == "CCDS,basic"
256
225
  assert python.iloc[0]["tag"] == "CCDS,basic"
257
-
258
-
259
- def test_read_gtf_restricted_python_matches_compiled(tmp_path: Path):
260
- path = _write_temp_gtf(
261
- tmp_path,
262
- "# header\n"
263
- 'chr1\thavana\texon\t12010\t12057\t.\t+\t.\tgene_id "ENSG1"; transcript_id "ENST1"; exon_number "1"; exon_id "ENSE1";\n',
264
- )
265
-
266
- skiprows = find_first_data_line_index(path)
267
- compiled = read_gtf_restricted(path, skiprows=skiprows)
268
- python = read_gtf_restricted_python(path, skiprows=skiprows)
269
-
270
- compiled = _normalize_frame_for_compare(compiled)
271
- python = _normalize_frame_for_compare(python)
272
-
273
- assert_frame_equal(compiled, python, check_dtype=False)
gtfreader-0.1.2/PKG-INFO DELETED
@@ -1,71 +0,0 @@
1
- Metadata-Version: 2.4
2
- Name: gtfreader
3
- Version: 0.1.2
4
- Summary: Fast Cython-backed parsing for GTF attribute columns.
5
- Author: Endre Bakken Stovner
6
- Classifier: Programming Language :: Python :: 3
7
- Classifier: Programming Language :: Cython
8
- Classifier: Programming Language :: Python :: 3 :: Only
9
- Classifier: Programming Language :: Python :: 3.12
10
- Classifier: Programming Language :: Python :: 3.13
11
- Classifier: Operating System :: OS Independent
12
- Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
13
- Requires-Python: >=3.12
14
- Description-Content-Type: text/markdown
15
- Requires-Dist: pandas>=2.0
16
-
17
- # gtfreader
18
-
19
- `gtfreader` is a small package for parsing and reading GTF files into pandas dataframes.
20
-
21
- Requires Python 3.12 or newer.
22
-
23
- ## Install
24
-
25
- ```bash
26
- python -m pip install -e .
27
- ```
28
-
29
- ## Usage
30
-
31
- ```python
32
- from gtfreader import read_gtf, read_gtf_python
33
-
34
- df = read_gtf("annotation.gtf")
35
- df_python = read_gtf_python("annotation.gtf")
36
- ```
37
-
38
- `read_gtf(...)` uses the compiled parser path when available. `read_gtf_python(...)` uses the high-level pure Python parser path used by the current `pyrunges` reader style.
39
-
40
- If you want to use the compiled low-level parser directly, pass it raw attribute strings from column 9 of the GTF before they have been expanded:
41
-
42
- ```python
43
- import pandas as pd
44
-
45
- from gtfreader import find_first_data_line_index, parse_chunk_columns
46
-
47
- skiprows = find_first_data_line_index("annotation.gtf")
48
- attribute_lines = pd.read_csv(
49
- "annotation.gtf",
50
- sep="\t",
51
- header=None,
52
- usecols=[8],
53
- names=["Attribute"],
54
- comment="#",
55
- skiprows=skiprows,
56
- )["Attribute"].tolist()
57
-
58
- compiled_columns = parse_chunk_columns(attribute_lines)
59
- ```
60
-
61
- ## Build
62
-
63
- ```bash
64
- python -m build
65
- ```
66
-
67
- ## Test
68
-
69
- ```bash
70
- python -m pytest -q
71
- ```
gtfreader-0.1.2/README.md DELETED
@@ -1,55 +0,0 @@
1
- # gtfreader
2
-
3
- `gtfreader` is a small package for parsing and reading GTF files into pandas dataframes.
4
-
5
- Requires Python 3.12 or newer.
6
-
7
- ## Install
8
-
9
- ```bash
10
- python -m pip install -e .
11
- ```
12
-
13
- ## Usage
14
-
15
- ```python
16
- from gtfreader import read_gtf, read_gtf_python
17
-
18
- df = read_gtf("annotation.gtf")
19
- df_python = read_gtf_python("annotation.gtf")
20
- ```
21
-
22
- `read_gtf(...)` uses the compiled parser path when available. `read_gtf_python(...)` uses the high-level pure Python parser path used by the current `pyrunges` reader style.
23
-
24
- If you want to use the compiled low-level parser directly, pass it raw attribute strings from column 9 of the GTF before they have been expanded:
25
-
26
- ```python
27
- import pandas as pd
28
-
29
- from gtfreader import find_first_data_line_index, parse_chunk_columns
30
-
31
- skiprows = find_first_data_line_index("annotation.gtf")
32
- attribute_lines = pd.read_csv(
33
- "annotation.gtf",
34
- sep="\t",
35
- header=None,
36
- usecols=[8],
37
- names=["Attribute"],
38
- comment="#",
39
- skiprows=skiprows,
40
- )["Attribute"].tolist()
41
-
42
- compiled_columns = parse_chunk_columns(attribute_lines)
43
- ```
44
-
45
- ## Build
46
-
47
- ```bash
48
- python -m build
49
- ```
50
-
51
- ## Test
52
-
53
- ```bash
54
- python -m pytest -q
55
- ```
@@ -1,352 +0,0 @@
1
- # cython: language_level=3
2
- # cython: boundscheck=False
3
- # cython: wraparound=False
4
- # cython: initializedcheck=False
5
- # cython: infer_types=True
6
- # cython: nonecheck=False
7
-
8
- cdef inline int _key_code(str s, Py_ssize_t a, Py_ssize_t b):
9
- cdef Py_ssize_t n
10
- n = b - a
11
-
12
- if n == 3:
13
- if s[a] == 't' and s[a + 1] == 'a' and s[a + 2] == 'g':
14
- return 9
15
- if s[a] == 'o' and s[a + 1] == 'n' and s[a + 2] == 't':
16
- return 17
17
-
18
- elif n == 5:
19
- if (
20
- s[a] == 'l' and s[a + 1] == 'e' and s[a + 2] == 'v' and
21
- s[a + 3] == 'e' and s[a + 4] == 'l'
22
- ):
23
- return 16
24
-
25
- elif n == 6:
26
- if (
27
- s[a] == 'c' and s[a + 1] == 'c' and s[a + 2] == 'd' and
28
- s[a + 3] == 's' and s[a + 4] == 'i' and s[a + 5] == 'd'
29
- ):
30
- return 14
31
-
32
- elif n == 7:
33
- if (
34
- s[a] == 'g' and s[a + 1] == 'e' and s[a + 2] == 'n' and
35
- s[a + 3] == 'e' and s[a + 4] == '_' and s[a + 5] == 'i' and
36
- s[a + 6] == 'd'
37
- ):
38
- return 1
39
- if (
40
- s[a] == 'e' and s[a + 1] == 'x' and s[a + 2] == 'o' and
41
- s[a + 3] == 'n' and s[a + 4] == '_' and s[a + 5] == 'i' and
42
- s[a + 6] == 'd'
43
- ):
44
- return 8
45
- if (
46
- s[a] == 'h' and s[a + 1] == 'g' and s[a + 2] == 'n' and
47
- s[a + 3] == 'c' and s[a + 4] == '_' and s[a + 5] == 'i' and
48
- s[a + 6] == 'd'
49
- ):
50
- return 13
51
-
52
- elif n == 9:
53
- if (
54
- s[a] == 'g' and s[a + 1] == 'e' and s[a + 2] == 'n' and
55
- s[a + 3] == 'e' and s[a + 4] == '_'
56
- ):
57
- if (
58
- s[a + 5] == 'n' and s[a + 6] == 'a' and
59
- s[a + 7] == 'm' and s[a + 8] == 'e'
60
- ):
61
- return 3
62
- if (
63
- s[a + 5] == 't' and s[a + 6] == 'y' and
64
- s[a + 7] == 'p' and s[a + 8] == 'e'
65
- ):
66
- return 4
67
-
68
- elif n == 10:
69
- if (
70
- s[a] == 'a' and s[a + 1] == 'r' and s[a + 2] == 't' and
71
- s[a + 3] == 'i' and s[a + 4] == 'f' and s[a + 5] == '_' and
72
- s[a + 6] == 'd' and s[a + 7] == 'u' and s[a + 8] == 'p' and
73
- s[a + 9] == 'l'
74
- ):
75
- return 15
76
-
77
- elif n == 11:
78
- if (
79
- s[a] == 'e' and s[a + 1] == 'x' and s[a + 2] == 'o' and
80
- s[a + 3] == 'n' and s[a + 4] == '_' and s[a + 5] == 'n' and
81
- s[a + 6] == 'u' and s[a + 7] == 'm' and s[a + 8] == 'b' and
82
- s[a + 9] == 'e' and s[a + 10] == 'r'
83
- ):
84
- return 7
85
- if (
86
- s[a] == 'h' and s[a + 1] == 'a' and s[a + 2] == 'v' and
87
- s[a + 3] == 'a' and s[a + 4] == 'n' and s[a + 5] == 'a' and
88
- s[a + 6] == '_' and s[a + 7] == 'g' and s[a + 8] == 'e' and
89
- s[a + 9] == 'n' and s[a + 10] == 'e'
90
- ):
91
- return 11
92
-
93
- elif n == 13:
94
- if (
95
- s[a] == 't' and s[a + 1] == 'r' and s[a + 2] == 'a' and
96
- s[a + 3] == 'n' and s[a + 4] == 's' and s[a + 5] == 'c' and
97
- s[a + 6] == 'r' and s[a + 7] == 'i' and s[a + 8] == 'p' and
98
- s[a + 9] == 't' and s[a + 10] == '_' and s[a + 11] == 'i' and
99
- s[a + 12] == 'd'
100
- ):
101
- return 2
102
-
103
- elif n == 15:
104
- if (
105
- s[a] == 't' and s[a + 1] == 'r' and s[a + 2] == 'a' and
106
- s[a + 3] == 'n' and s[a + 4] == 's' and s[a + 5] == 'c' and
107
- s[a + 6] == 'r' and s[a + 7] == 'i' and s[a + 8] == 'p' and
108
- s[a + 9] == 't' and s[a + 10] == '_'
109
- ):
110
- if (
111
- s[a + 11] == 'n' and s[a + 12] == 'a' and
112
- s[a + 13] == 'm' and s[a + 14] == 'e'
113
- ):
114
- return 5
115
- if (
116
- s[a + 11] == 't' and s[a + 12] == 'y' and
117
- s[a + 13] == 'p' and s[a + 14] == 'e'
118
- ):
119
- return 6
120
-
121
- elif n == 17:
122
- if (
123
- s[a] == 'h' and s[a + 1] == 'a' and s[a + 2] == 'v' and
124
- s[a + 3] == 'a' and s[a + 4] == 'n' and s[a + 5] == 'a' and
125
- s[a + 6] == '_' and s[a + 7] == 't' and s[a + 8] == 'r' and
126
- s[a + 9] == 'a' and s[a + 10] == 'n' and s[a + 11] == 's' and
127
- s[a + 12] == 'c' and s[a + 13] == 'r' and s[a + 14] == 'i' and
128
- s[a + 15] == 'p' and s[a + 16] == 't'
129
- ):
130
- return 10
131
-
132
- elif n == 24:
133
- if (
134
- s[a] == 't' and s[a + 1] == 'r' and s[a + 2] == 'a' and
135
- s[a + 3] == 'n' and s[a + 4] == 's' and s[a + 5] == 'c' and
136
- s[a + 6] == 'r' and s[a + 7] == 'i' and s[a + 8] == 'p' and
137
- s[a + 9] == 't' and s[a + 10] == '_' and s[a + 11] == 's' and
138
- s[a + 12] == 'u' and s[a + 13] == 'p' and s[a + 14] == 'p' and
139
- s[a + 15] == 'o' and s[a + 16] == 'r' and s[a + 17] == 't' and
140
- s[a + 18] == '_' and s[a + 19] == 'l' and s[a + 20] == 'e' and
141
- s[a + 21] == 'v' and s[a + 22] == 'e' and s[a + 23] == 'l'
142
- ):
143
- return 12
144
-
145
- return 0
146
-
147
-
148
- cdef inline str _cached_str(dict cache, str value):
149
- cdef object obj
150
- obj = cache.get(value)
151
- if obj is None:
152
- cache[value] = value
153
- return value
154
- return <str> obj
155
-
156
-
157
- def parse_chunk_columns(lines):
158
- cdef Py_ssize_t n, row_i
159
- cdef Py_ssize_t pos, m, key_start, key_end, val_start, val_end
160
- cdef dict columns
161
-
162
- cdef dict gene_name_cache
163
- cdef dict gene_type_cache
164
- cdef dict transcript_type_cache
165
- cdef dict tag_cache
166
- cdef dict level_cache
167
- cdef dict ont_cache
168
- cdef dict transcript_support_level_cache
169
- cdef dict artif_dupl_cache
170
-
171
- cdef str s, key, value
172
- cdef list col
173
- cdef int code
174
-
175
- cdef object obj_col
176
-
177
- cdef list gene_id_col
178
- cdef list transcript_id_col
179
- cdef list gene_name_col
180
- cdef list gene_type_col
181
- cdef list transcript_name_col
182
- cdef list transcript_type_col
183
- cdef list exon_number_col
184
- cdef list exon_id_col
185
- cdef list tag_col
186
- cdef list havana_transcript_col
187
- cdef list havana_gene_col
188
- cdef list transcript_support_level_col
189
- cdef list hgnc_id_col
190
- cdef list ccdsid_col
191
- cdef list artif_dupl_col
192
- cdef list level_col
193
- cdef list ont_col
194
-
195
- n = len(lines)
196
- columns = {}
197
-
198
- gene_name_cache = {}
199
- gene_type_cache = {}
200
- transcript_type_cache = {}
201
- tag_cache = {}
202
- level_cache = {}
203
- ont_cache = {}
204
- transcript_support_level_cache = {}
205
- artif_dupl_cache = {}
206
-
207
- gene_id_col = [None] * n
208
- transcript_id_col = [None] * n
209
- gene_name_col = [None] * n
210
- gene_type_col = [None] * n
211
- transcript_name_col = [None] * n
212
- transcript_type_col = [None] * n
213
- exon_number_col = [None] * n
214
- exon_id_col = [None] * n
215
- tag_col = [None] * n
216
- havana_transcript_col = [None] * n
217
- havana_gene_col = [None] * n
218
- transcript_support_level_col = [None] * n
219
- hgnc_id_col = [None] * n
220
- ccdsid_col = [None] * n
221
- artif_dupl_col = [None] * n
222
- level_col = [None] * n
223
- ont_col = [None] * n
224
-
225
- columns["gene_id"] = gene_id_col
226
- columns["transcript_id"] = transcript_id_col
227
- columns["gene_name"] = gene_name_col
228
- columns["gene_type"] = gene_type_col
229
- columns["transcript_name"] = transcript_name_col
230
- columns["transcript_type"] = transcript_type_col
231
- columns["exon_number"] = exon_number_col
232
- columns["exon_id"] = exon_id_col
233
- columns["tag"] = tag_col
234
- columns["havana_transcript"] = havana_transcript_col
235
- columns["havana_gene"] = havana_gene_col
236
- columns["transcript_support_level"] = transcript_support_level_col
237
- columns["hgnc_id"] = hgnc_id_col
238
- columns["ccdsid"] = ccdsid_col
239
- columns["artif_dupl"] = artif_dupl_col
240
- columns["level"] = level_col
241
- columns["ont"] = ont_col
242
-
243
- for row_i in range(n):
244
- s = <str> lines[row_i]
245
- m = len(s)
246
- pos = 0
247
-
248
- while pos < m:
249
- while pos < m and (s[pos] == ' ' or s[pos] == '\t' or s[pos] == ';'):
250
- pos += 1
251
- if pos >= m:
252
- break
253
-
254
- key_start = pos
255
-
256
- while pos < m and s[pos] != ' ' and s[pos] != '\t':
257
- pos += 1
258
- key_end = pos
259
-
260
- while pos < m and (s[pos] == ' ' or s[pos] == '\t'):
261
- pos += 1
262
- if pos >= m:
263
- break
264
-
265
- if s[pos] == '"':
266
- val_start = pos + 1
267
- pos += 1
268
-
269
- while pos < m and s[pos] != '"':
270
- pos += 1
271
- if pos >= m:
272
- break
273
-
274
- val_end = pos
275
- else:
276
- val_start = pos
277
- while pos < m and s[pos] != ';':
278
- pos += 1
279
- val_end = pos
280
- while val_end > val_start and (s[val_end - 1] == ' ' or s[val_end - 1] == '\t'):
281
- val_end -= 1
282
-
283
- value = s[val_start:val_end]
284
-
285
- code = _key_code(s, key_start, key_end)
286
-
287
- if code == 1:
288
- gene_id_col[row_i] = value
289
-
290
- elif code == 2:
291
- transcript_id_col[row_i] = value
292
-
293
- elif code == 3:
294
- gene_name_col[row_i] = _cached_str(gene_name_cache, value)
295
-
296
- elif code == 4:
297
- gene_type_col[row_i] = _cached_str(gene_type_cache, value)
298
-
299
- elif code == 5:
300
- transcript_name_col[row_i] = value
301
-
302
- elif code == 6:
303
- transcript_type_col[row_i] = _cached_str(transcript_type_cache, value)
304
-
305
- elif code == 7:
306
- exon_number_col[row_i] = value
307
-
308
- elif code == 8:
309
- exon_id_col[row_i] = value
310
-
311
- elif code == 9:
312
- tag_col[row_i] = _cached_str(tag_cache, value)
313
-
314
- elif code == 10:
315
- havana_transcript_col[row_i] = value
316
-
317
- elif code == 11:
318
- havana_gene_col[row_i] = value
319
-
320
- elif code == 12:
321
- transcript_support_level_col[row_i] = _cached_str(
322
- transcript_support_level_cache, value
323
- )
324
-
325
- elif code == 13:
326
- hgnc_id_col[row_i] = value
327
-
328
- elif code == 14:
329
- ccdsid_col[row_i] = value
330
-
331
- elif code == 15:
332
- artif_dupl_col[row_i] = _cached_str(artif_dupl_cache, value)
333
-
334
- elif code == 16:
335
- level_col[row_i] = _cached_str(level_cache, value)
336
-
337
- elif code == 17:
338
- ont_col[row_i] = _cached_str(ont_cache, value)
339
-
340
- else:
341
- key = s[key_start:key_end]
342
- obj_col = columns.get(key)
343
- if obj_col is None:
344
- col = [None] * n
345
- columns[key] = col
346
- else:
347
- col = <list> obj_col
348
- col[row_i] = value
349
-
350
- pos += 1
351
-
352
- return columns
@@ -1,71 +0,0 @@
1
- Metadata-Version: 2.4
2
- Name: gtfreader
3
- Version: 0.1.2
4
- Summary: Fast Cython-backed parsing for GTF attribute columns.
5
- Author: Endre Bakken Stovner
6
- Classifier: Programming Language :: Python :: 3
7
- Classifier: Programming Language :: Cython
8
- Classifier: Programming Language :: Python :: 3 :: Only
9
- Classifier: Programming Language :: Python :: 3.12
10
- Classifier: Programming Language :: Python :: 3.13
11
- Classifier: Operating System :: OS Independent
12
- Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
13
- Requires-Python: >=3.12
14
- Description-Content-Type: text/markdown
15
- Requires-Dist: pandas>=2.0
16
-
17
- # gtfreader
18
-
19
- `gtfreader` is a small package for parsing and reading GTF files into pandas dataframes.
20
-
21
- Requires Python 3.12 or newer.
22
-
23
- ## Install
24
-
25
- ```bash
26
- python -m pip install -e .
27
- ```
28
-
29
- ## Usage
30
-
31
- ```python
32
- from gtfreader import read_gtf, read_gtf_python
33
-
34
- df = read_gtf("annotation.gtf")
35
- df_python = read_gtf_python("annotation.gtf")
36
- ```
37
-
38
- `read_gtf(...)` uses the compiled parser path when available. `read_gtf_python(...)` uses the high-level pure Python parser path used by the current `pyrunges` reader style.
39
-
40
- If you want to use the compiled low-level parser directly, pass it raw attribute strings from column 9 of the GTF before they have been expanded:
41
-
42
- ```python
43
- import pandas as pd
44
-
45
- from gtfreader import find_first_data_line_index, parse_chunk_columns
46
-
47
- skiprows = find_first_data_line_index("annotation.gtf")
48
- attribute_lines = pd.read_csv(
49
- "annotation.gtf",
50
- sep="\t",
51
- header=None,
52
- usecols=[8],
53
- names=["Attribute"],
54
- comment="#",
55
- skiprows=skiprows,
56
- )["Attribute"].tolist()
57
-
58
- compiled_columns = parse_chunk_columns(attribute_lines)
59
- ```
60
-
61
- ## Build
62
-
63
- ```bash
64
- python -m build
65
- ```
66
-
67
- ## Test
68
-
69
- ```bash
70
- python -m pytest -q
71
- ```
File without changes
File without changes
File without changes