gtfreader 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,5 @@
1
+ include README.md
2
+ include pyproject.toml
3
+ include setup.py
4
+ recursive-include gtfreader *.py *.pyx *.c
5
+ recursive-include tests *.py
@@ -0,0 +1,66 @@
1
+ Metadata-Version: 2.4
2
+ Name: gtfreader
3
+ Version: 0.1.0
4
+ Summary: Fast Cython-backed parsing for GTF attribute columns.
5
+ Author: Endre Bakken Stovner
6
+ Classifier: Programming Language :: Python :: 3
7
+ Classifier: Programming Language :: Cython
8
+ Classifier: Programming Language :: Python :: 3 :: Only
9
+ Classifier: Operating System :: OS Independent
10
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
11
+ Requires-Python: >=3.9
12
+ Description-Content-Type: text/markdown
13
+ Requires-Dist: pandas>=2.0
14
+
15
+ # gtfreader
16
+
17
+ `gtfreader` is a small package for parsing and reading GTF files into pandas dataframes.
18
+
19
+ ## Install
20
+
21
+ ```bash
22
+ python -m pip install -e .
23
+ ```
24
+
25
+ ## Usage
26
+
27
+ ```python
28
+ from gtfreader import read_gtf, read_gtf_python
29
+
30
+ df = read_gtf("annotation.gtf")
31
+ df_python = read_gtf_python("annotation.gtf")
32
+ ```
33
+
34
+ `read_gtf(...)` uses the compiled parser path when available. `read_gtf_python(...)` uses the high-level pure Python parser path used by the current `pyrunges` reader style.
35
+
36
+ If you want to use the compiled low-level parser directly, pass it raw attribute strings from column 9 of the GTF before they have been expanded:
37
+
38
+ ```python
39
+ import pandas as pd
40
+
41
+ from gtfreader import find_first_data_line_index, parse_chunk_columns
42
+
43
+ skiprows = find_first_data_line_index("annotation.gtf")
44
+ attribute_lines = pd.read_csv(
45
+ "annotation.gtf",
46
+ sep="\t",
47
+ header=None,
48
+ usecols=[8],
49
+ names=["Attribute"],
50
+ skiprows=skiprows,
51
+ )["Attribute"].tolist()
52
+
53
+ compiled_columns = parse_chunk_columns(attribute_lines)
54
+ ```
55
+
56
+ ## Build
57
+
58
+ ```bash
59
+ python -m build
60
+ ```
61
+
62
+ ## Test
63
+
64
+ ```bash
65
+ python -m pytest -q
66
+ ```
@@ -0,0 +1,52 @@
1
+ # gtfreader
2
+
3
+ `gtfreader` is a small package for parsing and reading GTF files into pandas dataframes.
4
+
5
+ ## Install
6
+
7
+ ```bash
8
+ python -m pip install -e .
9
+ ```
10
+
11
+ ## Usage
12
+
13
+ ```python
14
+ from gtfreader import read_gtf, read_gtf_python
15
+
16
+ df = read_gtf("annotation.gtf")
17
+ df_python = read_gtf_python("annotation.gtf")
18
+ ```
19
+
20
+ `read_gtf(...)` uses the compiled parser path when available. `read_gtf_python(...)` uses the high-level pure Python parser path used by the current `pyrunges` reader style.
21
+
22
+ If you want to use the compiled low-level parser directly, pass it raw attribute strings from column 9 of the GTF before they have been expanded:
23
+
24
+ ```python
25
+ import pandas as pd
26
+
27
+ from gtfreader import find_first_data_line_index, parse_chunk_columns
28
+
29
+ skiprows = find_first_data_line_index("annotation.gtf")
30
+ attribute_lines = pd.read_csv(
31
+ "annotation.gtf",
32
+ sep="\t",
33
+ header=None,
34
+ usecols=[8],
35
+ names=["Attribute"],
36
+ skiprows=skiprows,
37
+ )["Attribute"].tolist()
38
+
39
+ compiled_columns = parse_chunk_columns(attribute_lines)
40
+ ```
41
+
42
+ ## Build
43
+
44
+ ```bash
45
+ python -m build
46
+ ```
47
+
48
+ ## Test
49
+
50
+ ```bash
51
+ python -m pytest -q
52
+ ```
@@ -0,0 +1,35 @@
1
+ """Public package API for gtfreader."""
2
+
3
+ try:
4
+ from ._parser import parse_chunk_columns
5
+ except ImportError:
6
+ def parse_chunk_columns(*_args, **_kwargs):
7
+ msg = "gtfreader.parse_chunk_columns requires the compiled extension. Use read_gtf_python or read_gtf_full_python for the pure Python fallback."
8
+ raise ImportError(msg)
9
+
10
+ from .readers import (
11
+ find_first_data_line_index,
12
+ parse_kv_fields,
13
+ read_gtf,
14
+ read_gtf_full,
15
+ read_gtf_full_python,
16
+ read_gtf_python,
17
+ read_gtf_restricted,
18
+ read_gtf_restricted_python,
19
+ to_rows,
20
+ to_rows_keep_duplicates,
21
+ )
22
+
23
+ __all__ = [
24
+ "find_first_data_line_index",
25
+ "parse_chunk_columns",
26
+ "parse_kv_fields",
27
+ "read_gtf",
28
+ "read_gtf_full",
29
+ "read_gtf_full_python",
30
+ "read_gtf_python",
31
+ "read_gtf_restricted",
32
+ "read_gtf_restricted_python",
33
+ "to_rows",
34
+ "to_rows_keep_duplicates",
35
+ ]