apb-fasta 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- apb_fasta-0.1.0/LICENSE +21 -0
- apb_fasta-0.1.0/PKG-INFO +116 -0
- apb_fasta-0.1.0/README.md +88 -0
- apb_fasta-0.1.0/pyproject.toml +146 -0
- apb_fasta-0.1.0/pyproject.toml.orig +107 -0
- apb_fasta-0.1.0/src/apb_fasta/__init__.py +0 -0
- apb_fasta-0.1.0/src/apb_fasta/api.py +164 -0
- apb_fasta-0.1.0/src/apb_fasta/calculation/__init__.py +0 -0
- apb_fasta-0.1.0/src/apb_fasta/calculation/matching.py +209 -0
- apb_fasta-0.1.0/src/apb_fasta/calculation/peptide_properties.py +25 -0
- apb_fasta-0.1.0/src/apb_fasta/calculation/protein_groups.py +231 -0
- apb_fasta-0.1.0/src/apb_fasta/calculation/results.py +57 -0
- apb_fasta-0.1.0/src/apb_fasta/cli.py +160 -0
- apb_fasta-0.1.0/src/apb_fasta/configuration.py +21 -0
- apb_fasta-0.1.0/src/apb_fasta/errors.py +5 -0
- apb_fasta-0.1.0/src/apb_fasta/integration.py +288 -0
- apb_fasta-0.1.0/src/apb_fasta/py.typed +0 -0
apb_fasta-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Witold Wolski
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
apb_fasta-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: apb-fasta
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: FASTA validation and protein annotation for APB2 results.
|
|
5
|
+
Keywords: proteomics,fasta,anndata,protein annotation,mass spectrometry
|
|
6
|
+
Author: Witold Wolski
|
|
7
|
+
Author-email: Witold Wolski <wew@fgcz.ethz.ch>
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Classifier: Development Status :: 3 - Alpha
|
|
11
|
+
Classifier: Intended Audience :: Science/Research
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
15
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
16
|
+
Classifier: Typing :: Typed
|
|
17
|
+
Requires-Dist: apb2>=0.1,<0.2
|
|
18
|
+
Requires-Dist: cyclopts>=4,<5
|
|
19
|
+
Requires-Dist: loguru>=0.7,<1
|
|
20
|
+
Requires-Dist: polars>=1.43,<2
|
|
21
|
+
Requires-Dist: protein-fasta>=0.3,<0.4
|
|
22
|
+
Requires-Dist: prozor>=0.1,<0.2
|
|
23
|
+
Requires-Python: >=3.13
|
|
24
|
+
Project-URL: Documentation, https://anndata-omics-bridge.github.io/apb-fasta/
|
|
25
|
+
Project-URL: Repository, https://github.com/anndata-omics-bridge/apb-fasta
|
|
26
|
+
Project-URL: Issues, https://github.com/anndata-omics-bridge/apb-fasta/issues
|
|
27
|
+
Description-Content-Type: text/markdown
|
|
28
|
+
|
|
29
|
+
# apb-fasta
|
|
30
|
+
|
|
31
|
+
FASTA verification and protein annotation for APB2 results.
|
|
32
|
+
|
|
33
|
+
**[Online documentation](https://anndata-omics-bridge.github.io/apb-fasta/)** or its [source index](https://github.com/anndata-omics-bridge/apb-fasta/blob/main/docs/index.md).
|
|
34
|
+
|
|
35
|
+
The first release exposes two independent operations:
|
|
36
|
+
|
|
37
|
+
- verify that APB2's modification-stripped peptide sequences occur in the supplied FASTA database;
|
|
38
|
+
- merge FASTA annotations for every reported protein-group member without collapsing the group to its leading accession.
|
|
39
|
+
|
|
40
|
+
Protein inference with Prozor is the planned third operation. It will remain explicit and opt-in.
|
|
41
|
+
|
|
42
|
+
## Installation
|
|
43
|
+
|
|
44
|
+
APB FASTA requires Python 3.13 or later.
|
|
45
|
+
|
|
46
|
+
```bash
|
|
47
|
+
pip install apb-fasta
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
## Python API
|
|
51
|
+
|
|
52
|
+
Read the FASTA files once and load the APB2 result:
|
|
53
|
+
|
|
54
|
+
```python
|
|
55
|
+
from pathlib import Path
|
|
56
|
+
|
|
57
|
+
from apb2.api import read_parsed_levels, write_parsed_levels
|
|
58
|
+
from apb_fasta.api import FastaAnnotator
|
|
59
|
+
|
|
60
|
+
annotator = FastaAnnotator.read((Path("human.fasta"), Path("contaminants.fasta")))
|
|
61
|
+
parsed = read_parsed_levels(Path("input.h5mu"))
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
Verify peptides only:
|
|
65
|
+
|
|
66
|
+
```python
|
|
67
|
+
verified = annotator.verify_peptides(parsed)
|
|
68
|
+
print(verified.reports.peptide_levels)
|
|
69
|
+
write_parsed_levels(verified.parsed, Path("verified.h5mu"))
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
Merge protein annotations only:
|
|
73
|
+
|
|
74
|
+
```python
|
|
75
|
+
annotated = annotator.merge_annotations(parsed)
|
|
76
|
+
print(annotated.reports.protein_groups)
|
|
77
|
+
write_parsed_levels(annotated.parsed, Path("annotated.h5mu"))
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
Apply both operations in memory:
|
|
81
|
+
|
|
82
|
+
```python
|
|
83
|
+
complete = annotator.annotate(parsed)
|
|
84
|
+
write_parsed_levels(complete.parsed, Path("complete.h5mu"))
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
The operations can also be chained explicitly without an intermediate file:
|
|
88
|
+
|
|
89
|
+
```python
|
|
90
|
+
verified = annotator.verify_peptides(parsed)
|
|
91
|
+
complete = annotator.merge_annotations(verified.parsed)
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
The annotator validates and binds the reusable protein frame once. Every method accepts one canonical `ParsedLevels` value and returns an immutable `FastaAnnotationResult` containing its replacement plus typed operation reports. `apb_fasta` neither opens result files nor receives raw AnnData or MuData objects.
|
|
95
|
+
|
|
96
|
+
## CLI
|
|
97
|
+
|
|
98
|
+
Verify peptides only:
|
|
99
|
+
|
|
100
|
+
```bash
|
|
101
|
+
apb-fasta verify-peptides input.h5mu human.fasta contaminants.fasta --output verified.h5mu
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
Merge protein annotations only:
|
|
105
|
+
|
|
106
|
+
```bash
|
|
107
|
+
apb-fasta merge-annotations input.h5mu human.fasta contaminants.fasta --output annotated.h5mu
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
Apply both operations together:
|
|
111
|
+
|
|
112
|
+
```bash
|
|
113
|
+
apb-fasta run input.h5mu human.fasta contaminants.fasta --output complete.h5mu
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
Peptide verification adds feature-aligned `varm["fasta_validation"]` tables. Protein annotation adds protein-aligned `varm["fasta"]`, the lossless `fasta_protein_group_members` annotation table, and its directed relation to the protein axis.
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
# apb-fasta
|
|
2
|
+
|
|
3
|
+
FASTA verification and protein annotation for APB2 results.
|
|
4
|
+
|
|
5
|
+
**[Online documentation](https://anndata-omics-bridge.github.io/apb-fasta/)** or its [source index](https://github.com/anndata-omics-bridge/apb-fasta/blob/main/docs/index.md).
|
|
6
|
+
|
|
7
|
+
The first release exposes two independent operations:
|
|
8
|
+
|
|
9
|
+
- verify that APB2's modification-stripped peptide sequences occur in the supplied FASTA database;
|
|
10
|
+
- merge FASTA annotations for every reported protein-group member without collapsing the group to its leading accession.
|
|
11
|
+
|
|
12
|
+
Protein inference with Prozor is the planned third operation. It will remain explicit and opt-in.
|
|
13
|
+
|
|
14
|
+
## Installation
|
|
15
|
+
|
|
16
|
+
APB FASTA requires Python 3.13 or later.
|
|
17
|
+
|
|
18
|
+
```bash
|
|
19
|
+
pip install apb-fasta
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
## Python API
|
|
23
|
+
|
|
24
|
+
Read the FASTA files once and load the APB2 result:
|
|
25
|
+
|
|
26
|
+
```python
|
|
27
|
+
from pathlib import Path
|
|
28
|
+
|
|
29
|
+
from apb2.api import read_parsed_levels, write_parsed_levels
|
|
30
|
+
from apb_fasta.api import FastaAnnotator
|
|
31
|
+
|
|
32
|
+
annotator = FastaAnnotator.read((Path("human.fasta"), Path("contaminants.fasta")))
|
|
33
|
+
parsed = read_parsed_levels(Path("input.h5mu"))
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
Verify peptides only:
|
|
37
|
+
|
|
38
|
+
```python
|
|
39
|
+
verified = annotator.verify_peptides(parsed)
|
|
40
|
+
print(verified.reports.peptide_levels)
|
|
41
|
+
write_parsed_levels(verified.parsed, Path("verified.h5mu"))
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
Merge protein annotations only:
|
|
45
|
+
|
|
46
|
+
```python
|
|
47
|
+
annotated = annotator.merge_annotations(parsed)
|
|
48
|
+
print(annotated.reports.protein_groups)
|
|
49
|
+
write_parsed_levels(annotated.parsed, Path("annotated.h5mu"))
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
Apply both operations in memory:
|
|
53
|
+
|
|
54
|
+
```python
|
|
55
|
+
complete = annotator.annotate(parsed)
|
|
56
|
+
write_parsed_levels(complete.parsed, Path("complete.h5mu"))
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
The operations can also be chained explicitly without an intermediate file:
|
|
60
|
+
|
|
61
|
+
```python
|
|
62
|
+
verified = annotator.verify_peptides(parsed)
|
|
63
|
+
complete = annotator.merge_annotations(verified.parsed)
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
The annotator validates and binds the reusable protein frame once. Every method accepts one canonical `ParsedLevels` value and returns an immutable `FastaAnnotationResult` containing its replacement plus typed operation reports. `apb_fasta` neither opens result files nor receives raw AnnData or MuData objects.
|
|
67
|
+
|
|
68
|
+
## CLI
|
|
69
|
+
|
|
70
|
+
Verify peptides only:
|
|
71
|
+
|
|
72
|
+
```bash
|
|
73
|
+
apb-fasta verify-peptides input.h5mu human.fasta contaminants.fasta --output verified.h5mu
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
Merge protein annotations only:
|
|
77
|
+
|
|
78
|
+
```bash
|
|
79
|
+
apb-fasta merge-annotations input.h5mu human.fasta contaminants.fasta --output annotated.h5mu
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
Apply both operations together:
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
apb-fasta run input.h5mu human.fasta contaminants.fasta --output complete.h5mu
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
Peptide verification adds feature-aligned `varm["fasta_validation"]` tables. Protein annotation adds protein-aligned `varm["fasta"]`, the lossless `fasta_protein_group_members` annotation table, and its directed relation to the protein axis.
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["uv_build>=0.9.26,<0.10.0"]
|
|
3
|
+
build-backend = "uv_build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "apb-fasta"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "FASTA validation and protein annotation for APB2 results."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.13"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
keywords = [
|
|
14
|
+
"proteomics",
|
|
15
|
+
"fasta",
|
|
16
|
+
"anndata",
|
|
17
|
+
"protein annotation",
|
|
18
|
+
"mass spectrometry",
|
|
19
|
+
]
|
|
20
|
+
classifiers = [
|
|
21
|
+
"Development Status :: 3 - Alpha",
|
|
22
|
+
"Intended Audience :: Science/Research",
|
|
23
|
+
"Operating System :: OS Independent",
|
|
24
|
+
"Programming Language :: Python :: 3",
|
|
25
|
+
"Programming Language :: Python :: 3.13",
|
|
26
|
+
"Topic :: Scientific/Engineering :: Bio-Informatics",
|
|
27
|
+
"Typing :: Typed",
|
|
28
|
+
]
|
|
29
|
+
dependencies = [
|
|
30
|
+
"apb2>=0.1,<0.2",
|
|
31
|
+
"cyclopts>=4,<5",
|
|
32
|
+
"loguru>=0.7,<1",
|
|
33
|
+
"polars>=1.43,<2",
|
|
34
|
+
"protein-fasta>=0.3,<0.4",
|
|
35
|
+
"prozor>=0.1,<0.2",
|
|
36
|
+
]
|
|
37
|
+
|
|
38
|
+
[[project.authors]]
|
|
39
|
+
name = "Witold Wolski"
|
|
40
|
+
email = "wew@fgcz.ethz.ch"
|
|
41
|
+
|
|
42
|
+
[project.scripts]
|
|
43
|
+
apb-fasta = "apb_fasta.cli:main"
|
|
44
|
+
|
|
45
|
+
[project.urls]
|
|
46
|
+
Documentation = "https://anndata-omics-bridge.github.io/apb-fasta/"
|
|
47
|
+
Repository = "https://github.com/anndata-omics-bridge/apb-fasta"
|
|
48
|
+
Issues = "https://github.com/anndata-omics-bridge/apb-fasta/issues"
|
|
49
|
+
|
|
50
|
+
[dependency-groups]
|
|
51
|
+
dev = [
|
|
52
|
+
"build>=1.3,<2",
|
|
53
|
+
"deptry>=0.24,<1",
|
|
54
|
+
"import-linter>=2.9,<3",
|
|
55
|
+
"pre-commit>=4,<5",
|
|
56
|
+
"pyright>=1.1.400,<2",
|
|
57
|
+
"pytest>=9,<10",
|
|
58
|
+
"pytest-cov>=7,<8",
|
|
59
|
+
"ruff>=0.15,<1",
|
|
60
|
+
"twine>=6,<7",
|
|
61
|
+
]
|
|
62
|
+
docs = [
|
|
63
|
+
"mkdocstrings[python]>=1.0,<2",
|
|
64
|
+
"pymdown-extensions>=11,<12",
|
|
65
|
+
"zensical==0.0.43",
|
|
66
|
+
]
|
|
67
|
+
|
|
68
|
+
[tool.uv.sources.apb2]
|
|
69
|
+
path = "../apb2"
|
|
70
|
+
editable = true
|
|
71
|
+
|
|
72
|
+
[tool.uv.sources.protein-fasta]
|
|
73
|
+
path = "../protein_fasta"
|
|
74
|
+
editable = true
|
|
75
|
+
|
|
76
|
+
[tool.uv.sources.prozor]
|
|
77
|
+
path = "../prozor"
|
|
78
|
+
editable = true
|
|
79
|
+
|
|
80
|
+
[tool.ruff]
|
|
81
|
+
line-length = 100
|
|
82
|
+
target-version = "py313"
|
|
83
|
+
src = [
|
|
84
|
+
"src",
|
|
85
|
+
"tests",
|
|
86
|
+
]
|
|
87
|
+
|
|
88
|
+
[tool.ruff.lint]
|
|
89
|
+
select = [
|
|
90
|
+
"ANN",
|
|
91
|
+
"B",
|
|
92
|
+
"C4",
|
|
93
|
+
"C90",
|
|
94
|
+
"E4",
|
|
95
|
+
"E7",
|
|
96
|
+
"E9",
|
|
97
|
+
"F",
|
|
98
|
+
"I",
|
|
99
|
+
"PGH",
|
|
100
|
+
"PIE",
|
|
101
|
+
"RUF",
|
|
102
|
+
"SIM",
|
|
103
|
+
"UP",
|
|
104
|
+
]
|
|
105
|
+
|
|
106
|
+
[tool.ruff.lint.mccabe]
|
|
107
|
+
max-complexity = 10
|
|
108
|
+
|
|
109
|
+
[tool.ruff.format]
|
|
110
|
+
docstring-code-format = true
|
|
111
|
+
|
|
112
|
+
[tool.pyright]
|
|
113
|
+
include = [
|
|
114
|
+
"src",
|
|
115
|
+
"tests",
|
|
116
|
+
]
|
|
117
|
+
venvPath = "."
|
|
118
|
+
venv = ".venv"
|
|
119
|
+
pythonVersion = "3.13"
|
|
120
|
+
typeCheckingMode = "strict"
|
|
121
|
+
reportImportCycles = "error"
|
|
122
|
+
reportMissingTypeStubs = "error"
|
|
123
|
+
reportUnnecessaryTypeIgnoreComment = "error"
|
|
124
|
+
reportImplicitOverride = "error"
|
|
125
|
+
enableTypeIgnoreComments = false
|
|
126
|
+
|
|
127
|
+
[tool.pytest.ini_options]
|
|
128
|
+
addopts = [
|
|
129
|
+
"--strict-config",
|
|
130
|
+
"--strict-markers",
|
|
131
|
+
"-ra",
|
|
132
|
+
]
|
|
133
|
+
testpaths = ["tests"]
|
|
134
|
+
xfail_strict = true
|
|
135
|
+
|
|
136
|
+
[tool.coverage.run]
|
|
137
|
+
branch = true
|
|
138
|
+
source = ["apb_fasta"]
|
|
139
|
+
|
|
140
|
+
[tool.coverage.report]
|
|
141
|
+
fail_under = 80
|
|
142
|
+
show_missing = true
|
|
143
|
+
skip_covered = true
|
|
144
|
+
|
|
145
|
+
[tool.deptry]
|
|
146
|
+
known_first_party = ["apb_fasta"]
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["uv_build>=0.9.26,<0.10.0"]
|
|
3
|
+
build-backend = "uv_build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "apb-fasta"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "FASTA validation and protein annotation for APB2 results."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.13"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [
|
|
14
|
+
{ name = "Witold Wolski", email = "wew@fgcz.ethz.ch" },
|
|
15
|
+
]
|
|
16
|
+
keywords = ["proteomics", "fasta", "anndata", "protein annotation", "mass spectrometry"]
|
|
17
|
+
classifiers = [
|
|
18
|
+
"Development Status :: 3 - Alpha",
|
|
19
|
+
"Intended Audience :: Science/Research",
|
|
20
|
+
"Operating System :: OS Independent",
|
|
21
|
+
"Programming Language :: Python :: 3",
|
|
22
|
+
"Programming Language :: Python :: 3.13",
|
|
23
|
+
"Topic :: Scientific/Engineering :: Bio-Informatics",
|
|
24
|
+
"Typing :: Typed",
|
|
25
|
+
]
|
|
26
|
+
dependencies = [
|
|
27
|
+
"apb2>=0.1,<0.2",
|
|
28
|
+
"cyclopts>=4,<5",
|
|
29
|
+
"loguru>=0.7,<1",
|
|
30
|
+
"polars>=1.43,<2",
|
|
31
|
+
"protein-fasta>=0.3,<0.4",
|
|
32
|
+
"prozor>=0.1,<0.2",
|
|
33
|
+
]
|
|
34
|
+
|
|
35
|
+
[project.scripts]
|
|
36
|
+
apb-fasta = "apb_fasta.cli:main"
|
|
37
|
+
|
|
38
|
+
[project.urls]
|
|
39
|
+
Documentation = "https://anndata-omics-bridge.github.io/apb-fasta/"
|
|
40
|
+
Repository = "https://github.com/anndata-omics-bridge/apb-fasta"
|
|
41
|
+
Issues = "https://github.com/anndata-omics-bridge/apb-fasta/issues"
|
|
42
|
+
|
|
43
|
+
[dependency-groups]
|
|
44
|
+
dev = [
|
|
45
|
+
"build>=1.3,<2",
|
|
46
|
+
"deptry>=0.24,<1",
|
|
47
|
+
"import-linter>=2.9,<3",
|
|
48
|
+
"pre-commit>=4,<5",
|
|
49
|
+
"pyright>=1.1.400,<2",
|
|
50
|
+
"pytest>=9,<10",
|
|
51
|
+
"pytest-cov>=7,<8",
|
|
52
|
+
"ruff>=0.15,<1",
|
|
53
|
+
"twine>=6,<7",
|
|
54
|
+
]
|
|
55
|
+
docs = [
|
|
56
|
+
"mkdocstrings[python]>=1.0,<2",
|
|
57
|
+
"pymdown-extensions>=11,<12",
|
|
58
|
+
"zensical==0.0.43",
|
|
59
|
+
]
|
|
60
|
+
|
|
61
|
+
[tool.uv.sources]
|
|
62
|
+
apb2 = { path = "../apb2", editable = true }
|
|
63
|
+
protein-fasta = { path = "../protein_fasta", editable = true }
|
|
64
|
+
prozor = { path = "../prozor", editable = true }
|
|
65
|
+
|
|
66
|
+
[tool.ruff]
|
|
67
|
+
line-length = 100
|
|
68
|
+
target-version = "py313"
|
|
69
|
+
src = ["src", "tests"]
|
|
70
|
+
|
|
71
|
+
[tool.ruff.lint]
|
|
72
|
+
select = ["ANN", "B", "C4", "C90", "E4", "E7", "E9", "F", "I", "PGH", "PIE", "RUF", "SIM", "UP"]
|
|
73
|
+
|
|
74
|
+
[tool.ruff.lint.mccabe]
|
|
75
|
+
max-complexity = 10
|
|
76
|
+
|
|
77
|
+
[tool.ruff.format]
|
|
78
|
+
docstring-code-format = true
|
|
79
|
+
|
|
80
|
+
[tool.pyright]
|
|
81
|
+
include = ["src", "tests"]
|
|
82
|
+
venvPath = "."
|
|
83
|
+
venv = ".venv"
|
|
84
|
+
pythonVersion = "3.13"
|
|
85
|
+
typeCheckingMode = "strict"
|
|
86
|
+
reportImportCycles = "error"
|
|
87
|
+
reportMissingTypeStubs = "error"
|
|
88
|
+
reportUnnecessaryTypeIgnoreComment = "error"
|
|
89
|
+
reportImplicitOverride = "error"
|
|
90
|
+
enableTypeIgnoreComments = false
|
|
91
|
+
|
|
92
|
+
[tool.pytest.ini_options]
|
|
93
|
+
addopts = ["--strict-config", "--strict-markers", "-ra"]
|
|
94
|
+
testpaths = ["tests"]
|
|
95
|
+
xfail_strict = true
|
|
96
|
+
|
|
97
|
+
[tool.coverage.run]
|
|
98
|
+
branch = true
|
|
99
|
+
source = ["apb_fasta"]
|
|
100
|
+
|
|
101
|
+
[tool.coverage.report]
|
|
102
|
+
fail_under = 80
|
|
103
|
+
show_missing = true
|
|
104
|
+
skip_covered = true
|
|
105
|
+
|
|
106
|
+
[tool.deptry]
|
|
107
|
+
known_first_party = ["apb_fasta"]
|
|
File without changes
|
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
"""Public in-memory FASTA annotation API."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Sequence
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from importlib.metadata import version
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
import polars as pl
|
|
11
|
+
from apb2.api import ParsedLevels
|
|
12
|
+
from protein_fasta.api import ProteinDatabase, ProteinFormat
|
|
13
|
+
from prozor.api import resolve_backend
|
|
14
|
+
|
|
15
|
+
from apb_fasta.calculation.matching import match_peptide_levels
|
|
16
|
+
from apb_fasta.calculation.peptide_properties import peptide_level_properties
|
|
17
|
+
from apb_fasta.calculation.protein_groups import match_protein_groups
|
|
18
|
+
from apb_fasta.calculation.results import FastaAnnotationReports
|
|
19
|
+
from apb_fasta.configuration import (
|
|
20
|
+
DEFAULT_FASTA_ANNOTATION_PARAMETERS,
|
|
21
|
+
FastaAnnotationParameters,
|
|
22
|
+
)
|
|
23
|
+
from apb_fasta.errors import FastaAnnotationError
|
|
24
|
+
from apb_fasta.integration import (
|
|
25
|
+
apply_peptide_matches,
|
|
26
|
+
apply_peptide_properties,
|
|
27
|
+
apply_protein_group_match,
|
|
28
|
+
peptide_inputs,
|
|
29
|
+
protein_frame_metadata,
|
|
30
|
+
protein_group_input,
|
|
31
|
+
validate_protein_frame,
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
_NO_PEPTIDE_LEVEL = (
|
|
35
|
+
"result contains no peptide-derived level with canonical ProForma_peptide values"
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass(frozen=True, slots=True)
|
|
40
|
+
class FastaAnnotationResult:
|
|
41
|
+
"""A replacement APB2 result and its FASTA reports."""
|
|
42
|
+
|
|
43
|
+
parsed: ParsedLevels
|
|
44
|
+
reports: FastaAnnotationReports
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class FastaAnnotator:
|
|
48
|
+
"""Bind one protein database and its FASTA annotation behavior."""
|
|
49
|
+
|
|
50
|
+
__slots__ = ("_parameters", "_proteins")
|
|
51
|
+
|
|
52
|
+
def __init__(
|
|
53
|
+
self,
|
|
54
|
+
proteins: pl.DataFrame,
|
|
55
|
+
parameters: FastaAnnotationParameters = DEFAULT_FASTA_ANNOTATION_PARAMETERS,
|
|
56
|
+
) -> None:
|
|
57
|
+
"""Validate and bind a reusable protein database and configuration."""
|
|
58
|
+
validate_protein_frame(proteins)
|
|
59
|
+
self._proteins = proteins
|
|
60
|
+
self._parameters = parameters
|
|
61
|
+
|
|
62
|
+
@classmethod
|
|
63
|
+
def read(
|
|
64
|
+
cls,
|
|
65
|
+
fasta: Sequence[Path],
|
|
66
|
+
formats: Sequence[str] = ("uniprotkb", "refseq"),
|
|
67
|
+
parameters: FastaAnnotationParameters = DEFAULT_FASTA_ANNOTATION_PARAMETERS,
|
|
68
|
+
) -> FastaAnnotator:
|
|
69
|
+
"""Read FASTA files, or the protein-fasta database Parquet written from them."""
|
|
70
|
+
if not fasta:
|
|
71
|
+
raise ValueError("at least one FASTA path is required")
|
|
72
|
+
database = ProteinDatabase(*(ProteinFormat(name) for name in formats))
|
|
73
|
+
return cls(database.parse(tuple(fasta)), parameters)
|
|
74
|
+
|
|
75
|
+
@property
|
|
76
|
+
def proteins(self) -> pl.DataFrame:
|
|
77
|
+
"""The bound protein table, one row per FASTA entry in file order."""
|
|
78
|
+
return self._proteins
|
|
79
|
+
|
|
80
|
+
def verify_peptides(self, parsed: ParsedLevels) -> FastaAnnotationResult:
|
|
81
|
+
"""Verify every canonical stripped peptide against the protein sequences."""
|
|
82
|
+
inputs = peptide_inputs(parsed)
|
|
83
|
+
if not inputs:
|
|
84
|
+
raise FastaAnnotationError(_NO_PEPTIDE_LEVEL)
|
|
85
|
+
peptide_levels = match_peptide_levels(
|
|
86
|
+
inputs,
|
|
87
|
+
self._proteins,
|
|
88
|
+
backend=self._parameters.matcher_backend,
|
|
89
|
+
il_equivalent=self._parameters.il_equivalent,
|
|
90
|
+
protein_group_separator=self._parameters.protein_group_separator,
|
|
91
|
+
)
|
|
92
|
+
replacement, reports = apply_peptide_matches(
|
|
93
|
+
parsed,
|
|
94
|
+
peptide_levels,
|
|
95
|
+
requested_backend=self._parameters.matcher_backend,
|
|
96
|
+
resolved_backend=resolve_backend(self._parameters.matcher_backend),
|
|
97
|
+
il_equivalent=self._parameters.il_equivalent,
|
|
98
|
+
protein_metadata=protein_frame_metadata(self._proteins),
|
|
99
|
+
)
|
|
100
|
+
return FastaAnnotationResult(parsed=replacement, reports=reports)
|
|
101
|
+
|
|
102
|
+
def merge_annotations(self, parsed: ParsedLevels) -> FastaAnnotationResult:
|
|
103
|
+
"""Merge FASTA annotations for every reported protein-group member."""
|
|
104
|
+
protein_input = protein_group_input(parsed)
|
|
105
|
+
if protein_input is None:
|
|
106
|
+
raise FastaAnnotationError("result contains no protein level to annotate")
|
|
107
|
+
protein_groups = match_protein_groups(
|
|
108
|
+
protein_input,
|
|
109
|
+
self._proteins,
|
|
110
|
+
separator=self._parameters.protein_group_separator,
|
|
111
|
+
)
|
|
112
|
+
replacement, reports = apply_protein_group_match(
|
|
113
|
+
parsed,
|
|
114
|
+
protein_groups,
|
|
115
|
+
protein_group_separator=self._parameters.protein_group_separator,
|
|
116
|
+
protein_metadata=protein_frame_metadata(self._proteins),
|
|
117
|
+
)
|
|
118
|
+
return FastaAnnotationResult(parsed=replacement, reports=reports)
|
|
119
|
+
|
|
120
|
+
def annotate(self, parsed: ParsedLevels) -> FastaAnnotationResult:
|
|
121
|
+
"""Verify peptides and then merge protein annotations in memory."""
|
|
122
|
+
verified = self.verify_peptides(parsed)
|
|
123
|
+
annotated = self.merge_annotations(verified.parsed)
|
|
124
|
+
return FastaAnnotationResult(
|
|
125
|
+
parsed=annotated.parsed,
|
|
126
|
+
reports=FastaAnnotationReports(
|
|
127
|
+
peptide_levels=verified.reports.peptide_levels,
|
|
128
|
+
protein_groups=annotated.reports.protein_groups,
|
|
129
|
+
),
|
|
130
|
+
)
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def add_peptide_properties(parsed: ParsedLevels) -> ParsedLevels:
|
|
134
|
+
"""Attach sequence-derived properties to every peptide-derived level.
|
|
135
|
+
|
|
136
|
+
Writes a feature-aligned ``varm["peptide_properties"]`` computed by
|
|
137
|
+
``protein_fasta.peptide_frame.peptide_property_frame`` from each level's ``ProForma_peptide``.
|
|
138
|
+
No protein database is needed. A feature without a sequence, or with a residue outside the 20
|
|
139
|
+
standard amino acids, has null properties.
|
|
140
|
+
|
|
141
|
+
Raises:
|
|
142
|
+
FastaAnnotationError: If the result has no peptide-derived level, already contains the
|
|
143
|
+
output, or already records the operation.
|
|
144
|
+
ValueError: If a sequence is not stripped upper-case letters.
|
|
145
|
+
"""
|
|
146
|
+
inputs = peptide_inputs(parsed)
|
|
147
|
+
if not inputs:
|
|
148
|
+
raise FastaAnnotationError(_NO_PEPTIDE_LEVEL)
|
|
149
|
+
properties = {
|
|
150
|
+
name: peptide_level_properties(level.frame, level.sequence_column)
|
|
151
|
+
for name, level in inputs.items()
|
|
152
|
+
}
|
|
153
|
+
return apply_peptide_properties(
|
|
154
|
+
parsed, properties, protein_fasta_version=version("protein-fasta")
|
|
155
|
+
)
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
__all__ = [
|
|
159
|
+
"FastaAnnotationParameters",
|
|
160
|
+
"FastaAnnotationReports",
|
|
161
|
+
"FastaAnnotationResult",
|
|
162
|
+
"FastaAnnotator",
|
|
163
|
+
"add_peptide_properties",
|
|
164
|
+
]
|
|
File without changes
|