prozor 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- prozor-0.1.0/LICENSE +21 -0
- prozor-0.1.0/PKG-INFO +122 -0
- prozor-0.1.0/README.md +96 -0
- prozor-0.1.0/pyproject.toml +142 -0
- prozor-0.1.0/pyproject.toml.orig +122 -0
- prozor-0.1.0/src/prozor/__init__.py +1 -0
- prozor-0.1.0/src/prozor/api.py +21 -0
- prozor-0.1.0/src/prozor/inference/__init__.py +0 -0
- prozor-0.1.0/src/prozor/inference/greedy.py +330 -0
- prozor-0.1.0/src/prozor/inference/ties.py +24 -0
- prozor-0.1.0/src/prozor/matching/__init__.py +0 -0
- prozor-0.1.0/src/prozor/matching/annotation.py +168 -0
- prozor-0.1.0/src/prozor/matching/automaton.py +164 -0
- prozor-0.1.0/src/prozor/py.typed +1 -0
prozor-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Witold Wolski
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
prozor-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: prozor
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Typed peptide-to-protein matching and parsimonious protein inference
|
|
5
|
+
Keywords: proteomics,protein inference,peptide matching,mass spectrometry
|
|
6
|
+
Author: Witold Wolski
|
|
7
|
+
Author-email: Witold Wolski <wew@fgcz.ethz.ch>
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Classifier: Development Status :: 3 - Alpha
|
|
11
|
+
Classifier: Intended Audience :: Science/Research
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
16
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
17
|
+
Classifier: Typing :: Typed
|
|
18
|
+
Requires-Dist: ahocorapy>=1.6,<2
|
|
19
|
+
Requires-Dist: ahocorasick-rs>=1.0,<2
|
|
20
|
+
Requires-Python: >=3.12
|
|
21
|
+
Project-URL: Homepage, https://github.com/anndata-omics-bridge/prozor
|
|
22
|
+
Project-URL: Documentation, https://anndata-omics-bridge.github.io/prozor/
|
|
23
|
+
Project-URL: Repository, https://github.com/anndata-omics-bridge/prozor.git
|
|
24
|
+
Project-URL: Issues, https://github.com/anndata-omics-bridge/prozor/issues
|
|
25
|
+
Description-Content-Type: text/markdown
|
|
26
|
+
|
|
27
|
+
# Prozor
|
|
28
|
+
|
|
29
|
+
[](https://github.com/anndata-omics-bridge/prozor/actions/workflows/quality.yml)
|
|
30
|
+
[](https://anndata-omics-bridge.github.io/prozor/)
|
|
31
|
+
[](https://www.python.org/)
|
|
32
|
+
[](https://spdx.org/licenses/MIT.html)
|
|
33
|
+
|
|
34
|
+
Typed peptide-to-protein matching and deterministic greedy-parsimony protein
|
|
35
|
+
inference.
|
|
36
|
+
|
|
37
|
+
This repository is a Python port of the R [`prozor`](https://github.com/protViz/prozor) package.
|
|
38
|
+
|
|
39
|
+
Prozor offers backend-neutral Aho--Corasick matching over mappings or streaming
|
|
40
|
+
protein records, occurrence-level annotations, unique peptide--protein edges,
|
|
41
|
+
and deterministic protein inference. The core deliberately does not
|
|
42
|
+
parse FASTA files or depend on pandas, AnnData, MuData, MuLink, workflow engines,
|
|
43
|
+
or consumer CLIs.
|
|
44
|
+
|
|
45
|
+
**[Documentation](https://anndata-omics-bridge.github.io/prozor/)** ·
|
|
46
|
+
**[Getting started](https://anndata-omics-bridge.github.io/prozor/getting-started/)** ·
|
|
47
|
+
**[API reference](https://anndata-omics-bridge.github.io/prozor/api/)**
|
|
48
|
+
|
|
49
|
+
Documentation: <https://anndata-omics-bridge.github.io/prozor/>
|
|
50
|
+
|
|
51
|
+
## Installation
|
|
52
|
+
|
|
53
|
+
```bash
|
|
54
|
+
python -m pip install prozor
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
Both matching implementations are installed. The public matching operations
|
|
58
|
+
default to `backend="auto"`, which selects `ahocorasick_rs`. The portable
|
|
59
|
+
`ahocorapy` implementation remains directly selectable and is the automatic
|
|
60
|
+
runtime fallback if Rust cannot be imported. Results record both the requested
|
|
61
|
+
and concrete backend.
|
|
62
|
+
|
|
63
|
+
## Quick start
|
|
64
|
+
|
|
65
|
+
```python
|
|
66
|
+
from dataclasses import dataclass
|
|
67
|
+
|
|
68
|
+
from prozor.api import annotate_peptides_streaming, greedy_parsimony
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
@dataclass(frozen=True, slots=True)
|
|
72
|
+
class Protein:
|
|
73
|
+
id: str
|
|
74
|
+
sequence: str
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
matches = annotate_peptides_streaming(
|
|
78
|
+
["PEPTIDE", "SEQUENCE"],
|
|
79
|
+
[
|
|
80
|
+
Protein("P1", "MYPEPTIDESEQUENCE"),
|
|
81
|
+
Protein("P2", "XXSEQUENCEXX"),
|
|
82
|
+
],
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
for match in matches:
|
|
86
|
+
print(match.peptide, match.protein_id, match.start, match.end)
|
|
87
|
+
|
|
88
|
+
edges = {(match.peptide, match.protein_id) for match in matches}
|
|
89
|
+
protein_groups = greedy_parsimony(edges)
|
|
90
|
+
print(protein_groups.to_dict())
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
Matches include nested, overlapping, and repeated sites using half-open
|
|
94
|
+
coordinates. Prozor matches the exact strings supplied by the consumer; FASTA
|
|
95
|
+
header interpretation, normalization, decoy classification, and persistence
|
|
96
|
+
remain application policy.
|
|
97
|
+
|
|
98
|
+
## Development
|
|
99
|
+
|
|
100
|
+
```bash
|
|
101
|
+
uv sync --group dev --group docs
|
|
102
|
+
.venv/bin/pre-commit install --hook-type pre-commit --hook-type pre-push
|
|
103
|
+
make check
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
`make check` runs Ruff, strict Pyright, dependency validation, branch-coverage
|
|
107
|
+
tests, a strict documentation build, and wheel/sdist validation. All Python
|
|
108
|
+
commands run from the synchronized project `.venv`.
|
|
109
|
+
|
|
110
|
+
## Provenance and license
|
|
111
|
+
|
|
112
|
+
This implementation ports the algorithmic behavior of the R [`prozor`](https://github.com/protViz/prozor) package and consolidates the matching and inference core for Python consumers such as APB and `diann_runner`.
|
|
113
|
+
|
|
114
|
+
The Python distribution is an independent implementation of that algorithm
|
|
115
|
+
rather than a translation of the R source, and it is licensed under MIT. The R
|
|
116
|
+
reference package remains GPL-3, and its license does not extend here: the
|
|
117
|
+
copyright holder of both packages is the same author, and algorithms are not
|
|
118
|
+
themselves subject to copyright.
|
|
119
|
+
|
|
120
|
+
MIT keeps the whole anndata-omics-bridge spine under one permissive license, so
|
|
121
|
+
that consumers such as APB can be redistributed without inheriting copyleft
|
|
122
|
+
obligations from this package.
|
prozor-0.1.0/README.md
ADDED
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
# Prozor
|
|
2
|
+
|
|
3
|
+
[](https://github.com/anndata-omics-bridge/prozor/actions/workflows/quality.yml)
|
|
4
|
+
[](https://anndata-omics-bridge.github.io/prozor/)
|
|
5
|
+
[](https://www.python.org/)
|
|
6
|
+
[](https://spdx.org/licenses/MIT.html)
|
|
7
|
+
|
|
8
|
+
Typed peptide-to-protein matching and deterministic greedy-parsimony protein
|
|
9
|
+
inference.
|
|
10
|
+
|
|
11
|
+
This repository is a Python port of the R [`prozor`](https://github.com/protViz/prozor) package.
|
|
12
|
+
|
|
13
|
+
Prozor offers backend-neutral Aho--Corasick matching over mappings or streaming
|
|
14
|
+
protein records, occurrence-level annotations, unique peptide--protein edges,
|
|
15
|
+
and deterministic protein inference. The core deliberately does not
|
|
16
|
+
parse FASTA files or depend on pandas, AnnData, MuData, MuLink, workflow engines,
|
|
17
|
+
or consumer CLIs.
|
|
18
|
+
|
|
19
|
+
**[Documentation](https://anndata-omics-bridge.github.io/prozor/)** ·
|
|
20
|
+
**[Getting started](https://anndata-omics-bridge.github.io/prozor/getting-started/)** ·
|
|
21
|
+
**[API reference](https://anndata-omics-bridge.github.io/prozor/api/)**
|
|
22
|
+
|
|
23
|
+
Documentation: <https://anndata-omics-bridge.github.io/prozor/>
|
|
24
|
+
|
|
25
|
+
## Installation
|
|
26
|
+
|
|
27
|
+
```bash
|
|
28
|
+
python -m pip install prozor
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
Both matching implementations are installed. The public matching operations
|
|
32
|
+
default to `backend="auto"`, which selects `ahocorasick_rs`. The portable
|
|
33
|
+
`ahocorapy` implementation remains directly selectable and is the automatic
|
|
34
|
+
runtime fallback if Rust cannot be imported. Results record both the requested
|
|
35
|
+
and concrete backend.
|
|
36
|
+
|
|
37
|
+
## Quick start
|
|
38
|
+
|
|
39
|
+
```python
|
|
40
|
+
from dataclasses import dataclass
|
|
41
|
+
|
|
42
|
+
from prozor.api import annotate_peptides_streaming, greedy_parsimony
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@dataclass(frozen=True, slots=True)
|
|
46
|
+
class Protein:
|
|
47
|
+
id: str
|
|
48
|
+
sequence: str
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
matches = annotate_peptides_streaming(
|
|
52
|
+
["PEPTIDE", "SEQUENCE"],
|
|
53
|
+
[
|
|
54
|
+
Protein("P1", "MYPEPTIDESEQUENCE"),
|
|
55
|
+
Protein("P2", "XXSEQUENCEXX"),
|
|
56
|
+
],
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
for match in matches:
|
|
60
|
+
print(match.peptide, match.protein_id, match.start, match.end)
|
|
61
|
+
|
|
62
|
+
edges = {(match.peptide, match.protein_id) for match in matches}
|
|
63
|
+
protein_groups = greedy_parsimony(edges)
|
|
64
|
+
print(protein_groups.to_dict())
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
Matches include nested, overlapping, and repeated sites using half-open
|
|
68
|
+
coordinates. Prozor matches the exact strings supplied by the consumer; FASTA
|
|
69
|
+
header interpretation, normalization, decoy classification, and persistence
|
|
70
|
+
remain application policy.
|
|
71
|
+
|
|
72
|
+
## Development
|
|
73
|
+
|
|
74
|
+
```bash
|
|
75
|
+
uv sync --group dev --group docs
|
|
76
|
+
.venv/bin/pre-commit install --hook-type pre-commit --hook-type pre-push
|
|
77
|
+
make check
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
`make check` runs Ruff, strict Pyright, dependency validation, branch-coverage
|
|
81
|
+
tests, a strict documentation build, and wheel/sdist validation. All Python
|
|
82
|
+
commands run from the synchronized project `.venv`.
|
|
83
|
+
|
|
84
|
+
## Provenance and license
|
|
85
|
+
|
|
86
|
+
This implementation ports the algorithmic behavior of the R [`prozor`](https://github.com/protViz/prozor) package and consolidates the matching and inference core for Python consumers such as APB and `diann_runner`.
|
|
87
|
+
|
|
88
|
+
The Python distribution is an independent implementation of that algorithm
|
|
89
|
+
rather than a translation of the R source, and it is licensed under MIT. The R
|
|
90
|
+
reference package remains GPL-3, and its license does not extend here: the
|
|
91
|
+
copyright holder of both packages is the same author, and algorithms are not
|
|
92
|
+
themselves subject to copyright.
|
|
93
|
+
|
|
94
|
+
MIT keeps the whole anndata-omics-bridge spine under one permissive license, so
|
|
95
|
+
that consumers such as APB can be redistributed without inheriting copyleft
|
|
96
|
+
obligations from this package.
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["uv_build>=0.9.26,<0.10.0"]
|
|
3
|
+
build-backend = "uv_build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "prozor"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Typed peptide-to-protein matching and parsimonious protein inference"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.12"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
keywords = [
|
|
14
|
+
"proteomics",
|
|
15
|
+
"protein inference",
|
|
16
|
+
"peptide matching",
|
|
17
|
+
"mass spectrometry",
|
|
18
|
+
]
|
|
19
|
+
classifiers = [
|
|
20
|
+
"Development Status :: 3 - Alpha",
|
|
21
|
+
"Intended Audience :: Science/Research",
|
|
22
|
+
"Operating System :: OS Independent",
|
|
23
|
+
"Programming Language :: Python :: 3",
|
|
24
|
+
"Programming Language :: Python :: 3.12",
|
|
25
|
+
"Programming Language :: Python :: 3.13",
|
|
26
|
+
"Topic :: Scientific/Engineering :: Bio-Informatics",
|
|
27
|
+
"Typing :: Typed",
|
|
28
|
+
]
|
|
29
|
+
dependencies = [
|
|
30
|
+
"ahocorapy>=1.6,<2",
|
|
31
|
+
"ahocorasick-rs>=1.0,<2",
|
|
32
|
+
]
|
|
33
|
+
|
|
34
|
+
[[project.authors]]
|
|
35
|
+
name = "Witold Wolski"
|
|
36
|
+
email = "wew@fgcz.ethz.ch"
|
|
37
|
+
|
|
38
|
+
[project.urls]
|
|
39
|
+
Homepage = "https://github.com/anndata-omics-bridge/prozor"
|
|
40
|
+
Documentation = "https://anndata-omics-bridge.github.io/prozor/"
|
|
41
|
+
Repository = "https://github.com/anndata-omics-bridge/prozor.git"
|
|
42
|
+
Issues = "https://github.com/anndata-omics-bridge/prozor/issues"
|
|
43
|
+
|
|
44
|
+
[tool.uv]
|
|
45
|
+
default-groups = [
|
|
46
|
+
"dev",
|
|
47
|
+
"docs",
|
|
48
|
+
]
|
|
49
|
+
|
|
50
|
+
[tool.uv.sources.carpet-scan]
|
|
51
|
+
git = "https://github.com/anndata-omics-bridge/carpet_scan"
|
|
52
|
+
rev = "a98f45d21884a95fa79b7660b153bbadb9268e03"
|
|
53
|
+
|
|
54
|
+
[tool.ruff]
|
|
55
|
+
line-length = 100
|
|
56
|
+
target-version = "py312"
|
|
57
|
+
src = [
|
|
58
|
+
"src",
|
|
59
|
+
"tests",
|
|
60
|
+
"benchmarks",
|
|
61
|
+
]
|
|
62
|
+
|
|
63
|
+
[tool.ruff.lint]
|
|
64
|
+
select = [
|
|
65
|
+
"ANN",
|
|
66
|
+
"B",
|
|
67
|
+
"C4",
|
|
68
|
+
"C90",
|
|
69
|
+
"E4",
|
|
70
|
+
"E7",
|
|
71
|
+
"E9",
|
|
72
|
+
"F",
|
|
73
|
+
"I",
|
|
74
|
+
"PGH",
|
|
75
|
+
"PIE",
|
|
76
|
+
"RUF",
|
|
77
|
+
"SIM",
|
|
78
|
+
"UP",
|
|
79
|
+
]
|
|
80
|
+
|
|
81
|
+
[tool.ruff.lint.mccabe]
|
|
82
|
+
max-complexity = 10
|
|
83
|
+
|
|
84
|
+
[tool.ruff.format]
|
|
85
|
+
docstring-code-format = true
|
|
86
|
+
|
|
87
|
+
[tool.pyright]
|
|
88
|
+
include = [
|
|
89
|
+
"src",
|
|
90
|
+
"tests",
|
|
91
|
+
"benchmarks",
|
|
92
|
+
]
|
|
93
|
+
stubPath = "typings"
|
|
94
|
+
venvPath = "."
|
|
95
|
+
venv = ".venv"
|
|
96
|
+
pythonVersion = "3.12"
|
|
97
|
+
typeCheckingMode = "strict"
|
|
98
|
+
reportImportCycles = "error"
|
|
99
|
+
reportMissingTypeStubs = "error"
|
|
100
|
+
reportUnnecessaryTypeIgnoreComment = "error"
|
|
101
|
+
reportImplicitOverride = "error"
|
|
102
|
+
enableTypeIgnoreComments = false
|
|
103
|
+
|
|
104
|
+
[tool.pytest.ini_options]
|
|
105
|
+
addopts = [
|
|
106
|
+
"--strict-config",
|
|
107
|
+
"--strict-markers",
|
|
108
|
+
"-ra",
|
|
109
|
+
]
|
|
110
|
+
testpaths = ["tests"]
|
|
111
|
+
xfail_strict = true
|
|
112
|
+
|
|
113
|
+
[tool.coverage.run]
|
|
114
|
+
branch = true
|
|
115
|
+
source = ["prozor"]
|
|
116
|
+
|
|
117
|
+
[tool.coverage.report]
|
|
118
|
+
fail_under = 80
|
|
119
|
+
show_missing = true
|
|
120
|
+
skip_covered = true
|
|
121
|
+
|
|
122
|
+
[tool.deptry]
|
|
123
|
+
known_first_party = ["prozor"]
|
|
124
|
+
|
|
125
|
+
[dependency-groups]
|
|
126
|
+
dev = [
|
|
127
|
+
"carpet-scan",
|
|
128
|
+
"build>=1.3,<2",
|
|
129
|
+
"deptry>=0.24,<1",
|
|
130
|
+
"import-linter>=2.8,<3",
|
|
131
|
+
"pre-commit>=4,<5",
|
|
132
|
+
"pyright>=1.1.400,<2",
|
|
133
|
+
"pytest>=9,<10",
|
|
134
|
+
"pytest-cov>=7,<8",
|
|
135
|
+
"ruff>=0.15,<1",
|
|
136
|
+
"twine>=6,<7",
|
|
137
|
+
]
|
|
138
|
+
docs = [
|
|
139
|
+
"mkdocstrings[python]>=1.0,<2",
|
|
140
|
+
"pymdown-extensions>=11,<12",
|
|
141
|
+
"zensical==0.0.43",
|
|
142
|
+
]
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["uv_build>=0.9.26,<0.10.0"]
|
|
3
|
+
build-backend = "uv_build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "prozor"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Typed peptide-to-protein matching and parsimonious protein inference"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.12"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [
|
|
14
|
+
{ name = "Witold Wolski", email = "wew@fgcz.ethz.ch" },
|
|
15
|
+
]
|
|
16
|
+
keywords = ["proteomics", "protein inference", "peptide matching", "mass spectrometry"]
|
|
17
|
+
classifiers = [
|
|
18
|
+
"Development Status :: 3 - Alpha",
|
|
19
|
+
"Intended Audience :: Science/Research",
|
|
20
|
+
"Operating System :: OS Independent",
|
|
21
|
+
"Programming Language :: Python :: 3",
|
|
22
|
+
"Programming Language :: Python :: 3.12",
|
|
23
|
+
"Programming Language :: Python :: 3.13",
|
|
24
|
+
"Topic :: Scientific/Engineering :: Bio-Informatics",
|
|
25
|
+
"Typing :: Typed",
|
|
26
|
+
]
|
|
27
|
+
dependencies = [
|
|
28
|
+
"ahocorapy>=1.6,<2",
|
|
29
|
+
"ahocorasick-rs>=1.0,<2",
|
|
30
|
+
]
|
|
31
|
+
|
|
32
|
+
[project.urls]
|
|
33
|
+
Homepage = "https://github.com/anndata-omics-bridge/prozor"
|
|
34
|
+
Documentation = "https://anndata-omics-bridge.github.io/prozor/"
|
|
35
|
+
Repository = "https://github.com/anndata-omics-bridge/prozor.git"
|
|
36
|
+
Issues = "https://github.com/anndata-omics-bridge/prozor/issues"
|
|
37
|
+
|
|
38
|
+
[tool.uv.sources]
|
|
39
|
+
carpet-scan = { git = "https://github.com/anndata-omics-bridge/carpet_scan", rev = "a98f45d21884a95fa79b7660b153bbadb9268e03" }
|
|
40
|
+
|
|
41
|
+
[dependency-groups]
|
|
42
|
+
dev = [
|
|
43
|
+
# Carpet diagnostics. Must live in Prozor's own environment rather than an isolated `uvx` one,
|
|
44
|
+
# because grimp resolves the target package through the Python path.
|
|
45
|
+
"carpet-scan",
|
|
46
|
+
"build>=1.3,<2",
|
|
47
|
+
"deptry>=0.24,<1",
|
|
48
|
+
"import-linter>=2.8,<3",
|
|
49
|
+
"pre-commit>=4,<5",
|
|
50
|
+
"pyright>=1.1.400,<2",
|
|
51
|
+
"pytest>=9,<10",
|
|
52
|
+
"pytest-cov>=7,<8",
|
|
53
|
+
"ruff>=0.15,<1",
|
|
54
|
+
"twine>=6,<7",
|
|
55
|
+
]
|
|
56
|
+
docs = [
|
|
57
|
+
"mkdocstrings[python]>=1.0,<2",
|
|
58
|
+
"pymdown-extensions>=11,<12",
|
|
59
|
+
"zensical==0.0.43",
|
|
60
|
+
]
|
|
61
|
+
|
|
62
|
+
[tool.uv]
|
|
63
|
+
default-groups = ["dev", "docs"]
|
|
64
|
+
|
|
65
|
+
[tool.ruff]
|
|
66
|
+
line-length = 100
|
|
67
|
+
target-version = "py312"
|
|
68
|
+
src = ["src", "tests", "benchmarks"]
|
|
69
|
+
|
|
70
|
+
[tool.ruff.lint]
|
|
71
|
+
select = [
|
|
72
|
+
"ANN",
|
|
73
|
+
"B",
|
|
74
|
+
"C4",
|
|
75
|
+
"C90",
|
|
76
|
+
"E4",
|
|
77
|
+
"E7",
|
|
78
|
+
"E9",
|
|
79
|
+
"F",
|
|
80
|
+
"I",
|
|
81
|
+
"PGH",
|
|
82
|
+
"PIE",
|
|
83
|
+
"RUF",
|
|
84
|
+
"SIM",
|
|
85
|
+
"UP",
|
|
86
|
+
]
|
|
87
|
+
|
|
88
|
+
[tool.ruff.lint.mccabe]
|
|
89
|
+
max-complexity = 10
|
|
90
|
+
|
|
91
|
+
[tool.ruff.format]
|
|
92
|
+
docstring-code-format = true
|
|
93
|
+
|
|
94
|
+
[tool.pyright]
|
|
95
|
+
include = ["src", "tests", "benchmarks"]
|
|
96
|
+
stubPath = "typings"
|
|
97
|
+
venvPath = "."
|
|
98
|
+
venv = ".venv"
|
|
99
|
+
pythonVersion = "3.12"
|
|
100
|
+
typeCheckingMode = "strict"
|
|
101
|
+
reportImportCycles = "error"
|
|
102
|
+
reportMissingTypeStubs = "error"
|
|
103
|
+
reportUnnecessaryTypeIgnoreComment = "error"
|
|
104
|
+
reportImplicitOverride = "error"
|
|
105
|
+
enableTypeIgnoreComments = false
|
|
106
|
+
|
|
107
|
+
[tool.pytest.ini_options]
|
|
108
|
+
addopts = ["--strict-config", "--strict-markers", "-ra"]
|
|
109
|
+
testpaths = ["tests"]
|
|
110
|
+
xfail_strict = true
|
|
111
|
+
|
|
112
|
+
[tool.coverage.run]
|
|
113
|
+
branch = true
|
|
114
|
+
source = ["prozor"]
|
|
115
|
+
|
|
116
|
+
[tool.coverage.report]
|
|
117
|
+
fail_under = 80
|
|
118
|
+
show_missing = true
|
|
119
|
+
skip_covered = true
|
|
120
|
+
|
|
121
|
+
[tool.deptry]
|
|
122
|
+
known_first_party = ["prozor"]
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""The one public prozor module for other anndata_bridge packages."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from prozor.inference.greedy import greedy_parsimony
|
|
6
|
+
from prozor.inference.ties import TieCandidate
|
|
7
|
+
from prozor.matching.annotation import (
|
|
8
|
+
ProteinSequenceRecord,
|
|
9
|
+
annotate_peptides,
|
|
10
|
+
annotate_peptides_streaming,
|
|
11
|
+
)
|
|
12
|
+
from prozor.matching.automaton import resolve_backend
|
|
13
|
+
|
|
14
|
+
__all__ = [
|
|
15
|
+
"ProteinSequenceRecord",
|
|
16
|
+
"TieCandidate",
|
|
17
|
+
"annotate_peptides",
|
|
18
|
+
"annotate_peptides_streaming",
|
|
19
|
+
"greedy_parsimony",
|
|
20
|
+
"resolve_backend",
|
|
21
|
+
]
|
|
File without changes
|
|
@@ -0,0 +1,330 @@
|
|
|
1
|
+
"""Deterministic greedy-parsimony protein inference over unique edges."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Iterable, Iterator, Sequence
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
|
|
8
|
+
from prozor.inference.ties import (
|
|
9
|
+
TieCandidate,
|
|
10
|
+
TieResolver,
|
|
11
|
+
resolve_current_tie,
|
|
12
|
+
)
|
|
13
|
+
|
|
14
|
+
type PeptideProteinEdge = tuple[str, str]
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@dataclass(slots=True)
|
|
18
|
+
class ProteinGroup:
|
|
19
|
+
"""Proteins selected together and the peptides assigned to them."""
|
|
20
|
+
|
|
21
|
+
proteins: list[str]
|
|
22
|
+
peptides: list[str]
|
|
23
|
+
|
|
24
|
+
@property
|
|
25
|
+
def protein_id(self) -> str:
|
|
26
|
+
"""Return the semicolon-joined protein group identifier."""
|
|
27
|
+
return ";".join(self.proteins)
|
|
28
|
+
|
|
29
|
+
@property
|
|
30
|
+
def n_peptides(self) -> int:
|
|
31
|
+
"""Return the number of assigned peptides."""
|
|
32
|
+
return len(self.peptides)
|
|
33
|
+
|
|
34
|
+
@property
|
|
35
|
+
def n_proteins(self) -> int:
|
|
36
|
+
"""Return the number of grouped proteins."""
|
|
37
|
+
return len(self.proteins)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@dataclass(slots=True)
|
|
41
|
+
class GreedyResult:
|
|
42
|
+
"""Protein groups selected by greedy parsimony."""
|
|
43
|
+
|
|
44
|
+
groups: list[ProteinGroup]
|
|
45
|
+
|
|
46
|
+
def __len__(self) -> int:
|
|
47
|
+
return len(self.groups)
|
|
48
|
+
|
|
49
|
+
def __iter__(self) -> Iterator[ProteinGroup]:
|
|
50
|
+
return iter(self.groups)
|
|
51
|
+
|
|
52
|
+
@property
|
|
53
|
+
def n_proteins(self) -> int:
|
|
54
|
+
"""Return the total number of selected and subsumed proteins."""
|
|
55
|
+
return sum(group.n_proteins for group in self.groups)
|
|
56
|
+
|
|
57
|
+
@property
|
|
58
|
+
def n_groups(self) -> int:
|
|
59
|
+
"""Return the number of inferred protein groups."""
|
|
60
|
+
return len(self.groups)
|
|
61
|
+
|
|
62
|
+
@property
|
|
63
|
+
def n_peptides(self) -> int:
|
|
64
|
+
"""Return the number of distinct assigned peptides."""
|
|
65
|
+
return len({peptide for group in self.groups for peptide in group.peptides})
|
|
66
|
+
|
|
67
|
+
def to_dict(self) -> dict[str, str]:
|
|
68
|
+
"""Return a peptide-to-protein-group mapping."""
|
|
69
|
+
return {peptide: group.protein_id for group in self.groups for peptide in group.peptides}
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
@dataclass(frozen=True, slots=True)
|
|
73
|
+
class _Selection:
|
|
74
|
+
winners: tuple[int, ...]
|
|
75
|
+
covered_peptides: frozenset[int]
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def greedy_parsimony(
|
|
79
|
+
edges: Iterable[PeptideProteinEdge],
|
|
80
|
+
resolve_tie: TieResolver = resolve_current_tie,
|
|
81
|
+
subsume: bool = True,
|
|
82
|
+
) -> GreedyResult:
|
|
83
|
+
"""Select a deterministic parsimonious set of protein groups.
|
|
84
|
+
|
|
85
|
+
Repeated ``(peptide, protein)`` edges collapse before inference. Proteins
|
|
86
|
+
with identical remaining peptide evidence are grouped. Disjoint equal
|
|
87
|
+
candidates are ordered deterministically, while overlapping non-identical
|
|
88
|
+
candidates are passed to ``resolve_tie``.
|
|
89
|
+
|
|
90
|
+
Args:
|
|
91
|
+
edges: Peptide and protein identifier pairs.
|
|
92
|
+
resolve_tie: Operation selecting one consequential tie candidate.
|
|
93
|
+
subsume: Whether proteins supported by a subset of selected evidence
|
|
94
|
+
remain in the selected group.
|
|
95
|
+
|
|
96
|
+
Returns:
|
|
97
|
+
Deterministically ordered inferred protein groups.
|
|
98
|
+
"""
|
|
99
|
+
peptides, proteins, peptide_proteins, protein_peptides = _build_incidence(edges)
|
|
100
|
+
active_peptides = set(range(len(peptides)))
|
|
101
|
+
active_proteins = set(range(len(proteins)))
|
|
102
|
+
counts = [len(protein_evidence) for protein_evidence in protein_peptides]
|
|
103
|
+
groups: list[ProteinGroup] = []
|
|
104
|
+
|
|
105
|
+
while active_peptides and active_proteins:
|
|
106
|
+
selections = _select_winners(
|
|
107
|
+
active_peptides,
|
|
108
|
+
active_proteins,
|
|
109
|
+
protein_peptides,
|
|
110
|
+
counts,
|
|
111
|
+
peptides,
|
|
112
|
+
proteins,
|
|
113
|
+
resolve_tie,
|
|
114
|
+
)
|
|
115
|
+
if not selections:
|
|
116
|
+
break
|
|
117
|
+
for selection in selections:
|
|
118
|
+
subsumed = (
|
|
119
|
+
_find_subsumed(
|
|
120
|
+
selection,
|
|
121
|
+
active_peptides,
|
|
122
|
+
active_proteins,
|
|
123
|
+
peptide_proteins,
|
|
124
|
+
protein_peptides,
|
|
125
|
+
)
|
|
126
|
+
if subsume
|
|
127
|
+
else ()
|
|
128
|
+
)
|
|
129
|
+
group_indices = tuple(sorted((*selection.winners, *subsumed)))
|
|
130
|
+
groups.append(
|
|
131
|
+
ProteinGroup(
|
|
132
|
+
proteins=sorted(proteins[index] for index in group_indices),
|
|
133
|
+
peptides=sorted(peptides[index] for index in selection.covered_peptides),
|
|
134
|
+
)
|
|
135
|
+
)
|
|
136
|
+
active_peptides.difference_update(selection.covered_peptides)
|
|
137
|
+
active_proteins.difference_update(group_indices)
|
|
138
|
+
_decrement_counts(
|
|
139
|
+
selection.covered_peptides,
|
|
140
|
+
active_proteins,
|
|
141
|
+
peptide_proteins,
|
|
142
|
+
counts,
|
|
143
|
+
)
|
|
144
|
+
|
|
145
|
+
groups.sort(
|
|
146
|
+
key=lambda group: (
|
|
147
|
+
-group.n_peptides,
|
|
148
|
+
-group.n_proteins,
|
|
149
|
+
tuple(group.proteins),
|
|
150
|
+
)
|
|
151
|
+
)
|
|
152
|
+
return GreedyResult(groups=groups)
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def _build_incidence(
|
|
156
|
+
edges: Iterable[PeptideProteinEdge],
|
|
157
|
+
) -> tuple[list[str], list[str], list[frozenset[int]], list[frozenset[int]]]:
|
|
158
|
+
unique_edges = set(edges)
|
|
159
|
+
peptides = sorted({peptide for peptide, _protein in unique_edges})
|
|
160
|
+
proteins = sorted({protein for _peptide, protein in unique_edges})
|
|
161
|
+
peptide_indices = {peptide: index for index, peptide in enumerate(peptides)}
|
|
162
|
+
protein_indices = {protein: index for index, protein in enumerate(proteins)}
|
|
163
|
+
peptide_proteins: list[set[int]] = [set() for _peptide in peptides]
|
|
164
|
+
protein_peptides: list[set[int]] = [set() for _protein in proteins]
|
|
165
|
+
for peptide, protein in unique_edges:
|
|
166
|
+
peptide_index = peptide_indices[peptide]
|
|
167
|
+
protein_index = protein_indices[protein]
|
|
168
|
+
peptide_proteins[peptide_index].add(protein_index)
|
|
169
|
+
protein_peptides[protein_index].add(peptide_index)
|
|
170
|
+
return (
|
|
171
|
+
peptides,
|
|
172
|
+
proteins,
|
|
173
|
+
[frozenset(indices) for indices in peptide_proteins],
|
|
174
|
+
[frozenset(indices) for indices in protein_peptides],
|
|
175
|
+
)
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _select_winners(
|
|
179
|
+
active_peptides: set[int],
|
|
180
|
+
active_proteins: set[int],
|
|
181
|
+
protein_peptides: Sequence[frozenset[int]],
|
|
182
|
+
counts: Sequence[int],
|
|
183
|
+
peptide_names: Sequence[str],
|
|
184
|
+
protein_names: Sequence[str],
|
|
185
|
+
resolve_tie: TieResolver,
|
|
186
|
+
) -> tuple[_Selection, ...]:
|
|
187
|
+
ordered_active = sorted(active_proteins)
|
|
188
|
+
if not ordered_active:
|
|
189
|
+
return ()
|
|
190
|
+
max_count = max(counts[index] for index in ordered_active)
|
|
191
|
+
if max_count == 0:
|
|
192
|
+
return ()
|
|
193
|
+
max_proteins = [index for index in ordered_active if counts[index] == max_count]
|
|
194
|
+
signature_groups: dict[frozenset[int], list[int]] = {}
|
|
195
|
+
for index in max_proteins:
|
|
196
|
+
signature = frozenset(protein_peptides[index] & active_peptides)
|
|
197
|
+
signature_groups.setdefault(signature, []).append(index)
|
|
198
|
+
|
|
199
|
+
grouped = tuple(
|
|
200
|
+
tuple(indices)
|
|
201
|
+
for _signature, indices in sorted(
|
|
202
|
+
signature_groups.items(),
|
|
203
|
+
key=lambda item: tuple(protein_names[index] for index in item[1]),
|
|
204
|
+
)
|
|
205
|
+
)
|
|
206
|
+
signatures = tuple(
|
|
207
|
+
frozenset(protein_peptides[indices[0]] & active_peptides) for indices in grouped
|
|
208
|
+
)
|
|
209
|
+
components = _overlap_components(signatures)
|
|
210
|
+
selections = [
|
|
211
|
+
_selection_for_component(
|
|
212
|
+
component,
|
|
213
|
+
grouped,
|
|
214
|
+
signatures,
|
|
215
|
+
peptide_names,
|
|
216
|
+
protein_names,
|
|
217
|
+
resolve_tie,
|
|
218
|
+
)
|
|
219
|
+
for component in components
|
|
220
|
+
]
|
|
221
|
+
return tuple(sorted(selections, key=lambda selection: _selection_key(selection, protein_names)))
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _selection_key(
|
|
225
|
+
selection: _Selection,
|
|
226
|
+
protein_names: Sequence[str],
|
|
227
|
+
) -> tuple[int, tuple[str, ...]]:
|
|
228
|
+
return (
|
|
229
|
+
-len(selection.winners),
|
|
230
|
+
tuple(protein_names[index] for index in selection.winners),
|
|
231
|
+
)
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def _overlap_components(signatures: Sequence[frozenset[int]]) -> tuple[tuple[int, ...], ...]:
|
|
235
|
+
peptide_groups: dict[int, list[int]] = {}
|
|
236
|
+
for group_index, signature in enumerate(signatures):
|
|
237
|
+
for peptide in signature:
|
|
238
|
+
peptide_groups.setdefault(peptide, []).append(group_index)
|
|
239
|
+
unseen = set(range(len(signatures)))
|
|
240
|
+
components: list[tuple[int, ...]] = []
|
|
241
|
+
while unseen:
|
|
242
|
+
pending = [min(unseen)]
|
|
243
|
+
component: set[int] = set()
|
|
244
|
+
while pending:
|
|
245
|
+
group_index = pending.pop()
|
|
246
|
+
if group_index not in unseen:
|
|
247
|
+
continue
|
|
248
|
+
unseen.remove(group_index)
|
|
249
|
+
component.add(group_index)
|
|
250
|
+
for peptide in signatures[group_index]:
|
|
251
|
+
pending.extend(peptide_groups[peptide])
|
|
252
|
+
components.append(tuple(sorted(component)))
|
|
253
|
+
return tuple(components)
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def _selection_for_component(
|
|
257
|
+
component: tuple[int, ...],
|
|
258
|
+
groups: Sequence[tuple[int, ...]],
|
|
259
|
+
signatures: Sequence[frozenset[int]],
|
|
260
|
+
peptide_names: Sequence[str],
|
|
261
|
+
protein_names: Sequence[str],
|
|
262
|
+
resolve_tie: TieResolver,
|
|
263
|
+
) -> _Selection:
|
|
264
|
+
component_groups = tuple(groups[index] for index in component)
|
|
265
|
+
component_signatures = tuple(signatures[index] for index in component)
|
|
266
|
+
winners = (
|
|
267
|
+
component_groups[0]
|
|
268
|
+
if len(component_groups) == 1
|
|
269
|
+
else _resolve_overlapping_tie(
|
|
270
|
+
component_groups,
|
|
271
|
+
component_signatures,
|
|
272
|
+
peptide_names,
|
|
273
|
+
protein_names,
|
|
274
|
+
resolve_tie,
|
|
275
|
+
)
|
|
276
|
+
)
|
|
277
|
+
winner_index = component_groups.index(winners)
|
|
278
|
+
return _Selection(winners=winners, covered_peptides=component_signatures[winner_index])
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
def _resolve_overlapping_tie(
|
|
282
|
+
groups: Sequence[tuple[int, ...]],
|
|
283
|
+
signatures: Sequence[frozenset[int]],
|
|
284
|
+
peptide_names: Sequence[str],
|
|
285
|
+
protein_names: Sequence[str],
|
|
286
|
+
resolve_tie: TieResolver,
|
|
287
|
+
) -> tuple[int, ...]:
|
|
288
|
+
candidates = tuple(
|
|
289
|
+
TieCandidate(
|
|
290
|
+
proteins=tuple(protein_names[index] for index in group),
|
|
291
|
+
unexplained_peptides=frozenset(peptide_names[index] for index in signature),
|
|
292
|
+
)
|
|
293
|
+
for group, signature in zip(groups, signatures, strict=True)
|
|
294
|
+
)
|
|
295
|
+
selected = resolve_tie(candidates)
|
|
296
|
+
try:
|
|
297
|
+
selected_index = candidates.index(selected)
|
|
298
|
+
except ValueError as error:
|
|
299
|
+
raise ValueError("tie resolver must return one of the candidates it received") from error
|
|
300
|
+
return groups[selected_index]
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
def _find_subsumed(
|
|
304
|
+
selection: _Selection,
|
|
305
|
+
active_peptides: set[int],
|
|
306
|
+
active_proteins: set[int],
|
|
307
|
+
peptide_proteins: Sequence[frozenset[int]],
|
|
308
|
+
protein_peptides: Sequence[frozenset[int]],
|
|
309
|
+
) -> tuple[int, ...]:
|
|
310
|
+
candidates = {
|
|
311
|
+
protein for peptide in selection.covered_peptides for protein in peptide_proteins[peptide]
|
|
312
|
+
}
|
|
313
|
+
candidates.difference_update(selection.winners)
|
|
314
|
+
candidates.intersection_update(active_proteins)
|
|
315
|
+
return tuple(
|
|
316
|
+
protein
|
|
317
|
+
for protein in sorted(candidates)
|
|
318
|
+
if protein_peptides[protein] & active_peptides <= selection.covered_peptides
|
|
319
|
+
)
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def _decrement_counts(
|
|
323
|
+
removed_peptides: frozenset[int],
|
|
324
|
+
active_proteins: set[int],
|
|
325
|
+
peptide_proteins: Sequence[frozenset[int]],
|
|
326
|
+
counts: list[int],
|
|
327
|
+
) -> None:
|
|
328
|
+
for peptide in removed_peptides:
|
|
329
|
+
for protein in peptide_proteins[peptide] & active_proteins:
|
|
330
|
+
counts[protein] -= 1
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
"""Tie candidates and resolution for greedy protein inference."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Callable, Sequence
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@dataclass(frozen=True, slots=True)
|
|
10
|
+
class TieCandidate:
|
|
11
|
+
"""One indistinguishable protein group competing in a consequential tie."""
|
|
12
|
+
|
|
13
|
+
proteins: tuple[str, ...]
|
|
14
|
+
unexplained_peptides: frozenset[str]
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
type TieResolver = Callable[[Sequence[TieCandidate]], TieCandidate]
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def resolve_current_tie(candidates: Sequence[TieCandidate]) -> TieCandidate:
|
|
21
|
+
"""Preserve Prozor's deterministic group-size and accession tie rule."""
|
|
22
|
+
if not candidates:
|
|
23
|
+
raise ValueError("at least one tie candidate is required")
|
|
24
|
+
return min(candidates, key=lambda candidate: (-len(candidate.proteins), candidate.proteins))
|
|
File without changes
|
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
"""Peptide-to-protein annotation over mappings or streaming protein records."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from bisect import bisect_right
|
|
6
|
+
from collections.abc import Iterable, Iterator, Mapping, Sequence
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from itertools import accumulate, islice
|
|
9
|
+
from typing import Protocol
|
|
10
|
+
|
|
11
|
+
from prozor.matching.automaton import (
|
|
12
|
+
BackendName,
|
|
13
|
+
BackendRequest,
|
|
14
|
+
create_automaton,
|
|
15
|
+
resolve_backend,
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
_SEPARATOR = "\n"
|
|
19
|
+
_BATCH_SIZE = 10_000
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class ProteinSequenceRecord(Protocol):
|
|
23
|
+
"""Smallest protein-record capability required by peptide matching."""
|
|
24
|
+
|
|
25
|
+
@property
|
|
26
|
+
def id(self) -> str:
|
|
27
|
+
"""Return the protein identifier attached to matching results."""
|
|
28
|
+
...
|
|
29
|
+
|
|
30
|
+
@property
|
|
31
|
+
def sequence(self) -> str:
|
|
32
|
+
"""Return the protein sequence to search."""
|
|
33
|
+
...
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass(frozen=True, slots=True)
|
|
37
|
+
class PeptideAnnotation:
|
|
38
|
+
"""One peptide occurrence in a protein sequence."""
|
|
39
|
+
|
|
40
|
+
peptide: str
|
|
41
|
+
protein_id: str
|
|
42
|
+
start: int
|
|
43
|
+
end: int
|
|
44
|
+
|
|
45
|
+
@property
|
|
46
|
+
def length(self) -> int:
|
|
47
|
+
"""Return the matched peptide length."""
|
|
48
|
+
return len(self.peptide)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
@dataclass(slots=True)
|
|
52
|
+
class AnnotationResult:
|
|
53
|
+
"""Collection of peptide matches and matching-backend provenance."""
|
|
54
|
+
|
|
55
|
+
annotations: list[PeptideAnnotation]
|
|
56
|
+
requested_backend: BackendRequest = "ahocorapy"
|
|
57
|
+
resolved_backend: BackendName = "ahocorapy"
|
|
58
|
+
|
|
59
|
+
def __len__(self) -> int:
|
|
60
|
+
return len(self.annotations)
|
|
61
|
+
|
|
62
|
+
def __iter__(self) -> Iterator[PeptideAnnotation]:
|
|
63
|
+
return iter(self.annotations)
|
|
64
|
+
|
|
65
|
+
@property
|
|
66
|
+
def peptides(self) -> set[str]:
|
|
67
|
+
"""Return the distinct matched peptide sequences."""
|
|
68
|
+
return {annotation.peptide for annotation in self.annotations}
|
|
69
|
+
|
|
70
|
+
@property
|
|
71
|
+
def proteins(self) -> set[str]:
|
|
72
|
+
"""Return the distinct matched protein identifiers."""
|
|
73
|
+
return {annotation.protein_id for annotation in self.annotations}
|
|
74
|
+
|
|
75
|
+
def filter_tryptic(
|
|
76
|
+
self,
|
|
77
|
+
proteins: Mapping[str, str],
|
|
78
|
+
prefix_residues: str = "RK",
|
|
79
|
+
allow_n_term: bool = True,
|
|
80
|
+
allow_after_init_met: bool = True,
|
|
81
|
+
) -> AnnotationResult:
|
|
82
|
+
"""Return matches with a supported tryptic N-terminal context."""
|
|
83
|
+
filtered: list[PeptideAnnotation] = []
|
|
84
|
+
for annotation in self.annotations:
|
|
85
|
+
sequence = proteins.get(annotation.protein_id, "")
|
|
86
|
+
if not sequence:
|
|
87
|
+
continue
|
|
88
|
+
valid_prefix = (
|
|
89
|
+
(annotation.start == 0 and allow_n_term)
|
|
90
|
+
or (annotation.start == 1 and allow_after_init_met and sequence.startswith("M"))
|
|
91
|
+
or (annotation.start > 0 and sequence[annotation.start - 1] in prefix_residues)
|
|
92
|
+
)
|
|
93
|
+
if valid_prefix:
|
|
94
|
+
filtered.append(annotation)
|
|
95
|
+
return AnnotationResult(
|
|
96
|
+
annotations=filtered,
|
|
97
|
+
requested_backend=self.requested_backend,
|
|
98
|
+
resolved_backend=self.resolved_backend,
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def annotate_peptides(
|
|
103
|
+
peptides: Iterable[str],
|
|
104
|
+
proteins: Mapping[str, str],
|
|
105
|
+
backend: str = "auto",
|
|
106
|
+
filter_tryptic: bool = False,
|
|
107
|
+
) -> AnnotationResult:
|
|
108
|
+
"""Annotate peptides against an in-memory protein mapping."""
|
|
109
|
+
result = _annotate(peptides, [(tuple(proteins), tuple(proteins.values()))], backend)
|
|
110
|
+
return result.filter_tryptic(proteins) if filter_tryptic else result
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def annotate_peptides_streaming(
|
|
114
|
+
peptides: Iterable[str],
|
|
115
|
+
protein_records: Iterable[ProteinSequenceRecord],
|
|
116
|
+
backend: str = "auto",
|
|
117
|
+
) -> AnnotationResult:
|
|
118
|
+
"""Annotate peptides against one-pass records exposing ``id`` and ``sequence``."""
|
|
119
|
+
records = iter(protein_records)
|
|
120
|
+
batches = (
|
|
121
|
+
(tuple(record.id for record in batch), tuple(record.sequence for record in batch))
|
|
122
|
+
for batch in iter(lambda: tuple(islice(records, _BATCH_SIZE)), ())
|
|
123
|
+
)
|
|
124
|
+
return _annotate(peptides, batches, backend)
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def _annotate(
|
|
128
|
+
peptides: Iterable[str],
|
|
129
|
+
batches: Iterable[tuple[Sequence[str], Sequence[str]]],
|
|
130
|
+
backend: str,
|
|
131
|
+
) -> AnnotationResult:
|
|
132
|
+
peptide_list = list(dict.fromkeys(peptides))
|
|
133
|
+
if not peptide_list:
|
|
134
|
+
return AnnotationResult(
|
|
135
|
+
annotations=[],
|
|
136
|
+
requested_backend=_requested_backend(backend),
|
|
137
|
+
resolved_backend=resolve_backend(backend),
|
|
138
|
+
)
|
|
139
|
+
if any(_SEPARATOR in peptide for peptide in peptide_list):
|
|
140
|
+
raise ValueError("peptides must not contain line breaks")
|
|
141
|
+
|
|
142
|
+
automaton = create_automaton(peptide_list, backend=backend)
|
|
143
|
+
annotations: list[PeptideAnnotation] = []
|
|
144
|
+
for ids, sequences in batches:
|
|
145
|
+
# One backend call per batch; the separator stops matches across sequences.
|
|
146
|
+
starts = list(accumulate((len(sequence) + 1 for sequence in sequences), initial=0))
|
|
147
|
+
for match in automaton.find_all(_SEPARATOR.join(sequences)):
|
|
148
|
+
index = bisect_right(starts, match.start) - 1
|
|
149
|
+
offset = starts[index]
|
|
150
|
+
annotations.append(
|
|
151
|
+
PeptideAnnotation(
|
|
152
|
+
match.keyword, ids[index], match.start - offset, match.end - offset
|
|
153
|
+
)
|
|
154
|
+
)
|
|
155
|
+
return AnnotationResult(
|
|
156
|
+
annotations=annotations,
|
|
157
|
+
requested_backend=automaton.requested_backend,
|
|
158
|
+
resolved_backend=automaton.resolved_backend,
|
|
159
|
+
)
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def _requested_backend(backend: str) -> BackendRequest:
|
|
163
|
+
resolve_backend(backend)
|
|
164
|
+
if backend == "ahocorapy":
|
|
165
|
+
return "ahocorapy"
|
|
166
|
+
if backend == "ahocorasick_rs":
|
|
167
|
+
return "ahocorasick_rs"
|
|
168
|
+
return "auto"
|
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
"""Backend-neutral Aho--Corasick peptide matching."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from abc import ABC, abstractmethod
|
|
6
|
+
from collections.abc import Iterable, Iterator
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from importlib.util import find_spec
|
|
9
|
+
from typing import Literal, cast, override
|
|
10
|
+
|
|
11
|
+
type BackendName = Literal["ahocorapy", "ahocorasick_rs"]
|
|
12
|
+
type BackendRequest = Literal["auto", "ahocorapy", "ahocorasick_rs"]
|
|
13
|
+
|
|
14
|
+
_VALID_BACKENDS = frozenset({"auto", "ahocorapy", "ahocorasick_rs"})
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@dataclass(frozen=True, slots=True)
|
|
18
|
+
class Match:
|
|
19
|
+
"""One half-open keyword match in a searched sequence."""
|
|
20
|
+
|
|
21
|
+
keyword: str
|
|
22
|
+
start: int
|
|
23
|
+
end: int
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class AhoCorasickBase(ABC):
|
|
27
|
+
"""Common interface implemented by every matching backend."""
|
|
28
|
+
|
|
29
|
+
requested_backend: BackendRequest
|
|
30
|
+
resolved_backend: BackendName
|
|
31
|
+
|
|
32
|
+
@abstractmethod
|
|
33
|
+
def find_all(self, text: str) -> Iterator[Match]:
|
|
34
|
+
"""Yield every match, including nested and overlapping matches."""
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class AhoCorasickPure(AhoCorasickBase):
|
|
38
|
+
"""Portable matcher implemented with :mod:`ahocorapy`."""
|
|
39
|
+
|
|
40
|
+
def __init__(
|
|
41
|
+
self,
|
|
42
|
+
keywords: Iterable[str],
|
|
43
|
+
*,
|
|
44
|
+
case_sensitive: bool = True,
|
|
45
|
+
requested_backend: BackendRequest = "ahocorapy",
|
|
46
|
+
) -> None:
|
|
47
|
+
from ahocorapy.keywordtree import KeywordTree
|
|
48
|
+
|
|
49
|
+
self.requested_backend = requested_backend
|
|
50
|
+
self.resolved_backend: BackendName = "ahocorapy"
|
|
51
|
+
self._tree = KeywordTree(case_insensitive=not case_sensitive)
|
|
52
|
+
for keyword in _unique_keywords(keywords):
|
|
53
|
+
self._tree.add(keyword)
|
|
54
|
+
self._tree.finalize()
|
|
55
|
+
|
|
56
|
+
@override
|
|
57
|
+
def find_all(self, text: str) -> Iterator[Match]:
|
|
58
|
+
"""Yield all pure-Python matches in ``text``."""
|
|
59
|
+
for keyword, start in self._tree.search_all(text):
|
|
60
|
+
yield Match(keyword=keyword, start=start, end=start + len(keyword))
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class AhoCorasickRust(AhoCorasickBase):
|
|
64
|
+
"""Accelerated matcher implemented with :mod:`ahocorasick_rs`."""
|
|
65
|
+
|
|
66
|
+
def __init__(
|
|
67
|
+
self,
|
|
68
|
+
keywords: Iterable[str],
|
|
69
|
+
*,
|
|
70
|
+
case_sensitive: bool = True,
|
|
71
|
+
requested_backend: BackendRequest = "ahocorasick_rs",
|
|
72
|
+
) -> None:
|
|
73
|
+
import ahocorasick_rs
|
|
74
|
+
|
|
75
|
+
self.requested_backend = requested_backend
|
|
76
|
+
self.resolved_backend: BackendName = "ahocorasick_rs"
|
|
77
|
+
self._keywords = _unique_keywords(keywords)
|
|
78
|
+
self._case_sensitive = case_sensitive
|
|
79
|
+
indexed_keywords = (
|
|
80
|
+
self._keywords
|
|
81
|
+
if case_sensitive
|
|
82
|
+
else tuple(keyword.lower() for keyword in self._keywords)
|
|
83
|
+
)
|
|
84
|
+
self._automaton = ahocorasick_rs.AhoCorasick(indexed_keywords)
|
|
85
|
+
|
|
86
|
+
@override
|
|
87
|
+
def find_all(self, text: str) -> Iterator[Match]:
|
|
88
|
+
"""Yield all Rust matches in ``text``."""
|
|
89
|
+
search_text = text if self._case_sensitive else text.lower()
|
|
90
|
+
for index, start, end in self._automaton.find_matches_as_indexes(
|
|
91
|
+
search_text,
|
|
92
|
+
overlapping=True,
|
|
93
|
+
):
|
|
94
|
+
yield Match(keyword=self._keywords[index], start=start, end=end)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def create_automaton(
|
|
98
|
+
keywords: Iterable[str],
|
|
99
|
+
backend: str = "auto",
|
|
100
|
+
case_sensitive: bool = True,
|
|
101
|
+
) -> AhoCorasickBase:
|
|
102
|
+
"""Create a matcher and expose both requested and resolved backend names.
|
|
103
|
+
|
|
104
|
+
Args:
|
|
105
|
+
keywords: Peptide patterns to search for.
|
|
106
|
+
backend: ``auto``, ``ahocorapy``, or ``ahocorasick_rs``.
|
|
107
|
+
case_sensitive: Whether matching distinguishes letter case.
|
|
108
|
+
|
|
109
|
+
Returns:
|
|
110
|
+
A backend-neutral matcher.
|
|
111
|
+
|
|
112
|
+
Raises:
|
|
113
|
+
ImportError: The Rust backend was requested but is not installed.
|
|
114
|
+
ValueError: ``backend`` is not supported.
|
|
115
|
+
"""
|
|
116
|
+
requested_backend = _validate_backend(backend)
|
|
117
|
+
resolved_backend = resolve_backend(requested_backend)
|
|
118
|
+
if resolved_backend == "ahocorasick_rs":
|
|
119
|
+
return AhoCorasickRust(
|
|
120
|
+
keywords,
|
|
121
|
+
case_sensitive=case_sensitive,
|
|
122
|
+
requested_backend=requested_backend,
|
|
123
|
+
)
|
|
124
|
+
return AhoCorasickPure(
|
|
125
|
+
keywords,
|
|
126
|
+
case_sensitive=case_sensitive,
|
|
127
|
+
requested_backend=requested_backend,
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def resolve_backend(backend: str = "auto") -> BackendName:
|
|
132
|
+
"""Resolve a requested backend to the concrete implementation name."""
|
|
133
|
+
requested_backend = _validate_backend(backend)
|
|
134
|
+
if requested_backend == "ahocorasick_rs":
|
|
135
|
+
if not _rust_available():
|
|
136
|
+
raise ImportError(
|
|
137
|
+
"backend 'ahocorasick_rs' requires a working ahocorasick-rs installation"
|
|
138
|
+
)
|
|
139
|
+
return "ahocorasick_rs"
|
|
140
|
+
if requested_backend == "ahocorapy":
|
|
141
|
+
return "ahocorapy"
|
|
142
|
+
return "ahocorasick_rs" if _rust_available() else "ahocorapy"
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def get_available_backends() -> list[BackendName]:
|
|
146
|
+
"""Return the concrete matching backends available in this environment."""
|
|
147
|
+
backends: list[BackendName] = ["ahocorapy"]
|
|
148
|
+
if _rust_available():
|
|
149
|
+
backends.append("ahocorasick_rs")
|
|
150
|
+
return backends
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _validate_backend(backend: str) -> BackendRequest:
|
|
154
|
+
if backend not in _VALID_BACKENDS:
|
|
155
|
+
raise ValueError(f"backend must be one of {sorted(_VALID_BACKENDS)}, got {backend!r}")
|
|
156
|
+
return cast(BackendRequest, backend)
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _rust_available() -> bool:
|
|
160
|
+
return find_spec("ahocorasick_rs") is not None
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def _unique_keywords(keywords: Iterable[str]) -> tuple[str, ...]:
|
|
164
|
+
return tuple(dict.fromkeys(keywords))
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|