prozor 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
prozor-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Witold Wolski
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
prozor-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,122 @@
1
+ Metadata-Version: 2.4
2
+ Name: prozor
3
+ Version: 0.1.0
4
+ Summary: Typed peptide-to-protein matching and parsimonious protein inference
5
+ Keywords: proteomics,protein inference,peptide matching,mass spectrometry
6
+ Author: Witold Wolski
7
+ Author-email: Witold Wolski <wew@fgcz.ethz.ch>
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Classifier: Development Status :: 3 - Alpha
11
+ Classifier: Intended Audience :: Science/Research
12
+ Classifier: Operating System :: OS Independent
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Programming Language :: Python :: 3.12
15
+ Classifier: Programming Language :: Python :: 3.13
16
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
17
+ Classifier: Typing :: Typed
18
+ Requires-Dist: ahocorapy>=1.6,<2
19
+ Requires-Dist: ahocorasick-rs>=1.0,<2
20
+ Requires-Python: >=3.12
21
+ Project-URL: Homepage, https://github.com/anndata-omics-bridge/prozor
22
+ Project-URL: Documentation, https://anndata-omics-bridge.github.io/prozor/
23
+ Project-URL: Repository, https://github.com/anndata-omics-bridge/prozor.git
24
+ Project-URL: Issues, https://github.com/anndata-omics-bridge/prozor/issues
25
+ Description-Content-Type: text/markdown
26
+
27
+ # Prozor
28
+
29
+ [![Quality](https://github.com/anndata-omics-bridge/prozor/actions/workflows/quality.yml/badge.svg)](https://github.com/anndata-omics-bridge/prozor/actions/workflows/quality.yml)
30
+ [![Documentation](https://github.com/anndata-omics-bridge/prozor/actions/workflows/docs.yml/badge.svg)](https://anndata-omics-bridge.github.io/prozor/)
31
+ [![Python](https://img.shields.io/badge/Python-%E2%89%A53.12-3776AB.svg)](https://www.python.org/)
32
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://spdx.org/licenses/MIT.html)
33
+
34
+ Typed peptide-to-protein matching and deterministic greedy-parsimony protein
35
+ inference.
36
+
37
+ This repository is a Python port of the R [`prozor`](https://github.com/protViz/prozor) package.
38
+
39
+ Prozor offers backend-neutral Aho--Corasick matching over mappings or streaming
40
+ protein records, occurrence-level annotations, unique peptide--protein edges,
41
+ and deterministic protein inference. The core deliberately does not
42
+ parse FASTA files or depend on pandas, AnnData, MuData, MuLink, workflow engines,
43
+ or consumer CLIs.
44
+
45
+ **[Documentation](https://anndata-omics-bridge.github.io/prozor/)** ·
46
+ **[Getting started](https://anndata-omics-bridge.github.io/prozor/getting-started/)** ·
47
+ **[API reference](https://anndata-omics-bridge.github.io/prozor/api/)**
48
+
49
+ Documentation: <https://anndata-omics-bridge.github.io/prozor/>
50
+
51
+ ## Installation
52
+
53
+ ```bash
54
+ python -m pip install prozor
55
+ ```
56
+
57
+ Both matching implementations are installed. The public matching operations
58
+ default to `backend="auto"`, which selects `ahocorasick_rs`. The portable
59
+ `ahocorapy` implementation remains directly selectable and is the automatic
60
+ runtime fallback if Rust cannot be imported. Results record both the requested
61
+ and concrete backend.
62
+
63
+ ## Quick start
64
+
65
+ ```python
66
+ from dataclasses import dataclass
67
+
68
+ from prozor.api import annotate_peptides_streaming, greedy_parsimony
69
+
70
+
71
+ @dataclass(frozen=True, slots=True)
72
+ class Protein:
73
+ id: str
74
+ sequence: str
75
+
76
+
77
+ matches = annotate_peptides_streaming(
78
+ ["PEPTIDE", "SEQUENCE"],
79
+ [
80
+ Protein("P1", "MYPEPTIDESEQUENCE"),
81
+ Protein("P2", "XXSEQUENCEXX"),
82
+ ],
83
+ )
84
+
85
+ for match in matches:
86
+ print(match.peptide, match.protein_id, match.start, match.end)
87
+
88
+ edges = {(match.peptide, match.protein_id) for match in matches}
89
+ protein_groups = greedy_parsimony(edges)
90
+ print(protein_groups.to_dict())
91
+ ```
92
+
93
+ Matches include nested, overlapping, and repeated sites using half-open
94
+ coordinates. Prozor matches the exact strings supplied by the consumer; FASTA
95
+ header interpretation, normalization, decoy classification, and persistence
96
+ remain application policy.
97
+
98
+ ## Development
99
+
100
+ ```bash
101
+ uv sync --group dev --group docs
102
+ .venv/bin/pre-commit install --hook-type pre-commit --hook-type pre-push
103
+ make check
104
+ ```
105
+
106
+ `make check` runs Ruff, strict Pyright, dependency validation, branch-coverage
107
+ tests, a strict documentation build, and wheel/sdist validation. All Python
108
+ commands run from the synchronized project `.venv`.
109
+
110
+ ## Provenance and license
111
+
112
+ This implementation ports the algorithmic behavior of the R [`prozor`](https://github.com/protViz/prozor) package and consolidates the matching and inference core for Python consumers such as APB and `diann_runner`.
113
+
114
+ The Python distribution is an independent implementation of that algorithm
115
+ rather than a translation of the R source, and it is licensed under MIT. The R
116
+ reference package remains GPL-3, and its license does not extend here: the
117
+ copyright holder of both packages is the same author, and algorithms are not
118
+ themselves subject to copyright.
119
+
120
+ MIT keeps the whole anndata-omics-bridge spine under one permissive license, so
121
+ that consumers such as APB can be redistributed without inheriting copyleft
122
+ obligations from this package.
prozor-0.1.0/README.md ADDED
@@ -0,0 +1,96 @@
1
+ # Prozor
2
+
3
+ [![Quality](https://github.com/anndata-omics-bridge/prozor/actions/workflows/quality.yml/badge.svg)](https://github.com/anndata-omics-bridge/prozor/actions/workflows/quality.yml)
4
+ [![Documentation](https://github.com/anndata-omics-bridge/prozor/actions/workflows/docs.yml/badge.svg)](https://anndata-omics-bridge.github.io/prozor/)
5
+ [![Python](https://img.shields.io/badge/Python-%E2%89%A53.12-3776AB.svg)](https://www.python.org/)
6
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://spdx.org/licenses/MIT.html)
7
+
8
+ Typed peptide-to-protein matching and deterministic greedy-parsimony protein
9
+ inference.
10
+
11
+ This repository is a Python port of the R [`prozor`](https://github.com/protViz/prozor) package.
12
+
13
+ Prozor offers backend-neutral Aho--Corasick matching over mappings or streaming
14
+ protein records, occurrence-level annotations, unique peptide--protein edges,
15
+ and deterministic protein inference. The core deliberately does not
16
+ parse FASTA files or depend on pandas, AnnData, MuData, MuLink, workflow engines,
17
+ or consumer CLIs.
18
+
19
+ **[Documentation](https://anndata-omics-bridge.github.io/prozor/)** ·
20
+ **[Getting started](https://anndata-omics-bridge.github.io/prozor/getting-started/)** ·
21
+ **[API reference](https://anndata-omics-bridge.github.io/prozor/api/)**
22
+
23
+ Documentation: <https://anndata-omics-bridge.github.io/prozor/>
24
+
25
+ ## Installation
26
+
27
+ ```bash
28
+ python -m pip install prozor
29
+ ```
30
+
31
+ Both matching implementations are installed. The public matching operations
32
+ default to `backend="auto"`, which selects `ahocorasick_rs`. The portable
33
+ `ahocorapy` implementation remains directly selectable and is the automatic
34
+ runtime fallback if Rust cannot be imported. Results record both the requested
35
+ and concrete backend.
36
+
37
+ ## Quick start
38
+
39
+ ```python
40
+ from dataclasses import dataclass
41
+
42
+ from prozor.api import annotate_peptides_streaming, greedy_parsimony
43
+
44
+
45
+ @dataclass(frozen=True, slots=True)
46
+ class Protein:
47
+ id: str
48
+ sequence: str
49
+
50
+
51
+ matches = annotate_peptides_streaming(
52
+ ["PEPTIDE", "SEQUENCE"],
53
+ [
54
+ Protein("P1", "MYPEPTIDESEQUENCE"),
55
+ Protein("P2", "XXSEQUENCEXX"),
56
+ ],
57
+ )
58
+
59
+ for match in matches:
60
+ print(match.peptide, match.protein_id, match.start, match.end)
61
+
62
+ edges = {(match.peptide, match.protein_id) for match in matches}
63
+ protein_groups = greedy_parsimony(edges)
64
+ print(protein_groups.to_dict())
65
+ ```
66
+
67
+ Matches include nested, overlapping, and repeated sites using half-open
68
+ coordinates. Prozor matches the exact strings supplied by the consumer; FASTA
69
+ header interpretation, normalization, decoy classification, and persistence
70
+ remain application policy.
71
+
72
+ ## Development
73
+
74
+ ```bash
75
+ uv sync --group dev --group docs
76
+ .venv/bin/pre-commit install --hook-type pre-commit --hook-type pre-push
77
+ make check
78
+ ```
79
+
80
+ `make check` runs Ruff, strict Pyright, dependency validation, branch-coverage
81
+ tests, a strict documentation build, and wheel/sdist validation. All Python
82
+ commands run from the synchronized project `.venv`.
83
+
84
+ ## Provenance and license
85
+
86
+ This implementation ports the algorithmic behavior of the R [`prozor`](https://github.com/protViz/prozor) package and consolidates the matching and inference core for Python consumers such as APB and `diann_runner`.
87
+
88
+ The Python distribution is an independent implementation of that algorithm
89
+ rather than a translation of the R source, and it is licensed under MIT. The R
90
+ reference package remains GPL-3, and its license does not extend here: the
91
+ copyright holder of both packages is the same author, and algorithms are not
92
+ themselves subject to copyright.
93
+
94
+ MIT keeps the whole anndata-omics-bridge spine under one permissive license, so
95
+ that consumers such as APB can be redistributed without inheriting copyleft
96
+ obligations from this package.
@@ -0,0 +1,142 @@
1
+ [build-system]
2
+ requires = ["uv_build>=0.9.26,<0.10.0"]
3
+ build-backend = "uv_build"
4
+
5
+ [project]
6
+ name = "prozor"
7
+ version = "0.1.0"
8
+ description = "Typed peptide-to-protein matching and parsimonious protein inference"
9
+ readme = "README.md"
10
+ requires-python = ">=3.12"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ keywords = [
14
+ "proteomics",
15
+ "protein inference",
16
+ "peptide matching",
17
+ "mass spectrometry",
18
+ ]
19
+ classifiers = [
20
+ "Development Status :: 3 - Alpha",
21
+ "Intended Audience :: Science/Research",
22
+ "Operating System :: OS Independent",
23
+ "Programming Language :: Python :: 3",
24
+ "Programming Language :: Python :: 3.12",
25
+ "Programming Language :: Python :: 3.13",
26
+ "Topic :: Scientific/Engineering :: Bio-Informatics",
27
+ "Typing :: Typed",
28
+ ]
29
+ dependencies = [
30
+ "ahocorapy>=1.6,<2",
31
+ "ahocorasick-rs>=1.0,<2",
32
+ ]
33
+
34
+ [[project.authors]]
35
+ name = "Witold Wolski"
36
+ email = "wew@fgcz.ethz.ch"
37
+
38
+ [project.urls]
39
+ Homepage = "https://github.com/anndata-omics-bridge/prozor"
40
+ Documentation = "https://anndata-omics-bridge.github.io/prozor/"
41
+ Repository = "https://github.com/anndata-omics-bridge/prozor.git"
42
+ Issues = "https://github.com/anndata-omics-bridge/prozor/issues"
43
+
44
+ [tool.uv]
45
+ default-groups = [
46
+ "dev",
47
+ "docs",
48
+ ]
49
+
50
+ [tool.uv.sources.carpet-scan]
51
+ git = "https://github.com/anndata-omics-bridge/carpet_scan"
52
+ rev = "a98f45d21884a95fa79b7660b153bbadb9268e03"
53
+
54
+ [tool.ruff]
55
+ line-length = 100
56
+ target-version = "py312"
57
+ src = [
58
+ "src",
59
+ "tests",
60
+ "benchmarks",
61
+ ]
62
+
63
+ [tool.ruff.lint]
64
+ select = [
65
+ "ANN",
66
+ "B",
67
+ "C4",
68
+ "C90",
69
+ "E4",
70
+ "E7",
71
+ "E9",
72
+ "F",
73
+ "I",
74
+ "PGH",
75
+ "PIE",
76
+ "RUF",
77
+ "SIM",
78
+ "UP",
79
+ ]
80
+
81
+ [tool.ruff.lint.mccabe]
82
+ max-complexity = 10
83
+
84
+ [tool.ruff.format]
85
+ docstring-code-format = true
86
+
87
+ [tool.pyright]
88
+ include = [
89
+ "src",
90
+ "tests",
91
+ "benchmarks",
92
+ ]
93
+ stubPath = "typings"
94
+ venvPath = "."
95
+ venv = ".venv"
96
+ pythonVersion = "3.12"
97
+ typeCheckingMode = "strict"
98
+ reportImportCycles = "error"
99
+ reportMissingTypeStubs = "error"
100
+ reportUnnecessaryTypeIgnoreComment = "error"
101
+ reportImplicitOverride = "error"
102
+ enableTypeIgnoreComments = false
103
+
104
+ [tool.pytest.ini_options]
105
+ addopts = [
106
+ "--strict-config",
107
+ "--strict-markers",
108
+ "-ra",
109
+ ]
110
+ testpaths = ["tests"]
111
+ xfail_strict = true
112
+
113
+ [tool.coverage.run]
114
+ branch = true
115
+ source = ["prozor"]
116
+
117
+ [tool.coverage.report]
118
+ fail_under = 80
119
+ show_missing = true
120
+ skip_covered = true
121
+
122
+ [tool.deptry]
123
+ known_first_party = ["prozor"]
124
+
125
+ [dependency-groups]
126
+ dev = [
127
+ "carpet-scan",
128
+ "build>=1.3,<2",
129
+ "deptry>=0.24,<1",
130
+ "import-linter>=2.8,<3",
131
+ "pre-commit>=4,<5",
132
+ "pyright>=1.1.400,<2",
133
+ "pytest>=9,<10",
134
+ "pytest-cov>=7,<8",
135
+ "ruff>=0.15,<1",
136
+ "twine>=6,<7",
137
+ ]
138
+ docs = [
139
+ "mkdocstrings[python]>=1.0,<2",
140
+ "pymdown-extensions>=11,<12",
141
+ "zensical==0.0.43",
142
+ ]
@@ -0,0 +1,122 @@
1
+ [build-system]
2
+ requires = ["uv_build>=0.9.26,<0.10.0"]
3
+ build-backend = "uv_build"
4
+
5
+ [project]
6
+ name = "prozor"
7
+ version = "0.1.0"
8
+ description = "Typed peptide-to-protein matching and parsimonious protein inference"
9
+ readme = "README.md"
10
+ requires-python = ">=3.12"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ authors = [
14
+ { name = "Witold Wolski", email = "wew@fgcz.ethz.ch" },
15
+ ]
16
+ keywords = ["proteomics", "protein inference", "peptide matching", "mass spectrometry"]
17
+ classifiers = [
18
+ "Development Status :: 3 - Alpha",
19
+ "Intended Audience :: Science/Research",
20
+ "Operating System :: OS Independent",
21
+ "Programming Language :: Python :: 3",
22
+ "Programming Language :: Python :: 3.12",
23
+ "Programming Language :: Python :: 3.13",
24
+ "Topic :: Scientific/Engineering :: Bio-Informatics",
25
+ "Typing :: Typed",
26
+ ]
27
+ dependencies = [
28
+ "ahocorapy>=1.6,<2",
29
+ "ahocorasick-rs>=1.0,<2",
30
+ ]
31
+
32
+ [project.urls]
33
+ Homepage = "https://github.com/anndata-omics-bridge/prozor"
34
+ Documentation = "https://anndata-omics-bridge.github.io/prozor/"
35
+ Repository = "https://github.com/anndata-omics-bridge/prozor.git"
36
+ Issues = "https://github.com/anndata-omics-bridge/prozor/issues"
37
+
38
+ [tool.uv.sources]
39
+ carpet-scan = { git = "https://github.com/anndata-omics-bridge/carpet_scan", rev = "a98f45d21884a95fa79b7660b153bbadb9268e03" }
40
+
41
+ [dependency-groups]
42
+ dev = [
43
+ # Carpet diagnostics. Must live in Prozor's own environment rather than an isolated `uvx` one,
44
+ # because grimp resolves the target package through the Python path.
45
+ "carpet-scan",
46
+ "build>=1.3,<2",
47
+ "deptry>=0.24,<1",
48
+ "import-linter>=2.8,<3",
49
+ "pre-commit>=4,<5",
50
+ "pyright>=1.1.400,<2",
51
+ "pytest>=9,<10",
52
+ "pytest-cov>=7,<8",
53
+ "ruff>=0.15,<1",
54
+ "twine>=6,<7",
55
+ ]
56
+ docs = [
57
+ "mkdocstrings[python]>=1.0,<2",
58
+ "pymdown-extensions>=11,<12",
59
+ "zensical==0.0.43",
60
+ ]
61
+
62
+ [tool.uv]
63
+ default-groups = ["dev", "docs"]
64
+
65
+ [tool.ruff]
66
+ line-length = 100
67
+ target-version = "py312"
68
+ src = ["src", "tests", "benchmarks"]
69
+
70
+ [tool.ruff.lint]
71
+ select = [
72
+ "ANN",
73
+ "B",
74
+ "C4",
75
+ "C90",
76
+ "E4",
77
+ "E7",
78
+ "E9",
79
+ "F",
80
+ "I",
81
+ "PGH",
82
+ "PIE",
83
+ "RUF",
84
+ "SIM",
85
+ "UP",
86
+ ]
87
+
88
+ [tool.ruff.lint.mccabe]
89
+ max-complexity = 10
90
+
91
+ [tool.ruff.format]
92
+ docstring-code-format = true
93
+
94
+ [tool.pyright]
95
+ include = ["src", "tests", "benchmarks"]
96
+ stubPath = "typings"
97
+ venvPath = "."
98
+ venv = ".venv"
99
+ pythonVersion = "3.12"
100
+ typeCheckingMode = "strict"
101
+ reportImportCycles = "error"
102
+ reportMissingTypeStubs = "error"
103
+ reportUnnecessaryTypeIgnoreComment = "error"
104
+ reportImplicitOverride = "error"
105
+ enableTypeIgnoreComments = false
106
+
107
+ [tool.pytest.ini_options]
108
+ addopts = ["--strict-config", "--strict-markers", "-ra"]
109
+ testpaths = ["tests"]
110
+ xfail_strict = true
111
+
112
+ [tool.coverage.run]
113
+ branch = true
114
+ source = ["prozor"]
115
+
116
+ [tool.coverage.report]
117
+ fail_under = 80
118
+ show_missing = true
119
+ skip_covered = true
120
+
121
+ [tool.deptry]
122
+ known_first_party = ["prozor"]
@@ -0,0 +1 @@
1
+
@@ -0,0 +1,21 @@
1
+ """The one public prozor module for other anndata_bridge packages."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from prozor.inference.greedy import greedy_parsimony
6
+ from prozor.inference.ties import TieCandidate
7
+ from prozor.matching.annotation import (
8
+ ProteinSequenceRecord,
9
+ annotate_peptides,
10
+ annotate_peptides_streaming,
11
+ )
12
+ from prozor.matching.automaton import resolve_backend
13
+
14
+ __all__ = [
15
+ "ProteinSequenceRecord",
16
+ "TieCandidate",
17
+ "annotate_peptides",
18
+ "annotate_peptides_streaming",
19
+ "greedy_parsimony",
20
+ "resolve_backend",
21
+ ]
File without changes
@@ -0,0 +1,330 @@
1
+ """Deterministic greedy-parsimony protein inference over unique edges."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Iterable, Iterator, Sequence
6
+ from dataclasses import dataclass
7
+
8
+ from prozor.inference.ties import (
9
+ TieCandidate,
10
+ TieResolver,
11
+ resolve_current_tie,
12
+ )
13
+
14
+ type PeptideProteinEdge = tuple[str, str]
15
+
16
+
17
+ @dataclass(slots=True)
18
+ class ProteinGroup:
19
+ """Proteins selected together and the peptides assigned to them."""
20
+
21
+ proteins: list[str]
22
+ peptides: list[str]
23
+
24
+ @property
25
+ def protein_id(self) -> str:
26
+ """Return the semicolon-joined protein group identifier."""
27
+ return ";".join(self.proteins)
28
+
29
+ @property
30
+ def n_peptides(self) -> int:
31
+ """Return the number of assigned peptides."""
32
+ return len(self.peptides)
33
+
34
+ @property
35
+ def n_proteins(self) -> int:
36
+ """Return the number of grouped proteins."""
37
+ return len(self.proteins)
38
+
39
+
40
+ @dataclass(slots=True)
41
+ class GreedyResult:
42
+ """Protein groups selected by greedy parsimony."""
43
+
44
+ groups: list[ProteinGroup]
45
+
46
+ def __len__(self) -> int:
47
+ return len(self.groups)
48
+
49
+ def __iter__(self) -> Iterator[ProteinGroup]:
50
+ return iter(self.groups)
51
+
52
+ @property
53
+ def n_proteins(self) -> int:
54
+ """Return the total number of selected and subsumed proteins."""
55
+ return sum(group.n_proteins for group in self.groups)
56
+
57
+ @property
58
+ def n_groups(self) -> int:
59
+ """Return the number of inferred protein groups."""
60
+ return len(self.groups)
61
+
62
+ @property
63
+ def n_peptides(self) -> int:
64
+ """Return the number of distinct assigned peptides."""
65
+ return len({peptide for group in self.groups for peptide in group.peptides})
66
+
67
+ def to_dict(self) -> dict[str, str]:
68
+ """Return a peptide-to-protein-group mapping."""
69
+ return {peptide: group.protein_id for group in self.groups for peptide in group.peptides}
70
+
71
+
72
+ @dataclass(frozen=True, slots=True)
73
+ class _Selection:
74
+ winners: tuple[int, ...]
75
+ covered_peptides: frozenset[int]
76
+
77
+
78
+ def greedy_parsimony(
79
+ edges: Iterable[PeptideProteinEdge],
80
+ resolve_tie: TieResolver = resolve_current_tie,
81
+ subsume: bool = True,
82
+ ) -> GreedyResult:
83
+ """Select a deterministic parsimonious set of protein groups.
84
+
85
+ Repeated ``(peptide, protein)`` edges collapse before inference. Proteins
86
+ with identical remaining peptide evidence are grouped. Disjoint equal
87
+ candidates are ordered deterministically, while overlapping non-identical
88
+ candidates are passed to ``resolve_tie``.
89
+
90
+ Args:
91
+ edges: Peptide and protein identifier pairs.
92
+ resolve_tie: Operation selecting one consequential tie candidate.
93
+ subsume: Whether proteins supported by a subset of selected evidence
94
+ remain in the selected group.
95
+
96
+ Returns:
97
+ Deterministically ordered inferred protein groups.
98
+ """
99
+ peptides, proteins, peptide_proteins, protein_peptides = _build_incidence(edges)
100
+ active_peptides = set(range(len(peptides)))
101
+ active_proteins = set(range(len(proteins)))
102
+ counts = [len(protein_evidence) for protein_evidence in protein_peptides]
103
+ groups: list[ProteinGroup] = []
104
+
105
+ while active_peptides and active_proteins:
106
+ selections = _select_winners(
107
+ active_peptides,
108
+ active_proteins,
109
+ protein_peptides,
110
+ counts,
111
+ peptides,
112
+ proteins,
113
+ resolve_tie,
114
+ )
115
+ if not selections:
116
+ break
117
+ for selection in selections:
118
+ subsumed = (
119
+ _find_subsumed(
120
+ selection,
121
+ active_peptides,
122
+ active_proteins,
123
+ peptide_proteins,
124
+ protein_peptides,
125
+ )
126
+ if subsume
127
+ else ()
128
+ )
129
+ group_indices = tuple(sorted((*selection.winners, *subsumed)))
130
+ groups.append(
131
+ ProteinGroup(
132
+ proteins=sorted(proteins[index] for index in group_indices),
133
+ peptides=sorted(peptides[index] for index in selection.covered_peptides),
134
+ )
135
+ )
136
+ active_peptides.difference_update(selection.covered_peptides)
137
+ active_proteins.difference_update(group_indices)
138
+ _decrement_counts(
139
+ selection.covered_peptides,
140
+ active_proteins,
141
+ peptide_proteins,
142
+ counts,
143
+ )
144
+
145
+ groups.sort(
146
+ key=lambda group: (
147
+ -group.n_peptides,
148
+ -group.n_proteins,
149
+ tuple(group.proteins),
150
+ )
151
+ )
152
+ return GreedyResult(groups=groups)
153
+
154
+
155
+ def _build_incidence(
156
+ edges: Iterable[PeptideProteinEdge],
157
+ ) -> tuple[list[str], list[str], list[frozenset[int]], list[frozenset[int]]]:
158
+ unique_edges = set(edges)
159
+ peptides = sorted({peptide for peptide, _protein in unique_edges})
160
+ proteins = sorted({protein for _peptide, protein in unique_edges})
161
+ peptide_indices = {peptide: index for index, peptide in enumerate(peptides)}
162
+ protein_indices = {protein: index for index, protein in enumerate(proteins)}
163
+ peptide_proteins: list[set[int]] = [set() for _peptide in peptides]
164
+ protein_peptides: list[set[int]] = [set() for _protein in proteins]
165
+ for peptide, protein in unique_edges:
166
+ peptide_index = peptide_indices[peptide]
167
+ protein_index = protein_indices[protein]
168
+ peptide_proteins[peptide_index].add(protein_index)
169
+ protein_peptides[protein_index].add(peptide_index)
170
+ return (
171
+ peptides,
172
+ proteins,
173
+ [frozenset(indices) for indices in peptide_proteins],
174
+ [frozenset(indices) for indices in protein_peptides],
175
+ )
176
+
177
+
178
+ def _select_winners(
179
+ active_peptides: set[int],
180
+ active_proteins: set[int],
181
+ protein_peptides: Sequence[frozenset[int]],
182
+ counts: Sequence[int],
183
+ peptide_names: Sequence[str],
184
+ protein_names: Sequence[str],
185
+ resolve_tie: TieResolver,
186
+ ) -> tuple[_Selection, ...]:
187
+ ordered_active = sorted(active_proteins)
188
+ if not ordered_active:
189
+ return ()
190
+ max_count = max(counts[index] for index in ordered_active)
191
+ if max_count == 0:
192
+ return ()
193
+ max_proteins = [index for index in ordered_active if counts[index] == max_count]
194
+ signature_groups: dict[frozenset[int], list[int]] = {}
195
+ for index in max_proteins:
196
+ signature = frozenset(protein_peptides[index] & active_peptides)
197
+ signature_groups.setdefault(signature, []).append(index)
198
+
199
+ grouped = tuple(
200
+ tuple(indices)
201
+ for _signature, indices in sorted(
202
+ signature_groups.items(),
203
+ key=lambda item: tuple(protein_names[index] for index in item[1]),
204
+ )
205
+ )
206
+ signatures = tuple(
207
+ frozenset(protein_peptides[indices[0]] & active_peptides) for indices in grouped
208
+ )
209
+ components = _overlap_components(signatures)
210
+ selections = [
211
+ _selection_for_component(
212
+ component,
213
+ grouped,
214
+ signatures,
215
+ peptide_names,
216
+ protein_names,
217
+ resolve_tie,
218
+ )
219
+ for component in components
220
+ ]
221
+ return tuple(sorted(selections, key=lambda selection: _selection_key(selection, protein_names)))
222
+
223
+
224
+ def _selection_key(
225
+ selection: _Selection,
226
+ protein_names: Sequence[str],
227
+ ) -> tuple[int, tuple[str, ...]]:
228
+ return (
229
+ -len(selection.winners),
230
+ tuple(protein_names[index] for index in selection.winners),
231
+ )
232
+
233
+
234
+ def _overlap_components(signatures: Sequence[frozenset[int]]) -> tuple[tuple[int, ...], ...]:
235
+ peptide_groups: dict[int, list[int]] = {}
236
+ for group_index, signature in enumerate(signatures):
237
+ for peptide in signature:
238
+ peptide_groups.setdefault(peptide, []).append(group_index)
239
+ unseen = set(range(len(signatures)))
240
+ components: list[tuple[int, ...]] = []
241
+ while unseen:
242
+ pending = [min(unseen)]
243
+ component: set[int] = set()
244
+ while pending:
245
+ group_index = pending.pop()
246
+ if group_index not in unseen:
247
+ continue
248
+ unseen.remove(group_index)
249
+ component.add(group_index)
250
+ for peptide in signatures[group_index]:
251
+ pending.extend(peptide_groups[peptide])
252
+ components.append(tuple(sorted(component)))
253
+ return tuple(components)
254
+
255
+
256
+ def _selection_for_component(
257
+ component: tuple[int, ...],
258
+ groups: Sequence[tuple[int, ...]],
259
+ signatures: Sequence[frozenset[int]],
260
+ peptide_names: Sequence[str],
261
+ protein_names: Sequence[str],
262
+ resolve_tie: TieResolver,
263
+ ) -> _Selection:
264
+ component_groups = tuple(groups[index] for index in component)
265
+ component_signatures = tuple(signatures[index] for index in component)
266
+ winners = (
267
+ component_groups[0]
268
+ if len(component_groups) == 1
269
+ else _resolve_overlapping_tie(
270
+ component_groups,
271
+ component_signatures,
272
+ peptide_names,
273
+ protein_names,
274
+ resolve_tie,
275
+ )
276
+ )
277
+ winner_index = component_groups.index(winners)
278
+ return _Selection(winners=winners, covered_peptides=component_signatures[winner_index])
279
+
280
+
281
+ def _resolve_overlapping_tie(
282
+ groups: Sequence[tuple[int, ...]],
283
+ signatures: Sequence[frozenset[int]],
284
+ peptide_names: Sequence[str],
285
+ protein_names: Sequence[str],
286
+ resolve_tie: TieResolver,
287
+ ) -> tuple[int, ...]:
288
+ candidates = tuple(
289
+ TieCandidate(
290
+ proteins=tuple(protein_names[index] for index in group),
291
+ unexplained_peptides=frozenset(peptide_names[index] for index in signature),
292
+ )
293
+ for group, signature in zip(groups, signatures, strict=True)
294
+ )
295
+ selected = resolve_tie(candidates)
296
+ try:
297
+ selected_index = candidates.index(selected)
298
+ except ValueError as error:
299
+ raise ValueError("tie resolver must return one of the candidates it received") from error
300
+ return groups[selected_index]
301
+
302
+
303
+ def _find_subsumed(
304
+ selection: _Selection,
305
+ active_peptides: set[int],
306
+ active_proteins: set[int],
307
+ peptide_proteins: Sequence[frozenset[int]],
308
+ protein_peptides: Sequence[frozenset[int]],
309
+ ) -> tuple[int, ...]:
310
+ candidates = {
311
+ protein for peptide in selection.covered_peptides for protein in peptide_proteins[peptide]
312
+ }
313
+ candidates.difference_update(selection.winners)
314
+ candidates.intersection_update(active_proteins)
315
+ return tuple(
316
+ protein
317
+ for protein in sorted(candidates)
318
+ if protein_peptides[protein] & active_peptides <= selection.covered_peptides
319
+ )
320
+
321
+
322
+ def _decrement_counts(
323
+ removed_peptides: frozenset[int],
324
+ active_proteins: set[int],
325
+ peptide_proteins: Sequence[frozenset[int]],
326
+ counts: list[int],
327
+ ) -> None:
328
+ for peptide in removed_peptides:
329
+ for protein in peptide_proteins[peptide] & active_proteins:
330
+ counts[protein] -= 1
@@ -0,0 +1,24 @@
1
+ """Tie candidates and resolution for greedy protein inference."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Callable, Sequence
6
+ from dataclasses import dataclass
7
+
8
+
9
+ @dataclass(frozen=True, slots=True)
10
+ class TieCandidate:
11
+ """One indistinguishable protein group competing in a consequential tie."""
12
+
13
+ proteins: tuple[str, ...]
14
+ unexplained_peptides: frozenset[str]
15
+
16
+
17
+ type TieResolver = Callable[[Sequence[TieCandidate]], TieCandidate]
18
+
19
+
20
+ def resolve_current_tie(candidates: Sequence[TieCandidate]) -> TieCandidate:
21
+ """Preserve Prozor's deterministic group-size and accession tie rule."""
22
+ if not candidates:
23
+ raise ValueError("at least one tie candidate is required")
24
+ return min(candidates, key=lambda candidate: (-len(candidate.proteins), candidate.proteins))
File without changes
@@ -0,0 +1,168 @@
1
+ """Peptide-to-protein annotation over mappings or streaming protein records."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from bisect import bisect_right
6
+ from collections.abc import Iterable, Iterator, Mapping, Sequence
7
+ from dataclasses import dataclass
8
+ from itertools import accumulate, islice
9
+ from typing import Protocol
10
+
11
+ from prozor.matching.automaton import (
12
+ BackendName,
13
+ BackendRequest,
14
+ create_automaton,
15
+ resolve_backend,
16
+ )
17
+
18
+ _SEPARATOR = "\n"
19
+ _BATCH_SIZE = 10_000
20
+
21
+
22
+ class ProteinSequenceRecord(Protocol):
23
+ """Smallest protein-record capability required by peptide matching."""
24
+
25
+ @property
26
+ def id(self) -> str:
27
+ """Return the protein identifier attached to matching results."""
28
+ ...
29
+
30
+ @property
31
+ def sequence(self) -> str:
32
+ """Return the protein sequence to search."""
33
+ ...
34
+
35
+
36
+ @dataclass(frozen=True, slots=True)
37
+ class PeptideAnnotation:
38
+ """One peptide occurrence in a protein sequence."""
39
+
40
+ peptide: str
41
+ protein_id: str
42
+ start: int
43
+ end: int
44
+
45
+ @property
46
+ def length(self) -> int:
47
+ """Return the matched peptide length."""
48
+ return len(self.peptide)
49
+
50
+
51
+ @dataclass(slots=True)
52
+ class AnnotationResult:
53
+ """Collection of peptide matches and matching-backend provenance."""
54
+
55
+ annotations: list[PeptideAnnotation]
56
+ requested_backend: BackendRequest = "ahocorapy"
57
+ resolved_backend: BackendName = "ahocorapy"
58
+
59
+ def __len__(self) -> int:
60
+ return len(self.annotations)
61
+
62
+ def __iter__(self) -> Iterator[PeptideAnnotation]:
63
+ return iter(self.annotations)
64
+
65
+ @property
66
+ def peptides(self) -> set[str]:
67
+ """Return the distinct matched peptide sequences."""
68
+ return {annotation.peptide for annotation in self.annotations}
69
+
70
+ @property
71
+ def proteins(self) -> set[str]:
72
+ """Return the distinct matched protein identifiers."""
73
+ return {annotation.protein_id for annotation in self.annotations}
74
+
75
+ def filter_tryptic(
76
+ self,
77
+ proteins: Mapping[str, str],
78
+ prefix_residues: str = "RK",
79
+ allow_n_term: bool = True,
80
+ allow_after_init_met: bool = True,
81
+ ) -> AnnotationResult:
82
+ """Return matches with a supported tryptic N-terminal context."""
83
+ filtered: list[PeptideAnnotation] = []
84
+ for annotation in self.annotations:
85
+ sequence = proteins.get(annotation.protein_id, "")
86
+ if not sequence:
87
+ continue
88
+ valid_prefix = (
89
+ (annotation.start == 0 and allow_n_term)
90
+ or (annotation.start == 1 and allow_after_init_met and sequence.startswith("M"))
91
+ or (annotation.start > 0 and sequence[annotation.start - 1] in prefix_residues)
92
+ )
93
+ if valid_prefix:
94
+ filtered.append(annotation)
95
+ return AnnotationResult(
96
+ annotations=filtered,
97
+ requested_backend=self.requested_backend,
98
+ resolved_backend=self.resolved_backend,
99
+ )
100
+
101
+
102
+ def annotate_peptides(
103
+ peptides: Iterable[str],
104
+ proteins: Mapping[str, str],
105
+ backend: str = "auto",
106
+ filter_tryptic: bool = False,
107
+ ) -> AnnotationResult:
108
+ """Annotate peptides against an in-memory protein mapping."""
109
+ result = _annotate(peptides, [(tuple(proteins), tuple(proteins.values()))], backend)
110
+ return result.filter_tryptic(proteins) if filter_tryptic else result
111
+
112
+
113
+ def annotate_peptides_streaming(
114
+ peptides: Iterable[str],
115
+ protein_records: Iterable[ProteinSequenceRecord],
116
+ backend: str = "auto",
117
+ ) -> AnnotationResult:
118
+ """Annotate peptides against one-pass records exposing ``id`` and ``sequence``."""
119
+ records = iter(protein_records)
120
+ batches = (
121
+ (tuple(record.id for record in batch), tuple(record.sequence for record in batch))
122
+ for batch in iter(lambda: tuple(islice(records, _BATCH_SIZE)), ())
123
+ )
124
+ return _annotate(peptides, batches, backend)
125
+
126
+
127
+ def _annotate(
128
+ peptides: Iterable[str],
129
+ batches: Iterable[tuple[Sequence[str], Sequence[str]]],
130
+ backend: str,
131
+ ) -> AnnotationResult:
132
+ peptide_list = list(dict.fromkeys(peptides))
133
+ if not peptide_list:
134
+ return AnnotationResult(
135
+ annotations=[],
136
+ requested_backend=_requested_backend(backend),
137
+ resolved_backend=resolve_backend(backend),
138
+ )
139
+ if any(_SEPARATOR in peptide for peptide in peptide_list):
140
+ raise ValueError("peptides must not contain line breaks")
141
+
142
+ automaton = create_automaton(peptide_list, backend=backend)
143
+ annotations: list[PeptideAnnotation] = []
144
+ for ids, sequences in batches:
145
+ # One backend call per batch; the separator stops matches across sequences.
146
+ starts = list(accumulate((len(sequence) + 1 for sequence in sequences), initial=0))
147
+ for match in automaton.find_all(_SEPARATOR.join(sequences)):
148
+ index = bisect_right(starts, match.start) - 1
149
+ offset = starts[index]
150
+ annotations.append(
151
+ PeptideAnnotation(
152
+ match.keyword, ids[index], match.start - offset, match.end - offset
153
+ )
154
+ )
155
+ return AnnotationResult(
156
+ annotations=annotations,
157
+ requested_backend=automaton.requested_backend,
158
+ resolved_backend=automaton.resolved_backend,
159
+ )
160
+
161
+
162
+ def _requested_backend(backend: str) -> BackendRequest:
163
+ resolve_backend(backend)
164
+ if backend == "ahocorapy":
165
+ return "ahocorapy"
166
+ if backend == "ahocorasick_rs":
167
+ return "ahocorasick_rs"
168
+ return "auto"
@@ -0,0 +1,164 @@
1
+ """Backend-neutral Aho--Corasick peptide matching."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from abc import ABC, abstractmethod
6
+ from collections.abc import Iterable, Iterator
7
+ from dataclasses import dataclass
8
+ from importlib.util import find_spec
9
+ from typing import Literal, cast, override
10
+
11
+ type BackendName = Literal["ahocorapy", "ahocorasick_rs"]
12
+ type BackendRequest = Literal["auto", "ahocorapy", "ahocorasick_rs"]
13
+
14
+ _VALID_BACKENDS = frozenset({"auto", "ahocorapy", "ahocorasick_rs"})
15
+
16
+
17
+ @dataclass(frozen=True, slots=True)
18
+ class Match:
19
+ """One half-open keyword match in a searched sequence."""
20
+
21
+ keyword: str
22
+ start: int
23
+ end: int
24
+
25
+
26
+ class AhoCorasickBase(ABC):
27
+ """Common interface implemented by every matching backend."""
28
+
29
+ requested_backend: BackendRequest
30
+ resolved_backend: BackendName
31
+
32
+ @abstractmethod
33
+ def find_all(self, text: str) -> Iterator[Match]:
34
+ """Yield every match, including nested and overlapping matches."""
35
+
36
+
37
+ class AhoCorasickPure(AhoCorasickBase):
38
+ """Portable matcher implemented with :mod:`ahocorapy`."""
39
+
40
+ def __init__(
41
+ self,
42
+ keywords: Iterable[str],
43
+ *,
44
+ case_sensitive: bool = True,
45
+ requested_backend: BackendRequest = "ahocorapy",
46
+ ) -> None:
47
+ from ahocorapy.keywordtree import KeywordTree
48
+
49
+ self.requested_backend = requested_backend
50
+ self.resolved_backend: BackendName = "ahocorapy"
51
+ self._tree = KeywordTree(case_insensitive=not case_sensitive)
52
+ for keyword in _unique_keywords(keywords):
53
+ self._tree.add(keyword)
54
+ self._tree.finalize()
55
+
56
+ @override
57
+ def find_all(self, text: str) -> Iterator[Match]:
58
+ """Yield all pure-Python matches in ``text``."""
59
+ for keyword, start in self._tree.search_all(text):
60
+ yield Match(keyword=keyword, start=start, end=start + len(keyword))
61
+
62
+
63
+ class AhoCorasickRust(AhoCorasickBase):
64
+ """Accelerated matcher implemented with :mod:`ahocorasick_rs`."""
65
+
66
+ def __init__(
67
+ self,
68
+ keywords: Iterable[str],
69
+ *,
70
+ case_sensitive: bool = True,
71
+ requested_backend: BackendRequest = "ahocorasick_rs",
72
+ ) -> None:
73
+ import ahocorasick_rs
74
+
75
+ self.requested_backend = requested_backend
76
+ self.resolved_backend: BackendName = "ahocorasick_rs"
77
+ self._keywords = _unique_keywords(keywords)
78
+ self._case_sensitive = case_sensitive
79
+ indexed_keywords = (
80
+ self._keywords
81
+ if case_sensitive
82
+ else tuple(keyword.lower() for keyword in self._keywords)
83
+ )
84
+ self._automaton = ahocorasick_rs.AhoCorasick(indexed_keywords)
85
+
86
+ @override
87
+ def find_all(self, text: str) -> Iterator[Match]:
88
+ """Yield all Rust matches in ``text``."""
89
+ search_text = text if self._case_sensitive else text.lower()
90
+ for index, start, end in self._automaton.find_matches_as_indexes(
91
+ search_text,
92
+ overlapping=True,
93
+ ):
94
+ yield Match(keyword=self._keywords[index], start=start, end=end)
95
+
96
+
97
+ def create_automaton(
98
+ keywords: Iterable[str],
99
+ backend: str = "auto",
100
+ case_sensitive: bool = True,
101
+ ) -> AhoCorasickBase:
102
+ """Create a matcher and expose both requested and resolved backend names.
103
+
104
+ Args:
105
+ keywords: Peptide patterns to search for.
106
+ backend: ``auto``, ``ahocorapy``, or ``ahocorasick_rs``.
107
+ case_sensitive: Whether matching distinguishes letter case.
108
+
109
+ Returns:
110
+ A backend-neutral matcher.
111
+
112
+ Raises:
113
+ ImportError: The Rust backend was requested but is not installed.
114
+ ValueError: ``backend`` is not supported.
115
+ """
116
+ requested_backend = _validate_backend(backend)
117
+ resolved_backend = resolve_backend(requested_backend)
118
+ if resolved_backend == "ahocorasick_rs":
119
+ return AhoCorasickRust(
120
+ keywords,
121
+ case_sensitive=case_sensitive,
122
+ requested_backend=requested_backend,
123
+ )
124
+ return AhoCorasickPure(
125
+ keywords,
126
+ case_sensitive=case_sensitive,
127
+ requested_backend=requested_backend,
128
+ )
129
+
130
+
131
+ def resolve_backend(backend: str = "auto") -> BackendName:
132
+ """Resolve a requested backend to the concrete implementation name."""
133
+ requested_backend = _validate_backend(backend)
134
+ if requested_backend == "ahocorasick_rs":
135
+ if not _rust_available():
136
+ raise ImportError(
137
+ "backend 'ahocorasick_rs' requires a working ahocorasick-rs installation"
138
+ )
139
+ return "ahocorasick_rs"
140
+ if requested_backend == "ahocorapy":
141
+ return "ahocorapy"
142
+ return "ahocorasick_rs" if _rust_available() else "ahocorapy"
143
+
144
+
145
+ def get_available_backends() -> list[BackendName]:
146
+ """Return the concrete matching backends available in this environment."""
147
+ backends: list[BackendName] = ["ahocorapy"]
148
+ if _rust_available():
149
+ backends.append("ahocorasick_rs")
150
+ return backends
151
+
152
+
153
+ def _validate_backend(backend: str) -> BackendRequest:
154
+ if backend not in _VALID_BACKENDS:
155
+ raise ValueError(f"backend must be one of {sorted(_VALID_BACKENDS)}, got {backend!r}")
156
+ return cast(BackendRequest, backend)
157
+
158
+
159
+ def _rust_available() -> bool:
160
+ return find_spec("ahocorasick_rs") is not None
161
+
162
+
163
+ def _unique_keywords(keywords: Iterable[str]) -> tuple[str, ...]:
164
+ return tuple(dict.fromkeys(keywords))
@@ -0,0 +1 @@
1
+