apb-proteobench 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. apb_proteobench-0.1.0/LICENSE +21 -0
  2. apb_proteobench-0.1.0/PKG-INFO +97 -0
  3. apb_proteobench-0.1.0/README.md +63 -0
  4. apb_proteobench-0.1.0/THIRD_PARTY_NOTICES.md +7 -0
  5. apb_proteobench-0.1.0/pyproject.toml +161 -0
  6. apb_proteobench-0.1.0/pyproject.toml.orig +135 -0
  7. apb_proteobench-0.1.0/src/apb_proteobench/__init__.py +1 -0
  8. apb_proteobench-0.1.0/src/apb_proteobench/annotation.py +93 -0
  9. apb_proteobench-0.1.0/src/apb_proteobench/api.py +113 -0
  10. apb_proteobench-0.1.0/src/apb_proteobench/calculation/__init__.py +0 -0
  11. apb_proteobench-0.1.0/src/apb_proteobench/calculation/contracts.py +35 -0
  12. apb_proteobench-0.1.0/src/apb_proteobench/calculation/entrapment.py +254 -0
  13. apb_proteobench-0.1.0/src/apb_proteobench/calculation/intermediate.py +560 -0
  14. apb_proteobench-0.1.0/src/apb_proteobench/calculation/mapping.py +44 -0
  15. apb_proteobench-0.1.0/src/apb_proteobench/calculation/metrics.py +283 -0
  16. apb_proteobench-0.1.0/src/apb_proteobench/cli/__init__.py +0 -0
  17. apb_proteobench-0.1.0/src/apb_proteobench/cli/app.py +754 -0
  18. apb_proteobench-0.1.0/src/apb_proteobench/cli/presentation.py +71 -0
  19. apb_proteobench-0.1.0/src/apb_proteobench/cli/result_performance.py +360 -0
  20. apb_proteobench-0.1.0/src/apb_proteobench/cli/timings.py +59 -0
  21. apb_proteobench-0.1.0/src/apb_proteobench/configuration/__init__.py +0 -0
  22. apb_proteobench-0.1.0/src/apb_proteobench/configuration/design.py +149 -0
  23. apb_proteobench-0.1.0/src/apb_proteobench/configuration/entrapment.py +56 -0
  24. apb_proteobench-0.1.0/src/apb_proteobench/configuration/load.py +151 -0
  25. apb_proteobench-0.1.0/src/apb_proteobench/configuration/schema.py +114 -0
  26. apb_proteobench-0.1.0/src/apb_proteobench/data/APACHE-2.0.txt +201 -0
  27. apb_proteobench-0.1.0/src/apb_proteobench/data/MODULES_NOTICE.md +27 -0
  28. apb_proteobench-0.1.0/src/apb_proteobench/data/RENDERER_NOTICE.md +5 -0
  29. apb_proteobench-0.1.0/src/apb_proteobench/data/__init__.py +0 -0
  30. apb_proteobench-0.1.0/src/apb_proteobench/data/modules/__init__.py +0 -0
  31. apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dda_astral.sdrf.tsv +7 -0
  32. apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dda_astral.toml +32 -0
  33. apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dda_peptidoform.sdrf.tsv +7 -0
  34. apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dda_peptidoform.toml +41 -0
  35. apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dda_qexactive.sdrf.tsv +7 -0
  36. apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dda_qexactive.toml +41 -0
  37. apb_proteobench-0.1.0/src/apb_proteobench/data/modules/denovo_dda_hcd.toml +6 -0
  38. apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dia_aif.sdrf.tsv +7 -0
  39. apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dia_aif.toml +32 -0
  40. apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dia_astral.sdrf.tsv +7 -0
  41. apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dia_astral.toml +41 -0
  42. apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dia_diapasef.sdrf.tsv +7 -0
  43. apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dia_diapasef.toml +41 -0
  44. apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dia_plasma.sdrf.tsv +13 -0
  45. apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dia_plasma.toml +50 -0
  46. apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dia_singlecell.sdrf.tsv +7 -0
  47. apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dia_singlecell.toml +36 -0
  48. apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dia_zenotof.sdrf.tsv +7 -0
  49. apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dia_zenotof.toml +32 -0
  50. apb_proteobench-0.1.0/src/apb_proteobench/data/modules/entrapment_dia_astral.sdrf.tsv +4 -0
  51. apb_proteobench-0.1.0/src/apb_proteobench/data/modules/entrapment_dia_astral.toml +5 -0
  52. apb_proteobench-0.1.0/src/apb_proteobench/entrapment.py +164 -0
  53. apb_proteobench-0.1.0/src/apb_proteobench/integration.py +326 -0
  54. apb_proteobench-0.1.0/src/apb_proteobench/py.typed +1 -0
  55. apb_proteobench-0.1.0/src/apb_proteobench/workflow.py +121 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Witold Wolski
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,97 @@
1
+ Metadata-Version: 2.4
2
+ Name: apb-proteobench
3
+ Version: 0.1.0
4
+ Summary: ProteoBench annotation, diagnostics, and scoring for APB2 results
5
+ Keywords: proteomics,proteobench,benchmarking,anndata,mass spectrometry
6
+ Author: Witold Wolski
7
+ Author-email: Witold Wolski <wew@fgcz.ethz.ch>
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ License-File: THIRD_PARTY_NOTICES.md
11
+ License-File: src/apb_proteobench/data/APACHE-2.0.txt
12
+ Classifier: Development Status :: 3 - Alpha
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.13
17
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
18
+ Classifier: Typing :: Typed
19
+ Requires-Dist: apb2>=0.1,<0.2
20
+ Requires-Dist: apb-catalog>=0.1,<0.2
21
+ Requires-Dist: apb-fasta>=0.1,<0.2
22
+ Requires-Dist: cyclopts>=4,<5
23
+ Requires-Dist: loguru>=0.7,<1
24
+ Requires-Dist: numpy>=2,<3
25
+ Requires-Dist: pandas>=2.3,<4
26
+ Requires-Dist: polars>=1.38,<2
27
+ Requires-Dist: pydantic>=2.12,<3
28
+ Requires-Dist: scipy>=1.16,<2
29
+ Requires-Python: >=3.13
30
+ Project-URL: Documentation, https://anndata-omics-bridge.github.io/apb-proteobench/
31
+ Project-URL: Repository, https://github.com/anndata-omics-bridge/apb-proteobench
32
+ Project-URL: Issues, https://github.com/anndata-omics-bridge/apb-proteobench/issues
33
+ Description-Content-Type: text/markdown
34
+
35
+ # APB ProteoBench
36
+
37
+ ProteoBench annotation, mixed-species diagnostics, and scoring for storage-neutral APB2
38
+ results. HYE and HY use the same configuration-driven calculation.
39
+
40
+ ## Installation
41
+
42
+ APB ProteoBench requires Python 3.13 or later.
43
+
44
+ ```bash
45
+ pip install apb-proteobench
46
+ ```
47
+
48
+ ## Usage
49
+
50
+ Run the complete workflow directly from vendor files:
51
+
52
+ ```bash
53
+ apb-proteobench run quant report.tsv proteins.fasta \
54
+ --params search-parameters.txt \
55
+ --module module_settings.toml \
56
+ --software spectronaut \
57
+ --level ion \
58
+ --output results/scored.h5ad
59
+ ```
60
+
61
+ This one call converts the selected APB2 quantification level, verifies modification-stripped peptide sequences against the FASTA database, applies the ProteoBench sample design, calculates diagnostics and scores, and writes one final APB2 result. Omit `--level` to convert every compatible level. Single-level H5AD and multi-level H5MU targets are supported alongside Parquet and DuckDB. Intermediate values remain storage-neutral `ParsedLevels`, and APB2's `write_parsed_levels` selects persistence from the target suffix. It scores the level named in `module_settings.toml` as the vendor table reports it and performs no quantitative aggregation.
62
+
63
+ The equivalent reusable workflow calls the existing tools directly:
64
+
65
+ ```bash
66
+ apb2 convert report.tsv --params search-parameters.txt --software spectronaut --output results/all
67
+ apb-fasta verify-peptides results/all.h5mu proteins.fasta --output results/fasta-checked.h5mu
68
+ apb-aggregate ion protein sum results/fasta-checked.h5mu results/aggregated.h5mu
69
+ apb-proteobench benchmark results/aggregated.h5mu module_settings.toml results/scored.h5mu
70
+ ```
71
+
72
+ The `apb-aggregate` step is optional: include it only when the scored level must be derived from a lower one. Omit it and `benchmark` reads `results/fasta-checked.h5mu` directly.
73
+
74
+ APB ProteoBench declares only `apb2` and `apb-fasta`, and reaches aggregation solely as a subprocess.
75
+
76
+ The CLI intentionally exposes only `run` and `benchmark`; APB2 owns conversion and persistence, while the Python API exposes in-memory ProteoBench analysis over canonical `ParsedLevels`. `benchmark` combines annotation and scoring for an existing APB2 result. Scoring uses one abundance layer: `--layer NAME`, by default `X`, the APB primary layer projected to AnnData `X`. Per-layer diagnostics live in `varm["proteobench:<layer-name>"]`, while selection provenance and layer-keyed scores live in `metadata["proteobench"]`. See the [documentation](https://anndata-omics-bridge.github.io/apb-proteobench/).
77
+
78
+ To feed the existing pMultiQC ProteoBench module and retain the matching ProteoBot result, add `--result-performance reports/result_performance.csv` to `run` or `benchmark`. The option publishes two staged files in the same directory: `result_performance.csv` plus `<intermediate_hash>.json` in the schema and naming convention used by `Proteobench/Results_quant_ion_DDA`. Each file is published atomically without overwrite, and a failed publication rolls back any file added by the same call. This export writes the scored layer, supports only an ion level, and refuses either existing target. Its SHA-256 submission hash is computed only during this export from the selected layer and scoring inputs. The new identity does not match historical SHA-1 uploads.
79
+
80
+ For optional operation-level timing files, pass `--timings-dir DIR` to `run`. It writes separate JSON files for APB2 conversion, FASTA verification, and ProteoBench benchmarking without changing the scored result or Studio's process-level runtime measurement.
81
+
82
+ The package owns all 11 ProteoBench module TOMLs: the nine quantitative HYE/HY and plasma modules
83
+ and the newer de novo and entrapment documents. Entrapment is scored by `run entrapment`; de novo
84
+ is packaged for planned support. Neither is accepted by the quantitative scorer. Every quantitative module's scores
85
+ include ProteoBench's plasma metrics; see [results](https://anndata-omics-bridge.github.io/apb-proteobench/results/). Stable names,
86
+ support status, and the Python loading API are documented under
87
+ [module configuration](https://anndata-omics-bridge.github.io/apb-proteobench/configuration/).
88
+
89
+ ## Development
90
+
91
+ ```bash
92
+ uv sync --group dev
93
+ make check
94
+ .venv/bin/pre-commit install --hook-type pre-commit --hook-type pre-push
95
+ ```
96
+
97
+ All Python commands run from the synchronized project `.venv`.
@@ -0,0 +1,63 @@
1
+ # APB ProteoBench
2
+
3
+ ProteoBench annotation, mixed-species diagnostics, and scoring for storage-neutral APB2
4
+ results. HYE and HY use the same configuration-driven calculation.
5
+
6
+ ## Installation
7
+
8
+ APB ProteoBench requires Python 3.13 or later.
9
+
10
+ ```bash
11
+ pip install apb-proteobench
12
+ ```
13
+
14
+ ## Usage
15
+
16
+ Run the complete workflow directly from vendor files:
17
+
18
+ ```bash
19
+ apb-proteobench run quant report.tsv proteins.fasta \
20
+ --params search-parameters.txt \
21
+ --module module_settings.toml \
22
+ --software spectronaut \
23
+ --level ion \
24
+ --output results/scored.h5ad
25
+ ```
26
+
27
+ This one call converts the selected APB2 quantification level, verifies modification-stripped peptide sequences against the FASTA database, applies the ProteoBench sample design, calculates diagnostics and scores, and writes one final APB2 result. Omit `--level` to convert every compatible level. Single-level H5AD and multi-level H5MU targets are supported alongside Parquet and DuckDB. Intermediate values remain storage-neutral `ParsedLevels`, and APB2's `write_parsed_levels` selects persistence from the target suffix. It scores the level named in `module_settings.toml` as the vendor table reports it and performs no quantitative aggregation.
28
+
29
+ The equivalent reusable workflow calls the existing tools directly:
30
+
31
+ ```bash
32
+ apb2 convert report.tsv --params search-parameters.txt --software spectronaut --output results/all
33
+ apb-fasta verify-peptides results/all.h5mu proteins.fasta --output results/fasta-checked.h5mu
34
+ apb-aggregate ion protein sum results/fasta-checked.h5mu results/aggregated.h5mu
35
+ apb-proteobench benchmark results/aggregated.h5mu module_settings.toml results/scored.h5mu
36
+ ```
37
+
38
+ The `apb-aggregate` step is optional: include it only when the scored level must be derived from a lower one. Omit it and `benchmark` reads `results/fasta-checked.h5mu` directly.
39
+
40
+ APB ProteoBench declares only `apb2` and `apb-fasta`, and reaches aggregation solely as a subprocess.
41
+
42
+ The CLI intentionally exposes only `run` and `benchmark`; APB2 owns conversion and persistence, while the Python API exposes in-memory ProteoBench analysis over canonical `ParsedLevels`. `benchmark` combines annotation and scoring for an existing APB2 result. Scoring uses one abundance layer: `--layer NAME`, by default `X`, the APB primary layer projected to AnnData `X`. Per-layer diagnostics live in `varm["proteobench:<layer-name>"]`, while selection provenance and layer-keyed scores live in `metadata["proteobench"]`. See the [documentation](https://anndata-omics-bridge.github.io/apb-proteobench/).
43
+
44
+ To feed the existing pMultiQC ProteoBench module and retain the matching ProteoBot result, add `--result-performance reports/result_performance.csv` to `run` or `benchmark`. The option publishes two staged files in the same directory: `result_performance.csv` plus `<intermediate_hash>.json` in the schema and naming convention used by `Proteobench/Results_quant_ion_DDA`. Each file is published atomically without overwrite, and a failed publication rolls back any file added by the same call. This export writes the scored layer, supports only an ion level, and refuses either existing target. Its SHA-256 submission hash is computed only during this export from the selected layer and scoring inputs. The new identity does not match historical SHA-1 uploads.
45
+
46
+ For optional operation-level timing files, pass `--timings-dir DIR` to `run`. It writes separate JSON files for APB2 conversion, FASTA verification, and ProteoBench benchmarking without changing the scored result or Studio's process-level runtime measurement.
47
+
48
+ The package owns all 11 ProteoBench module TOMLs: the nine quantitative HYE/HY and plasma modules
49
+ and the newer de novo and entrapment documents. Entrapment is scored by `run entrapment`; de novo
50
+ is packaged for planned support. Neither is accepted by the quantitative scorer. Every quantitative module's scores
51
+ include ProteoBench's plasma metrics; see [results](https://anndata-omics-bridge.github.io/apb-proteobench/results/). Stable names,
52
+ support status, and the Python loading API are documented under
53
+ [module configuration](https://anndata-omics-bridge.github.io/apb-proteobench/configuration/).
54
+
55
+ ## Development
56
+
57
+ ```bash
58
+ uv sync --group dev
59
+ make check
60
+ .venv/bin/pre-commit install --hook-type pre-commit --hook-type pre-push
61
+ ```
62
+
63
+ All Python commands run from the synchronized project `.venv`.
@@ -0,0 +1,7 @@
1
+ # Third-party notices
2
+
3
+ The modification renderer in `src/apb_proteobench/calculation/mapping.py` and the ProteoBench module TOMLs under
4
+ `src/apb_proteobench/data/modules/` are derived from the ProteoBench project and are distributed
5
+ under the Apache License 2.0. Their upstream attribution is retained in
6
+ `src/apb_proteobench/data/RENDERER_NOTICE.md` and `src/apb_proteobench/data/MODULES_NOTICE.md`. The
7
+ license text is retained beside them as `src/apb_proteobench/data/APACHE-2.0.txt`.
@@ -0,0 +1,161 @@
1
+ [build-system]
2
+ requires = ["uv_build>=0.9.26,<0.10.0"]
3
+ build-backend = "uv_build"
4
+
5
+ [project]
6
+ name = "apb-proteobench"
7
+ version = "0.1.0"
8
+ description = "ProteoBench annotation, diagnostics, and scoring for APB2 results"
9
+ readme = "README.md"
10
+ requires-python = ">=3.13"
11
+ license = "MIT"
12
+ license-files = [
13
+ "LICENSE",
14
+ "THIRD_PARTY_NOTICES.md",
15
+ "src/apb_proteobench/data/APACHE-2.0.txt",
16
+ ]
17
+ keywords = [
18
+ "proteomics",
19
+ "proteobench",
20
+ "benchmarking",
21
+ "anndata",
22
+ "mass spectrometry",
23
+ ]
24
+ classifiers = [
25
+ "Development Status :: 3 - Alpha",
26
+ "Intended Audience :: Science/Research",
27
+ "Operating System :: OS Independent",
28
+ "Programming Language :: Python :: 3",
29
+ "Programming Language :: Python :: 3.13",
30
+ "Topic :: Scientific/Engineering :: Bio-Informatics",
31
+ "Typing :: Typed",
32
+ ]
33
+ dependencies = [
34
+ "apb2>=0.1,<0.2",
35
+ "apb-catalog>=0.1,<0.2",
36
+ "apb-fasta>=0.1,<0.2",
37
+ "cyclopts>=4,<5",
38
+ "loguru>=0.7,<1",
39
+ "numpy>=2,<3",
40
+ "pandas>=2.3,<4",
41
+ "polars>=1.38,<2",
42
+ "pydantic>=2.12,<3",
43
+ "scipy>=1.16,<2",
44
+ ]
45
+
46
+ [[project.authors]]
47
+ name = "Witold Wolski"
48
+ email = "wew@fgcz.ethz.ch"
49
+
50
+ [project.scripts]
51
+ apb-proteobench = "apb_proteobench.cli.app:main"
52
+
53
+ [project.urls]
54
+ Documentation = "https://anndata-omics-bridge.github.io/apb-proteobench/"
55
+ Repository = "https://github.com/anndata-omics-bridge/apb-proteobench"
56
+ Issues = "https://github.com/anndata-omics-bridge/apb-proteobench/issues"
57
+
58
+ [dependency-groups]
59
+ dev = [
60
+ "build>=1.3,<2",
61
+ "deptry>=0.24,<1",
62
+ "import-linter>=2.9,<3",
63
+ "pandas-stubs>=2.3,<3",
64
+ "pre-commit>=4,<5",
65
+ "pyright>=1.1.400,<2",
66
+ "pytest>=9,<10",
67
+ "pytest-cov>=7,<8",
68
+ "ruff>=0.15,<1",
69
+ "sdrf-pipelines>=0.1.6,<0.2",
70
+ "twine>=6,<7",
71
+ ]
72
+ docs = [
73
+ "mkdocstrings[python]>=1.0,<2",
74
+ "pymdown-extensions>=11,<12",
75
+ "zensical==0.0.43",
76
+ ]
77
+
78
+ [tool.ruff]
79
+ line-length = 100
80
+ target-version = "py313"
81
+ src = [
82
+ "src",
83
+ "tests",
84
+ ]
85
+
86
+ [tool.ruff.lint]
87
+ select = [
88
+ "ANN",
89
+ "B",
90
+ "C4",
91
+ "C90",
92
+ "E4",
93
+ "E7",
94
+ "E9",
95
+ "F",
96
+ "I",
97
+ "PGH",
98
+ "PIE",
99
+ "RUF",
100
+ "SIM",
101
+ "UP",
102
+ ]
103
+
104
+ [tool.ruff.lint.mccabe]
105
+ max-complexity = 10
106
+
107
+ [tool.ruff.format]
108
+ docstring-code-format = true
109
+
110
+ [tool.pyright]
111
+ include = [
112
+ "src",
113
+ "tests",
114
+ ]
115
+ venvPath = "."
116
+ venv = ".venv"
117
+ pythonVersion = "3.13"
118
+ typeCheckingMode = "strict"
119
+ reportImportCycles = "error"
120
+ reportUnnecessaryTypeIgnoreComment = "error"
121
+ reportImplicitOverride = "error"
122
+ enableTypeIgnoreComments = false
123
+ reportMissingTypeStubs = "none"
124
+ reportUnknownMemberType = "none"
125
+ reportUnknownVariableType = "none"
126
+ reportUnknownArgumentType = "none"
127
+ reportUnknownParameterType = "none"
128
+ reportUnknownLambdaType = "none"
129
+
130
+ [tool.pytest.ini_options]
131
+ addopts = [
132
+ "--strict-config",
133
+ "--strict-markers",
134
+ "-ra",
135
+ ]
136
+ testpaths = ["tests"]
137
+ xfail_strict = true
138
+
139
+ [tool.coverage.run]
140
+ branch = true
141
+ source = ["apb_proteobench"]
142
+
143
+ [tool.coverage.report]
144
+ fail_under = 80
145
+ show_missing = true
146
+ skip_covered = true
147
+
148
+ [tool.deptry]
149
+ known_first_party = ["apb_proteobench"]
150
+
151
+ [tool.uv.sources.apb2]
152
+ path = "../apb2"
153
+ editable = true
154
+
155
+ [tool.uv.sources.apb-catalog]
156
+ path = "../apb-catalog"
157
+ editable = true
158
+
159
+ [tool.uv.sources.apb-fasta]
160
+ path = "../apb-fasta"
161
+ editable = true
@@ -0,0 +1,135 @@
1
+ [build-system]
2
+ requires = ["uv_build>=0.9.26,<0.10.0"]
3
+ build-backend = "uv_build"
4
+
5
+ [project]
6
+ name = "apb-proteobench"
7
+ version = "0.1.0"
8
+ description = "ProteoBench annotation, diagnostics, and scoring for APB2 results"
9
+ readme = "README.md"
10
+ requires-python = ">=3.13"
11
+ license = "MIT"
12
+ license-files = ["LICENSE", "THIRD_PARTY_NOTICES.md", "src/apb_proteobench/data/APACHE-2.0.txt"]
13
+ authors = [
14
+ { name = "Witold Wolski", email = "wew@fgcz.ethz.ch" },
15
+ ]
16
+ keywords = ["proteomics", "proteobench", "benchmarking", "anndata", "mass spectrometry"]
17
+ classifiers = [
18
+ "Development Status :: 3 - Alpha",
19
+ "Intended Audience :: Science/Research",
20
+ "Operating System :: OS Independent",
21
+ "Programming Language :: Python :: 3",
22
+ "Programming Language :: Python :: 3.13",
23
+ "Topic :: Scientific/Engineering :: Bio-Informatics",
24
+ "Typing :: Typed",
25
+ ]
26
+ dependencies = [
27
+ "apb2>=0.1,<0.2",
28
+ "apb-catalog>=0.1,<0.2",
29
+ "apb-fasta>=0.1,<0.2",
30
+ "cyclopts>=4,<5",
31
+ "loguru>=0.7,<1",
32
+ "numpy>=2,<3",
33
+ "pandas>=2.3,<4",
34
+ "polars>=1.38,<2",
35
+ "pydantic>=2.12,<3",
36
+ "scipy>=1.16,<2",
37
+ ]
38
+
39
+ [project.scripts]
40
+ apb-proteobench = "apb_proteobench.cli.app:main"
41
+
42
+ [project.urls]
43
+ Documentation = "https://anndata-omics-bridge.github.io/apb-proteobench/"
44
+ Repository = "https://github.com/anndata-omics-bridge/apb-proteobench"
45
+ Issues = "https://github.com/anndata-omics-bridge/apb-proteobench/issues"
46
+
47
+ [dependency-groups]
48
+ dev = [
49
+ "build>=1.3,<2",
50
+ "deptry>=0.24,<1",
51
+ "import-linter>=2.9,<3",
52
+ "pandas-stubs>=2.3,<3",
53
+ "pre-commit>=4,<5",
54
+ "pyright>=1.1.400,<2",
55
+ "pytest>=9,<10",
56
+ "pytest-cov>=7,<8",
57
+ "ruff>=0.15,<1",
58
+ "sdrf-pipelines>=0.1.6,<0.2",
59
+ "twine>=6,<7",
60
+ ]
61
+ docs = [
62
+ "mkdocstrings[python]>=1.0,<2",
63
+ "pymdown-extensions>=11,<12",
64
+ "zensical==0.0.43",
65
+ ]
66
+
67
+ [tool.ruff]
68
+ line-length = 100
69
+ target-version = "py313"
70
+ src = ["src", "tests"]
71
+
72
+ [tool.ruff.lint]
73
+ select = [
74
+ "ANN",
75
+ "B",
76
+ "C4",
77
+ "C90",
78
+ "E4",
79
+ "E7",
80
+ "E9",
81
+ "F",
82
+ "I",
83
+ "PGH",
84
+ "PIE",
85
+ "RUF",
86
+ "SIM",
87
+ "UP",
88
+ ]
89
+
90
+ [tool.ruff.lint.mccabe]
91
+ max-complexity = 10
92
+
93
+ [tool.ruff.format]
94
+ docstring-code-format = true
95
+
96
+ [tool.pyright]
97
+ include = ["src", "tests"]
98
+ venvPath = "."
99
+ venv = ".venv"
100
+ pythonVersion = "3.13"
101
+ typeCheckingMode = "strict"
102
+ reportImportCycles = "error"
103
+ reportUnnecessaryTypeIgnoreComment = "error"
104
+ reportImplicitOverride = "error"
105
+ enableTypeIgnoreComments = false
106
+ # SciPy and pandas still propagate Unknown through supported public operations.
107
+ # Keep strict first-party checking while suppressing diagnostics with only that signal.
108
+ reportMissingTypeStubs = "none"
109
+ reportUnknownMemberType = "none"
110
+ reportUnknownVariableType = "none"
111
+ reportUnknownArgumentType = "none"
112
+ reportUnknownParameterType = "none"
113
+ reportUnknownLambdaType = "none"
114
+
115
+ [tool.pytest.ini_options]
116
+ addopts = ["--strict-config", "--strict-markers", "-ra"]
117
+ testpaths = ["tests"]
118
+ xfail_strict = true
119
+
120
+ [tool.coverage.run]
121
+ branch = true
122
+ source = ["apb_proteobench"]
123
+
124
+ [tool.coverage.report]
125
+ fail_under = 80
126
+ show_missing = true
127
+ skip_covered = true
128
+
129
+ [tool.deptry]
130
+ known_first_party = ["apb_proteobench"]
131
+
132
+ [tool.uv.sources]
133
+ apb2 = { path = "../apb2", editable = true }
134
+ apb-catalog = { path = "../apb-catalog", editable = true }
135
+ apb-fasta = { path = "../apb-fasta", editable = true }
@@ -0,0 +1,93 @@
1
+ """Bind one ProteoBench module to the configured APB quantification level."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from copy import deepcopy
6
+ from dataclasses import dataclass, replace
7
+ from pathlib import Path
8
+
9
+ import polars as pl
10
+ from apb2.api import AnnotationCompiler, AnnotationError, JsonValue, ParsedLevels
11
+
12
+ from apb_proteobench.configuration.load import LoadedModule, load_module
13
+
14
+ _STORAGE_KEY = "proteobench"
15
+
16
+
17
+ @dataclass(frozen=True, slots=True)
18
+ class ProteoBenchAnnotation:
19
+ """A dataset annotated with every module sample, awaiting its module provenance."""
20
+
21
+ annotated: ParsedLevels
22
+ module: LoadedModule
23
+
24
+ def annotate(self) -> ParsedLevels:
25
+ """Store the normalized module beside APB2's own annotation provenance."""
26
+ metadata = deepcopy(self.annotated.metadata)
27
+ section = metadata.setdefault(_STORAGE_KEY, {})
28
+ if not isinstance(section, dict):
29
+ raise AnnotationError("ProteoBench metadata must be an object")
30
+ provenance = section.setdefault("provenance", {})
31
+ if not isinstance(provenance, dict):
32
+ raise AnnotationError("ProteoBench provenance must be an object")
33
+ record: dict[str, JsonValue] = {**self.module.metadata(), "schema_version": "2"}
34
+ provenance["annotation"] = record
35
+ return replace(self.annotated, metadata=metadata)
36
+
37
+
38
+ @dataclass(frozen=True, slots=True)
39
+ class ProteoBenchAnnotationParser:
40
+ """A validated module source ready to bind to one APB result."""
41
+
42
+ module: LoadedModule
43
+
44
+ @classmethod
45
+ def from_path(cls, path: Path, /) -> ProteoBenchAnnotationParser:
46
+ """Decode and validate the complete module document once."""
47
+ return cls(load_module(path))
48
+
49
+ def parse(self, parsed: ParsedLevels, /) -> ProteoBenchAnnotation:
50
+ """Match every run to exactly one module sample and every sample to a run."""
51
+ level_name = self.module.settings.general.level
52
+ if level_name not in parsed.levels:
53
+ raise AnnotationError(
54
+ f"ProteoBench module selects unavailable level {level_name!r}; "
55
+ f"available={list(parsed.levels)}"
56
+ )
57
+ annotation = AnnotationCompiler("error").compile(_sample_frame(self.module)).parse(parsed)
58
+ coverage = annotation.matches.levels[level_name].coverage
59
+ if coverage.annotation_only_count:
60
+ raise AnnotationError(
61
+ "ProteoBench module contains samples absent from quantification; "
62
+ f"count={coverage.annotation_only_count}, "
63
+ f"examples={list(coverage.annotation_only_examples)}"
64
+ )
65
+ return ProteoBenchAnnotation(annotated=annotation.annotate().parsed, module=self.module)
66
+
67
+
68
+ def _sample_frame(module: LoadedModule) -> pl.DataFrame:
69
+ """The module samples as a prolfquapp table keyed by raw file, with its aliases."""
70
+ samples = module.settings.samples
71
+ return pl.DataFrame(
72
+ {
73
+ "raw_file": [sample.raw_file for sample in samples],
74
+ "raw_file_aliases": [
75
+ list(
76
+ dict.fromkeys(
77
+ identifier
78
+ for identifier in (*sample.raw_file_aliases, sample.sample_name)
79
+ if identifier != sample.raw_file
80
+ )
81
+ )
82
+ for sample in samples
83
+ ],
84
+ "sample_name": [sample.sample_name for sample in samples],
85
+ "condition": [sample.condition for sample in samples],
86
+ },
87
+ schema={
88
+ "raw_file": pl.String,
89
+ "raw_file_aliases": pl.List(pl.String),
90
+ "sample_name": pl.String,
91
+ "condition": pl.String,
92
+ },
93
+ )