apb-proteobench 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- apb_proteobench-0.1.0/LICENSE +21 -0
- apb_proteobench-0.1.0/PKG-INFO +97 -0
- apb_proteobench-0.1.0/README.md +63 -0
- apb_proteobench-0.1.0/THIRD_PARTY_NOTICES.md +7 -0
- apb_proteobench-0.1.0/pyproject.toml +161 -0
- apb_proteobench-0.1.0/pyproject.toml.orig +135 -0
- apb_proteobench-0.1.0/src/apb_proteobench/__init__.py +1 -0
- apb_proteobench-0.1.0/src/apb_proteobench/annotation.py +93 -0
- apb_proteobench-0.1.0/src/apb_proteobench/api.py +113 -0
- apb_proteobench-0.1.0/src/apb_proteobench/calculation/__init__.py +0 -0
- apb_proteobench-0.1.0/src/apb_proteobench/calculation/contracts.py +35 -0
- apb_proteobench-0.1.0/src/apb_proteobench/calculation/entrapment.py +254 -0
- apb_proteobench-0.1.0/src/apb_proteobench/calculation/intermediate.py +560 -0
- apb_proteobench-0.1.0/src/apb_proteobench/calculation/mapping.py +44 -0
- apb_proteobench-0.1.0/src/apb_proteobench/calculation/metrics.py +283 -0
- apb_proteobench-0.1.0/src/apb_proteobench/cli/__init__.py +0 -0
- apb_proteobench-0.1.0/src/apb_proteobench/cli/app.py +754 -0
- apb_proteobench-0.1.0/src/apb_proteobench/cli/presentation.py +71 -0
- apb_proteobench-0.1.0/src/apb_proteobench/cli/result_performance.py +360 -0
- apb_proteobench-0.1.0/src/apb_proteobench/cli/timings.py +59 -0
- apb_proteobench-0.1.0/src/apb_proteobench/configuration/__init__.py +0 -0
- apb_proteobench-0.1.0/src/apb_proteobench/configuration/design.py +149 -0
- apb_proteobench-0.1.0/src/apb_proteobench/configuration/entrapment.py +56 -0
- apb_proteobench-0.1.0/src/apb_proteobench/configuration/load.py +151 -0
- apb_proteobench-0.1.0/src/apb_proteobench/configuration/schema.py +114 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/APACHE-2.0.txt +201 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/MODULES_NOTICE.md +27 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/RENDERER_NOTICE.md +5 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/__init__.py +0 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/modules/__init__.py +0 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dda_astral.sdrf.tsv +7 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dda_astral.toml +32 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dda_peptidoform.sdrf.tsv +7 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dda_peptidoform.toml +41 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dda_qexactive.sdrf.tsv +7 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dda_qexactive.toml +41 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/modules/denovo_dda_hcd.toml +6 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dia_aif.sdrf.tsv +7 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dia_aif.toml +32 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dia_astral.sdrf.tsv +7 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dia_astral.toml +41 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dia_diapasef.sdrf.tsv +7 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dia_diapasef.toml +41 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dia_plasma.sdrf.tsv +13 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dia_plasma.toml +50 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dia_singlecell.sdrf.tsv +7 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dia_singlecell.toml +36 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dia_zenotof.sdrf.tsv +7 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/modules/dia_zenotof.toml +32 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/modules/entrapment_dia_astral.sdrf.tsv +4 -0
- apb_proteobench-0.1.0/src/apb_proteobench/data/modules/entrapment_dia_astral.toml +5 -0
- apb_proteobench-0.1.0/src/apb_proteobench/entrapment.py +164 -0
- apb_proteobench-0.1.0/src/apb_proteobench/integration.py +326 -0
- apb_proteobench-0.1.0/src/apb_proteobench/py.typed +1 -0
- apb_proteobench-0.1.0/src/apb_proteobench/workflow.py +121 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Witold Wolski
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: apb-proteobench
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: ProteoBench annotation, diagnostics, and scoring for APB2 results
|
|
5
|
+
Keywords: proteomics,proteobench,benchmarking,anndata,mass spectrometry
|
|
6
|
+
Author: Witold Wolski
|
|
7
|
+
Author-email: Witold Wolski <wew@fgcz.ethz.ch>
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
License-File: THIRD_PARTY_NOTICES.md
|
|
11
|
+
License-File: src/apb_proteobench/data/APACHE-2.0.txt
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
17
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
18
|
+
Classifier: Typing :: Typed
|
|
19
|
+
Requires-Dist: apb2>=0.1,<0.2
|
|
20
|
+
Requires-Dist: apb-catalog>=0.1,<0.2
|
|
21
|
+
Requires-Dist: apb-fasta>=0.1,<0.2
|
|
22
|
+
Requires-Dist: cyclopts>=4,<5
|
|
23
|
+
Requires-Dist: loguru>=0.7,<1
|
|
24
|
+
Requires-Dist: numpy>=2,<3
|
|
25
|
+
Requires-Dist: pandas>=2.3,<4
|
|
26
|
+
Requires-Dist: polars>=1.38,<2
|
|
27
|
+
Requires-Dist: pydantic>=2.12,<3
|
|
28
|
+
Requires-Dist: scipy>=1.16,<2
|
|
29
|
+
Requires-Python: >=3.13
|
|
30
|
+
Project-URL: Documentation, https://anndata-omics-bridge.github.io/apb-proteobench/
|
|
31
|
+
Project-URL: Repository, https://github.com/anndata-omics-bridge/apb-proteobench
|
|
32
|
+
Project-URL: Issues, https://github.com/anndata-omics-bridge/apb-proteobench/issues
|
|
33
|
+
Description-Content-Type: text/markdown
|
|
34
|
+
|
|
35
|
+
# APB ProteoBench
|
|
36
|
+
|
|
37
|
+
ProteoBench annotation, mixed-species diagnostics, and scoring for storage-neutral APB2
|
|
38
|
+
results. HYE and HY use the same configuration-driven calculation.
|
|
39
|
+
|
|
40
|
+
## Installation
|
|
41
|
+
|
|
42
|
+
APB ProteoBench requires Python 3.13 or later.
|
|
43
|
+
|
|
44
|
+
```bash
|
|
45
|
+
pip install apb-proteobench
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
## Usage
|
|
49
|
+
|
|
50
|
+
Run the complete workflow directly from vendor files:
|
|
51
|
+
|
|
52
|
+
```bash
|
|
53
|
+
apb-proteobench run quant report.tsv proteins.fasta \
|
|
54
|
+
--params search-parameters.txt \
|
|
55
|
+
--module module_settings.toml \
|
|
56
|
+
--software spectronaut \
|
|
57
|
+
--level ion \
|
|
58
|
+
--output results/scored.h5ad
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
This one call converts the selected APB2 quantification level, verifies modification-stripped peptide sequences against the FASTA database, applies the ProteoBench sample design, calculates diagnostics and scores, and writes one final APB2 result. Omit `--level` to convert every compatible level. Single-level H5AD and multi-level H5MU targets are supported alongside Parquet and DuckDB. Intermediate values remain storage-neutral `ParsedLevels`, and APB2's `write_parsed_levels` selects persistence from the target suffix. It scores the level named in `module_settings.toml` as the vendor table reports it and performs no quantitative aggregation.
|
|
62
|
+
|
|
63
|
+
The equivalent reusable workflow calls the existing tools directly:
|
|
64
|
+
|
|
65
|
+
```bash
|
|
66
|
+
apb2 convert report.tsv --params search-parameters.txt --software spectronaut --output results/all
|
|
67
|
+
apb-fasta verify-peptides results/all.h5mu proteins.fasta --output results/fasta-checked.h5mu
|
|
68
|
+
apb-aggregate ion protein sum results/fasta-checked.h5mu results/aggregated.h5mu
|
|
69
|
+
apb-proteobench benchmark results/aggregated.h5mu module_settings.toml results/scored.h5mu
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
The `apb-aggregate` step is optional: include it only when the scored level must be derived from a lower one. Omit it and `benchmark` reads `results/fasta-checked.h5mu` directly.
|
|
73
|
+
|
|
74
|
+
APB ProteoBench declares only `apb2` and `apb-fasta`, and reaches aggregation solely as a subprocess.
|
|
75
|
+
|
|
76
|
+
The CLI intentionally exposes only `run` and `benchmark`; APB2 owns conversion and persistence, while the Python API exposes in-memory ProteoBench analysis over canonical `ParsedLevels`. `benchmark` combines annotation and scoring for an existing APB2 result. Scoring uses one abundance layer: `--layer NAME`, by default `X`, the APB primary layer projected to AnnData `X`. Per-layer diagnostics live in `varm["proteobench:<layer-name>"]`, while selection provenance and layer-keyed scores live in `metadata["proteobench"]`. See the [documentation](https://anndata-omics-bridge.github.io/apb-proteobench/).
|
|
77
|
+
|
|
78
|
+
To feed the existing pMultiQC ProteoBench module and retain the matching ProteoBot result, add `--result-performance reports/result_performance.csv` to `run` or `benchmark`. The option publishes two staged files in the same directory: `result_performance.csv` plus `<intermediate_hash>.json` in the schema and naming convention used by `Proteobench/Results_quant_ion_DDA`. Each file is published atomically without overwrite, and a failed publication rolls back any file added by the same call. This export writes the scored layer, supports only an ion level, and refuses either existing target. Its SHA-256 submission hash is computed only during this export from the selected layer and scoring inputs. The new identity does not match historical SHA-1 uploads.
|
|
79
|
+
|
|
80
|
+
For optional operation-level timing files, pass `--timings-dir DIR` to `run`. It writes separate JSON files for APB2 conversion, FASTA verification, and ProteoBench benchmarking without changing the scored result or Studio's process-level runtime measurement.
|
|
81
|
+
|
|
82
|
+
The package owns all 11 ProteoBench module TOMLs: the nine quantitative HYE/HY and plasma modules
|
|
83
|
+
and the newer de novo and entrapment documents. Entrapment is scored by `run entrapment`; de novo
|
|
84
|
+
is packaged for planned support. Neither is accepted by the quantitative scorer. Every quantitative module's scores
|
|
85
|
+
include ProteoBench's plasma metrics; see [results](https://anndata-omics-bridge.github.io/apb-proteobench/results/). Stable names,
|
|
86
|
+
support status, and the Python loading API are documented under
|
|
87
|
+
[module configuration](https://anndata-omics-bridge.github.io/apb-proteobench/configuration/).
|
|
88
|
+
|
|
89
|
+
## Development
|
|
90
|
+
|
|
91
|
+
```bash
|
|
92
|
+
uv sync --group dev
|
|
93
|
+
make check
|
|
94
|
+
.venv/bin/pre-commit install --hook-type pre-commit --hook-type pre-push
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
All Python commands run from the synchronized project `.venv`.
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
# APB ProteoBench
|
|
2
|
+
|
|
3
|
+
ProteoBench annotation, mixed-species diagnostics, and scoring for storage-neutral APB2
|
|
4
|
+
results. HYE and HY use the same configuration-driven calculation.
|
|
5
|
+
|
|
6
|
+
## Installation
|
|
7
|
+
|
|
8
|
+
APB ProteoBench requires Python 3.13 or later.
|
|
9
|
+
|
|
10
|
+
```bash
|
|
11
|
+
pip install apb-proteobench
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
## Usage
|
|
15
|
+
|
|
16
|
+
Run the complete workflow directly from vendor files:
|
|
17
|
+
|
|
18
|
+
```bash
|
|
19
|
+
apb-proteobench run quant report.tsv proteins.fasta \
|
|
20
|
+
--params search-parameters.txt \
|
|
21
|
+
--module module_settings.toml \
|
|
22
|
+
--software spectronaut \
|
|
23
|
+
--level ion \
|
|
24
|
+
--output results/scored.h5ad
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
This one call converts the selected APB2 quantification level, verifies modification-stripped peptide sequences against the FASTA database, applies the ProteoBench sample design, calculates diagnostics and scores, and writes one final APB2 result. Omit `--level` to convert every compatible level. Single-level H5AD and multi-level H5MU targets are supported alongside Parquet and DuckDB. Intermediate values remain storage-neutral `ParsedLevels`, and APB2's `write_parsed_levels` selects persistence from the target suffix. It scores the level named in `module_settings.toml` as the vendor table reports it and performs no quantitative aggregation.
|
|
28
|
+
|
|
29
|
+
The equivalent reusable workflow calls the existing tools directly:
|
|
30
|
+
|
|
31
|
+
```bash
|
|
32
|
+
apb2 convert report.tsv --params search-parameters.txt --software spectronaut --output results/all
|
|
33
|
+
apb-fasta verify-peptides results/all.h5mu proteins.fasta --output results/fasta-checked.h5mu
|
|
34
|
+
apb-aggregate ion protein sum results/fasta-checked.h5mu results/aggregated.h5mu
|
|
35
|
+
apb-proteobench benchmark results/aggregated.h5mu module_settings.toml results/scored.h5mu
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
The `apb-aggregate` step is optional: include it only when the scored level must be derived from a lower one. Omit it and `benchmark` reads `results/fasta-checked.h5mu` directly.
|
|
39
|
+
|
|
40
|
+
APB ProteoBench declares only `apb2` and `apb-fasta`, and reaches aggregation solely as a subprocess.
|
|
41
|
+
|
|
42
|
+
The CLI intentionally exposes only `run` and `benchmark`; APB2 owns conversion and persistence, while the Python API exposes in-memory ProteoBench analysis over canonical `ParsedLevels`. `benchmark` combines annotation and scoring for an existing APB2 result. Scoring uses one abundance layer: `--layer NAME`, by default `X`, the APB primary layer projected to AnnData `X`. Per-layer diagnostics live in `varm["proteobench:<layer-name>"]`, while selection provenance and layer-keyed scores live in `metadata["proteobench"]`. See the [documentation](https://anndata-omics-bridge.github.io/apb-proteobench/).
|
|
43
|
+
|
|
44
|
+
To feed the existing pMultiQC ProteoBench module and retain the matching ProteoBot result, add `--result-performance reports/result_performance.csv` to `run` or `benchmark`. The option publishes two staged files in the same directory: `result_performance.csv` plus `<intermediate_hash>.json` in the schema and naming convention used by `Proteobench/Results_quant_ion_DDA`. Each file is published atomically without overwrite, and a failed publication rolls back any file added by the same call. This export writes the scored layer, supports only an ion level, and refuses either existing target. Its SHA-256 submission hash is computed only during this export from the selected layer and scoring inputs. The new identity does not match historical SHA-1 uploads.
|
|
45
|
+
|
|
46
|
+
For optional operation-level timing files, pass `--timings-dir DIR` to `run`. It writes separate JSON files for APB2 conversion, FASTA verification, and ProteoBench benchmarking without changing the scored result or Studio's process-level runtime measurement.
|
|
47
|
+
|
|
48
|
+
The package owns all 11 ProteoBench module TOMLs: the nine quantitative HYE/HY and plasma modules
|
|
49
|
+
and the newer de novo and entrapment documents. Entrapment is scored by `run entrapment`; de novo
|
|
50
|
+
is packaged for planned support. Neither is accepted by the quantitative scorer. Every quantitative module's scores
|
|
51
|
+
include ProteoBench's plasma metrics; see [results](https://anndata-omics-bridge.github.io/apb-proteobench/results/). Stable names,
|
|
52
|
+
support status, and the Python loading API are documented under
|
|
53
|
+
[module configuration](https://anndata-omics-bridge.github.io/apb-proteobench/configuration/).
|
|
54
|
+
|
|
55
|
+
## Development
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
uv sync --group dev
|
|
59
|
+
make check
|
|
60
|
+
.venv/bin/pre-commit install --hook-type pre-commit --hook-type pre-push
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
All Python commands run from the synchronized project `.venv`.
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
# Third-party notices
|
|
2
|
+
|
|
3
|
+
The modification renderer in `src/apb_proteobench/calculation/mapping.py` and the ProteoBench module TOMLs under
|
|
4
|
+
`src/apb_proteobench/data/modules/` are derived from the ProteoBench project and are distributed
|
|
5
|
+
under the Apache License 2.0. Their upstream attribution is retained in
|
|
6
|
+
`src/apb_proteobench/data/RENDERER_NOTICE.md` and `src/apb_proteobench/data/MODULES_NOTICE.md`. The
|
|
7
|
+
license text is retained beside them as `src/apb_proteobench/data/APACHE-2.0.txt`.
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["uv_build>=0.9.26,<0.10.0"]
|
|
3
|
+
build-backend = "uv_build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "apb-proteobench"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "ProteoBench annotation, diagnostics, and scoring for APB2 results"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.13"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = [
|
|
13
|
+
"LICENSE",
|
|
14
|
+
"THIRD_PARTY_NOTICES.md",
|
|
15
|
+
"src/apb_proteobench/data/APACHE-2.0.txt",
|
|
16
|
+
]
|
|
17
|
+
keywords = [
|
|
18
|
+
"proteomics",
|
|
19
|
+
"proteobench",
|
|
20
|
+
"benchmarking",
|
|
21
|
+
"anndata",
|
|
22
|
+
"mass spectrometry",
|
|
23
|
+
]
|
|
24
|
+
classifiers = [
|
|
25
|
+
"Development Status :: 3 - Alpha",
|
|
26
|
+
"Intended Audience :: Science/Research",
|
|
27
|
+
"Operating System :: OS Independent",
|
|
28
|
+
"Programming Language :: Python :: 3",
|
|
29
|
+
"Programming Language :: Python :: 3.13",
|
|
30
|
+
"Topic :: Scientific/Engineering :: Bio-Informatics",
|
|
31
|
+
"Typing :: Typed",
|
|
32
|
+
]
|
|
33
|
+
dependencies = [
|
|
34
|
+
"apb2>=0.1,<0.2",
|
|
35
|
+
"apb-catalog>=0.1,<0.2",
|
|
36
|
+
"apb-fasta>=0.1,<0.2",
|
|
37
|
+
"cyclopts>=4,<5",
|
|
38
|
+
"loguru>=0.7,<1",
|
|
39
|
+
"numpy>=2,<3",
|
|
40
|
+
"pandas>=2.3,<4",
|
|
41
|
+
"polars>=1.38,<2",
|
|
42
|
+
"pydantic>=2.12,<3",
|
|
43
|
+
"scipy>=1.16,<2",
|
|
44
|
+
]
|
|
45
|
+
|
|
46
|
+
[[project.authors]]
|
|
47
|
+
name = "Witold Wolski"
|
|
48
|
+
email = "wew@fgcz.ethz.ch"
|
|
49
|
+
|
|
50
|
+
[project.scripts]
|
|
51
|
+
apb-proteobench = "apb_proteobench.cli.app:main"
|
|
52
|
+
|
|
53
|
+
[project.urls]
|
|
54
|
+
Documentation = "https://anndata-omics-bridge.github.io/apb-proteobench/"
|
|
55
|
+
Repository = "https://github.com/anndata-omics-bridge/apb-proteobench"
|
|
56
|
+
Issues = "https://github.com/anndata-omics-bridge/apb-proteobench/issues"
|
|
57
|
+
|
|
58
|
+
[dependency-groups]
|
|
59
|
+
dev = [
|
|
60
|
+
"build>=1.3,<2",
|
|
61
|
+
"deptry>=0.24,<1",
|
|
62
|
+
"import-linter>=2.9,<3",
|
|
63
|
+
"pandas-stubs>=2.3,<3",
|
|
64
|
+
"pre-commit>=4,<5",
|
|
65
|
+
"pyright>=1.1.400,<2",
|
|
66
|
+
"pytest>=9,<10",
|
|
67
|
+
"pytest-cov>=7,<8",
|
|
68
|
+
"ruff>=0.15,<1",
|
|
69
|
+
"sdrf-pipelines>=0.1.6,<0.2",
|
|
70
|
+
"twine>=6,<7",
|
|
71
|
+
]
|
|
72
|
+
docs = [
|
|
73
|
+
"mkdocstrings[python]>=1.0,<2",
|
|
74
|
+
"pymdown-extensions>=11,<12",
|
|
75
|
+
"zensical==0.0.43",
|
|
76
|
+
]
|
|
77
|
+
|
|
78
|
+
[tool.ruff]
|
|
79
|
+
line-length = 100
|
|
80
|
+
target-version = "py313"
|
|
81
|
+
src = [
|
|
82
|
+
"src",
|
|
83
|
+
"tests",
|
|
84
|
+
]
|
|
85
|
+
|
|
86
|
+
[tool.ruff.lint]
|
|
87
|
+
select = [
|
|
88
|
+
"ANN",
|
|
89
|
+
"B",
|
|
90
|
+
"C4",
|
|
91
|
+
"C90",
|
|
92
|
+
"E4",
|
|
93
|
+
"E7",
|
|
94
|
+
"E9",
|
|
95
|
+
"F",
|
|
96
|
+
"I",
|
|
97
|
+
"PGH",
|
|
98
|
+
"PIE",
|
|
99
|
+
"RUF",
|
|
100
|
+
"SIM",
|
|
101
|
+
"UP",
|
|
102
|
+
]
|
|
103
|
+
|
|
104
|
+
[tool.ruff.lint.mccabe]
|
|
105
|
+
max-complexity = 10
|
|
106
|
+
|
|
107
|
+
[tool.ruff.format]
|
|
108
|
+
docstring-code-format = true
|
|
109
|
+
|
|
110
|
+
[tool.pyright]
|
|
111
|
+
include = [
|
|
112
|
+
"src",
|
|
113
|
+
"tests",
|
|
114
|
+
]
|
|
115
|
+
venvPath = "."
|
|
116
|
+
venv = ".venv"
|
|
117
|
+
pythonVersion = "3.13"
|
|
118
|
+
typeCheckingMode = "strict"
|
|
119
|
+
reportImportCycles = "error"
|
|
120
|
+
reportUnnecessaryTypeIgnoreComment = "error"
|
|
121
|
+
reportImplicitOverride = "error"
|
|
122
|
+
enableTypeIgnoreComments = false
|
|
123
|
+
reportMissingTypeStubs = "none"
|
|
124
|
+
reportUnknownMemberType = "none"
|
|
125
|
+
reportUnknownVariableType = "none"
|
|
126
|
+
reportUnknownArgumentType = "none"
|
|
127
|
+
reportUnknownParameterType = "none"
|
|
128
|
+
reportUnknownLambdaType = "none"
|
|
129
|
+
|
|
130
|
+
[tool.pytest.ini_options]
|
|
131
|
+
addopts = [
|
|
132
|
+
"--strict-config",
|
|
133
|
+
"--strict-markers",
|
|
134
|
+
"-ra",
|
|
135
|
+
]
|
|
136
|
+
testpaths = ["tests"]
|
|
137
|
+
xfail_strict = true
|
|
138
|
+
|
|
139
|
+
[tool.coverage.run]
|
|
140
|
+
branch = true
|
|
141
|
+
source = ["apb_proteobench"]
|
|
142
|
+
|
|
143
|
+
[tool.coverage.report]
|
|
144
|
+
fail_under = 80
|
|
145
|
+
show_missing = true
|
|
146
|
+
skip_covered = true
|
|
147
|
+
|
|
148
|
+
[tool.deptry]
|
|
149
|
+
known_first_party = ["apb_proteobench"]
|
|
150
|
+
|
|
151
|
+
[tool.uv.sources.apb2]
|
|
152
|
+
path = "../apb2"
|
|
153
|
+
editable = true
|
|
154
|
+
|
|
155
|
+
[tool.uv.sources.apb-catalog]
|
|
156
|
+
path = "../apb-catalog"
|
|
157
|
+
editable = true
|
|
158
|
+
|
|
159
|
+
[tool.uv.sources.apb-fasta]
|
|
160
|
+
path = "../apb-fasta"
|
|
161
|
+
editable = true
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["uv_build>=0.9.26,<0.10.0"]
|
|
3
|
+
build-backend = "uv_build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "apb-proteobench"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "ProteoBench annotation, diagnostics, and scoring for APB2 results"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.13"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE", "THIRD_PARTY_NOTICES.md", "src/apb_proteobench/data/APACHE-2.0.txt"]
|
|
13
|
+
authors = [
|
|
14
|
+
{ name = "Witold Wolski", email = "wew@fgcz.ethz.ch" },
|
|
15
|
+
]
|
|
16
|
+
keywords = ["proteomics", "proteobench", "benchmarking", "anndata", "mass spectrometry"]
|
|
17
|
+
classifiers = [
|
|
18
|
+
"Development Status :: 3 - Alpha",
|
|
19
|
+
"Intended Audience :: Science/Research",
|
|
20
|
+
"Operating System :: OS Independent",
|
|
21
|
+
"Programming Language :: Python :: 3",
|
|
22
|
+
"Programming Language :: Python :: 3.13",
|
|
23
|
+
"Topic :: Scientific/Engineering :: Bio-Informatics",
|
|
24
|
+
"Typing :: Typed",
|
|
25
|
+
]
|
|
26
|
+
dependencies = [
|
|
27
|
+
"apb2>=0.1,<0.2",
|
|
28
|
+
"apb-catalog>=0.1,<0.2",
|
|
29
|
+
"apb-fasta>=0.1,<0.2",
|
|
30
|
+
"cyclopts>=4,<5",
|
|
31
|
+
"loguru>=0.7,<1",
|
|
32
|
+
"numpy>=2,<3",
|
|
33
|
+
"pandas>=2.3,<4",
|
|
34
|
+
"polars>=1.38,<2",
|
|
35
|
+
"pydantic>=2.12,<3",
|
|
36
|
+
"scipy>=1.16,<2",
|
|
37
|
+
]
|
|
38
|
+
|
|
39
|
+
[project.scripts]
|
|
40
|
+
apb-proteobench = "apb_proteobench.cli.app:main"
|
|
41
|
+
|
|
42
|
+
[project.urls]
|
|
43
|
+
Documentation = "https://anndata-omics-bridge.github.io/apb-proteobench/"
|
|
44
|
+
Repository = "https://github.com/anndata-omics-bridge/apb-proteobench"
|
|
45
|
+
Issues = "https://github.com/anndata-omics-bridge/apb-proteobench/issues"
|
|
46
|
+
|
|
47
|
+
[dependency-groups]
|
|
48
|
+
dev = [
|
|
49
|
+
"build>=1.3,<2",
|
|
50
|
+
"deptry>=0.24,<1",
|
|
51
|
+
"import-linter>=2.9,<3",
|
|
52
|
+
"pandas-stubs>=2.3,<3",
|
|
53
|
+
"pre-commit>=4,<5",
|
|
54
|
+
"pyright>=1.1.400,<2",
|
|
55
|
+
"pytest>=9,<10",
|
|
56
|
+
"pytest-cov>=7,<8",
|
|
57
|
+
"ruff>=0.15,<1",
|
|
58
|
+
"sdrf-pipelines>=0.1.6,<0.2",
|
|
59
|
+
"twine>=6,<7",
|
|
60
|
+
]
|
|
61
|
+
docs = [
|
|
62
|
+
"mkdocstrings[python]>=1.0,<2",
|
|
63
|
+
"pymdown-extensions>=11,<12",
|
|
64
|
+
"zensical==0.0.43",
|
|
65
|
+
]
|
|
66
|
+
|
|
67
|
+
[tool.ruff]
|
|
68
|
+
line-length = 100
|
|
69
|
+
target-version = "py313"
|
|
70
|
+
src = ["src", "tests"]
|
|
71
|
+
|
|
72
|
+
[tool.ruff.lint]
|
|
73
|
+
select = [
|
|
74
|
+
"ANN",
|
|
75
|
+
"B",
|
|
76
|
+
"C4",
|
|
77
|
+
"C90",
|
|
78
|
+
"E4",
|
|
79
|
+
"E7",
|
|
80
|
+
"E9",
|
|
81
|
+
"F",
|
|
82
|
+
"I",
|
|
83
|
+
"PGH",
|
|
84
|
+
"PIE",
|
|
85
|
+
"RUF",
|
|
86
|
+
"SIM",
|
|
87
|
+
"UP",
|
|
88
|
+
]
|
|
89
|
+
|
|
90
|
+
[tool.ruff.lint.mccabe]
|
|
91
|
+
max-complexity = 10
|
|
92
|
+
|
|
93
|
+
[tool.ruff.format]
|
|
94
|
+
docstring-code-format = true
|
|
95
|
+
|
|
96
|
+
[tool.pyright]
|
|
97
|
+
include = ["src", "tests"]
|
|
98
|
+
venvPath = "."
|
|
99
|
+
venv = ".venv"
|
|
100
|
+
pythonVersion = "3.13"
|
|
101
|
+
typeCheckingMode = "strict"
|
|
102
|
+
reportImportCycles = "error"
|
|
103
|
+
reportUnnecessaryTypeIgnoreComment = "error"
|
|
104
|
+
reportImplicitOverride = "error"
|
|
105
|
+
enableTypeIgnoreComments = false
|
|
106
|
+
# SciPy and pandas still propagate Unknown through supported public operations.
|
|
107
|
+
# Keep strict first-party checking while suppressing diagnostics with only that signal.
|
|
108
|
+
reportMissingTypeStubs = "none"
|
|
109
|
+
reportUnknownMemberType = "none"
|
|
110
|
+
reportUnknownVariableType = "none"
|
|
111
|
+
reportUnknownArgumentType = "none"
|
|
112
|
+
reportUnknownParameterType = "none"
|
|
113
|
+
reportUnknownLambdaType = "none"
|
|
114
|
+
|
|
115
|
+
[tool.pytest.ini_options]
|
|
116
|
+
addopts = ["--strict-config", "--strict-markers", "-ra"]
|
|
117
|
+
testpaths = ["tests"]
|
|
118
|
+
xfail_strict = true
|
|
119
|
+
|
|
120
|
+
[tool.coverage.run]
|
|
121
|
+
branch = true
|
|
122
|
+
source = ["apb_proteobench"]
|
|
123
|
+
|
|
124
|
+
[tool.coverage.report]
|
|
125
|
+
fail_under = 80
|
|
126
|
+
show_missing = true
|
|
127
|
+
skip_covered = true
|
|
128
|
+
|
|
129
|
+
[tool.deptry]
|
|
130
|
+
known_first_party = ["apb_proteobench"]
|
|
131
|
+
|
|
132
|
+
[tool.uv.sources]
|
|
133
|
+
apb2 = { path = "../apb2", editable = true }
|
|
134
|
+
apb-catalog = { path = "../apb-catalog", editable = true }
|
|
135
|
+
apb-fasta = { path = "../apb-fasta", editable = true }
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
"""Bind one ProteoBench module to the configured APB quantification level."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from copy import deepcopy
|
|
6
|
+
from dataclasses import dataclass, replace
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
import polars as pl
|
|
10
|
+
from apb2.api import AnnotationCompiler, AnnotationError, JsonValue, ParsedLevels
|
|
11
|
+
|
|
12
|
+
from apb_proteobench.configuration.load import LoadedModule, load_module
|
|
13
|
+
|
|
14
|
+
_STORAGE_KEY = "proteobench"
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@dataclass(frozen=True, slots=True)
|
|
18
|
+
class ProteoBenchAnnotation:
|
|
19
|
+
"""A dataset annotated with every module sample, awaiting its module provenance."""
|
|
20
|
+
|
|
21
|
+
annotated: ParsedLevels
|
|
22
|
+
module: LoadedModule
|
|
23
|
+
|
|
24
|
+
def annotate(self) -> ParsedLevels:
|
|
25
|
+
"""Store the normalized module beside APB2's own annotation provenance."""
|
|
26
|
+
metadata = deepcopy(self.annotated.metadata)
|
|
27
|
+
section = metadata.setdefault(_STORAGE_KEY, {})
|
|
28
|
+
if not isinstance(section, dict):
|
|
29
|
+
raise AnnotationError("ProteoBench metadata must be an object")
|
|
30
|
+
provenance = section.setdefault("provenance", {})
|
|
31
|
+
if not isinstance(provenance, dict):
|
|
32
|
+
raise AnnotationError("ProteoBench provenance must be an object")
|
|
33
|
+
record: dict[str, JsonValue] = {**self.module.metadata(), "schema_version": "2"}
|
|
34
|
+
provenance["annotation"] = record
|
|
35
|
+
return replace(self.annotated, metadata=metadata)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass(frozen=True, slots=True)
|
|
39
|
+
class ProteoBenchAnnotationParser:
|
|
40
|
+
"""A validated module source ready to bind to one APB result."""
|
|
41
|
+
|
|
42
|
+
module: LoadedModule
|
|
43
|
+
|
|
44
|
+
@classmethod
|
|
45
|
+
def from_path(cls, path: Path, /) -> ProteoBenchAnnotationParser:
|
|
46
|
+
"""Decode and validate the complete module document once."""
|
|
47
|
+
return cls(load_module(path))
|
|
48
|
+
|
|
49
|
+
def parse(self, parsed: ParsedLevels, /) -> ProteoBenchAnnotation:
|
|
50
|
+
"""Match every run to exactly one module sample and every sample to a run."""
|
|
51
|
+
level_name = self.module.settings.general.level
|
|
52
|
+
if level_name not in parsed.levels:
|
|
53
|
+
raise AnnotationError(
|
|
54
|
+
f"ProteoBench module selects unavailable level {level_name!r}; "
|
|
55
|
+
f"available={list(parsed.levels)}"
|
|
56
|
+
)
|
|
57
|
+
annotation = AnnotationCompiler("error").compile(_sample_frame(self.module)).parse(parsed)
|
|
58
|
+
coverage = annotation.matches.levels[level_name].coverage
|
|
59
|
+
if coverage.annotation_only_count:
|
|
60
|
+
raise AnnotationError(
|
|
61
|
+
"ProteoBench module contains samples absent from quantification; "
|
|
62
|
+
f"count={coverage.annotation_only_count}, "
|
|
63
|
+
f"examples={list(coverage.annotation_only_examples)}"
|
|
64
|
+
)
|
|
65
|
+
return ProteoBenchAnnotation(annotated=annotation.annotate().parsed, module=self.module)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _sample_frame(module: LoadedModule) -> pl.DataFrame:
|
|
69
|
+
"""The module samples as a prolfquapp table keyed by raw file, with its aliases."""
|
|
70
|
+
samples = module.settings.samples
|
|
71
|
+
return pl.DataFrame(
|
|
72
|
+
{
|
|
73
|
+
"raw_file": [sample.raw_file for sample in samples],
|
|
74
|
+
"raw_file_aliases": [
|
|
75
|
+
list(
|
|
76
|
+
dict.fromkeys(
|
|
77
|
+
identifier
|
|
78
|
+
for identifier in (*sample.raw_file_aliases, sample.sample_name)
|
|
79
|
+
if identifier != sample.raw_file
|
|
80
|
+
)
|
|
81
|
+
)
|
|
82
|
+
for sample in samples
|
|
83
|
+
],
|
|
84
|
+
"sample_name": [sample.sample_name for sample in samples],
|
|
85
|
+
"condition": [sample.condition for sample in samples],
|
|
86
|
+
},
|
|
87
|
+
schema={
|
|
88
|
+
"raw_file": pl.String,
|
|
89
|
+
"raw_file_aliases": pl.List(pl.String),
|
|
90
|
+
"sample_name": pl.String,
|
|
91
|
+
"condition": pl.String,
|
|
92
|
+
},
|
|
93
|
+
)
|