apb2 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (145) hide show
  1. apb2-0.1.0/LICENSE +21 -0
  2. apb2-0.1.0/PKG-INFO +185 -0
  3. apb2-0.1.0/README.md +149 -0
  4. apb2-0.1.0/pyproject.toml +157 -0
  5. apb2-0.1.0/pyproject.toml.orig +134 -0
  6. apb2-0.1.0/src/apb2/__init__.py +1 -0
  7. apb2-0.1.0/src/apb2/annotation/__init__.py +0 -0
  8. apb2-0.1.0/src/apb2/annotation/application/__init__.py +0 -0
  9. apb2-0.1.0/src/apb2/annotation/application/policies.py +306 -0
  10. apb2-0.1.0/src/apb2/annotation/compiler.py +76 -0
  11. apb2-0.1.0/src/apb2/annotation/contracts.py +29 -0
  12. apb2-0.1.0/src/apb2/annotation/data/__init__.py +0 -0
  13. apb2-0.1.0/src/apb2/annotation/data/model.py +112 -0
  14. apb2-0.1.0/src/apb2/annotation/matching/__init__.py +0 -0
  15. apb2-0.1.0/src/apb2/annotation/matching/core.py +467 -0
  16. apb2-0.1.0/src/apb2/annotation/prolfquapp.py +125 -0
  17. apb2-0.1.0/src/apb2/annotation/sdrf.py +205 -0
  18. apb2-0.1.0/src/apb2/annotation/source/__init__.py +0 -0
  19. apb2-0.1.0/src/apb2/annotation/source/load.py +68 -0
  20. apb2-0.1.0/src/apb2/api.py +65 -0
  21. apb2-0.1.0/src/apb2/cli/__init__.py +0 -0
  22. apb2-0.1.0/src/apb2/cli/annotation.py +53 -0
  23. apb2-0.1.0/src/apb2/cli/app.py +214 -0
  24. apb2-0.1.0/src/apb2/cli/conversion.py +291 -0
  25. apb2-0.1.0/src/apb2/parserV2/__init__.py +0 -0
  26. apb2-0.1.0/src/apb2/parserV2/compile.py +256 -0
  27. apb2-0.1.0/src/apb2/parserV2/detect_document.py +562 -0
  28. apb2-0.1.0/src/apb2/parserV2/joins/__init__.py +0 -0
  29. apb2-0.1.0/src/apb2/parserV2/joins/alphadia.py +60 -0
  30. apb2-0.1.0/src/apb2/parserV2/joins/maxquant.py +141 -0
  31. apb2-0.1.0/src/apb2/parserV2/parse_quant/__init__.py +0 -0
  32. apb2-0.1.0/src/apb2/parserV2/parse_quant/axis_columns.py +169 -0
  33. apb2-0.1.0/src/apb2/parserV2/parse_quant/contracts.py +138 -0
  34. apb2-0.1.0/src/apb2/parserV2/parse_quant/data/__init__.py +0 -0
  35. apb2-0.1.0/src/apb2/parserV2/parse_quant/data/errors.py +7 -0
  36. apb2-0.1.0/src/apb2/parserV2/parse_quant/data/layer_columns.py +56 -0
  37. apb2-0.1.0/src/apb2/parserV2/parse_quant/data/parsed.py +684 -0
  38. apb2-0.1.0/src/apb2/parserV2/parse_quant/data/raw.py +120 -0
  39. apb2-0.1.0/src/apb2/parserV2/parse_quant/data/source.py +27 -0
  40. apb2-0.1.0/src/apb2/parserV2/parse_quant/decomposition.py +289 -0
  41. apb2-0.1.0/src/apb2/parserV2/parse_quant/delimited_input.py +330 -0
  42. apb2-0.1.0/src/apb2/parserV2/parse_quant/duplicates.py +122 -0
  43. apb2-0.1.0/src/apb2/parserV2/parse_quant/errors.py +37 -0
  44. apb2-0.1.0/src/apb2/parserV2/parse_quant/excel_input.py +104 -0
  45. apb2-0.1.0/src/apb2/parserV2/parse_quant/fragments.py +104 -0
  46. apb2-0.1.0/src/apb2/parserV2/parse_quant/io/__init__.py +0 -0
  47. apb2-0.1.0/src/apb2/parserV2/parse_quant/io/anndata_reader.py +472 -0
  48. apb2-0.1.0/src/apb2/parserV2/parse_quant/io/anndata_writer.py +566 -0
  49. apb2-0.1.0/src/apb2/parserV2/parse_quant/io/duckdb.py +352 -0
  50. apb2-0.1.0/src/apb2/parserV2/parse_quant/io/errors.py +15 -0
  51. apb2-0.1.0/src/apb2/parserV2/parse_quant/io/formats.py +113 -0
  52. apb2-0.1.0/src/apb2/parserV2/parse_quant/io/json_representation.py +335 -0
  53. apb2-0.1.0/src/apb2/parserV2/parse_quant/io/layer_representation.py +243 -0
  54. apb2-0.1.0/src/apb2/parserV2/parse_quant/io/metadata.py +502 -0
  55. apb2-0.1.0/src/apb2/parserV2/parse_quant/io/parquet_reader.py +232 -0
  56. apb2-0.1.0/src/apb2/parserV2/parse_quant/io/parquet_writer.py +207 -0
  57. apb2-0.1.0/src/apb2/parserV2/parse_quant/io/uns_json.py +143 -0
  58. apb2-0.1.0/src/apb2/parserV2/parse_quant/io/validation.py +237 -0
  59. apb2-0.1.0/src/apb2/parserV2/parse_quant/layer_validation.py +64 -0
  60. apb2-0.1.0/src/apb2/parserV2/parse_quant/modifications.py +655 -0
  61. apb2-0.1.0/src/apb2/parserV2/parse_quant/numeric_text.py +66 -0
  62. apb2-0.1.0/src/apb2/parserV2/parse_quant/operations.py +163 -0
  63. apb2-0.1.0/src/apb2/parserV2/parse_quant/parameters/__init__.py +0 -0
  64. apb2-0.1.0/src/apb2/parserV2/parse_quant/parameters/axis.py +69 -0
  65. apb2-0.1.0/src/apb2/parserV2/parse_quant/parameters/level.py +10 -0
  66. apb2-0.1.0/src/apb2/parserV2/parse_quant/parameters/measurements.py +79 -0
  67. apb2-0.1.0/src/apb2/parserV2/parse_quant/parameters/source.py +293 -0
  68. apb2-0.1.0/src/apb2/parserV2/parse_quant/parquet_input.py +44 -0
  69. apb2-0.1.0/src/apb2/parserV2/parse_quant/parser.py +397 -0
  70. apb2-0.1.0/src/apb2/parserV2/parse_quant/plan_json.py +105 -0
  71. apb2-0.1.0/src/apb2/parserV2/parse_quant/prepared_input.py +28 -0
  72. apb2-0.1.0/src/apb2/parserV2/parse_quant/source_resolution.py +628 -0
  73. apb2-0.1.0/src/apb2/parserV2/parse_quant/value_parsing.py +232 -0
  74. apb2-0.1.0/src/apb2/parserV2/parse_rule_facade.py +509 -0
  75. apb2-0.1.0/src/apb2/parserV2/parser_factory.py +55 -0
  76. apb2-0.1.0/src/apb2/parserV2/prepare_source.py +102 -0
  77. apb2-0.1.0/src/apb2/parserV2/source_binding.py +121 -0
  78. apb2-0.1.0/src/apb2/parserV2/vendor_params/__init__.py +0 -0
  79. apb2-0.1.0/src/apb2/parserV2/vendor_params/parsers/__init__.py +0 -0
  80. apb2-0.1.0/src/apb2/parserV2/vendor_params/parsers/alphadia.py +199 -0
  81. apb2-0.1.0/src/apb2/parserV2/vendor_params/parsers/alphapept.py +127 -0
  82. apb2-0.1.0/src/apb2/parserV2/vendor_params/parsers/diann.py +522 -0
  83. apb2-0.1.0/src/apb2/parserV2/vendor_params/parsers/fragpipe.py +413 -0
  84. apb2-0.1.0/src/apb2/parserV2/vendor_params/parsers/i2masschroq.py +149 -0
  85. apb2-0.1.0/src/apb2/parserV2/vendor_params/parsers/maxquant.py +293 -0
  86. apb2-0.1.0/src/apb2/parserV2/vendor_params/parsers/metamorpheus.py +202 -0
  87. apb2-0.1.0/src/apb2/parserV2/vendor_params/parsers/msaid.py +66 -0
  88. apb2-0.1.0/src/apb2/parserV2/vendor_params/parsers/msangel.py +103 -0
  89. apb2-0.1.0/src/apb2/parserV2/vendor_params/parsers/peaks.py +187 -0
  90. apb2-0.1.0/src/apb2/parserV2/vendor_params/parsers/prolinestudio.py +120 -0
  91. apb2-0.1.0/src/apb2/parserV2/vendor_params/parsers/quantms.py +43 -0
  92. apb2-0.1.0/src/apb2/parserV2/vendor_params/parsers/sage.py +106 -0
  93. apb2-0.1.0/src/apb2/parserV2/vendor_params/parsers/shared/__init__.py +0 -0
  94. apb2-0.1.0/src/apb2/parserV2/vendor_params/parsers/shared/common.py +232 -0
  95. apb2-0.1.0/src/apb2/parserV2/vendor_params/parsers/shared/model.py +209 -0
  96. apb2-0.1.0/src/apb2/parserV2/vendor_params/parsers/shared/unimod.py +166 -0
  97. apb2-0.1.0/src/apb2/parserV2/vendor_params/parsers/shared/unimod_registry.json +55 -0
  98. apb2-0.1.0/src/apb2/parserV2/vendor_params/parsers/spectronaut.py +211 -0
  99. apb2-0.1.0/src/apb2/parserV2/vendor_params/parsers/wombat.py +110 -0
  100. apb2-0.1.0/src/apb2/parserV2/vendor_params/registry.py +133 -0
  101. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/__init__.py +0 -0
  102. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/catalog.json +25 -0
  103. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/catalog.py +179 -0
  104. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/document.py +319 -0
  105. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/documents/__init__.py +0 -0
  106. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/documents/_schema/document.schema.json +490 -0
  107. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/documents/_schema/rule.schema.json +1449 -0
  108. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/documents/alphadia/v1_10/rules.json +145 -0
  109. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/documents/alphadia/v1_12/rules.json +190 -0
  110. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/documents/alphadia/v2/rules.json +238 -0
  111. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/documents/alphapept/rules.json +217 -0
  112. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/documents/diann/v1_7/rules.json +365 -0
  113. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/documents/diann/v1_8/rules.json +391 -0
  114. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/documents/diann/v2/rules.json +292 -0
  115. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/documents/fragpipe/rules.json +182 -0
  116. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/documents/i2masschroq/rules.json +116 -0
  117. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/documents/maxquant/rules.json +601 -0
  118. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/documents/msangel/rules.json +177 -0
  119. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/documents/pb_custom/rules.json +70 -0
  120. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/documents/peaks/rules.json +191 -0
  121. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/documents/prolinestudio/rules.json +177 -0
  122. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/documents/quantms/rules.json +153 -0
  123. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/documents/sage/rules.json +159 -0
  124. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/documents/spectronaut/rules.json +628 -0
  125. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/documents/spectronaut/v15/rules.json +535 -0
  126. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/documents/spectronaut/v21/rules.json +636 -0
  127. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/documents/wombat/rules.json +156 -0
  128. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/loader.py +40 -0
  129. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/schema/__init__.py +0 -0
  130. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/schema/annotation.py +40 -0
  131. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/schema/axis.py +128 -0
  132. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/schema/base.py +31 -0
  133. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/schema/base_formats.py +62 -0
  134. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/schema/base_modifications.py +74 -0
  135. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/schema/fragments.py +42 -0
  136. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/schema/hierarchies.json +5 -0
  137. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/schema/hierarchy.py +11 -0
  138. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/schema/input.py +50 -0
  139. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/schema/measurements.py +127 -0
  140. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/schema/parameters.py +19 -0
  141. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/schema/role_policy.json +10 -0
  142. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/schema/roles.py +20 -0
  143. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/schema/rule.py +295 -0
  144. apb2-0.1.0/src/apb2/parserV2/vendor_parse_rules/schema_artifact.py +31 -0
  145. apb2-0.1.0/src/apb2/py.typed +1 -0
apb2-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Witold Wolski
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
apb2-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,185 @@
1
+ Metadata-Version: 2.4
2
+ Name: apb2
3
+ Version: 0.1.0
4
+ Summary: Convert proteomics software output to AnnData (rules-driven parser, second generation)
5
+ Keywords: proteomics,mass spectrometry,anndata,mudata,quantification
6
+ Author: Witold Wolski
7
+ Author-email: Witold Wolski <wew@fgcz.ethz.ch>
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Classifier: Development Status :: 3 - Alpha
11
+ Classifier: Intended Audience :: Science/Research
12
+ Classifier: Operating System :: OS Independent
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Programming Language :: Python :: 3.13
15
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
16
+ Classifier: Typing :: Typed
17
+ Requires-Dist: anndata>=0.11
18
+ Requires-Dist: cyclopts>=3
19
+ Requires-Dist: duckdb>=1.4,<2
20
+ Requires-Dist: loguru>=0.7
21
+ Requires-Dist: mudata>=0.4,<1
22
+ Requires-Dist: numpy>=2
23
+ Requires-Dist: packaging>=24
24
+ Requires-Dist: pandas>=2.2
25
+ Requires-Dist: polars>=1.43,<2
26
+ Requires-Dist: fastexcel>=0.21,<1
27
+ Requires-Dist: python-calamine>=0.5,<1
28
+ Requires-Dist: pydantic>=2.10
29
+ Requires-Dist: pyarrow>=15
30
+ Requires-Dist: pyyaml>=6
31
+ Requires-Dist: scipy>=1.15,<2
32
+ Requires-Python: >=3.13
33
+ Project-URL: Documentation, https://anndata-omics-bridge.github.io/apb2/
34
+ Project-URL: Repository, https://github.com/anndata-omics-bridge/apb2
35
+ Description-Content-Type: text/markdown
36
+
37
+ # apb2
38
+
39
+ APB2 is a [rules-driven framework](https://anndata-omics-bridge.github.io/apb2/rule-based/) for converting outputs from proteomics
40
+ software into AnnData or MuData. It supports ion, peptidoform, peptide, protein, and fragment
41
+ quantification levels and can also store the parsed data in Parquet or DuckDB.
42
+
43
+ “Rules-driven” means that declarative rule documents describe each vendor table: which columns
44
+ contain identifiers, measurements, and metadata, how those columns should be reshaped, and which
45
+ constraints the result must satisfy. One shared parser applies those rules, so a new or revised
46
+ input format can usually be supported by adding or updating a rule instead of writing a dedicated
47
+ reader.
48
+
49
+ Read the rendered [APB2 documentation](https://anndata-omics-bridge.github.io/apb2/) or its
50
+ [source index](https://github.com/anndata-omics-bridge/apb2/blob/main/docs/index.md). The [supported-software matrix](https://anndata-omics-bridge.github.io/apb2/supported_software/) lists
51
+ every packaged software version, quantification level, vendor input type, table shape, and
52
+ parameter parser.
53
+
54
+ Choose the documentation for your interface:
55
+
56
+ - [CLI reference](https://anndata-omics-bridge.github.io/apb2/cli/) and [command-line guides](https://anndata-omics-bridge.github.io/apb2/conversion/)
57
+ - [Python API reference](https://anndata-omics-bridge.github.io/apb2/api/)
58
+
59
+ ## Installation
60
+
61
+ APB2 requires Python 3.13 or later.
62
+
63
+ ```bash
64
+ pip install apb2
65
+ ```
66
+
67
+ To install only the `apb2` command, use `uv tool install apb2`.
68
+
69
+ ## Motivation and origin
70
+
71
+ APB2 is a refactoring and performance improvement of the now-discontinued [AnnData Proteomics Bridge (APB v1)](https://github.com/anndata-omics-bridge/anndata-proteomics-bridge).
72
+
73
+ APB2 is based on the work of [ProteoBench](https://github.com/proteobench/proteobench): it ports ProteoBench's parsing infrastructure — search-parameter parsing, vendor file-format parsing, modification parsing — into one rules-driven converter. See the ProteoBench preprint: [ProteoBench: the community-curated platform for comparing proteomics data analysis workflows](https://www.biorxiv.org/content/10.64898/2025.12.09.692895v2) (bioRxiv, 2025, doi:10.64898/2025.12.09.692895).
74
+
75
+ Packaged conversion rules include AlphaDIA, AlphaPept, DIA-NN, FragPipe, i2MassChroQ, MaxQuant, MSAngel, PEAKS, ProteoBench Custom, ProlineStudio, quantms, Sage, Spectronaut, and WOMBAT; the complete version and input-format matrix is in [supported software](https://anndata-omics-bridge.github.io/apb2/supported_software/).
76
+
77
+ The work that became APB2 was discussed and started during the Copenhagen ProteoBench Hackathon,
78
+ 13–17 April 2026, as one of the efforts to improve the backend of the
79
+ [ProteoBench platform](https://proteobench.cubimed.rub.de/). The hackathon included the public
80
+ [EuBIC-MS Seminar 2026 on 15 April](https://eubic-ms.org/events/latest-developments-and-tools-for-data-analysis/).
81
+
82
+ APB2 was also motivated by the vendor-specific readers maintained behind
83
+ [`prolfquapp::preprocess_software()`](https://github.com/prolfqua/prolfquapp/blob/master/R/preprocess_software.R#L137)
84
+ and in
85
+ [`prolfquappPTMreaders`](https://github.com/prolfqua/prolfquappPTMreaders). We plan to move their
86
+ remaining input variants and PTM/site-level formats into APB2 so one rules-driven parser can serve
87
+ both prolfquapp and ProteoBench, and hopefully other tools analysing quantification data.
88
+
89
+ ## Command-line interface
90
+
91
+ ### Convert
92
+
93
+ Use a packaged rule selected from the vendor parameter file and source header. `DATA` may be one vendor table or a vendor-result directory:
94
+
95
+ `--software` selects the parameter-file grammar and restricts result recognition to that vendor and its declared quantification software. For FragPipe parameters with DIA-NN output, pass `--software fragpipe`. Omit the hint for recognition across parameter-bearing rules; mismatches and ambiguity are errors, not a fallback to unrelated rules.
96
+
97
+ ProteoBench Custom uploads have no parameter file: `apb2 convert custom.txt ion --software pb_custom --output results/custom`.
98
+
99
+ ```bash
100
+ apb2 convert DATA LEVEL --params PARAMETER_FILE [--software VENDOR] [--output BASENAME]
101
+ ```
102
+
103
+ Omit `LEVEL` to convert every compatible level into one APB2 result:
104
+
105
+ ```bash
106
+ apb2 convert DATA --params PARAMETER_FILE [--software VENDOR] [--format FORMAT] [--output BASENAME]
107
+ ```
108
+
109
+ Use an explicit schema-0.8 rule document, with optional search-parameter evidence:
110
+
111
+ ```bash
112
+ apb2 convert DATA LEVEL --rule-config RULES_JSON [--params PARAMETER_FILE] \
113
+ [--software VENDOR] [--output BASENAME]
114
+ ```
115
+
116
+ `LEVEL` is one of `ion`, `peptidoform`, `peptide`, `protein`, or `fragment`. MaxQuant accepts any nonempty subset of evidence, modification-specific peptide, peptide and protein-group exports. Evidence stays separate from the higher-level join; an omitted level converts every available level. `--format` selects `hdf5`, `parquet`, or `duckdb`. HDF5 uses `.h5ad` with an explicit level and `.h5mu` otherwise. Complete one-to-one observation aliases are aligned; fractionated or unmapped resolutions produce separate outputs such as `result.raw_file.h5mu` and `result.experiment.h5mu`. See [output naming](https://anndata-omics-bridge.github.io/apb2/conversion/#output-naming). `--strict` promotes layer-contract warnings to errors. `--timings-output PATH` optionally writes a separate versioned JSON file containing internal compile, read, parse and write durations plus per-level read/parse durations; it does not enter the APB result or its scientific representation. The command performs conversion only; FASTA annotation and protein inference are outside Parser V2.
117
+
118
+ ### Reformat a parsed result
119
+
120
+ Change only the persisted format; no vendor parsing or annotation runs:
121
+
122
+ ```bash
123
+ apb2 reformat SOURCE TARGET
124
+ ```
125
+
126
+ The suffix selects h5ad, h5mu, an APB2 Parquet directory dataset, or DuckDB.
127
+
128
+ ### Annotate samples
129
+
130
+ Attach a generic prolfquapp-style CSV/TSV table to any APB2 result format:
131
+
132
+ ```bash
133
+ apb2 annotate INPUT ANNOTATION OUTPUT
134
+ ```
135
+
136
+ The default prolfquapp behavior retains unmatched quantitative observations and writes null
137
+ annotation fields. `--unmatched error` requires complete coverage; `--unmatched drop` explicitly
138
+ subsets every observation-aligned value. ProteoBench-specific module annotation and scoring live
139
+ in the separate `apb-proteobench` package. See the
140
+ [sample-annotation guide](https://anndata-omics-bridge.github.io/apb2/sample_annotation/).
141
+
142
+ ## Python API
143
+
144
+ The file-to-file facades mirror the CLI operations. The compiler/parser APIs expose
145
+ storage-neutral values for custom pipelines. Result formats also have explicit adapters:
146
+
147
+ ```python
148
+ from pathlib import Path
149
+
150
+ from apb2.api import read_parsed_levels, write_parsed_levels
151
+
152
+ parsed = read_parsed_levels(Path("result.parquet"))
153
+ write_parsed_levels(parsed, Path("result.duckdb"))
154
+ ```
155
+
156
+ Parquet and DuckDB preserve Polars result values exactly; h5ad and h5mu apply the stored
157
+ numeric/factor matrix projection. Every public result write also publishes an adjacent compact `.apb.json` scientific representation for inspection without loading the full result. See the [Python API reference](https://anndata-omics-bridge.github.io/apb2/api/) for vendor
158
+ conversion, annotation, result values, and errors.
159
+
160
+ ## Architecture
161
+
162
+ The CLI delegates conversion to Parser V2 and annotation to the independent annotation facade.
163
+ The controlling designs and dependency boundaries are documented in
164
+ [`docs/architecture_converter.md`](https://anndata-omics-bridge.github.io/apb2/architecture_converter/) and
165
+ [`docs/architecture_annotation.md`](https://anndata-omics-bridge.github.io/apb2/architecture_annotation/).
166
+
167
+ ## Development
168
+
169
+ ```bash
170
+ uv sync --group dev
171
+ make check
172
+ make docs
173
+ .venv/bin/pre-commit install --hook-type pre-commit --hook-type pre-push
174
+ ```
175
+
176
+ All Python commands run from the synchronized project `.venv`.
177
+ `make docs-serve` serves the user documentation locally. GitHub Actions publishes the strict
178
+ Zensical build to GitHub Pages from `main`.
179
+
180
+ The rule JSON Schema is a packaged artifact. Developers regenerate it from the Parser V2 rule
181
+ package rather than through a user-facing CLI command:
182
+
183
+ ```bash
184
+ uv run python -c 'from apb2.parserV2.vendor_parse_rules.schema_artifact import write_artifact; write_artifact()'
185
+ ```
apb2-0.1.0/README.md ADDED
@@ -0,0 +1,149 @@
1
+ # apb2
2
+
3
+ APB2 is a [rules-driven framework](https://anndata-omics-bridge.github.io/apb2/rule-based/) for converting outputs from proteomics
4
+ software into AnnData or MuData. It supports ion, peptidoform, peptide, protein, and fragment
5
+ quantification levels and can also store the parsed data in Parquet or DuckDB.
6
+
7
+ “Rules-driven” means that declarative rule documents describe each vendor table: which columns
8
+ contain identifiers, measurements, and metadata, how those columns should be reshaped, and which
9
+ constraints the result must satisfy. One shared parser applies those rules, so a new or revised
10
+ input format can usually be supported by adding or updating a rule instead of writing a dedicated
11
+ reader.
12
+
13
+ Read the rendered [APB2 documentation](https://anndata-omics-bridge.github.io/apb2/) or its
14
+ [source index](https://github.com/anndata-omics-bridge/apb2/blob/main/docs/index.md). The [supported-software matrix](https://anndata-omics-bridge.github.io/apb2/supported_software/) lists
15
+ every packaged software version, quantification level, vendor input type, table shape, and
16
+ parameter parser.
17
+
18
+ Choose the documentation for your interface:
19
+
20
+ - [CLI reference](https://anndata-omics-bridge.github.io/apb2/cli/) and [command-line guides](https://anndata-omics-bridge.github.io/apb2/conversion/)
21
+ - [Python API reference](https://anndata-omics-bridge.github.io/apb2/api/)
22
+
23
+ ## Installation
24
+
25
+ APB2 requires Python 3.13 or later.
26
+
27
+ ```bash
28
+ pip install apb2
29
+ ```
30
+
31
+ To install only the `apb2` command, use `uv tool install apb2`.
32
+
33
+ ## Motivation and origin
34
+
35
+ APB2 is a refactoring and performance improvement of the now-discontinued [AnnData Proteomics Bridge (APB v1)](https://github.com/anndata-omics-bridge/anndata-proteomics-bridge).
36
+
37
+ APB2 is based on the work of [ProteoBench](https://github.com/proteobench/proteobench): it ports ProteoBench's parsing infrastructure — search-parameter parsing, vendor file-format parsing, modification parsing — into one rules-driven converter. See the ProteoBench preprint: [ProteoBench: the community-curated platform for comparing proteomics data analysis workflows](https://www.biorxiv.org/content/10.64898/2025.12.09.692895v2) (bioRxiv, 2025, doi:10.64898/2025.12.09.692895).
38
+
39
+ Packaged conversion rules include AlphaDIA, AlphaPept, DIA-NN, FragPipe, i2MassChroQ, MaxQuant, MSAngel, PEAKS, ProteoBench Custom, ProlineStudio, quantms, Sage, Spectronaut, and WOMBAT; the complete version and input-format matrix is in [supported software](https://anndata-omics-bridge.github.io/apb2/supported_software/).
40
+
41
+ The work that became APB2 was discussed and started during the Copenhagen ProteoBench Hackathon,
42
+ 13–17 April 2026, as one of the efforts to improve the backend of the
43
+ [ProteoBench platform](https://proteobench.cubimed.rub.de/). The hackathon included the public
44
+ [EuBIC-MS Seminar 2026 on 15 April](https://eubic-ms.org/events/latest-developments-and-tools-for-data-analysis/).
45
+
46
+ APB2 was also motivated by the vendor-specific readers maintained behind
47
+ [`prolfquapp::preprocess_software()`](https://github.com/prolfqua/prolfquapp/blob/master/R/preprocess_software.R#L137)
48
+ and in
49
+ [`prolfquappPTMreaders`](https://github.com/prolfqua/prolfquappPTMreaders). We plan to move their
50
+ remaining input variants and PTM/site-level formats into APB2 so one rules-driven parser can serve
51
+ both prolfquapp and ProteoBench, and hopefully other tools analysing quantification data.
52
+
53
+ ## Command-line interface
54
+
55
+ ### Convert
56
+
57
+ Use a packaged rule selected from the vendor parameter file and source header. `DATA` may be one vendor table or a vendor-result directory:
58
+
59
+ `--software` selects the parameter-file grammar and restricts result recognition to that vendor and its declared quantification software. For FragPipe parameters with DIA-NN output, pass `--software fragpipe`. Omit the hint for recognition across parameter-bearing rules; mismatches and ambiguity are errors, not a fallback to unrelated rules.
60
+
61
+ ProteoBench Custom uploads have no parameter file: `apb2 convert custom.txt ion --software pb_custom --output results/custom`.
62
+
63
+ ```bash
64
+ apb2 convert DATA LEVEL --params PARAMETER_FILE [--software VENDOR] [--output BASENAME]
65
+ ```
66
+
67
+ Omit `LEVEL` to convert every compatible level into one APB2 result:
68
+
69
+ ```bash
70
+ apb2 convert DATA --params PARAMETER_FILE [--software VENDOR] [--format FORMAT] [--output BASENAME]
71
+ ```
72
+
73
+ Use an explicit schema-0.8 rule document, with optional search-parameter evidence:
74
+
75
+ ```bash
76
+ apb2 convert DATA LEVEL --rule-config RULES_JSON [--params PARAMETER_FILE] \
77
+ [--software VENDOR] [--output BASENAME]
78
+ ```
79
+
80
+ `LEVEL` is one of `ion`, `peptidoform`, `peptide`, `protein`, or `fragment`. MaxQuant accepts any nonempty subset of evidence, modification-specific peptide, peptide and protein-group exports. Evidence stays separate from the higher-level join; an omitted level converts every available level. `--format` selects `hdf5`, `parquet`, or `duckdb`. HDF5 uses `.h5ad` with an explicit level and `.h5mu` otherwise. Complete one-to-one observation aliases are aligned; fractionated or unmapped resolutions produce separate outputs such as `result.raw_file.h5mu` and `result.experiment.h5mu`. See [output naming](https://anndata-omics-bridge.github.io/apb2/conversion/#output-naming). `--strict` promotes layer-contract warnings to errors. `--timings-output PATH` optionally writes a separate versioned JSON file containing internal compile, read, parse and write durations plus per-level read/parse durations; it does not enter the APB result or its scientific representation. The command performs conversion only; FASTA annotation and protein inference are outside Parser V2.
81
+
82
+ ### Reformat a parsed result
83
+
84
+ Change only the persisted format; no vendor parsing or annotation runs:
85
+
86
+ ```bash
87
+ apb2 reformat SOURCE TARGET
88
+ ```
89
+
90
+ The suffix selects h5ad, h5mu, an APB2 Parquet directory dataset, or DuckDB.
91
+
92
+ ### Annotate samples
93
+
94
+ Attach a generic prolfquapp-style CSV/TSV table to any APB2 result format:
95
+
96
+ ```bash
97
+ apb2 annotate INPUT ANNOTATION OUTPUT
98
+ ```
99
+
100
+ The default prolfquapp behavior retains unmatched quantitative observations and writes null
101
+ annotation fields. `--unmatched error` requires complete coverage; `--unmatched drop` explicitly
102
+ subsets every observation-aligned value. ProteoBench-specific module annotation and scoring live
103
+ in the separate `apb-proteobench` package. See the
104
+ [sample-annotation guide](https://anndata-omics-bridge.github.io/apb2/sample_annotation/).
105
+
106
+ ## Python API
107
+
108
+ The file-to-file facades mirror the CLI operations. The compiler/parser APIs expose
109
+ storage-neutral values for custom pipelines. Result formats also have explicit adapters:
110
+
111
+ ```python
112
+ from pathlib import Path
113
+
114
+ from apb2.api import read_parsed_levels, write_parsed_levels
115
+
116
+ parsed = read_parsed_levels(Path("result.parquet"))
117
+ write_parsed_levels(parsed, Path("result.duckdb"))
118
+ ```
119
+
120
+ Parquet and DuckDB preserve Polars result values exactly; h5ad and h5mu apply the stored
121
+ numeric/factor matrix projection. Every public result write also publishes an adjacent compact `.apb.json` scientific representation for inspection without loading the full result. See the [Python API reference](https://anndata-omics-bridge.github.io/apb2/api/) for vendor
122
+ conversion, annotation, result values, and errors.
123
+
124
+ ## Architecture
125
+
126
+ The CLI delegates conversion to Parser V2 and annotation to the independent annotation facade.
127
+ The controlling designs and dependency boundaries are documented in
128
+ [`docs/architecture_converter.md`](https://anndata-omics-bridge.github.io/apb2/architecture_converter/) and
129
+ [`docs/architecture_annotation.md`](https://anndata-omics-bridge.github.io/apb2/architecture_annotation/).
130
+
131
+ ## Development
132
+
133
+ ```bash
134
+ uv sync --group dev
135
+ make check
136
+ make docs
137
+ .venv/bin/pre-commit install --hook-type pre-commit --hook-type pre-push
138
+ ```
139
+
140
+ All Python commands run from the synchronized project `.venv`.
141
+ `make docs-serve` serves the user documentation locally. GitHub Actions publishes the strict
142
+ Zensical build to GitHub Pages from `main`.
143
+
144
+ The rule JSON Schema is a packaged artifact. Developers regenerate it from the Parser V2 rule
145
+ package rather than through a user-facing CLI command:
146
+
147
+ ```bash
148
+ uv run python -c 'from apb2.parserV2.vendor_parse_rules.schema_artifact import write_artifact; write_artifact()'
149
+ ```
@@ -0,0 +1,157 @@
1
+ [build-system]
2
+ requires = ["uv_build>=0.9.26,<0.10.0"]
3
+ build-backend = "uv_build"
4
+
5
+ [project]
6
+ name = "apb2"
7
+ version = "0.1.0"
8
+ description = "Convert proteomics software output to AnnData (rules-driven parser, second generation)"
9
+ readme = "README.md"
10
+ requires-python = ">=3.13"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ keywords = [
14
+ "proteomics",
15
+ "mass spectrometry",
16
+ "anndata",
17
+ "mudata",
18
+ "quantification",
19
+ ]
20
+ classifiers = [
21
+ "Development Status :: 3 - Alpha",
22
+ "Intended Audience :: Science/Research",
23
+ "Operating System :: OS Independent",
24
+ "Programming Language :: Python :: 3",
25
+ "Programming Language :: Python :: 3.13",
26
+ "Topic :: Scientific/Engineering :: Bio-Informatics",
27
+ "Typing :: Typed",
28
+ ]
29
+ dependencies = [
30
+ "anndata>=0.11",
31
+ "cyclopts>=3",
32
+ "duckdb>=1.4,<2",
33
+ "loguru>=0.7",
34
+ "mudata>=0.4,<1",
35
+ "numpy>=2",
36
+ "packaging>=24",
37
+ "pandas>=2.2",
38
+ "polars>=1.43,<2",
39
+ "fastexcel>=0.21,<1",
40
+ "python-calamine>=0.5,<1",
41
+ "pydantic>=2.10",
42
+ "pyarrow>=15",
43
+ "pyyaml>=6",
44
+ "scipy>=1.15,<2",
45
+ ]
46
+
47
+ [[project.authors]]
48
+ name = "Witold Wolski"
49
+ email = "wew@fgcz.ethz.ch"
50
+
51
+ [project.urls]
52
+ Documentation = "https://anndata-omics-bridge.github.io/apb2/"
53
+ Repository = "https://github.com/anndata-omics-bridge/apb2"
54
+
55
+ [project.scripts]
56
+ apb2 = "apb2.cli.app:main"
57
+
58
+ [dependency-groups]
59
+ dev = [
60
+ "build>=1.3,<2",
61
+ "deptry>=0.24,<1",
62
+ "grimp>=3,<4",
63
+ "import-linter>=2,<3",
64
+ "pandas-stubs>=3.0.5.260730",
65
+ "pre-commit>=4,<5",
66
+ "pyright>=1.1.400,<2",
67
+ "pytest>=9,<10",
68
+ "pytest-cov>=7,<8",
69
+ "ruff>=0.15,<1",
70
+ "twine>=6,<7",
71
+ ]
72
+ docs = [
73
+ "pymdown-extensions>=11,<12",
74
+ "zensical==0.0.43",
75
+ ]
76
+
77
+ [tool.ruff]
78
+ line-length = 100
79
+ target-version = "py313"
80
+ src = [
81
+ "src",
82
+ "tests",
83
+ ]
84
+
85
+ [tool.ruff.lint]
86
+ select = [
87
+ "ANN",
88
+ "B",
89
+ "C4",
90
+ "C90",
91
+ "E4",
92
+ "E7",
93
+ "E9",
94
+ "F",
95
+ "I",
96
+ "PGH",
97
+ "PIE",
98
+ "RUF",
99
+ "SIM",
100
+ "UP",
101
+ ]
102
+
103
+ [tool.ruff.lint.mccabe]
104
+ max-complexity = 10
105
+
106
+ [tool.ruff.format]
107
+ docstring-code-format = true
108
+
109
+ [tool.pyright]
110
+ include = [
111
+ "src",
112
+ "tests",
113
+ "scripts",
114
+ ]
115
+ venvPath = "."
116
+ venv = ".venv"
117
+ pythonVersion = "3.13"
118
+ typeCheckingMode = "strict"
119
+ reportImportCycles = "error"
120
+ reportMissingTypeStubs = "none"
121
+ reportUnnecessaryTypeIgnoreComment = "error"
122
+ reportImplicitOverride = "error"
123
+ enableTypeIgnoreComments = false
124
+ reportUnknownMemberType = "none"
125
+ reportUnknownVariableType = "none"
126
+ reportUnknownArgumentType = "none"
127
+ reportUnknownParameterType = "none"
128
+ reportUnknownLambdaType = "none"
129
+
130
+ [tool.pytest.ini_options]
131
+ addopts = [
132
+ "--strict-config",
133
+ "--strict-markers",
134
+ "-ra",
135
+ ]
136
+ testpaths = ["tests"]
137
+ xfail_strict = true
138
+
139
+ [tool.deptry]
140
+ known_first_party = ["apb2"]
141
+ exclude = [
142
+ "documentation",
143
+ "scripts",
144
+ "tests",
145
+ "venv",
146
+ "[.]venv",
147
+ "[.]direnv",
148
+ "[.]git",
149
+ "[.]claude",
150
+ "setup[.]py",
151
+ ]
152
+
153
+ [tool.deptry.per_rule_ignores]
154
+ DEP002 = [
155
+ "pyarrow",
156
+ "python-calamine",
157
+ ]
@@ -0,0 +1,134 @@
1
+ [build-system]
2
+ requires = ["uv_build>=0.9.26,<0.10.0"]
3
+ build-backend = "uv_build"
4
+
5
+ [project]
6
+ name = "apb2"
7
+ version = "0.1.0"
8
+ description = "Convert proteomics software output to AnnData (rules-driven parser, second generation)"
9
+ readme = "README.md"
10
+ requires-python = ">=3.13"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ authors = [
14
+ { name = "Witold Wolski", email = "wew@fgcz.ethz.ch" },
15
+ ]
16
+ keywords = ["proteomics", "mass spectrometry", "anndata", "mudata", "quantification"]
17
+ classifiers = [
18
+ "Development Status :: 3 - Alpha",
19
+ "Intended Audience :: Science/Research",
20
+ "Operating System :: OS Independent",
21
+ "Programming Language :: Python :: 3",
22
+ "Programming Language :: Python :: 3.13",
23
+ "Topic :: Scientific/Engineering :: Bio-Informatics",
24
+ "Typing :: Typed",
25
+ ]
26
+ dependencies = [
27
+ "anndata>=0.11",
28
+ "cyclopts>=3",
29
+ "duckdb>=1.4,<2",
30
+ "loguru>=0.7",
31
+ "mudata>=0.4,<1",
32
+ "numpy>=2",
33
+ "packaging>=24",
34
+ "pandas>=2.2",
35
+ "polars>=1.43,<2",
36
+ "fastexcel>=0.21,<1",
37
+ "python-calamine>=0.5,<1",
38
+ "pydantic>=2.10",
39
+ "pyarrow>=15",
40
+ "pyyaml>=6",
41
+ "scipy>=1.15,<2",
42
+ ]
43
+
44
+ [project.urls]
45
+ Documentation = "https://anndata-omics-bridge.github.io/apb2/"
46
+ Repository = "https://github.com/anndata-omics-bridge/apb2"
47
+
48
+ [project.scripts]
49
+ apb2 = "apb2.cli.app:main"
50
+
51
+ [dependency-groups]
52
+ dev = [
53
+ "build>=1.3,<2",
54
+ "deptry>=0.24,<1",
55
+ "grimp>=3,<4",
56
+ "import-linter>=2,<3",
57
+ "pandas-stubs>=3.0.5.260730",
58
+ "pre-commit>=4,<5",
59
+ "pyright>=1.1.400,<2",
60
+ "pytest>=9,<10",
61
+ "pytest-cov>=7,<8",
62
+ "ruff>=0.15,<1",
63
+ "twine>=6,<7",
64
+ ]
65
+ docs = [
66
+ "pymdown-extensions>=11,<12",
67
+ "zensical==0.0.43",
68
+ ]
69
+ [tool.ruff]
70
+ line-length = 100
71
+ target-version = "py313"
72
+ src = ["src", "tests"]
73
+
74
+ [tool.ruff.lint]
75
+ select = [
76
+ "ANN",
77
+ "B",
78
+ "C4",
79
+ "C90",
80
+ "E4",
81
+ "E7",
82
+ "E9",
83
+ "F",
84
+ "I",
85
+ "PGH",
86
+ "PIE",
87
+ "RUF",
88
+ "SIM",
89
+ "UP",
90
+ ]
91
+
92
+ [tool.ruff.lint.mccabe]
93
+ max-complexity = 10
94
+
95
+ [tool.ruff.format]
96
+ docstring-code-format = true
97
+
98
+ [tool.pyright]
99
+ include = ["src", "tests", "scripts"]
100
+ venvPath = "."
101
+ venv = ".venv"
102
+ pythonVersion = "3.13"
103
+ typeCheckingMode = "strict"
104
+ reportImportCycles = "error"
105
+ reportMissingTypeStubs = "none"
106
+ reportUnnecessaryTypeIgnoreComment = "error"
107
+ reportImplicitOverride = "error"
108
+ enableTypeIgnoreComments = false
109
+ # Third-party scientific libraries expose incomplete types even with their
110
+ # canonical stub packages (same rationale and settings as apb). Keep strict
111
+ # first-party checks while suppressing diagnostics whose only signal is a
112
+ # propagated Unknown.
113
+ reportUnknownMemberType = "none"
114
+ reportUnknownVariableType = "none"
115
+ reportUnknownArgumentType = "none"
116
+ reportUnknownParameterType = "none"
117
+ reportUnknownLambdaType = "none"
118
+
119
+ [tool.pytest.ini_options]
120
+ addopts = ["--strict-config", "--strict-markers", "-ra"]
121
+ testpaths = ["tests"]
122
+ xfail_strict = true
123
+
124
+ [tool.deptry]
125
+ known_first_party = ["apb2"]
126
+ # documentation/ and scripts/ hold executable tooling, not importable package code; .claude/
127
+ # holds agent worktrees, each a full checkout.
128
+ exclude = ["documentation", "scripts", "tests", "venv", "[.]venv", "[.]direnv", "[.]git", "[.]claude", "setup[.]py"]
129
+
130
+ [tool.deptry.per_rule_ignores]
131
+ # DuckDB asks Polars for Arrow record batches when a DataFrame is registered. Polars imports
132
+ # PyArrow dynamically at that boundary, so no direct APB2 import exists for Deptry to observe.
133
+ # The parameter-workbook reader still loads python-calamine through Pandas' Excel engine.
134
+ DEP002 = ["pyarrow", "python-calamine"]
@@ -0,0 +1 @@
1
+ """Rules-driven proteomics vendor-table conversion."""
File without changes
File without changes