apb2 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (145) hide show
  1. apb2/__init__.py +1 -0
  2. apb2/annotation/__init__.py +0 -0
  3. apb2/annotation/application/__init__.py +0 -0
  4. apb2/annotation/application/policies.py +306 -0
  5. apb2/annotation/compiler.py +76 -0
  6. apb2/annotation/contracts.py +29 -0
  7. apb2/annotation/data/__init__.py +0 -0
  8. apb2/annotation/data/model.py +112 -0
  9. apb2/annotation/matching/__init__.py +0 -0
  10. apb2/annotation/matching/core.py +467 -0
  11. apb2/annotation/prolfquapp.py +125 -0
  12. apb2/annotation/sdrf.py +205 -0
  13. apb2/annotation/source/__init__.py +0 -0
  14. apb2/annotation/source/load.py +68 -0
  15. apb2/api.py +65 -0
  16. apb2/cli/__init__.py +0 -0
  17. apb2/cli/annotation.py +53 -0
  18. apb2/cli/app.py +214 -0
  19. apb2/cli/conversion.py +291 -0
  20. apb2/parserV2/__init__.py +0 -0
  21. apb2/parserV2/compile.py +256 -0
  22. apb2/parserV2/detect_document.py +562 -0
  23. apb2/parserV2/joins/__init__.py +0 -0
  24. apb2/parserV2/joins/alphadia.py +60 -0
  25. apb2/parserV2/joins/maxquant.py +141 -0
  26. apb2/parserV2/parse_quant/__init__.py +0 -0
  27. apb2/parserV2/parse_quant/axis_columns.py +169 -0
  28. apb2/parserV2/parse_quant/contracts.py +138 -0
  29. apb2/parserV2/parse_quant/data/__init__.py +0 -0
  30. apb2/parserV2/parse_quant/data/errors.py +7 -0
  31. apb2/parserV2/parse_quant/data/layer_columns.py +56 -0
  32. apb2/parserV2/parse_quant/data/parsed.py +684 -0
  33. apb2/parserV2/parse_quant/data/raw.py +120 -0
  34. apb2/parserV2/parse_quant/data/source.py +27 -0
  35. apb2/parserV2/parse_quant/decomposition.py +289 -0
  36. apb2/parserV2/parse_quant/delimited_input.py +330 -0
  37. apb2/parserV2/parse_quant/duplicates.py +122 -0
  38. apb2/parserV2/parse_quant/errors.py +37 -0
  39. apb2/parserV2/parse_quant/excel_input.py +104 -0
  40. apb2/parserV2/parse_quant/fragments.py +104 -0
  41. apb2/parserV2/parse_quant/io/__init__.py +0 -0
  42. apb2/parserV2/parse_quant/io/anndata_reader.py +472 -0
  43. apb2/parserV2/parse_quant/io/anndata_writer.py +566 -0
  44. apb2/parserV2/parse_quant/io/duckdb.py +352 -0
  45. apb2/parserV2/parse_quant/io/errors.py +15 -0
  46. apb2/parserV2/parse_quant/io/formats.py +113 -0
  47. apb2/parserV2/parse_quant/io/json_representation.py +335 -0
  48. apb2/parserV2/parse_quant/io/layer_representation.py +243 -0
  49. apb2/parserV2/parse_quant/io/metadata.py +502 -0
  50. apb2/parserV2/parse_quant/io/parquet_reader.py +232 -0
  51. apb2/parserV2/parse_quant/io/parquet_writer.py +207 -0
  52. apb2/parserV2/parse_quant/io/uns_json.py +143 -0
  53. apb2/parserV2/parse_quant/io/validation.py +237 -0
  54. apb2/parserV2/parse_quant/layer_validation.py +64 -0
  55. apb2/parserV2/parse_quant/modifications.py +655 -0
  56. apb2/parserV2/parse_quant/numeric_text.py +66 -0
  57. apb2/parserV2/parse_quant/operations.py +163 -0
  58. apb2/parserV2/parse_quant/parameters/__init__.py +0 -0
  59. apb2/parserV2/parse_quant/parameters/axis.py +69 -0
  60. apb2/parserV2/parse_quant/parameters/level.py +10 -0
  61. apb2/parserV2/parse_quant/parameters/measurements.py +79 -0
  62. apb2/parserV2/parse_quant/parameters/source.py +293 -0
  63. apb2/parserV2/parse_quant/parquet_input.py +44 -0
  64. apb2/parserV2/parse_quant/parser.py +397 -0
  65. apb2/parserV2/parse_quant/plan_json.py +105 -0
  66. apb2/parserV2/parse_quant/prepared_input.py +28 -0
  67. apb2/parserV2/parse_quant/source_resolution.py +628 -0
  68. apb2/parserV2/parse_quant/value_parsing.py +232 -0
  69. apb2/parserV2/parse_rule_facade.py +509 -0
  70. apb2/parserV2/parser_factory.py +55 -0
  71. apb2/parserV2/prepare_source.py +102 -0
  72. apb2/parserV2/source_binding.py +121 -0
  73. apb2/parserV2/vendor_params/__init__.py +0 -0
  74. apb2/parserV2/vendor_params/parsers/__init__.py +0 -0
  75. apb2/parserV2/vendor_params/parsers/alphadia.py +199 -0
  76. apb2/parserV2/vendor_params/parsers/alphapept.py +127 -0
  77. apb2/parserV2/vendor_params/parsers/diann.py +522 -0
  78. apb2/parserV2/vendor_params/parsers/fragpipe.py +413 -0
  79. apb2/parserV2/vendor_params/parsers/i2masschroq.py +149 -0
  80. apb2/parserV2/vendor_params/parsers/maxquant.py +293 -0
  81. apb2/parserV2/vendor_params/parsers/metamorpheus.py +202 -0
  82. apb2/parserV2/vendor_params/parsers/msaid.py +66 -0
  83. apb2/parserV2/vendor_params/parsers/msangel.py +103 -0
  84. apb2/parserV2/vendor_params/parsers/peaks.py +187 -0
  85. apb2/parserV2/vendor_params/parsers/prolinestudio.py +120 -0
  86. apb2/parserV2/vendor_params/parsers/quantms.py +43 -0
  87. apb2/parserV2/vendor_params/parsers/sage.py +106 -0
  88. apb2/parserV2/vendor_params/parsers/shared/__init__.py +0 -0
  89. apb2/parserV2/vendor_params/parsers/shared/common.py +232 -0
  90. apb2/parserV2/vendor_params/parsers/shared/model.py +209 -0
  91. apb2/parserV2/vendor_params/parsers/shared/unimod.py +166 -0
  92. apb2/parserV2/vendor_params/parsers/shared/unimod_registry.json +55 -0
  93. apb2/parserV2/vendor_params/parsers/spectronaut.py +211 -0
  94. apb2/parserV2/vendor_params/parsers/wombat.py +110 -0
  95. apb2/parserV2/vendor_params/registry.py +133 -0
  96. apb2/parserV2/vendor_parse_rules/__init__.py +0 -0
  97. apb2/parserV2/vendor_parse_rules/catalog.json +25 -0
  98. apb2/parserV2/vendor_parse_rules/catalog.py +179 -0
  99. apb2/parserV2/vendor_parse_rules/document.py +319 -0
  100. apb2/parserV2/vendor_parse_rules/documents/__init__.py +0 -0
  101. apb2/parserV2/vendor_parse_rules/documents/_schema/document.schema.json +490 -0
  102. apb2/parserV2/vendor_parse_rules/documents/_schema/rule.schema.json +1449 -0
  103. apb2/parserV2/vendor_parse_rules/documents/alphadia/v1_10/rules.json +145 -0
  104. apb2/parserV2/vendor_parse_rules/documents/alphadia/v1_12/rules.json +190 -0
  105. apb2/parserV2/vendor_parse_rules/documents/alphadia/v2/rules.json +238 -0
  106. apb2/parserV2/vendor_parse_rules/documents/alphapept/rules.json +217 -0
  107. apb2/parserV2/vendor_parse_rules/documents/diann/v1_7/rules.json +365 -0
  108. apb2/parserV2/vendor_parse_rules/documents/diann/v1_8/rules.json +391 -0
  109. apb2/parserV2/vendor_parse_rules/documents/diann/v2/rules.json +292 -0
  110. apb2/parserV2/vendor_parse_rules/documents/fragpipe/rules.json +182 -0
  111. apb2/parserV2/vendor_parse_rules/documents/i2masschroq/rules.json +116 -0
  112. apb2/parserV2/vendor_parse_rules/documents/maxquant/rules.json +601 -0
  113. apb2/parserV2/vendor_parse_rules/documents/msangel/rules.json +177 -0
  114. apb2/parserV2/vendor_parse_rules/documents/pb_custom/rules.json +70 -0
  115. apb2/parserV2/vendor_parse_rules/documents/peaks/rules.json +191 -0
  116. apb2/parserV2/vendor_parse_rules/documents/prolinestudio/rules.json +177 -0
  117. apb2/parserV2/vendor_parse_rules/documents/quantms/rules.json +153 -0
  118. apb2/parserV2/vendor_parse_rules/documents/sage/rules.json +159 -0
  119. apb2/parserV2/vendor_parse_rules/documents/spectronaut/rules.json +628 -0
  120. apb2/parserV2/vendor_parse_rules/documents/spectronaut/v15/rules.json +535 -0
  121. apb2/parserV2/vendor_parse_rules/documents/spectronaut/v21/rules.json +636 -0
  122. apb2/parserV2/vendor_parse_rules/documents/wombat/rules.json +156 -0
  123. apb2/parserV2/vendor_parse_rules/loader.py +40 -0
  124. apb2/parserV2/vendor_parse_rules/schema/__init__.py +0 -0
  125. apb2/parserV2/vendor_parse_rules/schema/annotation.py +40 -0
  126. apb2/parserV2/vendor_parse_rules/schema/axis.py +128 -0
  127. apb2/parserV2/vendor_parse_rules/schema/base.py +31 -0
  128. apb2/parserV2/vendor_parse_rules/schema/base_formats.py +62 -0
  129. apb2/parserV2/vendor_parse_rules/schema/base_modifications.py +74 -0
  130. apb2/parserV2/vendor_parse_rules/schema/fragments.py +42 -0
  131. apb2/parserV2/vendor_parse_rules/schema/hierarchies.json +5 -0
  132. apb2/parserV2/vendor_parse_rules/schema/hierarchy.py +11 -0
  133. apb2/parserV2/vendor_parse_rules/schema/input.py +50 -0
  134. apb2/parserV2/vendor_parse_rules/schema/measurements.py +127 -0
  135. apb2/parserV2/vendor_parse_rules/schema/parameters.py +19 -0
  136. apb2/parserV2/vendor_parse_rules/schema/role_policy.json +10 -0
  137. apb2/parserV2/vendor_parse_rules/schema/roles.py +20 -0
  138. apb2/parserV2/vendor_parse_rules/schema/rule.py +295 -0
  139. apb2/parserV2/vendor_parse_rules/schema_artifact.py +31 -0
  140. apb2/py.typed +1 -0
  141. apb2-0.1.0.dist-info/METADATA +185 -0
  142. apb2-0.1.0.dist-info/RECORD +145 -0
  143. apb2-0.1.0.dist-info/WHEEL +4 -0
  144. apb2-0.1.0.dist-info/entry_points.txt +3 -0
  145. apb2-0.1.0.dist-info/licenses/LICENSE +21 -0
apb2/__init__.py ADDED
@@ -0,0 +1 @@
1
+ """Rules-driven proteomics vendor-table conversion."""
File without changes
File without changes
@@ -0,0 +1,306 @@
1
+ """Runtime behaviors for attaching annotation and selecting observations."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Mapping
6
+ from dataclasses import dataclass, replace
7
+ from typing import Protocol
8
+
9
+ import polars as pl
10
+
11
+ from apb2.annotation.data.model import (
12
+ AnnotationError,
13
+ AnnotationFileOrigin,
14
+ AnnotationMatches,
15
+ AnnotationOrigin,
16
+ AnnotationResult,
17
+ LevelAnnotationMatch,
18
+ LevelAnnotationReport,
19
+ )
20
+ from apb2.parserV2.parse_quant.data.layer_columns import observation_labels
21
+ from apb2.parserV2.parse_quant.data.parsed import (
22
+ FinalLayerTable,
23
+ JsonValue,
24
+ ObsFinal,
25
+ ParsedLevel,
26
+ ParsedLevelName,
27
+ ParsedLevels,
28
+ )
29
+
30
+
31
+ class ObservationSelection(Protocol):
32
+ """Compute one validated Boolean selection per observation."""
33
+
34
+ def validate(self, match: LevelAnnotationMatch, /) -> None:
35
+ """Reject evidence from which this selection cannot be computed."""
36
+ ...
37
+
38
+ def selected_rows(self, match: LevelAnnotationMatch, /) -> pl.Series:
39
+ """Return one non-null Boolean per observation."""
40
+ ...
41
+
42
+
43
+ class AnnotationApplication(Protocol):
44
+ """Validate and apply one observation-retention behavior."""
45
+
46
+ def validate(self, matches: AnnotationMatches, /) -> None:
47
+ """Reject matching evidence that cannot produce this application."""
48
+ ...
49
+
50
+ def apply(self, parsed: ParsedLevels, matches: AnnotationMatches, /) -> AnnotationResult:
51
+ """Attach annotation and return a new storage-neutral result."""
52
+ ...
53
+
54
+
55
+ class MatchedAnnotationSelection:
56
+ """Select observations matched to any annotation row."""
57
+
58
+ __slots__ = ()
59
+
60
+ def validate(self, match: LevelAnnotationMatch, /) -> None:
61
+ del match
62
+
63
+ def selected_rows(self, match: LevelAnnotationMatch, /) -> pl.Series:
64
+ return match.matched_rows
65
+
66
+
67
+ @dataclass(frozen=True, slots=True)
68
+ class BooleanAnnotationSelection:
69
+ """Select matched observations whose annotation field is true."""
70
+
71
+ column: str
72
+
73
+ def validate(self, match: LevelAnnotationMatch, /) -> None:
74
+ if self.column not in match.aligned.columns:
75
+ raise AnnotationError(
76
+ f"annotation selection column {self.column!r} is absent; "
77
+ f"available={match.aligned.columns}"
78
+ )
79
+ values = match.aligned.get_column(self.column)
80
+ if values.dtype != pl.Boolean:
81
+ raise AnnotationError(
82
+ f"annotation selection column {self.column!r} must be Boolean, got {values.dtype}"
83
+ )
84
+ if values.filter(match.matched_rows).null_count():
85
+ raise AnnotationError(
86
+ f"annotation selection column {self.column!r} contains null matched values"
87
+ )
88
+
89
+ def selected_rows(self, match: LevelAnnotationMatch, /) -> pl.Series:
90
+ return match.aligned.get_column(self.column).fill_null(False)
91
+
92
+
93
+ @dataclass(frozen=True, slots=True)
94
+ class AllAnnotationSelections:
95
+ """Intersect several independently reusable observation selections."""
96
+
97
+ selections: tuple[ObservationSelection, ...]
98
+
99
+ def validate(self, match: LevelAnnotationMatch, /) -> None:
100
+ for selection in self.selections:
101
+ selection.validate(match)
102
+
103
+ def selected_rows(self, match: LevelAnnotationMatch, /) -> pl.Series:
104
+ selected = pl.Series("selected", [True] * len(match.matched_rows), dtype=pl.Boolean)
105
+ for selection in self.selections:
106
+ selected = selected & selection.selected_rows(match)
107
+ return selected.rename("selected")
108
+
109
+
110
+ class KeepUnmatchedAnnotation:
111
+ """Attach null metadata to unmatched observations and retain every observation."""
112
+
113
+ __slots__ = ()
114
+
115
+ def validate(self, matches: AnnotationMatches, /) -> None:
116
+ _require_any_match(matches)
117
+
118
+ def apply(self, parsed: ParsedLevels, matches: AnnotationMatches, /) -> AnnotationResult:
119
+ return _apply(parsed, matches, selections=None)
120
+
121
+
122
+ class RequireCompleteAnnotation:
123
+ """Require every observation to match before attaching annotation."""
124
+
125
+ __slots__ = ()
126
+
127
+ def validate(self, matches: AnnotationMatches, /) -> None:
128
+ _require_any_match(matches)
129
+ incomplete = {
130
+ name: match.coverage
131
+ for name, match in matches.levels.items()
132
+ if match.coverage.quant_only_count
133
+ }
134
+ if incomplete:
135
+ details = "; ".join(
136
+ f"{name}: {coverage.quant_only_count} unmatched "
137
+ f"{list(coverage.quant_only_examples)}, near_misses={dict(coverage.near_misses)}"
138
+ for name, coverage in incomplete.items()
139
+ )
140
+ raise AnnotationError(f"complete sample annotation required; {details}")
141
+
142
+ def apply(self, parsed: ParsedLevels, matches: AnnotationMatches, /) -> AnnotationResult:
143
+ return _apply(parsed, matches, selections=None)
144
+
145
+
146
+ @dataclass(frozen=True, slots=True)
147
+ class SelectAnnotatedObservations:
148
+ """Attach annotation and consistently retain only selected observations."""
149
+
150
+ selection: ObservationSelection
151
+
152
+ def validate(self, matches: AnnotationMatches, /) -> None:
153
+ _require_any_match(matches)
154
+ for match in matches.levels.values():
155
+ self.selection.validate(match)
156
+
157
+ def apply(self, parsed: ParsedLevels, matches: AnnotationMatches, /) -> AnnotationResult:
158
+ selections: dict[ParsedLevelName, pl.Series] = {
159
+ name: self.selection.selected_rows(match) for name, match in matches.levels.items()
160
+ }
161
+ return _apply(parsed, matches, selections=selections)
162
+
163
+
164
+ def record_annotation_provenance(
165
+ result: AnnotationResult,
166
+ convention: str,
167
+ origin: AnnotationOrigin,
168
+ /,
169
+ *,
170
+ metadata: Mapping[str, JsonValue] | None = None,
171
+ ) -> AnnotationResult:
172
+ """Record tool-owned source provenance and each level's annotation report."""
173
+ if convention in {"parse", "roles", "storage"}:
174
+ raise AnnotationError(f"annotation convention uses reserved APB section {convention!r}")
175
+ record: dict[str, JsonValue] = dict(metadata or {})
176
+ record["schema_version"] = "2"
177
+ if "source" not in record:
178
+ record["source"] = (
179
+ {"path": str(origin.path)} if isinstance(origin, AnnotationFileOrigin) else None
180
+ )
181
+ root_tool = _metadata_section(result.parsed.metadata, convention)
182
+ provenance = _metadata_section(root_tool, "provenance")
183
+ provenance["annotation"] = record
184
+ for name, level in result.parsed.levels.items():
185
+ report = _report_json(result.reports[name])
186
+ tool = _metadata_section(level.metadata, convention)
187
+ tool["annotation"] = report
188
+ return result
189
+
190
+
191
+ def _metadata_section(metadata: dict[str, JsonValue], name: str) -> dict[str, JsonValue]:
192
+ value = metadata.setdefault(name, {})
193
+ if not isinstance(value, dict):
194
+ raise AnnotationError(f"APB metadata section {name!r} must be an object")
195
+ return value
196
+
197
+
198
+ def _apply(
199
+ parsed: ParsedLevels,
200
+ matches: AnnotationMatches,
201
+ selections: Mapping[ParsedLevelName, pl.Series] | None,
202
+ ) -> AnnotationResult:
203
+ levels: dict[ParsedLevelName, ParsedLevel] = {}
204
+ reports: dict[ParsedLevelName, LevelAnnotationReport] = {}
205
+ for name, level in parsed.levels.items():
206
+ match = matches.levels[name]
207
+ selected = (
208
+ pl.Series("selected", [True] * level.obs.frame.height, dtype=pl.Boolean)
209
+ if selections is None
210
+ else selections[name]
211
+ )
212
+ levels[name] = _annotated_level(level, match, selected)
213
+ reports[name] = LevelAnnotationReport(
214
+ coverage=match.coverage,
215
+ corrections=match.corrections,
216
+ columns_added=tuple(match.aligned.columns),
217
+ )
218
+ return AnnotationResult(
219
+ parsed=ParsedLevels(
220
+ levels=levels,
221
+ uns=dict(parsed.uns),
222
+ metadata=dict(parsed.metadata),
223
+ annotation_tables=dict(parsed.annotation_tables),
224
+ feature_relations=dict(parsed.feature_relations),
225
+ ),
226
+ reports=reports,
227
+ )
228
+
229
+
230
+ def _require_any_match(matches: AnnotationMatches) -> None:
231
+ empty = [
232
+ name
233
+ for name, match in matches.levels.items()
234
+ if match.coverage.matched_observation_count == 0
235
+ ]
236
+ if empty:
237
+ raise AnnotationError(
238
+ f"sample annotation matched no observations for level(s) {empty}; "
239
+ "no dataset-bound annotation was constructed"
240
+ )
241
+
242
+
243
+ def _annotated_level(
244
+ level: ParsedLevel,
245
+ match: LevelAnnotationMatch,
246
+ selected: pl.Series,
247
+ ) -> ParsedLevel:
248
+ annotated_obs = level.obs.frame.hstack(match.aligned.get_columns()).filter(selected)
249
+ kept = [index for index, value in enumerate(selected) if value]
250
+ return replace(
251
+ level,
252
+ obs=ObsFinal(frame=annotated_obs, key_columns=level.obs.key_columns),
253
+ uns=dict(level.uns),
254
+ layers={name: _subset_layer(layer, kept) for name, layer in level.layers.items()},
255
+ obsm={name: frame.filter(selected) for name, frame in level.obsm.items()},
256
+ varm=dict(level.varm),
257
+ obsp={name: _subset_pairwise(frame, kept) for name, frame in level.obsp.items()},
258
+ varp=dict(level.varp),
259
+ metadata=dict(level.metadata),
260
+ )
261
+
262
+
263
+ def _subset_layer(layer: FinalLayerTable, kept: list[int]) -> FinalLayerTable:
264
+ value_columns = layer.values.columns
265
+ selected_names = [value_columns[index] for index in kept]
266
+ values = layer.values.select(selected_names)
267
+ replacement = observation_labels(len(kept), ())
268
+ values = values.rename(dict(zip(selected_names, replacement, strict=True)))
269
+ return replace(layer, values=values)
270
+
271
+
272
+ def _subset_pairwise(frame: pl.DataFrame, kept: list[int]) -> pl.DataFrame:
273
+ mapping = {old: new for new, old in enumerate(kept)}
274
+ if not mapping or frame.is_empty():
275
+ return frame.head(0)
276
+ return frame.filter(pl.col("row").is_in(kept) & pl.col("column").is_in(kept)).with_columns(
277
+ pl.col("row").replace_strict(mapping),
278
+ pl.col("column").replace_strict(mapping),
279
+ )
280
+
281
+
282
+ def _report_json(report: LevelAnnotationReport) -> dict[str, JsonValue]:
283
+ coverage = report.coverage
284
+ near_misses: dict[str, JsonValue] = {
285
+ key: dict(candidates) for key, candidates in coverage.near_misses.items()
286
+ }
287
+ corrections: dict[str, JsonValue] = {
288
+ correction.observed: {
289
+ "observed": correction.observed,
290
+ "expected": correction.expected,
291
+ "score": correction.score,
292
+ }
293
+ for correction in report.corrections
294
+ }
295
+ return {
296
+ "observation_count": coverage.observation_count,
297
+ "annotation_count": coverage.annotation_count,
298
+ "matched_observation_count": coverage.matched_observation_count,
299
+ "quant_only_count": coverage.quant_only_count,
300
+ "annotation_only_count": coverage.annotation_only_count,
301
+ "quant_only_examples": list(coverage.quant_only_examples),
302
+ "annotation_only_examples": list(coverage.annotation_only_examples),
303
+ "near_misses": near_misses,
304
+ "corrections": corrections,
305
+ "columns_added": list(report.columns_added),
306
+ }
@@ -0,0 +1,76 @@
1
+ """Compile one delimited annotation source into a source-bound parser."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from pathlib import Path
6
+ from typing import Literal
7
+
8
+ import polars as pl
9
+
10
+ from apb2.annotation.application.policies import (
11
+ AllAnnotationSelections,
12
+ AnnotationApplication,
13
+ BooleanAnnotationSelection,
14
+ KeepUnmatchedAnnotation,
15
+ MatchedAnnotationSelection,
16
+ ObservationSelection,
17
+ RequireCompleteAnnotation,
18
+ SelectAnnotatedObservations,
19
+ )
20
+ from apb2.annotation.contracts import AnnotationParser
21
+ from apb2.annotation.data.model import AnnotationError
22
+ from apb2.annotation.prolfquapp import ProlfquappAnnotationParser, prolfquapp_signature
23
+ from apb2.annotation.sdrf import SdrfAnnotationParser, SdrfSource, sdrf_signature
24
+ from apb2.annotation.source.load import load_annotation_file, load_annotation_frame
25
+
26
+
27
+ class AnnotationCompiler:
28
+ """User-configured parser for SDRF and generic delimited observation annotations."""
29
+
30
+ __slots__ = ("_application",)
31
+
32
+ def __init__(
33
+ self,
34
+ unmatched: Literal["keep", "error", "drop"] = "keep",
35
+ include: str | None = None,
36
+ ) -> None:
37
+ """Choose what happens to observations the annotation does not cover.
38
+
39
+ Args:
40
+ unmatched: ``keep`` them with null metadata, raise (``error``), or ``drop`` them.
41
+ include: A Boolean annotation column that further selects observations; requires
42
+ ``unmatched="drop"``.
43
+
44
+ Raises:
45
+ AnnotationError: ``include`` is given without ``unmatched="drop"``.
46
+ """
47
+ self._application = _application(unmatched, include)
48
+
49
+ def compile(self, source: Path | pl.DataFrame) -> AnnotationParser:
50
+ """Load once, verify the tabular convention, and return its bound parser.
51
+
52
+ SDRF headers take precedence; other tables must carry a prolfquapp observation key.
53
+ """
54
+ loaded = (
55
+ load_annotation_file(source)
56
+ if isinstance(source, Path)
57
+ else load_annotation_frame(source)
58
+ )
59
+ if sdrf_signature(loaded):
60
+ return SdrfAnnotationParser(source=SdrfSource(loaded), application=self._application)
61
+ if not prolfquapp_signature(loaded):
62
+ raise AnnotationError("annotation table has no supported observation key")
63
+ return ProlfquappAnnotationParser(source=loaded, application=self._application)
64
+
65
+
66
+ def _application(
67
+ unmatched: Literal["keep", "error", "drop"], include: str | None
68
+ ) -> AnnotationApplication:
69
+ if unmatched == "drop":
70
+ selections: list[ObservationSelection] = [MatchedAnnotationSelection()]
71
+ if include is not None:
72
+ selections.append(BooleanAnnotationSelection(include))
73
+ return SelectAnnotatedObservations(AllAnnotationSelections(tuple(selections)))
74
+ if include is not None:
75
+ raise AnnotationError("include requires unmatched='drop'")
76
+ return KeepUnmatchedAnnotation() if unmatched == "keep" else RequireCompleteAnnotation()
@@ -0,0 +1,29 @@
1
+ """Small public capabilities returned by sample-annotation compilation."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Protocol
6
+
7
+ from apb2.annotation.data.model import AnnotationMatches, AnnotationResult
8
+ from apb2.parserV2.parse_quant.data.parsed import ParsedLevels
9
+
10
+
11
+ class Annotation(Protocol):
12
+ """A validated sample annotation bound to exactly one parsed dataset."""
13
+
14
+ @property
15
+ def matches(self) -> AnnotationMatches:
16
+ """Return completed per-level matching evidence."""
17
+ ...
18
+
19
+ def annotate(self) -> AnnotationResult:
20
+ """Apply this annotation to the dataset against which it was validated."""
21
+ ...
22
+
23
+
24
+ class AnnotationParser(Protocol):
25
+ """A convention parser bound to one already loaded annotation source."""
26
+
27
+ def parse(self, parsed: ParsedLevels, /) -> Annotation:
28
+ """Validate and match the source, then construct a dataset-bound annotation."""
29
+ ...
File without changes
@@ -0,0 +1,112 @@
1
+ """Innermost values exchanged by the sample-annotation workflow."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Mapping
6
+ from dataclasses import dataclass
7
+ from pathlib import Path
8
+
9
+ import polars as pl
10
+
11
+ from apb2.parserV2.parse_quant.data.parsed import ParsedLevelName, ParsedLevels
12
+
13
+
14
+ class AnnotationError(ValueError):
15
+ """A sample annotation cannot be recognized, parsed, matched, or applied."""
16
+
17
+
18
+ @dataclass(frozen=True, slots=True)
19
+ class AnnotationFileOrigin:
20
+ """The resolved file from which an annotation was loaded."""
21
+
22
+ path: Path
23
+
24
+
25
+ @dataclass(frozen=True, slots=True)
26
+ class InMemoryAnnotationOrigin:
27
+ """Marker for a programmatically supplied Polars annotation."""
28
+
29
+
30
+ type AnnotationOrigin = AnnotationFileOrigin | InMemoryAnnotationOrigin
31
+
32
+ IN_MEMORY_ANNOTATION = InMemoryAnnotationOrigin()
33
+
34
+
35
+ @dataclass(frozen=True, slots=True)
36
+ class LoadedAnnotationSource:
37
+ """One physically decoded tabular annotation source.
38
+
39
+ ``headers`` holds the verbatim header text aligned with ``frame.columns``. Repeated headers
40
+ keep their text there while the frame columns stay unique.
41
+ """
42
+
43
+ frame: pl.DataFrame
44
+ origin: AnnotationOrigin
45
+ headers: tuple[str, ...]
46
+
47
+
48
+ @dataclass(frozen=True, slots=True)
49
+ class AnnotationTable:
50
+ """Validated sample rows and every identifier column accepted for matching."""
51
+
52
+ frame: pl.DataFrame
53
+ key_columns: tuple[str, ...]
54
+ alias_columns: tuple[str, ...]
55
+ origin: AnnotationOrigin
56
+
57
+
58
+ @dataclass(frozen=True, slots=True)
59
+ class AnnotationCoverage:
60
+ """Counts and bounded mismatch evidence for one quantification level."""
61
+
62
+ observation_count: int
63
+ annotation_count: int
64
+ matched_observation_count: int
65
+ quant_only_count: int
66
+ annotation_only_count: int
67
+ quant_only_examples: tuple[str, ...]
68
+ annotation_only_examples: tuple[str, ...]
69
+ near_misses: Mapping[str, tuple[tuple[str, float], ...]]
70
+
71
+
72
+ @dataclass(frozen=True, slots=True)
73
+ class KeyCorrection:
74
+ """One accepted fuzzy observation-to-annotation identifier correction."""
75
+
76
+ observed: str
77
+ expected: str
78
+ score: float
79
+
80
+
81
+ @dataclass(frozen=True, slots=True)
82
+ class LevelAnnotationMatch:
83
+ """Annotation aligned to one observation axis plus its matching evidence."""
84
+
85
+ aligned: pl.DataFrame
86
+ matched_rows: pl.Series
87
+ coverage: AnnotationCoverage
88
+ corrections: tuple[KeyCorrection, ...]
89
+
90
+
91
+ @dataclass(frozen=True, slots=True)
92
+ class AnnotationMatches:
93
+ """Independent matching evidence for every parsed quantification level."""
94
+
95
+ levels: Mapping[ParsedLevelName, LevelAnnotationMatch]
96
+
97
+
98
+ @dataclass(frozen=True, slots=True)
99
+ class LevelAnnotationReport:
100
+ """Persistable evidence produced by applying one level's annotation."""
101
+
102
+ coverage: AnnotationCoverage
103
+ corrections: tuple[KeyCorrection, ...]
104
+ columns_added: tuple[str, ...]
105
+
106
+
107
+ @dataclass(frozen=True, slots=True)
108
+ class AnnotationResult:
109
+ """A newly annotated storage-neutral result and its per-level reports."""
110
+
111
+ parsed: ParsedLevels
112
+ reports: Mapping[ParsedLevelName, LevelAnnotationReport]
File without changes