apb2 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- apb2/__init__.py +1 -0
- apb2/annotation/__init__.py +0 -0
- apb2/annotation/application/__init__.py +0 -0
- apb2/annotation/application/policies.py +306 -0
- apb2/annotation/compiler.py +76 -0
- apb2/annotation/contracts.py +29 -0
- apb2/annotation/data/__init__.py +0 -0
- apb2/annotation/data/model.py +112 -0
- apb2/annotation/matching/__init__.py +0 -0
- apb2/annotation/matching/core.py +467 -0
- apb2/annotation/prolfquapp.py +125 -0
- apb2/annotation/sdrf.py +205 -0
- apb2/annotation/source/__init__.py +0 -0
- apb2/annotation/source/load.py +68 -0
- apb2/api.py +65 -0
- apb2/cli/__init__.py +0 -0
- apb2/cli/annotation.py +53 -0
- apb2/cli/app.py +214 -0
- apb2/cli/conversion.py +291 -0
- apb2/parserV2/__init__.py +0 -0
- apb2/parserV2/compile.py +256 -0
- apb2/parserV2/detect_document.py +562 -0
- apb2/parserV2/joins/__init__.py +0 -0
- apb2/parserV2/joins/alphadia.py +60 -0
- apb2/parserV2/joins/maxquant.py +141 -0
- apb2/parserV2/parse_quant/__init__.py +0 -0
- apb2/parserV2/parse_quant/axis_columns.py +169 -0
- apb2/parserV2/parse_quant/contracts.py +138 -0
- apb2/parserV2/parse_quant/data/__init__.py +0 -0
- apb2/parserV2/parse_quant/data/errors.py +7 -0
- apb2/parserV2/parse_quant/data/layer_columns.py +56 -0
- apb2/parserV2/parse_quant/data/parsed.py +684 -0
- apb2/parserV2/parse_quant/data/raw.py +120 -0
- apb2/parserV2/parse_quant/data/source.py +27 -0
- apb2/parserV2/parse_quant/decomposition.py +289 -0
- apb2/parserV2/parse_quant/delimited_input.py +330 -0
- apb2/parserV2/parse_quant/duplicates.py +122 -0
- apb2/parserV2/parse_quant/errors.py +37 -0
- apb2/parserV2/parse_quant/excel_input.py +104 -0
- apb2/parserV2/parse_quant/fragments.py +104 -0
- apb2/parserV2/parse_quant/io/__init__.py +0 -0
- apb2/parserV2/parse_quant/io/anndata_reader.py +472 -0
- apb2/parserV2/parse_quant/io/anndata_writer.py +566 -0
- apb2/parserV2/parse_quant/io/duckdb.py +352 -0
- apb2/parserV2/parse_quant/io/errors.py +15 -0
- apb2/parserV2/parse_quant/io/formats.py +113 -0
- apb2/parserV2/parse_quant/io/json_representation.py +335 -0
- apb2/parserV2/parse_quant/io/layer_representation.py +243 -0
- apb2/parserV2/parse_quant/io/metadata.py +502 -0
- apb2/parserV2/parse_quant/io/parquet_reader.py +232 -0
- apb2/parserV2/parse_quant/io/parquet_writer.py +207 -0
- apb2/parserV2/parse_quant/io/uns_json.py +143 -0
- apb2/parserV2/parse_quant/io/validation.py +237 -0
- apb2/parserV2/parse_quant/layer_validation.py +64 -0
- apb2/parserV2/parse_quant/modifications.py +655 -0
- apb2/parserV2/parse_quant/numeric_text.py +66 -0
- apb2/parserV2/parse_quant/operations.py +163 -0
- apb2/parserV2/parse_quant/parameters/__init__.py +0 -0
- apb2/parserV2/parse_quant/parameters/axis.py +69 -0
- apb2/parserV2/parse_quant/parameters/level.py +10 -0
- apb2/parserV2/parse_quant/parameters/measurements.py +79 -0
- apb2/parserV2/parse_quant/parameters/source.py +293 -0
- apb2/parserV2/parse_quant/parquet_input.py +44 -0
- apb2/parserV2/parse_quant/parser.py +397 -0
- apb2/parserV2/parse_quant/plan_json.py +105 -0
- apb2/parserV2/parse_quant/prepared_input.py +28 -0
- apb2/parserV2/parse_quant/source_resolution.py +628 -0
- apb2/parserV2/parse_quant/value_parsing.py +232 -0
- apb2/parserV2/parse_rule_facade.py +509 -0
- apb2/parserV2/parser_factory.py +55 -0
- apb2/parserV2/prepare_source.py +102 -0
- apb2/parserV2/source_binding.py +121 -0
- apb2/parserV2/vendor_params/__init__.py +0 -0
- apb2/parserV2/vendor_params/parsers/__init__.py +0 -0
- apb2/parserV2/vendor_params/parsers/alphadia.py +199 -0
- apb2/parserV2/vendor_params/parsers/alphapept.py +127 -0
- apb2/parserV2/vendor_params/parsers/diann.py +522 -0
- apb2/parserV2/vendor_params/parsers/fragpipe.py +413 -0
- apb2/parserV2/vendor_params/parsers/i2masschroq.py +149 -0
- apb2/parserV2/vendor_params/parsers/maxquant.py +293 -0
- apb2/parserV2/vendor_params/parsers/metamorpheus.py +202 -0
- apb2/parserV2/vendor_params/parsers/msaid.py +66 -0
- apb2/parserV2/vendor_params/parsers/msangel.py +103 -0
- apb2/parserV2/vendor_params/parsers/peaks.py +187 -0
- apb2/parserV2/vendor_params/parsers/prolinestudio.py +120 -0
- apb2/parserV2/vendor_params/parsers/quantms.py +43 -0
- apb2/parserV2/vendor_params/parsers/sage.py +106 -0
- apb2/parserV2/vendor_params/parsers/shared/__init__.py +0 -0
- apb2/parserV2/vendor_params/parsers/shared/common.py +232 -0
- apb2/parserV2/vendor_params/parsers/shared/model.py +209 -0
- apb2/parserV2/vendor_params/parsers/shared/unimod.py +166 -0
- apb2/parserV2/vendor_params/parsers/shared/unimod_registry.json +55 -0
- apb2/parserV2/vendor_params/parsers/spectronaut.py +211 -0
- apb2/parserV2/vendor_params/parsers/wombat.py +110 -0
- apb2/parserV2/vendor_params/registry.py +133 -0
- apb2/parserV2/vendor_parse_rules/__init__.py +0 -0
- apb2/parserV2/vendor_parse_rules/catalog.json +25 -0
- apb2/parserV2/vendor_parse_rules/catalog.py +179 -0
- apb2/parserV2/vendor_parse_rules/document.py +319 -0
- apb2/parserV2/vendor_parse_rules/documents/__init__.py +0 -0
- apb2/parserV2/vendor_parse_rules/documents/_schema/document.schema.json +490 -0
- apb2/parserV2/vendor_parse_rules/documents/_schema/rule.schema.json +1449 -0
- apb2/parserV2/vendor_parse_rules/documents/alphadia/v1_10/rules.json +145 -0
- apb2/parserV2/vendor_parse_rules/documents/alphadia/v1_12/rules.json +190 -0
- apb2/parserV2/vendor_parse_rules/documents/alphadia/v2/rules.json +238 -0
- apb2/parserV2/vendor_parse_rules/documents/alphapept/rules.json +217 -0
- apb2/parserV2/vendor_parse_rules/documents/diann/v1_7/rules.json +365 -0
- apb2/parserV2/vendor_parse_rules/documents/diann/v1_8/rules.json +391 -0
- apb2/parserV2/vendor_parse_rules/documents/diann/v2/rules.json +292 -0
- apb2/parserV2/vendor_parse_rules/documents/fragpipe/rules.json +182 -0
- apb2/parserV2/vendor_parse_rules/documents/i2masschroq/rules.json +116 -0
- apb2/parserV2/vendor_parse_rules/documents/maxquant/rules.json +601 -0
- apb2/parserV2/vendor_parse_rules/documents/msangel/rules.json +177 -0
- apb2/parserV2/vendor_parse_rules/documents/pb_custom/rules.json +70 -0
- apb2/parserV2/vendor_parse_rules/documents/peaks/rules.json +191 -0
- apb2/parserV2/vendor_parse_rules/documents/prolinestudio/rules.json +177 -0
- apb2/parserV2/vendor_parse_rules/documents/quantms/rules.json +153 -0
- apb2/parserV2/vendor_parse_rules/documents/sage/rules.json +159 -0
- apb2/parserV2/vendor_parse_rules/documents/spectronaut/rules.json +628 -0
- apb2/parserV2/vendor_parse_rules/documents/spectronaut/v15/rules.json +535 -0
- apb2/parserV2/vendor_parse_rules/documents/spectronaut/v21/rules.json +636 -0
- apb2/parserV2/vendor_parse_rules/documents/wombat/rules.json +156 -0
- apb2/parserV2/vendor_parse_rules/loader.py +40 -0
- apb2/parserV2/vendor_parse_rules/schema/__init__.py +0 -0
- apb2/parserV2/vendor_parse_rules/schema/annotation.py +40 -0
- apb2/parserV2/vendor_parse_rules/schema/axis.py +128 -0
- apb2/parserV2/vendor_parse_rules/schema/base.py +31 -0
- apb2/parserV2/vendor_parse_rules/schema/base_formats.py +62 -0
- apb2/parserV2/vendor_parse_rules/schema/base_modifications.py +74 -0
- apb2/parserV2/vendor_parse_rules/schema/fragments.py +42 -0
- apb2/parserV2/vendor_parse_rules/schema/hierarchies.json +5 -0
- apb2/parserV2/vendor_parse_rules/schema/hierarchy.py +11 -0
- apb2/parserV2/vendor_parse_rules/schema/input.py +50 -0
- apb2/parserV2/vendor_parse_rules/schema/measurements.py +127 -0
- apb2/parserV2/vendor_parse_rules/schema/parameters.py +19 -0
- apb2/parserV2/vendor_parse_rules/schema/role_policy.json +10 -0
- apb2/parserV2/vendor_parse_rules/schema/roles.py +20 -0
- apb2/parserV2/vendor_parse_rules/schema/rule.py +295 -0
- apb2/parserV2/vendor_parse_rules/schema_artifact.py +31 -0
- apb2/py.typed +1 -0
- apb2-0.1.0.dist-info/METADATA +185 -0
- apb2-0.1.0.dist-info/RECORD +145 -0
- apb2-0.1.0.dist-info/WHEEL +4 -0
- apb2-0.1.0.dist-info/entry_points.txt +3 -0
- apb2-0.1.0.dist-info/licenses/LICENSE +21 -0
apb2/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Rules-driven proteomics vendor-table conversion."""
|
|
File without changes
|
|
File without changes
|
|
@@ -0,0 +1,306 @@
|
|
|
1
|
+
"""Runtime behaviors for attaching annotation and selecting observations."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Mapping
|
|
6
|
+
from dataclasses import dataclass, replace
|
|
7
|
+
from typing import Protocol
|
|
8
|
+
|
|
9
|
+
import polars as pl
|
|
10
|
+
|
|
11
|
+
from apb2.annotation.data.model import (
|
|
12
|
+
AnnotationError,
|
|
13
|
+
AnnotationFileOrigin,
|
|
14
|
+
AnnotationMatches,
|
|
15
|
+
AnnotationOrigin,
|
|
16
|
+
AnnotationResult,
|
|
17
|
+
LevelAnnotationMatch,
|
|
18
|
+
LevelAnnotationReport,
|
|
19
|
+
)
|
|
20
|
+
from apb2.parserV2.parse_quant.data.layer_columns import observation_labels
|
|
21
|
+
from apb2.parserV2.parse_quant.data.parsed import (
|
|
22
|
+
FinalLayerTable,
|
|
23
|
+
JsonValue,
|
|
24
|
+
ObsFinal,
|
|
25
|
+
ParsedLevel,
|
|
26
|
+
ParsedLevelName,
|
|
27
|
+
ParsedLevels,
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class ObservationSelection(Protocol):
|
|
32
|
+
"""Compute one validated Boolean selection per observation."""
|
|
33
|
+
|
|
34
|
+
def validate(self, match: LevelAnnotationMatch, /) -> None:
|
|
35
|
+
"""Reject evidence from which this selection cannot be computed."""
|
|
36
|
+
...
|
|
37
|
+
|
|
38
|
+
def selected_rows(self, match: LevelAnnotationMatch, /) -> pl.Series:
|
|
39
|
+
"""Return one non-null Boolean per observation."""
|
|
40
|
+
...
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class AnnotationApplication(Protocol):
|
|
44
|
+
"""Validate and apply one observation-retention behavior."""
|
|
45
|
+
|
|
46
|
+
def validate(self, matches: AnnotationMatches, /) -> None:
|
|
47
|
+
"""Reject matching evidence that cannot produce this application."""
|
|
48
|
+
...
|
|
49
|
+
|
|
50
|
+
def apply(self, parsed: ParsedLevels, matches: AnnotationMatches, /) -> AnnotationResult:
|
|
51
|
+
"""Attach annotation and return a new storage-neutral result."""
|
|
52
|
+
...
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class MatchedAnnotationSelection:
|
|
56
|
+
"""Select observations matched to any annotation row."""
|
|
57
|
+
|
|
58
|
+
__slots__ = ()
|
|
59
|
+
|
|
60
|
+
def validate(self, match: LevelAnnotationMatch, /) -> None:
|
|
61
|
+
del match
|
|
62
|
+
|
|
63
|
+
def selected_rows(self, match: LevelAnnotationMatch, /) -> pl.Series:
|
|
64
|
+
return match.matched_rows
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
@dataclass(frozen=True, slots=True)
|
|
68
|
+
class BooleanAnnotationSelection:
|
|
69
|
+
"""Select matched observations whose annotation field is true."""
|
|
70
|
+
|
|
71
|
+
column: str
|
|
72
|
+
|
|
73
|
+
def validate(self, match: LevelAnnotationMatch, /) -> None:
|
|
74
|
+
if self.column not in match.aligned.columns:
|
|
75
|
+
raise AnnotationError(
|
|
76
|
+
f"annotation selection column {self.column!r} is absent; "
|
|
77
|
+
f"available={match.aligned.columns}"
|
|
78
|
+
)
|
|
79
|
+
values = match.aligned.get_column(self.column)
|
|
80
|
+
if values.dtype != pl.Boolean:
|
|
81
|
+
raise AnnotationError(
|
|
82
|
+
f"annotation selection column {self.column!r} must be Boolean, got {values.dtype}"
|
|
83
|
+
)
|
|
84
|
+
if values.filter(match.matched_rows).null_count():
|
|
85
|
+
raise AnnotationError(
|
|
86
|
+
f"annotation selection column {self.column!r} contains null matched values"
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
def selected_rows(self, match: LevelAnnotationMatch, /) -> pl.Series:
|
|
90
|
+
return match.aligned.get_column(self.column).fill_null(False)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
@dataclass(frozen=True, slots=True)
|
|
94
|
+
class AllAnnotationSelections:
|
|
95
|
+
"""Intersect several independently reusable observation selections."""
|
|
96
|
+
|
|
97
|
+
selections: tuple[ObservationSelection, ...]
|
|
98
|
+
|
|
99
|
+
def validate(self, match: LevelAnnotationMatch, /) -> None:
|
|
100
|
+
for selection in self.selections:
|
|
101
|
+
selection.validate(match)
|
|
102
|
+
|
|
103
|
+
def selected_rows(self, match: LevelAnnotationMatch, /) -> pl.Series:
|
|
104
|
+
selected = pl.Series("selected", [True] * len(match.matched_rows), dtype=pl.Boolean)
|
|
105
|
+
for selection in self.selections:
|
|
106
|
+
selected = selected & selection.selected_rows(match)
|
|
107
|
+
return selected.rename("selected")
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
class KeepUnmatchedAnnotation:
|
|
111
|
+
"""Attach null metadata to unmatched observations and retain every observation."""
|
|
112
|
+
|
|
113
|
+
__slots__ = ()
|
|
114
|
+
|
|
115
|
+
def validate(self, matches: AnnotationMatches, /) -> None:
|
|
116
|
+
_require_any_match(matches)
|
|
117
|
+
|
|
118
|
+
def apply(self, parsed: ParsedLevels, matches: AnnotationMatches, /) -> AnnotationResult:
|
|
119
|
+
return _apply(parsed, matches, selections=None)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
class RequireCompleteAnnotation:
|
|
123
|
+
"""Require every observation to match before attaching annotation."""
|
|
124
|
+
|
|
125
|
+
__slots__ = ()
|
|
126
|
+
|
|
127
|
+
def validate(self, matches: AnnotationMatches, /) -> None:
|
|
128
|
+
_require_any_match(matches)
|
|
129
|
+
incomplete = {
|
|
130
|
+
name: match.coverage
|
|
131
|
+
for name, match in matches.levels.items()
|
|
132
|
+
if match.coverage.quant_only_count
|
|
133
|
+
}
|
|
134
|
+
if incomplete:
|
|
135
|
+
details = "; ".join(
|
|
136
|
+
f"{name}: {coverage.quant_only_count} unmatched "
|
|
137
|
+
f"{list(coverage.quant_only_examples)}, near_misses={dict(coverage.near_misses)}"
|
|
138
|
+
for name, coverage in incomplete.items()
|
|
139
|
+
)
|
|
140
|
+
raise AnnotationError(f"complete sample annotation required; {details}")
|
|
141
|
+
|
|
142
|
+
def apply(self, parsed: ParsedLevels, matches: AnnotationMatches, /) -> AnnotationResult:
|
|
143
|
+
return _apply(parsed, matches, selections=None)
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
@dataclass(frozen=True, slots=True)
|
|
147
|
+
class SelectAnnotatedObservations:
|
|
148
|
+
"""Attach annotation and consistently retain only selected observations."""
|
|
149
|
+
|
|
150
|
+
selection: ObservationSelection
|
|
151
|
+
|
|
152
|
+
def validate(self, matches: AnnotationMatches, /) -> None:
|
|
153
|
+
_require_any_match(matches)
|
|
154
|
+
for match in matches.levels.values():
|
|
155
|
+
self.selection.validate(match)
|
|
156
|
+
|
|
157
|
+
def apply(self, parsed: ParsedLevels, matches: AnnotationMatches, /) -> AnnotationResult:
|
|
158
|
+
selections: dict[ParsedLevelName, pl.Series] = {
|
|
159
|
+
name: self.selection.selected_rows(match) for name, match in matches.levels.items()
|
|
160
|
+
}
|
|
161
|
+
return _apply(parsed, matches, selections=selections)
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def record_annotation_provenance(
|
|
165
|
+
result: AnnotationResult,
|
|
166
|
+
convention: str,
|
|
167
|
+
origin: AnnotationOrigin,
|
|
168
|
+
/,
|
|
169
|
+
*,
|
|
170
|
+
metadata: Mapping[str, JsonValue] | None = None,
|
|
171
|
+
) -> AnnotationResult:
|
|
172
|
+
"""Record tool-owned source provenance and each level's annotation report."""
|
|
173
|
+
if convention in {"parse", "roles", "storage"}:
|
|
174
|
+
raise AnnotationError(f"annotation convention uses reserved APB section {convention!r}")
|
|
175
|
+
record: dict[str, JsonValue] = dict(metadata or {})
|
|
176
|
+
record["schema_version"] = "2"
|
|
177
|
+
if "source" not in record:
|
|
178
|
+
record["source"] = (
|
|
179
|
+
{"path": str(origin.path)} if isinstance(origin, AnnotationFileOrigin) else None
|
|
180
|
+
)
|
|
181
|
+
root_tool = _metadata_section(result.parsed.metadata, convention)
|
|
182
|
+
provenance = _metadata_section(root_tool, "provenance")
|
|
183
|
+
provenance["annotation"] = record
|
|
184
|
+
for name, level in result.parsed.levels.items():
|
|
185
|
+
report = _report_json(result.reports[name])
|
|
186
|
+
tool = _metadata_section(level.metadata, convention)
|
|
187
|
+
tool["annotation"] = report
|
|
188
|
+
return result
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def _metadata_section(metadata: dict[str, JsonValue], name: str) -> dict[str, JsonValue]:
|
|
192
|
+
value = metadata.setdefault(name, {})
|
|
193
|
+
if not isinstance(value, dict):
|
|
194
|
+
raise AnnotationError(f"APB metadata section {name!r} must be an object")
|
|
195
|
+
return value
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def _apply(
|
|
199
|
+
parsed: ParsedLevels,
|
|
200
|
+
matches: AnnotationMatches,
|
|
201
|
+
selections: Mapping[ParsedLevelName, pl.Series] | None,
|
|
202
|
+
) -> AnnotationResult:
|
|
203
|
+
levels: dict[ParsedLevelName, ParsedLevel] = {}
|
|
204
|
+
reports: dict[ParsedLevelName, LevelAnnotationReport] = {}
|
|
205
|
+
for name, level in parsed.levels.items():
|
|
206
|
+
match = matches.levels[name]
|
|
207
|
+
selected = (
|
|
208
|
+
pl.Series("selected", [True] * level.obs.frame.height, dtype=pl.Boolean)
|
|
209
|
+
if selections is None
|
|
210
|
+
else selections[name]
|
|
211
|
+
)
|
|
212
|
+
levels[name] = _annotated_level(level, match, selected)
|
|
213
|
+
reports[name] = LevelAnnotationReport(
|
|
214
|
+
coverage=match.coverage,
|
|
215
|
+
corrections=match.corrections,
|
|
216
|
+
columns_added=tuple(match.aligned.columns),
|
|
217
|
+
)
|
|
218
|
+
return AnnotationResult(
|
|
219
|
+
parsed=ParsedLevels(
|
|
220
|
+
levels=levels,
|
|
221
|
+
uns=dict(parsed.uns),
|
|
222
|
+
metadata=dict(parsed.metadata),
|
|
223
|
+
annotation_tables=dict(parsed.annotation_tables),
|
|
224
|
+
feature_relations=dict(parsed.feature_relations),
|
|
225
|
+
),
|
|
226
|
+
reports=reports,
|
|
227
|
+
)
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def _require_any_match(matches: AnnotationMatches) -> None:
|
|
231
|
+
empty = [
|
|
232
|
+
name
|
|
233
|
+
for name, match in matches.levels.items()
|
|
234
|
+
if match.coverage.matched_observation_count == 0
|
|
235
|
+
]
|
|
236
|
+
if empty:
|
|
237
|
+
raise AnnotationError(
|
|
238
|
+
f"sample annotation matched no observations for level(s) {empty}; "
|
|
239
|
+
"no dataset-bound annotation was constructed"
|
|
240
|
+
)
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def _annotated_level(
|
|
244
|
+
level: ParsedLevel,
|
|
245
|
+
match: LevelAnnotationMatch,
|
|
246
|
+
selected: pl.Series,
|
|
247
|
+
) -> ParsedLevel:
|
|
248
|
+
annotated_obs = level.obs.frame.hstack(match.aligned.get_columns()).filter(selected)
|
|
249
|
+
kept = [index for index, value in enumerate(selected) if value]
|
|
250
|
+
return replace(
|
|
251
|
+
level,
|
|
252
|
+
obs=ObsFinal(frame=annotated_obs, key_columns=level.obs.key_columns),
|
|
253
|
+
uns=dict(level.uns),
|
|
254
|
+
layers={name: _subset_layer(layer, kept) for name, layer in level.layers.items()},
|
|
255
|
+
obsm={name: frame.filter(selected) for name, frame in level.obsm.items()},
|
|
256
|
+
varm=dict(level.varm),
|
|
257
|
+
obsp={name: _subset_pairwise(frame, kept) for name, frame in level.obsp.items()},
|
|
258
|
+
varp=dict(level.varp),
|
|
259
|
+
metadata=dict(level.metadata),
|
|
260
|
+
)
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def _subset_layer(layer: FinalLayerTable, kept: list[int]) -> FinalLayerTable:
|
|
264
|
+
value_columns = layer.values.columns
|
|
265
|
+
selected_names = [value_columns[index] for index in kept]
|
|
266
|
+
values = layer.values.select(selected_names)
|
|
267
|
+
replacement = observation_labels(len(kept), ())
|
|
268
|
+
values = values.rename(dict(zip(selected_names, replacement, strict=True)))
|
|
269
|
+
return replace(layer, values=values)
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def _subset_pairwise(frame: pl.DataFrame, kept: list[int]) -> pl.DataFrame:
|
|
273
|
+
mapping = {old: new for new, old in enumerate(kept)}
|
|
274
|
+
if not mapping or frame.is_empty():
|
|
275
|
+
return frame.head(0)
|
|
276
|
+
return frame.filter(pl.col("row").is_in(kept) & pl.col("column").is_in(kept)).with_columns(
|
|
277
|
+
pl.col("row").replace_strict(mapping),
|
|
278
|
+
pl.col("column").replace_strict(mapping),
|
|
279
|
+
)
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def _report_json(report: LevelAnnotationReport) -> dict[str, JsonValue]:
|
|
283
|
+
coverage = report.coverage
|
|
284
|
+
near_misses: dict[str, JsonValue] = {
|
|
285
|
+
key: dict(candidates) for key, candidates in coverage.near_misses.items()
|
|
286
|
+
}
|
|
287
|
+
corrections: dict[str, JsonValue] = {
|
|
288
|
+
correction.observed: {
|
|
289
|
+
"observed": correction.observed,
|
|
290
|
+
"expected": correction.expected,
|
|
291
|
+
"score": correction.score,
|
|
292
|
+
}
|
|
293
|
+
for correction in report.corrections
|
|
294
|
+
}
|
|
295
|
+
return {
|
|
296
|
+
"observation_count": coverage.observation_count,
|
|
297
|
+
"annotation_count": coverage.annotation_count,
|
|
298
|
+
"matched_observation_count": coverage.matched_observation_count,
|
|
299
|
+
"quant_only_count": coverage.quant_only_count,
|
|
300
|
+
"annotation_only_count": coverage.annotation_only_count,
|
|
301
|
+
"quant_only_examples": list(coverage.quant_only_examples),
|
|
302
|
+
"annotation_only_examples": list(coverage.annotation_only_examples),
|
|
303
|
+
"near_misses": near_misses,
|
|
304
|
+
"corrections": corrections,
|
|
305
|
+
"columns_added": list(report.columns_added),
|
|
306
|
+
}
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
"""Compile one delimited annotation source into a source-bound parser."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Literal
|
|
7
|
+
|
|
8
|
+
import polars as pl
|
|
9
|
+
|
|
10
|
+
from apb2.annotation.application.policies import (
|
|
11
|
+
AllAnnotationSelections,
|
|
12
|
+
AnnotationApplication,
|
|
13
|
+
BooleanAnnotationSelection,
|
|
14
|
+
KeepUnmatchedAnnotation,
|
|
15
|
+
MatchedAnnotationSelection,
|
|
16
|
+
ObservationSelection,
|
|
17
|
+
RequireCompleteAnnotation,
|
|
18
|
+
SelectAnnotatedObservations,
|
|
19
|
+
)
|
|
20
|
+
from apb2.annotation.contracts import AnnotationParser
|
|
21
|
+
from apb2.annotation.data.model import AnnotationError
|
|
22
|
+
from apb2.annotation.prolfquapp import ProlfquappAnnotationParser, prolfquapp_signature
|
|
23
|
+
from apb2.annotation.sdrf import SdrfAnnotationParser, SdrfSource, sdrf_signature
|
|
24
|
+
from apb2.annotation.source.load import load_annotation_file, load_annotation_frame
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class AnnotationCompiler:
|
|
28
|
+
"""User-configured parser for SDRF and generic delimited observation annotations."""
|
|
29
|
+
|
|
30
|
+
__slots__ = ("_application",)
|
|
31
|
+
|
|
32
|
+
def __init__(
|
|
33
|
+
self,
|
|
34
|
+
unmatched: Literal["keep", "error", "drop"] = "keep",
|
|
35
|
+
include: str | None = None,
|
|
36
|
+
) -> None:
|
|
37
|
+
"""Choose what happens to observations the annotation does not cover.
|
|
38
|
+
|
|
39
|
+
Args:
|
|
40
|
+
unmatched: ``keep`` them with null metadata, raise (``error``), or ``drop`` them.
|
|
41
|
+
include: A Boolean annotation column that further selects observations; requires
|
|
42
|
+
``unmatched="drop"``.
|
|
43
|
+
|
|
44
|
+
Raises:
|
|
45
|
+
AnnotationError: ``include`` is given without ``unmatched="drop"``.
|
|
46
|
+
"""
|
|
47
|
+
self._application = _application(unmatched, include)
|
|
48
|
+
|
|
49
|
+
def compile(self, source: Path | pl.DataFrame) -> AnnotationParser:
|
|
50
|
+
"""Load once, verify the tabular convention, and return its bound parser.
|
|
51
|
+
|
|
52
|
+
SDRF headers take precedence; other tables must carry a prolfquapp observation key.
|
|
53
|
+
"""
|
|
54
|
+
loaded = (
|
|
55
|
+
load_annotation_file(source)
|
|
56
|
+
if isinstance(source, Path)
|
|
57
|
+
else load_annotation_frame(source)
|
|
58
|
+
)
|
|
59
|
+
if sdrf_signature(loaded):
|
|
60
|
+
return SdrfAnnotationParser(source=SdrfSource(loaded), application=self._application)
|
|
61
|
+
if not prolfquapp_signature(loaded):
|
|
62
|
+
raise AnnotationError("annotation table has no supported observation key")
|
|
63
|
+
return ProlfquappAnnotationParser(source=loaded, application=self._application)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _application(
|
|
67
|
+
unmatched: Literal["keep", "error", "drop"], include: str | None
|
|
68
|
+
) -> AnnotationApplication:
|
|
69
|
+
if unmatched == "drop":
|
|
70
|
+
selections: list[ObservationSelection] = [MatchedAnnotationSelection()]
|
|
71
|
+
if include is not None:
|
|
72
|
+
selections.append(BooleanAnnotationSelection(include))
|
|
73
|
+
return SelectAnnotatedObservations(AllAnnotationSelections(tuple(selections)))
|
|
74
|
+
if include is not None:
|
|
75
|
+
raise AnnotationError("include requires unmatched='drop'")
|
|
76
|
+
return KeepUnmatchedAnnotation() if unmatched == "keep" else RequireCompleteAnnotation()
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""Small public capabilities returned by sample-annotation compilation."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Protocol
|
|
6
|
+
|
|
7
|
+
from apb2.annotation.data.model import AnnotationMatches, AnnotationResult
|
|
8
|
+
from apb2.parserV2.parse_quant.data.parsed import ParsedLevels
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class Annotation(Protocol):
|
|
12
|
+
"""A validated sample annotation bound to exactly one parsed dataset."""
|
|
13
|
+
|
|
14
|
+
@property
|
|
15
|
+
def matches(self) -> AnnotationMatches:
|
|
16
|
+
"""Return completed per-level matching evidence."""
|
|
17
|
+
...
|
|
18
|
+
|
|
19
|
+
def annotate(self) -> AnnotationResult:
|
|
20
|
+
"""Apply this annotation to the dataset against which it was validated."""
|
|
21
|
+
...
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class AnnotationParser(Protocol):
|
|
25
|
+
"""A convention parser bound to one already loaded annotation source."""
|
|
26
|
+
|
|
27
|
+
def parse(self, parsed: ParsedLevels, /) -> Annotation:
|
|
28
|
+
"""Validate and match the source, then construct a dataset-bound annotation."""
|
|
29
|
+
...
|
|
File without changes
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
"""Innermost values exchanged by the sample-annotation workflow."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Mapping
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
import polars as pl
|
|
10
|
+
|
|
11
|
+
from apb2.parserV2.parse_quant.data.parsed import ParsedLevelName, ParsedLevels
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class AnnotationError(ValueError):
|
|
15
|
+
"""A sample annotation cannot be recognized, parsed, matched, or applied."""
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@dataclass(frozen=True, slots=True)
|
|
19
|
+
class AnnotationFileOrigin:
|
|
20
|
+
"""The resolved file from which an annotation was loaded."""
|
|
21
|
+
|
|
22
|
+
path: Path
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass(frozen=True, slots=True)
|
|
26
|
+
class InMemoryAnnotationOrigin:
|
|
27
|
+
"""Marker for a programmatically supplied Polars annotation."""
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
type AnnotationOrigin = AnnotationFileOrigin | InMemoryAnnotationOrigin
|
|
31
|
+
|
|
32
|
+
IN_MEMORY_ANNOTATION = InMemoryAnnotationOrigin()
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass(frozen=True, slots=True)
|
|
36
|
+
class LoadedAnnotationSource:
|
|
37
|
+
"""One physically decoded tabular annotation source.
|
|
38
|
+
|
|
39
|
+
``headers`` holds the verbatim header text aligned with ``frame.columns``. Repeated headers
|
|
40
|
+
keep their text there while the frame columns stay unique.
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
frame: pl.DataFrame
|
|
44
|
+
origin: AnnotationOrigin
|
|
45
|
+
headers: tuple[str, ...]
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
@dataclass(frozen=True, slots=True)
|
|
49
|
+
class AnnotationTable:
|
|
50
|
+
"""Validated sample rows and every identifier column accepted for matching."""
|
|
51
|
+
|
|
52
|
+
frame: pl.DataFrame
|
|
53
|
+
key_columns: tuple[str, ...]
|
|
54
|
+
alias_columns: tuple[str, ...]
|
|
55
|
+
origin: AnnotationOrigin
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@dataclass(frozen=True, slots=True)
|
|
59
|
+
class AnnotationCoverage:
|
|
60
|
+
"""Counts and bounded mismatch evidence for one quantification level."""
|
|
61
|
+
|
|
62
|
+
observation_count: int
|
|
63
|
+
annotation_count: int
|
|
64
|
+
matched_observation_count: int
|
|
65
|
+
quant_only_count: int
|
|
66
|
+
annotation_only_count: int
|
|
67
|
+
quant_only_examples: tuple[str, ...]
|
|
68
|
+
annotation_only_examples: tuple[str, ...]
|
|
69
|
+
near_misses: Mapping[str, tuple[tuple[str, float], ...]]
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
@dataclass(frozen=True, slots=True)
|
|
73
|
+
class KeyCorrection:
|
|
74
|
+
"""One accepted fuzzy observation-to-annotation identifier correction."""
|
|
75
|
+
|
|
76
|
+
observed: str
|
|
77
|
+
expected: str
|
|
78
|
+
score: float
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
@dataclass(frozen=True, slots=True)
|
|
82
|
+
class LevelAnnotationMatch:
|
|
83
|
+
"""Annotation aligned to one observation axis plus its matching evidence."""
|
|
84
|
+
|
|
85
|
+
aligned: pl.DataFrame
|
|
86
|
+
matched_rows: pl.Series
|
|
87
|
+
coverage: AnnotationCoverage
|
|
88
|
+
corrections: tuple[KeyCorrection, ...]
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
@dataclass(frozen=True, slots=True)
|
|
92
|
+
class AnnotationMatches:
|
|
93
|
+
"""Independent matching evidence for every parsed quantification level."""
|
|
94
|
+
|
|
95
|
+
levels: Mapping[ParsedLevelName, LevelAnnotationMatch]
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
@dataclass(frozen=True, slots=True)
|
|
99
|
+
class LevelAnnotationReport:
|
|
100
|
+
"""Persistable evidence produced by applying one level's annotation."""
|
|
101
|
+
|
|
102
|
+
coverage: AnnotationCoverage
|
|
103
|
+
corrections: tuple[KeyCorrection, ...]
|
|
104
|
+
columns_added: tuple[str, ...]
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
@dataclass(frozen=True, slots=True)
|
|
108
|
+
class AnnotationResult:
|
|
109
|
+
"""A newly annotated storage-neutral result and its per-level reports."""
|
|
110
|
+
|
|
111
|
+
parsed: ParsedLevels
|
|
112
|
+
reports: Mapping[ParsedLevelName, LevelAnnotationReport]
|
|
File without changes
|