apb-fasta 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
apb_fasta/__init__.py ADDED
File without changes
apb_fasta/api.py ADDED
@@ -0,0 +1,164 @@
1
+ """Public in-memory FASTA annotation API."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Sequence
6
+ from dataclasses import dataclass
7
+ from importlib.metadata import version
8
+ from pathlib import Path
9
+
10
+ import polars as pl
11
+ from apb2.api import ParsedLevels
12
+ from protein_fasta.api import ProteinDatabase, ProteinFormat
13
+ from prozor.api import resolve_backend
14
+
15
+ from apb_fasta.calculation.matching import match_peptide_levels
16
+ from apb_fasta.calculation.peptide_properties import peptide_level_properties
17
+ from apb_fasta.calculation.protein_groups import match_protein_groups
18
+ from apb_fasta.calculation.results import FastaAnnotationReports
19
+ from apb_fasta.configuration import (
20
+ DEFAULT_FASTA_ANNOTATION_PARAMETERS,
21
+ FastaAnnotationParameters,
22
+ )
23
+ from apb_fasta.errors import FastaAnnotationError
24
+ from apb_fasta.integration import (
25
+ apply_peptide_matches,
26
+ apply_peptide_properties,
27
+ apply_protein_group_match,
28
+ peptide_inputs,
29
+ protein_frame_metadata,
30
+ protein_group_input,
31
+ validate_protein_frame,
32
+ )
33
+
34
+ _NO_PEPTIDE_LEVEL = (
35
+ "result contains no peptide-derived level with canonical ProForma_peptide values"
36
+ )
37
+
38
+
39
+ @dataclass(frozen=True, slots=True)
40
+ class FastaAnnotationResult:
41
+ """A replacement APB2 result and its FASTA reports."""
42
+
43
+ parsed: ParsedLevels
44
+ reports: FastaAnnotationReports
45
+
46
+
47
+ class FastaAnnotator:
48
+ """Bind one protein database and its FASTA annotation behavior."""
49
+
50
+ __slots__ = ("_parameters", "_proteins")
51
+
52
+ def __init__(
53
+ self,
54
+ proteins: pl.DataFrame,
55
+ parameters: FastaAnnotationParameters = DEFAULT_FASTA_ANNOTATION_PARAMETERS,
56
+ ) -> None:
57
+ """Validate and bind a reusable protein database and configuration."""
58
+ validate_protein_frame(proteins)
59
+ self._proteins = proteins
60
+ self._parameters = parameters
61
+
62
+ @classmethod
63
+ def read(
64
+ cls,
65
+ fasta: Sequence[Path],
66
+ formats: Sequence[str] = ("uniprotkb", "refseq"),
67
+ parameters: FastaAnnotationParameters = DEFAULT_FASTA_ANNOTATION_PARAMETERS,
68
+ ) -> FastaAnnotator:
69
+ """Read FASTA files, or the protein-fasta database Parquet written from them."""
70
+ if not fasta:
71
+ raise ValueError("at least one FASTA path is required")
72
+ database = ProteinDatabase(*(ProteinFormat(name) for name in formats))
73
+ return cls(database.parse(tuple(fasta)), parameters)
74
+
75
+ @property
76
+ def proteins(self) -> pl.DataFrame:
77
+ """The bound protein table, one row per FASTA entry in file order."""
78
+ return self._proteins
79
+
80
+ def verify_peptides(self, parsed: ParsedLevels) -> FastaAnnotationResult:
81
+ """Verify every canonical stripped peptide against the protein sequences."""
82
+ inputs = peptide_inputs(parsed)
83
+ if not inputs:
84
+ raise FastaAnnotationError(_NO_PEPTIDE_LEVEL)
85
+ peptide_levels = match_peptide_levels(
86
+ inputs,
87
+ self._proteins,
88
+ backend=self._parameters.matcher_backend,
89
+ il_equivalent=self._parameters.il_equivalent,
90
+ protein_group_separator=self._parameters.protein_group_separator,
91
+ )
92
+ replacement, reports = apply_peptide_matches(
93
+ parsed,
94
+ peptide_levels,
95
+ requested_backend=self._parameters.matcher_backend,
96
+ resolved_backend=resolve_backend(self._parameters.matcher_backend),
97
+ il_equivalent=self._parameters.il_equivalent,
98
+ protein_metadata=protein_frame_metadata(self._proteins),
99
+ )
100
+ return FastaAnnotationResult(parsed=replacement, reports=reports)
101
+
102
+ def merge_annotations(self, parsed: ParsedLevels) -> FastaAnnotationResult:
103
+ """Merge FASTA annotations for every reported protein-group member."""
104
+ protein_input = protein_group_input(parsed)
105
+ if protein_input is None:
106
+ raise FastaAnnotationError("result contains no protein level to annotate")
107
+ protein_groups = match_protein_groups(
108
+ protein_input,
109
+ self._proteins,
110
+ separator=self._parameters.protein_group_separator,
111
+ )
112
+ replacement, reports = apply_protein_group_match(
113
+ parsed,
114
+ protein_groups,
115
+ protein_group_separator=self._parameters.protein_group_separator,
116
+ protein_metadata=protein_frame_metadata(self._proteins),
117
+ )
118
+ return FastaAnnotationResult(parsed=replacement, reports=reports)
119
+
120
+ def annotate(self, parsed: ParsedLevels) -> FastaAnnotationResult:
121
+ """Verify peptides and then merge protein annotations in memory."""
122
+ verified = self.verify_peptides(parsed)
123
+ annotated = self.merge_annotations(verified.parsed)
124
+ return FastaAnnotationResult(
125
+ parsed=annotated.parsed,
126
+ reports=FastaAnnotationReports(
127
+ peptide_levels=verified.reports.peptide_levels,
128
+ protein_groups=annotated.reports.protein_groups,
129
+ ),
130
+ )
131
+
132
+
133
+ def add_peptide_properties(parsed: ParsedLevels) -> ParsedLevels:
134
+ """Attach sequence-derived properties to every peptide-derived level.
135
+
136
+ Writes a feature-aligned ``varm["peptide_properties"]`` computed by
137
+ ``protein_fasta.peptide_frame.peptide_property_frame`` from each level's ``ProForma_peptide``.
138
+ No protein database is needed. A feature without a sequence, or with a residue outside the 20
139
+ standard amino acids, has null properties.
140
+
141
+ Raises:
142
+ FastaAnnotationError: If the result has no peptide-derived level, already contains the
143
+ output, or already records the operation.
144
+ ValueError: If a sequence is not stripped upper-case letters.
145
+ """
146
+ inputs = peptide_inputs(parsed)
147
+ if not inputs:
148
+ raise FastaAnnotationError(_NO_PEPTIDE_LEVEL)
149
+ properties = {
150
+ name: peptide_level_properties(level.frame, level.sequence_column)
151
+ for name, level in inputs.items()
152
+ }
153
+ return apply_peptide_properties(
154
+ parsed, properties, protein_fasta_version=version("protein-fasta")
155
+ )
156
+
157
+
158
+ __all__ = [
159
+ "FastaAnnotationParameters",
160
+ "FastaAnnotationReports",
161
+ "FastaAnnotationResult",
162
+ "FastaAnnotator",
163
+ "add_peptide_properties",
164
+ ]
File without changes
@@ -0,0 +1,209 @@
1
+ """Peptide-to-protein matching and feature-aligned summaries."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Mapping
6
+ from dataclasses import dataclass
7
+
8
+ import polars as pl
9
+ from prozor.api import annotate_peptides
10
+
11
+ from apb_fasta.calculation.results import PeptideCoverage, PeptideLevelMatch
12
+
13
+
14
+ @dataclass(frozen=True, slots=True)
15
+ class PeptideLevelInput:
16
+ """The columns needed to validate one peptide-derived feature level."""
17
+
18
+ frame: pl.DataFrame
19
+ sequence_column: str
20
+ accession_column: str | None
21
+
22
+
23
+ def match_peptide_levels(
24
+ levels: Mapping[str, PeptideLevelInput],
25
+ proteins: pl.DataFrame,
26
+ /,
27
+ *,
28
+ backend: str,
29
+ il_equivalent: bool,
30
+ protein_group_separator: str,
31
+ ) -> dict[str, PeptideLevelMatch]:
32
+ """Match distinct normalized peptides once and summarize every source feature."""
33
+ if not levels:
34
+ return {}
35
+ prepared = {name: _prepare_features(level, il_equivalent) for name, level in levels.items()}
36
+ peptides = (
37
+ pl.concat([frame.select("peptide") for frame in prepared.values()])
38
+ .drop_nulls()
39
+ .unique(maintain_order=True)
40
+ )
41
+ sequence = _text_column(proteins, "sequence")
42
+ if il_equivalent:
43
+ sequence = sequence.str.replace_all("I", "L", literal=True)
44
+ sequences = proteins.select(sequence).to_series()
45
+ if sequences.null_count():
46
+ raise ValueError("protein frame column 'sequence' contains a non-text value")
47
+ matched = annotate_peptides(
48
+ peptides.get_column("peptide").to_list(),
49
+ dict(zip(map(str, range(sequences.len())), sequences.to_list(), strict=True)),
50
+ backend=backend,
51
+ )
52
+ occurrences = pl.DataFrame(
53
+ [(match.peptide, int(match.protein_id)) for match in matched],
54
+ schema={"peptide": pl.String, "record": pl.UInt32},
55
+ orient="row",
56
+ )
57
+ site_counts = occurrences.group_by("peptide").agg(
58
+ pl.len().cast(pl.UInt64).alias("fasta_match_site_count")
59
+ )
60
+ columns = set(proteins.columns)
61
+ indexed = proteins.select(
62
+ id=_text_column(proteins, "id"),
63
+ organism=_text_column(
64
+ proteins, "organism_mnemonic" if "organism_mnemonic" in columns else None
65
+ ),
66
+ is_contaminant=pl.col("is_contaminant") if "is_contaminant" in columns else pl.lit(False),
67
+ ).with_row_index("record")
68
+ distinct = occurrences.unique(maintain_order=True).join(
69
+ indexed, on="record", how="left", maintain_order="left"
70
+ )
71
+ if distinct.get_column("id").null_count():
72
+ raise ValueError("protein frame column 'id' contains a non-text value")
73
+ matches = distinct.group_by("peptide", maintain_order=True).agg(
74
+ pl.col("record").alias("matched_records"),
75
+ pl.len().cast(pl.UInt64).alias("fasta_matching_protein_count"),
76
+ pl.col("id").str.join(";").alias("fasta_matching_protein_ids"),
77
+ fasta_matching_organisms=pl.col("organism").drop_nulls().unique().sort().str.join(";"),
78
+ fasta_matches_contaminant=pl.col("is_contaminant").any(),
79
+ )
80
+ matches = matches.join(site_counts, on="peptide", how="left")
81
+ assignments = _reported_assignments(prepared, proteins, protein_group_separator)
82
+ return {
83
+ name: _level_match(frame, matches, assignments, levels[name].accession_column is not None)
84
+ for name, frame in prepared.items()
85
+ }
86
+
87
+
88
+ def _text_column(frame: pl.DataFrame, column: str | None) -> pl.Expr:
89
+ if column is None:
90
+ return pl.lit(None, dtype=pl.String)
91
+ dtype = frame.schema[column]
92
+ if isinstance(dtype, (pl.String, pl.Categorical, pl.Enum)):
93
+ return pl.col(column).cast(pl.String)
94
+ if dtype == pl.Object:
95
+ # Only foreign mixed-object columns need scalar type checking.
96
+ return pl.col(column).map_elements(_optional_text, return_dtype=pl.String)
97
+ return pl.lit(None, dtype=pl.String)
98
+
99
+
100
+ def _optional_text(value: object) -> str | None:
101
+ return value if isinstance(value, str) else None
102
+
103
+
104
+ def _prepare_features(level: PeptideLevelInput, il_equivalent: bool) -> pl.DataFrame:
105
+ sequence = _text_column(level.frame, level.sequence_column).str.strip_chars().str.to_uppercase()
106
+ if il_equivalent:
107
+ sequence = sequence.str.replace_all("I", "L", literal=True)
108
+ # with_columns keeps the source height even when both expressions are scalar nulls.
109
+ return level.frame.with_columns(
110
+ sequence.alias("peptide"),
111
+ _text_column(level.frame, level.accession_column).alias("assignment"),
112
+ ).select(
113
+ pl.when(pl.col("peptide") != "").then(pl.col("peptide")).alias("peptide"),
114
+ "assignment",
115
+ )
116
+
117
+
118
+ def _reported_assignments(
119
+ prepared: Mapping[str, pl.DataFrame], proteins: pl.DataFrame, separator: str
120
+ ) -> pl.DataFrame:
121
+ aliases = pl.concat(
122
+ [
123
+ proteins.with_columns(_text_column(proteins, column).alias("member"))
124
+ .select("member")
125
+ .with_row_index("record")
126
+ for column in ("id", "accession")
127
+ if column in proteins.columns
128
+ ]
129
+ )
130
+ aliases = (
131
+ aliases.filter(pl.col("member").is_not_null() & (pl.col("member") != ""))
132
+ .unique()
133
+ .group_by("member")
134
+ .agg("record")
135
+ )
136
+ assignments = (
137
+ pl.concat([frame.select("assignment") for frame in prepared.values()]).drop_nulls().unique()
138
+ )
139
+ return (
140
+ assignments.with_columns(
141
+ pl.col("assignment")
142
+ .str.split(separator)
143
+ .list.eval(pl.element().str.strip_chars())
144
+ .alias("member")
145
+ )
146
+ .explode("member", empty_as_null=True)
147
+ .filter(pl.col("member") != "")
148
+ .join(aliases, on="member", how="left")
149
+ .group_by("assignment")
150
+ .agg(
151
+ pl.len().cast(pl.UInt64).alias("reported_member_count"),
152
+ pl.col("record")
153
+ .is_not_null()
154
+ .sum()
155
+ .cast(pl.UInt64)
156
+ .alias("reported_members_in_fasta_count"),
157
+ pl.col("record")
158
+ .explode(empty_as_null=True)
159
+ .drop_nulls()
160
+ .unique()
161
+ .alias("reported_records"),
162
+ )
163
+ )
164
+
165
+
166
+ def _level_match(
167
+ features: pl.DataFrame,
168
+ matches: pl.DataFrame,
169
+ assignments: pl.DataFrame,
170
+ has_assignment: bool,
171
+ ) -> PeptideLevelMatch:
172
+ joined = features.join(matches, on="peptide", how="left", maintain_order="left").join(
173
+ assignments, on="assignment", how="left", maintain_order="left"
174
+ )
175
+ summary = joined.select(
176
+ (pl.col("fasta_matching_protein_count").fill_null(0) > 0).alias("peptide_in_fasta"),
177
+ pl.col("fasta_match_site_count").fill_null(0),
178
+ pl.col("fasta_matching_protein_count").fill_null(0),
179
+ pl.col("fasta_matching_protein_ids").fill_null(""),
180
+ pl.col("fasta_matching_organisms").fill_null(""),
181
+ pl.col("fasta_matches_contaminant").fill_null(False),
182
+ pl.col("reported_member_count").fill_null(0),
183
+ pl.col("reported_members_in_fasta_count").fill_null(0),
184
+ (
185
+ pl.col("matched_records")
186
+ .fill_null([])
187
+ .list.set_intersection(pl.col("reported_records").fill_null([]))
188
+ .list.len()
189
+ > 0
190
+ ).alias("peptide_in_reported_protein"),
191
+ )
192
+ if not has_assignment:
193
+ summary = summary.with_columns(
194
+ pl.lit(None, dtype=pl.UInt64).alias("reported_member_count"),
195
+ pl.lit(None, dtype=pl.UInt64).alias("reported_members_in_fasta_count"),
196
+ pl.lit(None, dtype=pl.Boolean).alias("peptide_in_reported_protein"),
197
+ )
198
+ unique = joined.select("peptide", "fasta_match_site_count").drop_nulls("peptide").unique()
199
+ matched_count = int(summary.get_column("peptide_in_fasta").sum() or 0)
200
+ return PeptideLevelMatch(
201
+ summary=summary,
202
+ coverage=PeptideCoverage(
203
+ feature_count=features.height,
204
+ unique_sequence_count=unique.height,
205
+ matched_feature_count=matched_count,
206
+ unmatched_feature_count=features.height - matched_count,
207
+ match_site_count=int(unique.get_column("fasta_match_site_count").sum() or 0),
208
+ ),
209
+ )
@@ -0,0 +1,25 @@
1
+ """Feature-aligned sequence properties of one peptide-derived level."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import polars as pl
6
+ from protein_fasta.api import peptide_property_frame
7
+
8
+
9
+ def peptide_level_properties(frame: pl.DataFrame, sequence_column: str, /) -> pl.DataFrame:
10
+ """Return one property row per feature of ``frame``, in feature order.
11
+
12
+ Each distinct sequence is computed once; a feature without a sequence has null properties.
13
+ """
14
+ features = frame.select(
15
+ pl.col(sequence_column)
16
+ .cast(pl.String)
17
+ .str.strip_chars()
18
+ .str.to_uppercase()
19
+ .alias("sequence")
20
+ ).select(pl.when(pl.col("sequence") != "").then(pl.col("sequence")).alias("sequence"))
21
+ sequences = features.get_column("sequence").drop_nulls().unique(maintain_order=True)
22
+ properties = peptide_property_frame(sequences.to_list())
23
+ return features.join(properties, on="sequence", how="left", maintain_order="left").drop(
24
+ "sequence"
25
+ )
@@ -0,0 +1,231 @@
1
+ """Protein-group expansion and FASTA annotation over Polars values."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections import defaultdict
6
+ from collections.abc import Mapping
7
+ from dataclasses import dataclass
8
+
9
+ import polars as pl
10
+
11
+ from apb_fasta.calculation.results import ProteinGroupCoverage, ProteinGroupMatch
12
+
13
+ _OPTIONAL_FASTA_COLUMNS = (
14
+ "is_decoy",
15
+ "is_contaminant",
16
+ "accession",
17
+ "protein_name",
18
+ "gene_name",
19
+ "organism_name",
20
+ "taxonomy_id",
21
+ "database",
22
+ "review_status",
23
+ "fasta_source_path",
24
+ "fasta_source_checksum",
25
+ "fasta_source_ordinal",
26
+ "fasta_record_ordinal",
27
+ )
28
+
29
+
30
+ @dataclass(frozen=True, slots=True)
31
+ class ProteinGroupInput:
32
+ """The protein feature axis and its declared FASTA accession column."""
33
+
34
+ frame: pl.DataFrame
35
+ key_columns: tuple[str, ...]
36
+ accession_column: str
37
+
38
+
39
+ def match_protein_groups(
40
+ source: ProteinGroupInput,
41
+ proteins: pl.DataFrame,
42
+ /,
43
+ *,
44
+ separator: str,
45
+ ) -> ProteinGroupMatch:
46
+ """Expand every reported member and retain every matching FASTA record."""
47
+ protein_rows = proteins.to_dicts()
48
+ aliases = _protein_aliases(protein_rows)
49
+ members_rows: list[dict[str, object]] = []
50
+ summaries: list[dict[str, object]] = []
51
+ relation_rows: list[dict[str, object]] = []
52
+ matched_members = 0
53
+ unmatched_members = 0
54
+ ambiguous_members = 0
55
+ member_count = 0
56
+ for group_index, group in enumerate(source.frame.to_dicts()):
57
+ raw_group = group.get(source.accession_column)
58
+ members = _members(raw_group, separator)
59
+ group_matched = 0
60
+ group_unmatched = 0
61
+ group_ambiguous = 0
62
+ descriptions: list[str] = []
63
+ gene_names: list[str] = []
64
+ organism_names: list[str] = []
65
+ for member_ordinal, member in enumerate(members):
66
+ member_count += 1
67
+ matches = aliases.get(member, ())
68
+ if not matches:
69
+ unmatched_members += 1
70
+ group_unmatched += 1
71
+ members_rows.append(
72
+ _member_row(
73
+ source,
74
+ group,
75
+ raw_group,
76
+ member,
77
+ member_ordinal,
78
+ 0,
79
+ "unmatched",
80
+ None,
81
+ )
82
+ )
83
+ relation_rows.append(
84
+ {"row": len(members_rows) - 1, "column": group_index, "value": 1.0}
85
+ )
86
+ continue
87
+ matched_members += 1
88
+ group_matched += 1
89
+ status = "ambiguous" if len(matches) > 1 else "matched"
90
+ if len(matches) > 1:
91
+ ambiguous_members += 1
92
+ group_ambiguous += 1
93
+ for match_ordinal, protein_index in enumerate(matches):
94
+ protein = protein_rows[protein_index]
95
+ members_rows.append(
96
+ _member_row(
97
+ source,
98
+ group,
99
+ raw_group,
100
+ member,
101
+ member_ordinal,
102
+ match_ordinal,
103
+ status,
104
+ protein,
105
+ )
106
+ )
107
+ relation_rows.append(
108
+ {"row": len(members_rows) - 1, "column": group_index, "value": 1.0}
109
+ )
110
+ _append_text(descriptions, protein.get("description"))
111
+ _append_text(gene_names, protein.get("gene_name"))
112
+ _append_text(organism_names, protein.get("organism_name"))
113
+ summaries.append(
114
+ {
115
+ "reported_member_count": len(members),
116
+ "matched_member_count": group_matched,
117
+ "unmatched_member_count": group_unmatched,
118
+ "ambiguous_member_count": group_ambiguous,
119
+ "all_members_in_fasta": bool(members) and group_unmatched == 0,
120
+ "any_member_in_fasta": group_matched > 0,
121
+ "fasta_descriptions": ";".join(descriptions),
122
+ "fasta_gene_names": ";".join(gene_names),
123
+ "fasta_organism_names": ";".join(organism_names),
124
+ }
125
+ )
126
+ member_keys = ("source_level", *source.key_columns, "member_ordinal", "match_ordinal")
127
+ return ProteinGroupMatch(
128
+ members=pl.DataFrame(members_rows, schema=_member_schema(source, proteins)),
129
+ member_key_columns=member_keys,
130
+ summary=pl.DataFrame(summaries, schema=_summary_schema()),
131
+ relation=pl.DataFrame(
132
+ relation_rows,
133
+ schema={"row": pl.Int64, "column": pl.Int64, "value": pl.Float64},
134
+ ),
135
+ coverage=ProteinGroupCoverage(
136
+ group_count=source.frame.height,
137
+ member_count=member_count,
138
+ matched_member_count=matched_members,
139
+ unmatched_member_count=unmatched_members,
140
+ ambiguous_member_count=ambiguous_members,
141
+ ),
142
+ )
143
+
144
+
145
+ def _member_row(
146
+ source: ProteinGroupInput,
147
+ group: Mapping[str, object],
148
+ raw_group: object,
149
+ member: str,
150
+ member_ordinal: int,
151
+ match_ordinal: int,
152
+ status: str,
153
+ protein: Mapping[str, object] | None,
154
+ ) -> dict[str, object]:
155
+ sequence = None if protein is None else protein.get("sequence")
156
+ row: dict[str, object] = {
157
+ "source_level": "protein",
158
+ **{name: group[name] for name in source.key_columns},
159
+ "member_ordinal": member_ordinal,
160
+ "match_ordinal": match_ordinal,
161
+ "protein_group": raw_group,
162
+ "protein_member": member,
163
+ "normalized_matching_key": member,
164
+ "match_status": status,
165
+ "fasta_id": None if protein is None else protein.get("id"),
166
+ "fasta_description": None if protein is None else protein.get("description"),
167
+ "fasta_sequence_length": (None if not isinstance(sequence, str) else len(sequence)),
168
+ }
169
+ for name in _OPTIONAL_FASTA_COLUMNS:
170
+ row[name] = None if protein is None else protein.get(name)
171
+ return row
172
+
173
+
174
+ def _member_schema(
175
+ source: ProteinGroupInput,
176
+ proteins: pl.DataFrame,
177
+ ) -> dict[str, pl.DataType | type[pl.DataType]]:
178
+ schema: dict[str, pl.DataType | type[pl.DataType]] = {
179
+ "source_level": pl.String,
180
+ **{name: source.frame.schema[name] for name in source.key_columns},
181
+ "member_ordinal": pl.Int64,
182
+ "match_ordinal": pl.Int64,
183
+ "protein_group": source.frame.schema[source.accession_column],
184
+ "protein_member": pl.String,
185
+ "normalized_matching_key": pl.String,
186
+ "match_status": pl.String,
187
+ "fasta_id": pl.String,
188
+ "fasta_description": pl.String,
189
+ "fasta_sequence_length": pl.UInt64,
190
+ }
191
+ for name in _OPTIONAL_FASTA_COLUMNS:
192
+ schema[name] = proteins.schema.get(name, pl.Null)
193
+ return schema
194
+
195
+
196
+ def _summary_schema() -> dict[str, type[pl.DataType]]:
197
+ return {
198
+ "reported_member_count": pl.UInt64,
199
+ "matched_member_count": pl.UInt64,
200
+ "unmatched_member_count": pl.UInt64,
201
+ "ambiguous_member_count": pl.UInt64,
202
+ "all_members_in_fasta": pl.Boolean,
203
+ "any_member_in_fasta": pl.Boolean,
204
+ "fasta_descriptions": pl.String,
205
+ "fasta_gene_names": pl.String,
206
+ "fasta_organism_names": pl.String,
207
+ }
208
+
209
+
210
+ def _protein_aliases(rows: list[dict[str, object]]) -> dict[str, tuple[int, ...]]:
211
+ collected: dict[str, list[int]] = defaultdict(list)
212
+ for index, row in enumerate(rows):
213
+ aliases = {
214
+ value
215
+ for name in ("id", "accession")
216
+ if isinstance((value := row.get(name)), str) and value
217
+ }
218
+ for alias in aliases:
219
+ collected[alias].append(index)
220
+ return {name: tuple(indices) for name, indices in collected.items()}
221
+
222
+
223
+ def _members(value: object, separator: str) -> tuple[str, ...]:
224
+ if not isinstance(value, str):
225
+ return ()
226
+ return tuple(token.strip() for token in value.split(separator) if token.strip())
227
+
228
+
229
+ def _append_text(values: list[str], value: object) -> None:
230
+ if isinstance(value, str) and value and value not in values:
231
+ values.append(value)
@@ -0,0 +1,57 @@
1
+ """Immutable calculation values returned to the APB2 integration boundary."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Mapping
6
+ from dataclasses import dataclass
7
+
8
+ import polars as pl
9
+
10
+
11
+ @dataclass(frozen=True, slots=True)
12
+ class PeptideCoverage:
13
+ """Aggregate FASTA coverage for one APB2 feature level."""
14
+
15
+ feature_count: int
16
+ unique_sequence_count: int
17
+ matched_feature_count: int
18
+ unmatched_feature_count: int
19
+ match_site_count: int
20
+
21
+
22
+ @dataclass(frozen=True, slots=True)
23
+ class PeptideLevelMatch:
24
+ """Feature-aligned peptide facts and their aggregate coverage."""
25
+
26
+ summary: pl.DataFrame
27
+ coverage: PeptideCoverage
28
+
29
+
30
+ @dataclass(frozen=True, slots=True)
31
+ class ProteinGroupCoverage:
32
+ """Aggregate FASTA coverage for reported protein-group members."""
33
+
34
+ group_count: int
35
+ member_count: int
36
+ matched_member_count: int
37
+ unmatched_member_count: int
38
+ ambiguous_member_count: int
39
+
40
+
41
+ @dataclass(frozen=True, slots=True)
42
+ class ProteinGroupMatch:
43
+ """Lossless member expansion, group summaries, and source relation."""
44
+
45
+ members: pl.DataFrame
46
+ member_key_columns: tuple[str, ...]
47
+ summary: pl.DataFrame
48
+ relation: pl.DataFrame
49
+ coverage: ProteinGroupCoverage
50
+
51
+
52
+ @dataclass(frozen=True, slots=True)
53
+ class FastaAnnotationReports:
54
+ """Persistable aggregate evidence from one annotation application."""
55
+
56
+ peptide_levels: Mapping[str, PeptideCoverage]
57
+ protein_groups: ProteinGroupCoverage | None
apb_fasta/cli.py ADDED
@@ -0,0 +1,160 @@
1
+ """Thin command-line composition for independent APB2 FASTA operations."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import sys
6
+ from pathlib import Path
7
+ from typing import Literal
8
+
9
+ from apb2.api import read_parsed_levels, write_parsed_levels
10
+ from cyclopts import App
11
+ from loguru import logger
12
+
13
+ from apb_fasta.api import FastaAnnotationParameters, FastaAnnotationResult, FastaAnnotator
14
+
15
+ app = App(
16
+ name="apb-fasta",
17
+ help="Verify peptides and merge protein annotations using FASTA files",
18
+ help_on_error=True,
19
+ )
20
+
21
+
22
+ @app.command
23
+ def verify_peptides(
24
+ source: Path,
25
+ *fasta_paths: Path,
26
+ output: Path,
27
+ formats: tuple[str, ...] = ("uniprotkb", "refseq"),
28
+ backend: Literal["auto", "ahocorapy", "ahocorasick_rs"] = "auto",
29
+ il_equivalent: bool = False,
30
+ protein_group_separator: str = ";",
31
+ ) -> int:
32
+ """Verify stripped peptide sequences in SOURCE against FASTA_PATHS."""
33
+ try:
34
+ _require_new_target(source, output)
35
+ annotator = FastaAnnotator.read(
36
+ fasta_paths,
37
+ formats,
38
+ FastaAnnotationParameters(
39
+ protein_group_separator=protein_group_separator,
40
+ matcher_backend=backend,
41
+ il_equivalent=il_equivalent,
42
+ ),
43
+ )
44
+ result = annotator.verify_peptides(read_parsed_levels(source))
45
+ _write_result(result, output)
46
+ _report_peptide_verification(result)
47
+ except (OSError, ValueError) as error:
48
+ logger.error(str(error))
49
+ return 1
50
+ logger.info("wrote peptide-verified APB2 result to {}", output)
51
+ return 0
52
+
53
+
54
+ @app.command
55
+ def merge_annotations(
56
+ source: Path,
57
+ *fasta_paths: Path,
58
+ output: Path,
59
+ formats: tuple[str, ...] = ("uniprotkb", "refseq"),
60
+ protein_group_separator: str = ";",
61
+ ) -> int:
62
+ """Merge FASTA annotations into the reported protein groups in SOURCE."""
63
+ try:
64
+ _require_new_target(source, output)
65
+ annotator = FastaAnnotator.read(
66
+ fasta_paths,
67
+ formats,
68
+ FastaAnnotationParameters(
69
+ protein_group_separator=protein_group_separator,
70
+ ),
71
+ )
72
+ result = annotator.merge_annotations(read_parsed_levels(source))
73
+ _write_result(result, output)
74
+ _report_protein_annotations(result)
75
+ except (OSError, ValueError) as error:
76
+ logger.error(str(error))
77
+ return 1
78
+ logger.info("wrote FASTA-annotated APB2 result to {}", output)
79
+ return 0
80
+
81
+
82
+ @app.command
83
+ def run(
84
+ source: Path,
85
+ *fasta_paths: Path,
86
+ output: Path,
87
+ formats: tuple[str, ...] = ("uniprotkb", "refseq"),
88
+ backend: Literal["auto", "ahocorapy", "ahocorasick_rs"] = "auto",
89
+ il_equivalent: bool = False,
90
+ protein_group_separator: str = ";",
91
+ ) -> int:
92
+ """Verify peptides and merge protein annotations in one in-memory run."""
93
+ try:
94
+ _require_new_target(source, output)
95
+ annotator = FastaAnnotator.read(
96
+ fasta_paths,
97
+ formats,
98
+ FastaAnnotationParameters(
99
+ protein_group_separator=protein_group_separator,
100
+ matcher_backend=backend,
101
+ il_equivalent=il_equivalent,
102
+ ),
103
+ )
104
+ result = annotator.annotate(read_parsed_levels(source))
105
+ _write_result(result, output)
106
+ _report_peptide_verification(result)
107
+ _report_protein_annotations(result)
108
+ except (OSError, ValueError) as error:
109
+ logger.error(str(error))
110
+ return 1
111
+ logger.info("wrote peptide-verified and FASTA-annotated APB2 result to {}", output)
112
+ return 0
113
+
114
+
115
+ def _require_new_target(source: Path, target: Path, /) -> None:
116
+ if source.resolve() == target.resolve():
117
+ raise ValueError("output must differ from source")
118
+ if target.exists():
119
+ raise ValueError(f"output already exists: {target}")
120
+
121
+
122
+ def _write_result(result: FastaAnnotationResult, output: Path) -> None:
123
+ write_parsed_levels(result.parsed, output)
124
+
125
+
126
+ def _report_peptide_verification(result: FastaAnnotationResult) -> None:
127
+ for level, coverage in result.reports.peptide_levels.items():
128
+ logger.info(
129
+ "level={} peptides_in_fasta={}/{} unmatched={} unique_sequences={} match_sites={}",
130
+ level,
131
+ coverage.matched_feature_count,
132
+ coverage.feature_count,
133
+ coverage.unmatched_feature_count,
134
+ coverage.unique_sequence_count,
135
+ coverage.match_site_count,
136
+ )
137
+
138
+
139
+ def _report_protein_annotations(result: FastaAnnotationResult) -> None:
140
+ coverage = result.reports.protein_groups
141
+ if coverage is None:
142
+ return
143
+ logger.info(
144
+ "protein_groups={} members_in_fasta={}/{} unmatched={} ambiguous={}",
145
+ coverage.group_count,
146
+ coverage.matched_member_count,
147
+ coverage.member_count,
148
+ coverage.unmatched_member_count,
149
+ coverage.ambiguous_member_count,
150
+ )
151
+
152
+
153
+ def main() -> int:
154
+ """Run the console application."""
155
+ result = app()
156
+ return int(result) if result is not None else 0
157
+
158
+
159
+ if __name__ == "__main__":
160
+ sys.exit(main())
@@ -0,0 +1,21 @@
1
+ """User-selected FASTA annotation behavior."""
2
+
3
+ from dataclasses import dataclass
4
+ from typing import Literal
5
+
6
+
7
+ @dataclass(frozen=True, slots=True)
8
+ class FastaAnnotationParameters:
9
+ """Configuration shared by peptide validation and protein annotation."""
10
+
11
+ protein_group_separator: str = ";"
12
+ matcher_backend: Literal["auto", "ahocorapy", "ahocorasick_rs"] = "auto"
13
+ il_equivalent: bool = False
14
+
15
+ def __post_init__(self) -> None:
16
+ """Reject a separator that cannot define protein-group members."""
17
+ if not self.protein_group_separator:
18
+ raise ValueError("protein_group_separator must not be empty")
19
+
20
+
21
+ DEFAULT_FASTA_ANNOTATION_PARAMETERS = FastaAnnotationParameters()
apb_fasta/errors.py ADDED
@@ -0,0 +1,5 @@
1
+ """Expected APB FASTA annotation failures."""
2
+
3
+
4
+ class FastaAnnotationError(ValueError):
5
+ """The supplied APB2 result or protein frame cannot be annotated safely."""
@@ -0,0 +1,288 @@
1
+ """Translate between APB2 result values and FASTA calculations."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from copy import deepcopy
6
+ from dataclasses import asdict
7
+ from typing import cast
8
+
9
+ import polars as pl
10
+ from apb2.api import JsonValue, ParsedLevels
11
+
12
+ from apb_fasta.calculation.matching import PeptideLevelInput
13
+ from apb_fasta.calculation.protein_groups import ProteinGroupInput
14
+ from apb_fasta.calculation.results import (
15
+ FastaAnnotationReports,
16
+ PeptideLevelMatch,
17
+ ProteinGroupMatch,
18
+ )
19
+ from apb_fasta.errors import FastaAnnotationError
20
+
21
+ FASTA_VALIDATION_NAME = "fasta_validation"
22
+ FASTA_SUMMARY_NAME = "fasta"
23
+ MEMBER_TABLE_NAME = "fasta_protein_group_members"
24
+ MEMBER_RELATION_NAME = "fasta_protein_group_membership"
25
+ PEPTIDE_PROPERTIES_NAME = "peptide_properties"
26
+ PEPTIDE_VERIFICATION_OPERATION = "peptide_verification"
27
+ PROTEIN_ANNOTATION_OPERATION = "protein_annotation"
28
+ PEPTIDE_PROPERTIES_OPERATION = "peptide_properties"
29
+ _PEPTIDE_COLUMN = "ProForma_peptide"
30
+ _RESERVED_MEMBER_COLUMNS = frozenset({"source_level", "member_ordinal", "match_ordinal"})
31
+
32
+
33
+ def validate_protein_frame(proteins: pl.DataFrame, /) -> None:
34
+ """Require the stable protein-fasta columns used by both calculations."""
35
+ required = {
36
+ "id",
37
+ "description",
38
+ "sequence",
39
+ "fasta_source_path",
40
+ "fasta_source_checksum",
41
+ "fasta_source_ordinal",
42
+ "fasta_record_ordinal",
43
+ }
44
+ missing = sorted(required.difference(proteins.columns))
45
+ if missing:
46
+ raise FastaAnnotationError(f"protein frame is missing required columns {missing}")
47
+ if proteins.get_column("id").null_count():
48
+ raise FastaAnnotationError("protein frame contains a missing id")
49
+ if proteins.get_column("sequence").null_count():
50
+ raise FastaAnnotationError("protein frame contains a missing sequence")
51
+
52
+
53
+ def peptide_inputs(parsed: ParsedLevels, /) -> dict[str, PeptideLevelInput]:
54
+ """Extract every peptide-derived level carrying APB2's canonical sequence."""
55
+ result: dict[str, PeptideLevelInput] = {}
56
+ for name, level in parsed.levels.items():
57
+ if name == "protein" or _PEPTIDE_COLUMN not in level.var.frame.columns:
58
+ continue
59
+ result[name] = PeptideLevelInput(
60
+ frame=level.var.frame,
61
+ sequence_column=_PEPTIDE_COLUMN,
62
+ accession_column=_fasta_accession_column(level.var.roles, level.var.frame),
63
+ )
64
+ return result
65
+
66
+
67
+ def protein_group_input(parsed: ParsedLevels, /) -> ProteinGroupInput | None:
68
+ """Extract the protein axis and its persisted FASTA-accession role."""
69
+ level = parsed.levels.get("protein")
70
+ if level is None:
71
+ return None
72
+ accession_column = _fasta_accession_column(level.var.roles, level.var.frame)
73
+ if accession_column is None:
74
+ raise FastaAnnotationError("protein level does not declare column_roles.fasta_accessions")
75
+ collisions = _RESERVED_MEMBER_COLUMNS.intersection(level.var.key_columns)
76
+ if collisions:
77
+ raise FastaAnnotationError(
78
+ f"protein variable keys collide with FASTA member columns {sorted(collisions)}"
79
+ )
80
+ return ProteinGroupInput(
81
+ frame=level.var.frame,
82
+ key_columns=level.var.key_columns,
83
+ accession_column=accession_column,
84
+ )
85
+
86
+
87
+ def validate_peptide_output_names(
88
+ parsed: ParsedLevels,
89
+ matches: dict[str, PeptideLevelMatch],
90
+ /,
91
+ ) -> None:
92
+ """Reject peptide-verification collisions before constructing a replacement."""
93
+ _validate_operation_metadata(parsed, PEPTIDE_VERIFICATION_OPERATION)
94
+ for name in matches:
95
+ level_name = name
96
+ if FASTA_VALIDATION_NAME in parsed.levels[level_name].varm:
97
+ raise FastaAnnotationError(
98
+ f"level {name!r} already contains varm[{FASTA_VALIDATION_NAME!r}]"
99
+ )
100
+ if "fasta" in parsed.levels[level_name].metadata:
101
+ raise FastaAnnotationError(f"level {name!r} metadata already contains 'fasta'")
102
+
103
+
104
+ def validate_protein_output_names(parsed: ParsedLevels, /) -> None:
105
+ """Reject protein-annotation collisions before constructing a replacement."""
106
+ _validate_operation_metadata(parsed, PROTEIN_ANNOTATION_OPERATION)
107
+ protein = parsed.levels["protein"]
108
+ if FASTA_SUMMARY_NAME in protein.varm:
109
+ raise FastaAnnotationError(f"level 'protein' already contains varm[{FASTA_SUMMARY_NAME!r}]")
110
+ if MEMBER_TABLE_NAME in parsed.annotation_tables:
111
+ raise FastaAnnotationError(
112
+ f"result already contains annotation table {MEMBER_TABLE_NAME!r}"
113
+ )
114
+ if MEMBER_RELATION_NAME in parsed.feature_relations:
115
+ raise FastaAnnotationError(
116
+ f"result already contains feature relation {MEMBER_RELATION_NAME!r}"
117
+ )
118
+
119
+
120
+ def apply_peptide_matches(
121
+ parsed: ParsedLevels,
122
+ matches: dict[str, PeptideLevelMatch],
123
+ /,
124
+ *,
125
+ requested_backend: str,
126
+ resolved_backend: str,
127
+ il_equivalent: bool,
128
+ protein_metadata: dict[str, JsonValue],
129
+ ) -> tuple[ParsedLevels, FastaAnnotationReports]:
130
+ """Attach peptide verification to a deep replacement of the APB2 result."""
131
+ validate_peptide_output_names(parsed, matches)
132
+ result = deepcopy(parsed)
133
+ for name, match in matches.items():
134
+ level_name = name
135
+ result.levels[level_name].varm[FASTA_VALIDATION_NAME] = match.summary.clone()
136
+ level_metadata = result.levels[level_name].metadata
137
+ level_metadata["fasta"] = {
138
+ PEPTIDE_VERIFICATION_OPERATION: cast(dict[str, JsonValue], asdict(match.coverage))
139
+ }
140
+ _record_operation_metadata(
141
+ result,
142
+ PEPTIDE_VERIFICATION_OPERATION,
143
+ {
144
+ "requested_backend": requested_backend,
145
+ "resolved_backend": resolved_backend,
146
+ "il_equivalent": il_equivalent,
147
+ **protein_metadata,
148
+ },
149
+ )
150
+ reports = FastaAnnotationReports(
151
+ peptide_levels={name: match.coverage for name, match in matches.items()},
152
+ protein_groups=None,
153
+ )
154
+ return result, reports
155
+
156
+
157
+ def apply_protein_group_match(
158
+ parsed: ParsedLevels,
159
+ match: ProteinGroupMatch,
160
+ /,
161
+ *,
162
+ protein_group_separator: str,
163
+ protein_metadata: dict[str, JsonValue],
164
+ ) -> tuple[ParsedLevels, FastaAnnotationReports]:
165
+ """Attach protein FASTA annotations to a deep replacement of the APB2 result."""
166
+ validate_protein_output_names(parsed)
167
+ result = deepcopy(parsed)
168
+ result.levels["protein"].varm[FASTA_SUMMARY_NAME] = match.summary.clone()
169
+ result = result.with_annotation_table(
170
+ MEMBER_TABLE_NAME,
171
+ match.members.clone(),
172
+ match.member_key_columns,
173
+ {"producer": "apb-fasta", "schema_version": "1"},
174
+ ).with_feature_relation(
175
+ MEMBER_RELATION_NAME,
176
+ MEMBER_TABLE_NAME,
177
+ "protein",
178
+ match.relation.clone(),
179
+ {"producer": "apb-fasta", "semantic": "member_of"},
180
+ )
181
+ _record_operation_metadata(
182
+ result,
183
+ PROTEIN_ANNOTATION_OPERATION,
184
+ {
185
+ "protein_group_separator": protein_group_separator,
186
+ **protein_metadata,
187
+ },
188
+ )
189
+ return result, FastaAnnotationReports(
190
+ peptide_levels={},
191
+ protein_groups=match.coverage,
192
+ )
193
+
194
+
195
+ def apply_peptide_properties(
196
+ parsed: ParsedLevels,
197
+ properties: dict[str, pl.DataFrame],
198
+ /,
199
+ *,
200
+ protein_fasta_version: str,
201
+ ) -> ParsedLevels:
202
+ """Attach feature-aligned peptide properties to a deep replacement of the APB2 result."""
203
+ _validate_operation_metadata(parsed, PEPTIDE_PROPERTIES_OPERATION)
204
+ for name in properties:
205
+ if PEPTIDE_PROPERTIES_NAME in parsed.levels[name].varm:
206
+ raise FastaAnnotationError(
207
+ f"level {name!r} already contains varm[{PEPTIDE_PROPERTIES_NAME!r}]"
208
+ )
209
+ result = deepcopy(parsed)
210
+ for name, frame in properties.items():
211
+ result.levels[name].varm[PEPTIDE_PROPERTIES_NAME] = frame.clone()
212
+ _record_operation_metadata(
213
+ result,
214
+ PEPTIDE_PROPERTIES_OPERATION,
215
+ {"protein_fasta_version": protein_fasta_version},
216
+ )
217
+ return result
218
+
219
+
220
+ def protein_frame_metadata(proteins: pl.DataFrame, /) -> dict[str, JsonValue]:
221
+ """Return bounded source and schema provenance for one protein frame."""
222
+ sources = (
223
+ proteins.select(
224
+ "fasta_source_ordinal",
225
+ "fasta_source_path",
226
+ "fasta_source_checksum",
227
+ )
228
+ .unique(maintain_order=True)
229
+ .sort("fasta_source_ordinal")
230
+ .to_dicts()
231
+ )
232
+ return {
233
+ "protein_count": proteins.height,
234
+ "protein_columns": list(proteins.columns),
235
+ "sources": {
236
+ str(_json_scalar(source["fasta_source_ordinal"])): {
237
+ "ordinal": _json_scalar(source["fasta_source_ordinal"]),
238
+ "path": _json_scalar(source["fasta_source_path"]),
239
+ "checksum": _json_scalar(source["fasta_source_checksum"]),
240
+ }
241
+ for source in sources
242
+ },
243
+ }
244
+
245
+
246
+ def _validate_operation_metadata(parsed: ParsedLevels, operation: str) -> None:
247
+ metadata = parsed.metadata.get("fasta")
248
+ if metadata is None:
249
+ return
250
+ if not isinstance(metadata, dict):
251
+ raise FastaAnnotationError("result metadata 'fasta' section is not an object")
252
+ provenance = metadata.get("provenance", {})
253
+ if not isinstance(provenance, dict):
254
+ raise FastaAnnotationError("result FASTA provenance is not an object")
255
+ if operation in provenance:
256
+ raise FastaAnnotationError(f"result already contains FASTA operation {operation!r}")
257
+
258
+
259
+ def _record_operation_metadata(
260
+ parsed: ParsedLevels,
261
+ operation: str,
262
+ operation_metadata: dict[str, JsonValue],
263
+ ) -> None:
264
+ existing = parsed.metadata.get("fasta")
265
+ metadata: dict[str, JsonValue]
266
+ if existing is None:
267
+ metadata = {"schema_version": "2", "provenance": {}}
268
+ else:
269
+ metadata = cast(dict[str, JsonValue], deepcopy(existing))
270
+ provenance = metadata.setdefault("provenance", {})
271
+ if not isinstance(provenance, dict):
272
+ raise FastaAnnotationError("result FASTA provenance is not an object")
273
+ provenance[operation] = operation_metadata
274
+ parsed.metadata["fasta"] = metadata
275
+
276
+
277
+ def _fasta_accession_column(
278
+ roles: dict[str, str],
279
+ frame: pl.DataFrame,
280
+ ) -> str | None:
281
+ value = roles.get("fasta_accessions")
282
+ return value if isinstance(value, str) and value in frame.columns else None
283
+
284
+
285
+ def _json_scalar(value: object) -> bool | int | float | str | None:
286
+ if value is None or isinstance(value, bool | int | float | str):
287
+ return value
288
+ raise FastaAnnotationError(f"protein source provenance contains {type(value).__name__}")
apb_fasta/py.typed ADDED
File without changes
@@ -0,0 +1,116 @@
1
+ Metadata-Version: 2.4
2
+ Name: apb-fasta
3
+ Version: 0.1.0
4
+ Summary: FASTA validation and protein annotation for APB2 results.
5
+ Keywords: proteomics,fasta,anndata,protein annotation,mass spectrometry
6
+ Author: Witold Wolski
7
+ Author-email: Witold Wolski <wew@fgcz.ethz.ch>
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Classifier: Development Status :: 3 - Alpha
11
+ Classifier: Intended Audience :: Science/Research
12
+ Classifier: Operating System :: OS Independent
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Programming Language :: Python :: 3.13
15
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
16
+ Classifier: Typing :: Typed
17
+ Requires-Dist: apb2>=0.1,<0.2
18
+ Requires-Dist: cyclopts>=4,<5
19
+ Requires-Dist: loguru>=0.7,<1
20
+ Requires-Dist: polars>=1.43,<2
21
+ Requires-Dist: protein-fasta>=0.3,<0.4
22
+ Requires-Dist: prozor>=0.1,<0.2
23
+ Requires-Python: >=3.13
24
+ Project-URL: Documentation, https://anndata-omics-bridge.github.io/apb-fasta/
25
+ Project-URL: Repository, https://github.com/anndata-omics-bridge/apb-fasta
26
+ Project-URL: Issues, https://github.com/anndata-omics-bridge/apb-fasta/issues
27
+ Description-Content-Type: text/markdown
28
+
29
+ # apb-fasta
30
+
31
+ FASTA verification and protein annotation for APB2 results.
32
+
33
+ **[Online documentation](https://anndata-omics-bridge.github.io/apb-fasta/)** or its [source index](https://github.com/anndata-omics-bridge/apb-fasta/blob/main/docs/index.md).
34
+
35
+ The first release exposes two independent operations:
36
+
37
+ - verify that APB2's modification-stripped peptide sequences occur in the supplied FASTA database;
38
+ - merge FASTA annotations for every reported protein-group member without collapsing the group to its leading accession.
39
+
40
+ Protein inference with Prozor is the planned third operation. It will remain explicit and opt-in.
41
+
42
+ ## Installation
43
+
44
+ APB FASTA requires Python 3.13 or later.
45
+
46
+ ```bash
47
+ pip install apb-fasta
48
+ ```
49
+
50
+ ## Python API
51
+
52
+ Read the FASTA files once and load the APB2 result:
53
+
54
+ ```python
55
+ from pathlib import Path
56
+
57
+ from apb2.api import read_parsed_levels, write_parsed_levels
58
+ from apb_fasta.api import FastaAnnotator
59
+
60
+ annotator = FastaAnnotator.read((Path("human.fasta"), Path("contaminants.fasta")))
61
+ parsed = read_parsed_levels(Path("input.h5mu"))
62
+ ```
63
+
64
+ Verify peptides only:
65
+
66
+ ```python
67
+ verified = annotator.verify_peptides(parsed)
68
+ print(verified.reports.peptide_levels)
69
+ write_parsed_levels(verified.parsed, Path("verified.h5mu"))
70
+ ```
71
+
72
+ Merge protein annotations only:
73
+
74
+ ```python
75
+ annotated = annotator.merge_annotations(parsed)
76
+ print(annotated.reports.protein_groups)
77
+ write_parsed_levels(annotated.parsed, Path("annotated.h5mu"))
78
+ ```
79
+
80
+ Apply both operations in memory:
81
+
82
+ ```python
83
+ complete = annotator.annotate(parsed)
84
+ write_parsed_levels(complete.parsed, Path("complete.h5mu"))
85
+ ```
86
+
87
+ The operations can also be chained explicitly without an intermediate file:
88
+
89
+ ```python
90
+ verified = annotator.verify_peptides(parsed)
91
+ complete = annotator.merge_annotations(verified.parsed)
92
+ ```
93
+
94
+ The annotator validates and binds the reusable protein frame once. Every method accepts one canonical `ParsedLevels` value and returns an immutable `FastaAnnotationResult` containing its replacement plus typed operation reports. `apb_fasta` neither opens result files nor receives raw AnnData or MuData objects.
95
+
96
+ ## CLI
97
+
98
+ Verify peptides only:
99
+
100
+ ```bash
101
+ apb-fasta verify-peptides input.h5mu human.fasta contaminants.fasta --output verified.h5mu
102
+ ```
103
+
104
+ Merge protein annotations only:
105
+
106
+ ```bash
107
+ apb-fasta merge-annotations input.h5mu human.fasta contaminants.fasta --output annotated.h5mu
108
+ ```
109
+
110
+ Apply both operations together:
111
+
112
+ ```bash
113
+ apb-fasta run input.h5mu human.fasta contaminants.fasta --output complete.h5mu
114
+ ```
115
+
116
+ Peptide verification adds feature-aligned `varm["fasta_validation"]` tables. Protein annotation adds protein-aligned `varm["fasta"]`, the lossless `fasta_protein_group_members` annotation table, and its directed relation to the protein axis.
@@ -0,0 +1,17 @@
1
+ apb_fasta/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
2
+ apb_fasta/api.py,sha256=2squHIacOk51JchAEupW2A6rrvCCGBFf2tgBkbhzSL0,6202
3
+ apb_fasta/calculation/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
4
+ apb_fasta/calculation/matching.py,sha256=Sw8WeWyW89hGzkxAgAWmh_mTccxC9hkfbewPiSMxqg4,8097
5
+ apb_fasta/calculation/peptide_properties.py,sha256=80A_PjI_kgG0_kX_KjccRmAkLfkOaAc_oL6Dhy-DGl8,982
6
+ apb_fasta/calculation/protein_groups.py,sha256=FKg7kqvb1Tsisgd-HiJIUlZOTdGq8bufJqZIfNWGQ9o,8178
7
+ apb_fasta/calculation/results.py,sha256=9iOPBPfeekJ04W7jErm0k4uskPMkSumXx_8sd0IdE9s,1474
8
+ apb_fasta/cli.py,sha256=V6VqCzmuogjZj8EaIh53SWRhhkVoIgQqnxWat2eYMAc,5064
9
+ apb_fasta/configuration.py,sha256=blQIb9ywCp00Xr6ypWfQzUQ-x8aF7aegAB3veD0cgF4,706
10
+ apb_fasta/errors.py,sha256=_oLU8leci5hMXN9XJa0rD6HEnvfzv8CYlGn3e50u7P0,168
11
+ apb_fasta/integration.py,sha256=TcH0jtL_Aby2RKZydhLD_-qCXsSqv0wEYYKZ8_xWY6I,10643
12
+ apb_fasta/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
13
+ apb_fasta-0.1.0.dist-info/licenses/LICENSE,sha256=OPECQFbzDYwcWIzs8D8qw6o9AZsc97_u2fwqbe-ojAM,1070
14
+ apb_fasta-0.1.0.dist-info/WHEEL,sha256=cmC5s21ojypbVslldL7IJq3hjZH-tINy4rziKePFsG0,81
15
+ apb_fasta-0.1.0.dist-info/entry_points.txt,sha256=mPjMZ9aiQIs5fOumE4mwi2Kf-dm8BxDeLg-RFaXyNVQ,50
16
+ apb_fasta-0.1.0.dist-info/METADATA,sha256=McjRlQmEL64BaJuEAU2_Bd0kR7PRTFqURKySPP99knE,3864
17
+ apb_fasta-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: uv 0.12.23
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,3 @@
1
+ [console_scripts]
2
+ apb-fasta = apb_fasta.cli:main
3
+
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Witold Wolski
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.