apb-fasta 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- apb_fasta/__init__.py +0 -0
- apb_fasta/api.py +164 -0
- apb_fasta/calculation/__init__.py +0 -0
- apb_fasta/calculation/matching.py +209 -0
- apb_fasta/calculation/peptide_properties.py +25 -0
- apb_fasta/calculation/protein_groups.py +231 -0
- apb_fasta/calculation/results.py +57 -0
- apb_fasta/cli.py +160 -0
- apb_fasta/configuration.py +21 -0
- apb_fasta/errors.py +5 -0
- apb_fasta/integration.py +288 -0
- apb_fasta/py.typed +0 -0
- apb_fasta-0.1.0.dist-info/METADATA +116 -0
- apb_fasta-0.1.0.dist-info/RECORD +17 -0
- apb_fasta-0.1.0.dist-info/WHEEL +4 -0
- apb_fasta-0.1.0.dist-info/entry_points.txt +3 -0
- apb_fasta-0.1.0.dist-info/licenses/LICENSE +21 -0
apb_fasta/__init__.py
ADDED
|
File without changes
|
apb_fasta/api.py
ADDED
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
"""Public in-memory FASTA annotation API."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Sequence
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from importlib.metadata import version
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
import polars as pl
|
|
11
|
+
from apb2.api import ParsedLevels
|
|
12
|
+
from protein_fasta.api import ProteinDatabase, ProteinFormat
|
|
13
|
+
from prozor.api import resolve_backend
|
|
14
|
+
|
|
15
|
+
from apb_fasta.calculation.matching import match_peptide_levels
|
|
16
|
+
from apb_fasta.calculation.peptide_properties import peptide_level_properties
|
|
17
|
+
from apb_fasta.calculation.protein_groups import match_protein_groups
|
|
18
|
+
from apb_fasta.calculation.results import FastaAnnotationReports
|
|
19
|
+
from apb_fasta.configuration import (
|
|
20
|
+
DEFAULT_FASTA_ANNOTATION_PARAMETERS,
|
|
21
|
+
FastaAnnotationParameters,
|
|
22
|
+
)
|
|
23
|
+
from apb_fasta.errors import FastaAnnotationError
|
|
24
|
+
from apb_fasta.integration import (
|
|
25
|
+
apply_peptide_matches,
|
|
26
|
+
apply_peptide_properties,
|
|
27
|
+
apply_protein_group_match,
|
|
28
|
+
peptide_inputs,
|
|
29
|
+
protein_frame_metadata,
|
|
30
|
+
protein_group_input,
|
|
31
|
+
validate_protein_frame,
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
_NO_PEPTIDE_LEVEL = (
|
|
35
|
+
"result contains no peptide-derived level with canonical ProForma_peptide values"
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass(frozen=True, slots=True)
|
|
40
|
+
class FastaAnnotationResult:
|
|
41
|
+
"""A replacement APB2 result and its FASTA reports."""
|
|
42
|
+
|
|
43
|
+
parsed: ParsedLevels
|
|
44
|
+
reports: FastaAnnotationReports
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class FastaAnnotator:
|
|
48
|
+
"""Bind one protein database and its FASTA annotation behavior."""
|
|
49
|
+
|
|
50
|
+
__slots__ = ("_parameters", "_proteins")
|
|
51
|
+
|
|
52
|
+
def __init__(
|
|
53
|
+
self,
|
|
54
|
+
proteins: pl.DataFrame,
|
|
55
|
+
parameters: FastaAnnotationParameters = DEFAULT_FASTA_ANNOTATION_PARAMETERS,
|
|
56
|
+
) -> None:
|
|
57
|
+
"""Validate and bind a reusable protein database and configuration."""
|
|
58
|
+
validate_protein_frame(proteins)
|
|
59
|
+
self._proteins = proteins
|
|
60
|
+
self._parameters = parameters
|
|
61
|
+
|
|
62
|
+
@classmethod
|
|
63
|
+
def read(
|
|
64
|
+
cls,
|
|
65
|
+
fasta: Sequence[Path],
|
|
66
|
+
formats: Sequence[str] = ("uniprotkb", "refseq"),
|
|
67
|
+
parameters: FastaAnnotationParameters = DEFAULT_FASTA_ANNOTATION_PARAMETERS,
|
|
68
|
+
) -> FastaAnnotator:
|
|
69
|
+
"""Read FASTA files, or the protein-fasta database Parquet written from them."""
|
|
70
|
+
if not fasta:
|
|
71
|
+
raise ValueError("at least one FASTA path is required")
|
|
72
|
+
database = ProteinDatabase(*(ProteinFormat(name) for name in formats))
|
|
73
|
+
return cls(database.parse(tuple(fasta)), parameters)
|
|
74
|
+
|
|
75
|
+
@property
|
|
76
|
+
def proteins(self) -> pl.DataFrame:
|
|
77
|
+
"""The bound protein table, one row per FASTA entry in file order."""
|
|
78
|
+
return self._proteins
|
|
79
|
+
|
|
80
|
+
def verify_peptides(self, parsed: ParsedLevels) -> FastaAnnotationResult:
|
|
81
|
+
"""Verify every canonical stripped peptide against the protein sequences."""
|
|
82
|
+
inputs = peptide_inputs(parsed)
|
|
83
|
+
if not inputs:
|
|
84
|
+
raise FastaAnnotationError(_NO_PEPTIDE_LEVEL)
|
|
85
|
+
peptide_levels = match_peptide_levels(
|
|
86
|
+
inputs,
|
|
87
|
+
self._proteins,
|
|
88
|
+
backend=self._parameters.matcher_backend,
|
|
89
|
+
il_equivalent=self._parameters.il_equivalent,
|
|
90
|
+
protein_group_separator=self._parameters.protein_group_separator,
|
|
91
|
+
)
|
|
92
|
+
replacement, reports = apply_peptide_matches(
|
|
93
|
+
parsed,
|
|
94
|
+
peptide_levels,
|
|
95
|
+
requested_backend=self._parameters.matcher_backend,
|
|
96
|
+
resolved_backend=resolve_backend(self._parameters.matcher_backend),
|
|
97
|
+
il_equivalent=self._parameters.il_equivalent,
|
|
98
|
+
protein_metadata=protein_frame_metadata(self._proteins),
|
|
99
|
+
)
|
|
100
|
+
return FastaAnnotationResult(parsed=replacement, reports=reports)
|
|
101
|
+
|
|
102
|
+
def merge_annotations(self, parsed: ParsedLevels) -> FastaAnnotationResult:
|
|
103
|
+
"""Merge FASTA annotations for every reported protein-group member."""
|
|
104
|
+
protein_input = protein_group_input(parsed)
|
|
105
|
+
if protein_input is None:
|
|
106
|
+
raise FastaAnnotationError("result contains no protein level to annotate")
|
|
107
|
+
protein_groups = match_protein_groups(
|
|
108
|
+
protein_input,
|
|
109
|
+
self._proteins,
|
|
110
|
+
separator=self._parameters.protein_group_separator,
|
|
111
|
+
)
|
|
112
|
+
replacement, reports = apply_protein_group_match(
|
|
113
|
+
parsed,
|
|
114
|
+
protein_groups,
|
|
115
|
+
protein_group_separator=self._parameters.protein_group_separator,
|
|
116
|
+
protein_metadata=protein_frame_metadata(self._proteins),
|
|
117
|
+
)
|
|
118
|
+
return FastaAnnotationResult(parsed=replacement, reports=reports)
|
|
119
|
+
|
|
120
|
+
def annotate(self, parsed: ParsedLevels) -> FastaAnnotationResult:
|
|
121
|
+
"""Verify peptides and then merge protein annotations in memory."""
|
|
122
|
+
verified = self.verify_peptides(parsed)
|
|
123
|
+
annotated = self.merge_annotations(verified.parsed)
|
|
124
|
+
return FastaAnnotationResult(
|
|
125
|
+
parsed=annotated.parsed,
|
|
126
|
+
reports=FastaAnnotationReports(
|
|
127
|
+
peptide_levels=verified.reports.peptide_levels,
|
|
128
|
+
protein_groups=annotated.reports.protein_groups,
|
|
129
|
+
),
|
|
130
|
+
)
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def add_peptide_properties(parsed: ParsedLevels) -> ParsedLevels:
|
|
134
|
+
"""Attach sequence-derived properties to every peptide-derived level.
|
|
135
|
+
|
|
136
|
+
Writes a feature-aligned ``varm["peptide_properties"]`` computed by
|
|
137
|
+
``protein_fasta.peptide_frame.peptide_property_frame`` from each level's ``ProForma_peptide``.
|
|
138
|
+
No protein database is needed. A feature without a sequence, or with a residue outside the 20
|
|
139
|
+
standard amino acids, has null properties.
|
|
140
|
+
|
|
141
|
+
Raises:
|
|
142
|
+
FastaAnnotationError: If the result has no peptide-derived level, already contains the
|
|
143
|
+
output, or already records the operation.
|
|
144
|
+
ValueError: If a sequence is not stripped upper-case letters.
|
|
145
|
+
"""
|
|
146
|
+
inputs = peptide_inputs(parsed)
|
|
147
|
+
if not inputs:
|
|
148
|
+
raise FastaAnnotationError(_NO_PEPTIDE_LEVEL)
|
|
149
|
+
properties = {
|
|
150
|
+
name: peptide_level_properties(level.frame, level.sequence_column)
|
|
151
|
+
for name, level in inputs.items()
|
|
152
|
+
}
|
|
153
|
+
return apply_peptide_properties(
|
|
154
|
+
parsed, properties, protein_fasta_version=version("protein-fasta")
|
|
155
|
+
)
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
__all__ = [
|
|
159
|
+
"FastaAnnotationParameters",
|
|
160
|
+
"FastaAnnotationReports",
|
|
161
|
+
"FastaAnnotationResult",
|
|
162
|
+
"FastaAnnotator",
|
|
163
|
+
"add_peptide_properties",
|
|
164
|
+
]
|
|
File without changes
|
|
@@ -0,0 +1,209 @@
|
|
|
1
|
+
"""Peptide-to-protein matching and feature-aligned summaries."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Mapping
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
|
|
8
|
+
import polars as pl
|
|
9
|
+
from prozor.api import annotate_peptides
|
|
10
|
+
|
|
11
|
+
from apb_fasta.calculation.results import PeptideCoverage, PeptideLevelMatch
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass(frozen=True, slots=True)
|
|
15
|
+
class PeptideLevelInput:
|
|
16
|
+
"""The columns needed to validate one peptide-derived feature level."""
|
|
17
|
+
|
|
18
|
+
frame: pl.DataFrame
|
|
19
|
+
sequence_column: str
|
|
20
|
+
accession_column: str | None
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def match_peptide_levels(
|
|
24
|
+
levels: Mapping[str, PeptideLevelInput],
|
|
25
|
+
proteins: pl.DataFrame,
|
|
26
|
+
/,
|
|
27
|
+
*,
|
|
28
|
+
backend: str,
|
|
29
|
+
il_equivalent: bool,
|
|
30
|
+
protein_group_separator: str,
|
|
31
|
+
) -> dict[str, PeptideLevelMatch]:
|
|
32
|
+
"""Match distinct normalized peptides once and summarize every source feature."""
|
|
33
|
+
if not levels:
|
|
34
|
+
return {}
|
|
35
|
+
prepared = {name: _prepare_features(level, il_equivalent) for name, level in levels.items()}
|
|
36
|
+
peptides = (
|
|
37
|
+
pl.concat([frame.select("peptide") for frame in prepared.values()])
|
|
38
|
+
.drop_nulls()
|
|
39
|
+
.unique(maintain_order=True)
|
|
40
|
+
)
|
|
41
|
+
sequence = _text_column(proteins, "sequence")
|
|
42
|
+
if il_equivalent:
|
|
43
|
+
sequence = sequence.str.replace_all("I", "L", literal=True)
|
|
44
|
+
sequences = proteins.select(sequence).to_series()
|
|
45
|
+
if sequences.null_count():
|
|
46
|
+
raise ValueError("protein frame column 'sequence' contains a non-text value")
|
|
47
|
+
matched = annotate_peptides(
|
|
48
|
+
peptides.get_column("peptide").to_list(),
|
|
49
|
+
dict(zip(map(str, range(sequences.len())), sequences.to_list(), strict=True)),
|
|
50
|
+
backend=backend,
|
|
51
|
+
)
|
|
52
|
+
occurrences = pl.DataFrame(
|
|
53
|
+
[(match.peptide, int(match.protein_id)) for match in matched],
|
|
54
|
+
schema={"peptide": pl.String, "record": pl.UInt32},
|
|
55
|
+
orient="row",
|
|
56
|
+
)
|
|
57
|
+
site_counts = occurrences.group_by("peptide").agg(
|
|
58
|
+
pl.len().cast(pl.UInt64).alias("fasta_match_site_count")
|
|
59
|
+
)
|
|
60
|
+
columns = set(proteins.columns)
|
|
61
|
+
indexed = proteins.select(
|
|
62
|
+
id=_text_column(proteins, "id"),
|
|
63
|
+
organism=_text_column(
|
|
64
|
+
proteins, "organism_mnemonic" if "organism_mnemonic" in columns else None
|
|
65
|
+
),
|
|
66
|
+
is_contaminant=pl.col("is_contaminant") if "is_contaminant" in columns else pl.lit(False),
|
|
67
|
+
).with_row_index("record")
|
|
68
|
+
distinct = occurrences.unique(maintain_order=True).join(
|
|
69
|
+
indexed, on="record", how="left", maintain_order="left"
|
|
70
|
+
)
|
|
71
|
+
if distinct.get_column("id").null_count():
|
|
72
|
+
raise ValueError("protein frame column 'id' contains a non-text value")
|
|
73
|
+
matches = distinct.group_by("peptide", maintain_order=True).agg(
|
|
74
|
+
pl.col("record").alias("matched_records"),
|
|
75
|
+
pl.len().cast(pl.UInt64).alias("fasta_matching_protein_count"),
|
|
76
|
+
pl.col("id").str.join(";").alias("fasta_matching_protein_ids"),
|
|
77
|
+
fasta_matching_organisms=pl.col("organism").drop_nulls().unique().sort().str.join(";"),
|
|
78
|
+
fasta_matches_contaminant=pl.col("is_contaminant").any(),
|
|
79
|
+
)
|
|
80
|
+
matches = matches.join(site_counts, on="peptide", how="left")
|
|
81
|
+
assignments = _reported_assignments(prepared, proteins, protein_group_separator)
|
|
82
|
+
return {
|
|
83
|
+
name: _level_match(frame, matches, assignments, levels[name].accession_column is not None)
|
|
84
|
+
for name, frame in prepared.items()
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _text_column(frame: pl.DataFrame, column: str | None) -> pl.Expr:
|
|
89
|
+
if column is None:
|
|
90
|
+
return pl.lit(None, dtype=pl.String)
|
|
91
|
+
dtype = frame.schema[column]
|
|
92
|
+
if isinstance(dtype, (pl.String, pl.Categorical, pl.Enum)):
|
|
93
|
+
return pl.col(column).cast(pl.String)
|
|
94
|
+
if dtype == pl.Object:
|
|
95
|
+
# Only foreign mixed-object columns need scalar type checking.
|
|
96
|
+
return pl.col(column).map_elements(_optional_text, return_dtype=pl.String)
|
|
97
|
+
return pl.lit(None, dtype=pl.String)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _optional_text(value: object) -> str | None:
|
|
101
|
+
return value if isinstance(value, str) else None
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _prepare_features(level: PeptideLevelInput, il_equivalent: bool) -> pl.DataFrame:
|
|
105
|
+
sequence = _text_column(level.frame, level.sequence_column).str.strip_chars().str.to_uppercase()
|
|
106
|
+
if il_equivalent:
|
|
107
|
+
sequence = sequence.str.replace_all("I", "L", literal=True)
|
|
108
|
+
# with_columns keeps the source height even when both expressions are scalar nulls.
|
|
109
|
+
return level.frame.with_columns(
|
|
110
|
+
sequence.alias("peptide"),
|
|
111
|
+
_text_column(level.frame, level.accession_column).alias("assignment"),
|
|
112
|
+
).select(
|
|
113
|
+
pl.when(pl.col("peptide") != "").then(pl.col("peptide")).alias("peptide"),
|
|
114
|
+
"assignment",
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _reported_assignments(
|
|
119
|
+
prepared: Mapping[str, pl.DataFrame], proteins: pl.DataFrame, separator: str
|
|
120
|
+
) -> pl.DataFrame:
|
|
121
|
+
aliases = pl.concat(
|
|
122
|
+
[
|
|
123
|
+
proteins.with_columns(_text_column(proteins, column).alias("member"))
|
|
124
|
+
.select("member")
|
|
125
|
+
.with_row_index("record")
|
|
126
|
+
for column in ("id", "accession")
|
|
127
|
+
if column in proteins.columns
|
|
128
|
+
]
|
|
129
|
+
)
|
|
130
|
+
aliases = (
|
|
131
|
+
aliases.filter(pl.col("member").is_not_null() & (pl.col("member") != ""))
|
|
132
|
+
.unique()
|
|
133
|
+
.group_by("member")
|
|
134
|
+
.agg("record")
|
|
135
|
+
)
|
|
136
|
+
assignments = (
|
|
137
|
+
pl.concat([frame.select("assignment") for frame in prepared.values()]).drop_nulls().unique()
|
|
138
|
+
)
|
|
139
|
+
return (
|
|
140
|
+
assignments.with_columns(
|
|
141
|
+
pl.col("assignment")
|
|
142
|
+
.str.split(separator)
|
|
143
|
+
.list.eval(pl.element().str.strip_chars())
|
|
144
|
+
.alias("member")
|
|
145
|
+
)
|
|
146
|
+
.explode("member", empty_as_null=True)
|
|
147
|
+
.filter(pl.col("member") != "")
|
|
148
|
+
.join(aliases, on="member", how="left")
|
|
149
|
+
.group_by("assignment")
|
|
150
|
+
.agg(
|
|
151
|
+
pl.len().cast(pl.UInt64).alias("reported_member_count"),
|
|
152
|
+
pl.col("record")
|
|
153
|
+
.is_not_null()
|
|
154
|
+
.sum()
|
|
155
|
+
.cast(pl.UInt64)
|
|
156
|
+
.alias("reported_members_in_fasta_count"),
|
|
157
|
+
pl.col("record")
|
|
158
|
+
.explode(empty_as_null=True)
|
|
159
|
+
.drop_nulls()
|
|
160
|
+
.unique()
|
|
161
|
+
.alias("reported_records"),
|
|
162
|
+
)
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _level_match(
|
|
167
|
+
features: pl.DataFrame,
|
|
168
|
+
matches: pl.DataFrame,
|
|
169
|
+
assignments: pl.DataFrame,
|
|
170
|
+
has_assignment: bool,
|
|
171
|
+
) -> PeptideLevelMatch:
|
|
172
|
+
joined = features.join(matches, on="peptide", how="left", maintain_order="left").join(
|
|
173
|
+
assignments, on="assignment", how="left", maintain_order="left"
|
|
174
|
+
)
|
|
175
|
+
summary = joined.select(
|
|
176
|
+
(pl.col("fasta_matching_protein_count").fill_null(0) > 0).alias("peptide_in_fasta"),
|
|
177
|
+
pl.col("fasta_match_site_count").fill_null(0),
|
|
178
|
+
pl.col("fasta_matching_protein_count").fill_null(0),
|
|
179
|
+
pl.col("fasta_matching_protein_ids").fill_null(""),
|
|
180
|
+
pl.col("fasta_matching_organisms").fill_null(""),
|
|
181
|
+
pl.col("fasta_matches_contaminant").fill_null(False),
|
|
182
|
+
pl.col("reported_member_count").fill_null(0),
|
|
183
|
+
pl.col("reported_members_in_fasta_count").fill_null(0),
|
|
184
|
+
(
|
|
185
|
+
pl.col("matched_records")
|
|
186
|
+
.fill_null([])
|
|
187
|
+
.list.set_intersection(pl.col("reported_records").fill_null([]))
|
|
188
|
+
.list.len()
|
|
189
|
+
> 0
|
|
190
|
+
).alias("peptide_in_reported_protein"),
|
|
191
|
+
)
|
|
192
|
+
if not has_assignment:
|
|
193
|
+
summary = summary.with_columns(
|
|
194
|
+
pl.lit(None, dtype=pl.UInt64).alias("reported_member_count"),
|
|
195
|
+
pl.lit(None, dtype=pl.UInt64).alias("reported_members_in_fasta_count"),
|
|
196
|
+
pl.lit(None, dtype=pl.Boolean).alias("peptide_in_reported_protein"),
|
|
197
|
+
)
|
|
198
|
+
unique = joined.select("peptide", "fasta_match_site_count").drop_nulls("peptide").unique()
|
|
199
|
+
matched_count = int(summary.get_column("peptide_in_fasta").sum() or 0)
|
|
200
|
+
return PeptideLevelMatch(
|
|
201
|
+
summary=summary,
|
|
202
|
+
coverage=PeptideCoverage(
|
|
203
|
+
feature_count=features.height,
|
|
204
|
+
unique_sequence_count=unique.height,
|
|
205
|
+
matched_feature_count=matched_count,
|
|
206
|
+
unmatched_feature_count=features.height - matched_count,
|
|
207
|
+
match_site_count=int(unique.get_column("fasta_match_site_count").sum() or 0),
|
|
208
|
+
),
|
|
209
|
+
)
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
"""Feature-aligned sequence properties of one peptide-derived level."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import polars as pl
|
|
6
|
+
from protein_fasta.api import peptide_property_frame
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def peptide_level_properties(frame: pl.DataFrame, sequence_column: str, /) -> pl.DataFrame:
|
|
10
|
+
"""Return one property row per feature of ``frame``, in feature order.
|
|
11
|
+
|
|
12
|
+
Each distinct sequence is computed once; a feature without a sequence has null properties.
|
|
13
|
+
"""
|
|
14
|
+
features = frame.select(
|
|
15
|
+
pl.col(sequence_column)
|
|
16
|
+
.cast(pl.String)
|
|
17
|
+
.str.strip_chars()
|
|
18
|
+
.str.to_uppercase()
|
|
19
|
+
.alias("sequence")
|
|
20
|
+
).select(pl.when(pl.col("sequence") != "").then(pl.col("sequence")).alias("sequence"))
|
|
21
|
+
sequences = features.get_column("sequence").drop_nulls().unique(maintain_order=True)
|
|
22
|
+
properties = peptide_property_frame(sequences.to_list())
|
|
23
|
+
return features.join(properties, on="sequence", how="left", maintain_order="left").drop(
|
|
24
|
+
"sequence"
|
|
25
|
+
)
|
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
"""Protein-group expansion and FASTA annotation over Polars values."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections import defaultdict
|
|
6
|
+
from collections.abc import Mapping
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
|
|
9
|
+
import polars as pl
|
|
10
|
+
|
|
11
|
+
from apb_fasta.calculation.results import ProteinGroupCoverage, ProteinGroupMatch
|
|
12
|
+
|
|
13
|
+
_OPTIONAL_FASTA_COLUMNS = (
|
|
14
|
+
"is_decoy",
|
|
15
|
+
"is_contaminant",
|
|
16
|
+
"accession",
|
|
17
|
+
"protein_name",
|
|
18
|
+
"gene_name",
|
|
19
|
+
"organism_name",
|
|
20
|
+
"taxonomy_id",
|
|
21
|
+
"database",
|
|
22
|
+
"review_status",
|
|
23
|
+
"fasta_source_path",
|
|
24
|
+
"fasta_source_checksum",
|
|
25
|
+
"fasta_source_ordinal",
|
|
26
|
+
"fasta_record_ordinal",
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass(frozen=True, slots=True)
|
|
31
|
+
class ProteinGroupInput:
|
|
32
|
+
"""The protein feature axis and its declared FASTA accession column."""
|
|
33
|
+
|
|
34
|
+
frame: pl.DataFrame
|
|
35
|
+
key_columns: tuple[str, ...]
|
|
36
|
+
accession_column: str
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def match_protein_groups(
|
|
40
|
+
source: ProteinGroupInput,
|
|
41
|
+
proteins: pl.DataFrame,
|
|
42
|
+
/,
|
|
43
|
+
*,
|
|
44
|
+
separator: str,
|
|
45
|
+
) -> ProteinGroupMatch:
|
|
46
|
+
"""Expand every reported member and retain every matching FASTA record."""
|
|
47
|
+
protein_rows = proteins.to_dicts()
|
|
48
|
+
aliases = _protein_aliases(protein_rows)
|
|
49
|
+
members_rows: list[dict[str, object]] = []
|
|
50
|
+
summaries: list[dict[str, object]] = []
|
|
51
|
+
relation_rows: list[dict[str, object]] = []
|
|
52
|
+
matched_members = 0
|
|
53
|
+
unmatched_members = 0
|
|
54
|
+
ambiguous_members = 0
|
|
55
|
+
member_count = 0
|
|
56
|
+
for group_index, group in enumerate(source.frame.to_dicts()):
|
|
57
|
+
raw_group = group.get(source.accession_column)
|
|
58
|
+
members = _members(raw_group, separator)
|
|
59
|
+
group_matched = 0
|
|
60
|
+
group_unmatched = 0
|
|
61
|
+
group_ambiguous = 0
|
|
62
|
+
descriptions: list[str] = []
|
|
63
|
+
gene_names: list[str] = []
|
|
64
|
+
organism_names: list[str] = []
|
|
65
|
+
for member_ordinal, member in enumerate(members):
|
|
66
|
+
member_count += 1
|
|
67
|
+
matches = aliases.get(member, ())
|
|
68
|
+
if not matches:
|
|
69
|
+
unmatched_members += 1
|
|
70
|
+
group_unmatched += 1
|
|
71
|
+
members_rows.append(
|
|
72
|
+
_member_row(
|
|
73
|
+
source,
|
|
74
|
+
group,
|
|
75
|
+
raw_group,
|
|
76
|
+
member,
|
|
77
|
+
member_ordinal,
|
|
78
|
+
0,
|
|
79
|
+
"unmatched",
|
|
80
|
+
None,
|
|
81
|
+
)
|
|
82
|
+
)
|
|
83
|
+
relation_rows.append(
|
|
84
|
+
{"row": len(members_rows) - 1, "column": group_index, "value": 1.0}
|
|
85
|
+
)
|
|
86
|
+
continue
|
|
87
|
+
matched_members += 1
|
|
88
|
+
group_matched += 1
|
|
89
|
+
status = "ambiguous" if len(matches) > 1 else "matched"
|
|
90
|
+
if len(matches) > 1:
|
|
91
|
+
ambiguous_members += 1
|
|
92
|
+
group_ambiguous += 1
|
|
93
|
+
for match_ordinal, protein_index in enumerate(matches):
|
|
94
|
+
protein = protein_rows[protein_index]
|
|
95
|
+
members_rows.append(
|
|
96
|
+
_member_row(
|
|
97
|
+
source,
|
|
98
|
+
group,
|
|
99
|
+
raw_group,
|
|
100
|
+
member,
|
|
101
|
+
member_ordinal,
|
|
102
|
+
match_ordinal,
|
|
103
|
+
status,
|
|
104
|
+
protein,
|
|
105
|
+
)
|
|
106
|
+
)
|
|
107
|
+
relation_rows.append(
|
|
108
|
+
{"row": len(members_rows) - 1, "column": group_index, "value": 1.0}
|
|
109
|
+
)
|
|
110
|
+
_append_text(descriptions, protein.get("description"))
|
|
111
|
+
_append_text(gene_names, protein.get("gene_name"))
|
|
112
|
+
_append_text(organism_names, protein.get("organism_name"))
|
|
113
|
+
summaries.append(
|
|
114
|
+
{
|
|
115
|
+
"reported_member_count": len(members),
|
|
116
|
+
"matched_member_count": group_matched,
|
|
117
|
+
"unmatched_member_count": group_unmatched,
|
|
118
|
+
"ambiguous_member_count": group_ambiguous,
|
|
119
|
+
"all_members_in_fasta": bool(members) and group_unmatched == 0,
|
|
120
|
+
"any_member_in_fasta": group_matched > 0,
|
|
121
|
+
"fasta_descriptions": ";".join(descriptions),
|
|
122
|
+
"fasta_gene_names": ";".join(gene_names),
|
|
123
|
+
"fasta_organism_names": ";".join(organism_names),
|
|
124
|
+
}
|
|
125
|
+
)
|
|
126
|
+
member_keys = ("source_level", *source.key_columns, "member_ordinal", "match_ordinal")
|
|
127
|
+
return ProteinGroupMatch(
|
|
128
|
+
members=pl.DataFrame(members_rows, schema=_member_schema(source, proteins)),
|
|
129
|
+
member_key_columns=member_keys,
|
|
130
|
+
summary=pl.DataFrame(summaries, schema=_summary_schema()),
|
|
131
|
+
relation=pl.DataFrame(
|
|
132
|
+
relation_rows,
|
|
133
|
+
schema={"row": pl.Int64, "column": pl.Int64, "value": pl.Float64},
|
|
134
|
+
),
|
|
135
|
+
coverage=ProteinGroupCoverage(
|
|
136
|
+
group_count=source.frame.height,
|
|
137
|
+
member_count=member_count,
|
|
138
|
+
matched_member_count=matched_members,
|
|
139
|
+
unmatched_member_count=unmatched_members,
|
|
140
|
+
ambiguous_member_count=ambiguous_members,
|
|
141
|
+
),
|
|
142
|
+
)
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def _member_row(
|
|
146
|
+
source: ProteinGroupInput,
|
|
147
|
+
group: Mapping[str, object],
|
|
148
|
+
raw_group: object,
|
|
149
|
+
member: str,
|
|
150
|
+
member_ordinal: int,
|
|
151
|
+
match_ordinal: int,
|
|
152
|
+
status: str,
|
|
153
|
+
protein: Mapping[str, object] | None,
|
|
154
|
+
) -> dict[str, object]:
|
|
155
|
+
sequence = None if protein is None else protein.get("sequence")
|
|
156
|
+
row: dict[str, object] = {
|
|
157
|
+
"source_level": "protein",
|
|
158
|
+
**{name: group[name] for name in source.key_columns},
|
|
159
|
+
"member_ordinal": member_ordinal,
|
|
160
|
+
"match_ordinal": match_ordinal,
|
|
161
|
+
"protein_group": raw_group,
|
|
162
|
+
"protein_member": member,
|
|
163
|
+
"normalized_matching_key": member,
|
|
164
|
+
"match_status": status,
|
|
165
|
+
"fasta_id": None if protein is None else protein.get("id"),
|
|
166
|
+
"fasta_description": None if protein is None else protein.get("description"),
|
|
167
|
+
"fasta_sequence_length": (None if not isinstance(sequence, str) else len(sequence)),
|
|
168
|
+
}
|
|
169
|
+
for name in _OPTIONAL_FASTA_COLUMNS:
|
|
170
|
+
row[name] = None if protein is None else protein.get(name)
|
|
171
|
+
return row
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def _member_schema(
|
|
175
|
+
source: ProteinGroupInput,
|
|
176
|
+
proteins: pl.DataFrame,
|
|
177
|
+
) -> dict[str, pl.DataType | type[pl.DataType]]:
|
|
178
|
+
schema: dict[str, pl.DataType | type[pl.DataType]] = {
|
|
179
|
+
"source_level": pl.String,
|
|
180
|
+
**{name: source.frame.schema[name] for name in source.key_columns},
|
|
181
|
+
"member_ordinal": pl.Int64,
|
|
182
|
+
"match_ordinal": pl.Int64,
|
|
183
|
+
"protein_group": source.frame.schema[source.accession_column],
|
|
184
|
+
"protein_member": pl.String,
|
|
185
|
+
"normalized_matching_key": pl.String,
|
|
186
|
+
"match_status": pl.String,
|
|
187
|
+
"fasta_id": pl.String,
|
|
188
|
+
"fasta_description": pl.String,
|
|
189
|
+
"fasta_sequence_length": pl.UInt64,
|
|
190
|
+
}
|
|
191
|
+
for name in _OPTIONAL_FASTA_COLUMNS:
|
|
192
|
+
schema[name] = proteins.schema.get(name, pl.Null)
|
|
193
|
+
return schema
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _summary_schema() -> dict[str, type[pl.DataType]]:
|
|
197
|
+
return {
|
|
198
|
+
"reported_member_count": pl.UInt64,
|
|
199
|
+
"matched_member_count": pl.UInt64,
|
|
200
|
+
"unmatched_member_count": pl.UInt64,
|
|
201
|
+
"ambiguous_member_count": pl.UInt64,
|
|
202
|
+
"all_members_in_fasta": pl.Boolean,
|
|
203
|
+
"any_member_in_fasta": pl.Boolean,
|
|
204
|
+
"fasta_descriptions": pl.String,
|
|
205
|
+
"fasta_gene_names": pl.String,
|
|
206
|
+
"fasta_organism_names": pl.String,
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def _protein_aliases(rows: list[dict[str, object]]) -> dict[str, tuple[int, ...]]:
|
|
211
|
+
collected: dict[str, list[int]] = defaultdict(list)
|
|
212
|
+
for index, row in enumerate(rows):
|
|
213
|
+
aliases = {
|
|
214
|
+
value
|
|
215
|
+
for name in ("id", "accession")
|
|
216
|
+
if isinstance((value := row.get(name)), str) and value
|
|
217
|
+
}
|
|
218
|
+
for alias in aliases:
|
|
219
|
+
collected[alias].append(index)
|
|
220
|
+
return {name: tuple(indices) for name, indices in collected.items()}
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def _members(value: object, separator: str) -> tuple[str, ...]:
|
|
224
|
+
if not isinstance(value, str):
|
|
225
|
+
return ()
|
|
226
|
+
return tuple(token.strip() for token in value.split(separator) if token.strip())
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def _append_text(values: list[str], value: object) -> None:
|
|
230
|
+
if isinstance(value, str) and value and value not in values:
|
|
231
|
+
values.append(value)
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"""Immutable calculation values returned to the APB2 integration boundary."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Mapping
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
|
|
8
|
+
import polars as pl
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass(frozen=True, slots=True)
|
|
12
|
+
class PeptideCoverage:
|
|
13
|
+
"""Aggregate FASTA coverage for one APB2 feature level."""
|
|
14
|
+
|
|
15
|
+
feature_count: int
|
|
16
|
+
unique_sequence_count: int
|
|
17
|
+
matched_feature_count: int
|
|
18
|
+
unmatched_feature_count: int
|
|
19
|
+
match_site_count: int
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass(frozen=True, slots=True)
|
|
23
|
+
class PeptideLevelMatch:
|
|
24
|
+
"""Feature-aligned peptide facts and their aggregate coverage."""
|
|
25
|
+
|
|
26
|
+
summary: pl.DataFrame
|
|
27
|
+
coverage: PeptideCoverage
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass(frozen=True, slots=True)
|
|
31
|
+
class ProteinGroupCoverage:
|
|
32
|
+
"""Aggregate FASTA coverage for reported protein-group members."""
|
|
33
|
+
|
|
34
|
+
group_count: int
|
|
35
|
+
member_count: int
|
|
36
|
+
matched_member_count: int
|
|
37
|
+
unmatched_member_count: int
|
|
38
|
+
ambiguous_member_count: int
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@dataclass(frozen=True, slots=True)
|
|
42
|
+
class ProteinGroupMatch:
|
|
43
|
+
"""Lossless member expansion, group summaries, and source relation."""
|
|
44
|
+
|
|
45
|
+
members: pl.DataFrame
|
|
46
|
+
member_key_columns: tuple[str, ...]
|
|
47
|
+
summary: pl.DataFrame
|
|
48
|
+
relation: pl.DataFrame
|
|
49
|
+
coverage: ProteinGroupCoverage
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
@dataclass(frozen=True, slots=True)
|
|
53
|
+
class FastaAnnotationReports:
|
|
54
|
+
"""Persistable aggregate evidence from one annotation application."""
|
|
55
|
+
|
|
56
|
+
peptide_levels: Mapping[str, PeptideCoverage]
|
|
57
|
+
protein_groups: ProteinGroupCoverage | None
|
apb_fasta/cli.py
ADDED
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
"""Thin command-line composition for independent APB2 FASTA operations."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import sys
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Literal
|
|
8
|
+
|
|
9
|
+
from apb2.api import read_parsed_levels, write_parsed_levels
|
|
10
|
+
from cyclopts import App
|
|
11
|
+
from loguru import logger
|
|
12
|
+
|
|
13
|
+
from apb_fasta.api import FastaAnnotationParameters, FastaAnnotationResult, FastaAnnotator
|
|
14
|
+
|
|
15
|
+
app = App(
|
|
16
|
+
name="apb-fasta",
|
|
17
|
+
help="Verify peptides and merge protein annotations using FASTA files",
|
|
18
|
+
help_on_error=True,
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@app.command
|
|
23
|
+
def verify_peptides(
|
|
24
|
+
source: Path,
|
|
25
|
+
*fasta_paths: Path,
|
|
26
|
+
output: Path,
|
|
27
|
+
formats: tuple[str, ...] = ("uniprotkb", "refseq"),
|
|
28
|
+
backend: Literal["auto", "ahocorapy", "ahocorasick_rs"] = "auto",
|
|
29
|
+
il_equivalent: bool = False,
|
|
30
|
+
protein_group_separator: str = ";",
|
|
31
|
+
) -> int:
|
|
32
|
+
"""Verify stripped peptide sequences in SOURCE against FASTA_PATHS."""
|
|
33
|
+
try:
|
|
34
|
+
_require_new_target(source, output)
|
|
35
|
+
annotator = FastaAnnotator.read(
|
|
36
|
+
fasta_paths,
|
|
37
|
+
formats,
|
|
38
|
+
FastaAnnotationParameters(
|
|
39
|
+
protein_group_separator=protein_group_separator,
|
|
40
|
+
matcher_backend=backend,
|
|
41
|
+
il_equivalent=il_equivalent,
|
|
42
|
+
),
|
|
43
|
+
)
|
|
44
|
+
result = annotator.verify_peptides(read_parsed_levels(source))
|
|
45
|
+
_write_result(result, output)
|
|
46
|
+
_report_peptide_verification(result)
|
|
47
|
+
except (OSError, ValueError) as error:
|
|
48
|
+
logger.error(str(error))
|
|
49
|
+
return 1
|
|
50
|
+
logger.info("wrote peptide-verified APB2 result to {}", output)
|
|
51
|
+
return 0
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
@app.command
|
|
55
|
+
def merge_annotations(
|
|
56
|
+
source: Path,
|
|
57
|
+
*fasta_paths: Path,
|
|
58
|
+
output: Path,
|
|
59
|
+
formats: tuple[str, ...] = ("uniprotkb", "refseq"),
|
|
60
|
+
protein_group_separator: str = ";",
|
|
61
|
+
) -> int:
|
|
62
|
+
"""Merge FASTA annotations into the reported protein groups in SOURCE."""
|
|
63
|
+
try:
|
|
64
|
+
_require_new_target(source, output)
|
|
65
|
+
annotator = FastaAnnotator.read(
|
|
66
|
+
fasta_paths,
|
|
67
|
+
formats,
|
|
68
|
+
FastaAnnotationParameters(
|
|
69
|
+
protein_group_separator=protein_group_separator,
|
|
70
|
+
),
|
|
71
|
+
)
|
|
72
|
+
result = annotator.merge_annotations(read_parsed_levels(source))
|
|
73
|
+
_write_result(result, output)
|
|
74
|
+
_report_protein_annotations(result)
|
|
75
|
+
except (OSError, ValueError) as error:
|
|
76
|
+
logger.error(str(error))
|
|
77
|
+
return 1
|
|
78
|
+
logger.info("wrote FASTA-annotated APB2 result to {}", output)
|
|
79
|
+
return 0
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
@app.command
|
|
83
|
+
def run(
|
|
84
|
+
source: Path,
|
|
85
|
+
*fasta_paths: Path,
|
|
86
|
+
output: Path,
|
|
87
|
+
formats: tuple[str, ...] = ("uniprotkb", "refseq"),
|
|
88
|
+
backend: Literal["auto", "ahocorapy", "ahocorasick_rs"] = "auto",
|
|
89
|
+
il_equivalent: bool = False,
|
|
90
|
+
protein_group_separator: str = ";",
|
|
91
|
+
) -> int:
|
|
92
|
+
"""Verify peptides and merge protein annotations in one in-memory run."""
|
|
93
|
+
try:
|
|
94
|
+
_require_new_target(source, output)
|
|
95
|
+
annotator = FastaAnnotator.read(
|
|
96
|
+
fasta_paths,
|
|
97
|
+
formats,
|
|
98
|
+
FastaAnnotationParameters(
|
|
99
|
+
protein_group_separator=protein_group_separator,
|
|
100
|
+
matcher_backend=backend,
|
|
101
|
+
il_equivalent=il_equivalent,
|
|
102
|
+
),
|
|
103
|
+
)
|
|
104
|
+
result = annotator.annotate(read_parsed_levels(source))
|
|
105
|
+
_write_result(result, output)
|
|
106
|
+
_report_peptide_verification(result)
|
|
107
|
+
_report_protein_annotations(result)
|
|
108
|
+
except (OSError, ValueError) as error:
|
|
109
|
+
logger.error(str(error))
|
|
110
|
+
return 1
|
|
111
|
+
logger.info("wrote peptide-verified and FASTA-annotated APB2 result to {}", output)
|
|
112
|
+
return 0
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _require_new_target(source: Path, target: Path, /) -> None:
|
|
116
|
+
if source.resolve() == target.resolve():
|
|
117
|
+
raise ValueError("output must differ from source")
|
|
118
|
+
if target.exists():
|
|
119
|
+
raise ValueError(f"output already exists: {target}")
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _write_result(result: FastaAnnotationResult, output: Path) -> None:
|
|
123
|
+
write_parsed_levels(result.parsed, output)
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def _report_peptide_verification(result: FastaAnnotationResult) -> None:
|
|
127
|
+
for level, coverage in result.reports.peptide_levels.items():
|
|
128
|
+
logger.info(
|
|
129
|
+
"level={} peptides_in_fasta={}/{} unmatched={} unique_sequences={} match_sites={}",
|
|
130
|
+
level,
|
|
131
|
+
coverage.matched_feature_count,
|
|
132
|
+
coverage.feature_count,
|
|
133
|
+
coverage.unmatched_feature_count,
|
|
134
|
+
coverage.unique_sequence_count,
|
|
135
|
+
coverage.match_site_count,
|
|
136
|
+
)
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def _report_protein_annotations(result: FastaAnnotationResult) -> None:
|
|
140
|
+
coverage = result.reports.protein_groups
|
|
141
|
+
if coverage is None:
|
|
142
|
+
return
|
|
143
|
+
logger.info(
|
|
144
|
+
"protein_groups={} members_in_fasta={}/{} unmatched={} ambiguous={}",
|
|
145
|
+
coverage.group_count,
|
|
146
|
+
coverage.matched_member_count,
|
|
147
|
+
coverage.member_count,
|
|
148
|
+
coverage.unmatched_member_count,
|
|
149
|
+
coverage.ambiguous_member_count,
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def main() -> int:
|
|
154
|
+
"""Run the console application."""
|
|
155
|
+
result = app()
|
|
156
|
+
return int(result) if result is not None else 0
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
if __name__ == "__main__":
|
|
160
|
+
sys.exit(main())
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""User-selected FASTA annotation behavior."""
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
from typing import Literal
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
@dataclass(frozen=True, slots=True)
|
|
8
|
+
class FastaAnnotationParameters:
|
|
9
|
+
"""Configuration shared by peptide validation and protein annotation."""
|
|
10
|
+
|
|
11
|
+
protein_group_separator: str = ";"
|
|
12
|
+
matcher_backend: Literal["auto", "ahocorapy", "ahocorasick_rs"] = "auto"
|
|
13
|
+
il_equivalent: bool = False
|
|
14
|
+
|
|
15
|
+
def __post_init__(self) -> None:
|
|
16
|
+
"""Reject a separator that cannot define protein-group members."""
|
|
17
|
+
if not self.protein_group_separator:
|
|
18
|
+
raise ValueError("protein_group_separator must not be empty")
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
DEFAULT_FASTA_ANNOTATION_PARAMETERS = FastaAnnotationParameters()
|
apb_fasta/errors.py
ADDED
apb_fasta/integration.py
ADDED
|
@@ -0,0 +1,288 @@
|
|
|
1
|
+
"""Translate between APB2 result values and FASTA calculations."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from copy import deepcopy
|
|
6
|
+
from dataclasses import asdict
|
|
7
|
+
from typing import cast
|
|
8
|
+
|
|
9
|
+
import polars as pl
|
|
10
|
+
from apb2.api import JsonValue, ParsedLevels
|
|
11
|
+
|
|
12
|
+
from apb_fasta.calculation.matching import PeptideLevelInput
|
|
13
|
+
from apb_fasta.calculation.protein_groups import ProteinGroupInput
|
|
14
|
+
from apb_fasta.calculation.results import (
|
|
15
|
+
FastaAnnotationReports,
|
|
16
|
+
PeptideLevelMatch,
|
|
17
|
+
ProteinGroupMatch,
|
|
18
|
+
)
|
|
19
|
+
from apb_fasta.errors import FastaAnnotationError
|
|
20
|
+
|
|
21
|
+
FASTA_VALIDATION_NAME = "fasta_validation"
|
|
22
|
+
FASTA_SUMMARY_NAME = "fasta"
|
|
23
|
+
MEMBER_TABLE_NAME = "fasta_protein_group_members"
|
|
24
|
+
MEMBER_RELATION_NAME = "fasta_protein_group_membership"
|
|
25
|
+
PEPTIDE_PROPERTIES_NAME = "peptide_properties"
|
|
26
|
+
PEPTIDE_VERIFICATION_OPERATION = "peptide_verification"
|
|
27
|
+
PROTEIN_ANNOTATION_OPERATION = "protein_annotation"
|
|
28
|
+
PEPTIDE_PROPERTIES_OPERATION = "peptide_properties"
|
|
29
|
+
_PEPTIDE_COLUMN = "ProForma_peptide"
|
|
30
|
+
_RESERVED_MEMBER_COLUMNS = frozenset({"source_level", "member_ordinal", "match_ordinal"})
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def validate_protein_frame(proteins: pl.DataFrame, /) -> None:
|
|
34
|
+
"""Require the stable protein-fasta columns used by both calculations."""
|
|
35
|
+
required = {
|
|
36
|
+
"id",
|
|
37
|
+
"description",
|
|
38
|
+
"sequence",
|
|
39
|
+
"fasta_source_path",
|
|
40
|
+
"fasta_source_checksum",
|
|
41
|
+
"fasta_source_ordinal",
|
|
42
|
+
"fasta_record_ordinal",
|
|
43
|
+
}
|
|
44
|
+
missing = sorted(required.difference(proteins.columns))
|
|
45
|
+
if missing:
|
|
46
|
+
raise FastaAnnotationError(f"protein frame is missing required columns {missing}")
|
|
47
|
+
if proteins.get_column("id").null_count():
|
|
48
|
+
raise FastaAnnotationError("protein frame contains a missing id")
|
|
49
|
+
if proteins.get_column("sequence").null_count():
|
|
50
|
+
raise FastaAnnotationError("protein frame contains a missing sequence")
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def peptide_inputs(parsed: ParsedLevels, /) -> dict[str, PeptideLevelInput]:
|
|
54
|
+
"""Extract every peptide-derived level carrying APB2's canonical sequence."""
|
|
55
|
+
result: dict[str, PeptideLevelInput] = {}
|
|
56
|
+
for name, level in parsed.levels.items():
|
|
57
|
+
if name == "protein" or _PEPTIDE_COLUMN not in level.var.frame.columns:
|
|
58
|
+
continue
|
|
59
|
+
result[name] = PeptideLevelInput(
|
|
60
|
+
frame=level.var.frame,
|
|
61
|
+
sequence_column=_PEPTIDE_COLUMN,
|
|
62
|
+
accession_column=_fasta_accession_column(level.var.roles, level.var.frame),
|
|
63
|
+
)
|
|
64
|
+
return result
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def protein_group_input(parsed: ParsedLevels, /) -> ProteinGroupInput | None:
|
|
68
|
+
"""Extract the protein axis and its persisted FASTA-accession role."""
|
|
69
|
+
level = parsed.levels.get("protein")
|
|
70
|
+
if level is None:
|
|
71
|
+
return None
|
|
72
|
+
accession_column = _fasta_accession_column(level.var.roles, level.var.frame)
|
|
73
|
+
if accession_column is None:
|
|
74
|
+
raise FastaAnnotationError("protein level does not declare column_roles.fasta_accessions")
|
|
75
|
+
collisions = _RESERVED_MEMBER_COLUMNS.intersection(level.var.key_columns)
|
|
76
|
+
if collisions:
|
|
77
|
+
raise FastaAnnotationError(
|
|
78
|
+
f"protein variable keys collide with FASTA member columns {sorted(collisions)}"
|
|
79
|
+
)
|
|
80
|
+
return ProteinGroupInput(
|
|
81
|
+
frame=level.var.frame,
|
|
82
|
+
key_columns=level.var.key_columns,
|
|
83
|
+
accession_column=accession_column,
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def validate_peptide_output_names(
|
|
88
|
+
parsed: ParsedLevels,
|
|
89
|
+
matches: dict[str, PeptideLevelMatch],
|
|
90
|
+
/,
|
|
91
|
+
) -> None:
|
|
92
|
+
"""Reject peptide-verification collisions before constructing a replacement."""
|
|
93
|
+
_validate_operation_metadata(parsed, PEPTIDE_VERIFICATION_OPERATION)
|
|
94
|
+
for name in matches:
|
|
95
|
+
level_name = name
|
|
96
|
+
if FASTA_VALIDATION_NAME in parsed.levels[level_name].varm:
|
|
97
|
+
raise FastaAnnotationError(
|
|
98
|
+
f"level {name!r} already contains varm[{FASTA_VALIDATION_NAME!r}]"
|
|
99
|
+
)
|
|
100
|
+
if "fasta" in parsed.levels[level_name].metadata:
|
|
101
|
+
raise FastaAnnotationError(f"level {name!r} metadata already contains 'fasta'")
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def validate_protein_output_names(parsed: ParsedLevels, /) -> None:
|
|
105
|
+
"""Reject protein-annotation collisions before constructing a replacement."""
|
|
106
|
+
_validate_operation_metadata(parsed, PROTEIN_ANNOTATION_OPERATION)
|
|
107
|
+
protein = parsed.levels["protein"]
|
|
108
|
+
if FASTA_SUMMARY_NAME in protein.varm:
|
|
109
|
+
raise FastaAnnotationError(f"level 'protein' already contains varm[{FASTA_SUMMARY_NAME!r}]")
|
|
110
|
+
if MEMBER_TABLE_NAME in parsed.annotation_tables:
|
|
111
|
+
raise FastaAnnotationError(
|
|
112
|
+
f"result already contains annotation table {MEMBER_TABLE_NAME!r}"
|
|
113
|
+
)
|
|
114
|
+
if MEMBER_RELATION_NAME in parsed.feature_relations:
|
|
115
|
+
raise FastaAnnotationError(
|
|
116
|
+
f"result already contains feature relation {MEMBER_RELATION_NAME!r}"
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def apply_peptide_matches(
|
|
121
|
+
parsed: ParsedLevels,
|
|
122
|
+
matches: dict[str, PeptideLevelMatch],
|
|
123
|
+
/,
|
|
124
|
+
*,
|
|
125
|
+
requested_backend: str,
|
|
126
|
+
resolved_backend: str,
|
|
127
|
+
il_equivalent: bool,
|
|
128
|
+
protein_metadata: dict[str, JsonValue],
|
|
129
|
+
) -> tuple[ParsedLevels, FastaAnnotationReports]:
|
|
130
|
+
"""Attach peptide verification to a deep replacement of the APB2 result."""
|
|
131
|
+
validate_peptide_output_names(parsed, matches)
|
|
132
|
+
result = deepcopy(parsed)
|
|
133
|
+
for name, match in matches.items():
|
|
134
|
+
level_name = name
|
|
135
|
+
result.levels[level_name].varm[FASTA_VALIDATION_NAME] = match.summary.clone()
|
|
136
|
+
level_metadata = result.levels[level_name].metadata
|
|
137
|
+
level_metadata["fasta"] = {
|
|
138
|
+
PEPTIDE_VERIFICATION_OPERATION: cast(dict[str, JsonValue], asdict(match.coverage))
|
|
139
|
+
}
|
|
140
|
+
_record_operation_metadata(
|
|
141
|
+
result,
|
|
142
|
+
PEPTIDE_VERIFICATION_OPERATION,
|
|
143
|
+
{
|
|
144
|
+
"requested_backend": requested_backend,
|
|
145
|
+
"resolved_backend": resolved_backend,
|
|
146
|
+
"il_equivalent": il_equivalent,
|
|
147
|
+
**protein_metadata,
|
|
148
|
+
},
|
|
149
|
+
)
|
|
150
|
+
reports = FastaAnnotationReports(
|
|
151
|
+
peptide_levels={name: match.coverage for name, match in matches.items()},
|
|
152
|
+
protein_groups=None,
|
|
153
|
+
)
|
|
154
|
+
return result, reports
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def apply_protein_group_match(
|
|
158
|
+
parsed: ParsedLevels,
|
|
159
|
+
match: ProteinGroupMatch,
|
|
160
|
+
/,
|
|
161
|
+
*,
|
|
162
|
+
protein_group_separator: str,
|
|
163
|
+
protein_metadata: dict[str, JsonValue],
|
|
164
|
+
) -> tuple[ParsedLevels, FastaAnnotationReports]:
|
|
165
|
+
"""Attach protein FASTA annotations to a deep replacement of the APB2 result."""
|
|
166
|
+
validate_protein_output_names(parsed)
|
|
167
|
+
result = deepcopy(parsed)
|
|
168
|
+
result.levels["protein"].varm[FASTA_SUMMARY_NAME] = match.summary.clone()
|
|
169
|
+
result = result.with_annotation_table(
|
|
170
|
+
MEMBER_TABLE_NAME,
|
|
171
|
+
match.members.clone(),
|
|
172
|
+
match.member_key_columns,
|
|
173
|
+
{"producer": "apb-fasta", "schema_version": "1"},
|
|
174
|
+
).with_feature_relation(
|
|
175
|
+
MEMBER_RELATION_NAME,
|
|
176
|
+
MEMBER_TABLE_NAME,
|
|
177
|
+
"protein",
|
|
178
|
+
match.relation.clone(),
|
|
179
|
+
{"producer": "apb-fasta", "semantic": "member_of"},
|
|
180
|
+
)
|
|
181
|
+
_record_operation_metadata(
|
|
182
|
+
result,
|
|
183
|
+
PROTEIN_ANNOTATION_OPERATION,
|
|
184
|
+
{
|
|
185
|
+
"protein_group_separator": protein_group_separator,
|
|
186
|
+
**protein_metadata,
|
|
187
|
+
},
|
|
188
|
+
)
|
|
189
|
+
return result, FastaAnnotationReports(
|
|
190
|
+
peptide_levels={},
|
|
191
|
+
protein_groups=match.coverage,
|
|
192
|
+
)
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def apply_peptide_properties(
|
|
196
|
+
parsed: ParsedLevels,
|
|
197
|
+
properties: dict[str, pl.DataFrame],
|
|
198
|
+
/,
|
|
199
|
+
*,
|
|
200
|
+
protein_fasta_version: str,
|
|
201
|
+
) -> ParsedLevels:
|
|
202
|
+
"""Attach feature-aligned peptide properties to a deep replacement of the APB2 result."""
|
|
203
|
+
_validate_operation_metadata(parsed, PEPTIDE_PROPERTIES_OPERATION)
|
|
204
|
+
for name in properties:
|
|
205
|
+
if PEPTIDE_PROPERTIES_NAME in parsed.levels[name].varm:
|
|
206
|
+
raise FastaAnnotationError(
|
|
207
|
+
f"level {name!r} already contains varm[{PEPTIDE_PROPERTIES_NAME!r}]"
|
|
208
|
+
)
|
|
209
|
+
result = deepcopy(parsed)
|
|
210
|
+
for name, frame in properties.items():
|
|
211
|
+
result.levels[name].varm[PEPTIDE_PROPERTIES_NAME] = frame.clone()
|
|
212
|
+
_record_operation_metadata(
|
|
213
|
+
result,
|
|
214
|
+
PEPTIDE_PROPERTIES_OPERATION,
|
|
215
|
+
{"protein_fasta_version": protein_fasta_version},
|
|
216
|
+
)
|
|
217
|
+
return result
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def protein_frame_metadata(proteins: pl.DataFrame, /) -> dict[str, JsonValue]:
|
|
221
|
+
"""Return bounded source and schema provenance for one protein frame."""
|
|
222
|
+
sources = (
|
|
223
|
+
proteins.select(
|
|
224
|
+
"fasta_source_ordinal",
|
|
225
|
+
"fasta_source_path",
|
|
226
|
+
"fasta_source_checksum",
|
|
227
|
+
)
|
|
228
|
+
.unique(maintain_order=True)
|
|
229
|
+
.sort("fasta_source_ordinal")
|
|
230
|
+
.to_dicts()
|
|
231
|
+
)
|
|
232
|
+
return {
|
|
233
|
+
"protein_count": proteins.height,
|
|
234
|
+
"protein_columns": list(proteins.columns),
|
|
235
|
+
"sources": {
|
|
236
|
+
str(_json_scalar(source["fasta_source_ordinal"])): {
|
|
237
|
+
"ordinal": _json_scalar(source["fasta_source_ordinal"]),
|
|
238
|
+
"path": _json_scalar(source["fasta_source_path"]),
|
|
239
|
+
"checksum": _json_scalar(source["fasta_source_checksum"]),
|
|
240
|
+
}
|
|
241
|
+
for source in sources
|
|
242
|
+
},
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def _validate_operation_metadata(parsed: ParsedLevels, operation: str) -> None:
|
|
247
|
+
metadata = parsed.metadata.get("fasta")
|
|
248
|
+
if metadata is None:
|
|
249
|
+
return
|
|
250
|
+
if not isinstance(metadata, dict):
|
|
251
|
+
raise FastaAnnotationError("result metadata 'fasta' section is not an object")
|
|
252
|
+
provenance = metadata.get("provenance", {})
|
|
253
|
+
if not isinstance(provenance, dict):
|
|
254
|
+
raise FastaAnnotationError("result FASTA provenance is not an object")
|
|
255
|
+
if operation in provenance:
|
|
256
|
+
raise FastaAnnotationError(f"result already contains FASTA operation {operation!r}")
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def _record_operation_metadata(
|
|
260
|
+
parsed: ParsedLevels,
|
|
261
|
+
operation: str,
|
|
262
|
+
operation_metadata: dict[str, JsonValue],
|
|
263
|
+
) -> None:
|
|
264
|
+
existing = parsed.metadata.get("fasta")
|
|
265
|
+
metadata: dict[str, JsonValue]
|
|
266
|
+
if existing is None:
|
|
267
|
+
metadata = {"schema_version": "2", "provenance": {}}
|
|
268
|
+
else:
|
|
269
|
+
metadata = cast(dict[str, JsonValue], deepcopy(existing))
|
|
270
|
+
provenance = metadata.setdefault("provenance", {})
|
|
271
|
+
if not isinstance(provenance, dict):
|
|
272
|
+
raise FastaAnnotationError("result FASTA provenance is not an object")
|
|
273
|
+
provenance[operation] = operation_metadata
|
|
274
|
+
parsed.metadata["fasta"] = metadata
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
def _fasta_accession_column(
|
|
278
|
+
roles: dict[str, str],
|
|
279
|
+
frame: pl.DataFrame,
|
|
280
|
+
) -> str | None:
|
|
281
|
+
value = roles.get("fasta_accessions")
|
|
282
|
+
return value if isinstance(value, str) and value in frame.columns else None
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
def _json_scalar(value: object) -> bool | int | float | str | None:
|
|
286
|
+
if value is None or isinstance(value, bool | int | float | str):
|
|
287
|
+
return value
|
|
288
|
+
raise FastaAnnotationError(f"protein source provenance contains {type(value).__name__}")
|
apb_fasta/py.typed
ADDED
|
File without changes
|
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: apb-fasta
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: FASTA validation and protein annotation for APB2 results.
|
|
5
|
+
Keywords: proteomics,fasta,anndata,protein annotation,mass spectrometry
|
|
6
|
+
Author: Witold Wolski
|
|
7
|
+
Author-email: Witold Wolski <wew@fgcz.ethz.ch>
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Classifier: Development Status :: 3 - Alpha
|
|
11
|
+
Classifier: Intended Audience :: Science/Research
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
15
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
16
|
+
Classifier: Typing :: Typed
|
|
17
|
+
Requires-Dist: apb2>=0.1,<0.2
|
|
18
|
+
Requires-Dist: cyclopts>=4,<5
|
|
19
|
+
Requires-Dist: loguru>=0.7,<1
|
|
20
|
+
Requires-Dist: polars>=1.43,<2
|
|
21
|
+
Requires-Dist: protein-fasta>=0.3,<0.4
|
|
22
|
+
Requires-Dist: prozor>=0.1,<0.2
|
|
23
|
+
Requires-Python: >=3.13
|
|
24
|
+
Project-URL: Documentation, https://anndata-omics-bridge.github.io/apb-fasta/
|
|
25
|
+
Project-URL: Repository, https://github.com/anndata-omics-bridge/apb-fasta
|
|
26
|
+
Project-URL: Issues, https://github.com/anndata-omics-bridge/apb-fasta/issues
|
|
27
|
+
Description-Content-Type: text/markdown
|
|
28
|
+
|
|
29
|
+
# apb-fasta
|
|
30
|
+
|
|
31
|
+
FASTA verification and protein annotation for APB2 results.
|
|
32
|
+
|
|
33
|
+
**[Online documentation](https://anndata-omics-bridge.github.io/apb-fasta/)** or its [source index](https://github.com/anndata-omics-bridge/apb-fasta/blob/main/docs/index.md).
|
|
34
|
+
|
|
35
|
+
The first release exposes two independent operations:
|
|
36
|
+
|
|
37
|
+
- verify that APB2's modification-stripped peptide sequences occur in the supplied FASTA database;
|
|
38
|
+
- merge FASTA annotations for every reported protein-group member without collapsing the group to its leading accession.
|
|
39
|
+
|
|
40
|
+
Protein inference with Prozor is the planned third operation. It will remain explicit and opt-in.
|
|
41
|
+
|
|
42
|
+
## Installation
|
|
43
|
+
|
|
44
|
+
APB FASTA requires Python 3.13 or later.
|
|
45
|
+
|
|
46
|
+
```bash
|
|
47
|
+
pip install apb-fasta
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
## Python API
|
|
51
|
+
|
|
52
|
+
Read the FASTA files once and load the APB2 result:
|
|
53
|
+
|
|
54
|
+
```python
|
|
55
|
+
from pathlib import Path
|
|
56
|
+
|
|
57
|
+
from apb2.api import read_parsed_levels, write_parsed_levels
|
|
58
|
+
from apb_fasta.api import FastaAnnotator
|
|
59
|
+
|
|
60
|
+
annotator = FastaAnnotator.read((Path("human.fasta"), Path("contaminants.fasta")))
|
|
61
|
+
parsed = read_parsed_levels(Path("input.h5mu"))
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
Verify peptides only:
|
|
65
|
+
|
|
66
|
+
```python
|
|
67
|
+
verified = annotator.verify_peptides(parsed)
|
|
68
|
+
print(verified.reports.peptide_levels)
|
|
69
|
+
write_parsed_levels(verified.parsed, Path("verified.h5mu"))
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
Merge protein annotations only:
|
|
73
|
+
|
|
74
|
+
```python
|
|
75
|
+
annotated = annotator.merge_annotations(parsed)
|
|
76
|
+
print(annotated.reports.protein_groups)
|
|
77
|
+
write_parsed_levels(annotated.parsed, Path("annotated.h5mu"))
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
Apply both operations in memory:
|
|
81
|
+
|
|
82
|
+
```python
|
|
83
|
+
complete = annotator.annotate(parsed)
|
|
84
|
+
write_parsed_levels(complete.parsed, Path("complete.h5mu"))
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
The operations can also be chained explicitly without an intermediate file:
|
|
88
|
+
|
|
89
|
+
```python
|
|
90
|
+
verified = annotator.verify_peptides(parsed)
|
|
91
|
+
complete = annotator.merge_annotations(verified.parsed)
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
The annotator validates and binds the reusable protein frame once. Every method accepts one canonical `ParsedLevels` value and returns an immutable `FastaAnnotationResult` containing its replacement plus typed operation reports. `apb_fasta` neither opens result files nor receives raw AnnData or MuData objects.
|
|
95
|
+
|
|
96
|
+
## CLI
|
|
97
|
+
|
|
98
|
+
Verify peptides only:
|
|
99
|
+
|
|
100
|
+
```bash
|
|
101
|
+
apb-fasta verify-peptides input.h5mu human.fasta contaminants.fasta --output verified.h5mu
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
Merge protein annotations only:
|
|
105
|
+
|
|
106
|
+
```bash
|
|
107
|
+
apb-fasta merge-annotations input.h5mu human.fasta contaminants.fasta --output annotated.h5mu
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
Apply both operations together:
|
|
111
|
+
|
|
112
|
+
```bash
|
|
113
|
+
apb-fasta run input.h5mu human.fasta contaminants.fasta --output complete.h5mu
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
Peptide verification adds feature-aligned `varm["fasta_validation"]` tables. Protein annotation adds protein-aligned `varm["fasta"]`, the lossless `fasta_protein_group_members` annotation table, and its directed relation to the protein axis.
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
apb_fasta/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
2
|
+
apb_fasta/api.py,sha256=2squHIacOk51JchAEupW2A6rrvCCGBFf2tgBkbhzSL0,6202
|
|
3
|
+
apb_fasta/calculation/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
4
|
+
apb_fasta/calculation/matching.py,sha256=Sw8WeWyW89hGzkxAgAWmh_mTccxC9hkfbewPiSMxqg4,8097
|
|
5
|
+
apb_fasta/calculation/peptide_properties.py,sha256=80A_PjI_kgG0_kX_KjccRmAkLfkOaAc_oL6Dhy-DGl8,982
|
|
6
|
+
apb_fasta/calculation/protein_groups.py,sha256=FKg7kqvb1Tsisgd-HiJIUlZOTdGq8bufJqZIfNWGQ9o,8178
|
|
7
|
+
apb_fasta/calculation/results.py,sha256=9iOPBPfeekJ04W7jErm0k4uskPMkSumXx_8sd0IdE9s,1474
|
|
8
|
+
apb_fasta/cli.py,sha256=V6VqCzmuogjZj8EaIh53SWRhhkVoIgQqnxWat2eYMAc,5064
|
|
9
|
+
apb_fasta/configuration.py,sha256=blQIb9ywCp00Xr6ypWfQzUQ-x8aF7aegAB3veD0cgF4,706
|
|
10
|
+
apb_fasta/errors.py,sha256=_oLU8leci5hMXN9XJa0rD6HEnvfzv8CYlGn3e50u7P0,168
|
|
11
|
+
apb_fasta/integration.py,sha256=TcH0jtL_Aby2RKZydhLD_-qCXsSqv0wEYYKZ8_xWY6I,10643
|
|
12
|
+
apb_fasta/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
13
|
+
apb_fasta-0.1.0.dist-info/licenses/LICENSE,sha256=OPECQFbzDYwcWIzs8D8qw6o9AZsc97_u2fwqbe-ojAM,1070
|
|
14
|
+
apb_fasta-0.1.0.dist-info/WHEEL,sha256=cmC5s21ojypbVslldL7IJq3hjZH-tINy4rziKePFsG0,81
|
|
15
|
+
apb_fasta-0.1.0.dist-info/entry_points.txt,sha256=mPjMZ9aiQIs5fOumE4mwi2Kf-dm8BxDeLg-RFaXyNVQ,50
|
|
16
|
+
apb_fasta-0.1.0.dist-info/METADATA,sha256=McjRlQmEL64BaJuEAU2_Bd0kR7PRTFqURKySPP99knE,3864
|
|
17
|
+
apb_fasta-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Witold Wolski
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|