fusion-function 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,149 @@
1
+ from __future__ import annotations
2
+
3
+ from pathlib import Path
4
+ from typing import TYPE_CHECKING, Literal, NotRequired, TypedDict
5
+
6
+ from .reference import get_reference
7
+
8
+ if TYPE_CHECKING:
9
+ from .data import ReferenceReader
10
+
11
+
12
+ Strand = Literal[-1, 1]
13
+
14
+
15
+ class ProteinFeatureAnnotation(TypedDict):
16
+ name: str | None
17
+ entry_type: str | None
18
+ interpro_id: str | None
19
+
20
+
21
+ class TranscriptExon(TypedDict):
22
+ exon_number: int
23
+ chromosome: str | None
24
+ genomic_start: int
25
+ genomic_end: int
26
+ premrna_start: int
27
+ premrna_end: int
28
+
29
+
30
+ class CDSBlock(TypedDict):
31
+ chromosome: str | None
32
+ genomic_start: int
33
+ genomic_end: int
34
+ premrna_start: int
35
+ premrna_end: int
36
+ strand: Strand
37
+ assembly_name: str | None
38
+ cds_start: int
39
+ cds_end: int
40
+
41
+
42
+ class GenomicSegment(TypedDict):
43
+ chromosome: str | None
44
+ start: int
45
+ end: int
46
+ strand: Strand
47
+ assembly_name: str | None
48
+
49
+
50
+ class FeatureEvidence(TypedDict):
51
+ code: str
52
+ source: NotRequired[str]
53
+ id: NotRequired[str]
54
+
55
+
56
+ class ProteinFeature(TypedDict):
57
+ feature_id: str | None
58
+ source: str | None
59
+ interpro_id: str | None
60
+ start: int
61
+ end: int
62
+ cds_start: int
63
+ cds_end: int
64
+ chromosome: str | None
65
+ genomic_start: int | None
66
+ genomic_end: int | None
67
+ strand: Strand | None
68
+ assembly_name: str | None
69
+ description: str | None
70
+ panther_subfamily_id: str | None
71
+ panther_subfamily_description: str | None
72
+ interpro_name: NotRequired[str | None]
73
+ interpro_entry_type: NotRequired[str | None]
74
+ genomic_segments: NotRequired[list[GenomicSegment]]
75
+ feature_type: NotRequired[str]
76
+ uniprot_accession: NotRequired[str]
77
+ uniprot_isoform: NotRequired[str]
78
+ evidence: NotRequired[list[FeatureEvidence]]
79
+
80
+
81
+ def feature_identity(feature: ProteinFeature) -> tuple[object, ...]:
82
+ """Collapse only identical feature intervals, never overlapping boundaries.
83
+
84
+ Integrated signatures may share an InterPro ID and exact interval. Features
85
+ without one need their source, accession and description to distinguish, for
86
+ example, different ligand-binding annotations at the same residue.
87
+ """
88
+ identity = (
89
+ (feature["interpro_id"],)
90
+ if feature["interpro_id"]
91
+ else (
92
+ feature["source"],
93
+ feature["feature_id"],
94
+ feature["description"],
95
+ feature.get("uniprot_accession"),
96
+ feature.get("uniprot_isoform"),
97
+ )
98
+ )
99
+ return (
100
+ *identity,
101
+ feature["start"],
102
+ feature["end"],
103
+ feature.get("feature_type") or feature.get("interpro_entry_type"),
104
+ )
105
+
106
+
107
+ class ReferenceSpliceSite(TypedDict):
108
+ type: Literal["donor", "acceptor"]
109
+ exon_number: int
110
+ genomic_position: int
111
+ premrna_position: int
112
+ disruption_start: int
113
+ disruption_end: int
114
+
115
+
116
+ class TranscriptProteinFeatureResult(TypedDict):
117
+ translation_id: str | None
118
+ protein_length: int | None
119
+ chromosome: str | None
120
+ strand: Strand
121
+ transcript_genomic_start: int
122
+ transcript_genomic_end: int
123
+ premrna_length: int
124
+ premrna_sequence: NotRequired[str]
125
+ transcript_exons: list[TranscriptExon]
126
+ cds_blocks: list[CDSBlock]
127
+ cds_start_phase: NotRequired[int]
128
+ protein_features: list[ProteinFeature]
129
+ splice_sites: list[ReferenceSpliceSite]
130
+ assembly_name: NotRequired[str | None]
131
+
132
+
133
+ class EnsemblError(TypedDict):
134
+ error: str
135
+
136
+
137
+ ProteinFeatureResponse = TranscriptProteinFeatureResult | EnsemblError
138
+
139
+
140
+ def get_protein_domains(
141
+ transcript_id: str,
142
+ *,
143
+ reference: ReferenceReader | None = None,
144
+ database: str | Path | None = None,
145
+ release: int | None = None,
146
+ ) -> ProteinFeatureResponse:
147
+ """Read prepared transcript structure, sequence and protein features locally."""
148
+ reader = get_reference(reference=reference, database=database, release=release)
149
+ return reader.get_transcript(transcript_id)