fusion-function 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,82 @@
1
+ {
2
+ "catalog_version": 1,
3
+ "references": [
4
+ {
5
+ "species": "homo_sapiens",
6
+ "assembly": "GRCh38",
7
+ "release": 116,
8
+ "build_revision": 1,
9
+ "filename": "homo_sapiens.GRCh38.ensembl-116.ff-v1.build-1.sqlite.gz",
10
+ "compression": "gzip",
11
+ "archive_bytes": 2245743354,
12
+ "archive_sha256": "65222b7637ecfcb1ecff9d6862eeb20da9ac2d471a49d8c05568bc3f978c5866",
13
+ "database_bytes": 2727596032,
14
+ "database_sha256": "d3ffd6b684598d392f3d4525bd7f368ed3563dff8534a2c80d8cb361350aa1d4",
15
+ "doi": "10.5281/zenodo.23111108",
16
+ "url": "https://zenodo.org/records/23111108/files/homo_sapiens.GRCh38.ensembl-116.ff-v1.build-1.sqlite.gz?download=1",
17
+ "metadata": {
18
+ "format_version": "1",
19
+ "preprocessing_version": "1",
20
+ "cds_mapping_version": "2",
21
+ "feature_annotation_version": "3",
22
+ "transcript_payload_codec": "zlib-json-v1",
23
+ "sequence_chunk_size": "1048576",
24
+ "sequence_codec": "zlib"
25
+ },
26
+ "data_source_notices": {
27
+ "Ensembl": {
28
+ "terms": "Unrestricted project-generated data; third-party constraints may apply.",
29
+ "url": "https://www.ensembl.org/info/about/legal/disclaimer.html"
30
+ },
31
+ "UniProt Consortium": {
32
+ "terms": "CC BY 4.0. Credit UniProt, link to the license and identify modifications.",
33
+ "url": "https://www.uniprot.org/help/license",
34
+ "license_url": "https://creativecommons.org/licenses/by/4.0/"
35
+ },
36
+ "InterPro Consortium": {
37
+ "terms": "Current InterPro downloads: CC0 1.0. Retain historical-source notices.",
38
+ "url": "https://interpro-documentation.readthedocs.io/en/latest/license.html",
39
+ "license_url": "https://creativecommons.org/publicdomain/zero/1.0/"
40
+ },
41
+ "PANTHER": {
42
+ "terms": "Classification release 14.1 and 17.0 READMEs carry GPL-2.0-or-later notices. Confirm terms for the imported release and derived classifications before redistribution.",
43
+ "url": "https://data.pantherdb.org/ftp/sequence_classifications/"
44
+ },
45
+ "Ensembl member annotations": {
46
+ "terms": "Member-source terms are not replaced by Ensembl or InterPro terms. PROSITE database terms are CC BY-NC-ND 4.0 with commercial licensing; SMART models require a license. Confirm terms for derived match annotations.",
47
+ "url": "https://prosite.expasy.org/prosite_license.html",
48
+ "smart_url": "https://smart.embl.de/about.cgi"
49
+ }
50
+ },
51
+ "source_metadata": {
52
+ "format_version": "1",
53
+ "release": "116",
54
+ "species": "homo_sapiens",
55
+ "core_database": "homo_sapiens_core_116_38",
56
+ "assembly": "GRCh38",
57
+ "created_utc": "2026-10-01T01:17:06.715364+00:00",
58
+ "sequence_chunk_size": "1048576",
59
+ "sequence_codec": "zlib",
60
+ "tables": "[\"meta\", \"coord_system\", \"seq_region\", \"gene\", \"transcript\", \"exon\", \"exon_transcript\", \"translation\", \"analysis\", \"protein_feature\", \"interpro\", \"xref\", \"external_db\", \"object_xref\"]",
61
+ "sequence_types": "[\"dna\", \"pep\"]",
62
+ "counts": "{\"meta\": 235, \"coord_system\": 9, \"seq_region\": 268640, \"gene\": 87735, \"transcript\": 672310, \"exon\": 1422168, \"exon_transcript\": 5252227, \"translation\": 384051, \"analysis\": 100, \"protein_feature\": 6690040, \"interpro\": 25300, \"xref\": 2751838, \"external_db\": 501, \"object_xref\": 7737725, \"dna_sequences\": 706, \"pep_sequences\": 382428}",
63
+ "preprocessing_version": "1",
64
+ "preprocessing_counts": "{\"ready\": 368066, \"noncoding\": 288242, \"error\": 16002, \"protein_features\": 6296276}",
65
+ "preprocessed_utc": "2026-10-02T17:30:44.172625+00:00",
66
+ "interpro_entries_sha256": "1d247a42670d4a2913cec9f2a58813135c736d65128c1078d9c5eac797ffef1c",
67
+ "interpro_entry_types_loaded": "true",
68
+ "interpro_historical_entries": "{\"IPR001423\": \"109.0\", \"IPR004734\": \"106.0\", \"IPR009133\": \"109.0\", \"IPR010307\": \"108.0\", \"IPR017198\": \"106.0\", \"IPR018798\": \"107.0\", \"IPR019321\": \"108.0\", \"IPR024812\": \"109.0\", \"IPR026516\": \"109.0\", \"IPR026570\": \"109.0\", \"IPR029065\": \"107.0\", \"IPR031101\": \"109.0\", \"IPR032922\": \"107.0\", \"IPR033489\": \"109.0\", \"IPR033509\": \"109.0\", \"IPR039163\": \"109.0\", \"IPR039681\": \"109.0\", \"IPR039715\": \"109.0\", \"IPR039747\": \"109.0\", \"IPR042012\": \"106.0\", \"IPR043040\": \"107.0\", \"IPR043041\": \"107.0\", \"IPR043042\": \"107.0\", \"IPR043504\": \"107.0\", \"IPR044613\": \"109.0\", \"IPR044986\": \"109.0\", \"IPR045150\": \"107.0\", \"IPR048456\": \"106.0\", \"IPR050058\": \"108.0\", \"IPR050079\": \"108.0\", \"IPR050082\": \"108.0\", \"IPR050089\": \"107.0\", \"IPR050115\": \"108.0\", \"IPR050123\": \"107.0\", \"IPR050149\": \"108.0\", \"IPR050174\": \"108.0\", \"IPR050203\": \"108.0\", \"IPR050223\": \"108.0\", \"IPR050224\": \"108.0\", \"IPR050247\": \"108.0\", \"IPR050332\": \"109.0\", \"IPR050369\": \"108.0\", \"IPR050374\": \"108.0\", \"IPR050382\": \"109.0\", \"IPR050383\": \"109.0\", \"IPR050514\": \"109.0\", \"IPR050518\": \"108.0\", \"IPR050527\": \"107.0\", \"IPR050544\": \"109.0\", \"IPR050569\": \"109.0\", \"IPR050628\": \"107.0\", \"IPR050650\": \"107.0\", \"IPR050657\": \"107.0\", \"IPR050741\": \"107.0\", \"IPR050778\": \"107.0\", \"IPR050802\": \"107.0\", \"IPR050803\": \"106.0\", \"IPR050888\": \"108.0\", \"IPR050904\": \"108.0\", \"IPR050912\": \"107.0\", \"IPR050915\": \"108.0\", \"IPR050937\": \"107.0\", \"IPR050951\": \"107.0\", \"IPR050958\": \"109.0\", \"IPR050969\": \"109.0\", \"IPR051012\": \"107.0\", \"IPR051022\": \"107.0\", \"IPR051027\": \"107.0\", \"IPR051029\": \"107.0\", \"IPR051061\": \"107.0\", \"IPR051064\": \"107.0\", \"IPR051110\": \"109.0\", \"IPR051170\": \"109.0\", \"IPR051179\": \"108.0\", \"IPR051205\": \"108.0\", \"IPR051232\": \"108.0\", \"IPR051237\": \"107.0\", \"IPR051266\": \"109.0\", \"IPR051270\": \"107.0\", \"IPR051288\": \"108.0\", \"IPR051334\": \"107.0\", \"IPR051387\": \"108.0\", \"IPR051412\": \"109.0\", \"IPR051425\": \"109.0\", \"IPR051441\": \"109.0\", \"IPR051458\": \"109.0\", \"IPR051478\": \"107.0\", \"IPR051550\": \"109.0\", \"IPR051569\": \"107.0\", \"IPR051630\": \"108.0\", \"IPR051637\": \"109.0\", \"IPR051647\": \"109.0\", \"IPR051694\": \"109.0\", \"IPR051728\": \"109.0\", \"IPR051742\": \"108.0\", \"IPR051827\": \"109.0\", \"IPR051969\": \"107.0\", \"IPR052083\": \"109.0\", \"IPR052097\": \"108.0\", \"IPR052130\": \"109.0\", \"IPR052145\": \"109.0\", \"IPR052273\": \"109.0\", \"IPR052301\": \"109.0\", \"IPR052346\": \"106.0\", \"IPR052376\": \"109.0\", \"IPR052392\": \"109.0\", \"IPR052504\": \"109.0\", \"IPR052555\": \"109.0\", \"IPR052579\": \"109.0\", \"IPR052660\": \"109.0\", \"IPR052696\": \"109.0\", \"IPR052881\": \"109.0\", \"IPR053019\": \"109.0\", \"IPR053099\": \"109.0\", \"IPR053374\": \"107.0\", \"IPR054602\": \"107.0\", \"IPR055047\": \"107.0\", \"IPR055074\": \"107.0\", \"IPR055351\": \"107.0\", \"IPR056552\": \"107.0\"}",
69
+ "interpro_historical_sources": "[{\"release\": \"109.0\", \"sha256\": \"00b86a43b14652470d1110133b26f91314d5bddf9f164da76c494d7b322752ec\", \"url\": \"https://ftp.ebi.ac.uk/pub/databases/interpro/releases/109.0/entry.list\"}, {\"release\": \"108.0\", \"sha256\": \"11ddc8e6624a3ddaf40d5528e5aaf0a87308bd1c58a3f7087208091ab5eefc29\", \"url\": \"https://ftp.ebi.ac.uk/pub/databases/interpro/releases/108.0/entry.list\"}, {\"release\": \"107.0\", \"sha256\": \"333770e997fc6b88a836650ac5c9475b8934b4beb52a5a084d7d2e10dcc4af05\", \"url\": \"https://ftp.ebi.ac.uk/pub/databases/interpro/releases/107.0/entry.list\"}, {\"release\": \"106.0\", \"sha256\": \"15d98e0805a05cdbe4935ab2086e65c6e204181ae87c0b06cc33a315eea4db4a\", \"url\": \"https://ftp.ebi.ac.uk/pub/databases/interpro/releases/106.0/entry.list\"}]",
70
+ "interpro_preserved_metadata_sha256": "1d247a42670d4a2913cec9f2a58813135c736d65128c1078d9c5eac797ffef1c",
71
+ "panther_classifications_source": "{\"sha256\": \"0635357eeae76cb6dce8ce498f20e7ad5e101fe5a81717003c97acb2ed009b32\", \"url\": \"https://data.pantherdb.org/ftp/sequence_classifications/19.0/PANTHER_Sequence_Classification_files/PTHR19.0_human\", \"version\": \"19.0\"}",
72
+ "feature_annotation_version": "3",
73
+ "uniprot_features_source": "{\"sha256\": \"3296aef04049bbf83a21f3335436c9f2d3a00f6186fc10e4e66854ac37b2e5ab\", \"url\": \"https://rest.uniprot.org/uniprotkb/stream?format=xml&compressed=true&query=organism_id%3A9606%20AND%20reviewed%3Atrue\"}",
74
+ "uniprot_counts": "{\"entries\": 20431, \"features\": 332987, \"matched_proteins\": 97684, \"reviewed_human_entries\": 20431, \"sequence_mismatches\": 19693}",
75
+ "cds_mapping_version": "2",
76
+ "preprocessing_errors": "{\"CDS/peptide length mismatch\": {\"count\": 14362, \"example_transcripts\": [\"ENST00000361390\", \"ENST00000361453\", \"ENST00000362079\"]}, \"Unsupported transcript coordinate system\": {\"count\": 1640, \"example_transcripts\": [\"LRG_1t1\", \"LRG_1000t1\", \"LRG_1001t1\"]}}",
77
+ "uniprot_lookup_fingerprint": "7f8c19064714799b35cca12a4320c26e40f21faef2ac0e9a3220ad005779ca33",
78
+ "transcript_payload_codec": "zlib-json-v1"
79
+ }
80
+ }
81
+ ]
82
+ }
@@ -0,0 +1,315 @@
1
+ """Reviewed human UniProt features, imported once during reference preparation.
2
+
3
+ XML is streamed with the standard library. Isoforms are reconstructed only from
4
+ explicit UniProt splice variants; canonical features are transferred only across
5
+ unchanged sequence. Ensembl mappings additionally require an exact whole-protein
6
+ sequence match. No gene-name matching, approximate alignment or runtime network
7
+ access is used.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import json
13
+ import sqlite3
14
+ from collections import OrderedDict
15
+ from collections.abc import Iterator
16
+ from pathlib import Path
17
+ from typing import TypedDict
18
+ from xml.etree import ElementTree as ET
19
+
20
+ from .ensembl import FeatureEvidence
21
+
22
+
23
+ UNIPROT_HUMAN_URL = (
24
+ "https://rest.uniprot.org/uniprotkb/stream?format=xml&compressed=true"
25
+ "&query=organism_id%3A9606%20AND%20reviewed%3Atrue"
26
+ )
27
+ NS = "{http://uniprot.org/uniprot}"
28
+ FEATURE_TYPES = {
29
+ "domain": "domain",
30
+ "active site": "active_site",
31
+ "binding site": "binding_site",
32
+ "site": "conserved_site",
33
+ "motif": "motif",
34
+ "short sequence motif": "motif",
35
+ }
36
+
37
+
38
+ class CuratedFeature(TypedDict):
39
+ feature_id: str
40
+ feature_type: str
41
+ start: int
42
+ end: int
43
+ description: str
44
+ uniprot_accession: str
45
+ uniprot_isoform: str
46
+ evidence: list[FeatureEvidence]
47
+
48
+
49
+ class CuratedProtein(TypedDict):
50
+ sequence: str
51
+ features: list[CuratedFeature]
52
+ accessions: list[str]
53
+ protein_ids: list[str]
54
+
55
+
56
+ def exact_interval(element: ET.Element) -> tuple[int, int] | None:
57
+ """Reject uncertain/fuzzy coordinates rather than inventing exact bounds."""
58
+ location = element.find(NS + "location")
59
+ if location is None:
60
+ return None
61
+ position = location.find(NS + "position")
62
+ nodes = (
63
+ [position, position]
64
+ if position is not None
65
+ else [location.find(NS + "begin"), location.find(NS + "end")]
66
+ )
67
+ if any(
68
+ node is None
69
+ or node.get("status") not in {None, "certain"}
70
+ or not (node.get("position") or "").isdigit()
71
+ for node in nodes
72
+ ):
73
+ return None
74
+ start, end = [int(node.attrib["position"]) for node in nodes if node is not None]
75
+ return (start, end) if 1 <= start <= end else None
76
+
77
+
78
+ def entry_proteins(entry: ET.Element) -> Iterator[CuratedProtein]:
79
+ """Yield canonical/described isoforms with safely mapped functional features."""
80
+ if entry.get("dataset") != "Swiss-Prot" or not any(
81
+ ref.get("type") == "NCBI Taxonomy" and ref.get("id") == "9606"
82
+ for ref in entry.findall(f"{NS}organism/{NS}dbReference")
83
+ ):
84
+ return
85
+ accessions = [node.text for node in entry.findall(NS + "accession") if node.text]
86
+ canonical = "".join((entry.findtext(NS + "sequence") or "").split())
87
+ if not accessions or not canonical:
88
+ raise ValueError("Reviewed human UniProt entry lacks accession or sequence")
89
+ accession = accessions[0]
90
+ evidence: dict[str, FeatureEvidence] = {}
91
+ for node in entry.findall(NS + "evidence"):
92
+ item: FeatureEvidence = {"code": node.get("type", "")}
93
+ reference = node.find(f"{NS}source/{NS}dbReference")
94
+ if reference is not None:
95
+ item["source"] = reference.get("type", "")
96
+ item["id"] = reference.get("id", "")
97
+ evidence[node.get("key", "")] = item
98
+ features = entry.findall(NS + "feature")
99
+ variants = {node.get("id"): node for node in features if node.get("type") == "splice variant"}
100
+ isoforms = entry.findall(f"{NS}comment[@type='alternative products']/{NS}isoform")
101
+ # A canonical entry need not declare alternative products.
102
+ sequences: dict[str, tuple[str, list[tuple[int, int, int]]]] = {accession: (canonical, [])}
103
+ displayed = accession
104
+ for isoform in isoforms:
105
+ sequence_element = isoform.find(NS + "sequence")
106
+ ids = [node.text for node in isoform.findall(NS + "id") if node.text]
107
+ if sequence_element is None or not ids:
108
+ continue
109
+ edits: list[tuple[int, int, str]] = []
110
+ if sequence_element.get("type") == "displayed":
111
+ displayed = ids[0]
112
+ elif sequence_element.get("type") == "described":
113
+ refs = (sequence_element.get("ref") or "").split()
114
+ if not refs:
115
+ continue
116
+ for variant_id in refs:
117
+ variant = variants.get(variant_id)
118
+ interval = exact_interval(variant) if variant is not None else None
119
+ if variant is None or interval is None or interval[1] > len(canonical):
120
+ break
121
+ start, end = interval
122
+ original = variant.findtext(NS + "original")
123
+ if original and original != canonical[start - 1 : end]:
124
+ break
125
+ replacement = variant.findtext(NS + "variation") or ""
126
+ edits.append((start, end, replacement))
127
+ if len(edits) != len(refs):
128
+ continue
129
+ else:
130
+ # External/not-described isoform sequences cannot be reconstructed.
131
+ continue
132
+ edits.sort()
133
+ if any(left[1] >= right[0] for left, right in zip(edits, edits[1:])):
134
+ continue
135
+ parts: list[str] = []
136
+ cursor = 0
137
+ for start, end, replacement in edits:
138
+ parts.extend((canonical[cursor : start - 1], replacement))
139
+ cursor = end
140
+ parts.append(canonical[cursor:])
141
+ reconstructed = "".join(parts)
142
+ if not reconstructed:
143
+ continue
144
+ for isoform_id in ids:
145
+ sequences[isoform_id] = (reconstructed, [(a, b, len(c)) for a, b, c in edits])
146
+ if displayed != accession:
147
+ sequences.pop(accession)
148
+ for isoform_id, (sequence, edits_map) in sequences.items():
149
+ # The accession and displayed isoform are aliases for the same product.
150
+ aliases = {isoform_id, accession} if isoform_id in {accession, displayed} else {isoform_id}
151
+ protein_ids: list[str] = []
152
+ for ref in entry.findall(f"{NS}dbReference[@type='Ensembl']"):
153
+ molecule = ref.find(NS + "molecule")
154
+ if (molecule.get("id") if molecule is not None else accession) not in aliases:
155
+ continue
156
+ protein_ids.extend(
157
+ node.get("value", "").split(".")[0]
158
+ for node in ref.findall(f"{NS}property[@type='protein sequence ID']")
159
+ )
160
+ curated: list[CuratedFeature] = []
161
+ for index, node in enumerate(features, 1):
162
+ feature_type = FEATURE_TYPES.get(node.get("type", ""))
163
+ interval = exact_interval(node)
164
+ if feature_type is None or interval is None:
165
+ continue
166
+ start, end = interval
167
+ location = node.find(NS + "location")
168
+ target = location.get("sequence") if location is not None else None
169
+ if target:
170
+ if target not in aliases:
171
+ continue
172
+ else:
173
+ # A feature touching edited residues is omitted, even when the
174
+ # replacement has equal length. Unchanged features can shift.
175
+ if any(start <= b and end >= a for a, b, _ in edits_map):
176
+ continue
177
+ shift = sum(size - (b - a + 1) for a, b, size in edits_map if b < start)
178
+ start, end = start + shift, end + shift
179
+ if not 1 <= start <= end <= len(sequence):
180
+ continue
181
+ description = (
182
+ node.get("description")
183
+ or node.findtext(f"{NS}ligand/{NS}name")
184
+ or node.get("type", "")
185
+ or "Unnamed UniProt feature"
186
+ )
187
+ curated.append(
188
+ {
189
+ "feature_id": node.get("id") or f"{accession}:{index}",
190
+ "feature_type": feature_type,
191
+ "start": start,
192
+ "end": end,
193
+ "description": description,
194
+ "uniprot_accession": accession,
195
+ "uniprot_isoform": displayed if isoform_id == accession else isoform_id,
196
+ "evidence": [
197
+ evidence[key]
198
+ for key in (node.get("evidence") or "").split()
199
+ if key in evidence
200
+ ],
201
+ }
202
+ )
203
+ yield {
204
+ "sequence": sequence,
205
+ "features": curated,
206
+ "accessions": list(
207
+ aliases | (set(accessions) if isoform_id in {accession, displayed} else set())
208
+ ),
209
+ "protein_ids": protein_ids,
210
+ }
211
+
212
+
213
+ def import_features(db: sqlite3.Connection, path: Path) -> dict[str, int]:
214
+ """Store sequence-verified feature lookups in the caller's atomic transaction.
215
+
216
+ Cross-references narrow candidates before reading peptides. Each distinct
217
+ peptide is decompressed at most once from a bounded chunk cache; unmatched
218
+ UniProt isoforms and changed Ensembl sequences are counted and omitted.
219
+ """
220
+ from .data import LOG, fetch_sequence, input_progress, sqlite_activity
221
+
222
+ candidates: dict[str, set[str]] = {}
223
+ for stable_id, sequence_id in db.execute(
224
+ "SELECT stable_id, sequence_id FROM sequences WHERE kind='pep'"
225
+ ):
226
+ candidates.setdefault(stable_id, set()).add(sequence_id)
227
+ xrefs: dict[str, set[str]] = {}
228
+ tables = {row[0] for row in db.execute("SELECT name FROM sqlite_master WHERE type='table'")}
229
+ if "ensembl_object_xref" in tables and {"xref_id", "dbprimary_acc"} <= {
230
+ row[1] for row in db.execute("PRAGMA table_info(ensembl_xref)")
231
+ }:
232
+ # Some tiny fixtures do not have a full object_xref table.
233
+ columns = {row[1] for row in db.execute("PRAGMA table_info(ensembl_object_xref)")}
234
+ if {"xref_id", "ensembl_id", "ensembl_object_type"} <= columns:
235
+ external_join = ""
236
+ external_filter = ""
237
+ if (
238
+ "ensembl_external_db" in tables
239
+ and "external_db_id"
240
+ in {row[1] for row in db.execute("PRAGMA table_info(ensembl_xref)")}
241
+ and {"external_db_id", "db_name"}
242
+ <= {row[1] for row in db.execute("PRAGMA table_info(ensembl_external_db)")}
243
+ ):
244
+ external_join = "JOIN ensembl_external_db d ON d.external_db_id=x.external_db_id"
245
+ external_filter = "AND LOWER(d.db_name) LIKE '%uniprot%'"
246
+ with sqlite_activity("Read UniProt protein cross-references"):
247
+ for accession, stable_id in db.execute(f"""
248
+ SELECT x.dbprimary_acc, t.stable_id FROM ensembl_xref x
249
+ JOIN ensembl_object_xref o USING (xref_id)
250
+ JOIN ensembl_translation t ON t.translation_id=o.ensembl_id
251
+ {external_join}
252
+ WHERE o.ensembl_object_type='Translation' {external_filter}
253
+ """):
254
+ if stable_id in candidates:
255
+ xrefs.setdefault(accession, set()).add(stable_id)
256
+ db.execute("DROP TABLE IF EXISTS ff_uniprot_features")
257
+ db.execute(
258
+ "CREATE TABLE ff_uniprot_features (sequence_id TEXT NOT NULL, feature_id TEXT NOT NULL, "
259
+ "payload TEXT NOT NULL, PRIMARY KEY(sequence_id, feature_id)) WITHOUT ROWID"
260
+ )
261
+ counts = {
262
+ "entries": 0,
263
+ "reviewed_human_entries": 0,
264
+ "matched_proteins": 0,
265
+ "features": 0,
266
+ "sequence_mismatches": 0,
267
+ }
268
+ matched: set[str] = set()
269
+ cache: OrderedDict[tuple[str, str, int], bytes] = OrderedDict()
270
+ with path.open("rb") as source:
271
+ compressed = source.read(2) == b"\x1f\x8b"
272
+ with input_progress(path, "Import reviewed human UniProt", compressed=compressed) as (
273
+ stream,
274
+ bar,
275
+ update,
276
+ ):
277
+ parser = ET.iterparse(stream, events=("start", "end"))
278
+ _, root = next(parser)
279
+ if root.tag != NS + "uniprot":
280
+ raise ValueError("Expected UniProt XML")
281
+ for event, entry in parser:
282
+ if event != "end" or entry.tag != NS + "entry":
283
+ continue
284
+ proteins = entry_proteins(entry)
285
+ reviewed = False
286
+ for protein in proteins:
287
+ reviewed = True
288
+ stable_ids = set(protein["protein_ids"])
289
+ for accession in protein["accessions"]:
290
+ stable_ids.update(xrefs.get(accession, ()))
291
+ for stable_id in stable_ids:
292
+ for sequence_id in candidates.get(stable_id, ()):
293
+ peptide = fetch_sequence(db, "pep", sequence_id, _chunk_cache=cache)
294
+ if peptide != protein["sequence"]:
295
+ counts["sequence_mismatches"] += 1
296
+ continue
297
+ matched.add(sequence_id)
298
+ for feature in protein["features"]:
299
+ payload = json.dumps(feature, separators=(",", ":"))
300
+ db.execute(
301
+ "INSERT OR IGNORE INTO ff_uniprot_features VALUES (?, ?, ?)",
302
+ (sequence_id, feature["feature_id"], payload),
303
+ )
304
+ counts["entries"] += 1
305
+ counts["reviewed_human_entries"] += int(reviewed)
306
+ update()
307
+ if counts["entries"] % 100 == 0:
308
+ bar.set_postfix(entries=counts["entries"], matched=len(matched), refresh=False)
309
+ root.clear() # Avoid retaining the full human XML tree in memory.
310
+ if counts["reviewed_human_entries"] == 0:
311
+ raise ValueError("UniProt XML contains no reviewed human entries")
312
+ counts["matched_proteins"] = len(matched)
313
+ counts["features"] = db.execute("SELECT COUNT(*) FROM ff_uniprot_features").fetchone()[0]
314
+ LOG.info("UniProt import: %s", counts)
315
+ return counts
@@ -0,0 +1,98 @@
1
+ Metadata-Version: 2.4
2
+ Name: fusion-function
3
+ Version: 0.2.1
4
+ Summary: Fusion frame and protein-domain prediction using a local Ensembl reference
5
+ License-File: LICENSE
6
+ Author: Caralyn Reisle
7
+ Author-email: creisle@bcgsc.ca
8
+ Requires-Python: >=3.11,<4.0
9
+ Classifier: Programming Language :: Python :: 3
10
+ Classifier: Programming Language :: Python :: 3.11
11
+ Classifier: Programming Language :: Python :: 3.12
12
+ Classifier: Programming Language :: Python :: 3.13
13
+ Classifier: Programming Language :: Python :: 3.14
14
+ Classifier: Programming Language :: Python :: 3.15
15
+ Requires-Dist: tqdm (>=4.67,<5)
16
+ Description-Content-Type: text/markdown
17
+
18
+ # Fusion-function
19
+
20
+ ![Coverage](https://raw.githubusercontent.com/creisle/fusion_function/badges/badges/main/coverage.svg)
21
+
22
+ Predict fusion reading frames and retained, disrupted, or excluded functional protein features from human GRCh38 transcript breakpoints. Predictions include effects of splicing, translation initiation, and premature termination. Annotation uses a preprocessed local SQLite reference.
23
+
24
+ ## Quick start
25
+
26
+ ```bash
27
+ pip install fusion-function
28
+ fusion-function prepare-data
29
+ ```
30
+
31
+ Preparation downloads a compatible prebuilt reference for the latest Ensembl release when available, otherwise builds it from source. The reference is stored in the default user cache; source builds can take an hour or longer.
32
+
33
+ Use the annotate_fusion_domains function to get information about the status of various domains in the expected fusion product
34
+
35
+ ```python
36
+ from fusion_function import ReferenceDatabase, annotate_fusion_domains
37
+
38
+ with ReferenceDatabase() as ref:
39
+ # ex. BCR::ABL1
40
+ result = annotate_fusion_domains(
41
+ transcript1_id="ENST00000305877", # BCR
42
+ transcript2_id="ENST00000318560", # ABL1
43
+ breakpoint1="22:23290413",
44
+ breakpoint2="9:130854064",
45
+ gene1_terminus="N",
46
+ gene2_terminus="C",
47
+ reference=ref,
48
+ )
49
+ ```
50
+
51
+ This will return an object with the following shape. See the [api](./docs/api.md) for details.
52
+
53
+ ```json
54
+ {
55
+ "frame_status": "in_frame",
56
+ "domains": [
57
+ {
58
+ "transcript_id": "ENST00000305877",
59
+ "interpro_id": "IPR036481",
60
+ "name": "Bcr-Abl oncoprotein oligomerisation domain superfamily",
61
+ "domain_type": "homologous_superfamily",
62
+ "start": 1,
63
+ "end": 67,
64
+ "sources": ["SuperFamily"],
65
+ "feature_ids": ["SSF69036"],
66
+ "breakpoint_based_status": "included",
67
+ "breakpoint_retained_percent": 100.0,
68
+ "post_splicing_status": "preserved",
69
+ "post_translation_status": "preserved"
70
+ },
71
+ ...
72
+ ],
73
+ "translation_start": {"ENST00000305877": "native_start_retained"}
74
+ }
75
+ ```
76
+
77
+ Analysis uses the newest prepared local release by default.
78
+
79
+ ## Documentation
80
+
81
+ - [API](docs/api.md)
82
+ - [Reference configuration and preparation](docs/reference.md)
83
+ - [Methodology](docs/methodology.md)
84
+ - [Testing](docs/testing.md)
85
+
86
+ ## Important Limitations
87
+
88
+ - This package uses the splicing model defined by [MAVIS](https://github.com/bcgsc/mavis), this is non-exhaustive. It assumes splice sites to be disrupted based on a breakpoint being within 2bp but there are many other ways to disrupt splicing that are difficult to predict computationally (ex. deep intronic). This package only covers the standard scenarios
89
+ - Currently we only support Hg38
90
+ - Only exact breakpoints are supported
91
+ - Predicted structural consequences do not establish fusion expression, oncogenicity, pathogenicity, or clinical actionability.
92
+
93
+ ## Citation
94
+
95
+ This package ports and extends selected fusion-annotation logic from [MAVIS](https://github.com/bcgsc/mavis). Please cite:
96
+
97
+ Reisle C, Mungall KL, Choo C, et al. MAVIS: merging, annotation, validation, and illustration of structural variants. *Bioinformatics*. 2019;35(3):515–517. PMID:30016509.
98
+
@@ -0,0 +1,16 @@
1
+ fusion_function/__init__.py,sha256=kGDFqDAJP8D_9M43wiPD_rtymDlENkvrH6VR0Uxg6vI,258
2
+ fusion_function/__main__.py,sha256=k1ocEWawweo1qCJWNFAAvyxz3tcY13dzvCenHszij30,48
3
+ fusion_function/cli.py,sha256=DqDcyYK0R4d88hDe8D-uLE2cHMks95rxl7exScrqbHo,767
4
+ fusion_function/data.py,sha256=Tu230UT1SAhjhRIMNTDGme_X5xCuG8vdVevfrS-m4mk,121039
5
+ fusion_function/ensembl.py,sha256=Rip1k7rQlDrsNVPiF4xQxIMjjInoIDeMqIg5GmM21Gc,3889
6
+ fusion_function/fusion.py,sha256=9_KPXMkvpkdrA9coMVQqILYnrnWGkm8xNi2_AwBDxUM,40495
7
+ fusion_function/interpro.py,sha256=TVQxzR3xO5y1dQb5-_kHmkRNTvFNjGwVZpxw5Bt1Hjc,731
8
+ fusion_function/prebuilt.py,sha256=Lyrc0WMGwormvrEkr8eac3Fg2xRF2CJ366vJl3WDon4,23713
9
+ fusion_function/reference.py,sha256=OOgLN_4MuEzczYamfJ5rbQv3bSBWFLocjE2V9jV5Jho,4407
10
+ fusion_function/reference_catalog.json,sha256=8REBPh6B46K-Z9eWp-Ef-gLKsyTlEZ4ifFoXcwMwAaU,9260
11
+ fusion_function/uniprot.py,sha256=izymwyMI6fog3mUc2vwWuhgs80fyH8BfEeg4hCy6On0,13862
12
+ fusion_function-0.2.1.dist-info/METADATA,sha256=zTidauqLnSTCufH65Tcr6eR8w2TamVHdvb0_-y18Ln0,3853
13
+ fusion_function-0.2.1.dist-info/WHEEL,sha256=L-WvLdSBvJWHXv5zrQxDvNRnIgoBzLUpcbuiw19KCuQ,88
14
+ fusion_function-0.2.1.dist-info/entry_points.txt,sha256=-cTv1JByye6d8iVffWHKz6yNApXduTlch2JooZK8KFY,60
15
+ fusion_function-0.2.1.dist-info/licenses/LICENSE,sha256=OXLcl0T2SZ8Pmy2_dmlvKuetivmyPd5m1q-Gyd-zaYY,35149
16
+ fusion_function-0.2.1.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: poetry-core 2.5.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,3 @@
1
+ [console_scripts]
2
+ fusion-function=fusion_function.cli:main
3
+