fusion-function 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fusion_function/__init__.py +9 -0
- fusion_function/__main__.py +3 -0
- fusion_function/cli.py +27 -0
- fusion_function/data.py +2683 -0
- fusion_function/ensembl.py +149 -0
- fusion_function/fusion.py +1024 -0
- fusion_function/interpro.py +29 -0
- fusion_function/prebuilt.py +518 -0
- fusion_function/reference.py +112 -0
- fusion_function/reference_catalog.json +82 -0
- fusion_function/uniprot.py +315 -0
- fusion_function-0.2.1.dist-info/METADATA +98 -0
- fusion_function-0.2.1.dist-info/RECORD +16 -0
- fusion_function-0.2.1.dist-info/WHEEL +4 -0
- fusion_function-0.2.1.dist-info/entry_points.txt +3 -0
- fusion_function-0.2.1.dist-info/licenses/LICENSE +674 -0
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
{
|
|
2
|
+
"catalog_version": 1,
|
|
3
|
+
"references": [
|
|
4
|
+
{
|
|
5
|
+
"species": "homo_sapiens",
|
|
6
|
+
"assembly": "GRCh38",
|
|
7
|
+
"release": 116,
|
|
8
|
+
"build_revision": 1,
|
|
9
|
+
"filename": "homo_sapiens.GRCh38.ensembl-116.ff-v1.build-1.sqlite.gz",
|
|
10
|
+
"compression": "gzip",
|
|
11
|
+
"archive_bytes": 2245743354,
|
|
12
|
+
"archive_sha256": "65222b7637ecfcb1ecff9d6862eeb20da9ac2d471a49d8c05568bc3f978c5866",
|
|
13
|
+
"database_bytes": 2727596032,
|
|
14
|
+
"database_sha256": "d3ffd6b684598d392f3d4525bd7f368ed3563dff8534a2c80d8cb361350aa1d4",
|
|
15
|
+
"doi": "10.5281/zenodo.23111108",
|
|
16
|
+
"url": "https://zenodo.org/records/23111108/files/homo_sapiens.GRCh38.ensembl-116.ff-v1.build-1.sqlite.gz?download=1",
|
|
17
|
+
"metadata": {
|
|
18
|
+
"format_version": "1",
|
|
19
|
+
"preprocessing_version": "1",
|
|
20
|
+
"cds_mapping_version": "2",
|
|
21
|
+
"feature_annotation_version": "3",
|
|
22
|
+
"transcript_payload_codec": "zlib-json-v1",
|
|
23
|
+
"sequence_chunk_size": "1048576",
|
|
24
|
+
"sequence_codec": "zlib"
|
|
25
|
+
},
|
|
26
|
+
"data_source_notices": {
|
|
27
|
+
"Ensembl": {
|
|
28
|
+
"terms": "Unrestricted project-generated data; third-party constraints may apply.",
|
|
29
|
+
"url": "https://www.ensembl.org/info/about/legal/disclaimer.html"
|
|
30
|
+
},
|
|
31
|
+
"UniProt Consortium": {
|
|
32
|
+
"terms": "CC BY 4.0. Credit UniProt, link to the license and identify modifications.",
|
|
33
|
+
"url": "https://www.uniprot.org/help/license",
|
|
34
|
+
"license_url": "https://creativecommons.org/licenses/by/4.0/"
|
|
35
|
+
},
|
|
36
|
+
"InterPro Consortium": {
|
|
37
|
+
"terms": "Current InterPro downloads: CC0 1.0. Retain historical-source notices.",
|
|
38
|
+
"url": "https://interpro-documentation.readthedocs.io/en/latest/license.html",
|
|
39
|
+
"license_url": "https://creativecommons.org/publicdomain/zero/1.0/"
|
|
40
|
+
},
|
|
41
|
+
"PANTHER": {
|
|
42
|
+
"terms": "Classification release 14.1 and 17.0 READMEs carry GPL-2.0-or-later notices. Confirm terms for the imported release and derived classifications before redistribution.",
|
|
43
|
+
"url": "https://data.pantherdb.org/ftp/sequence_classifications/"
|
|
44
|
+
},
|
|
45
|
+
"Ensembl member annotations": {
|
|
46
|
+
"terms": "Member-source terms are not replaced by Ensembl or InterPro terms. PROSITE database terms are CC BY-NC-ND 4.0 with commercial licensing; SMART models require a license. Confirm terms for derived match annotations.",
|
|
47
|
+
"url": "https://prosite.expasy.org/prosite_license.html",
|
|
48
|
+
"smart_url": "https://smart.embl.de/about.cgi"
|
|
49
|
+
}
|
|
50
|
+
},
|
|
51
|
+
"source_metadata": {
|
|
52
|
+
"format_version": "1",
|
|
53
|
+
"release": "116",
|
|
54
|
+
"species": "homo_sapiens",
|
|
55
|
+
"core_database": "homo_sapiens_core_116_38",
|
|
56
|
+
"assembly": "GRCh38",
|
|
57
|
+
"created_utc": "2026-10-01T01:17:06.715364+00:00",
|
|
58
|
+
"sequence_chunk_size": "1048576",
|
|
59
|
+
"sequence_codec": "zlib",
|
|
60
|
+
"tables": "[\"meta\", \"coord_system\", \"seq_region\", \"gene\", \"transcript\", \"exon\", \"exon_transcript\", \"translation\", \"analysis\", \"protein_feature\", \"interpro\", \"xref\", \"external_db\", \"object_xref\"]",
|
|
61
|
+
"sequence_types": "[\"dna\", \"pep\"]",
|
|
62
|
+
"counts": "{\"meta\": 235, \"coord_system\": 9, \"seq_region\": 268640, \"gene\": 87735, \"transcript\": 672310, \"exon\": 1422168, \"exon_transcript\": 5252227, \"translation\": 384051, \"analysis\": 100, \"protein_feature\": 6690040, \"interpro\": 25300, \"xref\": 2751838, \"external_db\": 501, \"object_xref\": 7737725, \"dna_sequences\": 706, \"pep_sequences\": 382428}",
|
|
63
|
+
"preprocessing_version": "1",
|
|
64
|
+
"preprocessing_counts": "{\"ready\": 368066, \"noncoding\": 288242, \"error\": 16002, \"protein_features\": 6296276}",
|
|
65
|
+
"preprocessed_utc": "2026-10-02T17:30:44.172625+00:00",
|
|
66
|
+
"interpro_entries_sha256": "1d247a42670d4a2913cec9f2a58813135c736d65128c1078d9c5eac797ffef1c",
|
|
67
|
+
"interpro_entry_types_loaded": "true",
|
|
68
|
+
"interpro_historical_entries": "{\"IPR001423\": \"109.0\", \"IPR004734\": \"106.0\", \"IPR009133\": \"109.0\", \"IPR010307\": \"108.0\", \"IPR017198\": \"106.0\", \"IPR018798\": \"107.0\", \"IPR019321\": \"108.0\", \"IPR024812\": \"109.0\", \"IPR026516\": \"109.0\", \"IPR026570\": \"109.0\", \"IPR029065\": \"107.0\", \"IPR031101\": \"109.0\", \"IPR032922\": \"107.0\", \"IPR033489\": \"109.0\", \"IPR033509\": \"109.0\", \"IPR039163\": \"109.0\", \"IPR039681\": \"109.0\", \"IPR039715\": \"109.0\", \"IPR039747\": \"109.0\", \"IPR042012\": \"106.0\", \"IPR043040\": \"107.0\", \"IPR043041\": \"107.0\", \"IPR043042\": \"107.0\", \"IPR043504\": \"107.0\", \"IPR044613\": \"109.0\", \"IPR044986\": \"109.0\", \"IPR045150\": \"107.0\", \"IPR048456\": \"106.0\", \"IPR050058\": \"108.0\", \"IPR050079\": \"108.0\", \"IPR050082\": \"108.0\", \"IPR050089\": \"107.0\", \"IPR050115\": \"108.0\", \"IPR050123\": \"107.0\", \"IPR050149\": \"108.0\", \"IPR050174\": \"108.0\", \"IPR050203\": \"108.0\", \"IPR050223\": \"108.0\", \"IPR050224\": \"108.0\", \"IPR050247\": \"108.0\", \"IPR050332\": \"109.0\", \"IPR050369\": \"108.0\", \"IPR050374\": \"108.0\", \"IPR050382\": \"109.0\", \"IPR050383\": \"109.0\", \"IPR050514\": \"109.0\", \"IPR050518\": \"108.0\", \"IPR050527\": \"107.0\", \"IPR050544\": \"109.0\", \"IPR050569\": \"109.0\", \"IPR050628\": \"107.0\", \"IPR050650\": \"107.0\", \"IPR050657\": \"107.0\", \"IPR050741\": \"107.0\", \"IPR050778\": \"107.0\", \"IPR050802\": \"107.0\", \"IPR050803\": \"106.0\", \"IPR050888\": \"108.0\", \"IPR050904\": \"108.0\", \"IPR050912\": \"107.0\", \"IPR050915\": \"108.0\", \"IPR050937\": \"107.0\", \"IPR050951\": \"107.0\", \"IPR050958\": \"109.0\", \"IPR050969\": \"109.0\", \"IPR051012\": \"107.0\", \"IPR051022\": \"107.0\", \"IPR051027\": \"107.0\", \"IPR051029\": \"107.0\", \"IPR051061\": \"107.0\", \"IPR051064\": \"107.0\", \"IPR051110\": \"109.0\", \"IPR051170\": \"109.0\", \"IPR051179\": \"108.0\", \"IPR051205\": \"108.0\", \"IPR051232\": \"108.0\", \"IPR051237\": \"107.0\", \"IPR051266\": \"109.0\", \"IPR051270\": \"107.0\", \"IPR051288\": \"108.0\", \"IPR051334\": \"107.0\", \"IPR051387\": \"108.0\", \"IPR051412\": \"109.0\", \"IPR051425\": \"109.0\", \"IPR051441\": \"109.0\", \"IPR051458\": \"109.0\", \"IPR051478\": \"107.0\", \"IPR051550\": \"109.0\", \"IPR051569\": \"107.0\", \"IPR051630\": \"108.0\", \"IPR051637\": \"109.0\", \"IPR051647\": \"109.0\", \"IPR051694\": \"109.0\", \"IPR051728\": \"109.0\", \"IPR051742\": \"108.0\", \"IPR051827\": \"109.0\", \"IPR051969\": \"107.0\", \"IPR052083\": \"109.0\", \"IPR052097\": \"108.0\", \"IPR052130\": \"109.0\", \"IPR052145\": \"109.0\", \"IPR052273\": \"109.0\", \"IPR052301\": \"109.0\", \"IPR052346\": \"106.0\", \"IPR052376\": \"109.0\", \"IPR052392\": \"109.0\", \"IPR052504\": \"109.0\", \"IPR052555\": \"109.0\", \"IPR052579\": \"109.0\", \"IPR052660\": \"109.0\", \"IPR052696\": \"109.0\", \"IPR052881\": \"109.0\", \"IPR053019\": \"109.0\", \"IPR053099\": \"109.0\", \"IPR053374\": \"107.0\", \"IPR054602\": \"107.0\", \"IPR055047\": \"107.0\", \"IPR055074\": \"107.0\", \"IPR055351\": \"107.0\", \"IPR056552\": \"107.0\"}",
|
|
69
|
+
"interpro_historical_sources": "[{\"release\": \"109.0\", \"sha256\": \"00b86a43b14652470d1110133b26f91314d5bddf9f164da76c494d7b322752ec\", \"url\": \"https://ftp.ebi.ac.uk/pub/databases/interpro/releases/109.0/entry.list\"}, {\"release\": \"108.0\", \"sha256\": \"11ddc8e6624a3ddaf40d5528e5aaf0a87308bd1c58a3f7087208091ab5eefc29\", \"url\": \"https://ftp.ebi.ac.uk/pub/databases/interpro/releases/108.0/entry.list\"}, {\"release\": \"107.0\", \"sha256\": \"333770e997fc6b88a836650ac5c9475b8934b4beb52a5a084d7d2e10dcc4af05\", \"url\": \"https://ftp.ebi.ac.uk/pub/databases/interpro/releases/107.0/entry.list\"}, {\"release\": \"106.0\", \"sha256\": \"15d98e0805a05cdbe4935ab2086e65c6e204181ae87c0b06cc33a315eea4db4a\", \"url\": \"https://ftp.ebi.ac.uk/pub/databases/interpro/releases/106.0/entry.list\"}]",
|
|
70
|
+
"interpro_preserved_metadata_sha256": "1d247a42670d4a2913cec9f2a58813135c736d65128c1078d9c5eac797ffef1c",
|
|
71
|
+
"panther_classifications_source": "{\"sha256\": \"0635357eeae76cb6dce8ce498f20e7ad5e101fe5a81717003c97acb2ed009b32\", \"url\": \"https://data.pantherdb.org/ftp/sequence_classifications/19.0/PANTHER_Sequence_Classification_files/PTHR19.0_human\", \"version\": \"19.0\"}",
|
|
72
|
+
"feature_annotation_version": "3",
|
|
73
|
+
"uniprot_features_source": "{\"sha256\": \"3296aef04049bbf83a21f3335436c9f2d3a00f6186fc10e4e66854ac37b2e5ab\", \"url\": \"https://rest.uniprot.org/uniprotkb/stream?format=xml&compressed=true&query=organism_id%3A9606%20AND%20reviewed%3Atrue\"}",
|
|
74
|
+
"uniprot_counts": "{\"entries\": 20431, \"features\": 332987, \"matched_proteins\": 97684, \"reviewed_human_entries\": 20431, \"sequence_mismatches\": 19693}",
|
|
75
|
+
"cds_mapping_version": "2",
|
|
76
|
+
"preprocessing_errors": "{\"CDS/peptide length mismatch\": {\"count\": 14362, \"example_transcripts\": [\"ENST00000361390\", \"ENST00000361453\", \"ENST00000362079\"]}, \"Unsupported transcript coordinate system\": {\"count\": 1640, \"example_transcripts\": [\"LRG_1t1\", \"LRG_1000t1\", \"LRG_1001t1\"]}}",
|
|
77
|
+
"uniprot_lookup_fingerprint": "7f8c19064714799b35cca12a4320c26e40f21faef2ac0e9a3220ad005779ca33",
|
|
78
|
+
"transcript_payload_codec": "zlib-json-v1"
|
|
79
|
+
}
|
|
80
|
+
}
|
|
81
|
+
]
|
|
82
|
+
}
|
|
@@ -0,0 +1,315 @@
|
|
|
1
|
+
"""Reviewed human UniProt features, imported once during reference preparation.
|
|
2
|
+
|
|
3
|
+
XML is streamed with the standard library. Isoforms are reconstructed only from
|
|
4
|
+
explicit UniProt splice variants; canonical features are transferred only across
|
|
5
|
+
unchanged sequence. Ensembl mappings additionally require an exact whole-protein
|
|
6
|
+
sequence match. No gene-name matching, approximate alignment or runtime network
|
|
7
|
+
access is used.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import json
|
|
13
|
+
import sqlite3
|
|
14
|
+
from collections import OrderedDict
|
|
15
|
+
from collections.abc import Iterator
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from typing import TypedDict
|
|
18
|
+
from xml.etree import ElementTree as ET
|
|
19
|
+
|
|
20
|
+
from .ensembl import FeatureEvidence
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
UNIPROT_HUMAN_URL = (
|
|
24
|
+
"https://rest.uniprot.org/uniprotkb/stream?format=xml&compressed=true"
|
|
25
|
+
"&query=organism_id%3A9606%20AND%20reviewed%3Atrue"
|
|
26
|
+
)
|
|
27
|
+
NS = "{http://uniprot.org/uniprot}"
|
|
28
|
+
FEATURE_TYPES = {
|
|
29
|
+
"domain": "domain",
|
|
30
|
+
"active site": "active_site",
|
|
31
|
+
"binding site": "binding_site",
|
|
32
|
+
"site": "conserved_site",
|
|
33
|
+
"motif": "motif",
|
|
34
|
+
"short sequence motif": "motif",
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class CuratedFeature(TypedDict):
|
|
39
|
+
feature_id: str
|
|
40
|
+
feature_type: str
|
|
41
|
+
start: int
|
|
42
|
+
end: int
|
|
43
|
+
description: str
|
|
44
|
+
uniprot_accession: str
|
|
45
|
+
uniprot_isoform: str
|
|
46
|
+
evidence: list[FeatureEvidence]
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class CuratedProtein(TypedDict):
|
|
50
|
+
sequence: str
|
|
51
|
+
features: list[CuratedFeature]
|
|
52
|
+
accessions: list[str]
|
|
53
|
+
protein_ids: list[str]
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def exact_interval(element: ET.Element) -> tuple[int, int] | None:
|
|
57
|
+
"""Reject uncertain/fuzzy coordinates rather than inventing exact bounds."""
|
|
58
|
+
location = element.find(NS + "location")
|
|
59
|
+
if location is None:
|
|
60
|
+
return None
|
|
61
|
+
position = location.find(NS + "position")
|
|
62
|
+
nodes = (
|
|
63
|
+
[position, position]
|
|
64
|
+
if position is not None
|
|
65
|
+
else [location.find(NS + "begin"), location.find(NS + "end")]
|
|
66
|
+
)
|
|
67
|
+
if any(
|
|
68
|
+
node is None
|
|
69
|
+
or node.get("status") not in {None, "certain"}
|
|
70
|
+
or not (node.get("position") or "").isdigit()
|
|
71
|
+
for node in nodes
|
|
72
|
+
):
|
|
73
|
+
return None
|
|
74
|
+
start, end = [int(node.attrib["position"]) for node in nodes if node is not None]
|
|
75
|
+
return (start, end) if 1 <= start <= end else None
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def entry_proteins(entry: ET.Element) -> Iterator[CuratedProtein]:
|
|
79
|
+
"""Yield canonical/described isoforms with safely mapped functional features."""
|
|
80
|
+
if entry.get("dataset") != "Swiss-Prot" or not any(
|
|
81
|
+
ref.get("type") == "NCBI Taxonomy" and ref.get("id") == "9606"
|
|
82
|
+
for ref in entry.findall(f"{NS}organism/{NS}dbReference")
|
|
83
|
+
):
|
|
84
|
+
return
|
|
85
|
+
accessions = [node.text for node in entry.findall(NS + "accession") if node.text]
|
|
86
|
+
canonical = "".join((entry.findtext(NS + "sequence") or "").split())
|
|
87
|
+
if not accessions or not canonical:
|
|
88
|
+
raise ValueError("Reviewed human UniProt entry lacks accession or sequence")
|
|
89
|
+
accession = accessions[0]
|
|
90
|
+
evidence: dict[str, FeatureEvidence] = {}
|
|
91
|
+
for node in entry.findall(NS + "evidence"):
|
|
92
|
+
item: FeatureEvidence = {"code": node.get("type", "")}
|
|
93
|
+
reference = node.find(f"{NS}source/{NS}dbReference")
|
|
94
|
+
if reference is not None:
|
|
95
|
+
item["source"] = reference.get("type", "")
|
|
96
|
+
item["id"] = reference.get("id", "")
|
|
97
|
+
evidence[node.get("key", "")] = item
|
|
98
|
+
features = entry.findall(NS + "feature")
|
|
99
|
+
variants = {node.get("id"): node for node in features if node.get("type") == "splice variant"}
|
|
100
|
+
isoforms = entry.findall(f"{NS}comment[@type='alternative products']/{NS}isoform")
|
|
101
|
+
# A canonical entry need not declare alternative products.
|
|
102
|
+
sequences: dict[str, tuple[str, list[tuple[int, int, int]]]] = {accession: (canonical, [])}
|
|
103
|
+
displayed = accession
|
|
104
|
+
for isoform in isoforms:
|
|
105
|
+
sequence_element = isoform.find(NS + "sequence")
|
|
106
|
+
ids = [node.text for node in isoform.findall(NS + "id") if node.text]
|
|
107
|
+
if sequence_element is None or not ids:
|
|
108
|
+
continue
|
|
109
|
+
edits: list[tuple[int, int, str]] = []
|
|
110
|
+
if sequence_element.get("type") == "displayed":
|
|
111
|
+
displayed = ids[0]
|
|
112
|
+
elif sequence_element.get("type") == "described":
|
|
113
|
+
refs = (sequence_element.get("ref") or "").split()
|
|
114
|
+
if not refs:
|
|
115
|
+
continue
|
|
116
|
+
for variant_id in refs:
|
|
117
|
+
variant = variants.get(variant_id)
|
|
118
|
+
interval = exact_interval(variant) if variant is not None else None
|
|
119
|
+
if variant is None or interval is None or interval[1] > len(canonical):
|
|
120
|
+
break
|
|
121
|
+
start, end = interval
|
|
122
|
+
original = variant.findtext(NS + "original")
|
|
123
|
+
if original and original != canonical[start - 1 : end]:
|
|
124
|
+
break
|
|
125
|
+
replacement = variant.findtext(NS + "variation") or ""
|
|
126
|
+
edits.append((start, end, replacement))
|
|
127
|
+
if len(edits) != len(refs):
|
|
128
|
+
continue
|
|
129
|
+
else:
|
|
130
|
+
# External/not-described isoform sequences cannot be reconstructed.
|
|
131
|
+
continue
|
|
132
|
+
edits.sort()
|
|
133
|
+
if any(left[1] >= right[0] for left, right in zip(edits, edits[1:])):
|
|
134
|
+
continue
|
|
135
|
+
parts: list[str] = []
|
|
136
|
+
cursor = 0
|
|
137
|
+
for start, end, replacement in edits:
|
|
138
|
+
parts.extend((canonical[cursor : start - 1], replacement))
|
|
139
|
+
cursor = end
|
|
140
|
+
parts.append(canonical[cursor:])
|
|
141
|
+
reconstructed = "".join(parts)
|
|
142
|
+
if not reconstructed:
|
|
143
|
+
continue
|
|
144
|
+
for isoform_id in ids:
|
|
145
|
+
sequences[isoform_id] = (reconstructed, [(a, b, len(c)) for a, b, c in edits])
|
|
146
|
+
if displayed != accession:
|
|
147
|
+
sequences.pop(accession)
|
|
148
|
+
for isoform_id, (sequence, edits_map) in sequences.items():
|
|
149
|
+
# The accession and displayed isoform are aliases for the same product.
|
|
150
|
+
aliases = {isoform_id, accession} if isoform_id in {accession, displayed} else {isoform_id}
|
|
151
|
+
protein_ids: list[str] = []
|
|
152
|
+
for ref in entry.findall(f"{NS}dbReference[@type='Ensembl']"):
|
|
153
|
+
molecule = ref.find(NS + "molecule")
|
|
154
|
+
if (molecule.get("id") if molecule is not None else accession) not in aliases:
|
|
155
|
+
continue
|
|
156
|
+
protein_ids.extend(
|
|
157
|
+
node.get("value", "").split(".")[0]
|
|
158
|
+
for node in ref.findall(f"{NS}property[@type='protein sequence ID']")
|
|
159
|
+
)
|
|
160
|
+
curated: list[CuratedFeature] = []
|
|
161
|
+
for index, node in enumerate(features, 1):
|
|
162
|
+
feature_type = FEATURE_TYPES.get(node.get("type", ""))
|
|
163
|
+
interval = exact_interval(node)
|
|
164
|
+
if feature_type is None or interval is None:
|
|
165
|
+
continue
|
|
166
|
+
start, end = interval
|
|
167
|
+
location = node.find(NS + "location")
|
|
168
|
+
target = location.get("sequence") if location is not None else None
|
|
169
|
+
if target:
|
|
170
|
+
if target not in aliases:
|
|
171
|
+
continue
|
|
172
|
+
else:
|
|
173
|
+
# A feature touching edited residues is omitted, even when the
|
|
174
|
+
# replacement has equal length. Unchanged features can shift.
|
|
175
|
+
if any(start <= b and end >= a for a, b, _ in edits_map):
|
|
176
|
+
continue
|
|
177
|
+
shift = sum(size - (b - a + 1) for a, b, size in edits_map if b < start)
|
|
178
|
+
start, end = start + shift, end + shift
|
|
179
|
+
if not 1 <= start <= end <= len(sequence):
|
|
180
|
+
continue
|
|
181
|
+
description = (
|
|
182
|
+
node.get("description")
|
|
183
|
+
or node.findtext(f"{NS}ligand/{NS}name")
|
|
184
|
+
or node.get("type", "")
|
|
185
|
+
or "Unnamed UniProt feature"
|
|
186
|
+
)
|
|
187
|
+
curated.append(
|
|
188
|
+
{
|
|
189
|
+
"feature_id": node.get("id") or f"{accession}:{index}",
|
|
190
|
+
"feature_type": feature_type,
|
|
191
|
+
"start": start,
|
|
192
|
+
"end": end,
|
|
193
|
+
"description": description,
|
|
194
|
+
"uniprot_accession": accession,
|
|
195
|
+
"uniprot_isoform": displayed if isoform_id == accession else isoform_id,
|
|
196
|
+
"evidence": [
|
|
197
|
+
evidence[key]
|
|
198
|
+
for key in (node.get("evidence") or "").split()
|
|
199
|
+
if key in evidence
|
|
200
|
+
],
|
|
201
|
+
}
|
|
202
|
+
)
|
|
203
|
+
yield {
|
|
204
|
+
"sequence": sequence,
|
|
205
|
+
"features": curated,
|
|
206
|
+
"accessions": list(
|
|
207
|
+
aliases | (set(accessions) if isoform_id in {accession, displayed} else set())
|
|
208
|
+
),
|
|
209
|
+
"protein_ids": protein_ids,
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def import_features(db: sqlite3.Connection, path: Path) -> dict[str, int]:
|
|
214
|
+
"""Store sequence-verified feature lookups in the caller's atomic transaction.
|
|
215
|
+
|
|
216
|
+
Cross-references narrow candidates before reading peptides. Each distinct
|
|
217
|
+
peptide is decompressed at most once from a bounded chunk cache; unmatched
|
|
218
|
+
UniProt isoforms and changed Ensembl sequences are counted and omitted.
|
|
219
|
+
"""
|
|
220
|
+
from .data import LOG, fetch_sequence, input_progress, sqlite_activity
|
|
221
|
+
|
|
222
|
+
candidates: dict[str, set[str]] = {}
|
|
223
|
+
for stable_id, sequence_id in db.execute(
|
|
224
|
+
"SELECT stable_id, sequence_id FROM sequences WHERE kind='pep'"
|
|
225
|
+
):
|
|
226
|
+
candidates.setdefault(stable_id, set()).add(sequence_id)
|
|
227
|
+
xrefs: dict[str, set[str]] = {}
|
|
228
|
+
tables = {row[0] for row in db.execute("SELECT name FROM sqlite_master WHERE type='table'")}
|
|
229
|
+
if "ensembl_object_xref" in tables and {"xref_id", "dbprimary_acc"} <= {
|
|
230
|
+
row[1] for row in db.execute("PRAGMA table_info(ensembl_xref)")
|
|
231
|
+
}:
|
|
232
|
+
# Some tiny fixtures do not have a full object_xref table.
|
|
233
|
+
columns = {row[1] for row in db.execute("PRAGMA table_info(ensembl_object_xref)")}
|
|
234
|
+
if {"xref_id", "ensembl_id", "ensembl_object_type"} <= columns:
|
|
235
|
+
external_join = ""
|
|
236
|
+
external_filter = ""
|
|
237
|
+
if (
|
|
238
|
+
"ensembl_external_db" in tables
|
|
239
|
+
and "external_db_id"
|
|
240
|
+
in {row[1] for row in db.execute("PRAGMA table_info(ensembl_xref)")}
|
|
241
|
+
and {"external_db_id", "db_name"}
|
|
242
|
+
<= {row[1] for row in db.execute("PRAGMA table_info(ensembl_external_db)")}
|
|
243
|
+
):
|
|
244
|
+
external_join = "JOIN ensembl_external_db d ON d.external_db_id=x.external_db_id"
|
|
245
|
+
external_filter = "AND LOWER(d.db_name) LIKE '%uniprot%'"
|
|
246
|
+
with sqlite_activity("Read UniProt protein cross-references"):
|
|
247
|
+
for accession, stable_id in db.execute(f"""
|
|
248
|
+
SELECT x.dbprimary_acc, t.stable_id FROM ensembl_xref x
|
|
249
|
+
JOIN ensembl_object_xref o USING (xref_id)
|
|
250
|
+
JOIN ensembl_translation t ON t.translation_id=o.ensembl_id
|
|
251
|
+
{external_join}
|
|
252
|
+
WHERE o.ensembl_object_type='Translation' {external_filter}
|
|
253
|
+
"""):
|
|
254
|
+
if stable_id in candidates:
|
|
255
|
+
xrefs.setdefault(accession, set()).add(stable_id)
|
|
256
|
+
db.execute("DROP TABLE IF EXISTS ff_uniprot_features")
|
|
257
|
+
db.execute(
|
|
258
|
+
"CREATE TABLE ff_uniprot_features (sequence_id TEXT NOT NULL, feature_id TEXT NOT NULL, "
|
|
259
|
+
"payload TEXT NOT NULL, PRIMARY KEY(sequence_id, feature_id)) WITHOUT ROWID"
|
|
260
|
+
)
|
|
261
|
+
counts = {
|
|
262
|
+
"entries": 0,
|
|
263
|
+
"reviewed_human_entries": 0,
|
|
264
|
+
"matched_proteins": 0,
|
|
265
|
+
"features": 0,
|
|
266
|
+
"sequence_mismatches": 0,
|
|
267
|
+
}
|
|
268
|
+
matched: set[str] = set()
|
|
269
|
+
cache: OrderedDict[tuple[str, str, int], bytes] = OrderedDict()
|
|
270
|
+
with path.open("rb") as source:
|
|
271
|
+
compressed = source.read(2) == b"\x1f\x8b"
|
|
272
|
+
with input_progress(path, "Import reviewed human UniProt", compressed=compressed) as (
|
|
273
|
+
stream,
|
|
274
|
+
bar,
|
|
275
|
+
update,
|
|
276
|
+
):
|
|
277
|
+
parser = ET.iterparse(stream, events=("start", "end"))
|
|
278
|
+
_, root = next(parser)
|
|
279
|
+
if root.tag != NS + "uniprot":
|
|
280
|
+
raise ValueError("Expected UniProt XML")
|
|
281
|
+
for event, entry in parser:
|
|
282
|
+
if event != "end" or entry.tag != NS + "entry":
|
|
283
|
+
continue
|
|
284
|
+
proteins = entry_proteins(entry)
|
|
285
|
+
reviewed = False
|
|
286
|
+
for protein in proteins:
|
|
287
|
+
reviewed = True
|
|
288
|
+
stable_ids = set(protein["protein_ids"])
|
|
289
|
+
for accession in protein["accessions"]:
|
|
290
|
+
stable_ids.update(xrefs.get(accession, ()))
|
|
291
|
+
for stable_id in stable_ids:
|
|
292
|
+
for sequence_id in candidates.get(stable_id, ()):
|
|
293
|
+
peptide = fetch_sequence(db, "pep", sequence_id, _chunk_cache=cache)
|
|
294
|
+
if peptide != protein["sequence"]:
|
|
295
|
+
counts["sequence_mismatches"] += 1
|
|
296
|
+
continue
|
|
297
|
+
matched.add(sequence_id)
|
|
298
|
+
for feature in protein["features"]:
|
|
299
|
+
payload = json.dumps(feature, separators=(",", ":"))
|
|
300
|
+
db.execute(
|
|
301
|
+
"INSERT OR IGNORE INTO ff_uniprot_features VALUES (?, ?, ?)",
|
|
302
|
+
(sequence_id, feature["feature_id"], payload),
|
|
303
|
+
)
|
|
304
|
+
counts["entries"] += 1
|
|
305
|
+
counts["reviewed_human_entries"] += int(reviewed)
|
|
306
|
+
update()
|
|
307
|
+
if counts["entries"] % 100 == 0:
|
|
308
|
+
bar.set_postfix(entries=counts["entries"], matched=len(matched), refresh=False)
|
|
309
|
+
root.clear() # Avoid retaining the full human XML tree in memory.
|
|
310
|
+
if counts["reviewed_human_entries"] == 0:
|
|
311
|
+
raise ValueError("UniProt XML contains no reviewed human entries")
|
|
312
|
+
counts["matched_proteins"] = len(matched)
|
|
313
|
+
counts["features"] = db.execute("SELECT COUNT(*) FROM ff_uniprot_features").fetchone()[0]
|
|
314
|
+
LOG.info("UniProt import: %s", counts)
|
|
315
|
+
return counts
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: fusion-function
|
|
3
|
+
Version: 0.2.1
|
|
4
|
+
Summary: Fusion frame and protein-domain prediction using a local Ensembl reference
|
|
5
|
+
License-File: LICENSE
|
|
6
|
+
Author: Caralyn Reisle
|
|
7
|
+
Author-email: creisle@bcgsc.ca
|
|
8
|
+
Requires-Python: >=3.11,<4.0
|
|
9
|
+
Classifier: Programming Language :: Python :: 3
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.15
|
|
15
|
+
Requires-Dist: tqdm (>=4.67,<5)
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
|
|
18
|
+
# Fusion-function
|
|
19
|
+
|
|
20
|
+

|
|
21
|
+
|
|
22
|
+
Predict fusion reading frames and retained, disrupted, or excluded functional protein features from human GRCh38 transcript breakpoints. Predictions include effects of splicing, translation initiation, and premature termination. Annotation uses a preprocessed local SQLite reference.
|
|
23
|
+
|
|
24
|
+
## Quick start
|
|
25
|
+
|
|
26
|
+
```bash
|
|
27
|
+
pip install fusion-function
|
|
28
|
+
fusion-function prepare-data
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
Preparation downloads a compatible prebuilt reference for the latest Ensembl release when available, otherwise builds it from source. The reference is stored in the default user cache; source builds can take an hour or longer.
|
|
32
|
+
|
|
33
|
+
Use the annotate_fusion_domains function to get information about the status of various domains in the expected fusion product
|
|
34
|
+
|
|
35
|
+
```python
|
|
36
|
+
from fusion_function import ReferenceDatabase, annotate_fusion_domains
|
|
37
|
+
|
|
38
|
+
with ReferenceDatabase() as ref:
|
|
39
|
+
# ex. BCR::ABL1
|
|
40
|
+
result = annotate_fusion_domains(
|
|
41
|
+
transcript1_id="ENST00000305877", # BCR
|
|
42
|
+
transcript2_id="ENST00000318560", # ABL1
|
|
43
|
+
breakpoint1="22:23290413",
|
|
44
|
+
breakpoint2="9:130854064",
|
|
45
|
+
gene1_terminus="N",
|
|
46
|
+
gene2_terminus="C",
|
|
47
|
+
reference=ref,
|
|
48
|
+
)
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
This will return an object with the following shape. See the [api](./docs/api.md) for details.
|
|
52
|
+
|
|
53
|
+
```json
|
|
54
|
+
{
|
|
55
|
+
"frame_status": "in_frame",
|
|
56
|
+
"domains": [
|
|
57
|
+
{
|
|
58
|
+
"transcript_id": "ENST00000305877",
|
|
59
|
+
"interpro_id": "IPR036481",
|
|
60
|
+
"name": "Bcr-Abl oncoprotein oligomerisation domain superfamily",
|
|
61
|
+
"domain_type": "homologous_superfamily",
|
|
62
|
+
"start": 1,
|
|
63
|
+
"end": 67,
|
|
64
|
+
"sources": ["SuperFamily"],
|
|
65
|
+
"feature_ids": ["SSF69036"],
|
|
66
|
+
"breakpoint_based_status": "included",
|
|
67
|
+
"breakpoint_retained_percent": 100.0,
|
|
68
|
+
"post_splicing_status": "preserved",
|
|
69
|
+
"post_translation_status": "preserved"
|
|
70
|
+
},
|
|
71
|
+
...
|
|
72
|
+
],
|
|
73
|
+
"translation_start": {"ENST00000305877": "native_start_retained"}
|
|
74
|
+
}
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
Analysis uses the newest prepared local release by default.
|
|
78
|
+
|
|
79
|
+
## Documentation
|
|
80
|
+
|
|
81
|
+
- [API](docs/api.md)
|
|
82
|
+
- [Reference configuration and preparation](docs/reference.md)
|
|
83
|
+
- [Methodology](docs/methodology.md)
|
|
84
|
+
- [Testing](docs/testing.md)
|
|
85
|
+
|
|
86
|
+
## Important Limitations
|
|
87
|
+
|
|
88
|
+
- This package uses the splicing model defined by [MAVIS](https://github.com/bcgsc/mavis), this is non-exhaustive. It assumes splice sites to be disrupted based on a breakpoint being within 2bp but there are many other ways to disrupt splicing that are difficult to predict computationally (ex. deep intronic). This package only covers the standard scenarios
|
|
89
|
+
- Currently we only support Hg38
|
|
90
|
+
- Only exact breakpoints are supported
|
|
91
|
+
- Predicted structural consequences do not establish fusion expression, oncogenicity, pathogenicity, or clinical actionability.
|
|
92
|
+
|
|
93
|
+
## Citation
|
|
94
|
+
|
|
95
|
+
This package ports and extends selected fusion-annotation logic from [MAVIS](https://github.com/bcgsc/mavis). Please cite:
|
|
96
|
+
|
|
97
|
+
Reisle C, Mungall KL, Choo C, et al. MAVIS: merging, annotation, validation, and illustration of structural variants. *Bioinformatics*. 2019;35(3):515–517. PMID:30016509.
|
|
98
|
+
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
fusion_function/__init__.py,sha256=kGDFqDAJP8D_9M43wiPD_rtymDlENkvrH6VR0Uxg6vI,258
|
|
2
|
+
fusion_function/__main__.py,sha256=k1ocEWawweo1qCJWNFAAvyxz3tcY13dzvCenHszij30,48
|
|
3
|
+
fusion_function/cli.py,sha256=DqDcyYK0R4d88hDe8D-uLE2cHMks95rxl7exScrqbHo,767
|
|
4
|
+
fusion_function/data.py,sha256=Tu230UT1SAhjhRIMNTDGme_X5xCuG8vdVevfrS-m4mk,121039
|
|
5
|
+
fusion_function/ensembl.py,sha256=Rip1k7rQlDrsNVPiF4xQxIMjjInoIDeMqIg5GmM21Gc,3889
|
|
6
|
+
fusion_function/fusion.py,sha256=9_KPXMkvpkdrA9coMVQqILYnrnWGkm8xNi2_AwBDxUM,40495
|
|
7
|
+
fusion_function/interpro.py,sha256=TVQxzR3xO5y1dQb5-_kHmkRNTvFNjGwVZpxw5Bt1Hjc,731
|
|
8
|
+
fusion_function/prebuilt.py,sha256=Lyrc0WMGwormvrEkr8eac3Fg2xRF2CJ366vJl3WDon4,23713
|
|
9
|
+
fusion_function/reference.py,sha256=OOgLN_4MuEzczYamfJ5rbQv3bSBWFLocjE2V9jV5Jho,4407
|
|
10
|
+
fusion_function/reference_catalog.json,sha256=8REBPh6B46K-Z9eWp-Ef-gLKsyTlEZ4ifFoXcwMwAaU,9260
|
|
11
|
+
fusion_function/uniprot.py,sha256=izymwyMI6fog3mUc2vwWuhgs80fyH8BfEeg4hCy6On0,13862
|
|
12
|
+
fusion_function-0.2.1.dist-info/METADATA,sha256=zTidauqLnSTCufH65Tcr6eR8w2TamVHdvb0_-y18Ln0,3853
|
|
13
|
+
fusion_function-0.2.1.dist-info/WHEEL,sha256=L-WvLdSBvJWHXv5zrQxDvNRnIgoBzLUpcbuiw19KCuQ,88
|
|
14
|
+
fusion_function-0.2.1.dist-info/entry_points.txt,sha256=-cTv1JByye6d8iVffWHKz6yNApXduTlch2JooZK8KFY,60
|
|
15
|
+
fusion_function-0.2.1.dist-info/licenses/LICENSE,sha256=OXLcl0T2SZ8Pmy2_dmlvKuetivmyPd5m1q-Gyd-zaYY,35149
|
|
16
|
+
fusion_function-0.2.1.dist-info/RECORD,,
|