fusion-function 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fusion_function/__init__.py +9 -0
- fusion_function/__main__.py +3 -0
- fusion_function/cli.py +27 -0
- fusion_function/data.py +2683 -0
- fusion_function/ensembl.py +149 -0
- fusion_function/fusion.py +1024 -0
- fusion_function/interpro.py +29 -0
- fusion_function/prebuilt.py +518 -0
- fusion_function/reference.py +112 -0
- fusion_function/reference_catalog.json +82 -0
- fusion_function/uniprot.py +315 -0
- fusion_function-0.2.1.dist-info/METADATA +98 -0
- fusion_function-0.2.1.dist-info/RECORD +16 -0
- fusion_function-0.2.1.dist-info/WHEEL +4 -0
- fusion_function-0.2.1.dist-info/entry_points.txt +3 -0
- fusion_function-0.2.1.dist-info/licenses/LICENSE +674 -0
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""Local InterPro metadata access (no network requests)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import TYPE_CHECKING, TypedDict
|
|
7
|
+
|
|
8
|
+
from .reference import get_reference
|
|
9
|
+
|
|
10
|
+
if TYPE_CHECKING:
|
|
11
|
+
from .data import ReferenceReader
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class ProteinFeatureAnnotation(TypedDict):
|
|
15
|
+
name: str | None
|
|
16
|
+
entry_type: str | None
|
|
17
|
+
interpro_id: str | None
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def get_interpro_annotation(
|
|
21
|
+
interpro_id: str,
|
|
22
|
+
*,
|
|
23
|
+
reference: ReferenceReader | None = None,
|
|
24
|
+
database: str | Path | None = None,
|
|
25
|
+
release: int | None = None,
|
|
26
|
+
) -> ProteinFeatureAnnotation | None:
|
|
27
|
+
return get_reference(
|
|
28
|
+
reference=reference, database=database, release=release
|
|
29
|
+
).get_interpro_annotation(interpro_id)
|
|
@@ -0,0 +1,518 @@
|
|
|
1
|
+
"""Compact reference exports and checksum-pinned, resumable installation.
|
|
2
|
+
|
|
3
|
+
The publication catalog contains exact URLs, not a Zenodo username. Only
|
|
4
|
+
preparation uses it; runtime annotation continues to use local SQLite files.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import argparse
|
|
10
|
+
import gzip
|
|
11
|
+
import hashlib
|
|
12
|
+
import json
|
|
13
|
+
import logging
|
|
14
|
+
import os
|
|
15
|
+
import re
|
|
16
|
+
import sqlite3
|
|
17
|
+
import tempfile
|
|
18
|
+
import urllib.error
|
|
19
|
+
import urllib.parse
|
|
20
|
+
import urllib.request
|
|
21
|
+
from collections.abc import Mapping, Sequence
|
|
22
|
+
from contextlib import closing
|
|
23
|
+
from datetime import datetime, timezone
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
from typing import TYPE_CHECKING, Any
|
|
26
|
+
|
|
27
|
+
from . import data
|
|
28
|
+
|
|
29
|
+
if TYPE_CHECKING:
|
|
30
|
+
from http.client import HTTPResponse
|
|
31
|
+
|
|
32
|
+
CATALOG_URL = "https://raw.githubusercontent.com/creisle/fusion_function/main/src/fusion_function/reference_catalog.json"
|
|
33
|
+
CATALOG_PATH = Path(__file__).with_name("reference_catalog.json")
|
|
34
|
+
RUNTIME_TABLES = ("build_metadata", "ff_transcripts", "ff_interpro", "sequences", "sequence_chunks")
|
|
35
|
+
COMPATIBILITY = {
|
|
36
|
+
"format_version": "1",
|
|
37
|
+
"preprocessing_version": "1",
|
|
38
|
+
"cds_mapping_version": "2",
|
|
39
|
+
"feature_annotation_version": "3",
|
|
40
|
+
"transcript_payload_codec": data.TRANSCRIPT_PAYLOAD_CODEC,
|
|
41
|
+
"sequence_chunk_size": str(data.CHUNK_SIZE),
|
|
42
|
+
"sequence_codec": "zlib",
|
|
43
|
+
}
|
|
44
|
+
DATA_NOTICES = {
|
|
45
|
+
"Ensembl": {
|
|
46
|
+
"terms": "Unrestricted project-generated data; third-party constraints may apply.",
|
|
47
|
+
"url": "https://www.ensembl.org/info/about/legal/disclaimer.html",
|
|
48
|
+
},
|
|
49
|
+
"UniProt Consortium": {
|
|
50
|
+
"terms": "CC BY 4.0. Credit UniProt, link to the license and identify modifications.",
|
|
51
|
+
"url": "https://www.uniprot.org/help/license",
|
|
52
|
+
"license_url": "https://creativecommons.org/licenses/by/4.0/",
|
|
53
|
+
},
|
|
54
|
+
"InterPro Consortium": {
|
|
55
|
+
"terms": "Current InterPro downloads: CC0 1.0. Retain historical-source notices.",
|
|
56
|
+
"url": "https://interpro-documentation.readthedocs.io/en/latest/license.html",
|
|
57
|
+
"license_url": "https://creativecommons.org/publicdomain/zero/1.0/",
|
|
58
|
+
},
|
|
59
|
+
"PANTHER": {
|
|
60
|
+
"terms": "Classification release 14.1 and 17.0 READMEs carry GPL-2.0-or-later notices. Confirm terms for the imported release and derived classifications before redistribution.",
|
|
61
|
+
"url": "https://data.pantherdb.org/ftp/sequence_classifications/",
|
|
62
|
+
},
|
|
63
|
+
"Ensembl member annotations": {
|
|
64
|
+
"terms": "Member-source terms are not replaced by Ensembl or InterPro terms. PROSITE database terms are CC BY-NC-ND 4.0 with commercial licensing; SMART models require a license. Confirm terms for derived match annotations.",
|
|
65
|
+
"url": "https://prosite.expasy.org/prosite_license.html",
|
|
66
|
+
"smart_url": "https://smart.embl.de/about.cgi",
|
|
67
|
+
},
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def https_url(value: object) -> str:
|
|
72
|
+
"""Require public HTTPS URLs without embedded credentials."""
|
|
73
|
+
if not isinstance(value, str):
|
|
74
|
+
raise ValueError("Reference URL must be an HTTPS URL")
|
|
75
|
+
parsed = urllib.parse.urlparse(value)
|
|
76
|
+
if (
|
|
77
|
+
parsed.scheme != "https"
|
|
78
|
+
or not parsed.netloc
|
|
79
|
+
or parsed.username
|
|
80
|
+
or parsed.password
|
|
81
|
+
or parsed.fragment
|
|
82
|
+
):
|
|
83
|
+
raise ValueError("Reference URL must be HTTPS without credentials or a fragment")
|
|
84
|
+
return value
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def read_catalog(source: str | None = None) -> list[dict[str, Any]]:
|
|
88
|
+
"""Refresh the default catalog, falling back to its bundled copy on outages.
|
|
89
|
+
|
|
90
|
+
Explicit catalogs and malformed data fail clearly. A failed artifact
|
|
91
|
+
download never silently starts an expensive source build.
|
|
92
|
+
"""
|
|
93
|
+
selected = source or os.environ.get("FUSION_FUNCTION_REFERENCE_CATALOG")
|
|
94
|
+
if selected and "://" not in selected:
|
|
95
|
+
raw = Path(selected).expanduser().read_bytes()
|
|
96
|
+
else:
|
|
97
|
+
try:
|
|
98
|
+
with data.open_url(https_url(selected or CATALOG_URL)) as response:
|
|
99
|
+
raw = response.read(4 * data.CHUNK_SIZE + 1)
|
|
100
|
+
if len(raw) > 4 * data.CHUNK_SIZE:
|
|
101
|
+
raise ValueError("Reference catalog exceeds 4 MiB")
|
|
102
|
+
except (OSError, urllib.error.URLError) as exc:
|
|
103
|
+
if selected:
|
|
104
|
+
raise
|
|
105
|
+
data.LOG.warning("Reference catalog unavailable (%s); using bundled catalog", exc)
|
|
106
|
+
raw = CATALOG_PATH.read_bytes()
|
|
107
|
+
catalog = json.loads(raw)
|
|
108
|
+
if (
|
|
109
|
+
not isinstance(catalog, dict)
|
|
110
|
+
or catalog.get("catalog_version") != 1
|
|
111
|
+
or not isinstance(catalog.get("references"), list)
|
|
112
|
+
):
|
|
113
|
+
raise ValueError(
|
|
114
|
+
"Invalid reference catalog; expected catalog_version=1 and references list"
|
|
115
|
+
)
|
|
116
|
+
entries = []
|
|
117
|
+
for entry in catalog["references"]:
|
|
118
|
+
if not isinstance(entry, dict):
|
|
119
|
+
raise ValueError("Reference catalog entries must be objects")
|
|
120
|
+
# Export manifests can exist before their Zenodo record is published.
|
|
121
|
+
if entry.get("url") is None:
|
|
122
|
+
continue
|
|
123
|
+
https_url(entry["url"])
|
|
124
|
+
for key in ("release", "build_revision", "archive_bytes", "database_bytes"):
|
|
125
|
+
if type(entry.get(key)) is not int or entry[key] <= 0:
|
|
126
|
+
raise ValueError(f"Reference catalog {key} must be a positive integer")
|
|
127
|
+
for key in ("archive_sha256", "database_sha256"):
|
|
128
|
+
if not isinstance(entry.get(key), str) or not re.fullmatch(r"[0-9a-f]{64}", entry[key]):
|
|
129
|
+
raise ValueError(f"Invalid reference catalog {key}")
|
|
130
|
+
if entry.get("compression") != "gzip" or not isinstance(entry.get("metadata"), dict):
|
|
131
|
+
raise ValueError("Reference catalog requires gzip compression and metadata")
|
|
132
|
+
entries.append(entry)
|
|
133
|
+
return entries
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def select_reference(
|
|
137
|
+
entries: Sequence[Mapping[str, Any]], release: int
|
|
138
|
+
) -> Mapping[str, Any] | None:
|
|
139
|
+
"""Choose the newest compatible build of the exact requested release."""
|
|
140
|
+
candidates = [
|
|
141
|
+
entry
|
|
142
|
+
for entry in entries
|
|
143
|
+
if entry["release"] == release
|
|
144
|
+
and entry.get("species") == data.SPECIES
|
|
145
|
+
and entry.get("assembly") == data.ASSEMBLY
|
|
146
|
+
and all(entry["metadata"].get(key) == value for key, value in COMPATIBILITY.items())
|
|
147
|
+
]
|
|
148
|
+
if not candidates:
|
|
149
|
+
return None
|
|
150
|
+
revision = max(entry["build_revision"] for entry in candidates)
|
|
151
|
+
newest = [entry for entry in candidates if entry["build_revision"] == revision]
|
|
152
|
+
if len(newest) != 1:
|
|
153
|
+
raise ValueError(f"Ambiguous reference for release {release}, build {revision}")
|
|
154
|
+
return newest[0]
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def open_download(request: urllib.request.Request) -> HTTPResponse:
|
|
158
|
+
"""Open an anonymous download, including Range headers for interrupted files."""
|
|
159
|
+
return urllib.request.urlopen(request, timeout=60) # type: ignore[no-any-return]
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def download_archive(entry: Mapping[str, Any], cache: Path) -> Path:
|
|
163
|
+
"""Lock the shared archive cache, so installs to different outputs cannot race."""
|
|
164
|
+
directory = cache / "prebuilt" / entry["archive_sha256"]
|
|
165
|
+
directory.mkdir(parents=True, exist_ok=True)
|
|
166
|
+
with data.build_lock(directory / ".download.lock"):
|
|
167
|
+
return _download_archive(entry, directory)
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _download_archive(entry: Mapping[str, Any], directory: Path) -> Path:
|
|
171
|
+
"""Resume partial bytes; restart when Range is ignored and verify checksums.
|
|
172
|
+
|
|
173
|
+
Interrupted transfers retain partial bytes. Complete but invalid files are
|
|
174
|
+
deleted so the next attempt can start again.
|
|
175
|
+
"""
|
|
176
|
+
archive = directory / "ensembl.sqlite.gz"
|
|
177
|
+
partial = archive.with_name(archive.name + ".part")
|
|
178
|
+
size = entry["archive_bytes"]
|
|
179
|
+
data.LOG.info("Prebuilt archive: %s; resumable partial: %s", archive, partial)
|
|
180
|
+
if not archive.exists():
|
|
181
|
+
offset = partial.stat().st_size if partial.exists() else 0
|
|
182
|
+
if offset > size:
|
|
183
|
+
partial.unlink()
|
|
184
|
+
offset = 0
|
|
185
|
+
if offset < size:
|
|
186
|
+
request = urllib.request.Request(
|
|
187
|
+
entry["url"], headers={"Range": f"bytes={offset}-"} if offset else {}
|
|
188
|
+
)
|
|
189
|
+
with open_download(request) as response:
|
|
190
|
+
if response.status == 206:
|
|
191
|
+
if response.headers.get("Content-Range") != f"bytes {offset}-{size - 1}/{size}":
|
|
192
|
+
raise ValueError("Unexpected Content-Range for prebuilt reference")
|
|
193
|
+
elif response.status == 200:
|
|
194
|
+
offset = 0
|
|
195
|
+
else:
|
|
196
|
+
raise OSError(f"Unexpected reference download status: {response.status}")
|
|
197
|
+
with (
|
|
198
|
+
partial.open("ab" if offset else "wb") as stream,
|
|
199
|
+
data.progress(
|
|
200
|
+
total=size, desc="Download prebuilt reference", unit="B", unit_scale=True
|
|
201
|
+
) as bar,
|
|
202
|
+
):
|
|
203
|
+
bar.update(offset)
|
|
204
|
+
received = offset
|
|
205
|
+
while block := response.read(4 * data.CHUNK_SIZE):
|
|
206
|
+
if received + len(block) > size:
|
|
207
|
+
raise ValueError("Reference archive exceeds the catalog size")
|
|
208
|
+
stream.write(block)
|
|
209
|
+
received += len(block)
|
|
210
|
+
bar.update(len(block))
|
|
211
|
+
if received != size:
|
|
212
|
+
raise OSError(
|
|
213
|
+
f"Incomplete reference download: {received}/{size} bytes; rerun to resume"
|
|
214
|
+
)
|
|
215
|
+
if (
|
|
216
|
+
partial.stat().st_size != size
|
|
217
|
+
or data.sha256_file(partial, show_progress=True) != entry["archive_sha256"]
|
|
218
|
+
):
|
|
219
|
+
partial.unlink()
|
|
220
|
+
raise ValueError("Prebuilt archive checksum/size mismatch; rerun to download again")
|
|
221
|
+
partial.replace(archive)
|
|
222
|
+
elif (
|
|
223
|
+
archive.stat().st_size != size
|
|
224
|
+
or data.sha256_file(archive, show_progress=True) != entry["archive_sha256"]
|
|
225
|
+
):
|
|
226
|
+
archive.unlink()
|
|
227
|
+
raise ValueError("Cached prebuilt archive checksum/size mismatch; rerun to download again")
|
|
228
|
+
return archive
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def validate_database(path: Path, release: int, metadata: Mapping[str, str]) -> None:
|
|
232
|
+
"""Check metadata and runtime tables without another expensive integrity scan."""
|
|
233
|
+
from .reference import ReferenceDatabase
|
|
234
|
+
|
|
235
|
+
with ReferenceDatabase(path, release=release) as reader:
|
|
236
|
+
for key, value in metadata.items():
|
|
237
|
+
if reader.metadata.get(key) != value:
|
|
238
|
+
raise ValueError(f"Prebuilt database metadata mismatch: {key}")
|
|
239
|
+
tables = {
|
|
240
|
+
row[0] for row in reader.db.execute("SELECT name FROM sqlite_master WHERE type='table'")
|
|
241
|
+
}
|
|
242
|
+
if set(RUNTIME_TABLES) - tables:
|
|
243
|
+
raise ValueError("Prebuilt database is missing runtime tables")
|
|
244
|
+
if (
|
|
245
|
+
reader.db.execute(
|
|
246
|
+
"SELECT 1 FROM ff_transcripts WHERE status='ready' LIMIT 1"
|
|
247
|
+
).fetchone()
|
|
248
|
+
is None
|
|
249
|
+
):
|
|
250
|
+
raise ValueError("Prebuilt database contains no usable coding transcripts")
|
|
251
|
+
if (
|
|
252
|
+
reader.db.execute("SELECT 1 FROM sequence_chunks WHERE kind='dna' LIMIT 1").fetchone()
|
|
253
|
+
is None
|
|
254
|
+
):
|
|
255
|
+
raise ValueError("Prebuilt database contains no genome sequence chunks")
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def install_reference(
|
|
259
|
+
*,
|
|
260
|
+
release: int | None,
|
|
261
|
+
cache_dir: Path,
|
|
262
|
+
output: Path | None = None,
|
|
263
|
+
force: bool = False,
|
|
264
|
+
catalog: str | None = None,
|
|
265
|
+
) -> Path | None:
|
|
266
|
+
"""Install a published reference, or return None when a source build is needed."""
|
|
267
|
+
entries = read_catalog(catalog)
|
|
268
|
+
if not entries:
|
|
269
|
+
data.LOG.info("No published prebuilt references in the catalog; building from Ensembl FTP")
|
|
270
|
+
return None
|
|
271
|
+
if release is None:
|
|
272
|
+
# Latest still means latest Ensembl, not latest uploaded database.
|
|
273
|
+
release, _, _ = data.resolve_core(data.BASE_URL, None)
|
|
274
|
+
entry = select_reference(entries, release)
|
|
275
|
+
if entry is None:
|
|
276
|
+
data.LOG.info("No compatible prebuilt for Ensembl %s; building from Ensembl FTP", release)
|
|
277
|
+
return None
|
|
278
|
+
path = (
|
|
279
|
+
(
|
|
280
|
+
output
|
|
281
|
+
or cache_dir / data.SPECIES / data.ASSEMBLY / f"release-{release}" / "ensembl.sqlite"
|
|
282
|
+
)
|
|
283
|
+
.expanduser()
|
|
284
|
+
.resolve()
|
|
285
|
+
)
|
|
286
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
287
|
+
data.LOG.info(
|
|
288
|
+
"Selected prebuilt: Ensembl %s, build %s, %s",
|
|
289
|
+
release,
|
|
290
|
+
entry["build_revision"],
|
|
291
|
+
entry["url"],
|
|
292
|
+
)
|
|
293
|
+
data.LOG.info(
|
|
294
|
+
"Final reference: %s; expected installed size: %s bytes", path, entry["database_bytes"]
|
|
295
|
+
)
|
|
296
|
+
staging = path.with_name(path.name + ".prebuilt.part")
|
|
297
|
+
data.LOG.info("Expanded partial database: %s", staging)
|
|
298
|
+
steps = data.BuildSteps(4)
|
|
299
|
+
with data.build_lock(path.with_name(path.name + ".prepare.lock")):
|
|
300
|
+
if path.exists() and not force:
|
|
301
|
+
validate_database(path, release, COMPATIBILITY)
|
|
302
|
+
data.LOG.info("Using existing reference: %s; --force replaces it", path)
|
|
303
|
+
return path
|
|
304
|
+
try:
|
|
305
|
+
with steps.step("Download and verify prebuilt archive"):
|
|
306
|
+
data.LOG.info("A large reference download can take over 10 minutes")
|
|
307
|
+
archive = download_archive(entry, cache_dir)
|
|
308
|
+
with steps.step("Expand reference database"):
|
|
309
|
+
data.LOG.info("Expanding a large reference can take over 10 minutes")
|
|
310
|
+
digest = hashlib.sha256()
|
|
311
|
+
size = 0
|
|
312
|
+
with (
|
|
313
|
+
gzip.open(archive, "rb") as source,
|
|
314
|
+
staging.open("wb") as target,
|
|
315
|
+
data.progress(
|
|
316
|
+
total=entry["database_bytes"],
|
|
317
|
+
desc="Expand prebuilt reference",
|
|
318
|
+
unit="B",
|
|
319
|
+
unit_scale=True,
|
|
320
|
+
) as bar,
|
|
321
|
+
):
|
|
322
|
+
while block := source.read(4 * data.CHUNK_SIZE):
|
|
323
|
+
size += len(block)
|
|
324
|
+
if size > entry["database_bytes"]:
|
|
325
|
+
raise ValueError("Expanded reference exceeds the catalog size")
|
|
326
|
+
digest.update(block)
|
|
327
|
+
target.write(block)
|
|
328
|
+
bar.update(len(block))
|
|
329
|
+
if (
|
|
330
|
+
size != entry["database_bytes"]
|
|
331
|
+
or digest.hexdigest() != entry["database_sha256"]
|
|
332
|
+
):
|
|
333
|
+
raise ValueError("Expanded reference checksum/size mismatch")
|
|
334
|
+
with steps.step("Validate reference format and release"):
|
|
335
|
+
validate_database(staging, release, entry["metadata"])
|
|
336
|
+
with steps.step("Install completed reference"):
|
|
337
|
+
os.replace(staging, path)
|
|
338
|
+
finally:
|
|
339
|
+
staging.unlink(missing_ok=True)
|
|
340
|
+
steps.complete()
|
|
341
|
+
return path
|
|
342
|
+
|
|
343
|
+
|
|
344
|
+
def export_reference(
|
|
345
|
+
database: Path, directory: Path, *, build_revision: int = 1, zenodo_record: int | None = None
|
|
346
|
+
) -> Path:
|
|
347
|
+
"""Export a read-only snapshot of runtime data; preserve the maintainer's DB.
|
|
348
|
+
|
|
349
|
+
Bulk INSERT SELECT copies compressed payloads without Python decoding. The
|
|
350
|
+
attached source is read-only and one transaction holds a consistent snapshot.
|
|
351
|
+
"""
|
|
352
|
+
if build_revision <= 0 or (zenodo_record is not None and zenodo_record <= 0):
|
|
353
|
+
raise ValueError("Build revision and Zenodo record must be positive")
|
|
354
|
+
database = database.expanduser().resolve()
|
|
355
|
+
directory = directory.expanduser().resolve()
|
|
356
|
+
directory.mkdir(parents=True, exist_ok=True)
|
|
357
|
+
with data.ReferenceReader(database) as reader:
|
|
358
|
+
metadata = dict(reader.metadata)
|
|
359
|
+
release = int(metadata["release"])
|
|
360
|
+
validate_database(database, release, COMPATIBILITY)
|
|
361
|
+
stem = f"{data.SPECIES}.{data.ASSEMBLY}.ensembl-{release}.ff-v1.build-{build_revision}"
|
|
362
|
+
archive, manifest = directory / (stem + ".sqlite.gz"), directory / (stem + ".json")
|
|
363
|
+
with data.build_lock(directory / (stem + ".export.lock")):
|
|
364
|
+
if archive.exists() or manifest.exists():
|
|
365
|
+
raise FileExistsError("Export already exists; use a new build revision or directory")
|
|
366
|
+
with tempfile.TemporaryDirectory(prefix=stem + ".", dir=directory) as workspace:
|
|
367
|
+
compact = Path(workspace) / "ensembl.sqlite"
|
|
368
|
+
data.LOG.info("Export destination: %s; temporary database: %s", archive, compact)
|
|
369
|
+
steps = data.BuildSteps(4)
|
|
370
|
+
with closing(sqlite3.connect(compact, uri=True)) as target:
|
|
371
|
+
target.execute("ATTACH DATABASE ? AS original", (database.as_uri() + "?mode=ro",))
|
|
372
|
+
target.execute("PRAGMA secure_delete=OFF")
|
|
373
|
+
target.execute("PRAGMA user_version=1")
|
|
374
|
+
target.execute("BEGIN")
|
|
375
|
+
with steps.step("Copy runtime tables into a compact database"):
|
|
376
|
+
data.LOG.info("Copying a large reference can take over 10 minutes")
|
|
377
|
+
for table in (*RUNTIME_TABLES, "source_files"):
|
|
378
|
+
schema = target.execute(
|
|
379
|
+
"SELECT sql FROM original.sqlite_master WHERE type='table' AND name=?",
|
|
380
|
+
(table,),
|
|
381
|
+
).fetchone()
|
|
382
|
+
if schema is None:
|
|
383
|
+
if table == "source_files":
|
|
384
|
+
continue
|
|
385
|
+
raise ValueError(f"Missing runtime table: {table}")
|
|
386
|
+
target.execute(schema[0])
|
|
387
|
+
data.sqlite_phase(
|
|
388
|
+
target,
|
|
389
|
+
f'INSERT INTO "{table}" SELECT * FROM original."{table}"',
|
|
390
|
+
f"Export {table}",
|
|
391
|
+
)
|
|
392
|
+
for (sql,) in target.execute(
|
|
393
|
+
"SELECT sql FROM original.sqlite_master WHERE type='index' AND tbl_name=? AND sql IS NOT NULL",
|
|
394
|
+
(table,),
|
|
395
|
+
).fetchall():
|
|
396
|
+
data.sqlite_phase(target, sql, f"Index exported {table}")
|
|
397
|
+
# Keep original provenance, plus notices describing our transformation.
|
|
398
|
+
metadata = dict(target.execute("SELECT * FROM build_metadata"))
|
|
399
|
+
target.executemany(
|
|
400
|
+
"INSERT OR REPLACE INTO build_metadata VALUES (?,?)",
|
|
401
|
+
[
|
|
402
|
+
("reference_kind", "runtime"),
|
|
403
|
+
("reference_build_revision", str(build_revision)),
|
|
404
|
+
("exported_utc", datetime.now(timezone.utc).isoformat()),
|
|
405
|
+
("data_source_notices", json.dumps(DATA_NOTICES, sort_keys=True)),
|
|
406
|
+
(
|
|
407
|
+
"data_modifications",
|
|
408
|
+
"Runtime subset; derived transcript, feature and splice mappings; lossless compression.",
|
|
409
|
+
),
|
|
410
|
+
],
|
|
411
|
+
)
|
|
412
|
+
target.commit()
|
|
413
|
+
with steps.step("Validate exported SQLite integrity"):
|
|
414
|
+
if data.sqlite_phase(target, "PRAGMA integrity_check", "Check database") != [
|
|
415
|
+
("ok",)
|
|
416
|
+
]:
|
|
417
|
+
raise ValueError("Exported SQLite integrity check failed")
|
|
418
|
+
validate_database(compact, release, COMPATIBILITY)
|
|
419
|
+
with steps.step("Compress exported reference"):
|
|
420
|
+
data.LOG.info("Compressing a large reference can take over 10 minutes")
|
|
421
|
+
digest = hashlib.sha256()
|
|
422
|
+
temporary_archive = Path(workspace) / archive.name
|
|
423
|
+
with (
|
|
424
|
+
compact.open("rb") as source,
|
|
425
|
+
temporary_archive.open("wb") as output,
|
|
426
|
+
gzip.GzipFile(
|
|
427
|
+
filename="", fileobj=output, mode="wb", compresslevel=1, mtime=0
|
|
428
|
+
) as compressed,
|
|
429
|
+
data.progress(
|
|
430
|
+
total=compact.stat().st_size,
|
|
431
|
+
desc="Compress reference",
|
|
432
|
+
unit="B",
|
|
433
|
+
unit_scale=True,
|
|
434
|
+
) as bar,
|
|
435
|
+
):
|
|
436
|
+
while block := source.read(4 * data.CHUNK_SIZE):
|
|
437
|
+
digest.update(block)
|
|
438
|
+
compressed.write(block)
|
|
439
|
+
bar.update(len(block))
|
|
440
|
+
with steps.step("Write manifest and publish export"):
|
|
441
|
+
entry = {
|
|
442
|
+
"species": data.SPECIES,
|
|
443
|
+
"assembly": data.ASSEMBLY,
|
|
444
|
+
"release": release,
|
|
445
|
+
"build_revision": build_revision,
|
|
446
|
+
"url": f"https://zenodo.org/records/{zenodo_record}/files/{archive.name}?download=1"
|
|
447
|
+
if zenodo_record
|
|
448
|
+
else None,
|
|
449
|
+
"filename": archive.name,
|
|
450
|
+
"compression": "gzip",
|
|
451
|
+
"archive_bytes": temporary_archive.stat().st_size,
|
|
452
|
+
"archive_sha256": data.sha256_file(temporary_archive, show_progress=True),
|
|
453
|
+
"database_bytes": compact.stat().st_size,
|
|
454
|
+
"database_sha256": digest.hexdigest(),
|
|
455
|
+
"metadata": COMPATIBILITY,
|
|
456
|
+
"data_source_notices": DATA_NOTICES,
|
|
457
|
+
"source_metadata": metadata,
|
|
458
|
+
}
|
|
459
|
+
temporary_manifest = Path(workspace) / manifest.name
|
|
460
|
+
temporary_manifest.write_text(
|
|
461
|
+
json.dumps({"catalog_version": 1, "references": [entry]}, indent=2) + "\n"
|
|
462
|
+
)
|
|
463
|
+
temporary_archive.replace(archive)
|
|
464
|
+
temporary_manifest.replace(manifest)
|
|
465
|
+
steps.complete()
|
|
466
|
+
if zenodo_record is None:
|
|
467
|
+
data.LOG.info(
|
|
468
|
+
"Manifest URL is unset; add the published Zenodo file URL before registering it"
|
|
469
|
+
)
|
|
470
|
+
data.LOG.info("Export archive: %s; manifest: %s", archive, manifest)
|
|
471
|
+
return manifest
|
|
472
|
+
|
|
473
|
+
|
|
474
|
+
def main(argv: Sequence[str] | None = None) -> int:
|
|
475
|
+
"""Maintainer CLI to export a tested reference for publication."""
|
|
476
|
+
from .reference import ReferenceDatabase
|
|
477
|
+
|
|
478
|
+
parser = argparse.ArgumentParser(prog="fusion-function export-reference")
|
|
479
|
+
parser.add_argument(
|
|
480
|
+
"database",
|
|
481
|
+
type=Path,
|
|
482
|
+
nargs="?",
|
|
483
|
+
help="Existing database; default: FUSION_FUNCTION_DB or newest installed local release",
|
|
484
|
+
)
|
|
485
|
+
parser.add_argument(
|
|
486
|
+
"--release", type=int, help="Installed Ensembl release; default: newest installed release"
|
|
487
|
+
)
|
|
488
|
+
parser.add_argument(
|
|
489
|
+
"--output", required=True, type=Path, help="Directory for compressed DB and manifest"
|
|
490
|
+
)
|
|
491
|
+
parser.add_argument("--build-revision", type=int, default=1)
|
|
492
|
+
parser.add_argument(
|
|
493
|
+
"--zenodo-record", type=int, help="Reserved/published Zenodo record ID; not a username"
|
|
494
|
+
)
|
|
495
|
+
args = parser.parse_args(argv)
|
|
496
|
+
if args.release is not None and args.release <= 0:
|
|
497
|
+
parser.error("--release must be positive")
|
|
498
|
+
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
|
|
499
|
+
try:
|
|
500
|
+
# Reuse analysis selection, including cache overrides and release checks.
|
|
501
|
+
# Selection reads local files only; export never builds or downloads a DB.
|
|
502
|
+
with ReferenceDatabase(args.database, release=args.release) as reference:
|
|
503
|
+
database = reference.path
|
|
504
|
+
data.LOG.info("Export source database: %s", database)
|
|
505
|
+
manifest = export_reference(
|
|
506
|
+
database,
|
|
507
|
+
args.output,
|
|
508
|
+
build_revision=args.build_revision,
|
|
509
|
+
zenodo_record=args.zenodo_record,
|
|
510
|
+
)
|
|
511
|
+
except (OSError, ValueError, sqlite3.Error) as exc:
|
|
512
|
+
data.LOG.error("Export failed: %s", exc)
|
|
513
|
+
return 1
|
|
514
|
+
except KeyboardInterrupt:
|
|
515
|
+
data.LOG.error("Export interrupted; source database unchanged")
|
|
516
|
+
return 130
|
|
517
|
+
print(manifest)
|
|
518
|
+
return 0
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
"""Reference configuration and read-only access to prepared human databases."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import os
|
|
6
|
+
import re
|
|
7
|
+
import threading
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
from .data import ASSEMBLY, SPECIES, ReferenceReader, default_cache_dir
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def resolve_database(
|
|
14
|
+
database: str | Path | None = None,
|
|
15
|
+
*,
|
|
16
|
+
release: int | None = None,
|
|
17
|
+
cache_dir: str | Path | None = None,
|
|
18
|
+
) -> Path:
|
|
19
|
+
"""Choose explicit path, FUSION_FUNCTION_DB, or newest prepared local release."""
|
|
20
|
+
if release is not None and (not isinstance(release, int) or release <= 0):
|
|
21
|
+
raise ValueError("release must be a positive Ensembl release number")
|
|
22
|
+
explicit = database or os.environ.get("FUSION_FUNCTION_DB")
|
|
23
|
+
if explicit:
|
|
24
|
+
path = Path(explicit).expanduser().resolve()
|
|
25
|
+
else:
|
|
26
|
+
root = Path(cache_dir or default_cache_dir()).expanduser() / SPECIES / ASSEMBLY
|
|
27
|
+
if release is not None:
|
|
28
|
+
path = root / f"release-{release}" / "ensembl.sqlite"
|
|
29
|
+
else:
|
|
30
|
+
candidates = []
|
|
31
|
+
if root.is_dir():
|
|
32
|
+
for directory in root.iterdir():
|
|
33
|
+
match = re.fullmatch(r"release-(\d+)", directory.name)
|
|
34
|
+
if match and (directory / "ensembl.sqlite").is_file():
|
|
35
|
+
candidates.append((int(match[1]), directory / "ensembl.sqlite"))
|
|
36
|
+
if not candidates:
|
|
37
|
+
raise FileNotFoundError(
|
|
38
|
+
f"No prepared human reference in {root}. Run 'fusion-function prepare-data' "
|
|
39
|
+
"once, or set FUSION_FUNCTION_DB to an existing preprocessed database."
|
|
40
|
+
)
|
|
41
|
+
path = max(candidates)[1]
|
|
42
|
+
path = path.resolve()
|
|
43
|
+
if not path.is_file():
|
|
44
|
+
raise FileNotFoundError(
|
|
45
|
+
f"Reference database not found: {path}. Run 'fusion-function prepare-data'."
|
|
46
|
+
)
|
|
47
|
+
return path
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class ReferenceDatabase(ReferenceReader):
|
|
51
|
+
"""Reusable per-thread reference reader; close explicitly or use a context manager."""
|
|
52
|
+
|
|
53
|
+
def __init__(
|
|
54
|
+
self,
|
|
55
|
+
database: str | Path | None = None,
|
|
56
|
+
*,
|
|
57
|
+
release: int | None = None,
|
|
58
|
+
cache_dir: str | Path | None = None,
|
|
59
|
+
cached_chunks: int = 64,
|
|
60
|
+
) -> None:
|
|
61
|
+
self.path = resolve_database(database, release=release, cache_dir=cache_dir)
|
|
62
|
+
super().__init__(self.path, cached_chunks=cached_chunks)
|
|
63
|
+
try:
|
|
64
|
+
if self.metadata.get("format_version") != "1":
|
|
65
|
+
raise ValueError(
|
|
66
|
+
"Unsupported reference format; rebuild with 'fusion-function prepare-data'"
|
|
67
|
+
)
|
|
68
|
+
if self.metadata.get("species") != SPECIES:
|
|
69
|
+
raise ValueError("Only human reference databases are supported")
|
|
70
|
+
if self.metadata.get("assembly") != ASSEMBLY:
|
|
71
|
+
raise ValueError("Only GRCh38 reference databases are supported")
|
|
72
|
+
if release is not None and self.metadata.get("release") != str(release):
|
|
73
|
+
raise ValueError(
|
|
74
|
+
f"Requested release {release}, but database has release {self.metadata.get('release')}"
|
|
75
|
+
)
|
|
76
|
+
except BaseException:
|
|
77
|
+
self.close()
|
|
78
|
+
raise
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
_local = threading.local()
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def close_default_reference() -> None:
|
|
85
|
+
"""Release this thread's automatically cached connection and genome chunks."""
|
|
86
|
+
reader = getattr(_local, "reader", None)
|
|
87
|
+
if reader is not None:
|
|
88
|
+
reader.close()
|
|
89
|
+
del _local.reader
|
|
90
|
+
del _local.signature
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def get_reference(
|
|
94
|
+
*,
|
|
95
|
+
reference: ReferenceReader | None = None,
|
|
96
|
+
database: str | Path | None = None,
|
|
97
|
+
release: int | None = None,
|
|
98
|
+
) -> ReferenceReader:
|
|
99
|
+
if reference is not None:
|
|
100
|
+
if database is not None or release is not None:
|
|
101
|
+
raise ValueError("Use reference, or database/release; these cannot be combined")
|
|
102
|
+
return reference
|
|
103
|
+
path = resolve_database(database, release=release)
|
|
104
|
+
stat = path.stat()
|
|
105
|
+
signature = (os.getpid(), str(path), stat.st_ino, stat.st_size, stat.st_mtime_ns)
|
|
106
|
+
if getattr(_local, "signature", None) != signature:
|
|
107
|
+
close_default_reference()
|
|
108
|
+
_local.reader = ReferenceDatabase(path, release=release)
|
|
109
|
+
_local.signature = signature
|
|
110
|
+
elif release is not None and _local.reader.metadata.get("release") != str(release):
|
|
111
|
+
raise ValueError("Requested release does not match the selected database")
|
|
112
|
+
return _local.reader
|