fusion-function 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,29 @@
1
+ """Local InterPro metadata access (no network requests)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from pathlib import Path
6
+ from typing import TYPE_CHECKING, TypedDict
7
+
8
+ from .reference import get_reference
9
+
10
+ if TYPE_CHECKING:
11
+ from .data import ReferenceReader
12
+
13
+
14
+ class ProteinFeatureAnnotation(TypedDict):
15
+ name: str | None
16
+ entry_type: str | None
17
+ interpro_id: str | None
18
+
19
+
20
+ def get_interpro_annotation(
21
+ interpro_id: str,
22
+ *,
23
+ reference: ReferenceReader | None = None,
24
+ database: str | Path | None = None,
25
+ release: int | None = None,
26
+ ) -> ProteinFeatureAnnotation | None:
27
+ return get_reference(
28
+ reference=reference, database=database, release=release
29
+ ).get_interpro_annotation(interpro_id)
@@ -0,0 +1,518 @@
1
+ """Compact reference exports and checksum-pinned, resumable installation.
2
+
3
+ The publication catalog contains exact URLs, not a Zenodo username. Only
4
+ preparation uses it; runtime annotation continues to use local SQLite files.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import argparse
10
+ import gzip
11
+ import hashlib
12
+ import json
13
+ import logging
14
+ import os
15
+ import re
16
+ import sqlite3
17
+ import tempfile
18
+ import urllib.error
19
+ import urllib.parse
20
+ import urllib.request
21
+ from collections.abc import Mapping, Sequence
22
+ from contextlib import closing
23
+ from datetime import datetime, timezone
24
+ from pathlib import Path
25
+ from typing import TYPE_CHECKING, Any
26
+
27
+ from . import data
28
+
29
+ if TYPE_CHECKING:
30
+ from http.client import HTTPResponse
31
+
32
+ CATALOG_URL = "https://raw.githubusercontent.com/creisle/fusion_function/main/src/fusion_function/reference_catalog.json"
33
+ CATALOG_PATH = Path(__file__).with_name("reference_catalog.json")
34
+ RUNTIME_TABLES = ("build_metadata", "ff_transcripts", "ff_interpro", "sequences", "sequence_chunks")
35
+ COMPATIBILITY = {
36
+ "format_version": "1",
37
+ "preprocessing_version": "1",
38
+ "cds_mapping_version": "2",
39
+ "feature_annotation_version": "3",
40
+ "transcript_payload_codec": data.TRANSCRIPT_PAYLOAD_CODEC,
41
+ "sequence_chunk_size": str(data.CHUNK_SIZE),
42
+ "sequence_codec": "zlib",
43
+ }
44
+ DATA_NOTICES = {
45
+ "Ensembl": {
46
+ "terms": "Unrestricted project-generated data; third-party constraints may apply.",
47
+ "url": "https://www.ensembl.org/info/about/legal/disclaimer.html",
48
+ },
49
+ "UniProt Consortium": {
50
+ "terms": "CC BY 4.0. Credit UniProt, link to the license and identify modifications.",
51
+ "url": "https://www.uniprot.org/help/license",
52
+ "license_url": "https://creativecommons.org/licenses/by/4.0/",
53
+ },
54
+ "InterPro Consortium": {
55
+ "terms": "Current InterPro downloads: CC0 1.0. Retain historical-source notices.",
56
+ "url": "https://interpro-documentation.readthedocs.io/en/latest/license.html",
57
+ "license_url": "https://creativecommons.org/publicdomain/zero/1.0/",
58
+ },
59
+ "PANTHER": {
60
+ "terms": "Classification release 14.1 and 17.0 READMEs carry GPL-2.0-or-later notices. Confirm terms for the imported release and derived classifications before redistribution.",
61
+ "url": "https://data.pantherdb.org/ftp/sequence_classifications/",
62
+ },
63
+ "Ensembl member annotations": {
64
+ "terms": "Member-source terms are not replaced by Ensembl or InterPro terms. PROSITE database terms are CC BY-NC-ND 4.0 with commercial licensing; SMART models require a license. Confirm terms for derived match annotations.",
65
+ "url": "https://prosite.expasy.org/prosite_license.html",
66
+ "smart_url": "https://smart.embl.de/about.cgi",
67
+ },
68
+ }
69
+
70
+
71
+ def https_url(value: object) -> str:
72
+ """Require public HTTPS URLs without embedded credentials."""
73
+ if not isinstance(value, str):
74
+ raise ValueError("Reference URL must be an HTTPS URL")
75
+ parsed = urllib.parse.urlparse(value)
76
+ if (
77
+ parsed.scheme != "https"
78
+ or not parsed.netloc
79
+ or parsed.username
80
+ or parsed.password
81
+ or parsed.fragment
82
+ ):
83
+ raise ValueError("Reference URL must be HTTPS without credentials or a fragment")
84
+ return value
85
+
86
+
87
+ def read_catalog(source: str | None = None) -> list[dict[str, Any]]:
88
+ """Refresh the default catalog, falling back to its bundled copy on outages.
89
+
90
+ Explicit catalogs and malformed data fail clearly. A failed artifact
91
+ download never silently starts an expensive source build.
92
+ """
93
+ selected = source or os.environ.get("FUSION_FUNCTION_REFERENCE_CATALOG")
94
+ if selected and "://" not in selected:
95
+ raw = Path(selected).expanduser().read_bytes()
96
+ else:
97
+ try:
98
+ with data.open_url(https_url(selected or CATALOG_URL)) as response:
99
+ raw = response.read(4 * data.CHUNK_SIZE + 1)
100
+ if len(raw) > 4 * data.CHUNK_SIZE:
101
+ raise ValueError("Reference catalog exceeds 4 MiB")
102
+ except (OSError, urllib.error.URLError) as exc:
103
+ if selected:
104
+ raise
105
+ data.LOG.warning("Reference catalog unavailable (%s); using bundled catalog", exc)
106
+ raw = CATALOG_PATH.read_bytes()
107
+ catalog = json.loads(raw)
108
+ if (
109
+ not isinstance(catalog, dict)
110
+ or catalog.get("catalog_version") != 1
111
+ or not isinstance(catalog.get("references"), list)
112
+ ):
113
+ raise ValueError(
114
+ "Invalid reference catalog; expected catalog_version=1 and references list"
115
+ )
116
+ entries = []
117
+ for entry in catalog["references"]:
118
+ if not isinstance(entry, dict):
119
+ raise ValueError("Reference catalog entries must be objects")
120
+ # Export manifests can exist before their Zenodo record is published.
121
+ if entry.get("url") is None:
122
+ continue
123
+ https_url(entry["url"])
124
+ for key in ("release", "build_revision", "archive_bytes", "database_bytes"):
125
+ if type(entry.get(key)) is not int or entry[key] <= 0:
126
+ raise ValueError(f"Reference catalog {key} must be a positive integer")
127
+ for key in ("archive_sha256", "database_sha256"):
128
+ if not isinstance(entry.get(key), str) or not re.fullmatch(r"[0-9a-f]{64}", entry[key]):
129
+ raise ValueError(f"Invalid reference catalog {key}")
130
+ if entry.get("compression") != "gzip" or not isinstance(entry.get("metadata"), dict):
131
+ raise ValueError("Reference catalog requires gzip compression and metadata")
132
+ entries.append(entry)
133
+ return entries
134
+
135
+
136
+ def select_reference(
137
+ entries: Sequence[Mapping[str, Any]], release: int
138
+ ) -> Mapping[str, Any] | None:
139
+ """Choose the newest compatible build of the exact requested release."""
140
+ candidates = [
141
+ entry
142
+ for entry in entries
143
+ if entry["release"] == release
144
+ and entry.get("species") == data.SPECIES
145
+ and entry.get("assembly") == data.ASSEMBLY
146
+ and all(entry["metadata"].get(key) == value for key, value in COMPATIBILITY.items())
147
+ ]
148
+ if not candidates:
149
+ return None
150
+ revision = max(entry["build_revision"] for entry in candidates)
151
+ newest = [entry for entry in candidates if entry["build_revision"] == revision]
152
+ if len(newest) != 1:
153
+ raise ValueError(f"Ambiguous reference for release {release}, build {revision}")
154
+ return newest[0]
155
+
156
+
157
+ def open_download(request: urllib.request.Request) -> HTTPResponse:
158
+ """Open an anonymous download, including Range headers for interrupted files."""
159
+ return urllib.request.urlopen(request, timeout=60) # type: ignore[no-any-return]
160
+
161
+
162
+ def download_archive(entry: Mapping[str, Any], cache: Path) -> Path:
163
+ """Lock the shared archive cache, so installs to different outputs cannot race."""
164
+ directory = cache / "prebuilt" / entry["archive_sha256"]
165
+ directory.mkdir(parents=True, exist_ok=True)
166
+ with data.build_lock(directory / ".download.lock"):
167
+ return _download_archive(entry, directory)
168
+
169
+
170
+ def _download_archive(entry: Mapping[str, Any], directory: Path) -> Path:
171
+ """Resume partial bytes; restart when Range is ignored and verify checksums.
172
+
173
+ Interrupted transfers retain partial bytes. Complete but invalid files are
174
+ deleted so the next attempt can start again.
175
+ """
176
+ archive = directory / "ensembl.sqlite.gz"
177
+ partial = archive.with_name(archive.name + ".part")
178
+ size = entry["archive_bytes"]
179
+ data.LOG.info("Prebuilt archive: %s; resumable partial: %s", archive, partial)
180
+ if not archive.exists():
181
+ offset = partial.stat().st_size if partial.exists() else 0
182
+ if offset > size:
183
+ partial.unlink()
184
+ offset = 0
185
+ if offset < size:
186
+ request = urllib.request.Request(
187
+ entry["url"], headers={"Range": f"bytes={offset}-"} if offset else {}
188
+ )
189
+ with open_download(request) as response:
190
+ if response.status == 206:
191
+ if response.headers.get("Content-Range") != f"bytes {offset}-{size - 1}/{size}":
192
+ raise ValueError("Unexpected Content-Range for prebuilt reference")
193
+ elif response.status == 200:
194
+ offset = 0
195
+ else:
196
+ raise OSError(f"Unexpected reference download status: {response.status}")
197
+ with (
198
+ partial.open("ab" if offset else "wb") as stream,
199
+ data.progress(
200
+ total=size, desc="Download prebuilt reference", unit="B", unit_scale=True
201
+ ) as bar,
202
+ ):
203
+ bar.update(offset)
204
+ received = offset
205
+ while block := response.read(4 * data.CHUNK_SIZE):
206
+ if received + len(block) > size:
207
+ raise ValueError("Reference archive exceeds the catalog size")
208
+ stream.write(block)
209
+ received += len(block)
210
+ bar.update(len(block))
211
+ if received != size:
212
+ raise OSError(
213
+ f"Incomplete reference download: {received}/{size} bytes; rerun to resume"
214
+ )
215
+ if (
216
+ partial.stat().st_size != size
217
+ or data.sha256_file(partial, show_progress=True) != entry["archive_sha256"]
218
+ ):
219
+ partial.unlink()
220
+ raise ValueError("Prebuilt archive checksum/size mismatch; rerun to download again")
221
+ partial.replace(archive)
222
+ elif (
223
+ archive.stat().st_size != size
224
+ or data.sha256_file(archive, show_progress=True) != entry["archive_sha256"]
225
+ ):
226
+ archive.unlink()
227
+ raise ValueError("Cached prebuilt archive checksum/size mismatch; rerun to download again")
228
+ return archive
229
+
230
+
231
+ def validate_database(path: Path, release: int, metadata: Mapping[str, str]) -> None:
232
+ """Check metadata and runtime tables without another expensive integrity scan."""
233
+ from .reference import ReferenceDatabase
234
+
235
+ with ReferenceDatabase(path, release=release) as reader:
236
+ for key, value in metadata.items():
237
+ if reader.metadata.get(key) != value:
238
+ raise ValueError(f"Prebuilt database metadata mismatch: {key}")
239
+ tables = {
240
+ row[0] for row in reader.db.execute("SELECT name FROM sqlite_master WHERE type='table'")
241
+ }
242
+ if set(RUNTIME_TABLES) - tables:
243
+ raise ValueError("Prebuilt database is missing runtime tables")
244
+ if (
245
+ reader.db.execute(
246
+ "SELECT 1 FROM ff_transcripts WHERE status='ready' LIMIT 1"
247
+ ).fetchone()
248
+ is None
249
+ ):
250
+ raise ValueError("Prebuilt database contains no usable coding transcripts")
251
+ if (
252
+ reader.db.execute("SELECT 1 FROM sequence_chunks WHERE kind='dna' LIMIT 1").fetchone()
253
+ is None
254
+ ):
255
+ raise ValueError("Prebuilt database contains no genome sequence chunks")
256
+
257
+
258
+ def install_reference(
259
+ *,
260
+ release: int | None,
261
+ cache_dir: Path,
262
+ output: Path | None = None,
263
+ force: bool = False,
264
+ catalog: str | None = None,
265
+ ) -> Path | None:
266
+ """Install a published reference, or return None when a source build is needed."""
267
+ entries = read_catalog(catalog)
268
+ if not entries:
269
+ data.LOG.info("No published prebuilt references in the catalog; building from Ensembl FTP")
270
+ return None
271
+ if release is None:
272
+ # Latest still means latest Ensembl, not latest uploaded database.
273
+ release, _, _ = data.resolve_core(data.BASE_URL, None)
274
+ entry = select_reference(entries, release)
275
+ if entry is None:
276
+ data.LOG.info("No compatible prebuilt for Ensembl %s; building from Ensembl FTP", release)
277
+ return None
278
+ path = (
279
+ (
280
+ output
281
+ or cache_dir / data.SPECIES / data.ASSEMBLY / f"release-{release}" / "ensembl.sqlite"
282
+ )
283
+ .expanduser()
284
+ .resolve()
285
+ )
286
+ path.parent.mkdir(parents=True, exist_ok=True)
287
+ data.LOG.info(
288
+ "Selected prebuilt: Ensembl %s, build %s, %s",
289
+ release,
290
+ entry["build_revision"],
291
+ entry["url"],
292
+ )
293
+ data.LOG.info(
294
+ "Final reference: %s; expected installed size: %s bytes", path, entry["database_bytes"]
295
+ )
296
+ staging = path.with_name(path.name + ".prebuilt.part")
297
+ data.LOG.info("Expanded partial database: %s", staging)
298
+ steps = data.BuildSteps(4)
299
+ with data.build_lock(path.with_name(path.name + ".prepare.lock")):
300
+ if path.exists() and not force:
301
+ validate_database(path, release, COMPATIBILITY)
302
+ data.LOG.info("Using existing reference: %s; --force replaces it", path)
303
+ return path
304
+ try:
305
+ with steps.step("Download and verify prebuilt archive"):
306
+ data.LOG.info("A large reference download can take over 10 minutes")
307
+ archive = download_archive(entry, cache_dir)
308
+ with steps.step("Expand reference database"):
309
+ data.LOG.info("Expanding a large reference can take over 10 minutes")
310
+ digest = hashlib.sha256()
311
+ size = 0
312
+ with (
313
+ gzip.open(archive, "rb") as source,
314
+ staging.open("wb") as target,
315
+ data.progress(
316
+ total=entry["database_bytes"],
317
+ desc="Expand prebuilt reference",
318
+ unit="B",
319
+ unit_scale=True,
320
+ ) as bar,
321
+ ):
322
+ while block := source.read(4 * data.CHUNK_SIZE):
323
+ size += len(block)
324
+ if size > entry["database_bytes"]:
325
+ raise ValueError("Expanded reference exceeds the catalog size")
326
+ digest.update(block)
327
+ target.write(block)
328
+ bar.update(len(block))
329
+ if (
330
+ size != entry["database_bytes"]
331
+ or digest.hexdigest() != entry["database_sha256"]
332
+ ):
333
+ raise ValueError("Expanded reference checksum/size mismatch")
334
+ with steps.step("Validate reference format and release"):
335
+ validate_database(staging, release, entry["metadata"])
336
+ with steps.step("Install completed reference"):
337
+ os.replace(staging, path)
338
+ finally:
339
+ staging.unlink(missing_ok=True)
340
+ steps.complete()
341
+ return path
342
+
343
+
344
+ def export_reference(
345
+ database: Path, directory: Path, *, build_revision: int = 1, zenodo_record: int | None = None
346
+ ) -> Path:
347
+ """Export a read-only snapshot of runtime data; preserve the maintainer's DB.
348
+
349
+ Bulk INSERT SELECT copies compressed payloads without Python decoding. The
350
+ attached source is read-only and one transaction holds a consistent snapshot.
351
+ """
352
+ if build_revision <= 0 or (zenodo_record is not None and zenodo_record <= 0):
353
+ raise ValueError("Build revision and Zenodo record must be positive")
354
+ database = database.expanduser().resolve()
355
+ directory = directory.expanduser().resolve()
356
+ directory.mkdir(parents=True, exist_ok=True)
357
+ with data.ReferenceReader(database) as reader:
358
+ metadata = dict(reader.metadata)
359
+ release = int(metadata["release"])
360
+ validate_database(database, release, COMPATIBILITY)
361
+ stem = f"{data.SPECIES}.{data.ASSEMBLY}.ensembl-{release}.ff-v1.build-{build_revision}"
362
+ archive, manifest = directory / (stem + ".sqlite.gz"), directory / (stem + ".json")
363
+ with data.build_lock(directory / (stem + ".export.lock")):
364
+ if archive.exists() or manifest.exists():
365
+ raise FileExistsError("Export already exists; use a new build revision or directory")
366
+ with tempfile.TemporaryDirectory(prefix=stem + ".", dir=directory) as workspace:
367
+ compact = Path(workspace) / "ensembl.sqlite"
368
+ data.LOG.info("Export destination: %s; temporary database: %s", archive, compact)
369
+ steps = data.BuildSteps(4)
370
+ with closing(sqlite3.connect(compact, uri=True)) as target:
371
+ target.execute("ATTACH DATABASE ? AS original", (database.as_uri() + "?mode=ro",))
372
+ target.execute("PRAGMA secure_delete=OFF")
373
+ target.execute("PRAGMA user_version=1")
374
+ target.execute("BEGIN")
375
+ with steps.step("Copy runtime tables into a compact database"):
376
+ data.LOG.info("Copying a large reference can take over 10 minutes")
377
+ for table in (*RUNTIME_TABLES, "source_files"):
378
+ schema = target.execute(
379
+ "SELECT sql FROM original.sqlite_master WHERE type='table' AND name=?",
380
+ (table,),
381
+ ).fetchone()
382
+ if schema is None:
383
+ if table == "source_files":
384
+ continue
385
+ raise ValueError(f"Missing runtime table: {table}")
386
+ target.execute(schema[0])
387
+ data.sqlite_phase(
388
+ target,
389
+ f'INSERT INTO "{table}" SELECT * FROM original."{table}"',
390
+ f"Export {table}",
391
+ )
392
+ for (sql,) in target.execute(
393
+ "SELECT sql FROM original.sqlite_master WHERE type='index' AND tbl_name=? AND sql IS NOT NULL",
394
+ (table,),
395
+ ).fetchall():
396
+ data.sqlite_phase(target, sql, f"Index exported {table}")
397
+ # Keep original provenance, plus notices describing our transformation.
398
+ metadata = dict(target.execute("SELECT * FROM build_metadata"))
399
+ target.executemany(
400
+ "INSERT OR REPLACE INTO build_metadata VALUES (?,?)",
401
+ [
402
+ ("reference_kind", "runtime"),
403
+ ("reference_build_revision", str(build_revision)),
404
+ ("exported_utc", datetime.now(timezone.utc).isoformat()),
405
+ ("data_source_notices", json.dumps(DATA_NOTICES, sort_keys=True)),
406
+ (
407
+ "data_modifications",
408
+ "Runtime subset; derived transcript, feature and splice mappings; lossless compression.",
409
+ ),
410
+ ],
411
+ )
412
+ target.commit()
413
+ with steps.step("Validate exported SQLite integrity"):
414
+ if data.sqlite_phase(target, "PRAGMA integrity_check", "Check database") != [
415
+ ("ok",)
416
+ ]:
417
+ raise ValueError("Exported SQLite integrity check failed")
418
+ validate_database(compact, release, COMPATIBILITY)
419
+ with steps.step("Compress exported reference"):
420
+ data.LOG.info("Compressing a large reference can take over 10 minutes")
421
+ digest = hashlib.sha256()
422
+ temporary_archive = Path(workspace) / archive.name
423
+ with (
424
+ compact.open("rb") as source,
425
+ temporary_archive.open("wb") as output,
426
+ gzip.GzipFile(
427
+ filename="", fileobj=output, mode="wb", compresslevel=1, mtime=0
428
+ ) as compressed,
429
+ data.progress(
430
+ total=compact.stat().st_size,
431
+ desc="Compress reference",
432
+ unit="B",
433
+ unit_scale=True,
434
+ ) as bar,
435
+ ):
436
+ while block := source.read(4 * data.CHUNK_SIZE):
437
+ digest.update(block)
438
+ compressed.write(block)
439
+ bar.update(len(block))
440
+ with steps.step("Write manifest and publish export"):
441
+ entry = {
442
+ "species": data.SPECIES,
443
+ "assembly": data.ASSEMBLY,
444
+ "release": release,
445
+ "build_revision": build_revision,
446
+ "url": f"https://zenodo.org/records/{zenodo_record}/files/{archive.name}?download=1"
447
+ if zenodo_record
448
+ else None,
449
+ "filename": archive.name,
450
+ "compression": "gzip",
451
+ "archive_bytes": temporary_archive.stat().st_size,
452
+ "archive_sha256": data.sha256_file(temporary_archive, show_progress=True),
453
+ "database_bytes": compact.stat().st_size,
454
+ "database_sha256": digest.hexdigest(),
455
+ "metadata": COMPATIBILITY,
456
+ "data_source_notices": DATA_NOTICES,
457
+ "source_metadata": metadata,
458
+ }
459
+ temporary_manifest = Path(workspace) / manifest.name
460
+ temporary_manifest.write_text(
461
+ json.dumps({"catalog_version": 1, "references": [entry]}, indent=2) + "\n"
462
+ )
463
+ temporary_archive.replace(archive)
464
+ temporary_manifest.replace(manifest)
465
+ steps.complete()
466
+ if zenodo_record is None:
467
+ data.LOG.info(
468
+ "Manifest URL is unset; add the published Zenodo file URL before registering it"
469
+ )
470
+ data.LOG.info("Export archive: %s; manifest: %s", archive, manifest)
471
+ return manifest
472
+
473
+
474
+ def main(argv: Sequence[str] | None = None) -> int:
475
+ """Maintainer CLI to export a tested reference for publication."""
476
+ from .reference import ReferenceDatabase
477
+
478
+ parser = argparse.ArgumentParser(prog="fusion-function export-reference")
479
+ parser.add_argument(
480
+ "database",
481
+ type=Path,
482
+ nargs="?",
483
+ help="Existing database; default: FUSION_FUNCTION_DB or newest installed local release",
484
+ )
485
+ parser.add_argument(
486
+ "--release", type=int, help="Installed Ensembl release; default: newest installed release"
487
+ )
488
+ parser.add_argument(
489
+ "--output", required=True, type=Path, help="Directory for compressed DB and manifest"
490
+ )
491
+ parser.add_argument("--build-revision", type=int, default=1)
492
+ parser.add_argument(
493
+ "--zenodo-record", type=int, help="Reserved/published Zenodo record ID; not a username"
494
+ )
495
+ args = parser.parse_args(argv)
496
+ if args.release is not None and args.release <= 0:
497
+ parser.error("--release must be positive")
498
+ logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
499
+ try:
500
+ # Reuse analysis selection, including cache overrides and release checks.
501
+ # Selection reads local files only; export never builds or downloads a DB.
502
+ with ReferenceDatabase(args.database, release=args.release) as reference:
503
+ database = reference.path
504
+ data.LOG.info("Export source database: %s", database)
505
+ manifest = export_reference(
506
+ database,
507
+ args.output,
508
+ build_revision=args.build_revision,
509
+ zenodo_record=args.zenodo_record,
510
+ )
511
+ except (OSError, ValueError, sqlite3.Error) as exc:
512
+ data.LOG.error("Export failed: %s", exc)
513
+ return 1
514
+ except KeyboardInterrupt:
515
+ data.LOG.error("Export interrupted; source database unchanged")
516
+ return 130
517
+ print(manifest)
518
+ return 0
@@ -0,0 +1,112 @@
1
+ """Reference configuration and read-only access to prepared human databases."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import os
6
+ import re
7
+ import threading
8
+ from pathlib import Path
9
+
10
+ from .data import ASSEMBLY, SPECIES, ReferenceReader, default_cache_dir
11
+
12
+
13
+ def resolve_database(
14
+ database: str | Path | None = None,
15
+ *,
16
+ release: int | None = None,
17
+ cache_dir: str | Path | None = None,
18
+ ) -> Path:
19
+ """Choose explicit path, FUSION_FUNCTION_DB, or newest prepared local release."""
20
+ if release is not None and (not isinstance(release, int) or release <= 0):
21
+ raise ValueError("release must be a positive Ensembl release number")
22
+ explicit = database or os.environ.get("FUSION_FUNCTION_DB")
23
+ if explicit:
24
+ path = Path(explicit).expanduser().resolve()
25
+ else:
26
+ root = Path(cache_dir or default_cache_dir()).expanduser() / SPECIES / ASSEMBLY
27
+ if release is not None:
28
+ path = root / f"release-{release}" / "ensembl.sqlite"
29
+ else:
30
+ candidates = []
31
+ if root.is_dir():
32
+ for directory in root.iterdir():
33
+ match = re.fullmatch(r"release-(\d+)", directory.name)
34
+ if match and (directory / "ensembl.sqlite").is_file():
35
+ candidates.append((int(match[1]), directory / "ensembl.sqlite"))
36
+ if not candidates:
37
+ raise FileNotFoundError(
38
+ f"No prepared human reference in {root}. Run 'fusion-function prepare-data' "
39
+ "once, or set FUSION_FUNCTION_DB to an existing preprocessed database."
40
+ )
41
+ path = max(candidates)[1]
42
+ path = path.resolve()
43
+ if not path.is_file():
44
+ raise FileNotFoundError(
45
+ f"Reference database not found: {path}. Run 'fusion-function prepare-data'."
46
+ )
47
+ return path
48
+
49
+
50
+ class ReferenceDatabase(ReferenceReader):
51
+ """Reusable per-thread reference reader; close explicitly or use a context manager."""
52
+
53
+ def __init__(
54
+ self,
55
+ database: str | Path | None = None,
56
+ *,
57
+ release: int | None = None,
58
+ cache_dir: str | Path | None = None,
59
+ cached_chunks: int = 64,
60
+ ) -> None:
61
+ self.path = resolve_database(database, release=release, cache_dir=cache_dir)
62
+ super().__init__(self.path, cached_chunks=cached_chunks)
63
+ try:
64
+ if self.metadata.get("format_version") != "1":
65
+ raise ValueError(
66
+ "Unsupported reference format; rebuild with 'fusion-function prepare-data'"
67
+ )
68
+ if self.metadata.get("species") != SPECIES:
69
+ raise ValueError("Only human reference databases are supported")
70
+ if self.metadata.get("assembly") != ASSEMBLY:
71
+ raise ValueError("Only GRCh38 reference databases are supported")
72
+ if release is not None and self.metadata.get("release") != str(release):
73
+ raise ValueError(
74
+ f"Requested release {release}, but database has release {self.metadata.get('release')}"
75
+ )
76
+ except BaseException:
77
+ self.close()
78
+ raise
79
+
80
+
81
+ _local = threading.local()
82
+
83
+
84
+ def close_default_reference() -> None:
85
+ """Release this thread's automatically cached connection and genome chunks."""
86
+ reader = getattr(_local, "reader", None)
87
+ if reader is not None:
88
+ reader.close()
89
+ del _local.reader
90
+ del _local.signature
91
+
92
+
93
+ def get_reference(
94
+ *,
95
+ reference: ReferenceReader | None = None,
96
+ database: str | Path | None = None,
97
+ release: int | None = None,
98
+ ) -> ReferenceReader:
99
+ if reference is not None:
100
+ if database is not None or release is not None:
101
+ raise ValueError("Use reference, or database/release; these cannot be combined")
102
+ return reference
103
+ path = resolve_database(database, release=release)
104
+ stat = path.stat()
105
+ signature = (os.getpid(), str(path), stat.st_ino, stat.st_size, stat.st_mtime_ns)
106
+ if getattr(_local, "signature", None) != signature:
107
+ close_default_reference()
108
+ _local.reader = ReferenceDatabase(path, release=release)
109
+ _local.signature = signature
110
+ elif release is not None and _local.reader.metadata.get("release") != str(release):
111
+ raise ValueError("Requested release does not match the selected database")
112
+ return _local.reader