fusion-function 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,2683 @@
1
+ #!/usr/bin/env python3
2
+ """Build and read the human GRCh38 reference used by fusion-function.
3
+
4
+ Commands (Python 3.11+, tqdm is the only external dependency):
5
+ fusion-function prepare-data # latest Ensembl release
6
+ fusion-function prepare-data --release 116 # pinned release
7
+ fusion-function prepare-data --preprocess-only DB # regenerate derived data
8
+
9
+ Default output: <cache>/homo_sapiens/GRCh38/release-<release>/ensembl.sqlite.
10
+ Set FUSION_FUNCTION_CACHEDIR or --cache-dir to change the cache root.
11
+
12
+ Storage: ensembl_* tables preserve the FTP dump columns. DNA and peptides use
13
+ independently compressed 1 MiB chunks. ff_transcripts stores prepared exon/CDS
14
+ coordinates, splice windows and protein features; ff_interpro stores names and
15
+ types. Transcript models use compressed JSON; error messages remain plain JSON.
16
+ ff_uniprot_features stores sequence-verified reviewed human sites,
17
+ domains and motifs. Coordinates are 1-based and inclusive; CDS blocks follow transcript
18
+ orientation, including on the negative strand. Pre-mRNA is reconstructed when
19
+ requested, rather than stored for every transcript.
20
+
21
+ Preparation downloads FTP files over HTTPS and verifies cached files by SHA-256.
22
+ InterPro metadata is fetched by default, with archived entry lists supplementing
23
+ missing accessions. --interpro-entries FILE supplies local metadata instead.
24
+ Archive checksums and source releases are recorded in build_metadata.
25
+ Human PANTHER subfamily classifications are fetched for the version recorded
26
+ by Ensembl, or supplied with --panther-classifications FILE. Protein cross-
27
+ references enrich existing family hits; no domain coordinates are invented.
28
+ Reviewed human UniProt XML is fetched during preparation only, or supplied with
29
+ --uniprot-features FILE. Features require exact protein sequence identity and
30
+ carry UniProt accessions, isoforms and evidence codes.
31
+
32
+ New databases resume from <output>.building and are published only after
33
+ integrity checking; --force permits replacement. Repeating the same build skips
34
+ completed imports and reuses committed FASTA records from unchanged inputs.
35
+ Reprocessing updates derived tables in one transaction and rolls
36
+ back on failure. Numbered steps and terminal progress bars show build progress.
37
+ Runtime readers are local and read-only; no REST API, MySQL server or pysam is
38
+ required. See README.md and docs/reference.md for configuration details.
39
+ """
40
+
41
+ from __future__ import annotations
42
+
43
+ import argparse
44
+ import gzip
45
+ import hashlib
46
+ import io
47
+ import json
48
+ import logging
49
+ import os
50
+ import re
51
+ import sqlite3
52
+ import sys
53
+ import tempfile
54
+ import time
55
+ import urllib.error
56
+ import urllib.parse
57
+ import urllib.request
58
+ import zlib
59
+ from bisect import bisect_left
60
+ from collections import OrderedDict
61
+ from collections.abc import Callable, Iterable, Iterator, Mapping, Sequence
62
+ from contextlib import AbstractContextManager, ExitStack, closing, contextmanager
63
+ from contextvars import ContextVar
64
+ from datetime import datetime, timezone
65
+ from html.parser import HTMLParser
66
+ from itertools import groupby
67
+ from pathlib import Path
68
+ from types import TracebackType
69
+ from typing import TYPE_CHECKING, Any, Literal, Self, TextIO, TypedDict, cast, overload
70
+ from xml.etree.ElementTree import ParseError
71
+
72
+ from tqdm import tqdm
73
+ from tqdm.contrib.logging import logging_redirect_tqdm
74
+
75
+ if TYPE_CHECKING:
76
+ from http.client import HTTPResponse
77
+
78
+ from .ensembl import (
79
+ CDSBlock,
80
+ EnsemblError,
81
+ GenomicSegment,
82
+ ProteinFeatureAnnotation,
83
+ ProteinFeatureResponse,
84
+ TranscriptProteinFeatureResult,
85
+ )
86
+
87
+
88
+ # Raw SQL rows have different columns per source table; derived structures use
89
+ # the same typed records as the annotation API.
90
+ DatabaseRow = dict[str, Any]
91
+ AnalysisMetadata = dict[int, tuple[str | None, str | None]]
92
+ InterProMetadata = dict[str, tuple[str | None, str | None]]
93
+ ChunkCache = OrderedDict[tuple[str, str, int], bytes]
94
+ STRUCTURE_SOURCES = frozenset({"sifts", "alphafold"})
95
+ TRANSCRIPT_PAYLOAD_CODEC = "zlib-json-v1"
96
+
97
+
98
+ class SourceRecord(TypedDict):
99
+ url: str
100
+ sha256: str
101
+ bytes: int
102
+
103
+
104
+ class TranscriptErrorSummary(TypedDict):
105
+ count: int
106
+ example_transcripts: list[str]
107
+
108
+
109
+ class CheckpointRecord(TypedDict):
110
+ source: dict[str, Any]
111
+ fingerprint: str
112
+ count: int
113
+ complete: bool
114
+
115
+
116
+ BASE_URL = "https://ftp.ensembl.org/pub/"
117
+ INTERPRO_ENTRIES_URL = "https://ftp.ebi.ac.uk/pub/databases/interpro/current_release/entry.list"
118
+ INTERPRO_RELEASES_URL = "https://ftp.ebi.ac.uk/pub/databases/interpro/releases/"
119
+ PANTHER_CLASSIFICATIONS_URL = "https://data.pantherdb.org/ftp/sequence_classifications/"
120
+ # These inputs are fixed: omitting a table or sequence type can break preprocessing.
121
+ TABLES = (
122
+ "meta",
123
+ "coord_system",
124
+ "seq_region",
125
+ "gene",
126
+ "transcript",
127
+ "exon",
128
+ "exon_transcript",
129
+ "translation",
130
+ "analysis",
131
+ "protein_feature",
132
+ "interpro",
133
+ "xref",
134
+ "external_db",
135
+ "object_xref",
136
+ )
137
+ PREPROCESS_TABLES = (
138
+ "transcript",
139
+ "translation",
140
+ "exon",
141
+ "exon_transcript",
142
+ "seq_region",
143
+ "coord_system",
144
+ "protein_feature",
145
+ "analysis",
146
+ "interpro",
147
+ "xref",
148
+ )
149
+ SEQUENCE_TYPES = ("dna", "pep")
150
+ SPECIES = "homo_sapiens"
151
+ ASSEMBLY = "GRCh38"
152
+ CHUNK_SIZE = 1024 * 1024
153
+ LOG = logging.getLogger("ensembl_to_sqlite")
154
+ STEP_PREFIX = ContextVar("fusion_function_prepare_step", default="")
155
+
156
+
157
+ # Progress reporting and cache configuration.
158
+
159
+
160
+ class BuildSteps:
161
+ """Label the fixed build stages; their durations differ, so no total ETA."""
162
+
163
+ def __init__(self, total: int) -> None:
164
+ """Track the expected stage count and total elapsed build time."""
165
+ self.total = total
166
+ self.current = 0
167
+ self.started = time.monotonic()
168
+
169
+ @contextmanager
170
+ def step(self, description: str) -> Iterator[None]:
171
+ """Label a stage and its nested bars; restore the label even on failure."""
172
+ self.current += 1
173
+ if self.current > self.total:
174
+ raise ValueError("Preparation step count exceeds its declared total")
175
+ LOG.info("[Step %s/%s] %s", self.current, self.total, description)
176
+ token = STEP_PREFIX.set(f"[{self.current}/{self.total}] ")
177
+ try:
178
+ yield
179
+ finally:
180
+ STEP_PREFIX.reset(token)
181
+
182
+ def complete(self) -> None:
183
+ """Check that every declared stage ran, then log the total duration."""
184
+ if self.current != self.total:
185
+ raise ValueError("Preparation completed fewer steps than expected")
186
+ LOG.info(
187
+ "Completed all %s steps in %s",
188
+ self.total,
189
+ tqdm.format_interval(time.monotonic() - self.started),
190
+ )
191
+
192
+
193
+ def progress(
194
+ iterable: Iterable[Any] | None = None,
195
+ *,
196
+ total: int | None = None,
197
+ desc: str = "",
198
+ unit: str = "it",
199
+ unit_scale: bool = False,
200
+ unit_divisor: int = 1000,
201
+ disable: bool | None = None,
202
+ file: TextIO | None = None,
203
+ ) -> tqdm:
204
+ """Use compact, rate-limited bars; suppress terminal redraws in log files."""
205
+ long_process_notice(desc)
206
+ return tqdm(
207
+ iterable,
208
+ total=total,
209
+ desc=STEP_PREFIX.get() + desc,
210
+ unit=unit,
211
+ unit_scale=unit_scale,
212
+ unit_divisor=unit_divisor,
213
+ disable=disable,
214
+ file=file,
215
+ dynamic_ncols=True,
216
+ mininterval=0.5,
217
+ )
218
+
219
+
220
+ def file_label(path: Path) -> str:
221
+ """Short labels keep progress output on one terminal line."""
222
+ if path.name.endswith(".sql.gz"):
223
+ return "core schema"
224
+ if path.name == "stream":
225
+ return "UniProt features"
226
+ if path.name == "entry.list":
227
+ return "InterPro"
228
+ for kind in SEQUENCE_TYPES:
229
+ if f".{kind}." in path.name:
230
+ return f"{kind} FASTA"
231
+ return path.name.removesuffix(".txt.gz")[:24]
232
+
233
+
234
+ @overload
235
+ def input_progress(
236
+ path: Path, description: str, *, compressed: bool = True, text: Literal[True]
237
+ ) -> AbstractContextManager[tuple[io.TextIOWrapper, tqdm, Callable[[], None]]]: ...
238
+
239
+
240
+ @overload
241
+ def input_progress(
242
+ path: Path, description: str, *, compressed: bool = True, text: Literal[False] = False
243
+ ) -> AbstractContextManager[tuple[gzip.GzipFile | io.BufferedReader, tqdm, Callable[[], None]]]: ...
244
+
245
+
246
+ @overload
247
+ def input_progress(
248
+ path: Path, description: str, *, compressed: bool = True, text: bool
249
+ ) -> AbstractContextManager[
250
+ tuple[io.TextIOWrapper | gzip.GzipFile | io.BufferedReader, tqdm, Callable[[], None]]
251
+ ]: ...
252
+
253
+
254
+ @contextmanager
255
+ def input_progress(
256
+ path: Path, description: str, *, compressed: bool = True, text: bool = False
257
+ ) -> Iterator[tuple[Any, tqdm, Callable[[], None]]]:
258
+ """Track source bytes during parsing, with no pre-count or gzip size guess."""
259
+ with ExitStack() as stack:
260
+ raw = stack.enter_context(path.open("rb"))
261
+ source = stack.enter_context(gzip.GzipFile(fileobj=raw)) if compressed else raw
262
+ stream = (
263
+ stack.enter_context(io.TextIOWrapper(source, encoding="utf-8", newline=""))
264
+ if text
265
+ else source
266
+ )
267
+ bar = stack.enter_context(
268
+ progress(
269
+ total=path.stat().st_size,
270
+ desc=description,
271
+ unit="B",
272
+ unit_scale=True,
273
+ unit_divisor=1024,
274
+ )
275
+ )
276
+ position = 0
277
+
278
+ def update() -> None:
279
+ """Advance by source bytes read, including gzip buffering."""
280
+ nonlocal position
281
+ consumed = raw.tell()
282
+ bar.update(consumed - position)
283
+ position = consumed
284
+
285
+ yield stream, bar, update
286
+ update()
287
+
288
+
289
+ def long_process_notice(description: str) -> None:
290
+ """Print one advance notice for work that can exceed ten minutes.
291
+
292
+ These are possibilities for large human references, not completion estimates.
293
+ Emit notices only when work starts, so cached or checkpointed stages stay quiet.
294
+ """
295
+ if description == "Drop previous ff_transcripts":
296
+ note = "Removing old transcript models can take an hour or longer on some systems."
297
+ elif description == "Preprocess transcripts":
298
+ note = "Preparing all human transcripts can take an hour or longer."
299
+ elif description == "Check database":
300
+ note = "The full database integrity check can take over 10 minutes."
301
+ elif (
302
+ description.startswith(("Download ", "Import ", "Index "))
303
+ and description not in {"Import InterPro", "Import PANTHER", "Index analysis", "Index meta"}
304
+ ) or description in {
305
+ "Analyze database",
306
+ "Read UniProt protein cross-references",
307
+ "Read genome and protein sequence lengths",
308
+ "Prepare ordered exon stream",
309
+ "Prepare ordered protein feature stream",
310
+ "Prepare transcript stream",
311
+ "Commit preprocessed reference",
312
+ "Roll back preprocessing transaction",
313
+ }:
314
+ note = (
315
+ "This operation can take over 10 minutes, depending on reference size and system load."
316
+ )
317
+ else:
318
+ return
319
+ LOG.info("%s%s: %s", STEP_PREFIX.get(), description, note)
320
+
321
+
322
+ @contextmanager
323
+ def sqlite_activity(description: str) -> Iterator[None]:
324
+ """Log phase boundaries without background output or periodic redraws."""
325
+ label = STEP_PREFIX.get() + description
326
+ LOG.info("%s", label)
327
+ long_process_notice(description)
328
+ outcome = "failed"
329
+ try:
330
+ yield
331
+ outcome = "done"
332
+ except KeyboardInterrupt:
333
+ outcome = "interrupted"
334
+ raise
335
+ finally:
336
+ LOG.info("%s: %s", label, outcome)
337
+
338
+
339
+ def sqlite_phase(db: sqlite3.Connection, sql: str, description: str) -> list[tuple[Any, ...]]:
340
+ """Run one SQLite statement with start/completion logs and signal handling."""
341
+ with sqlite_activity(description):
342
+
343
+ def checkpoint() -> int:
344
+ """Keep Python signal handling responsive during long SQLite statements."""
345
+ return 0
346
+
347
+ db.set_progress_handler(checkpoint, 100000)
348
+ try:
349
+ return db.execute(sql).fetchall()
350
+ finally:
351
+ db.set_progress_handler(None, 0)
352
+
353
+
354
+ def default_cache_dir() -> Path:
355
+ """Use the package override, then the platform's conventional cache root."""
356
+ if value := os.environ.get("FUSION_FUNCTION_CACHEDIR"):
357
+ return Path(value).expanduser()
358
+ if sys.platform == "win32":
359
+ root = Path(os.environ.get("LOCALAPPDATA", Path.home() / "AppData/Local"))
360
+ elif sys.platform == "darwin":
361
+ root = Path.home() / "Library/Caches"
362
+ else:
363
+ root = Path(os.environ.get("XDG_CACHE_HOME", Path.home() / ".cache"))
364
+ return root / "fusion_function"
365
+
366
+
367
+ # Source discovery, verified downloads and MySQL dump imports.
368
+
369
+
370
+ def open_url(url: str) -> HTTPResponse:
371
+ """Open a source URL with an identifying user agent and a bounded timeout."""
372
+ request = urllib.request.Request(url, headers={"User-Agent": "ensembl-to-sqlite/1"})
373
+ return urllib.request.urlopen(request, timeout=60)
374
+
375
+
376
+ class Links(HTMLParser):
377
+ """Collect filenames from the HTML directory listings used by FTP mirrors."""
378
+
379
+ def __init__(self) -> None:
380
+ """Start with no names; repeated links are collapsed into a set."""
381
+ super().__init__()
382
+ self.names: set[str] = set()
383
+
384
+ def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
385
+ """Keep link basenames, excluding navigation to the current/parent path."""
386
+ if tag == "a":
387
+ href = dict(attrs).get("href", "")
388
+ path = urllib.parse.unquote(urllib.parse.urlparse(href).path)
389
+ name = path.rstrip("/").rsplit("/", 1)[-1]
390
+ if name not in {"", ".", ".."}:
391
+ self.names.add(name)
392
+
393
+
394
+ def listing(url: str) -> set[str]:
395
+ """Return directory entry names without depending on HTML table formatting."""
396
+ with open_url(url) as response:
397
+ text = response.read().decode("utf-8")
398
+ parser = Links()
399
+ parser.feed(text)
400
+ return parser.names
401
+
402
+
403
+ def resolve_core(base: str, release: int | None) -> tuple[int, str, str]:
404
+ """Find a published release and its human GRCh38 core directory.
405
+
406
+ Latest means the largest numbered release, not an unversioned FTP alias.
407
+ Choosing the exact assembly suffix avoids mixing GRCh37 and GRCh38 data.
408
+ """
409
+ if release is None:
410
+ releases = [
411
+ int(m[1]) for name in listing(base) if (m := re.fullmatch(r"release-(\d+)", name))
412
+ ]
413
+ if not releases:
414
+ raise ValueError("No numbered Ensembl releases found; supply --release")
415
+ release = max(releases)
416
+ root = f"{base}release-{release}/"
417
+ core_name = f"{SPECIES}_core_{release}_38"
418
+ if core_name not in listing(root + "mysql/"):
419
+ raise ValueError(f"Human GRCh38 core database {core_name} is absent from release {release}")
420
+ return release, root, core_name
421
+
422
+
423
+ def sha256_file(path: Path, *, show_progress: bool = False) -> str:
424
+ """Hash a file in bounded reads, optionally displaying verification progress."""
425
+ digest = hashlib.sha256()
426
+ with (
427
+ path.open("rb") as stream,
428
+ progress(
429
+ total=path.stat().st_size,
430
+ desc="Verify " + file_label(path),
431
+ unit="B",
432
+ unit_scale=True,
433
+ unit_divisor=1024,
434
+ disable=None if show_progress else True,
435
+ ) as bar,
436
+ ):
437
+ while block := stream.read(4 * CHUNK_SIZE):
438
+ digest.update(block)
439
+ bar.update(len(block))
440
+ return digest.hexdigest()
441
+
442
+
443
+ def download(url: str, directory: Path) -> tuple[Path, str]:
444
+ """Reuse verified cached bytes or atomically publish a complete download.
445
+
446
+ The caller supplies a directory that separates releases/source locations.
447
+ Recoverable failures are retried; partial files are removed on every exit.
448
+ """
449
+ directory = directory.expanduser().resolve()
450
+ directory.mkdir(parents=True, exist_ok=True)
451
+ name = urllib.parse.urlparse(url).path.rsplit("/", 1)[-1]
452
+ path = directory / name
453
+ sidecar = path.with_name(path.name + ".sha256")
454
+ if path.is_file() and sidecar.is_file():
455
+ cached_digest = sha256_file(path, show_progress=True)
456
+ if cached_digest == sidecar.read_text().strip():
457
+ LOG.info("Using cached %s", path)
458
+ return path, cached_digest
459
+ LOG.warning("Cached checksum mismatch: downloading %s again", name)
460
+ LOG.info("Downloading %s", url)
461
+ LOG.info(" Cached file: %s; checksum: %s", path, sidecar)
462
+ for attempt in range(3):
463
+ temporary = None
464
+ try:
465
+ digest = hashlib.sha256()
466
+ received = 0
467
+ with open_url(url) as response:
468
+ expected = response.headers.get("Content-Length")
469
+ with tempfile.NamedTemporaryFile(
470
+ dir=directory, prefix=name + ".", suffix=".part", delete=False
471
+ ) as out:
472
+ temporary = Path(out.name)
473
+ LOG.info(" Partial download: %s", temporary)
474
+ with progress(
475
+ total=int(expected) if expected is not None else None,
476
+ desc="Download " + file_label(path),
477
+ unit="B",
478
+ unit_scale=True,
479
+ unit_divisor=1024,
480
+ ) as bar:
481
+ while block := response.read(4 * CHUNK_SIZE):
482
+ out.write(block)
483
+ digest.update(block)
484
+ received += len(block)
485
+ bar.update(len(block))
486
+ if expected is not None and received != int(expected):
487
+ raise OSError(f"Truncated download: expected {expected}, received {received}")
488
+ temporary.replace(path)
489
+ checksum = digest.hexdigest()
490
+ sidecar.write_text(checksum + "\n")
491
+ return path, checksum
492
+ except (OSError, urllib.error.URLError) as exc:
493
+ if isinstance(exc, urllib.error.HTTPError) and exc.code in {400, 403, 404}:
494
+ raise
495
+ if attempt == 2:
496
+ raise
497
+ LOG.warning("Download failed (%s); retrying", exc)
498
+ time.sleep(2**attempt)
499
+ finally:
500
+ if temporary is not None:
501
+ temporary.unlink(missing_ok=True)
502
+ raise AssertionError("Unreachable")
503
+
504
+
505
+ def sql_name(value: str) -> str:
506
+ """Quote a SQLite identifier; bound parameters can only represent values."""
507
+ return '"' + value.replace('"', '""') + '"'
508
+
509
+
510
+ def parse_schema(path: Path) -> dict[str, list[tuple[str, str]]]:
511
+ """Read column order/types from MySQL DDL without executing source SQL."""
512
+ with gzip.open(path, "rt", encoding="utf-8") as stream:
513
+ text = stream.read()
514
+ tables = {}
515
+ for match in re.finditer(
516
+ r"CREATE TABLE(?: IF NOT EXISTS)? `([^`]+)`\s*\((.*?)\n\)", text, re.S
517
+ ):
518
+ columns = []
519
+ for column in re.finditer(r"^\s*`([^`]+)`\s+([A-Za-z]+)", match[2], re.M):
520
+ mysql_type = column[2].lower()
521
+ # Ignore MySQL constraints/options; imports need only SQLite affinities.
522
+ if mysql_type in {"tinyint", "smallint", "mediumint", "int", "bigint"}:
523
+ affinity = "INTEGER"
524
+ elif mysql_type in {"float", "double", "decimal"}:
525
+ affinity = "REAL"
526
+ else:
527
+ affinity = "TEXT"
528
+ columns.append((column[1], affinity))
529
+ if columns:
530
+ tables[match[1]] = columns
531
+ if not tables:
532
+ raise ValueError("Cannot parse any CREATE TABLE definitions from the source schema")
533
+ return tables
534
+
535
+
536
+ MYSQL_ESCAPES = {"0": "\0", "b": "\b", "n": "\n", "r": "\r", "t": "\t", "Z": "\x1a", "\\": "\\"}
537
+
538
+
539
+ def mysql_value(value: str) -> str | None:
540
+ r"""Decode MySQL dump escapes, distinguishing SQL NULL (\N) from text."""
541
+ if value == r"\N":
542
+ return None
543
+ return re.sub(r"\\(.)", lambda m: MYSQL_ESCAPES.get(m[1], m[1]), value)
544
+
545
+
546
+ def import_table(
547
+ db: sqlite3.Connection, table: str, columns: Sequence[tuple[str, str]], path: Path
548
+ ) -> int:
549
+ """Import one gzipped tab-separated dump, add lookup indexes and commit."""
550
+ destination = "ensembl_" + table
551
+ definitions = ", ".join(f"{sql_name(name)} {kind}" for name, kind in columns)
552
+ db.execute(f"CREATE TABLE {sql_name(destination)} ({definitions})")
553
+ statement = f"INSERT INTO {sql_name(destination)} VALUES ({','.join('?' for _ in columns)})"
554
+ count = 0
555
+ batch = []
556
+ with input_progress(path, "Import " + table, text=True) as (stream, bar, update):
557
+ for number, line in enumerate(stream, 1):
558
+ values = line.rstrip("\n").removesuffix("\r").split("\t")
559
+ if len(values) != len(columns):
560
+ raise ValueError(
561
+ f"{path.name}:{number}: {len(values)} fields; expected {len(columns)}"
562
+ )
563
+ batch.append(tuple(mysql_value(value) for value in values))
564
+ if len(batch) == 10000:
565
+ db.executemany(statement, batch)
566
+ count += len(batch)
567
+ batch.clear()
568
+ bar.set_postfix(rows=f"{count:,}", refresh=False)
569
+ update()
570
+ if batch:
571
+ db.executemany(statement, batch)
572
+ count += len(batch)
573
+ bar.set_postfix(rows=f"{count:,}", refresh=False)
574
+ names = {name for name, _ in columns}
575
+ # Primary lookup IDs plus joins commonly needed for transcript/domain data.
576
+ index_columns = list(
577
+ dict.fromkeys(
578
+ name
579
+ for name in (
580
+ table + "_id",
581
+ "stable_id",
582
+ "transcript_id",
583
+ "translation_id",
584
+ "exon_id",
585
+ "gene_id",
586
+ "seq_region_id",
587
+ "analysis_id",
588
+ "xref_id",
589
+ "ensembl_id",
590
+ "dbprimary_acc",
591
+ "interpro_ac",
592
+ "id",
593
+ "meta_key",
594
+ )
595
+ if name in names
596
+ )
597
+ )
598
+ for name in progress(index_columns, desc="Index " + table, unit="index"):
599
+ db.execute(
600
+ f"CREATE INDEX IF NOT EXISTS {sql_name(destination + '_' + name)} "
601
+ f"ON {sql_name(destination)} ({sql_name(name)})"
602
+ )
603
+ db.commit()
604
+ LOG.info("Imported %s: %s rows", table, f"{count:,}")
605
+ return count
606
+
607
+
608
+ # Chunked sequence storage and inclusive interval reads.
609
+
610
+
611
+ def create_sequence_tables(db: sqlite3.Connection) -> None:
612
+ """Create sequence metadata and separately addressable compressed chunks."""
613
+ db.executescript("""
614
+ CREATE TABLE sequences (
615
+ kind TEXT NOT NULL, sequence_id TEXT NOT NULL, stable_id TEXT NOT NULL,
616
+ header TEXT NOT NULL, length INTEGER NOT NULL, sha256 TEXT NOT NULL,
617
+ PRIMARY KEY (kind, sequence_id)
618
+ ) WITHOUT ROWID;
619
+ CREATE INDEX sequence_stable_id ON sequences (kind, stable_id);
620
+ CREATE TABLE sequence_chunks (
621
+ kind TEXT NOT NULL, sequence_id TEXT NOT NULL,
622
+ start INTEGER NOT NULL, data BLOB NOT NULL,
623
+ PRIMARY KEY (kind, sequence_id, start)
624
+ ) WITHOUT ROWID;
625
+ """)
626
+
627
+
628
+ def validate_dna_regions(db: sqlite3.Connection) -> None:
629
+ """Names are unique within coordinate systems, not across genome assemblies."""
630
+ db.execute("CREATE INDEX IF NOT EXISTS ensembl_seq_region_name ON ensembl_seq_region (name)")
631
+ unmatched = db.execute(
632
+ "SELECT s.sequence_id, s.length FROM sequences s WHERE s.kind='dna' AND NOT EXISTS ("
633
+ "SELECT 1 FROM ensembl_seq_region r JOIN ensembl_coord_system c USING (coord_system_id) "
634
+ "WHERE r.name=s.sequence_id AND c.version=? AND r.length=s.length) LIMIT 1",
635
+ (ASSEMBLY,),
636
+ ).fetchone()
637
+ if unmatched:
638
+ sequence_id, length = unmatched
639
+ candidates = db.execute(
640
+ "SELECT r.length FROM ensembl_seq_region r "
641
+ "JOIN ensembl_coord_system c USING (coord_system_id) "
642
+ "WHERE r.name=? AND c.version=?",
643
+ (sequence_id, ASSEMBLY),
644
+ ).fetchall()
645
+ lengths = sorted({row[0] for row in candidates})
646
+ raise ValueError(
647
+ f"FASTA/core sequence-region mismatch for {sequence_id}: FASTA length {length}; "
648
+ f"{ASSEMBLY} core lengths {lengths or 'no matching region'}"
649
+ )
650
+ LOG.info("Validated DNA sequence-region lengths against %s core records", ASSEMBLY)
651
+
652
+
653
+ def import_fasta(db: sqlite3.Connection, kind: str, path: Path, *, resume: bool = False) -> int:
654
+ """Stream FASTA, optionally reusing committed records from the same input.
655
+
656
+ The builder checks the source checksum before enabling resume. Record rows
657
+ exist only after their chunks are complete; discard any orphaned chunks.
658
+ """
659
+ completed = set()
660
+ if resume:
661
+ completed = {
662
+ row[0] for row in db.execute("SELECT sequence_id FROM sequences WHERE kind=?", (kind,))
663
+ }
664
+ db.execute(
665
+ "DELETE FROM sequence_chunks WHERE kind=? AND NOT EXISTS ("
666
+ "SELECT 1 FROM sequences s WHERE s.kind=sequence_chunks.kind "
667
+ "AND s.sequence_id=sequence_chunks.sequence_id)",
668
+ (kind,),
669
+ )
670
+ db.commit()
671
+ if completed:
672
+ LOG.info(
673
+ "Resuming %s FASTA: reusing %s complete sequences; rereading gzip to the next record",
674
+ kind,
675
+ f"{len(completed):,}",
676
+ )
677
+ sequence_id = header = None
678
+ skip_record = False
679
+ buffer = bytearray()
680
+ offset = count = 0
681
+ digest = hashlib.sha256()
682
+
683
+ def flush(size: int) -> None:
684
+ """Store the next chunk; its database start is 1-based, not a byte offset."""
685
+ nonlocal offset
686
+ chunk = bytes(buffer[:size])
687
+ del buffer[:size]
688
+ db.execute(
689
+ "INSERT INTO sequence_chunks VALUES (?, ?, ?, ?)",
690
+ (kind, sequence_id, offset + 1, zlib.compress(chunk, level=1)),
691
+ )
692
+ offset += len(chunk)
693
+ if size == CHUNK_SIZE:
694
+ update()
695
+
696
+ def finish() -> None:
697
+ """Flush one FASTA record and record its full length, header and checksum."""
698
+ nonlocal count
699
+ if sequence_id is None:
700
+ return
701
+ if skip_record:
702
+ count += 1
703
+ return
704
+ if buffer:
705
+ flush(len(buffer))
706
+ if offset == 0:
707
+ raise ValueError(f"Empty FASTA sequence: {sequence_id}")
708
+ # Assembly sequence names may contain dots; only peptide version suffixes
709
+ # are removed to support lookups by either stable or versioned protein ID.
710
+ stable_id = sequence_id if kind == "dna" else re.sub(r"\.\d+$", "", sequence_id)
711
+ db.execute(
712
+ "INSERT INTO sequences VALUES (?, ?, ?, ?, ?, ?)",
713
+ (kind, sequence_id, stable_id, header, offset, digest.hexdigest()),
714
+ )
715
+ count += 1
716
+ if count % 1000 == 0 or kind == "dna":
717
+ db.commit()
718
+ bar.set_postfix(records=f"{count:,}", refresh=False)
719
+ update()
720
+
721
+ with input_progress(path, "Import " + kind + " FASTA") as (stream, bar, update):
722
+ for line in stream:
723
+ if line.startswith(b">"):
724
+ finish()
725
+ header = line[1:].strip().decode("utf-8")
726
+ sequence_id = header.split()[0]
727
+ skip_record = sequence_id in completed
728
+ offset = 0
729
+ digest = hashlib.sha256()
730
+ else:
731
+ if skip_record:
732
+ continue
733
+ sequence = line.strip().upper()
734
+ if not sequence:
735
+ continue
736
+ if sequence_id is None:
737
+ raise ValueError("FASTA sequence precedes its header")
738
+ alphabet = b"ABCDEFGHIJKLMNOPQRSTUVWXYZ*" if kind == "pep" else b"ACGTRYSWKMBDHVN"
739
+ if sequence.translate(None, alphabet):
740
+ raise ValueError(f"Invalid {kind} sequence characters in {sequence_id}")
741
+ digest.update(sequence)
742
+ buffer.extend(sequence)
743
+ while len(buffer) >= CHUNK_SIZE:
744
+ flush(CHUNK_SIZE)
745
+ finish()
746
+ bar.set_postfix(records=f"{count:,}", refresh=False)
747
+ if not count:
748
+ raise ValueError(f"No FASTA records in {path}")
749
+ db.commit()
750
+ LOG.info("Imported %s: %s sequences", kind, f"{count:,}")
751
+ return count
752
+
753
+
754
+ def fetch_sequence(
755
+ db: sqlite3.Connection,
756
+ kind: str,
757
+ sequence_id: str,
758
+ start: int = 1,
759
+ end: int | None = None,
760
+ strand: int = 1,
761
+ *,
762
+ _chunk_cache: ChunkCache | None = None,
763
+ _cache_limit: int = 64,
764
+ ) -> str:
765
+ """Read a 1-based inclusive interval, optionally using a bounded chunk cache.
766
+
767
+ Unversioned protein IDs are accepted only when they match one sequence.
768
+ Negative-strand nucleotide reads return the reverse complement.
769
+ """
770
+ row = db.execute(
771
+ "SELECT sequence_id, length FROM sequences WHERE kind=? AND sequence_id=?",
772
+ (kind, sequence_id),
773
+ ).fetchone()
774
+ if row is None:
775
+ # Never silently choose among several versions of an unversioned ID.
776
+ rows = db.execute(
777
+ "SELECT sequence_id, length FROM sequences WHERE kind=? AND stable_id=?",
778
+ (kind, sequence_id),
779
+ ).fetchall()
780
+ if len(rows) != 1:
781
+ raise KeyError(f"Missing or ambiguous sequence: {kind}/{sequence_id}")
782
+ row = rows[0]
783
+ sequence_id, length = row
784
+ end = length if end is None else end
785
+ if not 1 <= start <= end <= length:
786
+ raise ValueError(f"Invalid interval {start}-{end}; sequence length is {length}")
787
+ if strand not in {-1, 1} or (kind == "pep" and strand != 1):
788
+ raise ValueError("Strand must be +1/-1 for nucleotides or +1 for peptides")
789
+ # Convert the requested position to the 1-based start of its containing chunk.
790
+ first_chunk = ((start - 1) // CHUNK_SIZE) * CHUNK_SIZE + 1
791
+ pieces = []
792
+ for position in range(first_chunk, end + 1, CHUNK_SIZE):
793
+ key = kind, sequence_id, position
794
+ if _chunk_cache is not None and key in _chunk_cache:
795
+ sequence = _chunk_cache[key]
796
+ _chunk_cache.move_to_end(key)
797
+ else:
798
+ record = db.execute(
799
+ "SELECT data FROM sequence_chunks WHERE kind=? AND sequence_id=? AND start=?", key
800
+ ).fetchone()
801
+ if record is None:
802
+ raise ValueError(f"Missing sequence chunk: {key}")
803
+ sequence = zlib.decompress(record[0])
804
+ if _chunk_cache is not None and _cache_limit:
805
+ _chunk_cache[key] = sequence
806
+ while len(_chunk_cache) > _cache_limit:
807
+ _chunk_cache.popitem(last=False)
808
+ # Python slices are 0-based/exclusive; database intervals are inclusive.
809
+ pieces.append(sequence[max(0, start - position) : end - position + 1])
810
+ result = b"".join(pieces).decode("ascii")
811
+ if len(result) != end - start + 1:
812
+ raise ValueError("Missing or damaged sequence chunks")
813
+ if strand == -1:
814
+ result = result.translate(str.maketrans("ACGTRYSWKMBDHVN", "TGCAYRSWMKVHDBN"))[::-1]
815
+ return result
816
+
817
+
818
+ # Streaming transcript models and coordinate mapping.
819
+
820
+
821
+ def dict_rows(
822
+ db: sqlite3.Connection, query: str, *, description: str | None = None
823
+ ) -> Iterator[DatabaseRow]:
824
+ """Yield named rows from a query without changing the connection row factory."""
825
+ # Sorted queries can do substantial work before returning their first row.
826
+ if description is not None:
827
+ with sqlite_activity(description):
828
+ cursor = db.execute(query)
829
+ else:
830
+ cursor = db.execute(query)
831
+ names = [column[0] for column in cursor.description]
832
+ for row in cursor:
833
+ yield dict(zip(names, row))
834
+
835
+
836
+ class GroupStream:
837
+ """Consume a query grouped by internal transcript ID without loading it all.
838
+
839
+ Input rows and calls to take() must both follow ascending transcript ID.
840
+ This lets exon/feature queries advance alongside the main transcript query.
841
+ """
842
+
843
+ def __init__(self, rows: Iterable[DatabaseRow]) -> None:
844
+ """Prime the first group from an already ordered row iterator."""
845
+ self.groups = iter(groupby(rows, key=lambda row: row["transcript_id"]))
846
+ self.current = next(self.groups, None)
847
+
848
+ def take(self, transcript_id: int) -> list[DatabaseRow]:
849
+ """Return this transcript's rows, or an empty list when it has no group."""
850
+ while self.current is not None and self.current[0] < transcript_id:
851
+ self.current = next(self.groups, None)
852
+ if self.current is None or self.current[0] != transcript_id:
853
+ return []
854
+ rows = list(self.current[1])
855
+ self.current = next(self.groups, None)
856
+ return rows
857
+
858
+
859
+ def reference_structure(
860
+ transcript: Mapping[str, Any],
861
+ exons: Sequence[DatabaseRow],
862
+ translation: Mapping[str, Any] | None,
863
+ ) -> TranscriptProteinFeatureResult:
864
+ """Validate a model and derive exon, splice-window and CDS coordinates.
865
+
866
+ Genomic positions remain on the reference strand. Pre-mRNA and CDS positions
867
+ follow transcription direction; pre-mRNA includes introns, CDS does not.
868
+ A missing translation produces a noncoding model with no CDS blocks.
869
+ Missing leading codon bases occupy peptide coordinates but have no genome
870
+ coordinates; the first CDS block starts after that offset.
871
+ """
872
+ strand = transcript["strand"]
873
+ start, end = transcript["start"], transcript["end"]
874
+ chromosome, assembly = transcript["chromosome"], transcript["assembly_name"]
875
+ if strand not in {-1, 1} or not 1 <= start <= end or not exons:
876
+ raise ValueError("Invalid transcript strand/span or missing exons")
877
+ # Negative-strand transcripts start at the highest genomic exon coordinate.
878
+ oriented = sorted(exons, key=lambda exon: exon["start"], reverse=strand == -1)
879
+ if [exon["rank"] for exon in oriented] != list(range(1, len(exons) + 1)):
880
+ raise ValueError("Exon ranks disagree with transcript strand/order")
881
+ result: TranscriptProteinFeatureResult = {
882
+ "translation_id": translation["stable_id"] if translation else None,
883
+ "protein_length": None,
884
+ "chromosome": chromosome,
885
+ "strand": strand,
886
+ "transcript_genomic_start": start,
887
+ "transcript_genomic_end": end,
888
+ "premrna_length": end - start + 1,
889
+ "transcript_exons": [],
890
+ "cds_blocks": [],
891
+ "protein_features": [],
892
+ "splice_sites": [],
893
+ "assembly_name": assembly,
894
+ }
895
+ for number, exon in enumerate(oriented, 1):
896
+ lo, hi = exon["start"], exon["end"]
897
+ if not start <= lo <= hi <= end or exon["strand"] != strand:
898
+ raise ValueError("Exon coordinates/strand inconsistent with transcript")
899
+ if exon["seq_region_id"] != transcript["seq_region_id"]:
900
+ raise ValueError("Exon and transcript use different sequence regions")
901
+ if number > 1:
902
+ previous = oriented[number - 2]
903
+ if (strand == 1 and previous["end"] >= lo) or (
904
+ strand == -1 and hi >= previous["start"]
905
+ ):
906
+ raise ValueError("Overlapping exons within transcript")
907
+ premrna_start = lo - start + 1 if strand == 1 else end - hi + 1
908
+ premrna_end = hi - start + 1 if strand == 1 else end - lo + 1
909
+ result["transcript_exons"].append(
910
+ {
911
+ "exon_number": number,
912
+ "chromosome": chromosome,
913
+ "genomic_start": lo,
914
+ "genomic_end": hi,
915
+ "premrna_start": premrna_start,
916
+ "premrna_end": premrna_end,
917
+ }
918
+ )
919
+ for site_type, position, genomic in (
920
+ ("acceptor", premrna_start, lo if strand == 1 else hi),
921
+ ("donor", premrna_end, hi if strand == 1 else lo),
922
+ ):
923
+ # Outer transcript ends are not internal splice junctions. Each retained
924
+ # window spans two exon bases and the two adjacent intron bases.
925
+ if (site_type == "acceptor" and number == 1) or (
926
+ site_type == "donor" and number == len(oriented)
927
+ ):
928
+ continue
929
+ if (strand, site_type) in {(-1, "donor"), (1, "acceptor")}:
930
+ disruption_start, disruption_end = genomic - 2, genomic + 1
931
+ else:
932
+ disruption_start, disruption_end = genomic - 1, genomic + 2
933
+ result["splice_sites"].append(
934
+ {
935
+ "type": cast(Literal["acceptor", "donor"], site_type),
936
+ "exon_number": number,
937
+ "genomic_position": genomic,
938
+ "premrna_position": position,
939
+ "disruption_start": disruption_start,
940
+ "disruption_end": disruption_end,
941
+ }
942
+ )
943
+ if translation is None:
944
+ return result
945
+ exon_by_id = {exon["exon_id"]: exon for exon in oriented}
946
+ first = exon_by_id.get(translation["start_exon_id"])
947
+ last = exon_by_id.get(translation["end_exon_id"])
948
+ if first is None or last is None or first["rank"] > last["rank"]:
949
+ raise ValueError("Translation endpoints do not belong to ordered transcript exons")
950
+ for exon, offset in ((first, translation["seq_start"]), (last, translation["seq_end"])):
951
+ if not 1 <= offset <= exon["end"] - exon["start"] + 1:
952
+ raise ValueError("Translation endpoint offset lies outside exon")
953
+ # Translation offsets are 1-based within the endpoint exons, counted in
954
+ # transcription direction (backwards from exon end on the negative strand).
955
+ first_pos = (
956
+ first["start"] + translation["seq_start"] - 1
957
+ if strand == 1
958
+ else first["end"] - translation["seq_start"] + 1
959
+ )
960
+ last_pos = (
961
+ last["start"] + translation["seq_end"] - 1
962
+ if strand == 1
963
+ else last["end"] - translation["seq_end"] + 1
964
+ )
965
+ if (strand == 1 and first_pos > last_pos) or (strand == -1 and first_pos < last_pos):
966
+ raise ValueError("Translation start follows translation end")
967
+ coding_lo, coding_hi = sorted((first_pos, last_pos))
968
+ # A 5'-incomplete CDS starts partway through the first peptide codon.
969
+ # Ensembl pads that codon with `phase` unknown bases when translating.
970
+ # Keep those peptide coordinates, but map only bases present in the genome:
971
+ # starting at phase + 1 also prevents a missing native start being retained.
972
+ phase = first["phase"]
973
+ if phase not in {-1, 0, 1, 2}:
974
+ raise ValueError("Invalid translation start exon phase")
975
+ start_phase = max(0, phase)
976
+ if start_phase:
977
+ result["cds_start_phase"] = start_phase
978
+ cds_pos = start_phase + 1
979
+ # Clip each exon to the coding span, then concatenate its CDS coordinates.
980
+ for prepared_exon in result["transcript_exons"]:
981
+ lo, hi = (
982
+ max(prepared_exon["genomic_start"], coding_lo),
983
+ min(prepared_exon["genomic_end"], coding_hi),
984
+ )
985
+ if lo > hi:
986
+ continue
987
+ length = hi - lo + 1
988
+ result["cds_blocks"].append(
989
+ {
990
+ "chromosome": chromosome,
991
+ "genomic_start": lo,
992
+ "genomic_end": hi,
993
+ "premrna_start": lo - start + 1 if strand == 1 else end - hi + 1,
994
+ "premrna_end": hi - start + 1 if strand == 1 else end - lo + 1,
995
+ "strand": strand,
996
+ "assembly_name": assembly,
997
+ "cds_start": cds_pos,
998
+ "cds_end": cds_pos + length - 1,
999
+ }
1000
+ )
1001
+ cds_pos += length
1002
+ return result
1003
+
1004
+
1005
+ def protein_segments(
1006
+ aa_start: int,
1007
+ aa_end: int,
1008
+ blocks: Sequence[CDSBlock],
1009
+ *,
1010
+ block_ends: Sequence[int] | None = None,
1011
+ ) -> list[GenomicSegment]:
1012
+ """Map an inclusive amino-acid interval to genomic segments across CDS exons.
1013
+
1014
+ Blocks are ordered by CDS position. Passing their end positions enables a
1015
+ binary search; callers mapping many features can reuse that index.
1016
+ """
1017
+ segments: list[GenomicSegment] = []
1018
+ cds_start, cds_end = (aa_start - 1) * 3 + 1, aa_end * 3
1019
+ # The preprocessor supplies a once-per-transcript index of ordered CDS blocks.
1020
+ if block_ends is not None:
1021
+ first = bisect_left(block_ends, cds_start)
1022
+ last = bisect_left(block_ends, cds_end) + 1
1023
+ selected = blocks[first:last]
1024
+ else:
1025
+ selected = blocks
1026
+ for block in selected:
1027
+ lo = max(cds_start, block["cds_start"])
1028
+ hi = min(cds_end, block["cds_end"])
1029
+ if lo > hi:
1030
+ continue
1031
+ start_offset = lo - block["cds_start"]
1032
+ end_offset = hi - block["cds_start"]
1033
+ # Return ascending genomic bounds even when CDS direction is reversed.
1034
+ if block["strand"] == 1:
1035
+ start = block["genomic_start"] + start_offset
1036
+ end = block["genomic_start"] + end_offset
1037
+ else:
1038
+ start = block["genomic_end"] - end_offset
1039
+ end = block["genomic_end"] - start_offset
1040
+ segments.append(
1041
+ {
1042
+ "chromosome": block["chromosome"],
1043
+ "start": start,
1044
+ "end": end,
1045
+ "strand": block["strand"],
1046
+ "assembly_name": block["assembly_name"],
1047
+ }
1048
+ )
1049
+ return segments
1050
+
1051
+
1052
+ # Source compatibility checks and bulk protein annotation metadata.
1053
+
1054
+
1055
+ def protein_analyses(db: sqlite3.Connection) -> AnalysisMetadata:
1056
+ """Use database names for feature sources, not Ensembl pipeline names.
1057
+
1058
+ For example, logic_name='hmmpanther' describes how the analysis ran;
1059
+ db='PANTHER' identifies the annotation source stored with each feature.
1060
+ Small offline fixtures may have only logic_name, so retain that fallback.
1061
+ Structure pipelines override database labels so their mappings can always
1062
+ be excluded from functional annotations.
1063
+ """
1064
+ analyses = {}
1065
+ for row in dict_rows(db, "SELECT * FROM ensembl_analysis"):
1066
+ source = row.get("db") or row["logic_name"]
1067
+ logic_name = (row["logic_name"] or "").lower()
1068
+ if logic_name in STRUCTURE_SOURCES:
1069
+ source = logic_name
1070
+ if (source or "").lower() in {"panther", "hmmpanther"}:
1071
+ source = "PANTHER"
1072
+ analyses[row["analysis_id"]] = (source, row.get("db_version"))
1073
+ return analyses
1074
+
1075
+
1076
+ def download_panther_classifications(version: str, cache: Path) -> tuple[Path, str, str]:
1077
+ """Fetch the exact release, accepting the historical trailing underscore.
1078
+
1079
+ Prefer an already cached filename so archived files also work offline.
1080
+ Only a 404 permits the alternate name; other HTTP/network errors propagate.
1081
+ """
1082
+ directory = cache / version
1083
+ root = PANTHER_CLASSIFICATIONS_URL + version + "/PANTHER_Sequence_Classification_files/"
1084
+ names = [f"PTHR{version}_human", f"PTHR{version}_human_"]
1085
+ names.sort(
1086
+ key=lambda name: (
1087
+ not ((directory / name).is_file() and (directory / (name + ".sha256")).is_file())
1088
+ )
1089
+ )
1090
+ for index, name in enumerate(names):
1091
+ url = root + name
1092
+ try:
1093
+ path, digest = download(url, directory)
1094
+ return path, digest, url
1095
+ except urllib.error.HTTPError as exc:
1096
+ if exc.code != 404 or index == len(names) - 1:
1097
+ raise
1098
+ LOG.info(
1099
+ "PANTHER %s filename unavailable; trying the alternate archived filename", version
1100
+ )
1101
+ raise AssertionError("Unreachable")
1102
+
1103
+
1104
+ def panther_release(analyses: AnalysisMetadata) -> str | None:
1105
+ """Require one exact PANTHER release when automatic metadata is needed."""
1106
+ versions = {version for source, version in analyses.values() if source == "PANTHER"}
1107
+ if not versions:
1108
+ return None
1109
+ if len(versions) != 1 or not re.fullmatch(r"\d+(?:\.\d+)*", next(iter(versions)) or ""):
1110
+ raise ValueError(
1111
+ "Cannot determine one PANTHER release from Ensembl analysis metadata; "
1112
+ "supply --panther-classifications FILE"
1113
+ )
1114
+ return next(iter(versions))
1115
+
1116
+
1117
+ def panther_entry_rows(path: Path) -> Iterator[tuple[str, str, str]]:
1118
+ """Read human UniProt-to-subfamily assignments from PANTHER's bulk TSV.
1119
+
1120
+ Older files place the subfamily ID/name in columns 3/5; newer files use
1121
+ columns 4/6. Detect the layout from the classification ID, independently of
1122
+ the filename. Protein accessions always come from the identifier in column 1.
1123
+ Family/subfamily classifications do not supply new domain coordinates.
1124
+ """
1125
+ with input_progress(path, "Import PANTHER", compressed=False, text=True) as (
1126
+ stream,
1127
+ bar,
1128
+ update,
1129
+ ):
1130
+ count = 0
1131
+ columns = None
1132
+ for number, line in enumerate(stream, 1):
1133
+ if not line.strip():
1134
+ continue
1135
+ fields = line.rstrip("\r\n").split("\t")
1136
+ if len(fields) < 5 or not fields[0].startswith("HUMAN|"):
1137
+ raise ValueError(f"Invalid human PANTHER classification at {path}:{number}")
1138
+ accession = re.search(r"(?:^|\|)UniProtKB[:=]([^|]+)", fields[0])
1139
+ if accession is None:
1140
+ raise ValueError(f"Missing UniProt accession at {path}:{number}")
1141
+ if columns is None:
1142
+ layouts = [
1143
+ (id_col, name_col)
1144
+ for id_col, name_col in ((3, 5), (2, 4))
1145
+ if len(fields) > name_col
1146
+ ]
1147
+ columns = next(
1148
+ (
1149
+ layout
1150
+ for layout in layouts
1151
+ if re.fullmatch(r"PTHR\d+:SF\d+", fields[layout[0]])
1152
+ ),
1153
+ None,
1154
+ )
1155
+ # Unassigned leading records can still identify an empty ID
1156
+ # column. Once detected, keep the layout fixed to catch bad rows.
1157
+ if columns is None:
1158
+ columns = next((layout for layout in layouts if not fields[layout[0]]), None)
1159
+ if columns is None:
1160
+ raise ValueError(
1161
+ f"Invalid PANTHER subfamily at {path}:{number}; "
1162
+ "expected an ID in column 3 or 4"
1163
+ )
1164
+ id_col, name_col = columns
1165
+ if len(fields) <= name_col:
1166
+ raise ValueError(f"Invalid human PANTHER classification at {path}:{number}")
1167
+ subfamily, name = fields[id_col], fields[name_col]
1168
+ # Records without a named subfamily cannot support the family fallback.
1169
+ if not subfamily:
1170
+ continue
1171
+ if not re.fullmatch(r"PTHR\d+:SF\d+", subfamily):
1172
+ raise ValueError(f"Invalid PANTHER subfamily at {path}:{number}: {subfamily}")
1173
+ if not name:
1174
+ continue
1175
+ count += 1
1176
+ yield accession[1], subfamily, name
1177
+ if number % 1000 == 0:
1178
+ bar.set_postfix(records=count, refresh=False)
1179
+ update()
1180
+ if not count:
1181
+ raise ValueError(f"No named human PANTHER subfamilies in {path}")
1182
+ bar.set_postfix(records=count, refresh=False)
1183
+
1184
+
1185
+ def panther_translation_annotations(db: sqlite3.Connection) -> dict[int, tuple[str, str]]:
1186
+ """Match exact protein cross-references; never propagate by gene name.
1187
+
1188
+ An ambiguous protein assignment is omitted. The caller also requires its
1189
+ family to agree with the coordinate-bearing Ensembl PANTHER hit.
1190
+ """
1191
+ tables = {row[0] for row in db.execute("SELECT name FROM sqlite_master WHERE type='table'")}
1192
+ if "ensembl_object_xref" not in tables:
1193
+ return {}
1194
+ matches: dict[int, set[tuple[str, str]]] = {}
1195
+ # Start from the small human lookup and use the imported accession/xref
1196
+ # indexes, rather than scanning every row in the much larger core tables.
1197
+ for translation_id, subfamily, name in db.execute("""
1198
+ SELECT o.ensembl_id, p.subfamily_id, p.name
1199
+ FROM ff_panther p CROSS JOIN ensembl_xref x ON x.dbprimary_acc=p.accession
1200
+ CROSS JOIN ensembl_object_xref o ON o.xref_id=x.xref_id
1201
+ WHERE o.ensembl_object_type='Translation'
1202
+ """):
1203
+ matches.setdefault(translation_id, set()).add((subfamily, name))
1204
+ return {key: next(iter(values)) for key, values in matches.items() if len(values) == 1}
1205
+
1206
+
1207
+ def validate_reference_source(db: sqlite3.Connection) -> dict[str, str]:
1208
+ """Reject incompatible or incomplete source databases before changing them."""
1209
+ available = {row[0] for row in db.execute("SELECT name FROM sqlite_master WHERE type='table'")}
1210
+ required = {"ensembl_" + name for name in PREPROCESS_TABLES}
1211
+ required.update({"build_metadata", "sequences", "sequence_chunks"})
1212
+ missing = required - available
1213
+ if missing:
1214
+ if (
1215
+ "build_metadata" in available
1216
+ and dict(db.execute("SELECT key, value FROM build_metadata")).get("reference_kind")
1217
+ == "runtime"
1218
+ ):
1219
+ raise ValueError(
1220
+ "This compact prebuilt reference has no Ensembl source tables. "
1221
+ "Install a newer prebuilt with prepare-data --release RELEASE --force, "
1222
+ "or build a full reference with --from-source before reprocessing."
1223
+ )
1224
+ raise ValueError(
1225
+ f"Preprocessing needs {sorted(missing)}; rebuild with 'fusion-function prepare-data --from-source'"
1226
+ )
1227
+ metadata = dict(db.execute("SELECT key, value FROM build_metadata"))
1228
+ if metadata.get("format_version") != "1":
1229
+ raise ValueError(
1230
+ "Unsupported reference format; rebuild with 'fusion-function prepare-data'"
1231
+ )
1232
+ if metadata.get("species") != SPECIES:
1233
+ raise ValueError("Only human reference databases are supported")
1234
+ if (
1235
+ metadata.get("assembly") != ASSEMBLY
1236
+ or not db.execute(
1237
+ "SELECT 1 FROM ensembl_coord_system WHERE version=? LIMIT 1", (ASSEMBLY,)
1238
+ ).fetchone()
1239
+ ):
1240
+ raise ValueError("Only GRCh38 reference databases are supported")
1241
+ if not re.fullmatch(r"[1-9]\d*", metadata.get("release", "")):
1242
+ raise ValueError("Reference metadata must contain a positive Ensembl release number")
1243
+ if (
1244
+ metadata.get("sequence_chunk_size") != str(CHUNK_SIZE)
1245
+ or metadata.get("sequence_codec") != "zlib"
1246
+ ):
1247
+ raise ValueError("Unsupported sequence chunk format")
1248
+ if metadata.get("preprocessing_version") not in {None, "1"}:
1249
+ raise ValueError("Unsupported preprocessing version")
1250
+ kinds = {row[0] for row in db.execute("SELECT DISTINCT kind FROM sequences")}
1251
+ if not set(SEQUENCE_TYPES) <= kinds:
1252
+ raise ValueError(
1253
+ "Preprocessing requires DNA and peptide sequences; rebuild with 'fusion-function prepare-data'"
1254
+ )
1255
+ return metadata
1256
+
1257
+
1258
+ def interpro_entry_rows(path: Path) -> Iterator[tuple[str, str | None, str | None]]:
1259
+ """Read current and archived InterPro entry lists with the same validation."""
1260
+ with input_progress(path, "Import InterPro", compressed=False, text=True) as (
1261
+ stream,
1262
+ bar,
1263
+ update,
1264
+ ):
1265
+ header = stream.readline().rstrip("\r\n").split("\t")
1266
+ expected = ["ENTRY_AC", "ENTRY_TYPE", "ENTRY_NAME"]
1267
+ if not set(expected) <= set(header):
1268
+ raise ValueError("InterPro TSV must have ENTRY_AC, ENTRY_TYPE and ENTRY_NAME columns")
1269
+ indices = [header.index(name) for name in expected]
1270
+ for number, line in enumerate(stream, 2):
1271
+ if not line.strip():
1272
+ continue
1273
+ values = line.rstrip("\r\n").split("\t")
1274
+ if len(values) != len(header):
1275
+ raise ValueError(f"InterPro TSV line {number}: wrong field count")
1276
+ accession, entry_type, name = [values[index] for index in indices]
1277
+ if not re.fullmatch(r"IPR\d+", accession):
1278
+ raise ValueError(f"Invalid InterPro accession on line {number}")
1279
+ entry_type = entry_type.strip().lower().replace(" ", "_").replace("-", "_")
1280
+ yield accession, name or None, entry_type or None
1281
+ if number % 1000 == 0:
1282
+ update()
1283
+
1284
+
1285
+ def restore_interpro_history(
1286
+ db: sqlite3.Connection, metadata: InterProMetadata, cache: Path
1287
+ ) -> tuple[dict[str, str], list[dict[str, str]]]:
1288
+ """Supplement missing types, newest archive first; never overwrite current types.
1289
+
1290
+ Explicit local entry lists do not trigger archive downloads.
1291
+ """
1292
+ # This small mapping table includes all possible core accessions. Only if
1293
+ # archives cannot resolve some do we scan the much larger feature table.
1294
+ required = {row[0] for row in db.execute("SELECT DISTINCT interpro_ac FROM ensembl_interpro")}
1295
+ missing = {accession for accession in required if not metadata.get(accession, (None, None))[1]}
1296
+ if not missing:
1297
+ return {}, []
1298
+ LOG.warning(
1299
+ "%s Ensembl InterPro accessions lack current metadata; searching archived entry lists before transcript preprocessing",
1300
+ len(missing),
1301
+ )
1302
+ versions = sorted(
1303
+ (name for name in listing(INTERPRO_RELEASES_URL) if re.fullmatch(r"\d+(?:\.\d+)?", name)),
1304
+ key=lambda name: tuple(int(part) for part in name.split(".")),
1305
+ reverse=True,
1306
+ )
1307
+ restored, sources = {}, []
1308
+ for version in versions:
1309
+ url = INTERPRO_RELEASES_URL + version + "/entry.list"
1310
+ try:
1311
+ path, digest = download(url, cache / "releases" / version)
1312
+ except urllib.error.HTTPError as exc:
1313
+ if exc.code == 404:
1314
+ continue
1315
+ raise
1316
+ recovered = []
1317
+ for accession, name, entry_type in interpro_entry_rows(path):
1318
+ if accession in missing and entry_type:
1319
+ recovered.append((accession, name, entry_type))
1320
+ metadata[accession] = name or metadata.get(accession, (None, None))[0], entry_type
1321
+ restored[accession] = version
1322
+ missing.remove(accession)
1323
+ if recovered:
1324
+ db.executemany(
1325
+ "INSERT INTO ff_interpro VALUES (?, ?, ?) ON CONFLICT(interpro_id) "
1326
+ "DO UPDATE SET name=COALESCE(excluded.name, ff_interpro.name), "
1327
+ "entry_type=excluded.entry_type",
1328
+ recovered,
1329
+ )
1330
+ sources.append({"release": version, "url": url, "sha256": digest})
1331
+ LOG.info(
1332
+ "Recovered %s historical InterPro entries from release %s; %s remaining",
1333
+ len(recovered),
1334
+ version,
1335
+ len(missing),
1336
+ )
1337
+ if not missing:
1338
+ break
1339
+ if missing:
1340
+ analyses = protein_analyses(db)
1341
+ used = {
1342
+ row[0]
1343
+ for row in sqlite_phase(
1344
+ db,
1345
+ """
1346
+ SELECT DISTINCT i.interpro_ac, pf.analysis_id FROM ensembl_protein_feature pf
1347
+ JOIN ensembl_translation tr USING (translation_id)
1348
+ JOIN ensembl_transcript t ON t.transcript_id=tr.transcript_id
1349
+ AND t.canonical_translation_id=tr.translation_id
1350
+ JOIN ensembl_interpro i ON i.id=pf.hit_name
1351
+ WHERE t.is_current=1
1352
+ """,
1353
+ "Check unresolved InterPro metadata",
1354
+ )
1355
+ if (analyses.get(row[1], (None, None))[0] or "").lower() not in STRUCTURE_SOURCES
1356
+ }
1357
+ missing.intersection_update(used)
1358
+ if missing:
1359
+ examples = ", ".join(sorted(missing)[:10])
1360
+ raise ValueError(
1361
+ f"Missing InterPro entry types for {len(missing)} accession(s) after checking archived metadata: "
1362
+ f"{examples}. Supply a complete --interpro-entries FILE; transcript preprocessing has not started."
1363
+ )
1364
+ return restored, sources
1365
+
1366
+
1367
+ # Derived-table preparation: one transaction, one transcript at a time.
1368
+
1369
+
1370
+ def _record_transcript_error(
1371
+ summary: dict[str, TranscriptErrorSummary], transcript_id: str, message: str
1372
+ ) -> None:
1373
+ """Group errors by reason while keeping variable details in the stored payload."""
1374
+ # CDS/peptide mismatch lengths and unsupported assembly names vary, so group
1375
+ # by the fixed message prefix. Each original full error remains retrievable.
1376
+ reason = message.partition(":")[0]
1377
+ entry = summary.setdefault(reason, {"count": 0, "example_transcripts": []})
1378
+ entry["count"] += 1
1379
+ if len(entry["example_transcripts"]) < 3:
1380
+ entry["example_transcripts"].append(transcript_id)
1381
+
1382
+
1383
+ def uniprot_lookup_fingerprint(
1384
+ db: sqlite3.Connection, metadata: Mapping[str, str], input_sha256: str
1385
+ ) -> str | None:
1386
+ """Key verified lookups by XML, mapping code and immutable imported sources.
1387
+
1388
+ Source checksums describe the raw tables and sequences imported by the
1389
+ builder. References without this provenance are reverified, never reused.
1390
+ """
1391
+ if not db.execute(
1392
+ "SELECT 1 FROM sqlite_master WHERE type='table' AND name='source_files'"
1393
+ ).fetchone():
1394
+ return None
1395
+ sources = list(db.execute("SELECT url, sha256, bytes FROM source_files ORDER BY url"))
1396
+ if not sources or not all(row[1] for row in sources):
1397
+ return None
1398
+ fingerprint = {
1399
+ "uniprot_sha256": input_sha256,
1400
+ "mapping_sha256": sha256_file(Path(__file__).with_name("uniprot.py")),
1401
+ "ensembl_sources": sources,
1402
+ "reference": {
1403
+ key: metadata.get(key)
1404
+ for key in (
1405
+ "format_version",
1406
+ "release",
1407
+ "species",
1408
+ "assembly",
1409
+ "sequence_codec",
1410
+ "sequence_chunk_size",
1411
+ )
1412
+ },
1413
+ }
1414
+ return hashlib.sha256(json.dumps(fingerprint, sort_keys=True).encode()).hexdigest()
1415
+
1416
+
1417
+ def encode_transcript_payload(payload: ProteinFeatureResponse) -> str | bytes:
1418
+ """Compress large models while keeping small error records SQL-readable.
1419
+
1420
+ Level 1 limits preprocessing CPU cost. This is lossless storage compression;
1421
+ it does not remove features or change the annotation API's returned model.
1422
+ """
1423
+ encoded = json.dumps(payload, separators=(",", ":"))
1424
+ return encoded if "error" in payload else zlib.compress(encoded.encode("utf-8"), level=1)
1425
+
1426
+
1427
+ def decode_transcript_payload(encoded: str | bytes) -> ProteinFeatureResponse:
1428
+ """Decode compressed models or plain JSON errors and existing references."""
1429
+ if isinstance(encoded, bytes):
1430
+ encoded = zlib.decompress(encoded).decode("utf-8")
1431
+ return cast("ProteinFeatureResponse", json.loads(encoded))
1432
+
1433
+
1434
+ def preprocess_reference(
1435
+ db: sqlite3.Connection,
1436
+ interpro_entries: Path | None = None,
1437
+ *,
1438
+ interpro_archive_cache: Path | None = None,
1439
+ panther_classifications: Path | None = None,
1440
+ panther_cache: Path | None = None,
1441
+ uniprot_features: Path | None = None,
1442
+ uniprot_source: dict[str, str] | None = None,
1443
+ ) -> dict[str, int]:
1444
+ """Build all derived tables in one transaction; leave prior tables on failure.
1445
+
1446
+ Requires raw annotation tables and DNA/peptide FASTA. Invalid individual
1447
+ transcript models are recorded as errors, rather than silently corrected.
1448
+ No nucleotide sequences are duplicated in the derived transcript records.
1449
+ """
1450
+ from .uniprot import CuratedFeature, import_features
1451
+
1452
+ metadata = validate_reference_source(db)
1453
+ analyses = protein_analyses(db)
1454
+ panther_source = None
1455
+ if panther_classifications is None and panther_cache is not None:
1456
+ version = panther_release(analyses)
1457
+ if version:
1458
+ panther_classifications, digest, url = download_panther_classifications(
1459
+ version, panther_cache
1460
+ )
1461
+ panther_source = {"version": version, "url": url, "sha256": digest}
1462
+ if uniprot_features is not None and uniprot_source is None:
1463
+ uniprot_source = {
1464
+ "path": str(uniprot_features.resolve()),
1465
+ "sha256": sha256_file(uniprot_features),
1466
+ }
1467
+ uniprot_fingerprint = (
1468
+ uniprot_lookup_fingerprint(db, metadata, uniprot_source["sha256"])
1469
+ if uniprot_features is not None and uniprot_source is not None
1470
+ else None
1471
+ )
1472
+ if panther_classifications is not None and panther_source is None:
1473
+ panther_source = {
1474
+ "path": str(panther_classifications.resolve()),
1475
+ "sha256": sha256_file(panther_classifications),
1476
+ }
1477
+ LOG.info("Preprocessing reference transcript models, domains and splice sites")
1478
+ db.commit()
1479
+ # These are reproducible public reference annotations. Some SQLite builds
1480
+ # default to zeroing deleted content, which rewrites entire large tables and
1481
+ # their rollback journals. Reclaim pages without scrubbing; journaling and
1482
+ # atomic rollback remain enabled. Restore the caller's setting on every exit.
1483
+ secure_delete = int(db.execute("PRAGMA main.secure_delete").fetchone()[0])
1484
+ # FAST reads back as 2, but setting the numeric value 2 means ON. Restore
1485
+ # the keyword so a caller using FAST keeps that exact policy.
1486
+ previous_secure_delete = ("OFF", "ON", "FAST")[secure_delete]
1487
+ db.execute("PRAGMA main.secure_delete=OFF")
1488
+ # Derived-table replacement, including DDL, is atomic. A failed metadata
1489
+ # lookup or model build rolls back to the prior usable reference tables.
1490
+ transcript_progress = None
1491
+ try:
1492
+ LOG.info(
1493
+ "SQLite reference preprocessing: secure_delete=%s (previously %s), "
1494
+ "journal_mode=%s, synchronous=%s, auto_vacuum=%s, page_size=%s",
1495
+ db.execute("PRAGMA main.secure_delete").fetchone()[0],
1496
+ previous_secure_delete,
1497
+ db.execute("PRAGMA main.journal_mode").fetchone()[0],
1498
+ db.execute("PRAGMA main.synchronous").fetchone()[0],
1499
+ db.execute("PRAGMA main.auto_vacuum").fetchone()[0],
1500
+ db.execute("PRAGMA main.page_size").fetchone()[0],
1501
+ )
1502
+ db.execute("BEGIN IMMEDIATE")
1503
+ if uniprot_features is not None:
1504
+ if (
1505
+ uniprot_fingerprint is not None
1506
+ and uniprot_fingerprint == metadata.get("uniprot_lookup_fingerprint")
1507
+ and metadata.get("uniprot_counts") is not None
1508
+ and db.execute(
1509
+ "SELECT 1 FROM sqlite_master WHERE type='table' AND name='ff_uniprot_features'"
1510
+ ).fetchone()
1511
+ ):
1512
+ LOG.info("Reusing sequence-verified UniProt lookups: inputs and mapping unchanged")
1513
+ uniprot_counts = json.loads(metadata["uniprot_counts"])
1514
+ else:
1515
+ uniprot_counts = import_features(db, uniprot_features)
1516
+ else:
1517
+ uniprot_counts = None
1518
+ # An offline low-level call can reuse already verified UniProt lookups.
1519
+ db.execute(
1520
+ "CREATE TABLE IF NOT EXISTS ff_uniprot_features (sequence_id TEXT NOT NULL, "
1521
+ "feature_id TEXT NOT NULL, payload TEXT NOT NULL, "
1522
+ "PRIMARY KEY(sequence_id, feature_id)) WITHOUT ROWID"
1523
+ )
1524
+ # Persist the bulk lookup so subsequent offline preprocessing can reuse it.
1525
+ # Import and all derived changes participate in the same rollback boundary.
1526
+ db.execute(
1527
+ "CREATE TABLE IF NOT EXISTS ff_panther (accession TEXT PRIMARY KEY, "
1528
+ "subfamily_id TEXT NOT NULL, name TEXT NOT NULL) WITHOUT ROWID"
1529
+ )
1530
+ if panther_classifications is not None:
1531
+ db.execute("DELETE FROM ff_panther")
1532
+ db.executemany(
1533
+ "INSERT INTO ff_panther VALUES (?, ?, ?)",
1534
+ panther_entry_rows(panther_classifications),
1535
+ )
1536
+ with sqlite_activity("Match PANTHER subfamilies to Ensembl proteins"):
1537
+ panther_annotations = (
1538
+ panther_translation_annotations(db)
1539
+ if db.execute("SELECT 1 FROM ff_panther LIMIT 1").fetchone()
1540
+ else {}
1541
+ )
1542
+ LOG.info(
1543
+ "PANTHER subfamily assignments matched %s Ensembl translations by protein cross-reference",
1544
+ f"{len(panther_annotations):,}",
1545
+ )
1546
+ with sqlite_activity("Preserve existing InterPro metadata"):
1547
+ has_interpro = db.execute(
1548
+ "SELECT 1 FROM sqlite_master WHERE type='table' AND name='ff_interpro'"
1549
+ ).fetchone()
1550
+ previous_interpro = (
1551
+ list(db.execute("SELECT interpro_id, name, entry_type FROM ff_interpro"))
1552
+ if has_interpro
1553
+ else []
1554
+ )
1555
+ for table in ("ff_transcripts", "ff_interpro"):
1556
+ with sqlite_activity("Drop previous " + table):
1557
+ db.execute("DROP TABLE IF EXISTS " + sql_name(table))
1558
+ with sqlite_activity("Create derived transcript and InterPro tables"):
1559
+ # Large model payloads belong in ordinary rowid-table leaves, not
1560
+ # the intermediate nodes of a WITHOUT ROWID primary-key tree.
1561
+ # Internal transcript-ID order then appends model rows sequentially;
1562
+ # only the much smaller stable-ID index needs random inserts.
1563
+ db.execute(
1564
+ "CREATE TABLE ff_transcripts (transcript_id TEXT NOT NULL PRIMARY KEY, version INTEGER, "
1565
+ "translation_id TEXT, translation_version INTEGER, status TEXT NOT NULL, "
1566
+ "payload BLOB NOT NULL)"
1567
+ )
1568
+ db.execute(
1569
+ "CREATE TABLE ff_interpro (interpro_id TEXT PRIMARY KEY, name TEXT, entry_type TEXT) WITHOUT ROWID"
1570
+ )
1571
+ # Priority: core names < previously stored types < supplied/current list.
1572
+ # History fills only remaining gaps; it never overwrites a known type.
1573
+ with sqlite_activity("Load InterPro names from Ensembl cross-references"):
1574
+ db.execute(
1575
+ "INSERT INTO ff_interpro SELECT dbprimary_acc, "
1576
+ "COALESCE(MAX(NULLIF(description,'')), MAX(NULLIF(display_label,''))), NULL "
1577
+ "FROM ensembl_xref WHERE dbprimary_acc LIKE 'IPR%' GROUP BY dbprimary_acc"
1578
+ )
1579
+ db.executemany(
1580
+ "INSERT INTO ff_interpro VALUES (?, ?, ?) ON CONFLICT(interpro_id) "
1581
+ "DO UPDATE SET name=COALESCE(excluded.name, ff_interpro.name), "
1582
+ "entry_type=excluded.entry_type",
1583
+ previous_interpro,
1584
+ )
1585
+ provided_accessions = set()
1586
+ if interpro_entries:
1587
+ LOG.info("Importing InterPro entry names/types from %s", interpro_entries)
1588
+ for row in interpro_entry_rows(interpro_entries):
1589
+ provided_accessions.add(row[0])
1590
+ db.execute(
1591
+ "INSERT INTO ff_interpro VALUES (?, ?, ?) ON CONFLICT(interpro_id) "
1592
+ "DO UPDATE SET name=excluded.name, entry_type=excluded.entry_type",
1593
+ row,
1594
+ )
1595
+ ipr_metadata = {row[0]: (row[1], row[2]) for row in db.execute("SELECT * FROM ff_interpro")}
1596
+ historical_entries = json.loads(metadata.get("interpro_historical_entries", "{}"))
1597
+ historical_sources = json.loads(metadata.get("interpro_historical_sources", "[]"))
1598
+ # Entries explicitly present in the new list take precedence over history.
1599
+ for accession in provided_accessions:
1600
+ historical_entries.pop(accession, None)
1601
+ if interpro_archive_cache is not None:
1602
+ restored, sources = restore_interpro_history(db, ipr_metadata, interpro_archive_cache)
1603
+ historical_entries.update(restored)
1604
+ historical_sources.extend(
1605
+ source for source in sources if source not in historical_sources
1606
+ )
1607
+ with sqlite_activity("Read genome and protein sequence lengths"):
1608
+ genome_lengths = dict(
1609
+ db.execute("SELECT sequence_id, length FROM sequences WHERE kind='dna'")
1610
+ )
1611
+ protein_lengths: dict[str, list[tuple[str, int]]] = {}
1612
+ for stable_id, sequence_id, length in db.execute(
1613
+ "SELECT stable_id, sequence_id, length FROM sequences WHERE kind='pep'"
1614
+ ):
1615
+ protein_lengths.setdefault(stable_id, []).append((sequence_id, length))
1616
+ curated_features = GroupStream(
1617
+ dict_rows(
1618
+ db,
1619
+ """
1620
+ SELECT t.transcript_id, u.payload
1621
+ FROM ensembl_transcript t JOIN ensembl_translation tr
1622
+ ON tr.translation_id=t.canonical_translation_id
1623
+ JOIN ff_uniprot_features u ON u.sequence_id IN
1624
+ (tr.stable_id, tr.stable_id || '.' || tr.version)
1625
+ WHERE t.is_current=1 ORDER BY t.transcript_id, u.feature_id
1626
+ """,
1627
+ description="Prepare ordered UniProt feature stream",
1628
+ )
1629
+ )
1630
+ # All three queries use the same ascending internal ID order. GroupStream
1631
+ # avoids per-transcript SQL queries and loading all exon/features into RAM.
1632
+ exons = GroupStream(
1633
+ dict_rows(
1634
+ db,
1635
+ """
1636
+ SELECT et.transcript_id, et.rank, e.exon_id, e.seq_region_id,
1637
+ e.seq_region_start AS start, e.seq_region_end AS end,
1638
+ e.seq_region_strand AS strand, e.phase
1639
+ FROM ensembl_exon_transcript et JOIN ensembl_exon e USING (exon_id)
1640
+ ORDER BY et.transcript_id, et.rank
1641
+ """,
1642
+ description="Prepare ordered exon stream",
1643
+ )
1644
+ )
1645
+ # Restrict hits to each current transcript's canonical translation; using
1646
+ # alternate translations here would mix incompatible peptide coordinates.
1647
+ features = GroupStream(
1648
+ dict_rows(
1649
+ db,
1650
+ """
1651
+ SELECT tr.transcript_id, pf.protein_feature_id, pf.seq_start AS start,
1652
+ pf.seq_end AS end, pf.hit_name AS feature_id,
1653
+ pf.hit_description AS description, a.logic_name AS source, pf.analysis_id,
1654
+ i.interpro_ac AS interpro_id
1655
+ FROM ensembl_protein_feature pf
1656
+ JOIN ensembl_translation tr USING (translation_id)
1657
+ JOIN ensembl_transcript t ON t.transcript_id=tr.transcript_id
1658
+ AND t.canonical_translation_id=tr.translation_id
1659
+ LEFT JOIN ensembl_analysis a USING (analysis_id)
1660
+ LEFT JOIN ensembl_interpro i ON i.id=pf.hit_name
1661
+ WHERE t.is_current=1
1662
+ ORDER BY tr.transcript_id, pf.protein_feature_id, i.interpro_ac
1663
+ """,
1664
+ description="Prepare ordered protein feature stream",
1665
+ )
1666
+ )
1667
+ transcripts = dict_rows(
1668
+ db,
1669
+ """
1670
+ SELECT t.transcript_id AS internal_id, t.stable_id AS transcript_id,
1671
+ t.version, t.seq_region_id, t.seq_region_start AS start,
1672
+ t.seq_region_end AS end, t.seq_region_strand AS strand,
1673
+ t.canonical_translation_id, r.name AS chromosome,
1674
+ cs.version AS assembly_name, tr.stable_id AS translation_id,
1675
+ tr.version AS translation_version, tr.seq_start, tr.start_exon_id,
1676
+ tr.seq_end, tr.end_exon_id,
1677
+ EXISTS(SELECT 1 FROM ensembl_translation alt
1678
+ WHERE alt.transcript_id=t.transcript_id) AS has_translation
1679
+ FROM ensembl_transcript t JOIN ensembl_seq_region r USING (seq_region_id)
1680
+ JOIN ensembl_coord_system cs USING (coord_system_id)
1681
+ LEFT JOIN ensembl_translation tr ON tr.translation_id=t.canonical_translation_id
1682
+ WHERE t.is_current=1 ORDER BY t.transcript_id
1683
+ """,
1684
+ description="Prepare transcript stream",
1685
+ )
1686
+ counts = {"ready": 0, "noncoding": 0, "error": 0, "protein_features": 0}
1687
+ errors: dict[str, TranscriptErrorSummary] = {}
1688
+ missing_entry_types: set[str] = set()
1689
+ total = sqlite_phase(
1690
+ db,
1691
+ "SELECT COUNT(*) FROM ensembl_transcript t "
1692
+ "JOIN ensembl_seq_region r USING (seq_region_id) "
1693
+ "JOIN ensembl_coord_system cs USING (coord_system_id) "
1694
+ "WHERE t.is_current=1",
1695
+ "Count current transcripts",
1696
+ )[0][0]
1697
+ transcript_progress = progress(
1698
+ transcripts, total=total, desc="Preprocess transcripts", unit="transcript"
1699
+ )
1700
+ payload: ProteinFeatureResponse
1701
+ for number, transcript in enumerate(transcript_progress, 1):
1702
+ exon_rows = exons.take(transcript["internal_id"])
1703
+ feature_rows = features.take(transcript["internal_id"])
1704
+ curated_rows = curated_features.take(transcript["internal_id"])
1705
+ translation = None
1706
+ if transcript["canonical_translation_id"] is not None:
1707
+ translation = {**transcript, "stable_id": transcript["translation_id"]}
1708
+ try:
1709
+ if transcript["assembly_name"] != ASSEMBLY:
1710
+ raise ValueError(
1711
+ f"Unsupported transcript coordinate system: {transcript['assembly_name']}"
1712
+ )
1713
+ if translation is not None and not translation["stable_id"]:
1714
+ raise ValueError("Canonical translation is missing from source table")
1715
+ if translation is None and transcript["has_translation"]:
1716
+ raise ValueError("Transcript has translations but no canonical translation")
1717
+ payload = reference_structure(transcript, exon_rows, translation)
1718
+ if transcript["chromosome"] not in genome_lengths:
1719
+ raise ValueError("Transcript sequence region is absent from genomic FASTA")
1720
+ if transcript["end"] > genome_lengths[transcript["chromosome"]]:
1721
+ raise ValueError("Transcript span exceeds genomic FASTA sequence")
1722
+ status = "noncoding" if translation is None else "ready"
1723
+ if translation:
1724
+ candidates = protein_lengths.get(translation["stable_id"], [])
1725
+ expected_id = f"{translation['stable_id']}.{transcript['translation_version']}"
1726
+ candidates = [
1727
+ record
1728
+ for record in candidates
1729
+ if record[0] in {expected_id, translation["stable_id"]}
1730
+ ]
1731
+ if len(candidates) != 1:
1732
+ raise ValueError(
1733
+ "Canonical translation peptide is missing/ambiguous or has a different version"
1734
+ )
1735
+ protein_length = candidates[0][1]
1736
+ payload["protein_length"] = protein_length
1737
+ cds_length = sum(
1738
+ block["cds_end"] - block["cds_start"] + 1 for block in payload["cds_blocks"]
1739
+ )
1740
+ # Include Ensembl's virtual leading codon bases in the length
1741
+ # comparison, without adding them to genomic CDS blocks.
1742
+ # Peptide FASTA omits the optional terminal stop codon.
1743
+ start_phase = payload.get("cds_start_phase", 0)
1744
+ if cds_length + start_phase not in {protein_length * 3, protein_length * 3 + 3}:
1745
+ raise ValueError(
1746
+ f"CDS/peptide length mismatch: {cds_length} bp, {protein_length} aa, "
1747
+ f"start phase {start_phase}"
1748
+ )
1749
+ block_ends = [block["cds_end"] for block in payload["cds_blocks"]]
1750
+ # Multiple signatures/InterPro mappings can share an interval.
1751
+ # Reuse coordinates within this transcript, then discard them.
1752
+ segment_cache: dict[tuple[int, int], tuple[list[GenomicSegment], int, int]] = {}
1753
+ for feature in feature_rows:
1754
+ source = analyses.get(feature["analysis_id"], (feature["source"], None))[0]
1755
+ if (source or "").lower() in STRUCTURE_SOURCES:
1756
+ # Whole-protein structure mappings are not domains/sites.
1757
+ # Exclude them before validating functional coordinates:
1758
+ # their intervals may exceed this transcript's peptide.
1759
+ continue
1760
+ if not 1 <= feature["start"] <= feature["end"] <= protein_length:
1761
+ raise ValueError(
1762
+ f"Protein feature falls outside peptide: {source} "
1763
+ f"{feature['feature_id']} {feature['start']}-{feature['end']}, "
1764
+ f"peptide length {protein_length}"
1765
+ )
1766
+ interval = feature["start"], feature["end"]
1767
+ mapped = segment_cache.get(interval)
1768
+ if mapped is None:
1769
+ segments = protein_segments(
1770
+ *interval, payload["cds_blocks"], block_ends=block_ends
1771
+ )
1772
+ mapped = (
1773
+ segments,
1774
+ min(segment["start"] for segment in segments),
1775
+ max(segment["end"] for segment in segments),
1776
+ )
1777
+ segment_cache[interval] = mapped
1778
+ segments, genomic_start, genomic_end = mapped
1779
+ name, entry_type = ipr_metadata.get(feature["interpro_id"], (None, None))
1780
+ subfamily = (
1781
+ feature["feature_id"]
1782
+ if (source or "").upper() == "PANTHER"
1783
+ and ":SF" in (feature["feature_id"] or "")
1784
+ else None
1785
+ )
1786
+ subfamily_name = feature["description"] if subfamily else None
1787
+ if source == "PANTHER":
1788
+ assignment = panther_annotations.get(
1789
+ transcript["canonical_translation_id"]
1790
+ )
1791
+ hit = feature["feature_id"] or ""
1792
+ if assignment and assignment[0].split(":")[0] == hit.split(":")[0]:
1793
+ if subfamily is None or subfamily == assignment[0]:
1794
+ subfamily, subfamily_name = assignment
1795
+ payload["protein_features"].append(
1796
+ {
1797
+ "feature_id": feature["feature_id"] or None,
1798
+ "source": source or None,
1799
+ "interpro_id": feature["interpro_id"],
1800
+ "start": feature["start"],
1801
+ "end": feature["end"],
1802
+ "cds_start": (feature["start"] - 1) * 3 + 1,
1803
+ "cds_end": feature["end"] * 3,
1804
+ "chromosome": transcript["chromosome"],
1805
+ "strand": transcript["strand"],
1806
+ "assembly_name": transcript["assembly_name"],
1807
+ "genomic_start": genomic_start,
1808
+ "genomic_end": genomic_end,
1809
+ "genomic_segments": segments,
1810
+ "description": feature["description"] or name,
1811
+ "interpro_name": name,
1812
+ "interpro_entry_type": entry_type,
1813
+ "panther_subfamily_id": subfamily,
1814
+ "panther_subfamily_description": subfamily_name,
1815
+ }
1816
+ )
1817
+ for curated_row in curated_rows:
1818
+ curated = cast(CuratedFeature, json.loads(curated_row["payload"]))
1819
+ start, end = curated["start"], curated["end"]
1820
+ if not 1 <= start <= end <= protein_length:
1821
+ raise ValueError("Verified UniProt feature falls outside peptide")
1822
+ segments = protein_segments(
1823
+ start, end, payload["cds_blocks"], block_ends=block_ends
1824
+ )
1825
+ payload["protein_features"].append(
1826
+ {
1827
+ "feature_id": curated["feature_id"],
1828
+ "feature_type": curated["feature_type"],
1829
+ "description": curated["description"],
1830
+ "start": start,
1831
+ "end": end,
1832
+ "uniprot_accession": curated["uniprot_accession"],
1833
+ "uniprot_isoform": curated["uniprot_isoform"],
1834
+ "evidence": curated["evidence"],
1835
+ "source": "UniProtKB",
1836
+ "interpro_id": None,
1837
+ "cds_start": (start - 1) * 3 + 1,
1838
+ "cds_end": end * 3,
1839
+ "chromosome": transcript["chromosome"],
1840
+ "strand": transcript["strand"],
1841
+ "assembly_name": transcript["assembly_name"],
1842
+ "genomic_start": min(segment["start"] for segment in segments),
1843
+ "genomic_end": max(segment["end"] for segment in segments),
1844
+ "genomic_segments": segments,
1845
+ "interpro_name": None,
1846
+ "interpro_entry_type": None,
1847
+ "panther_subfamily_id": None,
1848
+ "panther_subfamily_description": None,
1849
+ }
1850
+ )
1851
+ counts["protein_features"] += len(payload["protein_features"])
1852
+ except ValueError as exc:
1853
+ # Preserve the reason for unsupported models so readers can explain
1854
+ # the failure, rather than reporting that the transcript is absent.
1855
+ status, payload = "error", {"error": f"{transcript['transcript_id']}: {exc}"}
1856
+ _record_transcript_error(errors, transcript["transcript_id"], str(exc))
1857
+ if status == "ready":
1858
+ payload = cast("TranscriptProteinFeatureResult", payload)
1859
+ missing_entry_types.update(
1860
+ feature["interpro_id"]
1861
+ for feature in payload["protein_features"]
1862
+ if feature["interpro_id"] and not feature["interpro_entry_type"]
1863
+ )
1864
+ counts[status] += 1
1865
+ db.execute(
1866
+ "INSERT INTO ff_transcripts VALUES (?, ?, ?, ?, ?, ?)",
1867
+ (
1868
+ transcript["transcript_id"],
1869
+ transcript["version"],
1870
+ transcript["translation_id"],
1871
+ transcript["translation_version"],
1872
+ status,
1873
+ encode_transcript_payload(payload),
1874
+ ),
1875
+ )
1876
+ if number % 1000 == 0:
1877
+ transcript_progress.set_postfix(
1878
+ ready=counts["ready"], errors=counts["error"], refresh=False
1879
+ )
1880
+ if missing_entry_types:
1881
+ examples = ", ".join(sorted(missing_entry_types)[:10])
1882
+ raise ValueError(
1883
+ f"Missing InterPro entry types for {len(missing_entry_types)} accession(s): {examples}. "
1884
+ "Supply a complete --interpro-entries FILE or fetch compatible metadata; "
1885
+ "the previous reference has been preserved."
1886
+ )
1887
+ with sqlite_activity("Index prepared transcripts"):
1888
+ db.execute("CREATE INDEX ff_transcript_translation ON ff_transcripts (translation_id)")
1889
+ db.execute("CREATE INDEX ff_transcript_status ON ff_transcripts (status)")
1890
+ # Keep provenance in the same transaction as the tables it describes.
1891
+ errors = dict(sorted(errors.items(), key=lambda item: (-item[1]["count"], item[0])))
1892
+ preprocessing_metadata = {
1893
+ "preprocessing_version": "1",
1894
+ "feature_annotation_version": "3",
1895
+ "cds_mapping_version": "2",
1896
+ "transcript_payload_codec": TRANSCRIPT_PAYLOAD_CODEC,
1897
+ "preprocessing_counts": json.dumps(counts),
1898
+ "preprocessing_errors": json.dumps(errors),
1899
+ "preprocessed_utc": datetime.now(timezone.utc).isoformat(),
1900
+ "interpro_entries_sha256": (
1901
+ sha256_file(interpro_entries)
1902
+ if interpro_entries
1903
+ else metadata.get("interpro_entries_sha256", "")
1904
+ ),
1905
+ "interpro_entry_types_loaded": str(
1906
+ any(value[1] for value in ipr_metadata.values())
1907
+ ).lower(),
1908
+ "interpro_historical_entries": json.dumps(historical_entries, sort_keys=True),
1909
+ "interpro_historical_sources": json.dumps(historical_sources, sort_keys=True),
1910
+ "interpro_preserved_metadata_sha256": (
1911
+ metadata.get("interpro_entries_sha256", "")
1912
+ if interpro_entries and previous_interpro
1913
+ else metadata.get("interpro_preserved_metadata_sha256", "")
1914
+ ),
1915
+ }
1916
+ if uniprot_source is not None:
1917
+ preprocessing_metadata["uniprot_features_source"] = json.dumps(
1918
+ uniprot_source, sort_keys=True
1919
+ )
1920
+ preprocessing_metadata["uniprot_counts"] = json.dumps(uniprot_counts, sort_keys=True)
1921
+ # An import without recorded source provenance invalidates any
1922
+ # prior reuse key. Empty values never permit reuse.
1923
+ preprocessing_metadata["uniprot_lookup_fingerprint"] = uniprot_fingerprint or ""
1924
+ if panther_source is not None:
1925
+ preprocessing_metadata["panther_classifications_source"] = json.dumps(
1926
+ panther_source, sort_keys=True
1927
+ )
1928
+ for key, value in preprocessing_metadata.items():
1929
+ db.execute(
1930
+ "INSERT INTO build_metadata VALUES (?, ?) ON CONFLICT(key) DO UPDATE SET value=excluded.value",
1931
+ (key, value),
1932
+ )
1933
+ if counts["ready"] == 0:
1934
+ raise ValueError(
1935
+ "Preprocessing produced no usable coding transcripts; check source tables and FASTA"
1936
+ )
1937
+ with sqlite_activity("Commit preprocessed reference"):
1938
+ db.commit()
1939
+ except BaseException:
1940
+ with sqlite_activity("Roll back preprocessing transaction"):
1941
+ db.rollback()
1942
+ raise
1943
+ finally:
1944
+ try:
1945
+ if transcript_progress is not None:
1946
+ transcript_progress.close()
1947
+ finally:
1948
+ db.execute(f"PRAGMA main.secure_delete={previous_secure_delete}")
1949
+ LOG.info("Preprocessing complete: %s", counts)
1950
+ if counts["error"]:
1951
+ LOG.warning(
1952
+ "%s transcripts have stored errors; inspect ff_transcripts WHERE status='error'",
1953
+ counts["error"],
1954
+ )
1955
+ for reason, summary in errors.items():
1956
+ LOG.warning(
1957
+ " %s: %s transcript(s); examples: %s",
1958
+ reason,
1959
+ f"{summary['count']:,}",
1960
+ ", ".join(summary["example_transcripts"]),
1961
+ )
1962
+ return counts
1963
+
1964
+
1965
+ # Read-only runtime access to a completed reference.
1966
+
1967
+
1968
+ class ReferenceReader:
1969
+ """Read-only local reference access; reuse one reader across fusion calls.
1970
+
1971
+ get_transcript() returns prepared coordinates, splice sites and features,
1972
+ reconstructing pre-mRNA from the genome when requested.
1973
+ Genome chunk caching is LRU and bounded (default at most 64 MiB); no
1974
+ transcript sequences are cached.
1975
+ The reader is intended for one thread; use a separate reader per worker.
1976
+ """
1977
+
1978
+ def __init__(self, database: str | Path, cached_chunks: int = 64) -> None:
1979
+ """Open an existing preprocessed database and configure the chunk cache."""
1980
+ if cached_chunks < 0:
1981
+ raise ValueError("cached_chunks must be nonnegative")
1982
+ path = Path(database).expanduser().resolve()
1983
+ self.db = sqlite3.connect(path.as_uri() + "?mode=ro", uri=True)
1984
+ self.cached_chunks = cached_chunks
1985
+ self.chunk_cache: ChunkCache = OrderedDict()
1986
+ try:
1987
+ metadata = dict(self.db.execute("SELECT key, value FROM build_metadata"))
1988
+ if metadata.get("preprocessing_version") != "1":
1989
+ raise ValueError("Database is not preprocessed; run --preprocess-only DB")
1990
+ if metadata.get("sequence_chunk_size") != str(CHUNK_SIZE):
1991
+ raise ValueError("Unsupported sequence chunk size")
1992
+ if metadata.get("transcript_payload_codec") not in {None, TRANSCRIPT_PAYLOAD_CODEC}:
1993
+ raise ValueError("Unsupported transcript payload codec")
1994
+ self.metadata = metadata
1995
+ except BaseException:
1996
+ self.db.close()
1997
+ raise
1998
+
1999
+ def transcript_error_summary(self) -> dict[str, TranscriptErrorSummary]:
2000
+ """Read grouped failures with example IDs without rebuilding the reference.
2001
+
2002
+ Prepared summaries are cheap metadata reads. If one is absent, inspect
2003
+ only stored error payloads; usable transcript payloads are never loaded.
2004
+ """
2005
+ if "preprocessing_errors" in self.metadata:
2006
+ return cast(
2007
+ dict[str, TranscriptErrorSummary], json.loads(self.metadata["preprocessing_errors"])
2008
+ )
2009
+ errors: dict[str, TranscriptErrorSummary] = {}
2010
+ for transcript_id, encoded in self.db.execute(
2011
+ "SELECT transcript_id, payload FROM ff_transcripts WHERE status='error' ORDER BY transcript_id"
2012
+ ):
2013
+ payload = cast("EnsemblError", decode_transcript_payload(encoded))
2014
+ message = payload["error"].removeprefix(transcript_id + ": ")
2015
+ _record_transcript_error(errors, transcript_id, message)
2016
+ return dict(sorted(errors.items(), key=lambda item: (-item[1]["count"], item[0])))
2017
+
2018
+ def get_transcript(
2019
+ self, transcript_id: str, *, include_sequence: bool = True
2020
+ ) -> ProteinFeatureResponse:
2021
+ """Unknown IDs, version mismatches and unsupported models return errors."""
2022
+ match = re.fullmatch(r"(ENST\d+)(?:\.(\d+))?", transcript_id)
2023
+ if not match:
2024
+ return {"error": f"Invalid human Ensembl transcript ID: {transcript_id}"}
2025
+ stable_id, requested_version = match.groups()
2026
+ row = self.db.execute(
2027
+ "SELECT version, status, payload FROM ff_transcripts WHERE transcript_id=?",
2028
+ (stable_id,),
2029
+ ).fetchone()
2030
+ if row is None:
2031
+ return {"error": f"Transcript {transcript_id} is absent from this reference release"}
2032
+ version, status, encoded = row
2033
+ if requested_version is not None and int(requested_version) != version:
2034
+ return {
2035
+ "error": f"Transcript version mismatch: requested {transcript_id}; stored {stable_id}.{version}"
2036
+ }
2037
+ if status == "noncoding":
2038
+ return {"error": "Transcript is non-coding or has no translation object"}
2039
+ payload = decode_transcript_payload(encoded)
2040
+ if status == "error":
2041
+ return payload
2042
+ payload = cast("TranscriptProteinFeatureResult", payload)
2043
+ if include_sequence:
2044
+ payload["premrna_sequence"] = self.sequence(
2045
+ "dna",
2046
+ cast(str, payload["chromosome"]),
2047
+ payload["transcript_genomic_start"],
2048
+ payload["transcript_genomic_end"],
2049
+ payload["strand"],
2050
+ )
2051
+ return payload
2052
+
2053
+ def sequence(
2054
+ self, kind: str, sequence_id: str, start: int = 1, end: int | None = None, strand: int = 1
2055
+ ) -> str:
2056
+ """Read a sequence interval through this reader's shared bounded cache."""
2057
+ return fetch_sequence(
2058
+ self.db,
2059
+ kind,
2060
+ sequence_id,
2061
+ start,
2062
+ end,
2063
+ strand,
2064
+ _chunk_cache=self.chunk_cache,
2065
+ _cache_limit=self.cached_chunks,
2066
+ )
2067
+
2068
+ def get_interpro_annotation(self, accession: str) -> ProteinFeatureAnnotation | None:
2069
+ """Return stored metadata for an accession, or None when it is absent."""
2070
+ row = self.db.execute(
2071
+ "SELECT name, entry_type FROM ff_interpro WHERE interpro_id=?", (accession,)
2072
+ ).fetchone()
2073
+ if row is None:
2074
+ return None
2075
+ return {"name": row[0], "entry_type": row[1], "interpro_id": accession}
2076
+
2077
+ def close(self) -> None:
2078
+ """Release both the database handle and decompressed genome chunks."""
2079
+ self.chunk_cache.clear()
2080
+ self.db.close()
2081
+
2082
+ def __enter__(self) -> Self:
2083
+ """Use this reader as a context manager."""
2084
+ return self
2085
+
2086
+ def __exit__(
2087
+ self,
2088
+ exc_type: type[BaseException] | None,
2089
+ exc_value: BaseException | None,
2090
+ traceback: TracebackType | None,
2091
+ ) -> None:
2092
+ """Close the reader even when annotation raises an exception."""
2093
+ self.close()
2094
+
2095
+
2096
+ # Full-build orchestration and the prepare-data command.
2097
+
2098
+
2099
+ @contextmanager
2100
+ def build_lock(path: Path) -> Iterator[None]:
2101
+ """Keep concurrent builders from modifying the same resumable database.
2102
+
2103
+ Advisory locks are released by the OS even after a killed process. Leave the
2104
+ small lock file in place so another process cannot lock a different inode.
2105
+ """
2106
+ with path.open("a+b") as stream:
2107
+ try:
2108
+ if os.name == "nt":
2109
+ import msvcrt
2110
+
2111
+ stream.write(b"\0")
2112
+ stream.flush()
2113
+ stream.seek(0)
2114
+ msvcrt.locking(stream.fileno(), msvcrt.LK_NBLCK, 1) # type: ignore[attr-defined] # Windows-only API
2115
+ else:
2116
+ import fcntl
2117
+
2118
+ fcntl.flock(stream.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB)
2119
+ except OSError as exc:
2120
+ raise ValueError(f"Another build is using this output: {path}") from exc
2121
+ try:
2122
+ yield
2123
+ finally:
2124
+ if os.name == "nt":
2125
+ stream.seek(0)
2126
+ msvcrt.locking(stream.fileno(), msvcrt.LK_UNLCK, 1) # type: ignore[attr-defined] # Windows-only API
2127
+ else:
2128
+ fcntl.flock(stream.fileno(), fcntl.LOCK_UN)
2129
+
2130
+
2131
+ class BuildCheckpoint:
2132
+ """Record completed stages only after their SQLite work has committed."""
2133
+
2134
+ def __init__(self, db: sqlite3.Connection, configuration: Mapping[str, object]) -> None:
2135
+ self.db = db
2136
+ db.execute(
2137
+ "CREATE TABLE IF NOT EXISTS build_checkpoints ("
2138
+ "stage TEXT PRIMARY KEY, source TEXT NOT NULL, fingerprint TEXT NOT NULL, "
2139
+ "count INTEGER NOT NULL, complete INTEGER NOT NULL)"
2140
+ )
2141
+ fingerprint = json.dumps(configuration, sort_keys=True)
2142
+ previous = self.read("configuration")
2143
+ if previous and previous["fingerprint"] != fingerprint:
2144
+ raise ValueError(
2145
+ "Build checkpoint belongs to different source/settings; "
2146
+ "use a different --output or remove its .building database"
2147
+ )
2148
+ if previous is None:
2149
+ self.save("configuration", fingerprint=fingerprint)
2150
+
2151
+ def read(self, stage: str) -> CheckpointRecord | None:
2152
+ """Read one stage's source provenance and completion marker."""
2153
+ row = self.db.execute(
2154
+ "SELECT source, fingerprint, count, complete FROM build_checkpoints WHERE stage=?",
2155
+ (stage,),
2156
+ ).fetchone()
2157
+ if row is None:
2158
+ return None
2159
+ return {
2160
+ "source": json.loads(row[0]),
2161
+ "fingerprint": row[1],
2162
+ "count": row[2],
2163
+ "complete": bool(row[3]),
2164
+ }
2165
+
2166
+ def save(
2167
+ self,
2168
+ stage: str,
2169
+ *,
2170
+ source: Mapping[str, Any] | None = None,
2171
+ fingerprint: str = "",
2172
+ count: int = 0,
2173
+ complete: bool = True,
2174
+ ) -> None:
2175
+ """Commit a marker with the work it describes; incomplete stages can resume."""
2176
+ self.db.execute(
2177
+ "INSERT OR REPLACE INTO build_checkpoints VALUES (?, ?, ?, ?, ?)",
2178
+ (stage, json.dumps(source or {}, sort_keys=True), fingerprint, count, int(complete)),
2179
+ )
2180
+ self.db.commit()
2181
+
2182
+
2183
+ def build(args: argparse.Namespace) -> Path:
2184
+ """Resume committed imports in a staging database, then publish atomically.
2185
+
2186
+ Failures preserve the staging database and downloads. Repeating the same
2187
+ command skips completed stages; no incomplete reference replaces the output.
2188
+ """
2189
+ from .uniprot import UNIPROT_HUMAN_URL
2190
+
2191
+ cache = args.cache_dir.expanduser().resolve()
2192
+ base = args.base_url.rstrip("/") + "/"
2193
+ # Resolution, metadata/schema, tables, FASTA, preprocessing, analysis, check, publication.
2194
+ steps = BuildSteps(1 + 3 + len(TABLES) + len(SEQUENCE_TYPES) + 4)
2195
+ with steps.step("Resolve Ensembl release and human GRCh38 core directory"):
2196
+ release, root, core = resolve_core(base, args.release)
2197
+ LOG.info("Resolved Ensembl release %s; core database %s", release, core)
2198
+ output = (
2199
+ (args.output or cache / args.species / ASSEMBLY / f"release-{release}" / "ensembl.sqlite")
2200
+ .expanduser()
2201
+ .resolve()
2202
+ )
2203
+ output.parent.mkdir(parents=True, exist_ok=True)
2204
+ if output.exists() and not args.force:
2205
+ raise FileExistsError(f"Output exists: {output}. Use --force to replace it.")
2206
+ downloads = cache / "downloads" / hashlib.sha256(base.encode()).hexdigest()[:12] / core
2207
+ interpro_cache = cache / "metadata" / "interpro"
2208
+ LOG.info("Final reference database: %s", output)
2209
+ for group in ("core", *SEQUENCE_TYPES):
2210
+ LOG.info("%s download cache: %s", group, downloads / group)
2211
+ LOG.info("InterPro download cache: %s", interpro_cache)
2212
+ LOG.info("Historical InterPro entry-list cache: %s", interpro_cache / "releases")
2213
+ LOG.info(
2214
+ "Partial downloads: <download-cache>/<filename>.<random>.part; checksums: <filename>.sha256"
2215
+ )
2216
+ temporary = output.with_name(output.name + ".building")
2217
+ lock_path = output.with_name(output.name + ".prepare.lock")
2218
+ LOG.info("Resumable build checkpoint: %s; SQLite journal: %s-journal", temporary, temporary)
2219
+ LOG.info("Build lock: %s", lock_path)
2220
+ source_rows = []
2221
+ fetch_interpro = args.interpro_entries is None
2222
+
2223
+ def get(url: str, group: str = "core", *, directory: Path | None = None) -> Path:
2224
+ """Download/cache one source and retain its checksum/size for provenance."""
2225
+ path, digest = download(url, directory if directory is not None else downloads / group)
2226
+ source_rows.append((url, digest, path.stat().st_size))
2227
+ return path
2228
+
2229
+ core_url = root + f"mysql/{core}/"
2230
+ LOG.info("Temporary reference database: %s; rerun the same command to resume", temporary)
2231
+ try:
2232
+ with steps.step("Prepare InterPro metadata"):
2233
+ if fetch_interpro:
2234
+ interpro_entries = get(INTERPRO_ENTRIES_URL, directory=interpro_cache)
2235
+ else:
2236
+ interpro_entries = args.interpro_entries
2237
+ LOG.info("Using local InterPro entries: %s", interpro_entries)
2238
+ with steps.step("Prepare reviewed human UniProt features"):
2239
+ if args.uniprot_features is None:
2240
+ uniprot_features = get(UNIPROT_HUMAN_URL, directory=cache / "metadata" / "uniprot")
2241
+ uniprot_source = {"url": UNIPROT_HUMAN_URL, "sha256": source_rows[-1][1]}
2242
+ else:
2243
+ uniprot_features = args.uniprot_features
2244
+ uniprot_source = {
2245
+ "path": str(uniprot_features),
2246
+ "sha256": sha256_file(uniprot_features),
2247
+ }
2248
+ LOG.info("Using local UniProt features: %s", uniprot_features)
2249
+ with steps.step("Download and read Ensembl table definitions"):
2250
+ schema = parse_schema(get(core_url + core + ".sql.gz"))
2251
+ absent = set(TABLES) - set(schema)
2252
+ if absent:
2253
+ raise ValueError(f"Required tables absent from source schema: {sorted(absent)}")
2254
+ with build_lock(lock_path), closing(sqlite3.connect(temporary)) as db:
2255
+ if output.exists() and not args.force:
2256
+ raise FileExistsError(f"Output appeared during the build: {output}")
2257
+ # Apply build-only settings to the unpublished database. The 64 MiB
2258
+ # page cache reduces repeated reads without caching the entire input.
2259
+ db.execute("PRAGMA journal_mode=DELETE")
2260
+ db.execute("PRAGMA synchronous=NORMAL")
2261
+ db.execute("PRAGMA cache_size=-65536")
2262
+ db.execute("PRAGMA user_version=1")
2263
+ checkpoint = BuildCheckpoint(
2264
+ db,
2265
+ {
2266
+ "checkpoint_version": 1,
2267
+ "base_url": base,
2268
+ "release": release,
2269
+ "species": args.species,
2270
+ "core_database": core,
2271
+ "assembly": ASSEMBLY,
2272
+ "schema_sha256": source_rows[-1][1],
2273
+ "sequence_chunk_size": CHUNK_SIZE,
2274
+ "tables": TABLES,
2275
+ "sequence_types": SEQUENCE_TYPES,
2276
+ },
2277
+ )
2278
+
2279
+ def reuse(stage: str) -> int | None:
2280
+ """Reuse a completed import and its recorded source provenance."""
2281
+ saved = checkpoint.read(stage)
2282
+ if saved and saved["complete"]:
2283
+ source = saved["source"]
2284
+ source_rows.append((source["url"], source["sha256"], source["bytes"]))
2285
+ LOG.info("Reusing checkpoint for %s: %s records", stage, f"{saved['count']:,}")
2286
+ return saved["count"]
2287
+ return None
2288
+
2289
+ def last_source() -> SourceRecord:
2290
+ """Describe the most recently verified download."""
2291
+ url, digest, size = source_rows[-1]
2292
+ return {"url": url, "sha256": digest, "bytes": size}
2293
+
2294
+ counts = {}
2295
+ panther_digest = None
2296
+ for table in TABLES:
2297
+ with steps.step(f"Download, import and index {table}"):
2298
+ count = reuse("table:" + table)
2299
+ if count is None:
2300
+ # A failed import may have created its table but has no
2301
+ # completion marker. Recreate only that unfinished table.
2302
+ db.execute(f"DROP TABLE IF EXISTS {sql_name('ensembl_' + table)}")
2303
+ count = import_table(
2304
+ db, table, schema[table], get(core_url + table + ".txt.gz")
2305
+ )
2306
+ checkpoint.save("table:" + table, source=last_source(), count=count)
2307
+ counts[table] = count
2308
+ if table == "analysis" and args.panther_classifications is None:
2309
+ version = panther_release(protein_analyses(db))
2310
+ if version:
2311
+ LOG.info(
2312
+ "Check/download PANTHER %s metadata before importing genome sequences",
2313
+ version,
2314
+ )
2315
+ _, panther_digest, _ = download_panther_classifications(
2316
+ version, cache / "metadata" / "panther"
2317
+ )
2318
+ # Guard against accidentally mixing source database releases.
2319
+ versions = {
2320
+ str(row[0])
2321
+ for row in db.execute(
2322
+ "SELECT meta_value FROM ensembl_meta WHERE meta_key='schema_version'"
2323
+ )
2324
+ }
2325
+ if versions and versions != {str(release)}:
2326
+ raise ValueError(f"Core schema_version {versions} does not match release {release}")
2327
+ if not checkpoint.read("sequence_tables"):
2328
+ # SQLite executes CREATE TABLE outside implicit DML transactions;
2329
+ # recover a process killed partway through schema creation.
2330
+ db.execute("DROP TABLE IF EXISTS sequence_chunks")
2331
+ db.execute("DROP TABLE IF EXISTS sequences")
2332
+ create_sequence_tables(db)
2333
+ checkpoint.save("sequence_tables")
2334
+ for kind in SEQUENCE_TYPES:
2335
+ with steps.step(f"Download and import {kind} FASTA sequences"):
2336
+ count = reuse("fasta:" + kind)
2337
+ if count is not None:
2338
+ counts[kind + "_sequences"] = count
2339
+ continue
2340
+ directory = root + f"fasta/{args.species}/{kind}/"
2341
+ suffix = ".dna.toplevel.fa.gz" if kind == "dna" else f".{kind}.all.fa.gz"
2342
+ candidates = sorted(
2343
+ name for name in listing(directory) if name.endswith(suffix)
2344
+ )
2345
+ if len(candidates) != 1:
2346
+ raise ValueError(
2347
+ f"Expected one {suffix} file at {directory}; found {candidates}"
2348
+ )
2349
+ path = get(directory + candidates[0], kind)
2350
+ source = last_source()
2351
+ saved = checkpoint.read("fasta:" + kind)
2352
+ if saved is None or saved["source"] != source:
2353
+ # Never reuse records after the source bytes change.
2354
+ db.execute("DELETE FROM sequence_chunks WHERE kind=?", (kind,))
2355
+ db.execute("DELETE FROM sequences WHERE kind=?", (kind,))
2356
+ checkpoint.save("fasta:" + kind, source=source, complete=False)
2357
+ count = import_fasta(db, kind, path, resume=True)
2358
+ counts[kind + "_sequences"] = count
2359
+ if kind == "dna":
2360
+ validate_dna_regions(db)
2361
+ checkpoint.save("fasta:" + kind, source=source, count=count)
2362
+ db.execute(
2363
+ "CREATE TABLE IF NOT EXISTS build_metadata (key TEXT PRIMARY KEY, value TEXT NOT NULL)"
2364
+ )
2365
+ metadata = {
2366
+ "format_version": "1",
2367
+ "release": str(release),
2368
+ "species": args.species,
2369
+ "core_database": core,
2370
+ "assembly": ASSEMBLY,
2371
+ "created_utc": datetime.now(timezone.utc).isoformat(),
2372
+ "sequence_chunk_size": str(CHUNK_SIZE),
2373
+ "sequence_codec": "zlib",
2374
+ "tables": json.dumps(TABLES),
2375
+ "sequence_types": json.dumps(SEQUENCE_TYPES),
2376
+ "counts": json.dumps(counts),
2377
+ }
2378
+ db.executemany("INSERT OR REPLACE INTO build_metadata VALUES (?, ?)", metadata.items())
2379
+ db.execute(
2380
+ "CREATE TABLE IF NOT EXISTS source_files (url TEXT PRIMARY KEY, sha256 TEXT, bytes INTEGER)"
2381
+ )
2382
+ db.executemany("INSERT OR REPLACE INTO source_files VALUES (?, ?, ?)", source_rows)
2383
+ db.commit()
2384
+ with steps.step("Preprocess transcripts, domains and splice sites"):
2385
+ # If preprocessing completed before a later failure, preserve it
2386
+ # unless its metadata input or implementation has changed.
2387
+ fingerprint = json.dumps(
2388
+ {
2389
+ "implementation": sha256_file(Path(__file__)),
2390
+ "uniprot_implementation": sha256_file(
2391
+ Path(__file__).with_name("uniprot.py")
2392
+ ),
2393
+ "ensembl_implementation": sha256_file(
2394
+ Path(__file__).with_name("ensembl.py")
2395
+ ),
2396
+ "uniprot_sha256": sha256_file(uniprot_features),
2397
+ "interpro_sha256": sha256_file(interpro_entries),
2398
+ "fetch_interpro": fetch_interpro,
2399
+ "panther_sha256": (
2400
+ sha256_file(args.panther_classifications)
2401
+ if args.panther_classifications
2402
+ else panther_digest
2403
+ ),
2404
+ },
2405
+ sort_keys=True,
2406
+ )
2407
+ saved = checkpoint.read("preprocessing")
2408
+ if saved and saved["complete"] and saved["fingerprint"] == fingerprint:
2409
+ LOG.info("Reusing completed transcript/domain preprocessing")
2410
+ else:
2411
+ db.execute("DELETE FROM build_checkpoints WHERE stage='analyze'")
2412
+ checkpoint.save("preprocessing", fingerprint=fingerprint, complete=False)
2413
+ preprocess_reference(
2414
+ db,
2415
+ interpro_entries,
2416
+ interpro_archive_cache=interpro_cache if fetch_interpro else None,
2417
+ panther_classifications=args.panther_classifications,
2418
+ panther_cache=cache / "metadata" / "panther",
2419
+ uniprot_features=uniprot_features,
2420
+ uniprot_source=uniprot_source,
2421
+ )
2422
+ checkpoint.save("preprocessing", fingerprint=fingerprint)
2423
+ with steps.step("Analyze database indexes"):
2424
+ if checkpoint.read("analyze"):
2425
+ LOG.info("Reusing completed index analysis")
2426
+ else:
2427
+ sqlite_phase(db, "ANALYZE", "Analyze database")
2428
+ checkpoint.save("analyze")
2429
+ with steps.step("Validate SQLite database integrity"):
2430
+ result = sqlite_phase(db, "PRAGMA integrity_check", "Check database")
2431
+ if result != [("ok",)]:
2432
+ raise ValueError(f"SQLite integrity check failed: {result[:5]}")
2433
+ db.commit()
2434
+ # Close SQLite before publication, while still holding the build
2435
+ # lock. This also permits atomic replacement on Windows.
2436
+ db.close()
2437
+ with steps.step("Publish completed reference database"):
2438
+ if output.exists() and not args.force:
2439
+ raise FileExistsError(f"Output appeared during the build: {output}")
2440
+ os.replace(temporary, output)
2441
+ except BaseException:
2442
+ if temporary.exists():
2443
+ LOG.error(
2444
+ "Build checkpoint preserved at %s; rerun the same command to resume", temporary
2445
+ )
2446
+ raise
2447
+ LOG.info("Created %s (%.1f MiB)", output, output.stat().st_size / CHUNK_SIZE)
2448
+ steps.complete()
2449
+ return output
2450
+
2451
+
2452
+ def main(argv: Sequence[str] | None = None) -> int:
2453
+ """Validate CLI options, then run a full build or derived-table regeneration."""
2454
+ parser = argparse.ArgumentParser(
2455
+ prog="fusion-function prepare-data",
2456
+ description=__doc__,
2457
+ formatter_class=argparse.RawDescriptionHelpFormatter,
2458
+ )
2459
+ parser.add_argument(
2460
+ "--release", type=int, help="Ensembl release; default: latest numbered FTP release"
2461
+ )
2462
+ parser.add_argument(
2463
+ "--species",
2464
+ choices=[SPECIES],
2465
+ help="Species component of the output path; this script supports human data",
2466
+ )
2467
+ parser.add_argument(
2468
+ "--output", type=Path, help="SQLite file; default: release-specific file in cache"
2469
+ )
2470
+ parser.add_argument(
2471
+ "--cache-dir",
2472
+ type=Path,
2473
+ default=default_cache_dir(),
2474
+ help="Download/cache directory; defaults to FUSION_FUNCTION_CACHEDIR or OS user cache",
2475
+ )
2476
+ parser.add_argument("--base-url", help="Advanced: FTP HTTP(S) mirror root, ending at /pub/")
2477
+ parser.add_argument(
2478
+ "--from-source",
2479
+ action="store_true",
2480
+ default=None,
2481
+ help="Build from Ensembl FTP instead of downloading a compatible prebuilt reference",
2482
+ )
2483
+ parser.add_argument(
2484
+ "--reference-catalog",
2485
+ help="Prebuilt catalog: local JSON or HTTPS URL; default: maintained catalog",
2486
+ )
2487
+ parser.add_argument(
2488
+ "--force",
2489
+ action="store_true",
2490
+ default=None,
2491
+ help="Replace output after successful installation or build",
2492
+ )
2493
+ parser.add_argument(
2494
+ "--preprocess-only",
2495
+ type=Path,
2496
+ metavar="DB",
2497
+ help="Regenerate derived tables in a full source-built database",
2498
+ )
2499
+ parser.add_argument(
2500
+ "--interpro-entries",
2501
+ type=Path,
2502
+ metavar="TSV",
2503
+ help="Optional local InterPro entry.list TSV for canonical entry names/types",
2504
+ )
2505
+ parser.add_argument(
2506
+ "--panther-classifications",
2507
+ type=Path,
2508
+ metavar="TSV",
2509
+ help="Optional local PANTHER human classification TSV; default: Ensembl's PANTHER release",
2510
+ )
2511
+ parser.add_argument(
2512
+ "--uniprot-features",
2513
+ type=Path,
2514
+ metavar="XML",
2515
+ help="Local UniProt XML (optionally gzip); default: reviewed human bulk download",
2516
+ )
2517
+ args = parser.parse_args(argv)
2518
+ if args.release is not None and args.release <= 0:
2519
+ parser.error("--release must be positive")
2520
+ if args.preprocess_only:
2521
+ # Reprocessing uses the database's source release/assembly, so new-build
2522
+ # options cannot change where its existing coordinates came from.
2523
+ incompatible = [
2524
+ "--" + name.replace("_", "-")
2525
+ for name in (
2526
+ "release",
2527
+ "species",
2528
+ "output",
2529
+ "base_url",
2530
+ "force",
2531
+ "from_source",
2532
+ "reference_catalog",
2533
+ )
2534
+ if getattr(args, name) is not None
2535
+ ]
2536
+ if incompatible:
2537
+ parser.error("--preprocess-only cannot be combined with " + ", ".join(incompatible))
2538
+ source_options = args.from_source or any(
2539
+ getattr(args, name) is not None
2540
+ for name in ("base_url", "interpro_entries", "panther_classifications", "uniprot_features")
2541
+ )
2542
+ if args.reference_catalog and source_options:
2543
+ parser.error("--reference-catalog cannot be combined with source-build options")
2544
+ if args.interpro_entries:
2545
+ args.interpro_entries = args.interpro_entries.expanduser().resolve()
2546
+ if not args.interpro_entries.is_file():
2547
+ parser.error(f"InterPro entry file does not exist: {args.interpro_entries}")
2548
+ if args.panther_classifications:
2549
+ args.panther_classifications = args.panther_classifications.expanduser().resolve()
2550
+ if not args.panther_classifications.is_file():
2551
+ parser.error(
2552
+ f"PANTHER classification file does not exist: {args.panther_classifications}"
2553
+ )
2554
+ if args.uniprot_features:
2555
+ args.uniprot_features = args.uniprot_features.expanduser().resolve()
2556
+ if not args.uniprot_features.is_file():
2557
+ parser.error(f"UniProt feature file does not exist: {args.uniprot_features}")
2558
+ args.species = args.species or SPECIES
2559
+ args.base_url = args.base_url or BASE_URL
2560
+ args.force = bool(args.force)
2561
+ parsed_base = urllib.parse.urlparse(args.base_url)
2562
+ if (
2563
+ parsed_base.scheme not in {"http", "https"}
2564
+ or not parsed_base.netloc
2565
+ or parsed_base.query
2566
+ or parsed_base.fragment
2567
+ ):
2568
+ parser.error("--base-url must be an HTTP(S) mirror URL without a query or fragment")
2569
+ args.cache_dir = args.cache_dir.expanduser().resolve()
2570
+ logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
2571
+ LOG.info("Cache root: %s", args.cache_dir)
2572
+ LOG.info("Ensembl download root: %s", args.cache_dir / "downloads")
2573
+ LOG.info("InterPro cache: %s", args.cache_dir / "metadata" / "interpro")
2574
+ LOG.info(
2575
+ "Historical InterPro entry-list cache: %s",
2576
+ args.cache_dir / "metadata" / "interpro" / "releases",
2577
+ )
2578
+ LOG.info("UniProt feature cache: %s", args.cache_dir / "metadata" / "uniprot")
2579
+ LOG.info("PANTHER human classification cache: %s", args.cache_dir / "metadata" / "panther")
2580
+ LOG.info("Prebuilt download cache: %s", args.cache_dir / "prebuilt")
2581
+ LOG.info("Source partial downloads: <filename>.<random>.part beside cached files")
2582
+ LOG.info("Prebuilt partial downloads: <cache>/prebuilt/<sha256>/ensembl.sqlite.gz.part")
2583
+ if args.interpro_entries:
2584
+ LOG.info("Local InterPro entries: %s", args.interpro_entries)
2585
+ if args.panther_classifications:
2586
+ LOG.info("Local PANTHER classifications: %s", args.panther_classifications)
2587
+ with logging_redirect_tqdm():
2588
+ try:
2589
+ if args.preprocess_only:
2590
+ output = args.preprocess_only.expanduser().resolve()
2591
+ LOG.info("Reprocessing existing reference in place: %s", output)
2592
+ LOG.info(
2593
+ "SQLite transaction files, if used: %s-journal; %s-wal; %s-shm",
2594
+ output,
2595
+ output,
2596
+ output,
2597
+ )
2598
+ steps = BuildSteps(4)
2599
+ with (
2600
+ build_lock(output.with_name(output.name + ".prepare.lock")),
2601
+ closing(sqlite3.connect(output.as_uri() + "?mode=rw", uri=True)) as db,
2602
+ ):
2603
+ with steps.step("Validate existing reference database"):
2604
+ validate_reference_source(db)
2605
+ with steps.step("Prepare InterPro metadata"):
2606
+ if args.interpro_entries is None:
2607
+ archive_cache = args.cache_dir / "metadata" / "interpro"
2608
+ interpro_entries, _ = download(INTERPRO_ENTRIES_URL, archive_cache)
2609
+ else:
2610
+ interpro_entries = args.interpro_entries
2611
+ archive_cache = None
2612
+ LOG.info("Using local InterPro entries: %s", interpro_entries)
2613
+ with steps.step("Prepare reviewed human UniProt features"):
2614
+ from .uniprot import UNIPROT_HUMAN_URL
2615
+
2616
+ if args.uniprot_features is None:
2617
+ uniprot_features, digest = download(
2618
+ UNIPROT_HUMAN_URL, args.cache_dir / "metadata" / "uniprot"
2619
+ )
2620
+ uniprot_source = {"url": UNIPROT_HUMAN_URL, "sha256": digest}
2621
+ else:
2622
+ uniprot_features = args.uniprot_features
2623
+ uniprot_source = {
2624
+ "path": str(uniprot_features),
2625
+ "sha256": sha256_file(uniprot_features),
2626
+ }
2627
+ LOG.info("Using local UniProt features: %s", uniprot_features)
2628
+ with steps.step("Preprocess transcripts, domains and splice sites"):
2629
+ preprocess_reference(
2630
+ db,
2631
+ interpro_entries,
2632
+ interpro_archive_cache=archive_cache,
2633
+ panther_classifications=args.panther_classifications,
2634
+ panther_cache=args.cache_dir / "metadata" / "panther",
2635
+ uniprot_features=uniprot_features,
2636
+ uniprot_source=uniprot_source,
2637
+ )
2638
+ steps.complete()
2639
+ else:
2640
+ LOG.info(
2641
+ "Reference destination: %s",
2642
+ args.output.expanduser().resolve()
2643
+ if args.output
2644
+ else args.cache_dir
2645
+ / args.species
2646
+ / ASSEMBLY
2647
+ / f"release-{args.release or '<latest>'}"
2648
+ / "ensembl.sqlite",
2649
+ )
2650
+ output = None
2651
+ if not source_options:
2652
+ from .prebuilt import install_reference
2653
+
2654
+ output = install_reference(
2655
+ release=args.release,
2656
+ cache_dir=args.cache_dir,
2657
+ output=args.output,
2658
+ force=args.force,
2659
+ catalog=args.reference_catalog,
2660
+ )
2661
+ else:
2662
+ LOG.info("Source-build options selected; skipping the prebuilt catalog")
2663
+ if output is None:
2664
+ output = build(args)
2665
+ except (
2666
+ OSError,
2667
+ ValueError,
2668
+ sqlite3.Error,
2669
+ urllib.error.URLError,
2670
+ EOFError,
2671
+ ParseError,
2672
+ ) as exc:
2673
+ LOG.error("Build failed: %s", exc)
2674
+ return 1
2675
+ except KeyboardInterrupt:
2676
+ LOG.error("Interrupted; no incomplete database was published")
2677
+ return 130
2678
+ print(output)
2679
+ return 0
2680
+
2681
+
2682
+ if __name__ == "__main__":
2683
+ raise SystemExit(main())