fusion-function 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fusion_function/__init__.py +9 -0
- fusion_function/__main__.py +3 -0
- fusion_function/cli.py +27 -0
- fusion_function/data.py +2683 -0
- fusion_function/ensembl.py +149 -0
- fusion_function/fusion.py +1024 -0
- fusion_function/interpro.py +29 -0
- fusion_function/prebuilt.py +518 -0
- fusion_function/reference.py +112 -0
- fusion_function/reference_catalog.json +82 -0
- fusion_function/uniprot.py +315 -0
- fusion_function-0.2.1.dist-info/METADATA +98 -0
- fusion_function-0.2.1.dist-info/RECORD +16 -0
- fusion_function-0.2.1.dist-info/WHEEL +4 -0
- fusion_function-0.2.1.dist-info/entry_points.txt +3 -0
- fusion_function-0.2.1.dist-info/licenses/LICENSE +674 -0
fusion_function/data.py
ADDED
|
@@ -0,0 +1,2683 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Build and read the human GRCh38 reference used by fusion-function.
|
|
3
|
+
|
|
4
|
+
Commands (Python 3.11+, tqdm is the only external dependency):
|
|
5
|
+
fusion-function prepare-data # latest Ensembl release
|
|
6
|
+
fusion-function prepare-data --release 116 # pinned release
|
|
7
|
+
fusion-function prepare-data --preprocess-only DB # regenerate derived data
|
|
8
|
+
|
|
9
|
+
Default output: <cache>/homo_sapiens/GRCh38/release-<release>/ensembl.sqlite.
|
|
10
|
+
Set FUSION_FUNCTION_CACHEDIR or --cache-dir to change the cache root.
|
|
11
|
+
|
|
12
|
+
Storage: ensembl_* tables preserve the FTP dump columns. DNA and peptides use
|
|
13
|
+
independently compressed 1 MiB chunks. ff_transcripts stores prepared exon/CDS
|
|
14
|
+
coordinates, splice windows and protein features; ff_interpro stores names and
|
|
15
|
+
types. Transcript models use compressed JSON; error messages remain plain JSON.
|
|
16
|
+
ff_uniprot_features stores sequence-verified reviewed human sites,
|
|
17
|
+
domains and motifs. Coordinates are 1-based and inclusive; CDS blocks follow transcript
|
|
18
|
+
orientation, including on the negative strand. Pre-mRNA is reconstructed when
|
|
19
|
+
requested, rather than stored for every transcript.
|
|
20
|
+
|
|
21
|
+
Preparation downloads FTP files over HTTPS and verifies cached files by SHA-256.
|
|
22
|
+
InterPro metadata is fetched by default, with archived entry lists supplementing
|
|
23
|
+
missing accessions. --interpro-entries FILE supplies local metadata instead.
|
|
24
|
+
Archive checksums and source releases are recorded in build_metadata.
|
|
25
|
+
Human PANTHER subfamily classifications are fetched for the version recorded
|
|
26
|
+
by Ensembl, or supplied with --panther-classifications FILE. Protein cross-
|
|
27
|
+
references enrich existing family hits; no domain coordinates are invented.
|
|
28
|
+
Reviewed human UniProt XML is fetched during preparation only, or supplied with
|
|
29
|
+
--uniprot-features FILE. Features require exact protein sequence identity and
|
|
30
|
+
carry UniProt accessions, isoforms and evidence codes.
|
|
31
|
+
|
|
32
|
+
New databases resume from <output>.building and are published only after
|
|
33
|
+
integrity checking; --force permits replacement. Repeating the same build skips
|
|
34
|
+
completed imports and reuses committed FASTA records from unchanged inputs.
|
|
35
|
+
Reprocessing updates derived tables in one transaction and rolls
|
|
36
|
+
back on failure. Numbered steps and terminal progress bars show build progress.
|
|
37
|
+
Runtime readers are local and read-only; no REST API, MySQL server or pysam is
|
|
38
|
+
required. See README.md and docs/reference.md for configuration details.
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
from __future__ import annotations
|
|
42
|
+
|
|
43
|
+
import argparse
|
|
44
|
+
import gzip
|
|
45
|
+
import hashlib
|
|
46
|
+
import io
|
|
47
|
+
import json
|
|
48
|
+
import logging
|
|
49
|
+
import os
|
|
50
|
+
import re
|
|
51
|
+
import sqlite3
|
|
52
|
+
import sys
|
|
53
|
+
import tempfile
|
|
54
|
+
import time
|
|
55
|
+
import urllib.error
|
|
56
|
+
import urllib.parse
|
|
57
|
+
import urllib.request
|
|
58
|
+
import zlib
|
|
59
|
+
from bisect import bisect_left
|
|
60
|
+
from collections import OrderedDict
|
|
61
|
+
from collections.abc import Callable, Iterable, Iterator, Mapping, Sequence
|
|
62
|
+
from contextlib import AbstractContextManager, ExitStack, closing, contextmanager
|
|
63
|
+
from contextvars import ContextVar
|
|
64
|
+
from datetime import datetime, timezone
|
|
65
|
+
from html.parser import HTMLParser
|
|
66
|
+
from itertools import groupby
|
|
67
|
+
from pathlib import Path
|
|
68
|
+
from types import TracebackType
|
|
69
|
+
from typing import TYPE_CHECKING, Any, Literal, Self, TextIO, TypedDict, cast, overload
|
|
70
|
+
from xml.etree.ElementTree import ParseError
|
|
71
|
+
|
|
72
|
+
from tqdm import tqdm
|
|
73
|
+
from tqdm.contrib.logging import logging_redirect_tqdm
|
|
74
|
+
|
|
75
|
+
if TYPE_CHECKING:
|
|
76
|
+
from http.client import HTTPResponse
|
|
77
|
+
|
|
78
|
+
from .ensembl import (
|
|
79
|
+
CDSBlock,
|
|
80
|
+
EnsemblError,
|
|
81
|
+
GenomicSegment,
|
|
82
|
+
ProteinFeatureAnnotation,
|
|
83
|
+
ProteinFeatureResponse,
|
|
84
|
+
TranscriptProteinFeatureResult,
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
# Raw SQL rows have different columns per source table; derived structures use
|
|
89
|
+
# the same typed records as the annotation API.
|
|
90
|
+
DatabaseRow = dict[str, Any]
|
|
91
|
+
AnalysisMetadata = dict[int, tuple[str | None, str | None]]
|
|
92
|
+
InterProMetadata = dict[str, tuple[str | None, str | None]]
|
|
93
|
+
ChunkCache = OrderedDict[tuple[str, str, int], bytes]
|
|
94
|
+
STRUCTURE_SOURCES = frozenset({"sifts", "alphafold"})
|
|
95
|
+
TRANSCRIPT_PAYLOAD_CODEC = "zlib-json-v1"
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
class SourceRecord(TypedDict):
|
|
99
|
+
url: str
|
|
100
|
+
sha256: str
|
|
101
|
+
bytes: int
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
class TranscriptErrorSummary(TypedDict):
|
|
105
|
+
count: int
|
|
106
|
+
example_transcripts: list[str]
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
class CheckpointRecord(TypedDict):
|
|
110
|
+
source: dict[str, Any]
|
|
111
|
+
fingerprint: str
|
|
112
|
+
count: int
|
|
113
|
+
complete: bool
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
BASE_URL = "https://ftp.ensembl.org/pub/"
|
|
117
|
+
INTERPRO_ENTRIES_URL = "https://ftp.ebi.ac.uk/pub/databases/interpro/current_release/entry.list"
|
|
118
|
+
INTERPRO_RELEASES_URL = "https://ftp.ebi.ac.uk/pub/databases/interpro/releases/"
|
|
119
|
+
PANTHER_CLASSIFICATIONS_URL = "https://data.pantherdb.org/ftp/sequence_classifications/"
|
|
120
|
+
# These inputs are fixed: omitting a table or sequence type can break preprocessing.
|
|
121
|
+
TABLES = (
|
|
122
|
+
"meta",
|
|
123
|
+
"coord_system",
|
|
124
|
+
"seq_region",
|
|
125
|
+
"gene",
|
|
126
|
+
"transcript",
|
|
127
|
+
"exon",
|
|
128
|
+
"exon_transcript",
|
|
129
|
+
"translation",
|
|
130
|
+
"analysis",
|
|
131
|
+
"protein_feature",
|
|
132
|
+
"interpro",
|
|
133
|
+
"xref",
|
|
134
|
+
"external_db",
|
|
135
|
+
"object_xref",
|
|
136
|
+
)
|
|
137
|
+
PREPROCESS_TABLES = (
|
|
138
|
+
"transcript",
|
|
139
|
+
"translation",
|
|
140
|
+
"exon",
|
|
141
|
+
"exon_transcript",
|
|
142
|
+
"seq_region",
|
|
143
|
+
"coord_system",
|
|
144
|
+
"protein_feature",
|
|
145
|
+
"analysis",
|
|
146
|
+
"interpro",
|
|
147
|
+
"xref",
|
|
148
|
+
)
|
|
149
|
+
SEQUENCE_TYPES = ("dna", "pep")
|
|
150
|
+
SPECIES = "homo_sapiens"
|
|
151
|
+
ASSEMBLY = "GRCh38"
|
|
152
|
+
CHUNK_SIZE = 1024 * 1024
|
|
153
|
+
LOG = logging.getLogger("ensembl_to_sqlite")
|
|
154
|
+
STEP_PREFIX = ContextVar("fusion_function_prepare_step", default="")
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
# Progress reporting and cache configuration.
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
class BuildSteps:
|
|
161
|
+
"""Label the fixed build stages; their durations differ, so no total ETA."""
|
|
162
|
+
|
|
163
|
+
def __init__(self, total: int) -> None:
|
|
164
|
+
"""Track the expected stage count and total elapsed build time."""
|
|
165
|
+
self.total = total
|
|
166
|
+
self.current = 0
|
|
167
|
+
self.started = time.monotonic()
|
|
168
|
+
|
|
169
|
+
@contextmanager
|
|
170
|
+
def step(self, description: str) -> Iterator[None]:
|
|
171
|
+
"""Label a stage and its nested bars; restore the label even on failure."""
|
|
172
|
+
self.current += 1
|
|
173
|
+
if self.current > self.total:
|
|
174
|
+
raise ValueError("Preparation step count exceeds its declared total")
|
|
175
|
+
LOG.info("[Step %s/%s] %s", self.current, self.total, description)
|
|
176
|
+
token = STEP_PREFIX.set(f"[{self.current}/{self.total}] ")
|
|
177
|
+
try:
|
|
178
|
+
yield
|
|
179
|
+
finally:
|
|
180
|
+
STEP_PREFIX.reset(token)
|
|
181
|
+
|
|
182
|
+
def complete(self) -> None:
|
|
183
|
+
"""Check that every declared stage ran, then log the total duration."""
|
|
184
|
+
if self.current != self.total:
|
|
185
|
+
raise ValueError("Preparation completed fewer steps than expected")
|
|
186
|
+
LOG.info(
|
|
187
|
+
"Completed all %s steps in %s",
|
|
188
|
+
self.total,
|
|
189
|
+
tqdm.format_interval(time.monotonic() - self.started),
|
|
190
|
+
)
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def progress(
|
|
194
|
+
iterable: Iterable[Any] | None = None,
|
|
195
|
+
*,
|
|
196
|
+
total: int | None = None,
|
|
197
|
+
desc: str = "",
|
|
198
|
+
unit: str = "it",
|
|
199
|
+
unit_scale: bool = False,
|
|
200
|
+
unit_divisor: int = 1000,
|
|
201
|
+
disable: bool | None = None,
|
|
202
|
+
file: TextIO | None = None,
|
|
203
|
+
) -> tqdm:
|
|
204
|
+
"""Use compact, rate-limited bars; suppress terminal redraws in log files."""
|
|
205
|
+
long_process_notice(desc)
|
|
206
|
+
return tqdm(
|
|
207
|
+
iterable,
|
|
208
|
+
total=total,
|
|
209
|
+
desc=STEP_PREFIX.get() + desc,
|
|
210
|
+
unit=unit,
|
|
211
|
+
unit_scale=unit_scale,
|
|
212
|
+
unit_divisor=unit_divisor,
|
|
213
|
+
disable=disable,
|
|
214
|
+
file=file,
|
|
215
|
+
dynamic_ncols=True,
|
|
216
|
+
mininterval=0.5,
|
|
217
|
+
)
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def file_label(path: Path) -> str:
|
|
221
|
+
"""Short labels keep progress output on one terminal line."""
|
|
222
|
+
if path.name.endswith(".sql.gz"):
|
|
223
|
+
return "core schema"
|
|
224
|
+
if path.name == "stream":
|
|
225
|
+
return "UniProt features"
|
|
226
|
+
if path.name == "entry.list":
|
|
227
|
+
return "InterPro"
|
|
228
|
+
for kind in SEQUENCE_TYPES:
|
|
229
|
+
if f".{kind}." in path.name:
|
|
230
|
+
return f"{kind} FASTA"
|
|
231
|
+
return path.name.removesuffix(".txt.gz")[:24]
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
@overload
|
|
235
|
+
def input_progress(
|
|
236
|
+
path: Path, description: str, *, compressed: bool = True, text: Literal[True]
|
|
237
|
+
) -> AbstractContextManager[tuple[io.TextIOWrapper, tqdm, Callable[[], None]]]: ...
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
@overload
|
|
241
|
+
def input_progress(
|
|
242
|
+
path: Path, description: str, *, compressed: bool = True, text: Literal[False] = False
|
|
243
|
+
) -> AbstractContextManager[tuple[gzip.GzipFile | io.BufferedReader, tqdm, Callable[[], None]]]: ...
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
@overload
|
|
247
|
+
def input_progress(
|
|
248
|
+
path: Path, description: str, *, compressed: bool = True, text: bool
|
|
249
|
+
) -> AbstractContextManager[
|
|
250
|
+
tuple[io.TextIOWrapper | gzip.GzipFile | io.BufferedReader, tqdm, Callable[[], None]]
|
|
251
|
+
]: ...
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
@contextmanager
|
|
255
|
+
def input_progress(
|
|
256
|
+
path: Path, description: str, *, compressed: bool = True, text: bool = False
|
|
257
|
+
) -> Iterator[tuple[Any, tqdm, Callable[[], None]]]:
|
|
258
|
+
"""Track source bytes during parsing, with no pre-count or gzip size guess."""
|
|
259
|
+
with ExitStack() as stack:
|
|
260
|
+
raw = stack.enter_context(path.open("rb"))
|
|
261
|
+
source = stack.enter_context(gzip.GzipFile(fileobj=raw)) if compressed else raw
|
|
262
|
+
stream = (
|
|
263
|
+
stack.enter_context(io.TextIOWrapper(source, encoding="utf-8", newline=""))
|
|
264
|
+
if text
|
|
265
|
+
else source
|
|
266
|
+
)
|
|
267
|
+
bar = stack.enter_context(
|
|
268
|
+
progress(
|
|
269
|
+
total=path.stat().st_size,
|
|
270
|
+
desc=description,
|
|
271
|
+
unit="B",
|
|
272
|
+
unit_scale=True,
|
|
273
|
+
unit_divisor=1024,
|
|
274
|
+
)
|
|
275
|
+
)
|
|
276
|
+
position = 0
|
|
277
|
+
|
|
278
|
+
def update() -> None:
|
|
279
|
+
"""Advance by source bytes read, including gzip buffering."""
|
|
280
|
+
nonlocal position
|
|
281
|
+
consumed = raw.tell()
|
|
282
|
+
bar.update(consumed - position)
|
|
283
|
+
position = consumed
|
|
284
|
+
|
|
285
|
+
yield stream, bar, update
|
|
286
|
+
update()
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
def long_process_notice(description: str) -> None:
|
|
290
|
+
"""Print one advance notice for work that can exceed ten minutes.
|
|
291
|
+
|
|
292
|
+
These are possibilities for large human references, not completion estimates.
|
|
293
|
+
Emit notices only when work starts, so cached or checkpointed stages stay quiet.
|
|
294
|
+
"""
|
|
295
|
+
if description == "Drop previous ff_transcripts":
|
|
296
|
+
note = "Removing old transcript models can take an hour or longer on some systems."
|
|
297
|
+
elif description == "Preprocess transcripts":
|
|
298
|
+
note = "Preparing all human transcripts can take an hour or longer."
|
|
299
|
+
elif description == "Check database":
|
|
300
|
+
note = "The full database integrity check can take over 10 minutes."
|
|
301
|
+
elif (
|
|
302
|
+
description.startswith(("Download ", "Import ", "Index "))
|
|
303
|
+
and description not in {"Import InterPro", "Import PANTHER", "Index analysis", "Index meta"}
|
|
304
|
+
) or description in {
|
|
305
|
+
"Analyze database",
|
|
306
|
+
"Read UniProt protein cross-references",
|
|
307
|
+
"Read genome and protein sequence lengths",
|
|
308
|
+
"Prepare ordered exon stream",
|
|
309
|
+
"Prepare ordered protein feature stream",
|
|
310
|
+
"Prepare transcript stream",
|
|
311
|
+
"Commit preprocessed reference",
|
|
312
|
+
"Roll back preprocessing transaction",
|
|
313
|
+
}:
|
|
314
|
+
note = (
|
|
315
|
+
"This operation can take over 10 minutes, depending on reference size and system load."
|
|
316
|
+
)
|
|
317
|
+
else:
|
|
318
|
+
return
|
|
319
|
+
LOG.info("%s%s: %s", STEP_PREFIX.get(), description, note)
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
@contextmanager
|
|
323
|
+
def sqlite_activity(description: str) -> Iterator[None]:
|
|
324
|
+
"""Log phase boundaries without background output or periodic redraws."""
|
|
325
|
+
label = STEP_PREFIX.get() + description
|
|
326
|
+
LOG.info("%s", label)
|
|
327
|
+
long_process_notice(description)
|
|
328
|
+
outcome = "failed"
|
|
329
|
+
try:
|
|
330
|
+
yield
|
|
331
|
+
outcome = "done"
|
|
332
|
+
except KeyboardInterrupt:
|
|
333
|
+
outcome = "interrupted"
|
|
334
|
+
raise
|
|
335
|
+
finally:
|
|
336
|
+
LOG.info("%s: %s", label, outcome)
|
|
337
|
+
|
|
338
|
+
|
|
339
|
+
def sqlite_phase(db: sqlite3.Connection, sql: str, description: str) -> list[tuple[Any, ...]]:
|
|
340
|
+
"""Run one SQLite statement with start/completion logs and signal handling."""
|
|
341
|
+
with sqlite_activity(description):
|
|
342
|
+
|
|
343
|
+
def checkpoint() -> int:
|
|
344
|
+
"""Keep Python signal handling responsive during long SQLite statements."""
|
|
345
|
+
return 0
|
|
346
|
+
|
|
347
|
+
db.set_progress_handler(checkpoint, 100000)
|
|
348
|
+
try:
|
|
349
|
+
return db.execute(sql).fetchall()
|
|
350
|
+
finally:
|
|
351
|
+
db.set_progress_handler(None, 0)
|
|
352
|
+
|
|
353
|
+
|
|
354
|
+
def default_cache_dir() -> Path:
|
|
355
|
+
"""Use the package override, then the platform's conventional cache root."""
|
|
356
|
+
if value := os.environ.get("FUSION_FUNCTION_CACHEDIR"):
|
|
357
|
+
return Path(value).expanduser()
|
|
358
|
+
if sys.platform == "win32":
|
|
359
|
+
root = Path(os.environ.get("LOCALAPPDATA", Path.home() / "AppData/Local"))
|
|
360
|
+
elif sys.platform == "darwin":
|
|
361
|
+
root = Path.home() / "Library/Caches"
|
|
362
|
+
else:
|
|
363
|
+
root = Path(os.environ.get("XDG_CACHE_HOME", Path.home() / ".cache"))
|
|
364
|
+
return root / "fusion_function"
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
# Source discovery, verified downloads and MySQL dump imports.
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
def open_url(url: str) -> HTTPResponse:
|
|
371
|
+
"""Open a source URL with an identifying user agent and a bounded timeout."""
|
|
372
|
+
request = urllib.request.Request(url, headers={"User-Agent": "ensembl-to-sqlite/1"})
|
|
373
|
+
return urllib.request.urlopen(request, timeout=60)
|
|
374
|
+
|
|
375
|
+
|
|
376
|
+
class Links(HTMLParser):
|
|
377
|
+
"""Collect filenames from the HTML directory listings used by FTP mirrors."""
|
|
378
|
+
|
|
379
|
+
def __init__(self) -> None:
|
|
380
|
+
"""Start with no names; repeated links are collapsed into a set."""
|
|
381
|
+
super().__init__()
|
|
382
|
+
self.names: set[str] = set()
|
|
383
|
+
|
|
384
|
+
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
|
|
385
|
+
"""Keep link basenames, excluding navigation to the current/parent path."""
|
|
386
|
+
if tag == "a":
|
|
387
|
+
href = dict(attrs).get("href", "")
|
|
388
|
+
path = urllib.parse.unquote(urllib.parse.urlparse(href).path)
|
|
389
|
+
name = path.rstrip("/").rsplit("/", 1)[-1]
|
|
390
|
+
if name not in {"", ".", ".."}:
|
|
391
|
+
self.names.add(name)
|
|
392
|
+
|
|
393
|
+
|
|
394
|
+
def listing(url: str) -> set[str]:
|
|
395
|
+
"""Return directory entry names without depending on HTML table formatting."""
|
|
396
|
+
with open_url(url) as response:
|
|
397
|
+
text = response.read().decode("utf-8")
|
|
398
|
+
parser = Links()
|
|
399
|
+
parser.feed(text)
|
|
400
|
+
return parser.names
|
|
401
|
+
|
|
402
|
+
|
|
403
|
+
def resolve_core(base: str, release: int | None) -> tuple[int, str, str]:
|
|
404
|
+
"""Find a published release and its human GRCh38 core directory.
|
|
405
|
+
|
|
406
|
+
Latest means the largest numbered release, not an unversioned FTP alias.
|
|
407
|
+
Choosing the exact assembly suffix avoids mixing GRCh37 and GRCh38 data.
|
|
408
|
+
"""
|
|
409
|
+
if release is None:
|
|
410
|
+
releases = [
|
|
411
|
+
int(m[1]) for name in listing(base) if (m := re.fullmatch(r"release-(\d+)", name))
|
|
412
|
+
]
|
|
413
|
+
if not releases:
|
|
414
|
+
raise ValueError("No numbered Ensembl releases found; supply --release")
|
|
415
|
+
release = max(releases)
|
|
416
|
+
root = f"{base}release-{release}/"
|
|
417
|
+
core_name = f"{SPECIES}_core_{release}_38"
|
|
418
|
+
if core_name not in listing(root + "mysql/"):
|
|
419
|
+
raise ValueError(f"Human GRCh38 core database {core_name} is absent from release {release}")
|
|
420
|
+
return release, root, core_name
|
|
421
|
+
|
|
422
|
+
|
|
423
|
+
def sha256_file(path: Path, *, show_progress: bool = False) -> str:
|
|
424
|
+
"""Hash a file in bounded reads, optionally displaying verification progress."""
|
|
425
|
+
digest = hashlib.sha256()
|
|
426
|
+
with (
|
|
427
|
+
path.open("rb") as stream,
|
|
428
|
+
progress(
|
|
429
|
+
total=path.stat().st_size,
|
|
430
|
+
desc="Verify " + file_label(path),
|
|
431
|
+
unit="B",
|
|
432
|
+
unit_scale=True,
|
|
433
|
+
unit_divisor=1024,
|
|
434
|
+
disable=None if show_progress else True,
|
|
435
|
+
) as bar,
|
|
436
|
+
):
|
|
437
|
+
while block := stream.read(4 * CHUNK_SIZE):
|
|
438
|
+
digest.update(block)
|
|
439
|
+
bar.update(len(block))
|
|
440
|
+
return digest.hexdigest()
|
|
441
|
+
|
|
442
|
+
|
|
443
|
+
def download(url: str, directory: Path) -> tuple[Path, str]:
|
|
444
|
+
"""Reuse verified cached bytes or atomically publish a complete download.
|
|
445
|
+
|
|
446
|
+
The caller supplies a directory that separates releases/source locations.
|
|
447
|
+
Recoverable failures are retried; partial files are removed on every exit.
|
|
448
|
+
"""
|
|
449
|
+
directory = directory.expanduser().resolve()
|
|
450
|
+
directory.mkdir(parents=True, exist_ok=True)
|
|
451
|
+
name = urllib.parse.urlparse(url).path.rsplit("/", 1)[-1]
|
|
452
|
+
path = directory / name
|
|
453
|
+
sidecar = path.with_name(path.name + ".sha256")
|
|
454
|
+
if path.is_file() and sidecar.is_file():
|
|
455
|
+
cached_digest = sha256_file(path, show_progress=True)
|
|
456
|
+
if cached_digest == sidecar.read_text().strip():
|
|
457
|
+
LOG.info("Using cached %s", path)
|
|
458
|
+
return path, cached_digest
|
|
459
|
+
LOG.warning("Cached checksum mismatch: downloading %s again", name)
|
|
460
|
+
LOG.info("Downloading %s", url)
|
|
461
|
+
LOG.info(" Cached file: %s; checksum: %s", path, sidecar)
|
|
462
|
+
for attempt in range(3):
|
|
463
|
+
temporary = None
|
|
464
|
+
try:
|
|
465
|
+
digest = hashlib.sha256()
|
|
466
|
+
received = 0
|
|
467
|
+
with open_url(url) as response:
|
|
468
|
+
expected = response.headers.get("Content-Length")
|
|
469
|
+
with tempfile.NamedTemporaryFile(
|
|
470
|
+
dir=directory, prefix=name + ".", suffix=".part", delete=False
|
|
471
|
+
) as out:
|
|
472
|
+
temporary = Path(out.name)
|
|
473
|
+
LOG.info(" Partial download: %s", temporary)
|
|
474
|
+
with progress(
|
|
475
|
+
total=int(expected) if expected is not None else None,
|
|
476
|
+
desc="Download " + file_label(path),
|
|
477
|
+
unit="B",
|
|
478
|
+
unit_scale=True,
|
|
479
|
+
unit_divisor=1024,
|
|
480
|
+
) as bar:
|
|
481
|
+
while block := response.read(4 * CHUNK_SIZE):
|
|
482
|
+
out.write(block)
|
|
483
|
+
digest.update(block)
|
|
484
|
+
received += len(block)
|
|
485
|
+
bar.update(len(block))
|
|
486
|
+
if expected is not None and received != int(expected):
|
|
487
|
+
raise OSError(f"Truncated download: expected {expected}, received {received}")
|
|
488
|
+
temporary.replace(path)
|
|
489
|
+
checksum = digest.hexdigest()
|
|
490
|
+
sidecar.write_text(checksum + "\n")
|
|
491
|
+
return path, checksum
|
|
492
|
+
except (OSError, urllib.error.URLError) as exc:
|
|
493
|
+
if isinstance(exc, urllib.error.HTTPError) and exc.code in {400, 403, 404}:
|
|
494
|
+
raise
|
|
495
|
+
if attempt == 2:
|
|
496
|
+
raise
|
|
497
|
+
LOG.warning("Download failed (%s); retrying", exc)
|
|
498
|
+
time.sleep(2**attempt)
|
|
499
|
+
finally:
|
|
500
|
+
if temporary is not None:
|
|
501
|
+
temporary.unlink(missing_ok=True)
|
|
502
|
+
raise AssertionError("Unreachable")
|
|
503
|
+
|
|
504
|
+
|
|
505
|
+
def sql_name(value: str) -> str:
|
|
506
|
+
"""Quote a SQLite identifier; bound parameters can only represent values."""
|
|
507
|
+
return '"' + value.replace('"', '""') + '"'
|
|
508
|
+
|
|
509
|
+
|
|
510
|
+
def parse_schema(path: Path) -> dict[str, list[tuple[str, str]]]:
|
|
511
|
+
"""Read column order/types from MySQL DDL without executing source SQL."""
|
|
512
|
+
with gzip.open(path, "rt", encoding="utf-8") as stream:
|
|
513
|
+
text = stream.read()
|
|
514
|
+
tables = {}
|
|
515
|
+
for match in re.finditer(
|
|
516
|
+
r"CREATE TABLE(?: IF NOT EXISTS)? `([^`]+)`\s*\((.*?)\n\)", text, re.S
|
|
517
|
+
):
|
|
518
|
+
columns = []
|
|
519
|
+
for column in re.finditer(r"^\s*`([^`]+)`\s+([A-Za-z]+)", match[2], re.M):
|
|
520
|
+
mysql_type = column[2].lower()
|
|
521
|
+
# Ignore MySQL constraints/options; imports need only SQLite affinities.
|
|
522
|
+
if mysql_type in {"tinyint", "smallint", "mediumint", "int", "bigint"}:
|
|
523
|
+
affinity = "INTEGER"
|
|
524
|
+
elif mysql_type in {"float", "double", "decimal"}:
|
|
525
|
+
affinity = "REAL"
|
|
526
|
+
else:
|
|
527
|
+
affinity = "TEXT"
|
|
528
|
+
columns.append((column[1], affinity))
|
|
529
|
+
if columns:
|
|
530
|
+
tables[match[1]] = columns
|
|
531
|
+
if not tables:
|
|
532
|
+
raise ValueError("Cannot parse any CREATE TABLE definitions from the source schema")
|
|
533
|
+
return tables
|
|
534
|
+
|
|
535
|
+
|
|
536
|
+
MYSQL_ESCAPES = {"0": "\0", "b": "\b", "n": "\n", "r": "\r", "t": "\t", "Z": "\x1a", "\\": "\\"}
|
|
537
|
+
|
|
538
|
+
|
|
539
|
+
def mysql_value(value: str) -> str | None:
|
|
540
|
+
r"""Decode MySQL dump escapes, distinguishing SQL NULL (\N) from text."""
|
|
541
|
+
if value == r"\N":
|
|
542
|
+
return None
|
|
543
|
+
return re.sub(r"\\(.)", lambda m: MYSQL_ESCAPES.get(m[1], m[1]), value)
|
|
544
|
+
|
|
545
|
+
|
|
546
|
+
def import_table(
|
|
547
|
+
db: sqlite3.Connection, table: str, columns: Sequence[tuple[str, str]], path: Path
|
|
548
|
+
) -> int:
|
|
549
|
+
"""Import one gzipped tab-separated dump, add lookup indexes and commit."""
|
|
550
|
+
destination = "ensembl_" + table
|
|
551
|
+
definitions = ", ".join(f"{sql_name(name)} {kind}" for name, kind in columns)
|
|
552
|
+
db.execute(f"CREATE TABLE {sql_name(destination)} ({definitions})")
|
|
553
|
+
statement = f"INSERT INTO {sql_name(destination)} VALUES ({','.join('?' for _ in columns)})"
|
|
554
|
+
count = 0
|
|
555
|
+
batch = []
|
|
556
|
+
with input_progress(path, "Import " + table, text=True) as (stream, bar, update):
|
|
557
|
+
for number, line in enumerate(stream, 1):
|
|
558
|
+
values = line.rstrip("\n").removesuffix("\r").split("\t")
|
|
559
|
+
if len(values) != len(columns):
|
|
560
|
+
raise ValueError(
|
|
561
|
+
f"{path.name}:{number}: {len(values)} fields; expected {len(columns)}"
|
|
562
|
+
)
|
|
563
|
+
batch.append(tuple(mysql_value(value) for value in values))
|
|
564
|
+
if len(batch) == 10000:
|
|
565
|
+
db.executemany(statement, batch)
|
|
566
|
+
count += len(batch)
|
|
567
|
+
batch.clear()
|
|
568
|
+
bar.set_postfix(rows=f"{count:,}", refresh=False)
|
|
569
|
+
update()
|
|
570
|
+
if batch:
|
|
571
|
+
db.executemany(statement, batch)
|
|
572
|
+
count += len(batch)
|
|
573
|
+
bar.set_postfix(rows=f"{count:,}", refresh=False)
|
|
574
|
+
names = {name for name, _ in columns}
|
|
575
|
+
# Primary lookup IDs plus joins commonly needed for transcript/domain data.
|
|
576
|
+
index_columns = list(
|
|
577
|
+
dict.fromkeys(
|
|
578
|
+
name
|
|
579
|
+
for name in (
|
|
580
|
+
table + "_id",
|
|
581
|
+
"stable_id",
|
|
582
|
+
"transcript_id",
|
|
583
|
+
"translation_id",
|
|
584
|
+
"exon_id",
|
|
585
|
+
"gene_id",
|
|
586
|
+
"seq_region_id",
|
|
587
|
+
"analysis_id",
|
|
588
|
+
"xref_id",
|
|
589
|
+
"ensembl_id",
|
|
590
|
+
"dbprimary_acc",
|
|
591
|
+
"interpro_ac",
|
|
592
|
+
"id",
|
|
593
|
+
"meta_key",
|
|
594
|
+
)
|
|
595
|
+
if name in names
|
|
596
|
+
)
|
|
597
|
+
)
|
|
598
|
+
for name in progress(index_columns, desc="Index " + table, unit="index"):
|
|
599
|
+
db.execute(
|
|
600
|
+
f"CREATE INDEX IF NOT EXISTS {sql_name(destination + '_' + name)} "
|
|
601
|
+
f"ON {sql_name(destination)} ({sql_name(name)})"
|
|
602
|
+
)
|
|
603
|
+
db.commit()
|
|
604
|
+
LOG.info("Imported %s: %s rows", table, f"{count:,}")
|
|
605
|
+
return count
|
|
606
|
+
|
|
607
|
+
|
|
608
|
+
# Chunked sequence storage and inclusive interval reads.
|
|
609
|
+
|
|
610
|
+
|
|
611
|
+
def create_sequence_tables(db: sqlite3.Connection) -> None:
|
|
612
|
+
"""Create sequence metadata and separately addressable compressed chunks."""
|
|
613
|
+
db.executescript("""
|
|
614
|
+
CREATE TABLE sequences (
|
|
615
|
+
kind TEXT NOT NULL, sequence_id TEXT NOT NULL, stable_id TEXT NOT NULL,
|
|
616
|
+
header TEXT NOT NULL, length INTEGER NOT NULL, sha256 TEXT NOT NULL,
|
|
617
|
+
PRIMARY KEY (kind, sequence_id)
|
|
618
|
+
) WITHOUT ROWID;
|
|
619
|
+
CREATE INDEX sequence_stable_id ON sequences (kind, stable_id);
|
|
620
|
+
CREATE TABLE sequence_chunks (
|
|
621
|
+
kind TEXT NOT NULL, sequence_id TEXT NOT NULL,
|
|
622
|
+
start INTEGER NOT NULL, data BLOB NOT NULL,
|
|
623
|
+
PRIMARY KEY (kind, sequence_id, start)
|
|
624
|
+
) WITHOUT ROWID;
|
|
625
|
+
""")
|
|
626
|
+
|
|
627
|
+
|
|
628
|
+
def validate_dna_regions(db: sqlite3.Connection) -> None:
|
|
629
|
+
"""Names are unique within coordinate systems, not across genome assemblies."""
|
|
630
|
+
db.execute("CREATE INDEX IF NOT EXISTS ensembl_seq_region_name ON ensembl_seq_region (name)")
|
|
631
|
+
unmatched = db.execute(
|
|
632
|
+
"SELECT s.sequence_id, s.length FROM sequences s WHERE s.kind='dna' AND NOT EXISTS ("
|
|
633
|
+
"SELECT 1 FROM ensembl_seq_region r JOIN ensembl_coord_system c USING (coord_system_id) "
|
|
634
|
+
"WHERE r.name=s.sequence_id AND c.version=? AND r.length=s.length) LIMIT 1",
|
|
635
|
+
(ASSEMBLY,),
|
|
636
|
+
).fetchone()
|
|
637
|
+
if unmatched:
|
|
638
|
+
sequence_id, length = unmatched
|
|
639
|
+
candidates = db.execute(
|
|
640
|
+
"SELECT r.length FROM ensembl_seq_region r "
|
|
641
|
+
"JOIN ensembl_coord_system c USING (coord_system_id) "
|
|
642
|
+
"WHERE r.name=? AND c.version=?",
|
|
643
|
+
(sequence_id, ASSEMBLY),
|
|
644
|
+
).fetchall()
|
|
645
|
+
lengths = sorted({row[0] for row in candidates})
|
|
646
|
+
raise ValueError(
|
|
647
|
+
f"FASTA/core sequence-region mismatch for {sequence_id}: FASTA length {length}; "
|
|
648
|
+
f"{ASSEMBLY} core lengths {lengths or 'no matching region'}"
|
|
649
|
+
)
|
|
650
|
+
LOG.info("Validated DNA sequence-region lengths against %s core records", ASSEMBLY)
|
|
651
|
+
|
|
652
|
+
|
|
653
|
+
def import_fasta(db: sqlite3.Connection, kind: str, path: Path, *, resume: bool = False) -> int:
|
|
654
|
+
"""Stream FASTA, optionally reusing committed records from the same input.
|
|
655
|
+
|
|
656
|
+
The builder checks the source checksum before enabling resume. Record rows
|
|
657
|
+
exist only after their chunks are complete; discard any orphaned chunks.
|
|
658
|
+
"""
|
|
659
|
+
completed = set()
|
|
660
|
+
if resume:
|
|
661
|
+
completed = {
|
|
662
|
+
row[0] for row in db.execute("SELECT sequence_id FROM sequences WHERE kind=?", (kind,))
|
|
663
|
+
}
|
|
664
|
+
db.execute(
|
|
665
|
+
"DELETE FROM sequence_chunks WHERE kind=? AND NOT EXISTS ("
|
|
666
|
+
"SELECT 1 FROM sequences s WHERE s.kind=sequence_chunks.kind "
|
|
667
|
+
"AND s.sequence_id=sequence_chunks.sequence_id)",
|
|
668
|
+
(kind,),
|
|
669
|
+
)
|
|
670
|
+
db.commit()
|
|
671
|
+
if completed:
|
|
672
|
+
LOG.info(
|
|
673
|
+
"Resuming %s FASTA: reusing %s complete sequences; rereading gzip to the next record",
|
|
674
|
+
kind,
|
|
675
|
+
f"{len(completed):,}",
|
|
676
|
+
)
|
|
677
|
+
sequence_id = header = None
|
|
678
|
+
skip_record = False
|
|
679
|
+
buffer = bytearray()
|
|
680
|
+
offset = count = 0
|
|
681
|
+
digest = hashlib.sha256()
|
|
682
|
+
|
|
683
|
+
def flush(size: int) -> None:
|
|
684
|
+
"""Store the next chunk; its database start is 1-based, not a byte offset."""
|
|
685
|
+
nonlocal offset
|
|
686
|
+
chunk = bytes(buffer[:size])
|
|
687
|
+
del buffer[:size]
|
|
688
|
+
db.execute(
|
|
689
|
+
"INSERT INTO sequence_chunks VALUES (?, ?, ?, ?)",
|
|
690
|
+
(kind, sequence_id, offset + 1, zlib.compress(chunk, level=1)),
|
|
691
|
+
)
|
|
692
|
+
offset += len(chunk)
|
|
693
|
+
if size == CHUNK_SIZE:
|
|
694
|
+
update()
|
|
695
|
+
|
|
696
|
+
def finish() -> None:
|
|
697
|
+
"""Flush one FASTA record and record its full length, header and checksum."""
|
|
698
|
+
nonlocal count
|
|
699
|
+
if sequence_id is None:
|
|
700
|
+
return
|
|
701
|
+
if skip_record:
|
|
702
|
+
count += 1
|
|
703
|
+
return
|
|
704
|
+
if buffer:
|
|
705
|
+
flush(len(buffer))
|
|
706
|
+
if offset == 0:
|
|
707
|
+
raise ValueError(f"Empty FASTA sequence: {sequence_id}")
|
|
708
|
+
# Assembly sequence names may contain dots; only peptide version suffixes
|
|
709
|
+
# are removed to support lookups by either stable or versioned protein ID.
|
|
710
|
+
stable_id = sequence_id if kind == "dna" else re.sub(r"\.\d+$", "", sequence_id)
|
|
711
|
+
db.execute(
|
|
712
|
+
"INSERT INTO sequences VALUES (?, ?, ?, ?, ?, ?)",
|
|
713
|
+
(kind, sequence_id, stable_id, header, offset, digest.hexdigest()),
|
|
714
|
+
)
|
|
715
|
+
count += 1
|
|
716
|
+
if count % 1000 == 0 or kind == "dna":
|
|
717
|
+
db.commit()
|
|
718
|
+
bar.set_postfix(records=f"{count:,}", refresh=False)
|
|
719
|
+
update()
|
|
720
|
+
|
|
721
|
+
with input_progress(path, "Import " + kind + " FASTA") as (stream, bar, update):
|
|
722
|
+
for line in stream:
|
|
723
|
+
if line.startswith(b">"):
|
|
724
|
+
finish()
|
|
725
|
+
header = line[1:].strip().decode("utf-8")
|
|
726
|
+
sequence_id = header.split()[0]
|
|
727
|
+
skip_record = sequence_id in completed
|
|
728
|
+
offset = 0
|
|
729
|
+
digest = hashlib.sha256()
|
|
730
|
+
else:
|
|
731
|
+
if skip_record:
|
|
732
|
+
continue
|
|
733
|
+
sequence = line.strip().upper()
|
|
734
|
+
if not sequence:
|
|
735
|
+
continue
|
|
736
|
+
if sequence_id is None:
|
|
737
|
+
raise ValueError("FASTA sequence precedes its header")
|
|
738
|
+
alphabet = b"ABCDEFGHIJKLMNOPQRSTUVWXYZ*" if kind == "pep" else b"ACGTRYSWKMBDHVN"
|
|
739
|
+
if sequence.translate(None, alphabet):
|
|
740
|
+
raise ValueError(f"Invalid {kind} sequence characters in {sequence_id}")
|
|
741
|
+
digest.update(sequence)
|
|
742
|
+
buffer.extend(sequence)
|
|
743
|
+
while len(buffer) >= CHUNK_SIZE:
|
|
744
|
+
flush(CHUNK_SIZE)
|
|
745
|
+
finish()
|
|
746
|
+
bar.set_postfix(records=f"{count:,}", refresh=False)
|
|
747
|
+
if not count:
|
|
748
|
+
raise ValueError(f"No FASTA records in {path}")
|
|
749
|
+
db.commit()
|
|
750
|
+
LOG.info("Imported %s: %s sequences", kind, f"{count:,}")
|
|
751
|
+
return count
|
|
752
|
+
|
|
753
|
+
|
|
754
|
+
def fetch_sequence(
|
|
755
|
+
db: sqlite3.Connection,
|
|
756
|
+
kind: str,
|
|
757
|
+
sequence_id: str,
|
|
758
|
+
start: int = 1,
|
|
759
|
+
end: int | None = None,
|
|
760
|
+
strand: int = 1,
|
|
761
|
+
*,
|
|
762
|
+
_chunk_cache: ChunkCache | None = None,
|
|
763
|
+
_cache_limit: int = 64,
|
|
764
|
+
) -> str:
|
|
765
|
+
"""Read a 1-based inclusive interval, optionally using a bounded chunk cache.
|
|
766
|
+
|
|
767
|
+
Unversioned protein IDs are accepted only when they match one sequence.
|
|
768
|
+
Negative-strand nucleotide reads return the reverse complement.
|
|
769
|
+
"""
|
|
770
|
+
row = db.execute(
|
|
771
|
+
"SELECT sequence_id, length FROM sequences WHERE kind=? AND sequence_id=?",
|
|
772
|
+
(kind, sequence_id),
|
|
773
|
+
).fetchone()
|
|
774
|
+
if row is None:
|
|
775
|
+
# Never silently choose among several versions of an unversioned ID.
|
|
776
|
+
rows = db.execute(
|
|
777
|
+
"SELECT sequence_id, length FROM sequences WHERE kind=? AND stable_id=?",
|
|
778
|
+
(kind, sequence_id),
|
|
779
|
+
).fetchall()
|
|
780
|
+
if len(rows) != 1:
|
|
781
|
+
raise KeyError(f"Missing or ambiguous sequence: {kind}/{sequence_id}")
|
|
782
|
+
row = rows[0]
|
|
783
|
+
sequence_id, length = row
|
|
784
|
+
end = length if end is None else end
|
|
785
|
+
if not 1 <= start <= end <= length:
|
|
786
|
+
raise ValueError(f"Invalid interval {start}-{end}; sequence length is {length}")
|
|
787
|
+
if strand not in {-1, 1} or (kind == "pep" and strand != 1):
|
|
788
|
+
raise ValueError("Strand must be +1/-1 for nucleotides or +1 for peptides")
|
|
789
|
+
# Convert the requested position to the 1-based start of its containing chunk.
|
|
790
|
+
first_chunk = ((start - 1) // CHUNK_SIZE) * CHUNK_SIZE + 1
|
|
791
|
+
pieces = []
|
|
792
|
+
for position in range(first_chunk, end + 1, CHUNK_SIZE):
|
|
793
|
+
key = kind, sequence_id, position
|
|
794
|
+
if _chunk_cache is not None and key in _chunk_cache:
|
|
795
|
+
sequence = _chunk_cache[key]
|
|
796
|
+
_chunk_cache.move_to_end(key)
|
|
797
|
+
else:
|
|
798
|
+
record = db.execute(
|
|
799
|
+
"SELECT data FROM sequence_chunks WHERE kind=? AND sequence_id=? AND start=?", key
|
|
800
|
+
).fetchone()
|
|
801
|
+
if record is None:
|
|
802
|
+
raise ValueError(f"Missing sequence chunk: {key}")
|
|
803
|
+
sequence = zlib.decompress(record[0])
|
|
804
|
+
if _chunk_cache is not None and _cache_limit:
|
|
805
|
+
_chunk_cache[key] = sequence
|
|
806
|
+
while len(_chunk_cache) > _cache_limit:
|
|
807
|
+
_chunk_cache.popitem(last=False)
|
|
808
|
+
# Python slices are 0-based/exclusive; database intervals are inclusive.
|
|
809
|
+
pieces.append(sequence[max(0, start - position) : end - position + 1])
|
|
810
|
+
result = b"".join(pieces).decode("ascii")
|
|
811
|
+
if len(result) != end - start + 1:
|
|
812
|
+
raise ValueError("Missing or damaged sequence chunks")
|
|
813
|
+
if strand == -1:
|
|
814
|
+
result = result.translate(str.maketrans("ACGTRYSWKMBDHVN", "TGCAYRSWMKVHDBN"))[::-1]
|
|
815
|
+
return result
|
|
816
|
+
|
|
817
|
+
|
|
818
|
+
# Streaming transcript models and coordinate mapping.
|
|
819
|
+
|
|
820
|
+
|
|
821
|
+
def dict_rows(
|
|
822
|
+
db: sqlite3.Connection, query: str, *, description: str | None = None
|
|
823
|
+
) -> Iterator[DatabaseRow]:
|
|
824
|
+
"""Yield named rows from a query without changing the connection row factory."""
|
|
825
|
+
# Sorted queries can do substantial work before returning their first row.
|
|
826
|
+
if description is not None:
|
|
827
|
+
with sqlite_activity(description):
|
|
828
|
+
cursor = db.execute(query)
|
|
829
|
+
else:
|
|
830
|
+
cursor = db.execute(query)
|
|
831
|
+
names = [column[0] for column in cursor.description]
|
|
832
|
+
for row in cursor:
|
|
833
|
+
yield dict(zip(names, row))
|
|
834
|
+
|
|
835
|
+
|
|
836
|
+
class GroupStream:
|
|
837
|
+
"""Consume a query grouped by internal transcript ID without loading it all.
|
|
838
|
+
|
|
839
|
+
Input rows and calls to take() must both follow ascending transcript ID.
|
|
840
|
+
This lets exon/feature queries advance alongside the main transcript query.
|
|
841
|
+
"""
|
|
842
|
+
|
|
843
|
+
def __init__(self, rows: Iterable[DatabaseRow]) -> None:
|
|
844
|
+
"""Prime the first group from an already ordered row iterator."""
|
|
845
|
+
self.groups = iter(groupby(rows, key=lambda row: row["transcript_id"]))
|
|
846
|
+
self.current = next(self.groups, None)
|
|
847
|
+
|
|
848
|
+
def take(self, transcript_id: int) -> list[DatabaseRow]:
|
|
849
|
+
"""Return this transcript's rows, or an empty list when it has no group."""
|
|
850
|
+
while self.current is not None and self.current[0] < transcript_id:
|
|
851
|
+
self.current = next(self.groups, None)
|
|
852
|
+
if self.current is None or self.current[0] != transcript_id:
|
|
853
|
+
return []
|
|
854
|
+
rows = list(self.current[1])
|
|
855
|
+
self.current = next(self.groups, None)
|
|
856
|
+
return rows
|
|
857
|
+
|
|
858
|
+
|
|
859
|
+
def reference_structure(
|
|
860
|
+
transcript: Mapping[str, Any],
|
|
861
|
+
exons: Sequence[DatabaseRow],
|
|
862
|
+
translation: Mapping[str, Any] | None,
|
|
863
|
+
) -> TranscriptProteinFeatureResult:
|
|
864
|
+
"""Validate a model and derive exon, splice-window and CDS coordinates.
|
|
865
|
+
|
|
866
|
+
Genomic positions remain on the reference strand. Pre-mRNA and CDS positions
|
|
867
|
+
follow transcription direction; pre-mRNA includes introns, CDS does not.
|
|
868
|
+
A missing translation produces a noncoding model with no CDS blocks.
|
|
869
|
+
Missing leading codon bases occupy peptide coordinates but have no genome
|
|
870
|
+
coordinates; the first CDS block starts after that offset.
|
|
871
|
+
"""
|
|
872
|
+
strand = transcript["strand"]
|
|
873
|
+
start, end = transcript["start"], transcript["end"]
|
|
874
|
+
chromosome, assembly = transcript["chromosome"], transcript["assembly_name"]
|
|
875
|
+
if strand not in {-1, 1} or not 1 <= start <= end or not exons:
|
|
876
|
+
raise ValueError("Invalid transcript strand/span or missing exons")
|
|
877
|
+
# Negative-strand transcripts start at the highest genomic exon coordinate.
|
|
878
|
+
oriented = sorted(exons, key=lambda exon: exon["start"], reverse=strand == -1)
|
|
879
|
+
if [exon["rank"] for exon in oriented] != list(range(1, len(exons) + 1)):
|
|
880
|
+
raise ValueError("Exon ranks disagree with transcript strand/order")
|
|
881
|
+
result: TranscriptProteinFeatureResult = {
|
|
882
|
+
"translation_id": translation["stable_id"] if translation else None,
|
|
883
|
+
"protein_length": None,
|
|
884
|
+
"chromosome": chromosome,
|
|
885
|
+
"strand": strand,
|
|
886
|
+
"transcript_genomic_start": start,
|
|
887
|
+
"transcript_genomic_end": end,
|
|
888
|
+
"premrna_length": end - start + 1,
|
|
889
|
+
"transcript_exons": [],
|
|
890
|
+
"cds_blocks": [],
|
|
891
|
+
"protein_features": [],
|
|
892
|
+
"splice_sites": [],
|
|
893
|
+
"assembly_name": assembly,
|
|
894
|
+
}
|
|
895
|
+
for number, exon in enumerate(oriented, 1):
|
|
896
|
+
lo, hi = exon["start"], exon["end"]
|
|
897
|
+
if not start <= lo <= hi <= end or exon["strand"] != strand:
|
|
898
|
+
raise ValueError("Exon coordinates/strand inconsistent with transcript")
|
|
899
|
+
if exon["seq_region_id"] != transcript["seq_region_id"]:
|
|
900
|
+
raise ValueError("Exon and transcript use different sequence regions")
|
|
901
|
+
if number > 1:
|
|
902
|
+
previous = oriented[number - 2]
|
|
903
|
+
if (strand == 1 and previous["end"] >= lo) or (
|
|
904
|
+
strand == -1 and hi >= previous["start"]
|
|
905
|
+
):
|
|
906
|
+
raise ValueError("Overlapping exons within transcript")
|
|
907
|
+
premrna_start = lo - start + 1 if strand == 1 else end - hi + 1
|
|
908
|
+
premrna_end = hi - start + 1 if strand == 1 else end - lo + 1
|
|
909
|
+
result["transcript_exons"].append(
|
|
910
|
+
{
|
|
911
|
+
"exon_number": number,
|
|
912
|
+
"chromosome": chromosome,
|
|
913
|
+
"genomic_start": lo,
|
|
914
|
+
"genomic_end": hi,
|
|
915
|
+
"premrna_start": premrna_start,
|
|
916
|
+
"premrna_end": premrna_end,
|
|
917
|
+
}
|
|
918
|
+
)
|
|
919
|
+
for site_type, position, genomic in (
|
|
920
|
+
("acceptor", premrna_start, lo if strand == 1 else hi),
|
|
921
|
+
("donor", premrna_end, hi if strand == 1 else lo),
|
|
922
|
+
):
|
|
923
|
+
# Outer transcript ends are not internal splice junctions. Each retained
|
|
924
|
+
# window spans two exon bases and the two adjacent intron bases.
|
|
925
|
+
if (site_type == "acceptor" and number == 1) or (
|
|
926
|
+
site_type == "donor" and number == len(oriented)
|
|
927
|
+
):
|
|
928
|
+
continue
|
|
929
|
+
if (strand, site_type) in {(-1, "donor"), (1, "acceptor")}:
|
|
930
|
+
disruption_start, disruption_end = genomic - 2, genomic + 1
|
|
931
|
+
else:
|
|
932
|
+
disruption_start, disruption_end = genomic - 1, genomic + 2
|
|
933
|
+
result["splice_sites"].append(
|
|
934
|
+
{
|
|
935
|
+
"type": cast(Literal["acceptor", "donor"], site_type),
|
|
936
|
+
"exon_number": number,
|
|
937
|
+
"genomic_position": genomic,
|
|
938
|
+
"premrna_position": position,
|
|
939
|
+
"disruption_start": disruption_start,
|
|
940
|
+
"disruption_end": disruption_end,
|
|
941
|
+
}
|
|
942
|
+
)
|
|
943
|
+
if translation is None:
|
|
944
|
+
return result
|
|
945
|
+
exon_by_id = {exon["exon_id"]: exon for exon in oriented}
|
|
946
|
+
first = exon_by_id.get(translation["start_exon_id"])
|
|
947
|
+
last = exon_by_id.get(translation["end_exon_id"])
|
|
948
|
+
if first is None or last is None or first["rank"] > last["rank"]:
|
|
949
|
+
raise ValueError("Translation endpoints do not belong to ordered transcript exons")
|
|
950
|
+
for exon, offset in ((first, translation["seq_start"]), (last, translation["seq_end"])):
|
|
951
|
+
if not 1 <= offset <= exon["end"] - exon["start"] + 1:
|
|
952
|
+
raise ValueError("Translation endpoint offset lies outside exon")
|
|
953
|
+
# Translation offsets are 1-based within the endpoint exons, counted in
|
|
954
|
+
# transcription direction (backwards from exon end on the negative strand).
|
|
955
|
+
first_pos = (
|
|
956
|
+
first["start"] + translation["seq_start"] - 1
|
|
957
|
+
if strand == 1
|
|
958
|
+
else first["end"] - translation["seq_start"] + 1
|
|
959
|
+
)
|
|
960
|
+
last_pos = (
|
|
961
|
+
last["start"] + translation["seq_end"] - 1
|
|
962
|
+
if strand == 1
|
|
963
|
+
else last["end"] - translation["seq_end"] + 1
|
|
964
|
+
)
|
|
965
|
+
if (strand == 1 and first_pos > last_pos) or (strand == -1 and first_pos < last_pos):
|
|
966
|
+
raise ValueError("Translation start follows translation end")
|
|
967
|
+
coding_lo, coding_hi = sorted((first_pos, last_pos))
|
|
968
|
+
# A 5'-incomplete CDS starts partway through the first peptide codon.
|
|
969
|
+
# Ensembl pads that codon with `phase` unknown bases when translating.
|
|
970
|
+
# Keep those peptide coordinates, but map only bases present in the genome:
|
|
971
|
+
# starting at phase + 1 also prevents a missing native start being retained.
|
|
972
|
+
phase = first["phase"]
|
|
973
|
+
if phase not in {-1, 0, 1, 2}:
|
|
974
|
+
raise ValueError("Invalid translation start exon phase")
|
|
975
|
+
start_phase = max(0, phase)
|
|
976
|
+
if start_phase:
|
|
977
|
+
result["cds_start_phase"] = start_phase
|
|
978
|
+
cds_pos = start_phase + 1
|
|
979
|
+
# Clip each exon to the coding span, then concatenate its CDS coordinates.
|
|
980
|
+
for prepared_exon in result["transcript_exons"]:
|
|
981
|
+
lo, hi = (
|
|
982
|
+
max(prepared_exon["genomic_start"], coding_lo),
|
|
983
|
+
min(prepared_exon["genomic_end"], coding_hi),
|
|
984
|
+
)
|
|
985
|
+
if lo > hi:
|
|
986
|
+
continue
|
|
987
|
+
length = hi - lo + 1
|
|
988
|
+
result["cds_blocks"].append(
|
|
989
|
+
{
|
|
990
|
+
"chromosome": chromosome,
|
|
991
|
+
"genomic_start": lo,
|
|
992
|
+
"genomic_end": hi,
|
|
993
|
+
"premrna_start": lo - start + 1 if strand == 1 else end - hi + 1,
|
|
994
|
+
"premrna_end": hi - start + 1 if strand == 1 else end - lo + 1,
|
|
995
|
+
"strand": strand,
|
|
996
|
+
"assembly_name": assembly,
|
|
997
|
+
"cds_start": cds_pos,
|
|
998
|
+
"cds_end": cds_pos + length - 1,
|
|
999
|
+
}
|
|
1000
|
+
)
|
|
1001
|
+
cds_pos += length
|
|
1002
|
+
return result
|
|
1003
|
+
|
|
1004
|
+
|
|
1005
|
+
def protein_segments(
|
|
1006
|
+
aa_start: int,
|
|
1007
|
+
aa_end: int,
|
|
1008
|
+
blocks: Sequence[CDSBlock],
|
|
1009
|
+
*,
|
|
1010
|
+
block_ends: Sequence[int] | None = None,
|
|
1011
|
+
) -> list[GenomicSegment]:
|
|
1012
|
+
"""Map an inclusive amino-acid interval to genomic segments across CDS exons.
|
|
1013
|
+
|
|
1014
|
+
Blocks are ordered by CDS position. Passing their end positions enables a
|
|
1015
|
+
binary search; callers mapping many features can reuse that index.
|
|
1016
|
+
"""
|
|
1017
|
+
segments: list[GenomicSegment] = []
|
|
1018
|
+
cds_start, cds_end = (aa_start - 1) * 3 + 1, aa_end * 3
|
|
1019
|
+
# The preprocessor supplies a once-per-transcript index of ordered CDS blocks.
|
|
1020
|
+
if block_ends is not None:
|
|
1021
|
+
first = bisect_left(block_ends, cds_start)
|
|
1022
|
+
last = bisect_left(block_ends, cds_end) + 1
|
|
1023
|
+
selected = blocks[first:last]
|
|
1024
|
+
else:
|
|
1025
|
+
selected = blocks
|
|
1026
|
+
for block in selected:
|
|
1027
|
+
lo = max(cds_start, block["cds_start"])
|
|
1028
|
+
hi = min(cds_end, block["cds_end"])
|
|
1029
|
+
if lo > hi:
|
|
1030
|
+
continue
|
|
1031
|
+
start_offset = lo - block["cds_start"]
|
|
1032
|
+
end_offset = hi - block["cds_start"]
|
|
1033
|
+
# Return ascending genomic bounds even when CDS direction is reversed.
|
|
1034
|
+
if block["strand"] == 1:
|
|
1035
|
+
start = block["genomic_start"] + start_offset
|
|
1036
|
+
end = block["genomic_start"] + end_offset
|
|
1037
|
+
else:
|
|
1038
|
+
start = block["genomic_end"] - end_offset
|
|
1039
|
+
end = block["genomic_end"] - start_offset
|
|
1040
|
+
segments.append(
|
|
1041
|
+
{
|
|
1042
|
+
"chromosome": block["chromosome"],
|
|
1043
|
+
"start": start,
|
|
1044
|
+
"end": end,
|
|
1045
|
+
"strand": block["strand"],
|
|
1046
|
+
"assembly_name": block["assembly_name"],
|
|
1047
|
+
}
|
|
1048
|
+
)
|
|
1049
|
+
return segments
|
|
1050
|
+
|
|
1051
|
+
|
|
1052
|
+
# Source compatibility checks and bulk protein annotation metadata.
|
|
1053
|
+
|
|
1054
|
+
|
|
1055
|
+
def protein_analyses(db: sqlite3.Connection) -> AnalysisMetadata:
|
|
1056
|
+
"""Use database names for feature sources, not Ensembl pipeline names.
|
|
1057
|
+
|
|
1058
|
+
For example, logic_name='hmmpanther' describes how the analysis ran;
|
|
1059
|
+
db='PANTHER' identifies the annotation source stored with each feature.
|
|
1060
|
+
Small offline fixtures may have only logic_name, so retain that fallback.
|
|
1061
|
+
Structure pipelines override database labels so their mappings can always
|
|
1062
|
+
be excluded from functional annotations.
|
|
1063
|
+
"""
|
|
1064
|
+
analyses = {}
|
|
1065
|
+
for row in dict_rows(db, "SELECT * FROM ensembl_analysis"):
|
|
1066
|
+
source = row.get("db") or row["logic_name"]
|
|
1067
|
+
logic_name = (row["logic_name"] or "").lower()
|
|
1068
|
+
if logic_name in STRUCTURE_SOURCES:
|
|
1069
|
+
source = logic_name
|
|
1070
|
+
if (source or "").lower() in {"panther", "hmmpanther"}:
|
|
1071
|
+
source = "PANTHER"
|
|
1072
|
+
analyses[row["analysis_id"]] = (source, row.get("db_version"))
|
|
1073
|
+
return analyses
|
|
1074
|
+
|
|
1075
|
+
|
|
1076
|
+
def download_panther_classifications(version: str, cache: Path) -> tuple[Path, str, str]:
|
|
1077
|
+
"""Fetch the exact release, accepting the historical trailing underscore.
|
|
1078
|
+
|
|
1079
|
+
Prefer an already cached filename so archived files also work offline.
|
|
1080
|
+
Only a 404 permits the alternate name; other HTTP/network errors propagate.
|
|
1081
|
+
"""
|
|
1082
|
+
directory = cache / version
|
|
1083
|
+
root = PANTHER_CLASSIFICATIONS_URL + version + "/PANTHER_Sequence_Classification_files/"
|
|
1084
|
+
names = [f"PTHR{version}_human", f"PTHR{version}_human_"]
|
|
1085
|
+
names.sort(
|
|
1086
|
+
key=lambda name: (
|
|
1087
|
+
not ((directory / name).is_file() and (directory / (name + ".sha256")).is_file())
|
|
1088
|
+
)
|
|
1089
|
+
)
|
|
1090
|
+
for index, name in enumerate(names):
|
|
1091
|
+
url = root + name
|
|
1092
|
+
try:
|
|
1093
|
+
path, digest = download(url, directory)
|
|
1094
|
+
return path, digest, url
|
|
1095
|
+
except urllib.error.HTTPError as exc:
|
|
1096
|
+
if exc.code != 404 or index == len(names) - 1:
|
|
1097
|
+
raise
|
|
1098
|
+
LOG.info(
|
|
1099
|
+
"PANTHER %s filename unavailable; trying the alternate archived filename", version
|
|
1100
|
+
)
|
|
1101
|
+
raise AssertionError("Unreachable")
|
|
1102
|
+
|
|
1103
|
+
|
|
1104
|
+
def panther_release(analyses: AnalysisMetadata) -> str | None:
|
|
1105
|
+
"""Require one exact PANTHER release when automatic metadata is needed."""
|
|
1106
|
+
versions = {version for source, version in analyses.values() if source == "PANTHER"}
|
|
1107
|
+
if not versions:
|
|
1108
|
+
return None
|
|
1109
|
+
if len(versions) != 1 or not re.fullmatch(r"\d+(?:\.\d+)*", next(iter(versions)) or ""):
|
|
1110
|
+
raise ValueError(
|
|
1111
|
+
"Cannot determine one PANTHER release from Ensembl analysis metadata; "
|
|
1112
|
+
"supply --panther-classifications FILE"
|
|
1113
|
+
)
|
|
1114
|
+
return next(iter(versions))
|
|
1115
|
+
|
|
1116
|
+
|
|
1117
|
+
def panther_entry_rows(path: Path) -> Iterator[tuple[str, str, str]]:
|
|
1118
|
+
"""Read human UniProt-to-subfamily assignments from PANTHER's bulk TSV.
|
|
1119
|
+
|
|
1120
|
+
Older files place the subfamily ID/name in columns 3/5; newer files use
|
|
1121
|
+
columns 4/6. Detect the layout from the classification ID, independently of
|
|
1122
|
+
the filename. Protein accessions always come from the identifier in column 1.
|
|
1123
|
+
Family/subfamily classifications do not supply new domain coordinates.
|
|
1124
|
+
"""
|
|
1125
|
+
with input_progress(path, "Import PANTHER", compressed=False, text=True) as (
|
|
1126
|
+
stream,
|
|
1127
|
+
bar,
|
|
1128
|
+
update,
|
|
1129
|
+
):
|
|
1130
|
+
count = 0
|
|
1131
|
+
columns = None
|
|
1132
|
+
for number, line in enumerate(stream, 1):
|
|
1133
|
+
if not line.strip():
|
|
1134
|
+
continue
|
|
1135
|
+
fields = line.rstrip("\r\n").split("\t")
|
|
1136
|
+
if len(fields) < 5 or not fields[0].startswith("HUMAN|"):
|
|
1137
|
+
raise ValueError(f"Invalid human PANTHER classification at {path}:{number}")
|
|
1138
|
+
accession = re.search(r"(?:^|\|)UniProtKB[:=]([^|]+)", fields[0])
|
|
1139
|
+
if accession is None:
|
|
1140
|
+
raise ValueError(f"Missing UniProt accession at {path}:{number}")
|
|
1141
|
+
if columns is None:
|
|
1142
|
+
layouts = [
|
|
1143
|
+
(id_col, name_col)
|
|
1144
|
+
for id_col, name_col in ((3, 5), (2, 4))
|
|
1145
|
+
if len(fields) > name_col
|
|
1146
|
+
]
|
|
1147
|
+
columns = next(
|
|
1148
|
+
(
|
|
1149
|
+
layout
|
|
1150
|
+
for layout in layouts
|
|
1151
|
+
if re.fullmatch(r"PTHR\d+:SF\d+", fields[layout[0]])
|
|
1152
|
+
),
|
|
1153
|
+
None,
|
|
1154
|
+
)
|
|
1155
|
+
# Unassigned leading records can still identify an empty ID
|
|
1156
|
+
# column. Once detected, keep the layout fixed to catch bad rows.
|
|
1157
|
+
if columns is None:
|
|
1158
|
+
columns = next((layout for layout in layouts if not fields[layout[0]]), None)
|
|
1159
|
+
if columns is None:
|
|
1160
|
+
raise ValueError(
|
|
1161
|
+
f"Invalid PANTHER subfamily at {path}:{number}; "
|
|
1162
|
+
"expected an ID in column 3 or 4"
|
|
1163
|
+
)
|
|
1164
|
+
id_col, name_col = columns
|
|
1165
|
+
if len(fields) <= name_col:
|
|
1166
|
+
raise ValueError(f"Invalid human PANTHER classification at {path}:{number}")
|
|
1167
|
+
subfamily, name = fields[id_col], fields[name_col]
|
|
1168
|
+
# Records without a named subfamily cannot support the family fallback.
|
|
1169
|
+
if not subfamily:
|
|
1170
|
+
continue
|
|
1171
|
+
if not re.fullmatch(r"PTHR\d+:SF\d+", subfamily):
|
|
1172
|
+
raise ValueError(f"Invalid PANTHER subfamily at {path}:{number}: {subfamily}")
|
|
1173
|
+
if not name:
|
|
1174
|
+
continue
|
|
1175
|
+
count += 1
|
|
1176
|
+
yield accession[1], subfamily, name
|
|
1177
|
+
if number % 1000 == 0:
|
|
1178
|
+
bar.set_postfix(records=count, refresh=False)
|
|
1179
|
+
update()
|
|
1180
|
+
if not count:
|
|
1181
|
+
raise ValueError(f"No named human PANTHER subfamilies in {path}")
|
|
1182
|
+
bar.set_postfix(records=count, refresh=False)
|
|
1183
|
+
|
|
1184
|
+
|
|
1185
|
+
def panther_translation_annotations(db: sqlite3.Connection) -> dict[int, tuple[str, str]]:
|
|
1186
|
+
"""Match exact protein cross-references; never propagate by gene name.
|
|
1187
|
+
|
|
1188
|
+
An ambiguous protein assignment is omitted. The caller also requires its
|
|
1189
|
+
family to agree with the coordinate-bearing Ensembl PANTHER hit.
|
|
1190
|
+
"""
|
|
1191
|
+
tables = {row[0] for row in db.execute("SELECT name FROM sqlite_master WHERE type='table'")}
|
|
1192
|
+
if "ensembl_object_xref" not in tables:
|
|
1193
|
+
return {}
|
|
1194
|
+
matches: dict[int, set[tuple[str, str]]] = {}
|
|
1195
|
+
# Start from the small human lookup and use the imported accession/xref
|
|
1196
|
+
# indexes, rather than scanning every row in the much larger core tables.
|
|
1197
|
+
for translation_id, subfamily, name in db.execute("""
|
|
1198
|
+
SELECT o.ensembl_id, p.subfamily_id, p.name
|
|
1199
|
+
FROM ff_panther p CROSS JOIN ensembl_xref x ON x.dbprimary_acc=p.accession
|
|
1200
|
+
CROSS JOIN ensembl_object_xref o ON o.xref_id=x.xref_id
|
|
1201
|
+
WHERE o.ensembl_object_type='Translation'
|
|
1202
|
+
"""):
|
|
1203
|
+
matches.setdefault(translation_id, set()).add((subfamily, name))
|
|
1204
|
+
return {key: next(iter(values)) for key, values in matches.items() if len(values) == 1}
|
|
1205
|
+
|
|
1206
|
+
|
|
1207
|
+
def validate_reference_source(db: sqlite3.Connection) -> dict[str, str]:
|
|
1208
|
+
"""Reject incompatible or incomplete source databases before changing them."""
|
|
1209
|
+
available = {row[0] for row in db.execute("SELECT name FROM sqlite_master WHERE type='table'")}
|
|
1210
|
+
required = {"ensembl_" + name for name in PREPROCESS_TABLES}
|
|
1211
|
+
required.update({"build_metadata", "sequences", "sequence_chunks"})
|
|
1212
|
+
missing = required - available
|
|
1213
|
+
if missing:
|
|
1214
|
+
if (
|
|
1215
|
+
"build_metadata" in available
|
|
1216
|
+
and dict(db.execute("SELECT key, value FROM build_metadata")).get("reference_kind")
|
|
1217
|
+
== "runtime"
|
|
1218
|
+
):
|
|
1219
|
+
raise ValueError(
|
|
1220
|
+
"This compact prebuilt reference has no Ensembl source tables. "
|
|
1221
|
+
"Install a newer prebuilt with prepare-data --release RELEASE --force, "
|
|
1222
|
+
"or build a full reference with --from-source before reprocessing."
|
|
1223
|
+
)
|
|
1224
|
+
raise ValueError(
|
|
1225
|
+
f"Preprocessing needs {sorted(missing)}; rebuild with 'fusion-function prepare-data --from-source'"
|
|
1226
|
+
)
|
|
1227
|
+
metadata = dict(db.execute("SELECT key, value FROM build_metadata"))
|
|
1228
|
+
if metadata.get("format_version") != "1":
|
|
1229
|
+
raise ValueError(
|
|
1230
|
+
"Unsupported reference format; rebuild with 'fusion-function prepare-data'"
|
|
1231
|
+
)
|
|
1232
|
+
if metadata.get("species") != SPECIES:
|
|
1233
|
+
raise ValueError("Only human reference databases are supported")
|
|
1234
|
+
if (
|
|
1235
|
+
metadata.get("assembly") != ASSEMBLY
|
|
1236
|
+
or not db.execute(
|
|
1237
|
+
"SELECT 1 FROM ensembl_coord_system WHERE version=? LIMIT 1", (ASSEMBLY,)
|
|
1238
|
+
).fetchone()
|
|
1239
|
+
):
|
|
1240
|
+
raise ValueError("Only GRCh38 reference databases are supported")
|
|
1241
|
+
if not re.fullmatch(r"[1-9]\d*", metadata.get("release", "")):
|
|
1242
|
+
raise ValueError("Reference metadata must contain a positive Ensembl release number")
|
|
1243
|
+
if (
|
|
1244
|
+
metadata.get("sequence_chunk_size") != str(CHUNK_SIZE)
|
|
1245
|
+
or metadata.get("sequence_codec") != "zlib"
|
|
1246
|
+
):
|
|
1247
|
+
raise ValueError("Unsupported sequence chunk format")
|
|
1248
|
+
if metadata.get("preprocessing_version") not in {None, "1"}:
|
|
1249
|
+
raise ValueError("Unsupported preprocessing version")
|
|
1250
|
+
kinds = {row[0] for row in db.execute("SELECT DISTINCT kind FROM sequences")}
|
|
1251
|
+
if not set(SEQUENCE_TYPES) <= kinds:
|
|
1252
|
+
raise ValueError(
|
|
1253
|
+
"Preprocessing requires DNA and peptide sequences; rebuild with 'fusion-function prepare-data'"
|
|
1254
|
+
)
|
|
1255
|
+
return metadata
|
|
1256
|
+
|
|
1257
|
+
|
|
1258
|
+
def interpro_entry_rows(path: Path) -> Iterator[tuple[str, str | None, str | None]]:
|
|
1259
|
+
"""Read current and archived InterPro entry lists with the same validation."""
|
|
1260
|
+
with input_progress(path, "Import InterPro", compressed=False, text=True) as (
|
|
1261
|
+
stream,
|
|
1262
|
+
bar,
|
|
1263
|
+
update,
|
|
1264
|
+
):
|
|
1265
|
+
header = stream.readline().rstrip("\r\n").split("\t")
|
|
1266
|
+
expected = ["ENTRY_AC", "ENTRY_TYPE", "ENTRY_NAME"]
|
|
1267
|
+
if not set(expected) <= set(header):
|
|
1268
|
+
raise ValueError("InterPro TSV must have ENTRY_AC, ENTRY_TYPE and ENTRY_NAME columns")
|
|
1269
|
+
indices = [header.index(name) for name in expected]
|
|
1270
|
+
for number, line in enumerate(stream, 2):
|
|
1271
|
+
if not line.strip():
|
|
1272
|
+
continue
|
|
1273
|
+
values = line.rstrip("\r\n").split("\t")
|
|
1274
|
+
if len(values) != len(header):
|
|
1275
|
+
raise ValueError(f"InterPro TSV line {number}: wrong field count")
|
|
1276
|
+
accession, entry_type, name = [values[index] for index in indices]
|
|
1277
|
+
if not re.fullmatch(r"IPR\d+", accession):
|
|
1278
|
+
raise ValueError(f"Invalid InterPro accession on line {number}")
|
|
1279
|
+
entry_type = entry_type.strip().lower().replace(" ", "_").replace("-", "_")
|
|
1280
|
+
yield accession, name or None, entry_type or None
|
|
1281
|
+
if number % 1000 == 0:
|
|
1282
|
+
update()
|
|
1283
|
+
|
|
1284
|
+
|
|
1285
|
+
def restore_interpro_history(
|
|
1286
|
+
db: sqlite3.Connection, metadata: InterProMetadata, cache: Path
|
|
1287
|
+
) -> tuple[dict[str, str], list[dict[str, str]]]:
|
|
1288
|
+
"""Supplement missing types, newest archive first; never overwrite current types.
|
|
1289
|
+
|
|
1290
|
+
Explicit local entry lists do not trigger archive downloads.
|
|
1291
|
+
"""
|
|
1292
|
+
# This small mapping table includes all possible core accessions. Only if
|
|
1293
|
+
# archives cannot resolve some do we scan the much larger feature table.
|
|
1294
|
+
required = {row[0] for row in db.execute("SELECT DISTINCT interpro_ac FROM ensembl_interpro")}
|
|
1295
|
+
missing = {accession for accession in required if not metadata.get(accession, (None, None))[1]}
|
|
1296
|
+
if not missing:
|
|
1297
|
+
return {}, []
|
|
1298
|
+
LOG.warning(
|
|
1299
|
+
"%s Ensembl InterPro accessions lack current metadata; searching archived entry lists before transcript preprocessing",
|
|
1300
|
+
len(missing),
|
|
1301
|
+
)
|
|
1302
|
+
versions = sorted(
|
|
1303
|
+
(name for name in listing(INTERPRO_RELEASES_URL) if re.fullmatch(r"\d+(?:\.\d+)?", name)),
|
|
1304
|
+
key=lambda name: tuple(int(part) for part in name.split(".")),
|
|
1305
|
+
reverse=True,
|
|
1306
|
+
)
|
|
1307
|
+
restored, sources = {}, []
|
|
1308
|
+
for version in versions:
|
|
1309
|
+
url = INTERPRO_RELEASES_URL + version + "/entry.list"
|
|
1310
|
+
try:
|
|
1311
|
+
path, digest = download(url, cache / "releases" / version)
|
|
1312
|
+
except urllib.error.HTTPError as exc:
|
|
1313
|
+
if exc.code == 404:
|
|
1314
|
+
continue
|
|
1315
|
+
raise
|
|
1316
|
+
recovered = []
|
|
1317
|
+
for accession, name, entry_type in interpro_entry_rows(path):
|
|
1318
|
+
if accession in missing and entry_type:
|
|
1319
|
+
recovered.append((accession, name, entry_type))
|
|
1320
|
+
metadata[accession] = name or metadata.get(accession, (None, None))[0], entry_type
|
|
1321
|
+
restored[accession] = version
|
|
1322
|
+
missing.remove(accession)
|
|
1323
|
+
if recovered:
|
|
1324
|
+
db.executemany(
|
|
1325
|
+
"INSERT INTO ff_interpro VALUES (?, ?, ?) ON CONFLICT(interpro_id) "
|
|
1326
|
+
"DO UPDATE SET name=COALESCE(excluded.name, ff_interpro.name), "
|
|
1327
|
+
"entry_type=excluded.entry_type",
|
|
1328
|
+
recovered,
|
|
1329
|
+
)
|
|
1330
|
+
sources.append({"release": version, "url": url, "sha256": digest})
|
|
1331
|
+
LOG.info(
|
|
1332
|
+
"Recovered %s historical InterPro entries from release %s; %s remaining",
|
|
1333
|
+
len(recovered),
|
|
1334
|
+
version,
|
|
1335
|
+
len(missing),
|
|
1336
|
+
)
|
|
1337
|
+
if not missing:
|
|
1338
|
+
break
|
|
1339
|
+
if missing:
|
|
1340
|
+
analyses = protein_analyses(db)
|
|
1341
|
+
used = {
|
|
1342
|
+
row[0]
|
|
1343
|
+
for row in sqlite_phase(
|
|
1344
|
+
db,
|
|
1345
|
+
"""
|
|
1346
|
+
SELECT DISTINCT i.interpro_ac, pf.analysis_id FROM ensembl_protein_feature pf
|
|
1347
|
+
JOIN ensembl_translation tr USING (translation_id)
|
|
1348
|
+
JOIN ensembl_transcript t ON t.transcript_id=tr.transcript_id
|
|
1349
|
+
AND t.canonical_translation_id=tr.translation_id
|
|
1350
|
+
JOIN ensembl_interpro i ON i.id=pf.hit_name
|
|
1351
|
+
WHERE t.is_current=1
|
|
1352
|
+
""",
|
|
1353
|
+
"Check unresolved InterPro metadata",
|
|
1354
|
+
)
|
|
1355
|
+
if (analyses.get(row[1], (None, None))[0] or "").lower() not in STRUCTURE_SOURCES
|
|
1356
|
+
}
|
|
1357
|
+
missing.intersection_update(used)
|
|
1358
|
+
if missing:
|
|
1359
|
+
examples = ", ".join(sorted(missing)[:10])
|
|
1360
|
+
raise ValueError(
|
|
1361
|
+
f"Missing InterPro entry types for {len(missing)} accession(s) after checking archived metadata: "
|
|
1362
|
+
f"{examples}. Supply a complete --interpro-entries FILE; transcript preprocessing has not started."
|
|
1363
|
+
)
|
|
1364
|
+
return restored, sources
|
|
1365
|
+
|
|
1366
|
+
|
|
1367
|
+
# Derived-table preparation: one transaction, one transcript at a time.
|
|
1368
|
+
|
|
1369
|
+
|
|
1370
|
+
def _record_transcript_error(
|
|
1371
|
+
summary: dict[str, TranscriptErrorSummary], transcript_id: str, message: str
|
|
1372
|
+
) -> None:
|
|
1373
|
+
"""Group errors by reason while keeping variable details in the stored payload."""
|
|
1374
|
+
# CDS/peptide mismatch lengths and unsupported assembly names vary, so group
|
|
1375
|
+
# by the fixed message prefix. Each original full error remains retrievable.
|
|
1376
|
+
reason = message.partition(":")[0]
|
|
1377
|
+
entry = summary.setdefault(reason, {"count": 0, "example_transcripts": []})
|
|
1378
|
+
entry["count"] += 1
|
|
1379
|
+
if len(entry["example_transcripts"]) < 3:
|
|
1380
|
+
entry["example_transcripts"].append(transcript_id)
|
|
1381
|
+
|
|
1382
|
+
|
|
1383
|
+
def uniprot_lookup_fingerprint(
|
|
1384
|
+
db: sqlite3.Connection, metadata: Mapping[str, str], input_sha256: str
|
|
1385
|
+
) -> str | None:
|
|
1386
|
+
"""Key verified lookups by XML, mapping code and immutable imported sources.
|
|
1387
|
+
|
|
1388
|
+
Source checksums describe the raw tables and sequences imported by the
|
|
1389
|
+
builder. References without this provenance are reverified, never reused.
|
|
1390
|
+
"""
|
|
1391
|
+
if not db.execute(
|
|
1392
|
+
"SELECT 1 FROM sqlite_master WHERE type='table' AND name='source_files'"
|
|
1393
|
+
).fetchone():
|
|
1394
|
+
return None
|
|
1395
|
+
sources = list(db.execute("SELECT url, sha256, bytes FROM source_files ORDER BY url"))
|
|
1396
|
+
if not sources or not all(row[1] for row in sources):
|
|
1397
|
+
return None
|
|
1398
|
+
fingerprint = {
|
|
1399
|
+
"uniprot_sha256": input_sha256,
|
|
1400
|
+
"mapping_sha256": sha256_file(Path(__file__).with_name("uniprot.py")),
|
|
1401
|
+
"ensembl_sources": sources,
|
|
1402
|
+
"reference": {
|
|
1403
|
+
key: metadata.get(key)
|
|
1404
|
+
for key in (
|
|
1405
|
+
"format_version",
|
|
1406
|
+
"release",
|
|
1407
|
+
"species",
|
|
1408
|
+
"assembly",
|
|
1409
|
+
"sequence_codec",
|
|
1410
|
+
"sequence_chunk_size",
|
|
1411
|
+
)
|
|
1412
|
+
},
|
|
1413
|
+
}
|
|
1414
|
+
return hashlib.sha256(json.dumps(fingerprint, sort_keys=True).encode()).hexdigest()
|
|
1415
|
+
|
|
1416
|
+
|
|
1417
|
+
def encode_transcript_payload(payload: ProteinFeatureResponse) -> str | bytes:
|
|
1418
|
+
"""Compress large models while keeping small error records SQL-readable.
|
|
1419
|
+
|
|
1420
|
+
Level 1 limits preprocessing CPU cost. This is lossless storage compression;
|
|
1421
|
+
it does not remove features or change the annotation API's returned model.
|
|
1422
|
+
"""
|
|
1423
|
+
encoded = json.dumps(payload, separators=(",", ":"))
|
|
1424
|
+
return encoded if "error" in payload else zlib.compress(encoded.encode("utf-8"), level=1)
|
|
1425
|
+
|
|
1426
|
+
|
|
1427
|
+
def decode_transcript_payload(encoded: str | bytes) -> ProteinFeatureResponse:
|
|
1428
|
+
"""Decode compressed models or plain JSON errors and existing references."""
|
|
1429
|
+
if isinstance(encoded, bytes):
|
|
1430
|
+
encoded = zlib.decompress(encoded).decode("utf-8")
|
|
1431
|
+
return cast("ProteinFeatureResponse", json.loads(encoded))
|
|
1432
|
+
|
|
1433
|
+
|
|
1434
|
+
def preprocess_reference(
|
|
1435
|
+
db: sqlite3.Connection,
|
|
1436
|
+
interpro_entries: Path | None = None,
|
|
1437
|
+
*,
|
|
1438
|
+
interpro_archive_cache: Path | None = None,
|
|
1439
|
+
panther_classifications: Path | None = None,
|
|
1440
|
+
panther_cache: Path | None = None,
|
|
1441
|
+
uniprot_features: Path | None = None,
|
|
1442
|
+
uniprot_source: dict[str, str] | None = None,
|
|
1443
|
+
) -> dict[str, int]:
|
|
1444
|
+
"""Build all derived tables in one transaction; leave prior tables on failure.
|
|
1445
|
+
|
|
1446
|
+
Requires raw annotation tables and DNA/peptide FASTA. Invalid individual
|
|
1447
|
+
transcript models are recorded as errors, rather than silently corrected.
|
|
1448
|
+
No nucleotide sequences are duplicated in the derived transcript records.
|
|
1449
|
+
"""
|
|
1450
|
+
from .uniprot import CuratedFeature, import_features
|
|
1451
|
+
|
|
1452
|
+
metadata = validate_reference_source(db)
|
|
1453
|
+
analyses = protein_analyses(db)
|
|
1454
|
+
panther_source = None
|
|
1455
|
+
if panther_classifications is None and panther_cache is not None:
|
|
1456
|
+
version = panther_release(analyses)
|
|
1457
|
+
if version:
|
|
1458
|
+
panther_classifications, digest, url = download_panther_classifications(
|
|
1459
|
+
version, panther_cache
|
|
1460
|
+
)
|
|
1461
|
+
panther_source = {"version": version, "url": url, "sha256": digest}
|
|
1462
|
+
if uniprot_features is not None and uniprot_source is None:
|
|
1463
|
+
uniprot_source = {
|
|
1464
|
+
"path": str(uniprot_features.resolve()),
|
|
1465
|
+
"sha256": sha256_file(uniprot_features),
|
|
1466
|
+
}
|
|
1467
|
+
uniprot_fingerprint = (
|
|
1468
|
+
uniprot_lookup_fingerprint(db, metadata, uniprot_source["sha256"])
|
|
1469
|
+
if uniprot_features is not None and uniprot_source is not None
|
|
1470
|
+
else None
|
|
1471
|
+
)
|
|
1472
|
+
if panther_classifications is not None and panther_source is None:
|
|
1473
|
+
panther_source = {
|
|
1474
|
+
"path": str(panther_classifications.resolve()),
|
|
1475
|
+
"sha256": sha256_file(panther_classifications),
|
|
1476
|
+
}
|
|
1477
|
+
LOG.info("Preprocessing reference transcript models, domains and splice sites")
|
|
1478
|
+
db.commit()
|
|
1479
|
+
# These are reproducible public reference annotations. Some SQLite builds
|
|
1480
|
+
# default to zeroing deleted content, which rewrites entire large tables and
|
|
1481
|
+
# their rollback journals. Reclaim pages without scrubbing; journaling and
|
|
1482
|
+
# atomic rollback remain enabled. Restore the caller's setting on every exit.
|
|
1483
|
+
secure_delete = int(db.execute("PRAGMA main.secure_delete").fetchone()[0])
|
|
1484
|
+
# FAST reads back as 2, but setting the numeric value 2 means ON. Restore
|
|
1485
|
+
# the keyword so a caller using FAST keeps that exact policy.
|
|
1486
|
+
previous_secure_delete = ("OFF", "ON", "FAST")[secure_delete]
|
|
1487
|
+
db.execute("PRAGMA main.secure_delete=OFF")
|
|
1488
|
+
# Derived-table replacement, including DDL, is atomic. A failed metadata
|
|
1489
|
+
# lookup or model build rolls back to the prior usable reference tables.
|
|
1490
|
+
transcript_progress = None
|
|
1491
|
+
try:
|
|
1492
|
+
LOG.info(
|
|
1493
|
+
"SQLite reference preprocessing: secure_delete=%s (previously %s), "
|
|
1494
|
+
"journal_mode=%s, synchronous=%s, auto_vacuum=%s, page_size=%s",
|
|
1495
|
+
db.execute("PRAGMA main.secure_delete").fetchone()[0],
|
|
1496
|
+
previous_secure_delete,
|
|
1497
|
+
db.execute("PRAGMA main.journal_mode").fetchone()[0],
|
|
1498
|
+
db.execute("PRAGMA main.synchronous").fetchone()[0],
|
|
1499
|
+
db.execute("PRAGMA main.auto_vacuum").fetchone()[0],
|
|
1500
|
+
db.execute("PRAGMA main.page_size").fetchone()[0],
|
|
1501
|
+
)
|
|
1502
|
+
db.execute("BEGIN IMMEDIATE")
|
|
1503
|
+
if uniprot_features is not None:
|
|
1504
|
+
if (
|
|
1505
|
+
uniprot_fingerprint is not None
|
|
1506
|
+
and uniprot_fingerprint == metadata.get("uniprot_lookup_fingerprint")
|
|
1507
|
+
and metadata.get("uniprot_counts") is not None
|
|
1508
|
+
and db.execute(
|
|
1509
|
+
"SELECT 1 FROM sqlite_master WHERE type='table' AND name='ff_uniprot_features'"
|
|
1510
|
+
).fetchone()
|
|
1511
|
+
):
|
|
1512
|
+
LOG.info("Reusing sequence-verified UniProt lookups: inputs and mapping unchanged")
|
|
1513
|
+
uniprot_counts = json.loads(metadata["uniprot_counts"])
|
|
1514
|
+
else:
|
|
1515
|
+
uniprot_counts = import_features(db, uniprot_features)
|
|
1516
|
+
else:
|
|
1517
|
+
uniprot_counts = None
|
|
1518
|
+
# An offline low-level call can reuse already verified UniProt lookups.
|
|
1519
|
+
db.execute(
|
|
1520
|
+
"CREATE TABLE IF NOT EXISTS ff_uniprot_features (sequence_id TEXT NOT NULL, "
|
|
1521
|
+
"feature_id TEXT NOT NULL, payload TEXT NOT NULL, "
|
|
1522
|
+
"PRIMARY KEY(sequence_id, feature_id)) WITHOUT ROWID"
|
|
1523
|
+
)
|
|
1524
|
+
# Persist the bulk lookup so subsequent offline preprocessing can reuse it.
|
|
1525
|
+
# Import and all derived changes participate in the same rollback boundary.
|
|
1526
|
+
db.execute(
|
|
1527
|
+
"CREATE TABLE IF NOT EXISTS ff_panther (accession TEXT PRIMARY KEY, "
|
|
1528
|
+
"subfamily_id TEXT NOT NULL, name TEXT NOT NULL) WITHOUT ROWID"
|
|
1529
|
+
)
|
|
1530
|
+
if panther_classifications is not None:
|
|
1531
|
+
db.execute("DELETE FROM ff_panther")
|
|
1532
|
+
db.executemany(
|
|
1533
|
+
"INSERT INTO ff_panther VALUES (?, ?, ?)",
|
|
1534
|
+
panther_entry_rows(panther_classifications),
|
|
1535
|
+
)
|
|
1536
|
+
with sqlite_activity("Match PANTHER subfamilies to Ensembl proteins"):
|
|
1537
|
+
panther_annotations = (
|
|
1538
|
+
panther_translation_annotations(db)
|
|
1539
|
+
if db.execute("SELECT 1 FROM ff_panther LIMIT 1").fetchone()
|
|
1540
|
+
else {}
|
|
1541
|
+
)
|
|
1542
|
+
LOG.info(
|
|
1543
|
+
"PANTHER subfamily assignments matched %s Ensembl translations by protein cross-reference",
|
|
1544
|
+
f"{len(panther_annotations):,}",
|
|
1545
|
+
)
|
|
1546
|
+
with sqlite_activity("Preserve existing InterPro metadata"):
|
|
1547
|
+
has_interpro = db.execute(
|
|
1548
|
+
"SELECT 1 FROM sqlite_master WHERE type='table' AND name='ff_interpro'"
|
|
1549
|
+
).fetchone()
|
|
1550
|
+
previous_interpro = (
|
|
1551
|
+
list(db.execute("SELECT interpro_id, name, entry_type FROM ff_interpro"))
|
|
1552
|
+
if has_interpro
|
|
1553
|
+
else []
|
|
1554
|
+
)
|
|
1555
|
+
for table in ("ff_transcripts", "ff_interpro"):
|
|
1556
|
+
with sqlite_activity("Drop previous " + table):
|
|
1557
|
+
db.execute("DROP TABLE IF EXISTS " + sql_name(table))
|
|
1558
|
+
with sqlite_activity("Create derived transcript and InterPro tables"):
|
|
1559
|
+
# Large model payloads belong in ordinary rowid-table leaves, not
|
|
1560
|
+
# the intermediate nodes of a WITHOUT ROWID primary-key tree.
|
|
1561
|
+
# Internal transcript-ID order then appends model rows sequentially;
|
|
1562
|
+
# only the much smaller stable-ID index needs random inserts.
|
|
1563
|
+
db.execute(
|
|
1564
|
+
"CREATE TABLE ff_transcripts (transcript_id TEXT NOT NULL PRIMARY KEY, version INTEGER, "
|
|
1565
|
+
"translation_id TEXT, translation_version INTEGER, status TEXT NOT NULL, "
|
|
1566
|
+
"payload BLOB NOT NULL)"
|
|
1567
|
+
)
|
|
1568
|
+
db.execute(
|
|
1569
|
+
"CREATE TABLE ff_interpro (interpro_id TEXT PRIMARY KEY, name TEXT, entry_type TEXT) WITHOUT ROWID"
|
|
1570
|
+
)
|
|
1571
|
+
# Priority: core names < previously stored types < supplied/current list.
|
|
1572
|
+
# History fills only remaining gaps; it never overwrites a known type.
|
|
1573
|
+
with sqlite_activity("Load InterPro names from Ensembl cross-references"):
|
|
1574
|
+
db.execute(
|
|
1575
|
+
"INSERT INTO ff_interpro SELECT dbprimary_acc, "
|
|
1576
|
+
"COALESCE(MAX(NULLIF(description,'')), MAX(NULLIF(display_label,''))), NULL "
|
|
1577
|
+
"FROM ensembl_xref WHERE dbprimary_acc LIKE 'IPR%' GROUP BY dbprimary_acc"
|
|
1578
|
+
)
|
|
1579
|
+
db.executemany(
|
|
1580
|
+
"INSERT INTO ff_interpro VALUES (?, ?, ?) ON CONFLICT(interpro_id) "
|
|
1581
|
+
"DO UPDATE SET name=COALESCE(excluded.name, ff_interpro.name), "
|
|
1582
|
+
"entry_type=excluded.entry_type",
|
|
1583
|
+
previous_interpro,
|
|
1584
|
+
)
|
|
1585
|
+
provided_accessions = set()
|
|
1586
|
+
if interpro_entries:
|
|
1587
|
+
LOG.info("Importing InterPro entry names/types from %s", interpro_entries)
|
|
1588
|
+
for row in interpro_entry_rows(interpro_entries):
|
|
1589
|
+
provided_accessions.add(row[0])
|
|
1590
|
+
db.execute(
|
|
1591
|
+
"INSERT INTO ff_interpro VALUES (?, ?, ?) ON CONFLICT(interpro_id) "
|
|
1592
|
+
"DO UPDATE SET name=excluded.name, entry_type=excluded.entry_type",
|
|
1593
|
+
row,
|
|
1594
|
+
)
|
|
1595
|
+
ipr_metadata = {row[0]: (row[1], row[2]) for row in db.execute("SELECT * FROM ff_interpro")}
|
|
1596
|
+
historical_entries = json.loads(metadata.get("interpro_historical_entries", "{}"))
|
|
1597
|
+
historical_sources = json.loads(metadata.get("interpro_historical_sources", "[]"))
|
|
1598
|
+
# Entries explicitly present in the new list take precedence over history.
|
|
1599
|
+
for accession in provided_accessions:
|
|
1600
|
+
historical_entries.pop(accession, None)
|
|
1601
|
+
if interpro_archive_cache is not None:
|
|
1602
|
+
restored, sources = restore_interpro_history(db, ipr_metadata, interpro_archive_cache)
|
|
1603
|
+
historical_entries.update(restored)
|
|
1604
|
+
historical_sources.extend(
|
|
1605
|
+
source for source in sources if source not in historical_sources
|
|
1606
|
+
)
|
|
1607
|
+
with sqlite_activity("Read genome and protein sequence lengths"):
|
|
1608
|
+
genome_lengths = dict(
|
|
1609
|
+
db.execute("SELECT sequence_id, length FROM sequences WHERE kind='dna'")
|
|
1610
|
+
)
|
|
1611
|
+
protein_lengths: dict[str, list[tuple[str, int]]] = {}
|
|
1612
|
+
for stable_id, sequence_id, length in db.execute(
|
|
1613
|
+
"SELECT stable_id, sequence_id, length FROM sequences WHERE kind='pep'"
|
|
1614
|
+
):
|
|
1615
|
+
protein_lengths.setdefault(stable_id, []).append((sequence_id, length))
|
|
1616
|
+
curated_features = GroupStream(
|
|
1617
|
+
dict_rows(
|
|
1618
|
+
db,
|
|
1619
|
+
"""
|
|
1620
|
+
SELECT t.transcript_id, u.payload
|
|
1621
|
+
FROM ensembl_transcript t JOIN ensembl_translation tr
|
|
1622
|
+
ON tr.translation_id=t.canonical_translation_id
|
|
1623
|
+
JOIN ff_uniprot_features u ON u.sequence_id IN
|
|
1624
|
+
(tr.stable_id, tr.stable_id || '.' || tr.version)
|
|
1625
|
+
WHERE t.is_current=1 ORDER BY t.transcript_id, u.feature_id
|
|
1626
|
+
""",
|
|
1627
|
+
description="Prepare ordered UniProt feature stream",
|
|
1628
|
+
)
|
|
1629
|
+
)
|
|
1630
|
+
# All three queries use the same ascending internal ID order. GroupStream
|
|
1631
|
+
# avoids per-transcript SQL queries and loading all exon/features into RAM.
|
|
1632
|
+
exons = GroupStream(
|
|
1633
|
+
dict_rows(
|
|
1634
|
+
db,
|
|
1635
|
+
"""
|
|
1636
|
+
SELECT et.transcript_id, et.rank, e.exon_id, e.seq_region_id,
|
|
1637
|
+
e.seq_region_start AS start, e.seq_region_end AS end,
|
|
1638
|
+
e.seq_region_strand AS strand, e.phase
|
|
1639
|
+
FROM ensembl_exon_transcript et JOIN ensembl_exon e USING (exon_id)
|
|
1640
|
+
ORDER BY et.transcript_id, et.rank
|
|
1641
|
+
""",
|
|
1642
|
+
description="Prepare ordered exon stream",
|
|
1643
|
+
)
|
|
1644
|
+
)
|
|
1645
|
+
# Restrict hits to each current transcript's canonical translation; using
|
|
1646
|
+
# alternate translations here would mix incompatible peptide coordinates.
|
|
1647
|
+
features = GroupStream(
|
|
1648
|
+
dict_rows(
|
|
1649
|
+
db,
|
|
1650
|
+
"""
|
|
1651
|
+
SELECT tr.transcript_id, pf.protein_feature_id, pf.seq_start AS start,
|
|
1652
|
+
pf.seq_end AS end, pf.hit_name AS feature_id,
|
|
1653
|
+
pf.hit_description AS description, a.logic_name AS source, pf.analysis_id,
|
|
1654
|
+
i.interpro_ac AS interpro_id
|
|
1655
|
+
FROM ensembl_protein_feature pf
|
|
1656
|
+
JOIN ensembl_translation tr USING (translation_id)
|
|
1657
|
+
JOIN ensembl_transcript t ON t.transcript_id=tr.transcript_id
|
|
1658
|
+
AND t.canonical_translation_id=tr.translation_id
|
|
1659
|
+
LEFT JOIN ensembl_analysis a USING (analysis_id)
|
|
1660
|
+
LEFT JOIN ensembl_interpro i ON i.id=pf.hit_name
|
|
1661
|
+
WHERE t.is_current=1
|
|
1662
|
+
ORDER BY tr.transcript_id, pf.protein_feature_id, i.interpro_ac
|
|
1663
|
+
""",
|
|
1664
|
+
description="Prepare ordered protein feature stream",
|
|
1665
|
+
)
|
|
1666
|
+
)
|
|
1667
|
+
transcripts = dict_rows(
|
|
1668
|
+
db,
|
|
1669
|
+
"""
|
|
1670
|
+
SELECT t.transcript_id AS internal_id, t.stable_id AS transcript_id,
|
|
1671
|
+
t.version, t.seq_region_id, t.seq_region_start AS start,
|
|
1672
|
+
t.seq_region_end AS end, t.seq_region_strand AS strand,
|
|
1673
|
+
t.canonical_translation_id, r.name AS chromosome,
|
|
1674
|
+
cs.version AS assembly_name, tr.stable_id AS translation_id,
|
|
1675
|
+
tr.version AS translation_version, tr.seq_start, tr.start_exon_id,
|
|
1676
|
+
tr.seq_end, tr.end_exon_id,
|
|
1677
|
+
EXISTS(SELECT 1 FROM ensembl_translation alt
|
|
1678
|
+
WHERE alt.transcript_id=t.transcript_id) AS has_translation
|
|
1679
|
+
FROM ensembl_transcript t JOIN ensembl_seq_region r USING (seq_region_id)
|
|
1680
|
+
JOIN ensembl_coord_system cs USING (coord_system_id)
|
|
1681
|
+
LEFT JOIN ensembl_translation tr ON tr.translation_id=t.canonical_translation_id
|
|
1682
|
+
WHERE t.is_current=1 ORDER BY t.transcript_id
|
|
1683
|
+
""",
|
|
1684
|
+
description="Prepare transcript stream",
|
|
1685
|
+
)
|
|
1686
|
+
counts = {"ready": 0, "noncoding": 0, "error": 0, "protein_features": 0}
|
|
1687
|
+
errors: dict[str, TranscriptErrorSummary] = {}
|
|
1688
|
+
missing_entry_types: set[str] = set()
|
|
1689
|
+
total = sqlite_phase(
|
|
1690
|
+
db,
|
|
1691
|
+
"SELECT COUNT(*) FROM ensembl_transcript t "
|
|
1692
|
+
"JOIN ensembl_seq_region r USING (seq_region_id) "
|
|
1693
|
+
"JOIN ensembl_coord_system cs USING (coord_system_id) "
|
|
1694
|
+
"WHERE t.is_current=1",
|
|
1695
|
+
"Count current transcripts",
|
|
1696
|
+
)[0][0]
|
|
1697
|
+
transcript_progress = progress(
|
|
1698
|
+
transcripts, total=total, desc="Preprocess transcripts", unit="transcript"
|
|
1699
|
+
)
|
|
1700
|
+
payload: ProteinFeatureResponse
|
|
1701
|
+
for number, transcript in enumerate(transcript_progress, 1):
|
|
1702
|
+
exon_rows = exons.take(transcript["internal_id"])
|
|
1703
|
+
feature_rows = features.take(transcript["internal_id"])
|
|
1704
|
+
curated_rows = curated_features.take(transcript["internal_id"])
|
|
1705
|
+
translation = None
|
|
1706
|
+
if transcript["canonical_translation_id"] is not None:
|
|
1707
|
+
translation = {**transcript, "stable_id": transcript["translation_id"]}
|
|
1708
|
+
try:
|
|
1709
|
+
if transcript["assembly_name"] != ASSEMBLY:
|
|
1710
|
+
raise ValueError(
|
|
1711
|
+
f"Unsupported transcript coordinate system: {transcript['assembly_name']}"
|
|
1712
|
+
)
|
|
1713
|
+
if translation is not None and not translation["stable_id"]:
|
|
1714
|
+
raise ValueError("Canonical translation is missing from source table")
|
|
1715
|
+
if translation is None and transcript["has_translation"]:
|
|
1716
|
+
raise ValueError("Transcript has translations but no canonical translation")
|
|
1717
|
+
payload = reference_structure(transcript, exon_rows, translation)
|
|
1718
|
+
if transcript["chromosome"] not in genome_lengths:
|
|
1719
|
+
raise ValueError("Transcript sequence region is absent from genomic FASTA")
|
|
1720
|
+
if transcript["end"] > genome_lengths[transcript["chromosome"]]:
|
|
1721
|
+
raise ValueError("Transcript span exceeds genomic FASTA sequence")
|
|
1722
|
+
status = "noncoding" if translation is None else "ready"
|
|
1723
|
+
if translation:
|
|
1724
|
+
candidates = protein_lengths.get(translation["stable_id"], [])
|
|
1725
|
+
expected_id = f"{translation['stable_id']}.{transcript['translation_version']}"
|
|
1726
|
+
candidates = [
|
|
1727
|
+
record
|
|
1728
|
+
for record in candidates
|
|
1729
|
+
if record[0] in {expected_id, translation["stable_id"]}
|
|
1730
|
+
]
|
|
1731
|
+
if len(candidates) != 1:
|
|
1732
|
+
raise ValueError(
|
|
1733
|
+
"Canonical translation peptide is missing/ambiguous or has a different version"
|
|
1734
|
+
)
|
|
1735
|
+
protein_length = candidates[0][1]
|
|
1736
|
+
payload["protein_length"] = protein_length
|
|
1737
|
+
cds_length = sum(
|
|
1738
|
+
block["cds_end"] - block["cds_start"] + 1 for block in payload["cds_blocks"]
|
|
1739
|
+
)
|
|
1740
|
+
# Include Ensembl's virtual leading codon bases in the length
|
|
1741
|
+
# comparison, without adding them to genomic CDS blocks.
|
|
1742
|
+
# Peptide FASTA omits the optional terminal stop codon.
|
|
1743
|
+
start_phase = payload.get("cds_start_phase", 0)
|
|
1744
|
+
if cds_length + start_phase not in {protein_length * 3, protein_length * 3 + 3}:
|
|
1745
|
+
raise ValueError(
|
|
1746
|
+
f"CDS/peptide length mismatch: {cds_length} bp, {protein_length} aa, "
|
|
1747
|
+
f"start phase {start_phase}"
|
|
1748
|
+
)
|
|
1749
|
+
block_ends = [block["cds_end"] for block in payload["cds_blocks"]]
|
|
1750
|
+
# Multiple signatures/InterPro mappings can share an interval.
|
|
1751
|
+
# Reuse coordinates within this transcript, then discard them.
|
|
1752
|
+
segment_cache: dict[tuple[int, int], tuple[list[GenomicSegment], int, int]] = {}
|
|
1753
|
+
for feature in feature_rows:
|
|
1754
|
+
source = analyses.get(feature["analysis_id"], (feature["source"], None))[0]
|
|
1755
|
+
if (source or "").lower() in STRUCTURE_SOURCES:
|
|
1756
|
+
# Whole-protein structure mappings are not domains/sites.
|
|
1757
|
+
# Exclude them before validating functional coordinates:
|
|
1758
|
+
# their intervals may exceed this transcript's peptide.
|
|
1759
|
+
continue
|
|
1760
|
+
if not 1 <= feature["start"] <= feature["end"] <= protein_length:
|
|
1761
|
+
raise ValueError(
|
|
1762
|
+
f"Protein feature falls outside peptide: {source} "
|
|
1763
|
+
f"{feature['feature_id']} {feature['start']}-{feature['end']}, "
|
|
1764
|
+
f"peptide length {protein_length}"
|
|
1765
|
+
)
|
|
1766
|
+
interval = feature["start"], feature["end"]
|
|
1767
|
+
mapped = segment_cache.get(interval)
|
|
1768
|
+
if mapped is None:
|
|
1769
|
+
segments = protein_segments(
|
|
1770
|
+
*interval, payload["cds_blocks"], block_ends=block_ends
|
|
1771
|
+
)
|
|
1772
|
+
mapped = (
|
|
1773
|
+
segments,
|
|
1774
|
+
min(segment["start"] for segment in segments),
|
|
1775
|
+
max(segment["end"] for segment in segments),
|
|
1776
|
+
)
|
|
1777
|
+
segment_cache[interval] = mapped
|
|
1778
|
+
segments, genomic_start, genomic_end = mapped
|
|
1779
|
+
name, entry_type = ipr_metadata.get(feature["interpro_id"], (None, None))
|
|
1780
|
+
subfamily = (
|
|
1781
|
+
feature["feature_id"]
|
|
1782
|
+
if (source or "").upper() == "PANTHER"
|
|
1783
|
+
and ":SF" in (feature["feature_id"] or "")
|
|
1784
|
+
else None
|
|
1785
|
+
)
|
|
1786
|
+
subfamily_name = feature["description"] if subfamily else None
|
|
1787
|
+
if source == "PANTHER":
|
|
1788
|
+
assignment = panther_annotations.get(
|
|
1789
|
+
transcript["canonical_translation_id"]
|
|
1790
|
+
)
|
|
1791
|
+
hit = feature["feature_id"] or ""
|
|
1792
|
+
if assignment and assignment[0].split(":")[0] == hit.split(":")[0]:
|
|
1793
|
+
if subfamily is None or subfamily == assignment[0]:
|
|
1794
|
+
subfamily, subfamily_name = assignment
|
|
1795
|
+
payload["protein_features"].append(
|
|
1796
|
+
{
|
|
1797
|
+
"feature_id": feature["feature_id"] or None,
|
|
1798
|
+
"source": source or None,
|
|
1799
|
+
"interpro_id": feature["interpro_id"],
|
|
1800
|
+
"start": feature["start"],
|
|
1801
|
+
"end": feature["end"],
|
|
1802
|
+
"cds_start": (feature["start"] - 1) * 3 + 1,
|
|
1803
|
+
"cds_end": feature["end"] * 3,
|
|
1804
|
+
"chromosome": transcript["chromosome"],
|
|
1805
|
+
"strand": transcript["strand"],
|
|
1806
|
+
"assembly_name": transcript["assembly_name"],
|
|
1807
|
+
"genomic_start": genomic_start,
|
|
1808
|
+
"genomic_end": genomic_end,
|
|
1809
|
+
"genomic_segments": segments,
|
|
1810
|
+
"description": feature["description"] or name,
|
|
1811
|
+
"interpro_name": name,
|
|
1812
|
+
"interpro_entry_type": entry_type,
|
|
1813
|
+
"panther_subfamily_id": subfamily,
|
|
1814
|
+
"panther_subfamily_description": subfamily_name,
|
|
1815
|
+
}
|
|
1816
|
+
)
|
|
1817
|
+
for curated_row in curated_rows:
|
|
1818
|
+
curated = cast(CuratedFeature, json.loads(curated_row["payload"]))
|
|
1819
|
+
start, end = curated["start"], curated["end"]
|
|
1820
|
+
if not 1 <= start <= end <= protein_length:
|
|
1821
|
+
raise ValueError("Verified UniProt feature falls outside peptide")
|
|
1822
|
+
segments = protein_segments(
|
|
1823
|
+
start, end, payload["cds_blocks"], block_ends=block_ends
|
|
1824
|
+
)
|
|
1825
|
+
payload["protein_features"].append(
|
|
1826
|
+
{
|
|
1827
|
+
"feature_id": curated["feature_id"],
|
|
1828
|
+
"feature_type": curated["feature_type"],
|
|
1829
|
+
"description": curated["description"],
|
|
1830
|
+
"start": start,
|
|
1831
|
+
"end": end,
|
|
1832
|
+
"uniprot_accession": curated["uniprot_accession"],
|
|
1833
|
+
"uniprot_isoform": curated["uniprot_isoform"],
|
|
1834
|
+
"evidence": curated["evidence"],
|
|
1835
|
+
"source": "UniProtKB",
|
|
1836
|
+
"interpro_id": None,
|
|
1837
|
+
"cds_start": (start - 1) * 3 + 1,
|
|
1838
|
+
"cds_end": end * 3,
|
|
1839
|
+
"chromosome": transcript["chromosome"],
|
|
1840
|
+
"strand": transcript["strand"],
|
|
1841
|
+
"assembly_name": transcript["assembly_name"],
|
|
1842
|
+
"genomic_start": min(segment["start"] for segment in segments),
|
|
1843
|
+
"genomic_end": max(segment["end"] for segment in segments),
|
|
1844
|
+
"genomic_segments": segments,
|
|
1845
|
+
"interpro_name": None,
|
|
1846
|
+
"interpro_entry_type": None,
|
|
1847
|
+
"panther_subfamily_id": None,
|
|
1848
|
+
"panther_subfamily_description": None,
|
|
1849
|
+
}
|
|
1850
|
+
)
|
|
1851
|
+
counts["protein_features"] += len(payload["protein_features"])
|
|
1852
|
+
except ValueError as exc:
|
|
1853
|
+
# Preserve the reason for unsupported models so readers can explain
|
|
1854
|
+
# the failure, rather than reporting that the transcript is absent.
|
|
1855
|
+
status, payload = "error", {"error": f"{transcript['transcript_id']}: {exc}"}
|
|
1856
|
+
_record_transcript_error(errors, transcript["transcript_id"], str(exc))
|
|
1857
|
+
if status == "ready":
|
|
1858
|
+
payload = cast("TranscriptProteinFeatureResult", payload)
|
|
1859
|
+
missing_entry_types.update(
|
|
1860
|
+
feature["interpro_id"]
|
|
1861
|
+
for feature in payload["protein_features"]
|
|
1862
|
+
if feature["interpro_id"] and not feature["interpro_entry_type"]
|
|
1863
|
+
)
|
|
1864
|
+
counts[status] += 1
|
|
1865
|
+
db.execute(
|
|
1866
|
+
"INSERT INTO ff_transcripts VALUES (?, ?, ?, ?, ?, ?)",
|
|
1867
|
+
(
|
|
1868
|
+
transcript["transcript_id"],
|
|
1869
|
+
transcript["version"],
|
|
1870
|
+
transcript["translation_id"],
|
|
1871
|
+
transcript["translation_version"],
|
|
1872
|
+
status,
|
|
1873
|
+
encode_transcript_payload(payload),
|
|
1874
|
+
),
|
|
1875
|
+
)
|
|
1876
|
+
if number % 1000 == 0:
|
|
1877
|
+
transcript_progress.set_postfix(
|
|
1878
|
+
ready=counts["ready"], errors=counts["error"], refresh=False
|
|
1879
|
+
)
|
|
1880
|
+
if missing_entry_types:
|
|
1881
|
+
examples = ", ".join(sorted(missing_entry_types)[:10])
|
|
1882
|
+
raise ValueError(
|
|
1883
|
+
f"Missing InterPro entry types for {len(missing_entry_types)} accession(s): {examples}. "
|
|
1884
|
+
"Supply a complete --interpro-entries FILE or fetch compatible metadata; "
|
|
1885
|
+
"the previous reference has been preserved."
|
|
1886
|
+
)
|
|
1887
|
+
with sqlite_activity("Index prepared transcripts"):
|
|
1888
|
+
db.execute("CREATE INDEX ff_transcript_translation ON ff_transcripts (translation_id)")
|
|
1889
|
+
db.execute("CREATE INDEX ff_transcript_status ON ff_transcripts (status)")
|
|
1890
|
+
# Keep provenance in the same transaction as the tables it describes.
|
|
1891
|
+
errors = dict(sorted(errors.items(), key=lambda item: (-item[1]["count"], item[0])))
|
|
1892
|
+
preprocessing_metadata = {
|
|
1893
|
+
"preprocessing_version": "1",
|
|
1894
|
+
"feature_annotation_version": "3",
|
|
1895
|
+
"cds_mapping_version": "2",
|
|
1896
|
+
"transcript_payload_codec": TRANSCRIPT_PAYLOAD_CODEC,
|
|
1897
|
+
"preprocessing_counts": json.dumps(counts),
|
|
1898
|
+
"preprocessing_errors": json.dumps(errors),
|
|
1899
|
+
"preprocessed_utc": datetime.now(timezone.utc).isoformat(),
|
|
1900
|
+
"interpro_entries_sha256": (
|
|
1901
|
+
sha256_file(interpro_entries)
|
|
1902
|
+
if interpro_entries
|
|
1903
|
+
else metadata.get("interpro_entries_sha256", "")
|
|
1904
|
+
),
|
|
1905
|
+
"interpro_entry_types_loaded": str(
|
|
1906
|
+
any(value[1] for value in ipr_metadata.values())
|
|
1907
|
+
).lower(),
|
|
1908
|
+
"interpro_historical_entries": json.dumps(historical_entries, sort_keys=True),
|
|
1909
|
+
"interpro_historical_sources": json.dumps(historical_sources, sort_keys=True),
|
|
1910
|
+
"interpro_preserved_metadata_sha256": (
|
|
1911
|
+
metadata.get("interpro_entries_sha256", "")
|
|
1912
|
+
if interpro_entries and previous_interpro
|
|
1913
|
+
else metadata.get("interpro_preserved_metadata_sha256", "")
|
|
1914
|
+
),
|
|
1915
|
+
}
|
|
1916
|
+
if uniprot_source is not None:
|
|
1917
|
+
preprocessing_metadata["uniprot_features_source"] = json.dumps(
|
|
1918
|
+
uniprot_source, sort_keys=True
|
|
1919
|
+
)
|
|
1920
|
+
preprocessing_metadata["uniprot_counts"] = json.dumps(uniprot_counts, sort_keys=True)
|
|
1921
|
+
# An import without recorded source provenance invalidates any
|
|
1922
|
+
# prior reuse key. Empty values never permit reuse.
|
|
1923
|
+
preprocessing_metadata["uniprot_lookup_fingerprint"] = uniprot_fingerprint or ""
|
|
1924
|
+
if panther_source is not None:
|
|
1925
|
+
preprocessing_metadata["panther_classifications_source"] = json.dumps(
|
|
1926
|
+
panther_source, sort_keys=True
|
|
1927
|
+
)
|
|
1928
|
+
for key, value in preprocessing_metadata.items():
|
|
1929
|
+
db.execute(
|
|
1930
|
+
"INSERT INTO build_metadata VALUES (?, ?) ON CONFLICT(key) DO UPDATE SET value=excluded.value",
|
|
1931
|
+
(key, value),
|
|
1932
|
+
)
|
|
1933
|
+
if counts["ready"] == 0:
|
|
1934
|
+
raise ValueError(
|
|
1935
|
+
"Preprocessing produced no usable coding transcripts; check source tables and FASTA"
|
|
1936
|
+
)
|
|
1937
|
+
with sqlite_activity("Commit preprocessed reference"):
|
|
1938
|
+
db.commit()
|
|
1939
|
+
except BaseException:
|
|
1940
|
+
with sqlite_activity("Roll back preprocessing transaction"):
|
|
1941
|
+
db.rollback()
|
|
1942
|
+
raise
|
|
1943
|
+
finally:
|
|
1944
|
+
try:
|
|
1945
|
+
if transcript_progress is not None:
|
|
1946
|
+
transcript_progress.close()
|
|
1947
|
+
finally:
|
|
1948
|
+
db.execute(f"PRAGMA main.secure_delete={previous_secure_delete}")
|
|
1949
|
+
LOG.info("Preprocessing complete: %s", counts)
|
|
1950
|
+
if counts["error"]:
|
|
1951
|
+
LOG.warning(
|
|
1952
|
+
"%s transcripts have stored errors; inspect ff_transcripts WHERE status='error'",
|
|
1953
|
+
counts["error"],
|
|
1954
|
+
)
|
|
1955
|
+
for reason, summary in errors.items():
|
|
1956
|
+
LOG.warning(
|
|
1957
|
+
" %s: %s transcript(s); examples: %s",
|
|
1958
|
+
reason,
|
|
1959
|
+
f"{summary['count']:,}",
|
|
1960
|
+
", ".join(summary["example_transcripts"]),
|
|
1961
|
+
)
|
|
1962
|
+
return counts
|
|
1963
|
+
|
|
1964
|
+
|
|
1965
|
+
# Read-only runtime access to a completed reference.
|
|
1966
|
+
|
|
1967
|
+
|
|
1968
|
+
class ReferenceReader:
|
|
1969
|
+
"""Read-only local reference access; reuse one reader across fusion calls.
|
|
1970
|
+
|
|
1971
|
+
get_transcript() returns prepared coordinates, splice sites and features,
|
|
1972
|
+
reconstructing pre-mRNA from the genome when requested.
|
|
1973
|
+
Genome chunk caching is LRU and bounded (default at most 64 MiB); no
|
|
1974
|
+
transcript sequences are cached.
|
|
1975
|
+
The reader is intended for one thread; use a separate reader per worker.
|
|
1976
|
+
"""
|
|
1977
|
+
|
|
1978
|
+
def __init__(self, database: str | Path, cached_chunks: int = 64) -> None:
|
|
1979
|
+
"""Open an existing preprocessed database and configure the chunk cache."""
|
|
1980
|
+
if cached_chunks < 0:
|
|
1981
|
+
raise ValueError("cached_chunks must be nonnegative")
|
|
1982
|
+
path = Path(database).expanduser().resolve()
|
|
1983
|
+
self.db = sqlite3.connect(path.as_uri() + "?mode=ro", uri=True)
|
|
1984
|
+
self.cached_chunks = cached_chunks
|
|
1985
|
+
self.chunk_cache: ChunkCache = OrderedDict()
|
|
1986
|
+
try:
|
|
1987
|
+
metadata = dict(self.db.execute("SELECT key, value FROM build_metadata"))
|
|
1988
|
+
if metadata.get("preprocessing_version") != "1":
|
|
1989
|
+
raise ValueError("Database is not preprocessed; run --preprocess-only DB")
|
|
1990
|
+
if metadata.get("sequence_chunk_size") != str(CHUNK_SIZE):
|
|
1991
|
+
raise ValueError("Unsupported sequence chunk size")
|
|
1992
|
+
if metadata.get("transcript_payload_codec") not in {None, TRANSCRIPT_PAYLOAD_CODEC}:
|
|
1993
|
+
raise ValueError("Unsupported transcript payload codec")
|
|
1994
|
+
self.metadata = metadata
|
|
1995
|
+
except BaseException:
|
|
1996
|
+
self.db.close()
|
|
1997
|
+
raise
|
|
1998
|
+
|
|
1999
|
+
def transcript_error_summary(self) -> dict[str, TranscriptErrorSummary]:
|
|
2000
|
+
"""Read grouped failures with example IDs without rebuilding the reference.
|
|
2001
|
+
|
|
2002
|
+
Prepared summaries are cheap metadata reads. If one is absent, inspect
|
|
2003
|
+
only stored error payloads; usable transcript payloads are never loaded.
|
|
2004
|
+
"""
|
|
2005
|
+
if "preprocessing_errors" in self.metadata:
|
|
2006
|
+
return cast(
|
|
2007
|
+
dict[str, TranscriptErrorSummary], json.loads(self.metadata["preprocessing_errors"])
|
|
2008
|
+
)
|
|
2009
|
+
errors: dict[str, TranscriptErrorSummary] = {}
|
|
2010
|
+
for transcript_id, encoded in self.db.execute(
|
|
2011
|
+
"SELECT transcript_id, payload FROM ff_transcripts WHERE status='error' ORDER BY transcript_id"
|
|
2012
|
+
):
|
|
2013
|
+
payload = cast("EnsemblError", decode_transcript_payload(encoded))
|
|
2014
|
+
message = payload["error"].removeprefix(transcript_id + ": ")
|
|
2015
|
+
_record_transcript_error(errors, transcript_id, message)
|
|
2016
|
+
return dict(sorted(errors.items(), key=lambda item: (-item[1]["count"], item[0])))
|
|
2017
|
+
|
|
2018
|
+
def get_transcript(
|
|
2019
|
+
self, transcript_id: str, *, include_sequence: bool = True
|
|
2020
|
+
) -> ProteinFeatureResponse:
|
|
2021
|
+
"""Unknown IDs, version mismatches and unsupported models return errors."""
|
|
2022
|
+
match = re.fullmatch(r"(ENST\d+)(?:\.(\d+))?", transcript_id)
|
|
2023
|
+
if not match:
|
|
2024
|
+
return {"error": f"Invalid human Ensembl transcript ID: {transcript_id}"}
|
|
2025
|
+
stable_id, requested_version = match.groups()
|
|
2026
|
+
row = self.db.execute(
|
|
2027
|
+
"SELECT version, status, payload FROM ff_transcripts WHERE transcript_id=?",
|
|
2028
|
+
(stable_id,),
|
|
2029
|
+
).fetchone()
|
|
2030
|
+
if row is None:
|
|
2031
|
+
return {"error": f"Transcript {transcript_id} is absent from this reference release"}
|
|
2032
|
+
version, status, encoded = row
|
|
2033
|
+
if requested_version is not None and int(requested_version) != version:
|
|
2034
|
+
return {
|
|
2035
|
+
"error": f"Transcript version mismatch: requested {transcript_id}; stored {stable_id}.{version}"
|
|
2036
|
+
}
|
|
2037
|
+
if status == "noncoding":
|
|
2038
|
+
return {"error": "Transcript is non-coding or has no translation object"}
|
|
2039
|
+
payload = decode_transcript_payload(encoded)
|
|
2040
|
+
if status == "error":
|
|
2041
|
+
return payload
|
|
2042
|
+
payload = cast("TranscriptProteinFeatureResult", payload)
|
|
2043
|
+
if include_sequence:
|
|
2044
|
+
payload["premrna_sequence"] = self.sequence(
|
|
2045
|
+
"dna",
|
|
2046
|
+
cast(str, payload["chromosome"]),
|
|
2047
|
+
payload["transcript_genomic_start"],
|
|
2048
|
+
payload["transcript_genomic_end"],
|
|
2049
|
+
payload["strand"],
|
|
2050
|
+
)
|
|
2051
|
+
return payload
|
|
2052
|
+
|
|
2053
|
+
def sequence(
|
|
2054
|
+
self, kind: str, sequence_id: str, start: int = 1, end: int | None = None, strand: int = 1
|
|
2055
|
+
) -> str:
|
|
2056
|
+
"""Read a sequence interval through this reader's shared bounded cache."""
|
|
2057
|
+
return fetch_sequence(
|
|
2058
|
+
self.db,
|
|
2059
|
+
kind,
|
|
2060
|
+
sequence_id,
|
|
2061
|
+
start,
|
|
2062
|
+
end,
|
|
2063
|
+
strand,
|
|
2064
|
+
_chunk_cache=self.chunk_cache,
|
|
2065
|
+
_cache_limit=self.cached_chunks,
|
|
2066
|
+
)
|
|
2067
|
+
|
|
2068
|
+
def get_interpro_annotation(self, accession: str) -> ProteinFeatureAnnotation | None:
|
|
2069
|
+
"""Return stored metadata for an accession, or None when it is absent."""
|
|
2070
|
+
row = self.db.execute(
|
|
2071
|
+
"SELECT name, entry_type FROM ff_interpro WHERE interpro_id=?", (accession,)
|
|
2072
|
+
).fetchone()
|
|
2073
|
+
if row is None:
|
|
2074
|
+
return None
|
|
2075
|
+
return {"name": row[0], "entry_type": row[1], "interpro_id": accession}
|
|
2076
|
+
|
|
2077
|
+
def close(self) -> None:
|
|
2078
|
+
"""Release both the database handle and decompressed genome chunks."""
|
|
2079
|
+
self.chunk_cache.clear()
|
|
2080
|
+
self.db.close()
|
|
2081
|
+
|
|
2082
|
+
def __enter__(self) -> Self:
|
|
2083
|
+
"""Use this reader as a context manager."""
|
|
2084
|
+
return self
|
|
2085
|
+
|
|
2086
|
+
def __exit__(
|
|
2087
|
+
self,
|
|
2088
|
+
exc_type: type[BaseException] | None,
|
|
2089
|
+
exc_value: BaseException | None,
|
|
2090
|
+
traceback: TracebackType | None,
|
|
2091
|
+
) -> None:
|
|
2092
|
+
"""Close the reader even when annotation raises an exception."""
|
|
2093
|
+
self.close()
|
|
2094
|
+
|
|
2095
|
+
|
|
2096
|
+
# Full-build orchestration and the prepare-data command.
|
|
2097
|
+
|
|
2098
|
+
|
|
2099
|
+
@contextmanager
|
|
2100
|
+
def build_lock(path: Path) -> Iterator[None]:
|
|
2101
|
+
"""Keep concurrent builders from modifying the same resumable database.
|
|
2102
|
+
|
|
2103
|
+
Advisory locks are released by the OS even after a killed process. Leave the
|
|
2104
|
+
small lock file in place so another process cannot lock a different inode.
|
|
2105
|
+
"""
|
|
2106
|
+
with path.open("a+b") as stream:
|
|
2107
|
+
try:
|
|
2108
|
+
if os.name == "nt":
|
|
2109
|
+
import msvcrt
|
|
2110
|
+
|
|
2111
|
+
stream.write(b"\0")
|
|
2112
|
+
stream.flush()
|
|
2113
|
+
stream.seek(0)
|
|
2114
|
+
msvcrt.locking(stream.fileno(), msvcrt.LK_NBLCK, 1) # type: ignore[attr-defined] # Windows-only API
|
|
2115
|
+
else:
|
|
2116
|
+
import fcntl
|
|
2117
|
+
|
|
2118
|
+
fcntl.flock(stream.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB)
|
|
2119
|
+
except OSError as exc:
|
|
2120
|
+
raise ValueError(f"Another build is using this output: {path}") from exc
|
|
2121
|
+
try:
|
|
2122
|
+
yield
|
|
2123
|
+
finally:
|
|
2124
|
+
if os.name == "nt":
|
|
2125
|
+
stream.seek(0)
|
|
2126
|
+
msvcrt.locking(stream.fileno(), msvcrt.LK_UNLCK, 1) # type: ignore[attr-defined] # Windows-only API
|
|
2127
|
+
else:
|
|
2128
|
+
fcntl.flock(stream.fileno(), fcntl.LOCK_UN)
|
|
2129
|
+
|
|
2130
|
+
|
|
2131
|
+
class BuildCheckpoint:
|
|
2132
|
+
"""Record completed stages only after their SQLite work has committed."""
|
|
2133
|
+
|
|
2134
|
+
def __init__(self, db: sqlite3.Connection, configuration: Mapping[str, object]) -> None:
|
|
2135
|
+
self.db = db
|
|
2136
|
+
db.execute(
|
|
2137
|
+
"CREATE TABLE IF NOT EXISTS build_checkpoints ("
|
|
2138
|
+
"stage TEXT PRIMARY KEY, source TEXT NOT NULL, fingerprint TEXT NOT NULL, "
|
|
2139
|
+
"count INTEGER NOT NULL, complete INTEGER NOT NULL)"
|
|
2140
|
+
)
|
|
2141
|
+
fingerprint = json.dumps(configuration, sort_keys=True)
|
|
2142
|
+
previous = self.read("configuration")
|
|
2143
|
+
if previous and previous["fingerprint"] != fingerprint:
|
|
2144
|
+
raise ValueError(
|
|
2145
|
+
"Build checkpoint belongs to different source/settings; "
|
|
2146
|
+
"use a different --output or remove its .building database"
|
|
2147
|
+
)
|
|
2148
|
+
if previous is None:
|
|
2149
|
+
self.save("configuration", fingerprint=fingerprint)
|
|
2150
|
+
|
|
2151
|
+
def read(self, stage: str) -> CheckpointRecord | None:
|
|
2152
|
+
"""Read one stage's source provenance and completion marker."""
|
|
2153
|
+
row = self.db.execute(
|
|
2154
|
+
"SELECT source, fingerprint, count, complete FROM build_checkpoints WHERE stage=?",
|
|
2155
|
+
(stage,),
|
|
2156
|
+
).fetchone()
|
|
2157
|
+
if row is None:
|
|
2158
|
+
return None
|
|
2159
|
+
return {
|
|
2160
|
+
"source": json.loads(row[0]),
|
|
2161
|
+
"fingerprint": row[1],
|
|
2162
|
+
"count": row[2],
|
|
2163
|
+
"complete": bool(row[3]),
|
|
2164
|
+
}
|
|
2165
|
+
|
|
2166
|
+
def save(
|
|
2167
|
+
self,
|
|
2168
|
+
stage: str,
|
|
2169
|
+
*,
|
|
2170
|
+
source: Mapping[str, Any] | None = None,
|
|
2171
|
+
fingerprint: str = "",
|
|
2172
|
+
count: int = 0,
|
|
2173
|
+
complete: bool = True,
|
|
2174
|
+
) -> None:
|
|
2175
|
+
"""Commit a marker with the work it describes; incomplete stages can resume."""
|
|
2176
|
+
self.db.execute(
|
|
2177
|
+
"INSERT OR REPLACE INTO build_checkpoints VALUES (?, ?, ?, ?, ?)",
|
|
2178
|
+
(stage, json.dumps(source or {}, sort_keys=True), fingerprint, count, int(complete)),
|
|
2179
|
+
)
|
|
2180
|
+
self.db.commit()
|
|
2181
|
+
|
|
2182
|
+
|
|
2183
|
+
def build(args: argparse.Namespace) -> Path:
|
|
2184
|
+
"""Resume committed imports in a staging database, then publish atomically.
|
|
2185
|
+
|
|
2186
|
+
Failures preserve the staging database and downloads. Repeating the same
|
|
2187
|
+
command skips completed stages; no incomplete reference replaces the output.
|
|
2188
|
+
"""
|
|
2189
|
+
from .uniprot import UNIPROT_HUMAN_URL
|
|
2190
|
+
|
|
2191
|
+
cache = args.cache_dir.expanduser().resolve()
|
|
2192
|
+
base = args.base_url.rstrip("/") + "/"
|
|
2193
|
+
# Resolution, metadata/schema, tables, FASTA, preprocessing, analysis, check, publication.
|
|
2194
|
+
steps = BuildSteps(1 + 3 + len(TABLES) + len(SEQUENCE_TYPES) + 4)
|
|
2195
|
+
with steps.step("Resolve Ensembl release and human GRCh38 core directory"):
|
|
2196
|
+
release, root, core = resolve_core(base, args.release)
|
|
2197
|
+
LOG.info("Resolved Ensembl release %s; core database %s", release, core)
|
|
2198
|
+
output = (
|
|
2199
|
+
(args.output or cache / args.species / ASSEMBLY / f"release-{release}" / "ensembl.sqlite")
|
|
2200
|
+
.expanduser()
|
|
2201
|
+
.resolve()
|
|
2202
|
+
)
|
|
2203
|
+
output.parent.mkdir(parents=True, exist_ok=True)
|
|
2204
|
+
if output.exists() and not args.force:
|
|
2205
|
+
raise FileExistsError(f"Output exists: {output}. Use --force to replace it.")
|
|
2206
|
+
downloads = cache / "downloads" / hashlib.sha256(base.encode()).hexdigest()[:12] / core
|
|
2207
|
+
interpro_cache = cache / "metadata" / "interpro"
|
|
2208
|
+
LOG.info("Final reference database: %s", output)
|
|
2209
|
+
for group in ("core", *SEQUENCE_TYPES):
|
|
2210
|
+
LOG.info("%s download cache: %s", group, downloads / group)
|
|
2211
|
+
LOG.info("InterPro download cache: %s", interpro_cache)
|
|
2212
|
+
LOG.info("Historical InterPro entry-list cache: %s", interpro_cache / "releases")
|
|
2213
|
+
LOG.info(
|
|
2214
|
+
"Partial downloads: <download-cache>/<filename>.<random>.part; checksums: <filename>.sha256"
|
|
2215
|
+
)
|
|
2216
|
+
temporary = output.with_name(output.name + ".building")
|
|
2217
|
+
lock_path = output.with_name(output.name + ".prepare.lock")
|
|
2218
|
+
LOG.info("Resumable build checkpoint: %s; SQLite journal: %s-journal", temporary, temporary)
|
|
2219
|
+
LOG.info("Build lock: %s", lock_path)
|
|
2220
|
+
source_rows = []
|
|
2221
|
+
fetch_interpro = args.interpro_entries is None
|
|
2222
|
+
|
|
2223
|
+
def get(url: str, group: str = "core", *, directory: Path | None = None) -> Path:
|
|
2224
|
+
"""Download/cache one source and retain its checksum/size for provenance."""
|
|
2225
|
+
path, digest = download(url, directory if directory is not None else downloads / group)
|
|
2226
|
+
source_rows.append((url, digest, path.stat().st_size))
|
|
2227
|
+
return path
|
|
2228
|
+
|
|
2229
|
+
core_url = root + f"mysql/{core}/"
|
|
2230
|
+
LOG.info("Temporary reference database: %s; rerun the same command to resume", temporary)
|
|
2231
|
+
try:
|
|
2232
|
+
with steps.step("Prepare InterPro metadata"):
|
|
2233
|
+
if fetch_interpro:
|
|
2234
|
+
interpro_entries = get(INTERPRO_ENTRIES_URL, directory=interpro_cache)
|
|
2235
|
+
else:
|
|
2236
|
+
interpro_entries = args.interpro_entries
|
|
2237
|
+
LOG.info("Using local InterPro entries: %s", interpro_entries)
|
|
2238
|
+
with steps.step("Prepare reviewed human UniProt features"):
|
|
2239
|
+
if args.uniprot_features is None:
|
|
2240
|
+
uniprot_features = get(UNIPROT_HUMAN_URL, directory=cache / "metadata" / "uniprot")
|
|
2241
|
+
uniprot_source = {"url": UNIPROT_HUMAN_URL, "sha256": source_rows[-1][1]}
|
|
2242
|
+
else:
|
|
2243
|
+
uniprot_features = args.uniprot_features
|
|
2244
|
+
uniprot_source = {
|
|
2245
|
+
"path": str(uniprot_features),
|
|
2246
|
+
"sha256": sha256_file(uniprot_features),
|
|
2247
|
+
}
|
|
2248
|
+
LOG.info("Using local UniProt features: %s", uniprot_features)
|
|
2249
|
+
with steps.step("Download and read Ensembl table definitions"):
|
|
2250
|
+
schema = parse_schema(get(core_url + core + ".sql.gz"))
|
|
2251
|
+
absent = set(TABLES) - set(schema)
|
|
2252
|
+
if absent:
|
|
2253
|
+
raise ValueError(f"Required tables absent from source schema: {sorted(absent)}")
|
|
2254
|
+
with build_lock(lock_path), closing(sqlite3.connect(temporary)) as db:
|
|
2255
|
+
if output.exists() and not args.force:
|
|
2256
|
+
raise FileExistsError(f"Output appeared during the build: {output}")
|
|
2257
|
+
# Apply build-only settings to the unpublished database. The 64 MiB
|
|
2258
|
+
# page cache reduces repeated reads without caching the entire input.
|
|
2259
|
+
db.execute("PRAGMA journal_mode=DELETE")
|
|
2260
|
+
db.execute("PRAGMA synchronous=NORMAL")
|
|
2261
|
+
db.execute("PRAGMA cache_size=-65536")
|
|
2262
|
+
db.execute("PRAGMA user_version=1")
|
|
2263
|
+
checkpoint = BuildCheckpoint(
|
|
2264
|
+
db,
|
|
2265
|
+
{
|
|
2266
|
+
"checkpoint_version": 1,
|
|
2267
|
+
"base_url": base,
|
|
2268
|
+
"release": release,
|
|
2269
|
+
"species": args.species,
|
|
2270
|
+
"core_database": core,
|
|
2271
|
+
"assembly": ASSEMBLY,
|
|
2272
|
+
"schema_sha256": source_rows[-1][1],
|
|
2273
|
+
"sequence_chunk_size": CHUNK_SIZE,
|
|
2274
|
+
"tables": TABLES,
|
|
2275
|
+
"sequence_types": SEQUENCE_TYPES,
|
|
2276
|
+
},
|
|
2277
|
+
)
|
|
2278
|
+
|
|
2279
|
+
def reuse(stage: str) -> int | None:
|
|
2280
|
+
"""Reuse a completed import and its recorded source provenance."""
|
|
2281
|
+
saved = checkpoint.read(stage)
|
|
2282
|
+
if saved and saved["complete"]:
|
|
2283
|
+
source = saved["source"]
|
|
2284
|
+
source_rows.append((source["url"], source["sha256"], source["bytes"]))
|
|
2285
|
+
LOG.info("Reusing checkpoint for %s: %s records", stage, f"{saved['count']:,}")
|
|
2286
|
+
return saved["count"]
|
|
2287
|
+
return None
|
|
2288
|
+
|
|
2289
|
+
def last_source() -> SourceRecord:
|
|
2290
|
+
"""Describe the most recently verified download."""
|
|
2291
|
+
url, digest, size = source_rows[-1]
|
|
2292
|
+
return {"url": url, "sha256": digest, "bytes": size}
|
|
2293
|
+
|
|
2294
|
+
counts = {}
|
|
2295
|
+
panther_digest = None
|
|
2296
|
+
for table in TABLES:
|
|
2297
|
+
with steps.step(f"Download, import and index {table}"):
|
|
2298
|
+
count = reuse("table:" + table)
|
|
2299
|
+
if count is None:
|
|
2300
|
+
# A failed import may have created its table but has no
|
|
2301
|
+
# completion marker. Recreate only that unfinished table.
|
|
2302
|
+
db.execute(f"DROP TABLE IF EXISTS {sql_name('ensembl_' + table)}")
|
|
2303
|
+
count = import_table(
|
|
2304
|
+
db, table, schema[table], get(core_url + table + ".txt.gz")
|
|
2305
|
+
)
|
|
2306
|
+
checkpoint.save("table:" + table, source=last_source(), count=count)
|
|
2307
|
+
counts[table] = count
|
|
2308
|
+
if table == "analysis" and args.panther_classifications is None:
|
|
2309
|
+
version = panther_release(protein_analyses(db))
|
|
2310
|
+
if version:
|
|
2311
|
+
LOG.info(
|
|
2312
|
+
"Check/download PANTHER %s metadata before importing genome sequences",
|
|
2313
|
+
version,
|
|
2314
|
+
)
|
|
2315
|
+
_, panther_digest, _ = download_panther_classifications(
|
|
2316
|
+
version, cache / "metadata" / "panther"
|
|
2317
|
+
)
|
|
2318
|
+
# Guard against accidentally mixing source database releases.
|
|
2319
|
+
versions = {
|
|
2320
|
+
str(row[0])
|
|
2321
|
+
for row in db.execute(
|
|
2322
|
+
"SELECT meta_value FROM ensembl_meta WHERE meta_key='schema_version'"
|
|
2323
|
+
)
|
|
2324
|
+
}
|
|
2325
|
+
if versions and versions != {str(release)}:
|
|
2326
|
+
raise ValueError(f"Core schema_version {versions} does not match release {release}")
|
|
2327
|
+
if not checkpoint.read("sequence_tables"):
|
|
2328
|
+
# SQLite executes CREATE TABLE outside implicit DML transactions;
|
|
2329
|
+
# recover a process killed partway through schema creation.
|
|
2330
|
+
db.execute("DROP TABLE IF EXISTS sequence_chunks")
|
|
2331
|
+
db.execute("DROP TABLE IF EXISTS sequences")
|
|
2332
|
+
create_sequence_tables(db)
|
|
2333
|
+
checkpoint.save("sequence_tables")
|
|
2334
|
+
for kind in SEQUENCE_TYPES:
|
|
2335
|
+
with steps.step(f"Download and import {kind} FASTA sequences"):
|
|
2336
|
+
count = reuse("fasta:" + kind)
|
|
2337
|
+
if count is not None:
|
|
2338
|
+
counts[kind + "_sequences"] = count
|
|
2339
|
+
continue
|
|
2340
|
+
directory = root + f"fasta/{args.species}/{kind}/"
|
|
2341
|
+
suffix = ".dna.toplevel.fa.gz" if kind == "dna" else f".{kind}.all.fa.gz"
|
|
2342
|
+
candidates = sorted(
|
|
2343
|
+
name for name in listing(directory) if name.endswith(suffix)
|
|
2344
|
+
)
|
|
2345
|
+
if len(candidates) != 1:
|
|
2346
|
+
raise ValueError(
|
|
2347
|
+
f"Expected one {suffix} file at {directory}; found {candidates}"
|
|
2348
|
+
)
|
|
2349
|
+
path = get(directory + candidates[0], kind)
|
|
2350
|
+
source = last_source()
|
|
2351
|
+
saved = checkpoint.read("fasta:" + kind)
|
|
2352
|
+
if saved is None or saved["source"] != source:
|
|
2353
|
+
# Never reuse records after the source bytes change.
|
|
2354
|
+
db.execute("DELETE FROM sequence_chunks WHERE kind=?", (kind,))
|
|
2355
|
+
db.execute("DELETE FROM sequences WHERE kind=?", (kind,))
|
|
2356
|
+
checkpoint.save("fasta:" + kind, source=source, complete=False)
|
|
2357
|
+
count = import_fasta(db, kind, path, resume=True)
|
|
2358
|
+
counts[kind + "_sequences"] = count
|
|
2359
|
+
if kind == "dna":
|
|
2360
|
+
validate_dna_regions(db)
|
|
2361
|
+
checkpoint.save("fasta:" + kind, source=source, count=count)
|
|
2362
|
+
db.execute(
|
|
2363
|
+
"CREATE TABLE IF NOT EXISTS build_metadata (key TEXT PRIMARY KEY, value TEXT NOT NULL)"
|
|
2364
|
+
)
|
|
2365
|
+
metadata = {
|
|
2366
|
+
"format_version": "1",
|
|
2367
|
+
"release": str(release),
|
|
2368
|
+
"species": args.species,
|
|
2369
|
+
"core_database": core,
|
|
2370
|
+
"assembly": ASSEMBLY,
|
|
2371
|
+
"created_utc": datetime.now(timezone.utc).isoformat(),
|
|
2372
|
+
"sequence_chunk_size": str(CHUNK_SIZE),
|
|
2373
|
+
"sequence_codec": "zlib",
|
|
2374
|
+
"tables": json.dumps(TABLES),
|
|
2375
|
+
"sequence_types": json.dumps(SEQUENCE_TYPES),
|
|
2376
|
+
"counts": json.dumps(counts),
|
|
2377
|
+
}
|
|
2378
|
+
db.executemany("INSERT OR REPLACE INTO build_metadata VALUES (?, ?)", metadata.items())
|
|
2379
|
+
db.execute(
|
|
2380
|
+
"CREATE TABLE IF NOT EXISTS source_files (url TEXT PRIMARY KEY, sha256 TEXT, bytes INTEGER)"
|
|
2381
|
+
)
|
|
2382
|
+
db.executemany("INSERT OR REPLACE INTO source_files VALUES (?, ?, ?)", source_rows)
|
|
2383
|
+
db.commit()
|
|
2384
|
+
with steps.step("Preprocess transcripts, domains and splice sites"):
|
|
2385
|
+
# If preprocessing completed before a later failure, preserve it
|
|
2386
|
+
# unless its metadata input or implementation has changed.
|
|
2387
|
+
fingerprint = json.dumps(
|
|
2388
|
+
{
|
|
2389
|
+
"implementation": sha256_file(Path(__file__)),
|
|
2390
|
+
"uniprot_implementation": sha256_file(
|
|
2391
|
+
Path(__file__).with_name("uniprot.py")
|
|
2392
|
+
),
|
|
2393
|
+
"ensembl_implementation": sha256_file(
|
|
2394
|
+
Path(__file__).with_name("ensembl.py")
|
|
2395
|
+
),
|
|
2396
|
+
"uniprot_sha256": sha256_file(uniprot_features),
|
|
2397
|
+
"interpro_sha256": sha256_file(interpro_entries),
|
|
2398
|
+
"fetch_interpro": fetch_interpro,
|
|
2399
|
+
"panther_sha256": (
|
|
2400
|
+
sha256_file(args.panther_classifications)
|
|
2401
|
+
if args.panther_classifications
|
|
2402
|
+
else panther_digest
|
|
2403
|
+
),
|
|
2404
|
+
},
|
|
2405
|
+
sort_keys=True,
|
|
2406
|
+
)
|
|
2407
|
+
saved = checkpoint.read("preprocessing")
|
|
2408
|
+
if saved and saved["complete"] and saved["fingerprint"] == fingerprint:
|
|
2409
|
+
LOG.info("Reusing completed transcript/domain preprocessing")
|
|
2410
|
+
else:
|
|
2411
|
+
db.execute("DELETE FROM build_checkpoints WHERE stage='analyze'")
|
|
2412
|
+
checkpoint.save("preprocessing", fingerprint=fingerprint, complete=False)
|
|
2413
|
+
preprocess_reference(
|
|
2414
|
+
db,
|
|
2415
|
+
interpro_entries,
|
|
2416
|
+
interpro_archive_cache=interpro_cache if fetch_interpro else None,
|
|
2417
|
+
panther_classifications=args.panther_classifications,
|
|
2418
|
+
panther_cache=cache / "metadata" / "panther",
|
|
2419
|
+
uniprot_features=uniprot_features,
|
|
2420
|
+
uniprot_source=uniprot_source,
|
|
2421
|
+
)
|
|
2422
|
+
checkpoint.save("preprocessing", fingerprint=fingerprint)
|
|
2423
|
+
with steps.step("Analyze database indexes"):
|
|
2424
|
+
if checkpoint.read("analyze"):
|
|
2425
|
+
LOG.info("Reusing completed index analysis")
|
|
2426
|
+
else:
|
|
2427
|
+
sqlite_phase(db, "ANALYZE", "Analyze database")
|
|
2428
|
+
checkpoint.save("analyze")
|
|
2429
|
+
with steps.step("Validate SQLite database integrity"):
|
|
2430
|
+
result = sqlite_phase(db, "PRAGMA integrity_check", "Check database")
|
|
2431
|
+
if result != [("ok",)]:
|
|
2432
|
+
raise ValueError(f"SQLite integrity check failed: {result[:5]}")
|
|
2433
|
+
db.commit()
|
|
2434
|
+
# Close SQLite before publication, while still holding the build
|
|
2435
|
+
# lock. This also permits atomic replacement on Windows.
|
|
2436
|
+
db.close()
|
|
2437
|
+
with steps.step("Publish completed reference database"):
|
|
2438
|
+
if output.exists() and not args.force:
|
|
2439
|
+
raise FileExistsError(f"Output appeared during the build: {output}")
|
|
2440
|
+
os.replace(temporary, output)
|
|
2441
|
+
except BaseException:
|
|
2442
|
+
if temporary.exists():
|
|
2443
|
+
LOG.error(
|
|
2444
|
+
"Build checkpoint preserved at %s; rerun the same command to resume", temporary
|
|
2445
|
+
)
|
|
2446
|
+
raise
|
|
2447
|
+
LOG.info("Created %s (%.1f MiB)", output, output.stat().st_size / CHUNK_SIZE)
|
|
2448
|
+
steps.complete()
|
|
2449
|
+
return output
|
|
2450
|
+
|
|
2451
|
+
|
|
2452
|
+
def main(argv: Sequence[str] | None = None) -> int:
|
|
2453
|
+
"""Validate CLI options, then run a full build or derived-table regeneration."""
|
|
2454
|
+
parser = argparse.ArgumentParser(
|
|
2455
|
+
prog="fusion-function prepare-data",
|
|
2456
|
+
description=__doc__,
|
|
2457
|
+
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
2458
|
+
)
|
|
2459
|
+
parser.add_argument(
|
|
2460
|
+
"--release", type=int, help="Ensembl release; default: latest numbered FTP release"
|
|
2461
|
+
)
|
|
2462
|
+
parser.add_argument(
|
|
2463
|
+
"--species",
|
|
2464
|
+
choices=[SPECIES],
|
|
2465
|
+
help="Species component of the output path; this script supports human data",
|
|
2466
|
+
)
|
|
2467
|
+
parser.add_argument(
|
|
2468
|
+
"--output", type=Path, help="SQLite file; default: release-specific file in cache"
|
|
2469
|
+
)
|
|
2470
|
+
parser.add_argument(
|
|
2471
|
+
"--cache-dir",
|
|
2472
|
+
type=Path,
|
|
2473
|
+
default=default_cache_dir(),
|
|
2474
|
+
help="Download/cache directory; defaults to FUSION_FUNCTION_CACHEDIR or OS user cache",
|
|
2475
|
+
)
|
|
2476
|
+
parser.add_argument("--base-url", help="Advanced: FTP HTTP(S) mirror root, ending at /pub/")
|
|
2477
|
+
parser.add_argument(
|
|
2478
|
+
"--from-source",
|
|
2479
|
+
action="store_true",
|
|
2480
|
+
default=None,
|
|
2481
|
+
help="Build from Ensembl FTP instead of downloading a compatible prebuilt reference",
|
|
2482
|
+
)
|
|
2483
|
+
parser.add_argument(
|
|
2484
|
+
"--reference-catalog",
|
|
2485
|
+
help="Prebuilt catalog: local JSON or HTTPS URL; default: maintained catalog",
|
|
2486
|
+
)
|
|
2487
|
+
parser.add_argument(
|
|
2488
|
+
"--force",
|
|
2489
|
+
action="store_true",
|
|
2490
|
+
default=None,
|
|
2491
|
+
help="Replace output after successful installation or build",
|
|
2492
|
+
)
|
|
2493
|
+
parser.add_argument(
|
|
2494
|
+
"--preprocess-only",
|
|
2495
|
+
type=Path,
|
|
2496
|
+
metavar="DB",
|
|
2497
|
+
help="Regenerate derived tables in a full source-built database",
|
|
2498
|
+
)
|
|
2499
|
+
parser.add_argument(
|
|
2500
|
+
"--interpro-entries",
|
|
2501
|
+
type=Path,
|
|
2502
|
+
metavar="TSV",
|
|
2503
|
+
help="Optional local InterPro entry.list TSV for canonical entry names/types",
|
|
2504
|
+
)
|
|
2505
|
+
parser.add_argument(
|
|
2506
|
+
"--panther-classifications",
|
|
2507
|
+
type=Path,
|
|
2508
|
+
metavar="TSV",
|
|
2509
|
+
help="Optional local PANTHER human classification TSV; default: Ensembl's PANTHER release",
|
|
2510
|
+
)
|
|
2511
|
+
parser.add_argument(
|
|
2512
|
+
"--uniprot-features",
|
|
2513
|
+
type=Path,
|
|
2514
|
+
metavar="XML",
|
|
2515
|
+
help="Local UniProt XML (optionally gzip); default: reviewed human bulk download",
|
|
2516
|
+
)
|
|
2517
|
+
args = parser.parse_args(argv)
|
|
2518
|
+
if args.release is not None and args.release <= 0:
|
|
2519
|
+
parser.error("--release must be positive")
|
|
2520
|
+
if args.preprocess_only:
|
|
2521
|
+
# Reprocessing uses the database's source release/assembly, so new-build
|
|
2522
|
+
# options cannot change where its existing coordinates came from.
|
|
2523
|
+
incompatible = [
|
|
2524
|
+
"--" + name.replace("_", "-")
|
|
2525
|
+
for name in (
|
|
2526
|
+
"release",
|
|
2527
|
+
"species",
|
|
2528
|
+
"output",
|
|
2529
|
+
"base_url",
|
|
2530
|
+
"force",
|
|
2531
|
+
"from_source",
|
|
2532
|
+
"reference_catalog",
|
|
2533
|
+
)
|
|
2534
|
+
if getattr(args, name) is not None
|
|
2535
|
+
]
|
|
2536
|
+
if incompatible:
|
|
2537
|
+
parser.error("--preprocess-only cannot be combined with " + ", ".join(incompatible))
|
|
2538
|
+
source_options = args.from_source or any(
|
|
2539
|
+
getattr(args, name) is not None
|
|
2540
|
+
for name in ("base_url", "interpro_entries", "panther_classifications", "uniprot_features")
|
|
2541
|
+
)
|
|
2542
|
+
if args.reference_catalog and source_options:
|
|
2543
|
+
parser.error("--reference-catalog cannot be combined with source-build options")
|
|
2544
|
+
if args.interpro_entries:
|
|
2545
|
+
args.interpro_entries = args.interpro_entries.expanduser().resolve()
|
|
2546
|
+
if not args.interpro_entries.is_file():
|
|
2547
|
+
parser.error(f"InterPro entry file does not exist: {args.interpro_entries}")
|
|
2548
|
+
if args.panther_classifications:
|
|
2549
|
+
args.panther_classifications = args.panther_classifications.expanduser().resolve()
|
|
2550
|
+
if not args.panther_classifications.is_file():
|
|
2551
|
+
parser.error(
|
|
2552
|
+
f"PANTHER classification file does not exist: {args.panther_classifications}"
|
|
2553
|
+
)
|
|
2554
|
+
if args.uniprot_features:
|
|
2555
|
+
args.uniprot_features = args.uniprot_features.expanduser().resolve()
|
|
2556
|
+
if not args.uniprot_features.is_file():
|
|
2557
|
+
parser.error(f"UniProt feature file does not exist: {args.uniprot_features}")
|
|
2558
|
+
args.species = args.species or SPECIES
|
|
2559
|
+
args.base_url = args.base_url or BASE_URL
|
|
2560
|
+
args.force = bool(args.force)
|
|
2561
|
+
parsed_base = urllib.parse.urlparse(args.base_url)
|
|
2562
|
+
if (
|
|
2563
|
+
parsed_base.scheme not in {"http", "https"}
|
|
2564
|
+
or not parsed_base.netloc
|
|
2565
|
+
or parsed_base.query
|
|
2566
|
+
or parsed_base.fragment
|
|
2567
|
+
):
|
|
2568
|
+
parser.error("--base-url must be an HTTP(S) mirror URL without a query or fragment")
|
|
2569
|
+
args.cache_dir = args.cache_dir.expanduser().resolve()
|
|
2570
|
+
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
|
|
2571
|
+
LOG.info("Cache root: %s", args.cache_dir)
|
|
2572
|
+
LOG.info("Ensembl download root: %s", args.cache_dir / "downloads")
|
|
2573
|
+
LOG.info("InterPro cache: %s", args.cache_dir / "metadata" / "interpro")
|
|
2574
|
+
LOG.info(
|
|
2575
|
+
"Historical InterPro entry-list cache: %s",
|
|
2576
|
+
args.cache_dir / "metadata" / "interpro" / "releases",
|
|
2577
|
+
)
|
|
2578
|
+
LOG.info("UniProt feature cache: %s", args.cache_dir / "metadata" / "uniprot")
|
|
2579
|
+
LOG.info("PANTHER human classification cache: %s", args.cache_dir / "metadata" / "panther")
|
|
2580
|
+
LOG.info("Prebuilt download cache: %s", args.cache_dir / "prebuilt")
|
|
2581
|
+
LOG.info("Source partial downloads: <filename>.<random>.part beside cached files")
|
|
2582
|
+
LOG.info("Prebuilt partial downloads: <cache>/prebuilt/<sha256>/ensembl.sqlite.gz.part")
|
|
2583
|
+
if args.interpro_entries:
|
|
2584
|
+
LOG.info("Local InterPro entries: %s", args.interpro_entries)
|
|
2585
|
+
if args.panther_classifications:
|
|
2586
|
+
LOG.info("Local PANTHER classifications: %s", args.panther_classifications)
|
|
2587
|
+
with logging_redirect_tqdm():
|
|
2588
|
+
try:
|
|
2589
|
+
if args.preprocess_only:
|
|
2590
|
+
output = args.preprocess_only.expanduser().resolve()
|
|
2591
|
+
LOG.info("Reprocessing existing reference in place: %s", output)
|
|
2592
|
+
LOG.info(
|
|
2593
|
+
"SQLite transaction files, if used: %s-journal; %s-wal; %s-shm",
|
|
2594
|
+
output,
|
|
2595
|
+
output,
|
|
2596
|
+
output,
|
|
2597
|
+
)
|
|
2598
|
+
steps = BuildSteps(4)
|
|
2599
|
+
with (
|
|
2600
|
+
build_lock(output.with_name(output.name + ".prepare.lock")),
|
|
2601
|
+
closing(sqlite3.connect(output.as_uri() + "?mode=rw", uri=True)) as db,
|
|
2602
|
+
):
|
|
2603
|
+
with steps.step("Validate existing reference database"):
|
|
2604
|
+
validate_reference_source(db)
|
|
2605
|
+
with steps.step("Prepare InterPro metadata"):
|
|
2606
|
+
if args.interpro_entries is None:
|
|
2607
|
+
archive_cache = args.cache_dir / "metadata" / "interpro"
|
|
2608
|
+
interpro_entries, _ = download(INTERPRO_ENTRIES_URL, archive_cache)
|
|
2609
|
+
else:
|
|
2610
|
+
interpro_entries = args.interpro_entries
|
|
2611
|
+
archive_cache = None
|
|
2612
|
+
LOG.info("Using local InterPro entries: %s", interpro_entries)
|
|
2613
|
+
with steps.step("Prepare reviewed human UniProt features"):
|
|
2614
|
+
from .uniprot import UNIPROT_HUMAN_URL
|
|
2615
|
+
|
|
2616
|
+
if args.uniprot_features is None:
|
|
2617
|
+
uniprot_features, digest = download(
|
|
2618
|
+
UNIPROT_HUMAN_URL, args.cache_dir / "metadata" / "uniprot"
|
|
2619
|
+
)
|
|
2620
|
+
uniprot_source = {"url": UNIPROT_HUMAN_URL, "sha256": digest}
|
|
2621
|
+
else:
|
|
2622
|
+
uniprot_features = args.uniprot_features
|
|
2623
|
+
uniprot_source = {
|
|
2624
|
+
"path": str(uniprot_features),
|
|
2625
|
+
"sha256": sha256_file(uniprot_features),
|
|
2626
|
+
}
|
|
2627
|
+
LOG.info("Using local UniProt features: %s", uniprot_features)
|
|
2628
|
+
with steps.step("Preprocess transcripts, domains and splice sites"):
|
|
2629
|
+
preprocess_reference(
|
|
2630
|
+
db,
|
|
2631
|
+
interpro_entries,
|
|
2632
|
+
interpro_archive_cache=archive_cache,
|
|
2633
|
+
panther_classifications=args.panther_classifications,
|
|
2634
|
+
panther_cache=args.cache_dir / "metadata" / "panther",
|
|
2635
|
+
uniprot_features=uniprot_features,
|
|
2636
|
+
uniprot_source=uniprot_source,
|
|
2637
|
+
)
|
|
2638
|
+
steps.complete()
|
|
2639
|
+
else:
|
|
2640
|
+
LOG.info(
|
|
2641
|
+
"Reference destination: %s",
|
|
2642
|
+
args.output.expanduser().resolve()
|
|
2643
|
+
if args.output
|
|
2644
|
+
else args.cache_dir
|
|
2645
|
+
/ args.species
|
|
2646
|
+
/ ASSEMBLY
|
|
2647
|
+
/ f"release-{args.release or '<latest>'}"
|
|
2648
|
+
/ "ensembl.sqlite",
|
|
2649
|
+
)
|
|
2650
|
+
output = None
|
|
2651
|
+
if not source_options:
|
|
2652
|
+
from .prebuilt import install_reference
|
|
2653
|
+
|
|
2654
|
+
output = install_reference(
|
|
2655
|
+
release=args.release,
|
|
2656
|
+
cache_dir=args.cache_dir,
|
|
2657
|
+
output=args.output,
|
|
2658
|
+
force=args.force,
|
|
2659
|
+
catalog=args.reference_catalog,
|
|
2660
|
+
)
|
|
2661
|
+
else:
|
|
2662
|
+
LOG.info("Source-build options selected; skipping the prebuilt catalog")
|
|
2663
|
+
if output is None:
|
|
2664
|
+
output = build(args)
|
|
2665
|
+
except (
|
|
2666
|
+
OSError,
|
|
2667
|
+
ValueError,
|
|
2668
|
+
sqlite3.Error,
|
|
2669
|
+
urllib.error.URLError,
|
|
2670
|
+
EOFError,
|
|
2671
|
+
ParseError,
|
|
2672
|
+
) as exc:
|
|
2673
|
+
LOG.error("Build failed: %s", exc)
|
|
2674
|
+
return 1
|
|
2675
|
+
except KeyboardInterrupt:
|
|
2676
|
+
LOG.error("Interrupted; no incomplete database was published")
|
|
2677
|
+
return 130
|
|
2678
|
+
print(output)
|
|
2679
|
+
return 0
|
|
2680
|
+
|
|
2681
|
+
|
|
2682
|
+
if __name__ == "__main__":
|
|
2683
|
+
raise SystemExit(main())
|