seqevi 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- seqevi/__init__.py +7 -0
- seqevi/__main__.py +6 -0
- seqevi/adapters/__init__.py +41 -0
- seqevi/adapters/base.py +131 -0
- seqevi/adapters/dbcan_cazyme.py +588 -0
- seqevi/adapters/eggnog.py +674 -0
- seqevi/adapters/interpro_pfam.py +757 -0
- seqevi/adapters/registry.py +68 -0
- seqevi/annotate.py +413 -0
- seqevi/api.py +390 -0
- seqevi/cli.py +610 -0
- seqevi/distribution/__init__.py +13 -0
- seqevi/distribution/manifest.py +199 -0
- seqevi/distribution/oci.py +490 -0
- seqevi/distribution/setup.py +752 -0
- seqevi/errors.py +73 -0
- seqevi/evidence.py +295 -0
- seqevi/execution_profile.py +526 -0
- seqevi/hashing.py +13 -0
- seqevi/kits/__init__.py +1 -0
- seqevi/kits/dbcan-cazyme.toml +35 -0
- seqevi/resource_lock.py +438 -0
- seqevi/result.py +682 -0
- seqevi/runner.py +163 -0
- seqevi/runtime_identity.py +104 -0
- seqevi/sequence.py +383 -0
- seqevi/service/__init__.py +11 -0
- seqevi/service/app.py +213 -0
- seqevi/service/config.py +38 -0
- seqevi/service/persistence.py +360 -0
- seqevi/store/__init__.py +14 -0
- seqevi/store/artifact.py +225 -0
- seqevi/store/client.py +311 -0
- seqevi/store/contract.py +33 -0
- seqevi/store/factory.py +38 -0
- seqevi/store/local.py +479 -0
- seqevi/store/migration.py +62 -0
- seqevi/store/migrations/__init__.py +1 -0
- seqevi/store/migrations/env.py +30 -0
- seqevi/store/migrations/versions/0001_initial_store.py +103 -0
- seqevi/store/migrations/versions/0002_artifact_byte_size_bigint.py +40 -0
- seqevi/store/migrations/versions/__init__.py +1 -0
- seqevi/store/schema.py +86 -0
- seqevi/store/transport.py +224 -0
- seqevi-0.2.0.dist-info/METADATA +333 -0
- seqevi-0.2.0.dist-info/RECORD +49 -0
- seqevi-0.2.0.dist-info/WHEEL +4 -0
- seqevi-0.2.0.dist-info/entry_points.txt +5 -0
- seqevi-0.2.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,757 @@
|
|
|
1
|
+
"""Direct InterProScan/Pfam annotation adapter."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import csv
|
|
6
|
+
import gzip
|
|
7
|
+
import hashlib
|
|
8
|
+
import json
|
|
9
|
+
import math
|
|
10
|
+
import os
|
|
11
|
+
import re
|
|
12
|
+
import shutil
|
|
13
|
+
import tempfile
|
|
14
|
+
from collections.abc import Mapping
|
|
15
|
+
from dataclasses import asdict, dataclass
|
|
16
|
+
from datetime import datetime
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from typing import Any
|
|
19
|
+
|
|
20
|
+
import polars as pl
|
|
21
|
+
|
|
22
|
+
from seqevi.errors import AdapterError
|
|
23
|
+
from seqevi.evidence import ArtifactFile, EvidenceStatus, sha256_digest
|
|
24
|
+
from seqevi.resource_lock import ResourceComponent, resolve_resource_lock
|
|
25
|
+
from seqevi.runner import ToolCommand, ToolRunner, ToolTimeoutError
|
|
26
|
+
from seqevi.runtime_identity import RuntimeComponent, calculate_runtime_digest
|
|
27
|
+
from seqevi.sequence import SequenceIdentity
|
|
28
|
+
|
|
29
|
+
from .base import AdapterBatchResult, AdapterContract, AdapterSequenceResult
|
|
30
|
+
|
|
31
|
+
ADAPTER_CONTRACT_VERSION = "interpro-pfam/1"
|
|
32
|
+
|
|
33
|
+
INTERPRO_PFAM_EVIDENCE_SCHEMA: Mapping[str, pl.DataType] = {
|
|
34
|
+
"SequenceID": pl.String(),
|
|
35
|
+
"ProteinAccession": pl.String(),
|
|
36
|
+
"SequenceMD5": pl.String(),
|
|
37
|
+
"SequenceLength": pl.Int64(),
|
|
38
|
+
"Analysis": pl.String(),
|
|
39
|
+
"SignatureAccession": pl.String(),
|
|
40
|
+
"SignatureDescription": pl.String(),
|
|
41
|
+
"Start": pl.Int64(),
|
|
42
|
+
"Stop": pl.Int64(),
|
|
43
|
+
"Score": pl.Float64(),
|
|
44
|
+
"Status": pl.String(),
|
|
45
|
+
"InterProAccession": pl.String(),
|
|
46
|
+
"InterProDescription": pl.String(),
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
_VERSION_PATTERN = re.compile(r"\b5\.\d+-\d+\.\d+\b")
|
|
50
|
+
_PFAM_VERSION_PATTERN = re.compile(r"\d+(?:\.\d+)+\Z")
|
|
51
|
+
_PFAM_ACCESSION_PATTERN = re.compile(r"PF\d{5}\Z")
|
|
52
|
+
_INTERPRO_ACCESSION_PATTERN = re.compile(r"IPR\d{6}\Z")
|
|
53
|
+
_TSV_COLUMN_COUNT = 15
|
|
54
|
+
_PROBE_TIMEOUT_SECONDS = 120.0
|
|
55
|
+
_NORMALIZED_ROW_BATCH_SIZE = 1000
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@dataclass(frozen=True, slots=True)
|
|
59
|
+
class InterProPfamParameters:
|
|
60
|
+
"""Fixed scientific parameters for the v1 direct Pfam contract."""
|
|
61
|
+
|
|
62
|
+
application: str = "Pfam"
|
|
63
|
+
output_format: str = "TSV"
|
|
64
|
+
disable_precalculated_lookup: bool = True
|
|
65
|
+
include_go_terms: bool = False
|
|
66
|
+
include_pathways: bool = False
|
|
67
|
+
sequence_type: str = "protein"
|
|
68
|
+
|
|
69
|
+
def __post_init__(self) -> None:
|
|
70
|
+
values = (
|
|
71
|
+
self.application,
|
|
72
|
+
self.output_format,
|
|
73
|
+
self.disable_precalculated_lookup,
|
|
74
|
+
self.include_go_terms,
|
|
75
|
+
self.include_pathways,
|
|
76
|
+
self.sequence_type,
|
|
77
|
+
)
|
|
78
|
+
if values != ("Pfam", "TSV", True, False, False, "protein"):
|
|
79
|
+
raise ValueError(
|
|
80
|
+
"interpro-pfam/1 uses one fixed direct-scan scientific contract"
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
def as_semantic_parameters(self) -> dict[str, object]:
|
|
84
|
+
"""Return every result-affecting parameter with explicit defaults."""
|
|
85
|
+
|
|
86
|
+
return asdict(self)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
class InterProPfamAdapter:
|
|
90
|
+
"""Run local InterProScan Pfam and validate its native TSV output."""
|
|
91
|
+
|
|
92
|
+
def __init__(
|
|
93
|
+
self,
|
|
94
|
+
*,
|
|
95
|
+
executable: Path,
|
|
96
|
+
database: Path,
|
|
97
|
+
parameters: InterProPfamParameters | None = None,
|
|
98
|
+
verify_resource: bool = False,
|
|
99
|
+
environment: Mapping[str, str] | None = None,
|
|
100
|
+
) -> None:
|
|
101
|
+
self.executable = executable.resolve()
|
|
102
|
+
self.database = database.resolve()
|
|
103
|
+
self.parameters = parameters or InterProPfamParameters()
|
|
104
|
+
self.environment = dict(environment or {})
|
|
105
|
+
self.install_dir = self.executable.parent
|
|
106
|
+
self.properties_path = self.install_dir / "interproscan.properties"
|
|
107
|
+
self.jar_path = self.install_dir / "interproscan-5.jar"
|
|
108
|
+
|
|
109
|
+
self._validate_installation_paths()
|
|
110
|
+
self.properties = _read_properties(self.properties_path)
|
|
111
|
+
self.interproscan_version = _probe_interproscan_version(
|
|
112
|
+
self.executable,
|
|
113
|
+
environment=self.environment,
|
|
114
|
+
)
|
|
115
|
+
pfam_version, model_path = _resolve_pfam_model(
|
|
116
|
+
database=self.database,
|
|
117
|
+
properties=self.properties,
|
|
118
|
+
)
|
|
119
|
+
runtime_digest = _calculate_runtime_digest(
|
|
120
|
+
install_dir=self.install_dir,
|
|
121
|
+
executable=self.executable,
|
|
122
|
+
jar_path=self.jar_path,
|
|
123
|
+
properties=self.properties,
|
|
124
|
+
version=self.interproscan_version,
|
|
125
|
+
environment=self.environment,
|
|
126
|
+
)
|
|
127
|
+
resource_id = _calculate_resource_id(
|
|
128
|
+
database=self.database,
|
|
129
|
+
interproscan_version=self.interproscan_version,
|
|
130
|
+
pfam_version=pfam_version,
|
|
131
|
+
model_path=model_path,
|
|
132
|
+
verify=verify_resource,
|
|
133
|
+
)
|
|
134
|
+
self._contract = AdapterContract.from_parameters(
|
|
135
|
+
name="interpro-pfam",
|
|
136
|
+
version=ADAPTER_CONTRACT_VERSION,
|
|
137
|
+
tool_runtime_digest=f"sha256:{runtime_digest}",
|
|
138
|
+
resource_id=resource_id,
|
|
139
|
+
semantic_parameters=self.parameters.as_semantic_parameters(),
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
@property
|
|
143
|
+
def contract(self) -> AdapterContract:
|
|
144
|
+
return self._contract
|
|
145
|
+
|
|
146
|
+
@property
|
|
147
|
+
def evidence_schema(self) -> Mapping[str, pl.DataType]:
|
|
148
|
+
return INTERPRO_PFAM_EVIDENCE_SCHEMA
|
|
149
|
+
|
|
150
|
+
def run_batch(
|
|
151
|
+
self,
|
|
152
|
+
*,
|
|
153
|
+
identities: tuple[SequenceIdentity, ...],
|
|
154
|
+
input_fasta: Path,
|
|
155
|
+
work_dir: Path,
|
|
156
|
+
runner: ToolRunner,
|
|
157
|
+
timeout_seconds: float | None,
|
|
158
|
+
threads: int,
|
|
159
|
+
) -> AdapterBatchResult:
|
|
160
|
+
"""Run one deterministic cache-miss batch and validate every result."""
|
|
161
|
+
|
|
162
|
+
if not identities:
|
|
163
|
+
raise AdapterError("interpro-pfam batch must not be empty")
|
|
164
|
+
ids = [identity.sequence_id for identity in identities]
|
|
165
|
+
if len(ids) != len(set(ids)):
|
|
166
|
+
raise AdapterError("interpro-pfam batch contains duplicate SequenceIDs")
|
|
167
|
+
|
|
168
|
+
raw_path = work_dir / "interpro-pfam.tsv"
|
|
169
|
+
properties_path = work_dir / "interproscan.properties"
|
|
170
|
+
temporary_dir = work_dir / "interproscan-temp"
|
|
171
|
+
_write_runtime_properties(
|
|
172
|
+
source=self.properties_path,
|
|
173
|
+
target=properties_path,
|
|
174
|
+
database=self.database,
|
|
175
|
+
)
|
|
176
|
+
result = runner.run(
|
|
177
|
+
ToolCommand(
|
|
178
|
+
arguments=(
|
|
179
|
+
str(self.executable),
|
|
180
|
+
"--input",
|
|
181
|
+
str(input_fasta),
|
|
182
|
+
"--applications",
|
|
183
|
+
self.parameters.application,
|
|
184
|
+
"--formats",
|
|
185
|
+
self.parameters.output_format,
|
|
186
|
+
"--disable-precalc",
|
|
187
|
+
"--cpu",
|
|
188
|
+
str(threads),
|
|
189
|
+
"--outfile",
|
|
190
|
+
str(raw_path),
|
|
191
|
+
"--tempdir",
|
|
192
|
+
str(temporary_dir),
|
|
193
|
+
),
|
|
194
|
+
working_dir=work_dir,
|
|
195
|
+
stdout_path=work_dir / "interproscan.stdout.log",
|
|
196
|
+
stderr_path=work_dir / "interproscan.stderr.log",
|
|
197
|
+
environment={
|
|
198
|
+
**self.environment,
|
|
199
|
+
"INTERPROSCAN_CONF": str(properties_path),
|
|
200
|
+
},
|
|
201
|
+
),
|
|
202
|
+
timeout_seconds=timeout_seconds,
|
|
203
|
+
)
|
|
204
|
+
if result.return_code != 0:
|
|
205
|
+
raise AdapterError(
|
|
206
|
+
f"InterProScan exited with {result.return_code}; "
|
|
207
|
+
f"stderr: {result.stderr_path}"
|
|
208
|
+
)
|
|
209
|
+
if not raw_path.is_file():
|
|
210
|
+
raise AdapterError("InterProScan did not create its TSV output")
|
|
211
|
+
|
|
212
|
+
normalized, payload_digest_by_id = _parse_tsv(
|
|
213
|
+
raw_path,
|
|
214
|
+
identities=identities,
|
|
215
|
+
normalized_path=work_dir / "interpro-pfam.normalized.parquet",
|
|
216
|
+
)
|
|
217
|
+
sequence_results = tuple(
|
|
218
|
+
_sequence_result(
|
|
219
|
+
identity,
|
|
220
|
+
payload_digest=payload_digest_by_id.get(identity.sequence_id),
|
|
221
|
+
)
|
|
222
|
+
for identity in sorted(identities, key=lambda item: item.sequence_id)
|
|
223
|
+
)
|
|
224
|
+
return AdapterBatchResult(
|
|
225
|
+
sequences=sequence_results,
|
|
226
|
+
raw_artifact=_gzip_artifact(
|
|
227
|
+
raw_path,
|
|
228
|
+
work_dir / "interpro-pfam.tsv.gz",
|
|
229
|
+
),
|
|
230
|
+
normalized_artifact=normalized,
|
|
231
|
+
)
|
|
232
|
+
|
|
233
|
+
def _validate_installation_paths(self) -> None:
|
|
234
|
+
if not self.executable.is_file():
|
|
235
|
+
raise AdapterError(
|
|
236
|
+
f"InterProScan executable is not a file: {self.executable}"
|
|
237
|
+
)
|
|
238
|
+
if not self.database.is_dir():
|
|
239
|
+
raise AdapterError(
|
|
240
|
+
f"InterProScan data directory does not exist: {self.database}"
|
|
241
|
+
)
|
|
242
|
+
if not self.properties_path.is_file():
|
|
243
|
+
raise AdapterError(
|
|
244
|
+
"InterProScan launcher must be beside interproscan.properties: "
|
|
245
|
+
f"{self.properties_path}"
|
|
246
|
+
)
|
|
247
|
+
if not self.jar_path.is_file():
|
|
248
|
+
raise AdapterError(
|
|
249
|
+
f"InterProScan runtime jar does not exist: {self.jar_path}"
|
|
250
|
+
)
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def _probe_interproscan_version(
|
|
254
|
+
executable: Path,
|
|
255
|
+
*,
|
|
256
|
+
environment: Mapping[str, str],
|
|
257
|
+
) -> str:
|
|
258
|
+
with tempfile.TemporaryDirectory(prefix="seqevi-interpro-probe-") as raw_dir:
|
|
259
|
+
probe_dir = Path(raw_dir)
|
|
260
|
+
stdout_path = probe_dir / "stdout.log"
|
|
261
|
+
stderr_path = probe_dir / "stderr.log"
|
|
262
|
+
try:
|
|
263
|
+
result = ToolRunner().run(
|
|
264
|
+
ToolCommand(
|
|
265
|
+
arguments=(str(executable), "--version"),
|
|
266
|
+
working_dir=executable.parent,
|
|
267
|
+
stdout_path=stdout_path,
|
|
268
|
+
stderr_path=stderr_path,
|
|
269
|
+
environment=environment,
|
|
270
|
+
),
|
|
271
|
+
timeout_seconds=_PROBE_TIMEOUT_SECONDS,
|
|
272
|
+
)
|
|
273
|
+
except (OSError, ToolTimeoutError) as error:
|
|
274
|
+
raise AdapterError(f"InterProScan version probe failed: {error}") from error
|
|
275
|
+
|
|
276
|
+
output = "\n".join(
|
|
277
|
+
(
|
|
278
|
+
stdout_path.read_text(encoding="utf-8", errors="replace"),
|
|
279
|
+
stderr_path.read_text(encoding="utf-8", errors="replace"),
|
|
280
|
+
)
|
|
281
|
+
)
|
|
282
|
+
if result.return_code != 0:
|
|
283
|
+
raise AdapterError(
|
|
284
|
+
"InterProScan version probe exited with "
|
|
285
|
+
f"{result.return_code}: {output.strip()}"
|
|
286
|
+
)
|
|
287
|
+
versions = sorted(set(_VERSION_PATTERN.findall(output)))
|
|
288
|
+
if len(versions) != 1:
|
|
289
|
+
raise AdapterError(
|
|
290
|
+
"InterProScan version probe must report exactly one release: "
|
|
291
|
+
f"{versions}"
|
|
292
|
+
)
|
|
293
|
+
return versions[0]
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def _read_properties(path: Path) -> dict[str, str]:
|
|
297
|
+
properties: dict[str, str] = {}
|
|
298
|
+
for line_number, line in enumerate(
|
|
299
|
+
path.read_text(encoding="utf-8").splitlines(), start=1
|
|
300
|
+
):
|
|
301
|
+
stripped = line.strip()
|
|
302
|
+
if not stripped or stripped.startswith(("#", "!")):
|
|
303
|
+
continue
|
|
304
|
+
if "=" not in stripped:
|
|
305
|
+
raise AdapterError(f"invalid InterProScan property at {path}:{line_number}")
|
|
306
|
+
key, value = (part.strip() for part in stripped.split("=", maxsplit=1))
|
|
307
|
+
if not key or key in properties:
|
|
308
|
+
raise AdapterError(
|
|
309
|
+
f"duplicate or empty InterProScan property at {path}:{line_number}"
|
|
310
|
+
)
|
|
311
|
+
properties[key] = value
|
|
312
|
+
|
|
313
|
+
required = {
|
|
314
|
+
"bin.directory",
|
|
315
|
+
"binary.hmmer3.path",
|
|
316
|
+
"data.directory",
|
|
317
|
+
"pfam-a.hmm.path",
|
|
318
|
+
}
|
|
319
|
+
missing = sorted(required - properties.keys())
|
|
320
|
+
if missing:
|
|
321
|
+
raise AdapterError(
|
|
322
|
+
f"InterProScan properties are missing required keys: {missing}"
|
|
323
|
+
)
|
|
324
|
+
return properties
|
|
325
|
+
|
|
326
|
+
|
|
327
|
+
def _resolve_pfam_model(
|
|
328
|
+
*, database: Path, properties: Mapping[str, str]
|
|
329
|
+
) -> tuple[str, Path]:
|
|
330
|
+
configured = properties["pfam-a.hmm.path"]
|
|
331
|
+
prefix = "${data.directory}/"
|
|
332
|
+
if not configured.startswith(prefix):
|
|
333
|
+
raise AdapterError("pfam-a.hmm.path must be relative to ${data.directory}")
|
|
334
|
+
relative = Path(configured.removeprefix(prefix))
|
|
335
|
+
if relative.is_absolute() or ".." in relative.parts:
|
|
336
|
+
raise AdapterError("pfam-a.hmm.path escapes the InterProScan data directory")
|
|
337
|
+
model_path = (database / relative).resolve()
|
|
338
|
+
if not model_path.is_relative_to(database) or not model_path.is_file():
|
|
339
|
+
raise AdapterError(f"Pfam model file does not exist: {model_path}")
|
|
340
|
+
if len(relative.parts) < 3 or relative.parts[0].lower() != "pfam":
|
|
341
|
+
raise AdapterError(f"unexpected Pfam model path: {relative}")
|
|
342
|
+
pfam_version = relative.parts[1]
|
|
343
|
+
if not _PFAM_VERSION_PATTERN.fullmatch(pfam_version):
|
|
344
|
+
raise AdapterError(f"invalid Pfam release in model path: {pfam_version}")
|
|
345
|
+
return pfam_version, model_path
|
|
346
|
+
|
|
347
|
+
|
|
348
|
+
def _calculate_runtime_digest(
|
|
349
|
+
*,
|
|
350
|
+
install_dir: Path,
|
|
351
|
+
executable: Path,
|
|
352
|
+
jar_path: Path,
|
|
353
|
+
properties: Mapping[str, str],
|
|
354
|
+
version: str,
|
|
355
|
+
environment: Mapping[str, str],
|
|
356
|
+
) -> str:
|
|
357
|
+
bin_dir = _resolve_install_path(
|
|
358
|
+
install_dir,
|
|
359
|
+
properties["bin.directory"],
|
|
360
|
+
variables={},
|
|
361
|
+
)
|
|
362
|
+
hmmer_dir = _resolve_install_path(
|
|
363
|
+
install_dir,
|
|
364
|
+
properties["binary.hmmer3.path"],
|
|
365
|
+
variables={"bin.directory": bin_dir},
|
|
366
|
+
)
|
|
367
|
+
if not hmmer_dir.is_dir():
|
|
368
|
+
raise AdapterError(f"InterProScan HMMER directory does not exist: {hmmer_dir}")
|
|
369
|
+
hmmer_files = sorted(path for path in hmmer_dir.rglob("*") if path.is_file())
|
|
370
|
+
if not hmmer_files:
|
|
371
|
+
raise AdapterError(f"InterProScan HMMER directory is empty: {hmmer_dir}")
|
|
372
|
+
|
|
373
|
+
java = shutil.which(
|
|
374
|
+
"java",
|
|
375
|
+
path=environment.get("PATH", os.environ.get("PATH")),
|
|
376
|
+
)
|
|
377
|
+
if java is None:
|
|
378
|
+
raise AdapterError("InterProScan runtime has no Java executable")
|
|
379
|
+
normalized_properties = {
|
|
380
|
+
**properties,
|
|
381
|
+
"data.directory": "${data.directory}",
|
|
382
|
+
}
|
|
383
|
+
properties_digest = sha256_digest(
|
|
384
|
+
json.dumps(
|
|
385
|
+
normalized_properties,
|
|
386
|
+
ensure_ascii=True,
|
|
387
|
+
sort_keys=True,
|
|
388
|
+
separators=(",", ":"),
|
|
389
|
+
).encode("utf-8")
|
|
390
|
+
)
|
|
391
|
+
components = (
|
|
392
|
+
RuntimeComponent("launcher", executable),
|
|
393
|
+
RuntimeComponent("java", Path(java).resolve()),
|
|
394
|
+
RuntimeComponent("interproscan-5.jar", jar_path),
|
|
395
|
+
*(
|
|
396
|
+
RuntimeComponent(
|
|
397
|
+
f"hmmer/{path.relative_to(hmmer_dir).as_posix()}",
|
|
398
|
+
path,
|
|
399
|
+
)
|
|
400
|
+
for path in hmmer_files
|
|
401
|
+
),
|
|
402
|
+
)
|
|
403
|
+
return calculate_runtime_digest(
|
|
404
|
+
runtime_name="interproscan-pfam",
|
|
405
|
+
versions={
|
|
406
|
+
"interproscan": version,
|
|
407
|
+
"properties": f"sha256:{properties_digest}",
|
|
408
|
+
},
|
|
409
|
+
components=components,
|
|
410
|
+
)
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
def _resolve_install_path(
|
|
414
|
+
install_dir: Path,
|
|
415
|
+
raw_value: str,
|
|
416
|
+
*,
|
|
417
|
+
variables: Mapping[str, Path],
|
|
418
|
+
) -> Path:
|
|
419
|
+
expanded = raw_value
|
|
420
|
+
for name, path in variables.items():
|
|
421
|
+
expanded = expanded.replace(f"${{{name}}}", str(path))
|
|
422
|
+
if "${" in expanded:
|
|
423
|
+
raise AdapterError(f"unsupported InterProScan property expression: {raw_value}")
|
|
424
|
+
candidate = Path(expanded)
|
|
425
|
+
resolved = (
|
|
426
|
+
candidate.resolve()
|
|
427
|
+
if candidate.is_absolute()
|
|
428
|
+
else (install_dir / candidate).resolve()
|
|
429
|
+
)
|
|
430
|
+
if not resolved.is_relative_to(install_dir):
|
|
431
|
+
raise AdapterError(f"InterProScan runtime path escapes its install: {resolved}")
|
|
432
|
+
return resolved
|
|
433
|
+
|
|
434
|
+
|
|
435
|
+
def _calculate_resource_id(
|
|
436
|
+
*,
|
|
437
|
+
database: Path,
|
|
438
|
+
interproscan_version: str,
|
|
439
|
+
pfam_version: str,
|
|
440
|
+
model_path: Path,
|
|
441
|
+
verify: bool = False,
|
|
442
|
+
) -> str:
|
|
443
|
+
interpro_release = interproscan_version.split("-", maxsplit=1)[1]
|
|
444
|
+
metadata_path = model_path.with_name("pfam_a.dat")
|
|
445
|
+
if not metadata_path.is_file():
|
|
446
|
+
raise AdapterError(f"Pfam metadata file does not exist: {metadata_path}")
|
|
447
|
+
declarations = (
|
|
448
|
+
ResourceComponent(
|
|
449
|
+
"pfam-model",
|
|
450
|
+
model_path.relative_to(database).as_posix(),
|
|
451
|
+
),
|
|
452
|
+
ResourceComponent(
|
|
453
|
+
"pfam-metadata",
|
|
454
|
+
metadata_path.relative_to(database).as_posix(),
|
|
455
|
+
),
|
|
456
|
+
)
|
|
457
|
+
locked = resolve_resource_lock(
|
|
458
|
+
database=database,
|
|
459
|
+
resource_name="interpro",
|
|
460
|
+
resource_version=interpro_release,
|
|
461
|
+
components=declarations,
|
|
462
|
+
verify=verify,
|
|
463
|
+
)
|
|
464
|
+
digest = sha256_digest(
|
|
465
|
+
json.dumps(
|
|
466
|
+
[
|
|
467
|
+
(component.name, locked.hash_for(component.name))
|
|
468
|
+
for component in declarations
|
|
469
|
+
],
|
|
470
|
+
separators=(",", ":"),
|
|
471
|
+
).encode("utf-8")
|
|
472
|
+
)
|
|
473
|
+
return f"interpro/{interpro_release}/pfam/{pfam_version}/sha256:{digest}"
|
|
474
|
+
|
|
475
|
+
|
|
476
|
+
def _write_runtime_properties(*, source: Path, target: Path, database: Path) -> None:
|
|
477
|
+
lines = source.read_text(encoding="utf-8").splitlines()
|
|
478
|
+
replaced = False
|
|
479
|
+
output = []
|
|
480
|
+
for line in lines:
|
|
481
|
+
stripped = line.strip()
|
|
482
|
+
if (
|
|
483
|
+
stripped
|
|
484
|
+
and not stripped.startswith(("#", "!"))
|
|
485
|
+
and stripped.split("=", maxsplit=1)[0].strip() == "data.directory"
|
|
486
|
+
):
|
|
487
|
+
output.append(f"data.directory={database}")
|
|
488
|
+
replaced = True
|
|
489
|
+
else:
|
|
490
|
+
output.append(line)
|
|
491
|
+
if not replaced:
|
|
492
|
+
raise AdapterError("InterProScan properties have no data.directory entry")
|
|
493
|
+
target.write_text("\n".join(output) + "\n", encoding="utf-8")
|
|
494
|
+
|
|
495
|
+
|
|
496
|
+
def _parse_tsv(
|
|
497
|
+
path: Path,
|
|
498
|
+
*,
|
|
499
|
+
identities: tuple[SequenceIdentity, ...],
|
|
500
|
+
normalized_path: Path,
|
|
501
|
+
) -> tuple[ArtifactFile | None, dict[str, str]]:
|
|
502
|
+
expected = {identity.sequence_id: identity for identity in identities}
|
|
503
|
+
rows: list[dict[str, object]] = []
|
|
504
|
+
with tempfile.TemporaryDirectory(
|
|
505
|
+
prefix=".interpro-pfam-normalized-", dir=normalized_path.parent
|
|
506
|
+
) as raw_parts_dir:
|
|
507
|
+
parts_dir = Path(raw_parts_dir)
|
|
508
|
+
part_paths: list[Path] = []
|
|
509
|
+
try:
|
|
510
|
+
with path.open("r", encoding="utf-8", newline="") as handle:
|
|
511
|
+
reader = csv.reader(handle, delimiter="\t")
|
|
512
|
+
for line_number, fields in enumerate(reader, start=1):
|
|
513
|
+
if not fields or fields == [""]:
|
|
514
|
+
raise AdapterError(
|
|
515
|
+
"InterProScan TSV contains a blank row at line "
|
|
516
|
+
f"{line_number}"
|
|
517
|
+
)
|
|
518
|
+
if len(fields) != _TSV_COLUMN_COUNT:
|
|
519
|
+
raise AdapterError(
|
|
520
|
+
f"InterProScan TSV line {line_number} has "
|
|
521
|
+
f"{len(fields)} columns; expected {_TSV_COLUMN_COUNT}"
|
|
522
|
+
)
|
|
523
|
+
row = _parse_tsv_row(
|
|
524
|
+
fields, line_number=line_number, expected=expected
|
|
525
|
+
)
|
|
526
|
+
rows.append(row)
|
|
527
|
+
if len(rows) >= _NORMALIZED_ROW_BATCH_SIZE:
|
|
528
|
+
part_paths.append(_write_normalized_part(rows, parts_dir))
|
|
529
|
+
rows.clear()
|
|
530
|
+
except UnicodeDecodeError as error:
|
|
531
|
+
raise AdapterError(
|
|
532
|
+
f"InterProScan TSV is not valid UTF-8: {error}"
|
|
533
|
+
) from error
|
|
534
|
+
|
|
535
|
+
if rows:
|
|
536
|
+
part_paths.append(_write_normalized_part(rows, parts_dir))
|
|
537
|
+
if not part_paths:
|
|
538
|
+
return None, {}
|
|
539
|
+
normalized = pl.concat([pl.scan_parquet(part) for part in part_paths]).sort(
|
|
540
|
+
*INTERPRO_PFAM_EVIDENCE_SCHEMA,
|
|
541
|
+
nulls_last=True,
|
|
542
|
+
)
|
|
543
|
+
payload_digest_by_id = _payload_digests(normalized)
|
|
544
|
+
normalized.sink_parquet(
|
|
545
|
+
normalized_path, compression="zstd", maintain_order=True
|
|
546
|
+
)
|
|
547
|
+
return (
|
|
548
|
+
ArtifactFile.from_path(normalized_path, "application/vnd.apache.parquet"),
|
|
549
|
+
payload_digest_by_id,
|
|
550
|
+
)
|
|
551
|
+
|
|
552
|
+
|
|
553
|
+
def _write_normalized_part(rows: list[dict[str, object]], directory: Path) -> Path:
|
|
554
|
+
path = directory / f"part-{len(tuple(directory.iterdir())):06d}.parquet"
|
|
555
|
+
pl.DataFrame(rows, schema=INTERPRO_PFAM_EVIDENCE_SCHEMA).write_parquet(path)
|
|
556
|
+
return path
|
|
557
|
+
|
|
558
|
+
|
|
559
|
+
def _payload_digests(frame: pl.LazyFrame) -> dict[str, str]:
|
|
560
|
+
digests: dict[str, str] = {}
|
|
561
|
+
current_sequence_id: str | None = None
|
|
562
|
+
current_hasher: Any | None = None
|
|
563
|
+
previous_row: tuple[object, ...] | None = None
|
|
564
|
+
row_index = 0
|
|
565
|
+
for batch in frame.collect_batches(
|
|
566
|
+
chunk_size=_NORMALIZED_ROW_BATCH_SIZE,
|
|
567
|
+
engine="streaming",
|
|
568
|
+
):
|
|
569
|
+
for row in batch.iter_rows(named=True):
|
|
570
|
+
row_key = tuple(row[column] for column in INTERPRO_PFAM_EVIDENCE_SCHEMA)
|
|
571
|
+
if row_key == previous_row:
|
|
572
|
+
raise AdapterError("InterProScan TSV contains a duplicate match")
|
|
573
|
+
previous_row = row_key
|
|
574
|
+
sequence_id = str(row["SequenceID"])
|
|
575
|
+
if sequence_id != current_sequence_id:
|
|
576
|
+
if current_sequence_id is not None and current_hasher is not None:
|
|
577
|
+
current_hasher.update(b"]")
|
|
578
|
+
digests[current_sequence_id] = current_hasher.hexdigest()
|
|
579
|
+
current_sequence_id = sequence_id
|
|
580
|
+
current_hasher = hashlib.sha256()
|
|
581
|
+
current_hasher.update(b"[")
|
|
582
|
+
row_index = 0
|
|
583
|
+
if current_hasher is None:
|
|
584
|
+
raise AssertionError("payload digest state was not initialized")
|
|
585
|
+
if row_index:
|
|
586
|
+
current_hasher.update(b",")
|
|
587
|
+
current_hasher.update(
|
|
588
|
+
json.dumps(
|
|
589
|
+
row,
|
|
590
|
+
ensure_ascii=True,
|
|
591
|
+
allow_nan=False,
|
|
592
|
+
sort_keys=True,
|
|
593
|
+
separators=(",", ":"),
|
|
594
|
+
).encode("utf-8")
|
|
595
|
+
)
|
|
596
|
+
row_index += 1
|
|
597
|
+
if current_sequence_id is not None and current_hasher is not None:
|
|
598
|
+
current_hasher.update(b"]")
|
|
599
|
+
digests[current_sequence_id] = current_hasher.hexdigest()
|
|
600
|
+
return digests
|
|
601
|
+
|
|
602
|
+
|
|
603
|
+
def _parse_tsv_row(
|
|
604
|
+
fields: list[str],
|
|
605
|
+
*,
|
|
606
|
+
line_number: int,
|
|
607
|
+
expected: Mapping[str, SequenceIdentity],
|
|
608
|
+
) -> dict[str, object]:
|
|
609
|
+
(
|
|
610
|
+
accession,
|
|
611
|
+
sequence_md5,
|
|
612
|
+
raw_length,
|
|
613
|
+
analysis,
|
|
614
|
+
signature_accession,
|
|
615
|
+
signature_description,
|
|
616
|
+
raw_start,
|
|
617
|
+
raw_stop,
|
|
618
|
+
raw_score,
|
|
619
|
+
status,
|
|
620
|
+
run_date,
|
|
621
|
+
interpro_accession,
|
|
622
|
+
interpro_description,
|
|
623
|
+
go_annotations,
|
|
624
|
+
pathway_annotations,
|
|
625
|
+
) = fields
|
|
626
|
+
identity = expected.get(accession)
|
|
627
|
+
if identity is None:
|
|
628
|
+
raise AdapterError(
|
|
629
|
+
f"InterProScan TSV line {line_number} has an unknown SequenceID: "
|
|
630
|
+
f"{accession}"
|
|
631
|
+
)
|
|
632
|
+
if sequence_md5.lower() != identity.md5:
|
|
633
|
+
raise AdapterError(
|
|
634
|
+
f"InterProScan TSV line {line_number} MD5 does not match {accession}"
|
|
635
|
+
)
|
|
636
|
+
sequence_length = _parse_int(raw_length, field="sequence length", line=line_number)
|
|
637
|
+
start = _parse_int(raw_start, field="start", line=line_number)
|
|
638
|
+
stop = _parse_int(raw_stop, field="stop", line=line_number)
|
|
639
|
+
if sequence_length != identity.length:
|
|
640
|
+
raise AdapterError(
|
|
641
|
+
f"InterProScan TSV line {line_number} length does not match {accession}"
|
|
642
|
+
)
|
|
643
|
+
if not 1 <= start <= stop <= sequence_length:
|
|
644
|
+
raise AdapterError(
|
|
645
|
+
f"InterProScan TSV line {line_number} has invalid match coordinates"
|
|
646
|
+
)
|
|
647
|
+
if analysis != "Pfam":
|
|
648
|
+
raise AdapterError(
|
|
649
|
+
f"InterProScan TSV line {line_number} is not a Pfam match: {analysis}"
|
|
650
|
+
)
|
|
651
|
+
if not _PFAM_ACCESSION_PATTERN.fullmatch(signature_accession):
|
|
652
|
+
raise AdapterError(
|
|
653
|
+
f"InterProScan TSV line {line_number} has invalid Pfam accession: "
|
|
654
|
+
f"{signature_accession}"
|
|
655
|
+
)
|
|
656
|
+
score = _parse_float(raw_score, field="score", line=line_number)
|
|
657
|
+
if status != "T":
|
|
658
|
+
raise AdapterError(
|
|
659
|
+
f"InterProScan TSV line {line_number} has invalid status: {status}"
|
|
660
|
+
)
|
|
661
|
+
try:
|
|
662
|
+
datetime.strptime(run_date, "%d-%m-%Y")
|
|
663
|
+
except ValueError as error:
|
|
664
|
+
raise AdapterError(
|
|
665
|
+
f"InterProScan TSV line {line_number} has invalid run date: {run_date}"
|
|
666
|
+
) from error
|
|
667
|
+
normalized_interpro = _optional_text(interpro_accession)
|
|
668
|
+
if normalized_interpro is not None and not _INTERPRO_ACCESSION_PATTERN.fullmatch(
|
|
669
|
+
normalized_interpro
|
|
670
|
+
):
|
|
671
|
+
raise AdapterError(
|
|
672
|
+
f"InterProScan TSV line {line_number} has invalid InterPro accession: "
|
|
673
|
+
f"{normalized_interpro}"
|
|
674
|
+
)
|
|
675
|
+
if go_annotations != "-" or pathway_annotations != "-":
|
|
676
|
+
raise AdapterError(
|
|
677
|
+
f"InterProScan TSV line {line_number} contains GO or pathway annotations "
|
|
678
|
+
"outside the interpro-pfam/1 contract"
|
|
679
|
+
)
|
|
680
|
+
|
|
681
|
+
return {
|
|
682
|
+
"SequenceID": identity.sequence_id,
|
|
683
|
+
"ProteinAccession": accession,
|
|
684
|
+
"SequenceMD5": identity.md5,
|
|
685
|
+
"SequenceLength": sequence_length,
|
|
686
|
+
"Analysis": analysis,
|
|
687
|
+
"SignatureAccession": signature_accession,
|
|
688
|
+
"SignatureDescription": _optional_text(signature_description),
|
|
689
|
+
"Start": start,
|
|
690
|
+
"Stop": stop,
|
|
691
|
+
"Score": score,
|
|
692
|
+
"Status": status,
|
|
693
|
+
"InterProAccession": normalized_interpro,
|
|
694
|
+
"InterProDescription": _optional_text(interpro_description),
|
|
695
|
+
}
|
|
696
|
+
|
|
697
|
+
|
|
698
|
+
def _parse_int(value: str, *, field: str, line: int) -> int:
|
|
699
|
+
try:
|
|
700
|
+
return int(value)
|
|
701
|
+
except ValueError as error:
|
|
702
|
+
raise AdapterError(
|
|
703
|
+
f"InterProScan TSV line {line} has invalid {field}: {value}"
|
|
704
|
+
) from error
|
|
705
|
+
|
|
706
|
+
|
|
707
|
+
def _parse_float(value: str, *, field: str, line: int) -> float:
|
|
708
|
+
try:
|
|
709
|
+
parsed = float(value)
|
|
710
|
+
except ValueError as error:
|
|
711
|
+
raise AdapterError(
|
|
712
|
+
f"InterProScan TSV line {line} has invalid {field}: {value}"
|
|
713
|
+
) from error
|
|
714
|
+
if not math.isfinite(parsed):
|
|
715
|
+
raise AdapterError(
|
|
716
|
+
f"InterProScan TSV line {line} has non-finite {field}: {value}"
|
|
717
|
+
)
|
|
718
|
+
return parsed
|
|
719
|
+
|
|
720
|
+
|
|
721
|
+
def _optional_text(value: str) -> str | None:
|
|
722
|
+
return None if value == "-" else value
|
|
723
|
+
|
|
724
|
+
|
|
725
|
+
def _gzip_artifact(source: Path, target: Path) -> ArtifactFile:
|
|
726
|
+
with (
|
|
727
|
+
source.open("rb") as source_handle,
|
|
728
|
+
target.open("wb") as target_handle,
|
|
729
|
+
gzip.GzipFile(fileobj=target_handle, mode="wb", mtime=0) as compressed,
|
|
730
|
+
):
|
|
731
|
+
shutil.copyfileobj(source_handle, compressed)
|
|
732
|
+
return ArtifactFile.from_path(target, "application/gzip")
|
|
733
|
+
|
|
734
|
+
|
|
735
|
+
def _sequence_result(
|
|
736
|
+
identity: SequenceIdentity,
|
|
737
|
+
*,
|
|
738
|
+
payload_digest: str | None,
|
|
739
|
+
) -> AdapterSequenceResult:
|
|
740
|
+
if payload_digest is None:
|
|
741
|
+
payload = json.dumps(
|
|
742
|
+
{"SequenceID": identity.sequence_id, "Status": "no_hit"},
|
|
743
|
+
ensure_ascii=True,
|
|
744
|
+
sort_keys=True,
|
|
745
|
+
separators=(",", ":"),
|
|
746
|
+
).encode("utf-8")
|
|
747
|
+
return AdapterSequenceResult(
|
|
748
|
+
sequence_id=identity.sequence_id,
|
|
749
|
+
status=EvidenceStatus.NO_HIT,
|
|
750
|
+
payload_digest=sha256_digest(payload),
|
|
751
|
+
)
|
|
752
|
+
|
|
753
|
+
return AdapterSequenceResult(
|
|
754
|
+
sequence_id=identity.sequence_id,
|
|
755
|
+
status=EvidenceStatus.HIT,
|
|
756
|
+
payload_digest=payload_digest,
|
|
757
|
+
)
|