seqevi 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- seqevi/__init__.py +7 -0
- seqevi/__main__.py +6 -0
- seqevi/adapters/__init__.py +41 -0
- seqevi/adapters/base.py +131 -0
- seqevi/adapters/dbcan_cazyme.py +588 -0
- seqevi/adapters/eggnog.py +674 -0
- seqevi/adapters/interpro_pfam.py +757 -0
- seqevi/adapters/registry.py +68 -0
- seqevi/annotate.py +413 -0
- seqevi/api.py +390 -0
- seqevi/cli.py +610 -0
- seqevi/distribution/__init__.py +13 -0
- seqevi/distribution/manifest.py +199 -0
- seqevi/distribution/oci.py +490 -0
- seqevi/distribution/setup.py +752 -0
- seqevi/errors.py +73 -0
- seqevi/evidence.py +295 -0
- seqevi/execution_profile.py +526 -0
- seqevi/hashing.py +13 -0
- seqevi/kits/__init__.py +1 -0
- seqevi/kits/dbcan-cazyme.toml +35 -0
- seqevi/resource_lock.py +438 -0
- seqevi/result.py +682 -0
- seqevi/runner.py +163 -0
- seqevi/runtime_identity.py +104 -0
- seqevi/sequence.py +383 -0
- seqevi/service/__init__.py +11 -0
- seqevi/service/app.py +213 -0
- seqevi/service/config.py +38 -0
- seqevi/service/persistence.py +360 -0
- seqevi/store/__init__.py +14 -0
- seqevi/store/artifact.py +225 -0
- seqevi/store/client.py +311 -0
- seqevi/store/contract.py +33 -0
- seqevi/store/factory.py +38 -0
- seqevi/store/local.py +479 -0
- seqevi/store/migration.py +62 -0
- seqevi/store/migrations/__init__.py +1 -0
- seqevi/store/migrations/env.py +30 -0
- seqevi/store/migrations/versions/0001_initial_store.py +103 -0
- seqevi/store/migrations/versions/0002_artifact_byte_size_bigint.py +40 -0
- seqevi/store/migrations/versions/__init__.py +1 -0
- seqevi/store/schema.py +86 -0
- seqevi/store/transport.py +224 -0
- seqevi-0.2.0.dist-info/METADATA +333 -0
- seqevi-0.2.0.dist-info/RECORD +49 -0
- seqevi-0.2.0.dist-info/WHEEL +4 -0
- seqevi-0.2.0.dist-info/entry_points.txt +5 -0
- seqevi-0.2.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,674 @@
|
|
|
1
|
+
"""eggNOG-mapper protein annotation adapter."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import csv
|
|
6
|
+
import gzip
|
|
7
|
+
import json
|
|
8
|
+
import math
|
|
9
|
+
import os
|
|
10
|
+
import re
|
|
11
|
+
import shlex
|
|
12
|
+
import shutil
|
|
13
|
+
import tempfile
|
|
14
|
+
from collections.abc import Mapping
|
|
15
|
+
from dataclasses import asdict, dataclass
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
import polars as pl
|
|
19
|
+
|
|
20
|
+
from seqevi.errors import AdapterError
|
|
21
|
+
from seqevi.evidence import ArtifactFile, EvidenceStatus, sha256_digest
|
|
22
|
+
from seqevi.resource_lock import ResourceComponent, resolve_resource_lock
|
|
23
|
+
from seqevi.runner import ToolCommand, ToolRunner, ToolTimeoutError
|
|
24
|
+
from seqevi.runtime_identity import RuntimeComponent, calculate_runtime_digest
|
|
25
|
+
from seqevi.sequence import SequenceIdentity
|
|
26
|
+
|
|
27
|
+
from .base import AdapterBatchResult, AdapterContract, AdapterSequenceResult
|
|
28
|
+
|
|
29
|
+
ADAPTER_CONTRACT_VERSION = "eggnog/1"
|
|
30
|
+
|
|
31
|
+
_NATIVE_COLUMNS = (
|
|
32
|
+
"query",
|
|
33
|
+
"seed_ortholog",
|
|
34
|
+
"evalue",
|
|
35
|
+
"score",
|
|
36
|
+
"eggNOG_OGs",
|
|
37
|
+
"max_annot_lvl",
|
|
38
|
+
"COG_category",
|
|
39
|
+
"Description",
|
|
40
|
+
"Preferred_name",
|
|
41
|
+
"GOs",
|
|
42
|
+
"EC",
|
|
43
|
+
"KEGG_ko",
|
|
44
|
+
"KEGG_Pathway",
|
|
45
|
+
"KEGG_Module",
|
|
46
|
+
"KEGG_Reaction",
|
|
47
|
+
"KEGG_rclass",
|
|
48
|
+
"BRITE",
|
|
49
|
+
"KEGG_TC",
|
|
50
|
+
"CAZy",
|
|
51
|
+
"BiGG_Reaction",
|
|
52
|
+
"PFAMs",
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
EGGNOG_EVIDENCE_SCHEMA: Mapping[str, pl.DataType] = {
|
|
56
|
+
"SequenceID": pl.String(),
|
|
57
|
+
**{
|
|
58
|
+
column: pl.Float64() if column in {"evalue", "score"} else pl.String()
|
|
59
|
+
for column in _NATIVE_COLUMNS
|
|
60
|
+
},
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
_VERSION_PATTERN = re.compile(r"\bemapper-(2\.\d+\.\d+)\b")
|
|
64
|
+
_EXPECTED_DB_PATTERN = re.compile(r"Expected eggNOG DB version:\s*([^\s/]+)")
|
|
65
|
+
_INSTALLED_DB_PATTERN = re.compile(r"Installed eggNOG DB version:\s*([^\s/]+)")
|
|
66
|
+
_DIAMOND_VERSION_PATTERN = re.compile(r"\bdiamond version\s+([^\s/]+)")
|
|
67
|
+
_PROBE_TIMEOUT_SECONDS = 120.0
|
|
68
|
+
_REQUIRED_DATABASE_FILES = (
|
|
69
|
+
"eggnog.db",
|
|
70
|
+
"eggnog.taxa.db",
|
|
71
|
+
"eggnog_proteins.dmnd",
|
|
72
|
+
)
|
|
73
|
+
_OPTIONAL_DATABASE_FILES = ("eggnog.taxa.db.traverse.pkl",)
|
|
74
|
+
_NORMALIZED_ROW_BATCH_SIZE = 1000
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
@dataclass(frozen=True, slots=True)
|
|
78
|
+
class EggnogParameters:
|
|
79
|
+
"""Fixed scientific parameters for the v1 eggNOG protein contract."""
|
|
80
|
+
|
|
81
|
+
search_mode: str = "diamond"
|
|
82
|
+
input_type: str = "proteins"
|
|
83
|
+
seed_ortholog_evalue: float = 0.001
|
|
84
|
+
tax_scope: str = "auto"
|
|
85
|
+
target_orthologs: str = "all"
|
|
86
|
+
go_evidence: str = "non-electronic"
|
|
87
|
+
pfam_realign: str = "none"
|
|
88
|
+
|
|
89
|
+
def __post_init__(self) -> None:
|
|
90
|
+
values = tuple(asdict(self).values())
|
|
91
|
+
if values != (
|
|
92
|
+
"diamond",
|
|
93
|
+
"proteins",
|
|
94
|
+
0.001,
|
|
95
|
+
"auto",
|
|
96
|
+
"all",
|
|
97
|
+
"non-electronic",
|
|
98
|
+
"none",
|
|
99
|
+
):
|
|
100
|
+
raise ValueError("eggnog/1 uses one fixed protein annotation contract")
|
|
101
|
+
|
|
102
|
+
def as_semantic_parameters(self) -> dict[str, object]:
|
|
103
|
+
"""Return every result-affecting parameter with explicit defaults."""
|
|
104
|
+
|
|
105
|
+
return asdict(self)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
class EggnogAdapter:
|
|
109
|
+
"""Run eggNOG-mapper 2.x and validate its native annotations table."""
|
|
110
|
+
|
|
111
|
+
def __init__(
|
|
112
|
+
self,
|
|
113
|
+
*,
|
|
114
|
+
executable: Path,
|
|
115
|
+
database: Path,
|
|
116
|
+
parameters: EggnogParameters | None = None,
|
|
117
|
+
verify_resource: bool = False,
|
|
118
|
+
environment: Mapping[str, str] | None = None,
|
|
119
|
+
) -> None:
|
|
120
|
+
self.executable = executable.resolve()
|
|
121
|
+
self.database = database.resolve()
|
|
122
|
+
self.parameters = parameters or EggnogParameters()
|
|
123
|
+
self.environment = dict(environment or {})
|
|
124
|
+
if not self.executable.is_file():
|
|
125
|
+
raise AdapterError(f"eggNOG-mapper executable is not a file: {executable}")
|
|
126
|
+
if not self.database.is_dir():
|
|
127
|
+
raise AdapterError(f"eggNOG database is not a directory: {database}")
|
|
128
|
+
|
|
129
|
+
version_output = _probe_version(
|
|
130
|
+
self.executable,
|
|
131
|
+
self.database,
|
|
132
|
+
environment=self.environment,
|
|
133
|
+
)
|
|
134
|
+
tool_version, database_version, reported_diamond_version = (
|
|
135
|
+
_parse_version_output(version_output)
|
|
136
|
+
)
|
|
137
|
+
diamond = _resolve_diamond(self.executable, environment=self.environment)
|
|
138
|
+
diamond_version = _probe_diamond_version(
|
|
139
|
+
diamond,
|
|
140
|
+
environment=self.environment,
|
|
141
|
+
)
|
|
142
|
+
if diamond_version != reported_diamond_version:
|
|
143
|
+
raise AdapterError(
|
|
144
|
+
"eggNOG-mapper and the selected DIAMOND executable report "
|
|
145
|
+
"different versions"
|
|
146
|
+
)
|
|
147
|
+
runtime_digest = _runtime_digest(
|
|
148
|
+
self.executable,
|
|
149
|
+
tool_version=tool_version,
|
|
150
|
+
diamond=diamond,
|
|
151
|
+
diamond_version=diamond_version,
|
|
152
|
+
environment=self.environment,
|
|
153
|
+
)
|
|
154
|
+
resource_id = _resource_id(
|
|
155
|
+
self.database,
|
|
156
|
+
database_version,
|
|
157
|
+
verify=verify_resource,
|
|
158
|
+
)
|
|
159
|
+
self._contract = AdapterContract.from_parameters(
|
|
160
|
+
name="eggnog",
|
|
161
|
+
version=ADAPTER_CONTRACT_VERSION,
|
|
162
|
+
tool_runtime_digest=f"sha256:{runtime_digest}",
|
|
163
|
+
resource_id=resource_id,
|
|
164
|
+
semantic_parameters=self.parameters.as_semantic_parameters(),
|
|
165
|
+
)
|
|
166
|
+
self.tool_version = tool_version
|
|
167
|
+
self.diamond_version = diamond_version
|
|
168
|
+
self.database_version = database_version
|
|
169
|
+
|
|
170
|
+
@property
|
|
171
|
+
def contract(self) -> AdapterContract:
|
|
172
|
+
return self._contract
|
|
173
|
+
|
|
174
|
+
@property
|
|
175
|
+
def evidence_schema(self) -> Mapping[str, pl.DataType]:
|
|
176
|
+
return EGGNOG_EVIDENCE_SCHEMA
|
|
177
|
+
|
|
178
|
+
def run_batch(
|
|
179
|
+
self,
|
|
180
|
+
*,
|
|
181
|
+
identities: tuple[SequenceIdentity, ...],
|
|
182
|
+
input_fasta: Path,
|
|
183
|
+
work_dir: Path,
|
|
184
|
+
runner: ToolRunner,
|
|
185
|
+
timeout_seconds: float | None,
|
|
186
|
+
threads: int,
|
|
187
|
+
) -> AdapterBatchResult:
|
|
188
|
+
"""Run one deterministic cache-miss batch and validate every row."""
|
|
189
|
+
|
|
190
|
+
if not identities:
|
|
191
|
+
raise AdapterError("eggnog batch must not be empty")
|
|
192
|
+
output_name = "seqevi"
|
|
193
|
+
raw_path = work_dir / f"{output_name}.emapper.annotations"
|
|
194
|
+
parameters = self.parameters
|
|
195
|
+
result = runner.run(
|
|
196
|
+
ToolCommand(
|
|
197
|
+
arguments=(
|
|
198
|
+
str(self.executable),
|
|
199
|
+
"-i",
|
|
200
|
+
str(input_fasta),
|
|
201
|
+
"--itype",
|
|
202
|
+
parameters.input_type,
|
|
203
|
+
"--output",
|
|
204
|
+
output_name,
|
|
205
|
+
"--output_dir",
|
|
206
|
+
str(work_dir),
|
|
207
|
+
"--data_dir",
|
|
208
|
+
str(self.database),
|
|
209
|
+
"--cpu",
|
|
210
|
+
str(threads),
|
|
211
|
+
"--override",
|
|
212
|
+
"-m",
|
|
213
|
+
parameters.search_mode,
|
|
214
|
+
"--seed_ortholog_evalue",
|
|
215
|
+
str(parameters.seed_ortholog_evalue),
|
|
216
|
+
"--tax_scope",
|
|
217
|
+
parameters.tax_scope,
|
|
218
|
+
"--target_orthologs",
|
|
219
|
+
parameters.target_orthologs,
|
|
220
|
+
"--go_evidence",
|
|
221
|
+
parameters.go_evidence,
|
|
222
|
+
"--pfam_realign",
|
|
223
|
+
parameters.pfam_realign,
|
|
224
|
+
),
|
|
225
|
+
working_dir=work_dir,
|
|
226
|
+
stdout_path=work_dir / "eggnog.stdout.log",
|
|
227
|
+
stderr_path=work_dir / "eggnog.stderr.log",
|
|
228
|
+
environment=_runtime_environment(
|
|
229
|
+
self.executable,
|
|
230
|
+
overlay=self.environment,
|
|
231
|
+
),
|
|
232
|
+
),
|
|
233
|
+
timeout_seconds=timeout_seconds,
|
|
234
|
+
)
|
|
235
|
+
if result.return_code != 0:
|
|
236
|
+
raise AdapterError(
|
|
237
|
+
f"eggNOG-mapper exited with {result.return_code}; "
|
|
238
|
+
f"stderr: {result.stderr_path}"
|
|
239
|
+
)
|
|
240
|
+
if not raw_path.is_file():
|
|
241
|
+
raise AdapterError("eggNOG-mapper did not create its annotations output")
|
|
242
|
+
|
|
243
|
+
normalized, payload_digest_by_id = _parse_annotations(
|
|
244
|
+
raw_path,
|
|
245
|
+
identities=identities,
|
|
246
|
+
normalized_path=work_dir / "eggnog.normalized.parquet",
|
|
247
|
+
)
|
|
248
|
+
sequence_results = tuple(
|
|
249
|
+
_sequence_result(
|
|
250
|
+
identity,
|
|
251
|
+
payload_digest=payload_digest_by_id.get(identity.sequence_id),
|
|
252
|
+
)
|
|
253
|
+
for identity in sorted(identities, key=lambda item: item.sequence_id)
|
|
254
|
+
)
|
|
255
|
+
return AdapterBatchResult(
|
|
256
|
+
sequences=sequence_results,
|
|
257
|
+
raw_artifact=_gzip_annotations_artifact(
|
|
258
|
+
raw_path,
|
|
259
|
+
work_dir / "eggnog.annotations.tsv.gz",
|
|
260
|
+
),
|
|
261
|
+
normalized_artifact=normalized,
|
|
262
|
+
)
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
def _probe_version(
|
|
266
|
+
executable: Path,
|
|
267
|
+
database: Path,
|
|
268
|
+
*,
|
|
269
|
+
environment: Mapping[str, str],
|
|
270
|
+
) -> str:
|
|
271
|
+
with tempfile.TemporaryDirectory(prefix="seqevi-eggnog-probe-") as raw_dir:
|
|
272
|
+
root = Path(raw_dir)
|
|
273
|
+
stdout_path = root / "stdout.log"
|
|
274
|
+
stderr_path = root / "stderr.log"
|
|
275
|
+
try:
|
|
276
|
+
result = ToolRunner().run(
|
|
277
|
+
ToolCommand(
|
|
278
|
+
arguments=(
|
|
279
|
+
str(executable),
|
|
280
|
+
"--version",
|
|
281
|
+
"--data_dir",
|
|
282
|
+
str(database),
|
|
283
|
+
),
|
|
284
|
+
working_dir=executable.parent,
|
|
285
|
+
stdout_path=stdout_path,
|
|
286
|
+
stderr_path=stderr_path,
|
|
287
|
+
environment=_runtime_environment(
|
|
288
|
+
executable,
|
|
289
|
+
overlay=environment,
|
|
290
|
+
),
|
|
291
|
+
),
|
|
292
|
+
timeout_seconds=_PROBE_TIMEOUT_SECONDS,
|
|
293
|
+
)
|
|
294
|
+
except (OSError, ToolTimeoutError) as error:
|
|
295
|
+
raise AdapterError(
|
|
296
|
+
f"eggNOG-mapper version probe failed: {error}"
|
|
297
|
+
) from error
|
|
298
|
+
output = "\n".join(
|
|
299
|
+
(
|
|
300
|
+
stdout_path.read_text(encoding="utf-8", errors="replace"),
|
|
301
|
+
stderr_path.read_text(encoding="utf-8", errors="replace"),
|
|
302
|
+
)
|
|
303
|
+
).strip()
|
|
304
|
+
if result.return_code != 0:
|
|
305
|
+
raise AdapterError(
|
|
306
|
+
f"eggNOG-mapper version probe exited with {result.return_code}: {output}"
|
|
307
|
+
)
|
|
308
|
+
return output
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
def _runtime_environment(
|
|
312
|
+
executable: Path,
|
|
313
|
+
*,
|
|
314
|
+
overlay: Mapping[str, str] | None = None,
|
|
315
|
+
) -> dict[str, str]:
|
|
316
|
+
runtime_bin = str(executable.parent)
|
|
317
|
+
environment = dict(overlay or {})
|
|
318
|
+
inherited_path = environment.get("PATH", os.environ.get("PATH"))
|
|
319
|
+
environment["PATH"] = (
|
|
320
|
+
runtime_bin
|
|
321
|
+
if not inherited_path
|
|
322
|
+
else os.pathsep.join((runtime_bin, inherited_path))
|
|
323
|
+
)
|
|
324
|
+
return environment
|
|
325
|
+
|
|
326
|
+
|
|
327
|
+
def _parse_version_output(output: str) -> tuple[str, str, str]:
|
|
328
|
+
tool_versions = sorted(set(_VERSION_PATTERN.findall(output)))
|
|
329
|
+
expected = sorted(set(_EXPECTED_DB_PATTERN.findall(output)))
|
|
330
|
+
installed = sorted(set(_INSTALLED_DB_PATTERN.findall(output)))
|
|
331
|
+
diamond_versions = sorted(set(_DIAMOND_VERSION_PATTERN.findall(output)))
|
|
332
|
+
if len(tool_versions) != 1:
|
|
333
|
+
raise AdapterError(
|
|
334
|
+
"eggnog/1 requires exactly one eggNOG-mapper 2.x release in --version"
|
|
335
|
+
)
|
|
336
|
+
if len(expected) != 1 or len(installed) != 1 or expected != installed:
|
|
337
|
+
raise AdapterError(
|
|
338
|
+
"eggNOG-mapper must report one matching expected and installed DB version"
|
|
339
|
+
)
|
|
340
|
+
if len(diamond_versions) != 1:
|
|
341
|
+
raise AdapterError("eggnog/1 requires exactly one DIAMOND release in --version")
|
|
342
|
+
return tool_versions[0], installed[0], diamond_versions[0]
|
|
343
|
+
|
|
344
|
+
|
|
345
|
+
def _runtime_digest(
|
|
346
|
+
executable: Path,
|
|
347
|
+
*,
|
|
348
|
+
tool_version: str,
|
|
349
|
+
diamond: Path,
|
|
350
|
+
diamond_version: str,
|
|
351
|
+
environment: Mapping[str, str],
|
|
352
|
+
) -> str:
|
|
353
|
+
package_root = _resolve_eggnog_package(executable)
|
|
354
|
+
package_components = tuple(
|
|
355
|
+
RuntimeComponent(
|
|
356
|
+
name=f"eggnogmapper/{path.relative_to(package_root).as_posix()}",
|
|
357
|
+
path=path,
|
|
358
|
+
)
|
|
359
|
+
for path in sorted(package_root.rglob("*"))
|
|
360
|
+
if path.is_file()
|
|
361
|
+
and "__pycache__" not in path.parts
|
|
362
|
+
and path.suffix not in {".pyc", ".pyo"}
|
|
363
|
+
)
|
|
364
|
+
interpreter = _resolve_python_interpreter(executable, environment=environment)
|
|
365
|
+
distribution_records = tuple(
|
|
366
|
+
RuntimeComponent(
|
|
367
|
+
name=f"python-distributions/{path.parent.name}/RECORD",
|
|
368
|
+
path=path,
|
|
369
|
+
)
|
|
370
|
+
for path in sorted(package_root.parent.glob("*.dist-info/RECORD"))
|
|
371
|
+
)
|
|
372
|
+
return calculate_runtime_digest(
|
|
373
|
+
runtime_name="eggnog-mapper",
|
|
374
|
+
versions={"diamond": diamond_version, "eggnog-mapper": tool_version},
|
|
375
|
+
components=(
|
|
376
|
+
RuntimeComponent("launcher", executable),
|
|
377
|
+
RuntimeComponent("python", interpreter),
|
|
378
|
+
RuntimeComponent("diamond", diamond),
|
|
379
|
+
*distribution_records,
|
|
380
|
+
*package_components,
|
|
381
|
+
),
|
|
382
|
+
)
|
|
383
|
+
|
|
384
|
+
|
|
385
|
+
def _resolve_python_interpreter(
|
|
386
|
+
executable: Path, *, environment: Mapping[str, str]
|
|
387
|
+
) -> Path:
|
|
388
|
+
with executable.open("rb") as handle:
|
|
389
|
+
first_line = handle.readline(4096).decode("utf-8", errors="strict")
|
|
390
|
+
if not first_line.startswith("#!"):
|
|
391
|
+
raise AdapterError("eggNOG-mapper launcher has no Python shebang")
|
|
392
|
+
command = shlex.split(first_line[2:].strip())
|
|
393
|
+
if not command:
|
|
394
|
+
raise AdapterError("eggNOG-mapper launcher has an empty shebang")
|
|
395
|
+
if Path(command[0]).name == "env":
|
|
396
|
+
candidates = [item for item in command[1:] if not item.startswith("-")]
|
|
397
|
+
if len(candidates) != 1:
|
|
398
|
+
raise AdapterError("unsupported eggNOG-mapper env shebang")
|
|
399
|
+
resolved = shutil.which(
|
|
400
|
+
candidates[0],
|
|
401
|
+
path=_runtime_environment(executable, overlay=environment)["PATH"],
|
|
402
|
+
)
|
|
403
|
+
else:
|
|
404
|
+
resolved = shutil.which(
|
|
405
|
+
command[0],
|
|
406
|
+
path=_runtime_environment(executable, overlay=environment)["PATH"],
|
|
407
|
+
)
|
|
408
|
+
if resolved is None:
|
|
409
|
+
raise AdapterError("eggNOG-mapper Python interpreter cannot be resolved")
|
|
410
|
+
return Path(resolved).resolve()
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
def _resolve_eggnog_package(executable: Path) -> Path:
|
|
414
|
+
runtime_root = executable.parent.parent
|
|
415
|
+
candidates = []
|
|
416
|
+
for python_dir in sorted((runtime_root / "lib").glob("python*")):
|
|
417
|
+
for package_dir_name in ("site-packages", "dist-packages"):
|
|
418
|
+
candidate = python_dir / package_dir_name / "eggnogmapper"
|
|
419
|
+
if candidate.is_dir():
|
|
420
|
+
candidates.append(candidate.resolve())
|
|
421
|
+
unique = sorted(set(candidates))
|
|
422
|
+
if len(unique) != 1:
|
|
423
|
+
raise AdapterError(
|
|
424
|
+
"eggNOG-mapper runtime must contain exactly one installed "
|
|
425
|
+
"eggnogmapper package directory"
|
|
426
|
+
)
|
|
427
|
+
return unique[0]
|
|
428
|
+
|
|
429
|
+
|
|
430
|
+
def _resolve_diamond(
|
|
431
|
+
executable: Path,
|
|
432
|
+
*,
|
|
433
|
+
environment: Mapping[str, str],
|
|
434
|
+
) -> Path:
|
|
435
|
+
resolved = shutil.which(
|
|
436
|
+
"diamond",
|
|
437
|
+
path=_runtime_environment(executable, overlay=environment)["PATH"],
|
|
438
|
+
)
|
|
439
|
+
if resolved is None:
|
|
440
|
+
raise AdapterError("eggNOG-mapper runtime has no DIAMOND executable")
|
|
441
|
+
return Path(resolved).resolve()
|
|
442
|
+
|
|
443
|
+
|
|
444
|
+
def _probe_diamond_version(
|
|
445
|
+
executable: Path,
|
|
446
|
+
*,
|
|
447
|
+
environment: Mapping[str, str],
|
|
448
|
+
) -> str:
|
|
449
|
+
with tempfile.TemporaryDirectory(prefix="seqevi-diamond-probe-") as raw_dir:
|
|
450
|
+
root = Path(raw_dir)
|
|
451
|
+
stdout_path = root / "stdout.log"
|
|
452
|
+
stderr_path = root / "stderr.log"
|
|
453
|
+
try:
|
|
454
|
+
result = ToolRunner().run(
|
|
455
|
+
ToolCommand(
|
|
456
|
+
arguments=(str(executable), "version"),
|
|
457
|
+
working_dir=executable.parent,
|
|
458
|
+
stdout_path=stdout_path,
|
|
459
|
+
stderr_path=stderr_path,
|
|
460
|
+
environment=environment,
|
|
461
|
+
),
|
|
462
|
+
timeout_seconds=_PROBE_TIMEOUT_SECONDS,
|
|
463
|
+
)
|
|
464
|
+
except (OSError, ToolTimeoutError) as error:
|
|
465
|
+
raise AdapterError(f"DIAMOND version probe failed: {error}") from error
|
|
466
|
+
output = "\n".join(
|
|
467
|
+
(
|
|
468
|
+
stdout_path.read_text(encoding="utf-8", errors="replace"),
|
|
469
|
+
stderr_path.read_text(encoding="utf-8", errors="replace"),
|
|
470
|
+
)
|
|
471
|
+
).strip()
|
|
472
|
+
if result.return_code != 0:
|
|
473
|
+
raise AdapterError(
|
|
474
|
+
f"DIAMOND version probe exited with {result.return_code}: {output}"
|
|
475
|
+
)
|
|
476
|
+
versions = sorted(set(_DIAMOND_VERSION_PATTERN.findall(output)))
|
|
477
|
+
if len(versions) != 1:
|
|
478
|
+
raise AdapterError("DIAMOND executable did not report exactly one version")
|
|
479
|
+
return versions[0]
|
|
480
|
+
|
|
481
|
+
|
|
482
|
+
def _resource_id(database: Path, version: str, *, verify: bool = False) -> str:
|
|
483
|
+
declarations = tuple(
|
|
484
|
+
ResourceComponent(name=name, relative_path=name)
|
|
485
|
+
for name in _REQUIRED_DATABASE_FILES
|
|
486
|
+
) + tuple(
|
|
487
|
+
ResourceComponent(name=name, relative_path=name)
|
|
488
|
+
for name in _OPTIONAL_DATABASE_FILES
|
|
489
|
+
if (database / name).is_file()
|
|
490
|
+
)
|
|
491
|
+
locked = resolve_resource_lock(
|
|
492
|
+
database=database,
|
|
493
|
+
resource_name="eggnog",
|
|
494
|
+
resource_version=version,
|
|
495
|
+
components=declarations,
|
|
496
|
+
verify=verify,
|
|
497
|
+
)
|
|
498
|
+
components = [
|
|
499
|
+
(component.name, locked.hash_for(component.name)) for component in declarations
|
|
500
|
+
]
|
|
501
|
+
digest = sha256_digest(
|
|
502
|
+
json.dumps(components, separators=(",", ":")).encode("utf-8")
|
|
503
|
+
)
|
|
504
|
+
return f"eggnog/{version}/sha256:{digest}"
|
|
505
|
+
|
|
506
|
+
|
|
507
|
+
def _parse_annotations(
|
|
508
|
+
path: Path,
|
|
509
|
+
*,
|
|
510
|
+
identities: tuple[SequenceIdentity, ...],
|
|
511
|
+
normalized_path: Path,
|
|
512
|
+
) -> tuple[ArtifactFile | None, dict[str, str]]:
|
|
513
|
+
expected = {identity.sequence_id: identity for identity in identities}
|
|
514
|
+
header: tuple[str, ...] | None = None
|
|
515
|
+
rows: list[dict[str, object]] = []
|
|
516
|
+
seen_queries: set[str] = set()
|
|
517
|
+
payload_digest_by_id: dict[str, str] = {}
|
|
518
|
+
with tempfile.TemporaryDirectory(
|
|
519
|
+
prefix=".eggnog-normalized-", dir=normalized_path.parent
|
|
520
|
+
) as raw_parts_dir:
|
|
521
|
+
parts_dir = Path(raw_parts_dir)
|
|
522
|
+
part_paths: list[Path] = []
|
|
523
|
+
try:
|
|
524
|
+
with path.open("r", encoding="utf-8", newline="") as handle:
|
|
525
|
+
for line_number, raw_line in enumerate(handle, start=1):
|
|
526
|
+
line = raw_line.removesuffix("\n").removesuffix("\r")
|
|
527
|
+
if not line:
|
|
528
|
+
raise AdapterError(
|
|
529
|
+
f"eggNOG annotations contain a blank line at {line_number}"
|
|
530
|
+
)
|
|
531
|
+
if line.startswith("##"):
|
|
532
|
+
continue
|
|
533
|
+
if line.startswith("#"):
|
|
534
|
+
candidate = tuple(line.removeprefix("#").split("\t"))
|
|
535
|
+
if candidate[0] == "query":
|
|
536
|
+
if header is not None:
|
|
537
|
+
raise AdapterError(
|
|
538
|
+
"eggNOG annotations contain duplicate headers"
|
|
539
|
+
)
|
|
540
|
+
header = candidate
|
|
541
|
+
continue
|
|
542
|
+
if header is None:
|
|
543
|
+
raise AdapterError(
|
|
544
|
+
"eggNOG annotations data appeared before its header"
|
|
545
|
+
)
|
|
546
|
+
fields = next(csv.reader((line,), delimiter="\t"))
|
|
547
|
+
if len(fields) != len(header):
|
|
548
|
+
raise AdapterError(
|
|
549
|
+
f"eggNOG annotations line {line_number} has "
|
|
550
|
+
f"{len(fields)} columns; expected {len(header)}"
|
|
551
|
+
)
|
|
552
|
+
if header != _NATIVE_COLUMNS:
|
|
553
|
+
raise AdapterError(
|
|
554
|
+
"eggnog/1 requires the canonical eggNOG-mapper 2.x "
|
|
555
|
+
"annotations schema"
|
|
556
|
+
)
|
|
557
|
+
row = _parse_row(
|
|
558
|
+
header, fields, expected=expected, line_number=line_number
|
|
559
|
+
)
|
|
560
|
+
query = str(row["SequenceID"])
|
|
561
|
+
if query in seen_queries:
|
|
562
|
+
raise AdapterError(
|
|
563
|
+
f"eggNOG annotations contain duplicate query: {query}"
|
|
564
|
+
)
|
|
565
|
+
seen_queries.add(query)
|
|
566
|
+
payload_digest_by_id[query] = sha256_digest(
|
|
567
|
+
json.dumps(
|
|
568
|
+
row,
|
|
569
|
+
allow_nan=False,
|
|
570
|
+
sort_keys=True,
|
|
571
|
+
separators=(",", ":"),
|
|
572
|
+
).encode("utf-8")
|
|
573
|
+
)
|
|
574
|
+
rows.append(row)
|
|
575
|
+
if len(rows) >= _NORMALIZED_ROW_BATCH_SIZE:
|
|
576
|
+
part_paths.append(_write_normalized_part(rows, parts_dir))
|
|
577
|
+
rows.clear()
|
|
578
|
+
except UnicodeDecodeError as error:
|
|
579
|
+
raise AdapterError(
|
|
580
|
+
f"eggNOG annotations are not valid UTF-8: {error}"
|
|
581
|
+
) from error
|
|
582
|
+
|
|
583
|
+
if header is None:
|
|
584
|
+
raise AdapterError("eggNOG annotations are missing the #query header")
|
|
585
|
+
if header != _NATIVE_COLUMNS:
|
|
586
|
+
raise AdapterError(
|
|
587
|
+
"eggnog/1 requires the canonical eggNOG-mapper 2.x annotations schema"
|
|
588
|
+
)
|
|
589
|
+
if rows:
|
|
590
|
+
part_paths.append(_write_normalized_part(rows, parts_dir))
|
|
591
|
+
if not part_paths:
|
|
592
|
+
return None, payload_digest_by_id
|
|
593
|
+
pl.concat([pl.scan_parquet(part) for part in part_paths]).sort(
|
|
594
|
+
"SequenceID"
|
|
595
|
+
).sink_parquet(normalized_path, compression="zstd", maintain_order=True)
|
|
596
|
+
return (
|
|
597
|
+
ArtifactFile.from_path(normalized_path, "application/vnd.apache.parquet"),
|
|
598
|
+
payload_digest_by_id,
|
|
599
|
+
)
|
|
600
|
+
|
|
601
|
+
|
|
602
|
+
def _write_normalized_part(rows: list[dict[str, object]], directory: Path) -> Path:
|
|
603
|
+
path = directory / f"part-{len(tuple(directory.iterdir())):06d}.parquet"
|
|
604
|
+
pl.DataFrame(rows, schema=EGGNOG_EVIDENCE_SCHEMA).write_parquet(path)
|
|
605
|
+
return path
|
|
606
|
+
|
|
607
|
+
|
|
608
|
+
def _parse_row(
|
|
609
|
+
header: tuple[str, ...],
|
|
610
|
+
fields: list[str],
|
|
611
|
+
*,
|
|
612
|
+
expected: Mapping[str, SequenceIdentity],
|
|
613
|
+
line_number: int,
|
|
614
|
+
) -> dict[str, object]:
|
|
615
|
+
native = dict(zip(header, fields, strict=True))
|
|
616
|
+
query = native["query"]
|
|
617
|
+
if query not in expected:
|
|
618
|
+
raise AdapterError(
|
|
619
|
+
f"eggNOG annotations line {line_number} has unknown SequenceID: {query}"
|
|
620
|
+
)
|
|
621
|
+
row: dict[str, object] = {"SequenceID": query}
|
|
622
|
+
for column in _NATIVE_COLUMNS:
|
|
623
|
+
value = native[column]
|
|
624
|
+
if column in {"evalue", "score"}:
|
|
625
|
+
try:
|
|
626
|
+
parsed = float(value)
|
|
627
|
+
except ValueError as error:
|
|
628
|
+
raise AdapterError(
|
|
629
|
+
f"eggNOG annotations line {line_number} has invalid {column}: {value}"
|
|
630
|
+
) from error
|
|
631
|
+
if not math.isfinite(parsed):
|
|
632
|
+
raise AdapterError(
|
|
633
|
+
f"eggNOG annotations line {line_number} has non-finite {column}"
|
|
634
|
+
)
|
|
635
|
+
row[column] = parsed
|
|
636
|
+
else:
|
|
637
|
+
row[column] = None if value == "-" else value
|
|
638
|
+
return row
|
|
639
|
+
|
|
640
|
+
|
|
641
|
+
def _gzip_annotations_artifact(source: Path, target: Path) -> ArtifactFile:
|
|
642
|
+
with (
|
|
643
|
+
source.open("rb") as source_handle,
|
|
644
|
+
target.open("wb") as target_handle,
|
|
645
|
+
gzip.GzipFile(fileobj=target_handle, mode="wb", mtime=0) as compressed,
|
|
646
|
+
):
|
|
647
|
+
for line in source_handle:
|
|
648
|
+
if line.startswith(b"#") and not line.startswith(b"#query\t"):
|
|
649
|
+
continue
|
|
650
|
+
compressed.write(line)
|
|
651
|
+
return ArtifactFile.from_path(target, "application/gzip")
|
|
652
|
+
|
|
653
|
+
|
|
654
|
+
def _sequence_result(
|
|
655
|
+
identity: SequenceIdentity,
|
|
656
|
+
*,
|
|
657
|
+
payload_digest: str | None,
|
|
658
|
+
) -> AdapterSequenceResult:
|
|
659
|
+
if payload_digest is None:
|
|
660
|
+
payload = json.dumps(
|
|
661
|
+
{"SequenceID": identity.sequence_id, "Status": "no_hit"},
|
|
662
|
+
sort_keys=True,
|
|
663
|
+
separators=(",", ":"),
|
|
664
|
+
).encode("utf-8")
|
|
665
|
+
return AdapterSequenceResult(
|
|
666
|
+
sequence_id=identity.sequence_id,
|
|
667
|
+
status=EvidenceStatus.NO_HIT,
|
|
668
|
+
payload_digest=sha256_digest(payload),
|
|
669
|
+
)
|
|
670
|
+
return AdapterSequenceResult(
|
|
671
|
+
sequence_id=identity.sequence_id,
|
|
672
|
+
status=EvidenceStatus.HIT,
|
|
673
|
+
payload_digest=payload_digest,
|
|
674
|
+
)
|