seqevi 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- seqevi/__init__.py +7 -0
- seqevi/__main__.py +6 -0
- seqevi/adapters/__init__.py +41 -0
- seqevi/adapters/base.py +131 -0
- seqevi/adapters/dbcan_cazyme.py +588 -0
- seqevi/adapters/eggnog.py +674 -0
- seqevi/adapters/interpro_pfam.py +757 -0
- seqevi/adapters/registry.py +68 -0
- seqevi/annotate.py +413 -0
- seqevi/api.py +390 -0
- seqevi/cli.py +610 -0
- seqevi/distribution/__init__.py +13 -0
- seqevi/distribution/manifest.py +199 -0
- seqevi/distribution/oci.py +490 -0
- seqevi/distribution/setup.py +752 -0
- seqevi/errors.py +73 -0
- seqevi/evidence.py +295 -0
- seqevi/execution_profile.py +526 -0
- seqevi/hashing.py +13 -0
- seqevi/kits/__init__.py +1 -0
- seqevi/kits/dbcan-cazyme.toml +35 -0
- seqevi/resource_lock.py +438 -0
- seqevi/result.py +682 -0
- seqevi/runner.py +163 -0
- seqevi/runtime_identity.py +104 -0
- seqevi/sequence.py +383 -0
- seqevi/service/__init__.py +11 -0
- seqevi/service/app.py +213 -0
- seqevi/service/config.py +38 -0
- seqevi/service/persistence.py +360 -0
- seqevi/store/__init__.py +14 -0
- seqevi/store/artifact.py +225 -0
- seqevi/store/client.py +311 -0
- seqevi/store/contract.py +33 -0
- seqevi/store/factory.py +38 -0
- seqevi/store/local.py +479 -0
- seqevi/store/migration.py +62 -0
- seqevi/store/migrations/__init__.py +1 -0
- seqevi/store/migrations/env.py +30 -0
- seqevi/store/migrations/versions/0001_initial_store.py +103 -0
- seqevi/store/migrations/versions/0002_artifact_byte_size_bigint.py +40 -0
- seqevi/store/migrations/versions/__init__.py +1 -0
- seqevi/store/schema.py +86 -0
- seqevi/store/transport.py +224 -0
- seqevi-0.2.0.dist-info/METADATA +333 -0
- seqevi-0.2.0.dist-info/RECORD +49 -0
- seqevi-0.2.0.dist-info/WHEEL +4 -0
- seqevi-0.2.0.dist-info/entry_points.txt +5 -0
- seqevi-0.2.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,588 @@
|
|
|
1
|
+
"""Direct dbCAN v5 protein CAZyme annotation adapter."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import csv
|
|
6
|
+
import gzip
|
|
7
|
+
import json
|
|
8
|
+
import os
|
|
9
|
+
import re
|
|
10
|
+
import shlex
|
|
11
|
+
import shutil
|
|
12
|
+
import tempfile
|
|
13
|
+
from collections.abc import Mapping
|
|
14
|
+
from dataclasses import asdict, dataclass
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
import polars as pl
|
|
18
|
+
|
|
19
|
+
from seqevi.errors import AdapterError
|
|
20
|
+
from seqevi.evidence import ArtifactFile, EvidenceStatus, sha256_digest
|
|
21
|
+
from seqevi.resource_lock import ResourceComponent, resolve_resource_lock
|
|
22
|
+
from seqevi.runner import ToolCommand, ToolRunner, ToolTimeoutError
|
|
23
|
+
from seqevi.runtime_identity import RuntimeComponent, calculate_runtime_digest
|
|
24
|
+
from seqevi.sequence import SequenceIdentity
|
|
25
|
+
|
|
26
|
+
from .base import AdapterBatchResult, AdapterContract, AdapterSequenceResult
|
|
27
|
+
|
|
28
|
+
ADAPTER_CONTRACT_VERSION = "dbcan-cazyme/1"
|
|
29
|
+
RESOURCE_RELEASE = "db_v5-2-9_5-5-2026"
|
|
30
|
+
|
|
31
|
+
_REQUIRED_RESOURCE_COMPONENTS = (
|
|
32
|
+
ResourceComponent("CAZy-diamond", "CAZy.dmnd"),
|
|
33
|
+
ResourceComponent("dbCAN-HMM", "dbCAN.hmm"),
|
|
34
|
+
ResourceComponent("dbCAN-sub-HMM", "dbCAN-sub.hmm"),
|
|
35
|
+
ResourceComponent("fam-substrate-mapping", "fam-substrate-mapping.tsv"),
|
|
36
|
+
)
|
|
37
|
+
_OVERVIEW_COLUMNS = (
|
|
38
|
+
"Gene ID",
|
|
39
|
+
"EC#",
|
|
40
|
+
"dbCAN_hmm",
|
|
41
|
+
"dbCAN_sub",
|
|
42
|
+
"DIAMOND",
|
|
43
|
+
"#ofTools",
|
|
44
|
+
"Recommend Results",
|
|
45
|
+
"Substrate",
|
|
46
|
+
)
|
|
47
|
+
DBCAN_EVIDENCE_SCHEMA: Mapping[str, pl.DataType] = {
|
|
48
|
+
"SequenceID": pl.String(),
|
|
49
|
+
"Gene ID": pl.String(),
|
|
50
|
+
"EC#": pl.String(),
|
|
51
|
+
"dbCAN_hmm": pl.String(),
|
|
52
|
+
"dbCAN_sub": pl.String(),
|
|
53
|
+
"DIAMOND": pl.String(),
|
|
54
|
+
"#ofTools": pl.Int64(),
|
|
55
|
+
"Recommend Results": pl.String(),
|
|
56
|
+
"Substrate": pl.String(),
|
|
57
|
+
}
|
|
58
|
+
_VERSION_PATTERN = (
|
|
59
|
+
r"(?:dbcan|run_dbcan)\s*(?:version\s*)?:?\s*"
|
|
60
|
+
r"v?([0-9]+\.[0-9]+\.[0-9]+)"
|
|
61
|
+
)
|
|
62
|
+
_DIAMOND_VERSION_PATTERN = r"\bdiamond version\s+([^\s]+)"
|
|
63
|
+
_PROBE_TIMEOUT_SECONDS = 120.0
|
|
64
|
+
_NORMALIZED_ROW_BATCH_SIZE = 1000
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
@dataclass(frozen=True, slots=True)
|
|
68
|
+
class DBCanParameters:
|
|
69
|
+
"""Fixed scientific parameters for the dbCAN protein contract."""
|
|
70
|
+
|
|
71
|
+
mode: str = "protein"
|
|
72
|
+
methods: str = "diamond,hmm,dbCANsub"
|
|
73
|
+
diamond_evalue: float = 1e-102
|
|
74
|
+
dbcan_coverage: float = 0.35
|
|
75
|
+
dbcan_evalue: float = 1e-15
|
|
76
|
+
dbcan_sub_coverage: float = 0.35
|
|
77
|
+
dbcan_sub_evalue: float = 1e-15
|
|
78
|
+
|
|
79
|
+
def __post_init__(self) -> None:
|
|
80
|
+
values = tuple(asdict(self).values())
|
|
81
|
+
if values != (
|
|
82
|
+
"protein",
|
|
83
|
+
"diamond,hmm,dbCANsub",
|
|
84
|
+
1e-102,
|
|
85
|
+
0.35,
|
|
86
|
+
1e-15,
|
|
87
|
+
0.35,
|
|
88
|
+
1e-15,
|
|
89
|
+
):
|
|
90
|
+
raise ValueError("dbcan-cazyme/1 uses one fixed protein CAZyme contract")
|
|
91
|
+
|
|
92
|
+
def as_semantic_parameters(self) -> dict[str, object]:
|
|
93
|
+
"""Return every result-affecting dbCAN parameter."""
|
|
94
|
+
|
|
95
|
+
return asdict(self)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
class DBCanCazymeAdapter:
|
|
99
|
+
"""Run ``run_dbcan CAZyme_annotation --mode protein`` and validate overview."""
|
|
100
|
+
|
|
101
|
+
def __init__(
|
|
102
|
+
self,
|
|
103
|
+
*,
|
|
104
|
+
executable: Path,
|
|
105
|
+
database: Path,
|
|
106
|
+
parameters: DBCanParameters | None = None,
|
|
107
|
+
verify_resource: bool = False,
|
|
108
|
+
environment: Mapping[str, str] | None = None,
|
|
109
|
+
) -> None:
|
|
110
|
+
self.executable = executable.resolve()
|
|
111
|
+
self.database = database.resolve()
|
|
112
|
+
self.parameters = parameters or DBCanParameters()
|
|
113
|
+
self.environment = dict(environment or {})
|
|
114
|
+
self._validate_installation()
|
|
115
|
+
self.dbcan_version = _probe_dbcan_version(
|
|
116
|
+
self.executable, environment=self.environment
|
|
117
|
+
)
|
|
118
|
+
diamond = _resolve_diamond(self.executable, environment=self.environment)
|
|
119
|
+
self.diamond_version = _probe_diamond_version(
|
|
120
|
+
diamond, environment=self.environment
|
|
121
|
+
)
|
|
122
|
+
runtime_digest = _runtime_digest(
|
|
123
|
+
self.executable,
|
|
124
|
+
dbcan_version=self.dbcan_version,
|
|
125
|
+
diamond=diamond,
|
|
126
|
+
diamond_version=self.diamond_version,
|
|
127
|
+
environment=self.environment,
|
|
128
|
+
)
|
|
129
|
+
resource_id = _resource_id(
|
|
130
|
+
self.database,
|
|
131
|
+
verify=verify_resource,
|
|
132
|
+
)
|
|
133
|
+
self._contract = AdapterContract.from_parameters(
|
|
134
|
+
name="dbcan-cazyme",
|
|
135
|
+
version=ADAPTER_CONTRACT_VERSION,
|
|
136
|
+
tool_runtime_digest=f"sha256:{runtime_digest}",
|
|
137
|
+
resource_id=resource_id,
|
|
138
|
+
semantic_parameters=self.parameters.as_semantic_parameters(),
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
@property
|
|
142
|
+
def contract(self) -> AdapterContract:
|
|
143
|
+
return self._contract
|
|
144
|
+
|
|
145
|
+
@property
|
|
146
|
+
def evidence_schema(self) -> Mapping[str, pl.DataType]:
|
|
147
|
+
return DBCAN_EVIDENCE_SCHEMA
|
|
148
|
+
|
|
149
|
+
def run_batch(
|
|
150
|
+
self,
|
|
151
|
+
*,
|
|
152
|
+
identities: tuple[SequenceIdentity, ...],
|
|
153
|
+
input_fasta: Path,
|
|
154
|
+
work_dir: Path,
|
|
155
|
+
runner: ToolRunner,
|
|
156
|
+
timeout_seconds: float | None,
|
|
157
|
+
threads: int,
|
|
158
|
+
) -> AdapterBatchResult:
|
|
159
|
+
"""Run one cache-miss batch and normalize its protein overview."""
|
|
160
|
+
|
|
161
|
+
if not identities:
|
|
162
|
+
raise AdapterError("dbcan-cazyme batch must not be empty")
|
|
163
|
+
output_dir = work_dir / "dbcan-output"
|
|
164
|
+
output_dir.mkdir()
|
|
165
|
+
parameters = self.parameters
|
|
166
|
+
result = runner.run(
|
|
167
|
+
ToolCommand(
|
|
168
|
+
arguments=(
|
|
169
|
+
str(self.executable),
|
|
170
|
+
"CAZyme_annotation",
|
|
171
|
+
"--input_raw_data",
|
|
172
|
+
str(input_fasta),
|
|
173
|
+
"--mode",
|
|
174
|
+
parameters.mode,
|
|
175
|
+
"--output_dir",
|
|
176
|
+
str(output_dir),
|
|
177
|
+
"--db_dir",
|
|
178
|
+
str(self.database),
|
|
179
|
+
"--methods",
|
|
180
|
+
parameters.methods,
|
|
181
|
+
"--threads",
|
|
182
|
+
str(threads),
|
|
183
|
+
"--e_value_threshold",
|
|
184
|
+
str(parameters.diamond_evalue),
|
|
185
|
+
"--coverage_threshold_dbcan",
|
|
186
|
+
str(parameters.dbcan_coverage),
|
|
187
|
+
"--e_value_threshold_dbcan",
|
|
188
|
+
str(parameters.dbcan_evalue),
|
|
189
|
+
"--coverage_threshold_dbsub",
|
|
190
|
+
str(parameters.dbcan_sub_coverage),
|
|
191
|
+
"--e_value_threshold_dbsub",
|
|
192
|
+
str(parameters.dbcan_sub_evalue),
|
|
193
|
+
),
|
|
194
|
+
working_dir=work_dir,
|
|
195
|
+
stdout_path=work_dir / "dbcan.stdout.log",
|
|
196
|
+
stderr_path=work_dir / "dbcan.stderr.log",
|
|
197
|
+
environment=_runtime_environment(
|
|
198
|
+
self.executable, overlay=self.environment
|
|
199
|
+
),
|
|
200
|
+
),
|
|
201
|
+
timeout_seconds=timeout_seconds,
|
|
202
|
+
)
|
|
203
|
+
if result.return_code != 0:
|
|
204
|
+
raise AdapterError(
|
|
205
|
+
f"dbCAN exited with {result.return_code}; stderr: {result.stderr_path}"
|
|
206
|
+
)
|
|
207
|
+
overview = output_dir / "overview.tsv"
|
|
208
|
+
if not overview.is_file():
|
|
209
|
+
raise AdapterError("dbCAN did not create overview.tsv")
|
|
210
|
+
normalized, payloads = _parse_overview(
|
|
211
|
+
overview,
|
|
212
|
+
identities=identities,
|
|
213
|
+
normalized_path=work_dir / "dbcan.normalized.parquet",
|
|
214
|
+
)
|
|
215
|
+
sequence_results = tuple(
|
|
216
|
+
_sequence_result(
|
|
217
|
+
identity,
|
|
218
|
+
payload_digest=payloads.get(identity.sequence_id),
|
|
219
|
+
)
|
|
220
|
+
for identity in sorted(identities, key=lambda item: item.sequence_id)
|
|
221
|
+
)
|
|
222
|
+
return AdapterBatchResult(
|
|
223
|
+
sequences=sequence_results,
|
|
224
|
+
raw_artifact=_gzip_artifact(
|
|
225
|
+
overview,
|
|
226
|
+
work_dir / "dbcan.overview.tsv.gz",
|
|
227
|
+
),
|
|
228
|
+
normalized_artifact=normalized,
|
|
229
|
+
)
|
|
230
|
+
|
|
231
|
+
def _validate_installation(self) -> None:
|
|
232
|
+
if not self.executable.is_file() or not os.access(self.executable, os.X_OK):
|
|
233
|
+
raise AdapterError(
|
|
234
|
+
f"dbCAN executable is not an executable file: {self.executable}"
|
|
235
|
+
)
|
|
236
|
+
if not self.database.is_dir():
|
|
237
|
+
raise AdapterError(f"dbCAN resource is not a directory: {self.database}")
|
|
238
|
+
for component in _REQUIRED_RESOURCE_COMPONENTS:
|
|
239
|
+
path = self.database / component.relative_path
|
|
240
|
+
if not path.is_file():
|
|
241
|
+
raise AdapterError(
|
|
242
|
+
f"dbCAN resource is missing {component.relative_path}: {path}"
|
|
243
|
+
)
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def _runtime_environment(
|
|
247
|
+
executable: Path,
|
|
248
|
+
*,
|
|
249
|
+
overlay: Mapping[str, str] | None = None,
|
|
250
|
+
) -> dict[str, str]:
|
|
251
|
+
environment = dict(overlay or {})
|
|
252
|
+
inherited = environment.get("PATH", os.environ.get("PATH", ""))
|
|
253
|
+
environment["PATH"] = (
|
|
254
|
+
str(executable.parent)
|
|
255
|
+
if not inherited
|
|
256
|
+
else os.pathsep.join((str(executable.parent), inherited))
|
|
257
|
+
)
|
|
258
|
+
return environment
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def _probe_dbcan_version(
|
|
262
|
+
executable: Path,
|
|
263
|
+
*,
|
|
264
|
+
environment: Mapping[str, str],
|
|
265
|
+
) -> str:
|
|
266
|
+
output = _probe_command(executable, ("version",), environment=environment)
|
|
267
|
+
versions = sorted(set(re.findall(_VERSION_PATTERN, output, flags=re.IGNORECASE)))
|
|
268
|
+
if len(versions) != 1:
|
|
269
|
+
raise AdapterError(
|
|
270
|
+
"dbcan-cazyme/1 requires exactly one dbCAN 5.2.9 version in "
|
|
271
|
+
"run_dbcan version output"
|
|
272
|
+
)
|
|
273
|
+
if versions[0] != "5.2.9":
|
|
274
|
+
raise AdapterError(f"unsupported dbCAN release: {versions[0]}")
|
|
275
|
+
return versions[0]
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
def _probe_command(
|
|
279
|
+
executable: Path,
|
|
280
|
+
arguments: tuple[str, ...],
|
|
281
|
+
*,
|
|
282
|
+
environment: Mapping[str, str],
|
|
283
|
+
) -> str:
|
|
284
|
+
with tempfile.TemporaryDirectory(prefix="seqevi-dbcan-probe-") as raw_dir:
|
|
285
|
+
root = Path(raw_dir)
|
|
286
|
+
stdout_path = root / "stdout.log"
|
|
287
|
+
stderr_path = root / "stderr.log"
|
|
288
|
+
try:
|
|
289
|
+
result = ToolRunner().run(
|
|
290
|
+
ToolCommand(
|
|
291
|
+
arguments=(str(executable), *arguments),
|
|
292
|
+
working_dir=executable.parent,
|
|
293
|
+
stdout_path=stdout_path,
|
|
294
|
+
stderr_path=stderr_path,
|
|
295
|
+
environment=_runtime_environment(executable, overlay=environment),
|
|
296
|
+
),
|
|
297
|
+
timeout_seconds=_PROBE_TIMEOUT_SECONDS,
|
|
298
|
+
)
|
|
299
|
+
except (OSError, ToolTimeoutError) as error:
|
|
300
|
+
raise AdapterError(f"dbCAN version probe failed: {error}") from error
|
|
301
|
+
output = "\n".join(
|
|
302
|
+
(
|
|
303
|
+
stdout_path.read_text(encoding="utf-8", errors="replace"),
|
|
304
|
+
stderr_path.read_text(encoding="utf-8", errors="replace"),
|
|
305
|
+
)
|
|
306
|
+
)
|
|
307
|
+
if result.return_code != 0:
|
|
308
|
+
raise AdapterError(
|
|
309
|
+
f"dbCAN version probe exited with {result.return_code}: "
|
|
310
|
+
f"{output.strip()}"
|
|
311
|
+
)
|
|
312
|
+
return output
|
|
313
|
+
|
|
314
|
+
|
|
315
|
+
def _resolve_diamond(
|
|
316
|
+
executable: Path,
|
|
317
|
+
*,
|
|
318
|
+
environment: Mapping[str, str],
|
|
319
|
+
) -> Path:
|
|
320
|
+
resolved = shutil.which(
|
|
321
|
+
"diamond",
|
|
322
|
+
path=_runtime_environment(executable, overlay=environment)["PATH"],
|
|
323
|
+
)
|
|
324
|
+
if resolved is None:
|
|
325
|
+
raise AdapterError("dbCAN runtime has no DIAMOND executable")
|
|
326
|
+
return Path(resolved).resolve()
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
def _probe_diamond_version(
|
|
330
|
+
executable: Path,
|
|
331
|
+
*,
|
|
332
|
+
environment: Mapping[str, str],
|
|
333
|
+
) -> str:
|
|
334
|
+
output = _probe_command(executable, ("version",), environment=environment)
|
|
335
|
+
versions = sorted(
|
|
336
|
+
set(re.findall(_DIAMOND_VERSION_PATTERN, output, flags=re.IGNORECASE))
|
|
337
|
+
)
|
|
338
|
+
if len(versions) != 1:
|
|
339
|
+
raise AdapterError("dbcan-cazyme/1 requires exactly one DIAMOND release")
|
|
340
|
+
if versions[0] != "2.1.15":
|
|
341
|
+
raise AdapterError(f"unsupported DIAMOND release: {versions[0]}")
|
|
342
|
+
return versions[0]
|
|
343
|
+
|
|
344
|
+
|
|
345
|
+
def _runtime_digest(
|
|
346
|
+
executable: Path,
|
|
347
|
+
*,
|
|
348
|
+
dbcan_version: str,
|
|
349
|
+
diamond: Path,
|
|
350
|
+
diamond_version: str,
|
|
351
|
+
environment: Mapping[str, str],
|
|
352
|
+
) -> str:
|
|
353
|
+
components = [
|
|
354
|
+
RuntimeComponent("launcher", executable),
|
|
355
|
+
RuntimeComponent("diamond", diamond),
|
|
356
|
+
]
|
|
357
|
+
interpreter = _resolve_python_interpreter(executable, environment=environment)
|
|
358
|
+
components.append(RuntimeComponent("python", interpreter))
|
|
359
|
+
package_root = _resolve_dbcan_package(executable)
|
|
360
|
+
components.extend(
|
|
361
|
+
RuntimeComponent(f"dbcan/{path.relative_to(package_root).as_posix()}", path)
|
|
362
|
+
for path in sorted(package_root.rglob("*"))
|
|
363
|
+
if path.is_file() and "__pycache__" not in path.parts
|
|
364
|
+
)
|
|
365
|
+
components.extend(
|
|
366
|
+
RuntimeComponent(f"python-distributions/{path.parent.name}/RECORD", path)
|
|
367
|
+
for path in sorted(package_root.parent.glob("*.dist-info/RECORD"))
|
|
368
|
+
)
|
|
369
|
+
return calculate_runtime_digest(
|
|
370
|
+
runtime_name="dbcan-cazyme",
|
|
371
|
+
versions={"dbcan": dbcan_version, "diamond": diamond_version},
|
|
372
|
+
components=tuple(components),
|
|
373
|
+
)
|
|
374
|
+
|
|
375
|
+
|
|
376
|
+
def _resolve_python_interpreter(
|
|
377
|
+
executable: Path,
|
|
378
|
+
*,
|
|
379
|
+
environment: Mapping[str, str],
|
|
380
|
+
) -> Path:
|
|
381
|
+
try:
|
|
382
|
+
with executable.open("rb") as handle:
|
|
383
|
+
first_line = handle.readline(4096).decode("utf-8")
|
|
384
|
+
except (OSError, UnicodeDecodeError) as error:
|
|
385
|
+
raise AdapterError(f"dbCAN launcher cannot be read: {error}") from error
|
|
386
|
+
if not first_line.startswith("#!"):
|
|
387
|
+
raise AdapterError("dbCAN launcher has no Python shebang")
|
|
388
|
+
command = shlex.split(first_line[2:].strip())
|
|
389
|
+
if not command:
|
|
390
|
+
raise AdapterError("dbCAN launcher has an empty shebang")
|
|
391
|
+
name = command[0]
|
|
392
|
+
if Path(name).name == "env":
|
|
393
|
+
candidates = [item for item in command[1:] if not item.startswith("-")]
|
|
394
|
+
if len(candidates) != 1:
|
|
395
|
+
raise AdapterError("unsupported dbCAN env shebang")
|
|
396
|
+
name = candidates[0]
|
|
397
|
+
resolved = shutil.which(
|
|
398
|
+
name,
|
|
399
|
+
path=_runtime_environment(executable, overlay=environment)["PATH"],
|
|
400
|
+
)
|
|
401
|
+
if resolved is None:
|
|
402
|
+
raise AdapterError("dbCAN Python interpreter cannot be resolved")
|
|
403
|
+
return Path(resolved).resolve()
|
|
404
|
+
|
|
405
|
+
|
|
406
|
+
def _resolve_dbcan_package(executable: Path) -> Path:
|
|
407
|
+
runtime_root = executable.parent.parent
|
|
408
|
+
candidates = []
|
|
409
|
+
for python_dir in sorted((runtime_root / "lib").glob("python*")):
|
|
410
|
+
for package_dir_name in ("site-packages", "dist-packages"):
|
|
411
|
+
candidate = python_dir / package_dir_name / "dbcan"
|
|
412
|
+
if candidate.is_dir():
|
|
413
|
+
candidates.append(candidate.resolve())
|
|
414
|
+
unique = sorted(set(candidates))
|
|
415
|
+
if len(unique) != 1:
|
|
416
|
+
raise AdapterError(
|
|
417
|
+
"dbCAN runtime must contain exactly one installed dbcan package"
|
|
418
|
+
)
|
|
419
|
+
return unique[0]
|
|
420
|
+
|
|
421
|
+
|
|
422
|
+
def _resource_id(database: Path, *, verify: bool) -> str:
|
|
423
|
+
lock = resolve_resource_lock(
|
|
424
|
+
database=database,
|
|
425
|
+
resource_name="dbcan",
|
|
426
|
+
resource_version=RESOURCE_RELEASE,
|
|
427
|
+
components=_REQUIRED_RESOURCE_COMPONENTS,
|
|
428
|
+
verify=verify,
|
|
429
|
+
)
|
|
430
|
+
values = [
|
|
431
|
+
(component.name, lock.hash_for(component.name))
|
|
432
|
+
for component in _REQUIRED_RESOURCE_COMPONENTS
|
|
433
|
+
]
|
|
434
|
+
digest = sha256_digest(
|
|
435
|
+
json.dumps(values, separators=(",", ":"), ensure_ascii=True).encode()
|
|
436
|
+
)
|
|
437
|
+
return f"dbcan/{RESOURCE_RELEASE}/sha256:{digest}"
|
|
438
|
+
|
|
439
|
+
|
|
440
|
+
def _parse_overview(
|
|
441
|
+
path: Path,
|
|
442
|
+
*,
|
|
443
|
+
identities: tuple[SequenceIdentity, ...],
|
|
444
|
+
normalized_path: Path,
|
|
445
|
+
) -> tuple[ArtifactFile | None, dict[str, str]]:
|
|
446
|
+
expected = {identity.sequence_id: identity for identity in identities}
|
|
447
|
+
rows: list[dict[str, object]] = []
|
|
448
|
+
payloads: dict[str, str] = {}
|
|
449
|
+
seen: set[str] = set()
|
|
450
|
+
with tempfile.TemporaryDirectory(
|
|
451
|
+
prefix=".dbcan-normalized-", dir=normalized_path.parent
|
|
452
|
+
) as raw_parts_dir:
|
|
453
|
+
parts_dir = Path(raw_parts_dir)
|
|
454
|
+
part_paths: list[Path] = []
|
|
455
|
+
try:
|
|
456
|
+
with path.open("r", encoding="utf-8", newline="") as handle:
|
|
457
|
+
reader = csv.reader(handle, delimiter="\t")
|
|
458
|
+
try:
|
|
459
|
+
header = tuple(next(reader))
|
|
460
|
+
except StopIteration as error:
|
|
461
|
+
raise AdapterError("dbCAN overview.tsv is empty") from error
|
|
462
|
+
if header != _OVERVIEW_COLUMNS:
|
|
463
|
+
raise AdapterError(
|
|
464
|
+
"dbCAN overview.tsv has an unexpected header; expected "
|
|
465
|
+
+ "\\t".join(_OVERVIEW_COLUMNS)
|
|
466
|
+
)
|
|
467
|
+
for line_number, fields in enumerate(reader, start=2):
|
|
468
|
+
if not fields or all(not field.strip() for field in fields):
|
|
469
|
+
raise AdapterError(
|
|
470
|
+
f"dbCAN overview line {line_number} is blank"
|
|
471
|
+
)
|
|
472
|
+
if len(fields) != len(_OVERVIEW_COLUMNS):
|
|
473
|
+
raise AdapterError(
|
|
474
|
+
f"dbCAN overview line {line_number} has {len(fields)} "
|
|
475
|
+
f"columns; expected {len(_OVERVIEW_COLUMNS)}"
|
|
476
|
+
)
|
|
477
|
+
row = _parse_overview_row(
|
|
478
|
+
fields, expected=expected, line_number=line_number
|
|
479
|
+
)
|
|
480
|
+
sequence_id = str(row["SequenceID"])
|
|
481
|
+
if sequence_id in seen:
|
|
482
|
+
raise AdapterError(
|
|
483
|
+
f"dbCAN overview contains duplicate Gene ID: {sequence_id}"
|
|
484
|
+
)
|
|
485
|
+
seen.add(sequence_id)
|
|
486
|
+
payloads[sequence_id] = sha256_digest(
|
|
487
|
+
json.dumps(
|
|
488
|
+
row,
|
|
489
|
+
ensure_ascii=True,
|
|
490
|
+
sort_keys=True,
|
|
491
|
+
separators=(",", ":"),
|
|
492
|
+
).encode()
|
|
493
|
+
)
|
|
494
|
+
rows.append(row)
|
|
495
|
+
if len(rows) >= _NORMALIZED_ROW_BATCH_SIZE:
|
|
496
|
+
part_paths.append(_write_normalized_part(rows, parts_dir))
|
|
497
|
+
rows.clear()
|
|
498
|
+
except UnicodeDecodeError as error:
|
|
499
|
+
raise AdapterError(f"dbCAN overview is not valid UTF-8: {error}") from error
|
|
500
|
+
if rows:
|
|
501
|
+
part_paths.append(_write_normalized_part(rows, parts_dir))
|
|
502
|
+
if not part_paths:
|
|
503
|
+
return None, payloads
|
|
504
|
+
pl.concat([pl.scan_parquet(part) for part in part_paths]).sort(
|
|
505
|
+
"SequenceID"
|
|
506
|
+
).sink_parquet(normalized_path, compression="zstd", maintain_order=True)
|
|
507
|
+
return ArtifactFile.from_path(
|
|
508
|
+
normalized_path, "application/vnd.apache.parquet"
|
|
509
|
+
), payloads
|
|
510
|
+
|
|
511
|
+
|
|
512
|
+
def _write_normalized_part(rows: list[dict[str, object]], directory: Path) -> Path:
|
|
513
|
+
path = directory / f"part-{len(tuple(directory.iterdir())):06d}.parquet"
|
|
514
|
+
pl.DataFrame(rows, schema=DBCAN_EVIDENCE_SCHEMA).write_parquet(path)
|
|
515
|
+
return path
|
|
516
|
+
|
|
517
|
+
|
|
518
|
+
def _parse_overview_row(
|
|
519
|
+
fields: list[str],
|
|
520
|
+
*,
|
|
521
|
+
expected: Mapping[str, SequenceIdentity],
|
|
522
|
+
line_number: int,
|
|
523
|
+
) -> dict[str, object]:
|
|
524
|
+
native = dict(zip(_OVERVIEW_COLUMNS, fields, strict=True))
|
|
525
|
+
sequence_id = native["Gene ID"]
|
|
526
|
+
if sequence_id not in expected:
|
|
527
|
+
raise AdapterError(
|
|
528
|
+
f"dbCAN overview line {line_number} has unknown SequenceID: {sequence_id}"
|
|
529
|
+
)
|
|
530
|
+
raw_tools = native["#ofTools"]
|
|
531
|
+
try:
|
|
532
|
+
tools = int(raw_tools)
|
|
533
|
+
except ValueError as error:
|
|
534
|
+
raise AdapterError(
|
|
535
|
+
f"dbCAN overview line {line_number} has invalid #ofTools: {raw_tools}"
|
|
536
|
+
) from error
|
|
537
|
+
if tools < 0 or tools > 3:
|
|
538
|
+
raise AdapterError(
|
|
539
|
+
f"dbCAN overview line {line_number} has invalid #ofTools: {raw_tools}"
|
|
540
|
+
)
|
|
541
|
+
return {
|
|
542
|
+
"SequenceID": sequence_id,
|
|
543
|
+
"Gene ID": sequence_id,
|
|
544
|
+
"EC#": _optional_text(native["EC#"]),
|
|
545
|
+
"dbCAN_hmm": _optional_text(native["dbCAN_hmm"]),
|
|
546
|
+
"dbCAN_sub": _optional_text(native["dbCAN_sub"]),
|
|
547
|
+
"DIAMOND": _optional_text(native["DIAMOND"]),
|
|
548
|
+
"#ofTools": tools,
|
|
549
|
+
"Recommend Results": _optional_text(native["Recommend Results"]),
|
|
550
|
+
"Substrate": _optional_text(native["Substrate"]),
|
|
551
|
+
}
|
|
552
|
+
|
|
553
|
+
|
|
554
|
+
def _optional_text(value: str) -> str | None:
|
|
555
|
+
return None if value.strip() in {"", "-", "NA", "None"} else value
|
|
556
|
+
|
|
557
|
+
|
|
558
|
+
def _gzip_artifact(source: Path, target: Path) -> ArtifactFile:
|
|
559
|
+
with (
|
|
560
|
+
source.open("rb") as source_handle,
|
|
561
|
+
target.open("wb") as target_handle,
|
|
562
|
+
gzip.GzipFile(fileobj=target_handle, mode="wb", mtime=0) as compressed,
|
|
563
|
+
):
|
|
564
|
+
shutil.copyfileobj(source_handle, compressed)
|
|
565
|
+
return ArtifactFile.from_path(target, "application/gzip")
|
|
566
|
+
|
|
567
|
+
|
|
568
|
+
def _sequence_result(
|
|
569
|
+
identity: SequenceIdentity,
|
|
570
|
+
*,
|
|
571
|
+
payload_digest: str | None,
|
|
572
|
+
) -> AdapterSequenceResult:
|
|
573
|
+
if payload_digest is None:
|
|
574
|
+
payload = json.dumps(
|
|
575
|
+
{"SequenceID": identity.sequence_id, "Status": "no_hit"},
|
|
576
|
+
sort_keys=True,
|
|
577
|
+
separators=(",", ":"),
|
|
578
|
+
).encode()
|
|
579
|
+
return AdapterSequenceResult(
|
|
580
|
+
sequence_id=identity.sequence_id,
|
|
581
|
+
status=EvidenceStatus.NO_HIT,
|
|
582
|
+
payload_digest=sha256_digest(payload),
|
|
583
|
+
)
|
|
584
|
+
return AdapterSequenceResult(
|
|
585
|
+
sequence_id=identity.sequence_id,
|
|
586
|
+
status=EvidenceStatus.HIT,
|
|
587
|
+
payload_digest=payload_digest,
|
|
588
|
+
)
|