seqevi 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. seqevi/__init__.py +7 -0
  2. seqevi/__main__.py +6 -0
  3. seqevi/adapters/__init__.py +41 -0
  4. seqevi/adapters/base.py +131 -0
  5. seqevi/adapters/dbcan_cazyme.py +588 -0
  6. seqevi/adapters/eggnog.py +674 -0
  7. seqevi/adapters/interpro_pfam.py +757 -0
  8. seqevi/adapters/registry.py +68 -0
  9. seqevi/annotate.py +413 -0
  10. seqevi/api.py +390 -0
  11. seqevi/cli.py +610 -0
  12. seqevi/distribution/__init__.py +13 -0
  13. seqevi/distribution/manifest.py +199 -0
  14. seqevi/distribution/oci.py +490 -0
  15. seqevi/distribution/setup.py +752 -0
  16. seqevi/errors.py +73 -0
  17. seqevi/evidence.py +295 -0
  18. seqevi/execution_profile.py +526 -0
  19. seqevi/hashing.py +13 -0
  20. seqevi/kits/__init__.py +1 -0
  21. seqevi/kits/dbcan-cazyme.toml +35 -0
  22. seqevi/resource_lock.py +438 -0
  23. seqevi/result.py +682 -0
  24. seqevi/runner.py +163 -0
  25. seqevi/runtime_identity.py +104 -0
  26. seqevi/sequence.py +383 -0
  27. seqevi/service/__init__.py +11 -0
  28. seqevi/service/app.py +213 -0
  29. seqevi/service/config.py +38 -0
  30. seqevi/service/persistence.py +360 -0
  31. seqevi/store/__init__.py +14 -0
  32. seqevi/store/artifact.py +225 -0
  33. seqevi/store/client.py +311 -0
  34. seqevi/store/contract.py +33 -0
  35. seqevi/store/factory.py +38 -0
  36. seqevi/store/local.py +479 -0
  37. seqevi/store/migration.py +62 -0
  38. seqevi/store/migrations/__init__.py +1 -0
  39. seqevi/store/migrations/env.py +30 -0
  40. seqevi/store/migrations/versions/0001_initial_store.py +103 -0
  41. seqevi/store/migrations/versions/0002_artifact_byte_size_bigint.py +40 -0
  42. seqevi/store/migrations/versions/__init__.py +1 -0
  43. seqevi/store/schema.py +86 -0
  44. seqevi/store/transport.py +224 -0
  45. seqevi-0.2.0.dist-info/METADATA +333 -0
  46. seqevi-0.2.0.dist-info/RECORD +49 -0
  47. seqevi-0.2.0.dist-info/WHEEL +4 -0
  48. seqevi-0.2.0.dist-info/entry_points.txt +5 -0
  49. seqevi-0.2.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,757 @@
1
+ """Direct InterProScan/Pfam annotation adapter."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import csv
6
+ import gzip
7
+ import hashlib
8
+ import json
9
+ import math
10
+ import os
11
+ import re
12
+ import shutil
13
+ import tempfile
14
+ from collections.abc import Mapping
15
+ from dataclasses import asdict, dataclass
16
+ from datetime import datetime
17
+ from pathlib import Path
18
+ from typing import Any
19
+
20
+ import polars as pl
21
+
22
+ from seqevi.errors import AdapterError
23
+ from seqevi.evidence import ArtifactFile, EvidenceStatus, sha256_digest
24
+ from seqevi.resource_lock import ResourceComponent, resolve_resource_lock
25
+ from seqevi.runner import ToolCommand, ToolRunner, ToolTimeoutError
26
+ from seqevi.runtime_identity import RuntimeComponent, calculate_runtime_digest
27
+ from seqevi.sequence import SequenceIdentity
28
+
29
+ from .base import AdapterBatchResult, AdapterContract, AdapterSequenceResult
30
+
31
+ ADAPTER_CONTRACT_VERSION = "interpro-pfam/1"
32
+
33
+ INTERPRO_PFAM_EVIDENCE_SCHEMA: Mapping[str, pl.DataType] = {
34
+ "SequenceID": pl.String(),
35
+ "ProteinAccession": pl.String(),
36
+ "SequenceMD5": pl.String(),
37
+ "SequenceLength": pl.Int64(),
38
+ "Analysis": pl.String(),
39
+ "SignatureAccession": pl.String(),
40
+ "SignatureDescription": pl.String(),
41
+ "Start": pl.Int64(),
42
+ "Stop": pl.Int64(),
43
+ "Score": pl.Float64(),
44
+ "Status": pl.String(),
45
+ "InterProAccession": pl.String(),
46
+ "InterProDescription": pl.String(),
47
+ }
48
+
49
+ _VERSION_PATTERN = re.compile(r"\b5\.\d+-\d+\.\d+\b")
50
+ _PFAM_VERSION_PATTERN = re.compile(r"\d+(?:\.\d+)+\Z")
51
+ _PFAM_ACCESSION_PATTERN = re.compile(r"PF\d{5}\Z")
52
+ _INTERPRO_ACCESSION_PATTERN = re.compile(r"IPR\d{6}\Z")
53
+ _TSV_COLUMN_COUNT = 15
54
+ _PROBE_TIMEOUT_SECONDS = 120.0
55
+ _NORMALIZED_ROW_BATCH_SIZE = 1000
56
+
57
+
58
+ @dataclass(frozen=True, slots=True)
59
+ class InterProPfamParameters:
60
+ """Fixed scientific parameters for the v1 direct Pfam contract."""
61
+
62
+ application: str = "Pfam"
63
+ output_format: str = "TSV"
64
+ disable_precalculated_lookup: bool = True
65
+ include_go_terms: bool = False
66
+ include_pathways: bool = False
67
+ sequence_type: str = "protein"
68
+
69
+ def __post_init__(self) -> None:
70
+ values = (
71
+ self.application,
72
+ self.output_format,
73
+ self.disable_precalculated_lookup,
74
+ self.include_go_terms,
75
+ self.include_pathways,
76
+ self.sequence_type,
77
+ )
78
+ if values != ("Pfam", "TSV", True, False, False, "protein"):
79
+ raise ValueError(
80
+ "interpro-pfam/1 uses one fixed direct-scan scientific contract"
81
+ )
82
+
83
+ def as_semantic_parameters(self) -> dict[str, object]:
84
+ """Return every result-affecting parameter with explicit defaults."""
85
+
86
+ return asdict(self)
87
+
88
+
89
+ class InterProPfamAdapter:
90
+ """Run local InterProScan Pfam and validate its native TSV output."""
91
+
92
+ def __init__(
93
+ self,
94
+ *,
95
+ executable: Path,
96
+ database: Path,
97
+ parameters: InterProPfamParameters | None = None,
98
+ verify_resource: bool = False,
99
+ environment: Mapping[str, str] | None = None,
100
+ ) -> None:
101
+ self.executable = executable.resolve()
102
+ self.database = database.resolve()
103
+ self.parameters = parameters or InterProPfamParameters()
104
+ self.environment = dict(environment or {})
105
+ self.install_dir = self.executable.parent
106
+ self.properties_path = self.install_dir / "interproscan.properties"
107
+ self.jar_path = self.install_dir / "interproscan-5.jar"
108
+
109
+ self._validate_installation_paths()
110
+ self.properties = _read_properties(self.properties_path)
111
+ self.interproscan_version = _probe_interproscan_version(
112
+ self.executable,
113
+ environment=self.environment,
114
+ )
115
+ pfam_version, model_path = _resolve_pfam_model(
116
+ database=self.database,
117
+ properties=self.properties,
118
+ )
119
+ runtime_digest = _calculate_runtime_digest(
120
+ install_dir=self.install_dir,
121
+ executable=self.executable,
122
+ jar_path=self.jar_path,
123
+ properties=self.properties,
124
+ version=self.interproscan_version,
125
+ environment=self.environment,
126
+ )
127
+ resource_id = _calculate_resource_id(
128
+ database=self.database,
129
+ interproscan_version=self.interproscan_version,
130
+ pfam_version=pfam_version,
131
+ model_path=model_path,
132
+ verify=verify_resource,
133
+ )
134
+ self._contract = AdapterContract.from_parameters(
135
+ name="interpro-pfam",
136
+ version=ADAPTER_CONTRACT_VERSION,
137
+ tool_runtime_digest=f"sha256:{runtime_digest}",
138
+ resource_id=resource_id,
139
+ semantic_parameters=self.parameters.as_semantic_parameters(),
140
+ )
141
+
142
+ @property
143
+ def contract(self) -> AdapterContract:
144
+ return self._contract
145
+
146
+ @property
147
+ def evidence_schema(self) -> Mapping[str, pl.DataType]:
148
+ return INTERPRO_PFAM_EVIDENCE_SCHEMA
149
+
150
+ def run_batch(
151
+ self,
152
+ *,
153
+ identities: tuple[SequenceIdentity, ...],
154
+ input_fasta: Path,
155
+ work_dir: Path,
156
+ runner: ToolRunner,
157
+ timeout_seconds: float | None,
158
+ threads: int,
159
+ ) -> AdapterBatchResult:
160
+ """Run one deterministic cache-miss batch and validate every result."""
161
+
162
+ if not identities:
163
+ raise AdapterError("interpro-pfam batch must not be empty")
164
+ ids = [identity.sequence_id for identity in identities]
165
+ if len(ids) != len(set(ids)):
166
+ raise AdapterError("interpro-pfam batch contains duplicate SequenceIDs")
167
+
168
+ raw_path = work_dir / "interpro-pfam.tsv"
169
+ properties_path = work_dir / "interproscan.properties"
170
+ temporary_dir = work_dir / "interproscan-temp"
171
+ _write_runtime_properties(
172
+ source=self.properties_path,
173
+ target=properties_path,
174
+ database=self.database,
175
+ )
176
+ result = runner.run(
177
+ ToolCommand(
178
+ arguments=(
179
+ str(self.executable),
180
+ "--input",
181
+ str(input_fasta),
182
+ "--applications",
183
+ self.parameters.application,
184
+ "--formats",
185
+ self.parameters.output_format,
186
+ "--disable-precalc",
187
+ "--cpu",
188
+ str(threads),
189
+ "--outfile",
190
+ str(raw_path),
191
+ "--tempdir",
192
+ str(temporary_dir),
193
+ ),
194
+ working_dir=work_dir,
195
+ stdout_path=work_dir / "interproscan.stdout.log",
196
+ stderr_path=work_dir / "interproscan.stderr.log",
197
+ environment={
198
+ **self.environment,
199
+ "INTERPROSCAN_CONF": str(properties_path),
200
+ },
201
+ ),
202
+ timeout_seconds=timeout_seconds,
203
+ )
204
+ if result.return_code != 0:
205
+ raise AdapterError(
206
+ f"InterProScan exited with {result.return_code}; "
207
+ f"stderr: {result.stderr_path}"
208
+ )
209
+ if not raw_path.is_file():
210
+ raise AdapterError("InterProScan did not create its TSV output")
211
+
212
+ normalized, payload_digest_by_id = _parse_tsv(
213
+ raw_path,
214
+ identities=identities,
215
+ normalized_path=work_dir / "interpro-pfam.normalized.parquet",
216
+ )
217
+ sequence_results = tuple(
218
+ _sequence_result(
219
+ identity,
220
+ payload_digest=payload_digest_by_id.get(identity.sequence_id),
221
+ )
222
+ for identity in sorted(identities, key=lambda item: item.sequence_id)
223
+ )
224
+ return AdapterBatchResult(
225
+ sequences=sequence_results,
226
+ raw_artifact=_gzip_artifact(
227
+ raw_path,
228
+ work_dir / "interpro-pfam.tsv.gz",
229
+ ),
230
+ normalized_artifact=normalized,
231
+ )
232
+
233
+ def _validate_installation_paths(self) -> None:
234
+ if not self.executable.is_file():
235
+ raise AdapterError(
236
+ f"InterProScan executable is not a file: {self.executable}"
237
+ )
238
+ if not self.database.is_dir():
239
+ raise AdapterError(
240
+ f"InterProScan data directory does not exist: {self.database}"
241
+ )
242
+ if not self.properties_path.is_file():
243
+ raise AdapterError(
244
+ "InterProScan launcher must be beside interproscan.properties: "
245
+ f"{self.properties_path}"
246
+ )
247
+ if not self.jar_path.is_file():
248
+ raise AdapterError(
249
+ f"InterProScan runtime jar does not exist: {self.jar_path}"
250
+ )
251
+
252
+
253
+ def _probe_interproscan_version(
254
+ executable: Path,
255
+ *,
256
+ environment: Mapping[str, str],
257
+ ) -> str:
258
+ with tempfile.TemporaryDirectory(prefix="seqevi-interpro-probe-") as raw_dir:
259
+ probe_dir = Path(raw_dir)
260
+ stdout_path = probe_dir / "stdout.log"
261
+ stderr_path = probe_dir / "stderr.log"
262
+ try:
263
+ result = ToolRunner().run(
264
+ ToolCommand(
265
+ arguments=(str(executable), "--version"),
266
+ working_dir=executable.parent,
267
+ stdout_path=stdout_path,
268
+ stderr_path=stderr_path,
269
+ environment=environment,
270
+ ),
271
+ timeout_seconds=_PROBE_TIMEOUT_SECONDS,
272
+ )
273
+ except (OSError, ToolTimeoutError) as error:
274
+ raise AdapterError(f"InterProScan version probe failed: {error}") from error
275
+
276
+ output = "\n".join(
277
+ (
278
+ stdout_path.read_text(encoding="utf-8", errors="replace"),
279
+ stderr_path.read_text(encoding="utf-8", errors="replace"),
280
+ )
281
+ )
282
+ if result.return_code != 0:
283
+ raise AdapterError(
284
+ "InterProScan version probe exited with "
285
+ f"{result.return_code}: {output.strip()}"
286
+ )
287
+ versions = sorted(set(_VERSION_PATTERN.findall(output)))
288
+ if len(versions) != 1:
289
+ raise AdapterError(
290
+ "InterProScan version probe must report exactly one release: "
291
+ f"{versions}"
292
+ )
293
+ return versions[0]
294
+
295
+
296
+ def _read_properties(path: Path) -> dict[str, str]:
297
+ properties: dict[str, str] = {}
298
+ for line_number, line in enumerate(
299
+ path.read_text(encoding="utf-8").splitlines(), start=1
300
+ ):
301
+ stripped = line.strip()
302
+ if not stripped or stripped.startswith(("#", "!")):
303
+ continue
304
+ if "=" not in stripped:
305
+ raise AdapterError(f"invalid InterProScan property at {path}:{line_number}")
306
+ key, value = (part.strip() for part in stripped.split("=", maxsplit=1))
307
+ if not key or key in properties:
308
+ raise AdapterError(
309
+ f"duplicate or empty InterProScan property at {path}:{line_number}"
310
+ )
311
+ properties[key] = value
312
+
313
+ required = {
314
+ "bin.directory",
315
+ "binary.hmmer3.path",
316
+ "data.directory",
317
+ "pfam-a.hmm.path",
318
+ }
319
+ missing = sorted(required - properties.keys())
320
+ if missing:
321
+ raise AdapterError(
322
+ f"InterProScan properties are missing required keys: {missing}"
323
+ )
324
+ return properties
325
+
326
+
327
+ def _resolve_pfam_model(
328
+ *, database: Path, properties: Mapping[str, str]
329
+ ) -> tuple[str, Path]:
330
+ configured = properties["pfam-a.hmm.path"]
331
+ prefix = "${data.directory}/"
332
+ if not configured.startswith(prefix):
333
+ raise AdapterError("pfam-a.hmm.path must be relative to ${data.directory}")
334
+ relative = Path(configured.removeprefix(prefix))
335
+ if relative.is_absolute() or ".." in relative.parts:
336
+ raise AdapterError("pfam-a.hmm.path escapes the InterProScan data directory")
337
+ model_path = (database / relative).resolve()
338
+ if not model_path.is_relative_to(database) or not model_path.is_file():
339
+ raise AdapterError(f"Pfam model file does not exist: {model_path}")
340
+ if len(relative.parts) < 3 or relative.parts[0].lower() != "pfam":
341
+ raise AdapterError(f"unexpected Pfam model path: {relative}")
342
+ pfam_version = relative.parts[1]
343
+ if not _PFAM_VERSION_PATTERN.fullmatch(pfam_version):
344
+ raise AdapterError(f"invalid Pfam release in model path: {pfam_version}")
345
+ return pfam_version, model_path
346
+
347
+
348
+ def _calculate_runtime_digest(
349
+ *,
350
+ install_dir: Path,
351
+ executable: Path,
352
+ jar_path: Path,
353
+ properties: Mapping[str, str],
354
+ version: str,
355
+ environment: Mapping[str, str],
356
+ ) -> str:
357
+ bin_dir = _resolve_install_path(
358
+ install_dir,
359
+ properties["bin.directory"],
360
+ variables={},
361
+ )
362
+ hmmer_dir = _resolve_install_path(
363
+ install_dir,
364
+ properties["binary.hmmer3.path"],
365
+ variables={"bin.directory": bin_dir},
366
+ )
367
+ if not hmmer_dir.is_dir():
368
+ raise AdapterError(f"InterProScan HMMER directory does not exist: {hmmer_dir}")
369
+ hmmer_files = sorted(path for path in hmmer_dir.rglob("*") if path.is_file())
370
+ if not hmmer_files:
371
+ raise AdapterError(f"InterProScan HMMER directory is empty: {hmmer_dir}")
372
+
373
+ java = shutil.which(
374
+ "java",
375
+ path=environment.get("PATH", os.environ.get("PATH")),
376
+ )
377
+ if java is None:
378
+ raise AdapterError("InterProScan runtime has no Java executable")
379
+ normalized_properties = {
380
+ **properties,
381
+ "data.directory": "${data.directory}",
382
+ }
383
+ properties_digest = sha256_digest(
384
+ json.dumps(
385
+ normalized_properties,
386
+ ensure_ascii=True,
387
+ sort_keys=True,
388
+ separators=(",", ":"),
389
+ ).encode("utf-8")
390
+ )
391
+ components = (
392
+ RuntimeComponent("launcher", executable),
393
+ RuntimeComponent("java", Path(java).resolve()),
394
+ RuntimeComponent("interproscan-5.jar", jar_path),
395
+ *(
396
+ RuntimeComponent(
397
+ f"hmmer/{path.relative_to(hmmer_dir).as_posix()}",
398
+ path,
399
+ )
400
+ for path in hmmer_files
401
+ ),
402
+ )
403
+ return calculate_runtime_digest(
404
+ runtime_name="interproscan-pfam",
405
+ versions={
406
+ "interproscan": version,
407
+ "properties": f"sha256:{properties_digest}",
408
+ },
409
+ components=components,
410
+ )
411
+
412
+
413
+ def _resolve_install_path(
414
+ install_dir: Path,
415
+ raw_value: str,
416
+ *,
417
+ variables: Mapping[str, Path],
418
+ ) -> Path:
419
+ expanded = raw_value
420
+ for name, path in variables.items():
421
+ expanded = expanded.replace(f"${{{name}}}", str(path))
422
+ if "${" in expanded:
423
+ raise AdapterError(f"unsupported InterProScan property expression: {raw_value}")
424
+ candidate = Path(expanded)
425
+ resolved = (
426
+ candidate.resolve()
427
+ if candidate.is_absolute()
428
+ else (install_dir / candidate).resolve()
429
+ )
430
+ if not resolved.is_relative_to(install_dir):
431
+ raise AdapterError(f"InterProScan runtime path escapes its install: {resolved}")
432
+ return resolved
433
+
434
+
435
+ def _calculate_resource_id(
436
+ *,
437
+ database: Path,
438
+ interproscan_version: str,
439
+ pfam_version: str,
440
+ model_path: Path,
441
+ verify: bool = False,
442
+ ) -> str:
443
+ interpro_release = interproscan_version.split("-", maxsplit=1)[1]
444
+ metadata_path = model_path.with_name("pfam_a.dat")
445
+ if not metadata_path.is_file():
446
+ raise AdapterError(f"Pfam metadata file does not exist: {metadata_path}")
447
+ declarations = (
448
+ ResourceComponent(
449
+ "pfam-model",
450
+ model_path.relative_to(database).as_posix(),
451
+ ),
452
+ ResourceComponent(
453
+ "pfam-metadata",
454
+ metadata_path.relative_to(database).as_posix(),
455
+ ),
456
+ )
457
+ locked = resolve_resource_lock(
458
+ database=database,
459
+ resource_name="interpro",
460
+ resource_version=interpro_release,
461
+ components=declarations,
462
+ verify=verify,
463
+ )
464
+ digest = sha256_digest(
465
+ json.dumps(
466
+ [
467
+ (component.name, locked.hash_for(component.name))
468
+ for component in declarations
469
+ ],
470
+ separators=(",", ":"),
471
+ ).encode("utf-8")
472
+ )
473
+ return f"interpro/{interpro_release}/pfam/{pfam_version}/sha256:{digest}"
474
+
475
+
476
+ def _write_runtime_properties(*, source: Path, target: Path, database: Path) -> None:
477
+ lines = source.read_text(encoding="utf-8").splitlines()
478
+ replaced = False
479
+ output = []
480
+ for line in lines:
481
+ stripped = line.strip()
482
+ if (
483
+ stripped
484
+ and not stripped.startswith(("#", "!"))
485
+ and stripped.split("=", maxsplit=1)[0].strip() == "data.directory"
486
+ ):
487
+ output.append(f"data.directory={database}")
488
+ replaced = True
489
+ else:
490
+ output.append(line)
491
+ if not replaced:
492
+ raise AdapterError("InterProScan properties have no data.directory entry")
493
+ target.write_text("\n".join(output) + "\n", encoding="utf-8")
494
+
495
+
496
+ def _parse_tsv(
497
+ path: Path,
498
+ *,
499
+ identities: tuple[SequenceIdentity, ...],
500
+ normalized_path: Path,
501
+ ) -> tuple[ArtifactFile | None, dict[str, str]]:
502
+ expected = {identity.sequence_id: identity for identity in identities}
503
+ rows: list[dict[str, object]] = []
504
+ with tempfile.TemporaryDirectory(
505
+ prefix=".interpro-pfam-normalized-", dir=normalized_path.parent
506
+ ) as raw_parts_dir:
507
+ parts_dir = Path(raw_parts_dir)
508
+ part_paths: list[Path] = []
509
+ try:
510
+ with path.open("r", encoding="utf-8", newline="") as handle:
511
+ reader = csv.reader(handle, delimiter="\t")
512
+ for line_number, fields in enumerate(reader, start=1):
513
+ if not fields or fields == [""]:
514
+ raise AdapterError(
515
+ "InterProScan TSV contains a blank row at line "
516
+ f"{line_number}"
517
+ )
518
+ if len(fields) != _TSV_COLUMN_COUNT:
519
+ raise AdapterError(
520
+ f"InterProScan TSV line {line_number} has "
521
+ f"{len(fields)} columns; expected {_TSV_COLUMN_COUNT}"
522
+ )
523
+ row = _parse_tsv_row(
524
+ fields, line_number=line_number, expected=expected
525
+ )
526
+ rows.append(row)
527
+ if len(rows) >= _NORMALIZED_ROW_BATCH_SIZE:
528
+ part_paths.append(_write_normalized_part(rows, parts_dir))
529
+ rows.clear()
530
+ except UnicodeDecodeError as error:
531
+ raise AdapterError(
532
+ f"InterProScan TSV is not valid UTF-8: {error}"
533
+ ) from error
534
+
535
+ if rows:
536
+ part_paths.append(_write_normalized_part(rows, parts_dir))
537
+ if not part_paths:
538
+ return None, {}
539
+ normalized = pl.concat([pl.scan_parquet(part) for part in part_paths]).sort(
540
+ *INTERPRO_PFAM_EVIDENCE_SCHEMA,
541
+ nulls_last=True,
542
+ )
543
+ payload_digest_by_id = _payload_digests(normalized)
544
+ normalized.sink_parquet(
545
+ normalized_path, compression="zstd", maintain_order=True
546
+ )
547
+ return (
548
+ ArtifactFile.from_path(normalized_path, "application/vnd.apache.parquet"),
549
+ payload_digest_by_id,
550
+ )
551
+
552
+
553
+ def _write_normalized_part(rows: list[dict[str, object]], directory: Path) -> Path:
554
+ path = directory / f"part-{len(tuple(directory.iterdir())):06d}.parquet"
555
+ pl.DataFrame(rows, schema=INTERPRO_PFAM_EVIDENCE_SCHEMA).write_parquet(path)
556
+ return path
557
+
558
+
559
+ def _payload_digests(frame: pl.LazyFrame) -> dict[str, str]:
560
+ digests: dict[str, str] = {}
561
+ current_sequence_id: str | None = None
562
+ current_hasher: Any | None = None
563
+ previous_row: tuple[object, ...] | None = None
564
+ row_index = 0
565
+ for batch in frame.collect_batches(
566
+ chunk_size=_NORMALIZED_ROW_BATCH_SIZE,
567
+ engine="streaming",
568
+ ):
569
+ for row in batch.iter_rows(named=True):
570
+ row_key = tuple(row[column] for column in INTERPRO_PFAM_EVIDENCE_SCHEMA)
571
+ if row_key == previous_row:
572
+ raise AdapterError("InterProScan TSV contains a duplicate match")
573
+ previous_row = row_key
574
+ sequence_id = str(row["SequenceID"])
575
+ if sequence_id != current_sequence_id:
576
+ if current_sequence_id is not None and current_hasher is not None:
577
+ current_hasher.update(b"]")
578
+ digests[current_sequence_id] = current_hasher.hexdigest()
579
+ current_sequence_id = sequence_id
580
+ current_hasher = hashlib.sha256()
581
+ current_hasher.update(b"[")
582
+ row_index = 0
583
+ if current_hasher is None:
584
+ raise AssertionError("payload digest state was not initialized")
585
+ if row_index:
586
+ current_hasher.update(b",")
587
+ current_hasher.update(
588
+ json.dumps(
589
+ row,
590
+ ensure_ascii=True,
591
+ allow_nan=False,
592
+ sort_keys=True,
593
+ separators=(",", ":"),
594
+ ).encode("utf-8")
595
+ )
596
+ row_index += 1
597
+ if current_sequence_id is not None and current_hasher is not None:
598
+ current_hasher.update(b"]")
599
+ digests[current_sequence_id] = current_hasher.hexdigest()
600
+ return digests
601
+
602
+
603
+ def _parse_tsv_row(
604
+ fields: list[str],
605
+ *,
606
+ line_number: int,
607
+ expected: Mapping[str, SequenceIdentity],
608
+ ) -> dict[str, object]:
609
+ (
610
+ accession,
611
+ sequence_md5,
612
+ raw_length,
613
+ analysis,
614
+ signature_accession,
615
+ signature_description,
616
+ raw_start,
617
+ raw_stop,
618
+ raw_score,
619
+ status,
620
+ run_date,
621
+ interpro_accession,
622
+ interpro_description,
623
+ go_annotations,
624
+ pathway_annotations,
625
+ ) = fields
626
+ identity = expected.get(accession)
627
+ if identity is None:
628
+ raise AdapterError(
629
+ f"InterProScan TSV line {line_number} has an unknown SequenceID: "
630
+ f"{accession}"
631
+ )
632
+ if sequence_md5.lower() != identity.md5:
633
+ raise AdapterError(
634
+ f"InterProScan TSV line {line_number} MD5 does not match {accession}"
635
+ )
636
+ sequence_length = _parse_int(raw_length, field="sequence length", line=line_number)
637
+ start = _parse_int(raw_start, field="start", line=line_number)
638
+ stop = _parse_int(raw_stop, field="stop", line=line_number)
639
+ if sequence_length != identity.length:
640
+ raise AdapterError(
641
+ f"InterProScan TSV line {line_number} length does not match {accession}"
642
+ )
643
+ if not 1 <= start <= stop <= sequence_length:
644
+ raise AdapterError(
645
+ f"InterProScan TSV line {line_number} has invalid match coordinates"
646
+ )
647
+ if analysis != "Pfam":
648
+ raise AdapterError(
649
+ f"InterProScan TSV line {line_number} is not a Pfam match: {analysis}"
650
+ )
651
+ if not _PFAM_ACCESSION_PATTERN.fullmatch(signature_accession):
652
+ raise AdapterError(
653
+ f"InterProScan TSV line {line_number} has invalid Pfam accession: "
654
+ f"{signature_accession}"
655
+ )
656
+ score = _parse_float(raw_score, field="score", line=line_number)
657
+ if status != "T":
658
+ raise AdapterError(
659
+ f"InterProScan TSV line {line_number} has invalid status: {status}"
660
+ )
661
+ try:
662
+ datetime.strptime(run_date, "%d-%m-%Y")
663
+ except ValueError as error:
664
+ raise AdapterError(
665
+ f"InterProScan TSV line {line_number} has invalid run date: {run_date}"
666
+ ) from error
667
+ normalized_interpro = _optional_text(interpro_accession)
668
+ if normalized_interpro is not None and not _INTERPRO_ACCESSION_PATTERN.fullmatch(
669
+ normalized_interpro
670
+ ):
671
+ raise AdapterError(
672
+ f"InterProScan TSV line {line_number} has invalid InterPro accession: "
673
+ f"{normalized_interpro}"
674
+ )
675
+ if go_annotations != "-" or pathway_annotations != "-":
676
+ raise AdapterError(
677
+ f"InterProScan TSV line {line_number} contains GO or pathway annotations "
678
+ "outside the interpro-pfam/1 contract"
679
+ )
680
+
681
+ return {
682
+ "SequenceID": identity.sequence_id,
683
+ "ProteinAccession": accession,
684
+ "SequenceMD5": identity.md5,
685
+ "SequenceLength": sequence_length,
686
+ "Analysis": analysis,
687
+ "SignatureAccession": signature_accession,
688
+ "SignatureDescription": _optional_text(signature_description),
689
+ "Start": start,
690
+ "Stop": stop,
691
+ "Score": score,
692
+ "Status": status,
693
+ "InterProAccession": normalized_interpro,
694
+ "InterProDescription": _optional_text(interpro_description),
695
+ }
696
+
697
+
698
+ def _parse_int(value: str, *, field: str, line: int) -> int:
699
+ try:
700
+ return int(value)
701
+ except ValueError as error:
702
+ raise AdapterError(
703
+ f"InterProScan TSV line {line} has invalid {field}: {value}"
704
+ ) from error
705
+
706
+
707
+ def _parse_float(value: str, *, field: str, line: int) -> float:
708
+ try:
709
+ parsed = float(value)
710
+ except ValueError as error:
711
+ raise AdapterError(
712
+ f"InterProScan TSV line {line} has invalid {field}: {value}"
713
+ ) from error
714
+ if not math.isfinite(parsed):
715
+ raise AdapterError(
716
+ f"InterProScan TSV line {line} has non-finite {field}: {value}"
717
+ )
718
+ return parsed
719
+
720
+
721
+ def _optional_text(value: str) -> str | None:
722
+ return None if value == "-" else value
723
+
724
+
725
+ def _gzip_artifact(source: Path, target: Path) -> ArtifactFile:
726
+ with (
727
+ source.open("rb") as source_handle,
728
+ target.open("wb") as target_handle,
729
+ gzip.GzipFile(fileobj=target_handle, mode="wb", mtime=0) as compressed,
730
+ ):
731
+ shutil.copyfileobj(source_handle, compressed)
732
+ return ArtifactFile.from_path(target, "application/gzip")
733
+
734
+
735
+ def _sequence_result(
736
+ identity: SequenceIdentity,
737
+ *,
738
+ payload_digest: str | None,
739
+ ) -> AdapterSequenceResult:
740
+ if payload_digest is None:
741
+ payload = json.dumps(
742
+ {"SequenceID": identity.sequence_id, "Status": "no_hit"},
743
+ ensure_ascii=True,
744
+ sort_keys=True,
745
+ separators=(",", ":"),
746
+ ).encode("utf-8")
747
+ return AdapterSequenceResult(
748
+ sequence_id=identity.sequence_id,
749
+ status=EvidenceStatus.NO_HIT,
750
+ payload_digest=sha256_digest(payload),
751
+ )
752
+
753
+ return AdapterSequenceResult(
754
+ sequence_id=identity.sequence_id,
755
+ status=EvidenceStatus.HIT,
756
+ payload_digest=payload_digest,
757
+ )