seqevi 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- seqevi/__init__.py +7 -0
- seqevi/__main__.py +6 -0
- seqevi/adapters/__init__.py +41 -0
- seqevi/adapters/base.py +131 -0
- seqevi/adapters/dbcan_cazyme.py +588 -0
- seqevi/adapters/eggnog.py +674 -0
- seqevi/adapters/interpro_pfam.py +757 -0
- seqevi/adapters/registry.py +68 -0
- seqevi/annotate.py +413 -0
- seqevi/api.py +390 -0
- seqevi/cli.py +610 -0
- seqevi/distribution/__init__.py +13 -0
- seqevi/distribution/manifest.py +199 -0
- seqevi/distribution/oci.py +490 -0
- seqevi/distribution/setup.py +752 -0
- seqevi/errors.py +73 -0
- seqevi/evidence.py +295 -0
- seqevi/execution_profile.py +526 -0
- seqevi/hashing.py +13 -0
- seqevi/kits/__init__.py +1 -0
- seqevi/kits/dbcan-cazyme.toml +35 -0
- seqevi/resource_lock.py +438 -0
- seqevi/result.py +682 -0
- seqevi/runner.py +163 -0
- seqevi/runtime_identity.py +104 -0
- seqevi/sequence.py +383 -0
- seqevi/service/__init__.py +11 -0
- seqevi/service/app.py +213 -0
- seqevi/service/config.py +38 -0
- seqevi/service/persistence.py +360 -0
- seqevi/store/__init__.py +14 -0
- seqevi/store/artifact.py +225 -0
- seqevi/store/client.py +311 -0
- seqevi/store/contract.py +33 -0
- seqevi/store/factory.py +38 -0
- seqevi/store/local.py +479 -0
- seqevi/store/migration.py +62 -0
- seqevi/store/migrations/__init__.py +1 -0
- seqevi/store/migrations/env.py +30 -0
- seqevi/store/migrations/versions/0001_initial_store.py +103 -0
- seqevi/store/migrations/versions/0002_artifact_byte_size_bigint.py +40 -0
- seqevi/store/migrations/versions/__init__.py +1 -0
- seqevi/store/schema.py +86 -0
- seqevi/store/transport.py +224 -0
- seqevi-0.2.0.dist-info/METADATA +333 -0
- seqevi-0.2.0.dist-info/RECORD +49 -0
- seqevi-0.2.0.dist-info/WHEEL +4 -0
- seqevi-0.2.0.dist-info/entry_points.txt +5 -0
- seqevi-0.2.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
"""Explicit registry for official SeqEvi adapters."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from enum import StrEnum
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import NoReturn
|
|
9
|
+
|
|
10
|
+
from seqevi.errors import AdapterUnavailableError
|
|
11
|
+
|
|
12
|
+
from .base import AnnotationAdapter
|
|
13
|
+
from .dbcan_cazyme import DBCanCazymeAdapter
|
|
14
|
+
from .eggnog import EggnogAdapter
|
|
15
|
+
from .interpro_pfam import InterProPfamAdapter
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class AdapterName(StrEnum):
|
|
19
|
+
"""Official v1 adapter names accepted by the public CLI."""
|
|
20
|
+
|
|
21
|
+
EGGNOG = "eggnog"
|
|
22
|
+
INTERPRO_PFAM = "interpro-pfam"
|
|
23
|
+
DBCAN_CAZYME = "dbcan-cazyme"
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass(frozen=True, slots=True)
|
|
27
|
+
class AdapterConfiguration:
|
|
28
|
+
"""Concrete external tool and database locations supplied by a caller."""
|
|
29
|
+
|
|
30
|
+
name: AdapterName
|
|
31
|
+
executable: Path
|
|
32
|
+
database: Path
|
|
33
|
+
verify_resource: bool = False
|
|
34
|
+
environment: tuple[tuple[str, str], ...] = ()
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def create_adapter(configuration: AdapterConfiguration) -> AnnotationAdapter:
|
|
38
|
+
"""Create an official adapter implemented by the installed release."""
|
|
39
|
+
|
|
40
|
+
if configuration.name is AdapterName.INTERPRO_PFAM:
|
|
41
|
+
return InterProPfamAdapter(
|
|
42
|
+
executable=configuration.executable,
|
|
43
|
+
database=configuration.database,
|
|
44
|
+
verify_resource=configuration.verify_resource,
|
|
45
|
+
environment=dict(configuration.environment),
|
|
46
|
+
)
|
|
47
|
+
if configuration.name is AdapterName.EGGNOG:
|
|
48
|
+
return EggnogAdapter(
|
|
49
|
+
executable=configuration.executable,
|
|
50
|
+
database=configuration.database,
|
|
51
|
+
verify_resource=configuration.verify_resource,
|
|
52
|
+
environment=dict(configuration.environment),
|
|
53
|
+
)
|
|
54
|
+
if configuration.name is AdapterName.DBCAN_CAZYME:
|
|
55
|
+
return DBCanCazymeAdapter(
|
|
56
|
+
executable=configuration.executable,
|
|
57
|
+
database=configuration.database,
|
|
58
|
+
verify_resource=configuration.verify_resource,
|
|
59
|
+
environment=dict(configuration.environment),
|
|
60
|
+
)
|
|
61
|
+
_raise_unavailable(configuration.name, 4)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _raise_unavailable(name: AdapterName, phase: int) -> NoReturn:
|
|
65
|
+
raise AdapterUnavailableError(
|
|
66
|
+
f"adapter {name.value!r} is declared for v1 but is implemented in Phase "
|
|
67
|
+
f"{phase}; it is not implemented in this release"
|
|
68
|
+
)
|
seqevi/annotate.py
ADDED
|
@@ -0,0 +1,413 @@
|
|
|
1
|
+
"""Shallow orchestration for one exact annotation invocation."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import shutil
|
|
6
|
+
import sys
|
|
7
|
+
import tempfile
|
|
8
|
+
import time
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
from collections.abc import Mapping
|
|
11
|
+
from datetime import UTC, datetime
|
|
12
|
+
from itertools import batched
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from typing import Literal
|
|
15
|
+
|
|
16
|
+
from . import __version__
|
|
17
|
+
from .adapters.base import AdapterBatchResult, AnnotationAdapter
|
|
18
|
+
from .errors import AnnotationError, OutputPackageError
|
|
19
|
+
from .evidence import (
|
|
20
|
+
EvidenceCommit,
|
|
21
|
+
EvidenceQuery,
|
|
22
|
+
EvidenceSource,
|
|
23
|
+
EvidenceStatus,
|
|
24
|
+
)
|
|
25
|
+
from .result import RESULT_FORMAT_VERSION, materialize_result_database
|
|
26
|
+
from .runner import ToolCommand, ToolRunResult, ToolRunner
|
|
27
|
+
from .sequence import (
|
|
28
|
+
SequenceIdentity,
|
|
29
|
+
iter_staged_identities,
|
|
30
|
+
iter_staged_records,
|
|
31
|
+
iter_fasta_lines,
|
|
32
|
+
stage_fasta,
|
|
33
|
+
)
|
|
34
|
+
from .store.contract import EvidenceStore
|
|
35
|
+
|
|
36
|
+
_STORE_BATCH_SIZE = 1_000
|
|
37
|
+
_ANNOTATION_BATCH_SIZE = 10_000
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@dataclass(frozen=True, slots=True)
|
|
41
|
+
class AnnotationMetrics:
|
|
42
|
+
"""Operational measurements for one successful annotation invocation."""
|
|
43
|
+
|
|
44
|
+
elapsed_seconds: float
|
|
45
|
+
fasta_staging_seconds: float
|
|
46
|
+
store_lookup_seconds: float
|
|
47
|
+
adapter_seconds: float
|
|
48
|
+
external_tool_seconds: float
|
|
49
|
+
store_commit_seconds: float
|
|
50
|
+
store_fetch_seconds: float
|
|
51
|
+
package_seconds: float
|
|
52
|
+
peak_rss_kib: int | None
|
|
53
|
+
store_lookup_batches: int
|
|
54
|
+
store_commit_batches: int
|
|
55
|
+
store_fetch_batches: int
|
|
56
|
+
tool_batches: int
|
|
57
|
+
unique_artifact_reads: int
|
|
58
|
+
configured_threads: int
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
@dataclass(frozen=True, slots=True)
|
|
62
|
+
class AnnotationSummary:
|
|
63
|
+
"""Counts and output location for one successful invocation."""
|
|
64
|
+
|
|
65
|
+
input_records: int
|
|
66
|
+
unique_sequences: int
|
|
67
|
+
cache_hits: int
|
|
68
|
+
computed: int
|
|
69
|
+
hits: int
|
|
70
|
+
no_hits: int
|
|
71
|
+
output_dir: Path
|
|
72
|
+
metrics: AnnotationMetrics
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
@dataclass(frozen=True, slots=True)
|
|
76
|
+
class _BatchMetrics:
|
|
77
|
+
commit_batches: int
|
|
78
|
+
adapter_seconds: float
|
|
79
|
+
store_commit_seconds: float
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
class _MeasuringToolRunner(ToolRunner):
|
|
83
|
+
def __init__(self, delegate: ToolRunner) -> None:
|
|
84
|
+
self.delegate = delegate
|
|
85
|
+
self.duration_seconds = 0.0
|
|
86
|
+
|
|
87
|
+
def run(
|
|
88
|
+
self,
|
|
89
|
+
command: ToolCommand,
|
|
90
|
+
*,
|
|
91
|
+
timeout_seconds: float | None = None,
|
|
92
|
+
) -> ToolRunResult:
|
|
93
|
+
result = self.delegate.run(command, timeout_seconds=timeout_seconds)
|
|
94
|
+
self.duration_seconds += result.duration_seconds
|
|
95
|
+
return result
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def run_annotation(
|
|
99
|
+
*,
|
|
100
|
+
fasta_path: Path,
|
|
101
|
+
output_dir: Path,
|
|
102
|
+
adapter: AnnotationAdapter,
|
|
103
|
+
store: EvidenceStore,
|
|
104
|
+
runner: ToolRunner | None = None,
|
|
105
|
+
timeout_seconds: float | None = None,
|
|
106
|
+
threads: int = 1,
|
|
107
|
+
output_format: Literal["duckdb"] = "duckdb",
|
|
108
|
+
result_metadata: Mapping[str, str] | None = None,
|
|
109
|
+
) -> AnnotationSummary:
|
|
110
|
+
"""Resolve exact evidence, compute misses, and publish one DuckDB result."""
|
|
111
|
+
|
|
112
|
+
if threads < 1:
|
|
113
|
+
raise ValueError("threads must be positive")
|
|
114
|
+
if output_dir.exists():
|
|
115
|
+
raise OutputPackageError(f"output path already exists: {output_dir}")
|
|
116
|
+
if not output_dir.parent.is_dir():
|
|
117
|
+
raise OutputPackageError(
|
|
118
|
+
f"output parent directory does not exist: {output_dir.parent}"
|
|
119
|
+
)
|
|
120
|
+
|
|
121
|
+
started = time.perf_counter()
|
|
122
|
+
fasta_started = time.perf_counter()
|
|
123
|
+
fasta_stage_root = Path(
|
|
124
|
+
tempfile.mkdtemp(prefix=".seqevi-fasta-", dir=output_dir.parent)
|
|
125
|
+
)
|
|
126
|
+
stage = stage_fasta(fasta_path, fasta_stage_root)
|
|
127
|
+
fasta_staging_seconds = time.perf_counter() - fasta_started
|
|
128
|
+
try:
|
|
129
|
+
work_dir = Path(
|
|
130
|
+
tempfile.mkdtemp(prefix=".seqevi-annotate-", dir=output_dir.parent)
|
|
131
|
+
)
|
|
132
|
+
except Exception:
|
|
133
|
+
shutil.rmtree(stage.root, ignore_errors=True)
|
|
134
|
+
raise
|
|
135
|
+
lookup_batches = 0
|
|
136
|
+
commit_batches = 0
|
|
137
|
+
tool_batches = 0
|
|
138
|
+
store_lookup_seconds = 0.0
|
|
139
|
+
adapter_seconds = 0.0
|
|
140
|
+
store_commit_seconds = 0.0
|
|
141
|
+
try:
|
|
142
|
+
keys_by_sequence_id = {}
|
|
143
|
+
computed_ids: set[str] = set()
|
|
144
|
+
pending_misses: list[SequenceIdentity] = []
|
|
145
|
+
tool_runner = _MeasuringToolRunner(runner or ToolRunner())
|
|
146
|
+
|
|
147
|
+
for identity_batch in batched(iter_staged_identities(stage), _STORE_BATCH_SIZE):
|
|
148
|
+
queries = tuple(
|
|
149
|
+
EvidenceQuery(identity, adapter.contract.evidence_key(identity))
|
|
150
|
+
for identity in identity_batch
|
|
151
|
+
)
|
|
152
|
+
keys_by_sequence_id.update(
|
|
153
|
+
(query.identity.sequence_id, query.key) for query in queries
|
|
154
|
+
)
|
|
155
|
+
lookup_started = time.perf_counter()
|
|
156
|
+
cached = store.lookup_many(queries)
|
|
157
|
+
store_lookup_seconds += time.perf_counter() - lookup_started
|
|
158
|
+
lookup_batches += 1
|
|
159
|
+
for query in queries:
|
|
160
|
+
if query.key not in cached:
|
|
161
|
+
pending_misses.append(query.identity)
|
|
162
|
+
computed_ids.add(query.identity.sequence_id)
|
|
163
|
+
while len(pending_misses) >= _ANNOTATION_BATCH_SIZE:
|
|
164
|
+
annotation_identities = tuple(pending_misses[:_ANNOTATION_BATCH_SIZE])
|
|
165
|
+
del pending_misses[:_ANNOTATION_BATCH_SIZE]
|
|
166
|
+
tool_batches += 1
|
|
167
|
+
batch_metrics = _run_annotation_batch(
|
|
168
|
+
identities=annotation_identities,
|
|
169
|
+
batch_number=tool_batches,
|
|
170
|
+
work_dir=work_dir,
|
|
171
|
+
adapter=adapter,
|
|
172
|
+
store=store,
|
|
173
|
+
runner=tool_runner,
|
|
174
|
+
timeout_seconds=timeout_seconds,
|
|
175
|
+
threads=threads,
|
|
176
|
+
)
|
|
177
|
+
commit_batches += batch_metrics.commit_batches
|
|
178
|
+
adapter_seconds += batch_metrics.adapter_seconds
|
|
179
|
+
store_commit_seconds += batch_metrics.store_commit_seconds
|
|
180
|
+
|
|
181
|
+
if pending_misses:
|
|
182
|
+
tool_batches += 1
|
|
183
|
+
batch_metrics = _run_annotation_batch(
|
|
184
|
+
identities=tuple(pending_misses),
|
|
185
|
+
batch_number=tool_batches,
|
|
186
|
+
work_dir=work_dir,
|
|
187
|
+
adapter=adapter,
|
|
188
|
+
store=store,
|
|
189
|
+
runner=tool_runner,
|
|
190
|
+
timeout_seconds=timeout_seconds,
|
|
191
|
+
threads=threads,
|
|
192
|
+
)
|
|
193
|
+
commit_batches += batch_metrics.commit_batches
|
|
194
|
+
adapter_seconds += batch_metrics.adapter_seconds
|
|
195
|
+
store_commit_seconds += batch_metrics.store_commit_seconds
|
|
196
|
+
|
|
197
|
+
fetch_started = time.perf_counter()
|
|
198
|
+
fetched_by_key = store.fetch_many(keys_by_sequence_id.values())
|
|
199
|
+
store_fetch_seconds = time.perf_counter() - fetch_started
|
|
200
|
+
missing_keys = set(keys_by_sequence_id.values()) - fetched_by_key.keys()
|
|
201
|
+
if missing_keys:
|
|
202
|
+
raise AnnotationError(
|
|
203
|
+
f"Store did not expose {len(missing_keys)} terminal evidence records"
|
|
204
|
+
)
|
|
205
|
+
fetched_by_sequence_id = {
|
|
206
|
+
key.sequence_id: fetched for key, fetched in fetched_by_key.items()
|
|
207
|
+
}
|
|
208
|
+
source_by_sequence_id = {
|
|
209
|
+
sequence_id: (
|
|
210
|
+
EvidenceSource.COMPUTED
|
|
211
|
+
if sequence_id in computed_ids
|
|
212
|
+
else EvidenceSource.CACHE
|
|
213
|
+
)
|
|
214
|
+
for sequence_id in keys_by_sequence_id
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
statuses = [
|
|
218
|
+
fetched.record.status for fetched in fetched_by_sequence_id.values()
|
|
219
|
+
]
|
|
220
|
+
package_started = time.perf_counter()
|
|
221
|
+
if output_format != "duckdb":
|
|
222
|
+
raise AnnotationError(f"unsupported output format: {output_format}")
|
|
223
|
+
metadata = dict(result_metadata or _default_result_metadata(adapter))
|
|
224
|
+
metadata["InputDigest"] = stage.input_digest
|
|
225
|
+
metadata.setdefault("CreatedAt", datetime.now(UTC).isoformat())
|
|
226
|
+
materialize_result_database(
|
|
227
|
+
output_path=output_dir,
|
|
228
|
+
records=iter_staged_records(stage),
|
|
229
|
+
input_record_count=stage.input_records,
|
|
230
|
+
fetched_by_sequence_id=fetched_by_sequence_id,
|
|
231
|
+
source_by_sequence_id=source_by_sequence_id,
|
|
232
|
+
evidence_schema=adapter.evidence_schema,
|
|
233
|
+
adapter_contract=adapter.contract,
|
|
234
|
+
input_digest=stage.input_digest,
|
|
235
|
+
metadata=metadata,
|
|
236
|
+
run_metrics={
|
|
237
|
+
"input_records": stage.input_records,
|
|
238
|
+
"unique_sequences": stage.unique_sequences,
|
|
239
|
+
"cache_hits": stage.unique_sequences - len(computed_ids),
|
|
240
|
+
"computed": len(computed_ids),
|
|
241
|
+
"hits": statuses.count(EvidenceStatus.HIT),
|
|
242
|
+
"no_hits": statuses.count(EvidenceStatus.NO_HIT),
|
|
243
|
+
},
|
|
244
|
+
)
|
|
245
|
+
package_seconds = time.perf_counter() - package_started
|
|
246
|
+
except Exception as error:
|
|
247
|
+
raise AnnotationError(
|
|
248
|
+
f"annotation failed; diagnostics retained at {work_dir}: {error}"
|
|
249
|
+
) from error
|
|
250
|
+
else:
|
|
251
|
+
shutil.rmtree(work_dir, ignore_errors=True)
|
|
252
|
+
finally:
|
|
253
|
+
shutil.rmtree(stage.root, ignore_errors=True)
|
|
254
|
+
|
|
255
|
+
statuses = [fetched.record.status for fetched in fetched_by_sequence_id.values()]
|
|
256
|
+
artifact_digests = {
|
|
257
|
+
digest
|
|
258
|
+
for fetched in fetched_by_sequence_id.values()
|
|
259
|
+
for digest in (
|
|
260
|
+
fetched.record.normalized_artifact_digest,
|
|
261
|
+
fetched.record.raw_artifact_digest,
|
|
262
|
+
)
|
|
263
|
+
if digest is not None
|
|
264
|
+
}
|
|
265
|
+
return AnnotationSummary(
|
|
266
|
+
input_records=stage.input_records,
|
|
267
|
+
unique_sequences=stage.unique_sequences,
|
|
268
|
+
cache_hits=stage.unique_sequences - len(computed_ids),
|
|
269
|
+
computed=len(computed_ids),
|
|
270
|
+
hits=statuses.count(EvidenceStatus.HIT),
|
|
271
|
+
no_hits=statuses.count(EvidenceStatus.NO_HIT),
|
|
272
|
+
output_dir=output_dir,
|
|
273
|
+
metrics=AnnotationMetrics(
|
|
274
|
+
elapsed_seconds=time.perf_counter() - started,
|
|
275
|
+
fasta_staging_seconds=fasta_staging_seconds,
|
|
276
|
+
store_lookup_seconds=store_lookup_seconds,
|
|
277
|
+
adapter_seconds=adapter_seconds,
|
|
278
|
+
external_tool_seconds=tool_runner.duration_seconds,
|
|
279
|
+
store_commit_seconds=store_commit_seconds,
|
|
280
|
+
store_fetch_seconds=store_fetch_seconds,
|
|
281
|
+
package_seconds=package_seconds,
|
|
282
|
+
peak_rss_kib=_peak_rss_kib(),
|
|
283
|
+
store_lookup_batches=lookup_batches,
|
|
284
|
+
store_commit_batches=commit_batches,
|
|
285
|
+
store_fetch_batches=1,
|
|
286
|
+
tool_batches=tool_batches,
|
|
287
|
+
unique_artifact_reads=len(artifact_digests),
|
|
288
|
+
configured_threads=threads,
|
|
289
|
+
),
|
|
290
|
+
)
|
|
291
|
+
|
|
292
|
+
|
|
293
|
+
def _default_result_metadata(adapter: AnnotationAdapter) -> dict[str, str]:
|
|
294
|
+
"""Build a complete generic result identity for direct orchestration calls."""
|
|
295
|
+
|
|
296
|
+
adapter_name = adapter.contract.name
|
|
297
|
+
if adapter_name == "eggnog":
|
|
298
|
+
upstream_tool = "eggNOG-mapper"
|
|
299
|
+
result_schema = "eggnog-mapper/2"
|
|
300
|
+
elif adapter_name == "interpro-pfam":
|
|
301
|
+
upstream_tool = "InterProScan"
|
|
302
|
+
result_schema = "interproscan-pfam/5"
|
|
303
|
+
elif adapter_name == "dbcan-cazyme":
|
|
304
|
+
upstream_tool = "dbCAN"
|
|
305
|
+
result_schema = "dbcan-cazyme/5"
|
|
306
|
+
else:
|
|
307
|
+
upstream_tool = adapter_name
|
|
308
|
+
result_schema = f"{adapter_name}/1"
|
|
309
|
+
return {
|
|
310
|
+
"ResultFormatVersion": RESULT_FORMAT_VERSION,
|
|
311
|
+
"ResultSchemaID": result_schema,
|
|
312
|
+
"SeqEviVersion": __version__,
|
|
313
|
+
"Adapter": adapter_name,
|
|
314
|
+
"AdapterContractVersion": adapter.contract.version,
|
|
315
|
+
"UpstreamTool": upstream_tool,
|
|
316
|
+
"UpstreamToolVersion": "unknown",
|
|
317
|
+
"ToolRuntimeDigest": adapter.contract.tool_runtime_digest,
|
|
318
|
+
"ResourceID": adapter.contract.resource_id,
|
|
319
|
+
"InputDigest": "unknown",
|
|
320
|
+
"CreatedAt": datetime.now(UTC).isoformat(),
|
|
321
|
+
}
|
|
322
|
+
|
|
323
|
+
|
|
324
|
+
def _peak_rss_kib() -> int | None:
|
|
325
|
+
try:
|
|
326
|
+
import resource
|
|
327
|
+
except ImportError:
|
|
328
|
+
return None
|
|
329
|
+
peak = resource.getrusage(resource.RUSAGE_SELF).ru_maxrss
|
|
330
|
+
return peak // 1024 if sys.platform == "darwin" else peak
|
|
331
|
+
|
|
332
|
+
|
|
333
|
+
def _run_annotation_batch(
|
|
334
|
+
*,
|
|
335
|
+
identities: tuple[SequenceIdentity, ...],
|
|
336
|
+
batch_number: int,
|
|
337
|
+
work_dir: Path,
|
|
338
|
+
adapter: AnnotationAdapter,
|
|
339
|
+
store: EvidenceStore,
|
|
340
|
+
runner: ToolRunner,
|
|
341
|
+
timeout_seconds: float | None,
|
|
342
|
+
threads: int,
|
|
343
|
+
) -> _BatchMetrics:
|
|
344
|
+
batch_dir = work_dir / f"batch-{batch_number:06d}"
|
|
345
|
+
batch_dir.mkdir()
|
|
346
|
+
misses_fasta = batch_dir / "cache-misses.fasta"
|
|
347
|
+
with misses_fasta.open("w", encoding="ascii", newline="\n") as handle:
|
|
348
|
+
handle.writelines(iter_fasta_lines(identities))
|
|
349
|
+
adapter_started = time.perf_counter()
|
|
350
|
+
batch = adapter.run_batch(
|
|
351
|
+
identities=identities,
|
|
352
|
+
input_fasta=misses_fasta,
|
|
353
|
+
work_dir=batch_dir,
|
|
354
|
+
runner=runner,
|
|
355
|
+
timeout_seconds=timeout_seconds,
|
|
356
|
+
threads=threads,
|
|
357
|
+
)
|
|
358
|
+
adapter_seconds = time.perf_counter() - adapter_started
|
|
359
|
+
commits = _build_commits(
|
|
360
|
+
batch=batch,
|
|
361
|
+
identities=identities,
|
|
362
|
+
adapter=adapter,
|
|
363
|
+
)
|
|
364
|
+
batches = 0
|
|
365
|
+
commit_started = time.perf_counter()
|
|
366
|
+
for commit_batch in batched(commits, _STORE_BATCH_SIZE):
|
|
367
|
+
store.commit_many(commit_batch)
|
|
368
|
+
batches += 1
|
|
369
|
+
return _BatchMetrics(
|
|
370
|
+
commit_batches=batches,
|
|
371
|
+
adapter_seconds=adapter_seconds,
|
|
372
|
+
store_commit_seconds=time.perf_counter() - commit_started,
|
|
373
|
+
)
|
|
374
|
+
|
|
375
|
+
|
|
376
|
+
def _build_commits(
|
|
377
|
+
*,
|
|
378
|
+
batch: AdapterBatchResult,
|
|
379
|
+
identities: tuple[SequenceIdentity, ...],
|
|
380
|
+
adapter: AnnotationAdapter,
|
|
381
|
+
) -> tuple[EvidenceCommit, ...]:
|
|
382
|
+
identity_by_sequence_id = {
|
|
383
|
+
identity.sequence_id: identity for identity in identities
|
|
384
|
+
}
|
|
385
|
+
result_by_sequence_id = {result.sequence_id: result for result in batch.sequences}
|
|
386
|
+
expected = set(identity_by_sequence_id)
|
|
387
|
+
observed = set(result_by_sequence_id)
|
|
388
|
+
if expected != observed:
|
|
389
|
+
missing = ", ".join(sorted(expected - observed)) or "<none>"
|
|
390
|
+
extra = ", ".join(sorted(observed - expected)) or "<none>"
|
|
391
|
+
raise AnnotationError(
|
|
392
|
+
f"adapter sequence accounting mismatch; missing: {missing}; extra: {extra}"
|
|
393
|
+
)
|
|
394
|
+
|
|
395
|
+
commits = []
|
|
396
|
+
for sequence_id in sorted(expected):
|
|
397
|
+
identity = identity_by_sequence_id[sequence_id]
|
|
398
|
+
result = result_by_sequence_id[sequence_id]
|
|
399
|
+
commits.append(
|
|
400
|
+
EvidenceCommit(
|
|
401
|
+
identity=identity,
|
|
402
|
+
key=adapter.contract.evidence_key(identity),
|
|
403
|
+
status=result.status,
|
|
404
|
+
payload_digest=result.payload_digest,
|
|
405
|
+
normalized_artifact=(
|
|
406
|
+
batch.normalized_artifact
|
|
407
|
+
if result.status is EvidenceStatus.HIT
|
|
408
|
+
else None
|
|
409
|
+
),
|
|
410
|
+
raw_artifact=batch.raw_artifact,
|
|
411
|
+
)
|
|
412
|
+
)
|
|
413
|
+
return tuple(commits)
|