gapit 0.2.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gapit/__init__.py +3 -0
- gapit/blast.py +239 -0
- gapit/cli.py +128 -0
- gapit/cmd_db.py +113 -0
- gapit/cmd_db_build.py +72 -0
- gapit/cmd_db_install.py +126 -0
- gapit/cmd_db_outdated.py +51 -0
- gapit/cmd_db_search.py +66 -0
- gapit/cmd_screen.py +214 -0
- gapit/cmd_summary.py +65 -0
- gapit/config.py +51 -0
- gapit/data/snapshots/card.tar.gz +0 -0
- gapit/data/snapshots/vfdb.tar.gz +0 -0
- gapit/db.py +226 -0
- gapit/db_build_ops.py +210 -0
- gapit/db_ops.py +128 -0
- gapit/db_query_ops.py +252 -0
- gapit/dbbuild.py +207 -0
- gapit/dbcodec.py +117 -0
- gapit/dispatch.py +31 -0
- gapit/errors.py +87 -0
- gapit/fasta.py +123 -0
- gapit/formats/__init__.py +1 -0
- gapit/formats/json.py +309 -0
- gapit/formats/md.py +190 -0
- gapit/formats/schemas.py +30 -0
- gapit/formats/summary.py +103 -0
- gapit/formats/tsv.py +45 -0
- gapit/hits.py +107 -0
- gapit/mcp.py +158 -0
- gapit/mcp_schemas.py +123 -0
- gapit/mcp_tools.py +289 -0
- gapit/minimap.py +20 -0
- gapit/minimap2_run.py +114 -0
- gapit/paf.py +115 -0
- gapit/proctools.py +24 -0
- gapit/providers/__init__.py +39 -0
- gapit/providers/argannot.py +94 -0
- gapit/providers/bacmet2.py +59 -0
- gapit/providers/card.py +150 -0
- gapit/providers/common.py +245 -0
- gapit/providers/ecoh.py +63 -0
- gapit/providers/ecoli_vf.py +74 -0
- gapit/providers/megares.py +71 -0
- gapit/providers/ncbi.py +103 -0
- gapit/providers/plasmidfinder.py +69 -0
- gapit/providers/resfinder.py +123 -0
- gapit/providers/snapshots.py +119 -0
- gapit/providers/upec_expec_vf.py +85 -0
- gapit/providers/vfdb.py +92 -0
- gapit/providers/victors.py +109 -0
- gapit/py.typed +0 -0
- gapit/reads.py +221 -0
- gapit/records.py +152 -0
- gapit/report.py +25 -0
- gapit/screening.py +145 -0
- gapit/screening_reads.py +255 -0
- gapit/seqconvert.py +203 -0
- gapit/summary.py +151 -0
- gapit-0.2.2.dist-info/METADATA +183 -0
- gapit-0.2.2.dist-info/RECORD +64 -0
- gapit-0.2.2.dist-info/WHEEL +4 -0
- gapit-0.2.2.dist-info/entry_points.txt +3 -0
- gapit-0.2.2.dist-info/licenses/LICENSE +21 -0
gapit/minimap.py
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
"""COVERAGE_MAP minimap — exact abricate arithmetic (SPEC.md §4)."""
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
def minimap(x: int, y: int, length: int, broken: int) -> str:
|
|
5
|
+
"""15-char coverage bar: '=' over [x, y] scaled onto the map width.
|
|
6
|
+
|
|
7
|
+
Replicates the Perl int() truncation quirks verbatim — do not "fix" the
|
|
8
|
+
1-based coordinate artifacts (yi may exceed width-1). ``broken`` is the
|
|
9
|
+
gap-open count; a broken map is 14 boxes with '/' after box width//2.
|
|
10
|
+
"""
|
|
11
|
+
width = 15 - (1 if broken > 0 else 0)
|
|
12
|
+
scale = length / width
|
|
13
|
+
xi = int(x / scale)
|
|
14
|
+
yi = int(y / scale)
|
|
15
|
+
chars: list[str] = []
|
|
16
|
+
for i in range(width):
|
|
17
|
+
chars.append("=" if xi <= i <= yi else ".")
|
|
18
|
+
if broken > 0 and i == width // 2:
|
|
19
|
+
chars.append("/")
|
|
20
|
+
return "".join(chars)
|
gapit/minimap2_run.py
ADDED
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
"""The minimap2 invocation layer for reads mode (SPEC.md §10).
|
|
2
|
+
|
|
3
|
+
Split from reads.py (which was over the 250 pure-LOC ceiling) so this module
|
|
4
|
+
owns exactly one thing: running minimap2 and streaming its PAF rows. The
|
|
5
|
+
minimap2 invocation's own preset vocabulary (``ReadType``) lives in paf.py,
|
|
6
|
+
the lowest layer of the reads stack. Not to be confused with minimap.py,
|
|
7
|
+
which owns the blastn-side COVERAGE_MAP arithmetic.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import shlex
|
|
11
|
+
import subprocess
|
|
12
|
+
import sys
|
|
13
|
+
import threading
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
|
|
16
|
+
from gapit.db import Database
|
|
17
|
+
from gapit.errors import DependencyError, GapitError
|
|
18
|
+
from gapit.paf import PafRecord, ReadType, parse_paf_row
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _stream_minimap2(argv: list[str], r1: Path) -> list[PafRecord]:
|
|
22
|
+
"""Run one minimap2 invocation, streaming PAF rows off stdout as lines
|
|
23
|
+
arrive (no full-PAF materialization). stderr is drained by a background
|
|
24
|
+
thread: an undrained stderr pipe fills (~64 KB) and blocks the child
|
|
25
|
+
mid-run. Returncode != 0 raises MINIMAP2_FAILED with the captured
|
|
26
|
+
stderr text."""
|
|
27
|
+
try:
|
|
28
|
+
process = subprocess.Popen(
|
|
29
|
+
argv,
|
|
30
|
+
stdin=subprocess.DEVNULL,
|
|
31
|
+
stdout=subprocess.PIPE,
|
|
32
|
+
stderr=subprocess.PIPE,
|
|
33
|
+
text=True,
|
|
34
|
+
)
|
|
35
|
+
except FileNotFoundError as exc:
|
|
36
|
+
raise DependencyError(
|
|
37
|
+
"required binary not found on PATH: minimap2",
|
|
38
|
+
code="MISSING_DEPENDENCY",
|
|
39
|
+
context={"binary": "minimap2"},
|
|
40
|
+
) from exc
|
|
41
|
+
stderr_chunks: list[str] = []
|
|
42
|
+
|
|
43
|
+
def drain() -> None:
|
|
44
|
+
if process.stderr is not None:
|
|
45
|
+
stderr_chunks.append(process.stderr.read())
|
|
46
|
+
|
|
47
|
+
thread = threading.Thread(target=drain)
|
|
48
|
+
thread.start()
|
|
49
|
+
rows: list[PafRecord] = []
|
|
50
|
+
try:
|
|
51
|
+
if process.stdout is not None:
|
|
52
|
+
for line in process.stdout:
|
|
53
|
+
stripped = line.rstrip("\n")
|
|
54
|
+
if stripped.strip():
|
|
55
|
+
rows.append(parse_paf_row(stripped))
|
|
56
|
+
finally:
|
|
57
|
+
if process.stdout is not None:
|
|
58
|
+
process.stdout.close() # on a parse error, SIGPIPE stops the child
|
|
59
|
+
process.wait()
|
|
60
|
+
thread.join()
|
|
61
|
+
if process.returncode != 0:
|
|
62
|
+
raise GapitError(
|
|
63
|
+
f"minimap2 failed: {''.join(stderr_chunks).strip()}",
|
|
64
|
+
code="MINIMAP2_FAILED",
|
|
65
|
+
context={"binary": "minimap2", "file": str(r1)},
|
|
66
|
+
)
|
|
67
|
+
return rows
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def run_minimap2(
|
|
71
|
+
lanes: list[tuple[Path, Path | None]],
|
|
72
|
+
database: Database,
|
|
73
|
+
*,
|
|
74
|
+
read_type: ReadType,
|
|
75
|
+
threads: int,
|
|
76
|
+
debug: bool = False,
|
|
77
|
+
nm_tags: bool = False,
|
|
78
|
+
) -> list[PafRecord]:
|
|
79
|
+
"""Run one minimap2 invocation per lane (PAF on stdout) and concatenate
|
|
80
|
+
the rows, streaming each lane's output row by row. The index argument is
|
|
81
|
+
always the ``sequences`` FASTA — minimap2 indexes it in memory with the
|
|
82
|
+
invocation preset's own parameters. A persisted ``.mmi`` is deliberately
|
|
83
|
+
rejected even when one sits beside the FASTA: a default-built index
|
|
84
|
+
overrides the ``-x`` preset's indexing parameters (``-k, -w or -H
|
|
85
|
+
overridden by prebuilt index``), which misassigns close homologs and
|
|
86
|
+
benchmarks slower than in-memory indexing (2026-09-19: blaCTX-M/blaSHV
|
|
87
|
+
allele divergence, +1.2 s on the ncbi db). minimap2's pairing semantics
|
|
88
|
+
for >2 input files are undocumented; per-lane runs (r1[i] alone or with
|
|
89
|
+
its mate r2[i]) are deterministic. With ``nm_tags``, ``--cs`` is added so
|
|
90
|
+
rows carry ``NM:i:`` (minimap2 omits NM from PAF output without it; the
|
|
91
|
+
alignments themselves are unchanged) — used by gapit.reads/2 identity
|
|
92
|
+
filtering. With ``debug``, echo each argv to stderr (abricate --debug
|
|
93
|
+
parity)."""
|
|
94
|
+
rows: list[PafRecord] = []
|
|
95
|
+
for r1, r2 in lanes:
|
|
96
|
+
argv = [
|
|
97
|
+
"minimap2",
|
|
98
|
+
"-x",
|
|
99
|
+
read_type,
|
|
100
|
+
"-t",
|
|
101
|
+
str(threads),
|
|
102
|
+
]
|
|
103
|
+
if nm_tags:
|
|
104
|
+
argv.append("--cs")
|
|
105
|
+
# Input paths go to argv absolutized: an absolute path starts with
|
|
106
|
+
# "/" and can never parse as a minimap2 option, so a query file
|
|
107
|
+
# named "-d" cannot make minimap2 dump an index over its mate path.
|
|
108
|
+
argv += [str(database.sequences_path.absolute()), str(r1.absolute())]
|
|
109
|
+
if r2 is not None:
|
|
110
|
+
argv.append(str(r2.absolute()))
|
|
111
|
+
if debug:
|
|
112
|
+
print(f"gapit: run: {shlex.join(argv)}", file=sys.stderr)
|
|
113
|
+
rows.extend(_stream_minimap2(argv, r1))
|
|
114
|
+
return rows
|
gapit/paf.py
ADDED
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
"""PAF row parsing and interval arithmetic (minimap2 output boundary).
|
|
2
|
+
|
|
3
|
+
Typed parsing of minimap2's PAF rows (12 required fields + tags), the
|
|
4
|
+
half-open interval math used to aggregate alignment spans into coverage, and
|
|
5
|
+
the per-alignment identity rule behind gapit.reads/2 filtering.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from collections.abc import Iterable
|
|
9
|
+
from typing import Literal
|
|
10
|
+
|
|
11
|
+
from pydantic import BaseModel
|
|
12
|
+
|
|
13
|
+
from gapit.errors import GapitError
|
|
14
|
+
|
|
15
|
+
ReadType = Literal["sr", "map-ont", "map-hifi"]
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class PafRecord(BaseModel, frozen=True):
|
|
19
|
+
"""One PAF alignment row: 12 required fields + primary flag (tp:A:P) and
|
|
20
|
+
mismatch count (NM:i:, None when minimap2 emitted no NM tag — it only
|
|
21
|
+
does with ``--cs``)."""
|
|
22
|
+
|
|
23
|
+
qname: str
|
|
24
|
+
qlen: int
|
|
25
|
+
qstart: int
|
|
26
|
+
qend: int
|
|
27
|
+
strand: Literal["+", "-"]
|
|
28
|
+
tname: str
|
|
29
|
+
tlen: int
|
|
30
|
+
tstart: int
|
|
31
|
+
tend: int
|
|
32
|
+
nmatch: int
|
|
33
|
+
alen: int
|
|
34
|
+
mapq: int
|
|
35
|
+
is_primary: bool = True
|
|
36
|
+
nm: int | None = None
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def parse_paf_row(line: str) -> PafRecord:
|
|
40
|
+
"""Parse one tab-delimited PAF line; <12 fields is a hard error. Records
|
|
41
|
+
without a tp tag count as primary; NM:i: is extracted when present."""
|
|
42
|
+
fields = line.split("\t")
|
|
43
|
+
if len(fields) < 12:
|
|
44
|
+
raise GapitError(
|
|
45
|
+
f"can not parse PAF row (expected >= 12 fields): {line!r}",
|
|
46
|
+
code="PAF_PARSE_FAILED",
|
|
47
|
+
)
|
|
48
|
+
is_primary = True
|
|
49
|
+
nm: int | None = None
|
|
50
|
+
for tag in fields[12:]:
|
|
51
|
+
if tag.startswith("tp:A:"):
|
|
52
|
+
is_primary = tag == "tp:A:P"
|
|
53
|
+
elif tag.startswith("NM:i:"):
|
|
54
|
+
nm = int(tag[len("NM:i:") :])
|
|
55
|
+
strand = fields[4]
|
|
56
|
+
if strand not in ("+", "-"):
|
|
57
|
+
raise GapitError(
|
|
58
|
+
f"can not parse PAF strand (expected + or -): {strand!r}",
|
|
59
|
+
code="PAF_PARSE_FAILED",
|
|
60
|
+
)
|
|
61
|
+
# model_construct: the manual int()/strand checks above already guarantee
|
|
62
|
+
# the field types; re-validating per PAF row would be redundant.
|
|
63
|
+
return PafRecord.model_construct(
|
|
64
|
+
qname=fields[0],
|
|
65
|
+
qlen=int(fields[1]),
|
|
66
|
+
qstart=int(fields[2]),
|
|
67
|
+
qend=int(fields[3]),
|
|
68
|
+
strand=strand,
|
|
69
|
+
tname=fields[5],
|
|
70
|
+
tlen=int(fields[6]),
|
|
71
|
+
tstart=int(fields[7]),
|
|
72
|
+
tend=int(fields[8]),
|
|
73
|
+
nmatch=int(fields[9]),
|
|
74
|
+
alen=int(fields[10]),
|
|
75
|
+
mapq=int(fields[11]),
|
|
76
|
+
is_primary=is_primary,
|
|
77
|
+
nm=nm,
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def alignment_identity(record: PafRecord) -> float:
|
|
82
|
+
"""Per-alignment identity for gapit.reads/2: ``100 * (alen - nm) / alen``
|
|
83
|
+
over the block length (column 10) and the NM tag (mismatches + gaps per
|
|
84
|
+
the PAF spec). A row without NM cannot be assessed and counts as 100.0;
|
|
85
|
+
a degenerate zero-length block also counts as 100.0 (no division)."""
|
|
86
|
+
if record.nm is None or record.alen <= 0:
|
|
87
|
+
return 100.0
|
|
88
|
+
return 100.0 * (record.alen - record.nm) / record.alen
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def filter_alignments(
|
|
92
|
+
rows: Iterable[PafRecord], *, min_identity: float = 0.0, min_mapq: int = 0
|
|
93
|
+
) -> list[PafRecord]:
|
|
94
|
+
"""gapit.reads/2 row filter: keep alignments with identity >= min_identity
|
|
95
|
+
and mapq >= min_mapq (applied after the primary-only rule upstream; both
|
|
96
|
+
thresholds default off, in which case every row is kept verbatim — the
|
|
97
|
+
gapit.reads/1 path must stay byte-identical, even for pathological rows
|
|
98
|
+
whose arithmetic identity is negative)."""
|
|
99
|
+
if min_identity <= 0.0 and min_mapq <= 0:
|
|
100
|
+
return list(rows)
|
|
101
|
+
return [row for row in rows if row.mapq >= min_mapq and alignment_identity(row) >= min_identity]
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def union_length(intervals: Iterable[tuple[int, int]]) -> int:
|
|
105
|
+
"""Total length of the union of half-open [start, end) intervals:
|
|
106
|
+
sort + linear sweep. Identical arithmetic to counting positions whose
|
|
107
|
+
per-base depth is > 0, without materializing any per-base array."""
|
|
108
|
+
covered = 0
|
|
109
|
+
reach = -1
|
|
110
|
+
for start, end in sorted(intervals):
|
|
111
|
+
if end <= reach:
|
|
112
|
+
continue
|
|
113
|
+
covered += end - max(start, reach)
|
|
114
|
+
reach = end
|
|
115
|
+
return covered
|
gapit/proctools.py
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
"""Shared external-tool plumbing: the argv subprocess runner and stderr notes."""
|
|
2
|
+
|
|
3
|
+
import subprocess
|
|
4
|
+
import sys
|
|
5
|
+
|
|
6
|
+
from gapit.errors import DependencyError
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def run_tool(argv: list[str]) -> subprocess.CompletedProcess[str]:
|
|
10
|
+
"""Run an external tool with an argv list (never a shell)."""
|
|
11
|
+
try:
|
|
12
|
+
return subprocess.run(argv, check=False, capture_output=True, text=True)
|
|
13
|
+
except FileNotFoundError as exc:
|
|
14
|
+
raise DependencyError(
|
|
15
|
+
f"required binary not found on PATH: {argv[0]}",
|
|
16
|
+
code="MISSING_DEPENDENCY",
|
|
17
|
+
context={"binary": argv[0]},
|
|
18
|
+
) from exc
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def note(quiet: bool, message: str) -> None:
|
|
22
|
+
"""Per-step progress on stderr when quiet is disabled (stdout stays pure)."""
|
|
23
|
+
if not quiet:
|
|
24
|
+
print(f"gapit: {message}", file=sys.stderr)
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
"""Database providers: the Provider contract and per-provider modules (Wave B).
|
|
2
|
+
|
|
3
|
+
Each concrete provider (Wave B1..B12) is a module exporting a ``Provider``
|
|
4
|
+
value consumed by :func:`gapit.providers.common.fetch_provider`.
|
|
5
|
+
``REGISTRY`` maps every provider name to its ``Provider`` — the CLI's
|
|
6
|
+
``gapit db fetch`` / ``gapit db list`` lookup table (Wave C-a). Imports are
|
|
7
|
+
static and alphabetized; no dynamic import magic.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from gapit.providers import (
|
|
11
|
+
argannot,
|
|
12
|
+
bacmet2,
|
|
13
|
+
card,
|
|
14
|
+
ecoh,
|
|
15
|
+
ecoli_vf,
|
|
16
|
+
megares,
|
|
17
|
+
ncbi,
|
|
18
|
+
plasmidfinder,
|
|
19
|
+
resfinder,
|
|
20
|
+
upec_expec_vf,
|
|
21
|
+
vfdb,
|
|
22
|
+
victors,
|
|
23
|
+
)
|
|
24
|
+
from gapit.providers.common import Provider
|
|
25
|
+
|
|
26
|
+
REGISTRY: dict[str, Provider] = {
|
|
27
|
+
argannot.PROVIDER.name: argannot.PROVIDER,
|
|
28
|
+
bacmet2.PROVIDER.name: bacmet2.PROVIDER,
|
|
29
|
+
card.PROVIDER.name: card.PROVIDER,
|
|
30
|
+
ecoh.PROVIDER.name: ecoh.PROVIDER,
|
|
31
|
+
ecoli_vf.PROVIDER.name: ecoli_vf.PROVIDER,
|
|
32
|
+
megares.PROVIDER.name: megares.PROVIDER,
|
|
33
|
+
ncbi.PROVIDER.name: ncbi.PROVIDER,
|
|
34
|
+
plasmidfinder.PROVIDER.name: plasmidfinder.PROVIDER,
|
|
35
|
+
resfinder.PROVIDER.name: resfinder.PROVIDER,
|
|
36
|
+
upec_expec_vf.PROVIDER.name: upec_expec_vf.PROVIDER,
|
|
37
|
+
vfdb.PROVIDER.name: vfdb.PROVIDER,
|
|
38
|
+
victors.PROVIDER.name: victors.PROVIDER,
|
|
39
|
+
}
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
"""ARG-ANNOT provider (acquired resistance genes) — transform only.
|
|
2
|
+
|
|
3
|
+
Upstream ``get_argannot`` (abricate-get_db 1.4.0) downloads
|
|
4
|
+
``ARG-ANNOT_NT_V6_July2019.txt`` — a nucleotide FASTA whose raw text is
|
|
5
|
+
littered with stray backslashes (upstream calls this "fix syntax errors in
|
|
6
|
+
the FASTA file"), so every backslash is stripped BEFORE parsing. Headers
|
|
7
|
+
look like::
|
|
8
|
+
|
|
9
|
+
>(AGly)aac2-Ie:NC_011896:3039059-3039607:549
|
|
10
|
+
|
|
11
|
+
The id token splits on ``:``: gene is ``x[0]``, accession is ``x[1]:x[2]``
|
|
12
|
+
(later fields, e.g. the trailing length, are ignored — upstream never reads
|
|
13
|
+
``x[3]``). Product falls back to the gene when the header carries no
|
|
14
|
+
description (save_fasta's ``DESC || ID``); real ARG-ANNOT headers carry
|
|
15
|
+
none. Records with fewer than 3 colon-separated fields are SKIPPED
|
|
16
|
+
(documented deviation): upstream perl would splice an undefined ``x[2]``
|
|
17
|
+
into the accession; a typed Record refuses to carry that.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from collections.abc import Iterator
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
from tempfile import NamedTemporaryFile
|
|
23
|
+
|
|
24
|
+
from gapit.fasta import iter_fasta
|
|
25
|
+
from gapit.providers.common import Provider
|
|
26
|
+
from gapit.records import Record
|
|
27
|
+
|
|
28
|
+
NAME = "argannot"
|
|
29
|
+
SOURCE_FILE = "ARG-ANNOT_NT_V6_July2019.txt"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def transform(workdir: Path) -> Iterator[Record]:
|
|
33
|
+
"""Yield Records from ``workdir/ARG-ANNOT_NT_V6_July2019.txt``.
|
|
34
|
+
|
|
35
|
+
The raw text is repaired first — every backslash stripped, upstream's
|
|
36
|
+
global ``s/\\\\//g`` — and written to a scratch file inside the workdir
|
|
37
|
+
so :func:`gapit.fasta.iter_fasta` can parse it (it takes a path). The
|
|
38
|
+
scratch file is removed on completion or early generator close.
|
|
39
|
+
Sequence arrives raw — fetch_provider N-normalizes it for the nucl
|
|
40
|
+
dbtype.
|
|
41
|
+
"""
|
|
42
|
+
repaired = (workdir / SOURCE_FILE).read_text(encoding="utf-8").replace("\\", "")
|
|
43
|
+
with NamedTemporaryFile(
|
|
44
|
+
mode="w",
|
|
45
|
+
encoding="utf-8",
|
|
46
|
+
newline="\n",
|
|
47
|
+
prefix=".argannot.",
|
|
48
|
+
suffix=".txt",
|
|
49
|
+
dir=workdir,
|
|
50
|
+
delete=False,
|
|
51
|
+
) as handle:
|
|
52
|
+
handle.write(repaired)
|
|
53
|
+
scratch = Path(handle.name)
|
|
54
|
+
try:
|
|
55
|
+
for fasta in iter_fasta(scratch):
|
|
56
|
+
parts = fasta.id.split(":")
|
|
57
|
+
if len(parts) < 3:
|
|
58
|
+
continue
|
|
59
|
+
gene = parts[0]
|
|
60
|
+
yield Record(
|
|
61
|
+
db=NAME,
|
|
62
|
+
gene=gene,
|
|
63
|
+
sequence=fasta.sequence,
|
|
64
|
+
accession=":".join(parts[1:3]),
|
|
65
|
+
function=(),
|
|
66
|
+
product=fasta.description or gene,
|
|
67
|
+
source_id=fasta.id,
|
|
68
|
+
)
|
|
69
|
+
finally:
|
|
70
|
+
scratch.unlink(missing_ok=True)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
PROVIDER = Provider(
|
|
74
|
+
name=NAME,
|
|
75
|
+
description="ARG-ANNOT acquired resistance genes",
|
|
76
|
+
# Upstream 301-migrated; the deep link 404s on both domains (verified 2026-09-17).
|
|
77
|
+
# Wayback CDX: 2020-06-26 + 2026-01-14 captures share digest 5ZDVKAPZ4ZGIUDSYFVTFUYYY4CS5MUXZ
|
|
78
|
+
# (2024 differs: suspect partial). The 2026-01-14 memento intermittently serves a 9KB
|
|
79
|
+
# Wayback outage interstitial (LoadShardBlock datanode, observed 2026-09-18); the
|
|
80
|
+
# 2020-06-26 memento probed good (text/plain, 2,139,519 bytes, same stable digest
|
|
81
|
+
# family). Snapshot pinned, not latest; veto to upstream if it returns.
|
|
82
|
+
# The id_ suffix requests the original WARC bytes with no Wayback rewrite/redirect
|
|
83
|
+
# layer — the right form for machine fetching: verified 2026-09-18 (200, text/plain,
|
|
84
|
+
# 2139519B, FASTA head ">(AGly)aac:AJ628983:1985-2539:555") while the plain
|
|
85
|
+
# memento form flapped (interstitial twice, 45s apart).
|
|
86
|
+
source_urls=(
|
|
87
|
+
"https://web.archive.org/web/20200626214628id_/"
|
|
88
|
+
"https://www.mediterranee-infection.com/wp-content/uploads/2019/09/"
|
|
89
|
+
"ARG-ANNOT_NT_V6_July2019.txt",
|
|
90
|
+
),
|
|
91
|
+
dbtype="nucl",
|
|
92
|
+
transform=transform,
|
|
93
|
+
snapshot=None,
|
|
94
|
+
)
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
"""BacMet2 provider (experimentally confirmed, PROTEIN) — transform only.
|
|
2
|
+
|
|
3
|
+
Upstream ``get_bacmet2`` (abricate-get_db 1.4.0) downloads
|
|
4
|
+
``BacMet2_EXP_database.fasta`` — a protein file, hence the set's one
|
|
5
|
+
``dbtype="prot"`` provider (screening uses blastx). Headers look like::
|
|
6
|
+
|
|
7
|
+
>BAC0098|ctpC|sp|P0A502|CTPC_MYCTU Probable manganese/zinc-exporting
|
|
8
|
+
|
|
9
|
+
The id token splits on ``|``: ``bac_id|gene|db|uniprot`` plus an optional
|
|
10
|
+
5th field (the UniProt entry name) that upstream ignores. Upstream
|
|
11
|
+
recombines ``gene-bac_id`` as the gene name and ``db:uniprot`` as the
|
|
12
|
+
accession; product falls back to the gene when the description is empty
|
|
13
|
+
(save_fasta's ``DESC || ID``). Function is the locked ``biocide`` constant:
|
|
14
|
+
BacMet covers biocides + metals but the 4-field id carries no class, so
|
|
15
|
+
``biocide`` is the user-approved approximation (Wave F2c).
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from collections.abc import Iterable
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
from gapit.fasta import iter_fasta
|
|
22
|
+
from gapit.providers.common import Provider
|
|
23
|
+
from gapit.records import Record
|
|
24
|
+
|
|
25
|
+
_NAME = "bacmet2"
|
|
26
|
+
_FUNCTION = ("biocide",) # locked gapit/v1 func vocabulary (Wave F2c)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def transform(workdir: Path) -> Iterable[Record]:
|
|
30
|
+
"""Yield Records from ``workdir/BacMet2_EXP_database.fasta``.
|
|
31
|
+
|
|
32
|
+
A record whose id has fewer than 4 pipe-separated fields is skipped
|
|
33
|
+
(upstream perl would emit undef-indexed garbage). Sequence arrives raw
|
|
34
|
+
— fetch_provider X-normalizes it for the prot dbtype.
|
|
35
|
+
"""
|
|
36
|
+
for fasta in iter_fasta(workdir / "BacMet2_EXP_database.fasta"):
|
|
37
|
+
parts = fasta.id.split("|")
|
|
38
|
+
if len(parts) < 4:
|
|
39
|
+
continue
|
|
40
|
+
gene = f"{parts[1]}-{parts[0]}"
|
|
41
|
+
yield Record(
|
|
42
|
+
db=_NAME,
|
|
43
|
+
gene=gene,
|
|
44
|
+
sequence=fasta.sequence,
|
|
45
|
+
accession=f"{parts[2]}:{parts[3]}",
|
|
46
|
+
function=_FUNCTION,
|
|
47
|
+
product=fasta.description or gene,
|
|
48
|
+
source_id=fasta.id,
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
PROVIDER = Provider(
|
|
53
|
+
name=_NAME,
|
|
54
|
+
description="BacMet2 experimentally confirmed biocide/resistance genes (protein)",
|
|
55
|
+
source_urls=("http://bacmet.biomedicine.gu.se/download/BacMet2_EXP_database.fasta",),
|
|
56
|
+
dbtype="prot",
|
|
57
|
+
transform=transform,
|
|
58
|
+
snapshot=None,
|
|
59
|
+
)
|
gapit/providers/card.py
ADDED
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
"""CARD provider (Wave B2): transform card.mcmaster.ca data into Records.
|
|
2
|
+
|
|
3
|
+
Upstream quirk: the source URL ``https://card.mcmaster.ca/latest/data`` has no
|
|
4
|
+
extension but serves a tar.bz2, so the download lands at ``<workdir>/data``.
|
|
5
|
+
Only the root-level ``card.json`` is extracted (upstream:
|
|
6
|
+
``tar xf card.tar.bz2 card.json``), matched by root-normalized member name
|
|
7
|
+
because the re-tarred archive prefixes members with ``./``. Every dict-valued
|
|
8
|
+
top-level entry whose
|
|
9
|
+
``model_type`` is ``"protein homolog model"`` becomes one Record; a homolog
|
|
10
|
+
model carrying ``model_param.snp`` is a hard error (upstream ``err()``); every
|
|
11
|
+
other model type — and any non-dict entry, like the ``_version`` string — is
|
|
12
|
+
skipped. Sequence/function normalization is NOT done here: that is
|
|
13
|
+
fetch_provider's job (B0 contract).
|
|
14
|
+
|
|
15
|
+
``Any`` is confined to the parsed-JSON values flowing through these helpers:
|
|
16
|
+
CARD node shapes are heterogeneous, and checking them into pydantic models
|
|
17
|
+
would dwarf the module. Every value crossing out to a Record is a str/int.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
import json
|
|
21
|
+
import re
|
|
22
|
+
import tarfile
|
|
23
|
+
from collections.abc import Iterator
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
from typing import Any
|
|
26
|
+
|
|
27
|
+
from gapit.errors import DatabaseError
|
|
28
|
+
from gapit.providers.common import Dbtype, Provider
|
|
29
|
+
from gapit.records import Record
|
|
30
|
+
|
|
31
|
+
NAME = "card"
|
|
32
|
+
DESCRIPTION = "CARD protein homolog resistance models"
|
|
33
|
+
SOURCE_URLS = ("https://card.mcmaster.ca/latest/data",)
|
|
34
|
+
DBTYPE: Dbtype = "nucl"
|
|
35
|
+
|
|
36
|
+
_MEMBER = "card.json" # the one member we extract from the tarball root
|
|
37
|
+
_CHAR_SPACE = re.compile(r"\s") # per character, like perl s/\s/_/g (drug classes)
|
|
38
|
+
_RUN_SPACE = re.compile(r"\s+") # per run, like perl s/\s+/_/g (model names)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _root_name(member_name: str) -> str:
|
|
42
|
+
"""Member path with a leading './' and a single leading '/' stripped:
|
|
43
|
+
card.mcmaster.ca re-tarred its archive with './'-prefixed members
|
|
44
|
+
(Wave E), so exact-name lookups miss."""
|
|
45
|
+
return member_name.removeprefix("./").removeprefix("/")
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _models(tarball: Path) -> Iterator[dict[str, Any]]:
|
|
49
|
+
"""Extract card.json from the tarball; yield its dict-valued entries.
|
|
50
|
+
|
|
51
|
+
The member is matched by root-normalized name (first hit in archive
|
|
52
|
+
order). Non-dict top-level values (the ``_version`` metadata string) are
|
|
53
|
+
skipped — upstream's ``next unless ref($g) eq 'HASH'``.
|
|
54
|
+
"""
|
|
55
|
+
with tarfile.open(tarball, "r:bz2") as tar:
|
|
56
|
+
member = next(
|
|
57
|
+
(m for m in tar.getmembers() if _root_name(m.name) == _MEMBER),
|
|
58
|
+
None,
|
|
59
|
+
)
|
|
60
|
+
if member is None:
|
|
61
|
+
raise DatabaseError(
|
|
62
|
+
f"no {_MEMBER} member in {tarball}",
|
|
63
|
+
code="PROVIDER_INVALID",
|
|
64
|
+
context={"archive": str(tarball), "expected": _MEMBER},
|
|
65
|
+
)
|
|
66
|
+
extracted = tar.extractfile(member)
|
|
67
|
+
if extracted is None:
|
|
68
|
+
raise DatabaseError(
|
|
69
|
+
f"{_MEMBER} in {tarball} is not a regular file",
|
|
70
|
+
code="PROVIDER_INVALID",
|
|
71
|
+
context={"db": NAME},
|
|
72
|
+
)
|
|
73
|
+
card: Any = json.loads(extracted.read())
|
|
74
|
+
for model in card.values():
|
|
75
|
+
if isinstance(model, dict):
|
|
76
|
+
yield model
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _drug_classes(model: dict[str, Any]) -> tuple[str, ...]:
|
|
80
|
+
"""Drug Class category names: ' antibiotic' stripped once, then each
|
|
81
|
+
whitespace character -> '_' (upstream's two s/// on $abx), in category
|
|
82
|
+
order — sorting is fetch_provider's job."""
|
|
83
|
+
categories: Any = model.get("ARO_category", {})
|
|
84
|
+
classes: list[str] = []
|
|
85
|
+
for category in categories.values():
|
|
86
|
+
if category.get("category_aro_class_name") == "Drug Class":
|
|
87
|
+
name: str = category.get("category_aro_name", "")
|
|
88
|
+
classes.append(_CHAR_SPACE.sub("_", name.replace(" antibiotic", "", 1)))
|
|
89
|
+
return tuple(classes)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _first_dna(model: dict[str, Any]) -> Any:
|
|
93
|
+
"""The dna_sequence dict of the lexicographically-first key under
|
|
94
|
+
model_sequences.sequence (perl: ``my ($key) = sort keys %$dna``).
|
|
95
|
+
|
|
96
|
+
Its siblings carry accession/strand/fmin/fmax; the sequence itself is the
|
|
97
|
+
``sequence`` key of that same dict.
|
|
98
|
+
"""
|
|
99
|
+
sequences: Any = model.get("model_sequences", {}).get("sequence", {})
|
|
100
|
+
first = sorted(sequences)[0]
|
|
101
|
+
return sequences[first].get("dna_sequence", {})
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _model_record(model: dict[str, Any]) -> Record:
|
|
105
|
+
"""One 'protein homolog model' -> one Record (db=card)."""
|
|
106
|
+
model_name: str = model.get("model_name", "")
|
|
107
|
+
model_param: Any = model.get("model_param") or {}
|
|
108
|
+
if "snp" in model_param:
|
|
109
|
+
raise DatabaseError(
|
|
110
|
+
f"{model_name} has model_param.snp",
|
|
111
|
+
code="PROVIDER_INVALID",
|
|
112
|
+
context={"model": model_name},
|
|
113
|
+
)
|
|
114
|
+
dna: Any = _first_dna(model)
|
|
115
|
+
strand: Any = dna.get("strand", "+")
|
|
116
|
+
fmin: Any = dna.get("fmin", 0)
|
|
117
|
+
fmax: Any = dna.get("fmax", 0)
|
|
118
|
+
start, stop = (fmax, fmin) if strand == "-" else (fmin, fmax)
|
|
119
|
+
product: Any = model.get("ARO_description") or model.get("ARO_accession", "")
|
|
120
|
+
return Record(
|
|
121
|
+
db=NAME,
|
|
122
|
+
gene=_RUN_SPACE.sub("_", model_name),
|
|
123
|
+
sequence=dna.get("sequence", ""),
|
|
124
|
+
accession=f"{dna.get('accession', '')}:{start}-{stop}",
|
|
125
|
+
function=_drug_classes(model),
|
|
126
|
+
product=product,
|
|
127
|
+
source_id=model.get("ARO_accession", ""),
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def transform(workdir: Path) -> Iterator[Record]:
|
|
132
|
+
"""Provider transform: the tarball at ``<workdir>/data`` -> Records.
|
|
133
|
+
|
|
134
|
+
Non-homolog model types are skipped BEFORE the model_param.snp check
|
|
135
|
+
(upstream's statement order), so a variant model carrying an snp is
|
|
136
|
+
skipped, not an error.
|
|
137
|
+
"""
|
|
138
|
+
for model in _models(workdir / "data"):
|
|
139
|
+
if model.get("model_type") == "protein homolog model":
|
|
140
|
+
yield _model_record(model)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
PROVIDER = Provider(
|
|
144
|
+
name=NAME,
|
|
145
|
+
description=DESCRIPTION,
|
|
146
|
+
source_urls=SOURCE_URLS,
|
|
147
|
+
dbtype=DBTYPE,
|
|
148
|
+
transform=transform,
|
|
149
|
+
snapshot="card.tar.gz",
|
|
150
|
+
)
|