gapit 0.2.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gapit/__init__.py +3 -0
- gapit/blast.py +239 -0
- gapit/cli.py +128 -0
- gapit/cmd_db.py +113 -0
- gapit/cmd_db_build.py +72 -0
- gapit/cmd_db_install.py +126 -0
- gapit/cmd_db_outdated.py +51 -0
- gapit/cmd_db_search.py +66 -0
- gapit/cmd_screen.py +214 -0
- gapit/cmd_summary.py +65 -0
- gapit/config.py +51 -0
- gapit/data/snapshots/card.tar.gz +0 -0
- gapit/data/snapshots/vfdb.tar.gz +0 -0
- gapit/db.py +226 -0
- gapit/db_build_ops.py +210 -0
- gapit/db_ops.py +128 -0
- gapit/db_query_ops.py +252 -0
- gapit/dbbuild.py +207 -0
- gapit/dbcodec.py +117 -0
- gapit/dispatch.py +31 -0
- gapit/errors.py +87 -0
- gapit/fasta.py +123 -0
- gapit/formats/__init__.py +1 -0
- gapit/formats/json.py +309 -0
- gapit/formats/md.py +190 -0
- gapit/formats/schemas.py +30 -0
- gapit/formats/summary.py +103 -0
- gapit/formats/tsv.py +45 -0
- gapit/hits.py +107 -0
- gapit/mcp.py +158 -0
- gapit/mcp_schemas.py +123 -0
- gapit/mcp_tools.py +289 -0
- gapit/minimap.py +20 -0
- gapit/minimap2_run.py +114 -0
- gapit/paf.py +115 -0
- gapit/proctools.py +24 -0
- gapit/providers/__init__.py +39 -0
- gapit/providers/argannot.py +94 -0
- gapit/providers/bacmet2.py +59 -0
- gapit/providers/card.py +150 -0
- gapit/providers/common.py +245 -0
- gapit/providers/ecoh.py +63 -0
- gapit/providers/ecoli_vf.py +74 -0
- gapit/providers/megares.py +71 -0
- gapit/providers/ncbi.py +103 -0
- gapit/providers/plasmidfinder.py +69 -0
- gapit/providers/resfinder.py +123 -0
- gapit/providers/snapshots.py +119 -0
- gapit/providers/upec_expec_vf.py +85 -0
- gapit/providers/vfdb.py +92 -0
- gapit/providers/victors.py +109 -0
- gapit/py.typed +0 -0
- gapit/reads.py +221 -0
- gapit/records.py +152 -0
- gapit/report.py +25 -0
- gapit/screening.py +145 -0
- gapit/screening_reads.py +255 -0
- gapit/seqconvert.py +203 -0
- gapit/summary.py +151 -0
- gapit-0.2.2.dist-info/METADATA +183 -0
- gapit-0.2.2.dist-info/RECORD +64 -0
- gapit-0.2.2.dist-info/WHEEL +4 -0
- gapit-0.2.2.dist-info/entry_points.txt +3 -0
- gapit-0.2.2.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,245 @@
|
|
|
1
|
+
"""Wave B0 shared provider infrastructure: the generic get_db pipeline.
|
|
2
|
+
|
|
3
|
+
Every concrete provider (Wave B1..B12) is a :class:`Provider` value — a name,
|
|
4
|
+
a description, a tuple of source URLs, a dbtype, and a ``transform`` that
|
|
5
|
+
turns downloaded files into typed ``Record``s. :func:`fetch_provider` runs
|
|
6
|
+
the generic abricate-get_db flow around it: download each URL into a scratch
|
|
7
|
+
workdir under ``db_dir``, transform, normalize sequences and function classes
|
|
8
|
+
(upstream load_fasta semantics), dedupe by exact normalized sequence
|
|
9
|
+
(first wins), sort by gene, persist ``records.jsonl``, then delegate to
|
|
10
|
+
:func:`gapit.dbbuild.build_database` for the ``sequences`` FASTA, the BLAST
|
|
11
|
+
index, and the manifest (written last, certifying the build).
|
|
12
|
+
|
|
13
|
+
Upstream's ``is_full_gene`` is deliberately NOT ported: its map result is
|
|
14
|
+
discarded — a no-op.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
import os
|
|
18
|
+
import re
|
|
19
|
+
import urllib.request
|
|
20
|
+
from collections.abc import Callable, Iterable, Sequence
|
|
21
|
+
from dataclasses import dataclass
|
|
22
|
+
from importlib.resources import files
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
from tempfile import TemporaryDirectory
|
|
25
|
+
from typing import Literal
|
|
26
|
+
from urllib.parse import urlparse
|
|
27
|
+
|
|
28
|
+
from gapit import __version__
|
|
29
|
+
from gapit.dbbuild import build_database
|
|
30
|
+
from gapit.errors import DatabaseError
|
|
31
|
+
from gapit.proctools import note
|
|
32
|
+
from gapit.providers.snapshots import extract_snapshot, read_snapshot_manifest
|
|
33
|
+
from gapit.records import Manifest, Record, write_records
|
|
34
|
+
|
|
35
|
+
_NUCL_JUNK = re.compile(r"[^AGCT]")
|
|
36
|
+
_PROT_JUNK = re.compile(r"[^A-Z]")
|
|
37
|
+
_WHITESPACE = re.compile(r"\s+")
|
|
38
|
+
_CHUNK_BYTES = 1 << 20
|
|
39
|
+
# Bounded fetch: socket.timeout (an OSError) already maps to DOWNLOAD_FAILED below.
|
|
40
|
+
_DOWNLOAD_TIMEOUT_SECONDS = 60
|
|
41
|
+
|
|
42
|
+
# mgc.ac.cn (vfdb) 403s "Python-urllib" UAs specifically while serving
|
|
43
|
+
# browser-ish clients (Wave E diagnosis): the Mozilla compatibility token
|
|
44
|
+
# passes those naive UA filters and the gapit/<version> suffix still
|
|
45
|
+
# identifies the tool to servers that log it.
|
|
46
|
+
_USER_AGENT = f"Mozilla/5.0 (compatible; gapit/{__version__})"
|
|
47
|
+
|
|
48
|
+
Dbtype = Literal["nucl", "prot"]
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
@dataclass(frozen=True, slots=True)
|
|
52
|
+
class Provider:
|
|
53
|
+
"""One database provider: everything fetch_provider needs to know.
|
|
54
|
+
|
|
55
|
+
``transform`` receives the download workdir (each source URL saved under
|
|
56
|
+
its basename) and yields Records; setting ``db`` to the provider name is
|
|
57
|
+
the provider module's job. Sequences and function classes may arrive
|
|
58
|
+
raw — fetch_provider normalizes both.
|
|
59
|
+
|
|
60
|
+
``snapshot`` names a bundled archive (Wave G) under
|
|
61
|
+
``gapit/data/snapshots/`` — None (default) for network-only providers.
|
|
62
|
+
"""
|
|
63
|
+
|
|
64
|
+
name: str
|
|
65
|
+
description: str
|
|
66
|
+
source_urls: tuple[str, ...]
|
|
67
|
+
dbtype: Dbtype
|
|
68
|
+
transform: Callable[[Path], Iterable[Record]]
|
|
69
|
+
snapshot: str | None = None
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _basename(url: str) -> str:
|
|
73
|
+
"""Final path segment of a URL, query string excluded ('' for a bare host)."""
|
|
74
|
+
return urlparse(url).path.rstrip("/").rsplit("/", 1)[-1]
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _download(url: str, dest: Path) -> None:
|
|
78
|
+
"""Fetch ``url`` into ``dest`` atomically: urlopen a Request carrying
|
|
79
|
+
the module User-Agent, stream the body into a hidden .part file beside
|
|
80
|
+
``dest`` in chunks, then os.replace. file:// URLs work (tests depend on
|
|
81
|
+
it). Any URLError/OSError/ValueError becomes DatabaseError
|
|
82
|
+
``DOWNLOAD_FAILED`` with the URL in context. No shell, ever.
|
|
83
|
+
"""
|
|
84
|
+
temp = dest.with_name(f".{dest.name}.part")
|
|
85
|
+
request = urllib.request.Request(url, headers={"User-Agent": _USER_AGENT})
|
|
86
|
+
try:
|
|
87
|
+
with (
|
|
88
|
+
urllib.request.urlopen(request, timeout=_DOWNLOAD_TIMEOUT_SECONDS) as response,
|
|
89
|
+
temp.open("wb") as out,
|
|
90
|
+
):
|
|
91
|
+
while chunk := response.read(_CHUNK_BYTES):
|
|
92
|
+
out.write(chunk)
|
|
93
|
+
os.replace(temp, dest)
|
|
94
|
+
except (OSError, ValueError) as exc:
|
|
95
|
+
raise DatabaseError(
|
|
96
|
+
f"failed to download {url}: {exc}",
|
|
97
|
+
code="DOWNLOAD_FAILED",
|
|
98
|
+
context={"url": url},
|
|
99
|
+
) from exc
|
|
100
|
+
finally:
|
|
101
|
+
temp.unlink(missing_ok=True)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _normalize_sequence(sequence: str, dbtype: Dbtype) -> str:
|
|
105
|
+
"""Uppercase; then nucl: non-AGCT -> 'N', prot: non-A-Z -> 'X'
|
|
106
|
+
(upstream load_fasta semantics)."""
|
|
107
|
+
upper = sequence.upper()
|
|
108
|
+
match dbtype:
|
|
109
|
+
case "nucl":
|
|
110
|
+
return _NUCL_JUNK.sub("N", upper)
|
|
111
|
+
case "prot":
|
|
112
|
+
return _PROT_JUNK.sub("X", upper)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _normalize_function(classes: Sequence[str]) -> tuple[str, ...]:
|
|
116
|
+
"""Sorted classes with each whitespace run collapsed to one '_'.
|
|
117
|
+
|
|
118
|
+
Upstream sorts, joins with ';', then globally substitutes s/\\s+/_/g —
|
|
119
|
+
';' carries no whitespace, so per-class substitution is equivalent.
|
|
120
|
+
"""
|
|
121
|
+
return tuple(_WHITESPACE.sub("_", drug_class) for drug_class in sorted(classes))
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _normalize_record(record: Record, dbtype: Dbtype) -> Record:
|
|
125
|
+
"""Apply the two per-record normalizations; db, gene, and the rest stay."""
|
|
126
|
+
return record.model_copy(
|
|
127
|
+
update={
|
|
128
|
+
"sequence": _normalize_sequence(record.sequence, dbtype),
|
|
129
|
+
"function": _normalize_function(record.function),
|
|
130
|
+
}
|
|
131
|
+
)
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _dedupe(records: Sequence[Record]) -> tuple[tuple[Record, ...], int]:
|
|
135
|
+
"""Drop later records whose (already normalized) SEQUENCE was seen; first
|
|
136
|
+
wins. Returns the kept records and the dropped count (for the caller's
|
|
137
|
+
stderr note). Duplicate gene NAMES are allowed and kept — upstream too.
|
|
138
|
+
"""
|
|
139
|
+
seen: set[str] = set()
|
|
140
|
+
kept: list[Record] = []
|
|
141
|
+
dropped = 0
|
|
142
|
+
for record in records:
|
|
143
|
+
if record.sequence in seen:
|
|
144
|
+
dropped += 1
|
|
145
|
+
else:
|
|
146
|
+
seen.add(record.sequence)
|
|
147
|
+
kept.append(record)
|
|
148
|
+
return tuple(kept), dropped
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def _snapshot_path(provider: Provider) -> Path | None:
|
|
152
|
+
"""Filesystem path of the provider's bundled snapshot archive, or None
|
|
153
|
+
when the provider has none / the package data is absent (missing
|
|
154
|
+
archives degrade silently to the network path — Wave G). Tests
|
|
155
|
+
monkeypatch THIS seam (string setattr), never importlib itself; gapit
|
|
156
|
+
ships as a regular filesystem package, so the str() round-trip is safe.
|
|
157
|
+
"""
|
|
158
|
+
if provider.snapshot is None:
|
|
159
|
+
return None
|
|
160
|
+
archive = files("gapit").joinpath("data").joinpath("snapshots").joinpath(provider.snapshot)
|
|
161
|
+
return Path(str(archive)) if archive.is_file() else None
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def bundled_snapshot_manifest(provider: Provider) -> Manifest | None:
|
|
165
|
+
"""The provider's bundled snapshot manifest read in-memory, or None when
|
|
166
|
+
the provider ships no resolvable snapshot archive (read-only queries:
|
|
167
|
+
`db outdated` — same seam fetch_provider uses, so tests patch
|
|
168
|
+
``_snapshot_path`` and both paths see the fake)."""
|
|
169
|
+
archive = _snapshot_path(provider)
|
|
170
|
+
return None if archive is None else read_snapshot_manifest(archive)
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def fetch_provider(
|
|
174
|
+
provider: Provider,
|
|
175
|
+
db_dir: Path,
|
|
176
|
+
*,
|
|
177
|
+
fetched_at: str,
|
|
178
|
+
force: bool = False,
|
|
179
|
+
quiet: bool = True,
|
|
180
|
+
from_source: bool = False,
|
|
181
|
+
debug: bool = False,
|
|
182
|
+
) -> Manifest:
|
|
183
|
+
"""Run the generic provider pipeline into ``db_dir`` (created if needed).
|
|
184
|
+
|
|
185
|
+
Snapshot path (Wave G): unless ``from_source`` is set, a provider with a
|
|
186
|
+
resolvable bundled snapshot installs its archived ``records.jsonl`` and
|
|
187
|
+
rebuilds every index locally — ``upstream_version`` comes from the
|
|
188
|
+
archived manifest, everything else is fresh. Otherwise the network path
|
|
189
|
+
runs: download each URL, transform, normalize, dedupe, sort.
|
|
190
|
+
|
|
191
|
+
Raises DatabaseError (exit 4): ``DB_ALREADY_EXISTS`` when a manifest is
|
|
192
|
+
present and force is off (upstream: "Won't overwrite existing (use
|
|
193
|
+
--force)"), ``DOWNLOAD_FAILED`` for a failed source download,
|
|
194
|
+
``SNAPSHOT_INVALID`` for a corrupt snapshot archive, and
|
|
195
|
+
``PROVIDER_EMPTY`` when the transform+dedupe leaves zero records.
|
|
196
|
+
"""
|
|
197
|
+
if (db_dir / "gapit-manifest.json").is_file() and not force:
|
|
198
|
+
raise DatabaseError(
|
|
199
|
+
f"won't overwrite existing database {provider.name} (use --force)",
|
|
200
|
+
code="DB_ALREADY_EXISTS",
|
|
201
|
+
context={"db": provider.name},
|
|
202
|
+
)
|
|
203
|
+
db_dir.mkdir(parents=True, exist_ok=True)
|
|
204
|
+
if not from_source:
|
|
205
|
+
snapshot = _snapshot_path(provider)
|
|
206
|
+
if snapshot is not None:
|
|
207
|
+
archived = extract_snapshot(snapshot, db_dir)
|
|
208
|
+
note(quiet, f"installed {provider.name} from bundled snapshot {snapshot.name}")
|
|
209
|
+
return build_database(
|
|
210
|
+
db_dir,
|
|
211
|
+
name=provider.name,
|
|
212
|
+
dbtype=provider.dbtype,
|
|
213
|
+
source_urls=provider.source_urls,
|
|
214
|
+
fetched_at=fetched_at,
|
|
215
|
+
upstream_version=archived.upstream_version,
|
|
216
|
+
quiet=quiet,
|
|
217
|
+
debug=debug,
|
|
218
|
+
)
|
|
219
|
+
with TemporaryDirectory(dir=db_dir, prefix=".download.") as workdir_name:
|
|
220
|
+
workdir = Path(workdir_name)
|
|
221
|
+
for url in provider.source_urls:
|
|
222
|
+
_download(url, workdir / _basename(url))
|
|
223
|
+
note(quiet, f"downloaded {len(provider.source_urls)} source file(s)")
|
|
224
|
+
records = tuple(
|
|
225
|
+
_normalize_record(record, provider.dbtype) for record in provider.transform(workdir)
|
|
226
|
+
)
|
|
227
|
+
kept, dropped = _dedupe(records)
|
|
228
|
+
note(quiet, f"read {len(records)} records from {provider.name}")
|
|
229
|
+
note(quiet, f"dropped {dropped} duplicate sequence(s), kept {len(kept)}")
|
|
230
|
+
if not kept:
|
|
231
|
+
raise DatabaseError(
|
|
232
|
+
f"provider {provider.name} yielded no records",
|
|
233
|
+
code="PROVIDER_EMPTY",
|
|
234
|
+
context={"db": provider.name},
|
|
235
|
+
)
|
|
236
|
+
write_records(sorted(kept, key=lambda record: record.gene), db_dir / "records.jsonl")
|
|
237
|
+
return build_database(
|
|
238
|
+
db_dir,
|
|
239
|
+
name=provider.name,
|
|
240
|
+
dbtype=provider.dbtype,
|
|
241
|
+
source_urls=provider.source_urls,
|
|
242
|
+
fetched_at=fetched_at,
|
|
243
|
+
quiet=quiet,
|
|
244
|
+
debug=debug,
|
|
245
|
+
)
|
gapit/providers/ecoh.py
ADDED
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
"""ecoh database provider — E. coli O and H antigen genes (srst2 EcOH).
|
|
2
|
+
|
|
3
|
+
Transform-only Wave B10 module: ``PROVIDER`` wires the pinned metadata into
|
|
4
|
+
the frozen B0 :class:`gapit.providers.common.Provider` contract and
|
|
5
|
+
:func:`transform` parses the downloaded ``EcOH.fasta`` into typed Records.
|
|
6
|
+
Functional categories are allele-prefix derived — ``fliC*`` → H-antigen,
|
|
7
|
+
``wzx``/``wzy``/``wzt``/``wzm`` → O-antigen — fallback ``antigen``.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from collections.abc import Iterator
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
from gapit.fasta import iter_fasta
|
|
14
|
+
from gapit.providers.common import Provider
|
|
15
|
+
from gapit.records import Record
|
|
16
|
+
|
|
17
|
+
NAME = "ecoh"
|
|
18
|
+
SOURCE_FILE = "EcOH.fasta"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def transform(workdir: Path) -> Iterator[Record]:
|
|
22
|
+
"""Parse ``workdir/EcOH.fasta`` into Records (db = ecoh).
|
|
23
|
+
|
|
24
|
+
srst2 header convention: ``>cluster__gene__allele__seq_id acc;word;word``
|
|
25
|
+
(abricate-get_db ``get_ecoh``). The id splits on ``__`` into exactly 4
|
|
26
|
+
parts and the gene is part 3 — the allele, e.g. ``fliC-H1``. The
|
|
27
|
+
description splits on ``;``: the first piece is the accession, the
|
|
28
|
+
space-joined rest is the product. Ids without exactly 4 ``__``-parts are
|
|
29
|
+
skipped: upstream perl would read an undefined gene field there, and a
|
|
30
|
+
typed Record cannot carry one. ``function`` is allele-prefix derived
|
|
31
|
+
per the module-docstring rule (fallback ``antigen``).
|
|
32
|
+
"""
|
|
33
|
+
for fasta in iter_fasta(workdir / SOURCE_FILE):
|
|
34
|
+
parts = fasta.id.split("__")
|
|
35
|
+
if len(parts) != 4:
|
|
36
|
+
continue
|
|
37
|
+
gene = parts[2]
|
|
38
|
+
if gene.startswith("fliC"):
|
|
39
|
+
function: tuple[str, ...] = ("H-antigen",)
|
|
40
|
+
elif gene.startswith(("wzx", "wzy", "wzt", "wzm")):
|
|
41
|
+
function = ("O-antigen",)
|
|
42
|
+
else:
|
|
43
|
+
function = ("antigen",)
|
|
44
|
+
description = fasta.description.split(";")
|
|
45
|
+
yield Record(
|
|
46
|
+
db=NAME,
|
|
47
|
+
gene=gene,
|
|
48
|
+
accession=description[0],
|
|
49
|
+
function=function,
|
|
50
|
+
product=" ".join(description[1:]),
|
|
51
|
+
sequence=fasta.sequence,
|
|
52
|
+
source_id=fasta.id,
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
PROVIDER = Provider(
|
|
57
|
+
name=NAME,
|
|
58
|
+
description="E. coli O and H antigens (srst2 EcOH)",
|
|
59
|
+
source_urls=("https://raw.githubusercontent.com/katholt/srst2/master/data/EcOH.fasta",),
|
|
60
|
+
dbtype="nucl",
|
|
61
|
+
transform=transform,
|
|
62
|
+
snapshot=None,
|
|
63
|
+
)
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
r"""ecoli_vf provider (phac-nml E. coli virulence factors) — transform only.
|
|
2
|
+
|
|
3
|
+
Upstream ``get_ecoli_vf`` (abricate-get_db 1.4.0) parses
|
|
4
|
+
``repaired_ecoli_vfs_shortnames.ffn`` in three steps per record::
|
|
5
|
+
|
|
6
|
+
>VFG000748(gi:2865308) (espF) EspF [EspF (VF0182)] [Escherichia coli ...]
|
|
7
|
+
|
|
8
|
+
1. id ``^(\w+)(?:\((.*?)\))?$`` -> base id + optional paren accession
|
|
9
|
+
(upstream ``die``s on a non-match; we SKIP the record instead —
|
|
10
|
+
same deviation family as vfdb/ecoh, one bad header must not kill
|
|
11
|
+
a fetch)
|
|
12
|
+
2. accession = ``$2 || $1`` -> paren content; an absent OR EMPTY
|
|
13
|
+
capture falls back to the base id (perl ``||`` is falsy-based)
|
|
14
|
+
3. description: repeatedly strip trailing bracket groups
|
|
15
|
+
(``s/\s\[.*?\]$//g``), then ``^(?:\((.*?)\)\s+)?(.*)$`` -> a
|
|
16
|
+
leading paren group RENAMES the gene (overrides the base id), the
|
|
17
|
+
remainder becomes the product (``$DESC || $ID`` fallback).
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
import re
|
|
21
|
+
from collections.abc import Iterable
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
|
|
24
|
+
from gapit.fasta import iter_fasta
|
|
25
|
+
from gapit.providers.common import Provider
|
|
26
|
+
from gapit.records import Record
|
|
27
|
+
|
|
28
|
+
_NAME = "ecoli_vf"
|
|
29
|
+
_SOURCE_FILE = "repaired_ecoli_vfs_shortnames.ffn"
|
|
30
|
+
|
|
31
|
+
# Upstream regexes verbatim (the /x id regex ignores whitespace; joined here).
|
|
32
|
+
_ID = re.compile(r"^(\w+)(?:\((.*?)\))?$")
|
|
33
|
+
_TRAILING_BRACKETS = re.compile(r"\s\[.*?\]$")
|
|
34
|
+
_DESC = re.compile(r"^(?:\((.*?)\)\s+)?(.*)$")
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def transform(workdir: Path) -> Iterable[Record]:
|
|
38
|
+
"""Yield Records from ``workdir/repaired_ecoli_vfs_shortnames.ffn``.
|
|
39
|
+
|
|
40
|
+
Sequence and the leading/trailing-stripped description arrive via
|
|
41
|
+
gapit.fasta; gene/accession/product derive per the docstring steps.
|
|
42
|
+
"""
|
|
43
|
+
for fasta in iter_fasta(workdir / _SOURCE_FILE):
|
|
44
|
+
id_match = _ID.match(fasta.id)
|
|
45
|
+
if id_match is None:
|
|
46
|
+
continue
|
|
47
|
+
base = id_match.group(1)
|
|
48
|
+
accession = id_match.group(2) or base # perl $2 || $1: '' is falsy too
|
|
49
|
+
description = fasta.description
|
|
50
|
+
while (stripped := _TRAILING_BRACKETS.sub("", description)) != description:
|
|
51
|
+
description = stripped
|
|
52
|
+
desc_match = _DESC.match(description)
|
|
53
|
+
assert desc_match is not None # (.*) matches any string — cannot fail
|
|
54
|
+
gene = desc_match.group(1) or base
|
|
55
|
+
product = desc_match.group(2)
|
|
56
|
+
yield Record(
|
|
57
|
+
db=_NAME,
|
|
58
|
+
gene=gene,
|
|
59
|
+
sequence=fasta.sequence,
|
|
60
|
+
accession=accession,
|
|
61
|
+
function=("virulence",),
|
|
62
|
+
product=product or gene, # save_fasta: -desc => ($DESC || $ID)
|
|
63
|
+
source_id=base,
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
PROVIDER = Provider(
|
|
68
|
+
name=_NAME,
|
|
69
|
+
description="E. coli virulence factors (phac-nml)",
|
|
70
|
+
source_urls=("https://github.com/phac-nml/ecoli_vf/raw/master/data/" + _SOURCE_FILE,),
|
|
71
|
+
dbtype="nucl",
|
|
72
|
+
transform=transform,
|
|
73
|
+
snapshot=None,
|
|
74
|
+
)
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
"""megares database provider — MEGARes v3 antimicrobial resistance genes.
|
|
2
|
+
|
|
3
|
+
Transform-only Wave B9 module: ``PROVIDER`` wires the pinned metadata into
|
|
4
|
+
the frozen B0 :class:`gapit.providers.common.Provider` contract and
|
|
5
|
+
:func:`transform` unpacks the downloaded ``megares_v3.00.zip`` and parses
|
|
6
|
+
every ``megares_drugs_*.fasta`` inside it into typed Records.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import zipfile
|
|
10
|
+
from collections.abc import Iterator
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
from gapit.fasta import iter_fasta
|
|
14
|
+
from gapit.providers.common import Provider
|
|
15
|
+
from gapit.records import Record
|
|
16
|
+
|
|
17
|
+
NAME = "megares"
|
|
18
|
+
ARCHIVE = "megares_v3.00.zip"
|
|
19
|
+
_DATABASE_GLOB = "megares_drugs_*.fasta"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def transform(workdir: Path) -> Iterator[Record]:
|
|
23
|
+
"""Extract ``workdir/megares_v3.00.zip`` and parse every nested
|
|
24
|
+
``megares_drugs_*.fasta`` into Records (db = megares).
|
|
25
|
+
|
|
26
|
+
MEGARes v3 header convention (abricate-get_db ``get_megares``):
|
|
27
|
+
``>id|type|class|mech|group|note`` where the note field exists only on
|
|
28
|
+
records requiring SNP confirmation — a non-empty note skips the record
|
|
29
|
+
(SPEC §8). Keepers map gene=group, accession=source_id=id (the MEG_
|
|
30
|
+
number), product=colon-joined type/class/mech/group; function = the
|
|
31
|
+
``class`` field (x[2], e.g. Tetracyclines) as a 1-tuple — upstream
|
|
32
|
+
sets no ABX, the class IS the functional category. Archives often nest
|
|
33
|
+
the fasta one directory deep — upstream
|
|
34
|
+
``unzip -j`` flattens, here the glob is recursive (sorted, so multi-file
|
|
35
|
+
archives parse deterministically).
|
|
36
|
+
|
|
37
|
+
Documented deviation: upstream perl splits into six list variables and
|
|
38
|
+
keeps records even when fields are missing (undef gene, empty product
|
|
39
|
+
pieces); gapit skips ids with fewer than 5 pipe-fields or an empty among
|
|
40
|
+
the 5 leading ones. ``split('|', 5)`` mirrors perl's implicit
|
|
41
|
+
list-assignment limit: any pipes past the group fold into the note, and
|
|
42
|
+
a folded note is non-empty, so over-long ids skip exactly like upstream.
|
|
43
|
+
"""
|
|
44
|
+
with zipfile.ZipFile(workdir / ARCHIVE) as archive:
|
|
45
|
+
archive.extractall(workdir)
|
|
46
|
+
for fasta_path in sorted(workdir.rglob(_DATABASE_GLOB)):
|
|
47
|
+
for fasta in iter_fasta(fasta_path):
|
|
48
|
+
parts = fasta.id.split("|", 5)
|
|
49
|
+
if len(parts) < 5 or "" in parts[:5]:
|
|
50
|
+
continue
|
|
51
|
+
if len(parts) == 6 and parts[5]:
|
|
52
|
+
continue
|
|
53
|
+
yield Record(
|
|
54
|
+
db=NAME,
|
|
55
|
+
gene=parts[4],
|
|
56
|
+
accession=parts[0],
|
|
57
|
+
function=(parts[2],),
|
|
58
|
+
product=":".join(parts[1:5]),
|
|
59
|
+
sequence=fasta.sequence,
|
|
60
|
+
source_id=parts[0],
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
PROVIDER = Provider(
|
|
65
|
+
name=NAME,
|
|
66
|
+
description="MEGARes antimicrobial resistance genes",
|
|
67
|
+
source_urls=("https://www.meglab.org/downloads/megares_v3.00.zip",),
|
|
68
|
+
dbtype="nucl",
|
|
69
|
+
transform=transform,
|
|
70
|
+
snapshot=None,
|
|
71
|
+
)
|
gapit/providers/ncbi.py
ADDED
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
"""ncbi database provider — NCBI AMRFinderPlus curated AMR (Wave B1).
|
|
2
|
+
|
|
3
|
+
Transform-only module mirroring abricate-get_db ``get_ncbi``: pair
|
|
4
|
+
``AMR_CDS.fa`` records with ``ReferenceGeneCatalog.txt`` rows keyed by
|
|
5
|
+
``refseq_nucleotide_accession`` (column 10, 0-based), keeping only plain
|
|
6
|
+
(non-fusion) genes whose catalog row is scope core / type AMR / subtype AMR.
|
|
7
|
+
``PROVIDER`` wires the pinned metadata into the frozen B0
|
|
8
|
+
:class:`gapit.providers.common.Provider` contract.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
import re
|
|
12
|
+
from collections.abc import Iterator
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
from gapit.fasta import iter_fasta
|
|
16
|
+
from gapit.providers.common import Provider
|
|
17
|
+
from gapit.records import Record
|
|
18
|
+
|
|
19
|
+
NAME = "ncbi"
|
|
20
|
+
AMR_CDS_FILE = "AMR_CDS.fa"
|
|
21
|
+
CATALOG_FILE = "ReferenceGeneCatalog.txt"
|
|
22
|
+
_ACCESSION_COLUMN = 10
|
|
23
|
+
_VERSIONED_ACCESSION = re.compile(r"\.\d+$")
|
|
24
|
+
_LATEST = (
|
|
25
|
+
"https://ftp.ncbi.nlm.nih.gov/pathogen/Antimicrobial_resistance/AMRFinderPlus/database/latest"
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _load_catalog(path: Path) -> dict[str, dict[str, str]]:
|
|
30
|
+
"""ReferenceGeneCatalog rows as ``{accession: {header name: value}}``.
|
|
31
|
+
|
|
32
|
+
The first line is the header; every later row is keyed by column 10
|
|
33
|
+
(``refseq_nucleotide_accession``). Duplicate accessions keep the FIRST
|
|
34
|
+
row (upstream's ``||=``), and rows are padded to the header width so
|
|
35
|
+
lookups of unused trailing columns cannot fail on short rows.
|
|
36
|
+
"""
|
|
37
|
+
rows: dict[str, dict[str, str]] = {}
|
|
38
|
+
header: list[str] | None = None
|
|
39
|
+
with path.open(encoding="utf-8") as handle:
|
|
40
|
+
for line in handle:
|
|
41
|
+
columns = line.rstrip("\r\n").split("\t")
|
|
42
|
+
if header is None:
|
|
43
|
+
header = columns
|
|
44
|
+
continue
|
|
45
|
+
width = max(len(header), _ACCESSION_COLUMN + 1)
|
|
46
|
+
padded = columns + [""] * (width - len(columns))
|
|
47
|
+
if (acc := padded[_ACCESSION_COLUMN]) and acc not in rows:
|
|
48
|
+
rows[acc] = dict(zip(header, padded, strict=False)) # perl zip: shorter wins
|
|
49
|
+
return rows
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def transform(workdir: Path) -> Iterator[Record]:
|
|
53
|
+
"""Parse ``workdir/AMR_CDS.fa`` + ``ReferenceGeneCatalog.txt`` (db = ncbi).
|
|
54
|
+
|
|
55
|
+
AMRFinderPlus fasta ids are ``pi|acc|fp|fn|gene|fam|prod`` — perl's
|
|
56
|
+
7-variable ``split`` assignment imposes LIMIT 7, so everything past the
|
|
57
|
+
sixth pipe belongs to ``prod``. Records are skipped — passively, like
|
|
58
|
+
upstream — unless the id yields 7 non-empty fields, fp/fn are both "1"
|
|
59
|
+
(fusion filter), and the accession (``.1`` appended unless already
|
|
60
|
+
versioned, e.g. ``NG_050200`` -> ``NG_050200.1``) matches a catalog row
|
|
61
|
+
with scope ``core``, type ``AMR``, and subtype ``AMR``. Sequences and
|
|
62
|
+
function categories stay raw: fetch_provider owns normalization.
|
|
63
|
+
"""
|
|
64
|
+
catalog = _load_catalog(workdir / CATALOG_FILE)
|
|
65
|
+
for fasta in iter_fasta(workdir / AMR_CDS_FILE):
|
|
66
|
+
fields = fasta.id.split("|")
|
|
67
|
+
if len(fields) < 7:
|
|
68
|
+
continue
|
|
69
|
+
pi, acc, fp, fn, gene, fam = fields[:6]
|
|
70
|
+
prod = "|".join(fields[6:])
|
|
71
|
+
if not all((pi, acc, fp, fn, gene, fam, prod)):
|
|
72
|
+
continue
|
|
73
|
+
if fp != "1" or fn != "1":
|
|
74
|
+
continue
|
|
75
|
+
if not _VERSIONED_ACCESSION.search(acc):
|
|
76
|
+
acc += ".1"
|
|
77
|
+
row = catalog.get(acc)
|
|
78
|
+
if row is None:
|
|
79
|
+
continue
|
|
80
|
+
if row["scope"] != "core" or row["type"] != "AMR" or row["subtype"] != "AMR":
|
|
81
|
+
continue
|
|
82
|
+
yield Record(
|
|
83
|
+
db=NAME,
|
|
84
|
+
gene=gene,
|
|
85
|
+
accession=row["refseq_nucleotide_accession"],
|
|
86
|
+
function=tuple(part for part in row["subclass"].split("/") if part),
|
|
87
|
+
product=prod.replace("_", " "),
|
|
88
|
+
sequence=fasta.sequence,
|
|
89
|
+
source_id=pi,
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
PROVIDER = Provider(
|
|
94
|
+
name=NAME,
|
|
95
|
+
description="NCBI AMRFinderPlus (reference finder) curated AMR",
|
|
96
|
+
source_urls=(
|
|
97
|
+
f"{_LATEST}/AMR_CDS.fa",
|
|
98
|
+
f"{_LATEST}/ReferenceGeneCatalog.txt",
|
|
99
|
+
),
|
|
100
|
+
dbtype="nucl",
|
|
101
|
+
transform=transform,
|
|
102
|
+
snapshot=None,
|
|
103
|
+
)
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"""plasmidfinder database provider — CGE PlasmidFinder replicons (transform only).
|
|
2
|
+
|
|
3
|
+
Upstream ``get_plasmidfinder`` (abricate-get_db 1.4.0) downloads the bitbucket
|
|
4
|
+
HEAD.zip — an archive with an arbitrary top-level directory — unzips every
|
|
5
|
+
``*.fsa`` and, per record, splits the accession suffix off the id::
|
|
6
|
+
|
|
7
|
+
>IncFII_1_NC_004631.1 -> gene IncFII_1, accession NC_004631.1
|
|
8
|
+
|
|
9
|
+
Perl semantics preserved exactly (verified against the parity env): the
|
|
10
|
+
product is the ORIGINAL full id (DESC is assigned before the regex munging),
|
|
11
|
+
trailing underscore runs are stripped from the captured id, and a non-matching
|
|
12
|
+
id — or one that strips to empty — keeps the ORIGINAL id as gene with an empty
|
|
13
|
+
accession. Upstream additionally warns on the empty-id case; the transform has
|
|
14
|
+
no diagnostic channel, so the record is kept silently (``fetch_provider`` owns
|
|
15
|
+
stderr). The ``_1`` copy number STAYS on the gene: only the accession suffix
|
|
16
|
+
is removed.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
import re
|
|
20
|
+
import zipfile
|
|
21
|
+
from collections.abc import Iterator
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
|
|
24
|
+
from gapit.fasta import iter_fasta
|
|
25
|
+
from gapit.providers.common import Provider
|
|
26
|
+
from gapit.records import Record
|
|
27
|
+
|
|
28
|
+
NAME = "plasmidfinder"
|
|
29
|
+
ARCHIVE = "HEAD.zip"
|
|
30
|
+
|
|
31
|
+
# Upstream regex verbatim: group 2 is the accession ([A-Z]+ prefix or the
|
|
32
|
+
# literal NC_, digits, optional .version); group 1 is everything before the
|
|
33
|
+
# last viable underscore — greedy, so the copy number stays in group 1.
|
|
34
|
+
_ID = re.compile(r"^(.*)_(([A-Z]+|NC_)\d+(\.\d+)?)$")
|
|
35
|
+
_FUNCTION = ("replicon",) # locked gapit/v1 func vocabulary (Wave F2c)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def transform(workdir: Path) -> Iterator[Record]:
|
|
39
|
+
"""Yield Records from every ``*.fsa`` inside ``workdir/HEAD.zip``.
|
|
40
|
+
|
|
41
|
+
Members are extracted into the workdir (bitbucket archives nest under an
|
|
42
|
+
arbitrary top-level directory) and matched with a recursive glob, in
|
|
43
|
+
sorted order for determinism.
|
|
44
|
+
"""
|
|
45
|
+
with zipfile.ZipFile(workdir / ARCHIVE) as archive:
|
|
46
|
+
archive.extractall(workdir)
|
|
47
|
+
for fsa in sorted(workdir.glob("**/*.fsa")):
|
|
48
|
+
for fasta in iter_fasta(fsa):
|
|
49
|
+
id_match = _ID.match(fasta.id)
|
|
50
|
+
captured = id_match.group(1).rstrip("_") if id_match is not None else ""
|
|
51
|
+
yield Record(
|
|
52
|
+
db=NAME,
|
|
53
|
+
gene=captured or fasta.id,
|
|
54
|
+
accession=id_match.group(2) if id_match is not None else "",
|
|
55
|
+
function=_FUNCTION,
|
|
56
|
+
product=fasta.id,
|
|
57
|
+
sequence=fasta.sequence,
|
|
58
|
+
source_id=fasta.id,
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
PROVIDER = Provider(
|
|
63
|
+
name=NAME,
|
|
64
|
+
description="CGE PlasmidFinder replicons",
|
|
65
|
+
source_urls=("https://bitbucket.org/genomicepidemiology/plasmidfinder_db/get/HEAD.zip",),
|
|
66
|
+
dbtype="nucl",
|
|
67
|
+
transform=transform,
|
|
68
|
+
snapshot=None,
|
|
69
|
+
)
|