gapit 0.2.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gapit/__init__.py +3 -0
- gapit/blast.py +239 -0
- gapit/cli.py +128 -0
- gapit/cmd_db.py +113 -0
- gapit/cmd_db_build.py +72 -0
- gapit/cmd_db_install.py +126 -0
- gapit/cmd_db_outdated.py +51 -0
- gapit/cmd_db_search.py +66 -0
- gapit/cmd_screen.py +214 -0
- gapit/cmd_summary.py +65 -0
- gapit/config.py +51 -0
- gapit/data/snapshots/card.tar.gz +0 -0
- gapit/data/snapshots/vfdb.tar.gz +0 -0
- gapit/db.py +226 -0
- gapit/db_build_ops.py +210 -0
- gapit/db_ops.py +128 -0
- gapit/db_query_ops.py +252 -0
- gapit/dbbuild.py +207 -0
- gapit/dbcodec.py +117 -0
- gapit/dispatch.py +31 -0
- gapit/errors.py +87 -0
- gapit/fasta.py +123 -0
- gapit/formats/__init__.py +1 -0
- gapit/formats/json.py +309 -0
- gapit/formats/md.py +190 -0
- gapit/formats/schemas.py +30 -0
- gapit/formats/summary.py +103 -0
- gapit/formats/tsv.py +45 -0
- gapit/hits.py +107 -0
- gapit/mcp.py +158 -0
- gapit/mcp_schemas.py +123 -0
- gapit/mcp_tools.py +289 -0
- gapit/minimap.py +20 -0
- gapit/minimap2_run.py +114 -0
- gapit/paf.py +115 -0
- gapit/proctools.py +24 -0
- gapit/providers/__init__.py +39 -0
- gapit/providers/argannot.py +94 -0
- gapit/providers/bacmet2.py +59 -0
- gapit/providers/card.py +150 -0
- gapit/providers/common.py +245 -0
- gapit/providers/ecoh.py +63 -0
- gapit/providers/ecoli_vf.py +74 -0
- gapit/providers/megares.py +71 -0
- gapit/providers/ncbi.py +103 -0
- gapit/providers/plasmidfinder.py +69 -0
- gapit/providers/resfinder.py +123 -0
- gapit/providers/snapshots.py +119 -0
- gapit/providers/upec_expec_vf.py +85 -0
- gapit/providers/vfdb.py +92 -0
- gapit/providers/victors.py +109 -0
- gapit/py.typed +0 -0
- gapit/reads.py +221 -0
- gapit/records.py +152 -0
- gapit/report.py +25 -0
- gapit/screening.py +145 -0
- gapit/screening_reads.py +255 -0
- gapit/seqconvert.py +203 -0
- gapit/summary.py +151 -0
- gapit-0.2.2.dist-info/METADATA +183 -0
- gapit-0.2.2.dist-info/RECORD +64 -0
- gapit-0.2.2.dist-info/WHEEL +4 -0
- gapit-0.2.2.dist-info/entry_points.txt +3 -0
- gapit-0.2.2.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
"""resfinder database provider — CGE ResFinder acquired resistance genes.
|
|
2
|
+
|
|
3
|
+
Upstream ``get_resfinder`` (abricate-get_db 1.4.0) git-clones the resfinder_db
|
|
4
|
+
repo; gapit downloads the same tree as the bitbucket ``HEAD.zip`` archive (no
|
|
5
|
+
git dependency) whose members sit under an arbitrary top-level directory.
|
|
6
|
+
``phenotypes.txt`` maps each gene to its function categories; every ``*.fsa``
|
|
7
|
+
carries ids like ``demoA_1_FAKE0001`` — gene prefix, copy number, accession.
|
|
8
|
+
|
|
9
|
+
Issue #62 repair: resfinder .fsa files can glue a record header onto the end
|
|
10
|
+
of the previous sequence line (a letter immediately followed by ``>``); a
|
|
11
|
+
newline is inserted there in the raw text before parsing, exactly like
|
|
12
|
+
upstream's ``sed s/([A-Z])>/\\1\\n>/gi`` (the /i makes it both cases).
|
|
13
|
+
|
|
14
|
+
Perl semantics kept: Class-cell pieces are filtered of unknown/notes/none
|
|
15
|
+
markers BEFORE stripping (upstream greps, then trims), the LAST phenotypes
|
|
16
|
+
row for a gene wins (plain hash assign), and the product is always the gene
|
|
17
|
+
prefix (upstream's ``$anno{$id}{DESC}`` is never assigned). Deviation: a
|
|
18
|
+
record whose id has no ``_<digits>_<accession>`` structure is SKIPPED —
|
|
19
|
+
upstream would emit it with undef fields (same policy as vfdb).
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
import re
|
|
23
|
+
import zipfile
|
|
24
|
+
from collections.abc import Iterator
|
|
25
|
+
from pathlib import Path
|
|
26
|
+
from tempfile import TemporaryDirectory
|
|
27
|
+
|
|
28
|
+
from gapit.errors import DatabaseError
|
|
29
|
+
from gapit.fasta import iter_fasta
|
|
30
|
+
from gapit.providers.common import Provider
|
|
31
|
+
from gapit.records import Record
|
|
32
|
+
|
|
33
|
+
NAME = "resfinder"
|
|
34
|
+
ARCHIVE = "HEAD.zip"
|
|
35
|
+
|
|
36
|
+
_PHENOTYPES = "phenotypes.txt"
|
|
37
|
+
_GENE = re.compile(r"^(.*?)_\w+$") # col0 minus its final _<word> chunk
|
|
38
|
+
_CLASS_SPLIT = re.compile(r",\s*")
|
|
39
|
+
_CLASS_FILTER = re.compile(r"(unknown|notes|^none)", re.IGNORECASE)
|
|
40
|
+
# abricate issue #62: a letter directly followed by '>' is a glued header.
|
|
41
|
+
_INLINE_HEADER = re.compile(r"([A-Za-z])>")
|
|
42
|
+
_ID = re.compile(r"^(.*?)_(\d+)_(\S+)$") # base _ copy _ accession
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _read_classes(root: Path) -> dict[str, tuple[str, ...]]:
|
|
46
|
+
"""phenotypes.txt under the extracted archive: {gene prefix: classes}.
|
|
47
|
+
|
|
48
|
+
Comment lines start with '#'; rows split on tabs; the gene prefix is
|
|
49
|
+
col0 minus its final ``_<word>`` chunk; classes are the col2 pieces
|
|
50
|
+
(split on comma + whitespace) that do not match unknown/notes/none —
|
|
51
|
+
filtered BEFORE stripping, like upstream — each stripped. A later row
|
|
52
|
+
for the same gene overwrites the earlier one (plain hash assign), and a
|
|
53
|
+
missing Class cell means no classes (upstream splits undef into an
|
|
54
|
+
empty list).
|
|
55
|
+
"""
|
|
56
|
+
found = sorted(root.glob(f"**/{_PHENOTYPES}"))
|
|
57
|
+
if not found:
|
|
58
|
+
raise DatabaseError(
|
|
59
|
+
f"{ARCHIVE} contains no {_PHENOTYPES}",
|
|
60
|
+
code="PROVIDER_INVALID",
|
|
61
|
+
context={"db": NAME},
|
|
62
|
+
)
|
|
63
|
+
classes: dict[str, tuple[str, ...]] = {}
|
|
64
|
+
for line in found[0].read_text(encoding="utf-8").splitlines():
|
|
65
|
+
if line.startswith("#"):
|
|
66
|
+
continue
|
|
67
|
+
row = line.split("\t")
|
|
68
|
+
gene = _GENE.match(row[0])
|
|
69
|
+
if gene is None:
|
|
70
|
+
continue
|
|
71
|
+
cell = row[2] if len(row) > 2 else ""
|
|
72
|
+
classes[gene.group(1)] = tuple(
|
|
73
|
+
piece.strip()
|
|
74
|
+
for piece in (_CLASS_SPLIT.split(cell) if cell else [])
|
|
75
|
+
if not _CLASS_FILTER.search(piece)
|
|
76
|
+
)
|
|
77
|
+
return classes
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def transform(workdir: Path) -> Iterator[Record]:
|
|
81
|
+
"""Yield Records from ``workdir/HEAD.zip`` (the bitbucket archive).
|
|
82
|
+
|
|
83
|
+
The archive is extracted into a scratch directory (members sit under an
|
|
84
|
+
arbitrary top-level directory); each ``*.fsa`` is repaired for glued
|
|
85
|
+
headers into a scratch copy before FASTA parsing, so the download is
|
|
86
|
+
never modified. Files are visited in sorted order for determinism.
|
|
87
|
+
"""
|
|
88
|
+
with (
|
|
89
|
+
TemporaryDirectory(prefix=".resfinder-extract.") as extract_name,
|
|
90
|
+
TemporaryDirectory(prefix=".resfinder-repair.") as repair_name,
|
|
91
|
+
):
|
|
92
|
+
extract_dir = Path(extract_name)
|
|
93
|
+
with zipfile.ZipFile(workdir / ARCHIVE) as archive:
|
|
94
|
+
archive.extractall(extract_dir)
|
|
95
|
+
classes = _read_classes(extract_dir)
|
|
96
|
+
for index, fsa in enumerate(sorted(extract_dir.glob("**/*.fsa"))):
|
|
97
|
+
repaired = _INLINE_HEADER.sub(r"\1\n>", fsa.read_text(encoding="utf-8"))
|
|
98
|
+
temp = Path(repair_name) / f"{index:06d}.fsa"
|
|
99
|
+
temp.write_text(repaired, encoding="utf-8")
|
|
100
|
+
for fasta in iter_fasta(temp):
|
|
101
|
+
id_match = _ID.match(fasta.id)
|
|
102
|
+
if id_match is None:
|
|
103
|
+
continue
|
|
104
|
+
base = id_match.group(1)
|
|
105
|
+
yield Record(
|
|
106
|
+
db=NAME,
|
|
107
|
+
gene=f"{base}_{id_match.group(2)}",
|
|
108
|
+
sequence=fasta.sequence,
|
|
109
|
+
accession=id_match.group(3),
|
|
110
|
+
function=classes.get(base, ()),
|
|
111
|
+
product=base,
|
|
112
|
+
source_id=fasta.id,
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
PROVIDER = Provider(
|
|
117
|
+
name=NAME,
|
|
118
|
+
description="CGE ResFinder acquired resistance genes",
|
|
119
|
+
source_urls=("https://bitbucket.org/genomicepidemiology/resfinder_db/get/HEAD.zip",),
|
|
120
|
+
dbtype="nucl",
|
|
121
|
+
transform=transform,
|
|
122
|
+
snapshot=None,
|
|
123
|
+
)
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
"""Bundled database snapshots (Wave G): deterministic archives + extraction.
|
|
2
|
+
|
|
3
|
+
A snapshot is ``<name>.tar.gz`` containing EXACTLY ``records.jsonl`` and
|
|
4
|
+
``gapit-manifest.json`` from an installed gapit-native database directory.
|
|
5
|
+
card and vfdb ship inside the wheel (``src/gapit/data/snapshots/``) so the
|
|
6
|
+
two default databases install with zero network — snapshots eliminate the
|
|
7
|
+
flaky-source failure modes for exactly those two (Wave E/F3: mgc.ac.cn UA
|
|
8
|
+
blocks, card re-tarring, Wayback interstitials). Every other provider keeps
|
|
9
|
+
the upstream fetch; a future DB is bundled by dropping a ``<name>.tar.gz``
|
|
10
|
+
into the snapshots dir and setting ``snapshot=`` on its Provider (one line).
|
|
11
|
+
|
|
12
|
+
``records.jsonl`` (post-normalize records) is what gets snapshotted — NOT
|
|
13
|
+
the BLAST index: index bytes are BLAST-version-sensitive while a
|
|
14
|
+
local rebuild from records is deterministic and fast. The archived manifest
|
|
15
|
+
contributes only ``upstream_version``; fetched_at, sha256 and tool versions
|
|
16
|
+
are rebuilt locally by :func:`gapit.dbbuild.build_database`.
|
|
17
|
+
|
|
18
|
+
Determinism (frozen layout contract): entries added in sorted name order,
|
|
19
|
+
PAX format, uid/gid 0, mtime 0, and a gzip stream with mtime 0 and no
|
|
20
|
+
embedded filename — two builds from the same inputs are byte-identical.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
import gzip
|
|
24
|
+
import io
|
|
25
|
+
import os
|
|
26
|
+
import tarfile
|
|
27
|
+
from pathlib import Path
|
|
28
|
+
from tempfile import NamedTemporaryFile
|
|
29
|
+
|
|
30
|
+
from gapit.errors import DatabaseError
|
|
31
|
+
from gapit.records import Manifest
|
|
32
|
+
|
|
33
|
+
_MEMBERS = ("gapit-manifest.json", "records.jsonl") # sorted: manifest < records
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _snapshot_error(db_dir: Path, missing: str) -> DatabaseError:
|
|
37
|
+
"""The uniform archive-content failure: located, machine-stable."""
|
|
38
|
+
return DatabaseError(
|
|
39
|
+
f"cannot snapshot {db_dir}: missing {missing}",
|
|
40
|
+
code="SNAPSHOT_INVALID",
|
|
41
|
+
context={"db_dir": str(db_dir), "missing": missing},
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def make_snapshot(db_dir: Path, dest: Path) -> None:
|
|
46
|
+
"""Write the deterministic snapshot archive of ``db_dir`` to ``dest``.
|
|
47
|
+
|
|
48
|
+
Both members are required (a snapshot of a half-built db dir refuses
|
|
49
|
+
loudly instead of shipping an uninstallable archive).
|
|
50
|
+
"""
|
|
51
|
+
with (
|
|
52
|
+
dest.open("wb") as raw,
|
|
53
|
+
gzip.GzipFile(
|
|
54
|
+
# filename="" keeps the dest path OUT of the gzip header (the
|
|
55
|
+
# fileobj's .name would otherwise be embedded — path-dependent bytes).
|
|
56
|
+
filename="",
|
|
57
|
+
mode="wb",
|
|
58
|
+
fileobj=raw,
|
|
59
|
+
mtime=0,
|
|
60
|
+
) as gz,
|
|
61
|
+
tarfile.open(fileobj=gz, mode="w", format=tarfile.PAX_FORMAT) as tar,
|
|
62
|
+
):
|
|
63
|
+
for name in _MEMBERS:
|
|
64
|
+
source = db_dir / name
|
|
65
|
+
if not source.is_file():
|
|
66
|
+
raise _snapshot_error(db_dir, name)
|
|
67
|
+
data = source.read_bytes()
|
|
68
|
+
info = tarfile.TarInfo(name)
|
|
69
|
+
info.size = len(data)
|
|
70
|
+
info.mtime = 0
|
|
71
|
+
tar.addfile(info, io.BytesIO(data))
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _member(tar: tarfile.TarFile, name: str, archive: Path) -> bytes:
|
|
75
|
+
"""Read one regular-file member; typed SNAPSHOT_INVALID when absent
|
|
76
|
+
(no untyped KeyError escape — the Wave E card.py lesson)."""
|
|
77
|
+
member = next((m for m in tar.getmembers() if m.name == name and m.isfile()), None)
|
|
78
|
+
if member is None:
|
|
79
|
+
raise DatabaseError(
|
|
80
|
+
f"snapshot {archive} has no {name} member",
|
|
81
|
+
code="SNAPSHOT_INVALID",
|
|
82
|
+
context={"archive": str(archive), "missing": name},
|
|
83
|
+
)
|
|
84
|
+
extracted = tar.extractfile(member)
|
|
85
|
+
if extracted is None:
|
|
86
|
+
raise DatabaseError(
|
|
87
|
+
f"{name} in {archive} is not a regular file",
|
|
88
|
+
code="SNAPSHOT_INVALID",
|
|
89
|
+
context={"archive": str(archive), "missing": name},
|
|
90
|
+
)
|
|
91
|
+
return extracted.read()
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def read_snapshot_manifest(archive: Path) -> Manifest:
|
|
95
|
+
"""The archived gapit-manifest.json parsed in-memory (no extraction,
|
|
96
|
+
nothing written) — read-only queries such as `db outdated`."""
|
|
97
|
+
with tarfile.open(archive, "r:gz") as tar:
|
|
98
|
+
return Manifest.model_validate_json(_member(tar, "gapit-manifest.json", archive))
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def extract_snapshot(archive: Path, db_dir: Path) -> Manifest:
|
|
102
|
+
"""Install ``records.jsonl`` from ``archive`` into ``db_dir`` atomically;
|
|
103
|
+
return the ARCHIVED manifest (its upstream_version seeds the fresh build)."""
|
|
104
|
+
with tarfile.open(archive, "r:gz") as tar:
|
|
105
|
+
records = _member(tar, "records.jsonl", archive)
|
|
106
|
+
archived = Manifest.model_validate_json(_member(tar, "gapit-manifest.json", archive))
|
|
107
|
+
temp_path: Path | None = None
|
|
108
|
+
try:
|
|
109
|
+
with NamedTemporaryFile(dir=db_dir, prefix=".records.", suffix=".jsonl", delete=False) as (
|
|
110
|
+
temp
|
|
111
|
+
):
|
|
112
|
+
temp_path = Path(temp.name)
|
|
113
|
+
temp.write(records)
|
|
114
|
+
assert temp_path is not None
|
|
115
|
+
os.replace(temp_path, db_dir / "records.jsonl")
|
|
116
|
+
finally:
|
|
117
|
+
if temp_path is not None:
|
|
118
|
+
temp_path.unlink(missing_ok=True)
|
|
119
|
+
return archived
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
"""upec_expec_vf provider — UPEC/ExPEC virulence genes (FordeGenomics).
|
|
2
|
+
|
|
3
|
+
Transform-only module (Wave B12): turns the downloaded
|
|
4
|
+
``UPEC_ExPEC_VF.tsv`` into typed ``Record``s exactly like upstream
|
|
5
|
+
``abricate-get_db get_upec_expec_vf`` — ``load_tabular`` keys every row by
|
|
6
|
+
its ``Gene name`` column with first occurrence winning (perl ``||=``), then
|
|
7
|
+
ID=Gene name, ACC=``Accession/Source:Begin-End``, DESC=Description,
|
|
8
|
+
SEQ=Sequence. Function is the locked ``virulence`` constant: the TSV Class
|
|
9
|
+
column is source metadata, NOT the function slot (Wave F2c). Sequences
|
|
10
|
+
stay RAW here: normalization, sequence dedupe, and sorting belong to the
|
|
11
|
+
generic :func:`gapit.providers.common.fetch_provider` pipeline.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
import csv
|
|
15
|
+
from collections.abc import Iterator
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
from gapit.errors import DatabaseError
|
|
19
|
+
from gapit.providers.common import Provider
|
|
20
|
+
from gapit.records import Record
|
|
21
|
+
|
|
22
|
+
NAME = "upec_expec_vf"
|
|
23
|
+
|
|
24
|
+
_FUNCTION = ("virulence",) # locked gapit/v1 func vocabulary (Wave F2c)
|
|
25
|
+
|
|
26
|
+
_FILENAME = "UPEC_ExPEC_VF.tsv"
|
|
27
|
+
_REQUIRED_COLUMNS = (
|
|
28
|
+
"Gene name",
|
|
29
|
+
"Accession/Source",
|
|
30
|
+
"Begin",
|
|
31
|
+
"End",
|
|
32
|
+
"Description",
|
|
33
|
+
"Class",
|
|
34
|
+
"Sequence",
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def transform(workdir: Path) -> Iterator[Record]:
|
|
39
|
+
"""Yield one Record per well-formed row of ``workdir/UPEC_ExPEC_VF.tsv``.
|
|
40
|
+
|
|
41
|
+
First row is the header; every column is looked up by NAME. Rows whose
|
|
42
|
+
Gene name or Sequence is empty are skipped (upstream would splice undef
|
|
43
|
+
pieces into the record — deviation noted in the Phase 7 notepad). A
|
|
44
|
+
header missing required columns raises DatabaseError PROVIDER_MALFORMED.
|
|
45
|
+
"""
|
|
46
|
+
path = workdir / _FILENAME
|
|
47
|
+
with path.open(encoding="utf-8", newline="") as handle:
|
|
48
|
+
rows = csv.DictReader(handle, delimiter="\t", restval="")
|
|
49
|
+
missing = [
|
|
50
|
+
column for column in _REQUIRED_COLUMNS if column not in set(rows.fieldnames or ())
|
|
51
|
+
]
|
|
52
|
+
if missing:
|
|
53
|
+
raise DatabaseError(
|
|
54
|
+
f"upec_expec_vf source is missing column(s): {', '.join(missing)}",
|
|
55
|
+
code="PROVIDER_MALFORMED",
|
|
56
|
+
context={"file": str(path), "missing_columns": ",".join(missing)},
|
|
57
|
+
)
|
|
58
|
+
seen: set[str] = set()
|
|
59
|
+
for row in rows:
|
|
60
|
+
gene = row["Gene name"]
|
|
61
|
+
sequence = row["Sequence"]
|
|
62
|
+
if not gene or not sequence or gene in seen:
|
|
63
|
+
continue
|
|
64
|
+
seen.add(gene)
|
|
65
|
+
yield Record(
|
|
66
|
+
db=NAME,
|
|
67
|
+
gene=gene,
|
|
68
|
+
accession=f"{row['Accession/Source']}:{row['Begin']}-{row['End']}",
|
|
69
|
+
function=_FUNCTION,
|
|
70
|
+
product=row["Description"],
|
|
71
|
+
sequence=sequence,
|
|
72
|
+
source_id=gene,
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
PROVIDER = Provider(
|
|
77
|
+
name=NAME,
|
|
78
|
+
description="UPEC/ExPEC virulence genes (FordeGenomics)",
|
|
79
|
+
source_urls=(
|
|
80
|
+
"https://raw.githubusercontent.com/FordeGenomics/ST167_Code/refs/heads/main/UPEC-ExPEC_VF/UPEC_ExPEC_VF.tsv",
|
|
81
|
+
),
|
|
82
|
+
dbtype="nucl",
|
|
83
|
+
transform=transform,
|
|
84
|
+
snapshot=None,
|
|
85
|
+
)
|
gapit/providers/vfdb.py
ADDED
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
"""VFDB provider (set A, nucleotide) — transform only.
|
|
2
|
+
|
|
3
|
+
Upstream ``get_vfdb`` (abricate-get_db 1.4.0) decompresses
|
|
4
|
+
``VFDB_setA_nt.fas.gz`` and, per record, pulls the accession from the
|
|
5
|
+
``<gene>(<db>|<acc>.<version>)`` id suffix and renames the gene to the
|
|
6
|
+
leading paren group of the description::
|
|
7
|
+
|
|
8
|
+
>VFG000676(gb|AAD32411) (lef) anthrax toxin lethal factor precursor ...
|
|
9
|
+
|
|
10
|
+
Perl quirk: a record failing either regex keeps the PREVIOUS record's $1/$2
|
|
11
|
+
(stale fields). We skip such records instead — see the phase 7 notepad.
|
|
12
|
+
|
|
13
|
+
Encoding quirk (Wave E): the real ``VFDB_setA_nt.fas.gz`` is not UTF-8-clean
|
|
14
|
+
(a latin-1 0xA0 nbsp crashes ``gapit.fasta``'s strict decoder), so the
|
|
15
|
+
decompressed bytes are decoded as latin-1 before parsing — see the comment
|
|
16
|
+
in :func:`transform`.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
import gzip
|
|
20
|
+
import re
|
|
21
|
+
from collections.abc import Iterable
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
from tempfile import NamedTemporaryFile
|
|
24
|
+
|
|
25
|
+
from gapit.fasta import iter_fasta
|
|
26
|
+
from gapit.providers.common import Provider
|
|
27
|
+
from gapit.records import Record
|
|
28
|
+
|
|
29
|
+
_NAME = "vfdb"
|
|
30
|
+
|
|
31
|
+
# Upstream regexes verbatim (abricate issue #64 comment by @VGalata):
|
|
32
|
+
# accession = group 2; group 3 (".2" version suffix) is consumed, not used.
|
|
33
|
+
_ID = re.compile(r"^(\w+)\(\w+\|(\w+)(\.\d+)?\)$")
|
|
34
|
+
_DESC = re.compile(r"^\((.*?)\)")
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def transform(workdir: Path) -> Iterable[Record]:
|
|
38
|
+
"""Yield Records from ``workdir/VFDB_setA_nt.fas.gz``.
|
|
39
|
+
|
|
40
|
+
The gunzipped text is decoded as latin-1 into a scratch UTF-8 file for
|
|
41
|
+
:func:`gapit.fasta.iter_fasta` (it takes a path); the scratch file is
|
|
42
|
+
removed on completion or early generator close (argannot precedent).
|
|
43
|
+
|
|
44
|
+
A record failing EITHER regex is skipped; product keeps the full
|
|
45
|
+
description (upstream leaves DESC untouched) and source_id keeps the
|
|
46
|
+
original id.
|
|
47
|
+
"""
|
|
48
|
+
# Byte-preserving decode: upstream perl is byte-oriented here, and
|
|
49
|
+
# latin-1 maps every byte 1:1 — descriptions keep the character and
|
|
50
|
+
# sequence junk flows on to fetch_provider's N-normalization. Our
|
|
51
|
+
# JSONL/FASTA outputs are UTF-8 by contract, so writing the scratch
|
|
52
|
+
# file as UTF-8 re-encodes the character deterministically. Never
|
|
53
|
+
# errors="replace": that would silently rewrite the data.
|
|
54
|
+
text = gzip.decompress((workdir / "VFDB_setA_nt.fas.gz").read_bytes()).decode("latin-1")
|
|
55
|
+
with NamedTemporaryFile(
|
|
56
|
+
mode="w",
|
|
57
|
+
encoding="utf-8",
|
|
58
|
+
newline="\n",
|
|
59
|
+
prefix=".vfdb.",
|
|
60
|
+
suffix=".fas",
|
|
61
|
+
dir=workdir,
|
|
62
|
+
delete=False,
|
|
63
|
+
) as handle:
|
|
64
|
+
handle.write(text)
|
|
65
|
+
scratch = Path(handle.name)
|
|
66
|
+
try:
|
|
67
|
+
for fasta in iter_fasta(scratch):
|
|
68
|
+
id_match = _ID.match(fasta.id)
|
|
69
|
+
desc_match = _DESC.match(fasta.description)
|
|
70
|
+
if id_match is None or desc_match is None:
|
|
71
|
+
continue
|
|
72
|
+
yield Record(
|
|
73
|
+
db=_NAME,
|
|
74
|
+
gene=desc_match.group(1),
|
|
75
|
+
sequence=fasta.sequence,
|
|
76
|
+
accession=id_match.group(2),
|
|
77
|
+
function=("virulence",),
|
|
78
|
+
product=fasta.description,
|
|
79
|
+
source_id=fasta.id,
|
|
80
|
+
)
|
|
81
|
+
finally:
|
|
82
|
+
scratch.unlink(missing_ok=True)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
PROVIDER = Provider(
|
|
86
|
+
name=_NAME,
|
|
87
|
+
description="VFDB virulence factors (set A, nucleotide)",
|
|
88
|
+
source_urls=("http://www.mgc.ac.cn/VFs/Down/VFDB_setA_nt.fas.gz",),
|
|
89
|
+
dbtype="nucl",
|
|
90
|
+
transform=transform,
|
|
91
|
+
snapshot="vfdb.tar.gz",
|
|
92
|
+
)
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
"""Victors provider (virulence factors, nucleotide) — transform only.
|
|
2
|
+
|
|
3
|
+
Upstream ``get_victors`` (abricate-get_db 1.4.0) cross-references two
|
|
4
|
+
phidias.us downloads, both served under ``.php`` URLs:
|
|
5
|
+
|
|
6
|
+
- ``gen_downloads.php`` — nucleotide CDS content (.ffn), ids like
|
|
7
|
+
``gi|115534241:2616-3152`` carrying the source GI and coordinates;
|
|
8
|
+
- ``gen_downloads_protein.php`` — protein content (.faa), headers like
|
|
9
|
+
``>gi|115534244|ref|YP_783826.1| hypothetical protein pCJ01p4
|
|
10
|
+
[Campylobacter jejuni]``.
|
|
11
|
+
|
|
12
|
+
Protein headers are keyed by gi; each .ffn record looks its gi up in that
|
|
13
|
+
map for accession and product, falling back to ``gi|<gi>:<start>-<stop>``
|
|
14
|
+
and ``hypothetical protein``. The gene stays the ORIGINAL .ffn id —
|
|
15
|
+
upstream never renames it here.
|
|
16
|
+
|
|
17
|
+
Perl quirks (phase 7 notepad): upstream's ``.`` in both regexes matches
|
|
18
|
+
the literal ``|`` (escaped here); a protein header failing the regex is
|
|
19
|
+
skipped (``next unless``); an .ffn id failing its regex would reuse the
|
|
20
|
+
PREVIOUS match's stale capture variables — such records are skipped, the
|
|
21
|
+
same policy as the vfdb provider. Upstream's product line
|
|
22
|
+
(``$s->{DESC} =~ ... || 'hypothetical protein'``) is a void-context match
|
|
23
|
+
whose result is discarded; the notepad-locked digest semantics (map
|
|
24
|
+
product or ``hypothetical protein``) are implemented instead.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
import re
|
|
28
|
+
from collections.abc import Iterable
|
|
29
|
+
from pathlib import Path
|
|
30
|
+
|
|
31
|
+
from gapit.fasta import iter_fasta
|
|
32
|
+
from gapit.providers.common import Provider
|
|
33
|
+
from gapit.records import Record
|
|
34
|
+
|
|
35
|
+
_NAME = "victors"
|
|
36
|
+
_FFN_NAME = "gen_downloads.php" # nucleotide CDS content
|
|
37
|
+
_FAA_NAME = "gen_downloads_protein.php" # protein content
|
|
38
|
+
_HYPOTHETICAL = "hypothetical protein"
|
|
39
|
+
_FUNCTION = ("virulence",) # locked gapit/v1 func vocabulary (Wave F2c)
|
|
40
|
+
|
|
41
|
+
# Upstream m"^>gi.(\d+).ref.([^|]+). ([^[]+)" — its dots matched the pipes.
|
|
42
|
+
# \s+ eats the id/product separator; the product runs to end of line.
|
|
43
|
+
_FAA_HEADER = re.compile(r">gi\|(\d+)\|ref\|([^|]+)\|\s+(.+)")
|
|
44
|
+
# Upstream m/gi.(\d+):(\d+)-(\d+)/ searched against the .ffn record id.
|
|
45
|
+
_FFN_ID = re.compile(r"gi\|(\d+):(\d+)-(\d+)")
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _read_protein_map(path: Path) -> dict[str, tuple[str, str]]:
|
|
49
|
+
"""gi -> (accession, product) from .faa header lines.
|
|
50
|
+
|
|
51
|
+
Headers failing the pattern are skipped (upstream ``next unless``);
|
|
52
|
+
the product stops at the first `` [`` strain bracket (upstream
|
|
53
|
+
``([^[]+)``). A later duplicate gi overwrites an earlier one, exactly
|
|
54
|
+
like upstream's hash assignment.
|
|
55
|
+
"""
|
|
56
|
+
gi_map: dict[str, tuple[str, str]] = {}
|
|
57
|
+
with path.open(encoding="utf-8") as handle:
|
|
58
|
+
for line in handle:
|
|
59
|
+
header = _FAA_HEADER.match(line)
|
|
60
|
+
if header is None:
|
|
61
|
+
continue
|
|
62
|
+
gi, accession, product = (
|
|
63
|
+
header.group(1),
|
|
64
|
+
header.group(2),
|
|
65
|
+
header.group(3).split(" [", 1)[0],
|
|
66
|
+
)
|
|
67
|
+
gi_map[gi] = (accession, product)
|
|
68
|
+
return gi_map
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def transform(workdir: Path) -> Iterable[Record]:
|
|
72
|
+
"""Yield Records from the two phidias.us downloads in ``workdir``.
|
|
73
|
+
|
|
74
|
+
Each .ffn record keeps its ORIGINAL id as the gene; the gi carved out
|
|
75
|
+
of that id looks accession/product up in the protein map, falling back
|
|
76
|
+
to ``gi|<gi>:<start>-<stop>`` / ``hypothetical protein`` for an unknown
|
|
77
|
+
gi. Records whose id carries no gi coordinates are skipped (upstream
|
|
78
|
+
would build them from the previous match's stale captures).
|
|
79
|
+
"""
|
|
80
|
+
gi_map = _read_protein_map(workdir / _FAA_NAME)
|
|
81
|
+
for fasta in iter_fasta(workdir / _FFN_NAME):
|
|
82
|
+
coords = _FFN_ID.search(fasta.id)
|
|
83
|
+
if coords is None:
|
|
84
|
+
continue
|
|
85
|
+
gi = coords.group(1)
|
|
86
|
+
fallback = f"gi|{gi}:{coords.group(2)}-{coords.group(3)}"
|
|
87
|
+
accession, product = gi_map.get(gi, (fallback, _HYPOTHETICAL))
|
|
88
|
+
yield Record(
|
|
89
|
+
db=_NAME,
|
|
90
|
+
gene=fasta.id,
|
|
91
|
+
sequence=fasta.sequence,
|
|
92
|
+
accession=accession,
|
|
93
|
+
function=_FUNCTION,
|
|
94
|
+
product=product,
|
|
95
|
+
source_id=gi,
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
PROVIDER = Provider(
|
|
100
|
+
name=_NAME,
|
|
101
|
+
description="Victors virulence factors",
|
|
102
|
+
source_urls=(
|
|
103
|
+
"http://phidias.us/victors/downloads/gen_downloads.php",
|
|
104
|
+
"http://phidias.us/victors/downloads/gen_downloads_protein.php",
|
|
105
|
+
),
|
|
106
|
+
dbtype="nucl",
|
|
107
|
+
transform=transform,
|
|
108
|
+
snapshot=None,
|
|
109
|
+
)
|
gapit/py.typed
ADDED
|
File without changes
|