gapit 0.2.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. gapit/__init__.py +3 -0
  2. gapit/blast.py +239 -0
  3. gapit/cli.py +128 -0
  4. gapit/cmd_db.py +113 -0
  5. gapit/cmd_db_build.py +72 -0
  6. gapit/cmd_db_install.py +126 -0
  7. gapit/cmd_db_outdated.py +51 -0
  8. gapit/cmd_db_search.py +66 -0
  9. gapit/cmd_screen.py +214 -0
  10. gapit/cmd_summary.py +65 -0
  11. gapit/config.py +51 -0
  12. gapit/data/snapshots/card.tar.gz +0 -0
  13. gapit/data/snapshots/vfdb.tar.gz +0 -0
  14. gapit/db.py +226 -0
  15. gapit/db_build_ops.py +210 -0
  16. gapit/db_ops.py +128 -0
  17. gapit/db_query_ops.py +252 -0
  18. gapit/dbbuild.py +207 -0
  19. gapit/dbcodec.py +117 -0
  20. gapit/dispatch.py +31 -0
  21. gapit/errors.py +87 -0
  22. gapit/fasta.py +123 -0
  23. gapit/formats/__init__.py +1 -0
  24. gapit/formats/json.py +309 -0
  25. gapit/formats/md.py +190 -0
  26. gapit/formats/schemas.py +30 -0
  27. gapit/formats/summary.py +103 -0
  28. gapit/formats/tsv.py +45 -0
  29. gapit/hits.py +107 -0
  30. gapit/mcp.py +158 -0
  31. gapit/mcp_schemas.py +123 -0
  32. gapit/mcp_tools.py +289 -0
  33. gapit/minimap.py +20 -0
  34. gapit/minimap2_run.py +114 -0
  35. gapit/paf.py +115 -0
  36. gapit/proctools.py +24 -0
  37. gapit/providers/__init__.py +39 -0
  38. gapit/providers/argannot.py +94 -0
  39. gapit/providers/bacmet2.py +59 -0
  40. gapit/providers/card.py +150 -0
  41. gapit/providers/common.py +245 -0
  42. gapit/providers/ecoh.py +63 -0
  43. gapit/providers/ecoli_vf.py +74 -0
  44. gapit/providers/megares.py +71 -0
  45. gapit/providers/ncbi.py +103 -0
  46. gapit/providers/plasmidfinder.py +69 -0
  47. gapit/providers/resfinder.py +123 -0
  48. gapit/providers/snapshots.py +119 -0
  49. gapit/providers/upec_expec_vf.py +85 -0
  50. gapit/providers/vfdb.py +92 -0
  51. gapit/providers/victors.py +109 -0
  52. gapit/py.typed +0 -0
  53. gapit/reads.py +221 -0
  54. gapit/records.py +152 -0
  55. gapit/report.py +25 -0
  56. gapit/screening.py +145 -0
  57. gapit/screening_reads.py +255 -0
  58. gapit/seqconvert.py +203 -0
  59. gapit/summary.py +151 -0
  60. gapit-0.2.2.dist-info/METADATA +183 -0
  61. gapit-0.2.2.dist-info/RECORD +64 -0
  62. gapit-0.2.2.dist-info/WHEEL +4 -0
  63. gapit-0.2.2.dist-info/entry_points.txt +3 -0
  64. gapit-0.2.2.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,245 @@
1
+ """Wave B0 shared provider infrastructure: the generic get_db pipeline.
2
+
3
+ Every concrete provider (Wave B1..B12) is a :class:`Provider` value — a name,
4
+ a description, a tuple of source URLs, a dbtype, and a ``transform`` that
5
+ turns downloaded files into typed ``Record``s. :func:`fetch_provider` runs
6
+ the generic abricate-get_db flow around it: download each URL into a scratch
7
+ workdir under ``db_dir``, transform, normalize sequences and function classes
8
+ (upstream load_fasta semantics), dedupe by exact normalized sequence
9
+ (first wins), sort by gene, persist ``records.jsonl``, then delegate to
10
+ :func:`gapit.dbbuild.build_database` for the ``sequences`` FASTA, the BLAST
11
+ index, and the manifest (written last, certifying the build).
12
+
13
+ Upstream's ``is_full_gene`` is deliberately NOT ported: its map result is
14
+ discarded — a no-op.
15
+ """
16
+
17
+ import os
18
+ import re
19
+ import urllib.request
20
+ from collections.abc import Callable, Iterable, Sequence
21
+ from dataclasses import dataclass
22
+ from importlib.resources import files
23
+ from pathlib import Path
24
+ from tempfile import TemporaryDirectory
25
+ from typing import Literal
26
+ from urllib.parse import urlparse
27
+
28
+ from gapit import __version__
29
+ from gapit.dbbuild import build_database
30
+ from gapit.errors import DatabaseError
31
+ from gapit.proctools import note
32
+ from gapit.providers.snapshots import extract_snapshot, read_snapshot_manifest
33
+ from gapit.records import Manifest, Record, write_records
34
+
35
+ _NUCL_JUNK = re.compile(r"[^AGCT]")
36
+ _PROT_JUNK = re.compile(r"[^A-Z]")
37
+ _WHITESPACE = re.compile(r"\s+")
38
+ _CHUNK_BYTES = 1 << 20
39
+ # Bounded fetch: socket.timeout (an OSError) already maps to DOWNLOAD_FAILED below.
40
+ _DOWNLOAD_TIMEOUT_SECONDS = 60
41
+
42
+ # mgc.ac.cn (vfdb) 403s "Python-urllib" UAs specifically while serving
43
+ # browser-ish clients (Wave E diagnosis): the Mozilla compatibility token
44
+ # passes those naive UA filters and the gapit/<version> suffix still
45
+ # identifies the tool to servers that log it.
46
+ _USER_AGENT = f"Mozilla/5.0 (compatible; gapit/{__version__})"
47
+
48
+ Dbtype = Literal["nucl", "prot"]
49
+
50
+
51
+ @dataclass(frozen=True, slots=True)
52
+ class Provider:
53
+ """One database provider: everything fetch_provider needs to know.
54
+
55
+ ``transform`` receives the download workdir (each source URL saved under
56
+ its basename) and yields Records; setting ``db`` to the provider name is
57
+ the provider module's job. Sequences and function classes may arrive
58
+ raw — fetch_provider normalizes both.
59
+
60
+ ``snapshot`` names a bundled archive (Wave G) under
61
+ ``gapit/data/snapshots/`` — None (default) for network-only providers.
62
+ """
63
+
64
+ name: str
65
+ description: str
66
+ source_urls: tuple[str, ...]
67
+ dbtype: Dbtype
68
+ transform: Callable[[Path], Iterable[Record]]
69
+ snapshot: str | None = None
70
+
71
+
72
+ def _basename(url: str) -> str:
73
+ """Final path segment of a URL, query string excluded ('' for a bare host)."""
74
+ return urlparse(url).path.rstrip("/").rsplit("/", 1)[-1]
75
+
76
+
77
+ def _download(url: str, dest: Path) -> None:
78
+ """Fetch ``url`` into ``dest`` atomically: urlopen a Request carrying
79
+ the module User-Agent, stream the body into a hidden .part file beside
80
+ ``dest`` in chunks, then os.replace. file:// URLs work (tests depend on
81
+ it). Any URLError/OSError/ValueError becomes DatabaseError
82
+ ``DOWNLOAD_FAILED`` with the URL in context. No shell, ever.
83
+ """
84
+ temp = dest.with_name(f".{dest.name}.part")
85
+ request = urllib.request.Request(url, headers={"User-Agent": _USER_AGENT})
86
+ try:
87
+ with (
88
+ urllib.request.urlopen(request, timeout=_DOWNLOAD_TIMEOUT_SECONDS) as response,
89
+ temp.open("wb") as out,
90
+ ):
91
+ while chunk := response.read(_CHUNK_BYTES):
92
+ out.write(chunk)
93
+ os.replace(temp, dest)
94
+ except (OSError, ValueError) as exc:
95
+ raise DatabaseError(
96
+ f"failed to download {url}: {exc}",
97
+ code="DOWNLOAD_FAILED",
98
+ context={"url": url},
99
+ ) from exc
100
+ finally:
101
+ temp.unlink(missing_ok=True)
102
+
103
+
104
+ def _normalize_sequence(sequence: str, dbtype: Dbtype) -> str:
105
+ """Uppercase; then nucl: non-AGCT -> 'N', prot: non-A-Z -> 'X'
106
+ (upstream load_fasta semantics)."""
107
+ upper = sequence.upper()
108
+ match dbtype:
109
+ case "nucl":
110
+ return _NUCL_JUNK.sub("N", upper)
111
+ case "prot":
112
+ return _PROT_JUNK.sub("X", upper)
113
+
114
+
115
+ def _normalize_function(classes: Sequence[str]) -> tuple[str, ...]:
116
+ """Sorted classes with each whitespace run collapsed to one '_'.
117
+
118
+ Upstream sorts, joins with ';', then globally substitutes s/\\s+/_/g —
119
+ ';' carries no whitespace, so per-class substitution is equivalent.
120
+ """
121
+ return tuple(_WHITESPACE.sub("_", drug_class) for drug_class in sorted(classes))
122
+
123
+
124
+ def _normalize_record(record: Record, dbtype: Dbtype) -> Record:
125
+ """Apply the two per-record normalizations; db, gene, and the rest stay."""
126
+ return record.model_copy(
127
+ update={
128
+ "sequence": _normalize_sequence(record.sequence, dbtype),
129
+ "function": _normalize_function(record.function),
130
+ }
131
+ )
132
+
133
+
134
+ def _dedupe(records: Sequence[Record]) -> tuple[tuple[Record, ...], int]:
135
+ """Drop later records whose (already normalized) SEQUENCE was seen; first
136
+ wins. Returns the kept records and the dropped count (for the caller's
137
+ stderr note). Duplicate gene NAMES are allowed and kept — upstream too.
138
+ """
139
+ seen: set[str] = set()
140
+ kept: list[Record] = []
141
+ dropped = 0
142
+ for record in records:
143
+ if record.sequence in seen:
144
+ dropped += 1
145
+ else:
146
+ seen.add(record.sequence)
147
+ kept.append(record)
148
+ return tuple(kept), dropped
149
+
150
+
151
+ def _snapshot_path(provider: Provider) -> Path | None:
152
+ """Filesystem path of the provider's bundled snapshot archive, or None
153
+ when the provider has none / the package data is absent (missing
154
+ archives degrade silently to the network path — Wave G). Tests
155
+ monkeypatch THIS seam (string setattr), never importlib itself; gapit
156
+ ships as a regular filesystem package, so the str() round-trip is safe.
157
+ """
158
+ if provider.snapshot is None:
159
+ return None
160
+ archive = files("gapit").joinpath("data").joinpath("snapshots").joinpath(provider.snapshot)
161
+ return Path(str(archive)) if archive.is_file() else None
162
+
163
+
164
+ def bundled_snapshot_manifest(provider: Provider) -> Manifest | None:
165
+ """The provider's bundled snapshot manifest read in-memory, or None when
166
+ the provider ships no resolvable snapshot archive (read-only queries:
167
+ `db outdated` — same seam fetch_provider uses, so tests patch
168
+ ``_snapshot_path`` and both paths see the fake)."""
169
+ archive = _snapshot_path(provider)
170
+ return None if archive is None else read_snapshot_manifest(archive)
171
+
172
+
173
+ def fetch_provider(
174
+ provider: Provider,
175
+ db_dir: Path,
176
+ *,
177
+ fetched_at: str,
178
+ force: bool = False,
179
+ quiet: bool = True,
180
+ from_source: bool = False,
181
+ debug: bool = False,
182
+ ) -> Manifest:
183
+ """Run the generic provider pipeline into ``db_dir`` (created if needed).
184
+
185
+ Snapshot path (Wave G): unless ``from_source`` is set, a provider with a
186
+ resolvable bundled snapshot installs its archived ``records.jsonl`` and
187
+ rebuilds every index locally — ``upstream_version`` comes from the
188
+ archived manifest, everything else is fresh. Otherwise the network path
189
+ runs: download each URL, transform, normalize, dedupe, sort.
190
+
191
+ Raises DatabaseError (exit 4): ``DB_ALREADY_EXISTS`` when a manifest is
192
+ present and force is off (upstream: "Won't overwrite existing (use
193
+ --force)"), ``DOWNLOAD_FAILED`` for a failed source download,
194
+ ``SNAPSHOT_INVALID`` for a corrupt snapshot archive, and
195
+ ``PROVIDER_EMPTY`` when the transform+dedupe leaves zero records.
196
+ """
197
+ if (db_dir / "gapit-manifest.json").is_file() and not force:
198
+ raise DatabaseError(
199
+ f"won't overwrite existing database {provider.name} (use --force)",
200
+ code="DB_ALREADY_EXISTS",
201
+ context={"db": provider.name},
202
+ )
203
+ db_dir.mkdir(parents=True, exist_ok=True)
204
+ if not from_source:
205
+ snapshot = _snapshot_path(provider)
206
+ if snapshot is not None:
207
+ archived = extract_snapshot(snapshot, db_dir)
208
+ note(quiet, f"installed {provider.name} from bundled snapshot {snapshot.name}")
209
+ return build_database(
210
+ db_dir,
211
+ name=provider.name,
212
+ dbtype=provider.dbtype,
213
+ source_urls=provider.source_urls,
214
+ fetched_at=fetched_at,
215
+ upstream_version=archived.upstream_version,
216
+ quiet=quiet,
217
+ debug=debug,
218
+ )
219
+ with TemporaryDirectory(dir=db_dir, prefix=".download.") as workdir_name:
220
+ workdir = Path(workdir_name)
221
+ for url in provider.source_urls:
222
+ _download(url, workdir / _basename(url))
223
+ note(quiet, f"downloaded {len(provider.source_urls)} source file(s)")
224
+ records = tuple(
225
+ _normalize_record(record, provider.dbtype) for record in provider.transform(workdir)
226
+ )
227
+ kept, dropped = _dedupe(records)
228
+ note(quiet, f"read {len(records)} records from {provider.name}")
229
+ note(quiet, f"dropped {dropped} duplicate sequence(s), kept {len(kept)}")
230
+ if not kept:
231
+ raise DatabaseError(
232
+ f"provider {provider.name} yielded no records",
233
+ code="PROVIDER_EMPTY",
234
+ context={"db": provider.name},
235
+ )
236
+ write_records(sorted(kept, key=lambda record: record.gene), db_dir / "records.jsonl")
237
+ return build_database(
238
+ db_dir,
239
+ name=provider.name,
240
+ dbtype=provider.dbtype,
241
+ source_urls=provider.source_urls,
242
+ fetched_at=fetched_at,
243
+ quiet=quiet,
244
+ debug=debug,
245
+ )
@@ -0,0 +1,63 @@
1
+ """ecoh database provider — E. coli O and H antigen genes (srst2 EcOH).
2
+
3
+ Transform-only Wave B10 module: ``PROVIDER`` wires the pinned metadata into
4
+ the frozen B0 :class:`gapit.providers.common.Provider` contract and
5
+ :func:`transform` parses the downloaded ``EcOH.fasta`` into typed Records.
6
+ Functional categories are allele-prefix derived — ``fliC*`` → H-antigen,
7
+ ``wzx``/``wzy``/``wzt``/``wzm`` → O-antigen — fallback ``antigen``.
8
+ """
9
+
10
+ from collections.abc import Iterator
11
+ from pathlib import Path
12
+
13
+ from gapit.fasta import iter_fasta
14
+ from gapit.providers.common import Provider
15
+ from gapit.records import Record
16
+
17
+ NAME = "ecoh"
18
+ SOURCE_FILE = "EcOH.fasta"
19
+
20
+
21
+ def transform(workdir: Path) -> Iterator[Record]:
22
+ """Parse ``workdir/EcOH.fasta`` into Records (db = ecoh).
23
+
24
+ srst2 header convention: ``>cluster__gene__allele__seq_id acc;word;word``
25
+ (abricate-get_db ``get_ecoh``). The id splits on ``__`` into exactly 4
26
+ parts and the gene is part 3 — the allele, e.g. ``fliC-H1``. The
27
+ description splits on ``;``: the first piece is the accession, the
28
+ space-joined rest is the product. Ids without exactly 4 ``__``-parts are
29
+ skipped: upstream perl would read an undefined gene field there, and a
30
+ typed Record cannot carry one. ``function`` is allele-prefix derived
31
+ per the module-docstring rule (fallback ``antigen``).
32
+ """
33
+ for fasta in iter_fasta(workdir / SOURCE_FILE):
34
+ parts = fasta.id.split("__")
35
+ if len(parts) != 4:
36
+ continue
37
+ gene = parts[2]
38
+ if gene.startswith("fliC"):
39
+ function: tuple[str, ...] = ("H-antigen",)
40
+ elif gene.startswith(("wzx", "wzy", "wzt", "wzm")):
41
+ function = ("O-antigen",)
42
+ else:
43
+ function = ("antigen",)
44
+ description = fasta.description.split(";")
45
+ yield Record(
46
+ db=NAME,
47
+ gene=gene,
48
+ accession=description[0],
49
+ function=function,
50
+ product=" ".join(description[1:]),
51
+ sequence=fasta.sequence,
52
+ source_id=fasta.id,
53
+ )
54
+
55
+
56
+ PROVIDER = Provider(
57
+ name=NAME,
58
+ description="E. coli O and H antigens (srst2 EcOH)",
59
+ source_urls=("https://raw.githubusercontent.com/katholt/srst2/master/data/EcOH.fasta",),
60
+ dbtype="nucl",
61
+ transform=transform,
62
+ snapshot=None,
63
+ )
@@ -0,0 +1,74 @@
1
+ r"""ecoli_vf provider (phac-nml E. coli virulence factors) — transform only.
2
+
3
+ Upstream ``get_ecoli_vf`` (abricate-get_db 1.4.0) parses
4
+ ``repaired_ecoli_vfs_shortnames.ffn`` in three steps per record::
5
+
6
+ >VFG000748(gi:2865308) (espF) EspF [EspF (VF0182)] [Escherichia coli ...]
7
+
8
+ 1. id ``^(\w+)(?:\((.*?)\))?$`` -> base id + optional paren accession
9
+ (upstream ``die``s on a non-match; we SKIP the record instead —
10
+ same deviation family as vfdb/ecoh, one bad header must not kill
11
+ a fetch)
12
+ 2. accession = ``$2 || $1`` -> paren content; an absent OR EMPTY
13
+ capture falls back to the base id (perl ``||`` is falsy-based)
14
+ 3. description: repeatedly strip trailing bracket groups
15
+ (``s/\s\[.*?\]$//g``), then ``^(?:\((.*?)\)\s+)?(.*)$`` -> a
16
+ leading paren group RENAMES the gene (overrides the base id), the
17
+ remainder becomes the product (``$DESC || $ID`` fallback).
18
+ """
19
+
20
+ import re
21
+ from collections.abc import Iterable
22
+ from pathlib import Path
23
+
24
+ from gapit.fasta import iter_fasta
25
+ from gapit.providers.common import Provider
26
+ from gapit.records import Record
27
+
28
+ _NAME = "ecoli_vf"
29
+ _SOURCE_FILE = "repaired_ecoli_vfs_shortnames.ffn"
30
+
31
+ # Upstream regexes verbatim (the /x id regex ignores whitespace; joined here).
32
+ _ID = re.compile(r"^(\w+)(?:\((.*?)\))?$")
33
+ _TRAILING_BRACKETS = re.compile(r"\s\[.*?\]$")
34
+ _DESC = re.compile(r"^(?:\((.*?)\)\s+)?(.*)$")
35
+
36
+
37
+ def transform(workdir: Path) -> Iterable[Record]:
38
+ """Yield Records from ``workdir/repaired_ecoli_vfs_shortnames.ffn``.
39
+
40
+ Sequence and the leading/trailing-stripped description arrive via
41
+ gapit.fasta; gene/accession/product derive per the docstring steps.
42
+ """
43
+ for fasta in iter_fasta(workdir / _SOURCE_FILE):
44
+ id_match = _ID.match(fasta.id)
45
+ if id_match is None:
46
+ continue
47
+ base = id_match.group(1)
48
+ accession = id_match.group(2) or base # perl $2 || $1: '' is falsy too
49
+ description = fasta.description
50
+ while (stripped := _TRAILING_BRACKETS.sub("", description)) != description:
51
+ description = stripped
52
+ desc_match = _DESC.match(description)
53
+ assert desc_match is not None # (.*) matches any string — cannot fail
54
+ gene = desc_match.group(1) or base
55
+ product = desc_match.group(2)
56
+ yield Record(
57
+ db=_NAME,
58
+ gene=gene,
59
+ sequence=fasta.sequence,
60
+ accession=accession,
61
+ function=("virulence",),
62
+ product=product or gene, # save_fasta: -desc => ($DESC || $ID)
63
+ source_id=base,
64
+ )
65
+
66
+
67
+ PROVIDER = Provider(
68
+ name=_NAME,
69
+ description="E. coli virulence factors (phac-nml)",
70
+ source_urls=("https://github.com/phac-nml/ecoli_vf/raw/master/data/" + _SOURCE_FILE,),
71
+ dbtype="nucl",
72
+ transform=transform,
73
+ snapshot=None,
74
+ )
@@ -0,0 +1,71 @@
1
+ """megares database provider — MEGARes v3 antimicrobial resistance genes.
2
+
3
+ Transform-only Wave B9 module: ``PROVIDER`` wires the pinned metadata into
4
+ the frozen B0 :class:`gapit.providers.common.Provider` contract and
5
+ :func:`transform` unpacks the downloaded ``megares_v3.00.zip`` and parses
6
+ every ``megares_drugs_*.fasta`` inside it into typed Records.
7
+ """
8
+
9
+ import zipfile
10
+ from collections.abc import Iterator
11
+ from pathlib import Path
12
+
13
+ from gapit.fasta import iter_fasta
14
+ from gapit.providers.common import Provider
15
+ from gapit.records import Record
16
+
17
+ NAME = "megares"
18
+ ARCHIVE = "megares_v3.00.zip"
19
+ _DATABASE_GLOB = "megares_drugs_*.fasta"
20
+
21
+
22
+ def transform(workdir: Path) -> Iterator[Record]:
23
+ """Extract ``workdir/megares_v3.00.zip`` and parse every nested
24
+ ``megares_drugs_*.fasta`` into Records (db = megares).
25
+
26
+ MEGARes v3 header convention (abricate-get_db ``get_megares``):
27
+ ``>id|type|class|mech|group|note`` where the note field exists only on
28
+ records requiring SNP confirmation — a non-empty note skips the record
29
+ (SPEC §8). Keepers map gene=group, accession=source_id=id (the MEG_
30
+ number), product=colon-joined type/class/mech/group; function = the
31
+ ``class`` field (x[2], e.g. Tetracyclines) as a 1-tuple — upstream
32
+ sets no ABX, the class IS the functional category. Archives often nest
33
+ the fasta one directory deep — upstream
34
+ ``unzip -j`` flattens, here the glob is recursive (sorted, so multi-file
35
+ archives parse deterministically).
36
+
37
+ Documented deviation: upstream perl splits into six list variables and
38
+ keeps records even when fields are missing (undef gene, empty product
39
+ pieces); gapit skips ids with fewer than 5 pipe-fields or an empty among
40
+ the 5 leading ones. ``split('|', 5)`` mirrors perl's implicit
41
+ list-assignment limit: any pipes past the group fold into the note, and
42
+ a folded note is non-empty, so over-long ids skip exactly like upstream.
43
+ """
44
+ with zipfile.ZipFile(workdir / ARCHIVE) as archive:
45
+ archive.extractall(workdir)
46
+ for fasta_path in sorted(workdir.rglob(_DATABASE_GLOB)):
47
+ for fasta in iter_fasta(fasta_path):
48
+ parts = fasta.id.split("|", 5)
49
+ if len(parts) < 5 or "" in parts[:5]:
50
+ continue
51
+ if len(parts) == 6 and parts[5]:
52
+ continue
53
+ yield Record(
54
+ db=NAME,
55
+ gene=parts[4],
56
+ accession=parts[0],
57
+ function=(parts[2],),
58
+ product=":".join(parts[1:5]),
59
+ sequence=fasta.sequence,
60
+ source_id=parts[0],
61
+ )
62
+
63
+
64
+ PROVIDER = Provider(
65
+ name=NAME,
66
+ description="MEGARes antimicrobial resistance genes",
67
+ source_urls=("https://www.meglab.org/downloads/megares_v3.00.zip",),
68
+ dbtype="nucl",
69
+ transform=transform,
70
+ snapshot=None,
71
+ )
@@ -0,0 +1,103 @@
1
+ """ncbi database provider — NCBI AMRFinderPlus curated AMR (Wave B1).
2
+
3
+ Transform-only module mirroring abricate-get_db ``get_ncbi``: pair
4
+ ``AMR_CDS.fa`` records with ``ReferenceGeneCatalog.txt`` rows keyed by
5
+ ``refseq_nucleotide_accession`` (column 10, 0-based), keeping only plain
6
+ (non-fusion) genes whose catalog row is scope core / type AMR / subtype AMR.
7
+ ``PROVIDER`` wires the pinned metadata into the frozen B0
8
+ :class:`gapit.providers.common.Provider` contract.
9
+ """
10
+
11
+ import re
12
+ from collections.abc import Iterator
13
+ from pathlib import Path
14
+
15
+ from gapit.fasta import iter_fasta
16
+ from gapit.providers.common import Provider
17
+ from gapit.records import Record
18
+
19
+ NAME = "ncbi"
20
+ AMR_CDS_FILE = "AMR_CDS.fa"
21
+ CATALOG_FILE = "ReferenceGeneCatalog.txt"
22
+ _ACCESSION_COLUMN = 10
23
+ _VERSIONED_ACCESSION = re.compile(r"\.\d+$")
24
+ _LATEST = (
25
+ "https://ftp.ncbi.nlm.nih.gov/pathogen/Antimicrobial_resistance/AMRFinderPlus/database/latest"
26
+ )
27
+
28
+
29
+ def _load_catalog(path: Path) -> dict[str, dict[str, str]]:
30
+ """ReferenceGeneCatalog rows as ``{accession: {header name: value}}``.
31
+
32
+ The first line is the header; every later row is keyed by column 10
33
+ (``refseq_nucleotide_accession``). Duplicate accessions keep the FIRST
34
+ row (upstream's ``||=``), and rows are padded to the header width so
35
+ lookups of unused trailing columns cannot fail on short rows.
36
+ """
37
+ rows: dict[str, dict[str, str]] = {}
38
+ header: list[str] | None = None
39
+ with path.open(encoding="utf-8") as handle:
40
+ for line in handle:
41
+ columns = line.rstrip("\r\n").split("\t")
42
+ if header is None:
43
+ header = columns
44
+ continue
45
+ width = max(len(header), _ACCESSION_COLUMN + 1)
46
+ padded = columns + [""] * (width - len(columns))
47
+ if (acc := padded[_ACCESSION_COLUMN]) and acc not in rows:
48
+ rows[acc] = dict(zip(header, padded, strict=False)) # perl zip: shorter wins
49
+ return rows
50
+
51
+
52
+ def transform(workdir: Path) -> Iterator[Record]:
53
+ """Parse ``workdir/AMR_CDS.fa`` + ``ReferenceGeneCatalog.txt`` (db = ncbi).
54
+
55
+ AMRFinderPlus fasta ids are ``pi|acc|fp|fn|gene|fam|prod`` — perl's
56
+ 7-variable ``split`` assignment imposes LIMIT 7, so everything past the
57
+ sixth pipe belongs to ``prod``. Records are skipped — passively, like
58
+ upstream — unless the id yields 7 non-empty fields, fp/fn are both "1"
59
+ (fusion filter), and the accession (``.1`` appended unless already
60
+ versioned, e.g. ``NG_050200`` -> ``NG_050200.1``) matches a catalog row
61
+ with scope ``core``, type ``AMR``, and subtype ``AMR``. Sequences and
62
+ function categories stay raw: fetch_provider owns normalization.
63
+ """
64
+ catalog = _load_catalog(workdir / CATALOG_FILE)
65
+ for fasta in iter_fasta(workdir / AMR_CDS_FILE):
66
+ fields = fasta.id.split("|")
67
+ if len(fields) < 7:
68
+ continue
69
+ pi, acc, fp, fn, gene, fam = fields[:6]
70
+ prod = "|".join(fields[6:])
71
+ if not all((pi, acc, fp, fn, gene, fam, prod)):
72
+ continue
73
+ if fp != "1" or fn != "1":
74
+ continue
75
+ if not _VERSIONED_ACCESSION.search(acc):
76
+ acc += ".1"
77
+ row = catalog.get(acc)
78
+ if row is None:
79
+ continue
80
+ if row["scope"] != "core" or row["type"] != "AMR" or row["subtype"] != "AMR":
81
+ continue
82
+ yield Record(
83
+ db=NAME,
84
+ gene=gene,
85
+ accession=row["refseq_nucleotide_accession"],
86
+ function=tuple(part for part in row["subclass"].split("/") if part),
87
+ product=prod.replace("_", " "),
88
+ sequence=fasta.sequence,
89
+ source_id=pi,
90
+ )
91
+
92
+
93
+ PROVIDER = Provider(
94
+ name=NAME,
95
+ description="NCBI AMRFinderPlus (reference finder) curated AMR",
96
+ source_urls=(
97
+ f"{_LATEST}/AMR_CDS.fa",
98
+ f"{_LATEST}/ReferenceGeneCatalog.txt",
99
+ ),
100
+ dbtype="nucl",
101
+ transform=transform,
102
+ snapshot=None,
103
+ )
@@ -0,0 +1,69 @@
1
+ """plasmidfinder database provider — CGE PlasmidFinder replicons (transform only).
2
+
3
+ Upstream ``get_plasmidfinder`` (abricate-get_db 1.4.0) downloads the bitbucket
4
+ HEAD.zip — an archive with an arbitrary top-level directory — unzips every
5
+ ``*.fsa`` and, per record, splits the accession suffix off the id::
6
+
7
+ >IncFII_1_NC_004631.1 -> gene IncFII_1, accession NC_004631.1
8
+
9
+ Perl semantics preserved exactly (verified against the parity env): the
10
+ product is the ORIGINAL full id (DESC is assigned before the regex munging),
11
+ trailing underscore runs are stripped from the captured id, and a non-matching
12
+ id — or one that strips to empty — keeps the ORIGINAL id as gene with an empty
13
+ accession. Upstream additionally warns on the empty-id case; the transform has
14
+ no diagnostic channel, so the record is kept silently (``fetch_provider`` owns
15
+ stderr). The ``_1`` copy number STAYS on the gene: only the accession suffix
16
+ is removed.
17
+ """
18
+
19
+ import re
20
+ import zipfile
21
+ from collections.abc import Iterator
22
+ from pathlib import Path
23
+
24
+ from gapit.fasta import iter_fasta
25
+ from gapit.providers.common import Provider
26
+ from gapit.records import Record
27
+
28
+ NAME = "plasmidfinder"
29
+ ARCHIVE = "HEAD.zip"
30
+
31
+ # Upstream regex verbatim: group 2 is the accession ([A-Z]+ prefix or the
32
+ # literal NC_, digits, optional .version); group 1 is everything before the
33
+ # last viable underscore — greedy, so the copy number stays in group 1.
34
+ _ID = re.compile(r"^(.*)_(([A-Z]+|NC_)\d+(\.\d+)?)$")
35
+ _FUNCTION = ("replicon",) # locked gapit/v1 func vocabulary (Wave F2c)
36
+
37
+
38
+ def transform(workdir: Path) -> Iterator[Record]:
39
+ """Yield Records from every ``*.fsa`` inside ``workdir/HEAD.zip``.
40
+
41
+ Members are extracted into the workdir (bitbucket archives nest under an
42
+ arbitrary top-level directory) and matched with a recursive glob, in
43
+ sorted order for determinism.
44
+ """
45
+ with zipfile.ZipFile(workdir / ARCHIVE) as archive:
46
+ archive.extractall(workdir)
47
+ for fsa in sorted(workdir.glob("**/*.fsa")):
48
+ for fasta in iter_fasta(fsa):
49
+ id_match = _ID.match(fasta.id)
50
+ captured = id_match.group(1).rstrip("_") if id_match is not None else ""
51
+ yield Record(
52
+ db=NAME,
53
+ gene=captured or fasta.id,
54
+ accession=id_match.group(2) if id_match is not None else "",
55
+ function=_FUNCTION,
56
+ product=fasta.id,
57
+ sequence=fasta.sequence,
58
+ source_id=fasta.id,
59
+ )
60
+
61
+
62
+ PROVIDER = Provider(
63
+ name=NAME,
64
+ description="CGE PlasmidFinder replicons",
65
+ source_urls=("https://bitbucket.org/genomicepidemiology/plasmidfinder_db/get/HEAD.zip",),
66
+ dbtype="nucl",
67
+ transform=transform,
68
+ snapshot=None,
69
+ )