gapit 0.2.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. gapit/__init__.py +3 -0
  2. gapit/blast.py +239 -0
  3. gapit/cli.py +128 -0
  4. gapit/cmd_db.py +113 -0
  5. gapit/cmd_db_build.py +72 -0
  6. gapit/cmd_db_install.py +126 -0
  7. gapit/cmd_db_outdated.py +51 -0
  8. gapit/cmd_db_search.py +66 -0
  9. gapit/cmd_screen.py +214 -0
  10. gapit/cmd_summary.py +65 -0
  11. gapit/config.py +51 -0
  12. gapit/data/snapshots/card.tar.gz +0 -0
  13. gapit/data/snapshots/vfdb.tar.gz +0 -0
  14. gapit/db.py +226 -0
  15. gapit/db_build_ops.py +210 -0
  16. gapit/db_ops.py +128 -0
  17. gapit/db_query_ops.py +252 -0
  18. gapit/dbbuild.py +207 -0
  19. gapit/dbcodec.py +117 -0
  20. gapit/dispatch.py +31 -0
  21. gapit/errors.py +87 -0
  22. gapit/fasta.py +123 -0
  23. gapit/formats/__init__.py +1 -0
  24. gapit/formats/json.py +309 -0
  25. gapit/formats/md.py +190 -0
  26. gapit/formats/schemas.py +30 -0
  27. gapit/formats/summary.py +103 -0
  28. gapit/formats/tsv.py +45 -0
  29. gapit/hits.py +107 -0
  30. gapit/mcp.py +158 -0
  31. gapit/mcp_schemas.py +123 -0
  32. gapit/mcp_tools.py +289 -0
  33. gapit/minimap.py +20 -0
  34. gapit/minimap2_run.py +114 -0
  35. gapit/paf.py +115 -0
  36. gapit/proctools.py +24 -0
  37. gapit/providers/__init__.py +39 -0
  38. gapit/providers/argannot.py +94 -0
  39. gapit/providers/bacmet2.py +59 -0
  40. gapit/providers/card.py +150 -0
  41. gapit/providers/common.py +245 -0
  42. gapit/providers/ecoh.py +63 -0
  43. gapit/providers/ecoli_vf.py +74 -0
  44. gapit/providers/megares.py +71 -0
  45. gapit/providers/ncbi.py +103 -0
  46. gapit/providers/plasmidfinder.py +69 -0
  47. gapit/providers/resfinder.py +123 -0
  48. gapit/providers/snapshots.py +119 -0
  49. gapit/providers/upec_expec_vf.py +85 -0
  50. gapit/providers/vfdb.py +92 -0
  51. gapit/providers/victors.py +109 -0
  52. gapit/py.typed +0 -0
  53. gapit/reads.py +221 -0
  54. gapit/records.py +152 -0
  55. gapit/report.py +25 -0
  56. gapit/screening.py +145 -0
  57. gapit/screening_reads.py +255 -0
  58. gapit/seqconvert.py +203 -0
  59. gapit/summary.py +151 -0
  60. gapit-0.2.2.dist-info/METADATA +183 -0
  61. gapit-0.2.2.dist-info/RECORD +64 -0
  62. gapit-0.2.2.dist-info/WHEEL +4 -0
  63. gapit-0.2.2.dist-info/entry_points.txt +3 -0
  64. gapit-0.2.2.dist-info/licenses/LICENSE +21 -0
gapit/minimap.py ADDED
@@ -0,0 +1,20 @@
1
+ """COVERAGE_MAP minimap — exact abricate arithmetic (SPEC.md §4)."""
2
+
3
+
4
+ def minimap(x: int, y: int, length: int, broken: int) -> str:
5
+ """15-char coverage bar: '=' over [x, y] scaled onto the map width.
6
+
7
+ Replicates the Perl int() truncation quirks verbatim — do not "fix" the
8
+ 1-based coordinate artifacts (yi may exceed width-1). ``broken`` is the
9
+ gap-open count; a broken map is 14 boxes with '/' after box width//2.
10
+ """
11
+ width = 15 - (1 if broken > 0 else 0)
12
+ scale = length / width
13
+ xi = int(x / scale)
14
+ yi = int(y / scale)
15
+ chars: list[str] = []
16
+ for i in range(width):
17
+ chars.append("=" if xi <= i <= yi else ".")
18
+ if broken > 0 and i == width // 2:
19
+ chars.append("/")
20
+ return "".join(chars)
gapit/minimap2_run.py ADDED
@@ -0,0 +1,114 @@
1
+ """The minimap2 invocation layer for reads mode (SPEC.md §10).
2
+
3
+ Split from reads.py (which was over the 250 pure-LOC ceiling) so this module
4
+ owns exactly one thing: running minimap2 and streaming its PAF rows. The
5
+ minimap2 invocation's own preset vocabulary (``ReadType``) lives in paf.py,
6
+ the lowest layer of the reads stack. Not to be confused with minimap.py,
7
+ which owns the blastn-side COVERAGE_MAP arithmetic.
8
+ """
9
+
10
+ import shlex
11
+ import subprocess
12
+ import sys
13
+ import threading
14
+ from pathlib import Path
15
+
16
+ from gapit.db import Database
17
+ from gapit.errors import DependencyError, GapitError
18
+ from gapit.paf import PafRecord, ReadType, parse_paf_row
19
+
20
+
21
+ def _stream_minimap2(argv: list[str], r1: Path) -> list[PafRecord]:
22
+ """Run one minimap2 invocation, streaming PAF rows off stdout as lines
23
+ arrive (no full-PAF materialization). stderr is drained by a background
24
+ thread: an undrained stderr pipe fills (~64 KB) and blocks the child
25
+ mid-run. Returncode != 0 raises MINIMAP2_FAILED with the captured
26
+ stderr text."""
27
+ try:
28
+ process = subprocess.Popen(
29
+ argv,
30
+ stdin=subprocess.DEVNULL,
31
+ stdout=subprocess.PIPE,
32
+ stderr=subprocess.PIPE,
33
+ text=True,
34
+ )
35
+ except FileNotFoundError as exc:
36
+ raise DependencyError(
37
+ "required binary not found on PATH: minimap2",
38
+ code="MISSING_DEPENDENCY",
39
+ context={"binary": "minimap2"},
40
+ ) from exc
41
+ stderr_chunks: list[str] = []
42
+
43
+ def drain() -> None:
44
+ if process.stderr is not None:
45
+ stderr_chunks.append(process.stderr.read())
46
+
47
+ thread = threading.Thread(target=drain)
48
+ thread.start()
49
+ rows: list[PafRecord] = []
50
+ try:
51
+ if process.stdout is not None:
52
+ for line in process.stdout:
53
+ stripped = line.rstrip("\n")
54
+ if stripped.strip():
55
+ rows.append(parse_paf_row(stripped))
56
+ finally:
57
+ if process.stdout is not None:
58
+ process.stdout.close() # on a parse error, SIGPIPE stops the child
59
+ process.wait()
60
+ thread.join()
61
+ if process.returncode != 0:
62
+ raise GapitError(
63
+ f"minimap2 failed: {''.join(stderr_chunks).strip()}",
64
+ code="MINIMAP2_FAILED",
65
+ context={"binary": "minimap2", "file": str(r1)},
66
+ )
67
+ return rows
68
+
69
+
70
+ def run_minimap2(
71
+ lanes: list[tuple[Path, Path | None]],
72
+ database: Database,
73
+ *,
74
+ read_type: ReadType,
75
+ threads: int,
76
+ debug: bool = False,
77
+ nm_tags: bool = False,
78
+ ) -> list[PafRecord]:
79
+ """Run one minimap2 invocation per lane (PAF on stdout) and concatenate
80
+ the rows, streaming each lane's output row by row. The index argument is
81
+ always the ``sequences`` FASTA — minimap2 indexes it in memory with the
82
+ invocation preset's own parameters. A persisted ``.mmi`` is deliberately
83
+ rejected even when one sits beside the FASTA: a default-built index
84
+ overrides the ``-x`` preset's indexing parameters (``-k, -w or -H
85
+ overridden by prebuilt index``), which misassigns close homologs and
86
+ benchmarks slower than in-memory indexing (2026-09-19: blaCTX-M/blaSHV
87
+ allele divergence, +1.2 s on the ncbi db). minimap2's pairing semantics
88
+ for >2 input files are undocumented; per-lane runs (r1[i] alone or with
89
+ its mate r2[i]) are deterministic. With ``nm_tags``, ``--cs`` is added so
90
+ rows carry ``NM:i:`` (minimap2 omits NM from PAF output without it; the
91
+ alignments themselves are unchanged) — used by gapit.reads/2 identity
92
+ filtering. With ``debug``, echo each argv to stderr (abricate --debug
93
+ parity)."""
94
+ rows: list[PafRecord] = []
95
+ for r1, r2 in lanes:
96
+ argv = [
97
+ "minimap2",
98
+ "-x",
99
+ read_type,
100
+ "-t",
101
+ str(threads),
102
+ ]
103
+ if nm_tags:
104
+ argv.append("--cs")
105
+ # Input paths go to argv absolutized: an absolute path starts with
106
+ # "/" and can never parse as a minimap2 option, so a query file
107
+ # named "-d" cannot make minimap2 dump an index over its mate path.
108
+ argv += [str(database.sequences_path.absolute()), str(r1.absolute())]
109
+ if r2 is not None:
110
+ argv.append(str(r2.absolute()))
111
+ if debug:
112
+ print(f"gapit: run: {shlex.join(argv)}", file=sys.stderr)
113
+ rows.extend(_stream_minimap2(argv, r1))
114
+ return rows
gapit/paf.py ADDED
@@ -0,0 +1,115 @@
1
+ """PAF row parsing and interval arithmetic (minimap2 output boundary).
2
+
3
+ Typed parsing of minimap2's PAF rows (12 required fields + tags), the
4
+ half-open interval math used to aggregate alignment spans into coverage, and
5
+ the per-alignment identity rule behind gapit.reads/2 filtering.
6
+ """
7
+
8
+ from collections.abc import Iterable
9
+ from typing import Literal
10
+
11
+ from pydantic import BaseModel
12
+
13
+ from gapit.errors import GapitError
14
+
15
+ ReadType = Literal["sr", "map-ont", "map-hifi"]
16
+
17
+
18
+ class PafRecord(BaseModel, frozen=True):
19
+ """One PAF alignment row: 12 required fields + primary flag (tp:A:P) and
20
+ mismatch count (NM:i:, None when minimap2 emitted no NM tag — it only
21
+ does with ``--cs``)."""
22
+
23
+ qname: str
24
+ qlen: int
25
+ qstart: int
26
+ qend: int
27
+ strand: Literal["+", "-"]
28
+ tname: str
29
+ tlen: int
30
+ tstart: int
31
+ tend: int
32
+ nmatch: int
33
+ alen: int
34
+ mapq: int
35
+ is_primary: bool = True
36
+ nm: int | None = None
37
+
38
+
39
+ def parse_paf_row(line: str) -> PafRecord:
40
+ """Parse one tab-delimited PAF line; <12 fields is a hard error. Records
41
+ without a tp tag count as primary; NM:i: is extracted when present."""
42
+ fields = line.split("\t")
43
+ if len(fields) < 12:
44
+ raise GapitError(
45
+ f"can not parse PAF row (expected >= 12 fields): {line!r}",
46
+ code="PAF_PARSE_FAILED",
47
+ )
48
+ is_primary = True
49
+ nm: int | None = None
50
+ for tag in fields[12:]:
51
+ if tag.startswith("tp:A:"):
52
+ is_primary = tag == "tp:A:P"
53
+ elif tag.startswith("NM:i:"):
54
+ nm = int(tag[len("NM:i:") :])
55
+ strand = fields[4]
56
+ if strand not in ("+", "-"):
57
+ raise GapitError(
58
+ f"can not parse PAF strand (expected + or -): {strand!r}",
59
+ code="PAF_PARSE_FAILED",
60
+ )
61
+ # model_construct: the manual int()/strand checks above already guarantee
62
+ # the field types; re-validating per PAF row would be redundant.
63
+ return PafRecord.model_construct(
64
+ qname=fields[0],
65
+ qlen=int(fields[1]),
66
+ qstart=int(fields[2]),
67
+ qend=int(fields[3]),
68
+ strand=strand,
69
+ tname=fields[5],
70
+ tlen=int(fields[6]),
71
+ tstart=int(fields[7]),
72
+ tend=int(fields[8]),
73
+ nmatch=int(fields[9]),
74
+ alen=int(fields[10]),
75
+ mapq=int(fields[11]),
76
+ is_primary=is_primary,
77
+ nm=nm,
78
+ )
79
+
80
+
81
+ def alignment_identity(record: PafRecord) -> float:
82
+ """Per-alignment identity for gapit.reads/2: ``100 * (alen - nm) / alen``
83
+ over the block length (column 10) and the NM tag (mismatches + gaps per
84
+ the PAF spec). A row without NM cannot be assessed and counts as 100.0;
85
+ a degenerate zero-length block also counts as 100.0 (no division)."""
86
+ if record.nm is None or record.alen <= 0:
87
+ return 100.0
88
+ return 100.0 * (record.alen - record.nm) / record.alen
89
+
90
+
91
+ def filter_alignments(
92
+ rows: Iterable[PafRecord], *, min_identity: float = 0.0, min_mapq: int = 0
93
+ ) -> list[PafRecord]:
94
+ """gapit.reads/2 row filter: keep alignments with identity >= min_identity
95
+ and mapq >= min_mapq (applied after the primary-only rule upstream; both
96
+ thresholds default off, in which case every row is kept verbatim — the
97
+ gapit.reads/1 path must stay byte-identical, even for pathological rows
98
+ whose arithmetic identity is negative)."""
99
+ if min_identity <= 0.0 and min_mapq <= 0:
100
+ return list(rows)
101
+ return [row for row in rows if row.mapq >= min_mapq and alignment_identity(row) >= min_identity]
102
+
103
+
104
+ def union_length(intervals: Iterable[tuple[int, int]]) -> int:
105
+ """Total length of the union of half-open [start, end) intervals:
106
+ sort + linear sweep. Identical arithmetic to counting positions whose
107
+ per-base depth is > 0, without materializing any per-base array."""
108
+ covered = 0
109
+ reach = -1
110
+ for start, end in sorted(intervals):
111
+ if end <= reach:
112
+ continue
113
+ covered += end - max(start, reach)
114
+ reach = end
115
+ return covered
gapit/proctools.py ADDED
@@ -0,0 +1,24 @@
1
+ """Shared external-tool plumbing: the argv subprocess runner and stderr notes."""
2
+
3
+ import subprocess
4
+ import sys
5
+
6
+ from gapit.errors import DependencyError
7
+
8
+
9
+ def run_tool(argv: list[str]) -> subprocess.CompletedProcess[str]:
10
+ """Run an external tool with an argv list (never a shell)."""
11
+ try:
12
+ return subprocess.run(argv, check=False, capture_output=True, text=True)
13
+ except FileNotFoundError as exc:
14
+ raise DependencyError(
15
+ f"required binary not found on PATH: {argv[0]}",
16
+ code="MISSING_DEPENDENCY",
17
+ context={"binary": argv[0]},
18
+ ) from exc
19
+
20
+
21
+ def note(quiet: bool, message: str) -> None:
22
+ """Per-step progress on stderr when quiet is disabled (stdout stays pure)."""
23
+ if not quiet:
24
+ print(f"gapit: {message}", file=sys.stderr)
@@ -0,0 +1,39 @@
1
+ """Database providers: the Provider contract and per-provider modules (Wave B).
2
+
3
+ Each concrete provider (Wave B1..B12) is a module exporting a ``Provider``
4
+ value consumed by :func:`gapit.providers.common.fetch_provider`.
5
+ ``REGISTRY`` maps every provider name to its ``Provider`` — the CLI's
6
+ ``gapit db fetch`` / ``gapit db list`` lookup table (Wave C-a). Imports are
7
+ static and alphabetized; no dynamic import magic.
8
+ """
9
+
10
+ from gapit.providers import (
11
+ argannot,
12
+ bacmet2,
13
+ card,
14
+ ecoh,
15
+ ecoli_vf,
16
+ megares,
17
+ ncbi,
18
+ plasmidfinder,
19
+ resfinder,
20
+ upec_expec_vf,
21
+ vfdb,
22
+ victors,
23
+ )
24
+ from gapit.providers.common import Provider
25
+
26
+ REGISTRY: dict[str, Provider] = {
27
+ argannot.PROVIDER.name: argannot.PROVIDER,
28
+ bacmet2.PROVIDER.name: bacmet2.PROVIDER,
29
+ card.PROVIDER.name: card.PROVIDER,
30
+ ecoh.PROVIDER.name: ecoh.PROVIDER,
31
+ ecoli_vf.PROVIDER.name: ecoli_vf.PROVIDER,
32
+ megares.PROVIDER.name: megares.PROVIDER,
33
+ ncbi.PROVIDER.name: ncbi.PROVIDER,
34
+ plasmidfinder.PROVIDER.name: plasmidfinder.PROVIDER,
35
+ resfinder.PROVIDER.name: resfinder.PROVIDER,
36
+ upec_expec_vf.PROVIDER.name: upec_expec_vf.PROVIDER,
37
+ vfdb.PROVIDER.name: vfdb.PROVIDER,
38
+ victors.PROVIDER.name: victors.PROVIDER,
39
+ }
@@ -0,0 +1,94 @@
1
+ """ARG-ANNOT provider (acquired resistance genes) — transform only.
2
+
3
+ Upstream ``get_argannot`` (abricate-get_db 1.4.0) downloads
4
+ ``ARG-ANNOT_NT_V6_July2019.txt`` — a nucleotide FASTA whose raw text is
5
+ littered with stray backslashes (upstream calls this "fix syntax errors in
6
+ the FASTA file"), so every backslash is stripped BEFORE parsing. Headers
7
+ look like::
8
+
9
+ >(AGly)aac2-Ie:NC_011896:3039059-3039607:549
10
+
11
+ The id token splits on ``:``: gene is ``x[0]``, accession is ``x[1]:x[2]``
12
+ (later fields, e.g. the trailing length, are ignored — upstream never reads
13
+ ``x[3]``). Product falls back to the gene when the header carries no
14
+ description (save_fasta's ``DESC || ID``); real ARG-ANNOT headers carry
15
+ none. Records with fewer than 3 colon-separated fields are SKIPPED
16
+ (documented deviation): upstream perl would splice an undefined ``x[2]``
17
+ into the accession; a typed Record refuses to carry that.
18
+ """
19
+
20
+ from collections.abc import Iterator
21
+ from pathlib import Path
22
+ from tempfile import NamedTemporaryFile
23
+
24
+ from gapit.fasta import iter_fasta
25
+ from gapit.providers.common import Provider
26
+ from gapit.records import Record
27
+
28
+ NAME = "argannot"
29
+ SOURCE_FILE = "ARG-ANNOT_NT_V6_July2019.txt"
30
+
31
+
32
+ def transform(workdir: Path) -> Iterator[Record]:
33
+ """Yield Records from ``workdir/ARG-ANNOT_NT_V6_July2019.txt``.
34
+
35
+ The raw text is repaired first — every backslash stripped, upstream's
36
+ global ``s/\\\\//g`` — and written to a scratch file inside the workdir
37
+ so :func:`gapit.fasta.iter_fasta` can parse it (it takes a path). The
38
+ scratch file is removed on completion or early generator close.
39
+ Sequence arrives raw — fetch_provider N-normalizes it for the nucl
40
+ dbtype.
41
+ """
42
+ repaired = (workdir / SOURCE_FILE).read_text(encoding="utf-8").replace("\\", "")
43
+ with NamedTemporaryFile(
44
+ mode="w",
45
+ encoding="utf-8",
46
+ newline="\n",
47
+ prefix=".argannot.",
48
+ suffix=".txt",
49
+ dir=workdir,
50
+ delete=False,
51
+ ) as handle:
52
+ handle.write(repaired)
53
+ scratch = Path(handle.name)
54
+ try:
55
+ for fasta in iter_fasta(scratch):
56
+ parts = fasta.id.split(":")
57
+ if len(parts) < 3:
58
+ continue
59
+ gene = parts[0]
60
+ yield Record(
61
+ db=NAME,
62
+ gene=gene,
63
+ sequence=fasta.sequence,
64
+ accession=":".join(parts[1:3]),
65
+ function=(),
66
+ product=fasta.description or gene,
67
+ source_id=fasta.id,
68
+ )
69
+ finally:
70
+ scratch.unlink(missing_ok=True)
71
+
72
+
73
+ PROVIDER = Provider(
74
+ name=NAME,
75
+ description="ARG-ANNOT acquired resistance genes",
76
+ # Upstream 301-migrated; the deep link 404s on both domains (verified 2026-09-17).
77
+ # Wayback CDX: 2020-06-26 + 2026-01-14 captures share digest 5ZDVKAPZ4ZGIUDSYFVTFUYYY4CS5MUXZ
78
+ # (2024 differs: suspect partial). The 2026-01-14 memento intermittently serves a 9KB
79
+ # Wayback outage interstitial (LoadShardBlock datanode, observed 2026-09-18); the
80
+ # 2020-06-26 memento probed good (text/plain, 2,139,519 bytes, same stable digest
81
+ # family). Snapshot pinned, not latest; veto to upstream if it returns.
82
+ # The id_ suffix requests the original WARC bytes with no Wayback rewrite/redirect
83
+ # layer — the right form for machine fetching: verified 2026-09-18 (200, text/plain,
84
+ # 2139519B, FASTA head ">(AGly)aac:AJ628983:1985-2539:555") while the plain
85
+ # memento form flapped (interstitial twice, 45s apart).
86
+ source_urls=(
87
+ "https://web.archive.org/web/20200626214628id_/"
88
+ "https://www.mediterranee-infection.com/wp-content/uploads/2019/09/"
89
+ "ARG-ANNOT_NT_V6_July2019.txt",
90
+ ),
91
+ dbtype="nucl",
92
+ transform=transform,
93
+ snapshot=None,
94
+ )
@@ -0,0 +1,59 @@
1
+ """BacMet2 provider (experimentally confirmed, PROTEIN) — transform only.
2
+
3
+ Upstream ``get_bacmet2`` (abricate-get_db 1.4.0) downloads
4
+ ``BacMet2_EXP_database.fasta`` — a protein file, hence the set's one
5
+ ``dbtype="prot"`` provider (screening uses blastx). Headers look like::
6
+
7
+ >BAC0098|ctpC|sp|P0A502|CTPC_MYCTU Probable manganese/zinc-exporting
8
+
9
+ The id token splits on ``|``: ``bac_id|gene|db|uniprot`` plus an optional
10
+ 5th field (the UniProt entry name) that upstream ignores. Upstream
11
+ recombines ``gene-bac_id`` as the gene name and ``db:uniprot`` as the
12
+ accession; product falls back to the gene when the description is empty
13
+ (save_fasta's ``DESC || ID``). Function is the locked ``biocide`` constant:
14
+ BacMet covers biocides + metals but the 4-field id carries no class, so
15
+ ``biocide`` is the user-approved approximation (Wave F2c).
16
+ """
17
+
18
+ from collections.abc import Iterable
19
+ from pathlib import Path
20
+
21
+ from gapit.fasta import iter_fasta
22
+ from gapit.providers.common import Provider
23
+ from gapit.records import Record
24
+
25
+ _NAME = "bacmet2"
26
+ _FUNCTION = ("biocide",) # locked gapit/v1 func vocabulary (Wave F2c)
27
+
28
+
29
+ def transform(workdir: Path) -> Iterable[Record]:
30
+ """Yield Records from ``workdir/BacMet2_EXP_database.fasta``.
31
+
32
+ A record whose id has fewer than 4 pipe-separated fields is skipped
33
+ (upstream perl would emit undef-indexed garbage). Sequence arrives raw
34
+ — fetch_provider X-normalizes it for the prot dbtype.
35
+ """
36
+ for fasta in iter_fasta(workdir / "BacMet2_EXP_database.fasta"):
37
+ parts = fasta.id.split("|")
38
+ if len(parts) < 4:
39
+ continue
40
+ gene = f"{parts[1]}-{parts[0]}"
41
+ yield Record(
42
+ db=_NAME,
43
+ gene=gene,
44
+ sequence=fasta.sequence,
45
+ accession=f"{parts[2]}:{parts[3]}",
46
+ function=_FUNCTION,
47
+ product=fasta.description or gene,
48
+ source_id=fasta.id,
49
+ )
50
+
51
+
52
+ PROVIDER = Provider(
53
+ name=_NAME,
54
+ description="BacMet2 experimentally confirmed biocide/resistance genes (protein)",
55
+ source_urls=("http://bacmet.biomedicine.gu.se/download/BacMet2_EXP_database.fasta",),
56
+ dbtype="prot",
57
+ transform=transform,
58
+ snapshot=None,
59
+ )
@@ -0,0 +1,150 @@
1
+ """CARD provider (Wave B2): transform card.mcmaster.ca data into Records.
2
+
3
+ Upstream quirk: the source URL ``https://card.mcmaster.ca/latest/data`` has no
4
+ extension but serves a tar.bz2, so the download lands at ``<workdir>/data``.
5
+ Only the root-level ``card.json`` is extracted (upstream:
6
+ ``tar xf card.tar.bz2 card.json``), matched by root-normalized member name
7
+ because the re-tarred archive prefixes members with ``./``. Every dict-valued
8
+ top-level entry whose
9
+ ``model_type`` is ``"protein homolog model"`` becomes one Record; a homolog
10
+ model carrying ``model_param.snp`` is a hard error (upstream ``err()``); every
11
+ other model type — and any non-dict entry, like the ``_version`` string — is
12
+ skipped. Sequence/function normalization is NOT done here: that is
13
+ fetch_provider's job (B0 contract).
14
+
15
+ ``Any`` is confined to the parsed-JSON values flowing through these helpers:
16
+ CARD node shapes are heterogeneous, and checking them into pydantic models
17
+ would dwarf the module. Every value crossing out to a Record is a str/int.
18
+ """
19
+
20
+ import json
21
+ import re
22
+ import tarfile
23
+ from collections.abc import Iterator
24
+ from pathlib import Path
25
+ from typing import Any
26
+
27
+ from gapit.errors import DatabaseError
28
+ from gapit.providers.common import Dbtype, Provider
29
+ from gapit.records import Record
30
+
31
+ NAME = "card"
32
+ DESCRIPTION = "CARD protein homolog resistance models"
33
+ SOURCE_URLS = ("https://card.mcmaster.ca/latest/data",)
34
+ DBTYPE: Dbtype = "nucl"
35
+
36
+ _MEMBER = "card.json" # the one member we extract from the tarball root
37
+ _CHAR_SPACE = re.compile(r"\s") # per character, like perl s/\s/_/g (drug classes)
38
+ _RUN_SPACE = re.compile(r"\s+") # per run, like perl s/\s+/_/g (model names)
39
+
40
+
41
+ def _root_name(member_name: str) -> str:
42
+ """Member path with a leading './' and a single leading '/' stripped:
43
+ card.mcmaster.ca re-tarred its archive with './'-prefixed members
44
+ (Wave E), so exact-name lookups miss."""
45
+ return member_name.removeprefix("./").removeprefix("/")
46
+
47
+
48
+ def _models(tarball: Path) -> Iterator[dict[str, Any]]:
49
+ """Extract card.json from the tarball; yield its dict-valued entries.
50
+
51
+ The member is matched by root-normalized name (first hit in archive
52
+ order). Non-dict top-level values (the ``_version`` metadata string) are
53
+ skipped — upstream's ``next unless ref($g) eq 'HASH'``.
54
+ """
55
+ with tarfile.open(tarball, "r:bz2") as tar:
56
+ member = next(
57
+ (m for m in tar.getmembers() if _root_name(m.name) == _MEMBER),
58
+ None,
59
+ )
60
+ if member is None:
61
+ raise DatabaseError(
62
+ f"no {_MEMBER} member in {tarball}",
63
+ code="PROVIDER_INVALID",
64
+ context={"archive": str(tarball), "expected": _MEMBER},
65
+ )
66
+ extracted = tar.extractfile(member)
67
+ if extracted is None:
68
+ raise DatabaseError(
69
+ f"{_MEMBER} in {tarball} is not a regular file",
70
+ code="PROVIDER_INVALID",
71
+ context={"db": NAME},
72
+ )
73
+ card: Any = json.loads(extracted.read())
74
+ for model in card.values():
75
+ if isinstance(model, dict):
76
+ yield model
77
+
78
+
79
+ def _drug_classes(model: dict[str, Any]) -> tuple[str, ...]:
80
+ """Drug Class category names: ' antibiotic' stripped once, then each
81
+ whitespace character -> '_' (upstream's two s/// on $abx), in category
82
+ order — sorting is fetch_provider's job."""
83
+ categories: Any = model.get("ARO_category", {})
84
+ classes: list[str] = []
85
+ for category in categories.values():
86
+ if category.get("category_aro_class_name") == "Drug Class":
87
+ name: str = category.get("category_aro_name", "")
88
+ classes.append(_CHAR_SPACE.sub("_", name.replace(" antibiotic", "", 1)))
89
+ return tuple(classes)
90
+
91
+
92
+ def _first_dna(model: dict[str, Any]) -> Any:
93
+ """The dna_sequence dict of the lexicographically-first key under
94
+ model_sequences.sequence (perl: ``my ($key) = sort keys %$dna``).
95
+
96
+ Its siblings carry accession/strand/fmin/fmax; the sequence itself is the
97
+ ``sequence`` key of that same dict.
98
+ """
99
+ sequences: Any = model.get("model_sequences", {}).get("sequence", {})
100
+ first = sorted(sequences)[0]
101
+ return sequences[first].get("dna_sequence", {})
102
+
103
+
104
+ def _model_record(model: dict[str, Any]) -> Record:
105
+ """One 'protein homolog model' -> one Record (db=card)."""
106
+ model_name: str = model.get("model_name", "")
107
+ model_param: Any = model.get("model_param") or {}
108
+ if "snp" in model_param:
109
+ raise DatabaseError(
110
+ f"{model_name} has model_param.snp",
111
+ code="PROVIDER_INVALID",
112
+ context={"model": model_name},
113
+ )
114
+ dna: Any = _first_dna(model)
115
+ strand: Any = dna.get("strand", "+")
116
+ fmin: Any = dna.get("fmin", 0)
117
+ fmax: Any = dna.get("fmax", 0)
118
+ start, stop = (fmax, fmin) if strand == "-" else (fmin, fmax)
119
+ product: Any = model.get("ARO_description") or model.get("ARO_accession", "")
120
+ return Record(
121
+ db=NAME,
122
+ gene=_RUN_SPACE.sub("_", model_name),
123
+ sequence=dna.get("sequence", ""),
124
+ accession=f"{dna.get('accession', '')}:{start}-{stop}",
125
+ function=_drug_classes(model),
126
+ product=product,
127
+ source_id=model.get("ARO_accession", ""),
128
+ )
129
+
130
+
131
+ def transform(workdir: Path) -> Iterator[Record]:
132
+ """Provider transform: the tarball at ``<workdir>/data`` -> Records.
133
+
134
+ Non-homolog model types are skipped BEFORE the model_param.snp check
135
+ (upstream's statement order), so a variant model carrying an snp is
136
+ skipped, not an error.
137
+ """
138
+ for model in _models(workdir / "data"):
139
+ if model.get("model_type") == "protein homolog model":
140
+ yield _model_record(model)
141
+
142
+
143
+ PROVIDER = Provider(
144
+ name=NAME,
145
+ description=DESCRIPTION,
146
+ source_urls=SOURCE_URLS,
147
+ dbtype=DBTYPE,
148
+ transform=transform,
149
+ snapshot="card.tar.gz",
150
+ )