gapit 0.2.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. gapit/__init__.py +3 -0
  2. gapit/blast.py +239 -0
  3. gapit/cli.py +128 -0
  4. gapit/cmd_db.py +113 -0
  5. gapit/cmd_db_build.py +72 -0
  6. gapit/cmd_db_install.py +126 -0
  7. gapit/cmd_db_outdated.py +51 -0
  8. gapit/cmd_db_search.py +66 -0
  9. gapit/cmd_screen.py +214 -0
  10. gapit/cmd_summary.py +65 -0
  11. gapit/config.py +51 -0
  12. gapit/data/snapshots/card.tar.gz +0 -0
  13. gapit/data/snapshots/vfdb.tar.gz +0 -0
  14. gapit/db.py +226 -0
  15. gapit/db_build_ops.py +210 -0
  16. gapit/db_ops.py +128 -0
  17. gapit/db_query_ops.py +252 -0
  18. gapit/dbbuild.py +207 -0
  19. gapit/dbcodec.py +117 -0
  20. gapit/dispatch.py +31 -0
  21. gapit/errors.py +87 -0
  22. gapit/fasta.py +123 -0
  23. gapit/formats/__init__.py +1 -0
  24. gapit/formats/json.py +309 -0
  25. gapit/formats/md.py +190 -0
  26. gapit/formats/schemas.py +30 -0
  27. gapit/formats/summary.py +103 -0
  28. gapit/formats/tsv.py +45 -0
  29. gapit/hits.py +107 -0
  30. gapit/mcp.py +158 -0
  31. gapit/mcp_schemas.py +123 -0
  32. gapit/mcp_tools.py +289 -0
  33. gapit/minimap.py +20 -0
  34. gapit/minimap2_run.py +114 -0
  35. gapit/paf.py +115 -0
  36. gapit/proctools.py +24 -0
  37. gapit/providers/__init__.py +39 -0
  38. gapit/providers/argannot.py +94 -0
  39. gapit/providers/bacmet2.py +59 -0
  40. gapit/providers/card.py +150 -0
  41. gapit/providers/common.py +245 -0
  42. gapit/providers/ecoh.py +63 -0
  43. gapit/providers/ecoli_vf.py +74 -0
  44. gapit/providers/megares.py +71 -0
  45. gapit/providers/ncbi.py +103 -0
  46. gapit/providers/plasmidfinder.py +69 -0
  47. gapit/providers/resfinder.py +123 -0
  48. gapit/providers/snapshots.py +119 -0
  49. gapit/providers/upec_expec_vf.py +85 -0
  50. gapit/providers/vfdb.py +92 -0
  51. gapit/providers/victors.py +109 -0
  52. gapit/py.typed +0 -0
  53. gapit/reads.py +221 -0
  54. gapit/records.py +152 -0
  55. gapit/report.py +25 -0
  56. gapit/screening.py +145 -0
  57. gapit/screening_reads.py +255 -0
  58. gapit/seqconvert.py +203 -0
  59. gapit/summary.py +151 -0
  60. gapit-0.2.2.dist-info/METADATA +183 -0
  61. gapit-0.2.2.dist-info/RECORD +64 -0
  62. gapit-0.2.2.dist-info/WHEEL +4 -0
  63. gapit-0.2.2.dist-info/entry_points.txt +3 -0
  64. gapit-0.2.2.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,123 @@
1
+ """resfinder database provider — CGE ResFinder acquired resistance genes.
2
+
3
+ Upstream ``get_resfinder`` (abricate-get_db 1.4.0) git-clones the resfinder_db
4
+ repo; gapit downloads the same tree as the bitbucket ``HEAD.zip`` archive (no
5
+ git dependency) whose members sit under an arbitrary top-level directory.
6
+ ``phenotypes.txt`` maps each gene to its function categories; every ``*.fsa``
7
+ carries ids like ``demoA_1_FAKE0001`` — gene prefix, copy number, accession.
8
+
9
+ Issue #62 repair: resfinder .fsa files can glue a record header onto the end
10
+ of the previous sequence line (a letter immediately followed by ``>``); a
11
+ newline is inserted there in the raw text before parsing, exactly like
12
+ upstream's ``sed s/([A-Z])>/\\1\\n>/gi`` (the /i makes it both cases).
13
+
14
+ Perl semantics kept: Class-cell pieces are filtered of unknown/notes/none
15
+ markers BEFORE stripping (upstream greps, then trims), the LAST phenotypes
16
+ row for a gene wins (plain hash assign), and the product is always the gene
17
+ prefix (upstream's ``$anno{$id}{DESC}`` is never assigned). Deviation: a
18
+ record whose id has no ``_<digits>_<accession>`` structure is SKIPPED —
19
+ upstream would emit it with undef fields (same policy as vfdb).
20
+ """
21
+
22
+ import re
23
+ import zipfile
24
+ from collections.abc import Iterator
25
+ from pathlib import Path
26
+ from tempfile import TemporaryDirectory
27
+
28
+ from gapit.errors import DatabaseError
29
+ from gapit.fasta import iter_fasta
30
+ from gapit.providers.common import Provider
31
+ from gapit.records import Record
32
+
33
+ NAME = "resfinder"
34
+ ARCHIVE = "HEAD.zip"
35
+
36
+ _PHENOTYPES = "phenotypes.txt"
37
+ _GENE = re.compile(r"^(.*?)_\w+$") # col0 minus its final _<word> chunk
38
+ _CLASS_SPLIT = re.compile(r",\s*")
39
+ _CLASS_FILTER = re.compile(r"(unknown|notes|^none)", re.IGNORECASE)
40
+ # abricate issue #62: a letter directly followed by '>' is a glued header.
41
+ _INLINE_HEADER = re.compile(r"([A-Za-z])>")
42
+ _ID = re.compile(r"^(.*?)_(\d+)_(\S+)$") # base _ copy _ accession
43
+
44
+
45
+ def _read_classes(root: Path) -> dict[str, tuple[str, ...]]:
46
+ """phenotypes.txt under the extracted archive: {gene prefix: classes}.
47
+
48
+ Comment lines start with '#'; rows split on tabs; the gene prefix is
49
+ col0 minus its final ``_<word>`` chunk; classes are the col2 pieces
50
+ (split on comma + whitespace) that do not match unknown/notes/none —
51
+ filtered BEFORE stripping, like upstream — each stripped. A later row
52
+ for the same gene overwrites the earlier one (plain hash assign), and a
53
+ missing Class cell means no classes (upstream splits undef into an
54
+ empty list).
55
+ """
56
+ found = sorted(root.glob(f"**/{_PHENOTYPES}"))
57
+ if not found:
58
+ raise DatabaseError(
59
+ f"{ARCHIVE} contains no {_PHENOTYPES}",
60
+ code="PROVIDER_INVALID",
61
+ context={"db": NAME},
62
+ )
63
+ classes: dict[str, tuple[str, ...]] = {}
64
+ for line in found[0].read_text(encoding="utf-8").splitlines():
65
+ if line.startswith("#"):
66
+ continue
67
+ row = line.split("\t")
68
+ gene = _GENE.match(row[0])
69
+ if gene is None:
70
+ continue
71
+ cell = row[2] if len(row) > 2 else ""
72
+ classes[gene.group(1)] = tuple(
73
+ piece.strip()
74
+ for piece in (_CLASS_SPLIT.split(cell) if cell else [])
75
+ if not _CLASS_FILTER.search(piece)
76
+ )
77
+ return classes
78
+
79
+
80
+ def transform(workdir: Path) -> Iterator[Record]:
81
+ """Yield Records from ``workdir/HEAD.zip`` (the bitbucket archive).
82
+
83
+ The archive is extracted into a scratch directory (members sit under an
84
+ arbitrary top-level directory); each ``*.fsa`` is repaired for glued
85
+ headers into a scratch copy before FASTA parsing, so the download is
86
+ never modified. Files are visited in sorted order for determinism.
87
+ """
88
+ with (
89
+ TemporaryDirectory(prefix=".resfinder-extract.") as extract_name,
90
+ TemporaryDirectory(prefix=".resfinder-repair.") as repair_name,
91
+ ):
92
+ extract_dir = Path(extract_name)
93
+ with zipfile.ZipFile(workdir / ARCHIVE) as archive:
94
+ archive.extractall(extract_dir)
95
+ classes = _read_classes(extract_dir)
96
+ for index, fsa in enumerate(sorted(extract_dir.glob("**/*.fsa"))):
97
+ repaired = _INLINE_HEADER.sub(r"\1\n>", fsa.read_text(encoding="utf-8"))
98
+ temp = Path(repair_name) / f"{index:06d}.fsa"
99
+ temp.write_text(repaired, encoding="utf-8")
100
+ for fasta in iter_fasta(temp):
101
+ id_match = _ID.match(fasta.id)
102
+ if id_match is None:
103
+ continue
104
+ base = id_match.group(1)
105
+ yield Record(
106
+ db=NAME,
107
+ gene=f"{base}_{id_match.group(2)}",
108
+ sequence=fasta.sequence,
109
+ accession=id_match.group(3),
110
+ function=classes.get(base, ()),
111
+ product=base,
112
+ source_id=fasta.id,
113
+ )
114
+
115
+
116
+ PROVIDER = Provider(
117
+ name=NAME,
118
+ description="CGE ResFinder acquired resistance genes",
119
+ source_urls=("https://bitbucket.org/genomicepidemiology/resfinder_db/get/HEAD.zip",),
120
+ dbtype="nucl",
121
+ transform=transform,
122
+ snapshot=None,
123
+ )
@@ -0,0 +1,119 @@
1
+ """Bundled database snapshots (Wave G): deterministic archives + extraction.
2
+
3
+ A snapshot is ``<name>.tar.gz`` containing EXACTLY ``records.jsonl`` and
4
+ ``gapit-manifest.json`` from an installed gapit-native database directory.
5
+ card and vfdb ship inside the wheel (``src/gapit/data/snapshots/``) so the
6
+ two default databases install with zero network — snapshots eliminate the
7
+ flaky-source failure modes for exactly those two (Wave E/F3: mgc.ac.cn UA
8
+ blocks, card re-tarring, Wayback interstitials). Every other provider keeps
9
+ the upstream fetch; a future DB is bundled by dropping a ``<name>.tar.gz``
10
+ into the snapshots dir and setting ``snapshot=`` on its Provider (one line).
11
+
12
+ ``records.jsonl`` (post-normalize records) is what gets snapshotted — NOT
13
+ the BLAST index: index bytes are BLAST-version-sensitive while a
14
+ local rebuild from records is deterministic and fast. The archived manifest
15
+ contributes only ``upstream_version``; fetched_at, sha256 and tool versions
16
+ are rebuilt locally by :func:`gapit.dbbuild.build_database`.
17
+
18
+ Determinism (frozen layout contract): entries added in sorted name order,
19
+ PAX format, uid/gid 0, mtime 0, and a gzip stream with mtime 0 and no
20
+ embedded filename — two builds from the same inputs are byte-identical.
21
+ """
22
+
23
+ import gzip
24
+ import io
25
+ import os
26
+ import tarfile
27
+ from pathlib import Path
28
+ from tempfile import NamedTemporaryFile
29
+
30
+ from gapit.errors import DatabaseError
31
+ from gapit.records import Manifest
32
+
33
+ _MEMBERS = ("gapit-manifest.json", "records.jsonl") # sorted: manifest < records
34
+
35
+
36
+ def _snapshot_error(db_dir: Path, missing: str) -> DatabaseError:
37
+ """The uniform archive-content failure: located, machine-stable."""
38
+ return DatabaseError(
39
+ f"cannot snapshot {db_dir}: missing {missing}",
40
+ code="SNAPSHOT_INVALID",
41
+ context={"db_dir": str(db_dir), "missing": missing},
42
+ )
43
+
44
+
45
+ def make_snapshot(db_dir: Path, dest: Path) -> None:
46
+ """Write the deterministic snapshot archive of ``db_dir`` to ``dest``.
47
+
48
+ Both members are required (a snapshot of a half-built db dir refuses
49
+ loudly instead of shipping an uninstallable archive).
50
+ """
51
+ with (
52
+ dest.open("wb") as raw,
53
+ gzip.GzipFile(
54
+ # filename="" keeps the dest path OUT of the gzip header (the
55
+ # fileobj's .name would otherwise be embedded — path-dependent bytes).
56
+ filename="",
57
+ mode="wb",
58
+ fileobj=raw,
59
+ mtime=0,
60
+ ) as gz,
61
+ tarfile.open(fileobj=gz, mode="w", format=tarfile.PAX_FORMAT) as tar,
62
+ ):
63
+ for name in _MEMBERS:
64
+ source = db_dir / name
65
+ if not source.is_file():
66
+ raise _snapshot_error(db_dir, name)
67
+ data = source.read_bytes()
68
+ info = tarfile.TarInfo(name)
69
+ info.size = len(data)
70
+ info.mtime = 0
71
+ tar.addfile(info, io.BytesIO(data))
72
+
73
+
74
+ def _member(tar: tarfile.TarFile, name: str, archive: Path) -> bytes:
75
+ """Read one regular-file member; typed SNAPSHOT_INVALID when absent
76
+ (no untyped KeyError escape — the Wave E card.py lesson)."""
77
+ member = next((m for m in tar.getmembers() if m.name == name and m.isfile()), None)
78
+ if member is None:
79
+ raise DatabaseError(
80
+ f"snapshot {archive} has no {name} member",
81
+ code="SNAPSHOT_INVALID",
82
+ context={"archive": str(archive), "missing": name},
83
+ )
84
+ extracted = tar.extractfile(member)
85
+ if extracted is None:
86
+ raise DatabaseError(
87
+ f"{name} in {archive} is not a regular file",
88
+ code="SNAPSHOT_INVALID",
89
+ context={"archive": str(archive), "missing": name},
90
+ )
91
+ return extracted.read()
92
+
93
+
94
+ def read_snapshot_manifest(archive: Path) -> Manifest:
95
+ """The archived gapit-manifest.json parsed in-memory (no extraction,
96
+ nothing written) — read-only queries such as `db outdated`."""
97
+ with tarfile.open(archive, "r:gz") as tar:
98
+ return Manifest.model_validate_json(_member(tar, "gapit-manifest.json", archive))
99
+
100
+
101
+ def extract_snapshot(archive: Path, db_dir: Path) -> Manifest:
102
+ """Install ``records.jsonl`` from ``archive`` into ``db_dir`` atomically;
103
+ return the ARCHIVED manifest (its upstream_version seeds the fresh build)."""
104
+ with tarfile.open(archive, "r:gz") as tar:
105
+ records = _member(tar, "records.jsonl", archive)
106
+ archived = Manifest.model_validate_json(_member(tar, "gapit-manifest.json", archive))
107
+ temp_path: Path | None = None
108
+ try:
109
+ with NamedTemporaryFile(dir=db_dir, prefix=".records.", suffix=".jsonl", delete=False) as (
110
+ temp
111
+ ):
112
+ temp_path = Path(temp.name)
113
+ temp.write(records)
114
+ assert temp_path is not None
115
+ os.replace(temp_path, db_dir / "records.jsonl")
116
+ finally:
117
+ if temp_path is not None:
118
+ temp_path.unlink(missing_ok=True)
119
+ return archived
@@ -0,0 +1,85 @@
1
+ """upec_expec_vf provider — UPEC/ExPEC virulence genes (FordeGenomics).
2
+
3
+ Transform-only module (Wave B12): turns the downloaded
4
+ ``UPEC_ExPEC_VF.tsv`` into typed ``Record``s exactly like upstream
5
+ ``abricate-get_db get_upec_expec_vf`` — ``load_tabular`` keys every row by
6
+ its ``Gene name`` column with first occurrence winning (perl ``||=``), then
7
+ ID=Gene name, ACC=``Accession/Source:Begin-End``, DESC=Description,
8
+ SEQ=Sequence. Function is the locked ``virulence`` constant: the TSV Class
9
+ column is source metadata, NOT the function slot (Wave F2c). Sequences
10
+ stay RAW here: normalization, sequence dedupe, and sorting belong to the
11
+ generic :func:`gapit.providers.common.fetch_provider` pipeline.
12
+ """
13
+
14
+ import csv
15
+ from collections.abc import Iterator
16
+ from pathlib import Path
17
+
18
+ from gapit.errors import DatabaseError
19
+ from gapit.providers.common import Provider
20
+ from gapit.records import Record
21
+
22
+ NAME = "upec_expec_vf"
23
+
24
+ _FUNCTION = ("virulence",) # locked gapit/v1 func vocabulary (Wave F2c)
25
+
26
+ _FILENAME = "UPEC_ExPEC_VF.tsv"
27
+ _REQUIRED_COLUMNS = (
28
+ "Gene name",
29
+ "Accession/Source",
30
+ "Begin",
31
+ "End",
32
+ "Description",
33
+ "Class",
34
+ "Sequence",
35
+ )
36
+
37
+
38
+ def transform(workdir: Path) -> Iterator[Record]:
39
+ """Yield one Record per well-formed row of ``workdir/UPEC_ExPEC_VF.tsv``.
40
+
41
+ First row is the header; every column is looked up by NAME. Rows whose
42
+ Gene name or Sequence is empty are skipped (upstream would splice undef
43
+ pieces into the record — deviation noted in the Phase 7 notepad). A
44
+ header missing required columns raises DatabaseError PROVIDER_MALFORMED.
45
+ """
46
+ path = workdir / _FILENAME
47
+ with path.open(encoding="utf-8", newline="") as handle:
48
+ rows = csv.DictReader(handle, delimiter="\t", restval="")
49
+ missing = [
50
+ column for column in _REQUIRED_COLUMNS if column not in set(rows.fieldnames or ())
51
+ ]
52
+ if missing:
53
+ raise DatabaseError(
54
+ f"upec_expec_vf source is missing column(s): {', '.join(missing)}",
55
+ code="PROVIDER_MALFORMED",
56
+ context={"file": str(path), "missing_columns": ",".join(missing)},
57
+ )
58
+ seen: set[str] = set()
59
+ for row in rows:
60
+ gene = row["Gene name"]
61
+ sequence = row["Sequence"]
62
+ if not gene or not sequence or gene in seen:
63
+ continue
64
+ seen.add(gene)
65
+ yield Record(
66
+ db=NAME,
67
+ gene=gene,
68
+ accession=f"{row['Accession/Source']}:{row['Begin']}-{row['End']}",
69
+ function=_FUNCTION,
70
+ product=row["Description"],
71
+ sequence=sequence,
72
+ source_id=gene,
73
+ )
74
+
75
+
76
+ PROVIDER = Provider(
77
+ name=NAME,
78
+ description="UPEC/ExPEC virulence genes (FordeGenomics)",
79
+ source_urls=(
80
+ "https://raw.githubusercontent.com/FordeGenomics/ST167_Code/refs/heads/main/UPEC-ExPEC_VF/UPEC_ExPEC_VF.tsv",
81
+ ),
82
+ dbtype="nucl",
83
+ transform=transform,
84
+ snapshot=None,
85
+ )
@@ -0,0 +1,92 @@
1
+ """VFDB provider (set A, nucleotide) — transform only.
2
+
3
+ Upstream ``get_vfdb`` (abricate-get_db 1.4.0) decompresses
4
+ ``VFDB_setA_nt.fas.gz`` and, per record, pulls the accession from the
5
+ ``<gene>(<db>|<acc>.<version>)`` id suffix and renames the gene to the
6
+ leading paren group of the description::
7
+
8
+ >VFG000676(gb|AAD32411) (lef) anthrax toxin lethal factor precursor ...
9
+
10
+ Perl quirk: a record failing either regex keeps the PREVIOUS record's $1/$2
11
+ (stale fields). We skip such records instead — see the phase 7 notepad.
12
+
13
+ Encoding quirk (Wave E): the real ``VFDB_setA_nt.fas.gz`` is not UTF-8-clean
14
+ (a latin-1 0xA0 nbsp crashes ``gapit.fasta``'s strict decoder), so the
15
+ decompressed bytes are decoded as latin-1 before parsing — see the comment
16
+ in :func:`transform`.
17
+ """
18
+
19
+ import gzip
20
+ import re
21
+ from collections.abc import Iterable
22
+ from pathlib import Path
23
+ from tempfile import NamedTemporaryFile
24
+
25
+ from gapit.fasta import iter_fasta
26
+ from gapit.providers.common import Provider
27
+ from gapit.records import Record
28
+
29
+ _NAME = "vfdb"
30
+
31
+ # Upstream regexes verbatim (abricate issue #64 comment by @VGalata):
32
+ # accession = group 2; group 3 (".2" version suffix) is consumed, not used.
33
+ _ID = re.compile(r"^(\w+)\(\w+\|(\w+)(\.\d+)?\)$")
34
+ _DESC = re.compile(r"^\((.*?)\)")
35
+
36
+
37
+ def transform(workdir: Path) -> Iterable[Record]:
38
+ """Yield Records from ``workdir/VFDB_setA_nt.fas.gz``.
39
+
40
+ The gunzipped text is decoded as latin-1 into a scratch UTF-8 file for
41
+ :func:`gapit.fasta.iter_fasta` (it takes a path); the scratch file is
42
+ removed on completion or early generator close (argannot precedent).
43
+
44
+ A record failing EITHER regex is skipped; product keeps the full
45
+ description (upstream leaves DESC untouched) and source_id keeps the
46
+ original id.
47
+ """
48
+ # Byte-preserving decode: upstream perl is byte-oriented here, and
49
+ # latin-1 maps every byte 1:1 — descriptions keep the character and
50
+ # sequence junk flows on to fetch_provider's N-normalization. Our
51
+ # JSONL/FASTA outputs are UTF-8 by contract, so writing the scratch
52
+ # file as UTF-8 re-encodes the character deterministically. Never
53
+ # errors="replace": that would silently rewrite the data.
54
+ text = gzip.decompress((workdir / "VFDB_setA_nt.fas.gz").read_bytes()).decode("latin-1")
55
+ with NamedTemporaryFile(
56
+ mode="w",
57
+ encoding="utf-8",
58
+ newline="\n",
59
+ prefix=".vfdb.",
60
+ suffix=".fas",
61
+ dir=workdir,
62
+ delete=False,
63
+ ) as handle:
64
+ handle.write(text)
65
+ scratch = Path(handle.name)
66
+ try:
67
+ for fasta in iter_fasta(scratch):
68
+ id_match = _ID.match(fasta.id)
69
+ desc_match = _DESC.match(fasta.description)
70
+ if id_match is None or desc_match is None:
71
+ continue
72
+ yield Record(
73
+ db=_NAME,
74
+ gene=desc_match.group(1),
75
+ sequence=fasta.sequence,
76
+ accession=id_match.group(2),
77
+ function=("virulence",),
78
+ product=fasta.description,
79
+ source_id=fasta.id,
80
+ )
81
+ finally:
82
+ scratch.unlink(missing_ok=True)
83
+
84
+
85
+ PROVIDER = Provider(
86
+ name=_NAME,
87
+ description="VFDB virulence factors (set A, nucleotide)",
88
+ source_urls=("http://www.mgc.ac.cn/VFs/Down/VFDB_setA_nt.fas.gz",),
89
+ dbtype="nucl",
90
+ transform=transform,
91
+ snapshot="vfdb.tar.gz",
92
+ )
@@ -0,0 +1,109 @@
1
+ """Victors provider (virulence factors, nucleotide) — transform only.
2
+
3
+ Upstream ``get_victors`` (abricate-get_db 1.4.0) cross-references two
4
+ phidias.us downloads, both served under ``.php`` URLs:
5
+
6
+ - ``gen_downloads.php`` — nucleotide CDS content (.ffn), ids like
7
+ ``gi|115534241:2616-3152`` carrying the source GI and coordinates;
8
+ - ``gen_downloads_protein.php`` — protein content (.faa), headers like
9
+ ``>gi|115534244|ref|YP_783826.1| hypothetical protein pCJ01p4
10
+ [Campylobacter jejuni]``.
11
+
12
+ Protein headers are keyed by gi; each .ffn record looks its gi up in that
13
+ map for accession and product, falling back to ``gi|<gi>:<start>-<stop>``
14
+ and ``hypothetical protein``. The gene stays the ORIGINAL .ffn id —
15
+ upstream never renames it here.
16
+
17
+ Perl quirks (phase 7 notepad): upstream's ``.`` in both regexes matches
18
+ the literal ``|`` (escaped here); a protein header failing the regex is
19
+ skipped (``next unless``); an .ffn id failing its regex would reuse the
20
+ PREVIOUS match's stale capture variables — such records are skipped, the
21
+ same policy as the vfdb provider. Upstream's product line
22
+ (``$s->{DESC} =~ ... || 'hypothetical protein'``) is a void-context match
23
+ whose result is discarded; the notepad-locked digest semantics (map
24
+ product or ``hypothetical protein``) are implemented instead.
25
+ """
26
+
27
+ import re
28
+ from collections.abc import Iterable
29
+ from pathlib import Path
30
+
31
+ from gapit.fasta import iter_fasta
32
+ from gapit.providers.common import Provider
33
+ from gapit.records import Record
34
+
35
+ _NAME = "victors"
36
+ _FFN_NAME = "gen_downloads.php" # nucleotide CDS content
37
+ _FAA_NAME = "gen_downloads_protein.php" # protein content
38
+ _HYPOTHETICAL = "hypothetical protein"
39
+ _FUNCTION = ("virulence",) # locked gapit/v1 func vocabulary (Wave F2c)
40
+
41
+ # Upstream m"^>gi.(\d+).ref.([^|]+). ([^[]+)" — its dots matched the pipes.
42
+ # \s+ eats the id/product separator; the product runs to end of line.
43
+ _FAA_HEADER = re.compile(r">gi\|(\d+)\|ref\|([^|]+)\|\s+(.+)")
44
+ # Upstream m/gi.(\d+):(\d+)-(\d+)/ searched against the .ffn record id.
45
+ _FFN_ID = re.compile(r"gi\|(\d+):(\d+)-(\d+)")
46
+
47
+
48
+ def _read_protein_map(path: Path) -> dict[str, tuple[str, str]]:
49
+ """gi -> (accession, product) from .faa header lines.
50
+
51
+ Headers failing the pattern are skipped (upstream ``next unless``);
52
+ the product stops at the first `` [`` strain bracket (upstream
53
+ ``([^[]+)``). A later duplicate gi overwrites an earlier one, exactly
54
+ like upstream's hash assignment.
55
+ """
56
+ gi_map: dict[str, tuple[str, str]] = {}
57
+ with path.open(encoding="utf-8") as handle:
58
+ for line in handle:
59
+ header = _FAA_HEADER.match(line)
60
+ if header is None:
61
+ continue
62
+ gi, accession, product = (
63
+ header.group(1),
64
+ header.group(2),
65
+ header.group(3).split(" [", 1)[0],
66
+ )
67
+ gi_map[gi] = (accession, product)
68
+ return gi_map
69
+
70
+
71
+ def transform(workdir: Path) -> Iterable[Record]:
72
+ """Yield Records from the two phidias.us downloads in ``workdir``.
73
+
74
+ Each .ffn record keeps its ORIGINAL id as the gene; the gi carved out
75
+ of that id looks accession/product up in the protein map, falling back
76
+ to ``gi|<gi>:<start>-<stop>`` / ``hypothetical protein`` for an unknown
77
+ gi. Records whose id carries no gi coordinates are skipped (upstream
78
+ would build them from the previous match's stale captures).
79
+ """
80
+ gi_map = _read_protein_map(workdir / _FAA_NAME)
81
+ for fasta in iter_fasta(workdir / _FFN_NAME):
82
+ coords = _FFN_ID.search(fasta.id)
83
+ if coords is None:
84
+ continue
85
+ gi = coords.group(1)
86
+ fallback = f"gi|{gi}:{coords.group(2)}-{coords.group(3)}"
87
+ accession, product = gi_map.get(gi, (fallback, _HYPOTHETICAL))
88
+ yield Record(
89
+ db=_NAME,
90
+ gene=fasta.id,
91
+ sequence=fasta.sequence,
92
+ accession=accession,
93
+ function=_FUNCTION,
94
+ product=product,
95
+ source_id=gi,
96
+ )
97
+
98
+
99
+ PROVIDER = Provider(
100
+ name=_NAME,
101
+ description="Victors virulence factors",
102
+ source_urls=(
103
+ "http://phidias.us/victors/downloads/gen_downloads.php",
104
+ "http://phidias.us/victors/downloads/gen_downloads_protein.php",
105
+ ),
106
+ dbtype="nucl",
107
+ transform=transform,
108
+ snapshot=None,
109
+ )
gapit/py.typed ADDED
File without changes