gapit 0.2.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. gapit/__init__.py +3 -0
  2. gapit/blast.py +239 -0
  3. gapit/cli.py +128 -0
  4. gapit/cmd_db.py +113 -0
  5. gapit/cmd_db_build.py +72 -0
  6. gapit/cmd_db_install.py +126 -0
  7. gapit/cmd_db_outdated.py +51 -0
  8. gapit/cmd_db_search.py +66 -0
  9. gapit/cmd_screen.py +214 -0
  10. gapit/cmd_summary.py +65 -0
  11. gapit/config.py +51 -0
  12. gapit/data/snapshots/card.tar.gz +0 -0
  13. gapit/data/snapshots/vfdb.tar.gz +0 -0
  14. gapit/db.py +226 -0
  15. gapit/db_build_ops.py +210 -0
  16. gapit/db_ops.py +128 -0
  17. gapit/db_query_ops.py +252 -0
  18. gapit/dbbuild.py +207 -0
  19. gapit/dbcodec.py +117 -0
  20. gapit/dispatch.py +31 -0
  21. gapit/errors.py +87 -0
  22. gapit/fasta.py +123 -0
  23. gapit/formats/__init__.py +1 -0
  24. gapit/formats/json.py +309 -0
  25. gapit/formats/md.py +190 -0
  26. gapit/formats/schemas.py +30 -0
  27. gapit/formats/summary.py +103 -0
  28. gapit/formats/tsv.py +45 -0
  29. gapit/hits.py +107 -0
  30. gapit/mcp.py +158 -0
  31. gapit/mcp_schemas.py +123 -0
  32. gapit/mcp_tools.py +289 -0
  33. gapit/minimap.py +20 -0
  34. gapit/minimap2_run.py +114 -0
  35. gapit/paf.py +115 -0
  36. gapit/proctools.py +24 -0
  37. gapit/providers/__init__.py +39 -0
  38. gapit/providers/argannot.py +94 -0
  39. gapit/providers/bacmet2.py +59 -0
  40. gapit/providers/card.py +150 -0
  41. gapit/providers/common.py +245 -0
  42. gapit/providers/ecoh.py +63 -0
  43. gapit/providers/ecoli_vf.py +74 -0
  44. gapit/providers/megares.py +71 -0
  45. gapit/providers/ncbi.py +103 -0
  46. gapit/providers/plasmidfinder.py +69 -0
  47. gapit/providers/resfinder.py +123 -0
  48. gapit/providers/snapshots.py +119 -0
  49. gapit/providers/upec_expec_vf.py +85 -0
  50. gapit/providers/vfdb.py +92 -0
  51. gapit/providers/victors.py +109 -0
  52. gapit/py.typed +0 -0
  53. gapit/reads.py +221 -0
  54. gapit/records.py +152 -0
  55. gapit/report.py +25 -0
  56. gapit/screening.py +145 -0
  57. gapit/screening_reads.py +255 -0
  58. gapit/seqconvert.py +203 -0
  59. gapit/summary.py +151 -0
  60. gapit-0.2.2.dist-info/METADATA +183 -0
  61. gapit-0.2.2.dist-info/RECORD +64 -0
  62. gapit-0.2.2.dist-info/WHEEL +4 -0
  63. gapit-0.2.2.dist-info/entry_points.txt +3 -0
  64. gapit-0.2.2.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,51 @@
1
+ """The `gapit db outdated` command: typer shell over the use-case
2
+ (:mod:`gapit.db_query_ops`) — installed-database staleness report.
3
+
4
+ Reports every installed database's age against a staleness threshold and
5
+ the bundled snapshot date. Staleness is a REPORT, never an error state:
6
+ exit 0 even when everything is stale. Read-only: no builds, no network,
7
+ no datadir writes.
8
+ """
9
+
10
+ from typing import Annotated
11
+
12
+ import typer
13
+
14
+ from gapit.db_query_ops import (
15
+ DEFAULT_STALE_DAYS,
16
+ DbOutdatedDocument,
17
+ outdated_tsv_lines,
18
+ perform_outdated,
19
+ )
20
+ from gapit.dispatch import Datadir, dispatch
21
+
22
+
23
+ def db_outdated_command(
24
+ datadir: Datadir = None,
25
+ days: Annotated[
26
+ int,
27
+ typer.Option("--days", min=0, help="Staleness threshold in days (stale past this age)."),
28
+ ] = DEFAULT_STALE_DAYS,
29
+ as_json: Annotated[
30
+ bool,
31
+ typer.Option("--json", help="Print machine-readable JSON instead of a table."),
32
+ ] = False,
33
+ quiet: Annotated[bool, typer.Option("--quiet", help="Silence stderr diagnostics.")] = False,
34
+ ) -> None:
35
+ """Report installed database ages and available updates (exit 0 however
36
+ stale things are — a report, not an error).
37
+
38
+ A database is `stale` past --days (default 90) and `snapshot-update`
39
+ when its provider's bundled snapshot is newer than the installed copy.
40
+ """
41
+
42
+ def run() -> None:
43
+ entries = perform_outdated(datadir, days=days)
44
+ if as_json:
45
+ document = DbOutdatedDocument(databases=tuple(entries))
46
+ typer.echo(document.model_dump_json(indent=2, by_alias=True))
47
+ return
48
+ for line in outdated_tsv_lines(entries):
49
+ typer.echo(line)
50
+
51
+ dispatch(run)
gapit/cmd_db_search.py ADDED
@@ -0,0 +1,66 @@
1
+ """The `gapit db search` command: typer shell over the use-case
2
+ (:mod:`gapit.db_query_ops`) — case-insensitive lookup across the
3
+ ``records.jsonl`` truth stores of every installed database (SPEC.md §11).
4
+
5
+ Read-only: no builds, no network, no datadir writes.
6
+ """
7
+
8
+ from typing import Annotated
9
+
10
+ import typer
11
+
12
+ from gapit.db_query_ops import (
13
+ DEFAULT_LIMIT,
14
+ SearchField,
15
+ perform_search,
16
+ search_json_line,
17
+ search_tsv_row,
18
+ )
19
+ from gapit.dispatch import Datadir, dispatch
20
+ from gapit.proctools import note
21
+
22
+
23
+ def db_search_command(
24
+ term: Annotated[str, typer.Argument(help="Search term (case-insensitive).")],
25
+ datadir: Datadir = None,
26
+ db: Annotated[
27
+ str | None,
28
+ typer.Option("--db", help="Restrict the scan to one installed database."),
29
+ ] = None,
30
+ field: Annotated[
31
+ SearchField,
32
+ typer.Option("--field", help="Field to match: gene, accession, function, product, any."),
33
+ ] = SearchField.any,
34
+ exact: Annotated[
35
+ bool,
36
+ typer.Option("--exact", help="Full-field equality instead of substring."),
37
+ ] = False,
38
+ limit: Annotated[
39
+ int,
40
+ typer.Option("--limit", min=0, help="Max hits to print (0 = unlimited)."),
41
+ ] = DEFAULT_LIMIT,
42
+ as_json: Annotated[
43
+ bool,
44
+ typer.Option("--json", help="Print JSONL lines instead of TSV rows."),
45
+ ] = False,
46
+ quiet: Annotated[bool, typer.Option("--quiet", help="Silence stderr diagnostics.")] = False,
47
+ ) -> None:
48
+ """Search records.jsonl across installed databases (zero hits exit 0)."""
49
+
50
+ def run() -> None:
51
+ hits, total = perform_search(
52
+ term,
53
+ datadir,
54
+ db=db,
55
+ field=field,
56
+ exact=exact,
57
+ limit=limit,
58
+ render=search_json_line if as_json else search_tsv_row,
59
+ quiet=quiet,
60
+ )
61
+ if limit > 0 and total > limit:
62
+ note(quiet, f"truncated to {limit} of {total} matching records (use --limit 0 for all)")
63
+ for line in hits:
64
+ typer.echo(line)
65
+
66
+ dispatch(run)
gapit/cmd_screen.py ADDED
@@ -0,0 +1,214 @@
1
+ """The `gapit screen` command: contig files (blastn) or reads and assemblies
2
+ (minimap2) screening.
3
+
4
+ Lives outside cli.py to keep that module small; cli.py registers it via
5
+ ``register_screen_command``.
6
+ """
7
+
8
+ from pathlib import Path
9
+ from typing import Annotated
10
+
11
+ import typer
12
+
13
+ from gapit.dispatch import Datadir, dispatch
14
+ from gapit.errors import usage_fail
15
+ from gapit.reads import ReadTypeEnum
16
+ from gapit.screening import AlignerEnum, OutputFormat, run_screen
17
+ from gapit.screening_reads import run_screen_assemblies, run_screen_reads
18
+
19
+
20
+ def _split_read_list(raw: str, flag: str) -> list[Path]:
21
+ """Split a comma-separated --r1/--r2 value into paths; empty elements
22
+ are usage errors (an empty string would silently become the cwd)."""
23
+ parts = [part.strip() for part in raw.split(",")]
24
+ if any(not part for part in parts):
25
+ usage_fail(f"{flag} contains an empty element: {raw!r}")
26
+ return [Path(part) for part in parts]
27
+
28
+
29
+ def screen_command(
30
+ files: Annotated[
31
+ list[Path] | None,
32
+ typer.Argument(help="Input FASTA/GBK/EMBL contig file(s) to screen."),
33
+ ] = None,
34
+ r1: Annotated[
35
+ str | None,
36
+ typer.Option(
37
+ "--r1",
38
+ help="Comma-separated FASTQ reads or assembly FASTA file(s), one per lane.",
39
+ ),
40
+ ] = None,
41
+ r2: Annotated[
42
+ str | None,
43
+ typer.Option("--r2", help="Comma-separated mate FASTQ file(s); must match --r1 count."),
44
+ ] = None,
45
+ read_type: Annotated[
46
+ ReadTypeEnum | None,
47
+ typer.Option(
48
+ "--read-type",
49
+ help=(
50
+ "minimap2 preset for reads mode (default: sr for FASTQ, map-ont"
51
+ " for assembly FASTA)."
52
+ ),
53
+ ),
54
+ ] = None,
55
+ min_breadth: Annotated[
56
+ float,
57
+ typer.Option("--min-breadth", help="Reads mode: minimum %breadth for presence."),
58
+ ] = 90.0,
59
+ min_identity: Annotated[
60
+ float,
61
+ typer.Option(
62
+ "--min-identity",
63
+ help=(
64
+ "Reads mode: minimum %identity per alignment, 0 <= x <= 100 (0 = off;"
65
+ " any nonzero value emits gapit.reads/2)."
66
+ ),
67
+ ),
68
+ ] = 0.0,
69
+ min_mapq: Annotated[
70
+ int,
71
+ typer.Option(
72
+ "--min-mapq",
73
+ help=(
74
+ "Reads mode: minimum MAPQ per alignment (0 = off; any nonzero value"
75
+ " emits gapit.reads/2)."
76
+ ),
77
+ ),
78
+ ] = 0,
79
+ aligner: Annotated[
80
+ AlignerEnum | None,
81
+ typer.Option(
82
+ "--aligner",
83
+ help=(
84
+ "Alignment engine (default: blastn for contig files, minimap2 for --r1/--r2 reads)."
85
+ ),
86
+ ),
87
+ ] = None,
88
+ db: Annotated[
89
+ str, typer.Option("--db", help="Database to screen against (datadir subdir).")
90
+ ] = "ncbi",
91
+ datadir: Datadir = None,
92
+ minid: Annotated[
93
+ float, typer.Option("--minid", help="Minimum %identity, 0 < x <= 100.")
94
+ ] = 80.0,
95
+ mincov: Annotated[
96
+ float, typer.Option("--mincov", help="Minimum %coverage, 0 <= x <= 100.")
97
+ ] = 80.0,
98
+ threads: Annotated[int, typer.Option("--threads", help="BLAST worker threads.")] = 1,
99
+ jobs: Annotated[
100
+ int,
101
+ typer.Option(
102
+ "--jobs",
103
+ help=(
104
+ "Screen N input files concurrently (gapit extension; output order is"
105
+ " always input order). Each worker runs its own BLAST against the"
106
+ " shared db index, which BLAST mmaps — concurrent readers are fine."
107
+ ),
108
+ ),
109
+ ] = 1,
110
+ fofn: Annotated[
111
+ Path | None,
112
+ typer.Option("--fofn", help="File of filenames; replaces the positional FILEs."),
113
+ ] = None,
114
+ quiet: Annotated[bool, typer.Option("--quiet", help="Silence stderr diagnostics.")] = False,
115
+ noheader: Annotated[bool, typer.Option("--noheader", help="Suppress the header row.")] = False,
116
+ nopath: Annotated[bool, typer.Option("--nopath", help="Basename the FILE column.")] = False,
117
+ debug: Annotated[bool, typer.Option("--debug", help="Verbose stderr diagnostics.")] = False,
118
+ output_format: Annotated[
119
+ OutputFormat | None,
120
+ typer.Option("--format", help="Output format (reads mode defaults to json)."),
121
+ ] = None,
122
+ ) -> None:
123
+ """Screen contig files or FASTQ reads (R1 and R2 comma-lists, one lane
124
+ each) for AMR/virulence genes."""
125
+
126
+ def run() -> None:
127
+ if (r1 is not None or r2 is not None) and files:
128
+ usage_fail("--r1/--r2 and positional contig FILEs are mutually exclusive")
129
+ if r2 is not None and r1 is None:
130
+ usage_fail("--r2 requires --r1")
131
+ if (
132
+ (min_identity > 0 or min_mapq > 0)
133
+ and r1 is None
134
+ and aligner is not AlignerEnum.minimap2
135
+ ):
136
+ usage_fail("--min-identity/--min-mapq are reads-mode only (minimap2 engine)")
137
+ if r1 is not None or r2 is not None:
138
+ if fofn is not None:
139
+ usage_fail("--fofn is not available in reads mode")
140
+ if noheader:
141
+ usage_fail("--noheader is not available in reads mode")
142
+ if nopath:
143
+ usage_fail("--nopath is not available in reads mode")
144
+ if jobs != 1:
145
+ usage_fail("--jobs is not available in reads mode")
146
+ typer.echo(
147
+ run_screen_reads(
148
+ _split_read_list(r1 or "", "--r1"),
149
+ _split_read_list(r2, "--r2") if r2 is not None else None,
150
+ db,
151
+ datadir,
152
+ read_type,
153
+ min_breadth,
154
+ min_identity,
155
+ min_mapq,
156
+ threads,
157
+ output_format,
158
+ quiet,
159
+ debug,
160
+ aligner=aligner,
161
+ minid=minid,
162
+ mincov=mincov,
163
+ ),
164
+ nl=False,
165
+ )
166
+ elif aligner is AlignerEnum.minimap2:
167
+ typer.echo(
168
+ run_screen_assemblies(
169
+ files,
170
+ fofn,
171
+ db,
172
+ datadir,
173
+ read_type,
174
+ min_breadth,
175
+ min_identity,
176
+ min_mapq,
177
+ threads,
178
+ jobs,
179
+ noheader,
180
+ nopath,
181
+ output_format,
182
+ quiet,
183
+ debug,
184
+ minid=minid,
185
+ mincov=mincov,
186
+ ),
187
+ nl=False,
188
+ )
189
+ else:
190
+ typer.echo(
191
+ run_screen(
192
+ files,
193
+ db,
194
+ datadir,
195
+ minid,
196
+ mincov,
197
+ threads,
198
+ jobs,
199
+ fofn,
200
+ quiet,
201
+ noheader,
202
+ nopath,
203
+ debug,
204
+ output_format or OutputFormat.tsv,
205
+ ),
206
+ nl=False,
207
+ )
208
+
209
+ dispatch(run)
210
+
211
+
212
+ def register_screen_command(app: typer.Typer) -> None:
213
+ """Attach the screen command to the CLI app."""
214
+ app.command("screen")(screen_command)
gapit/cmd_summary.py ADDED
@@ -0,0 +1,65 @@
1
+ """The `gapit summary` command: matrix over abricate-format report tables.
2
+
3
+ Lives outside cli.py to keep that module small; cli.py registers it via
4
+ ``register_summary_command``.
5
+ """
6
+
7
+ from datetime import UTC, datetime
8
+ from pathlib import Path
9
+ from typing import Annotated
10
+
11
+ import typer
12
+
13
+ from gapit.dispatch import dispatch
14
+ from gapit.errors import usage_fail
15
+ from gapit.formats.summary import format_summary_tsv, render_summary_json, render_summary_md
16
+ from gapit.screening import OutputFormat
17
+ from gapit.summary import SummaryParams, build_summary
18
+
19
+
20
+ def summary_command(
21
+ files: Annotated[
22
+ list[Path] | None,
23
+ typer.Argument(help="Abricate-format report file(s) to summarize."),
24
+ ] = None,
25
+ identity: Annotated[
26
+ bool,
27
+ typer.Option("--identity", help="Cells show %IDENTITY instead of %COVERAGE."),
28
+ ] = False,
29
+ nopath: Annotated[
30
+ bool,
31
+ typer.Option("--nopath", help="Basename row keys (FILE values / input filenames)."),
32
+ ] = False,
33
+ quiet: Annotated[bool, typer.Option("--quiet", help="Silence stderr diagnostics.")] = False,
34
+ output_format: Annotated[
35
+ OutputFormat, typer.Option("--format", help="Output format.")
36
+ ] = OutputFormat.tsv,
37
+ ) -> None:
38
+ """Summarize report table(s) into a gene presence/absence matrix."""
39
+
40
+ def run() -> None:
41
+ if not files:
42
+ usage_fail("summary needs >= 1 report file(s)")
43
+ if identity and not quiet:
44
+ typer.echo("Using %IDENTITY for the summary table instead of %COVERAGE", err=True)
45
+ params = SummaryParams(identity=identity, nopath=nopath)
46
+
47
+ def warn(message: str) -> None:
48
+ if not quiet:
49
+ typer.echo(f"WARNING: {message}", err=True)
50
+
51
+ matrix = build_summary(list(files), params, warn=warn)
52
+ now = datetime.now(UTC)
53
+ if output_format is OutputFormat.json:
54
+ typer.echo(render_summary_json(matrix, now=now), nl=False)
55
+ elif output_format is OutputFormat.md:
56
+ typer.echo(render_summary_md(matrix, now=now), nl=False)
57
+ else:
58
+ typer.echo(format_summary_tsv(matrix, csv=output_format is OutputFormat.csv), nl=False)
59
+
60
+ dispatch(run)
61
+
62
+
63
+ def register_summary_command(app: typer.Typer) -> None:
64
+ """Attach the summary command to the CLI app."""
65
+ app.command("summary")(summary_command)
gapit/config.py ADDED
@@ -0,0 +1,51 @@
1
+ """Datadir resolution and configuration defaults."""
2
+
3
+ import os
4
+ from pathlib import Path
5
+
6
+ from gapit.errors import DatabaseError
7
+
8
+ ENV_DATADIR = "GAPIT_DATADIR"
9
+ DEFAULT_DATADIR = Path("~/.local/share/gapit/db")
10
+
11
+
12
+ def resolve_datadir(cli_value: Path | None) -> Path:
13
+ """Resolve the datadir: CLI ``--datadir`` > ``$GAPIT_DATADIR`` > ``~/.local/share/gapit/db``.
14
+
15
+ Expands ``~`` and resolves to an absolute path; raises DatabaseError when the
16
+ resolved directory does not exist.
17
+ """
18
+ raw = cli_value if cli_value is not None else os.environ.get(ENV_DATADIR, DEFAULT_DATADIR)
19
+ datadir = Path(raw).expanduser().resolve()
20
+ if not datadir.is_dir():
21
+ raise DatabaseError(
22
+ f"datadir does not exist: {datadir}",
23
+ code="DATADIR_NOT_FOUND",
24
+ context={"datadir": str(datadir)},
25
+ )
26
+ return datadir
27
+
28
+
29
+ def ensure_datadir(cli_value: Path | None) -> Path:
30
+ """Resolve the datadir and mkdir it when absent (fresh-machine bootstrap).
31
+
32
+ Write paths (``db fetch``, ``db build``) bootstrap a missing root; every
33
+ read path still demands it via ``resolve_datadir``. The resolved path is
34
+ recovered from ``resolve_datadir``'s DATADIR_NOT_FOUND context — this
35
+ module stays the single owner of resolution.
36
+ """
37
+ try:
38
+ root = resolve_datadir(cli_value)
39
+ except DatabaseError as exc:
40
+ if exc.code != "DATADIR_NOT_FOUND":
41
+ raise
42
+ root = Path(exc.context["datadir"])
43
+ try:
44
+ root.mkdir(parents=True, exist_ok=True)
45
+ except OSError as exc:
46
+ raise DatabaseError(
47
+ f"cannot create datadir: {root}",
48
+ code="DATADIR_CREATE_FAILED",
49
+ context={"datadir": str(root)},
50
+ ) from exc
51
+ return root
Binary file
Binary file
gapit/db.py ADDED
@@ -0,0 +1,226 @@
1
+ """Database layer: discovery, ``~~~`` header parsing, makeblastdb/blastdbcmd wrappers."""
2
+
3
+ import os
4
+ import re
5
+ import shlex
6
+ import sys
7
+ from pathlib import Path
8
+ from typing import Literal
9
+
10
+ from pydantic import BaseModel
11
+
12
+ from gapit.errors import DatabaseError, InputError
13
+ from gapit.fasta import iter_fasta
14
+ from gapit.proctools import run_tool
15
+
16
+ IDSEP = "~~~"
17
+
18
+
19
+ class DbHeader(BaseModel, frozen=True):
20
+ """Parsed ``~~~`` fields of a database sequence id (SPEC.md §4 step 5).
21
+
22
+ ``function`` carries functional categories (AMR classes, virulence, ...):
23
+ the legacy ``~~~`` branch fills it with abricate's 4th resistance field
24
+ verbatim — same bytes, better name (Wave F1).
25
+ """
26
+
27
+ database: str
28
+ gene: str
29
+ accession: str
30
+ function: str
31
+
32
+
33
+ class BlastDbInfo(BaseModel, frozen=True):
34
+ """Introspection of one built BLAST database (``blastdbcmd -info``)."""
35
+
36
+ n_sequences: int
37
+ dbtype: Literal["nucl", "prot"]
38
+ title: str
39
+ date: str
40
+
41
+
42
+ class Database(BaseModel, frozen=True):
43
+ """One discovered database directory under the datadir."""
44
+
45
+ name: str
46
+ path: Path
47
+ sequences_path: Path
48
+
49
+
50
+ class DatabaseInfo(BaseModel, frozen=True):
51
+ """One row of ``gapit list`` output."""
52
+
53
+ name: str
54
+ n_sequences: int
55
+ dbtype: Literal["nucl", "prot"]
56
+ date: str
57
+
58
+
59
+ def parse_db_header(seqid: str, default_db: str) -> DbHeader:
60
+ """Split a seqid on ``~~~`` into (database, gene, accession, function).
61
+
62
+ Without any ``~~~`` the whole seqid is the gene and the default database is
63
+ used; partial headers leave trailing fields empty; an empty database field
64
+ also falls back to ``default_db`` (SPEC.md §4 step 5).
65
+ """
66
+ fields = seqid.split(IDSEP)
67
+ if len(fields) == 1:
68
+ return DbHeader(database=default_db, gene=seqid, accession="", function="")
69
+ fields += [""] * (4 - len(fields))
70
+ database, gene, accession, function = fields[:4]
71
+ return DbHeader(
72
+ database=database or default_db,
73
+ gene=gene,
74
+ accession=accession,
75
+ function=function,
76
+ )
77
+
78
+
79
+ _AGTC_TABLE = str.maketrans("", "", "AGTCagtc")
80
+
81
+
82
+ def mol_type(letters: str) -> Literal["nucl", "prot"]:
83
+ """Abricate heuristic (SPEC.md §2): protein iff non-[AGTC] letters are
84
+ strictly more than half of the input (empty input is nucleotide)."""
85
+ non_agtc = len(letters.translate(_AGTC_TABLE))
86
+ return "prot" if 2 * non_agtc > len(letters) else "nucl"
87
+
88
+
89
+ def make_blast_db(
90
+ sequences_path: Path,
91
+ name: str,
92
+ *,
93
+ dbtype: Literal["nucl", "prot"] | None = None,
94
+ debug: bool = False,
95
+ ) -> None:
96
+ """(Re)build the BLAST index for one database directory.
97
+
98
+ ``dbtype=None`` (the default) keeps the abricate ``mol_type`` heuristic;
99
+ an explicit ``"nucl"``/``"prot"`` — e.g. from a gapit manifest, which
100
+ declares the type — skips the heuristic entirely. With ``debug``, echo
101
+ the makeblastdb argv to stderr (abricate --debug parity).
102
+ """
103
+ if dbtype is None:
104
+ letters = "".join(record.sequence for record in iter_fasta(sequences_path))
105
+ dbtype = mol_type(letters)
106
+ for index_file in sequences_path.parent.glob(f"{sequences_path.name}.[np]??"):
107
+ index_file.unlink()
108
+ argv = [
109
+ "makeblastdb",
110
+ "-in",
111
+ str(sequences_path),
112
+ "-title",
113
+ name,
114
+ "-dbtype",
115
+ dbtype,
116
+ "-logfile",
117
+ "/dev/null",
118
+ ]
119
+ if debug:
120
+ print(f"gapit: run: {shlex.join(argv)}", file=sys.stderr)
121
+ result = run_tool(argv)
122
+ if result.returncode != 0:
123
+ raise DatabaseError(
124
+ f"makeblastdb failed for {name}: {result.stderr.strip()}",
125
+ code="MAKEBLASTDB_FAILED",
126
+ )
127
+
128
+
129
+ _SEQUENCES_RE = re.compile(r"([\d,]+)\s+sequences;")
130
+ _TITLE_RE = re.compile(r"^Database:[ \t]*(.*)$", re.MULTILINE)
131
+ _DATE_RE = re.compile(r"Date:\s*([A-Za-z]{3})\s+(\d{1,2}),\s+(\d{4})")
132
+
133
+
134
+ def parse_blastdbcmd_info(text: str) -> BlastDbInfo:
135
+ """Parse ``blastdbcmd -info`` output (SPEC.md §2)."""
136
+ sequences_match = _SEQUENCES_RE.search(text)
137
+ title_match = _TITLE_RE.search(text)
138
+ date_match = _DATE_RE.search(text)
139
+ if sequences_match is None or title_match is None or date_match is None:
140
+ raise DatabaseError(
141
+ f"could not parse blastdbcmd output: {text!r}",
142
+ code="BLASTDBCMD_PARSE_FAILED",
143
+ )
144
+ month, day, year = date_match.group(1), date_match.group(2), date_match.group(3)
145
+ return BlastDbInfo(
146
+ n_sequences=int(sequences_match.group(1).replace(",", "")),
147
+ dbtype="prot" if "total residues" in text else "nucl",
148
+ title=title_match.group(1).strip(),
149
+ date=f"{year}-{month}-{int(day):02d}",
150
+ )
151
+
152
+
153
+ def blast_db_info(db_prefix: Path) -> BlastDbInfo:
154
+ """Introspect a built BLAST database via ``blastdbcmd -info``."""
155
+ result = run_tool(["blastdbcmd", "-info", "-db", str(db_prefix)])
156
+ if result.returncode != 0:
157
+ raise DatabaseError(
158
+ f"Database {db_prefix} is not indexed, please try: gapit setupdb",
159
+ code="DATABASE_NOT_INDEXED",
160
+ context={"db": str(db_prefix)},
161
+ )
162
+ return parse_blastdbcmd_info(result.stdout)
163
+
164
+
165
+ def discover_databases(datadir: Path) -> list[Database]:
166
+ """Immediate subdirectories of the datadir holding a readable ``sequences``
167
+ file, sorted by name."""
168
+ databases: list[Database] = []
169
+ for child in datadir.iterdir():
170
+ sequences = child / "sequences"
171
+ if child.is_dir() and sequences.is_file() and os.access(sequences, os.R_OK):
172
+ databases.append(Database(name=child.name, path=child, sequences_path=sequences))
173
+ return sorted(databases, key=lambda database: database.name)
174
+
175
+
176
+ def _manifest_dbtype(db_dir: Path) -> Literal["nucl", "prot"] | None:
177
+ """The dbtype declared by a database's gapit manifest, or ``None`` when the
178
+ directory has none (abricate-built) — the caller then falls back to the
179
+ ``mol_type`` heuristic. A malformed manifest propagates (no silent
180
+ degradation).
181
+
182
+ Lazy import: records -> formats.json -> reads -> db is a cycle at module
183
+ level (Wave C-a notepad hazard).
184
+ """
185
+ from gapit.records import read_manifest
186
+
187
+ try:
188
+ return read_manifest(db_dir / "gapit-manifest.json").dbtype
189
+ except InputError as exc:
190
+ if exc.code != "INPUT_NOT_FOUND":
191
+ raise
192
+ return None
193
+
194
+
195
+ def list_databases(datadir: Path, *, setupdb: bool, debug: bool = False) -> list[DatabaseInfo]:
196
+ """Enumerate databases (building indices first when ``setupdb``), requiring
197
+ every database to be indexed; returns one DatabaseInfo per database."""
198
+ infos: list[DatabaseInfo] = []
199
+ for database in discover_databases(datadir):
200
+ if setupdb:
201
+ make_blast_db(
202
+ database.sequences_path,
203
+ database.name,
204
+ dbtype=_manifest_dbtype(database.path),
205
+ debug=debug,
206
+ )
207
+ sequences = database.sequences_path
208
+ index_exists = any(
209
+ (sequences.parent / f"{sequences.name}{suffix}").exists() for suffix in (".nin", ".pin")
210
+ )
211
+ if not index_exists:
212
+ raise DatabaseError(
213
+ f"Database {database.name} is not indexed, please try: gapit setupdb",
214
+ code="DATABASE_NOT_INDEXED",
215
+ context={"db": database.name},
216
+ )
217
+ info = blast_db_info(sequences)
218
+ infos.append(
219
+ DatabaseInfo(
220
+ name=database.name,
221
+ n_sequences=info.n_sequences,
222
+ dbtype=info.dbtype,
223
+ date=info.date,
224
+ )
225
+ )
226
+ return infos