gapit 0.2.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gapit/__init__.py +3 -0
- gapit/blast.py +239 -0
- gapit/cli.py +128 -0
- gapit/cmd_db.py +113 -0
- gapit/cmd_db_build.py +72 -0
- gapit/cmd_db_install.py +126 -0
- gapit/cmd_db_outdated.py +51 -0
- gapit/cmd_db_search.py +66 -0
- gapit/cmd_screen.py +214 -0
- gapit/cmd_summary.py +65 -0
- gapit/config.py +51 -0
- gapit/data/snapshots/card.tar.gz +0 -0
- gapit/data/snapshots/vfdb.tar.gz +0 -0
- gapit/db.py +226 -0
- gapit/db_build_ops.py +210 -0
- gapit/db_ops.py +128 -0
- gapit/db_query_ops.py +252 -0
- gapit/dbbuild.py +207 -0
- gapit/dbcodec.py +117 -0
- gapit/dispatch.py +31 -0
- gapit/errors.py +87 -0
- gapit/fasta.py +123 -0
- gapit/formats/__init__.py +1 -0
- gapit/formats/json.py +309 -0
- gapit/formats/md.py +190 -0
- gapit/formats/schemas.py +30 -0
- gapit/formats/summary.py +103 -0
- gapit/formats/tsv.py +45 -0
- gapit/hits.py +107 -0
- gapit/mcp.py +158 -0
- gapit/mcp_schemas.py +123 -0
- gapit/mcp_tools.py +289 -0
- gapit/minimap.py +20 -0
- gapit/minimap2_run.py +114 -0
- gapit/paf.py +115 -0
- gapit/proctools.py +24 -0
- gapit/providers/__init__.py +39 -0
- gapit/providers/argannot.py +94 -0
- gapit/providers/bacmet2.py +59 -0
- gapit/providers/card.py +150 -0
- gapit/providers/common.py +245 -0
- gapit/providers/ecoh.py +63 -0
- gapit/providers/ecoli_vf.py +74 -0
- gapit/providers/megares.py +71 -0
- gapit/providers/ncbi.py +103 -0
- gapit/providers/plasmidfinder.py +69 -0
- gapit/providers/resfinder.py +123 -0
- gapit/providers/snapshots.py +119 -0
- gapit/providers/upec_expec_vf.py +85 -0
- gapit/providers/vfdb.py +92 -0
- gapit/providers/victors.py +109 -0
- gapit/py.typed +0 -0
- gapit/reads.py +221 -0
- gapit/records.py +152 -0
- gapit/report.py +25 -0
- gapit/screening.py +145 -0
- gapit/screening_reads.py +255 -0
- gapit/seqconvert.py +203 -0
- gapit/summary.py +151 -0
- gapit-0.2.2.dist-info/METADATA +183 -0
- gapit-0.2.2.dist-info/RECORD +64 -0
- gapit-0.2.2.dist-info/WHEEL +4 -0
- gapit-0.2.2.dist-info/entry_points.txt +3 -0
- gapit-0.2.2.dist-info/licenses/LICENSE +21 -0
gapit/cmd_db_outdated.py
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""The `gapit db outdated` command: typer shell over the use-case
|
|
2
|
+
(:mod:`gapit.db_query_ops`) — installed-database staleness report.
|
|
3
|
+
|
|
4
|
+
Reports every installed database's age against a staleness threshold and
|
|
5
|
+
the bundled snapshot date. Staleness is a REPORT, never an error state:
|
|
6
|
+
exit 0 even when everything is stale. Read-only: no builds, no network,
|
|
7
|
+
no datadir writes.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from typing import Annotated
|
|
11
|
+
|
|
12
|
+
import typer
|
|
13
|
+
|
|
14
|
+
from gapit.db_query_ops import (
|
|
15
|
+
DEFAULT_STALE_DAYS,
|
|
16
|
+
DbOutdatedDocument,
|
|
17
|
+
outdated_tsv_lines,
|
|
18
|
+
perform_outdated,
|
|
19
|
+
)
|
|
20
|
+
from gapit.dispatch import Datadir, dispatch
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def db_outdated_command(
|
|
24
|
+
datadir: Datadir = None,
|
|
25
|
+
days: Annotated[
|
|
26
|
+
int,
|
|
27
|
+
typer.Option("--days", min=0, help="Staleness threshold in days (stale past this age)."),
|
|
28
|
+
] = DEFAULT_STALE_DAYS,
|
|
29
|
+
as_json: Annotated[
|
|
30
|
+
bool,
|
|
31
|
+
typer.Option("--json", help="Print machine-readable JSON instead of a table."),
|
|
32
|
+
] = False,
|
|
33
|
+
quiet: Annotated[bool, typer.Option("--quiet", help="Silence stderr diagnostics.")] = False,
|
|
34
|
+
) -> None:
|
|
35
|
+
"""Report installed database ages and available updates (exit 0 however
|
|
36
|
+
stale things are — a report, not an error).
|
|
37
|
+
|
|
38
|
+
A database is `stale` past --days (default 90) and `snapshot-update`
|
|
39
|
+
when its provider's bundled snapshot is newer than the installed copy.
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
def run() -> None:
|
|
43
|
+
entries = perform_outdated(datadir, days=days)
|
|
44
|
+
if as_json:
|
|
45
|
+
document = DbOutdatedDocument(databases=tuple(entries))
|
|
46
|
+
typer.echo(document.model_dump_json(indent=2, by_alias=True))
|
|
47
|
+
return
|
|
48
|
+
for line in outdated_tsv_lines(entries):
|
|
49
|
+
typer.echo(line)
|
|
50
|
+
|
|
51
|
+
dispatch(run)
|
gapit/cmd_db_search.py
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
"""The `gapit db search` command: typer shell over the use-case
|
|
2
|
+
(:mod:`gapit.db_query_ops`) — case-insensitive lookup across the
|
|
3
|
+
``records.jsonl`` truth stores of every installed database (SPEC.md §11).
|
|
4
|
+
|
|
5
|
+
Read-only: no builds, no network, no datadir writes.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from typing import Annotated
|
|
9
|
+
|
|
10
|
+
import typer
|
|
11
|
+
|
|
12
|
+
from gapit.db_query_ops import (
|
|
13
|
+
DEFAULT_LIMIT,
|
|
14
|
+
SearchField,
|
|
15
|
+
perform_search,
|
|
16
|
+
search_json_line,
|
|
17
|
+
search_tsv_row,
|
|
18
|
+
)
|
|
19
|
+
from gapit.dispatch import Datadir, dispatch
|
|
20
|
+
from gapit.proctools import note
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def db_search_command(
|
|
24
|
+
term: Annotated[str, typer.Argument(help="Search term (case-insensitive).")],
|
|
25
|
+
datadir: Datadir = None,
|
|
26
|
+
db: Annotated[
|
|
27
|
+
str | None,
|
|
28
|
+
typer.Option("--db", help="Restrict the scan to one installed database."),
|
|
29
|
+
] = None,
|
|
30
|
+
field: Annotated[
|
|
31
|
+
SearchField,
|
|
32
|
+
typer.Option("--field", help="Field to match: gene, accession, function, product, any."),
|
|
33
|
+
] = SearchField.any,
|
|
34
|
+
exact: Annotated[
|
|
35
|
+
bool,
|
|
36
|
+
typer.Option("--exact", help="Full-field equality instead of substring."),
|
|
37
|
+
] = False,
|
|
38
|
+
limit: Annotated[
|
|
39
|
+
int,
|
|
40
|
+
typer.Option("--limit", min=0, help="Max hits to print (0 = unlimited)."),
|
|
41
|
+
] = DEFAULT_LIMIT,
|
|
42
|
+
as_json: Annotated[
|
|
43
|
+
bool,
|
|
44
|
+
typer.Option("--json", help="Print JSONL lines instead of TSV rows."),
|
|
45
|
+
] = False,
|
|
46
|
+
quiet: Annotated[bool, typer.Option("--quiet", help="Silence stderr diagnostics.")] = False,
|
|
47
|
+
) -> None:
|
|
48
|
+
"""Search records.jsonl across installed databases (zero hits exit 0)."""
|
|
49
|
+
|
|
50
|
+
def run() -> None:
|
|
51
|
+
hits, total = perform_search(
|
|
52
|
+
term,
|
|
53
|
+
datadir,
|
|
54
|
+
db=db,
|
|
55
|
+
field=field,
|
|
56
|
+
exact=exact,
|
|
57
|
+
limit=limit,
|
|
58
|
+
render=search_json_line if as_json else search_tsv_row,
|
|
59
|
+
quiet=quiet,
|
|
60
|
+
)
|
|
61
|
+
if limit > 0 and total > limit:
|
|
62
|
+
note(quiet, f"truncated to {limit} of {total} matching records (use --limit 0 for all)")
|
|
63
|
+
for line in hits:
|
|
64
|
+
typer.echo(line)
|
|
65
|
+
|
|
66
|
+
dispatch(run)
|
gapit/cmd_screen.py
ADDED
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
"""The `gapit screen` command: contig files (blastn) or reads and assemblies
|
|
2
|
+
(minimap2) screening.
|
|
3
|
+
|
|
4
|
+
Lives outside cli.py to keep that module small; cli.py registers it via
|
|
5
|
+
``register_screen_command``.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Annotated
|
|
10
|
+
|
|
11
|
+
import typer
|
|
12
|
+
|
|
13
|
+
from gapit.dispatch import Datadir, dispatch
|
|
14
|
+
from gapit.errors import usage_fail
|
|
15
|
+
from gapit.reads import ReadTypeEnum
|
|
16
|
+
from gapit.screening import AlignerEnum, OutputFormat, run_screen
|
|
17
|
+
from gapit.screening_reads import run_screen_assemblies, run_screen_reads
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _split_read_list(raw: str, flag: str) -> list[Path]:
|
|
21
|
+
"""Split a comma-separated --r1/--r2 value into paths; empty elements
|
|
22
|
+
are usage errors (an empty string would silently become the cwd)."""
|
|
23
|
+
parts = [part.strip() for part in raw.split(",")]
|
|
24
|
+
if any(not part for part in parts):
|
|
25
|
+
usage_fail(f"{flag} contains an empty element: {raw!r}")
|
|
26
|
+
return [Path(part) for part in parts]
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def screen_command(
|
|
30
|
+
files: Annotated[
|
|
31
|
+
list[Path] | None,
|
|
32
|
+
typer.Argument(help="Input FASTA/GBK/EMBL contig file(s) to screen."),
|
|
33
|
+
] = None,
|
|
34
|
+
r1: Annotated[
|
|
35
|
+
str | None,
|
|
36
|
+
typer.Option(
|
|
37
|
+
"--r1",
|
|
38
|
+
help="Comma-separated FASTQ reads or assembly FASTA file(s), one per lane.",
|
|
39
|
+
),
|
|
40
|
+
] = None,
|
|
41
|
+
r2: Annotated[
|
|
42
|
+
str | None,
|
|
43
|
+
typer.Option("--r2", help="Comma-separated mate FASTQ file(s); must match --r1 count."),
|
|
44
|
+
] = None,
|
|
45
|
+
read_type: Annotated[
|
|
46
|
+
ReadTypeEnum | None,
|
|
47
|
+
typer.Option(
|
|
48
|
+
"--read-type",
|
|
49
|
+
help=(
|
|
50
|
+
"minimap2 preset for reads mode (default: sr for FASTQ, map-ont"
|
|
51
|
+
" for assembly FASTA)."
|
|
52
|
+
),
|
|
53
|
+
),
|
|
54
|
+
] = None,
|
|
55
|
+
min_breadth: Annotated[
|
|
56
|
+
float,
|
|
57
|
+
typer.Option("--min-breadth", help="Reads mode: minimum %breadth for presence."),
|
|
58
|
+
] = 90.0,
|
|
59
|
+
min_identity: Annotated[
|
|
60
|
+
float,
|
|
61
|
+
typer.Option(
|
|
62
|
+
"--min-identity",
|
|
63
|
+
help=(
|
|
64
|
+
"Reads mode: minimum %identity per alignment, 0 <= x <= 100 (0 = off;"
|
|
65
|
+
" any nonzero value emits gapit.reads/2)."
|
|
66
|
+
),
|
|
67
|
+
),
|
|
68
|
+
] = 0.0,
|
|
69
|
+
min_mapq: Annotated[
|
|
70
|
+
int,
|
|
71
|
+
typer.Option(
|
|
72
|
+
"--min-mapq",
|
|
73
|
+
help=(
|
|
74
|
+
"Reads mode: minimum MAPQ per alignment (0 = off; any nonzero value"
|
|
75
|
+
" emits gapit.reads/2)."
|
|
76
|
+
),
|
|
77
|
+
),
|
|
78
|
+
] = 0,
|
|
79
|
+
aligner: Annotated[
|
|
80
|
+
AlignerEnum | None,
|
|
81
|
+
typer.Option(
|
|
82
|
+
"--aligner",
|
|
83
|
+
help=(
|
|
84
|
+
"Alignment engine (default: blastn for contig files, minimap2 for --r1/--r2 reads)."
|
|
85
|
+
),
|
|
86
|
+
),
|
|
87
|
+
] = None,
|
|
88
|
+
db: Annotated[
|
|
89
|
+
str, typer.Option("--db", help="Database to screen against (datadir subdir).")
|
|
90
|
+
] = "ncbi",
|
|
91
|
+
datadir: Datadir = None,
|
|
92
|
+
minid: Annotated[
|
|
93
|
+
float, typer.Option("--minid", help="Minimum %identity, 0 < x <= 100.")
|
|
94
|
+
] = 80.0,
|
|
95
|
+
mincov: Annotated[
|
|
96
|
+
float, typer.Option("--mincov", help="Minimum %coverage, 0 <= x <= 100.")
|
|
97
|
+
] = 80.0,
|
|
98
|
+
threads: Annotated[int, typer.Option("--threads", help="BLAST worker threads.")] = 1,
|
|
99
|
+
jobs: Annotated[
|
|
100
|
+
int,
|
|
101
|
+
typer.Option(
|
|
102
|
+
"--jobs",
|
|
103
|
+
help=(
|
|
104
|
+
"Screen N input files concurrently (gapit extension; output order is"
|
|
105
|
+
" always input order). Each worker runs its own BLAST against the"
|
|
106
|
+
" shared db index, which BLAST mmaps — concurrent readers are fine."
|
|
107
|
+
),
|
|
108
|
+
),
|
|
109
|
+
] = 1,
|
|
110
|
+
fofn: Annotated[
|
|
111
|
+
Path | None,
|
|
112
|
+
typer.Option("--fofn", help="File of filenames; replaces the positional FILEs."),
|
|
113
|
+
] = None,
|
|
114
|
+
quiet: Annotated[bool, typer.Option("--quiet", help="Silence stderr diagnostics.")] = False,
|
|
115
|
+
noheader: Annotated[bool, typer.Option("--noheader", help="Suppress the header row.")] = False,
|
|
116
|
+
nopath: Annotated[bool, typer.Option("--nopath", help="Basename the FILE column.")] = False,
|
|
117
|
+
debug: Annotated[bool, typer.Option("--debug", help="Verbose stderr diagnostics.")] = False,
|
|
118
|
+
output_format: Annotated[
|
|
119
|
+
OutputFormat | None,
|
|
120
|
+
typer.Option("--format", help="Output format (reads mode defaults to json)."),
|
|
121
|
+
] = None,
|
|
122
|
+
) -> None:
|
|
123
|
+
"""Screen contig files or FASTQ reads (R1 and R2 comma-lists, one lane
|
|
124
|
+
each) for AMR/virulence genes."""
|
|
125
|
+
|
|
126
|
+
def run() -> None:
|
|
127
|
+
if (r1 is not None or r2 is not None) and files:
|
|
128
|
+
usage_fail("--r1/--r2 and positional contig FILEs are mutually exclusive")
|
|
129
|
+
if r2 is not None and r1 is None:
|
|
130
|
+
usage_fail("--r2 requires --r1")
|
|
131
|
+
if (
|
|
132
|
+
(min_identity > 0 or min_mapq > 0)
|
|
133
|
+
and r1 is None
|
|
134
|
+
and aligner is not AlignerEnum.minimap2
|
|
135
|
+
):
|
|
136
|
+
usage_fail("--min-identity/--min-mapq are reads-mode only (minimap2 engine)")
|
|
137
|
+
if r1 is not None or r2 is not None:
|
|
138
|
+
if fofn is not None:
|
|
139
|
+
usage_fail("--fofn is not available in reads mode")
|
|
140
|
+
if noheader:
|
|
141
|
+
usage_fail("--noheader is not available in reads mode")
|
|
142
|
+
if nopath:
|
|
143
|
+
usage_fail("--nopath is not available in reads mode")
|
|
144
|
+
if jobs != 1:
|
|
145
|
+
usage_fail("--jobs is not available in reads mode")
|
|
146
|
+
typer.echo(
|
|
147
|
+
run_screen_reads(
|
|
148
|
+
_split_read_list(r1 or "", "--r1"),
|
|
149
|
+
_split_read_list(r2, "--r2") if r2 is not None else None,
|
|
150
|
+
db,
|
|
151
|
+
datadir,
|
|
152
|
+
read_type,
|
|
153
|
+
min_breadth,
|
|
154
|
+
min_identity,
|
|
155
|
+
min_mapq,
|
|
156
|
+
threads,
|
|
157
|
+
output_format,
|
|
158
|
+
quiet,
|
|
159
|
+
debug,
|
|
160
|
+
aligner=aligner,
|
|
161
|
+
minid=minid,
|
|
162
|
+
mincov=mincov,
|
|
163
|
+
),
|
|
164
|
+
nl=False,
|
|
165
|
+
)
|
|
166
|
+
elif aligner is AlignerEnum.minimap2:
|
|
167
|
+
typer.echo(
|
|
168
|
+
run_screen_assemblies(
|
|
169
|
+
files,
|
|
170
|
+
fofn,
|
|
171
|
+
db,
|
|
172
|
+
datadir,
|
|
173
|
+
read_type,
|
|
174
|
+
min_breadth,
|
|
175
|
+
min_identity,
|
|
176
|
+
min_mapq,
|
|
177
|
+
threads,
|
|
178
|
+
jobs,
|
|
179
|
+
noheader,
|
|
180
|
+
nopath,
|
|
181
|
+
output_format,
|
|
182
|
+
quiet,
|
|
183
|
+
debug,
|
|
184
|
+
minid=minid,
|
|
185
|
+
mincov=mincov,
|
|
186
|
+
),
|
|
187
|
+
nl=False,
|
|
188
|
+
)
|
|
189
|
+
else:
|
|
190
|
+
typer.echo(
|
|
191
|
+
run_screen(
|
|
192
|
+
files,
|
|
193
|
+
db,
|
|
194
|
+
datadir,
|
|
195
|
+
minid,
|
|
196
|
+
mincov,
|
|
197
|
+
threads,
|
|
198
|
+
jobs,
|
|
199
|
+
fofn,
|
|
200
|
+
quiet,
|
|
201
|
+
noheader,
|
|
202
|
+
nopath,
|
|
203
|
+
debug,
|
|
204
|
+
output_format or OutputFormat.tsv,
|
|
205
|
+
),
|
|
206
|
+
nl=False,
|
|
207
|
+
)
|
|
208
|
+
|
|
209
|
+
dispatch(run)
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def register_screen_command(app: typer.Typer) -> None:
|
|
213
|
+
"""Attach the screen command to the CLI app."""
|
|
214
|
+
app.command("screen")(screen_command)
|
gapit/cmd_summary.py
ADDED
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
"""The `gapit summary` command: matrix over abricate-format report tables.
|
|
2
|
+
|
|
3
|
+
Lives outside cli.py to keep that module small; cli.py registers it via
|
|
4
|
+
``register_summary_command``.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from datetime import UTC, datetime
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Annotated
|
|
10
|
+
|
|
11
|
+
import typer
|
|
12
|
+
|
|
13
|
+
from gapit.dispatch import dispatch
|
|
14
|
+
from gapit.errors import usage_fail
|
|
15
|
+
from gapit.formats.summary import format_summary_tsv, render_summary_json, render_summary_md
|
|
16
|
+
from gapit.screening import OutputFormat
|
|
17
|
+
from gapit.summary import SummaryParams, build_summary
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def summary_command(
|
|
21
|
+
files: Annotated[
|
|
22
|
+
list[Path] | None,
|
|
23
|
+
typer.Argument(help="Abricate-format report file(s) to summarize."),
|
|
24
|
+
] = None,
|
|
25
|
+
identity: Annotated[
|
|
26
|
+
bool,
|
|
27
|
+
typer.Option("--identity", help="Cells show %IDENTITY instead of %COVERAGE."),
|
|
28
|
+
] = False,
|
|
29
|
+
nopath: Annotated[
|
|
30
|
+
bool,
|
|
31
|
+
typer.Option("--nopath", help="Basename row keys (FILE values / input filenames)."),
|
|
32
|
+
] = False,
|
|
33
|
+
quiet: Annotated[bool, typer.Option("--quiet", help="Silence stderr diagnostics.")] = False,
|
|
34
|
+
output_format: Annotated[
|
|
35
|
+
OutputFormat, typer.Option("--format", help="Output format.")
|
|
36
|
+
] = OutputFormat.tsv,
|
|
37
|
+
) -> None:
|
|
38
|
+
"""Summarize report table(s) into a gene presence/absence matrix."""
|
|
39
|
+
|
|
40
|
+
def run() -> None:
|
|
41
|
+
if not files:
|
|
42
|
+
usage_fail("summary needs >= 1 report file(s)")
|
|
43
|
+
if identity and not quiet:
|
|
44
|
+
typer.echo("Using %IDENTITY for the summary table instead of %COVERAGE", err=True)
|
|
45
|
+
params = SummaryParams(identity=identity, nopath=nopath)
|
|
46
|
+
|
|
47
|
+
def warn(message: str) -> None:
|
|
48
|
+
if not quiet:
|
|
49
|
+
typer.echo(f"WARNING: {message}", err=True)
|
|
50
|
+
|
|
51
|
+
matrix = build_summary(list(files), params, warn=warn)
|
|
52
|
+
now = datetime.now(UTC)
|
|
53
|
+
if output_format is OutputFormat.json:
|
|
54
|
+
typer.echo(render_summary_json(matrix, now=now), nl=False)
|
|
55
|
+
elif output_format is OutputFormat.md:
|
|
56
|
+
typer.echo(render_summary_md(matrix, now=now), nl=False)
|
|
57
|
+
else:
|
|
58
|
+
typer.echo(format_summary_tsv(matrix, csv=output_format is OutputFormat.csv), nl=False)
|
|
59
|
+
|
|
60
|
+
dispatch(run)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def register_summary_command(app: typer.Typer) -> None:
|
|
64
|
+
"""Attach the summary command to the CLI app."""
|
|
65
|
+
app.command("summary")(summary_command)
|
gapit/config.py
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""Datadir resolution and configuration defaults."""
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
from gapit.errors import DatabaseError
|
|
7
|
+
|
|
8
|
+
ENV_DATADIR = "GAPIT_DATADIR"
|
|
9
|
+
DEFAULT_DATADIR = Path("~/.local/share/gapit/db")
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def resolve_datadir(cli_value: Path | None) -> Path:
|
|
13
|
+
"""Resolve the datadir: CLI ``--datadir`` > ``$GAPIT_DATADIR`` > ``~/.local/share/gapit/db``.
|
|
14
|
+
|
|
15
|
+
Expands ``~`` and resolves to an absolute path; raises DatabaseError when the
|
|
16
|
+
resolved directory does not exist.
|
|
17
|
+
"""
|
|
18
|
+
raw = cli_value if cli_value is not None else os.environ.get(ENV_DATADIR, DEFAULT_DATADIR)
|
|
19
|
+
datadir = Path(raw).expanduser().resolve()
|
|
20
|
+
if not datadir.is_dir():
|
|
21
|
+
raise DatabaseError(
|
|
22
|
+
f"datadir does not exist: {datadir}",
|
|
23
|
+
code="DATADIR_NOT_FOUND",
|
|
24
|
+
context={"datadir": str(datadir)},
|
|
25
|
+
)
|
|
26
|
+
return datadir
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def ensure_datadir(cli_value: Path | None) -> Path:
|
|
30
|
+
"""Resolve the datadir and mkdir it when absent (fresh-machine bootstrap).
|
|
31
|
+
|
|
32
|
+
Write paths (``db fetch``, ``db build``) bootstrap a missing root; every
|
|
33
|
+
read path still demands it via ``resolve_datadir``. The resolved path is
|
|
34
|
+
recovered from ``resolve_datadir``'s DATADIR_NOT_FOUND context — this
|
|
35
|
+
module stays the single owner of resolution.
|
|
36
|
+
"""
|
|
37
|
+
try:
|
|
38
|
+
root = resolve_datadir(cli_value)
|
|
39
|
+
except DatabaseError as exc:
|
|
40
|
+
if exc.code != "DATADIR_NOT_FOUND":
|
|
41
|
+
raise
|
|
42
|
+
root = Path(exc.context["datadir"])
|
|
43
|
+
try:
|
|
44
|
+
root.mkdir(parents=True, exist_ok=True)
|
|
45
|
+
except OSError as exc:
|
|
46
|
+
raise DatabaseError(
|
|
47
|
+
f"cannot create datadir: {root}",
|
|
48
|
+
code="DATADIR_CREATE_FAILED",
|
|
49
|
+
context={"datadir": str(root)},
|
|
50
|
+
) from exc
|
|
51
|
+
return root
|
|
Binary file
|
|
Binary file
|
gapit/db.py
ADDED
|
@@ -0,0 +1,226 @@
|
|
|
1
|
+
"""Database layer: discovery, ``~~~`` header parsing, makeblastdb/blastdbcmd wrappers."""
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
import re
|
|
5
|
+
import shlex
|
|
6
|
+
import sys
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Literal
|
|
9
|
+
|
|
10
|
+
from pydantic import BaseModel
|
|
11
|
+
|
|
12
|
+
from gapit.errors import DatabaseError, InputError
|
|
13
|
+
from gapit.fasta import iter_fasta
|
|
14
|
+
from gapit.proctools import run_tool
|
|
15
|
+
|
|
16
|
+
IDSEP = "~~~"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class DbHeader(BaseModel, frozen=True):
|
|
20
|
+
"""Parsed ``~~~`` fields of a database sequence id (SPEC.md §4 step 5).
|
|
21
|
+
|
|
22
|
+
``function`` carries functional categories (AMR classes, virulence, ...):
|
|
23
|
+
the legacy ``~~~`` branch fills it with abricate's 4th resistance field
|
|
24
|
+
verbatim — same bytes, better name (Wave F1).
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
database: str
|
|
28
|
+
gene: str
|
|
29
|
+
accession: str
|
|
30
|
+
function: str
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class BlastDbInfo(BaseModel, frozen=True):
|
|
34
|
+
"""Introspection of one built BLAST database (``blastdbcmd -info``)."""
|
|
35
|
+
|
|
36
|
+
n_sequences: int
|
|
37
|
+
dbtype: Literal["nucl", "prot"]
|
|
38
|
+
title: str
|
|
39
|
+
date: str
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class Database(BaseModel, frozen=True):
|
|
43
|
+
"""One discovered database directory under the datadir."""
|
|
44
|
+
|
|
45
|
+
name: str
|
|
46
|
+
path: Path
|
|
47
|
+
sequences_path: Path
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class DatabaseInfo(BaseModel, frozen=True):
|
|
51
|
+
"""One row of ``gapit list`` output."""
|
|
52
|
+
|
|
53
|
+
name: str
|
|
54
|
+
n_sequences: int
|
|
55
|
+
dbtype: Literal["nucl", "prot"]
|
|
56
|
+
date: str
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def parse_db_header(seqid: str, default_db: str) -> DbHeader:
|
|
60
|
+
"""Split a seqid on ``~~~`` into (database, gene, accession, function).
|
|
61
|
+
|
|
62
|
+
Without any ``~~~`` the whole seqid is the gene and the default database is
|
|
63
|
+
used; partial headers leave trailing fields empty; an empty database field
|
|
64
|
+
also falls back to ``default_db`` (SPEC.md §4 step 5).
|
|
65
|
+
"""
|
|
66
|
+
fields = seqid.split(IDSEP)
|
|
67
|
+
if len(fields) == 1:
|
|
68
|
+
return DbHeader(database=default_db, gene=seqid, accession="", function="")
|
|
69
|
+
fields += [""] * (4 - len(fields))
|
|
70
|
+
database, gene, accession, function = fields[:4]
|
|
71
|
+
return DbHeader(
|
|
72
|
+
database=database or default_db,
|
|
73
|
+
gene=gene,
|
|
74
|
+
accession=accession,
|
|
75
|
+
function=function,
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
_AGTC_TABLE = str.maketrans("", "", "AGTCagtc")
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def mol_type(letters: str) -> Literal["nucl", "prot"]:
|
|
83
|
+
"""Abricate heuristic (SPEC.md §2): protein iff non-[AGTC] letters are
|
|
84
|
+
strictly more than half of the input (empty input is nucleotide)."""
|
|
85
|
+
non_agtc = len(letters.translate(_AGTC_TABLE))
|
|
86
|
+
return "prot" if 2 * non_agtc > len(letters) else "nucl"
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def make_blast_db(
|
|
90
|
+
sequences_path: Path,
|
|
91
|
+
name: str,
|
|
92
|
+
*,
|
|
93
|
+
dbtype: Literal["nucl", "prot"] | None = None,
|
|
94
|
+
debug: bool = False,
|
|
95
|
+
) -> None:
|
|
96
|
+
"""(Re)build the BLAST index for one database directory.
|
|
97
|
+
|
|
98
|
+
``dbtype=None`` (the default) keeps the abricate ``mol_type`` heuristic;
|
|
99
|
+
an explicit ``"nucl"``/``"prot"`` — e.g. from a gapit manifest, which
|
|
100
|
+
declares the type — skips the heuristic entirely. With ``debug``, echo
|
|
101
|
+
the makeblastdb argv to stderr (abricate --debug parity).
|
|
102
|
+
"""
|
|
103
|
+
if dbtype is None:
|
|
104
|
+
letters = "".join(record.sequence for record in iter_fasta(sequences_path))
|
|
105
|
+
dbtype = mol_type(letters)
|
|
106
|
+
for index_file in sequences_path.parent.glob(f"{sequences_path.name}.[np]??"):
|
|
107
|
+
index_file.unlink()
|
|
108
|
+
argv = [
|
|
109
|
+
"makeblastdb",
|
|
110
|
+
"-in",
|
|
111
|
+
str(sequences_path),
|
|
112
|
+
"-title",
|
|
113
|
+
name,
|
|
114
|
+
"-dbtype",
|
|
115
|
+
dbtype,
|
|
116
|
+
"-logfile",
|
|
117
|
+
"/dev/null",
|
|
118
|
+
]
|
|
119
|
+
if debug:
|
|
120
|
+
print(f"gapit: run: {shlex.join(argv)}", file=sys.stderr)
|
|
121
|
+
result = run_tool(argv)
|
|
122
|
+
if result.returncode != 0:
|
|
123
|
+
raise DatabaseError(
|
|
124
|
+
f"makeblastdb failed for {name}: {result.stderr.strip()}",
|
|
125
|
+
code="MAKEBLASTDB_FAILED",
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
_SEQUENCES_RE = re.compile(r"([\d,]+)\s+sequences;")
|
|
130
|
+
_TITLE_RE = re.compile(r"^Database:[ \t]*(.*)$", re.MULTILINE)
|
|
131
|
+
_DATE_RE = re.compile(r"Date:\s*([A-Za-z]{3})\s+(\d{1,2}),\s+(\d{4})")
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def parse_blastdbcmd_info(text: str) -> BlastDbInfo:
|
|
135
|
+
"""Parse ``blastdbcmd -info`` output (SPEC.md §2)."""
|
|
136
|
+
sequences_match = _SEQUENCES_RE.search(text)
|
|
137
|
+
title_match = _TITLE_RE.search(text)
|
|
138
|
+
date_match = _DATE_RE.search(text)
|
|
139
|
+
if sequences_match is None or title_match is None or date_match is None:
|
|
140
|
+
raise DatabaseError(
|
|
141
|
+
f"could not parse blastdbcmd output: {text!r}",
|
|
142
|
+
code="BLASTDBCMD_PARSE_FAILED",
|
|
143
|
+
)
|
|
144
|
+
month, day, year = date_match.group(1), date_match.group(2), date_match.group(3)
|
|
145
|
+
return BlastDbInfo(
|
|
146
|
+
n_sequences=int(sequences_match.group(1).replace(",", "")),
|
|
147
|
+
dbtype="prot" if "total residues" in text else "nucl",
|
|
148
|
+
title=title_match.group(1).strip(),
|
|
149
|
+
date=f"{year}-{month}-{int(day):02d}",
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def blast_db_info(db_prefix: Path) -> BlastDbInfo:
|
|
154
|
+
"""Introspect a built BLAST database via ``blastdbcmd -info``."""
|
|
155
|
+
result = run_tool(["blastdbcmd", "-info", "-db", str(db_prefix)])
|
|
156
|
+
if result.returncode != 0:
|
|
157
|
+
raise DatabaseError(
|
|
158
|
+
f"Database {db_prefix} is not indexed, please try: gapit setupdb",
|
|
159
|
+
code="DATABASE_NOT_INDEXED",
|
|
160
|
+
context={"db": str(db_prefix)},
|
|
161
|
+
)
|
|
162
|
+
return parse_blastdbcmd_info(result.stdout)
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def discover_databases(datadir: Path) -> list[Database]:
|
|
166
|
+
"""Immediate subdirectories of the datadir holding a readable ``sequences``
|
|
167
|
+
file, sorted by name."""
|
|
168
|
+
databases: list[Database] = []
|
|
169
|
+
for child in datadir.iterdir():
|
|
170
|
+
sequences = child / "sequences"
|
|
171
|
+
if child.is_dir() and sequences.is_file() and os.access(sequences, os.R_OK):
|
|
172
|
+
databases.append(Database(name=child.name, path=child, sequences_path=sequences))
|
|
173
|
+
return sorted(databases, key=lambda database: database.name)
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _manifest_dbtype(db_dir: Path) -> Literal["nucl", "prot"] | None:
|
|
177
|
+
"""The dbtype declared by a database's gapit manifest, or ``None`` when the
|
|
178
|
+
directory has none (abricate-built) — the caller then falls back to the
|
|
179
|
+
``mol_type`` heuristic. A malformed manifest propagates (no silent
|
|
180
|
+
degradation).
|
|
181
|
+
|
|
182
|
+
Lazy import: records -> formats.json -> reads -> db is a cycle at module
|
|
183
|
+
level (Wave C-a notepad hazard).
|
|
184
|
+
"""
|
|
185
|
+
from gapit.records import read_manifest
|
|
186
|
+
|
|
187
|
+
try:
|
|
188
|
+
return read_manifest(db_dir / "gapit-manifest.json").dbtype
|
|
189
|
+
except InputError as exc:
|
|
190
|
+
if exc.code != "INPUT_NOT_FOUND":
|
|
191
|
+
raise
|
|
192
|
+
return None
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def list_databases(datadir: Path, *, setupdb: bool, debug: bool = False) -> list[DatabaseInfo]:
|
|
196
|
+
"""Enumerate databases (building indices first when ``setupdb``), requiring
|
|
197
|
+
every database to be indexed; returns one DatabaseInfo per database."""
|
|
198
|
+
infos: list[DatabaseInfo] = []
|
|
199
|
+
for database in discover_databases(datadir):
|
|
200
|
+
if setupdb:
|
|
201
|
+
make_blast_db(
|
|
202
|
+
database.sequences_path,
|
|
203
|
+
database.name,
|
|
204
|
+
dbtype=_manifest_dbtype(database.path),
|
|
205
|
+
debug=debug,
|
|
206
|
+
)
|
|
207
|
+
sequences = database.sequences_path
|
|
208
|
+
index_exists = any(
|
|
209
|
+
(sequences.parent / f"{sequences.name}{suffix}").exists() for suffix in (".nin", ".pin")
|
|
210
|
+
)
|
|
211
|
+
if not index_exists:
|
|
212
|
+
raise DatabaseError(
|
|
213
|
+
f"Database {database.name} is not indexed, please try: gapit setupdb",
|
|
214
|
+
code="DATABASE_NOT_INDEXED",
|
|
215
|
+
context={"db": database.name},
|
|
216
|
+
)
|
|
217
|
+
info = blast_db_info(sequences)
|
|
218
|
+
infos.append(
|
|
219
|
+
DatabaseInfo(
|
|
220
|
+
name=database.name,
|
|
221
|
+
n_sequences=info.n_sequences,
|
|
222
|
+
dbtype=info.dbtype,
|
|
223
|
+
date=info.date,
|
|
224
|
+
)
|
|
225
|
+
)
|
|
226
|
+
return infos
|