gapit 0.2.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gapit/__init__.py +3 -0
- gapit/blast.py +239 -0
- gapit/cli.py +128 -0
- gapit/cmd_db.py +113 -0
- gapit/cmd_db_build.py +72 -0
- gapit/cmd_db_install.py +126 -0
- gapit/cmd_db_outdated.py +51 -0
- gapit/cmd_db_search.py +66 -0
- gapit/cmd_screen.py +214 -0
- gapit/cmd_summary.py +65 -0
- gapit/config.py +51 -0
- gapit/data/snapshots/card.tar.gz +0 -0
- gapit/data/snapshots/vfdb.tar.gz +0 -0
- gapit/db.py +226 -0
- gapit/db_build_ops.py +210 -0
- gapit/db_ops.py +128 -0
- gapit/db_query_ops.py +252 -0
- gapit/dbbuild.py +207 -0
- gapit/dbcodec.py +117 -0
- gapit/dispatch.py +31 -0
- gapit/errors.py +87 -0
- gapit/fasta.py +123 -0
- gapit/formats/__init__.py +1 -0
- gapit/formats/json.py +309 -0
- gapit/formats/md.py +190 -0
- gapit/formats/schemas.py +30 -0
- gapit/formats/summary.py +103 -0
- gapit/formats/tsv.py +45 -0
- gapit/hits.py +107 -0
- gapit/mcp.py +158 -0
- gapit/mcp_schemas.py +123 -0
- gapit/mcp_tools.py +289 -0
- gapit/minimap.py +20 -0
- gapit/minimap2_run.py +114 -0
- gapit/paf.py +115 -0
- gapit/proctools.py +24 -0
- gapit/providers/__init__.py +39 -0
- gapit/providers/argannot.py +94 -0
- gapit/providers/bacmet2.py +59 -0
- gapit/providers/card.py +150 -0
- gapit/providers/common.py +245 -0
- gapit/providers/ecoh.py +63 -0
- gapit/providers/ecoli_vf.py +74 -0
- gapit/providers/megares.py +71 -0
- gapit/providers/ncbi.py +103 -0
- gapit/providers/plasmidfinder.py +69 -0
- gapit/providers/resfinder.py +123 -0
- gapit/providers/snapshots.py +119 -0
- gapit/providers/upec_expec_vf.py +85 -0
- gapit/providers/vfdb.py +92 -0
- gapit/providers/victors.py +109 -0
- gapit/py.typed +0 -0
- gapit/reads.py +221 -0
- gapit/records.py +152 -0
- gapit/report.py +25 -0
- gapit/screening.py +145 -0
- gapit/screening_reads.py +255 -0
- gapit/seqconvert.py +203 -0
- gapit/summary.py +151 -0
- gapit-0.2.2.dist-info/METADATA +183 -0
- gapit-0.2.2.dist-info/RECORD +64 -0
- gapit-0.2.2.dist-info/WHEEL +4 -0
- gapit-0.2.2.dist-info/entry_points.txt +3 -0
- gapit-0.2.2.dist-info/licenses/LICENSE +21 -0
gapit/screening_reads.py
ADDED
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
"""The minimap2 screening use-cases (SPEC.md §10): ``--r1``/``--r2`` reads and
|
|
2
|
+
positional assemblies via ``--aligner minimap2``.
|
|
3
|
+
|
|
4
|
+
Split from screening.py so each use-case module stays under the 250 pure-LOC
|
|
5
|
+
ceiling; screening.py keeps the blastn contig pipeline and shared helpers.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from datetime import UTC, datetime
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
import typer
|
|
12
|
+
|
|
13
|
+
from gapit import config
|
|
14
|
+
from gapit.errors import ensure_input_file, usage_fail
|
|
15
|
+
from gapit.formats.json import render_reads2_json, render_reads_json
|
|
16
|
+
from gapit.formats.md import render_reads2_markdown, render_reads_markdown
|
|
17
|
+
from gapit.reads import (
|
|
18
|
+
ReadFileKind,
|
|
19
|
+
ReadsParams,
|
|
20
|
+
ReadTypeEnum,
|
|
21
|
+
detect_read_kind,
|
|
22
|
+
screen_reads,
|
|
23
|
+
)
|
|
24
|
+
from gapit.screening import AlignerEnum, OutputFormat, find_database
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _pair_read_lanes(r1: list[Path], r2: list[Path] | None) -> list[tuple[Path, Path | None]]:
|
|
28
|
+
"""Pair per-lane --r1/--r2 file lists (r2 None = single-end)."""
|
|
29
|
+
if r2 is not None and len(r2) != len(r1):
|
|
30
|
+
usage_fail(f"--r2 has {len(r2)} files but --r1 has {len(r1)} (lanes must pair up)")
|
|
31
|
+
if r2 is None:
|
|
32
|
+
return [(path, None) for path in r1]
|
|
33
|
+
return list(zip(r1, r2, strict=True))
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _resolve_read_preset(
|
|
37
|
+
lanes: list[tuple[Path, Path | None]], read_type: ReadTypeEnum | None, quiet: bool
|
|
38
|
+
) -> ReadTypeEnum:
|
|
39
|
+
"""Detect every input's kind from content and resolve the minimap2 preset:
|
|
40
|
+
assembly FASTA forces map-ont (one stderr note unless quiet/explicit),
|
|
41
|
+
FASTQ keeps sr unless a preset was given. Mixed kinds and FASTA paired-end
|
|
42
|
+
are usage errors."""
|
|
43
|
+
r1_kinds = {detect_read_kind(r1_path) for r1_path, _ in lanes}
|
|
44
|
+
if len(r1_kinds) > 1:
|
|
45
|
+
usage_fail("mixed FASTA and FASTQ inputs")
|
|
46
|
+
r2_kinds = {detect_read_kind(r2_path) for _, r2_path in lanes if r2_path is not None}
|
|
47
|
+
if r2_kinds and ReadFileKind.fasta in r1_kinds | r2_kinds:
|
|
48
|
+
usage_fail("paired-end requires FASTQ")
|
|
49
|
+
if ReadFileKind.fasta not in r1_kinds:
|
|
50
|
+
return read_type if read_type is not None else ReadTypeEnum.sr
|
|
51
|
+
if read_type is not None and read_type is not ReadTypeEnum.map_ont:
|
|
52
|
+
usage_fail("assembly FASTA requires map-ont")
|
|
53
|
+
if read_type is None and not quiet:
|
|
54
|
+
typer.echo("assembly FASTA detected; using map-ont", err=True)
|
|
55
|
+
return ReadTypeEnum.map_ont
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _reject_blastn_thresholds(minid: float, mincov: float, mode: str) -> None:
|
|
59
|
+
"""--minid/--mincov are blastn-only; every minimap2 entry point rejects
|
|
60
|
+
them instead of silently ignoring them (reads thresholds have their own
|
|
61
|
+
flags). ``mode`` names the invocation in the frozen message."""
|
|
62
|
+
if minid != 80.0 or mincov != 80.0:
|
|
63
|
+
usage_fail(
|
|
64
|
+
f"--minid/--mincov apply to blastn only; use --min-identity/--min-breadth with {mode}"
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _validate_reads_usage(
|
|
69
|
+
output_format: OutputFormat | None,
|
|
70
|
+
min_breadth: float,
|
|
71
|
+
min_identity: float,
|
|
72
|
+
min_mapq: int,
|
|
73
|
+
threads: int,
|
|
74
|
+
) -> None:
|
|
75
|
+
"""Usage gates shared by both minimap2 entry points, in the frozen order
|
|
76
|
+
(format first, then thresholds), before any file is touched."""
|
|
77
|
+
if output_format is OutputFormat.tsv or output_format is OutputFormat.csv:
|
|
78
|
+
usage_fail("--format tsv|csv is not available in reads mode (use json or md)")
|
|
79
|
+
if not 0.0 <= min_breadth <= 100.0:
|
|
80
|
+
usage_fail(f"--min-breadth must be in [0, 100]: got {min_breadth}")
|
|
81
|
+
if not 0.0 <= min_identity <= 100.0:
|
|
82
|
+
usage_fail(f"--min-identity must be in [0, 100]: got {min_identity}")
|
|
83
|
+
if min_mapq < 0:
|
|
84
|
+
usage_fail(f"--min-mapq must be >= 0: got {min_mapq}")
|
|
85
|
+
if threads < 1:
|
|
86
|
+
usage_fail(f"--threads must be >= 1: got {threads}")
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _screen_lanes(
|
|
90
|
+
lanes: list[tuple[Path, Path | None]],
|
|
91
|
+
db_name: str,
|
|
92
|
+
datadir: Path | None,
|
|
93
|
+
read_type: ReadTypeEnum | None,
|
|
94
|
+
min_breadth: float,
|
|
95
|
+
min_identity: float,
|
|
96
|
+
min_mapq: int,
|
|
97
|
+
threads: int,
|
|
98
|
+
output_format: OutputFormat | None,
|
|
99
|
+
quiet: bool,
|
|
100
|
+
debug: bool,
|
|
101
|
+
) -> str:
|
|
102
|
+
"""Minimap2 engine core shared by both entry points: preset resolution,
|
|
103
|
+
screening, rendering; json is the default format (SPEC.md §10). Either
|
|
104
|
+
reads/2 threshold on selects the gapit.reads/2 document; both off keep
|
|
105
|
+
gapit.reads/1 byte-identical. Returns the rendered output for the caller
|
|
106
|
+
to echo."""
|
|
107
|
+
resolved = _resolve_read_preset(lanes, read_type, quiet)
|
|
108
|
+
database = find_database(config.resolve_datadir(datadir), db_name)
|
|
109
|
+
read_files = [r1_path for r1_path, _ in lanes] + [
|
|
110
|
+
r2_path for _, r2_path in lanes if r2_path is not None
|
|
111
|
+
]
|
|
112
|
+
read_list = ", ".join(str(path) for path in read_files)
|
|
113
|
+
if not quiet:
|
|
114
|
+
typer.echo(f"Screening reads: {read_list}", err=True)
|
|
115
|
+
report = screen_reads(
|
|
116
|
+
lanes,
|
|
117
|
+
database,
|
|
118
|
+
read_type=resolved.value,
|
|
119
|
+
min_breadth=min_breadth,
|
|
120
|
+
threads=threads,
|
|
121
|
+
debug=debug,
|
|
122
|
+
min_identity=min_identity,
|
|
123
|
+
min_mapq=min_mapq,
|
|
124
|
+
)
|
|
125
|
+
present = sum(1 for gene in report.genes if gene.present)
|
|
126
|
+
if not quiet:
|
|
127
|
+
typer.echo(f"Detected {present} present genes in {read_list}", err=True)
|
|
128
|
+
params = ReadsParams(
|
|
129
|
+
db=db_name,
|
|
130
|
+
read_type=resolved.value,
|
|
131
|
+
min_breadth=min_breadth,
|
|
132
|
+
threads=threads,
|
|
133
|
+
min_identity=min_identity,
|
|
134
|
+
min_mapq=min_mapq,
|
|
135
|
+
)
|
|
136
|
+
now = datetime.now(UTC)
|
|
137
|
+
reads2 = min_identity > 0.0 or min_mapq > 0
|
|
138
|
+
if reads2:
|
|
139
|
+
output = (
|
|
140
|
+
render_reads2_markdown([report], params, now=now)
|
|
141
|
+
if output_format is OutputFormat.md
|
|
142
|
+
else render_reads2_json([report], params, now=now)
|
|
143
|
+
)
|
|
144
|
+
else:
|
|
145
|
+
output = (
|
|
146
|
+
render_reads_markdown([report], params, now=now)
|
|
147
|
+
if output_format is OutputFormat.md
|
|
148
|
+
else render_reads_json([report], params, now=now)
|
|
149
|
+
)
|
|
150
|
+
return output
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def run_screen_reads(
|
|
154
|
+
r1: list[Path],
|
|
155
|
+
r2: list[Path] | None,
|
|
156
|
+
db_name: str,
|
|
157
|
+
datadir: Path | None,
|
|
158
|
+
read_type: ReadTypeEnum | None,
|
|
159
|
+
min_breadth: float,
|
|
160
|
+
min_identity: float,
|
|
161
|
+
min_mapq: int,
|
|
162
|
+
threads: int,
|
|
163
|
+
output_format: OutputFormat | None,
|
|
164
|
+
quiet: bool,
|
|
165
|
+
debug: bool = False,
|
|
166
|
+
aligner: AlignerEnum | None = None,
|
|
167
|
+
minid: float = 80.0,
|
|
168
|
+
mincov: float = 80.0,
|
|
169
|
+
) -> str:
|
|
170
|
+
"""Screen FASTQ reads or assembly FASTA given as already-split per-lane
|
|
171
|
+
--r1/--r2 file lists (the CLI owns the comma-splitting; MCP passes arrays
|
|
172
|
+
natively, so commas in filenames survive). Per-lane minimap2, sample-level
|
|
173
|
+
union; json is the default format (SPEC.md §10). A nonzero
|
|
174
|
+
--min-identity/--min-mapq turns on gapit.reads/2 alignment filtering.
|
|
175
|
+
Returns the rendered output."""
|
|
176
|
+
if aligner is AlignerEnum.blastn:
|
|
177
|
+
usage_fail("--aligner blastn is not available for --r1/--r2 reads input")
|
|
178
|
+
_validate_reads_usage(output_format, min_breadth, min_identity, min_mapq, threads)
|
|
179
|
+
_reject_blastn_thresholds(minid, mincov, "--r1/--r2")
|
|
180
|
+
lanes = _pair_read_lanes(r1, r2)
|
|
181
|
+
for path in [r1_path for r1_path, _ in lanes] + [
|
|
182
|
+
r2_path for _, r2_path in lanes if r2_path is not None
|
|
183
|
+
]:
|
|
184
|
+
ensure_input_file(path, "reads file")
|
|
185
|
+
return _screen_lanes(
|
|
186
|
+
lanes,
|
|
187
|
+
db_name,
|
|
188
|
+
datadir,
|
|
189
|
+
read_type,
|
|
190
|
+
min_breadth,
|
|
191
|
+
min_identity,
|
|
192
|
+
min_mapq,
|
|
193
|
+
threads,
|
|
194
|
+
output_format,
|
|
195
|
+
quiet,
|
|
196
|
+
debug,
|
|
197
|
+
)
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def run_screen_assemblies(
|
|
201
|
+
files: list[Path] | None,
|
|
202
|
+
fofn: Path | None,
|
|
203
|
+
db_name: str,
|
|
204
|
+
datadir: Path | None,
|
|
205
|
+
read_type: ReadTypeEnum | None,
|
|
206
|
+
min_breadth: float,
|
|
207
|
+
min_identity: float,
|
|
208
|
+
min_mapq: int,
|
|
209
|
+
threads: int,
|
|
210
|
+
jobs: int,
|
|
211
|
+
noheader: bool,
|
|
212
|
+
nopath: bool,
|
|
213
|
+
output_format: OutputFormat | None,
|
|
214
|
+
quiet: bool,
|
|
215
|
+
debug: bool = False,
|
|
216
|
+
minid: float = 80.0,
|
|
217
|
+
mincov: float = 80.0,
|
|
218
|
+
) -> str:
|
|
219
|
+
"""Screen positional assembly FASTA file(s) with the minimap2 engine
|
|
220
|
+
(--aligner minimap2): every input must be FASTA(.gz) content — FASTQ
|
|
221
|
+
content is a usage error, undetectable content keeps the typed input
|
|
222
|
+
error. Preset resolution and output follow the reads contract (SPEC §10);
|
|
223
|
+
the blastn-engine-only flags --fofn/--jobs/--noheader/--nopath and the
|
|
224
|
+
blastn thresholds --minid/--mincov are rejected here instead of silently
|
|
225
|
+
ignored. A nonzero --min-identity/--min-mapq turns on gapit.reads/2
|
|
226
|
+
filtering. Returns the rendered output."""
|
|
227
|
+
_validate_reads_usage(output_format, min_breadth, min_identity, min_mapq, threads)
|
|
228
|
+
if fofn is not None:
|
|
229
|
+
usage_fail("--fofn is not available with --aligner minimap2")
|
|
230
|
+
if jobs != 1:
|
|
231
|
+
usage_fail("--jobs is not available with --aligner minimap2")
|
|
232
|
+
if noheader:
|
|
233
|
+
usage_fail("--noheader is not available with --aligner minimap2")
|
|
234
|
+
if nopath:
|
|
235
|
+
usage_fail("--nopath is not available with --aligner minimap2")
|
|
236
|
+
_reject_blastn_thresholds(minid, mincov, "--aligner minimap2")
|
|
237
|
+
if not files:
|
|
238
|
+
usage_fail("no input files given (positional FILEs)")
|
|
239
|
+
for path in files:
|
|
240
|
+
ensure_input_file(path)
|
|
241
|
+
if detect_read_kind(path) is not ReadFileKind.fasta:
|
|
242
|
+
usage_fail("minimap2 engine requires FASTA assemblies")
|
|
243
|
+
return _screen_lanes(
|
|
244
|
+
[(path, None) for path in files],
|
|
245
|
+
db_name,
|
|
246
|
+
datadir,
|
|
247
|
+
read_type,
|
|
248
|
+
min_breadth,
|
|
249
|
+
min_identity,
|
|
250
|
+
min_mapq,
|
|
251
|
+
threads,
|
|
252
|
+
output_format,
|
|
253
|
+
quiet,
|
|
254
|
+
debug,
|
|
255
|
+
)
|
gapit/seqconvert.py
ADDED
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
"""Native input normalization: FASTA/FASTQ/GenBank/EMBL (.gz/.bz2) → FASTA lines.
|
|
2
|
+
|
|
3
|
+
Replaces the external ``any2fasta -q -u <file>`` stage of the screening
|
|
4
|
+
pipeline (blast.py). Semantics extracted from any2fasta 0.8.1 (Perl,
|
|
5
|
+
bioconda; ``.pixi/envs/default/bin/any2fasta``) invoked with ``-q -u`` only —
|
|
6
|
+
no ``-n`` (N-purification), ``-l``, ``-g`` (VERSION), or ``-s`` (description
|
|
7
|
+
stripping). Perl line references below are that file.
|
|
8
|
+
|
|
9
|
+
- Detection (L89-99, L136-153): first line of the decompressed stream, in
|
|
10
|
+
order GENBANK ``^LOCUS\\h``, EMBL ``^ID\\h``, FASTA ``^>\\S``, FASTQ
|
|
11
|
+
``^@\\S`` (``\\h`` = space/tab; ``\\S`` = non-whitespace after the marker).
|
|
12
|
+
Empty input dies with "The input appears to be empty" (L131-133); an
|
|
13
|
+
unrecognized first line dies with "Unfamilar format with first line: ..."
|
|
14
|
+
(L153 — typo theirs, kept for message parity with the retired binary).
|
|
15
|
+
- purify_dna (L164-170): with ``-u`` only uc() — sequences uppercased, nothing
|
|
16
|
+
else touched. purify_id (L174-181) is a no-op without ``-s``: headers pass
|
|
17
|
+
through verbatim.
|
|
18
|
+
- FASTA (L185-200): header lines printed verbatim; sequence lines uppercased;
|
|
19
|
+
blank/whitespace-only lines skipped entirely (L189).
|
|
20
|
+
- FASTQ (L204-215): strict 4-line stride. Header = the ``@`` line minus its
|
|
21
|
+
leading ``@``, verbatim (id + description); sequence = uc() of line 2;
|
|
22
|
+
``+``/quality lines dropped. A trailing lone ``@`` header emits nothing
|
|
23
|
+
(loop guard ``$i < $#lines``, L208).
|
|
24
|
+
- GenBank (L255-296): id = first token after ``LOCUS\\s+`` (L285-287).
|
|
25
|
+
DEFINITION is never read — records carry no description. VERSION overrides
|
|
26
|
+
the id only under ``-g`` (L290-292; unused). ORIGIN data lines: drop the
|
|
27
|
+
first 10 columns (coordinate prefix, L279), then strip whitespace (L280) —
|
|
28
|
+
digits past column 10 are kept. Records flush at ``//`` (L264-269); a
|
|
29
|
+
record never terminated by ``//`` is dropped.
|
|
30
|
+
- EMBL (L300-339): id = text after ``ID\\s+`` up to the first ``;`` (L330-331,
|
|
31
|
+
not trimmed). DE is never read. SQ data lines: strip all whitespace AND
|
|
32
|
+
digits (L325). Flush at ``//``.
|
|
33
|
+
|
|
34
|
+
Deliberate divergences, invisible to BLAST: sequence is re-wrapped at 60
|
|
35
|
+
columns (the Perl reuses input wrapping) and input is read with universal
|
|
36
|
+
newlines (the Perl preserves ``\\r``). Parsed records — id, description,
|
|
37
|
+
uppercased sequence — are identical, so the blastn query is unchanged and
|
|
38
|
+
abricate parity is unaffected. gapit owns no any2fasta dependency; the
|
|
39
|
+
differential suite runs the real binary from the parity env's PATH, where
|
|
40
|
+
abricate provides it transitively (tests/test_seqconvert_differential.py).
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
import re
|
|
44
|
+
from collections.abc import Iterator
|
|
45
|
+
from enum import Enum
|
|
46
|
+
from pathlib import Path
|
|
47
|
+
from typing import IO
|
|
48
|
+
|
|
49
|
+
from gapit.errors import InputError
|
|
50
|
+
from gapit.fasta import open_text
|
|
51
|
+
|
|
52
|
+
_WRAP = 60
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class SeqFormat(Enum):
|
|
56
|
+
"""Input formats detected from the first decompressed line."""
|
|
57
|
+
|
|
58
|
+
fasta = "fasta"
|
|
59
|
+
fastq = "fastq"
|
|
60
|
+
genbank = "genbank"
|
|
61
|
+
embl = "embl"
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _tagged(line: str, tag: str) -> bool:
|
|
65
|
+
"""``^tag\\h`` — tag at column 1 followed by one space or tab."""
|
|
66
|
+
return line.startswith(tag) and line[len(tag) : len(tag) + 1] in (" ", "\t")
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def detect_format(path: Path) -> SeqFormat:
|
|
70
|
+
"""Content-sniff the first decompressed line (reads.py detect_read_kind style)."""
|
|
71
|
+
try:
|
|
72
|
+
with open_text(path) as handle:
|
|
73
|
+
first = handle.readline()
|
|
74
|
+
except (OSError, UnicodeDecodeError, EOFError) as exc:
|
|
75
|
+
raise InputError(
|
|
76
|
+
f"could not read input: {exc}",
|
|
77
|
+
code="INVALID_INPUT",
|
|
78
|
+
context={"file": str(path)},
|
|
79
|
+
) from exc
|
|
80
|
+
if not first:
|
|
81
|
+
raise InputError(
|
|
82
|
+
"The input appears to be empty",
|
|
83
|
+
code="INVALID_INPUT",
|
|
84
|
+
context={"file": str(path)},
|
|
85
|
+
)
|
|
86
|
+
line = first.rstrip("\r\n")
|
|
87
|
+
if _tagged(line, "LOCUS"):
|
|
88
|
+
return SeqFormat.genbank
|
|
89
|
+
if _tagged(line, "ID"):
|
|
90
|
+
return SeqFormat.embl
|
|
91
|
+
if len(line) > 1 and line[0] == ">" and not line[1].isspace():
|
|
92
|
+
return SeqFormat.fasta
|
|
93
|
+
if len(line) > 1 and line[0] == "@" and not line[1].isspace():
|
|
94
|
+
return SeqFormat.fastq
|
|
95
|
+
raise InputError(
|
|
96
|
+
f"Unfamilar format with first line: {line}",
|
|
97
|
+
code="INVALID_INPUT",
|
|
98
|
+
context={"file": str(path)},
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _iter_fasta(handle: IO[str]) -> Iterator[tuple[str, str]]:
|
|
103
|
+
"""(header, sequence) pairs; headers verbatim, blank lines skipped."""
|
|
104
|
+
header: str | None = None
|
|
105
|
+
chunks: list[str] = []
|
|
106
|
+
for raw in handle:
|
|
107
|
+
line = raw.rstrip("\n")
|
|
108
|
+
if not line.strip():
|
|
109
|
+
continue
|
|
110
|
+
if line.startswith(">"):
|
|
111
|
+
if header is not None:
|
|
112
|
+
yield (header, "".join(chunks).upper())
|
|
113
|
+
header, chunks = line[1:], []
|
|
114
|
+
else:
|
|
115
|
+
chunks.append(line)
|
|
116
|
+
if header is not None:
|
|
117
|
+
yield (header, "".join(chunks).upper())
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _iter_fastq(handle: IO[str]) -> Iterator[tuple[str, str]]:
|
|
121
|
+
"""Stride-4 records; a trailing lone '@' header emits nothing (perl guard)."""
|
|
122
|
+
lines = iter(handle)
|
|
123
|
+
for header_line in lines:
|
|
124
|
+
sequence_line = next(lines, None)
|
|
125
|
+
if sequence_line is None:
|
|
126
|
+
return
|
|
127
|
+
yield (header_line.rstrip("\n")[1:], sequence_line.rstrip("\n").upper())
|
|
128
|
+
next(lines, None) # '+' separator
|
|
129
|
+
next(lines, None) # quality line
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def _iter_genbank(handle: IO[str]) -> Iterator[tuple[str, str]]:
|
|
133
|
+
"""LOCUS-name ids without description; ORIGIN columns 11+ minus whitespace."""
|
|
134
|
+
acc = ""
|
|
135
|
+
chunks: list[str] = []
|
|
136
|
+
in_seq = False
|
|
137
|
+
for raw in handle:
|
|
138
|
+
line = raw.rstrip("\n")
|
|
139
|
+
if line.startswith("//"):
|
|
140
|
+
yield (acc, "".join(chunks).upper())
|
|
141
|
+
acc, chunks, in_seq = "", [], False
|
|
142
|
+
elif line.startswith("ORIGIN"):
|
|
143
|
+
in_seq = True
|
|
144
|
+
elif in_seq:
|
|
145
|
+
chunks.append(re.sub(r"\s+", "", line[10:]))
|
|
146
|
+
elif (match := re.match(r"LOCUS\s+(\S+)", line)) is not None:
|
|
147
|
+
acc = match.group(1)
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _iter_embl(handle: IO[str]) -> Iterator[tuple[str, str]]:
|
|
151
|
+
"""ID-before-';' ids without description; SQ lines minus whitespace+digits."""
|
|
152
|
+
acc = ""
|
|
153
|
+
chunks: list[str] = []
|
|
154
|
+
in_seq = False
|
|
155
|
+
for raw in handle:
|
|
156
|
+
line = raw.rstrip("\n")
|
|
157
|
+
if line.startswith("//"):
|
|
158
|
+
yield (acc, "".join(chunks).upper())
|
|
159
|
+
acc, chunks, in_seq = "", [], False
|
|
160
|
+
elif re.match(r"SQ\s", line) is not None:
|
|
161
|
+
in_seq = True
|
|
162
|
+
elif in_seq:
|
|
163
|
+
chunks.append(re.sub(r"[\s\d]", "", line))
|
|
164
|
+
elif (match := re.match(r"ID\s+([^;]+)", line)) is not None:
|
|
165
|
+
acc = match.group(1)
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _records(fmt: SeqFormat, handle: IO[str]) -> Iterator[tuple[str, str]]:
|
|
169
|
+
records: Iterator[tuple[str, str]]
|
|
170
|
+
match fmt:
|
|
171
|
+
case SeqFormat.fasta:
|
|
172
|
+
records = _iter_fasta(handle)
|
|
173
|
+
case SeqFormat.fastq:
|
|
174
|
+
records = _iter_fastq(handle)
|
|
175
|
+
case SeqFormat.genbank:
|
|
176
|
+
records = _iter_genbank(handle)
|
|
177
|
+
case SeqFormat.embl:
|
|
178
|
+
records = _iter_embl(handle)
|
|
179
|
+
return records
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def to_fasta_lines(path: Path, fmt: SeqFormat | None = None) -> Iterator[str]:
|
|
183
|
+
"""Yield complete FASTA lines for one input file: one verbatim header line
|
|
184
|
+
per record, then its uppercased sequence wrapped at 60 columns.
|
|
185
|
+
|
|
186
|
+
``fmt`` may be pre-supplied to reuse a sniff (blast.run_blastn's debug
|
|
187
|
+
echo shares one detection). Raises InputError (INVALID_INPUT) on unreadable,
|
|
188
|
+
empty, or unrecognized input — the retired any2fasta's fatal path.
|
|
189
|
+
"""
|
|
190
|
+
if fmt is None:
|
|
191
|
+
fmt = detect_format(path)
|
|
192
|
+
try:
|
|
193
|
+
with open_text(path) as handle:
|
|
194
|
+
for header, sequence in _records(fmt, handle):
|
|
195
|
+
yield f">{header}\n"
|
|
196
|
+
for offset in range(0, len(sequence), _WRAP):
|
|
197
|
+
yield f"{sequence[offset : offset + _WRAP]}\n"
|
|
198
|
+
except (OSError, UnicodeDecodeError, EOFError) as exc:
|
|
199
|
+
raise InputError(
|
|
200
|
+
f"could not read input: {exc}",
|
|
201
|
+
code="INVALID_INPUT",
|
|
202
|
+
context={"file": str(path)},
|
|
203
|
+
) from exc
|
gapit/summary.py
ADDED
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
"""Summary mode core: parse abricate-format report tables into a gene matrix.
|
|
2
|
+
|
|
3
|
+
Mirrors abricate 1.4.0 ``summary_table`` (SPEC.md §6) exactly where it is
|
|
4
|
+
defined, and replaces its silent-undef edges with typed InputErrors
|
|
5
|
+
(``SUMMARY_MALFORMED``) — a documented [gapit-extension] divergence.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from collections.abc import Callable
|
|
9
|
+
from pathlib import Path, PurePath
|
|
10
|
+
from typing import Literal
|
|
11
|
+
|
|
12
|
+
from pydantic import BaseModel
|
|
13
|
+
|
|
14
|
+
from gapit.errors import InputError
|
|
15
|
+
|
|
16
|
+
FIELDSEP = ";"
|
|
17
|
+
ABSENT = "."
|
|
18
|
+
|
|
19
|
+
Warn = Callable[[str], None]
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class SummaryParams(BaseModel, frozen=True):
|
|
23
|
+
"""Summary parameters in effect (metric + path display)."""
|
|
24
|
+
|
|
25
|
+
identity: bool = False
|
|
26
|
+
nopath: bool = False
|
|
27
|
+
|
|
28
|
+
@property
|
|
29
|
+
def metric(self) -> Literal["%COVERAGE", "%IDENTITY"]:
|
|
30
|
+
"""The report column summarized into cells."""
|
|
31
|
+
return "%IDENTITY" if self.identity else "%COVERAGE"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class SummaryRow(BaseModel, frozen=True):
|
|
35
|
+
"""One matrix row: a display label, its distinct-gene count, and the
|
|
36
|
+
per-gene cell values as the original report strings in row order."""
|
|
37
|
+
|
|
38
|
+
file: str
|
|
39
|
+
num_found: int
|
|
40
|
+
cells: dict[str, tuple[str, ...]]
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class SummaryMatrix(BaseModel, frozen=True):
|
|
44
|
+
"""Canonical in-memory summary result: sorted gene universe + ordered rows."""
|
|
45
|
+
|
|
46
|
+
params: SummaryParams
|
|
47
|
+
genes: tuple[str, ...]
|
|
48
|
+
rows: tuple[SummaryRow, ...]
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _read_report(path: Path) -> str:
|
|
52
|
+
"""Read one report table; missing/unreadable/non-UTF-8 is an InputError."""
|
|
53
|
+
if not path.is_file():
|
|
54
|
+
raise InputError(
|
|
55
|
+
f"report file not found or unreadable: {path}",
|
|
56
|
+
code="INPUT_NOT_FOUND",
|
|
57
|
+
context={"file": str(path)},
|
|
58
|
+
)
|
|
59
|
+
try:
|
|
60
|
+
return path.read_text(encoding="utf-8")
|
|
61
|
+
except UnicodeDecodeError as exc:
|
|
62
|
+
raise InputError(
|
|
63
|
+
f"report is not valid UTF-8: {path}",
|
|
64
|
+
code="SUMMARY_MALFORMED",
|
|
65
|
+
context={"file": str(path)},
|
|
66
|
+
) from exc
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _lines(text: str) -> list[str]:
|
|
70
|
+
"""Split like Perl's ``while (<$fh>)``: on \\n, no phantom final line."""
|
|
71
|
+
if not text:
|
|
72
|
+
return []
|
|
73
|
+
parts = text.split("\n")
|
|
74
|
+
if parts[-1] == "":
|
|
75
|
+
parts.pop()
|
|
76
|
+
return parts
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _detect_separator(text: str) -> str:
|
|
80
|
+
"""[gapit-extension] auto-detect the table separator per file: tab if the
|
|
81
|
+
first line has any, else comma, else tab (single-column degenerate)."""
|
|
82
|
+
first = _lines(text)[0] if text else ""
|
|
83
|
+
if "\t" in first:
|
|
84
|
+
return "\t"
|
|
85
|
+
if "," in first:
|
|
86
|
+
return ","
|
|
87
|
+
return "\t"
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _malformed(path: Path, line_number: int, detail: str) -> InputError:
|
|
91
|
+
return InputError(
|
|
92
|
+
f"malformed report row: {detail} ({path}:{line_number})",
|
|
93
|
+
code="SUMMARY_MALFORMED",
|
|
94
|
+
context={"file": str(path), "line": str(line_number)},
|
|
95
|
+
)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def build_summary(paths: list[Path], params: SummaryParams, *, warn: Warn) -> SummaryMatrix:
|
|
99
|
+
"""Aggregate report table(s) into the summary matrix.
|
|
100
|
+
|
|
101
|
+
Dutch mode (exactly one input) keys rows by each table row's FILE column;
|
|
102
|
+
otherwise rows are keyed by input filename as given. Rows are sorted by
|
|
103
|
+
that key — not by display label — matching upstream. The first row seen
|
|
104
|
+
anywhere is the column map (upstream ``@hdr``); later '#' rows are skipped.
|
|
105
|
+
"""
|
|
106
|
+
dutch = len(paths) == 1
|
|
107
|
+
data: dict[str, dict[str, list[str]]] = {}
|
|
108
|
+
seen: set[str] = set()
|
|
109
|
+
indexes: dict[str, int] | None = None
|
|
110
|
+
for path in paths:
|
|
111
|
+
key = str(path)
|
|
112
|
+
if key in seen:
|
|
113
|
+
warn(f"Skipping duplicate file: {key}")
|
|
114
|
+
continue
|
|
115
|
+
seen.add(key)
|
|
116
|
+
text = _read_report(path)
|
|
117
|
+
if not dutch:
|
|
118
|
+
data[key] = {}
|
|
119
|
+
separator = _detect_separator(text)
|
|
120
|
+
for line_number, line in enumerate(_lines(text), start=1):
|
|
121
|
+
columns = line.split(separator)
|
|
122
|
+
if indexes is None:
|
|
123
|
+
# Header-name -> column index with Perl zip semantics (later dups win).
|
|
124
|
+
indexes = dict(zip(columns, range(len(columns)), strict=True))
|
|
125
|
+
if columns[0].startswith("#"):
|
|
126
|
+
continue
|
|
127
|
+
assert indexes is not None # set on the first line, before any data row
|
|
128
|
+
gene_at = indexes.get("GENE")
|
|
129
|
+
metric_at = indexes.get(params.metric)
|
|
130
|
+
if gene_at is None:
|
|
131
|
+
raise _malformed(path, line_number, "header has no GENE column")
|
|
132
|
+
if metric_at is None:
|
|
133
|
+
raise _malformed(path, line_number, f"header has no {params.metric} column")
|
|
134
|
+
if len(columns) <= max(gene_at, metric_at):
|
|
135
|
+
raise _malformed(
|
|
136
|
+
path, line_number, f"expected >= {max(gene_at, metric_at) + 1} columns"
|
|
137
|
+
)
|
|
138
|
+
gene, value = columns[gene_at], columns[metric_at]
|
|
139
|
+
file_key = PurePath(columns[0]).name if params.nopath else columns[0]
|
|
140
|
+
row_key = file_key if dutch else key
|
|
141
|
+
data.setdefault(row_key, {}).setdefault(gene, []).append(value)
|
|
142
|
+
genes = tuple(sorted({gene for hits in data.values() for gene in hits}))
|
|
143
|
+
rows = tuple(
|
|
144
|
+
SummaryRow(
|
|
145
|
+
file=PurePath(row_key).name if params.nopath else row_key,
|
|
146
|
+
num_found=len(data[row_key]),
|
|
147
|
+
cells={gene: tuple(data[row_key][gene]) for gene in genes if gene in data[row_key]},
|
|
148
|
+
)
|
|
149
|
+
for row_key in sorted(data)
|
|
150
|
+
)
|
|
151
|
+
return SummaryMatrix(params=params, genes=genes, rows=rows)
|