gapit 0.2.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gapit/__init__.py +3 -0
- gapit/blast.py +239 -0
- gapit/cli.py +128 -0
- gapit/cmd_db.py +113 -0
- gapit/cmd_db_build.py +72 -0
- gapit/cmd_db_install.py +126 -0
- gapit/cmd_db_outdated.py +51 -0
- gapit/cmd_db_search.py +66 -0
- gapit/cmd_screen.py +214 -0
- gapit/cmd_summary.py +65 -0
- gapit/config.py +51 -0
- gapit/data/snapshots/card.tar.gz +0 -0
- gapit/data/snapshots/vfdb.tar.gz +0 -0
- gapit/db.py +226 -0
- gapit/db_build_ops.py +210 -0
- gapit/db_ops.py +128 -0
- gapit/db_query_ops.py +252 -0
- gapit/dbbuild.py +207 -0
- gapit/dbcodec.py +117 -0
- gapit/dispatch.py +31 -0
- gapit/errors.py +87 -0
- gapit/fasta.py +123 -0
- gapit/formats/__init__.py +1 -0
- gapit/formats/json.py +309 -0
- gapit/formats/md.py +190 -0
- gapit/formats/schemas.py +30 -0
- gapit/formats/summary.py +103 -0
- gapit/formats/tsv.py +45 -0
- gapit/hits.py +107 -0
- gapit/mcp.py +158 -0
- gapit/mcp_schemas.py +123 -0
- gapit/mcp_tools.py +289 -0
- gapit/minimap.py +20 -0
- gapit/minimap2_run.py +114 -0
- gapit/paf.py +115 -0
- gapit/proctools.py +24 -0
- gapit/providers/__init__.py +39 -0
- gapit/providers/argannot.py +94 -0
- gapit/providers/bacmet2.py +59 -0
- gapit/providers/card.py +150 -0
- gapit/providers/common.py +245 -0
- gapit/providers/ecoh.py +63 -0
- gapit/providers/ecoli_vf.py +74 -0
- gapit/providers/megares.py +71 -0
- gapit/providers/ncbi.py +103 -0
- gapit/providers/plasmidfinder.py +69 -0
- gapit/providers/resfinder.py +123 -0
- gapit/providers/snapshots.py +119 -0
- gapit/providers/upec_expec_vf.py +85 -0
- gapit/providers/vfdb.py +92 -0
- gapit/providers/victors.py +109 -0
- gapit/py.typed +0 -0
- gapit/reads.py +221 -0
- gapit/records.py +152 -0
- gapit/report.py +25 -0
- gapit/screening.py +145 -0
- gapit/screening_reads.py +255 -0
- gapit/seqconvert.py +203 -0
- gapit/summary.py +151 -0
- gapit-0.2.2.dist-info/METADATA +183 -0
- gapit-0.2.2.dist-info/RECORD +64 -0
- gapit-0.2.2.dist-info/WHEEL +4 -0
- gapit-0.2.2.dist-info/entry_points.txt +3 -0
- gapit-0.2.2.dist-info/licenses/LICENSE +21 -0
gapit/dbbuild.py
ADDED
|
@@ -0,0 +1,207 @@
|
|
|
1
|
+
"""Wave A3 build pipeline: records.jsonl -> a fully built gapit-native database.
|
|
2
|
+
|
|
3
|
+
``build_database`` turns a directory's truth source (``records.jsonl``) into
|
|
4
|
+
the three artifacts consumers rely on — the ``sequences`` FASTA projection,
|
|
5
|
+
its BLAST index, and the ``gapit-manifest.json`` provenance sidecar written
|
|
6
|
+
LAST (it certifies the artifacts). Every step is deterministic, atomic where
|
|
7
|
+
it matters, and self-verifying: the generated headers must decode back
|
|
8
|
+
through :mod:`gapit.dbcodec` before anything is indexed. No minimap2 index
|
|
9
|
+
is persisted: reads mode indexes the ``sequences`` FASTA in memory only
|
|
10
|
+
(SPEC.md §10).
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import hashlib
|
|
14
|
+
import os
|
|
15
|
+
from collections.abc import Sequence
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from tempfile import NamedTemporaryFile
|
|
18
|
+
from typing import Literal
|
|
19
|
+
|
|
20
|
+
from gapit.db import DbHeader, make_blast_db
|
|
21
|
+
from gapit.dbcodec import decode_seqid, encode_seqid
|
|
22
|
+
from gapit.errors import DatabaseError
|
|
23
|
+
from gapit.fasta import iter_fasta
|
|
24
|
+
from gapit.proctools import note, run_tool
|
|
25
|
+
from gapit.records import Manifest, Record, count_records, read_records, write_manifest
|
|
26
|
+
|
|
27
|
+
_WRAP_COLUMNS = 60
|
|
28
|
+
_CHUNK_BYTES = 1 << 20
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def generate_sequences(records_path: Path, sequences_path: Path) -> None:
|
|
32
|
+
"""Stream ``records.jsonl`` into the ``sequences`` FASTA projection.
|
|
33
|
+
|
|
34
|
+
One header line per record (``encode_seqid(...)`` + space + product),
|
|
35
|
+
sequence wrapped at 60 columns, LF endings, UTF-8. Bytes land in a temp
|
|
36
|
+
file in the target directory and are ``os.replace``d only after the whole
|
|
37
|
+
file was written, so a failure never leaves a partial ``sequences``.
|
|
38
|
+
|
|
39
|
+
Raises DatabaseError ``BUILD_INVALID`` (exit 4) for a record whose product
|
|
40
|
+
contains a line break or whose sequence is empty, and for a records file
|
|
41
|
+
with zero records (makeblastdb cannot index nothing); the per-record
|
|
42
|
+
errors carry the record's gene in context. Duplicate ``(db, gene)`` pairs
|
|
43
|
+
are allowed and preserved — upstream databases contain them and
|
|
44
|
+
records.jsonl order is the truth.
|
|
45
|
+
"""
|
|
46
|
+
temp_path: Path | None = None
|
|
47
|
+
n_written = 0
|
|
48
|
+
try:
|
|
49
|
+
with NamedTemporaryFile(
|
|
50
|
+
mode="w",
|
|
51
|
+
encoding="utf-8",
|
|
52
|
+
newline="\n",
|
|
53
|
+
dir=sequences_path.parent,
|
|
54
|
+
prefix=f".{sequences_path.name}.",
|
|
55
|
+
suffix=".tmp",
|
|
56
|
+
delete=False,
|
|
57
|
+
) as temp:
|
|
58
|
+
temp_path = Path(temp.name)
|
|
59
|
+
for record in read_records(records_path):
|
|
60
|
+
_check_record(record)
|
|
61
|
+
seqid = encode_seqid(record.db, record.gene, record.accession, record.function)
|
|
62
|
+
temp.write(f">{seqid} {record.product}\n")
|
|
63
|
+
for offset in range(0, len(record.sequence), _WRAP_COLUMNS):
|
|
64
|
+
temp.write(f"{record.sequence[offset : offset + _WRAP_COLUMNS]}\n")
|
|
65
|
+
n_written += 1
|
|
66
|
+
if n_written == 0:
|
|
67
|
+
raise DatabaseError(
|
|
68
|
+
f"cannot build {sequences_path}: {records_path} contains no records",
|
|
69
|
+
code="BUILD_INVALID",
|
|
70
|
+
context={"file": str(records_path)},
|
|
71
|
+
)
|
|
72
|
+
assert temp_path is not None
|
|
73
|
+
os.replace(temp_path, sequences_path)
|
|
74
|
+
finally:
|
|
75
|
+
if temp_path is not None:
|
|
76
|
+
temp_path.unlink(missing_ok=True)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _check_record(record: Record) -> None:
|
|
80
|
+
"""Generation-time validation: the two record shapes that would produce a
|
|
81
|
+
broken FASTA (headers must stay one line; empty sequences cannot be
|
|
82
|
+
indexed)."""
|
|
83
|
+
if "\n" in record.product or "\r" in record.product:
|
|
84
|
+
raise DatabaseError(
|
|
85
|
+
f"record {record.gene!r}: product contains a line break",
|
|
86
|
+
code="BUILD_INVALID",
|
|
87
|
+
context={"gene": record.gene},
|
|
88
|
+
)
|
|
89
|
+
if not record.sequence:
|
|
90
|
+
raise DatabaseError(
|
|
91
|
+
f"record {record.gene!r}: empty sequence",
|
|
92
|
+
code="BUILD_INVALID",
|
|
93
|
+
context={"gene": record.gene},
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _self_check_failed(gene: str, reason: str) -> DatabaseError:
|
|
98
|
+
"""The uniform self-check failure: gene located, machine-stable reason."""
|
|
99
|
+
return DatabaseError(
|
|
100
|
+
f"self-check failed at gene {gene!r} ({reason})",
|
|
101
|
+
code="BUILD_SELF_CHECK_FAILED",
|
|
102
|
+
context={"gene": gene, "reason": reason},
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def verify_sequences(sequences_path: Path, records_path: Path, name: str) -> None:
|
|
107
|
+
"""Self-check: every generated FASTA record decodes back to its record.
|
|
108
|
+
|
|
109
|
+
Streams ``sequences`` and ``records.jsonl`` in parallel; the header (via
|
|
110
|
+
:func:`gapit.dbcodec.decode_seqid` with ``default_db=name``) and the
|
|
111
|
+
sequence must match record-for-record, in order. The FASTA description
|
|
112
|
+
(product) is deliberately not compared: the reader strips leading and
|
|
113
|
+
trailing whitespace by design, so it is a display field, not a truth
|
|
114
|
+
field. Any decode failure, field mismatch, or count mismatch raises
|
|
115
|
+
``BUILD_SELF_CHECK_FAILED`` (a structurally invalid FASTA surfaces as its
|
|
116
|
+
native InputError instead — either way the build stops before indexing).
|
|
117
|
+
"""
|
|
118
|
+
fasta_stream = iter_fasta(sequences_path)
|
|
119
|
+
for record in read_records(records_path):
|
|
120
|
+
fasta = next(fasta_stream, None)
|
|
121
|
+
if fasta is None:
|
|
122
|
+
raise _self_check_failed(record.gene, "missing_fasta_record")
|
|
123
|
+
try:
|
|
124
|
+
header = decode_seqid(fasta.id, default_db=name)
|
|
125
|
+
except DatabaseError as exc:
|
|
126
|
+
raise _self_check_failed(record.gene, f"decode:{exc.code}") from exc
|
|
127
|
+
expected = DbHeader(
|
|
128
|
+
database=record.db or name,
|
|
129
|
+
gene=record.gene,
|
|
130
|
+
accession=record.accession,
|
|
131
|
+
function=";".join(record.function),
|
|
132
|
+
)
|
|
133
|
+
if header != expected:
|
|
134
|
+
raise _self_check_failed(record.gene, "header_mismatch")
|
|
135
|
+
if fasta.sequence != record.sequence:
|
|
136
|
+
raise _self_check_failed(record.gene, "sequence_mismatch")
|
|
137
|
+
if next(fasta_stream, None) is not None:
|
|
138
|
+
raise _self_check_failed("", "extra_fasta_record")
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _sha256(path: Path) -> str:
|
|
142
|
+
"""Streaming SHA256 of a file's bytes (1 MiB chunks)."""
|
|
143
|
+
hasher = hashlib.sha256()
|
|
144
|
+
with path.open("rb") as handle:
|
|
145
|
+
while chunk := handle.read(_CHUNK_BYTES):
|
|
146
|
+
hasher.update(chunk)
|
|
147
|
+
return hasher.hexdigest()
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _version_line(argv: list[str]) -> str:
|
|
151
|
+
"""First line of a version command's stdout ('' when it printed nothing)."""
|
|
152
|
+
stdout = run_tool(argv).stdout
|
|
153
|
+
return stdout.splitlines()[0].strip() if stdout else ""
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def build_database(
|
|
157
|
+
db_dir: Path,
|
|
158
|
+
*,
|
|
159
|
+
name: str,
|
|
160
|
+
dbtype: Literal["nucl", "prot"],
|
|
161
|
+
source_urls: Sequence[str],
|
|
162
|
+
fetched_at: str,
|
|
163
|
+
upstream_version: str = "",
|
|
164
|
+
quiet: bool = True,
|
|
165
|
+
debug: bool = False,
|
|
166
|
+
) -> Manifest:
|
|
167
|
+
"""Build every gapit-native artifact in ``db_dir`` from its records.jsonl.
|
|
168
|
+
|
|
169
|
+
Pipeline — each step runs only after the previous one succeeded, and the
|
|
170
|
+
manifest is written LAST (it certifies the artifacts):
|
|
171
|
+
|
|
172
|
+
1. ``generate_sequences`` -> ``db_dir/sequences``
|
|
173
|
+
2. self-check: every header decodes back (``BUILD_SELF_CHECK_FAILED``)
|
|
174
|
+
3. streaming SHA256 of ``sequences``
|
|
175
|
+
4. ``makeblastdb`` with the EXPLICIT ``dbtype`` — the manifest declares
|
|
176
|
+
the type, so the mol_type heuristic is skipped
|
|
177
|
+
5. count records and capture ``blastn -version`` / ``minimap2 --version``
|
|
178
|
+
6. write ``db_dir/gapit-manifest.json`` and return the Manifest
|
|
179
|
+
|
|
180
|
+
No ``.mmi`` is built for either dbtype: reads mode indexes the FASTA in
|
|
181
|
+
memory with the invocation preset's own parameters (minimap2 is
|
|
182
|
+
nucleotide-only, which is why prot databases never participated anyway).
|
|
183
|
+
"""
|
|
184
|
+
records_path = db_dir / "records.jsonl"
|
|
185
|
+
sequences_path = db_dir / "sequences"
|
|
186
|
+
generate_sequences(records_path, sequences_path)
|
|
187
|
+
note(quiet, f"generated {sequences_path}")
|
|
188
|
+
verify_sequences(sequences_path, records_path, name)
|
|
189
|
+
note(quiet, f"self-check passed for {name}")
|
|
190
|
+
sha256 = _sha256(sequences_path)
|
|
191
|
+
make_blast_db(sequences_path, name, dbtype=dbtype, debug=debug)
|
|
192
|
+
note(quiet, f"BLAST index built ({dbtype})")
|
|
193
|
+
manifest = Manifest(
|
|
194
|
+
name=name,
|
|
195
|
+
source_urls=tuple(source_urls),
|
|
196
|
+
fetched_at=fetched_at,
|
|
197
|
+
sha256=sha256,
|
|
198
|
+
n_records=count_records(records_path),
|
|
199
|
+
dbtype=dbtype,
|
|
200
|
+
upstream_version=upstream_version,
|
|
201
|
+
makeblastdb_version=_version_line(["blastn", "-version"]),
|
|
202
|
+
# Kept although no .mmi is built: environment provenance for the
|
|
203
|
+
# machine that produced the artifacts (spec'd in gapit.manifest/1).
|
|
204
|
+
minimap2_version=_version_line(["minimap2", "--version"]),
|
|
205
|
+
)
|
|
206
|
+
write_manifest(manifest, db_dir / "gapit-manifest.json")
|
|
207
|
+
return manifest
|
gapit/dbcodec.py
ADDED
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
"""gapit/v1 database header codec: tagged, percent-encoded sequence ids.
|
|
2
|
+
|
|
3
|
+
Native header format (Phase 7): ``gapit|db=<v>|gene=<v>|acc=<v>|func=<v>`` with
|
|
4
|
+
fixed key order and lowercase-hex percent escapes inside values. Encode is
|
|
5
|
+
total for any input; decode is strict (``DatabaseError HEADER_MALFORMED``).
|
|
6
|
+
Seqids without the ``gapit|`` prefix delegate to the legacy ``~~~`` rules in
|
|
7
|
+
``gapit.db`` (SPEC.md §4 step 5).
|
|
8
|
+
|
|
9
|
+
The fourth slot is named ``func`` because gapit answers presence/absence — the
|
|
10
|
+
values it carries are functional categories (AMR classes, virulence, O-antigen,
|
|
11
|
+
replicon, biocide), not only resistance (Wave F1 rename; outputs keep the
|
|
12
|
+
frozen ``RESISTANCE``/``resistance`` names).
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
import re
|
|
16
|
+
from collections.abc import Sequence
|
|
17
|
+
|
|
18
|
+
from gapit.db import DbHeader, parse_db_header
|
|
19
|
+
from gapit.errors import DatabaseError
|
|
20
|
+
|
|
21
|
+
PREFIX = "gapit|"
|
|
22
|
+
|
|
23
|
+
# Values-only escape alphabet (probe-verified byte-identical through
|
|
24
|
+
# makeblastdb/blastn/minimap2); every other char travels verbatim.
|
|
25
|
+
_ENCODE_TABLE = str.maketrans(
|
|
26
|
+
{
|
|
27
|
+
"%": "%25",
|
|
28
|
+
"|": "%7C",
|
|
29
|
+
"=": "%3D",
|
|
30
|
+
";": "%3B",
|
|
31
|
+
" ": "%20",
|
|
32
|
+
"\t": "%09",
|
|
33
|
+
"\n": "%0A",
|
|
34
|
+
"\r": "%0D",
|
|
35
|
+
}
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
_ESCAPE_RE = re.compile("%([0-9a-fA-F]{2})")
|
|
39
|
+
_BAD_ESCAPE_RE = re.compile(r"%(?![0-9a-fA-F]{2})")
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def is_gapit_header(seqid: str) -> bool:
|
|
43
|
+
"""True iff the seqid carries the native ``gapit|`` tagged prefix."""
|
|
44
|
+
return seqid.startswith(PREFIX)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def encode_seqid(db: str, gene: str, accession: str, function: Sequence[str]) -> str:
|
|
48
|
+
"""Render one record as a native gapit/v1 seqid (total: never raises).
|
|
49
|
+
|
|
50
|
+
The function list is joined with ``;`` first, then encoded as one value,
|
|
51
|
+
so it decodes back to the abricate display string ``a;b``.
|
|
52
|
+
"""
|
|
53
|
+
return "|".join(
|
|
54
|
+
(
|
|
55
|
+
"gapit",
|
|
56
|
+
f"db={db.translate(_ENCODE_TABLE)}",
|
|
57
|
+
f"gene={gene.translate(_ENCODE_TABLE)}",
|
|
58
|
+
f"acc={accession.translate(_ENCODE_TABLE)}",
|
|
59
|
+
f"func={';'.join(function).translate(_ENCODE_TABLE)}",
|
|
60
|
+
)
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def decode_seqid(seqid: str, default_db: str) -> DbHeader:
|
|
65
|
+
"""Parse a seqid into a ``DbHeader``.
|
|
66
|
+
|
|
67
|
+
``gapit|``-prefixed seqids use the strict native format: bad escapes,
|
|
68
|
+
missing/empty gene, a missing ``db``/``acc`` KEY, and duplicate keys raise
|
|
69
|
+
``HEADER_MALFORMED``; unknown keys are skipped (forward compat); an empty
|
|
70
|
+
``db`` value falls back to ``default_db`` (symmetric with legacy) and an
|
|
71
|
+
empty ``acc`` value is allowed (unpublished sequences); the ``func`` key is
|
|
72
|
+
intentionally optional and defaults to ``""``. Anything else delegates to
|
|
73
|
+
``gapit.db.parse_db_header``.
|
|
74
|
+
"""
|
|
75
|
+
if not is_gapit_header(seqid):
|
|
76
|
+
return parse_db_header(seqid, default_db)
|
|
77
|
+
fields: dict[str, str] = {}
|
|
78
|
+
for segment in seqid.split("|")[1:]:
|
|
79
|
+
key, sep, raw_value = segment.partition("=")
|
|
80
|
+
if not sep:
|
|
81
|
+
raise _malformed(seqid, "segment_without_key")
|
|
82
|
+
if not key:
|
|
83
|
+
raise _malformed(seqid, "empty_key")
|
|
84
|
+
if key in fields:
|
|
85
|
+
raise _malformed(seqid, f"duplicate_key:{key}")
|
|
86
|
+
fields[key] = _decode_value(raw_value, key=key, seqid=seqid)
|
|
87
|
+
gene = fields.get("gene", "")
|
|
88
|
+
if not gene:
|
|
89
|
+
raise _malformed(seqid, "missing_gene")
|
|
90
|
+
if "db" not in fields:
|
|
91
|
+
raise _malformed(seqid, "missing_db")
|
|
92
|
+
if "acc" not in fields:
|
|
93
|
+
raise _malformed(seqid, "missing_acc")
|
|
94
|
+
return DbHeader(
|
|
95
|
+
database=fields["db"] or default_db,
|
|
96
|
+
gene=gene,
|
|
97
|
+
accession=fields["acc"],
|
|
98
|
+
# func was encoded as one ;-joined value; the decoded string already is
|
|
99
|
+
# the abricate display form "a;b".
|
|
100
|
+
function=fields.get("func", ""),
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _decode_value(raw: str, *, key: str, seqid: str) -> str:
|
|
105
|
+
"""Percent-decode one raw value; every ``%`` must precede two hex digits."""
|
|
106
|
+
if _BAD_ESCAPE_RE.search(raw):
|
|
107
|
+
raise _malformed(seqid, f"invalid_percent_escape:{key}")
|
|
108
|
+
return _ESCAPE_RE.sub(lambda match: chr(int(match.group(1), 16)), raw)
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _malformed(seqid: str, reason: str) -> DatabaseError:
|
|
112
|
+
"""Build the strict-decode failure (exit 4) with seqid + reason context."""
|
|
113
|
+
return DatabaseError(
|
|
114
|
+
f"malformed gapit/v1 sequence header: {reason}",
|
|
115
|
+
code="HEADER_MALFORMED",
|
|
116
|
+
context={"seqid": seqid, "reason": reason},
|
|
117
|
+
)
|
gapit/dispatch.py
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
"""Shared CLI plumbing: error-envelope dispatch and the ``--datadir`` option.
|
|
2
|
+
|
|
3
|
+
Imported by every command module (cli.py, cmd_*.py); imports nothing from
|
|
4
|
+
gapit except :mod:`gapit.errors`, so it can never participate in a cycle.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from collections.abc import Callable
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Annotated
|
|
10
|
+
|
|
11
|
+
import typer
|
|
12
|
+
|
|
13
|
+
from gapit.errors import GapitError, render_error
|
|
14
|
+
|
|
15
|
+
Datadir = Annotated[
|
|
16
|
+
Path | None,
|
|
17
|
+
typer.Option(
|
|
18
|
+
"--datadir",
|
|
19
|
+
help="Database directory (default: $GAPIT_DATADIR, then ~/.local/share/gapit/db).",
|
|
20
|
+
),
|
|
21
|
+
]
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def dispatch(action: Callable[[], None]) -> None:
|
|
25
|
+
"""Run a command body; any failure renders the gapit.error/1 envelope on
|
|
26
|
+
stderr and exits with the documented code (UNEXPECTED/1 for non-GapitError)."""
|
|
27
|
+
try:
|
|
28
|
+
action()
|
|
29
|
+
except Exception as exc:
|
|
30
|
+
typer.echo(render_error(exc), err=True)
|
|
31
|
+
raise typer.Exit(code=exc.exit_code if isinstance(exc, GapitError) else 1) from exc
|
gapit/errors.py
ADDED
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
"""Typed errors, shared raise-helpers, and the JSON error envelope
|
|
2
|
+
(gapit.error/1).
|
|
3
|
+
|
|
4
|
+
Exit-code contract (AGENTS.md §5): 2 usage, 3 missing dependency, 4 db error,
|
|
5
|
+
5 input error, 1 unexpected. Failures render as a one-line JSON envelope on
|
|
6
|
+
stderr via ``render_error``.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Literal, NoReturn
|
|
11
|
+
|
|
12
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class GapitError(Exception):
|
|
16
|
+
"""Base class for all typed gapit errors."""
|
|
17
|
+
|
|
18
|
+
code: str
|
|
19
|
+
exit_code: int = 1
|
|
20
|
+
|
|
21
|
+
def __init__(self, message: str, *, code: str, context: dict[str, str] | None = None) -> None:
|
|
22
|
+
super().__init__(message)
|
|
23
|
+
self.code = code
|
|
24
|
+
self.context: dict[str, str] = context if context is not None else {}
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class UsageError(GapitError):
|
|
28
|
+
"""Invalid command-line usage."""
|
|
29
|
+
|
|
30
|
+
exit_code = 2
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class DependencyError(GapitError):
|
|
34
|
+
"""An external binary (BLAST+, minimap2) is missing from PATH."""
|
|
35
|
+
|
|
36
|
+
exit_code = 3
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class DatabaseError(GapitError):
|
|
40
|
+
"""A datadir/database is missing, unindexed, or failed to build."""
|
|
41
|
+
|
|
42
|
+
exit_code = 4
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class InputError(GapitError):
|
|
46
|
+
"""User-supplied input is malformed."""
|
|
47
|
+
|
|
48
|
+
exit_code = 5
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def usage_fail(message: str) -> NoReturn:
|
|
52
|
+
"""Raise a usage error (gapit.error/1 envelope, exit 2)."""
|
|
53
|
+
raise UsageError(message, code="USAGE_ERROR")
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def ensure_input_file(path: Path, what: str = "input file") -> None:
|
|
57
|
+
"""Raise INPUT_NOT_FOUND for a missing/unreadable input path; ``what``
|
|
58
|
+
names the kind in the message (e.g. "reads file")."""
|
|
59
|
+
if not path.is_file():
|
|
60
|
+
raise InputError(
|
|
61
|
+
f"{what} not found or unreadable: {path}",
|
|
62
|
+
code="INPUT_NOT_FOUND",
|
|
63
|
+
context={"file": str(path)},
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
class ErrorEnvelope(BaseModel, frozen=True):
|
|
68
|
+
"""The gapit.error/1 stderr envelope."""
|
|
69
|
+
|
|
70
|
+
model_config = ConfigDict(populate_by_name=True)
|
|
71
|
+
|
|
72
|
+
schema_name: Literal["gapit.error/1"] = Field(default="gapit.error/1", alias="schema")
|
|
73
|
+
code: str
|
|
74
|
+
message: str
|
|
75
|
+
context: dict[str, str]
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def render_error(exc: BaseException) -> str:
|
|
79
|
+
"""One-line compact JSON envelope for any failure (non-GapitError becomes
|
|
80
|
+
UNEXPECTED, exit 1)."""
|
|
81
|
+
if isinstance(exc, GapitError):
|
|
82
|
+
envelope = ErrorEnvelope(code=exc.code, message=str(exc), context=exc.context)
|
|
83
|
+
else:
|
|
84
|
+
envelope = ErrorEnvelope(
|
|
85
|
+
code="UNEXPECTED", message=f"{type(exc).__name__}: {exc}", context={}
|
|
86
|
+
)
|
|
87
|
+
return envelope.model_dump_json(by_alias=True)
|
gapit/fasta.py
ADDED
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
"""Streaming FASTA reader (plain, gzip, bzip2)."""
|
|
2
|
+
|
|
3
|
+
import bz2
|
|
4
|
+
import gzip
|
|
5
|
+
from collections.abc import Iterator
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import IO
|
|
8
|
+
|
|
9
|
+
from pydantic import BaseModel
|
|
10
|
+
|
|
11
|
+
from gapit.errors import InputError
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class FastaRecord(BaseModel, frozen=True):
|
|
15
|
+
"""One FASTA record: ``id`` is the first whitespace-delimited header token."""
|
|
16
|
+
|
|
17
|
+
id: str
|
|
18
|
+
description: str
|
|
19
|
+
sequence: str
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def open_text(path: Path) -> IO[str]:
|
|
23
|
+
"""Open a sequence file as UTF-8 text, transparently decompressing .gz/.bz2.
|
|
24
|
+
|
|
25
|
+
Shared by the FASTA readers and seqconvert's normalizer; the suffix
|
|
26
|
+
(not content) picks the decompressor, matching the any2fasta CLI contract.
|
|
27
|
+
"""
|
|
28
|
+
if path.name.endswith(".gz"):
|
|
29
|
+
return gzip.open(path, "rt", encoding="utf-8")
|
|
30
|
+
if path.name.endswith(".bz2"):
|
|
31
|
+
return bz2.open(path, "rt", encoding="utf-8")
|
|
32
|
+
return path.open("rt", encoding="utf-8")
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _finalize(path: Path, pending: tuple[str, str, list[str]]) -> FastaRecord:
|
|
36
|
+
"""Turn accumulated (id, description, sequence lines) into a FastaRecord."""
|
|
37
|
+
seqid, description, chunks = pending
|
|
38
|
+
if not chunks:
|
|
39
|
+
raise InputError(
|
|
40
|
+
f"{path}: record {seqid!r} has an empty sequence",
|
|
41
|
+
code="INVALID_FASTA",
|
|
42
|
+
context={"file": str(path)},
|
|
43
|
+
)
|
|
44
|
+
return FastaRecord(id=seqid, description=description, sequence="".join(chunks))
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def iter_fasta(path: Path) -> Iterator[FastaRecord]:
|
|
48
|
+
"""Stream FastaRecords from a FASTA file (plain/.gz/.bz2, UTF-8).
|
|
49
|
+
|
|
50
|
+
Raises InputError on content before the first ``>`` header, on a record
|
|
51
|
+
with an empty sequence, or on an unreadable/truncated stream (a .gz/.bz2
|
|
52
|
+
cut short raises EOFError mid-read, not an OSError); an empty file yields
|
|
53
|
+
zero records.
|
|
54
|
+
"""
|
|
55
|
+
try:
|
|
56
|
+
with open_text(path) as handle:
|
|
57
|
+
# pending = (id, description, sequence lines) of the record being read
|
|
58
|
+
pending: tuple[str, str, list[str]] | None = None
|
|
59
|
+
for lineno, raw_line in enumerate(handle, start=1):
|
|
60
|
+
line = raw_line.strip()
|
|
61
|
+
if not line:
|
|
62
|
+
continue
|
|
63
|
+
if not line.startswith(">"):
|
|
64
|
+
if pending is None:
|
|
65
|
+
raise InputError(
|
|
66
|
+
f"{path}: content before first '>' header at line {lineno}",
|
|
67
|
+
code="INVALID_FASTA",
|
|
68
|
+
context={"file": str(path)},
|
|
69
|
+
)
|
|
70
|
+
pending[2].append(line)
|
|
71
|
+
continue
|
|
72
|
+
if pending is not None:
|
|
73
|
+
yield _finalize(path, pending)
|
|
74
|
+
parts = line[1:].strip().split(maxsplit=1)
|
|
75
|
+
pending = (
|
|
76
|
+
parts[0] if parts else "",
|
|
77
|
+
parts[1] if len(parts) > 1 else "",
|
|
78
|
+
[],
|
|
79
|
+
)
|
|
80
|
+
if pending is not None:
|
|
81
|
+
yield _finalize(path, pending)
|
|
82
|
+
except (OSError, UnicodeDecodeError, EOFError) as exc:
|
|
83
|
+
raise InputError(
|
|
84
|
+
f"{path}: could not read input: {exc}",
|
|
85
|
+
code="INVALID_FASTA",
|
|
86
|
+
context={"file": str(path)},
|
|
87
|
+
) from exc
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def iter_fasta_headers(path: Path) -> Iterator[tuple[str, str]]:
|
|
91
|
+
"""Stream (id, description) pairs from FASTA headers only.
|
|
92
|
+
|
|
93
|
+
Sequence lines are skipped, not accumulated — the cheap iterator for
|
|
94
|
+
header-only consumers (e.g. product lookups on a large db). Unlike
|
|
95
|
+
iter_fasta, the empty-sequence InputError does NOT apply: no sequence
|
|
96
|
+
check is performed because sequences are never read. Content before the
|
|
97
|
+
first ``>`` header and an unreadable/truncated stream still raise
|
|
98
|
+
InputError; header parsing (id is the first whitespace token, description
|
|
99
|
+
the rest) matches iter_fasta; an empty file yields nothing.
|
|
100
|
+
"""
|
|
101
|
+
try:
|
|
102
|
+
with open_text(path) as handle:
|
|
103
|
+
seen_header = False
|
|
104
|
+
for lineno, raw_line in enumerate(handle, start=1):
|
|
105
|
+
line = raw_line.strip()
|
|
106
|
+
if line.startswith(">"):
|
|
107
|
+
seen_header = True
|
|
108
|
+
parts = line[1:].strip().split(maxsplit=1)
|
|
109
|
+
yield (parts[0] if parts else "", parts[1] if len(parts) > 1 else "")
|
|
110
|
+
elif not line or seen_header:
|
|
111
|
+
continue
|
|
112
|
+
else:
|
|
113
|
+
raise InputError(
|
|
114
|
+
f"{path}: content before first '>' header at line {lineno}",
|
|
115
|
+
code="INVALID_FASTA",
|
|
116
|
+
context={"file": str(path)},
|
|
117
|
+
)
|
|
118
|
+
except (OSError, UnicodeDecodeError, EOFError) as exc:
|
|
119
|
+
raise InputError(
|
|
120
|
+
f"{path}: could not read input: {exc}",
|
|
121
|
+
code="INVALID_FASTA",
|
|
122
|
+
context={"file": str(path)},
|
|
123
|
+
) from exc
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Output formatters for gapit reports (TSV/CSV today; JSON and Markdown in Phase 4)."""
|