gapit 0.2.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. gapit/__init__.py +3 -0
  2. gapit/blast.py +239 -0
  3. gapit/cli.py +128 -0
  4. gapit/cmd_db.py +113 -0
  5. gapit/cmd_db_build.py +72 -0
  6. gapit/cmd_db_install.py +126 -0
  7. gapit/cmd_db_outdated.py +51 -0
  8. gapit/cmd_db_search.py +66 -0
  9. gapit/cmd_screen.py +214 -0
  10. gapit/cmd_summary.py +65 -0
  11. gapit/config.py +51 -0
  12. gapit/data/snapshots/card.tar.gz +0 -0
  13. gapit/data/snapshots/vfdb.tar.gz +0 -0
  14. gapit/db.py +226 -0
  15. gapit/db_build_ops.py +210 -0
  16. gapit/db_ops.py +128 -0
  17. gapit/db_query_ops.py +252 -0
  18. gapit/dbbuild.py +207 -0
  19. gapit/dbcodec.py +117 -0
  20. gapit/dispatch.py +31 -0
  21. gapit/errors.py +87 -0
  22. gapit/fasta.py +123 -0
  23. gapit/formats/__init__.py +1 -0
  24. gapit/formats/json.py +309 -0
  25. gapit/formats/md.py +190 -0
  26. gapit/formats/schemas.py +30 -0
  27. gapit/formats/summary.py +103 -0
  28. gapit/formats/tsv.py +45 -0
  29. gapit/hits.py +107 -0
  30. gapit/mcp.py +158 -0
  31. gapit/mcp_schemas.py +123 -0
  32. gapit/mcp_tools.py +289 -0
  33. gapit/minimap.py +20 -0
  34. gapit/minimap2_run.py +114 -0
  35. gapit/paf.py +115 -0
  36. gapit/proctools.py +24 -0
  37. gapit/providers/__init__.py +39 -0
  38. gapit/providers/argannot.py +94 -0
  39. gapit/providers/bacmet2.py +59 -0
  40. gapit/providers/card.py +150 -0
  41. gapit/providers/common.py +245 -0
  42. gapit/providers/ecoh.py +63 -0
  43. gapit/providers/ecoli_vf.py +74 -0
  44. gapit/providers/megares.py +71 -0
  45. gapit/providers/ncbi.py +103 -0
  46. gapit/providers/plasmidfinder.py +69 -0
  47. gapit/providers/resfinder.py +123 -0
  48. gapit/providers/snapshots.py +119 -0
  49. gapit/providers/upec_expec_vf.py +85 -0
  50. gapit/providers/vfdb.py +92 -0
  51. gapit/providers/victors.py +109 -0
  52. gapit/py.typed +0 -0
  53. gapit/reads.py +221 -0
  54. gapit/records.py +152 -0
  55. gapit/report.py +25 -0
  56. gapit/screening.py +145 -0
  57. gapit/screening_reads.py +255 -0
  58. gapit/seqconvert.py +203 -0
  59. gapit/summary.py +151 -0
  60. gapit-0.2.2.dist-info/METADATA +183 -0
  61. gapit-0.2.2.dist-info/RECORD +64 -0
  62. gapit-0.2.2.dist-info/WHEEL +4 -0
  63. gapit-0.2.2.dist-info/entry_points.txt +3 -0
  64. gapit-0.2.2.dist-info/licenses/LICENSE +21 -0
gapit/dbbuild.py ADDED
@@ -0,0 +1,207 @@
1
+ """Wave A3 build pipeline: records.jsonl -> a fully built gapit-native database.
2
+
3
+ ``build_database`` turns a directory's truth source (``records.jsonl``) into
4
+ the three artifacts consumers rely on — the ``sequences`` FASTA projection,
5
+ its BLAST index, and the ``gapit-manifest.json`` provenance sidecar written
6
+ LAST (it certifies the artifacts). Every step is deterministic, atomic where
7
+ it matters, and self-verifying: the generated headers must decode back
8
+ through :mod:`gapit.dbcodec` before anything is indexed. No minimap2 index
9
+ is persisted: reads mode indexes the ``sequences`` FASTA in memory only
10
+ (SPEC.md §10).
11
+ """
12
+
13
+ import hashlib
14
+ import os
15
+ from collections.abc import Sequence
16
+ from pathlib import Path
17
+ from tempfile import NamedTemporaryFile
18
+ from typing import Literal
19
+
20
+ from gapit.db import DbHeader, make_blast_db
21
+ from gapit.dbcodec import decode_seqid, encode_seqid
22
+ from gapit.errors import DatabaseError
23
+ from gapit.fasta import iter_fasta
24
+ from gapit.proctools import note, run_tool
25
+ from gapit.records import Manifest, Record, count_records, read_records, write_manifest
26
+
27
+ _WRAP_COLUMNS = 60
28
+ _CHUNK_BYTES = 1 << 20
29
+
30
+
31
+ def generate_sequences(records_path: Path, sequences_path: Path) -> None:
32
+ """Stream ``records.jsonl`` into the ``sequences`` FASTA projection.
33
+
34
+ One header line per record (``encode_seqid(...)`` + space + product),
35
+ sequence wrapped at 60 columns, LF endings, UTF-8. Bytes land in a temp
36
+ file in the target directory and are ``os.replace``d only after the whole
37
+ file was written, so a failure never leaves a partial ``sequences``.
38
+
39
+ Raises DatabaseError ``BUILD_INVALID`` (exit 4) for a record whose product
40
+ contains a line break or whose sequence is empty, and for a records file
41
+ with zero records (makeblastdb cannot index nothing); the per-record
42
+ errors carry the record's gene in context. Duplicate ``(db, gene)`` pairs
43
+ are allowed and preserved — upstream databases contain them and
44
+ records.jsonl order is the truth.
45
+ """
46
+ temp_path: Path | None = None
47
+ n_written = 0
48
+ try:
49
+ with NamedTemporaryFile(
50
+ mode="w",
51
+ encoding="utf-8",
52
+ newline="\n",
53
+ dir=sequences_path.parent,
54
+ prefix=f".{sequences_path.name}.",
55
+ suffix=".tmp",
56
+ delete=False,
57
+ ) as temp:
58
+ temp_path = Path(temp.name)
59
+ for record in read_records(records_path):
60
+ _check_record(record)
61
+ seqid = encode_seqid(record.db, record.gene, record.accession, record.function)
62
+ temp.write(f">{seqid} {record.product}\n")
63
+ for offset in range(0, len(record.sequence), _WRAP_COLUMNS):
64
+ temp.write(f"{record.sequence[offset : offset + _WRAP_COLUMNS]}\n")
65
+ n_written += 1
66
+ if n_written == 0:
67
+ raise DatabaseError(
68
+ f"cannot build {sequences_path}: {records_path} contains no records",
69
+ code="BUILD_INVALID",
70
+ context={"file": str(records_path)},
71
+ )
72
+ assert temp_path is not None
73
+ os.replace(temp_path, sequences_path)
74
+ finally:
75
+ if temp_path is not None:
76
+ temp_path.unlink(missing_ok=True)
77
+
78
+
79
+ def _check_record(record: Record) -> None:
80
+ """Generation-time validation: the two record shapes that would produce a
81
+ broken FASTA (headers must stay one line; empty sequences cannot be
82
+ indexed)."""
83
+ if "\n" in record.product or "\r" in record.product:
84
+ raise DatabaseError(
85
+ f"record {record.gene!r}: product contains a line break",
86
+ code="BUILD_INVALID",
87
+ context={"gene": record.gene},
88
+ )
89
+ if not record.sequence:
90
+ raise DatabaseError(
91
+ f"record {record.gene!r}: empty sequence",
92
+ code="BUILD_INVALID",
93
+ context={"gene": record.gene},
94
+ )
95
+
96
+
97
+ def _self_check_failed(gene: str, reason: str) -> DatabaseError:
98
+ """The uniform self-check failure: gene located, machine-stable reason."""
99
+ return DatabaseError(
100
+ f"self-check failed at gene {gene!r} ({reason})",
101
+ code="BUILD_SELF_CHECK_FAILED",
102
+ context={"gene": gene, "reason": reason},
103
+ )
104
+
105
+
106
+ def verify_sequences(sequences_path: Path, records_path: Path, name: str) -> None:
107
+ """Self-check: every generated FASTA record decodes back to its record.
108
+
109
+ Streams ``sequences`` and ``records.jsonl`` in parallel; the header (via
110
+ :func:`gapit.dbcodec.decode_seqid` with ``default_db=name``) and the
111
+ sequence must match record-for-record, in order. The FASTA description
112
+ (product) is deliberately not compared: the reader strips leading and
113
+ trailing whitespace by design, so it is a display field, not a truth
114
+ field. Any decode failure, field mismatch, or count mismatch raises
115
+ ``BUILD_SELF_CHECK_FAILED`` (a structurally invalid FASTA surfaces as its
116
+ native InputError instead — either way the build stops before indexing).
117
+ """
118
+ fasta_stream = iter_fasta(sequences_path)
119
+ for record in read_records(records_path):
120
+ fasta = next(fasta_stream, None)
121
+ if fasta is None:
122
+ raise _self_check_failed(record.gene, "missing_fasta_record")
123
+ try:
124
+ header = decode_seqid(fasta.id, default_db=name)
125
+ except DatabaseError as exc:
126
+ raise _self_check_failed(record.gene, f"decode:{exc.code}") from exc
127
+ expected = DbHeader(
128
+ database=record.db or name,
129
+ gene=record.gene,
130
+ accession=record.accession,
131
+ function=";".join(record.function),
132
+ )
133
+ if header != expected:
134
+ raise _self_check_failed(record.gene, "header_mismatch")
135
+ if fasta.sequence != record.sequence:
136
+ raise _self_check_failed(record.gene, "sequence_mismatch")
137
+ if next(fasta_stream, None) is not None:
138
+ raise _self_check_failed("", "extra_fasta_record")
139
+
140
+
141
+ def _sha256(path: Path) -> str:
142
+ """Streaming SHA256 of a file's bytes (1 MiB chunks)."""
143
+ hasher = hashlib.sha256()
144
+ with path.open("rb") as handle:
145
+ while chunk := handle.read(_CHUNK_BYTES):
146
+ hasher.update(chunk)
147
+ return hasher.hexdigest()
148
+
149
+
150
+ def _version_line(argv: list[str]) -> str:
151
+ """First line of a version command's stdout ('' when it printed nothing)."""
152
+ stdout = run_tool(argv).stdout
153
+ return stdout.splitlines()[0].strip() if stdout else ""
154
+
155
+
156
+ def build_database(
157
+ db_dir: Path,
158
+ *,
159
+ name: str,
160
+ dbtype: Literal["nucl", "prot"],
161
+ source_urls: Sequence[str],
162
+ fetched_at: str,
163
+ upstream_version: str = "",
164
+ quiet: bool = True,
165
+ debug: bool = False,
166
+ ) -> Manifest:
167
+ """Build every gapit-native artifact in ``db_dir`` from its records.jsonl.
168
+
169
+ Pipeline — each step runs only after the previous one succeeded, and the
170
+ manifest is written LAST (it certifies the artifacts):
171
+
172
+ 1. ``generate_sequences`` -> ``db_dir/sequences``
173
+ 2. self-check: every header decodes back (``BUILD_SELF_CHECK_FAILED``)
174
+ 3. streaming SHA256 of ``sequences``
175
+ 4. ``makeblastdb`` with the EXPLICIT ``dbtype`` — the manifest declares
176
+ the type, so the mol_type heuristic is skipped
177
+ 5. count records and capture ``blastn -version`` / ``minimap2 --version``
178
+ 6. write ``db_dir/gapit-manifest.json`` and return the Manifest
179
+
180
+ No ``.mmi`` is built for either dbtype: reads mode indexes the FASTA in
181
+ memory with the invocation preset's own parameters (minimap2 is
182
+ nucleotide-only, which is why prot databases never participated anyway).
183
+ """
184
+ records_path = db_dir / "records.jsonl"
185
+ sequences_path = db_dir / "sequences"
186
+ generate_sequences(records_path, sequences_path)
187
+ note(quiet, f"generated {sequences_path}")
188
+ verify_sequences(sequences_path, records_path, name)
189
+ note(quiet, f"self-check passed for {name}")
190
+ sha256 = _sha256(sequences_path)
191
+ make_blast_db(sequences_path, name, dbtype=dbtype, debug=debug)
192
+ note(quiet, f"BLAST index built ({dbtype})")
193
+ manifest = Manifest(
194
+ name=name,
195
+ source_urls=tuple(source_urls),
196
+ fetched_at=fetched_at,
197
+ sha256=sha256,
198
+ n_records=count_records(records_path),
199
+ dbtype=dbtype,
200
+ upstream_version=upstream_version,
201
+ makeblastdb_version=_version_line(["blastn", "-version"]),
202
+ # Kept although no .mmi is built: environment provenance for the
203
+ # machine that produced the artifacts (spec'd in gapit.manifest/1).
204
+ minimap2_version=_version_line(["minimap2", "--version"]),
205
+ )
206
+ write_manifest(manifest, db_dir / "gapit-manifest.json")
207
+ return manifest
gapit/dbcodec.py ADDED
@@ -0,0 +1,117 @@
1
+ """gapit/v1 database header codec: tagged, percent-encoded sequence ids.
2
+
3
+ Native header format (Phase 7): ``gapit|db=<v>|gene=<v>|acc=<v>|func=<v>`` with
4
+ fixed key order and lowercase-hex percent escapes inside values. Encode is
5
+ total for any input; decode is strict (``DatabaseError HEADER_MALFORMED``).
6
+ Seqids without the ``gapit|`` prefix delegate to the legacy ``~~~`` rules in
7
+ ``gapit.db`` (SPEC.md §4 step 5).
8
+
9
+ The fourth slot is named ``func`` because gapit answers presence/absence — the
10
+ values it carries are functional categories (AMR classes, virulence, O-antigen,
11
+ replicon, biocide), not only resistance (Wave F1 rename; outputs keep the
12
+ frozen ``RESISTANCE``/``resistance`` names).
13
+ """
14
+
15
+ import re
16
+ from collections.abc import Sequence
17
+
18
+ from gapit.db import DbHeader, parse_db_header
19
+ from gapit.errors import DatabaseError
20
+
21
+ PREFIX = "gapit|"
22
+
23
+ # Values-only escape alphabet (probe-verified byte-identical through
24
+ # makeblastdb/blastn/minimap2); every other char travels verbatim.
25
+ _ENCODE_TABLE = str.maketrans(
26
+ {
27
+ "%": "%25",
28
+ "|": "%7C",
29
+ "=": "%3D",
30
+ ";": "%3B",
31
+ " ": "%20",
32
+ "\t": "%09",
33
+ "\n": "%0A",
34
+ "\r": "%0D",
35
+ }
36
+ )
37
+
38
+ _ESCAPE_RE = re.compile("%([0-9a-fA-F]{2})")
39
+ _BAD_ESCAPE_RE = re.compile(r"%(?![0-9a-fA-F]{2})")
40
+
41
+
42
+ def is_gapit_header(seqid: str) -> bool:
43
+ """True iff the seqid carries the native ``gapit|`` tagged prefix."""
44
+ return seqid.startswith(PREFIX)
45
+
46
+
47
+ def encode_seqid(db: str, gene: str, accession: str, function: Sequence[str]) -> str:
48
+ """Render one record as a native gapit/v1 seqid (total: never raises).
49
+
50
+ The function list is joined with ``;`` first, then encoded as one value,
51
+ so it decodes back to the abricate display string ``a;b``.
52
+ """
53
+ return "|".join(
54
+ (
55
+ "gapit",
56
+ f"db={db.translate(_ENCODE_TABLE)}",
57
+ f"gene={gene.translate(_ENCODE_TABLE)}",
58
+ f"acc={accession.translate(_ENCODE_TABLE)}",
59
+ f"func={';'.join(function).translate(_ENCODE_TABLE)}",
60
+ )
61
+ )
62
+
63
+
64
+ def decode_seqid(seqid: str, default_db: str) -> DbHeader:
65
+ """Parse a seqid into a ``DbHeader``.
66
+
67
+ ``gapit|``-prefixed seqids use the strict native format: bad escapes,
68
+ missing/empty gene, a missing ``db``/``acc`` KEY, and duplicate keys raise
69
+ ``HEADER_MALFORMED``; unknown keys are skipped (forward compat); an empty
70
+ ``db`` value falls back to ``default_db`` (symmetric with legacy) and an
71
+ empty ``acc`` value is allowed (unpublished sequences); the ``func`` key is
72
+ intentionally optional and defaults to ``""``. Anything else delegates to
73
+ ``gapit.db.parse_db_header``.
74
+ """
75
+ if not is_gapit_header(seqid):
76
+ return parse_db_header(seqid, default_db)
77
+ fields: dict[str, str] = {}
78
+ for segment in seqid.split("|")[1:]:
79
+ key, sep, raw_value = segment.partition("=")
80
+ if not sep:
81
+ raise _malformed(seqid, "segment_without_key")
82
+ if not key:
83
+ raise _malformed(seqid, "empty_key")
84
+ if key in fields:
85
+ raise _malformed(seqid, f"duplicate_key:{key}")
86
+ fields[key] = _decode_value(raw_value, key=key, seqid=seqid)
87
+ gene = fields.get("gene", "")
88
+ if not gene:
89
+ raise _malformed(seqid, "missing_gene")
90
+ if "db" not in fields:
91
+ raise _malformed(seqid, "missing_db")
92
+ if "acc" not in fields:
93
+ raise _malformed(seqid, "missing_acc")
94
+ return DbHeader(
95
+ database=fields["db"] or default_db,
96
+ gene=gene,
97
+ accession=fields["acc"],
98
+ # func was encoded as one ;-joined value; the decoded string already is
99
+ # the abricate display form "a;b".
100
+ function=fields.get("func", ""),
101
+ )
102
+
103
+
104
+ def _decode_value(raw: str, *, key: str, seqid: str) -> str:
105
+ """Percent-decode one raw value; every ``%`` must precede two hex digits."""
106
+ if _BAD_ESCAPE_RE.search(raw):
107
+ raise _malformed(seqid, f"invalid_percent_escape:{key}")
108
+ return _ESCAPE_RE.sub(lambda match: chr(int(match.group(1), 16)), raw)
109
+
110
+
111
+ def _malformed(seqid: str, reason: str) -> DatabaseError:
112
+ """Build the strict-decode failure (exit 4) with seqid + reason context."""
113
+ return DatabaseError(
114
+ f"malformed gapit/v1 sequence header: {reason}",
115
+ code="HEADER_MALFORMED",
116
+ context={"seqid": seqid, "reason": reason},
117
+ )
gapit/dispatch.py ADDED
@@ -0,0 +1,31 @@
1
+ """Shared CLI plumbing: error-envelope dispatch and the ``--datadir`` option.
2
+
3
+ Imported by every command module (cli.py, cmd_*.py); imports nothing from
4
+ gapit except :mod:`gapit.errors`, so it can never participate in a cycle.
5
+ """
6
+
7
+ from collections.abc import Callable
8
+ from pathlib import Path
9
+ from typing import Annotated
10
+
11
+ import typer
12
+
13
+ from gapit.errors import GapitError, render_error
14
+
15
+ Datadir = Annotated[
16
+ Path | None,
17
+ typer.Option(
18
+ "--datadir",
19
+ help="Database directory (default: $GAPIT_DATADIR, then ~/.local/share/gapit/db).",
20
+ ),
21
+ ]
22
+
23
+
24
+ def dispatch(action: Callable[[], None]) -> None:
25
+ """Run a command body; any failure renders the gapit.error/1 envelope on
26
+ stderr and exits with the documented code (UNEXPECTED/1 for non-GapitError)."""
27
+ try:
28
+ action()
29
+ except Exception as exc:
30
+ typer.echo(render_error(exc), err=True)
31
+ raise typer.Exit(code=exc.exit_code if isinstance(exc, GapitError) else 1) from exc
gapit/errors.py ADDED
@@ -0,0 +1,87 @@
1
+ """Typed errors, shared raise-helpers, and the JSON error envelope
2
+ (gapit.error/1).
3
+
4
+ Exit-code contract (AGENTS.md §5): 2 usage, 3 missing dependency, 4 db error,
5
+ 5 input error, 1 unexpected. Failures render as a one-line JSON envelope on
6
+ stderr via ``render_error``.
7
+ """
8
+
9
+ from pathlib import Path
10
+ from typing import Literal, NoReturn
11
+
12
+ from pydantic import BaseModel, ConfigDict, Field
13
+
14
+
15
+ class GapitError(Exception):
16
+ """Base class for all typed gapit errors."""
17
+
18
+ code: str
19
+ exit_code: int = 1
20
+
21
+ def __init__(self, message: str, *, code: str, context: dict[str, str] | None = None) -> None:
22
+ super().__init__(message)
23
+ self.code = code
24
+ self.context: dict[str, str] = context if context is not None else {}
25
+
26
+
27
+ class UsageError(GapitError):
28
+ """Invalid command-line usage."""
29
+
30
+ exit_code = 2
31
+
32
+
33
+ class DependencyError(GapitError):
34
+ """An external binary (BLAST+, minimap2) is missing from PATH."""
35
+
36
+ exit_code = 3
37
+
38
+
39
+ class DatabaseError(GapitError):
40
+ """A datadir/database is missing, unindexed, or failed to build."""
41
+
42
+ exit_code = 4
43
+
44
+
45
+ class InputError(GapitError):
46
+ """User-supplied input is malformed."""
47
+
48
+ exit_code = 5
49
+
50
+
51
+ def usage_fail(message: str) -> NoReturn:
52
+ """Raise a usage error (gapit.error/1 envelope, exit 2)."""
53
+ raise UsageError(message, code="USAGE_ERROR")
54
+
55
+
56
+ def ensure_input_file(path: Path, what: str = "input file") -> None:
57
+ """Raise INPUT_NOT_FOUND for a missing/unreadable input path; ``what``
58
+ names the kind in the message (e.g. "reads file")."""
59
+ if not path.is_file():
60
+ raise InputError(
61
+ f"{what} not found or unreadable: {path}",
62
+ code="INPUT_NOT_FOUND",
63
+ context={"file": str(path)},
64
+ )
65
+
66
+
67
+ class ErrorEnvelope(BaseModel, frozen=True):
68
+ """The gapit.error/1 stderr envelope."""
69
+
70
+ model_config = ConfigDict(populate_by_name=True)
71
+
72
+ schema_name: Literal["gapit.error/1"] = Field(default="gapit.error/1", alias="schema")
73
+ code: str
74
+ message: str
75
+ context: dict[str, str]
76
+
77
+
78
+ def render_error(exc: BaseException) -> str:
79
+ """One-line compact JSON envelope for any failure (non-GapitError becomes
80
+ UNEXPECTED, exit 1)."""
81
+ if isinstance(exc, GapitError):
82
+ envelope = ErrorEnvelope(code=exc.code, message=str(exc), context=exc.context)
83
+ else:
84
+ envelope = ErrorEnvelope(
85
+ code="UNEXPECTED", message=f"{type(exc).__name__}: {exc}", context={}
86
+ )
87
+ return envelope.model_dump_json(by_alias=True)
gapit/fasta.py ADDED
@@ -0,0 +1,123 @@
1
+ """Streaming FASTA reader (plain, gzip, bzip2)."""
2
+
3
+ import bz2
4
+ import gzip
5
+ from collections.abc import Iterator
6
+ from pathlib import Path
7
+ from typing import IO
8
+
9
+ from pydantic import BaseModel
10
+
11
+ from gapit.errors import InputError
12
+
13
+
14
+ class FastaRecord(BaseModel, frozen=True):
15
+ """One FASTA record: ``id`` is the first whitespace-delimited header token."""
16
+
17
+ id: str
18
+ description: str
19
+ sequence: str
20
+
21
+
22
+ def open_text(path: Path) -> IO[str]:
23
+ """Open a sequence file as UTF-8 text, transparently decompressing .gz/.bz2.
24
+
25
+ Shared by the FASTA readers and seqconvert's normalizer; the suffix
26
+ (not content) picks the decompressor, matching the any2fasta CLI contract.
27
+ """
28
+ if path.name.endswith(".gz"):
29
+ return gzip.open(path, "rt", encoding="utf-8")
30
+ if path.name.endswith(".bz2"):
31
+ return bz2.open(path, "rt", encoding="utf-8")
32
+ return path.open("rt", encoding="utf-8")
33
+
34
+
35
+ def _finalize(path: Path, pending: tuple[str, str, list[str]]) -> FastaRecord:
36
+ """Turn accumulated (id, description, sequence lines) into a FastaRecord."""
37
+ seqid, description, chunks = pending
38
+ if not chunks:
39
+ raise InputError(
40
+ f"{path}: record {seqid!r} has an empty sequence",
41
+ code="INVALID_FASTA",
42
+ context={"file": str(path)},
43
+ )
44
+ return FastaRecord(id=seqid, description=description, sequence="".join(chunks))
45
+
46
+
47
+ def iter_fasta(path: Path) -> Iterator[FastaRecord]:
48
+ """Stream FastaRecords from a FASTA file (plain/.gz/.bz2, UTF-8).
49
+
50
+ Raises InputError on content before the first ``>`` header, on a record
51
+ with an empty sequence, or on an unreadable/truncated stream (a .gz/.bz2
52
+ cut short raises EOFError mid-read, not an OSError); an empty file yields
53
+ zero records.
54
+ """
55
+ try:
56
+ with open_text(path) as handle:
57
+ # pending = (id, description, sequence lines) of the record being read
58
+ pending: tuple[str, str, list[str]] | None = None
59
+ for lineno, raw_line in enumerate(handle, start=1):
60
+ line = raw_line.strip()
61
+ if not line:
62
+ continue
63
+ if not line.startswith(">"):
64
+ if pending is None:
65
+ raise InputError(
66
+ f"{path}: content before first '>' header at line {lineno}",
67
+ code="INVALID_FASTA",
68
+ context={"file": str(path)},
69
+ )
70
+ pending[2].append(line)
71
+ continue
72
+ if pending is not None:
73
+ yield _finalize(path, pending)
74
+ parts = line[1:].strip().split(maxsplit=1)
75
+ pending = (
76
+ parts[0] if parts else "",
77
+ parts[1] if len(parts) > 1 else "",
78
+ [],
79
+ )
80
+ if pending is not None:
81
+ yield _finalize(path, pending)
82
+ except (OSError, UnicodeDecodeError, EOFError) as exc:
83
+ raise InputError(
84
+ f"{path}: could not read input: {exc}",
85
+ code="INVALID_FASTA",
86
+ context={"file": str(path)},
87
+ ) from exc
88
+
89
+
90
+ def iter_fasta_headers(path: Path) -> Iterator[tuple[str, str]]:
91
+ """Stream (id, description) pairs from FASTA headers only.
92
+
93
+ Sequence lines are skipped, not accumulated — the cheap iterator for
94
+ header-only consumers (e.g. product lookups on a large db). Unlike
95
+ iter_fasta, the empty-sequence InputError does NOT apply: no sequence
96
+ check is performed because sequences are never read. Content before the
97
+ first ``>`` header and an unreadable/truncated stream still raise
98
+ InputError; header parsing (id is the first whitespace token, description
99
+ the rest) matches iter_fasta; an empty file yields nothing.
100
+ """
101
+ try:
102
+ with open_text(path) as handle:
103
+ seen_header = False
104
+ for lineno, raw_line in enumerate(handle, start=1):
105
+ line = raw_line.strip()
106
+ if line.startswith(">"):
107
+ seen_header = True
108
+ parts = line[1:].strip().split(maxsplit=1)
109
+ yield (parts[0] if parts else "", parts[1] if len(parts) > 1 else "")
110
+ elif not line or seen_header:
111
+ continue
112
+ else:
113
+ raise InputError(
114
+ f"{path}: content before first '>' header at line {lineno}",
115
+ code="INVALID_FASTA",
116
+ context={"file": str(path)},
117
+ )
118
+ except (OSError, UnicodeDecodeError, EOFError) as exc:
119
+ raise InputError(
120
+ f"{path}: could not read input: {exc}",
121
+ code="INVALID_FASTA",
122
+ context={"file": str(path)},
123
+ ) from exc
@@ -0,0 +1 @@
1
+ """Output formatters for gapit reports (TSV/CSV today; JSON and Markdown in Phase 4)."""