gapit 0.2.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. gapit/__init__.py +3 -0
  2. gapit/blast.py +239 -0
  3. gapit/cli.py +128 -0
  4. gapit/cmd_db.py +113 -0
  5. gapit/cmd_db_build.py +72 -0
  6. gapit/cmd_db_install.py +126 -0
  7. gapit/cmd_db_outdated.py +51 -0
  8. gapit/cmd_db_search.py +66 -0
  9. gapit/cmd_screen.py +214 -0
  10. gapit/cmd_summary.py +65 -0
  11. gapit/config.py +51 -0
  12. gapit/data/snapshots/card.tar.gz +0 -0
  13. gapit/data/snapshots/vfdb.tar.gz +0 -0
  14. gapit/db.py +226 -0
  15. gapit/db_build_ops.py +210 -0
  16. gapit/db_ops.py +128 -0
  17. gapit/db_query_ops.py +252 -0
  18. gapit/dbbuild.py +207 -0
  19. gapit/dbcodec.py +117 -0
  20. gapit/dispatch.py +31 -0
  21. gapit/errors.py +87 -0
  22. gapit/fasta.py +123 -0
  23. gapit/formats/__init__.py +1 -0
  24. gapit/formats/json.py +309 -0
  25. gapit/formats/md.py +190 -0
  26. gapit/formats/schemas.py +30 -0
  27. gapit/formats/summary.py +103 -0
  28. gapit/formats/tsv.py +45 -0
  29. gapit/hits.py +107 -0
  30. gapit/mcp.py +158 -0
  31. gapit/mcp_schemas.py +123 -0
  32. gapit/mcp_tools.py +289 -0
  33. gapit/minimap.py +20 -0
  34. gapit/minimap2_run.py +114 -0
  35. gapit/paf.py +115 -0
  36. gapit/proctools.py +24 -0
  37. gapit/providers/__init__.py +39 -0
  38. gapit/providers/argannot.py +94 -0
  39. gapit/providers/bacmet2.py +59 -0
  40. gapit/providers/card.py +150 -0
  41. gapit/providers/common.py +245 -0
  42. gapit/providers/ecoh.py +63 -0
  43. gapit/providers/ecoli_vf.py +74 -0
  44. gapit/providers/megares.py +71 -0
  45. gapit/providers/ncbi.py +103 -0
  46. gapit/providers/plasmidfinder.py +69 -0
  47. gapit/providers/resfinder.py +123 -0
  48. gapit/providers/snapshots.py +119 -0
  49. gapit/providers/upec_expec_vf.py +85 -0
  50. gapit/providers/vfdb.py +92 -0
  51. gapit/providers/victors.py +109 -0
  52. gapit/py.typed +0 -0
  53. gapit/reads.py +221 -0
  54. gapit/records.py +152 -0
  55. gapit/report.py +25 -0
  56. gapit/screening.py +145 -0
  57. gapit/screening_reads.py +255 -0
  58. gapit/seqconvert.py +203 -0
  59. gapit/summary.py +151 -0
  60. gapit-0.2.2.dist-info/METADATA +183 -0
  61. gapit-0.2.2.dist-info/RECORD +64 -0
  62. gapit-0.2.2.dist-info/WHEEL +4 -0
  63. gapit-0.2.2.dist-info/entry_points.txt +3 -0
  64. gapit-0.2.2.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,255 @@
1
+ """The minimap2 screening use-cases (SPEC.md §10): ``--r1``/``--r2`` reads and
2
+ positional assemblies via ``--aligner minimap2``.
3
+
4
+ Split from screening.py so each use-case module stays under the 250 pure-LOC
5
+ ceiling; screening.py keeps the blastn contig pipeline and shared helpers.
6
+ """
7
+
8
+ from datetime import UTC, datetime
9
+ from pathlib import Path
10
+
11
+ import typer
12
+
13
+ from gapit import config
14
+ from gapit.errors import ensure_input_file, usage_fail
15
+ from gapit.formats.json import render_reads2_json, render_reads_json
16
+ from gapit.formats.md import render_reads2_markdown, render_reads_markdown
17
+ from gapit.reads import (
18
+ ReadFileKind,
19
+ ReadsParams,
20
+ ReadTypeEnum,
21
+ detect_read_kind,
22
+ screen_reads,
23
+ )
24
+ from gapit.screening import AlignerEnum, OutputFormat, find_database
25
+
26
+
27
+ def _pair_read_lanes(r1: list[Path], r2: list[Path] | None) -> list[tuple[Path, Path | None]]:
28
+ """Pair per-lane --r1/--r2 file lists (r2 None = single-end)."""
29
+ if r2 is not None and len(r2) != len(r1):
30
+ usage_fail(f"--r2 has {len(r2)} files but --r1 has {len(r1)} (lanes must pair up)")
31
+ if r2 is None:
32
+ return [(path, None) for path in r1]
33
+ return list(zip(r1, r2, strict=True))
34
+
35
+
36
+ def _resolve_read_preset(
37
+ lanes: list[tuple[Path, Path | None]], read_type: ReadTypeEnum | None, quiet: bool
38
+ ) -> ReadTypeEnum:
39
+ """Detect every input's kind from content and resolve the minimap2 preset:
40
+ assembly FASTA forces map-ont (one stderr note unless quiet/explicit),
41
+ FASTQ keeps sr unless a preset was given. Mixed kinds and FASTA paired-end
42
+ are usage errors."""
43
+ r1_kinds = {detect_read_kind(r1_path) for r1_path, _ in lanes}
44
+ if len(r1_kinds) > 1:
45
+ usage_fail("mixed FASTA and FASTQ inputs")
46
+ r2_kinds = {detect_read_kind(r2_path) for _, r2_path in lanes if r2_path is not None}
47
+ if r2_kinds and ReadFileKind.fasta in r1_kinds | r2_kinds:
48
+ usage_fail("paired-end requires FASTQ")
49
+ if ReadFileKind.fasta not in r1_kinds:
50
+ return read_type if read_type is not None else ReadTypeEnum.sr
51
+ if read_type is not None and read_type is not ReadTypeEnum.map_ont:
52
+ usage_fail("assembly FASTA requires map-ont")
53
+ if read_type is None and not quiet:
54
+ typer.echo("assembly FASTA detected; using map-ont", err=True)
55
+ return ReadTypeEnum.map_ont
56
+
57
+
58
+ def _reject_blastn_thresholds(minid: float, mincov: float, mode: str) -> None:
59
+ """--minid/--mincov are blastn-only; every minimap2 entry point rejects
60
+ them instead of silently ignoring them (reads thresholds have their own
61
+ flags). ``mode`` names the invocation in the frozen message."""
62
+ if minid != 80.0 or mincov != 80.0:
63
+ usage_fail(
64
+ f"--minid/--mincov apply to blastn only; use --min-identity/--min-breadth with {mode}"
65
+ )
66
+
67
+
68
+ def _validate_reads_usage(
69
+ output_format: OutputFormat | None,
70
+ min_breadth: float,
71
+ min_identity: float,
72
+ min_mapq: int,
73
+ threads: int,
74
+ ) -> None:
75
+ """Usage gates shared by both minimap2 entry points, in the frozen order
76
+ (format first, then thresholds), before any file is touched."""
77
+ if output_format is OutputFormat.tsv or output_format is OutputFormat.csv:
78
+ usage_fail("--format tsv|csv is not available in reads mode (use json or md)")
79
+ if not 0.0 <= min_breadth <= 100.0:
80
+ usage_fail(f"--min-breadth must be in [0, 100]: got {min_breadth}")
81
+ if not 0.0 <= min_identity <= 100.0:
82
+ usage_fail(f"--min-identity must be in [0, 100]: got {min_identity}")
83
+ if min_mapq < 0:
84
+ usage_fail(f"--min-mapq must be >= 0: got {min_mapq}")
85
+ if threads < 1:
86
+ usage_fail(f"--threads must be >= 1: got {threads}")
87
+
88
+
89
+ def _screen_lanes(
90
+ lanes: list[tuple[Path, Path | None]],
91
+ db_name: str,
92
+ datadir: Path | None,
93
+ read_type: ReadTypeEnum | None,
94
+ min_breadth: float,
95
+ min_identity: float,
96
+ min_mapq: int,
97
+ threads: int,
98
+ output_format: OutputFormat | None,
99
+ quiet: bool,
100
+ debug: bool,
101
+ ) -> str:
102
+ """Minimap2 engine core shared by both entry points: preset resolution,
103
+ screening, rendering; json is the default format (SPEC.md §10). Either
104
+ reads/2 threshold on selects the gapit.reads/2 document; both off keep
105
+ gapit.reads/1 byte-identical. Returns the rendered output for the caller
106
+ to echo."""
107
+ resolved = _resolve_read_preset(lanes, read_type, quiet)
108
+ database = find_database(config.resolve_datadir(datadir), db_name)
109
+ read_files = [r1_path for r1_path, _ in lanes] + [
110
+ r2_path for _, r2_path in lanes if r2_path is not None
111
+ ]
112
+ read_list = ", ".join(str(path) for path in read_files)
113
+ if not quiet:
114
+ typer.echo(f"Screening reads: {read_list}", err=True)
115
+ report = screen_reads(
116
+ lanes,
117
+ database,
118
+ read_type=resolved.value,
119
+ min_breadth=min_breadth,
120
+ threads=threads,
121
+ debug=debug,
122
+ min_identity=min_identity,
123
+ min_mapq=min_mapq,
124
+ )
125
+ present = sum(1 for gene in report.genes if gene.present)
126
+ if not quiet:
127
+ typer.echo(f"Detected {present} present genes in {read_list}", err=True)
128
+ params = ReadsParams(
129
+ db=db_name,
130
+ read_type=resolved.value,
131
+ min_breadth=min_breadth,
132
+ threads=threads,
133
+ min_identity=min_identity,
134
+ min_mapq=min_mapq,
135
+ )
136
+ now = datetime.now(UTC)
137
+ reads2 = min_identity > 0.0 or min_mapq > 0
138
+ if reads2:
139
+ output = (
140
+ render_reads2_markdown([report], params, now=now)
141
+ if output_format is OutputFormat.md
142
+ else render_reads2_json([report], params, now=now)
143
+ )
144
+ else:
145
+ output = (
146
+ render_reads_markdown([report], params, now=now)
147
+ if output_format is OutputFormat.md
148
+ else render_reads_json([report], params, now=now)
149
+ )
150
+ return output
151
+
152
+
153
+ def run_screen_reads(
154
+ r1: list[Path],
155
+ r2: list[Path] | None,
156
+ db_name: str,
157
+ datadir: Path | None,
158
+ read_type: ReadTypeEnum | None,
159
+ min_breadth: float,
160
+ min_identity: float,
161
+ min_mapq: int,
162
+ threads: int,
163
+ output_format: OutputFormat | None,
164
+ quiet: bool,
165
+ debug: bool = False,
166
+ aligner: AlignerEnum | None = None,
167
+ minid: float = 80.0,
168
+ mincov: float = 80.0,
169
+ ) -> str:
170
+ """Screen FASTQ reads or assembly FASTA given as already-split per-lane
171
+ --r1/--r2 file lists (the CLI owns the comma-splitting; MCP passes arrays
172
+ natively, so commas in filenames survive). Per-lane minimap2, sample-level
173
+ union; json is the default format (SPEC.md §10). A nonzero
174
+ --min-identity/--min-mapq turns on gapit.reads/2 alignment filtering.
175
+ Returns the rendered output."""
176
+ if aligner is AlignerEnum.blastn:
177
+ usage_fail("--aligner blastn is not available for --r1/--r2 reads input")
178
+ _validate_reads_usage(output_format, min_breadth, min_identity, min_mapq, threads)
179
+ _reject_blastn_thresholds(minid, mincov, "--r1/--r2")
180
+ lanes = _pair_read_lanes(r1, r2)
181
+ for path in [r1_path for r1_path, _ in lanes] + [
182
+ r2_path for _, r2_path in lanes if r2_path is not None
183
+ ]:
184
+ ensure_input_file(path, "reads file")
185
+ return _screen_lanes(
186
+ lanes,
187
+ db_name,
188
+ datadir,
189
+ read_type,
190
+ min_breadth,
191
+ min_identity,
192
+ min_mapq,
193
+ threads,
194
+ output_format,
195
+ quiet,
196
+ debug,
197
+ )
198
+
199
+
200
+ def run_screen_assemblies(
201
+ files: list[Path] | None,
202
+ fofn: Path | None,
203
+ db_name: str,
204
+ datadir: Path | None,
205
+ read_type: ReadTypeEnum | None,
206
+ min_breadth: float,
207
+ min_identity: float,
208
+ min_mapq: int,
209
+ threads: int,
210
+ jobs: int,
211
+ noheader: bool,
212
+ nopath: bool,
213
+ output_format: OutputFormat | None,
214
+ quiet: bool,
215
+ debug: bool = False,
216
+ minid: float = 80.0,
217
+ mincov: float = 80.0,
218
+ ) -> str:
219
+ """Screen positional assembly FASTA file(s) with the minimap2 engine
220
+ (--aligner minimap2): every input must be FASTA(.gz) content — FASTQ
221
+ content is a usage error, undetectable content keeps the typed input
222
+ error. Preset resolution and output follow the reads contract (SPEC §10);
223
+ the blastn-engine-only flags --fofn/--jobs/--noheader/--nopath and the
224
+ blastn thresholds --minid/--mincov are rejected here instead of silently
225
+ ignored. A nonzero --min-identity/--min-mapq turns on gapit.reads/2
226
+ filtering. Returns the rendered output."""
227
+ _validate_reads_usage(output_format, min_breadth, min_identity, min_mapq, threads)
228
+ if fofn is not None:
229
+ usage_fail("--fofn is not available with --aligner minimap2")
230
+ if jobs != 1:
231
+ usage_fail("--jobs is not available with --aligner minimap2")
232
+ if noheader:
233
+ usage_fail("--noheader is not available with --aligner minimap2")
234
+ if nopath:
235
+ usage_fail("--nopath is not available with --aligner minimap2")
236
+ _reject_blastn_thresholds(minid, mincov, "--aligner minimap2")
237
+ if not files:
238
+ usage_fail("no input files given (positional FILEs)")
239
+ for path in files:
240
+ ensure_input_file(path)
241
+ if detect_read_kind(path) is not ReadFileKind.fasta:
242
+ usage_fail("minimap2 engine requires FASTA assemblies")
243
+ return _screen_lanes(
244
+ [(path, None) for path in files],
245
+ db_name,
246
+ datadir,
247
+ read_type,
248
+ min_breadth,
249
+ min_identity,
250
+ min_mapq,
251
+ threads,
252
+ output_format,
253
+ quiet,
254
+ debug,
255
+ )
gapit/seqconvert.py ADDED
@@ -0,0 +1,203 @@
1
+ """Native input normalization: FASTA/FASTQ/GenBank/EMBL (.gz/.bz2) → FASTA lines.
2
+
3
+ Replaces the external ``any2fasta -q -u <file>`` stage of the screening
4
+ pipeline (blast.py). Semantics extracted from any2fasta 0.8.1 (Perl,
5
+ bioconda; ``.pixi/envs/default/bin/any2fasta``) invoked with ``-q -u`` only —
6
+ no ``-n`` (N-purification), ``-l``, ``-g`` (VERSION), or ``-s`` (description
7
+ stripping). Perl line references below are that file.
8
+
9
+ - Detection (L89-99, L136-153): first line of the decompressed stream, in
10
+ order GENBANK ``^LOCUS\\h``, EMBL ``^ID\\h``, FASTA ``^>\\S``, FASTQ
11
+ ``^@\\S`` (``\\h`` = space/tab; ``\\S`` = non-whitespace after the marker).
12
+ Empty input dies with "The input appears to be empty" (L131-133); an
13
+ unrecognized first line dies with "Unfamilar format with first line: ..."
14
+ (L153 — typo theirs, kept for message parity with the retired binary).
15
+ - purify_dna (L164-170): with ``-u`` only uc() — sequences uppercased, nothing
16
+ else touched. purify_id (L174-181) is a no-op without ``-s``: headers pass
17
+ through verbatim.
18
+ - FASTA (L185-200): header lines printed verbatim; sequence lines uppercased;
19
+ blank/whitespace-only lines skipped entirely (L189).
20
+ - FASTQ (L204-215): strict 4-line stride. Header = the ``@`` line minus its
21
+ leading ``@``, verbatim (id + description); sequence = uc() of line 2;
22
+ ``+``/quality lines dropped. A trailing lone ``@`` header emits nothing
23
+ (loop guard ``$i < $#lines``, L208).
24
+ - GenBank (L255-296): id = first token after ``LOCUS\\s+`` (L285-287).
25
+ DEFINITION is never read — records carry no description. VERSION overrides
26
+ the id only under ``-g`` (L290-292; unused). ORIGIN data lines: drop the
27
+ first 10 columns (coordinate prefix, L279), then strip whitespace (L280) —
28
+ digits past column 10 are kept. Records flush at ``//`` (L264-269); a
29
+ record never terminated by ``//`` is dropped.
30
+ - EMBL (L300-339): id = text after ``ID\\s+`` up to the first ``;`` (L330-331,
31
+ not trimmed). DE is never read. SQ data lines: strip all whitespace AND
32
+ digits (L325). Flush at ``//``.
33
+
34
+ Deliberate divergences, invisible to BLAST: sequence is re-wrapped at 60
35
+ columns (the Perl reuses input wrapping) and input is read with universal
36
+ newlines (the Perl preserves ``\\r``). Parsed records — id, description,
37
+ uppercased sequence — are identical, so the blastn query is unchanged and
38
+ abricate parity is unaffected. gapit owns no any2fasta dependency; the
39
+ differential suite runs the real binary from the parity env's PATH, where
40
+ abricate provides it transitively (tests/test_seqconvert_differential.py).
41
+ """
42
+
43
+ import re
44
+ from collections.abc import Iterator
45
+ from enum import Enum
46
+ from pathlib import Path
47
+ from typing import IO
48
+
49
+ from gapit.errors import InputError
50
+ from gapit.fasta import open_text
51
+
52
+ _WRAP = 60
53
+
54
+
55
+ class SeqFormat(Enum):
56
+ """Input formats detected from the first decompressed line."""
57
+
58
+ fasta = "fasta"
59
+ fastq = "fastq"
60
+ genbank = "genbank"
61
+ embl = "embl"
62
+
63
+
64
+ def _tagged(line: str, tag: str) -> bool:
65
+ """``^tag\\h`` — tag at column 1 followed by one space or tab."""
66
+ return line.startswith(tag) and line[len(tag) : len(tag) + 1] in (" ", "\t")
67
+
68
+
69
+ def detect_format(path: Path) -> SeqFormat:
70
+ """Content-sniff the first decompressed line (reads.py detect_read_kind style)."""
71
+ try:
72
+ with open_text(path) as handle:
73
+ first = handle.readline()
74
+ except (OSError, UnicodeDecodeError, EOFError) as exc:
75
+ raise InputError(
76
+ f"could not read input: {exc}",
77
+ code="INVALID_INPUT",
78
+ context={"file": str(path)},
79
+ ) from exc
80
+ if not first:
81
+ raise InputError(
82
+ "The input appears to be empty",
83
+ code="INVALID_INPUT",
84
+ context={"file": str(path)},
85
+ )
86
+ line = first.rstrip("\r\n")
87
+ if _tagged(line, "LOCUS"):
88
+ return SeqFormat.genbank
89
+ if _tagged(line, "ID"):
90
+ return SeqFormat.embl
91
+ if len(line) > 1 and line[0] == ">" and not line[1].isspace():
92
+ return SeqFormat.fasta
93
+ if len(line) > 1 and line[0] == "@" and not line[1].isspace():
94
+ return SeqFormat.fastq
95
+ raise InputError(
96
+ f"Unfamilar format with first line: {line}",
97
+ code="INVALID_INPUT",
98
+ context={"file": str(path)},
99
+ )
100
+
101
+
102
+ def _iter_fasta(handle: IO[str]) -> Iterator[tuple[str, str]]:
103
+ """(header, sequence) pairs; headers verbatim, blank lines skipped."""
104
+ header: str | None = None
105
+ chunks: list[str] = []
106
+ for raw in handle:
107
+ line = raw.rstrip("\n")
108
+ if not line.strip():
109
+ continue
110
+ if line.startswith(">"):
111
+ if header is not None:
112
+ yield (header, "".join(chunks).upper())
113
+ header, chunks = line[1:], []
114
+ else:
115
+ chunks.append(line)
116
+ if header is not None:
117
+ yield (header, "".join(chunks).upper())
118
+
119
+
120
+ def _iter_fastq(handle: IO[str]) -> Iterator[tuple[str, str]]:
121
+ """Stride-4 records; a trailing lone '@' header emits nothing (perl guard)."""
122
+ lines = iter(handle)
123
+ for header_line in lines:
124
+ sequence_line = next(lines, None)
125
+ if sequence_line is None:
126
+ return
127
+ yield (header_line.rstrip("\n")[1:], sequence_line.rstrip("\n").upper())
128
+ next(lines, None) # '+' separator
129
+ next(lines, None) # quality line
130
+
131
+
132
+ def _iter_genbank(handle: IO[str]) -> Iterator[tuple[str, str]]:
133
+ """LOCUS-name ids without description; ORIGIN columns 11+ minus whitespace."""
134
+ acc = ""
135
+ chunks: list[str] = []
136
+ in_seq = False
137
+ for raw in handle:
138
+ line = raw.rstrip("\n")
139
+ if line.startswith("//"):
140
+ yield (acc, "".join(chunks).upper())
141
+ acc, chunks, in_seq = "", [], False
142
+ elif line.startswith("ORIGIN"):
143
+ in_seq = True
144
+ elif in_seq:
145
+ chunks.append(re.sub(r"\s+", "", line[10:]))
146
+ elif (match := re.match(r"LOCUS\s+(\S+)", line)) is not None:
147
+ acc = match.group(1)
148
+
149
+
150
+ def _iter_embl(handle: IO[str]) -> Iterator[tuple[str, str]]:
151
+ """ID-before-';' ids without description; SQ lines minus whitespace+digits."""
152
+ acc = ""
153
+ chunks: list[str] = []
154
+ in_seq = False
155
+ for raw in handle:
156
+ line = raw.rstrip("\n")
157
+ if line.startswith("//"):
158
+ yield (acc, "".join(chunks).upper())
159
+ acc, chunks, in_seq = "", [], False
160
+ elif re.match(r"SQ\s", line) is not None:
161
+ in_seq = True
162
+ elif in_seq:
163
+ chunks.append(re.sub(r"[\s\d]", "", line))
164
+ elif (match := re.match(r"ID\s+([^;]+)", line)) is not None:
165
+ acc = match.group(1)
166
+
167
+
168
+ def _records(fmt: SeqFormat, handle: IO[str]) -> Iterator[tuple[str, str]]:
169
+ records: Iterator[tuple[str, str]]
170
+ match fmt:
171
+ case SeqFormat.fasta:
172
+ records = _iter_fasta(handle)
173
+ case SeqFormat.fastq:
174
+ records = _iter_fastq(handle)
175
+ case SeqFormat.genbank:
176
+ records = _iter_genbank(handle)
177
+ case SeqFormat.embl:
178
+ records = _iter_embl(handle)
179
+ return records
180
+
181
+
182
+ def to_fasta_lines(path: Path, fmt: SeqFormat | None = None) -> Iterator[str]:
183
+ """Yield complete FASTA lines for one input file: one verbatim header line
184
+ per record, then its uppercased sequence wrapped at 60 columns.
185
+
186
+ ``fmt`` may be pre-supplied to reuse a sniff (blast.run_blastn's debug
187
+ echo shares one detection). Raises InputError (INVALID_INPUT) on unreadable,
188
+ empty, or unrecognized input — the retired any2fasta's fatal path.
189
+ """
190
+ if fmt is None:
191
+ fmt = detect_format(path)
192
+ try:
193
+ with open_text(path) as handle:
194
+ for header, sequence in _records(fmt, handle):
195
+ yield f">{header}\n"
196
+ for offset in range(0, len(sequence), _WRAP):
197
+ yield f"{sequence[offset : offset + _WRAP]}\n"
198
+ except (OSError, UnicodeDecodeError, EOFError) as exc:
199
+ raise InputError(
200
+ f"could not read input: {exc}",
201
+ code="INVALID_INPUT",
202
+ context={"file": str(path)},
203
+ ) from exc
gapit/summary.py ADDED
@@ -0,0 +1,151 @@
1
+ """Summary mode core: parse abricate-format report tables into a gene matrix.
2
+
3
+ Mirrors abricate 1.4.0 ``summary_table`` (SPEC.md §6) exactly where it is
4
+ defined, and replaces its silent-undef edges with typed InputErrors
5
+ (``SUMMARY_MALFORMED``) — a documented [gapit-extension] divergence.
6
+ """
7
+
8
+ from collections.abc import Callable
9
+ from pathlib import Path, PurePath
10
+ from typing import Literal
11
+
12
+ from pydantic import BaseModel
13
+
14
+ from gapit.errors import InputError
15
+
16
+ FIELDSEP = ";"
17
+ ABSENT = "."
18
+
19
+ Warn = Callable[[str], None]
20
+
21
+
22
+ class SummaryParams(BaseModel, frozen=True):
23
+ """Summary parameters in effect (metric + path display)."""
24
+
25
+ identity: bool = False
26
+ nopath: bool = False
27
+
28
+ @property
29
+ def metric(self) -> Literal["%COVERAGE", "%IDENTITY"]:
30
+ """The report column summarized into cells."""
31
+ return "%IDENTITY" if self.identity else "%COVERAGE"
32
+
33
+
34
+ class SummaryRow(BaseModel, frozen=True):
35
+ """One matrix row: a display label, its distinct-gene count, and the
36
+ per-gene cell values as the original report strings in row order."""
37
+
38
+ file: str
39
+ num_found: int
40
+ cells: dict[str, tuple[str, ...]]
41
+
42
+
43
+ class SummaryMatrix(BaseModel, frozen=True):
44
+ """Canonical in-memory summary result: sorted gene universe + ordered rows."""
45
+
46
+ params: SummaryParams
47
+ genes: tuple[str, ...]
48
+ rows: tuple[SummaryRow, ...]
49
+
50
+
51
+ def _read_report(path: Path) -> str:
52
+ """Read one report table; missing/unreadable/non-UTF-8 is an InputError."""
53
+ if not path.is_file():
54
+ raise InputError(
55
+ f"report file not found or unreadable: {path}",
56
+ code="INPUT_NOT_FOUND",
57
+ context={"file": str(path)},
58
+ )
59
+ try:
60
+ return path.read_text(encoding="utf-8")
61
+ except UnicodeDecodeError as exc:
62
+ raise InputError(
63
+ f"report is not valid UTF-8: {path}",
64
+ code="SUMMARY_MALFORMED",
65
+ context={"file": str(path)},
66
+ ) from exc
67
+
68
+
69
+ def _lines(text: str) -> list[str]:
70
+ """Split like Perl's ``while (<$fh>)``: on \\n, no phantom final line."""
71
+ if not text:
72
+ return []
73
+ parts = text.split("\n")
74
+ if parts[-1] == "":
75
+ parts.pop()
76
+ return parts
77
+
78
+
79
+ def _detect_separator(text: str) -> str:
80
+ """[gapit-extension] auto-detect the table separator per file: tab if the
81
+ first line has any, else comma, else tab (single-column degenerate)."""
82
+ first = _lines(text)[0] if text else ""
83
+ if "\t" in first:
84
+ return "\t"
85
+ if "," in first:
86
+ return ","
87
+ return "\t"
88
+
89
+
90
+ def _malformed(path: Path, line_number: int, detail: str) -> InputError:
91
+ return InputError(
92
+ f"malformed report row: {detail} ({path}:{line_number})",
93
+ code="SUMMARY_MALFORMED",
94
+ context={"file": str(path), "line": str(line_number)},
95
+ )
96
+
97
+
98
+ def build_summary(paths: list[Path], params: SummaryParams, *, warn: Warn) -> SummaryMatrix:
99
+ """Aggregate report table(s) into the summary matrix.
100
+
101
+ Dutch mode (exactly one input) keys rows by each table row's FILE column;
102
+ otherwise rows are keyed by input filename as given. Rows are sorted by
103
+ that key — not by display label — matching upstream. The first row seen
104
+ anywhere is the column map (upstream ``@hdr``); later '#' rows are skipped.
105
+ """
106
+ dutch = len(paths) == 1
107
+ data: dict[str, dict[str, list[str]]] = {}
108
+ seen: set[str] = set()
109
+ indexes: dict[str, int] | None = None
110
+ for path in paths:
111
+ key = str(path)
112
+ if key in seen:
113
+ warn(f"Skipping duplicate file: {key}")
114
+ continue
115
+ seen.add(key)
116
+ text = _read_report(path)
117
+ if not dutch:
118
+ data[key] = {}
119
+ separator = _detect_separator(text)
120
+ for line_number, line in enumerate(_lines(text), start=1):
121
+ columns = line.split(separator)
122
+ if indexes is None:
123
+ # Header-name -> column index with Perl zip semantics (later dups win).
124
+ indexes = dict(zip(columns, range(len(columns)), strict=True))
125
+ if columns[0].startswith("#"):
126
+ continue
127
+ assert indexes is not None # set on the first line, before any data row
128
+ gene_at = indexes.get("GENE")
129
+ metric_at = indexes.get(params.metric)
130
+ if gene_at is None:
131
+ raise _malformed(path, line_number, "header has no GENE column")
132
+ if metric_at is None:
133
+ raise _malformed(path, line_number, f"header has no {params.metric} column")
134
+ if len(columns) <= max(gene_at, metric_at):
135
+ raise _malformed(
136
+ path, line_number, f"expected >= {max(gene_at, metric_at) + 1} columns"
137
+ )
138
+ gene, value = columns[gene_at], columns[metric_at]
139
+ file_key = PurePath(columns[0]).name if params.nopath else columns[0]
140
+ row_key = file_key if dutch else key
141
+ data.setdefault(row_key, {}).setdefault(gene, []).append(value)
142
+ genes = tuple(sorted({gene for hits in data.values() for gene in hits}))
143
+ rows = tuple(
144
+ SummaryRow(
145
+ file=PurePath(row_key).name if params.nopath else row_key,
146
+ num_found=len(data[row_key]),
147
+ cells={gene: tuple(data[row_key][gene]) for gene in genes if gene in data[row_key]},
148
+ )
149
+ for row_key in sorted(data)
150
+ )
151
+ return SummaryMatrix(params=params, genes=genes, rows=rows)