gapit 0.2.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. gapit/__init__.py +3 -0
  2. gapit/blast.py +239 -0
  3. gapit/cli.py +128 -0
  4. gapit/cmd_db.py +113 -0
  5. gapit/cmd_db_build.py +72 -0
  6. gapit/cmd_db_install.py +126 -0
  7. gapit/cmd_db_outdated.py +51 -0
  8. gapit/cmd_db_search.py +66 -0
  9. gapit/cmd_screen.py +214 -0
  10. gapit/cmd_summary.py +65 -0
  11. gapit/config.py +51 -0
  12. gapit/data/snapshots/card.tar.gz +0 -0
  13. gapit/data/snapshots/vfdb.tar.gz +0 -0
  14. gapit/db.py +226 -0
  15. gapit/db_build_ops.py +210 -0
  16. gapit/db_ops.py +128 -0
  17. gapit/db_query_ops.py +252 -0
  18. gapit/dbbuild.py +207 -0
  19. gapit/dbcodec.py +117 -0
  20. gapit/dispatch.py +31 -0
  21. gapit/errors.py +87 -0
  22. gapit/fasta.py +123 -0
  23. gapit/formats/__init__.py +1 -0
  24. gapit/formats/json.py +309 -0
  25. gapit/formats/md.py +190 -0
  26. gapit/formats/schemas.py +30 -0
  27. gapit/formats/summary.py +103 -0
  28. gapit/formats/tsv.py +45 -0
  29. gapit/hits.py +107 -0
  30. gapit/mcp.py +158 -0
  31. gapit/mcp_schemas.py +123 -0
  32. gapit/mcp_tools.py +289 -0
  33. gapit/minimap.py +20 -0
  34. gapit/minimap2_run.py +114 -0
  35. gapit/paf.py +115 -0
  36. gapit/proctools.py +24 -0
  37. gapit/providers/__init__.py +39 -0
  38. gapit/providers/argannot.py +94 -0
  39. gapit/providers/bacmet2.py +59 -0
  40. gapit/providers/card.py +150 -0
  41. gapit/providers/common.py +245 -0
  42. gapit/providers/ecoh.py +63 -0
  43. gapit/providers/ecoli_vf.py +74 -0
  44. gapit/providers/megares.py +71 -0
  45. gapit/providers/ncbi.py +103 -0
  46. gapit/providers/plasmidfinder.py +69 -0
  47. gapit/providers/resfinder.py +123 -0
  48. gapit/providers/snapshots.py +119 -0
  49. gapit/providers/upec_expec_vf.py +85 -0
  50. gapit/providers/vfdb.py +92 -0
  51. gapit/providers/victors.py +109 -0
  52. gapit/py.typed +0 -0
  53. gapit/reads.py +221 -0
  54. gapit/records.py +152 -0
  55. gapit/report.py +25 -0
  56. gapit/screening.py +145 -0
  57. gapit/screening_reads.py +255 -0
  58. gapit/seqconvert.py +203 -0
  59. gapit/summary.py +151 -0
  60. gapit-0.2.2.dist-info/METADATA +183 -0
  61. gapit-0.2.2.dist-info/RECORD +64 -0
  62. gapit-0.2.2.dist-info/WHEEL +4 -0
  63. gapit-0.2.2.dist-info/entry_points.txt +3 -0
  64. gapit-0.2.2.dist-info/licenses/LICENSE +21 -0
gapit/hits.py ADDED
@@ -0,0 +1,107 @@
1
+ """Hit processing: the SPEC.md §4 algorithm, in order, nothing more.
2
+
3
+ No interval merging of any kind — upstream reports overlapping genes at
4
+ different query spans, and so do we. Subject ids decode through
5
+ :mod:`gapit.dbcodec`: legacy ``~~~`` ids parse by the frozen db.py rules,
6
+ ``gapit|``-tagged ids by the strict native codec — a malformed native header
7
+ raises DatabaseError ``HEADER_MALFORMED`` (exit 4), never a silent fallback.
8
+ """
9
+
10
+ import re
11
+ from collections.abc import Iterable
12
+ from typing import TYPE_CHECKING, Literal
13
+
14
+ from pydantic import BaseModel
15
+
16
+ from gapit.db import IDSEP
17
+ from gapit.dbcodec import decode_seqid, is_gapit_header
18
+ from gapit.minimap import minimap
19
+
20
+ # Perl: $product =~ s/^\S+\s+// if $product =~ m/~~~/ — strips the leading
21
+ # makeblastdb id token only when followed by whitespace; a bare ~~~id stays.
22
+ # Native gapit| ids get the same substitution: makeblastdb prefixes stitle
23
+ # with the id, and there the prefix is the whitespace-free tagged seqid.
24
+ _LEADING_TOKEN_RE = re.compile(r"^\S+\s+")
25
+
26
+ if TYPE_CHECKING:
27
+ # TYPE_CHECKING-only: blast -> report -> hits would otherwise be a cycle.
28
+ from gapit.blast import BlastRow
29
+
30
+
31
+ class Hit(BaseModel, frozen=True):
32
+ """One surviving hit — raw semantic values only (display formatting is Phase 3)."""
33
+
34
+ sequence: str
35
+ start: int
36
+ end: int
37
+ strand: Literal["+", "-"]
38
+ gene: str
39
+ database: str
40
+ accession: str
41
+ product: str
42
+ function: str
43
+ s_start: int
44
+ s_end: int
45
+ s_len: int
46
+ coverage_map: str
47
+ gap_openings: int
48
+ gaps: int
49
+ identity_pct: float
50
+ coverage_pct: float
51
+
52
+
53
+ def process_rows(rows: Iterable["BlastRow"], *, mincov: float, default_db: str) -> list[Hit]:
54
+ """Turn BLAST rows into hits, strictly in SPEC.md §4 order:
55
+
56
+ 1. minus-strand swap (subject coords only), 2. dedup on
57
+ (qseqid, qstart, qend) — first row wins, key ignores strand, and the key
58
+ is claimed even if the row is later coverage-filtered, 3. coverage filter
59
+ on the unrounded float, 4. header decode (native ``gapit|`` codec or
60
+ legacy ``~~~`` rules), 5. product cleanup.
61
+ """
62
+ hits: list[Hit] = []
63
+ seen: set[tuple[str, int, int]] = set()
64
+ for row in rows:
65
+ # 1. minus-strand normalize: swap sstart/send (query coords untouched)
66
+ if row.sstrand == "minus":
67
+ s_start, s_end = row.send, row.sstart
68
+ else:
69
+ s_start, s_end = row.sstart, row.send
70
+ # 2. dedup on the query span
71
+ key = (row.qseqid, row.qstart, row.qend)
72
+ if key in seen:
73
+ continue
74
+ seen.add(key)
75
+ # 3. coverage filter on the unrounded float
76
+ coverage_pct = 100.0 * (row.length - row.gaps) / row.slen
77
+ if coverage_pct < mincov:
78
+ continue
79
+ # 4. subject id decode: native gapit| codec, legacy ~~~ delegated
80
+ header = decode_seqid(row.sseqid, default_db)
81
+ # 5. product cleanup: n/a fallback, strip ',' and tab, drop leading id token
82
+ product = row.stitle or "n/a"
83
+ product = product.replace(",", "").replace("\t", "")
84
+ if IDSEP in product or is_gapit_header(row.sseqid):
85
+ product = _LEADING_TOKEN_RE.sub("", product, count=1)
86
+ hits.append(
87
+ Hit(
88
+ sequence=row.qseqid,
89
+ start=row.qstart,
90
+ end=row.qend,
91
+ strand="-" if row.sstrand == "minus" else "+",
92
+ gene=header.gene,
93
+ database=header.database,
94
+ accession=header.accession,
95
+ function=header.function,
96
+ product=product,
97
+ s_start=s_start,
98
+ s_end=s_end,
99
+ s_len=row.slen,
100
+ coverage_map=minimap(s_start, s_end, row.slen, row.gapopen),
101
+ gap_openings=row.gapopen,
102
+ gaps=row.gaps,
103
+ identity_pct=row.pident,
104
+ coverage_pct=coverage_pct,
105
+ )
106
+ )
107
+ return hits
gapit/mcp.py ADDED
@@ -0,0 +1,158 @@
1
+ r"""Minimal MCP stdio server: gapit as a tool provider for agents.
2
+
3
+ Hand-rolled on purpose (AGENTS.md §2 keeps the dependency list deliberately
4
+ short): no `mcp` SDK — just the essential MCP stdio behavior, JSON-RPC 2.0,
5
+ one message per line on stdin and one response line on stdout. Limitations:
6
+ single messages only (no batch arrays); non-JSON lines are ignored silently
7
+ (robustness over -32700); JSON lines that fail frame validation get a
8
+ -32600/-32602 error when they carry a request id, so no client ever hangs.
9
+ This module is the PROTOCOL only — frames,
10
+ dispatch, and the serve loop; the tool implementations live in
11
+ :mod:`gapit.mcp_tools` and the tools/list declarations (inputSchemas) in
12
+ :mod:`gapit.mcp_schemas` (screen, summary, schema, db_list, db_fetch,
13
+ db_build, db_search, db_outdated). Tool failures return isError=true with
14
+ the gapit.error/1 envelope as text; stdout is protocol-only.
15
+ """
16
+
17
+ import json
18
+ import sys
19
+ from collections.abc import Iterable
20
+ from typing import Any, TextIO, TypeGuard
21
+
22
+ import typer
23
+ from pydantic import BaseModel, ConfigDict, ValidationError
24
+
25
+ from gapit import __version__
26
+ from gapit.errors import render_error
27
+ from gapit.mcp_schemas import TOOLS
28
+ from gapit.mcp_tools import TOOL_HANDLERS
29
+
30
+ # Parsed JSON-RPC values are the one sanctioned Any boundary (card.py
31
+ # precedent): frame/argument containers stay dict[str, Any] until the
32
+ # per-field checks in mcp_tools pin concrete types.
33
+ JsonRpcId = str | int | float | None
34
+
35
+ PROTOCOL_VERSION = "2025-06-18"
36
+
37
+
38
+ class _Frame(BaseModel, frozen=True):
39
+ """One stdin frame (request or notification); ``id`` presence in
40
+ model_fields_set distinguishes requests, per JSON-RPC 2.0."""
41
+
42
+ model_config = ConfigDict(extra="ignore")
43
+
44
+ method: str | None = None
45
+ params: dict[str, Any] | None = None
46
+ id: JsonRpcId = None
47
+
48
+
49
+ class _ToolCall(BaseModel, frozen=True):
50
+ """tools/call params: the tool ``name`` and its ``arguments`` object."""
51
+
52
+ model_config = ConfigDict(extra="ignore")
53
+
54
+ name: str = ""
55
+ arguments: dict[str, Any] | None = None
56
+
57
+
58
+ def _result(id_value: JsonRpcId, result: dict[str, object]) -> dict[str, object]:
59
+ return {"jsonrpc": "2.0", "id": id_value, "result": result}
60
+
61
+
62
+ def _error(id_value: JsonRpcId, code: int, message: str) -> dict[str, object]:
63
+ return {"jsonrpc": "2.0", "id": id_value, "error": {"code": code, "message": message}}
64
+
65
+
66
+ def _text(text: str, *, is_error: bool) -> dict[str, object]:
67
+ return {"content": [{"type": "text", "text": text}], "isError": is_error}
68
+
69
+
70
+ def _initialize(params: dict[str, Any] | None) -> dict[str, object]:
71
+ version = (params or {}).get("protocolVersion", "")
72
+ requested = version if isinstance(version, str) else ""
73
+ return {
74
+ "protocolVersion": requested or PROTOCOL_VERSION,
75
+ "capabilities": {"tools": {}},
76
+ "serverInfo": {"name": "gapit", "version": __version__},
77
+ }
78
+
79
+
80
+ def _tools_call(id_value: JsonRpcId, params: dict[str, Any] | None) -> dict[str, object]:
81
+ try:
82
+ call = _ToolCall.model_validate(params or {})
83
+ except ValidationError:
84
+ return _error(id_value, -32602, "invalid tools/call params")
85
+ tool = TOOL_HANDLERS.get(call.name)
86
+ if tool is None:
87
+ return _error(id_value, -32602, f"unknown tool: {call.name or '(none)'}")
88
+ try:
89
+ text = tool(call.arguments or {})
90
+ except Exception as exc: # gapit.dispatch semantics: envelope, not a crash
91
+ return _result(id_value, _text(render_error(exc), is_error=True))
92
+ return _result(id_value, _text(text, is_error=False))
93
+
94
+
95
+ def _is_json_object(value: Any) -> TypeGuard[dict[str, Any]]:
96
+ """Narrow a parsed JSON value to the sanctioned dict[str, Any] boundary
97
+ (plain isinstance would surface dict[Unknown, Unknown])."""
98
+ return isinstance(value, dict)
99
+
100
+
101
+ def _reject(value: Any) -> dict[str, object] | None:
102
+ """A line that parsed as JSON but failed _Frame validation: answer the
103
+ request id with -32600 (missing/non-string method) or -32602 (params not
104
+ an object); id-less messages stay ignored like notifications."""
105
+ if not _is_json_object(value):
106
+ return None
107
+ id_value = value.get("id")
108
+ if id_value is None:
109
+ return None
110
+ if not isinstance(value.get("method"), str):
111
+ return _error(id_value, -32600, "invalid request: method must be a string")
112
+ return _error(id_value, -32602, "invalid request: params must be an object")
113
+
114
+
115
+ def _handle(frame: _Frame) -> dict[str, object] | None:
116
+ """One parsed frame -> one response object; None = no response."""
117
+ if "id" not in frame.model_fields_set:
118
+ return None # notifications never get responses
119
+ if frame.method is None:
120
+ return _error(frame.id, -32600, "invalid request: method is required")
121
+ id_value = frame.id
122
+ if frame.method == "initialize":
123
+ return _result(id_value, _initialize(frame.params))
124
+ if frame.method == "tools/list":
125
+ return _result(id_value, {"tools": list(TOOLS)})
126
+ if frame.method == "tools/call":
127
+ return _tools_call(id_value, frame.params)
128
+ return _error(id_value, -32601, f"method not found: {frame.method}")
129
+
130
+
131
+ def serve(stdin: Iterable[str], stdout: TextIO) -> None:
132
+ """Serve newline-delimited JSON-RPC 2.0 until EOF; non-JSON lines and
133
+ id-less invalid messages are ignored, requests that fail frame
134
+ validation get a -32600/-32602 error instead of silence."""
135
+ for line in stdin:
136
+ try:
137
+ value: object = json.loads(line)
138
+ except json.JSONDecodeError:
139
+ continue
140
+ try:
141
+ frame = _Frame.model_validate(value)
142
+ except ValidationError:
143
+ response = _reject(value)
144
+ else:
145
+ response = _handle(frame)
146
+ if response is not None:
147
+ stdout.write(json.dumps(response, separators=(",", ":")) + "\n")
148
+ stdout.flush()
149
+
150
+
151
+ def register_mcp_command(app: typer.Typer) -> None:
152
+ """Attach the `mcp` subcommand (cli.py stays import + registration only)."""
153
+ app.command("mcp")(main)
154
+
155
+
156
+ def main() -> None:
157
+ """`gapit-mcp` console script; also backs the `gapit mcp` subcommand."""
158
+ serve(sys.stdin, sys.stdout)
gapit/mcp_schemas.py ADDED
@@ -0,0 +1,123 @@
1
+ """MCP tools/list declarations: names, descriptions, inputSchemas (protocol: gapit.mcp).
2
+
3
+ Pure data — the roster mirrors the handler map in :mod:`gapit.mcp_tools`.
4
+ Descriptions are wire-visible wording; treat any edit like a schema change
5
+ (AGENTS.md §5).
6
+ """
7
+
8
+ from gapit.db_query_ops import DEFAULT_LIMIT, DEFAULT_STALE_DAYS, SearchField
9
+ from gapit.formats.schemas import SCHEMA_MODELS
10
+ from gapit.reads import ReadTypeEnum
11
+
12
+ _FILES: dict[str, object] = {"type": "array", "items": {"type": "string"}}
13
+ _FLAG: dict[str, object] = {"type": "boolean", "default": False}
14
+ _STR: dict[str, object] = {"type": "string"}
15
+ _DAYS: dict[str, object] = {"type": "integer", "minimum": 0, "default": DEFAULT_STALE_DAYS}
16
+ _SEARCH_FIELDS: list[str] = [field.value for field in SearchField]
17
+ _READ_TYPES: list[str] = [preset.value for preset in ReadTypeEnum]
18
+
19
+
20
+ def _tool_entry(
21
+ name: str,
22
+ description: str,
23
+ properties: dict[str, object] | None = None,
24
+ required: list[str] | None = None,
25
+ ) -> dict[str, object]:
26
+ schema = {"type": "object", "properties": properties or {}, "required": required or []}
27
+ return {"name": name, "description": description, "inputSchema": schema}
28
+
29
+
30
+ TOOLS: list[dict[str, object]] = [
31
+ _tool_entry(
32
+ "screen",
33
+ "Screen contig files for AMR/virulence genes (json = gapit.report/1;"
34
+ " aligner minimap2 = fast assembly survey emitting gapit.reads/1).",
35
+ dict(
36
+ files=_FILES,
37
+ db={"type": "string", "default": "ncbi"},
38
+ minid={"type": "number"},
39
+ mincov={"type": "number"},
40
+ format={"type": "string", "enum": ["json", "tsv", "md"], "default": "json"},
41
+ aligner={"type": "string", "enum": ["blastn", "minimap2"], "default": "blastn"},
42
+ min_breadth={"type": "number", "minimum": 0, "maximum": 100, "default": 90},
43
+ min_identity={"type": "number", "minimum": 0, "maximum": 100, "default": 0},
44
+ min_mapq={"type": "integer", "minimum": 0, "default": 0},
45
+ datadir=_STR,
46
+ ),
47
+ ["files"],
48
+ ),
49
+ _tool_entry(
50
+ "screen_reads",
51
+ "Screen FASTQ reads for genes via minimap2 (json = gapit.reads/1;"
52
+ " min_identity/min_mapq > 0 emits gapit.reads/2). Returns the rendered"
53
+ " document.",
54
+ dict(
55
+ r1=_FILES,
56
+ r2=_FILES,
57
+ read_type={"type": "string", "enum": _READ_TYPES, "default": "sr"},
58
+ min_breadth={"type": "number", "minimum": 0, "maximum": 100, "default": 90},
59
+ min_identity={"type": "number", "minimum": 0, "maximum": 100, "default": 0},
60
+ min_mapq={"type": "integer", "minimum": 0, "default": 0},
61
+ format={"type": "string", "enum": ["json", "md"], "default": "json"},
62
+ db={"type": "string", "default": "ncbi"},
63
+ datadir=_STR,
64
+ ),
65
+ ["r1"],
66
+ ),
67
+ _tool_entry(
68
+ "summary",
69
+ "Summarize report tables into a gapit.summary/1 matrix.",
70
+ dict(files=_FILES, identity={"type": "boolean"}, nopath={"type": "boolean"}),
71
+ ["files"],
72
+ ),
73
+ _tool_entry(
74
+ "schema",
75
+ "Print the JSON Schema of a gapit output document.",
76
+ dict(name={"type": "string", "enum": sorted(SCHEMA_MODELS)}),
77
+ ["name"],
78
+ ),
79
+ _tool_entry("db_list", "List database providers and their installed state (gapit.dblist/1)."),
80
+ _tool_entry(
81
+ "db_fetch",
82
+ "Fetch provider database(s) into the datadir (name omitted: card+vfdb from bundled"
83
+ " snapshots; network installs can take minutes). One JSON receipt line per database"
84
+ " (db, records, dbtype, destination).",
85
+ dict(name=_STR, datadir=_STR, force=_FLAG),
86
+ ),
87
+ _tool_entry(
88
+ "db_build",
89
+ "Build a custom database from a LOCAL FASTA filesystem path (plain, abricate ~~~, or"
90
+ " gapit| headers; .gz/.bz2 accepted). Returns a JSON receipt (db, records, dbtype,"
91
+ " destination).",
92
+ dict(
93
+ name=_STR,
94
+ fasta=_STR,
95
+ tsv=_STR,
96
+ dbtype={"type": "string", "enum": ["nucl", "prot"]},
97
+ description=_STR,
98
+ datadir=_STR,
99
+ force=_FLAG,
100
+ ),
101
+ ["name", "fasta"],
102
+ ),
103
+ _tool_entry(
104
+ "db_search",
105
+ "Search installed databases for a term (case-insensitive substring, or exact"
106
+ " full-field equality). Hit rows are TSV: DB, GENE, ACCESSION, FUNCTION, PRODUCT, LENGTH.",
107
+ dict(
108
+ term=_STR,
109
+ db=_STR,
110
+ field={"type": "string", "enum": _SEARCH_FIELDS, "default": "any"},
111
+ exact=_FLAG,
112
+ limit={"type": "integer", "minimum": 0, "default": DEFAULT_LIMIT},
113
+ datadir=_STR,
114
+ ),
115
+ ["term"],
116
+ ),
117
+ _tool_entry(
118
+ "db_outdated",
119
+ "Report installed database ages against the staleness threshold (days) and newer"
120
+ " bundled snapshots. Rows are TSV with columns NAME, FETCHED_AT, AGE_DAYS, STATUS.",
121
+ dict(days=_DAYS, datadir=_STR),
122
+ ),
123
+ ]
gapit/mcp_tools.py ADDED
@@ -0,0 +1,289 @@
1
+ """MCP tool implementations (protocol: gapit.mcp; declarations: gapit.mcp_schemas).
2
+
3
+ Every tool mirrors its CLI twin by calling the SAME shared callables —
4
+ never a reimplementation. Failures raise typed errors; the protocol layer
5
+ renders them as isError envelopes.
6
+ """
7
+
8
+ import json
9
+ from collections.abc import Callable
10
+ from datetime import UTC, datetime
11
+ from pathlib import Path
12
+ from typing import Any
13
+
14
+ from gapit import config
15
+ from gapit.db_build_ops import perform_build
16
+ from gapit.db_ops import db_list_json, perform_fetch
17
+ from gapit.db_query_ops import (
18
+ DEFAULT_LIMIT,
19
+ DEFAULT_STALE_DAYS,
20
+ SearchField,
21
+ outdated_tsv_lines,
22
+ perform_outdated,
23
+ perform_search,
24
+ )
25
+ from gapit.errors import usage_fail
26
+ from gapit.formats.schemas import SCHEMA_MODELS
27
+ from gapit.formats.summary import render_summary_json
28
+ from gapit.reads import ReadTypeEnum
29
+ from gapit.screening import AlignerEnum, OutputFormat, run_screen
30
+ from gapit.screening_reads import run_screen_assemblies, run_screen_reads
31
+ from gapit.summary import SummaryParams, build_summary
32
+
33
+ # Parsed JSON-RPC argument containers stay dict[str, Any] until the
34
+ # per-field isinstance checks in the helpers below pin concrete types (the
35
+ # sanctioned Any boundary, card.py precedent).
36
+
37
+
38
+ def _paths(arguments: dict[str, Any], key: str) -> list[Path]:
39
+ value: Any = arguments.get(key, [])
40
+ # items stays unnarrowed Any: iterating the isinstance-narrowed value
41
+ # would leak Unknown into basedpyright strict; value proves list-ness.
42
+ items: Any = arguments.get(key, [])
43
+ if not isinstance(value, list) or not all(isinstance(item, str) for item in items):
44
+ usage_fail(f"{key} must be an array of file path strings")
45
+ return [Path(item) for item in items]
46
+
47
+
48
+ def _string(arguments: dict[str, Any], key: str, default: str) -> str:
49
+ value = arguments.get(key, default)
50
+ if not isinstance(value, str):
51
+ usage_fail(f"{key} must be a string")
52
+ return value
53
+
54
+
55
+ def _optional_string(arguments: dict[str, Any], key: str) -> str | None:
56
+ value = arguments.get(key)
57
+ if value is not None and not isinstance(value, str):
58
+ usage_fail(f"{key} must be a string")
59
+ return value
60
+
61
+
62
+ def _required_string(arguments: dict[str, Any], key: str) -> str:
63
+ value = arguments.get(key)
64
+ if not isinstance(value, str) or not value:
65
+ usage_fail(f"{key} must be a non-empty string")
66
+ return value
67
+
68
+
69
+ def _optional_path(arguments: dict[str, Any], key: str) -> Path | None:
70
+ value = _optional_string(arguments, key)
71
+ return None if value is None else Path(value)
72
+
73
+
74
+ def _integer(arguments: dict[str, Any], key: str, default: int) -> int:
75
+ value = arguments.get(key, default)
76
+ if isinstance(value, bool) or not isinstance(value, int) or value < 0:
77
+ usage_fail(f"{key} must be a non-negative integer")
78
+ return value
79
+
80
+
81
+ def _number(arguments: dict[str, Any], key: str, default: float) -> float:
82
+ value = arguments.get(key, default)
83
+ if isinstance(value, bool) or not isinstance(value, (int, float)):
84
+ usage_fail(f"{key} must be a number")
85
+ return float(value)
86
+
87
+
88
+ def _flag(arguments: dict[str, Any], key: str) -> bool:
89
+ value = arguments.get(key, False)
90
+ if not isinstance(value, bool):
91
+ usage_fail(f"{key} must be a boolean")
92
+ return value
93
+
94
+
95
+ def _optional_output_format(name: str) -> OutputFormat | None:
96
+ """Map an already-validated format name onto the reads use-case's
97
+ OutputFormat|None (None = the json default, SPEC.md §10)."""
98
+ return None if name == "json" else OutputFormat(name)
99
+
100
+
101
+ def _tool_screen(arguments: dict[str, Any]) -> str:
102
+ files = _paths(arguments, "files")
103
+ if not files:
104
+ usage_fail("no input files given (files is required)")
105
+ db_name = _string(arguments, "db", "ncbi")
106
+ datadir = _optional_path(arguments, "datadir")
107
+ minid = _number(arguments, "minid", 80.0)
108
+ mincov = _number(arguments, "mincov", 80.0)
109
+ output_format = _string(arguments, "format", "json")
110
+ if output_format not in ("json", "tsv", "md"):
111
+ usage_fail(f"format must be one of json, tsv, md: got {output_format}")
112
+ aligner_name = _string(arguments, "aligner", "blastn")
113
+ try:
114
+ aligner = AlignerEnum(aligner_name)
115
+ except ValueError:
116
+ usage_fail(f"aligner must be blastn or minimap2: got {aligner_name}")
117
+ min_breadth = _number(arguments, "min_breadth", 90.0)
118
+ min_identity = _number(arguments, "min_identity", 0.0)
119
+ min_mapq = _integer(arguments, "min_mapq", 0)
120
+ if aligner is AlignerEnum.minimap2:
121
+ # tsv is rejected by the use-case itself (reads-mode format rule);
122
+ # non-default minid/mincov too (blastn-only thresholds).
123
+ return run_screen_assemblies(
124
+ files,
125
+ None,
126
+ db_name,
127
+ datadir,
128
+ None,
129
+ min_breadth,
130
+ min_identity,
131
+ min_mapq,
132
+ threads=1,
133
+ jobs=1,
134
+ noheader=False,
135
+ nopath=False,
136
+ output_format=_optional_output_format(output_format),
137
+ quiet=True,
138
+ minid=minid,
139
+ mincov=mincov,
140
+ )
141
+ if (min_breadth, min_identity, min_mapq) != (90.0, 0.0, 0):
142
+ usage_fail("reads-mode parameters require aligner minimap2")
143
+ return run_screen(
144
+ files=files,
145
+ db_name=db_name,
146
+ datadir=datadir,
147
+ minid=minid,
148
+ mincov=mincov,
149
+ threads=1,
150
+ jobs=1,
151
+ fofn=None,
152
+ quiet=True,
153
+ noheader=False,
154
+ nopath=False,
155
+ debug=False,
156
+ output_format=OutputFormat(output_format),
157
+ )
158
+
159
+
160
+ def _reads_paths(arguments: dict[str, Any], key: str) -> list[Path]:
161
+ """Reads path array -> the native list the use-case consumes (no
162
+ join/split round trip, so commas in filenames survive). Empty elements
163
+ are usage errors: an empty string would silently become the cwd."""
164
+ _paths(arguments, key)
165
+ # list[str] proven by the _paths isinstance checks above
166
+ raw: list[str] = arguments.get(key, [])
167
+ if any(item == "" for item in raw):
168
+ usage_fail(f"{key} contains an empty element")
169
+ return [Path(item) for item in raw]
170
+
171
+
172
+ def _tool_screen_reads(arguments: dict[str, Any]) -> str:
173
+ r1 = _reads_paths(arguments, "r1")
174
+ if not r1:
175
+ usage_fail("no reads files given (r1 is required)")
176
+ r2 = _reads_paths(arguments, "r2") or None
177
+ read_type_name = _string(arguments, "read_type", ReadTypeEnum.sr.value)
178
+ try:
179
+ read_type = ReadTypeEnum(read_type_name)
180
+ except ValueError:
181
+ usage_fail(f"read_type must be one of {', '.join(t.value for t in ReadTypeEnum)}")
182
+ output_format = _string(arguments, "format", "json")
183
+ if output_format not in ("json", "md"):
184
+ usage_fail(f"format must be one of json, md: got {output_format}")
185
+ min_breadth = _number(arguments, "min_breadth", 90.0)
186
+ if not 0.0 <= min_breadth <= 100.0:
187
+ usage_fail(f"min_breadth must be in [0, 100]: got {min_breadth}")
188
+ min_identity = _number(arguments, "min_identity", 0.0)
189
+ if not 0.0 <= min_identity <= 100.0:
190
+ usage_fail(f"min_identity must be in [0, 100]: got {min_identity}")
191
+ min_mapq = _integer(arguments, "min_mapq", 0)
192
+ # lane pairing is validated by the use-case (frozen CLI message)
193
+ return run_screen_reads(
194
+ r1,
195
+ r2,
196
+ _string(arguments, "db", "ncbi"),
197
+ _optional_path(arguments, "datadir"),
198
+ read_type,
199
+ min_breadth,
200
+ min_identity,
201
+ min_mapq,
202
+ threads=1,
203
+ output_format=_optional_output_format(output_format),
204
+ quiet=True,
205
+ )
206
+
207
+
208
+ def _tool_summary(arguments: dict[str, Any]) -> str:
209
+ files = _paths(arguments, "files")
210
+ if not files:
211
+ usage_fail("summary needs >= 1 report file(s)")
212
+ params = SummaryParams(identity=_flag(arguments, "identity"), nopath=_flag(arguments, "nopath"))
213
+ # The CLI warns about duplicate inputs on stderr; MCP reserves stderr for
214
+ # protocol-internal errors, so those warnings are dropped here.
215
+ matrix = build_summary(files, params, warn=lambda message: None)
216
+ return render_summary_json(matrix, now=datetime.now(UTC))
217
+
218
+
219
+ def _tool_schema(arguments: dict[str, Any]) -> str:
220
+ name = _string(arguments, "name", "")
221
+ model = SCHEMA_MODELS.get(name)
222
+ if model is None:
223
+ usage_fail(f"unknown schema name: {name} (choose from: {', '.join(SCHEMA_MODELS)})")
224
+ return json.dumps(model.model_json_schema(by_alias=True), indent=2)
225
+
226
+
227
+ def _tool_db_list(arguments: dict[str, Any]) -> str:
228
+ return db_list_json(config.resolve_datadir(None))
229
+
230
+
231
+ def _tool_db_fetch(arguments: dict[str, Any]) -> str:
232
+ receipts = perform_fetch(
233
+ _optional_string(arguments, "name"),
234
+ _optional_path(arguments, "datadir"),
235
+ force=_flag(arguments, "force"),
236
+ )
237
+ return "\n".join(receipt.model_dump_json() for receipt in receipts)
238
+
239
+
240
+ def _tool_db_build(arguments: dict[str, Any]) -> str:
241
+ dbtype = _optional_string(arguments, "dbtype")
242
+ if dbtype is not None and dbtype not in ("nucl", "prot"):
243
+ usage_fail(f"dbtype must be nucl or prot: got {dbtype}")
244
+ receipt = perform_build(
245
+ _required_string(arguments, "name"),
246
+ Path(_required_string(arguments, "fasta")),
247
+ _optional_path(arguments, "tsv"),
248
+ dbtype,
249
+ _string(arguments, "description", ""),
250
+ _optional_path(arguments, "datadir"),
251
+ _flag(arguments, "force"),
252
+ warn=lambda message: None,
253
+ )
254
+ return receipt.model_dump_json()
255
+
256
+
257
+ def _tool_db_search(arguments: dict[str, Any]) -> str:
258
+ try:
259
+ field = SearchField(_string(arguments, "field", "any"))
260
+ except ValueError:
261
+ usage_fail(f"field must be one of {', '.join(f.value for f in SearchField)}")
262
+ hits, _total = perform_search(
263
+ _required_string(arguments, "term"),
264
+ _optional_path(arguments, "datadir"),
265
+ db=_optional_string(arguments, "db"),
266
+ field=field,
267
+ exact=_flag(arguments, "exact"),
268
+ limit=_integer(arguments, "limit", DEFAULT_LIMIT),
269
+ )
270
+ return "\n".join(hits)
271
+
272
+
273
+ def _tool_db_outdated(arguments: dict[str, Any]) -> str:
274
+ days = _integer(arguments, "days", DEFAULT_STALE_DAYS)
275
+ lines = outdated_tsv_lines(perform_outdated(_optional_path(arguments, "datadir"), days=days))
276
+ return "\n".join(lines)
277
+
278
+
279
+ TOOL_HANDLERS: dict[str, Callable[[dict[str, Any]], str]] = {
280
+ "screen": _tool_screen,
281
+ "screen_reads": _tool_screen_reads,
282
+ "summary": _tool_summary,
283
+ "schema": _tool_schema,
284
+ "db_list": _tool_db_list,
285
+ "db_fetch": _tool_db_fetch,
286
+ "db_build": _tool_db_build,
287
+ "db_search": _tool_db_search,
288
+ "db_outdated": _tool_db_outdated,
289
+ }