gapit 0.2.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gapit/__init__.py +3 -0
- gapit/blast.py +239 -0
- gapit/cli.py +128 -0
- gapit/cmd_db.py +113 -0
- gapit/cmd_db_build.py +72 -0
- gapit/cmd_db_install.py +126 -0
- gapit/cmd_db_outdated.py +51 -0
- gapit/cmd_db_search.py +66 -0
- gapit/cmd_screen.py +214 -0
- gapit/cmd_summary.py +65 -0
- gapit/config.py +51 -0
- gapit/data/snapshots/card.tar.gz +0 -0
- gapit/data/snapshots/vfdb.tar.gz +0 -0
- gapit/db.py +226 -0
- gapit/db_build_ops.py +210 -0
- gapit/db_ops.py +128 -0
- gapit/db_query_ops.py +252 -0
- gapit/dbbuild.py +207 -0
- gapit/dbcodec.py +117 -0
- gapit/dispatch.py +31 -0
- gapit/errors.py +87 -0
- gapit/fasta.py +123 -0
- gapit/formats/__init__.py +1 -0
- gapit/formats/json.py +309 -0
- gapit/formats/md.py +190 -0
- gapit/formats/schemas.py +30 -0
- gapit/formats/summary.py +103 -0
- gapit/formats/tsv.py +45 -0
- gapit/hits.py +107 -0
- gapit/mcp.py +158 -0
- gapit/mcp_schemas.py +123 -0
- gapit/mcp_tools.py +289 -0
- gapit/minimap.py +20 -0
- gapit/minimap2_run.py +114 -0
- gapit/paf.py +115 -0
- gapit/proctools.py +24 -0
- gapit/providers/__init__.py +39 -0
- gapit/providers/argannot.py +94 -0
- gapit/providers/bacmet2.py +59 -0
- gapit/providers/card.py +150 -0
- gapit/providers/common.py +245 -0
- gapit/providers/ecoh.py +63 -0
- gapit/providers/ecoli_vf.py +74 -0
- gapit/providers/megares.py +71 -0
- gapit/providers/ncbi.py +103 -0
- gapit/providers/plasmidfinder.py +69 -0
- gapit/providers/resfinder.py +123 -0
- gapit/providers/snapshots.py +119 -0
- gapit/providers/upec_expec_vf.py +85 -0
- gapit/providers/vfdb.py +92 -0
- gapit/providers/victors.py +109 -0
- gapit/py.typed +0 -0
- gapit/reads.py +221 -0
- gapit/records.py +152 -0
- gapit/report.py +25 -0
- gapit/screening.py +145 -0
- gapit/screening_reads.py +255 -0
- gapit/seqconvert.py +203 -0
- gapit/summary.py +151 -0
- gapit-0.2.2.dist-info/METADATA +183 -0
- gapit-0.2.2.dist-info/RECORD +64 -0
- gapit-0.2.2.dist-info/WHEEL +4 -0
- gapit-0.2.2.dist-info/entry_points.txt +3 -0
- gapit-0.2.2.dist-info/licenses/LICENSE +21 -0
gapit/db_build_ops.py
ADDED
|
@@ -0,0 +1,210 @@
|
|
|
1
|
+
"""The custom-database build use-case, shared by the CLI and MCP.
|
|
2
|
+
|
|
3
|
+
Turns a user-supplied FASTA into a fully built gapit-native database via the
|
|
4
|
+
existing pipeline (records.jsonl -> sequences + BLAST index + manifest,
|
|
5
|
+
written last). This module is orchestration plus a metadata merge only — the
|
|
6
|
+
building blocks live in records/dbbuild/fasta/dbcodec/db (SPEC.md §11 covers
|
|
7
|
+
the build pipeline itself). The typer command (:mod:`gapit.cmd_db_build`)
|
|
8
|
+
and the MCP tool (:mod:`gapit.mcp_tools`) are thin callers.
|
|
9
|
+
|
|
10
|
+
Header kind is detected PER RECORD (mixed files allowed): a ``gapit|``
|
|
11
|
+
prefix decodes through the strict tagged codec, anything else through the
|
|
12
|
+
abricate ``~~~`` rules (a plain id carries no ``~~~`` and decodes to itself).
|
|
13
|
+
``db`` is always the TARGET name and ``source_id`` keeps the original id
|
|
14
|
+
token; record order is input order. Sequences are stored verbatim — no
|
|
15
|
+
provider-style normalization, this is the user's curated truth.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
import csv
|
|
19
|
+
from collections.abc import Callable
|
|
20
|
+
from dataclasses import dataclass
|
|
21
|
+
from datetime import UTC, datetime
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
from typing import Literal
|
|
24
|
+
|
|
25
|
+
from pydantic import BaseModel
|
|
26
|
+
|
|
27
|
+
from gapit import config
|
|
28
|
+
from gapit.db import mol_type
|
|
29
|
+
from gapit.dbbuild import build_database
|
|
30
|
+
from gapit.dbcodec import decode_seqid
|
|
31
|
+
from gapit.errors import DatabaseError, InputError, UsageError
|
|
32
|
+
from gapit.fasta import FastaRecord, iter_fasta
|
|
33
|
+
from gapit.records import Record, write_records
|
|
34
|
+
|
|
35
|
+
Dbtype = Literal["nucl", "prot"]
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class BuildReceipt(BaseModel, frozen=True):
|
|
39
|
+
"""One-line JSON success receipt for `db build` (mirrors db fetch)."""
|
|
40
|
+
|
|
41
|
+
db: str
|
|
42
|
+
records: int
|
|
43
|
+
dbtype: Dbtype
|
|
44
|
+
destination: str
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@dataclass(frozen=True, slots=True)
|
|
48
|
+
class Metadata:
|
|
49
|
+
"""Parsed --tsv: which merge columns the header carries, and the row
|
|
50
|
+
values per gene (accession, ';'-joined function; '' for absent columns)."""
|
|
51
|
+
|
|
52
|
+
columns: frozenset[str]
|
|
53
|
+
rows: dict[str, tuple[str, str]]
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _split_function(joined: str) -> tuple[str, ...]:
|
|
57
|
+
"""';'-joined classes -> tuple, empty pieces dropped."""
|
|
58
|
+
return tuple(part for part in joined.split(";") if part)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _to_record(fasta: FastaRecord, name: str, default_product: str) -> Record:
|
|
62
|
+
"""One FASTA record -> one Record.
|
|
63
|
+
|
|
64
|
+
decode_seqid routes by header kind; the product prefers the FASTA
|
|
65
|
+
description, then --description, then the gene (the provider-side
|
|
66
|
+
``description or gene`` pattern). Malformed ``gapit|`` headers raise
|
|
67
|
+
their native HEADER_MALFORMED DatabaseError.
|
|
68
|
+
"""
|
|
69
|
+
header = decode_seqid(fasta.id, default_db=name)
|
|
70
|
+
return Record(
|
|
71
|
+
db=name,
|
|
72
|
+
gene=header.gene,
|
|
73
|
+
sequence=fasta.sequence,
|
|
74
|
+
accession=header.accession,
|
|
75
|
+
function=_split_function(header.function),
|
|
76
|
+
product=fasta.description or default_product or header.gene,
|
|
77
|
+
source_id=fasta.id,
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _read_metadata(path: Path, warn: Callable[[str], None]) -> Metadata:
|
|
82
|
+
"""Parse the metadata TSV: header row mandatory and must carry ``gene``
|
|
83
|
+
(InputError METADATA_MALFORMED otherwise); ``accession``/``function``
|
|
84
|
+
are optional per-file, extra columns are ignored. Duplicate genes keep
|
|
85
|
+
the FIRST row (warn); short rows read their missing cells as empty.
|
|
86
|
+
"""
|
|
87
|
+
if not path.is_file():
|
|
88
|
+
raise InputError(
|
|
89
|
+
f"metadata file not found or unreadable: {path}",
|
|
90
|
+
code="INPUT_NOT_FOUND",
|
|
91
|
+
context={"file": str(path)},
|
|
92
|
+
)
|
|
93
|
+
with path.open(encoding="utf-8", newline="") as handle:
|
|
94
|
+
rows = csv.reader(handle, delimiter="\t")
|
|
95
|
+
header = next((row for row in rows if row), None)
|
|
96
|
+
if header is None or "gene" not in header:
|
|
97
|
+
raise InputError(
|
|
98
|
+
f"metadata TSV needs a header row with a 'gene' column: {path}",
|
|
99
|
+
code="METADATA_MALFORMED",
|
|
100
|
+
context={"file": str(path), "needed": "gene"},
|
|
101
|
+
)
|
|
102
|
+
columns = {column: index for index, column in enumerate(header)}
|
|
103
|
+
|
|
104
|
+
def cell(padded: list[str], column: str) -> str:
|
|
105
|
+
return padded[columns[column]]
|
|
106
|
+
|
|
107
|
+
metadata: dict[str, tuple[str, str]] = {}
|
|
108
|
+
for row in rows:
|
|
109
|
+
if not any(value.strip() for value in row):
|
|
110
|
+
continue
|
|
111
|
+
padded = row + [""] * (len(header) - len(row))
|
|
112
|
+
gene = cell(padded, "gene")
|
|
113
|
+
if not gene:
|
|
114
|
+
continue
|
|
115
|
+
if gene in metadata:
|
|
116
|
+
warn(f"duplicate gene {gene!r} in metadata TSV: keeping the first row")
|
|
117
|
+
continue
|
|
118
|
+
metadata[gene] = (
|
|
119
|
+
cell(padded, "accession") if "accession" in columns else "",
|
|
120
|
+
cell(padded, "function") if "function" in columns else "",
|
|
121
|
+
)
|
|
122
|
+
return Metadata(columns=frozenset(columns), rows=metadata)
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _merge(records: list[Record], metadata: Metadata, warn: Callable[[str], None]) -> list[Record]:
|
|
126
|
+
"""Overwrite accession/function on records whose gene has a metadata row
|
|
127
|
+
(only the columns the TSV actually carries); genes only present in the
|
|
128
|
+
TSV are warned about and skipped."""
|
|
129
|
+
merged: list[Record] = []
|
|
130
|
+
for record in records:
|
|
131
|
+
row = metadata.rows.get(record.gene)
|
|
132
|
+
if row is None:
|
|
133
|
+
merged.append(record)
|
|
134
|
+
continue
|
|
135
|
+
update: dict[str, object] = {}
|
|
136
|
+
if "accession" in metadata.columns:
|
|
137
|
+
update["accession"] = row[0]
|
|
138
|
+
if "function" in metadata.columns:
|
|
139
|
+
update["function"] = _split_function(row[1])
|
|
140
|
+
merged.append(record.model_copy(update=update))
|
|
141
|
+
fasta_genes = {record.gene for record in records}
|
|
142
|
+
for gene in sorted(set(metadata.rows) - fasta_genes):
|
|
143
|
+
warn(f"gene {gene!r} in metadata TSV not found in FASTA: skipped")
|
|
144
|
+
return merged
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _resolve_dbtype(flag: Dbtype | None, records: list[Record]) -> Dbtype:
|
|
148
|
+
"""Explicit --dbtype wins; else the abricate mol_type heuristic over the
|
|
149
|
+
concatenated input sequences (db.mol_type, SPEC.md §2)."""
|
|
150
|
+
if flag is not None:
|
|
151
|
+
return flag
|
|
152
|
+
return mol_type("".join(record.sequence for record in records))
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def perform_build(
|
|
156
|
+
name: str,
|
|
157
|
+
fasta: Path,
|
|
158
|
+
tsv: Path | None,
|
|
159
|
+
dbtype: Dbtype | None,
|
|
160
|
+
description: str,
|
|
161
|
+
datadir: Path | None,
|
|
162
|
+
force: bool,
|
|
163
|
+
*,
|
|
164
|
+
warn: Callable[[str], None],
|
|
165
|
+
quiet: bool = True,
|
|
166
|
+
) -> BuildReceipt:
|
|
167
|
+
"""Run the custom-build pipeline and return the receipt — the shared CLI
|
|
168
|
+
+ MCP path. Warnings go to the caller-supplied ``warn`` (CLI: stderr;
|
|
169
|
+
MCP: dropped — stderr is reserved for the protocol)."""
|
|
170
|
+
|
|
171
|
+
# Security/frozen rule: `Path(datadir) / name` REPLACES the base when name
|
|
172
|
+
# is absolute (and `..` escapes it); plain names only, all else allowed.
|
|
173
|
+
if not name or name in (".", "..") or "/" in name or "\\" in name:
|
|
174
|
+
raise UsageError(
|
|
175
|
+
f"database name must be a plain name without path separators: {name!r}",
|
|
176
|
+
code="USAGE_ERROR",
|
|
177
|
+
context={"name": name},
|
|
178
|
+
)
|
|
179
|
+
if not fasta.is_file():
|
|
180
|
+
raise InputError(
|
|
181
|
+
f"FASTA file not found or unreadable: {fasta}",
|
|
182
|
+
code="INPUT_NOT_FOUND",
|
|
183
|
+
context={"file": str(fasta)},
|
|
184
|
+
)
|
|
185
|
+
db_dir = config.ensure_datadir(datadir) / name
|
|
186
|
+
if (db_dir / "gapit-manifest.json").is_file() and not force:
|
|
187
|
+
raise DatabaseError(
|
|
188
|
+
f"won't overwrite existing database {name} (use --force)",
|
|
189
|
+
code="DB_ALREADY_EXISTS",
|
|
190
|
+
context={"db": name},
|
|
191
|
+
)
|
|
192
|
+
records = [_to_record(fasta_record, name, description) for fasta_record in iter_fasta(fasta)]
|
|
193
|
+
if tsv is not None:
|
|
194
|
+
records = _merge(records, _read_metadata(tsv, warn), warn)
|
|
195
|
+
db_dir.mkdir(parents=True, exist_ok=True)
|
|
196
|
+
write_records(records, db_dir / "records.jsonl")
|
|
197
|
+
manifest = build_database(
|
|
198
|
+
db_dir,
|
|
199
|
+
name=name,
|
|
200
|
+
dbtype=_resolve_dbtype(dbtype, records),
|
|
201
|
+
source_urls=("local",),
|
|
202
|
+
fetched_at=datetime.now(UTC).strftime("%Y-%m-%dT%H:%M:%SZ"),
|
|
203
|
+
quiet=quiet,
|
|
204
|
+
)
|
|
205
|
+
return BuildReceipt(
|
|
206
|
+
db=name,
|
|
207
|
+
records=manifest.n_records,
|
|
208
|
+
dbtype=manifest.dbtype,
|
|
209
|
+
destination=str(db_dir),
|
|
210
|
+
)
|
gapit/db_ops.py
ADDED
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
"""Database provider use-cases shared by the CLI and MCP: fetch and list.
|
|
2
|
+
|
|
3
|
+
``perform_fetch`` installs provider database(s) under the datadir (bundled
|
|
4
|
+
snapshot first, ``from_source`` forces upstream); the ``db_list_*``
|
|
5
|
+
callables build the gapit.dblist/1 provider listing. The typer commands
|
|
6
|
+
(:mod:`gapit.cmd_db`) and the MCP tools (:mod:`gapit.mcp_tools`) are thin
|
|
7
|
+
callers — this module owns the behavior and imports no CLI plumbing.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from collections.abc import Iterator
|
|
11
|
+
from datetime import UTC, datetime
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import Literal
|
|
14
|
+
|
|
15
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
16
|
+
|
|
17
|
+
from gapit import config
|
|
18
|
+
from gapit.errors import UsageError
|
|
19
|
+
from gapit.providers import REGISTRY
|
|
20
|
+
from gapit.providers.common import Dbtype, fetch_provider
|
|
21
|
+
from gapit.records import read_manifest
|
|
22
|
+
|
|
23
|
+
# Bare `gapit db fetch` installs these, in order — the two providers whose
|
|
24
|
+
# snapshots ship inside the wheel (Wave G: zero-network bootstrap).
|
|
25
|
+
DEFAULT_DBS = ("card", "vfdb")
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class ProviderReceipt(BaseModel, frozen=True):
|
|
29
|
+
"""One-line JSON success receipt for `db fetch` (no biological data)."""
|
|
30
|
+
|
|
31
|
+
db: str
|
|
32
|
+
records: int
|
|
33
|
+
dbtype: Dbtype
|
|
34
|
+
destination: str
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class DbListEntry(BaseModel, frozen=True):
|
|
38
|
+
"""One provider row in the gapit.dblist/1 listing document."""
|
|
39
|
+
|
|
40
|
+
name: str
|
|
41
|
+
description: str
|
|
42
|
+
dbtype: str
|
|
43
|
+
installed: bool
|
|
44
|
+
records: int | None = None
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class DbListDocument(BaseModel, frozen=True):
|
|
48
|
+
"""gapit.dblist/1 — `gapit db list --json` output.
|
|
49
|
+
|
|
50
|
+
A CLI listing, deliberately local to this module: NOT registered in
|
|
51
|
+
`gapit schema` (the public schema surface stays unchanged this wave).
|
|
52
|
+
"""
|
|
53
|
+
|
|
54
|
+
model_config = ConfigDict(populate_by_name=True)
|
|
55
|
+
|
|
56
|
+
schema_name: Literal["gapit.dblist/1"] = Field(default="gapit.dblist/1", alias="schema")
|
|
57
|
+
providers: tuple[DbListEntry, ...]
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def perform_fetch(
|
|
61
|
+
name: str | None,
|
|
62
|
+
datadir: Path | None,
|
|
63
|
+
*,
|
|
64
|
+
force: bool = False,
|
|
65
|
+
from_source: bool = False,
|
|
66
|
+
quiet: bool = True,
|
|
67
|
+
debug: bool = False,
|
|
68
|
+
) -> Iterator[ProviderReceipt]:
|
|
69
|
+
"""Fetch provider database(s) into <datadir>/NAME — the shared CLI + MCP
|
|
70
|
+
path. NAME None installs every database in DEFAULT_DBS order; each
|
|
71
|
+
receipt yields as its install completes (streaming, like the CLI's
|
|
72
|
+
per-db stdout lines).
|
|
73
|
+
"""
|
|
74
|
+
|
|
75
|
+
def fetch_one(provider_name: str) -> ProviderReceipt:
|
|
76
|
+
provider = REGISTRY.get(provider_name)
|
|
77
|
+
if provider is None:
|
|
78
|
+
raise UsageError(
|
|
79
|
+
f"unknown provider: {provider_name} (available: {', '.join(sorted(REGISTRY))})",
|
|
80
|
+
code="USAGE_ERROR",
|
|
81
|
+
context={"provider": provider_name},
|
|
82
|
+
)
|
|
83
|
+
db_dir = config.ensure_datadir(datadir) / provider_name
|
|
84
|
+
manifest = fetch_provider(
|
|
85
|
+
provider,
|
|
86
|
+
db_dir,
|
|
87
|
+
fetched_at=datetime.now(UTC).strftime("%Y-%m-%dT%H:%M:%SZ"),
|
|
88
|
+
force=force,
|
|
89
|
+
quiet=quiet,
|
|
90
|
+
from_source=from_source,
|
|
91
|
+
debug=debug,
|
|
92
|
+
)
|
|
93
|
+
return ProviderReceipt(
|
|
94
|
+
db=provider_name,
|
|
95
|
+
records=manifest.n_records,
|
|
96
|
+
dbtype=manifest.dbtype,
|
|
97
|
+
destination=str(db_dir),
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
for provider_name in (name,) if name is not None else DEFAULT_DBS:
|
|
101
|
+
yield fetch_one(provider_name)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def db_list_entries(root: Path) -> list[DbListEntry]:
|
|
105
|
+
"""Provider rows for the gapit.dblist/1 listing — shared CLI + MCP path."""
|
|
106
|
+
entries: list[DbListEntry] = []
|
|
107
|
+
for provider_name in sorted(REGISTRY):
|
|
108
|
+
provider = REGISTRY[provider_name]
|
|
109
|
+
manifest_path = root / provider_name / "gapit-manifest.json"
|
|
110
|
+
installed = manifest_path.is_file()
|
|
111
|
+
entries.append(
|
|
112
|
+
DbListEntry(
|
|
113
|
+
name=provider_name,
|
|
114
|
+
description=provider.description,
|
|
115
|
+
dbtype=provider.dbtype,
|
|
116
|
+
installed=installed,
|
|
117
|
+
records=read_manifest(manifest_path).n_records if installed else None,
|
|
118
|
+
)
|
|
119
|
+
)
|
|
120
|
+
return entries
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def db_list_json(root: Path) -> str:
|
|
124
|
+
"""The gapit.dblist/1 document JSON — shared CLI + MCP path."""
|
|
125
|
+
entries = db_list_entries(root)
|
|
126
|
+
return DbListDocument(providers=tuple(entries)).model_dump_json(
|
|
127
|
+
indent=2, by_alias=True, exclude_none=True
|
|
128
|
+
)
|
gapit/db_query_ops.py
ADDED
|
@@ -0,0 +1,252 @@
|
|
|
1
|
+
"""Read-only queries over installed databases, shared by the CLI and MCP.
|
|
2
|
+
|
|
3
|
+
- ``perform_search``: case-insensitive lookup across the ``records.jsonl``
|
|
4
|
+
truth stores of every installed database (SPEC.md §11).
|
|
5
|
+
- ``perform_outdated``: installed-database staleness report against a
|
|
6
|
+
threshold (``days``) and the bundled snapshot dates. Staleness is a
|
|
7
|
+
REPORT, never an error state.
|
|
8
|
+
|
|
9
|
+
Both are read-only: no builds, no network, no datadir writes. The typer
|
|
10
|
+
commands (:mod:`gapit.cmd_db_search`, :mod:`gapit.cmd_db_outdated`) and the
|
|
11
|
+
MCP tools (:mod:`gapit.mcp_tools`) are thin callers.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
import enum
|
|
15
|
+
from collections.abc import Callable, Iterable
|
|
16
|
+
from datetime import UTC, datetime
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from typing import Literal
|
|
19
|
+
|
|
20
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
21
|
+
|
|
22
|
+
from gapit import config
|
|
23
|
+
from gapit.errors import DatabaseError, InputError, UsageError
|
|
24
|
+
from gapit.proctools import note
|
|
25
|
+
from gapit.providers import REGISTRY
|
|
26
|
+
from gapit.providers.common import bundled_snapshot_manifest
|
|
27
|
+
from gapit.records import Record, installed_db_dirs, read_manifest, read_records
|
|
28
|
+
|
|
29
|
+
DEFAULT_LIMIT = 100
|
|
30
|
+
DEFAULT_STALE_DAYS = 90
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class SearchField(enum.Enum):
|
|
34
|
+
"""Record fields `db search` can match against."""
|
|
35
|
+
|
|
36
|
+
gene = "gene"
|
|
37
|
+
accession = "accession"
|
|
38
|
+
function = "function"
|
|
39
|
+
product = "product"
|
|
40
|
+
any = "any"
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class SearchEntry(BaseModel, frozen=True):
|
|
44
|
+
"""One search hit: the JSONL line (and the TSV row's source), a CLI
|
|
45
|
+
listing shape — not a registered schema."""
|
|
46
|
+
|
|
47
|
+
db: str
|
|
48
|
+
gene: str
|
|
49
|
+
accession: str
|
|
50
|
+
function: tuple[str, ...]
|
|
51
|
+
product: str
|
|
52
|
+
length: int
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _candidates(record: Record, field: SearchField) -> tuple[str, ...]:
|
|
56
|
+
"""The string values of the chosen field (function contributes one per class)."""
|
|
57
|
+
match field:
|
|
58
|
+
case SearchField.gene:
|
|
59
|
+
return (record.gene,)
|
|
60
|
+
case SearchField.accession:
|
|
61
|
+
return (record.accession,)
|
|
62
|
+
case SearchField.function:
|
|
63
|
+
return record.function
|
|
64
|
+
case SearchField.product:
|
|
65
|
+
return (record.product,)
|
|
66
|
+
case SearchField.any:
|
|
67
|
+
return (record.gene, record.accession, *record.function, record.product)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _matches(record: Record, needle: str, field: SearchField, exact: bool) -> bool:
|
|
71
|
+
"""Case-insensitive substring by default; --exact is full-field equality."""
|
|
72
|
+
values = _candidates(record, field)
|
|
73
|
+
if exact:
|
|
74
|
+
return any(value.lower() == needle for value in values)
|
|
75
|
+
return any(needle in value.lower() for value in values)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def search_tsv_row(record: Record) -> str:
|
|
79
|
+
"""The hit row: DB\tGENE\tACCESSION\tFUNCTION\tPRODUCT\tLENGTH."""
|
|
80
|
+
return (
|
|
81
|
+
f"{record.db}\t{record.gene}\t{record.accession}\t"
|
|
82
|
+
f"{';'.join(record.function)}\t{record.product}\t{len(record.sequence)}"
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def search_json_line(record: Record) -> str:
|
|
87
|
+
"""The hit as one JSONL line (same fields as the TSV row, snake_case)."""
|
|
88
|
+
return SearchEntry(
|
|
89
|
+
db=record.db,
|
|
90
|
+
gene=record.gene,
|
|
91
|
+
accession=record.accession,
|
|
92
|
+
function=record.function,
|
|
93
|
+
product=record.product,
|
|
94
|
+
length=len(record.sequence),
|
|
95
|
+
).model_dump_json()
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def perform_search(
|
|
99
|
+
term: str,
|
|
100
|
+
datadir: Path | None,
|
|
101
|
+
*,
|
|
102
|
+
db: str | None = None,
|
|
103
|
+
field: SearchField = SearchField.any,
|
|
104
|
+
exact: bool = False,
|
|
105
|
+
limit: int = DEFAULT_LIMIT,
|
|
106
|
+
render: Callable[[Record], str] = search_tsv_row,
|
|
107
|
+
quiet: bool = True,
|
|
108
|
+
) -> tuple[list[str], int]:
|
|
109
|
+
"""Scan the installed records.jsonl stores — the shared CLI + MCP path.
|
|
110
|
+
|
|
111
|
+
Returns ``(hit lines up to limit, total matching count)``; the caller
|
|
112
|
+
owns the truncation note (a stderr diagnostic) and the printing.
|
|
113
|
+
"""
|
|
114
|
+
root = config.resolve_datadir(datadir)
|
|
115
|
+
db_dirs = installed_db_dirs(root)
|
|
116
|
+
if db is not None and db not in {db_dir.name for db_dir in db_dirs}:
|
|
117
|
+
installed = " ".join(sorted(db_dir.name for db_dir in db_dirs))
|
|
118
|
+
raise UsageError(
|
|
119
|
+
f"unknown database: {db} (installed: {installed or 'none'})",
|
|
120
|
+
code="USAGE_ERROR",
|
|
121
|
+
context={"db": db, "installed": installed},
|
|
122
|
+
)
|
|
123
|
+
needle = term.lower()
|
|
124
|
+
hits: list[str] = []
|
|
125
|
+
total = 0
|
|
126
|
+
for db_dir in db_dirs:
|
|
127
|
+
if db is not None and db_dir.name != db:
|
|
128
|
+
continue
|
|
129
|
+
records_path = db_dir / "records.jsonl"
|
|
130
|
+
if not records_path.is_file():
|
|
131
|
+
if db is not None:
|
|
132
|
+
raise DatabaseError(
|
|
133
|
+
f"database {db_dir.name} is incomplete: records.jsonl missing",
|
|
134
|
+
code="DB_INCOMPLETE",
|
|
135
|
+
context={"db": db_dir.name},
|
|
136
|
+
)
|
|
137
|
+
note(quiet, f"skipping {db_dir.name}: no records.jsonl (incomplete database)")
|
|
138
|
+
continue
|
|
139
|
+
for record in read_records(records_path):
|
|
140
|
+
if _matches(record, needle, field, exact):
|
|
141
|
+
total += 1
|
|
142
|
+
if limit == 0 or len(hits) < limit:
|
|
143
|
+
hits.append(render(record))
|
|
144
|
+
return hits, total
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
class DbOutdatedEntry(BaseModel, frozen=True):
|
|
148
|
+
"""One installed-database row in the gapit.dboutdated/1 report."""
|
|
149
|
+
|
|
150
|
+
db: str
|
|
151
|
+
fetched_at: str
|
|
152
|
+
age_days: float
|
|
153
|
+
status: Literal["ok", "stale", "snapshot-update", "stale+snapshot-update"]
|
|
154
|
+
upstream_version: str
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
class DbOutdatedDocument(BaseModel, frozen=True):
|
|
158
|
+
"""gapit.dboutdated/1 — `gapit db outdated --json` output.
|
|
159
|
+
|
|
160
|
+
A CLI listing, deliberately local to this module: NOT registered in
|
|
161
|
+
`gapit schema` (mirrors the gapit.dblist/1 precedent).
|
|
162
|
+
"""
|
|
163
|
+
|
|
164
|
+
model_config = ConfigDict(populate_by_name=True)
|
|
165
|
+
|
|
166
|
+
schema_name: Literal["gapit.dboutdated/1"] = Field(default="gapit.dboutdated/1", alias="schema")
|
|
167
|
+
databases: tuple[DbOutdatedEntry, ...]
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _utc_timestamp(value: str, where: str) -> datetime:
|
|
171
|
+
"""Parse a manifest fetched_at into aware UTC; anything else is a
|
|
172
|
+
malformed manifest (trusted metadata: writers emit ISO-8601 with Z)."""
|
|
173
|
+
try:
|
|
174
|
+
parsed = datetime.fromisoformat(value)
|
|
175
|
+
except ValueError as exc:
|
|
176
|
+
raise InputError(
|
|
177
|
+
f"malformed manifest {where}: invalid fetched_at: {value}",
|
|
178
|
+
code="MANIFEST_MALFORMED",
|
|
179
|
+
context={"file": where},
|
|
180
|
+
) from exc
|
|
181
|
+
if parsed.tzinfo is None:
|
|
182
|
+
raise InputError(
|
|
183
|
+
f"malformed manifest {where}: fetched_at has no UTC offset: {value}",
|
|
184
|
+
code="MANIFEST_MALFORMED",
|
|
185
|
+
context={"file": where},
|
|
186
|
+
)
|
|
187
|
+
return parsed.astimezone(UTC)
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def _snapshot_newer(name: str, fetched: datetime) -> bool:
|
|
191
|
+
"""True when the provider's bundled snapshot manifest is newer than the
|
|
192
|
+
installed ``fetched`` (unknown provider / no bundle -> False)."""
|
|
193
|
+
provider = REGISTRY.get(name)
|
|
194
|
+
if provider is None:
|
|
195
|
+
return False
|
|
196
|
+
archived = bundled_snapshot_manifest(provider)
|
|
197
|
+
return archived is not None and fetched < _utc_timestamp(
|
|
198
|
+
archived.fetched_at, f"bundled snapshot of {name}"
|
|
199
|
+
)
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def perform_outdated(
|
|
203
|
+
datadir: Path | None, *, days: int = DEFAULT_STALE_DAYS
|
|
204
|
+
) -> list[DbOutdatedEntry]:
|
|
205
|
+
"""Compute the staleness report entries — the shared CLI + MCP path.
|
|
206
|
+
|
|
207
|
+
A database is `stale` past ``days`` and `snapshot-update` when its
|
|
208
|
+
provider's bundled snapshot is newer than the installed copy.
|
|
209
|
+
"""
|
|
210
|
+
root = config.resolve_datadir(datadir)
|
|
211
|
+
db_dirs = installed_db_dirs(root)
|
|
212
|
+
if not db_dirs:
|
|
213
|
+
raise DatabaseError(
|
|
214
|
+
f"no installed databases in datadir: {root}",
|
|
215
|
+
code="DATADIR_EMPTY",
|
|
216
|
+
context={"datadir": str(root)},
|
|
217
|
+
)
|
|
218
|
+
now = datetime.now(UTC)
|
|
219
|
+
entries: list[DbOutdatedEntry] = []
|
|
220
|
+
for db_dir in db_dirs:
|
|
221
|
+
manifest_path = db_dir / "gapit-manifest.json"
|
|
222
|
+
manifest = read_manifest(manifest_path)
|
|
223
|
+
fetched = _utc_timestamp(manifest.fetched_at, str(manifest_path))
|
|
224
|
+
age = (now - fetched).total_seconds() / 86400
|
|
225
|
+
match (age > days, _snapshot_newer(db_dir.name, fetched)):
|
|
226
|
+
case (True, True):
|
|
227
|
+
status = "stale+snapshot-update"
|
|
228
|
+
case (True, False):
|
|
229
|
+
status = "stale"
|
|
230
|
+
case (False, True):
|
|
231
|
+
status = "snapshot-update"
|
|
232
|
+
case (False, False):
|
|
233
|
+
status = "ok"
|
|
234
|
+
entries.append(
|
|
235
|
+
DbOutdatedEntry(
|
|
236
|
+
db=db_dir.name,
|
|
237
|
+
fetched_at=manifest.fetched_at,
|
|
238
|
+
age_days=round(age, 2),
|
|
239
|
+
status=status,
|
|
240
|
+
upstream_version=manifest.upstream_version,
|
|
241
|
+
)
|
|
242
|
+
)
|
|
243
|
+
return entries
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def outdated_tsv_lines(entries: Iterable[DbOutdatedEntry]) -> list[str]:
|
|
247
|
+
"""The NAME/FETCHED_AT/AGE_DAYS/STATUS table lines (CLI stdout = MCP text)."""
|
|
248
|
+
lines = ["NAME\tFETCHED_AT\tAGE_DAYS\tSTATUS"]
|
|
249
|
+
lines.extend(
|
|
250
|
+
f"{entry.db}\t{entry.fetched_at}\t{entry.age_days:.2f}\t{entry.status}" for entry in entries
|
|
251
|
+
)
|
|
252
|
+
return lines
|