gapit 0.2.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. gapit/__init__.py +3 -0
  2. gapit/blast.py +239 -0
  3. gapit/cli.py +128 -0
  4. gapit/cmd_db.py +113 -0
  5. gapit/cmd_db_build.py +72 -0
  6. gapit/cmd_db_install.py +126 -0
  7. gapit/cmd_db_outdated.py +51 -0
  8. gapit/cmd_db_search.py +66 -0
  9. gapit/cmd_screen.py +214 -0
  10. gapit/cmd_summary.py +65 -0
  11. gapit/config.py +51 -0
  12. gapit/data/snapshots/card.tar.gz +0 -0
  13. gapit/data/snapshots/vfdb.tar.gz +0 -0
  14. gapit/db.py +226 -0
  15. gapit/db_build_ops.py +210 -0
  16. gapit/db_ops.py +128 -0
  17. gapit/db_query_ops.py +252 -0
  18. gapit/dbbuild.py +207 -0
  19. gapit/dbcodec.py +117 -0
  20. gapit/dispatch.py +31 -0
  21. gapit/errors.py +87 -0
  22. gapit/fasta.py +123 -0
  23. gapit/formats/__init__.py +1 -0
  24. gapit/formats/json.py +309 -0
  25. gapit/formats/md.py +190 -0
  26. gapit/formats/schemas.py +30 -0
  27. gapit/formats/summary.py +103 -0
  28. gapit/formats/tsv.py +45 -0
  29. gapit/hits.py +107 -0
  30. gapit/mcp.py +158 -0
  31. gapit/mcp_schemas.py +123 -0
  32. gapit/mcp_tools.py +289 -0
  33. gapit/minimap.py +20 -0
  34. gapit/minimap2_run.py +114 -0
  35. gapit/paf.py +115 -0
  36. gapit/proctools.py +24 -0
  37. gapit/providers/__init__.py +39 -0
  38. gapit/providers/argannot.py +94 -0
  39. gapit/providers/bacmet2.py +59 -0
  40. gapit/providers/card.py +150 -0
  41. gapit/providers/common.py +245 -0
  42. gapit/providers/ecoh.py +63 -0
  43. gapit/providers/ecoli_vf.py +74 -0
  44. gapit/providers/megares.py +71 -0
  45. gapit/providers/ncbi.py +103 -0
  46. gapit/providers/plasmidfinder.py +69 -0
  47. gapit/providers/resfinder.py +123 -0
  48. gapit/providers/snapshots.py +119 -0
  49. gapit/providers/upec_expec_vf.py +85 -0
  50. gapit/providers/vfdb.py +92 -0
  51. gapit/providers/victors.py +109 -0
  52. gapit/py.typed +0 -0
  53. gapit/reads.py +221 -0
  54. gapit/records.py +152 -0
  55. gapit/report.py +25 -0
  56. gapit/screening.py +145 -0
  57. gapit/screening_reads.py +255 -0
  58. gapit/seqconvert.py +203 -0
  59. gapit/summary.py +151 -0
  60. gapit-0.2.2.dist-info/METADATA +183 -0
  61. gapit-0.2.2.dist-info/RECORD +64 -0
  62. gapit-0.2.2.dist-info/WHEEL +4 -0
  63. gapit-0.2.2.dist-info/entry_points.txt +3 -0
  64. gapit-0.2.2.dist-info/licenses/LICENSE +21 -0
gapit/db_build_ops.py ADDED
@@ -0,0 +1,210 @@
1
+ """The custom-database build use-case, shared by the CLI and MCP.
2
+
3
+ Turns a user-supplied FASTA into a fully built gapit-native database via the
4
+ existing pipeline (records.jsonl -> sequences + BLAST index + manifest,
5
+ written last). This module is orchestration plus a metadata merge only — the
6
+ building blocks live in records/dbbuild/fasta/dbcodec/db (SPEC.md §11 covers
7
+ the build pipeline itself). The typer command (:mod:`gapit.cmd_db_build`)
8
+ and the MCP tool (:mod:`gapit.mcp_tools`) are thin callers.
9
+
10
+ Header kind is detected PER RECORD (mixed files allowed): a ``gapit|``
11
+ prefix decodes through the strict tagged codec, anything else through the
12
+ abricate ``~~~`` rules (a plain id carries no ``~~~`` and decodes to itself).
13
+ ``db`` is always the TARGET name and ``source_id`` keeps the original id
14
+ token; record order is input order. Sequences are stored verbatim — no
15
+ provider-style normalization, this is the user's curated truth.
16
+ """
17
+
18
+ import csv
19
+ from collections.abc import Callable
20
+ from dataclasses import dataclass
21
+ from datetime import UTC, datetime
22
+ from pathlib import Path
23
+ from typing import Literal
24
+
25
+ from pydantic import BaseModel
26
+
27
+ from gapit import config
28
+ from gapit.db import mol_type
29
+ from gapit.dbbuild import build_database
30
+ from gapit.dbcodec import decode_seqid
31
+ from gapit.errors import DatabaseError, InputError, UsageError
32
+ from gapit.fasta import FastaRecord, iter_fasta
33
+ from gapit.records import Record, write_records
34
+
35
+ Dbtype = Literal["nucl", "prot"]
36
+
37
+
38
+ class BuildReceipt(BaseModel, frozen=True):
39
+ """One-line JSON success receipt for `db build` (mirrors db fetch)."""
40
+
41
+ db: str
42
+ records: int
43
+ dbtype: Dbtype
44
+ destination: str
45
+
46
+
47
+ @dataclass(frozen=True, slots=True)
48
+ class Metadata:
49
+ """Parsed --tsv: which merge columns the header carries, and the row
50
+ values per gene (accession, ';'-joined function; '' for absent columns)."""
51
+
52
+ columns: frozenset[str]
53
+ rows: dict[str, tuple[str, str]]
54
+
55
+
56
+ def _split_function(joined: str) -> tuple[str, ...]:
57
+ """';'-joined classes -> tuple, empty pieces dropped."""
58
+ return tuple(part for part in joined.split(";") if part)
59
+
60
+
61
+ def _to_record(fasta: FastaRecord, name: str, default_product: str) -> Record:
62
+ """One FASTA record -> one Record.
63
+
64
+ decode_seqid routes by header kind; the product prefers the FASTA
65
+ description, then --description, then the gene (the provider-side
66
+ ``description or gene`` pattern). Malformed ``gapit|`` headers raise
67
+ their native HEADER_MALFORMED DatabaseError.
68
+ """
69
+ header = decode_seqid(fasta.id, default_db=name)
70
+ return Record(
71
+ db=name,
72
+ gene=header.gene,
73
+ sequence=fasta.sequence,
74
+ accession=header.accession,
75
+ function=_split_function(header.function),
76
+ product=fasta.description or default_product or header.gene,
77
+ source_id=fasta.id,
78
+ )
79
+
80
+
81
+ def _read_metadata(path: Path, warn: Callable[[str], None]) -> Metadata:
82
+ """Parse the metadata TSV: header row mandatory and must carry ``gene``
83
+ (InputError METADATA_MALFORMED otherwise); ``accession``/``function``
84
+ are optional per-file, extra columns are ignored. Duplicate genes keep
85
+ the FIRST row (warn); short rows read their missing cells as empty.
86
+ """
87
+ if not path.is_file():
88
+ raise InputError(
89
+ f"metadata file not found or unreadable: {path}",
90
+ code="INPUT_NOT_FOUND",
91
+ context={"file": str(path)},
92
+ )
93
+ with path.open(encoding="utf-8", newline="") as handle:
94
+ rows = csv.reader(handle, delimiter="\t")
95
+ header = next((row for row in rows if row), None)
96
+ if header is None or "gene" not in header:
97
+ raise InputError(
98
+ f"metadata TSV needs a header row with a 'gene' column: {path}",
99
+ code="METADATA_MALFORMED",
100
+ context={"file": str(path), "needed": "gene"},
101
+ )
102
+ columns = {column: index for index, column in enumerate(header)}
103
+
104
+ def cell(padded: list[str], column: str) -> str:
105
+ return padded[columns[column]]
106
+
107
+ metadata: dict[str, tuple[str, str]] = {}
108
+ for row in rows:
109
+ if not any(value.strip() for value in row):
110
+ continue
111
+ padded = row + [""] * (len(header) - len(row))
112
+ gene = cell(padded, "gene")
113
+ if not gene:
114
+ continue
115
+ if gene in metadata:
116
+ warn(f"duplicate gene {gene!r} in metadata TSV: keeping the first row")
117
+ continue
118
+ metadata[gene] = (
119
+ cell(padded, "accession") if "accession" in columns else "",
120
+ cell(padded, "function") if "function" in columns else "",
121
+ )
122
+ return Metadata(columns=frozenset(columns), rows=metadata)
123
+
124
+
125
+ def _merge(records: list[Record], metadata: Metadata, warn: Callable[[str], None]) -> list[Record]:
126
+ """Overwrite accession/function on records whose gene has a metadata row
127
+ (only the columns the TSV actually carries); genes only present in the
128
+ TSV are warned about and skipped."""
129
+ merged: list[Record] = []
130
+ for record in records:
131
+ row = metadata.rows.get(record.gene)
132
+ if row is None:
133
+ merged.append(record)
134
+ continue
135
+ update: dict[str, object] = {}
136
+ if "accession" in metadata.columns:
137
+ update["accession"] = row[0]
138
+ if "function" in metadata.columns:
139
+ update["function"] = _split_function(row[1])
140
+ merged.append(record.model_copy(update=update))
141
+ fasta_genes = {record.gene for record in records}
142
+ for gene in sorted(set(metadata.rows) - fasta_genes):
143
+ warn(f"gene {gene!r} in metadata TSV not found in FASTA: skipped")
144
+ return merged
145
+
146
+
147
+ def _resolve_dbtype(flag: Dbtype | None, records: list[Record]) -> Dbtype:
148
+ """Explicit --dbtype wins; else the abricate mol_type heuristic over the
149
+ concatenated input sequences (db.mol_type, SPEC.md §2)."""
150
+ if flag is not None:
151
+ return flag
152
+ return mol_type("".join(record.sequence for record in records))
153
+
154
+
155
+ def perform_build(
156
+ name: str,
157
+ fasta: Path,
158
+ tsv: Path | None,
159
+ dbtype: Dbtype | None,
160
+ description: str,
161
+ datadir: Path | None,
162
+ force: bool,
163
+ *,
164
+ warn: Callable[[str], None],
165
+ quiet: bool = True,
166
+ ) -> BuildReceipt:
167
+ """Run the custom-build pipeline and return the receipt — the shared CLI
168
+ + MCP path. Warnings go to the caller-supplied ``warn`` (CLI: stderr;
169
+ MCP: dropped — stderr is reserved for the protocol)."""
170
+
171
+ # Security/frozen rule: `Path(datadir) / name` REPLACES the base when name
172
+ # is absolute (and `..` escapes it); plain names only, all else allowed.
173
+ if not name or name in (".", "..") or "/" in name or "\\" in name:
174
+ raise UsageError(
175
+ f"database name must be a plain name without path separators: {name!r}",
176
+ code="USAGE_ERROR",
177
+ context={"name": name},
178
+ )
179
+ if not fasta.is_file():
180
+ raise InputError(
181
+ f"FASTA file not found or unreadable: {fasta}",
182
+ code="INPUT_NOT_FOUND",
183
+ context={"file": str(fasta)},
184
+ )
185
+ db_dir = config.ensure_datadir(datadir) / name
186
+ if (db_dir / "gapit-manifest.json").is_file() and not force:
187
+ raise DatabaseError(
188
+ f"won't overwrite existing database {name} (use --force)",
189
+ code="DB_ALREADY_EXISTS",
190
+ context={"db": name},
191
+ )
192
+ records = [_to_record(fasta_record, name, description) for fasta_record in iter_fasta(fasta)]
193
+ if tsv is not None:
194
+ records = _merge(records, _read_metadata(tsv, warn), warn)
195
+ db_dir.mkdir(parents=True, exist_ok=True)
196
+ write_records(records, db_dir / "records.jsonl")
197
+ manifest = build_database(
198
+ db_dir,
199
+ name=name,
200
+ dbtype=_resolve_dbtype(dbtype, records),
201
+ source_urls=("local",),
202
+ fetched_at=datetime.now(UTC).strftime("%Y-%m-%dT%H:%M:%SZ"),
203
+ quiet=quiet,
204
+ )
205
+ return BuildReceipt(
206
+ db=name,
207
+ records=manifest.n_records,
208
+ dbtype=manifest.dbtype,
209
+ destination=str(db_dir),
210
+ )
gapit/db_ops.py ADDED
@@ -0,0 +1,128 @@
1
+ """Database provider use-cases shared by the CLI and MCP: fetch and list.
2
+
3
+ ``perform_fetch`` installs provider database(s) under the datadir (bundled
4
+ snapshot first, ``from_source`` forces upstream); the ``db_list_*``
5
+ callables build the gapit.dblist/1 provider listing. The typer commands
6
+ (:mod:`gapit.cmd_db`) and the MCP tools (:mod:`gapit.mcp_tools`) are thin
7
+ callers — this module owns the behavior and imports no CLI plumbing.
8
+ """
9
+
10
+ from collections.abc import Iterator
11
+ from datetime import UTC, datetime
12
+ from pathlib import Path
13
+ from typing import Literal
14
+
15
+ from pydantic import BaseModel, ConfigDict, Field
16
+
17
+ from gapit import config
18
+ from gapit.errors import UsageError
19
+ from gapit.providers import REGISTRY
20
+ from gapit.providers.common import Dbtype, fetch_provider
21
+ from gapit.records import read_manifest
22
+
23
+ # Bare `gapit db fetch` installs these, in order — the two providers whose
24
+ # snapshots ship inside the wheel (Wave G: zero-network bootstrap).
25
+ DEFAULT_DBS = ("card", "vfdb")
26
+
27
+
28
+ class ProviderReceipt(BaseModel, frozen=True):
29
+ """One-line JSON success receipt for `db fetch` (no biological data)."""
30
+
31
+ db: str
32
+ records: int
33
+ dbtype: Dbtype
34
+ destination: str
35
+
36
+
37
+ class DbListEntry(BaseModel, frozen=True):
38
+ """One provider row in the gapit.dblist/1 listing document."""
39
+
40
+ name: str
41
+ description: str
42
+ dbtype: str
43
+ installed: bool
44
+ records: int | None = None
45
+
46
+
47
+ class DbListDocument(BaseModel, frozen=True):
48
+ """gapit.dblist/1 — `gapit db list --json` output.
49
+
50
+ A CLI listing, deliberately local to this module: NOT registered in
51
+ `gapit schema` (the public schema surface stays unchanged this wave).
52
+ """
53
+
54
+ model_config = ConfigDict(populate_by_name=True)
55
+
56
+ schema_name: Literal["gapit.dblist/1"] = Field(default="gapit.dblist/1", alias="schema")
57
+ providers: tuple[DbListEntry, ...]
58
+
59
+
60
+ def perform_fetch(
61
+ name: str | None,
62
+ datadir: Path | None,
63
+ *,
64
+ force: bool = False,
65
+ from_source: bool = False,
66
+ quiet: bool = True,
67
+ debug: bool = False,
68
+ ) -> Iterator[ProviderReceipt]:
69
+ """Fetch provider database(s) into <datadir>/NAME — the shared CLI + MCP
70
+ path. NAME None installs every database in DEFAULT_DBS order; each
71
+ receipt yields as its install completes (streaming, like the CLI's
72
+ per-db stdout lines).
73
+ """
74
+
75
+ def fetch_one(provider_name: str) -> ProviderReceipt:
76
+ provider = REGISTRY.get(provider_name)
77
+ if provider is None:
78
+ raise UsageError(
79
+ f"unknown provider: {provider_name} (available: {', '.join(sorted(REGISTRY))})",
80
+ code="USAGE_ERROR",
81
+ context={"provider": provider_name},
82
+ )
83
+ db_dir = config.ensure_datadir(datadir) / provider_name
84
+ manifest = fetch_provider(
85
+ provider,
86
+ db_dir,
87
+ fetched_at=datetime.now(UTC).strftime("%Y-%m-%dT%H:%M:%SZ"),
88
+ force=force,
89
+ quiet=quiet,
90
+ from_source=from_source,
91
+ debug=debug,
92
+ )
93
+ return ProviderReceipt(
94
+ db=provider_name,
95
+ records=manifest.n_records,
96
+ dbtype=manifest.dbtype,
97
+ destination=str(db_dir),
98
+ )
99
+
100
+ for provider_name in (name,) if name is not None else DEFAULT_DBS:
101
+ yield fetch_one(provider_name)
102
+
103
+
104
+ def db_list_entries(root: Path) -> list[DbListEntry]:
105
+ """Provider rows for the gapit.dblist/1 listing — shared CLI + MCP path."""
106
+ entries: list[DbListEntry] = []
107
+ for provider_name in sorted(REGISTRY):
108
+ provider = REGISTRY[provider_name]
109
+ manifest_path = root / provider_name / "gapit-manifest.json"
110
+ installed = manifest_path.is_file()
111
+ entries.append(
112
+ DbListEntry(
113
+ name=provider_name,
114
+ description=provider.description,
115
+ dbtype=provider.dbtype,
116
+ installed=installed,
117
+ records=read_manifest(manifest_path).n_records if installed else None,
118
+ )
119
+ )
120
+ return entries
121
+
122
+
123
+ def db_list_json(root: Path) -> str:
124
+ """The gapit.dblist/1 document JSON — shared CLI + MCP path."""
125
+ entries = db_list_entries(root)
126
+ return DbListDocument(providers=tuple(entries)).model_dump_json(
127
+ indent=2, by_alias=True, exclude_none=True
128
+ )
gapit/db_query_ops.py ADDED
@@ -0,0 +1,252 @@
1
+ """Read-only queries over installed databases, shared by the CLI and MCP.
2
+
3
+ - ``perform_search``: case-insensitive lookup across the ``records.jsonl``
4
+ truth stores of every installed database (SPEC.md §11).
5
+ - ``perform_outdated``: installed-database staleness report against a
6
+ threshold (``days``) and the bundled snapshot dates. Staleness is a
7
+ REPORT, never an error state.
8
+
9
+ Both are read-only: no builds, no network, no datadir writes. The typer
10
+ commands (:mod:`gapit.cmd_db_search`, :mod:`gapit.cmd_db_outdated`) and the
11
+ MCP tools (:mod:`gapit.mcp_tools`) are thin callers.
12
+ """
13
+
14
+ import enum
15
+ from collections.abc import Callable, Iterable
16
+ from datetime import UTC, datetime
17
+ from pathlib import Path
18
+ from typing import Literal
19
+
20
+ from pydantic import BaseModel, ConfigDict, Field
21
+
22
+ from gapit import config
23
+ from gapit.errors import DatabaseError, InputError, UsageError
24
+ from gapit.proctools import note
25
+ from gapit.providers import REGISTRY
26
+ from gapit.providers.common import bundled_snapshot_manifest
27
+ from gapit.records import Record, installed_db_dirs, read_manifest, read_records
28
+
29
+ DEFAULT_LIMIT = 100
30
+ DEFAULT_STALE_DAYS = 90
31
+
32
+
33
+ class SearchField(enum.Enum):
34
+ """Record fields `db search` can match against."""
35
+
36
+ gene = "gene"
37
+ accession = "accession"
38
+ function = "function"
39
+ product = "product"
40
+ any = "any"
41
+
42
+
43
+ class SearchEntry(BaseModel, frozen=True):
44
+ """One search hit: the JSONL line (and the TSV row's source), a CLI
45
+ listing shape — not a registered schema."""
46
+
47
+ db: str
48
+ gene: str
49
+ accession: str
50
+ function: tuple[str, ...]
51
+ product: str
52
+ length: int
53
+
54
+
55
+ def _candidates(record: Record, field: SearchField) -> tuple[str, ...]:
56
+ """The string values of the chosen field (function contributes one per class)."""
57
+ match field:
58
+ case SearchField.gene:
59
+ return (record.gene,)
60
+ case SearchField.accession:
61
+ return (record.accession,)
62
+ case SearchField.function:
63
+ return record.function
64
+ case SearchField.product:
65
+ return (record.product,)
66
+ case SearchField.any:
67
+ return (record.gene, record.accession, *record.function, record.product)
68
+
69
+
70
+ def _matches(record: Record, needle: str, field: SearchField, exact: bool) -> bool:
71
+ """Case-insensitive substring by default; --exact is full-field equality."""
72
+ values = _candidates(record, field)
73
+ if exact:
74
+ return any(value.lower() == needle for value in values)
75
+ return any(needle in value.lower() for value in values)
76
+
77
+
78
+ def search_tsv_row(record: Record) -> str:
79
+ """The hit row: DB\tGENE\tACCESSION\tFUNCTION\tPRODUCT\tLENGTH."""
80
+ return (
81
+ f"{record.db}\t{record.gene}\t{record.accession}\t"
82
+ f"{';'.join(record.function)}\t{record.product}\t{len(record.sequence)}"
83
+ )
84
+
85
+
86
+ def search_json_line(record: Record) -> str:
87
+ """The hit as one JSONL line (same fields as the TSV row, snake_case)."""
88
+ return SearchEntry(
89
+ db=record.db,
90
+ gene=record.gene,
91
+ accession=record.accession,
92
+ function=record.function,
93
+ product=record.product,
94
+ length=len(record.sequence),
95
+ ).model_dump_json()
96
+
97
+
98
+ def perform_search(
99
+ term: str,
100
+ datadir: Path | None,
101
+ *,
102
+ db: str | None = None,
103
+ field: SearchField = SearchField.any,
104
+ exact: bool = False,
105
+ limit: int = DEFAULT_LIMIT,
106
+ render: Callable[[Record], str] = search_tsv_row,
107
+ quiet: bool = True,
108
+ ) -> tuple[list[str], int]:
109
+ """Scan the installed records.jsonl stores — the shared CLI + MCP path.
110
+
111
+ Returns ``(hit lines up to limit, total matching count)``; the caller
112
+ owns the truncation note (a stderr diagnostic) and the printing.
113
+ """
114
+ root = config.resolve_datadir(datadir)
115
+ db_dirs = installed_db_dirs(root)
116
+ if db is not None and db not in {db_dir.name for db_dir in db_dirs}:
117
+ installed = " ".join(sorted(db_dir.name for db_dir in db_dirs))
118
+ raise UsageError(
119
+ f"unknown database: {db} (installed: {installed or 'none'})",
120
+ code="USAGE_ERROR",
121
+ context={"db": db, "installed": installed},
122
+ )
123
+ needle = term.lower()
124
+ hits: list[str] = []
125
+ total = 0
126
+ for db_dir in db_dirs:
127
+ if db is not None and db_dir.name != db:
128
+ continue
129
+ records_path = db_dir / "records.jsonl"
130
+ if not records_path.is_file():
131
+ if db is not None:
132
+ raise DatabaseError(
133
+ f"database {db_dir.name} is incomplete: records.jsonl missing",
134
+ code="DB_INCOMPLETE",
135
+ context={"db": db_dir.name},
136
+ )
137
+ note(quiet, f"skipping {db_dir.name}: no records.jsonl (incomplete database)")
138
+ continue
139
+ for record in read_records(records_path):
140
+ if _matches(record, needle, field, exact):
141
+ total += 1
142
+ if limit == 0 or len(hits) < limit:
143
+ hits.append(render(record))
144
+ return hits, total
145
+
146
+
147
+ class DbOutdatedEntry(BaseModel, frozen=True):
148
+ """One installed-database row in the gapit.dboutdated/1 report."""
149
+
150
+ db: str
151
+ fetched_at: str
152
+ age_days: float
153
+ status: Literal["ok", "stale", "snapshot-update", "stale+snapshot-update"]
154
+ upstream_version: str
155
+
156
+
157
+ class DbOutdatedDocument(BaseModel, frozen=True):
158
+ """gapit.dboutdated/1 — `gapit db outdated --json` output.
159
+
160
+ A CLI listing, deliberately local to this module: NOT registered in
161
+ `gapit schema` (mirrors the gapit.dblist/1 precedent).
162
+ """
163
+
164
+ model_config = ConfigDict(populate_by_name=True)
165
+
166
+ schema_name: Literal["gapit.dboutdated/1"] = Field(default="gapit.dboutdated/1", alias="schema")
167
+ databases: tuple[DbOutdatedEntry, ...]
168
+
169
+
170
+ def _utc_timestamp(value: str, where: str) -> datetime:
171
+ """Parse a manifest fetched_at into aware UTC; anything else is a
172
+ malformed manifest (trusted metadata: writers emit ISO-8601 with Z)."""
173
+ try:
174
+ parsed = datetime.fromisoformat(value)
175
+ except ValueError as exc:
176
+ raise InputError(
177
+ f"malformed manifest {where}: invalid fetched_at: {value}",
178
+ code="MANIFEST_MALFORMED",
179
+ context={"file": where},
180
+ ) from exc
181
+ if parsed.tzinfo is None:
182
+ raise InputError(
183
+ f"malformed manifest {where}: fetched_at has no UTC offset: {value}",
184
+ code="MANIFEST_MALFORMED",
185
+ context={"file": where},
186
+ )
187
+ return parsed.astimezone(UTC)
188
+
189
+
190
+ def _snapshot_newer(name: str, fetched: datetime) -> bool:
191
+ """True when the provider's bundled snapshot manifest is newer than the
192
+ installed ``fetched`` (unknown provider / no bundle -> False)."""
193
+ provider = REGISTRY.get(name)
194
+ if provider is None:
195
+ return False
196
+ archived = bundled_snapshot_manifest(provider)
197
+ return archived is not None and fetched < _utc_timestamp(
198
+ archived.fetched_at, f"bundled snapshot of {name}"
199
+ )
200
+
201
+
202
+ def perform_outdated(
203
+ datadir: Path | None, *, days: int = DEFAULT_STALE_DAYS
204
+ ) -> list[DbOutdatedEntry]:
205
+ """Compute the staleness report entries — the shared CLI + MCP path.
206
+
207
+ A database is `stale` past ``days`` and `snapshot-update` when its
208
+ provider's bundled snapshot is newer than the installed copy.
209
+ """
210
+ root = config.resolve_datadir(datadir)
211
+ db_dirs = installed_db_dirs(root)
212
+ if not db_dirs:
213
+ raise DatabaseError(
214
+ f"no installed databases in datadir: {root}",
215
+ code="DATADIR_EMPTY",
216
+ context={"datadir": str(root)},
217
+ )
218
+ now = datetime.now(UTC)
219
+ entries: list[DbOutdatedEntry] = []
220
+ for db_dir in db_dirs:
221
+ manifest_path = db_dir / "gapit-manifest.json"
222
+ manifest = read_manifest(manifest_path)
223
+ fetched = _utc_timestamp(manifest.fetched_at, str(manifest_path))
224
+ age = (now - fetched).total_seconds() / 86400
225
+ match (age > days, _snapshot_newer(db_dir.name, fetched)):
226
+ case (True, True):
227
+ status = "stale+snapshot-update"
228
+ case (True, False):
229
+ status = "stale"
230
+ case (False, True):
231
+ status = "snapshot-update"
232
+ case (False, False):
233
+ status = "ok"
234
+ entries.append(
235
+ DbOutdatedEntry(
236
+ db=db_dir.name,
237
+ fetched_at=manifest.fetched_at,
238
+ age_days=round(age, 2),
239
+ status=status,
240
+ upstream_version=manifest.upstream_version,
241
+ )
242
+ )
243
+ return entries
244
+
245
+
246
+ def outdated_tsv_lines(entries: Iterable[DbOutdatedEntry]) -> list[str]:
247
+ """The NAME/FETCHED_AT/AGE_DAYS/STATUS table lines (CLI stdout = MCP text)."""
248
+ lines = ["NAME\tFETCHED_AT\tAGE_DAYS\tSTATUS"]
249
+ lines.extend(
250
+ f"{entry.db}\t{entry.fetched_at}\t{entry.age_days:.2f}\t{entry.status}" for entry in entries
251
+ )
252
+ return lines