gapit 0.2.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gapit/__init__.py +3 -0
- gapit/blast.py +239 -0
- gapit/cli.py +128 -0
- gapit/cmd_db.py +113 -0
- gapit/cmd_db_build.py +72 -0
- gapit/cmd_db_install.py +126 -0
- gapit/cmd_db_outdated.py +51 -0
- gapit/cmd_db_search.py +66 -0
- gapit/cmd_screen.py +214 -0
- gapit/cmd_summary.py +65 -0
- gapit/config.py +51 -0
- gapit/data/snapshots/card.tar.gz +0 -0
- gapit/data/snapshots/vfdb.tar.gz +0 -0
- gapit/db.py +226 -0
- gapit/db_build_ops.py +210 -0
- gapit/db_ops.py +128 -0
- gapit/db_query_ops.py +252 -0
- gapit/dbbuild.py +207 -0
- gapit/dbcodec.py +117 -0
- gapit/dispatch.py +31 -0
- gapit/errors.py +87 -0
- gapit/fasta.py +123 -0
- gapit/formats/__init__.py +1 -0
- gapit/formats/json.py +309 -0
- gapit/formats/md.py +190 -0
- gapit/formats/schemas.py +30 -0
- gapit/formats/summary.py +103 -0
- gapit/formats/tsv.py +45 -0
- gapit/hits.py +107 -0
- gapit/mcp.py +158 -0
- gapit/mcp_schemas.py +123 -0
- gapit/mcp_tools.py +289 -0
- gapit/minimap.py +20 -0
- gapit/minimap2_run.py +114 -0
- gapit/paf.py +115 -0
- gapit/proctools.py +24 -0
- gapit/providers/__init__.py +39 -0
- gapit/providers/argannot.py +94 -0
- gapit/providers/bacmet2.py +59 -0
- gapit/providers/card.py +150 -0
- gapit/providers/common.py +245 -0
- gapit/providers/ecoh.py +63 -0
- gapit/providers/ecoli_vf.py +74 -0
- gapit/providers/megares.py +71 -0
- gapit/providers/ncbi.py +103 -0
- gapit/providers/plasmidfinder.py +69 -0
- gapit/providers/resfinder.py +123 -0
- gapit/providers/snapshots.py +119 -0
- gapit/providers/upec_expec_vf.py +85 -0
- gapit/providers/vfdb.py +92 -0
- gapit/providers/victors.py +109 -0
- gapit/py.typed +0 -0
- gapit/reads.py +221 -0
- gapit/records.py +152 -0
- gapit/report.py +25 -0
- gapit/screening.py +145 -0
- gapit/screening_reads.py +255 -0
- gapit/seqconvert.py +203 -0
- gapit/summary.py +151 -0
- gapit-0.2.2.dist-info/METADATA +183 -0
- gapit-0.2.2.dist-info/RECORD +64 -0
- gapit-0.2.2.dist-info/WHEEL +4 -0
- gapit-0.2.2.dist-info/entry_points.txt +3 -0
- gapit-0.2.2.dist-info/licenses/LICENSE +21 -0
gapit/formats/json.py
ADDED
|
@@ -0,0 +1,309 @@
|
|
|
1
|
+
"""Agent-facing JSON documents: gapit.report/1, gapit.list/1, gapit.version/1."""
|
|
2
|
+
|
|
3
|
+
from collections.abc import Iterable
|
|
4
|
+
from datetime import UTC, datetime
|
|
5
|
+
from typing import Literal
|
|
6
|
+
|
|
7
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
8
|
+
|
|
9
|
+
from gapit import __version__
|
|
10
|
+
from gapit.hits import Hit
|
|
11
|
+
from gapit.reads import GeneCoverage, ReadsParams, ReadsReport
|
|
12
|
+
from gapit.report import Report, ScreeningParams
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class HitDocument(BaseModel, frozen=True):
|
|
16
|
+
"""One hit — 1:1 with the TSV columns, snake_case, same string values."""
|
|
17
|
+
|
|
18
|
+
sequence: str
|
|
19
|
+
start: int
|
|
20
|
+
end: int
|
|
21
|
+
strand: str
|
|
22
|
+
gene: str
|
|
23
|
+
coverage: str
|
|
24
|
+
coverage_map: str
|
|
25
|
+
gaps: str
|
|
26
|
+
coverage_pct: float
|
|
27
|
+
identity_pct: float
|
|
28
|
+
database: str
|
|
29
|
+
accession: str
|
|
30
|
+
product: str
|
|
31
|
+
# "resistance" is frozen by gapit.report/1 (1:1 with the TSV RESISTANCE
|
|
32
|
+
# column); the value flows from Hit.function and carries functional
|
|
33
|
+
# categories for native DBs (Wave F1 renamed the internal slot only).
|
|
34
|
+
resistance: str
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class FileDocument(BaseModel, frozen=True):
|
|
38
|
+
"""One screened file and its hits (zero hits -> empty list)."""
|
|
39
|
+
|
|
40
|
+
file: str
|
|
41
|
+
hits: list[HitDocument]
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class ParamsDocument(BaseModel, frozen=True):
|
|
45
|
+
"""The screening parameters in effect."""
|
|
46
|
+
|
|
47
|
+
db: str
|
|
48
|
+
minid: float
|
|
49
|
+
mincov: float
|
|
50
|
+
threads: int
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class ToolDocument(BaseModel, frozen=True):
|
|
54
|
+
"""Tool self-identification."""
|
|
55
|
+
|
|
56
|
+
name: str = "gapit"
|
|
57
|
+
version: str = __version__
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
class ReportDocument(BaseModel, frozen=True):
|
|
61
|
+
"""gapit.report/1 — the canonical machine-readable screening output."""
|
|
62
|
+
|
|
63
|
+
model_config = ConfigDict(populate_by_name=True)
|
|
64
|
+
|
|
65
|
+
schema_name: Literal["gapit.report/1"] = Field(default="gapit.report/1", alias="schema")
|
|
66
|
+
tool: ToolDocument = ToolDocument()
|
|
67
|
+
created_at: str
|
|
68
|
+
params: ParamsDocument
|
|
69
|
+
files: list[FileDocument]
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
class ListEntryDocument(BaseModel, frozen=True):
|
|
73
|
+
"""One row of `gapit list --json`."""
|
|
74
|
+
|
|
75
|
+
name: str
|
|
76
|
+
sequences: int
|
|
77
|
+
dbtype: str
|
|
78
|
+
date: str
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
class ListDocument(BaseModel, frozen=True):
|
|
82
|
+
"""gapit.list/1."""
|
|
83
|
+
|
|
84
|
+
model_config = ConfigDict(populate_by_name=True)
|
|
85
|
+
|
|
86
|
+
schema_name: Literal["gapit.list/1"] = Field(default="gapit.list/1", alias="schema")
|
|
87
|
+
databases: list[ListEntryDocument]
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
class VersionDocument(BaseModel, frozen=True):
|
|
91
|
+
"""gapit.version/1 — compact one-line self-description."""
|
|
92
|
+
|
|
93
|
+
model_config = ConfigDict(populate_by_name=True)
|
|
94
|
+
|
|
95
|
+
schema_name: Literal["gapit.version/1"] = Field(default="gapit.version/1", alias="schema")
|
|
96
|
+
name: str = "gapit"
|
|
97
|
+
version: str
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
class GeneCoverageDocument(BaseModel, frozen=True):
|
|
101
|
+
"""One gene's presence call in reads mode."""
|
|
102
|
+
|
|
103
|
+
gene: str
|
|
104
|
+
database: str
|
|
105
|
+
accession: str
|
|
106
|
+
product: str
|
|
107
|
+
# "resistance" is frozen by gapit.reads/1; the value flows from
|
|
108
|
+
# GeneCoverage.function (functional categories for native DBs).
|
|
109
|
+
resistance: str
|
|
110
|
+
tlen: int
|
|
111
|
+
breadth_pct: float
|
|
112
|
+
mean_depth: float
|
|
113
|
+
reads_mapped: int
|
|
114
|
+
present: bool
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
class ReadsFileDocument(BaseModel, frozen=True):
|
|
118
|
+
"""One screened read set."""
|
|
119
|
+
|
|
120
|
+
reads: list[str]
|
|
121
|
+
genes: list[GeneCoverageDocument]
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
class ReadsParamsDocument(BaseModel, frozen=True):
|
|
125
|
+
"""Read-screening parameters in effect."""
|
|
126
|
+
|
|
127
|
+
db: str
|
|
128
|
+
read_type: str
|
|
129
|
+
min_breadth: float
|
|
130
|
+
threads: int
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
class ReadsDocument(BaseModel, frozen=True):
|
|
134
|
+
"""gapit.reads/1 — machine-readable read-screening output."""
|
|
135
|
+
|
|
136
|
+
model_config = ConfigDict(populate_by_name=True)
|
|
137
|
+
|
|
138
|
+
schema_name: Literal["gapit.reads/1"] = Field(default="gapit.reads/1", alias="schema")
|
|
139
|
+
tool: ToolDocument = ToolDocument()
|
|
140
|
+
created_at: str
|
|
141
|
+
params: ReadsParamsDocument
|
|
142
|
+
files: list[ReadsFileDocument]
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
class GeneCoverage2Document(BaseModel, frozen=True):
|
|
146
|
+
"""One gene's presence call in reads/2 mode: the reads/1 fields plus the
|
|
147
|
+
alen-weighted mean per-alignment identity."""
|
|
148
|
+
|
|
149
|
+
gene: str
|
|
150
|
+
database: str
|
|
151
|
+
accession: str
|
|
152
|
+
product: str
|
|
153
|
+
resistance: str
|
|
154
|
+
tlen: int
|
|
155
|
+
breadth_pct: float
|
|
156
|
+
mean_depth: float
|
|
157
|
+
reads_mapped: int
|
|
158
|
+
present: bool
|
|
159
|
+
mean_identity_pct: float
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
class Reads2FileDocument(BaseModel, frozen=True):
|
|
163
|
+
"""One screened read set (reads/2)."""
|
|
164
|
+
|
|
165
|
+
reads: list[str]
|
|
166
|
+
genes: list[GeneCoverage2Document]
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
class Reads2ParamsDocument(BaseModel, frozen=True):
|
|
170
|
+
"""Read-screening parameters in effect (reads/2): reads/1 params plus the
|
|
171
|
+
opt-in alignment filters."""
|
|
172
|
+
|
|
173
|
+
db: str
|
|
174
|
+
read_type: str
|
|
175
|
+
min_breadth: float
|
|
176
|
+
threads: int
|
|
177
|
+
min_identity: float
|
|
178
|
+
min_mapq: int
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
class Reads2Document(BaseModel, frozen=True):
|
|
182
|
+
"""gapit.reads/2 — read-screening output under opt-in identity/MAPQ
|
|
183
|
+
alignment filtering (reads/1 stays the default and is frozen)."""
|
|
184
|
+
|
|
185
|
+
model_config = ConfigDict(populate_by_name=True)
|
|
186
|
+
|
|
187
|
+
schema_name: Literal["gapit.reads/2"] = Field(default="gapit.reads/2", alias="schema")
|
|
188
|
+
tool: ToolDocument = ToolDocument()
|
|
189
|
+
created_at: str
|
|
190
|
+
params: Reads2ParamsDocument
|
|
191
|
+
files: list[Reads2FileDocument]
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def utc_timestamp(now: datetime) -> str:
|
|
195
|
+
"""ISO-8601 UTC with a trailing Z, second precision."""
|
|
196
|
+
return now.astimezone(UTC).strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def _hit_document(hit: Hit) -> HitDocument:
|
|
200
|
+
return HitDocument(
|
|
201
|
+
sequence=hit.sequence,
|
|
202
|
+
start=hit.start,
|
|
203
|
+
end=hit.end,
|
|
204
|
+
strand=hit.strand,
|
|
205
|
+
gene=hit.gene,
|
|
206
|
+
coverage=f"{hit.s_start}-{hit.s_end}/{hit.s_len}",
|
|
207
|
+
coverage_map=hit.coverage_map,
|
|
208
|
+
gaps=f"{hit.gap_openings}/{hit.gaps}",
|
|
209
|
+
coverage_pct=round(hit.coverage_pct, 2),
|
|
210
|
+
identity_pct=round(hit.identity_pct, 2),
|
|
211
|
+
database=hit.database,
|
|
212
|
+
accession=hit.accession,
|
|
213
|
+
product=hit.product,
|
|
214
|
+
resistance=hit.function,
|
|
215
|
+
)
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def render_json(reports: Iterable[Report], params: ScreeningParams, *, now: datetime) -> str:
|
|
219
|
+
"""Serialize screening results as gapit.report/1 (indented, schema first)."""
|
|
220
|
+
document = ReportDocument(
|
|
221
|
+
created_at=utc_timestamp(now),
|
|
222
|
+
params=ParamsDocument(
|
|
223
|
+
db=params.db, minid=params.minid, mincov=params.mincov, threads=params.threads
|
|
224
|
+
),
|
|
225
|
+
files=[
|
|
226
|
+
FileDocument(file=report.file, hits=[_hit_document(hit) for hit in report.hits])
|
|
227
|
+
for report in reports
|
|
228
|
+
],
|
|
229
|
+
)
|
|
230
|
+
return document.model_dump_json(indent=2, by_alias=True)
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def _gene_coverage_document(gene: GeneCoverage) -> GeneCoverageDocument:
|
|
234
|
+
return GeneCoverageDocument(
|
|
235
|
+
gene=gene.gene,
|
|
236
|
+
database=gene.database,
|
|
237
|
+
accession=gene.accession,
|
|
238
|
+
product=gene.product,
|
|
239
|
+
resistance=gene.function,
|
|
240
|
+
tlen=gene.tlen,
|
|
241
|
+
breadth_pct=round(gene.breadth_pct, 2),
|
|
242
|
+
mean_depth=round(gene.mean_depth, 2),
|
|
243
|
+
reads_mapped=gene.reads_mapped,
|
|
244
|
+
present=gene.present,
|
|
245
|
+
)
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def render_reads_json(reports: Iterable[ReadsReport], params: ReadsParams, *, now: datetime) -> str:
|
|
249
|
+
"""Serialize read-screening results as gapit.reads/1 (indented, schema first)."""
|
|
250
|
+
document = ReadsDocument(
|
|
251
|
+
created_at=utc_timestamp(now),
|
|
252
|
+
params=ReadsParamsDocument(
|
|
253
|
+
db=params.db,
|
|
254
|
+
read_type=params.read_type,
|
|
255
|
+
min_breadth=params.min_breadth,
|
|
256
|
+
threads=params.threads,
|
|
257
|
+
),
|
|
258
|
+
files=[
|
|
259
|
+
ReadsFileDocument(
|
|
260
|
+
reads=list(report.reads),
|
|
261
|
+
genes=[_gene_coverage_document(gene) for gene in report.genes],
|
|
262
|
+
)
|
|
263
|
+
for report in reports
|
|
264
|
+
],
|
|
265
|
+
)
|
|
266
|
+
return document.model_dump_json(indent=2, by_alias=True)
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def _gene_coverage2_document(gene: GeneCoverage) -> GeneCoverage2Document:
|
|
270
|
+
return GeneCoverage2Document(
|
|
271
|
+
gene=gene.gene,
|
|
272
|
+
database=gene.database,
|
|
273
|
+
accession=gene.accession,
|
|
274
|
+
product=gene.product,
|
|
275
|
+
resistance=gene.function,
|
|
276
|
+
tlen=gene.tlen,
|
|
277
|
+
breadth_pct=round(gene.breadth_pct, 2),
|
|
278
|
+
mean_depth=round(gene.mean_depth, 2),
|
|
279
|
+
reads_mapped=gene.reads_mapped,
|
|
280
|
+
present=gene.present,
|
|
281
|
+
mean_identity_pct=round(gene.mean_identity_pct, 2),
|
|
282
|
+
)
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
def render_reads2_json(
|
|
286
|
+
reports: Iterable[ReadsReport], params: ReadsParams, *, now: datetime
|
|
287
|
+
) -> str:
|
|
288
|
+
"""Serialize filtered read-screening results as gapit.reads/2 (indented,
|
|
289
|
+
schema first; same shape as /1 plus the filter params and per-gene
|
|
290
|
+
mean_identity_pct)."""
|
|
291
|
+
document = Reads2Document(
|
|
292
|
+
created_at=utc_timestamp(now),
|
|
293
|
+
params=Reads2ParamsDocument(
|
|
294
|
+
db=params.db,
|
|
295
|
+
read_type=params.read_type,
|
|
296
|
+
min_breadth=params.min_breadth,
|
|
297
|
+
threads=params.threads,
|
|
298
|
+
min_identity=params.min_identity,
|
|
299
|
+
min_mapq=params.min_mapq,
|
|
300
|
+
),
|
|
301
|
+
files=[
|
|
302
|
+
Reads2FileDocument(
|
|
303
|
+
reads=list(report.reads),
|
|
304
|
+
genes=[_gene_coverage2_document(gene) for gene in report.genes],
|
|
305
|
+
)
|
|
306
|
+
for report in reports
|
|
307
|
+
],
|
|
308
|
+
)
|
|
309
|
+
return document.model_dump_json(indent=2, by_alias=True)
|
gapit/formats/md.py
ADDED
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
"""Human- and agent-readable Markdown reports (gapit.report/1, gapit.reads/1)."""
|
|
2
|
+
|
|
3
|
+
from collections.abc import Iterable
|
|
4
|
+
from datetime import UTC, datetime
|
|
5
|
+
|
|
6
|
+
from gapit import __version__
|
|
7
|
+
from gapit.reads import ReadsParams, ReadsReport
|
|
8
|
+
from gapit.report import Report, ScreeningParams
|
|
9
|
+
|
|
10
|
+
_COLUMNS = (
|
|
11
|
+
"Sequence",
|
|
12
|
+
"Start",
|
|
13
|
+
"End",
|
|
14
|
+
"Strand",
|
|
15
|
+
"Gene",
|
|
16
|
+
"Coverage",
|
|
17
|
+
"Map",
|
|
18
|
+
"Gaps",
|
|
19
|
+
"%Coverage",
|
|
20
|
+
"%Identity",
|
|
21
|
+
"Database",
|
|
22
|
+
"Accession",
|
|
23
|
+
"Product",
|
|
24
|
+
"Resistance",
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _md_cell(value: str) -> str:
|
|
29
|
+
"""Escape Markdown table metacharacters so cells cannot break the table
|
|
30
|
+
(victors gene ids like ``gi|115534241:2616-3152`` contain pipes)."""
|
|
31
|
+
return value.replace("\\", "\\\\").replace("|", "\\|")
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def render_markdown(reports: Iterable[Report], params: ScreeningParams, *, now: datetime) -> str:
|
|
35
|
+
"""Render screening results as deterministic Markdown: YAML frontmatter,
|
|
36
|
+
one section per file, stable table columns."""
|
|
37
|
+
files = list(reports)
|
|
38
|
+
total_hits = sum(len(report.hits) for report in files)
|
|
39
|
+
lines: list[str] = [
|
|
40
|
+
"---",
|
|
41
|
+
"schema: gapit.report/1",
|
|
42
|
+
f"tool: gapit {__version__}",
|
|
43
|
+
f"created_at: {now.astimezone(UTC).strftime('%Y-%m-%dT%H:%M:%SZ')}",
|
|
44
|
+
f"db: {params.db}",
|
|
45
|
+
f"minid: {params.minid}",
|
|
46
|
+
f"mincov: {params.mincov}",
|
|
47
|
+
f"threads: {params.threads}",
|
|
48
|
+
f"files: {len(files)}",
|
|
49
|
+
f"hits: {total_hits}",
|
|
50
|
+
"---",
|
|
51
|
+
"",
|
|
52
|
+
"# gapit screening report",
|
|
53
|
+
"",
|
|
54
|
+
]
|
|
55
|
+
for report in files:
|
|
56
|
+
lines.append(f"## `{report.file}`")
|
|
57
|
+
lines.append("")
|
|
58
|
+
if not report.hits:
|
|
59
|
+
lines.append("_No hits._")
|
|
60
|
+
lines.append("")
|
|
61
|
+
continue
|
|
62
|
+
lines.append("| " + " | ".join(_COLUMNS) + " |")
|
|
63
|
+
lines.append("|" + "---|" * len(_COLUMNS))
|
|
64
|
+
for hit in report.hits:
|
|
65
|
+
cells = (
|
|
66
|
+
hit.sequence,
|
|
67
|
+
str(hit.start),
|
|
68
|
+
str(hit.end),
|
|
69
|
+
hit.strand,
|
|
70
|
+
hit.gene,
|
|
71
|
+
f"{hit.s_start}-{hit.s_end}/{hit.s_len}",
|
|
72
|
+
hit.coverage_map,
|
|
73
|
+
f"{hit.gap_openings}/{hit.gaps}",
|
|
74
|
+
f"{hit.coverage_pct:.2f}",
|
|
75
|
+
f"{hit.identity_pct:.2f}",
|
|
76
|
+
hit.database,
|
|
77
|
+
hit.accession,
|
|
78
|
+
hit.product,
|
|
79
|
+
hit.function,
|
|
80
|
+
)
|
|
81
|
+
lines.append("| " + " | ".join(_md_cell(cell) for cell in cells) + " |")
|
|
82
|
+
lines.append("")
|
|
83
|
+
return "\n".join(lines) + "\n"
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
_READS_COLUMNS = (
|
|
87
|
+
"Gene",
|
|
88
|
+
"Breadth%",
|
|
89
|
+
"Depth",
|
|
90
|
+
"Reads",
|
|
91
|
+
"Present",
|
|
92
|
+
"Database",
|
|
93
|
+
"Accession",
|
|
94
|
+
"Product",
|
|
95
|
+
"Resistance",
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
_READS2_COLUMNS = (*_READS_COLUMNS, "Identity%")
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _render_reads_markdown(
|
|
102
|
+
reports: Iterable[ReadsReport],
|
|
103
|
+
params: ReadsParams,
|
|
104
|
+
*,
|
|
105
|
+
now: datetime,
|
|
106
|
+
schema: str,
|
|
107
|
+
columns: tuple[str, ...],
|
|
108
|
+
with_identity: bool,
|
|
109
|
+
) -> str:
|
|
110
|
+
"""Shared reads-report Markdown core; gapit.reads/2 adds the two filter
|
|
111
|
+
lines to the frontmatter and an Identity% cell per gene."""
|
|
112
|
+
files = list(reports)
|
|
113
|
+
genes_found = sum(1 for report in files for gene in report.genes if gene.present)
|
|
114
|
+
lines: list[str] = [
|
|
115
|
+
"---",
|
|
116
|
+
f"schema: {schema}",
|
|
117
|
+
f"tool: gapit {__version__}",
|
|
118
|
+
f"created_at: {now.astimezone(UTC).strftime('%Y-%m-%dT%H:%M:%SZ')}",
|
|
119
|
+
f"db: {params.db}",
|
|
120
|
+
f"read_type: {params.read_type}",
|
|
121
|
+
f"min_breadth: {params.min_breadth}",
|
|
122
|
+
f"threads: {params.threads}",
|
|
123
|
+
]
|
|
124
|
+
if with_identity:
|
|
125
|
+
lines.append(f"min_identity: {params.min_identity}")
|
|
126
|
+
lines.append(f"min_mapq: {params.min_mapq}")
|
|
127
|
+
lines += [
|
|
128
|
+
f"files: {len(files)}",
|
|
129
|
+
f"genes_found: {genes_found}",
|
|
130
|
+
"---",
|
|
131
|
+
"",
|
|
132
|
+
"# gapit read screening report",
|
|
133
|
+
"",
|
|
134
|
+
]
|
|
135
|
+
for report in files:
|
|
136
|
+
lines.append(f"## `{', '.join(report.reads)}`")
|
|
137
|
+
lines.append("")
|
|
138
|
+
if not report.genes:
|
|
139
|
+
lines.append("_No genes detected._")
|
|
140
|
+
lines.append("")
|
|
141
|
+
continue
|
|
142
|
+
lines.append("| " + " | ".join(columns) + " |")
|
|
143
|
+
lines.append("|" + "---|" * len(columns))
|
|
144
|
+
for gene in report.genes:
|
|
145
|
+
cells = (
|
|
146
|
+
gene.gene,
|
|
147
|
+
f"{gene.breadth_pct:.2f}",
|
|
148
|
+
f"{gene.mean_depth:.2f}",
|
|
149
|
+
str(gene.reads_mapped),
|
|
150
|
+
"yes" if gene.present else "no",
|
|
151
|
+
gene.database,
|
|
152
|
+
gene.accession,
|
|
153
|
+
gene.product,
|
|
154
|
+
gene.function,
|
|
155
|
+
)
|
|
156
|
+
if with_identity:
|
|
157
|
+
cells += (f"{gene.mean_identity_pct:.2f}",)
|
|
158
|
+
lines.append("| " + " | ".join(_md_cell(cell) for cell in cells) + " |")
|
|
159
|
+
lines.append("")
|
|
160
|
+
return "\n".join(lines) + "\n"
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def render_reads_markdown(
|
|
164
|
+
reports: Iterable[ReadsReport], params: ReadsParams, *, now: datetime
|
|
165
|
+
) -> str:
|
|
166
|
+
"""Render read-screening results as deterministic Markdown."""
|
|
167
|
+
return _render_reads_markdown(
|
|
168
|
+
reports,
|
|
169
|
+
params,
|
|
170
|
+
now=now,
|
|
171
|
+
schema="gapit.reads/1",
|
|
172
|
+
columns=_READS_COLUMNS,
|
|
173
|
+
with_identity=False,
|
|
174
|
+
)
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def render_reads2_markdown(
|
|
178
|
+
reports: Iterable[ReadsReport], params: ReadsParams, *, now: datetime
|
|
179
|
+
) -> str:
|
|
180
|
+
"""Render filtered read-screening results as deterministic Markdown
|
|
181
|
+
(gapit.reads/2): the reads/1 shape plus the filter thresholds in
|
|
182
|
+
frontmatter and an Identity% column per gene."""
|
|
183
|
+
return _render_reads_markdown(
|
|
184
|
+
reports,
|
|
185
|
+
params,
|
|
186
|
+
now=now,
|
|
187
|
+
schema="gapit.reads/2",
|
|
188
|
+
columns=_READS2_COLUMNS,
|
|
189
|
+
with_identity=True,
|
|
190
|
+
)
|
gapit/formats/schemas.py
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
"""The schema registry: every document name ``gapit schema`` can introspect.
|
|
2
|
+
|
|
3
|
+
Lives beside the document models but one level up: it needs the summary
|
|
4
|
+
document (formats/summary.py) and the error envelope (errors.py), and
|
|
5
|
+
formats/summary.py already imports from formats/json.py — so this module, not
|
|
6
|
+
formats/json.py, is the cycle-free single home. Insertion order is part of the
|
|
7
|
+
public surface (``gapit schema`` unknown-name message iterates it).
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from pydantic import BaseModel
|
|
11
|
+
|
|
12
|
+
from gapit.errors import ErrorEnvelope
|
|
13
|
+
from gapit.formats.json import (
|
|
14
|
+
ListDocument,
|
|
15
|
+
Reads2Document,
|
|
16
|
+
ReadsDocument,
|
|
17
|
+
ReportDocument,
|
|
18
|
+
VersionDocument,
|
|
19
|
+
)
|
|
20
|
+
from gapit.formats.summary import SummaryDocument
|
|
21
|
+
|
|
22
|
+
SCHEMA_MODELS: dict[str, type[BaseModel]] = {
|
|
23
|
+
"report": ReportDocument,
|
|
24
|
+
"reads": ReadsDocument,
|
|
25
|
+
"reads2": Reads2Document,
|
|
26
|
+
"summary": SummaryDocument,
|
|
27
|
+
"list": ListDocument,
|
|
28
|
+
"error": ErrorEnvelope,
|
|
29
|
+
"version": VersionDocument,
|
|
30
|
+
}
|
gapit/formats/summary.py
ADDED
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
"""Summary output formats: abricate-parity TSV/CSV, gapit.summary/1 JSON, Markdown."""
|
|
2
|
+
|
|
3
|
+
from datetime import datetime
|
|
4
|
+
from typing import Literal
|
|
5
|
+
|
|
6
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
7
|
+
|
|
8
|
+
from gapit import __version__
|
|
9
|
+
from gapit.formats.json import ToolDocument, utc_timestamp
|
|
10
|
+
from gapit.summary import ABSENT, FIELDSEP, SummaryMatrix
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class SummaryParamsDocument(BaseModel, frozen=True):
|
|
14
|
+
"""The summary parameters in effect (metric is explicit)."""
|
|
15
|
+
|
|
16
|
+
metric: Literal["%COVERAGE", "%IDENTITY"]
|
|
17
|
+
nopath: bool
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class SummaryRowDocument(BaseModel, frozen=True):
|
|
21
|
+
"""One matrix row: label, distinct-gene count, per-gene value lists."""
|
|
22
|
+
|
|
23
|
+
file: str
|
|
24
|
+
num_found: int
|
|
25
|
+
cells: dict[str, list[str]]
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class SummaryDocument(BaseModel, frozen=True):
|
|
29
|
+
"""gapit.summary/1 — machine-readable summary matrix."""
|
|
30
|
+
|
|
31
|
+
model_config = ConfigDict(populate_by_name=True)
|
|
32
|
+
|
|
33
|
+
schema_name: Literal["gapit.summary/1"] = Field(default="gapit.summary/1", alias="schema")
|
|
34
|
+
tool: ToolDocument = ToolDocument()
|
|
35
|
+
created_at: str
|
|
36
|
+
params: SummaryParamsDocument
|
|
37
|
+
genes: list[str]
|
|
38
|
+
rows: list[SummaryRowDocument]
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def format_summary_tsv(matrix: SummaryMatrix, *, csv: bool) -> str:
|
|
42
|
+
"""Render the matrix in abricate --summary shape: '#FILE NUM_FOUND <genes>'
|
|
43
|
+
header, one row per input, ';' cells, '.' absent. Line = sep.join + '\\n'."""
|
|
44
|
+
sep = "," if csv else "\t"
|
|
45
|
+
lines = [sep.join(("#FILE", "NUM_FOUND", *matrix.genes))]
|
|
46
|
+
for row in matrix.rows:
|
|
47
|
+
cells = [
|
|
48
|
+
FIELDSEP.join(row.cells[gene]) if gene in row.cells else ABSENT for gene in matrix.genes
|
|
49
|
+
]
|
|
50
|
+
lines.append(sep.join((row.file, str(row.num_found), *cells)))
|
|
51
|
+
return "".join(line + "\n" for line in lines)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def render_summary_json(matrix: SummaryMatrix, *, now: datetime) -> str:
|
|
55
|
+
"""Serialize the matrix as gapit.summary/1 (indented, schema first)."""
|
|
56
|
+
document = SummaryDocument(
|
|
57
|
+
created_at=utc_timestamp(now),
|
|
58
|
+
params=SummaryParamsDocument(metric=matrix.params.metric, nopath=matrix.params.nopath),
|
|
59
|
+
genes=list(matrix.genes),
|
|
60
|
+
rows=[
|
|
61
|
+
SummaryRowDocument(
|
|
62
|
+
file=row.file,
|
|
63
|
+
num_found=row.num_found,
|
|
64
|
+
cells={gene: list(values) for gene, values in row.cells.items()},
|
|
65
|
+
)
|
|
66
|
+
for row in matrix.rows
|
|
67
|
+
],
|
|
68
|
+
)
|
|
69
|
+
return document.model_dump_json(indent=2, by_alias=True)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _md_cell(value: str) -> str:
|
|
73
|
+
"""Escape Markdown table metacharacters so cells cannot break the table."""
|
|
74
|
+
return value.replace("\\", "\\\\").replace("|", "\\|")
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def render_summary_md(matrix: SummaryMatrix, *, now: datetime) -> str:
|
|
78
|
+
"""Render the matrix as deterministic Markdown: YAML frontmatter, then a
|
|
79
|
+
File x gene table with escaped cells ('.' = absent, ';' = multi-value)."""
|
|
80
|
+
columns = ("File", "Num found", *matrix.genes)
|
|
81
|
+
lines: list[str] = [
|
|
82
|
+
"---",
|
|
83
|
+
"schema: gapit.summary/1",
|
|
84
|
+
f"tool: gapit {__version__}",
|
|
85
|
+
f"created_at: {utc_timestamp(now)}",
|
|
86
|
+
f"metric: '{matrix.params.metric}'",
|
|
87
|
+
f"nopath: {str(matrix.params.nopath).lower()}",
|
|
88
|
+
f"files: {len(matrix.rows)}",
|
|
89
|
+
f"genes: {len(matrix.genes)}",
|
|
90
|
+
"---",
|
|
91
|
+
"",
|
|
92
|
+
"# gapit summary matrix",
|
|
93
|
+
"",
|
|
94
|
+
"| " + " | ".join(_md_cell(column) for column in columns) + " |",
|
|
95
|
+
"|" + "---|" * len(columns),
|
|
96
|
+
]
|
|
97
|
+
for row in matrix.rows:
|
|
98
|
+
cells = [
|
|
99
|
+
_md_cell(FIELDSEP.join(row.cells[gene])) if gene in row.cells else ABSENT
|
|
100
|
+
for gene in matrix.genes
|
|
101
|
+
]
|
|
102
|
+
lines.append("| " + " | ".join((_md_cell(row.file), str(row.num_found), *cells)) + " |")
|
|
103
|
+
return "\n".join(lines) + "\n"
|
gapit/formats/tsv.py
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""abricate-byte-compatible TSV/CSV rendering of Reports (SPEC.md §5)."""
|
|
2
|
+
|
|
3
|
+
from collections.abc import Iterable
|
|
4
|
+
from pathlib import PurePath
|
|
5
|
+
|
|
6
|
+
from gapit.report import Report
|
|
7
|
+
|
|
8
|
+
HEADER = (
|
|
9
|
+
"#FILE\tSEQUENCE\tSTART\tEND\tSTRAND\tGENE\tCOVERAGE\tCOVERAGE_MAP\tGAPS\t"
|
|
10
|
+
"%COVERAGE\t%IDENTITY\tDATABASE\tACCESSION\tPRODUCT\tRESISTANCE"
|
|
11
|
+
)
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def format_tsv(reports: Iterable[Report], *, csv: bool, noheader: bool, nopath: bool) -> str:
|
|
15
|
+
"""Render reports as abricate-format lines: one header (unless noheader),
|
|
16
|
+
then one line per hit per report in report order. Line = sep.join(fields)
|
|
17
|
+
+ "\\n"; fields never contain the separator (product cleanup strips commas).
|
|
18
|
+
"""
|
|
19
|
+
sep = "," if csv else "\t"
|
|
20
|
+
lines: list[str] = [] if noheader else [HEADER.replace("\t", sep)]
|
|
21
|
+
for report in reports:
|
|
22
|
+
file_column = PurePath(report.file).name if nopath else report.file
|
|
23
|
+
for hit in report.hits:
|
|
24
|
+
lines.append(
|
|
25
|
+
sep.join(
|
|
26
|
+
(
|
|
27
|
+
file_column,
|
|
28
|
+
hit.sequence,
|
|
29
|
+
str(hit.start),
|
|
30
|
+
str(hit.end),
|
|
31
|
+
hit.strand,
|
|
32
|
+
hit.gene,
|
|
33
|
+
f"{hit.s_start}-{hit.s_end}/{hit.s_len}",
|
|
34
|
+
hit.coverage_map,
|
|
35
|
+
f"{hit.gap_openings}/{hit.gaps}",
|
|
36
|
+
f"{hit.coverage_pct:.2f}",
|
|
37
|
+
f"{hit.identity_pct:.2f}",
|
|
38
|
+
hit.database,
|
|
39
|
+
hit.accession,
|
|
40
|
+
hit.product,
|
|
41
|
+
hit.function,
|
|
42
|
+
)
|
|
43
|
+
)
|
|
44
|
+
)
|
|
45
|
+
return "".join(line + "\n" for line in lines)
|