giftag 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- giftag/__init__.py +11 -0
- giftag/__main__.py +3 -0
- giftag/annotate.py +163 -0
- giftag/build.py +610 -0
- giftag/cli.py +363 -0
- giftag/data/dbcan_sub_db_v5-2-9_5-5-2026.tsv +503 -0
- giftag/database.py +107 -0
- giftag/fetch.py +218 -0
- giftag/genes.py +111 -0
- giftag/markers.py +89 -0
- giftag/search.py +152 -0
- giftag/sources.py +86 -0
- giftag/ui.py +222 -0
- giftag-1.0.0.dist-info/METADATA +74 -0
- giftag-1.0.0.dist-info/RECORD +18 -0
- giftag-1.0.0.dist-info/WHEEL +4 -0
- giftag-1.0.0.dist-info/entry_points.txt +2 -0
- giftag-1.0.0.dist-info/licenses/LICENSE +21 -0
giftag/__init__.py
ADDED
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
"""giftag: annotate genomes with exactly the markers gifter evaluates."""
|
|
2
|
+
|
|
3
|
+
__version__ = "1.0.0"
|
|
4
|
+
|
|
5
|
+
# The layout of a database directory written by `giftag build`. `annotate`
|
|
6
|
+
# refuses a directory whose format it does not know.
|
|
7
|
+
DB_FORMAT = 1
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class GiftagError(Exception):
|
|
11
|
+
"""An error the command line reports as a message rather than a traceback."""
|
giftag/__main__.py
ADDED
giftag/annotate.py
ADDED
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
"""`giftag annotate`: genomes or proteins in, a gifter marker table out."""
|
|
2
|
+
|
|
3
|
+
import csv
|
|
4
|
+
import json
|
|
5
|
+
import time
|
|
6
|
+
from collections import Counter
|
|
7
|
+
from datetime import datetime, timezone
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
from rich.markup import escape
|
|
11
|
+
|
|
12
|
+
from giftag import GiftagError, __version__, ui
|
|
13
|
+
from giftag.build import cazy_family
|
|
14
|
+
from giftag.database import Database
|
|
15
|
+
from giftag.genes import call_genes, genome_id, looks_nucleotide, read_fasta, write_fasta
|
|
16
|
+
|
|
17
|
+
MARKER_COLUMNS = (
|
|
18
|
+
"genome_id", "gene_id", "namespace", "accession", "source", "profile", "rule",
|
|
19
|
+
"score", "evalue", "threshold", "coverage", "target_from", "target_to",
|
|
20
|
+
)
|
|
21
|
+
GENOME_COLUMNS = ("genome_id", "input", "input_type", "gene_calling", "sequences",
|
|
22
|
+
"length_bp", "proteins", "marker_genes", "markers")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def annotate(inputs, outdir, db_dir, threads=0, mode="auto", input_type="auto",
|
|
26
|
+
gate_subfamilies=False):
|
|
27
|
+
started = time.monotonic()
|
|
28
|
+
database = Database(db_dir)
|
|
29
|
+
inputs = [Path(p) for p in inputs]
|
|
30
|
+
if not inputs:
|
|
31
|
+
raise GiftagError("no input files")
|
|
32
|
+
ids = [genome_id(p) for p in inputs]
|
|
33
|
+
duplicated = sorted(gid for gid, n in Counter(ids).items() if n > 1)
|
|
34
|
+
if duplicated:
|
|
35
|
+
raise GiftagError(f"inputs share genome IDs: {', '.join(duplicated)}")
|
|
36
|
+
outdir = Path(outdir)
|
|
37
|
+
outdir.mkdir(parents=True, exist_ok=True)
|
|
38
|
+
|
|
39
|
+
unsearchable = database.unsearchable
|
|
40
|
+
if unsearchable:
|
|
41
|
+
ui.warning(f"{len(unsearchable)} gifter markers cannot be searched with this database "
|
|
42
|
+
f"and will read as absent in gifter (see markers.tsv in the database)")
|
|
43
|
+
with ui.Task("loading profiles", done="profiles loaded in {elapsed}"):
|
|
44
|
+
database.load_all()
|
|
45
|
+
|
|
46
|
+
# Tables are written genome by genome, so a long run that stops part way
|
|
47
|
+
# keeps what it finished. giftag_run.json is written last and marks a run
|
|
48
|
+
# as complete.
|
|
49
|
+
(outdir / "giftag_run.json").unlink(missing_ok=True)
|
|
50
|
+
n_rows = 0
|
|
51
|
+
summary = []
|
|
52
|
+
width = len(str(len(inputs)))
|
|
53
|
+
with _Table(outdir / "giftag_markers.tsv", MARKER_COLUMNS) as markers, \
|
|
54
|
+
_Table(outdir / "giftag_genomes.tsv", GENOME_COLUMNS) as genomes, \
|
|
55
|
+
ui.Task("annotating", total=len(inputs), steps=False) as task:
|
|
56
|
+
for index, (path, gid) in enumerate(zip(inputs, ids), start=1):
|
|
57
|
+
began = time.monotonic()
|
|
58
|
+
name = escape(gid)
|
|
59
|
+
|
|
60
|
+
def stage(step):
|
|
61
|
+
task.update(description=f"{name} · {step}")
|
|
62
|
+
|
|
63
|
+
stage("reading")
|
|
64
|
+
records = read_fasta(path)
|
|
65
|
+
kind = input_type
|
|
66
|
+
if kind == "auto":
|
|
67
|
+
kind = "nucleotide" if looks_nucleotide(records) else "protein"
|
|
68
|
+
if kind == "nucleotide":
|
|
69
|
+
stage("calling genes")
|
|
70
|
+
proteins, used = call_genes(records, mode)
|
|
71
|
+
(outdir / "proteins").mkdir(exist_ok=True)
|
|
72
|
+
write_fasta(outdir / "proteins" / f"{gid}.faa", proteins)
|
|
73
|
+
calling = f"pyrodigal {used}"
|
|
74
|
+
else:
|
|
75
|
+
proteins, calling = records, "none (protein input)"
|
|
76
|
+
|
|
77
|
+
calls = database.search(proteins, cpus=threads, gate=gate_subfamilies, stage=stage)
|
|
78
|
+
rows = _label(gid, calls, database.labels)
|
|
79
|
+
markers.write(rows)
|
|
80
|
+
n_rows += len(rows)
|
|
81
|
+
row = dict(
|
|
82
|
+
genome_id=gid, input=str(path), input_type=kind, gene_calling=calling,
|
|
83
|
+
sequences=len(records),
|
|
84
|
+
length_bp=sum(len(s) for _, s in records) if kind == "nucleotide" else "",
|
|
85
|
+
proteins=len(proteins), marker_genes=len({r["gene_id"] for r in rows}),
|
|
86
|
+
markers=len({(r["namespace"], r["accession"]) for r in rows}),
|
|
87
|
+
)
|
|
88
|
+
genomes.write([row])
|
|
89
|
+
summary.append(row)
|
|
90
|
+
task.advance()
|
|
91
|
+
ui.info(f"[dim]{index:>{width}}/{len(inputs)}[/dim] [bold]{name}[/bold] · "
|
|
92
|
+
f"{len(proteins):,} proteins · {row['markers']:,} markers · "
|
|
93
|
+
f"{ui.format_duration(time.monotonic() - began)}")
|
|
94
|
+
|
|
95
|
+
manifest = database.manifest
|
|
96
|
+
run = {
|
|
97
|
+
"giftag_version": __version__,
|
|
98
|
+
"finished_utc": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"),
|
|
99
|
+
"seconds": round(time.monotonic() - started, 1),
|
|
100
|
+
"database": {
|
|
101
|
+
"path": str(database.path.resolve()),
|
|
102
|
+
"built_utc": manifest["built_utc"],
|
|
103
|
+
"gifter": manifest["gifter"],
|
|
104
|
+
"sources": {name: {k: v for k, v in meta.items() if k != "terms"}
|
|
105
|
+
for name, meta in manifest["sources"].items()},
|
|
106
|
+
"marker_status": manifest["marker_status"],
|
|
107
|
+
},
|
|
108
|
+
"parameters": {"mode": mode, "input_type": input_type, "threads": threads,
|
|
109
|
+
"gate_subfamilies": gate_subfamilies},
|
|
110
|
+
"genomes": len(inputs),
|
|
111
|
+
"rows": n_rows,
|
|
112
|
+
}
|
|
113
|
+
with open(outdir / "giftag_run.json", "w") as handle:
|
|
114
|
+
json.dump(run, handle, indent=2)
|
|
115
|
+
handle.write("\n")
|
|
116
|
+
return {"outdir": outdir, "rows": n_rows, "genomes": summary,
|
|
117
|
+
"seconds": run["seconds"], "unsearchable": len(unsearchable)}
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _label(gid, calls, labels):
|
|
121
|
+
"""Translate accepted calls into gifter's namespace/accession, one row per
|
|
122
|
+
gene and marker, keeping the strongest supporting hit."""
|
|
123
|
+
best = {}
|
|
124
|
+
for call in calls:
|
|
125
|
+
found = list(labels.get((call.source, call.profile), ()))
|
|
126
|
+
if call.source == "dbcan" and cazy_family(call.profile) != call.profile:
|
|
127
|
+
# A hit to an official subfamily model (GH43_18) is a hit to its
|
|
128
|
+
# family: the narrower evidence licenses the broader marker. The
|
|
129
|
+
# row keeps the subfamily model as its `profile`.
|
|
130
|
+
found += labels.get(("dbcan", cazy_family(call.profile)), ())
|
|
131
|
+
for namespace, accession in found:
|
|
132
|
+
key = (call.gene_id, namespace, accession)
|
|
133
|
+
if key not in best or call.score > best[key].score:
|
|
134
|
+
best[key] = call
|
|
135
|
+
rows = []
|
|
136
|
+
for (gene, namespace, accession), call in best.items():
|
|
137
|
+
rows.append(dict(
|
|
138
|
+
genome_id=gid, gene_id=gene, namespace=namespace, accession=accession,
|
|
139
|
+
source=call.source, profile=call.profile, rule=call.rule,
|
|
140
|
+
score=f"{call.score:.1f}", evalue=f"{call.evalue:.3g}",
|
|
141
|
+
threshold=f"{call.threshold:g}",
|
|
142
|
+
coverage="" if call.coverage is None else f"{call.coverage:.3f}",
|
|
143
|
+
target_from=call.target_from or "", target_to=call.target_to or "",
|
|
144
|
+
))
|
|
145
|
+
rows.sort(key=lambda r: (r["gene_id"], r["namespace"], r["accession"]))
|
|
146
|
+
return rows
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
class _Table:
|
|
150
|
+
def __init__(self, path, columns):
|
|
151
|
+
self._handle = open(path, "w", newline="")
|
|
152
|
+
self._writer = csv.DictWriter(self._handle, columns, delimiter="\t", lineterminator="\n")
|
|
153
|
+
self._writer.writeheader()
|
|
154
|
+
|
|
155
|
+
def write(self, rows):
|
|
156
|
+
self._writer.writerows(rows)
|
|
157
|
+
self._handle.flush()
|
|
158
|
+
|
|
159
|
+
def __enter__(self):
|
|
160
|
+
return self
|
|
161
|
+
|
|
162
|
+
def __exit__(self, *exc):
|
|
163
|
+
self._handle.close()
|