giftag 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
giftag/__init__.py ADDED
@@ -0,0 +1,11 @@
1
+ """giftag: annotate genomes with exactly the markers gifter evaluates."""
2
+
3
+ __version__ = "1.0.0"
4
+
5
+ # The layout of a database directory written by `giftag build`. `annotate`
6
+ # refuses a directory whose format it does not know.
7
+ DB_FORMAT = 1
8
+
9
+
10
+ class GiftagError(Exception):
11
+ """An error the command line reports as a message rather than a traceback."""
giftag/__main__.py ADDED
@@ -0,0 +1,3 @@
1
+ from giftag.cli import main
2
+
3
+ raise SystemExit(main())
giftag/annotate.py ADDED
@@ -0,0 +1,163 @@
1
+ """`giftag annotate`: genomes or proteins in, a gifter marker table out."""
2
+
3
+ import csv
4
+ import json
5
+ import time
6
+ from collections import Counter
7
+ from datetime import datetime, timezone
8
+ from pathlib import Path
9
+
10
+ from rich.markup import escape
11
+
12
+ from giftag import GiftagError, __version__, ui
13
+ from giftag.build import cazy_family
14
+ from giftag.database import Database
15
+ from giftag.genes import call_genes, genome_id, looks_nucleotide, read_fasta, write_fasta
16
+
17
+ MARKER_COLUMNS = (
18
+ "genome_id", "gene_id", "namespace", "accession", "source", "profile", "rule",
19
+ "score", "evalue", "threshold", "coverage", "target_from", "target_to",
20
+ )
21
+ GENOME_COLUMNS = ("genome_id", "input", "input_type", "gene_calling", "sequences",
22
+ "length_bp", "proteins", "marker_genes", "markers")
23
+
24
+
25
+ def annotate(inputs, outdir, db_dir, threads=0, mode="auto", input_type="auto",
26
+ gate_subfamilies=False):
27
+ started = time.monotonic()
28
+ database = Database(db_dir)
29
+ inputs = [Path(p) for p in inputs]
30
+ if not inputs:
31
+ raise GiftagError("no input files")
32
+ ids = [genome_id(p) for p in inputs]
33
+ duplicated = sorted(gid for gid, n in Counter(ids).items() if n > 1)
34
+ if duplicated:
35
+ raise GiftagError(f"inputs share genome IDs: {', '.join(duplicated)}")
36
+ outdir = Path(outdir)
37
+ outdir.mkdir(parents=True, exist_ok=True)
38
+
39
+ unsearchable = database.unsearchable
40
+ if unsearchable:
41
+ ui.warning(f"{len(unsearchable)} gifter markers cannot be searched with this database "
42
+ f"and will read as absent in gifter (see markers.tsv in the database)")
43
+ with ui.Task("loading profiles", done="profiles loaded in {elapsed}"):
44
+ database.load_all()
45
+
46
+ # Tables are written genome by genome, so a long run that stops part way
47
+ # keeps what it finished. giftag_run.json is written last and marks a run
48
+ # as complete.
49
+ (outdir / "giftag_run.json").unlink(missing_ok=True)
50
+ n_rows = 0
51
+ summary = []
52
+ width = len(str(len(inputs)))
53
+ with _Table(outdir / "giftag_markers.tsv", MARKER_COLUMNS) as markers, \
54
+ _Table(outdir / "giftag_genomes.tsv", GENOME_COLUMNS) as genomes, \
55
+ ui.Task("annotating", total=len(inputs), steps=False) as task:
56
+ for index, (path, gid) in enumerate(zip(inputs, ids), start=1):
57
+ began = time.monotonic()
58
+ name = escape(gid)
59
+
60
+ def stage(step):
61
+ task.update(description=f"{name} · {step}")
62
+
63
+ stage("reading")
64
+ records = read_fasta(path)
65
+ kind = input_type
66
+ if kind == "auto":
67
+ kind = "nucleotide" if looks_nucleotide(records) else "protein"
68
+ if kind == "nucleotide":
69
+ stage("calling genes")
70
+ proteins, used = call_genes(records, mode)
71
+ (outdir / "proteins").mkdir(exist_ok=True)
72
+ write_fasta(outdir / "proteins" / f"{gid}.faa", proteins)
73
+ calling = f"pyrodigal {used}"
74
+ else:
75
+ proteins, calling = records, "none (protein input)"
76
+
77
+ calls = database.search(proteins, cpus=threads, gate=gate_subfamilies, stage=stage)
78
+ rows = _label(gid, calls, database.labels)
79
+ markers.write(rows)
80
+ n_rows += len(rows)
81
+ row = dict(
82
+ genome_id=gid, input=str(path), input_type=kind, gene_calling=calling,
83
+ sequences=len(records),
84
+ length_bp=sum(len(s) for _, s in records) if kind == "nucleotide" else "",
85
+ proteins=len(proteins), marker_genes=len({r["gene_id"] for r in rows}),
86
+ markers=len({(r["namespace"], r["accession"]) for r in rows}),
87
+ )
88
+ genomes.write([row])
89
+ summary.append(row)
90
+ task.advance()
91
+ ui.info(f"[dim]{index:>{width}}/{len(inputs)}[/dim] [bold]{name}[/bold] · "
92
+ f"{len(proteins):,} proteins · {row['markers']:,} markers · "
93
+ f"{ui.format_duration(time.monotonic() - began)}")
94
+
95
+ manifest = database.manifest
96
+ run = {
97
+ "giftag_version": __version__,
98
+ "finished_utc": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"),
99
+ "seconds": round(time.monotonic() - started, 1),
100
+ "database": {
101
+ "path": str(database.path.resolve()),
102
+ "built_utc": manifest["built_utc"],
103
+ "gifter": manifest["gifter"],
104
+ "sources": {name: {k: v for k, v in meta.items() if k != "terms"}
105
+ for name, meta in manifest["sources"].items()},
106
+ "marker_status": manifest["marker_status"],
107
+ },
108
+ "parameters": {"mode": mode, "input_type": input_type, "threads": threads,
109
+ "gate_subfamilies": gate_subfamilies},
110
+ "genomes": len(inputs),
111
+ "rows": n_rows,
112
+ }
113
+ with open(outdir / "giftag_run.json", "w") as handle:
114
+ json.dump(run, handle, indent=2)
115
+ handle.write("\n")
116
+ return {"outdir": outdir, "rows": n_rows, "genomes": summary,
117
+ "seconds": run["seconds"], "unsearchable": len(unsearchable)}
118
+
119
+
120
+ def _label(gid, calls, labels):
121
+ """Translate accepted calls into gifter's namespace/accession, one row per
122
+ gene and marker, keeping the strongest supporting hit."""
123
+ best = {}
124
+ for call in calls:
125
+ found = list(labels.get((call.source, call.profile), ()))
126
+ if call.source == "dbcan" and cazy_family(call.profile) != call.profile:
127
+ # A hit to an official subfamily model (GH43_18) is a hit to its
128
+ # family: the narrower evidence licenses the broader marker. The
129
+ # row keeps the subfamily model as its `profile`.
130
+ found += labels.get(("dbcan", cazy_family(call.profile)), ())
131
+ for namespace, accession in found:
132
+ key = (call.gene_id, namespace, accession)
133
+ if key not in best or call.score > best[key].score:
134
+ best[key] = call
135
+ rows = []
136
+ for (gene, namespace, accession), call in best.items():
137
+ rows.append(dict(
138
+ genome_id=gid, gene_id=gene, namespace=namespace, accession=accession,
139
+ source=call.source, profile=call.profile, rule=call.rule,
140
+ score=f"{call.score:.1f}", evalue=f"{call.evalue:.3g}",
141
+ threshold=f"{call.threshold:g}",
142
+ coverage="" if call.coverage is None else f"{call.coverage:.3f}",
143
+ target_from=call.target_from or "", target_to=call.target_to or "",
144
+ ))
145
+ rows.sort(key=lambda r: (r["gene_id"], r["namespace"], r["accession"]))
146
+ return rows
147
+
148
+
149
+ class _Table:
150
+ def __init__(self, path, columns):
151
+ self._handle = open(path, "w", newline="")
152
+ self._writer = csv.DictWriter(self._handle, columns, delimiter="\t", lineterminator="\n")
153
+ self._writer.writeheader()
154
+
155
+ def write(self, rows):
156
+ self._writer.writerows(rows)
157
+ self._handle.flush()
158
+
159
+ def __enter__(self):
160
+ return self
161
+
162
+ def __exit__(self, *exc):
163
+ self._handle.close()