origenomi 0.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- origenomi-0.1.1/PKG-INFO +47 -0
- origenomi-0.1.1/README.md +36 -0
- origenomi-0.1.1/pyproject.toml +29 -0
- origenomi-0.1.1/setup.cfg +4 -0
- origenomi-0.1.1/src/origami/blast.py +122 -0
- origenomi-0.1.1/src/origami/circos_tools.py +272 -0
- origenomi-0.1.1/src/origami/circularize.py +33 -0
- origenomi-0.1.1/src/origami/cli.py +61 -0
- origenomi-0.1.1/src/origami/config.py +5 -0
- origenomi-0.1.1/src/origami/dot.py +166 -0
- origenomi-0.1.1/src/origami/genome_stat.py +30 -0
- origenomi-0.1.1/src/origami/hits.py +26 -0
- origenomi-0.1.1/src/origami/io.py +33 -0
- origenomi-0.1.1/src/origami/linear.py +79 -0
- origenomi-0.1.1/src/origami/logging_setup.py +9 -0
- origenomi-0.1.1/src/origami/main.py +164 -0
- origenomi-0.1.1/src/origami/models.py +29 -0
- origenomi-0.1.1/src/origami/pairing.py +228 -0
- origenomi-0.1.1/src/origami/paths.py +35 -0
- origenomi-0.1.1/src/origami/report.py +177 -0
- origenomi-0.1.1/src/origami/rotate.py +4 -0
- origenomi-0.1.1/src/origami/setup.py +9 -0
- origenomi-0.1.1/src/origami/trim.py +77 -0
- origenomi-0.1.1/src/origami/utils.py +92 -0
- origenomi-0.1.1/src/origami/workflows.py +341 -0
- origenomi-0.1.1/src/origenomi.egg-info/PKG-INFO +47 -0
- origenomi-0.1.1/src/origenomi.egg-info/SOURCES.txt +29 -0
- origenomi-0.1.1/src/origenomi.egg-info/dependency_links.txt +1 -0
- origenomi-0.1.1/src/origenomi.egg-info/entry_points.txt +2 -0
- origenomi-0.1.1/src/origenomi.egg-info/requires.txt +3 -0
- origenomi-0.1.1/src/origenomi.egg-info/top_level.txt +1 -0
origenomi-0.1.1/PKG-INFO
ADDED
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: origenomi
|
|
3
|
+
Version: 0.1.1
|
|
4
|
+
Summary: Trim circular overlaps, find dnaA/oriC, rotate sequences, and report.
|
|
5
|
+
Author: You
|
|
6
|
+
Requires-Python: >=3.9
|
|
7
|
+
Description-Content-Type: text/markdown
|
|
8
|
+
Requires-Dist: biopython>=1.83
|
|
9
|
+
Requires-Dist: numpy>=1.26
|
|
10
|
+
Requires-Dist: matplotlib>=3.7
|
|
11
|
+
|
|
12
|
+
# Origami
|
|
13
|
+
|
|
14
|
+
Command-line tool that:
|
|
15
|
+
- Trims duplicated terminal overlaps in circular assemblies (half2 → DB(half1) via BLAST).
|
|
16
|
+
- Runs **one BLAST per DB** (dnaA and oriC) on the concatenated FASTA.
|
|
17
|
+
- Keeps **top 3 dnaA** and **top 7 oriC** per record (dedup by `sseqid`, ordered by evalue→identity→coverage→bitscore).
|
|
18
|
+
- Pairs dnaA–oriC via midpoint proximity (1% then 5% of record length, else closest), computes AT/GC on the oriC segment, and chooses the pair with **highest AT/GC ratio** (tie → higher AT%).
|
|
19
|
+
- Rotates so the **earlier** of dnaA/oriC starts the sequence (or the single site if only one found).
|
|
20
|
+
- Writes exactly two outputs: `origami_<prefix>.fna` and `origami_<prefix>_report.txt`.
|
|
21
|
+
|
|
22
|
+
## Install
|
|
23
|
+
```bash
|
|
24
|
+
pip install -e .
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
## Usage
|
|
28
|
+
|
|
29
|
+
### Full pipeline
|
|
30
|
+
```bash
|
|
31
|
+
origami run -i genome.fna --dnaA-db /path/new_dnaA_DB --oric-db /path/DoriC_DB
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
### Trim only
|
|
35
|
+
```bash
|
|
36
|
+
origami trim -i genome.fna
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
### OriC only (no trim)
|
|
40
|
+
```bash
|
|
41
|
+
origami oric -i genome.fna --dnaA-db /path/new_dnaA_DB --oric-db /path/DoriC_DB
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
## Outputs
|
|
45
|
+
1. origami_<prefix>.fna
|
|
46
|
+
2. origami_<prefix>_report.txt
|
|
47
|
+
3. Temps in ./temp/<prefix>/, cleaned unless --keep-temp.
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
# Origami
|
|
2
|
+
|
|
3
|
+
Command-line tool that:
|
|
4
|
+
- Trims duplicated terminal overlaps in circular assemblies (half2 → DB(half1) via BLAST).
|
|
5
|
+
- Runs **one BLAST per DB** (dnaA and oriC) on the concatenated FASTA.
|
|
6
|
+
- Keeps **top 3 dnaA** and **top 7 oriC** per record (dedup by `sseqid`, ordered by evalue→identity→coverage→bitscore).
|
|
7
|
+
- Pairs dnaA–oriC via midpoint proximity (1% then 5% of record length, else closest), computes AT/GC on the oriC segment, and chooses the pair with **highest AT/GC ratio** (tie → higher AT%).
|
|
8
|
+
- Rotates so the **earlier** of dnaA/oriC starts the sequence (or the single site if only one found).
|
|
9
|
+
- Writes exactly two outputs: `origami_<prefix>.fna` and `origami_<prefix>_report.txt`.
|
|
10
|
+
|
|
11
|
+
## Install
|
|
12
|
+
```bash
|
|
13
|
+
pip install -e .
|
|
14
|
+
```
|
|
15
|
+
|
|
16
|
+
## Usage
|
|
17
|
+
|
|
18
|
+
### Full pipeline
|
|
19
|
+
```bash
|
|
20
|
+
origami run -i genome.fna --dnaA-db /path/new_dnaA_DB --oric-db /path/DoriC_DB
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
### Trim only
|
|
24
|
+
```bash
|
|
25
|
+
origami trim -i genome.fna
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
### OriC only (no trim)
|
|
29
|
+
```bash
|
|
30
|
+
origami oric -i genome.fna --dnaA-db /path/new_dnaA_DB --oric-db /path/DoriC_DB
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
## Outputs
|
|
34
|
+
1. origami_<prefix>.fna
|
|
35
|
+
2. origami_<prefix>_report.txt
|
|
36
|
+
3. Temps in ./temp/<prefix>/, cleaned unless --keep-temp.
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "origenomi"
|
|
7
|
+
version = "0.1.1"
|
|
8
|
+
description = "Trim circular overlaps, find dnaA/oriC, rotate sequences, and report."
|
|
9
|
+
authors = [{ name = "You" }]
|
|
10
|
+
readme = "README.md"
|
|
11
|
+
requires-python = ">=3.9"
|
|
12
|
+
dependencies = [
|
|
13
|
+
"biopython>=1.83",
|
|
14
|
+
"numpy>=1.26",
|
|
15
|
+
"matplotlib>=3.7"
|
|
16
|
+
]
|
|
17
|
+
|
|
18
|
+
[project.scripts]
|
|
19
|
+
origenomi = "origami.cli:main"
|
|
20
|
+
|
|
21
|
+
[tool.setuptools]
|
|
22
|
+
package-dir = {"" = "src"}
|
|
23
|
+
packages = ["origami"]
|
|
24
|
+
|
|
25
|
+
[tool.setuptools.package-data]
|
|
26
|
+
origami = [
|
|
27
|
+
"database/**/"
|
|
28
|
+
]
|
|
29
|
+
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
import subprocess
|
|
2
|
+
from subprocess import DEVNULL
|
|
3
|
+
|
|
4
|
+
# === Annotation (dnaA / oriC) ===
|
|
5
|
+
# Keep your exact outfmt for annotation BLASTs
|
|
6
|
+
OUTFMT_ANN = (
|
|
7
|
+
"7 qseqid sseqid qstart qend sstart send qlen slen "
|
|
8
|
+
"pident evalue bitscore mismatch gaps length"
|
|
9
|
+
)
|
|
10
|
+
|
|
11
|
+
def run_blast_annotation(
|
|
12
|
+
query_fasta: str,
|
|
13
|
+
db_prefix: str,
|
|
14
|
+
out_path: str,
|
|
15
|
+
evalue: str = "1e-5",
|
|
16
|
+
quiet: bool = True,
|
|
17
|
+
) -> None:
|
|
18
|
+
"""
|
|
19
|
+
One BLAST per DB (dnaA, oriC) over the concatenated FASTA.
|
|
20
|
+
"""
|
|
21
|
+
args = [
|
|
22
|
+
"blastn",
|
|
23
|
+
"-query", query_fasta,
|
|
24
|
+
"-db", db_prefix,
|
|
25
|
+
"-out", out_path,
|
|
26
|
+
"-outfmt", OUTFMT_ANN,
|
|
27
|
+
"-evalue", str(evalue),
|
|
28
|
+
]
|
|
29
|
+
subprocess.run(
|
|
30
|
+
args,
|
|
31
|
+
check=True,
|
|
32
|
+
stdout=DEVNULL if quiet else None,
|
|
33
|
+
stderr=DEVNULL if quiet else None,
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
def parse_blast_annotation(path: str):
|
|
37
|
+
"""
|
|
38
|
+
Parse annotation outfmt into dict[qseqid] -> list[hit dicts].
|
|
39
|
+
"""
|
|
40
|
+
hits = {}
|
|
41
|
+
with open(path) as f:
|
|
42
|
+
for line in f:
|
|
43
|
+
if not line.strip() or line.startswith("#"):
|
|
44
|
+
continue
|
|
45
|
+
(qid, sid, qstart, qend, sstart, send,
|
|
46
|
+
qlen, slen, pident, evalue, bitscore,
|
|
47
|
+
mismatch, gaps, length) = line.strip().split()
|
|
48
|
+
|
|
49
|
+
qstart, qend = int(qstart), int(qend)
|
|
50
|
+
sstart, send = int(sstart), int(send)
|
|
51
|
+
|
|
52
|
+
entry = {
|
|
53
|
+
"sseqid": sid,
|
|
54
|
+
"qstart": min(qstart, qend),
|
|
55
|
+
"qend": max(qstart, qend),
|
|
56
|
+
"sstart": min(sstart, send),
|
|
57
|
+
"send": max(sstart, send),
|
|
58
|
+
"qlen": int(qlen),
|
|
59
|
+
"slen": int(slen),
|
|
60
|
+
"length": int(length),
|
|
61
|
+
"pident": float(pident),
|
|
62
|
+
"evalue": float(evalue),
|
|
63
|
+
"bitscore": float(bitscore),
|
|
64
|
+
"mismatch": int(mismatch),
|
|
65
|
+
"gaps": int(gaps),
|
|
66
|
+
}
|
|
67
|
+
hits.setdefault(qid, []).append(entry)
|
|
68
|
+
return hits
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
# === Trim (half2 → DB(half1)) ===
|
|
72
|
+
# Minimal columns, strict flags; we sort by length and a couple tie-breakers
|
|
73
|
+
OUTFMT_TRIM = "6 qseqid sseqid qstart qend sstart send qlen slen pident length evalue"
|
|
74
|
+
|
|
75
|
+
def run_blast_trim(
|
|
76
|
+
query_fasta: str,
|
|
77
|
+
db_prefix: str,
|
|
78
|
+
out_path: str,
|
|
79
|
+
evalue: str = "1e-10",
|
|
80
|
+
quiet: bool = True,
|
|
81
|
+
) -> None:
|
|
82
|
+
"""
|
|
83
|
+
BLAST for terminal overlap detection (half2 vs DB(half1)).
|
|
84
|
+
"""
|
|
85
|
+
args = [
|
|
86
|
+
"blastn",
|
|
87
|
+
"-query", query_fasta,
|
|
88
|
+
"-db", db_prefix,
|
|
89
|
+
"-out", out_path,
|
|
90
|
+
"-outfmt", OUTFMT_TRIM,
|
|
91
|
+
"-evalue", str(evalue),
|
|
92
|
+
"-dust", "no",
|
|
93
|
+
"-soft_masking", "false",
|
|
94
|
+
"-max_hsps", "1",
|
|
95
|
+
]
|
|
96
|
+
subprocess.run(
|
|
97
|
+
args,
|
|
98
|
+
check=True,
|
|
99
|
+
stdout=DEVNULL if quiet else None,
|
|
100
|
+
stderr=DEVNULL if quiet else None,
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
def parse_blast_trim(path: str):
|
|
104
|
+
"""
|
|
105
|
+
Parse trim outfmt into a flat list of hits (dicts).
|
|
106
|
+
"""
|
|
107
|
+
rows = []
|
|
108
|
+
with open(path) as f:
|
|
109
|
+
for line in f:
|
|
110
|
+
if not line.strip() or line.startswith("#"):
|
|
111
|
+
continue
|
|
112
|
+
(qid, sid, qstart, qend, sstart, send,
|
|
113
|
+
qlen, slen, pident, length, evalue) = line.strip().split()
|
|
114
|
+
rows.append({
|
|
115
|
+
"qseqid": qid, "sseqid": sid,
|
|
116
|
+
"qstart": int(qstart), "qend": int(qend),
|
|
117
|
+
"sstart": int(sstart), "send": int(send),
|
|
118
|
+
"qlen": int(qlen), "slen": int(slen),
|
|
119
|
+
"pident": float(pident), "length": int(length),
|
|
120
|
+
"evalue": float(evalue),
|
|
121
|
+
})
|
|
122
|
+
return rows
|
|
@@ -0,0 +1,272 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import re
|
|
3
|
+
import numpy as np
|
|
4
|
+
import matplotlib.pyplot as plt
|
|
5
|
+
from Bio import SeqIO
|
|
6
|
+
|
|
7
|
+
# SHARED FUNCTION
|
|
8
|
+
def fast_gc_skew(seq, win=1000):
|
|
9
|
+
"""Compute GC skew using cumulative NumPy operations."""
|
|
10
|
+
seq = np.frombuffer(seq.upper().encode(), dtype='S1')
|
|
11
|
+
g = (seq == b'G').astype(int)
|
|
12
|
+
c = (seq == b'C').astype(int)
|
|
13
|
+
|
|
14
|
+
cum_g = np.cumsum(g)
|
|
15
|
+
cum_c = np.cumsum(c)
|
|
16
|
+
|
|
17
|
+
skew = np.empty(len(seq) // win)
|
|
18
|
+
|
|
19
|
+
for i in range(len(skew)):
|
|
20
|
+
start = i * win
|
|
21
|
+
end = start + win
|
|
22
|
+
|
|
23
|
+
g_win = cum_g[end - 1] - (cum_g[start - 1] if start > 0 else 0)
|
|
24
|
+
c_win = cum_c[end - 1] - (cum_c[start - 1] if start > 0 else 0)
|
|
25
|
+
|
|
26
|
+
denom = g_win + c_win
|
|
27
|
+
skew[i] = (g_win - c_win) / denom if denom > 0 else 0
|
|
28
|
+
|
|
29
|
+
return skew
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def read_positions(report):
|
|
33
|
+
"""Extract dnaA and oriC coordinates from Origami report."""
|
|
34
|
+
dnaA = None
|
|
35
|
+
oriC_list = []
|
|
36
|
+
|
|
37
|
+
with open(report) as f:
|
|
38
|
+
for line in f:
|
|
39
|
+
# dnaA coordinates
|
|
40
|
+
if "Chosen dnaA:" in line:
|
|
41
|
+
next(f)
|
|
42
|
+
coords_line = next(f).strip()
|
|
43
|
+
m = re.search(r"Coords:\s*(\d+)\s*-\s*(\d+)", coords_line)
|
|
44
|
+
if m:
|
|
45
|
+
dnaA = int(m.group(1))
|
|
46
|
+
|
|
47
|
+
# oriC coordinates (can be multiple)
|
|
48
|
+
if "Chosen oriC:" in line:
|
|
49
|
+
next(f)
|
|
50
|
+
coords_line = next(f).strip()
|
|
51
|
+
m = re.findall(r"(\d+)\s*-\s*(\d+)", coords_line)
|
|
52
|
+
for a, b in m:
|
|
53
|
+
oriC_list.append((int(a), int(b)))
|
|
54
|
+
|
|
55
|
+
return dnaA, oriC_list
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
# ============================================================
|
|
59
|
+
# 1) NON-ROTATED CIRCOS PLOT → plot_circos()
|
|
60
|
+
# ============================================================
|
|
61
|
+
|
|
62
|
+
def plot_circos(fna_path, report_path, out_dir):
|
|
63
|
+
"""
|
|
64
|
+
This is the NON-rotated circos plot (Version B).
|
|
65
|
+
Used for original genome orientation.
|
|
66
|
+
"""
|
|
67
|
+
record = next(SeqIO.parse(fna_path, "fasta"))
|
|
68
|
+
seq = str(record.seq)
|
|
69
|
+
genome_len = len(seq)
|
|
70
|
+
print(f"Genome Length in non rotated circos:{genome_len}")
|
|
71
|
+
dnaA, oriCs = read_positions(report_path)
|
|
72
|
+
oriC = oriCs[0][0] if oriCs else None
|
|
73
|
+
|
|
74
|
+
# Compute GC skew
|
|
75
|
+
skew = fast_gc_skew(seq, win=1000)
|
|
76
|
+
angles = np.linspace(0, 2*np.pi, len(skew), endpoint=False)
|
|
77
|
+
|
|
78
|
+
# Polar plot setup
|
|
79
|
+
fig, ax = plt.subplots(figsize=(10, 10), subplot_kw={'projection': 'polar'})
|
|
80
|
+
ax.set_theta_direction(-1)
|
|
81
|
+
ax.set_theta_offset(np.pi / 2)
|
|
82
|
+
ax.set_axis_off()
|
|
83
|
+
|
|
84
|
+
# Outer genome ring
|
|
85
|
+
outer_r = 1.0
|
|
86
|
+
ax.plot(np.linspace(0, 2*np.pi, 2000),
|
|
87
|
+
[outer_r] * 2000,
|
|
88
|
+
color='navy', lw=4)
|
|
89
|
+
|
|
90
|
+
# GC skew bars
|
|
91
|
+
bar_r = 0.8
|
|
92
|
+
bar_scale = 0.25
|
|
93
|
+
|
|
94
|
+
ax.bar(
|
|
95
|
+
angles[skew > 0],
|
|
96
|
+
skew[skew > 0] * bar_scale,
|
|
97
|
+
width=2*np.pi/len(skew),
|
|
98
|
+
bottom=bar_r,
|
|
99
|
+
color='lightblue',
|
|
100
|
+
linewidth=0
|
|
101
|
+
)
|
|
102
|
+
ax.bar(
|
|
103
|
+
angles[skew < 0],
|
|
104
|
+
-skew[skew < 0] * bar_scale,
|
|
105
|
+
width=2*np.pi/len(skew),
|
|
106
|
+
bottom=bar_r - bar_scale * np.abs(skew[skew < 0]),
|
|
107
|
+
color='orange',
|
|
108
|
+
linewidth=0
|
|
109
|
+
)
|
|
110
|
+
|
|
111
|
+
# Ticks (always 28)
|
|
112
|
+
num_ticks = 28
|
|
113
|
+
tick_spacing = genome_len / num_ticks
|
|
114
|
+
label_offset = 0.10
|
|
115
|
+
|
|
116
|
+
for i in range(num_ticks):
|
|
117
|
+
pos = tick_spacing * i
|
|
118
|
+
theta = 2 * np.pi * (pos / genome_len)
|
|
119
|
+
|
|
120
|
+
ax.plot([theta, theta], [outer_r, outer_r + 0.03],
|
|
121
|
+
color='navy', lw=1.5)
|
|
122
|
+
ax.text(theta, outer_r + label_offset,
|
|
123
|
+
f"{pos/1e6:.2f} Mb",
|
|
124
|
+
ha='center', va='center', fontsize=9)
|
|
125
|
+
|
|
126
|
+
# dnaA / oriC markers
|
|
127
|
+
if dnaA:
|
|
128
|
+
theta_dnaA = 2 * np.pi * (dnaA / genome_len)
|
|
129
|
+
ax.plot([theta_dnaA], [outer_r], marker='s',
|
|
130
|
+
color='blue', markersize=10)
|
|
131
|
+
|
|
132
|
+
if oriC:
|
|
133
|
+
theta_oriC = 2 * np.pi * (oriC / genome_len)
|
|
134
|
+
ax.plot([theta_oriC], [outer_r], marker='o',
|
|
135
|
+
color='red', markersize=10)
|
|
136
|
+
|
|
137
|
+
# Legend
|
|
138
|
+
fig.text(0.05, 0.93, u'\u25A0 dnaA', color='blue', fontsize=13, fontweight='bold')
|
|
139
|
+
fig.text(0.05, 0.89, u'\u25CF oriC', color='red', fontsize=13, fontweight='bold')
|
|
140
|
+
|
|
141
|
+
# Title text
|
|
142
|
+
desc = record.description.split(",")[0]
|
|
143
|
+
parts = desc.split(" ", 1)
|
|
144
|
+
line1 = parts[0]
|
|
145
|
+
line2 = parts[1] if len(parts) > 1 else ""
|
|
146
|
+
|
|
147
|
+
ax.text(0, 0,
|
|
148
|
+
f"{line1}\n{line2}\n({genome_len:,} bp)",
|
|
149
|
+
ha='center', va='center',
|
|
150
|
+
fontsize=10, fontweight='bold')
|
|
151
|
+
|
|
152
|
+
# Save
|
|
153
|
+
base = os.path.splitext(os.path.basename(fna_path))[0]
|
|
154
|
+
png = os.path.join(out_dir, f"{base}_final_circos_v9.png")
|
|
155
|
+
svg = os.path.join(out_dir, f"{base}_final_circos_v9.svg")
|
|
156
|
+
|
|
157
|
+
plt.savefig(png, dpi=400, bbox_inches='tight')
|
|
158
|
+
plt.savefig(svg, bbox_inches='tight')
|
|
159
|
+
plt.close()
|
|
160
|
+
|
|
161
|
+
return png # main.py expects a PNG path
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
# ============================================================
|
|
165
|
+
# 2) ROTATED CIRCOS PLOT → plot_rotated_circos()
|
|
166
|
+
# ============================================================
|
|
167
|
+
|
|
168
|
+
def plot_rotated_circos(fna_path, report_path, out_dir):
|
|
169
|
+
"""
|
|
170
|
+
ROTATED circos plot (Version A).
|
|
171
|
+
Adjusts genome so the origin (dnaA or oriC) starts at 0°.
|
|
172
|
+
"""
|
|
173
|
+
record = next(SeqIO.parse(fna_path, "fasta"))
|
|
174
|
+
seq = str(record.seq)
|
|
175
|
+
genome_len = len(seq)
|
|
176
|
+
print(f"Genome Length in rotated circos:{genome_len}")
|
|
177
|
+
dnaA, oriC_list = read_positions(report_path)
|
|
178
|
+
oriC = oriC_list[0][0] if oriC_list else None
|
|
179
|
+
|
|
180
|
+
# Determine shift point: whichever (dnaA or oriC) is further.
|
|
181
|
+
a, b = sorted([dnaA, oriC])
|
|
182
|
+
d = b - a
|
|
183
|
+
|
|
184
|
+
if d > genome_len / 2:
|
|
185
|
+
shift_bp = b
|
|
186
|
+
rotated_from = "oriC" if b == oriC else "dnaA"
|
|
187
|
+
else:
|
|
188
|
+
shift_bp = a
|
|
189
|
+
rotated_from = "oriC" if a == oriC else "dnaA"
|
|
190
|
+
|
|
191
|
+
# Compute GC skew and rotate
|
|
192
|
+
skew = fast_gc_skew(seq, win=1000)
|
|
193
|
+
bins = len(skew)
|
|
194
|
+
bin_size = genome_len / bins
|
|
195
|
+
# shift_bins = int(shift_bp / bin_size)
|
|
196
|
+
# skew = np.roll(skew, -shift_bins)
|
|
197
|
+
|
|
198
|
+
# Polar plot
|
|
199
|
+
fig, ax = plt.subplots(figsize=(10, 10), subplot_kw={'projection': 'polar'})
|
|
200
|
+
ax.set_theta_direction(-1)
|
|
201
|
+
ax.set_theta_offset(np.pi / 2)
|
|
202
|
+
ax.set_axis_off()
|
|
203
|
+
|
|
204
|
+
# Outer ring
|
|
205
|
+
outer_r = 1.0
|
|
206
|
+
ax.plot(np.linspace(0, 2*np.pi, 2000),
|
|
207
|
+
[outer_r] * 2000,
|
|
208
|
+
color='navy', lw=4)
|
|
209
|
+
|
|
210
|
+
# GC skew bars
|
|
211
|
+
r_inner = 0.8
|
|
212
|
+
bar_scale = 0.25
|
|
213
|
+
angles = np.linspace(0, 2*np.pi, bins, endpoint=False)
|
|
214
|
+
|
|
215
|
+
ax.bar(angles[skew > 0], skew[skew > 0] * bar_scale,
|
|
216
|
+
width=2*np.pi/bins, bottom=r_inner,
|
|
217
|
+
color='lightblue', linewidth=0)
|
|
218
|
+
|
|
219
|
+
ax.bar(angles[skew < 0], -skew[skew < 0] * bar_scale,
|
|
220
|
+
width=2*np.pi/bins,
|
|
221
|
+
bottom=r_inner - bar_scale*np.abs(skew[skew < 0]),
|
|
222
|
+
color='orange', linewidth=0)
|
|
223
|
+
|
|
224
|
+
# Ticks (28)
|
|
225
|
+
num_ticks = 28
|
|
226
|
+
tick_spacing = genome_len / num_ticks
|
|
227
|
+
label_offset = 0.08
|
|
228
|
+
|
|
229
|
+
for i in range(num_ticks):
|
|
230
|
+
pos = tick_spacing * i
|
|
231
|
+
theta = 2 * np.pi * i / num_ticks
|
|
232
|
+
ax.plot([theta, theta], [outer_r, outer_r + 0.03],
|
|
233
|
+
color='navy', lw=1.5)
|
|
234
|
+
ax.text(theta, outer_r + label_offset,
|
|
235
|
+
f"{pos/1e6:.2f} Mb",
|
|
236
|
+
ha='center', va='center', fontsize=9)
|
|
237
|
+
|
|
238
|
+
# Rotated marker positions
|
|
239
|
+
dnaA_rot = (dnaA - shift_bp) % genome_len
|
|
240
|
+
oriC_rot = (oriC - shift_bp) % genome_len
|
|
241
|
+
|
|
242
|
+
ax.plot([2*np.pi*(dnaA_rot/genome_len)], [outer_r],
|
|
243
|
+
marker='s', color='blue', markersize=10)
|
|
244
|
+
ax.plot([2*np.pi*(oriC_rot/genome_len)], [outer_r],
|
|
245
|
+
marker='o', color='red', markersize=10)
|
|
246
|
+
|
|
247
|
+
# Center label
|
|
248
|
+
desc = record.description.split(",")[0]
|
|
249
|
+
parts = desc.split(" ", 1)
|
|
250
|
+
line1 = parts[0]
|
|
251
|
+
line2 = parts[1] if len(parts) > 1 else ""
|
|
252
|
+
|
|
253
|
+
ax.text(0, 0,
|
|
254
|
+
f"{line1}\n{line2}\n({genome_len:,} bp)",
|
|
255
|
+
ha='center', va='center',
|
|
256
|
+
fontsize=10, fontweight='bold')
|
|
257
|
+
|
|
258
|
+
# Legend
|
|
259
|
+
fig.text(0.05, 0.93, u'\u25A0 dnaA', color='blue', fontsize=13, fontweight='bold')
|
|
260
|
+
fig.text(0.05, 0.89, u'\u25CF oriC', color='red', fontsize=13, fontweight='bold')
|
|
261
|
+
|
|
262
|
+
# Save
|
|
263
|
+
base = os.path.splitext(os.path.basename(fna_path))[0]
|
|
264
|
+
png = os.path.join(out_dir, f"{base}_rotated_circular_gcskew.png")
|
|
265
|
+
svg = os.path.join(out_dir, f"{base}_rotated_circular_gcskew.svg")
|
|
266
|
+
|
|
267
|
+
plt.savefig(png, dpi=300, bbox_inches='tight')
|
|
268
|
+
plt.savefig(svg, bbox_inches='tight')
|
|
269
|
+
plt.close()
|
|
270
|
+
|
|
271
|
+
# Return the PNG, the chosen marker, and the coordinate
|
|
272
|
+
return png, rotated_from, shift_bp
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
|
|
2
|
+
def prepare_chromosome_sequence(header: str, sequence: str, duplicate_bp: int = 4000) -> str | None:
|
|
3
|
+
"""
|
|
4
|
+
Prepares a bacterial chromosome sequence for BLAST:
|
|
5
|
+
- Skips plasmids
|
|
6
|
+
- Adds the first `duplicate_bp` bases at the end to handle circular overlap
|
|
7
|
+
"""
|
|
8
|
+
# checking for plasmid in the header
|
|
9
|
+
if "plasmid" in header.lower():
|
|
10
|
+
return sequence # skip plasmids
|
|
11
|
+
|
|
12
|
+
# Make sure we don't exceed sequence length
|
|
13
|
+
bp_to_duplicate = min(len(sequence), duplicate_bp)
|
|
14
|
+
extended_seq = sequence + sequence[:bp_to_duplicate]
|
|
15
|
+
return extended_seq
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def trim_extended_if_needed(seq: str, hit: dict, extension_len: int = 4000):
|
|
19
|
+
"""
|
|
20
|
+
Adjust sequence and coordinates:
|
|
21
|
+
- If hit is in the extension (last extension_len bp),
|
|
22
|
+
drop first extension_len bp.
|
|
23
|
+
- Otherwise, drop artificial tail.
|
|
24
|
+
"""
|
|
25
|
+
seq_len = len(seq)
|
|
26
|
+
orig_len = seq_len - extension_len
|
|
27
|
+
|
|
28
|
+
if hit["qend"] > orig_len:
|
|
29
|
+
# Hit was in extension → drop front
|
|
30
|
+
return seq[extension_len:], "front"
|
|
31
|
+
else:
|
|
32
|
+
# No hit in extension → drop tail
|
|
33
|
+
return seq[:-extension_len], "tail"
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
import argparse
|
|
2
|
+
import os
|
|
3
|
+
from origami.workflows import run_full, run_trim_only, run_oric_only
|
|
4
|
+
from origami.logging_setup import setup_logging
|
|
5
|
+
from importlib.resources import files
|
|
6
|
+
|
|
7
|
+
# Default DB paths (relative to project root)
|
|
8
|
+
DNAA_PREFIX = "dnaA_clustered100_DB"
|
|
9
|
+
ORIC_PREFIX = "OriC_clustered100_DB"
|
|
10
|
+
|
|
11
|
+
def get_db_prefixes():
|
|
12
|
+
db_root = files("origami.data.clustered_DB")
|
|
13
|
+
dnaA_prefix = db_root / "dnaA_clustered100_DB"
|
|
14
|
+
oric_prefix = db_root / "OriC_clustered100_DB"
|
|
15
|
+
return str(dnaA_prefix), str(oric_prefix)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def main():
|
|
19
|
+
parser = argparse.ArgumentParser(
|
|
20
|
+
prog="origenomi",
|
|
21
|
+
description="Trim circular overlaps, find dnaA/oriC, rotate, and report."
|
|
22
|
+
)
|
|
23
|
+
subparsers = parser.add_subparsers(dest="command")
|
|
24
|
+
|
|
25
|
+
def add_common(p):
|
|
26
|
+
p.add_argument("-i", "--input", required=True, help="Input FASTA (single or multi-FASTA)")
|
|
27
|
+
p.add_argument("-o","--out-dir", default=".", help="Output directory (default: .)")
|
|
28
|
+
p.add_argument("--keep-temp", action="store_true", help="Keep temp files")
|
|
29
|
+
p.add_argument("--verbose", action="store_true", help="Verbose logging")
|
|
30
|
+
p.add_argument("--keep-plot", action="store_true", help="Keep the plots")
|
|
31
|
+
|
|
32
|
+
# Full pipeline
|
|
33
|
+
p_run = subparsers.add_parser("run", help="Trim + dnaA/oriC + rotate + report")
|
|
34
|
+
add_common(p_run)
|
|
35
|
+
|
|
36
|
+
# Trim-only
|
|
37
|
+
p_trim = subparsers.add_parser("trim", help="Trim duplicated terminal overlaps only")
|
|
38
|
+
add_common(p_trim)
|
|
39
|
+
|
|
40
|
+
# OriC-only
|
|
41
|
+
p_oric = subparsers.add_parser("oric", help="dnaA/oriC + rotate on original input")
|
|
42
|
+
add_common(p_oric)
|
|
43
|
+
|
|
44
|
+
args = parser.parse_args()
|
|
45
|
+
|
|
46
|
+
# Inject default DB paths automatically
|
|
47
|
+
args.dnaA_db, args.oric_db = get_db_prefixes()
|
|
48
|
+
|
|
49
|
+
args.out_dir = os.path.abspath(args.out_dir)
|
|
50
|
+
os.makedirs(args.out_dir, exist_ok=True)
|
|
51
|
+
|
|
52
|
+
logger = setup_logging(args.verbose)
|
|
53
|
+
|
|
54
|
+
if args.command == "run":
|
|
55
|
+
run_full(args, logger)
|
|
56
|
+
elif args.command == "trim":
|
|
57
|
+
run_trim_only(args, logger)
|
|
58
|
+
elif args.command == "oric":
|
|
59
|
+
run_oric_only(args, logger)
|
|
60
|
+
else:
|
|
61
|
+
parser.print_help()
|