OctopuSV 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- octopusv/__init__.py +4 -0
- octopusv/__main__.py +3 -0
- octopusv/bencher/__init__.py +0 -0
- octopusv/bencher/bench_utils.py +39 -0
- octopusv/bencher/sv_bencher.py +274 -0
- octopusv/cli/__init__.py +0 -0
- octopusv/cli/bench.py +91 -0
- octopusv/cli/cli.py +53 -0
- octopusv/cli/convert.py +336 -0
- octopusv/cli/merge.py +269 -0
- octopusv/cli/plot.py +27 -0
- octopusv/cli/stat.py +54 -0
- octopusv/cli/svcf2bed.py +49 -0
- octopusv/cli/svcf2bedpe.py +49 -0
- octopusv/cli/svcf2vcf.py +36 -0
- octopusv/converter/__init__.py +0 -0
- octopusv/converter/base.py +122 -0
- octopusv/converter/bnd2del.py +130 -0
- octopusv/converter/bnd2dup_pair.py +130 -0
- octopusv/converter/bnd2inv_pair.py +118 -0
- octopusv/converter/bnd_keeping.py +38 -0
- octopusv/converter/mpi2tra.py +33 -0
- octopusv/converter/mpm2tra.py +31 -0
- octopusv/converter/mprtra2tra.py +53 -0
- octopusv/converter/nobnd.py +69 -0
- octopusv/converter/snmd_dndpi2tra.py +32 -0
- octopusv/converter/snmd_dndpr_tra2tra.py +37 -0
- octopusv/converter/stra2tra.py +13 -0
- octopusv/filter/__init__.py +10 -0
- octopusv/filter/quality_filter.py +337 -0
- octopusv/formatter/svcf_to_bed_converter.py +61 -0
- octopusv/formatter/svcf_to_bedpe_converter.py +68 -0
- octopusv/formatter/svcf_to_vcf_converter.py +98 -0
- octopusv/merger/TRA_merge_logic.py +95 -0
- octopusv/merger/__init__.py +0 -0
- octopusv/merger/bnd_merge_logic.py +112 -0
- octopusv/merger/bnd_merger.py +113 -0
- octopusv/merger/multi_sample_writer.py +177 -0
- octopusv/merger/name_mapper.py +50 -0
- octopusv/merger/sv_merge_logic.py +81 -0
- octopusv/merger/sv_merger.py +399 -0
- octopusv/merger/sv_selector.py +125 -0
- octopusv/merger/tra_merger.py +139 -0
- octopusv/merger/upset_plotter.py +175 -0
- octopusv/ploter/__init__.py +0 -0
- octopusv/ploter/chromosome_plotter.py +146 -0
- octopusv/ploter/size_plotter.py +90 -0
- octopusv/ploter/type_plotter.py +121 -0
- octopusv/py.typed +0 -0
- octopusv/report/generator.py +88 -0
- octopusv/report/image2base.py +0 -0
- octopusv/report/logo.png +0 -0
- octopusv/report/template.html +677 -0
- octopusv/stater/__init__.py +0 -0
- octopusv/stater/chromosome_analyzer.py +28 -0
- octopusv/stater/genotype_analyzer.py +289 -0
- octopusv/stater/qc_analyzer.py +128 -0
- octopusv/stater/size_analyzer.py +57 -0
- octopusv/stater/sv_stater.py +255 -0
- octopusv/stater/type_analyzer.py +24 -0
- octopusv/sv.py +117 -0
- octopusv/transformer/__init__.py +0 -0
- octopusv/transformer/base.py +13 -0
- octopusv/transformer/mp_bnd.py +14 -0
- octopusv/transformer/no_bnd.py +16 -0
- octopusv/transformer/same_chr_sv.py +44 -0
- octopusv/transformer/snmd_bndp.py +14 -0
- octopusv/transformer/stra.py +11 -0
- octopusv/utils/SV_classifier_by_chromosome.py +95 -0
- octopusv/utils/SV_classifier_by_type.py +34 -0
- octopusv/utils/__init__.py +0 -0
- octopusv/utils/construct_sample_string.py +42 -0
- octopusv/utils/normal_vcf_parser.py +126 -0
- octopusv/utils/svcf_parser.py +210 -0
- octopusv/utils/svcf_utils.py +51 -0
- octopusv/vis/__init__.py +1 -0
- octopusv-0.2.0.dist-info/LICENSE +21 -0
- octopusv-0.2.0.dist-info/METADATA +214 -0
- octopusv-0.2.0.dist-info/RECORD +81 -0
- octopusv-0.2.0.dist-info/WHEEL +4 -0
- octopusv-0.2.0.dist-info/entry_points.txt +3 -0
octopusv/__init__.py
ADDED
octopusv/__main__.py
ADDED
|
File without changes
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
import json
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
def calculate_metrics(results: dict[str, list]):
|
|
6
|
+
"""Calculate benchmark metrics including precision, recall, and F1 score."""
|
|
7
|
+
tp = len(results["tp_call"])
|
|
8
|
+
fp = len(results["fp"])
|
|
9
|
+
fn = len(results["fn"])
|
|
10
|
+
|
|
11
|
+
precision = tp / (tp + fp) if tp + fp > 0 else 0
|
|
12
|
+
recall = tp / (tp + fn) if tp + fn > 0 else 0
|
|
13
|
+
f1 = 2 * (precision * recall) / (precision + recall) if precision + recall > 0 else 0
|
|
14
|
+
|
|
15
|
+
return {"TP": tp, "FP": fp, "FN": fn, "Precision": precision, "Recall": recall, "F1": f1}
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def write_vcf(file_path: Path, events: list[tuple | object]):
|
|
19
|
+
"""Write events to VCF format file."""
|
|
20
|
+
with file_path.open("w") as f:
|
|
21
|
+
f.write("#CHROM\tPOS\tID\tREF\tALT\tQUAL\tFILTER\tINFO\n")
|
|
22
|
+
for event in events:
|
|
23
|
+
if isinstance(event, tuple): # TRA event
|
|
24
|
+
chrom, start_chrom, end_chrom, start_pos, end_pos, source_file, bnd_pattern = event
|
|
25
|
+
f.write(
|
|
26
|
+
f"{start_chrom}\t{start_pos}\t.\tN\t{bnd_pattern}\t.\tPASS\t"
|
|
27
|
+
f"SVTYPE=TRA;END={end_pos};CHR2={end_chrom};SOURCES={source_file}\n"
|
|
28
|
+
)
|
|
29
|
+
else: # Other SV events
|
|
30
|
+
f.write(
|
|
31
|
+
f"{event.chrom}\t{event.pos}\t{event.sv_id}\t{event.ref}\t{event.alt}\t"
|
|
32
|
+
f"{event.quality}\t{event.filter}\t{';'.join(f'{k}={v}' for k, v in event.info.items())}\n"
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def write_summary(file_path: Path, metrics: dict):
|
|
37
|
+
"""Write benchmark summary metrics to JSON file."""
|
|
38
|
+
with file_path.open("w") as f:
|
|
39
|
+
json.dump(metrics, f, indent=2)
|
|
@@ -0,0 +1,274 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
|
|
4
|
+
from octopusv.utils.svcf_parser import SVCFFileEventCreator
|
|
5
|
+
|
|
6
|
+
from .bench_utils import calculate_metrics, write_summary, write_vcf
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class SVBencher:
|
|
10
|
+
"""Benchmark structural variants using GIAB standards."""
|
|
11
|
+
|
|
12
|
+
def __init__(
|
|
13
|
+
self,
|
|
14
|
+
truth_file: Path,
|
|
15
|
+
call_file: Path,
|
|
16
|
+
output_dir: Path,
|
|
17
|
+
reference_distance: int = 500,
|
|
18
|
+
sequence_similarity: float = 0.7,
|
|
19
|
+
size_similarity: float = 0.7,
|
|
20
|
+
reciprocal_overlap: float = 0.0,
|
|
21
|
+
type_ignore: bool = False,
|
|
22
|
+
size_min: int = 50,
|
|
23
|
+
size_max: int = 50000,
|
|
24
|
+
pass_only: bool = False,
|
|
25
|
+
enable_sequence_comparison: bool = False,
|
|
26
|
+
):
|
|
27
|
+
"""Initialize the SV benchmarker with GIAB standard parameters."""
|
|
28
|
+
self.truth_file = truth_file
|
|
29
|
+
self.call_file = call_file
|
|
30
|
+
self.output_dir = output_dir
|
|
31
|
+
self.reference_distance = reference_distance
|
|
32
|
+
self.sequence_similarity = sequence_similarity
|
|
33
|
+
self.size_similarity = size_similarity
|
|
34
|
+
self.reciprocal_overlap = reciprocal_overlap
|
|
35
|
+
self.type_ignore = type_ignore
|
|
36
|
+
self.size_min = size_min
|
|
37
|
+
self.size_max = size_max
|
|
38
|
+
self.pass_only = pass_only
|
|
39
|
+
self.enable_sequence_comparison = enable_sequence_comparison
|
|
40
|
+
|
|
41
|
+
self.truth_events = None
|
|
42
|
+
self.call_events = None
|
|
43
|
+
self.results = None
|
|
44
|
+
|
|
45
|
+
self.logger = logging.getLogger(__name__)
|
|
46
|
+
|
|
47
|
+
def run_benchmark(self):
|
|
48
|
+
"""Run the complete benchmarking process."""
|
|
49
|
+
try:
|
|
50
|
+
self._parse_files()
|
|
51
|
+
self._compare_events()
|
|
52
|
+
self._write_results()
|
|
53
|
+
except Exception as e:
|
|
54
|
+
self.logger.error(f"Benchmarking failed: {e!s}")
|
|
55
|
+
raise
|
|
56
|
+
|
|
57
|
+
def _parse_files(self):
|
|
58
|
+
"""Parse input VCF files."""
|
|
59
|
+
self.logger.info("Parsing truth file...")
|
|
60
|
+
truth_parser = SVCFFileEventCreator([str(self.truth_file)])
|
|
61
|
+
truth_parser.parse()
|
|
62
|
+
|
|
63
|
+
self.logger.info("Parsing call file...")
|
|
64
|
+
call_parser = SVCFFileEventCreator([str(self.call_file)])
|
|
65
|
+
call_parser.parse()
|
|
66
|
+
|
|
67
|
+
self.truth_events = truth_parser.events
|
|
68
|
+
self.call_events = call_parser.events
|
|
69
|
+
|
|
70
|
+
def _filter_events(self, events: list) -> list:
|
|
71
|
+
"""Filter events based on size and FILTER criteria."""
|
|
72
|
+
filtered = []
|
|
73
|
+
for event in events:
|
|
74
|
+
# Skip if not PASS and pass_only is True
|
|
75
|
+
if self.pass_only and event.filter != "PASS":
|
|
76
|
+
continue
|
|
77
|
+
|
|
78
|
+
# Get event size
|
|
79
|
+
try:
|
|
80
|
+
if event.sv_type == "TRA":
|
|
81
|
+
size = 0 # TRA events don't have a meaningful size
|
|
82
|
+
else:
|
|
83
|
+
size = abs(event.end_pos - event.start_pos)
|
|
84
|
+
|
|
85
|
+
# Skip if outside size range (except for TRA)
|
|
86
|
+
if event.sv_type != "TRA" and (size < self.size_min or size > self.size_max):
|
|
87
|
+
continue
|
|
88
|
+
|
|
89
|
+
filtered.append(event)
|
|
90
|
+
except AttributeError as e:
|
|
91
|
+
self.logger.warning(f"Skipping malformed event: {e!s}")
|
|
92
|
+
continue
|
|
93
|
+
|
|
94
|
+
return filtered
|
|
95
|
+
|
|
96
|
+
def _meets_matching_criteria(self, truth_event, call_event) -> bool:
|
|
97
|
+
"""Check if two events meet the matching criteria based on GIAB standards."""
|
|
98
|
+
try:
|
|
99
|
+
# Check SV type unless ignored
|
|
100
|
+
if not self.type_ignore and truth_event.sv_type != call_event.sv_type:
|
|
101
|
+
return False
|
|
102
|
+
|
|
103
|
+
# Special handling for translocations
|
|
104
|
+
if truth_event.sv_type == "TRA":
|
|
105
|
+
return self._compare_tra_events(truth_event, call_event)
|
|
106
|
+
|
|
107
|
+
# Check reference distance
|
|
108
|
+
start_dist = abs(truth_event.start_pos - call_event.start_pos)
|
|
109
|
+
end_dist = abs(truth_event.end_pos - call_event.end_pos)
|
|
110
|
+
if start_dist > self.reference_distance or end_dist > self.reference_distance:
|
|
111
|
+
return False
|
|
112
|
+
|
|
113
|
+
# Check size similarity
|
|
114
|
+
truth_size = abs(truth_event.end_pos - truth_event.start_pos)
|
|
115
|
+
call_size = abs(call_event.end_pos - call_event.start_pos)
|
|
116
|
+
if truth_size == 0 or call_size == 0:
|
|
117
|
+
return False
|
|
118
|
+
size_ratio = min(truth_size, call_size) / max(truth_size, call_size)
|
|
119
|
+
if size_ratio < self.size_similarity:
|
|
120
|
+
return False
|
|
121
|
+
|
|
122
|
+
# Check sequence similarity if enabled
|
|
123
|
+
if self.enable_sequence_comparison:
|
|
124
|
+
similarity = self._calculate_sequence_similarity(truth_event, call_event)
|
|
125
|
+
if similarity < self.sequence_similarity:
|
|
126
|
+
return False
|
|
127
|
+
|
|
128
|
+
# Check reciprocal overlap
|
|
129
|
+
overlap = self._calculate_overlap(truth_event, call_event)
|
|
130
|
+
return not overlap < self.reciprocal_overlap
|
|
131
|
+
|
|
132
|
+
except AttributeError as e:
|
|
133
|
+
self.logger.warning(f"Error comparing events: {e!s}")
|
|
134
|
+
return False
|
|
135
|
+
|
|
136
|
+
def _compare_tra_events(self, truth_event, call_event) -> bool:
|
|
137
|
+
"""Special comparison logic for translocation events."""
|
|
138
|
+
try:
|
|
139
|
+
# Check chromosomes match
|
|
140
|
+
if truth_event.start_chrom != call_event.start_chrom or truth_event.end_chrom != call_event.end_chrom:
|
|
141
|
+
return False
|
|
142
|
+
|
|
143
|
+
# Check positions within reference distance
|
|
144
|
+
start_dist = abs(truth_event.start_pos - call_event.start_pos)
|
|
145
|
+
end_dist = abs(truth_event.end_pos - call_event.end_pos)
|
|
146
|
+
if start_dist > self.reference_distance or end_dist > self.reference_distance:
|
|
147
|
+
return False
|
|
148
|
+
|
|
149
|
+
# Check strand consistency if available
|
|
150
|
+
if hasattr(truth_event, "strand") and hasattr(call_event, "strand"):
|
|
151
|
+
if truth_event.strand != call_event.strand:
|
|
152
|
+
return False
|
|
153
|
+
|
|
154
|
+
return True
|
|
155
|
+
|
|
156
|
+
except AttributeError as e:
|
|
157
|
+
self.logger.warning(f"Error comparing TRA events: {e!s}")
|
|
158
|
+
return False
|
|
159
|
+
|
|
160
|
+
def _calculate_sequence_similarity(self, truth_event, call_event) -> float:
|
|
161
|
+
"""Calculate sequence similarity between events."""
|
|
162
|
+
truth_seq = self._get_sequence_from_event(truth_event)
|
|
163
|
+
call_seq = self._get_sequence_from_event(call_event)
|
|
164
|
+
|
|
165
|
+
# If no sequence information available, assume similarity
|
|
166
|
+
if not truth_seq or not call_seq:
|
|
167
|
+
return 1.0
|
|
168
|
+
|
|
169
|
+
try:
|
|
170
|
+
from Levenshtein import ratio
|
|
171
|
+
|
|
172
|
+
return ratio(truth_seq, call_seq)
|
|
173
|
+
except ImportError:
|
|
174
|
+
self.logger.warning("Levenshtein package not available, using simple comparison")
|
|
175
|
+
return float(truth_seq == call_seq)
|
|
176
|
+
|
|
177
|
+
def _get_sequence_from_event(self, event) -> str | None:
|
|
178
|
+
"""Extract sequence information from an event."""
|
|
179
|
+
try:
|
|
180
|
+
if hasattr(event, "alt_seq") and event.alt_seq:
|
|
181
|
+
return event.alt_seq
|
|
182
|
+
|
|
183
|
+
if hasattr(event, "info"):
|
|
184
|
+
seq = event.info.get("SVSEQ", "")
|
|
185
|
+
if seq:
|
|
186
|
+
return seq
|
|
187
|
+
seq = event.info.get("SEQ", "")
|
|
188
|
+
if seq:
|
|
189
|
+
return seq
|
|
190
|
+
|
|
191
|
+
if hasattr(event, "alt") and len(event.alt) > 1 and not event.alt.startswith("<"):
|
|
192
|
+
return event.alt
|
|
193
|
+
|
|
194
|
+
return None
|
|
195
|
+
|
|
196
|
+
except AttributeError:
|
|
197
|
+
return None
|
|
198
|
+
|
|
199
|
+
def _calculate_overlap(self, event1, event2) -> float:
|
|
200
|
+
"""Calculate reciprocal overlap between two events."""
|
|
201
|
+
try:
|
|
202
|
+
if event1.sv_type == "TRA" or event2.sv_type == "TRA":
|
|
203
|
+
return 1.0 # TRA events are compared by breakpoints only
|
|
204
|
+
|
|
205
|
+
overlap_start = max(event1.start_pos, event2.start_pos)
|
|
206
|
+
overlap_end = min(event1.end_pos, event2.end_pos)
|
|
207
|
+
|
|
208
|
+
if overlap_start >= overlap_end:
|
|
209
|
+
return 0.0
|
|
210
|
+
|
|
211
|
+
overlap_length = overlap_end - overlap_start
|
|
212
|
+
event1_length = event1.end_pos - event1.start_pos
|
|
213
|
+
event2_length = event2.end_pos - event2.start_pos
|
|
214
|
+
|
|
215
|
+
if event1_length == 0 or event2_length == 0:
|
|
216
|
+
return 0.0
|
|
217
|
+
|
|
218
|
+
overlap_ratio1 = overlap_length / event1_length
|
|
219
|
+
overlap_ratio2 = overlap_length / event2_length
|
|
220
|
+
|
|
221
|
+
return min(overlap_ratio1, overlap_ratio2)
|
|
222
|
+
|
|
223
|
+
except AttributeError as e:
|
|
224
|
+
self.logger.warning(f"Error calculating overlap: {e!s}")
|
|
225
|
+
return 0.0
|
|
226
|
+
|
|
227
|
+
def _compare_events(self):
|
|
228
|
+
"""Compare truth and call events to identify matches."""
|
|
229
|
+
self.logger.info("Filtering events...")
|
|
230
|
+
filtered_truth = self._filter_events(self.truth_events)
|
|
231
|
+
filtered_calls = self._filter_events(self.call_events)
|
|
232
|
+
|
|
233
|
+
self.logger.info("Comparing events...")
|
|
234
|
+
tp_base, tp_call, fp = [], [], []
|
|
235
|
+
matched_truth = set()
|
|
236
|
+
matched_calls = set()
|
|
237
|
+
|
|
238
|
+
# Compare each call against truth
|
|
239
|
+
for call_event in filtered_calls:
|
|
240
|
+
found_match = False
|
|
241
|
+
for truth_event in filtered_truth:
|
|
242
|
+
if truth_event in matched_truth:
|
|
243
|
+
continue
|
|
244
|
+
|
|
245
|
+
if self._meets_matching_criteria(truth_event, call_event):
|
|
246
|
+
tp_call.append(call_event)
|
|
247
|
+
tp_base.append(truth_event)
|
|
248
|
+
matched_truth.add(truth_event)
|
|
249
|
+
matched_calls.add(call_event)
|
|
250
|
+
found_match = True
|
|
251
|
+
break
|
|
252
|
+
|
|
253
|
+
if not found_match:
|
|
254
|
+
fp.append(call_event)
|
|
255
|
+
|
|
256
|
+
# Collect unmatched truth events as FN
|
|
257
|
+
fn = [event for event in filtered_truth if event not in matched_truth]
|
|
258
|
+
|
|
259
|
+
self.results = {"tp_base": tp_base, "tp_call": tp_call, "fp": fp, "fn": fn}
|
|
260
|
+
|
|
261
|
+
self.logger.info(f"Found {len(tp_call)} true positives, {len(fp)} false positives, {len(fn)} false negatives")
|
|
262
|
+
|
|
263
|
+
def _write_results(self):
|
|
264
|
+
"""Write benchmark results to output directory."""
|
|
265
|
+
self.logger.info("Writing results...")
|
|
266
|
+
self.output_dir.mkdir(parents=True, exist_ok=True)
|
|
267
|
+
|
|
268
|
+
write_vcf(self.output_dir / "tp-base.vcf", self.results["tp_base"])
|
|
269
|
+
write_vcf(self.output_dir / "tp-call.vcf", self.results["tp_call"])
|
|
270
|
+
write_vcf(self.output_dir / "fp.vcf", self.results["fp"])
|
|
271
|
+
write_vcf(self.output_dir / "fn.vcf", self.results["fn"])
|
|
272
|
+
|
|
273
|
+
metrics = calculate_metrics(self.results)
|
|
274
|
+
write_summary(self.output_dir / "summary.json", metrics)
|
octopusv/cli/__init__.py
ADDED
|
File without changes
|
octopusv/cli/bench.py
ADDED
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
|
|
4
|
+
import typer
|
|
5
|
+
from rich.logging import RichHandler
|
|
6
|
+
|
|
7
|
+
from octopusv.bencher.sv_bencher import SVBencher
|
|
8
|
+
|
|
9
|
+
# Set logging style
|
|
10
|
+
FORMAT = "%(message)s"
|
|
11
|
+
logging.basicConfig(level="INFO", format=FORMAT, datefmt="[%X]", handlers=[RichHandler()])
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def bench(
|
|
15
|
+
truth_file: Path = typer.Argument(..., help="Path to the truth (ground truth) SVCF file."),
|
|
16
|
+
call_file: Path = typer.Argument(..., help="Path to the call (test) SVCF file."),
|
|
17
|
+
output_dir: Path = typer.Option(..., "--output-dir", "-o", help="Output directory for benchmark results."),
|
|
18
|
+
reference_distance: int = typer.Option(
|
|
19
|
+
500, "--reference-distance", "-r", help="Max reference location distance (default: 500)"
|
|
20
|
+
),
|
|
21
|
+
size_similarity: float = typer.Option(
|
|
22
|
+
0.7, "--size-similarity", "-P", help="Min pct size similarity (minsize/maxsize) (default: 0.7)"
|
|
23
|
+
),
|
|
24
|
+
reciprocal_overlap: float = typer.Option(
|
|
25
|
+
0.0, "--reciprocal-overlap", "-O", help="Min pct reciprocal overlap (default: 0.0)"
|
|
26
|
+
),
|
|
27
|
+
type_ignore: bool = typer.Option(
|
|
28
|
+
False, "--type-ignore", "-t", help="Variant types don't need to match to compare (default: False)"
|
|
29
|
+
),
|
|
30
|
+
size_min: int = typer.Option(
|
|
31
|
+
50, "--size-min", "-s", help="Minimum variant size to consider from test calls (default: 50)"
|
|
32
|
+
),
|
|
33
|
+
size_max: int = typer.Option(50000, "--size-max", help="Maximum variant size to consider (default: 50000)"),
|
|
34
|
+
pass_only: bool = typer.Option(
|
|
35
|
+
False, "--pass-only", help="Only consider variants with FILTER == PASS (default: False)"
|
|
36
|
+
),
|
|
37
|
+
enable_sequence_comparison: bool = typer.Option(
|
|
38
|
+
False,
|
|
39
|
+
"--enable-sequence-comparison",
|
|
40
|
+
help="Enable sequence similarity comparison if sequences available (default: False)",
|
|
41
|
+
),
|
|
42
|
+
sequence_similarity: float = typer.Option(
|
|
43
|
+
0.7,
|
|
44
|
+
"--sequence-similarity",
|
|
45
|
+
"-p",
|
|
46
|
+
help="Min sequence similarity when sequence comparison enabled (default: 0.7)",
|
|
47
|
+
),
|
|
48
|
+
):
|
|
49
|
+
"""Benchmark structural variation calls against a truth set using GIAB standards.
|
|
50
|
+
|
|
51
|
+
This tool implements the GIAB (Genome in a Bottle) consortium's recommendations for
|
|
52
|
+
structural variant benchmarking. It compares a test VCF file against a truth set
|
|
53
|
+
and calculates precision, recall, and F1 scores.
|
|
54
|
+
|
|
55
|
+
Default thresholds are set according to GIAB standards:
|
|
56
|
+
- 500bp maximum reference distance
|
|
57
|
+
- 70% size similarity
|
|
58
|
+
- 0% reciprocal overlap
|
|
59
|
+
- 50bp minimum variant size
|
|
60
|
+
- 50kb maximum variant size
|
|
61
|
+
|
|
62
|
+
Results are written to the specified output directory, including:
|
|
63
|
+
- True positives from the baseline (tp-base.vcf)
|
|
64
|
+
- True positives from the test set (tp-call.vcf)
|
|
65
|
+
- False positives (fp.vcf)
|
|
66
|
+
- False negatives (fn.vcf)
|
|
67
|
+
- Summary metrics in JSON format (summary.json)
|
|
68
|
+
"""
|
|
69
|
+
try:
|
|
70
|
+
bencher = SVBencher(
|
|
71
|
+
truth_file,
|
|
72
|
+
call_file,
|
|
73
|
+
output_dir,
|
|
74
|
+
reference_distance=reference_distance,
|
|
75
|
+
sequence_similarity=sequence_similarity,
|
|
76
|
+
size_similarity=size_similarity,
|
|
77
|
+
reciprocal_overlap=reciprocal_overlap,
|
|
78
|
+
type_ignore=type_ignore,
|
|
79
|
+
size_min=size_min,
|
|
80
|
+
size_max=size_max,
|
|
81
|
+
pass_only=pass_only,
|
|
82
|
+
enable_sequence_comparison=enable_sequence_comparison,
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
typer.echo("Starting benchmark...")
|
|
86
|
+
bencher.run_benchmark()
|
|
87
|
+
typer.echo(f"Benchmark results written to {output_dir}")
|
|
88
|
+
|
|
89
|
+
except Exception as e:
|
|
90
|
+
typer.echo(f"Error during benchmarking: {e!s}", err=True)
|
|
91
|
+
raise typer.Exit(code=1)
|
octopusv/cli/cli.py
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
import sys
|
|
3
|
+
|
|
4
|
+
import typer
|
|
5
|
+
from rich.logging import RichHandler
|
|
6
|
+
|
|
7
|
+
from octopusv import __version__
|
|
8
|
+
|
|
9
|
+
from .bench import bench
|
|
10
|
+
from .convert import correct
|
|
11
|
+
from .merge import merge
|
|
12
|
+
from .plot import plot
|
|
13
|
+
from .stat import stat
|
|
14
|
+
from .svcf2bed import svcf2bed
|
|
15
|
+
from .svcf2bedpe import svcf2bedpe
|
|
16
|
+
from .svcf2vcf import svcf2vcf
|
|
17
|
+
|
|
18
|
+
app = typer.Typer(
|
|
19
|
+
epilog=f"{typer.style('Agent Octopus Code V helps you dive deep into the structural variations ocean!', fg=typer.colors.GREEN, bold=True)}",
|
|
20
|
+
context_settings={"help_option_names": ["-h", "--help"]},
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
FORMAT = "%(message)s"
|
|
25
|
+
logging.basicConfig(level="INFO", format=FORMAT, datefmt="[%X]", handlers=[RichHandler()])
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
app.command()(correct) # Command to initiate convert functionality.
|
|
29
|
+
app.command()(merge) # Command to initiate merge functionality.
|
|
30
|
+
app.command(name="benchmark")(bench) # Command to initiate bench functionality.
|
|
31
|
+
app.command()(stat) # Command to initiate stat functionality.
|
|
32
|
+
app.command()(plot) # Command to initiate plot functionality.
|
|
33
|
+
app.command()(svcf2vcf)
|
|
34
|
+
app.command()(svcf2bed)
|
|
35
|
+
app.command()(svcf2bedpe)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@app.callback()
|
|
39
|
+
def display_version_info():
|
|
40
|
+
"""Display the version information."""
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
display_version_info.__doc__ = f"""{typer.style("Version", fg=typer.colors.YELLOW, bold=True)}: {typer.style(f"{__version__}", fg=typer.colors.GREEN, bold=True)}"""
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
# For documentation purposes
|
|
47
|
+
if "sphinx" in sys.modules and __name__ != "__main__":
|
|
48
|
+
# Create the typer click object to generate docs with sphinx-click
|
|
49
|
+
typer_click_object = typer.main.get_command(app)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
if __name__ == "__main__":
|
|
53
|
+
app()
|