evolutionary-stability-optimizer 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- eso/__init__.py +11 -0
- eso/cli.py +111 -0
- eso/codon_usage.py +96 -0
- eso/constraints.py +179 -0
- eso/custom_score.py +170 -0
- eso/data/__init__.py +0 -0
- eso/data/human-antibody-heavy-chain-codon-frequencies.csv +62 -0
- eso/data/human-antibody-light-chain-codon-frequencies.csv +62 -0
- eso/detection/__init__.py +0 -0
- eso/detection/_overlap.py +112 -0
- eso/detection/common_motifs.py +81 -0
- eso/detection/dispatch.py +185 -0
- eso/detection/methylation.py +96 -0
- eso/detection/motif_utils.py +77 -0
- eso/detection/recombination.py +329 -0
- eso/detection/slippage.py +249 -0
- eso/detection/staubility_variant.py +284 -0
- eso/io_utils.py +312 -0
- eso/optimize.py +266 -0
- eso/pipeline.py +322 -0
- eso/report.py +57 -0
- eso/sequence_utils.py +41 -0
- evolutionary_stability_optimizer-0.1.0.dist-info/METADATA +501 -0
- evolutionary_stability_optimizer-0.1.0.dist-info/RECORD +27 -0
- evolutionary_stability_optimizer-0.1.0.dist-info/WHEEL +4 -0
- evolutionary_stability_optimizer-0.1.0.dist-info/entry_points.txt +3 -0
- evolutionary_stability_optimizer-0.1.0.dist-info/licenses/LICENSE +21 -0
eso/optimize.py
ADDED
|
@@ -0,0 +1,266 @@
|
|
|
1
|
+
"""DNAChisel-based sequence optimization: enforce translation, GC-content
|
|
2
|
+
windows, codon usage, and avoidance of detected hypermutable sites.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
import warnings
|
|
6
|
+
|
|
7
|
+
import pandas as pd
|
|
8
|
+
import dnachisel
|
|
9
|
+
from dnachisel.DnaOptimizationProblem import NoSolutionError
|
|
10
|
+
|
|
11
|
+
from eso.codon_usage import CODON_USAGE_TABLES
|
|
12
|
+
from eso.constraints import (
|
|
13
|
+
convert_df_to_constraints,
|
|
14
|
+
exclusion_site_correcter,
|
|
15
|
+
recombination_to_multiple_avoidance_sites,
|
|
16
|
+
)
|
|
17
|
+
from eso.custom_score import CustomScore
|
|
18
|
+
from eso.detection.slippage import modify_df_slippage
|
|
19
|
+
|
|
20
|
+
# Substring DNAChisel raises when an AvoidPattern's location doesn't align to
|
|
21
|
+
# the codon grid under EnforceTranslation: DnaOptimizationProblem.resolve_constraint
|
|
22
|
+
# can compute a "jointly mutable" mutation-space region (accounting for codon
|
|
23
|
+
# interdependencies) that ends up not overlapping the constraint's own declared
|
|
24
|
+
# location at all. AvoidPattern.localized() correctly returns None for that
|
|
25
|
+
# (they genuinely don't overlap) - but the DNAChisel caller doesn't check for
|
|
26
|
+
# None before calling .evaluate() on it. Confirmed via direct reproduction with
|
|
27
|
+
# a non-codon-aligned 2nt repeat and, separately, a 1nt homopolymer run; a
|
|
28
|
+
# codon-aligned repeat (e.g. a 3nt unit) never triggers it.
|
|
29
|
+
#
|
|
30
|
+
# An earlier fix widened every avoid-window outward to whole codons before
|
|
31
|
+
# building the AvoidPattern, which does dodge this crash - but it also changes
|
|
32
|
+
# what's being asked for: "avoid this pattern at this spot" becomes "avoid this
|
|
33
|
+
# pattern ANYWHERE in the whole codon", which for a single-nucleotide pattern
|
|
34
|
+
# in a homopolymer run can make the constraint unsatisfiable for every codon in
|
|
35
|
+
# the run (e.g. all-Phe TTT/TTC codons always contain a T), silently dropping
|
|
36
|
+
# every constraint and leaving the site untouched - no crash, but no effect
|
|
37
|
+
# either (confirmed directly: a "ATG"+"T"*12+"TAA" homopolymer produced
|
|
38
|
+
# num_edits=0 with that approach). So instead of forcing this by widening the
|
|
39
|
+
# window, the retry loop below simply catches the crash like any other
|
|
40
|
+
# unsatisfiable-constraint case and drops whichever constraints are actually
|
|
41
|
+
# still failing, exactly as it already did for DNAChisel's own
|
|
42
|
+
# NoSolutionError-with-no-named-constraint case.
|
|
43
|
+
_LOCALIZED_NONE_CRASH_MESSAGE = "'NoneType' object has no attribute 'evaluate'"
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _warn_dropped_constraint(constraint):
|
|
47
|
+
warnings.warn(
|
|
48
|
+
f"Could not satisfy constraint {constraint} - dropping it and continuing without it. "
|
|
49
|
+
"This most often happens when disrupting a detected hypermutable site would require a "
|
|
50
|
+
"change that conflicts with another hard constraint, most commonly translation "
|
|
51
|
+
"preservation (e.g. every synonymous codon at that position still contains the pattern "
|
|
52
|
+
"being avoided, such as a homopolymer landing on a Met/Trp codon with no alternative). "
|
|
53
|
+
"The site this constraint was protecting was left unmodified in the final sequence.",
|
|
54
|
+
stacklevel=3,
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _codon_optimization_objectives(organism_name, orf_regions, method):
|
|
59
|
+
"""Build DNAChisel CodonOptimize objectives.
|
|
60
|
+
|
|
61
|
+
`organism_name` may be one of eso.codon_usage.CODON_USAGE_TABLES' custom
|
|
62
|
+
tables (not present in the Kazusa database), or any species/TaxID
|
|
63
|
+
supported by python-codon-tables.
|
|
64
|
+
"""
|
|
65
|
+
if organism_name in CODON_USAGE_TABLES:
|
|
66
|
+
codon_usage_table = CODON_USAGE_TABLES[organism_name]()
|
|
67
|
+
return [
|
|
68
|
+
dnachisel.CodonOptimize(location=orf, codon_usage_table=codon_usage_table.copy(), method=method)
|
|
69
|
+
for orf in orf_regions
|
|
70
|
+
]
|
|
71
|
+
|
|
72
|
+
try:
|
|
73
|
+
return [dnachisel.CodonOptimize(species=organism_name, location=orf, method=method) for orf in orf_regions]
|
|
74
|
+
except Exception:
|
|
75
|
+
print('unknown organism, no codon optimization!')
|
|
76
|
+
return []
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def optimization_engine(
|
|
80
|
+
seq,
|
|
81
|
+
mini_gc=0.3,
|
|
82
|
+
maxi_gc=0.7,
|
|
83
|
+
window_size_gc=50,
|
|
84
|
+
method='use_best_codon',
|
|
85
|
+
organism_name='not_specified',
|
|
86
|
+
custom_score_fn=None,
|
|
87
|
+
custom_score_minimize=False,
|
|
88
|
+
df_recombination=None,
|
|
89
|
+
df_slippage=None,
|
|
90
|
+
df_motifs=None,
|
|
91
|
+
orf_regions=(),
|
|
92
|
+
exclusion_regions=(),
|
|
93
|
+
):
|
|
94
|
+
"""Optimize `seq` for codon usage, GC content, and (if hotspot dataframes
|
|
95
|
+
are given) avoidance of detected recombination/slippage/methylation sites,
|
|
96
|
+
while preserving the amino-acid translation over `orf_regions`.
|
|
97
|
+
|
|
98
|
+
Citation: Jack, Leonard, Mishler, Renda, Leon, Suarez & Barrick (2015).
|
|
99
|
+
"Predicting the genetic stability of engineered DNA sequences with the
|
|
100
|
+
EFM Calculator." ACS Synthetic Biology. DOI: 10.1021/acssynbio.5b00068
|
|
101
|
+
|
|
102
|
+
Parameters
|
|
103
|
+
----------
|
|
104
|
+
seq: str
|
|
105
|
+
DNA sequence (ACGT alphabet).
|
|
106
|
+
mini_gc, maxi_gc: float in [0, 1]
|
|
107
|
+
Allowed GC-content range within any `window_size_gc`-nt window.
|
|
108
|
+
method: {"use_best_codon", "match_codon_usage", "harmonize_rca"}
|
|
109
|
+
Codon optimization strategy (see DNAChisel's CodonOptimize).
|
|
110
|
+
organism_name: str
|
|
111
|
+
Host organism for codon optimization: one of
|
|
112
|
+
eso.codon_usage.CODON_USAGE_TABLES' keys, a python-codon-tables
|
|
113
|
+
species name/TaxID, or "not_specified" to skip codon optimization.
|
|
114
|
+
custom_score_fn: callable(str) -> float, or None
|
|
115
|
+
If given, replaces the CodonOptimize (CAI/tAI-style) objective with
|
|
116
|
+
eso.custom_score.CustomScore wrapping this function (higher is
|
|
117
|
+
better sequence, unless custom_score_minimize=True). `organism_name`
|
|
118
|
+
and `method` are then ignored. Called once per ORF, on the whole ORF,
|
|
119
|
+
for every trial mutation during optimization - can be slow on long
|
|
120
|
+
sequences or an expensive custom_score_fn (a warning is raised).
|
|
121
|
+
custom_score_minimize: bool
|
|
122
|
+
If True, treats a lower custom_score_fn value as better.
|
|
123
|
+
df_recombination, df_slippage, df_motifs: pandas.DataFrame or None
|
|
124
|
+
Detected hotspots to avoid (from eso.detection.*); pass None or an
|
|
125
|
+
empty dataframe to skip a given constraint type.
|
|
126
|
+
orf_regions: sequence of (start, end) tuples
|
|
127
|
+
Regions to keep in-frame and translation-preserving. Defaults to the
|
|
128
|
+
whole sequence, trimmed to a multiple of 3.
|
|
129
|
+
exclusion_regions: sequence of (start, end) tuples
|
|
130
|
+
Regions that must not be modified.
|
|
131
|
+
|
|
132
|
+
Returns
|
|
133
|
+
-------
|
|
134
|
+
(final_sequence, objectives_summary, num_edits)
|
|
135
|
+
"""
|
|
136
|
+
if df_recombination is None:
|
|
137
|
+
df_recombination = pd.DataFrame()
|
|
138
|
+
if df_slippage is None:
|
|
139
|
+
df_slippage = pd.DataFrame()
|
|
140
|
+
if df_motifs is None:
|
|
141
|
+
df_motifs = pd.DataFrame()
|
|
142
|
+
|
|
143
|
+
if len(orf_regions) == 0:
|
|
144
|
+
new_last_index = ((len(seq) - 1) // 3) * 3
|
|
145
|
+
orf_regions = [(0, new_last_index)]
|
|
146
|
+
|
|
147
|
+
if custom_score_fn is not None:
|
|
148
|
+
# Scoped to each ORF individually, matching how _codon_optimization_objectives
|
|
149
|
+
# already scopes CodonOptimize to `location=orf` - without this, custom scoring
|
|
150
|
+
# would apply across the whole sequence including any non-ORF flanks (UTRs,
|
|
151
|
+
# locked regions), inconsistent with codon-usage scoring's own ORF-only behavior.
|
|
152
|
+
obj = [
|
|
153
|
+
CustomScore(custom_score_fn, location=orf, minimize=custom_score_minimize)
|
|
154
|
+
for orf in orf_regions
|
|
155
|
+
]
|
|
156
|
+
else:
|
|
157
|
+
obj = _codon_optimization_objectives(organism_name, orf_regions, method)
|
|
158
|
+
|
|
159
|
+
cnst = [dnachisel.EnforceGCContent(mini=mini_gc, maxi=maxi_gc, window=window_size_gc)]
|
|
160
|
+
for orf in orf_regions:
|
|
161
|
+
cnst.append(dnachisel.EnforceTranslation(location=orf))
|
|
162
|
+
|
|
163
|
+
cnst.extend(dnachisel.AvoidChanges(location=region) for region in exclusion_regions)
|
|
164
|
+
|
|
165
|
+
if not df_recombination.empty:
|
|
166
|
+
df_rec = recombination_to_multiple_avoidance_sites(df_recombination, exclusion_regions)
|
|
167
|
+
df_rec = exclusion_site_correcter(df_rec, exclusion_regions)
|
|
168
|
+
cnst.extend(convert_df_to_constraints(df_rec))
|
|
169
|
+
|
|
170
|
+
if not df_slippage.empty:
|
|
171
|
+
df_slip = modify_df_slippage(df_slippage)
|
|
172
|
+
df_slip = exclusion_site_correcter(df_slip.copy(), exclusion_regions)
|
|
173
|
+
cnst.extend(convert_df_to_constraints(df_slip))
|
|
174
|
+
|
|
175
|
+
if not df_motifs.empty:
|
|
176
|
+
df_mot = df_motifs.copy()[['start_index', 'end_index', 'actual_site']].rename(
|
|
177
|
+
columns={'start_index': 'start', 'end_index': 'end', 'actual_site': 'sequence'})
|
|
178
|
+
df_mot.loc[:, 'start'] = df_mot['start'].astype(int)
|
|
179
|
+
# eso.detection.methylation's end_index is INCLUSIVE (the index of the
|
|
180
|
+
# motif's last nucleotide), but everything downstream of here -
|
|
181
|
+
# exclusion_site_correcter, convert_df_to_constraints, and DNAChisel's
|
|
182
|
+
# own Location - uses EXCLUSIVE end (matching Python slicing). Without
|
|
183
|
+
# the +1, every motif's AvoidPattern location was exactly one
|
|
184
|
+
# nucleotide too short to ever contain its own pattern, so
|
|
185
|
+
# DNAChisel could never find it there and always reported the
|
|
186
|
+
# constraint as trivially satisfied - methylation-motif avoidance
|
|
187
|
+
# silently did nothing. Confirmed directly: before this fix, a GATC
|
|
188
|
+
# motif passed as df_motifs survived optimization completely
|
|
189
|
+
# untouched (0 edits); see tests/test_optimize.py.
|
|
190
|
+
df_mot.loc[:, 'end'] = df_mot['end'].astype(int) + 1
|
|
191
|
+
df_mot = exclusion_site_correcter(df_mot, exclusion_regions)
|
|
192
|
+
cnst.extend(convert_df_to_constraints(df_mot))
|
|
193
|
+
|
|
194
|
+
problem = None
|
|
195
|
+
flag = 0
|
|
196
|
+
# 60 was chosen empirically back when constraint counts were always small
|
|
197
|
+
# (num_sites capped how many hotspot-avoidance constraints existed at
|
|
198
|
+
# all). Since num_sites became report-only and constraint-building
|
|
199
|
+
# started using every raw, uncollapsed candidate (see
|
|
200
|
+
# docs/detector-comparisons.md's overlap-collapse coverage-gap entry), a
|
|
201
|
+
# single dense/repetitive sequence can legitimately produce well over 60
|
|
202
|
+
# individual AvoidPattern constraints - a fixed 60-round budget could then
|
|
203
|
+
# raise NoSolutionError purely from running out of retries, not from any
|
|
204
|
+
# actual unresolvable conflict. Scale the budget with how many
|
|
205
|
+
# constraints there actually are (computed once, before any get dropped,
|
|
206
|
+
# so it doesn't shrink alongside `cnst` as rounds progress), keeping 60 as
|
|
207
|
+
# the floor for the common, small-constraint-count case this was
|
|
208
|
+
# originally tuned for.
|
|
209
|
+
retry_budget = max(60, len(cnst))
|
|
210
|
+
while flag < retry_budget: # retry, dropping unsatisfiable/crashing constraints
|
|
211
|
+
problem = dnachisel.DnaOptimizationProblem(sequence=str(seq), constraints=cnst, objectives=obj)
|
|
212
|
+
try:
|
|
213
|
+
problem.resolve_constraints()
|
|
214
|
+
break
|
|
215
|
+
except NoSolutionError as e:
|
|
216
|
+
if e.constraint is not None:
|
|
217
|
+
_warn_dropped_constraint(e.constraint)
|
|
218
|
+
cnst.remove(e.constraint)
|
|
219
|
+
flag += 1
|
|
220
|
+
continue
|
|
221
|
+
drop_and_retry = True
|
|
222
|
+
except AttributeError as e:
|
|
223
|
+
# See _LOCALIZED_NONE_CRASH_MESSAGE above: a genuine DNAChisel-internal
|
|
224
|
+
# crash on non-codon-aligned AvoidPattern locations. Only swallow this
|
|
225
|
+
# exact crash - anything else with this type is a real bug, re-raise it.
|
|
226
|
+
if _LOCALIZED_NONE_CRASH_MESSAGE not in str(e):
|
|
227
|
+
raise
|
|
228
|
+
drop_and_retry = True
|
|
229
|
+
|
|
230
|
+
if drop_and_retry:
|
|
231
|
+
# Either DNAChisel's own final consistency check
|
|
232
|
+
# (perform_final_constraints_check, run at the end of
|
|
233
|
+
# resolve_constraints) failed without identifying a single culprit
|
|
234
|
+
# constraint - NoSolutionError.constraint defaults to None, seen in
|
|
235
|
+
# practice with several individually-resolvable but densely packed
|
|
236
|
+
# AvoidPattern constraints that regress each other by the time
|
|
237
|
+
# solving reaches the last one - or resolve_constraint itself
|
|
238
|
+
# crashed on a non-codon-aligned AvoidPattern location. Either way,
|
|
239
|
+
# `cnst.remove(None)` isn't possible here, so instead re-evaluate
|
|
240
|
+
# every constraint ourselves and drop whichever ones are still
|
|
241
|
+
# actually failing, so the retry loop can keep making progress the
|
|
242
|
+
# same way it does for the normal case. Constraints like
|
|
243
|
+
# EnforceGCContent have location=None until
|
|
244
|
+
# initialized_on_problem() fills it in (as a copy, not in-place) -
|
|
245
|
+
# evaluate that initialized copy, not the raw constraint, which
|
|
246
|
+
# would crash the same way on its own unset location.
|
|
247
|
+
failing = [
|
|
248
|
+
c for c in cnst
|
|
249
|
+
if not c.initialized_on_problem(problem, role='constraint').evaluate(problem).passes
|
|
250
|
+
]
|
|
251
|
+
if not failing:
|
|
252
|
+
raise
|
|
253
|
+
for constraint in failing:
|
|
254
|
+
_warn_dropped_constraint(constraint)
|
|
255
|
+
cnst.remove(constraint)
|
|
256
|
+
flag += 1
|
|
257
|
+
else:
|
|
258
|
+
raise NoSolutionError(
|
|
259
|
+
f"More than {retry_budget} hard constraints were not satisfied ({flag}).", problem=problem)
|
|
260
|
+
|
|
261
|
+
problem.optimize()
|
|
262
|
+
obj_description = problem.objectives_text_summary()
|
|
263
|
+
num_edits = problem.number_of_edits()
|
|
264
|
+
final_sequence = str(problem.sequence).upper()
|
|
265
|
+
|
|
266
|
+
return final_sequence, obj_description, num_edits
|
eso/pipeline.py
ADDED
|
@@ -0,0 +1,322 @@
|
|
|
1
|
+
"""End-to-end orchestration: for each sequence file, run codon/GC optimization,
|
|
2
|
+
detect hypermutable sites, re-optimize while avoiding them, and write out
|
|
3
|
+
per-file CSVs plus a Word comparison report.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from os import path
|
|
8
|
+
|
|
9
|
+
import numpy as np
|
|
10
|
+
import pandas as pd
|
|
11
|
+
|
|
12
|
+
from eso.detection.common_motifs import load_common_motifs
|
|
13
|
+
from eso.detection.dispatch import (
|
|
14
|
+
collapse_recombination_sites,
|
|
15
|
+
collapse_slippage_sites,
|
|
16
|
+
find_recombination_candidates,
|
|
17
|
+
find_slippage_candidates,
|
|
18
|
+
recombination_sites_for_constraints,
|
|
19
|
+
slippage_sites_for_constraints,
|
|
20
|
+
)
|
|
21
|
+
from eso.detection.methylation import load_motifs, find_motif_sites
|
|
22
|
+
from eso.io_utils import file_opener, file_stem, relevant_file_paths, test_input
|
|
23
|
+
from eso.optimize import optimization_engine
|
|
24
|
+
from eso.report import create_word_document_with_highlighted_differences
|
|
25
|
+
from eso.sequence_utils import parse_region
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def suspect_site_extractor(target_seq, compute_motifs, num_sites, motifs_path=None,
|
|
29
|
+
common_motifs=None, recombination_mode='thorough',
|
|
30
|
+
slippage_mode='default'):
|
|
31
|
+
"""Detect recombination and slippage sites (and, if `compute_motifs`, methylation
|
|
32
|
+
motif sites) in `target_seq`. Returns a dict of dataframes keyed by
|
|
33
|
+
'df_recombination', 'df_slippage', and optionally 'df_motifs' - each
|
|
34
|
+
collapsed to one representative per distinct site and limited to
|
|
35
|
+
`num_sites`, for reporting - plus 'df_recombination_raw'/'df_slippage_raw',
|
|
36
|
+
a reduced-but-not-collapsed view (see
|
|
37
|
+
eso.detection.slippage.slippage_sites_for_constraints) meant for building
|
|
38
|
+
correction constraints from instead.
|
|
39
|
+
|
|
40
|
+
Detection itself only runs once per category; the report and
|
|
41
|
+
constraint-building views are both derived from that single run. Use the
|
|
42
|
+
_raw dataframes (not the collapsed ones) to build correction constraints:
|
|
43
|
+
collapsing (as the report view does) can silently drop a genuinely
|
|
44
|
+
distinct, only-partially-overlapping hotspot entirely, leaving the region
|
|
45
|
+
it uniquely covered with no fix. `num_sites` is report-only for the same
|
|
46
|
+
reason - limiting it would reintroduce that exact silent-coverage-loss
|
|
47
|
+
risk for whichever candidates it cuts.
|
|
48
|
+
|
|
49
|
+
recombination_mode: see eso.detection.dispatch.find_recombination_sites -
|
|
50
|
+
"thorough" (default, Levenshtein-tolerant) or "fast" (exact-match only).
|
|
51
|
+
slippage_mode: see eso.detection.dispatch.find_slippage_sites -
|
|
52
|
+
"default" or "fast" (equivalent sensitivity; "default" is also faster).
|
|
53
|
+
common_motifs: list of str or None
|
|
54
|
+
Names from eso.detection.common_motifs.COMMON_MOTIFS (currently
|
|
55
|
+
"dam", "dcm") to include alongside any `motifs_path` file. At least
|
|
56
|
+
one of `motifs_path`/`common_motifs` is required if `compute_motifs`.
|
|
57
|
+
"""
|
|
58
|
+
df_recombination_candidates = find_recombination_candidates(target_seq, mode=recombination_mode)
|
|
59
|
+
df_slippage_candidates = find_slippage_candidates(target_seq, mode=slippage_mode)
|
|
60
|
+
|
|
61
|
+
sites_collector = {
|
|
62
|
+
'df_recombination': collapse_recombination_sites(
|
|
63
|
+
df_recombination_candidates, num_sites, mode=recombination_mode),
|
|
64
|
+
'df_slippage': collapse_slippage_sites(df_slippage_candidates, num_sites, mode=slippage_mode),
|
|
65
|
+
'df_recombination_raw': recombination_sites_for_constraints(df_recombination_candidates, mode=recombination_mode),
|
|
66
|
+
'df_slippage_raw': slippage_sites_for_constraints(df_slippage_candidates, mode=slippage_mode),
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
if compute_motifs:
|
|
70
|
+
relevant_motifs = list(load_motifs(motifs_path)) if motifs_path else []
|
|
71
|
+
if common_motifs:
|
|
72
|
+
relevant_motifs = relevant_motifs + load_common_motifs(common_motifs)
|
|
73
|
+
sites_collector['df_motifs'] = find_motif_sites(target_seq, num_sites, relevant_motifs)
|
|
74
|
+
|
|
75
|
+
return sites_collector
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _extract_cai(objectives_text_summary, num_codons):
|
|
79
|
+
"""Parse the CAI objective's score out of DNAChisel's summary text, or
|
|
80
|
+
return None if there wasn't one (e.g. `organism_name` wasn't recognized,
|
|
81
|
+
so no codon-optimization objective was ever added - DNAChisel then
|
|
82
|
+
summarizes as "===> No specifications", which has no ':' to parse and
|
|
83
|
+
used to crash this with an unhandled IndexError).
|
|
84
|
+
"""
|
|
85
|
+
first_line = objectives_text_summary.split('\n')[0]
|
|
86
|
+
if ':' not in first_line:
|
|
87
|
+
return None
|
|
88
|
+
cai_score = float(first_line.split(':')[1].strip())
|
|
89
|
+
return np.exp(cai_score / num_codons).round(4)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def backend(data, file, output_path, compute_motifs, num_sites, motifs_path,
|
|
93
|
+
optimize, mini_gc, maxi_gc, method, organism_name, indexes,
|
|
94
|
+
recombination_mode='thorough', slippage_mode='default', common_motifs=None,
|
|
95
|
+
custom_score_fn=None, custom_score_minimize=False):
|
|
96
|
+
"""Run the two-pass optimization (CAI/GC only, then + hotspot avoidance) over
|
|
97
|
+
every sequence record in `data`, and write out CSVs + a Word report to
|
|
98
|
+
`output_path/<file_stem>/`.
|
|
99
|
+
"""
|
|
100
|
+
recombination_collector = []
|
|
101
|
+
recombination_raw_collector = []
|
|
102
|
+
slippage_collector = []
|
|
103
|
+
slippage_raw_collector = []
|
|
104
|
+
motifs_collector = []
|
|
105
|
+
sequences_for_doc = []
|
|
106
|
+
|
|
107
|
+
filename_indexes = file_stem(file[0])
|
|
108
|
+
curr_output_path = path.join(output_path, filename_indexes)
|
|
109
|
+
Path(curr_output_path).mkdir(parents=True, exist_ok=True)
|
|
110
|
+
|
|
111
|
+
final_results = []
|
|
112
|
+
|
|
113
|
+
for ii, record in enumerate(data):
|
|
114
|
+
curr_seq = str(record.seq).upper()
|
|
115
|
+
original_seq = curr_seq
|
|
116
|
+
seq_indexes = str(ii)
|
|
117
|
+
|
|
118
|
+
if (filename_indexes, seq_indexes) not in indexes:
|
|
119
|
+
orf_regions = ()
|
|
120
|
+
exclusion_regions = ()
|
|
121
|
+
else:
|
|
122
|
+
relevant_index_data = indexes[(filename_indexes, seq_indexes)]
|
|
123
|
+
orf_regions = parse_region(relevant_index_data[0])
|
|
124
|
+
exclusion_regions = parse_region(relevant_index_data[1])
|
|
125
|
+
|
|
126
|
+
num_codons = sum((orf[1] - orf[0]) / 3 for orf in orf_regions) or len(curr_seq) // 3
|
|
127
|
+
|
|
128
|
+
maximal_cai = None
|
|
129
|
+
if optimize:
|
|
130
|
+
curr_seq, obj_description, _ = optimization_engine(
|
|
131
|
+
curr_seq, mini_gc=mini_gc, maxi_gc=maxi_gc, method=method, organism_name=organism_name,
|
|
132
|
+
custom_score_fn=custom_score_fn,
|
|
133
|
+
custom_score_minimize=custom_score_minimize,
|
|
134
|
+
orf_regions=orf_regions, exclusion_regions=exclusion_regions)
|
|
135
|
+
if custom_score_fn is None:
|
|
136
|
+
maximal_cai = _extract_cai(obj_description, num_codons)
|
|
137
|
+
|
|
138
|
+
curr_sites_collector = suspect_site_extractor(
|
|
139
|
+
curr_seq, compute_motifs, num_sites, motifs_path, common_motifs=common_motifs,
|
|
140
|
+
recombination_mode=recombination_mode, slippage_mode=slippage_mode)
|
|
141
|
+
|
|
142
|
+
# reporting (CSV) uses the collapsed, num_sites-limited view; the
|
|
143
|
+
# optimizer is given the raw, uncollapsed, unlimited view instead, so a
|
|
144
|
+
# genuinely distinct hotspot that only partially overlaps a
|
|
145
|
+
# higher-scoring one still gets its own correction constraint - see
|
|
146
|
+
# suspect_site_extractor's docstring and docs/detector-comparisons.md.
|
|
147
|
+
df_recombination = curr_sites_collector['df_recombination']
|
|
148
|
+
if len(df_recombination) > 0:
|
|
149
|
+
df_recombination.loc[:, 'sequence_number'] = str(ii)
|
|
150
|
+
recombination_collector.append(df_recombination)
|
|
151
|
+
df_recombination_raw = curr_sites_collector['df_recombination_raw']
|
|
152
|
+
if len(df_recombination_raw) > 0:
|
|
153
|
+
df_recombination_raw.loc[:, 'sequence_number'] = str(ii)
|
|
154
|
+
recombination_raw_collector.append(df_recombination_raw)
|
|
155
|
+
|
|
156
|
+
df_slippage = curr_sites_collector['df_slippage']
|
|
157
|
+
if len(df_slippage) > 0:
|
|
158
|
+
df_slippage.loc[:, 'sequence_number'] = str(ii)
|
|
159
|
+
slippage_collector.append(df_slippage)
|
|
160
|
+
df_slippage_raw = curr_sites_collector['df_slippage_raw']
|
|
161
|
+
if len(df_slippage_raw) > 0:
|
|
162
|
+
df_slippage_raw.loc[:, 'sequence_number'] = str(ii)
|
|
163
|
+
slippage_raw_collector.append(df_slippage_raw)
|
|
164
|
+
|
|
165
|
+
df_motifs = pd.DataFrame()
|
|
166
|
+
if compute_motifs:
|
|
167
|
+
df_motifs = curr_sites_collector['df_motifs']
|
|
168
|
+
if len(df_motifs) > 0:
|
|
169
|
+
df_motifs.loc[:, 'sequence_number'] = str(ii)
|
|
170
|
+
motifs_collector.append(df_motifs)
|
|
171
|
+
|
|
172
|
+
if optimize:
|
|
173
|
+
curr_seq, obj_description, num_edits = optimization_engine(
|
|
174
|
+
curr_seq, df_recombination=df_recombination_raw, df_slippage=df_slippage_raw, df_motifs=df_motifs,
|
|
175
|
+
mini_gc=mini_gc, maxi_gc=maxi_gc, method=method, organism_name=organism_name,
|
|
176
|
+
custom_score_fn=custom_score_fn,
|
|
177
|
+
custom_score_minimize=custom_score_minimize,
|
|
178
|
+
orf_regions=orf_regions, exclusion_regions=exclusion_regions)
|
|
179
|
+
|
|
180
|
+
with open(path.join(curr_output_path, 'final_sequence.txt'), "w", encoding="utf-8") as text_file:
|
|
181
|
+
if custom_score_fn is None and maximal_cai is not None:
|
|
182
|
+
cai_constrained = _extract_cai(obj_description, num_codons)
|
|
183
|
+
text_file.write('The maximal CAI of gene (with no constraints) objective:\n')
|
|
184
|
+
text_file.write(f'{maximal_cai}\n')
|
|
185
|
+
text_file.write('The CAI of gene (after constraints) objective:\n')
|
|
186
|
+
text_file.write(f'{cai_constrained}\n')
|
|
187
|
+
elif custom_score_fn is None:
|
|
188
|
+
text_file.write(
|
|
189
|
+
"No codon-usage objective was applied (organism_name wasn't recognized).\n")
|
|
190
|
+
else:
|
|
191
|
+
text_file.write('Optimized using a custom score function instead of CAI/tAI.\n')
|
|
192
|
+
text_file.write('The number of codons edited due to hypermutable site constraints:\n')
|
|
193
|
+
text_file.write(f'{num_edits}\n')
|
|
194
|
+
text_file.write('The final sequence is:\n')
|
|
195
|
+
for line_start in range(0, len(curr_seq), 70):
|
|
196
|
+
text_file.write(curr_seq[line_start:line_start + 70] + '\n')
|
|
197
|
+
|
|
198
|
+
sequences_for_doc.append((f"{filename_indexes}_{ii}", original_seq, curr_seq))
|
|
199
|
+
final_results.append((ii, curr_seq))
|
|
200
|
+
|
|
201
|
+
if recombination_collector:
|
|
202
|
+
pd.concat(recombination_collector, ignore_index=True).to_csv(
|
|
203
|
+
path.join(curr_output_path, 'recombination_sites.csv'), index=False)
|
|
204
|
+
|
|
205
|
+
if optimize and recombination_raw_collector:
|
|
206
|
+
# every candidate actually given a correction constraint during
|
|
207
|
+
# optimization, not just the collapsed "one representative per
|
|
208
|
+
# distinct site" view above - since those two views can now
|
|
209
|
+
# legitimately differ (see suspect_site_extractor's docstring and
|
|
210
|
+
# docs/detector-comparisons.md), this is what to check against
|
|
211
|
+
# final_sequence.txt if a diff shows an edit with no corresponding
|
|
212
|
+
# row in recombination_sites.csv.
|
|
213
|
+
pd.concat(recombination_raw_collector, ignore_index=True).to_csv(
|
|
214
|
+
path.join(curr_output_path, 'recombination_sites_corrected.csv'), index=False)
|
|
215
|
+
|
|
216
|
+
if slippage_collector:
|
|
217
|
+
pd.concat(slippage_collector, ignore_index=True).to_csv(
|
|
218
|
+
path.join(curr_output_path, 'slippage_sites.csv'), index=False)
|
|
219
|
+
|
|
220
|
+
if optimize and slippage_raw_collector:
|
|
221
|
+
# see the recombination_sites_corrected.csv comment above.
|
|
222
|
+
pd.concat(slippage_raw_collector, ignore_index=True).to_csv(
|
|
223
|
+
path.join(curr_output_path, 'slippage_sites_corrected.csv'), index=False)
|
|
224
|
+
|
|
225
|
+
if compute_motifs and motifs_collector:
|
|
226
|
+
pd.concat(motifs_collector, ignore_index=True).to_csv(
|
|
227
|
+
path.join(curr_output_path, 'motif_sites.csv'), index=False)
|
|
228
|
+
|
|
229
|
+
if sequences_for_doc:
|
|
230
|
+
create_word_document_with_highlighted_differences(sequences_for_doc, curr_output_path)
|
|
231
|
+
|
|
232
|
+
return final_results
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def main(input_folder=None, output_path=None, compute_motifs=False, num_sites=np.inf,
|
|
236
|
+
motifs_path=None, common_motifs=None, optimize=True, mini_gc=0.3, maxi_gc=0.7,
|
|
237
|
+
method='use_best_codon', organism_name='not_specified', indexes=None,
|
|
238
|
+
recombination_mode='thorough', slippage_mode='default', custom_score_fn=None,
|
|
239
|
+
custom_score_minimize=False):
|
|
240
|
+
"""Optimize every FASTA/GenBank file in `input_folder`, writing per-file CSVs
|
|
241
|
+
of detected hotspots and the optimized sequence into `output_path`.
|
|
242
|
+
|
|
243
|
+
Parameters
|
|
244
|
+
----------
|
|
245
|
+
input_folder: str
|
|
246
|
+
Directory to scan for .fasta/.fna/.ffn/.faa/.frn/.fa/.gb/.gbk/.genbank
|
|
247
|
+
files (optionally gzipped), directly inside or one level under it.
|
|
248
|
+
output_path: str
|
|
249
|
+
Directory to write results into (one subdirectory per input file).
|
|
250
|
+
compute_motifs: bool
|
|
251
|
+
Whether to also detect methylation motif sites (needs `motifs_path` and/or
|
|
252
|
+
`common_motifs`).
|
|
253
|
+
num_sites: int or float('inf')
|
|
254
|
+
Max number of hotspots to report (in the CSVs/collapsed dataframes)
|
|
255
|
+
per category. Default: all. Does NOT limit how many correction
|
|
256
|
+
constraints are built during optimization - every detected candidate
|
|
257
|
+
is always given a constraint, regardless of `num_sites`, since
|
|
258
|
+
limiting that too could silently leave part of a genuine hotspot
|
|
259
|
+
unconstrained (see eso.pipeline.suspect_site_extractor).
|
|
260
|
+
motifs_path: str or None
|
|
261
|
+
Path to a MEME-minimal-format PSSM file.
|
|
262
|
+
common_motifs: list of str or None
|
|
263
|
+
Names from eso.detection.common_motifs.COMMON_MOTIFS (currently "dam",
|
|
264
|
+
"dcm" - E. coli's methylation systems) to include alongside any
|
|
265
|
+
`motifs_path` file, with no file needed at all. At least one of
|
|
266
|
+
`motifs_path`/`common_motifs` is required if compute_motifs=True.
|
|
267
|
+
optimize: bool
|
|
268
|
+
Whether to codon/GC-optimize and avoid hotspots, vs. just detect them.
|
|
269
|
+
mini_gc, maxi_gc: float in [0, 1]
|
|
270
|
+
Allowed GC-content range within any 50nt window.
|
|
271
|
+
method: {"use_best_codon", "match_codon_usage", "harmonize_rca"}
|
|
272
|
+
Codon optimization strategy.
|
|
273
|
+
organism_name: str
|
|
274
|
+
Host organism for codon optimization (see eso.optimize._codon_optimization_objectives).
|
|
275
|
+
indexes: dict
|
|
276
|
+
Maps (file_stem, seq_index_str) -> (orf_region_string, exclusion_region_string),
|
|
277
|
+
1-indexed and inclusive, e.g. {("my_gene", "0"): ("1-6, 51-68", "1-6, 50-68")}.
|
|
278
|
+
Omit or pass {} to treat entire sequences as the ORF with no exclusions.
|
|
279
|
+
recombination_mode: {"thorough", "fast"}
|
|
280
|
+
See eso.detection.dispatch.find_recombination_sites. "thorough" (default)
|
|
281
|
+
catches near-duplicate hotspots and stays roughly linear (confirmed
|
|
282
|
+
practical up to 1,000,000nt - see docs/detector-comparisons.md);
|
|
283
|
+
"fast" is exact-match only, 19-34x faster, for workloads where that
|
|
284
|
+
speed gap actually matters.
|
|
285
|
+
slippage_mode: {"default", "fast"}
|
|
286
|
+
See eso.detection.dispatch.find_slippage_sites. Both detect identical
|
|
287
|
+
hotspots; "default" is also faster at every length tested.
|
|
288
|
+
custom_score_fn: callable(str) -> float, or None
|
|
289
|
+
If given, replaces CAI/tAI (organism_name/method) with this scoring
|
|
290
|
+
function - see eso.custom_score.CustomScore. Most users should use
|
|
291
|
+
the `--custom-score-file` CLI flag / eso.custom_score.load_custom_score_from_file
|
|
292
|
+
instead of passing a function directly here.
|
|
293
|
+
custom_score_minimize: bool
|
|
294
|
+
See eso.optimize.optimization_engine; only used if custom_score_fn is given.
|
|
295
|
+
|
|
296
|
+
Returns
|
|
297
|
+
-------
|
|
298
|
+
(message, results) where message is 'Success!' or a validation error, and
|
|
299
|
+
results is a list of (file, seq_index, optimized_sequence) tuples.
|
|
300
|
+
"""
|
|
301
|
+
indexes = indexes or {}
|
|
302
|
+
if output_path is None:
|
|
303
|
+
output_path = path.join(input_folder or '.', 'output')
|
|
304
|
+
|
|
305
|
+
files = relevant_file_paths(input_folder=input_folder)
|
|
306
|
+
message = test_input(mini_gc, maxi_gc, indexes, files)
|
|
307
|
+
|
|
308
|
+
if message != 'Success!':
|
|
309
|
+
return message, []
|
|
310
|
+
|
|
311
|
+
final_results = []
|
|
312
|
+
for file in files:
|
|
313
|
+
data = file_opener(file)
|
|
314
|
+
curr_results = backend(
|
|
315
|
+
data, file, output_path, compute_motifs, num_sites, motifs_path,
|
|
316
|
+
optimize=optimize, mini_gc=mini_gc, maxi_gc=maxi_gc, method=method,
|
|
317
|
+
organism_name=organism_name, indexes=indexes, recombination_mode=recombination_mode,
|
|
318
|
+
slippage_mode=slippage_mode, common_motifs=common_motifs, custom_score_fn=custom_score_fn,
|
|
319
|
+
custom_score_minimize=custom_score_minimize)
|
|
320
|
+
final_results.extend((file, seq_index, seq) for seq_index, seq in curr_results)
|
|
321
|
+
|
|
322
|
+
return message, final_results
|
eso/report.py
ADDED
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"""Optional Word-document report showing original vs. optimized sequences
|
|
2
|
+
with per-nucleotide differences highlighted.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
from os import path
|
|
6
|
+
|
|
7
|
+
try:
|
|
8
|
+
from docx import Document
|
|
9
|
+
from docx.enum.text import WD_COLOR_INDEX
|
|
10
|
+
DOCX_AVAILABLE = True
|
|
11
|
+
except ImportError:
|
|
12
|
+
DOCX_AVAILABLE = False
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def create_word_document_with_highlighted_differences(sequences_data, output_path):
|
|
16
|
+
"""
|
|
17
|
+
Parameters
|
|
18
|
+
----------
|
|
19
|
+
sequences_data: list of (sequence_name, original_seq, final_seq) tuples.
|
|
20
|
+
output_path: directory in which to save 'sequence_comparison.docx'.
|
|
21
|
+
"""
|
|
22
|
+
if not DOCX_AVAILABLE:
|
|
23
|
+
print("python-docx not available (install the 'docx-report' extra), skipping Word document generation")
|
|
24
|
+
return
|
|
25
|
+
|
|
26
|
+
doc = Document()
|
|
27
|
+
doc.add_heading('Sequence Optimization Results', 0)
|
|
28
|
+
|
|
29
|
+
for index, (seq_name, original_seq, final_seq) in enumerate(sequences_data):
|
|
30
|
+
doc.add_heading(f'Sequence: {seq_name}', level=1)
|
|
31
|
+
|
|
32
|
+
doc.add_heading('Original Sequence:', level=2)
|
|
33
|
+
original_paragraph = doc.add_paragraph()
|
|
34
|
+
|
|
35
|
+
doc.add_heading('Final Sequence:', level=2)
|
|
36
|
+
final_paragraph = doc.add_paragraph()
|
|
37
|
+
|
|
38
|
+
for i, char in enumerate(original_seq):
|
|
39
|
+
run = original_paragraph.add_run(char)
|
|
40
|
+
if i < len(final_seq) and char != final_seq[i]:
|
|
41
|
+
run.font.highlight_color = WD_COLOR_INDEX.YELLOW
|
|
42
|
+
|
|
43
|
+
for i, char in enumerate(final_seq):
|
|
44
|
+
run = final_paragraph.add_run(char)
|
|
45
|
+
if i < len(original_seq) and char != original_seq[i]:
|
|
46
|
+
run.font.highlight_color = WD_COLOR_INDEX.YELLOW
|
|
47
|
+
|
|
48
|
+
# was `if seq_name != sequences_data[-1][0]` - compared by name, so a
|
|
49
|
+
# sequence whose name happened to match the true last entry's name
|
|
50
|
+
# (e.g. two files/records that coincidentally share a stem) would
|
|
51
|
+
# wrongly skip its own page break. Compare by position instead.
|
|
52
|
+
if index != len(sequences_data) - 1:
|
|
53
|
+
doc.add_page_break()
|
|
54
|
+
|
|
55
|
+
doc_path = path.join(output_path, 'sequence_comparison.docx')
|
|
56
|
+
doc.save(doc_path)
|
|
57
|
+
print(f"Word document saved to: {doc_path}")
|