evolutionary-stability-optimizer 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
eso/optimize.py ADDED
@@ -0,0 +1,266 @@
1
+ """DNAChisel-based sequence optimization: enforce translation, GC-content
2
+ windows, codon usage, and avoidance of detected hypermutable sites.
3
+ """
4
+
5
+ import warnings
6
+
7
+ import pandas as pd
8
+ import dnachisel
9
+ from dnachisel.DnaOptimizationProblem import NoSolutionError
10
+
11
+ from eso.codon_usage import CODON_USAGE_TABLES
12
+ from eso.constraints import (
13
+ convert_df_to_constraints,
14
+ exclusion_site_correcter,
15
+ recombination_to_multiple_avoidance_sites,
16
+ )
17
+ from eso.custom_score import CustomScore
18
+ from eso.detection.slippage import modify_df_slippage
19
+
20
+ # Substring DNAChisel raises when an AvoidPattern's location doesn't align to
21
+ # the codon grid under EnforceTranslation: DnaOptimizationProblem.resolve_constraint
22
+ # can compute a "jointly mutable" mutation-space region (accounting for codon
23
+ # interdependencies) that ends up not overlapping the constraint's own declared
24
+ # location at all. AvoidPattern.localized() correctly returns None for that
25
+ # (they genuinely don't overlap) - but the DNAChisel caller doesn't check for
26
+ # None before calling .evaluate() on it. Confirmed via direct reproduction with
27
+ # a non-codon-aligned 2nt repeat and, separately, a 1nt homopolymer run; a
28
+ # codon-aligned repeat (e.g. a 3nt unit) never triggers it.
29
+ #
30
+ # An earlier fix widened every avoid-window outward to whole codons before
31
+ # building the AvoidPattern, which does dodge this crash - but it also changes
32
+ # what's being asked for: "avoid this pattern at this spot" becomes "avoid this
33
+ # pattern ANYWHERE in the whole codon", which for a single-nucleotide pattern
34
+ # in a homopolymer run can make the constraint unsatisfiable for every codon in
35
+ # the run (e.g. all-Phe TTT/TTC codons always contain a T), silently dropping
36
+ # every constraint and leaving the site untouched - no crash, but no effect
37
+ # either (confirmed directly: a "ATG"+"T"*12+"TAA" homopolymer produced
38
+ # num_edits=0 with that approach). So instead of forcing this by widening the
39
+ # window, the retry loop below simply catches the crash like any other
40
+ # unsatisfiable-constraint case and drops whichever constraints are actually
41
+ # still failing, exactly as it already did for DNAChisel's own
42
+ # NoSolutionError-with-no-named-constraint case.
43
+ _LOCALIZED_NONE_CRASH_MESSAGE = "'NoneType' object has no attribute 'evaluate'"
44
+
45
+
46
+ def _warn_dropped_constraint(constraint):
47
+ warnings.warn(
48
+ f"Could not satisfy constraint {constraint} - dropping it and continuing without it. "
49
+ "This most often happens when disrupting a detected hypermutable site would require a "
50
+ "change that conflicts with another hard constraint, most commonly translation "
51
+ "preservation (e.g. every synonymous codon at that position still contains the pattern "
52
+ "being avoided, such as a homopolymer landing on a Met/Trp codon with no alternative). "
53
+ "The site this constraint was protecting was left unmodified in the final sequence.",
54
+ stacklevel=3,
55
+ )
56
+
57
+
58
+ def _codon_optimization_objectives(organism_name, orf_regions, method):
59
+ """Build DNAChisel CodonOptimize objectives.
60
+
61
+ `organism_name` may be one of eso.codon_usage.CODON_USAGE_TABLES' custom
62
+ tables (not present in the Kazusa database), or any species/TaxID
63
+ supported by python-codon-tables.
64
+ """
65
+ if organism_name in CODON_USAGE_TABLES:
66
+ codon_usage_table = CODON_USAGE_TABLES[organism_name]()
67
+ return [
68
+ dnachisel.CodonOptimize(location=orf, codon_usage_table=codon_usage_table.copy(), method=method)
69
+ for orf in orf_regions
70
+ ]
71
+
72
+ try:
73
+ return [dnachisel.CodonOptimize(species=organism_name, location=orf, method=method) for orf in orf_regions]
74
+ except Exception:
75
+ print('unknown organism, no codon optimization!')
76
+ return []
77
+
78
+
79
+ def optimization_engine(
80
+ seq,
81
+ mini_gc=0.3,
82
+ maxi_gc=0.7,
83
+ window_size_gc=50,
84
+ method='use_best_codon',
85
+ organism_name='not_specified',
86
+ custom_score_fn=None,
87
+ custom_score_minimize=False,
88
+ df_recombination=None,
89
+ df_slippage=None,
90
+ df_motifs=None,
91
+ orf_regions=(),
92
+ exclusion_regions=(),
93
+ ):
94
+ """Optimize `seq` for codon usage, GC content, and (if hotspot dataframes
95
+ are given) avoidance of detected recombination/slippage/methylation sites,
96
+ while preserving the amino-acid translation over `orf_regions`.
97
+
98
+ Citation: Jack, Leonard, Mishler, Renda, Leon, Suarez & Barrick (2015).
99
+ "Predicting the genetic stability of engineered DNA sequences with the
100
+ EFM Calculator." ACS Synthetic Biology. DOI: 10.1021/acssynbio.5b00068
101
+
102
+ Parameters
103
+ ----------
104
+ seq: str
105
+ DNA sequence (ACGT alphabet).
106
+ mini_gc, maxi_gc: float in [0, 1]
107
+ Allowed GC-content range within any `window_size_gc`-nt window.
108
+ method: {"use_best_codon", "match_codon_usage", "harmonize_rca"}
109
+ Codon optimization strategy (see DNAChisel's CodonOptimize).
110
+ organism_name: str
111
+ Host organism for codon optimization: one of
112
+ eso.codon_usage.CODON_USAGE_TABLES' keys, a python-codon-tables
113
+ species name/TaxID, or "not_specified" to skip codon optimization.
114
+ custom_score_fn: callable(str) -> float, or None
115
+ If given, replaces the CodonOptimize (CAI/tAI-style) objective with
116
+ eso.custom_score.CustomScore wrapping this function (higher is
117
+ better sequence, unless custom_score_minimize=True). `organism_name`
118
+ and `method` are then ignored. Called once per ORF, on the whole ORF,
119
+ for every trial mutation during optimization - can be slow on long
120
+ sequences or an expensive custom_score_fn (a warning is raised).
121
+ custom_score_minimize: bool
122
+ If True, treats a lower custom_score_fn value as better.
123
+ df_recombination, df_slippage, df_motifs: pandas.DataFrame or None
124
+ Detected hotspots to avoid (from eso.detection.*); pass None or an
125
+ empty dataframe to skip a given constraint type.
126
+ orf_regions: sequence of (start, end) tuples
127
+ Regions to keep in-frame and translation-preserving. Defaults to the
128
+ whole sequence, trimmed to a multiple of 3.
129
+ exclusion_regions: sequence of (start, end) tuples
130
+ Regions that must not be modified.
131
+
132
+ Returns
133
+ -------
134
+ (final_sequence, objectives_summary, num_edits)
135
+ """
136
+ if df_recombination is None:
137
+ df_recombination = pd.DataFrame()
138
+ if df_slippage is None:
139
+ df_slippage = pd.DataFrame()
140
+ if df_motifs is None:
141
+ df_motifs = pd.DataFrame()
142
+
143
+ if len(orf_regions) == 0:
144
+ new_last_index = ((len(seq) - 1) // 3) * 3
145
+ orf_regions = [(0, new_last_index)]
146
+
147
+ if custom_score_fn is not None:
148
+ # Scoped to each ORF individually, matching how _codon_optimization_objectives
149
+ # already scopes CodonOptimize to `location=orf` - without this, custom scoring
150
+ # would apply across the whole sequence including any non-ORF flanks (UTRs,
151
+ # locked regions), inconsistent with codon-usage scoring's own ORF-only behavior.
152
+ obj = [
153
+ CustomScore(custom_score_fn, location=orf, minimize=custom_score_minimize)
154
+ for orf in orf_regions
155
+ ]
156
+ else:
157
+ obj = _codon_optimization_objectives(organism_name, orf_regions, method)
158
+
159
+ cnst = [dnachisel.EnforceGCContent(mini=mini_gc, maxi=maxi_gc, window=window_size_gc)]
160
+ for orf in orf_regions:
161
+ cnst.append(dnachisel.EnforceTranslation(location=orf))
162
+
163
+ cnst.extend(dnachisel.AvoidChanges(location=region) for region in exclusion_regions)
164
+
165
+ if not df_recombination.empty:
166
+ df_rec = recombination_to_multiple_avoidance_sites(df_recombination, exclusion_regions)
167
+ df_rec = exclusion_site_correcter(df_rec, exclusion_regions)
168
+ cnst.extend(convert_df_to_constraints(df_rec))
169
+
170
+ if not df_slippage.empty:
171
+ df_slip = modify_df_slippage(df_slippage)
172
+ df_slip = exclusion_site_correcter(df_slip.copy(), exclusion_regions)
173
+ cnst.extend(convert_df_to_constraints(df_slip))
174
+
175
+ if not df_motifs.empty:
176
+ df_mot = df_motifs.copy()[['start_index', 'end_index', 'actual_site']].rename(
177
+ columns={'start_index': 'start', 'end_index': 'end', 'actual_site': 'sequence'})
178
+ df_mot.loc[:, 'start'] = df_mot['start'].astype(int)
179
+ # eso.detection.methylation's end_index is INCLUSIVE (the index of the
180
+ # motif's last nucleotide), but everything downstream of here -
181
+ # exclusion_site_correcter, convert_df_to_constraints, and DNAChisel's
182
+ # own Location - uses EXCLUSIVE end (matching Python slicing). Without
183
+ # the +1, every motif's AvoidPattern location was exactly one
184
+ # nucleotide too short to ever contain its own pattern, so
185
+ # DNAChisel could never find it there and always reported the
186
+ # constraint as trivially satisfied - methylation-motif avoidance
187
+ # silently did nothing. Confirmed directly: before this fix, a GATC
188
+ # motif passed as df_motifs survived optimization completely
189
+ # untouched (0 edits); see tests/test_optimize.py.
190
+ df_mot.loc[:, 'end'] = df_mot['end'].astype(int) + 1
191
+ df_mot = exclusion_site_correcter(df_mot, exclusion_regions)
192
+ cnst.extend(convert_df_to_constraints(df_mot))
193
+
194
+ problem = None
195
+ flag = 0
196
+ # 60 was chosen empirically back when constraint counts were always small
197
+ # (num_sites capped how many hotspot-avoidance constraints existed at
198
+ # all). Since num_sites became report-only and constraint-building
199
+ # started using every raw, uncollapsed candidate (see
200
+ # docs/detector-comparisons.md's overlap-collapse coverage-gap entry), a
201
+ # single dense/repetitive sequence can legitimately produce well over 60
202
+ # individual AvoidPattern constraints - a fixed 60-round budget could then
203
+ # raise NoSolutionError purely from running out of retries, not from any
204
+ # actual unresolvable conflict. Scale the budget with how many
205
+ # constraints there actually are (computed once, before any get dropped,
206
+ # so it doesn't shrink alongside `cnst` as rounds progress), keeping 60 as
207
+ # the floor for the common, small-constraint-count case this was
208
+ # originally tuned for.
209
+ retry_budget = max(60, len(cnst))
210
+ while flag < retry_budget: # retry, dropping unsatisfiable/crashing constraints
211
+ problem = dnachisel.DnaOptimizationProblem(sequence=str(seq), constraints=cnst, objectives=obj)
212
+ try:
213
+ problem.resolve_constraints()
214
+ break
215
+ except NoSolutionError as e:
216
+ if e.constraint is not None:
217
+ _warn_dropped_constraint(e.constraint)
218
+ cnst.remove(e.constraint)
219
+ flag += 1
220
+ continue
221
+ drop_and_retry = True
222
+ except AttributeError as e:
223
+ # See _LOCALIZED_NONE_CRASH_MESSAGE above: a genuine DNAChisel-internal
224
+ # crash on non-codon-aligned AvoidPattern locations. Only swallow this
225
+ # exact crash - anything else with this type is a real bug, re-raise it.
226
+ if _LOCALIZED_NONE_CRASH_MESSAGE not in str(e):
227
+ raise
228
+ drop_and_retry = True
229
+
230
+ if drop_and_retry:
231
+ # Either DNAChisel's own final consistency check
232
+ # (perform_final_constraints_check, run at the end of
233
+ # resolve_constraints) failed without identifying a single culprit
234
+ # constraint - NoSolutionError.constraint defaults to None, seen in
235
+ # practice with several individually-resolvable but densely packed
236
+ # AvoidPattern constraints that regress each other by the time
237
+ # solving reaches the last one - or resolve_constraint itself
238
+ # crashed on a non-codon-aligned AvoidPattern location. Either way,
239
+ # `cnst.remove(None)` isn't possible here, so instead re-evaluate
240
+ # every constraint ourselves and drop whichever ones are still
241
+ # actually failing, so the retry loop can keep making progress the
242
+ # same way it does for the normal case. Constraints like
243
+ # EnforceGCContent have location=None until
244
+ # initialized_on_problem() fills it in (as a copy, not in-place) -
245
+ # evaluate that initialized copy, not the raw constraint, which
246
+ # would crash the same way on its own unset location.
247
+ failing = [
248
+ c for c in cnst
249
+ if not c.initialized_on_problem(problem, role='constraint').evaluate(problem).passes
250
+ ]
251
+ if not failing:
252
+ raise
253
+ for constraint in failing:
254
+ _warn_dropped_constraint(constraint)
255
+ cnst.remove(constraint)
256
+ flag += 1
257
+ else:
258
+ raise NoSolutionError(
259
+ f"More than {retry_budget} hard constraints were not satisfied ({flag}).", problem=problem)
260
+
261
+ problem.optimize()
262
+ obj_description = problem.objectives_text_summary()
263
+ num_edits = problem.number_of_edits()
264
+ final_sequence = str(problem.sequence).upper()
265
+
266
+ return final_sequence, obj_description, num_edits
eso/pipeline.py ADDED
@@ -0,0 +1,322 @@
1
+ """End-to-end orchestration: for each sequence file, run codon/GC optimization,
2
+ detect hypermutable sites, re-optimize while avoiding them, and write out
3
+ per-file CSVs plus a Word comparison report.
4
+ """
5
+
6
+ from pathlib import Path
7
+ from os import path
8
+
9
+ import numpy as np
10
+ import pandas as pd
11
+
12
+ from eso.detection.common_motifs import load_common_motifs
13
+ from eso.detection.dispatch import (
14
+ collapse_recombination_sites,
15
+ collapse_slippage_sites,
16
+ find_recombination_candidates,
17
+ find_slippage_candidates,
18
+ recombination_sites_for_constraints,
19
+ slippage_sites_for_constraints,
20
+ )
21
+ from eso.detection.methylation import load_motifs, find_motif_sites
22
+ from eso.io_utils import file_opener, file_stem, relevant_file_paths, test_input
23
+ from eso.optimize import optimization_engine
24
+ from eso.report import create_word_document_with_highlighted_differences
25
+ from eso.sequence_utils import parse_region
26
+
27
+
28
+ def suspect_site_extractor(target_seq, compute_motifs, num_sites, motifs_path=None,
29
+ common_motifs=None, recombination_mode='thorough',
30
+ slippage_mode='default'):
31
+ """Detect recombination and slippage sites (and, if `compute_motifs`, methylation
32
+ motif sites) in `target_seq`. Returns a dict of dataframes keyed by
33
+ 'df_recombination', 'df_slippage', and optionally 'df_motifs' - each
34
+ collapsed to one representative per distinct site and limited to
35
+ `num_sites`, for reporting - plus 'df_recombination_raw'/'df_slippage_raw',
36
+ a reduced-but-not-collapsed view (see
37
+ eso.detection.slippage.slippage_sites_for_constraints) meant for building
38
+ correction constraints from instead.
39
+
40
+ Detection itself only runs once per category; the report and
41
+ constraint-building views are both derived from that single run. Use the
42
+ _raw dataframes (not the collapsed ones) to build correction constraints:
43
+ collapsing (as the report view does) can silently drop a genuinely
44
+ distinct, only-partially-overlapping hotspot entirely, leaving the region
45
+ it uniquely covered with no fix. `num_sites` is report-only for the same
46
+ reason - limiting it would reintroduce that exact silent-coverage-loss
47
+ risk for whichever candidates it cuts.
48
+
49
+ recombination_mode: see eso.detection.dispatch.find_recombination_sites -
50
+ "thorough" (default, Levenshtein-tolerant) or "fast" (exact-match only).
51
+ slippage_mode: see eso.detection.dispatch.find_slippage_sites -
52
+ "default" or "fast" (equivalent sensitivity; "default" is also faster).
53
+ common_motifs: list of str or None
54
+ Names from eso.detection.common_motifs.COMMON_MOTIFS (currently
55
+ "dam", "dcm") to include alongside any `motifs_path` file. At least
56
+ one of `motifs_path`/`common_motifs` is required if `compute_motifs`.
57
+ """
58
+ df_recombination_candidates = find_recombination_candidates(target_seq, mode=recombination_mode)
59
+ df_slippage_candidates = find_slippage_candidates(target_seq, mode=slippage_mode)
60
+
61
+ sites_collector = {
62
+ 'df_recombination': collapse_recombination_sites(
63
+ df_recombination_candidates, num_sites, mode=recombination_mode),
64
+ 'df_slippage': collapse_slippage_sites(df_slippage_candidates, num_sites, mode=slippage_mode),
65
+ 'df_recombination_raw': recombination_sites_for_constraints(df_recombination_candidates, mode=recombination_mode),
66
+ 'df_slippage_raw': slippage_sites_for_constraints(df_slippage_candidates, mode=slippage_mode),
67
+ }
68
+
69
+ if compute_motifs:
70
+ relevant_motifs = list(load_motifs(motifs_path)) if motifs_path else []
71
+ if common_motifs:
72
+ relevant_motifs = relevant_motifs + load_common_motifs(common_motifs)
73
+ sites_collector['df_motifs'] = find_motif_sites(target_seq, num_sites, relevant_motifs)
74
+
75
+ return sites_collector
76
+
77
+
78
+ def _extract_cai(objectives_text_summary, num_codons):
79
+ """Parse the CAI objective's score out of DNAChisel's summary text, or
80
+ return None if there wasn't one (e.g. `organism_name` wasn't recognized,
81
+ so no codon-optimization objective was ever added - DNAChisel then
82
+ summarizes as "===> No specifications", which has no ':' to parse and
83
+ used to crash this with an unhandled IndexError).
84
+ """
85
+ first_line = objectives_text_summary.split('\n')[0]
86
+ if ':' not in first_line:
87
+ return None
88
+ cai_score = float(first_line.split(':')[1].strip())
89
+ return np.exp(cai_score / num_codons).round(4)
90
+
91
+
92
+ def backend(data, file, output_path, compute_motifs, num_sites, motifs_path,
93
+ optimize, mini_gc, maxi_gc, method, organism_name, indexes,
94
+ recombination_mode='thorough', slippage_mode='default', common_motifs=None,
95
+ custom_score_fn=None, custom_score_minimize=False):
96
+ """Run the two-pass optimization (CAI/GC only, then + hotspot avoidance) over
97
+ every sequence record in `data`, and write out CSVs + a Word report to
98
+ `output_path/<file_stem>/`.
99
+ """
100
+ recombination_collector = []
101
+ recombination_raw_collector = []
102
+ slippage_collector = []
103
+ slippage_raw_collector = []
104
+ motifs_collector = []
105
+ sequences_for_doc = []
106
+
107
+ filename_indexes = file_stem(file[0])
108
+ curr_output_path = path.join(output_path, filename_indexes)
109
+ Path(curr_output_path).mkdir(parents=True, exist_ok=True)
110
+
111
+ final_results = []
112
+
113
+ for ii, record in enumerate(data):
114
+ curr_seq = str(record.seq).upper()
115
+ original_seq = curr_seq
116
+ seq_indexes = str(ii)
117
+
118
+ if (filename_indexes, seq_indexes) not in indexes:
119
+ orf_regions = ()
120
+ exclusion_regions = ()
121
+ else:
122
+ relevant_index_data = indexes[(filename_indexes, seq_indexes)]
123
+ orf_regions = parse_region(relevant_index_data[0])
124
+ exclusion_regions = parse_region(relevant_index_data[1])
125
+
126
+ num_codons = sum((orf[1] - orf[0]) / 3 for orf in orf_regions) or len(curr_seq) // 3
127
+
128
+ maximal_cai = None
129
+ if optimize:
130
+ curr_seq, obj_description, _ = optimization_engine(
131
+ curr_seq, mini_gc=mini_gc, maxi_gc=maxi_gc, method=method, organism_name=organism_name,
132
+ custom_score_fn=custom_score_fn,
133
+ custom_score_minimize=custom_score_minimize,
134
+ orf_regions=orf_regions, exclusion_regions=exclusion_regions)
135
+ if custom_score_fn is None:
136
+ maximal_cai = _extract_cai(obj_description, num_codons)
137
+
138
+ curr_sites_collector = suspect_site_extractor(
139
+ curr_seq, compute_motifs, num_sites, motifs_path, common_motifs=common_motifs,
140
+ recombination_mode=recombination_mode, slippage_mode=slippage_mode)
141
+
142
+ # reporting (CSV) uses the collapsed, num_sites-limited view; the
143
+ # optimizer is given the raw, uncollapsed, unlimited view instead, so a
144
+ # genuinely distinct hotspot that only partially overlaps a
145
+ # higher-scoring one still gets its own correction constraint - see
146
+ # suspect_site_extractor's docstring and docs/detector-comparisons.md.
147
+ df_recombination = curr_sites_collector['df_recombination']
148
+ if len(df_recombination) > 0:
149
+ df_recombination.loc[:, 'sequence_number'] = str(ii)
150
+ recombination_collector.append(df_recombination)
151
+ df_recombination_raw = curr_sites_collector['df_recombination_raw']
152
+ if len(df_recombination_raw) > 0:
153
+ df_recombination_raw.loc[:, 'sequence_number'] = str(ii)
154
+ recombination_raw_collector.append(df_recombination_raw)
155
+
156
+ df_slippage = curr_sites_collector['df_slippage']
157
+ if len(df_slippage) > 0:
158
+ df_slippage.loc[:, 'sequence_number'] = str(ii)
159
+ slippage_collector.append(df_slippage)
160
+ df_slippage_raw = curr_sites_collector['df_slippage_raw']
161
+ if len(df_slippage_raw) > 0:
162
+ df_slippage_raw.loc[:, 'sequence_number'] = str(ii)
163
+ slippage_raw_collector.append(df_slippage_raw)
164
+
165
+ df_motifs = pd.DataFrame()
166
+ if compute_motifs:
167
+ df_motifs = curr_sites_collector['df_motifs']
168
+ if len(df_motifs) > 0:
169
+ df_motifs.loc[:, 'sequence_number'] = str(ii)
170
+ motifs_collector.append(df_motifs)
171
+
172
+ if optimize:
173
+ curr_seq, obj_description, num_edits = optimization_engine(
174
+ curr_seq, df_recombination=df_recombination_raw, df_slippage=df_slippage_raw, df_motifs=df_motifs,
175
+ mini_gc=mini_gc, maxi_gc=maxi_gc, method=method, organism_name=organism_name,
176
+ custom_score_fn=custom_score_fn,
177
+ custom_score_minimize=custom_score_minimize,
178
+ orf_regions=orf_regions, exclusion_regions=exclusion_regions)
179
+
180
+ with open(path.join(curr_output_path, 'final_sequence.txt'), "w", encoding="utf-8") as text_file:
181
+ if custom_score_fn is None and maximal_cai is not None:
182
+ cai_constrained = _extract_cai(obj_description, num_codons)
183
+ text_file.write('The maximal CAI of gene (with no constraints) objective:\n')
184
+ text_file.write(f'{maximal_cai}\n')
185
+ text_file.write('The CAI of gene (after constraints) objective:\n')
186
+ text_file.write(f'{cai_constrained}\n')
187
+ elif custom_score_fn is None:
188
+ text_file.write(
189
+ "No codon-usage objective was applied (organism_name wasn't recognized).\n")
190
+ else:
191
+ text_file.write('Optimized using a custom score function instead of CAI/tAI.\n')
192
+ text_file.write('The number of codons edited due to hypermutable site constraints:\n')
193
+ text_file.write(f'{num_edits}\n')
194
+ text_file.write('The final sequence is:\n')
195
+ for line_start in range(0, len(curr_seq), 70):
196
+ text_file.write(curr_seq[line_start:line_start + 70] + '\n')
197
+
198
+ sequences_for_doc.append((f"{filename_indexes}_{ii}", original_seq, curr_seq))
199
+ final_results.append((ii, curr_seq))
200
+
201
+ if recombination_collector:
202
+ pd.concat(recombination_collector, ignore_index=True).to_csv(
203
+ path.join(curr_output_path, 'recombination_sites.csv'), index=False)
204
+
205
+ if optimize and recombination_raw_collector:
206
+ # every candidate actually given a correction constraint during
207
+ # optimization, not just the collapsed "one representative per
208
+ # distinct site" view above - since those two views can now
209
+ # legitimately differ (see suspect_site_extractor's docstring and
210
+ # docs/detector-comparisons.md), this is what to check against
211
+ # final_sequence.txt if a diff shows an edit with no corresponding
212
+ # row in recombination_sites.csv.
213
+ pd.concat(recombination_raw_collector, ignore_index=True).to_csv(
214
+ path.join(curr_output_path, 'recombination_sites_corrected.csv'), index=False)
215
+
216
+ if slippage_collector:
217
+ pd.concat(slippage_collector, ignore_index=True).to_csv(
218
+ path.join(curr_output_path, 'slippage_sites.csv'), index=False)
219
+
220
+ if optimize and slippage_raw_collector:
221
+ # see the recombination_sites_corrected.csv comment above.
222
+ pd.concat(slippage_raw_collector, ignore_index=True).to_csv(
223
+ path.join(curr_output_path, 'slippage_sites_corrected.csv'), index=False)
224
+
225
+ if compute_motifs and motifs_collector:
226
+ pd.concat(motifs_collector, ignore_index=True).to_csv(
227
+ path.join(curr_output_path, 'motif_sites.csv'), index=False)
228
+
229
+ if sequences_for_doc:
230
+ create_word_document_with_highlighted_differences(sequences_for_doc, curr_output_path)
231
+
232
+ return final_results
233
+
234
+
235
+ def main(input_folder=None, output_path=None, compute_motifs=False, num_sites=np.inf,
236
+ motifs_path=None, common_motifs=None, optimize=True, mini_gc=0.3, maxi_gc=0.7,
237
+ method='use_best_codon', organism_name='not_specified', indexes=None,
238
+ recombination_mode='thorough', slippage_mode='default', custom_score_fn=None,
239
+ custom_score_minimize=False):
240
+ """Optimize every FASTA/GenBank file in `input_folder`, writing per-file CSVs
241
+ of detected hotspots and the optimized sequence into `output_path`.
242
+
243
+ Parameters
244
+ ----------
245
+ input_folder: str
246
+ Directory to scan for .fasta/.fna/.ffn/.faa/.frn/.fa/.gb/.gbk/.genbank
247
+ files (optionally gzipped), directly inside or one level under it.
248
+ output_path: str
249
+ Directory to write results into (one subdirectory per input file).
250
+ compute_motifs: bool
251
+ Whether to also detect methylation motif sites (needs `motifs_path` and/or
252
+ `common_motifs`).
253
+ num_sites: int or float('inf')
254
+ Max number of hotspots to report (in the CSVs/collapsed dataframes)
255
+ per category. Default: all. Does NOT limit how many correction
256
+ constraints are built during optimization - every detected candidate
257
+ is always given a constraint, regardless of `num_sites`, since
258
+ limiting that too could silently leave part of a genuine hotspot
259
+ unconstrained (see eso.pipeline.suspect_site_extractor).
260
+ motifs_path: str or None
261
+ Path to a MEME-minimal-format PSSM file.
262
+ common_motifs: list of str or None
263
+ Names from eso.detection.common_motifs.COMMON_MOTIFS (currently "dam",
264
+ "dcm" - E. coli's methylation systems) to include alongside any
265
+ `motifs_path` file, with no file needed at all. At least one of
266
+ `motifs_path`/`common_motifs` is required if compute_motifs=True.
267
+ optimize: bool
268
+ Whether to codon/GC-optimize and avoid hotspots, vs. just detect them.
269
+ mini_gc, maxi_gc: float in [0, 1]
270
+ Allowed GC-content range within any 50nt window.
271
+ method: {"use_best_codon", "match_codon_usage", "harmonize_rca"}
272
+ Codon optimization strategy.
273
+ organism_name: str
274
+ Host organism for codon optimization (see eso.optimize._codon_optimization_objectives).
275
+ indexes: dict
276
+ Maps (file_stem, seq_index_str) -> (orf_region_string, exclusion_region_string),
277
+ 1-indexed and inclusive, e.g. {("my_gene", "0"): ("1-6, 51-68", "1-6, 50-68")}.
278
+ Omit or pass {} to treat entire sequences as the ORF with no exclusions.
279
+ recombination_mode: {"thorough", "fast"}
280
+ See eso.detection.dispatch.find_recombination_sites. "thorough" (default)
281
+ catches near-duplicate hotspots and stays roughly linear (confirmed
282
+ practical up to 1,000,000nt - see docs/detector-comparisons.md);
283
+ "fast" is exact-match only, 19-34x faster, for workloads where that
284
+ speed gap actually matters.
285
+ slippage_mode: {"default", "fast"}
286
+ See eso.detection.dispatch.find_slippage_sites. Both detect identical
287
+ hotspots; "default" is also faster at every length tested.
288
+ custom_score_fn: callable(str) -> float, or None
289
+ If given, replaces CAI/tAI (organism_name/method) with this scoring
290
+ function - see eso.custom_score.CustomScore. Most users should use
291
+ the `--custom-score-file` CLI flag / eso.custom_score.load_custom_score_from_file
292
+ instead of passing a function directly here.
293
+ custom_score_minimize: bool
294
+ See eso.optimize.optimization_engine; only used if custom_score_fn is given.
295
+
296
+ Returns
297
+ -------
298
+ (message, results) where message is 'Success!' or a validation error, and
299
+ results is a list of (file, seq_index, optimized_sequence) tuples.
300
+ """
301
+ indexes = indexes or {}
302
+ if output_path is None:
303
+ output_path = path.join(input_folder or '.', 'output')
304
+
305
+ files = relevant_file_paths(input_folder=input_folder)
306
+ message = test_input(mini_gc, maxi_gc, indexes, files)
307
+
308
+ if message != 'Success!':
309
+ return message, []
310
+
311
+ final_results = []
312
+ for file in files:
313
+ data = file_opener(file)
314
+ curr_results = backend(
315
+ data, file, output_path, compute_motifs, num_sites, motifs_path,
316
+ optimize=optimize, mini_gc=mini_gc, maxi_gc=maxi_gc, method=method,
317
+ organism_name=organism_name, indexes=indexes, recombination_mode=recombination_mode,
318
+ slippage_mode=slippage_mode, common_motifs=common_motifs, custom_score_fn=custom_score_fn,
319
+ custom_score_minimize=custom_score_minimize)
320
+ final_results.extend((file, seq_index, seq) for seq_index, seq in curr_results)
321
+
322
+ return message, final_results
eso/report.py ADDED
@@ -0,0 +1,57 @@
1
+ """Optional Word-document report showing original vs. optimized sequences
2
+ with per-nucleotide differences highlighted.
3
+ """
4
+
5
+ from os import path
6
+
7
+ try:
8
+ from docx import Document
9
+ from docx.enum.text import WD_COLOR_INDEX
10
+ DOCX_AVAILABLE = True
11
+ except ImportError:
12
+ DOCX_AVAILABLE = False
13
+
14
+
15
+ def create_word_document_with_highlighted_differences(sequences_data, output_path):
16
+ """
17
+ Parameters
18
+ ----------
19
+ sequences_data: list of (sequence_name, original_seq, final_seq) tuples.
20
+ output_path: directory in which to save 'sequence_comparison.docx'.
21
+ """
22
+ if not DOCX_AVAILABLE:
23
+ print("python-docx not available (install the 'docx-report' extra), skipping Word document generation")
24
+ return
25
+
26
+ doc = Document()
27
+ doc.add_heading('Sequence Optimization Results', 0)
28
+
29
+ for index, (seq_name, original_seq, final_seq) in enumerate(sequences_data):
30
+ doc.add_heading(f'Sequence: {seq_name}', level=1)
31
+
32
+ doc.add_heading('Original Sequence:', level=2)
33
+ original_paragraph = doc.add_paragraph()
34
+
35
+ doc.add_heading('Final Sequence:', level=2)
36
+ final_paragraph = doc.add_paragraph()
37
+
38
+ for i, char in enumerate(original_seq):
39
+ run = original_paragraph.add_run(char)
40
+ if i < len(final_seq) and char != final_seq[i]:
41
+ run.font.highlight_color = WD_COLOR_INDEX.YELLOW
42
+
43
+ for i, char in enumerate(final_seq):
44
+ run = final_paragraph.add_run(char)
45
+ if i < len(original_seq) and char != original_seq[i]:
46
+ run.font.highlight_color = WD_COLOR_INDEX.YELLOW
47
+
48
+ # was `if seq_name != sequences_data[-1][0]` - compared by name, so a
49
+ # sequence whose name happened to match the true last entry's name
50
+ # (e.g. two files/records that coincidentally share a stem) would
51
+ # wrongly skip its own page break. Compare by position instead.
52
+ if index != len(sequences_data) - 1:
53
+ doc.add_page_break()
54
+
55
+ doc_path = path.join(output_path, 'sequence_comparison.docx')
56
+ doc.save(doc_path)
57
+ print(f"Word document saved to: {doc_path}")