evolutionary-stability-optimizer 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- eso/__init__.py +11 -0
- eso/cli.py +111 -0
- eso/codon_usage.py +96 -0
- eso/constraints.py +179 -0
- eso/custom_score.py +170 -0
- eso/data/__init__.py +0 -0
- eso/data/human-antibody-heavy-chain-codon-frequencies.csv +62 -0
- eso/data/human-antibody-light-chain-codon-frequencies.csv +62 -0
- eso/detection/__init__.py +0 -0
- eso/detection/_overlap.py +112 -0
- eso/detection/common_motifs.py +81 -0
- eso/detection/dispatch.py +185 -0
- eso/detection/methylation.py +96 -0
- eso/detection/motif_utils.py +77 -0
- eso/detection/recombination.py +329 -0
- eso/detection/slippage.py +249 -0
- eso/detection/staubility_variant.py +284 -0
- eso/io_utils.py +312 -0
- eso/optimize.py +266 -0
- eso/pipeline.py +322 -0
- eso/report.py +57 -0
- eso/sequence_utils.py +41 -0
- evolutionary_stability_optimizer-0.1.0.dist-info/METADATA +501 -0
- evolutionary_stability_optimizer-0.1.0.dist-info/RECORD +27 -0
- evolutionary_stability_optimizer-0.1.0.dist-info/WHEEL +4 -0
- evolutionary_stability_optimizer-0.1.0.dist-info/entry_points.txt +3 -0
- evolutionary_stability_optimizer-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,329 @@
|
|
|
1
|
+
"""Repeat-mediated deletion (RMD) / recombination hotspot detection.
|
|
2
|
+
|
|
3
|
+
Finds pairs of near-identical sites (length >= 16, Levenshtein distance <= 1,
|
|
4
|
+
non-overlapping) that are candidates for recombination-mediated deletion, and
|
|
5
|
+
scores them with the empirical mutation-rate formula from the EFM Calculator
|
|
6
|
+
paper (Jack et al. 2015, ACS Synthetic Biology, DOI: 10.1021/acssynbio.5b00068).
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import numpy as np
|
|
10
|
+
import pandas as pd
|
|
11
|
+
from Levenshtein import distance as levenshtein_distance
|
|
12
|
+
|
|
13
|
+
from eso.detection._overlap import ranges_overlap, range_contains
|
|
14
|
+
from eso.sequence_utils import add_backward_sites, shorten_sequences
|
|
15
|
+
|
|
16
|
+
# matches the columns actually produced by the non-empty path below
|
|
17
|
+
# (sequence_1/sequence_2, not a single 'sequence' column)
|
|
18
|
+
RECOMBINATION_COLUMNS = [
|
|
19
|
+
'sequence_1', 'start_1', 'end_1', 'sequence_2', 'start_2', 'end_2',
|
|
20
|
+
'location_delta', 'site_length', 'log10_prob_recombination_ecoli',
|
|
21
|
+
]
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _generate_all_recombination_sites(seq):
|
|
25
|
+
"""Generate candidate 16-17mers (forward + reverse complement) at every offset,
|
|
26
|
+
plus their single-insertion/deletion/substitution-shortened variants, so that
|
|
27
|
+
near-duplicate site pairs can be found by exact match instead of all-pairs comparison.
|
|
28
|
+
"""
|
|
29
|
+
forward_insertions = [(seq[ii:ii + 17], ii, ii + 16) for ii in range(len(seq))]
|
|
30
|
+
forward_insertions = [x for x in forward_insertions if len(x[0]) == 17]
|
|
31
|
+
|
|
32
|
+
df_forward_insertions = pd.DataFrame.from_records(
|
|
33
|
+
data=forward_insertions, columns=['sequence', 'start', 'end'])
|
|
34
|
+
|
|
35
|
+
df_insertions = add_backward_sites(df_forward_insertions)
|
|
36
|
+
df_first_sites = shorten_sequences(df_forward_insertions)
|
|
37
|
+
df_substitutions = add_backward_sites(df_first_sites)
|
|
38
|
+
|
|
39
|
+
df_deletions = shorten_sequences(df_first_sites)
|
|
40
|
+
df_deletions = add_backward_sites(df_deletions)
|
|
41
|
+
|
|
42
|
+
return df_first_sites, df_insertions, df_substitutions, df_deletions
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _generate_neighbors(curr_seq):
|
|
46
|
+
"""All sequences up to one insertion, deletion, or substitution away from curr_seq."""
|
|
47
|
+
neighbors_substitutions = set()
|
|
48
|
+
neighbors_deletions = set()
|
|
49
|
+
neighbors_insertions = set()
|
|
50
|
+
|
|
51
|
+
for ii in range(len(curr_seq)):
|
|
52
|
+
prefix = curr_seq[:ii]
|
|
53
|
+
suffix = curr_seq[ii:]
|
|
54
|
+
for nt in ['A', 'C', 'G', 'T']:
|
|
55
|
+
neighbors_insertions.add(prefix + nt + suffix)
|
|
56
|
+
suffix = suffix[1:]
|
|
57
|
+
for nt in ['A', 'C', 'G', 'T']:
|
|
58
|
+
neighbors_substitutions.add(prefix + nt + suffix)
|
|
59
|
+
neighbors_deletions.add(prefix + suffix)
|
|
60
|
+
|
|
61
|
+
return neighbors_substitutions, neighbors_deletions, neighbors_insertions
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _order_site12(df):
|
|
65
|
+
lower_second = (df.start_1 > df.start_2)
|
|
66
|
+
df_output = df.copy()
|
|
67
|
+
for col in ['sequence', 'start', 'end']:
|
|
68
|
+
df_output.loc[lower_second, f'{col}_1'] = df[f'{col}_2']
|
|
69
|
+
df_output.loc[lower_second, f'{col}_2'] = df[f'{col}_1']
|
|
70
|
+
return df_output
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _build_sequence_index(df):
|
|
74
|
+
"""Map sequence -> list of (start, end) tuples, for O(1) dict lookup instead
|
|
75
|
+
of a `df.sequence.isin(...)` filter (which rescans/reindexes the whole
|
|
76
|
+
frame) called once per candidate site.
|
|
77
|
+
"""
|
|
78
|
+
index = {}
|
|
79
|
+
for sequence, start, end in zip(df.sequence, df.start, df.end):
|
|
80
|
+
index.setdefault(sequence, []).append((start, end))
|
|
81
|
+
return index
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _generate_relevant_pairs(df_first_sites, df_insertions, df_substitutions, df_deletions):
|
|
85
|
+
"""Profiling showed this function dominates find_recombination_sites'
|
|
86
|
+
runtime (5.1s of 7.0s on an ~830nt sequence), almost entirely in pandas
|
|
87
|
+
overhead: `.isin()` plus the surrounding boolean-mask filtering and
|
|
88
|
+
`.iloc`/`.loc` indexing, called once per candidate site (potentially
|
|
89
|
+
O(n) sites). Building a plain-dict index once up front, and iterating
|
|
90
|
+
the input frames as plain Python tuples via zip() rather than
|
|
91
|
+
`.iloc[ii, col]`, avoids nearly all of that per-call pandas overhead
|
|
92
|
+
while producing identical pairs.
|
|
93
|
+
"""
|
|
94
|
+
pairs = []
|
|
95
|
+
|
|
96
|
+
substitutions_index = _build_sequence_index(df_substitutions)
|
|
97
|
+
deletions_index = _build_sequence_index(df_deletions)
|
|
98
|
+
insertions_index = _build_sequence_index(df_insertions)
|
|
99
|
+
|
|
100
|
+
for sequence_1, start_1, end_1 in zip(df_first_sites.sequence, df_first_sites.start, df_first_sites.end):
|
|
101
|
+
neighbors_substitutions, neighbors_deletions, neighbors_insertions = _generate_neighbors(sequence_1)
|
|
102
|
+
|
|
103
|
+
for neighbors, index in (
|
|
104
|
+
(neighbors_substitutions, substitutions_index),
|
|
105
|
+
(neighbors_deletions, deletions_index),
|
|
106
|
+
(neighbors_insertions, insertions_index),
|
|
107
|
+
):
|
|
108
|
+
for neighbor_seq in neighbors:
|
|
109
|
+
for start_2, end_2 in index.get(neighbor_seq, ()):
|
|
110
|
+
pairs.append((sequence_1, start_1, end_1, neighbor_seq, start_2, end_2))
|
|
111
|
+
|
|
112
|
+
df_pairs = pd.DataFrame.from_records(
|
|
113
|
+
pairs, columns=['sequence_1', 'start_1', 'end_1', 'sequence_2', 'start_2', 'end_2'])
|
|
114
|
+
|
|
115
|
+
df_pairs = _order_site12(df_pairs)
|
|
116
|
+
|
|
117
|
+
# drop overlapping pairs and duplicates
|
|
118
|
+
df_pairs = df_pairs[df_pairs.end_1 < df_pairs.start_2].drop_duplicates().reset_index(drop=True)
|
|
119
|
+
|
|
120
|
+
return df_pairs
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _elongate_sites(row, full_seq):
|
|
124
|
+
"""Extend each site pair outward while they remain within Levenshtein distance 1
|
|
125
|
+
and non-overlapping, so the reported site is the full recombination hotspot
|
|
126
|
+
rather than just its detected 16-mer core.
|
|
127
|
+
"""
|
|
128
|
+
sequence_1, start_1, end_1 = row.sequence_1, row.start_1, row.end_1 + 1
|
|
129
|
+
sequence_2, start_2, end_2 = row.sequence_2, row.start_2, row.end_2 + 1
|
|
130
|
+
|
|
131
|
+
while (levenshtein_distance(full_seq[start_1:end_1 + 1], full_seq[start_2:end_2 + 1], score_cutoff=1) < 2) \
|
|
132
|
+
and (end_1 + 1 < start_2):
|
|
133
|
+
end_1 += 1
|
|
134
|
+
end_2 += 1
|
|
135
|
+
|
|
136
|
+
while (levenshtein_distance(full_seq[start_1 - 1:end_1], full_seq[start_2 - 1:end_2], score_cutoff=1) < 2) \
|
|
137
|
+
and (end_1 + 1 < start_2):
|
|
138
|
+
start_1 -= 1
|
|
139
|
+
start_2 -= 1
|
|
140
|
+
|
|
141
|
+
row.sequence_1, row.start_1, row.end_1 = full_seq[start_1:end_1], start_1, end_1
|
|
142
|
+
row.sequence_2, row.start_2, row.end_2 = full_seq[start_2:end_2], start_2, end_2
|
|
143
|
+
return row
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def calc_recombination_score(location_delta, site_length):
|
|
147
|
+
"""Empirical log10 probability of recombination-mediated deletion.
|
|
148
|
+
|
|
149
|
+
Formula and constants per Oliveira et al. 2008 (Plasmid 60:159-165,
|
|
150
|
+
doi:10.1016/j.plasmid.2008.06.004), Eq. (4) and Table 3, recA+ row:
|
|
151
|
+
FR(LR,LS) = (A+LS)^(-a/LR) * LR/(1+B*LR+C*LS), with A=5.8, B=1465.6,
|
|
152
|
+
C=0 (not fitted for recA+), a=29.0 (this module's `alpha`).
|
|
153
|
+
|
|
154
|
+
The EFM Calculator's own source (github.com/barricklab/efm-calculator,
|
|
155
|
+
get_recombo_rate) hardcodes this same B=1465.6 and exponent=29, but pairs
|
|
156
|
+
them with A=8.8 - a value that does not appear anywhere in the recA+ row
|
|
157
|
+
of Table 3. It matches the *recA-* row's exponent (a=8.8 there, a
|
|
158
|
+
different parameter, in a different model, with a different B and an
|
|
159
|
+
extra C term) instead. That looks like a transcription mix-up in the
|
|
160
|
+
reference tool between the two rows of the source table, not a
|
|
161
|
+
correction - so this codebase's original 5.8 (from ESO_curr/STABLES) is
|
|
162
|
+
the value actually supported by the primary source, and the previous
|
|
163
|
+
change to 8.8 here was a mistake, reverted.
|
|
164
|
+
"""
|
|
165
|
+
a, b, c, alpha = 5.8, 1465.6, 0, 29
|
|
166
|
+
|
|
167
|
+
first_component = a + location_delta
|
|
168
|
+
second_component = -1 * (alpha / site_length)
|
|
169
|
+
third_component = site_length / (1 + b * site_length + c * location_delta)
|
|
170
|
+
|
|
171
|
+
recombination_probability = (first_component ** second_component) * third_component
|
|
172
|
+
return np.log10(recombination_probability)
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def _collapse_pairs_by_predicate(df_pairs, should_drop_range):
|
|
176
|
+
"""Shared non-max-suppression walk for site-pairs: process pairs in
|
|
177
|
+
descending score order (ties broken by descending combined span, then a
|
|
178
|
+
stable sort, for the same determinism reason documented on
|
|
179
|
+
eso.detection._overlap._collapse_by_predicate - found via property-based
|
|
180
|
+
testing on that same-shaped single-range logic, not reproduced here as
|
|
181
|
+
its own test but the same tiebreak applied preemptively since the walk
|
|
182
|
+
is structurally identical), dropping a pair exactly when
|
|
183
|
+
`should_drop_range` is True of BOTH its site_1 and site_2 ranges against
|
|
184
|
+
some already-kept pair's corresponding ranges, otherwise keeping it.
|
|
185
|
+
Requiring agreement on both sides (not just one) avoids merging two
|
|
186
|
+
genuinely distinct hotspots that happen to share one site. The two
|
|
187
|
+
functions below differ only in which predicate they pass in (overlap vs.
|
|
188
|
+
full containment).
|
|
189
|
+
"""
|
|
190
|
+
kept_rows = []
|
|
191
|
+
kept_ranges = [] # list of ((start_1,end_1), (start_2,end_2))
|
|
192
|
+
|
|
193
|
+
combined_span = (df_pairs.end_1 - df_pairs.start_1) + (df_pairs.end_2 - df_pairs.start_2)
|
|
194
|
+
ordering = df_pairs.assign(_combined_span=combined_span).sort_values(
|
|
195
|
+
['log10_prob_recombination_ecoli', '_combined_span'], ascending=[False, False], kind='mergesort')
|
|
196
|
+
|
|
197
|
+
for _, row in ordering.iterrows():
|
|
198
|
+
range_1 = (row.start_1, row.end_1)
|
|
199
|
+
range_2 = (row.start_2, row.end_2)
|
|
200
|
+
if any(
|
|
201
|
+
should_drop_range(range_1, r1) and should_drop_range(range_2, r2)
|
|
202
|
+
for r1, r2 in kept_ranges
|
|
203
|
+
):
|
|
204
|
+
continue
|
|
205
|
+
kept_rows.append(row.drop('_combined_span'))
|
|
206
|
+
kept_ranges.append((range_1, range_2))
|
|
207
|
+
|
|
208
|
+
return pd.DataFrame(kept_rows, columns=df_pairs.columns) if kept_rows else df_pairs.iloc[0:0]
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def _collapse_overlapping_pairs(df_pairs):
|
|
212
|
+
"""Different seed 16-mers for the same real hotspot converge, via
|
|
213
|
+
elongation, to slightly different (start, end) extents rather than one
|
|
214
|
+
canonical pair - so even exact-coordinate dedup leaves several
|
|
215
|
+
near-identical rows per real site. Collapse them via non-max suppression:
|
|
216
|
+
walk pairs in descending score order, keeping a pair only if BOTH its
|
|
217
|
+
site_1 and site_2 ranges are still free of an already-kept pair's
|
|
218
|
+
corresponding range.
|
|
219
|
+
"""
|
|
220
|
+
return _collapse_pairs_by_predicate(df_pairs, should_drop_range=ranges_overlap)
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def _collapse_overlapping_pairs_no_coverage_loss(df_pairs):
|
|
224
|
+
"""Like _collapse_overlapping_pairs, but only drops a pair when BOTH its
|
|
225
|
+
site_1 and site_2 ranges are FULLY CONTAINED in an already-kept pair's
|
|
226
|
+
corresponding ranges - not merely overlapping them. See
|
|
227
|
+
eso.detection._overlap.collapse_overlapping_intervals_no_coverage_loss
|
|
228
|
+
for why: plain overlap-based NMS can silently drop a genuinely distinct,
|
|
229
|
+
only-partially-overlapping hotspot pair, leaving whatever part of its
|
|
230
|
+
span isn't covered by the kept pair with no correction constraint. Rows
|
|
231
|
+
returned here can legitimately still overlap each other - use this to
|
|
232
|
+
build constraints, never for a "how many distinct sites" count (use
|
|
233
|
+
_collapse_overlapping_pairs for that instead, via collapse_recombination_sites).
|
|
234
|
+
"""
|
|
235
|
+
def _kept_fully_covers_current(current_range, kept_range):
|
|
236
|
+
return range_contains(outer=kept_range, inner=current_range)
|
|
237
|
+
|
|
238
|
+
return _collapse_pairs_by_predicate(df_pairs, should_drop_range=_kept_fully_covers_current)
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def recombination_sites_for_constraints(df_pairs):
|
|
242
|
+
"""Reduce raw recombination candidates (see find_recombination_candidates)
|
|
243
|
+
for feeding into correction-constraint building
|
|
244
|
+
(recombination_to_multiple_avoidance_sites), WITHOUT the coverage-loss
|
|
245
|
+
risk plain non-max suppression (collapse_recombination_sites) has - see
|
|
246
|
+
_collapse_overlapping_pairs_no_coverage_loss and
|
|
247
|
+
eso.detection.slippage.slippage_sites_for_constraints for the concrete
|
|
248
|
+
failure this avoids.
|
|
249
|
+
"""
|
|
250
|
+
if df_pairs.shape[0] == 0:
|
|
251
|
+
return df_pairs
|
|
252
|
+
return _collapse_overlapping_pairs_no_coverage_loss(df_pairs)
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def find_recombination_candidates(seq):
|
|
256
|
+
"""Find every candidate recombination (RMD) site-pair in `seq`, WITHOUT
|
|
257
|
+
collapsing overlapping pairs down to one representative per real hotspot.
|
|
258
|
+
|
|
259
|
+
A pair of sites is a candidate if both are >=16nt, within Levenshtein
|
|
260
|
+
distance 1 of each other, and non-overlapping.
|
|
261
|
+
|
|
262
|
+
This is the shared base both find_recombination_sites (collapsed report
|
|
263
|
+
view) and recombination_sites_for_constraints (constraint-building view)
|
|
264
|
+
are built from.
|
|
265
|
+
|
|
266
|
+
Returns a dataframe sorted by descending mutation risk
|
|
267
|
+
(`log10_prob_recombination_ecoli`), not limited by `num_sites` (that only
|
|
268
|
+
makes sense for the collapsed, human-facing view - see
|
|
269
|
+
find_recombination_sites/collapse_recombination_sites).
|
|
270
|
+
"""
|
|
271
|
+
empty_df = pd.DataFrame(columns=RECOMBINATION_COLUMNS)
|
|
272
|
+
|
|
273
|
+
df_first_sites, df_insertions, df_substitutions, df_deletions = _generate_all_recombination_sites(seq)
|
|
274
|
+
df_pairs = _generate_relevant_pairs(df_first_sites, df_insertions, df_substitutions, df_deletions)
|
|
275
|
+
|
|
276
|
+
if df_pairs.shape[0] == 0:
|
|
277
|
+
return empty_df
|
|
278
|
+
|
|
279
|
+
df_pairs = df_pairs.apply(lambda row: _elongate_sites(row, seq), axis=1)
|
|
280
|
+
|
|
281
|
+
df_pairs.loc[:, 'location_delta'] = df_pairs.start_2 - df_pairs.end_1
|
|
282
|
+
df_pairs.loc[:, 'site_length'] = df_pairs.apply(
|
|
283
|
+
lambda row: max(row.end_1 - row.start_1, row.end_2 - row.start_2), axis=1)
|
|
284
|
+
|
|
285
|
+
df_pairs.loc[:, 'log10_prob_recombination_ecoli'] = df_pairs.apply(
|
|
286
|
+
lambda x: calc_recombination_score(x.location_delta, x.site_length), axis=1)
|
|
287
|
+
|
|
288
|
+
df_pairs = df_pairs[df_pairs.log10_prob_recombination_ecoli > -9]
|
|
289
|
+
|
|
290
|
+
for col in ['start_1', 'end_1', 'start_2', 'end_2', 'location_delta', 'site_length']:
|
|
291
|
+
df_pairs.loc[:, col] = df_pairs[col].astype(int)
|
|
292
|
+
|
|
293
|
+
return df_pairs.sort_values('log10_prob_recombination_ecoli', ascending=False).reset_index(drop=True)
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def collapse_recombination_sites(df_pairs, num_sites=np.inf):
|
|
297
|
+
"""Collapse raw recombination candidates (see find_recombination_candidates)
|
|
298
|
+
down to one representative pair per group of overlapping pairs, for a
|
|
299
|
+
human-facing "distinct sites" report or count - NOT for building
|
|
300
|
+
correction constraints (see recombination_sites_for_constraints instead -
|
|
301
|
+
a dropped pair's uniquely-covered span would otherwise get no fix).
|
|
302
|
+
|
|
303
|
+
The many candidate-seed rows that converge (via elongation) on the same
|
|
304
|
+
or overlapping real hotspot are collapsed down to one representative row
|
|
305
|
+
each.
|
|
306
|
+
"""
|
|
307
|
+
if df_pairs.shape[0] == 0:
|
|
308
|
+
return df_pairs
|
|
309
|
+
|
|
310
|
+
df_pairs = _collapse_overlapping_pairs(df_pairs)
|
|
311
|
+
df_pairs = df_pairs.sort_values('log10_prob_recombination_ecoli', ascending=False)
|
|
312
|
+
|
|
313
|
+
if num_sites < np.inf:
|
|
314
|
+
df_pairs = df_pairs.head(int(num_sites))
|
|
315
|
+
|
|
316
|
+
return df_pairs
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
def find_recombination_sites(seq, num_sites=np.inf):
|
|
320
|
+
"""Find recombination (RMD) hotspots in `seq`, collapsed to one
|
|
321
|
+
representative pair per real hotspot (see collapse_recombination_sites)
|
|
322
|
+
and limited to `num_sites` distinct site-pairs if given - the
|
|
323
|
+
human-facing report/count view. Do NOT use this to build correction
|
|
324
|
+
constraints (see find_recombination_candidates).
|
|
325
|
+
|
|
326
|
+
Returns a dataframe sorted by descending mutation risk
|
|
327
|
+
(`log10_prob_recombination_ecoli`).
|
|
328
|
+
"""
|
|
329
|
+
return collapse_recombination_sites(find_recombination_candidates(seq), num_sites)
|
|
@@ -0,0 +1,249 @@
|
|
|
1
|
+
"""Replication-slippage (SSR - simple sequence repeat) hotspot detection.
|
|
2
|
+
|
|
3
|
+
Finds short tandem repeats (base units of length 1-15 repeated 3+ times, or a
|
|
4
|
+
single nucleotide repeated 12+ times) and scores them with the empirical
|
|
5
|
+
mutation-rate formula from the EFM Calculator paper (Jack et al. 2015, ACS
|
|
6
|
+
Synthetic Biology, DOI: 10.1021/acssynbio.5b00068).
|
|
7
|
+
|
|
8
|
+
The length-1 (homopolymer) minimum is 12, not 3, for two stacked reasons:
|
|
9
|
+
|
|
10
|
+
1. log10_prob = -12.9 + 0.729*n crosses the -9 filter cutoff (applied below)
|
|
11
|
+
at n=6 - runs of n=4/5 always fail it and are never worth detecting.
|
|
12
|
+
2. Every homopolymer run of n>=6 is *also* always detected as a length-2 site
|
|
13
|
+
(any 6+ identical characters trivially contain "XX" repeated 3+ times),
|
|
14
|
+
scored as -4.749 + 0.063*floor(n/2) - and collapse_overlapping_intervals
|
|
15
|
+
below only keeps the single highest-scoring representation per physical
|
|
16
|
+
site. Comparing the two formulas: the length-1 score grows ~0.729/nt while
|
|
17
|
+
the length-2 score grows only ~0.0315/nt, so length-2 always wins for
|
|
18
|
+
n=6..11 (e.g. n=8: length-1 -7.068 vs length-2 -4.497) - the length-1
|
|
19
|
+
detection is real but its result is always discarded. The crossover is
|
|
20
|
+
n=12 exactly (length-1 -4.152 vs length-2 -4.371) - only from there does
|
|
21
|
+
the length-1 reading start being the one that survives, correctly
|
|
22
|
+
reflecting a homopolymer run's much higher risk than a linear-in-repeats
|
|
23
|
+
length-2 score would suggest.
|
|
24
|
+
|
|
25
|
+
So detecting length-1 sites below n=12 is provably pure waste - the result
|
|
26
|
+
never changes the final output regardless, since length-2 detection (run
|
|
27
|
+
independently, over the same sequence) always produces a same-or-better-scoring
|
|
28
|
+
overlapping candidate for that region. This doesn't apply to length>1 units:
|
|
29
|
+
log10_prob = -4.749 + 0.063*n is already > -9 at the smallest detectable n=3,
|
|
30
|
+
so every length>1 candidate always passes the -9 filter on its own.
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
import numpy as np
|
|
34
|
+
import pandas as pd
|
|
35
|
+
|
|
36
|
+
from eso.detection._overlap import collapse_overlapping_intervals, collapse_overlapping_intervals_no_coverage_loss
|
|
37
|
+
|
|
38
|
+
SLIPPAGE_COLUMNS = ['start', 'end', 'length_base_unit', 'sequence']
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _find_all(a_str, sub):
|
|
42
|
+
"""Non-overlapping occurrences of `sub` in `a_str`."""
|
|
43
|
+
start = 0
|
|
44
|
+
while True:
|
|
45
|
+
start = a_str.find(sub, start)
|
|
46
|
+
if start == -1:
|
|
47
|
+
return
|
|
48
|
+
yield start
|
|
49
|
+
start += len(sub)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _find_longest_match(seq, subunit, start_index):
|
|
53
|
+
"""Given a start location and a repeat unit, find the full extent of the run."""
|
|
54
|
+
curr_seq = seq[start_index:]
|
|
55
|
+
len_sub = len(subunit)
|
|
56
|
+
end_shift = 0
|
|
57
|
+
while curr_seq[:len_sub] == subunit:
|
|
58
|
+
end_shift += len_sub
|
|
59
|
+
curr_seq = curr_seq[len_sub:]
|
|
60
|
+
end_index = start_index + end_shift
|
|
61
|
+
return start_index, end_index, seq[start_index:end_index], len_sub
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _generate_slippage_sites_current_subunit(seq, subunit):
|
|
65
|
+
curr_slippage_sites = []
|
|
66
|
+
# 12 for length-1 units (below that, length-2 detection of the same
|
|
67
|
+
# region always outscores it - see module docstring), 3 for longer units.
|
|
68
|
+
subseq = subunit * (12 if len(subunit) == 1 else 3)
|
|
69
|
+
|
|
70
|
+
indexes_curr = list(_find_all(seq, subseq))
|
|
71
|
+
if not indexes_curr:
|
|
72
|
+
return curr_slippage_sites
|
|
73
|
+
|
|
74
|
+
curr_slippage_sites.append(_find_longest_match(seq, subunit, indexes_curr[0]))
|
|
75
|
+
|
|
76
|
+
for index in indexes_curr[1:]:
|
|
77
|
+
end_previous_site = curr_slippage_sites[-1][1]
|
|
78
|
+
if index <= end_previous_site:
|
|
79
|
+
continue
|
|
80
|
+
curr_slippage_sites.append(_find_longest_match(seq, subunit, index))
|
|
81
|
+
|
|
82
|
+
return curr_slippage_sites
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _find_slippage_len1(seq):
|
|
86
|
+
slippage_sites = []
|
|
87
|
+
for nt in ['A', 'C', 'G', 'T']:
|
|
88
|
+
slippage_sites.extend(_generate_slippage_sites_current_subunit(seq, nt))
|
|
89
|
+
return pd.DataFrame.from_records(
|
|
90
|
+
data=slippage_sites, columns=['start', 'end', 'sequence', 'length_base_unit'],
|
|
91
|
+
).drop_duplicates().reset_index(drop=True)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _find_relevant_subunits_len_l(seq, length):
|
|
95
|
+
"""Unique substrings of `length` that appear 3+ times in a row, as repeat-unit candidates.
|
|
96
|
+
|
|
97
|
+
Profiling showed the original approach - build every unique substring,
|
|
98
|
+
then call `seq.find(subunit*3)` once per candidate - spends ~93% of total
|
|
99
|
+
find_slippage_sites runtime in `str.find` (88,151 calls on a 9.6kb random
|
|
100
|
+
sequence), since each call rescans the whole sequence: O(unique
|
|
101
|
+
candidates) x O(n) = O(n^2) for a sequence with mostly-unique substrings.
|
|
102
|
+
Building a position dict in one pass and checking adjacency via O(1) set
|
|
103
|
+
membership (positions p, p+length, p+2*length all present means
|
|
104
|
+
`subunit*3` occurs at p) is equivalent but O(n) instead.
|
|
105
|
+
"""
|
|
106
|
+
positions_by_subunit = {}
|
|
107
|
+
for ii in range(len(seq) - length + 1):
|
|
108
|
+
positions_by_subunit.setdefault(seq[ii:ii + length], []).append(ii)
|
|
109
|
+
|
|
110
|
+
relevant = []
|
|
111
|
+
for subunit, positions in positions_by_subunit.items():
|
|
112
|
+
position_set = set(positions)
|
|
113
|
+
if any((p + length) in position_set and (p + 2 * length) in position_set for p in positions):
|
|
114
|
+
relevant.append(subunit)
|
|
115
|
+
|
|
116
|
+
return sorted(relevant)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _find_slippage_len_l(seq, length):
|
|
120
|
+
"""Note: 1 < length < 16."""
|
|
121
|
+
relevant_subunits = _find_relevant_subunits_len_l(seq, length)
|
|
122
|
+
slippage_sites = []
|
|
123
|
+
for subunit in relevant_subunits:
|
|
124
|
+
slippage_sites.extend(_generate_slippage_sites_current_subunit(seq, subunit))
|
|
125
|
+
return pd.DataFrame.from_records(data=slippage_sites, columns=['start', 'end', 'sequence', 'length_base_unit'])
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def find_slippage_candidates(seq):
|
|
129
|
+
"""Find every candidate slippage (SSR) hotspot in `seq` across repeat-unit
|
|
130
|
+
lengths 1-15, WITHOUT collapsing overlapping candidates down to one
|
|
131
|
+
representative per physical site.
|
|
132
|
+
|
|
133
|
+
This is the shared base both find_slippage_sites (collapsed report view,
|
|
134
|
+
via collapse_slippage_sites) and slippage_sites_for_constraints
|
|
135
|
+
(constraint-building view) are built from - see the latter for why
|
|
136
|
+
building constraints needs something other than plain non-max
|
|
137
|
+
suppression.
|
|
138
|
+
|
|
139
|
+
Returns a dataframe sorted by descending mutation risk
|
|
140
|
+
(`log10_prob_slippage_ecoli`), not limited by `num_sites` (that only
|
|
141
|
+
makes sense for the collapsed, human-facing view - see
|
|
142
|
+
find_slippage_sites/collapse_slippage_sites).
|
|
143
|
+
"""
|
|
144
|
+
slippage_sites_list = [_find_slippage_len1(seq)]
|
|
145
|
+
for length in range(2, 16):
|
|
146
|
+
slippage_sites_list.append(_find_slippage_len_l(seq, length))
|
|
147
|
+
df_slippage = pd.concat(slippage_sites_list, ignore_index=True)[SLIPPAGE_COLUMNS]
|
|
148
|
+
|
|
149
|
+
df_slippage.loc[:, 'num_base_units'] = (
|
|
150
|
+
df_slippage.sequence.apply(len) / df_slippage.length_base_unit
|
|
151
|
+
).astype(int)
|
|
152
|
+
|
|
153
|
+
df_slippage.loc[:, 'log10_prob_slippage_ecoli'] = -4.749 + 0.063 * df_slippage['num_base_units']
|
|
154
|
+
df_slippage.loc[df_slippage.length_base_unit == 1, 'log10_prob_slippage_ecoli'] = (
|
|
155
|
+
-12.9 + 0.729 * df_slippage['num_base_units']
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
df_slippage = df_slippage[df_slippage.log10_prob_slippage_ecoli > -9]
|
|
159
|
+
return df_slippage.sort_values('log10_prob_slippage_ecoli', ascending=False).reset_index(drop=True)
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def slippage_sites_for_constraints(df_slippage):
|
|
163
|
+
"""Reduce raw slippage candidates (see find_slippage_candidates) for
|
|
164
|
+
feeding into correction-constraint building (modify_df_slippage), WITHOUT
|
|
165
|
+
the coverage-loss risk plain non-max suppression (collapse_slippage_sites)
|
|
166
|
+
has.
|
|
167
|
+
|
|
168
|
+
Regular NMS (`collapse_overlapping_intervals`, used by
|
|
169
|
+
collapse_slippage_sites) only requires two candidates' ranges to
|
|
170
|
+
*overlap*, not that one *contain* the other, before dropping the
|
|
171
|
+
lower-scoring one - so two genuinely distinct, only-partially-overlapping
|
|
172
|
+
hotspots can have the lower-scoring one dropped entirely, silently
|
|
173
|
+
leaving the part of its span the higher-scoring one doesn't cover with no
|
|
174
|
+
correction constraint at all. Concrete case: a length-2 site over
|
|
175
|
+
[0, 10) and a length-3 site over [8, 20) overlap only at [8, 10) -
|
|
176
|
+
collapsing keeps whichever scores higher and drops the other completely,
|
|
177
|
+
so positions [10, 20) (if the length-2 site wins) would get no
|
|
178
|
+
constraint, even though the length-3 site was a real,
|
|
179
|
+
independently-detected hotspot. See docs/detector-comparisons.md.
|
|
180
|
+
|
|
181
|
+
This uses `collapse_overlapping_intervals_no_coverage_loss` instead: a
|
|
182
|
+
candidate is only dropped if its ENTIRE range is already covered by a
|
|
183
|
+
higher-scoring kept candidate - the common case (several detections
|
|
184
|
+
converging on essentially the same physical site) still collapses down
|
|
185
|
+
to one row as before, but a candidate sticking out beyond every
|
|
186
|
+
higher-scoring one it overlaps is kept, so every real hotspot still gets
|
|
187
|
+
a constraint covering its full extent, not just whatever a single
|
|
188
|
+
NMS survivor happened to cover.
|
|
189
|
+
"""
|
|
190
|
+
return collapse_overlapping_intervals_no_coverage_loss(df_slippage, score_col='log10_prob_slippage_ecoli')
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def collapse_slippage_sites(df_slippage, num_sites=np.inf):
|
|
194
|
+
"""Collapse raw slippage candidates (see find_slippage_candidates) down to
|
|
195
|
+
one representative per group of overlapping candidates, for a
|
|
196
|
+
human-facing "distinct sites" report or count - NOT for building
|
|
197
|
+
correction constraints (see slippage_sites_for_constraints instead - a
|
|
198
|
+
dropped candidate's uniquely-covered span would otherwise get no fix).
|
|
199
|
+
|
|
200
|
+
different base-unit lengths (and different phase offsets within the same
|
|
201
|
+
length) can each detect the same physical repeat as a separate row -
|
|
202
|
+
e.g. "GCGCGCGC" is a valid length-2 run starting at position N, AND its
|
|
203
|
+
1-shifted substring "CGCGCG" is a separate valid length-2 run starting
|
|
204
|
+
at N+1. Collapsing by exact 'start' alone (an earlier approach) misses
|
|
205
|
+
this, since the rows don't share a start position. Keep one
|
|
206
|
+
representative (highest scoring) per group of overlapping ranges instead.
|
|
207
|
+
"""
|
|
208
|
+
df_slippage = collapse_overlapping_intervals(df_slippage, score_col='log10_prob_slippage_ecoli')
|
|
209
|
+
df_slippage = df_slippage.sort_values(['log10_prob_slippage_ecoli', 'length_base_unit'], ascending=[False, False])
|
|
210
|
+
|
|
211
|
+
if num_sites < np.inf:
|
|
212
|
+
df_slippage = df_slippage.head(int(num_sites))
|
|
213
|
+
|
|
214
|
+
return df_slippage
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def find_slippage_sites(seq, num_sites=np.inf):
|
|
218
|
+
"""Find slippage (SSR) hotspots in `seq`, collapsed to one representative
|
|
219
|
+
per physical site (see collapse_slippage_sites) and limited to
|
|
220
|
+
`num_sites` if given - the human-facing report/count view. Do NOT use
|
|
221
|
+
this to build correction constraints (see slippage_sites_for_constraints).
|
|
222
|
+
|
|
223
|
+
Returns a dataframe sorted by descending mutation risk
|
|
224
|
+
(`log10_prob_slippage_ecoli`).
|
|
225
|
+
"""
|
|
226
|
+
return collapse_slippage_sites(find_slippage_candidates(seq), num_sites)
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def modify_df_slippage(df_slippage):
|
|
230
|
+
"""Convert each slippage-site row (N repeated base units) into ~N/2 rows of
|
|
231
|
+
individual base units to avoid (skipping every other one) - enough to
|
|
232
|
+
disrupt the repeat without necessarily eliminating it entirely.
|
|
233
|
+
|
|
234
|
+
Example: a row for "TGTGTGTGTG" (base unit "TG", num_base_units=5) becomes
|
|
235
|
+
3 rows, each a single "TG" occurrence and its coordinates.
|
|
236
|
+
"""
|
|
237
|
+
df_list = []
|
|
238
|
+
|
|
239
|
+
for idx in df_slippage.index:
|
|
240
|
+
num_base_units = df_slippage.loc[idx].num_base_units
|
|
241
|
+
length = df_slippage.loc[idx].length_base_unit
|
|
242
|
+
for i in range(0, (num_base_units - 1), 2):
|
|
243
|
+
df_list.append({
|
|
244
|
+
'sequence': df_slippage.loc[idx].sequence[int(i * length):int((i + 1) * length)],
|
|
245
|
+
'start': int(df_slippage.loc[idx].start + i * length),
|
|
246
|
+
'end': int(df_slippage.loc[idx].start + (i + 1) * length),
|
|
247
|
+
})
|
|
248
|
+
|
|
249
|
+
return pd.DataFrame(df_list)
|