evolutionary-stability-optimizer 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,329 @@
1
+ """Repeat-mediated deletion (RMD) / recombination hotspot detection.
2
+
3
+ Finds pairs of near-identical sites (length >= 16, Levenshtein distance <= 1,
4
+ non-overlapping) that are candidates for recombination-mediated deletion, and
5
+ scores them with the empirical mutation-rate formula from the EFM Calculator
6
+ paper (Jack et al. 2015, ACS Synthetic Biology, DOI: 10.1021/acssynbio.5b00068).
7
+ """
8
+
9
+ import numpy as np
10
+ import pandas as pd
11
+ from Levenshtein import distance as levenshtein_distance
12
+
13
+ from eso.detection._overlap import ranges_overlap, range_contains
14
+ from eso.sequence_utils import add_backward_sites, shorten_sequences
15
+
16
+ # matches the columns actually produced by the non-empty path below
17
+ # (sequence_1/sequence_2, not a single 'sequence' column)
18
+ RECOMBINATION_COLUMNS = [
19
+ 'sequence_1', 'start_1', 'end_1', 'sequence_2', 'start_2', 'end_2',
20
+ 'location_delta', 'site_length', 'log10_prob_recombination_ecoli',
21
+ ]
22
+
23
+
24
+ def _generate_all_recombination_sites(seq):
25
+ """Generate candidate 16-17mers (forward + reverse complement) at every offset,
26
+ plus their single-insertion/deletion/substitution-shortened variants, so that
27
+ near-duplicate site pairs can be found by exact match instead of all-pairs comparison.
28
+ """
29
+ forward_insertions = [(seq[ii:ii + 17], ii, ii + 16) for ii in range(len(seq))]
30
+ forward_insertions = [x for x in forward_insertions if len(x[0]) == 17]
31
+
32
+ df_forward_insertions = pd.DataFrame.from_records(
33
+ data=forward_insertions, columns=['sequence', 'start', 'end'])
34
+
35
+ df_insertions = add_backward_sites(df_forward_insertions)
36
+ df_first_sites = shorten_sequences(df_forward_insertions)
37
+ df_substitutions = add_backward_sites(df_first_sites)
38
+
39
+ df_deletions = shorten_sequences(df_first_sites)
40
+ df_deletions = add_backward_sites(df_deletions)
41
+
42
+ return df_first_sites, df_insertions, df_substitutions, df_deletions
43
+
44
+
45
+ def _generate_neighbors(curr_seq):
46
+ """All sequences up to one insertion, deletion, or substitution away from curr_seq."""
47
+ neighbors_substitutions = set()
48
+ neighbors_deletions = set()
49
+ neighbors_insertions = set()
50
+
51
+ for ii in range(len(curr_seq)):
52
+ prefix = curr_seq[:ii]
53
+ suffix = curr_seq[ii:]
54
+ for nt in ['A', 'C', 'G', 'T']:
55
+ neighbors_insertions.add(prefix + nt + suffix)
56
+ suffix = suffix[1:]
57
+ for nt in ['A', 'C', 'G', 'T']:
58
+ neighbors_substitutions.add(prefix + nt + suffix)
59
+ neighbors_deletions.add(prefix + suffix)
60
+
61
+ return neighbors_substitutions, neighbors_deletions, neighbors_insertions
62
+
63
+
64
+ def _order_site12(df):
65
+ lower_second = (df.start_1 > df.start_2)
66
+ df_output = df.copy()
67
+ for col in ['sequence', 'start', 'end']:
68
+ df_output.loc[lower_second, f'{col}_1'] = df[f'{col}_2']
69
+ df_output.loc[lower_second, f'{col}_2'] = df[f'{col}_1']
70
+ return df_output
71
+
72
+
73
+ def _build_sequence_index(df):
74
+ """Map sequence -> list of (start, end) tuples, for O(1) dict lookup instead
75
+ of a `df.sequence.isin(...)` filter (which rescans/reindexes the whole
76
+ frame) called once per candidate site.
77
+ """
78
+ index = {}
79
+ for sequence, start, end in zip(df.sequence, df.start, df.end):
80
+ index.setdefault(sequence, []).append((start, end))
81
+ return index
82
+
83
+
84
+ def _generate_relevant_pairs(df_first_sites, df_insertions, df_substitutions, df_deletions):
85
+ """Profiling showed this function dominates find_recombination_sites'
86
+ runtime (5.1s of 7.0s on an ~830nt sequence), almost entirely in pandas
87
+ overhead: `.isin()` plus the surrounding boolean-mask filtering and
88
+ `.iloc`/`.loc` indexing, called once per candidate site (potentially
89
+ O(n) sites). Building a plain-dict index once up front, and iterating
90
+ the input frames as plain Python tuples via zip() rather than
91
+ `.iloc[ii, col]`, avoids nearly all of that per-call pandas overhead
92
+ while producing identical pairs.
93
+ """
94
+ pairs = []
95
+
96
+ substitutions_index = _build_sequence_index(df_substitutions)
97
+ deletions_index = _build_sequence_index(df_deletions)
98
+ insertions_index = _build_sequence_index(df_insertions)
99
+
100
+ for sequence_1, start_1, end_1 in zip(df_first_sites.sequence, df_first_sites.start, df_first_sites.end):
101
+ neighbors_substitutions, neighbors_deletions, neighbors_insertions = _generate_neighbors(sequence_1)
102
+
103
+ for neighbors, index in (
104
+ (neighbors_substitutions, substitutions_index),
105
+ (neighbors_deletions, deletions_index),
106
+ (neighbors_insertions, insertions_index),
107
+ ):
108
+ for neighbor_seq in neighbors:
109
+ for start_2, end_2 in index.get(neighbor_seq, ()):
110
+ pairs.append((sequence_1, start_1, end_1, neighbor_seq, start_2, end_2))
111
+
112
+ df_pairs = pd.DataFrame.from_records(
113
+ pairs, columns=['sequence_1', 'start_1', 'end_1', 'sequence_2', 'start_2', 'end_2'])
114
+
115
+ df_pairs = _order_site12(df_pairs)
116
+
117
+ # drop overlapping pairs and duplicates
118
+ df_pairs = df_pairs[df_pairs.end_1 < df_pairs.start_2].drop_duplicates().reset_index(drop=True)
119
+
120
+ return df_pairs
121
+
122
+
123
+ def _elongate_sites(row, full_seq):
124
+ """Extend each site pair outward while they remain within Levenshtein distance 1
125
+ and non-overlapping, so the reported site is the full recombination hotspot
126
+ rather than just its detected 16-mer core.
127
+ """
128
+ sequence_1, start_1, end_1 = row.sequence_1, row.start_1, row.end_1 + 1
129
+ sequence_2, start_2, end_2 = row.sequence_2, row.start_2, row.end_2 + 1
130
+
131
+ while (levenshtein_distance(full_seq[start_1:end_1 + 1], full_seq[start_2:end_2 + 1], score_cutoff=1) < 2) \
132
+ and (end_1 + 1 < start_2):
133
+ end_1 += 1
134
+ end_2 += 1
135
+
136
+ while (levenshtein_distance(full_seq[start_1 - 1:end_1], full_seq[start_2 - 1:end_2], score_cutoff=1) < 2) \
137
+ and (end_1 + 1 < start_2):
138
+ start_1 -= 1
139
+ start_2 -= 1
140
+
141
+ row.sequence_1, row.start_1, row.end_1 = full_seq[start_1:end_1], start_1, end_1
142
+ row.sequence_2, row.start_2, row.end_2 = full_seq[start_2:end_2], start_2, end_2
143
+ return row
144
+
145
+
146
+ def calc_recombination_score(location_delta, site_length):
147
+ """Empirical log10 probability of recombination-mediated deletion.
148
+
149
+ Formula and constants per Oliveira et al. 2008 (Plasmid 60:159-165,
150
+ doi:10.1016/j.plasmid.2008.06.004), Eq. (4) and Table 3, recA+ row:
151
+ FR(LR,LS) = (A+LS)^(-a/LR) * LR/(1+B*LR+C*LS), with A=5.8, B=1465.6,
152
+ C=0 (not fitted for recA+), a=29.0 (this module's `alpha`).
153
+
154
+ The EFM Calculator's own source (github.com/barricklab/efm-calculator,
155
+ get_recombo_rate) hardcodes this same B=1465.6 and exponent=29, but pairs
156
+ them with A=8.8 - a value that does not appear anywhere in the recA+ row
157
+ of Table 3. It matches the *recA-* row's exponent (a=8.8 there, a
158
+ different parameter, in a different model, with a different B and an
159
+ extra C term) instead. That looks like a transcription mix-up in the
160
+ reference tool between the two rows of the source table, not a
161
+ correction - so this codebase's original 5.8 (from ESO_curr/STABLES) is
162
+ the value actually supported by the primary source, and the previous
163
+ change to 8.8 here was a mistake, reverted.
164
+ """
165
+ a, b, c, alpha = 5.8, 1465.6, 0, 29
166
+
167
+ first_component = a + location_delta
168
+ second_component = -1 * (alpha / site_length)
169
+ third_component = site_length / (1 + b * site_length + c * location_delta)
170
+
171
+ recombination_probability = (first_component ** second_component) * third_component
172
+ return np.log10(recombination_probability)
173
+
174
+
175
+ def _collapse_pairs_by_predicate(df_pairs, should_drop_range):
176
+ """Shared non-max-suppression walk for site-pairs: process pairs in
177
+ descending score order (ties broken by descending combined span, then a
178
+ stable sort, for the same determinism reason documented on
179
+ eso.detection._overlap._collapse_by_predicate - found via property-based
180
+ testing on that same-shaped single-range logic, not reproduced here as
181
+ its own test but the same tiebreak applied preemptively since the walk
182
+ is structurally identical), dropping a pair exactly when
183
+ `should_drop_range` is True of BOTH its site_1 and site_2 ranges against
184
+ some already-kept pair's corresponding ranges, otherwise keeping it.
185
+ Requiring agreement on both sides (not just one) avoids merging two
186
+ genuinely distinct hotspots that happen to share one site. The two
187
+ functions below differ only in which predicate they pass in (overlap vs.
188
+ full containment).
189
+ """
190
+ kept_rows = []
191
+ kept_ranges = [] # list of ((start_1,end_1), (start_2,end_2))
192
+
193
+ combined_span = (df_pairs.end_1 - df_pairs.start_1) + (df_pairs.end_2 - df_pairs.start_2)
194
+ ordering = df_pairs.assign(_combined_span=combined_span).sort_values(
195
+ ['log10_prob_recombination_ecoli', '_combined_span'], ascending=[False, False], kind='mergesort')
196
+
197
+ for _, row in ordering.iterrows():
198
+ range_1 = (row.start_1, row.end_1)
199
+ range_2 = (row.start_2, row.end_2)
200
+ if any(
201
+ should_drop_range(range_1, r1) and should_drop_range(range_2, r2)
202
+ for r1, r2 in kept_ranges
203
+ ):
204
+ continue
205
+ kept_rows.append(row.drop('_combined_span'))
206
+ kept_ranges.append((range_1, range_2))
207
+
208
+ return pd.DataFrame(kept_rows, columns=df_pairs.columns) if kept_rows else df_pairs.iloc[0:0]
209
+
210
+
211
+ def _collapse_overlapping_pairs(df_pairs):
212
+ """Different seed 16-mers for the same real hotspot converge, via
213
+ elongation, to slightly different (start, end) extents rather than one
214
+ canonical pair - so even exact-coordinate dedup leaves several
215
+ near-identical rows per real site. Collapse them via non-max suppression:
216
+ walk pairs in descending score order, keeping a pair only if BOTH its
217
+ site_1 and site_2 ranges are still free of an already-kept pair's
218
+ corresponding range.
219
+ """
220
+ return _collapse_pairs_by_predicate(df_pairs, should_drop_range=ranges_overlap)
221
+
222
+
223
+ def _collapse_overlapping_pairs_no_coverage_loss(df_pairs):
224
+ """Like _collapse_overlapping_pairs, but only drops a pair when BOTH its
225
+ site_1 and site_2 ranges are FULLY CONTAINED in an already-kept pair's
226
+ corresponding ranges - not merely overlapping them. See
227
+ eso.detection._overlap.collapse_overlapping_intervals_no_coverage_loss
228
+ for why: plain overlap-based NMS can silently drop a genuinely distinct,
229
+ only-partially-overlapping hotspot pair, leaving whatever part of its
230
+ span isn't covered by the kept pair with no correction constraint. Rows
231
+ returned here can legitimately still overlap each other - use this to
232
+ build constraints, never for a "how many distinct sites" count (use
233
+ _collapse_overlapping_pairs for that instead, via collapse_recombination_sites).
234
+ """
235
+ def _kept_fully_covers_current(current_range, kept_range):
236
+ return range_contains(outer=kept_range, inner=current_range)
237
+
238
+ return _collapse_pairs_by_predicate(df_pairs, should_drop_range=_kept_fully_covers_current)
239
+
240
+
241
+ def recombination_sites_for_constraints(df_pairs):
242
+ """Reduce raw recombination candidates (see find_recombination_candidates)
243
+ for feeding into correction-constraint building
244
+ (recombination_to_multiple_avoidance_sites), WITHOUT the coverage-loss
245
+ risk plain non-max suppression (collapse_recombination_sites) has - see
246
+ _collapse_overlapping_pairs_no_coverage_loss and
247
+ eso.detection.slippage.slippage_sites_for_constraints for the concrete
248
+ failure this avoids.
249
+ """
250
+ if df_pairs.shape[0] == 0:
251
+ return df_pairs
252
+ return _collapse_overlapping_pairs_no_coverage_loss(df_pairs)
253
+
254
+
255
+ def find_recombination_candidates(seq):
256
+ """Find every candidate recombination (RMD) site-pair in `seq`, WITHOUT
257
+ collapsing overlapping pairs down to one representative per real hotspot.
258
+
259
+ A pair of sites is a candidate if both are >=16nt, within Levenshtein
260
+ distance 1 of each other, and non-overlapping.
261
+
262
+ This is the shared base both find_recombination_sites (collapsed report
263
+ view) and recombination_sites_for_constraints (constraint-building view)
264
+ are built from.
265
+
266
+ Returns a dataframe sorted by descending mutation risk
267
+ (`log10_prob_recombination_ecoli`), not limited by `num_sites` (that only
268
+ makes sense for the collapsed, human-facing view - see
269
+ find_recombination_sites/collapse_recombination_sites).
270
+ """
271
+ empty_df = pd.DataFrame(columns=RECOMBINATION_COLUMNS)
272
+
273
+ df_first_sites, df_insertions, df_substitutions, df_deletions = _generate_all_recombination_sites(seq)
274
+ df_pairs = _generate_relevant_pairs(df_first_sites, df_insertions, df_substitutions, df_deletions)
275
+
276
+ if df_pairs.shape[0] == 0:
277
+ return empty_df
278
+
279
+ df_pairs = df_pairs.apply(lambda row: _elongate_sites(row, seq), axis=1)
280
+
281
+ df_pairs.loc[:, 'location_delta'] = df_pairs.start_2 - df_pairs.end_1
282
+ df_pairs.loc[:, 'site_length'] = df_pairs.apply(
283
+ lambda row: max(row.end_1 - row.start_1, row.end_2 - row.start_2), axis=1)
284
+
285
+ df_pairs.loc[:, 'log10_prob_recombination_ecoli'] = df_pairs.apply(
286
+ lambda x: calc_recombination_score(x.location_delta, x.site_length), axis=1)
287
+
288
+ df_pairs = df_pairs[df_pairs.log10_prob_recombination_ecoli > -9]
289
+
290
+ for col in ['start_1', 'end_1', 'start_2', 'end_2', 'location_delta', 'site_length']:
291
+ df_pairs.loc[:, col] = df_pairs[col].astype(int)
292
+
293
+ return df_pairs.sort_values('log10_prob_recombination_ecoli', ascending=False).reset_index(drop=True)
294
+
295
+
296
+ def collapse_recombination_sites(df_pairs, num_sites=np.inf):
297
+ """Collapse raw recombination candidates (see find_recombination_candidates)
298
+ down to one representative pair per group of overlapping pairs, for a
299
+ human-facing "distinct sites" report or count - NOT for building
300
+ correction constraints (see recombination_sites_for_constraints instead -
301
+ a dropped pair's uniquely-covered span would otherwise get no fix).
302
+
303
+ The many candidate-seed rows that converge (via elongation) on the same
304
+ or overlapping real hotspot are collapsed down to one representative row
305
+ each.
306
+ """
307
+ if df_pairs.shape[0] == 0:
308
+ return df_pairs
309
+
310
+ df_pairs = _collapse_overlapping_pairs(df_pairs)
311
+ df_pairs = df_pairs.sort_values('log10_prob_recombination_ecoli', ascending=False)
312
+
313
+ if num_sites < np.inf:
314
+ df_pairs = df_pairs.head(int(num_sites))
315
+
316
+ return df_pairs
317
+
318
+
319
+ def find_recombination_sites(seq, num_sites=np.inf):
320
+ """Find recombination (RMD) hotspots in `seq`, collapsed to one
321
+ representative pair per real hotspot (see collapse_recombination_sites)
322
+ and limited to `num_sites` distinct site-pairs if given - the
323
+ human-facing report/count view. Do NOT use this to build correction
324
+ constraints (see find_recombination_candidates).
325
+
326
+ Returns a dataframe sorted by descending mutation risk
327
+ (`log10_prob_recombination_ecoli`).
328
+ """
329
+ return collapse_recombination_sites(find_recombination_candidates(seq), num_sites)
@@ -0,0 +1,249 @@
1
+ """Replication-slippage (SSR - simple sequence repeat) hotspot detection.
2
+
3
+ Finds short tandem repeats (base units of length 1-15 repeated 3+ times, or a
4
+ single nucleotide repeated 12+ times) and scores them with the empirical
5
+ mutation-rate formula from the EFM Calculator paper (Jack et al. 2015, ACS
6
+ Synthetic Biology, DOI: 10.1021/acssynbio.5b00068).
7
+
8
+ The length-1 (homopolymer) minimum is 12, not 3, for two stacked reasons:
9
+
10
+ 1. log10_prob = -12.9 + 0.729*n crosses the -9 filter cutoff (applied below)
11
+ at n=6 - runs of n=4/5 always fail it and are never worth detecting.
12
+ 2. Every homopolymer run of n>=6 is *also* always detected as a length-2 site
13
+ (any 6+ identical characters trivially contain "XX" repeated 3+ times),
14
+ scored as -4.749 + 0.063*floor(n/2) - and collapse_overlapping_intervals
15
+ below only keeps the single highest-scoring representation per physical
16
+ site. Comparing the two formulas: the length-1 score grows ~0.729/nt while
17
+ the length-2 score grows only ~0.0315/nt, so length-2 always wins for
18
+ n=6..11 (e.g. n=8: length-1 -7.068 vs length-2 -4.497) - the length-1
19
+ detection is real but its result is always discarded. The crossover is
20
+ n=12 exactly (length-1 -4.152 vs length-2 -4.371) - only from there does
21
+ the length-1 reading start being the one that survives, correctly
22
+ reflecting a homopolymer run's much higher risk than a linear-in-repeats
23
+ length-2 score would suggest.
24
+
25
+ So detecting length-1 sites below n=12 is provably pure waste - the result
26
+ never changes the final output regardless, since length-2 detection (run
27
+ independently, over the same sequence) always produces a same-or-better-scoring
28
+ overlapping candidate for that region. This doesn't apply to length>1 units:
29
+ log10_prob = -4.749 + 0.063*n is already > -9 at the smallest detectable n=3,
30
+ so every length>1 candidate always passes the -9 filter on its own.
31
+ """
32
+
33
+ import numpy as np
34
+ import pandas as pd
35
+
36
+ from eso.detection._overlap import collapse_overlapping_intervals, collapse_overlapping_intervals_no_coverage_loss
37
+
38
+ SLIPPAGE_COLUMNS = ['start', 'end', 'length_base_unit', 'sequence']
39
+
40
+
41
+ def _find_all(a_str, sub):
42
+ """Non-overlapping occurrences of `sub` in `a_str`."""
43
+ start = 0
44
+ while True:
45
+ start = a_str.find(sub, start)
46
+ if start == -1:
47
+ return
48
+ yield start
49
+ start += len(sub)
50
+
51
+
52
+ def _find_longest_match(seq, subunit, start_index):
53
+ """Given a start location and a repeat unit, find the full extent of the run."""
54
+ curr_seq = seq[start_index:]
55
+ len_sub = len(subunit)
56
+ end_shift = 0
57
+ while curr_seq[:len_sub] == subunit:
58
+ end_shift += len_sub
59
+ curr_seq = curr_seq[len_sub:]
60
+ end_index = start_index + end_shift
61
+ return start_index, end_index, seq[start_index:end_index], len_sub
62
+
63
+
64
+ def _generate_slippage_sites_current_subunit(seq, subunit):
65
+ curr_slippage_sites = []
66
+ # 12 for length-1 units (below that, length-2 detection of the same
67
+ # region always outscores it - see module docstring), 3 for longer units.
68
+ subseq = subunit * (12 if len(subunit) == 1 else 3)
69
+
70
+ indexes_curr = list(_find_all(seq, subseq))
71
+ if not indexes_curr:
72
+ return curr_slippage_sites
73
+
74
+ curr_slippage_sites.append(_find_longest_match(seq, subunit, indexes_curr[0]))
75
+
76
+ for index in indexes_curr[1:]:
77
+ end_previous_site = curr_slippage_sites[-1][1]
78
+ if index <= end_previous_site:
79
+ continue
80
+ curr_slippage_sites.append(_find_longest_match(seq, subunit, index))
81
+
82
+ return curr_slippage_sites
83
+
84
+
85
+ def _find_slippage_len1(seq):
86
+ slippage_sites = []
87
+ for nt in ['A', 'C', 'G', 'T']:
88
+ slippage_sites.extend(_generate_slippage_sites_current_subunit(seq, nt))
89
+ return pd.DataFrame.from_records(
90
+ data=slippage_sites, columns=['start', 'end', 'sequence', 'length_base_unit'],
91
+ ).drop_duplicates().reset_index(drop=True)
92
+
93
+
94
+ def _find_relevant_subunits_len_l(seq, length):
95
+ """Unique substrings of `length` that appear 3+ times in a row, as repeat-unit candidates.
96
+
97
+ Profiling showed the original approach - build every unique substring,
98
+ then call `seq.find(subunit*3)` once per candidate - spends ~93% of total
99
+ find_slippage_sites runtime in `str.find` (88,151 calls on a 9.6kb random
100
+ sequence), since each call rescans the whole sequence: O(unique
101
+ candidates) x O(n) = O(n^2) for a sequence with mostly-unique substrings.
102
+ Building a position dict in one pass and checking adjacency via O(1) set
103
+ membership (positions p, p+length, p+2*length all present means
104
+ `subunit*3` occurs at p) is equivalent but O(n) instead.
105
+ """
106
+ positions_by_subunit = {}
107
+ for ii in range(len(seq) - length + 1):
108
+ positions_by_subunit.setdefault(seq[ii:ii + length], []).append(ii)
109
+
110
+ relevant = []
111
+ for subunit, positions in positions_by_subunit.items():
112
+ position_set = set(positions)
113
+ if any((p + length) in position_set and (p + 2 * length) in position_set for p in positions):
114
+ relevant.append(subunit)
115
+
116
+ return sorted(relevant)
117
+
118
+
119
+ def _find_slippage_len_l(seq, length):
120
+ """Note: 1 < length < 16."""
121
+ relevant_subunits = _find_relevant_subunits_len_l(seq, length)
122
+ slippage_sites = []
123
+ for subunit in relevant_subunits:
124
+ slippage_sites.extend(_generate_slippage_sites_current_subunit(seq, subunit))
125
+ return pd.DataFrame.from_records(data=slippage_sites, columns=['start', 'end', 'sequence', 'length_base_unit'])
126
+
127
+
128
+ def find_slippage_candidates(seq):
129
+ """Find every candidate slippage (SSR) hotspot in `seq` across repeat-unit
130
+ lengths 1-15, WITHOUT collapsing overlapping candidates down to one
131
+ representative per physical site.
132
+
133
+ This is the shared base both find_slippage_sites (collapsed report view,
134
+ via collapse_slippage_sites) and slippage_sites_for_constraints
135
+ (constraint-building view) are built from - see the latter for why
136
+ building constraints needs something other than plain non-max
137
+ suppression.
138
+
139
+ Returns a dataframe sorted by descending mutation risk
140
+ (`log10_prob_slippage_ecoli`), not limited by `num_sites` (that only
141
+ makes sense for the collapsed, human-facing view - see
142
+ find_slippage_sites/collapse_slippage_sites).
143
+ """
144
+ slippage_sites_list = [_find_slippage_len1(seq)]
145
+ for length in range(2, 16):
146
+ slippage_sites_list.append(_find_slippage_len_l(seq, length))
147
+ df_slippage = pd.concat(slippage_sites_list, ignore_index=True)[SLIPPAGE_COLUMNS]
148
+
149
+ df_slippage.loc[:, 'num_base_units'] = (
150
+ df_slippage.sequence.apply(len) / df_slippage.length_base_unit
151
+ ).astype(int)
152
+
153
+ df_slippage.loc[:, 'log10_prob_slippage_ecoli'] = -4.749 + 0.063 * df_slippage['num_base_units']
154
+ df_slippage.loc[df_slippage.length_base_unit == 1, 'log10_prob_slippage_ecoli'] = (
155
+ -12.9 + 0.729 * df_slippage['num_base_units']
156
+ )
157
+
158
+ df_slippage = df_slippage[df_slippage.log10_prob_slippage_ecoli > -9]
159
+ return df_slippage.sort_values('log10_prob_slippage_ecoli', ascending=False).reset_index(drop=True)
160
+
161
+
162
+ def slippage_sites_for_constraints(df_slippage):
163
+ """Reduce raw slippage candidates (see find_slippage_candidates) for
164
+ feeding into correction-constraint building (modify_df_slippage), WITHOUT
165
+ the coverage-loss risk plain non-max suppression (collapse_slippage_sites)
166
+ has.
167
+
168
+ Regular NMS (`collapse_overlapping_intervals`, used by
169
+ collapse_slippage_sites) only requires two candidates' ranges to
170
+ *overlap*, not that one *contain* the other, before dropping the
171
+ lower-scoring one - so two genuinely distinct, only-partially-overlapping
172
+ hotspots can have the lower-scoring one dropped entirely, silently
173
+ leaving the part of its span the higher-scoring one doesn't cover with no
174
+ correction constraint at all. Concrete case: a length-2 site over
175
+ [0, 10) and a length-3 site over [8, 20) overlap only at [8, 10) -
176
+ collapsing keeps whichever scores higher and drops the other completely,
177
+ so positions [10, 20) (if the length-2 site wins) would get no
178
+ constraint, even though the length-3 site was a real,
179
+ independently-detected hotspot. See docs/detector-comparisons.md.
180
+
181
+ This uses `collapse_overlapping_intervals_no_coverage_loss` instead: a
182
+ candidate is only dropped if its ENTIRE range is already covered by a
183
+ higher-scoring kept candidate - the common case (several detections
184
+ converging on essentially the same physical site) still collapses down
185
+ to one row as before, but a candidate sticking out beyond every
186
+ higher-scoring one it overlaps is kept, so every real hotspot still gets
187
+ a constraint covering its full extent, not just whatever a single
188
+ NMS survivor happened to cover.
189
+ """
190
+ return collapse_overlapping_intervals_no_coverage_loss(df_slippage, score_col='log10_prob_slippage_ecoli')
191
+
192
+
193
+ def collapse_slippage_sites(df_slippage, num_sites=np.inf):
194
+ """Collapse raw slippage candidates (see find_slippage_candidates) down to
195
+ one representative per group of overlapping candidates, for a
196
+ human-facing "distinct sites" report or count - NOT for building
197
+ correction constraints (see slippage_sites_for_constraints instead - a
198
+ dropped candidate's uniquely-covered span would otherwise get no fix).
199
+
200
+ different base-unit lengths (and different phase offsets within the same
201
+ length) can each detect the same physical repeat as a separate row -
202
+ e.g. "GCGCGCGC" is a valid length-2 run starting at position N, AND its
203
+ 1-shifted substring "CGCGCG" is a separate valid length-2 run starting
204
+ at N+1. Collapsing by exact 'start' alone (an earlier approach) misses
205
+ this, since the rows don't share a start position. Keep one
206
+ representative (highest scoring) per group of overlapping ranges instead.
207
+ """
208
+ df_slippage = collapse_overlapping_intervals(df_slippage, score_col='log10_prob_slippage_ecoli')
209
+ df_slippage = df_slippage.sort_values(['log10_prob_slippage_ecoli', 'length_base_unit'], ascending=[False, False])
210
+
211
+ if num_sites < np.inf:
212
+ df_slippage = df_slippage.head(int(num_sites))
213
+
214
+ return df_slippage
215
+
216
+
217
+ def find_slippage_sites(seq, num_sites=np.inf):
218
+ """Find slippage (SSR) hotspots in `seq`, collapsed to one representative
219
+ per physical site (see collapse_slippage_sites) and limited to
220
+ `num_sites` if given - the human-facing report/count view. Do NOT use
221
+ this to build correction constraints (see slippage_sites_for_constraints).
222
+
223
+ Returns a dataframe sorted by descending mutation risk
224
+ (`log10_prob_slippage_ecoli`).
225
+ """
226
+ return collapse_slippage_sites(find_slippage_candidates(seq), num_sites)
227
+
228
+
229
+ def modify_df_slippage(df_slippage):
230
+ """Convert each slippage-site row (N repeated base units) into ~N/2 rows of
231
+ individual base units to avoid (skipping every other one) - enough to
232
+ disrupt the repeat without necessarily eliminating it entirely.
233
+
234
+ Example: a row for "TGTGTGTGTG" (base unit "TG", num_base_units=5) becomes
235
+ 3 rows, each a single "TG" occurrence and its coordinates.
236
+ """
237
+ df_list = []
238
+
239
+ for idx in df_slippage.index:
240
+ num_base_units = df_slippage.loc[idx].num_base_units
241
+ length = df_slippage.loc[idx].length_base_unit
242
+ for i in range(0, (num_base_units - 1), 2):
243
+ df_list.append({
244
+ 'sequence': df_slippage.loc[idx].sequence[int(i * length):int((i + 1) * length)],
245
+ 'start': int(df_slippage.loc[idx].start + i * length),
246
+ 'end': int(df_slippage.loc[idx].start + (i + 1) * length),
247
+ })
248
+
249
+ return pd.DataFrame(df_list)