evolutionary-stability-optimizer 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- eso/__init__.py +11 -0
- eso/cli.py +111 -0
- eso/codon_usage.py +96 -0
- eso/constraints.py +179 -0
- eso/custom_score.py +170 -0
- eso/data/__init__.py +0 -0
- eso/data/human-antibody-heavy-chain-codon-frequencies.csv +62 -0
- eso/data/human-antibody-light-chain-codon-frequencies.csv +62 -0
- eso/detection/__init__.py +0 -0
- eso/detection/_overlap.py +112 -0
- eso/detection/common_motifs.py +81 -0
- eso/detection/dispatch.py +185 -0
- eso/detection/methylation.py +96 -0
- eso/detection/motif_utils.py +77 -0
- eso/detection/recombination.py +329 -0
- eso/detection/slippage.py +249 -0
- eso/detection/staubility_variant.py +284 -0
- eso/io_utils.py +312 -0
- eso/optimize.py +266 -0
- eso/pipeline.py +322 -0
- eso/report.py +57 -0
- eso/sequence_utils.py +41 -0
- evolutionary_stability_optimizer-0.1.0.dist-info/METADATA +501 -0
- evolutionary_stability_optimizer-0.1.0.dist-info/RECORD +27 -0
- evolutionary_stability_optimizer-0.1.0.dist-info/WHEEL +4 -0
- evolutionary_stability_optimizer-0.1.0.dist-info/entry_points.txt +3 -0
- evolutionary_stability_optimizer-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
"codon","aa","freq_within_aa"
|
|
2
|
+
"AAA","K",0.5038634756949489
|
|
3
|
+
"AAC","N",0.5718761146593684
|
|
4
|
+
"AAG","K",0.4961365243050511
|
|
5
|
+
"AAT","N",0.4281238853406316
|
|
6
|
+
"ACA","T",0.13097182238346933
|
|
7
|
+
"ACC","T",0.47899857499742254
|
|
8
|
+
"ACG","T",0.09268272481419175
|
|
9
|
+
"ACT","T",0.29734687780491637
|
|
10
|
+
"AGA","R",0.1876054334663075
|
|
11
|
+
"AGC","S",0.19444975284400967
|
|
12
|
+
"AGG","R",0.3922372498119939
|
|
13
|
+
"AGT","S",0.20454951237884456
|
|
14
|
+
"ATA","I",0.03868630942512579
|
|
15
|
+
"ATC","I",0.7397642158016438
|
|
16
|
+
"ATG","M",1
|
|
17
|
+
"ATT","I",0.22154947477323048
|
|
18
|
+
"CAA","Q",0.19296824449591132
|
|
19
|
+
"CAC","H",0.6570207717474473
|
|
20
|
+
"CAG","Q",0.8070317555040887
|
|
21
|
+
"CAT","H",0.3429792282525527
|
|
22
|
+
"CCA","P",0.33448630048322064
|
|
23
|
+
"CCC","P",0.24284536707540752
|
|
24
|
+
"CCG","P",0.07291725166880268
|
|
25
|
+
"CCT","P",0.34975108077256917
|
|
26
|
+
"CGA","R",0.0842652823706148
|
|
27
|
+
"CGC","R",0.09752660203720129
|
|
28
|
+
"CGG","R",0.20034523246364142
|
|
29
|
+
"CGT","R",0.03802019985024105
|
|
30
|
+
"CTA","L",0.05924441120383728
|
|
31
|
+
"CTC","L",0.3665813872174048
|
|
32
|
+
"CTG","L",0.35403072244441774
|
|
33
|
+
"CTT","L",0.039708080776196295
|
|
34
|
+
"GAA","E",0.4260872689539162
|
|
35
|
+
"GAC","D",0.45230366253507515
|
|
36
|
+
"GAG","E",0.5739127310460839
|
|
37
|
+
"GAT","D",0.5476963374649249
|
|
38
|
+
"GCA","A",0.2191246284363758
|
|
39
|
+
"GCC","A",0.5061087270479293
|
|
40
|
+
"GCG","A",0.05734268094352331
|
|
41
|
+
"GCT","A",0.21742396357217159
|
|
42
|
+
"GGA","G",0.1959839346229272
|
|
43
|
+
"GGC","G",0.32278997536379045
|
|
44
|
+
"GGG","G",0.38173182752721285
|
|
45
|
+
"GGT","G",0.09949426248606953
|
|
46
|
+
"GTA","V",0.07877585325542301
|
|
47
|
+
"GTC","V",0.4179341238484603
|
|
48
|
+
"GTG","V",0.33411562644079185
|
|
49
|
+
"GTT","V",0.16917439645532487
|
|
50
|
+
"TAC","Y",0.39897973273328724
|
|
51
|
+
"TAT","Y",0.6010202672667128
|
|
52
|
+
"TCA","S",0.08820172572782266
|
|
53
|
+
"TCC","S",0.20245309080811363
|
|
54
|
+
"TCG","S",0.022302532086794738
|
|
55
|
+
"TCT","S",0.2880433861544147
|
|
56
|
+
"TGC","C",0.6363459475159284
|
|
57
|
+
"TGG","W",1
|
|
58
|
+
"TGT","C",0.3636540524840716
|
|
59
|
+
"TTA","L",0.07132527815882145
|
|
60
|
+
"TTC","F",0.7559554651014863
|
|
61
|
+
"TTG","L",0.10911012019932245
|
|
62
|
+
"TTT","F",0.24404453489851372
|
|
File without changes
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
"""Shared non-max-suppression helper: several detectors (independently, in both
|
|
2
|
+
the primary and STABLES-derived implementations) generate multiple candidate
|
|
3
|
+
rows - at different phases, frames, or seed positions - that describe the same
|
|
4
|
+
real hotspot rather than genuinely distinct ones. This collapses a dataframe
|
|
5
|
+
of scored, ranged candidates down to one representative per group of
|
|
6
|
+
mutually-overlapping candidates.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import pandas as pd
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def ranges_overlap(a, b):
|
|
13
|
+
"""True if ranges (start, end) overlap, using the same exclusive-end
|
|
14
|
+
convention as Python slicing (seq[start:end]) - so two ranges that merely
|
|
15
|
+
touch at a shared boundary (e.g. (0, 950) and (950, 1250)) do NOT count
|
|
16
|
+
as overlapping, since they share no actual character position. A
|
|
17
|
+
previous `<=` version treated touching-but-adjacent ranges as
|
|
18
|
+
overlapping, which silently discarded a genuinely distinct, adjacent
|
|
19
|
+
hotspot whenever it happened to sit immediately next to a higher-scoring
|
|
20
|
+
one - found via a chunk-boundary stress test for an unrelated prototype,
|
|
21
|
+
reproduced with two homopolymer runs separated by a third with no gap.
|
|
22
|
+
"""
|
|
23
|
+
return a[0] < b[1] and b[0] < a[1]
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def range_contains(outer, inner):
|
|
27
|
+
"""True if `inner` (start, end) is fully contained in `outer` (start, end),
|
|
28
|
+
exclusive-end throughout (matches ranges_overlap's convention).
|
|
29
|
+
"""
|
|
30
|
+
return outer[0] <= inner[0] and inner[1] <= outer[1]
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _collapse_by_predicate(df, score_col, start_col, end_col, should_drop):
|
|
34
|
+
"""Shared non-max-suppression walk: process rows in descending `score_col`
|
|
35
|
+
order (ties broken by descending span, then a stable sort, so tied rows
|
|
36
|
+
process in a fixed, reproducible order - see below), dropping a row
|
|
37
|
+
exactly when `should_drop(current_range, kept_range)` is True for some
|
|
38
|
+
already-kept range, otherwise keeping it. The two public collapse
|
|
39
|
+
functions below differ only in which predicate they pass in - everything
|
|
40
|
+
else (the sort, the kept_rows/kept_ranges bookkeeping, the empty-
|
|
41
|
+
dataframe fallback) is identical between them, so it lives here once.
|
|
42
|
+
|
|
43
|
+
The span tiebreak was added after property-based testing
|
|
44
|
+
(tests/test_detection_overlap_properties.py) found that two same-scored
|
|
45
|
+
candidates, one fully containing the other, could both survive or either
|
|
46
|
+
one survive depending purely on which happened to appear first in `df` -
|
|
47
|
+
score-only sorting doesn't order tied rows at all, so it fell back to
|
|
48
|
+
whatever incidental order the input arrived in, making the result
|
|
49
|
+
non-deterministic across runs/pandas versions for that (rare but real)
|
|
50
|
+
case. Processing the larger span first among ties makes the outcome
|
|
51
|
+
reproducible, and additionally means a fully-redundant smaller duplicate
|
|
52
|
+
gets dropped rather than kept alongside it - but this only resolves EXACT
|
|
53
|
+
ties: a smaller, genuinely higher-scoring row and a larger, lower-scoring
|
|
54
|
+
one that doesn't fully cover it can both legitimately survive regardless
|
|
55
|
+
of this tiebreak (each may represent a structurally different candidate
|
|
56
|
+
that just happens to share coordinates) - that's correct, not a bug.
|
|
57
|
+
"""
|
|
58
|
+
kept_rows = []
|
|
59
|
+
kept_ranges = []
|
|
60
|
+
|
|
61
|
+
ordering = df.assign(_span=df[end_col] - df[start_col]).sort_values(
|
|
62
|
+
[score_col, '_span'], ascending=[False, False], kind='mergesort')
|
|
63
|
+
|
|
64
|
+
for _, row in ordering.iterrows():
|
|
65
|
+
current_range = (row[start_col], row[end_col])
|
|
66
|
+
if any(should_drop(current_range, kept_range) for kept_range in kept_ranges):
|
|
67
|
+
continue
|
|
68
|
+
kept_rows.append(row.drop('_span'))
|
|
69
|
+
kept_ranges.append(current_range)
|
|
70
|
+
|
|
71
|
+
return pd.DataFrame(kept_rows, columns=df.columns) if kept_rows else df.iloc[0:0]
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def collapse_overlapping_intervals(df, score_col, start_col='start', end_col='end'):
|
|
75
|
+
"""Walk rows in descending `score_col` order, keeping a row only if its
|
|
76
|
+
[start_col, end_col] range doesn't overlap an already-kept row's range.
|
|
77
|
+
|
|
78
|
+
This is non-max suppression: it only requires ranges to *overlap*, not
|
|
79
|
+
that one *contain* the other, to drop the lower-scoring one. For a
|
|
80
|
+
"distinct sites" report or count that's the right call - but it means a
|
|
81
|
+
row can be dropped even though part of its range isn't actually covered
|
|
82
|
+
by anything kept, silently losing coverage of that part entirely. Do NOT
|
|
83
|
+
use this to build correction constraints for that reason - use
|
|
84
|
+
collapse_overlapping_intervals_no_coverage_loss instead. See
|
|
85
|
+
docs/detector-comparisons.md for the concrete failure this caused.
|
|
86
|
+
"""
|
|
87
|
+
return _collapse_by_predicate(df, score_col, start_col, end_col, should_drop=ranges_overlap)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def collapse_overlapping_intervals_no_coverage_loss(df, score_col, start_col='start', end_col='end'):
|
|
91
|
+
"""Like collapse_overlapping_intervals, but only drops a row when its
|
|
92
|
+
range is FULLY CONTAINED in an already-kept row's range - not merely
|
|
93
|
+
overlapping it.
|
|
94
|
+
|
|
95
|
+
Walking in descending `score_col` order, a redundant candidate (one whose
|
|
96
|
+
entire range is already covered by a higher-scoring kept row - the common
|
|
97
|
+
case: several detections of the same physical site converging on
|
|
98
|
+
essentially the same extent) is still dropped, exactly as before. But a
|
|
99
|
+
candidate that only *partially* overlaps a kept row - some real,
|
|
100
|
+
independently-detected hotspot whose span isn't a subset of the kept
|
|
101
|
+
row's - is kept too, so nothing downstream ever has to build a correction
|
|
102
|
+
constraint for a region no surviving row actually covers.
|
|
103
|
+
|
|
104
|
+
Rows returned by this function can legitimately still overlap each other
|
|
105
|
+
(that's the point) - use it to feed a correction/constraint-building step,
|
|
106
|
+
never for a "how many distinct sites" count or report (use
|
|
107
|
+
collapse_overlapping_intervals for that instead).
|
|
108
|
+
"""
|
|
109
|
+
def _kept_fully_covers_current(current_range, kept_range):
|
|
110
|
+
return range_contains(outer=kept_range, inner=current_range)
|
|
111
|
+
|
|
112
|
+
return _collapse_by_predicate(df, score_col, start_col, end_col, should_drop=_kept_fully_covers_current)
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
"""A small, ready-to-use library of commonly-referenced DNA motifs worth
|
|
2
|
+
checking for/avoiding when designing an engineered sequence - not just
|
|
3
|
+
methylation, but a few other properties that turn up across the synthetic
|
|
4
|
+
biology literature.
|
|
5
|
+
|
|
6
|
+
**Methylation** (E. coli's two systems, since E. coli is already a
|
|
7
|
+
first-class host here - see eso.codon_usage's bundled "e_coli" support):
|
|
8
|
+
- `dam`: GATC, N6-methyladenine (essentially universal across E. coli
|
|
9
|
+
strains and many other Gammaproteobacteria).
|
|
10
|
+
- `dcm`: CCWGG (W = A or T), C5-methylcytosine on the internal C.
|
|
11
|
+
Sources: NEB, "Dam and Dcm Methylases of E. coli"; EcoSal Plus, "DNA
|
|
12
|
+
Methylation" (doi.org/10.1128/ecosalplus.esp-0003-2013).
|
|
13
|
+
|
|
14
|
+
**Cryptic ribosome binding**:
|
|
15
|
+
- `shine_dalgarno`: AGGAGG, the canonical bacterial ribosome-binding-site
|
|
16
|
+
consensus. A copy of this sequence occurring *inside* a coding region
|
|
17
|
+
(not at an intended start codon) is a known source of unintended internal
|
|
18
|
+
translation initiation.
|
|
19
|
+
Source: Shine-Dalgarno sequences are measurably depleted from within
|
|
20
|
+
bacterial coding sequences, consistent with selection against this exact
|
|
21
|
+
risk (see e.g. Mol. Biol. Evol. 35(10):2487, and PMC6107199).
|
|
22
|
+
|
|
23
|
+
**Cryptic bacterial (sigma70) promoter elements**:
|
|
24
|
+
- `sigma70_minus35`: TTGACA, the -35 hexamer consensus.
|
|
25
|
+
- `sigma70_minus10`: TATAAT, the -10 ("Pribnow box") hexamer consensus.
|
|
26
|
+
These are the two core hexamers recognized by E. coli's housekeeping sigma
|
|
27
|
+
factor; an accidental occurrence of either inside a coding sequence is a
|
|
28
|
+
textbook source of unwanted "cryptic" transcription. **Caveat**: a real
|
|
29
|
+
sigma70 promoter needs BOTH hexamers at roughly the right spacing
|
|
30
|
+
(~17±1bp apart) - this module flags each hexamer independently (this
|
|
31
|
+
detector has no concept of "two motifs at a specific spacing"), so an
|
|
32
|
+
isolated hit is much weaker evidence than a "both boxes, correctly spaced"
|
|
33
|
+
finding would be. Treat isolated hits as a coarse, conservative screen, not
|
|
34
|
+
a confirmed cryptic promoter.
|
|
35
|
+
|
|
36
|
+
**Not included, and why**: transcription terminators (rho-independent
|
|
37
|
+
terminators are a hairpin + poly-U structure - a secondary-structure
|
|
38
|
+
property, not a fixed linear sequence motif this PSSM-based approach can
|
|
39
|
+
represent) and the eukaryotic Kozak sequence (a *desired* translation-
|
|
40
|
+
initiation context to match near a real start codon, not something to
|
|
41
|
+
avoid - a different problem from what this detector/module is for).
|
|
42
|
+
Restriction enzyme sites are also not duplicated here - DNAChisel (already
|
|
43
|
+
a dependency of this project) already bundles a comprehensive registry via
|
|
44
|
+
`dnachisel.list_common_enzymes()` / `dnachisel.EnzymeSitePattern`, usable
|
|
45
|
+
directly with `AvoidPattern` during optimization.
|
|
46
|
+
|
|
47
|
+
For anything beyond what's bundled here, REBASE (rebase.neb.com) is the
|
|
48
|
+
standard reference database for restriction/methylation motifs across
|
|
49
|
+
organisms - use `eso.detection.motif_utils.motif_from_consensus` to turn
|
|
50
|
+
any REBASE-style (or other literature) consensus sequence into a usable
|
|
51
|
+
motif, the same way everything in this module is built.
|
|
52
|
+
"""
|
|
53
|
+
|
|
54
|
+
from eso.detection.motif_utils import motif_from_consensus
|
|
55
|
+
|
|
56
|
+
#: name -> IUPAC consensus sequence
|
|
57
|
+
COMMON_MOTIFS = {
|
|
58
|
+
"dam": "GATC",
|
|
59
|
+
"dcm": "CCWGG",
|
|
60
|
+
"shine_dalgarno": "AGGAGG",
|
|
61
|
+
"sigma70_minus35": "TTGACA",
|
|
62
|
+
"sigma70_minus10": "TATAAT",
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def load_common_motifs(names=None):
|
|
67
|
+
"""Return a list of Bio.motifs Motif objects for the requested common
|
|
68
|
+
motifs (default: all of them - see COMMON_MOTIFS for the full list).
|
|
69
|
+
|
|
70
|
+
`names` is case-insensitive; see COMMON_MOTIFS for the available keys.
|
|
71
|
+
"""
|
|
72
|
+
if names is None:
|
|
73
|
+
names = list(COMMON_MOTIFS)
|
|
74
|
+
|
|
75
|
+
motifs_list = []
|
|
76
|
+
for name in names:
|
|
77
|
+
key = name.strip().lower()
|
|
78
|
+
if key not in COMMON_MOTIFS:
|
|
79
|
+
raise ValueError(f"Unknown common motif {name!r}; choose from {sorted(COMMON_MOTIFS)}")
|
|
80
|
+
motifs_list.append(motif_from_consensus(key, COMMON_MOTIFS[key]))
|
|
81
|
+
return motifs_list
|
|
@@ -0,0 +1,185 @@
|
|
|
1
|
+
"""Mode-based routing across independently-developed detector implementations,
|
|
2
|
+
so callers choose a tradeoff without needing to know which module originally
|
|
3
|
+
implemented which algorithm.
|
|
4
|
+
|
|
5
|
+
Covers recombination and slippage detection; methylation has only one
|
|
6
|
+
implementation (eso.detection.methylation) - a second was built and compared
|
|
7
|
+
here, but removed after it was found to disagree in accuracy (not just
|
|
8
|
+
speed) with the first - see docs/detector-comparisons.md.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
import numpy as np
|
|
12
|
+
|
|
13
|
+
from eso.detection import recombination, slippage, staubility_variant
|
|
14
|
+
|
|
15
|
+
# Each category (recombination/slippage) is dispatched at three stages -
|
|
16
|
+
# find the raw candidates, collapse them to a report view, and reduce them
|
|
17
|
+
# to a constraint-building view (see eso.detection.recombination's module
|
|
18
|
+
# docstring split for what each stage means) - across the same two
|
|
19
|
+
# implementations ("thorough"/"fast" or "default"/"fast"). That's 6 mode
|
|
20
|
+
# tables in total; `_dispatch` below is the one place that turns a mode
|
|
21
|
+
# table + a chosen mode into either a result or a friendly ValueError, so a
|
|
22
|
+
# bug fix or message tweak to that logic only has to happen once.
|
|
23
|
+
|
|
24
|
+
RECOMBINATION_MODES = {
|
|
25
|
+
"thorough": recombination.find_recombination_sites,
|
|
26
|
+
"fast": staubility_variant.find_recombination_sites,
|
|
27
|
+
}
|
|
28
|
+
RECOMBINATION_CANDIDATE_MODES = {
|
|
29
|
+
"thorough": recombination.find_recombination_candidates,
|
|
30
|
+
"fast": staubility_variant.find_recombination_candidates,
|
|
31
|
+
}
|
|
32
|
+
RECOMBINATION_COLLAPSE_MODES = {
|
|
33
|
+
"thorough": recombination.collapse_recombination_sites,
|
|
34
|
+
"fast": staubility_variant.collapse_recombination_sites,
|
|
35
|
+
}
|
|
36
|
+
RECOMBINATION_FOR_CONSTRAINTS_MODES = {
|
|
37
|
+
"thorough": recombination.recombination_sites_for_constraints,
|
|
38
|
+
"fast": staubility_variant.recombination_sites_for_constraints,
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
SLIPPAGE_MODES = {
|
|
42
|
+
"default": slippage.find_slippage_sites,
|
|
43
|
+
"fast": staubility_variant.find_slippage_sites,
|
|
44
|
+
}
|
|
45
|
+
SLIPPAGE_CANDIDATE_MODES = {
|
|
46
|
+
"default": slippage.find_slippage_candidates,
|
|
47
|
+
"fast": staubility_variant.find_slippage_candidates,
|
|
48
|
+
}
|
|
49
|
+
SLIPPAGE_COLLAPSE_MODES = {
|
|
50
|
+
"default": slippage.collapse_slippage_sites,
|
|
51
|
+
"fast": staubility_variant.collapse_slippage_sites,
|
|
52
|
+
}
|
|
53
|
+
SLIPPAGE_FOR_CONSTRAINTS_MODES = {
|
|
54
|
+
"default": slippage.slippage_sites_for_constraints,
|
|
55
|
+
"fast": staubility_variant.slippage_sites_for_constraints,
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _dispatch(modes, mode, category_label, *args):
|
|
60
|
+
"""Look up `mode` in `modes` and call it with `args`, or raise a
|
|
61
|
+
ValueError naming `category_label` ("recombination"/"slippage") and
|
|
62
|
+
listing the modes actually available - the one error-message format
|
|
63
|
+
every public function below shares.
|
|
64
|
+
"""
|
|
65
|
+
try:
|
|
66
|
+
implementation = modes[mode]
|
|
67
|
+
except KeyError:
|
|
68
|
+
raise ValueError(f"Unknown {category_label} mode {mode!r}; choose from {sorted(modes)}")
|
|
69
|
+
return implementation(*args)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def find_recombination_sites(seq, num_sites=np.inf, mode="thorough"):
|
|
73
|
+
"""Detect recombination (RMD) hotspots, routed to one of two independently
|
|
74
|
+
developed implementations.
|
|
75
|
+
|
|
76
|
+
mode="thorough" (default) - eso.detection.recombination: Levenshtein-tolerant,
|
|
77
|
+
catches pairs of sites within edit distance 1 of each other, not just
|
|
78
|
+
exact duplicates. Benchmarked as roughly linear from 51,400nt through
|
|
79
|
+
1,000,000nt (~126.7s at 1,000,000nt, in local benchmarks after fixing
|
|
80
|
+
a pandas-overhead bottleneck - see docs/detector-comparisons.md), with
|
|
81
|
+
no breakdown point found at any tested scale - recommended by default
|
|
82
|
+
at essentially any realistic sequence length.
|
|
83
|
+
|
|
84
|
+
mode="fast" - eso.detection.staubility_variant: exact 16-mer match only,
|
|
85
|
+
via vectorized n-gram counting. Will miss a near-duplicate whenever
|
|
86
|
+
its point of divergence sits centrally enough that no 16-consecutive-nt
|
|
87
|
+
exact window survives on either side (verified: catches a duplicate
|
|
88
|
+
with a 1nt substitution near either edge, since 16+nt of exact match
|
|
89
|
+
remains; misses the same case when the substitution is centered).
|
|
90
|
+
19-34x faster than "thorough" at every length tested - reach for this
|
|
91
|
+
only when that speed gap itself matters (e.g. many-sequence batch
|
|
92
|
+
workloads), not because "thorough" becomes intractable.
|
|
93
|
+
"""
|
|
94
|
+
return _dispatch(RECOMBINATION_MODES, mode, "recombination", seq, num_sites)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def find_slippage_sites(seq, num_sites=np.inf, mode="default"):
|
|
98
|
+
"""Detect slippage (SSR) hotspots, routed to one of two independently
|
|
99
|
+
developed implementations.
|
|
100
|
+
|
|
101
|
+
Unlike recombination, both modes detect exactly the same hotspots -
|
|
102
|
+
verified via 300 randomized trials with zero sensitivity or row-count
|
|
103
|
+
mismatches after fixing bugs in both implementations (see
|
|
104
|
+
docs/detector-comparisons.md). This is purely a speed choice, and
|
|
105
|
+
"default" wins it outright:
|
|
106
|
+
|
|
107
|
+
mode="default" - eso.detection.slippage: after fixing an O(n^2) candidate-
|
|
108
|
+
scan (see docs/detector-comparisons.md), this is faster than "fast"
|
|
109
|
+
at every length tested, from a few hundred nt through 300,000nt,
|
|
110
|
+
with the gap widening as length grows (10x faster at 300kb).
|
|
111
|
+
|
|
112
|
+
mode="fast" - eso.detection.staubility_variant: kept as an independent
|
|
113
|
+
second implementation (useful as a cross-check, and it's a distinct
|
|
114
|
+
algorithm, not just a slower copy) - but there is no longer a length
|
|
115
|
+
range where it's actually faster than "default".
|
|
116
|
+
"""
|
|
117
|
+
return _dispatch(SLIPPAGE_MODES, mode, "slippage", seq, num_sites)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def find_recombination_candidates(seq, mode="thorough"):
|
|
121
|
+
"""Every candidate recombination (RMD) site-pair, WITHOUT collapsing
|
|
122
|
+
overlapping pairs down to one representative per real hotspot - routed to
|
|
123
|
+
one of the two implementations documented on find_recombination_sites.
|
|
124
|
+
|
|
125
|
+
This is the shared base find_recombination_sites (report view, via
|
|
126
|
+
collapse_recombination_sites) and recombination_sites_for_constraints
|
|
127
|
+
(constraint-building view) are both derived from - use whichever of
|
|
128
|
+
those two fits, not this function directly, unless you specifically want
|
|
129
|
+
every raw candidate with nothing reduced at all.
|
|
130
|
+
"""
|
|
131
|
+
return _dispatch(RECOMBINATION_CANDIDATE_MODES, mode, "recombination", seq)
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def find_slippage_candidates(seq, mode="default"):
|
|
135
|
+
"""Every candidate slippage (SSR) hotspot, WITHOUT collapsing overlapping
|
|
136
|
+
candidates down to one representative per physical site - routed to one
|
|
137
|
+
of the two implementations documented on find_slippage_sites.
|
|
138
|
+
|
|
139
|
+
This is the shared base find_slippage_sites (report view, via
|
|
140
|
+
collapse_slippage_sites) and slippage_sites_for_constraints
|
|
141
|
+
(constraint-building view) are both derived from - use whichever of
|
|
142
|
+
those two fits, not this function directly, unless you specifically want
|
|
143
|
+
every raw candidate with nothing reduced at all.
|
|
144
|
+
"""
|
|
145
|
+
return _dispatch(SLIPPAGE_CANDIDATE_MODES, mode, "slippage", seq)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def collapse_recombination_sites(df_pairs, num_sites=np.inf, mode="thorough"):
|
|
149
|
+
"""Collapse a raw recombination-candidates dataframe (from
|
|
150
|
+
find_recombination_candidates, same `mode`) down to the human-facing
|
|
151
|
+
"distinct sites" view - one representative pair per real hotspot, limited
|
|
152
|
+
to `num_sites`. Lets a caller run detection once and derive both the raw
|
|
153
|
+
(constraint-building) and collapsed (report) views without re-detecting.
|
|
154
|
+
"""
|
|
155
|
+
return _dispatch(RECOMBINATION_COLLAPSE_MODES, mode, "recombination", df_pairs, num_sites)
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def collapse_slippage_sites(df_slippage, num_sites=np.inf, mode="default"):
|
|
159
|
+
"""Collapse a raw slippage-candidates dataframe (from
|
|
160
|
+
find_slippage_candidates, same `mode`) down to the human-facing "distinct
|
|
161
|
+
sites" view - one representative per physical site, limited to
|
|
162
|
+
`num_sites`. Lets a caller run detection once and derive both the raw
|
|
163
|
+
(constraint-building) and collapsed (report) views without re-detecting.
|
|
164
|
+
"""
|
|
165
|
+
return _dispatch(SLIPPAGE_COLLAPSE_MODES, mode, "slippage", df_slippage, num_sites)
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def recombination_sites_for_constraints(df_pairs, mode="thorough"):
|
|
169
|
+
"""Reduce a raw recombination-candidates dataframe (from
|
|
170
|
+
find_recombination_candidates, same `mode`) for feeding into
|
|
171
|
+
eso.constraints.recombination_to_multiple_avoidance_sites, WITHOUT the
|
|
172
|
+
coverage-loss risk collapse_recombination_sites has - see
|
|
173
|
+
eso.detection.recombination.recombination_sites_for_constraints.
|
|
174
|
+
"""
|
|
175
|
+
return _dispatch(RECOMBINATION_FOR_CONSTRAINTS_MODES, mode, "recombination", df_pairs)
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def slippage_sites_for_constraints(df_slippage, mode="default"):
|
|
179
|
+
"""Reduce a raw slippage-candidates dataframe (from
|
|
180
|
+
find_slippage_candidates, same `mode`) for feeding into
|
|
181
|
+
eso.detection.slippage.modify_df_slippage, WITHOUT the coverage-loss risk
|
|
182
|
+
collapse_slippage_sites has - see
|
|
183
|
+
eso.detection.slippage.slippage_sites_for_constraints.
|
|
184
|
+
"""
|
|
185
|
+
return _dispatch(SLIPPAGE_FOR_CONSTRAINTS_MODES, mode, "slippage", df_slippage)
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
"""Methylation motif detection via position-specific scoring matrices (PSSMs).
|
|
2
|
+
|
|
3
|
+
Scores every position of the sequence (forward and reverse complement)
|
|
4
|
+
against a set of methylation-enzyme recognition motifs and keeps the
|
|
5
|
+
best-scoring match per position above random-chance probability.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import numpy as np
|
|
9
|
+
import pandas as pd
|
|
10
|
+
from Bio import motifs
|
|
11
|
+
|
|
12
|
+
from eso.sequence_utils import reverse_complement_seq
|
|
13
|
+
|
|
14
|
+
SITE_COLUMNS = [
|
|
15
|
+
'start_index', 'end_index', 'matching_motif', 'PSSM_score',
|
|
16
|
+
'actual_site', 'actual_site_reverse_conjugate',
|
|
17
|
+
]
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def load_motifs(motifs_path):
|
|
21
|
+
"""Load PSSM motifs from a MEME-minimal-format file (e.g. topEnriched.*.meme.txt)."""
|
|
22
|
+
with open(motifs_path, "r", encoding="utf-8") as handle:
|
|
23
|
+
return list(motifs.parse(handle, "minimal"))
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def find_motif_sites(seq, num_sites, relevant_motifs):
|
|
27
|
+
"""Find the best-scoring motif match at every position of `seq` (forward or
|
|
28
|
+
reverse complement, across all of `relevant_motifs`) that scores above
|
|
29
|
+
random-chance probability (PSSM log-odds > 0). Returns at most `num_sites`
|
|
30
|
+
rows, highest-scoring first.
|
|
31
|
+
|
|
32
|
+
Builds one score matrix (2*len(relevant_motifs) rows x len(seq) columns,
|
|
33
|
+
-inf where a motif doesn't reach that far) and reduces it with vectorized
|
|
34
|
+
numpy calls, rather than a per-position/per-motif Python loop building an
|
|
35
|
+
intermediate long-format dataframe with 2*len(relevant_motifs)*len(seq)
|
|
36
|
+
rows - profiling showed that intermediate frame (and a df.apply(axis=1) to
|
|
37
|
+
extract each site's sequence, now also replaced with plain zip()) was the
|
|
38
|
+
dominant cost for realistic multi-motif sets on long sequences.
|
|
39
|
+
"""
|
|
40
|
+
seq_len = len(seq)
|
|
41
|
+
if not relevant_motifs or seq_len == 0:
|
|
42
|
+
return pd.DataFrame(columns=SITE_COLUMNS)
|
|
43
|
+
|
|
44
|
+
num_motifs = len(relevant_motifs)
|
|
45
|
+
scores_matrix = np.full((2 * num_motifs, seq_len), -np.inf)
|
|
46
|
+
motif_lengths = np.empty(num_motifs, dtype=int)
|
|
47
|
+
motif_names = []
|
|
48
|
+
|
|
49
|
+
for ii, motif in enumerate(relevant_motifs):
|
|
50
|
+
length = len(motif)
|
|
51
|
+
motif_lengths[ii] = length
|
|
52
|
+
motif_names.append(motif.name)
|
|
53
|
+
valid = seq_len - length + 1
|
|
54
|
+
if valid <= 0:
|
|
55
|
+
continue
|
|
56
|
+
scores_matrix[2 * ii, :valid] = motif.pssm.calculate(seq)
|
|
57
|
+
scores_matrix[2 * ii + 1, :valid] = motif.pssm.reverse_complement().calculate(seq)
|
|
58
|
+
|
|
59
|
+
# ties (equal score at the same position) resolve to the lowest motif_number,
|
|
60
|
+
# forward strand before backward - np.argmax returns the first max, and rows
|
|
61
|
+
# are laid out (motif0-fwd, motif0-rev, motif1-fwd, ...), matching this
|
|
62
|
+
# function's historical tie-breaking order.
|
|
63
|
+
best_row = np.argmax(scores_matrix, axis=0)
|
|
64
|
+
best_score = scores_matrix[best_row, np.arange(seq_len)]
|
|
65
|
+
|
|
66
|
+
keep_mask = best_score > 0
|
|
67
|
+
start_indices = np.nonzero(keep_mask)[0]
|
|
68
|
+
if start_indices.size == 0:
|
|
69
|
+
return pd.DataFrame(columns=SITE_COLUMNS)
|
|
70
|
+
|
|
71
|
+
winning_rows = best_row[keep_mask]
|
|
72
|
+
winning_scores = best_score[keep_mask]
|
|
73
|
+
winning_motif_numbers = winning_rows // 2
|
|
74
|
+
end_indices = start_indices + motif_lengths[winning_motif_numbers] - 1
|
|
75
|
+
|
|
76
|
+
# highest-scoring first, matching this function's historical output order
|
|
77
|
+
order = np.argsort(-winning_scores, kind='stable')
|
|
78
|
+
if num_sites < order.size:
|
|
79
|
+
order = order[:int(num_sites)]
|
|
80
|
+
|
|
81
|
+
start_indices = start_indices[order]
|
|
82
|
+
end_indices = end_indices[order]
|
|
83
|
+
winning_scores = winning_scores[order]
|
|
84
|
+
matching_motifs = [motif_names[m] for m in winning_motif_numbers[order]]
|
|
85
|
+
|
|
86
|
+
actual_sites = [seq[s:e + 1] for s, e in zip(start_indices, end_indices)]
|
|
87
|
+
actual_sites_rc = [reverse_complement_seq(site) for site in actual_sites]
|
|
88
|
+
|
|
89
|
+
return pd.DataFrame({
|
|
90
|
+
'start_index': start_indices,
|
|
91
|
+
'end_index': end_indices,
|
|
92
|
+
'matching_motif': matching_motifs,
|
|
93
|
+
'PSSM_score': winning_scores,
|
|
94
|
+
'actual_site': actual_sites,
|
|
95
|
+
'actual_site_reverse_conjugate': actual_sites_rc,
|
|
96
|
+
})
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""Build Bio.motifs Motif objects from a plain IUPAC consensus string (e.g.
|
|
2
|
+
"GATC", "CCWGG"), instead of requiring a full MEME-format PSSM file.
|
|
3
|
+
|
|
4
|
+
This is the easy on-ramp for custom motifs: most methylation/restriction
|
|
5
|
+
motifs are naturally described this way (REBASE, NEB, and the primary
|
|
6
|
+
literature all give them as a consensus sequence with ambiguity codes, not as
|
|
7
|
+
a position-probability matrix) - so there's no need to hand-author a MEME
|
|
8
|
+
file just to check for one.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from Bio import motifs
|
|
12
|
+
|
|
13
|
+
#: IUPAC nucleotide ambiguity codes -> the bases each one represents.
|
|
14
|
+
IUPAC_NUCLEOTIDE_CODES = {
|
|
15
|
+
'A': 'A', 'C': 'C', 'G': 'G', 'T': 'T',
|
|
16
|
+
'R': 'AG', 'Y': 'CT', 'S': 'GC', 'W': 'AT', 'K': 'GT', 'M': 'AC',
|
|
17
|
+
'B': 'CGT', 'D': 'AGT', 'H': 'ACT', 'V': 'ACG', 'N': 'ACGT',
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def motif_from_consensus(name, consensus, pseudocount=1):
|
|
22
|
+
"""Build a single Bio.motifs Motif from an IUPAC consensus string.
|
|
23
|
+
|
|
24
|
+
Each position scores an exact match to one of its allowed bases highest;
|
|
25
|
+
ambiguity codes (e.g. `W` = A or T) split probability evenly between the
|
|
26
|
+
bases they represent. A small `pseudocount` (added to every base at every
|
|
27
|
+
position, allowed or not) keeps every probability nonzero, so scoring
|
|
28
|
+
never divides by zero - it does not meaningfully weaken the requirement
|
|
29
|
+
that a real match stick to the allowed bases.
|
|
30
|
+
|
|
31
|
+
Parameters
|
|
32
|
+
----------
|
|
33
|
+
name: str
|
|
34
|
+
Name for the resulting motif (appears as `matching_motif` in
|
|
35
|
+
eso.detection.methylation.find_motif_sites' output).
|
|
36
|
+
consensus: str
|
|
37
|
+
An IUPAC nucleotide string, e.g. "GATC" (exact) or "CCWGG"
|
|
38
|
+
(W = A or T).
|
|
39
|
+
pseudocount: int or float
|
|
40
|
+
Added to every base's raw count at every position before
|
|
41
|
+
normalizing. Higher values make the motif more tolerant of
|
|
42
|
+
near-matches; the default (1, against an allowed-base count of 100)
|
|
43
|
+
is close to an exact-match requirement.
|
|
44
|
+
|
|
45
|
+
Returns
|
|
46
|
+
-------
|
|
47
|
+
A Bio.motifs Motif, usable anywhere a MEME-file-loaded motif is (e.g.
|
|
48
|
+
eso.detection.methylation.find_motif_sites' `relevant_motifs`).
|
|
49
|
+
"""
|
|
50
|
+
consensus = consensus.strip().upper()
|
|
51
|
+
if not consensus:
|
|
52
|
+
raise ValueError("consensus can't be empty.")
|
|
53
|
+
|
|
54
|
+
counts = {letter: [] for letter in 'ACGT'}
|
|
55
|
+
for position, char in enumerate(consensus):
|
|
56
|
+
allowed = IUPAC_NUCLEOTIDE_CODES.get(char)
|
|
57
|
+
if allowed is None:
|
|
58
|
+
raise ValueError(
|
|
59
|
+
f"'{char}' at position {position + 1} of '{consensus}' isn't a recognized "
|
|
60
|
+
f"IUPAC nucleotide code. Valid codes: A, C, G, T, or an ambiguity code "
|
|
61
|
+
f"({', '.join(sorted(c for c in IUPAC_NUCLEOTIDE_CODES if len(IUPAC_NUCLEOTIDE_CODES[c]) > 1))})."
|
|
62
|
+
)
|
|
63
|
+
for letter in 'ACGT':
|
|
64
|
+
counts[letter].append((100 if letter in allowed else 0) + pseudocount)
|
|
65
|
+
|
|
66
|
+
motif = motifs.Motif(counts=counts)
|
|
67
|
+
motif.name = name
|
|
68
|
+
return motif
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def motifs_from_consensus_dict(consensus_by_name, pseudocount=1):
|
|
72
|
+
"""Build a list of Motifs from a plain {name: consensus_string} dict -
|
|
73
|
+
e.g. `motifs_from_consensus_dict({"my_site": "GANTC"})`. This, plus
|
|
74
|
+
`motif_from_consensus`, is all that's needed to define custom motifs
|
|
75
|
+
without a MEME file.
|
|
76
|
+
"""
|
|
77
|
+
return [motif_from_consensus(name, consensus, pseudocount=pseudocount) for name, consensus in consensus_by_name.items()]
|