evolutionary-stability-optimizer 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,62 @@
1
+ "codon","aa","freq_within_aa"
2
+ "AAA","K",0.5038634756949489
3
+ "AAC","N",0.5718761146593684
4
+ "AAG","K",0.4961365243050511
5
+ "AAT","N",0.4281238853406316
6
+ "ACA","T",0.13097182238346933
7
+ "ACC","T",0.47899857499742254
8
+ "ACG","T",0.09268272481419175
9
+ "ACT","T",0.29734687780491637
10
+ "AGA","R",0.1876054334663075
11
+ "AGC","S",0.19444975284400967
12
+ "AGG","R",0.3922372498119939
13
+ "AGT","S",0.20454951237884456
14
+ "ATA","I",0.03868630942512579
15
+ "ATC","I",0.7397642158016438
16
+ "ATG","M",1
17
+ "ATT","I",0.22154947477323048
18
+ "CAA","Q",0.19296824449591132
19
+ "CAC","H",0.6570207717474473
20
+ "CAG","Q",0.8070317555040887
21
+ "CAT","H",0.3429792282525527
22
+ "CCA","P",0.33448630048322064
23
+ "CCC","P",0.24284536707540752
24
+ "CCG","P",0.07291725166880268
25
+ "CCT","P",0.34975108077256917
26
+ "CGA","R",0.0842652823706148
27
+ "CGC","R",0.09752660203720129
28
+ "CGG","R",0.20034523246364142
29
+ "CGT","R",0.03802019985024105
30
+ "CTA","L",0.05924441120383728
31
+ "CTC","L",0.3665813872174048
32
+ "CTG","L",0.35403072244441774
33
+ "CTT","L",0.039708080776196295
34
+ "GAA","E",0.4260872689539162
35
+ "GAC","D",0.45230366253507515
36
+ "GAG","E",0.5739127310460839
37
+ "GAT","D",0.5476963374649249
38
+ "GCA","A",0.2191246284363758
39
+ "GCC","A",0.5061087270479293
40
+ "GCG","A",0.05734268094352331
41
+ "GCT","A",0.21742396357217159
42
+ "GGA","G",0.1959839346229272
43
+ "GGC","G",0.32278997536379045
44
+ "GGG","G",0.38173182752721285
45
+ "GGT","G",0.09949426248606953
46
+ "GTA","V",0.07877585325542301
47
+ "GTC","V",0.4179341238484603
48
+ "GTG","V",0.33411562644079185
49
+ "GTT","V",0.16917439645532487
50
+ "TAC","Y",0.39897973273328724
51
+ "TAT","Y",0.6010202672667128
52
+ "TCA","S",0.08820172572782266
53
+ "TCC","S",0.20245309080811363
54
+ "TCG","S",0.022302532086794738
55
+ "TCT","S",0.2880433861544147
56
+ "TGC","C",0.6363459475159284
57
+ "TGG","W",1
58
+ "TGT","C",0.3636540524840716
59
+ "TTA","L",0.07132527815882145
60
+ "TTC","F",0.7559554651014863
61
+ "TTG","L",0.10911012019932245
62
+ "TTT","F",0.24404453489851372
File without changes
@@ -0,0 +1,112 @@
1
+ """Shared non-max-suppression helper: several detectors (independently, in both
2
+ the primary and STABLES-derived implementations) generate multiple candidate
3
+ rows - at different phases, frames, or seed positions - that describe the same
4
+ real hotspot rather than genuinely distinct ones. This collapses a dataframe
5
+ of scored, ranged candidates down to one representative per group of
6
+ mutually-overlapping candidates.
7
+ """
8
+
9
+ import pandas as pd
10
+
11
+
12
+ def ranges_overlap(a, b):
13
+ """True if ranges (start, end) overlap, using the same exclusive-end
14
+ convention as Python slicing (seq[start:end]) - so two ranges that merely
15
+ touch at a shared boundary (e.g. (0, 950) and (950, 1250)) do NOT count
16
+ as overlapping, since they share no actual character position. A
17
+ previous `<=` version treated touching-but-adjacent ranges as
18
+ overlapping, which silently discarded a genuinely distinct, adjacent
19
+ hotspot whenever it happened to sit immediately next to a higher-scoring
20
+ one - found via a chunk-boundary stress test for an unrelated prototype,
21
+ reproduced with two homopolymer runs separated by a third with no gap.
22
+ """
23
+ return a[0] < b[1] and b[0] < a[1]
24
+
25
+
26
+ def range_contains(outer, inner):
27
+ """True if `inner` (start, end) is fully contained in `outer` (start, end),
28
+ exclusive-end throughout (matches ranges_overlap's convention).
29
+ """
30
+ return outer[0] <= inner[0] and inner[1] <= outer[1]
31
+
32
+
33
+ def _collapse_by_predicate(df, score_col, start_col, end_col, should_drop):
34
+ """Shared non-max-suppression walk: process rows in descending `score_col`
35
+ order (ties broken by descending span, then a stable sort, so tied rows
36
+ process in a fixed, reproducible order - see below), dropping a row
37
+ exactly when `should_drop(current_range, kept_range)` is True for some
38
+ already-kept range, otherwise keeping it. The two public collapse
39
+ functions below differ only in which predicate they pass in - everything
40
+ else (the sort, the kept_rows/kept_ranges bookkeeping, the empty-
41
+ dataframe fallback) is identical between them, so it lives here once.
42
+
43
+ The span tiebreak was added after property-based testing
44
+ (tests/test_detection_overlap_properties.py) found that two same-scored
45
+ candidates, one fully containing the other, could both survive or either
46
+ one survive depending purely on which happened to appear first in `df` -
47
+ score-only sorting doesn't order tied rows at all, so it fell back to
48
+ whatever incidental order the input arrived in, making the result
49
+ non-deterministic across runs/pandas versions for that (rare but real)
50
+ case. Processing the larger span first among ties makes the outcome
51
+ reproducible, and additionally means a fully-redundant smaller duplicate
52
+ gets dropped rather than kept alongside it - but this only resolves EXACT
53
+ ties: a smaller, genuinely higher-scoring row and a larger, lower-scoring
54
+ one that doesn't fully cover it can both legitimately survive regardless
55
+ of this tiebreak (each may represent a structurally different candidate
56
+ that just happens to share coordinates) - that's correct, not a bug.
57
+ """
58
+ kept_rows = []
59
+ kept_ranges = []
60
+
61
+ ordering = df.assign(_span=df[end_col] - df[start_col]).sort_values(
62
+ [score_col, '_span'], ascending=[False, False], kind='mergesort')
63
+
64
+ for _, row in ordering.iterrows():
65
+ current_range = (row[start_col], row[end_col])
66
+ if any(should_drop(current_range, kept_range) for kept_range in kept_ranges):
67
+ continue
68
+ kept_rows.append(row.drop('_span'))
69
+ kept_ranges.append(current_range)
70
+
71
+ return pd.DataFrame(kept_rows, columns=df.columns) if kept_rows else df.iloc[0:0]
72
+
73
+
74
+ def collapse_overlapping_intervals(df, score_col, start_col='start', end_col='end'):
75
+ """Walk rows in descending `score_col` order, keeping a row only if its
76
+ [start_col, end_col] range doesn't overlap an already-kept row's range.
77
+
78
+ This is non-max suppression: it only requires ranges to *overlap*, not
79
+ that one *contain* the other, to drop the lower-scoring one. For a
80
+ "distinct sites" report or count that's the right call - but it means a
81
+ row can be dropped even though part of its range isn't actually covered
82
+ by anything kept, silently losing coverage of that part entirely. Do NOT
83
+ use this to build correction constraints for that reason - use
84
+ collapse_overlapping_intervals_no_coverage_loss instead. See
85
+ docs/detector-comparisons.md for the concrete failure this caused.
86
+ """
87
+ return _collapse_by_predicate(df, score_col, start_col, end_col, should_drop=ranges_overlap)
88
+
89
+
90
+ def collapse_overlapping_intervals_no_coverage_loss(df, score_col, start_col='start', end_col='end'):
91
+ """Like collapse_overlapping_intervals, but only drops a row when its
92
+ range is FULLY CONTAINED in an already-kept row's range - not merely
93
+ overlapping it.
94
+
95
+ Walking in descending `score_col` order, a redundant candidate (one whose
96
+ entire range is already covered by a higher-scoring kept row - the common
97
+ case: several detections of the same physical site converging on
98
+ essentially the same extent) is still dropped, exactly as before. But a
99
+ candidate that only *partially* overlaps a kept row - some real,
100
+ independently-detected hotspot whose span isn't a subset of the kept
101
+ row's - is kept too, so nothing downstream ever has to build a correction
102
+ constraint for a region no surviving row actually covers.
103
+
104
+ Rows returned by this function can legitimately still overlap each other
105
+ (that's the point) - use it to feed a correction/constraint-building step,
106
+ never for a "how many distinct sites" count or report (use
107
+ collapse_overlapping_intervals for that instead).
108
+ """
109
+ def _kept_fully_covers_current(current_range, kept_range):
110
+ return range_contains(outer=kept_range, inner=current_range)
111
+
112
+ return _collapse_by_predicate(df, score_col, start_col, end_col, should_drop=_kept_fully_covers_current)
@@ -0,0 +1,81 @@
1
+ """A small, ready-to-use library of commonly-referenced DNA motifs worth
2
+ checking for/avoiding when designing an engineered sequence - not just
3
+ methylation, but a few other properties that turn up across the synthetic
4
+ biology literature.
5
+
6
+ **Methylation** (E. coli's two systems, since E. coli is already a
7
+ first-class host here - see eso.codon_usage's bundled "e_coli" support):
8
+ - `dam`: GATC, N6-methyladenine (essentially universal across E. coli
9
+ strains and many other Gammaproteobacteria).
10
+ - `dcm`: CCWGG (W = A or T), C5-methylcytosine on the internal C.
11
+ Sources: NEB, "Dam and Dcm Methylases of E. coli"; EcoSal Plus, "DNA
12
+ Methylation" (doi.org/10.1128/ecosalplus.esp-0003-2013).
13
+
14
+ **Cryptic ribosome binding**:
15
+ - `shine_dalgarno`: AGGAGG, the canonical bacterial ribosome-binding-site
16
+ consensus. A copy of this sequence occurring *inside* a coding region
17
+ (not at an intended start codon) is a known source of unintended internal
18
+ translation initiation.
19
+ Source: Shine-Dalgarno sequences are measurably depleted from within
20
+ bacterial coding sequences, consistent with selection against this exact
21
+ risk (see e.g. Mol. Biol. Evol. 35(10):2487, and PMC6107199).
22
+
23
+ **Cryptic bacterial (sigma70) promoter elements**:
24
+ - `sigma70_minus35`: TTGACA, the -35 hexamer consensus.
25
+ - `sigma70_minus10`: TATAAT, the -10 ("Pribnow box") hexamer consensus.
26
+ These are the two core hexamers recognized by E. coli's housekeeping sigma
27
+ factor; an accidental occurrence of either inside a coding sequence is a
28
+ textbook source of unwanted "cryptic" transcription. **Caveat**: a real
29
+ sigma70 promoter needs BOTH hexamers at roughly the right spacing
30
+ (~17±1bp apart) - this module flags each hexamer independently (this
31
+ detector has no concept of "two motifs at a specific spacing"), so an
32
+ isolated hit is much weaker evidence than a "both boxes, correctly spaced"
33
+ finding would be. Treat isolated hits as a coarse, conservative screen, not
34
+ a confirmed cryptic promoter.
35
+
36
+ **Not included, and why**: transcription terminators (rho-independent
37
+ terminators are a hairpin + poly-U structure - a secondary-structure
38
+ property, not a fixed linear sequence motif this PSSM-based approach can
39
+ represent) and the eukaryotic Kozak sequence (a *desired* translation-
40
+ initiation context to match near a real start codon, not something to
41
+ avoid - a different problem from what this detector/module is for).
42
+ Restriction enzyme sites are also not duplicated here - DNAChisel (already
43
+ a dependency of this project) already bundles a comprehensive registry via
44
+ `dnachisel.list_common_enzymes()` / `dnachisel.EnzymeSitePattern`, usable
45
+ directly with `AvoidPattern` during optimization.
46
+
47
+ For anything beyond what's bundled here, REBASE (rebase.neb.com) is the
48
+ standard reference database for restriction/methylation motifs across
49
+ organisms - use `eso.detection.motif_utils.motif_from_consensus` to turn
50
+ any REBASE-style (or other literature) consensus sequence into a usable
51
+ motif, the same way everything in this module is built.
52
+ """
53
+
54
+ from eso.detection.motif_utils import motif_from_consensus
55
+
56
+ #: name -> IUPAC consensus sequence
57
+ COMMON_MOTIFS = {
58
+ "dam": "GATC",
59
+ "dcm": "CCWGG",
60
+ "shine_dalgarno": "AGGAGG",
61
+ "sigma70_minus35": "TTGACA",
62
+ "sigma70_minus10": "TATAAT",
63
+ }
64
+
65
+
66
+ def load_common_motifs(names=None):
67
+ """Return a list of Bio.motifs Motif objects for the requested common
68
+ motifs (default: all of them - see COMMON_MOTIFS for the full list).
69
+
70
+ `names` is case-insensitive; see COMMON_MOTIFS for the available keys.
71
+ """
72
+ if names is None:
73
+ names = list(COMMON_MOTIFS)
74
+
75
+ motifs_list = []
76
+ for name in names:
77
+ key = name.strip().lower()
78
+ if key not in COMMON_MOTIFS:
79
+ raise ValueError(f"Unknown common motif {name!r}; choose from {sorted(COMMON_MOTIFS)}")
80
+ motifs_list.append(motif_from_consensus(key, COMMON_MOTIFS[key]))
81
+ return motifs_list
@@ -0,0 +1,185 @@
1
+ """Mode-based routing across independently-developed detector implementations,
2
+ so callers choose a tradeoff without needing to know which module originally
3
+ implemented which algorithm.
4
+
5
+ Covers recombination and slippage detection; methylation has only one
6
+ implementation (eso.detection.methylation) - a second was built and compared
7
+ here, but removed after it was found to disagree in accuracy (not just
8
+ speed) with the first - see docs/detector-comparisons.md.
9
+ """
10
+
11
+ import numpy as np
12
+
13
+ from eso.detection import recombination, slippage, staubility_variant
14
+
15
+ # Each category (recombination/slippage) is dispatched at three stages -
16
+ # find the raw candidates, collapse them to a report view, and reduce them
17
+ # to a constraint-building view (see eso.detection.recombination's module
18
+ # docstring split for what each stage means) - across the same two
19
+ # implementations ("thorough"/"fast" or "default"/"fast"). That's 6 mode
20
+ # tables in total; `_dispatch` below is the one place that turns a mode
21
+ # table + a chosen mode into either a result or a friendly ValueError, so a
22
+ # bug fix or message tweak to that logic only has to happen once.
23
+
24
+ RECOMBINATION_MODES = {
25
+ "thorough": recombination.find_recombination_sites,
26
+ "fast": staubility_variant.find_recombination_sites,
27
+ }
28
+ RECOMBINATION_CANDIDATE_MODES = {
29
+ "thorough": recombination.find_recombination_candidates,
30
+ "fast": staubility_variant.find_recombination_candidates,
31
+ }
32
+ RECOMBINATION_COLLAPSE_MODES = {
33
+ "thorough": recombination.collapse_recombination_sites,
34
+ "fast": staubility_variant.collapse_recombination_sites,
35
+ }
36
+ RECOMBINATION_FOR_CONSTRAINTS_MODES = {
37
+ "thorough": recombination.recombination_sites_for_constraints,
38
+ "fast": staubility_variant.recombination_sites_for_constraints,
39
+ }
40
+
41
+ SLIPPAGE_MODES = {
42
+ "default": slippage.find_slippage_sites,
43
+ "fast": staubility_variant.find_slippage_sites,
44
+ }
45
+ SLIPPAGE_CANDIDATE_MODES = {
46
+ "default": slippage.find_slippage_candidates,
47
+ "fast": staubility_variant.find_slippage_candidates,
48
+ }
49
+ SLIPPAGE_COLLAPSE_MODES = {
50
+ "default": slippage.collapse_slippage_sites,
51
+ "fast": staubility_variant.collapse_slippage_sites,
52
+ }
53
+ SLIPPAGE_FOR_CONSTRAINTS_MODES = {
54
+ "default": slippage.slippage_sites_for_constraints,
55
+ "fast": staubility_variant.slippage_sites_for_constraints,
56
+ }
57
+
58
+
59
+ def _dispatch(modes, mode, category_label, *args):
60
+ """Look up `mode` in `modes` and call it with `args`, or raise a
61
+ ValueError naming `category_label` ("recombination"/"slippage") and
62
+ listing the modes actually available - the one error-message format
63
+ every public function below shares.
64
+ """
65
+ try:
66
+ implementation = modes[mode]
67
+ except KeyError:
68
+ raise ValueError(f"Unknown {category_label} mode {mode!r}; choose from {sorted(modes)}")
69
+ return implementation(*args)
70
+
71
+
72
+ def find_recombination_sites(seq, num_sites=np.inf, mode="thorough"):
73
+ """Detect recombination (RMD) hotspots, routed to one of two independently
74
+ developed implementations.
75
+
76
+ mode="thorough" (default) - eso.detection.recombination: Levenshtein-tolerant,
77
+ catches pairs of sites within edit distance 1 of each other, not just
78
+ exact duplicates. Benchmarked as roughly linear from 51,400nt through
79
+ 1,000,000nt (~126.7s at 1,000,000nt, in local benchmarks after fixing
80
+ a pandas-overhead bottleneck - see docs/detector-comparisons.md), with
81
+ no breakdown point found at any tested scale - recommended by default
82
+ at essentially any realistic sequence length.
83
+
84
+ mode="fast" - eso.detection.staubility_variant: exact 16-mer match only,
85
+ via vectorized n-gram counting. Will miss a near-duplicate whenever
86
+ its point of divergence sits centrally enough that no 16-consecutive-nt
87
+ exact window survives on either side (verified: catches a duplicate
88
+ with a 1nt substitution near either edge, since 16+nt of exact match
89
+ remains; misses the same case when the substitution is centered).
90
+ 19-34x faster than "thorough" at every length tested - reach for this
91
+ only when that speed gap itself matters (e.g. many-sequence batch
92
+ workloads), not because "thorough" becomes intractable.
93
+ """
94
+ return _dispatch(RECOMBINATION_MODES, mode, "recombination", seq, num_sites)
95
+
96
+
97
+ def find_slippage_sites(seq, num_sites=np.inf, mode="default"):
98
+ """Detect slippage (SSR) hotspots, routed to one of two independently
99
+ developed implementations.
100
+
101
+ Unlike recombination, both modes detect exactly the same hotspots -
102
+ verified via 300 randomized trials with zero sensitivity or row-count
103
+ mismatches after fixing bugs in both implementations (see
104
+ docs/detector-comparisons.md). This is purely a speed choice, and
105
+ "default" wins it outright:
106
+
107
+ mode="default" - eso.detection.slippage: after fixing an O(n^2) candidate-
108
+ scan (see docs/detector-comparisons.md), this is faster than "fast"
109
+ at every length tested, from a few hundred nt through 300,000nt,
110
+ with the gap widening as length grows (10x faster at 300kb).
111
+
112
+ mode="fast" - eso.detection.staubility_variant: kept as an independent
113
+ second implementation (useful as a cross-check, and it's a distinct
114
+ algorithm, not just a slower copy) - but there is no longer a length
115
+ range where it's actually faster than "default".
116
+ """
117
+ return _dispatch(SLIPPAGE_MODES, mode, "slippage", seq, num_sites)
118
+
119
+
120
+ def find_recombination_candidates(seq, mode="thorough"):
121
+ """Every candidate recombination (RMD) site-pair, WITHOUT collapsing
122
+ overlapping pairs down to one representative per real hotspot - routed to
123
+ one of the two implementations documented on find_recombination_sites.
124
+
125
+ This is the shared base find_recombination_sites (report view, via
126
+ collapse_recombination_sites) and recombination_sites_for_constraints
127
+ (constraint-building view) are both derived from - use whichever of
128
+ those two fits, not this function directly, unless you specifically want
129
+ every raw candidate with nothing reduced at all.
130
+ """
131
+ return _dispatch(RECOMBINATION_CANDIDATE_MODES, mode, "recombination", seq)
132
+
133
+
134
+ def find_slippage_candidates(seq, mode="default"):
135
+ """Every candidate slippage (SSR) hotspot, WITHOUT collapsing overlapping
136
+ candidates down to one representative per physical site - routed to one
137
+ of the two implementations documented on find_slippage_sites.
138
+
139
+ This is the shared base find_slippage_sites (report view, via
140
+ collapse_slippage_sites) and slippage_sites_for_constraints
141
+ (constraint-building view) are both derived from - use whichever of
142
+ those two fits, not this function directly, unless you specifically want
143
+ every raw candidate with nothing reduced at all.
144
+ """
145
+ return _dispatch(SLIPPAGE_CANDIDATE_MODES, mode, "slippage", seq)
146
+
147
+
148
+ def collapse_recombination_sites(df_pairs, num_sites=np.inf, mode="thorough"):
149
+ """Collapse a raw recombination-candidates dataframe (from
150
+ find_recombination_candidates, same `mode`) down to the human-facing
151
+ "distinct sites" view - one representative pair per real hotspot, limited
152
+ to `num_sites`. Lets a caller run detection once and derive both the raw
153
+ (constraint-building) and collapsed (report) views without re-detecting.
154
+ """
155
+ return _dispatch(RECOMBINATION_COLLAPSE_MODES, mode, "recombination", df_pairs, num_sites)
156
+
157
+
158
+ def collapse_slippage_sites(df_slippage, num_sites=np.inf, mode="default"):
159
+ """Collapse a raw slippage-candidates dataframe (from
160
+ find_slippage_candidates, same `mode`) down to the human-facing "distinct
161
+ sites" view - one representative per physical site, limited to
162
+ `num_sites`. Lets a caller run detection once and derive both the raw
163
+ (constraint-building) and collapsed (report) views without re-detecting.
164
+ """
165
+ return _dispatch(SLIPPAGE_COLLAPSE_MODES, mode, "slippage", df_slippage, num_sites)
166
+
167
+
168
+ def recombination_sites_for_constraints(df_pairs, mode="thorough"):
169
+ """Reduce a raw recombination-candidates dataframe (from
170
+ find_recombination_candidates, same `mode`) for feeding into
171
+ eso.constraints.recombination_to_multiple_avoidance_sites, WITHOUT the
172
+ coverage-loss risk collapse_recombination_sites has - see
173
+ eso.detection.recombination.recombination_sites_for_constraints.
174
+ """
175
+ return _dispatch(RECOMBINATION_FOR_CONSTRAINTS_MODES, mode, "recombination", df_pairs)
176
+
177
+
178
+ def slippage_sites_for_constraints(df_slippage, mode="default"):
179
+ """Reduce a raw slippage-candidates dataframe (from
180
+ find_slippage_candidates, same `mode`) for feeding into
181
+ eso.detection.slippage.modify_df_slippage, WITHOUT the coverage-loss risk
182
+ collapse_slippage_sites has - see
183
+ eso.detection.slippage.slippage_sites_for_constraints.
184
+ """
185
+ return _dispatch(SLIPPAGE_FOR_CONSTRAINTS_MODES, mode, "slippage", df_slippage)
@@ -0,0 +1,96 @@
1
+ """Methylation motif detection via position-specific scoring matrices (PSSMs).
2
+
3
+ Scores every position of the sequence (forward and reverse complement)
4
+ against a set of methylation-enzyme recognition motifs and keeps the
5
+ best-scoring match per position above random-chance probability.
6
+ """
7
+
8
+ import numpy as np
9
+ import pandas as pd
10
+ from Bio import motifs
11
+
12
+ from eso.sequence_utils import reverse_complement_seq
13
+
14
+ SITE_COLUMNS = [
15
+ 'start_index', 'end_index', 'matching_motif', 'PSSM_score',
16
+ 'actual_site', 'actual_site_reverse_conjugate',
17
+ ]
18
+
19
+
20
+ def load_motifs(motifs_path):
21
+ """Load PSSM motifs from a MEME-minimal-format file (e.g. topEnriched.*.meme.txt)."""
22
+ with open(motifs_path, "r", encoding="utf-8") as handle:
23
+ return list(motifs.parse(handle, "minimal"))
24
+
25
+
26
+ def find_motif_sites(seq, num_sites, relevant_motifs):
27
+ """Find the best-scoring motif match at every position of `seq` (forward or
28
+ reverse complement, across all of `relevant_motifs`) that scores above
29
+ random-chance probability (PSSM log-odds > 0). Returns at most `num_sites`
30
+ rows, highest-scoring first.
31
+
32
+ Builds one score matrix (2*len(relevant_motifs) rows x len(seq) columns,
33
+ -inf where a motif doesn't reach that far) and reduces it with vectorized
34
+ numpy calls, rather than a per-position/per-motif Python loop building an
35
+ intermediate long-format dataframe with 2*len(relevant_motifs)*len(seq)
36
+ rows - profiling showed that intermediate frame (and a df.apply(axis=1) to
37
+ extract each site's sequence, now also replaced with plain zip()) was the
38
+ dominant cost for realistic multi-motif sets on long sequences.
39
+ """
40
+ seq_len = len(seq)
41
+ if not relevant_motifs or seq_len == 0:
42
+ return pd.DataFrame(columns=SITE_COLUMNS)
43
+
44
+ num_motifs = len(relevant_motifs)
45
+ scores_matrix = np.full((2 * num_motifs, seq_len), -np.inf)
46
+ motif_lengths = np.empty(num_motifs, dtype=int)
47
+ motif_names = []
48
+
49
+ for ii, motif in enumerate(relevant_motifs):
50
+ length = len(motif)
51
+ motif_lengths[ii] = length
52
+ motif_names.append(motif.name)
53
+ valid = seq_len - length + 1
54
+ if valid <= 0:
55
+ continue
56
+ scores_matrix[2 * ii, :valid] = motif.pssm.calculate(seq)
57
+ scores_matrix[2 * ii + 1, :valid] = motif.pssm.reverse_complement().calculate(seq)
58
+
59
+ # ties (equal score at the same position) resolve to the lowest motif_number,
60
+ # forward strand before backward - np.argmax returns the first max, and rows
61
+ # are laid out (motif0-fwd, motif0-rev, motif1-fwd, ...), matching this
62
+ # function's historical tie-breaking order.
63
+ best_row = np.argmax(scores_matrix, axis=0)
64
+ best_score = scores_matrix[best_row, np.arange(seq_len)]
65
+
66
+ keep_mask = best_score > 0
67
+ start_indices = np.nonzero(keep_mask)[0]
68
+ if start_indices.size == 0:
69
+ return pd.DataFrame(columns=SITE_COLUMNS)
70
+
71
+ winning_rows = best_row[keep_mask]
72
+ winning_scores = best_score[keep_mask]
73
+ winning_motif_numbers = winning_rows // 2
74
+ end_indices = start_indices + motif_lengths[winning_motif_numbers] - 1
75
+
76
+ # highest-scoring first, matching this function's historical output order
77
+ order = np.argsort(-winning_scores, kind='stable')
78
+ if num_sites < order.size:
79
+ order = order[:int(num_sites)]
80
+
81
+ start_indices = start_indices[order]
82
+ end_indices = end_indices[order]
83
+ winning_scores = winning_scores[order]
84
+ matching_motifs = [motif_names[m] for m in winning_motif_numbers[order]]
85
+
86
+ actual_sites = [seq[s:e + 1] for s, e in zip(start_indices, end_indices)]
87
+ actual_sites_rc = [reverse_complement_seq(site) for site in actual_sites]
88
+
89
+ return pd.DataFrame({
90
+ 'start_index': start_indices,
91
+ 'end_index': end_indices,
92
+ 'matching_motif': matching_motifs,
93
+ 'PSSM_score': winning_scores,
94
+ 'actual_site': actual_sites,
95
+ 'actual_site_reverse_conjugate': actual_sites_rc,
96
+ })
@@ -0,0 +1,77 @@
1
+ """Build Bio.motifs Motif objects from a plain IUPAC consensus string (e.g.
2
+ "GATC", "CCWGG"), instead of requiring a full MEME-format PSSM file.
3
+
4
+ This is the easy on-ramp for custom motifs: most methylation/restriction
5
+ motifs are naturally described this way (REBASE, NEB, and the primary
6
+ literature all give them as a consensus sequence with ambiguity codes, not as
7
+ a position-probability matrix) - so there's no need to hand-author a MEME
8
+ file just to check for one.
9
+ """
10
+
11
+ from Bio import motifs
12
+
13
+ #: IUPAC nucleotide ambiguity codes -> the bases each one represents.
14
+ IUPAC_NUCLEOTIDE_CODES = {
15
+ 'A': 'A', 'C': 'C', 'G': 'G', 'T': 'T',
16
+ 'R': 'AG', 'Y': 'CT', 'S': 'GC', 'W': 'AT', 'K': 'GT', 'M': 'AC',
17
+ 'B': 'CGT', 'D': 'AGT', 'H': 'ACT', 'V': 'ACG', 'N': 'ACGT',
18
+ }
19
+
20
+
21
+ def motif_from_consensus(name, consensus, pseudocount=1):
22
+ """Build a single Bio.motifs Motif from an IUPAC consensus string.
23
+
24
+ Each position scores an exact match to one of its allowed bases highest;
25
+ ambiguity codes (e.g. `W` = A or T) split probability evenly between the
26
+ bases they represent. A small `pseudocount` (added to every base at every
27
+ position, allowed or not) keeps every probability nonzero, so scoring
28
+ never divides by zero - it does not meaningfully weaken the requirement
29
+ that a real match stick to the allowed bases.
30
+
31
+ Parameters
32
+ ----------
33
+ name: str
34
+ Name for the resulting motif (appears as `matching_motif` in
35
+ eso.detection.methylation.find_motif_sites' output).
36
+ consensus: str
37
+ An IUPAC nucleotide string, e.g. "GATC" (exact) or "CCWGG"
38
+ (W = A or T).
39
+ pseudocount: int or float
40
+ Added to every base's raw count at every position before
41
+ normalizing. Higher values make the motif more tolerant of
42
+ near-matches; the default (1, against an allowed-base count of 100)
43
+ is close to an exact-match requirement.
44
+
45
+ Returns
46
+ -------
47
+ A Bio.motifs Motif, usable anywhere a MEME-file-loaded motif is (e.g.
48
+ eso.detection.methylation.find_motif_sites' `relevant_motifs`).
49
+ """
50
+ consensus = consensus.strip().upper()
51
+ if not consensus:
52
+ raise ValueError("consensus can't be empty.")
53
+
54
+ counts = {letter: [] for letter in 'ACGT'}
55
+ for position, char in enumerate(consensus):
56
+ allowed = IUPAC_NUCLEOTIDE_CODES.get(char)
57
+ if allowed is None:
58
+ raise ValueError(
59
+ f"'{char}' at position {position + 1} of '{consensus}' isn't a recognized "
60
+ f"IUPAC nucleotide code. Valid codes: A, C, G, T, or an ambiguity code "
61
+ f"({', '.join(sorted(c for c in IUPAC_NUCLEOTIDE_CODES if len(IUPAC_NUCLEOTIDE_CODES[c]) > 1))})."
62
+ )
63
+ for letter in 'ACGT':
64
+ counts[letter].append((100 if letter in allowed else 0) + pseudocount)
65
+
66
+ motif = motifs.Motif(counts=counts)
67
+ motif.name = name
68
+ return motif
69
+
70
+
71
+ def motifs_from_consensus_dict(consensus_by_name, pseudocount=1):
72
+ """Build a list of Motifs from a plain {name: consensus_string} dict -
73
+ e.g. `motifs_from_consensus_dict({"my_site": "GANTC"})`. This, plus
74
+ `motif_from_consensus`, is all that's needed to define custom motifs
75
+ without a MEME file.
76
+ """
77
+ return [motif_from_consensus(name, consensus, pseudocount=pseudocount) for name, consensus in consensus_by_name.items()]