evolutionary-stability-optimizer 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,284 @@
1
+ """Alternate recombination/slippage detectors, developed independently inside the
2
+ STABLES project (Staubility_Code_shimshi.py) using a different algorithmic
3
+ approach than eso.detection.recombination/slippage:
4
+
5
+ - Recombination: ngram-counting (CountVectorizer, 16-mers) + exact re-match,
6
+ vs. the Levenshtein-neighbor generation approach in eso.detection.recombination.
7
+ - Slippage: per-frameshift scan for back-to-back repeats, vs. the
8
+ find-all-occurrences approach in eso.detection.slippage.
9
+
10
+ This module also carried a methylation-motif scorer (site_motif_grader/
11
+ calc_max_site) at one point; removed after comparison against
12
+ eso.detection.methylation found it scored candidates by raw sequence
13
+ probability with no background correction, which measurably disagrees with
14
+ (not just runs slower than) eso.detection.methylation's background-corrected
15
+ approach whenever a motif's background isn't uniform - see
16
+ docs/detector-comparisons.md for the full writeup, kept as a historical
17
+ record even though the code itself is gone.
18
+
19
+ Kept as a separate module rather than merged into the primary detectors -
20
+ these are two independently-arrived-at implementations of the same EFM
21
+ Calculator concept and haven't yet been reconciled into one canonical API.
22
+ """
23
+
24
+ import re
25
+
26
+ import numpy as np
27
+ import pandas as pd
28
+ from sklearn.feature_extraction.text import CountVectorizer
29
+
30
+ from eso.detection._overlap import collapse_overlapping_intervals, collapse_overlapping_intervals_no_coverage_loss
31
+
32
+
33
+ def genome_cutter(start, end, seq):
34
+ return seq[start:end]
35
+
36
+
37
+ def find_recombination_candidates(seq):
38
+ """ngram-counting variant of recombination (RMD) hotspot detection.
39
+
40
+ Unlike eso.detection.recombination's variant, this one never collapses
41
+ overlapping pairs (it only merges exact back-to-back 16-mer matches into
42
+ one site, which is a different, lossless operation) - so its output is
43
+ already the raw/candidate view. Kept as a separate name from
44
+ find_recombination_sites anyway, for a consistent candidates/collapsed
45
+ pair of names across both detector implementations - see
46
+ eso.detection.dispatch.
47
+ """
48
+ vectorizer = CountVectorizer(analyzer='char_wb', ngram_range=(16, 16))
49
+ counter = vectorizer.fit_transform([seq]).toarray()
50
+
51
+ sites_recombination = list(np.where(counter > 1)[1])
52
+ empty_columns = [
53
+ 'start_1', 'end_1', 'sequence', 'start_2', 'end_2',
54
+ 'location_delta', 'site_length', 'log10_prob_recombination_ecoli', 'sequence_number',
55
+ ]
56
+ if not sites_recombination:
57
+ return pd.DataFrame(columns=empty_columns)
58
+
59
+ all_sites = vectorizer.get_feature_names_out()
60
+
61
+ suspect_recombination = []
62
+ for site in sites_recombination:
63
+ curr_seq = all_sites[site]
64
+ list_regions = [match.span() for match in re.finditer(curr_seq.upper(), seq)]
65
+ suspect_recombination.extend(list_regions)
66
+
67
+ suspect_recombination = sorted(suspect_recombination)
68
+ df_recombination = pd.DataFrame(suspect_recombination, columns=['start', 'end'])
69
+
70
+ # merge back-to-back 16-mer matches (from repeats longer than 16nt) into one site
71
+ df_recombination.loc[:, 'start_delta'] = df_recombination['start'] - df_recombination['start'].shift()
72
+ df_recombination.loc[:, 'end_delta'] = df_recombination['end'].shift(-1) - df_recombination['end']
73
+ df_recombination = df_recombination[(df_recombination.start_delta != 1.0) | (df_recombination.end_delta != 1.0)]
74
+
75
+ df_recombination.loc[(df_recombination.end_delta == 1.0), 'end'] = None
76
+ # `df.loc[:, 'end'] = ...` would silently keep the column's existing float64
77
+ # dtype (from the None assignment above) even after .astype(int) below;
78
+ # only whole-column reassignment (df['end'] = ...) actually changes dtype.
79
+ df_recombination['end'] = df_recombination.loc[:, 'end'].bfill().astype(int) - 1
80
+ df_recombination = df_recombination[df_recombination.start_delta != 1.0][['start', 'end']]
81
+
82
+ df_recombination.loc[:, 'sequence'] = df_recombination.apply(
83
+ lambda x: genome_cutter(x['start'], x['end'], seq), axis=1)
84
+
85
+ df_recombination = df_recombination.merge(df_recombination, on='sequence', suffixes=('_1', '_2'))
86
+ df_recombination = df_recombination[df_recombination.end_1 < df_recombination.start_2]
87
+
88
+ df_recombination.loc[:, 'location_delta'] = df_recombination.start_2 - df_recombination.end_1
89
+ df_recombination.loc[:, 'site_length'] = df_recombination.end_1 - df_recombination.start_1
90
+
91
+ # a=5.8 per Oliveira et al. 2008, Table 3, recA+ row (see
92
+ # eso.detection.recombination.calc_recombination_score's docstring for
93
+ # the full story, including why the EFM Calculator's own 8.8 is a
94
+ # mix-up with a different row/parameter of that same table).
95
+ a, b, c, alpha = 5.8, 1465.6, 0, 29
96
+ base = a + df_recombination['location_delta']
97
+ exponent = -1 * alpha / df_recombination['site_length']
98
+ scale = df_recombination['site_length'] / (
99
+ 1 + b * df_recombination['site_length'] + c * df_recombination['location_delta'])
100
+ df_recombination.loc[:, 'log10_prob_recombination_ecoli'] = np.log10((base ** exponent) * scale)
101
+
102
+ return df_recombination.sort_values('log10_prob_recombination_ecoli', ascending=False).reset_index(drop=True)
103
+
104
+
105
+ def collapse_recombination_sites(df_recombination, num_sites=np.inf):
106
+ """No-op collapse: this detector never produces overlapping-but-distinct
107
+ pairs the way eso.detection.recombination's does (it only merges exact
108
+ back-to-back matches, already handled in find_recombination_candidates),
109
+ so there is nothing to collapse here - just applies `num_sites` and drops
110
+ any column that's entirely null.
111
+
112
+ The null-column drop happens here (on the num_sites-limited report view),
113
+ not in find_recombination_candidates, matching the original pre-split
114
+ behavior exactly: which columns are "entirely null" can differ between
115
+ the full candidate set and a num_sites-limited subset of it, so doing
116
+ this before truncating (as an earlier version of this split briefly did)
117
+ could report a different column set for the same `num_sites` value than
118
+ before the candidates/collapse split existed.
119
+ """
120
+ if num_sites < np.inf:
121
+ df_recombination = df_recombination.head(int(num_sites))
122
+
123
+ for col in df_recombination:
124
+ if df_recombination[col].isnull().all():
125
+ del df_recombination[col]
126
+
127
+ return df_recombination
128
+
129
+
130
+ def recombination_sites_for_constraints(df_recombination):
131
+ """Identity: this detector never collapses overlapping pairs in the first
132
+ place (see collapse_recombination_sites), so its raw candidates are
133
+ already the constraint-building view - kept as a same-named counterpart
134
+ to eso.detection.recombination's version so eso.detection.dispatch can
135
+ call either implementation uniformly.
136
+ """
137
+ return df_recombination
138
+
139
+
140
+ def find_recombination_sites(seq, num_sites=np.inf):
141
+ """ngram-counting variant of recombination (RMD) hotspot detection,
142
+ limited to `num_sites` if given. See find_recombination_candidates -
143
+ this detector never needs to collapse, so this is only a thin
144
+ num_sites-limiting wrapper.
145
+ """
146
+ return collapse_recombination_sites(find_recombination_candidates(seq), num_sites)
147
+
148
+
149
+ def find_slippage_sites_length_l(seq, length):
150
+ """Per-frameshift scan for a repeated base unit of size `length`
151
+ (repeated 3+ times, or 12+ times for length=1 - see
152
+ eso.detection.slippage's module docstring for why 12: below that, an
153
+ independently-run length-2 detection of the same physical region always
154
+ outscores it, so its result is always discarded downstream).
155
+ """
156
+ from textwrap import wrap
157
+
158
+ slippage_sites = []
159
+ homopolymer_window = 12
160
+
161
+ for frameshift in range(length):
162
+ curr_seq = seq[frameshift:]
163
+ curr_seq_split = wrap(curr_seq, length)
164
+
165
+ end_of_range = len(curr_seq_split) - 2
166
+
167
+ for ii in range(end_of_range):
168
+ # `length == 1` must be checked FIRST so Python's `and`
169
+ # short-circuits before touching curr_seq_split[ii + 1/2] when
170
+ # length > 1 (needed for is_followed2 below); is_followed1 itself
171
+ # uses slicing, which never raises IndexError even past the list
172
+ # end, so it needs no equivalent bounds bug to worry about - a
173
+ # slice shorter than `homopolymer_window` just fails the length
174
+ # check instead of crashing.
175
+ is_followed2 = (
176
+ length > 1
177
+ and curr_seq_split[ii] == curr_seq_split[ii + 1]
178
+ and curr_seq_split[ii] == curr_seq_split[ii + 2]
179
+ )
180
+ window = curr_seq_split[ii:ii + homopolymer_window]
181
+ is_followed1 = length == 1 and len(window) == homopolymer_window and len(set(window)) == 1
182
+
183
+ if is_followed2:
184
+ slippage_sites.append((frameshift + ii * length, frameshift + length * (ii + 3)))
185
+ if is_followed1:
186
+ slippage_sites.append((ii, ii + homopolymer_window))
187
+
188
+ if not slippage_sites:
189
+ return pd.DataFrame(columns=['start', 'end', 'sequence', 'length_base_unit'])
190
+
191
+ df_slippage = pd.DataFrame(sorted(slippage_sites), columns=['start', 'end'])
192
+
193
+ df_slippage.loc[:, 'start_delta'] = df_slippage['start'] - df_slippage['start'].shift()
194
+ df_slippage.loc[:, 'end_delta'] = df_slippage['end'].shift(-1) - df_slippage['end']
195
+ df_slippage = df_slippage[(df_slippage.start_delta != 1.0) | (df_slippage.end_delta != 1.0)]
196
+ df_slippage.loc[(df_slippage.end_delta == 1.0), 'end'] = None
197
+ # see the identical fix in find_recombination_sites above: whole-column
198
+ # reassignment is required for .astype(int) to actually stick here
199
+ df_slippage['end'] = df_slippage.loc[:, 'end'].bfill().astype(int)
200
+ df_slippage = df_slippage[df_slippage.start_delta != 1.0][['start', 'end']]
201
+
202
+ df_slippage.loc[:, 'end'] = (
203
+ ((df_slippage['end'] - df_slippage['start']) / length).astype(int) * length
204
+ ) + df_slippage['start']
205
+
206
+ df_slippage.loc[:, 'sequence'] = df_slippage.apply(
207
+ lambda x: genome_cutter(x['start'], x['end'], seq), axis=1)
208
+ df_slippage.loc[:, 'length_base_unit'] = length
209
+
210
+ return df_slippage
211
+
212
+
213
+ def find_slippage_candidates(seq):
214
+ """ngram/frameshift-scan variant of slippage (SSR) hotspot detection -
215
+ every candidate, WITHOUT collapsing overlapping ones down to one
216
+ representative per physical site. See
217
+ eso.detection.slippage.find_slippage_candidates for why constraint-
218
+ building must use this, not find_slippage_sites.
219
+ """
220
+ slippage_sites_list = [find_slippage_sites_length_l(seq, length) for length in range(1, 16)]
221
+ df_slippage = pd.concat(slippage_sites_list, ignore_index=True)[['start', 'end', 'length_base_unit', 'sequence']]
222
+
223
+ df_slippage.loc[:, 'num_base_units'] = (
224
+ df_slippage.sequence.apply(len) / df_slippage.length_base_unit
225
+ ).astype(int)
226
+
227
+ df_slippage.loc[:, 'log10_prob_slippage_ecoli'] = -4.749 + 0.063 * df_slippage['num_base_units']
228
+ df_slippage.loc[df_slippage.length_base_unit == 1, 'log10_prob_slippage_ecoli'] = (
229
+ -12.9 + 0.729 * df_slippage['num_base_units']
230
+ )
231
+
232
+ # this filter was applied by the STABLES caller (STABLES_full_code's
233
+ # optimization pipeline), not inside find_slippage_sites itself - lost
234
+ # when this function was extracted in isolation. Restored here so the
235
+ # function is self-contained and consistent with its sibling
236
+ # find_recombination_candidates above, which does filter internally.
237
+ df_slippage = df_slippage[df_slippage.log10_prob_slippage_ecoli > -9]
238
+ return df_slippage.sort_values('log10_prob_slippage_ecoli', ascending=False).reset_index(drop=True)
239
+
240
+
241
+ def slippage_sites_for_constraints(df_slippage):
242
+ """Reduce raw slippage candidates (see find_slippage_candidates) for
243
+ feeding into correction-constraint building, WITHOUT the coverage-loss
244
+ risk plain non-max suppression (collapse_slippage_sites) has - see
245
+ eso.detection.slippage.slippage_sites_for_constraints for the concrete
246
+ failure this avoids. Kept as a same-named counterpart so
247
+ eso.detection.dispatch can call either implementation uniformly.
248
+ """
249
+ return collapse_overlapping_intervals_no_coverage_loss(df_slippage, score_col='log10_prob_slippage_ecoli')
250
+
251
+
252
+ def collapse_slippage_sites(df_slippage, num_sites=np.inf):
253
+ """Collapse raw slippage candidates (see find_slippage_candidates) down to
254
+ one representative per group of overlapping candidates, for a
255
+ human-facing "distinct sites" report or count - NOT for building
256
+ correction constraints (see slippage_sites_for_constraints instead). See
257
+ eso.detection.slippage's version of this function for why exact-start
258
+ dedup isn't enough - phase-shifted/differently-lengthed detections of the
259
+ same physical repeat need an overlap-based collapse instead.
260
+ """
261
+ df_slippage = collapse_overlapping_intervals(df_slippage, score_col='log10_prob_slippage_ecoli')
262
+ df_slippage = df_slippage.sort_values(['log10_prob_slippage_ecoli', 'length_base_unit'], ascending=[False, False])
263
+
264
+ if num_sites < np.inf:
265
+ df_slippage = df_slippage.head(int(num_sites))
266
+
267
+ return df_slippage
268
+
269
+
270
+ def find_slippage_sites(seq, num_sites=np.inf):
271
+ """ngram/frameshift-scan variant of slippage (SSR) hotspot detection,
272
+ collapsed to one representative per physical site and limited to
273
+ `num_sites` if given - the human-facing report/count view. Do NOT use
274
+ this to build correction constraints (see find_slippage_candidates).
275
+ """
276
+ return collapse_slippage_sites(find_slippage_candidates(seq), num_sites)
277
+
278
+
279
+ def suspect_site_extractor(seq, num_sites=np.inf, extension=''):
280
+ """Run both recombination and slippage detection (this module's ngram/frameshift variants)."""
281
+ return {
282
+ 'df_recombination' + extension: find_recombination_sites(seq, num_sites),
283
+ 'df_slippage' + extension: find_slippage_sites(seq, num_sites),
284
+ }
eso/io_utils.py ADDED
@@ -0,0 +1,312 @@
1
+ """FASTA/GenBank file discovery and input validation."""
2
+
3
+ import gzip
4
+ import json
5
+ import os
6
+ from os import path
7
+ from pathlib import Path
8
+
9
+ from Bio import SeqIO
10
+ from dnachisel import biotools
11
+
12
+ from eso.sequence_utils import parse_region
13
+
14
+ FILE_ENDINGS = {
15
+ 'fasta': ['fasta', 'fna', 'ffn', 'faa', 'frn', 'fa'],
16
+ 'genbank': ['genbank', 'gb', 'gbk'],
17
+ }
18
+
19
+
20
+ def file_stem(filepath):
21
+ """Basename with its extension(s) removed, preserving any other dots in
22
+ the stem itself. `path.basename(filepath).split('.')[0]` (used to
23
+ determine this) truncated at the FIRST dot, so a filename like
24
+ "sample.2024.fasta" silently became the stem "sample" instead of
25
+ "sample.2024" - meaning an `indexes` entry keyed on the intended full
26
+ stem would never match, silently falling back to no ORF/exclusion
27
+ regions instead of erroring.
28
+ """
29
+ name = path.basename(filepath)
30
+ if name.endswith('.gz'):
31
+ name = name[:-3]
32
+ return path.splitext(name)[0]
33
+
34
+
35
+ class IndexesFileError(Exception):
36
+ """An --indexes-file failed to load or was malformed.
37
+
38
+ Raised with a plain-English message aimed at someone filling in ORF and
39
+ exclusion regions for their sequences, not a Python or JSON expert - the
40
+ message alone should be enough to fix the problem.
41
+ """
42
+
43
+
44
+ def load_indexes_from_file(file_path):
45
+ """Load the `indexes` argument to eso.pipeline.main from a JSON file, for
46
+ use as the CLI's `--indexes-file`.
47
+
48
+ The file must be a JSON list of objects, each with:
49
+ - "file": the FASTA/GenBank file's stem (no extension - matches
50
+ eso.io_utils.file_stem), e.g. "my_gene" for "my_gene.fasta".
51
+ - "seq_index": which record within that file, as a string, 0-indexed in
52
+ file order, e.g. "0" for the first sequence.
53
+ - "orf_regions": 1-indexed, inclusive region string, e.g. "1-6, 51-68".
54
+ - "exclusion_regions": same format; omit or use "" for no exclusions.
55
+
56
+ Region strings themselves are validated later, the same way as when
57
+ `indexes` is passed directly to eso.pipeline.main - this only checks the
58
+ file's own structure (valid JSON, a list, well-formed entries).
59
+
60
+ Returns
61
+ -------
62
+ dict mapping (file_stem, seq_index) -> (orf_regions, exclusion_regions),
63
+ matching eso.pipeline.main's `indexes` parameter.
64
+ """
65
+ if not path.isfile(file_path):
66
+ raise IndexesFileError(
67
+ f"Can't find the indexes file '{file_path}'. Check the path is correct.")
68
+
69
+ with open(file_path, "r", encoding="utf-8") as handle:
70
+ try:
71
+ entries = json.load(handle)
72
+ except json.JSONDecodeError as e:
73
+ raise IndexesFileError(
74
+ f"'{file_path}' isn't valid JSON: {e}. Check for missing commas, quotes, or brackets."
75
+ ) from e
76
+
77
+ if not isinstance(entries, list):
78
+ raise IndexesFileError(
79
+ f"'{file_path}' must contain a JSON list of entries (got a {type(entries).__name__} instead). "
80
+ 'Each entry looks like {"file": "my_gene", "seq_index": "0", "orf_regions": "1-6, 51-68", '
81
+ '"exclusion_regions": "1-6, 50-68"}.')
82
+
83
+ indexes = {}
84
+ for ii, entry in enumerate(entries):
85
+ if not isinstance(entry, dict) or 'file' not in entry or 'seq_index' not in entry:
86
+ raise IndexesFileError(
87
+ f"Entry {ii} of '{file_path}' must be an object with at least \"file\" and "
88
+ f'"seq_index" keys - got {entry!r}.')
89
+
90
+ key = (str(entry['file']), str(entry['seq_index']))
91
+ orf_regions = str(entry.get('orf_regions', ''))
92
+ exclusion_regions = str(entry.get('exclusion_regions', ''))
93
+ indexes[key] = (orf_regions, exclusion_regions)
94
+
95
+ return indexes
96
+
97
+
98
+ def file_opener(file):
99
+ """`file` is a (filepath, filetype) tuple, filetype in {'fasta', 'genbank'}."""
100
+ filepath, filetype = file
101
+ if filepath.endswith('.gz'):
102
+ with gzip.open(filepath, "rt", encoding="utf-8") as handle:
103
+ return list(SeqIO.parse(handle, filetype))
104
+ with open(filepath, "rt", encoding="utf-8") as handle:
105
+ return list(SeqIO.parse(handle, filetype))
106
+
107
+
108
+ def _matching_filetype(filename):
109
+ """Return the FILE_ENDINGS key `filename` belongs to (case-insensitively,
110
+ ignoring a trailing .gz), or None if it doesn't match any known extension.
111
+
112
+ Matching case-insensitively (rather than relying on glob('*.fasta')) matters
113
+ because glob's case sensitivity is filesystem-dependent - Windows filesystems
114
+ are case-insensitive so `*.fasta` there also matches `GENE.FASTA`, but the
115
+ same glob call on a (case-sensitive) Mac/Linux filesystem would silently skip
116
+ it. Confirmed this is a real, reachable discrepancy, not hypothetical -
117
+ without this, a file discovery result depends on which OS the tool happens
118
+ to run on for the exact same input folder.
119
+ """
120
+ name = filename.lower()
121
+ if name.endswith('.gz'):
122
+ name = name[:-3]
123
+ if '.' not in name:
124
+ return None
125
+ ext = name.rsplit('.', 1)[-1]
126
+ for filetype, endings in FILE_ENDINGS.items():
127
+ if ext in endings:
128
+ return filetype
129
+ return None
130
+
131
+
132
+ def relevant_file_paths(input_folder=None):
133
+ """Find all FASTA/GenBank files (optionally gzipped) directly in or one level
134
+ under `input_folder`, returned as a list of (filepath, filetype) tuples.
135
+ """
136
+ if input_folder is None:
137
+ input_folder = os.getcwd()
138
+ input_as_path = Path(input_folder)
139
+
140
+ candidates = [p for p in input_as_path.glob('*') if p.is_file()]
141
+ for entry in input_as_path.glob('*'):
142
+ if entry.is_dir():
143
+ candidates.extend(p for p in entry.glob('*') if p.is_file())
144
+
145
+ files = []
146
+ for candidate in candidates:
147
+ filetype = _matching_filetype(candidate.name)
148
+ if filetype is not None:
149
+ files.append((str(candidate), filetype))
150
+ return files
151
+
152
+
153
+ def exclusion_gc_tester(file, indexes):
154
+ """Given a file's exclusion (locked) regions, compute the range of GC-content
155
+ limits that can still legally be enforced across a sliding window that
156
+ partially overlaps those regions.
157
+ """
158
+ data = file_opener(file)
159
+ filename_indexes = file_stem(file[0])
160
+
161
+ legal_mini_gc = 1.0
162
+ legal_maxi_gc = 0.0
163
+
164
+ for ii, record in enumerate(data):
165
+ example_seq = str(record.seq).upper()
166
+ windowed_gc_content = biotools.gc_content(example_seq, window_size=50)
167
+ seq_indexes = str(ii)
168
+
169
+ if (filename_indexes, seq_indexes) not in indexes:
170
+ exclusion_regions = ()
171
+ else:
172
+ relevant_index_data = indexes[(filename_indexes, seq_indexes)]
173
+ exclusion_regions = parse_region(relevant_index_data[1])
174
+
175
+ for region in exclusion_regions:
176
+ start_ind, end_ind = region
177
+ curr_min_gc, curr_max_gc = 1.0, 0.0
178
+
179
+ if start_ind + 50 <= end_ind:
180
+ curr_max_gc = max(windowed_gc_content[start_ind:end_ind - 50])
181
+ curr_min_gc = min(windowed_gc_content[start_ind:end_ind - 50])
182
+
183
+ for jj in range(50):
184
+ if start_ind - jj < 0:
185
+ break
186
+ curr_region = (start_ind - jj, start_ind + 49 - jj)
187
+ overlap = ''
188
+ for ex_reg in exclusion_regions:
189
+ overlap_reg = biotools.windows_overlap(curr_region, ex_reg)
190
+ if overlap_reg is not None:
191
+ overlap += example_seq[overlap_reg[0]:overlap_reg[1]]
192
+ # biotools.gc_content('') is 0/0 (NaN, with a RuntimeWarning) -
193
+ # `overlap` is empty whenever none of exclusion_regions actually
194
+ # overlaps this particular sliding window, which does happen (e.g.
195
+ # a window entirely before the region under test). The GC content
196
+ # of an empty overlap contributes nothing regardless, so skip the
197
+ # call rather than asking for a GC content that doesn't exist.
198
+ curr_gc = biotools.gc_content(overlap) if overlap else 0.0
199
+ curr_max_gc = max(curr_max_gc, curr_gc * len(overlap) / 50.0)
200
+ curr_min_gc = min(curr_min_gc, curr_gc * len(overlap) / 50.0 + (1.0 - len(overlap) / 50.0))
201
+
202
+ for jj in range(50):
203
+ if end_ind + jj >= len(example_seq):
204
+ break
205
+ curr_region = (end_ind - 49 + jj, end_ind + jj)
206
+ overlap = ''
207
+ for ex_reg in exclusion_regions:
208
+ overlap_reg = biotools.windows_overlap(curr_region, ex_reg)
209
+ if overlap_reg is not None:
210
+ overlap += example_seq[overlap_reg[0]:overlap_reg[1]]
211
+ curr_gc = biotools.gc_content(overlap) if overlap else 0.0
212
+ curr_max_gc = max(curr_max_gc, curr_gc * len(overlap) / 50.0)
213
+ curr_min_gc = min(curr_min_gc, curr_gc * len(overlap) / 50.0 + (1.0 - len(overlap) / 50.0))
214
+
215
+ legal_mini_gc = min(curr_min_gc, legal_mini_gc)
216
+ legal_maxi_gc = max(curr_max_gc, legal_maxi_gc)
217
+
218
+ return legal_mini_gc, legal_maxi_gc
219
+
220
+
221
+ def _sequence_lengths_by_index_key(files):
222
+ """Map (file_stem, seq_index_str) -> sequence length, for every record in
223
+ `files` - lets ORF/exclusion regions be checked against the actual
224
+ sequence they'll apply to, not just against each other.
225
+ """
226
+ lengths = {}
227
+ for file in files:
228
+ filename_indexes = file_stem(file[0])
229
+ for ii, record in enumerate(file_opener(file)):
230
+ lengths[(filename_indexes, str(ii))] = len(record.seq)
231
+ return lengths
232
+
233
+
234
+ def test_input(mini_gc, maxi_gc, indexes, files):
235
+ """Validate GC-content bounds and ORF/exclusion region formatting before running.
236
+
237
+ `indexes` maps (filename, seq_index) -> (orf_region_string, exclusion_region_string),
238
+ e.g. {"myfile": ("1-9, 21-29", "")} - 1-indexed, inclusive, as entered by a biologist.
239
+ """
240
+ if mini_gc < 0.0:
241
+ return 'The minimal GC content must be at least 0!'
242
+ if maxi_gc > 1.0:
243
+ return 'The maximal GC content must be no more than 1!'
244
+ if mini_gc >= maxi_gc:
245
+ return 'The minimal GC content must be less than the maximum!'
246
+
247
+ if len(indexes) > 0:
248
+ # only checked against a record when the (file, seq_index) key
249
+ # actually matches one - an unmatched key is a separate, pre-existing
250
+ # leniency (eso.io_utils.file_stem's docstring, eso.pipeline.backend's
251
+ # `if key not in indexes` check) this doesn't attempt to change.
252
+ seq_lengths = _sequence_lengths_by_index_key(files)
253
+
254
+ for key, index_value in indexes.items():
255
+ seq_length = seq_lengths.get(key)
256
+
257
+ orf_indexes = parse_region(index_value[0])
258
+ if orf_indexes == 'error':
259
+ return 'ORF regions must be formatted as "start_1-end_1,start_2-end_2,...", for example "1-9, 21-29"'
260
+ for curr_ind in orf_indexes:
261
+ if curr_ind[0] < 0:
262
+ return 'Start index must be greater than 0!'
263
+ if curr_ind[0] >= curr_ind[1]:
264
+ return 'Start index must be smaller than end index!'
265
+ if (curr_ind[1] - curr_ind[0]) % 3 != 0:
266
+ return 'Indexes must describe sequence length divisible by 3!'
267
+ # without this, an ORF region's end past the actual sequence
268
+ # length isn't rejected here - Python's forgiving slicing
269
+ # (seq[0:6000] on a 500nt string just returns 500nt) means a
270
+ # region meant for a different/longer sequence would silently
271
+ # get truncated instead of erroring, with no indication the
272
+ # requested region wasn't what was actually applied.
273
+ if seq_length is not None and curr_ind[1] > seq_length:
274
+ return (
275
+ f'ORF region {curr_ind[0] + 1}-{curr_ind[1]} (1-indexed) goes past the end of '
276
+ f'{key[0]!r} sequence {key[1]} ({seq_length}nt long)!'
277
+ )
278
+
279
+ if index_value[1] not in ('', 'None'):
280
+ region_indexes = parse_region(index_value[1])
281
+ if region_indexes == 'error':
282
+ return 'Exclusion regions must be formatted as "start_1-end_1,start_2-end_2,...", for example "1-8, 50-103"'
283
+ # was `for curr_ind in orf_indexes` - validated the already-checked
284
+ # ORF indexes a second time and never actually checked the
285
+ # exclusion regions themselves, so a malformed exclusion region
286
+ # (e.g. start >= end) slipped through validation entirely and
287
+ # only surfaced later as a confusing, unrelated GC-limit error.
288
+ for curr_ind in region_indexes:
289
+ if curr_ind[0] < 0:
290
+ return 'Start index must be greater than 0, also for exclusion sites!!'
291
+ if curr_ind[0] >= curr_ind[1]:
292
+ return 'Start index must be smaller than end index, also for exclusion sites!'
293
+ if seq_length is not None and curr_ind[1] > seq_length:
294
+ return (
295
+ f'Exclusion region {curr_ind[0] + 1}-{curr_ind[1]} (1-indexed) goes past the end '
296
+ f'of {key[0]!r} sequence {key[1]} ({seq_length}nt long)!'
297
+ )
298
+
299
+ legal_mini_gc, legal_maxi_gc = 0.0, 1.0
300
+ for file in files:
301
+ curr_mini_gc, curr_maxi_gc = exclusion_gc_tester(file, indexes)
302
+ legal_mini_gc = max(legal_mini_gc, curr_mini_gc) + 0.01
303
+ legal_maxi_gc = min(legal_maxi_gc, curr_maxi_gc) - 0.01
304
+
305
+ if legal_mini_gc < mini_gc or legal_maxi_gc > maxi_gc:
306
+ return (
307
+ f'Given the exclusion regions defined by the user, the maximal GC limit must be at least '
308
+ f'{legal_maxi_gc} and the minimal GC limit must be at most {legal_mini_gc}. Note that the '
309
+ f'maximal/minimal GC limit required might be a bit higher/lower to account also for ORF regions.'
310
+ )
311
+
312
+ return 'Success!'