evolutionary-stability-optimizer 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- eso/__init__.py +11 -0
- eso/cli.py +111 -0
- eso/codon_usage.py +96 -0
- eso/constraints.py +179 -0
- eso/custom_score.py +170 -0
- eso/data/__init__.py +0 -0
- eso/data/human-antibody-heavy-chain-codon-frequencies.csv +62 -0
- eso/data/human-antibody-light-chain-codon-frequencies.csv +62 -0
- eso/detection/__init__.py +0 -0
- eso/detection/_overlap.py +112 -0
- eso/detection/common_motifs.py +81 -0
- eso/detection/dispatch.py +185 -0
- eso/detection/methylation.py +96 -0
- eso/detection/motif_utils.py +77 -0
- eso/detection/recombination.py +329 -0
- eso/detection/slippage.py +249 -0
- eso/detection/staubility_variant.py +284 -0
- eso/io_utils.py +312 -0
- eso/optimize.py +266 -0
- eso/pipeline.py +322 -0
- eso/report.py +57 -0
- eso/sequence_utils.py +41 -0
- evolutionary_stability_optimizer-0.1.0.dist-info/METADATA +501 -0
- evolutionary_stability_optimizer-0.1.0.dist-info/RECORD +27 -0
- evolutionary_stability_optimizer-0.1.0.dist-info/WHEEL +4 -0
- evolutionary_stability_optimizer-0.1.0.dist-info/entry_points.txt +3 -0
- evolutionary_stability_optimizer-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,284 @@
|
|
|
1
|
+
"""Alternate recombination/slippage detectors, developed independently inside the
|
|
2
|
+
STABLES project (Staubility_Code_shimshi.py) using a different algorithmic
|
|
3
|
+
approach than eso.detection.recombination/slippage:
|
|
4
|
+
|
|
5
|
+
- Recombination: ngram-counting (CountVectorizer, 16-mers) + exact re-match,
|
|
6
|
+
vs. the Levenshtein-neighbor generation approach in eso.detection.recombination.
|
|
7
|
+
- Slippage: per-frameshift scan for back-to-back repeats, vs. the
|
|
8
|
+
find-all-occurrences approach in eso.detection.slippage.
|
|
9
|
+
|
|
10
|
+
This module also carried a methylation-motif scorer (site_motif_grader/
|
|
11
|
+
calc_max_site) at one point; removed after comparison against
|
|
12
|
+
eso.detection.methylation found it scored candidates by raw sequence
|
|
13
|
+
probability with no background correction, which measurably disagrees with
|
|
14
|
+
(not just runs slower than) eso.detection.methylation's background-corrected
|
|
15
|
+
approach whenever a motif's background isn't uniform - see
|
|
16
|
+
docs/detector-comparisons.md for the full writeup, kept as a historical
|
|
17
|
+
record even though the code itself is gone.
|
|
18
|
+
|
|
19
|
+
Kept as a separate module rather than merged into the primary detectors -
|
|
20
|
+
these are two independently-arrived-at implementations of the same EFM
|
|
21
|
+
Calculator concept and haven't yet been reconciled into one canonical API.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
import re
|
|
25
|
+
|
|
26
|
+
import numpy as np
|
|
27
|
+
import pandas as pd
|
|
28
|
+
from sklearn.feature_extraction.text import CountVectorizer
|
|
29
|
+
|
|
30
|
+
from eso.detection._overlap import collapse_overlapping_intervals, collapse_overlapping_intervals_no_coverage_loss
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def genome_cutter(start, end, seq):
|
|
34
|
+
return seq[start:end]
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def find_recombination_candidates(seq):
|
|
38
|
+
"""ngram-counting variant of recombination (RMD) hotspot detection.
|
|
39
|
+
|
|
40
|
+
Unlike eso.detection.recombination's variant, this one never collapses
|
|
41
|
+
overlapping pairs (it only merges exact back-to-back 16-mer matches into
|
|
42
|
+
one site, which is a different, lossless operation) - so its output is
|
|
43
|
+
already the raw/candidate view. Kept as a separate name from
|
|
44
|
+
find_recombination_sites anyway, for a consistent candidates/collapsed
|
|
45
|
+
pair of names across both detector implementations - see
|
|
46
|
+
eso.detection.dispatch.
|
|
47
|
+
"""
|
|
48
|
+
vectorizer = CountVectorizer(analyzer='char_wb', ngram_range=(16, 16))
|
|
49
|
+
counter = vectorizer.fit_transform([seq]).toarray()
|
|
50
|
+
|
|
51
|
+
sites_recombination = list(np.where(counter > 1)[1])
|
|
52
|
+
empty_columns = [
|
|
53
|
+
'start_1', 'end_1', 'sequence', 'start_2', 'end_2',
|
|
54
|
+
'location_delta', 'site_length', 'log10_prob_recombination_ecoli', 'sequence_number',
|
|
55
|
+
]
|
|
56
|
+
if not sites_recombination:
|
|
57
|
+
return pd.DataFrame(columns=empty_columns)
|
|
58
|
+
|
|
59
|
+
all_sites = vectorizer.get_feature_names_out()
|
|
60
|
+
|
|
61
|
+
suspect_recombination = []
|
|
62
|
+
for site in sites_recombination:
|
|
63
|
+
curr_seq = all_sites[site]
|
|
64
|
+
list_regions = [match.span() for match in re.finditer(curr_seq.upper(), seq)]
|
|
65
|
+
suspect_recombination.extend(list_regions)
|
|
66
|
+
|
|
67
|
+
suspect_recombination = sorted(suspect_recombination)
|
|
68
|
+
df_recombination = pd.DataFrame(suspect_recombination, columns=['start', 'end'])
|
|
69
|
+
|
|
70
|
+
# merge back-to-back 16-mer matches (from repeats longer than 16nt) into one site
|
|
71
|
+
df_recombination.loc[:, 'start_delta'] = df_recombination['start'] - df_recombination['start'].shift()
|
|
72
|
+
df_recombination.loc[:, 'end_delta'] = df_recombination['end'].shift(-1) - df_recombination['end']
|
|
73
|
+
df_recombination = df_recombination[(df_recombination.start_delta != 1.0) | (df_recombination.end_delta != 1.0)]
|
|
74
|
+
|
|
75
|
+
df_recombination.loc[(df_recombination.end_delta == 1.0), 'end'] = None
|
|
76
|
+
# `df.loc[:, 'end'] = ...` would silently keep the column's existing float64
|
|
77
|
+
# dtype (from the None assignment above) even after .astype(int) below;
|
|
78
|
+
# only whole-column reassignment (df['end'] = ...) actually changes dtype.
|
|
79
|
+
df_recombination['end'] = df_recombination.loc[:, 'end'].bfill().astype(int) - 1
|
|
80
|
+
df_recombination = df_recombination[df_recombination.start_delta != 1.0][['start', 'end']]
|
|
81
|
+
|
|
82
|
+
df_recombination.loc[:, 'sequence'] = df_recombination.apply(
|
|
83
|
+
lambda x: genome_cutter(x['start'], x['end'], seq), axis=1)
|
|
84
|
+
|
|
85
|
+
df_recombination = df_recombination.merge(df_recombination, on='sequence', suffixes=('_1', '_2'))
|
|
86
|
+
df_recombination = df_recombination[df_recombination.end_1 < df_recombination.start_2]
|
|
87
|
+
|
|
88
|
+
df_recombination.loc[:, 'location_delta'] = df_recombination.start_2 - df_recombination.end_1
|
|
89
|
+
df_recombination.loc[:, 'site_length'] = df_recombination.end_1 - df_recombination.start_1
|
|
90
|
+
|
|
91
|
+
# a=5.8 per Oliveira et al. 2008, Table 3, recA+ row (see
|
|
92
|
+
# eso.detection.recombination.calc_recombination_score's docstring for
|
|
93
|
+
# the full story, including why the EFM Calculator's own 8.8 is a
|
|
94
|
+
# mix-up with a different row/parameter of that same table).
|
|
95
|
+
a, b, c, alpha = 5.8, 1465.6, 0, 29
|
|
96
|
+
base = a + df_recombination['location_delta']
|
|
97
|
+
exponent = -1 * alpha / df_recombination['site_length']
|
|
98
|
+
scale = df_recombination['site_length'] / (
|
|
99
|
+
1 + b * df_recombination['site_length'] + c * df_recombination['location_delta'])
|
|
100
|
+
df_recombination.loc[:, 'log10_prob_recombination_ecoli'] = np.log10((base ** exponent) * scale)
|
|
101
|
+
|
|
102
|
+
return df_recombination.sort_values('log10_prob_recombination_ecoli', ascending=False).reset_index(drop=True)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def collapse_recombination_sites(df_recombination, num_sites=np.inf):
|
|
106
|
+
"""No-op collapse: this detector never produces overlapping-but-distinct
|
|
107
|
+
pairs the way eso.detection.recombination's does (it only merges exact
|
|
108
|
+
back-to-back matches, already handled in find_recombination_candidates),
|
|
109
|
+
so there is nothing to collapse here - just applies `num_sites` and drops
|
|
110
|
+
any column that's entirely null.
|
|
111
|
+
|
|
112
|
+
The null-column drop happens here (on the num_sites-limited report view),
|
|
113
|
+
not in find_recombination_candidates, matching the original pre-split
|
|
114
|
+
behavior exactly: which columns are "entirely null" can differ between
|
|
115
|
+
the full candidate set and a num_sites-limited subset of it, so doing
|
|
116
|
+
this before truncating (as an earlier version of this split briefly did)
|
|
117
|
+
could report a different column set for the same `num_sites` value than
|
|
118
|
+
before the candidates/collapse split existed.
|
|
119
|
+
"""
|
|
120
|
+
if num_sites < np.inf:
|
|
121
|
+
df_recombination = df_recombination.head(int(num_sites))
|
|
122
|
+
|
|
123
|
+
for col in df_recombination:
|
|
124
|
+
if df_recombination[col].isnull().all():
|
|
125
|
+
del df_recombination[col]
|
|
126
|
+
|
|
127
|
+
return df_recombination
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def recombination_sites_for_constraints(df_recombination):
|
|
131
|
+
"""Identity: this detector never collapses overlapping pairs in the first
|
|
132
|
+
place (see collapse_recombination_sites), so its raw candidates are
|
|
133
|
+
already the constraint-building view - kept as a same-named counterpart
|
|
134
|
+
to eso.detection.recombination's version so eso.detection.dispatch can
|
|
135
|
+
call either implementation uniformly.
|
|
136
|
+
"""
|
|
137
|
+
return df_recombination
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def find_recombination_sites(seq, num_sites=np.inf):
|
|
141
|
+
"""ngram-counting variant of recombination (RMD) hotspot detection,
|
|
142
|
+
limited to `num_sites` if given. See find_recombination_candidates -
|
|
143
|
+
this detector never needs to collapse, so this is only a thin
|
|
144
|
+
num_sites-limiting wrapper.
|
|
145
|
+
"""
|
|
146
|
+
return collapse_recombination_sites(find_recombination_candidates(seq), num_sites)
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def find_slippage_sites_length_l(seq, length):
|
|
150
|
+
"""Per-frameshift scan for a repeated base unit of size `length`
|
|
151
|
+
(repeated 3+ times, or 12+ times for length=1 - see
|
|
152
|
+
eso.detection.slippage's module docstring for why 12: below that, an
|
|
153
|
+
independently-run length-2 detection of the same physical region always
|
|
154
|
+
outscores it, so its result is always discarded downstream).
|
|
155
|
+
"""
|
|
156
|
+
from textwrap import wrap
|
|
157
|
+
|
|
158
|
+
slippage_sites = []
|
|
159
|
+
homopolymer_window = 12
|
|
160
|
+
|
|
161
|
+
for frameshift in range(length):
|
|
162
|
+
curr_seq = seq[frameshift:]
|
|
163
|
+
curr_seq_split = wrap(curr_seq, length)
|
|
164
|
+
|
|
165
|
+
end_of_range = len(curr_seq_split) - 2
|
|
166
|
+
|
|
167
|
+
for ii in range(end_of_range):
|
|
168
|
+
# `length == 1` must be checked FIRST so Python's `and`
|
|
169
|
+
# short-circuits before touching curr_seq_split[ii + 1/2] when
|
|
170
|
+
# length > 1 (needed for is_followed2 below); is_followed1 itself
|
|
171
|
+
# uses slicing, which never raises IndexError even past the list
|
|
172
|
+
# end, so it needs no equivalent bounds bug to worry about - a
|
|
173
|
+
# slice shorter than `homopolymer_window` just fails the length
|
|
174
|
+
# check instead of crashing.
|
|
175
|
+
is_followed2 = (
|
|
176
|
+
length > 1
|
|
177
|
+
and curr_seq_split[ii] == curr_seq_split[ii + 1]
|
|
178
|
+
and curr_seq_split[ii] == curr_seq_split[ii + 2]
|
|
179
|
+
)
|
|
180
|
+
window = curr_seq_split[ii:ii + homopolymer_window]
|
|
181
|
+
is_followed1 = length == 1 and len(window) == homopolymer_window and len(set(window)) == 1
|
|
182
|
+
|
|
183
|
+
if is_followed2:
|
|
184
|
+
slippage_sites.append((frameshift + ii * length, frameshift + length * (ii + 3)))
|
|
185
|
+
if is_followed1:
|
|
186
|
+
slippage_sites.append((ii, ii + homopolymer_window))
|
|
187
|
+
|
|
188
|
+
if not slippage_sites:
|
|
189
|
+
return pd.DataFrame(columns=['start', 'end', 'sequence', 'length_base_unit'])
|
|
190
|
+
|
|
191
|
+
df_slippage = pd.DataFrame(sorted(slippage_sites), columns=['start', 'end'])
|
|
192
|
+
|
|
193
|
+
df_slippage.loc[:, 'start_delta'] = df_slippage['start'] - df_slippage['start'].shift()
|
|
194
|
+
df_slippage.loc[:, 'end_delta'] = df_slippage['end'].shift(-1) - df_slippage['end']
|
|
195
|
+
df_slippage = df_slippage[(df_slippage.start_delta != 1.0) | (df_slippage.end_delta != 1.0)]
|
|
196
|
+
df_slippage.loc[(df_slippage.end_delta == 1.0), 'end'] = None
|
|
197
|
+
# see the identical fix in find_recombination_sites above: whole-column
|
|
198
|
+
# reassignment is required for .astype(int) to actually stick here
|
|
199
|
+
df_slippage['end'] = df_slippage.loc[:, 'end'].bfill().astype(int)
|
|
200
|
+
df_slippage = df_slippage[df_slippage.start_delta != 1.0][['start', 'end']]
|
|
201
|
+
|
|
202
|
+
df_slippage.loc[:, 'end'] = (
|
|
203
|
+
((df_slippage['end'] - df_slippage['start']) / length).astype(int) * length
|
|
204
|
+
) + df_slippage['start']
|
|
205
|
+
|
|
206
|
+
df_slippage.loc[:, 'sequence'] = df_slippage.apply(
|
|
207
|
+
lambda x: genome_cutter(x['start'], x['end'], seq), axis=1)
|
|
208
|
+
df_slippage.loc[:, 'length_base_unit'] = length
|
|
209
|
+
|
|
210
|
+
return df_slippage
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def find_slippage_candidates(seq):
|
|
214
|
+
"""ngram/frameshift-scan variant of slippage (SSR) hotspot detection -
|
|
215
|
+
every candidate, WITHOUT collapsing overlapping ones down to one
|
|
216
|
+
representative per physical site. See
|
|
217
|
+
eso.detection.slippage.find_slippage_candidates for why constraint-
|
|
218
|
+
building must use this, not find_slippage_sites.
|
|
219
|
+
"""
|
|
220
|
+
slippage_sites_list = [find_slippage_sites_length_l(seq, length) for length in range(1, 16)]
|
|
221
|
+
df_slippage = pd.concat(slippage_sites_list, ignore_index=True)[['start', 'end', 'length_base_unit', 'sequence']]
|
|
222
|
+
|
|
223
|
+
df_slippage.loc[:, 'num_base_units'] = (
|
|
224
|
+
df_slippage.sequence.apply(len) / df_slippage.length_base_unit
|
|
225
|
+
).astype(int)
|
|
226
|
+
|
|
227
|
+
df_slippage.loc[:, 'log10_prob_slippage_ecoli'] = -4.749 + 0.063 * df_slippage['num_base_units']
|
|
228
|
+
df_slippage.loc[df_slippage.length_base_unit == 1, 'log10_prob_slippage_ecoli'] = (
|
|
229
|
+
-12.9 + 0.729 * df_slippage['num_base_units']
|
|
230
|
+
)
|
|
231
|
+
|
|
232
|
+
# this filter was applied by the STABLES caller (STABLES_full_code's
|
|
233
|
+
# optimization pipeline), not inside find_slippage_sites itself - lost
|
|
234
|
+
# when this function was extracted in isolation. Restored here so the
|
|
235
|
+
# function is self-contained and consistent with its sibling
|
|
236
|
+
# find_recombination_candidates above, which does filter internally.
|
|
237
|
+
df_slippage = df_slippage[df_slippage.log10_prob_slippage_ecoli > -9]
|
|
238
|
+
return df_slippage.sort_values('log10_prob_slippage_ecoli', ascending=False).reset_index(drop=True)
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def slippage_sites_for_constraints(df_slippage):
|
|
242
|
+
"""Reduce raw slippage candidates (see find_slippage_candidates) for
|
|
243
|
+
feeding into correction-constraint building, WITHOUT the coverage-loss
|
|
244
|
+
risk plain non-max suppression (collapse_slippage_sites) has - see
|
|
245
|
+
eso.detection.slippage.slippage_sites_for_constraints for the concrete
|
|
246
|
+
failure this avoids. Kept as a same-named counterpart so
|
|
247
|
+
eso.detection.dispatch can call either implementation uniformly.
|
|
248
|
+
"""
|
|
249
|
+
return collapse_overlapping_intervals_no_coverage_loss(df_slippage, score_col='log10_prob_slippage_ecoli')
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def collapse_slippage_sites(df_slippage, num_sites=np.inf):
|
|
253
|
+
"""Collapse raw slippage candidates (see find_slippage_candidates) down to
|
|
254
|
+
one representative per group of overlapping candidates, for a
|
|
255
|
+
human-facing "distinct sites" report or count - NOT for building
|
|
256
|
+
correction constraints (see slippage_sites_for_constraints instead). See
|
|
257
|
+
eso.detection.slippage's version of this function for why exact-start
|
|
258
|
+
dedup isn't enough - phase-shifted/differently-lengthed detections of the
|
|
259
|
+
same physical repeat need an overlap-based collapse instead.
|
|
260
|
+
"""
|
|
261
|
+
df_slippage = collapse_overlapping_intervals(df_slippage, score_col='log10_prob_slippage_ecoli')
|
|
262
|
+
df_slippage = df_slippage.sort_values(['log10_prob_slippage_ecoli', 'length_base_unit'], ascending=[False, False])
|
|
263
|
+
|
|
264
|
+
if num_sites < np.inf:
|
|
265
|
+
df_slippage = df_slippage.head(int(num_sites))
|
|
266
|
+
|
|
267
|
+
return df_slippage
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def find_slippage_sites(seq, num_sites=np.inf):
|
|
271
|
+
"""ngram/frameshift-scan variant of slippage (SSR) hotspot detection,
|
|
272
|
+
collapsed to one representative per physical site and limited to
|
|
273
|
+
`num_sites` if given - the human-facing report/count view. Do NOT use
|
|
274
|
+
this to build correction constraints (see find_slippage_candidates).
|
|
275
|
+
"""
|
|
276
|
+
return collapse_slippage_sites(find_slippage_candidates(seq), num_sites)
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
def suspect_site_extractor(seq, num_sites=np.inf, extension=''):
|
|
280
|
+
"""Run both recombination and slippage detection (this module's ngram/frameshift variants)."""
|
|
281
|
+
return {
|
|
282
|
+
'df_recombination' + extension: find_recombination_sites(seq, num_sites),
|
|
283
|
+
'df_slippage' + extension: find_slippage_sites(seq, num_sites),
|
|
284
|
+
}
|
eso/io_utils.py
ADDED
|
@@ -0,0 +1,312 @@
|
|
|
1
|
+
"""FASTA/GenBank file discovery and input validation."""
|
|
2
|
+
|
|
3
|
+
import gzip
|
|
4
|
+
import json
|
|
5
|
+
import os
|
|
6
|
+
from os import path
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
from Bio import SeqIO
|
|
10
|
+
from dnachisel import biotools
|
|
11
|
+
|
|
12
|
+
from eso.sequence_utils import parse_region
|
|
13
|
+
|
|
14
|
+
FILE_ENDINGS = {
|
|
15
|
+
'fasta': ['fasta', 'fna', 'ffn', 'faa', 'frn', 'fa'],
|
|
16
|
+
'genbank': ['genbank', 'gb', 'gbk'],
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def file_stem(filepath):
|
|
21
|
+
"""Basename with its extension(s) removed, preserving any other dots in
|
|
22
|
+
the stem itself. `path.basename(filepath).split('.')[0]` (used to
|
|
23
|
+
determine this) truncated at the FIRST dot, so a filename like
|
|
24
|
+
"sample.2024.fasta" silently became the stem "sample" instead of
|
|
25
|
+
"sample.2024" - meaning an `indexes` entry keyed on the intended full
|
|
26
|
+
stem would never match, silently falling back to no ORF/exclusion
|
|
27
|
+
regions instead of erroring.
|
|
28
|
+
"""
|
|
29
|
+
name = path.basename(filepath)
|
|
30
|
+
if name.endswith('.gz'):
|
|
31
|
+
name = name[:-3]
|
|
32
|
+
return path.splitext(name)[0]
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class IndexesFileError(Exception):
|
|
36
|
+
"""An --indexes-file failed to load or was malformed.
|
|
37
|
+
|
|
38
|
+
Raised with a plain-English message aimed at someone filling in ORF and
|
|
39
|
+
exclusion regions for their sequences, not a Python or JSON expert - the
|
|
40
|
+
message alone should be enough to fix the problem.
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def load_indexes_from_file(file_path):
|
|
45
|
+
"""Load the `indexes` argument to eso.pipeline.main from a JSON file, for
|
|
46
|
+
use as the CLI's `--indexes-file`.
|
|
47
|
+
|
|
48
|
+
The file must be a JSON list of objects, each with:
|
|
49
|
+
- "file": the FASTA/GenBank file's stem (no extension - matches
|
|
50
|
+
eso.io_utils.file_stem), e.g. "my_gene" for "my_gene.fasta".
|
|
51
|
+
- "seq_index": which record within that file, as a string, 0-indexed in
|
|
52
|
+
file order, e.g. "0" for the first sequence.
|
|
53
|
+
- "orf_regions": 1-indexed, inclusive region string, e.g. "1-6, 51-68".
|
|
54
|
+
- "exclusion_regions": same format; omit or use "" for no exclusions.
|
|
55
|
+
|
|
56
|
+
Region strings themselves are validated later, the same way as when
|
|
57
|
+
`indexes` is passed directly to eso.pipeline.main - this only checks the
|
|
58
|
+
file's own structure (valid JSON, a list, well-formed entries).
|
|
59
|
+
|
|
60
|
+
Returns
|
|
61
|
+
-------
|
|
62
|
+
dict mapping (file_stem, seq_index) -> (orf_regions, exclusion_regions),
|
|
63
|
+
matching eso.pipeline.main's `indexes` parameter.
|
|
64
|
+
"""
|
|
65
|
+
if not path.isfile(file_path):
|
|
66
|
+
raise IndexesFileError(
|
|
67
|
+
f"Can't find the indexes file '{file_path}'. Check the path is correct.")
|
|
68
|
+
|
|
69
|
+
with open(file_path, "r", encoding="utf-8") as handle:
|
|
70
|
+
try:
|
|
71
|
+
entries = json.load(handle)
|
|
72
|
+
except json.JSONDecodeError as e:
|
|
73
|
+
raise IndexesFileError(
|
|
74
|
+
f"'{file_path}' isn't valid JSON: {e}. Check for missing commas, quotes, or brackets."
|
|
75
|
+
) from e
|
|
76
|
+
|
|
77
|
+
if not isinstance(entries, list):
|
|
78
|
+
raise IndexesFileError(
|
|
79
|
+
f"'{file_path}' must contain a JSON list of entries (got a {type(entries).__name__} instead). "
|
|
80
|
+
'Each entry looks like {"file": "my_gene", "seq_index": "0", "orf_regions": "1-6, 51-68", '
|
|
81
|
+
'"exclusion_regions": "1-6, 50-68"}.')
|
|
82
|
+
|
|
83
|
+
indexes = {}
|
|
84
|
+
for ii, entry in enumerate(entries):
|
|
85
|
+
if not isinstance(entry, dict) or 'file' not in entry or 'seq_index' not in entry:
|
|
86
|
+
raise IndexesFileError(
|
|
87
|
+
f"Entry {ii} of '{file_path}' must be an object with at least \"file\" and "
|
|
88
|
+
f'"seq_index" keys - got {entry!r}.')
|
|
89
|
+
|
|
90
|
+
key = (str(entry['file']), str(entry['seq_index']))
|
|
91
|
+
orf_regions = str(entry.get('orf_regions', ''))
|
|
92
|
+
exclusion_regions = str(entry.get('exclusion_regions', ''))
|
|
93
|
+
indexes[key] = (orf_regions, exclusion_regions)
|
|
94
|
+
|
|
95
|
+
return indexes
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def file_opener(file):
|
|
99
|
+
"""`file` is a (filepath, filetype) tuple, filetype in {'fasta', 'genbank'}."""
|
|
100
|
+
filepath, filetype = file
|
|
101
|
+
if filepath.endswith('.gz'):
|
|
102
|
+
with gzip.open(filepath, "rt", encoding="utf-8") as handle:
|
|
103
|
+
return list(SeqIO.parse(handle, filetype))
|
|
104
|
+
with open(filepath, "rt", encoding="utf-8") as handle:
|
|
105
|
+
return list(SeqIO.parse(handle, filetype))
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _matching_filetype(filename):
|
|
109
|
+
"""Return the FILE_ENDINGS key `filename` belongs to (case-insensitively,
|
|
110
|
+
ignoring a trailing .gz), or None if it doesn't match any known extension.
|
|
111
|
+
|
|
112
|
+
Matching case-insensitively (rather than relying on glob('*.fasta')) matters
|
|
113
|
+
because glob's case sensitivity is filesystem-dependent - Windows filesystems
|
|
114
|
+
are case-insensitive so `*.fasta` there also matches `GENE.FASTA`, but the
|
|
115
|
+
same glob call on a (case-sensitive) Mac/Linux filesystem would silently skip
|
|
116
|
+
it. Confirmed this is a real, reachable discrepancy, not hypothetical -
|
|
117
|
+
without this, a file discovery result depends on which OS the tool happens
|
|
118
|
+
to run on for the exact same input folder.
|
|
119
|
+
"""
|
|
120
|
+
name = filename.lower()
|
|
121
|
+
if name.endswith('.gz'):
|
|
122
|
+
name = name[:-3]
|
|
123
|
+
if '.' not in name:
|
|
124
|
+
return None
|
|
125
|
+
ext = name.rsplit('.', 1)[-1]
|
|
126
|
+
for filetype, endings in FILE_ENDINGS.items():
|
|
127
|
+
if ext in endings:
|
|
128
|
+
return filetype
|
|
129
|
+
return None
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def relevant_file_paths(input_folder=None):
|
|
133
|
+
"""Find all FASTA/GenBank files (optionally gzipped) directly in or one level
|
|
134
|
+
under `input_folder`, returned as a list of (filepath, filetype) tuples.
|
|
135
|
+
"""
|
|
136
|
+
if input_folder is None:
|
|
137
|
+
input_folder = os.getcwd()
|
|
138
|
+
input_as_path = Path(input_folder)
|
|
139
|
+
|
|
140
|
+
candidates = [p for p in input_as_path.glob('*') if p.is_file()]
|
|
141
|
+
for entry in input_as_path.glob('*'):
|
|
142
|
+
if entry.is_dir():
|
|
143
|
+
candidates.extend(p for p in entry.glob('*') if p.is_file())
|
|
144
|
+
|
|
145
|
+
files = []
|
|
146
|
+
for candidate in candidates:
|
|
147
|
+
filetype = _matching_filetype(candidate.name)
|
|
148
|
+
if filetype is not None:
|
|
149
|
+
files.append((str(candidate), filetype))
|
|
150
|
+
return files
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def exclusion_gc_tester(file, indexes):
|
|
154
|
+
"""Given a file's exclusion (locked) regions, compute the range of GC-content
|
|
155
|
+
limits that can still legally be enforced across a sliding window that
|
|
156
|
+
partially overlaps those regions.
|
|
157
|
+
"""
|
|
158
|
+
data = file_opener(file)
|
|
159
|
+
filename_indexes = file_stem(file[0])
|
|
160
|
+
|
|
161
|
+
legal_mini_gc = 1.0
|
|
162
|
+
legal_maxi_gc = 0.0
|
|
163
|
+
|
|
164
|
+
for ii, record in enumerate(data):
|
|
165
|
+
example_seq = str(record.seq).upper()
|
|
166
|
+
windowed_gc_content = biotools.gc_content(example_seq, window_size=50)
|
|
167
|
+
seq_indexes = str(ii)
|
|
168
|
+
|
|
169
|
+
if (filename_indexes, seq_indexes) not in indexes:
|
|
170
|
+
exclusion_regions = ()
|
|
171
|
+
else:
|
|
172
|
+
relevant_index_data = indexes[(filename_indexes, seq_indexes)]
|
|
173
|
+
exclusion_regions = parse_region(relevant_index_data[1])
|
|
174
|
+
|
|
175
|
+
for region in exclusion_regions:
|
|
176
|
+
start_ind, end_ind = region
|
|
177
|
+
curr_min_gc, curr_max_gc = 1.0, 0.0
|
|
178
|
+
|
|
179
|
+
if start_ind + 50 <= end_ind:
|
|
180
|
+
curr_max_gc = max(windowed_gc_content[start_ind:end_ind - 50])
|
|
181
|
+
curr_min_gc = min(windowed_gc_content[start_ind:end_ind - 50])
|
|
182
|
+
|
|
183
|
+
for jj in range(50):
|
|
184
|
+
if start_ind - jj < 0:
|
|
185
|
+
break
|
|
186
|
+
curr_region = (start_ind - jj, start_ind + 49 - jj)
|
|
187
|
+
overlap = ''
|
|
188
|
+
for ex_reg in exclusion_regions:
|
|
189
|
+
overlap_reg = biotools.windows_overlap(curr_region, ex_reg)
|
|
190
|
+
if overlap_reg is not None:
|
|
191
|
+
overlap += example_seq[overlap_reg[0]:overlap_reg[1]]
|
|
192
|
+
# biotools.gc_content('') is 0/0 (NaN, with a RuntimeWarning) -
|
|
193
|
+
# `overlap` is empty whenever none of exclusion_regions actually
|
|
194
|
+
# overlaps this particular sliding window, which does happen (e.g.
|
|
195
|
+
# a window entirely before the region under test). The GC content
|
|
196
|
+
# of an empty overlap contributes nothing regardless, so skip the
|
|
197
|
+
# call rather than asking for a GC content that doesn't exist.
|
|
198
|
+
curr_gc = biotools.gc_content(overlap) if overlap else 0.0
|
|
199
|
+
curr_max_gc = max(curr_max_gc, curr_gc * len(overlap) / 50.0)
|
|
200
|
+
curr_min_gc = min(curr_min_gc, curr_gc * len(overlap) / 50.0 + (1.0 - len(overlap) / 50.0))
|
|
201
|
+
|
|
202
|
+
for jj in range(50):
|
|
203
|
+
if end_ind + jj >= len(example_seq):
|
|
204
|
+
break
|
|
205
|
+
curr_region = (end_ind - 49 + jj, end_ind + jj)
|
|
206
|
+
overlap = ''
|
|
207
|
+
for ex_reg in exclusion_regions:
|
|
208
|
+
overlap_reg = biotools.windows_overlap(curr_region, ex_reg)
|
|
209
|
+
if overlap_reg is not None:
|
|
210
|
+
overlap += example_seq[overlap_reg[0]:overlap_reg[1]]
|
|
211
|
+
curr_gc = biotools.gc_content(overlap) if overlap else 0.0
|
|
212
|
+
curr_max_gc = max(curr_max_gc, curr_gc * len(overlap) / 50.0)
|
|
213
|
+
curr_min_gc = min(curr_min_gc, curr_gc * len(overlap) / 50.0 + (1.0 - len(overlap) / 50.0))
|
|
214
|
+
|
|
215
|
+
legal_mini_gc = min(curr_min_gc, legal_mini_gc)
|
|
216
|
+
legal_maxi_gc = max(curr_max_gc, legal_maxi_gc)
|
|
217
|
+
|
|
218
|
+
return legal_mini_gc, legal_maxi_gc
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def _sequence_lengths_by_index_key(files):
|
|
222
|
+
"""Map (file_stem, seq_index_str) -> sequence length, for every record in
|
|
223
|
+
`files` - lets ORF/exclusion regions be checked against the actual
|
|
224
|
+
sequence they'll apply to, not just against each other.
|
|
225
|
+
"""
|
|
226
|
+
lengths = {}
|
|
227
|
+
for file in files:
|
|
228
|
+
filename_indexes = file_stem(file[0])
|
|
229
|
+
for ii, record in enumerate(file_opener(file)):
|
|
230
|
+
lengths[(filename_indexes, str(ii))] = len(record.seq)
|
|
231
|
+
return lengths
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def test_input(mini_gc, maxi_gc, indexes, files):
|
|
235
|
+
"""Validate GC-content bounds and ORF/exclusion region formatting before running.
|
|
236
|
+
|
|
237
|
+
`indexes` maps (filename, seq_index) -> (orf_region_string, exclusion_region_string),
|
|
238
|
+
e.g. {"myfile": ("1-9, 21-29", "")} - 1-indexed, inclusive, as entered by a biologist.
|
|
239
|
+
"""
|
|
240
|
+
if mini_gc < 0.0:
|
|
241
|
+
return 'The minimal GC content must be at least 0!'
|
|
242
|
+
if maxi_gc > 1.0:
|
|
243
|
+
return 'The maximal GC content must be no more than 1!'
|
|
244
|
+
if mini_gc >= maxi_gc:
|
|
245
|
+
return 'The minimal GC content must be less than the maximum!'
|
|
246
|
+
|
|
247
|
+
if len(indexes) > 0:
|
|
248
|
+
# only checked against a record when the (file, seq_index) key
|
|
249
|
+
# actually matches one - an unmatched key is a separate, pre-existing
|
|
250
|
+
# leniency (eso.io_utils.file_stem's docstring, eso.pipeline.backend's
|
|
251
|
+
# `if key not in indexes` check) this doesn't attempt to change.
|
|
252
|
+
seq_lengths = _sequence_lengths_by_index_key(files)
|
|
253
|
+
|
|
254
|
+
for key, index_value in indexes.items():
|
|
255
|
+
seq_length = seq_lengths.get(key)
|
|
256
|
+
|
|
257
|
+
orf_indexes = parse_region(index_value[0])
|
|
258
|
+
if orf_indexes == 'error':
|
|
259
|
+
return 'ORF regions must be formatted as "start_1-end_1,start_2-end_2,...", for example "1-9, 21-29"'
|
|
260
|
+
for curr_ind in orf_indexes:
|
|
261
|
+
if curr_ind[0] < 0:
|
|
262
|
+
return 'Start index must be greater than 0!'
|
|
263
|
+
if curr_ind[0] >= curr_ind[1]:
|
|
264
|
+
return 'Start index must be smaller than end index!'
|
|
265
|
+
if (curr_ind[1] - curr_ind[0]) % 3 != 0:
|
|
266
|
+
return 'Indexes must describe sequence length divisible by 3!'
|
|
267
|
+
# without this, an ORF region's end past the actual sequence
|
|
268
|
+
# length isn't rejected here - Python's forgiving slicing
|
|
269
|
+
# (seq[0:6000] on a 500nt string just returns 500nt) means a
|
|
270
|
+
# region meant for a different/longer sequence would silently
|
|
271
|
+
# get truncated instead of erroring, with no indication the
|
|
272
|
+
# requested region wasn't what was actually applied.
|
|
273
|
+
if seq_length is not None and curr_ind[1] > seq_length:
|
|
274
|
+
return (
|
|
275
|
+
f'ORF region {curr_ind[0] + 1}-{curr_ind[1]} (1-indexed) goes past the end of '
|
|
276
|
+
f'{key[0]!r} sequence {key[1]} ({seq_length}nt long)!'
|
|
277
|
+
)
|
|
278
|
+
|
|
279
|
+
if index_value[1] not in ('', 'None'):
|
|
280
|
+
region_indexes = parse_region(index_value[1])
|
|
281
|
+
if region_indexes == 'error':
|
|
282
|
+
return 'Exclusion regions must be formatted as "start_1-end_1,start_2-end_2,...", for example "1-8, 50-103"'
|
|
283
|
+
# was `for curr_ind in orf_indexes` - validated the already-checked
|
|
284
|
+
# ORF indexes a second time and never actually checked the
|
|
285
|
+
# exclusion regions themselves, so a malformed exclusion region
|
|
286
|
+
# (e.g. start >= end) slipped through validation entirely and
|
|
287
|
+
# only surfaced later as a confusing, unrelated GC-limit error.
|
|
288
|
+
for curr_ind in region_indexes:
|
|
289
|
+
if curr_ind[0] < 0:
|
|
290
|
+
return 'Start index must be greater than 0, also for exclusion sites!!'
|
|
291
|
+
if curr_ind[0] >= curr_ind[1]:
|
|
292
|
+
return 'Start index must be smaller than end index, also for exclusion sites!'
|
|
293
|
+
if seq_length is not None and curr_ind[1] > seq_length:
|
|
294
|
+
return (
|
|
295
|
+
f'Exclusion region {curr_ind[0] + 1}-{curr_ind[1]} (1-indexed) goes past the end '
|
|
296
|
+
f'of {key[0]!r} sequence {key[1]} ({seq_length}nt long)!'
|
|
297
|
+
)
|
|
298
|
+
|
|
299
|
+
legal_mini_gc, legal_maxi_gc = 0.0, 1.0
|
|
300
|
+
for file in files:
|
|
301
|
+
curr_mini_gc, curr_maxi_gc = exclusion_gc_tester(file, indexes)
|
|
302
|
+
legal_mini_gc = max(legal_mini_gc, curr_mini_gc) + 0.01
|
|
303
|
+
legal_maxi_gc = min(legal_maxi_gc, curr_maxi_gc) - 0.01
|
|
304
|
+
|
|
305
|
+
if legal_mini_gc < mini_gc or legal_maxi_gc > maxi_gc:
|
|
306
|
+
return (
|
|
307
|
+
f'Given the exclusion regions defined by the user, the maximal GC limit must be at least '
|
|
308
|
+
f'{legal_maxi_gc} and the minimal GC limit must be at most {legal_mini_gc}. Note that the '
|
|
309
|
+
f'maximal/minimal GC limit required might be a bit higher/lower to account also for ORF regions.'
|
|
310
|
+
)
|
|
311
|
+
|
|
312
|
+
return 'Success!'
|