strkit 0.11.0__tar.gz → 0.11.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {strkit-0.11.0/strkit.egg-info → strkit-0.11.1}/PKG-INFO +41 -37
- {strkit-0.11.0 → strkit-0.11.1}/README.md +39 -35
- {strkit-0.11.0 → strkit-0.11.1}/setup.py +1 -1
- strkit-0.11.1/strkit/VERSION +1 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/call/allele.py +14 -9
- {strkit-0.11.0 → strkit-0.11.1}/strkit/call/call_locus.py +47 -31
- {strkit-0.11.0 → strkit-0.11.1}/strkit/mi/base.py +15 -17
- {strkit-0.11.0 → strkit-0.11.1}/strkit/mi/result.py +4 -2
- {strkit-0.11.0 → strkit-0.11.1/strkit.egg-info}/PKG-INFO +41 -37
- {strkit-0.11.0 → strkit-0.11.1}/strkit.egg-info/requires.txt +1 -1
- strkit-0.11.0/strkit/VERSION +0 -1
- {strkit-0.11.0 → strkit-0.11.1}/LICENSE +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/MANIFEST.in +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/pyproject.toml +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/setup.cfg +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/__init__.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/call/__init__.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/call/align_matrix.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/call/call_sample.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/call/cigar.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/call/non_daemonic_pool.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/call/realign.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/call/repeats.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/call/snvs.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/call/types.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/call/utils.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/catalog/__init__.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/catalog/combine.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/constants.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/convert/__init__.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/convert/_bed_4.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/convert/converter.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/convert/expansionhunter.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/convert/gangstr.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/convert/hipstr.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/convert/straglr.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/convert/tandem_genotypes.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/entry.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/exceptions.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/json.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/logger.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/mi/__init__.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/mi/expansionhunter.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/mi/gangstr.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/mi/repeathmm.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/mi/straglr.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/mi/strkit.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/mi/tandem_genotypes.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/mi/vcf_utils.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/utils.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/viz/__init__.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/viz/server.py +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/viz/static/logo.png +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit/viz/templates/browser.html +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit.egg-info/SOURCES.txt +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit.egg-info/dependency_links.txt +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit.egg-info/entry_points.txt +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/strkit.egg-info/top_level.txt +0 -0
- {strkit-0.11.0 → strkit-0.11.1}/tests/test_caller_utils.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: strkit
|
|
3
|
-
Version: 0.11.
|
|
3
|
+
Version: 0.11.1
|
|
4
4
|
Summary: A toolkit for analyzing variation in short(ish) tandem repeats.
|
|
5
5
|
Home-page: https://github.com/davidlougheed/strkit
|
|
6
6
|
Author: David Lougheed
|
|
@@ -13,7 +13,7 @@ Requires-Python: ~=3.9
|
|
|
13
13
|
Description-Content-Type: text/markdown
|
|
14
14
|
License-File: LICENSE
|
|
15
15
|
Requires-Dist: Flask<2.3,>=2.2.5
|
|
16
|
-
Requires-Dist: pysam<0.
|
|
16
|
+
Requires-Dist: pysam<0.23,>=0.19
|
|
17
17
|
Requires-Dist: numpy<1.27,>=1.23.4
|
|
18
18
|
Requires-Dist: parasail<1.4,>=1.2.4
|
|
19
19
|
Requires-Dist: scikit-learn<1.4,>=1.2.1
|
|
@@ -31,13 +31,46 @@ A genotyping and analysis toolkit for short(ish) tandem repeats.
|
|
|
31
31
|
|
|
32
32
|
<img src="./docs/images/strkit_logo_small.png" alt="" width="500" height="324" />
|
|
33
33
|
|
|
34
|
-
© David Lougheed and McGill University 2021-2023 (versions up to and including `0.8.0a1`).
|
|
35
34
|
|
|
36
|
-
|
|
37
|
-
|
|
35
|
+
## Copyright and License
|
|
36
|
+
|
|
37
|
+
* 2021-2023: © David Lougheed (DL) and McGill University 2021-2023 (versions up to and including `0.8.0a1`),
|
|
38
|
+
created during graduate research by DL.
|
|
39
|
+
* 2023+: (versions beyond `0.8.0a1`):
|
|
40
|
+
* Portions © DL and McGill University 2021-2023
|
|
41
|
+
* Portions © McGill University 2023
|
|
42
|
+
* Portions © DL 2023
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
### Notice
|
|
46
|
+
|
|
47
|
+
This program is free software: you can redistribute it and/or modify
|
|
48
|
+
it under the terms of the GNU General Public License as published by
|
|
49
|
+
the Free Software Foundation, either version 3 of the License, or
|
|
50
|
+
(at your option) any later version.
|
|
51
|
+
|
|
52
|
+
This program is distributed in the hope that it will be useful,
|
|
53
|
+
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
54
|
+
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
55
|
+
GNU General Public License for more details.
|
|
56
|
+
|
|
57
|
+
You should have received a copy of the GNU General Public License
|
|
58
|
+
along with this program. If not, see <https://www.gnu.org/licenses/>.
|
|
59
|
+
|
|
60
|
+
### Exceptions
|
|
61
|
+
|
|
62
|
+
**Some exclusions to this license apply; specifically portions of
|
|
63
|
+
[`strkit/viz/templates/browser.html`](strkit/viz/templates/browser.html) and
|
|
64
|
+
the STRkit logo files ([./docs/images/strkit_logo_small.png](./docs/images/strkit_logo_small.png)
|
|
65
|
+
and [./strkit/viz/static/logo.png](./strkit/viz/static/logo.png).)**
|
|
38
66
|
|
|
39
67
|
The STRkit logo is © David Lougheed 2022, and was designed by Evelyn Lougheed. It is not licensed
|
|
40
|
-
under the terms of the GPL 3.0;
|
|
68
|
+
under the terms of the GPL 3.0; it is instead licensed under the terms of the
|
|
69
|
+
[CC BY-ND 4.0](https://creativecommons.org/licenses/by-nd/4.0/).
|
|
70
|
+
|
|
71
|
+
Portions of `viz/templates/browser.html` copyright (C) 2021-2022 Observable, Inc.
|
|
72
|
+
Used under the terms of the ISC license.
|
|
73
|
+
|
|
41
74
|
|
|
42
75
|
## Installation
|
|
43
76
|
|
|
@@ -99,7 +132,8 @@ strkit call \
|
|
|
99
132
|
--min-reads 4 \ # Minimum number of supporting reads needed to make a call
|
|
100
133
|
--min-allele-reads 2 \ # Minimum number of supporting reads needed to call a specific allele size
|
|
101
134
|
--flank-size 70 \ # Size of the flanking region to use on either side of a region to properly anchor reads
|
|
102
|
-
--seed 183 # Fixed random number generator seed for replicability
|
|
135
|
+
--seed 183 \ # Fixed random number generator seed for replicability
|
|
136
|
+
--processes 10 # Number of parallel processes to use; DEFAULT: 1
|
|
103
137
|
```
|
|
104
138
|
|
|
105
139
|
Ideally, you should be using a read file aligned with parameters tuned for tandem repeats.
|
|
@@ -215,33 +249,3 @@ by looking at inheritance patterns. This tool provides a few different ways to d
|
|
|
215
249
|
|
|
216
250
|
**For more information on what kind of analyses can be done with this data**, see the
|
|
217
251
|
[Trio analyses with STRkit](./docs/trio_analyses.md) page.
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
## Copyright and License
|
|
222
|
-
|
|
223
|
-
**Some exclusions to this license apply; specifically portions of
|
|
224
|
-
[`strkit/viz/templates/browser.html`](strkit/viz/templates/browser.html)
|
|
225
|
-
and files in the `catalogs` directory.**
|
|
226
|
-
|
|
227
|
-
© David Lougheed 2021-2023 (versions up to and including `0.8.0a1`).
|
|
228
|
-
|
|
229
|
-
© David Lougheed 2021-2023 with portions © McGill University 2023 (versions beyond `0.8.0a1`).
|
|
230
|
-
|
|
231
|
-
Portions of `viz/templates/browser.html` copyright (C) 2021-2022 Observable, Inc.
|
|
232
|
-
Used under the terms of the ISC license.
|
|
233
|
-
|
|
234
|
-
### Notice
|
|
235
|
-
|
|
236
|
-
This program is free software: you can redistribute it and/or modify
|
|
237
|
-
it under the terms of the GNU General Public License as published by
|
|
238
|
-
the Free Software Foundation, either version 3 of the License, or
|
|
239
|
-
(at your option) any later version.
|
|
240
|
-
|
|
241
|
-
This program is distributed in the hope that it will be useful,
|
|
242
|
-
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
243
|
-
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
244
|
-
GNU General Public License for more details.
|
|
245
|
-
|
|
246
|
-
You should have received a copy of the GNU General Public License
|
|
247
|
-
along with this program. If not, see <https://www.gnu.org/licenses/>.
|
|
@@ -6,13 +6,46 @@ A genotyping and analysis toolkit for short(ish) tandem repeats.
|
|
|
6
6
|
|
|
7
7
|
<img src="./docs/images/strkit_logo_small.png" alt="" width="500" height="324" />
|
|
8
8
|
|
|
9
|
-
© David Lougheed and McGill University 2021-2023 (versions up to and including `0.8.0a1`).
|
|
10
9
|
|
|
11
|
-
|
|
12
|
-
|
|
10
|
+
## Copyright and License
|
|
11
|
+
|
|
12
|
+
* 2021-2023: © David Lougheed (DL) and McGill University 2021-2023 (versions up to and including `0.8.0a1`),
|
|
13
|
+
created during graduate research by DL.
|
|
14
|
+
* 2023+: (versions beyond `0.8.0a1`):
|
|
15
|
+
* Portions © DL and McGill University 2021-2023
|
|
16
|
+
* Portions © McGill University 2023
|
|
17
|
+
* Portions © DL 2023
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
### Notice
|
|
21
|
+
|
|
22
|
+
This program is free software: you can redistribute it and/or modify
|
|
23
|
+
it under the terms of the GNU General Public License as published by
|
|
24
|
+
the Free Software Foundation, either version 3 of the License, or
|
|
25
|
+
(at your option) any later version.
|
|
26
|
+
|
|
27
|
+
This program is distributed in the hope that it will be useful,
|
|
28
|
+
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
29
|
+
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
30
|
+
GNU General Public License for more details.
|
|
31
|
+
|
|
32
|
+
You should have received a copy of the GNU General Public License
|
|
33
|
+
along with this program. If not, see <https://www.gnu.org/licenses/>.
|
|
34
|
+
|
|
35
|
+
### Exceptions
|
|
36
|
+
|
|
37
|
+
**Some exclusions to this license apply; specifically portions of
|
|
38
|
+
[`strkit/viz/templates/browser.html`](strkit/viz/templates/browser.html) and
|
|
39
|
+
the STRkit logo files ([./docs/images/strkit_logo_small.png](./docs/images/strkit_logo_small.png)
|
|
40
|
+
and [./strkit/viz/static/logo.png](./strkit/viz/static/logo.png).)**
|
|
13
41
|
|
|
14
42
|
The STRkit logo is © David Lougheed 2022, and was designed by Evelyn Lougheed. It is not licensed
|
|
15
|
-
under the terms of the GPL 3.0;
|
|
43
|
+
under the terms of the GPL 3.0; it is instead licensed under the terms of the
|
|
44
|
+
[CC BY-ND 4.0](https://creativecommons.org/licenses/by-nd/4.0/).
|
|
45
|
+
|
|
46
|
+
Portions of `viz/templates/browser.html` copyright (C) 2021-2022 Observable, Inc.
|
|
47
|
+
Used under the terms of the ISC license.
|
|
48
|
+
|
|
16
49
|
|
|
17
50
|
## Installation
|
|
18
51
|
|
|
@@ -74,7 +107,8 @@ strkit call \
|
|
|
74
107
|
--min-reads 4 \ # Minimum number of supporting reads needed to make a call
|
|
75
108
|
--min-allele-reads 2 \ # Minimum number of supporting reads needed to call a specific allele size
|
|
76
109
|
--flank-size 70 \ # Size of the flanking region to use on either side of a region to properly anchor reads
|
|
77
|
-
--seed 183 # Fixed random number generator seed for replicability
|
|
110
|
+
--seed 183 \ # Fixed random number generator seed for replicability
|
|
111
|
+
--processes 10 # Number of parallel processes to use; DEFAULT: 1
|
|
78
112
|
```
|
|
79
113
|
|
|
80
114
|
Ideally, you should be using a read file aligned with parameters tuned for tandem repeats.
|
|
@@ -190,33 +224,3 @@ by looking at inheritance patterns. This tool provides a few different ways to d
|
|
|
190
224
|
|
|
191
225
|
**For more information on what kind of analyses can be done with this data**, see the
|
|
192
226
|
[Trio analyses with STRkit](./docs/trio_analyses.md) page.
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
## Copyright and License
|
|
197
|
-
|
|
198
|
-
**Some exclusions to this license apply; specifically portions of
|
|
199
|
-
[`strkit/viz/templates/browser.html`](strkit/viz/templates/browser.html)
|
|
200
|
-
and files in the `catalogs` directory.**
|
|
201
|
-
|
|
202
|
-
© David Lougheed 2021-2023 (versions up to and including `0.8.0a1`).
|
|
203
|
-
|
|
204
|
-
© David Lougheed 2021-2023 with portions © McGill University 2023 (versions beyond `0.8.0a1`).
|
|
205
|
-
|
|
206
|
-
Portions of `viz/templates/browser.html` copyright (C) 2021-2022 Observable, Inc.
|
|
207
|
-
Used under the terms of the ISC license.
|
|
208
|
-
|
|
209
|
-
### Notice
|
|
210
|
-
|
|
211
|
-
This program is free software: you can redistribute it and/or modify
|
|
212
|
-
it under the terms of the GNU General Public License as published by
|
|
213
|
-
the Free Software Foundation, either version 3 of the License, or
|
|
214
|
-
(at your option) any later version.
|
|
215
|
-
|
|
216
|
-
This program is distributed in the hope that it will be useful,
|
|
217
|
-
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
218
|
-
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
219
|
-
GNU General Public License for more details.
|
|
220
|
-
|
|
221
|
-
You should have received a copy of the GNU General Public License
|
|
222
|
-
along with this program. If not, see <https://www.gnu.org/licenses/>.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
0.11.1
|
|
@@ -42,19 +42,24 @@ N_GM_INIT = 3
|
|
|
42
42
|
|
|
43
43
|
FLOAT_32_EPSILON = np.finfo(np.float32).eps
|
|
44
44
|
|
|
45
|
+
CI_PERCENTILE_RANGES = {
|
|
46
|
+
"95": (2.5, 97.5),
|
|
47
|
+
"99": (0.5, 99.5),
|
|
48
|
+
}
|
|
49
|
+
|
|
45
50
|
|
|
46
51
|
def _array_as_int(n: Union[NDArray[np.int_], NDArray[np.float_]]) -> NDArray[np.int32]:
|
|
47
52
|
return np.rint(n).astype(np.int32)
|
|
48
53
|
|
|
49
54
|
|
|
50
|
-
def _calculate_cis(
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
55
|
+
def _calculate_cis(
|
|
56
|
+
samples,
|
|
57
|
+
force_int: bool = False,
|
|
58
|
+
ci: str = Literal["95", "99"],
|
|
59
|
+
) -> Union[NDArray[np.int32], NDArray[np.float_]]:
|
|
60
|
+
percentiles = np.percentile(
|
|
61
|
+
samples, CI_PERCENTILE_RANGES[ci], axis=1, method="interpolated_inverted_cdf"
|
|
62
|
+
).transpose()
|
|
58
63
|
return _array_as_int(percentiles) if force_int else percentiles
|
|
59
64
|
|
|
60
65
|
|
|
@@ -139,7 +144,7 @@ def fit_gmm(
|
|
|
139
144
|
|
|
140
145
|
|
|
141
146
|
class CallDict(TypedDict):
|
|
142
|
-
call: Union[NDArray[np.int32], NDArray[np.
|
|
147
|
+
call: Union[NDArray[np.int32], NDArray[np.float_]]
|
|
143
148
|
call_95_cis: Union[NDArray[NDArray[np.int32]], NDArray[NDArray[np.float_]]]
|
|
144
149
|
call_99_cis: Union[NDArray[NDArray[np.int32]], NDArray[NDArray[np.float_]]]
|
|
145
150
|
peaks: NDArray[np.float_]
|
|
@@ -1,10 +1,12 @@
|
|
|
1
1
|
from __future__ import annotations
|
|
2
2
|
|
|
3
|
+
import functools
|
|
3
4
|
import itertools
|
|
4
5
|
import logging
|
|
5
6
|
import multiprocessing as mp
|
|
6
7
|
import numpy as np
|
|
7
8
|
import pysam
|
|
9
|
+
import operator
|
|
8
10
|
import queue
|
|
9
11
|
|
|
10
12
|
from collections import Counter
|
|
@@ -55,10 +57,21 @@ significant_clip_threshold = 100
|
|
|
55
57
|
significant_clip_snv_take_in = 250
|
|
56
58
|
|
|
57
59
|
|
|
60
|
+
# property getters & other partials
|
|
61
|
+
cn_getter = operator.itemgetter("cn")
|
|
62
|
+
weight_getter = operator.itemgetter("w")
|
|
63
|
+
cat_strs = "".join
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
@functools.cache
|
|
67
|
+
def _mask_low_q_base(base_and_qual: tuple[str, int]) -> str:
|
|
68
|
+
return base_and_qual[0] if base_and_qual[1] > base_wildcard_threshold else "X"
|
|
69
|
+
|
|
70
|
+
|
|
58
71
|
def calculate_seq_with_wildcards(qs: str, quals: Optional[list[int]]) -> str:
|
|
59
72
|
if quals is None:
|
|
60
73
|
return qs # No quality information, so don't do anything
|
|
61
|
-
return
|
|
74
|
+
return cat_strs(map(_mask_low_q_base, zip(qs, quals)))
|
|
62
75
|
|
|
63
76
|
|
|
64
77
|
def get_read_coords_from_matched_pairs(
|
|
@@ -79,7 +92,7 @@ def get_read_coords_from_matched_pairs(
|
|
|
79
92
|
|
|
80
93
|
# Skip gaps on either side to find mapped flank indices
|
|
81
94
|
|
|
82
|
-
# Binary search for left flank start
|
|
95
|
+
# Binary search for left flank start -------------------------------------------------------------------------------
|
|
83
96
|
lhs, found = find_pair_by_ref_pos(matched_pairs, left_flank_coord)
|
|
84
97
|
|
|
85
98
|
# lhs now contains the index for the closest starting coordinate to left_flank_coord
|
|
@@ -94,6 +107,7 @@ def get_read_coords_from_matched_pairs(
|
|
|
94
107
|
lhs -= 1
|
|
95
108
|
|
|
96
109
|
left_flank_start, _ = matched_pairs[lhs]
|
|
110
|
+
# ------------------------------------------------------------------------------------------------------------------
|
|
97
111
|
|
|
98
112
|
for query_coord, ref_coord in matched_pairs[lhs+1:]:
|
|
99
113
|
# Skip gaps on either side to find mapped flank indices
|
|
@@ -216,7 +230,7 @@ def calculate_read_distance(
|
|
|
216
230
|
r1 = read_dict_items[i][1]
|
|
217
231
|
r1_snv_u = r1["snvu"]
|
|
218
232
|
|
|
219
|
-
r1_out_of_range: set[int] =
|
|
233
|
+
r1_out_of_range: set[int] = set(filter(lambda y: r1_snv_u[y] == SNV_OUT_OF_RANGE_CHAR, useful_snvs_range))
|
|
220
234
|
|
|
221
235
|
for j in range(i + 1, n_reads):
|
|
222
236
|
r2 = read_dict_items[j][1]
|
|
@@ -359,11 +373,13 @@ def call_alleles_with_incorporated_snvs(
|
|
|
359
373
|
continue
|
|
360
374
|
|
|
361
375
|
read_useful_snv_bases = tuple(snv_bases[bi] for bi, _pos in useful_snvs)
|
|
362
|
-
|
|
376
|
+
n_non_blank_read_useful_snv_bases = len(
|
|
377
|
+
tuple(filter(lambda bb: bb != SNV_OUT_OF_RANGE_CHAR, read_useful_snv_bases))
|
|
378
|
+
)
|
|
363
379
|
|
|
364
|
-
if
|
|
380
|
+
if n_non_blank_read_useful_snv_bases: # TODO: parametrize
|
|
365
381
|
read_dict_items_with_at_least_one_snv.append(read_item)
|
|
366
|
-
if
|
|
382
|
+
if n_non_blank_read_useful_snv_bases >= 2: # TODO: parametrize
|
|
367
383
|
read_dict_items_with_many_snvs.append(read_item)
|
|
368
384
|
else:
|
|
369
385
|
read_dict_items_with_no_snvs.append(read_item)
|
|
@@ -371,7 +387,7 @@ def call_alleles_with_incorporated_snvs(
|
|
|
371
387
|
read["snvu"] = read_useful_snv_bases # Store read-level 'useful' SNVs
|
|
372
388
|
|
|
373
389
|
if print_snvs:
|
|
374
|
-
print(rn, f"\t{read['cn']:.0f}", "\t",
|
|
390
|
+
print(rn, f"\t{read['cn']:.0f}", "\t", cat_strs(read_useful_snv_bases), n_non_blank_read_useful_snv_bases)
|
|
375
391
|
|
|
376
392
|
n_reads_with_many_snvs: int = len(read_dict_items_with_many_snvs)
|
|
377
393
|
n_reads_with_at_least_one_snv: int = len(read_dict_items_with_at_least_one_snv)
|
|
@@ -424,25 +440,23 @@ def call_alleles_with_incorporated_snvs(
|
|
|
424
440
|
cluster_labels = c.labels_
|
|
425
441
|
cluster_indices = tuple(range(n_alleles))
|
|
426
442
|
|
|
427
|
-
cluster_reads: list[
|
|
443
|
+
cluster_reads: list[tuple[ReadDict, ...]] = []
|
|
428
444
|
cns: Union[list[list[int]], list[list[float]]] = []
|
|
429
445
|
c_ws: list[Union[NDArray[np.int_], NDArray[np.float_]]] = []
|
|
430
446
|
|
|
431
447
|
for ci in cluster_indices:
|
|
432
|
-
crs: list[ReadDict] = []
|
|
433
|
-
|
|
434
448
|
# Find reads for cluster
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
449
|
+
crs: tuple[ReadDict, ...] = tuple(
|
|
450
|
+
r for i, (_, r) in enumerate(read_dict_items_with_at_least_one_snv)
|
|
451
|
+
if cluster_labels[i] == ci
|
|
452
|
+
)
|
|
438
453
|
|
|
439
454
|
# Calculate copy number set
|
|
440
|
-
cns.append(
|
|
455
|
+
cns.append(list(map(cn_getter, crs)))
|
|
441
456
|
|
|
442
457
|
# Calculate weights array
|
|
443
|
-
ws = np.fromiter((
|
|
444
|
-
ws
|
|
445
|
-
c_ws.append(ws)
|
|
458
|
+
ws = np.fromiter(map(weight_getter, crs), dtype=np.float_)
|
|
459
|
+
c_ws.append(ws / np.sum(ws))
|
|
446
460
|
|
|
447
461
|
cluster_reads.append(crs)
|
|
448
462
|
|
|
@@ -501,9 +515,7 @@ def call_alleles_with_incorporated_snvs(
|
|
|
501
515
|
for rd in cluster_reads_ordered[i]:
|
|
502
516
|
rd["p"] = i
|
|
503
517
|
|
|
504
|
-
|
|
505
|
-
peak_weights = np.concatenate(tuple(cc["peak_weights"] for cc in cdd), axis=0)
|
|
506
|
-
peak_weights /= np.sum(peak_weights)
|
|
518
|
+
peak_weights_pre_adj = np.concatenate(tuple(cc["peak_weights"] for cc in cdd), axis=0)
|
|
507
519
|
|
|
508
520
|
# All call_datas are truth-y; all arrays should be ordered by peak_order
|
|
509
521
|
call_data = {
|
|
@@ -513,7 +525,8 @@ def call_alleles_with_incorporated_snvs(
|
|
|
513
525
|
"peaks": np.concatenate(tuple(cc["peaks"] for cc in cdd_ordered), axis=None),
|
|
514
526
|
|
|
515
527
|
# TODO: Readjust peak weights when combining or don't include
|
|
516
|
-
|
|
528
|
+
# Make peak weights sum to 1
|
|
529
|
+
"peak_weights": peak_weights_pre_adj / np.sum(peak_weights_pre_adj),
|
|
517
530
|
|
|
518
531
|
"peak_stdevs": np.concatenate(tuple(cc["peak_stdevs"] for cc in cdd_ordered), axis=0),
|
|
519
532
|
"modal_n_peaks": n_alleles, # n. of alleles = n. of peaks always -- if we phased using SNVs
|
|
@@ -904,7 +917,8 @@ def call_locus(
|
|
|
904
917
|
# reference genome. If we have a rare realignment (e.g., a large expansion), we cannot use SNVs.
|
|
905
918
|
have_rare_realigns: bool = False
|
|
906
919
|
for rn, read in read_dict_items:
|
|
907
|
-
|
|
920
|
+
read_cn = read["cn"]
|
|
921
|
+
n_same_cn_no_realign = sum(1 for _, r2 in read_dict_items if not r2.get("realn") and r2["cn"] == read_cn)
|
|
908
922
|
if read.get("realn") and n_same_cn_no_realign == 0:
|
|
909
923
|
have_rare_realigns = True
|
|
910
924
|
break
|
|
@@ -960,9 +974,9 @@ def call_locus(
|
|
|
960
974
|
if single_or_dist_assign: # Didn't use SNVs, so call the 'old-fashioned' way - using only copy number
|
|
961
975
|
# Dicts are ordered in Python; very nice :)
|
|
962
976
|
rdvs = tuple(read_dict.values())
|
|
963
|
-
rcns = tuple(map(
|
|
977
|
+
rcns = tuple(map(cn_getter, rdvs))
|
|
964
978
|
read_cns = np.fromiter(rcns, dtype=np.float_ if fractional else np.int_)
|
|
965
|
-
read_weights = np.fromiter(map(
|
|
979
|
+
read_weights = np.fromiter(map(weight_getter, rdvs), dtype=np.float_)
|
|
966
980
|
read_weights = read_weights / np.sum(read_weights) # Normalize to probabilities
|
|
967
981
|
|
|
968
982
|
call_data = call_alleles(
|
|
@@ -1005,9 +1019,7 @@ def call_locus(
|
|
|
1005
1019
|
stdevs: NDArray[np.float_] = call_stdevs[:call_modal_n]
|
|
1006
1020
|
weights: NDArray[np.float_] = call_weights[:call_modal_n]
|
|
1007
1021
|
|
|
1008
|
-
allele_reads: list[list[str]] = []
|
|
1009
|
-
for _ in range(call_modal_n):
|
|
1010
|
-
allele_reads.append([])
|
|
1022
|
+
allele_reads: list[list[str]] = [list() for _ in range(call_modal_n)]
|
|
1011
1023
|
|
|
1012
1024
|
for r, rd in read_dict_items:
|
|
1013
1025
|
# Need latter term for peaks that we overwrite if we revert to "dist" assignment:
|
|
@@ -1053,7 +1065,7 @@ def call_locus(
|
|
|
1053
1065
|
"stdevs": call_stdevs.tolist(), # from np.ndarray
|
|
1054
1066
|
"modal_n": call_modal_n,
|
|
1055
1067
|
"n_reads": call_peak_n_reads,
|
|
1056
|
-
**({"kmers":
|
|
1068
|
+
**({"kmers": list(map(dict, peak_kmers))} if count_kmers in ("peak", "both") else {}),
|
|
1057
1069
|
} if call_data else None
|
|
1058
1070
|
|
|
1059
1071
|
# Calculate call time ----------------------------------------------------------------------------------------------
|
|
@@ -1066,11 +1078,15 @@ def call_locus(
|
|
|
1066
1078
|
|
|
1067
1079
|
# Finally, compile the call into a dictionary with all information to return ---------------------------------------
|
|
1068
1080
|
|
|
1069
|
-
|
|
1070
|
-
|
|
1081
|
+
if fractional:
|
|
1082
|
+
def _ndarray_serialize(x: Iterable) -> list[Union[float, np.float_]]:
|
|
1083
|
+
return [round_to_base_pos(y, motif_size) for y in x]
|
|
1084
|
+
else:
|
|
1085
|
+
def _ndarray_serialize(x: Iterable) -> list[Union[int, float, np.int_, np.float_]]:
|
|
1086
|
+
return list(map(round, x))
|
|
1071
1087
|
|
|
1072
1088
|
def _nested_ndarray_serialize(x: Iterable) -> list[list[Union[int, float, np.int_, np.float_]]]:
|
|
1073
|
-
return
|
|
1089
|
+
return list(map(_ndarray_serialize, x))
|
|
1074
1090
|
|
|
1075
1091
|
call_val = apply_or_none(_ndarray_serialize, call)
|
|
1076
1092
|
call_95_cis_val = apply_or_none(_nested_ndarray_serialize, call_95_cis)
|
|
@@ -13,6 +13,15 @@ __all__ = [
|
|
|
13
13
|
]
|
|
14
14
|
|
|
15
15
|
|
|
16
|
+
def _line_filter_fn(s: str) -> bool:
|
|
17
|
+
"""
|
|
18
|
+
Filter function to skip blank lines and comments
|
|
19
|
+
:param s: line of a file
|
|
20
|
+
:return: whether the line is not blank and is not a comment
|
|
21
|
+
"""
|
|
22
|
+
return s and not s.startswith("#")
|
|
23
|
+
|
|
24
|
+
|
|
16
25
|
# noinspection PyUnusedLocal
|
|
17
26
|
class BaseCalculator(ABC):
|
|
18
27
|
def __init__(
|
|
@@ -73,33 +82,22 @@ class BaseCalculator(ABC):
|
|
|
73
82
|
def mt_corr(self) -> str:
|
|
74
83
|
return self._mt_corr
|
|
75
84
|
|
|
76
|
-
def _make_loci_dict(self) -> dict:
|
|
85
|
+
def _make_loci_dict(self) -> dict[tuple[str, str, str], list[str, ...]]:
|
|
77
86
|
if not self._loci_file:
|
|
78
87
|
return {}
|
|
79
88
|
|
|
80
89
|
with open(self._loci_file, "r") as lf:
|
|
81
90
|
return {
|
|
82
91
|
tuple(d[:3]): d[3:]
|
|
83
|
-
for d in (
|
|
84
|
-
line.split("\t")
|
|
85
|
-
for line in map(lambda x: x.strip(), lf)
|
|
86
|
-
if line and not line.startswith("#")
|
|
87
|
-
)
|
|
92
|
+
for d in map(lambda line: line.split("\t"), filter(_line_filter_fn, map(str.strip, lf)))
|
|
88
93
|
}
|
|
89
94
|
|
|
90
|
-
def _make_exclude_set(self) -> set:
|
|
95
|
+
def _make_exclude_set(self) -> set[tuple[str, str, str]]:
|
|
91
96
|
if not self._exclude_file:
|
|
92
97
|
return set()
|
|
93
98
|
|
|
94
99
|
with open(self._exclude_file, "r") as lf:
|
|
95
|
-
return
|
|
96
|
-
tuple(d[:3])
|
|
97
|
-
for d in (
|
|
98
|
-
line.split("\t")
|
|
99
|
-
for line in map(lambda x: x.strip(), lf)
|
|
100
|
-
if line and not line.startswith("#")
|
|
101
|
-
)
|
|
102
|
-
}
|
|
100
|
+
return set(map(lambda line: tuple(line.split("\t")[:3]), filter(_line_filter_fn, map(str.strip, lf))))
|
|
103
101
|
|
|
104
102
|
def should_exclude_locus(self, locus: tuple[str, str, str]) -> bool:
|
|
105
103
|
return locus in self._exclude_set
|
|
@@ -136,8 +134,8 @@ class BaseCalculator(ABC):
|
|
|
136
134
|
def calculate(self, included_contigs: set) -> Optional[MIResult]:
|
|
137
135
|
res: float = 0
|
|
138
136
|
res_pm1: float = 0
|
|
139
|
-
res_95_ci: float =
|
|
140
|
-
res_99_ci: float =
|
|
137
|
+
res_95_ci: Optional[float] = None
|
|
138
|
+
res_99_ci: Optional[float] = None
|
|
141
139
|
n_total: int = 0
|
|
142
140
|
|
|
143
141
|
contig_results = []
|
|
@@ -486,11 +486,13 @@ class MIResult:
|
|
|
486
486
|
self._logger.warning("Cannot correct for multiple testing when test is not enabled")
|
|
487
487
|
return
|
|
488
488
|
|
|
489
|
-
loci = [locus for cr in self.contig_results for locus in cr]
|
|
490
489
|
p_values = [locus.p_value for cr in self.contig_results for locus in cr]
|
|
491
490
|
not_tested = {i for i, p in enumerate(p_values) if p is None}
|
|
492
491
|
|
|
493
|
-
loci_filt = [
|
|
492
|
+
loci_filt = [
|
|
493
|
+
locus for i, locus in enumerate(locus for cr in self.contig_results for locus in cr)
|
|
494
|
+
if i not in not_tested
|
|
495
|
+
]
|
|
494
496
|
p_values_filt = np.fromiter((p for i, p in enumerate(p_values) if i not in not_tested), dtype=np.float64)
|
|
495
497
|
|
|
496
498
|
if (mtm := self._mt_corr) != "none":
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: strkit
|
|
3
|
-
Version: 0.11.
|
|
3
|
+
Version: 0.11.1
|
|
4
4
|
Summary: A toolkit for analyzing variation in short(ish) tandem repeats.
|
|
5
5
|
Home-page: https://github.com/davidlougheed/strkit
|
|
6
6
|
Author: David Lougheed
|
|
@@ -13,7 +13,7 @@ Requires-Python: ~=3.9
|
|
|
13
13
|
Description-Content-Type: text/markdown
|
|
14
14
|
License-File: LICENSE
|
|
15
15
|
Requires-Dist: Flask<2.3,>=2.2.5
|
|
16
|
-
Requires-Dist: pysam<0.
|
|
16
|
+
Requires-Dist: pysam<0.23,>=0.19
|
|
17
17
|
Requires-Dist: numpy<1.27,>=1.23.4
|
|
18
18
|
Requires-Dist: parasail<1.4,>=1.2.4
|
|
19
19
|
Requires-Dist: scikit-learn<1.4,>=1.2.1
|
|
@@ -31,13 +31,46 @@ A genotyping and analysis toolkit for short(ish) tandem repeats.
|
|
|
31
31
|
|
|
32
32
|
<img src="./docs/images/strkit_logo_small.png" alt="" width="500" height="324" />
|
|
33
33
|
|
|
34
|
-
© David Lougheed and McGill University 2021-2023 (versions up to and including `0.8.0a1`).
|
|
35
34
|
|
|
36
|
-
|
|
37
|
-
|
|
35
|
+
## Copyright and License
|
|
36
|
+
|
|
37
|
+
* 2021-2023: © David Lougheed (DL) and McGill University 2021-2023 (versions up to and including `0.8.0a1`),
|
|
38
|
+
created during graduate research by DL.
|
|
39
|
+
* 2023+: (versions beyond `0.8.0a1`):
|
|
40
|
+
* Portions © DL and McGill University 2021-2023
|
|
41
|
+
* Portions © McGill University 2023
|
|
42
|
+
* Portions © DL 2023
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
### Notice
|
|
46
|
+
|
|
47
|
+
This program is free software: you can redistribute it and/or modify
|
|
48
|
+
it under the terms of the GNU General Public License as published by
|
|
49
|
+
the Free Software Foundation, either version 3 of the License, or
|
|
50
|
+
(at your option) any later version.
|
|
51
|
+
|
|
52
|
+
This program is distributed in the hope that it will be useful,
|
|
53
|
+
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
54
|
+
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
55
|
+
GNU General Public License for more details.
|
|
56
|
+
|
|
57
|
+
You should have received a copy of the GNU General Public License
|
|
58
|
+
along with this program. If not, see <https://www.gnu.org/licenses/>.
|
|
59
|
+
|
|
60
|
+
### Exceptions
|
|
61
|
+
|
|
62
|
+
**Some exclusions to this license apply; specifically portions of
|
|
63
|
+
[`strkit/viz/templates/browser.html`](strkit/viz/templates/browser.html) and
|
|
64
|
+
the STRkit logo files ([./docs/images/strkit_logo_small.png](./docs/images/strkit_logo_small.png)
|
|
65
|
+
and [./strkit/viz/static/logo.png](./strkit/viz/static/logo.png).)**
|
|
38
66
|
|
|
39
67
|
The STRkit logo is © David Lougheed 2022, and was designed by Evelyn Lougheed. It is not licensed
|
|
40
|
-
under the terms of the GPL 3.0;
|
|
68
|
+
under the terms of the GPL 3.0; it is instead licensed under the terms of the
|
|
69
|
+
[CC BY-ND 4.0](https://creativecommons.org/licenses/by-nd/4.0/).
|
|
70
|
+
|
|
71
|
+
Portions of `viz/templates/browser.html` copyright (C) 2021-2022 Observable, Inc.
|
|
72
|
+
Used under the terms of the ISC license.
|
|
73
|
+
|
|
41
74
|
|
|
42
75
|
## Installation
|
|
43
76
|
|
|
@@ -99,7 +132,8 @@ strkit call \
|
|
|
99
132
|
--min-reads 4 \ # Minimum number of supporting reads needed to make a call
|
|
100
133
|
--min-allele-reads 2 \ # Minimum number of supporting reads needed to call a specific allele size
|
|
101
134
|
--flank-size 70 \ # Size of the flanking region to use on either side of a region to properly anchor reads
|
|
102
|
-
--seed 183 # Fixed random number generator seed for replicability
|
|
135
|
+
--seed 183 \ # Fixed random number generator seed for replicability
|
|
136
|
+
--processes 10 # Number of parallel processes to use; DEFAULT: 1
|
|
103
137
|
```
|
|
104
138
|
|
|
105
139
|
Ideally, you should be using a read file aligned with parameters tuned for tandem repeats.
|
|
@@ -215,33 +249,3 @@ by looking at inheritance patterns. This tool provides a few different ways to d
|
|
|
215
249
|
|
|
216
250
|
**For more information on what kind of analyses can be done with this data**, see the
|
|
217
251
|
[Trio analyses with STRkit](./docs/trio_analyses.md) page.
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
## Copyright and License
|
|
222
|
-
|
|
223
|
-
**Some exclusions to this license apply; specifically portions of
|
|
224
|
-
[`strkit/viz/templates/browser.html`](strkit/viz/templates/browser.html)
|
|
225
|
-
and files in the `catalogs` directory.**
|
|
226
|
-
|
|
227
|
-
© David Lougheed 2021-2023 (versions up to and including `0.8.0a1`).
|
|
228
|
-
|
|
229
|
-
© David Lougheed 2021-2023 with portions © McGill University 2023 (versions beyond `0.8.0a1`).
|
|
230
|
-
|
|
231
|
-
Portions of `viz/templates/browser.html` copyright (C) 2021-2022 Observable, Inc.
|
|
232
|
-
Used under the terms of the ISC license.
|
|
233
|
-
|
|
234
|
-
### Notice
|
|
235
|
-
|
|
236
|
-
This program is free software: you can redistribute it and/or modify
|
|
237
|
-
it under the terms of the GNU General Public License as published by
|
|
238
|
-
the Free Software Foundation, either version 3 of the License, or
|
|
239
|
-
(at your option) any later version.
|
|
240
|
-
|
|
241
|
-
This program is distributed in the hope that it will be useful,
|
|
242
|
-
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
243
|
-
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
244
|
-
GNU General Public License for more details.
|
|
245
|
-
|
|
246
|
-
You should have received a copy of the GNU General Public License
|
|
247
|
-
along with this program. If not, see <https://www.gnu.org/licenses/>.
|
strkit-0.11.0/strkit/VERSION
DELETED
|
@@ -1 +0,0 @@
|
|
|
1
|
-
0.11.0
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|