primer-finder 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- primer_finder-1.0.0/LICENSE +21 -0
- primer_finder-1.0.0/PKG-INFO +115 -0
- primer_finder-1.0.0/README.md +92 -0
- primer_finder-1.0.0/primer_finder/__init__.py +8 -0
- primer_finder-1.0.0/primer_finder/__main__.py +5 -0
- primer_finder-1.0.0/primer_finder/assemble.py +58 -0
- primer_finder-1.0.0/primer_finder/blast.py +238 -0
- primer_finder-1.0.0/primer_finder/cli.py +135 -0
- primer_finder-1.0.0/primer_finder/idt.py +256 -0
- primer_finder-1.0.0/primer_finder/kmers.py +105 -0
- primer_finder-1.0.0/primer_finder/mapping.py +185 -0
- primer_finder-1.0.0/primer_finder/pipeline.py +324 -0
- primer_finder-1.0.0/primer_finder/seqio.py +148 -0
- primer_finder-1.0.0/primer_finder/system.py +84 -0
- primer_finder-1.0.0/primer_finder/tools.py +109 -0
- primer_finder-1.0.0/primer_finder.egg-info/PKG-INFO +115 -0
- primer_finder-1.0.0/primer_finder.egg-info/SOURCES.txt +31 -0
- primer_finder-1.0.0/primer_finder.egg-info/dependency_links.txt +1 -0
- primer_finder-1.0.0/primer_finder.egg-info/entry_points.txt +2 -0
- primer_finder-1.0.0/primer_finder.egg-info/requires.txt +6 -0
- primer_finder-1.0.0/primer_finder.egg-info/top_level.txt +1 -0
- primer_finder-1.0.0/pyproject.toml +53 -0
- primer_finder-1.0.0/setup.cfg +4 -0
- primer_finder-1.0.0/tests/test_blast.py +230 -0
- primer_finder-1.0.0/tests/test_cli.py +147 -0
- primer_finder-1.0.0/tests/test_example.py +139 -0
- primer_finder-1.0.0/tests/test_idt.py +312 -0
- primer_finder-1.0.0/tests/test_kmers.py +77 -0
- primer_finder-1.0.0/tests/test_mapping.py +134 -0
- primer_finder-1.0.0/tests/test_pipeline.py +300 -0
- primer_finder-1.0.0/tests/test_seqio.py +120 -0
- primer_finder-1.0.0/tests/test_system.py +76 -0
- primer_finder-1.0.0/tests/test_tools.py +75 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2019 Marc-Olivier Duceppe
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: primer-finder
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Find group-specific kmers to design selective qPCR assays.
|
|
5
|
+
Author: Marc-Olivier Duceppe
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/duceppemo/primer-finder
|
|
8
|
+
Project-URL: Documentation, https://github.com/duceppemo/primer-finder/wiki
|
|
9
|
+
Project-URL: Changelog, https://github.com/duceppemo/primer-finder/blob/master/CHANGELOG.md
|
|
10
|
+
Keywords: bioinformatics,qPCR,primer design,kmer,diagnostics
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Operating System :: POSIX
|
|
13
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
14
|
+
Requires-Python: >=3.10
|
|
15
|
+
Description-Content-Type: text/markdown
|
|
16
|
+
License-File: LICENSE
|
|
17
|
+
Provides-Extra: test
|
|
18
|
+
Requires-Dist: pytest>=7; extra == "test"
|
|
19
|
+
Requires-Dist: pytest-cov>=5; extra == "test"
|
|
20
|
+
Requires-Dist: ruff; extra == "test"
|
|
21
|
+
Requires-Dist: pre-commit; extra == "test"
|
|
22
|
+
Dynamic: license-file
|
|
23
|
+
|
|
24
|
+
# primer-finder
|
|
25
|
+
|
|
26
|
+
<p align="center">
|
|
27
|
+
<a href="https://github.com/duceppemo/primer-finder/actions/workflows/ci.yml"><img src="https://github.com/duceppemo/primer-finder/actions/workflows/ci.yml/badge.svg" alt="CI"></a>
|
|
28
|
+
<a href="https://codecov.io/gh/duceppemo/primer-finder"><img src="https://codecov.io/gh/duceppemo/primer-finder/graph/badge.svg" alt="Coverage"></a>
|
|
29
|
+
<a href="https://github.com/duceppemo/primer-finder/releases/latest"><img src="https://img.shields.io/github/v/release/duceppemo/primer-finder?label=release&cacheSeconds=3600" alt="Latest release"></a>
|
|
30
|
+
<a href="https://anaconda.org/bioconda/primer-finder"><img src="https://img.shields.io/conda/vn/bioconda/primer-finder?label=bioconda" alt="Bioconda"></a>
|
|
31
|
+
<img src="https://img.shields.io/badge/python-3.10%2B-blue" alt="Python 3.10+">
|
|
32
|
+
<a href="LICENSE"><img src="https://img.shields.io/github/license/duceppemo/primer-finder" alt="License: MIT"></a>
|
|
33
|
+
<a href="https://github.com/duceppemo/primer-finder/wiki"><img src="https://img.shields.io/badge/docs-wiki-informational" alt="Documentation"></a>
|
|
34
|
+
<a href="https://zenodo.org/badge/latestdoi/222737677"><img src="https://zenodo.org/badge/222737677.svg" alt="DOI"></a>
|
|
35
|
+
</p>
|
|
36
|
+
|
|
37
|
+
primer-finder finds the sequences that tell one group of genomes from another, so that a selective (q)PCR
|
|
38
|
+
assay can be designed on them. Give it two folders of assembled genomes — the ones the assay should amplify
|
|
39
|
+
(inclusion) and the ones it must not (exclusion) — and it reports the regions that every inclusion genome
|
|
40
|
+
carries, that no exclusion genome carries, and whose differences are close enough together to sit in one
|
|
41
|
+
primer or probe.
|
|
42
|
+
|
|
43
|
+
```
|
|
44
|
+
inclusion/ ──┐ ┌──► 1. kmers shared by all inclusion genomes (KMC)
|
|
45
|
+
├──► kmers ──► subtract ──► assemble ──► map ──► blast
|
|
46
|
+
exclusion/ ──┘ (KMC) (SKESA or (minimap2) (every genome)
|
|
47
|
+
SPAdes)
|
|
48
|
+
└──► final_kmers.fasta: the candidate regions,
|
|
49
|
+
with the specific bases in lower case
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
## Quick start
|
|
53
|
+
|
|
54
|
+
```bash
|
|
55
|
+
conda create -n primer-finder -c conda-forge -c bioconda primer-finder
|
|
56
|
+
conda activate primer-finder
|
|
57
|
+
|
|
58
|
+
primer-finder -i inclusion/ -e exclusion/ -o results/
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
`inclusion/` and `exclusion/` hold one assembled genome per file (`.fasta`, `.fna`, `.fa`, gzipped or not;
|
|
62
|
+
subfolders and symbolic links are followed). The answer is `results/final_kmers.fasta`: one record per
|
|
63
|
+
candidate region, the specific bases in lower case and their positions in the header, the most promising
|
|
64
|
+
first. `results/run_info.json` records the parameters, the genomes and the version of every program used.
|
|
65
|
+
|
|
66
|
+
To install from the source code instead, see
|
|
67
|
+
[Installation](https://github.com/duceppemo/primer-finder/wiki/Installation).
|
|
68
|
+
|
|
69
|
+
primer-finder only keeps perfect matches: a kmer must be in **all** the inclusion genomes with no mismatch,
|
|
70
|
+
and in **none** of the exclusion genomes. It is therefore very sensitive to the quality of the assemblies and
|
|
71
|
+
to how the genomes were assigned to the two groups. Curate the input genomes;
|
|
72
|
+
[genome_comparator](https://github.com/duceppemo/genome_comparator) helps with that.
|
|
73
|
+
|
|
74
|
+
To check an installation, run the bundled example (simulated genomes with a known answer, a few seconds).
|
|
75
|
+
It is in the repository, not in the conda package:
|
|
76
|
+
|
|
77
|
+
```bash
|
|
78
|
+
curl -sL https://github.com/duceppemo/primer-finder/archive/refs/tags/v1.0.0.tar.gz | tar -xz --strip-components=1 primer-finder-1.0.0/example
|
|
79
|
+
bash example/run_example.sh
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
## Ordering the assays
|
|
83
|
+
|
|
84
|
+
Once an assay has been designed from a candidate region and ordered, `primer-finder idt` turns the IDT order
|
|
85
|
+
sheet into a fasta file of oligos, one record per primer and probe:
|
|
86
|
+
|
|
87
|
+
```bash
|
|
88
|
+
primer-finder idt order.xlsx assays.fasta my_target
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
## Documentation
|
|
92
|
+
|
|
93
|
+
Everything else is in the [wiki](https://github.com/duceppemo/primer-finder/wiki), whose sources are
|
|
94
|
+
maintained in [`docs/wiki`](docs/wiki):
|
|
95
|
+
|
|
96
|
+
| Page | Contents |
|
|
97
|
+
|---|---|
|
|
98
|
+
| [Installation](https://github.com/duceppemo/primer-finder/wiki/Installation) | conda, bioconda, from source, checking the installation |
|
|
99
|
+
| [Usage](https://github.com/duceppemo/primer-finder/wiki/Usage) | inputs, every option, choosing the two groups, performance |
|
|
100
|
+
| [Methods](https://github.com/duceppemo/primer-finder/wiki/Methods) | what each step does, the filtering rules, limits |
|
|
101
|
+
| [Outputs](https://github.com/duceppemo/primer-finder/wiki/Outputs) | every file and field |
|
|
102
|
+
| [Example](https://github.com/duceppemo/primer-finder/wiki/Example) | the simulated dataset and its expected result |
|
|
103
|
+
| [FAQ](https://github.com/duceppemo/primer-finder/wiki/FAQ) | troubleshooting, "no contig passed" |
|
|
104
|
+
| [Development](https://github.com/duceppemo/primer-finder/wiki/Development) | tests, continuous integration, releases |
|
|
105
|
+
|
|
106
|
+
## Citation
|
|
107
|
+
|
|
108
|
+
If primer-finder helped your work, please cite it (see [CITATION.cff](CITATION.cff)) together with the
|
|
109
|
+
programs it runs: [KMC](https://github.com/refresh-bio/KMC),
|
|
110
|
+
[SKESA](https://github.com/ncbi/SKESA) or [SPAdes](https://github.com/ablab/spades),
|
|
111
|
+
[minimap2](https://github.com/lh3/minimap2) and [BLAST](https://blast.ncbi.nlm.nih.gov/).
|
|
112
|
+
|
|
113
|
+
## License
|
|
114
|
+
|
|
115
|
+
[MIT](LICENSE)
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
# primer-finder
|
|
2
|
+
|
|
3
|
+
<p align="center">
|
|
4
|
+
<a href="https://github.com/duceppemo/primer-finder/actions/workflows/ci.yml"><img src="https://github.com/duceppemo/primer-finder/actions/workflows/ci.yml/badge.svg" alt="CI"></a>
|
|
5
|
+
<a href="https://codecov.io/gh/duceppemo/primer-finder"><img src="https://codecov.io/gh/duceppemo/primer-finder/graph/badge.svg" alt="Coverage"></a>
|
|
6
|
+
<a href="https://github.com/duceppemo/primer-finder/releases/latest"><img src="https://img.shields.io/github/v/release/duceppemo/primer-finder?label=release&cacheSeconds=3600" alt="Latest release"></a>
|
|
7
|
+
<a href="https://anaconda.org/bioconda/primer-finder"><img src="https://img.shields.io/conda/vn/bioconda/primer-finder?label=bioconda" alt="Bioconda"></a>
|
|
8
|
+
<img src="https://img.shields.io/badge/python-3.10%2B-blue" alt="Python 3.10+">
|
|
9
|
+
<a href="LICENSE"><img src="https://img.shields.io/github/license/duceppemo/primer-finder" alt="License: MIT"></a>
|
|
10
|
+
<a href="https://github.com/duceppemo/primer-finder/wiki"><img src="https://img.shields.io/badge/docs-wiki-informational" alt="Documentation"></a>
|
|
11
|
+
<a href="https://zenodo.org/badge/latestdoi/222737677"><img src="https://zenodo.org/badge/222737677.svg" alt="DOI"></a>
|
|
12
|
+
</p>
|
|
13
|
+
|
|
14
|
+
primer-finder finds the sequences that tell one group of genomes from another, so that a selective (q)PCR
|
|
15
|
+
assay can be designed on them. Give it two folders of assembled genomes — the ones the assay should amplify
|
|
16
|
+
(inclusion) and the ones it must not (exclusion) — and it reports the regions that every inclusion genome
|
|
17
|
+
carries, that no exclusion genome carries, and whose differences are close enough together to sit in one
|
|
18
|
+
primer or probe.
|
|
19
|
+
|
|
20
|
+
```
|
|
21
|
+
inclusion/ ──┐ ┌──► 1. kmers shared by all inclusion genomes (KMC)
|
|
22
|
+
├──► kmers ──► subtract ──► assemble ──► map ──► blast
|
|
23
|
+
exclusion/ ──┘ (KMC) (SKESA or (minimap2) (every genome)
|
|
24
|
+
SPAdes)
|
|
25
|
+
└──► final_kmers.fasta: the candidate regions,
|
|
26
|
+
with the specific bases in lower case
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
## Quick start
|
|
30
|
+
|
|
31
|
+
```bash
|
|
32
|
+
conda create -n primer-finder -c conda-forge -c bioconda primer-finder
|
|
33
|
+
conda activate primer-finder
|
|
34
|
+
|
|
35
|
+
primer-finder -i inclusion/ -e exclusion/ -o results/
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
`inclusion/` and `exclusion/` hold one assembled genome per file (`.fasta`, `.fna`, `.fa`, gzipped or not;
|
|
39
|
+
subfolders and symbolic links are followed). The answer is `results/final_kmers.fasta`: one record per
|
|
40
|
+
candidate region, the specific bases in lower case and their positions in the header, the most promising
|
|
41
|
+
first. `results/run_info.json` records the parameters, the genomes and the version of every program used.
|
|
42
|
+
|
|
43
|
+
To install from the source code instead, see
|
|
44
|
+
[Installation](https://github.com/duceppemo/primer-finder/wiki/Installation).
|
|
45
|
+
|
|
46
|
+
primer-finder only keeps perfect matches: a kmer must be in **all** the inclusion genomes with no mismatch,
|
|
47
|
+
and in **none** of the exclusion genomes. It is therefore very sensitive to the quality of the assemblies and
|
|
48
|
+
to how the genomes were assigned to the two groups. Curate the input genomes;
|
|
49
|
+
[genome_comparator](https://github.com/duceppemo/genome_comparator) helps with that.
|
|
50
|
+
|
|
51
|
+
To check an installation, run the bundled example (simulated genomes with a known answer, a few seconds).
|
|
52
|
+
It is in the repository, not in the conda package:
|
|
53
|
+
|
|
54
|
+
```bash
|
|
55
|
+
curl -sL https://github.com/duceppemo/primer-finder/archive/refs/tags/v1.0.0.tar.gz | tar -xz --strip-components=1 primer-finder-1.0.0/example
|
|
56
|
+
bash example/run_example.sh
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
## Ordering the assays
|
|
60
|
+
|
|
61
|
+
Once an assay has been designed from a candidate region and ordered, `primer-finder idt` turns the IDT order
|
|
62
|
+
sheet into a fasta file of oligos, one record per primer and probe:
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
primer-finder idt order.xlsx assays.fasta my_target
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
## Documentation
|
|
69
|
+
|
|
70
|
+
Everything else is in the [wiki](https://github.com/duceppemo/primer-finder/wiki), whose sources are
|
|
71
|
+
maintained in [`docs/wiki`](docs/wiki):
|
|
72
|
+
|
|
73
|
+
| Page | Contents |
|
|
74
|
+
|---|---|
|
|
75
|
+
| [Installation](https://github.com/duceppemo/primer-finder/wiki/Installation) | conda, bioconda, from source, checking the installation |
|
|
76
|
+
| [Usage](https://github.com/duceppemo/primer-finder/wiki/Usage) | inputs, every option, choosing the two groups, performance |
|
|
77
|
+
| [Methods](https://github.com/duceppemo/primer-finder/wiki/Methods) | what each step does, the filtering rules, limits |
|
|
78
|
+
| [Outputs](https://github.com/duceppemo/primer-finder/wiki/Outputs) | every file and field |
|
|
79
|
+
| [Example](https://github.com/duceppemo/primer-finder/wiki/Example) | the simulated dataset and its expected result |
|
|
80
|
+
| [FAQ](https://github.com/duceppemo/primer-finder/wiki/FAQ) | troubleshooting, "no contig passed" |
|
|
81
|
+
| [Development](https://github.com/duceppemo/primer-finder/wiki/Development) | tests, continuous integration, releases |
|
|
82
|
+
|
|
83
|
+
## Citation
|
|
84
|
+
|
|
85
|
+
If primer-finder helped your work, please cite it (see [CITATION.cff](CITATION.cff)) together with the
|
|
86
|
+
programs it runs: [KMC](https://github.com/refresh-bio/KMC),
|
|
87
|
+
[SKESA](https://github.com/ncbi/SKESA) or [SPAdes](https://github.com/ablab/spades),
|
|
88
|
+
[minimap2](https://github.com/lh3/minimap2) and [BLAST](https://blast.ncbi.nlm.nih.gov/).
|
|
89
|
+
|
|
90
|
+
## License
|
|
91
|
+
|
|
92
|
+
[MIT](LICENSE)
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
"""primer-finder: find group-specific kmers to design selective qPCR assays."""
|
|
2
|
+
|
|
3
|
+
__version__ = "1.0.0"
|
|
4
|
+
__author__ = "Marc-Olivier Duceppe"
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class PrimerFinderError(Exception):
|
|
8
|
+
"""A user-facing error: bad input, a missing program, or a failed external command."""
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""Assembling the inclusion-specific kmers into contigs, with SKESA or SPAdes."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
import shutil
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
from primer_finder import PrimerFinderError, tools
|
|
10
|
+
|
|
11
|
+
log = logging.getLogger(__name__)
|
|
12
|
+
|
|
13
|
+
ASSEMBLERS = ("skesa", "spades")
|
|
14
|
+
PROGRAMS = {"skesa": "skesa", "spades": "spades.py"}
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def assemble(kmer_fasta: Path, output: Path, assembler: str, threads: int, memory_gb: int) -> Path:
|
|
18
|
+
"""Assemble a fasta file of kmers into `output`. Returns `output`."""
|
|
19
|
+
if assembler == "skesa":
|
|
20
|
+
_skesa(kmer_fasta, output, threads, memory_gb)
|
|
21
|
+
elif assembler == "spades":
|
|
22
|
+
_spades(kmer_fasta, output, threads, memory_gb)
|
|
23
|
+
else: # pragma: no cover - the command line only accepts the two
|
|
24
|
+
raise PrimerFinderError(f'Unknown assembler "{assembler}". Choose one of: {", ".join(ASSEMBLERS)}')
|
|
25
|
+
if not output.exists() or output.stat().st_size == 0:
|
|
26
|
+
raise PrimerFinderError(
|
|
27
|
+
f"{assembler} could not assemble the inclusion-specific kmers into contigs. "
|
|
28
|
+
"There may be too few of them; a smaller kmer size (-k) may help."
|
|
29
|
+
)
|
|
30
|
+
return output
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _skesa(kmer_fasta: Path, output: Path, threads: int, memory_gb: int) -> None:
|
|
34
|
+
tools.run([
|
|
35
|
+
"skesa",
|
|
36
|
+
"--cores", str(threads),
|
|
37
|
+
"--mem", str(memory_gb),
|
|
38
|
+
"--fasta", kmer_fasta,
|
|
39
|
+
"--contigs_out", output,
|
|
40
|
+
])
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _spades(kmer_fasta: Path, output: Path, threads: int, memory_gb: int) -> None:
|
|
44
|
+
work_dir = output.parent / "spades"
|
|
45
|
+
tools.run([
|
|
46
|
+
"spades.py",
|
|
47
|
+
"--s", "1", kmer_fasta,
|
|
48
|
+
"--isolate",
|
|
49
|
+
"--only-assembler", # The kmers carry no quality values: there is nothing to correct
|
|
50
|
+
"--threads", str(threads),
|
|
51
|
+
"--memory", str(memory_gb),
|
|
52
|
+
"-o", work_dir,
|
|
53
|
+
])
|
|
54
|
+
contigs = work_dir / "contigs.fasta"
|
|
55
|
+
if not contigs.exists():
|
|
56
|
+
raise PrimerFinderError(f"SPAdes wrote no contigs ({contigs} missing)")
|
|
57
|
+
shutil.move(str(contigs), output)
|
|
58
|
+
shutil.rmtree(work_dir, ignore_errors=True)
|
|
@@ -0,0 +1,238 @@
|
|
|
1
|
+
"""Checking the candidate contigs against every inclusion and exclusion genome with blast."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import gzip
|
|
6
|
+
import logging
|
|
7
|
+
import os
|
|
8
|
+
import shutil
|
|
9
|
+
from collections.abc import Iterable, Sequence
|
|
10
|
+
from concurrent import futures
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
from primer_finder import tools
|
|
15
|
+
from primer_finder.mapping import PRIMER_LENGTH
|
|
16
|
+
from primer_finder.seqio import base_name, iter_records
|
|
17
|
+
|
|
18
|
+
log = logging.getLogger(__name__)
|
|
19
|
+
|
|
20
|
+
# A hit has to be at least this good to count as "the contig is there".
|
|
21
|
+
MAX_EVALUE = 1e-10
|
|
22
|
+
# A variant position is kept when it differs in at least this fraction of the exclusion genomes that hold
|
|
23
|
+
# the region: a few exclusion genomes may carry the inclusion allele without making the assay useless.
|
|
24
|
+
MIN_EXCLUSION_FRACTION = 0.90
|
|
25
|
+
|
|
26
|
+
PRESENCE_FIELDS = ("qseqid", "evalue")
|
|
27
|
+
ALIGNMENT_FIELDS = ("qseqid", "qstart", "qend", "evalue", "qseq", "sseq")
|
|
28
|
+
|
|
29
|
+
# BLAST splits its path arguments on whitespace, so it is never given a path that could hold a space: each
|
|
30
|
+
# genome is linked into its own folder under these fixed names, and blast runs in that folder.
|
|
31
|
+
LOCAL_FASTA = "genome.fasta"
|
|
32
|
+
DB_NAME = "db"
|
|
33
|
+
HITS_NAME = "hits.tsv"
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass
|
|
37
|
+
class Hit:
|
|
38
|
+
"""One blast high-scoring pair, with the aligned query and subject sequences."""
|
|
39
|
+
|
|
40
|
+
query: str
|
|
41
|
+
qstart: int
|
|
42
|
+
qend: int
|
|
43
|
+
evalue: float
|
|
44
|
+
qseq: str = ""
|
|
45
|
+
sseq: str = ""
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
@dataclass
|
|
49
|
+
class Alignments:
|
|
50
|
+
"""What one exclusion genome holds of one contig: every part of the contig it aligns, and where each
|
|
51
|
+
of those alignments differs from it."""
|
|
52
|
+
|
|
53
|
+
spans: list[tuple[int, int, set[int]]] = field(default_factory=list) # start, end (0-based), variants
|
|
54
|
+
|
|
55
|
+
def add(self, hit: Hit) -> None:
|
|
56
|
+
self.spans.append((hit.qstart - 1, hit.qend, set(variant_positions(hit))))
|
|
57
|
+
|
|
58
|
+
def covers(self, position: int) -> bool:
|
|
59
|
+
return any(start <= position < end for start, end, _ in self.spans)
|
|
60
|
+
|
|
61
|
+
def differs_at(self, position: int) -> bool:
|
|
62
|
+
"""True when every copy of this region in the genome differs from the contig at this position. One
|
|
63
|
+
matching copy is enough for an assay to amplify the genome, so one is enough to say no."""
|
|
64
|
+
covering = [(start, end, variants) for start, end, variants in self.spans if start <= position < end]
|
|
65
|
+
return bool(covering) and all(position in variants for _, _, variants in covering)
|
|
66
|
+
|
|
67
|
+
@property
|
|
68
|
+
def variants(self) -> set[int]:
|
|
69
|
+
return {position for _, _, variants in self.spans for position in variants}
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
@dataclass
|
|
73
|
+
class ExclusionResult:
|
|
74
|
+
"""What the exclusion genomes say about one contig: one entry per genome that holds any part of it."""
|
|
75
|
+
|
|
76
|
+
genomes: list[Alignments] = field(default_factory=list)
|
|
77
|
+
|
|
78
|
+
@property
|
|
79
|
+
def genomes_hit(self) -> int:
|
|
80
|
+
return len(self.genomes)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def make_db(genome: Path, work_dir: Path) -> Path:
|
|
84
|
+
"""Make a blast database for one genome, inside `work_dir` so that the input folders are left alone.
|
|
85
|
+
|
|
86
|
+
The genome is linked into `work_dir` under a fixed name (a gzipped one is decompressed: makeblastdb does
|
|
87
|
+
not read gzip), and makeblastdb runs there with relative paths, so neither a space in the path nor two
|
|
88
|
+
genomes with the same file name can reach BLAST.
|
|
89
|
+
"""
|
|
90
|
+
work_dir.mkdir(parents=True, exist_ok=True)
|
|
91
|
+
local = work_dir / LOCAL_FASTA
|
|
92
|
+
if local.exists() or local.is_symlink():
|
|
93
|
+
local.unlink()
|
|
94
|
+
if genome.name.endswith(".gz"):
|
|
95
|
+
with gzip.open(genome, "rb") as src, local.open("wb") as dst:
|
|
96
|
+
shutil.copyfileobj(src, dst)
|
|
97
|
+
else:
|
|
98
|
+
try:
|
|
99
|
+
local.symlink_to(genome.resolve())
|
|
100
|
+
except OSError: # pragma: no cover - a filesystem without symbolic links
|
|
101
|
+
shutil.copyfile(genome, local)
|
|
102
|
+
tools.run(["makeblastdb", "-in", LOCAL_FASTA, "-dbtype", "nucl", "-out", DB_NAME], cwd=work_dir)
|
|
103
|
+
return work_dir / DB_NAME
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def blastn(db: Path, query: Path, out_file: Path, fields: Sequence[str], max_targets: int = 10) -> Path:
|
|
107
|
+
"""Run blastn in the database's folder, writing the chosen tabular fields to `out_file`."""
|
|
108
|
+
work_dir = db.parent
|
|
109
|
+
tools.run([
|
|
110
|
+
"blastn",
|
|
111
|
+
"-db", db.name,
|
|
112
|
+
"-query", os.path.relpath(query, work_dir),
|
|
113
|
+
"-out", os.path.relpath(out_file, work_dir),
|
|
114
|
+
"-evalue", str(MAX_EVALUE),
|
|
115
|
+
"-max_target_seqs", str(max_targets),
|
|
116
|
+
"-num_threads", "1",
|
|
117
|
+
"-outfmt", "6 " + " ".join(fields),
|
|
118
|
+
], cwd=work_dir)
|
|
119
|
+
return out_file
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def parse_hits(out_file: Path, fields: Sequence[str]) -> list[Hit]:
|
|
123
|
+
"""Read a tabular blast output written with `fields`."""
|
|
124
|
+
hits: list[Hit] = []
|
|
125
|
+
index = {name: position for position, name in enumerate(fields)}
|
|
126
|
+
with out_file.open() as fh:
|
|
127
|
+
for line in fh:
|
|
128
|
+
values = line.rstrip("\n").split("\t")
|
|
129
|
+
if len(values) != len(fields):
|
|
130
|
+
continue
|
|
131
|
+
hits.append(Hit(
|
|
132
|
+
query=values[index["qseqid"]],
|
|
133
|
+
qstart=int(values[index["qstart"]]) if "qstart" in index else 0,
|
|
134
|
+
qend=int(values[index["qend"]]) if "qend" in index else 0,
|
|
135
|
+
evalue=float(values[index["evalue"]]),
|
|
136
|
+
qseq=values[index["qseq"]] if "qseq" in index else "",
|
|
137
|
+
sseq=values[index["sseq"]] if "sseq" in index else "",
|
|
138
|
+
))
|
|
139
|
+
return hits
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def variant_positions(hit: Hit) -> list[int]:
|
|
143
|
+
"""The positions of the contig (0-based) that differ from this exclusion sequence: mismatches, inserted
|
|
144
|
+
bases, and the base next to a deletion."""
|
|
145
|
+
positions: list[int] = []
|
|
146
|
+
position = hit.qstart - 1
|
|
147
|
+
for query_base, subject_base in zip(hit.qseq, hit.sseq, strict=False):
|
|
148
|
+
if query_base == "-": # The exclusion genome has bases the contig does not: mark where they fit in
|
|
149
|
+
positions.append(position)
|
|
150
|
+
continue
|
|
151
|
+
if subject_base == "-" or query_base.upper() != subject_base.upper():
|
|
152
|
+
positions.append(position)
|
|
153
|
+
position += 1
|
|
154
|
+
return sorted(set(positions))
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def has_close_variants(positions: Sequence[int], window: int = PRIMER_LENGTH) -> bool:
|
|
158
|
+
"""True if at least two of these positions are less than `window` bases apart."""
|
|
159
|
+
return any(second - first < window for first, second in zip(positions, positions[1:], strict=False))
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def _parallel(work: Iterable[tuple], function, threads: int) -> list:
|
|
163
|
+
"""Run `function(*arguments)` for every item, at most `threads` at a time, keeping the input order."""
|
|
164
|
+
work = list(work)
|
|
165
|
+
if not work:
|
|
166
|
+
return []
|
|
167
|
+
with futures.ThreadPoolExecutor(max_workers=max(1, min(threads, len(work)))) as executor:
|
|
168
|
+
return list(executor.map(lambda arguments: function(*arguments), work))
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def genome_folder(work_dir: Path, index: int, genome: Path) -> Path:
|
|
172
|
+
"""A folder of its own for each genome: two genomes in different subfolders can have the same file
|
|
173
|
+
name, so the position in the list is what makes it unique."""
|
|
174
|
+
return work_dir / f"{index:04d}_{base_name(genome)}"
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def presence_in_genomes(
|
|
178
|
+
query: Path, genomes: Sequence[Path], work_dir: Path, threads: int
|
|
179
|
+
) -> dict[str, dict[str, bool]]:
|
|
180
|
+
"""For every contig of `query`, whether it is present in each genome. Keyed by contig, then by the
|
|
181
|
+
genome's path, which is what tells two genomes with the same file name apart."""
|
|
182
|
+
def one(index: int, genome: Path) -> set[str]:
|
|
183
|
+
folder = genome_folder(work_dir, index, genome)
|
|
184
|
+
db = make_db(genome, folder)
|
|
185
|
+
hits = parse_hits(blastn(db, query, folder / HITS_NAME, PRESENCE_FIELDS, max_targets=1),
|
|
186
|
+
PRESENCE_FIELDS)
|
|
187
|
+
return {hit.query for hit in hits if hit.evalue <= MAX_EVALUE}
|
|
188
|
+
|
|
189
|
+
found = _parallel(enumerate(genomes), one, threads)
|
|
190
|
+
contigs = _query_names(query)
|
|
191
|
+
presence: dict[str, dict[str, bool]] = {contig: {} for contig in contigs}
|
|
192
|
+
for genome, hits in zip(genomes, found, strict=True):
|
|
193
|
+
for contig in contigs:
|
|
194
|
+
presence[contig][str(genome)] = contig in hits
|
|
195
|
+
return presence
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def _query_names(query: Path) -> list[str]:
|
|
199
|
+
"""The names of the contigs in a fasta file, in file order."""
|
|
200
|
+
return [record.name for record in iter_records(query)]
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def exclusion_variants(
|
|
204
|
+
query: Path, genomes: Sequence[Path], work_dir: Path, threads: int
|
|
205
|
+
) -> dict[str, ExclusionResult]:
|
|
206
|
+
"""For every contig of `query`, what each exclusion genome that holds part of it looks like there."""
|
|
207
|
+
def one(index: int, genome: Path) -> list[Hit]:
|
|
208
|
+
folder = genome_folder(work_dir, index, genome)
|
|
209
|
+
db = make_db(genome, folder)
|
|
210
|
+
return parse_hits(blastn(db, query, folder / HITS_NAME, ALIGNMENT_FIELDS), ALIGNMENT_FIELDS)
|
|
211
|
+
|
|
212
|
+
results: dict[str, ExclusionResult] = {name: ExclusionResult() for name in _query_names(query)}
|
|
213
|
+
for hits in _parallel(enumerate(genomes), one, threads):
|
|
214
|
+
per_contig: dict[str, Alignments] = {}
|
|
215
|
+
for hit in hits:
|
|
216
|
+
if hit.evalue > MAX_EVALUE:
|
|
217
|
+
continue
|
|
218
|
+
per_contig.setdefault(hit.query, Alignments()).add(hit)
|
|
219
|
+
for contig, alignments in per_contig.items():
|
|
220
|
+
results.setdefault(contig, ExclusionResult()).genomes.append(alignments)
|
|
221
|
+
return results
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def shared_variants(result: ExclusionResult, fraction: float = MIN_EXCLUSION_FRACTION) -> list[int]:
|
|
225
|
+
"""The contig positions worth designing a primer on: those that differ from at least `fraction` of the
|
|
226
|
+
exclusion genomes that hold that part of the contig.
|
|
227
|
+
|
|
228
|
+
A genome that does not align a position says nothing about it — it does not hold that region, which only
|
|
229
|
+
makes the assay more selective — so it is left out of the count rather than counted as identical.
|
|
230
|
+
"""
|
|
231
|
+
kept: list[int] = []
|
|
232
|
+
candidates = {position for alignments in result.genomes for position in alignments.variants}
|
|
233
|
+
for position in sorted(candidates):
|
|
234
|
+
covering = [alignments for alignments in result.genomes if alignments.covers(position)]
|
|
235
|
+
differing = [alignments for alignments in covering if alignments.differs_at(position)]
|
|
236
|
+
if covering and len(differing) >= fraction * len(covering):
|
|
237
|
+
kept.append(position)
|
|
238
|
+
return kept
|