primer-finder 1.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. primer_finder-1.0.0/LICENSE +21 -0
  2. primer_finder-1.0.0/PKG-INFO +115 -0
  3. primer_finder-1.0.0/README.md +92 -0
  4. primer_finder-1.0.0/primer_finder/__init__.py +8 -0
  5. primer_finder-1.0.0/primer_finder/__main__.py +5 -0
  6. primer_finder-1.0.0/primer_finder/assemble.py +58 -0
  7. primer_finder-1.0.0/primer_finder/blast.py +238 -0
  8. primer_finder-1.0.0/primer_finder/cli.py +135 -0
  9. primer_finder-1.0.0/primer_finder/idt.py +256 -0
  10. primer_finder-1.0.0/primer_finder/kmers.py +105 -0
  11. primer_finder-1.0.0/primer_finder/mapping.py +185 -0
  12. primer_finder-1.0.0/primer_finder/pipeline.py +324 -0
  13. primer_finder-1.0.0/primer_finder/seqio.py +148 -0
  14. primer_finder-1.0.0/primer_finder/system.py +84 -0
  15. primer_finder-1.0.0/primer_finder/tools.py +109 -0
  16. primer_finder-1.0.0/primer_finder.egg-info/PKG-INFO +115 -0
  17. primer_finder-1.0.0/primer_finder.egg-info/SOURCES.txt +31 -0
  18. primer_finder-1.0.0/primer_finder.egg-info/dependency_links.txt +1 -0
  19. primer_finder-1.0.0/primer_finder.egg-info/entry_points.txt +2 -0
  20. primer_finder-1.0.0/primer_finder.egg-info/requires.txt +6 -0
  21. primer_finder-1.0.0/primer_finder.egg-info/top_level.txt +1 -0
  22. primer_finder-1.0.0/pyproject.toml +53 -0
  23. primer_finder-1.0.0/setup.cfg +4 -0
  24. primer_finder-1.0.0/tests/test_blast.py +230 -0
  25. primer_finder-1.0.0/tests/test_cli.py +147 -0
  26. primer_finder-1.0.0/tests/test_example.py +139 -0
  27. primer_finder-1.0.0/tests/test_idt.py +312 -0
  28. primer_finder-1.0.0/tests/test_kmers.py +77 -0
  29. primer_finder-1.0.0/tests/test_mapping.py +134 -0
  30. primer_finder-1.0.0/tests/test_pipeline.py +300 -0
  31. primer_finder-1.0.0/tests/test_seqio.py +120 -0
  32. primer_finder-1.0.0/tests/test_system.py +76 -0
  33. primer_finder-1.0.0/tests/test_tools.py +75 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2019 Marc-Olivier Duceppe
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,115 @@
1
+ Metadata-Version: 2.4
2
+ Name: primer-finder
3
+ Version: 1.0.0
4
+ Summary: Find group-specific kmers to design selective qPCR assays.
5
+ Author: Marc-Olivier Duceppe
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/duceppemo/primer-finder
8
+ Project-URL: Documentation, https://github.com/duceppemo/primer-finder/wiki
9
+ Project-URL: Changelog, https://github.com/duceppemo/primer-finder/blob/master/CHANGELOG.md
10
+ Keywords: bioinformatics,qPCR,primer design,kmer,diagnostics
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Operating System :: POSIX
13
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
14
+ Requires-Python: >=3.10
15
+ Description-Content-Type: text/markdown
16
+ License-File: LICENSE
17
+ Provides-Extra: test
18
+ Requires-Dist: pytest>=7; extra == "test"
19
+ Requires-Dist: pytest-cov>=5; extra == "test"
20
+ Requires-Dist: ruff; extra == "test"
21
+ Requires-Dist: pre-commit; extra == "test"
22
+ Dynamic: license-file
23
+
24
+ # primer-finder
25
+
26
+ <p align="center">
27
+ <a href="https://github.com/duceppemo/primer-finder/actions/workflows/ci.yml"><img src="https://github.com/duceppemo/primer-finder/actions/workflows/ci.yml/badge.svg" alt="CI"></a>
28
+ <a href="https://codecov.io/gh/duceppemo/primer-finder"><img src="https://codecov.io/gh/duceppemo/primer-finder/graph/badge.svg" alt="Coverage"></a>
29
+ <a href="https://github.com/duceppemo/primer-finder/releases/latest"><img src="https://img.shields.io/github/v/release/duceppemo/primer-finder?label=release&cacheSeconds=3600" alt="Latest release"></a>
30
+ <a href="https://anaconda.org/bioconda/primer-finder"><img src="https://img.shields.io/conda/vn/bioconda/primer-finder?label=bioconda" alt="Bioconda"></a>
31
+ <img src="https://img.shields.io/badge/python-3.10%2B-blue" alt="Python 3.10+">
32
+ <a href="LICENSE"><img src="https://img.shields.io/github/license/duceppemo/primer-finder" alt="License: MIT"></a>
33
+ <a href="https://github.com/duceppemo/primer-finder/wiki"><img src="https://img.shields.io/badge/docs-wiki-informational" alt="Documentation"></a>
34
+ <a href="https://zenodo.org/badge/latestdoi/222737677"><img src="https://zenodo.org/badge/222737677.svg" alt="DOI"></a>
35
+ </p>
36
+
37
+ primer-finder finds the sequences that tell one group of genomes from another, so that a selective (q)PCR
38
+ assay can be designed on them. Give it two folders of assembled genomes — the ones the assay should amplify
39
+ (inclusion) and the ones it must not (exclusion) — and it reports the regions that every inclusion genome
40
+ carries, that no exclusion genome carries, and whose differences are close enough together to sit in one
41
+ primer or probe.
42
+
43
+ ```
44
+ inclusion/ ──┐ ┌──► 1. kmers shared by all inclusion genomes (KMC)
45
+ ├──► kmers ──► subtract ──► assemble ──► map ──► blast
46
+ exclusion/ ──┘ (KMC) (SKESA or (minimap2) (every genome)
47
+ SPAdes)
48
+ └──► final_kmers.fasta: the candidate regions,
49
+ with the specific bases in lower case
50
+ ```
51
+
52
+ ## Quick start
53
+
54
+ ```bash
55
+ conda create -n primer-finder -c conda-forge -c bioconda primer-finder
56
+ conda activate primer-finder
57
+
58
+ primer-finder -i inclusion/ -e exclusion/ -o results/
59
+ ```
60
+
61
+ `inclusion/` and `exclusion/` hold one assembled genome per file (`.fasta`, `.fna`, `.fa`, gzipped or not;
62
+ subfolders and symbolic links are followed). The answer is `results/final_kmers.fasta`: one record per
63
+ candidate region, the specific bases in lower case and their positions in the header, the most promising
64
+ first. `results/run_info.json` records the parameters, the genomes and the version of every program used.
65
+
66
+ To install from the source code instead, see
67
+ [Installation](https://github.com/duceppemo/primer-finder/wiki/Installation).
68
+
69
+ primer-finder only keeps perfect matches: a kmer must be in **all** the inclusion genomes with no mismatch,
70
+ and in **none** of the exclusion genomes. It is therefore very sensitive to the quality of the assemblies and
71
+ to how the genomes were assigned to the two groups. Curate the input genomes;
72
+ [genome_comparator](https://github.com/duceppemo/genome_comparator) helps with that.
73
+
74
+ To check an installation, run the bundled example (simulated genomes with a known answer, a few seconds).
75
+ It is in the repository, not in the conda package:
76
+
77
+ ```bash
78
+ curl -sL https://github.com/duceppemo/primer-finder/archive/refs/tags/v1.0.0.tar.gz | tar -xz --strip-components=1 primer-finder-1.0.0/example
79
+ bash example/run_example.sh
80
+ ```
81
+
82
+ ## Ordering the assays
83
+
84
+ Once an assay has been designed from a candidate region and ordered, `primer-finder idt` turns the IDT order
85
+ sheet into a fasta file of oligos, one record per primer and probe:
86
+
87
+ ```bash
88
+ primer-finder idt order.xlsx assays.fasta my_target
89
+ ```
90
+
91
+ ## Documentation
92
+
93
+ Everything else is in the [wiki](https://github.com/duceppemo/primer-finder/wiki), whose sources are
94
+ maintained in [`docs/wiki`](docs/wiki):
95
+
96
+ | Page | Contents |
97
+ |---|---|
98
+ | [Installation](https://github.com/duceppemo/primer-finder/wiki/Installation) | conda, bioconda, from source, checking the installation |
99
+ | [Usage](https://github.com/duceppemo/primer-finder/wiki/Usage) | inputs, every option, choosing the two groups, performance |
100
+ | [Methods](https://github.com/duceppemo/primer-finder/wiki/Methods) | what each step does, the filtering rules, limits |
101
+ | [Outputs](https://github.com/duceppemo/primer-finder/wiki/Outputs) | every file and field |
102
+ | [Example](https://github.com/duceppemo/primer-finder/wiki/Example) | the simulated dataset and its expected result |
103
+ | [FAQ](https://github.com/duceppemo/primer-finder/wiki/FAQ) | troubleshooting, "no contig passed" |
104
+ | [Development](https://github.com/duceppemo/primer-finder/wiki/Development) | tests, continuous integration, releases |
105
+
106
+ ## Citation
107
+
108
+ If primer-finder helped your work, please cite it (see [CITATION.cff](CITATION.cff)) together with the
109
+ programs it runs: [KMC](https://github.com/refresh-bio/KMC),
110
+ [SKESA](https://github.com/ncbi/SKESA) or [SPAdes](https://github.com/ablab/spades),
111
+ [minimap2](https://github.com/lh3/minimap2) and [BLAST](https://blast.ncbi.nlm.nih.gov/).
112
+
113
+ ## License
114
+
115
+ [MIT](LICENSE)
@@ -0,0 +1,92 @@
1
+ # primer-finder
2
+
3
+ <p align="center">
4
+ <a href="https://github.com/duceppemo/primer-finder/actions/workflows/ci.yml"><img src="https://github.com/duceppemo/primer-finder/actions/workflows/ci.yml/badge.svg" alt="CI"></a>
5
+ <a href="https://codecov.io/gh/duceppemo/primer-finder"><img src="https://codecov.io/gh/duceppemo/primer-finder/graph/badge.svg" alt="Coverage"></a>
6
+ <a href="https://github.com/duceppemo/primer-finder/releases/latest"><img src="https://img.shields.io/github/v/release/duceppemo/primer-finder?label=release&cacheSeconds=3600" alt="Latest release"></a>
7
+ <a href="https://anaconda.org/bioconda/primer-finder"><img src="https://img.shields.io/conda/vn/bioconda/primer-finder?label=bioconda" alt="Bioconda"></a>
8
+ <img src="https://img.shields.io/badge/python-3.10%2B-blue" alt="Python 3.10+">
9
+ <a href="LICENSE"><img src="https://img.shields.io/github/license/duceppemo/primer-finder" alt="License: MIT"></a>
10
+ <a href="https://github.com/duceppemo/primer-finder/wiki"><img src="https://img.shields.io/badge/docs-wiki-informational" alt="Documentation"></a>
11
+ <a href="https://zenodo.org/badge/latestdoi/222737677"><img src="https://zenodo.org/badge/222737677.svg" alt="DOI"></a>
12
+ </p>
13
+
14
+ primer-finder finds the sequences that tell one group of genomes from another, so that a selective (q)PCR
15
+ assay can be designed on them. Give it two folders of assembled genomes — the ones the assay should amplify
16
+ (inclusion) and the ones it must not (exclusion) — and it reports the regions that every inclusion genome
17
+ carries, that no exclusion genome carries, and whose differences are close enough together to sit in one
18
+ primer or probe.
19
+
20
+ ```
21
+ inclusion/ ──┐ ┌──► 1. kmers shared by all inclusion genomes (KMC)
22
+ ├──► kmers ──► subtract ──► assemble ──► map ──► blast
23
+ exclusion/ ──┘ (KMC) (SKESA or (minimap2) (every genome)
24
+ SPAdes)
25
+ └──► final_kmers.fasta: the candidate regions,
26
+ with the specific bases in lower case
27
+ ```
28
+
29
+ ## Quick start
30
+
31
+ ```bash
32
+ conda create -n primer-finder -c conda-forge -c bioconda primer-finder
33
+ conda activate primer-finder
34
+
35
+ primer-finder -i inclusion/ -e exclusion/ -o results/
36
+ ```
37
+
38
+ `inclusion/` and `exclusion/` hold one assembled genome per file (`.fasta`, `.fna`, `.fa`, gzipped or not;
39
+ subfolders and symbolic links are followed). The answer is `results/final_kmers.fasta`: one record per
40
+ candidate region, the specific bases in lower case and their positions in the header, the most promising
41
+ first. `results/run_info.json` records the parameters, the genomes and the version of every program used.
42
+
43
+ To install from the source code instead, see
44
+ [Installation](https://github.com/duceppemo/primer-finder/wiki/Installation).
45
+
46
+ primer-finder only keeps perfect matches: a kmer must be in **all** the inclusion genomes with no mismatch,
47
+ and in **none** of the exclusion genomes. It is therefore very sensitive to the quality of the assemblies and
48
+ to how the genomes were assigned to the two groups. Curate the input genomes;
49
+ [genome_comparator](https://github.com/duceppemo/genome_comparator) helps with that.
50
+
51
+ To check an installation, run the bundled example (simulated genomes with a known answer, a few seconds).
52
+ It is in the repository, not in the conda package:
53
+
54
+ ```bash
55
+ curl -sL https://github.com/duceppemo/primer-finder/archive/refs/tags/v1.0.0.tar.gz | tar -xz --strip-components=1 primer-finder-1.0.0/example
56
+ bash example/run_example.sh
57
+ ```
58
+
59
+ ## Ordering the assays
60
+
61
+ Once an assay has been designed from a candidate region and ordered, `primer-finder idt` turns the IDT order
62
+ sheet into a fasta file of oligos, one record per primer and probe:
63
+
64
+ ```bash
65
+ primer-finder idt order.xlsx assays.fasta my_target
66
+ ```
67
+
68
+ ## Documentation
69
+
70
+ Everything else is in the [wiki](https://github.com/duceppemo/primer-finder/wiki), whose sources are
71
+ maintained in [`docs/wiki`](docs/wiki):
72
+
73
+ | Page | Contents |
74
+ |---|---|
75
+ | [Installation](https://github.com/duceppemo/primer-finder/wiki/Installation) | conda, bioconda, from source, checking the installation |
76
+ | [Usage](https://github.com/duceppemo/primer-finder/wiki/Usage) | inputs, every option, choosing the two groups, performance |
77
+ | [Methods](https://github.com/duceppemo/primer-finder/wiki/Methods) | what each step does, the filtering rules, limits |
78
+ | [Outputs](https://github.com/duceppemo/primer-finder/wiki/Outputs) | every file and field |
79
+ | [Example](https://github.com/duceppemo/primer-finder/wiki/Example) | the simulated dataset and its expected result |
80
+ | [FAQ](https://github.com/duceppemo/primer-finder/wiki/FAQ) | troubleshooting, "no contig passed" |
81
+ | [Development](https://github.com/duceppemo/primer-finder/wiki/Development) | tests, continuous integration, releases |
82
+
83
+ ## Citation
84
+
85
+ If primer-finder helped your work, please cite it (see [CITATION.cff](CITATION.cff)) together with the
86
+ programs it runs: [KMC](https://github.com/refresh-bio/KMC),
87
+ [SKESA](https://github.com/ncbi/SKESA) or [SPAdes](https://github.com/ablab/spades),
88
+ [minimap2](https://github.com/lh3/minimap2) and [BLAST](https://blast.ncbi.nlm.nih.gov/).
89
+
90
+ ## License
91
+
92
+ [MIT](LICENSE)
@@ -0,0 +1,8 @@
1
+ """primer-finder: find group-specific kmers to design selective qPCR assays."""
2
+
3
+ __version__ = "1.0.0"
4
+ __author__ = "Marc-Olivier Duceppe"
5
+
6
+
7
+ class PrimerFinderError(Exception):
8
+ """A user-facing error: bad input, a missing program, or a failed external command."""
@@ -0,0 +1,5 @@
1
+ import sys
2
+
3
+ from primer_finder.cli import main
4
+
5
+ sys.exit(main())
@@ -0,0 +1,58 @@
1
+ """Assembling the inclusion-specific kmers into contigs, with SKESA or SPAdes."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ import shutil
7
+ from pathlib import Path
8
+
9
+ from primer_finder import PrimerFinderError, tools
10
+
11
+ log = logging.getLogger(__name__)
12
+
13
+ ASSEMBLERS = ("skesa", "spades")
14
+ PROGRAMS = {"skesa": "skesa", "spades": "spades.py"}
15
+
16
+
17
+ def assemble(kmer_fasta: Path, output: Path, assembler: str, threads: int, memory_gb: int) -> Path:
18
+ """Assemble a fasta file of kmers into `output`. Returns `output`."""
19
+ if assembler == "skesa":
20
+ _skesa(kmer_fasta, output, threads, memory_gb)
21
+ elif assembler == "spades":
22
+ _spades(kmer_fasta, output, threads, memory_gb)
23
+ else: # pragma: no cover - the command line only accepts the two
24
+ raise PrimerFinderError(f'Unknown assembler "{assembler}". Choose one of: {", ".join(ASSEMBLERS)}')
25
+ if not output.exists() or output.stat().st_size == 0:
26
+ raise PrimerFinderError(
27
+ f"{assembler} could not assemble the inclusion-specific kmers into contigs. "
28
+ "There may be too few of them; a smaller kmer size (-k) may help."
29
+ )
30
+ return output
31
+
32
+
33
+ def _skesa(kmer_fasta: Path, output: Path, threads: int, memory_gb: int) -> None:
34
+ tools.run([
35
+ "skesa",
36
+ "--cores", str(threads),
37
+ "--mem", str(memory_gb),
38
+ "--fasta", kmer_fasta,
39
+ "--contigs_out", output,
40
+ ])
41
+
42
+
43
+ def _spades(kmer_fasta: Path, output: Path, threads: int, memory_gb: int) -> None:
44
+ work_dir = output.parent / "spades"
45
+ tools.run([
46
+ "spades.py",
47
+ "--s", "1", kmer_fasta,
48
+ "--isolate",
49
+ "--only-assembler", # The kmers carry no quality values: there is nothing to correct
50
+ "--threads", str(threads),
51
+ "--memory", str(memory_gb),
52
+ "-o", work_dir,
53
+ ])
54
+ contigs = work_dir / "contigs.fasta"
55
+ if not contigs.exists():
56
+ raise PrimerFinderError(f"SPAdes wrote no contigs ({contigs} missing)")
57
+ shutil.move(str(contigs), output)
58
+ shutil.rmtree(work_dir, ignore_errors=True)
@@ -0,0 +1,238 @@
1
+ """Checking the candidate contigs against every inclusion and exclusion genome with blast."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import gzip
6
+ import logging
7
+ import os
8
+ import shutil
9
+ from collections.abc import Iterable, Sequence
10
+ from concurrent import futures
11
+ from dataclasses import dataclass, field
12
+ from pathlib import Path
13
+
14
+ from primer_finder import tools
15
+ from primer_finder.mapping import PRIMER_LENGTH
16
+ from primer_finder.seqio import base_name, iter_records
17
+
18
+ log = logging.getLogger(__name__)
19
+
20
+ # A hit has to be at least this good to count as "the contig is there".
21
+ MAX_EVALUE = 1e-10
22
+ # A variant position is kept when it differs in at least this fraction of the exclusion genomes that hold
23
+ # the region: a few exclusion genomes may carry the inclusion allele without making the assay useless.
24
+ MIN_EXCLUSION_FRACTION = 0.90
25
+
26
+ PRESENCE_FIELDS = ("qseqid", "evalue")
27
+ ALIGNMENT_FIELDS = ("qseqid", "qstart", "qend", "evalue", "qseq", "sseq")
28
+
29
+ # BLAST splits its path arguments on whitespace, so it is never given a path that could hold a space: each
30
+ # genome is linked into its own folder under these fixed names, and blast runs in that folder.
31
+ LOCAL_FASTA = "genome.fasta"
32
+ DB_NAME = "db"
33
+ HITS_NAME = "hits.tsv"
34
+
35
+
36
+ @dataclass
37
+ class Hit:
38
+ """One blast high-scoring pair, with the aligned query and subject sequences."""
39
+
40
+ query: str
41
+ qstart: int
42
+ qend: int
43
+ evalue: float
44
+ qseq: str = ""
45
+ sseq: str = ""
46
+
47
+
48
+ @dataclass
49
+ class Alignments:
50
+ """What one exclusion genome holds of one contig: every part of the contig it aligns, and where each
51
+ of those alignments differs from it."""
52
+
53
+ spans: list[tuple[int, int, set[int]]] = field(default_factory=list) # start, end (0-based), variants
54
+
55
+ def add(self, hit: Hit) -> None:
56
+ self.spans.append((hit.qstart - 1, hit.qend, set(variant_positions(hit))))
57
+
58
+ def covers(self, position: int) -> bool:
59
+ return any(start <= position < end for start, end, _ in self.spans)
60
+
61
+ def differs_at(self, position: int) -> bool:
62
+ """True when every copy of this region in the genome differs from the contig at this position. One
63
+ matching copy is enough for an assay to amplify the genome, so one is enough to say no."""
64
+ covering = [(start, end, variants) for start, end, variants in self.spans if start <= position < end]
65
+ return bool(covering) and all(position in variants for _, _, variants in covering)
66
+
67
+ @property
68
+ def variants(self) -> set[int]:
69
+ return {position for _, _, variants in self.spans for position in variants}
70
+
71
+
72
+ @dataclass
73
+ class ExclusionResult:
74
+ """What the exclusion genomes say about one contig: one entry per genome that holds any part of it."""
75
+
76
+ genomes: list[Alignments] = field(default_factory=list)
77
+
78
+ @property
79
+ def genomes_hit(self) -> int:
80
+ return len(self.genomes)
81
+
82
+
83
+ def make_db(genome: Path, work_dir: Path) -> Path:
84
+ """Make a blast database for one genome, inside `work_dir` so that the input folders are left alone.
85
+
86
+ The genome is linked into `work_dir` under a fixed name (a gzipped one is decompressed: makeblastdb does
87
+ not read gzip), and makeblastdb runs there with relative paths, so neither a space in the path nor two
88
+ genomes with the same file name can reach BLAST.
89
+ """
90
+ work_dir.mkdir(parents=True, exist_ok=True)
91
+ local = work_dir / LOCAL_FASTA
92
+ if local.exists() or local.is_symlink():
93
+ local.unlink()
94
+ if genome.name.endswith(".gz"):
95
+ with gzip.open(genome, "rb") as src, local.open("wb") as dst:
96
+ shutil.copyfileobj(src, dst)
97
+ else:
98
+ try:
99
+ local.symlink_to(genome.resolve())
100
+ except OSError: # pragma: no cover - a filesystem without symbolic links
101
+ shutil.copyfile(genome, local)
102
+ tools.run(["makeblastdb", "-in", LOCAL_FASTA, "-dbtype", "nucl", "-out", DB_NAME], cwd=work_dir)
103
+ return work_dir / DB_NAME
104
+
105
+
106
+ def blastn(db: Path, query: Path, out_file: Path, fields: Sequence[str], max_targets: int = 10) -> Path:
107
+ """Run blastn in the database's folder, writing the chosen tabular fields to `out_file`."""
108
+ work_dir = db.parent
109
+ tools.run([
110
+ "blastn",
111
+ "-db", db.name,
112
+ "-query", os.path.relpath(query, work_dir),
113
+ "-out", os.path.relpath(out_file, work_dir),
114
+ "-evalue", str(MAX_EVALUE),
115
+ "-max_target_seqs", str(max_targets),
116
+ "-num_threads", "1",
117
+ "-outfmt", "6 " + " ".join(fields),
118
+ ], cwd=work_dir)
119
+ return out_file
120
+
121
+
122
+ def parse_hits(out_file: Path, fields: Sequence[str]) -> list[Hit]:
123
+ """Read a tabular blast output written with `fields`."""
124
+ hits: list[Hit] = []
125
+ index = {name: position for position, name in enumerate(fields)}
126
+ with out_file.open() as fh:
127
+ for line in fh:
128
+ values = line.rstrip("\n").split("\t")
129
+ if len(values) != len(fields):
130
+ continue
131
+ hits.append(Hit(
132
+ query=values[index["qseqid"]],
133
+ qstart=int(values[index["qstart"]]) if "qstart" in index else 0,
134
+ qend=int(values[index["qend"]]) if "qend" in index else 0,
135
+ evalue=float(values[index["evalue"]]),
136
+ qseq=values[index["qseq"]] if "qseq" in index else "",
137
+ sseq=values[index["sseq"]] if "sseq" in index else "",
138
+ ))
139
+ return hits
140
+
141
+
142
+ def variant_positions(hit: Hit) -> list[int]:
143
+ """The positions of the contig (0-based) that differ from this exclusion sequence: mismatches, inserted
144
+ bases, and the base next to a deletion."""
145
+ positions: list[int] = []
146
+ position = hit.qstart - 1
147
+ for query_base, subject_base in zip(hit.qseq, hit.sseq, strict=False):
148
+ if query_base == "-": # The exclusion genome has bases the contig does not: mark where they fit in
149
+ positions.append(position)
150
+ continue
151
+ if subject_base == "-" or query_base.upper() != subject_base.upper():
152
+ positions.append(position)
153
+ position += 1
154
+ return sorted(set(positions))
155
+
156
+
157
+ def has_close_variants(positions: Sequence[int], window: int = PRIMER_LENGTH) -> bool:
158
+ """True if at least two of these positions are less than `window` bases apart."""
159
+ return any(second - first < window for first, second in zip(positions, positions[1:], strict=False))
160
+
161
+
162
+ def _parallel(work: Iterable[tuple], function, threads: int) -> list:
163
+ """Run `function(*arguments)` for every item, at most `threads` at a time, keeping the input order."""
164
+ work = list(work)
165
+ if not work:
166
+ return []
167
+ with futures.ThreadPoolExecutor(max_workers=max(1, min(threads, len(work)))) as executor:
168
+ return list(executor.map(lambda arguments: function(*arguments), work))
169
+
170
+
171
+ def genome_folder(work_dir: Path, index: int, genome: Path) -> Path:
172
+ """A folder of its own for each genome: two genomes in different subfolders can have the same file
173
+ name, so the position in the list is what makes it unique."""
174
+ return work_dir / f"{index:04d}_{base_name(genome)}"
175
+
176
+
177
+ def presence_in_genomes(
178
+ query: Path, genomes: Sequence[Path], work_dir: Path, threads: int
179
+ ) -> dict[str, dict[str, bool]]:
180
+ """For every contig of `query`, whether it is present in each genome. Keyed by contig, then by the
181
+ genome's path, which is what tells two genomes with the same file name apart."""
182
+ def one(index: int, genome: Path) -> set[str]:
183
+ folder = genome_folder(work_dir, index, genome)
184
+ db = make_db(genome, folder)
185
+ hits = parse_hits(blastn(db, query, folder / HITS_NAME, PRESENCE_FIELDS, max_targets=1),
186
+ PRESENCE_FIELDS)
187
+ return {hit.query for hit in hits if hit.evalue <= MAX_EVALUE}
188
+
189
+ found = _parallel(enumerate(genomes), one, threads)
190
+ contigs = _query_names(query)
191
+ presence: dict[str, dict[str, bool]] = {contig: {} for contig in contigs}
192
+ for genome, hits in zip(genomes, found, strict=True):
193
+ for contig in contigs:
194
+ presence[contig][str(genome)] = contig in hits
195
+ return presence
196
+
197
+
198
+ def _query_names(query: Path) -> list[str]:
199
+ """The names of the contigs in a fasta file, in file order."""
200
+ return [record.name for record in iter_records(query)]
201
+
202
+
203
+ def exclusion_variants(
204
+ query: Path, genomes: Sequence[Path], work_dir: Path, threads: int
205
+ ) -> dict[str, ExclusionResult]:
206
+ """For every contig of `query`, what each exclusion genome that holds part of it looks like there."""
207
+ def one(index: int, genome: Path) -> list[Hit]:
208
+ folder = genome_folder(work_dir, index, genome)
209
+ db = make_db(genome, folder)
210
+ return parse_hits(blastn(db, query, folder / HITS_NAME, ALIGNMENT_FIELDS), ALIGNMENT_FIELDS)
211
+
212
+ results: dict[str, ExclusionResult] = {name: ExclusionResult() for name in _query_names(query)}
213
+ for hits in _parallel(enumerate(genomes), one, threads):
214
+ per_contig: dict[str, Alignments] = {}
215
+ for hit in hits:
216
+ if hit.evalue > MAX_EVALUE:
217
+ continue
218
+ per_contig.setdefault(hit.query, Alignments()).add(hit)
219
+ for contig, alignments in per_contig.items():
220
+ results.setdefault(contig, ExclusionResult()).genomes.append(alignments)
221
+ return results
222
+
223
+
224
+ def shared_variants(result: ExclusionResult, fraction: float = MIN_EXCLUSION_FRACTION) -> list[int]:
225
+ """The contig positions worth designing a primer on: those that differ from at least `fraction` of the
226
+ exclusion genomes that hold that part of the contig.
227
+
228
+ A genome that does not align a position says nothing about it — it does not hold that region, which only
229
+ makes the assay more selective — so it is left out of the count rather than counted as identical.
230
+ """
231
+ kept: list[int] = []
232
+ candidates = {position for alignments in result.genomes for position in alignments.variants}
233
+ for position in sorted(candidates):
234
+ covering = [alignments for alignments in result.genomes if alignments.covers(position)]
235
+ differing = [alignments for alignments in covering if alignments.differs_at(position)]
236
+ if covering and len(differing) >= fraction * len(covering):
237
+ kept.append(position)
238
+ return kept