prismalign 0.2.2__tar.gz → 0.2.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Potentially problematic release.
This version of prismalign might be problematic. Click here for more details.
- {prismalign-0.2.2/prismalign.egg-info → prismalign-0.2.4}/PKG-INFO +74 -17
- prismalign-0.2.4/README.md +172 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/__init__.py +26 -3
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/adapters/__init__.py +10 -1
- prismalign-0.2.4/prismalign/adapters/base.py +471 -0
- prismalign-0.2.4/prismalign/adapters/bowtie2/__init__.py +52 -0
- prismalign-0.2.4/prismalign/adapters/bwa/__init__.py +33 -0
- prismalign-0.2.4/prismalign/adapters/bwa_mem2/__init__.py +44 -0
- prismalign-0.2.4/prismalign/adapters/hisat2/__init__.py +42 -0
- prismalign-0.2.4/prismalign/adapters/minimap2/__init__.py +29 -0
- prismalign-0.2.4/prismalign/adapters/sam/__init__.py +61 -0
- prismalign-0.2.4/prismalign/adapters/strobealign/__init__.py +60 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/__init__.py +9 -1
- prismalign-0.2.4/prismalign/backends/bwamem2/__init__.py +102 -0
- prismalign-0.2.4/prismalign/backends/registry.py +106 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/cli.py +4 -1
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/engine.py +139 -51
- {prismalign-0.2.2 → prismalign-0.2.4/prismalign.egg-info}/PKG-INFO +74 -17
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign.egg-info/SOURCES.txt +6 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign.egg-info/requires.txt +3 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/pyproject.toml +15 -1
- prismalign-0.2.4/tests/test_adapters.py +191 -0
- prismalign-0.2.4/tests/test_backends.py +149 -0
- prismalign-0.2.4/tests/test_batch_backend.py +63 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/tests/test_sam_backend.py +4 -4
- prismalign-0.2.2/README.md +0 -117
- prismalign-0.2.2/prismalign/adapters/bowtie2/__init__.py +0 -73
- prismalign-0.2.2/prismalign/adapters/bwa_mem2/__init__.py +0 -68
- prismalign-0.2.2/prismalign/adapters/sam/__init__.py +0 -140
- prismalign-0.2.2/prismalign/adapters/strobealign/__init__.py +0 -52
- prismalign-0.2.2/prismalign/backends/registry.py +0 -36
- prismalign-0.2.2/tests/test_adapters.py +0 -95
- prismalign-0.2.2/tests/test_backends.py +0 -84
- {prismalign-0.2.2 → prismalign-0.2.4}/LICENSE +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/MANIFEST.in +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/base.py +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/bwamem/__init__.py +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/mappy/__init__.py +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/minibwa/__init__.py +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/__init__.py +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/LICENSE.wfa2 +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/alignment/affine2p_penalties.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/alignment/affine2p_penalties.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/alignment/affine_penalties.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/alignment/affine_penalties.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/alignment/cigar.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/alignment/cigar.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/alignment/cigar_utils.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/alignment/cigar_utils.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/alignment/linear_penalties.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/alignment/score_matrix.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/alignment/score_matrix.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/system/mm_allocator.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/system/mm_allocator.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/system/mm_stack.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/system/mm_stack.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/system/profiler_counter.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/system/profiler_counter.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/system/profiler_timer.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/system/profiler_timer.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/utils/bitmap.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/utils/bitmap.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/utils/commons.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/utils/commons.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/utils/dna_text.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/utils/dna_text.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/utils/heatmap.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/utils/heatmap.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/utils/sequence_buffer.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/utils/sequence_buffer.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/utils/vector.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/utils/vector.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_align.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_align.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_aligner.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_aligner.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_attributes.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_attributes.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_backtrace.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_backtrace.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_backtrace_buffer.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_backtrace_buffer.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_backtrace_offload.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_backtrace_offload.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_bialign.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_bialign.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_bialigner.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_bialigner.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_components.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_components.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_compute.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_compute.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_compute_affine.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_compute_affine.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_compute_affine2p.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_compute_affine2p.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_compute_edit.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_compute_edit.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_compute_linear.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_compute_linear.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_debug.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_debug.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_display.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_display.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_extend.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_extend.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_extend_kernels.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_extend_kernels.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_extend_kernels_avx.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_heuristic.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_heuristic.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_offset.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_pcigar.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_pcigar.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_penalties.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_penalties.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_plot.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_plot.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_sequences.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_sequences.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_slab.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_slab.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_termination.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_termination.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_unialign.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wavefront_unialign.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wfa2lib/wavefront/wfa.h +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/backends/wfa2/wrap.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/colorops.py +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/em.py +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/schemes.py +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign/seqops.c +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign.egg-info/dependency_links.txt +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign.egg-info/entry_points.txt +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/prismalign.egg-info/top_level.txt +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/setup.cfg +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/setup.py +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/tests/test_backends_matrix.py +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/tests/test_colorops.py +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/tests/test_em.py +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/tests/test_engine.py +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/tests/test_hierarchy.py +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/tests/test_parity.py +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/tests/test_pe_minibwa.py +0 -0
- {prismalign-0.2.2 → prismalign-0.2.4}/tests/test_wfa2_backend.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: prismalign
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.4
|
|
4
4
|
Summary: N-color (2-color / 3-color / 3-nt) nucleotide-conversion alignment engine with pluggable backends
|
|
5
5
|
Author-email: Chang Ye <yech1990@gmail.com>
|
|
6
6
|
License-Expression: GPL-3.0-only
|
|
@@ -17,6 +17,8 @@ Provides-Extra: mappy
|
|
|
17
17
|
Requires-Dist: mappy>=2.24; extra == "mappy"
|
|
18
18
|
Provides-Extra: minibwa
|
|
19
19
|
Requires-Dist: minibwa>=0.1.7; extra == "minibwa"
|
|
20
|
+
Provides-Extra: bwamem2
|
|
21
|
+
Requires-Dist: bwamem2>=0.1.0; extra == "bwamem2"
|
|
20
22
|
Dynamic: license-file
|
|
21
23
|
|
|
22
24
|
# Prismalign
|
|
@@ -62,8 +64,12 @@ with ps.NColorMapper(scheme=ps.BS, backend="bwamem") as mapper:
|
|
|
62
64
|
## Usage — CLI
|
|
63
65
|
|
|
64
66
|
```bash
|
|
65
|
-
# classic two-color (MK: A->G + C->T)
|
|
66
|
-
|
|
67
|
+
# classic two-color (MK: A->G + C->T); --backend auto picks the fastest
|
|
68
|
+
# importable backend (minibwa > mappy > bwamem)
|
|
69
|
+
prismalign map -s MK -r ref.fa -o out.bam reads.fq
|
|
70
|
+
|
|
71
|
+
# force the fastest native backend (bwa-mem2 speed tier)
|
|
72
|
+
prismalign map -s MK --backend minibwa -r ref.fa -o out.bam reads.fq
|
|
67
73
|
|
|
68
74
|
# bisulfite-seq (3-nt single channel C->T)
|
|
69
75
|
prismalign map -s BS -r genome.fa -o bs.bam --index-dir idx reads.fq
|
|
@@ -96,37 +102,88 @@ mapper.map_file(r1_file="reads.fq", ref_files=["genome.fa"],
|
|
|
96
102
|
output_files=["out.bam"])
|
|
97
103
|
```
|
|
98
104
|
|
|
99
|
-
## Backends
|
|
105
|
+
## Backends vs Adapters — two integration layers
|
|
106
|
+
|
|
107
|
+
Prismalign plugs in aligners at **two layers**:
|
|
100
108
|
|
|
101
|
-
|
|
102
|
-
|
|
109
|
+
1. **Backends** — **in-process** (compiled C / a Python binding). The engine's per-read
|
|
110
|
+
`map_one` calls `backend.align(seq)` directly. Selectable via `--backend`.
|
|
111
|
+
2. **Adapters** — **subprocess** wrappers for external *command-line* aligners. They
|
|
112
|
+
map reads at three granularities (`map_read`/`map_batch`/`map_file`), unified on the
|
|
113
|
+
`CliAdapter` base; *not* `--backend`-selectable.
|
|
103
114
|
|
|
115
|
+
### Backends (in-process, `--backend`)
|
|
104
116
|
| backend | engine | notes |
|
|
105
117
|
|---------|--------|-------|
|
|
106
118
|
| `bwamem` | BWA-MEM via the `bwamem` package | default, fast C backend (SE + PE) |
|
|
107
119
|
| `minibwa` | **lh3/minibwa** (bwa-mem successor) via **PyO3 pip binding `minibwa`** (fg-labs) | ~2-3x faster than bwa-mem; `pip install minibwa` (SE + PE) |
|
|
108
120
|
| `mappy` | minimap2 via `mappy` | official minimap2 Python binding (SE + PE) |
|
|
121
|
+
| `bwamem2` | **bwa-mem2** native in-process via the `bwamem2` Cython binding | correct, but **per-read**; `pip install 'prismalign[bwamem2]'` (standard, non-free-threaded CPython) |
|
|
109
122
|
| `wfa2` | **WFA2-lib** (vendored v2.3.6, MIT) compiled in-process | exact gapped (indel-aware) wavefront alignment; SE only |
|
|
110
123
|
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
124
|
+
### Adapters (subprocess, `prismalign.adapters`)
|
|
125
|
+
All built on the shared `CliAdapter` base, which maps reads at three granularities:
|
|
126
|
+
|
|
127
|
+
| method | granularity | notes |
|
|
128
|
+
|---|---|---|
|
|
129
|
+
| `map_read(seq)` | 1 read | compat / debug (slow: one subprocess per read) |
|
|
130
|
+
| `map_batch(reads)` | 1 batch (`(name,seq,qual)` list) | one tool invocation |
|
|
131
|
+
| `map_file(fastq, batch_size, threads, workers)` | whole FASTQ (plain/.gz) | **path prismalign reads + batches itself** (default `batch_size=8192` = 8×bwa-mem2's internal 512); `workers>1` runs batches **concurrently** (each an independent tool subprocess — the real throughput lever, since a tool's own threads saturate on a shared index) |
|
|
132
|
+
|
|
133
|
+
| adapter | tool | index |
|
|
134
|
+
|---|---|---|
|
|
135
|
+
| `BwaMemAdapter` | `bwa mem` | `bwa index` |
|
|
136
|
+
| `BwaMem2Adapter` | `bwa-mem2 mem` | `bwa-mem2 index` |
|
|
137
|
+
| `Bowtie2Adapter` | `bowtie2` | `bowtie2-build` |
|
|
138
|
+
| `Minimap2Adapter` | `minimap2 -a` | none (reads FASTA directly) |
|
|
139
|
+
| `Hisat2Adapter` | `hisat2` | `hisat2-build` |
|
|
140
|
+
| `StrobealignAdapter` | `strobealign` | `.sti` |
|
|
141
|
+
| `SamAdapter` | any SAM mapper | (generic; you supply the command template) |
|
|
142
|
+
|
|
143
|
+
Each adapter needs its binary on PATH (or an env var: `BWA_BIN`, `BWA_MEM2_BIN`,
|
|
144
|
+
`BOWTIE2_BIN`, `MINIMAP2_BIN`, `HISAT2_BIN`, `STROBEALIGN_BIN`).
|
|
145
|
+
|
|
146
|
+
**Throughput tip:** for a large job use ``map_file(path, batch_size=8192,
|
|
147
|
+
workers=cpu/2..cpu, threads=1)`` — batching at 8192 (a multiple of the tool's
|
|
148
|
+
internal 512) amortizes the subprocess startup, and ``workers`` parallelizes
|
|
149
|
+
batches across **separate tool processes** (each with its own memory bandwidth;
|
|
150
|
+
the tool's own ``-t`` saturates, so parallelize via processes). `.gz` input is
|
|
151
|
+
read with an internal fast library (`isal`/`xopen`/`rapidgzip`) when available.
|
|
152
|
+
|
|
153
|
+
### bwa-mem2 appears at BOTH layers (intentional)
|
|
154
|
+
- **`BwaMem2Backend`** (backend) = bwa-mem2 **in-process**, per-read — for embedding / `--backend bwamem2`.
|
|
155
|
+
- **`BwaMem2Adapter`** (adapter) = bwa-mem2 **batched** CLI (`map_file`) — for throughput.
|
|
156
|
+
|
|
157
|
+
For large references, bwa-mem2's ~2-3x (from SIMD FM-index search) shows up; build it
|
|
158
|
+
**cleanly for AVX2/AVX-512** (stale object files cause SIGILL). The in-process backend
|
|
159
|
+
is per-read (batch the calls, or use the adapter, for throughput).
|
|
160
|
+
|
|
161
|
+
Full inventory — including where each wrapper lives — in [`docs/backends.md`](docs/backends.md).
|
|
119
162
|
|
|
120
163
|
## Speed & IO
|
|
121
164
|
|
|
122
|
-
* **
|
|
123
|
-
|
|
124
|
-
|
|
165
|
+
* **Auto backend**: `--backend auto` (explicit) picks the fastest *importable*
|
|
166
|
+
native backend — `minibwa` (~2-3x BWA-MEM, the bwa-mem2 speed tier), else
|
|
167
|
+
`mappy` (in-process), else `bwamem`. (Default is `bwamem`.) The chosen
|
|
168
|
+
backend is printed at startup (`[prismalign] backend auto=minibwa`).
|
|
169
|
+
* **Process parallelism is the lever**: `-t/--threads N` maps reads in an
|
|
170
|
+
ordered fork+COW process pool (any backend); batches are drained in read
|
|
171
|
+
order so the BAM is **byte-identical** to `threads=1`. Processes (each with
|
|
172
|
+
its own index) scale; bwa-mem2's internal `-t` threads do *not* (they share
|
|
173
|
+
one index and are memory-bound). So scale **processes/workers**, not
|
|
174
|
+
bwa-mem2 threads.
|
|
125
175
|
* **Reduced repeated IO**: references are copy+converted **once** even when
|
|
126
176
|
reused across layers (cache keyed by path+scheme); per-hit reference fetch
|
|
127
177
|
is cached in memory for small contigs (RNA/transcript references), so only
|
|
128
178
|
one indexed read per contig.
|
|
129
179
|
|
|
180
|
+
> **Throughput ceiling.** The heavy per-read work (BWA-MEM / minibwa / minimap2
|
|
181
|
+
> / WFA2 kernels) is native C. With BWA-MEM the runtime is essentially that
|
|
182
|
+
> kernel's throughput — prismalign's glue (conversion, re-scoring, tag
|
|
183
|
+
> emission) adds only a small fraction. For the largest runs, use `minibwa`
|
|
184
|
+
> (runs on standard CPython; no free-threaded-3.14 wheel) or `mappy`; both are
|
|
185
|
+
> far faster than BWA-MEM.
|
|
186
|
+
|
|
130
187
|
## Limitations (v0.2.x)
|
|
131
188
|
|
|
132
189
|
* paired-end is supported natively by the `bwamem`, `mappy` and `minibwa`
|
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
# Prismalign
|
|
2
|
+
|
|
3
|
+
**N-color nucleotide-conversion alignment engine** with pluggable backends.
|
|
4
|
+
|
|
5
|
+
Prismalign maps sequencing reads from any nucleotide-conversion chemistry
|
|
6
|
+
(bisulfite-seq `C→T`, SLAM-seq `T→C`, m6A / A-to-I `A→G`, MK/KM dual-base,
|
|
7
|
+
or a custom 3rd channel) using a **HISAT-3N-style** strategy:
|
|
8
|
+
|
|
9
|
+
1. build a *converted* reference index (`scheme.ref_from → ref_to`)
|
|
10
|
+
2. transform each read per color channel and align it to the converted index
|
|
11
|
+
via a pluggable backend (**bwamem** by default; **WFA2-lib**;
|
|
12
|
+
**minimap2/mappy**; **minibwa**, all optional)
|
|
13
|
+
3. **re-score every hit against the *original* reference** so that real
|
|
14
|
+
conversions are rewarded (not counted as mismatches), emitting a
|
|
15
|
+
color-correct `MD` plus per-channel `Y`/`Z` counts in BAM tags.
|
|
16
|
+
|
|
17
|
+
All per-read heavy kernels are native C (BWA-MEM / WFA2-lib / minimap2 /
|
|
18
|
+
minibwa); the Python layer is a thin, friendly wrapper.
|
|
19
|
+
|
|
20
|
+
## Install
|
|
21
|
+
|
|
22
|
+
```bash
|
|
23
|
+
pip install -e . # bwamem + built-in WFA2 C backends
|
|
24
|
+
pip install -e "./[mappy]" # + minimap2 backend
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
## Usage — Python (clean wrapper)
|
|
28
|
+
|
|
29
|
+
```python
|
|
30
|
+
import prismalign as ps
|
|
31
|
+
|
|
32
|
+
# one-shot mapping -> BAM (builds indexes, maps, cleans up)
|
|
33
|
+
ps.map_reads("reads.fq", "ref.fa", "out.bam",
|
|
34
|
+
scheme="MK", backend="bwamem", threads=4)
|
|
35
|
+
|
|
36
|
+
# object API / reuse
|
|
37
|
+
with ps.NColorMapper(scheme=ps.BS, backend="bwamem") as mapper:
|
|
38
|
+
mapper.map_file("reads.fq", ref_files=["ref.fa"], output_files=["bs.bam"])
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
## Usage — CLI
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
# classic two-color (MK: A->G + C->T); --backend auto picks the fastest
|
|
45
|
+
# importable backend (minibwa > mappy > bwamem)
|
|
46
|
+
prismalign map -s MK -r ref.fa -o out.bam reads.fq
|
|
47
|
+
|
|
48
|
+
# force the fastest native backend (bwa-mem2 speed tier)
|
|
49
|
+
prismalign map -s MK --backend minibwa -r ref.fa -o out.bam reads.fq
|
|
50
|
+
|
|
51
|
+
# bisulfite-seq (3-nt single channel C->T)
|
|
52
|
+
prismalign map -s BS -r genome.fa -o bs.bam --index-dir idx reads.fq
|
|
53
|
+
|
|
54
|
+
# parallel (2 copies of the reads, byte-identical output to -t 1)
|
|
55
|
+
prismalign map -s MK -r ref.fa -o out.bam -t 4 reads.fq
|
|
56
|
+
|
|
57
|
+
# list built-in schemes
|
|
58
|
+
prismalign schemes
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
## Schemes
|
|
62
|
+
|
|
63
|
+
| name | reference index | channels | use case |
|
|
64
|
+
|-------|-----------------|----------|----------|
|
|
65
|
+
| `MK` | `AC→GT` | 2 | dual-base conversion A→G + C→T (classic two-color) |
|
|
66
|
+
| `KM` | `GT→AC` | 2 | reverse of MK |
|
|
67
|
+
| `BS` | `C→T` | 1 | bisulfite-seq (3-nt) |
|
|
68
|
+
| `SLAM`| `T→C` | 1 | SLAM-seq |
|
|
69
|
+
| `A2G` | `A→G` | 1 | m6A / A-to-I editing |
|
|
70
|
+
| `THREE`| `AC→GT` | 3 | three-color demo (add your 3rd base pair in `schemes.py`) |
|
|
71
|
+
|
|
72
|
+
## Python API
|
|
73
|
+
|
|
74
|
+
```python
|
|
75
|
+
from prismalign import NColorMapper, BS
|
|
76
|
+
|
|
77
|
+
mapper = NColorMapper(scheme=BS, backend="bwamem", index_dir="idx")
|
|
78
|
+
mapper.map_file(r1_file="reads.fq", ref_files=["genome.fa"],
|
|
79
|
+
output_files=["out.bam"])
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
## Backends vs Adapters — two integration layers
|
|
83
|
+
|
|
84
|
+
Prismalign plugs in aligners at **two layers**:
|
|
85
|
+
|
|
86
|
+
1. **Backends** — **in-process** (compiled C / a Python binding). The engine's per-read
|
|
87
|
+
`map_one` calls `backend.align(seq)` directly. Selectable via `--backend`.
|
|
88
|
+
2. **Adapters** — **subprocess** wrappers for external *command-line* aligners. They
|
|
89
|
+
map reads at three granularities (`map_read`/`map_batch`/`map_file`), unified on the
|
|
90
|
+
`CliAdapter` base; *not* `--backend`-selectable.
|
|
91
|
+
|
|
92
|
+
### Backends (in-process, `--backend`)
|
|
93
|
+
| backend | engine | notes |
|
|
94
|
+
|---------|--------|-------|
|
|
95
|
+
| `bwamem` | BWA-MEM via the `bwamem` package | default, fast C backend (SE + PE) |
|
|
96
|
+
| `minibwa` | **lh3/minibwa** (bwa-mem successor) via **PyO3 pip binding `minibwa`** (fg-labs) | ~2-3x faster than bwa-mem; `pip install minibwa` (SE + PE) |
|
|
97
|
+
| `mappy` | minimap2 via `mappy` | official minimap2 Python binding (SE + PE) |
|
|
98
|
+
| `bwamem2` | **bwa-mem2** native in-process via the `bwamem2` Cython binding | correct, but **per-read**; `pip install 'prismalign[bwamem2]'` (standard, non-free-threaded CPython) |
|
|
99
|
+
| `wfa2` | **WFA2-lib** (vendored v2.3.6, MIT) compiled in-process | exact gapped (indel-aware) wavefront alignment; SE only |
|
|
100
|
+
|
|
101
|
+
### Adapters (subprocess, `prismalign.adapters`)
|
|
102
|
+
All built on the shared `CliAdapter` base, which maps reads at three granularities:
|
|
103
|
+
|
|
104
|
+
| method | granularity | notes |
|
|
105
|
+
|---|---|---|
|
|
106
|
+
| `map_read(seq)` | 1 read | compat / debug (slow: one subprocess per read) |
|
|
107
|
+
| `map_batch(reads)` | 1 batch (`(name,seq,qual)` list) | one tool invocation |
|
|
108
|
+
| `map_file(fastq, batch_size, threads, workers)` | whole FASTQ (plain/.gz) | **path prismalign reads + batches itself** (default `batch_size=8192` = 8×bwa-mem2's internal 512); `workers>1` runs batches **concurrently** (each an independent tool subprocess — the real throughput lever, since a tool's own threads saturate on a shared index) |
|
|
109
|
+
|
|
110
|
+
| adapter | tool | index |
|
|
111
|
+
|---|---|---|
|
|
112
|
+
| `BwaMemAdapter` | `bwa mem` | `bwa index` |
|
|
113
|
+
| `BwaMem2Adapter` | `bwa-mem2 mem` | `bwa-mem2 index` |
|
|
114
|
+
| `Bowtie2Adapter` | `bowtie2` | `bowtie2-build` |
|
|
115
|
+
| `Minimap2Adapter` | `minimap2 -a` | none (reads FASTA directly) |
|
|
116
|
+
| `Hisat2Adapter` | `hisat2` | `hisat2-build` |
|
|
117
|
+
| `StrobealignAdapter` | `strobealign` | `.sti` |
|
|
118
|
+
| `SamAdapter` | any SAM mapper | (generic; you supply the command template) |
|
|
119
|
+
|
|
120
|
+
Each adapter needs its binary on PATH (or an env var: `BWA_BIN`, `BWA_MEM2_BIN`,
|
|
121
|
+
`BOWTIE2_BIN`, `MINIMAP2_BIN`, `HISAT2_BIN`, `STROBEALIGN_BIN`).
|
|
122
|
+
|
|
123
|
+
**Throughput tip:** for a large job use ``map_file(path, batch_size=8192,
|
|
124
|
+
workers=cpu/2..cpu, threads=1)`` — batching at 8192 (a multiple of the tool's
|
|
125
|
+
internal 512) amortizes the subprocess startup, and ``workers`` parallelizes
|
|
126
|
+
batches across **separate tool processes** (each with its own memory bandwidth;
|
|
127
|
+
the tool's own ``-t`` saturates, so parallelize via processes). `.gz` input is
|
|
128
|
+
read with an internal fast library (`isal`/`xopen`/`rapidgzip`) when available.
|
|
129
|
+
|
|
130
|
+
### bwa-mem2 appears at BOTH layers (intentional)
|
|
131
|
+
- **`BwaMem2Backend`** (backend) = bwa-mem2 **in-process**, per-read — for embedding / `--backend bwamem2`.
|
|
132
|
+
- **`BwaMem2Adapter`** (adapter) = bwa-mem2 **batched** CLI (`map_file`) — for throughput.
|
|
133
|
+
|
|
134
|
+
For large references, bwa-mem2's ~2-3x (from SIMD FM-index search) shows up; build it
|
|
135
|
+
**cleanly for AVX2/AVX-512** (stale object files cause SIGILL). The in-process backend
|
|
136
|
+
is per-read (batch the calls, or use the adapter, for throughput).
|
|
137
|
+
|
|
138
|
+
Full inventory — including where each wrapper lives — in [`docs/backends.md`](docs/backends.md).
|
|
139
|
+
|
|
140
|
+
## Speed & IO
|
|
141
|
+
|
|
142
|
+
* **Auto backend**: `--backend auto` (explicit) picks the fastest *importable*
|
|
143
|
+
native backend — `minibwa` (~2-3x BWA-MEM, the bwa-mem2 speed tier), else
|
|
144
|
+
`mappy` (in-process), else `bwamem`. (Default is `bwamem`.) The chosen
|
|
145
|
+
backend is printed at startup (`[prismalign] backend auto=minibwa`).
|
|
146
|
+
* **Process parallelism is the lever**: `-t/--threads N` maps reads in an
|
|
147
|
+
ordered fork+COW process pool (any backend); batches are drained in read
|
|
148
|
+
order so the BAM is **byte-identical** to `threads=1`. Processes (each with
|
|
149
|
+
its own index) scale; bwa-mem2's internal `-t` threads do *not* (they share
|
|
150
|
+
one index and are memory-bound). So scale **processes/workers**, not
|
|
151
|
+
bwa-mem2 threads.
|
|
152
|
+
* **Reduced repeated IO**: references are copy+converted **once** even when
|
|
153
|
+
reused across layers (cache keyed by path+scheme); per-hit reference fetch
|
|
154
|
+
is cached in memory for small contigs (RNA/transcript references), so only
|
|
155
|
+
one indexed read per contig.
|
|
156
|
+
|
|
157
|
+
> **Throughput ceiling.** The heavy per-read work (BWA-MEM / minibwa / minimap2
|
|
158
|
+
> / WFA2 kernels) is native C. With BWA-MEM the runtime is essentially that
|
|
159
|
+
> kernel's throughput — prismalign's glue (conversion, re-scoring, tag
|
|
160
|
+
> emission) adds only a small fraction. For the largest runs, use `minibwa`
|
|
161
|
+
> (runs on standard CPython; no free-threaded-3.14 wheel) or `mappy`; both are
|
|
162
|
+
> far faster than BWA-MEM.
|
|
163
|
+
|
|
164
|
+
## Limitations (v0.2.x)
|
|
165
|
+
|
|
166
|
+
* paired-end is supported natively by the `bwamem`, `mappy` and `minibwa`
|
|
167
|
+
backends; `wfa2` is single-end for now (the subprocess `sam`/
|
|
168
|
+
`strobealign`/`bowtie2`/`bwa_mem2` *adapters* are also SE).
|
|
169
|
+
* hierarchical (layered) mapping uses the `PLAIN` identity scheme for
|
|
170
|
+
non-converted short-RNA references. `minibwa` is the fastest native backend
|
|
171
|
+
but needs standard (non-free-threaded) CPython ≤ 3.13, so on 3.14t `mappy`
|
|
172
|
+
is the fastest available.
|
|
@@ -39,10 +39,14 @@ from .em import em_allocate, EMResult, nm_from_md, read_nm
|
|
|
39
39
|
from .engine import NColorMapper, score_to_mapq
|
|
40
40
|
from . import adapters
|
|
41
41
|
from .adapters import (
|
|
42
|
+
CliAdapter,
|
|
42
43
|
SamAdapter,
|
|
43
44
|
StrobealignAdapter,
|
|
44
45
|
Bowtie2Adapter,
|
|
46
|
+
BwaMemAdapter,
|
|
45
47
|
BwaMem2Adapter,
|
|
48
|
+
Minimap2Adapter,
|
|
49
|
+
Hisat2Adapter,
|
|
46
50
|
)
|
|
47
51
|
from .backends import (
|
|
48
52
|
RawHit,
|
|
@@ -52,7 +56,14 @@ from .backends import (
|
|
|
52
56
|
MinibwaBackend,
|
|
53
57
|
Wfa2Backend,
|
|
54
58
|
get_backend,
|
|
59
|
+
recommended_backend,
|
|
60
|
+
backend_available,
|
|
55
61
|
)
|
|
62
|
+
|
|
63
|
+
try:
|
|
64
|
+
from .backends import BwaMem2Backend
|
|
65
|
+
except ImportError:
|
|
66
|
+
BwaMem2Backend = None
|
|
56
67
|
from .schemes import MK, KM, BS, SLAM, A2G, THREE, PLAIN, ColorScheme, ColorChannel
|
|
57
68
|
|
|
58
69
|
# Single source of truth is pyproject.toml; never hardcode here.
|
|
@@ -83,16 +94,23 @@ __all__ = [
|
|
|
83
94
|
"read_nm",
|
|
84
95
|
# native backends
|
|
85
96
|
"get_backend",
|
|
97
|
+
"recommended_backend",
|
|
98
|
+
"backend_available",
|
|
86
99
|
"BwaMemBackend",
|
|
87
100
|
"MappyBackend",
|
|
88
101
|
"MinibwaBackend",
|
|
89
102
|
"Wfa2Backend",
|
|
103
|
+
"BwaMem2Backend",
|
|
90
104
|
# external-tool adapters (not selectable via --backend)
|
|
91
105
|
"adapters",
|
|
106
|
+
"CliAdapter",
|
|
92
107
|
"SamAdapter",
|
|
93
108
|
"StrobealignAdapter",
|
|
94
109
|
"Bowtie2Adapter",
|
|
110
|
+
"BwaMemAdapter",
|
|
95
111
|
"BwaMem2Adapter",
|
|
112
|
+
"Minimap2Adapter",
|
|
113
|
+
"Hisat2Adapter",
|
|
96
114
|
# schemes
|
|
97
115
|
"MK",
|
|
98
116
|
"KM",
|
|
@@ -122,10 +140,15 @@ def available_schemes() -> list:
|
|
|
122
140
|
|
|
123
141
|
|
|
124
142
|
def available_backends() -> list:
|
|
125
|
-
"""Names of
|
|
126
|
-
|
|
143
|
+
"""Names of native backends actually usable on this interpreter.
|
|
144
|
+
|
|
145
|
+
Unlike ``_BACKENDS`` (every registered backend), this probes each backend's
|
|
146
|
+
runtime dependency, so optional ones with no installable wheel here (e.g.
|
|
147
|
+
``minibwa`` on free-threaded CPython 3.14t) are omitted.
|
|
148
|
+
"""
|
|
149
|
+
from .backends import _BACKENDS, backend_available
|
|
127
150
|
|
|
128
|
-
return sorted(_BACKENDS)
|
|
151
|
+
return sorted(b for b in _BACKENDS if b != "auto" and backend_available(b))
|
|
129
152
|
|
|
130
153
|
|
|
131
154
|
def map_reads(
|
|
@@ -14,16 +14,25 @@ Every adapter still conforms to the ``Backend`` interface, so it can be passed
|
|
|
14
14
|
to an ``NColorMapper`` via ``backend_kwargs``/a custom ``get_backend`` override.
|
|
15
15
|
"""
|
|
16
16
|
|
|
17
|
-
from .
|
|
17
|
+
from .base import CliAdapter, _cigar_refspan, _line_to_rawhit, _sam_to_rawhits
|
|
18
|
+
from .sam import SamAdapter
|
|
18
19
|
from .strobealign import StrobealignAdapter
|
|
19
20
|
from .bowtie2 import Bowtie2Adapter
|
|
21
|
+
from .bwa import BwaMemAdapter
|
|
20
22
|
from .bwa_mem2 import BwaMem2Adapter
|
|
23
|
+
from .minimap2 import Minimap2Adapter
|
|
24
|
+
from .hisat2 import Hisat2Adapter
|
|
21
25
|
|
|
22
26
|
__all__ = [
|
|
27
|
+
"CliAdapter",
|
|
23
28
|
"SamAdapter",
|
|
24
29
|
"StrobealignAdapter",
|
|
25
30
|
"Bowtie2Adapter",
|
|
31
|
+
"BwaMemAdapter",
|
|
26
32
|
"BwaMem2Adapter",
|
|
33
|
+
"Minimap2Adapter",
|
|
34
|
+
"Hisat2Adapter",
|
|
27
35
|
"_cigar_refspan",
|
|
36
|
+
"_line_to_rawhit",
|
|
28
37
|
"_sam_to_rawhits",
|
|
29
38
|
]
|