rfmix-reader 0.2.0__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rfmix_reader-0.2.1/PKG-INFO +259 -0
- rfmix_reader-0.2.1/README.md +215 -0
- {rfmix_reader-0.2.0 → rfmix_reader-0.2.1}/pyproject.toml +38 -17
- rfmix_reader-0.2.1/rfmix_reader/__init__.py +92 -0
- {rfmix_reader-0.2.0 → rfmix_reader-0.2.1}/rfmix_reader/_fb_read.py +2 -3
- {rfmix_reader-0.2.0 → rfmix_reader-0.2.1}/rfmix_reader/_imputation.py +27 -16
- {rfmix_reader-0.2.0 → rfmix_reader-0.2.1}/rfmix_reader/_loci_bed.py +6 -6
- rfmix_reader-0.2.1/rfmix_reader/_read_flare.py +438 -0
- {rfmix_reader-0.2.0 → rfmix_reader-0.2.1}/rfmix_reader/_read_rfmix.py +72 -124
- rfmix_reader-0.2.1/rfmix_reader/_read_simu.py +420 -0
- {rfmix_reader-0.2.0 → rfmix_reader-0.2.1}/rfmix_reader/_testing.py +5 -5
- {rfmix_reader-0.2.0 → rfmix_reader-0.2.1}/rfmix_reader/_utils.py +112 -67
- {rfmix_reader-0.2.0 → rfmix_reader-0.2.1}/rfmix_reader/_visualization.py +17 -17
- {rfmix_reader-0.2.0 → rfmix_reader-0.2.1}/rfmix_reader/_write_data.py +15 -15
- rfmix_reader-0.2.0/PKG-INFO +0 -278
- rfmix_reader-0.2.0/README.md +0 -247
- rfmix_reader-0.2.0/rfmix_reader/__init__.py +0 -53
- {rfmix_reader-0.2.0 → rfmix_reader-0.2.1}/LICENSE +0 -0
- {rfmix_reader-0.2.0 → rfmix_reader-0.2.1}/rfmix_reader/.local.py +0 -0
- {rfmix_reader-0.2.0 → rfmix_reader-0.2.1}/rfmix_reader/_chunk.py +0 -0
- {rfmix_reader-0.2.0 → rfmix_reader-0.2.1}/rfmix_reader/_cli.py +0 -0
- {rfmix_reader-0.2.0 → rfmix_reader-0.2.1}/rfmix_reader/_constants.py +0 -0
- {rfmix_reader-0.2.0 → rfmix_reader-0.2.1}/rfmix_reader/_errorhandling.py +0 -0
- {rfmix_reader-0.2.0 → rfmix_reader-0.2.1}/rfmix_reader/_tagore.py +0 -0
- {rfmix_reader-0.2.0 → rfmix_reader-0.2.1}/rfmix_reader/base.svg.p +0 -0
|
@@ -0,0 +1,259 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: rfmix-reader
|
|
3
|
+
Version: 0.2.1
|
|
4
|
+
Summary: RFMix-reader is a Python package designed to efficiently read and process output files generated by RFMix, a popular tool for estimating local ancestry in admixed populations. The package employs a lazy loading approach, which minimizes memory consumption by reading only the loci that are accessed by the user, rather than loading the entire dataset into memory at once.
|
|
5
|
+
License: GPL-3.0-or-later
|
|
6
|
+
License-File: LICENSE
|
|
7
|
+
Keywords: file parser,rfmix,gpu acceleration,local ancestry
|
|
8
|
+
Author: Kynon J.M. Benjamin
|
|
9
|
+
Author-email: kj.benjamin90@gmail.com
|
|
10
|
+
Maintainer: Kynon J.M. Benjamin
|
|
11
|
+
Maintainer-email: kj.benjamin90@gmail.com
|
|
12
|
+
Requires-Python: >=3.11,<3.15
|
|
13
|
+
Classifier: License :: OSI Approved :: GNU General Public License v3 or later (GPLv3+)
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
20
|
+
Provides-Extra: docs
|
|
21
|
+
Provides-Extra: gpu
|
|
22
|
+
Provides-Extra: io
|
|
23
|
+
Provides-Extra: tests
|
|
24
|
+
Provides-Extra: viz
|
|
25
|
+
Requires-Dist: cairosvg (>=2.7,<3.0) ; extra == "viz"
|
|
26
|
+
Requires-Dist: cyvcf2 (>=0.31)
|
|
27
|
+
Requires-Dist: dask (>=2025.1,<2026.0)
|
|
28
|
+
Requires-Dist: matplotlib (>=3.10,<4.0) ; extra == "viz"
|
|
29
|
+
Requires-Dist: numpy (>=1.23,<3)
|
|
30
|
+
Requires-Dist: pandas (>=2.0)
|
|
31
|
+
Requires-Dist: psutil (>=6,<7)
|
|
32
|
+
Requires-Dist: seaborn (>=0.13,<0.14) ; extra == "viz"
|
|
33
|
+
Requires-Dist: sphinx (>=7,<9) ; extra == "docs"
|
|
34
|
+
Requires-Dist: sphinx-autodoc-typehints (>=2,<3) ; extra == "docs"
|
|
35
|
+
Requires-Dist: sphinx-copybutton (>=0.5,<0.6) ; extra == "docs"
|
|
36
|
+
Requires-Dist: sphinx-rtd-theme (>=2,<3) ; extra == "docs"
|
|
37
|
+
Requires-Dist: torch (>=2.8) ; extra == "gpu"
|
|
38
|
+
Requires-Dist: tqdm (>=4.66)
|
|
39
|
+
Project-URL: Bug Tracker, https://github.com/heart-gen/rfmix_reader/issues
|
|
40
|
+
Project-URL: homepage, https://rfmix-reader.readthedocs.io/en/latest/
|
|
41
|
+
Project-URL: repository, https://github.com/heart-gen/rfmix_reader.git
|
|
42
|
+
Description-Content-Type: text/markdown
|
|
43
|
+
|
|
44
|
+
# RFMix-reader
|
|
45
|
+
`RFMix-reader` is a Python package for efficiently reading and processing output
|
|
46
|
+
files generated by [`RFMix`](https://github.com/slowkoni/rfmix), a widely used tool
|
|
47
|
+
for estimating local ancestry in admixed populations.
|
|
48
|
+
It employs a **lazy loading approach** to minimize memory usage, and leverages **GPU acceleration**
|
|
49
|
+
for major speedups when available.
|
|
50
|
+
|
|
51
|
+
---
|
|
52
|
+
|
|
53
|
+
## Installation
|
|
54
|
+
|
|
55
|
+
`rfmix-reader` requires **Python 3.10+**. Install from PyPI:
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
pip install rfmix-reader
|
|
59
|
+
````
|
|
60
|
+
|
|
61
|
+
### Installation Options
|
|
62
|
+
|
|
63
|
+
* **Basic install** (CPU only):
|
|
64
|
+
|
|
65
|
+
```bash
|
|
66
|
+
pip install rfmix-reader
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
* **With GPU acceleration** (`cupy`, `cudf`, `dask-cudf`):
|
|
70
|
+
|
|
71
|
+
```bash
|
|
72
|
+
pip install rfmix-reader[gpu]
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
* **With documentation tools** (`sphinx`, `sphinx-rtd-theme`):
|
|
76
|
+
|
|
77
|
+
```bash
|
|
78
|
+
pip install rfmix-reader[docs]
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
* **With testing tools** (`pytest`):
|
|
82
|
+
|
|
83
|
+
```bash
|
|
84
|
+
pip install rfmix-reader[tests]
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
### GPU Notes
|
|
88
|
+
|
|
89
|
+
* `torch` is installed automatically.
|
|
90
|
+
* For CUDA builds, install a matching GPU-enabled wheel for your system following the [PyTorch guide](https://pytorch.org/get-started/locally/).
|
|
91
|
+
* RAPIDS (`cudf`, `cupy`) wheels are version- and CUDA-specific. See the [RAPIDS install guide](https://docs.rapids.ai/install).
|
|
92
|
+
* CPU-only installations will still run efficiently, just without GPU acceleration.
|
|
93
|
+
|
|
94
|
+
---
|
|
95
|
+
|
|
96
|
+
## Quickstart
|
|
97
|
+
|
|
98
|
+
```python
|
|
99
|
+
from rfmix_reader import read_rfmix
|
|
100
|
+
|
|
101
|
+
# Load RFMix outputs (two-population admixture example)
|
|
102
|
+
file_path = "examples/two_populations/out/"
|
|
103
|
+
loci_df, g_anc, local_array = read_rfmix(file_path)
|
|
104
|
+
|
|
105
|
+
print(loci_df.head())
|
|
106
|
+
print(g_anc.head())
|
|
107
|
+
print(local_array.shape)
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
---
|
|
111
|
+
|
|
112
|
+
## Key Features
|
|
113
|
+
|
|
114
|
+
* **Lazy Loading**: Reads data on-the-fly, reducing memory footprint.
|
|
115
|
+
* **Efficient Access**: Query specific loci or regions of interest.
|
|
116
|
+
* **Seamless Integration**: Works smoothly with `pandas`, `dask`, and other analysis tools.
|
|
117
|
+
* **Loci Imputation**: Impute local ancestry loci to dense genotype variant sites.
|
|
118
|
+
* **GPU Acceleration**: Automatic CUDA acceleration via PyTorch/CuPy when available.
|
|
119
|
+
|
|
120
|
+
---
|
|
121
|
+
|
|
122
|
+
## Simulation Data
|
|
123
|
+
|
|
124
|
+
Test datasets for two- and three-population admixture are available on Synapse:
|
|
125
|
+
[Synapse Project syn61691659](https://www.synapse.org/Synapse:syn61691659).
|
|
126
|
+
|
|
127
|
+
---
|
|
128
|
+
|
|
129
|
+
## Usage
|
|
130
|
+
|
|
131
|
+
### Binary Conversion
|
|
132
|
+
|
|
133
|
+
RFMix does not generate binary files directly.
|
|
134
|
+
Use `create_binaries` to generate them (also available as a CLI):
|
|
135
|
+
|
|
136
|
+
```bash
|
|
137
|
+
create-binaries two_pops/out/
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
```python
|
|
141
|
+
from rfmix_reader import create_binaries
|
|
142
|
+
|
|
143
|
+
create_binaries("two_pops/out/", binary_dir="./binary_files")
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
### Main Function
|
|
147
|
+
|
|
148
|
+
Once binaries are available, process RFMix results:
|
|
149
|
+
|
|
150
|
+
```python
|
|
151
|
+
from rfmix_reader import read_rfmix
|
|
152
|
+
|
|
153
|
+
loci, g_anc, admix = read_rfmix("two_pops/out/")
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
### Three Population Example
|
|
157
|
+
|
|
158
|
+
Binaries can also be generated on-the-fly within `read_rfmix` with
|
|
159
|
+
`generate_binary` set to `True`.
|
|
160
|
+
|
|
161
|
+
```python
|
|
162
|
+
loci, g_anc, admix = read_rfmix("examples/three_populations/out/",
|
|
163
|
+
binary_dir="./binary_files",
|
|
164
|
+
generate_binary=True)
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
### Loci Imputation
|
|
168
|
+
|
|
169
|
+
Impute local ancestry loci to variant positions for integration with genotype data:
|
|
170
|
+
|
|
171
|
+
```python
|
|
172
|
+
from rfmix_reader import interpolate_array
|
|
173
|
+
import pandas as pd
|
|
174
|
+
import dask.array as da
|
|
175
|
+
|
|
176
|
+
variant_loci_df = pd.DataFrame({
|
|
177
|
+
"chrom": ["1", "1", "1", "1"],
|
|
178
|
+
"pos": [100, 200, 300, 400],
|
|
179
|
+
"i": [1, None, None, 2]
|
|
180
|
+
})
|
|
181
|
+
admix = da.random.random((2, 3)) # mock admixture data
|
|
182
|
+
|
|
183
|
+
z = interpolate_array(variant_loci_df, admix, "/path/to/output")
|
|
184
|
+
print(z.shape)
|
|
185
|
+
```
|
|
186
|
+
|
|
187
|
+
### Reading Haptools simulations
|
|
188
|
+
|
|
189
|
+
Use `read_simu` to load BGZF-compressed VCF files created by
|
|
190
|
+
`haptools simgenotype --pop_field`:
|
|
191
|
+
|
|
192
|
+
```python
|
|
193
|
+
from rfmix_reader import read_simu
|
|
194
|
+
|
|
195
|
+
loci_df, g_anc, admix = read_simu("/path/to/simulations/")
|
|
196
|
+
```
|
|
197
|
+
|
|
198
|
+
Haptools does **not** include the chromosome length in the `##contig`
|
|
199
|
+
header lines, but `read_simu` requires that metadata to index each VCF.
|
|
200
|
+
Copy the `contigs.txt` file Haptools generates from the FASTA you used
|
|
201
|
+
for simulation and reheader every file with the appropriate contig entry
|
|
202
|
+
before calling `read_simu`. The following snippet shows one approach
|
|
203
|
+
using `bcftools` and `tabix`:
|
|
204
|
+
|
|
205
|
+
```bash
|
|
206
|
+
CONTIGS="../../three_populations/_m/contigs.txt"
|
|
207
|
+
VCFDIR="gt-files"
|
|
208
|
+
CHR="chr${SLURM_ARRAY_TASK_ID}"
|
|
209
|
+
OUT="${VCFDIR}/${CHR}.vcf.gz"
|
|
210
|
+
IN="${VCFDIR}/back/${CHR}.vcf.gz"
|
|
211
|
+
|
|
212
|
+
CONTIG_LINE=$(grep -w "ID=${CHR}" "$CONTIGS")
|
|
213
|
+
if [[ -z "$CONTIG_LINE" ]]; then
|
|
214
|
+
echo "ERROR: No contig line found for ${CHR} in $CONTIGS"
|
|
215
|
+
exit 1
|
|
216
|
+
fi
|
|
217
|
+
|
|
218
|
+
bcftools view -h "$IN" \
|
|
219
|
+
| sed "s/^##contig=<ID=${CHR}>.*/${CONTIG_LINE}/" > header.${CHR}.tmp
|
|
220
|
+
bcftools reheader -h header.${CHR}.tmp -o "$OUT" "$IN"
|
|
221
|
+
tabix -p vcf "$OUT"
|
|
222
|
+
```
|
|
223
|
+
|
|
224
|
+
---
|
|
225
|
+
|
|
226
|
+
## Development Install
|
|
227
|
+
|
|
228
|
+
For contributors:
|
|
229
|
+
|
|
230
|
+
```bash
|
|
231
|
+
git clone https://github.com/heart-gen/rfmix_reader.git
|
|
232
|
+
cd rfmix_reader
|
|
233
|
+
pip install -e ".[gpu,docs,tests]"
|
|
234
|
+
```
|
|
235
|
+
|
|
236
|
+
---
|
|
237
|
+
|
|
238
|
+
## Citation
|
|
239
|
+
|
|
240
|
+
If you use this software, please cite:
|
|
241
|
+
|
|
242
|
+
[](https://zenodo.org/doi/10.5281/zenodo.12629787)
|
|
243
|
+
|
|
244
|
+
Benjamin, K. J. M. (2024). **RFMix-reader (Version 0.2.0)** \[Computer software].
|
|
245
|
+
[https://github.com/heart-gen/rfmix\_reader](https://github.com/heart-gen/rfmix_reader)
|
|
246
|
+
|
|
247
|
+
Kynon JM Benjamin. *"RFMix-reader: Accelerated reading and processing for local ancestry studies."*
|
|
248
|
+
**bioRxiv** (2024).
|
|
249
|
+
DOI: [10.1101/2024.07.13.603370](https://www.biorxiv.org/content/10.1101/2024.07.13.603370v2).
|
|
250
|
+
|
|
251
|
+
---
|
|
252
|
+
|
|
253
|
+
## Funding
|
|
254
|
+
|
|
255
|
+
This work was supported by the National Institutes of Health,
|
|
256
|
+
National Institute on Minority Health and Health Disparities (NIMHD)
|
|
257
|
+
K99MD016964 / R00MD016964.
|
|
258
|
+
|
|
259
|
+
|
|
@@ -0,0 +1,215 @@
|
|
|
1
|
+
# RFMix-reader
|
|
2
|
+
`RFMix-reader` is a Python package for efficiently reading and processing output
|
|
3
|
+
files generated by [`RFMix`](https://github.com/slowkoni/rfmix), a widely used tool
|
|
4
|
+
for estimating local ancestry in admixed populations.
|
|
5
|
+
It employs a **lazy loading approach** to minimize memory usage, and leverages **GPU acceleration**
|
|
6
|
+
for major speedups when available.
|
|
7
|
+
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
## Installation
|
|
11
|
+
|
|
12
|
+
`rfmix-reader` requires **Python 3.10+**. Install from PyPI:
|
|
13
|
+
|
|
14
|
+
```bash
|
|
15
|
+
pip install rfmix-reader
|
|
16
|
+
````
|
|
17
|
+
|
|
18
|
+
### Installation Options
|
|
19
|
+
|
|
20
|
+
* **Basic install** (CPU only):
|
|
21
|
+
|
|
22
|
+
```bash
|
|
23
|
+
pip install rfmix-reader
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
* **With GPU acceleration** (`cupy`, `cudf`, `dask-cudf`):
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
pip install rfmix-reader[gpu]
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
* **With documentation tools** (`sphinx`, `sphinx-rtd-theme`):
|
|
33
|
+
|
|
34
|
+
```bash
|
|
35
|
+
pip install rfmix-reader[docs]
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
* **With testing tools** (`pytest`):
|
|
39
|
+
|
|
40
|
+
```bash
|
|
41
|
+
pip install rfmix-reader[tests]
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
### GPU Notes
|
|
45
|
+
|
|
46
|
+
* `torch` is installed automatically.
|
|
47
|
+
* For CUDA builds, install a matching GPU-enabled wheel for your system following the [PyTorch guide](https://pytorch.org/get-started/locally/).
|
|
48
|
+
* RAPIDS (`cudf`, `cupy`) wheels are version- and CUDA-specific. See the [RAPIDS install guide](https://docs.rapids.ai/install).
|
|
49
|
+
* CPU-only installations will still run efficiently, just without GPU acceleration.
|
|
50
|
+
|
|
51
|
+
---
|
|
52
|
+
|
|
53
|
+
## Quickstart
|
|
54
|
+
|
|
55
|
+
```python
|
|
56
|
+
from rfmix_reader import read_rfmix
|
|
57
|
+
|
|
58
|
+
# Load RFMix outputs (two-population admixture example)
|
|
59
|
+
file_path = "examples/two_populations/out/"
|
|
60
|
+
loci_df, g_anc, local_array = read_rfmix(file_path)
|
|
61
|
+
|
|
62
|
+
print(loci_df.head())
|
|
63
|
+
print(g_anc.head())
|
|
64
|
+
print(local_array.shape)
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
---
|
|
68
|
+
|
|
69
|
+
## Key Features
|
|
70
|
+
|
|
71
|
+
* **Lazy Loading**: Reads data on-the-fly, reducing memory footprint.
|
|
72
|
+
* **Efficient Access**: Query specific loci or regions of interest.
|
|
73
|
+
* **Seamless Integration**: Works smoothly with `pandas`, `dask`, and other analysis tools.
|
|
74
|
+
* **Loci Imputation**: Impute local ancestry loci to dense genotype variant sites.
|
|
75
|
+
* **GPU Acceleration**: Automatic CUDA acceleration via PyTorch/CuPy when available.
|
|
76
|
+
|
|
77
|
+
---
|
|
78
|
+
|
|
79
|
+
## Simulation Data
|
|
80
|
+
|
|
81
|
+
Test datasets for two- and three-population admixture are available on Synapse:
|
|
82
|
+
[Synapse Project syn61691659](https://www.synapse.org/Synapse:syn61691659).
|
|
83
|
+
|
|
84
|
+
---
|
|
85
|
+
|
|
86
|
+
## Usage
|
|
87
|
+
|
|
88
|
+
### Binary Conversion
|
|
89
|
+
|
|
90
|
+
RFMix does not generate binary files directly.
|
|
91
|
+
Use `create_binaries` to generate them (also available as a CLI):
|
|
92
|
+
|
|
93
|
+
```bash
|
|
94
|
+
create-binaries two_pops/out/
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
```python
|
|
98
|
+
from rfmix_reader import create_binaries
|
|
99
|
+
|
|
100
|
+
create_binaries("two_pops/out/", binary_dir="./binary_files")
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
### Main Function
|
|
104
|
+
|
|
105
|
+
Once binaries are available, process RFMix results:
|
|
106
|
+
|
|
107
|
+
```python
|
|
108
|
+
from rfmix_reader import read_rfmix
|
|
109
|
+
|
|
110
|
+
loci, g_anc, admix = read_rfmix("two_pops/out/")
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
### Three Population Example
|
|
114
|
+
|
|
115
|
+
Binaries can also be generated on-the-fly within `read_rfmix` with
|
|
116
|
+
`generate_binary` set to `True`.
|
|
117
|
+
|
|
118
|
+
```python
|
|
119
|
+
loci, g_anc, admix = read_rfmix("examples/three_populations/out/",
|
|
120
|
+
binary_dir="./binary_files",
|
|
121
|
+
generate_binary=True)
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
### Loci Imputation
|
|
125
|
+
|
|
126
|
+
Impute local ancestry loci to variant positions for integration with genotype data:
|
|
127
|
+
|
|
128
|
+
```python
|
|
129
|
+
from rfmix_reader import interpolate_array
|
|
130
|
+
import pandas as pd
|
|
131
|
+
import dask.array as da
|
|
132
|
+
|
|
133
|
+
variant_loci_df = pd.DataFrame({
|
|
134
|
+
"chrom": ["1", "1", "1", "1"],
|
|
135
|
+
"pos": [100, 200, 300, 400],
|
|
136
|
+
"i": [1, None, None, 2]
|
|
137
|
+
})
|
|
138
|
+
admix = da.random.random((2, 3)) # mock admixture data
|
|
139
|
+
|
|
140
|
+
z = interpolate_array(variant_loci_df, admix, "/path/to/output")
|
|
141
|
+
print(z.shape)
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
### Reading Haptools simulations
|
|
145
|
+
|
|
146
|
+
Use `read_simu` to load BGZF-compressed VCF files created by
|
|
147
|
+
`haptools simgenotype --pop_field`:
|
|
148
|
+
|
|
149
|
+
```python
|
|
150
|
+
from rfmix_reader import read_simu
|
|
151
|
+
|
|
152
|
+
loci_df, g_anc, admix = read_simu("/path/to/simulations/")
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
Haptools does **not** include the chromosome length in the `##contig`
|
|
156
|
+
header lines, but `read_simu` requires that metadata to index each VCF.
|
|
157
|
+
Copy the `contigs.txt` file Haptools generates from the FASTA you used
|
|
158
|
+
for simulation and reheader every file with the appropriate contig entry
|
|
159
|
+
before calling `read_simu`. The following snippet shows one approach
|
|
160
|
+
using `bcftools` and `tabix`:
|
|
161
|
+
|
|
162
|
+
```bash
|
|
163
|
+
CONTIGS="../../three_populations/_m/contigs.txt"
|
|
164
|
+
VCFDIR="gt-files"
|
|
165
|
+
CHR="chr${SLURM_ARRAY_TASK_ID}"
|
|
166
|
+
OUT="${VCFDIR}/${CHR}.vcf.gz"
|
|
167
|
+
IN="${VCFDIR}/back/${CHR}.vcf.gz"
|
|
168
|
+
|
|
169
|
+
CONTIG_LINE=$(grep -w "ID=${CHR}" "$CONTIGS")
|
|
170
|
+
if [[ -z "$CONTIG_LINE" ]]; then
|
|
171
|
+
echo "ERROR: No contig line found for ${CHR} in $CONTIGS"
|
|
172
|
+
exit 1
|
|
173
|
+
fi
|
|
174
|
+
|
|
175
|
+
bcftools view -h "$IN" \
|
|
176
|
+
| sed "s/^##contig=<ID=${CHR}>.*/${CONTIG_LINE}/" > header.${CHR}.tmp
|
|
177
|
+
bcftools reheader -h header.${CHR}.tmp -o "$OUT" "$IN"
|
|
178
|
+
tabix -p vcf "$OUT"
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
---
|
|
182
|
+
|
|
183
|
+
## Development Install
|
|
184
|
+
|
|
185
|
+
For contributors:
|
|
186
|
+
|
|
187
|
+
```bash
|
|
188
|
+
git clone https://github.com/heart-gen/rfmix_reader.git
|
|
189
|
+
cd rfmix_reader
|
|
190
|
+
pip install -e ".[gpu,docs,tests]"
|
|
191
|
+
```
|
|
192
|
+
|
|
193
|
+
---
|
|
194
|
+
|
|
195
|
+
## Citation
|
|
196
|
+
|
|
197
|
+
If you use this software, please cite:
|
|
198
|
+
|
|
199
|
+
[](https://zenodo.org/doi/10.5281/zenodo.12629787)
|
|
200
|
+
|
|
201
|
+
Benjamin, K. J. M. (2024). **RFMix-reader (Version 0.2.0)** \[Computer software].
|
|
202
|
+
[https://github.com/heart-gen/rfmix\_reader](https://github.com/heart-gen/rfmix_reader)
|
|
203
|
+
|
|
204
|
+
Kynon JM Benjamin. *"RFMix-reader: Accelerated reading and processing for local ancestry studies."*
|
|
205
|
+
**bioRxiv** (2024).
|
|
206
|
+
DOI: [10.1101/2024.07.13.603370](https://www.biorxiv.org/content/10.1101/2024.07.13.603370v2).
|
|
207
|
+
|
|
208
|
+
---
|
|
209
|
+
|
|
210
|
+
## Funding
|
|
211
|
+
|
|
212
|
+
This work was supported by the National Institutes of Health,
|
|
213
|
+
National Institute on Minority Health and Health Disparities (NIMHD)
|
|
214
|
+
K99MD016964 / R00MD016964.
|
|
215
|
+
|
|
@@ -1,14 +1,12 @@
|
|
|
1
1
|
[tool.poetry]
|
|
2
2
|
name = "rfmix-reader"
|
|
3
|
-
version = "0.2.
|
|
3
|
+
version = "0.2.1" # This will be overridden if dynamic versioning is enabled
|
|
4
4
|
description = "RFMix-reader is a Python package designed to efficiently read and process output files generated by RFMix, a popular tool for estimating local ancestry in admixed populations. The package employs a lazy loading approach, which minimizes memory consumption by reading only the loci that are accessed by the user, rather than loading the entire dataset into memory at once."
|
|
5
5
|
authors = ["Kynon J.M. Benjamin <kj.benjamin90@gmail.com>"]
|
|
6
6
|
maintainers = ["Kynon J.M. Benjamin <kj.benjamin90@gmail.com>"]
|
|
7
7
|
readme = "README.md"
|
|
8
8
|
license = "GPL-3.0-or-later"
|
|
9
9
|
keywords = ["file parser", "rfmix", "gpu acceleration", "local ancestry"]
|
|
10
|
-
homepage = "https://rfmix-reader.readthedocs.io/en/latest/"
|
|
11
|
-
repository = "https://github.com/heart-gen/rfmix_reader.git"
|
|
12
10
|
packages = [{ include = "rfmix_reader" }]
|
|
13
11
|
classifiers = [
|
|
14
12
|
"Programming Language :: Python :: 3",
|
|
@@ -17,28 +15,49 @@ classifiers = [
|
|
|
17
15
|
]
|
|
18
16
|
|
|
19
17
|
[tool.poetry.dependencies]
|
|
20
|
-
python = ">=3.
|
|
18
|
+
python = ">=3.11,<3.15"
|
|
19
|
+
numpy = ">=1.23,<3"
|
|
21
20
|
pandas = ">=2.0"
|
|
22
|
-
dask = ">=2025.1"
|
|
23
21
|
tqdm = ">=4.66"
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
22
|
+
psutil = "^6"
|
|
23
|
+
cyvcf2 = ">=0.31"
|
|
24
|
+
dask = "^2025.1"
|
|
25
|
+
# Plotting
|
|
26
|
+
matplotlib = { version = "^3.10", optional = true }
|
|
27
|
+
seaborn = { version = "^0.13", optional = true }
|
|
28
|
+
cairosvg = { version = "^2.7", optional = true }
|
|
29
|
+
# GPU
|
|
30
|
+
torch = { version = ">=2.8", optional = true }
|
|
31
|
+
# Docs
|
|
32
|
+
sphinx = { version = ">=7,<9", optional = true }
|
|
33
|
+
sphinx-rtd-theme = { version = "^2", optional = true }
|
|
34
|
+
sphinx-autodoc-typehints = { version = "^2", optional = true }
|
|
35
|
+
sphinx-copybutton = { version = "^0.5", optional = true }
|
|
36
|
+
|
|
37
|
+
[tool.poetry.extras]
|
|
38
|
+
viz = ["matplotlib", "seaborn", "cairosvg"]
|
|
39
|
+
gpu = ["cupy", "cudf", "dask-cudf", "torch"]
|
|
40
|
+
io = ["pyarrow"] # parquet export
|
|
41
|
+
docs = ["sphinx", "sphinx-rtd-theme", "sphinx-autodoc-typehints", "sphinx-copybutton"]
|
|
42
|
+
tests = ["pytest"]
|
|
43
|
+
|
|
44
|
+
[tool.poetry.urls]
|
|
45
|
+
homepage = "https://rfmix-reader.readthedocs.io/en/latest/"
|
|
46
|
+
repository = "https://github.com/heart-gen/rfmix_reader.git"
|
|
47
|
+
"Bug Tracker" = "https://github.com/heart-gen/rfmix_reader/issues"
|
|
29
48
|
|
|
30
49
|
[tool.poetry.scripts]
|
|
31
50
|
create-binaries = "rfmix_reader._cli:main"
|
|
32
51
|
|
|
33
52
|
[tool.poetry.group.test.dependencies]
|
|
34
|
-
pytest = "^8
|
|
53
|
+
pytest = "^8"
|
|
54
|
+
pytest-cov = "^5"
|
|
35
55
|
|
|
36
56
|
[tool.poetry.group.docs.dependencies]
|
|
37
|
-
sphinx = "
|
|
38
|
-
sphinx-rtd-theme = "^2
|
|
39
|
-
sphinx-autodoc-typehints = "^2
|
|
40
|
-
|
|
41
|
-
torchvision = {version = "^0.21+cpu", source = "pytorch_cpu"}
|
|
57
|
+
sphinx = ">=7,<9"
|
|
58
|
+
sphinx-rtd-theme = "^2"
|
|
59
|
+
sphinx-autodoc-typehints = "^2"
|
|
60
|
+
sphinx-copybutton = "^0.5"
|
|
42
61
|
|
|
43
62
|
[[tool.poetry.source]]
|
|
44
63
|
name = "pytorch_cpu"
|
|
@@ -47,7 +66,9 @@ priority = "explicit"
|
|
|
47
66
|
|
|
48
67
|
[tool.poetry-dynamic-versioning]
|
|
49
68
|
enable = false
|
|
50
|
-
|
|
69
|
+
|
|
70
|
+
[tool.poetry-dynamic-versioning.metadata]
|
|
71
|
+
file = "rfmix_reader/_version.py"
|
|
51
72
|
|
|
52
73
|
[build-system]
|
|
53
74
|
requires = ["poetry-core>=2.0.0", "poetry-dynamic-versioning"]
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from importlib.metadata import version as _v, PackageNotFoundError
|
|
4
|
+
from typing import TYPE_CHECKING
|
|
5
|
+
|
|
6
|
+
try:
|
|
7
|
+
__version__ = _v("rfmix-reader") # distribution name
|
|
8
|
+
except PackageNotFoundError:
|
|
9
|
+
try:
|
|
10
|
+
from ._version import __version__ # fallback for local builds
|
|
11
|
+
except Exception:
|
|
12
|
+
__version__ = "0.0.0"
|
|
13
|
+
|
|
14
|
+
# Public API
|
|
15
|
+
__all__ = [
|
|
16
|
+
"Chunk",
|
|
17
|
+
"read_fb", "read_simu", "read_rfmix", "read_flare",
|
|
18
|
+
"write_data",
|
|
19
|
+
"admix_to_bed_individual",
|
|
20
|
+
"CHROM_SIZES", "COORDINATES",
|
|
21
|
+
"BinaryFileNotFoundError",
|
|
22
|
+
"interpolate_array",
|
|
23
|
+
"get_pops", "get_prefixes", "create_binaries", "get_sample_names",
|
|
24
|
+
"set_gpu_environment", "delete_files_or_directories",
|
|
25
|
+
"save_multi_format", "generate_tagore_bed",
|
|
26
|
+
"plot_global_ancestry", "plot_ancestry_by_chromosome",
|
|
27
|
+
"plot_local_ancestry_tagore",
|
|
28
|
+
]
|
|
29
|
+
|
|
30
|
+
# Map public names for lazy loading
|
|
31
|
+
_lazy = {
|
|
32
|
+
"Chunk": ("._chunk", "Chunk"),
|
|
33
|
+
"read_fb": ("._fb_read", "read_fb"),
|
|
34
|
+
"read_simu": ("._read_simu", "read_simu"),
|
|
35
|
+
"read_rfmix": ("._read_rfmix", "read_rfmix"),
|
|
36
|
+
"read_flare": ("._read_flare", "read_flare"),
|
|
37
|
+
"write_data": ("._write_data", "write_data"),
|
|
38
|
+
"admix_to_bed_individual": ("._loci_bed", "admix_to_bed_individual"),
|
|
39
|
+
"CHROM_SIZES": ("._constants", "CHROM_SIZES"),
|
|
40
|
+
"COORDINATES": ("._constants", "COORDINATES"),
|
|
41
|
+
"BinaryFileNotFoundError": ("._errorhandling", "BinaryFileNotFoundError"),
|
|
42
|
+
"interpolate_array": ("._imputation", "interpolate_array"),
|
|
43
|
+
"get_pops": ("._utils", "get_pops"),
|
|
44
|
+
"get_prefixes": ("._utils", "get_prefixes"),
|
|
45
|
+
"create_binaries": ("._utils", "create_binaries"),
|
|
46
|
+
"get_sample_names": ("._utils", "get_sample_names"),
|
|
47
|
+
"set_gpu_environment": ("._utils", "set_gpu_environment"),
|
|
48
|
+
"delete_files_or_directories": ("._utils", "delete_files_or_directories"),
|
|
49
|
+
"save_multi_format": ("._visualization", "save_multi_format"),
|
|
50
|
+
"generate_tagore_bed": ("._visualization", "generate_tagore_bed"),
|
|
51
|
+
"plot_global_ancestry": ("._visualization", "plot_global_ancestry"),
|
|
52
|
+
"plot_ancestry_by_chromosome": ("._visualization", "plot_ancestry_by_chromosome"),
|
|
53
|
+
"plot_local_ancestry_tagore": ("._tagore", "plot_local_ancestry_tagore"),
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
def __getattr__(name: str):
|
|
57
|
+
"""Lazy attribute loader to keep import-time light."""
|
|
58
|
+
if name in _lazy:
|
|
59
|
+
import importlib
|
|
60
|
+
mod_name, attr_name = _lazy[name]
|
|
61
|
+
mod = importlib.import_module(mod_name, __name__)
|
|
62
|
+
obj = getattr(mod, attr_name)
|
|
63
|
+
globals()[name] = obj # cache for future access
|
|
64
|
+
return obj
|
|
65
|
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def __dir__():
|
|
69
|
+
# help() and tab-complete show public API
|
|
70
|
+
return sorted(list(globals().keys()) + __all__)
|
|
71
|
+
|
|
72
|
+
# Make type checkers happy without importing heavy deps at runtime
|
|
73
|
+
if TYPE_CHECKING:
|
|
74
|
+
from ._chunk import Chunk
|
|
75
|
+
from ._fb_read import read_fb
|
|
76
|
+
from ._read_simu import read_simu
|
|
77
|
+
from ._read_rfmix import read_rfmix
|
|
78
|
+
from ._read_flare import read_flare
|
|
79
|
+
from ._write_data import write_data
|
|
80
|
+
from ._loci_bed import admix_to_bed_individual
|
|
81
|
+
from ._constants import CHROM_SIZES, COORDINATES
|
|
82
|
+
from ._errorhandling import BinaryFileNotFoundError
|
|
83
|
+
from ._imputation import interpolate_array
|
|
84
|
+
from ._utils import (
|
|
85
|
+
get_pops, get_prefixes, create_binaries, get_sample_names,
|
|
86
|
+
set_gpu_environment, delete_files_or_directories,
|
|
87
|
+
)
|
|
88
|
+
from ._visualization import (
|
|
89
|
+
save_multi_format, generate_tagore_bed,
|
|
90
|
+
plot_global_ancestry, plot_ancestry_by_chromosome,
|
|
91
|
+
)
|
|
92
|
+
from ._tagore import plot_local_ancestry_tagore
|
|
@@ -5,7 +5,6 @@ Source: https://github.com/limix/pandas-plink/blob/main/pandas_plink/_bed_read.p
|
|
|
5
5
|
from dask.delayed import delayed
|
|
6
6
|
from dask.array import from_delayed, Array, concatenate
|
|
7
7
|
from numpy import (
|
|
8
|
-
ascontiguousarray,
|
|
9
8
|
float32,
|
|
10
9
|
memmap,
|
|
11
10
|
int32
|
|
@@ -66,7 +65,7 @@ def read_fb(
|
|
|
66
65
|
col_end,
|
|
67
66
|
)
|
|
68
67
|
shape = (row_end - row_start, col_end - col_start)
|
|
69
|
-
row_sx.append(from_delayed(x, shape, dtype=
|
|
68
|
+
row_sx.append(from_delayed(x, shape, dtype=int32))
|
|
70
69
|
col_start = col_end
|
|
71
70
|
col_sx.append(concatenate(row_sx, 1, True))
|
|
72
71
|
row_start = row_end
|
|
@@ -103,5 +102,5 @@ def _read_chunk(
|
|
|
103
102
|
|
|
104
103
|
buff = memmap(filepath, dtype=float32, mode="r",
|
|
105
104
|
offset=offset, shape=size)
|
|
106
|
-
return
|
|
105
|
+
return buff.astype(int32, copy=False)
|
|
107
106
|
|