rfmix-reader 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rfmix_reader/__init__.py +12 -0
- rfmix_reader/chunk.py +34 -0
- rfmix_reader/fb_read.py +133 -0
- rfmix_reader/fb_reader.c +54 -0
- rfmix_reader/include/fb_reader.h +8 -0
- rfmix_reader/read_rfmix.py +405 -0
- rfmix_reader/test/data_files/two_populations/1k_sampleinfo.tsv +2504 -0
- rfmix_reader/test/data_files/two_populations/AFR_admixed.dat +2 -0
- rfmix_reader/test/data_files/two_populations/chr20.bp +12682 -0
- rfmix_reader/test/data_files/two_populations/chr21.bp +8197 -0
- rfmix_reader/test/data_files/two_populations/chr21.vcf.gz +0 -0
- rfmix_reader/test/data_files/two_populations/chr22.bp +9240 -0
- rfmix_reader/test/data_files/two_populations/chr22.vcf.gz +0 -0
- rfmix_reader/test/data_files/two_populations/step_0.sh +39 -0
- rfmix_reader/test/data_files/two_populations/step_1.sh +46 -0
- rfmix_reader/test/data_files/two_populations/step_2.sh +45 -0
- rfmix_reader-0.1.0.dist-info/LICENSE +674 -0
- rfmix_reader-0.1.0.dist-info/METADATA +47 -0
- rfmix_reader-0.1.0.dist-info/RECORD +21 -0
- rfmix_reader-0.1.0.dist-info/WHEEL +4 -0
- rfmix_reader-0.1.0.dist-info/entry_points.txt +3 -0
rfmix_reader/__init__.py
ADDED
rfmix_reader/chunk.py
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Adapted from the `_chunk.py` script in the `pandas-plink` package.
|
|
3
|
+
Source: https://github.com/limix/pandas-plink/blob/main/pandas_plink/_chunk.py
|
|
4
|
+
"""
|
|
5
|
+
from typing import Optional
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
|
|
8
|
+
__all__ = ["Chunk"]
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass
|
|
12
|
+
class Chunk:
|
|
13
|
+
"""
|
|
14
|
+
Chunk specification for a contiguous submatrix of the haplotype matrix.
|
|
15
|
+
|
|
16
|
+
Parameters
|
|
17
|
+
----------
|
|
18
|
+
nsamples : Optional[int], default=1024
|
|
19
|
+
Number of samples in a single chunk, limited by the total number of
|
|
20
|
+
samples. Set to `None` to include all samples.
|
|
21
|
+
nloci : Optional[int], default=1024
|
|
22
|
+
Number of loci in a single chunk, limited by the total number of
|
|
23
|
+
loci. Set to `None` to include all loci.
|
|
24
|
+
|
|
25
|
+
Notes
|
|
26
|
+
-----
|
|
27
|
+
- Small chunks may increase computational time, while large chunks may increase
|
|
28
|
+
memory usage.
|
|
29
|
+
- For small datasets, try setting both `nsamples` and `nloci` to `None`.
|
|
30
|
+
- For large datasets where you need to use every sample, try setting `nsamples=None`
|
|
31
|
+
and choose a small value for `nloci`.
|
|
32
|
+
"""
|
|
33
|
+
nsamples: Optional[int] = 1024
|
|
34
|
+
nloci: Optional[int] = 1024
|
rfmix_reader/fb_read.py
ADDED
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Adapted from `_bed_read.py` script in the `pandas-plink` package.
|
|
3
|
+
Source: https://github.com/limix/pandas-plink/blob/main/pandas_plink/_bed_read.py
|
|
4
|
+
"""
|
|
5
|
+
from numpy import (
|
|
6
|
+
ascontiguousarray,
|
|
7
|
+
empty,
|
|
8
|
+
float32,
|
|
9
|
+
memmap,
|
|
10
|
+
uint8,
|
|
11
|
+
uint64,
|
|
12
|
+
zeros,
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
__all__ = ["read_fb"]
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def read_fb(filepath, nrows, ncols, row_chunk, col_chunk):
|
|
19
|
+
"""
|
|
20
|
+
Read and process data from a file in chunks, skipping the first
|
|
21
|
+
2 rows (comments) and 4 columns (loci annotation).
|
|
22
|
+
|
|
23
|
+
Parameters:
|
|
24
|
+
filepath (str): Path to the file.
|
|
25
|
+
nrows (int): Total number of rows in the dataset.
|
|
26
|
+
ncols (int): Total number of columns in the dataset.
|
|
27
|
+
row_chunk (int): Number of rows to process in each chunk.
|
|
28
|
+
col_chunk (int): Number of columns to process in each chunk.
|
|
29
|
+
|
|
30
|
+
Returns:
|
|
31
|
+
dask.array: Concatenated array of processed data.
|
|
32
|
+
"""
|
|
33
|
+
from dask.delayed import delayed
|
|
34
|
+
from dask.array import concatenate, from_delayed
|
|
35
|
+
|
|
36
|
+
# Validate input parameters
|
|
37
|
+
if nrows <= 2 or ncols <= 4:
|
|
38
|
+
raise ValueError("Number of rows must be greater than 2 and number of columns must be greater than 4.")
|
|
39
|
+
if row_chunk <= 0 or col_chunk <= 0:
|
|
40
|
+
raise ValueError("row_chunk and col_chunk must be positive integers.")
|
|
41
|
+
|
|
42
|
+
# Calculate row size and total size for memory mapping
|
|
43
|
+
row_size = (ncols + 3) // 4
|
|
44
|
+
size = nrows * row_size
|
|
45
|
+
|
|
46
|
+
try:
|
|
47
|
+
buff = memmap(filepath, uint8, "r", 3, shape=(size,))
|
|
48
|
+
except Exception as e:
|
|
49
|
+
raise IOError(f"Error reading file: {e}")
|
|
50
|
+
|
|
51
|
+
row_start = 2 # Skip the first 2 rows
|
|
52
|
+
column_chunks = []
|
|
53
|
+
|
|
54
|
+
while row_start < nrows:
|
|
55
|
+
row_end = min(row_start + row_chunk, nrows)
|
|
56
|
+
col_start = 4 # Skip the first 4 columns
|
|
57
|
+
row_chunks = []
|
|
58
|
+
|
|
59
|
+
while col_start < ncols:
|
|
60
|
+
col_end = min(col_start + col_chunk, ncols)
|
|
61
|
+
x = delayed(_read_fb_chunk, None, True, None, False)(
|
|
62
|
+
buff,
|
|
63
|
+
nrows,
|
|
64
|
+
ncols,
|
|
65
|
+
row_start,
|
|
66
|
+
row_end,
|
|
67
|
+
col_start,
|
|
68
|
+
col_end,
|
|
69
|
+
)
|
|
70
|
+
shape = (row_end - row_start, (col_end - col_start))
|
|
71
|
+
row_chunks.append(from_delayed(x, shape, float32))
|
|
72
|
+
col_start = col_end
|
|
73
|
+
|
|
74
|
+
column_chunks.append(concatenate(row_chunks, 1, True))
|
|
75
|
+
row_start = row_end
|
|
76
|
+
|
|
77
|
+
return concatenate(column_chunks, 0, True)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _read_fb_chunk(
|
|
81
|
+
buff, nrows, ncols, row_start, row_end, col_start, col_end
|
|
82
|
+
):
|
|
83
|
+
"""
|
|
84
|
+
Read a chunk of data from the buffer and process it based on populations.
|
|
85
|
+
|
|
86
|
+
Parameters:
|
|
87
|
+
buff (memmap): Memory-mapped buffer containing the data.
|
|
88
|
+
nrows (int): Total number of rows in the dataset.
|
|
89
|
+
ncols (int): Total number of columns in the dataset.
|
|
90
|
+
row_start (int): Starting row index for the chunk.
|
|
91
|
+
row_end (int): Ending row index for the chunk.
|
|
92
|
+
col_start (int): Starting column index for the chunk.
|
|
93
|
+
col_end (int): Ending column index for the chunk.
|
|
94
|
+
|
|
95
|
+
Returns:
|
|
96
|
+
dask.array: Processed array with adjacent columns summed for each population subset.
|
|
97
|
+
"""
|
|
98
|
+
from fb_reader import ffi, lib
|
|
99
|
+
|
|
100
|
+
base_type = uint8
|
|
101
|
+
base_size = base_type().nbytes
|
|
102
|
+
base_repr = "uint8_t"
|
|
103
|
+
|
|
104
|
+
# Ensure the number of columns to be processed is even
|
|
105
|
+
num_cols = col_end - col_start
|
|
106
|
+
if num_cols % 2 != 0:
|
|
107
|
+
raise ValueError("Number of columns must be even.")
|
|
108
|
+
|
|
109
|
+
X = zeros((row_end - row_start, num_cols), base_type)
|
|
110
|
+
assert X.flags.aligned
|
|
111
|
+
|
|
112
|
+
strides = empty(2, uint64)
|
|
113
|
+
strides[:] = X.strides
|
|
114
|
+
strides //= base_size
|
|
115
|
+
|
|
116
|
+
try:
|
|
117
|
+
lib.read_fb_chunk(
|
|
118
|
+
ffi.cast(f"{base_repr} *", buff.ctypes.data),
|
|
119
|
+
nrows,
|
|
120
|
+
ncols,
|
|
121
|
+
row_start,
|
|
122
|
+
col_start,
|
|
123
|
+
row_end,
|
|
124
|
+
col_end,
|
|
125
|
+
ffi.cast(f"{base_repr} *", X.ctypes.data),
|
|
126
|
+
ffi.cast("uint64_t *", strides.ctypes.data),
|
|
127
|
+
)
|
|
128
|
+
except Exception as e:
|
|
129
|
+
raise IOError(f"Error reading data chunk: {e}")
|
|
130
|
+
|
|
131
|
+
# Convert to contiguous array of type float32
|
|
132
|
+
return ascontiguousarray(X, float32)
|
|
133
|
+
|
rfmix_reader/fb_reader.c
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* Adapted from the `_bed_reader.h` script in the `pandas-plink` package.
|
|
3
|
+
* Source: https://github.com/limix/pandas-plink/blob/main/pandas_plink/_bed_reader.h
|
|
4
|
+
*/
|
|
5
|
+
|
|
6
|
+
#include <math.h>
|
|
7
|
+
#include <stdio.h>
|
|
8
|
+
#include <stdlib.h>
|
|
9
|
+
|
|
10
|
+
#define MIN(a,b) ((a > b) ? b : a)
|
|
11
|
+
|
|
12
|
+
void read_fb_chunk(uint8_t *buff, uint64_t nrows, uint64_t ncols,
|
|
13
|
+
uint64_t row_start, uint64_t col_start, uint64_t row_end,
|
|
14
|
+
uint64_t col_end, uint8_t *out, uint64_t *strides)
|
|
15
|
+
{
|
|
16
|
+
char b, b0, b1, p0, p1;
|
|
17
|
+
uint64_t r;
|
|
18
|
+
uint64_t c, ce;
|
|
19
|
+
uint64_t row_chunk;
|
|
20
|
+
uint64_t row_size;
|
|
21
|
+
|
|
22
|
+
// in bytes
|
|
23
|
+
row_chunk = (col_end - col_start + 3) / 4;
|
|
24
|
+
// in bytes
|
|
25
|
+
row_size = (ncols + 3) / 4;
|
|
26
|
+
|
|
27
|
+
r = row_start;
|
|
28
|
+
buff += r * row_size + col_start / 4;
|
|
29
|
+
|
|
30
|
+
while (r < row_end)
|
|
31
|
+
{
|
|
32
|
+
for (c = col_start; c < col_end;)
|
|
33
|
+
{
|
|
34
|
+
b = buff[(c - col_start) / 4];
|
|
35
|
+
|
|
36
|
+
b0 = b & 0x55;
|
|
37
|
+
b1 = (b & 0xAA) >> 1;
|
|
38
|
+
|
|
39
|
+
p0 = b0 ^ b1;
|
|
40
|
+
p1 = (b0 | b1) & b0;
|
|
41
|
+
p1 <<= 1;
|
|
42
|
+
p0 |= p1;
|
|
43
|
+
ce = MIN(c + 4, col_end);
|
|
44
|
+
for (; c < ce; ++c)
|
|
45
|
+
{
|
|
46
|
+
out[(r - row_start) * strides[0] +
|
|
47
|
+
(c - col_start) * strides[1]] = p0 & 3;
|
|
48
|
+
p0 >>= 2;
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
++r;
|
|
52
|
+
buff += row_size;
|
|
53
|
+
}
|
|
54
|
+
}
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* Adapted from the `_bed_reader.h` script in the `pandas-plink` package.
|
|
3
|
+
* Source: https://github.com/limix/pandas-plink/blob/main/pandas_plink/_bed_reader.h
|
|
4
|
+
*/
|
|
5
|
+
|
|
6
|
+
void read_fb_chunk(uint8_t *buff, uint64_t nrows, uint64_t ncols,
|
|
7
|
+
uint64_t row_start, uint64_t col_start, uint64_t row_end,
|
|
8
|
+
uint64_t col_end, uint8_t *out, uint64_t *strides);
|
|
@@ -0,0 +1,405 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Adapted from `_read.py` script in the `pandas-plink` package.
|
|
3
|
+
Source: https://github.com/limix/pandas-plink/blob/main/pandas_plink/_read.py
|
|
4
|
+
"""
|
|
5
|
+
import warnings
|
|
6
|
+
from glob import glob
|
|
7
|
+
from os.path import basename, dirname, join
|
|
8
|
+
from collections import OrderedDict as odict
|
|
9
|
+
from typing import Optional, Callable, List, Tuple
|
|
10
|
+
|
|
11
|
+
from dask.array import Array
|
|
12
|
+
from pandas import DataFrame, read_csv
|
|
13
|
+
|
|
14
|
+
from chunk import Chunk
|
|
15
|
+
from fb_read import read_fb
|
|
16
|
+
|
|
17
|
+
__all__ = ["read_rfmix"]
|
|
18
|
+
|
|
19
|
+
def read_rfmix(
|
|
20
|
+
file_prefix: str, verbose: bool = True
|
|
21
|
+
) -> Tuple[DataFrame, DataFrame, Array]:
|
|
22
|
+
"""
|
|
23
|
+
Read RFMix files into data frames and a Dask array.
|
|
24
|
+
|
|
25
|
+
Notes
|
|
26
|
+
-----
|
|
27
|
+
Local ancestry can be either :const:`0`, :const:`1`, :const:`2`, or
|
|
28
|
+
:data:`math.nan`:
|
|
29
|
+
|
|
30
|
+
- :const:`0` No alleles are associated with this ancestry
|
|
31
|
+
- :const:`1` One allele is associated with this ancestry
|
|
32
|
+
- :const:`2` Both alleles are associated with this ancestry
|
|
33
|
+
|
|
34
|
+
Parameters
|
|
35
|
+
----------
|
|
36
|
+
file_prefix : str
|
|
37
|
+
Path prefix to the set of RFMix files. It will load all of the chromosomes
|
|
38
|
+
at once.
|
|
39
|
+
verbose : bool, optional
|
|
40
|
+
:const:`True` for progress information; :const:`False` otherwise.
|
|
41
|
+
|
|
42
|
+
Returns
|
|
43
|
+
-------
|
|
44
|
+
loci : :class:`pandas.DataFrame`
|
|
45
|
+
Loci information for the FB data.
|
|
46
|
+
rf_q : :class:`pandas.DataFrame`
|
|
47
|
+
Global ancestry by chromosome from RFMix.
|
|
48
|
+
admix : :class:`dask.array.Array`
|
|
49
|
+
Local ancestry per population (columns pop1*nsamples ... popX*nsamples).
|
|
50
|
+
This is in order of the populations see `rf_q`.
|
|
51
|
+
"""
|
|
52
|
+
from tqdm import tqdm
|
|
53
|
+
from pandas import concat
|
|
54
|
+
from dask.array import concatenate
|
|
55
|
+
|
|
56
|
+
# Get file prefixes
|
|
57
|
+
file_prefixes = sorted(glob(file_prefix))
|
|
58
|
+
if len(file_prefixes) == 1:
|
|
59
|
+
file_prefixes = sorted(glob(join(file_prefix, "*")))
|
|
60
|
+
|
|
61
|
+
file_prefixes = sorted(_clean_prefixes(file_prefixes))
|
|
62
|
+
fn = [{s: f"{fp}.{s}" for s in ["fb.tsv", "rfmix.Q"]} for fp in file_prefixes]
|
|
63
|
+
|
|
64
|
+
# Load loci information
|
|
65
|
+
pbar = tqdm(desc="Mapping loci files", total=len(fn), disable=not verbose)
|
|
66
|
+
loci = _read_file(fn, lambda f: _read_loci(f["fb.tsv"]), pbar)
|
|
67
|
+
pbar.close()
|
|
68
|
+
if len(file_prefixes) > 1 and verbose:
|
|
69
|
+
msg = "Multiple files read in this order:"
|
|
70
|
+
print(f"{msg} {[basename(f) for f in file_prefixes]}")
|
|
71
|
+
|
|
72
|
+
# Adjust loci indices and concatenate
|
|
73
|
+
nmarkers = {}
|
|
74
|
+
index_offset = 0
|
|
75
|
+
for i, bi in enumerate(loci):
|
|
76
|
+
nmarkers[fn[i]["fb.tsv"]] = bi.shape[0]
|
|
77
|
+
bi["i"] += index_offset
|
|
78
|
+
index_offset += bi.shape[0]
|
|
79
|
+
loci = concat(loci, axis=0, ignore_index=True)
|
|
80
|
+
|
|
81
|
+
# Load global ancestry per chromosome
|
|
82
|
+
pbar = tqdm(desc="Mapping Q files", total=len(fn), disable=not verbose)
|
|
83
|
+
rf_q = _read_file(fn, lambda f: _read_Q(f["rfmix.Q"]), pbar)
|
|
84
|
+
pbar.close()
|
|
85
|
+
|
|
86
|
+
nsamples = rf_q[0].shape[0]
|
|
87
|
+
pops = rf_q[0].drop(["sample_id", "chrom"], axis=1).columns.values
|
|
88
|
+
rf_q = concat(rf_q, axis=0, ignore_index=True)
|
|
89
|
+
|
|
90
|
+
# Loading local ancestry by loci
|
|
91
|
+
pbar = tqdm(desc="Mapping fb files", total=len(fn), disable=not verbose)
|
|
92
|
+
admix = _read_file(
|
|
93
|
+
fn,
|
|
94
|
+
lambda f: _read_fb(f["fb.tsv"], nsamples,
|
|
95
|
+
nmarkers[f["fb.tsv"]], pops, Chunk()),
|
|
96
|
+
pbar,
|
|
97
|
+
)
|
|
98
|
+
pbar.close()
|
|
99
|
+
admix = concatenate(admix, axis=0)
|
|
100
|
+
|
|
101
|
+
return loci, rf_q, admix
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _read_file(fn: List[str], read_func: Callable, pbar=None) -> List:
|
|
105
|
+
"""
|
|
106
|
+
Read data from multiple files using a provided read function.
|
|
107
|
+
|
|
108
|
+
Parameters:
|
|
109
|
+
----------
|
|
110
|
+
fn (List[str]): A list of file paths to read.
|
|
111
|
+
read_func (Callable): A function to read data from each file.
|
|
112
|
+
pbar (Optional): A progress bar object to update during reading.
|
|
113
|
+
|
|
114
|
+
Returns:
|
|
115
|
+
-------
|
|
116
|
+
List: A list containing the data read from each file.
|
|
117
|
+
"""
|
|
118
|
+
data = [];
|
|
119
|
+
for file_name in fn:
|
|
120
|
+
data.append(read_func(file_name))
|
|
121
|
+
if pbar:
|
|
122
|
+
pbar.update(1)
|
|
123
|
+
return data
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def _read_csv(fn: str, header: dict) -> DataFrame:
|
|
127
|
+
"""
|
|
128
|
+
Read a CSV file into a pandas DataFrame with specified data types.
|
|
129
|
+
|
|
130
|
+
Parameters:
|
|
131
|
+
----------
|
|
132
|
+
fn (str): The file path of the CSV file.
|
|
133
|
+
header (dict): A dictionary mapping column names to data types.
|
|
134
|
+
|
|
135
|
+
Returns:
|
|
136
|
+
-------
|
|
137
|
+
DataFrame: The data read from the CSV file as a pandas DataFrame.
|
|
138
|
+
"""
|
|
139
|
+
try:
|
|
140
|
+
df = read_csv(
|
|
141
|
+
fn,
|
|
142
|
+
delim_whitespace=True,
|
|
143
|
+
header=None,
|
|
144
|
+
names=list(header.keys()),
|
|
145
|
+
dtype=header,
|
|
146
|
+
comment="#",
|
|
147
|
+
compression=None,
|
|
148
|
+
engine="c",
|
|
149
|
+
iterator=False,
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
except Exception as e:
|
|
153
|
+
raise IOError(f"Error reading file '{fn}': {e}")
|
|
154
|
+
# Validate that resulting DataFrame is correct type
|
|
155
|
+
if not isinstance(df, DataFrame):
|
|
156
|
+
raise ValueError(f"Expected a DataFrame but got {type(df)} instead.")
|
|
157
|
+
|
|
158
|
+
return df
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _read_tsv(fn: str) -> DataFrame:
|
|
162
|
+
"""
|
|
163
|
+
Read a TSV file into a pandas DataFrame.
|
|
164
|
+
|
|
165
|
+
Parameters:
|
|
166
|
+
----------
|
|
167
|
+
fn (str): File name of the TSV file.
|
|
168
|
+
|
|
169
|
+
Returns:
|
|
170
|
+
-------
|
|
171
|
+
DataFrame: DataFrame containing specified columns from the TSV file.
|
|
172
|
+
"""
|
|
173
|
+
from numpy import int32
|
|
174
|
+
from pandas import StringDtype, concat
|
|
175
|
+
|
|
176
|
+
header = {"chromosome": StringDtype(), "physical_position": int32}
|
|
177
|
+
columns = ["chromosome", "physical_position"]
|
|
178
|
+
|
|
179
|
+
try:
|
|
180
|
+
chunks = read_csv(
|
|
181
|
+
fn,
|
|
182
|
+
delim_whitespace=True,
|
|
183
|
+
header=0,
|
|
184
|
+
usecols=columns,
|
|
185
|
+
dtype=header,
|
|
186
|
+
comment="#",
|
|
187
|
+
chunksize=100000, # Low memory chunks
|
|
188
|
+
)
|
|
189
|
+
|
|
190
|
+
# Concatenate chunks into single DataFrame
|
|
191
|
+
df = concat(chunks, ignore_index=True)
|
|
192
|
+
except FileNotFoundError:
|
|
193
|
+
raise FileNotFoundError(f"File {fn} not found.")
|
|
194
|
+
except Exception as e:
|
|
195
|
+
raise IOError(f"Error reading file {fn}: {e}")
|
|
196
|
+
|
|
197
|
+
# Validate that resulting DataFrame is correct type
|
|
198
|
+
if not isinstance(df, DataFrame):
|
|
199
|
+
raise ValueError(f"Expected a DataFrame but got {type(df)} instead.")
|
|
200
|
+
|
|
201
|
+
# Ensure DataFrame contains correct columns
|
|
202
|
+
if not all(column in df.columns for column in columns):
|
|
203
|
+
raise ValueError(f"DataFrame does not contain expected columns: {columns}")
|
|
204
|
+
|
|
205
|
+
return df
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def _read_loci(fn: str) -> DataFrame:
|
|
209
|
+
"""
|
|
210
|
+
Read loci information from a TSV file and add a sequential index column.
|
|
211
|
+
|
|
212
|
+
Parameters:
|
|
213
|
+
----------
|
|
214
|
+
fn (str): The file path of the TSV file containing loci information.
|
|
215
|
+
|
|
216
|
+
Returns:
|
|
217
|
+
-------
|
|
218
|
+
DataFrame: A pandas DataFrame containing the loci information with an additional 'i' column for indexing.
|
|
219
|
+
"""
|
|
220
|
+
df = _read_tsv(fn)
|
|
221
|
+
df["i"] = range(df.shape[0])
|
|
222
|
+
return df
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def _read_Q(fn: str) -> DataFrame:
|
|
226
|
+
"""
|
|
227
|
+
Read the Q matrix from a file and add the chromosome information.
|
|
228
|
+
|
|
229
|
+
Parameters:
|
|
230
|
+
----------
|
|
231
|
+
fn (str): The file path of the Q matrix file.
|
|
232
|
+
|
|
233
|
+
Returns:
|
|
234
|
+
-------
|
|
235
|
+
DataFrame: The Q matrix with the chromosome information added.
|
|
236
|
+
"""
|
|
237
|
+
from re import search
|
|
238
|
+
|
|
239
|
+
df = _read_Q_noi(fn)
|
|
240
|
+
match = search(r'chr(\d+)', fn)
|
|
241
|
+
if match:
|
|
242
|
+
chrom = match.group(0)
|
|
243
|
+
df["chrom"] = chrom
|
|
244
|
+
else:
|
|
245
|
+
print(f"Warning: Could not extract chromosome information from '{fn}'")
|
|
246
|
+
return df
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def _read_Q_noi(fn: str) -> DataFrame:
|
|
250
|
+
"""
|
|
251
|
+
Read the Q matrix from a file without adding chromosome information.
|
|
252
|
+
|
|
253
|
+
Parameters:
|
|
254
|
+
----------
|
|
255
|
+
fn (str): The file path of the Q matrix file.
|
|
256
|
+
|
|
257
|
+
Returns:
|
|
258
|
+
-------
|
|
259
|
+
DataFrame: The Q matrix without chromosome information.
|
|
260
|
+
"""
|
|
261
|
+
try:
|
|
262
|
+
header = odict(_types(fn))
|
|
263
|
+
return _read_csv(fn, header)
|
|
264
|
+
except Exception as e:
|
|
265
|
+
raise IOError(f"Error reading Q matrix from '{fn}': {e}")
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
def _read_fb(fn: str, nsamples: int, nloci: int, pops: list, chunk: Optional[Chunk] = None) -> Array:
|
|
269
|
+
"""
|
|
270
|
+
Read the forward-backward matrix from a file as a Dask Array.
|
|
271
|
+
|
|
272
|
+
Parameters:
|
|
273
|
+
----------
|
|
274
|
+
fn (str): The file path of the forward-backward matrix file.
|
|
275
|
+
nsamples (int): The number of samples in the dataset.
|
|
276
|
+
nloci (int): The number of loci in the dataset.
|
|
277
|
+
pops (list): A list of population labels.
|
|
278
|
+
chunk (Chunk, optional): A Chunk object specifying the chunk size for reading.
|
|
279
|
+
|
|
280
|
+
Returns:
|
|
281
|
+
-------
|
|
282
|
+
dask.array.Array: The forward-backward matrix as a Dask Array.
|
|
283
|
+
"""
|
|
284
|
+
npops = len(pops)
|
|
285
|
+
nrows = nloci
|
|
286
|
+
ncols = nsamples * npops * 2
|
|
287
|
+
row_chunk = nrows if chunk.nloci is None else min(nrows, chunk.nloci)
|
|
288
|
+
col_chunk = ncols if chunk.nsamples is None else min(ncols, chunk.nsamples)
|
|
289
|
+
max_npartitions = 16_384
|
|
290
|
+
row_chunk = max(nrows // max_npartitions, row_chunk)
|
|
291
|
+
col_chunk = max(ncols // max_npartitions, col_chunk)
|
|
292
|
+
X = read_fb(fn, nrows, ncols, row_chunk, col_chunk)
|
|
293
|
+
|
|
294
|
+
# Subset populations and sum adjacent columns
|
|
295
|
+
return _subset_populations(X, npops)
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
def _subset_populations(X: Array, npops: int) -> Array:
|
|
299
|
+
"""
|
|
300
|
+
Subset and process the input array X based on populations.
|
|
301
|
+
|
|
302
|
+
Parameters:
|
|
303
|
+
X (dask.array): Input array where columns represent data for different populations.
|
|
304
|
+
npops (int): Number of populations for column processing.
|
|
305
|
+
|
|
306
|
+
Returns:
|
|
307
|
+
dask.array: Processed array with adjacent columns summed for each population subset.
|
|
308
|
+
"""
|
|
309
|
+
from dask.array import concatenate
|
|
310
|
+
|
|
311
|
+
pop_subset = []
|
|
312
|
+
pop_start = 0
|
|
313
|
+
ncols = X.shape[1]
|
|
314
|
+
|
|
315
|
+
if ncols % npops != 0:
|
|
316
|
+
raise ValueError("The number of columns in X must be divisible by npops.")
|
|
317
|
+
|
|
318
|
+
while pop_start < npops:
|
|
319
|
+
X0 = X[:, pop_start::npops] # Subset based on populations
|
|
320
|
+
if X0.shape[1] % 2 != 0:
|
|
321
|
+
raise ValueError("Number of columns must be even.")
|
|
322
|
+
X0_summed = X0[:, ::2] + X0[:, 1::2] # Sum adjacent columns
|
|
323
|
+
pop_subset.append(X0_summed)
|
|
324
|
+
pop_start += 1
|
|
325
|
+
|
|
326
|
+
return concatenate(pop_subset, 1, True)
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
def _types(fn: str) -> dict:
|
|
330
|
+
"""
|
|
331
|
+
Infer the data types of columns in a TSV file.
|
|
332
|
+
|
|
333
|
+
Parameters:
|
|
334
|
+
----------
|
|
335
|
+
fn (str) : File name of the TSV file.
|
|
336
|
+
|
|
337
|
+
Returns:
|
|
338
|
+
-------
|
|
339
|
+
dict : Dictionary mapping column names to their inferred data types.
|
|
340
|
+
"""
|
|
341
|
+
from pandas import StringDtype
|
|
342
|
+
|
|
343
|
+
try:
|
|
344
|
+
# Read the first two rows of the file, skipping the first row
|
|
345
|
+
df = read_csv(
|
|
346
|
+
fn,
|
|
347
|
+
delim_whitespace=True,
|
|
348
|
+
nrows=2,
|
|
349
|
+
skiprows=1,
|
|
350
|
+
)
|
|
351
|
+
|
|
352
|
+
except FileNotFoundError:
|
|
353
|
+
raise FileNotFoundError(f"File '{fn}' not found.")
|
|
354
|
+
except Exception as e:
|
|
355
|
+
raise IOError(f"Error reading file '{fn}': {e}")
|
|
356
|
+
|
|
357
|
+
# Validate that the resulting DataFrame is of the correct type
|
|
358
|
+
if not isinstance(df, DataFrame):
|
|
359
|
+
raise ValueError(f"Expected a DataFrame but got {type(df)} instead.")
|
|
360
|
+
|
|
361
|
+
# Ensure the DataFrame contains at least one column
|
|
362
|
+
if df.shape[1] < 1:
|
|
363
|
+
raise ValueError("The DataFrame does not contain any columns.")
|
|
364
|
+
|
|
365
|
+
# Initialize the header dictionary with the sample_id column
|
|
366
|
+
header = {"sample_id": StringDtype()}
|
|
367
|
+
# Update the header dictionary with the data types of the remaining columns
|
|
368
|
+
header.update(df.dtypes[1:].to_dict())
|
|
369
|
+
|
|
370
|
+
return header
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
def _clean_prefixes(prefixes):
|
|
374
|
+
"""
|
|
375
|
+
Clean and filter a list of file prefixes.
|
|
376
|
+
|
|
377
|
+
Parameters:
|
|
378
|
+
----------
|
|
379
|
+
prefixes (list): A list of file prefixes (paths).
|
|
380
|
+
|
|
381
|
+
Returns:
|
|
382
|
+
-------
|
|
383
|
+
list: A list of unique, cleaned file prefixes without the file extensions.
|
|
384
|
+
|
|
385
|
+
Notes:
|
|
386
|
+
-----
|
|
387
|
+
- The function removes any prefixes that end with ".logs".
|
|
388
|
+
- It also removes any duplicate prefixes after cleaning.
|
|
389
|
+
"""
|
|
390
|
+
cleaned_prefixes = []
|
|
391
|
+
for prefix in prefixes:
|
|
392
|
+
# Split the prefix into directory and base name
|
|
393
|
+
dir_path = dirname(prefix)
|
|
394
|
+
base_name = basename(prefix)
|
|
395
|
+
|
|
396
|
+
# Remove the file extensions from the base name
|
|
397
|
+
base = base_name.split(".")[0]
|
|
398
|
+
|
|
399
|
+
# Skip prefixes that end with ".logs"
|
|
400
|
+
if base != "logs":
|
|
401
|
+
cleaned_prefix = join(dir_path, base)
|
|
402
|
+
cleaned_prefixes.append(cleaned_prefix)
|
|
403
|
+
|
|
404
|
+
# Remove duplicate prefixes
|
|
405
|
+
return list(set(cleaned_prefixes))
|