rfmix-reader 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,12 @@
1
+ from chunk import Chunk
2
+ from fb_read import read_fb
3
+ from read_rfmix import read_rfmix
4
+
5
+ __version__ = "0.1.0"
6
+
7
+ __all__ = [
8
+ "Chunk",
9
+ "__version__",
10
+ "read_fb",
11
+ "read_rfmix",
12
+ ]
rfmix_reader/chunk.py ADDED
@@ -0,0 +1,34 @@
1
+ """
2
+ Adapted from the `_chunk.py` script in the `pandas-plink` package.
3
+ Source: https://github.com/limix/pandas-plink/blob/main/pandas_plink/_chunk.py
4
+ """
5
+ from typing import Optional
6
+ from dataclasses import dataclass
7
+
8
+ __all__ = ["Chunk"]
9
+
10
+
11
+ @dataclass
12
+ class Chunk:
13
+ """
14
+ Chunk specification for a contiguous submatrix of the haplotype matrix.
15
+
16
+ Parameters
17
+ ----------
18
+ nsamples : Optional[int], default=1024
19
+ Number of samples in a single chunk, limited by the total number of
20
+ samples. Set to `None` to include all samples.
21
+ nloci : Optional[int], default=1024
22
+ Number of loci in a single chunk, limited by the total number of
23
+ loci. Set to `None` to include all loci.
24
+
25
+ Notes
26
+ -----
27
+ - Small chunks may increase computational time, while large chunks may increase
28
+ memory usage.
29
+ - For small datasets, try setting both `nsamples` and `nloci` to `None`.
30
+ - For large datasets where you need to use every sample, try setting `nsamples=None`
31
+ and choose a small value for `nloci`.
32
+ """
33
+ nsamples: Optional[int] = 1024
34
+ nloci: Optional[int] = 1024
@@ -0,0 +1,133 @@
1
+ """
2
+ Adapted from `_bed_read.py` script in the `pandas-plink` package.
3
+ Source: https://github.com/limix/pandas-plink/blob/main/pandas_plink/_bed_read.py
4
+ """
5
+ from numpy import (
6
+ ascontiguousarray,
7
+ empty,
8
+ float32,
9
+ memmap,
10
+ uint8,
11
+ uint64,
12
+ zeros,
13
+ )
14
+
15
+ __all__ = ["read_fb"]
16
+
17
+
18
+ def read_fb(filepath, nrows, ncols, row_chunk, col_chunk):
19
+ """
20
+ Read and process data from a file in chunks, skipping the first
21
+ 2 rows (comments) and 4 columns (loci annotation).
22
+
23
+ Parameters:
24
+ filepath (str): Path to the file.
25
+ nrows (int): Total number of rows in the dataset.
26
+ ncols (int): Total number of columns in the dataset.
27
+ row_chunk (int): Number of rows to process in each chunk.
28
+ col_chunk (int): Number of columns to process in each chunk.
29
+
30
+ Returns:
31
+ dask.array: Concatenated array of processed data.
32
+ """
33
+ from dask.delayed import delayed
34
+ from dask.array import concatenate, from_delayed
35
+
36
+ # Validate input parameters
37
+ if nrows <= 2 or ncols <= 4:
38
+ raise ValueError("Number of rows must be greater than 2 and number of columns must be greater than 4.")
39
+ if row_chunk <= 0 or col_chunk <= 0:
40
+ raise ValueError("row_chunk and col_chunk must be positive integers.")
41
+
42
+ # Calculate row size and total size for memory mapping
43
+ row_size = (ncols + 3) // 4
44
+ size = nrows * row_size
45
+
46
+ try:
47
+ buff = memmap(filepath, uint8, "r", 3, shape=(size,))
48
+ except Exception as e:
49
+ raise IOError(f"Error reading file: {e}")
50
+
51
+ row_start = 2 # Skip the first 2 rows
52
+ column_chunks = []
53
+
54
+ while row_start < nrows:
55
+ row_end = min(row_start + row_chunk, nrows)
56
+ col_start = 4 # Skip the first 4 columns
57
+ row_chunks = []
58
+
59
+ while col_start < ncols:
60
+ col_end = min(col_start + col_chunk, ncols)
61
+ x = delayed(_read_fb_chunk, None, True, None, False)(
62
+ buff,
63
+ nrows,
64
+ ncols,
65
+ row_start,
66
+ row_end,
67
+ col_start,
68
+ col_end,
69
+ )
70
+ shape = (row_end - row_start, (col_end - col_start))
71
+ row_chunks.append(from_delayed(x, shape, float32))
72
+ col_start = col_end
73
+
74
+ column_chunks.append(concatenate(row_chunks, 1, True))
75
+ row_start = row_end
76
+
77
+ return concatenate(column_chunks, 0, True)
78
+
79
+
80
+ def _read_fb_chunk(
81
+ buff, nrows, ncols, row_start, row_end, col_start, col_end
82
+ ):
83
+ """
84
+ Read a chunk of data from the buffer and process it based on populations.
85
+
86
+ Parameters:
87
+ buff (memmap): Memory-mapped buffer containing the data.
88
+ nrows (int): Total number of rows in the dataset.
89
+ ncols (int): Total number of columns in the dataset.
90
+ row_start (int): Starting row index for the chunk.
91
+ row_end (int): Ending row index for the chunk.
92
+ col_start (int): Starting column index for the chunk.
93
+ col_end (int): Ending column index for the chunk.
94
+
95
+ Returns:
96
+ dask.array: Processed array with adjacent columns summed for each population subset.
97
+ """
98
+ from fb_reader import ffi, lib
99
+
100
+ base_type = uint8
101
+ base_size = base_type().nbytes
102
+ base_repr = "uint8_t"
103
+
104
+ # Ensure the number of columns to be processed is even
105
+ num_cols = col_end - col_start
106
+ if num_cols % 2 != 0:
107
+ raise ValueError("Number of columns must be even.")
108
+
109
+ X = zeros((row_end - row_start, num_cols), base_type)
110
+ assert X.flags.aligned
111
+
112
+ strides = empty(2, uint64)
113
+ strides[:] = X.strides
114
+ strides //= base_size
115
+
116
+ try:
117
+ lib.read_fb_chunk(
118
+ ffi.cast(f"{base_repr} *", buff.ctypes.data),
119
+ nrows,
120
+ ncols,
121
+ row_start,
122
+ col_start,
123
+ row_end,
124
+ col_end,
125
+ ffi.cast(f"{base_repr} *", X.ctypes.data),
126
+ ffi.cast("uint64_t *", strides.ctypes.data),
127
+ )
128
+ except Exception as e:
129
+ raise IOError(f"Error reading data chunk: {e}")
130
+
131
+ # Convert to contiguous array of type float32
132
+ return ascontiguousarray(X, float32)
133
+
@@ -0,0 +1,54 @@
1
+ /*
2
+ * Adapted from the `_bed_reader.h` script in the `pandas-plink` package.
3
+ * Source: https://github.com/limix/pandas-plink/blob/main/pandas_plink/_bed_reader.h
4
+ */
5
+
6
+ #include <math.h>
7
+ #include <stdio.h>
8
+ #include <stdlib.h>
9
+
10
+ #define MIN(a,b) ((a > b) ? b : a)
11
+
12
+ void read_fb_chunk(uint8_t *buff, uint64_t nrows, uint64_t ncols,
13
+ uint64_t row_start, uint64_t col_start, uint64_t row_end,
14
+ uint64_t col_end, uint8_t *out, uint64_t *strides)
15
+ {
16
+ char b, b0, b1, p0, p1;
17
+ uint64_t r;
18
+ uint64_t c, ce;
19
+ uint64_t row_chunk;
20
+ uint64_t row_size;
21
+
22
+ // in bytes
23
+ row_chunk = (col_end - col_start + 3) / 4;
24
+ // in bytes
25
+ row_size = (ncols + 3) / 4;
26
+
27
+ r = row_start;
28
+ buff += r * row_size + col_start / 4;
29
+
30
+ while (r < row_end)
31
+ {
32
+ for (c = col_start; c < col_end;)
33
+ {
34
+ b = buff[(c - col_start) / 4];
35
+
36
+ b0 = b & 0x55;
37
+ b1 = (b & 0xAA) >> 1;
38
+
39
+ p0 = b0 ^ b1;
40
+ p1 = (b0 | b1) & b0;
41
+ p1 <<= 1;
42
+ p0 |= p1;
43
+ ce = MIN(c + 4, col_end);
44
+ for (; c < ce; ++c)
45
+ {
46
+ out[(r - row_start) * strides[0] +
47
+ (c - col_start) * strides[1]] = p0 & 3;
48
+ p0 >>= 2;
49
+ }
50
+ }
51
+ ++r;
52
+ buff += row_size;
53
+ }
54
+ }
@@ -0,0 +1,8 @@
1
+ /*
2
+ * Adapted from the `_bed_reader.h` script in the `pandas-plink` package.
3
+ * Source: https://github.com/limix/pandas-plink/blob/main/pandas_plink/_bed_reader.h
4
+ */
5
+
6
+ void read_fb_chunk(uint8_t *buff, uint64_t nrows, uint64_t ncols,
7
+ uint64_t row_start, uint64_t col_start, uint64_t row_end,
8
+ uint64_t col_end, uint8_t *out, uint64_t *strides);
@@ -0,0 +1,405 @@
1
+ """
2
+ Adapted from `_read.py` script in the `pandas-plink` package.
3
+ Source: https://github.com/limix/pandas-plink/blob/main/pandas_plink/_read.py
4
+ """
5
+ import warnings
6
+ from glob import glob
7
+ from os.path import basename, dirname, join
8
+ from collections import OrderedDict as odict
9
+ from typing import Optional, Callable, List, Tuple
10
+
11
+ from dask.array import Array
12
+ from pandas import DataFrame, read_csv
13
+
14
+ from chunk import Chunk
15
+ from fb_read import read_fb
16
+
17
+ __all__ = ["read_rfmix"]
18
+
19
+ def read_rfmix(
20
+ file_prefix: str, verbose: bool = True
21
+ ) -> Tuple[DataFrame, DataFrame, Array]:
22
+ """
23
+ Read RFMix files into data frames and a Dask array.
24
+
25
+ Notes
26
+ -----
27
+ Local ancestry can be either :const:`0`, :const:`1`, :const:`2`, or
28
+ :data:`math.nan`:
29
+
30
+ - :const:`0` No alleles are associated with this ancestry
31
+ - :const:`1` One allele is associated with this ancestry
32
+ - :const:`2` Both alleles are associated with this ancestry
33
+
34
+ Parameters
35
+ ----------
36
+ file_prefix : str
37
+ Path prefix to the set of RFMix files. It will load all of the chromosomes
38
+ at once.
39
+ verbose : bool, optional
40
+ :const:`True` for progress information; :const:`False` otherwise.
41
+
42
+ Returns
43
+ -------
44
+ loci : :class:`pandas.DataFrame`
45
+ Loci information for the FB data.
46
+ rf_q : :class:`pandas.DataFrame`
47
+ Global ancestry by chromosome from RFMix.
48
+ admix : :class:`dask.array.Array`
49
+ Local ancestry per population (columns pop1*nsamples ... popX*nsamples).
50
+ This is in order of the populations see `rf_q`.
51
+ """
52
+ from tqdm import tqdm
53
+ from pandas import concat
54
+ from dask.array import concatenate
55
+
56
+ # Get file prefixes
57
+ file_prefixes = sorted(glob(file_prefix))
58
+ if len(file_prefixes) == 1:
59
+ file_prefixes = sorted(glob(join(file_prefix, "*")))
60
+
61
+ file_prefixes = sorted(_clean_prefixes(file_prefixes))
62
+ fn = [{s: f"{fp}.{s}" for s in ["fb.tsv", "rfmix.Q"]} for fp in file_prefixes]
63
+
64
+ # Load loci information
65
+ pbar = tqdm(desc="Mapping loci files", total=len(fn), disable=not verbose)
66
+ loci = _read_file(fn, lambda f: _read_loci(f["fb.tsv"]), pbar)
67
+ pbar.close()
68
+ if len(file_prefixes) > 1 and verbose:
69
+ msg = "Multiple files read in this order:"
70
+ print(f"{msg} {[basename(f) for f in file_prefixes]}")
71
+
72
+ # Adjust loci indices and concatenate
73
+ nmarkers = {}
74
+ index_offset = 0
75
+ for i, bi in enumerate(loci):
76
+ nmarkers[fn[i]["fb.tsv"]] = bi.shape[0]
77
+ bi["i"] += index_offset
78
+ index_offset += bi.shape[0]
79
+ loci = concat(loci, axis=0, ignore_index=True)
80
+
81
+ # Load global ancestry per chromosome
82
+ pbar = tqdm(desc="Mapping Q files", total=len(fn), disable=not verbose)
83
+ rf_q = _read_file(fn, lambda f: _read_Q(f["rfmix.Q"]), pbar)
84
+ pbar.close()
85
+
86
+ nsamples = rf_q[0].shape[0]
87
+ pops = rf_q[0].drop(["sample_id", "chrom"], axis=1).columns.values
88
+ rf_q = concat(rf_q, axis=0, ignore_index=True)
89
+
90
+ # Loading local ancestry by loci
91
+ pbar = tqdm(desc="Mapping fb files", total=len(fn), disable=not verbose)
92
+ admix = _read_file(
93
+ fn,
94
+ lambda f: _read_fb(f["fb.tsv"], nsamples,
95
+ nmarkers[f["fb.tsv"]], pops, Chunk()),
96
+ pbar,
97
+ )
98
+ pbar.close()
99
+ admix = concatenate(admix, axis=0)
100
+
101
+ return loci, rf_q, admix
102
+
103
+
104
+ def _read_file(fn: List[str], read_func: Callable, pbar=None) -> List:
105
+ """
106
+ Read data from multiple files using a provided read function.
107
+
108
+ Parameters:
109
+ ----------
110
+ fn (List[str]): A list of file paths to read.
111
+ read_func (Callable): A function to read data from each file.
112
+ pbar (Optional): A progress bar object to update during reading.
113
+
114
+ Returns:
115
+ -------
116
+ List: A list containing the data read from each file.
117
+ """
118
+ data = [];
119
+ for file_name in fn:
120
+ data.append(read_func(file_name))
121
+ if pbar:
122
+ pbar.update(1)
123
+ return data
124
+
125
+
126
+ def _read_csv(fn: str, header: dict) -> DataFrame:
127
+ """
128
+ Read a CSV file into a pandas DataFrame with specified data types.
129
+
130
+ Parameters:
131
+ ----------
132
+ fn (str): The file path of the CSV file.
133
+ header (dict): A dictionary mapping column names to data types.
134
+
135
+ Returns:
136
+ -------
137
+ DataFrame: The data read from the CSV file as a pandas DataFrame.
138
+ """
139
+ try:
140
+ df = read_csv(
141
+ fn,
142
+ delim_whitespace=True,
143
+ header=None,
144
+ names=list(header.keys()),
145
+ dtype=header,
146
+ comment="#",
147
+ compression=None,
148
+ engine="c",
149
+ iterator=False,
150
+ )
151
+
152
+ except Exception as e:
153
+ raise IOError(f"Error reading file '{fn}': {e}")
154
+ # Validate that resulting DataFrame is correct type
155
+ if not isinstance(df, DataFrame):
156
+ raise ValueError(f"Expected a DataFrame but got {type(df)} instead.")
157
+
158
+ return df
159
+
160
+
161
+ def _read_tsv(fn: str) -> DataFrame:
162
+ """
163
+ Read a TSV file into a pandas DataFrame.
164
+
165
+ Parameters:
166
+ ----------
167
+ fn (str): File name of the TSV file.
168
+
169
+ Returns:
170
+ -------
171
+ DataFrame: DataFrame containing specified columns from the TSV file.
172
+ """
173
+ from numpy import int32
174
+ from pandas import StringDtype, concat
175
+
176
+ header = {"chromosome": StringDtype(), "physical_position": int32}
177
+ columns = ["chromosome", "physical_position"]
178
+
179
+ try:
180
+ chunks = read_csv(
181
+ fn,
182
+ delim_whitespace=True,
183
+ header=0,
184
+ usecols=columns,
185
+ dtype=header,
186
+ comment="#",
187
+ chunksize=100000, # Low memory chunks
188
+ )
189
+
190
+ # Concatenate chunks into single DataFrame
191
+ df = concat(chunks, ignore_index=True)
192
+ except FileNotFoundError:
193
+ raise FileNotFoundError(f"File {fn} not found.")
194
+ except Exception as e:
195
+ raise IOError(f"Error reading file {fn}: {e}")
196
+
197
+ # Validate that resulting DataFrame is correct type
198
+ if not isinstance(df, DataFrame):
199
+ raise ValueError(f"Expected a DataFrame but got {type(df)} instead.")
200
+
201
+ # Ensure DataFrame contains correct columns
202
+ if not all(column in df.columns for column in columns):
203
+ raise ValueError(f"DataFrame does not contain expected columns: {columns}")
204
+
205
+ return df
206
+
207
+
208
+ def _read_loci(fn: str) -> DataFrame:
209
+ """
210
+ Read loci information from a TSV file and add a sequential index column.
211
+
212
+ Parameters:
213
+ ----------
214
+ fn (str): The file path of the TSV file containing loci information.
215
+
216
+ Returns:
217
+ -------
218
+ DataFrame: A pandas DataFrame containing the loci information with an additional 'i' column for indexing.
219
+ """
220
+ df = _read_tsv(fn)
221
+ df["i"] = range(df.shape[0])
222
+ return df
223
+
224
+
225
+ def _read_Q(fn: str) -> DataFrame:
226
+ """
227
+ Read the Q matrix from a file and add the chromosome information.
228
+
229
+ Parameters:
230
+ ----------
231
+ fn (str): The file path of the Q matrix file.
232
+
233
+ Returns:
234
+ -------
235
+ DataFrame: The Q matrix with the chromosome information added.
236
+ """
237
+ from re import search
238
+
239
+ df = _read_Q_noi(fn)
240
+ match = search(r'chr(\d+)', fn)
241
+ if match:
242
+ chrom = match.group(0)
243
+ df["chrom"] = chrom
244
+ else:
245
+ print(f"Warning: Could not extract chromosome information from '{fn}'")
246
+ return df
247
+
248
+
249
+ def _read_Q_noi(fn: str) -> DataFrame:
250
+ """
251
+ Read the Q matrix from a file without adding chromosome information.
252
+
253
+ Parameters:
254
+ ----------
255
+ fn (str): The file path of the Q matrix file.
256
+
257
+ Returns:
258
+ -------
259
+ DataFrame: The Q matrix without chromosome information.
260
+ """
261
+ try:
262
+ header = odict(_types(fn))
263
+ return _read_csv(fn, header)
264
+ except Exception as e:
265
+ raise IOError(f"Error reading Q matrix from '{fn}': {e}")
266
+
267
+
268
+ def _read_fb(fn: str, nsamples: int, nloci: int, pops: list, chunk: Optional[Chunk] = None) -> Array:
269
+ """
270
+ Read the forward-backward matrix from a file as a Dask Array.
271
+
272
+ Parameters:
273
+ ----------
274
+ fn (str): The file path of the forward-backward matrix file.
275
+ nsamples (int): The number of samples in the dataset.
276
+ nloci (int): The number of loci in the dataset.
277
+ pops (list): A list of population labels.
278
+ chunk (Chunk, optional): A Chunk object specifying the chunk size for reading.
279
+
280
+ Returns:
281
+ -------
282
+ dask.array.Array: The forward-backward matrix as a Dask Array.
283
+ """
284
+ npops = len(pops)
285
+ nrows = nloci
286
+ ncols = nsamples * npops * 2
287
+ row_chunk = nrows if chunk.nloci is None else min(nrows, chunk.nloci)
288
+ col_chunk = ncols if chunk.nsamples is None else min(ncols, chunk.nsamples)
289
+ max_npartitions = 16_384
290
+ row_chunk = max(nrows // max_npartitions, row_chunk)
291
+ col_chunk = max(ncols // max_npartitions, col_chunk)
292
+ X = read_fb(fn, nrows, ncols, row_chunk, col_chunk)
293
+
294
+ # Subset populations and sum adjacent columns
295
+ return _subset_populations(X, npops)
296
+
297
+
298
+ def _subset_populations(X: Array, npops: int) -> Array:
299
+ """
300
+ Subset and process the input array X based on populations.
301
+
302
+ Parameters:
303
+ X (dask.array): Input array where columns represent data for different populations.
304
+ npops (int): Number of populations for column processing.
305
+
306
+ Returns:
307
+ dask.array: Processed array with adjacent columns summed for each population subset.
308
+ """
309
+ from dask.array import concatenate
310
+
311
+ pop_subset = []
312
+ pop_start = 0
313
+ ncols = X.shape[1]
314
+
315
+ if ncols % npops != 0:
316
+ raise ValueError("The number of columns in X must be divisible by npops.")
317
+
318
+ while pop_start < npops:
319
+ X0 = X[:, pop_start::npops] # Subset based on populations
320
+ if X0.shape[1] % 2 != 0:
321
+ raise ValueError("Number of columns must be even.")
322
+ X0_summed = X0[:, ::2] + X0[:, 1::2] # Sum adjacent columns
323
+ pop_subset.append(X0_summed)
324
+ pop_start += 1
325
+
326
+ return concatenate(pop_subset, 1, True)
327
+
328
+
329
+ def _types(fn: str) -> dict:
330
+ """
331
+ Infer the data types of columns in a TSV file.
332
+
333
+ Parameters:
334
+ ----------
335
+ fn (str) : File name of the TSV file.
336
+
337
+ Returns:
338
+ -------
339
+ dict : Dictionary mapping column names to their inferred data types.
340
+ """
341
+ from pandas import StringDtype
342
+
343
+ try:
344
+ # Read the first two rows of the file, skipping the first row
345
+ df = read_csv(
346
+ fn,
347
+ delim_whitespace=True,
348
+ nrows=2,
349
+ skiprows=1,
350
+ )
351
+
352
+ except FileNotFoundError:
353
+ raise FileNotFoundError(f"File '{fn}' not found.")
354
+ except Exception as e:
355
+ raise IOError(f"Error reading file '{fn}': {e}")
356
+
357
+ # Validate that the resulting DataFrame is of the correct type
358
+ if not isinstance(df, DataFrame):
359
+ raise ValueError(f"Expected a DataFrame but got {type(df)} instead.")
360
+
361
+ # Ensure the DataFrame contains at least one column
362
+ if df.shape[1] < 1:
363
+ raise ValueError("The DataFrame does not contain any columns.")
364
+
365
+ # Initialize the header dictionary with the sample_id column
366
+ header = {"sample_id": StringDtype()}
367
+ # Update the header dictionary with the data types of the remaining columns
368
+ header.update(df.dtypes[1:].to_dict())
369
+
370
+ return header
371
+
372
+
373
+ def _clean_prefixes(prefixes):
374
+ """
375
+ Clean and filter a list of file prefixes.
376
+
377
+ Parameters:
378
+ ----------
379
+ prefixes (list): A list of file prefixes (paths).
380
+
381
+ Returns:
382
+ -------
383
+ list: A list of unique, cleaned file prefixes without the file extensions.
384
+
385
+ Notes:
386
+ -----
387
+ - The function removes any prefixes that end with ".logs".
388
+ - It also removes any duplicate prefixes after cleaning.
389
+ """
390
+ cleaned_prefixes = []
391
+ for prefix in prefixes:
392
+ # Split the prefix into directory and base name
393
+ dir_path = dirname(prefix)
394
+ base_name = basename(prefix)
395
+
396
+ # Remove the file extensions from the base name
397
+ base = base_name.split(".")[0]
398
+
399
+ # Skip prefixes that end with ".logs"
400
+ if base != "logs":
401
+ cleaned_prefix = join(dir_path, base)
402
+ cleaned_prefixes.append(cleaned_prefix)
403
+
404
+ # Remove duplicate prefixes
405
+ return list(set(cleaned_prefixes))