multineuronchat 2025.11.10.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,159 @@
1
+ import os
2
+
3
+ import tempfile
4
+
5
+ import loompy
6
+
7
+ import numpy as np
8
+
9
+ from tqdm import tqdm
10
+
11
+ from .normalize import subject_wise_max_normalization
12
+ from .loompy_utils import create_empty_loom_file
13
+
14
+ from typing import Set, Optional, List
15
+
16
+
17
+ def filter_genes(
18
+ path_to_loom: str,
19
+ set_of_genes: Set[str],
20
+ gene_label_row: str = 'Gene',
21
+ path_to_filtered_loom: Optional[str] = '',
22
+ chunk_size: Optional[int] = -1,
23
+ verbose: Optional[bool] = False
24
+ ) -> str:
25
+ """
26
+ Filters out all genes that are not in the set_of_genes from the loom file at path_to_loom. The filtered loom file
27
+ is saved at path_to_filtered_loom. If path_to_filtered_loom is not provided, the filtered loom file is saved in the
28
+ same directory as the original loom file with the suffix '_GeneSubsampled.loom'.
29
+
30
+ If the system you are running this on has limited memory, you can set the chunk_size to a lower value to reduce the
31
+ memory usage. If the chunk_size is set to zero or a negative number, the complete matrix is loaded into memory.
32
+
33
+ :param path_to_loom: Path to the loom file that should be filtered
34
+ :param set_of_genes: Set of genes that should be kept in the filtered loom file
35
+ (usually provided by MultiNeuronChat)
36
+ :param gene_label_row: Name of the row attribute in the loom file that contains the gene labels (default: 'Gene')
37
+ :param path_to_filtered_loom: Path where the filtered loom file should be saved (default: same directory as the
38
+ original loom file with the suffix '_GeneSubsampled.loom')
39
+ :param chunk_size: Number of columns that are loaded into memory at once (default: -1)
40
+ :param verbose: If True, the progress is printed to the console (default: False)
41
+ :return: Path to the filtered loom file. If path_to_filtered_loom is provided, the same path is returned.
42
+ """
43
+
44
+ if not os.path.isfile(path_to_loom):
45
+ raise FileNotFoundError(f"File {path_to_loom} not found")
46
+
47
+ if not path_to_loom.endswith('.loom'):
48
+ raise ValueError('The file must be a loom file')
49
+
50
+ # if path_to_filtered_loom is not provided, save the filtered loom file in the same directory as the original loom
51
+ # file with the suffix '_GeneSubsampled.loom'
52
+ if path_to_filtered_loom == '' or path_to_filtered_loom is None:
53
+ path_to_filtered_loom: str = path_to_loom.replace('.loom', '_GeneSubsampled.loom')
54
+
55
+ with loompy.connect(path_to_loom, mode='r') as src:
56
+ n_rows, n_cols = src.shape
57
+
58
+ # if the chunk_size is set to zero or a negative number, we load the complete matrix into memory
59
+ if chunk_size <= 0:
60
+ chunk_size = n_cols
61
+
62
+ genes_to_keep_mask: List[bool] = [x in set_of_genes for x in src.ra[gene_label_row]]
63
+ genes_to_keep_idx: np.array = np.arange(n_rows)[genes_to_keep_mask]
64
+
65
+ n_rows_to_keep: int = len(genes_to_keep_idx)
66
+
67
+ row_attrs = {k: src.ra[k][genes_to_keep_idx] for k in src.row_attrs.keys()}
68
+ col_attrs = src.ca
69
+
70
+ # Create a new loom file with the filtered genes
71
+ create_empty_loom_file(
72
+ path_to_loom=path_to_filtered_loom,
73
+ shape=(n_rows_to_keep, n_cols),
74
+ row_attrs=row_attrs,
75
+ col_attrs=col_attrs
76
+ )
77
+
78
+ with loompy.connect(path_to_filtered_loom, mode='r+') as dst:
79
+ if verbose:
80
+ print('Filter the dataset for required genes:')
81
+
82
+ for i in tqdm(range(0, n_cols, chunk_size), disable=(not verbose)):
83
+ end_i: int = min(i + chunk_size, n_cols)
84
+
85
+ dst[:, i:end_i] = src[genes_to_keep_idx, i:end_i]
86
+
87
+ return path_to_filtered_loom
88
+
89
+
90
+ def gene_filter_and_subject_wise_normalize_dataset(
91
+ path_to_loom: str,
92
+ gene_set: Set[str],
93
+ subject_label_column: str,
94
+ path_to_normalized_loom: Optional[str] = '',
95
+ gene_label_row: str = 'Gene',
96
+ chunk_size: Optional[int] = -1,
97
+ tmp_path: Optional[str] = None,
98
+ verbose: Optional[bool] = False
99
+ ) -> str:
100
+ """
101
+ Filters out all genes that are not in the gene_set from the loom file at path_to_loom and performs a subject-wise
102
+ max normalization of the data. The normalized loom file is saved at path_to_normalized_loom.
103
+
104
+ If path_to_normalized_loom is not provided, the normalized loom file is saved in the same directory as the original
105
+ loom file with the suffix '_normalized.loom'.
106
+
107
+ If the system you are running this on has limited memory, you can set the chunk_size to a lower value to reduce the
108
+ memory usage. If the chunk_size is set to zero or a negative number, the complete matrix is loaded into memory.
109
+
110
+ :param path_to_loom: Path to the loom file that should be filtered and normalized
111
+ :param gene_set: Set of genes that should be kept in the filtered loom file
112
+ :param subject_label_column: Name of the column in the column attributes of the loom file that contains the subject
113
+ labels
114
+ :param path_to_normalized_loom: Path where the normalized loom file should be saved
115
+ (default: same directory as the original loom file with the suffix '_normalized.loom')
116
+ :param gene_label_row: Name of the row attribute in the loom file that contains the gene labels (default: 'Gene')
117
+ :param chunk_size: Number of columns that are loaded into memory at once (default: -1)
118
+ :param tmp_path: Path to the temporary directory where the filtered loom file is saved (default: system's temp dir)
119
+ :param verbose: If True, the progress is printed to the console (default: False)
120
+ :return: Path to the gene-filtered and subject-wise max-normalized loom file. If path_to_normalized_loom is provided,
121
+ the same path is returned.
122
+ """
123
+ if tmp_path is None:
124
+ tmp_path: str = tempfile.gettempdir()
125
+ else:
126
+ os.makedirs(tmp_path, exist_ok=True)
127
+
128
+ file_path: str = os.path.dirname(path_to_loom)
129
+ file_name: str = os.path.basename(path_to_loom)
130
+
131
+ gene_filtered_file_name: str = file_name.replace('.loom', '_filtered.loom')
132
+ gene_filtered_path: str = os.path.join(tmp_path, gene_filtered_file_name)
133
+
134
+ # if path_to_normalized_loom is not provided, save the normalized loom file in the same directory as the original
135
+ # loom file with the suffix '_normalized.loom'
136
+ if path_to_normalized_loom == '' or path_to_normalized_loom is None:
137
+ normalized_file_name: str = gene_filtered_file_name.replace('.loom', '_normalized.loom')
138
+ path_to_normalized_loom: str = os.path.join(file_path, normalized_file_name)
139
+
140
+ filter_genes(
141
+ path_to_loom=path_to_loom,
142
+ set_of_genes=gene_set,
143
+ path_to_filtered_loom=gene_filtered_path,
144
+ gene_label_row=gene_label_row,
145
+ chunk_size=chunk_size,
146
+ verbose=verbose
147
+ )
148
+
149
+ subject_wise_max_normalization(
150
+ path_to_loom=gene_filtered_path,
151
+ subject_label_column=subject_label_column,
152
+ path_to_normalized_loom=path_to_normalized_loom,
153
+ chunk_size=chunk_size
154
+ )
155
+
156
+ # remove tmp files
157
+ os.remove(gene_filtered_path)
158
+
159
+ return path_to_normalized_loom