NetAnalyzer 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- NetAnalyzer/__init__.py +11 -0
- NetAnalyzer/adv_mat_calc.py +106 -0
- NetAnalyzer/cli_manager.py +331 -0
- NetAnalyzer/graph2sim.py +106 -0
- NetAnalyzer/integration.py +165 -0
- NetAnalyzer/main_modules.py +610 -0
- NetAnalyzer/net_parser.py +78 -0
- NetAnalyzer/net_plotter.py +200 -0
- NetAnalyzer/netanalyzer.py +1257 -0
- NetAnalyzer/performancer.py +83 -0
- NetAnalyzer/ranker.py +353 -0
- NetAnalyzer/seed_parser.py +27 -0
- NetAnalyzer/templates/net_explorer.txt +83 -0
- NetAnalyzer/templates/network.txt +1 -0
- NetAnalyzer-1.0.0.dist-info/LICENSE.txt +21 -0
- NetAnalyzer-1.0.0.dist-info/METADATA +86 -0
- NetAnalyzer-1.0.0.dist-info/RECORD +20 -0
- NetAnalyzer-1.0.0.dist-info/WHEEL +5 -0
- NetAnalyzer-1.0.0.dist-info/entry_points.txt +8 -0
- NetAnalyzer-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
class Performancer:
|
|
2
|
+
def __init__(self):
|
|
3
|
+
self.control = {}
|
|
4
|
+
|
|
5
|
+
def load_control(self, ref_array):
|
|
6
|
+
for pair in ref_array:
|
|
7
|
+
node1, node2 = pair
|
|
8
|
+
if node2 != '-':
|
|
9
|
+
query = self.control.get(node1)
|
|
10
|
+
if query == None:
|
|
11
|
+
self.control[node1] = [node2]
|
|
12
|
+
else:
|
|
13
|
+
query.append(node2)
|
|
14
|
+
|
|
15
|
+
# Pandey 2007, Association Analysis-based Transformations for Protein Interaction Networks: A Function Prediction Case Study
|
|
16
|
+
def get_pred_rec(self, predictions, cut_number = 100, top_number = 10000):
|
|
17
|
+
preds, limits = self.load_prediction(predictions)
|
|
18
|
+
cuts = self.get_cuts(limits, cut_number)
|
|
19
|
+
performance = [] #cut, pred, rec
|
|
20
|
+
for cut in cuts:
|
|
21
|
+
prec, rec = self.pred_rec(preds, cut, top_number)
|
|
22
|
+
performance.append([cut, prec, rec])
|
|
23
|
+
return performance
|
|
24
|
+
|
|
25
|
+
def load_prediction(self, pairs_array):
|
|
26
|
+
pred = {}
|
|
27
|
+
min = None
|
|
28
|
+
max = None
|
|
29
|
+
for rec in pairs_array:
|
|
30
|
+
key, label, score = rec
|
|
31
|
+
if min != None and max != None:
|
|
32
|
+
if score < min: min = score
|
|
33
|
+
if score > max: max = score
|
|
34
|
+
else:
|
|
35
|
+
min = score; max = score
|
|
36
|
+
query = pred.get(key)
|
|
37
|
+
if query == None:
|
|
38
|
+
pred[key] = [[label], [score]]
|
|
39
|
+
else:
|
|
40
|
+
query[0].append(label)
|
|
41
|
+
query[1].append(score)
|
|
42
|
+
return pred, [min, max]
|
|
43
|
+
|
|
44
|
+
def pred_rec(self, preds, cut, top):
|
|
45
|
+
predicted_labels = 0 #m
|
|
46
|
+
true_labels = 0 #n
|
|
47
|
+
common_labels = 0 # k
|
|
48
|
+
for ctrl in self.control:
|
|
49
|
+
key, c_labels = ctrl
|
|
50
|
+
true_labels += len(c_labels) #n
|
|
51
|
+
pred_info = preds[key]
|
|
52
|
+
if pred_info != None:
|
|
53
|
+
labels, scores = pred_info
|
|
54
|
+
reliable_labels = self.get_reliable_labels(labels, scores, cut, top)
|
|
55
|
+
predicted_labels += len(reliable_labels) #m
|
|
56
|
+
common_labels += len(set(c_labels) & set(reliable_labels)) #k
|
|
57
|
+
if predicted_labels > 0:
|
|
58
|
+
prec = common_labels/predicted_labels
|
|
59
|
+
else:
|
|
60
|
+
prec = 0.0
|
|
61
|
+
|
|
62
|
+
if true_labels > 0:
|
|
63
|
+
rec = common_labels/true_labels
|
|
64
|
+
else:
|
|
65
|
+
rec = 0.0
|
|
66
|
+
|
|
67
|
+
return prec, rec
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def get_cuts(self, limits, n_cuts):
|
|
71
|
+
min, max = limits
|
|
72
|
+
span = (max - min)/n_cuts
|
|
73
|
+
cut = min
|
|
74
|
+
cuts = [ cut + n * span for n in range(n_cuts + 1) ]
|
|
75
|
+
return cuts
|
|
76
|
+
|
|
77
|
+
def get_reliable_labels(self, labels, scores, cut, top):
|
|
78
|
+
reliable_labels = [ [labels[i], score] for i, score in enumerable(scores) if score >= cut ]
|
|
79
|
+
def _(e):
|
|
80
|
+
return e[1]
|
|
81
|
+
reliable_labels.sort(reverse=True, key=_)
|
|
82
|
+
reliable_labels = [ pred[0] for pred in reliable_labels[0:top-1:1] ]
|
|
83
|
+
return reliable_labels
|
NetAnalyzer/ranker.py
ADDED
|
@@ -0,0 +1,353 @@
|
|
|
1
|
+
import numpy as np
|
|
2
|
+
import os
|
|
3
|
+
from itertools import combinations
|
|
4
|
+
from sklearn.model_selection import KFold, LeaveOneOut
|
|
5
|
+
from NetAnalyzer.adv_mat_calc import Adv_mat_calc
|
|
6
|
+
from NetAnalyzer.seed_parser import SeedParser
|
|
7
|
+
import py_exp_calc.exp_calc as pxc
|
|
8
|
+
|
|
9
|
+
class Ranker:
|
|
10
|
+
|
|
11
|
+
def __init__(self):
|
|
12
|
+
self.matrix = None
|
|
13
|
+
self.nodes = [] # kernel nodes
|
|
14
|
+
self.seeds = {} # node seeds
|
|
15
|
+
self.weights = {}
|
|
16
|
+
self.negatives = None # Could be 'all' or a dict with the negative nodes
|
|
17
|
+
self.seed_nodes2idx = {}
|
|
18
|
+
self.reference_nodes = {}
|
|
19
|
+
self.ranking = {}
|
|
20
|
+
self.discarded_seeds = {}
|
|
21
|
+
self.in_names = True
|
|
22
|
+
self.seed_presence = True
|
|
23
|
+
self.attributes = {"header": ["candidates", "score", "normalized_rank", "rank", "uniq_rank"]}
|
|
24
|
+
|
|
25
|
+
# Normalization and filtering matrix
|
|
26
|
+
def normalize_matrix(self, mode="by_column"):
|
|
27
|
+
degree_matrix = self.matrix.sum(0)
|
|
28
|
+
inv_degree_matrix = np.diag(1/degree_matrix)
|
|
29
|
+
if mode == "by_column":
|
|
30
|
+
self.matrix = self.matrix @ inv_degree_matrix
|
|
31
|
+
elif mode == "by_row":
|
|
32
|
+
self.matrix = inv_degree_matrix @ self.matrix
|
|
33
|
+
elif mode == "by_row_col":
|
|
34
|
+
self.matrix = np.sqrt(
|
|
35
|
+
inv_degree_matrix) @ self.matrix @ np.sqrt(inv_degree_matrix)
|
|
36
|
+
|
|
37
|
+
def filter_matrix(self, whitelist):
|
|
38
|
+
self.matrix, self.nodes, _ = Adv_mat_calc.filter_rowcols_by_whitelist(
|
|
39
|
+
self.matrix, self.nodes, self.nodes, whitelist, symmetric=True)
|
|
40
|
+
|
|
41
|
+
# loading and cleaning list of nodes
|
|
42
|
+
def load_nodes_from_file(self, file):
|
|
43
|
+
with open(file) as f:
|
|
44
|
+
for line in f:
|
|
45
|
+
self.nodes.append(line.rstrip())
|
|
46
|
+
|
|
47
|
+
def load_seeds(self, node_groups, sep=',', uniq=True):
|
|
48
|
+
self.seeds, self.weights = SeedParser.load_nodes_by_group(node_groups, sep=sep)
|
|
49
|
+
if uniq:
|
|
50
|
+
self.seeds = {seed: pxc.uniq(nodes) for seed, nodes in self.seeds.items()}
|
|
51
|
+
|
|
52
|
+
def load_negatives(self, negative_file, sep=','):
|
|
53
|
+
if os.path.exists(negative_file):
|
|
54
|
+
self.negatives, _ = SeedParser.load_node_groups_from_file(negative_file, sep=sep)
|
|
55
|
+
elif negative_file == "all":
|
|
56
|
+
self.negatives = negative_file
|
|
57
|
+
|
|
58
|
+
def load_references(self, node_groups, sep=','):
|
|
59
|
+
self.reference_nodes, _ = SeedParser.load_nodes_by_group(node_groups, sep=sep)
|
|
60
|
+
|
|
61
|
+
def clean_seeds(self, minimum_size = 1):
|
|
62
|
+
cleaned_seeds = {}
|
|
63
|
+
cleaned_weights = {}
|
|
64
|
+
for seed_name, seed in self.seeds.items():
|
|
65
|
+
cleaned_seed = [ node for node in seed if node in self.nodes ]
|
|
66
|
+
if len(cleaned_seed) < minimum_size:
|
|
67
|
+
self.discarded_seeds[seed_name] = seed
|
|
68
|
+
else:
|
|
69
|
+
cleaned_seeds[seed_name] = cleaned_seed
|
|
70
|
+
if self.weights.get(seed_name) is not None:
|
|
71
|
+
cleaned_weights[seed_name] = { node: self.weights[seed_name][node] for node in cleaned_seed }
|
|
72
|
+
self.seeds = cleaned_seeds
|
|
73
|
+
self.weights = cleaned_weights
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
# id <-> name translation
|
|
77
|
+
def get_nodes_indexes(self, nodes):
|
|
78
|
+
node_indxs = []
|
|
79
|
+
#if type(nodes) == int: nodes = [nodes]
|
|
80
|
+
for node in nodes:
|
|
81
|
+
index_node = self.seed_nodes2idx[node]
|
|
82
|
+
node_indxs.append(index_node)
|
|
83
|
+
|
|
84
|
+
return node_indxs
|
|
85
|
+
|
|
86
|
+
def get_seed_indexes(self):
|
|
87
|
+
indexes = {}
|
|
88
|
+
for node in sum(self.seeds.values(), []):
|
|
89
|
+
if indexes.get(node) is None:
|
|
90
|
+
indx = self.nodes.index(node)
|
|
91
|
+
indexes[node] = indx
|
|
92
|
+
self.seed_nodes2idx = indexes
|
|
93
|
+
|
|
94
|
+
def translate2idx(self):
|
|
95
|
+
self.in_names = False
|
|
96
|
+
for seed_name, seed in self.seeds.items():
|
|
97
|
+
self.seeds[seed_name] = self.get_nodes_indexes(seed)
|
|
98
|
+
for seed_name, node2weight in self.weights.items():
|
|
99
|
+
node2weight = { self.seed_nodes2idx[node]: weight for node, weight in node2weight.items()}
|
|
100
|
+
self.weights[seed_name] = node2weight
|
|
101
|
+
|
|
102
|
+
def translate2names(self):
|
|
103
|
+
self.in_names = True
|
|
104
|
+
for seed_name, seed in self.seeds.items():
|
|
105
|
+
self.seeds[seed_name] = [self.nodes[node] for node in seed]
|
|
106
|
+
for seed_name, node2weight in self.weights.items():
|
|
107
|
+
node2weight = { self.nodes[node]: weight for node, weight in node2weight.items()}
|
|
108
|
+
self.weights[seed_name] = node2weight
|
|
109
|
+
|
|
110
|
+
# Ranking
|
|
111
|
+
## Cross validation
|
|
112
|
+
|
|
113
|
+
def get_seed_cross_validation(self, k_fold=None):
|
|
114
|
+
new_seeds = {}
|
|
115
|
+
new_weights = {}
|
|
116
|
+
genes2predict = {}
|
|
117
|
+
|
|
118
|
+
for seed_name, seed in self.seeds.items():
|
|
119
|
+
if k_fold != None:
|
|
120
|
+
cv = KFold(n_splits=k_fold)
|
|
121
|
+
else:
|
|
122
|
+
cv = LeaveOneOut()
|
|
123
|
+
|
|
124
|
+
for indx, (train_index, test_index) in enumerate(cv.split(seed)):
|
|
125
|
+
seed_name_one_out = str(seed_name) + "_iteration_" + str(indx)
|
|
126
|
+
#print("train index is:", train_index)
|
|
127
|
+
new_seeds[seed_name_one_out] = [seed[i] for i in train_index]
|
|
128
|
+
|
|
129
|
+
if self.weights.get(seed_name):
|
|
130
|
+
new_weights[seed_name_one_out] = {node: self.weights[seed_name][node] for node in new_seeds[seed_name_one_out]}
|
|
131
|
+
|
|
132
|
+
if self.in_names:
|
|
133
|
+
genes2predict[seed_name_one_out] = [seed[i] for i in test_index]
|
|
134
|
+
else:
|
|
135
|
+
genes2predict[seed_name_one_out] = [seed[i] for i in test_index]
|
|
136
|
+
genes2predict[seed_name_one_out] = [self.nodes[idx] for idx in genes2predict[seed_name_one_out]]
|
|
137
|
+
if self.reference_nodes.get(seed_name) is not None:
|
|
138
|
+
genes2predict[seed_name_one_out] += self.reference_nodes[seed_name]
|
|
139
|
+
|
|
140
|
+
genes2predict[seed_name_one_out] = list(
|
|
141
|
+
set(genes2predict[seed_name_one_out]))
|
|
142
|
+
|
|
143
|
+
self.seeds = new_seeds
|
|
144
|
+
self.reference_nodes = genes2predict
|
|
145
|
+
self.weights = new_weights
|
|
146
|
+
|
|
147
|
+
def get_loo_ranking(self, seed_name, seed, metric = "mean"):
|
|
148
|
+
# Generate new seeds
|
|
149
|
+
# Get matrix values
|
|
150
|
+
cv = LeaveOneOut()
|
|
151
|
+
ncols= len(self.nodes)
|
|
152
|
+
nrows = len(seed)
|
|
153
|
+
W = np.zeros((nrows,ncols))
|
|
154
|
+
nodes2predict_pos = []
|
|
155
|
+
nodes2predict_names = []
|
|
156
|
+
new_seed_names = []
|
|
157
|
+
for indx, (train_index, test_index) in enumerate(cv.split(seed)):
|
|
158
|
+
new_seed_names.append(str(seed_name) + "_iteration_" + str(indx))
|
|
159
|
+
nodes = [seed[i] for i in train_index]
|
|
160
|
+
if self.weights.get(seed_name):
|
|
161
|
+
w = [self.weights[seed_name][node] for node in nodes]
|
|
162
|
+
W[indx, nodes] = w
|
|
163
|
+
else:
|
|
164
|
+
W[indx, nodes] = 1
|
|
165
|
+
#print(W)
|
|
166
|
+
node2predict = seed[test_index[0]]
|
|
167
|
+
nodes2predict_pos.append(node2predict)
|
|
168
|
+
nodes2predict_names.append(self.nodes[node2predict])
|
|
169
|
+
# recoger los valores correspondientes
|
|
170
|
+
# Calcular los score
|
|
171
|
+
if metric == "mean":
|
|
172
|
+
R = W @ self.matrix / (nrows - 1)
|
|
173
|
+
elif metric == "max":
|
|
174
|
+
R = np.zeros(sel.matrix.shape)
|
|
175
|
+
for row in range(0,self.matrix.shape[0]):
|
|
176
|
+
R[row] = (W[row,:] * sefl.matrix).max(0)
|
|
177
|
+
scores = R[range(0,nrows), nodes2predict_pos]
|
|
178
|
+
# Calcular los members_below
|
|
179
|
+
if self.seed_presence:
|
|
180
|
+
members_below = (R >= scores.reshape(nrows,1)).sum(1)
|
|
181
|
+
# Calcular los porcentage
|
|
182
|
+
rank_percentage = members_below/ncols
|
|
183
|
+
else:
|
|
184
|
+
R[W != 0] = np.min(scores) -1
|
|
185
|
+
members_below = (R >= scores.reshape(nrows, 1)).sum(1)
|
|
186
|
+
rank_percentage = members_below/(ncols-nrows+1)
|
|
187
|
+
# Calcular los absolute rank
|
|
188
|
+
return list(zip(new_seed_names, nodes2predict_names, scores, rank_percentage, members_below))
|
|
189
|
+
|
|
190
|
+
## Ranking by seed
|
|
191
|
+
def do_ranking(self, cross_validation=False, k_fold=None, propagate=False, metric = "mean", options={"tolerance": 1e-9, "iteration_limit": 100, "with_restart": 0}):
|
|
192
|
+
self.get_seed_indexes()
|
|
193
|
+
self.translate2idx()
|
|
194
|
+
|
|
195
|
+
if cross_validation and k_fold is not None:
|
|
196
|
+
self.get_seed_cross_validation(k_fold=k_fold)
|
|
197
|
+
|
|
198
|
+
ranked_lists = []
|
|
199
|
+
for seed_name, seed in self.seeds.items():
|
|
200
|
+
# The code in this block CANNOT modify nothing outside
|
|
201
|
+
if cross_validation and k_fold is None:
|
|
202
|
+
self.attributes["header"] = ["candidates", "score", "normalized_rank", "rank"]
|
|
203
|
+
rank_list = self.get_loo_ranking(seed_name, seed, metric=metric)
|
|
204
|
+
print(rank_list)
|
|
205
|
+
for row in rank_list:
|
|
206
|
+
ranked_lists.append([row[0], [list(row[1:])]])
|
|
207
|
+
else:
|
|
208
|
+
rank_list = self.rank_by_seed(seed, weights=self.weights.get(seed_name), propagate=propagate, metric = metric, options=options) # Production mode
|
|
209
|
+
if cross_validation and k_fold is not None:
|
|
210
|
+
rank_list = self.delete_seed_from_rank(rank_list, [self.nodes[pos] for pos in self.seeds[seed_name]])
|
|
211
|
+
ranked_lists.append([seed_name, rank_list])
|
|
212
|
+
|
|
213
|
+
for seed_name, rank_list in ranked_lists: # Transfer resuls to hash
|
|
214
|
+
self.ranking[seed_name] = rank_list
|
|
215
|
+
|
|
216
|
+
self.translate2names()
|
|
217
|
+
|
|
218
|
+
def rank_by_seed(self, seed, weights=None, propagate=False, metric = "mean", options={"tolerance": 1e-9, "iteration_limit": 100, "with_restart": 0}):
|
|
219
|
+
ordered_gene_score = []
|
|
220
|
+
genes_pos = seed
|
|
221
|
+
number_of_all_nodes = len(self.nodes)
|
|
222
|
+
if self.seed_presence:
|
|
223
|
+
genes_of_interest = list(range(0,number_of_all_nodes))
|
|
224
|
+
nodes = self.nodes
|
|
225
|
+
else:
|
|
226
|
+
genes_of_interest = list(set(range(0,number_of_all_nodes))-set(genes_pos))
|
|
227
|
+
nodes = [self.nodes[idx] for idx in genes_of_interest]
|
|
228
|
+
if weights: weights = np.array([weights[s] for s in seed])
|
|
229
|
+
number_of_seed_genes = len(genes_pos)
|
|
230
|
+
|
|
231
|
+
if number_of_seed_genes > 0:
|
|
232
|
+
gen_list = self.update_seed(
|
|
233
|
+
genes_pos, genes_of_interest, weights=weights, propagate=propagate, metric = metric, options=options)
|
|
234
|
+
ordered_gene_score = pxc.get_rank_metrics(gen_list, ids=nodes)
|
|
235
|
+
|
|
236
|
+
return ordered_gene_score
|
|
237
|
+
|
|
238
|
+
def update_seed(self, genes_pos, genes_of_interest, weights=None, propagate=False, metric = "mean", options={"tolerance": 1e-9, "iteration_limit": 100, "with_restart": 0}):
|
|
239
|
+
number_of_seed_genes = len(genes_pos)
|
|
240
|
+
number_of_all_nodes = len(self.nodes)
|
|
241
|
+
|
|
242
|
+
if propagate:
|
|
243
|
+
# TODO: Weight extension on this area
|
|
244
|
+
# TODO: This option soe not accept remove the seeds.
|
|
245
|
+
seed_vector = np.zeros((number_of_all_nodes))
|
|
246
|
+
seed_vector[genes_pos] = 1
|
|
247
|
+
updated_seed = self.propagate_seed(
|
|
248
|
+
self.matrix, seed_attr=seed_vector, tol=options["tolerance"], n=options["iteration_limit"], restart_factor=options["with_restart"])
|
|
249
|
+
gen_list = updated_seed[genes_of_interest]
|
|
250
|
+
else:
|
|
251
|
+
subsets_gen_values = self.matrix[genes_pos,:][:,genes_of_interest]
|
|
252
|
+
|
|
253
|
+
if metric == "max":
|
|
254
|
+
gen_list = subsets_gen_values.max(0)
|
|
255
|
+
elif metric == "mean":
|
|
256
|
+
if weights is not None:
|
|
257
|
+
integrated_gen_values = weights @ subsets_gen_values
|
|
258
|
+
gen_list = (1/weights.sum()) * integrated_gen_values
|
|
259
|
+
else:
|
|
260
|
+
integrated_gen_values = subsets_gen_values.sum(0)
|
|
261
|
+
gen_list = (1/number_of_seed_genes) * integrated_gen_values
|
|
262
|
+
return gen_list
|
|
263
|
+
|
|
264
|
+
def propagate_seed(self, matrix, seed_attr, tol=1e-6, n=1000, restart_factor=0):
|
|
265
|
+
seed_attr_old = seed_attr
|
|
266
|
+
error = tol + 1 # Now error is more than tol
|
|
267
|
+
k = 0
|
|
268
|
+
while error > tol and k < n:
|
|
269
|
+
if restart_factor > 0:
|
|
270
|
+
seed_attr_new = (1 - restart_factor) * \
|
|
271
|
+
matrix@seed_attr_old + restart_factor*seed_attr
|
|
272
|
+
else:
|
|
273
|
+
seed_attr_new = matrix @ seed_attr_old
|
|
274
|
+
# Normalization
|
|
275
|
+
seed_attr_new = seed_attr_new / seed_attr_new.sum(0)
|
|
276
|
+
# Take error
|
|
277
|
+
error = np.linalg.norm(seed_attr_new - seed_attr_old, ord=np.inf)
|
|
278
|
+
seed_attr_old = seed_attr_new
|
|
279
|
+
k += 1
|
|
280
|
+
|
|
281
|
+
if k >= n: # TODO, let see the error.
|
|
282
|
+
print("Convergence not achieved")
|
|
283
|
+
|
|
284
|
+
return seed_attr_old
|
|
285
|
+
|
|
286
|
+
# Output and parsing ranking
|
|
287
|
+
|
|
288
|
+
def get_filtered_ranks_by_reference(self):
|
|
289
|
+
filtered_ranked_genes = {}
|
|
290
|
+
|
|
291
|
+
for seed_name, ranking in self.ranking.items():
|
|
292
|
+
if self.reference_nodes.get(seed_name) is None or not ranking:
|
|
293
|
+
continue
|
|
294
|
+
|
|
295
|
+
ranking = self.array2hash(ranking, 0, range(0, len(ranking[0])))
|
|
296
|
+
references = self.reference_nodes[seed_name]
|
|
297
|
+
filtered_ranked_genes[seed_name] = []
|
|
298
|
+
|
|
299
|
+
for reference in references:
|
|
300
|
+
rank = ranking.get(reference)
|
|
301
|
+
if rank is not None:
|
|
302
|
+
filtered_ranked_genes[seed_name].append(rank)
|
|
303
|
+
|
|
304
|
+
filtered_ranked_genes[seed_name] = sorted(
|
|
305
|
+
filtered_ranked_genes[seed_name], key=lambda rank: -rank[1])
|
|
306
|
+
|
|
307
|
+
return filtered_ranked_genes
|
|
308
|
+
|
|
309
|
+
def write_ranking(self, output_name, add_header=True, top_n=None):
|
|
310
|
+
if add_header:
|
|
311
|
+
header = self.attributes["header"]
|
|
312
|
+
header.append("seed_group")
|
|
313
|
+
if top_n is not None:
|
|
314
|
+
rankings = self.get_top(top_n)
|
|
315
|
+
else:
|
|
316
|
+
rankings = self.ranking
|
|
317
|
+
with open(output_name, 'w') as f:
|
|
318
|
+
if add_header: f.write('\t'.join(header)+"\n")
|
|
319
|
+
for seed_name, ranking in rankings.items():
|
|
320
|
+
for row in ranking:
|
|
321
|
+
f.write('\t'.join(map(str,row)) + "\t" + f"{seed_name}" + "\n")
|
|
322
|
+
|
|
323
|
+
def add_candidate_tag_types(self):
|
|
324
|
+
rankings = {}
|
|
325
|
+
for seed_name, rankings_by_seed in self.ranking.items():
|
|
326
|
+
added_ranking_column = []
|
|
327
|
+
for ranking in rankings_by_seed:
|
|
328
|
+
if ranking[0] in self.seeds[seed_name]:
|
|
329
|
+
ranking.insert(5, "seed")
|
|
330
|
+
else:
|
|
331
|
+
ranking.insert(5, "new")
|
|
332
|
+
added_ranking_column.append(ranking)
|
|
333
|
+
rankings[seed_name] = added_ranking_column
|
|
334
|
+
self.ranking = rankings
|
|
335
|
+
self.attributes["header"].insert(5,"type")
|
|
336
|
+
|
|
337
|
+
def get_top(self, top_n):
|
|
338
|
+
top_ranked_genes = {}
|
|
339
|
+
for seed_name, ranking in self.ranking.items():
|
|
340
|
+
if ranking is not None:
|
|
341
|
+
top_ranked_genes[seed_name] = ranking[0:top_n]
|
|
342
|
+
return top_ranked_genes
|
|
343
|
+
|
|
344
|
+
# Auxiliar methods
|
|
345
|
+
def array2hash(self, arr, key, values): # 2exp?
|
|
346
|
+
h = {}
|
|
347
|
+
for els in arr:
|
|
348
|
+
h[els[0]] = [els[value] for value in values]
|
|
349
|
+
return h
|
|
350
|
+
|
|
351
|
+
def delete_seed_from_rank(self, rank_list, seed):
|
|
352
|
+
rank_list = [row for row in rank_list if row[0] not in seed]
|
|
353
|
+
return rank_list
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
import os
|
|
2
|
+
|
|
3
|
+
class SeedParser:
|
|
4
|
+
|
|
5
|
+
@staticmethod
|
|
6
|
+
def load_nodes_by_group(node_groups, sep= ','):
|
|
7
|
+
group_nodes = {}
|
|
8
|
+
weights_nodes = {}
|
|
9
|
+
if os.path.exists(node_groups):
|
|
10
|
+
group_nodes, weights_nodes = SeedParser.load_node_groups_from_file(node_groups, sep=sep)
|
|
11
|
+
else:
|
|
12
|
+
group_nodes = {"seed_genes": node_groups.split(sep)}
|
|
13
|
+
return group_nodes, weights_nodes
|
|
14
|
+
|
|
15
|
+
@staticmethod
|
|
16
|
+
def load_node_groups_from_file(file, sep=','):
|
|
17
|
+
group_nodes = {}
|
|
18
|
+
weights_nodes = {}
|
|
19
|
+
with open(file) as f:
|
|
20
|
+
for line in f:
|
|
21
|
+
fields = line.rstrip().split("\t")
|
|
22
|
+
set_name, nodes, *weights = fields
|
|
23
|
+
group_nodes[set_name] = nodes.split(sep)
|
|
24
|
+
if weights:
|
|
25
|
+
weights = weights[0].split(sep)
|
|
26
|
+
weights_nodes[set_name] = {group_nodes[set_name][i]: float(weight) for i, weight in enumerate(weights)} # TODO: check this section
|
|
27
|
+
return group_nodes, weights_nodes
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
<%
|
|
2
|
+
import pandas as pd
|
|
3
|
+
def plot_embdedding(data, plotter_list):
|
|
4
|
+
plotter_list["sns"].scatterplot(data=data[data['group_seed'] == 'background'], x='coord1', y='coord2', color='lightgrey', label='background', zorder=1)
|
|
5
|
+
plotter_list["sns"].scatterplot(data=df[df['group_seed'] != 'background'], x='coord1', y='coord2', hue='group_seed', palette='Set2', legend=True, zorder=2)
|
|
6
|
+
|
|
7
|
+
node2seed = {}
|
|
8
|
+
for seed, nodes in plotter.hash_vars["seeds2explore"].items():
|
|
9
|
+
for node in nodes:
|
|
10
|
+
if not node2seed.get(node): node2seed[node] = seed
|
|
11
|
+
%>
|
|
12
|
+
|
|
13
|
+
## Seeds LCC.
|
|
14
|
+
% if plotter.hash_vars.get("seeds2lcc"):
|
|
15
|
+
${plotter.create_title("LCC by seed", id='net_study', hlevel=1, indexable=True, clickable=False)}
|
|
16
|
+
<div style="overflow: hidden; display: flex; flex-direction: col; justify-content: center;">
|
|
17
|
+
<%
|
|
18
|
+
table_lcc = [["seed_name","net","lcc"]]
|
|
19
|
+
for seed, net_lcc in plotter.hash_vars["seeds2lcc"].items():
|
|
20
|
+
for net, lcc in net_lcc.items():
|
|
21
|
+
table_lcc.append([seed, net, lcc])
|
|
22
|
+
plotter.hash_vars["table_lcc"] = table_lcc
|
|
23
|
+
%>
|
|
24
|
+
${plotter.barplot(id='table_lcc', responsive= False, header=True,
|
|
25
|
+
fields = [0,2],
|
|
26
|
+
x_label = 'LCC size',
|
|
27
|
+
height = '400px', width= '400px',
|
|
28
|
+
var_attr = [0,1],
|
|
29
|
+
config = {
|
|
30
|
+
'showLegend' : True,
|
|
31
|
+
'colorBy' : 'seed_name',
|
|
32
|
+
'setMinX': 0,
|
|
33
|
+
"titleFontStyle": "italic",
|
|
34
|
+
"titleScaleFontFactor": 0.7,
|
|
35
|
+
"smpLabelScaleFontFactor": 1,
|
|
36
|
+
"axisTickScaleFontFactor": 0.7,
|
|
37
|
+
"segregateSamplesBy": "net"
|
|
38
|
+
})}
|
|
39
|
+
</div>
|
|
40
|
+
%endif
|
|
41
|
+
## Seeds subgraphs.
|
|
42
|
+
% if plotter.hash_vars.get("seeds2subgraph"):
|
|
43
|
+
<% plotter.set_header() %>
|
|
44
|
+
${plotter.create_title("Network study on every group of genes", id='net_study', hlevel=1, indexable=True, clickable=False)}
|
|
45
|
+
% for seed, net2subgraph in plotter.hash_vars["seeds2subgraph"].items():
|
|
46
|
+
${plotter.create_title(f"Group of genes: {seed}", id='net_study', hlevel=2, indexable=True, clickable=False)}
|
|
47
|
+
% for net, subgraph in net2subgraph.items():
|
|
48
|
+
${plotter.create_title(f"<b>{net}</b>", id='net_study', hlevel=3, indexable=True, clickable=False)}
|
|
49
|
+
<% key = net+"_"+seed
|
|
50
|
+
plotter.hash_vars[key] = {"graph": subgraph, "layers": ["layer", "layer"], "reference_nodes": plotter.hash_vars["seeds2explore"][seed], "group_nodes": []}
|
|
51
|
+
layout = "forcedir" if plotter.hash_vars["plot_method"] == "elgrapho" else None
|
|
52
|
+
%>
|
|
53
|
+
${plotter.network(id = key, method = plotter.hash_vars["plot_method"], layout=layout)}
|
|
54
|
+
%endfor
|
|
55
|
+
%endfor
|
|
56
|
+
%endif
|
|
57
|
+
|
|
58
|
+
% if plotter.hash_vars["net2embedding_proj"]:
|
|
59
|
+
${plotter.create_title("Network embedding and projection", id='net_emb', hlevel=1, indexable=True, clickable=False)}
|
|
60
|
+
% for graph, embedding in plotter.hash_vars["net2embedding_proj"].items():
|
|
61
|
+
${plotter.create_title(f"embedding in {graph}", id=f"net_emb_{graph}", hlevel=2, indexable=True, clickable=False)}
|
|
62
|
+
<%
|
|
63
|
+
coords, nodes = embedding
|
|
64
|
+
df = pd.DataFrame(coords, columns= ["coord1","coord2"])
|
|
65
|
+
df = df.assign(group_seed=[node2seed[node] if node2seed.get(node) else "background" for node in nodes])
|
|
66
|
+
plotter.hash_vars["net_proj"] = df
|
|
67
|
+
%>
|
|
68
|
+
${plotter.static_plot_main(id="net_proj", raw= True, plotting_function= lambda data, plotter_list: plot_embdedding(data, plotter_list))}
|
|
69
|
+
% endfor
|
|
70
|
+
% endif
|
|
71
|
+
|
|
72
|
+
## Net sims
|
|
73
|
+
% if plotter.hash_vars.get("net_sims"):
|
|
74
|
+
${plotter.create_title("Network similarity edge", id='net_edge', hlevel=1, indexable=True, clickable=False)}
|
|
75
|
+
<%
|
|
76
|
+
sim_nets, network_ids = plotter.hash_vars["net_sims"]
|
|
77
|
+
table_sims = [["",*network_ids]]
|
|
78
|
+
for i, net_id in enumerate(network_ids):
|
|
79
|
+
table_sims.append([net_id,*sim_nets[i,:].tolist()])
|
|
80
|
+
plotter.hash_vars["table_sims"] = table_sims
|
|
81
|
+
%>
|
|
82
|
+
${plotter.corplot(id = 'table_sims', header = True, row_names = True, correlationAxis = 'variables', config= {"correlationType":"circle"})}
|
|
83
|
+
%endif
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
${plotter.network(id = 'net_data', **build_options) }
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
The MIT License (MIT)
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2023 federedef
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: NetAnalyzer
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Python package for network analysis, operations and priorization.
|
|
5
|
+
Home-page: https://github.com/seoanezonjic/NetAnalyzer/
|
|
6
|
+
Author: seoanezonjic
|
|
7
|
+
Author-email: seoanezonjic@uma.es
|
|
8
|
+
License: MIT
|
|
9
|
+
Project-URL: Documentation, https://github.com/seoanezonjic/NetAnalyzer/
|
|
10
|
+
Platform: any
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Programming Language :: Python
|
|
13
|
+
Description-Content-Type: text/x-rst; charset=UTF-8
|
|
14
|
+
License-File: LICENSE.txt
|
|
15
|
+
Requires-Dist: NetworkX
|
|
16
|
+
Requires-Dist: numpy
|
|
17
|
+
Requires-Dist: scipy ==1.10.1
|
|
18
|
+
Requires-Dist: statsmodels
|
|
19
|
+
Requires-Dist: graphviz
|
|
20
|
+
Requires-Dist: mako
|
|
21
|
+
Requires-Dist: matplotlib
|
|
22
|
+
Requires-Dist: cdlib
|
|
23
|
+
Requires-Dist: scikit-learn
|
|
24
|
+
Requires-Dist: py-semtools
|
|
25
|
+
Requires-Dist: umap-learn
|
|
26
|
+
Requires-Dist: py-report-html
|
|
27
|
+
Requires-Dist: pecanpy
|
|
28
|
+
Requires-Dist: typing-extensions
|
|
29
|
+
Requires-Dist: py-cmdtabs
|
|
30
|
+
Requires-Dist: py-exp-calc
|
|
31
|
+
Requires-Dist: clusim
|
|
32
|
+
Requires-Dist: importlib-metadata ; python_version < "3.8"
|
|
33
|
+
Provides-Extra: testing
|
|
34
|
+
Requires-Dist: setuptools ; extra == 'testing'
|
|
35
|
+
Requires-Dist: pytest ; extra == 'testing'
|
|
36
|
+
Requires-Dist: pytest-cov ; extra == 'testing'
|
|
37
|
+
|
|
38
|
+
.. These are examples of badges you might want to add to your README:
|
|
39
|
+
please update the URLs accordingly
|
|
40
|
+
|
|
41
|
+
.. image:: https://api.cirrus-ci.com/github/<USER>/NetAnalyzer.svg?branch=main
|
|
42
|
+
:alt: Built Status
|
|
43
|
+
:target: https://cirrus-ci.com/github/<USER>/NetAnalyzer
|
|
44
|
+
.. image:: https://readthedocs.org/projects/NetAnalyzer/badge/?version=latest
|
|
45
|
+
:alt: ReadTheDocs
|
|
46
|
+
:target: https://NetAnalyzer.readthedocs.io/en/stable/
|
|
47
|
+
.. image:: https://img.shields.io/coveralls/github/<USER>/NetAnalyzer/main.svg
|
|
48
|
+
:alt: Coveralls
|
|
49
|
+
:target: https://coveralls.io/r/<USER>/NetAnalyzer
|
|
50
|
+
.. image:: https://img.shields.io/pypi/v/NetAnalyzer.svg
|
|
51
|
+
:alt: PyPI-Server
|
|
52
|
+
:target: https://pypi.org/project/NetAnalyzer/
|
|
53
|
+
.. image:: https://img.shields.io/conda/vn/conda-forge/NetAnalyzer.svg
|
|
54
|
+
:alt: Conda-Forge
|
|
55
|
+
:target: https://anaconda.org/conda-forge/NetAnalyzer
|
|
56
|
+
.. image:: https://pepy.tech/badge/NetAnalyzer/month
|
|
57
|
+
:alt: Monthly Downloads
|
|
58
|
+
:target: https://pepy.tech/project/NetAnalyzer
|
|
59
|
+
.. image:: https://img.shields.io/twitter/url/http/shields.io.svg?style=social&label=Twitter
|
|
60
|
+
:alt: Twitter
|
|
61
|
+
:target: https://twitter.com/NetAnalyzer
|
|
62
|
+
|
|
63
|
+
.. image:: https://img.shields.io/badge/-PyScaffold-005CA0?logo=pyscaffold
|
|
64
|
+
:alt: Project generated with PyScaffold
|
|
65
|
+
:target: https://pyscaffold.org/
|
|
66
|
+
|
|
67
|
+
|
|
|
68
|
+
|
|
69
|
+
===========
|
|
70
|
+
NetAnalyzer
|
|
71
|
+
===========
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
Python library for network analysis, operations and priorization.
|
|
75
|
+
|
|
76
|
+
This package is designed to perform various steps in network analysis and processing through a modular design. Key features include:
|
|
77
|
+
|
|
78
|
+
* Randomization: Enables randomization of both clustered and individual nodes or edges within networks.
|
|
79
|
+
* Projections: Simplifies network complexity by reducing the number of layers based on connections from an excluded layer. For example, it can transition from a Phenotype-Patient-Mutation network to a Patient-Mutation network, connecting layers based on common nodes between patients and mutations.
|
|
80
|
+
* Topological Analysis: Computes various topological metrics for nodes (e.g., degree, betweenness) and provides summary statistics for entire networks.
|
|
81
|
+
* Cluster analysis: Performs metrics on predefined clusters and applies clustering algorithms based on the cdlib library.
|
|
82
|
+
* Embedding of networks (Kernels and node2vec): Defines node similarity using methods for processing context information in networks, including classical Kernel approaches and node2vec. It also supports integration of multiple layers.
|
|
83
|
+
* Prioritization: Applies propagation algorithms to prioritize nodes based on similarity metrics, such as the adjacency matrix, and a set of seed nodes.
|
|
84
|
+
* Net plotting: Provides several tools for graphing networks from different net plotter packages (igraph, cytoscape, graphviz).
|
|
85
|
+
|
|
86
|
+
Please, cite this library as: Rojano E., Seoane-Zonjic P., Bueno-Amorós A., Perkins JR., and Ranea JAG. Revealing the Relationship Between Human Genome Regions and Pathological Phenotypes Through Network Analysis. Lecture Notes in Computer Science, DOI: 10.1007/978-3-319-56148-6_17.
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
NetAnalyzer/__init__.py,sha256=aZmS9IxJg31hnW_0-y1FOfHDKw3GatdL1IcDVMpQtks,501
|
|
2
|
+
NetAnalyzer/adv_mat_calc.py,sha256=e5qnSFf9C74ZPh8VIfEvnAO5M6YLRKSGWF-an_oTSKg,4074
|
|
3
|
+
NetAnalyzer/cli_manager.py,sha256=S6VTHuO2i-kV9pKxhC2iUgWb_gpJU0183GOTgUxD05U,22250
|
|
4
|
+
NetAnalyzer/graph2sim.py,sha256=_Ww_DgdMkjU8SDOm8ww3uv7i-h_2dc6ZwqsHObU2134,6545
|
|
5
|
+
NetAnalyzer/integration.py,sha256=24we4WBVo7F63PTydwBvT938DIJk4I-vgX7-TekPQUM,6343
|
|
6
|
+
NetAnalyzer/main_modules.py,sha256=jinn3F9UfvPHjQY4PEKC-vv7_haWpG9-4opqFL7cmWI,26244
|
|
7
|
+
NetAnalyzer/net_parser.py,sha256=Lgn7nS_J1RHGO2imiQzYmo6hd5zOTTBOYwU2_Ev91cs,3026
|
|
8
|
+
NetAnalyzer/net_plotter.py,sha256=fXoO6-3SQQbI_s_bOaHXFTJfGDLvg3ywAkieH7j_7Bo,8310
|
|
9
|
+
NetAnalyzer/netanalyzer.py,sha256=b1scKyvhwQhay3waBrLO4VxP6N-CnfTXOZFL11yMrn4,63194
|
|
10
|
+
NetAnalyzer/performancer.py,sha256=Mw3hDOrEGJBVoOfxIdv88YCxmg9UUMWWfYTAvmikNHA,2428
|
|
11
|
+
NetAnalyzer/ranker.py,sha256=eOSVH6GK0JaK2dXmcDCMSkgT0-7NPMdRIi_IpJD3isw,14978
|
|
12
|
+
NetAnalyzer/seed_parser.py,sha256=zFhapldMenhEYto0FIDlcsUYU0HZmm_hDuU5s8lLGuU,1056
|
|
13
|
+
NetAnalyzer/templates/net_explorer.txt,sha256=RdJBoK4ebxkJ6VJEcjUJHNeUgXS30xvw3nHMwYNAEqY,4146
|
|
14
|
+
NetAnalyzer/templates/network.txt,sha256=df1AaGbrWqmnN6MI-EGIpu_6LA82SdMiPXiFT3B5U1w,54
|
|
15
|
+
NetAnalyzer-1.0.0.dist-info/LICENSE.txt,sha256=o731a6_5f5RM3QnOR0gHLL2teNGoww7dsclPuUtm1NA,1076
|
|
16
|
+
NetAnalyzer-1.0.0.dist-info/METADATA,sha256=Q6K0DFfwDYT8jj9yHU-dhRHmrQU7IfuGtNB0quMuUuQ,4282
|
|
17
|
+
NetAnalyzer-1.0.0.dist-info/WHEEL,sha256=Wyh-_nZ0DJYolHNn1_hMa4lM7uDedD_RGVwbmTjyItk,91
|
|
18
|
+
NetAnalyzer-1.0.0.dist-info/entry_points.txt,sha256=UuP_GLgeGGdZEvljM0q9oqsCMUxFQS3fEabGfXfx9NI,416
|
|
19
|
+
NetAnalyzer-1.0.0.dist-info/top_level.txt,sha256=M2clp1DIUgApO2xUDyAUt49q5D8W37Sf_Vz5d_peO2o,12
|
|
20
|
+
NetAnalyzer-1.0.0.dist-info/RECORD,,
|