NetAnalyzer 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1257 @@
1
+ import random
2
+ import sys
3
+ import re
4
+ import copy
5
+ import networkx as nx
6
+ import math
7
+ import numpy as np
8
+ import scipy.stats as stats
9
+ import pandas as pd
10
+ import statsmodels.api as sm
11
+ import itertools
12
+ import warnings
13
+ import logging
14
+ from cdlib import algorithms, viz, evaluation
15
+ from cdlib import NodeClustering
16
+ import py_semtools # For external_data
17
+ from py_semtools import Ontology
18
+ from NetAnalyzer.adv_mat_calc import Adv_mat_calc
19
+ import py_exp_calc.exp_calc as pxc
20
+ from NetAnalyzer.net_plotter import Net_plotter
21
+ from NetAnalyzer.graph2sim import Graph2sim
22
+ from sklearn.preprocessing import StandardScaler
23
+ from sklearn.decomposition import PCA
24
+ from scipy.stats import zscore
25
+ # https://stackoverflow.com/questions/60392940/multi-layer-graph-in-networkx
26
+ # http://mkivela.com/pymnet
27
+
28
+ class NetAnalyzer:
29
+
30
+ def __init__(self, layers):
31
+ self.threads = 2
32
+ self.graph = nx.Graph() # Talk with PSZ, problem with directed graphs on loading edges.
33
+ self.layers = layers
34
+ self.association_values = {}
35
+ self.compute_autorelations = True
36
+ self.compute_pairs = 'conn'
37
+ self.matrices = {"adjacency_matrices": {}, # {layers => Mat, rowIds, colIds}
38
+ "kernels": {}, # {layers => {method_type => Mat, rowIds, colIds}}
39
+ "associations": {}, # {layers => {method_type => Mat, rowIds, colIds}}
40
+ "semantic_sims": {}, # {layers => {method_type => Mat, rowIds, colIds}}
41
+ }
42
+ self.embedding_coords = {}
43
+ self.group_nodes = {} # Communities are lists {community_id : [Node1, Node2,...]}
44
+ self.group_reference = {}
45
+ self.reference_nodes = []
46
+ self.loaded_obos = []
47
+ self.ontologies = []
48
+ self.layer_ontologies = {}
49
+
50
+ def __eq__(self, other): # https://igeorgiev.eu/python/tdd/python-unittest-assert-custom-objects-are-equal/
51
+ return nx.utils.misc.graphs_equal(self.graph, other.graph) and \
52
+ self.layers == other.layers and \
53
+ self.association_values == other.association_values and \
54
+ self.compute_autorelations == other.compute_autorelations and \
55
+ self.compute_pairs == other.compute_pairs and \
56
+ self.matrices == other.matrices and \
57
+ self.embedding_coords == other.embedding_coords and \
58
+ self.group_nodes == other.group_nodes and \
59
+ self.reference_nodes == other.reference_nodes and \
60
+ self.loaded_obos == other.loaded_obos and \
61
+ self.ontologies == other.ontologies and \
62
+ self.layer_ontologies == other.layer_ontologies
63
+
64
+ def clone(self):
65
+ network_clone = NetAnalyzer(copy.copy(self.layers))
66
+ network_clone.graph = copy.deepcopy(self.graph)
67
+ network_clone.association_values = self.association_values.copy()
68
+ network_clone.set_compute_pairs(self.compute_pairs, self.compute_autorelations)
69
+ network_clone.embedding_coords = self.embedding_coords.copy()
70
+ network_clone.matrices = self.matrices.copy()
71
+ network_clone.group_nodes = copy.deepcopy(self.group_nodes)
72
+ network_clone.reference_nodes = self.reference_nodes.copy()
73
+ network_clone.loaded_obos = self.loaded_obos.copy()
74
+ network_clone.ontologies = self.ontologies.deepcopy() if self.ontologies != [] else []
75
+ network_clone.layer_ontologies = self.layer_ontologies.deepcopy() if self.layer_ontologies != {} else {}
76
+ return network_clone
77
+
78
+ # THE PREVIOUS METHODS NEED TO DEFINE/ACCESS THE VERY SAME ATTRIBUTES, WATCH OUT ABOUT THIS !!!!!!!!!!!!!
79
+
80
+ def set_compute_pairs(self, use_pairs, get_autorelations):
81
+ self.compute_pairs = use_pairs
82
+ self.compute_autorelations = get_autorelations
83
+
84
+ def add_node(self, nodeID, layer):
85
+ self.graph.add_node(nodeID, layer=layer)
86
+
87
+ def add_edge(self, node1, node2, **attribs):
88
+ self.graph.add_edge(node1, node2, **attribs) # Talk with PSZ, problem with directed graphs on loading edges.
89
+
90
+ def set_layer(self, layer_definitions, node_name):
91
+ layer = None
92
+ if len(layer_definitions) > 1:
93
+ for layer_name, regexp in layer_definitions:
94
+ if re.search(regexp, node_name):
95
+ layer = layer_name
96
+ break
97
+ if layer == None: raise Exception("The node '" + node_name + "' not match with any layer regex")
98
+ else:
99
+ layer = layer_definitions[0][0]
100
+ if layer not in self.layers: self.layers.append(layer)
101
+ return layer
102
+
103
+ def set_groups(self, groups):
104
+ for group_id, nodes in groups.items():
105
+ for node in nodes:
106
+ if node in self.graph.nodes:
107
+ if self.group_nodes.get(group_id) is None:
108
+ self.group_nodes[group_id] = [node]
109
+ else:
110
+ self.group_nodes[group_id].append(node)
111
+ else:
112
+ #print("Group id: " + str(group_id) + " with member not in network:" + str(node), file=sys.stderr)
113
+ logging.warning("Group id: " + str(group_id) + " with member not in network: " + str(node))
114
+
115
+
116
+ def generate_adjacency_matrix(self, layerA, layerB):
117
+ layerAidNodes = [ node[0] for node in self.graph.nodes('layer') if node[1] == layerA]
118
+ layerBidNodes = [ node[0] for node in self.graph.nodes('layer') if node[1] == layerB]
119
+ matrix = np.zeros((len(layerAidNodes), len(layerBidNodes)))
120
+
121
+ has_weight = 'weight' if nx.get_edge_attributes(self.graph, 'weight') else None
122
+
123
+ if layerA == layerB:
124
+ # The method biadjacency matrix for this cases, fill the triangular upper matrix.
125
+ matrix_triu = np.array(nx.bipartite.biadjacency_matrix(self.graph, row_order=layerAidNodes, column_order=layerBidNodes, weight=has_weight, format='csr').todense())
126
+ matrix_tril = np.array(nx.bipartite.biadjacency_matrix(self.graph, row_order=layerBidNodes, column_order=layerAidNodes, weight=has_weight, format='csr').todense())
127
+ matrix = np.triu(matrix_triu) + np.tril(np.transpose(matrix_tril), k = -1)
128
+ else:
129
+ matrix = np.array(nx.bipartite.biadjacency_matrix(self.graph, row_order=layerAidNodes, column_order=layerBidNodes, weight=has_weight, format='csr').todense())
130
+
131
+ all_info_matrix = [matrix, layerAidNodes, layerBidNodes]
132
+
133
+ self.matrices["adjacency_matrices"][(layerA, layerB)] = all_info_matrix
134
+
135
+ return all_info_matrix
136
+
137
+ def generate_all_biadjs(self):
138
+ for layerA, layerB in itertools.product(self.layers, self.layers):
139
+ self.generate_adjacency_matrix(layerA, layerB)
140
+
141
+ def adjMat2netObj(self, layerA, layerB):
142
+ matrix, rowIds, colIds = self.matrices["adjacency_matrices"][(layerA, layerB)]
143
+
144
+ self.graph = nx.Graph()
145
+ for rowId in rowIds: self.add_node(rowId, layerA)
146
+ for colId in colIds: self.add_node(colId, layerB)
147
+
148
+ for rowPos, rowId in enumerate(rowIds):
149
+ for colPos, colId in enumerate(colIds):
150
+ associationValue = matrix[rowPos, colPos]
151
+ if associationValue > 0: self.graph.add_edge(rowId, colId, weight=associationValue)
152
+ return self.graph
153
+
154
+ def delete_nodes(self, node_list, mode='d'):
155
+ if mode == 'd':
156
+ self.graph.remove_nodes_from(node_list)
157
+ elif mode == 'r': # reverse selection
158
+ self.graph.remove_nodes_from(list(n for n in self.graph.nodes if n not in node_list ))
159
+
160
+ def get_connected_nodes(self, node_id, from_layer):
161
+ return [n for n in self.graph.neighbors(node_id) if self.graph.nodes[n]['layer'] == from_layer ]
162
+
163
+ def get_layers_as_dict(self, from_layers, to_layer):
164
+ relations = {}
165
+ from_nodes = self.get_nodes_layer(from_layers)
166
+ for fr_node in from_nodes:
167
+ relations[fr_node] = self.get_connected_nodes(fr_node, to_layer)
168
+ return relations
169
+
170
+ def link_ontology(self, ontology_file_path, layer_name):
171
+ if ontology_file_path not in self.loaded_obos: #Load new ontology
172
+ ontology = Ontology(file = ontology_file_path, load_file = True)
173
+ ontology.precompute()
174
+ ontology.threads = self.threads
175
+ self.loaded_obos.append(ontology_file_path)
176
+ self.ontologies.append(ontology)
177
+ else: #Link loaded ontology to current layer
178
+ ontology = self.ontologies[self.loaded_obos.index(ontology_file_path)]
179
+ self.layer_ontologies[layer_name] = ontology
180
+
181
+ def get_bipartite_subgraph(self, from_layer_node_ids, from_layer, to_layer):
182
+ bipartite_subgraph = {}
183
+ for from_layer_node_id in from_layer_node_ids:
184
+ connected_nodes = self.graph.neighbors(from_layer_node_id)
185
+ for connected_node in connected_nodes:
186
+ if self.graph.nodes[connected_node]['layer'] == to_layer:
187
+ query = bipartite_subgraph.get(connected_node)
188
+ if query == None:
189
+ bipartite_subgraph[connected_node] = self.get_connected_nodes(connected_node, from_layer)
190
+ return bipartite_subgraph
191
+
192
+ def get_nodes_by_attr(self, attrib, value):
193
+ return [nodeID for nodeID, attr in self.graph.nodes(data=True) if attr[attrib] == value]
194
+
195
+ def get_nodes_layer(self, layers):
196
+ nodes = []
197
+ for layer in layers:
198
+ nodes.extend(self.get_nodes_by_attr('layer', layer))
199
+ return nodes
200
+
201
+ def get_node_layer(self, node_id):
202
+ return self.graph.nodes(data=True)[node_id]['layer']
203
+
204
+ def get_edge_number(self):
205
+ return len(self.graph.edges())
206
+
207
+ def get_degree(self, zscore = True):
208
+ degree = dict(self.graph.degree())
209
+ if zscore:
210
+ degree = self.znormalize_dic_by_values(degree)
211
+ return degree
212
+
213
+ def get_betweenness_centrality(self, zscore = True):
214
+ betweenness_centrality = dict(nx.betweenness_centrality(self.graph))
215
+ if zscore:
216
+ betweenness_centrality = self.znormalize_dic_by_values(betweenness_centrality)
217
+ return betweenness_centrality
218
+
219
+
220
+ def znormalize_dic_by_values(self, dic2znormalize):
221
+ data = np.array([d for n, d in dic2znormalize.items()])
222
+ # data_z = Adv_mat_calc.zscore_normalize(data)
223
+ data_z = zscore(data, axis=None)
224
+ znormalized_dic = {}
225
+ count = 0
226
+ for n, d in dic2znormalize.items():
227
+ znormalized_dic[n] = data_z[count]
228
+ count += 1
229
+ dic2znormalize = znormalized_dic
230
+ return dic2znormalize
231
+
232
+
233
+ def collect_nodes(self, layers = 'all'):
234
+ nodeIDsA = []
235
+ nodeIDsB = []
236
+ if self.compute_autorelations: # TODO: remove the compute_autorrelations attrib.
237
+ if layers == 'all':
238
+ nodeIDsA = self.graph.nodes
239
+ else:
240
+ nodeIDsA = self.get_nodes_layer(layers)
241
+ else:
242
+ if layers != 'all': # layers contains two layer IDs
243
+ nodeIDsA = self.get_nodes_layer([layers[0]])
244
+ nodeIDsB = self.get_nodes_layer([layers[1]])
245
+ return nodeIDsA, nodeIDsB
246
+
247
+ def intersection(self, node1, node2):
248
+ shared_nodes = nx.common_neighbors(self.graph, node1, node2)
249
+ return shared_nodes
250
+
251
+ def get_all_intersections(self, layers = 'all'):
252
+ def _(node1, node2):
253
+ node_intersection = self.intersection(node1, node2)
254
+ return len(list(node_intersection))
255
+ intersection_lengths = self.get_all_pairs(_, layers = layers)
256
+ return intersection_lengths
257
+
258
+ def connections(self, ids_connected_to_n1, ids_connected_to_n2):
259
+ res = False
260
+ if ids_connected_to_n1 != None and ids_connected_to_n2 != None and len(ids_connected_to_n1 & ids_connected_to_n2) > 0 : # check that at least exists one node that connect to n1 and n2
261
+ res = True
262
+ return res
263
+
264
+ def get_all_pairs(self, pair_operation = None , layers = 'all'):
265
+ all_pairs = []
266
+ nodeIDsA, nodeIDsB = self.collect_nodes(layers = layers)
267
+ if pair_operation != None:
268
+ if self.compute_autorelations:
269
+ node_list = [ n for n in nodeIDsA] # Is this conversion needed?
270
+ while len(node_list) > 0:
271
+ node1 = node_list.pop(0)
272
+ if self.compute_pairs == 'all':
273
+ for node2 in node_list:
274
+ res = pair_operation(node1, node2)
275
+ all_pairs.append(res)
276
+ elif self.compute_pairs == 'conn':
277
+ ids_connected_to_n1 = set(self.graph.neighbors(node1))
278
+ for node2 in node_list:
279
+ ids_connected_to_n2 = set(self.graph.neighbors(node2))
280
+ if self.connections(ids_connected_to_n1, ids_connected_to_n2):
281
+ res = pair_operation(node1, node2)
282
+ all_pairs.append(res)
283
+ else:
284
+ if self.compute_pairs == 'conn': #MAIN METHOD
285
+ for node1 in nodeIDsA:
286
+ ids_connected_to_n1 = set(self.graph.neighbors(node1))
287
+ for node2 in nodeIDsB:
288
+ ids_connected_to_n2 = set(self.graph.neighbors(node2))
289
+ if self.connections(ids_connected_to_n1, ids_connected_to_n2):
290
+ res = pair_operation(node1, node2)
291
+ all_pairs.append(res)
292
+ elif self.compute_pairs == 'all':
293
+ raise NotImplementedError('Not implemented')
294
+
295
+ return all_pairs
296
+
297
+
298
+ ## association methods adjacency matrix based
299
+ #---------------------------------------------------------
300
+
301
+ def clean_autorelations_on_association_values(self):
302
+ for meth, values in self.association_values.items():
303
+ self.association_values[meth] = [relation for relation in values if self.graph.nodes[relation[0]]["layer"] != self.graph.nodes[relation[1]]["layer"]]
304
+
305
+ def get_association_values(self, layers, base_layer, meth, output_filename=None, outFormat='pair', add_to_object= False, **options): #TODO: Talk with PSZ about **optinos or options= {}
306
+
307
+ default_options = {"n_neighbors": 15, "min_dist": 0.1, "n_components": 2, "metric": 'euclidean',
308
+ "corr_type": "pearson", "pvalue": 0.05, "pvalue_adj_method": None, "alternative": 'greater', "coords2sim_type": 'dotProduct' }
309
+ default_options.update(options)
310
+
311
+ relations = [] #node A, node B, val
312
+ if meth == 'counts':
313
+ relations = self.get_counts_associations(layers, base_layer)
314
+ elif meth == 'jaccard': #all networks
315
+ relations = self.get_jaccard_associations(layers, base_layer)
316
+ elif meth == 'simpson': #all networks
317
+ relations = self.get_simpson_associations(layers, base_layer)
318
+ elif meth == 'geometric': #all networks
319
+ relations = self.get_geometric_associations(layers, base_layer)
320
+ elif meth == 'cosine': #all networks
321
+ relations = self.get_cosine_associations(layers, base_layer)
322
+ elif meth == 'pcc': #all networks
323
+ relations = self.get_pcc_associations(layers, base_layer)
324
+ elif meth == 'hypergeometric': #all networks
325
+ relations = self.get_hypergeometric_associations(layers, base_layer)
326
+ elif meth == 'hypergeometric_bf': #all networks
327
+ relations = self.get_hypergeometric_associations(layers, base_layer, pvalue_adj_method = 'bonferroni')
328
+ elif meth == 'hypergeometric_bh': #all networks
329
+ relations = self.get_hypergeometric_associations(layers, base_layer, pvalue_adj_method = 'benjamini_hochberg')
330
+ elif meth == 'csi': #all networks
331
+ relations = self.get_csi_associations(layers, base_layer)
332
+ elif meth == 'transference': #tripartite networks
333
+ relations = self.get_association_by_transference_resources(layers, base_layer)
334
+ elif meth == "correlation":
335
+ relations = self.get_corr_associations(layers, base_layer, corr_type = default_options["corr_type"], pvalue = default_options["pvalue"], pvalue_adj_method = default_options["pvalue_adj_method"], alternative = default_options["alternative"])
336
+ elif meth == "umap":
337
+ relations = self.get_umap_associations(layers, base_layer, n_neighbors = default_options["n_neighbors"], min_dist = default_options["min_dist"], n_components = default_options["n_components"], metric = default_options["metric"])
338
+ elif meth == "pca":
339
+ relations = self.get_pca_associations(layers, base_layer, n_components = default_options["n_components"], coords2sim_type = default_options["coords2sim_type"])
340
+ elif meth == "bicm":
341
+ relations = self.get_bicm_associations(layers, base_layer, pvalue = default_options["pvalue"])
342
+
343
+ if len(layers) == 1: layers = (layers[0], layers[0])
344
+ self.control_output(values = relations, output_filename=output_filename, inFormat="pair",
345
+ outFormat=outFormat, add_to_object=add_to_object, matrix_keys= ("associations", layers, meth))
346
+
347
+ return relations
348
+
349
+ def get_matrix_from_keys(self, matrix_keys, symm = True): # TODO: Implement this option in the code.
350
+ layers = matrix_keys[1]
351
+ matrix = pxc.dig(self.matrices,*matrix_keys)
352
+ if matrix is None and symm:
353
+ matrix_keys = list(matrix_keys)
354
+ matrix_keys[1] = (layers[1], layers[0])
355
+ matrix_keys = tuple(matrix_keys)
356
+ matrix = pxc.dig(self.matrices,*matrix_keys)
357
+ return matrix
358
+
359
+ def get_bicm_associations(self, layers, base_layer, pvalue = 0.05, pvalue_adj_method = 'fdr'):
360
+ # TODO: Need test.
361
+ biadj_matrix = pxc.dig(self.matrices, "adjacency_matrices",tuple(layers),base_layer)
362
+ if biadj_matrix is None:
363
+ biadj_matrix = self.generate_adjacency_matrix(*layers, base_layer)
364
+
365
+ matrix, rowIds, _ = biadj_matrix
366
+
367
+ from bicm.graph_classes import BipartiteGraph
368
+ miGraph = BipartiteGraph(biadjacency=matrix)
369
+ relations = miGraph.get_rows_projection(
370
+ alpha=pvalue,
371
+ method='poisson',
372
+ progress_bar=False,
373
+ fmt='edgelist',
374
+ validation_method=pvalue_adj_method)
375
+ relations = [[rowIds[relation[0]], rowIds[relation[1]], 1] for relation in relations] # TODO: Check 0-based numeration.
376
+ return relations
377
+
378
+
379
+ def get_corr_associations(self, layers, base_layer, corr_type = "pearson", pvalue = 0.05, pvalue_adj_method = None, alternative = 'greater'):
380
+ biadj_matrix = pxc.dig(self.matrices,"adjacency_matrices",tuple(layers),base_layer)
381
+ if biadj_matrix is None:
382
+ biadj_matrix = self.generate_adjacency_matrix(*layers, base_layer)
383
+
384
+ matrix, rowIds, _ = biadj_matrix
385
+
386
+ if corr_type == "pearson":
387
+ corr_mat, corr_pvalue = pxc.get_corr(x=matrix.T, alternative=alternative, corr_type = "pearson")
388
+ elif corr_type == "spearman":
389
+ corr_mat, corr_pvalue = pxc.get_corr(x=matrix.T, alternative= alternative, corr_type = "spearman")
390
+
391
+ relations = pxc.matrixes2pairs([corr_pvalue, corr_mat], rowIds, rowIds, symm = True)
392
+ relations = [ relation for relation in relations if relation[0] != relation[1] and not np.isnan(relation[2]) and relation[3] != 0 ]
393
+ if pvalue_adj_method is not None: self.adjust_pval_association(relations, pvalue_adj_method)
394
+ relations = [[relation[0], relation[1], relation[3]] for relation in relations if relation[2] < pvalue]
395
+ return relations
396
+
397
+ def get_umap_associations(self, layers, base_layer, n_neighbors = 15, min_dist = 0.1, n_components = 2, metric = 'euclidean', random_seed = None):
398
+ biadj_matrix = pxc.dig(self.matrices,"adjacency_matrices",tuple(layers),base_layer)
399
+ if biadj_matrix is None:
400
+ biadj_matrix = self.generate_adjacency_matrix(*layers, base_layer)
401
+ data, rowIds, _ = biadj_matrix
402
+ umap_coords = Adv_mat_calc.data2umap(data, n_neighbors=n_neighbors, min_dist=min_dist, n_components=n_components, metric=metric, random_seed = random_seed)
403
+ umap_sims = np.triu(pxc.coords2sim(umap_coords, sim="euclidean"), k = 1)
404
+ relations = pxc.matrix2relations(umap_sims, rowIds, rowIds)
405
+ relations = [relation for relation in relations if relation[2] != 0]
406
+ return relations
407
+
408
+ def get_pca_associations(self, layers, base_layer, n_components = 2, coords2sim_type = "dotProduct"):
409
+ # TODO: Try to select the correct number of n_componentes (automatically)
410
+ biadj_matrix = pxc.dig(self.matrices,"adjacency_matrices",tuple(layers),base_layer)
411
+ if biadj_matrix is None:
412
+ biadj_matrix = self.generate_adjacency_matrix(*layers, base_layer)
413
+
414
+ matrix, rowIds, _ = biadj_matrix
415
+
416
+ x = StandardScaler().fit_transform(matrix)
417
+ pca = PCA(n_components=n_components)
418
+ pca_coords = pca.fit_transform(x)
419
+ pca_matrix_sim = pxc.coords2sim(pca_coords, sim = coords2sim_type)
420
+ relations = pxc.matrix2relations(pca_matrix_sim , rowIds, rowIds)
421
+ return relations
422
+
423
+
424
+ def get_association_by_transference_resources(self, firstPairLayers, secondPairLayers, lambda_value1 = 0.5, lambda_value2 = 0.5):
425
+ relations = []
426
+ matrix1 = self.matrices["adjacency_matrices"][firstPairLayers][0]
427
+ matrix2 = self.matrices["adjacency_matrices"][secondPairLayers][0]
428
+ finalMatrix = Adv_mat_calc.tranference_resources(matrix1, matrix2, lambda_value1 = lambda_value1, lambda_value2 = lambda_value2)
429
+ rowIds = self.matrices["adjacency_matrices"][firstPairLayers][1]
430
+ colIds = self.matrices["adjacency_matrices"][secondPairLayers][2]
431
+ relations = pxc.matrix2relations(finalMatrix, rowIds, colIds)
432
+ self.association_values['transference'] = relations
433
+ return relations
434
+
435
+ def get_associations(self, layers, base_layer, compute_association): # BASE METHOD
436
+ base_nodes = set(self.get_nodes_layer([base_layer]))
437
+ def _(node1, node2):
438
+ associatedIDs_node1 = set(self.graph.neighbors(node1))
439
+ associatedIDs_node2 = set(self.graph.neighbors(node2))
440
+ intersectedIDs = (associatedIDs_node1 & associatedIDs_node2) & base_nodes
441
+ associationValue = compute_association(associatedIDs_node1, associatedIDs_node2, intersectedIDs, node1, node2)
442
+ return [node1, node2, associationValue]
443
+ associations = self.get_all_pairs(_, layers = layers)
444
+ return associations
445
+
446
+ #https://stackoverflow.com/questions/55063978/ruby-like-yield-in-python-3
447
+ def get_counts_associations(self, layers, base_layer):
448
+ def _(associatedIDs_node1, associatedIDs_node2, intersectedIDs, node1, node2):
449
+ return len(intersectedIDs)
450
+ relations = self.get_associations(layers, base_layer, _)
451
+ self.association_values['counts'] = relations
452
+ return relations
453
+
454
+ def get_jaccard_associations(self, layers, base_layer):
455
+ def _(associatedIDs_node1, associatedIDs_node2, intersectedIDs, node1, node2):
456
+ unionIDS = associatedIDs_node1 | associatedIDs_node2
457
+ return len(intersectedIDs)/len(unionIDS)
458
+ relations = self.get_associations(layers, base_layer, _)
459
+ self.association_values['jaccard'] = relations
460
+ return relations
461
+
462
+ def get_simpson_associations(self, layers, base_layer):
463
+ def _(associatedIDs_node1, associatedIDs_node2, intersectedIDs, node1, node2):
464
+ minLength = min([len(associatedIDs_node1), len(associatedIDs_node2)])
465
+ return len(intersectedIDs)/minLength
466
+ relations = self.get_associations(layers, base_layer, _)
467
+ self.association_values['simpson'] = relations
468
+ return relations
469
+
470
+ def get_geometric_associations(self, layers, base_layer):
471
+ #wang 2016 method
472
+ def _(associatedIDs_node1, associatedIDs_node2, intersectedIDs, node1, node2):
473
+ intersectedIDs = len(intersectedIDs)**2
474
+ productLength = math.sqrt(len(associatedIDs_node1) * len(associatedIDs_node2))
475
+ return intersectedIDs/productLength
476
+ relations = self.get_associations(layers, base_layer, _)
477
+ self.association_values['geometric'] = relations
478
+ return relations
479
+
480
+ def get_cosine_associations(self, layers, base_layer):
481
+ def _(associatedIDs_node1, associatedIDs_node2, intersectedIDs, node1, node2):
482
+ productLength = math.sqrt(len(associatedIDs_node1) * len(associatedIDs_node2))
483
+ return len(intersectedIDs)/productLength
484
+ relations = self.get_associations(layers, base_layer, _)
485
+ self.association_values['cosine'] = relations
486
+ return relations
487
+
488
+ def get_pcc_associations(self, layers, base_layer, weighted = False ):
489
+ #for Ny calcule use get_nodes_layer
490
+ base_layer_nodes = self.get_nodes_layer([base_layer])
491
+ ny = len(base_layer_nodes)
492
+ def _(associatedIDs_node1, associatedIDs_node2, intersectedIDs, node1, node2):
493
+ intersProd = len(intersectedIDs) * ny
494
+ nodesProd = len(associatedIDs_node1) * len(associatedIDs_node2)
495
+ nodesSubs = intersProd - nodesProd
496
+ nodesAInNetwork = ny - len(associatedIDs_node1)
497
+ nodesBInNetwork = ny - len(associatedIDs_node2)
498
+ return np.float64(nodesSubs) / math.sqrt(nodesProd * nodesAInNetwork * nodesBInNetwork) # TODO: np.float64 is used to handle division by 0. Fix the implementation/test to avoid this case
499
+ relations = self.get_associations(layers, base_layer, _)
500
+ self.association_values['pcc'] = relations
501
+ return relations
502
+
503
+ def get_csi_associations(self, layers, base_layer):
504
+ pcc_relations = self.get_pcc_associations(layers, base_layer)
505
+ pcc_relations = [row for row in pcc_relations if not math.isnan(row[2])]
506
+ if len(layers) > 1:
507
+ self.clean_autorelations_on_association_values()
508
+
509
+ nx = len(self.get_nodes_layer(layers))
510
+ pcc_vals = {}
511
+ node_rels = {}
512
+
513
+ for node1, node2, assoc_index in pcc_relations:
514
+ pxc.add_nested_value(pcc_vals, (node1, node2), np.abs(assoc_index))
515
+ pxc.add_nested_value(pcc_vals, (node2, node1), np.abs(assoc_index))
516
+ pxc.add_record(node_rels, node1, node2)
517
+ pxc.add_record(node_rels, node2, node1)
518
+
519
+ relations = []
520
+ for node1, node2, assoc_index in pcc_relations:
521
+ pccAB = assoc_index - 0.05
522
+ valid_nodes = 0
523
+
524
+ significant_nodes_from_node1 = set([node for node in node_rels[node1] if pcc_vals[node1][node] >= pccAB])
525
+ significant_nodes_from_node2 = set([node for node in node_rels[node2] if pcc_vals[node2][node] >= pccAB])
526
+ all_significant_nodes = significant_nodes_from_node2 | significant_nodes_from_node1
527
+ all_nodes = set(node_rels[node1]) | set(node_rels[node2])
528
+
529
+ csiValue = 1 - (len(all_significant_nodes))/(len(all_nodes))
530
+ relations.append([node1, node2, csiValue])
531
+
532
+ self.association_values['csi'] = relations
533
+ return relations
534
+
535
+ def get_hypergeometric_associations(self, layers, base_layer, pvalue_adj_method= None):
536
+ ny = len(self.get_nodes_layer([base_layer]))
537
+ def _(associatedIDs_node1, associatedIDs_node2, intersectedIDs, node1, node2):
538
+ # Analogous formulation with stats.fisher_exact(data, alternative='greater')
539
+ intersection_lengths = len(intersectedIDs)
540
+ if intersection_lengths > 0:
541
+ n1_items = len(associatedIDs_node1)
542
+ n2_items = len(associatedIDs_node2)
543
+ p_value = stats.hypergeom.sf(intersection_lengths-1, ny, n1_items, n2_items)
544
+
545
+ return p_value
546
+ relations = self.get_associations(layers, base_layer, _)
547
+
548
+ if pvalue_adj_method == 'bonferroni':
549
+ meth = 'hypergeometric_bf'
550
+ self.adjust_pval_association(relations, 'bonferroni')
551
+ elif pvalue_adj_method == 'benjamini_hochberg':
552
+ meth = 'hypergeometric_bh'
553
+ self.adjust_pval_association(relations, 'fdr_bh')
554
+ else:
555
+ meth = 'hypergeometric'
556
+ relations = [[assoc[0], assoc[1], -np.log10(assoc[2])] for assoc in relations if assoc[2] > 0]
557
+ self.association_values[meth] = relations
558
+ return relations
559
+
560
+ def adjust_pval_association(self, associations, method): # TODO TEST
561
+ pvals = np.array([val[2] for val in associations])
562
+ adj_pvals = sm.stats.multipletests(pvals, method=method, is_sorted=False, returnsorted=False)[1] #2expcalc?
563
+ for idx, adj_pval in enumerate(adj_pvals):
564
+ associations[idx][2] = adj_pval
565
+
566
+ ## filter methods
567
+ #----------------
568
+
569
+ def get_filter(self, layers, method="cutoff", options={}):
570
+ default_options = {"cutoff": None, "compute_autorelations": False, "binarize": False}
571
+ default_options.update(options)
572
+
573
+ if method == "cutoff":
574
+ filtered_function = self.filter_cutoff
575
+ else:
576
+ raise Exception('Not defined method')
577
+
578
+ edges_with_filtered_values = []
579
+ for layers_pairs in itertools.pairwise(layers):
580
+ edges_with_filtered_values += filtered_function(layers_pairs, cutoff= default_options["cutoff"], compute_autorelations = default_options["compute_autorelations"])
581
+
582
+ if default_options["binarize"] == True:
583
+ edges_with_filtered_values = [[edge[0], edge[1], float(edge[2]>0)] for edge in edges_with_filtered_values]
584
+
585
+ self.update_edges_with(edges_with_filtered_values)
586
+
587
+
588
+ def filter_cutoff(self, layers, cutoff=0.5, compute_autorelations= False):
589
+ edges_with_filtered_values = []
590
+
591
+ def _(node1, node2):
592
+ edges_attr = self.graph.edges()[node1,node2]
593
+ weight = edges_attr["weight"] if edges_attr.get("weight") else 1
594
+ if weight < cutoff:
595
+ weight = 0
596
+ return [node1, node2, weight]
597
+
598
+ edges_with_filtered_values = self.get_direct_conns(_, layers = layers, compute_autorelations= compute_autorelations)
599
+
600
+ return edges_with_filtered_values
601
+
602
+ def write_subgraph(self, layers, output_filename, outFormat='pair'):
603
+ if outFormat == "pair":
604
+ nodes = [node for node, data in self.graph.nodes(data=True) if data.get("layer") in layers]
605
+ subgraph = self.graph.subgraph(nodes)
606
+ with open(output_filename, "w") as f:
607
+ for nodeA, nodeB, data in subgraph.edges(data=True):
608
+ weight = data.get("weight")
609
+ if weight is not None:
610
+ f.write(f"{nodeA}\t{nodeB}\t{str(weight)}" + "\n")
611
+ else:
612
+ f.write(f"{nodeA}\t{nodeB}" + "\n")
613
+ elif outFormat == "matrix":
614
+ if self.matrices["adjacency_matrices"].get(layers) is None:
615
+ self.generate_adjacency_matrix(layers[0], layers[1])
616
+ matrix, rowIds, colIds = pxc.dig(self.matrices,"adjacency_matrices", layers)
617
+ self.write_obj(matrix, output_filename=output_filename, Format=outFormat, rowIds=rowIds, colIds=colIds)
618
+
619
+
620
+ def get_direct_conns(self, pair_operation = None, layers = None, compute_autorelations = False):
621
+ direct_edges = []
622
+ nodeIDsA = self.get_nodes_layer([layers[0]])
623
+ if layers[0] == layers[1]:
624
+ nodeIDsB = nodeIDsA
625
+ else:
626
+ nodeIDsB = self.get_nodes_layer([layers[1]])
627
+
628
+
629
+ if compute_autorelations:
630
+ all_nodes = list(set(nodeIDsA).union(set(nodeIDsB)))
631
+ for nodeA, nodeB in itertools.combinations(all_nodes,2): # Watchout! Just valids for non directed graphs
632
+ if self.graph.has_edge(nodeA, nodeB):
633
+ res = pair_operation(nodeA, nodeB)
634
+ direct_edges.append(res)
635
+ else:
636
+ for nodeA in nodeIDsA:
637
+ for nodeB in nodeIDsB:
638
+ if self.graph.has_edge(nodeA, nodeB):
639
+ res = pair_operation(nodeA, nodeB)
640
+ direct_edges.append(res)
641
+
642
+ return direct_edges
643
+
644
+
645
+ def update_edges_with(self, relations):
646
+ for nodeA, nodeB, weight in relations:
647
+ if weight > 0:
648
+ self.graph.add_edge(nodeA, nodeB, weight=weight)
649
+ elif self.graph.has_edge(nodeA, nodeB):
650
+ self.graph.remove_edge(nodeA, nodeB)
651
+
652
+
653
+ ## Kernel and similarity methods
654
+ #------------------------------------
655
+
656
+ def get_kernel(self, layers, method, normalization=False, sim_type= "dotProduct", embedding_kwargs={}, output_filename=None, outFormat='matrix', add_to_object= False):
657
+ #embedding_kwargs accept: dimensions, walk_length, num_walks, p, q, workers, window, min_count, seed, quiet, batch_words
658
+
659
+ if method in Graph2sim.allowed_embeddings:
660
+ adj_mat, embedding_nodes, _ = self.matrices["adjacency_matrices"][(layers[0],layers[0])]
661
+ emb_coords = Graph2sim.get_embedding(adj_mat, embedding = method, embedding_nodes=embedding_nodes, **embedding_kwargs)
662
+ kernel = Graph2sim.emb_coords2kernel(emb_coords, normalization, sim_type= sim_type)
663
+ rowIds = embedding_nodes
664
+ colIds = rowIds
665
+ elif method[0:2] in Graph2sim.allowed_kernels:
666
+ adj_mat, rowIds, colIds = self.matrices["adjacency_matrices"][(layers[0],layers[0])]
667
+ kernel = Graph2sim.get_kernel(adj_mat, method, normalization=normalization)
668
+
669
+ self.control_output(values = kernel, rowIds = rowIds, colIds = colIds, output_filename=output_filename, inFormat="matrix",
670
+ outFormat=outFormat, add_to_object=add_to_object, matrix_keys= ("kernels", layers, method))
671
+ return kernel, rowIds, colIds
672
+
673
+ def write_kernel(self, layers, kernel_type, output_file):
674
+ kernel, rowIds, colIds = self.matrices["kernels"][layers][kernel_type]
675
+ np.save(output_file, kernel)
676
+ self.write_nodelist(rowIds, output_file + "_rowIds")
677
+ self.write_nodelist(rowIds, output_file + "_colIds")
678
+
679
+ def get_similarity(self, layers, base_layer, sim_type='lin', options={}, output_filename=None, outFormat='pair', add_to_object= False):
680
+ # options--> options['term_filter'] = GO:00001
681
+ ontology = self.layer_ontologies[base_layer]
682
+ relations = self.get_layers_as_dict(layers, base_layer)
683
+ ontology.load_profiles(relations)
684
+ ontology.clean_profiles(store = True,options=options)
685
+ similarity_pairs = ontology.compare_profiles(sim_type = sim_type)
686
+
687
+ if len(layers) == 1: layers = (layers[0],layers[0])
688
+ self.control_output(values = similarity_pairs, output_filename=output_filename, inFormat="nested_pairs",
689
+ outFormat=outFormat, add_to_object=add_to_object, matrix_keys= ("semantic_sims", layers, sim_type))
690
+ return similarity_pairs
691
+
692
+
693
+
694
+ def shortest_path(self, source, target):
695
+ return nx.shortest_path(self.graph, source, target)
696
+
697
+ def average_shortest_path_length(self, community):
698
+ weight_attr_name = "weight" if nx.get_edge_attributes(self.graph, 'weight') else None
699
+ try:
700
+ com = community.copy()
701
+ path_lens = []
702
+ while len(com) > 1:
703
+ source = com.pop()
704
+ for target in com:
705
+ path_lens.append(nx.shortest_path_length(self.graph, source, target, weight= weight_attr_name))
706
+ if path_lens:
707
+ asp_com = np.mean(path_lens)
708
+ else:
709
+ asp_com = None
710
+ except nx.exception.NetworkXNoPath:
711
+ asp_com = None
712
+ if asp_com and pd.isna(asp_com): raise Exception("Na value not expected on avg sht path")
713
+ return asp_com
714
+
715
+ def shortest_paths(self, community):
716
+ return nx.all_pairs_shortest_path(community)
717
+
718
+
719
+ def get_node_attributes(self, attr_names, layers = "all", summary = False, output_filename = None):
720
+ if type(layers) == str and layers != "all": layers = [layers]
721
+ if layers == "all": layers = self.layers
722
+ node_universe = self.get_nodes_layer(layers)
723
+
724
+ attrs = {}
725
+ for attr_name in attr_names:
726
+ if attr_name == 'get_degree':
727
+ attrs["get_degree"] = self.get_degree(zscore=False)
728
+ elif attr_name == 'get_degreeZ':
729
+ attrs["get_degreeZ"] = self.get_degree()
730
+ elif attr_name == "betweenness_centrality":
731
+ attrs["betweenness_centrality"] = self.get_betweenness_centrality(zscore=False)
732
+ elif attr_name == "betweenness_centralityZ":
733
+ attrs["betweenness_centralityZ"] = self.get_betweenness_centrality()
734
+
735
+ node_ids = attrs[list(attrs.keys())[0]].keys() # TODO: This line of code should be replaced for an option to select node for each attr.
736
+ node_ids = [node_id for node_id in node_ids if node_id in node_universe]
737
+
738
+ node_attrs = []
739
+ if summary:
740
+ for at in attr_names:
741
+ stats = pxc.get_stats_from_list(list(attrs[at].values()))
742
+ node_attrs += [[at] + stat for stat in stats]
743
+ else:
744
+ for n in node_ids:
745
+ n_attrs = [ attrs[at][n] for at in attr_names ]
746
+ node_attrs.append([n] + n_attrs)
747
+
748
+ self.control_output(values = node_attrs, output_filename=output_filename, inFormat="pair", outFormat="pair", add_to_object=False)
749
+
750
+ return node_attrs
751
+
752
+ def get_graph_attributes(self, attr_names, layers = "all", summary = False, output_filename = None):
753
+ if type(layers) == str and layers != "all": layers = [layers]
754
+ if layers == "all":
755
+ subgraph = self.graph
756
+ else:
757
+ node_universe = self.get_nodes_layer(layers)
758
+ subgraph = self.graph.subgraph(node_universe)
759
+
760
+ attrs = {}
761
+ for attr_name in attr_names:
762
+ if attr_name == 'size':
763
+ attrs[attr_name] = len(subgraph.nodes())
764
+ elif attr_name == 'edge_density':
765
+ attrs[attr_name] = nx.density(subgraph)
766
+ elif attr_name == 'transitivity' or attr_name == 'global_clustering':
767
+ attrs[attr_name] = nx.transitivity(subgraph)
768
+ elif attr_name == "assorciativity":
769
+ attrs[attr_name] = nx.degree_assortativity_coefficient(subgraph)
770
+
771
+ graph_attrs = []
772
+ for attr_name, attr_value in attrs.items():
773
+ graph_attrs.append([attr_name, attr_value])
774
+
775
+ self.control_output(values = graph_attrs, output_filename=output_filename, inFormat="pair", outFormat="pair", add_to_object=False)
776
+ return graph_attrs
777
+
778
+
779
+
780
+ ## Ploting method
781
+ #----------------
782
+
783
+ def plot_network(self, options = {}):
784
+ net_data = {
785
+ 'group_nodes': self.group_nodes,
786
+ 'reference_nodes': self.reference_nodes,
787
+ 'graph': self.graph,
788
+ 'layers': self.layers
789
+ }
790
+ Net_plotter(net_data, options)
791
+
792
+ ## Matrix information and manipulation
793
+ #-------------------------------------
794
+
795
+ def write_matrix(self, mat_keys, output_filename):
796
+ matrix_row_col = pxc.dig(self.matrices,*mat_keys)
797
+ if matrix_row_col is not None:
798
+ mat, rowIds, colIds = matrix_row_col
799
+ self.write_obj(mat, output_filename, Format= "matrix", rowIds=rowIds, colIds=colIds)
800
+ else:
801
+ raise Exception("keys for matrices which dont exist yet")
802
+
803
+ def write_stats_from_matrix(self, mat_keys, output_filename="stats_from_matrix"):
804
+ matrix_data = pxc.dig(self.matrices,*mat_keys)
805
+ if matrix_data == None: raise Exception("keys for matrices which dont exist yet")
806
+ matrix, _, _ = matrix_data
807
+
808
+ stats = pxc.get_stats_from_matrix(matrix)
809
+ self.write_obj(stats, output_filename, Format="pair")
810
+
811
+ def normalize_matrix(self, mat_keys, by= "rows_cols"):
812
+ matrix_data = pxc.dig(self.matrices,*mat_keys)
813
+ if matrix_data == None: raise Exception("keys for matrices which dont exist yet")
814
+ matrix, rowIds, colIds = matrix_data
815
+
816
+ matrix = pxc.normalize_matrix(matrix, by)
817
+
818
+ self.control_output(values = matrix, rowIds = rowIds, colIds = colIds, output_filename = None, outFormat = "matrix",
819
+ inFormat = "matrix", add_to_object = True, matrix_keys = mat_keys)
820
+
821
+
822
+ def mat_vs_mat(self, mat1_rowcol, mat2_rowcol, operation="cutoff", options={"cutoff": 0, "cutoff_type": "greater"}): #2exp?
823
+ default_options={"cutoff": 0, "cutoff_type": "greater"}
824
+ default_options.update(options)
825
+ mat1, rows1, cols1 = mat1_rowcol
826
+ mat2, rows2, cols2 = mat2_rowcol
827
+
828
+ if operation == "filter":
829
+ if default_options["cutoff_type"] == "greater":
830
+ mat2 = mat2 >= default_options["cutoff"]
831
+ elif default_options["cutoff_type"] == "less":
832
+ mat2 = mat2 <= default_options["cutoff"]
833
+ mat_result = mat1 * mat2
834
+ rows_result, cols_result = rows1, cols1
835
+ mat_result, rows_result, cols_result = pxc.remove_zero_lines(mat_result, rows_result, cols_result)
836
+
837
+ return mat_result, rows_result, cols_result
838
+
839
+
840
+ def mat_vs_mat_operation(self, mat1_keys, mat2_keys, operation, options, output_filename=None, outFormat='matrix', add_to_object= False):
841
+ result = (None, None, None)
842
+
843
+ mat1 = pxc.dig(self.matrices,*mat1_keys)
844
+ mat2 = pxc.dig(self.matrices,*mat2_keys)
845
+
846
+ if mat1 is None or mat2 is None:
847
+ raise Exception("keys for matrices which dont exist yet")
848
+
849
+ mat_result, rows_result, cols_result = self.mat_vs_mat(mat1, mat2, operation, options)
850
+
851
+
852
+ self.control_output(values = mat_result, rowIds = rows_result, colIds = cols_result, output_filename = output_filename,
853
+ inFormat = "matrix", outFormat = outFormat, add_to_object = add_to_object, matrix_keys = mat1_keys)
854
+ return mat_result, rows_result, cols_result
855
+
856
+
857
+ def filter_matrix(self, mat_keys, operation, options, output_filename=None, outFormat='matrix', add_to_object= False):
858
+ result = (None, None, None)
859
+
860
+ mat1 = pxc.dig(self.matrices,*mat_keys)
861
+
862
+ if mat1 is None:
863
+ raise Exception("keys for matrices which dont exist yet")
864
+ else:
865
+ mat1, rows1, cols1 = mat1
866
+ layers = mat_keys[1]
867
+
868
+ if operation == "filter_cutoff":
869
+ filtered_mat = mat1 >= options["cutoff"]
870
+ mat_result = mat1 * filtered_mat
871
+ pxc.filter_cutoff_mat(mat1, cutoff = options["cutoff"])
872
+ rows_result, cols_result = rows1, cols1
873
+ elif operation == "filter_disparity":
874
+ mat_result, rows_result, cols_result = Adv_mat_calc.disparity_filter_mat(mat1, rows1, cols1, pval_threshold = options["pval_threshold"])
875
+ elif operation == "filter_by_percentile":
876
+ mat_result = pxc.percentile_filter(mat1, options["percentile"]) # TODO: Check is this is valid for non-square matrix
877
+ rows_result, cols_result = rows1, cols1
878
+
879
+ if options.get("binarize"):
880
+ mat_result = pxc.binarize_mat(mat_result)
881
+
882
+ mat_result, rows_result, cols_result = pxc.remove_zero_lines(mat_result, rows_result, cols_result)
883
+
884
+ self.control_output(values = mat_result, rowIds = rows_result, colIds = cols_result, output_filename = output_filename,
885
+ inFormat = "matrix", outFormat = outFormat, add_to_object = add_to_object, matrix_keys = mat_keys)
886
+ return mat_result, rows_result, cols_result
887
+
888
+
889
+ ## Community Methods
890
+ #-------------------
891
+
892
+ # Cluster (community) dicovery #
893
+
894
+ def get_communities_as_cdlibObj(self,communities,overlaping=False): # communites is a hash like group_nodes
895
+ coms = [list(c) for c in communities.values()]
896
+ communities = NodeClustering(coms, self.graph, "external", method_parameters={}, overlap=overlaping)
897
+ return communities
898
+
899
+ def discover_clusters(self, cluster_method, clust_kwargs, **user_options):
900
+ if user_options.get("seed") != None: self.set_seed(user_options.get("seed"))
901
+ communities = self.get_clusters_by_algorithm(cluster_method, clust_kwargs)
902
+ if cluster_method in ['hlc', 'hlc_f']: communities = self.link_to_node_communities(communities)
903
+ communities = { str(idx): community for idx, community in enumerate(communities)}
904
+ self.group_nodes.update(communities) # If external coms added, thay will not be removed!
905
+
906
+ def link_to_node_communities(self, communities):
907
+ comm_nodes = []
908
+ for com in communities:
909
+ if len(com) > 1 :
910
+ nodes = []
911
+ for e in com:
912
+ a, b = e
913
+ if not a in nodes: nodes.append(a)
914
+ if not b in nodes: nodes.append(b)
915
+ if len(nodes) > 2: comm_nodes.append(nodes)
916
+ return comm_nodes
917
+
918
+ def get_clusters_by_algorithm(self, cluster_method, clust_kwargs={}):
919
+ if(cluster_method == 'leiden'):
920
+ communities = algorithms.leiden(self.graph, weights='weight', **clust_kwargs)
921
+ elif(cluster_method == 'louvain'):
922
+ communities = algorithms.louvain(self.graph, weight='weight', **clust_kwargs)
923
+ elif(cluster_method == 'cpm'):
924
+ communities = algorithms.cpm(self.graph, weights='weight', **clust_kwargs)
925
+ elif(cluster_method == 'der'):
926
+ communities = algorithms.der(self.graph, **clust_kwargs)
927
+ elif(cluster_method == 'edmot'):
928
+ communities = algorithms.edmot(self.graph, **clust_kwargs)
929
+ elif(cluster_method == 'eigenvector'):
930
+ communities = algorithms.eigenvector(self.graph, **clust_kwargs)
931
+ elif(cluster_method == 'gdmp2'):
932
+ communities = algorithms.gdmp2(self.graph, **clust_kwargs)
933
+ elif(cluster_method == 'greedy_modularity'):
934
+ communities = algorithms.greedy_modularity(self.graph, weight='weight', **clust_kwargs)
935
+ elif(cluster_method == 'label_propagation'):
936
+ communities = algorithms.label_propagation(self.graph, **clust_kwargs)
937
+ elif(cluster_method == 'markov_clustering'):
938
+ communities = algorithms.markov_clustering(self.graph, **clust_kwargs)
939
+ elif(cluster_method == 'rber_pots'):
940
+ communities = algorithms.rber_pots(self.graph, weights='weight', **clust_kwargs)
941
+ elif(cluster_method == 'rb_pots'):
942
+ communities = algorithms.rb_pots(self.graph, weights='weight', **clust_kwargs)
943
+ elif(cluster_method == 'significance_communities'):
944
+ communities = algorithms.significance_communities(self.graph, **clust_kwargs)
945
+ elif(cluster_method == 'spinglass'):
946
+ communities = algorithms.spinglass(self.graph, **clust_kwargs)
947
+ elif(cluster_method == 'surprise_communities'):
948
+ communities = algorithms.surprise_communities(self.graph, **clust_kwargs)
949
+ elif(cluster_method == 'walktrap'):
950
+ communities = algorithms.walktrap(self.graph, **clust_kwargs)
951
+ elif(cluster_method == 'lais2'):
952
+ communities = algorithms.lais2(self.graph, **clust_kwargs)
953
+ elif(cluster_method == 'big_clam'):
954
+ communities = algorithms.big_clam(self.graph, **clust_kwargs)
955
+ elif(cluster_method == 'danmf'):
956
+ communities = algorithms.danmf(self.graph, **clust_kwargs)
957
+ elif(cluster_method == 'ego_networks'):
958
+ communities = algorithms.ego_networks(self.graph, **clust_kwargs)
959
+ elif(cluster_method == 'egonet_splitter'):
960
+ communities = algorithms.egonet_splitter(self.graph, **clust_kwargs)
961
+ elif(cluster_method == 'mnmf'):
962
+ communities = algorithms.mnmf(self.graph, **clust_kwargs)
963
+ elif(cluster_method == 'nnsed'):
964
+ communities = algorithms.nnsed(self.graph, **clust_kwargs)
965
+ elif(cluster_method == 'slpa'):
966
+ communities = algorithms.slpa(self.graph, **clust_kwargs)
967
+ elif(cluster_method == 'bimlpa'):
968
+ communities = algorithms.bimlpa(self.graph, **clust_kwargs)
969
+ elif(cluster_method == 'wcommunity'):
970
+ communities = algorithms.wCommunity(self.graph, **clust_kwargs)
971
+ elif(cluster_method == 'kclique'):
972
+ communities = algorithms.kclique(self.graph, **clust_kwargs)
973
+ elif(cluster_method == 'hlc'):
974
+ communities = algorithms.hierarchical_link_community(self.graph, **clust_kwargs)
975
+ elif(cluster_method == 'hlc_f'):
976
+ communities = algorithms.hierarchical_link_community_full(self.graph, **clust_kwargs)
977
+ elif(cluster_method == 'aslpaw'):
978
+ with warnings.catch_warnings():
979
+ warnings.filterwarnings("ignore")
980
+ communities = algorithms.aslpaw(self.graph)
981
+ else:
982
+ raise Exception('Not defined method')
983
+ print(communities.method_parameters, file=sys.stderr)
984
+ print(communities.overlap, file=sys.stderr)
985
+ print(communities.node_coverage, file=sys.stderr)
986
+
987
+ return communities.communities # To return a list of list with each of the nodes names for each communities.
988
+
989
+ # Metrics
990
+
991
+ # Evaluating one community
992
+
993
+ def compute_comparative_degree(self, com): # see Girvan-Newman Benchmark control parameter in http://networksciencebook.com/chapter/9#testing (communities chapter)
994
+ internal_degree = 0
995
+ external_degree = 0
996
+ com_nodes = set(com)
997
+ for nodeID in com_nodes:
998
+ nodeIDneigh = set(self.graph.neighbors(nodeID))
999
+ if nodeIDneigh == None: next
1000
+ internal_degree += len(nodeIDneigh & com_nodes)
1001
+ external_degree += len(nodeIDneigh - com_nodes)
1002
+ comparative_degree = external_degree / (external_degree + internal_degree)
1003
+ return comparative_degree
1004
+
1005
+ def compute_node_com_assoc(self, com, ref_node):
1006
+ ref_edges = 0
1007
+ ref_secondary_edges = 0
1008
+ secondary_nodes = {}
1009
+ other_edges = 0
1010
+ other_nodes = {}
1011
+
1012
+ refNneigh = set(self.graph.neighbors(ref_node))
1013
+ for nodeID in com: # Change this to put as a list of nodes
1014
+ nodeIDneigh = set(self.graph.neighbors(nodeID))
1015
+ if nodeIDneigh == None: next
1016
+ if ref_node in nodeIDneigh: ref_edges += 1
1017
+ if refNneigh != None:
1018
+ common_nodes = nodeIDneigh & refNneigh
1019
+ for id in common_nodes: secondary_nodes[id] = True
1020
+ ref_secondary_edges += len(common_nodes)
1021
+ specific_nodes = nodeIDneigh - refNneigh - {ref_node}
1022
+ for id in specific_nodes: other_nodes[id] = True
1023
+ other_edges += len(specific_nodes)
1024
+ by_edge = (ref_edges + ref_secondary_edges) / other_edges
1025
+ by_node = (ref_edges + len(secondary_nodes)) / len(other_nodes)
1026
+ return [by_edge, by_node]
1027
+
1028
+ # Evaluating all communities
1029
+
1030
+ def communities_avg_sht_path(self, coms):
1031
+ asp_coms = []
1032
+ for com_id, com in coms.items():
1033
+ asp_com = self.average_shortest_path_length(com)
1034
+ asp_coms.append(asp_com)
1035
+ return asp_coms
1036
+
1037
+ def communities_comparative_degree(self, coms):
1038
+ return [ self.compute_comparative_degree(com) for com_id, com in coms.items()]
1039
+
1040
+ def communities_node_com_assoc(self, coms, ref_node):
1041
+ return [ self.compute_node_com_assoc(com, ref_node) for com_id, com in coms.items()]
1042
+
1043
+ def compute_summarized_group_metrics(self, output_filename, metrics = ['size', 'avg_transitivity', 'internal_edge_density',
1044
+ 'conductance', 'triangle_participation_ratio', 'max_odf', 'avg_odf', 'avg_embeddedness', 'average_internal_degree','cut_ratio',
1045
+ 'fraction_over_median_degree', 'scaled_density']):
1046
+ # HAS NOT SUMMARY: 'surprise', 'significance', 'comparative_degree', 'avg_sht_path', 'node_com_assoc'
1047
+ communities = NodeClustering(list(self.group_nodes.values()), self.graph, "external", overlap=True)
1048
+ results = []
1049
+ for metric in metrics:
1050
+ # https://www.kite.com/python/answers/how-to-call-a-function-by-its-name-as-a-string-in-python
1051
+ class_method = getattr(evaluation, metric)
1052
+ res = class_method(self.graph, communities)
1053
+ results.append(res)
1054
+
1055
+ with open(output_filename, 'w') as out_file:
1056
+ out_file.write("\t".join(["Metric", "Mean", "Max", "Min", "Std"]) + "\n")
1057
+ count = 0
1058
+ for res in results:
1059
+ metric_name = metrics[count]
1060
+ out_file.write("\t".join([metric_name, str(res.score), str(res.max), str(res.min), str(res.std)]) + "\n")
1061
+ count += 1
1062
+
1063
+ def compute_group_metrics(self, output_filename, metrics = ['comparative_degree', 'avg_sht_path', 'node_com_assoc']): #metics by each clusters
1064
+ output_metrics = [[k] for k in self.group_nodes.keys()]
1065
+ header = ['group']
1066
+
1067
+ for metric in metrics:
1068
+ self.add_metrics(header, output_metrics, metric)
1069
+
1070
+ with open(output_filename, 'w') as out_file:
1071
+ out_file.write("\t".join(header) + "\n")
1072
+ for line in output_metrics:
1073
+ out_file.write("\t".join(list(map(str, line))) + "\n")
1074
+
1075
+ def add_metrics(self, header, output_metrics, metric):
1076
+ # Fusion of cdlib stats methods with NetAnalyzer "original" methods.
1077
+ if metric == 'comparative_degree':
1078
+ comparative_degree = self.communities_comparative_degree(self.group_nodes)
1079
+ for i, val in enumerate(comparative_degree): output_metrics[i].append(self.replace_none_vals(val)) # Add to metrics
1080
+ header.append(metric)
1081
+ elif metric == 'avg_sht_path':
1082
+ avg_sht_path = self.communities_avg_sht_path(self.group_nodes)
1083
+ for i, val in enumerate(avg_sht_path): output_metrics[i].append(self.replace_none_vals(val)) # Add to metrics
1084
+ header.append(metric)
1085
+ elif metric == 'node_com_assoc':
1086
+ if len(self.reference_nodes) > 0:
1087
+ header.extend(['node_com_assoc_by_edge', 'node_com_assoc_by_node'])
1088
+ node_com_assoc = self.communities_node_com_assoc(self.group_nodes, self.reference_nodes[0]) # Assume only obe reference node
1089
+ for i, val in enumerate(node_com_assoc): output_metrics[i].extend(val) # Add to metrics
1090
+ else:
1091
+ # https://www.kite.com/python/answers/how-to-call-a-function-by-its-name-as-a-string-in-python
1092
+ communities = NodeClustering(list(self.group_nodes.values()), self.graph, "external", overlap=True) # TODO Maybe this is not the most efficient way (?)
1093
+ class_method = getattr(evaluation, metric)
1094
+ res = class_method(self.graph, communities, summary=False)
1095
+ for i, val in enumerate(res): output_metrics[i].append(self.replace_none_vals(val))
1096
+ header.append(metric)
1097
+
1098
+ # Evaluating comparison between partitions (EXTERNAL EVALUATION IN CDLIB)
1099
+
1100
+ def join_clusters(self, join_strategy):
1101
+ membersG = set([node for nodes in self.group_nodes.values() for node in nodes])
1102
+ membersR = set([node for nodes in self.group_reference.values() for node in nodes])
1103
+
1104
+ if join_strategy == "union":
1105
+ members = membersG | membersR
1106
+ for i, member in enumerate(members - membersR):
1107
+ self.group_reference[f"single_added_ref_{i}"] = [member]
1108
+ for i, member in enumerate(members - membersG):
1109
+ self.group_nodes[f"single_added_group_{i}"] = [member]
1110
+
1111
+ def compare_partitions(self, overlaping=False):
1112
+ communities = self.get_communities_as_cdlibObj(self.group_nodes,overlaping=overlaping)
1113
+ ref_communities = self.get_communities_as_cdlibObj(self.group_reference, overlaping=overlaping)
1114
+ res = {}
1115
+ if overlaping:
1116
+ res["mutual_information_MGH"] = evaluation.overlapping_normalized_mutual_information_MGH(ref_communities,communities).score
1117
+ res["mutual_information_LFK"] = evaluation.overlapping_normalized_mutual_information_LFK(ref_communities,communities).score
1118
+ res["omega"] = evaluation.omega(ref_communities,communities).score
1119
+ #res["overlap_quality"] = evaluation.overlap_quality(ref_communities,communities).score
1120
+ else:
1121
+ res["mutual_information"] = evaluation.normalized_mutual_information(ref_communities,communities).score
1122
+ return res
1123
+
1124
+ # TODO: Add ranker evalutation for set of clusterings (This is told to be added in a posterior expansion phase of lib)
1125
+
1126
+ # Cluster Expansions #
1127
+ def expand_clusters(self, expand_method, one_sht_paths = False):
1128
+ clusters = {}
1129
+
1130
+ if one_sht_paths:
1131
+ get_sht_path = lambda G, nodeA, nodeB: [nx.shortest_path(G, NodeA, NodeB)]
1132
+ else:
1133
+ get_sht_path = lambda G, nodeA, nodeB: nx.all_shortest_paths(G, NodeA, NodeB)
1134
+
1135
+ for id, com in self.group_nodes.items():
1136
+ if expand_method == 'sht_path':
1137
+ new_nodes = set(com)
1138
+ # Community nodes are included in the set above and then this set is expanded with shortest path nodes
1139
+ # between community nodes and assigned as the new cluster nodes list, otherwise updating the current list
1140
+ # could potentially add original community nodes again if they are found in the shortest path between other community nodes.
1141
+ sht_paths = []
1142
+
1143
+ for NodeA, NodeB in itertools.combinations(com, 2):
1144
+ if NodeA in self.graph.nodes and NodeB in self.graph.nodes:
1145
+ sht_path = None
1146
+ try:
1147
+ sht_path = get_sht_path(self.graph, NodeA, NodeB)
1148
+ except nx.exception.NetworkXNoPath:
1149
+ continue
1150
+ sht_paths.append(sht_path)
1151
+
1152
+ for node_pair_sht_paths in sht_paths:
1153
+ for path in node_pair_sht_paths:
1154
+ new_nodes = new_nodes.union(set(path))
1155
+ self.group_nodes[id] = new_nodes # Originally it was modified inplace with "com.add_nodes_from(list(new_nodes))" because "com" were networkx objects
1156
+ clusters[id] = new_nodes
1157
+ return clusters
1158
+
1159
+ ## RAMDOMIZATION METHODS
1160
+ ############################################################
1161
+ def randomize_monopartite_net_by_nodes(self):
1162
+ nodeIds = list(self.graph.nodes)
1163
+ random.shuffle(nodeIds)
1164
+ new_mapping = dict(zip(self.graph.nodes, nodeIds))
1165
+ random_network = self.clone() # TODO # Change to new instance with only an empty graph and layers defined
1166
+ random_network.graph = nx.relabel_nodes(self.graph, new_mapping)
1167
+ return random_network
1168
+
1169
+ def randomize_monopartite_net_by_links(self):
1170
+ source = []
1171
+ target = []
1172
+ weigth = []
1173
+ for e, datadict in self.graph.edges.items():
1174
+ source.append(e[0])
1175
+ target.append(e[1])
1176
+ w = datadict.get('weigth')
1177
+ if w != None: weigth.append(w)
1178
+ random.shuffle(target)
1179
+ random_network = self.clone() # TODO # Change to new instance with only an empty graph and layers defined
1180
+ random_network.graph.clear()
1181
+ for src in source:
1182
+ i = 0
1183
+ while src == target[i] or (random_network.graph.has_node(src) and target[i] in random_network.graph[src]):
1184
+ i += 1
1185
+ targ = target.pop(i)
1186
+ if len(weigth) > 0:
1187
+ random_network.graph.add_edge(src, targ, {'weigth' : weigth.pop()})
1188
+ else:
1189
+ random_network.graph.add_edge(src, targ)
1190
+ return random_network
1191
+
1192
+
1193
+ def randomize_network(self, random_type, **user_options):
1194
+ if user_options.get("seed") != None: self.set_seed(user_options.get("seed"))
1195
+
1196
+ if random_type == 'nodes':
1197
+ if len(self.layers) == 1:
1198
+ random_network = self.randomize_monopartite_net_by_nodes()
1199
+ elif len(self.layers) == 2:
1200
+ random_network = self.randomize_bipartite_net_by_nodes()
1201
+ elif random_type == 'links':
1202
+ if len(self.layers) == 1:
1203
+ random_network = self.randomize_monopartite_net_by_links()
1204
+ elif len(self.layers) == 2:
1205
+ random_network = self.randomize_bipartite_net_by_links()
1206
+ else:
1207
+ raise(f"ERROR: The randomization is not available for {random_type} types of nodes")
1208
+ return random_network
1209
+
1210
+ ## AUXILIAR METHODS
1211
+ #######################################################################################
1212
+
1213
+ def control_output(self, values, output_filename = None, inFormat = "pair", outFormat = "matrix" , add_to_object = False, matrix_keys = None, rowIds = None, colIds = None):
1214
+ if add_to_object:
1215
+ matrix, rowIds, colIds = pxc.transform2obj(values, inFormat= inFormat, outFormat= "matrix", rowIds = rowIds, colIds = colIds)
1216
+ path_keys = matrix_keys[:-1]
1217
+ final_key = matrix_keys[-1]
1218
+ result_dic = pxc.dig(self.matrices,*path_keys)
1219
+ if result_dic is not None:
1220
+ result_dic[final_key] = [matrix, rowIds, colIds]
1221
+ else:
1222
+ # This is posible just when len(matrix_keys) == 3
1223
+ self.matrices[path_keys[0]][path_keys[1]] = {final_key: [matrix, rowIds, colIds]}
1224
+ elif output_filename != None:
1225
+ obj, rowIds, colIds = pxc.transform2obj(values, inFormat= inFormat, outFormat= outFormat, rowIds = rowIds, colIds = colIds)
1226
+ self.write_obj(obj, output_filename, Format=outFormat, rowIds=rowIds, colIds=colIds)
1227
+
1228
+
1229
+ def write_obj(self, obj, output_filename, Format=None, rowIds=None, colIds=None):
1230
+ if Format == 'pair':
1231
+ with open(output_filename, 'w') as f:
1232
+ for pair in obj: f.write("\t".join([str(item) for item in pair]) + "\n")
1233
+ elif Format == 'matrix':
1234
+ np.save(output_filename, obj)
1235
+ if rowIds != None:
1236
+ with open(output_filename + '_rowIds', 'w') as f:
1237
+ for item in rowIds: f.write(item + "\n")
1238
+ if colIds != None:
1239
+ with open(output_filename + '_colIds', 'w') as f:
1240
+ for item in colIds: f.write(item + "\n")
1241
+
1242
+
1243
+ def write_nodelist(self, nodes, file_name):
1244
+ with open(file_name, "w") as f:
1245
+ for node in nodes:
1246
+ f.write(node + "\n")
1247
+
1248
+ def replace_none_vals(self, val): #2exp?
1249
+ return 'NULL' if val == None else val
1250
+
1251
+ def set_seed(self, seed):
1252
+ try:
1253
+ random.seed(int(seed))
1254
+ np.random.seed(int(seed))
1255
+ except ValueError:
1256
+ #np seed cannot used something else but integers, and although random allows it, 200, 200.0 and "200" gives different results, so in order to avoid weird results, we force the seed to be an integer
1257
+ raise(f"ERROR: The seed must be a valid number")