NetAnalyzer 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- NetAnalyzer/__init__.py +11 -0
- NetAnalyzer/adv_mat_calc.py +106 -0
- NetAnalyzer/cli_manager.py +331 -0
- NetAnalyzer/graph2sim.py +106 -0
- NetAnalyzer/integration.py +165 -0
- NetAnalyzer/main_modules.py +610 -0
- NetAnalyzer/net_parser.py +78 -0
- NetAnalyzer/net_plotter.py +200 -0
- NetAnalyzer/netanalyzer.py +1257 -0
- NetAnalyzer/performancer.py +83 -0
- NetAnalyzer/ranker.py +353 -0
- NetAnalyzer/seed_parser.py +27 -0
- NetAnalyzer/templates/net_explorer.txt +83 -0
- NetAnalyzer/templates/network.txt +1 -0
- NetAnalyzer-1.0.0.dist-info/LICENSE.txt +21 -0
- NetAnalyzer-1.0.0.dist-info/METADATA +86 -0
- NetAnalyzer-1.0.0.dist-info/RECORD +20 -0
- NetAnalyzer-1.0.0.dist-info/WHEEL +5 -0
- NetAnalyzer-1.0.0.dist-info/entry_points.txt +8 -0
- NetAnalyzer-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1257 @@
|
|
|
1
|
+
import random
|
|
2
|
+
import sys
|
|
3
|
+
import re
|
|
4
|
+
import copy
|
|
5
|
+
import networkx as nx
|
|
6
|
+
import math
|
|
7
|
+
import numpy as np
|
|
8
|
+
import scipy.stats as stats
|
|
9
|
+
import pandas as pd
|
|
10
|
+
import statsmodels.api as sm
|
|
11
|
+
import itertools
|
|
12
|
+
import warnings
|
|
13
|
+
import logging
|
|
14
|
+
from cdlib import algorithms, viz, evaluation
|
|
15
|
+
from cdlib import NodeClustering
|
|
16
|
+
import py_semtools # For external_data
|
|
17
|
+
from py_semtools import Ontology
|
|
18
|
+
from NetAnalyzer.adv_mat_calc import Adv_mat_calc
|
|
19
|
+
import py_exp_calc.exp_calc as pxc
|
|
20
|
+
from NetAnalyzer.net_plotter import Net_plotter
|
|
21
|
+
from NetAnalyzer.graph2sim import Graph2sim
|
|
22
|
+
from sklearn.preprocessing import StandardScaler
|
|
23
|
+
from sklearn.decomposition import PCA
|
|
24
|
+
from scipy.stats import zscore
|
|
25
|
+
# https://stackoverflow.com/questions/60392940/multi-layer-graph-in-networkx
|
|
26
|
+
# http://mkivela.com/pymnet
|
|
27
|
+
|
|
28
|
+
class NetAnalyzer:
|
|
29
|
+
|
|
30
|
+
def __init__(self, layers):
|
|
31
|
+
self.threads = 2
|
|
32
|
+
self.graph = nx.Graph() # Talk with PSZ, problem with directed graphs on loading edges.
|
|
33
|
+
self.layers = layers
|
|
34
|
+
self.association_values = {}
|
|
35
|
+
self.compute_autorelations = True
|
|
36
|
+
self.compute_pairs = 'conn'
|
|
37
|
+
self.matrices = {"adjacency_matrices": {}, # {layers => Mat, rowIds, colIds}
|
|
38
|
+
"kernels": {}, # {layers => {method_type => Mat, rowIds, colIds}}
|
|
39
|
+
"associations": {}, # {layers => {method_type => Mat, rowIds, colIds}}
|
|
40
|
+
"semantic_sims": {}, # {layers => {method_type => Mat, rowIds, colIds}}
|
|
41
|
+
}
|
|
42
|
+
self.embedding_coords = {}
|
|
43
|
+
self.group_nodes = {} # Communities are lists {community_id : [Node1, Node2,...]}
|
|
44
|
+
self.group_reference = {}
|
|
45
|
+
self.reference_nodes = []
|
|
46
|
+
self.loaded_obos = []
|
|
47
|
+
self.ontologies = []
|
|
48
|
+
self.layer_ontologies = {}
|
|
49
|
+
|
|
50
|
+
def __eq__(self, other): # https://igeorgiev.eu/python/tdd/python-unittest-assert-custom-objects-are-equal/
|
|
51
|
+
return nx.utils.misc.graphs_equal(self.graph, other.graph) and \
|
|
52
|
+
self.layers == other.layers and \
|
|
53
|
+
self.association_values == other.association_values and \
|
|
54
|
+
self.compute_autorelations == other.compute_autorelations and \
|
|
55
|
+
self.compute_pairs == other.compute_pairs and \
|
|
56
|
+
self.matrices == other.matrices and \
|
|
57
|
+
self.embedding_coords == other.embedding_coords and \
|
|
58
|
+
self.group_nodes == other.group_nodes and \
|
|
59
|
+
self.reference_nodes == other.reference_nodes and \
|
|
60
|
+
self.loaded_obos == other.loaded_obos and \
|
|
61
|
+
self.ontologies == other.ontologies and \
|
|
62
|
+
self.layer_ontologies == other.layer_ontologies
|
|
63
|
+
|
|
64
|
+
def clone(self):
|
|
65
|
+
network_clone = NetAnalyzer(copy.copy(self.layers))
|
|
66
|
+
network_clone.graph = copy.deepcopy(self.graph)
|
|
67
|
+
network_clone.association_values = self.association_values.copy()
|
|
68
|
+
network_clone.set_compute_pairs(self.compute_pairs, self.compute_autorelations)
|
|
69
|
+
network_clone.embedding_coords = self.embedding_coords.copy()
|
|
70
|
+
network_clone.matrices = self.matrices.copy()
|
|
71
|
+
network_clone.group_nodes = copy.deepcopy(self.group_nodes)
|
|
72
|
+
network_clone.reference_nodes = self.reference_nodes.copy()
|
|
73
|
+
network_clone.loaded_obos = self.loaded_obos.copy()
|
|
74
|
+
network_clone.ontologies = self.ontologies.deepcopy() if self.ontologies != [] else []
|
|
75
|
+
network_clone.layer_ontologies = self.layer_ontologies.deepcopy() if self.layer_ontologies != {} else {}
|
|
76
|
+
return network_clone
|
|
77
|
+
|
|
78
|
+
# THE PREVIOUS METHODS NEED TO DEFINE/ACCESS THE VERY SAME ATTRIBUTES, WATCH OUT ABOUT THIS !!!!!!!!!!!!!
|
|
79
|
+
|
|
80
|
+
def set_compute_pairs(self, use_pairs, get_autorelations):
|
|
81
|
+
self.compute_pairs = use_pairs
|
|
82
|
+
self.compute_autorelations = get_autorelations
|
|
83
|
+
|
|
84
|
+
def add_node(self, nodeID, layer):
|
|
85
|
+
self.graph.add_node(nodeID, layer=layer)
|
|
86
|
+
|
|
87
|
+
def add_edge(self, node1, node2, **attribs):
|
|
88
|
+
self.graph.add_edge(node1, node2, **attribs) # Talk with PSZ, problem with directed graphs on loading edges.
|
|
89
|
+
|
|
90
|
+
def set_layer(self, layer_definitions, node_name):
|
|
91
|
+
layer = None
|
|
92
|
+
if len(layer_definitions) > 1:
|
|
93
|
+
for layer_name, regexp in layer_definitions:
|
|
94
|
+
if re.search(regexp, node_name):
|
|
95
|
+
layer = layer_name
|
|
96
|
+
break
|
|
97
|
+
if layer == None: raise Exception("The node '" + node_name + "' not match with any layer regex")
|
|
98
|
+
else:
|
|
99
|
+
layer = layer_definitions[0][0]
|
|
100
|
+
if layer not in self.layers: self.layers.append(layer)
|
|
101
|
+
return layer
|
|
102
|
+
|
|
103
|
+
def set_groups(self, groups):
|
|
104
|
+
for group_id, nodes in groups.items():
|
|
105
|
+
for node in nodes:
|
|
106
|
+
if node in self.graph.nodes:
|
|
107
|
+
if self.group_nodes.get(group_id) is None:
|
|
108
|
+
self.group_nodes[group_id] = [node]
|
|
109
|
+
else:
|
|
110
|
+
self.group_nodes[group_id].append(node)
|
|
111
|
+
else:
|
|
112
|
+
#print("Group id: " + str(group_id) + " with member not in network:" + str(node), file=sys.stderr)
|
|
113
|
+
logging.warning("Group id: " + str(group_id) + " with member not in network: " + str(node))
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def generate_adjacency_matrix(self, layerA, layerB):
|
|
117
|
+
layerAidNodes = [ node[0] for node in self.graph.nodes('layer') if node[1] == layerA]
|
|
118
|
+
layerBidNodes = [ node[0] for node in self.graph.nodes('layer') if node[1] == layerB]
|
|
119
|
+
matrix = np.zeros((len(layerAidNodes), len(layerBidNodes)))
|
|
120
|
+
|
|
121
|
+
has_weight = 'weight' if nx.get_edge_attributes(self.graph, 'weight') else None
|
|
122
|
+
|
|
123
|
+
if layerA == layerB:
|
|
124
|
+
# The method biadjacency matrix for this cases, fill the triangular upper matrix.
|
|
125
|
+
matrix_triu = np.array(nx.bipartite.biadjacency_matrix(self.graph, row_order=layerAidNodes, column_order=layerBidNodes, weight=has_weight, format='csr').todense())
|
|
126
|
+
matrix_tril = np.array(nx.bipartite.biadjacency_matrix(self.graph, row_order=layerBidNodes, column_order=layerAidNodes, weight=has_weight, format='csr').todense())
|
|
127
|
+
matrix = np.triu(matrix_triu) + np.tril(np.transpose(matrix_tril), k = -1)
|
|
128
|
+
else:
|
|
129
|
+
matrix = np.array(nx.bipartite.biadjacency_matrix(self.graph, row_order=layerAidNodes, column_order=layerBidNodes, weight=has_weight, format='csr').todense())
|
|
130
|
+
|
|
131
|
+
all_info_matrix = [matrix, layerAidNodes, layerBidNodes]
|
|
132
|
+
|
|
133
|
+
self.matrices["adjacency_matrices"][(layerA, layerB)] = all_info_matrix
|
|
134
|
+
|
|
135
|
+
return all_info_matrix
|
|
136
|
+
|
|
137
|
+
def generate_all_biadjs(self):
|
|
138
|
+
for layerA, layerB in itertools.product(self.layers, self.layers):
|
|
139
|
+
self.generate_adjacency_matrix(layerA, layerB)
|
|
140
|
+
|
|
141
|
+
def adjMat2netObj(self, layerA, layerB):
|
|
142
|
+
matrix, rowIds, colIds = self.matrices["adjacency_matrices"][(layerA, layerB)]
|
|
143
|
+
|
|
144
|
+
self.graph = nx.Graph()
|
|
145
|
+
for rowId in rowIds: self.add_node(rowId, layerA)
|
|
146
|
+
for colId in colIds: self.add_node(colId, layerB)
|
|
147
|
+
|
|
148
|
+
for rowPos, rowId in enumerate(rowIds):
|
|
149
|
+
for colPos, colId in enumerate(colIds):
|
|
150
|
+
associationValue = matrix[rowPos, colPos]
|
|
151
|
+
if associationValue > 0: self.graph.add_edge(rowId, colId, weight=associationValue)
|
|
152
|
+
return self.graph
|
|
153
|
+
|
|
154
|
+
def delete_nodes(self, node_list, mode='d'):
|
|
155
|
+
if mode == 'd':
|
|
156
|
+
self.graph.remove_nodes_from(node_list)
|
|
157
|
+
elif mode == 'r': # reverse selection
|
|
158
|
+
self.graph.remove_nodes_from(list(n for n in self.graph.nodes if n not in node_list ))
|
|
159
|
+
|
|
160
|
+
def get_connected_nodes(self, node_id, from_layer):
|
|
161
|
+
return [n for n in self.graph.neighbors(node_id) if self.graph.nodes[n]['layer'] == from_layer ]
|
|
162
|
+
|
|
163
|
+
def get_layers_as_dict(self, from_layers, to_layer):
|
|
164
|
+
relations = {}
|
|
165
|
+
from_nodes = self.get_nodes_layer(from_layers)
|
|
166
|
+
for fr_node in from_nodes:
|
|
167
|
+
relations[fr_node] = self.get_connected_nodes(fr_node, to_layer)
|
|
168
|
+
return relations
|
|
169
|
+
|
|
170
|
+
def link_ontology(self, ontology_file_path, layer_name):
|
|
171
|
+
if ontology_file_path not in self.loaded_obos: #Load new ontology
|
|
172
|
+
ontology = Ontology(file = ontology_file_path, load_file = True)
|
|
173
|
+
ontology.precompute()
|
|
174
|
+
ontology.threads = self.threads
|
|
175
|
+
self.loaded_obos.append(ontology_file_path)
|
|
176
|
+
self.ontologies.append(ontology)
|
|
177
|
+
else: #Link loaded ontology to current layer
|
|
178
|
+
ontology = self.ontologies[self.loaded_obos.index(ontology_file_path)]
|
|
179
|
+
self.layer_ontologies[layer_name] = ontology
|
|
180
|
+
|
|
181
|
+
def get_bipartite_subgraph(self, from_layer_node_ids, from_layer, to_layer):
|
|
182
|
+
bipartite_subgraph = {}
|
|
183
|
+
for from_layer_node_id in from_layer_node_ids:
|
|
184
|
+
connected_nodes = self.graph.neighbors(from_layer_node_id)
|
|
185
|
+
for connected_node in connected_nodes:
|
|
186
|
+
if self.graph.nodes[connected_node]['layer'] == to_layer:
|
|
187
|
+
query = bipartite_subgraph.get(connected_node)
|
|
188
|
+
if query == None:
|
|
189
|
+
bipartite_subgraph[connected_node] = self.get_connected_nodes(connected_node, from_layer)
|
|
190
|
+
return bipartite_subgraph
|
|
191
|
+
|
|
192
|
+
def get_nodes_by_attr(self, attrib, value):
|
|
193
|
+
return [nodeID for nodeID, attr in self.graph.nodes(data=True) if attr[attrib] == value]
|
|
194
|
+
|
|
195
|
+
def get_nodes_layer(self, layers):
|
|
196
|
+
nodes = []
|
|
197
|
+
for layer in layers:
|
|
198
|
+
nodes.extend(self.get_nodes_by_attr('layer', layer))
|
|
199
|
+
return nodes
|
|
200
|
+
|
|
201
|
+
def get_node_layer(self, node_id):
|
|
202
|
+
return self.graph.nodes(data=True)[node_id]['layer']
|
|
203
|
+
|
|
204
|
+
def get_edge_number(self):
|
|
205
|
+
return len(self.graph.edges())
|
|
206
|
+
|
|
207
|
+
def get_degree(self, zscore = True):
|
|
208
|
+
degree = dict(self.graph.degree())
|
|
209
|
+
if zscore:
|
|
210
|
+
degree = self.znormalize_dic_by_values(degree)
|
|
211
|
+
return degree
|
|
212
|
+
|
|
213
|
+
def get_betweenness_centrality(self, zscore = True):
|
|
214
|
+
betweenness_centrality = dict(nx.betweenness_centrality(self.graph))
|
|
215
|
+
if zscore:
|
|
216
|
+
betweenness_centrality = self.znormalize_dic_by_values(betweenness_centrality)
|
|
217
|
+
return betweenness_centrality
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def znormalize_dic_by_values(self, dic2znormalize):
|
|
221
|
+
data = np.array([d for n, d in dic2znormalize.items()])
|
|
222
|
+
# data_z = Adv_mat_calc.zscore_normalize(data)
|
|
223
|
+
data_z = zscore(data, axis=None)
|
|
224
|
+
znormalized_dic = {}
|
|
225
|
+
count = 0
|
|
226
|
+
for n, d in dic2znormalize.items():
|
|
227
|
+
znormalized_dic[n] = data_z[count]
|
|
228
|
+
count += 1
|
|
229
|
+
dic2znormalize = znormalized_dic
|
|
230
|
+
return dic2znormalize
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def collect_nodes(self, layers = 'all'):
|
|
234
|
+
nodeIDsA = []
|
|
235
|
+
nodeIDsB = []
|
|
236
|
+
if self.compute_autorelations: # TODO: remove the compute_autorrelations attrib.
|
|
237
|
+
if layers == 'all':
|
|
238
|
+
nodeIDsA = self.graph.nodes
|
|
239
|
+
else:
|
|
240
|
+
nodeIDsA = self.get_nodes_layer(layers)
|
|
241
|
+
else:
|
|
242
|
+
if layers != 'all': # layers contains two layer IDs
|
|
243
|
+
nodeIDsA = self.get_nodes_layer([layers[0]])
|
|
244
|
+
nodeIDsB = self.get_nodes_layer([layers[1]])
|
|
245
|
+
return nodeIDsA, nodeIDsB
|
|
246
|
+
|
|
247
|
+
def intersection(self, node1, node2):
|
|
248
|
+
shared_nodes = nx.common_neighbors(self.graph, node1, node2)
|
|
249
|
+
return shared_nodes
|
|
250
|
+
|
|
251
|
+
def get_all_intersections(self, layers = 'all'):
|
|
252
|
+
def _(node1, node2):
|
|
253
|
+
node_intersection = self.intersection(node1, node2)
|
|
254
|
+
return len(list(node_intersection))
|
|
255
|
+
intersection_lengths = self.get_all_pairs(_, layers = layers)
|
|
256
|
+
return intersection_lengths
|
|
257
|
+
|
|
258
|
+
def connections(self, ids_connected_to_n1, ids_connected_to_n2):
|
|
259
|
+
res = False
|
|
260
|
+
if ids_connected_to_n1 != None and ids_connected_to_n2 != None and len(ids_connected_to_n1 & ids_connected_to_n2) > 0 : # check that at least exists one node that connect to n1 and n2
|
|
261
|
+
res = True
|
|
262
|
+
return res
|
|
263
|
+
|
|
264
|
+
def get_all_pairs(self, pair_operation = None , layers = 'all'):
|
|
265
|
+
all_pairs = []
|
|
266
|
+
nodeIDsA, nodeIDsB = self.collect_nodes(layers = layers)
|
|
267
|
+
if pair_operation != None:
|
|
268
|
+
if self.compute_autorelations:
|
|
269
|
+
node_list = [ n for n in nodeIDsA] # Is this conversion needed?
|
|
270
|
+
while len(node_list) > 0:
|
|
271
|
+
node1 = node_list.pop(0)
|
|
272
|
+
if self.compute_pairs == 'all':
|
|
273
|
+
for node2 in node_list:
|
|
274
|
+
res = pair_operation(node1, node2)
|
|
275
|
+
all_pairs.append(res)
|
|
276
|
+
elif self.compute_pairs == 'conn':
|
|
277
|
+
ids_connected_to_n1 = set(self.graph.neighbors(node1))
|
|
278
|
+
for node2 in node_list:
|
|
279
|
+
ids_connected_to_n2 = set(self.graph.neighbors(node2))
|
|
280
|
+
if self.connections(ids_connected_to_n1, ids_connected_to_n2):
|
|
281
|
+
res = pair_operation(node1, node2)
|
|
282
|
+
all_pairs.append(res)
|
|
283
|
+
else:
|
|
284
|
+
if self.compute_pairs == 'conn': #MAIN METHOD
|
|
285
|
+
for node1 in nodeIDsA:
|
|
286
|
+
ids_connected_to_n1 = set(self.graph.neighbors(node1))
|
|
287
|
+
for node2 in nodeIDsB:
|
|
288
|
+
ids_connected_to_n2 = set(self.graph.neighbors(node2))
|
|
289
|
+
if self.connections(ids_connected_to_n1, ids_connected_to_n2):
|
|
290
|
+
res = pair_operation(node1, node2)
|
|
291
|
+
all_pairs.append(res)
|
|
292
|
+
elif self.compute_pairs == 'all':
|
|
293
|
+
raise NotImplementedError('Not implemented')
|
|
294
|
+
|
|
295
|
+
return all_pairs
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
## association methods adjacency matrix based
|
|
299
|
+
#---------------------------------------------------------
|
|
300
|
+
|
|
301
|
+
def clean_autorelations_on_association_values(self):
|
|
302
|
+
for meth, values in self.association_values.items():
|
|
303
|
+
self.association_values[meth] = [relation for relation in values if self.graph.nodes[relation[0]]["layer"] != self.graph.nodes[relation[1]]["layer"]]
|
|
304
|
+
|
|
305
|
+
def get_association_values(self, layers, base_layer, meth, output_filename=None, outFormat='pair', add_to_object= False, **options): #TODO: Talk with PSZ about **optinos or options= {}
|
|
306
|
+
|
|
307
|
+
default_options = {"n_neighbors": 15, "min_dist": 0.1, "n_components": 2, "metric": 'euclidean',
|
|
308
|
+
"corr_type": "pearson", "pvalue": 0.05, "pvalue_adj_method": None, "alternative": 'greater', "coords2sim_type": 'dotProduct' }
|
|
309
|
+
default_options.update(options)
|
|
310
|
+
|
|
311
|
+
relations = [] #node A, node B, val
|
|
312
|
+
if meth == 'counts':
|
|
313
|
+
relations = self.get_counts_associations(layers, base_layer)
|
|
314
|
+
elif meth == 'jaccard': #all networks
|
|
315
|
+
relations = self.get_jaccard_associations(layers, base_layer)
|
|
316
|
+
elif meth == 'simpson': #all networks
|
|
317
|
+
relations = self.get_simpson_associations(layers, base_layer)
|
|
318
|
+
elif meth == 'geometric': #all networks
|
|
319
|
+
relations = self.get_geometric_associations(layers, base_layer)
|
|
320
|
+
elif meth == 'cosine': #all networks
|
|
321
|
+
relations = self.get_cosine_associations(layers, base_layer)
|
|
322
|
+
elif meth == 'pcc': #all networks
|
|
323
|
+
relations = self.get_pcc_associations(layers, base_layer)
|
|
324
|
+
elif meth == 'hypergeometric': #all networks
|
|
325
|
+
relations = self.get_hypergeometric_associations(layers, base_layer)
|
|
326
|
+
elif meth == 'hypergeometric_bf': #all networks
|
|
327
|
+
relations = self.get_hypergeometric_associations(layers, base_layer, pvalue_adj_method = 'bonferroni')
|
|
328
|
+
elif meth == 'hypergeometric_bh': #all networks
|
|
329
|
+
relations = self.get_hypergeometric_associations(layers, base_layer, pvalue_adj_method = 'benjamini_hochberg')
|
|
330
|
+
elif meth == 'csi': #all networks
|
|
331
|
+
relations = self.get_csi_associations(layers, base_layer)
|
|
332
|
+
elif meth == 'transference': #tripartite networks
|
|
333
|
+
relations = self.get_association_by_transference_resources(layers, base_layer)
|
|
334
|
+
elif meth == "correlation":
|
|
335
|
+
relations = self.get_corr_associations(layers, base_layer, corr_type = default_options["corr_type"], pvalue = default_options["pvalue"], pvalue_adj_method = default_options["pvalue_adj_method"], alternative = default_options["alternative"])
|
|
336
|
+
elif meth == "umap":
|
|
337
|
+
relations = self.get_umap_associations(layers, base_layer, n_neighbors = default_options["n_neighbors"], min_dist = default_options["min_dist"], n_components = default_options["n_components"], metric = default_options["metric"])
|
|
338
|
+
elif meth == "pca":
|
|
339
|
+
relations = self.get_pca_associations(layers, base_layer, n_components = default_options["n_components"], coords2sim_type = default_options["coords2sim_type"])
|
|
340
|
+
elif meth == "bicm":
|
|
341
|
+
relations = self.get_bicm_associations(layers, base_layer, pvalue = default_options["pvalue"])
|
|
342
|
+
|
|
343
|
+
if len(layers) == 1: layers = (layers[0], layers[0])
|
|
344
|
+
self.control_output(values = relations, output_filename=output_filename, inFormat="pair",
|
|
345
|
+
outFormat=outFormat, add_to_object=add_to_object, matrix_keys= ("associations", layers, meth))
|
|
346
|
+
|
|
347
|
+
return relations
|
|
348
|
+
|
|
349
|
+
def get_matrix_from_keys(self, matrix_keys, symm = True): # TODO: Implement this option in the code.
|
|
350
|
+
layers = matrix_keys[1]
|
|
351
|
+
matrix = pxc.dig(self.matrices,*matrix_keys)
|
|
352
|
+
if matrix is None and symm:
|
|
353
|
+
matrix_keys = list(matrix_keys)
|
|
354
|
+
matrix_keys[1] = (layers[1], layers[0])
|
|
355
|
+
matrix_keys = tuple(matrix_keys)
|
|
356
|
+
matrix = pxc.dig(self.matrices,*matrix_keys)
|
|
357
|
+
return matrix
|
|
358
|
+
|
|
359
|
+
def get_bicm_associations(self, layers, base_layer, pvalue = 0.05, pvalue_adj_method = 'fdr'):
|
|
360
|
+
# TODO: Need test.
|
|
361
|
+
biadj_matrix = pxc.dig(self.matrices, "adjacency_matrices",tuple(layers),base_layer)
|
|
362
|
+
if biadj_matrix is None:
|
|
363
|
+
biadj_matrix = self.generate_adjacency_matrix(*layers, base_layer)
|
|
364
|
+
|
|
365
|
+
matrix, rowIds, _ = biadj_matrix
|
|
366
|
+
|
|
367
|
+
from bicm.graph_classes import BipartiteGraph
|
|
368
|
+
miGraph = BipartiteGraph(biadjacency=matrix)
|
|
369
|
+
relations = miGraph.get_rows_projection(
|
|
370
|
+
alpha=pvalue,
|
|
371
|
+
method='poisson',
|
|
372
|
+
progress_bar=False,
|
|
373
|
+
fmt='edgelist',
|
|
374
|
+
validation_method=pvalue_adj_method)
|
|
375
|
+
relations = [[rowIds[relation[0]], rowIds[relation[1]], 1] for relation in relations] # TODO: Check 0-based numeration.
|
|
376
|
+
return relations
|
|
377
|
+
|
|
378
|
+
|
|
379
|
+
def get_corr_associations(self, layers, base_layer, corr_type = "pearson", pvalue = 0.05, pvalue_adj_method = None, alternative = 'greater'):
|
|
380
|
+
biadj_matrix = pxc.dig(self.matrices,"adjacency_matrices",tuple(layers),base_layer)
|
|
381
|
+
if biadj_matrix is None:
|
|
382
|
+
biadj_matrix = self.generate_adjacency_matrix(*layers, base_layer)
|
|
383
|
+
|
|
384
|
+
matrix, rowIds, _ = biadj_matrix
|
|
385
|
+
|
|
386
|
+
if corr_type == "pearson":
|
|
387
|
+
corr_mat, corr_pvalue = pxc.get_corr(x=matrix.T, alternative=alternative, corr_type = "pearson")
|
|
388
|
+
elif corr_type == "spearman":
|
|
389
|
+
corr_mat, corr_pvalue = pxc.get_corr(x=matrix.T, alternative= alternative, corr_type = "spearman")
|
|
390
|
+
|
|
391
|
+
relations = pxc.matrixes2pairs([corr_pvalue, corr_mat], rowIds, rowIds, symm = True)
|
|
392
|
+
relations = [ relation for relation in relations if relation[0] != relation[1] and not np.isnan(relation[2]) and relation[3] != 0 ]
|
|
393
|
+
if pvalue_adj_method is not None: self.adjust_pval_association(relations, pvalue_adj_method)
|
|
394
|
+
relations = [[relation[0], relation[1], relation[3]] for relation in relations if relation[2] < pvalue]
|
|
395
|
+
return relations
|
|
396
|
+
|
|
397
|
+
def get_umap_associations(self, layers, base_layer, n_neighbors = 15, min_dist = 0.1, n_components = 2, metric = 'euclidean', random_seed = None):
|
|
398
|
+
biadj_matrix = pxc.dig(self.matrices,"adjacency_matrices",tuple(layers),base_layer)
|
|
399
|
+
if biadj_matrix is None:
|
|
400
|
+
biadj_matrix = self.generate_adjacency_matrix(*layers, base_layer)
|
|
401
|
+
data, rowIds, _ = biadj_matrix
|
|
402
|
+
umap_coords = Adv_mat_calc.data2umap(data, n_neighbors=n_neighbors, min_dist=min_dist, n_components=n_components, metric=metric, random_seed = random_seed)
|
|
403
|
+
umap_sims = np.triu(pxc.coords2sim(umap_coords, sim="euclidean"), k = 1)
|
|
404
|
+
relations = pxc.matrix2relations(umap_sims, rowIds, rowIds)
|
|
405
|
+
relations = [relation for relation in relations if relation[2] != 0]
|
|
406
|
+
return relations
|
|
407
|
+
|
|
408
|
+
def get_pca_associations(self, layers, base_layer, n_components = 2, coords2sim_type = "dotProduct"):
|
|
409
|
+
# TODO: Try to select the correct number of n_componentes (automatically)
|
|
410
|
+
biadj_matrix = pxc.dig(self.matrices,"adjacency_matrices",tuple(layers),base_layer)
|
|
411
|
+
if biadj_matrix is None:
|
|
412
|
+
biadj_matrix = self.generate_adjacency_matrix(*layers, base_layer)
|
|
413
|
+
|
|
414
|
+
matrix, rowIds, _ = biadj_matrix
|
|
415
|
+
|
|
416
|
+
x = StandardScaler().fit_transform(matrix)
|
|
417
|
+
pca = PCA(n_components=n_components)
|
|
418
|
+
pca_coords = pca.fit_transform(x)
|
|
419
|
+
pca_matrix_sim = pxc.coords2sim(pca_coords, sim = coords2sim_type)
|
|
420
|
+
relations = pxc.matrix2relations(pca_matrix_sim , rowIds, rowIds)
|
|
421
|
+
return relations
|
|
422
|
+
|
|
423
|
+
|
|
424
|
+
def get_association_by_transference_resources(self, firstPairLayers, secondPairLayers, lambda_value1 = 0.5, lambda_value2 = 0.5):
|
|
425
|
+
relations = []
|
|
426
|
+
matrix1 = self.matrices["adjacency_matrices"][firstPairLayers][0]
|
|
427
|
+
matrix2 = self.matrices["adjacency_matrices"][secondPairLayers][0]
|
|
428
|
+
finalMatrix = Adv_mat_calc.tranference_resources(matrix1, matrix2, lambda_value1 = lambda_value1, lambda_value2 = lambda_value2)
|
|
429
|
+
rowIds = self.matrices["adjacency_matrices"][firstPairLayers][1]
|
|
430
|
+
colIds = self.matrices["adjacency_matrices"][secondPairLayers][2]
|
|
431
|
+
relations = pxc.matrix2relations(finalMatrix, rowIds, colIds)
|
|
432
|
+
self.association_values['transference'] = relations
|
|
433
|
+
return relations
|
|
434
|
+
|
|
435
|
+
def get_associations(self, layers, base_layer, compute_association): # BASE METHOD
|
|
436
|
+
base_nodes = set(self.get_nodes_layer([base_layer]))
|
|
437
|
+
def _(node1, node2):
|
|
438
|
+
associatedIDs_node1 = set(self.graph.neighbors(node1))
|
|
439
|
+
associatedIDs_node2 = set(self.graph.neighbors(node2))
|
|
440
|
+
intersectedIDs = (associatedIDs_node1 & associatedIDs_node2) & base_nodes
|
|
441
|
+
associationValue = compute_association(associatedIDs_node1, associatedIDs_node2, intersectedIDs, node1, node2)
|
|
442
|
+
return [node1, node2, associationValue]
|
|
443
|
+
associations = self.get_all_pairs(_, layers = layers)
|
|
444
|
+
return associations
|
|
445
|
+
|
|
446
|
+
#https://stackoverflow.com/questions/55063978/ruby-like-yield-in-python-3
|
|
447
|
+
def get_counts_associations(self, layers, base_layer):
|
|
448
|
+
def _(associatedIDs_node1, associatedIDs_node2, intersectedIDs, node1, node2):
|
|
449
|
+
return len(intersectedIDs)
|
|
450
|
+
relations = self.get_associations(layers, base_layer, _)
|
|
451
|
+
self.association_values['counts'] = relations
|
|
452
|
+
return relations
|
|
453
|
+
|
|
454
|
+
def get_jaccard_associations(self, layers, base_layer):
|
|
455
|
+
def _(associatedIDs_node1, associatedIDs_node2, intersectedIDs, node1, node2):
|
|
456
|
+
unionIDS = associatedIDs_node1 | associatedIDs_node2
|
|
457
|
+
return len(intersectedIDs)/len(unionIDS)
|
|
458
|
+
relations = self.get_associations(layers, base_layer, _)
|
|
459
|
+
self.association_values['jaccard'] = relations
|
|
460
|
+
return relations
|
|
461
|
+
|
|
462
|
+
def get_simpson_associations(self, layers, base_layer):
|
|
463
|
+
def _(associatedIDs_node1, associatedIDs_node2, intersectedIDs, node1, node2):
|
|
464
|
+
minLength = min([len(associatedIDs_node1), len(associatedIDs_node2)])
|
|
465
|
+
return len(intersectedIDs)/minLength
|
|
466
|
+
relations = self.get_associations(layers, base_layer, _)
|
|
467
|
+
self.association_values['simpson'] = relations
|
|
468
|
+
return relations
|
|
469
|
+
|
|
470
|
+
def get_geometric_associations(self, layers, base_layer):
|
|
471
|
+
#wang 2016 method
|
|
472
|
+
def _(associatedIDs_node1, associatedIDs_node2, intersectedIDs, node1, node2):
|
|
473
|
+
intersectedIDs = len(intersectedIDs)**2
|
|
474
|
+
productLength = math.sqrt(len(associatedIDs_node1) * len(associatedIDs_node2))
|
|
475
|
+
return intersectedIDs/productLength
|
|
476
|
+
relations = self.get_associations(layers, base_layer, _)
|
|
477
|
+
self.association_values['geometric'] = relations
|
|
478
|
+
return relations
|
|
479
|
+
|
|
480
|
+
def get_cosine_associations(self, layers, base_layer):
|
|
481
|
+
def _(associatedIDs_node1, associatedIDs_node2, intersectedIDs, node1, node2):
|
|
482
|
+
productLength = math.sqrt(len(associatedIDs_node1) * len(associatedIDs_node2))
|
|
483
|
+
return len(intersectedIDs)/productLength
|
|
484
|
+
relations = self.get_associations(layers, base_layer, _)
|
|
485
|
+
self.association_values['cosine'] = relations
|
|
486
|
+
return relations
|
|
487
|
+
|
|
488
|
+
def get_pcc_associations(self, layers, base_layer, weighted = False ):
|
|
489
|
+
#for Ny calcule use get_nodes_layer
|
|
490
|
+
base_layer_nodes = self.get_nodes_layer([base_layer])
|
|
491
|
+
ny = len(base_layer_nodes)
|
|
492
|
+
def _(associatedIDs_node1, associatedIDs_node2, intersectedIDs, node1, node2):
|
|
493
|
+
intersProd = len(intersectedIDs) * ny
|
|
494
|
+
nodesProd = len(associatedIDs_node1) * len(associatedIDs_node2)
|
|
495
|
+
nodesSubs = intersProd - nodesProd
|
|
496
|
+
nodesAInNetwork = ny - len(associatedIDs_node1)
|
|
497
|
+
nodesBInNetwork = ny - len(associatedIDs_node2)
|
|
498
|
+
return np.float64(nodesSubs) / math.sqrt(nodesProd * nodesAInNetwork * nodesBInNetwork) # TODO: np.float64 is used to handle division by 0. Fix the implementation/test to avoid this case
|
|
499
|
+
relations = self.get_associations(layers, base_layer, _)
|
|
500
|
+
self.association_values['pcc'] = relations
|
|
501
|
+
return relations
|
|
502
|
+
|
|
503
|
+
def get_csi_associations(self, layers, base_layer):
|
|
504
|
+
pcc_relations = self.get_pcc_associations(layers, base_layer)
|
|
505
|
+
pcc_relations = [row for row in pcc_relations if not math.isnan(row[2])]
|
|
506
|
+
if len(layers) > 1:
|
|
507
|
+
self.clean_autorelations_on_association_values()
|
|
508
|
+
|
|
509
|
+
nx = len(self.get_nodes_layer(layers))
|
|
510
|
+
pcc_vals = {}
|
|
511
|
+
node_rels = {}
|
|
512
|
+
|
|
513
|
+
for node1, node2, assoc_index in pcc_relations:
|
|
514
|
+
pxc.add_nested_value(pcc_vals, (node1, node2), np.abs(assoc_index))
|
|
515
|
+
pxc.add_nested_value(pcc_vals, (node2, node1), np.abs(assoc_index))
|
|
516
|
+
pxc.add_record(node_rels, node1, node2)
|
|
517
|
+
pxc.add_record(node_rels, node2, node1)
|
|
518
|
+
|
|
519
|
+
relations = []
|
|
520
|
+
for node1, node2, assoc_index in pcc_relations:
|
|
521
|
+
pccAB = assoc_index - 0.05
|
|
522
|
+
valid_nodes = 0
|
|
523
|
+
|
|
524
|
+
significant_nodes_from_node1 = set([node for node in node_rels[node1] if pcc_vals[node1][node] >= pccAB])
|
|
525
|
+
significant_nodes_from_node2 = set([node for node in node_rels[node2] if pcc_vals[node2][node] >= pccAB])
|
|
526
|
+
all_significant_nodes = significant_nodes_from_node2 | significant_nodes_from_node1
|
|
527
|
+
all_nodes = set(node_rels[node1]) | set(node_rels[node2])
|
|
528
|
+
|
|
529
|
+
csiValue = 1 - (len(all_significant_nodes))/(len(all_nodes))
|
|
530
|
+
relations.append([node1, node2, csiValue])
|
|
531
|
+
|
|
532
|
+
self.association_values['csi'] = relations
|
|
533
|
+
return relations
|
|
534
|
+
|
|
535
|
+
def get_hypergeometric_associations(self, layers, base_layer, pvalue_adj_method= None):
|
|
536
|
+
ny = len(self.get_nodes_layer([base_layer]))
|
|
537
|
+
def _(associatedIDs_node1, associatedIDs_node2, intersectedIDs, node1, node2):
|
|
538
|
+
# Analogous formulation with stats.fisher_exact(data, alternative='greater')
|
|
539
|
+
intersection_lengths = len(intersectedIDs)
|
|
540
|
+
if intersection_lengths > 0:
|
|
541
|
+
n1_items = len(associatedIDs_node1)
|
|
542
|
+
n2_items = len(associatedIDs_node2)
|
|
543
|
+
p_value = stats.hypergeom.sf(intersection_lengths-1, ny, n1_items, n2_items)
|
|
544
|
+
|
|
545
|
+
return p_value
|
|
546
|
+
relations = self.get_associations(layers, base_layer, _)
|
|
547
|
+
|
|
548
|
+
if pvalue_adj_method == 'bonferroni':
|
|
549
|
+
meth = 'hypergeometric_bf'
|
|
550
|
+
self.adjust_pval_association(relations, 'bonferroni')
|
|
551
|
+
elif pvalue_adj_method == 'benjamini_hochberg':
|
|
552
|
+
meth = 'hypergeometric_bh'
|
|
553
|
+
self.adjust_pval_association(relations, 'fdr_bh')
|
|
554
|
+
else:
|
|
555
|
+
meth = 'hypergeometric'
|
|
556
|
+
relations = [[assoc[0], assoc[1], -np.log10(assoc[2])] for assoc in relations if assoc[2] > 0]
|
|
557
|
+
self.association_values[meth] = relations
|
|
558
|
+
return relations
|
|
559
|
+
|
|
560
|
+
def adjust_pval_association(self, associations, method): # TODO TEST
|
|
561
|
+
pvals = np.array([val[2] for val in associations])
|
|
562
|
+
adj_pvals = sm.stats.multipletests(pvals, method=method, is_sorted=False, returnsorted=False)[1] #2expcalc?
|
|
563
|
+
for idx, adj_pval in enumerate(adj_pvals):
|
|
564
|
+
associations[idx][2] = adj_pval
|
|
565
|
+
|
|
566
|
+
## filter methods
|
|
567
|
+
#----------------
|
|
568
|
+
|
|
569
|
+
def get_filter(self, layers, method="cutoff", options={}):
|
|
570
|
+
default_options = {"cutoff": None, "compute_autorelations": False, "binarize": False}
|
|
571
|
+
default_options.update(options)
|
|
572
|
+
|
|
573
|
+
if method == "cutoff":
|
|
574
|
+
filtered_function = self.filter_cutoff
|
|
575
|
+
else:
|
|
576
|
+
raise Exception('Not defined method')
|
|
577
|
+
|
|
578
|
+
edges_with_filtered_values = []
|
|
579
|
+
for layers_pairs in itertools.pairwise(layers):
|
|
580
|
+
edges_with_filtered_values += filtered_function(layers_pairs, cutoff= default_options["cutoff"], compute_autorelations = default_options["compute_autorelations"])
|
|
581
|
+
|
|
582
|
+
if default_options["binarize"] == True:
|
|
583
|
+
edges_with_filtered_values = [[edge[0], edge[1], float(edge[2]>0)] for edge in edges_with_filtered_values]
|
|
584
|
+
|
|
585
|
+
self.update_edges_with(edges_with_filtered_values)
|
|
586
|
+
|
|
587
|
+
|
|
588
|
+
def filter_cutoff(self, layers, cutoff=0.5, compute_autorelations= False):
|
|
589
|
+
edges_with_filtered_values = []
|
|
590
|
+
|
|
591
|
+
def _(node1, node2):
|
|
592
|
+
edges_attr = self.graph.edges()[node1,node2]
|
|
593
|
+
weight = edges_attr["weight"] if edges_attr.get("weight") else 1
|
|
594
|
+
if weight < cutoff:
|
|
595
|
+
weight = 0
|
|
596
|
+
return [node1, node2, weight]
|
|
597
|
+
|
|
598
|
+
edges_with_filtered_values = self.get_direct_conns(_, layers = layers, compute_autorelations= compute_autorelations)
|
|
599
|
+
|
|
600
|
+
return edges_with_filtered_values
|
|
601
|
+
|
|
602
|
+
def write_subgraph(self, layers, output_filename, outFormat='pair'):
|
|
603
|
+
if outFormat == "pair":
|
|
604
|
+
nodes = [node for node, data in self.graph.nodes(data=True) if data.get("layer") in layers]
|
|
605
|
+
subgraph = self.graph.subgraph(nodes)
|
|
606
|
+
with open(output_filename, "w") as f:
|
|
607
|
+
for nodeA, nodeB, data in subgraph.edges(data=True):
|
|
608
|
+
weight = data.get("weight")
|
|
609
|
+
if weight is not None:
|
|
610
|
+
f.write(f"{nodeA}\t{nodeB}\t{str(weight)}" + "\n")
|
|
611
|
+
else:
|
|
612
|
+
f.write(f"{nodeA}\t{nodeB}" + "\n")
|
|
613
|
+
elif outFormat == "matrix":
|
|
614
|
+
if self.matrices["adjacency_matrices"].get(layers) is None:
|
|
615
|
+
self.generate_adjacency_matrix(layers[0], layers[1])
|
|
616
|
+
matrix, rowIds, colIds = pxc.dig(self.matrices,"adjacency_matrices", layers)
|
|
617
|
+
self.write_obj(matrix, output_filename=output_filename, Format=outFormat, rowIds=rowIds, colIds=colIds)
|
|
618
|
+
|
|
619
|
+
|
|
620
|
+
def get_direct_conns(self, pair_operation = None, layers = None, compute_autorelations = False):
|
|
621
|
+
direct_edges = []
|
|
622
|
+
nodeIDsA = self.get_nodes_layer([layers[0]])
|
|
623
|
+
if layers[0] == layers[1]:
|
|
624
|
+
nodeIDsB = nodeIDsA
|
|
625
|
+
else:
|
|
626
|
+
nodeIDsB = self.get_nodes_layer([layers[1]])
|
|
627
|
+
|
|
628
|
+
|
|
629
|
+
if compute_autorelations:
|
|
630
|
+
all_nodes = list(set(nodeIDsA).union(set(nodeIDsB)))
|
|
631
|
+
for nodeA, nodeB in itertools.combinations(all_nodes,2): # Watchout! Just valids for non directed graphs
|
|
632
|
+
if self.graph.has_edge(nodeA, nodeB):
|
|
633
|
+
res = pair_operation(nodeA, nodeB)
|
|
634
|
+
direct_edges.append(res)
|
|
635
|
+
else:
|
|
636
|
+
for nodeA in nodeIDsA:
|
|
637
|
+
for nodeB in nodeIDsB:
|
|
638
|
+
if self.graph.has_edge(nodeA, nodeB):
|
|
639
|
+
res = pair_operation(nodeA, nodeB)
|
|
640
|
+
direct_edges.append(res)
|
|
641
|
+
|
|
642
|
+
return direct_edges
|
|
643
|
+
|
|
644
|
+
|
|
645
|
+
def update_edges_with(self, relations):
|
|
646
|
+
for nodeA, nodeB, weight in relations:
|
|
647
|
+
if weight > 0:
|
|
648
|
+
self.graph.add_edge(nodeA, nodeB, weight=weight)
|
|
649
|
+
elif self.graph.has_edge(nodeA, nodeB):
|
|
650
|
+
self.graph.remove_edge(nodeA, nodeB)
|
|
651
|
+
|
|
652
|
+
|
|
653
|
+
## Kernel and similarity methods
|
|
654
|
+
#------------------------------------
|
|
655
|
+
|
|
656
|
+
def get_kernel(self, layers, method, normalization=False, sim_type= "dotProduct", embedding_kwargs={}, output_filename=None, outFormat='matrix', add_to_object= False):
|
|
657
|
+
#embedding_kwargs accept: dimensions, walk_length, num_walks, p, q, workers, window, min_count, seed, quiet, batch_words
|
|
658
|
+
|
|
659
|
+
if method in Graph2sim.allowed_embeddings:
|
|
660
|
+
adj_mat, embedding_nodes, _ = self.matrices["adjacency_matrices"][(layers[0],layers[0])]
|
|
661
|
+
emb_coords = Graph2sim.get_embedding(adj_mat, embedding = method, embedding_nodes=embedding_nodes, **embedding_kwargs)
|
|
662
|
+
kernel = Graph2sim.emb_coords2kernel(emb_coords, normalization, sim_type= sim_type)
|
|
663
|
+
rowIds = embedding_nodes
|
|
664
|
+
colIds = rowIds
|
|
665
|
+
elif method[0:2] in Graph2sim.allowed_kernels:
|
|
666
|
+
adj_mat, rowIds, colIds = self.matrices["adjacency_matrices"][(layers[0],layers[0])]
|
|
667
|
+
kernel = Graph2sim.get_kernel(adj_mat, method, normalization=normalization)
|
|
668
|
+
|
|
669
|
+
self.control_output(values = kernel, rowIds = rowIds, colIds = colIds, output_filename=output_filename, inFormat="matrix",
|
|
670
|
+
outFormat=outFormat, add_to_object=add_to_object, matrix_keys= ("kernels", layers, method))
|
|
671
|
+
return kernel, rowIds, colIds
|
|
672
|
+
|
|
673
|
+
def write_kernel(self, layers, kernel_type, output_file):
|
|
674
|
+
kernel, rowIds, colIds = self.matrices["kernels"][layers][kernel_type]
|
|
675
|
+
np.save(output_file, kernel)
|
|
676
|
+
self.write_nodelist(rowIds, output_file + "_rowIds")
|
|
677
|
+
self.write_nodelist(rowIds, output_file + "_colIds")
|
|
678
|
+
|
|
679
|
+
def get_similarity(self, layers, base_layer, sim_type='lin', options={}, output_filename=None, outFormat='pair', add_to_object= False):
|
|
680
|
+
# options--> options['term_filter'] = GO:00001
|
|
681
|
+
ontology = self.layer_ontologies[base_layer]
|
|
682
|
+
relations = self.get_layers_as_dict(layers, base_layer)
|
|
683
|
+
ontology.load_profiles(relations)
|
|
684
|
+
ontology.clean_profiles(store = True,options=options)
|
|
685
|
+
similarity_pairs = ontology.compare_profiles(sim_type = sim_type)
|
|
686
|
+
|
|
687
|
+
if len(layers) == 1: layers = (layers[0],layers[0])
|
|
688
|
+
self.control_output(values = similarity_pairs, output_filename=output_filename, inFormat="nested_pairs",
|
|
689
|
+
outFormat=outFormat, add_to_object=add_to_object, matrix_keys= ("semantic_sims", layers, sim_type))
|
|
690
|
+
return similarity_pairs
|
|
691
|
+
|
|
692
|
+
|
|
693
|
+
|
|
694
|
+
def shortest_path(self, source, target):
|
|
695
|
+
return nx.shortest_path(self.graph, source, target)
|
|
696
|
+
|
|
697
|
+
def average_shortest_path_length(self, community):
|
|
698
|
+
weight_attr_name = "weight" if nx.get_edge_attributes(self.graph, 'weight') else None
|
|
699
|
+
try:
|
|
700
|
+
com = community.copy()
|
|
701
|
+
path_lens = []
|
|
702
|
+
while len(com) > 1:
|
|
703
|
+
source = com.pop()
|
|
704
|
+
for target in com:
|
|
705
|
+
path_lens.append(nx.shortest_path_length(self.graph, source, target, weight= weight_attr_name))
|
|
706
|
+
if path_lens:
|
|
707
|
+
asp_com = np.mean(path_lens)
|
|
708
|
+
else:
|
|
709
|
+
asp_com = None
|
|
710
|
+
except nx.exception.NetworkXNoPath:
|
|
711
|
+
asp_com = None
|
|
712
|
+
if asp_com and pd.isna(asp_com): raise Exception("Na value not expected on avg sht path")
|
|
713
|
+
return asp_com
|
|
714
|
+
|
|
715
|
+
def shortest_paths(self, community):
|
|
716
|
+
return nx.all_pairs_shortest_path(community)
|
|
717
|
+
|
|
718
|
+
|
|
719
|
+
def get_node_attributes(self, attr_names, layers = "all", summary = False, output_filename = None):
|
|
720
|
+
if type(layers) == str and layers != "all": layers = [layers]
|
|
721
|
+
if layers == "all": layers = self.layers
|
|
722
|
+
node_universe = self.get_nodes_layer(layers)
|
|
723
|
+
|
|
724
|
+
attrs = {}
|
|
725
|
+
for attr_name in attr_names:
|
|
726
|
+
if attr_name == 'get_degree':
|
|
727
|
+
attrs["get_degree"] = self.get_degree(zscore=False)
|
|
728
|
+
elif attr_name == 'get_degreeZ':
|
|
729
|
+
attrs["get_degreeZ"] = self.get_degree()
|
|
730
|
+
elif attr_name == "betweenness_centrality":
|
|
731
|
+
attrs["betweenness_centrality"] = self.get_betweenness_centrality(zscore=False)
|
|
732
|
+
elif attr_name == "betweenness_centralityZ":
|
|
733
|
+
attrs["betweenness_centralityZ"] = self.get_betweenness_centrality()
|
|
734
|
+
|
|
735
|
+
node_ids = attrs[list(attrs.keys())[0]].keys() # TODO: This line of code should be replaced for an option to select node for each attr.
|
|
736
|
+
node_ids = [node_id for node_id in node_ids if node_id in node_universe]
|
|
737
|
+
|
|
738
|
+
node_attrs = []
|
|
739
|
+
if summary:
|
|
740
|
+
for at in attr_names:
|
|
741
|
+
stats = pxc.get_stats_from_list(list(attrs[at].values()))
|
|
742
|
+
node_attrs += [[at] + stat for stat in stats]
|
|
743
|
+
else:
|
|
744
|
+
for n in node_ids:
|
|
745
|
+
n_attrs = [ attrs[at][n] for at in attr_names ]
|
|
746
|
+
node_attrs.append([n] + n_attrs)
|
|
747
|
+
|
|
748
|
+
self.control_output(values = node_attrs, output_filename=output_filename, inFormat="pair", outFormat="pair", add_to_object=False)
|
|
749
|
+
|
|
750
|
+
return node_attrs
|
|
751
|
+
|
|
752
|
+
def get_graph_attributes(self, attr_names, layers = "all", summary = False, output_filename = None):
|
|
753
|
+
if type(layers) == str and layers != "all": layers = [layers]
|
|
754
|
+
if layers == "all":
|
|
755
|
+
subgraph = self.graph
|
|
756
|
+
else:
|
|
757
|
+
node_universe = self.get_nodes_layer(layers)
|
|
758
|
+
subgraph = self.graph.subgraph(node_universe)
|
|
759
|
+
|
|
760
|
+
attrs = {}
|
|
761
|
+
for attr_name in attr_names:
|
|
762
|
+
if attr_name == 'size':
|
|
763
|
+
attrs[attr_name] = len(subgraph.nodes())
|
|
764
|
+
elif attr_name == 'edge_density':
|
|
765
|
+
attrs[attr_name] = nx.density(subgraph)
|
|
766
|
+
elif attr_name == 'transitivity' or attr_name == 'global_clustering':
|
|
767
|
+
attrs[attr_name] = nx.transitivity(subgraph)
|
|
768
|
+
elif attr_name == "assorciativity":
|
|
769
|
+
attrs[attr_name] = nx.degree_assortativity_coefficient(subgraph)
|
|
770
|
+
|
|
771
|
+
graph_attrs = []
|
|
772
|
+
for attr_name, attr_value in attrs.items():
|
|
773
|
+
graph_attrs.append([attr_name, attr_value])
|
|
774
|
+
|
|
775
|
+
self.control_output(values = graph_attrs, output_filename=output_filename, inFormat="pair", outFormat="pair", add_to_object=False)
|
|
776
|
+
return graph_attrs
|
|
777
|
+
|
|
778
|
+
|
|
779
|
+
|
|
780
|
+
## Ploting method
|
|
781
|
+
#----------------
|
|
782
|
+
|
|
783
|
+
def plot_network(self, options = {}):
|
|
784
|
+
net_data = {
|
|
785
|
+
'group_nodes': self.group_nodes,
|
|
786
|
+
'reference_nodes': self.reference_nodes,
|
|
787
|
+
'graph': self.graph,
|
|
788
|
+
'layers': self.layers
|
|
789
|
+
}
|
|
790
|
+
Net_plotter(net_data, options)
|
|
791
|
+
|
|
792
|
+
## Matrix information and manipulation
|
|
793
|
+
#-------------------------------------
|
|
794
|
+
|
|
795
|
+
def write_matrix(self, mat_keys, output_filename):
|
|
796
|
+
matrix_row_col = pxc.dig(self.matrices,*mat_keys)
|
|
797
|
+
if matrix_row_col is not None:
|
|
798
|
+
mat, rowIds, colIds = matrix_row_col
|
|
799
|
+
self.write_obj(mat, output_filename, Format= "matrix", rowIds=rowIds, colIds=colIds)
|
|
800
|
+
else:
|
|
801
|
+
raise Exception("keys for matrices which dont exist yet")
|
|
802
|
+
|
|
803
|
+
def write_stats_from_matrix(self, mat_keys, output_filename="stats_from_matrix"):
|
|
804
|
+
matrix_data = pxc.dig(self.matrices,*mat_keys)
|
|
805
|
+
if matrix_data == None: raise Exception("keys for matrices which dont exist yet")
|
|
806
|
+
matrix, _, _ = matrix_data
|
|
807
|
+
|
|
808
|
+
stats = pxc.get_stats_from_matrix(matrix)
|
|
809
|
+
self.write_obj(stats, output_filename, Format="pair")
|
|
810
|
+
|
|
811
|
+
def normalize_matrix(self, mat_keys, by= "rows_cols"):
|
|
812
|
+
matrix_data = pxc.dig(self.matrices,*mat_keys)
|
|
813
|
+
if matrix_data == None: raise Exception("keys for matrices which dont exist yet")
|
|
814
|
+
matrix, rowIds, colIds = matrix_data
|
|
815
|
+
|
|
816
|
+
matrix = pxc.normalize_matrix(matrix, by)
|
|
817
|
+
|
|
818
|
+
self.control_output(values = matrix, rowIds = rowIds, colIds = colIds, output_filename = None, outFormat = "matrix",
|
|
819
|
+
inFormat = "matrix", add_to_object = True, matrix_keys = mat_keys)
|
|
820
|
+
|
|
821
|
+
|
|
822
|
+
def mat_vs_mat(self, mat1_rowcol, mat2_rowcol, operation="cutoff", options={"cutoff": 0, "cutoff_type": "greater"}): #2exp?
|
|
823
|
+
default_options={"cutoff": 0, "cutoff_type": "greater"}
|
|
824
|
+
default_options.update(options)
|
|
825
|
+
mat1, rows1, cols1 = mat1_rowcol
|
|
826
|
+
mat2, rows2, cols2 = mat2_rowcol
|
|
827
|
+
|
|
828
|
+
if operation == "filter":
|
|
829
|
+
if default_options["cutoff_type"] == "greater":
|
|
830
|
+
mat2 = mat2 >= default_options["cutoff"]
|
|
831
|
+
elif default_options["cutoff_type"] == "less":
|
|
832
|
+
mat2 = mat2 <= default_options["cutoff"]
|
|
833
|
+
mat_result = mat1 * mat2
|
|
834
|
+
rows_result, cols_result = rows1, cols1
|
|
835
|
+
mat_result, rows_result, cols_result = pxc.remove_zero_lines(mat_result, rows_result, cols_result)
|
|
836
|
+
|
|
837
|
+
return mat_result, rows_result, cols_result
|
|
838
|
+
|
|
839
|
+
|
|
840
|
+
def mat_vs_mat_operation(self, mat1_keys, mat2_keys, operation, options, output_filename=None, outFormat='matrix', add_to_object= False):
|
|
841
|
+
result = (None, None, None)
|
|
842
|
+
|
|
843
|
+
mat1 = pxc.dig(self.matrices,*mat1_keys)
|
|
844
|
+
mat2 = pxc.dig(self.matrices,*mat2_keys)
|
|
845
|
+
|
|
846
|
+
if mat1 is None or mat2 is None:
|
|
847
|
+
raise Exception("keys for matrices which dont exist yet")
|
|
848
|
+
|
|
849
|
+
mat_result, rows_result, cols_result = self.mat_vs_mat(mat1, mat2, operation, options)
|
|
850
|
+
|
|
851
|
+
|
|
852
|
+
self.control_output(values = mat_result, rowIds = rows_result, colIds = cols_result, output_filename = output_filename,
|
|
853
|
+
inFormat = "matrix", outFormat = outFormat, add_to_object = add_to_object, matrix_keys = mat1_keys)
|
|
854
|
+
return mat_result, rows_result, cols_result
|
|
855
|
+
|
|
856
|
+
|
|
857
|
+
def filter_matrix(self, mat_keys, operation, options, output_filename=None, outFormat='matrix', add_to_object= False):
|
|
858
|
+
result = (None, None, None)
|
|
859
|
+
|
|
860
|
+
mat1 = pxc.dig(self.matrices,*mat_keys)
|
|
861
|
+
|
|
862
|
+
if mat1 is None:
|
|
863
|
+
raise Exception("keys for matrices which dont exist yet")
|
|
864
|
+
else:
|
|
865
|
+
mat1, rows1, cols1 = mat1
|
|
866
|
+
layers = mat_keys[1]
|
|
867
|
+
|
|
868
|
+
if operation == "filter_cutoff":
|
|
869
|
+
filtered_mat = mat1 >= options["cutoff"]
|
|
870
|
+
mat_result = mat1 * filtered_mat
|
|
871
|
+
pxc.filter_cutoff_mat(mat1, cutoff = options["cutoff"])
|
|
872
|
+
rows_result, cols_result = rows1, cols1
|
|
873
|
+
elif operation == "filter_disparity":
|
|
874
|
+
mat_result, rows_result, cols_result = Adv_mat_calc.disparity_filter_mat(mat1, rows1, cols1, pval_threshold = options["pval_threshold"])
|
|
875
|
+
elif operation == "filter_by_percentile":
|
|
876
|
+
mat_result = pxc.percentile_filter(mat1, options["percentile"]) # TODO: Check is this is valid for non-square matrix
|
|
877
|
+
rows_result, cols_result = rows1, cols1
|
|
878
|
+
|
|
879
|
+
if options.get("binarize"):
|
|
880
|
+
mat_result = pxc.binarize_mat(mat_result)
|
|
881
|
+
|
|
882
|
+
mat_result, rows_result, cols_result = pxc.remove_zero_lines(mat_result, rows_result, cols_result)
|
|
883
|
+
|
|
884
|
+
self.control_output(values = mat_result, rowIds = rows_result, colIds = cols_result, output_filename = output_filename,
|
|
885
|
+
inFormat = "matrix", outFormat = outFormat, add_to_object = add_to_object, matrix_keys = mat_keys)
|
|
886
|
+
return mat_result, rows_result, cols_result
|
|
887
|
+
|
|
888
|
+
|
|
889
|
+
## Community Methods
|
|
890
|
+
#-------------------
|
|
891
|
+
|
|
892
|
+
# Cluster (community) dicovery #
|
|
893
|
+
|
|
894
|
+
def get_communities_as_cdlibObj(self,communities,overlaping=False): # communites is a hash like group_nodes
|
|
895
|
+
coms = [list(c) for c in communities.values()]
|
|
896
|
+
communities = NodeClustering(coms, self.graph, "external", method_parameters={}, overlap=overlaping)
|
|
897
|
+
return communities
|
|
898
|
+
|
|
899
|
+
def discover_clusters(self, cluster_method, clust_kwargs, **user_options):
|
|
900
|
+
if user_options.get("seed") != None: self.set_seed(user_options.get("seed"))
|
|
901
|
+
communities = self.get_clusters_by_algorithm(cluster_method, clust_kwargs)
|
|
902
|
+
if cluster_method in ['hlc', 'hlc_f']: communities = self.link_to_node_communities(communities)
|
|
903
|
+
communities = { str(idx): community for idx, community in enumerate(communities)}
|
|
904
|
+
self.group_nodes.update(communities) # If external coms added, thay will not be removed!
|
|
905
|
+
|
|
906
|
+
def link_to_node_communities(self, communities):
|
|
907
|
+
comm_nodes = []
|
|
908
|
+
for com in communities:
|
|
909
|
+
if len(com) > 1 :
|
|
910
|
+
nodes = []
|
|
911
|
+
for e in com:
|
|
912
|
+
a, b = e
|
|
913
|
+
if not a in nodes: nodes.append(a)
|
|
914
|
+
if not b in nodes: nodes.append(b)
|
|
915
|
+
if len(nodes) > 2: comm_nodes.append(nodes)
|
|
916
|
+
return comm_nodes
|
|
917
|
+
|
|
918
|
+
def get_clusters_by_algorithm(self, cluster_method, clust_kwargs={}):
|
|
919
|
+
if(cluster_method == 'leiden'):
|
|
920
|
+
communities = algorithms.leiden(self.graph, weights='weight', **clust_kwargs)
|
|
921
|
+
elif(cluster_method == 'louvain'):
|
|
922
|
+
communities = algorithms.louvain(self.graph, weight='weight', **clust_kwargs)
|
|
923
|
+
elif(cluster_method == 'cpm'):
|
|
924
|
+
communities = algorithms.cpm(self.graph, weights='weight', **clust_kwargs)
|
|
925
|
+
elif(cluster_method == 'der'):
|
|
926
|
+
communities = algorithms.der(self.graph, **clust_kwargs)
|
|
927
|
+
elif(cluster_method == 'edmot'):
|
|
928
|
+
communities = algorithms.edmot(self.graph, **clust_kwargs)
|
|
929
|
+
elif(cluster_method == 'eigenvector'):
|
|
930
|
+
communities = algorithms.eigenvector(self.graph, **clust_kwargs)
|
|
931
|
+
elif(cluster_method == 'gdmp2'):
|
|
932
|
+
communities = algorithms.gdmp2(self.graph, **clust_kwargs)
|
|
933
|
+
elif(cluster_method == 'greedy_modularity'):
|
|
934
|
+
communities = algorithms.greedy_modularity(self.graph, weight='weight', **clust_kwargs)
|
|
935
|
+
elif(cluster_method == 'label_propagation'):
|
|
936
|
+
communities = algorithms.label_propagation(self.graph, **clust_kwargs)
|
|
937
|
+
elif(cluster_method == 'markov_clustering'):
|
|
938
|
+
communities = algorithms.markov_clustering(self.graph, **clust_kwargs)
|
|
939
|
+
elif(cluster_method == 'rber_pots'):
|
|
940
|
+
communities = algorithms.rber_pots(self.graph, weights='weight', **clust_kwargs)
|
|
941
|
+
elif(cluster_method == 'rb_pots'):
|
|
942
|
+
communities = algorithms.rb_pots(self.graph, weights='weight', **clust_kwargs)
|
|
943
|
+
elif(cluster_method == 'significance_communities'):
|
|
944
|
+
communities = algorithms.significance_communities(self.graph, **clust_kwargs)
|
|
945
|
+
elif(cluster_method == 'spinglass'):
|
|
946
|
+
communities = algorithms.spinglass(self.graph, **clust_kwargs)
|
|
947
|
+
elif(cluster_method == 'surprise_communities'):
|
|
948
|
+
communities = algorithms.surprise_communities(self.graph, **clust_kwargs)
|
|
949
|
+
elif(cluster_method == 'walktrap'):
|
|
950
|
+
communities = algorithms.walktrap(self.graph, **clust_kwargs)
|
|
951
|
+
elif(cluster_method == 'lais2'):
|
|
952
|
+
communities = algorithms.lais2(self.graph, **clust_kwargs)
|
|
953
|
+
elif(cluster_method == 'big_clam'):
|
|
954
|
+
communities = algorithms.big_clam(self.graph, **clust_kwargs)
|
|
955
|
+
elif(cluster_method == 'danmf'):
|
|
956
|
+
communities = algorithms.danmf(self.graph, **clust_kwargs)
|
|
957
|
+
elif(cluster_method == 'ego_networks'):
|
|
958
|
+
communities = algorithms.ego_networks(self.graph, **clust_kwargs)
|
|
959
|
+
elif(cluster_method == 'egonet_splitter'):
|
|
960
|
+
communities = algorithms.egonet_splitter(self.graph, **clust_kwargs)
|
|
961
|
+
elif(cluster_method == 'mnmf'):
|
|
962
|
+
communities = algorithms.mnmf(self.graph, **clust_kwargs)
|
|
963
|
+
elif(cluster_method == 'nnsed'):
|
|
964
|
+
communities = algorithms.nnsed(self.graph, **clust_kwargs)
|
|
965
|
+
elif(cluster_method == 'slpa'):
|
|
966
|
+
communities = algorithms.slpa(self.graph, **clust_kwargs)
|
|
967
|
+
elif(cluster_method == 'bimlpa'):
|
|
968
|
+
communities = algorithms.bimlpa(self.graph, **clust_kwargs)
|
|
969
|
+
elif(cluster_method == 'wcommunity'):
|
|
970
|
+
communities = algorithms.wCommunity(self.graph, **clust_kwargs)
|
|
971
|
+
elif(cluster_method == 'kclique'):
|
|
972
|
+
communities = algorithms.kclique(self.graph, **clust_kwargs)
|
|
973
|
+
elif(cluster_method == 'hlc'):
|
|
974
|
+
communities = algorithms.hierarchical_link_community(self.graph, **clust_kwargs)
|
|
975
|
+
elif(cluster_method == 'hlc_f'):
|
|
976
|
+
communities = algorithms.hierarchical_link_community_full(self.graph, **clust_kwargs)
|
|
977
|
+
elif(cluster_method == 'aslpaw'):
|
|
978
|
+
with warnings.catch_warnings():
|
|
979
|
+
warnings.filterwarnings("ignore")
|
|
980
|
+
communities = algorithms.aslpaw(self.graph)
|
|
981
|
+
else:
|
|
982
|
+
raise Exception('Not defined method')
|
|
983
|
+
print(communities.method_parameters, file=sys.stderr)
|
|
984
|
+
print(communities.overlap, file=sys.stderr)
|
|
985
|
+
print(communities.node_coverage, file=sys.stderr)
|
|
986
|
+
|
|
987
|
+
return communities.communities # To return a list of list with each of the nodes names for each communities.
|
|
988
|
+
|
|
989
|
+
# Metrics
|
|
990
|
+
|
|
991
|
+
# Evaluating one community
|
|
992
|
+
|
|
993
|
+
def compute_comparative_degree(self, com): # see Girvan-Newman Benchmark control parameter in http://networksciencebook.com/chapter/9#testing (communities chapter)
|
|
994
|
+
internal_degree = 0
|
|
995
|
+
external_degree = 0
|
|
996
|
+
com_nodes = set(com)
|
|
997
|
+
for nodeID in com_nodes:
|
|
998
|
+
nodeIDneigh = set(self.graph.neighbors(nodeID))
|
|
999
|
+
if nodeIDneigh == None: next
|
|
1000
|
+
internal_degree += len(nodeIDneigh & com_nodes)
|
|
1001
|
+
external_degree += len(nodeIDneigh - com_nodes)
|
|
1002
|
+
comparative_degree = external_degree / (external_degree + internal_degree)
|
|
1003
|
+
return comparative_degree
|
|
1004
|
+
|
|
1005
|
+
def compute_node_com_assoc(self, com, ref_node):
|
|
1006
|
+
ref_edges = 0
|
|
1007
|
+
ref_secondary_edges = 0
|
|
1008
|
+
secondary_nodes = {}
|
|
1009
|
+
other_edges = 0
|
|
1010
|
+
other_nodes = {}
|
|
1011
|
+
|
|
1012
|
+
refNneigh = set(self.graph.neighbors(ref_node))
|
|
1013
|
+
for nodeID in com: # Change this to put as a list of nodes
|
|
1014
|
+
nodeIDneigh = set(self.graph.neighbors(nodeID))
|
|
1015
|
+
if nodeIDneigh == None: next
|
|
1016
|
+
if ref_node in nodeIDneigh: ref_edges += 1
|
|
1017
|
+
if refNneigh != None:
|
|
1018
|
+
common_nodes = nodeIDneigh & refNneigh
|
|
1019
|
+
for id in common_nodes: secondary_nodes[id] = True
|
|
1020
|
+
ref_secondary_edges += len(common_nodes)
|
|
1021
|
+
specific_nodes = nodeIDneigh - refNneigh - {ref_node}
|
|
1022
|
+
for id in specific_nodes: other_nodes[id] = True
|
|
1023
|
+
other_edges += len(specific_nodes)
|
|
1024
|
+
by_edge = (ref_edges + ref_secondary_edges) / other_edges
|
|
1025
|
+
by_node = (ref_edges + len(secondary_nodes)) / len(other_nodes)
|
|
1026
|
+
return [by_edge, by_node]
|
|
1027
|
+
|
|
1028
|
+
# Evaluating all communities
|
|
1029
|
+
|
|
1030
|
+
def communities_avg_sht_path(self, coms):
|
|
1031
|
+
asp_coms = []
|
|
1032
|
+
for com_id, com in coms.items():
|
|
1033
|
+
asp_com = self.average_shortest_path_length(com)
|
|
1034
|
+
asp_coms.append(asp_com)
|
|
1035
|
+
return asp_coms
|
|
1036
|
+
|
|
1037
|
+
def communities_comparative_degree(self, coms):
|
|
1038
|
+
return [ self.compute_comparative_degree(com) for com_id, com in coms.items()]
|
|
1039
|
+
|
|
1040
|
+
def communities_node_com_assoc(self, coms, ref_node):
|
|
1041
|
+
return [ self.compute_node_com_assoc(com, ref_node) for com_id, com in coms.items()]
|
|
1042
|
+
|
|
1043
|
+
def compute_summarized_group_metrics(self, output_filename, metrics = ['size', 'avg_transitivity', 'internal_edge_density',
|
|
1044
|
+
'conductance', 'triangle_participation_ratio', 'max_odf', 'avg_odf', 'avg_embeddedness', 'average_internal_degree','cut_ratio',
|
|
1045
|
+
'fraction_over_median_degree', 'scaled_density']):
|
|
1046
|
+
# HAS NOT SUMMARY: 'surprise', 'significance', 'comparative_degree', 'avg_sht_path', 'node_com_assoc'
|
|
1047
|
+
communities = NodeClustering(list(self.group_nodes.values()), self.graph, "external", overlap=True)
|
|
1048
|
+
results = []
|
|
1049
|
+
for metric in metrics:
|
|
1050
|
+
# https://www.kite.com/python/answers/how-to-call-a-function-by-its-name-as-a-string-in-python
|
|
1051
|
+
class_method = getattr(evaluation, metric)
|
|
1052
|
+
res = class_method(self.graph, communities)
|
|
1053
|
+
results.append(res)
|
|
1054
|
+
|
|
1055
|
+
with open(output_filename, 'w') as out_file:
|
|
1056
|
+
out_file.write("\t".join(["Metric", "Mean", "Max", "Min", "Std"]) + "\n")
|
|
1057
|
+
count = 0
|
|
1058
|
+
for res in results:
|
|
1059
|
+
metric_name = metrics[count]
|
|
1060
|
+
out_file.write("\t".join([metric_name, str(res.score), str(res.max), str(res.min), str(res.std)]) + "\n")
|
|
1061
|
+
count += 1
|
|
1062
|
+
|
|
1063
|
+
def compute_group_metrics(self, output_filename, metrics = ['comparative_degree', 'avg_sht_path', 'node_com_assoc']): #metics by each clusters
|
|
1064
|
+
output_metrics = [[k] for k in self.group_nodes.keys()]
|
|
1065
|
+
header = ['group']
|
|
1066
|
+
|
|
1067
|
+
for metric in metrics:
|
|
1068
|
+
self.add_metrics(header, output_metrics, metric)
|
|
1069
|
+
|
|
1070
|
+
with open(output_filename, 'w') as out_file:
|
|
1071
|
+
out_file.write("\t".join(header) + "\n")
|
|
1072
|
+
for line in output_metrics:
|
|
1073
|
+
out_file.write("\t".join(list(map(str, line))) + "\n")
|
|
1074
|
+
|
|
1075
|
+
def add_metrics(self, header, output_metrics, metric):
|
|
1076
|
+
# Fusion of cdlib stats methods with NetAnalyzer "original" methods.
|
|
1077
|
+
if metric == 'comparative_degree':
|
|
1078
|
+
comparative_degree = self.communities_comparative_degree(self.group_nodes)
|
|
1079
|
+
for i, val in enumerate(comparative_degree): output_metrics[i].append(self.replace_none_vals(val)) # Add to metrics
|
|
1080
|
+
header.append(metric)
|
|
1081
|
+
elif metric == 'avg_sht_path':
|
|
1082
|
+
avg_sht_path = self.communities_avg_sht_path(self.group_nodes)
|
|
1083
|
+
for i, val in enumerate(avg_sht_path): output_metrics[i].append(self.replace_none_vals(val)) # Add to metrics
|
|
1084
|
+
header.append(metric)
|
|
1085
|
+
elif metric == 'node_com_assoc':
|
|
1086
|
+
if len(self.reference_nodes) > 0:
|
|
1087
|
+
header.extend(['node_com_assoc_by_edge', 'node_com_assoc_by_node'])
|
|
1088
|
+
node_com_assoc = self.communities_node_com_assoc(self.group_nodes, self.reference_nodes[0]) # Assume only obe reference node
|
|
1089
|
+
for i, val in enumerate(node_com_assoc): output_metrics[i].extend(val) # Add to metrics
|
|
1090
|
+
else:
|
|
1091
|
+
# https://www.kite.com/python/answers/how-to-call-a-function-by-its-name-as-a-string-in-python
|
|
1092
|
+
communities = NodeClustering(list(self.group_nodes.values()), self.graph, "external", overlap=True) # TODO Maybe this is not the most efficient way (?)
|
|
1093
|
+
class_method = getattr(evaluation, metric)
|
|
1094
|
+
res = class_method(self.graph, communities, summary=False)
|
|
1095
|
+
for i, val in enumerate(res): output_metrics[i].append(self.replace_none_vals(val))
|
|
1096
|
+
header.append(metric)
|
|
1097
|
+
|
|
1098
|
+
# Evaluating comparison between partitions (EXTERNAL EVALUATION IN CDLIB)
|
|
1099
|
+
|
|
1100
|
+
def join_clusters(self, join_strategy):
|
|
1101
|
+
membersG = set([node for nodes in self.group_nodes.values() for node in nodes])
|
|
1102
|
+
membersR = set([node for nodes in self.group_reference.values() for node in nodes])
|
|
1103
|
+
|
|
1104
|
+
if join_strategy == "union":
|
|
1105
|
+
members = membersG | membersR
|
|
1106
|
+
for i, member in enumerate(members - membersR):
|
|
1107
|
+
self.group_reference[f"single_added_ref_{i}"] = [member]
|
|
1108
|
+
for i, member in enumerate(members - membersG):
|
|
1109
|
+
self.group_nodes[f"single_added_group_{i}"] = [member]
|
|
1110
|
+
|
|
1111
|
+
def compare_partitions(self, overlaping=False):
|
|
1112
|
+
communities = self.get_communities_as_cdlibObj(self.group_nodes,overlaping=overlaping)
|
|
1113
|
+
ref_communities = self.get_communities_as_cdlibObj(self.group_reference, overlaping=overlaping)
|
|
1114
|
+
res = {}
|
|
1115
|
+
if overlaping:
|
|
1116
|
+
res["mutual_information_MGH"] = evaluation.overlapping_normalized_mutual_information_MGH(ref_communities,communities).score
|
|
1117
|
+
res["mutual_information_LFK"] = evaluation.overlapping_normalized_mutual_information_LFK(ref_communities,communities).score
|
|
1118
|
+
res["omega"] = evaluation.omega(ref_communities,communities).score
|
|
1119
|
+
#res["overlap_quality"] = evaluation.overlap_quality(ref_communities,communities).score
|
|
1120
|
+
else:
|
|
1121
|
+
res["mutual_information"] = evaluation.normalized_mutual_information(ref_communities,communities).score
|
|
1122
|
+
return res
|
|
1123
|
+
|
|
1124
|
+
# TODO: Add ranker evalutation for set of clusterings (This is told to be added in a posterior expansion phase of lib)
|
|
1125
|
+
|
|
1126
|
+
# Cluster Expansions #
|
|
1127
|
+
def expand_clusters(self, expand_method, one_sht_paths = False):
|
|
1128
|
+
clusters = {}
|
|
1129
|
+
|
|
1130
|
+
if one_sht_paths:
|
|
1131
|
+
get_sht_path = lambda G, nodeA, nodeB: [nx.shortest_path(G, NodeA, NodeB)]
|
|
1132
|
+
else:
|
|
1133
|
+
get_sht_path = lambda G, nodeA, nodeB: nx.all_shortest_paths(G, NodeA, NodeB)
|
|
1134
|
+
|
|
1135
|
+
for id, com in self.group_nodes.items():
|
|
1136
|
+
if expand_method == 'sht_path':
|
|
1137
|
+
new_nodes = set(com)
|
|
1138
|
+
# Community nodes are included in the set above and then this set is expanded with shortest path nodes
|
|
1139
|
+
# between community nodes and assigned as the new cluster nodes list, otherwise updating the current list
|
|
1140
|
+
# could potentially add original community nodes again if they are found in the shortest path between other community nodes.
|
|
1141
|
+
sht_paths = []
|
|
1142
|
+
|
|
1143
|
+
for NodeA, NodeB in itertools.combinations(com, 2):
|
|
1144
|
+
if NodeA in self.graph.nodes and NodeB in self.graph.nodes:
|
|
1145
|
+
sht_path = None
|
|
1146
|
+
try:
|
|
1147
|
+
sht_path = get_sht_path(self.graph, NodeA, NodeB)
|
|
1148
|
+
except nx.exception.NetworkXNoPath:
|
|
1149
|
+
continue
|
|
1150
|
+
sht_paths.append(sht_path)
|
|
1151
|
+
|
|
1152
|
+
for node_pair_sht_paths in sht_paths:
|
|
1153
|
+
for path in node_pair_sht_paths:
|
|
1154
|
+
new_nodes = new_nodes.union(set(path))
|
|
1155
|
+
self.group_nodes[id] = new_nodes # Originally it was modified inplace with "com.add_nodes_from(list(new_nodes))" because "com" were networkx objects
|
|
1156
|
+
clusters[id] = new_nodes
|
|
1157
|
+
return clusters
|
|
1158
|
+
|
|
1159
|
+
## RAMDOMIZATION METHODS
|
|
1160
|
+
############################################################
|
|
1161
|
+
def randomize_monopartite_net_by_nodes(self):
|
|
1162
|
+
nodeIds = list(self.graph.nodes)
|
|
1163
|
+
random.shuffle(nodeIds)
|
|
1164
|
+
new_mapping = dict(zip(self.graph.nodes, nodeIds))
|
|
1165
|
+
random_network = self.clone() # TODO # Change to new instance with only an empty graph and layers defined
|
|
1166
|
+
random_network.graph = nx.relabel_nodes(self.graph, new_mapping)
|
|
1167
|
+
return random_network
|
|
1168
|
+
|
|
1169
|
+
def randomize_monopartite_net_by_links(self):
|
|
1170
|
+
source = []
|
|
1171
|
+
target = []
|
|
1172
|
+
weigth = []
|
|
1173
|
+
for e, datadict in self.graph.edges.items():
|
|
1174
|
+
source.append(e[0])
|
|
1175
|
+
target.append(e[1])
|
|
1176
|
+
w = datadict.get('weigth')
|
|
1177
|
+
if w != None: weigth.append(w)
|
|
1178
|
+
random.shuffle(target)
|
|
1179
|
+
random_network = self.clone() # TODO # Change to new instance with only an empty graph and layers defined
|
|
1180
|
+
random_network.graph.clear()
|
|
1181
|
+
for src in source:
|
|
1182
|
+
i = 0
|
|
1183
|
+
while src == target[i] or (random_network.graph.has_node(src) and target[i] in random_network.graph[src]):
|
|
1184
|
+
i += 1
|
|
1185
|
+
targ = target.pop(i)
|
|
1186
|
+
if len(weigth) > 0:
|
|
1187
|
+
random_network.graph.add_edge(src, targ, {'weigth' : weigth.pop()})
|
|
1188
|
+
else:
|
|
1189
|
+
random_network.graph.add_edge(src, targ)
|
|
1190
|
+
return random_network
|
|
1191
|
+
|
|
1192
|
+
|
|
1193
|
+
def randomize_network(self, random_type, **user_options):
|
|
1194
|
+
if user_options.get("seed") != None: self.set_seed(user_options.get("seed"))
|
|
1195
|
+
|
|
1196
|
+
if random_type == 'nodes':
|
|
1197
|
+
if len(self.layers) == 1:
|
|
1198
|
+
random_network = self.randomize_monopartite_net_by_nodes()
|
|
1199
|
+
elif len(self.layers) == 2:
|
|
1200
|
+
random_network = self.randomize_bipartite_net_by_nodes()
|
|
1201
|
+
elif random_type == 'links':
|
|
1202
|
+
if len(self.layers) == 1:
|
|
1203
|
+
random_network = self.randomize_monopartite_net_by_links()
|
|
1204
|
+
elif len(self.layers) == 2:
|
|
1205
|
+
random_network = self.randomize_bipartite_net_by_links()
|
|
1206
|
+
else:
|
|
1207
|
+
raise(f"ERROR: The randomization is not available for {random_type} types of nodes")
|
|
1208
|
+
return random_network
|
|
1209
|
+
|
|
1210
|
+
## AUXILIAR METHODS
|
|
1211
|
+
#######################################################################################
|
|
1212
|
+
|
|
1213
|
+
def control_output(self, values, output_filename = None, inFormat = "pair", outFormat = "matrix" , add_to_object = False, matrix_keys = None, rowIds = None, colIds = None):
|
|
1214
|
+
if add_to_object:
|
|
1215
|
+
matrix, rowIds, colIds = pxc.transform2obj(values, inFormat= inFormat, outFormat= "matrix", rowIds = rowIds, colIds = colIds)
|
|
1216
|
+
path_keys = matrix_keys[:-1]
|
|
1217
|
+
final_key = matrix_keys[-1]
|
|
1218
|
+
result_dic = pxc.dig(self.matrices,*path_keys)
|
|
1219
|
+
if result_dic is not None:
|
|
1220
|
+
result_dic[final_key] = [matrix, rowIds, colIds]
|
|
1221
|
+
else:
|
|
1222
|
+
# This is posible just when len(matrix_keys) == 3
|
|
1223
|
+
self.matrices[path_keys[0]][path_keys[1]] = {final_key: [matrix, rowIds, colIds]}
|
|
1224
|
+
elif output_filename != None:
|
|
1225
|
+
obj, rowIds, colIds = pxc.transform2obj(values, inFormat= inFormat, outFormat= outFormat, rowIds = rowIds, colIds = colIds)
|
|
1226
|
+
self.write_obj(obj, output_filename, Format=outFormat, rowIds=rowIds, colIds=colIds)
|
|
1227
|
+
|
|
1228
|
+
|
|
1229
|
+
def write_obj(self, obj, output_filename, Format=None, rowIds=None, colIds=None):
|
|
1230
|
+
if Format == 'pair':
|
|
1231
|
+
with open(output_filename, 'w') as f:
|
|
1232
|
+
for pair in obj: f.write("\t".join([str(item) for item in pair]) + "\n")
|
|
1233
|
+
elif Format == 'matrix':
|
|
1234
|
+
np.save(output_filename, obj)
|
|
1235
|
+
if rowIds != None:
|
|
1236
|
+
with open(output_filename + '_rowIds', 'w') as f:
|
|
1237
|
+
for item in rowIds: f.write(item + "\n")
|
|
1238
|
+
if colIds != None:
|
|
1239
|
+
with open(output_filename + '_colIds', 'w') as f:
|
|
1240
|
+
for item in colIds: f.write(item + "\n")
|
|
1241
|
+
|
|
1242
|
+
|
|
1243
|
+
def write_nodelist(self, nodes, file_name):
|
|
1244
|
+
with open(file_name, "w") as f:
|
|
1245
|
+
for node in nodes:
|
|
1246
|
+
f.write(node + "\n")
|
|
1247
|
+
|
|
1248
|
+
def replace_none_vals(self, val): #2exp?
|
|
1249
|
+
return 'NULL' if val == None else val
|
|
1250
|
+
|
|
1251
|
+
def set_seed(self, seed):
|
|
1252
|
+
try:
|
|
1253
|
+
random.seed(int(seed))
|
|
1254
|
+
np.random.seed(int(seed))
|
|
1255
|
+
except ValueError:
|
|
1256
|
+
#np seed cannot used something else but integers, and although random allows it, 200, 200.0 and "200" gives different results, so in order to avoid weird results, we force the seed to be an integer
|
|
1257
|
+
raise(f"ERROR: The seed must be a valid number")
|