NetAnalyzer 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- NetAnalyzer/__init__.py +11 -0
- NetAnalyzer/adv_mat_calc.py +106 -0
- NetAnalyzer/cli_manager.py +331 -0
- NetAnalyzer/graph2sim.py +106 -0
- NetAnalyzer/integration.py +165 -0
- NetAnalyzer/main_modules.py +610 -0
- NetAnalyzer/net_parser.py +78 -0
- NetAnalyzer/net_plotter.py +200 -0
- NetAnalyzer/netanalyzer.py +1257 -0
- NetAnalyzer/performancer.py +83 -0
- NetAnalyzer/ranker.py +353 -0
- NetAnalyzer/seed_parser.py +27 -0
- NetAnalyzer/templates/net_explorer.txt +83 -0
- NetAnalyzer/templates/network.txt +1 -0
- NetAnalyzer-1.0.0.dist-info/LICENSE.txt +21 -0
- NetAnalyzer-1.0.0.dist-info/METADATA +86 -0
- NetAnalyzer-1.0.0.dist-info/RECORD +20 -0
- NetAnalyzer-1.0.0.dist-info/WHEEL +5 -0
- NetAnalyzer-1.0.0.dist-info/entry_points.txt +8 -0
- NetAnalyzer-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,610 @@
|
|
|
1
|
+
import sys
|
|
2
|
+
import os
|
|
3
|
+
import numpy as np
|
|
4
|
+
import random
|
|
5
|
+
import copy
|
|
6
|
+
from multiprocessing import Process, Manager, Lock
|
|
7
|
+
from py_cmdtabs.cmdtabs import CmdTabs
|
|
8
|
+
import py_exp_calc.exp_calc as pxc
|
|
9
|
+
from py_report_html import Py_report_html
|
|
10
|
+
from NetAnalyzer import Net_parser, NetAnalyzer
|
|
11
|
+
from NetAnalyzer import Kernels
|
|
12
|
+
from NetAnalyzer import Net_parser, NetAnalyzer
|
|
13
|
+
from NetAnalyzer import Ranker
|
|
14
|
+
from NetAnalyzer import Graph2sim
|
|
15
|
+
from NetAnalyzer import Adv_mat_calc
|
|
16
|
+
from NetAnalyzer.performancer import Performancer
|
|
17
|
+
from NetAnalyzer.seed_parser import SeedParser
|
|
18
|
+
import networkx as nx
|
|
19
|
+
|
|
20
|
+
def main_net_explorer(options, test = False):
|
|
21
|
+
# loading gene seeds.
|
|
22
|
+
options = vars(options)
|
|
23
|
+
if options["seed_nodes"]: seeds2explore, _ = SeedParser.load_nodes_by_group(options["seed_nodes"], sep=options["seed_sep"])
|
|
24
|
+
|
|
25
|
+
# Loading multinet operations
|
|
26
|
+
multinet = {}
|
|
27
|
+
for net_id, net_file in options["input_file"].items():
|
|
28
|
+
nodes = options['node_files'][net_id]
|
|
29
|
+
multinet[net_id] = Net_parser.load_network_by_bin_matrix(net_file, [nodes], [['layer', '-']])
|
|
30
|
+
cutoff = options["layer_cutoff"].get(net_id)
|
|
31
|
+
if options["no_autorelations"]: np.fill_diagonal(multinet[net_id].matrices['adjacency_matrices'][('layer','layer')][0], 0)
|
|
32
|
+
if cutoff:
|
|
33
|
+
cutoff = float(cutoff)
|
|
34
|
+
multinet[net_id].filter_matrix( mat_keys=('adjacency_matrices',('layer','layer')),operation='filter_cutoff', options={'cutoff': cutoff}, add_to_object=True)
|
|
35
|
+
multinet[net_id].adjMat2netObj('layer','layer')
|
|
36
|
+
|
|
37
|
+
# extract a subgraph for each
|
|
38
|
+
if options["seed_nodes"]:
|
|
39
|
+
seeds2subgraph = {}
|
|
40
|
+
seeds2lcc = {}
|
|
41
|
+
for seed, nodes in seeds2explore.items():
|
|
42
|
+
seeds2subgraph[seed] = {}
|
|
43
|
+
seeds2lcc[seed] = {}
|
|
44
|
+
for net_id, net in multinet.items():
|
|
45
|
+
# get neighbor from node
|
|
46
|
+
nodes_with_neigh = set(nodes)
|
|
47
|
+
neigh_level = int(options["neigh_level"].get(net_id)) if options["neigh_level"].get(net_id) else 0
|
|
48
|
+
for i in range(0, neigh_level): nodes_with_neigh = get_neigh_set(net, nodes_with_neigh)
|
|
49
|
+
seeds2subgraph[seed][net_id] = net.graph.subgraph(nodes_with_neigh)
|
|
50
|
+
largest_cc = len(max(nx.connected_components(seeds2subgraph[seed][net_id]), key=len))
|
|
51
|
+
seeds2lcc[seed][net_id] = largest_cc
|
|
52
|
+
|
|
53
|
+
# # If mention, add node2vec coordinates with a tnse proyection.
|
|
54
|
+
net2embedding_proj = None
|
|
55
|
+
if options["embedding_proj"]:
|
|
56
|
+
net2embedding_proj = {}
|
|
57
|
+
for net_id, net in multinet.items():
|
|
58
|
+
adj_mat, embedding_nodes, _ = net.matrices["adjacency_matrices"][("layer", "layer")]
|
|
59
|
+
emb_coords = Graph2sim.get_embedding(adj_mat, embedding = "node2vec", embedding_nodes=embedding_nodes)
|
|
60
|
+
umap_coords = Adv_mat_calc.data2umap(emb_coords, n_neighbors = 15, min_dist = 0.1, n_components = 2, metric = 'euclidean', random_seed = None)
|
|
61
|
+
net2embedding_proj[net_id] = [umap_coords, embedding_nodes]
|
|
62
|
+
|
|
63
|
+
# Comparing nets:
|
|
64
|
+
if options["compare_nets"]:
|
|
65
|
+
network_ids = list(options["input_file"].keys())
|
|
66
|
+
num_nets = len(network_ids)
|
|
67
|
+
sim_nets = np.zeros((num_nets,num_nets))
|
|
68
|
+
for i, net_i in enumerate(network_ids):
|
|
69
|
+
for j,net_j in enumerate(network_ids[i:len(network_ids)]):
|
|
70
|
+
edges_i = set(multinet[net_i].graph.edges)
|
|
71
|
+
edges_j = set(multinet[net_j].graph.edges)
|
|
72
|
+
sim_i_j = len(edges_i & edges_j) / min(len(edges_i), len(edges_j))
|
|
73
|
+
sim_nets[i,j+i] = sim_i_j
|
|
74
|
+
sim_nets[j+i,i] = sim_i_j
|
|
75
|
+
net_sims = [sim_nets, network_ids]
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
# Execute the reports in the process.
|
|
79
|
+
template = open(str(os.path.join(os.path.dirname(__file__), 'templates','net_explorer.txt'))).read()
|
|
80
|
+
container = {"seeds2explore": seeds2explore, "seeds2subgraph": seeds2subgraph, "seeds2lcc":seeds2lcc, "net2embedding_proj": net2embedding_proj, "net_sims": net_sims, "plot_method": options["plot_network_method"]}
|
|
81
|
+
|
|
82
|
+
report = Py_report_html(container, 'Network explorer')
|
|
83
|
+
report.build(template)
|
|
84
|
+
report.write(options["output_file"]+".html")
|
|
85
|
+
|
|
86
|
+
if test: return container
|
|
87
|
+
|
|
88
|
+
def get_neigh_set(net, nodes):
|
|
89
|
+
neigh = set(nodes)
|
|
90
|
+
for i,n in enumerate(nodes):
|
|
91
|
+
try:
|
|
92
|
+
neigh = neigh | set(net.graph.neighbors(n))
|
|
93
|
+
except:
|
|
94
|
+
continue
|
|
95
|
+
return list(neigh)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def main_integrate_kernels(options):
|
|
99
|
+
kernels = Kernels()
|
|
100
|
+
|
|
101
|
+
if not options.kernel_ids:
|
|
102
|
+
options.kernel_ids = list(range(0,len(options.kernel_files)))
|
|
103
|
+
options.kernel_ids = [str(k) for k in options.kernel_ids]
|
|
104
|
+
|
|
105
|
+
# TODO: Consider adding more options to integrate to accept other formats
|
|
106
|
+
if options.input_format == "bin":
|
|
107
|
+
kernels.load_kernels_by_bin_matrixes(options.kernel_files, options.node_files, options.kernel_ids)
|
|
108
|
+
kernels.create_general_index()
|
|
109
|
+
|
|
110
|
+
if not options.raw_values:
|
|
111
|
+
kernels.move2zero_reference()
|
|
112
|
+
|
|
113
|
+
if options.integration_type is not None:
|
|
114
|
+
print(options.threads)
|
|
115
|
+
kernels.integrate_matrix(options.integration_type, options.threads, options.symmetric)
|
|
116
|
+
|
|
117
|
+
if options.output_file is not None:
|
|
118
|
+
kernel, names = kernels.integrated_kernel
|
|
119
|
+
np.save(options.output_file, kernel)
|
|
120
|
+
|
|
121
|
+
with open(options.output_file +'.lst', 'w') as f:
|
|
122
|
+
for name in names:
|
|
123
|
+
f.write(name + "\n")
|
|
124
|
+
|
|
125
|
+
def main_netanalyzer(options):
|
|
126
|
+
print("Loading network data")
|
|
127
|
+
opts = vars(options)
|
|
128
|
+
# FRED: Remove this part of vars and modify the loads methods (Tlk wth PSZ)
|
|
129
|
+
fullNet = Net_parser.load(opts)
|
|
130
|
+
fullNet.set_compute_pairs(options.use_pairs, not options.no_autorelations)
|
|
131
|
+
fullNet.threads = options.threads
|
|
132
|
+
|
|
133
|
+
fullNet.reference_nodes = options.reference_nodes
|
|
134
|
+
for ont_data in opts['ontologies']:
|
|
135
|
+
layer_name, ontology_file_path = ont_data
|
|
136
|
+
fullNet.link_ontology(ontology_file_path, layer_name)
|
|
137
|
+
fullNet.ontologies
|
|
138
|
+
|
|
139
|
+
if options.group_nodes:
|
|
140
|
+
fullNet.set_groups(options.group_nodes)
|
|
141
|
+
|
|
142
|
+
if options.delete_nodes:
|
|
143
|
+
node_list = CmdTabs.load_input_data(options.delete_nodes[0])
|
|
144
|
+
node_list = [item for sublist in node_list for item in sublist]
|
|
145
|
+
mode = options.delete_nodes[1] if len(options.delete_nodes) > 1 else 'd'
|
|
146
|
+
fullNet.delete_nodes(node_list, mode)
|
|
147
|
+
|
|
148
|
+
if options.dsl_script is not None:
|
|
149
|
+
execute_dsl_script(fullNet, options.dsl_script)
|
|
150
|
+
sys.exit()
|
|
151
|
+
|
|
152
|
+
if options.meth is not None:
|
|
153
|
+
print(f"Performing association method {options.meth} on network \n")
|
|
154
|
+
if options.meth == "transference":
|
|
155
|
+
if not (options.use_layers[0][0], options.use_layers[1][0]) in fullNet.matrices["adjacency_matrices"]:
|
|
156
|
+
fullNet.generate_adjacency_matrix(
|
|
157
|
+
options.use_layers[0][0], options.use_layers[1][0])
|
|
158
|
+
if not (options.use_layers[1][0], options.use_layers[0][1]) in fullNet.matrices["adjacency_matrices"]:
|
|
159
|
+
fullNet.generate_adjacency_matrix(
|
|
160
|
+
options.use_layers[1][0], options.use_layers[0][1])
|
|
161
|
+
|
|
162
|
+
fullNet.get_association_values(
|
|
163
|
+
(options.use_layers[0][0], options.use_layers[1][0]),
|
|
164
|
+
(options.use_layers[1][0], options.use_layers[0][1]),
|
|
165
|
+
"transference")
|
|
166
|
+
else:
|
|
167
|
+
fullNet.get_association_values(
|
|
168
|
+
options.use_layers[0],
|
|
169
|
+
options.use_layers[1][0],
|
|
170
|
+
options.meth)
|
|
171
|
+
|
|
172
|
+
with open(options.assoc_file, 'w') as f:
|
|
173
|
+
for val in fullNet.association_values[options.meth][:-1]:
|
|
174
|
+
f.write("\t".join(map(str, val)) + "\n")
|
|
175
|
+
f.write(
|
|
176
|
+
"\t".join(map(str, fullNet.association_values[options.meth][-1])))
|
|
177
|
+
print(f"End of analysis: {options.meth}")
|
|
178
|
+
|
|
179
|
+
if options.control_file != None:
|
|
180
|
+
with open(options.control_file, "r") as f:
|
|
181
|
+
control = [control.append(line.rstrip().split("\t")) for line in f]
|
|
182
|
+
Performancer.load_control(control)
|
|
183
|
+
predictions = fullNet.association_values[options.meth]
|
|
184
|
+
performance = Performancer.get_pred_rec(predictions)
|
|
185
|
+
with open(options.performance_file, 'r') as f:
|
|
186
|
+
f.write("\t".join(['cut', 'prec', 'rec', 'meth']) + "\n")
|
|
187
|
+
for item in performance:
|
|
188
|
+
item.append(options['meth'])
|
|
189
|
+
f.write("\t".join(item) + "\n")
|
|
190
|
+
|
|
191
|
+
if options.kernel is not None:
|
|
192
|
+
# This allows inject custom arguments for each embedding method
|
|
193
|
+
embedding_kwargs = eval('{' +options.embedding_add_options +'}')
|
|
194
|
+
if len(options.use_layers) == 1:
|
|
195
|
+
# we use only a layer to perform the kernel, so only one item it is selected.
|
|
196
|
+
layers2kernel = (options.use_layers[0][0], options.use_layers[0][0])
|
|
197
|
+
else:
|
|
198
|
+
layers2kernel = tuple(options.use_layers[0])
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
fullNet.get_kernel(layers2kernel, options.kernel, options.normalize_kernel,
|
|
202
|
+
options.coords2sim_type, embedding_kwargs, add_to_object=True)
|
|
203
|
+
fullNet.write_kernel(layers2kernel, options.kernel, options.kernel_file)
|
|
204
|
+
|
|
205
|
+
if options.graph_file is not None:
|
|
206
|
+
options.graph_options['output_file'] = options.graph_file
|
|
207
|
+
fullNet.plot_network(options.graph_options)
|
|
208
|
+
|
|
209
|
+
# Group creation
|
|
210
|
+
if options.build_cluster_alg is not None:
|
|
211
|
+
clust_kwargs = eval('{' + options.build_clusters_add_options +'}')
|
|
212
|
+
fullNet.discover_clusters(
|
|
213
|
+
options.build_cluster_alg, clust_kwargs, **{'seed': options.seed})
|
|
214
|
+
|
|
215
|
+
if options.output_build_clusters is None:
|
|
216
|
+
options.output_build_clusters = options.build_cluster_alg + \
|
|
217
|
+
'_' + 'discovered_clusters.txt'
|
|
218
|
+
|
|
219
|
+
with open(options.output_build_clusters, 'w') as out_file:
|
|
220
|
+
for cl_id, nodes in fullNet.group_nodes.items():
|
|
221
|
+
for node in nodes:
|
|
222
|
+
out_file.write(f"{cl_id}\t{node}\n")
|
|
223
|
+
|
|
224
|
+
# Group metrics by cluster.
|
|
225
|
+
if options.group_metrics:
|
|
226
|
+
fullNet.compute_group_metrics(
|
|
227
|
+
output_filename=options.output_metrics_by_cluster, metrics=options.group_metrics)
|
|
228
|
+
|
|
229
|
+
# Group metrics summarized.
|
|
230
|
+
if options.summarize_metrics:
|
|
231
|
+
fullNet.compute_summarized_group_metrics(
|
|
232
|
+
output_filename=options.output_summarized_metrics, metrics=options.summarize_metrics)
|
|
233
|
+
|
|
234
|
+
# Comparing Group Families (Two by now)
|
|
235
|
+
if options.compare_clusters_reference is not None:
|
|
236
|
+
fullNet.group_reference = options.compare_clusters_reference
|
|
237
|
+
fullNet.join_clusters(options.compare_clusters_join)
|
|
238
|
+
results = fullNet.compare_partitions(options.overlapping_communities)
|
|
239
|
+
for metric_name, metric_value in results.items():
|
|
240
|
+
print(f"{metric_name}\t{metric_value}")
|
|
241
|
+
|
|
242
|
+
# Group Expansion
|
|
243
|
+
if options.expand_clusters is not None:
|
|
244
|
+
expanded_clusters = fullNet.expand_clusters(
|
|
245
|
+
options.expand_clusters, options.one_sht_pairs)
|
|
246
|
+
with open(options.output_expand_clusters, 'w') as out_file:
|
|
247
|
+
for cl_id, nodes in expanded_clusters.items():
|
|
248
|
+
for node in nodes:
|
|
249
|
+
out_file.write(f"{cl_id}\t{node}\n")
|
|
250
|
+
|
|
251
|
+
if len(options.get_attributes) > 0:
|
|
252
|
+
fullNet.get_node_attributes(
|
|
253
|
+
options.get_attributes, summary=options.attributes_summarize, output_filename="node_attributes.txt")
|
|
254
|
+
if len(options.get_graph_attributes) > 0:
|
|
255
|
+
fullNet.get_graph_attributes(
|
|
256
|
+
options.get_graph_attributes, output_filename="graph_attributes.txt")
|
|
257
|
+
|
|
258
|
+
def main_randomize_clustering(options):
|
|
259
|
+
options = vars(options)
|
|
260
|
+
if options["seed"]: random.seed(options["seed"])
|
|
261
|
+
cluster2nodes = load_clusters(options)
|
|
262
|
+
random_clusters = {}
|
|
263
|
+
# Reverse should go to expcalc
|
|
264
|
+
node2clusters = {}
|
|
265
|
+
for cluster, nodes in cluster2nodes.items():
|
|
266
|
+
for node in nodes:
|
|
267
|
+
if not node2clusters.get(node):
|
|
268
|
+
node2clusters[node] = [cluster]
|
|
269
|
+
else:
|
|
270
|
+
node2clusters[node].append(cluster)
|
|
271
|
+
if options["random_type"][0] == "hard_fixed" or options["random_type"] == "soft_fixed":
|
|
272
|
+
# Setting universe com and exiled:
|
|
273
|
+
cluster_universe = list(cluster2nodes.keys())
|
|
274
|
+
for node, clusters in node2clusters.items():
|
|
275
|
+
if options["random_type"][0] == "hard_fixed":
|
|
276
|
+
number_clust = len(clusters)
|
|
277
|
+
elif options["random_type"][1] == "soft_fixed":
|
|
278
|
+
u_size = len(cluster_universe)
|
|
279
|
+
k = len(pxc.intersection(clusters,cluster_universe))
|
|
280
|
+
p = k/u_size
|
|
281
|
+
number_clust = random.binomial(u_size, p)
|
|
282
|
+
for clust in random.sample(cluster_universe,number_clust):
|
|
283
|
+
if random_clusters.get(clust):
|
|
284
|
+
random_clusters[clust].append(node)
|
|
285
|
+
else:
|
|
286
|
+
random_clusters[clust] = [node]
|
|
287
|
+
if len(random_clusters[clust]) == len(cluster2nodes[clust]): cluster_universe.remove(clust)
|
|
288
|
+
elif options["random_type"][0] == "not_fixed":
|
|
289
|
+
all_nodes = [ n for cl in cluster2nodes.values() for n in cl ]
|
|
290
|
+
uniq_nodes = pxc.uniq(all_nodes)
|
|
291
|
+
if len(uniq_nodes) < len(all_nodes):
|
|
292
|
+
for cluster, nodes in cluster2nodes.items():
|
|
293
|
+
random_clusters[cluster] = random.sample(uniq_nodes, len(nodes))
|
|
294
|
+
else:
|
|
295
|
+
random.shuffle(all_nodes)
|
|
296
|
+
i = 0
|
|
297
|
+
for cluster, nodes in cluster2nodes:
|
|
298
|
+
random_clusters[cluster] = all_nodes[i, i+len(nodes)]
|
|
299
|
+
i += len(nodes)
|
|
300
|
+
else:
|
|
301
|
+
# Hacemos el procedimiento que se seguia con anterioridad r/nr
|
|
302
|
+
all_sizes = [int(options['random_type'][2])] * int(options['random_type'][1])
|
|
303
|
+
all_nodes = list(node2clusters.keys())
|
|
304
|
+
random_clusters = random_sample(all_nodes, options['random_type'][3] == "r", all_sizes, options['seed'])
|
|
305
|
+
|
|
306
|
+
write_clusters(random_clusters, options['output_file'], options['aggregate_sep'])
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
def main_randomize_network(options):
|
|
310
|
+
fullNet = Net_parser.load(vars(options))
|
|
311
|
+
randomNet = fullNet.randomize_network(options.type_random, **{"seed":options.seed})
|
|
312
|
+
|
|
313
|
+
with open(options.output_file, "w") as outfile:
|
|
314
|
+
for e in randomNet.graph.edges:
|
|
315
|
+
outfile.write(f"{e[0]}\t{e[1]}\n")
|
|
316
|
+
|
|
317
|
+
def worker_ranker(seed_groups, seed_weight, opts, nodes, all_rankings, lock):
|
|
318
|
+
ranker = Ranker()
|
|
319
|
+
if opts["seed_presence"] == "remove":
|
|
320
|
+
ranker.seed_presence = False
|
|
321
|
+
ranker.nodes = nodes
|
|
322
|
+
for sg_id, sg in seed_groups: ranker.seeds[sg_id] = sg
|
|
323
|
+
for sg_id, ws in seed_weight:
|
|
324
|
+
if ws is not None: ranker.weights[sg_id] = ws
|
|
325
|
+
# LOAD KERNEL
|
|
326
|
+
load_kernel(ranker, opts)
|
|
327
|
+
# DO RANKING
|
|
328
|
+
propagate_options = eval('{' + opts["propagate_options"] +'}')
|
|
329
|
+
ranker.do_ranking(cross_validation=opts.get("cross_validation"), propagate=opts["propagate"],
|
|
330
|
+
k_fold=opts.get("k_fold"), metric = opts["representation_seed_metric"], options=propagate_options)
|
|
331
|
+
with lock: # lock avoids that several processes write at same time in the dictionary
|
|
332
|
+
data_package = {} # Create chunks of results to reduce using too much RAM in pickle process and process piping overload
|
|
333
|
+
added_records = 0
|
|
334
|
+
for key, vals in ranker.ranking.items():
|
|
335
|
+
data_package[key] = vals
|
|
336
|
+
added_records += len(vals)
|
|
337
|
+
if added_records > 10000:
|
|
338
|
+
all_rankings.update(data_package)
|
|
339
|
+
data_package = {}
|
|
340
|
+
if len(data_package) > 0: all_rankings.update(data_package) # Write buffered records not writed during loop execution
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
def load_kernel(ranker, opts):
|
|
344
|
+
ranker.matrix = np.load(opts["kernel_files"])
|
|
345
|
+
if opts.get('normalize_matrix') is not None:
|
|
346
|
+
ranker.normalize_matrix(mode=opts["normalize_matrix"])
|
|
347
|
+
if opts.get('whitelist') is not None:
|
|
348
|
+
ranker.filter_matrix(opts["whitelist"])
|
|
349
|
+
ranker.clean_seeds()
|
|
350
|
+
|
|
351
|
+
def sort_records_by_load(records):
|
|
352
|
+
recs = []
|
|
353
|
+
slices = int(chunk_size/2)
|
|
354
|
+
r = chunk_size % 2
|
|
355
|
+
while len(records) > 0:
|
|
356
|
+
recs.append(records.pop(0))
|
|
357
|
+
if len(records) > 0: recs.append(records.pop())
|
|
358
|
+
return recs
|
|
359
|
+
|
|
360
|
+
def main_ranker(options):
|
|
361
|
+
# LOAD RANKER
|
|
362
|
+
ranker = Ranker()
|
|
363
|
+
if options.seed_presence == "remove": # TODO: Probably, this is not necessary right here but on worker_ranker
|
|
364
|
+
ranker.seed_presence = False
|
|
365
|
+
# LOAD SEEDS
|
|
366
|
+
ranker.load_nodes_from_file(options.node_files)
|
|
367
|
+
ranker.load_seeds(options.seed_nodes, sep=options.seed_sep) # TODO: Add when 3 columns is needed for weigths
|
|
368
|
+
ranker.clean_seeds(options.minimum_size)
|
|
369
|
+
discarded_seeds = [ [seed_name, seed] for seed_name, seed in ranker.discarded_seeds.items()]
|
|
370
|
+
if discarded_seeds:
|
|
371
|
+
with open(options.output_file + "_discarded", "w") as f:
|
|
372
|
+
for seed_name, seed in discarded_seeds: f.write(f"{seed_name}\t{options.seed_sep.join(seed)}"+"\n")
|
|
373
|
+
|
|
374
|
+
# DO PARALLEL RANKING
|
|
375
|
+
chunk_size = int(len(ranker.seeds)/options.threads)
|
|
376
|
+
seeds = list(ranker.seeds.items())
|
|
377
|
+
opts = vars(options)
|
|
378
|
+
lock = Lock()
|
|
379
|
+
if options.cross_validation and options.k_fold is None:
|
|
380
|
+
header = ["candidates", "score", "normalized_rank", "rank"]
|
|
381
|
+
else:
|
|
382
|
+
header = ["candidates", "score", "normalized_rank", "rank", "uniq_rank"]
|
|
383
|
+
|
|
384
|
+
with Manager() as manager:
|
|
385
|
+
all_rankings = manager.dict()
|
|
386
|
+
processes = []
|
|
387
|
+
if options.threads > 1:
|
|
388
|
+
worker_threads = options.threads - 1
|
|
389
|
+
seeds.sort(reverse=True, key=lambda x: len(x[1]))
|
|
390
|
+
seeds = sort_records_by_load(seeds)
|
|
391
|
+
else:
|
|
392
|
+
worker_threads = options.threads
|
|
393
|
+
for i in range(worker_threads):
|
|
394
|
+
length = len(seeds)
|
|
395
|
+
offset = length-chunk_size
|
|
396
|
+
records = seeds[offset:length]
|
|
397
|
+
seeds = seeds[0:offset]
|
|
398
|
+
if len(seeds) < chunk_size: records.extend(seeds) # There is no enough record for other chunk so we merge the remanent records t this chunk
|
|
399
|
+
records_weight = [ (record[0], ranker.weights.get(record[0])) for record in records ]
|
|
400
|
+
p = Process(target=worker_ranker, args=(records, records_weight, opts, ranker.nodes, all_rankings, lock))
|
|
401
|
+
processes.append(p)
|
|
402
|
+
p.start()
|
|
403
|
+
for p in processes: p.join() # first we have to start ALL the processes before ask to wait for their termination. For this reason, the join MUST be in an independent loop
|
|
404
|
+
ranker.ranking.update(all_rankings) # COPY RESULTS OUT OF THE MEMORY MAP MANAGER!!!
|
|
405
|
+
ranker.attributes["header"] = header
|
|
406
|
+
|
|
407
|
+
# WRITE RANKING
|
|
408
|
+
if options.seed_presence == "annotate":
|
|
409
|
+
ranker.add_candidate_tag_types()
|
|
410
|
+
|
|
411
|
+
if options.top_n is not None:
|
|
412
|
+
if options.output_top is None:
|
|
413
|
+
ranker.ranking = ranker.get_top(options.top_n)
|
|
414
|
+
else:
|
|
415
|
+
ranker.write_ranking(options.output_top, add_header=options.header, top_n=options.top_n)
|
|
416
|
+
|
|
417
|
+
if options.filter is not None:
|
|
418
|
+
ranker.load_references(options.filter, sep=",")
|
|
419
|
+
if options.cross_validation and options.k_fold is not None:
|
|
420
|
+
ranker.get_seed_cross_validation(k_fold=options.k_fold)
|
|
421
|
+
ranker.ranking = ranker.get_filtered_ranks_by_reference()
|
|
422
|
+
|
|
423
|
+
if options.add_tags is not None:
|
|
424
|
+
tags = load_tags(options.add_tags)
|
|
425
|
+
if options.cross_validation:
|
|
426
|
+
for seed, nodes in ranker.seeds.keys():
|
|
427
|
+
tags[seed] = tags[seed.split("_")[0]]
|
|
428
|
+
seed_col = len(ranker.attributes["header"]) -1
|
|
429
|
+
print(tags)
|
|
430
|
+
print(ranker.ranking)
|
|
431
|
+
for seed, ranking_by_seed in ranker.ranking.items():
|
|
432
|
+
ranker.ranking[seed] = pxc.add_tags(ranking_by_seed, tags[seed], (0,), default_tag = False)
|
|
433
|
+
print(ranker.ranking)
|
|
434
|
+
|
|
435
|
+
if ranker.ranking:
|
|
436
|
+
ranker.write_ranking(f"{options.output_file}_all_candidates", add_header=options.header)
|
|
437
|
+
|
|
438
|
+
def main_text2binary_matrix(options):
|
|
439
|
+
if options.input_file == '-':
|
|
440
|
+
source = sys.stdin
|
|
441
|
+
else:
|
|
442
|
+
source = open(options.input_file)
|
|
443
|
+
|
|
444
|
+
if options.input_type == 'bin':
|
|
445
|
+
matrix = np.load(options.input_file)
|
|
446
|
+
elif options.input_type == 'matrix':
|
|
447
|
+
matrix = load_matrix_file(source)
|
|
448
|
+
elif options.input_type == 'pair':
|
|
449
|
+
matrix, names = load_pair_file(source, options.byte_format)
|
|
450
|
+
with open(options.output_file + ".lst", 'w') as f:
|
|
451
|
+
f.write("\n".join(names))
|
|
452
|
+
|
|
453
|
+
source.close()
|
|
454
|
+
|
|
455
|
+
if options.set_diagonal:
|
|
456
|
+
elements = matrix.shape[-1]
|
|
457
|
+
for n in range(elements):
|
|
458
|
+
matrix[n, n] = 1.0
|
|
459
|
+
|
|
460
|
+
if options.binarize is not None and options.cutoff is None:
|
|
461
|
+
matrix = pxc.filter_cutoff_mat(matrix, options.binarize)
|
|
462
|
+
matrix = pxc.binarize_mat(matrix)
|
|
463
|
+
|
|
464
|
+
if options.cutoff is not None and options.binarize is None:
|
|
465
|
+
matrix = pxc.filter_cutoff_mat(matrix, options.cutoff)
|
|
466
|
+
|
|
467
|
+
if options.stats is not None:
|
|
468
|
+
stats = pxc.get_stats_from_matrix(matrix)
|
|
469
|
+
with open(options.stats, 'w') as f:
|
|
470
|
+
for row in stats:
|
|
471
|
+
f.write("\t".join([str(item) for item in row]) + "\n")
|
|
472
|
+
|
|
473
|
+
if options.output_type == 'bin':
|
|
474
|
+
np.save(options.output_file, matrix)
|
|
475
|
+
elif options.output_type == 'mat':
|
|
476
|
+
np.savetxt(options.output_file, matrix, delimiter='\t')
|
|
477
|
+
|
|
478
|
+
# METHODS FOR NETANALYZER
|
|
479
|
+
#########################
|
|
480
|
+
|
|
481
|
+
def get_args(dsl_args):
|
|
482
|
+
"""return args, kwargs"""
|
|
483
|
+
args = []
|
|
484
|
+
kwargs = {}
|
|
485
|
+
for dsl_arg in dsl_args:
|
|
486
|
+
if '=' in dsl_arg:
|
|
487
|
+
k, v = dsl_arg.split('=', 1)
|
|
488
|
+
kwargs[k] = eval(v)
|
|
489
|
+
else:
|
|
490
|
+
args.append(eval(dsl_arg))
|
|
491
|
+
return args, kwargs
|
|
492
|
+
|
|
493
|
+
def execute_dsl_script(net_obj, dsl_path):
|
|
494
|
+
with open(dsl_path, 'r') as file:
|
|
495
|
+
for line in file:
|
|
496
|
+
line = line.strip()
|
|
497
|
+
if not line or line[0] == '#': continue
|
|
498
|
+
command = line.split()
|
|
499
|
+
func = getattr(net_obj, command.pop(0))
|
|
500
|
+
args, kwargs = get_args(command)
|
|
501
|
+
func(*args, **kwargs)
|
|
502
|
+
|
|
503
|
+
# METHODS FOR RANDOMIZE CLUSTERING
|
|
504
|
+
###################################
|
|
505
|
+
|
|
506
|
+
def load_clusters(options):
|
|
507
|
+
clusters = {}
|
|
508
|
+
with open(options['input_file'], "r") as f:
|
|
509
|
+
for line in f:
|
|
510
|
+
line = line.rstrip().split(options['column_sep'])
|
|
511
|
+
cluster = line[options['cluster_index']]
|
|
512
|
+
node = line[options['node_index']]
|
|
513
|
+
if options.get('node_sep') != None:
|
|
514
|
+
node = node.split(options['node_sep'])
|
|
515
|
+
clusters[cluster] = node
|
|
516
|
+
else:
|
|
517
|
+
query = clusters.get(cluster)
|
|
518
|
+
if query == None:
|
|
519
|
+
clusters[cluster] = [node]
|
|
520
|
+
else:
|
|
521
|
+
query.append(node)
|
|
522
|
+
return clusters
|
|
523
|
+
|
|
524
|
+
def random_sample(nodes, replacement, all_sizes, seed):
|
|
525
|
+
random_clusters = {}
|
|
526
|
+
node_list = copy.deepcopy(nodes)
|
|
527
|
+
random.seed(seed)
|
|
528
|
+
for counter, cluster_size in enumerate(all_sizes):
|
|
529
|
+
if cluster_size > len(node_list) and not replacement: sys.exit("Not enough nodes to generate clusters. Please activate replacement or change random mode")
|
|
530
|
+
random_nodes = random.sample(node_list, cluster_size)
|
|
531
|
+
if not replacement:
|
|
532
|
+
node_list = pxc.diff(node_list, random_nodes)
|
|
533
|
+
random_clusters[f"{counter}_random"] = random_nodes
|
|
534
|
+
return random_clusters
|
|
535
|
+
|
|
536
|
+
def write_clusters(clusters, output_file, sep): #2expcalc
|
|
537
|
+
with open(output_file, "w") as outfile:
|
|
538
|
+
for cluster, nodes in clusters.items():
|
|
539
|
+
if sep != None: nodes = [sep.join(nodes)]
|
|
540
|
+
for node in nodes:
|
|
541
|
+
outfile.write(f"{cluster}\t{node}\n")
|
|
542
|
+
|
|
543
|
+
# METHODS FOR RANKER
|
|
544
|
+
#####################
|
|
545
|
+
|
|
546
|
+
def open_whitelist(file):
|
|
547
|
+
whitelist = []
|
|
548
|
+
with open(file, "r") as f:
|
|
549
|
+
for line in f:
|
|
550
|
+
node = line.strip()
|
|
551
|
+
whitelist.append(node)
|
|
552
|
+
return whitelist
|
|
553
|
+
|
|
554
|
+
def load_tags(file):
|
|
555
|
+
tags = {}
|
|
556
|
+
with open(file, "r") as f:
|
|
557
|
+
for line in f:
|
|
558
|
+
line = line.strip().split("\t")
|
|
559
|
+
pxc.add_nested_value(tags, (line[0],line[1]), line[2])
|
|
560
|
+
return tags
|
|
561
|
+
|
|
562
|
+
# METHODS FOR TEXT2BINARY
|
|
563
|
+
#########################
|
|
564
|
+
|
|
565
|
+
def load_matrix_file(source, splitChar = "\t"):
|
|
566
|
+
matrix = None
|
|
567
|
+
counter = 0
|
|
568
|
+
for line in source:
|
|
569
|
+
line = line.strip()
|
|
570
|
+
|
|
571
|
+
row = [float(c) for c in line.split(splitChar)]
|
|
572
|
+
if matrix is None:
|
|
573
|
+
matrix = np.zeros((len(row), len(row)))
|
|
574
|
+
for i, val in enumerate(row):
|
|
575
|
+
matrix[counter, i] = val
|
|
576
|
+
counter += 1
|
|
577
|
+
|
|
578
|
+
return matrix
|
|
579
|
+
|
|
580
|
+
def load_pair_file(source, byte_format = "float32"):
|
|
581
|
+
# Not used byte_forma parameter
|
|
582
|
+
connections = {}
|
|
583
|
+
for line in source:
|
|
584
|
+
line = line.strip().split("\t")
|
|
585
|
+
if len(line) == 3:
|
|
586
|
+
node_a, node_b, weight = line
|
|
587
|
+
weight = float(weight)
|
|
588
|
+
else:
|
|
589
|
+
node_a, node_b = line
|
|
590
|
+
weight = 1.0
|
|
591
|
+
pxc.add_nested_value(connections, (node_a, node_b), weight)
|
|
592
|
+
pxc.add_nested_value(connections, (node_b, node_a), weight)
|
|
593
|
+
|
|
594
|
+
matrix, names = dicti2wmatrix_squared(connections)
|
|
595
|
+
return matrix, names
|
|
596
|
+
|
|
597
|
+
def dicti2wmatrix_squared(dicti,symm= True):
|
|
598
|
+
element_names = dicti.keys()
|
|
599
|
+
matrix = np.zeros((len(element_names), len(element_names)))
|
|
600
|
+
i = 0
|
|
601
|
+
for elementA, relations in dicti.items():
|
|
602
|
+
for j, elementB in enumerate(element_names):
|
|
603
|
+
if elementA != elementB:
|
|
604
|
+
query = relations.get(elementB)
|
|
605
|
+
if query is not None:
|
|
606
|
+
matrix[i, j] = query
|
|
607
|
+
if symm:
|
|
608
|
+
matrix[j, i] = query
|
|
609
|
+
i += 1
|
|
610
|
+
return matrix, element_names
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
import numpy
|
|
2
|
+
from NetAnalyzer.netanalyzer import NetAnalyzer
|
|
3
|
+
|
|
4
|
+
class Net_parser:
|
|
5
|
+
|
|
6
|
+
def load(options):
|
|
7
|
+
net = None
|
|
8
|
+
if options['input_format'] == 'pair':
|
|
9
|
+
net = Net_parser.load_network_by_pairs(options['input_file'], options['layers'], options['split_char'])
|
|
10
|
+
elif options['input_format'] == 'bin':
|
|
11
|
+
net = Net_parser.load_network_by_bin_matrix(options['input_file'], options['node_files'], options['layers'])
|
|
12
|
+
elif options['input_format'] == 'matrix':
|
|
13
|
+
net = Net_parser.load_network_by_plain_matrix(options['input_file'], options['node_files'], options['layers'], options['split_char'])
|
|
14
|
+
else:
|
|
15
|
+
raise("ERROR: The format " + options['input_format'] + " is not defined")
|
|
16
|
+
|
|
17
|
+
if options.get('load_both'): # TODO: Not tested Yet.
|
|
18
|
+
if not net.graph:
|
|
19
|
+
layerA, layerB = list(net.matrices["adjacency_matrices"].keys())[0]
|
|
20
|
+
net.adjMat2netObj(layerA,layerB)
|
|
21
|
+
if net.matrices["adjacency_matrices"] == {}:
|
|
22
|
+
net.generate_all_biadjs()
|
|
23
|
+
|
|
24
|
+
return net
|
|
25
|
+
|
|
26
|
+
def load_network_by_pairs(file, layers, split_character="\t"):
|
|
27
|
+
net = NetAnalyzer([layer[0] for layer in layers])
|
|
28
|
+
f = open(file)
|
|
29
|
+
for line in f:
|
|
30
|
+
pair = line.rstrip().split(split_character)
|
|
31
|
+
node1 = pair[0]
|
|
32
|
+
node2 = pair[1]
|
|
33
|
+
net.add_node(node1, net.set_layer(layers, node1))
|
|
34
|
+
net.add_node(node2, net.set_layer(layers, node2))
|
|
35
|
+
if len(pair) == 3:
|
|
36
|
+
net.add_edge(node1, node2, weight=float(pair[2]))
|
|
37
|
+
else:
|
|
38
|
+
net.add_edge(node1, node2)
|
|
39
|
+
|
|
40
|
+
net.add_edge(node1, node2)
|
|
41
|
+
f.close()
|
|
42
|
+
return net
|
|
43
|
+
|
|
44
|
+
def load_network_by_bin_matrix(input_file, node_file, layers):
|
|
45
|
+
tag_layers = tuple([layer[0] for layer in layers])
|
|
46
|
+
net = NetAnalyzer(tag_layers)
|
|
47
|
+
if len(node_file) == 1:
|
|
48
|
+
node_names = Net_parser.load_input_list(node_file[0])
|
|
49
|
+
row_names = col_names = node_names
|
|
50
|
+
else:
|
|
51
|
+
row_names = Net_parser.load_input_list(node_file[0])
|
|
52
|
+
col_names = Net_parser.load_input_list(node_file[1])
|
|
53
|
+
if len(tag_layers) == 1:
|
|
54
|
+
net.matrices["adjacency_matrices"][(tag_layers[0],tag_layers[0])] = [numpy.load(input_file), row_names, col_names]
|
|
55
|
+
else:
|
|
56
|
+
net.matrices["adjacency_matrices"][tag_layers] = [numpy.load(input_file), row_names, col_names]
|
|
57
|
+
return net
|
|
58
|
+
|
|
59
|
+
def load_network_by_plain_matrix(input_file, node_file, layers, splitChar="\t"):
|
|
60
|
+
tag_layers = tuple([layer[0] for layer in layers])
|
|
61
|
+
net = NetAnalyzer(tag_layers)
|
|
62
|
+
if len(node_file) == 1:
|
|
63
|
+
node_names = Net_parser.load_input_list(node_file[0])
|
|
64
|
+
row_names = col_names = node_names
|
|
65
|
+
else:
|
|
66
|
+
row_names = Net_parser.load_input_list(node_file[0])
|
|
67
|
+
col_names = Net_parser.load_input_list(node_file[1])
|
|
68
|
+
if len(tag_layers) == 1:
|
|
69
|
+
net.matrices["adjacency_matrices"][(tag_layers[0],tag_layers[0])] = [numpy.genfromtxt(input_file, delimiter=splitChar), row_names, col_names]
|
|
70
|
+
else:
|
|
71
|
+
net.matrices["adjacency_matrices"][tag_layers] = [numpy.genfromtxt(input_file, delimiter=splitChar), row_names, col_names]
|
|
72
|
+
return net
|
|
73
|
+
|
|
74
|
+
def load_input_list(input_path):
|
|
75
|
+
file = open(input_path, "r")
|
|
76
|
+
input_data = file.readlines()
|
|
77
|
+
file.close()
|
|
78
|
+
return [line.rstrip() for line in input_data]
|