NetAnalyzer 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,610 @@
1
+ import sys
2
+ import os
3
+ import numpy as np
4
+ import random
5
+ import copy
6
+ from multiprocessing import Process, Manager, Lock
7
+ from py_cmdtabs.cmdtabs import CmdTabs
8
+ import py_exp_calc.exp_calc as pxc
9
+ from py_report_html import Py_report_html
10
+ from NetAnalyzer import Net_parser, NetAnalyzer
11
+ from NetAnalyzer import Kernels
12
+ from NetAnalyzer import Net_parser, NetAnalyzer
13
+ from NetAnalyzer import Ranker
14
+ from NetAnalyzer import Graph2sim
15
+ from NetAnalyzer import Adv_mat_calc
16
+ from NetAnalyzer.performancer import Performancer
17
+ from NetAnalyzer.seed_parser import SeedParser
18
+ import networkx as nx
19
+
20
+ def main_net_explorer(options, test = False):
21
+ # loading gene seeds.
22
+ options = vars(options)
23
+ if options["seed_nodes"]: seeds2explore, _ = SeedParser.load_nodes_by_group(options["seed_nodes"], sep=options["seed_sep"])
24
+
25
+ # Loading multinet operations
26
+ multinet = {}
27
+ for net_id, net_file in options["input_file"].items():
28
+ nodes = options['node_files'][net_id]
29
+ multinet[net_id] = Net_parser.load_network_by_bin_matrix(net_file, [nodes], [['layer', '-']])
30
+ cutoff = options["layer_cutoff"].get(net_id)
31
+ if options["no_autorelations"]: np.fill_diagonal(multinet[net_id].matrices['adjacency_matrices'][('layer','layer')][0], 0)
32
+ if cutoff:
33
+ cutoff = float(cutoff)
34
+ multinet[net_id].filter_matrix( mat_keys=('adjacency_matrices',('layer','layer')),operation='filter_cutoff', options={'cutoff': cutoff}, add_to_object=True)
35
+ multinet[net_id].adjMat2netObj('layer','layer')
36
+
37
+ # extract a subgraph for each
38
+ if options["seed_nodes"]:
39
+ seeds2subgraph = {}
40
+ seeds2lcc = {}
41
+ for seed, nodes in seeds2explore.items():
42
+ seeds2subgraph[seed] = {}
43
+ seeds2lcc[seed] = {}
44
+ for net_id, net in multinet.items():
45
+ # get neighbor from node
46
+ nodes_with_neigh = set(nodes)
47
+ neigh_level = int(options["neigh_level"].get(net_id)) if options["neigh_level"].get(net_id) else 0
48
+ for i in range(0, neigh_level): nodes_with_neigh = get_neigh_set(net, nodes_with_neigh)
49
+ seeds2subgraph[seed][net_id] = net.graph.subgraph(nodes_with_neigh)
50
+ largest_cc = len(max(nx.connected_components(seeds2subgraph[seed][net_id]), key=len))
51
+ seeds2lcc[seed][net_id] = largest_cc
52
+
53
+ # # If mention, add node2vec coordinates with a tnse proyection.
54
+ net2embedding_proj = None
55
+ if options["embedding_proj"]:
56
+ net2embedding_proj = {}
57
+ for net_id, net in multinet.items():
58
+ adj_mat, embedding_nodes, _ = net.matrices["adjacency_matrices"][("layer", "layer")]
59
+ emb_coords = Graph2sim.get_embedding(adj_mat, embedding = "node2vec", embedding_nodes=embedding_nodes)
60
+ umap_coords = Adv_mat_calc.data2umap(emb_coords, n_neighbors = 15, min_dist = 0.1, n_components = 2, metric = 'euclidean', random_seed = None)
61
+ net2embedding_proj[net_id] = [umap_coords, embedding_nodes]
62
+
63
+ # Comparing nets:
64
+ if options["compare_nets"]:
65
+ network_ids = list(options["input_file"].keys())
66
+ num_nets = len(network_ids)
67
+ sim_nets = np.zeros((num_nets,num_nets))
68
+ for i, net_i in enumerate(network_ids):
69
+ for j,net_j in enumerate(network_ids[i:len(network_ids)]):
70
+ edges_i = set(multinet[net_i].graph.edges)
71
+ edges_j = set(multinet[net_j].graph.edges)
72
+ sim_i_j = len(edges_i & edges_j) / min(len(edges_i), len(edges_j))
73
+ sim_nets[i,j+i] = sim_i_j
74
+ sim_nets[j+i,i] = sim_i_j
75
+ net_sims = [sim_nets, network_ids]
76
+
77
+
78
+ # Execute the reports in the process.
79
+ template = open(str(os.path.join(os.path.dirname(__file__), 'templates','net_explorer.txt'))).read()
80
+ container = {"seeds2explore": seeds2explore, "seeds2subgraph": seeds2subgraph, "seeds2lcc":seeds2lcc, "net2embedding_proj": net2embedding_proj, "net_sims": net_sims, "plot_method": options["plot_network_method"]}
81
+
82
+ report = Py_report_html(container, 'Network explorer')
83
+ report.build(template)
84
+ report.write(options["output_file"]+".html")
85
+
86
+ if test: return container
87
+
88
+ def get_neigh_set(net, nodes):
89
+ neigh = set(nodes)
90
+ for i,n in enumerate(nodes):
91
+ try:
92
+ neigh = neigh | set(net.graph.neighbors(n))
93
+ except:
94
+ continue
95
+ return list(neigh)
96
+
97
+
98
+ def main_integrate_kernels(options):
99
+ kernels = Kernels()
100
+
101
+ if not options.kernel_ids:
102
+ options.kernel_ids = list(range(0,len(options.kernel_files)))
103
+ options.kernel_ids = [str(k) for k in options.kernel_ids]
104
+
105
+ # TODO: Consider adding more options to integrate to accept other formats
106
+ if options.input_format == "bin":
107
+ kernels.load_kernels_by_bin_matrixes(options.kernel_files, options.node_files, options.kernel_ids)
108
+ kernels.create_general_index()
109
+
110
+ if not options.raw_values:
111
+ kernels.move2zero_reference()
112
+
113
+ if options.integration_type is not None:
114
+ print(options.threads)
115
+ kernels.integrate_matrix(options.integration_type, options.threads, options.symmetric)
116
+
117
+ if options.output_file is not None:
118
+ kernel, names = kernels.integrated_kernel
119
+ np.save(options.output_file, kernel)
120
+
121
+ with open(options.output_file +'.lst', 'w') as f:
122
+ for name in names:
123
+ f.write(name + "\n")
124
+
125
+ def main_netanalyzer(options):
126
+ print("Loading network data")
127
+ opts = vars(options)
128
+ # FRED: Remove this part of vars and modify the loads methods (Tlk wth PSZ)
129
+ fullNet = Net_parser.load(opts)
130
+ fullNet.set_compute_pairs(options.use_pairs, not options.no_autorelations)
131
+ fullNet.threads = options.threads
132
+
133
+ fullNet.reference_nodes = options.reference_nodes
134
+ for ont_data in opts['ontologies']:
135
+ layer_name, ontology_file_path = ont_data
136
+ fullNet.link_ontology(ontology_file_path, layer_name)
137
+ fullNet.ontologies
138
+
139
+ if options.group_nodes:
140
+ fullNet.set_groups(options.group_nodes)
141
+
142
+ if options.delete_nodes:
143
+ node_list = CmdTabs.load_input_data(options.delete_nodes[0])
144
+ node_list = [item for sublist in node_list for item in sublist]
145
+ mode = options.delete_nodes[1] if len(options.delete_nodes) > 1 else 'd'
146
+ fullNet.delete_nodes(node_list, mode)
147
+
148
+ if options.dsl_script is not None:
149
+ execute_dsl_script(fullNet, options.dsl_script)
150
+ sys.exit()
151
+
152
+ if options.meth is not None:
153
+ print(f"Performing association method {options.meth} on network \n")
154
+ if options.meth == "transference":
155
+ if not (options.use_layers[0][0], options.use_layers[1][0]) in fullNet.matrices["adjacency_matrices"]:
156
+ fullNet.generate_adjacency_matrix(
157
+ options.use_layers[0][0], options.use_layers[1][0])
158
+ if not (options.use_layers[1][0], options.use_layers[0][1]) in fullNet.matrices["adjacency_matrices"]:
159
+ fullNet.generate_adjacency_matrix(
160
+ options.use_layers[1][0], options.use_layers[0][1])
161
+
162
+ fullNet.get_association_values(
163
+ (options.use_layers[0][0], options.use_layers[1][0]),
164
+ (options.use_layers[1][0], options.use_layers[0][1]),
165
+ "transference")
166
+ else:
167
+ fullNet.get_association_values(
168
+ options.use_layers[0],
169
+ options.use_layers[1][0],
170
+ options.meth)
171
+
172
+ with open(options.assoc_file, 'w') as f:
173
+ for val in fullNet.association_values[options.meth][:-1]:
174
+ f.write("\t".join(map(str, val)) + "\n")
175
+ f.write(
176
+ "\t".join(map(str, fullNet.association_values[options.meth][-1])))
177
+ print(f"End of analysis: {options.meth}")
178
+
179
+ if options.control_file != None:
180
+ with open(options.control_file, "r") as f:
181
+ control = [control.append(line.rstrip().split("\t")) for line in f]
182
+ Performancer.load_control(control)
183
+ predictions = fullNet.association_values[options.meth]
184
+ performance = Performancer.get_pred_rec(predictions)
185
+ with open(options.performance_file, 'r') as f:
186
+ f.write("\t".join(['cut', 'prec', 'rec', 'meth']) + "\n")
187
+ for item in performance:
188
+ item.append(options['meth'])
189
+ f.write("\t".join(item) + "\n")
190
+
191
+ if options.kernel is not None:
192
+ # This allows inject custom arguments for each embedding method
193
+ embedding_kwargs = eval('{' +options.embedding_add_options +'}')
194
+ if len(options.use_layers) == 1:
195
+ # we use only a layer to perform the kernel, so only one item it is selected.
196
+ layers2kernel = (options.use_layers[0][0], options.use_layers[0][0])
197
+ else:
198
+ layers2kernel = tuple(options.use_layers[0])
199
+
200
+
201
+ fullNet.get_kernel(layers2kernel, options.kernel, options.normalize_kernel,
202
+ options.coords2sim_type, embedding_kwargs, add_to_object=True)
203
+ fullNet.write_kernel(layers2kernel, options.kernel, options.kernel_file)
204
+
205
+ if options.graph_file is not None:
206
+ options.graph_options['output_file'] = options.graph_file
207
+ fullNet.plot_network(options.graph_options)
208
+
209
+ # Group creation
210
+ if options.build_cluster_alg is not None:
211
+ clust_kwargs = eval('{' + options.build_clusters_add_options +'}')
212
+ fullNet.discover_clusters(
213
+ options.build_cluster_alg, clust_kwargs, **{'seed': options.seed})
214
+
215
+ if options.output_build_clusters is None:
216
+ options.output_build_clusters = options.build_cluster_alg + \
217
+ '_' + 'discovered_clusters.txt'
218
+
219
+ with open(options.output_build_clusters, 'w') as out_file:
220
+ for cl_id, nodes in fullNet.group_nodes.items():
221
+ for node in nodes:
222
+ out_file.write(f"{cl_id}\t{node}\n")
223
+
224
+ # Group metrics by cluster.
225
+ if options.group_metrics:
226
+ fullNet.compute_group_metrics(
227
+ output_filename=options.output_metrics_by_cluster, metrics=options.group_metrics)
228
+
229
+ # Group metrics summarized.
230
+ if options.summarize_metrics:
231
+ fullNet.compute_summarized_group_metrics(
232
+ output_filename=options.output_summarized_metrics, metrics=options.summarize_metrics)
233
+
234
+ # Comparing Group Families (Two by now)
235
+ if options.compare_clusters_reference is not None:
236
+ fullNet.group_reference = options.compare_clusters_reference
237
+ fullNet.join_clusters(options.compare_clusters_join)
238
+ results = fullNet.compare_partitions(options.overlapping_communities)
239
+ for metric_name, metric_value in results.items():
240
+ print(f"{metric_name}\t{metric_value}")
241
+
242
+ # Group Expansion
243
+ if options.expand_clusters is not None:
244
+ expanded_clusters = fullNet.expand_clusters(
245
+ options.expand_clusters, options.one_sht_pairs)
246
+ with open(options.output_expand_clusters, 'w') as out_file:
247
+ for cl_id, nodes in expanded_clusters.items():
248
+ for node in nodes:
249
+ out_file.write(f"{cl_id}\t{node}\n")
250
+
251
+ if len(options.get_attributes) > 0:
252
+ fullNet.get_node_attributes(
253
+ options.get_attributes, summary=options.attributes_summarize, output_filename="node_attributes.txt")
254
+ if len(options.get_graph_attributes) > 0:
255
+ fullNet.get_graph_attributes(
256
+ options.get_graph_attributes, output_filename="graph_attributes.txt")
257
+
258
+ def main_randomize_clustering(options):
259
+ options = vars(options)
260
+ if options["seed"]: random.seed(options["seed"])
261
+ cluster2nodes = load_clusters(options)
262
+ random_clusters = {}
263
+ # Reverse should go to expcalc
264
+ node2clusters = {}
265
+ for cluster, nodes in cluster2nodes.items():
266
+ for node in nodes:
267
+ if not node2clusters.get(node):
268
+ node2clusters[node] = [cluster]
269
+ else:
270
+ node2clusters[node].append(cluster)
271
+ if options["random_type"][0] == "hard_fixed" or options["random_type"] == "soft_fixed":
272
+ # Setting universe com and exiled:
273
+ cluster_universe = list(cluster2nodes.keys())
274
+ for node, clusters in node2clusters.items():
275
+ if options["random_type"][0] == "hard_fixed":
276
+ number_clust = len(clusters)
277
+ elif options["random_type"][1] == "soft_fixed":
278
+ u_size = len(cluster_universe)
279
+ k = len(pxc.intersection(clusters,cluster_universe))
280
+ p = k/u_size
281
+ number_clust = random.binomial(u_size, p)
282
+ for clust in random.sample(cluster_universe,number_clust):
283
+ if random_clusters.get(clust):
284
+ random_clusters[clust].append(node)
285
+ else:
286
+ random_clusters[clust] = [node]
287
+ if len(random_clusters[clust]) == len(cluster2nodes[clust]): cluster_universe.remove(clust)
288
+ elif options["random_type"][0] == "not_fixed":
289
+ all_nodes = [ n for cl in cluster2nodes.values() for n in cl ]
290
+ uniq_nodes = pxc.uniq(all_nodes)
291
+ if len(uniq_nodes) < len(all_nodes):
292
+ for cluster, nodes in cluster2nodes.items():
293
+ random_clusters[cluster] = random.sample(uniq_nodes, len(nodes))
294
+ else:
295
+ random.shuffle(all_nodes)
296
+ i = 0
297
+ for cluster, nodes in cluster2nodes:
298
+ random_clusters[cluster] = all_nodes[i, i+len(nodes)]
299
+ i += len(nodes)
300
+ else:
301
+ # Hacemos el procedimiento que se seguia con anterioridad r/nr
302
+ all_sizes = [int(options['random_type'][2])] * int(options['random_type'][1])
303
+ all_nodes = list(node2clusters.keys())
304
+ random_clusters = random_sample(all_nodes, options['random_type'][3] == "r", all_sizes, options['seed'])
305
+
306
+ write_clusters(random_clusters, options['output_file'], options['aggregate_sep'])
307
+
308
+
309
+ def main_randomize_network(options):
310
+ fullNet = Net_parser.load(vars(options))
311
+ randomNet = fullNet.randomize_network(options.type_random, **{"seed":options.seed})
312
+
313
+ with open(options.output_file, "w") as outfile:
314
+ for e in randomNet.graph.edges:
315
+ outfile.write(f"{e[0]}\t{e[1]}\n")
316
+
317
+ def worker_ranker(seed_groups, seed_weight, opts, nodes, all_rankings, lock):
318
+ ranker = Ranker()
319
+ if opts["seed_presence"] == "remove":
320
+ ranker.seed_presence = False
321
+ ranker.nodes = nodes
322
+ for sg_id, sg in seed_groups: ranker.seeds[sg_id] = sg
323
+ for sg_id, ws in seed_weight:
324
+ if ws is not None: ranker.weights[sg_id] = ws
325
+ # LOAD KERNEL
326
+ load_kernel(ranker, opts)
327
+ # DO RANKING
328
+ propagate_options = eval('{' + opts["propagate_options"] +'}')
329
+ ranker.do_ranking(cross_validation=opts.get("cross_validation"), propagate=opts["propagate"],
330
+ k_fold=opts.get("k_fold"), metric = opts["representation_seed_metric"], options=propagate_options)
331
+ with lock: # lock avoids that several processes write at same time in the dictionary
332
+ data_package = {} # Create chunks of results to reduce using too much RAM in pickle process and process piping overload
333
+ added_records = 0
334
+ for key, vals in ranker.ranking.items():
335
+ data_package[key] = vals
336
+ added_records += len(vals)
337
+ if added_records > 10000:
338
+ all_rankings.update(data_package)
339
+ data_package = {}
340
+ if len(data_package) > 0: all_rankings.update(data_package) # Write buffered records not writed during loop execution
341
+
342
+
343
+ def load_kernel(ranker, opts):
344
+ ranker.matrix = np.load(opts["kernel_files"])
345
+ if opts.get('normalize_matrix') is not None:
346
+ ranker.normalize_matrix(mode=opts["normalize_matrix"])
347
+ if opts.get('whitelist') is not None:
348
+ ranker.filter_matrix(opts["whitelist"])
349
+ ranker.clean_seeds()
350
+
351
+ def sort_records_by_load(records):
352
+ recs = []
353
+ slices = int(chunk_size/2)
354
+ r = chunk_size % 2
355
+ while len(records) > 0:
356
+ recs.append(records.pop(0))
357
+ if len(records) > 0: recs.append(records.pop())
358
+ return recs
359
+
360
+ def main_ranker(options):
361
+ # LOAD RANKER
362
+ ranker = Ranker()
363
+ if options.seed_presence == "remove": # TODO: Probably, this is not necessary right here but on worker_ranker
364
+ ranker.seed_presence = False
365
+ # LOAD SEEDS
366
+ ranker.load_nodes_from_file(options.node_files)
367
+ ranker.load_seeds(options.seed_nodes, sep=options.seed_sep) # TODO: Add when 3 columns is needed for weigths
368
+ ranker.clean_seeds(options.minimum_size)
369
+ discarded_seeds = [ [seed_name, seed] for seed_name, seed in ranker.discarded_seeds.items()]
370
+ if discarded_seeds:
371
+ with open(options.output_file + "_discarded", "w") as f:
372
+ for seed_name, seed in discarded_seeds: f.write(f"{seed_name}\t{options.seed_sep.join(seed)}"+"\n")
373
+
374
+ # DO PARALLEL RANKING
375
+ chunk_size = int(len(ranker.seeds)/options.threads)
376
+ seeds = list(ranker.seeds.items())
377
+ opts = vars(options)
378
+ lock = Lock()
379
+ if options.cross_validation and options.k_fold is None:
380
+ header = ["candidates", "score", "normalized_rank", "rank"]
381
+ else:
382
+ header = ["candidates", "score", "normalized_rank", "rank", "uniq_rank"]
383
+
384
+ with Manager() as manager:
385
+ all_rankings = manager.dict()
386
+ processes = []
387
+ if options.threads > 1:
388
+ worker_threads = options.threads - 1
389
+ seeds.sort(reverse=True, key=lambda x: len(x[1]))
390
+ seeds = sort_records_by_load(seeds)
391
+ else:
392
+ worker_threads = options.threads
393
+ for i in range(worker_threads):
394
+ length = len(seeds)
395
+ offset = length-chunk_size
396
+ records = seeds[offset:length]
397
+ seeds = seeds[0:offset]
398
+ if len(seeds) < chunk_size: records.extend(seeds) # There is no enough record for other chunk so we merge the remanent records t this chunk
399
+ records_weight = [ (record[0], ranker.weights.get(record[0])) for record in records ]
400
+ p = Process(target=worker_ranker, args=(records, records_weight, opts, ranker.nodes, all_rankings, lock))
401
+ processes.append(p)
402
+ p.start()
403
+ for p in processes: p.join() # first we have to start ALL the processes before ask to wait for their termination. For this reason, the join MUST be in an independent loop
404
+ ranker.ranking.update(all_rankings) # COPY RESULTS OUT OF THE MEMORY MAP MANAGER!!!
405
+ ranker.attributes["header"] = header
406
+
407
+ # WRITE RANKING
408
+ if options.seed_presence == "annotate":
409
+ ranker.add_candidate_tag_types()
410
+
411
+ if options.top_n is not None:
412
+ if options.output_top is None:
413
+ ranker.ranking = ranker.get_top(options.top_n)
414
+ else:
415
+ ranker.write_ranking(options.output_top, add_header=options.header, top_n=options.top_n)
416
+
417
+ if options.filter is not None:
418
+ ranker.load_references(options.filter, sep=",")
419
+ if options.cross_validation and options.k_fold is not None:
420
+ ranker.get_seed_cross_validation(k_fold=options.k_fold)
421
+ ranker.ranking = ranker.get_filtered_ranks_by_reference()
422
+
423
+ if options.add_tags is not None:
424
+ tags = load_tags(options.add_tags)
425
+ if options.cross_validation:
426
+ for seed, nodes in ranker.seeds.keys():
427
+ tags[seed] = tags[seed.split("_")[0]]
428
+ seed_col = len(ranker.attributes["header"]) -1
429
+ print(tags)
430
+ print(ranker.ranking)
431
+ for seed, ranking_by_seed in ranker.ranking.items():
432
+ ranker.ranking[seed] = pxc.add_tags(ranking_by_seed, tags[seed], (0,), default_tag = False)
433
+ print(ranker.ranking)
434
+
435
+ if ranker.ranking:
436
+ ranker.write_ranking(f"{options.output_file}_all_candidates", add_header=options.header)
437
+
438
+ def main_text2binary_matrix(options):
439
+ if options.input_file == '-':
440
+ source = sys.stdin
441
+ else:
442
+ source = open(options.input_file)
443
+
444
+ if options.input_type == 'bin':
445
+ matrix = np.load(options.input_file)
446
+ elif options.input_type == 'matrix':
447
+ matrix = load_matrix_file(source)
448
+ elif options.input_type == 'pair':
449
+ matrix, names = load_pair_file(source, options.byte_format)
450
+ with open(options.output_file + ".lst", 'w') as f:
451
+ f.write("\n".join(names))
452
+
453
+ source.close()
454
+
455
+ if options.set_diagonal:
456
+ elements = matrix.shape[-1]
457
+ for n in range(elements):
458
+ matrix[n, n] = 1.0
459
+
460
+ if options.binarize is not None and options.cutoff is None:
461
+ matrix = pxc.filter_cutoff_mat(matrix, options.binarize)
462
+ matrix = pxc.binarize_mat(matrix)
463
+
464
+ if options.cutoff is not None and options.binarize is None:
465
+ matrix = pxc.filter_cutoff_mat(matrix, options.cutoff)
466
+
467
+ if options.stats is not None:
468
+ stats = pxc.get_stats_from_matrix(matrix)
469
+ with open(options.stats, 'w') as f:
470
+ for row in stats:
471
+ f.write("\t".join([str(item) for item in row]) + "\n")
472
+
473
+ if options.output_type == 'bin':
474
+ np.save(options.output_file, matrix)
475
+ elif options.output_type == 'mat':
476
+ np.savetxt(options.output_file, matrix, delimiter='\t')
477
+
478
+ # METHODS FOR NETANALYZER
479
+ #########################
480
+
481
+ def get_args(dsl_args):
482
+ """return args, kwargs"""
483
+ args = []
484
+ kwargs = {}
485
+ for dsl_arg in dsl_args:
486
+ if '=' in dsl_arg:
487
+ k, v = dsl_arg.split('=', 1)
488
+ kwargs[k] = eval(v)
489
+ else:
490
+ args.append(eval(dsl_arg))
491
+ return args, kwargs
492
+
493
+ def execute_dsl_script(net_obj, dsl_path):
494
+ with open(dsl_path, 'r') as file:
495
+ for line in file:
496
+ line = line.strip()
497
+ if not line or line[0] == '#': continue
498
+ command = line.split()
499
+ func = getattr(net_obj, command.pop(0))
500
+ args, kwargs = get_args(command)
501
+ func(*args, **kwargs)
502
+
503
+ # METHODS FOR RANDOMIZE CLUSTERING
504
+ ###################################
505
+
506
+ def load_clusters(options):
507
+ clusters = {}
508
+ with open(options['input_file'], "r") as f:
509
+ for line in f:
510
+ line = line.rstrip().split(options['column_sep'])
511
+ cluster = line[options['cluster_index']]
512
+ node = line[options['node_index']]
513
+ if options.get('node_sep') != None:
514
+ node = node.split(options['node_sep'])
515
+ clusters[cluster] = node
516
+ else:
517
+ query = clusters.get(cluster)
518
+ if query == None:
519
+ clusters[cluster] = [node]
520
+ else:
521
+ query.append(node)
522
+ return clusters
523
+
524
+ def random_sample(nodes, replacement, all_sizes, seed):
525
+ random_clusters = {}
526
+ node_list = copy.deepcopy(nodes)
527
+ random.seed(seed)
528
+ for counter, cluster_size in enumerate(all_sizes):
529
+ if cluster_size > len(node_list) and not replacement: sys.exit("Not enough nodes to generate clusters. Please activate replacement or change random mode")
530
+ random_nodes = random.sample(node_list, cluster_size)
531
+ if not replacement:
532
+ node_list = pxc.diff(node_list, random_nodes)
533
+ random_clusters[f"{counter}_random"] = random_nodes
534
+ return random_clusters
535
+
536
+ def write_clusters(clusters, output_file, sep): #2expcalc
537
+ with open(output_file, "w") as outfile:
538
+ for cluster, nodes in clusters.items():
539
+ if sep != None: nodes = [sep.join(nodes)]
540
+ for node in nodes:
541
+ outfile.write(f"{cluster}\t{node}\n")
542
+
543
+ # METHODS FOR RANKER
544
+ #####################
545
+
546
+ def open_whitelist(file):
547
+ whitelist = []
548
+ with open(file, "r") as f:
549
+ for line in f:
550
+ node = line.strip()
551
+ whitelist.append(node)
552
+ return whitelist
553
+
554
+ def load_tags(file):
555
+ tags = {}
556
+ with open(file, "r") as f:
557
+ for line in f:
558
+ line = line.strip().split("\t")
559
+ pxc.add_nested_value(tags, (line[0],line[1]), line[2])
560
+ return tags
561
+
562
+ # METHODS FOR TEXT2BINARY
563
+ #########################
564
+
565
+ def load_matrix_file(source, splitChar = "\t"):
566
+ matrix = None
567
+ counter = 0
568
+ for line in source:
569
+ line = line.strip()
570
+
571
+ row = [float(c) for c in line.split(splitChar)]
572
+ if matrix is None:
573
+ matrix = np.zeros((len(row), len(row)))
574
+ for i, val in enumerate(row):
575
+ matrix[counter, i] = val
576
+ counter += 1
577
+
578
+ return matrix
579
+
580
+ def load_pair_file(source, byte_format = "float32"):
581
+ # Not used byte_forma parameter
582
+ connections = {}
583
+ for line in source:
584
+ line = line.strip().split("\t")
585
+ if len(line) == 3:
586
+ node_a, node_b, weight = line
587
+ weight = float(weight)
588
+ else:
589
+ node_a, node_b = line
590
+ weight = 1.0
591
+ pxc.add_nested_value(connections, (node_a, node_b), weight)
592
+ pxc.add_nested_value(connections, (node_b, node_a), weight)
593
+
594
+ matrix, names = dicti2wmatrix_squared(connections)
595
+ return matrix, names
596
+
597
+ def dicti2wmatrix_squared(dicti,symm= True):
598
+ element_names = dicti.keys()
599
+ matrix = np.zeros((len(element_names), len(element_names)))
600
+ i = 0
601
+ for elementA, relations in dicti.items():
602
+ for j, elementB in enumerate(element_names):
603
+ if elementA != elementB:
604
+ query = relations.get(elementB)
605
+ if query is not None:
606
+ matrix[i, j] = query
607
+ if symm:
608
+ matrix[j, i] = query
609
+ i += 1
610
+ return matrix, element_names
@@ -0,0 +1,78 @@
1
+ import numpy
2
+ from NetAnalyzer.netanalyzer import NetAnalyzer
3
+
4
+ class Net_parser:
5
+
6
+ def load(options):
7
+ net = None
8
+ if options['input_format'] == 'pair':
9
+ net = Net_parser.load_network_by_pairs(options['input_file'], options['layers'], options['split_char'])
10
+ elif options['input_format'] == 'bin':
11
+ net = Net_parser.load_network_by_bin_matrix(options['input_file'], options['node_files'], options['layers'])
12
+ elif options['input_format'] == 'matrix':
13
+ net = Net_parser.load_network_by_plain_matrix(options['input_file'], options['node_files'], options['layers'], options['split_char'])
14
+ else:
15
+ raise("ERROR: The format " + options['input_format'] + " is not defined")
16
+
17
+ if options.get('load_both'): # TODO: Not tested Yet.
18
+ if not net.graph:
19
+ layerA, layerB = list(net.matrices["adjacency_matrices"].keys())[0]
20
+ net.adjMat2netObj(layerA,layerB)
21
+ if net.matrices["adjacency_matrices"] == {}:
22
+ net.generate_all_biadjs()
23
+
24
+ return net
25
+
26
+ def load_network_by_pairs(file, layers, split_character="\t"):
27
+ net = NetAnalyzer([layer[0] for layer in layers])
28
+ f = open(file)
29
+ for line in f:
30
+ pair = line.rstrip().split(split_character)
31
+ node1 = pair[0]
32
+ node2 = pair[1]
33
+ net.add_node(node1, net.set_layer(layers, node1))
34
+ net.add_node(node2, net.set_layer(layers, node2))
35
+ if len(pair) == 3:
36
+ net.add_edge(node1, node2, weight=float(pair[2]))
37
+ else:
38
+ net.add_edge(node1, node2)
39
+
40
+ net.add_edge(node1, node2)
41
+ f.close()
42
+ return net
43
+
44
+ def load_network_by_bin_matrix(input_file, node_file, layers):
45
+ tag_layers = tuple([layer[0] for layer in layers])
46
+ net = NetAnalyzer(tag_layers)
47
+ if len(node_file) == 1:
48
+ node_names = Net_parser.load_input_list(node_file[0])
49
+ row_names = col_names = node_names
50
+ else:
51
+ row_names = Net_parser.load_input_list(node_file[0])
52
+ col_names = Net_parser.load_input_list(node_file[1])
53
+ if len(tag_layers) == 1:
54
+ net.matrices["adjacency_matrices"][(tag_layers[0],tag_layers[0])] = [numpy.load(input_file), row_names, col_names]
55
+ else:
56
+ net.matrices["adjacency_matrices"][tag_layers] = [numpy.load(input_file), row_names, col_names]
57
+ return net
58
+
59
+ def load_network_by_plain_matrix(input_file, node_file, layers, splitChar="\t"):
60
+ tag_layers = tuple([layer[0] for layer in layers])
61
+ net = NetAnalyzer(tag_layers)
62
+ if len(node_file) == 1:
63
+ node_names = Net_parser.load_input_list(node_file[0])
64
+ row_names = col_names = node_names
65
+ else:
66
+ row_names = Net_parser.load_input_list(node_file[0])
67
+ col_names = Net_parser.load_input_list(node_file[1])
68
+ if len(tag_layers) == 1:
69
+ net.matrices["adjacency_matrices"][(tag_layers[0],tag_layers[0])] = [numpy.genfromtxt(input_file, delimiter=splitChar), row_names, col_names]
70
+ else:
71
+ net.matrices["adjacency_matrices"][tag_layers] = [numpy.genfromtxt(input_file, delimiter=splitChar), row_names, col_names]
72
+ return net
73
+
74
+ def load_input_list(input_path):
75
+ file = open(input_path, "r")
76
+ input_data = file.readlines()
77
+ file.close()
78
+ return [line.rstrip() for line in input_data]