NetAnalyzer 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- NetAnalyzer/__init__.py +11 -0
- NetAnalyzer/adv_mat_calc.py +106 -0
- NetAnalyzer/cli_manager.py +331 -0
- NetAnalyzer/graph2sim.py +106 -0
- NetAnalyzer/integration.py +165 -0
- NetAnalyzer/main_modules.py +610 -0
- NetAnalyzer/net_parser.py +78 -0
- NetAnalyzer/net_plotter.py +200 -0
- NetAnalyzer/netanalyzer.py +1257 -0
- NetAnalyzer/performancer.py +83 -0
- NetAnalyzer/ranker.py +353 -0
- NetAnalyzer/seed_parser.py +27 -0
- NetAnalyzer/templates/net_explorer.txt +83 -0
- NetAnalyzer/templates/network.txt +1 -0
- NetAnalyzer-1.0.0.dist-info/LICENSE.txt +21 -0
- NetAnalyzer-1.0.0.dist-info/METADATA +86 -0
- NetAnalyzer-1.0.0.dist-info/RECORD +20 -0
- NetAnalyzer-1.0.0.dist-info/WHEEL +5 -0
- NetAnalyzer-1.0.0.dist-info/entry_points.txt +8 -0
- NetAnalyzer-1.0.0.dist-info/top_level.txt +1 -0
NetAnalyzer/__init__.py
ADDED
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
from NetAnalyzer.netanalyzer import NetAnalyzer
|
|
2
|
+
from NetAnalyzer.net_parser import Net_parser
|
|
3
|
+
from NetAnalyzer.adv_mat_calc import Adv_mat_calc
|
|
4
|
+
from NetAnalyzer.graph2sim import Graph2sim
|
|
5
|
+
from NetAnalyzer.net_plotter import Net_plotter
|
|
6
|
+
from NetAnalyzer.ranker import Ranker
|
|
7
|
+
from NetAnalyzer.performancer import Performancer
|
|
8
|
+
from NetAnalyzer.integration import Kernels
|
|
9
|
+
from NetAnalyzer.main_modules import *
|
|
10
|
+
from NetAnalyzer.cli_manager import *
|
|
11
|
+
from NetAnalyzer.seed_parser import SeedParser
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
import sys
|
|
2
|
+
import numpy as np
|
|
3
|
+
from scipy import linalg
|
|
4
|
+
import scipy.stats as stats
|
|
5
|
+
import umap
|
|
6
|
+
from warnings import warn
|
|
7
|
+
class Adv_mat_calc:
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@staticmethod
|
|
11
|
+
def data2umap(data, n_neighbors = 15, min_dist = 0.1, n_components = 2, metric = 'euclidean', random_seed = None): #2exp?
|
|
12
|
+
reducer = umap.UMAP(n_neighbors=n_neighbors, min_dist=min_dist, n_components=n_components, metric=metric, random_state= random_seed)
|
|
13
|
+
umap_coords = reducer.fit_transform(data)
|
|
14
|
+
return umap_coords
|
|
15
|
+
|
|
16
|
+
# Alaimo 2014, doi: 10.3389/fbioe.2014.00071
|
|
17
|
+
@staticmethod
|
|
18
|
+
def tranference_resources(matrix1, matrix2, lambda_value1 = 0.5, lambda_value2 = 0.5): #2exp?
|
|
19
|
+
# TODO (Fede,19/12/22) An extension to n layers would be possible with an iterative process.
|
|
20
|
+
m1rowNumber, m1colNumber = matrix1.shape
|
|
21
|
+
m2rowNumber, m2colNumber = matrix2.shape
|
|
22
|
+
matrix1Weight = Adv_mat_calc.graphWeights(m1colNumber, m1rowNumber, matrix1.T, lambda_value1)
|
|
23
|
+
matrix2Weight = Adv_mat_calc.graphWeights(m2colNumber, m2rowNumber, matrix2.T, lambda_value2)
|
|
24
|
+
matrixWeightProduct = np.dot(matrix1Weight, np.dot(matrix2, matrix2Weight))
|
|
25
|
+
finalMatrix = np.dot(matrix1, matrixWeightProduct)
|
|
26
|
+
return finalMatrix
|
|
27
|
+
|
|
28
|
+
@staticmethod
|
|
29
|
+
def graphWeights(rowsNumber, colsNumber, inputMatrix, lambdaValue = 0.5): #2exp?
|
|
30
|
+
ky = np.diag((1.0 / inputMatrix.sum(0))) #sum cols
|
|
31
|
+
weigth = np.dot(inputMatrix, ky).T
|
|
32
|
+
weigth[np.isnan(weigth)] = 0 # if there is no neighbors, there is no weight
|
|
33
|
+
ky = None #free memory
|
|
34
|
+
weigth = np.dot(inputMatrix, weigth)
|
|
35
|
+
|
|
36
|
+
kx = inputMatrix.sum(1) #sum rows
|
|
37
|
+
|
|
38
|
+
kx_lamb = kx ** lambdaValue
|
|
39
|
+
kx_lamb_mat = np.zeros((rowsNumber, rowsNumber))
|
|
40
|
+
for j in range(0,rowsNumber):
|
|
41
|
+
for i in range(0,rowsNumber):
|
|
42
|
+
kx_lamb_mat[j,i] = kx_lamb[i]
|
|
43
|
+
kx_lamb = None #free memory
|
|
44
|
+
|
|
45
|
+
kx_inv_lamb = kx ** (1 - lambdaValue)
|
|
46
|
+
kx_inv_lamb_mat = np.zeros((rowsNumber, rowsNumber))
|
|
47
|
+
for j in range(0,rowsNumber):
|
|
48
|
+
for i in range(0,rowsNumber):
|
|
49
|
+
kx_inv_lamb_mat[j,i] = kx_inv_lamb[i]
|
|
50
|
+
kx_inv_lamb = None #free memory
|
|
51
|
+
|
|
52
|
+
nx = 1.0/(kx_lamb_mat * kx_inv_lamb_mat) # inplace marks a matrix to be used by reference, not for value
|
|
53
|
+
nx[nx == np.inf] = 0 # if there is no neighbors, there is no weight
|
|
54
|
+
kx_lamb_mat = None #free memory
|
|
55
|
+
kx_inv_lamb_mat = None #free memory
|
|
56
|
+
weigth = weigth * nx
|
|
57
|
+
return weigth
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
@staticmethod
|
|
61
|
+
def disparity_filter_mat(matrix, rowIds, colIds, pval_threshold = 0.05): #2exp?
|
|
62
|
+
pval_mat = Adv_mat_calc.get_disparity_backbone_pval(matrix)
|
|
63
|
+
print(pval_mat)
|
|
64
|
+
# Create edge list from that p value matrix
|
|
65
|
+
result_mat = pval_mat < pval_threshold
|
|
66
|
+
# adjacency matrix, obtained when p[i,j] OR p[j,i] match the criteria
|
|
67
|
+
new_adj = result_mat.transpose() + result_mat
|
|
68
|
+
print(new_adj)
|
|
69
|
+
matrix[~new_adj] = 0
|
|
70
|
+
|
|
71
|
+
# remove genes with no significance
|
|
72
|
+
k = np.sum(new_adj, axis=0)
|
|
73
|
+
|
|
74
|
+
final_adj_mat = matrix[:,k>0]
|
|
75
|
+
final_adj_mat = final_adj_mat[k>0,:]
|
|
76
|
+
final_rowIds = [node_id for node_id, is_good in zip(rowIds, list(k>0)) if is_good]
|
|
77
|
+
final_colIds = [node_id for node_id, is_good in zip(colIds, list(k>0)) if is_good]
|
|
78
|
+
|
|
79
|
+
return final_adj_mat, final_rowIds, final_colIds
|
|
80
|
+
|
|
81
|
+
@staticmethod
|
|
82
|
+
def filter_rowcols_by_whitelist(matrix, rowIds, colIds, whitelist, symmetric = False): #2exp?
|
|
83
|
+
row_index = [ i for i, rowId in enumerate(rowIds) if rowId in whitelist ]
|
|
84
|
+
if symmetric:
|
|
85
|
+
col_index = row_index
|
|
86
|
+
else:
|
|
87
|
+
col_index = [ i for i, colId in enumerate(colIds) if colId in whitelist ]
|
|
88
|
+
matrix = matrix[row_index]
|
|
89
|
+
matrix = matrix[:,col_index]
|
|
90
|
+
rowIds = [rowIds[i] for i in row_index]
|
|
91
|
+
colIds = [colIds[i] for i in col_index]
|
|
92
|
+
return matrix, rowIds, colIds
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
@staticmethod
|
|
96
|
+
def get_disparity_backbone_pval(matrix): #2exp?
|
|
97
|
+
# by the moment, implementetion square (?)
|
|
98
|
+
# TODO: Add a warning when not square matrix.
|
|
99
|
+
pval_mat = matrix
|
|
100
|
+
W = np.sum(pval_mat, axis=0)
|
|
101
|
+
k = (pval_mat > 0).sum(0)
|
|
102
|
+
# operacion vectorizada.
|
|
103
|
+
pval_mat = np.ones(matrix.shape)
|
|
104
|
+
for i in range(0,pval_mat.shape[1]):
|
|
105
|
+
pval_mat[:,i] = (1-(matrix[:,i]/W[i]))**(k[i]-1)
|
|
106
|
+
return pval_mat
|
|
@@ -0,0 +1,331 @@
|
|
|
1
|
+
import argparse
|
|
2
|
+
import os
|
|
3
|
+
from py_cmdtabs.cmdtabs import CmdTabs
|
|
4
|
+
from NetAnalyzer.main_modules import *
|
|
5
|
+
|
|
6
|
+
## TYPES
|
|
7
|
+
def based_0(string): return int(string) - 1
|
|
8
|
+
|
|
9
|
+
def list_based_0(string): return CmdTabs.parse_column_indices(",", string)
|
|
10
|
+
|
|
11
|
+
def single_split(string, sep = ","):
|
|
12
|
+
return string.strip().split(sep)
|
|
13
|
+
|
|
14
|
+
def double_split(string, sep1=";", sep2=","):
|
|
15
|
+
return [sublst.split(sep2) for sublst in string.strip().split(sep1)]
|
|
16
|
+
|
|
17
|
+
def loading_dic(string, sep1=";", sep2=","):
|
|
18
|
+
return {key: value for key, value in double_split(string, sep1=";", sep2=",")}
|
|
19
|
+
|
|
20
|
+
def group_nodes_parse(string):
|
|
21
|
+
group_nodes = {}
|
|
22
|
+
if os.path.isfile(string):
|
|
23
|
+
with open(string) as file:
|
|
24
|
+
for line in file:
|
|
25
|
+
groupID, nodeID = line.strip().split("\t")
|
|
26
|
+
query = group_nodes.get(groupID)
|
|
27
|
+
if query is None:
|
|
28
|
+
group_nodes[groupID] = [nodeID]
|
|
29
|
+
else:
|
|
30
|
+
query.append(nodeID)
|
|
31
|
+
else:
|
|
32
|
+
for i, group in enumerate(string.split(";")):
|
|
33
|
+
group_nodes[i] = group.split(',')
|
|
34
|
+
|
|
35
|
+
return group_nodes
|
|
36
|
+
|
|
37
|
+
def graph_options_parse(string):
|
|
38
|
+
graph_options = {}
|
|
39
|
+
for pair in string.split(','):
|
|
40
|
+
fields = pair.split('=')
|
|
41
|
+
graph_options[fields[0]] = fields[1]
|
|
42
|
+
return graph_options
|
|
43
|
+
|
|
44
|
+
## Common options
|
|
45
|
+
##############################################
|
|
46
|
+
|
|
47
|
+
def add_kernel_flags(parser, multiple = False):
|
|
48
|
+
if multiple:
|
|
49
|
+
type_parse = lambda x: single_split(x, sep=";")
|
|
50
|
+
else:
|
|
51
|
+
type_parse = lambda x: str(x)
|
|
52
|
+
|
|
53
|
+
parser.add_argument("-k", "--input_kernels", dest="kernel_files", default= None, type= lambda x: type_parse(x),
|
|
54
|
+
help="The roots from each kernel to integrate")
|
|
55
|
+
parser.add_argument("-n", "--input_nodes", dest="node_files", default= None, type = lambda x: type_parse(x),
|
|
56
|
+
help="The list of node for each kernel in lst format")
|
|
57
|
+
|
|
58
|
+
def add_seed_flags(parser):
|
|
59
|
+
parser.add_argument("--seed_nodes", dest="seed_nodes", default=None,
|
|
60
|
+
help="The name of the nodes to use as seeds")
|
|
61
|
+
parser.add_argument("--seed_sep", dest="seed_sep", default=",",
|
|
62
|
+
help="Separator of seed genes. Only use when -s point to a file")
|
|
63
|
+
|
|
64
|
+
def add_random_seed(parser, default_seed=None):
|
|
65
|
+
parser.add_argument("--seed", dest="seed", default= default_seed, type=int,
|
|
66
|
+
help="Allows to set a seed for the randomization process. Set to a number. Otherwise results are not reproducible.")
|
|
67
|
+
|
|
68
|
+
def add_output_flags(parser, default_opt={"output_file": "output_file"}):
|
|
69
|
+
parser.add_argument("-o","--output_file", dest="output_file", default=default_opt['output_file'],
|
|
70
|
+
help="Output file name")
|
|
71
|
+
|
|
72
|
+
def add_input_graph_flags(parser, multinet = False):
|
|
73
|
+
if multinet:
|
|
74
|
+
parser.add_argument("-i", "--input_file", dest="input_file", default= None, type = lambda string: loading_dic(string, sep1=";", sep2=","),
|
|
75
|
+
help="Input file to create networks for further analysis")
|
|
76
|
+
parser.add_argument("-n","--node_names_file", dest="node_files", default=None, type = lambda string: loading_dic(string, sep1=";", sep2=","),
|
|
77
|
+
help="Files with node names corresponding to the input matrix, only use when -i is set to bin or matrix, could be two paths, indicating rows and cols, respectively. If just one path added, it is assumed to be for rows and cols")
|
|
78
|
+
# parser.add_argument("-l","--layers", dest="layers", default=[['layer', '-']], type= lambda x: double_split(x, sep1=";",sep2=","),
|
|
79
|
+
# help="Layer definition on network: layer1name,regexp1;layer2name,regexp2...")
|
|
80
|
+
else:
|
|
81
|
+
parser.add_argument("-i", "--input_file", dest="input_file", default= None,
|
|
82
|
+
help="Input file to create networks for further analysis")
|
|
83
|
+
parser.add_argument("-n","--node_names_file", dest="node_files", default=None, type = lambda x: single_split(x, sep=","),
|
|
84
|
+
help="Files with node names corresponding to the input matrix, only use when -i is set to bin or matrix, could be two paths, indicating rows and cols, respectively. If just one path added, it is assumed to be for rows and cols")
|
|
85
|
+
parser.add_argument("-l","--layers", dest="layers", default=[['layer', '-']], type= lambda x: double_split(x, sep1=";",sep2=","),
|
|
86
|
+
help="Layer definition on network: layer1name,regexp1;layer2name,regexp2...")
|
|
87
|
+
parser.add_argument("-s","--split_char", dest="split_char", default='\t',
|
|
88
|
+
help = "Character for splitting input file. Default: tab")
|
|
89
|
+
parser.add_argument("-f","--input_format", dest="input_format", default='pair',
|
|
90
|
+
help="Input file format: pair (default), bin, matrix")
|
|
91
|
+
parser.add_argument("--both_repre_formats", dest="load_both", default=False, action='store_true',
|
|
92
|
+
help="If we need to load the adjacency matrixes and the graph object")
|
|
93
|
+
|
|
94
|
+
def add_resources_flags(parser, default_opt={"threads": 1}):
|
|
95
|
+
parser.add_argument("-T", "--threads", dest="threads", default=default_opt["threads"], type=int,
|
|
96
|
+
help="Number of threads to use in computation.")
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def add_common_relations_process(parser):
|
|
100
|
+
parser.add_argument("-N","--no_autorelations", dest="no_autorelations", default=False, action='store_true',
|
|
101
|
+
help="No processing autorelations")
|
|
102
|
+
|
|
103
|
+
##############################################
|
|
104
|
+
|
|
105
|
+
def integrate_kernels(args=None):
|
|
106
|
+
parser = argparse.ArgumentParser(description='Integrate kernels or embedding in matrix format')
|
|
107
|
+
add_kernel_flags(parser, multiple = True)
|
|
108
|
+
add_output_flags(parser, default_opt={"output_file": "general_matrix"})
|
|
109
|
+
parser.add_argument("-f","--format_kernel",dest= "input_format", default="bin",
|
|
110
|
+
help= "The format of the kernels to integrate")
|
|
111
|
+
parser.add_argument("-I", "--kernel_ids", dest="kernel_ids", default= None, type = lambda x: single_split(x, sep=";"),
|
|
112
|
+
help="The names of each kernel")
|
|
113
|
+
# Integration conf
|
|
114
|
+
parser.add_argument("-i","--integration_type",dest= "integration_type", default=None,
|
|
115
|
+
help= "It specifies how to integrate the kernels")
|
|
116
|
+
parser.add_argument("--raw_values", dest="raw_values", default=False, action='store_true',help="Select this option to use the negatives and positives values, without translation")
|
|
117
|
+
parser.add_argument("--asym",dest= "symmetric", default=True, action='store_false',
|
|
118
|
+
help= "It specifies if the kernel matrixes are or not symmetric")
|
|
119
|
+
|
|
120
|
+
# Resources
|
|
121
|
+
add_resources_flags(parser=parser, default_opt={"threads": 8})
|
|
122
|
+
opts = parser.parse_args(args)
|
|
123
|
+
main_integrate_kernels(opts)
|
|
124
|
+
|
|
125
|
+
def netanalyzer(args=None):
|
|
126
|
+
parser = argparse.ArgumentParser(description='Perform Network analysis from NetAnalyzer package')
|
|
127
|
+
add_common_relations_process(parser)
|
|
128
|
+
add_input_graph_flags(parser)
|
|
129
|
+
add_output_flags(parser, default_opt={"output_file": "output_file"})
|
|
130
|
+
add_random_seed(parser, default_seed=None)
|
|
131
|
+
parser.add_argument("-O", "--ontology", dest="ontologies", default=[], type=lambda x: double_split(x, sep1=";",sep2=","),
|
|
132
|
+
help="String that define which ontologies must be used with each layer. String definition:'layer_name1,path_to_obo_file1;layer_name2,path_to_obo_file2'")
|
|
133
|
+
# Assoc
|
|
134
|
+
parser.add_argument("-P","--use_pairs", dest="use_pairs", default='conn',
|
|
135
|
+
help="Which pairs must be computed. 'all' means all posible pair node combinations and 'conn' means the pair are truly connected in the network. Default 'conn' ")
|
|
136
|
+
parser.add_argument("-m","--association_method", dest="meth", default=None,
|
|
137
|
+
help="select association method to perform the projections: counts, jaccard, simpson, geometric, cosine, pcc, hypergeometric, hypergeometric_bf, hypergeometric_bh, csi, transference, correlation, umap, pca, bicm")
|
|
138
|
+
parser.add_argument("-a","--assoc_file", dest="assoc_file", default='assoc_values.txt',
|
|
139
|
+
help="Output file name for association values")
|
|
140
|
+
parser.add_argument("-p","--performance_file", dest="performance_file", default='perf_values.txt',
|
|
141
|
+
help="Output file name for performance values")
|
|
142
|
+
parser.add_argument("-u","--use_layers", dest="use_layers", default=[], type= lambda x: double_split(x, sep1=";",sep2=","),
|
|
143
|
+
help="Set which layers must be used on association methods: layer1,layer2;layerA,layerB")
|
|
144
|
+
parser.add_argument("-c","--control_file", dest="control_file", default=None,
|
|
145
|
+
help="Control file name")
|
|
146
|
+
# Kernel
|
|
147
|
+
parser.add_argument("-k","--kernel_method", dest="kernel", default=None,
|
|
148
|
+
help="Kernel operation to perform with the adjacency matrix")
|
|
149
|
+
parser.add_argument("--embedding_add_options", dest="embedding_add_options", default="",
|
|
150
|
+
help="Additional options for embedding kernel methods. It must be defines as '\"opt_name1\" : value1, \"opt_name2\" : value2,...' ")
|
|
151
|
+
parser.add_argument("-z","--normalize_kernel_values", dest="normalize_kernel", default=False, action='store_true',
|
|
152
|
+
help="Apply cosine normalization to the obtained kernel")
|
|
153
|
+
parser.add_argument("--coords2sim_type", dest="coords2sim_type", default="dotProduct", help= "Select the type of transformation from coords to similarity: dotProduct, normalizedScaling, infinity and int or float numbers")
|
|
154
|
+
parser.add_argument("-K","--kernel_file", dest="kernel_file", default='kernel_file',
|
|
155
|
+
help="Output file name for kernel values")
|
|
156
|
+
# Plotting
|
|
157
|
+
parser.add_argument("-g", "--graph_file", dest="graph_file", default=None,
|
|
158
|
+
help="Build a graphic representation of the network")
|
|
159
|
+
parser.add_argument("--graph_options", dest="graph_options", default={'method': 'elgrapho', 'layout': 'forcedir', 'steps': '30'}, type= graph_options_parse,
|
|
160
|
+
help="Set graph parameters as 'NAME1=value1,NAME2=value2,...")
|
|
161
|
+
# Nodes states
|
|
162
|
+
parser.add_argument("-r","--reference_nodes", dest="reference_nodes", default=[], type= lambda x: single_split(x, sep=","),
|
|
163
|
+
help="Node ids comma separared")
|
|
164
|
+
parser.add_argument("-G","--group_nodes", dest="group_nodes", default={}, type= group_nodes_parse,
|
|
165
|
+
help="File path or groups separated by ';' and group node ids comma separared")
|
|
166
|
+
parser.add_argument("-d","--delete", dest="delete_nodes", default=[], type= lambda x: single_split(x, sep=";"),
|
|
167
|
+
help="Remove nodes from file. If PATH;r then nodes not included in file are removed")
|
|
168
|
+
# Compare cluster
|
|
169
|
+
parser.add_argument("--overlapping_communities", dest ="overlapping_communities", default=False, action="store_true",
|
|
170
|
+
help=" This is needed to activate overlapping sensitive operations in communities analysis")
|
|
171
|
+
parser.add_argument("-R","--compare_clusters_reference", dest="compare_clusters_reference", default=None, type= group_nodes_parse,
|
|
172
|
+
help="File path or groups separated by ';' and group node ids comma separared")
|
|
173
|
+
parser.add_argument("--compare_clusters_join", dest="compare_clusters_join", default=None,
|
|
174
|
+
help="Join strategy on comparing two types of clusters: union")
|
|
175
|
+
# Build cluster
|
|
176
|
+
parser.add_argument("-b", "--build_clusters_alg", dest="build_cluster_alg", default=None,
|
|
177
|
+
help="Type of cluster algorithm")
|
|
178
|
+
parser.add_argument("-B", "--build_clusters_add_options", dest="build_clusters_add_options", default="",
|
|
179
|
+
help="Additional options for clustering methods. It must be defines as '\"opt_name1\" : value1, \"opt_name2\" : value2,...'")
|
|
180
|
+
parser.add_argument("--output_build_clusters", dest="output_build_clusters", default=None, help= "output name for discovered clusters")
|
|
181
|
+
# Expand cluster
|
|
182
|
+
parser.add_argument("-x","--expand_clusters", dest="expand_clusters", default=None,
|
|
183
|
+
help="Method to expand clusters Available methods: sht_path")
|
|
184
|
+
parser.add_argument("--one_sht_pairs", dest="one_sht_pairs", default=False, action='store_true',
|
|
185
|
+
help="add this flag if expand cluster needed with just one of the shortest paths")
|
|
186
|
+
parser.add_argument("--output_expand_clusters", dest= "output_expand_clusters", default="expand_clusters.txt", help="outputname fopr expand clusters file")
|
|
187
|
+
# Cluster metrics
|
|
188
|
+
parser.add_argument("-M", "--group_metrics", dest="group_metrics", default=None, type= lambda x: single_split(x, sep=";"),
|
|
189
|
+
help="Perform group group_metrics")
|
|
190
|
+
parser.add_argument("--output_metrics_by_cluster", dest="output_metrics_by_cluster", default='group_metrics.txt', help= "output name for metrics by cluster file")
|
|
191
|
+
parser.add_argument("-S", "--summarize_metrics", dest="summarize_metrics", default=None, type= lambda x: single_split(x, sep=";"),
|
|
192
|
+
help="Summarize metrics from groups")
|
|
193
|
+
parser.add_argument("--output_summarized_metrics", dest="output_summarized_metrics", default='group_metrics_summarized.txt', help= "output name for summarized metrics file")
|
|
194
|
+
# Graph metrics by net or node
|
|
195
|
+
parser.add_argument("-A", "--attributes", dest="get_attributes", default=[], type =lambda x: single_split(x, sep=","),
|
|
196
|
+
help="String separated by commas with the name of node attribute")
|
|
197
|
+
parser.add_argument("--graph_attributes", dest="get_graph_attributes", default=[], type =lambda x: single_split(x, sep=","),
|
|
198
|
+
help="String separated by commas with the name of network attribute")
|
|
199
|
+
parser.add_argument("--attributes_summarize", dest="attributes_summarize", default= False, action = "store_true", help="If the attribtes needs to be obtained summarized")
|
|
200
|
+
# DSL section
|
|
201
|
+
parser.add_argument("--dsl_script", dest="dsl_script", default=None,
|
|
202
|
+
help="Path to dsl script to perform complex analysis")
|
|
203
|
+
# Resources
|
|
204
|
+
add_resources_flags(parser=parser, default_opt={"threads": 2})
|
|
205
|
+
|
|
206
|
+
opts = parser.parse_args(args)
|
|
207
|
+
main_netanalyzer(opts)
|
|
208
|
+
|
|
209
|
+
def randomize_clustering(args=None):
|
|
210
|
+
parser = argparse.ArgumentParser(description='Perform clusters randomization')
|
|
211
|
+
add_output_flags(parser, default_opt={"output_file": "random_clusters.txt"})
|
|
212
|
+
parser.add_argument("-i", "--input_file", dest="input_file", default= None,
|
|
213
|
+
help="Input file to create networks for further analysis")
|
|
214
|
+
parser.add_argument("-S", "--split_char", dest="column_sep", default = "\t",
|
|
215
|
+
help="Character for splitting input file. Default: tab")
|
|
216
|
+
parser.add_argument("-a", "--aggregate_sep", dest="aggregate_sep", default = None,
|
|
217
|
+
help="This option activates aggregation in output. Separator character must be provided")
|
|
218
|
+
parser.add_argument("-N", "--node_column", dest="node_index", default= 1, type=based_0,
|
|
219
|
+
help="Number of the nodes column")
|
|
220
|
+
parser.add_argument("-C", "--cluster_column", dest="cluster_index", default= 0, type=based_0,
|
|
221
|
+
help="Number of the clusters column")
|
|
222
|
+
parser.add_argument("-s", "--node_sep", dest="node_sep", default = None,
|
|
223
|
+
help="Node split character. This option must to be used when input file is aggregated")
|
|
224
|
+
# random conf
|
|
225
|
+
parser.add_argument("-r", "--random_type", dest="random_type", default = ["hard_fixed"], type = lambda x: single_split(x,sep=":"),
|
|
226
|
+
help="""Indicate random mode. First, the not custom randomization, where cluster size is the same:
|
|
227
|
+
'hard_fixed' (fixing the number of communities a node belongs),
|
|
228
|
+
'soft_fixed' (soft finxing the number of communities a node belongs by binomial model),
|
|
229
|
+
'not_fixed' (just a sampling with replacement).
|
|
230
|
+
Secondly, the custom size can be used with the format 'custom:n:s:r|nr' for generate 'n' clusters of 's' nodes and
|
|
231
|
+
r|nr for replacement or not, respectively. Default = 'hard_fixed'""")
|
|
232
|
+
add_random_seed(parser, default_seed=123)
|
|
233
|
+
opts = parser.parse_args(args)
|
|
234
|
+
main_randomize_clustering(opts)
|
|
235
|
+
|
|
236
|
+
def randomize_network(args=None):
|
|
237
|
+
parser = argparse.ArgumentParser(description='Perform Network analysis from NetAnalyzer package')
|
|
238
|
+
add_input_graph_flags(parser)
|
|
239
|
+
add_output_flags(parser, default_opt={"output_file": None})
|
|
240
|
+
add_random_seed(parser, default_seed=None)
|
|
241
|
+
# random conf
|
|
242
|
+
parser.add_argument("-r", "--type_random", dest="type_random", default= None,
|
|
243
|
+
help="Randomized basis. 'nodes' Node-baseds randomize or 'links' Links-baseds randomize")
|
|
244
|
+
opts = parser.parse_args(args)
|
|
245
|
+
main_randomize_network(opts)
|
|
246
|
+
|
|
247
|
+
def ranker(args=None):
|
|
248
|
+
parser = argparse.ArgumentParser(description='Get the ranks from a matrix similarity score and a list of seeds')
|
|
249
|
+
add_seed_flags(parser)
|
|
250
|
+
add_output_flags(parser, default_opt={"output_file": "ranked_genes"})
|
|
251
|
+
add_kernel_flags(parser, multiple = False)
|
|
252
|
+
|
|
253
|
+
# Output ranking
|
|
254
|
+
parser.add_argument("--seed_presence", dest="seed_presence", default=None,
|
|
255
|
+
help="Seed presence on list: 'remove', when seed is not added on the ranker calculation; 'annotate', when just tag on output is needed, and calculation is obtained inclusing the seeds: new, seed.")
|
|
256
|
+
parser.add_argument("--header", dest="header", default=False,
|
|
257
|
+
action='store_true', help="Select this option if header needed")
|
|
258
|
+
# filtering
|
|
259
|
+
parser.add_argument("-f", "--filter", dest="filter", default=None,
|
|
260
|
+
help="PATH to file with seed_name and genes to keep in output")
|
|
261
|
+
parser.add_argument("--whitelist", dest="whitelist", default=None, type = open_whitelist, help= "File Path with the whitelist of nodes to take into account in the ranker process")
|
|
262
|
+
parser.add_argument("--minimum_size", dest="minimum_size", default=1, type=int)
|
|
263
|
+
# Options in ranker alg
|
|
264
|
+
parser.add_argument("-N","--normalize_matrix", dest="normalize_matrix", default= None,
|
|
265
|
+
help="Select the type of normalization, options are: None, by_column, by_row, by_row_col")
|
|
266
|
+
parser.add_argument("-p", "--propagate", dest="propagate", default = False, action = "store_true")
|
|
267
|
+
parser.add_argument("--propagate_options", dest="propagate_options", default = '"tolerance": 1e-5, "iteration_limit": 100, "with_restart": 0',
|
|
268
|
+
help="Additional options for propagation methods. It must be defines as '\"opt_name1\" : value1, \"opt_name2\" : value2,...'")
|
|
269
|
+
# Benchmarking mode
|
|
270
|
+
parser.add_argument("-l", "--cross_validation", dest="cross_validation", default=False,
|
|
271
|
+
action='store_true', help="To activate cross validation")
|
|
272
|
+
parser.add_argument("-K", "--k_fold", dest="k_fold", default=None, type=int,
|
|
273
|
+
help="Indicate the number of itrations needed, not used for leave one out cross validation (loocv)")
|
|
274
|
+
# Select tops
|
|
275
|
+
parser.add_argument("-t", "--top_n", dest="top_n", default=None, type=int,
|
|
276
|
+
help="Top N genes to print in output")
|
|
277
|
+
parser.add_argument("--output_top", dest="output_top", default=None,
|
|
278
|
+
help="File to save Top N genes")
|
|
279
|
+
parser.add_argument("--add_tags", dest="add_tags", default=None, help="Adding node attribute by seed: format seed\\tnode\\tattr")
|
|
280
|
+
parser.add_argument("--representation_seed_metric", dest = "representation_seed_metric", default = "mean",
|
|
281
|
+
help = "select the type of representation on seed, default mean, options: mean and max")
|
|
282
|
+
# Resources
|
|
283
|
+
add_resources_flags(parser=parser, default_opt={"threads": 1})
|
|
284
|
+
opts = parser.parse_args(args)
|
|
285
|
+
main_ranker(opts)
|
|
286
|
+
|
|
287
|
+
def text2binary_matrix(args=None):
|
|
288
|
+
parser = argparse.ArgumentParser(description="Transforming matrix format and obtaining statistics")
|
|
289
|
+
add_output_flags(parser, default_opt={"output_file": None})
|
|
290
|
+
parser.add_argument('-i', '--input_file', dest="input_file", default=None,
|
|
291
|
+
help="input file")
|
|
292
|
+
parser.add_argument('-b', '--byte_format', dest="byte_format", default="float64",
|
|
293
|
+
help='Format of the numeric values stored in matrix. Default: float64, warning set this to less precission can modify computation results using this matrix.')
|
|
294
|
+
parser.add_argument('-t', '--input_type', dest="input_type", default='pair',
|
|
295
|
+
help='Set input format file. "pair", "matrix" or "bin"')
|
|
296
|
+
parser.add_argument('-O', '--output_type', dest="output_type", default='bin',
|
|
297
|
+
help='Set output format file. "bin" for binary (default) or "mat" for tabulated text file matrix')
|
|
298
|
+
# Process matrix
|
|
299
|
+
parser.add_argument('-d', '--set_diagonal', dest="set_diagonal", default=False, action='store_true',
|
|
300
|
+
help='Set to 1.0 the main diagonal')
|
|
301
|
+
parser.add_argument('-B', '--binarize', dest="binarize", default=None, type = float,
|
|
302
|
+
help='Binarize matrix changin x >= thr to one and any other to zero into matrix given')
|
|
303
|
+
parser.add_argument('-c', '--cutoff', dest="cutoff", default=None, type = float,
|
|
304
|
+
help='Cutoff matrix values keeping just x >= and setting any other to zero into matrix given')
|
|
305
|
+
# Get stats
|
|
306
|
+
parser.add_argument('-s', '--get_stats', dest="stats", default=None,
|
|
307
|
+
help='Get stats from the processed matrix')
|
|
308
|
+
|
|
309
|
+
opts = parser.parse_args(args)
|
|
310
|
+
main_text2binary_matrix(opts)
|
|
311
|
+
|
|
312
|
+
def net_explorer(args=None, test=False):
|
|
313
|
+
parser = argparse.ArgumentParser(description="Transforming matrix format and obtaining statistics")
|
|
314
|
+
add_common_relations_process(parser) # Common relations options
|
|
315
|
+
add_input_graph_flags(parser, multinet = True) # Input graph
|
|
316
|
+
add_seed_flags(parser) # Adding seeds
|
|
317
|
+
add_output_flags(parser, default_opt={"output_file": "output_file"})
|
|
318
|
+
# layer processing
|
|
319
|
+
parser.add_argument('-c', '--layer_cutoff', dest="layer_cutoff", default={}, type = lambda string: loading_dic(string, sep1=";", sep2=","),
|
|
320
|
+
help='Cutoff to apply to every layer in the multiplexed one')
|
|
321
|
+
# Analysis options
|
|
322
|
+
parser.add_argument("-l", "--neigh_level", dest="neigh_level", default={}, type = lambda string: loading_dic(string, sep1=";", sep2=","),
|
|
323
|
+
help="Defining the level of neighbourhood on the initial set of nodes")
|
|
324
|
+
parser.add_argument("--plot_network_method", dest="plot_network_method", default="pyvis",
|
|
325
|
+
help="Defining the plot method used on report")
|
|
326
|
+
parser.add_argument("--embedding_proj", dest="embedding_proj", default=None,
|
|
327
|
+
help="Select different projections methods: umap")
|
|
328
|
+
parser.add_argument("--compare_nets", dest="compare_nets", default=False, action="store_true")
|
|
329
|
+
opts = parser.parse_args(args)
|
|
330
|
+
to_test = main_net_explorer(opts, test)
|
|
331
|
+
if test: return to_test
|
NetAnalyzer/graph2sim.py
ADDED
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
import sys
|
|
2
|
+
import numpy as np
|
|
3
|
+
from scipy import linalg
|
|
4
|
+
from warnings import warn
|
|
5
|
+
import networkx as nx
|
|
6
|
+
#from NetAnalyzer.adv_mat_calc import Adv_mat_calc
|
|
7
|
+
import py_exp_calc.exp_calc as pxc
|
|
8
|
+
from pecanpy import pecanpy
|
|
9
|
+
from gensim.models import Word2Vec
|
|
10
|
+
|
|
11
|
+
class Graph2sim:
|
|
12
|
+
|
|
13
|
+
allowed_embeddings = ['node2vec', 'deepwalk']
|
|
14
|
+
allowed_kernels = ['el', 'ct', 'rf', 'me', 'vn', 'rl', 'ka', 'md']
|
|
15
|
+
|
|
16
|
+
def get_embedding(adj_mat, embedding_nodes, embedding = "node2vec", dimensions = 64, walk_length=30, num_walks = 200, p = 1, q = 1, workers = 16, window = 10, min_count=1, seed = None, quiet=False, batch_words=4):
|
|
17
|
+
|
|
18
|
+
emb_coords = None
|
|
19
|
+
if embedding in ['node2vec', 'deepwalk']: # TODO 'metapath2vec',
|
|
20
|
+
if embedding == 'node2vec' or embedding == "deepwalk":
|
|
21
|
+
if embedding == "deepwalk":
|
|
22
|
+
p = 1
|
|
23
|
+
q = 1
|
|
24
|
+
verbose = not quiet
|
|
25
|
+
g = pecanpy.DenseOTF(p=p, q=q, workers=workers, verbose= verbose)
|
|
26
|
+
g = g.from_mat(adj_mat=adj_mat, node_ids=embedding_nodes)
|
|
27
|
+
walks = g.simulate_walks(num_walks=num_walks, walk_length=walk_length)
|
|
28
|
+
# use random walks to train embeddings
|
|
29
|
+
model = Word2Vec(walks, vector_size=dimensions, window=window, min_count=min_count, workers=workers) # point to extend
|
|
30
|
+
list_arrays=[model.wv.get_vector(str(n)) for n in embedding_nodes]
|
|
31
|
+
n_cols=list_arrays[0].shape[0] # Number of col
|
|
32
|
+
n_rows=len(list_arrays)# Number of rows
|
|
33
|
+
emb_coords = np.concatenate(list_arrays).reshape([n_rows,n_cols]) # Concat all the arrays at one.
|
|
34
|
+
else:
|
|
35
|
+
print('Warning: The embedding method was not specified or not exists.')
|
|
36
|
+
sys.exit(0)
|
|
37
|
+
|
|
38
|
+
return emb_coords
|
|
39
|
+
|
|
40
|
+
def emb_coords2kernel(emb_coords, normalization = False, sim_type= "dotProduct"):
|
|
41
|
+
kernel = pxc.coords2sim(emb_coords, sim = sim_type)
|
|
42
|
+
if normalization: kernel = pxc.cosine_normalization(kernel)
|
|
43
|
+
return kernel
|
|
44
|
+
|
|
45
|
+
def get_kernel(matrix, kernel, normalization=False): #2exp?
|
|
46
|
+
#I = identity matrix
|
|
47
|
+
#D = Diagonal matrix
|
|
48
|
+
#A = adjacency matrix
|
|
49
|
+
#L = laplacian matrix = D − A
|
|
50
|
+
matrix_result = None
|
|
51
|
+
dimension_elements = np.shape(matrix)[1]
|
|
52
|
+
# In scuba code, the diagonal values of A is set to 0. In weighted matrix the kernel result is the same with or without this operation. Maybe increases the computing performance?
|
|
53
|
+
# In the md kernel this operation affects the values of the final kernel
|
|
54
|
+
#dimension_elements.times do |n|
|
|
55
|
+
# matrix[n,n] = 0.0
|
|
56
|
+
#end
|
|
57
|
+
if kernel in ['el', 'ct', 'rf', 'me'] or 'vn' in kernel or 'rl' in kernel:
|
|
58
|
+
diagonal_matrix = np.zeros((dimension_elements,dimension_elements))
|
|
59
|
+
np.fill_diagonal(diagonal_matrix,matrix.sum(axis=1)) # get the total sum for each row, for this reason the sum method takes the 1 value. If sum colums is desired, use 0
|
|
60
|
+
# Make a matrix whose diagonal is row_sum
|
|
61
|
+
matrix_L = diagonal_matrix - matrix
|
|
62
|
+
if kernel == 'el': #Exponential Laplacian diffusion kernel(active). F Fouss 2012 | doi: 10.1016/j.neunet.2012.03.001
|
|
63
|
+
beta = 0.02
|
|
64
|
+
beta_product = matrix_L * -beta
|
|
65
|
+
#matrix_result = beta_product.expm
|
|
66
|
+
matrix_result = linalg.expm(beta_product)
|
|
67
|
+
elif kernel == 'ct': # Commute time kernel (active). J.-K. Heriche 2014 | doi: 10.1091/mbc.E13-04-0221
|
|
68
|
+
matrix_result = np.linalg.pinv(matrix_L, hermitian=True) # Hermitian parameter added to ensure convergence, just for real symmetric matrixes.
|
|
69
|
+
# Anibal saids that this kernel was normalized. Why?. Paper do not seem to describe this operation for ct, it describes for Kvn or for all kernels, it is not clear.
|
|
70
|
+
elif kernel == 'rf': # Random forest kernel. J.-K. Heriche 2014 | doi: 10.1091/mbc.E13-04-0221
|
|
71
|
+
matrix_result = np.linalg.inv(np.eye(dimension_elements) + matrix_L) #Krf = (I +L ) ^ −1
|
|
72
|
+
elif 'vn' in kernel: # von Neumann diffusion kernel. J.-K. Heriche 2014 | doi: 10.1091/mbc.E13-04-0221
|
|
73
|
+
alpha = float(kernel.replace('vn', '')) * max(np.linalg.eigvals(matrix)) ** -1 # alpha = impact_of_penalization (1, 0.5 or 0.1) * spectral radius of A. spectral radius of A = absolute value of max eigenvalue of A
|
|
74
|
+
# TODO: The expresion max(np.linalg.eigvals(matrix)) obtain the max eigen computing all but in ruby was used a power series to compute directly this value. Implement this
|
|
75
|
+
matrix_result = np.linalg.inv(np.eye(dimension_elements) - matrix * alpha ) # (I -alphaA ) ^ −1
|
|
76
|
+
elif 'rl' in kernel: # Regularized Laplacian kernel matrix (active)
|
|
77
|
+
alpha = float(kernel.replace('rl', '')) * max(np.linalg.eigvals(matrix)) ** -1 # alpha = impact_of_penalization (1, 0.5 or 0.1) * spectral radius of A. spectral radius of A = absolute value of max eigenvalue of A
|
|
78
|
+
matrix_result = np.linalg.inv(np.eye(dimension_elements) + matrix_L * alpha ) # (I + alphaL ) ^ −1
|
|
79
|
+
elif kernel == 'me': # Markov exponential diffusion kernel (active). G Zampieri 2018 | doi.org/10.1186/s12859-018-2025-5 . Taken from compute_kernel script
|
|
80
|
+
beta=0.04
|
|
81
|
+
#(beta/N)*(N*I - D + A)
|
|
82
|
+
id_mat = np.eye(dimension_elements)
|
|
83
|
+
m_matrix = (id_mat * dimension_elements - diagonal_matrix + matrix ) * (beta/dimension_elements)
|
|
84
|
+
#matrix_result = m_matrix.expm
|
|
85
|
+
matrix_result = linalg.expm(m_matrix)
|
|
86
|
+
elif kernel == 'ka': # Kernelized adjacency matrix (active). J.-K. Heriche 2014 | doi: 10.1091/mbc.E13-04-0221
|
|
87
|
+
lambda_value = min(np.linalg.eigvals(matrix)) # TODO implent as power series as shown in ruby equivalent
|
|
88
|
+
matrix_result = matrix + np.eye(dimension_elements) * abs(lambda_value) # Ka = A + lambda*I # lambda = the absolute value of the smallest eigenvalue of A
|
|
89
|
+
elif 'md' in kernel: # Markov diffusion kernel matrix. G Zampieri 2018 | doi.org/10.1186/s12859-018-2025-5 . Taken from compute_kernel script
|
|
90
|
+
t = int(kernel.replace('md', ''))
|
|
91
|
+
#TODO: check implementation
|
|
92
|
+
col_sum = matrix.sum(axis=1)
|
|
93
|
+
p_mat = np.divide(matrix.T,col_sum).T
|
|
94
|
+
p_temp_mat = p_mat.copy()
|
|
95
|
+
zt_mat = p_mat.copy()
|
|
96
|
+
for i in range(0, t-1):
|
|
97
|
+
p_temp_mat = np.dot(p_temp_mat,p_mat)
|
|
98
|
+
zt_mat = zt_mat + p_temp_mat
|
|
99
|
+
zt_mat = zt_mat * (1.0/t)
|
|
100
|
+
matrix_result = np.dot(zt_mat, zt_mat.T)
|
|
101
|
+
else:
|
|
102
|
+
matrix_result = matrix
|
|
103
|
+
warn('Warning: The kernel method was not specified or not exists. The adjacency matrix will be given as result')
|
|
104
|
+
# This allows process a previous kernel and perform the normalization in a separated step.
|
|
105
|
+
if normalization: matrix_result = pxc.cosine_normalization(matrix_result) #TODO: check implementation with Numo::array
|
|
106
|
+
return matrix_result
|