d3graph 3.0.0__tar.gz → 3.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,12 +1,12 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: d3graph
3
- Version: 3.0.0
3
+ Version: 3.1.0
4
4
  Summary: Python package to create interactive network based on d3js.
5
5
  Author-email: Erdogan Taskesen <erdogant@gmail.com>
6
6
  License-Expression: BSD-3-Clause
7
7
  Project-URL: Homepage, https://erdogant.github.io/d3graph
8
8
  Project-URL: Download, https://github.com/erdogant/d3graph/archive/{version}.tar.gz
9
- Keywords: Python,network,graph,network analysis,visualization,interactive,d3js,d3graph,data visualization,web,modular,customizable
9
+ Keywords: Python,network,graph,network analysis,visualization,interactive,d3js,d3graph,data visualization,web,modular,customizable,Network significance
10
10
  Classifier: Programming Language :: Python :: 3
11
11
  Classifier: Operating System :: OS Independent
12
12
  Classifier: Intended Audience :: Education
@@ -27,6 +27,8 @@ Requires-Dist: packaging
27
27
  Requires-Dist: markupsafe
28
28
  Requires-Dist: python-louvain
29
29
  Requires-Dist: datazets
30
+ Requires-Dist: distfit
31
+ Requires-Dist: tqdm
30
32
  Dynamic: license-file
31
33
 
32
34
  # D3graph: Interactive force-directed networks
@@ -17,7 +17,7 @@ from d3graph.d3graph import (
17
17
 
18
18
  __author__ = 'Erdogan Tasksen'
19
19
  __email__ = 'erdogant@gmail.com'
20
- __version__ = '3.0.0'
20
+ __version__ = '3.1.0'
21
21
 
22
22
  # Setup root logger
23
23
  _logger = logging.getLogger('d3graph')
@@ -15,6 +15,7 @@ from sys import platform
15
15
  from tempfile import gettempdir
16
16
  from unicodedata import normalize
17
17
  import uuid
18
+ from tqdm import tqdm
18
19
 
19
20
  from community import community_louvain
20
21
  import colourmap as cm
@@ -26,6 +27,7 @@ from ismember import ismember
26
27
  from jinja2 import Environment, PackageLoader
27
28
  from packaging import version
28
29
  import datazets as dz
30
+ from distfit import distfit
29
31
 
30
32
  logger = logging.getLogger(__name__)
31
33
 
@@ -543,6 +545,7 @@ class d3graph:
543
545
  'color': color of the node
544
546
  'opacity': Opacity of the node
545
547
  'size': size of the node
548
+ 'proba': Significance (p-value) of the node from network_significance(). NaN until that method is run.
546
549
  'fontcolor': color of the node text
547
550
  'fontsize': node text size
548
551
  'edge_size': edge_size of the node
@@ -682,6 +685,7 @@ class d3graph:
682
685
  'fontcolor': str(fontcolor[i]),
683
686
  'fontsize': str(fontsize[i]),
684
687
  'size': size[i],
688
+ 'proba': np.nan,
685
689
  'edge_size': edge_size[i],
686
690
  'edge_color': edge_color[i],
687
691
  'group': group[i]}
@@ -886,6 +890,7 @@ class d3graph:
886
890
  'highlight_full_network': self.config.get('highlight_full_network', True),
887
891
  'save_button': self.config['save_button'],
888
892
  'node_text_inside': self.config.get('node_text_inside', False),
893
+ 'SIGNIFICANCE_ALPHA': self.config.get('significance_alpha', 0.05),
889
894
  'CLICK_COMMENT': CLICK_COMMENT,
890
895
  'CLICK_FILL': click_properties['fill'],
891
896
  'CLICK_STROKE': click_properties['stroke'],
@@ -956,6 +961,289 @@ class d3graph:
956
961
  # Set to config
957
962
  self.config['filepath'] = Path(filepath)
958
963
 
964
+ # Network statistics
965
+ def network_statistic(self, adjmat, statistic):
966
+ """
967
+ Calculate node-level network statistics.
968
+
969
+ Parameters
970
+ ----------
971
+ adjmat : pd.DataFrame
972
+ Weighted adjacency matrix.
973
+
974
+ statistic : str
975
+ One of:
976
+ - pagerank
977
+ - hits_hub
978
+ - hits_authority
979
+ - degree
980
+ - closeness
981
+ - betweenness
982
+
983
+ Returns
984
+ -------
985
+ pd.Series
986
+ """
987
+
988
+ if not isinstance(adjmat, pd.DataFrame):
989
+ raise TypeError("adjmat must be a pandas DataFrame.")
990
+
991
+ if adjmat.shape[0] != adjmat.shape[1]:
992
+ raise ValueError("adjmat must be square.")
993
+
994
+ G = nx.from_pandas_adjacency(adjmat, create_using=nx.DiGraph)
995
+ statistic = statistic.lower()
996
+
997
+ if statistic == "pagerank":
998
+ values = nx.pagerank(G, weight="weight")
999
+ elif statistic in ["hits_hub", "hub"]:
1000
+ hubs, authorities = nx.hits(G, max_iter=1000, normalized=True)
1001
+ values = hubs
1002
+ elif statistic in ["hits_authority", "authority"]:
1003
+ hubs, authorities = nx.hits(G, max_iter=1000, normalized=True)
1004
+ values = authorities
1005
+ elif statistic == "degree":
1006
+ values = dict(G.degree(weight="weight"))
1007
+ elif statistic == "closeness":
1008
+ values = nx.closeness_centrality(G, distance="weight")
1009
+ elif statistic == "betweenness":
1010
+ values = nx.betweenness_centrality(G, weight="weight")
1011
+ else:
1012
+ raise ValueError(f"Unknown statistic: {statistic}")
1013
+
1014
+ return pd.Series(values, name=statistic)
1015
+
1016
+ def network_randomize(self, adjmat, nswap=None, max_tries=None, seed=None):
1017
+ """
1018
+ Degree-preserving randomization using Maslov-Sneppen edge rewiring.
1019
+
1020
+ Preserves:
1021
+ - Out-degree for directed graphs
1022
+ - In-degree for directed graphs
1023
+ - Degree for undirected graphs
1024
+ - Edge weights
1025
+
1026
+ Parameters
1027
+ ----------
1028
+ adjmat : pd.DataFrame
1029
+ Square adjacency matrix.
1030
+
1031
+ nswap : int, optional
1032
+ Number of successful edge swaps.
1033
+
1034
+ max_tries : int, optional
1035
+ Maximum number of attempted swaps.
1036
+
1037
+ seed : int, optional
1038
+ Random seed.
1039
+
1040
+ Returns
1041
+ -------
1042
+ pd.DataFrame
1043
+ Randomized adjacency matrix.
1044
+ """
1045
+
1046
+ if not isinstance(adjmat, pd.DataFrame):
1047
+ raise TypeError("adjmat must be a pandas DataFrame.")
1048
+
1049
+ if adjmat.shape[0] != adjmat.shape[1]:
1050
+ raise ValueError("adjmat must be square.")
1051
+
1052
+ index = adjmat.index
1053
+ columns = adjmat.columns
1054
+ A = adjmat.to_numpy(copy=True)
1055
+
1056
+ # Store edges as list of tuples (more stable than numpy array)
1057
+ edges = list(zip(*np.where(A > 0)))
1058
+
1059
+ if len(edges) < 2:
1060
+ return adjmat.copy()
1061
+ if nswap is None:
1062
+ nswap = len(edges) * 10
1063
+ if max_tries is None:
1064
+ max_tries = nswap * 20
1065
+
1066
+ rng = np.random.default_rng(seed)
1067
+ # O(1) lookup
1068
+ edge_set = set(edges)
1069
+ successful = 0
1070
+ tries = 0
1071
+
1072
+ while successful < nswap and tries < max_tries:
1073
+ tries += 1
1074
+ idx1, idx2 = rng.choice(len(edges), size=2, replace=False)
1075
+ u, v = edges[idx1]
1076
+ x, y = edges[idx2]
1077
+ # Need four distinct nodes
1078
+ if len({u, v, x, y}) < 4:
1079
+ continue
1080
+
1081
+ # New edges
1082
+ e1 = (u, y)
1083
+ e2 = (x, v)
1084
+
1085
+ # No self loops
1086
+ if u == y or x == v:
1087
+ continue
1088
+
1089
+ # Avoid duplicates
1090
+ if e1 in edge_set or e2 in edge_set:
1091
+ continue
1092
+
1093
+ # Store weights
1094
+ w1 = A[u, v]
1095
+ w2 = A[x, y]
1096
+
1097
+ # Perform swap
1098
+ A[u, v] = 0
1099
+ A[x, y] = 0
1100
+
1101
+ A[u, y] = w1
1102
+ A[x, v] = w2
1103
+
1104
+ # Update edge tracking
1105
+ edge_set.remove((u, v))
1106
+ edge_set.remove((x, y))
1107
+
1108
+ edge_set.add(e1)
1109
+ edge_set.add(e2)
1110
+
1111
+ edges[idx1] = e1
1112
+ edges[idx2] = e2
1113
+
1114
+ successful += 1
1115
+
1116
+ return pd.DataFrame(A, index=index, columns=columns)
1117
+
1118
+ def network_significance(
1119
+ self,
1120
+ adjmat,
1121
+ statistic,
1122
+ n_top=100,
1123
+ n_random=1000,
1124
+ nswap=None,
1125
+ alpha=0.05,
1126
+ seed=None,
1127
+ ):
1128
+ """
1129
+ Test significance of top scoring nodes against
1130
+ degree-preserving randomized networks.
1131
+
1132
+ Parameters
1133
+ ----------
1134
+ adjmat : pd.DataFrame
1135
+ Weighted adjacency matrix. Node labels are sanitized the same
1136
+ way graph() does (matching self.adjmat / self.node_properties),
1137
+ so passing your original, unsanitized adjmat here is safe —
1138
+ but for identical results, prefer passing self.adjmat.
1139
+ statistic : str
1140
+ 'pagerank'
1141
+ 'hits_hub'
1142
+ 'hits_authority'
1143
+ 'closeness'
1144
+ 'betweenness'
1145
+ Note: 'degree' is not supported here — network_randomize() preserves every node's in/out-degree exactly,
1146
+ so there is no valid null distribution to test it against (see Note below).
1147
+ n_top : int
1148
+ Number of highest scoring nodes tested.
1149
+ n_random : int
1150
+ Number of randomized networks.
1151
+ nswap : int
1152
+ Number of edge swaps per random network.
1153
+ alpha : float
1154
+ Significance threshold.
1155
+ seed : int
1156
+ Random seed.
1157
+
1158
+ Note
1159
+ ----
1160
+ Call this *after* set_node_properties(), not before: set_node_properties()
1161
+ rebuilds self.node_properties from scratch on every call, so calling it
1162
+ again after network_significance() would silently erase the computed
1163
+ 'proba' values.
1164
+
1165
+ 'degree' is deliberately unsupported as a `statistic`: network_randomize()
1166
+ (Maslov-Sneppen edge swaps) holds every node's in/out-degree exactly
1167
+ fixed by construction, so the null distribution for 'degree' has
1168
+ ~zero variance and any p-value computed from it is a fitting
1169
+ artifact, not a real result — it's tested here and raises ValueError.
1170
+
1171
+ Returns
1172
+ -------
1173
+ pd.DataFrame
1174
+ Significance results.
1175
+ """
1176
+ rng = np.random.default_rng(seed)
1177
+ # Make copy
1178
+ adjmat = data_checks(adjmat.copy())
1179
+
1180
+ # 'degree' can never be meaningfully tested against this null model:
1181
+ # the Maslov-Sneppen double-edge-swap is defined to hold every
1182
+ # node's in-degree and out-degree exactly fixed (that's the "degree-
1183
+ # preserving" part), so network_statistic(random_adj, 'degree') is
1184
+ # identical to the real degree in every single randomization for an
1185
+ # unweighted/binary graph.
1186
+ if statistic.lower() == 'degree':
1187
+ raise ValueError("Network_significance(statistic='degree') is not meaningful.")
1188
+
1189
+ # Remember alpha so write_html() can pass it into the browser as SIGNIFICANCE_ALPHA
1190
+ self.config['significance_alpha'] = alpha
1191
+
1192
+ # Real network scores
1193
+ real_scores = self.network_statistic(adjmat, statistic)
1194
+ top_nodes = real_scores.sort_values(ascending=False).head(n_top).index
1195
+
1196
+ # Null distributions
1197
+ logger.info('Creating degree-preserving randomized networks')
1198
+ null_scores = {node: [] for node in top_nodes}
1199
+
1200
+ for i in tqdm(range(n_random)):
1201
+ random_adj = self.network_randomize(adjmat, nswap=nswap, seed=rng.integers(0, 1_000_000))
1202
+ scores = self.network_statistic(random_adj, statistic)
1203
+ # Store only the relevant node-scores
1204
+ for node in top_nodes:
1205
+ null_scores[node].append(scores[node])
1206
+
1207
+ # Statistics
1208
+ logger.info(f'Computing {statistic} significance for the top {len(top_nodes)} nodes.')
1209
+ results = []
1210
+ for node in tqdm(top_nodes):
1211
+ # Get the real score
1212
+ y = real_scores[node]
1213
+ # Null distribution for this node
1214
+ rand_scores = np.asarray(null_scores[node])
1215
+ # Set very small values to 0
1216
+ rand_scores = np.where(np.abs(rand_scores) < 1e-10, 0, rand_scores)
1217
+ # Initialize
1218
+ model = distfit(distr="popular", method='parametric', alpha=alpha, verbose=None)
1219
+ # Fit null distribution
1220
+ model_results = model.fit_transform(rand_scores)
1221
+ # model.plot()
1222
+
1223
+ # Calculate probability of observing score >= observed
1224
+ Pout = model.predict(y)
1225
+ # Get Probability
1226
+ y_proba = Pout['y_proba'][0]
1227
+ # FDR Multiple test correction
1228
+ y_proba = np.minimum(y_proba * n_top, 1)
1229
+
1230
+ # Store
1231
+ results.append({"node": node,
1232
+ "score_real": y,
1233
+ "score_random_mean": np.mean(rand_scores),
1234
+ "proba": y_proba,
1235
+ "significant": y_proba < alpha,
1236
+ "statistic": statistic,
1237
+ "distribution": model_results['model']['name'],
1238
+ })
1239
+
1240
+ # Store in node
1241
+ if self.node_properties.get(node):
1242
+ self.node_properties[node]['proba'] = y_proba
1243
+
1244
+ return pd.DataFrame(results).sort_values("proba", ascending=True).reset_index(drop=True)
1245
+
1246
+
959
1247
  def import_example(self, data='energy', url=None, sep=','):
960
1248
  """Import example dataset from github source.
961
1249
 
@@ -1173,6 +1461,7 @@ def json_create(G: nx.Graph, compute_stats: bool = True) -> str:
1173
1461
  nodes[i]['node_color'] = nodes[i].pop('color')
1174
1462
  nodes[i]['node_opacity'] = nodes[i].pop('opacity')
1175
1463
  nodes[i]['node_size'] = nodes[i].pop('size')
1464
+ nodes[i]['node_proba'] = nodes[i].pop('proba')
1176
1465
  nodes[i]['node_size_edge'] = nodes[i].pop('edge_size')
1177
1466
  nodes[i]['node_color_edge'] = nodes[i].pop('edge_color')
1178
1467
  nodes[i]['node_fontcolor'] = nodes[i].pop('fontcolor')
@@ -1743,46 +2032,6 @@ def vec2adjmat(source, target, weight=None, symmetric: bool = True, aggfunc='sum
1743
2032
  return adjmat
1744
2033
 
1745
2034
 
1746
- # %% Convert adjacency matrix to vector
1747
- # def adjmat2vec(adjmat, min_weight: float = 1.0) -> pd.DataFrame:
1748
- # """Convert adjacency matrix into vector with source and target.
1749
-
1750
- # Parameters
1751
- # ----------
1752
- # adjmat : pd.DataFrame()
1753
- # Adjacency matrix.
1754
-
1755
- # min_weight : float
1756
- # edges are returned with a minimum weight.
1757
-
1758
- # Returns
1759
- # -------
1760
- # pd.DataFrame()
1761
- # nodes that are connected based on source and target
1762
-
1763
- # Examples
1764
- # --------
1765
- # >>> source = ['Cloudy', 'Cloudy', 'Sprinkler', 'Rain']
1766
- # >>> target = ['Sprinkler', 'Rain', 'Wet_Grass', 'Wet_Grass']
1767
- # >>> adjmat = vec2adjmat(source, target, weight=[1, 2, 1, 3])
1768
- # >>> vector = adjmat2vec(adjmat)
1769
-
1770
- # """
1771
- # # Convert adjacency matrix into vector
1772
- # logger.info('Converting adjacency matrix into source-target..')
1773
- # adjmat = adjmat.stack().reset_index()
1774
- # # Set columns
1775
- # adjmat.columns = ['source', 'target', 'weight']
1776
- # # Remove self loops and no-connected edges
1777
- # Iloc1 = adjmat['source'] != adjmat['target']
1778
- # Iloc2 = adjmat['weight'] >= min_weight
1779
- # Iloc = Iloc1 & Iloc2
1780
- # # Take only connected nodes
1781
- # adjmat = adjmat.loc[Iloc, :]
1782
- # adjmat.reset_index(drop=True, inplace=True)
1783
- # return adjmat
1784
-
1785
-
1786
2035
  def adjmat2vec(adjmat, min_weight: float = 1.0) -> pd.DataFrame:
1787
2036
  """
1788
2037
  Fast conversion of adjacency matrix → edge list.
@@ -2074,6 +2323,8 @@ def import_example(data='energy', url=None, sep=','):
2074
2323
  return dz.get(data=data, url=url, sep=sep)
2075
2324
 
2076
2325
 
2326
+ # %%
2327
+
2077
2328
  def get_support(support):
2078
2329
  """Support."""
2079
2330
  script=''