graphify-vault 0.3.2__tar.gz → 0.3.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {graphify_vault-0.3.2/graphify_vault.egg-info → graphify_vault-0.3.4}/PKG-INFO +5 -1
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/__main__.py +2 -2
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/cluster.py +110 -1
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/doctor.py +7 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/extract.py +310 -5
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/serve.py +236 -7
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/watch.py +2 -2
- {graphify_vault-0.3.2 → graphify_vault-0.3.4/graphify_vault.egg-info}/PKG-INFO +5 -1
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify_vault.egg-info/entry_points.txt +1 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify_vault.egg-info/requires.txt +4 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/pyproject.toml +4 -3
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/LICENSE +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/README.md +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/__init__.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/analyze.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/benchmark.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/build.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/cache.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/detect.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/export.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/hooks.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/ingest.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/manifest.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/report.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/security.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/skill-aider.md +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/skill-claw.md +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/skill-codex.md +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/skill-copilot.md +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/skill-droid.md +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/skill-kiro.md +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/skill-opencode.md +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/skill-trae.md +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/skill-vscode.md +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/skill-windows.md +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/skill.md +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/transcribe.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/validate.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/wiki.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify_vault.egg-info/SOURCES.txt +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify_vault.egg-info/dependency_links.txt +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify_vault.egg-info/top_level.txt +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/setup.cfg +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_analyze.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_benchmark.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_build.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_cache.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_claude_md.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_cluster.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_confidence.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_detect.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_export.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_extract.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_graph_build_perf.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_hooks.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_hypergraph.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_ingest.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_install.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_languages.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_multilang.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_pipeline.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_rationale.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_report.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_security.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_semantic_similarity.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_serve.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_sort_by.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_spatial_craft_build.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_spatialsdk_query_regression.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_transcribe.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_validate.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_watch.py +0 -0
- {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_wiki.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: graphify_vault
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.4
|
|
4
4
|
Summary: Fork of github.com/safishamsi/graphify with GRAPHIFY_WORKSPACE path resolution for distributed graph artifacts, same-label node merging, incremental build fixes, and local knowledge vault capabilities
|
|
5
5
|
License: MIT License
|
|
6
6
|
|
|
@@ -56,6 +56,9 @@ Requires-Dist: tree-sitter-julia
|
|
|
56
56
|
Requires-Dist: tree-sitter-verilog
|
|
57
57
|
Provides-Extra: mcp
|
|
58
58
|
Requires-Dist: mcp; extra == "mcp"
|
|
59
|
+
Requires-Dist: deep-translator; extra == "mcp"
|
|
60
|
+
Requires-Dist: pypdf; extra == "mcp"
|
|
61
|
+
Requires-Dist: html2text; extra == "mcp"
|
|
59
62
|
Provides-Extra: neo4j
|
|
60
63
|
Requires-Dist: neo4j; extra == "neo4j"
|
|
61
64
|
Provides-Extra: pdf
|
|
@@ -75,6 +78,7 @@ Requires-Dist: faster-whisper; extra == "video"
|
|
|
75
78
|
Requires-Dist: yt-dlp; extra == "video"
|
|
76
79
|
Provides-Extra: all
|
|
77
80
|
Requires-Dist: mcp; extra == "all"
|
|
81
|
+
Requires-Dist: deep-translator; extra == "all"
|
|
78
82
|
Requires-Dist: neo4j; extra == "all"
|
|
79
83
|
Requires-Dist: pypdf; extra == "all"
|
|
80
84
|
Requires-Dist: html2text; extra == "all"
|
|
@@ -1422,7 +1422,7 @@ def main() -> None:
|
|
|
1422
1422
|
sys.exit(1)
|
|
1423
1423
|
from networkx.readwrite import json_graph as _jg
|
|
1424
1424
|
from graphify.build import build_from_json
|
|
1425
|
-
from graphify.cluster import cluster, score_all
|
|
1425
|
+
from graphify.cluster import cluster, score_all, label_communities
|
|
1426
1426
|
from graphify.analyze import god_nodes, surprising_connections, suggest_questions
|
|
1427
1427
|
from graphify.report import generate
|
|
1428
1428
|
from graphify.export import to_json, to_html
|
|
@@ -1435,7 +1435,7 @@ def main() -> None:
|
|
|
1435
1435
|
cohesion = score_all(G, communities)
|
|
1436
1436
|
gods = god_nodes(G)
|
|
1437
1437
|
surprises = surprising_connections(G, communities)
|
|
1438
|
-
labels =
|
|
1438
|
+
labels = label_communities(G, communities)
|
|
1439
1439
|
questions = suggest_questions(G, communities, labels)
|
|
1440
1440
|
tokens = {"input": 0, "output": 0}
|
|
1441
1441
|
report = generate(G, communities, cohesion, labels, gods, surprises,
|
|
@@ -3,6 +3,7 @@ from __future__ import annotations
|
|
|
3
3
|
import contextlib
|
|
4
4
|
import inspect
|
|
5
5
|
import io
|
|
6
|
+
import re
|
|
6
7
|
import sys
|
|
7
8
|
import networkx as nx
|
|
8
9
|
|
|
@@ -18,6 +19,87 @@ def _suppress_output():
|
|
|
18
19
|
return contextlib.redirect_stdout(io.StringIO())
|
|
19
20
|
|
|
20
21
|
|
|
22
|
+
def _label_token_overlap(label_a: str, label_b: str) -> float:
|
|
23
|
+
"""Compute token overlap ratio between two labels.
|
|
24
|
+
|
|
25
|
+
Splits labels into lowercase tokens using both non-alphanumeric
|
|
26
|
+
boundaries AND camelCase/PascalCase boundaries. Returns the
|
|
27
|
+
Jaccard similarity of the token sets. This provides a lightweight
|
|
28
|
+
semantic similarity signal without requiring any external NLP models.
|
|
29
|
+
"""
|
|
30
|
+
def _tokens(text: str) -> set[str]:
|
|
31
|
+
# Split on non-alphanumeric boundaries
|
|
32
|
+
parts = re.split(r'[^a-zA-Z0-9]+', text)
|
|
33
|
+
# Also split camelCase/PascalCase: "VideoComponent" -> "Video", "Component"
|
|
34
|
+
expanded = []
|
|
35
|
+
for part in parts:
|
|
36
|
+
# Split on camelCase boundaries: lowercase followed by uppercase
|
|
37
|
+
sub_parts = re.sub(r'([a-z])([A-Z])', r'\1 \2', part).split()
|
|
38
|
+
# Split on acronyms: uppercase followed by uppercase+lowercase (e.g., "XMLParser" -> "XML", "Parser")
|
|
39
|
+
sub_parts = [re.sub(r'([A-Z]+)([A-Z][a-z])', r'\1 \2', s).split() for s in sub_parts]
|
|
40
|
+
expanded.extend([t for group in sub_parts for t in group])
|
|
41
|
+
return {t.lower() for t in expanded if len(t) > 2}
|
|
42
|
+
ta = _tokens(label_a)
|
|
43
|
+
tb = _tokens(label_b)
|
|
44
|
+
if not ta or not tb:
|
|
45
|
+
return 0.0
|
|
46
|
+
intersection = ta & tb
|
|
47
|
+
union = ta | tb
|
|
48
|
+
return len(intersection) / len(union) if union else 0.0
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _add_semantic_edges(G: nx.Graph, weight: float = 0.3) -> nx.Graph:
|
|
52
|
+
"""Add weak semantic edges between nodes with overlapping labels.
|
|
53
|
+
|
|
54
|
+
For each pair of nodes in the same connected component that share
|
|
55
|
+
label tokens (Jaccard > 0.3), adds a 'semantically_similar_to' edge
|
|
56
|
+
with the given weight. These edges guide Leiden/Louvain to group
|
|
57
|
+
semantically related nodes together even without structural edges.
|
|
58
|
+
|
|
59
|
+
This is a non-destructive overlay: the original graph is copied and
|
|
60
|
+
only new edges are added (no existing edges are modified).
|
|
61
|
+
"""
|
|
62
|
+
if G.number_of_nodes() < 2:
|
|
63
|
+
return G.copy()
|
|
64
|
+
|
|
65
|
+
# Only consider nodes with meaningful labels
|
|
66
|
+
candidates = []
|
|
67
|
+
for nid in G.nodes:
|
|
68
|
+
label = str(G.nodes[nid].get("label", ""))
|
|
69
|
+
if label and len(label) > 2:
|
|
70
|
+
candidates.append((nid, label))
|
|
71
|
+
|
|
72
|
+
if len(candidates) < 2:
|
|
73
|
+
return G.copy()
|
|
74
|
+
|
|
75
|
+
G_augmented = G.copy()
|
|
76
|
+
added = 0
|
|
77
|
+
|
|
78
|
+
# For efficiency, only check pairs within the same connected component
|
|
79
|
+
# and limit to reasonable graph sizes
|
|
80
|
+
if len(candidates) > 5000:
|
|
81
|
+
# For very large graphs, sample pairs from same-neighborhood
|
|
82
|
+
return G_augmented
|
|
83
|
+
|
|
84
|
+
for i in range(len(candidates)):
|
|
85
|
+
nid_a, label_a = candidates[i]
|
|
86
|
+
for j in range(i + 1, len(candidates)):
|
|
87
|
+
nid_b, label_b = candidates[j]
|
|
88
|
+
# Skip if already connected by an edge
|
|
89
|
+
if G_augmented.has_edge(nid_a, nid_b):
|
|
90
|
+
continue
|
|
91
|
+
similarity = _label_token_overlap(label_a, label_b)
|
|
92
|
+
if similarity > 0.3:
|
|
93
|
+
G_augmented.add_edge(nid_a, nid_b,
|
|
94
|
+
relation="semantically_similar_to",
|
|
95
|
+
confidence="INFERRED",
|
|
96
|
+
confidence_score=round(similarity, 2),
|
|
97
|
+
weight=weight * similarity)
|
|
98
|
+
added += 1
|
|
99
|
+
|
|
100
|
+
return G_augmented
|
|
101
|
+
|
|
102
|
+
|
|
21
103
|
def _partition(G: nx.Graph) -> dict[str, int]:
|
|
22
104
|
"""Run community detection. Returns {node_id: community_id}.
|
|
23
105
|
|
|
@@ -56,7 +138,7 @@ _MAX_COMMUNITY_FRACTION = 0.25 # communities larger than 25% of graph get spli
|
|
|
56
138
|
_MIN_SPLIT_SIZE = 10 # only split if community has at least this many nodes
|
|
57
139
|
|
|
58
140
|
|
|
59
|
-
def cluster(G: nx.Graph) -> dict[int, list[str]]:
|
|
141
|
+
def cluster(G: nx.Graph, *, semantic: bool = True) -> dict[int, list[str]]:
|
|
60
142
|
"""Run Leiden community detection. Returns {community_id: [node_ids]}.
|
|
61
143
|
|
|
62
144
|
Community IDs are stable across runs: 0 = largest community after splitting.
|
|
@@ -65,6 +147,10 @@ def cluster(G: nx.Graph) -> dict[int, list[str]]:
|
|
|
65
147
|
|
|
66
148
|
Accepts directed or undirected graphs. DiGraphs are converted to undirected
|
|
67
149
|
internally since Louvain/Leiden require undirected input.
|
|
150
|
+
|
|
151
|
+
If semantic=True (default), weak semantic edges are added between nodes
|
|
152
|
+
with overlapping labels before clustering, helping semantically related
|
|
153
|
+
nodes group together even without structural edges.
|
|
68
154
|
"""
|
|
69
155
|
if G.number_of_nodes() == 0:
|
|
70
156
|
return {}
|
|
@@ -73,6 +159,10 @@ def cluster(G: nx.Graph) -> dict[int, list[str]]:
|
|
|
73
159
|
if G.number_of_edges() == 0:
|
|
74
160
|
return {i: [n] for i, n in enumerate(sorted(G.nodes))}
|
|
75
161
|
|
|
162
|
+
# Optionally augment graph with semantic edges for better community detection
|
|
163
|
+
if semantic:
|
|
164
|
+
G = _add_semantic_edges(G)
|
|
165
|
+
|
|
76
166
|
# Leiden warns and drops isolates - handle them separately
|
|
77
167
|
isolates = [n for n in G.nodes() if G.degree(n) == 0]
|
|
78
168
|
connected_nodes = [n for n in G.nodes() if G.degree(n) > 0]
|
|
@@ -133,5 +223,24 @@ def cohesion_score(G: nx.Graph, community_nodes: list[str]) -> float:
|
|
|
133
223
|
return round(actual / possible, 2) if possible > 0 else 0.0
|
|
134
224
|
|
|
135
225
|
|
|
226
|
+
def label_communities(G: nx.Graph, communities: dict[int, list[str]]) -> dict[int, str]:
|
|
227
|
+
"""Generate semantic labels for communities based on highest-degree node.
|
|
228
|
+
|
|
229
|
+
For each community, the node with the highest degree (most connections)
|
|
230
|
+
is chosen as the representative label. This is a simple, universal heuristic
|
|
231
|
+
that works for any graph without domain-specific knowledge.
|
|
232
|
+
"""
|
|
233
|
+
labels: dict[int, str] = {}
|
|
234
|
+
for cid, node_ids in communities.items():
|
|
235
|
+
if not node_ids:
|
|
236
|
+
labels[cid] = f"Community {cid}"
|
|
237
|
+
continue
|
|
238
|
+
# Find the node with highest degree in this community
|
|
239
|
+
best_node = max(node_ids, key=lambda n: G.degree(n) if n in G else 0)
|
|
240
|
+
label = G.nodes[best_node].get("label", best_node) if best_node in G else best_node
|
|
241
|
+
labels[cid] = label
|
|
242
|
+
return labels
|
|
243
|
+
|
|
244
|
+
|
|
136
245
|
def score_all(G: nx.Graph, communities: dict[int, list[str]]) -> dict[int, float]:
|
|
137
246
|
return {cid: cohesion_score(G, nodes) for cid, nodes in communities.items()}
|
|
@@ -66,6 +66,13 @@ def _check_graphify_installed() -> bool:
|
|
|
66
66
|
)
|
|
67
67
|
if result.returncode == 0:
|
|
68
68
|
_ok(f"graphify {__version__}")
|
|
69
|
+
# Check for conflicting graphifyy installation
|
|
70
|
+
try:
|
|
71
|
+
from importlib.metadata import version as _v, PackageNotFoundError
|
|
72
|
+
_v("graphifyy")
|
|
73
|
+
_warn("Both graphify_vault and graphifyy detected. 'graphify' command may point to graphifyy. Use 'gv' command instead, or uninstall graphifyy.")
|
|
74
|
+
except PackageNotFoundError:
|
|
75
|
+
pass
|
|
69
76
|
return True
|
|
70
77
|
except Exception:
|
|
71
78
|
pass
|
|
@@ -231,7 +231,7 @@ def extract_document(path: Path, root: Path | None = None) -> dict:
|
|
|
231
231
|
return {"nodes": [], "edges": []}
|
|
232
232
|
|
|
233
233
|
definitions: list[dict[str, Any]] = []
|
|
234
|
-
headings: list[tuple[str, int]] = []
|
|
234
|
+
headings: list[tuple[str, int, int]] = [] # (heading, lineno, level)
|
|
235
235
|
seen_labels: set[str] = set()
|
|
236
236
|
|
|
237
237
|
# Pre-strip lines that are raw HTML (SVG, embedded widgets, etc.)
|
|
@@ -246,8 +246,9 @@ def extract_document(path: Path, root: Path | None = None) -> dict:
|
|
|
246
246
|
m = re.match(r'^(#{1,6})\s+(.+)', line)
|
|
247
247
|
if m:
|
|
248
248
|
heading = m.group(2).strip()
|
|
249
|
+
level = len(m.group(1))
|
|
249
250
|
if _looks_like_symbol_heading(heading):
|
|
250
|
-
headings.append((heading, lineno))
|
|
251
|
+
headings.append((heading, lineno, level))
|
|
251
252
|
|
|
252
253
|
match = _DOC_DEF_RE.search(line)
|
|
253
254
|
if not match:
|
|
@@ -270,7 +271,7 @@ def extract_document(path: Path, root: Path | None = None) -> dict:
|
|
|
270
271
|
definitions.append({"label": label, "lineno": lineno, "bases": bases})
|
|
271
272
|
seen_labels.add(label)
|
|
272
273
|
|
|
273
|
-
for heading, lineno in headings:
|
|
274
|
+
for heading, lineno, _level in headings:
|
|
274
275
|
if heading not in seen_labels:
|
|
275
276
|
definitions.append({"label": heading, "lineno": lineno, "bases": []})
|
|
276
277
|
seen_labels.add(heading)
|
|
@@ -297,7 +298,7 @@ def extract_document(path: Path, root: Path | None = None) -> dict:
|
|
|
297
298
|
seen_labels.add(name)
|
|
298
299
|
|
|
299
300
|
heading_positions: dict[str, list[int]] = {}
|
|
300
|
-
for heading, lineno in headings:
|
|
301
|
+
for heading, lineno, _level in headings:
|
|
301
302
|
heading_positions.setdefault(heading, []).append(lineno)
|
|
302
303
|
|
|
303
304
|
lines = text.splitlines()
|
|
@@ -316,7 +317,7 @@ def extract_document(path: Path, root: Path | None = None) -> dict:
|
|
|
316
317
|
for idx, item in enumerate(definitions):
|
|
317
318
|
next_heading_lineno = None
|
|
318
319
|
current_lineno = int(item["lineno"])
|
|
319
|
-
for heading, heading_lineno in headings:
|
|
320
|
+
for heading, heading_lineno, _level in headings:
|
|
320
321
|
if heading_lineno > current_lineno:
|
|
321
322
|
next_heading_lineno = heading_lineno
|
|
322
323
|
break
|
|
@@ -468,6 +469,289 @@ def _resolve_document_cross_references(
|
|
|
468
469
|
return new_edges
|
|
469
470
|
|
|
470
471
|
|
|
472
|
+
def _add_heading_structure_edges(
|
|
473
|
+
doc_results: list[dict], doc_paths: list[Path], root: Path | None = None,
|
|
474
|
+
) -> list[dict]:
|
|
475
|
+
"""Add ``contains`` edges based on heading hierarchy within documents.
|
|
476
|
+
|
|
477
|
+
For each document, re-parses headings to build a parent-child tree:
|
|
478
|
+
an H2 heading "contains" all H3 headings under it until the next H2,
|
|
479
|
+
and so on. Also creates ``contains`` edges from heading nodes to
|
|
480
|
+
concept nodes (definitions) that fall under that heading section.
|
|
481
|
+
"""
|
|
482
|
+
_HEADING_RE = re.compile(r'^(#{1,6})\s+(.+)', re.MULTILINE)
|
|
483
|
+
|
|
484
|
+
new_edges: list[dict] = []
|
|
485
|
+
seen_pairs: set[tuple[str, str]] = set()
|
|
486
|
+
|
|
487
|
+
for result, path in zip(doc_results, doc_paths):
|
|
488
|
+
relpath = _doc_relpath(path, root)
|
|
489
|
+
stem = _doc_file_stem(path)
|
|
490
|
+
nodes = result.get("nodes", [])
|
|
491
|
+
if not nodes:
|
|
492
|
+
continue
|
|
493
|
+
|
|
494
|
+
try:
|
|
495
|
+
text = path.read_text(encoding="utf-8", errors="replace")
|
|
496
|
+
except OSError:
|
|
497
|
+
continue
|
|
498
|
+
|
|
499
|
+
# Parse all headings with levels
|
|
500
|
+
all_headings: list[tuple[str, int, int]] = [] # (heading, lineno, level)
|
|
501
|
+
for m in _HEADING_RE.finditer(text):
|
|
502
|
+
heading = m.group(2).strip()
|
|
503
|
+
level = len(m.group(1))
|
|
504
|
+
# Calculate line number from match position
|
|
505
|
+
lineno = text[:m.start()].count("\n") + 1
|
|
506
|
+
all_headings.append((heading, lineno, level))
|
|
507
|
+
|
|
508
|
+
if not all_headings:
|
|
509
|
+
continue
|
|
510
|
+
|
|
511
|
+
# Build heading node ID map: heading label → nid
|
|
512
|
+
heading_nid_map: dict[str, str] = {}
|
|
513
|
+
for n in nodes:
|
|
514
|
+
heading_nid_map[n.get("label", "")] = n.get("id", "")
|
|
515
|
+
|
|
516
|
+
# Build parent-child heading relationships using a stack
|
|
517
|
+
# Stack contains (heading, lineno, level, nid)
|
|
518
|
+
stack: list[tuple[str, int, int, str]] = []
|
|
519
|
+
heading_contains: list[tuple[str, str]] = [] # (parent_nid, child_nid)
|
|
520
|
+
|
|
521
|
+
for heading, lineno, level in all_headings:
|
|
522
|
+
nid = heading_nid_map.get(heading, _make_id(stem, heading))
|
|
523
|
+
# Pop stack until we find a parent (lower level number = higher rank)
|
|
524
|
+
while stack and stack[-1][2] >= level:
|
|
525
|
+
stack.pop()
|
|
526
|
+
if stack:
|
|
527
|
+
parent_nid = stack[-1][3]
|
|
528
|
+
heading_contains.append((parent_nid, nid))
|
|
529
|
+
stack.append((heading, lineno, level, nid))
|
|
530
|
+
|
|
531
|
+
# Create contains edges for heading hierarchy
|
|
532
|
+
for parent_nid, child_nid in heading_contains:
|
|
533
|
+
if parent_nid == child_nid:
|
|
534
|
+
continue
|
|
535
|
+
pair = (parent_nid, child_nid)
|
|
536
|
+
if pair in seen_pairs:
|
|
537
|
+
continue
|
|
538
|
+
seen_pairs.add(pair)
|
|
539
|
+
new_edges.append({
|
|
540
|
+
"source": parent_nid,
|
|
541
|
+
"target": child_nid,
|
|
542
|
+
"relation": "contains",
|
|
543
|
+
"confidence": "EXTRACTED",
|
|
544
|
+
"confidence_score": 1.0,
|
|
545
|
+
"source_file": relpath,
|
|
546
|
+
"weight": 1.0,
|
|
547
|
+
})
|
|
548
|
+
|
|
549
|
+
# Create contains edges from heading to concept nodes in that section
|
|
550
|
+
# A concept node belongs to the nearest preceding heading
|
|
551
|
+
sorted_headings = sorted(all_headings, key=lambda h: h[1])
|
|
552
|
+
for n in nodes:
|
|
553
|
+
nid = n.get("id", "")
|
|
554
|
+
label = n.get("label", "")
|
|
555
|
+
start_line = n.get("start_line")
|
|
556
|
+
if not isinstance(start_line, int):
|
|
557
|
+
continue
|
|
558
|
+
# Find the nearest preceding heading
|
|
559
|
+
parent_heading_nid = None
|
|
560
|
+
for heading, h_lineno, h_level in reversed(sorted_headings):
|
|
561
|
+
if h_lineno <= start_line:
|
|
562
|
+
parent_heading_nid = heading_nid_map.get(heading, _make_id(stem, heading))
|
|
563
|
+
break
|
|
564
|
+
if parent_heading_nid and parent_heading_nid != nid:
|
|
565
|
+
pair = (parent_heading_nid, nid)
|
|
566
|
+
if pair not in seen_pairs:
|
|
567
|
+
seen_pairs.add(pair)
|
|
568
|
+
new_edges.append({
|
|
569
|
+
"source": parent_heading_nid,
|
|
570
|
+
"target": nid,
|
|
571
|
+
"relation": "contains",
|
|
572
|
+
"confidence": "EXTRACTED",
|
|
573
|
+
"confidence_score": 0.9,
|
|
574
|
+
"source_file": relpath,
|
|
575
|
+
"weight": 0.9,
|
|
576
|
+
})
|
|
577
|
+
|
|
578
|
+
return new_edges
|
|
579
|
+
|
|
580
|
+
|
|
581
|
+
def _add_co_occurrence_edges(
|
|
582
|
+
doc_results: list[dict], doc_paths: list[Path], root: Path | None = None,
|
|
583
|
+
) -> list[dict]:
|
|
584
|
+
"""Add ``co_occurs_in`` edges between concept nodes in the same document.
|
|
585
|
+
|
|
586
|
+
Nodes appearing in the same paragraph get a co-occurrence edge with
|
|
587
|
+
weight=1.0. Weight decays with paragraph distance:
|
|
588
|
+
weight = 1.0 / (1 + abs(paragraph_distance)).
|
|
589
|
+
Only creates edges between nodes from the same document.
|
|
590
|
+
"""
|
|
591
|
+
new_edges: list[dict] = []
|
|
592
|
+
seen_pairs: set[tuple[str, str]] = set()
|
|
593
|
+
|
|
594
|
+
for result, path in zip(doc_results, doc_paths):
|
|
595
|
+
relpath = _doc_relpath(path, root)
|
|
596
|
+
nodes = result.get("nodes", [])
|
|
597
|
+
if len(nodes) < 2:
|
|
598
|
+
continue
|
|
599
|
+
|
|
600
|
+
# Compute paragraph index for each node based on start_line
|
|
601
|
+
# Paragraphs are separated by blank lines
|
|
602
|
+
try:
|
|
603
|
+
text = path.read_text(encoding="utf-8", errors="replace")
|
|
604
|
+
except OSError:
|
|
605
|
+
continue
|
|
606
|
+
lines = text.splitlines()
|
|
607
|
+
|
|
608
|
+
# Build line → paragraph_number map
|
|
609
|
+
para_map: dict[int, int] = {}
|
|
610
|
+
para_idx = 0
|
|
611
|
+
for i, line in enumerate(lines, start=1):
|
|
612
|
+
if i == 1 or (line.strip() == "" and i > 1):
|
|
613
|
+
# New paragraph starts after blank line (or at line 1)
|
|
614
|
+
if i > 1:
|
|
615
|
+
para_idx += 1
|
|
616
|
+
para_map[i] = para_idx
|
|
617
|
+
|
|
618
|
+
# Get paragraph index for each node
|
|
619
|
+
node_paras: list[tuple[str, int]] = [] # (nid, paragraph)
|
|
620
|
+
for n in nodes:
|
|
621
|
+
nid = n.get("id", "")
|
|
622
|
+
start_line = n.get("start_line") or n.get("source_location", "")
|
|
623
|
+
# Parse line number
|
|
624
|
+
if isinstance(start_line, int):
|
|
625
|
+
lineno = start_line
|
|
626
|
+
elif isinstance(start_line, str) and start_line.startswith("L"):
|
|
627
|
+
try:
|
|
628
|
+
lineno = int(start_line[1:])
|
|
629
|
+
except ValueError:
|
|
630
|
+
continue
|
|
631
|
+
else:
|
|
632
|
+
continue
|
|
633
|
+
para = para_map.get(lineno, 0)
|
|
634
|
+
node_paras.append((nid, para))
|
|
635
|
+
|
|
636
|
+
# Create co-occurrence edges between all pairs in the same document
|
|
637
|
+
for i in range(len(node_paras)):
|
|
638
|
+
nid_i, para_i = node_paras[i]
|
|
639
|
+
for j in range(i + 1, len(node_paras)):
|
|
640
|
+
nid_j, para_j = node_paras[j]
|
|
641
|
+
if nid_i == nid_j:
|
|
642
|
+
continue
|
|
643
|
+
# Sort pair to avoid duplicates
|
|
644
|
+
pair = tuple(sorted([nid_i, nid_j]))
|
|
645
|
+
if pair in seen_pairs:
|
|
646
|
+
continue
|
|
647
|
+
dist = abs(para_i - para_j)
|
|
648
|
+
# Only create edges within a reasonable distance (same doc, ≤5 paragraphs)
|
|
649
|
+
if dist > 5:
|
|
650
|
+
continue
|
|
651
|
+
weight = round(1.0 / (1 + dist), 2)
|
|
652
|
+
seen_pairs.add(pair)
|
|
653
|
+
new_edges.append({
|
|
654
|
+
"source": pair[0],
|
|
655
|
+
"target": pair[1],
|
|
656
|
+
"relation": "co_occurs_in",
|
|
657
|
+
"confidence": "INFERRED",
|
|
658
|
+
"confidence_score": weight,
|
|
659
|
+
"source_file": relpath,
|
|
660
|
+
"weight": weight,
|
|
661
|
+
})
|
|
662
|
+
|
|
663
|
+
return new_edges
|
|
664
|
+
|
|
665
|
+
|
|
666
|
+
def _resolve_markdown_link_references(
|
|
667
|
+
doc_results: list[dict], doc_paths: list[Path], root: Path | None = None,
|
|
668
|
+
) -> list[dict]:
|
|
669
|
+
"""Resolve Markdown ``[text](url)`` links into ``references`` edges.
|
|
670
|
+
|
|
671
|
+
Scans all document nodes' raws for Markdown links. When the URL
|
|
672
|
+
resolves to another document in the corpus, creates a ``references``
|
|
673
|
+
edge from the source concept node to the target document's file-level
|
|
674
|
+
node (or the best matching concept node within the target document).
|
|
675
|
+
"""
|
|
676
|
+
_MD_LINK = re.compile(r'\[([^\]]+)\]\(([^)]+)\)')
|
|
677
|
+
|
|
678
|
+
# Build file-level node ID map: relpath → nid
|
|
679
|
+
file_nid_map: dict[str, str] = {}
|
|
680
|
+
# Build label → nid map for intra-document concept matching
|
|
681
|
+
label_to_nid: dict[str, str] = {}
|
|
682
|
+
for result, path in zip(doc_results, doc_paths):
|
|
683
|
+
relpath = _doc_relpath(path, root)
|
|
684
|
+
stem = _doc_file_stem(path)
|
|
685
|
+
file_nid = _make_id(stem)
|
|
686
|
+
file_nid_map[relpath] = file_nid
|
|
687
|
+
# Also map by stem name for partial matches
|
|
688
|
+
file_nid_map[path.name] = file_nid
|
|
689
|
+
for n in result.get("nodes", []):
|
|
690
|
+
lbl = n.get("label", "")
|
|
691
|
+
nid = n.get("id", "")
|
|
692
|
+
if lbl and nid:
|
|
693
|
+
label_to_nid.setdefault(lbl, nid)
|
|
694
|
+
|
|
695
|
+
# Build a set of known document relative paths for fast lookup
|
|
696
|
+
known_relpaths = set(file_nid_map.keys())
|
|
697
|
+
|
|
698
|
+
new_edges: list[dict] = []
|
|
699
|
+
seen_pairs: set[tuple[str, str]] = set()
|
|
700
|
+
|
|
701
|
+
for result, path in zip(doc_results, doc_paths):
|
|
702
|
+
relpath = _doc_relpath(path, root)
|
|
703
|
+
for n in result.get("nodes", []):
|
|
704
|
+
nid = n.get("id", "")
|
|
705
|
+
raws_blocks = n.get("raws", [])
|
|
706
|
+
if not raws_blocks:
|
|
707
|
+
continue
|
|
708
|
+
raws_text = "\n".join(raws_blocks) if isinstance(raws_blocks, list) else str(raws_blocks)
|
|
709
|
+
|
|
710
|
+
for m in _MD_LINK.finditer(raws_text):
|
|
711
|
+
link_text = m.group(1).strip()
|
|
712
|
+
link_url = m.group(2).strip()
|
|
713
|
+
|
|
714
|
+
# Skip external URLs, anchors, and image links
|
|
715
|
+
if link_url.startswith(("http://", "https://", "#", "mailto:", "data:")):
|
|
716
|
+
continue
|
|
717
|
+
# Strip anchor fragment
|
|
718
|
+
link_url = link_url.split("#")[0].split("?")[0]
|
|
719
|
+
if not link_url:
|
|
720
|
+
continue
|
|
721
|
+
|
|
722
|
+
# Try to resolve the URL to a known document
|
|
723
|
+
target_nid = None
|
|
724
|
+
# Direct relative path match
|
|
725
|
+
if link_url in known_relpaths:
|
|
726
|
+
target_nid = file_nid_map.get(link_url)
|
|
727
|
+
else:
|
|
728
|
+
# Try matching by filename
|
|
729
|
+
target_name = Path(link_url).name
|
|
730
|
+
if target_name in file_nid_map:
|
|
731
|
+
target_nid = file_nid_map[target_name]
|
|
732
|
+
|
|
733
|
+
# If no file-level match, try matching link text to a concept label
|
|
734
|
+
if target_nid is None and link_text in label_to_nid:
|
|
735
|
+
target_nid = label_to_nid[link_text]
|
|
736
|
+
|
|
737
|
+
if target_nid is None or target_nid == nid:
|
|
738
|
+
continue
|
|
739
|
+
if (nid, target_nid) in seen_pairs:
|
|
740
|
+
continue
|
|
741
|
+
seen_pairs.add((nid, target_nid))
|
|
742
|
+
new_edges.append({
|
|
743
|
+
"source": nid,
|
|
744
|
+
"target": target_nid,
|
|
745
|
+
"relation": "references",
|
|
746
|
+
"confidence": "EXTRACTED",
|
|
747
|
+
"confidence_score": 1.0,
|
|
748
|
+
"source_file": relpath,
|
|
749
|
+
"weight": 1.0,
|
|
750
|
+
})
|
|
751
|
+
|
|
752
|
+
return new_edges
|
|
753
|
+
|
|
754
|
+
|
|
471
755
|
def merge_document_semantic_extracts(doc_extract: dict[str, Any], semantic_extract: dict[str, Any]) -> dict[str, Any]:
|
|
472
756
|
"""Merge deterministic doc symbols with semantic extraction.
|
|
473
757
|
|
|
@@ -3888,6 +4172,27 @@ def extract(paths: list[Path], cache_root: Path | None = None) -> dict:
|
|
|
3888
4172
|
import logging
|
|
3889
4173
|
logging.getLogger(__name__).warning("Document cross-reference resolution failed, skipping: %s", exc)
|
|
3890
4174
|
|
|
4175
|
+
try:
|
|
4176
|
+
md_link_edges = _resolve_markdown_link_references(doc_results, doc_paths, root)
|
|
4177
|
+
all_edges.extend(md_link_edges)
|
|
4178
|
+
except Exception as exc:
|
|
4179
|
+
import logging
|
|
4180
|
+
logging.getLogger(__name__).warning("Markdown link reference resolution failed, skipping: %s", exc)
|
|
4181
|
+
|
|
4182
|
+
try:
|
|
4183
|
+
co_occurs_edges = _add_co_occurrence_edges(doc_results, doc_paths, root)
|
|
4184
|
+
all_edges.extend(co_occurs_edges)
|
|
4185
|
+
except Exception as exc:
|
|
4186
|
+
import logging
|
|
4187
|
+
logging.getLogger(__name__).warning("Co-occurrence edge creation failed, skipping: %s", exc)
|
|
4188
|
+
|
|
4189
|
+
try:
|
|
4190
|
+
heading_edges = _add_heading_structure_edges(doc_results, doc_paths, root)
|
|
4191
|
+
all_edges.extend(heading_edges)
|
|
4192
|
+
except Exception as exc:
|
|
4193
|
+
import logging
|
|
4194
|
+
logging.getLogger(__name__).warning("Heading structure edge creation failed, skipping: %s", exc)
|
|
4195
|
+
|
|
3891
4196
|
return {
|
|
3892
4197
|
"nodes": all_nodes,
|
|
3893
4198
|
"edges": all_edges,
|
|
@@ -11,6 +11,7 @@ from pathlib import Path
|
|
|
11
11
|
import networkx as nx
|
|
12
12
|
from networkx.readwrite import json_graph
|
|
13
13
|
from graphify.security import sanitize_label
|
|
14
|
+
from graphify.cluster import label_communities
|
|
14
15
|
|
|
15
16
|
|
|
16
17
|
def _resolve_graph_path(graph_path: str) -> str:
|
|
@@ -105,6 +106,12 @@ _TASK_MARKERS = (
|
|
|
105
106
|
"relevant", "entry", "workflow", "flow", "path", "components", "component", "types",
|
|
106
107
|
)
|
|
107
108
|
|
|
109
|
+
_CONNECTION_MARKERS = (
|
|
110
|
+
"连接", "桥接", "关联", "中间", "之间", "关系",
|
|
111
|
+
"connect", "bridge", "link", "between", "relation", "intermediate",
|
|
112
|
+
"how does", "how do", "what connects", "what links", "what relates",
|
|
113
|
+
)
|
|
114
|
+
|
|
108
115
|
_STOP_TERMS = {
|
|
109
116
|
"a", "an", "the", "and", "or", "but", "if", "then", "how", "what", "which", "where", "when",
|
|
110
117
|
"why", "who", "whom", "whose", "should", "would", "could", "can", "do", "does", "did", "done",
|
|
@@ -483,6 +490,9 @@ def _classify_query_mode(question: str, terms: list[str] | None = None) -> str:
|
|
|
483
490
|
identifier_like = bool(re.fullmatch(r"[A-Za-z_][A-Za-z0-9_\.]*", stripped))
|
|
484
491
|
if identifier_like:
|
|
485
492
|
return "entity"
|
|
493
|
+
# Connection queries: detect "connect/bridge/link/between" patterns
|
|
494
|
+
if any(marker in normalized for marker in _CONNECTION_MARKERS):
|
|
495
|
+
return "connection"
|
|
486
496
|
if any(marker in question for marker in _TASK_MARKERS[:10]) or any(marker in normalized for marker in _TASK_MARKERS[10:]):
|
|
487
497
|
return "task"
|
|
488
498
|
if len(terms) >= 4:
|
|
@@ -715,6 +725,91 @@ def _compute_hub_labels(G: nx.Graph) -> set[str]:
|
|
|
715
725
|
return hub_labels
|
|
716
726
|
|
|
717
727
|
|
|
728
|
+
def _compute_graph_degree_stats(G: nx.Graph) -> dict[str, float]:
|
|
729
|
+
"""Compute degree statistics for adaptive threshold calculation.
|
|
730
|
+
|
|
731
|
+
Returns median_degree and mean_degree for use in adaptive penalties.
|
|
732
|
+
These replace hard-coded penalty values with graph-relative ones.
|
|
733
|
+
"""
|
|
734
|
+
if G.number_of_nodes() == 0:
|
|
735
|
+
return {"median_degree": 1.0, "mean_degree": 1.0}
|
|
736
|
+
degrees = [G.degree(n) for n in G.nodes]
|
|
737
|
+
sorted_degrees = sorted(degrees)
|
|
738
|
+
n = len(sorted_degrees)
|
|
739
|
+
median_degree = sorted_degrees[n // 2] if n % 2 == 1 else (sorted_degrees[n // 2 - 1] + sorted_degrees[n // 2]) / 2
|
|
740
|
+
mean_degree = sum(degrees) / n
|
|
741
|
+
return {"median_degree": max(median_degree, 1.0), "mean_degree": max(mean_degree, 1.0)}
|
|
742
|
+
|
|
743
|
+
|
|
744
|
+
def _adaptive_penalty_score(G: nx.Graph, node_id: str, base_penalty: float, degree_stats: dict[str, float]) -> float:
|
|
745
|
+
"""Compute adaptive penalty based on node degree relative to graph statistics.
|
|
746
|
+
|
|
747
|
+
Instead of hard-coded penalties (e.g., score -= 5.0), we scale penalties
|
|
748
|
+
by the node's degree relative to the median degree. High-degree nodes
|
|
749
|
+
get full penalties (they are hubs — if they don't match discriminative
|
|
750
|
+
terms, they are almost certainly irrelevant), while low-degree nodes
|
|
751
|
+
get reduced penalties (they may be specific but under-connected).
|
|
752
|
+
|
|
753
|
+
base_penalty: the maximum penalty to apply (e.g., 5.0 for non-discriminative)
|
|
754
|
+
"""
|
|
755
|
+
degree = G.degree(node_id) if node_id in G else 0
|
|
756
|
+
median_degree = degree_stats.get("median_degree", 1.0)
|
|
757
|
+
# Scale penalty: high-degree nodes get full penalty, low-degree get reduced
|
|
758
|
+
# degree >= median: penalty = base_penalty * 1.0
|
|
759
|
+
# degree << median: penalty = base_penalty * 0.5
|
|
760
|
+
scale = min(1.0, max(0.5, degree / median_degree)) if median_degree > 0 else 1.0
|
|
761
|
+
return base_penalty * scale
|
|
762
|
+
|
|
763
|
+
|
|
764
|
+
def _adaptive_threshold(scores: list[float], min_threshold: float = 1.0) -> float:
|
|
765
|
+
"""Find adaptive threshold using elbow detection.
|
|
766
|
+
|
|
767
|
+
Instead of hard-coded threshold = top_score * 0.65, we find the "elbow"
|
|
768
|
+
in the score distribution - the point where scores start dropping rapidly.
|
|
769
|
+
This naturally adapts to different query result distributions.
|
|
770
|
+
"""
|
|
771
|
+
if not scores:
|
|
772
|
+
return min_threshold
|
|
773
|
+
if len(scores) <= 2:
|
|
774
|
+
return max(scores[0] * 0.65, min_threshold)
|
|
775
|
+
|
|
776
|
+
# Sort scores descending
|
|
777
|
+
sorted_scores = sorted(scores, reverse=True)
|
|
778
|
+
|
|
779
|
+
# Find elbow: point with maximum second derivative (curvature)
|
|
780
|
+
# Simple approach: find where score drop exceeds average drop
|
|
781
|
+
drops = [sorted_scores[i] - sorted_scores[i + 1] for i in range(len(sorted_scores) - 1)]
|
|
782
|
+
if not drops:
|
|
783
|
+
return max(sorted_scores[0] * 0.65, min_threshold)
|
|
784
|
+
|
|
785
|
+
avg_drop = sum(drops) / len(drops)
|
|
786
|
+
|
|
787
|
+
# Find first index where drop exceeds 1.5x average
|
|
788
|
+
for i, drop in enumerate(drops):
|
|
789
|
+
if drop > avg_drop * 1.5 and i > 0:
|
|
790
|
+
# Threshold is the score at the elbow point (before the big drop)
|
|
791
|
+
return max(sorted_scores[i], min_threshold)
|
|
792
|
+
|
|
793
|
+
# No clear elbow found, use percentile-based threshold
|
|
794
|
+
return max(sorted_scores[len(sorted_scores) // 3], min_threshold)
|
|
795
|
+
|
|
796
|
+
|
|
797
|
+
def _adaptive_community_quota(top_k: int, num_communities_in_graph: int) -> int:
|
|
798
|
+
"""Compute adaptive community quota for seed diversity.
|
|
799
|
+
|
|
800
|
+
Instead of hard-coded community_quota = max(top_k // 2, 2), we scale
|
|
801
|
+
based on the number of communities in the graph. More communities
|
|
802
|
+
means we need more seeds per community to achieve coverage.
|
|
803
|
+
"""
|
|
804
|
+
if num_communities_in_graph <= 2:
|
|
805
|
+
return max(top_k // 2, 2)
|
|
806
|
+
# Scale quota: more communities → smaller quota per community
|
|
807
|
+
# but ensure at least 2 seeds per community for diversity
|
|
808
|
+
base_quota = max(top_k // 2, 2)
|
|
809
|
+
scale = min(1.0, max(0.5, 3.0 / num_communities_in_graph))
|
|
810
|
+
return max(int(base_quota * scale), 1)
|
|
811
|
+
|
|
812
|
+
|
|
718
813
|
def _parse_god_labels_from_report(report_text: str) -> set[str]:
|
|
719
814
|
"""Parse god node labels from GRAPH_REPORT.md text.
|
|
720
815
|
|
|
@@ -878,6 +973,124 @@ def _relation_family_weight(relation: str, confidence: str = "") -> float:
|
|
|
878
973
|
return base + confidence_bonus
|
|
879
974
|
|
|
880
975
|
|
|
976
|
+
def _connection_query(G: nx.Graph, question: str, terms: list[str], hub_labels: set[str] | None = None, god_labels: set[str] | None = None) -> dict[str, object]:
|
|
977
|
+
"""Handle connection/bridge queries: find paths between concept clusters.
|
|
978
|
+
|
|
979
|
+
Strategy:
|
|
980
|
+
1. Find seeds from different communities (cross-domain seeds)
|
|
981
|
+
2. Compute shortest paths between cross-community seed pairs
|
|
982
|
+
3. Identify bridge nodes (nodes appearing on multiple paths)
|
|
983
|
+
4. Return bridge concepts + paths + neighbor summaries
|
|
984
|
+
"""
|
|
985
|
+
# Find seeds with community diversity emphasis
|
|
986
|
+
seeds = _select_seed_nodes(G, question, terms, top_k=7, hub_labels=hub_labels, god_labels=god_labels)
|
|
987
|
+
if not seeds:
|
|
988
|
+
return {
|
|
989
|
+
"question": question,
|
|
990
|
+
"query_mode": "connection",
|
|
991
|
+
"terms": terms,
|
|
992
|
+
"bridges": [],
|
|
993
|
+
"paths": [],
|
|
994
|
+
"text": "No matching nodes found for connection query.",
|
|
995
|
+
}
|
|
996
|
+
|
|
997
|
+
# Group seeds by community
|
|
998
|
+
seed_by_community: dict[int, list[dict]] = {}
|
|
999
|
+
for seed in seeds:
|
|
1000
|
+
cid = G.nodes.get(seed["node_id"], {}).get("community")
|
|
1001
|
+
if cid is not None:
|
|
1002
|
+
seed_by_community.setdefault(cid, []).append(seed)
|
|
1003
|
+
|
|
1004
|
+
# Need at least 2 communities for a connection query
|
|
1005
|
+
if len(seed_by_community) < 2:
|
|
1006
|
+
# Fall back to regular task query
|
|
1007
|
+
return None # Signal to caller to use task mode
|
|
1008
|
+
|
|
1009
|
+
# Find cross-community seed pairs and compute shortest paths
|
|
1010
|
+
paths: list[dict] = []
|
|
1011
|
+
bridge_counter: dict[str, int] = {} # node_id -> count of paths it appears on
|
|
1012
|
+
community_pairs = []
|
|
1013
|
+
communities = list(seed_by_community.keys())
|
|
1014
|
+
for i in range(len(communities)):
|
|
1015
|
+
for j in range(i + 1, len(communities)):
|
|
1016
|
+
community_pairs.append((communities[i], communities[j]))
|
|
1017
|
+
|
|
1018
|
+
for cid_a, cid_b in community_pairs[:6]: # Limit to 6 pairs for performance
|
|
1019
|
+
seeds_a = seed_by_community[cid_a]
|
|
1020
|
+
seeds_b = seed_by_community[cid_b]
|
|
1021
|
+
# Try shortest path between best seed from each community
|
|
1022
|
+
for sa in seeds_a[:2]:
|
|
1023
|
+
for sb in seeds_b[:2]:
|
|
1024
|
+
try:
|
|
1025
|
+
path = nx.shortest_path(G, sa["node_id"], sb["node_id"])
|
|
1026
|
+
except (nx.NetworkXNoPath, nx.NodeNotFound):
|
|
1027
|
+
continue
|
|
1028
|
+
if len(path) < 2:
|
|
1029
|
+
continue
|
|
1030
|
+
# Record path
|
|
1031
|
+
path_labels = [G.nodes[n].get("label", n) for n in path]
|
|
1032
|
+
path_relations = []
|
|
1033
|
+
for k in range(len(path) - 1):
|
|
1034
|
+
edge = _edge_payload(G, path[k], path[k + 1])
|
|
1035
|
+
path_relations.append(edge.get("relation", ""))
|
|
1036
|
+
paths.append({
|
|
1037
|
+
"from": path_labels[0],
|
|
1038
|
+
"to": path_labels[-1],
|
|
1039
|
+
"from_community": cid_a,
|
|
1040
|
+
"to_community": cid_b,
|
|
1041
|
+
"path": path_labels,
|
|
1042
|
+
"relations": path_relations,
|
|
1043
|
+
"length": len(path) - 1,
|
|
1044
|
+
})
|
|
1045
|
+
# Count bridge nodes (intermediate nodes on paths)
|
|
1046
|
+
for node_id in path[1:-1]: # Exclude endpoints
|
|
1047
|
+
bridge_counter[node_id] = bridge_counter.get(node_id, 0) + 1
|
|
1048
|
+
|
|
1049
|
+
# Identify bridge nodes: appear on multiple paths
|
|
1050
|
+
bridges: list[dict] = []
|
|
1051
|
+
for node_id, count in sorted(bridge_counter.items(), key=lambda x: -x[1]):
|
|
1052
|
+
if count < 1:
|
|
1053
|
+
continue
|
|
1054
|
+
label = G.nodes[node_id].get("label", node_id)
|
|
1055
|
+
degree = G.degree(node_id)
|
|
1056
|
+
# Get neighbor summary
|
|
1057
|
+
neighbors = list(G.neighbors(node_id))
|
|
1058
|
+
neighbor_labels = [G.nodes[n].get("label", n) for n in neighbors[:8]]
|
|
1059
|
+
neighbor_communities = {G.nodes[n].get("community") for n in neighbors if G.nodes[n].get("community") is not None}
|
|
1060
|
+
bridges.append({
|
|
1061
|
+
"node_id": node_id,
|
|
1062
|
+
"label": label,
|
|
1063
|
+
"degree": degree,
|
|
1064
|
+
"path_count": count,
|
|
1065
|
+
"neighbor_count": len(neighbors),
|
|
1066
|
+
"neighbor_communities": len(neighbor_communities),
|
|
1067
|
+
"sample_neighbors": neighbor_labels,
|
|
1068
|
+
})
|
|
1069
|
+
|
|
1070
|
+
# Build text summary
|
|
1071
|
+
text_parts = [f"Connection query: {question}"]
|
|
1072
|
+
if bridges:
|
|
1073
|
+
text_parts.append(f"\nBridge concepts ({len(bridges)}):")
|
|
1074
|
+
for b in bridges[:5]:
|
|
1075
|
+
text_parts.append(f" - {b['label']} (appears on {b['path_count']} paths, connects {b['neighbor_communities']} communities)")
|
|
1076
|
+
if paths:
|
|
1077
|
+
text_parts.append(f"\nPaths found: {len(paths)}")
|
|
1078
|
+
for p in paths[:3]:
|
|
1079
|
+
text_parts.append(f" {p['from']} --[{', '.join(p['relations'])}]--> {p['to']} (length {p['length']})")
|
|
1080
|
+
if len(p['path']) > 2:
|
|
1081
|
+
text_parts.append(f" via: {' -> '.join(p['path'][1:-1])}")
|
|
1082
|
+
|
|
1083
|
+
return {
|
|
1084
|
+
"question": question,
|
|
1085
|
+
"query_mode": "connection",
|
|
1086
|
+
"terms": terms,
|
|
1087
|
+
"seeds": [{"node_id": s["node_id"], "label": s["label"], "seed_score": s["seed_score"]} for s in seeds],
|
|
1088
|
+
"bridges": bridges[:10],
|
|
1089
|
+
"paths": paths[:6],
|
|
1090
|
+
"text": "\n".join(text_parts),
|
|
1091
|
+
}
|
|
1092
|
+
|
|
1093
|
+
|
|
881
1094
|
def _select_seed_nodes(G: nx.Graph, question: str, terms: list[str], top_k: int = 5, hub_labels: set[str] | None = None, god_labels: set[str] | None = None) -> list[dict[str, object]]:
|
|
882
1095
|
scored_nodes = _score_nodes(G, terms)
|
|
883
1096
|
if not scored_nodes:
|
|
@@ -885,6 +1098,10 @@ def _select_seed_nodes(G: nx.Graph, question: str, terms: list[str], top_k: int
|
|
|
885
1098
|
base_scores = {nid: score for score, nid in scored_nodes}
|
|
886
1099
|
lexical_frontier = {nid for _, nid in scored_nodes[: max(top_k * 3, 12)]}
|
|
887
1100
|
compact_question = _normalize_query_text(question).replace(" ", "")
|
|
1101
|
+
# Compute graph degree statistics for adaptive penalties
|
|
1102
|
+
degree_stats = _compute_graph_degree_stats(G)
|
|
1103
|
+
# Count communities for adaptive quota
|
|
1104
|
+
all_communities = {G.nodes[n].get("community") for n in G.nodes if G.nodes[n].get("community") is not None}
|
|
888
1105
|
# Identify discriminative terms: those NOT matching generic/non-discriminative words.
|
|
889
1106
|
# Seeds that only match generic terms (e.g. "component") but miss
|
|
890
1107
|
# discriminative terms (e.g. "video") should be penalized.
|
|
@@ -933,26 +1150,29 @@ def _select_seed_nodes(G: nx.Graph, question: str, terms: list[str], top_k: int
|
|
|
933
1150
|
score += structural_support
|
|
934
1151
|
reasons.append("supported by nearby graph structure")
|
|
935
1152
|
if _is_noise_node(data, hub_labels):
|
|
936
|
-
score -= 3.0
|
|
1153
|
+
score -= _adaptive_penalty_score(G, nid, 3.0, degree_stats)
|
|
937
1154
|
# Enum/config types (ending in Mode, Flag, Type, etc.) make poor
|
|
938
1155
|
# navigation seeds — users ask about components, not enum values.
|
|
939
1156
|
if re.match(r".*(Mode|Flag|Type|Options|Config|Style|Format|Encoding|Space|Completed|Started|Stopped|Paused|Unknown)$", label):
|
|
940
|
-
score -= 2.0
|
|
1157
|
+
score -= _adaptive_penalty_score(G, nid, 2.0, degree_stats)
|
|
941
1158
|
# Penalize seeds that only match generic terms (e.g. "component")
|
|
942
1159
|
# but miss all discriminative terms (e.g. "video", "animation").
|
|
943
1160
|
if discriminative_compacts:
|
|
944
1161
|
label_spaced, label_compact = _normalized_label_forms(data.get("norm_label") or label)
|
|
945
1162
|
hits_discriminative = any(dt in label_compact or dt in label_spaced for dt in discriminative_compacts)
|
|
946
1163
|
if not hits_discriminative:
|
|
947
|
-
score -= 5.0
|
|
1164
|
+
score -= _adaptive_penalty_score(G, nid, 5.0, degree_stats)
|
|
948
1165
|
if score > 0:
|
|
949
1166
|
scored.append((score, nid, reasons))
|
|
950
1167
|
scored.sort(key=lambda item: item[0], reverse=True)
|
|
951
1168
|
if not scored:
|
|
952
1169
|
return []
|
|
953
|
-
threshold
|
|
1170
|
+
# Adaptive threshold: find elbow in score distribution instead of hard-coded 0.65
|
|
1171
|
+
all_scores = [s for s, _, _ in scored]
|
|
1172
|
+
threshold = _adaptive_threshold(all_scores)
|
|
954
1173
|
seeds: list[dict[str, object]] = []
|
|
955
|
-
|
|
1174
|
+
# Adaptive community quota: scale based on number of communities
|
|
1175
|
+
community_quota = _adaptive_community_quota(top_k, len(all_communities))
|
|
956
1176
|
community_count: dict[int, int] = {}
|
|
957
1177
|
for score, nid, reasons in scored:
|
|
958
1178
|
if len(seeds) >= top_k:
|
|
@@ -1652,6 +1872,13 @@ def _query_graph(
|
|
|
1652
1872
|
query_mode = _classify_query_mode(question, terms)
|
|
1653
1873
|
# Compute hub labels from graph structure (replaces hard-coded _GENERIC_LABELS)
|
|
1654
1874
|
hub_labels = _compute_hub_labels(G)
|
|
1875
|
+
if query_mode == "connection":
|
|
1876
|
+
result = _connection_query(G, question, terms, hub_labels=hub_labels, god_labels=god_labels)
|
|
1877
|
+
if result is not None:
|
|
1878
|
+
result["query_variants"] = query_texts
|
|
1879
|
+
return result
|
|
1880
|
+
# Fall back to task mode if not enough communities
|
|
1881
|
+
query_mode = "task"
|
|
1655
1882
|
if query_mode == "task":
|
|
1656
1883
|
seeds = _select_seed_nodes(G, question, terms, top_k=5, hub_labels=hub_labels, god_labels=god_labels)
|
|
1657
1884
|
if not seeds:
|
|
@@ -1707,7 +1934,9 @@ def _query_graph(
|
|
|
1707
1934
|
"text": "No matching nodes found.",
|
|
1708
1935
|
}
|
|
1709
1936
|
top_score = scored[0][0]
|
|
1710
|
-
threshold
|
|
1937
|
+
# Adaptive threshold: use elbow detection instead of hard-coded 0.5
|
|
1938
|
+
all_scores = [s for s, _ in scored]
|
|
1939
|
+
threshold = _adaptive_threshold(all_scores, min_threshold=0.5)
|
|
1711
1940
|
start_nodes = [nid for s, nid in scored[:5] if s >= threshold]
|
|
1712
1941
|
if not start_nodes:
|
|
1713
1942
|
return {
|
|
@@ -2170,7 +2399,7 @@ def serve(graph_path: str = "graphify-out/graph.json", workspace: str | None = N
|
|
|
2170
2399
|
return {int(k): v for k, v in json.loads(labels_path.read_text(encoding="utf-8")).items()}
|
|
2171
2400
|
except Exception:
|
|
2172
2401
|
pass
|
|
2173
|
-
return
|
|
2402
|
+
return label_communities(G, communities)
|
|
2174
2403
|
|
|
2175
2404
|
@server.list_resources()
|
|
2176
2405
|
async def list_resources() -> list[types.Resource]:
|
|
@@ -22,7 +22,7 @@ def _rebuild_code(watch_path: Path, *, follow_symlinks: bool = False) -> bool:
|
|
|
22
22
|
from graphify.extract import extract
|
|
23
23
|
from graphify.detect import detect
|
|
24
24
|
from graphify.build import build_from_json
|
|
25
|
-
from graphify.cluster import cluster, score_all
|
|
25
|
+
from graphify.cluster import cluster, score_all, label_communities
|
|
26
26
|
from graphify.analyze import god_nodes, surprising_connections, suggest_questions
|
|
27
27
|
from graphify.report import generate
|
|
28
28
|
from graphify.export import to_json, to_html
|
|
@@ -72,7 +72,7 @@ def _rebuild_code(watch_path: Path, *, follow_symlinks: bool = False) -> bool:
|
|
|
72
72
|
cohesion = score_all(G, communities)
|
|
73
73
|
gods = god_nodes(G)
|
|
74
74
|
surprises = surprising_connections(G, communities)
|
|
75
|
-
labels =
|
|
75
|
+
labels = label_communities(G, communities)
|
|
76
76
|
questions = suggest_questions(G, communities, labels)
|
|
77
77
|
|
|
78
78
|
out.mkdir(exist_ok=True)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: graphify_vault
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.4
|
|
4
4
|
Summary: Fork of github.com/safishamsi/graphify with GRAPHIFY_WORKSPACE path resolution for distributed graph artifacts, same-label node merging, incremental build fixes, and local knowledge vault capabilities
|
|
5
5
|
License: MIT License
|
|
6
6
|
|
|
@@ -56,6 +56,9 @@ Requires-Dist: tree-sitter-julia
|
|
|
56
56
|
Requires-Dist: tree-sitter-verilog
|
|
57
57
|
Provides-Extra: mcp
|
|
58
58
|
Requires-Dist: mcp; extra == "mcp"
|
|
59
|
+
Requires-Dist: deep-translator; extra == "mcp"
|
|
60
|
+
Requires-Dist: pypdf; extra == "mcp"
|
|
61
|
+
Requires-Dist: html2text; extra == "mcp"
|
|
59
62
|
Provides-Extra: neo4j
|
|
60
63
|
Requires-Dist: neo4j; extra == "neo4j"
|
|
61
64
|
Provides-Extra: pdf
|
|
@@ -75,6 +78,7 @@ Requires-Dist: faster-whisper; extra == "video"
|
|
|
75
78
|
Requires-Dist: yt-dlp; extra == "video"
|
|
76
79
|
Provides-Extra: all
|
|
77
80
|
Requires-Dist: mcp; extra == "all"
|
|
81
|
+
Requires-Dist: deep-translator; extra == "all"
|
|
78
82
|
Requires-Dist: neo4j; extra == "all"
|
|
79
83
|
Requires-Dist: pypdf; extra == "all"
|
|
80
84
|
Requires-Dist: html2text; extra == "all"
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "graphify_vault"
|
|
7
|
-
version = "0.3.
|
|
7
|
+
version = "0.3.4"
|
|
8
8
|
description = "Fork of github.com/safishamsi/graphify with GRAPHIFY_WORKSPACE path resolution for distributed graph artifacts, same-label node merging, incremental build fixes, and local knowledge vault capabilities"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = { file = "LICENSE" }
|
|
@@ -42,7 +42,7 @@ Repository = "https://github.com/gengwb/graphify"
|
|
|
42
42
|
Issues = "https://github.com/gengwb/graphify/issues"
|
|
43
43
|
|
|
44
44
|
[project.optional-dependencies]
|
|
45
|
-
mcp = ["mcp"]
|
|
45
|
+
mcp = ["mcp", "deep-translator", "pypdf", "html2text"]
|
|
46
46
|
neo4j = ["neo4j"]
|
|
47
47
|
pdf = ["pypdf", "html2text"]
|
|
48
48
|
watch = ["watchdog"]
|
|
@@ -50,10 +50,11 @@ svg = ["matplotlib"]
|
|
|
50
50
|
leiden = ["graspologic; python_version < '3.13'"]
|
|
51
51
|
office = ["python-docx", "openpyxl"]
|
|
52
52
|
video = ["faster-whisper", "yt-dlp"]
|
|
53
|
-
all = ["mcp", "neo4j", "pypdf", "html2text", "watchdog", "graspologic; python_version < '3.13'", "python-docx", "openpyxl", "faster-whisper", "yt-dlp", "matplotlib"]
|
|
53
|
+
all = ["mcp", "deep-translator", "neo4j", "pypdf", "html2text", "watchdog", "graspologic; python_version < '3.13'", "python-docx", "openpyxl", "faster-whisper", "yt-dlp", "matplotlib"]
|
|
54
54
|
|
|
55
55
|
[project.scripts]
|
|
56
56
|
graphify = "graphify.__main__:main"
|
|
57
|
+
gv = "graphify.__main__:main"
|
|
57
58
|
|
|
58
59
|
[tool.setuptools.packages.find]
|
|
59
60
|
where = ["."]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|