graphify-vault 0.3.2__tar.gz → 0.3.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. {graphify_vault-0.3.2/graphify_vault.egg-info → graphify_vault-0.3.4}/PKG-INFO +5 -1
  2. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/__main__.py +2 -2
  3. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/cluster.py +110 -1
  4. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/doctor.py +7 -0
  5. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/extract.py +310 -5
  6. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/serve.py +236 -7
  7. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/watch.py +2 -2
  8. {graphify_vault-0.3.2 → graphify_vault-0.3.4/graphify_vault.egg-info}/PKG-INFO +5 -1
  9. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify_vault.egg-info/entry_points.txt +1 -0
  10. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify_vault.egg-info/requires.txt +4 -0
  11. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/pyproject.toml +4 -3
  12. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/LICENSE +0 -0
  13. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/README.md +0 -0
  14. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/__init__.py +0 -0
  15. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/analyze.py +0 -0
  16. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/benchmark.py +0 -0
  17. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/build.py +0 -0
  18. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/cache.py +0 -0
  19. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/detect.py +0 -0
  20. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/export.py +0 -0
  21. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/hooks.py +0 -0
  22. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/ingest.py +0 -0
  23. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/manifest.py +0 -0
  24. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/report.py +0 -0
  25. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/security.py +0 -0
  26. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/skill-aider.md +0 -0
  27. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/skill-claw.md +0 -0
  28. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/skill-codex.md +0 -0
  29. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/skill-copilot.md +0 -0
  30. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/skill-droid.md +0 -0
  31. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/skill-kiro.md +0 -0
  32. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/skill-opencode.md +0 -0
  33. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/skill-trae.md +0 -0
  34. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/skill-vscode.md +0 -0
  35. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/skill-windows.md +0 -0
  36. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/skill.md +0 -0
  37. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/transcribe.py +0 -0
  38. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/validate.py +0 -0
  39. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify/wiki.py +0 -0
  40. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify_vault.egg-info/SOURCES.txt +0 -0
  41. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify_vault.egg-info/dependency_links.txt +0 -0
  42. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/graphify_vault.egg-info/top_level.txt +0 -0
  43. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/setup.cfg +0 -0
  44. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_analyze.py +0 -0
  45. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_benchmark.py +0 -0
  46. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_build.py +0 -0
  47. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_cache.py +0 -0
  48. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_claude_md.py +0 -0
  49. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_cluster.py +0 -0
  50. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_confidence.py +0 -0
  51. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_detect.py +0 -0
  52. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_export.py +0 -0
  53. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_extract.py +0 -0
  54. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_graph_build_perf.py +0 -0
  55. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_hooks.py +0 -0
  56. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_hypergraph.py +0 -0
  57. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_ingest.py +0 -0
  58. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_install.py +0 -0
  59. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_languages.py +0 -0
  60. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_multilang.py +0 -0
  61. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_pipeline.py +0 -0
  62. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_rationale.py +0 -0
  63. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_report.py +0 -0
  64. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_security.py +0 -0
  65. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_semantic_similarity.py +0 -0
  66. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_serve.py +0 -0
  67. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_sort_by.py +0 -0
  68. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_spatial_craft_build.py +0 -0
  69. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_spatialsdk_query_regression.py +0 -0
  70. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_transcribe.py +0 -0
  71. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_validate.py +0 -0
  72. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_watch.py +0 -0
  73. {graphify_vault-0.3.2 → graphify_vault-0.3.4}/tests/test_wiki.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: graphify_vault
3
- Version: 0.3.2
3
+ Version: 0.3.4
4
4
  Summary: Fork of github.com/safishamsi/graphify with GRAPHIFY_WORKSPACE path resolution for distributed graph artifacts, same-label node merging, incremental build fixes, and local knowledge vault capabilities
5
5
  License: MIT License
6
6
 
@@ -56,6 +56,9 @@ Requires-Dist: tree-sitter-julia
56
56
  Requires-Dist: tree-sitter-verilog
57
57
  Provides-Extra: mcp
58
58
  Requires-Dist: mcp; extra == "mcp"
59
+ Requires-Dist: deep-translator; extra == "mcp"
60
+ Requires-Dist: pypdf; extra == "mcp"
61
+ Requires-Dist: html2text; extra == "mcp"
59
62
  Provides-Extra: neo4j
60
63
  Requires-Dist: neo4j; extra == "neo4j"
61
64
  Provides-Extra: pdf
@@ -75,6 +78,7 @@ Requires-Dist: faster-whisper; extra == "video"
75
78
  Requires-Dist: yt-dlp; extra == "video"
76
79
  Provides-Extra: all
77
80
  Requires-Dist: mcp; extra == "all"
81
+ Requires-Dist: deep-translator; extra == "all"
78
82
  Requires-Dist: neo4j; extra == "all"
79
83
  Requires-Dist: pypdf; extra == "all"
80
84
  Requires-Dist: html2text; extra == "all"
@@ -1422,7 +1422,7 @@ def main() -> None:
1422
1422
  sys.exit(1)
1423
1423
  from networkx.readwrite import json_graph as _jg
1424
1424
  from graphify.build import build_from_json
1425
- from graphify.cluster import cluster, score_all
1425
+ from graphify.cluster import cluster, score_all, label_communities
1426
1426
  from graphify.analyze import god_nodes, surprising_connections, suggest_questions
1427
1427
  from graphify.report import generate
1428
1428
  from graphify.export import to_json, to_html
@@ -1435,7 +1435,7 @@ def main() -> None:
1435
1435
  cohesion = score_all(G, communities)
1436
1436
  gods = god_nodes(G)
1437
1437
  surprises = surprising_connections(G, communities)
1438
- labels = {cid: f"Community {cid}" for cid in communities}
1438
+ labels = label_communities(G, communities)
1439
1439
  questions = suggest_questions(G, communities, labels)
1440
1440
  tokens = {"input": 0, "output": 0}
1441
1441
  report = generate(G, communities, cohesion, labels, gods, surprises,
@@ -3,6 +3,7 @@ from __future__ import annotations
3
3
  import contextlib
4
4
  import inspect
5
5
  import io
6
+ import re
6
7
  import sys
7
8
  import networkx as nx
8
9
 
@@ -18,6 +19,87 @@ def _suppress_output():
18
19
  return contextlib.redirect_stdout(io.StringIO())
19
20
 
20
21
 
22
+ def _label_token_overlap(label_a: str, label_b: str) -> float:
23
+ """Compute token overlap ratio between two labels.
24
+
25
+ Splits labels into lowercase tokens using both non-alphanumeric
26
+ boundaries AND camelCase/PascalCase boundaries. Returns the
27
+ Jaccard similarity of the token sets. This provides a lightweight
28
+ semantic similarity signal without requiring any external NLP models.
29
+ """
30
+ def _tokens(text: str) -> set[str]:
31
+ # Split on non-alphanumeric boundaries
32
+ parts = re.split(r'[^a-zA-Z0-9]+', text)
33
+ # Also split camelCase/PascalCase: "VideoComponent" -> "Video", "Component"
34
+ expanded = []
35
+ for part in parts:
36
+ # Split on camelCase boundaries: lowercase followed by uppercase
37
+ sub_parts = re.sub(r'([a-z])([A-Z])', r'\1 \2', part).split()
38
+ # Split on acronyms: uppercase followed by uppercase+lowercase (e.g., "XMLParser" -> "XML", "Parser")
39
+ sub_parts = [re.sub(r'([A-Z]+)([A-Z][a-z])', r'\1 \2', s).split() for s in sub_parts]
40
+ expanded.extend([t for group in sub_parts for t in group])
41
+ return {t.lower() for t in expanded if len(t) > 2}
42
+ ta = _tokens(label_a)
43
+ tb = _tokens(label_b)
44
+ if not ta or not tb:
45
+ return 0.0
46
+ intersection = ta & tb
47
+ union = ta | tb
48
+ return len(intersection) / len(union) if union else 0.0
49
+
50
+
51
+ def _add_semantic_edges(G: nx.Graph, weight: float = 0.3) -> nx.Graph:
52
+ """Add weak semantic edges between nodes with overlapping labels.
53
+
54
+ For each pair of nodes in the same connected component that share
55
+ label tokens (Jaccard > 0.3), adds a 'semantically_similar_to' edge
56
+ with the given weight. These edges guide Leiden/Louvain to group
57
+ semantically related nodes together even without structural edges.
58
+
59
+ This is a non-destructive overlay: the original graph is copied and
60
+ only new edges are added (no existing edges are modified).
61
+ """
62
+ if G.number_of_nodes() < 2:
63
+ return G.copy()
64
+
65
+ # Only consider nodes with meaningful labels
66
+ candidates = []
67
+ for nid in G.nodes:
68
+ label = str(G.nodes[nid].get("label", ""))
69
+ if label and len(label) > 2:
70
+ candidates.append((nid, label))
71
+
72
+ if len(candidates) < 2:
73
+ return G.copy()
74
+
75
+ G_augmented = G.copy()
76
+ added = 0
77
+
78
+ # For efficiency, only check pairs within the same connected component
79
+ # and limit to reasonable graph sizes
80
+ if len(candidates) > 5000:
81
+ # For very large graphs, sample pairs from same-neighborhood
82
+ return G_augmented
83
+
84
+ for i in range(len(candidates)):
85
+ nid_a, label_a = candidates[i]
86
+ for j in range(i + 1, len(candidates)):
87
+ nid_b, label_b = candidates[j]
88
+ # Skip if already connected by an edge
89
+ if G_augmented.has_edge(nid_a, nid_b):
90
+ continue
91
+ similarity = _label_token_overlap(label_a, label_b)
92
+ if similarity > 0.3:
93
+ G_augmented.add_edge(nid_a, nid_b,
94
+ relation="semantically_similar_to",
95
+ confidence="INFERRED",
96
+ confidence_score=round(similarity, 2),
97
+ weight=weight * similarity)
98
+ added += 1
99
+
100
+ return G_augmented
101
+
102
+
21
103
  def _partition(G: nx.Graph) -> dict[str, int]:
22
104
  """Run community detection. Returns {node_id: community_id}.
23
105
 
@@ -56,7 +138,7 @@ _MAX_COMMUNITY_FRACTION = 0.25 # communities larger than 25% of graph get spli
56
138
  _MIN_SPLIT_SIZE = 10 # only split if community has at least this many nodes
57
139
 
58
140
 
59
- def cluster(G: nx.Graph) -> dict[int, list[str]]:
141
+ def cluster(G: nx.Graph, *, semantic: bool = True) -> dict[int, list[str]]:
60
142
  """Run Leiden community detection. Returns {community_id: [node_ids]}.
61
143
 
62
144
  Community IDs are stable across runs: 0 = largest community after splitting.
@@ -65,6 +147,10 @@ def cluster(G: nx.Graph) -> dict[int, list[str]]:
65
147
 
66
148
  Accepts directed or undirected graphs. DiGraphs are converted to undirected
67
149
  internally since Louvain/Leiden require undirected input.
150
+
151
+ If semantic=True (default), weak semantic edges are added between nodes
152
+ with overlapping labels before clustering, helping semantically related
153
+ nodes group together even without structural edges.
68
154
  """
69
155
  if G.number_of_nodes() == 0:
70
156
  return {}
@@ -73,6 +159,10 @@ def cluster(G: nx.Graph) -> dict[int, list[str]]:
73
159
  if G.number_of_edges() == 0:
74
160
  return {i: [n] for i, n in enumerate(sorted(G.nodes))}
75
161
 
162
+ # Optionally augment graph with semantic edges for better community detection
163
+ if semantic:
164
+ G = _add_semantic_edges(G)
165
+
76
166
  # Leiden warns and drops isolates - handle them separately
77
167
  isolates = [n for n in G.nodes() if G.degree(n) == 0]
78
168
  connected_nodes = [n for n in G.nodes() if G.degree(n) > 0]
@@ -133,5 +223,24 @@ def cohesion_score(G: nx.Graph, community_nodes: list[str]) -> float:
133
223
  return round(actual / possible, 2) if possible > 0 else 0.0
134
224
 
135
225
 
226
+ def label_communities(G: nx.Graph, communities: dict[int, list[str]]) -> dict[int, str]:
227
+ """Generate semantic labels for communities based on highest-degree node.
228
+
229
+ For each community, the node with the highest degree (most connections)
230
+ is chosen as the representative label. This is a simple, universal heuristic
231
+ that works for any graph without domain-specific knowledge.
232
+ """
233
+ labels: dict[int, str] = {}
234
+ for cid, node_ids in communities.items():
235
+ if not node_ids:
236
+ labels[cid] = f"Community {cid}"
237
+ continue
238
+ # Find the node with highest degree in this community
239
+ best_node = max(node_ids, key=lambda n: G.degree(n) if n in G else 0)
240
+ label = G.nodes[best_node].get("label", best_node) if best_node in G else best_node
241
+ labels[cid] = label
242
+ return labels
243
+
244
+
136
245
  def score_all(G: nx.Graph, communities: dict[int, list[str]]) -> dict[int, float]:
137
246
  return {cid: cohesion_score(G, nodes) for cid, nodes in communities.items()}
@@ -66,6 +66,13 @@ def _check_graphify_installed() -> bool:
66
66
  )
67
67
  if result.returncode == 0:
68
68
  _ok(f"graphify {__version__}")
69
+ # Check for conflicting graphifyy installation
70
+ try:
71
+ from importlib.metadata import version as _v, PackageNotFoundError
72
+ _v("graphifyy")
73
+ _warn("Both graphify_vault and graphifyy detected. 'graphify' command may point to graphifyy. Use 'gv' command instead, or uninstall graphifyy.")
74
+ except PackageNotFoundError:
75
+ pass
69
76
  return True
70
77
  except Exception:
71
78
  pass
@@ -231,7 +231,7 @@ def extract_document(path: Path, root: Path | None = None) -> dict:
231
231
  return {"nodes": [], "edges": []}
232
232
 
233
233
  definitions: list[dict[str, Any]] = []
234
- headings: list[tuple[str, int]] = []
234
+ headings: list[tuple[str, int, int]] = [] # (heading, lineno, level)
235
235
  seen_labels: set[str] = set()
236
236
 
237
237
  # Pre-strip lines that are raw HTML (SVG, embedded widgets, etc.)
@@ -246,8 +246,9 @@ def extract_document(path: Path, root: Path | None = None) -> dict:
246
246
  m = re.match(r'^(#{1,6})\s+(.+)', line)
247
247
  if m:
248
248
  heading = m.group(2).strip()
249
+ level = len(m.group(1))
249
250
  if _looks_like_symbol_heading(heading):
250
- headings.append((heading, lineno))
251
+ headings.append((heading, lineno, level))
251
252
 
252
253
  match = _DOC_DEF_RE.search(line)
253
254
  if not match:
@@ -270,7 +271,7 @@ def extract_document(path: Path, root: Path | None = None) -> dict:
270
271
  definitions.append({"label": label, "lineno": lineno, "bases": bases})
271
272
  seen_labels.add(label)
272
273
 
273
- for heading, lineno in headings:
274
+ for heading, lineno, _level in headings:
274
275
  if heading not in seen_labels:
275
276
  definitions.append({"label": heading, "lineno": lineno, "bases": []})
276
277
  seen_labels.add(heading)
@@ -297,7 +298,7 @@ def extract_document(path: Path, root: Path | None = None) -> dict:
297
298
  seen_labels.add(name)
298
299
 
299
300
  heading_positions: dict[str, list[int]] = {}
300
- for heading, lineno in headings:
301
+ for heading, lineno, _level in headings:
301
302
  heading_positions.setdefault(heading, []).append(lineno)
302
303
 
303
304
  lines = text.splitlines()
@@ -316,7 +317,7 @@ def extract_document(path: Path, root: Path | None = None) -> dict:
316
317
  for idx, item in enumerate(definitions):
317
318
  next_heading_lineno = None
318
319
  current_lineno = int(item["lineno"])
319
- for heading, heading_lineno in headings:
320
+ for heading, heading_lineno, _level in headings:
320
321
  if heading_lineno > current_lineno:
321
322
  next_heading_lineno = heading_lineno
322
323
  break
@@ -468,6 +469,289 @@ def _resolve_document_cross_references(
468
469
  return new_edges
469
470
 
470
471
 
472
+ def _add_heading_structure_edges(
473
+ doc_results: list[dict], doc_paths: list[Path], root: Path | None = None,
474
+ ) -> list[dict]:
475
+ """Add ``contains`` edges based on heading hierarchy within documents.
476
+
477
+ For each document, re-parses headings to build a parent-child tree:
478
+ an H2 heading "contains" all H3 headings under it until the next H2,
479
+ and so on. Also creates ``contains`` edges from heading nodes to
480
+ concept nodes (definitions) that fall under that heading section.
481
+ """
482
+ _HEADING_RE = re.compile(r'^(#{1,6})\s+(.+)', re.MULTILINE)
483
+
484
+ new_edges: list[dict] = []
485
+ seen_pairs: set[tuple[str, str]] = set()
486
+
487
+ for result, path in zip(doc_results, doc_paths):
488
+ relpath = _doc_relpath(path, root)
489
+ stem = _doc_file_stem(path)
490
+ nodes = result.get("nodes", [])
491
+ if not nodes:
492
+ continue
493
+
494
+ try:
495
+ text = path.read_text(encoding="utf-8", errors="replace")
496
+ except OSError:
497
+ continue
498
+
499
+ # Parse all headings with levels
500
+ all_headings: list[tuple[str, int, int]] = [] # (heading, lineno, level)
501
+ for m in _HEADING_RE.finditer(text):
502
+ heading = m.group(2).strip()
503
+ level = len(m.group(1))
504
+ # Calculate line number from match position
505
+ lineno = text[:m.start()].count("\n") + 1
506
+ all_headings.append((heading, lineno, level))
507
+
508
+ if not all_headings:
509
+ continue
510
+
511
+ # Build heading node ID map: heading label → nid
512
+ heading_nid_map: dict[str, str] = {}
513
+ for n in nodes:
514
+ heading_nid_map[n.get("label", "")] = n.get("id", "")
515
+
516
+ # Build parent-child heading relationships using a stack
517
+ # Stack contains (heading, lineno, level, nid)
518
+ stack: list[tuple[str, int, int, str]] = []
519
+ heading_contains: list[tuple[str, str]] = [] # (parent_nid, child_nid)
520
+
521
+ for heading, lineno, level in all_headings:
522
+ nid = heading_nid_map.get(heading, _make_id(stem, heading))
523
+ # Pop stack until we find a parent (lower level number = higher rank)
524
+ while stack and stack[-1][2] >= level:
525
+ stack.pop()
526
+ if stack:
527
+ parent_nid = stack[-1][3]
528
+ heading_contains.append((parent_nid, nid))
529
+ stack.append((heading, lineno, level, nid))
530
+
531
+ # Create contains edges for heading hierarchy
532
+ for parent_nid, child_nid in heading_contains:
533
+ if parent_nid == child_nid:
534
+ continue
535
+ pair = (parent_nid, child_nid)
536
+ if pair in seen_pairs:
537
+ continue
538
+ seen_pairs.add(pair)
539
+ new_edges.append({
540
+ "source": parent_nid,
541
+ "target": child_nid,
542
+ "relation": "contains",
543
+ "confidence": "EXTRACTED",
544
+ "confidence_score": 1.0,
545
+ "source_file": relpath,
546
+ "weight": 1.0,
547
+ })
548
+
549
+ # Create contains edges from heading to concept nodes in that section
550
+ # A concept node belongs to the nearest preceding heading
551
+ sorted_headings = sorted(all_headings, key=lambda h: h[1])
552
+ for n in nodes:
553
+ nid = n.get("id", "")
554
+ label = n.get("label", "")
555
+ start_line = n.get("start_line")
556
+ if not isinstance(start_line, int):
557
+ continue
558
+ # Find the nearest preceding heading
559
+ parent_heading_nid = None
560
+ for heading, h_lineno, h_level in reversed(sorted_headings):
561
+ if h_lineno <= start_line:
562
+ parent_heading_nid = heading_nid_map.get(heading, _make_id(stem, heading))
563
+ break
564
+ if parent_heading_nid and parent_heading_nid != nid:
565
+ pair = (parent_heading_nid, nid)
566
+ if pair not in seen_pairs:
567
+ seen_pairs.add(pair)
568
+ new_edges.append({
569
+ "source": parent_heading_nid,
570
+ "target": nid,
571
+ "relation": "contains",
572
+ "confidence": "EXTRACTED",
573
+ "confidence_score": 0.9,
574
+ "source_file": relpath,
575
+ "weight": 0.9,
576
+ })
577
+
578
+ return new_edges
579
+
580
+
581
+ def _add_co_occurrence_edges(
582
+ doc_results: list[dict], doc_paths: list[Path], root: Path | None = None,
583
+ ) -> list[dict]:
584
+ """Add ``co_occurs_in`` edges between concept nodes in the same document.
585
+
586
+ Nodes appearing in the same paragraph get a co-occurrence edge with
587
+ weight=1.0. Weight decays with paragraph distance:
588
+ weight = 1.0 / (1 + abs(paragraph_distance)).
589
+ Only creates edges between nodes from the same document.
590
+ """
591
+ new_edges: list[dict] = []
592
+ seen_pairs: set[tuple[str, str]] = set()
593
+
594
+ for result, path in zip(doc_results, doc_paths):
595
+ relpath = _doc_relpath(path, root)
596
+ nodes = result.get("nodes", [])
597
+ if len(nodes) < 2:
598
+ continue
599
+
600
+ # Compute paragraph index for each node based on start_line
601
+ # Paragraphs are separated by blank lines
602
+ try:
603
+ text = path.read_text(encoding="utf-8", errors="replace")
604
+ except OSError:
605
+ continue
606
+ lines = text.splitlines()
607
+
608
+ # Build line → paragraph_number map
609
+ para_map: dict[int, int] = {}
610
+ para_idx = 0
611
+ for i, line in enumerate(lines, start=1):
612
+ if i == 1 or (line.strip() == "" and i > 1):
613
+ # New paragraph starts after blank line (or at line 1)
614
+ if i > 1:
615
+ para_idx += 1
616
+ para_map[i] = para_idx
617
+
618
+ # Get paragraph index for each node
619
+ node_paras: list[tuple[str, int]] = [] # (nid, paragraph)
620
+ for n in nodes:
621
+ nid = n.get("id", "")
622
+ start_line = n.get("start_line") or n.get("source_location", "")
623
+ # Parse line number
624
+ if isinstance(start_line, int):
625
+ lineno = start_line
626
+ elif isinstance(start_line, str) and start_line.startswith("L"):
627
+ try:
628
+ lineno = int(start_line[1:])
629
+ except ValueError:
630
+ continue
631
+ else:
632
+ continue
633
+ para = para_map.get(lineno, 0)
634
+ node_paras.append((nid, para))
635
+
636
+ # Create co-occurrence edges between all pairs in the same document
637
+ for i in range(len(node_paras)):
638
+ nid_i, para_i = node_paras[i]
639
+ for j in range(i + 1, len(node_paras)):
640
+ nid_j, para_j = node_paras[j]
641
+ if nid_i == nid_j:
642
+ continue
643
+ # Sort pair to avoid duplicates
644
+ pair = tuple(sorted([nid_i, nid_j]))
645
+ if pair in seen_pairs:
646
+ continue
647
+ dist = abs(para_i - para_j)
648
+ # Only create edges within a reasonable distance (same doc, ≤5 paragraphs)
649
+ if dist > 5:
650
+ continue
651
+ weight = round(1.0 / (1 + dist), 2)
652
+ seen_pairs.add(pair)
653
+ new_edges.append({
654
+ "source": pair[0],
655
+ "target": pair[1],
656
+ "relation": "co_occurs_in",
657
+ "confidence": "INFERRED",
658
+ "confidence_score": weight,
659
+ "source_file": relpath,
660
+ "weight": weight,
661
+ })
662
+
663
+ return new_edges
664
+
665
+
666
+ def _resolve_markdown_link_references(
667
+ doc_results: list[dict], doc_paths: list[Path], root: Path | None = None,
668
+ ) -> list[dict]:
669
+ """Resolve Markdown ``[text](url)`` links into ``references`` edges.
670
+
671
+ Scans all document nodes' raws for Markdown links. When the URL
672
+ resolves to another document in the corpus, creates a ``references``
673
+ edge from the source concept node to the target document's file-level
674
+ node (or the best matching concept node within the target document).
675
+ """
676
+ _MD_LINK = re.compile(r'\[([^\]]+)\]\(([^)]+)\)')
677
+
678
+ # Build file-level node ID map: relpath → nid
679
+ file_nid_map: dict[str, str] = {}
680
+ # Build label → nid map for intra-document concept matching
681
+ label_to_nid: dict[str, str] = {}
682
+ for result, path in zip(doc_results, doc_paths):
683
+ relpath = _doc_relpath(path, root)
684
+ stem = _doc_file_stem(path)
685
+ file_nid = _make_id(stem)
686
+ file_nid_map[relpath] = file_nid
687
+ # Also map by stem name for partial matches
688
+ file_nid_map[path.name] = file_nid
689
+ for n in result.get("nodes", []):
690
+ lbl = n.get("label", "")
691
+ nid = n.get("id", "")
692
+ if lbl and nid:
693
+ label_to_nid.setdefault(lbl, nid)
694
+
695
+ # Build a set of known document relative paths for fast lookup
696
+ known_relpaths = set(file_nid_map.keys())
697
+
698
+ new_edges: list[dict] = []
699
+ seen_pairs: set[tuple[str, str]] = set()
700
+
701
+ for result, path in zip(doc_results, doc_paths):
702
+ relpath = _doc_relpath(path, root)
703
+ for n in result.get("nodes", []):
704
+ nid = n.get("id", "")
705
+ raws_blocks = n.get("raws", [])
706
+ if not raws_blocks:
707
+ continue
708
+ raws_text = "\n".join(raws_blocks) if isinstance(raws_blocks, list) else str(raws_blocks)
709
+
710
+ for m in _MD_LINK.finditer(raws_text):
711
+ link_text = m.group(1).strip()
712
+ link_url = m.group(2).strip()
713
+
714
+ # Skip external URLs, anchors, and image links
715
+ if link_url.startswith(("http://", "https://", "#", "mailto:", "data:")):
716
+ continue
717
+ # Strip anchor fragment
718
+ link_url = link_url.split("#")[0].split("?")[0]
719
+ if not link_url:
720
+ continue
721
+
722
+ # Try to resolve the URL to a known document
723
+ target_nid = None
724
+ # Direct relative path match
725
+ if link_url in known_relpaths:
726
+ target_nid = file_nid_map.get(link_url)
727
+ else:
728
+ # Try matching by filename
729
+ target_name = Path(link_url).name
730
+ if target_name in file_nid_map:
731
+ target_nid = file_nid_map[target_name]
732
+
733
+ # If no file-level match, try matching link text to a concept label
734
+ if target_nid is None and link_text in label_to_nid:
735
+ target_nid = label_to_nid[link_text]
736
+
737
+ if target_nid is None or target_nid == nid:
738
+ continue
739
+ if (nid, target_nid) in seen_pairs:
740
+ continue
741
+ seen_pairs.add((nid, target_nid))
742
+ new_edges.append({
743
+ "source": nid,
744
+ "target": target_nid,
745
+ "relation": "references",
746
+ "confidence": "EXTRACTED",
747
+ "confidence_score": 1.0,
748
+ "source_file": relpath,
749
+ "weight": 1.0,
750
+ })
751
+
752
+ return new_edges
753
+
754
+
471
755
  def merge_document_semantic_extracts(doc_extract: dict[str, Any], semantic_extract: dict[str, Any]) -> dict[str, Any]:
472
756
  """Merge deterministic doc symbols with semantic extraction.
473
757
 
@@ -3888,6 +4172,27 @@ def extract(paths: list[Path], cache_root: Path | None = None) -> dict:
3888
4172
  import logging
3889
4173
  logging.getLogger(__name__).warning("Document cross-reference resolution failed, skipping: %s", exc)
3890
4174
 
4175
+ try:
4176
+ md_link_edges = _resolve_markdown_link_references(doc_results, doc_paths, root)
4177
+ all_edges.extend(md_link_edges)
4178
+ except Exception as exc:
4179
+ import logging
4180
+ logging.getLogger(__name__).warning("Markdown link reference resolution failed, skipping: %s", exc)
4181
+
4182
+ try:
4183
+ co_occurs_edges = _add_co_occurrence_edges(doc_results, doc_paths, root)
4184
+ all_edges.extend(co_occurs_edges)
4185
+ except Exception as exc:
4186
+ import logging
4187
+ logging.getLogger(__name__).warning("Co-occurrence edge creation failed, skipping: %s", exc)
4188
+
4189
+ try:
4190
+ heading_edges = _add_heading_structure_edges(doc_results, doc_paths, root)
4191
+ all_edges.extend(heading_edges)
4192
+ except Exception as exc:
4193
+ import logging
4194
+ logging.getLogger(__name__).warning("Heading structure edge creation failed, skipping: %s", exc)
4195
+
3891
4196
  return {
3892
4197
  "nodes": all_nodes,
3893
4198
  "edges": all_edges,
@@ -11,6 +11,7 @@ from pathlib import Path
11
11
  import networkx as nx
12
12
  from networkx.readwrite import json_graph
13
13
  from graphify.security import sanitize_label
14
+ from graphify.cluster import label_communities
14
15
 
15
16
 
16
17
  def _resolve_graph_path(graph_path: str) -> str:
@@ -105,6 +106,12 @@ _TASK_MARKERS = (
105
106
  "relevant", "entry", "workflow", "flow", "path", "components", "component", "types",
106
107
  )
107
108
 
109
+ _CONNECTION_MARKERS = (
110
+ "连接", "桥接", "关联", "中间", "之间", "关系",
111
+ "connect", "bridge", "link", "between", "relation", "intermediate",
112
+ "how does", "how do", "what connects", "what links", "what relates",
113
+ )
114
+
108
115
  _STOP_TERMS = {
109
116
  "a", "an", "the", "and", "or", "but", "if", "then", "how", "what", "which", "where", "when",
110
117
  "why", "who", "whom", "whose", "should", "would", "could", "can", "do", "does", "did", "done",
@@ -483,6 +490,9 @@ def _classify_query_mode(question: str, terms: list[str] | None = None) -> str:
483
490
  identifier_like = bool(re.fullmatch(r"[A-Za-z_][A-Za-z0-9_\.]*", stripped))
484
491
  if identifier_like:
485
492
  return "entity"
493
+ # Connection queries: detect "connect/bridge/link/between" patterns
494
+ if any(marker in normalized for marker in _CONNECTION_MARKERS):
495
+ return "connection"
486
496
  if any(marker in question for marker in _TASK_MARKERS[:10]) or any(marker in normalized for marker in _TASK_MARKERS[10:]):
487
497
  return "task"
488
498
  if len(terms) >= 4:
@@ -715,6 +725,91 @@ def _compute_hub_labels(G: nx.Graph) -> set[str]:
715
725
  return hub_labels
716
726
 
717
727
 
728
+ def _compute_graph_degree_stats(G: nx.Graph) -> dict[str, float]:
729
+ """Compute degree statistics for adaptive threshold calculation.
730
+
731
+ Returns median_degree and mean_degree for use in adaptive penalties.
732
+ These replace hard-coded penalty values with graph-relative ones.
733
+ """
734
+ if G.number_of_nodes() == 0:
735
+ return {"median_degree": 1.0, "mean_degree": 1.0}
736
+ degrees = [G.degree(n) for n in G.nodes]
737
+ sorted_degrees = sorted(degrees)
738
+ n = len(sorted_degrees)
739
+ median_degree = sorted_degrees[n // 2] if n % 2 == 1 else (sorted_degrees[n // 2 - 1] + sorted_degrees[n // 2]) / 2
740
+ mean_degree = sum(degrees) / n
741
+ return {"median_degree": max(median_degree, 1.0), "mean_degree": max(mean_degree, 1.0)}
742
+
743
+
744
+ def _adaptive_penalty_score(G: nx.Graph, node_id: str, base_penalty: float, degree_stats: dict[str, float]) -> float:
745
+ """Compute adaptive penalty based on node degree relative to graph statistics.
746
+
747
+ Instead of hard-coded penalties (e.g., score -= 5.0), we scale penalties
748
+ by the node's degree relative to the median degree. High-degree nodes
749
+ get full penalties (they are hubs — if they don't match discriminative
750
+ terms, they are almost certainly irrelevant), while low-degree nodes
751
+ get reduced penalties (they may be specific but under-connected).
752
+
753
+ base_penalty: the maximum penalty to apply (e.g., 5.0 for non-discriminative)
754
+ """
755
+ degree = G.degree(node_id) if node_id in G else 0
756
+ median_degree = degree_stats.get("median_degree", 1.0)
757
+ # Scale penalty: high-degree nodes get full penalty, low-degree get reduced
758
+ # degree >= median: penalty = base_penalty * 1.0
759
+ # degree << median: penalty = base_penalty * 0.5
760
+ scale = min(1.0, max(0.5, degree / median_degree)) if median_degree > 0 else 1.0
761
+ return base_penalty * scale
762
+
763
+
764
+ def _adaptive_threshold(scores: list[float], min_threshold: float = 1.0) -> float:
765
+ """Find adaptive threshold using elbow detection.
766
+
767
+ Instead of hard-coded threshold = top_score * 0.65, we find the "elbow"
768
+ in the score distribution - the point where scores start dropping rapidly.
769
+ This naturally adapts to different query result distributions.
770
+ """
771
+ if not scores:
772
+ return min_threshold
773
+ if len(scores) <= 2:
774
+ return max(scores[0] * 0.65, min_threshold)
775
+
776
+ # Sort scores descending
777
+ sorted_scores = sorted(scores, reverse=True)
778
+
779
+ # Find elbow: point with maximum second derivative (curvature)
780
+ # Simple approach: find where score drop exceeds average drop
781
+ drops = [sorted_scores[i] - sorted_scores[i + 1] for i in range(len(sorted_scores) - 1)]
782
+ if not drops:
783
+ return max(sorted_scores[0] * 0.65, min_threshold)
784
+
785
+ avg_drop = sum(drops) / len(drops)
786
+
787
+ # Find first index where drop exceeds 1.5x average
788
+ for i, drop in enumerate(drops):
789
+ if drop > avg_drop * 1.5 and i > 0:
790
+ # Threshold is the score at the elbow point (before the big drop)
791
+ return max(sorted_scores[i], min_threshold)
792
+
793
+ # No clear elbow found, use percentile-based threshold
794
+ return max(sorted_scores[len(sorted_scores) // 3], min_threshold)
795
+
796
+
797
+ def _adaptive_community_quota(top_k: int, num_communities_in_graph: int) -> int:
798
+ """Compute adaptive community quota for seed diversity.
799
+
800
+ Instead of hard-coded community_quota = max(top_k // 2, 2), we scale
801
+ based on the number of communities in the graph. More communities
802
+ means we need more seeds per community to achieve coverage.
803
+ """
804
+ if num_communities_in_graph <= 2:
805
+ return max(top_k // 2, 2)
806
+ # Scale quota: more communities → smaller quota per community
807
+ # but ensure at least 2 seeds per community for diversity
808
+ base_quota = max(top_k // 2, 2)
809
+ scale = min(1.0, max(0.5, 3.0 / num_communities_in_graph))
810
+ return max(int(base_quota * scale), 1)
811
+
812
+
718
813
  def _parse_god_labels_from_report(report_text: str) -> set[str]:
719
814
  """Parse god node labels from GRAPH_REPORT.md text.
720
815
 
@@ -878,6 +973,124 @@ def _relation_family_weight(relation: str, confidence: str = "") -> float:
878
973
  return base + confidence_bonus
879
974
 
880
975
 
976
+ def _connection_query(G: nx.Graph, question: str, terms: list[str], hub_labels: set[str] | None = None, god_labels: set[str] | None = None) -> dict[str, object]:
977
+ """Handle connection/bridge queries: find paths between concept clusters.
978
+
979
+ Strategy:
980
+ 1. Find seeds from different communities (cross-domain seeds)
981
+ 2. Compute shortest paths between cross-community seed pairs
982
+ 3. Identify bridge nodes (nodes appearing on multiple paths)
983
+ 4. Return bridge concepts + paths + neighbor summaries
984
+ """
985
+ # Find seeds with community diversity emphasis
986
+ seeds = _select_seed_nodes(G, question, terms, top_k=7, hub_labels=hub_labels, god_labels=god_labels)
987
+ if not seeds:
988
+ return {
989
+ "question": question,
990
+ "query_mode": "connection",
991
+ "terms": terms,
992
+ "bridges": [],
993
+ "paths": [],
994
+ "text": "No matching nodes found for connection query.",
995
+ }
996
+
997
+ # Group seeds by community
998
+ seed_by_community: dict[int, list[dict]] = {}
999
+ for seed in seeds:
1000
+ cid = G.nodes.get(seed["node_id"], {}).get("community")
1001
+ if cid is not None:
1002
+ seed_by_community.setdefault(cid, []).append(seed)
1003
+
1004
+ # Need at least 2 communities for a connection query
1005
+ if len(seed_by_community) < 2:
1006
+ # Fall back to regular task query
1007
+ return None # Signal to caller to use task mode
1008
+
1009
+ # Find cross-community seed pairs and compute shortest paths
1010
+ paths: list[dict] = []
1011
+ bridge_counter: dict[str, int] = {} # node_id -> count of paths it appears on
1012
+ community_pairs = []
1013
+ communities = list(seed_by_community.keys())
1014
+ for i in range(len(communities)):
1015
+ for j in range(i + 1, len(communities)):
1016
+ community_pairs.append((communities[i], communities[j]))
1017
+
1018
+ for cid_a, cid_b in community_pairs[:6]: # Limit to 6 pairs for performance
1019
+ seeds_a = seed_by_community[cid_a]
1020
+ seeds_b = seed_by_community[cid_b]
1021
+ # Try shortest path between best seed from each community
1022
+ for sa in seeds_a[:2]:
1023
+ for sb in seeds_b[:2]:
1024
+ try:
1025
+ path = nx.shortest_path(G, sa["node_id"], sb["node_id"])
1026
+ except (nx.NetworkXNoPath, nx.NodeNotFound):
1027
+ continue
1028
+ if len(path) < 2:
1029
+ continue
1030
+ # Record path
1031
+ path_labels = [G.nodes[n].get("label", n) for n in path]
1032
+ path_relations = []
1033
+ for k in range(len(path) - 1):
1034
+ edge = _edge_payload(G, path[k], path[k + 1])
1035
+ path_relations.append(edge.get("relation", ""))
1036
+ paths.append({
1037
+ "from": path_labels[0],
1038
+ "to": path_labels[-1],
1039
+ "from_community": cid_a,
1040
+ "to_community": cid_b,
1041
+ "path": path_labels,
1042
+ "relations": path_relations,
1043
+ "length": len(path) - 1,
1044
+ })
1045
+ # Count bridge nodes (intermediate nodes on paths)
1046
+ for node_id in path[1:-1]: # Exclude endpoints
1047
+ bridge_counter[node_id] = bridge_counter.get(node_id, 0) + 1
1048
+
1049
+ # Identify bridge nodes: appear on multiple paths
1050
+ bridges: list[dict] = []
1051
+ for node_id, count in sorted(bridge_counter.items(), key=lambda x: -x[1]):
1052
+ if count < 1:
1053
+ continue
1054
+ label = G.nodes[node_id].get("label", node_id)
1055
+ degree = G.degree(node_id)
1056
+ # Get neighbor summary
1057
+ neighbors = list(G.neighbors(node_id))
1058
+ neighbor_labels = [G.nodes[n].get("label", n) for n in neighbors[:8]]
1059
+ neighbor_communities = {G.nodes[n].get("community") for n in neighbors if G.nodes[n].get("community") is not None}
1060
+ bridges.append({
1061
+ "node_id": node_id,
1062
+ "label": label,
1063
+ "degree": degree,
1064
+ "path_count": count,
1065
+ "neighbor_count": len(neighbors),
1066
+ "neighbor_communities": len(neighbor_communities),
1067
+ "sample_neighbors": neighbor_labels,
1068
+ })
1069
+
1070
+ # Build text summary
1071
+ text_parts = [f"Connection query: {question}"]
1072
+ if bridges:
1073
+ text_parts.append(f"\nBridge concepts ({len(bridges)}):")
1074
+ for b in bridges[:5]:
1075
+ text_parts.append(f" - {b['label']} (appears on {b['path_count']} paths, connects {b['neighbor_communities']} communities)")
1076
+ if paths:
1077
+ text_parts.append(f"\nPaths found: {len(paths)}")
1078
+ for p in paths[:3]:
1079
+ text_parts.append(f" {p['from']} --[{', '.join(p['relations'])}]--> {p['to']} (length {p['length']})")
1080
+ if len(p['path']) > 2:
1081
+ text_parts.append(f" via: {' -> '.join(p['path'][1:-1])}")
1082
+
1083
+ return {
1084
+ "question": question,
1085
+ "query_mode": "connection",
1086
+ "terms": terms,
1087
+ "seeds": [{"node_id": s["node_id"], "label": s["label"], "seed_score": s["seed_score"]} for s in seeds],
1088
+ "bridges": bridges[:10],
1089
+ "paths": paths[:6],
1090
+ "text": "\n".join(text_parts),
1091
+ }
1092
+
1093
+
881
1094
  def _select_seed_nodes(G: nx.Graph, question: str, terms: list[str], top_k: int = 5, hub_labels: set[str] | None = None, god_labels: set[str] | None = None) -> list[dict[str, object]]:
882
1095
  scored_nodes = _score_nodes(G, terms)
883
1096
  if not scored_nodes:
@@ -885,6 +1098,10 @@ def _select_seed_nodes(G: nx.Graph, question: str, terms: list[str], top_k: int
885
1098
  base_scores = {nid: score for score, nid in scored_nodes}
886
1099
  lexical_frontier = {nid for _, nid in scored_nodes[: max(top_k * 3, 12)]}
887
1100
  compact_question = _normalize_query_text(question).replace(" ", "")
1101
+ # Compute graph degree statistics for adaptive penalties
1102
+ degree_stats = _compute_graph_degree_stats(G)
1103
+ # Count communities for adaptive quota
1104
+ all_communities = {G.nodes[n].get("community") for n in G.nodes if G.nodes[n].get("community") is not None}
888
1105
  # Identify discriminative terms: those NOT matching generic/non-discriminative words.
889
1106
  # Seeds that only match generic terms (e.g. "component") but miss
890
1107
  # discriminative terms (e.g. "video") should be penalized.
@@ -933,26 +1150,29 @@ def _select_seed_nodes(G: nx.Graph, question: str, terms: list[str], top_k: int
933
1150
  score += structural_support
934
1151
  reasons.append("supported by nearby graph structure")
935
1152
  if _is_noise_node(data, hub_labels):
936
- score -= 3.0
1153
+ score -= _adaptive_penalty_score(G, nid, 3.0, degree_stats)
937
1154
  # Enum/config types (ending in Mode, Flag, Type, etc.) make poor
938
1155
  # navigation seeds — users ask about components, not enum values.
939
1156
  if re.match(r".*(Mode|Flag|Type|Options|Config|Style|Format|Encoding|Space|Completed|Started|Stopped|Paused|Unknown)$", label):
940
- score -= 2.0
1157
+ score -= _adaptive_penalty_score(G, nid, 2.0, degree_stats)
941
1158
  # Penalize seeds that only match generic terms (e.g. "component")
942
1159
  # but miss all discriminative terms (e.g. "video", "animation").
943
1160
  if discriminative_compacts:
944
1161
  label_spaced, label_compact = _normalized_label_forms(data.get("norm_label") or label)
945
1162
  hits_discriminative = any(dt in label_compact or dt in label_spaced for dt in discriminative_compacts)
946
1163
  if not hits_discriminative:
947
- score -= 5.0
1164
+ score -= _adaptive_penalty_score(G, nid, 5.0, degree_stats)
948
1165
  if score > 0:
949
1166
  scored.append((score, nid, reasons))
950
1167
  scored.sort(key=lambda item: item[0], reverse=True)
951
1168
  if not scored:
952
1169
  return []
953
- threshold = max(scored[0][0] * 0.65, 1.0)
1170
+ # Adaptive threshold: find elbow in score distribution instead of hard-coded 0.65
1171
+ all_scores = [s for s, _, _ in scored]
1172
+ threshold = _adaptive_threshold(all_scores)
954
1173
  seeds: list[dict[str, object]] = []
955
- community_quota = max(top_k // 2, 2)
1174
+ # Adaptive community quota: scale based on number of communities
1175
+ community_quota = _adaptive_community_quota(top_k, len(all_communities))
956
1176
  community_count: dict[int, int] = {}
957
1177
  for score, nid, reasons in scored:
958
1178
  if len(seeds) >= top_k:
@@ -1652,6 +1872,13 @@ def _query_graph(
1652
1872
  query_mode = _classify_query_mode(question, terms)
1653
1873
  # Compute hub labels from graph structure (replaces hard-coded _GENERIC_LABELS)
1654
1874
  hub_labels = _compute_hub_labels(G)
1875
+ if query_mode == "connection":
1876
+ result = _connection_query(G, question, terms, hub_labels=hub_labels, god_labels=god_labels)
1877
+ if result is not None:
1878
+ result["query_variants"] = query_texts
1879
+ return result
1880
+ # Fall back to task mode if not enough communities
1881
+ query_mode = "task"
1655
1882
  if query_mode == "task":
1656
1883
  seeds = _select_seed_nodes(G, question, terms, top_k=5, hub_labels=hub_labels, god_labels=god_labels)
1657
1884
  if not seeds:
@@ -1707,7 +1934,9 @@ def _query_graph(
1707
1934
  "text": "No matching nodes found.",
1708
1935
  }
1709
1936
  top_score = scored[0][0]
1710
- threshold = top_score * 0.5
1937
+ # Adaptive threshold: use elbow detection instead of hard-coded 0.5
1938
+ all_scores = [s for s, _ in scored]
1939
+ threshold = _adaptive_threshold(all_scores, min_threshold=0.5)
1711
1940
  start_nodes = [nid for s, nid in scored[:5] if s >= threshold]
1712
1941
  if not start_nodes:
1713
1942
  return {
@@ -2170,7 +2399,7 @@ def serve(graph_path: str = "graphify-out/graph.json", workspace: str | None = N
2170
2399
  return {int(k): v for k, v in json.loads(labels_path.read_text(encoding="utf-8")).items()}
2171
2400
  except Exception:
2172
2401
  pass
2173
- return {cid: f"Community {cid}" for cid in communities}
2402
+ return label_communities(G, communities)
2174
2403
 
2175
2404
  @server.list_resources()
2176
2405
  async def list_resources() -> list[types.Resource]:
@@ -22,7 +22,7 @@ def _rebuild_code(watch_path: Path, *, follow_symlinks: bool = False) -> bool:
22
22
  from graphify.extract import extract
23
23
  from graphify.detect import detect
24
24
  from graphify.build import build_from_json
25
- from graphify.cluster import cluster, score_all
25
+ from graphify.cluster import cluster, score_all, label_communities
26
26
  from graphify.analyze import god_nodes, surprising_connections, suggest_questions
27
27
  from graphify.report import generate
28
28
  from graphify.export import to_json, to_html
@@ -72,7 +72,7 @@ def _rebuild_code(watch_path: Path, *, follow_symlinks: bool = False) -> bool:
72
72
  cohesion = score_all(G, communities)
73
73
  gods = god_nodes(G)
74
74
  surprises = surprising_connections(G, communities)
75
- labels = {cid: "Community " + str(cid) for cid in communities}
75
+ labels = label_communities(G, communities)
76
76
  questions = suggest_questions(G, communities, labels)
77
77
 
78
78
  out.mkdir(exist_ok=True)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: graphify_vault
3
- Version: 0.3.2
3
+ Version: 0.3.4
4
4
  Summary: Fork of github.com/safishamsi/graphify with GRAPHIFY_WORKSPACE path resolution for distributed graph artifacts, same-label node merging, incremental build fixes, and local knowledge vault capabilities
5
5
  License: MIT License
6
6
 
@@ -56,6 +56,9 @@ Requires-Dist: tree-sitter-julia
56
56
  Requires-Dist: tree-sitter-verilog
57
57
  Provides-Extra: mcp
58
58
  Requires-Dist: mcp; extra == "mcp"
59
+ Requires-Dist: deep-translator; extra == "mcp"
60
+ Requires-Dist: pypdf; extra == "mcp"
61
+ Requires-Dist: html2text; extra == "mcp"
59
62
  Provides-Extra: neo4j
60
63
  Requires-Dist: neo4j; extra == "neo4j"
61
64
  Provides-Extra: pdf
@@ -75,6 +78,7 @@ Requires-Dist: faster-whisper; extra == "video"
75
78
  Requires-Dist: yt-dlp; extra == "video"
76
79
  Provides-Extra: all
77
80
  Requires-Dist: mcp; extra == "all"
81
+ Requires-Dist: deep-translator; extra == "all"
78
82
  Requires-Dist: neo4j; extra == "all"
79
83
  Requires-Dist: pypdf; extra == "all"
80
84
  Requires-Dist: html2text; extra == "all"
@@ -1,2 +1,3 @@
1
1
  [console_scripts]
2
2
  graphify = graphify.__main__:main
3
+ gv = graphify.__main__:main
@@ -24,6 +24,7 @@ tree-sitter-verilog
24
24
 
25
25
  [all]
26
26
  mcp
27
+ deep-translator
27
28
  neo4j
28
29
  pypdf
29
30
  html2text
@@ -44,6 +45,9 @@ graspologic
44
45
 
45
46
  [mcp]
46
47
  mcp
48
+ deep-translator
49
+ pypdf
50
+ html2text
47
51
 
48
52
  [neo4j]
49
53
  neo4j
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "graphify_vault"
7
- version = "0.3.2"
7
+ version = "0.3.4"
8
8
  description = "Fork of github.com/safishamsi/graphify with GRAPHIFY_WORKSPACE path resolution for distributed graph artifacts, same-label node merging, incremental build fixes, and local knowledge vault capabilities"
9
9
  readme = "README.md"
10
10
  license = { file = "LICENSE" }
@@ -42,7 +42,7 @@ Repository = "https://github.com/gengwb/graphify"
42
42
  Issues = "https://github.com/gengwb/graphify/issues"
43
43
 
44
44
  [project.optional-dependencies]
45
- mcp = ["mcp"]
45
+ mcp = ["mcp", "deep-translator", "pypdf", "html2text"]
46
46
  neo4j = ["neo4j"]
47
47
  pdf = ["pypdf", "html2text"]
48
48
  watch = ["watchdog"]
@@ -50,10 +50,11 @@ svg = ["matplotlib"]
50
50
  leiden = ["graspologic; python_version < '3.13'"]
51
51
  office = ["python-docx", "openpyxl"]
52
52
  video = ["faster-whisper", "yt-dlp"]
53
- all = ["mcp", "neo4j", "pypdf", "html2text", "watchdog", "graspologic; python_version < '3.13'", "python-docx", "openpyxl", "faster-whisper", "yt-dlp", "matplotlib"]
53
+ all = ["mcp", "deep-translator", "neo4j", "pypdf", "html2text", "watchdog", "graspologic; python_version < '3.13'", "python-docx", "openpyxl", "faster-whisper", "yt-dlp", "matplotlib"]
54
54
 
55
55
  [project.scripts]
56
56
  graphify = "graphify.__main__:main"
57
+ gv = "graphify.__main__:main"
57
58
 
58
59
  [tool.setuptools.packages.find]
59
60
  where = ["."]
File without changes
File without changes
File without changes