tscode-kg 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tscode_kg/__init__.py +42 -0
- tscode_kg/__main__.py +6 -0
- tscode_kg/analysis.py +1829 -0
- tscode_kg/app.py +1355 -0
- tscode_kg/bridge.py +114 -0
- tscode_kg/centrality.py +434 -0
- tscode_kg/cli/__init__.py +1 -0
- tscode_kg/cli/cmd_analyze.py +69 -0
- tscode_kg/cli/cmd_bridges.py +38 -0
- tscode_kg/cli/cmd_build.py +86 -0
- tscode_kg/cli/cmd_centrality.py +124 -0
- tscode_kg/cli/cmd_explain.py +58 -0
- tscode_kg/cli/cmd_framework_nodes.py +43 -0
- tscode_kg/cli/cmd_hooks.py +125 -0
- tscode_kg/cli/cmd_init.py +234 -0
- tscode_kg/cli/cmd_mcp.py +35 -0
- tscode_kg/cli/cmd_model.py +52 -0
- tscode_kg/cli/cmd_query.py +75 -0
- tscode_kg/cli/cmd_snapshot.py +431 -0
- tscode_kg/cli/cmd_viz.py +175 -0
- tscode_kg/cli/main.py +56 -0
- tscode_kg/coderank.py +564 -0
- tscode_kg/config.py +36 -0
- tscode_kg/explain.py +270 -0
- tscode_kg/extractor.py +827 -0
- tscode_kg/framework_detector.py +106 -0
- tscode_kg/kg.py +193 -0
- tscode_kg/layout3d.py +492 -0
- tscode_kg/mcp_server.py +1412 -0
- tscode_kg/snapshots.py +64 -0
- tscode_kg/viz3d.py +1457 -0
- tscode_kg/viz3d_timeline.py +369 -0
- tscode_kg-0.2.0.dist-info/METADATA +196 -0
- tscode_kg-0.2.0.dist-info/RECORD +37 -0
- tscode_kg-0.2.0.dist-info/WHEEL +4 -0
- tscode_kg-0.2.0.dist-info/entry_points.txt +15 -0
- tscode_kg-0.2.0.dist-info/licenses/LICENSE +24 -0
tscode_kg/mcp_server.py
ADDED
|
@@ -0,0 +1,1412 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
mcp_server.py — TypeScriptKG MCP Server
|
|
4
|
+
|
|
5
|
+
Exposes the TypeScriptKG hybrid query and snippet-pack pipeline as
|
|
6
|
+
Model Context Protocol (MCP) tools, allowing any MCP-compatible agent
|
|
7
|
+
(Claude Desktop, Cursor, Continue, etc.) to query a TypeScript/JavaScript
|
|
8
|
+
codebase knowledge graph directly.
|
|
9
|
+
|
|
10
|
+
Tools
|
|
11
|
+
-----
|
|
12
|
+
query_codebase(q, k, hop, rels, max_nodes, min_score, max_per_module, rerank_mode)
|
|
13
|
+
Hybrid semantic + structural query. Returns ranked nodes and edges.
|
|
14
|
+
|
|
15
|
+
pack_snippets(q, k, hop, rels, context, max_lines, max_nodes, min_score, rerank_mode)
|
|
16
|
+
Hybrid query + source-grounded snippet extraction.
|
|
17
|
+
|
|
18
|
+
callers(node_id, rel, paths)
|
|
19
|
+
Reverse lookup: find all callers of a node, resolving sym: stubs.
|
|
20
|
+
|
|
21
|
+
get_node(node_id, include_edges)
|
|
22
|
+
Fetch a single node by its stable ID.
|
|
23
|
+
|
|
24
|
+
graph_stats()
|
|
25
|
+
Return node and edge counts by kind/relation as Markdown.
|
|
26
|
+
|
|
27
|
+
list_nodes(module_path, kind)
|
|
28
|
+
List nodes filtered by module path prefix and/or kind.
|
|
29
|
+
|
|
30
|
+
find_node(name, kind)
|
|
31
|
+
Find nodes by plain name or qualname substring.
|
|
32
|
+
|
|
33
|
+
centrality(top, kinds, group_by)
|
|
34
|
+
SIR PageRank — rank nodes or modules by structural importance.
|
|
35
|
+
|
|
36
|
+
bridge_centrality(top, include_imports)
|
|
37
|
+
Module connectivity — unique module interactions per module.
|
|
38
|
+
|
|
39
|
+
framework_nodes(top)
|
|
40
|
+
Framework-like hub modules via SIR + module connectivity.
|
|
41
|
+
|
|
42
|
+
find_definition_at(file, line)
|
|
43
|
+
Reverse-resolve a (file, line) location to a node and explain it.
|
|
44
|
+
|
|
45
|
+
analyze_repo()
|
|
46
|
+
Run structural analysis and return a Markdown report.
|
|
47
|
+
|
|
48
|
+
explain(node_id, limit)
|
|
49
|
+
Natural-language explanation of a node: callers, callees, role.
|
|
50
|
+
|
|
51
|
+
rank_nodes(top, rels, persist_metric, exclude_tests)
|
|
52
|
+
Global weighted CodeRank (PageRank) over the repository graph.
|
|
53
|
+
|
|
54
|
+
query_ranked(q, k, mode, top, rels, radius, exclude_tests)
|
|
55
|
+
CodeRank-enhanced hybrid or personalized-PageRank query ranking.
|
|
56
|
+
|
|
57
|
+
explain_rank(node_id, q)
|
|
58
|
+
Explain the CodeRank score components for a specific node.
|
|
59
|
+
|
|
60
|
+
snapshot_list(limit, branch)
|
|
61
|
+
List saved temporal metric snapshots, most recent first.
|
|
62
|
+
|
|
63
|
+
snapshot_show(key)
|
|
64
|
+
Show full details of one snapshot ("latest" for the most recent).
|
|
65
|
+
|
|
66
|
+
snapshot_diff(key_a, key_b)
|
|
67
|
+
Compare two metric snapshots side-by-side.
|
|
68
|
+
|
|
69
|
+
Author: Eric G. Suchanek, PhD
|
|
70
|
+
"""
|
|
71
|
+
|
|
72
|
+
from __future__ import annotations
|
|
73
|
+
|
|
74
|
+
import argparse
|
|
75
|
+
import json
|
|
76
|
+
import sys
|
|
77
|
+
from pathlib import Path
|
|
78
|
+
|
|
79
|
+
from kg_utils.semantic import DEFAULT_MODEL
|
|
80
|
+
from kg_utils.store import DEFAULT_RELS
|
|
81
|
+
from mcp.server.fastmcp import FastMCP
|
|
82
|
+
|
|
83
|
+
from tscode_kg.kg import TypeScriptKG
|
|
84
|
+
from tscode_kg.snapshots import SnapshotManager
|
|
85
|
+
|
|
86
|
+
# ---------------------------------------------------------------------------
|
|
87
|
+
# Global state — initialised in main()
|
|
88
|
+
# ---------------------------------------------------------------------------
|
|
89
|
+
|
|
90
|
+
_kg: TypeScriptKG | None = None
|
|
91
|
+
_snapshot_mgr: SnapshotManager | None = None
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _get_kg() -> TypeScriptKG:
|
|
95
|
+
if _kg is None:
|
|
96
|
+
raise RuntimeError(
|
|
97
|
+
"TypeScriptKG not initialised. Run the server via 'tscodekg-mcp --repo /path/to/repo'"
|
|
98
|
+
)
|
|
99
|
+
return _kg
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _get_snapshot_mgr() -> SnapshotManager:
|
|
103
|
+
if _snapshot_mgr is None:
|
|
104
|
+
raise RuntimeError(
|
|
105
|
+
"SnapshotManager not initialised. Run the server via 'tscodekg-mcp --repo /path/to/repo'"
|
|
106
|
+
)
|
|
107
|
+
return _snapshot_mgr
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _snapshot_freshness(snapshot_total_nodes: int) -> dict:
|
|
111
|
+
"""Compare a snapshot's node count against the currently loaded graph DB.
|
|
112
|
+
|
|
113
|
+
:param snapshot_total_nodes: ``metrics.total_nodes`` from a snapshot object.
|
|
114
|
+
:return: Freshness metadata payload.
|
|
115
|
+
"""
|
|
116
|
+
current = _get_kg().stats()
|
|
117
|
+
current_nodes = int(current.get("total_nodes", 0))
|
|
118
|
+
delta = current_nodes - int(snapshot_total_nodes)
|
|
119
|
+
|
|
120
|
+
is_fresh = delta == 0
|
|
121
|
+
status = "fresh" if delta == 0 else ("behind" if delta > 0 else "ahead")
|
|
122
|
+
note = None
|
|
123
|
+
|
|
124
|
+
if 0 < delta < 50:
|
|
125
|
+
is_fresh = True
|
|
126
|
+
status = "near_fresh"
|
|
127
|
+
note = "Within tolerance (sym: stubs often accumulate between rebuilds)"
|
|
128
|
+
|
|
129
|
+
out = {
|
|
130
|
+
"snapshot_total_nodes": int(snapshot_total_nodes),
|
|
131
|
+
"current_total_nodes": current_nodes,
|
|
132
|
+
"delta_nodes": delta,
|
|
133
|
+
"is_fresh": is_fresh,
|
|
134
|
+
"status": status,
|
|
135
|
+
}
|
|
136
|
+
if note:
|
|
137
|
+
out["note"] = note
|
|
138
|
+
return out
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
# ---------------------------------------------------------------------------
|
|
142
|
+
# MCP server
|
|
143
|
+
# ---------------------------------------------------------------------------
|
|
144
|
+
|
|
145
|
+
mcp = FastMCP(
|
|
146
|
+
"tscodekg",
|
|
147
|
+
instructions=(
|
|
148
|
+
"TypeScriptKG is a hybrid semantic + structural knowledge graph for TypeScript and "
|
|
149
|
+
"JavaScript codebases. It indexes every module, class, interface, type alias, enum, "
|
|
150
|
+
"function, and method as a node, with typed edges "
|
|
151
|
+
"(CALLS, IMPORTS, CONTAINS, INHERITS, IMPLEMENTS, EXTENDS) connecting them.\n\n"
|
|
152
|
+
"## Tools\n\n"
|
|
153
|
+
"**graph_stats()** — Start here. Returns node/edge counts by kind and relation "
|
|
154
|
+
"plus JSDoc coverage. Use before issuing query_codebase() or pack_snippets().\n\n"
|
|
155
|
+
"**query_codebase(q, k, hop, rels, max_nodes, min_score, max_per_module, rerank_mode)** — "
|
|
156
|
+
"Hybrid semantic + structural search. Seeds on vector similarity then expands through "
|
|
157
|
+
"the graph. Returns ranked nodes and edges as JSON.\n\n"
|
|
158
|
+
"**pack_snippets(q, k, hop, rels, context, max_lines, max_nodes, min_score, rerank_mode)** — "
|
|
159
|
+
"Same hybrid search, but returns actual source code as a Markdown context pack "
|
|
160
|
+
"with ranked, deduplicated snippets and line numbers.\n\n"
|
|
161
|
+
"**callers(node_id, rel, paths)** — Precise reverse lookup: every node that calls "
|
|
162
|
+
"(or inherits from, imports, …) the given node, resolving cross-module sym: stubs. "
|
|
163
|
+
"Filter with paths='src/' to exclude test callers.\n\n"
|
|
164
|
+
"**get_node(node_id, include_edges)** — Precise lookup of a single node by its stable "
|
|
165
|
+
"ID (e.g. 'cls:src/auth/middleware.ts:AuthMiddleware').\n\n"
|
|
166
|
+
"**list_nodes(module_path, kind)** — List nodes filtered by module path and/or kind.\n\n"
|
|
167
|
+
"**find_node(name, kind)** — Find nodes by name substring when the stable ID is unknown.\n\n"
|
|
168
|
+
"**centrality(top, kinds, group_by)** — Structural Importance Ranking (SIR): "
|
|
169
|
+
"deterministic weighted PageRank over the graph. group_by='node' ranks individual "
|
|
170
|
+
"nodes; group_by='module' aggregates per module. Use to find hotspots before "
|
|
171
|
+
"refactoring or review.\n\n"
|
|
172
|
+
"**bridge_centrality(top, include_imports)** — Module connectivity: how many unique "
|
|
173
|
+
"modules each module calls or is called by. Identifies orchestrator/hub modules.\n\n"
|
|
174
|
+
"**framework_nodes(top)** — Framework-like hub modules: 0.6 × SIR + 0.4 × "
|
|
175
|
+
"connectivity (both normalized). Surfaces the repo-defining abstractions.\n\n"
|
|
176
|
+
"**find_definition_at(file, line)** — Reverse-resolve a (file, line) location to the "
|
|
177
|
+
"innermost enclosing node and return its explain() report.\n\n"
|
|
178
|
+
"**analyze_repo()** — Structural analysis: node counts, edge counts, JSDoc coverage, "
|
|
179
|
+
"node distribution by kind.\n\n"
|
|
180
|
+
"**explain(node_id, limit)** — Natural-language Markdown explanation of a node: "
|
|
181
|
+
"metadata, JSDoc, callers, callees, and its role in the codebase.\n\n"
|
|
182
|
+
"**rank_nodes(top, rels, persist_metric, exclude_tests)** — Global weighted CodeRank "
|
|
183
|
+
"(PageRank). Returns the most structurally important nodes as JSON.\n\n"
|
|
184
|
+
"**query_ranked(q, k, mode, top, rels, radius, exclude_tests)** — Query ranking that "
|
|
185
|
+
"combines semantic seeds with centrality and graph proximity ('hybrid') or "
|
|
186
|
+
"personalized PageRank ('ppr'); results include 'why' explanations.\n\n"
|
|
187
|
+
"**explain_rank(node_id, q)** — Break down a node's CodeRank score: global rank, "
|
|
188
|
+
"inbound/outbound structural edges, and optional query-conditioned scores.\n\n"
|
|
189
|
+
"**snapshot_list(limit, branch)** — List saved temporal metric snapshots (most recent "
|
|
190
|
+
"first) with per-snapshot deltas and freshness vs. the live graph.\n\n"
|
|
191
|
+
"**snapshot_show(key)** — Full details of one snapshot; pass 'latest' (default) for "
|
|
192
|
+
"the most recent.\n\n"
|
|
193
|
+
"**snapshot_diff(key_a, key_b)** — Side-by-side comparison of two snapshots with "
|
|
194
|
+
"computed deltas (b − a).\n\n"
|
|
195
|
+
"## Recommended Workflows\n\n"
|
|
196
|
+
"- **Explore unfamiliar TS/JS repo**: graph_stats → query_codebase → pack_snippets\n"
|
|
197
|
+
"- **Find a specific class/function**: find_node(name) → get_node(include_edges=True)\n"
|
|
198
|
+
"- **Understand an interface**: get_node → pack_snippets\n"
|
|
199
|
+
"- **Trace usage of a symbol**: find_node(name) → callers(node_id)\n"
|
|
200
|
+
"- **Identify structural hotspots**: centrality(top=20) or centrality(group_by='module')\n"
|
|
201
|
+
"- **Architecture review**: analyze_repo\n"
|
|
202
|
+
"- **Track codebase evolution**: snapshot_list → snapshot_diff(key_a, key_b)\n"
|
|
203
|
+
"- **Answer 'how does X work?'**: pack_snippets with a descriptive query\n"
|
|
204
|
+
),
|
|
205
|
+
)
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
@mcp.tool()
|
|
209
|
+
def query_codebase(
|
|
210
|
+
q: str,
|
|
211
|
+
k: int = 8,
|
|
212
|
+
hop: int = 1,
|
|
213
|
+
rels: str = "CONTAINS,CALLS,IMPORTS,INHERITS,IMPLEMENTS,EXTENDS",
|
|
214
|
+
max_nodes: int = 25,
|
|
215
|
+
min_score: float = 0.0,
|
|
216
|
+
max_per_module: int = 3,
|
|
217
|
+
rerank_mode: str = "hybrid",
|
|
218
|
+
rerank_semantic_weight: float = 0.7,
|
|
219
|
+
rerank_lexical_weight: float = 0.3,
|
|
220
|
+
format: str = "json",
|
|
221
|
+
) -> str:
|
|
222
|
+
"""
|
|
223
|
+
Hybrid semantic + structural query over the TypeScript/JavaScript codebase graph.
|
|
224
|
+
|
|
225
|
+
:param q: Natural-language query, e.g. "authentication middleware".
|
|
226
|
+
:param k: Number of semantic seed nodes (default 8).
|
|
227
|
+
:param hop: Graph expansion hops (default 1).
|
|
228
|
+
:param rels: Comma-separated edge types to follow.
|
|
229
|
+
:param max_nodes: Maximum nodes to return (default 25).
|
|
230
|
+
:param min_score: Minimum semantic score for seed inclusion in [0, 1].
|
|
231
|
+
:param max_per_module: Maximum nodes per module (default 3; 0 disables).
|
|
232
|
+
:param rerank_mode: 'hybrid' (default), 'semantic', or 'legacy'.
|
|
233
|
+
:param rerank_semantic_weight: Semantic weight for hybrid mode (default 0.7).
|
|
234
|
+
:param rerank_lexical_weight: Lexical weight for hybrid mode (default 0.3).
|
|
235
|
+
:param format: 'json' (default) or 'markdown'.
|
|
236
|
+
:return: JSON string or Markdown table.
|
|
237
|
+
"""
|
|
238
|
+
rel_tuple = tuple(r.strip() for r in rels.split(",") if r.strip())
|
|
239
|
+
result = _get_kg().query(
|
|
240
|
+
q,
|
|
241
|
+
k=k,
|
|
242
|
+
hop=hop,
|
|
243
|
+
rels=rel_tuple or DEFAULT_RELS,
|
|
244
|
+
max_nodes=max_nodes,
|
|
245
|
+
min_score=min_score,
|
|
246
|
+
max_per_module=max_per_module if max_per_module > 0 else None,
|
|
247
|
+
rerank_mode=rerank_mode,
|
|
248
|
+
rerank_semantic_weight=rerank_semantic_weight,
|
|
249
|
+
rerank_lexical_weight=rerank_lexical_weight,
|
|
250
|
+
)
|
|
251
|
+
data = json.loads(result.to_json())
|
|
252
|
+
|
|
253
|
+
if format == "markdown":
|
|
254
|
+
out: list[str] = [
|
|
255
|
+
f"## Query Results: `{q}`\n",
|
|
256
|
+
f"**Seeds:** {data['seeds']} | "
|
|
257
|
+
f"**Expanded:** {data['expanded_nodes']} | "
|
|
258
|
+
f"**Returned:** {data['returned_nodes']}\n",
|
|
259
|
+
"| Rank | Score | Kind | Name | Module |",
|
|
260
|
+
"|-----:|------:|------|------|--------|",
|
|
261
|
+
]
|
|
262
|
+
for rank_idx, node in enumerate(data["nodes"], start=1):
|
|
263
|
+
score = node.get("relevance", {}).get("score", 0.0)
|
|
264
|
+
kind = node.get("kind", "?")
|
|
265
|
+
name = node.get("qualname") or node.get("name", "?")
|
|
266
|
+
module = node.get("module_path", "")
|
|
267
|
+
out.append(f"| {rank_idx} | {score:.3f} | {kind} | `{name}` | `{module}` |")
|
|
268
|
+
return "\n".join(out)
|
|
269
|
+
|
|
270
|
+
return json.dumps(data, indent=2, ensure_ascii=False)
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
@mcp.tool()
|
|
274
|
+
def pack_snippets(
|
|
275
|
+
q: str,
|
|
276
|
+
k: int = 8,
|
|
277
|
+
hop: int = 1,
|
|
278
|
+
rels: str = "CONTAINS,CALLS,IMPORTS,INHERITS,IMPLEMENTS,EXTENDS",
|
|
279
|
+
context: int = 5,
|
|
280
|
+
max_lines: int = 60,
|
|
281
|
+
max_nodes: int = 15,
|
|
282
|
+
min_score: float = 0.0,
|
|
283
|
+
max_per_module: int = 3,
|
|
284
|
+
rerank_mode: str = "hybrid",
|
|
285
|
+
rerank_semantic_weight: float = 0.7,
|
|
286
|
+
rerank_lexical_weight: float = 0.3,
|
|
287
|
+
) -> str:
|
|
288
|
+
"""
|
|
289
|
+
Hybrid query + source-grounded TypeScript/JS snippet extraction.
|
|
290
|
+
|
|
291
|
+
Returns a Markdown context pack with ranked, deduplicated code snippets
|
|
292
|
+
and line numbers — ready for direct LLM ingestion.
|
|
293
|
+
|
|
294
|
+
:param q: Natural-language query, e.g. "error handling middleware".
|
|
295
|
+
:param k: Number of semantic seed nodes (default 8).
|
|
296
|
+
:param hop: Graph expansion hops (default 1).
|
|
297
|
+
:param rels: Comma-separated edge types to follow.
|
|
298
|
+
:param context: Extra context lines around each definition (default 5).
|
|
299
|
+
:param max_lines: Maximum lines per snippet block (default 60).
|
|
300
|
+
:param max_nodes: Maximum nodes to include in the pack (default 15).
|
|
301
|
+
:param min_score: Minimum semantic score for seed inclusion in [0, 1].
|
|
302
|
+
:param max_per_module: Maximum nodes per module (default 3; 0 disables).
|
|
303
|
+
:param rerank_mode: 'hybrid' (default), 'semantic', or 'legacy'.
|
|
304
|
+
:param rerank_semantic_weight: Semantic weight for hybrid mode (default 0.7).
|
|
305
|
+
:param rerank_lexical_weight: Lexical weight for hybrid mode (default 0.3).
|
|
306
|
+
:return: Markdown string with source-grounded code snippets.
|
|
307
|
+
"""
|
|
308
|
+
rel_tuple = tuple(r.strip() for r in rels.split(",") if r.strip())
|
|
309
|
+
pack = _get_kg().pack(
|
|
310
|
+
q,
|
|
311
|
+
k=k,
|
|
312
|
+
hop=hop,
|
|
313
|
+
rels=rel_tuple or DEFAULT_RELS,
|
|
314
|
+
context=context,
|
|
315
|
+
max_lines=max_lines,
|
|
316
|
+
max_nodes=max_nodes,
|
|
317
|
+
min_score=min_score,
|
|
318
|
+
max_per_module=max_per_module if max_per_module > 0 else None,
|
|
319
|
+
rerank_mode=rerank_mode,
|
|
320
|
+
rerank_semantic_weight=rerank_semantic_weight,
|
|
321
|
+
rerank_lexical_weight=rerank_lexical_weight,
|
|
322
|
+
)
|
|
323
|
+
return pack.to_markdown()
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
@mcp.tool()
|
|
327
|
+
def callers(node_id: str, rel: str = "CALLS", paths: str = "") -> str:
|
|
328
|
+
"""
|
|
329
|
+
Return all nodes that call a given node, resolving through ``sym:`` stubs.
|
|
330
|
+
|
|
331
|
+
Unlike ``query_codebase`` (which seeds on semantics and expands outward),
|
|
332
|
+
this tool performs a precise reverse lookup: it finds every caller of the
|
|
333
|
+
specified node, including cross-module callers that reference it via an
|
|
334
|
+
import alias recorded as a ``sym:`` stub.
|
|
335
|
+
|
|
336
|
+
The ``rel`` parameter accepts any edge relation, not just ``CALLS``::
|
|
337
|
+
|
|
338
|
+
callers(node_id, rel="INHERITS") # find all subclasses
|
|
339
|
+
callers(node_id, rel="IMPLEMENTS") # find all implementations
|
|
340
|
+
callers(node_id, rel="IMPORTS") # find all importers
|
|
341
|
+
|
|
342
|
+
Typical workflow::
|
|
343
|
+
|
|
344
|
+
# 1. Resolve the exact node ID
|
|
345
|
+
get_node("fn:src/utils/helpers.ts:formatDate")
|
|
346
|
+
|
|
347
|
+
# 2. Find all callers (production code only)
|
|
348
|
+
callers("fn:src/utils/helpers.ts:formatDate", paths="src/")
|
|
349
|
+
|
|
350
|
+
:param node_id: Target node identifier, e.g.
|
|
351
|
+
``cls:src/auth/middleware.ts:AuthMiddleware``.
|
|
352
|
+
:param rel: Relation type to invert (default ``"CALLS"``).
|
|
353
|
+
:param paths: Comma-separated module path prefixes to include, e.g.
|
|
354
|
+
``"src/"`` to exclude test callers.
|
|
355
|
+
Empty string (default) returns all callers.
|
|
356
|
+
:return: JSON with ``node_id``, ``rel``, ``caller_count``, and
|
|
357
|
+
``callers`` list of node dicts.
|
|
358
|
+
"""
|
|
359
|
+
caller_list = _get_kg().callers(node_id, rel=rel)
|
|
360
|
+
if paths:
|
|
361
|
+
path_prefixes = [p.strip() for p in paths.split(",") if p.strip()]
|
|
362
|
+
caller_list = [
|
|
363
|
+
c
|
|
364
|
+
for c in caller_list
|
|
365
|
+
if any((c.get("module_path") or "").startswith(pfx) for pfx in path_prefixes)
|
|
366
|
+
]
|
|
367
|
+
return json.dumps(
|
|
368
|
+
{
|
|
369
|
+
"node_id": node_id,
|
|
370
|
+
"rel": rel,
|
|
371
|
+
"caller_count": len(caller_list),
|
|
372
|
+
"callers": caller_list,
|
|
373
|
+
},
|
|
374
|
+
indent=2,
|
|
375
|
+
ensure_ascii=False,
|
|
376
|
+
)
|
|
377
|
+
|
|
378
|
+
|
|
379
|
+
@mcp.tool()
|
|
380
|
+
def get_node(node_id: str, include_edges: bool = False) -> str:
|
|
381
|
+
"""
|
|
382
|
+
Fetch a single TypeScript/JS node by its stable ID and render as Markdown.
|
|
383
|
+
|
|
384
|
+
Node IDs follow the pattern ``<kind>:<module_path>:<qualname>``, e.g.
|
|
385
|
+
``cls:src/auth/middleware.ts:AuthMiddleware`` or
|
|
386
|
+
``fn:src/utils/helpers.ts:formatDate``.
|
|
387
|
+
|
|
388
|
+
:param node_id: Stable node identifier.
|
|
389
|
+
:param include_edges: If True, append outgoing edges and incoming callers.
|
|
390
|
+
:return: Markdown-formatted node summary.
|
|
391
|
+
"""
|
|
392
|
+
kg = _get_kg()
|
|
393
|
+
node = kg.node(node_id)
|
|
394
|
+
if node is None:
|
|
395
|
+
return f"## Node Not Found\n\nNode ID `{node_id}` does not exist in the knowledge graph."
|
|
396
|
+
|
|
397
|
+
kind = node.get("kind", "unknown")
|
|
398
|
+
name = node.get("qualname") or node.get("name", "unknown")
|
|
399
|
+
out: list[str] = [f"## `{name}` ({kind})\n"]
|
|
400
|
+
|
|
401
|
+
module = node.get("module_path", "")
|
|
402
|
+
lineno = node.get("lineno")
|
|
403
|
+
end_lineno = node.get("end_lineno")
|
|
404
|
+
if module:
|
|
405
|
+
out.append(f"- **Module:** `{module}`")
|
|
406
|
+
if lineno is not None:
|
|
407
|
+
loc = f"line {lineno}"
|
|
408
|
+
if end_lineno:
|
|
409
|
+
loc += f"–{end_lineno}"
|
|
410
|
+
out.append(f"- **Location:** {loc}")
|
|
411
|
+
out.append(f"- **ID:** `{node_id}`")
|
|
412
|
+
out.append("")
|
|
413
|
+
|
|
414
|
+
docstring = node.get("docstring", "").strip()
|
|
415
|
+
if docstring:
|
|
416
|
+
out.append("### JSDoc\n")
|
|
417
|
+
out.append(docstring)
|
|
418
|
+
out.append("")
|
|
419
|
+
|
|
420
|
+
if not include_edges:
|
|
421
|
+
return "\n".join(out)
|
|
422
|
+
|
|
423
|
+
store = getattr(kg, "_store", None)
|
|
424
|
+
if store is not None:
|
|
425
|
+
for rel in ("CALLS", "CONTAINS", "IMPORTS", "INHERITS", "IMPLEMENTS", "EXTENDS"):
|
|
426
|
+
edges = store.edges_from(node_id, rel=rel)
|
|
427
|
+
visible = [e for e in edges if not e["dst"].startswith("sym:")] if edges else []
|
|
428
|
+
if visible:
|
|
429
|
+
out.append(f"### Outgoing {rel}\n")
|
|
430
|
+
for e in visible:
|
|
431
|
+
out.append(f"- `{e['dst']}`")
|
|
432
|
+
out.append("")
|
|
433
|
+
|
|
434
|
+
try:
|
|
435
|
+
caller_nodes = kg.callers(node_id, rel="CALLS")
|
|
436
|
+
if caller_nodes:
|
|
437
|
+
out.append("### Incoming Calls\n")
|
|
438
|
+
for c in caller_nodes:
|
|
439
|
+
cname = c.get("qualname") or c.get("name", "")
|
|
440
|
+
cmod = c.get("module_path", "")
|
|
441
|
+
cline = c.get("lineno")
|
|
442
|
+
cid = c.get("id", "")
|
|
443
|
+
loc_str = f" (line {cline})" if cline else ""
|
|
444
|
+
out.append(f"- `{cid}` — `{cname}` in `{cmod}`{loc_str}")
|
|
445
|
+
out.append("")
|
|
446
|
+
except (AttributeError, ValueError, RuntimeError):
|
|
447
|
+
pass
|
|
448
|
+
|
|
449
|
+
return "\n".join(out)
|
|
450
|
+
|
|
451
|
+
|
|
452
|
+
@mcp.tool()
|
|
453
|
+
def graph_stats() -> str:
|
|
454
|
+
"""
|
|
455
|
+
Return node and edge counts by kind and relation as Markdown.
|
|
456
|
+
|
|
457
|
+
Call this first when engaging with a new TypeScript/JavaScript repo.
|
|
458
|
+
Reports JSDoc coverage (fraction of functions/methods with JSDoc comments).
|
|
459
|
+
|
|
460
|
+
:return: Markdown summary with total counts, nodes-by-kind, and edges-by-relation tables.
|
|
461
|
+
"""
|
|
462
|
+
stats = _get_kg().stats()
|
|
463
|
+
out: list[str] = ["## TypeScriptKG Graph Statistics\n"]
|
|
464
|
+
out.append(f"- **Database:** `{stats.get('db_path', '')}`")
|
|
465
|
+
out.append(f"- **Total nodes:** {stats.get('total_nodes', 0):,}")
|
|
466
|
+
out.append(
|
|
467
|
+
f"- **Meaningful nodes:** {stats.get('meaningful_nodes', 0):,} *(excludes sym: stubs)*"
|
|
468
|
+
)
|
|
469
|
+
out.append(f"- **Total edges:** {stats.get('total_edges', 0):,}")
|
|
470
|
+
cov = stats.get("docstring_coverage")
|
|
471
|
+
if cov is not None:
|
|
472
|
+
out.append(f"- **JSDoc coverage:** {cov:.1%} *(functions + methods)*")
|
|
473
|
+
out.append("")
|
|
474
|
+
|
|
475
|
+
node_counts: dict = stats.get("node_counts", {})
|
|
476
|
+
if node_counts:
|
|
477
|
+
out.append("### Nodes by Kind\n")
|
|
478
|
+
out.append("| Kind | Count |")
|
|
479
|
+
out.append("|------|------:|")
|
|
480
|
+
for kind, count in sorted(node_counts.items(), key=lambda x: -x[1]):
|
|
481
|
+
out.append(f"| {kind} | {count:,} |")
|
|
482
|
+
out.append("")
|
|
483
|
+
|
|
484
|
+
edge_counts: dict = stats.get("edge_counts", {})
|
|
485
|
+
if edge_counts:
|
|
486
|
+
out.append("### Edges by Relation\n")
|
|
487
|
+
out.append("| Relation | Count |")
|
|
488
|
+
out.append("|----------|------:|")
|
|
489
|
+
for rel, count in sorted(edge_counts.items(), key=lambda x: -x[1]):
|
|
490
|
+
out.append(f"| {rel} | {count:,} |")
|
|
491
|
+
out.append("")
|
|
492
|
+
|
|
493
|
+
out.append(
|
|
494
|
+
"> `sym:` nodes are import stub placeholders for external packages — "
|
|
495
|
+
"they are not local code entities."
|
|
496
|
+
)
|
|
497
|
+
return "\n".join(out)
|
|
498
|
+
|
|
499
|
+
|
|
500
|
+
@mcp.tool()
|
|
501
|
+
def list_nodes(
|
|
502
|
+
module_path: str = "",
|
|
503
|
+
kind: str = "",
|
|
504
|
+
) -> str:
|
|
505
|
+
"""
|
|
506
|
+
List nodes filtered by module path prefix and/or kind.
|
|
507
|
+
|
|
508
|
+
:param module_path: Module path prefix filter (e.g. "src/auth/middleware.ts").
|
|
509
|
+
:param kind: Node kind filter: module | class | interface | type_alias | enum |
|
|
510
|
+
namespace | function | method.
|
|
511
|
+
:return: JSON array of matching node dicts.
|
|
512
|
+
"""
|
|
513
|
+
kg = _get_kg()
|
|
514
|
+
store = getattr(kg, "_store", None)
|
|
515
|
+
if not store:
|
|
516
|
+
return json.dumps({"error": "No database store available."}, indent=2)
|
|
517
|
+
|
|
518
|
+
q = "SELECT id, name, qualname, kind, module_path, lineno, docstring FROM nodes WHERE 1=1"
|
|
519
|
+
q += " AND id NOT LIKE 'sym:%'"
|
|
520
|
+
params = []
|
|
521
|
+
|
|
522
|
+
if module_path:
|
|
523
|
+
q += " AND module_path LIKE ?"
|
|
524
|
+
params.append(f"{module_path}%")
|
|
525
|
+
if kind:
|
|
526
|
+
q += " AND kind = ?"
|
|
527
|
+
params.append(kind)
|
|
528
|
+
|
|
529
|
+
q += " ORDER BY module_path, lineno"
|
|
530
|
+
|
|
531
|
+
try:
|
|
532
|
+
rows = store.con.execute(q, params).fetchall()
|
|
533
|
+
result = []
|
|
534
|
+
for r in rows:
|
|
535
|
+
doc = r[6]
|
|
536
|
+
if doc and len(doc) > 120:
|
|
537
|
+
doc = doc[:120] + "..."
|
|
538
|
+
result.append(
|
|
539
|
+
{
|
|
540
|
+
"id": r[0],
|
|
541
|
+
"name": r[1],
|
|
542
|
+
"qualname": r[2],
|
|
543
|
+
"kind": r[3],
|
|
544
|
+
"module_path": r[4],
|
|
545
|
+
"lineno": r[5],
|
|
546
|
+
"docstring": doc,
|
|
547
|
+
}
|
|
548
|
+
)
|
|
549
|
+
return json.dumps(result, indent=2, ensure_ascii=False)
|
|
550
|
+
except Exception as e: # pylint: disable=broad-except
|
|
551
|
+
return json.dumps({"error": str(e)}, indent=2)
|
|
552
|
+
|
|
553
|
+
|
|
554
|
+
@mcp.tool()
|
|
555
|
+
def find_node(name: str, kind: str = "") -> str:
|
|
556
|
+
"""
|
|
557
|
+
Find graph nodes by name without knowing their full stable ID.
|
|
558
|
+
|
|
559
|
+
Case-insensitive match against name and qualname. Use when you know a
|
|
560
|
+
function or class name from reading code and need its stable ID.
|
|
561
|
+
|
|
562
|
+
:param name: Function, class, or interface name to search for.
|
|
563
|
+
:param kind: Optional kind filter: module | class | interface | function | method | etc.
|
|
564
|
+
:return: JSON array of matching node dicts.
|
|
565
|
+
"""
|
|
566
|
+
kg = _get_kg()
|
|
567
|
+
store = getattr(kg, "_store", None)
|
|
568
|
+
if not store:
|
|
569
|
+
return json.dumps({"error": "No database store available."}, indent=2)
|
|
570
|
+
|
|
571
|
+
name_lower = name.lower()
|
|
572
|
+
q = (
|
|
573
|
+
"SELECT id, name, qualname, kind, module_path, lineno, docstring "
|
|
574
|
+
"FROM nodes WHERE (LOWER(name) = ? OR LOWER(qualname) LIKE ?)"
|
|
575
|
+
" AND id NOT LIKE 'sym:%'"
|
|
576
|
+
)
|
|
577
|
+
params: list = [name_lower, f"%{name_lower}%"]
|
|
578
|
+
if kind:
|
|
579
|
+
q += " AND kind = ?"
|
|
580
|
+
params.append(kind)
|
|
581
|
+
q += " ORDER BY module_path, lineno"
|
|
582
|
+
|
|
583
|
+
try:
|
|
584
|
+
rows = store.con.execute(q, params).fetchall()
|
|
585
|
+
result = []
|
|
586
|
+
for r in rows:
|
|
587
|
+
doc = r[6]
|
|
588
|
+
if doc and len(doc) > 120:
|
|
589
|
+
doc = doc[:120] + "..."
|
|
590
|
+
result.append(
|
|
591
|
+
{
|
|
592
|
+
"id": r[0],
|
|
593
|
+
"name": r[1],
|
|
594
|
+
"qualname": r[2],
|
|
595
|
+
"kind": r[3],
|
|
596
|
+
"module_path": r[4],
|
|
597
|
+
"lineno": r[5],
|
|
598
|
+
"docstring": doc,
|
|
599
|
+
}
|
|
600
|
+
)
|
|
601
|
+
return json.dumps(result, indent=2, ensure_ascii=False)
|
|
602
|
+
except Exception as e: # pylint: disable=broad-except
|
|
603
|
+
return json.dumps({"error": str(e)}, indent=2)
|
|
604
|
+
|
|
605
|
+
|
|
606
|
+
@mcp.tool()
|
|
607
|
+
def centrality(
|
|
608
|
+
top: int = 20,
|
|
609
|
+
kinds: str = "",
|
|
610
|
+
group_by: str = "node",
|
|
611
|
+
) -> str:
|
|
612
|
+
"""
|
|
613
|
+
Compute Structural Importance Ranking (SIR) for the indexed codebase.
|
|
614
|
+
|
|
615
|
+
Runs a deterministic weighted PageRank over the sym-stub-resolved call
|
|
616
|
+
graph. Edge weights are tuned per relation type
|
|
617
|
+
(CALLS > INHERITS/IMPLEMENTS > IMPORTS > CONTAINS) and amplified for
|
|
618
|
+
cross-module links; private symbols receive a post-convergence penalty.
|
|
619
|
+
Scores are normalized to sum to 1.0.
|
|
620
|
+
|
|
621
|
+
Use this to:
|
|
622
|
+
|
|
623
|
+
- Identify the most structurally critical functions, classes, and interfaces
|
|
624
|
+
- Understand which modules are most depended upon
|
|
625
|
+
- Prioritize code review, refactoring, or test coverage efforts
|
|
626
|
+
|
|
627
|
+
:param top: Maximum number of ranked entries to return (default 20).
|
|
628
|
+
:param kinds: Comma-separated node kinds to include: ``module``, ``class``,
|
|
629
|
+
``interface``, ``function``, ``method``. Empty string returns
|
|
630
|
+
all kinds. Ignored when ``group_by='module'`` (all kinds
|
|
631
|
+
contribute to module aggregation).
|
|
632
|
+
:param group_by: ``node`` (default) returns individual node rankings with
|
|
633
|
+
score, inbound edge count, and cross-module inbound count;
|
|
634
|
+
``module`` aggregates node scores per module.
|
|
635
|
+
:return: Markdown-formatted ranking table.
|
|
636
|
+
"""
|
|
637
|
+
try:
|
|
638
|
+
from tscode_kg.centrality import ( # noqa: PLC0415
|
|
639
|
+
StructuralImportanceRanker,
|
|
640
|
+
aggregate_module_scores,
|
|
641
|
+
)
|
|
642
|
+
|
|
643
|
+
db_path = _get_kg().db_path
|
|
644
|
+
ranker = StructuralImportanceRanker(db_path)
|
|
645
|
+
all_records = ranker.compute()
|
|
646
|
+
except Exception as e: # noqa: BLE001
|
|
647
|
+
return f"## Centrality Error\n\nFailed to compute SIR scores: `{e}`"
|
|
648
|
+
|
|
649
|
+
out: list[str] = ["## Structural Importance Ranking (SIR)\n"]
|
|
650
|
+
|
|
651
|
+
if group_by == "module":
|
|
652
|
+
payload = aggregate_module_scores(all_records)[:top]
|
|
653
|
+
out.append(f"**Group by:** module | **Top:** {top}\n")
|
|
654
|
+
out.append("| Rank | Score | Members | Module |")
|
|
655
|
+
out.append("|-----:|------:|--------:|--------|")
|
|
656
|
+
for row in payload:
|
|
657
|
+
out.append(
|
|
658
|
+
f"| {row['rank']} | {row['score']:.6f}"
|
|
659
|
+
f" | {row['member_count']} | `{row['module_path']}` |"
|
|
660
|
+
)
|
|
661
|
+
else:
|
|
662
|
+
kind_set: set[str] | None = None
|
|
663
|
+
if kinds.strip():
|
|
664
|
+
kind_set = {k.strip().lower() for k in kinds.split(",") if k.strip()}
|
|
665
|
+
|
|
666
|
+
filtered = [r for r in all_records if kind_set is None or r.kind in kind_set][:top]
|
|
667
|
+
label = kinds if kind_set else "all kinds"
|
|
668
|
+
out.append(f"**Group by:** node | **Top:** {top} | **Filter:** {label}\n")
|
|
669
|
+
out.append("| Rank | Score | Kind | Name | Module | Inbound | XMod |")
|
|
670
|
+
out.append("|-----:|------:|------|------|--------|--------:|-----:|")
|
|
671
|
+
for r in filtered:
|
|
672
|
+
module = f"`{r.module_path}`" if r.module_path else "—"
|
|
673
|
+
out.append(
|
|
674
|
+
f"| {r.rank} | {r.score:.6f} | {r.kind} | `{r.name}`"
|
|
675
|
+
f" | {module} | {r.inbound_count} | {r.cross_module_inbound} |"
|
|
676
|
+
)
|
|
677
|
+
|
|
678
|
+
out.append("")
|
|
679
|
+
out.append(
|
|
680
|
+
"> SIR scores are normalized to sum 1.0 across all nodes. "
|
|
681
|
+
"Higher score = more structurally central. "
|
|
682
|
+
"XMod = cross-module inbound edges."
|
|
683
|
+
)
|
|
684
|
+
return "\n".join(out)
|
|
685
|
+
|
|
686
|
+
|
|
687
|
+
@mcp.tool()
|
|
688
|
+
def bridge_centrality(
|
|
689
|
+
top: int = 20,
|
|
690
|
+
include_imports: bool = True,
|
|
691
|
+
) -> str:
|
|
692
|
+
"""
|
|
693
|
+
Compute module connectivity: how many unique modules each module interacts with.
|
|
694
|
+
|
|
695
|
+
For well-modularized codebases, identifies orchestrator and hub modules that
|
|
696
|
+
touch many other modules. Replaces betweenness centrality (which is meaningless
|
|
697
|
+
when inter-module edges are zero).
|
|
698
|
+
|
|
699
|
+
**Connectivity score** = (unique modules called + unique modules calling this) / 30 + frequency / 50
|
|
700
|
+
Higher score = more complex coupling with other modules.
|
|
701
|
+
|
|
702
|
+
Scores are persisted to the ``centrality_scores`` table under the
|
|
703
|
+
``module_connectivity`` metric for use by ``framework_nodes()``.
|
|
704
|
+
|
|
705
|
+
:param top: Number of top connectivity modules to return (default 20).
|
|
706
|
+
:param include_imports: Whether to include IMPORTS in connectivity (default True).
|
|
707
|
+
:return: Markdown-formatted ranking table of modules by connectivity.
|
|
708
|
+
"""
|
|
709
|
+
try:
|
|
710
|
+
from tscode_kg.bridge import compute_bridge_centrality # noqa: PLC0415
|
|
711
|
+
|
|
712
|
+
db_path = str(_get_kg().db_path)
|
|
713
|
+
modules = compute_bridge_centrality(
|
|
714
|
+
kind="module",
|
|
715
|
+
include_imports=include_imports,
|
|
716
|
+
top=top,
|
|
717
|
+
db_path=db_path,
|
|
718
|
+
)
|
|
719
|
+
except Exception as e: # noqa: BLE001
|
|
720
|
+
return f"## Module Connectivity Error\n\nFailed to compute connectivity: `{e}`"
|
|
721
|
+
|
|
722
|
+
out: list[str] = ["## Module Connectivity (Interaction Complexity)\n"]
|
|
723
|
+
out.append(f"**Top:** {top} | **Include imports:** {include_imports}\n")
|
|
724
|
+
out.append("| Rank | Connectivity | Module |")
|
|
725
|
+
out.append("|-----:|-------------:|--------|")
|
|
726
|
+
for rank_idx, (mod, score) in enumerate(modules, start=1):
|
|
727
|
+
out.append(f"| {rank_idx} | {score:.6f} | `{mod}` |")
|
|
728
|
+
out.append("")
|
|
729
|
+
out.append(
|
|
730
|
+
"> Connectivity = unique modules called + unique modules calling this module. "
|
|
731
|
+
"Higher score = orchestrator/hub module with complex interactions. "
|
|
732
|
+
"Scores are persisted as `module_connectivity` metric for use by `framework_nodes()`."
|
|
733
|
+
)
|
|
734
|
+
return "\n".join(out)
|
|
735
|
+
|
|
736
|
+
|
|
737
|
+
@mcp.tool()
|
|
738
|
+
def framework_nodes(top: int = 20) -> str:
|
|
739
|
+
"""
|
|
740
|
+
Identify framework-like (hub) modules using SIR + module connectivity.
|
|
741
|
+
|
|
742
|
+
A "framework node" is a module that is both:
|
|
743
|
+
- Structurally important (high SIR/PageRank — central to the graph)
|
|
744
|
+
- Highly connected (calls/imports many modules — orchestrator/hub role)
|
|
745
|
+
|
|
746
|
+
Framework score = 0.6 × normalized SIR + 0.4 × normalized connectivity,
|
|
747
|
+
both auto-computed on first call. High-scoring modules are critical hubs:
|
|
748
|
+
architecturally central AND complex in their interactions.
|
|
749
|
+
|
|
750
|
+
:param top: Number of top framework-like modules to return (default 20).
|
|
751
|
+
:return: Markdown-formatted ranking table of framework nodes.
|
|
752
|
+
"""
|
|
753
|
+
try:
|
|
754
|
+
from tscode_kg.bridge import compute_bridge_centrality # noqa: PLC0415
|
|
755
|
+
from tscode_kg.centrality import StructuralImportanceRanker # noqa: PLC0415
|
|
756
|
+
from tscode_kg.framework_detector import detect_framework_nodes # noqa: PLC0415
|
|
757
|
+
|
|
758
|
+
kg = _get_kg()
|
|
759
|
+
db_path = str(kg.db_path)
|
|
760
|
+
|
|
761
|
+
# Compute and persist SIR scores (structural importance)
|
|
762
|
+
try:
|
|
763
|
+
ranker = StructuralImportanceRanker(db_path)
|
|
764
|
+
records = ranker.compute()
|
|
765
|
+
ranker.write_scores(records, metric="sir_pagerank")
|
|
766
|
+
except Exception as e: # noqa: BLE001
|
|
767
|
+
return f"## Framework Nodes Error\n\nFailed to compute SIR scores: `{e}`"
|
|
768
|
+
|
|
769
|
+
# Compute and persist module connectivity scores (interaction complexity)
|
|
770
|
+
try:
|
|
771
|
+
compute_bridge_centrality(kind="module", include_imports=True, top=25, db_path=db_path)
|
|
772
|
+
except Exception as e: # noqa: BLE001
|
|
773
|
+
return f"## Framework Nodes Error\n\nFailed to compute module connectivity: `{e}`"
|
|
774
|
+
|
|
775
|
+
# Detect framework nodes by combining both metrics
|
|
776
|
+
nodes = detect_framework_nodes(limit=top, db_path=db_path)
|
|
777
|
+
except Exception as e: # noqa: BLE001
|
|
778
|
+
return f"## Framework Nodes Error\n\nFailed to detect framework nodes: `{e}`"
|
|
779
|
+
|
|
780
|
+
out: list[str] = ["## Framework-like Modules (Critical Hubs)\n"]
|
|
781
|
+
out.append(f"**Top:** {top} | **Score:** 0.6 × SIR + 0.4 × connectivity (both normalized)\n")
|
|
782
|
+
out.append("| Rank | Score | Module |")
|
|
783
|
+
out.append("|-----:|------:|--------|")
|
|
784
|
+
for rank_idx, (_, score, label) in enumerate(nodes, start=1):
|
|
785
|
+
out.append(f"| {rank_idx} | {score:.6f} | `{label}` |")
|
|
786
|
+
out.append("")
|
|
787
|
+
out.append(
|
|
788
|
+
"> Framework nodes: both architecturally central (SIR) AND heavily connected "
|
|
789
|
+
"(calls/imports many modules). High-scoring modules are critical orchestrators/hubs."
|
|
790
|
+
)
|
|
791
|
+
return "\n".join(out)
|
|
792
|
+
|
|
793
|
+
|
|
794
|
+
@mcp.tool()
|
|
795
|
+
def find_definition_at(file: str, line: int) -> str:
|
|
796
|
+
"""
|
|
797
|
+
Find the code node whose definition spans a given file location.
|
|
798
|
+
|
|
799
|
+
Reverse-resolves a ``(file, line)`` pair to a graph node ID and returns the
|
|
800
|
+
same Markdown report as ``explain()``. Useful when reading a file in an IDE
|
|
801
|
+
and wanting to understand the symbol at a specific line without constructing
|
|
802
|
+
a node ID manually.
|
|
803
|
+
|
|
804
|
+
Matches the innermost (most-specific) function, method, class, interface,
|
|
805
|
+
type alias, or enum whose ``lineno ≤ line ≤ end_lineno``. Falls back to
|
|
806
|
+
the module node when no narrower match exists.
|
|
807
|
+
|
|
808
|
+
:param file: Module path as stored in the graph, e.g. ``src/auth/middleware.ts``.
|
|
809
|
+
Leading ``./`` is stripped automatically.
|
|
810
|
+
:param line: Line number (1-indexed) within the file.
|
|
811
|
+
:return: Markdown explanation from ``explain()``, or an informative error
|
|
812
|
+
message if no node spans that location.
|
|
813
|
+
"""
|
|
814
|
+
kg = _get_kg()
|
|
815
|
+
store = getattr(kg, "_store", None) or getattr(kg, "store", None)
|
|
816
|
+
if store is None:
|
|
817
|
+
return "## Error\n\nNo graph store available."
|
|
818
|
+
|
|
819
|
+
norm_file = file.lstrip("./")
|
|
820
|
+
|
|
821
|
+
# Innermost span: smallest (end_lineno - lineno) that still contains `line`.
|
|
822
|
+
rows = store.con.execute(
|
|
823
|
+
"""
|
|
824
|
+
SELECT id
|
|
825
|
+
FROM nodes
|
|
826
|
+
WHERE (module_path = :f OR module_path LIKE :like)
|
|
827
|
+
AND kind IN ('function', 'method', 'class', 'interface', 'type_alias', 'enum')
|
|
828
|
+
AND lineno IS NOT NULL
|
|
829
|
+
AND lineno <= :ln
|
|
830
|
+
AND (end_lineno IS NULL OR end_lineno >= :ln)
|
|
831
|
+
ORDER BY (COALESCE(end_lineno, lineno) - lineno) ASC
|
|
832
|
+
LIMIT 1
|
|
833
|
+
""",
|
|
834
|
+
{"f": norm_file, "like": f"%{norm_file}", "ln": line},
|
|
835
|
+
).fetchall()
|
|
836
|
+
|
|
837
|
+
if not rows:
|
|
838
|
+
# Fall back to the module node itself
|
|
839
|
+
mod_rows = store.con.execute(
|
|
840
|
+
"SELECT id FROM nodes WHERE kind = 'module' AND (module_path = ? OR module_path LIKE ?)",
|
|
841
|
+
(norm_file, f"%{norm_file}"),
|
|
842
|
+
).fetchall()
|
|
843
|
+
if not mod_rows:
|
|
844
|
+
return (
|
|
845
|
+
f"## No Definition Found\n\n"
|
|
846
|
+
f"No function, method, class, interface, type alias, or enum spans "
|
|
847
|
+
f"`{file}:{line}` in the graph.\n\n"
|
|
848
|
+
"Check that the file path matches the module path stored in the graph "
|
|
849
|
+
"(use `graph_stats()` or `list_nodes()` to browse available modules)."
|
|
850
|
+
)
|
|
851
|
+
node_id = mod_rows[0][0]
|
|
852
|
+
else:
|
|
853
|
+
node_id = rows[0][0]
|
|
854
|
+
|
|
855
|
+
return explain(node_id)
|
|
856
|
+
|
|
857
|
+
|
|
858
|
+
@mcp.tool()
|
|
859
|
+
def analyze_repo() -> str:
|
|
860
|
+
"""
|
|
861
|
+
Run a full structural analysis of the indexed TypeScript/JavaScript repository.
|
|
862
|
+
|
|
863
|
+
Executes the 14-phase TypeScriptKG analysis pipeline — baseline metrics,
|
|
864
|
+
CodeRank, fan-in/fan-out, module coupling, critical call chains, public API
|
|
865
|
+
surface, JSDoc coverage, class/interface hierarchy, insights, snapshot
|
|
866
|
+
history, and SIR centrality — and returns the results as Markdown.
|
|
867
|
+
|
|
868
|
+
:return: Markdown-formatted analysis report.
|
|
869
|
+
"""
|
|
870
|
+
from io import StringIO # noqa: PLC0415
|
|
871
|
+
|
|
872
|
+
from rich.console import Console # noqa: PLC0415
|
|
873
|
+
|
|
874
|
+
from tscode_kg.analysis import TSCodeKGAnalyzer # noqa: PLC0415
|
|
875
|
+
from tscode_kg.kg import _render_analysis # noqa: PLC0415
|
|
876
|
+
|
|
877
|
+
# Silence Rich output — stdout carries the MCP protocol on stdio transport.
|
|
878
|
+
silent = Console(file=StringIO(), highlight=False)
|
|
879
|
+
kg = _get_kg()
|
|
880
|
+
try:
|
|
881
|
+
analyzer = TSCodeKGAnalyzer(kg, console=silent, snapshot_mgr=_snapshot_mgr)
|
|
882
|
+
analyzer.run_analysis()
|
|
883
|
+
return analyzer.to_markdown()
|
|
884
|
+
except Exception as exc: # noqa: BLE001
|
|
885
|
+
# Lightweight stats-only fallback — never re-runs the noisy analyzer.
|
|
886
|
+
try:
|
|
887
|
+
return _render_analysis(str(kg.repo_root), kg.store.stats())
|
|
888
|
+
except Exception: # noqa: BLE001
|
|
889
|
+
return f"# TypeScriptKG Analysis\n\nAnalysis failed: {exc}\n"
|
|
890
|
+
|
|
891
|
+
|
|
892
|
+
@mcp.tool()
|
|
893
|
+
def explain(node_id: str, limit: int = 10) -> str:
|
|
894
|
+
"""
|
|
895
|
+
Return a natural-language explanation of a code node.
|
|
896
|
+
|
|
897
|
+
Given a node ID (e.g., ``fn:src/utils/helpers.ts:formatDate``),
|
|
898
|
+
returns a markdown-formatted explanation that includes:
|
|
899
|
+
|
|
900
|
+
- **What it is**: The node's kind, short description from its JSDoc
|
|
901
|
+
- **Where it lives**: Module path and source location
|
|
902
|
+
- **What calls it**: The callers (reverse call graph)
|
|
903
|
+
- **What it calls**: The callees (functions/methods this node invokes)
|
|
904
|
+
- **Documentation**: Full JSDoc if available
|
|
905
|
+
|
|
906
|
+
This is ideal for understanding the role and context of a specific node
|
|
907
|
+
without needing to read the full source code. Use ``pack_snippets()``
|
|
908
|
+
to then retrieve the actual implementation.
|
|
909
|
+
|
|
910
|
+
:param node_id: Stable node identifier, e.g.
|
|
911
|
+
``fn:src/utils/helpers.ts:formatDate``.
|
|
912
|
+
:param limit: Maximum callers and callees to list (default 10). Pass 0
|
|
913
|
+
to list all.
|
|
914
|
+
:return: Markdown-formatted explanation ready for LLM consumption.
|
|
915
|
+
"""
|
|
916
|
+
from tscode_kg.explain import render_explain # noqa: PLC0415
|
|
917
|
+
|
|
918
|
+
return render_explain(
|
|
919
|
+
_get_kg(),
|
|
920
|
+
node_id,
|
|
921
|
+
limit=limit,
|
|
922
|
+
snippets_hint="pack_snippets()",
|
|
923
|
+
)
|
|
924
|
+
|
|
925
|
+
|
|
926
|
+
@mcp.tool()
|
|
927
|
+
def rank_nodes(
|
|
928
|
+
top: int = 25,
|
|
929
|
+
rels: str = "CALLS,IMPORTS,INHERITS,IMPLEMENTS,EXTENDS",
|
|
930
|
+
persist_metric: str = "",
|
|
931
|
+
exclude_tests: bool = True,
|
|
932
|
+
) -> str:
|
|
933
|
+
"""
|
|
934
|
+
Compute global weighted CodeRank (PageRank) over the repository graph.
|
|
935
|
+
|
|
936
|
+
Builds a directed weighted graph from the SQLite store and runs weighted
|
|
937
|
+
PageRank to identify the most structurally important nodes. Relation
|
|
938
|
+
weights follow the CodeRank defaults: CALLS=1.0, IMPORTS=0.9,
|
|
939
|
+
INHERITS/IMPLEMENTS/EXTENDS=0.75. Test paths are excluded by default.
|
|
940
|
+
|
|
941
|
+
Optionally persists the scores into the ``node_metrics`` table under the
|
|
942
|
+
given metric name so they can be loaded at query time without recomputing.
|
|
943
|
+
|
|
944
|
+
:param top: Number of top-ranked nodes to return (default 25).
|
|
945
|
+
:param rels: Comma-separated relations to include in the graph
|
|
946
|
+
(default ``"CALLS,IMPORTS,INHERITS,IMPLEMENTS,EXTENDS"``).
|
|
947
|
+
:param persist_metric: If non-empty, persist scores to ``node_metrics``
|
|
948
|
+
under this metric name (e.g. ``"coderank_global"``).
|
|
949
|
+
:param exclude_tests: Exclude test-path nodes from the graph (default True).
|
|
950
|
+
:return: JSON array of ranked node dicts with ``node_id``, ``score``,
|
|
951
|
+
``top_pct`` (e.g. ``"top 0.5%"``), ``kind``, ``qualname``,
|
|
952
|
+
``module_path``, and ``rank`` fields.
|
|
953
|
+
"""
|
|
954
|
+
from tscode_kg.coderank import ( # noqa: PLC0415
|
|
955
|
+
build_code_graph,
|
|
956
|
+
compute_coderank,
|
|
957
|
+
persist_metric_scores,
|
|
958
|
+
)
|
|
959
|
+
|
|
960
|
+
db_path = str(_get_kg().db_path)
|
|
961
|
+
rel_list = [r.strip() for r in rels.split(",") if r.strip()]
|
|
962
|
+
|
|
963
|
+
try:
|
|
964
|
+
graph = build_code_graph(
|
|
965
|
+
db_path,
|
|
966
|
+
include_relations=rel_list,
|
|
967
|
+
exclude_test_paths=exclude_tests,
|
|
968
|
+
)
|
|
969
|
+
scores = compute_coderank(graph)
|
|
970
|
+
except Exception as exc: # noqa: BLE001
|
|
971
|
+
return json.dumps({"error": str(exc)}, indent=2)
|
|
972
|
+
|
|
973
|
+
if persist_metric:
|
|
974
|
+
try:
|
|
975
|
+
persist_metric_scores(db_path, persist_metric, scores)
|
|
976
|
+
except Exception: # noqa: BLE001
|
|
977
|
+
pass # non-fatal — still return results
|
|
978
|
+
|
|
979
|
+
# Filter out sym: stubs (import placeholders) — only return real code entities
|
|
980
|
+
all_real_nodes = [
|
|
981
|
+
(nid, s)
|
|
982
|
+
for nid, s in sorted(scores.items(), key=lambda kv: kv[1], reverse=True)
|
|
983
|
+
if not nid.startswith("sym:")
|
|
984
|
+
]
|
|
985
|
+
total_real = len(all_real_nodes)
|
|
986
|
+
results = []
|
|
987
|
+
for rank_idx, (node_id, score) in enumerate(all_real_nodes[:top], start=1):
|
|
988
|
+
attrs = graph.nodes.get(node_id, {})
|
|
989
|
+
top_pct = round(rank_idx / total_real * 100, 1) if total_real > 0 else 0.0
|
|
990
|
+
results.append(
|
|
991
|
+
{
|
|
992
|
+
"rank": rank_idx,
|
|
993
|
+
"node_id": node_id,
|
|
994
|
+
"score": round(score, 8),
|
|
995
|
+
"top_pct": f"top {top_pct:.1f}%",
|
|
996
|
+
"kind": attrs.get("kind"),
|
|
997
|
+
"qualname": attrs.get("qualname"),
|
|
998
|
+
"module_path": attrs.get("module_path"),
|
|
999
|
+
}
|
|
1000
|
+
)
|
|
1001
|
+
|
|
1002
|
+
return json.dumps(results, indent=2, ensure_ascii=False)
|
|
1003
|
+
|
|
1004
|
+
|
|
1005
|
+
@mcp.tool()
|
|
1006
|
+
def query_ranked(
|
|
1007
|
+
q: str,
|
|
1008
|
+
k: int = 8,
|
|
1009
|
+
mode: str = "hybrid",
|
|
1010
|
+
top: int = 25,
|
|
1011
|
+
rels: str = "CALLS,IMPORTS,INHERITS,IMPLEMENTS,EXTENDS",
|
|
1012
|
+
radius: int = 2,
|
|
1013
|
+
exclude_tests: bool = True,
|
|
1014
|
+
) -> str:
|
|
1015
|
+
"""
|
|
1016
|
+
Rank query results using CodeRank-enhanced hybrid or personalized PageRank.
|
|
1017
|
+
|
|
1018
|
+
Combines semantic seed scores from the vector index with structural
|
|
1019
|
+
centrality and graph proximity to produce a final ranked list with
|
|
1020
|
+
explainability components.
|
|
1021
|
+
|
|
1022
|
+
Two modes are available:
|
|
1023
|
+
|
|
1024
|
+
- ``hybrid`` (default): 0.60 × semantic + 0.25 × centrality + 0.15 × proximity
|
|
1025
|
+
- ``ppr``: 0.70 × personalized PageRank + 0.30 × semantic
|
|
1026
|
+
|
|
1027
|
+
:param q: Natural-language query string.
|
|
1028
|
+
:param k: Number of semantic seed nodes to retrieve (default 8).
|
|
1029
|
+
:param mode: Ranking mode — ``"hybrid"`` (default) or ``"ppr"``.
|
|
1030
|
+
:param top: Maximum ranked results to return (default 25).
|
|
1031
|
+
:param rels: Comma-separated relations to include in the local graph.
|
|
1032
|
+
:param radius: Graph expansion radius around seeds (default 2).
|
|
1033
|
+
:param exclude_tests: Exclude test-path nodes (default True).
|
|
1034
|
+
:return: JSON array of ranked result dicts with score components and
|
|
1035
|
+
``why`` explanation strings. ``sym:`` import stub nodes are
|
|
1036
|
+
always excluded from the output.
|
|
1037
|
+
"""
|
|
1038
|
+
from tscode_kg.coderank import ( # noqa: PLC0415
|
|
1039
|
+
build_code_graph,
|
|
1040
|
+
compute_coderank,
|
|
1041
|
+
rank_query_hybrid,
|
|
1042
|
+
rank_query_ppr,
|
|
1043
|
+
)
|
|
1044
|
+
|
|
1045
|
+
kg = _get_kg()
|
|
1046
|
+
db_path = str(kg.db_path)
|
|
1047
|
+
rel_list = [r.strip() for r in rels.split(",") if r.strip()]
|
|
1048
|
+
|
|
1049
|
+
# Get semantic seeds from the vector index
|
|
1050
|
+
try:
|
|
1051
|
+
raw = kg.query(q, k=k, hop=0, rels=tuple(rel_list))
|
|
1052
|
+
seed_data = json.loads(raw.to_json())
|
|
1053
|
+
seed_nodes = seed_data.get("nodes", [])
|
|
1054
|
+
semantic_scores: dict[str, float] = {
|
|
1055
|
+
n["id"]: float((n.get("relevance") or {}).get("score", 0.0))
|
|
1056
|
+
for n in seed_nodes
|
|
1057
|
+
if (n.get("relevance") or {}).get("score", 0.0) > 0
|
|
1058
|
+
}
|
|
1059
|
+
except Exception as exc: # noqa: BLE001
|
|
1060
|
+
return json.dumps({"error": f"Seed retrieval failed: {exc}"}, indent=2)
|
|
1061
|
+
|
|
1062
|
+
if not semantic_scores:
|
|
1063
|
+
return json.dumps({"error": "No semantic seeds found for query."}, indent=2)
|
|
1064
|
+
|
|
1065
|
+
try:
|
|
1066
|
+
graph = build_code_graph(
|
|
1067
|
+
db_path,
|
|
1068
|
+
include_relations=rel_list,
|
|
1069
|
+
exclude_test_paths=exclude_tests,
|
|
1070
|
+
)
|
|
1071
|
+
except Exception as exc: # noqa: BLE001
|
|
1072
|
+
return json.dumps({"error": f"Graph build failed: {exc}"}, indent=2)
|
|
1073
|
+
|
|
1074
|
+
global_cr = compute_coderank(graph)
|
|
1075
|
+
try:
|
|
1076
|
+
if mode == "ppr":
|
|
1077
|
+
results = rank_query_ppr(graph, semantic_scores, radius=radius, top_k=top)
|
|
1078
|
+
else:
|
|
1079
|
+
results = rank_query_hybrid(
|
|
1080
|
+
graph, semantic_scores, global_coderank=global_cr, radius=radius, top_k=top
|
|
1081
|
+
)
|
|
1082
|
+
except Exception as exc: # noqa: BLE001
|
|
1083
|
+
return json.dumps({"error": f"Ranking failed: {exc}"}, indent=2)
|
|
1084
|
+
|
|
1085
|
+
output = []
|
|
1086
|
+
for rank_idx, r in enumerate(results, start=1):
|
|
1087
|
+
if r.node_id.startswith("sym:"):
|
|
1088
|
+
continue
|
|
1089
|
+
output.append(
|
|
1090
|
+
{
|
|
1091
|
+
"rank": rank_idx,
|
|
1092
|
+
"node_id": r.node_id,
|
|
1093
|
+
"adjusted_score": round(r.adjusted_score, 6),
|
|
1094
|
+
"final_score": round(r.final_score, 6),
|
|
1095
|
+
"semantic_score": round(r.semantic_score, 6),
|
|
1096
|
+
"centrality_score": round(r.centrality_score, 6),
|
|
1097
|
+
"proximity_score": round(r.proximity_score, 6),
|
|
1098
|
+
"kind": r.kind,
|
|
1099
|
+
"qualname": r.qualname,
|
|
1100
|
+
"module_path": r.module_path,
|
|
1101
|
+
"why": list(r.why),
|
|
1102
|
+
}
|
|
1103
|
+
)
|
|
1104
|
+
|
|
1105
|
+
return json.dumps(
|
|
1106
|
+
{"query": q, "mode": mode, "returned": len(output), "results": output},
|
|
1107
|
+
indent=2,
|
|
1108
|
+
ensure_ascii=False,
|
|
1109
|
+
)
|
|
1110
|
+
|
|
1111
|
+
|
|
1112
|
+
@mcp.tool()
|
|
1113
|
+
def explain_rank(node_id: str, q: str = "") -> str:
|
|
1114
|
+
"""
|
|
1115
|
+
Explain the CodeRank score components for a specific node.
|
|
1116
|
+
|
|
1117
|
+
Returns a Markdown report showing the node's structural position in the
|
|
1118
|
+
graph: how many nodes call it, import it, or inherit from / implement /
|
|
1119
|
+
extend it; its global CodeRank score; and, when a query is provided, its
|
|
1120
|
+
semantic relevance and proximity to the query seed set.
|
|
1121
|
+
|
|
1122
|
+
:param node_id: Stable node identifier, e.g.
|
|
1123
|
+
``fn:src/utils/helpers.ts:formatDate``.
|
|
1124
|
+
:param q: Optional query string. When provided, semantic score and
|
|
1125
|
+
proximity to the query seed set are included in the report.
|
|
1126
|
+
:return: Markdown-formatted explanation of the node's rank components.
|
|
1127
|
+
"""
|
|
1128
|
+
from tscode_kg.coderank import ( # noqa: PLC0415
|
|
1129
|
+
DEFAULT_GLOBAL_RELS,
|
|
1130
|
+
build_code_graph,
|
|
1131
|
+
compute_coderank,
|
|
1132
|
+
compute_seed_proximity,
|
|
1133
|
+
)
|
|
1134
|
+
|
|
1135
|
+
kg = _get_kg()
|
|
1136
|
+
db_path = str(kg.db_path)
|
|
1137
|
+
|
|
1138
|
+
node = kg.node(node_id)
|
|
1139
|
+
if node is None:
|
|
1140
|
+
return f"## Node Not Found\n\nNode ID `{node_id}` does not exist."
|
|
1141
|
+
|
|
1142
|
+
kind = node.get("kind", "unknown")
|
|
1143
|
+
name = node.get("qualname") or node.get("name", "unknown")
|
|
1144
|
+
out: list[str] = [f"## CodeRank Explanation: `{name}` ({kind})\n"]
|
|
1145
|
+
out.append(f"- **ID:** `{node_id}`")
|
|
1146
|
+
if node.get("module_path"):
|
|
1147
|
+
out.append(f"- **Module:** `{node['module_path']}`")
|
|
1148
|
+
out.append("")
|
|
1149
|
+
|
|
1150
|
+
# Build graph and compute global CodeRank
|
|
1151
|
+
try:
|
|
1152
|
+
graph = build_code_graph(
|
|
1153
|
+
db_path,
|
|
1154
|
+
include_relations=list(DEFAULT_GLOBAL_RELS),
|
|
1155
|
+
exclude_test_paths=True,
|
|
1156
|
+
)
|
|
1157
|
+
scores = compute_coderank(graph)
|
|
1158
|
+
except Exception as exc: # noqa: BLE001
|
|
1159
|
+
return f"## Error\n\nFailed to build graph: `{exc}`"
|
|
1160
|
+
|
|
1161
|
+
global_score = scores.get(node_id, 0.0)
|
|
1162
|
+
meaningful_scores = sorted(
|
|
1163
|
+
(v for k, v in scores.items() if not k.startswith("sym:")), reverse=True
|
|
1164
|
+
)
|
|
1165
|
+
rank_pos = next(
|
|
1166
|
+
(i + 1 for i, s in enumerate(meaningful_scores) if s <= global_score),
|
|
1167
|
+
len(meaningful_scores),
|
|
1168
|
+
)
|
|
1169
|
+
|
|
1170
|
+
out.append("### Global CodeRank\n")
|
|
1171
|
+
out.append(f"- **Score:** `{global_score:.8f}`")
|
|
1172
|
+
out.append(f"- **Rank:** #{rank_pos} of {len(meaningful_scores)} meaningful nodes")
|
|
1173
|
+
out.append("")
|
|
1174
|
+
|
|
1175
|
+
# Structural context from graph
|
|
1176
|
+
if node_id in graph:
|
|
1177
|
+
in_edges = list(graph.in_edges(node_id, data=True))
|
|
1178
|
+
out.append("### Structural Inbound Edges\n")
|
|
1179
|
+
callers_count = sum(1 for _, _, d in in_edges if "CALLS" in d.get("relations", set()))
|
|
1180
|
+
importers_count = sum(1 for _, _, d in in_edges if "IMPORTS" in d.get("relations", set()))
|
|
1181
|
+
inheritors_count = sum(
|
|
1182
|
+
1
|
|
1183
|
+
for _, _, d in in_edges
|
|
1184
|
+
if d.get("relations", set()) & {"INHERITS", "IMPLEMENTS", "EXTENDS"}
|
|
1185
|
+
)
|
|
1186
|
+
if callers_count:
|
|
1187
|
+
out.append(f"- Called by **{callers_count}** upstream node(s)")
|
|
1188
|
+
if importers_count:
|
|
1189
|
+
out.append(f"- Imported by **{importers_count}** upstream node(s)")
|
|
1190
|
+
if inheritors_count:
|
|
1191
|
+
out.append(
|
|
1192
|
+
f"- Inherited/implemented/extended by **{inheritors_count}** downstream node(s)"
|
|
1193
|
+
)
|
|
1194
|
+
if not (callers_count or importers_count or inheritors_count):
|
|
1195
|
+
out.append("- No inbound structural edges found in the ranked graph")
|
|
1196
|
+
out.append("")
|
|
1197
|
+
|
|
1198
|
+
out_edges = list(graph.out_edges(node_id, data=True))
|
|
1199
|
+
if out_edges:
|
|
1200
|
+
out.append("### Structural Outbound Edges\n")
|
|
1201
|
+
out.append(f"- Calls/imports/inherits **{len(out_edges)}** downstream node(s)")
|
|
1202
|
+
out.append("")
|
|
1203
|
+
|
|
1204
|
+
# Optional query-conditioned scores
|
|
1205
|
+
if q:
|
|
1206
|
+
out.append("### Query-Conditioned Scores\n")
|
|
1207
|
+
try:
|
|
1208
|
+
raw = kg.query(q, k=8, hop=0)
|
|
1209
|
+
seed_data = json.loads(raw.to_json())
|
|
1210
|
+
seed_nodes = seed_data.get("nodes", [])
|
|
1211
|
+
semantic_scores: dict[str, float] = {
|
|
1212
|
+
n["id"]: float((n.get("relevance") or {}).get("score", 0.0)) for n in seed_nodes
|
|
1213
|
+
}
|
|
1214
|
+
this_semantic = semantic_scores.get(node_id, 0.0)
|
|
1215
|
+
out.append(f"- **Query:** `{q}`")
|
|
1216
|
+
out.append(f"- **Semantic score:** `{this_semantic:.4f}`")
|
|
1217
|
+
|
|
1218
|
+
if node_id in graph:
|
|
1219
|
+
seeds = list(semantic_scores.keys())
|
|
1220
|
+
proximity = compute_seed_proximity(graph, seeds)
|
|
1221
|
+
prox = proximity.get(node_id, 0.0)
|
|
1222
|
+
out.append(f"- **Proximity to seeds:** `{prox:.4f}`")
|
|
1223
|
+
if prox >= 1.0:
|
|
1224
|
+
out.append(" → Direct semantic seed")
|
|
1225
|
+
elif prox >= 0.5:
|
|
1226
|
+
out.append(" → One hop from a semantic seed")
|
|
1227
|
+
elif prox > 0:
|
|
1228
|
+
out.append(" → Within local query neighborhood")
|
|
1229
|
+
else:
|
|
1230
|
+
out.append(" → Outside query neighborhood")
|
|
1231
|
+
except Exception as exc: # noqa: BLE001
|
|
1232
|
+
out.append(f"- Query scoring failed: `{exc}`")
|
|
1233
|
+
out.append("")
|
|
1234
|
+
|
|
1235
|
+
out.append("---\n")
|
|
1236
|
+
out.append(
|
|
1237
|
+
"*Use `rank_nodes()` for global top-N ranking, or `query_ranked()` for query-conditioned ranking.*"
|
|
1238
|
+
)
|
|
1239
|
+
return "\n".join(out)
|
|
1240
|
+
|
|
1241
|
+
|
|
1242
|
+
@mcp.tool()
|
|
1243
|
+
def snapshot_list(limit: int = 10, branch: str = "") -> str:
|
|
1244
|
+
"""
|
|
1245
|
+
List saved temporal snapshots of codebase metrics in reverse chronological order.
|
|
1246
|
+
|
|
1247
|
+
Each entry in the returned list contains a ``key`` (tree hash snapshot
|
|
1248
|
+
identifier), ``branch``, ``timestamp``, ``version``, and a summary of
|
|
1249
|
+
key metrics (node count, edge count, JSDoc coverage) plus deltas vs. the
|
|
1250
|
+
previous snapshot. Use the ``key`` field when calling ``snapshot_show()``
|
|
1251
|
+
or ``snapshot_diff(key_a=..., key_b=...)``.
|
|
1252
|
+
|
|
1253
|
+
Use this tool to answer questions like "how has the codebase grown?" or
|
|
1254
|
+
"when did JSDoc coverage improve?" or "show me only main-branch snapshots".
|
|
1255
|
+
|
|
1256
|
+
:param limit: Maximum number of snapshots to return (default 10; pass 0 for all).
|
|
1257
|
+
:param branch: If provided, filter to snapshots from this branch only
|
|
1258
|
+
(e.g. ``"main"`` or ``"develop"``).
|
|
1259
|
+
:return: JSON array of snapshot metadata dicts, most recent first.
|
|
1260
|
+
"""
|
|
1261
|
+
mgr = _get_snapshot_mgr()
|
|
1262
|
+
snapshots = mgr.list_snapshots(
|
|
1263
|
+
limit=limit if limit > 0 else None,
|
|
1264
|
+
branch=branch if branch else None,
|
|
1265
|
+
)
|
|
1266
|
+
for snap in snapshots:
|
|
1267
|
+
snap_metrics = snap.get("metrics", {})
|
|
1268
|
+
snap["freshness"] = _snapshot_freshness(snap_metrics.get("total_nodes", 0))
|
|
1269
|
+
return json.dumps(snapshots, indent=2, ensure_ascii=False)
|
|
1270
|
+
|
|
1271
|
+
|
|
1272
|
+
@mcp.tool()
|
|
1273
|
+
def snapshot_show(key: str = "latest") -> str:
|
|
1274
|
+
"""
|
|
1275
|
+
Show full details of a specific codebase metrics snapshot.
|
|
1276
|
+
|
|
1277
|
+
Pass a snapshot key (tree hash) to retrieve that exact snapshot, or use
|
|
1278
|
+
the special value ``"latest"`` (default) to retrieve the most recent one.
|
|
1279
|
+
|
|
1280
|
+
Snapshot keys are the ``key`` field returned by ``snapshot_list()``.
|
|
1281
|
+
|
|
1282
|
+
The returned object contains the full metrics dict (total_nodes,
|
|
1283
|
+
total_edges, meaningful_nodes, docstring_coverage, node_counts,
|
|
1284
|
+
edge_counts, critical_issues, complexity_median), the top hotspots, and
|
|
1285
|
+
deltas computed vs. both the previous and the baseline (oldest) snapshots.
|
|
1286
|
+
|
|
1287
|
+
:param key: Snapshot key to load, or ``"latest"`` for the most
|
|
1288
|
+
recent snapshot (default ``"latest"``). Keys are tree
|
|
1289
|
+
hashes returned by ``snapshot_list()``.
|
|
1290
|
+
:return: JSON object with full snapshot details, or an error dict if
|
|
1291
|
+
the requested snapshot does not exist.
|
|
1292
|
+
"""
|
|
1293
|
+
mgr = _get_snapshot_mgr()
|
|
1294
|
+
|
|
1295
|
+
if key == "latest":
|
|
1296
|
+
entries = mgr.list_snapshots(limit=1)
|
|
1297
|
+
if not entries:
|
|
1298
|
+
return json.dumps({"error": "No snapshots found."})
|
|
1299
|
+
key = entries[0]["key"]
|
|
1300
|
+
|
|
1301
|
+
snapshot = mgr.load_snapshot(key)
|
|
1302
|
+
if snapshot is None:
|
|
1303
|
+
return json.dumps({"error": f"Snapshot not found for key: {key!r}"})
|
|
1304
|
+
out = snapshot.to_dict()
|
|
1305
|
+
out["freshness"] = _snapshot_freshness(snapshot.metrics.get("total_nodes", 0))
|
|
1306
|
+
return json.dumps(out, indent=2, ensure_ascii=False)
|
|
1307
|
+
|
|
1308
|
+
|
|
1309
|
+
@mcp.tool()
|
|
1310
|
+
def snapshot_diff(key_a: str, key_b: str) -> str:
|
|
1311
|
+
"""
|
|
1312
|
+
Compare two codebase metric snapshots side-by-side.
|
|
1313
|
+
|
|
1314
|
+
Returns the full metrics dict for both snapshots and a computed delta
|
|
1315
|
+
(b − a) covering node and edge counts, plus per-kind node count and
|
|
1316
|
+
per-relation edge count deltas.
|
|
1317
|
+
|
|
1318
|
+
Typical workflow::
|
|
1319
|
+
|
|
1320
|
+
# 1. List available snapshots — note the 'key' field in each entry
|
|
1321
|
+
snapshot_list()
|
|
1322
|
+
|
|
1323
|
+
# 2. Diff any two using the key= field values
|
|
1324
|
+
snapshot_diff(key_a="abc1234ef...", key_b="def5678ab...")
|
|
1325
|
+
|
|
1326
|
+
:param key_a: First (older) snapshot key — the ``key`` field from
|
|
1327
|
+
``snapshot_list()`` output (a tree-hash string).
|
|
1328
|
+
:param key_b: Second (newer) snapshot key — the ``key`` field from
|
|
1329
|
+
``snapshot_list()`` output (a tree-hash string).
|
|
1330
|
+
:return: JSON object with keys ``a`` (metrics + issues list for key_a),
|
|
1331
|
+
``b`` (metrics + issues list for key_b), ``delta`` (b − a),
|
|
1332
|
+
``node_counts_delta``, and ``edge_counts_delta``. Returns an
|
|
1333
|
+
error dict if either snapshot is missing.
|
|
1334
|
+
"""
|
|
1335
|
+
mgr = _get_snapshot_mgr()
|
|
1336
|
+
result = mgr.diff_snapshots(key_a, key_b)
|
|
1337
|
+
if "error" not in result:
|
|
1338
|
+
result["freshness"] = {
|
|
1339
|
+
"a": _snapshot_freshness(result.get("a", {}).get("metrics", {}).get("total_nodes", 0)),
|
|
1340
|
+
"b": _snapshot_freshness(result.get("b", {}).get("metrics", {}).get("total_nodes", 0)),
|
|
1341
|
+
}
|
|
1342
|
+
return json.dumps(result, indent=2, ensure_ascii=False)
|
|
1343
|
+
|
|
1344
|
+
|
|
1345
|
+
# ---------------------------------------------------------------------------
|
|
1346
|
+
# CLI entry point
|
|
1347
|
+
# ---------------------------------------------------------------------------
|
|
1348
|
+
|
|
1349
|
+
|
|
1350
|
+
def _parse_args(argv: list | None = None) -> argparse.Namespace:
|
|
1351
|
+
p = argparse.ArgumentParser(
|
|
1352
|
+
prog="tscodekg-mcp",
|
|
1353
|
+
description="TypeScriptKG MCP server — exposes TS/JS codebase query tools to AI agents.",
|
|
1354
|
+
)
|
|
1355
|
+
p.add_argument("--repo", default=".", help="Repository root directory (default: .)")
|
|
1356
|
+
p.add_argument(
|
|
1357
|
+
"--db",
|
|
1358
|
+
default=".tscodekg/graph.sqlite",
|
|
1359
|
+
help="Path to the SQLite knowledge graph",
|
|
1360
|
+
)
|
|
1361
|
+
p.add_argument(
|
|
1362
|
+
"--vectors",
|
|
1363
|
+
default=".tscodekg/vectors.sqlite",
|
|
1364
|
+
help="Path to the sqlite-vec vector store",
|
|
1365
|
+
)
|
|
1366
|
+
p.add_argument(
|
|
1367
|
+
"--model",
|
|
1368
|
+
default=DEFAULT_MODEL,
|
|
1369
|
+
help=f"Sentence-transformer model name (default: {DEFAULT_MODEL})",
|
|
1370
|
+
)
|
|
1371
|
+
p.add_argument(
|
|
1372
|
+
"--transport",
|
|
1373
|
+
choices=["stdio", "sse"],
|
|
1374
|
+
default="stdio",
|
|
1375
|
+
help="MCP transport: stdio (default) or sse (HTTP)",
|
|
1376
|
+
)
|
|
1377
|
+
return p.parse_args(argv)
|
|
1378
|
+
|
|
1379
|
+
|
|
1380
|
+
def main(argv: list | None = None) -> None:
|
|
1381
|
+
"""CLI entry point for the TypeScriptKG MCP server."""
|
|
1382
|
+
global _kg, _snapshot_mgr
|
|
1383
|
+
|
|
1384
|
+
args = _parse_args(argv)
|
|
1385
|
+
|
|
1386
|
+
repo = Path(args.repo).resolve()
|
|
1387
|
+
db = Path(args.db) if Path(args.db).is_absolute() else repo / args.db
|
|
1388
|
+
vectors = Path(args.vectors) if Path(args.vectors).is_absolute() else repo / args.vectors
|
|
1389
|
+
|
|
1390
|
+
if not db.exists():
|
|
1391
|
+
print(
|
|
1392
|
+
f"WARNING: SQLite database not found at '{db}'.\nRun 'tscodekg build --repo .' first.",
|
|
1393
|
+
file=sys.stderr,
|
|
1394
|
+
)
|
|
1395
|
+
|
|
1396
|
+
print(
|
|
1397
|
+
f"TypeScriptKG MCP server starting\n"
|
|
1398
|
+
f" repo : {repo}\n"
|
|
1399
|
+
f" db : {db}\n"
|
|
1400
|
+
f" vectors : {vectors}\n"
|
|
1401
|
+
f" model : {args.model}\n"
|
|
1402
|
+
f" transport: {args.transport}",
|
|
1403
|
+
file=sys.stderr,
|
|
1404
|
+
)
|
|
1405
|
+
|
|
1406
|
+
_kg = TypeScriptKG(repo_root=repo, db_path=db, vectors_path=vectors, model=args.model)
|
|
1407
|
+
_snapshot_mgr = SnapshotManager(repo / ".tscodekg" / "snapshots", db_path=db)
|
|
1408
|
+
mcp.run(transport=args.transport)
|
|
1409
|
+
|
|
1410
|
+
|
|
1411
|
+
if __name__ == "__main__":
|
|
1412
|
+
main()
|