tscode-kg 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1412 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ mcp_server.py — TypeScriptKG MCP Server
4
+
5
+ Exposes the TypeScriptKG hybrid query and snippet-pack pipeline as
6
+ Model Context Protocol (MCP) tools, allowing any MCP-compatible agent
7
+ (Claude Desktop, Cursor, Continue, etc.) to query a TypeScript/JavaScript
8
+ codebase knowledge graph directly.
9
+
10
+ Tools
11
+ -----
12
+ query_codebase(q, k, hop, rels, max_nodes, min_score, max_per_module, rerank_mode)
13
+ Hybrid semantic + structural query. Returns ranked nodes and edges.
14
+
15
+ pack_snippets(q, k, hop, rels, context, max_lines, max_nodes, min_score, rerank_mode)
16
+ Hybrid query + source-grounded snippet extraction.
17
+
18
+ callers(node_id, rel, paths)
19
+ Reverse lookup: find all callers of a node, resolving sym: stubs.
20
+
21
+ get_node(node_id, include_edges)
22
+ Fetch a single node by its stable ID.
23
+
24
+ graph_stats()
25
+ Return node and edge counts by kind/relation as Markdown.
26
+
27
+ list_nodes(module_path, kind)
28
+ List nodes filtered by module path prefix and/or kind.
29
+
30
+ find_node(name, kind)
31
+ Find nodes by plain name or qualname substring.
32
+
33
+ centrality(top, kinds, group_by)
34
+ SIR PageRank — rank nodes or modules by structural importance.
35
+
36
+ bridge_centrality(top, include_imports)
37
+ Module connectivity — unique module interactions per module.
38
+
39
+ framework_nodes(top)
40
+ Framework-like hub modules via SIR + module connectivity.
41
+
42
+ find_definition_at(file, line)
43
+ Reverse-resolve a (file, line) location to a node and explain it.
44
+
45
+ analyze_repo()
46
+ Run structural analysis and return a Markdown report.
47
+
48
+ explain(node_id, limit)
49
+ Natural-language explanation of a node: callers, callees, role.
50
+
51
+ rank_nodes(top, rels, persist_metric, exclude_tests)
52
+ Global weighted CodeRank (PageRank) over the repository graph.
53
+
54
+ query_ranked(q, k, mode, top, rels, radius, exclude_tests)
55
+ CodeRank-enhanced hybrid or personalized-PageRank query ranking.
56
+
57
+ explain_rank(node_id, q)
58
+ Explain the CodeRank score components for a specific node.
59
+
60
+ snapshot_list(limit, branch)
61
+ List saved temporal metric snapshots, most recent first.
62
+
63
+ snapshot_show(key)
64
+ Show full details of one snapshot ("latest" for the most recent).
65
+
66
+ snapshot_diff(key_a, key_b)
67
+ Compare two metric snapshots side-by-side.
68
+
69
+ Author: Eric G. Suchanek, PhD
70
+ """
71
+
72
+ from __future__ import annotations
73
+
74
+ import argparse
75
+ import json
76
+ import sys
77
+ from pathlib import Path
78
+
79
+ from kg_utils.semantic import DEFAULT_MODEL
80
+ from kg_utils.store import DEFAULT_RELS
81
+ from mcp.server.fastmcp import FastMCP
82
+
83
+ from tscode_kg.kg import TypeScriptKG
84
+ from tscode_kg.snapshots import SnapshotManager
85
+
86
+ # ---------------------------------------------------------------------------
87
+ # Global state — initialised in main()
88
+ # ---------------------------------------------------------------------------
89
+
90
+ _kg: TypeScriptKG | None = None
91
+ _snapshot_mgr: SnapshotManager | None = None
92
+
93
+
94
+ def _get_kg() -> TypeScriptKG:
95
+ if _kg is None:
96
+ raise RuntimeError(
97
+ "TypeScriptKG not initialised. Run the server via 'tscodekg-mcp --repo /path/to/repo'"
98
+ )
99
+ return _kg
100
+
101
+
102
+ def _get_snapshot_mgr() -> SnapshotManager:
103
+ if _snapshot_mgr is None:
104
+ raise RuntimeError(
105
+ "SnapshotManager not initialised. Run the server via 'tscodekg-mcp --repo /path/to/repo'"
106
+ )
107
+ return _snapshot_mgr
108
+
109
+
110
+ def _snapshot_freshness(snapshot_total_nodes: int) -> dict:
111
+ """Compare a snapshot's node count against the currently loaded graph DB.
112
+
113
+ :param snapshot_total_nodes: ``metrics.total_nodes`` from a snapshot object.
114
+ :return: Freshness metadata payload.
115
+ """
116
+ current = _get_kg().stats()
117
+ current_nodes = int(current.get("total_nodes", 0))
118
+ delta = current_nodes - int(snapshot_total_nodes)
119
+
120
+ is_fresh = delta == 0
121
+ status = "fresh" if delta == 0 else ("behind" if delta > 0 else "ahead")
122
+ note = None
123
+
124
+ if 0 < delta < 50:
125
+ is_fresh = True
126
+ status = "near_fresh"
127
+ note = "Within tolerance (sym: stubs often accumulate between rebuilds)"
128
+
129
+ out = {
130
+ "snapshot_total_nodes": int(snapshot_total_nodes),
131
+ "current_total_nodes": current_nodes,
132
+ "delta_nodes": delta,
133
+ "is_fresh": is_fresh,
134
+ "status": status,
135
+ }
136
+ if note:
137
+ out["note"] = note
138
+ return out
139
+
140
+
141
+ # ---------------------------------------------------------------------------
142
+ # MCP server
143
+ # ---------------------------------------------------------------------------
144
+
145
+ mcp = FastMCP(
146
+ "tscodekg",
147
+ instructions=(
148
+ "TypeScriptKG is a hybrid semantic + structural knowledge graph for TypeScript and "
149
+ "JavaScript codebases. It indexes every module, class, interface, type alias, enum, "
150
+ "function, and method as a node, with typed edges "
151
+ "(CALLS, IMPORTS, CONTAINS, INHERITS, IMPLEMENTS, EXTENDS) connecting them.\n\n"
152
+ "## Tools\n\n"
153
+ "**graph_stats()** — Start here. Returns node/edge counts by kind and relation "
154
+ "plus JSDoc coverage. Use before issuing query_codebase() or pack_snippets().\n\n"
155
+ "**query_codebase(q, k, hop, rels, max_nodes, min_score, max_per_module, rerank_mode)** — "
156
+ "Hybrid semantic + structural search. Seeds on vector similarity then expands through "
157
+ "the graph. Returns ranked nodes and edges as JSON.\n\n"
158
+ "**pack_snippets(q, k, hop, rels, context, max_lines, max_nodes, min_score, rerank_mode)** — "
159
+ "Same hybrid search, but returns actual source code as a Markdown context pack "
160
+ "with ranked, deduplicated snippets and line numbers.\n\n"
161
+ "**callers(node_id, rel, paths)** — Precise reverse lookup: every node that calls "
162
+ "(or inherits from, imports, …) the given node, resolving cross-module sym: stubs. "
163
+ "Filter with paths='src/' to exclude test callers.\n\n"
164
+ "**get_node(node_id, include_edges)** — Precise lookup of a single node by its stable "
165
+ "ID (e.g. 'cls:src/auth/middleware.ts:AuthMiddleware').\n\n"
166
+ "**list_nodes(module_path, kind)** — List nodes filtered by module path and/or kind.\n\n"
167
+ "**find_node(name, kind)** — Find nodes by name substring when the stable ID is unknown.\n\n"
168
+ "**centrality(top, kinds, group_by)** — Structural Importance Ranking (SIR): "
169
+ "deterministic weighted PageRank over the graph. group_by='node' ranks individual "
170
+ "nodes; group_by='module' aggregates per module. Use to find hotspots before "
171
+ "refactoring or review.\n\n"
172
+ "**bridge_centrality(top, include_imports)** — Module connectivity: how many unique "
173
+ "modules each module calls or is called by. Identifies orchestrator/hub modules.\n\n"
174
+ "**framework_nodes(top)** — Framework-like hub modules: 0.6 × SIR + 0.4 × "
175
+ "connectivity (both normalized). Surfaces the repo-defining abstractions.\n\n"
176
+ "**find_definition_at(file, line)** — Reverse-resolve a (file, line) location to the "
177
+ "innermost enclosing node and return its explain() report.\n\n"
178
+ "**analyze_repo()** — Structural analysis: node counts, edge counts, JSDoc coverage, "
179
+ "node distribution by kind.\n\n"
180
+ "**explain(node_id, limit)** — Natural-language Markdown explanation of a node: "
181
+ "metadata, JSDoc, callers, callees, and its role in the codebase.\n\n"
182
+ "**rank_nodes(top, rels, persist_metric, exclude_tests)** — Global weighted CodeRank "
183
+ "(PageRank). Returns the most structurally important nodes as JSON.\n\n"
184
+ "**query_ranked(q, k, mode, top, rels, radius, exclude_tests)** — Query ranking that "
185
+ "combines semantic seeds with centrality and graph proximity ('hybrid') or "
186
+ "personalized PageRank ('ppr'); results include 'why' explanations.\n\n"
187
+ "**explain_rank(node_id, q)** — Break down a node's CodeRank score: global rank, "
188
+ "inbound/outbound structural edges, and optional query-conditioned scores.\n\n"
189
+ "**snapshot_list(limit, branch)** — List saved temporal metric snapshots (most recent "
190
+ "first) with per-snapshot deltas and freshness vs. the live graph.\n\n"
191
+ "**snapshot_show(key)** — Full details of one snapshot; pass 'latest' (default) for "
192
+ "the most recent.\n\n"
193
+ "**snapshot_diff(key_a, key_b)** — Side-by-side comparison of two snapshots with "
194
+ "computed deltas (b − a).\n\n"
195
+ "## Recommended Workflows\n\n"
196
+ "- **Explore unfamiliar TS/JS repo**: graph_stats → query_codebase → pack_snippets\n"
197
+ "- **Find a specific class/function**: find_node(name) → get_node(include_edges=True)\n"
198
+ "- **Understand an interface**: get_node → pack_snippets\n"
199
+ "- **Trace usage of a symbol**: find_node(name) → callers(node_id)\n"
200
+ "- **Identify structural hotspots**: centrality(top=20) or centrality(group_by='module')\n"
201
+ "- **Architecture review**: analyze_repo\n"
202
+ "- **Track codebase evolution**: snapshot_list → snapshot_diff(key_a, key_b)\n"
203
+ "- **Answer 'how does X work?'**: pack_snippets with a descriptive query\n"
204
+ ),
205
+ )
206
+
207
+
208
+ @mcp.tool()
209
+ def query_codebase(
210
+ q: str,
211
+ k: int = 8,
212
+ hop: int = 1,
213
+ rels: str = "CONTAINS,CALLS,IMPORTS,INHERITS,IMPLEMENTS,EXTENDS",
214
+ max_nodes: int = 25,
215
+ min_score: float = 0.0,
216
+ max_per_module: int = 3,
217
+ rerank_mode: str = "hybrid",
218
+ rerank_semantic_weight: float = 0.7,
219
+ rerank_lexical_weight: float = 0.3,
220
+ format: str = "json",
221
+ ) -> str:
222
+ """
223
+ Hybrid semantic + structural query over the TypeScript/JavaScript codebase graph.
224
+
225
+ :param q: Natural-language query, e.g. "authentication middleware".
226
+ :param k: Number of semantic seed nodes (default 8).
227
+ :param hop: Graph expansion hops (default 1).
228
+ :param rels: Comma-separated edge types to follow.
229
+ :param max_nodes: Maximum nodes to return (default 25).
230
+ :param min_score: Minimum semantic score for seed inclusion in [0, 1].
231
+ :param max_per_module: Maximum nodes per module (default 3; 0 disables).
232
+ :param rerank_mode: 'hybrid' (default), 'semantic', or 'legacy'.
233
+ :param rerank_semantic_weight: Semantic weight for hybrid mode (default 0.7).
234
+ :param rerank_lexical_weight: Lexical weight for hybrid mode (default 0.3).
235
+ :param format: 'json' (default) or 'markdown'.
236
+ :return: JSON string or Markdown table.
237
+ """
238
+ rel_tuple = tuple(r.strip() for r in rels.split(",") if r.strip())
239
+ result = _get_kg().query(
240
+ q,
241
+ k=k,
242
+ hop=hop,
243
+ rels=rel_tuple or DEFAULT_RELS,
244
+ max_nodes=max_nodes,
245
+ min_score=min_score,
246
+ max_per_module=max_per_module if max_per_module > 0 else None,
247
+ rerank_mode=rerank_mode,
248
+ rerank_semantic_weight=rerank_semantic_weight,
249
+ rerank_lexical_weight=rerank_lexical_weight,
250
+ )
251
+ data = json.loads(result.to_json())
252
+
253
+ if format == "markdown":
254
+ out: list[str] = [
255
+ f"## Query Results: `{q}`\n",
256
+ f"**Seeds:** {data['seeds']} | "
257
+ f"**Expanded:** {data['expanded_nodes']} | "
258
+ f"**Returned:** {data['returned_nodes']}\n",
259
+ "| Rank | Score | Kind | Name | Module |",
260
+ "|-----:|------:|------|------|--------|",
261
+ ]
262
+ for rank_idx, node in enumerate(data["nodes"], start=1):
263
+ score = node.get("relevance", {}).get("score", 0.0)
264
+ kind = node.get("kind", "?")
265
+ name = node.get("qualname") or node.get("name", "?")
266
+ module = node.get("module_path", "")
267
+ out.append(f"| {rank_idx} | {score:.3f} | {kind} | `{name}` | `{module}` |")
268
+ return "\n".join(out)
269
+
270
+ return json.dumps(data, indent=2, ensure_ascii=False)
271
+
272
+
273
+ @mcp.tool()
274
+ def pack_snippets(
275
+ q: str,
276
+ k: int = 8,
277
+ hop: int = 1,
278
+ rels: str = "CONTAINS,CALLS,IMPORTS,INHERITS,IMPLEMENTS,EXTENDS",
279
+ context: int = 5,
280
+ max_lines: int = 60,
281
+ max_nodes: int = 15,
282
+ min_score: float = 0.0,
283
+ max_per_module: int = 3,
284
+ rerank_mode: str = "hybrid",
285
+ rerank_semantic_weight: float = 0.7,
286
+ rerank_lexical_weight: float = 0.3,
287
+ ) -> str:
288
+ """
289
+ Hybrid query + source-grounded TypeScript/JS snippet extraction.
290
+
291
+ Returns a Markdown context pack with ranked, deduplicated code snippets
292
+ and line numbers — ready for direct LLM ingestion.
293
+
294
+ :param q: Natural-language query, e.g. "error handling middleware".
295
+ :param k: Number of semantic seed nodes (default 8).
296
+ :param hop: Graph expansion hops (default 1).
297
+ :param rels: Comma-separated edge types to follow.
298
+ :param context: Extra context lines around each definition (default 5).
299
+ :param max_lines: Maximum lines per snippet block (default 60).
300
+ :param max_nodes: Maximum nodes to include in the pack (default 15).
301
+ :param min_score: Minimum semantic score for seed inclusion in [0, 1].
302
+ :param max_per_module: Maximum nodes per module (default 3; 0 disables).
303
+ :param rerank_mode: 'hybrid' (default), 'semantic', or 'legacy'.
304
+ :param rerank_semantic_weight: Semantic weight for hybrid mode (default 0.7).
305
+ :param rerank_lexical_weight: Lexical weight for hybrid mode (default 0.3).
306
+ :return: Markdown string with source-grounded code snippets.
307
+ """
308
+ rel_tuple = tuple(r.strip() for r in rels.split(",") if r.strip())
309
+ pack = _get_kg().pack(
310
+ q,
311
+ k=k,
312
+ hop=hop,
313
+ rels=rel_tuple or DEFAULT_RELS,
314
+ context=context,
315
+ max_lines=max_lines,
316
+ max_nodes=max_nodes,
317
+ min_score=min_score,
318
+ max_per_module=max_per_module if max_per_module > 0 else None,
319
+ rerank_mode=rerank_mode,
320
+ rerank_semantic_weight=rerank_semantic_weight,
321
+ rerank_lexical_weight=rerank_lexical_weight,
322
+ )
323
+ return pack.to_markdown()
324
+
325
+
326
+ @mcp.tool()
327
+ def callers(node_id: str, rel: str = "CALLS", paths: str = "") -> str:
328
+ """
329
+ Return all nodes that call a given node, resolving through ``sym:`` stubs.
330
+
331
+ Unlike ``query_codebase`` (which seeds on semantics and expands outward),
332
+ this tool performs a precise reverse lookup: it finds every caller of the
333
+ specified node, including cross-module callers that reference it via an
334
+ import alias recorded as a ``sym:`` stub.
335
+
336
+ The ``rel`` parameter accepts any edge relation, not just ``CALLS``::
337
+
338
+ callers(node_id, rel="INHERITS") # find all subclasses
339
+ callers(node_id, rel="IMPLEMENTS") # find all implementations
340
+ callers(node_id, rel="IMPORTS") # find all importers
341
+
342
+ Typical workflow::
343
+
344
+ # 1. Resolve the exact node ID
345
+ get_node("fn:src/utils/helpers.ts:formatDate")
346
+
347
+ # 2. Find all callers (production code only)
348
+ callers("fn:src/utils/helpers.ts:formatDate", paths="src/")
349
+
350
+ :param node_id: Target node identifier, e.g.
351
+ ``cls:src/auth/middleware.ts:AuthMiddleware``.
352
+ :param rel: Relation type to invert (default ``"CALLS"``).
353
+ :param paths: Comma-separated module path prefixes to include, e.g.
354
+ ``"src/"`` to exclude test callers.
355
+ Empty string (default) returns all callers.
356
+ :return: JSON with ``node_id``, ``rel``, ``caller_count``, and
357
+ ``callers`` list of node dicts.
358
+ """
359
+ caller_list = _get_kg().callers(node_id, rel=rel)
360
+ if paths:
361
+ path_prefixes = [p.strip() for p in paths.split(",") if p.strip()]
362
+ caller_list = [
363
+ c
364
+ for c in caller_list
365
+ if any((c.get("module_path") or "").startswith(pfx) for pfx in path_prefixes)
366
+ ]
367
+ return json.dumps(
368
+ {
369
+ "node_id": node_id,
370
+ "rel": rel,
371
+ "caller_count": len(caller_list),
372
+ "callers": caller_list,
373
+ },
374
+ indent=2,
375
+ ensure_ascii=False,
376
+ )
377
+
378
+
379
+ @mcp.tool()
380
+ def get_node(node_id: str, include_edges: bool = False) -> str:
381
+ """
382
+ Fetch a single TypeScript/JS node by its stable ID and render as Markdown.
383
+
384
+ Node IDs follow the pattern ``<kind>:<module_path>:<qualname>``, e.g.
385
+ ``cls:src/auth/middleware.ts:AuthMiddleware`` or
386
+ ``fn:src/utils/helpers.ts:formatDate``.
387
+
388
+ :param node_id: Stable node identifier.
389
+ :param include_edges: If True, append outgoing edges and incoming callers.
390
+ :return: Markdown-formatted node summary.
391
+ """
392
+ kg = _get_kg()
393
+ node = kg.node(node_id)
394
+ if node is None:
395
+ return f"## Node Not Found\n\nNode ID `{node_id}` does not exist in the knowledge graph."
396
+
397
+ kind = node.get("kind", "unknown")
398
+ name = node.get("qualname") or node.get("name", "unknown")
399
+ out: list[str] = [f"## `{name}` ({kind})\n"]
400
+
401
+ module = node.get("module_path", "")
402
+ lineno = node.get("lineno")
403
+ end_lineno = node.get("end_lineno")
404
+ if module:
405
+ out.append(f"- **Module:** `{module}`")
406
+ if lineno is not None:
407
+ loc = f"line {lineno}"
408
+ if end_lineno:
409
+ loc += f"–{end_lineno}"
410
+ out.append(f"- **Location:** {loc}")
411
+ out.append(f"- **ID:** `{node_id}`")
412
+ out.append("")
413
+
414
+ docstring = node.get("docstring", "").strip()
415
+ if docstring:
416
+ out.append("### JSDoc\n")
417
+ out.append(docstring)
418
+ out.append("")
419
+
420
+ if not include_edges:
421
+ return "\n".join(out)
422
+
423
+ store = getattr(kg, "_store", None)
424
+ if store is not None:
425
+ for rel in ("CALLS", "CONTAINS", "IMPORTS", "INHERITS", "IMPLEMENTS", "EXTENDS"):
426
+ edges = store.edges_from(node_id, rel=rel)
427
+ visible = [e for e in edges if not e["dst"].startswith("sym:")] if edges else []
428
+ if visible:
429
+ out.append(f"### Outgoing {rel}\n")
430
+ for e in visible:
431
+ out.append(f"- `{e['dst']}`")
432
+ out.append("")
433
+
434
+ try:
435
+ caller_nodes = kg.callers(node_id, rel="CALLS")
436
+ if caller_nodes:
437
+ out.append("### Incoming Calls\n")
438
+ for c in caller_nodes:
439
+ cname = c.get("qualname") or c.get("name", "")
440
+ cmod = c.get("module_path", "")
441
+ cline = c.get("lineno")
442
+ cid = c.get("id", "")
443
+ loc_str = f" (line {cline})" if cline else ""
444
+ out.append(f"- `{cid}` — `{cname}` in `{cmod}`{loc_str}")
445
+ out.append("")
446
+ except (AttributeError, ValueError, RuntimeError):
447
+ pass
448
+
449
+ return "\n".join(out)
450
+
451
+
452
+ @mcp.tool()
453
+ def graph_stats() -> str:
454
+ """
455
+ Return node and edge counts by kind and relation as Markdown.
456
+
457
+ Call this first when engaging with a new TypeScript/JavaScript repo.
458
+ Reports JSDoc coverage (fraction of functions/methods with JSDoc comments).
459
+
460
+ :return: Markdown summary with total counts, nodes-by-kind, and edges-by-relation tables.
461
+ """
462
+ stats = _get_kg().stats()
463
+ out: list[str] = ["## TypeScriptKG Graph Statistics\n"]
464
+ out.append(f"- **Database:** `{stats.get('db_path', '')}`")
465
+ out.append(f"- **Total nodes:** {stats.get('total_nodes', 0):,}")
466
+ out.append(
467
+ f"- **Meaningful nodes:** {stats.get('meaningful_nodes', 0):,} *(excludes sym: stubs)*"
468
+ )
469
+ out.append(f"- **Total edges:** {stats.get('total_edges', 0):,}")
470
+ cov = stats.get("docstring_coverage")
471
+ if cov is not None:
472
+ out.append(f"- **JSDoc coverage:** {cov:.1%} *(functions + methods)*")
473
+ out.append("")
474
+
475
+ node_counts: dict = stats.get("node_counts", {})
476
+ if node_counts:
477
+ out.append("### Nodes by Kind\n")
478
+ out.append("| Kind | Count |")
479
+ out.append("|------|------:|")
480
+ for kind, count in sorted(node_counts.items(), key=lambda x: -x[1]):
481
+ out.append(f"| {kind} | {count:,} |")
482
+ out.append("")
483
+
484
+ edge_counts: dict = stats.get("edge_counts", {})
485
+ if edge_counts:
486
+ out.append("### Edges by Relation\n")
487
+ out.append("| Relation | Count |")
488
+ out.append("|----------|------:|")
489
+ for rel, count in sorted(edge_counts.items(), key=lambda x: -x[1]):
490
+ out.append(f"| {rel} | {count:,} |")
491
+ out.append("")
492
+
493
+ out.append(
494
+ "> `sym:` nodes are import stub placeholders for external packages — "
495
+ "they are not local code entities."
496
+ )
497
+ return "\n".join(out)
498
+
499
+
500
+ @mcp.tool()
501
+ def list_nodes(
502
+ module_path: str = "",
503
+ kind: str = "",
504
+ ) -> str:
505
+ """
506
+ List nodes filtered by module path prefix and/or kind.
507
+
508
+ :param module_path: Module path prefix filter (e.g. "src/auth/middleware.ts").
509
+ :param kind: Node kind filter: module | class | interface | type_alias | enum |
510
+ namespace | function | method.
511
+ :return: JSON array of matching node dicts.
512
+ """
513
+ kg = _get_kg()
514
+ store = getattr(kg, "_store", None)
515
+ if not store:
516
+ return json.dumps({"error": "No database store available."}, indent=2)
517
+
518
+ q = "SELECT id, name, qualname, kind, module_path, lineno, docstring FROM nodes WHERE 1=1"
519
+ q += " AND id NOT LIKE 'sym:%'"
520
+ params = []
521
+
522
+ if module_path:
523
+ q += " AND module_path LIKE ?"
524
+ params.append(f"{module_path}%")
525
+ if kind:
526
+ q += " AND kind = ?"
527
+ params.append(kind)
528
+
529
+ q += " ORDER BY module_path, lineno"
530
+
531
+ try:
532
+ rows = store.con.execute(q, params).fetchall()
533
+ result = []
534
+ for r in rows:
535
+ doc = r[6]
536
+ if doc and len(doc) > 120:
537
+ doc = doc[:120] + "..."
538
+ result.append(
539
+ {
540
+ "id": r[0],
541
+ "name": r[1],
542
+ "qualname": r[2],
543
+ "kind": r[3],
544
+ "module_path": r[4],
545
+ "lineno": r[5],
546
+ "docstring": doc,
547
+ }
548
+ )
549
+ return json.dumps(result, indent=2, ensure_ascii=False)
550
+ except Exception as e: # pylint: disable=broad-except
551
+ return json.dumps({"error": str(e)}, indent=2)
552
+
553
+
554
+ @mcp.tool()
555
+ def find_node(name: str, kind: str = "") -> str:
556
+ """
557
+ Find graph nodes by name without knowing their full stable ID.
558
+
559
+ Case-insensitive match against name and qualname. Use when you know a
560
+ function or class name from reading code and need its stable ID.
561
+
562
+ :param name: Function, class, or interface name to search for.
563
+ :param kind: Optional kind filter: module | class | interface | function | method | etc.
564
+ :return: JSON array of matching node dicts.
565
+ """
566
+ kg = _get_kg()
567
+ store = getattr(kg, "_store", None)
568
+ if not store:
569
+ return json.dumps({"error": "No database store available."}, indent=2)
570
+
571
+ name_lower = name.lower()
572
+ q = (
573
+ "SELECT id, name, qualname, kind, module_path, lineno, docstring "
574
+ "FROM nodes WHERE (LOWER(name) = ? OR LOWER(qualname) LIKE ?)"
575
+ " AND id NOT LIKE 'sym:%'"
576
+ )
577
+ params: list = [name_lower, f"%{name_lower}%"]
578
+ if kind:
579
+ q += " AND kind = ?"
580
+ params.append(kind)
581
+ q += " ORDER BY module_path, lineno"
582
+
583
+ try:
584
+ rows = store.con.execute(q, params).fetchall()
585
+ result = []
586
+ for r in rows:
587
+ doc = r[6]
588
+ if doc and len(doc) > 120:
589
+ doc = doc[:120] + "..."
590
+ result.append(
591
+ {
592
+ "id": r[0],
593
+ "name": r[1],
594
+ "qualname": r[2],
595
+ "kind": r[3],
596
+ "module_path": r[4],
597
+ "lineno": r[5],
598
+ "docstring": doc,
599
+ }
600
+ )
601
+ return json.dumps(result, indent=2, ensure_ascii=False)
602
+ except Exception as e: # pylint: disable=broad-except
603
+ return json.dumps({"error": str(e)}, indent=2)
604
+
605
+
606
+ @mcp.tool()
607
+ def centrality(
608
+ top: int = 20,
609
+ kinds: str = "",
610
+ group_by: str = "node",
611
+ ) -> str:
612
+ """
613
+ Compute Structural Importance Ranking (SIR) for the indexed codebase.
614
+
615
+ Runs a deterministic weighted PageRank over the sym-stub-resolved call
616
+ graph. Edge weights are tuned per relation type
617
+ (CALLS > INHERITS/IMPLEMENTS > IMPORTS > CONTAINS) and amplified for
618
+ cross-module links; private symbols receive a post-convergence penalty.
619
+ Scores are normalized to sum to 1.0.
620
+
621
+ Use this to:
622
+
623
+ - Identify the most structurally critical functions, classes, and interfaces
624
+ - Understand which modules are most depended upon
625
+ - Prioritize code review, refactoring, or test coverage efforts
626
+
627
+ :param top: Maximum number of ranked entries to return (default 20).
628
+ :param kinds: Comma-separated node kinds to include: ``module``, ``class``,
629
+ ``interface``, ``function``, ``method``. Empty string returns
630
+ all kinds. Ignored when ``group_by='module'`` (all kinds
631
+ contribute to module aggregation).
632
+ :param group_by: ``node`` (default) returns individual node rankings with
633
+ score, inbound edge count, and cross-module inbound count;
634
+ ``module`` aggregates node scores per module.
635
+ :return: Markdown-formatted ranking table.
636
+ """
637
+ try:
638
+ from tscode_kg.centrality import ( # noqa: PLC0415
639
+ StructuralImportanceRanker,
640
+ aggregate_module_scores,
641
+ )
642
+
643
+ db_path = _get_kg().db_path
644
+ ranker = StructuralImportanceRanker(db_path)
645
+ all_records = ranker.compute()
646
+ except Exception as e: # noqa: BLE001
647
+ return f"## Centrality Error\n\nFailed to compute SIR scores: `{e}`"
648
+
649
+ out: list[str] = ["## Structural Importance Ranking (SIR)\n"]
650
+
651
+ if group_by == "module":
652
+ payload = aggregate_module_scores(all_records)[:top]
653
+ out.append(f"**Group by:** module | **Top:** {top}\n")
654
+ out.append("| Rank | Score | Members | Module |")
655
+ out.append("|-----:|------:|--------:|--------|")
656
+ for row in payload:
657
+ out.append(
658
+ f"| {row['rank']} | {row['score']:.6f}"
659
+ f" | {row['member_count']} | `{row['module_path']}` |"
660
+ )
661
+ else:
662
+ kind_set: set[str] | None = None
663
+ if kinds.strip():
664
+ kind_set = {k.strip().lower() for k in kinds.split(",") if k.strip()}
665
+
666
+ filtered = [r for r in all_records if kind_set is None or r.kind in kind_set][:top]
667
+ label = kinds if kind_set else "all kinds"
668
+ out.append(f"**Group by:** node | **Top:** {top} | **Filter:** {label}\n")
669
+ out.append("| Rank | Score | Kind | Name | Module | Inbound | XMod |")
670
+ out.append("|-----:|------:|------|------|--------|--------:|-----:|")
671
+ for r in filtered:
672
+ module = f"`{r.module_path}`" if r.module_path else "—"
673
+ out.append(
674
+ f"| {r.rank} | {r.score:.6f} | {r.kind} | `{r.name}`"
675
+ f" | {module} | {r.inbound_count} | {r.cross_module_inbound} |"
676
+ )
677
+
678
+ out.append("")
679
+ out.append(
680
+ "> SIR scores are normalized to sum 1.0 across all nodes. "
681
+ "Higher score = more structurally central. "
682
+ "XMod = cross-module inbound edges."
683
+ )
684
+ return "\n".join(out)
685
+
686
+
687
+ @mcp.tool()
688
+ def bridge_centrality(
689
+ top: int = 20,
690
+ include_imports: bool = True,
691
+ ) -> str:
692
+ """
693
+ Compute module connectivity: how many unique modules each module interacts with.
694
+
695
+ For well-modularized codebases, identifies orchestrator and hub modules that
696
+ touch many other modules. Replaces betweenness centrality (which is meaningless
697
+ when inter-module edges are zero).
698
+
699
+ **Connectivity score** = (unique modules called + unique modules calling this) / 30 + frequency / 50
700
+ Higher score = more complex coupling with other modules.
701
+
702
+ Scores are persisted to the ``centrality_scores`` table under the
703
+ ``module_connectivity`` metric for use by ``framework_nodes()``.
704
+
705
+ :param top: Number of top connectivity modules to return (default 20).
706
+ :param include_imports: Whether to include IMPORTS in connectivity (default True).
707
+ :return: Markdown-formatted ranking table of modules by connectivity.
708
+ """
709
+ try:
710
+ from tscode_kg.bridge import compute_bridge_centrality # noqa: PLC0415
711
+
712
+ db_path = str(_get_kg().db_path)
713
+ modules = compute_bridge_centrality(
714
+ kind="module",
715
+ include_imports=include_imports,
716
+ top=top,
717
+ db_path=db_path,
718
+ )
719
+ except Exception as e: # noqa: BLE001
720
+ return f"## Module Connectivity Error\n\nFailed to compute connectivity: `{e}`"
721
+
722
+ out: list[str] = ["## Module Connectivity (Interaction Complexity)\n"]
723
+ out.append(f"**Top:** {top} | **Include imports:** {include_imports}\n")
724
+ out.append("| Rank | Connectivity | Module |")
725
+ out.append("|-----:|-------------:|--------|")
726
+ for rank_idx, (mod, score) in enumerate(modules, start=1):
727
+ out.append(f"| {rank_idx} | {score:.6f} | `{mod}` |")
728
+ out.append("")
729
+ out.append(
730
+ "> Connectivity = unique modules called + unique modules calling this module. "
731
+ "Higher score = orchestrator/hub module with complex interactions. "
732
+ "Scores are persisted as `module_connectivity` metric for use by `framework_nodes()`."
733
+ )
734
+ return "\n".join(out)
735
+
736
+
737
+ @mcp.tool()
738
+ def framework_nodes(top: int = 20) -> str:
739
+ """
740
+ Identify framework-like (hub) modules using SIR + module connectivity.
741
+
742
+ A "framework node" is a module that is both:
743
+ - Structurally important (high SIR/PageRank — central to the graph)
744
+ - Highly connected (calls/imports many modules — orchestrator/hub role)
745
+
746
+ Framework score = 0.6 × normalized SIR + 0.4 × normalized connectivity,
747
+ both auto-computed on first call. High-scoring modules are critical hubs:
748
+ architecturally central AND complex in their interactions.
749
+
750
+ :param top: Number of top framework-like modules to return (default 20).
751
+ :return: Markdown-formatted ranking table of framework nodes.
752
+ """
753
+ try:
754
+ from tscode_kg.bridge import compute_bridge_centrality # noqa: PLC0415
755
+ from tscode_kg.centrality import StructuralImportanceRanker # noqa: PLC0415
756
+ from tscode_kg.framework_detector import detect_framework_nodes # noqa: PLC0415
757
+
758
+ kg = _get_kg()
759
+ db_path = str(kg.db_path)
760
+
761
+ # Compute and persist SIR scores (structural importance)
762
+ try:
763
+ ranker = StructuralImportanceRanker(db_path)
764
+ records = ranker.compute()
765
+ ranker.write_scores(records, metric="sir_pagerank")
766
+ except Exception as e: # noqa: BLE001
767
+ return f"## Framework Nodes Error\n\nFailed to compute SIR scores: `{e}`"
768
+
769
+ # Compute and persist module connectivity scores (interaction complexity)
770
+ try:
771
+ compute_bridge_centrality(kind="module", include_imports=True, top=25, db_path=db_path)
772
+ except Exception as e: # noqa: BLE001
773
+ return f"## Framework Nodes Error\n\nFailed to compute module connectivity: `{e}`"
774
+
775
+ # Detect framework nodes by combining both metrics
776
+ nodes = detect_framework_nodes(limit=top, db_path=db_path)
777
+ except Exception as e: # noqa: BLE001
778
+ return f"## Framework Nodes Error\n\nFailed to detect framework nodes: `{e}`"
779
+
780
+ out: list[str] = ["## Framework-like Modules (Critical Hubs)\n"]
781
+ out.append(f"**Top:** {top} | **Score:** 0.6 × SIR + 0.4 × connectivity (both normalized)\n")
782
+ out.append("| Rank | Score | Module |")
783
+ out.append("|-----:|------:|--------|")
784
+ for rank_idx, (_, score, label) in enumerate(nodes, start=1):
785
+ out.append(f"| {rank_idx} | {score:.6f} | `{label}` |")
786
+ out.append("")
787
+ out.append(
788
+ "> Framework nodes: both architecturally central (SIR) AND heavily connected "
789
+ "(calls/imports many modules). High-scoring modules are critical orchestrators/hubs."
790
+ )
791
+ return "\n".join(out)
792
+
793
+
794
+ @mcp.tool()
795
+ def find_definition_at(file: str, line: int) -> str:
796
+ """
797
+ Find the code node whose definition spans a given file location.
798
+
799
+ Reverse-resolves a ``(file, line)`` pair to a graph node ID and returns the
800
+ same Markdown report as ``explain()``. Useful when reading a file in an IDE
801
+ and wanting to understand the symbol at a specific line without constructing
802
+ a node ID manually.
803
+
804
+ Matches the innermost (most-specific) function, method, class, interface,
805
+ type alias, or enum whose ``lineno ≤ line ≤ end_lineno``. Falls back to
806
+ the module node when no narrower match exists.
807
+
808
+ :param file: Module path as stored in the graph, e.g. ``src/auth/middleware.ts``.
809
+ Leading ``./`` is stripped automatically.
810
+ :param line: Line number (1-indexed) within the file.
811
+ :return: Markdown explanation from ``explain()``, or an informative error
812
+ message if no node spans that location.
813
+ """
814
+ kg = _get_kg()
815
+ store = getattr(kg, "_store", None) or getattr(kg, "store", None)
816
+ if store is None:
817
+ return "## Error\n\nNo graph store available."
818
+
819
+ norm_file = file.lstrip("./")
820
+
821
+ # Innermost span: smallest (end_lineno - lineno) that still contains `line`.
822
+ rows = store.con.execute(
823
+ """
824
+ SELECT id
825
+ FROM nodes
826
+ WHERE (module_path = :f OR module_path LIKE :like)
827
+ AND kind IN ('function', 'method', 'class', 'interface', 'type_alias', 'enum')
828
+ AND lineno IS NOT NULL
829
+ AND lineno <= :ln
830
+ AND (end_lineno IS NULL OR end_lineno >= :ln)
831
+ ORDER BY (COALESCE(end_lineno, lineno) - lineno) ASC
832
+ LIMIT 1
833
+ """,
834
+ {"f": norm_file, "like": f"%{norm_file}", "ln": line},
835
+ ).fetchall()
836
+
837
+ if not rows:
838
+ # Fall back to the module node itself
839
+ mod_rows = store.con.execute(
840
+ "SELECT id FROM nodes WHERE kind = 'module' AND (module_path = ? OR module_path LIKE ?)",
841
+ (norm_file, f"%{norm_file}"),
842
+ ).fetchall()
843
+ if not mod_rows:
844
+ return (
845
+ f"## No Definition Found\n\n"
846
+ f"No function, method, class, interface, type alias, or enum spans "
847
+ f"`{file}:{line}` in the graph.\n\n"
848
+ "Check that the file path matches the module path stored in the graph "
849
+ "(use `graph_stats()` or `list_nodes()` to browse available modules)."
850
+ )
851
+ node_id = mod_rows[0][0]
852
+ else:
853
+ node_id = rows[0][0]
854
+
855
+ return explain(node_id)
856
+
857
+
858
+ @mcp.tool()
859
+ def analyze_repo() -> str:
860
+ """
861
+ Run a full structural analysis of the indexed TypeScript/JavaScript repository.
862
+
863
+ Executes the 14-phase TypeScriptKG analysis pipeline — baseline metrics,
864
+ CodeRank, fan-in/fan-out, module coupling, critical call chains, public API
865
+ surface, JSDoc coverage, class/interface hierarchy, insights, snapshot
866
+ history, and SIR centrality — and returns the results as Markdown.
867
+
868
+ :return: Markdown-formatted analysis report.
869
+ """
870
+ from io import StringIO # noqa: PLC0415
871
+
872
+ from rich.console import Console # noqa: PLC0415
873
+
874
+ from tscode_kg.analysis import TSCodeKGAnalyzer # noqa: PLC0415
875
+ from tscode_kg.kg import _render_analysis # noqa: PLC0415
876
+
877
+ # Silence Rich output — stdout carries the MCP protocol on stdio transport.
878
+ silent = Console(file=StringIO(), highlight=False)
879
+ kg = _get_kg()
880
+ try:
881
+ analyzer = TSCodeKGAnalyzer(kg, console=silent, snapshot_mgr=_snapshot_mgr)
882
+ analyzer.run_analysis()
883
+ return analyzer.to_markdown()
884
+ except Exception as exc: # noqa: BLE001
885
+ # Lightweight stats-only fallback — never re-runs the noisy analyzer.
886
+ try:
887
+ return _render_analysis(str(kg.repo_root), kg.store.stats())
888
+ except Exception: # noqa: BLE001
889
+ return f"# TypeScriptKG Analysis\n\nAnalysis failed: {exc}\n"
890
+
891
+
892
+ @mcp.tool()
893
+ def explain(node_id: str, limit: int = 10) -> str:
894
+ """
895
+ Return a natural-language explanation of a code node.
896
+
897
+ Given a node ID (e.g., ``fn:src/utils/helpers.ts:formatDate``),
898
+ returns a markdown-formatted explanation that includes:
899
+
900
+ - **What it is**: The node's kind, short description from its JSDoc
901
+ - **Where it lives**: Module path and source location
902
+ - **What calls it**: The callers (reverse call graph)
903
+ - **What it calls**: The callees (functions/methods this node invokes)
904
+ - **Documentation**: Full JSDoc if available
905
+
906
+ This is ideal for understanding the role and context of a specific node
907
+ without needing to read the full source code. Use ``pack_snippets()``
908
+ to then retrieve the actual implementation.
909
+
910
+ :param node_id: Stable node identifier, e.g.
911
+ ``fn:src/utils/helpers.ts:formatDate``.
912
+ :param limit: Maximum callers and callees to list (default 10). Pass 0
913
+ to list all.
914
+ :return: Markdown-formatted explanation ready for LLM consumption.
915
+ """
916
+ from tscode_kg.explain import render_explain # noqa: PLC0415
917
+
918
+ return render_explain(
919
+ _get_kg(),
920
+ node_id,
921
+ limit=limit,
922
+ snippets_hint="pack_snippets()",
923
+ )
924
+
925
+
926
+ @mcp.tool()
927
+ def rank_nodes(
928
+ top: int = 25,
929
+ rels: str = "CALLS,IMPORTS,INHERITS,IMPLEMENTS,EXTENDS",
930
+ persist_metric: str = "",
931
+ exclude_tests: bool = True,
932
+ ) -> str:
933
+ """
934
+ Compute global weighted CodeRank (PageRank) over the repository graph.
935
+
936
+ Builds a directed weighted graph from the SQLite store and runs weighted
937
+ PageRank to identify the most structurally important nodes. Relation
938
+ weights follow the CodeRank defaults: CALLS=1.0, IMPORTS=0.9,
939
+ INHERITS/IMPLEMENTS/EXTENDS=0.75. Test paths are excluded by default.
940
+
941
+ Optionally persists the scores into the ``node_metrics`` table under the
942
+ given metric name so they can be loaded at query time without recomputing.
943
+
944
+ :param top: Number of top-ranked nodes to return (default 25).
945
+ :param rels: Comma-separated relations to include in the graph
946
+ (default ``"CALLS,IMPORTS,INHERITS,IMPLEMENTS,EXTENDS"``).
947
+ :param persist_metric: If non-empty, persist scores to ``node_metrics``
948
+ under this metric name (e.g. ``"coderank_global"``).
949
+ :param exclude_tests: Exclude test-path nodes from the graph (default True).
950
+ :return: JSON array of ranked node dicts with ``node_id``, ``score``,
951
+ ``top_pct`` (e.g. ``"top 0.5%"``), ``kind``, ``qualname``,
952
+ ``module_path``, and ``rank`` fields.
953
+ """
954
+ from tscode_kg.coderank import ( # noqa: PLC0415
955
+ build_code_graph,
956
+ compute_coderank,
957
+ persist_metric_scores,
958
+ )
959
+
960
+ db_path = str(_get_kg().db_path)
961
+ rel_list = [r.strip() for r in rels.split(",") if r.strip()]
962
+
963
+ try:
964
+ graph = build_code_graph(
965
+ db_path,
966
+ include_relations=rel_list,
967
+ exclude_test_paths=exclude_tests,
968
+ )
969
+ scores = compute_coderank(graph)
970
+ except Exception as exc: # noqa: BLE001
971
+ return json.dumps({"error": str(exc)}, indent=2)
972
+
973
+ if persist_metric:
974
+ try:
975
+ persist_metric_scores(db_path, persist_metric, scores)
976
+ except Exception: # noqa: BLE001
977
+ pass # non-fatal — still return results
978
+
979
+ # Filter out sym: stubs (import placeholders) — only return real code entities
980
+ all_real_nodes = [
981
+ (nid, s)
982
+ for nid, s in sorted(scores.items(), key=lambda kv: kv[1], reverse=True)
983
+ if not nid.startswith("sym:")
984
+ ]
985
+ total_real = len(all_real_nodes)
986
+ results = []
987
+ for rank_idx, (node_id, score) in enumerate(all_real_nodes[:top], start=1):
988
+ attrs = graph.nodes.get(node_id, {})
989
+ top_pct = round(rank_idx / total_real * 100, 1) if total_real > 0 else 0.0
990
+ results.append(
991
+ {
992
+ "rank": rank_idx,
993
+ "node_id": node_id,
994
+ "score": round(score, 8),
995
+ "top_pct": f"top {top_pct:.1f}%",
996
+ "kind": attrs.get("kind"),
997
+ "qualname": attrs.get("qualname"),
998
+ "module_path": attrs.get("module_path"),
999
+ }
1000
+ )
1001
+
1002
+ return json.dumps(results, indent=2, ensure_ascii=False)
1003
+
1004
+
1005
+ @mcp.tool()
1006
+ def query_ranked(
1007
+ q: str,
1008
+ k: int = 8,
1009
+ mode: str = "hybrid",
1010
+ top: int = 25,
1011
+ rels: str = "CALLS,IMPORTS,INHERITS,IMPLEMENTS,EXTENDS",
1012
+ radius: int = 2,
1013
+ exclude_tests: bool = True,
1014
+ ) -> str:
1015
+ """
1016
+ Rank query results using CodeRank-enhanced hybrid or personalized PageRank.
1017
+
1018
+ Combines semantic seed scores from the vector index with structural
1019
+ centrality and graph proximity to produce a final ranked list with
1020
+ explainability components.
1021
+
1022
+ Two modes are available:
1023
+
1024
+ - ``hybrid`` (default): 0.60 × semantic + 0.25 × centrality + 0.15 × proximity
1025
+ - ``ppr``: 0.70 × personalized PageRank + 0.30 × semantic
1026
+
1027
+ :param q: Natural-language query string.
1028
+ :param k: Number of semantic seed nodes to retrieve (default 8).
1029
+ :param mode: Ranking mode — ``"hybrid"`` (default) or ``"ppr"``.
1030
+ :param top: Maximum ranked results to return (default 25).
1031
+ :param rels: Comma-separated relations to include in the local graph.
1032
+ :param radius: Graph expansion radius around seeds (default 2).
1033
+ :param exclude_tests: Exclude test-path nodes (default True).
1034
+ :return: JSON array of ranked result dicts with score components and
1035
+ ``why`` explanation strings. ``sym:`` import stub nodes are
1036
+ always excluded from the output.
1037
+ """
1038
+ from tscode_kg.coderank import ( # noqa: PLC0415
1039
+ build_code_graph,
1040
+ compute_coderank,
1041
+ rank_query_hybrid,
1042
+ rank_query_ppr,
1043
+ )
1044
+
1045
+ kg = _get_kg()
1046
+ db_path = str(kg.db_path)
1047
+ rel_list = [r.strip() for r in rels.split(",") if r.strip()]
1048
+
1049
+ # Get semantic seeds from the vector index
1050
+ try:
1051
+ raw = kg.query(q, k=k, hop=0, rels=tuple(rel_list))
1052
+ seed_data = json.loads(raw.to_json())
1053
+ seed_nodes = seed_data.get("nodes", [])
1054
+ semantic_scores: dict[str, float] = {
1055
+ n["id"]: float((n.get("relevance") or {}).get("score", 0.0))
1056
+ for n in seed_nodes
1057
+ if (n.get("relevance") or {}).get("score", 0.0) > 0
1058
+ }
1059
+ except Exception as exc: # noqa: BLE001
1060
+ return json.dumps({"error": f"Seed retrieval failed: {exc}"}, indent=2)
1061
+
1062
+ if not semantic_scores:
1063
+ return json.dumps({"error": "No semantic seeds found for query."}, indent=2)
1064
+
1065
+ try:
1066
+ graph = build_code_graph(
1067
+ db_path,
1068
+ include_relations=rel_list,
1069
+ exclude_test_paths=exclude_tests,
1070
+ )
1071
+ except Exception as exc: # noqa: BLE001
1072
+ return json.dumps({"error": f"Graph build failed: {exc}"}, indent=2)
1073
+
1074
+ global_cr = compute_coderank(graph)
1075
+ try:
1076
+ if mode == "ppr":
1077
+ results = rank_query_ppr(graph, semantic_scores, radius=radius, top_k=top)
1078
+ else:
1079
+ results = rank_query_hybrid(
1080
+ graph, semantic_scores, global_coderank=global_cr, radius=radius, top_k=top
1081
+ )
1082
+ except Exception as exc: # noqa: BLE001
1083
+ return json.dumps({"error": f"Ranking failed: {exc}"}, indent=2)
1084
+
1085
+ output = []
1086
+ for rank_idx, r in enumerate(results, start=1):
1087
+ if r.node_id.startswith("sym:"):
1088
+ continue
1089
+ output.append(
1090
+ {
1091
+ "rank": rank_idx,
1092
+ "node_id": r.node_id,
1093
+ "adjusted_score": round(r.adjusted_score, 6),
1094
+ "final_score": round(r.final_score, 6),
1095
+ "semantic_score": round(r.semantic_score, 6),
1096
+ "centrality_score": round(r.centrality_score, 6),
1097
+ "proximity_score": round(r.proximity_score, 6),
1098
+ "kind": r.kind,
1099
+ "qualname": r.qualname,
1100
+ "module_path": r.module_path,
1101
+ "why": list(r.why),
1102
+ }
1103
+ )
1104
+
1105
+ return json.dumps(
1106
+ {"query": q, "mode": mode, "returned": len(output), "results": output},
1107
+ indent=2,
1108
+ ensure_ascii=False,
1109
+ )
1110
+
1111
+
1112
+ @mcp.tool()
1113
+ def explain_rank(node_id: str, q: str = "") -> str:
1114
+ """
1115
+ Explain the CodeRank score components for a specific node.
1116
+
1117
+ Returns a Markdown report showing the node's structural position in the
1118
+ graph: how many nodes call it, import it, or inherit from / implement /
1119
+ extend it; its global CodeRank score; and, when a query is provided, its
1120
+ semantic relevance and proximity to the query seed set.
1121
+
1122
+ :param node_id: Stable node identifier, e.g.
1123
+ ``fn:src/utils/helpers.ts:formatDate``.
1124
+ :param q: Optional query string. When provided, semantic score and
1125
+ proximity to the query seed set are included in the report.
1126
+ :return: Markdown-formatted explanation of the node's rank components.
1127
+ """
1128
+ from tscode_kg.coderank import ( # noqa: PLC0415
1129
+ DEFAULT_GLOBAL_RELS,
1130
+ build_code_graph,
1131
+ compute_coderank,
1132
+ compute_seed_proximity,
1133
+ )
1134
+
1135
+ kg = _get_kg()
1136
+ db_path = str(kg.db_path)
1137
+
1138
+ node = kg.node(node_id)
1139
+ if node is None:
1140
+ return f"## Node Not Found\n\nNode ID `{node_id}` does not exist."
1141
+
1142
+ kind = node.get("kind", "unknown")
1143
+ name = node.get("qualname") or node.get("name", "unknown")
1144
+ out: list[str] = [f"## CodeRank Explanation: `{name}` ({kind})\n"]
1145
+ out.append(f"- **ID:** `{node_id}`")
1146
+ if node.get("module_path"):
1147
+ out.append(f"- **Module:** `{node['module_path']}`")
1148
+ out.append("")
1149
+
1150
+ # Build graph and compute global CodeRank
1151
+ try:
1152
+ graph = build_code_graph(
1153
+ db_path,
1154
+ include_relations=list(DEFAULT_GLOBAL_RELS),
1155
+ exclude_test_paths=True,
1156
+ )
1157
+ scores = compute_coderank(graph)
1158
+ except Exception as exc: # noqa: BLE001
1159
+ return f"## Error\n\nFailed to build graph: `{exc}`"
1160
+
1161
+ global_score = scores.get(node_id, 0.0)
1162
+ meaningful_scores = sorted(
1163
+ (v for k, v in scores.items() if not k.startswith("sym:")), reverse=True
1164
+ )
1165
+ rank_pos = next(
1166
+ (i + 1 for i, s in enumerate(meaningful_scores) if s <= global_score),
1167
+ len(meaningful_scores),
1168
+ )
1169
+
1170
+ out.append("### Global CodeRank\n")
1171
+ out.append(f"- **Score:** `{global_score:.8f}`")
1172
+ out.append(f"- **Rank:** #{rank_pos} of {len(meaningful_scores)} meaningful nodes")
1173
+ out.append("")
1174
+
1175
+ # Structural context from graph
1176
+ if node_id in graph:
1177
+ in_edges = list(graph.in_edges(node_id, data=True))
1178
+ out.append("### Structural Inbound Edges\n")
1179
+ callers_count = sum(1 for _, _, d in in_edges if "CALLS" in d.get("relations", set()))
1180
+ importers_count = sum(1 for _, _, d in in_edges if "IMPORTS" in d.get("relations", set()))
1181
+ inheritors_count = sum(
1182
+ 1
1183
+ for _, _, d in in_edges
1184
+ if d.get("relations", set()) & {"INHERITS", "IMPLEMENTS", "EXTENDS"}
1185
+ )
1186
+ if callers_count:
1187
+ out.append(f"- Called by **{callers_count}** upstream node(s)")
1188
+ if importers_count:
1189
+ out.append(f"- Imported by **{importers_count}** upstream node(s)")
1190
+ if inheritors_count:
1191
+ out.append(
1192
+ f"- Inherited/implemented/extended by **{inheritors_count}** downstream node(s)"
1193
+ )
1194
+ if not (callers_count or importers_count or inheritors_count):
1195
+ out.append("- No inbound structural edges found in the ranked graph")
1196
+ out.append("")
1197
+
1198
+ out_edges = list(graph.out_edges(node_id, data=True))
1199
+ if out_edges:
1200
+ out.append("### Structural Outbound Edges\n")
1201
+ out.append(f"- Calls/imports/inherits **{len(out_edges)}** downstream node(s)")
1202
+ out.append("")
1203
+
1204
+ # Optional query-conditioned scores
1205
+ if q:
1206
+ out.append("### Query-Conditioned Scores\n")
1207
+ try:
1208
+ raw = kg.query(q, k=8, hop=0)
1209
+ seed_data = json.loads(raw.to_json())
1210
+ seed_nodes = seed_data.get("nodes", [])
1211
+ semantic_scores: dict[str, float] = {
1212
+ n["id"]: float((n.get("relevance") or {}).get("score", 0.0)) for n in seed_nodes
1213
+ }
1214
+ this_semantic = semantic_scores.get(node_id, 0.0)
1215
+ out.append(f"- **Query:** `{q}`")
1216
+ out.append(f"- **Semantic score:** `{this_semantic:.4f}`")
1217
+
1218
+ if node_id in graph:
1219
+ seeds = list(semantic_scores.keys())
1220
+ proximity = compute_seed_proximity(graph, seeds)
1221
+ prox = proximity.get(node_id, 0.0)
1222
+ out.append(f"- **Proximity to seeds:** `{prox:.4f}`")
1223
+ if prox >= 1.0:
1224
+ out.append(" → Direct semantic seed")
1225
+ elif prox >= 0.5:
1226
+ out.append(" → One hop from a semantic seed")
1227
+ elif prox > 0:
1228
+ out.append(" → Within local query neighborhood")
1229
+ else:
1230
+ out.append(" → Outside query neighborhood")
1231
+ except Exception as exc: # noqa: BLE001
1232
+ out.append(f"- Query scoring failed: `{exc}`")
1233
+ out.append("")
1234
+
1235
+ out.append("---\n")
1236
+ out.append(
1237
+ "*Use `rank_nodes()` for global top-N ranking, or `query_ranked()` for query-conditioned ranking.*"
1238
+ )
1239
+ return "\n".join(out)
1240
+
1241
+
1242
+ @mcp.tool()
1243
+ def snapshot_list(limit: int = 10, branch: str = "") -> str:
1244
+ """
1245
+ List saved temporal snapshots of codebase metrics in reverse chronological order.
1246
+
1247
+ Each entry in the returned list contains a ``key`` (tree hash snapshot
1248
+ identifier), ``branch``, ``timestamp``, ``version``, and a summary of
1249
+ key metrics (node count, edge count, JSDoc coverage) plus deltas vs. the
1250
+ previous snapshot. Use the ``key`` field when calling ``snapshot_show()``
1251
+ or ``snapshot_diff(key_a=..., key_b=...)``.
1252
+
1253
+ Use this tool to answer questions like "how has the codebase grown?" or
1254
+ "when did JSDoc coverage improve?" or "show me only main-branch snapshots".
1255
+
1256
+ :param limit: Maximum number of snapshots to return (default 10; pass 0 for all).
1257
+ :param branch: If provided, filter to snapshots from this branch only
1258
+ (e.g. ``"main"`` or ``"develop"``).
1259
+ :return: JSON array of snapshot metadata dicts, most recent first.
1260
+ """
1261
+ mgr = _get_snapshot_mgr()
1262
+ snapshots = mgr.list_snapshots(
1263
+ limit=limit if limit > 0 else None,
1264
+ branch=branch if branch else None,
1265
+ )
1266
+ for snap in snapshots:
1267
+ snap_metrics = snap.get("metrics", {})
1268
+ snap["freshness"] = _snapshot_freshness(snap_metrics.get("total_nodes", 0))
1269
+ return json.dumps(snapshots, indent=2, ensure_ascii=False)
1270
+
1271
+
1272
+ @mcp.tool()
1273
+ def snapshot_show(key: str = "latest") -> str:
1274
+ """
1275
+ Show full details of a specific codebase metrics snapshot.
1276
+
1277
+ Pass a snapshot key (tree hash) to retrieve that exact snapshot, or use
1278
+ the special value ``"latest"`` (default) to retrieve the most recent one.
1279
+
1280
+ Snapshot keys are the ``key`` field returned by ``snapshot_list()``.
1281
+
1282
+ The returned object contains the full metrics dict (total_nodes,
1283
+ total_edges, meaningful_nodes, docstring_coverage, node_counts,
1284
+ edge_counts, critical_issues, complexity_median), the top hotspots, and
1285
+ deltas computed vs. both the previous and the baseline (oldest) snapshots.
1286
+
1287
+ :param key: Snapshot key to load, or ``"latest"`` for the most
1288
+ recent snapshot (default ``"latest"``). Keys are tree
1289
+ hashes returned by ``snapshot_list()``.
1290
+ :return: JSON object with full snapshot details, or an error dict if
1291
+ the requested snapshot does not exist.
1292
+ """
1293
+ mgr = _get_snapshot_mgr()
1294
+
1295
+ if key == "latest":
1296
+ entries = mgr.list_snapshots(limit=1)
1297
+ if not entries:
1298
+ return json.dumps({"error": "No snapshots found."})
1299
+ key = entries[0]["key"]
1300
+
1301
+ snapshot = mgr.load_snapshot(key)
1302
+ if snapshot is None:
1303
+ return json.dumps({"error": f"Snapshot not found for key: {key!r}"})
1304
+ out = snapshot.to_dict()
1305
+ out["freshness"] = _snapshot_freshness(snapshot.metrics.get("total_nodes", 0))
1306
+ return json.dumps(out, indent=2, ensure_ascii=False)
1307
+
1308
+
1309
+ @mcp.tool()
1310
+ def snapshot_diff(key_a: str, key_b: str) -> str:
1311
+ """
1312
+ Compare two codebase metric snapshots side-by-side.
1313
+
1314
+ Returns the full metrics dict for both snapshots and a computed delta
1315
+ (b − a) covering node and edge counts, plus per-kind node count and
1316
+ per-relation edge count deltas.
1317
+
1318
+ Typical workflow::
1319
+
1320
+ # 1. List available snapshots — note the 'key' field in each entry
1321
+ snapshot_list()
1322
+
1323
+ # 2. Diff any two using the key= field values
1324
+ snapshot_diff(key_a="abc1234ef...", key_b="def5678ab...")
1325
+
1326
+ :param key_a: First (older) snapshot key — the ``key`` field from
1327
+ ``snapshot_list()`` output (a tree-hash string).
1328
+ :param key_b: Second (newer) snapshot key — the ``key`` field from
1329
+ ``snapshot_list()`` output (a tree-hash string).
1330
+ :return: JSON object with keys ``a`` (metrics + issues list for key_a),
1331
+ ``b`` (metrics + issues list for key_b), ``delta`` (b − a),
1332
+ ``node_counts_delta``, and ``edge_counts_delta``. Returns an
1333
+ error dict if either snapshot is missing.
1334
+ """
1335
+ mgr = _get_snapshot_mgr()
1336
+ result = mgr.diff_snapshots(key_a, key_b)
1337
+ if "error" not in result:
1338
+ result["freshness"] = {
1339
+ "a": _snapshot_freshness(result.get("a", {}).get("metrics", {}).get("total_nodes", 0)),
1340
+ "b": _snapshot_freshness(result.get("b", {}).get("metrics", {}).get("total_nodes", 0)),
1341
+ }
1342
+ return json.dumps(result, indent=2, ensure_ascii=False)
1343
+
1344
+
1345
+ # ---------------------------------------------------------------------------
1346
+ # CLI entry point
1347
+ # ---------------------------------------------------------------------------
1348
+
1349
+
1350
+ def _parse_args(argv: list | None = None) -> argparse.Namespace:
1351
+ p = argparse.ArgumentParser(
1352
+ prog="tscodekg-mcp",
1353
+ description="TypeScriptKG MCP server — exposes TS/JS codebase query tools to AI agents.",
1354
+ )
1355
+ p.add_argument("--repo", default=".", help="Repository root directory (default: .)")
1356
+ p.add_argument(
1357
+ "--db",
1358
+ default=".tscodekg/graph.sqlite",
1359
+ help="Path to the SQLite knowledge graph",
1360
+ )
1361
+ p.add_argument(
1362
+ "--vectors",
1363
+ default=".tscodekg/vectors.sqlite",
1364
+ help="Path to the sqlite-vec vector store",
1365
+ )
1366
+ p.add_argument(
1367
+ "--model",
1368
+ default=DEFAULT_MODEL,
1369
+ help=f"Sentence-transformer model name (default: {DEFAULT_MODEL})",
1370
+ )
1371
+ p.add_argument(
1372
+ "--transport",
1373
+ choices=["stdio", "sse"],
1374
+ default="stdio",
1375
+ help="MCP transport: stdio (default) or sse (HTTP)",
1376
+ )
1377
+ return p.parse_args(argv)
1378
+
1379
+
1380
+ def main(argv: list | None = None) -> None:
1381
+ """CLI entry point for the TypeScriptKG MCP server."""
1382
+ global _kg, _snapshot_mgr
1383
+
1384
+ args = _parse_args(argv)
1385
+
1386
+ repo = Path(args.repo).resolve()
1387
+ db = Path(args.db) if Path(args.db).is_absolute() else repo / args.db
1388
+ vectors = Path(args.vectors) if Path(args.vectors).is_absolute() else repo / args.vectors
1389
+
1390
+ if not db.exists():
1391
+ print(
1392
+ f"WARNING: SQLite database not found at '{db}'.\nRun 'tscodekg build --repo .' first.",
1393
+ file=sys.stderr,
1394
+ )
1395
+
1396
+ print(
1397
+ f"TypeScriptKG MCP server starting\n"
1398
+ f" repo : {repo}\n"
1399
+ f" db : {db}\n"
1400
+ f" vectors : {vectors}\n"
1401
+ f" model : {args.model}\n"
1402
+ f" transport: {args.transport}",
1403
+ file=sys.stderr,
1404
+ )
1405
+
1406
+ _kg = TypeScriptKG(repo_root=repo, db_path=db, vectors_path=vectors, model=args.model)
1407
+ _snapshot_mgr = SnapshotManager(repo / ".tscodekg" / "snapshots", db_path=db)
1408
+ mcp.run(transport=args.transport)
1409
+
1410
+
1411
+ if __name__ == "__main__":
1412
+ main()