codegraph-engine 2.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. codegraph/__init__.py +37 -0
  2. codegraph/agent.py +26 -0
  3. codegraph/architecture.py +328 -0
  4. codegraph/audit.py +106 -0
  5. codegraph/cache.py +95 -0
  6. codegraph/cli.py +854 -0
  7. codegraph/config.py +43 -0
  8. codegraph/constraints.py +238 -0
  9. codegraph/context.py +1228 -0
  10. codegraph/epistemic.py +90 -0
  11. codegraph/errors.py +275 -0
  12. codegraph/evidence/__init__.py +15 -0
  13. codegraph/evidence/citations.py +397 -0
  14. codegraph/frameworks.py +434 -0
  15. codegraph/freshness.py +295 -0
  16. codegraph/git.py +278 -0
  17. codegraph/graph/__init__.py +46 -0
  18. codegraph/graph/models.py +41 -0
  19. codegraph/graph/traversal.py +1291 -0
  20. codegraph/indexing/__init__.py +4 -0
  21. codegraph/indexing/classifier.py +274 -0
  22. codegraph/indexing/indexer.py +943 -0
  23. codegraph/indexing/models.py +338 -0
  24. codegraph/indexing/parser.py +1240 -0
  25. codegraph/indexing/scanner.py +200 -0
  26. codegraph/indexing/test_framework.py +116 -0
  27. codegraph/interrogation.py +1582 -0
  28. codegraph/llm/__init__.py +3 -0
  29. codegraph/llm/base.py +15 -0
  30. codegraph/llm/context.py +20 -0
  31. codegraph/mcp/__init__.py +3 -0
  32. codegraph/mcp/server.py +736 -0
  33. codegraph/memory/__init__.py +3 -0
  34. codegraph/memory/store.py +46 -0
  35. codegraph/models.py +289 -0
  36. codegraph/observability.py +151 -0
  37. codegraph/optimizer.py +372 -0
  38. codegraph/planner.py +417 -0
  39. codegraph/py.typed +1 -0
  40. codegraph/query_expansion.py +199 -0
  41. codegraph/ranking.py +363 -0
  42. codegraph/resolver.py +843 -0
  43. codegraph/resources/__init__.py +45 -0
  44. codegraph/resources/cache.py +117 -0
  45. codegraph/resources/coalescer.py +83 -0
  46. codegraph/resources/debouncer.py +98 -0
  47. codegraph/resources/governor.py +232 -0
  48. codegraph/resources/policy.py +123 -0
  49. codegraph/retrieval_policy.py +220 -0
  50. codegraph/search/__init__.py +23 -0
  51. codegraph/search/hybrid.py +301 -0
  52. codegraph/search/semantic.py +28 -0
  53. codegraph/security/__init__.py +3 -0
  54. codegraph/security/paths.py +35 -0
  55. codegraph/target_resolver.py +348 -0
  56. codegraph/task.py +637 -0
  57. codegraph_engine-2.1.1.dist-info/METADATA +334 -0
  58. codegraph_engine-2.1.1.dist-info/RECORD +62 -0
  59. codegraph_engine-2.1.1.dist-info/WHEEL +5 -0
  60. codegraph_engine-2.1.1.dist-info/entry_points.txt +2 -0
  61. codegraph_engine-2.1.1.dist-info/licenses/LICENSE +21 -0
  62. codegraph_engine-2.1.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1291 @@
1
+ """Graph traversal and query engine: call graph, dependency graph, impact analysis, tests.
2
+
3
+ Confidence Vocabulary:
4
+ HIGH — parser-confirmed, symbol-resolved, source-backed (verified)
5
+ MEDIUM — structurally inferred (unique match, partial resolution, same module)
6
+ LOW — textual/heuristic match (unverified)
7
+ UNKNOWN — unresolved reference / dynamic call
8
+
9
+ Edge Classifications:
10
+ STRUCTURAL: DEFINES, CONTAINS, EXPORTS, REEXPORTS
11
+ SEMANTIC: IMPORTS, CALLS, EXTENDS, IMPLEMENTS, HANDLED_BY, TESTS
12
+ UNCERTAIN: POSSIBLE_CALLS
13
+
14
+ Architectural Invariant:
15
+ Graph is a projection of resolved source facts and references.
16
+ Heuristics are never promoted to verified CALLS.
17
+ Inverses (e.g. A --CALLS--> B queryable as CALLED_BY) do not duplicate independent facts.
18
+ """
19
+ from __future__ import annotations
20
+
21
+ import re
22
+ import sqlite3
23
+ from typing import Any
24
+
25
+ from codegraph.evidence.citations import get_repository_from_con, verify_source_hash
26
+ from codegraph.freshness import index_generation
27
+ from codegraph.indexing.models import Symbol
28
+ from codegraph.resources import get_global_governor, get_graph_cache
29
+
30
+ from .models import GraphEdge
31
+
32
+
33
+ class AttrDict(dict[str, Any]):
34
+ """Dict subclass that provides attribute access for ergonomics and JSON compatibility."""
35
+
36
+ def __getattr__(self, name: str) -> Any:
37
+ try:
38
+ return self[name]
39
+ except KeyError:
40
+ raise AttributeError(f"'AttrDict' object has no attribute '{name}'") from None
41
+
42
+ def __setattr__(self, name: str, value: Any) -> None:
43
+ self[name] = value
44
+
45
+
46
+ # ---------------------------------------------------------------------------
47
+ # Symbol Query API
48
+ # ---------------------------------------------------------------------------
49
+
50
+
51
+ def get_symbol(con: sqlite3.Connection, canonical_id: str) -> Symbol | None:
52
+ """Retrieve a single symbol by its exact canonical ID."""
53
+ row = con.execute(
54
+ "SELECT name, qualified_name, kind, path, start_line, end_line, decorators, "
55
+ "id, canonical_id, language, module, scope, signature, content_hash, "
56
+ "parent_symbol_id, visibility, return_type, parameter_count, documentation "
57
+ "FROM symbols WHERE canonical_id=? OR id=? LIMIT 1",
58
+ (canonical_id, canonical_id),
59
+ ).fetchone()
60
+ if not row:
61
+ return None
62
+ return _row_to_symbol(row)
63
+
64
+
65
+ def find_symbol_exact(con: sqlite3.Connection, qualified_name: str) -> list[Symbol]:
66
+ """Retrieve symbols matching exact canonical_id or qualified_name."""
67
+ rows = con.execute(
68
+ "SELECT name, qualified_name, kind, path, start_line, end_line, decorators, "
69
+ "id, canonical_id, language, module, scope, signature, content_hash, "
70
+ "parent_symbol_id, visibility, return_type, parameter_count, documentation "
71
+ "FROM symbols WHERE canonical_id=? OR qualified_name=? "
72
+ "ORDER BY path, start_line",
73
+ (qualified_name, qualified_name),
74
+ ).fetchall()
75
+ return [_row_to_symbol(r) for r in rows]
76
+
77
+
78
+ def find_symbols(
79
+ con: sqlite3.Connection, name: str, max_results: int = 50
80
+ ) -> list[Symbol]:
81
+ """Find symbols by canonical ID, qualified name, or short name with deterministic ranking.
82
+
83
+ - If query has dots and matches canonical_id, returns exact canonical match.
84
+ - If query has dots and matches qualified_name, returns qualified matches.
85
+ - If query has no dots, returns symbols where name = query, preserving all distinct modules.
86
+ """
87
+ clean = name.strip()
88
+ if not clean:
89
+ return []
90
+
91
+ # 1. Exact canonical ID match
92
+ canon_rows = con.execute(
93
+ "SELECT name, qualified_name, kind, path, start_line, end_line, decorators, "
94
+ "id, canonical_id, language, module, scope, signature, content_hash, "
95
+ "parent_symbol_id, visibility, return_type, parameter_count, documentation "
96
+ "FROM symbols WHERE canonical_id=? ORDER BY path, start_line LIMIT ?",
97
+ (clean, max_results),
98
+ ).fetchall()
99
+ if canon_rows:
100
+ return [_row_to_symbol(r) for r in canon_rows]
101
+
102
+ # 2. Qualified name match if dotted
103
+ if "." in clean:
104
+ q_rows = con.execute(
105
+ "SELECT name, qualified_name, kind, path, start_line, end_line, decorators, "
106
+ "id, canonical_id, language, module, scope, signature, content_hash, "
107
+ "parent_symbol_id, visibility, return_type, parameter_count, documentation "
108
+ "FROM symbols WHERE qualified_name=? OR canonical_id LIKE ? "
109
+ "ORDER BY path, start_line LIMIT ?",
110
+ (clean, f"%.{clean}", max_results),
111
+ ).fetchall()
112
+ if q_rows:
113
+ return [_row_to_symbol(r) for r in q_rows]
114
+
115
+ # 3. Short name match
116
+ short_rows = con.execute(
117
+ "SELECT name, qualified_name, kind, path, start_line, end_line, decorators, "
118
+ "id, canonical_id, language, module, scope, signature, content_hash, "
119
+ "parent_symbol_id, visibility, return_type, parameter_count, documentation "
120
+ "FROM symbols WHERE name=? ORDER BY path, start_line, canonical_id LIMIT ?",
121
+ (clean, max_results),
122
+ ).fetchall()
123
+ return [_row_to_symbol(r) for r in short_rows]
124
+
125
+
126
+ find_symbol = find_symbols
127
+
128
+
129
+ def _row_to_symbol(r: sqlite3.Row) -> Symbol:
130
+ dec_list = [d for d in str(r["decorators"]).split(",") if d]
131
+ return Symbol(
132
+ id=str(r["id"]),
133
+ canonical_id=str(r["canonical_id"]),
134
+ name=str(r["name"]),
135
+ qualified_name=str(r["qualified_name"]),
136
+ kind=str(r["kind"]),
137
+ start_line=int(r["start_line"]),
138
+ end_line=int(r["end_line"]),
139
+ file_path=str(r["path"]),
140
+ decorators=dec_list,
141
+ language=str(r["language"]),
142
+ module=str(r["module"]),
143
+ path=str(r["path"]),
144
+ scope=str(r["scope"]),
145
+ signature=str(r["signature"]),
146
+ content_hash=str(r["content_hash"]),
147
+ parent_symbol_id=r["parent_symbol_id"],
148
+ visibility=str(r["visibility"]),
149
+ return_type=r["return_type"],
150
+ parameter_count=r["parameter_count"],
151
+ documentation=r["documentation"],
152
+ )
153
+
154
+
155
+ # ---------------------------------------------------------------------------
156
+ # References & Call Graph API
157
+ # ---------------------------------------------------------------------------
158
+
159
+
160
+ def find_references(
161
+ con: sqlite3.Connection,
162
+ canonical_id: str,
163
+ max_results: int = 100,
164
+ ) -> list[AttrDict]:
165
+ """Return verified and pending references targeting the given symbol."""
166
+ repo = get_repository_from_con(con)
167
+ rows = con.execute(
168
+ "SELECT source_symbol_id, target_symbol_id, relationship, confidence, "
169
+ "path, start_line, end_line, evidence, source_hash, indexed_commit, evidence_status "
170
+ "FROM 'references' "
171
+ "WHERE target_symbol_id=? OR target_symbol_id LIKE ? "
172
+ "ORDER BY confidence = 'HIGH' DESC, path, start_line LIMIT ?",
173
+ (canonical_id, f"%.{canonical_id}", max_results),
174
+ ).fetchall()
175
+
176
+ results: list[AttrDict] = []
177
+ for r in rows:
178
+ st = verify_source_hash(repo, r["path"], r["source_hash"])
179
+ results.append(
180
+ AttrDict(
181
+ source_symbol_id=r["source_symbol_id"],
182
+ target_symbol_id=r["target_symbol_id"],
183
+ source=r["source_symbol_id"],
184
+ target=r["target_symbol_id"],
185
+ relationship=r["relationship"],
186
+ confidence=r["confidence"] if st == "current" else "LOW",
187
+ path=r["path"],
188
+ file=r["path"],
189
+ start_line=r["start_line"],
190
+ end_line=r["end_line"],
191
+ line=r["start_line"],
192
+ evidence=r["evidence"],
193
+ source_hash=r["source_hash"],
194
+ indexed_commit=r["indexed_commit"],
195
+ evidence_status=st,
196
+ )
197
+ )
198
+ return results
199
+
200
+
201
+ def find_callers(
202
+ con: sqlite3.Connection,
203
+ symbol: str,
204
+ max_results: int = 50,
205
+ ) -> list[AttrDict]:
206
+ """Return verified callers from resolved references, falling back to static calls table.
207
+
208
+ Verified callers carry confidence 'HIGH' or 'MEDIUM' and relationship 'CALLS'.
209
+ Unverified textual matches carry confidence 'LOW' and relationship 'POSSIBLE_CALLS'.
210
+ """
211
+ gov = get_global_governor()
212
+ max_results = min(max_results, gov.policy.max_graph_nodes_per_query)
213
+ gen = index_generation(con)
214
+ cache = get_graph_cache()
215
+ cache_key = (gen, "callers", symbol, max_results)
216
+ cached = cache.get(cache_key)
217
+ if cached is not None and isinstance(cached, list):
218
+ return cached
219
+
220
+ repo = get_repository_from_con(con)
221
+ short_name = symbol.split(".")[-1]
222
+ results: list[AttrDict] = []
223
+ seen: set[tuple[str, str | None, int]] = set()
224
+
225
+ # 1. Resolved CALLS edges from references table
226
+ rows = con.execute(
227
+ "SELECT source_symbol_id, target_symbol_id, relationship, confidence, "
228
+ "path, start_line, end_line, evidence, source_hash "
229
+ "FROM 'references' "
230
+ "WHERE relationship='CALLS' AND (target_symbol_id=? OR target_symbol_id LIKE ? OR target_symbol_id LIKE ?) "
231
+ "ORDER BY confidence = 'HIGH' DESC, path, start_line LIMIT ?",
232
+ (symbol, f"%.{symbol}", f"%.{short_name}", max_results),
233
+ ).fetchall()
234
+
235
+ for r in rows:
236
+ key = (r["path"], r["source_symbol_id"], r["start_line"])
237
+ if key in seen:
238
+ continue
239
+ seen.add(key)
240
+ st = verify_source_hash(repo, r["path"], r["source_hash"])
241
+ results.append(
242
+ AttrDict(
243
+ file=r["path"],
244
+ path=r["path"],
245
+ symbol=r["source_symbol_id"],
246
+ source_symbol_id=r["source_symbol_id"],
247
+ callee=short_name,
248
+ target_symbol_id=r["target_symbol_id"],
249
+ line=r["start_line"],
250
+ start_line=r["start_line"],
251
+ end_line=r["end_line"],
252
+ relationship="CALLS",
253
+ confidence=r["confidence"] if st == "current" else "LOW",
254
+ evidence=r["evidence"],
255
+ evidence_status=st,
256
+ )
257
+ )
258
+
259
+ # 2. Check static calls table (for unresolved or partially resolved calls)
260
+ if len(results) < max_results:
261
+ rem = max_results - len(results)
262
+ call_rows = con.execute(
263
+ "SELECT source_path, callee, qualified_callee, line, confidence, source_symbol_id, resolved_symbol_id "
264
+ "FROM calls WHERE callee=? OR qualified_callee=? OR resolved_symbol_id=? LIMIT ?",
265
+ (short_name, symbol, symbol, rem),
266
+ ).fetchall()
267
+
268
+ for cr in call_rows:
269
+ key = (cr["source_path"], cr["source_symbol_id"], cr["line"])
270
+ if key in seen:
271
+ continue
272
+ seen.add(key)
273
+ conf = cr["confidence"] or "LOW"
274
+ rel = "CALLS" if conf in ("HIGH", "MEDIUM") else "POSSIBLE_CALLS"
275
+ results.append(
276
+ AttrDict(
277
+ file=cr["source_path"],
278
+ path=cr["source_path"],
279
+ symbol=cr["source_symbol_id"] or cr["source_path"],
280
+ source_symbol_id=cr["source_symbol_id"],
281
+ callee=cr["callee"],
282
+ qualified_callee=cr["qualified_callee"],
283
+ line=cr["line"],
284
+ start_line=cr["line"],
285
+ end_line=cr["line"],
286
+ relationship=rel,
287
+ confidence=conf,
288
+ evidence=f"Static call site '{cr['callee']}()' in {cr['source_path']}:{cr['line']}",
289
+ )
290
+ )
291
+
292
+ final_callers = results[:max_results]
293
+ cache.set(cache_key, final_callers)
294
+ return final_callers
295
+
296
+
297
+ get_callers = find_callers
298
+
299
+
300
+ def find_callees(
301
+ con: sqlite3.Connection,
302
+ symbol: str,
303
+ max_results: int = 50,
304
+ ) -> list[AttrDict]:
305
+ """Return callees called by the given symbol's body."""
306
+ gov = get_global_governor()
307
+ max_results = min(max_results, gov.policy.max_graph_nodes_per_query)
308
+ gen = index_generation(con)
309
+ cache = get_graph_cache()
310
+ cache_key = (gen, "callees", symbol, max_results)
311
+ cached = cache.get(cache_key)
312
+ if cached is not None and isinstance(cached, list):
313
+ return cached
314
+
315
+ repo = get_repository_from_con(con)
316
+ short_name = symbol.split(".")[-1]
317
+ results: list[AttrDict] = []
318
+ seen: set[tuple[str, str | None, int]] = set()
319
+
320
+ # 1. Check resolved references where source_symbol_id matches symbol
321
+ rows = con.execute(
322
+ "SELECT source_symbol_id, target_symbol_id, relationship, confidence, "
323
+ "path, start_line, end_line, evidence, source_hash "
324
+ "FROM 'references' "
325
+ "WHERE (source_symbol_id=? OR source_symbol_id LIKE ? OR source_symbol_id LIKE ?) "
326
+ "AND relationship IN ('CALLS', 'UNRESOLVED_REFERENCE') "
327
+ "ORDER BY confidence = 'HIGH' DESC, path, start_line LIMIT ?",
328
+ (symbol, f"%.{symbol}", f"%.{short_name}", max_results),
329
+ ).fetchall()
330
+
331
+ for r in rows:
332
+ key = (r["path"], r["target_symbol_id"], r["start_line"])
333
+ if key in seen:
334
+ continue
335
+ seen.add(key)
336
+ callee_nm = r["target_symbol_id"].split(".")[-1] if r["target_symbol_id"] else "unresolved"
337
+ st = verify_source_hash(repo, r["path"], r["source_hash"])
338
+ results.append(
339
+ AttrDict(
340
+ callee=callee_nm,
341
+ qualified_callee=r["target_symbol_id"],
342
+ target_symbol_id=r["target_symbol_id"],
343
+ source_symbol_id=r["source_symbol_id"],
344
+ file=r["path"],
345
+ path=r["path"],
346
+ line=r["start_line"],
347
+ start_line=r["start_line"],
348
+ end_line=r["end_line"],
349
+ relationship=r["relationship"],
350
+ confidence=r["confidence"] if st == "current" else "LOW",
351
+ evidence=r["evidence"],
352
+ evidence_status=st,
353
+ )
354
+ )
355
+
356
+ # 2. Check chunks / calls table if references did not contain calls
357
+ if not results:
358
+ chunk = con.execute(
359
+ "SELECT path, start_line, end_line, content FROM chunks "
360
+ "WHERE symbol=? OR symbol LIKE ? LIMIT 1",
361
+ (symbol, f"%.{symbol}"),
362
+ ).fetchone()
363
+ if chunk:
364
+ call_rows = con.execute(
365
+ "SELECT callee, qualified_callee, line, confidence FROM calls "
366
+ "WHERE source_path=? AND line >= ? AND line <= ? LIMIT ?",
367
+ (chunk["path"], chunk["start_line"], chunk["end_line"], max_results),
368
+ ).fetchall()
369
+ for cr in call_rows:
370
+ results.append(
371
+ AttrDict(
372
+ callee=cr["callee"],
373
+ qualified_callee=cr["qualified_callee"],
374
+ line=cr["line"],
375
+ start_line=cr["line"],
376
+ end_line=cr["line"],
377
+ file=chunk["path"],
378
+ path=chunk["path"],
379
+ relationship="CALLS" if cr["confidence"] in ("HIGH", "MEDIUM") else "POSSIBLE_CALLS",
380
+ confidence=cr["confidence"] or "LOW",
381
+ evidence=f"Call site in {chunk['path']}:{cr['line']}",
382
+ )
383
+ )
384
+
385
+ final_callees = results[:max_results]
386
+ cache.set(cache_key, final_callees)
387
+ return final_callees
388
+
389
+
390
+ get_callees = find_callees
391
+
392
+
393
+ def find_implementations(
394
+ con: sqlite3.Connection,
395
+ canonical_id: str,
396
+ max_results: int = 50,
397
+ ) -> list[AttrDict]:
398
+ """Find classes or interfaces that EXTENDS or IMPLEMENTS canonical_id."""
399
+ short = canonical_id.split(".")[-1]
400
+ rows = con.execute(
401
+ "SELECT source_symbol_id, target_symbol_id, relationship, confidence, path, start_line, end_line, evidence "
402
+ "FROM 'references' "
403
+ "WHERE relationship IN ('EXTENDS', 'IMPLEMENTS') AND (target_symbol_id=? OR target_symbol_id LIKE ? OR target_symbol_id=?) "
404
+ "ORDER BY path, start_line LIMIT ?",
405
+ (canonical_id, f"%.{canonical_id}", short, max_results),
406
+ ).fetchall()
407
+
408
+ return [
409
+ AttrDict(
410
+ implementor=r["source_symbol_id"],
411
+ target=r["target_symbol_id"],
412
+ relationship=r["relationship"],
413
+ confidence=r["confidence"],
414
+ file=r["path"],
415
+ line=r["start_line"],
416
+ evidence=r["evidence"],
417
+ )
418
+ for r in rows
419
+ ]
420
+
421
+
422
+ def get_call_graph(
423
+ con: sqlite3.Connection,
424
+ symbol: str | None = None,
425
+ depth: int = 2,
426
+ max_results: int = 100,
427
+ ) -> list[AttrDict]:
428
+ """Return call graph edges up to `depth` hops."""
429
+ if depth < 1:
430
+ depth = 1
431
+ if depth > 5:
432
+ depth = 5
433
+
434
+ edges: list[AttrDict] = []
435
+ visited: set[tuple[str, str]] = set()
436
+
437
+ if symbol:
438
+ frontier = {symbol}
439
+ for _ in range(depth):
440
+ if not frontier:
441
+ break
442
+ next_frontier: set[str] = set()
443
+ for sym in frontier:
444
+ callers = find_callers(con, sym, max_results=max_results)
445
+ for c in callers:
446
+ src = c.get("source_symbol_id") or c.get("file")
447
+ tgt = c.get("target_symbol_id") or sym
448
+ edge_key = (str(src), str(tgt))
449
+ if edge_key not in visited:
450
+ visited.add(edge_key)
451
+ edges.append(
452
+ AttrDict(
453
+ source=src,
454
+ target=tgt,
455
+ relationship=c.get("relationship", "CALLS"),
456
+ confidence=c.get("confidence", "HIGH"),
457
+ file=c.get("file"),
458
+ line=c.get("line"),
459
+ evidence=c.get("evidence", ""),
460
+ )
461
+ )
462
+ if src and str(src) not in frontier:
463
+ next_frontier.add(str(src))
464
+ frontier = next_frontier
465
+ return edges[:max_results]
466
+ else:
467
+ rows = con.execute(
468
+ "SELECT source, target, relationship, confidence, file, start_line, evidence "
469
+ "FROM graph_edges WHERE relationship IN ('CALLS', 'POSSIBLE_CALLS', 'HANDLED_BY', 'ROUTES_TO') "
470
+ "LIMIT ?",
471
+ (max_results,),
472
+ ).fetchall()
473
+ return [
474
+ AttrDict(
475
+ source=r["source"],
476
+ target=r["target"],
477
+ relationship=r["relationship"],
478
+ confidence=r["confidence"],
479
+ file=r["file"],
480
+ line=r["start_line"],
481
+ evidence=r["evidence"],
482
+ )
483
+ for r in rows
484
+ ]
485
+
486
+
487
+ # ---------------------------------------------------------------------------
488
+ # Dependency & Import Graph API
489
+ # ---------------------------------------------------------------------------
490
+
491
+
492
+ def get_dependency_graph(
493
+ con: sqlite3.Connection,
494
+ path: str | None = None,
495
+ depth: int = 2,
496
+ max_results: int = 200,
497
+ ) -> list[dict[str, object]]:
498
+ """Return import dependency edges, optionally scoped to a starting file."""
499
+ if depth < 1:
500
+ depth = 1
501
+ if depth > 5:
502
+ depth = 5
503
+
504
+ if path:
505
+ visited: set[str] = set()
506
+ frontier = {path}
507
+ edges: list[dict[str, object]] = []
508
+ for _ in range(depth):
509
+ if not frontier:
510
+ break
511
+ next_frontier: set[str] = set()
512
+ for src in frontier:
513
+ if src in visited:
514
+ continue
515
+ visited.add(src)
516
+ rows = con.execute(
517
+ "SELECT module, resolved_path FROM imports WHERE source_path=? LIMIT ?",
518
+ (src, max_results),
519
+ ).fetchall()
520
+ for row in rows:
521
+ edges.append(
522
+ {
523
+ "source": src,
524
+ "target": row["module"],
525
+ "resolved_target": row["resolved_path"],
526
+ "relationship": "IMPORTS",
527
+ "confidence": "HIGH",
528
+ "evidence": f"Import '{row['module']}' in {src}",
529
+ }
530
+ )
531
+ if row["resolved_path"] and row["resolved_path"] not in visited:
532
+ next_frontier.add(row["resolved_path"])
533
+ frontier = next_frontier - visited
534
+ return edges[:max_results]
535
+ else:
536
+ rows = con.execute(
537
+ "SELECT source_path, module, resolved_path FROM imports LIMIT ?", (max_results,)
538
+ ).fetchall()
539
+ return [
540
+ {
541
+ "source": r["source_path"],
542
+ "target": r["module"],
543
+ "resolved_target": r["resolved_path"],
544
+ "relationship": "IMPORTS",
545
+ "confidence": "HIGH",
546
+ "evidence": "Parser-extracted import declaration",
547
+ }
548
+ for r in rows
549
+ ]
550
+
551
+
552
+ def find_importers(
553
+ con: sqlite3.Connection,
554
+ module: str,
555
+ max_results: int = 100,
556
+ ) -> list[dict[str, object]]:
557
+ """Return files that import a given module or file."""
558
+ rows = con.execute(
559
+ "SELECT source_path, module, name, alias, resolved_path FROM imports "
560
+ "WHERE module=? OR resolved_module=? OR resolved_path=? OR full_name LIKE ? "
561
+ "ORDER BY source_path LIMIT ?",
562
+ (module, module, module, f"{module}%", max_results),
563
+ ).fetchall()
564
+ return [
565
+ {
566
+ "importer": r["source_path"],
567
+ "module": r["module"],
568
+ "name": r["name"],
569
+ "alias": r["alias"],
570
+ "resolved_path": r["resolved_path"],
571
+ "confidence": "HIGH",
572
+ "evidence": "Parser-extracted import declaration",
573
+ }
574
+ for r in rows
575
+ ]
576
+
577
+
578
+ # ---------------------------------------------------------------------------
579
+ # Test Intelligence API
580
+ # ---------------------------------------------------------------------------
581
+
582
+ _TEST_FILE_RE = re.compile(
583
+ r"(^|[_/])test[_s]?[_/]|[_/]spec[_/]|test[_s]?\.(py|js|ts|jsx|tsx)$|spec\.(py|js|ts|jsx|tsx)$",
584
+ re.IGNORECASE,
585
+ )
586
+
587
+
588
+ def find_related_tests(
589
+ con: sqlite3.Connection,
590
+ symbol_or_path: str,
591
+ max_results: int = 20,
592
+ ) -> list[dict[str, object]]:
593
+ """Find test files statically linked to a symbol or file path with verified provenance."""
594
+ gov = get_global_governor()
595
+ max_results = min(max_results, gov.policy.max_graph_nodes_per_query)
596
+ gen = index_generation(con)
597
+ cache = get_graph_cache()
598
+ cache_key = (gen, "related_tests", symbol_or_path, max_results)
599
+ cached = cache.get(cache_key)
600
+ if cached is not None and isinstance(cached, list):
601
+ return cached
602
+
603
+ short = symbol_or_path.split(".")[-1]
604
+ module_name = (
605
+ symbol_or_path.replace("/", ".").rsplit(".", 1)[0]
606
+ if "." in symbol_or_path
607
+ else symbol_or_path
608
+ )
609
+ tests: list[dict[str, object]] = []
610
+ seen: set[str] = set()
611
+
612
+ # 1. VERIFIED: Tests that directly call or reference the symbol in 'references' table
613
+ ref_rows = con.execute(
614
+ "SELECT path, source_symbol_id, start_line, evidence FROM 'references' "
615
+ "WHERE (target_symbol_id=? OR target_symbol_id LIKE ? OR target_symbol_id=?) "
616
+ "AND relationship IN ('CALLS', 'TESTS') LIMIT ?",
617
+ (symbol_or_path, f"%.{short}", short, max_results),
618
+ ).fetchall()
619
+ for row in ref_rows:
620
+ if _TEST_FILE_RE.search(row["path"]) and row["path"] not in seen:
621
+ seen.add(row["path"])
622
+ tests.append(
623
+ {
624
+ "file": row["path"],
625
+ "symbol": row["source_symbol_id"],
626
+ "relationship": "TESTS_SYMBOL",
627
+ "classification": "VERIFIED_TEST",
628
+ "confidence": "HIGH",
629
+ "evidence": f"Test verifies symbol: {row['evidence']}",
630
+ }
631
+ )
632
+
633
+ # 2. VERIFIED: Tests that import the containing module
634
+ imp_rows = con.execute(
635
+ "SELECT DISTINCT source_path FROM imports "
636
+ "WHERE module=? OR resolved_module=? OR module LIKE ? LIMIT ?",
637
+ (module_name, module_name, f"%{short}%", max_results),
638
+ ).fetchall()
639
+ for row in imp_rows:
640
+ p = row["source_path"]
641
+ if _TEST_FILE_RE.search(p) and p not in seen:
642
+ seen.add(p)
643
+ tests.append(
644
+ {
645
+ "file": p,
646
+ "symbol": None,
647
+ "relationship": "TEST_IMPORTS_MODULE",
648
+ "classification": "VERIFIED_TEST",
649
+ "confidence": "HIGH",
650
+ "evidence": f"Test file imports module '{module_name}'",
651
+ }
652
+ )
653
+
654
+ # 3. POSSIBLE: Test symbols whose name contains the target name
655
+ sym_rows = con.execute(
656
+ "SELECT qualified_name, path, start_line FROM symbols "
657
+ "WHERE name LIKE ? AND path LIKE ? LIMIT ?",
658
+ (f"%{short}%", "%test%", max_results),
659
+ ).fetchall()
660
+ for row in sym_rows:
661
+ if _TEST_FILE_RE.search(row["path"]) and row["path"] not in seen:
662
+ seen.add(row["path"])
663
+ tests.append(
664
+ {
665
+ "file": row["path"],
666
+ "symbol": row["qualified_name"],
667
+ "start_line": row["start_line"],
668
+ "relationship": "TEST_COVERS_SYMBOL",
669
+ "classification": "POSSIBLE_TEST",
670
+ "confidence": "MEDIUM",
671
+ "evidence": f"Test name contains '{short}'",
672
+ }
673
+ )
674
+
675
+ if not tests:
676
+ return [
677
+ {
678
+ "result": "no_static_link_found",
679
+ "detail": (
680
+ f"No statically linked test found for '{symbol_or_path}'. "
681
+ "This does not mean tests are completely absent — dynamic test frameworks "
682
+ "or naming conventions outside our patterns may cover this symbol."
683
+ ),
684
+ }
685
+ ]
686
+
687
+ final_tests = tests[:max_results]
688
+ cache.set(cache_key, final_tests)
689
+ return final_tests
690
+
691
+
692
+ def find_affected_tests(
693
+ con: sqlite3.Connection,
694
+ changed_targets: list[str],
695
+ max_results: int = 50,
696
+ ) -> list[dict[str, object]]:
697
+ """Return tests affected by changes to any of `changed_targets`."""
698
+ affected: list[dict[str, object]] = []
699
+ seen: set[str] = set()
700
+ for target in changed_targets:
701
+ res = find_related_tests(con, target, max_results=max_results)
702
+ for t in res:
703
+ if "result" not in t:
704
+ f = str(t.get("file", ""))
705
+ if f not in seen:
706
+ seen.add(f)
707
+ affected.append(t)
708
+ return affected[:max_results]
709
+
710
+
711
+ # ---------------------------------------------------------------------------
712
+ # Impact Analysis Engine
713
+ # ---------------------------------------------------------------------------
714
+
715
+
716
+ def analyze_impact(
717
+ con: sqlite3.Connection,
718
+ symbol_or_file: str,
719
+ max_depth: int = 3,
720
+ max_results: int = 100,
721
+ ) -> dict[str, object]:
722
+ """Perform a bounded reverse-graph impact analysis for a symbol or file."""
723
+ short_name = symbol_or_file.split(".")[-1]
724
+ is_file = "/" in symbol_or_file or symbol_or_file.endswith((".py", ".js", ".ts", ".jsx", ".tsx"))
725
+
726
+ direct_callers: list[dict[str, object]] = []
727
+ transitive_callers: list[dict[str, object]] = []
728
+ dependencies: list[dict[str, object]] = []
729
+ dependent_modules: list[dict[str, object]] = []
730
+ related_apis: list[dict[str, object]] = []
731
+ related_tests = find_related_tests(con, symbol_or_file, max_results=20)
732
+
733
+ # 1. Direct callers
734
+ direct = find_callers(con, symbol_or_file, max_results=max_results)
735
+ for d in direct:
736
+ direct_callers.append(
737
+ {
738
+ "file": d.get("file"),
739
+ "symbol": d.get("symbol"),
740
+ "line": d.get("line"),
741
+ "relationship": d.get("relationship", "CALLS"),
742
+ "confidence": d.get("confidence", "HIGH"),
743
+ "label": "verified" if d.get("confidence") == "HIGH" else "possible",
744
+ "evidence": d.get("evidence", ""),
745
+ }
746
+ )
747
+
748
+ # 2. Transitive callers
749
+ if max_depth > 1:
750
+ seen_files = {str(c_dict.get("file")) for c_dict in direct_callers}
751
+ for caller_entry in direct_callers[:15]:
752
+ src_sym = caller_entry.get("symbol") or caller_entry.get("file")
753
+ if src_sym:
754
+ second = find_callers(con, str(src_sym), max_results=10)
755
+ for s in second:
756
+ if str(s.get("file")) not in seen_files:
757
+ seen_files.add(str(s.get("file")))
758
+ transitive_callers.append(
759
+ {
760
+ "file": s.get("file"),
761
+ "symbol": s.get("symbol"),
762
+ "line": s.get("line"),
763
+ "relationship": s.get("relationship", "CALLS"),
764
+ "confidence": s.get("confidence", "LOW"),
765
+ "label": "possible",
766
+ "evidence": s.get("evidence", ""),
767
+ }
768
+ )
769
+
770
+ # 3. Dependencies
771
+ if is_file:
772
+ deps = get_dependency_graph(con, symbol_or_file, depth=1, max_results=50)
773
+ for dep_entry in deps:
774
+ dependencies.append(
775
+ {
776
+ "target": dep_entry.get("target"),
777
+ "resolved_target": dep_entry.get("resolved_target"),
778
+ "relationship": "IMPORTS",
779
+ "label": "verified",
780
+ "evidence": dep_entry.get("evidence", ""),
781
+ }
782
+ )
783
+ mod_name = symbol_or_file.replace("/", ".").rsplit(".", 1)[0]
784
+ importers = find_importers(con, mod_name, max_results=50)
785
+ for imp in importers:
786
+ dependent_modules.append(
787
+ {
788
+ "importer": imp.get("importer"),
789
+ "module": imp.get("module"),
790
+ "label": "verified",
791
+ "evidence": imp.get("evidence", ""),
792
+ }
793
+ )
794
+ else:
795
+ defn = con.execute(
796
+ "SELECT path FROM symbols WHERE canonical_id=? OR qualified_name=? OR name=? LIMIT 1",
797
+ (symbol_or_file, symbol_or_file, short_name),
798
+ ).fetchone()
799
+ if defn:
800
+ deps = get_dependency_graph(con, defn["path"], depth=1, max_results=50)
801
+ for dep_entry in deps:
802
+ dependencies.append(
803
+ {
804
+ "target": dep_entry.get("target"),
805
+ "resolved_target": dep_entry.get("resolved_target"),
806
+ "relationship": "IMPORTS",
807
+ "label": "verified",
808
+ "evidence": dep_entry.get("evidence", ""),
809
+ }
810
+ )
811
+
812
+ # 4. Related APIs (Framework endpoints routing to or matching this handler)
813
+ route_rows = con.execute(
814
+ "SELECT endpoint_id, framework, http_method, route_path, handler_canonical_id, file_path, line, evidence "
815
+ "FROM framework_routes WHERE handler_canonical_id LIKE ? OR handler_name=?",
816
+ (f"%{short_name}%", short_name),
817
+ ).fetchall()
818
+ for rr in route_rows:
819
+ related_apis.append(
820
+ {
821
+ "endpoint_id": rr["endpoint_id"],
822
+ "framework": rr["framework"],
823
+ "http_method": rr["http_method"],
824
+ "route": rr["route_path"],
825
+ "handler": rr["handler_canonical_id"],
826
+ "file": rr["file_path"],
827
+ "line": rr["line"],
828
+ "label": "verified",
829
+ "evidence": rr["evidence"],
830
+ }
831
+ )
832
+
833
+ return {
834
+ "subject": symbol_or_file,
835
+ "direct_callers": direct_callers[:max_results],
836
+ "transitive_callers": transitive_callers[:max_results],
837
+ "dependencies": dependencies,
838
+ "dependent_modules": dependent_modules,
839
+ "related_apis": related_apis,
840
+ "related_tests": related_tests,
841
+ "label_legend": {
842
+ "verified": "Parser-confirmed relationship with concrete source evidence",
843
+ "inferred": "Structurally inferred relationship",
844
+ "possible": "Heuristic match — not statically verified",
845
+ },
846
+ "note": "Static analysis cannot confirm runtime behavior. Do not assert failure solely from static data.",
847
+ }
848
+
849
+
850
+ # ---------------------------------------------------------------------------
851
+ # Backward-Compatible Public Trace / Import / Definition APIs
852
+ # ---------------------------------------------------------------------------
853
+
854
+
855
+ def trace_call(
856
+ con: sqlite3.Connection,
857
+ symbol: str,
858
+ max_depth: int = 2,
859
+ callers: bool = True,
860
+ callees: bool = False,
861
+ both: bool = False,
862
+ ) -> list[dict[str, object]]:
863
+ """Return definition, callers, and/or callees with explicit confidence and relationship labels.
864
+
865
+ Hard bound: max_depth is clamped to max 5.
866
+ Explicit labels: CALLER, CALLEE, DEFINES, IMPORTS, HANDLED_BY.
867
+ """
868
+ effective_depth = max(1, min(max_depth, 5))
869
+ if both:
870
+ do_callers = True
871
+ do_callees = True
872
+ elif callees and not callers:
873
+ do_callers = False
874
+ do_callees = True
875
+ elif callers and not callees:
876
+ do_callers = True
877
+ do_callees = False
878
+ elif not callers and not callees:
879
+ do_callers = True
880
+ do_callees = True
881
+ else:
882
+ do_callers = callers
883
+ do_callees = callees
884
+
885
+ definitions = con.execute(
886
+ "SELECT canonical_id, qualified_name, path, start_line, end_line FROM symbols "
887
+ "WHERE canonical_id=? OR qualified_name=? OR name=?",
888
+ (symbol, symbol, symbol),
889
+ ).fetchall()
890
+
891
+ output: list[dict[str, object]] = []
892
+ seen: set[tuple[str, str, int, str]] = set()
893
+
894
+ # 1. Definitions
895
+ for d in definitions:
896
+ key = (d["qualified_name"], d["path"], d["start_line"], "DEFINES")
897
+ if key not in seen:
898
+ seen.add(key)
899
+ output.append(
900
+ {
901
+ "symbol": d["qualified_name"],
902
+ "canonical_id": d["canonical_id"],
903
+ "file": d["path"],
904
+ "start_line": d["start_line"],
905
+ "end_line": d["end_line"],
906
+ "relationship": "DEFINES",
907
+ "confidence": "HIGH",
908
+ "evidence": "Parser-extracted symbol definition",
909
+ }
910
+ )
911
+
912
+ # 2. Handled-by routes
913
+ try:
914
+ route_rows = con.execute(
915
+ "SELECT endpoint_id, file_path, line FROM framework_routes "
916
+ "WHERE handler_name=? OR handler_canonical_id=?",
917
+ (symbol, symbol),
918
+ ).fetchall()
919
+ for rr in route_rows:
920
+ key = (rr["endpoint_id"], rr["file_path"], rr["line"], "HANDLED_BY")
921
+ if key not in seen:
922
+ seen.add(key)
923
+ output.append(
924
+ {
925
+ "symbol": rr["endpoint_id"],
926
+ "file": rr["file_path"],
927
+ "start_line": rr["line"],
928
+ "end_line": rr["line"],
929
+ "relationship": "HANDLED_BY",
930
+ "confidence": "HIGH",
931
+ "evidence": f"Route definition for {rr['endpoint_id']}",
932
+ }
933
+ )
934
+ except sqlite3.OperationalError:
935
+ pass
936
+
937
+ # 3. Callers traversal
938
+ if do_callers:
939
+ current_symbols = [symbol]
940
+ visited_callers: set[str] = set()
941
+ for depth_step in range(effective_depth):
942
+ next_symbols: list[str] = []
943
+ for sym in current_symbols:
944
+ if sym in visited_callers:
945
+ continue
946
+ visited_callers.add(sym)
947
+ caller_items = find_callers(con, sym, max_results=20)
948
+ for c in caller_items:
949
+ c_sym = str(c.get("symbol") or "")
950
+ c_file = str(c.get("file") or "")
951
+ c_line = int(c.get("start_line") or 1)
952
+ key = (c_sym, c_file, c_line, "CALLER")
953
+ if key not in seen:
954
+ seen.add(key)
955
+ output.append(
956
+ {
957
+ "symbol": c_sym,
958
+ "file": c_file,
959
+ "start_line": c_line,
960
+ "end_line": int(c.get("end_line") or c_line),
961
+ "relationship": "CALLER",
962
+ "confidence": str(c.get("confidence", "LOW")),
963
+ "evidence": str(c.get("evidence", f"Call reference to '{sym}'")),
964
+ "depth": depth_step + 1,
965
+ }
966
+ )
967
+ if c_sym:
968
+ next_symbols.append(c_sym)
969
+ current_symbols = next_symbols
970
+ if not current_symbols:
971
+ break
972
+
973
+ # 4. Callees traversal
974
+ if do_callees:
975
+ current_symbols = [symbol]
976
+ visited_callees: set[str] = set()
977
+ for depth_step in range(effective_depth):
978
+ next_symbols = []
979
+ for sym in current_symbols:
980
+ if sym in visited_callees:
981
+ continue
982
+ visited_callees.add(sym)
983
+ callee_items = find_callees(con, sym, max_results=20)
984
+ for c in callee_items:
985
+ c_sym = str(c.get("qualified_callee") or c.get("callee") or "")
986
+ c_file = str(c.get("file") or "")
987
+ c_line = int(c.get("start_line") or c.get("line") or 1)
988
+ key = (c_sym, c_file, c_line, "CALLEE")
989
+ if key not in seen:
990
+ seen.add(key)
991
+ output.append(
992
+ {
993
+ "symbol": c_sym,
994
+ "file": c_file,
995
+ "start_line": c_line,
996
+ "end_line": int(c.get("end_line") or c_line),
997
+ "relationship": "CALLEE",
998
+ "confidence": str(c.get("confidence", "LOW")),
999
+ "evidence": str(c.get("evidence", f"Direct callee of '{sym}'")),
1000
+ "depth": depth_step + 1,
1001
+ }
1002
+ )
1003
+ if c_sym:
1004
+ next_symbols.append(c_sym)
1005
+ current_symbols = next_symbols
1006
+ if not current_symbols:
1007
+ break
1008
+
1009
+ return output
1010
+
1011
+
1012
+ def find_parallel_implementations(
1013
+ con: sqlite3.Connection,
1014
+ symbol: str,
1015
+ max_results: int = 5,
1016
+ ) -> list[dict[str, Any]]:
1017
+ """Discover parallel implementations sharing algorithmic or domain roles."""
1018
+ clean = symbol.split(".")[-1]
1019
+ row = con.execute(
1020
+ "SELECT canonical_id, qualified_name, name, path, kind, parameter_count, return_type, module "
1021
+ "FROM symbols WHERE canonical_id=? OR qualified_name=? OR name=? LIMIT 1",
1022
+ (symbol, symbol, clean),
1023
+ ).fetchone()
1024
+ if not row:
1025
+ return []
1026
+
1027
+ source_path = row["path"]
1028
+ source_param_count = row["parameter_count"]
1029
+ source_kind = row["kind"]
1030
+ source_name = row["name"]
1031
+ source_canon = row["canonical_id"]
1032
+
1033
+ results: list[dict[str, Any]] = []
1034
+ seen: set[str] = {source_canon, source_name}
1035
+
1036
+ # Strategy A: Check shared callers
1037
+ shared_callee_rows = con.execute(
1038
+ "SELECT DISTINCT c2.callee, c2.qualified_callee, c2.source_path, s.canonical_id, s.path, s.start_line, s.end_line "
1039
+ "FROM calls c1 "
1040
+ "JOIN calls c2 ON c1.source_path = c2.source_path "
1041
+ "JOIN symbols s ON (s.canonical_id = c2.qualified_callee OR s.name = c2.callee) "
1042
+ "WHERE (c1.callee=? OR c1.qualified_callee=?) AND c2.callee != ? "
1043
+ "AND s.kind=? "
1044
+ "LIMIT ?",
1045
+ (source_name, source_canon, source_name, source_kind, max_results),
1046
+ ).fetchall()
1047
+
1048
+ for sc in shared_callee_rows:
1049
+ canon = sc["canonical_id"]
1050
+ if canon in seen:
1051
+ continue
1052
+ seen.add(canon)
1053
+ results.append(
1054
+ {
1055
+ "symbol": sc["qualified_callee"] or sc["callee"],
1056
+ "canonical_id": canon,
1057
+ "file": sc["path"],
1058
+ "start_line": sc["start_line"],
1059
+ "end_line": sc["end_line"],
1060
+ "relationship": "PARALLEL_IMPLEMENTATION",
1061
+ "confidence": "MEDIUM",
1062
+ "evidence": f"Parallel implementation: shared caller context with '{source_name}'",
1063
+ }
1064
+ )
1065
+
1066
+ # Strategy B: Sibling functions in same file/module with matching parameter count
1067
+ if len(results) < max_results and source_param_count is not None and source_param_count > 0:
1068
+ sibling_rows = con.execute(
1069
+ "SELECT canonical_id, qualified_name, name, path, start_line, end_line "
1070
+ "FROM symbols WHERE path=? AND kind=? AND parameter_count=? AND canonical_id != ? "
1071
+ "LIMIT ?",
1072
+ (source_path, source_kind, source_param_count, source_canon, max_results - len(results)),
1073
+ ).fetchall()
1074
+ for sib in sibling_rows:
1075
+ canon = sib["canonical_id"]
1076
+ if canon in seen:
1077
+ continue
1078
+ seen.add(canon)
1079
+ results.append(
1080
+ {
1081
+ "symbol": sib["qualified_name"],
1082
+ "canonical_id": canon,
1083
+ "file": sib["path"],
1084
+ "start_line": sib["start_line"],
1085
+ "end_line": sib["end_line"],
1086
+ "relationship": "PARALLEL_IMPLEMENTATION",
1087
+ "confidence": "MEDIUM",
1088
+ "evidence": f"Parallel implementation: sibling {source_kind} with matching signature in {source_path}",
1089
+ }
1090
+ )
1091
+
1092
+ return results[:max_results]
1093
+
1094
+
1095
+ def get_graph_summary(con: sqlite3.Connection) -> dict[str, object]:
1096
+ """Return bounded repository graph summary statistics."""
1097
+ node_counts_by_type: dict[str, int] = {}
1098
+ for r in con.execute("SELECT kind, count(*) FROM symbols GROUP BY kind").fetchall():
1099
+ node_counts_by_type[r[0]] = r[1]
1100
+
1101
+ edge_counts_by_type: dict[str, int] = {
1102
+ "IMPORTS": con.execute("SELECT count(*) FROM imports").fetchone()[0],
1103
+ "CALLS": con.execute("SELECT count(*) FROM calls").fetchone()[0],
1104
+ "EXTENDS": con.execute("SELECT count(*) FROM inheritance").fetchone()[0],
1105
+ }
1106
+ try:
1107
+ edge_counts_by_type["HANDLED_BY"] = con.execute("SELECT count(*) FROM framework_routes").fetchone()[0]
1108
+ except sqlite3.OperationalError:
1109
+ pass
1110
+
1111
+ modules = [
1112
+ r[0]
1113
+ for r in con.execute(
1114
+ "SELECT DISTINCT module FROM symbols WHERE module != '' ORDER BY module LIMIT 50"
1115
+ ).fetchall()
1116
+ ]
1117
+
1118
+ dependencies = [
1119
+ {"module": r[0], "count": r[1]}
1120
+ for r in con.execute(
1121
+ "SELECT module, count(*) AS c FROM imports GROUP BY module ORDER BY c DESC LIMIT 20"
1122
+ ).fetchall()
1123
+ ]
1124
+
1125
+ route_count = 0
1126
+ try:
1127
+ route_count = con.execute("SELECT count(*) FROM framework_routes").fetchone()[0]
1128
+ except sqlite3.OperationalError:
1129
+ pass
1130
+
1131
+ test_count = 0
1132
+ try:
1133
+ test_count = con.execute(
1134
+ "SELECT count(*) FROM files WHERE category='TEST' OR path LIKE 'tests/%' OR path LIKE 'test/%'"
1135
+ ).fetchone()[0]
1136
+ except sqlite3.OperationalError:
1137
+ pass
1138
+
1139
+ return {
1140
+ "node_counts": node_counts_by_type,
1141
+ "edge_counts": edge_counts_by_type,
1142
+ "modules": modules,
1143
+ "top_dependencies": dependencies,
1144
+ "route_count": route_count,
1145
+ "test_count": test_count,
1146
+ "external_service_count": len([d for d in dependencies if "." not in d["module"] and "/" not in d["module"]]),
1147
+ "total_symbols": sum(node_counts_by_type.values()),
1148
+ "total_edges": sum(edge_counts_by_type.values()),
1149
+ }
1150
+
1151
+
1152
+ def get_focused_graph(
1153
+ con: sqlite3.Connection,
1154
+ module: str | None = None,
1155
+ symbol: str | None = None,
1156
+ depth: int = 2,
1157
+ ) -> dict[str, object]:
1158
+ """Return bounded focused graph for a module or symbol."""
1159
+ bounded_depth = max(1, min(depth, 5))
1160
+ nodes: list[dict[str, object]] = []
1161
+ edges: list[dict[str, object]] = []
1162
+ seen_nodes: set[str] = set()
1163
+ seen_edges: set[tuple[str, str, str]] = set()
1164
+
1165
+ if symbol:
1166
+ tr = trace_call(con, symbol, max_depth=bounded_depth, both=True)
1167
+ for item in tr[:100]:
1168
+ sym_name = str(item.get("symbol") or "")
1169
+ if sym_name and sym_name not in seen_nodes:
1170
+ seen_nodes.add(sym_name)
1171
+ nodes.append({
1172
+ "id": sym_name,
1173
+ "file": item.get("file"),
1174
+ "line": item.get("start_line"),
1175
+ "relationship": item.get("relationship"),
1176
+ })
1177
+ rel = str(item.get("relationship") or "")
1178
+ if rel in ("CALLER", "CALLEE", "HANDLED_BY"):
1179
+ src = sym_name if rel == "CALLEE" else symbol
1180
+ tgt = symbol if rel == "CALLEE" else sym_name
1181
+ edge_key = (src, tgt, rel)
1182
+ if edge_key not in seen_edges:
1183
+ seen_edges.add(edge_key)
1184
+ edges.append({
1185
+ "source": src,
1186
+ "target": tgt,
1187
+ "relationship": rel,
1188
+ "confidence": item.get("confidence", "HIGH"),
1189
+ "evidence": item.get("evidence", ""),
1190
+ })
1191
+ elif module:
1192
+ rows = con.execute(
1193
+ "SELECT canonical_id, qualified_name, kind, path, start_line, end_line "
1194
+ "FROM symbols WHERE module=? OR path LIKE ? LIMIT 50",
1195
+ (module, f"%{module}%"),
1196
+ ).fetchall()
1197
+ for r in rows:
1198
+ cid = r["canonical_id"]
1199
+ if cid not in seen_nodes:
1200
+ seen_nodes.add(cid)
1201
+ nodes.append({
1202
+ "id": cid,
1203
+ "name": r["qualified_name"],
1204
+ "kind": r["kind"],
1205
+ "file": r["path"],
1206
+ "line": r["start_line"],
1207
+ })
1208
+ for n in nodes[:30]:
1209
+ sym = str(n.get("name") or "")
1210
+ callees = find_callees(con, sym, max_results=10)
1211
+ for c in callees:
1212
+ c_sym = str(c.get("qualified_callee") or c.get("callee") or "")
1213
+ if c_sym in seen_nodes:
1214
+ edge_key = (sym, c_sym, "CALLS")
1215
+ if edge_key not in seen_edges:
1216
+ seen_edges.add(edge_key)
1217
+ edges.append({
1218
+ "source": sym,
1219
+ "target": c_sym,
1220
+ "relationship": "CALLS",
1221
+ "confidence": str(c.get("confidence", "HIGH")),
1222
+ "evidence": str(c.get("evidence", "")),
1223
+ })
1224
+
1225
+ return {
1226
+ "module": module,
1227
+ "symbol": symbol,
1228
+ "depth": bounded_depth,
1229
+ "nodes": sorted(nodes, key=lambda n: str(n.get("id"))),
1230
+ "edges": sorted(edges, key=lambda e: (str(e.get("source")), str(e.get("target")))),
1231
+ "node_count": len(nodes),
1232
+ "edge_count": len(edges),
1233
+ }
1234
+
1235
+
1236
+ def import_edges(con: sqlite3.Connection) -> list[GraphEdge]:
1237
+ """Return parser-extracted import relationships with evidence metadata."""
1238
+ rows = con.execute(
1239
+ "SELECT source_path, module, resolved_path, line FROM imports ORDER BY source_path, line"
1240
+ ).fetchall()
1241
+ return [
1242
+ GraphEdge(
1243
+ source=r["source_path"],
1244
+ target=r["resolved_path"] or r["module"],
1245
+ relationship="IMPORTS",
1246
+ confidence="HIGH",
1247
+ file=r["source_path"],
1248
+ start_line=r["line"],
1249
+ end_line=r["line"],
1250
+ evidence=f"Parser-extracted import '{r['module']}'",
1251
+ )
1252
+ for r in rows
1253
+ ]
1254
+
1255
+
1256
+ def definition_edges(con: sqlite3.Connection, limit: int = 500) -> list[GraphEdge]:
1257
+ """Return parser-confirmed definition and containment edges."""
1258
+ rows = con.execute(
1259
+ "SELECT canonical_id, qualified_name, kind, path, start_line, end_line, parent_symbol_id "
1260
+ "FROM symbols ORDER BY path, start_line LIMIT ?",
1261
+ (limit,),
1262
+ ).fetchall()
1263
+ edges: list[GraphEdge] = []
1264
+ for r in rows:
1265
+ if r["parent_symbol_id"]:
1266
+ edges.append(
1267
+ GraphEdge(
1268
+ source=r["parent_symbol_id"],
1269
+ target=r["canonical_id"],
1270
+ relationship="CONTAINS",
1271
+ confidence="HIGH",
1272
+ file=r["path"],
1273
+ start_line=r["start_line"],
1274
+ end_line=r["end_line"],
1275
+ evidence=f"Containment {r['parent_symbol_id']} -> {r['canonical_id']}",
1276
+ )
1277
+ )
1278
+ else:
1279
+ edges.append(
1280
+ GraphEdge(
1281
+ source=r["path"],
1282
+ target=r["canonical_id"],
1283
+ relationship="DEFINES",
1284
+ confidence="HIGH",
1285
+ file=r["path"],
1286
+ start_line=r["start_line"],
1287
+ end_line=r["end_line"],
1288
+ evidence=f"File {r['path']} defines {r['canonical_id']}",
1289
+ )
1290
+ )
1291
+ return edges