codegraph-engine 2.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. codegraph/__init__.py +37 -0
  2. codegraph/agent.py +26 -0
  3. codegraph/architecture.py +328 -0
  4. codegraph/audit.py +106 -0
  5. codegraph/cache.py +95 -0
  6. codegraph/cli.py +854 -0
  7. codegraph/config.py +43 -0
  8. codegraph/constraints.py +238 -0
  9. codegraph/context.py +1228 -0
  10. codegraph/epistemic.py +90 -0
  11. codegraph/errors.py +275 -0
  12. codegraph/evidence/__init__.py +15 -0
  13. codegraph/evidence/citations.py +397 -0
  14. codegraph/frameworks.py +434 -0
  15. codegraph/freshness.py +295 -0
  16. codegraph/git.py +278 -0
  17. codegraph/graph/__init__.py +46 -0
  18. codegraph/graph/models.py +41 -0
  19. codegraph/graph/traversal.py +1291 -0
  20. codegraph/indexing/__init__.py +4 -0
  21. codegraph/indexing/classifier.py +274 -0
  22. codegraph/indexing/indexer.py +943 -0
  23. codegraph/indexing/models.py +338 -0
  24. codegraph/indexing/parser.py +1240 -0
  25. codegraph/indexing/scanner.py +200 -0
  26. codegraph/indexing/test_framework.py +116 -0
  27. codegraph/interrogation.py +1582 -0
  28. codegraph/llm/__init__.py +3 -0
  29. codegraph/llm/base.py +15 -0
  30. codegraph/llm/context.py +20 -0
  31. codegraph/mcp/__init__.py +3 -0
  32. codegraph/mcp/server.py +736 -0
  33. codegraph/memory/__init__.py +3 -0
  34. codegraph/memory/store.py +46 -0
  35. codegraph/models.py +289 -0
  36. codegraph/observability.py +151 -0
  37. codegraph/optimizer.py +372 -0
  38. codegraph/planner.py +417 -0
  39. codegraph/py.typed +1 -0
  40. codegraph/query_expansion.py +199 -0
  41. codegraph/ranking.py +363 -0
  42. codegraph/resolver.py +843 -0
  43. codegraph/resources/__init__.py +45 -0
  44. codegraph/resources/cache.py +117 -0
  45. codegraph/resources/coalescer.py +83 -0
  46. codegraph/resources/debouncer.py +98 -0
  47. codegraph/resources/governor.py +232 -0
  48. codegraph/resources/policy.py +123 -0
  49. codegraph/retrieval_policy.py +220 -0
  50. codegraph/search/__init__.py +23 -0
  51. codegraph/search/hybrid.py +301 -0
  52. codegraph/search/semantic.py +28 -0
  53. codegraph/security/__init__.py +3 -0
  54. codegraph/security/paths.py +35 -0
  55. codegraph/target_resolver.py +348 -0
  56. codegraph/task.py +637 -0
  57. codegraph_engine-2.1.1.dist-info/METADATA +334 -0
  58. codegraph_engine-2.1.1.dist-info/RECORD +62 -0
  59. codegraph_engine-2.1.1.dist-info/WHEEL +5 -0
  60. codegraph_engine-2.1.1.dist-info/entry_points.txt +2 -0
  61. codegraph_engine-2.1.1.dist-info/licenses/LICENSE +21 -0
  62. codegraph_engine-2.1.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1582 @@
1
+ """Core deterministic repository interrogation engine for CodeGraph MCP.
2
+
3
+ Principle:
4
+ "The AI understands the developer. CodeGraph interrogates the repository."
5
+ "The claim comes from the AI; the proof comes from CodeGraph."
6
+
7
+ Invariant:
8
+ same repository + same index generation + same MCP request = same deterministic result.
9
+ """
10
+ from __future__ import annotations
11
+
12
+ import sqlite3
13
+ from collections import deque
14
+ from pathlib import Path
15
+ from typing import Any
16
+
17
+ from codegraph.architecture import get_architecture as arch_get_architecture
18
+ from codegraph.errors import ErrorCode, SecurityError
19
+ from codegraph.freshness import check_freshness, index_generation
20
+ from codegraph.git import (
21
+ changed_files,
22
+ changed_symbols_since,
23
+ current_commit,
24
+ )
25
+ from codegraph.security import is_sensitive, safe_path
26
+
27
+
28
+ def make_evidence(
29
+ file: str,
30
+ start_line: int,
31
+ end_line: int,
32
+ evidence_type: str,
33
+ canonical_id: str | None = None,
34
+ ) -> dict[str, Any]:
35
+ """Create a standardized, evidence-backed citation record."""
36
+ return {
37
+ "file": file,
38
+ "start_line": start_line,
39
+ "end_line": end_line,
40
+ "type": evidence_type,
41
+ "canonical_id": canonical_id or "",
42
+ }
43
+
44
+
45
+ def get_index_metadata(con: sqlite3.Connection, repository: Path) -> dict[str, Any]:
46
+ """Return index generation, freshness, and repository commit identity."""
47
+ gen = index_generation(con)
48
+ freshness_rep = check_freshness(repository, con)
49
+ commit = current_commit(repository)
50
+
51
+ created_at = ""
52
+ try:
53
+ row = con.execute("SELECT value FROM metadata WHERE key='indexed_at'").fetchone()
54
+ if row:
55
+ created_at = str(row[0])
56
+ except Exception:
57
+ pass
58
+
59
+ return {
60
+ "index": {
61
+ "generation": gen,
62
+ "created_at": created_at,
63
+ "freshness": freshness_rep.status.value,
64
+ },
65
+ "repository": {
66
+ "commit": commit,
67
+ },
68
+ }
69
+
70
+
71
+ def check_index_available(con: sqlite3.Connection, repository: Path) -> dict[str, Any] | None:
72
+ """Return an INDEX_NOT_FOUND error envelope if the repository has not been indexed."""
73
+ try:
74
+ count = con.execute("SELECT count(*) FROM files").fetchone()[0]
75
+ except sqlite3.OperationalError:
76
+ count = 0
77
+ if count == 0:
78
+ return {
79
+ "status": "error",
80
+ "error": {
81
+ "code": ErrorCode.INDEX_NOT_FOUND.value,
82
+ "message": "No CodeGraph index exists for this repository.",
83
+ "next_action": {
84
+ "command": "codegraph init",
85
+ "reason": "Initialize the repository before querying it.",
86
+ },
87
+ },
88
+ }
89
+ return None
90
+
91
+
92
+ # ---------------------------------------------------------------------------
93
+ # 1. resolve_symbol
94
+ # ---------------------------------------------------------------------------
95
+ def resolve_symbol(
96
+ con: sqlite3.Connection,
97
+ repository: Path,
98
+ name: str,
99
+ ) -> dict[str, Any]:
100
+ """Determine whether an exact/canonical symbol exists and return its location and identity.
101
+
102
+ CodeGraph MUST NOT choose an intended candidate when ambiguous.
103
+ """
104
+ unindexed = check_index_available(con, repository)
105
+ if unindexed:
106
+ return unindexed
107
+
108
+ clean_name = name.strip()
109
+ meta = get_index_metadata(con, repository)
110
+ if not clean_name:
111
+ return {
112
+ "status": "invalid_request",
113
+ "error_code": ErrorCode.EMPTY_SYMBOL_NAME.value,
114
+ "error": {
115
+ "code": ErrorCode.EMPTY_SYMBOL_NAME.value,
116
+ "message": "Symbol name must be non-empty.",
117
+ "next_action": {
118
+ "command": "codegraph resolve <symbol>",
119
+ "reason": "Provide a non-empty symbol name to resolve.",
120
+ },
121
+ },
122
+ "query": name,
123
+ **meta,
124
+ }
125
+
126
+ # Query matching symbols
127
+ rows = con.execute(
128
+ "SELECT canonical_id, name, qualified_name, kind, path, start_line, end_line "
129
+ "FROM symbols "
130
+ "WHERE canonical_id=? OR qualified_name=? OR name=? "
131
+ "ORDER BY canonical_id ASC",
132
+ (clean_name, clean_name, clean_name),
133
+ ).fetchall()
134
+
135
+ if not rows:
136
+ return {
137
+ "status": "not_found",
138
+ "query": clean_name,
139
+ "error": {
140
+ "code": ErrorCode.SYMBOL_NOT_FOUND.value,
141
+ "message": f"No matching symbol was found for '{clean_name}'.",
142
+ "next_action": {
143
+ "command": f"codegraph search {clean_name}",
144
+ "reason": "Search with a broader query term.",
145
+ },
146
+ },
147
+ **meta,
148
+ "matches": [],
149
+ "candidates": [],
150
+ }
151
+
152
+ # Check for exact canonical_id match
153
+ exact_canonical = [r for r in rows if r["canonical_id"] == clean_name]
154
+ if len(exact_canonical) == 1:
155
+ match = exact_canonical[0]
156
+ ev = [
157
+ make_evidence(
158
+ file=match["path"],
159
+ start_line=match["start_line"],
160
+ end_line=match["end_line"],
161
+ evidence_type="definition",
162
+ canonical_id=match["canonical_id"],
163
+ )
164
+ ]
165
+ return {
166
+ "status": "ok",
167
+ **meta,
168
+ "symbol": {
169
+ "name": match["name"],
170
+ "canonical_id": match["canonical_id"],
171
+ "kind": match["kind"],
172
+ },
173
+ "location": {
174
+ "file": match["path"],
175
+ "start_line": match["start_line"],
176
+ "end_line": match["end_line"],
177
+ },
178
+ "evidence": ev,
179
+ }
180
+
181
+ # If only one match overall
182
+ if len(rows) == 1:
183
+ match = rows[0]
184
+ ev = [
185
+ make_evidence(
186
+ file=match["path"],
187
+ start_line=match["start_line"],
188
+ end_line=match["end_line"],
189
+ evidence_type="definition",
190
+ canonical_id=match["canonical_id"],
191
+ )
192
+ ]
193
+ return {
194
+ "status": "ok",
195
+ **meta,
196
+ "symbol": {
197
+ "name": match["name"],
198
+ "canonical_id": match["canonical_id"],
199
+ "kind": match["kind"],
200
+ },
201
+ "location": {
202
+ "file": match["path"],
203
+ "start_line": match["start_line"],
204
+ "end_line": match["end_line"],
205
+ },
206
+ "evidence": ev,
207
+ }
208
+
209
+ # Multiple candidates exist -> ambiguous! CodeGraph MUST NOT choose!
210
+ matches = [
211
+ {
212
+ "canonical_id": r["canonical_id"],
213
+ "symbol": r["name"],
214
+ "name": r["name"],
215
+ "file": r["path"],
216
+ "kind": r["kind"],
217
+ "line": r["start_line"],
218
+ "start_line": r["start_line"],
219
+ "end_line": r["end_line"],
220
+ }
221
+ for r in rows
222
+ ]
223
+ matches.sort(key=lambda m: str(m["canonical_id"]))
224
+ return {
225
+ "status": "ambiguous",
226
+ "query": clean_name,
227
+ "error": {
228
+ "code": ErrorCode.SYMBOL_AMBIGUOUS.value,
229
+ "message": f"Multiple symbols match '{clean_name}'. Provide a qualified name or canonical ID.",
230
+ "next_action": {
231
+ "command": "codegraph resolve <canonical_id>",
232
+ "reason": "Select an explicit canonical ID from candidates.",
233
+ },
234
+ },
235
+ **meta,
236
+ "matches": matches,
237
+ "candidates": matches,
238
+ }
239
+
240
+
241
+ # ---------------------------------------------------------------------------
242
+ # 2. search_symbols
243
+ # ---------------------------------------------------------------------------
244
+ def search_symbols(
245
+ con: sqlite3.Connection,
246
+ repository: Path,
247
+ query: str,
248
+ top_k: int = 20,
249
+ ) -> dict[str, Any]:
250
+ """Search indexed repository symbols using explicit search terms with deterministic ranking."""
251
+ unindexed = check_index_available(con, repository)
252
+ if unindexed:
253
+ return unindexed
254
+
255
+ clean_query = query.strip()
256
+ meta = get_index_metadata(con, repository)
257
+ if not clean_query:
258
+ return {
259
+ "status": "error",
260
+ "error_code": ErrorCode.INVALID_ARGUMENT.value,
261
+ "error": {
262
+ "code": ErrorCode.INVALID_ARGUMENT.value,
263
+ "message": "Search query must be non-empty.",
264
+ "next_action": {
265
+ "command": "codegraph search <query>",
266
+ "reason": "Provide a non-empty query term.",
267
+ },
268
+ },
269
+ "query": query,
270
+ **meta,
271
+ "count": 0,
272
+ "symbols": [],
273
+ "evidence": [],
274
+ }
275
+
276
+ bounded_k = max(1, min(top_k, 100))
277
+
278
+ # Search symbols joined with files to exclude GENERATED files
279
+ rows = con.execute(
280
+ "SELECT s.canonical_id, s.name, s.qualified_name, s.kind, s.path, s.start_line, s.end_line, f.category "
281
+ "FROM symbols s "
282
+ "JOIN files f ON f.path = s.path "
283
+ "WHERE (f.category != 'GENERATED' OR f.category IS NULL) "
284
+ "AND (s.name LIKE ? OR s.qualified_name LIKE ? OR s.canonical_id LIKE ?) "
285
+ "ORDER BY s.canonical_id ASC",
286
+ (f"%{clean_query}%", f"%{clean_query}%", f"%{clean_query}%"),
287
+ ).fetchall()
288
+
289
+ # Deterministic scoring:
290
+ # 1.0 = exact match on name or qualified_name
291
+ # 0.85 = prefix match on name
292
+ # 0.70 = substring match on name
293
+ # 0.60 = match elsewhere
294
+ scored_results: list[dict[str, Any]] = []
295
+ q_lower = clean_query.lower()
296
+
297
+ for r in rows:
298
+ name = r["name"]
299
+ qname = r["qualified_name"] or ""
300
+ cid = r["canonical_id"] or ""
301
+ name_lower = name.lower()
302
+
303
+ if name_lower == q_lower or qname.lower() == q_lower:
304
+ score = 1.00
305
+ elif name_lower.startswith(q_lower):
306
+ score = 0.85
307
+ elif q_lower in name_lower:
308
+ score = 0.70
309
+ else:
310
+ score = 0.60
311
+
312
+ scored_results.append(
313
+ {
314
+ "name": name,
315
+ "canonical_id": cid,
316
+ "kind": r["kind"],
317
+ "file": r["path"],
318
+ "start_line": r["start_line"],
319
+ "end_line": r["end_line"],
320
+ "score": round(score, 2),
321
+ "evidence": make_evidence(
322
+ file=r["path"],
323
+ start_line=r["start_line"],
324
+ end_line=r["end_line"],
325
+ evidence_type="symbol_match",
326
+ canonical_id=cid,
327
+ ),
328
+ }
329
+ )
330
+
331
+ # Sort deterministically: highest score first, ties broken by canonical_id ASC
332
+ scored_results.sort(key=lambda s: (-float(s["score"]), str(s["canonical_id"])))
333
+ selected = scored_results[:bounded_k]
334
+
335
+ evidence_list = [s["evidence"] for s in selected]
336
+
337
+ return {
338
+ "status": "ok",
339
+ "query": clean_query,
340
+ **meta,
341
+ "count": len(selected),
342
+ "symbols": selected,
343
+ "evidence": evidence_list,
344
+ }
345
+
346
+
347
+ # ---------------------------------------------------------------------------
348
+ # 3. get_symbol
349
+ # ---------------------------------------------------------------------------
350
+ def get_symbol(
351
+ con: sqlite3.Connection,
352
+ repository: Path,
353
+ canonical_id: str,
354
+ ) -> dict[str, Any]:
355
+ """Return complete structured information for a known canonical symbol."""
356
+ unindexed = check_index_available(con, repository)
357
+ if unindexed:
358
+ return unindexed
359
+
360
+ clean_id = canonical_id.strip()
361
+ meta = get_index_metadata(con, repository)
362
+ if not clean_id:
363
+ return {
364
+ "status": "error",
365
+ "error_code": ErrorCode.INVALID_ARGUMENT.value,
366
+ "error": {
367
+ "code": ErrorCode.INVALID_ARGUMENT.value,
368
+ "message": "Canonical ID or symbol name must be non-empty.",
369
+ "next_action": {
370
+ "command": "codegraph get-symbol <symbol>",
371
+ "reason": "Provide a non-empty canonical ID or symbol name.",
372
+ },
373
+ },
374
+ "canonical_id": canonical_id,
375
+ **meta,
376
+ }
377
+
378
+ row = con.execute(
379
+ "SELECT id, canonical_id, qualified_name, name, kind, path, start_line, end_line, "
380
+ "parent_symbol_id, module, scope, language, signature, content_hash, visibility, "
381
+ "return_type, parameter_count, documentation, decorators "
382
+ "FROM symbols "
383
+ "WHERE canonical_id=? OR qualified_name=? OR name=? "
384
+ "ORDER BY canonical_id=? DESC "
385
+ "LIMIT 1",
386
+ (clean_id, clean_id, clean_id, clean_id),
387
+ ).fetchone()
388
+
389
+ if not row:
390
+ return {
391
+ "status": "not_found",
392
+ "canonical_id": clean_id,
393
+ "error": {
394
+ "code": ErrorCode.SYMBOL_NOT_FOUND.value,
395
+ "message": f"No symbol found with canonical ID or name: '{clean_id}'",
396
+ "next_action": {
397
+ "command": f"codegraph search {clean_id}",
398
+ "reason": "Search for symbol candidates.",
399
+ },
400
+ },
401
+ **meta,
402
+ }
403
+
404
+ # Query parent symbol
405
+ parent_info: str | None = None
406
+ if row["parent_symbol_id"]:
407
+ p_row = con.execute(
408
+ "SELECT canonical_id, qualified_name FROM symbols WHERE id=?",
409
+ (row["parent_symbol_id"],),
410
+ ).fetchone()
411
+ if p_row:
412
+ parent_info = p_row["canonical_id"] or p_row["qualified_name"]
413
+
414
+ # Query children
415
+ children_rows = con.execute(
416
+ "SELECT canonical_id, name, kind, start_line, end_line FROM symbols "
417
+ "WHERE parent_symbol_id=? ORDER BY start_line ASC, canonical_id ASC",
418
+ (row["id"],),
419
+ ).fetchall()
420
+ children = [
421
+ {
422
+ "canonical_id": c["canonical_id"],
423
+ "name": c["name"],
424
+ "kind": c["kind"],
425
+ "start_line": c["start_line"],
426
+ "end_line": c["end_line"],
427
+ }
428
+ for c in children_rows
429
+ ]
430
+
431
+ # Parse decorators
432
+ decors_raw = row["decorators"] or ""
433
+ decorators: list[str] = []
434
+ if decors_raw:
435
+ decorators = [d.strip() for d in decors_raw.split(",") if d.strip()]
436
+
437
+ # Query relationships from graph_edges
438
+ cid = row["canonical_id"]
439
+ rel_rows = con.execute(
440
+ "SELECT target, relationship, confidence, file, start_line, end_line "
441
+ "FROM graph_edges WHERE source=? ORDER BY relationship ASC, target ASC",
442
+ (cid,),
443
+ ).fetchall()
444
+ relationships = [
445
+ {
446
+ "target": r["target"],
447
+ "relationship": r["relationship"],
448
+ "confidence": r["confidence"],
449
+ "file": r["file"],
450
+ "start_line": r["start_line"],
451
+ "end_line": r["end_line"],
452
+ }
453
+ for r in rel_rows
454
+ ]
455
+
456
+ ev = [
457
+ make_evidence(
458
+ file=row["path"],
459
+ start_line=row["start_line"],
460
+ end_line=row["end_line"],
461
+ evidence_type="definition",
462
+ canonical_id=row["canonical_id"],
463
+ )
464
+ ]
465
+
466
+ return {
467
+ "status": "ok",
468
+ **meta,
469
+ "evidence": ev,
470
+ "symbol": {
471
+ "canonical_id": row["canonical_id"],
472
+ "name": row["name"],
473
+ "qualified_name": row["qualified_name"],
474
+ "kind": row["kind"],
475
+ "file": row["path"],
476
+ "start_line": row["start_line"],
477
+ "end_line": row["end_line"],
478
+ "module": row["module"],
479
+ "scope": row["scope"],
480
+ "language": row["language"],
481
+ "signature": row["signature"] or "",
482
+ "parent": parent_info,
483
+ "children": children,
484
+ "decorators": decorators,
485
+ "docstring": row["documentation"] or "",
486
+ "visibility": row["visibility"] or "public",
487
+ "relationships": relationships,
488
+ },
489
+ }
490
+
491
+
492
+ # ---------------------------------------------------------------------------
493
+ # 4. get_file
494
+ # ---------------------------------------------------------------------------
495
+ def get_file(
496
+ con: sqlite3.Connection,
497
+ repository: Path,
498
+ path: str,
499
+ include_content: bool = False,
500
+ ) -> dict[str, Any]:
501
+ """Return the structural AST representation of an indexed file."""
502
+ unindexed = check_index_available(con, repository)
503
+ if unindexed:
504
+ return unindexed
505
+
506
+ meta = get_index_metadata(con, repository)
507
+ try:
508
+ resolved = safe_path(repository, path)
509
+ relative = resolved.relative_to(repository).as_posix()
510
+ except SecurityError:
511
+ return {
512
+ "status": "error",
513
+ "error_code": ErrorCode.PATH_OUTSIDE_REPOSITORY.value,
514
+ "error": {
515
+ "code": ErrorCode.PATH_OUTSIDE_REPOSITORY.value,
516
+ "message": f"Path traversal blocked: '{path}' resides outside repository boundary.",
517
+ "next_action": {
518
+ "command": "codegraph status",
519
+ "reason": "Ensure all queried paths reside within the repository root.",
520
+ },
521
+ },
522
+ "path": path,
523
+ **meta,
524
+ }
525
+ except Exception:
526
+ return {
527
+ "status": "error",
528
+ "error_code": ErrorCode.INVALID_PATH.value,
529
+ "error": {
530
+ "code": ErrorCode.INVALID_PATH.value,
531
+ "message": f"Invalid path: '{path}'",
532
+ "next_action": {
533
+ "command": "codegraph status",
534
+ "reason": "Check repository files and paths.",
535
+ },
536
+ },
537
+ "path": path,
538
+ **meta,
539
+ }
540
+
541
+ if is_sensitive(Path(relative)):
542
+ return {
543
+ "status": "invalid_request",
544
+ "error_code": ErrorCode.SENSITIVE_FILE_ACCESS_DENIED.value,
545
+ "error": {
546
+ "code": ErrorCode.SENSITIVE_FILE_ACCESS_DENIED.value,
547
+ "message": f"Access to sensitive file '{relative}' is blocked.",
548
+ "next_action": {
549
+ "command": "codegraph privacy",
550
+ "reason": "Review protected file boundaries.",
551
+ },
552
+ },
553
+ "path": relative,
554
+ **meta,
555
+ }
556
+
557
+ file_row = con.execute(
558
+ "SELECT hash, language, category FROM files WHERE path=?",
559
+ (relative,),
560
+ ).fetchone()
561
+
562
+ if not file_row and not resolved.is_file():
563
+ return {
564
+ "status": "not_found",
565
+ "error_code": ErrorCode.INVALID_PATH.value,
566
+ "error": {
567
+ "code": ErrorCode.INVALID_PATH.value,
568
+ "message": f"File does not exist: '{relative}'",
569
+ "next_action": {
570
+ "command": "codegraph status",
571
+ "reason": "Check repository files and paths.",
572
+ },
573
+ },
574
+ "path": relative,
575
+ **meta,
576
+ }
577
+
578
+ # Query symbols in file
579
+ symbols_rows = con.execute(
580
+ "SELECT canonical_id, name, qualified_name, kind, start_line, end_line "
581
+ "FROM symbols WHERE path=? ORDER BY start_line ASC, canonical_id ASC",
582
+ (relative,),
583
+ ).fetchall()
584
+
585
+ symbols = [
586
+ {
587
+ "canonical_id": s["canonical_id"],
588
+ "name": s["name"],
589
+ "qualified_name": s["qualified_name"],
590
+ "kind": s["kind"],
591
+ "start_line": s["start_line"],
592
+ "end_line": s["end_line"],
593
+ }
594
+ for s in symbols_rows
595
+ ]
596
+ classes = [s for s in symbols if s["kind"] in ("class", "interface")]
597
+ functions = [s for s in symbols if s["kind"] in ("function", "method")]
598
+
599
+ # Query imports
600
+ import_rows = con.execute(
601
+ "SELECT module, name, alias, line, imported_module, imported_name, resolved_path "
602
+ "FROM imports WHERE source_path=? ORDER BY line ASC, module ASC",
603
+ (relative,),
604
+ ).fetchall()
605
+ imports = [
606
+ {
607
+ "module": i["module"],
608
+ "name": i["name"],
609
+ "alias": i["alias"],
610
+ "line": i["line"],
611
+ "imported_module": i["imported_module"],
612
+ "imported_name": i["imported_name"],
613
+ "resolved_path": i["resolved_path"],
614
+ }
615
+ for i in import_rows
616
+ ]
617
+
618
+ # Query routes in file
619
+ route_rows = con.execute(
620
+ "SELECT endpoint_id, framework, http_method, route_path, handler_name, line "
621
+ "FROM framework_routes WHERE file_path=? ORDER BY line ASC, route_path ASC",
622
+ (relative,),
623
+ ).fetchall()
624
+ routes = [
625
+ {
626
+ "endpoint_id": r["endpoint_id"],
627
+ "framework": r["framework"],
628
+ "method": r["http_method"],
629
+ "path": r["route_path"],
630
+ "handler": r["handler_name"],
631
+ "line": r["line"],
632
+ }
633
+ for r in route_rows
634
+ ]
635
+
636
+ file_size = resolved.stat().st_size if resolved.exists() else 0
637
+ file_hash = file_row["hash"] if file_row else ""
638
+ language = file_row["language"] if file_row else "unknown"
639
+
640
+ content: str | None = None
641
+ if include_content and resolved.exists():
642
+ content = resolved.read_text(encoding="utf-8", errors="replace")
643
+ if len(content) > 50_000:
644
+ content = content[:50_000] + "\n...[truncated at 50,000 chars]"
645
+
646
+ ev = [
647
+ make_evidence(
648
+ file=relative,
649
+ start_line=1,
650
+ end_line=len(symbols_rows) or 1,
651
+ evidence_type="file_ast",
652
+ canonical_id=relative,
653
+ )
654
+ ]
655
+
656
+ result: dict[str, Any] = {
657
+ "status": "ok",
658
+ "file": relative,
659
+ **meta,
660
+ "hash": file_hash,
661
+ "language": language,
662
+ "size_bytes": file_size,
663
+ "classes": classes,
664
+ "functions": functions,
665
+ "imports": imports,
666
+ "routes": routes,
667
+ "evidence": ev,
668
+ }
669
+ if include_content:
670
+ result["content"] = content
671
+ return result
672
+
673
+
674
+ # ---------------------------------------------------------------------------
675
+ # 5. get_references
676
+ # ---------------------------------------------------------------------------
677
+ def get_references(
678
+ con: sqlite3.Connection,
679
+ repository: Path,
680
+ canonical_id: str,
681
+ ) -> dict[str, Any]:
682
+ """Return all known references and call sites to a canonical symbol with evidence."""
683
+ unindexed = check_index_available(con, repository)
684
+ if unindexed:
685
+ return unindexed
686
+
687
+ clean_id = canonical_id.strip()
688
+ meta = get_index_metadata(con, repository)
689
+ if not clean_id:
690
+ return {
691
+ "status": "error",
692
+ "error_code": ErrorCode.INVALID_ARGUMENT.value,
693
+ "error": {
694
+ "code": ErrorCode.INVALID_ARGUMENT.value,
695
+ "message": "Canonical ID must be non-empty.",
696
+ "next_action": {
697
+ "command": "codegraph resolve <symbol>",
698
+ "reason": "Resolve canonical ID before querying references.",
699
+ },
700
+ },
701
+ "canonical_id": canonical_id,
702
+ **meta,
703
+ "count": 0,
704
+ "references": [],
705
+ }
706
+
707
+ short_name = clean_id.split(":")[-1].split(".")[-1]
708
+
709
+ # 1. Search 'references' table
710
+ ref_rows = con.execute(
711
+ "SELECT source_symbol_id, relationship, confidence, path, start_line, end_line, evidence "
712
+ "FROM 'references' "
713
+ "WHERE target_symbol_id=? OR target_symbol_id LIKE ? "
714
+ "ORDER BY path ASC, start_line ASC",
715
+ (clean_id, f"%.{clean_id}"),
716
+ ).fetchall()
717
+
718
+ # 2. Search 'calls' table
719
+ call_rows = con.execute(
720
+ "SELECT source_path, callee, line, confidence, source_symbol_id "
721
+ "FROM calls "
722
+ "WHERE resolved_symbol_id=? OR qualified_callee=? OR callee=? "
723
+ "ORDER BY source_path ASC, line ASC",
724
+ (clean_id, clean_id, short_name),
725
+ ).fetchall()
726
+
727
+ references: list[dict[str, Any]] = []
728
+ seen: set[tuple[str, int, str]] = set()
729
+
730
+ for r in ref_rows:
731
+ key = (r["path"], r["start_line"], r["source_symbol_id"] or "")
732
+ if key not in seen:
733
+ seen.add(key)
734
+ references.append(
735
+ {
736
+ "canonical_id": r["source_symbol_id"] or "",
737
+ "file": r["path"],
738
+ "start_line": r["start_line"],
739
+ "end_line": r["end_line"],
740
+ "reference_type": r["relationship"].lower(),
741
+ "confidence": r["confidence"],
742
+ "evidence": make_evidence(
743
+ file=r["path"],
744
+ start_line=r["start_line"],
745
+ end_line=r["end_line"],
746
+ evidence_type="reference",
747
+ canonical_id=r["source_symbol_id"],
748
+ ),
749
+ }
750
+ )
751
+
752
+ for c in call_rows:
753
+ key = (c["source_path"], c["line"], c["source_symbol_id"] or "")
754
+ if key not in seen:
755
+ seen.add(key)
756
+ references.append(
757
+ {
758
+ "canonical_id": c["source_symbol_id"] or "",
759
+ "file": c["source_path"],
760
+ "start_line": c["line"],
761
+ "end_line": c["line"],
762
+ "reference_type": "call",
763
+ "confidence": c["confidence"],
764
+ "evidence": make_evidence(
765
+ file=c["source_path"],
766
+ start_line=c["line"],
767
+ end_line=c["line"],
768
+ evidence_type="call",
769
+ canonical_id=c["source_symbol_id"],
770
+ ),
771
+ }
772
+ )
773
+
774
+ references.sort(key=lambda r: (str(r["file"]), int(r["start_line"]), str(r["canonical_id"])))
775
+
776
+ return {
777
+ "status": "ok",
778
+ "canonical_id": clean_id,
779
+ **meta,
780
+ "count": len(references),
781
+ "references": references,
782
+ }
783
+
784
+
785
+ # ---------------------------------------------------------------------------
786
+ # 6. get_callers
787
+ # ---------------------------------------------------------------------------
788
+ def get_callers(
789
+ con: sqlite3.Connection,
790
+ repository: Path,
791
+ canonical_id: str,
792
+ ) -> dict[str, Any]:
793
+ """Return functions and methods that call the specified canonical symbol."""
794
+ unindexed = check_index_available(con, repository)
795
+ if unindexed:
796
+ return unindexed
797
+
798
+ clean_id = canonical_id.strip()
799
+ meta = get_index_metadata(con, repository)
800
+ if not clean_id:
801
+ return {
802
+ "status": "error",
803
+ "error_code": ErrorCode.INVALID_ARGUMENT.value,
804
+ "error": {
805
+ "code": ErrorCode.INVALID_ARGUMENT.value,
806
+ "message": "Canonical ID must be non-empty.",
807
+ "next_action": {
808
+ "command": "codegraph resolve <symbol>",
809
+ "reason": "Resolve canonical ID before querying callers.",
810
+ },
811
+ },
812
+ "canonical_id": canonical_id,
813
+ **meta,
814
+ "count": 0,
815
+ "callers": [],
816
+ }
817
+
818
+ short_name = clean_id.split(":")[-1].split(".")[-1]
819
+
820
+ # Query calls joined with symbols to get authoritative caller canonical IDs
821
+ rows = con.execute(
822
+ "SELECT c.source_path, c.callee, c.qualified_callee, c.line, c.confidence, "
823
+ "c.source_symbol_id, s.canonical_id AS caller_canon, s.name AS caller_name "
824
+ "FROM calls c "
825
+ "LEFT JOIN symbols s ON s.canonical_id = c.source_symbol_id "
826
+ "WHERE c.resolved_symbol_id=? OR c.qualified_callee=? OR c.callee=? "
827
+ "ORDER BY c.source_path ASC, c.line ASC",
828
+ (clean_id, clean_id, short_name),
829
+ ).fetchall()
830
+
831
+ callers: list[dict[str, Any]] = []
832
+ seen: set[tuple[str, int, str]] = set()
833
+
834
+ for r in rows:
835
+ caller_id = r["caller_canon"] or r["source_symbol_id"] or r["caller_name"] or r["source_path"]
836
+ key = (r["source_path"], r["line"], caller_id)
837
+ if key not in seen:
838
+ seen.add(key)
839
+ callers.append(
840
+ {
841
+ "caller": caller_id,
842
+ "canonical_id": r["caller_canon"] or r["source_symbol_id"] or "",
843
+ "file": r["source_path"],
844
+ "line": r["line"],
845
+ "confidence": r["confidence"],
846
+ "call_site": {
847
+ "file": r["source_path"],
848
+ "line": r["line"],
849
+ "callee": r["callee"],
850
+ },
851
+ "evidence": make_evidence(
852
+ file=r["source_path"],
853
+ start_line=r["line"],
854
+ end_line=r["line"],
855
+ evidence_type="call",
856
+ canonical_id=caller_id,
857
+ ),
858
+ }
859
+ )
860
+
861
+ callers.sort(key=lambda c: (str(c["file"]), int(c["line"]), str(c["caller"])))
862
+
863
+ return {
864
+ "status": "ok",
865
+ "canonical_id": clean_id,
866
+ **meta,
867
+ "count": len(callers),
868
+ "callers": callers,
869
+ }
870
+
871
+
872
+ # ---------------------------------------------------------------------------
873
+ # 7. get_callees
874
+ # ---------------------------------------------------------------------------
875
+ def get_callees(
876
+ con: sqlite3.Connection,
877
+ repository: Path,
878
+ canonical_id: str,
879
+ ) -> dict[str, Any]:
880
+ """Return functions and methods called by the specified canonical symbol."""
881
+ unindexed = check_index_available(con, repository)
882
+ if unindexed:
883
+ return unindexed
884
+
885
+ clean_id = canonical_id.strip()
886
+ meta = get_index_metadata(con, repository)
887
+ if not clean_id:
888
+ return {
889
+ "status": "error",
890
+ "error_code": ErrorCode.INVALID_ARGUMENT.value,
891
+ "error": {
892
+ "code": ErrorCode.INVALID_ARGUMENT.value,
893
+ "message": "Canonical ID must be non-empty.",
894
+ "next_action": {
895
+ "command": "codegraph resolve <symbol>",
896
+ "reason": "Resolve canonical ID before querying callees.",
897
+ },
898
+ },
899
+ "canonical_id": canonical_id,
900
+ **meta,
901
+ "count": 0,
902
+ "callees": [],
903
+ }
904
+
905
+ # Find the symbol record to get line bounds
906
+ sym_row = con.execute(
907
+ "SELECT id, canonical_id, path, start_line, end_line FROM symbols "
908
+ "WHERE canonical_id=? OR qualified_name=? OR name=? LIMIT 1",
909
+ (clean_id, clean_id, clean_id),
910
+ ).fetchone()
911
+
912
+ query_id = sym_row["canonical_id"] if sym_row else clean_id
913
+ query_path = sym_row["path"] if sym_row else None
914
+
915
+ if sym_row and query_path:
916
+ rows = con.execute(
917
+ "SELECT c.source_path, c.callee, c.qualified_callee, c.line, c.confidence, "
918
+ "c.source_symbol_id, c.resolved_symbol_id, "
919
+ "s.canonical_id AS callee_canon, s.path AS callee_path "
920
+ "FROM calls c "
921
+ "LEFT JOIN symbols s ON (s.canonical_id = c.resolved_symbol_id OR s.qualified_name = c.qualified_callee) "
922
+ "WHERE (c.source_symbol_id=? OR (c.source_path=? AND c.line >= ? AND c.line <= ?)) "
923
+ "ORDER BY c.line ASC",
924
+ (query_id, query_path, sym_row["start_line"], sym_row["end_line"]),
925
+ ).fetchall()
926
+ else:
927
+ rows = con.execute(
928
+ "SELECT c.source_path, c.callee, c.qualified_callee, c.line, c.confidence, "
929
+ "c.source_symbol_id, c.resolved_symbol_id, "
930
+ "s.canonical_id AS callee_canon, s.path AS callee_path "
931
+ "FROM calls c "
932
+ "LEFT JOIN symbols s ON (s.canonical_id = c.resolved_symbol_id OR s.qualified_name = c.qualified_callee) "
933
+ "WHERE c.source_symbol_id=? "
934
+ "ORDER BY c.line ASC",
935
+ (query_id,),
936
+ ).fetchall()
937
+
938
+ callees: list[dict[str, Any]] = []
939
+ seen: set[tuple[str, int, str]] = set()
940
+
941
+ for r in rows:
942
+ callee_name = r["callee"]
943
+ target_cid = r["callee_canon"] or r["resolved_symbol_id"]
944
+ if target_cid or r["callee_path"]:
945
+ call_type = "resolved_call"
946
+ elif "." in str(r["qualified_callee"] or ""):
947
+ call_type = "external_call"
948
+ else:
949
+ call_type = "unresolved_call"
950
+
951
+ key = (r["source_path"], r["line"], callee_name)
952
+ if key not in seen:
953
+ seen.add(key)
954
+ callees.append(
955
+ {
956
+ "callee": callee_name,
957
+ "canonical_id": target_cid,
958
+ "call_type": call_type,
959
+ "file": r["source_path"],
960
+ "line": r["line"],
961
+ "confidence": r["confidence"],
962
+ "evidence": make_evidence(
963
+ file=r["source_path"],
964
+ start_line=r["line"],
965
+ end_line=r["line"],
966
+ evidence_type="call",
967
+ canonical_id=target_cid or query_id,
968
+ ),
969
+ }
970
+ )
971
+
972
+ callees.sort(key=lambda c: (str(c["call_type"]), str(c["canonical_id"] or ""), int(c["line"])))
973
+
974
+ return {
975
+ "status": "ok",
976
+ "canonical_id": clean_id,
977
+ **meta,
978
+ "count": len(callees),
979
+ "callees": callees,
980
+ "resolved_calls": [c for c in callees if c["call_type"] == "resolved_call"],
981
+ "unresolved_calls": [c for c in callees if c["call_type"] == "unresolved_call"],
982
+ "external_calls": [c for c in callees if c["call_type"] == "external_call"],
983
+ }
984
+
985
+
986
+ # ---------------------------------------------------------------------------
987
+ # 8. trace_path
988
+ # ---------------------------------------------------------------------------
989
+ def trace_path(
990
+ con: sqlite3.Connection,
991
+ repository: Path,
992
+ from_symbol: str,
993
+ to_symbol: str,
994
+ max_depth: int = 5,
995
+ ) -> dict[str, Any]:
996
+ """Find a deterministic relationship path between two known symbols."""
997
+ unindexed = check_index_available(con, repository)
998
+ if unindexed:
999
+ return unindexed
1000
+
1001
+ clean_from = from_symbol.strip()
1002
+ clean_to = to_symbol.strip()
1003
+ meta = get_index_metadata(con, repository)
1004
+
1005
+ if max_depth < 1 or max_depth > 10:
1006
+ return {
1007
+ "status": "error",
1008
+ "error_code": ErrorCode.INVALID_DEPTH.value,
1009
+ "error": {
1010
+ "code": ErrorCode.INVALID_DEPTH.value,
1011
+ "message": f"Invalid graph traversal depth {max_depth}; allowed range is 1..5.",
1012
+ "next_action": {
1013
+ "command": "codegraph trace --depth 2 <symbol>",
1014
+ "reason": "Specify a traversal depth between 1 and 5.",
1015
+ },
1016
+ },
1017
+ "from": from_symbol,
1018
+ "to": to_symbol,
1019
+ **meta,
1020
+ "path": [],
1021
+ }
1022
+
1023
+ bounded_depth = max(1, min(max_depth, 5))
1024
+
1025
+ if not clean_from or not clean_to:
1026
+ return {
1027
+ "status": "error",
1028
+ "error_code": ErrorCode.INVALID_ARGUMENT.value,
1029
+ "error": {
1030
+ "code": ErrorCode.INVALID_ARGUMENT.value,
1031
+ "message": "Both source_symbol and target_symbol must be non-empty.",
1032
+ "next_action": {
1033
+ "command": "codegraph trace <source> <target>",
1034
+ "reason": "Provide valid source and target symbol names.",
1035
+ },
1036
+ },
1037
+ "from": from_symbol,
1038
+ "to": to_symbol,
1039
+ **meta,
1040
+ "path": [],
1041
+ }
1042
+
1043
+ # Resolve from and to targets
1044
+ from_row = con.execute(
1045
+ "SELECT canonical_id, qualified_name, name FROM symbols "
1046
+ "WHERE canonical_id=? OR qualified_name=? OR name=? LIMIT 1",
1047
+ (clean_from, clean_from, clean_from),
1048
+ ).fetchone()
1049
+
1050
+ to_row = con.execute(
1051
+ "SELECT canonical_id, qualified_name, name FROM symbols "
1052
+ "WHERE canonical_id=? OR qualified_name=? OR name=? LIMIT 1",
1053
+ (clean_to, clean_to, clean_to),
1054
+ ).fetchone()
1055
+
1056
+ if not from_row or not to_row:
1057
+ missing = clean_from if not from_row else clean_to
1058
+ return {
1059
+ "status": "not_found",
1060
+ "error": {
1061
+ "code": ErrorCode.SYMBOL_NOT_FOUND.value,
1062
+ "message": f"Symbol not found: '{missing}'",
1063
+ "next_action": {
1064
+ "command": f"codegraph search {missing}",
1065
+ "reason": "Verify symbol exists in the repository.",
1066
+ },
1067
+ },
1068
+ "from": clean_from,
1069
+ "to": clean_to,
1070
+ **meta,
1071
+ "path": [],
1072
+ }
1073
+
1074
+ start_canon = from_row["canonical_id"]
1075
+ target_canon = to_row["canonical_id"]
1076
+ target_names = {to_row["canonical_id"], to_row["qualified_name"], to_row["name"]}
1077
+
1078
+ if start_canon == target_canon:
1079
+ return {
1080
+ "status": "ok",
1081
+ "from": clean_from,
1082
+ "to": clean_to,
1083
+ **meta,
1084
+ "path_length": 0,
1085
+ "path": [],
1086
+ }
1087
+
1088
+ # Deterministic BFS search
1089
+ queue: deque[tuple[str, list[dict[str, Any]]]] = deque([(start_canon, [])])
1090
+ visited: set[str] = {start_canon}
1091
+
1092
+ while queue:
1093
+ curr_sym, path = queue.popleft()
1094
+ if len(path) >= bounded_depth:
1095
+ continue
1096
+
1097
+ curr_short = curr_sym.split(":")[-1].split(".")[-1]
1098
+
1099
+ # Query outgoing edges from calls
1100
+ edges = con.execute(
1101
+ "SELECT c.callee, c.qualified_callee, c.resolved_symbol_id, c.source_path, c.line "
1102
+ "FROM calls c "
1103
+ "WHERE c.source_symbol_id=? OR c.callee=? "
1104
+ "ORDER BY c.source_path ASC, c.line ASC",
1105
+ (curr_sym, curr_short),
1106
+ ).fetchall()
1107
+
1108
+ # Query outgoing edges from graph_edges
1109
+ graph_rows = con.execute(
1110
+ "SELECT target, relationship, file, start_line, end_line "
1111
+ "FROM graph_edges "
1112
+ "WHERE source=? AND relationship IN ('CALLS', 'IMPORTS', 'HANDLED_BY') "
1113
+ "ORDER BY relationship ASC, target ASC",
1114
+ (curr_sym,),
1115
+ ).fetchall()
1116
+
1117
+ all_steps: list[tuple[str, str, str, int, int]] = []
1118
+ for e in edges:
1119
+ dest = e["resolved_symbol_id"] or e["qualified_callee"] or e["callee"]
1120
+ all_steps.append((dest, "CALLS", e["source_path"], e["line"], e["line"]))
1121
+ for g in graph_rows:
1122
+ all_steps.append((g["target"], g["relationship"], g["file"], g["start_line"], g["end_line"]))
1123
+
1124
+ # Sort steps deterministically
1125
+ all_steps.sort(key=lambda s: (s[0], s[1], s[2], s[3]))
1126
+
1127
+ for dest_sym, rel, s_file, s_line, e_line in all_steps:
1128
+ dest_row = con.execute(
1129
+ "SELECT canonical_id FROM symbols WHERE canonical_id=? OR qualified_name=? OR name=? LIMIT 1",
1130
+ (dest_sym, dest_sym, dest_sym),
1131
+ ).fetchone()
1132
+ dest_canon = dest_row["canonical_id"] if dest_row else dest_sym
1133
+
1134
+ new_path = list(path) + [
1135
+ {
1136
+ "source": curr_sym,
1137
+ "target": dest_canon,
1138
+ "relationship": rel,
1139
+ "evidence": make_evidence(
1140
+ file=s_file,
1141
+ start_line=s_line,
1142
+ end_line=e_line,
1143
+ evidence_type="path_edge",
1144
+ canonical_id=dest_canon,
1145
+ ),
1146
+ }
1147
+ ]
1148
+
1149
+ if dest_canon == target_canon or dest_sym in target_names:
1150
+ return {
1151
+ "status": "ok",
1152
+ "from": clean_from,
1153
+ "to": clean_to,
1154
+ **meta,
1155
+ "path_length": len(new_path),
1156
+ "path": new_path,
1157
+ }
1158
+
1159
+ if dest_canon and dest_canon not in visited:
1160
+ visited.add(dest_canon)
1161
+ queue.append((dest_canon, new_path))
1162
+
1163
+ return {
1164
+ "status": "not_found",
1165
+ "from": clean_from,
1166
+ "to": clean_to,
1167
+ "reachable": False,
1168
+ "message": f"No relationship path found between '{clean_from}' and '{clean_to}' within depth {bounded_depth}.",
1169
+ **meta,
1170
+ "path": [],
1171
+ }
1172
+
1173
+
1174
+ # ---------------------------------------------------------------------------
1175
+ # 9. get_imports
1176
+ # ---------------------------------------------------------------------------
1177
+ def get_imports(
1178
+ con: sqlite3.Connection,
1179
+ repository: Path,
1180
+ file: str | None = None,
1181
+ canonical_id: str | None = None,
1182
+ ) -> dict[str, Any]:
1183
+ """Return imports and import relationships for a file or canonical symbol."""
1184
+ unindexed = check_index_available(con, repository)
1185
+ if unindexed:
1186
+ return unindexed
1187
+
1188
+ meta = get_index_metadata(con, repository)
1189
+ target_path = file
1190
+
1191
+ if canonical_id and not target_path:
1192
+ sym_row = con.execute(
1193
+ "SELECT path FROM symbols WHERE canonical_id=? OR qualified_name=? OR name=? LIMIT 1",
1194
+ (canonical_id, canonical_id, canonical_id),
1195
+ ).fetchone()
1196
+ if sym_row:
1197
+ target_path = sym_row["path"]
1198
+
1199
+ if not target_path:
1200
+ return {
1201
+ "status": "invalid_request",
1202
+ "error_code": ErrorCode.INVALID_ARGUMENT.value,
1203
+ "error": {
1204
+ "code": ErrorCode.INVALID_ARGUMENT.value,
1205
+ "message": "At least one of 'file' or 'canonical_id' must be specified.",
1206
+ "next_action": {
1207
+ "command": "codegraph get-file <path>",
1208
+ "reason": "Specify a valid file path or canonical ID.",
1209
+ },
1210
+ },
1211
+ **meta,
1212
+ "count": 0,
1213
+ "imports": [],
1214
+ }
1215
+
1216
+ rows = con.execute(
1217
+ "SELECT source_path, module, name, alias, imported_module, imported_name, resolved_path, line "
1218
+ "FROM imports WHERE source_path=? ORDER BY line ASC, module ASC",
1219
+ (target_path,),
1220
+ ).fetchall()
1221
+
1222
+ imports = [
1223
+ {
1224
+ "source": r["source_path"],
1225
+ "target": r["imported_module"] or r["module"],
1226
+ "import_type": "symbol" if r["name"] else "module",
1227
+ "file": r["source_path"],
1228
+ "line": r["line"],
1229
+ "resolved_target": r["resolved_path"],
1230
+ "evidence": make_evidence(
1231
+ file=r["source_path"],
1232
+ start_line=r["line"],
1233
+ end_line=r["line"],
1234
+ evidence_type="import",
1235
+ canonical_id=r["source_path"],
1236
+ ),
1237
+ }
1238
+ for r in rows
1239
+ ]
1240
+
1241
+ imports.sort(key=lambda i: (str(i["file"]), int(i["line"]), str(i["target"])))
1242
+
1243
+ return {
1244
+ "status": "ok",
1245
+ "file": target_path,
1246
+ "canonical_id": canonical_id,
1247
+ **meta,
1248
+ "count": len(imports),
1249
+ "imports": imports,
1250
+ }
1251
+
1252
+
1253
+ # ---------------------------------------------------------------------------
1254
+ # 10. get_dependents
1255
+ # ---------------------------------------------------------------------------
1256
+ def get_dependents(
1257
+ con: sqlite3.Connection,
1258
+ repository: Path,
1259
+ canonical_id: str | None = None,
1260
+ file: str | None = None,
1261
+ ) -> dict[str, Any]:
1262
+ """Reverse dependency query: return files and symbols that depend on the target."""
1263
+ unindexed = check_index_available(con, repository)
1264
+ if unindexed:
1265
+ return unindexed
1266
+
1267
+ meta = get_index_metadata(con, repository)
1268
+ target = canonical_id or file
1269
+ if not target:
1270
+ return {
1271
+ "status": "invalid_request",
1272
+ "error_code": ErrorCode.INVALID_ARGUMENT.value,
1273
+ "error": {
1274
+ "code": ErrorCode.INVALID_ARGUMENT.value,
1275
+ "message": "At least one of 'canonical_id' or 'file' must be specified.",
1276
+ "next_action": {
1277
+ "command": "codegraph resolve <symbol>",
1278
+ "reason": "Specify a valid canonical ID or file path.",
1279
+ },
1280
+ },
1281
+ **meta,
1282
+ "count": 0,
1283
+ "dependents": [],
1284
+ }
1285
+
1286
+ clean_target = target.strip()
1287
+ short_name = clean_target.split(":")[-1].split(".")[-1]
1288
+ dependents: list[dict[str, Any]] = []
1289
+ seen: set[tuple[str, str, int]] = set()
1290
+
1291
+ # 1. Reverse imports
1292
+ imp_rows = con.execute(
1293
+ "SELECT source_path, line, module FROM imports "
1294
+ "WHERE module=? OR imported_module=? OR resolved_path=? OR source_path=? "
1295
+ "ORDER BY source_path ASC, line ASC",
1296
+ (clean_target, clean_target, clean_target, clean_target),
1297
+ ).fetchall()
1298
+ for r in imp_rows:
1299
+ key = ("imports", r["source_path"], r["line"])
1300
+ if key not in seen:
1301
+ seen.add(key)
1302
+ dependents.append(
1303
+ {
1304
+ "dependent": r["source_path"],
1305
+ "dependent_type": "file",
1306
+ "relationship": "imports",
1307
+ "file": r["source_path"],
1308
+ "line": r["line"],
1309
+ "evidence": make_evidence(
1310
+ file=r["source_path"],
1311
+ start_line=r["line"],
1312
+ end_line=r["line"],
1313
+ evidence_type="import",
1314
+ canonical_id=r["source_path"],
1315
+ ),
1316
+ }
1317
+ )
1318
+
1319
+ # 2. Reverse calls
1320
+ call_rows = con.execute(
1321
+ "SELECT source_symbol_id, source_path, line FROM calls "
1322
+ "WHERE resolved_symbol_id=? OR qualified_callee=? OR callee=? "
1323
+ "ORDER BY source_path ASC, line ASC",
1324
+ (clean_target, clean_target, short_name),
1325
+ ).fetchall()
1326
+ for r in call_rows:
1327
+ dep_sym = r["source_symbol_id"] or r["source_path"]
1328
+ key = ("calls", r["source_path"], r["line"])
1329
+ if key not in seen:
1330
+ seen.add(key)
1331
+ dependents.append(
1332
+ {
1333
+ "dependent": dep_sym,
1334
+ "dependent_type": "symbol",
1335
+ "relationship": "calls",
1336
+ "file": r["source_path"],
1337
+ "line": r["line"],
1338
+ "evidence": make_evidence(
1339
+ file=r["source_path"],
1340
+ start_line=r["line"],
1341
+ end_line=r["line"],
1342
+ evidence_type="call",
1343
+ canonical_id=dep_sym,
1344
+ ),
1345
+ }
1346
+ )
1347
+
1348
+ # 3. Inheritance
1349
+ inh_rows = con.execute(
1350
+ "SELECT source_symbol, source_file, line, source_canonical_id FROM inheritance "
1351
+ "WHERE base_name=? OR base_name LIKE ? "
1352
+ "ORDER BY source_file ASC, line ASC",
1353
+ (clean_target, f"%.{short_name}"),
1354
+ ).fetchall()
1355
+ for r in inh_rows:
1356
+ key = ("inherits", r["source_file"], r["line"])
1357
+ if key not in seen:
1358
+ seen.add(key)
1359
+ dependents.append(
1360
+ {
1361
+ "dependent": r["source_canonical_id"] or r["source_symbol"],
1362
+ "dependent_type": "symbol",
1363
+ "relationship": "inherits",
1364
+ "file": r["source_file"],
1365
+ "line": r["line"],
1366
+ "evidence": make_evidence(
1367
+ file=r["source_file"],
1368
+ start_line=r["line"],
1369
+ end_line=r["line"],
1370
+ evidence_type="inheritance",
1371
+ canonical_id=r["source_canonical_id"],
1372
+ ),
1373
+ }
1374
+ )
1375
+
1376
+ dependents.sort(key=lambda d: (str(d["relationship"]), str(d["file"]), int(d["line"])))
1377
+
1378
+ return {
1379
+ "status": "ok",
1380
+ "target": clean_target,
1381
+ **meta,
1382
+ "count": len(dependents),
1383
+ "dependents": dependents,
1384
+ }
1385
+
1386
+
1387
+ # ---------------------------------------------------------------------------
1388
+ # 11. list_routes
1389
+ # ---------------------------------------------------------------------------
1390
+ def list_routes(
1391
+ con: sqlite3.Connection,
1392
+ repository: Path,
1393
+ framework: str | None = None,
1394
+ method: str | None = None,
1395
+ path: str | None = None,
1396
+ ) -> dict[str, Any]:
1397
+ """Expose application routes discovered from the repository."""
1398
+ unindexed = check_index_available(con, repository)
1399
+ if unindexed:
1400
+ return unindexed
1401
+
1402
+ meta = get_index_metadata(con, repository)
1403
+ query = (
1404
+ "SELECT endpoint_id, framework, http_method, route_path, handler_name, "
1405
+ "handler_canonical_id, file_path, line, evidence, confidence "
1406
+ "FROM framework_routes WHERE 1=1 "
1407
+ )
1408
+ params: list[str] = []
1409
+ if framework:
1410
+ query += "AND framework=? "
1411
+ params.append(framework.lower())
1412
+ if method:
1413
+ query += "AND UPPER(http_method)=? "
1414
+ params.append(method.upper())
1415
+ if path:
1416
+ query += "AND route_path LIKE ? "
1417
+ params.append(f"%{path}%")
1418
+
1419
+ query += "ORDER BY route_path ASC, http_method ASC"
1420
+ try:
1421
+ rows = con.execute(query, params).fetchall()
1422
+ except sqlite3.OperationalError:
1423
+ rows = []
1424
+
1425
+ routes = [
1426
+ {
1427
+ "method": r["http_method"],
1428
+ "path": r["route_path"],
1429
+ "handler": r["handler_canonical_id"] or r["handler_name"],
1430
+ "file": r["file_path"],
1431
+ "start_line": r["line"],
1432
+ "end_line": r["line"],
1433
+ "framework": r["framework"],
1434
+ "evidence": make_evidence(
1435
+ file=r["file_path"],
1436
+ start_line=r["line"],
1437
+ end_line=r["line"],
1438
+ evidence_type="route",
1439
+ canonical_id=r["handler_canonical_id"],
1440
+ ),
1441
+ }
1442
+ for r in rows
1443
+ ]
1444
+
1445
+ return {
1446
+ "status": "ok",
1447
+ **meta,
1448
+ "count": len(routes),
1449
+ "routes": routes,
1450
+ }
1451
+
1452
+
1453
+ # ---------------------------------------------------------------------------
1454
+ # 12. get_architecture
1455
+ # ---------------------------------------------------------------------------
1456
+ def get_architecture(
1457
+ con: sqlite3.Connection,
1458
+ repository: Path,
1459
+ ) -> dict[str, Any]:
1460
+ """Return structural overview of repository architecture."""
1461
+ unindexed = check_index_available(con, repository)
1462
+ if unindexed:
1463
+ return unindexed
1464
+
1465
+ meta = get_index_metadata(con, repository)
1466
+ arch = arch_get_architecture(con, repository)
1467
+
1468
+ route_summary = arch.get("route_summary")
1469
+ routes = route_summary.get("routes", []) if isinstance(route_summary, dict) else []
1470
+ dep_summary = arch.get("dependency_summary")
1471
+ major_deps = dep_summary.get("top_dependencies", []) if isinstance(dep_summary, dict) else []
1472
+ test_summary = arch.get("test_summary")
1473
+ test_frameworks = test_summary.get("frameworks", []) if isinstance(test_summary, dict) else []
1474
+ model_summary = arch.get("data_model_summary")
1475
+ data_models = model_summary.get("models", []) if isinstance(model_summary, dict) else []
1476
+
1477
+ return {
1478
+ "status": "ok",
1479
+ **meta,
1480
+ "repository": str(repository.resolve()),
1481
+ "languages": arch.get("languages", ["Python"]),
1482
+ "directories": arch.get("directories", []),
1483
+ "modules": arch.get("top_level_modules", []),
1484
+ "entrypoints": arch.get("entrypoints", []),
1485
+ "routes": routes,
1486
+ "major_dependencies": major_deps,
1487
+ "test_frameworks": test_frameworks,
1488
+ "data_models": data_models,
1489
+ }
1490
+
1491
+
1492
+ # ---------------------------------------------------------------------------
1493
+ # 13. get_git_impact
1494
+ # ---------------------------------------------------------------------------
1495
+ def get_git_impact(
1496
+ con: sqlite3.Connection,
1497
+ repository: Path,
1498
+ base: str = "HEAD~1",
1499
+ head: str = "HEAD",
1500
+ ) -> dict[str, Any]:
1501
+ """Determine code affected by Git changes between base and head."""
1502
+ unindexed = check_index_available(con, repository)
1503
+ if unindexed:
1504
+ return unindexed
1505
+
1506
+ meta = get_index_metadata(con, repository)
1507
+ diffs = changed_files(repository, since=base, until=head)
1508
+ changed_paths = [d.path for d in diffs]
1509
+
1510
+ changed_symbols: list[str] = []
1511
+ added_symbols: list[str] = []
1512
+ removed_symbols: list[str] = []
1513
+ modified_symbols: list[str] = []
1514
+
1515
+ affected_callers: list[dict[str, Any]] = []
1516
+ affected_callees: list[dict[str, Any]] = []
1517
+ affected_dependents: list[dict[str, Any]] = []
1518
+ affected_routes: list[dict[str, Any]] = []
1519
+
1520
+ for d in diffs:
1521
+ if d.status == "A":
1522
+ s_rows = con.execute("SELECT canonical_id, name FROM symbols WHERE path=?", (d.path,)).fetchall()
1523
+ for s in s_rows:
1524
+ added_symbols.append(s["canonical_id"] or s["name"])
1525
+ elif d.status == "D":
1526
+ removed_symbols.append(d.path)
1527
+ else:
1528
+ s_rows = con.execute("SELECT canonical_id, name FROM symbols WHERE path=?", (d.path,)).fetchall()
1529
+ sym_names = {r["name"] for r in s_rows}
1530
+ matched = changed_symbols_since(repository, base, sym_names)
1531
+ for m in matched:
1532
+ modified_symbols.append(m)
1533
+ changed_symbols.append(m)
1534
+
1535
+ # Compute callers and callees of modified symbols
1536
+ for sym in modified_symbols:
1537
+ callers_res = get_callers(con, repository, sym)
1538
+ for c in callers_res.get("callers", []):
1539
+ affected_callers.append(c)
1540
+
1541
+ callees_res = get_callees(con, repository, sym)
1542
+ for c in callees_res.get("callees", []):
1543
+ affected_callees.append(c)
1544
+
1545
+ dep_res = get_dependents(con, repository, canonical_id=sym)
1546
+ for d in dep_res.get("dependents", []):
1547
+ affected_dependents.append(d)
1548
+
1549
+ # Affected routes
1550
+ for p in changed_paths:
1551
+ r_rows = con.execute("SELECT * FROM framework_routes WHERE file_path=?", (p,)).fetchall()
1552
+ for r in r_rows:
1553
+ affected_routes.append(
1554
+ {
1555
+ "method": r["http_method"],
1556
+ "path": r["route_path"],
1557
+ "handler": r["handler_canonical_id"],
1558
+ "file": r["file_path"],
1559
+ }
1560
+ )
1561
+
1562
+ return {
1563
+ "status": "ok",
1564
+ "base": base,
1565
+ "head": head,
1566
+ **meta,
1567
+ "summary": {
1568
+ "changed_files_count": len(diffs),
1569
+ "affected_symbol_count": len(set(changed_symbols + added_symbols)),
1570
+ "affected_route_count": len(affected_routes),
1571
+ "affected_dependent_count": len(affected_dependents),
1572
+ },
1573
+ "changed_files": [d.as_dict() for d in diffs],
1574
+ "changed_symbols": sorted(list(set(changed_symbols + added_symbols))),
1575
+ "added_symbols": sorted(added_symbols),
1576
+ "removed_symbols": sorted(removed_symbols),
1577
+ "modified_symbols": sorted(modified_symbols),
1578
+ "affected_callers": affected_callers,
1579
+ "affected_callees": affected_callees,
1580
+ "affected_dependents": affected_dependents,
1581
+ "affected_routes": affected_routes,
1582
+ }