codegraph-engine 2.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codegraph/__init__.py +37 -0
- codegraph/agent.py +26 -0
- codegraph/architecture.py +328 -0
- codegraph/audit.py +106 -0
- codegraph/cache.py +95 -0
- codegraph/cli.py +854 -0
- codegraph/config.py +43 -0
- codegraph/constraints.py +238 -0
- codegraph/context.py +1228 -0
- codegraph/epistemic.py +90 -0
- codegraph/errors.py +275 -0
- codegraph/evidence/__init__.py +15 -0
- codegraph/evidence/citations.py +397 -0
- codegraph/frameworks.py +434 -0
- codegraph/freshness.py +295 -0
- codegraph/git.py +278 -0
- codegraph/graph/__init__.py +46 -0
- codegraph/graph/models.py +41 -0
- codegraph/graph/traversal.py +1291 -0
- codegraph/indexing/__init__.py +4 -0
- codegraph/indexing/classifier.py +274 -0
- codegraph/indexing/indexer.py +943 -0
- codegraph/indexing/models.py +338 -0
- codegraph/indexing/parser.py +1240 -0
- codegraph/indexing/scanner.py +200 -0
- codegraph/indexing/test_framework.py +116 -0
- codegraph/interrogation.py +1582 -0
- codegraph/llm/__init__.py +3 -0
- codegraph/llm/base.py +15 -0
- codegraph/llm/context.py +20 -0
- codegraph/mcp/__init__.py +3 -0
- codegraph/mcp/server.py +736 -0
- codegraph/memory/__init__.py +3 -0
- codegraph/memory/store.py +46 -0
- codegraph/models.py +289 -0
- codegraph/observability.py +151 -0
- codegraph/optimizer.py +372 -0
- codegraph/planner.py +417 -0
- codegraph/py.typed +1 -0
- codegraph/query_expansion.py +199 -0
- codegraph/ranking.py +363 -0
- codegraph/resolver.py +843 -0
- codegraph/resources/__init__.py +45 -0
- codegraph/resources/cache.py +117 -0
- codegraph/resources/coalescer.py +83 -0
- codegraph/resources/debouncer.py +98 -0
- codegraph/resources/governor.py +232 -0
- codegraph/resources/policy.py +123 -0
- codegraph/retrieval_policy.py +220 -0
- codegraph/search/__init__.py +23 -0
- codegraph/search/hybrid.py +301 -0
- codegraph/search/semantic.py +28 -0
- codegraph/security/__init__.py +3 -0
- codegraph/security/paths.py +35 -0
- codegraph/target_resolver.py +348 -0
- codegraph/task.py +637 -0
- codegraph_engine-2.1.1.dist-info/METADATA +334 -0
- codegraph_engine-2.1.1.dist-info/RECORD +62 -0
- codegraph_engine-2.1.1.dist-info/WHEEL +5 -0
- codegraph_engine-2.1.1.dist-info/entry_points.txt +2 -0
- codegraph_engine-2.1.1.dist-info/licenses/LICENSE +21 -0
- codegraph_engine-2.1.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1582 @@
|
|
|
1
|
+
"""Core deterministic repository interrogation engine for CodeGraph MCP.
|
|
2
|
+
|
|
3
|
+
Principle:
|
|
4
|
+
"The AI understands the developer. CodeGraph interrogates the repository."
|
|
5
|
+
"The claim comes from the AI; the proof comes from CodeGraph."
|
|
6
|
+
|
|
7
|
+
Invariant:
|
|
8
|
+
same repository + same index generation + same MCP request = same deterministic result.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import sqlite3
|
|
13
|
+
from collections import deque
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
from codegraph.architecture import get_architecture as arch_get_architecture
|
|
18
|
+
from codegraph.errors import ErrorCode, SecurityError
|
|
19
|
+
from codegraph.freshness import check_freshness, index_generation
|
|
20
|
+
from codegraph.git import (
|
|
21
|
+
changed_files,
|
|
22
|
+
changed_symbols_since,
|
|
23
|
+
current_commit,
|
|
24
|
+
)
|
|
25
|
+
from codegraph.security import is_sensitive, safe_path
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def make_evidence(
|
|
29
|
+
file: str,
|
|
30
|
+
start_line: int,
|
|
31
|
+
end_line: int,
|
|
32
|
+
evidence_type: str,
|
|
33
|
+
canonical_id: str | None = None,
|
|
34
|
+
) -> dict[str, Any]:
|
|
35
|
+
"""Create a standardized, evidence-backed citation record."""
|
|
36
|
+
return {
|
|
37
|
+
"file": file,
|
|
38
|
+
"start_line": start_line,
|
|
39
|
+
"end_line": end_line,
|
|
40
|
+
"type": evidence_type,
|
|
41
|
+
"canonical_id": canonical_id or "",
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def get_index_metadata(con: sqlite3.Connection, repository: Path) -> dict[str, Any]:
|
|
46
|
+
"""Return index generation, freshness, and repository commit identity."""
|
|
47
|
+
gen = index_generation(con)
|
|
48
|
+
freshness_rep = check_freshness(repository, con)
|
|
49
|
+
commit = current_commit(repository)
|
|
50
|
+
|
|
51
|
+
created_at = ""
|
|
52
|
+
try:
|
|
53
|
+
row = con.execute("SELECT value FROM metadata WHERE key='indexed_at'").fetchone()
|
|
54
|
+
if row:
|
|
55
|
+
created_at = str(row[0])
|
|
56
|
+
except Exception:
|
|
57
|
+
pass
|
|
58
|
+
|
|
59
|
+
return {
|
|
60
|
+
"index": {
|
|
61
|
+
"generation": gen,
|
|
62
|
+
"created_at": created_at,
|
|
63
|
+
"freshness": freshness_rep.status.value,
|
|
64
|
+
},
|
|
65
|
+
"repository": {
|
|
66
|
+
"commit": commit,
|
|
67
|
+
},
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def check_index_available(con: sqlite3.Connection, repository: Path) -> dict[str, Any] | None:
|
|
72
|
+
"""Return an INDEX_NOT_FOUND error envelope if the repository has not been indexed."""
|
|
73
|
+
try:
|
|
74
|
+
count = con.execute("SELECT count(*) FROM files").fetchone()[0]
|
|
75
|
+
except sqlite3.OperationalError:
|
|
76
|
+
count = 0
|
|
77
|
+
if count == 0:
|
|
78
|
+
return {
|
|
79
|
+
"status": "error",
|
|
80
|
+
"error": {
|
|
81
|
+
"code": ErrorCode.INDEX_NOT_FOUND.value,
|
|
82
|
+
"message": "No CodeGraph index exists for this repository.",
|
|
83
|
+
"next_action": {
|
|
84
|
+
"command": "codegraph init",
|
|
85
|
+
"reason": "Initialize the repository before querying it.",
|
|
86
|
+
},
|
|
87
|
+
},
|
|
88
|
+
}
|
|
89
|
+
return None
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
# ---------------------------------------------------------------------------
|
|
93
|
+
# 1. resolve_symbol
|
|
94
|
+
# ---------------------------------------------------------------------------
|
|
95
|
+
def resolve_symbol(
|
|
96
|
+
con: sqlite3.Connection,
|
|
97
|
+
repository: Path,
|
|
98
|
+
name: str,
|
|
99
|
+
) -> dict[str, Any]:
|
|
100
|
+
"""Determine whether an exact/canonical symbol exists and return its location and identity.
|
|
101
|
+
|
|
102
|
+
CodeGraph MUST NOT choose an intended candidate when ambiguous.
|
|
103
|
+
"""
|
|
104
|
+
unindexed = check_index_available(con, repository)
|
|
105
|
+
if unindexed:
|
|
106
|
+
return unindexed
|
|
107
|
+
|
|
108
|
+
clean_name = name.strip()
|
|
109
|
+
meta = get_index_metadata(con, repository)
|
|
110
|
+
if not clean_name:
|
|
111
|
+
return {
|
|
112
|
+
"status": "invalid_request",
|
|
113
|
+
"error_code": ErrorCode.EMPTY_SYMBOL_NAME.value,
|
|
114
|
+
"error": {
|
|
115
|
+
"code": ErrorCode.EMPTY_SYMBOL_NAME.value,
|
|
116
|
+
"message": "Symbol name must be non-empty.",
|
|
117
|
+
"next_action": {
|
|
118
|
+
"command": "codegraph resolve <symbol>",
|
|
119
|
+
"reason": "Provide a non-empty symbol name to resolve.",
|
|
120
|
+
},
|
|
121
|
+
},
|
|
122
|
+
"query": name,
|
|
123
|
+
**meta,
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
# Query matching symbols
|
|
127
|
+
rows = con.execute(
|
|
128
|
+
"SELECT canonical_id, name, qualified_name, kind, path, start_line, end_line "
|
|
129
|
+
"FROM symbols "
|
|
130
|
+
"WHERE canonical_id=? OR qualified_name=? OR name=? "
|
|
131
|
+
"ORDER BY canonical_id ASC",
|
|
132
|
+
(clean_name, clean_name, clean_name),
|
|
133
|
+
).fetchall()
|
|
134
|
+
|
|
135
|
+
if not rows:
|
|
136
|
+
return {
|
|
137
|
+
"status": "not_found",
|
|
138
|
+
"query": clean_name,
|
|
139
|
+
"error": {
|
|
140
|
+
"code": ErrorCode.SYMBOL_NOT_FOUND.value,
|
|
141
|
+
"message": f"No matching symbol was found for '{clean_name}'.",
|
|
142
|
+
"next_action": {
|
|
143
|
+
"command": f"codegraph search {clean_name}",
|
|
144
|
+
"reason": "Search with a broader query term.",
|
|
145
|
+
},
|
|
146
|
+
},
|
|
147
|
+
**meta,
|
|
148
|
+
"matches": [],
|
|
149
|
+
"candidates": [],
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
# Check for exact canonical_id match
|
|
153
|
+
exact_canonical = [r for r in rows if r["canonical_id"] == clean_name]
|
|
154
|
+
if len(exact_canonical) == 1:
|
|
155
|
+
match = exact_canonical[0]
|
|
156
|
+
ev = [
|
|
157
|
+
make_evidence(
|
|
158
|
+
file=match["path"],
|
|
159
|
+
start_line=match["start_line"],
|
|
160
|
+
end_line=match["end_line"],
|
|
161
|
+
evidence_type="definition",
|
|
162
|
+
canonical_id=match["canonical_id"],
|
|
163
|
+
)
|
|
164
|
+
]
|
|
165
|
+
return {
|
|
166
|
+
"status": "ok",
|
|
167
|
+
**meta,
|
|
168
|
+
"symbol": {
|
|
169
|
+
"name": match["name"],
|
|
170
|
+
"canonical_id": match["canonical_id"],
|
|
171
|
+
"kind": match["kind"],
|
|
172
|
+
},
|
|
173
|
+
"location": {
|
|
174
|
+
"file": match["path"],
|
|
175
|
+
"start_line": match["start_line"],
|
|
176
|
+
"end_line": match["end_line"],
|
|
177
|
+
},
|
|
178
|
+
"evidence": ev,
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
# If only one match overall
|
|
182
|
+
if len(rows) == 1:
|
|
183
|
+
match = rows[0]
|
|
184
|
+
ev = [
|
|
185
|
+
make_evidence(
|
|
186
|
+
file=match["path"],
|
|
187
|
+
start_line=match["start_line"],
|
|
188
|
+
end_line=match["end_line"],
|
|
189
|
+
evidence_type="definition",
|
|
190
|
+
canonical_id=match["canonical_id"],
|
|
191
|
+
)
|
|
192
|
+
]
|
|
193
|
+
return {
|
|
194
|
+
"status": "ok",
|
|
195
|
+
**meta,
|
|
196
|
+
"symbol": {
|
|
197
|
+
"name": match["name"],
|
|
198
|
+
"canonical_id": match["canonical_id"],
|
|
199
|
+
"kind": match["kind"],
|
|
200
|
+
},
|
|
201
|
+
"location": {
|
|
202
|
+
"file": match["path"],
|
|
203
|
+
"start_line": match["start_line"],
|
|
204
|
+
"end_line": match["end_line"],
|
|
205
|
+
},
|
|
206
|
+
"evidence": ev,
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
# Multiple candidates exist -> ambiguous! CodeGraph MUST NOT choose!
|
|
210
|
+
matches = [
|
|
211
|
+
{
|
|
212
|
+
"canonical_id": r["canonical_id"],
|
|
213
|
+
"symbol": r["name"],
|
|
214
|
+
"name": r["name"],
|
|
215
|
+
"file": r["path"],
|
|
216
|
+
"kind": r["kind"],
|
|
217
|
+
"line": r["start_line"],
|
|
218
|
+
"start_line": r["start_line"],
|
|
219
|
+
"end_line": r["end_line"],
|
|
220
|
+
}
|
|
221
|
+
for r in rows
|
|
222
|
+
]
|
|
223
|
+
matches.sort(key=lambda m: str(m["canonical_id"]))
|
|
224
|
+
return {
|
|
225
|
+
"status": "ambiguous",
|
|
226
|
+
"query": clean_name,
|
|
227
|
+
"error": {
|
|
228
|
+
"code": ErrorCode.SYMBOL_AMBIGUOUS.value,
|
|
229
|
+
"message": f"Multiple symbols match '{clean_name}'. Provide a qualified name or canonical ID.",
|
|
230
|
+
"next_action": {
|
|
231
|
+
"command": "codegraph resolve <canonical_id>",
|
|
232
|
+
"reason": "Select an explicit canonical ID from candidates.",
|
|
233
|
+
},
|
|
234
|
+
},
|
|
235
|
+
**meta,
|
|
236
|
+
"matches": matches,
|
|
237
|
+
"candidates": matches,
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
# ---------------------------------------------------------------------------
|
|
242
|
+
# 2. search_symbols
|
|
243
|
+
# ---------------------------------------------------------------------------
|
|
244
|
+
def search_symbols(
|
|
245
|
+
con: sqlite3.Connection,
|
|
246
|
+
repository: Path,
|
|
247
|
+
query: str,
|
|
248
|
+
top_k: int = 20,
|
|
249
|
+
) -> dict[str, Any]:
|
|
250
|
+
"""Search indexed repository symbols using explicit search terms with deterministic ranking."""
|
|
251
|
+
unindexed = check_index_available(con, repository)
|
|
252
|
+
if unindexed:
|
|
253
|
+
return unindexed
|
|
254
|
+
|
|
255
|
+
clean_query = query.strip()
|
|
256
|
+
meta = get_index_metadata(con, repository)
|
|
257
|
+
if not clean_query:
|
|
258
|
+
return {
|
|
259
|
+
"status": "error",
|
|
260
|
+
"error_code": ErrorCode.INVALID_ARGUMENT.value,
|
|
261
|
+
"error": {
|
|
262
|
+
"code": ErrorCode.INVALID_ARGUMENT.value,
|
|
263
|
+
"message": "Search query must be non-empty.",
|
|
264
|
+
"next_action": {
|
|
265
|
+
"command": "codegraph search <query>",
|
|
266
|
+
"reason": "Provide a non-empty query term.",
|
|
267
|
+
},
|
|
268
|
+
},
|
|
269
|
+
"query": query,
|
|
270
|
+
**meta,
|
|
271
|
+
"count": 0,
|
|
272
|
+
"symbols": [],
|
|
273
|
+
"evidence": [],
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
bounded_k = max(1, min(top_k, 100))
|
|
277
|
+
|
|
278
|
+
# Search symbols joined with files to exclude GENERATED files
|
|
279
|
+
rows = con.execute(
|
|
280
|
+
"SELECT s.canonical_id, s.name, s.qualified_name, s.kind, s.path, s.start_line, s.end_line, f.category "
|
|
281
|
+
"FROM symbols s "
|
|
282
|
+
"JOIN files f ON f.path = s.path "
|
|
283
|
+
"WHERE (f.category != 'GENERATED' OR f.category IS NULL) "
|
|
284
|
+
"AND (s.name LIKE ? OR s.qualified_name LIKE ? OR s.canonical_id LIKE ?) "
|
|
285
|
+
"ORDER BY s.canonical_id ASC",
|
|
286
|
+
(f"%{clean_query}%", f"%{clean_query}%", f"%{clean_query}%"),
|
|
287
|
+
).fetchall()
|
|
288
|
+
|
|
289
|
+
# Deterministic scoring:
|
|
290
|
+
# 1.0 = exact match on name or qualified_name
|
|
291
|
+
# 0.85 = prefix match on name
|
|
292
|
+
# 0.70 = substring match on name
|
|
293
|
+
# 0.60 = match elsewhere
|
|
294
|
+
scored_results: list[dict[str, Any]] = []
|
|
295
|
+
q_lower = clean_query.lower()
|
|
296
|
+
|
|
297
|
+
for r in rows:
|
|
298
|
+
name = r["name"]
|
|
299
|
+
qname = r["qualified_name"] or ""
|
|
300
|
+
cid = r["canonical_id"] or ""
|
|
301
|
+
name_lower = name.lower()
|
|
302
|
+
|
|
303
|
+
if name_lower == q_lower or qname.lower() == q_lower:
|
|
304
|
+
score = 1.00
|
|
305
|
+
elif name_lower.startswith(q_lower):
|
|
306
|
+
score = 0.85
|
|
307
|
+
elif q_lower in name_lower:
|
|
308
|
+
score = 0.70
|
|
309
|
+
else:
|
|
310
|
+
score = 0.60
|
|
311
|
+
|
|
312
|
+
scored_results.append(
|
|
313
|
+
{
|
|
314
|
+
"name": name,
|
|
315
|
+
"canonical_id": cid,
|
|
316
|
+
"kind": r["kind"],
|
|
317
|
+
"file": r["path"],
|
|
318
|
+
"start_line": r["start_line"],
|
|
319
|
+
"end_line": r["end_line"],
|
|
320
|
+
"score": round(score, 2),
|
|
321
|
+
"evidence": make_evidence(
|
|
322
|
+
file=r["path"],
|
|
323
|
+
start_line=r["start_line"],
|
|
324
|
+
end_line=r["end_line"],
|
|
325
|
+
evidence_type="symbol_match",
|
|
326
|
+
canonical_id=cid,
|
|
327
|
+
),
|
|
328
|
+
}
|
|
329
|
+
)
|
|
330
|
+
|
|
331
|
+
# Sort deterministically: highest score first, ties broken by canonical_id ASC
|
|
332
|
+
scored_results.sort(key=lambda s: (-float(s["score"]), str(s["canonical_id"])))
|
|
333
|
+
selected = scored_results[:bounded_k]
|
|
334
|
+
|
|
335
|
+
evidence_list = [s["evidence"] for s in selected]
|
|
336
|
+
|
|
337
|
+
return {
|
|
338
|
+
"status": "ok",
|
|
339
|
+
"query": clean_query,
|
|
340
|
+
**meta,
|
|
341
|
+
"count": len(selected),
|
|
342
|
+
"symbols": selected,
|
|
343
|
+
"evidence": evidence_list,
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
# ---------------------------------------------------------------------------
|
|
348
|
+
# 3. get_symbol
|
|
349
|
+
# ---------------------------------------------------------------------------
|
|
350
|
+
def get_symbol(
|
|
351
|
+
con: sqlite3.Connection,
|
|
352
|
+
repository: Path,
|
|
353
|
+
canonical_id: str,
|
|
354
|
+
) -> dict[str, Any]:
|
|
355
|
+
"""Return complete structured information for a known canonical symbol."""
|
|
356
|
+
unindexed = check_index_available(con, repository)
|
|
357
|
+
if unindexed:
|
|
358
|
+
return unindexed
|
|
359
|
+
|
|
360
|
+
clean_id = canonical_id.strip()
|
|
361
|
+
meta = get_index_metadata(con, repository)
|
|
362
|
+
if not clean_id:
|
|
363
|
+
return {
|
|
364
|
+
"status": "error",
|
|
365
|
+
"error_code": ErrorCode.INVALID_ARGUMENT.value,
|
|
366
|
+
"error": {
|
|
367
|
+
"code": ErrorCode.INVALID_ARGUMENT.value,
|
|
368
|
+
"message": "Canonical ID or symbol name must be non-empty.",
|
|
369
|
+
"next_action": {
|
|
370
|
+
"command": "codegraph get-symbol <symbol>",
|
|
371
|
+
"reason": "Provide a non-empty canonical ID or symbol name.",
|
|
372
|
+
},
|
|
373
|
+
},
|
|
374
|
+
"canonical_id": canonical_id,
|
|
375
|
+
**meta,
|
|
376
|
+
}
|
|
377
|
+
|
|
378
|
+
row = con.execute(
|
|
379
|
+
"SELECT id, canonical_id, qualified_name, name, kind, path, start_line, end_line, "
|
|
380
|
+
"parent_symbol_id, module, scope, language, signature, content_hash, visibility, "
|
|
381
|
+
"return_type, parameter_count, documentation, decorators "
|
|
382
|
+
"FROM symbols "
|
|
383
|
+
"WHERE canonical_id=? OR qualified_name=? OR name=? "
|
|
384
|
+
"ORDER BY canonical_id=? DESC "
|
|
385
|
+
"LIMIT 1",
|
|
386
|
+
(clean_id, clean_id, clean_id, clean_id),
|
|
387
|
+
).fetchone()
|
|
388
|
+
|
|
389
|
+
if not row:
|
|
390
|
+
return {
|
|
391
|
+
"status": "not_found",
|
|
392
|
+
"canonical_id": clean_id,
|
|
393
|
+
"error": {
|
|
394
|
+
"code": ErrorCode.SYMBOL_NOT_FOUND.value,
|
|
395
|
+
"message": f"No symbol found with canonical ID or name: '{clean_id}'",
|
|
396
|
+
"next_action": {
|
|
397
|
+
"command": f"codegraph search {clean_id}",
|
|
398
|
+
"reason": "Search for symbol candidates.",
|
|
399
|
+
},
|
|
400
|
+
},
|
|
401
|
+
**meta,
|
|
402
|
+
}
|
|
403
|
+
|
|
404
|
+
# Query parent symbol
|
|
405
|
+
parent_info: str | None = None
|
|
406
|
+
if row["parent_symbol_id"]:
|
|
407
|
+
p_row = con.execute(
|
|
408
|
+
"SELECT canonical_id, qualified_name FROM symbols WHERE id=?",
|
|
409
|
+
(row["parent_symbol_id"],),
|
|
410
|
+
).fetchone()
|
|
411
|
+
if p_row:
|
|
412
|
+
parent_info = p_row["canonical_id"] or p_row["qualified_name"]
|
|
413
|
+
|
|
414
|
+
# Query children
|
|
415
|
+
children_rows = con.execute(
|
|
416
|
+
"SELECT canonical_id, name, kind, start_line, end_line FROM symbols "
|
|
417
|
+
"WHERE parent_symbol_id=? ORDER BY start_line ASC, canonical_id ASC",
|
|
418
|
+
(row["id"],),
|
|
419
|
+
).fetchall()
|
|
420
|
+
children = [
|
|
421
|
+
{
|
|
422
|
+
"canonical_id": c["canonical_id"],
|
|
423
|
+
"name": c["name"],
|
|
424
|
+
"kind": c["kind"],
|
|
425
|
+
"start_line": c["start_line"],
|
|
426
|
+
"end_line": c["end_line"],
|
|
427
|
+
}
|
|
428
|
+
for c in children_rows
|
|
429
|
+
]
|
|
430
|
+
|
|
431
|
+
# Parse decorators
|
|
432
|
+
decors_raw = row["decorators"] or ""
|
|
433
|
+
decorators: list[str] = []
|
|
434
|
+
if decors_raw:
|
|
435
|
+
decorators = [d.strip() for d in decors_raw.split(",") if d.strip()]
|
|
436
|
+
|
|
437
|
+
# Query relationships from graph_edges
|
|
438
|
+
cid = row["canonical_id"]
|
|
439
|
+
rel_rows = con.execute(
|
|
440
|
+
"SELECT target, relationship, confidence, file, start_line, end_line "
|
|
441
|
+
"FROM graph_edges WHERE source=? ORDER BY relationship ASC, target ASC",
|
|
442
|
+
(cid,),
|
|
443
|
+
).fetchall()
|
|
444
|
+
relationships = [
|
|
445
|
+
{
|
|
446
|
+
"target": r["target"],
|
|
447
|
+
"relationship": r["relationship"],
|
|
448
|
+
"confidence": r["confidence"],
|
|
449
|
+
"file": r["file"],
|
|
450
|
+
"start_line": r["start_line"],
|
|
451
|
+
"end_line": r["end_line"],
|
|
452
|
+
}
|
|
453
|
+
for r in rel_rows
|
|
454
|
+
]
|
|
455
|
+
|
|
456
|
+
ev = [
|
|
457
|
+
make_evidence(
|
|
458
|
+
file=row["path"],
|
|
459
|
+
start_line=row["start_line"],
|
|
460
|
+
end_line=row["end_line"],
|
|
461
|
+
evidence_type="definition",
|
|
462
|
+
canonical_id=row["canonical_id"],
|
|
463
|
+
)
|
|
464
|
+
]
|
|
465
|
+
|
|
466
|
+
return {
|
|
467
|
+
"status": "ok",
|
|
468
|
+
**meta,
|
|
469
|
+
"evidence": ev,
|
|
470
|
+
"symbol": {
|
|
471
|
+
"canonical_id": row["canonical_id"],
|
|
472
|
+
"name": row["name"],
|
|
473
|
+
"qualified_name": row["qualified_name"],
|
|
474
|
+
"kind": row["kind"],
|
|
475
|
+
"file": row["path"],
|
|
476
|
+
"start_line": row["start_line"],
|
|
477
|
+
"end_line": row["end_line"],
|
|
478
|
+
"module": row["module"],
|
|
479
|
+
"scope": row["scope"],
|
|
480
|
+
"language": row["language"],
|
|
481
|
+
"signature": row["signature"] or "",
|
|
482
|
+
"parent": parent_info,
|
|
483
|
+
"children": children,
|
|
484
|
+
"decorators": decorators,
|
|
485
|
+
"docstring": row["documentation"] or "",
|
|
486
|
+
"visibility": row["visibility"] or "public",
|
|
487
|
+
"relationships": relationships,
|
|
488
|
+
},
|
|
489
|
+
}
|
|
490
|
+
|
|
491
|
+
|
|
492
|
+
# ---------------------------------------------------------------------------
|
|
493
|
+
# 4. get_file
|
|
494
|
+
# ---------------------------------------------------------------------------
|
|
495
|
+
def get_file(
|
|
496
|
+
con: sqlite3.Connection,
|
|
497
|
+
repository: Path,
|
|
498
|
+
path: str,
|
|
499
|
+
include_content: bool = False,
|
|
500
|
+
) -> dict[str, Any]:
|
|
501
|
+
"""Return the structural AST representation of an indexed file."""
|
|
502
|
+
unindexed = check_index_available(con, repository)
|
|
503
|
+
if unindexed:
|
|
504
|
+
return unindexed
|
|
505
|
+
|
|
506
|
+
meta = get_index_metadata(con, repository)
|
|
507
|
+
try:
|
|
508
|
+
resolved = safe_path(repository, path)
|
|
509
|
+
relative = resolved.relative_to(repository).as_posix()
|
|
510
|
+
except SecurityError:
|
|
511
|
+
return {
|
|
512
|
+
"status": "error",
|
|
513
|
+
"error_code": ErrorCode.PATH_OUTSIDE_REPOSITORY.value,
|
|
514
|
+
"error": {
|
|
515
|
+
"code": ErrorCode.PATH_OUTSIDE_REPOSITORY.value,
|
|
516
|
+
"message": f"Path traversal blocked: '{path}' resides outside repository boundary.",
|
|
517
|
+
"next_action": {
|
|
518
|
+
"command": "codegraph status",
|
|
519
|
+
"reason": "Ensure all queried paths reside within the repository root.",
|
|
520
|
+
},
|
|
521
|
+
},
|
|
522
|
+
"path": path,
|
|
523
|
+
**meta,
|
|
524
|
+
}
|
|
525
|
+
except Exception:
|
|
526
|
+
return {
|
|
527
|
+
"status": "error",
|
|
528
|
+
"error_code": ErrorCode.INVALID_PATH.value,
|
|
529
|
+
"error": {
|
|
530
|
+
"code": ErrorCode.INVALID_PATH.value,
|
|
531
|
+
"message": f"Invalid path: '{path}'",
|
|
532
|
+
"next_action": {
|
|
533
|
+
"command": "codegraph status",
|
|
534
|
+
"reason": "Check repository files and paths.",
|
|
535
|
+
},
|
|
536
|
+
},
|
|
537
|
+
"path": path,
|
|
538
|
+
**meta,
|
|
539
|
+
}
|
|
540
|
+
|
|
541
|
+
if is_sensitive(Path(relative)):
|
|
542
|
+
return {
|
|
543
|
+
"status": "invalid_request",
|
|
544
|
+
"error_code": ErrorCode.SENSITIVE_FILE_ACCESS_DENIED.value,
|
|
545
|
+
"error": {
|
|
546
|
+
"code": ErrorCode.SENSITIVE_FILE_ACCESS_DENIED.value,
|
|
547
|
+
"message": f"Access to sensitive file '{relative}' is blocked.",
|
|
548
|
+
"next_action": {
|
|
549
|
+
"command": "codegraph privacy",
|
|
550
|
+
"reason": "Review protected file boundaries.",
|
|
551
|
+
},
|
|
552
|
+
},
|
|
553
|
+
"path": relative,
|
|
554
|
+
**meta,
|
|
555
|
+
}
|
|
556
|
+
|
|
557
|
+
file_row = con.execute(
|
|
558
|
+
"SELECT hash, language, category FROM files WHERE path=?",
|
|
559
|
+
(relative,),
|
|
560
|
+
).fetchone()
|
|
561
|
+
|
|
562
|
+
if not file_row and not resolved.is_file():
|
|
563
|
+
return {
|
|
564
|
+
"status": "not_found",
|
|
565
|
+
"error_code": ErrorCode.INVALID_PATH.value,
|
|
566
|
+
"error": {
|
|
567
|
+
"code": ErrorCode.INVALID_PATH.value,
|
|
568
|
+
"message": f"File does not exist: '{relative}'",
|
|
569
|
+
"next_action": {
|
|
570
|
+
"command": "codegraph status",
|
|
571
|
+
"reason": "Check repository files and paths.",
|
|
572
|
+
},
|
|
573
|
+
},
|
|
574
|
+
"path": relative,
|
|
575
|
+
**meta,
|
|
576
|
+
}
|
|
577
|
+
|
|
578
|
+
# Query symbols in file
|
|
579
|
+
symbols_rows = con.execute(
|
|
580
|
+
"SELECT canonical_id, name, qualified_name, kind, start_line, end_line "
|
|
581
|
+
"FROM symbols WHERE path=? ORDER BY start_line ASC, canonical_id ASC",
|
|
582
|
+
(relative,),
|
|
583
|
+
).fetchall()
|
|
584
|
+
|
|
585
|
+
symbols = [
|
|
586
|
+
{
|
|
587
|
+
"canonical_id": s["canonical_id"],
|
|
588
|
+
"name": s["name"],
|
|
589
|
+
"qualified_name": s["qualified_name"],
|
|
590
|
+
"kind": s["kind"],
|
|
591
|
+
"start_line": s["start_line"],
|
|
592
|
+
"end_line": s["end_line"],
|
|
593
|
+
}
|
|
594
|
+
for s in symbols_rows
|
|
595
|
+
]
|
|
596
|
+
classes = [s for s in symbols if s["kind"] in ("class", "interface")]
|
|
597
|
+
functions = [s for s in symbols if s["kind"] in ("function", "method")]
|
|
598
|
+
|
|
599
|
+
# Query imports
|
|
600
|
+
import_rows = con.execute(
|
|
601
|
+
"SELECT module, name, alias, line, imported_module, imported_name, resolved_path "
|
|
602
|
+
"FROM imports WHERE source_path=? ORDER BY line ASC, module ASC",
|
|
603
|
+
(relative,),
|
|
604
|
+
).fetchall()
|
|
605
|
+
imports = [
|
|
606
|
+
{
|
|
607
|
+
"module": i["module"],
|
|
608
|
+
"name": i["name"],
|
|
609
|
+
"alias": i["alias"],
|
|
610
|
+
"line": i["line"],
|
|
611
|
+
"imported_module": i["imported_module"],
|
|
612
|
+
"imported_name": i["imported_name"],
|
|
613
|
+
"resolved_path": i["resolved_path"],
|
|
614
|
+
}
|
|
615
|
+
for i in import_rows
|
|
616
|
+
]
|
|
617
|
+
|
|
618
|
+
# Query routes in file
|
|
619
|
+
route_rows = con.execute(
|
|
620
|
+
"SELECT endpoint_id, framework, http_method, route_path, handler_name, line "
|
|
621
|
+
"FROM framework_routes WHERE file_path=? ORDER BY line ASC, route_path ASC",
|
|
622
|
+
(relative,),
|
|
623
|
+
).fetchall()
|
|
624
|
+
routes = [
|
|
625
|
+
{
|
|
626
|
+
"endpoint_id": r["endpoint_id"],
|
|
627
|
+
"framework": r["framework"],
|
|
628
|
+
"method": r["http_method"],
|
|
629
|
+
"path": r["route_path"],
|
|
630
|
+
"handler": r["handler_name"],
|
|
631
|
+
"line": r["line"],
|
|
632
|
+
}
|
|
633
|
+
for r in route_rows
|
|
634
|
+
]
|
|
635
|
+
|
|
636
|
+
file_size = resolved.stat().st_size if resolved.exists() else 0
|
|
637
|
+
file_hash = file_row["hash"] if file_row else ""
|
|
638
|
+
language = file_row["language"] if file_row else "unknown"
|
|
639
|
+
|
|
640
|
+
content: str | None = None
|
|
641
|
+
if include_content and resolved.exists():
|
|
642
|
+
content = resolved.read_text(encoding="utf-8", errors="replace")
|
|
643
|
+
if len(content) > 50_000:
|
|
644
|
+
content = content[:50_000] + "\n...[truncated at 50,000 chars]"
|
|
645
|
+
|
|
646
|
+
ev = [
|
|
647
|
+
make_evidence(
|
|
648
|
+
file=relative,
|
|
649
|
+
start_line=1,
|
|
650
|
+
end_line=len(symbols_rows) or 1,
|
|
651
|
+
evidence_type="file_ast",
|
|
652
|
+
canonical_id=relative,
|
|
653
|
+
)
|
|
654
|
+
]
|
|
655
|
+
|
|
656
|
+
result: dict[str, Any] = {
|
|
657
|
+
"status": "ok",
|
|
658
|
+
"file": relative,
|
|
659
|
+
**meta,
|
|
660
|
+
"hash": file_hash,
|
|
661
|
+
"language": language,
|
|
662
|
+
"size_bytes": file_size,
|
|
663
|
+
"classes": classes,
|
|
664
|
+
"functions": functions,
|
|
665
|
+
"imports": imports,
|
|
666
|
+
"routes": routes,
|
|
667
|
+
"evidence": ev,
|
|
668
|
+
}
|
|
669
|
+
if include_content:
|
|
670
|
+
result["content"] = content
|
|
671
|
+
return result
|
|
672
|
+
|
|
673
|
+
|
|
674
|
+
# ---------------------------------------------------------------------------
|
|
675
|
+
# 5. get_references
|
|
676
|
+
# ---------------------------------------------------------------------------
|
|
677
|
+
def get_references(
|
|
678
|
+
con: sqlite3.Connection,
|
|
679
|
+
repository: Path,
|
|
680
|
+
canonical_id: str,
|
|
681
|
+
) -> dict[str, Any]:
|
|
682
|
+
"""Return all known references and call sites to a canonical symbol with evidence."""
|
|
683
|
+
unindexed = check_index_available(con, repository)
|
|
684
|
+
if unindexed:
|
|
685
|
+
return unindexed
|
|
686
|
+
|
|
687
|
+
clean_id = canonical_id.strip()
|
|
688
|
+
meta = get_index_metadata(con, repository)
|
|
689
|
+
if not clean_id:
|
|
690
|
+
return {
|
|
691
|
+
"status": "error",
|
|
692
|
+
"error_code": ErrorCode.INVALID_ARGUMENT.value,
|
|
693
|
+
"error": {
|
|
694
|
+
"code": ErrorCode.INVALID_ARGUMENT.value,
|
|
695
|
+
"message": "Canonical ID must be non-empty.",
|
|
696
|
+
"next_action": {
|
|
697
|
+
"command": "codegraph resolve <symbol>",
|
|
698
|
+
"reason": "Resolve canonical ID before querying references.",
|
|
699
|
+
},
|
|
700
|
+
},
|
|
701
|
+
"canonical_id": canonical_id,
|
|
702
|
+
**meta,
|
|
703
|
+
"count": 0,
|
|
704
|
+
"references": [],
|
|
705
|
+
}
|
|
706
|
+
|
|
707
|
+
short_name = clean_id.split(":")[-1].split(".")[-1]
|
|
708
|
+
|
|
709
|
+
# 1. Search 'references' table
|
|
710
|
+
ref_rows = con.execute(
|
|
711
|
+
"SELECT source_symbol_id, relationship, confidence, path, start_line, end_line, evidence "
|
|
712
|
+
"FROM 'references' "
|
|
713
|
+
"WHERE target_symbol_id=? OR target_symbol_id LIKE ? "
|
|
714
|
+
"ORDER BY path ASC, start_line ASC",
|
|
715
|
+
(clean_id, f"%.{clean_id}"),
|
|
716
|
+
).fetchall()
|
|
717
|
+
|
|
718
|
+
# 2. Search 'calls' table
|
|
719
|
+
call_rows = con.execute(
|
|
720
|
+
"SELECT source_path, callee, line, confidence, source_symbol_id "
|
|
721
|
+
"FROM calls "
|
|
722
|
+
"WHERE resolved_symbol_id=? OR qualified_callee=? OR callee=? "
|
|
723
|
+
"ORDER BY source_path ASC, line ASC",
|
|
724
|
+
(clean_id, clean_id, short_name),
|
|
725
|
+
).fetchall()
|
|
726
|
+
|
|
727
|
+
references: list[dict[str, Any]] = []
|
|
728
|
+
seen: set[tuple[str, int, str]] = set()
|
|
729
|
+
|
|
730
|
+
for r in ref_rows:
|
|
731
|
+
key = (r["path"], r["start_line"], r["source_symbol_id"] or "")
|
|
732
|
+
if key not in seen:
|
|
733
|
+
seen.add(key)
|
|
734
|
+
references.append(
|
|
735
|
+
{
|
|
736
|
+
"canonical_id": r["source_symbol_id"] or "",
|
|
737
|
+
"file": r["path"],
|
|
738
|
+
"start_line": r["start_line"],
|
|
739
|
+
"end_line": r["end_line"],
|
|
740
|
+
"reference_type": r["relationship"].lower(),
|
|
741
|
+
"confidence": r["confidence"],
|
|
742
|
+
"evidence": make_evidence(
|
|
743
|
+
file=r["path"],
|
|
744
|
+
start_line=r["start_line"],
|
|
745
|
+
end_line=r["end_line"],
|
|
746
|
+
evidence_type="reference",
|
|
747
|
+
canonical_id=r["source_symbol_id"],
|
|
748
|
+
),
|
|
749
|
+
}
|
|
750
|
+
)
|
|
751
|
+
|
|
752
|
+
for c in call_rows:
|
|
753
|
+
key = (c["source_path"], c["line"], c["source_symbol_id"] or "")
|
|
754
|
+
if key not in seen:
|
|
755
|
+
seen.add(key)
|
|
756
|
+
references.append(
|
|
757
|
+
{
|
|
758
|
+
"canonical_id": c["source_symbol_id"] or "",
|
|
759
|
+
"file": c["source_path"],
|
|
760
|
+
"start_line": c["line"],
|
|
761
|
+
"end_line": c["line"],
|
|
762
|
+
"reference_type": "call",
|
|
763
|
+
"confidence": c["confidence"],
|
|
764
|
+
"evidence": make_evidence(
|
|
765
|
+
file=c["source_path"],
|
|
766
|
+
start_line=c["line"],
|
|
767
|
+
end_line=c["line"],
|
|
768
|
+
evidence_type="call",
|
|
769
|
+
canonical_id=c["source_symbol_id"],
|
|
770
|
+
),
|
|
771
|
+
}
|
|
772
|
+
)
|
|
773
|
+
|
|
774
|
+
references.sort(key=lambda r: (str(r["file"]), int(r["start_line"]), str(r["canonical_id"])))
|
|
775
|
+
|
|
776
|
+
return {
|
|
777
|
+
"status": "ok",
|
|
778
|
+
"canonical_id": clean_id,
|
|
779
|
+
**meta,
|
|
780
|
+
"count": len(references),
|
|
781
|
+
"references": references,
|
|
782
|
+
}
|
|
783
|
+
|
|
784
|
+
|
|
785
|
+
# ---------------------------------------------------------------------------
|
|
786
|
+
# 6. get_callers
|
|
787
|
+
# ---------------------------------------------------------------------------
|
|
788
|
+
def get_callers(
|
|
789
|
+
con: sqlite3.Connection,
|
|
790
|
+
repository: Path,
|
|
791
|
+
canonical_id: str,
|
|
792
|
+
) -> dict[str, Any]:
|
|
793
|
+
"""Return functions and methods that call the specified canonical symbol."""
|
|
794
|
+
unindexed = check_index_available(con, repository)
|
|
795
|
+
if unindexed:
|
|
796
|
+
return unindexed
|
|
797
|
+
|
|
798
|
+
clean_id = canonical_id.strip()
|
|
799
|
+
meta = get_index_metadata(con, repository)
|
|
800
|
+
if not clean_id:
|
|
801
|
+
return {
|
|
802
|
+
"status": "error",
|
|
803
|
+
"error_code": ErrorCode.INVALID_ARGUMENT.value,
|
|
804
|
+
"error": {
|
|
805
|
+
"code": ErrorCode.INVALID_ARGUMENT.value,
|
|
806
|
+
"message": "Canonical ID must be non-empty.",
|
|
807
|
+
"next_action": {
|
|
808
|
+
"command": "codegraph resolve <symbol>",
|
|
809
|
+
"reason": "Resolve canonical ID before querying callers.",
|
|
810
|
+
},
|
|
811
|
+
},
|
|
812
|
+
"canonical_id": canonical_id,
|
|
813
|
+
**meta,
|
|
814
|
+
"count": 0,
|
|
815
|
+
"callers": [],
|
|
816
|
+
}
|
|
817
|
+
|
|
818
|
+
short_name = clean_id.split(":")[-1].split(".")[-1]
|
|
819
|
+
|
|
820
|
+
# Query calls joined with symbols to get authoritative caller canonical IDs
|
|
821
|
+
rows = con.execute(
|
|
822
|
+
"SELECT c.source_path, c.callee, c.qualified_callee, c.line, c.confidence, "
|
|
823
|
+
"c.source_symbol_id, s.canonical_id AS caller_canon, s.name AS caller_name "
|
|
824
|
+
"FROM calls c "
|
|
825
|
+
"LEFT JOIN symbols s ON s.canonical_id = c.source_symbol_id "
|
|
826
|
+
"WHERE c.resolved_symbol_id=? OR c.qualified_callee=? OR c.callee=? "
|
|
827
|
+
"ORDER BY c.source_path ASC, c.line ASC",
|
|
828
|
+
(clean_id, clean_id, short_name),
|
|
829
|
+
).fetchall()
|
|
830
|
+
|
|
831
|
+
callers: list[dict[str, Any]] = []
|
|
832
|
+
seen: set[tuple[str, int, str]] = set()
|
|
833
|
+
|
|
834
|
+
for r in rows:
|
|
835
|
+
caller_id = r["caller_canon"] or r["source_symbol_id"] or r["caller_name"] or r["source_path"]
|
|
836
|
+
key = (r["source_path"], r["line"], caller_id)
|
|
837
|
+
if key not in seen:
|
|
838
|
+
seen.add(key)
|
|
839
|
+
callers.append(
|
|
840
|
+
{
|
|
841
|
+
"caller": caller_id,
|
|
842
|
+
"canonical_id": r["caller_canon"] or r["source_symbol_id"] or "",
|
|
843
|
+
"file": r["source_path"],
|
|
844
|
+
"line": r["line"],
|
|
845
|
+
"confidence": r["confidence"],
|
|
846
|
+
"call_site": {
|
|
847
|
+
"file": r["source_path"],
|
|
848
|
+
"line": r["line"],
|
|
849
|
+
"callee": r["callee"],
|
|
850
|
+
},
|
|
851
|
+
"evidence": make_evidence(
|
|
852
|
+
file=r["source_path"],
|
|
853
|
+
start_line=r["line"],
|
|
854
|
+
end_line=r["line"],
|
|
855
|
+
evidence_type="call",
|
|
856
|
+
canonical_id=caller_id,
|
|
857
|
+
),
|
|
858
|
+
}
|
|
859
|
+
)
|
|
860
|
+
|
|
861
|
+
callers.sort(key=lambda c: (str(c["file"]), int(c["line"]), str(c["caller"])))
|
|
862
|
+
|
|
863
|
+
return {
|
|
864
|
+
"status": "ok",
|
|
865
|
+
"canonical_id": clean_id,
|
|
866
|
+
**meta,
|
|
867
|
+
"count": len(callers),
|
|
868
|
+
"callers": callers,
|
|
869
|
+
}
|
|
870
|
+
|
|
871
|
+
|
|
872
|
+
# ---------------------------------------------------------------------------
|
|
873
|
+
# 7. get_callees
|
|
874
|
+
# ---------------------------------------------------------------------------
|
|
875
|
+
def get_callees(
|
|
876
|
+
con: sqlite3.Connection,
|
|
877
|
+
repository: Path,
|
|
878
|
+
canonical_id: str,
|
|
879
|
+
) -> dict[str, Any]:
|
|
880
|
+
"""Return functions and methods called by the specified canonical symbol."""
|
|
881
|
+
unindexed = check_index_available(con, repository)
|
|
882
|
+
if unindexed:
|
|
883
|
+
return unindexed
|
|
884
|
+
|
|
885
|
+
clean_id = canonical_id.strip()
|
|
886
|
+
meta = get_index_metadata(con, repository)
|
|
887
|
+
if not clean_id:
|
|
888
|
+
return {
|
|
889
|
+
"status": "error",
|
|
890
|
+
"error_code": ErrorCode.INVALID_ARGUMENT.value,
|
|
891
|
+
"error": {
|
|
892
|
+
"code": ErrorCode.INVALID_ARGUMENT.value,
|
|
893
|
+
"message": "Canonical ID must be non-empty.",
|
|
894
|
+
"next_action": {
|
|
895
|
+
"command": "codegraph resolve <symbol>",
|
|
896
|
+
"reason": "Resolve canonical ID before querying callees.",
|
|
897
|
+
},
|
|
898
|
+
},
|
|
899
|
+
"canonical_id": canonical_id,
|
|
900
|
+
**meta,
|
|
901
|
+
"count": 0,
|
|
902
|
+
"callees": [],
|
|
903
|
+
}
|
|
904
|
+
|
|
905
|
+
# Find the symbol record to get line bounds
|
|
906
|
+
sym_row = con.execute(
|
|
907
|
+
"SELECT id, canonical_id, path, start_line, end_line FROM symbols "
|
|
908
|
+
"WHERE canonical_id=? OR qualified_name=? OR name=? LIMIT 1",
|
|
909
|
+
(clean_id, clean_id, clean_id),
|
|
910
|
+
).fetchone()
|
|
911
|
+
|
|
912
|
+
query_id = sym_row["canonical_id"] if sym_row else clean_id
|
|
913
|
+
query_path = sym_row["path"] if sym_row else None
|
|
914
|
+
|
|
915
|
+
if sym_row and query_path:
|
|
916
|
+
rows = con.execute(
|
|
917
|
+
"SELECT c.source_path, c.callee, c.qualified_callee, c.line, c.confidence, "
|
|
918
|
+
"c.source_symbol_id, c.resolved_symbol_id, "
|
|
919
|
+
"s.canonical_id AS callee_canon, s.path AS callee_path "
|
|
920
|
+
"FROM calls c "
|
|
921
|
+
"LEFT JOIN symbols s ON (s.canonical_id = c.resolved_symbol_id OR s.qualified_name = c.qualified_callee) "
|
|
922
|
+
"WHERE (c.source_symbol_id=? OR (c.source_path=? AND c.line >= ? AND c.line <= ?)) "
|
|
923
|
+
"ORDER BY c.line ASC",
|
|
924
|
+
(query_id, query_path, sym_row["start_line"], sym_row["end_line"]),
|
|
925
|
+
).fetchall()
|
|
926
|
+
else:
|
|
927
|
+
rows = con.execute(
|
|
928
|
+
"SELECT c.source_path, c.callee, c.qualified_callee, c.line, c.confidence, "
|
|
929
|
+
"c.source_symbol_id, c.resolved_symbol_id, "
|
|
930
|
+
"s.canonical_id AS callee_canon, s.path AS callee_path "
|
|
931
|
+
"FROM calls c "
|
|
932
|
+
"LEFT JOIN symbols s ON (s.canonical_id = c.resolved_symbol_id OR s.qualified_name = c.qualified_callee) "
|
|
933
|
+
"WHERE c.source_symbol_id=? "
|
|
934
|
+
"ORDER BY c.line ASC",
|
|
935
|
+
(query_id,),
|
|
936
|
+
).fetchall()
|
|
937
|
+
|
|
938
|
+
callees: list[dict[str, Any]] = []
|
|
939
|
+
seen: set[tuple[str, int, str]] = set()
|
|
940
|
+
|
|
941
|
+
for r in rows:
|
|
942
|
+
callee_name = r["callee"]
|
|
943
|
+
target_cid = r["callee_canon"] or r["resolved_symbol_id"]
|
|
944
|
+
if target_cid or r["callee_path"]:
|
|
945
|
+
call_type = "resolved_call"
|
|
946
|
+
elif "." in str(r["qualified_callee"] or ""):
|
|
947
|
+
call_type = "external_call"
|
|
948
|
+
else:
|
|
949
|
+
call_type = "unresolved_call"
|
|
950
|
+
|
|
951
|
+
key = (r["source_path"], r["line"], callee_name)
|
|
952
|
+
if key not in seen:
|
|
953
|
+
seen.add(key)
|
|
954
|
+
callees.append(
|
|
955
|
+
{
|
|
956
|
+
"callee": callee_name,
|
|
957
|
+
"canonical_id": target_cid,
|
|
958
|
+
"call_type": call_type,
|
|
959
|
+
"file": r["source_path"],
|
|
960
|
+
"line": r["line"],
|
|
961
|
+
"confidence": r["confidence"],
|
|
962
|
+
"evidence": make_evidence(
|
|
963
|
+
file=r["source_path"],
|
|
964
|
+
start_line=r["line"],
|
|
965
|
+
end_line=r["line"],
|
|
966
|
+
evidence_type="call",
|
|
967
|
+
canonical_id=target_cid or query_id,
|
|
968
|
+
),
|
|
969
|
+
}
|
|
970
|
+
)
|
|
971
|
+
|
|
972
|
+
callees.sort(key=lambda c: (str(c["call_type"]), str(c["canonical_id"] or ""), int(c["line"])))
|
|
973
|
+
|
|
974
|
+
return {
|
|
975
|
+
"status": "ok",
|
|
976
|
+
"canonical_id": clean_id,
|
|
977
|
+
**meta,
|
|
978
|
+
"count": len(callees),
|
|
979
|
+
"callees": callees,
|
|
980
|
+
"resolved_calls": [c for c in callees if c["call_type"] == "resolved_call"],
|
|
981
|
+
"unresolved_calls": [c for c in callees if c["call_type"] == "unresolved_call"],
|
|
982
|
+
"external_calls": [c for c in callees if c["call_type"] == "external_call"],
|
|
983
|
+
}
|
|
984
|
+
|
|
985
|
+
|
|
986
|
+
# ---------------------------------------------------------------------------
|
|
987
|
+
# 8. trace_path
|
|
988
|
+
# ---------------------------------------------------------------------------
|
|
989
|
+
def trace_path(
|
|
990
|
+
con: sqlite3.Connection,
|
|
991
|
+
repository: Path,
|
|
992
|
+
from_symbol: str,
|
|
993
|
+
to_symbol: str,
|
|
994
|
+
max_depth: int = 5,
|
|
995
|
+
) -> dict[str, Any]:
|
|
996
|
+
"""Find a deterministic relationship path between two known symbols."""
|
|
997
|
+
unindexed = check_index_available(con, repository)
|
|
998
|
+
if unindexed:
|
|
999
|
+
return unindexed
|
|
1000
|
+
|
|
1001
|
+
clean_from = from_symbol.strip()
|
|
1002
|
+
clean_to = to_symbol.strip()
|
|
1003
|
+
meta = get_index_metadata(con, repository)
|
|
1004
|
+
|
|
1005
|
+
if max_depth < 1 or max_depth > 10:
|
|
1006
|
+
return {
|
|
1007
|
+
"status": "error",
|
|
1008
|
+
"error_code": ErrorCode.INVALID_DEPTH.value,
|
|
1009
|
+
"error": {
|
|
1010
|
+
"code": ErrorCode.INVALID_DEPTH.value,
|
|
1011
|
+
"message": f"Invalid graph traversal depth {max_depth}; allowed range is 1..5.",
|
|
1012
|
+
"next_action": {
|
|
1013
|
+
"command": "codegraph trace --depth 2 <symbol>",
|
|
1014
|
+
"reason": "Specify a traversal depth between 1 and 5.",
|
|
1015
|
+
},
|
|
1016
|
+
},
|
|
1017
|
+
"from": from_symbol,
|
|
1018
|
+
"to": to_symbol,
|
|
1019
|
+
**meta,
|
|
1020
|
+
"path": [],
|
|
1021
|
+
}
|
|
1022
|
+
|
|
1023
|
+
bounded_depth = max(1, min(max_depth, 5))
|
|
1024
|
+
|
|
1025
|
+
if not clean_from or not clean_to:
|
|
1026
|
+
return {
|
|
1027
|
+
"status": "error",
|
|
1028
|
+
"error_code": ErrorCode.INVALID_ARGUMENT.value,
|
|
1029
|
+
"error": {
|
|
1030
|
+
"code": ErrorCode.INVALID_ARGUMENT.value,
|
|
1031
|
+
"message": "Both source_symbol and target_symbol must be non-empty.",
|
|
1032
|
+
"next_action": {
|
|
1033
|
+
"command": "codegraph trace <source> <target>",
|
|
1034
|
+
"reason": "Provide valid source and target symbol names.",
|
|
1035
|
+
},
|
|
1036
|
+
},
|
|
1037
|
+
"from": from_symbol,
|
|
1038
|
+
"to": to_symbol,
|
|
1039
|
+
**meta,
|
|
1040
|
+
"path": [],
|
|
1041
|
+
}
|
|
1042
|
+
|
|
1043
|
+
# Resolve from and to targets
|
|
1044
|
+
from_row = con.execute(
|
|
1045
|
+
"SELECT canonical_id, qualified_name, name FROM symbols "
|
|
1046
|
+
"WHERE canonical_id=? OR qualified_name=? OR name=? LIMIT 1",
|
|
1047
|
+
(clean_from, clean_from, clean_from),
|
|
1048
|
+
).fetchone()
|
|
1049
|
+
|
|
1050
|
+
to_row = con.execute(
|
|
1051
|
+
"SELECT canonical_id, qualified_name, name FROM symbols "
|
|
1052
|
+
"WHERE canonical_id=? OR qualified_name=? OR name=? LIMIT 1",
|
|
1053
|
+
(clean_to, clean_to, clean_to),
|
|
1054
|
+
).fetchone()
|
|
1055
|
+
|
|
1056
|
+
if not from_row or not to_row:
|
|
1057
|
+
missing = clean_from if not from_row else clean_to
|
|
1058
|
+
return {
|
|
1059
|
+
"status": "not_found",
|
|
1060
|
+
"error": {
|
|
1061
|
+
"code": ErrorCode.SYMBOL_NOT_FOUND.value,
|
|
1062
|
+
"message": f"Symbol not found: '{missing}'",
|
|
1063
|
+
"next_action": {
|
|
1064
|
+
"command": f"codegraph search {missing}",
|
|
1065
|
+
"reason": "Verify symbol exists in the repository.",
|
|
1066
|
+
},
|
|
1067
|
+
},
|
|
1068
|
+
"from": clean_from,
|
|
1069
|
+
"to": clean_to,
|
|
1070
|
+
**meta,
|
|
1071
|
+
"path": [],
|
|
1072
|
+
}
|
|
1073
|
+
|
|
1074
|
+
start_canon = from_row["canonical_id"]
|
|
1075
|
+
target_canon = to_row["canonical_id"]
|
|
1076
|
+
target_names = {to_row["canonical_id"], to_row["qualified_name"], to_row["name"]}
|
|
1077
|
+
|
|
1078
|
+
if start_canon == target_canon:
|
|
1079
|
+
return {
|
|
1080
|
+
"status": "ok",
|
|
1081
|
+
"from": clean_from,
|
|
1082
|
+
"to": clean_to,
|
|
1083
|
+
**meta,
|
|
1084
|
+
"path_length": 0,
|
|
1085
|
+
"path": [],
|
|
1086
|
+
}
|
|
1087
|
+
|
|
1088
|
+
# Deterministic BFS search
|
|
1089
|
+
queue: deque[tuple[str, list[dict[str, Any]]]] = deque([(start_canon, [])])
|
|
1090
|
+
visited: set[str] = {start_canon}
|
|
1091
|
+
|
|
1092
|
+
while queue:
|
|
1093
|
+
curr_sym, path = queue.popleft()
|
|
1094
|
+
if len(path) >= bounded_depth:
|
|
1095
|
+
continue
|
|
1096
|
+
|
|
1097
|
+
curr_short = curr_sym.split(":")[-1].split(".")[-1]
|
|
1098
|
+
|
|
1099
|
+
# Query outgoing edges from calls
|
|
1100
|
+
edges = con.execute(
|
|
1101
|
+
"SELECT c.callee, c.qualified_callee, c.resolved_symbol_id, c.source_path, c.line "
|
|
1102
|
+
"FROM calls c "
|
|
1103
|
+
"WHERE c.source_symbol_id=? OR c.callee=? "
|
|
1104
|
+
"ORDER BY c.source_path ASC, c.line ASC",
|
|
1105
|
+
(curr_sym, curr_short),
|
|
1106
|
+
).fetchall()
|
|
1107
|
+
|
|
1108
|
+
# Query outgoing edges from graph_edges
|
|
1109
|
+
graph_rows = con.execute(
|
|
1110
|
+
"SELECT target, relationship, file, start_line, end_line "
|
|
1111
|
+
"FROM graph_edges "
|
|
1112
|
+
"WHERE source=? AND relationship IN ('CALLS', 'IMPORTS', 'HANDLED_BY') "
|
|
1113
|
+
"ORDER BY relationship ASC, target ASC",
|
|
1114
|
+
(curr_sym,),
|
|
1115
|
+
).fetchall()
|
|
1116
|
+
|
|
1117
|
+
all_steps: list[tuple[str, str, str, int, int]] = []
|
|
1118
|
+
for e in edges:
|
|
1119
|
+
dest = e["resolved_symbol_id"] or e["qualified_callee"] or e["callee"]
|
|
1120
|
+
all_steps.append((dest, "CALLS", e["source_path"], e["line"], e["line"]))
|
|
1121
|
+
for g in graph_rows:
|
|
1122
|
+
all_steps.append((g["target"], g["relationship"], g["file"], g["start_line"], g["end_line"]))
|
|
1123
|
+
|
|
1124
|
+
# Sort steps deterministically
|
|
1125
|
+
all_steps.sort(key=lambda s: (s[0], s[1], s[2], s[3]))
|
|
1126
|
+
|
|
1127
|
+
for dest_sym, rel, s_file, s_line, e_line in all_steps:
|
|
1128
|
+
dest_row = con.execute(
|
|
1129
|
+
"SELECT canonical_id FROM symbols WHERE canonical_id=? OR qualified_name=? OR name=? LIMIT 1",
|
|
1130
|
+
(dest_sym, dest_sym, dest_sym),
|
|
1131
|
+
).fetchone()
|
|
1132
|
+
dest_canon = dest_row["canonical_id"] if dest_row else dest_sym
|
|
1133
|
+
|
|
1134
|
+
new_path = list(path) + [
|
|
1135
|
+
{
|
|
1136
|
+
"source": curr_sym,
|
|
1137
|
+
"target": dest_canon,
|
|
1138
|
+
"relationship": rel,
|
|
1139
|
+
"evidence": make_evidence(
|
|
1140
|
+
file=s_file,
|
|
1141
|
+
start_line=s_line,
|
|
1142
|
+
end_line=e_line,
|
|
1143
|
+
evidence_type="path_edge",
|
|
1144
|
+
canonical_id=dest_canon,
|
|
1145
|
+
),
|
|
1146
|
+
}
|
|
1147
|
+
]
|
|
1148
|
+
|
|
1149
|
+
if dest_canon == target_canon or dest_sym in target_names:
|
|
1150
|
+
return {
|
|
1151
|
+
"status": "ok",
|
|
1152
|
+
"from": clean_from,
|
|
1153
|
+
"to": clean_to,
|
|
1154
|
+
**meta,
|
|
1155
|
+
"path_length": len(new_path),
|
|
1156
|
+
"path": new_path,
|
|
1157
|
+
}
|
|
1158
|
+
|
|
1159
|
+
if dest_canon and dest_canon not in visited:
|
|
1160
|
+
visited.add(dest_canon)
|
|
1161
|
+
queue.append((dest_canon, new_path))
|
|
1162
|
+
|
|
1163
|
+
return {
|
|
1164
|
+
"status": "not_found",
|
|
1165
|
+
"from": clean_from,
|
|
1166
|
+
"to": clean_to,
|
|
1167
|
+
"reachable": False,
|
|
1168
|
+
"message": f"No relationship path found between '{clean_from}' and '{clean_to}' within depth {bounded_depth}.",
|
|
1169
|
+
**meta,
|
|
1170
|
+
"path": [],
|
|
1171
|
+
}
|
|
1172
|
+
|
|
1173
|
+
|
|
1174
|
+
# ---------------------------------------------------------------------------
|
|
1175
|
+
# 9. get_imports
|
|
1176
|
+
# ---------------------------------------------------------------------------
|
|
1177
|
+
def get_imports(
|
|
1178
|
+
con: sqlite3.Connection,
|
|
1179
|
+
repository: Path,
|
|
1180
|
+
file: str | None = None,
|
|
1181
|
+
canonical_id: str | None = None,
|
|
1182
|
+
) -> dict[str, Any]:
|
|
1183
|
+
"""Return imports and import relationships for a file or canonical symbol."""
|
|
1184
|
+
unindexed = check_index_available(con, repository)
|
|
1185
|
+
if unindexed:
|
|
1186
|
+
return unindexed
|
|
1187
|
+
|
|
1188
|
+
meta = get_index_metadata(con, repository)
|
|
1189
|
+
target_path = file
|
|
1190
|
+
|
|
1191
|
+
if canonical_id and not target_path:
|
|
1192
|
+
sym_row = con.execute(
|
|
1193
|
+
"SELECT path FROM symbols WHERE canonical_id=? OR qualified_name=? OR name=? LIMIT 1",
|
|
1194
|
+
(canonical_id, canonical_id, canonical_id),
|
|
1195
|
+
).fetchone()
|
|
1196
|
+
if sym_row:
|
|
1197
|
+
target_path = sym_row["path"]
|
|
1198
|
+
|
|
1199
|
+
if not target_path:
|
|
1200
|
+
return {
|
|
1201
|
+
"status": "invalid_request",
|
|
1202
|
+
"error_code": ErrorCode.INVALID_ARGUMENT.value,
|
|
1203
|
+
"error": {
|
|
1204
|
+
"code": ErrorCode.INVALID_ARGUMENT.value,
|
|
1205
|
+
"message": "At least one of 'file' or 'canonical_id' must be specified.",
|
|
1206
|
+
"next_action": {
|
|
1207
|
+
"command": "codegraph get-file <path>",
|
|
1208
|
+
"reason": "Specify a valid file path or canonical ID.",
|
|
1209
|
+
},
|
|
1210
|
+
},
|
|
1211
|
+
**meta,
|
|
1212
|
+
"count": 0,
|
|
1213
|
+
"imports": [],
|
|
1214
|
+
}
|
|
1215
|
+
|
|
1216
|
+
rows = con.execute(
|
|
1217
|
+
"SELECT source_path, module, name, alias, imported_module, imported_name, resolved_path, line "
|
|
1218
|
+
"FROM imports WHERE source_path=? ORDER BY line ASC, module ASC",
|
|
1219
|
+
(target_path,),
|
|
1220
|
+
).fetchall()
|
|
1221
|
+
|
|
1222
|
+
imports = [
|
|
1223
|
+
{
|
|
1224
|
+
"source": r["source_path"],
|
|
1225
|
+
"target": r["imported_module"] or r["module"],
|
|
1226
|
+
"import_type": "symbol" if r["name"] else "module",
|
|
1227
|
+
"file": r["source_path"],
|
|
1228
|
+
"line": r["line"],
|
|
1229
|
+
"resolved_target": r["resolved_path"],
|
|
1230
|
+
"evidence": make_evidence(
|
|
1231
|
+
file=r["source_path"],
|
|
1232
|
+
start_line=r["line"],
|
|
1233
|
+
end_line=r["line"],
|
|
1234
|
+
evidence_type="import",
|
|
1235
|
+
canonical_id=r["source_path"],
|
|
1236
|
+
),
|
|
1237
|
+
}
|
|
1238
|
+
for r in rows
|
|
1239
|
+
]
|
|
1240
|
+
|
|
1241
|
+
imports.sort(key=lambda i: (str(i["file"]), int(i["line"]), str(i["target"])))
|
|
1242
|
+
|
|
1243
|
+
return {
|
|
1244
|
+
"status": "ok",
|
|
1245
|
+
"file": target_path,
|
|
1246
|
+
"canonical_id": canonical_id,
|
|
1247
|
+
**meta,
|
|
1248
|
+
"count": len(imports),
|
|
1249
|
+
"imports": imports,
|
|
1250
|
+
}
|
|
1251
|
+
|
|
1252
|
+
|
|
1253
|
+
# ---------------------------------------------------------------------------
|
|
1254
|
+
# 10. get_dependents
|
|
1255
|
+
# ---------------------------------------------------------------------------
|
|
1256
|
+
def get_dependents(
|
|
1257
|
+
con: sqlite3.Connection,
|
|
1258
|
+
repository: Path,
|
|
1259
|
+
canonical_id: str | None = None,
|
|
1260
|
+
file: str | None = None,
|
|
1261
|
+
) -> dict[str, Any]:
|
|
1262
|
+
"""Reverse dependency query: return files and symbols that depend on the target."""
|
|
1263
|
+
unindexed = check_index_available(con, repository)
|
|
1264
|
+
if unindexed:
|
|
1265
|
+
return unindexed
|
|
1266
|
+
|
|
1267
|
+
meta = get_index_metadata(con, repository)
|
|
1268
|
+
target = canonical_id or file
|
|
1269
|
+
if not target:
|
|
1270
|
+
return {
|
|
1271
|
+
"status": "invalid_request",
|
|
1272
|
+
"error_code": ErrorCode.INVALID_ARGUMENT.value,
|
|
1273
|
+
"error": {
|
|
1274
|
+
"code": ErrorCode.INVALID_ARGUMENT.value,
|
|
1275
|
+
"message": "At least one of 'canonical_id' or 'file' must be specified.",
|
|
1276
|
+
"next_action": {
|
|
1277
|
+
"command": "codegraph resolve <symbol>",
|
|
1278
|
+
"reason": "Specify a valid canonical ID or file path.",
|
|
1279
|
+
},
|
|
1280
|
+
},
|
|
1281
|
+
**meta,
|
|
1282
|
+
"count": 0,
|
|
1283
|
+
"dependents": [],
|
|
1284
|
+
}
|
|
1285
|
+
|
|
1286
|
+
clean_target = target.strip()
|
|
1287
|
+
short_name = clean_target.split(":")[-1].split(".")[-1]
|
|
1288
|
+
dependents: list[dict[str, Any]] = []
|
|
1289
|
+
seen: set[tuple[str, str, int]] = set()
|
|
1290
|
+
|
|
1291
|
+
# 1. Reverse imports
|
|
1292
|
+
imp_rows = con.execute(
|
|
1293
|
+
"SELECT source_path, line, module FROM imports "
|
|
1294
|
+
"WHERE module=? OR imported_module=? OR resolved_path=? OR source_path=? "
|
|
1295
|
+
"ORDER BY source_path ASC, line ASC",
|
|
1296
|
+
(clean_target, clean_target, clean_target, clean_target),
|
|
1297
|
+
).fetchall()
|
|
1298
|
+
for r in imp_rows:
|
|
1299
|
+
key = ("imports", r["source_path"], r["line"])
|
|
1300
|
+
if key not in seen:
|
|
1301
|
+
seen.add(key)
|
|
1302
|
+
dependents.append(
|
|
1303
|
+
{
|
|
1304
|
+
"dependent": r["source_path"],
|
|
1305
|
+
"dependent_type": "file",
|
|
1306
|
+
"relationship": "imports",
|
|
1307
|
+
"file": r["source_path"],
|
|
1308
|
+
"line": r["line"],
|
|
1309
|
+
"evidence": make_evidence(
|
|
1310
|
+
file=r["source_path"],
|
|
1311
|
+
start_line=r["line"],
|
|
1312
|
+
end_line=r["line"],
|
|
1313
|
+
evidence_type="import",
|
|
1314
|
+
canonical_id=r["source_path"],
|
|
1315
|
+
),
|
|
1316
|
+
}
|
|
1317
|
+
)
|
|
1318
|
+
|
|
1319
|
+
# 2. Reverse calls
|
|
1320
|
+
call_rows = con.execute(
|
|
1321
|
+
"SELECT source_symbol_id, source_path, line FROM calls "
|
|
1322
|
+
"WHERE resolved_symbol_id=? OR qualified_callee=? OR callee=? "
|
|
1323
|
+
"ORDER BY source_path ASC, line ASC",
|
|
1324
|
+
(clean_target, clean_target, short_name),
|
|
1325
|
+
).fetchall()
|
|
1326
|
+
for r in call_rows:
|
|
1327
|
+
dep_sym = r["source_symbol_id"] or r["source_path"]
|
|
1328
|
+
key = ("calls", r["source_path"], r["line"])
|
|
1329
|
+
if key not in seen:
|
|
1330
|
+
seen.add(key)
|
|
1331
|
+
dependents.append(
|
|
1332
|
+
{
|
|
1333
|
+
"dependent": dep_sym,
|
|
1334
|
+
"dependent_type": "symbol",
|
|
1335
|
+
"relationship": "calls",
|
|
1336
|
+
"file": r["source_path"],
|
|
1337
|
+
"line": r["line"],
|
|
1338
|
+
"evidence": make_evidence(
|
|
1339
|
+
file=r["source_path"],
|
|
1340
|
+
start_line=r["line"],
|
|
1341
|
+
end_line=r["line"],
|
|
1342
|
+
evidence_type="call",
|
|
1343
|
+
canonical_id=dep_sym,
|
|
1344
|
+
),
|
|
1345
|
+
}
|
|
1346
|
+
)
|
|
1347
|
+
|
|
1348
|
+
# 3. Inheritance
|
|
1349
|
+
inh_rows = con.execute(
|
|
1350
|
+
"SELECT source_symbol, source_file, line, source_canonical_id FROM inheritance "
|
|
1351
|
+
"WHERE base_name=? OR base_name LIKE ? "
|
|
1352
|
+
"ORDER BY source_file ASC, line ASC",
|
|
1353
|
+
(clean_target, f"%.{short_name}"),
|
|
1354
|
+
).fetchall()
|
|
1355
|
+
for r in inh_rows:
|
|
1356
|
+
key = ("inherits", r["source_file"], r["line"])
|
|
1357
|
+
if key not in seen:
|
|
1358
|
+
seen.add(key)
|
|
1359
|
+
dependents.append(
|
|
1360
|
+
{
|
|
1361
|
+
"dependent": r["source_canonical_id"] or r["source_symbol"],
|
|
1362
|
+
"dependent_type": "symbol",
|
|
1363
|
+
"relationship": "inherits",
|
|
1364
|
+
"file": r["source_file"],
|
|
1365
|
+
"line": r["line"],
|
|
1366
|
+
"evidence": make_evidence(
|
|
1367
|
+
file=r["source_file"],
|
|
1368
|
+
start_line=r["line"],
|
|
1369
|
+
end_line=r["line"],
|
|
1370
|
+
evidence_type="inheritance",
|
|
1371
|
+
canonical_id=r["source_canonical_id"],
|
|
1372
|
+
),
|
|
1373
|
+
}
|
|
1374
|
+
)
|
|
1375
|
+
|
|
1376
|
+
dependents.sort(key=lambda d: (str(d["relationship"]), str(d["file"]), int(d["line"])))
|
|
1377
|
+
|
|
1378
|
+
return {
|
|
1379
|
+
"status": "ok",
|
|
1380
|
+
"target": clean_target,
|
|
1381
|
+
**meta,
|
|
1382
|
+
"count": len(dependents),
|
|
1383
|
+
"dependents": dependents,
|
|
1384
|
+
}
|
|
1385
|
+
|
|
1386
|
+
|
|
1387
|
+
# ---------------------------------------------------------------------------
|
|
1388
|
+
# 11. list_routes
|
|
1389
|
+
# ---------------------------------------------------------------------------
|
|
1390
|
+
def list_routes(
|
|
1391
|
+
con: sqlite3.Connection,
|
|
1392
|
+
repository: Path,
|
|
1393
|
+
framework: str | None = None,
|
|
1394
|
+
method: str | None = None,
|
|
1395
|
+
path: str | None = None,
|
|
1396
|
+
) -> dict[str, Any]:
|
|
1397
|
+
"""Expose application routes discovered from the repository."""
|
|
1398
|
+
unindexed = check_index_available(con, repository)
|
|
1399
|
+
if unindexed:
|
|
1400
|
+
return unindexed
|
|
1401
|
+
|
|
1402
|
+
meta = get_index_metadata(con, repository)
|
|
1403
|
+
query = (
|
|
1404
|
+
"SELECT endpoint_id, framework, http_method, route_path, handler_name, "
|
|
1405
|
+
"handler_canonical_id, file_path, line, evidence, confidence "
|
|
1406
|
+
"FROM framework_routes WHERE 1=1 "
|
|
1407
|
+
)
|
|
1408
|
+
params: list[str] = []
|
|
1409
|
+
if framework:
|
|
1410
|
+
query += "AND framework=? "
|
|
1411
|
+
params.append(framework.lower())
|
|
1412
|
+
if method:
|
|
1413
|
+
query += "AND UPPER(http_method)=? "
|
|
1414
|
+
params.append(method.upper())
|
|
1415
|
+
if path:
|
|
1416
|
+
query += "AND route_path LIKE ? "
|
|
1417
|
+
params.append(f"%{path}%")
|
|
1418
|
+
|
|
1419
|
+
query += "ORDER BY route_path ASC, http_method ASC"
|
|
1420
|
+
try:
|
|
1421
|
+
rows = con.execute(query, params).fetchall()
|
|
1422
|
+
except sqlite3.OperationalError:
|
|
1423
|
+
rows = []
|
|
1424
|
+
|
|
1425
|
+
routes = [
|
|
1426
|
+
{
|
|
1427
|
+
"method": r["http_method"],
|
|
1428
|
+
"path": r["route_path"],
|
|
1429
|
+
"handler": r["handler_canonical_id"] or r["handler_name"],
|
|
1430
|
+
"file": r["file_path"],
|
|
1431
|
+
"start_line": r["line"],
|
|
1432
|
+
"end_line": r["line"],
|
|
1433
|
+
"framework": r["framework"],
|
|
1434
|
+
"evidence": make_evidence(
|
|
1435
|
+
file=r["file_path"],
|
|
1436
|
+
start_line=r["line"],
|
|
1437
|
+
end_line=r["line"],
|
|
1438
|
+
evidence_type="route",
|
|
1439
|
+
canonical_id=r["handler_canonical_id"],
|
|
1440
|
+
),
|
|
1441
|
+
}
|
|
1442
|
+
for r in rows
|
|
1443
|
+
]
|
|
1444
|
+
|
|
1445
|
+
return {
|
|
1446
|
+
"status": "ok",
|
|
1447
|
+
**meta,
|
|
1448
|
+
"count": len(routes),
|
|
1449
|
+
"routes": routes,
|
|
1450
|
+
}
|
|
1451
|
+
|
|
1452
|
+
|
|
1453
|
+
# ---------------------------------------------------------------------------
|
|
1454
|
+
# 12. get_architecture
|
|
1455
|
+
# ---------------------------------------------------------------------------
|
|
1456
|
+
def get_architecture(
|
|
1457
|
+
con: sqlite3.Connection,
|
|
1458
|
+
repository: Path,
|
|
1459
|
+
) -> dict[str, Any]:
|
|
1460
|
+
"""Return structural overview of repository architecture."""
|
|
1461
|
+
unindexed = check_index_available(con, repository)
|
|
1462
|
+
if unindexed:
|
|
1463
|
+
return unindexed
|
|
1464
|
+
|
|
1465
|
+
meta = get_index_metadata(con, repository)
|
|
1466
|
+
arch = arch_get_architecture(con, repository)
|
|
1467
|
+
|
|
1468
|
+
route_summary = arch.get("route_summary")
|
|
1469
|
+
routes = route_summary.get("routes", []) if isinstance(route_summary, dict) else []
|
|
1470
|
+
dep_summary = arch.get("dependency_summary")
|
|
1471
|
+
major_deps = dep_summary.get("top_dependencies", []) if isinstance(dep_summary, dict) else []
|
|
1472
|
+
test_summary = arch.get("test_summary")
|
|
1473
|
+
test_frameworks = test_summary.get("frameworks", []) if isinstance(test_summary, dict) else []
|
|
1474
|
+
model_summary = arch.get("data_model_summary")
|
|
1475
|
+
data_models = model_summary.get("models", []) if isinstance(model_summary, dict) else []
|
|
1476
|
+
|
|
1477
|
+
return {
|
|
1478
|
+
"status": "ok",
|
|
1479
|
+
**meta,
|
|
1480
|
+
"repository": str(repository.resolve()),
|
|
1481
|
+
"languages": arch.get("languages", ["Python"]),
|
|
1482
|
+
"directories": arch.get("directories", []),
|
|
1483
|
+
"modules": arch.get("top_level_modules", []),
|
|
1484
|
+
"entrypoints": arch.get("entrypoints", []),
|
|
1485
|
+
"routes": routes,
|
|
1486
|
+
"major_dependencies": major_deps,
|
|
1487
|
+
"test_frameworks": test_frameworks,
|
|
1488
|
+
"data_models": data_models,
|
|
1489
|
+
}
|
|
1490
|
+
|
|
1491
|
+
|
|
1492
|
+
# ---------------------------------------------------------------------------
|
|
1493
|
+
# 13. get_git_impact
|
|
1494
|
+
# ---------------------------------------------------------------------------
|
|
1495
|
+
def get_git_impact(
|
|
1496
|
+
con: sqlite3.Connection,
|
|
1497
|
+
repository: Path,
|
|
1498
|
+
base: str = "HEAD~1",
|
|
1499
|
+
head: str = "HEAD",
|
|
1500
|
+
) -> dict[str, Any]:
|
|
1501
|
+
"""Determine code affected by Git changes between base and head."""
|
|
1502
|
+
unindexed = check_index_available(con, repository)
|
|
1503
|
+
if unindexed:
|
|
1504
|
+
return unindexed
|
|
1505
|
+
|
|
1506
|
+
meta = get_index_metadata(con, repository)
|
|
1507
|
+
diffs = changed_files(repository, since=base, until=head)
|
|
1508
|
+
changed_paths = [d.path for d in diffs]
|
|
1509
|
+
|
|
1510
|
+
changed_symbols: list[str] = []
|
|
1511
|
+
added_symbols: list[str] = []
|
|
1512
|
+
removed_symbols: list[str] = []
|
|
1513
|
+
modified_symbols: list[str] = []
|
|
1514
|
+
|
|
1515
|
+
affected_callers: list[dict[str, Any]] = []
|
|
1516
|
+
affected_callees: list[dict[str, Any]] = []
|
|
1517
|
+
affected_dependents: list[dict[str, Any]] = []
|
|
1518
|
+
affected_routes: list[dict[str, Any]] = []
|
|
1519
|
+
|
|
1520
|
+
for d in diffs:
|
|
1521
|
+
if d.status == "A":
|
|
1522
|
+
s_rows = con.execute("SELECT canonical_id, name FROM symbols WHERE path=?", (d.path,)).fetchall()
|
|
1523
|
+
for s in s_rows:
|
|
1524
|
+
added_symbols.append(s["canonical_id"] or s["name"])
|
|
1525
|
+
elif d.status == "D":
|
|
1526
|
+
removed_symbols.append(d.path)
|
|
1527
|
+
else:
|
|
1528
|
+
s_rows = con.execute("SELECT canonical_id, name FROM symbols WHERE path=?", (d.path,)).fetchall()
|
|
1529
|
+
sym_names = {r["name"] for r in s_rows}
|
|
1530
|
+
matched = changed_symbols_since(repository, base, sym_names)
|
|
1531
|
+
for m in matched:
|
|
1532
|
+
modified_symbols.append(m)
|
|
1533
|
+
changed_symbols.append(m)
|
|
1534
|
+
|
|
1535
|
+
# Compute callers and callees of modified symbols
|
|
1536
|
+
for sym in modified_symbols:
|
|
1537
|
+
callers_res = get_callers(con, repository, sym)
|
|
1538
|
+
for c in callers_res.get("callers", []):
|
|
1539
|
+
affected_callers.append(c)
|
|
1540
|
+
|
|
1541
|
+
callees_res = get_callees(con, repository, sym)
|
|
1542
|
+
for c in callees_res.get("callees", []):
|
|
1543
|
+
affected_callees.append(c)
|
|
1544
|
+
|
|
1545
|
+
dep_res = get_dependents(con, repository, canonical_id=sym)
|
|
1546
|
+
for d in dep_res.get("dependents", []):
|
|
1547
|
+
affected_dependents.append(d)
|
|
1548
|
+
|
|
1549
|
+
# Affected routes
|
|
1550
|
+
for p in changed_paths:
|
|
1551
|
+
r_rows = con.execute("SELECT * FROM framework_routes WHERE file_path=?", (p,)).fetchall()
|
|
1552
|
+
for r in r_rows:
|
|
1553
|
+
affected_routes.append(
|
|
1554
|
+
{
|
|
1555
|
+
"method": r["http_method"],
|
|
1556
|
+
"path": r["route_path"],
|
|
1557
|
+
"handler": r["handler_canonical_id"],
|
|
1558
|
+
"file": r["file_path"],
|
|
1559
|
+
}
|
|
1560
|
+
)
|
|
1561
|
+
|
|
1562
|
+
return {
|
|
1563
|
+
"status": "ok",
|
|
1564
|
+
"base": base,
|
|
1565
|
+
"head": head,
|
|
1566
|
+
**meta,
|
|
1567
|
+
"summary": {
|
|
1568
|
+
"changed_files_count": len(diffs),
|
|
1569
|
+
"affected_symbol_count": len(set(changed_symbols + added_symbols)),
|
|
1570
|
+
"affected_route_count": len(affected_routes),
|
|
1571
|
+
"affected_dependent_count": len(affected_dependents),
|
|
1572
|
+
},
|
|
1573
|
+
"changed_files": [d.as_dict() for d in diffs],
|
|
1574
|
+
"changed_symbols": sorted(list(set(changed_symbols + added_symbols))),
|
|
1575
|
+
"added_symbols": sorted(added_symbols),
|
|
1576
|
+
"removed_symbols": sorted(removed_symbols),
|
|
1577
|
+
"modified_symbols": sorted(modified_symbols),
|
|
1578
|
+
"affected_callers": affected_callers,
|
|
1579
|
+
"affected_callees": affected_callees,
|
|
1580
|
+
"affected_dependents": affected_dependents,
|
|
1581
|
+
"affected_routes": affected_routes,
|
|
1582
|
+
}
|