codegraph-engine 2.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codegraph/__init__.py +37 -0
- codegraph/agent.py +26 -0
- codegraph/architecture.py +328 -0
- codegraph/audit.py +106 -0
- codegraph/cache.py +95 -0
- codegraph/cli.py +854 -0
- codegraph/config.py +43 -0
- codegraph/constraints.py +238 -0
- codegraph/context.py +1228 -0
- codegraph/epistemic.py +90 -0
- codegraph/errors.py +275 -0
- codegraph/evidence/__init__.py +15 -0
- codegraph/evidence/citations.py +397 -0
- codegraph/frameworks.py +434 -0
- codegraph/freshness.py +295 -0
- codegraph/git.py +278 -0
- codegraph/graph/__init__.py +46 -0
- codegraph/graph/models.py +41 -0
- codegraph/graph/traversal.py +1291 -0
- codegraph/indexing/__init__.py +4 -0
- codegraph/indexing/classifier.py +274 -0
- codegraph/indexing/indexer.py +943 -0
- codegraph/indexing/models.py +338 -0
- codegraph/indexing/parser.py +1240 -0
- codegraph/indexing/scanner.py +200 -0
- codegraph/indexing/test_framework.py +116 -0
- codegraph/interrogation.py +1582 -0
- codegraph/llm/__init__.py +3 -0
- codegraph/llm/base.py +15 -0
- codegraph/llm/context.py +20 -0
- codegraph/mcp/__init__.py +3 -0
- codegraph/mcp/server.py +736 -0
- codegraph/memory/__init__.py +3 -0
- codegraph/memory/store.py +46 -0
- codegraph/models.py +289 -0
- codegraph/observability.py +151 -0
- codegraph/optimizer.py +372 -0
- codegraph/planner.py +417 -0
- codegraph/py.typed +1 -0
- codegraph/query_expansion.py +199 -0
- codegraph/ranking.py +363 -0
- codegraph/resolver.py +843 -0
- codegraph/resources/__init__.py +45 -0
- codegraph/resources/cache.py +117 -0
- codegraph/resources/coalescer.py +83 -0
- codegraph/resources/debouncer.py +98 -0
- codegraph/resources/governor.py +232 -0
- codegraph/resources/policy.py +123 -0
- codegraph/retrieval_policy.py +220 -0
- codegraph/search/__init__.py +23 -0
- codegraph/search/hybrid.py +301 -0
- codegraph/search/semantic.py +28 -0
- codegraph/security/__init__.py +3 -0
- codegraph/security/paths.py +35 -0
- codegraph/target_resolver.py +348 -0
- codegraph/task.py +637 -0
- codegraph_engine-2.1.1.dist-info/METADATA +334 -0
- codegraph_engine-2.1.1.dist-info/RECORD +62 -0
- codegraph_engine-2.1.1.dist-info/WHEEL +5 -0
- codegraph_engine-2.1.1.dist-info/entry_points.txt +2 -0
- codegraph_engine-2.1.1.dist-info/licenses/LICENSE +21 -0
- codegraph_engine-2.1.1.dist-info/top_level.txt +1 -0
codegraph/__init__.py
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""CodeGraph MCP: local, evidence-backed codebase intelligence."""
|
|
2
|
+
|
|
3
|
+
__version__ = "2.1.1"
|
|
4
|
+
from codegraph.errors import (
|
|
5
|
+
CodeGraphError,
|
|
6
|
+
ErrorCode,
|
|
7
|
+
IndexStaleError,
|
|
8
|
+
InvalidArgumentError,
|
|
9
|
+
InvalidDepthError,
|
|
10
|
+
InvalidModuleError,
|
|
11
|
+
InvalidPathError,
|
|
12
|
+
NotIndexedError,
|
|
13
|
+
ParseFailureError,
|
|
14
|
+
RepositoryNotInitializedError,
|
|
15
|
+
SecurityError,
|
|
16
|
+
SymbolAmbiguousError,
|
|
17
|
+
SymbolNotFoundError,
|
|
18
|
+
UnsupportedLanguageError,
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
__all__ = [
|
|
22
|
+
"__version__",
|
|
23
|
+
"CodeGraphError",
|
|
24
|
+
"ErrorCode",
|
|
25
|
+
"IndexStaleError",
|
|
26
|
+
"InvalidArgumentError",
|
|
27
|
+
"InvalidDepthError",
|
|
28
|
+
"InvalidModuleError",
|
|
29
|
+
"InvalidPathError",
|
|
30
|
+
"NotIndexedError",
|
|
31
|
+
"ParseFailureError",
|
|
32
|
+
"RepositoryNotInitializedError",
|
|
33
|
+
"SecurityError",
|
|
34
|
+
"SymbolAmbiguousError",
|
|
35
|
+
"SymbolNotFoundError",
|
|
36
|
+
"UnsupportedLanguageError",
|
|
37
|
+
]
|
codegraph/agent.py
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import sqlite3
|
|
4
|
+
from dataclasses import asdict, dataclass
|
|
5
|
+
|
|
6
|
+
from codegraph.llm import LLMProvider
|
|
7
|
+
from codegraph.llm.context import assemble_context
|
|
8
|
+
from codegraph.search import search
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass(frozen=True)
|
|
12
|
+
class Answer:
|
|
13
|
+
answer: str
|
|
14
|
+
evidence: list[dict[str, object]]
|
|
15
|
+
generated: bool
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
async def answer_question(con: sqlite3.Connection, question: str, provider: LLMProvider | None, max_context: int, max_tool_calls: int) -> Answer:
|
|
19
|
+
results = search(con, question, min(max_tool_calls, 20))
|
|
20
|
+
evidence = [asdict(item) for item in results]
|
|
21
|
+
context = assemble_context(results, max_context)
|
|
22
|
+
if provider is None:
|
|
23
|
+
summary = "No LLM provider is configured. Relevant source evidence is returned below."
|
|
24
|
+
return Answer(summary, evidence, False)
|
|
25
|
+
answer = await provider.complete(question, context)
|
|
26
|
+
return Answer(answer, evidence, True)
|
|
@@ -0,0 +1,328 @@
|
|
|
1
|
+
"""Repository architecture intelligence with structured node/edge graph projection.
|
|
2
|
+
|
|
3
|
+
Returns a structured view of the repository: entry points, endpoints, controllers,
|
|
4
|
+
services, models, APIs, tests, config, and framework integration edges.
|
|
5
|
+
|
|
6
|
+
All results are source-derived — no fictional architecture is generated.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import re
|
|
11
|
+
import sqlite3
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
_ENTRY_PATTERNS = re.compile(
|
|
15
|
+
r"(^|[_/])(main|app|server|index|wsgi|asgi|manage|run)\.(py|js|ts)$",
|
|
16
|
+
re.IGNORECASE,
|
|
17
|
+
)
|
|
18
|
+
_TEST_PATTERNS = re.compile(
|
|
19
|
+
r"(^|[_/])test[_s]?[_/]|test[_s]?\.(py|js|ts)$|spec\.(py|js|ts)$",
|
|
20
|
+
re.IGNORECASE,
|
|
21
|
+
)
|
|
22
|
+
_MODEL_PATTERNS = re.compile(
|
|
23
|
+
r"(^|[_/])(model|schema|entity|orm|db)\.(py|js|ts)$|models?(\.py|\.ts|\.js|/)$",
|
|
24
|
+
re.IGNORECASE,
|
|
25
|
+
)
|
|
26
|
+
_CONFIG_PATTERNS = re.compile(
|
|
27
|
+
r"(^|[_/])(config|settings|conf|env)\.(py|js|ts)$",
|
|
28
|
+
re.IGNORECASE,
|
|
29
|
+
)
|
|
30
|
+
_API_PATTERNS = re.compile(
|
|
31
|
+
r"(^|[_/])(api|routes?|views?|controllers?|handlers?)\.(py|js|ts)$|/api/",
|
|
32
|
+
re.IGNORECASE,
|
|
33
|
+
)
|
|
34
|
+
_REPO_PATTERNS = re.compile(
|
|
35
|
+
r"(^|[_/])(repo|repository|store|dao)\.(py|js|ts)$",
|
|
36
|
+
re.IGNORECASE,
|
|
37
|
+
)
|
|
38
|
+
_SERVICE_PATTERNS = re.compile(
|
|
39
|
+
r"(^|[_/])(service|svc|manager|provider)\.(py|js|ts)$",
|
|
40
|
+
re.IGNORECASE,
|
|
41
|
+
)
|
|
42
|
+
_INTEGRATION_PATTERNS = re.compile(
|
|
43
|
+
r"(database|redis|celery|kafka|rabbitmq|s3|stripe|twilio|sendgrid|oauth|jwt|openai)",
|
|
44
|
+
re.IGNORECASE,
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
_FRAMEWORK_IMPORT_PATTERNS: dict[str, re.Pattern[str]] = {
|
|
48
|
+
"flask": re.compile(r"^flask(?:\.|$)", re.IGNORECASE),
|
|
49
|
+
"fastapi": re.compile(r"^fastapi(?:\.|$)", re.IGNORECASE),
|
|
50
|
+
"django": re.compile(r"^django(?:\.|$)", re.IGNORECASE),
|
|
51
|
+
"express": re.compile(r"^express(?:\/|$)", re.IGNORECASE),
|
|
52
|
+
"nextjs": re.compile(r"^next(?:\/|$)", re.IGNORECASE),
|
|
53
|
+
"tornado": re.compile(r"^tornado(?:\.|$)", re.IGNORECASE),
|
|
54
|
+
"aiohttp": re.compile(r"^aiohttp(?:\.|$)", re.IGNORECASE),
|
|
55
|
+
"starlette": re.compile(r"^starlette(?:\.|$)", re.IGNORECASE),
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
_MANIFEST_NAMES = (
|
|
59
|
+
"pyproject.toml",
|
|
60
|
+
"requirements.txt",
|
|
61
|
+
"package.json",
|
|
62
|
+
"setup.py",
|
|
63
|
+
"setup.cfg",
|
|
64
|
+
"Pipfile",
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def detect_frameworks(
|
|
69
|
+
con: sqlite3.Connection,
|
|
70
|
+
repository: Path,
|
|
71
|
+
) -> list[str]:
|
|
72
|
+
"""Detect application frameworks using routes, AST imports, and bounded manifest discovery."""
|
|
73
|
+
detected: set[str] = set()
|
|
74
|
+
|
|
75
|
+
# 1. Concrete routes registered in database
|
|
76
|
+
try:
|
|
77
|
+
fw_rows = con.execute("SELECT DISTINCT framework FROM framework_routes").fetchall()
|
|
78
|
+
for r in fw_rows:
|
|
79
|
+
if r[0]:
|
|
80
|
+
detected.add(str(r[0]).lower())
|
|
81
|
+
except sqlite3.OperationalError:
|
|
82
|
+
pass
|
|
83
|
+
|
|
84
|
+
# 2. AST imports from indexed source files
|
|
85
|
+
try:
|
|
86
|
+
imp_rows = con.execute("SELECT DISTINCT module FROM imports WHERE module != ''").fetchall()
|
|
87
|
+
for r in imp_rows:
|
|
88
|
+
mod = str(r[0]).strip()
|
|
89
|
+
for fw, pat in _FRAMEWORK_IMPORT_PATTERNS.items():
|
|
90
|
+
if pat.search(mod):
|
|
91
|
+
detected.add(fw)
|
|
92
|
+
except sqlite3.OperationalError:
|
|
93
|
+
pass
|
|
94
|
+
|
|
95
|
+
# 3. Bounded recursive manifest discovery (depth <= 3)
|
|
96
|
+
try:
|
|
97
|
+
root = repository.resolve()
|
|
98
|
+
dirs_to_check: list[Path] = [root]
|
|
99
|
+
|
|
100
|
+
def _subdirs(d: Path) -> list[Path]:
|
|
101
|
+
res: list[Path] = []
|
|
102
|
+
try:
|
|
103
|
+
for entry in d.iterdir():
|
|
104
|
+
if entry.is_dir() and not entry.name.startswith(".") and entry.name != "node_modules":
|
|
105
|
+
res.append(entry)
|
|
106
|
+
except OSError:
|
|
107
|
+
pass
|
|
108
|
+
return res
|
|
109
|
+
|
|
110
|
+
level1 = _subdirs(root)
|
|
111
|
+
dirs_to_check.extend(level1)
|
|
112
|
+
for d1 in level1:
|
|
113
|
+
level2 = _subdirs(d1)
|
|
114
|
+
dirs_to_check.extend(level2)
|
|
115
|
+
|
|
116
|
+
for d in dirs_to_check:
|
|
117
|
+
try:
|
|
118
|
+
for item in d.iterdir():
|
|
119
|
+
if not item.is_file():
|
|
120
|
+
continue
|
|
121
|
+
nm = item.name.lower()
|
|
122
|
+
if nm in _MANIFEST_NAMES or (nm.startswith("requirements") and nm.endswith(".txt")):
|
|
123
|
+
text = item.read_text(encoding="utf-8", errors="replace").lower()
|
|
124
|
+
for fw in ("flask", "fastapi", "django", "express", "tornado", "aiohttp", "starlette"):
|
|
125
|
+
if re.search(rf"(?:^|[^\w-]){fw}(?:[^\w-]|$)", text):
|
|
126
|
+
detected.add(fw)
|
|
127
|
+
if "next" in text and ("\"next\"" in text or "'next'" in text or "nextjs" in text):
|
|
128
|
+
detected.add("nextjs")
|
|
129
|
+
except OSError:
|
|
130
|
+
pass
|
|
131
|
+
except Exception:
|
|
132
|
+
pass
|
|
133
|
+
|
|
134
|
+
return sorted(detected)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def get_architecture(
|
|
138
|
+
con: sqlite3.Connection,
|
|
139
|
+
repository: Path,
|
|
140
|
+
max_per_category: int = 20,
|
|
141
|
+
) -> dict[str, object]:
|
|
142
|
+
"""Return a structured, source-derived architecture overview with nodes and edges."""
|
|
143
|
+
files: list[str] = [
|
|
144
|
+
r[0] for r in con.execute("SELECT path FROM files WHERE status='ok' ORDER BY path LIMIT 500")
|
|
145
|
+
]
|
|
146
|
+
|
|
147
|
+
entry_points = [f for f in files if _ENTRY_PATTERNS.search(f)]
|
|
148
|
+
test_files = [f for f in files if _TEST_PATTERNS.search(f)]
|
|
149
|
+
model_files = [f for f in files if _MODEL_PATTERNS.search(f)]
|
|
150
|
+
config_files = [f for f in files if _CONFIG_PATTERNS.search(f)]
|
|
151
|
+
api_files = [f for f in files if _API_PATTERNS.search(f)]
|
|
152
|
+
repo_files = [f for f in files if _REPO_PATTERNS.search(f)]
|
|
153
|
+
service_files = [f for f in files if _SERVICE_PATTERNS.search(f)]
|
|
154
|
+
|
|
155
|
+
# Integration hints from imports
|
|
156
|
+
integrations: list[str] = []
|
|
157
|
+
try:
|
|
158
|
+
imp_rows = con.execute("SELECT DISTINCT module FROM imports LIMIT 500").fetchall()
|
|
159
|
+
for row in imp_rows:
|
|
160
|
+
m = _INTEGRATION_PATTERNS.search(row[0])
|
|
161
|
+
if m:
|
|
162
|
+
integrations.append(m.group(0).lower())
|
|
163
|
+
except sqlite3.OperationalError:
|
|
164
|
+
pass
|
|
165
|
+
|
|
166
|
+
# Top-level classes
|
|
167
|
+
class_rows = con.execute(
|
|
168
|
+
"SELECT canonical_id, qualified_name, path, kind FROM symbols WHERE kind='class' LIMIT 100"
|
|
169
|
+
).fetchall()
|
|
170
|
+
classes = [
|
|
171
|
+
{"name": r["qualified_name"], "canonical_id": r["canonical_id"], "file": r["path"]}
|
|
172
|
+
for r in class_rows
|
|
173
|
+
]
|
|
174
|
+
|
|
175
|
+
# Concrete framework routes
|
|
176
|
+
endpoints: list[dict[str, object]] = []
|
|
177
|
+
nodes: list[dict[str, object]] = []
|
|
178
|
+
edges: list[dict[str, object]] = []
|
|
179
|
+
|
|
180
|
+
try:
|
|
181
|
+
route_rows = con.execute(
|
|
182
|
+
"SELECT endpoint_id, framework, http_method, route_path, normalized_route, "
|
|
183
|
+
"handler_name, handler_canonical_id, file_path, line, evidence, confidence "
|
|
184
|
+
"FROM framework_routes LIMIT 100"
|
|
185
|
+
).fetchall()
|
|
186
|
+
for rr in route_rows:
|
|
187
|
+
ep = {
|
|
188
|
+
"endpoint_id": rr["endpoint_id"],
|
|
189
|
+
"framework": rr["framework"],
|
|
190
|
+
"http_method": rr["http_method"],
|
|
191
|
+
"route_path": rr["route_path"],
|
|
192
|
+
"normalized_route": rr["normalized_route"],
|
|
193
|
+
"handler_name": rr["handler_name"],
|
|
194
|
+
"handler_canonical_id": rr["handler_canonical_id"],
|
|
195
|
+
"file": rr["file_path"],
|
|
196
|
+
"line": rr["line"],
|
|
197
|
+
"evidence": rr["evidence"],
|
|
198
|
+
"confidence": rr["confidence"],
|
|
199
|
+
}
|
|
200
|
+
endpoints.append(ep)
|
|
201
|
+
nodes.append(
|
|
202
|
+
{
|
|
203
|
+
"id": rr["endpoint_id"],
|
|
204
|
+
"name": f"{rr['http_method']} {rr['normalized_route']}",
|
|
205
|
+
"layer": "endpoint",
|
|
206
|
+
"file": rr["file_path"],
|
|
207
|
+
"line": rr["line"],
|
|
208
|
+
"confidence": rr["confidence"],
|
|
209
|
+
}
|
|
210
|
+
)
|
|
211
|
+
if rr["handler_canonical_id"]:
|
|
212
|
+
edges.append(
|
|
213
|
+
{
|
|
214
|
+
"source": rr["endpoint_id"],
|
|
215
|
+
"target": rr["handler_canonical_id"],
|
|
216
|
+
"relationship": "HANDLED_BY",
|
|
217
|
+
"confidence": rr["confidence"],
|
|
218
|
+
"evidence": rr["evidence"],
|
|
219
|
+
}
|
|
220
|
+
)
|
|
221
|
+
except sqlite3.OperationalError:
|
|
222
|
+
pass
|
|
223
|
+
|
|
224
|
+
# Frameworks
|
|
225
|
+
frameworks = detect_frameworks(con, repository)
|
|
226
|
+
|
|
227
|
+
# Top-level modules
|
|
228
|
+
modules: list[str] = []
|
|
229
|
+
try:
|
|
230
|
+
mod_rows = con.execute(
|
|
231
|
+
"SELECT DISTINCT module FROM symbols WHERE module != '' ORDER BY module LIMIT 20"
|
|
232
|
+
).fetchall()
|
|
233
|
+
modules = [r[0] for r in mod_rows]
|
|
234
|
+
except sqlite3.OperationalError:
|
|
235
|
+
pass
|
|
236
|
+
|
|
237
|
+
# Dependencies summary
|
|
238
|
+
dependencies: list[dict[str, object]] = []
|
|
239
|
+
try:
|
|
240
|
+
dep_rows = con.execute(
|
|
241
|
+
"SELECT module, count(*) AS c FROM imports GROUP BY module ORDER BY c DESC LIMIT 20"
|
|
242
|
+
).fetchall()
|
|
243
|
+
dependencies = [{"module": r[0], "count": r[1]} for r in dep_rows]
|
|
244
|
+
except sqlite3.OperationalError:
|
|
245
|
+
pass
|
|
246
|
+
|
|
247
|
+
# Generated artifacts
|
|
248
|
+
generated_files: list[str] = []
|
|
249
|
+
try:
|
|
250
|
+
gen_rows = con.execute("SELECT path FROM files WHERE category='GENERATED' ORDER BY path LIMIT 20").fetchall()
|
|
251
|
+
generated_files = [r[0] for r in gen_rows]
|
|
252
|
+
except sqlite3.OperationalError:
|
|
253
|
+
pass
|
|
254
|
+
|
|
255
|
+
# Test summary with test framework detection
|
|
256
|
+
from codegraph.indexing.test_framework import detect_test_framework
|
|
257
|
+
test_framework, tests_present, test_evidence = detect_test_framework(con, repository)
|
|
258
|
+
test_summary = {
|
|
259
|
+
"framework": test_framework.value,
|
|
260
|
+
"tests_present": tests_present,
|
|
261
|
+
"evidence": test_evidence,
|
|
262
|
+
"test_files_count": len(test_files),
|
|
263
|
+
"test_files": test_files[:max_per_category],
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
# Route summary
|
|
267
|
+
route_summary = {
|
|
268
|
+
"total_routes": len(endpoints),
|
|
269
|
+
"frameworks": frameworks,
|
|
270
|
+
"methods": {
|
|
271
|
+
m: len([e for e in endpoints if e.get("http_method") == m])
|
|
272
|
+
for m in sorted({str(e.get("http_method", "")) for e in endpoints if e.get("http_method")})
|
|
273
|
+
},
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
# Data / model summary
|
|
277
|
+
data_model_summary = {
|
|
278
|
+
"model_files": model_files[:max_per_category],
|
|
279
|
+
"model_classes": [c["name"] for c in classes[:max_per_category]],
|
|
280
|
+
"total_classes": len(classes),
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
# Module count by language
|
|
284
|
+
lang_counts: dict[str, int] = {}
|
|
285
|
+
for r in con.execute("SELECT language, count(*) AS n FROM files GROUP BY language"):
|
|
286
|
+
lang_counts[r["language"]] = r["n"]
|
|
287
|
+
|
|
288
|
+
total_symbols = con.execute("SELECT count(*) FROM symbols").fetchone()[0]
|
|
289
|
+
total_files = len(files)
|
|
290
|
+
|
|
291
|
+
return {
|
|
292
|
+
"source": "deterministic — parser-extracted from repository index",
|
|
293
|
+
"repository": str(repository),
|
|
294
|
+
"summary": {
|
|
295
|
+
"total_files": total_files,
|
|
296
|
+
"total_symbols": total_symbols,
|
|
297
|
+
"total_endpoints": len(endpoints),
|
|
298
|
+
"languages": lang_counts,
|
|
299
|
+
},
|
|
300
|
+
"languages": lang_counts,
|
|
301
|
+
"frameworks": frameworks,
|
|
302
|
+
"entry_points": entry_points[:max_per_category],
|
|
303
|
+
"top_level_modules": modules,
|
|
304
|
+
"dependency_summary": dependencies,
|
|
305
|
+
"route_summary": route_summary,
|
|
306
|
+
"test_summary": test_summary,
|
|
307
|
+
"data_model_summary": data_model_summary,
|
|
308
|
+
"external_services": sorted(set(integrations))[:max_per_category],
|
|
309
|
+
"generated_artifact_summary": {
|
|
310
|
+
"count": len(generated_files),
|
|
311
|
+
"files": generated_files,
|
|
312
|
+
},
|
|
313
|
+
"endpoints": endpoints[:max_per_category],
|
|
314
|
+
"services": service_files[:max_per_category],
|
|
315
|
+
"repositories": repo_files[:max_per_category],
|
|
316
|
+
"models": model_files[:max_per_category],
|
|
317
|
+
"apis": api_files[:max_per_category],
|
|
318
|
+
"tests": test_files[:max_per_category],
|
|
319
|
+
"config": config_files[:max_per_category],
|
|
320
|
+
"classes": classes[:max_per_category],
|
|
321
|
+
"external_integrations": sorted(set(integrations))[:max_per_category],
|
|
322
|
+
"architecture_nodes": nodes[:max_per_category * 2],
|
|
323
|
+
"architecture_edges": edges[:max_per_category * 2],
|
|
324
|
+
"note": (
|
|
325
|
+
"Architecture is inferred from source declarations, routes, and import patterns. "
|
|
326
|
+
"Framework endpoints are verified from AST route registrations."
|
|
327
|
+
),
|
|
328
|
+
}
|
codegraph/audit.py
ADDED
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
"""Optional local audit log for MCP tool operations.
|
|
2
|
+
|
|
3
|
+
Records: timestamp, tool, operation, repository, files accessed, duration.
|
|
4
|
+
File contents are NEVER recorded.
|
|
5
|
+
Log can be cleared via CLI. Disabled by default.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import json
|
|
10
|
+
import sqlite3
|
|
11
|
+
import time
|
|
12
|
+
from collections.abc import Iterator
|
|
13
|
+
from contextlib import contextmanager
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class AuditLog:
|
|
18
|
+
"""SQLite-backed local audit log. Contents never include source code."""
|
|
19
|
+
|
|
20
|
+
def __init__(self, db_path: Path, enabled: bool = False) -> None:
|
|
21
|
+
self.db_path = db_path
|
|
22
|
+
self.enabled = enabled
|
|
23
|
+
if enabled:
|
|
24
|
+
self._ensure_table()
|
|
25
|
+
|
|
26
|
+
def _connect(self) -> sqlite3.Connection:
|
|
27
|
+
con = sqlite3.connect(self.db_path)
|
|
28
|
+
con.row_factory = sqlite3.Row
|
|
29
|
+
return con
|
|
30
|
+
|
|
31
|
+
def _ensure_table(self) -> None:
|
|
32
|
+
with self._session() as con:
|
|
33
|
+
con.execute("""
|
|
34
|
+
CREATE TABLE IF NOT EXISTS audit_log (
|
|
35
|
+
id INTEGER PRIMARY KEY,
|
|
36
|
+
ts INTEGER NOT NULL,
|
|
37
|
+
tool TEXT NOT NULL,
|
|
38
|
+
operation TEXT NOT NULL,
|
|
39
|
+
repository TEXT NOT NULL,
|
|
40
|
+
files_accessed TEXT NOT NULL DEFAULT '[]',
|
|
41
|
+
duration_ms REAL NOT NULL DEFAULT 0
|
|
42
|
+
)
|
|
43
|
+
""")
|
|
44
|
+
|
|
45
|
+
@contextmanager
|
|
46
|
+
def _session(self) -> Iterator[sqlite3.Connection]:
|
|
47
|
+
con = self._connect()
|
|
48
|
+
try:
|
|
49
|
+
yield con
|
|
50
|
+
con.commit()
|
|
51
|
+
except BaseException:
|
|
52
|
+
con.rollback()
|
|
53
|
+
raise
|
|
54
|
+
finally:
|
|
55
|
+
con.close()
|
|
56
|
+
|
|
57
|
+
def record(
|
|
58
|
+
self,
|
|
59
|
+
tool: str,
|
|
60
|
+
operation: str,
|
|
61
|
+
repository: str,
|
|
62
|
+
files_accessed: list[str] | None = None,
|
|
63
|
+
duration_ms: float = 0.0,
|
|
64
|
+
) -> None:
|
|
65
|
+
if not self.enabled:
|
|
66
|
+
return
|
|
67
|
+
with self._session() as con:
|
|
68
|
+
con.execute(
|
|
69
|
+
"INSERT INTO audit_log(ts, tool, operation, repository, files_accessed, duration_ms) "
|
|
70
|
+
"VALUES (?, ?, ?, ?, ?, ?)",
|
|
71
|
+
(
|
|
72
|
+
int(time.time()),
|
|
73
|
+
tool,
|
|
74
|
+
operation,
|
|
75
|
+
repository,
|
|
76
|
+
json.dumps(files_accessed or []),
|
|
77
|
+
duration_ms,
|
|
78
|
+
),
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
def recent(self, limit: int = 100) -> list[dict[str, object]]:
|
|
82
|
+
if not self.enabled:
|
|
83
|
+
return []
|
|
84
|
+
with self._session() as con:
|
|
85
|
+
rows = con.execute(
|
|
86
|
+
"SELECT ts, tool, operation, repository, files_accessed, duration_ms "
|
|
87
|
+
"FROM audit_log ORDER BY id DESC LIMIT ?",
|
|
88
|
+
(limit,),
|
|
89
|
+
).fetchall()
|
|
90
|
+
return [
|
|
91
|
+
{
|
|
92
|
+
"ts": r["ts"],
|
|
93
|
+
"tool": r["tool"],
|
|
94
|
+
"operation": r["operation"],
|
|
95
|
+
"repository": r["repository"],
|
|
96
|
+
"files_accessed": json.loads(r["files_accessed"]),
|
|
97
|
+
"duration_ms": r["duration_ms"],
|
|
98
|
+
}
|
|
99
|
+
for r in rows
|
|
100
|
+
]
|
|
101
|
+
|
|
102
|
+
def clear(self) -> None:
|
|
103
|
+
if not self.enabled:
|
|
104
|
+
return
|
|
105
|
+
with self._session() as con:
|
|
106
|
+
con.execute("DELETE FROM audit_log")
|
codegraph/cache.py
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
"""Deterministic Task Fingerprinting and Context Caching Engine.
|
|
2
|
+
|
|
3
|
+
Guarantees repository generation consistency and cache invalidation on:
|
|
4
|
+
- Source code or index generation change
|
|
5
|
+
- Parser version change
|
|
6
|
+
- Database schema change
|
|
7
|
+
- Configuration change
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import hashlib
|
|
12
|
+
import json
|
|
13
|
+
import sqlite3
|
|
14
|
+
import time
|
|
15
|
+
from typing import Any, cast
|
|
16
|
+
|
|
17
|
+
from codegraph.indexing.parser import PARSER_VERSION
|
|
18
|
+
from codegraph.task import TaskSpec
|
|
19
|
+
|
|
20
|
+
CACHE_SCHEMA_VERSION = 1
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def compute_context_cache_key(
|
|
24
|
+
task_spec: TaskSpec,
|
|
25
|
+
repository_generation: int,
|
|
26
|
+
max_tokens: int,
|
|
27
|
+
parser_version: str = PARSER_VERSION,
|
|
28
|
+
extra_config: str = "",
|
|
29
|
+
) -> str:
|
|
30
|
+
"""Compute an immutable, deterministic cache key for a context compilation request."""
|
|
31
|
+
canonical_components = {
|
|
32
|
+
"intent": task_spec.intent,
|
|
33
|
+
"goal": task_spec.goal,
|
|
34
|
+
"targets": sorted(task_spec.targets),
|
|
35
|
+
"operations": sorted(task_spec.operations),
|
|
36
|
+
"constraints": sorted(task_spec.constraints),
|
|
37
|
+
"exclusions": sorted(task_spec.exclusions),
|
|
38
|
+
"scope_paths": sorted(task_spec.scope_paths),
|
|
39
|
+
"time_scope": task_spec.time_scope or "",
|
|
40
|
+
"repository_generation": repository_generation,
|
|
41
|
+
"max_tokens": max_tokens,
|
|
42
|
+
"parser_version": parser_version,
|
|
43
|
+
"cache_schema": CACHE_SCHEMA_VERSION,
|
|
44
|
+
"extra_config": extra_config,
|
|
45
|
+
}
|
|
46
|
+
raw = json.dumps(canonical_components, sort_keys=True, separators=(",", ":")).encode("utf-8")
|
|
47
|
+
return f"ctx:{hashlib.sha256(raw).hexdigest()}"
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def get_cached_context_packet(
|
|
51
|
+
con: sqlite3.Connection,
|
|
52
|
+
cache_key: str,
|
|
53
|
+
) -> dict[str, Any] | None:
|
|
54
|
+
"""Retrieve cached context packet if present in SQLite context_cache."""
|
|
55
|
+
try:
|
|
56
|
+
row = con.execute(
|
|
57
|
+
"SELECT data FROM context_cache WHERE cache_key=?",
|
|
58
|
+
(cache_key,),
|
|
59
|
+
).fetchone()
|
|
60
|
+
if row:
|
|
61
|
+
raw_data = row["data"] if isinstance(row, sqlite3.Row) else row[0]
|
|
62
|
+
parsed = json.loads(raw_data)
|
|
63
|
+
if isinstance(parsed, dict):
|
|
64
|
+
return cast(dict[str, Any], parsed)
|
|
65
|
+
return None
|
|
66
|
+
except Exception:
|
|
67
|
+
pass
|
|
68
|
+
return None
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def store_cached_context_packet(
|
|
72
|
+
con: sqlite3.Connection,
|
|
73
|
+
cache_key: str,
|
|
74
|
+
packet_dict: dict[str, Any],
|
|
75
|
+
) -> None:
|
|
76
|
+
"""Store compiled ContextPacket in SQLite context_cache table."""
|
|
77
|
+
try:
|
|
78
|
+
now = int(time.time())
|
|
79
|
+
serialized = json.dumps(packet_dict, sort_keys=True)
|
|
80
|
+
con.execute(
|
|
81
|
+
"INSERT OR REPLACE INTO context_cache (cache_key, data, created_at) VALUES (?, ?, ?)",
|
|
82
|
+
(cache_key, serialized, now),
|
|
83
|
+
)
|
|
84
|
+
con.commit()
|
|
85
|
+
except Exception:
|
|
86
|
+
pass
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def invalidate_context_cache(con: sqlite3.Connection) -> None:
|
|
90
|
+
"""Explicitly wipe context cache upon repository mutation or index generation advancement."""
|
|
91
|
+
try:
|
|
92
|
+
con.execute("DELETE FROM context_cache")
|
|
93
|
+
con.commit()
|
|
94
|
+
except Exception:
|
|
95
|
+
pass
|