codegraph-engine 2.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. codegraph/__init__.py +37 -0
  2. codegraph/agent.py +26 -0
  3. codegraph/architecture.py +328 -0
  4. codegraph/audit.py +106 -0
  5. codegraph/cache.py +95 -0
  6. codegraph/cli.py +854 -0
  7. codegraph/config.py +43 -0
  8. codegraph/constraints.py +238 -0
  9. codegraph/context.py +1228 -0
  10. codegraph/epistemic.py +90 -0
  11. codegraph/errors.py +275 -0
  12. codegraph/evidence/__init__.py +15 -0
  13. codegraph/evidence/citations.py +397 -0
  14. codegraph/frameworks.py +434 -0
  15. codegraph/freshness.py +295 -0
  16. codegraph/git.py +278 -0
  17. codegraph/graph/__init__.py +46 -0
  18. codegraph/graph/models.py +41 -0
  19. codegraph/graph/traversal.py +1291 -0
  20. codegraph/indexing/__init__.py +4 -0
  21. codegraph/indexing/classifier.py +274 -0
  22. codegraph/indexing/indexer.py +943 -0
  23. codegraph/indexing/models.py +338 -0
  24. codegraph/indexing/parser.py +1240 -0
  25. codegraph/indexing/scanner.py +200 -0
  26. codegraph/indexing/test_framework.py +116 -0
  27. codegraph/interrogation.py +1582 -0
  28. codegraph/llm/__init__.py +3 -0
  29. codegraph/llm/base.py +15 -0
  30. codegraph/llm/context.py +20 -0
  31. codegraph/mcp/__init__.py +3 -0
  32. codegraph/mcp/server.py +736 -0
  33. codegraph/memory/__init__.py +3 -0
  34. codegraph/memory/store.py +46 -0
  35. codegraph/models.py +289 -0
  36. codegraph/observability.py +151 -0
  37. codegraph/optimizer.py +372 -0
  38. codegraph/planner.py +417 -0
  39. codegraph/py.typed +1 -0
  40. codegraph/query_expansion.py +199 -0
  41. codegraph/ranking.py +363 -0
  42. codegraph/resolver.py +843 -0
  43. codegraph/resources/__init__.py +45 -0
  44. codegraph/resources/cache.py +117 -0
  45. codegraph/resources/coalescer.py +83 -0
  46. codegraph/resources/debouncer.py +98 -0
  47. codegraph/resources/governor.py +232 -0
  48. codegraph/resources/policy.py +123 -0
  49. codegraph/retrieval_policy.py +220 -0
  50. codegraph/search/__init__.py +23 -0
  51. codegraph/search/hybrid.py +301 -0
  52. codegraph/search/semantic.py +28 -0
  53. codegraph/security/__init__.py +3 -0
  54. codegraph/security/paths.py +35 -0
  55. codegraph/target_resolver.py +348 -0
  56. codegraph/task.py +637 -0
  57. codegraph_engine-2.1.1.dist-info/METADATA +334 -0
  58. codegraph_engine-2.1.1.dist-info/RECORD +62 -0
  59. codegraph_engine-2.1.1.dist-info/WHEEL +5 -0
  60. codegraph_engine-2.1.1.dist-info/entry_points.txt +2 -0
  61. codegraph_engine-2.1.1.dist-info/licenses/LICENSE +21 -0
  62. codegraph_engine-2.1.1.dist-info/top_level.txt +1 -0
codegraph/__init__.py ADDED
@@ -0,0 +1,37 @@
1
+ """CodeGraph MCP: local, evidence-backed codebase intelligence."""
2
+
3
+ __version__ = "2.1.1"
4
+ from codegraph.errors import (
5
+ CodeGraphError,
6
+ ErrorCode,
7
+ IndexStaleError,
8
+ InvalidArgumentError,
9
+ InvalidDepthError,
10
+ InvalidModuleError,
11
+ InvalidPathError,
12
+ NotIndexedError,
13
+ ParseFailureError,
14
+ RepositoryNotInitializedError,
15
+ SecurityError,
16
+ SymbolAmbiguousError,
17
+ SymbolNotFoundError,
18
+ UnsupportedLanguageError,
19
+ )
20
+
21
+ __all__ = [
22
+ "__version__",
23
+ "CodeGraphError",
24
+ "ErrorCode",
25
+ "IndexStaleError",
26
+ "InvalidArgumentError",
27
+ "InvalidDepthError",
28
+ "InvalidModuleError",
29
+ "InvalidPathError",
30
+ "NotIndexedError",
31
+ "ParseFailureError",
32
+ "RepositoryNotInitializedError",
33
+ "SecurityError",
34
+ "SymbolAmbiguousError",
35
+ "SymbolNotFoundError",
36
+ "UnsupportedLanguageError",
37
+ ]
codegraph/agent.py ADDED
@@ -0,0 +1,26 @@
1
+ from __future__ import annotations
2
+
3
+ import sqlite3
4
+ from dataclasses import asdict, dataclass
5
+
6
+ from codegraph.llm import LLMProvider
7
+ from codegraph.llm.context import assemble_context
8
+ from codegraph.search import search
9
+
10
+
11
+ @dataclass(frozen=True)
12
+ class Answer:
13
+ answer: str
14
+ evidence: list[dict[str, object]]
15
+ generated: bool
16
+
17
+
18
+ async def answer_question(con: sqlite3.Connection, question: str, provider: LLMProvider | None, max_context: int, max_tool_calls: int) -> Answer:
19
+ results = search(con, question, min(max_tool_calls, 20))
20
+ evidence = [asdict(item) for item in results]
21
+ context = assemble_context(results, max_context)
22
+ if provider is None:
23
+ summary = "No LLM provider is configured. Relevant source evidence is returned below."
24
+ return Answer(summary, evidence, False)
25
+ answer = await provider.complete(question, context)
26
+ return Answer(answer, evidence, True)
@@ -0,0 +1,328 @@
1
+ """Repository architecture intelligence with structured node/edge graph projection.
2
+
3
+ Returns a structured view of the repository: entry points, endpoints, controllers,
4
+ services, models, APIs, tests, config, and framework integration edges.
5
+
6
+ All results are source-derived — no fictional architecture is generated.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ import re
11
+ import sqlite3
12
+ from pathlib import Path
13
+
14
+ _ENTRY_PATTERNS = re.compile(
15
+ r"(^|[_/])(main|app|server|index|wsgi|asgi|manage|run)\.(py|js|ts)$",
16
+ re.IGNORECASE,
17
+ )
18
+ _TEST_PATTERNS = re.compile(
19
+ r"(^|[_/])test[_s]?[_/]|test[_s]?\.(py|js|ts)$|spec\.(py|js|ts)$",
20
+ re.IGNORECASE,
21
+ )
22
+ _MODEL_PATTERNS = re.compile(
23
+ r"(^|[_/])(model|schema|entity|orm|db)\.(py|js|ts)$|models?(\.py|\.ts|\.js|/)$",
24
+ re.IGNORECASE,
25
+ )
26
+ _CONFIG_PATTERNS = re.compile(
27
+ r"(^|[_/])(config|settings|conf|env)\.(py|js|ts)$",
28
+ re.IGNORECASE,
29
+ )
30
+ _API_PATTERNS = re.compile(
31
+ r"(^|[_/])(api|routes?|views?|controllers?|handlers?)\.(py|js|ts)$|/api/",
32
+ re.IGNORECASE,
33
+ )
34
+ _REPO_PATTERNS = re.compile(
35
+ r"(^|[_/])(repo|repository|store|dao)\.(py|js|ts)$",
36
+ re.IGNORECASE,
37
+ )
38
+ _SERVICE_PATTERNS = re.compile(
39
+ r"(^|[_/])(service|svc|manager|provider)\.(py|js|ts)$",
40
+ re.IGNORECASE,
41
+ )
42
+ _INTEGRATION_PATTERNS = re.compile(
43
+ r"(database|redis|celery|kafka|rabbitmq|s3|stripe|twilio|sendgrid|oauth|jwt|openai)",
44
+ re.IGNORECASE,
45
+ )
46
+
47
+ _FRAMEWORK_IMPORT_PATTERNS: dict[str, re.Pattern[str]] = {
48
+ "flask": re.compile(r"^flask(?:\.|$)", re.IGNORECASE),
49
+ "fastapi": re.compile(r"^fastapi(?:\.|$)", re.IGNORECASE),
50
+ "django": re.compile(r"^django(?:\.|$)", re.IGNORECASE),
51
+ "express": re.compile(r"^express(?:\/|$)", re.IGNORECASE),
52
+ "nextjs": re.compile(r"^next(?:\/|$)", re.IGNORECASE),
53
+ "tornado": re.compile(r"^tornado(?:\.|$)", re.IGNORECASE),
54
+ "aiohttp": re.compile(r"^aiohttp(?:\.|$)", re.IGNORECASE),
55
+ "starlette": re.compile(r"^starlette(?:\.|$)", re.IGNORECASE),
56
+ }
57
+
58
+ _MANIFEST_NAMES = (
59
+ "pyproject.toml",
60
+ "requirements.txt",
61
+ "package.json",
62
+ "setup.py",
63
+ "setup.cfg",
64
+ "Pipfile",
65
+ )
66
+
67
+
68
+ def detect_frameworks(
69
+ con: sqlite3.Connection,
70
+ repository: Path,
71
+ ) -> list[str]:
72
+ """Detect application frameworks using routes, AST imports, and bounded manifest discovery."""
73
+ detected: set[str] = set()
74
+
75
+ # 1. Concrete routes registered in database
76
+ try:
77
+ fw_rows = con.execute("SELECT DISTINCT framework FROM framework_routes").fetchall()
78
+ for r in fw_rows:
79
+ if r[0]:
80
+ detected.add(str(r[0]).lower())
81
+ except sqlite3.OperationalError:
82
+ pass
83
+
84
+ # 2. AST imports from indexed source files
85
+ try:
86
+ imp_rows = con.execute("SELECT DISTINCT module FROM imports WHERE module != ''").fetchall()
87
+ for r in imp_rows:
88
+ mod = str(r[0]).strip()
89
+ for fw, pat in _FRAMEWORK_IMPORT_PATTERNS.items():
90
+ if pat.search(mod):
91
+ detected.add(fw)
92
+ except sqlite3.OperationalError:
93
+ pass
94
+
95
+ # 3. Bounded recursive manifest discovery (depth <= 3)
96
+ try:
97
+ root = repository.resolve()
98
+ dirs_to_check: list[Path] = [root]
99
+
100
+ def _subdirs(d: Path) -> list[Path]:
101
+ res: list[Path] = []
102
+ try:
103
+ for entry in d.iterdir():
104
+ if entry.is_dir() and not entry.name.startswith(".") and entry.name != "node_modules":
105
+ res.append(entry)
106
+ except OSError:
107
+ pass
108
+ return res
109
+
110
+ level1 = _subdirs(root)
111
+ dirs_to_check.extend(level1)
112
+ for d1 in level1:
113
+ level2 = _subdirs(d1)
114
+ dirs_to_check.extend(level2)
115
+
116
+ for d in dirs_to_check:
117
+ try:
118
+ for item in d.iterdir():
119
+ if not item.is_file():
120
+ continue
121
+ nm = item.name.lower()
122
+ if nm in _MANIFEST_NAMES or (nm.startswith("requirements") and nm.endswith(".txt")):
123
+ text = item.read_text(encoding="utf-8", errors="replace").lower()
124
+ for fw in ("flask", "fastapi", "django", "express", "tornado", "aiohttp", "starlette"):
125
+ if re.search(rf"(?:^|[^\w-]){fw}(?:[^\w-]|$)", text):
126
+ detected.add(fw)
127
+ if "next" in text and ("\"next\"" in text or "'next'" in text or "nextjs" in text):
128
+ detected.add("nextjs")
129
+ except OSError:
130
+ pass
131
+ except Exception:
132
+ pass
133
+
134
+ return sorted(detected)
135
+
136
+
137
+ def get_architecture(
138
+ con: sqlite3.Connection,
139
+ repository: Path,
140
+ max_per_category: int = 20,
141
+ ) -> dict[str, object]:
142
+ """Return a structured, source-derived architecture overview with nodes and edges."""
143
+ files: list[str] = [
144
+ r[0] for r in con.execute("SELECT path FROM files WHERE status='ok' ORDER BY path LIMIT 500")
145
+ ]
146
+
147
+ entry_points = [f for f in files if _ENTRY_PATTERNS.search(f)]
148
+ test_files = [f for f in files if _TEST_PATTERNS.search(f)]
149
+ model_files = [f for f in files if _MODEL_PATTERNS.search(f)]
150
+ config_files = [f for f in files if _CONFIG_PATTERNS.search(f)]
151
+ api_files = [f for f in files if _API_PATTERNS.search(f)]
152
+ repo_files = [f for f in files if _REPO_PATTERNS.search(f)]
153
+ service_files = [f for f in files if _SERVICE_PATTERNS.search(f)]
154
+
155
+ # Integration hints from imports
156
+ integrations: list[str] = []
157
+ try:
158
+ imp_rows = con.execute("SELECT DISTINCT module FROM imports LIMIT 500").fetchall()
159
+ for row in imp_rows:
160
+ m = _INTEGRATION_PATTERNS.search(row[0])
161
+ if m:
162
+ integrations.append(m.group(0).lower())
163
+ except sqlite3.OperationalError:
164
+ pass
165
+
166
+ # Top-level classes
167
+ class_rows = con.execute(
168
+ "SELECT canonical_id, qualified_name, path, kind FROM symbols WHERE kind='class' LIMIT 100"
169
+ ).fetchall()
170
+ classes = [
171
+ {"name": r["qualified_name"], "canonical_id": r["canonical_id"], "file": r["path"]}
172
+ for r in class_rows
173
+ ]
174
+
175
+ # Concrete framework routes
176
+ endpoints: list[dict[str, object]] = []
177
+ nodes: list[dict[str, object]] = []
178
+ edges: list[dict[str, object]] = []
179
+
180
+ try:
181
+ route_rows = con.execute(
182
+ "SELECT endpoint_id, framework, http_method, route_path, normalized_route, "
183
+ "handler_name, handler_canonical_id, file_path, line, evidence, confidence "
184
+ "FROM framework_routes LIMIT 100"
185
+ ).fetchall()
186
+ for rr in route_rows:
187
+ ep = {
188
+ "endpoint_id": rr["endpoint_id"],
189
+ "framework": rr["framework"],
190
+ "http_method": rr["http_method"],
191
+ "route_path": rr["route_path"],
192
+ "normalized_route": rr["normalized_route"],
193
+ "handler_name": rr["handler_name"],
194
+ "handler_canonical_id": rr["handler_canonical_id"],
195
+ "file": rr["file_path"],
196
+ "line": rr["line"],
197
+ "evidence": rr["evidence"],
198
+ "confidence": rr["confidence"],
199
+ }
200
+ endpoints.append(ep)
201
+ nodes.append(
202
+ {
203
+ "id": rr["endpoint_id"],
204
+ "name": f"{rr['http_method']} {rr['normalized_route']}",
205
+ "layer": "endpoint",
206
+ "file": rr["file_path"],
207
+ "line": rr["line"],
208
+ "confidence": rr["confidence"],
209
+ }
210
+ )
211
+ if rr["handler_canonical_id"]:
212
+ edges.append(
213
+ {
214
+ "source": rr["endpoint_id"],
215
+ "target": rr["handler_canonical_id"],
216
+ "relationship": "HANDLED_BY",
217
+ "confidence": rr["confidence"],
218
+ "evidence": rr["evidence"],
219
+ }
220
+ )
221
+ except sqlite3.OperationalError:
222
+ pass
223
+
224
+ # Frameworks
225
+ frameworks = detect_frameworks(con, repository)
226
+
227
+ # Top-level modules
228
+ modules: list[str] = []
229
+ try:
230
+ mod_rows = con.execute(
231
+ "SELECT DISTINCT module FROM symbols WHERE module != '' ORDER BY module LIMIT 20"
232
+ ).fetchall()
233
+ modules = [r[0] for r in mod_rows]
234
+ except sqlite3.OperationalError:
235
+ pass
236
+
237
+ # Dependencies summary
238
+ dependencies: list[dict[str, object]] = []
239
+ try:
240
+ dep_rows = con.execute(
241
+ "SELECT module, count(*) AS c FROM imports GROUP BY module ORDER BY c DESC LIMIT 20"
242
+ ).fetchall()
243
+ dependencies = [{"module": r[0], "count": r[1]} for r in dep_rows]
244
+ except sqlite3.OperationalError:
245
+ pass
246
+
247
+ # Generated artifacts
248
+ generated_files: list[str] = []
249
+ try:
250
+ gen_rows = con.execute("SELECT path FROM files WHERE category='GENERATED' ORDER BY path LIMIT 20").fetchall()
251
+ generated_files = [r[0] for r in gen_rows]
252
+ except sqlite3.OperationalError:
253
+ pass
254
+
255
+ # Test summary with test framework detection
256
+ from codegraph.indexing.test_framework import detect_test_framework
257
+ test_framework, tests_present, test_evidence = detect_test_framework(con, repository)
258
+ test_summary = {
259
+ "framework": test_framework.value,
260
+ "tests_present": tests_present,
261
+ "evidence": test_evidence,
262
+ "test_files_count": len(test_files),
263
+ "test_files": test_files[:max_per_category],
264
+ }
265
+
266
+ # Route summary
267
+ route_summary = {
268
+ "total_routes": len(endpoints),
269
+ "frameworks": frameworks,
270
+ "methods": {
271
+ m: len([e for e in endpoints if e.get("http_method") == m])
272
+ for m in sorted({str(e.get("http_method", "")) for e in endpoints if e.get("http_method")})
273
+ },
274
+ }
275
+
276
+ # Data / model summary
277
+ data_model_summary = {
278
+ "model_files": model_files[:max_per_category],
279
+ "model_classes": [c["name"] for c in classes[:max_per_category]],
280
+ "total_classes": len(classes),
281
+ }
282
+
283
+ # Module count by language
284
+ lang_counts: dict[str, int] = {}
285
+ for r in con.execute("SELECT language, count(*) AS n FROM files GROUP BY language"):
286
+ lang_counts[r["language"]] = r["n"]
287
+
288
+ total_symbols = con.execute("SELECT count(*) FROM symbols").fetchone()[0]
289
+ total_files = len(files)
290
+
291
+ return {
292
+ "source": "deterministic — parser-extracted from repository index",
293
+ "repository": str(repository),
294
+ "summary": {
295
+ "total_files": total_files,
296
+ "total_symbols": total_symbols,
297
+ "total_endpoints": len(endpoints),
298
+ "languages": lang_counts,
299
+ },
300
+ "languages": lang_counts,
301
+ "frameworks": frameworks,
302
+ "entry_points": entry_points[:max_per_category],
303
+ "top_level_modules": modules,
304
+ "dependency_summary": dependencies,
305
+ "route_summary": route_summary,
306
+ "test_summary": test_summary,
307
+ "data_model_summary": data_model_summary,
308
+ "external_services": sorted(set(integrations))[:max_per_category],
309
+ "generated_artifact_summary": {
310
+ "count": len(generated_files),
311
+ "files": generated_files,
312
+ },
313
+ "endpoints": endpoints[:max_per_category],
314
+ "services": service_files[:max_per_category],
315
+ "repositories": repo_files[:max_per_category],
316
+ "models": model_files[:max_per_category],
317
+ "apis": api_files[:max_per_category],
318
+ "tests": test_files[:max_per_category],
319
+ "config": config_files[:max_per_category],
320
+ "classes": classes[:max_per_category],
321
+ "external_integrations": sorted(set(integrations))[:max_per_category],
322
+ "architecture_nodes": nodes[:max_per_category * 2],
323
+ "architecture_edges": edges[:max_per_category * 2],
324
+ "note": (
325
+ "Architecture is inferred from source declarations, routes, and import patterns. "
326
+ "Framework endpoints are verified from AST route registrations."
327
+ ),
328
+ }
codegraph/audit.py ADDED
@@ -0,0 +1,106 @@
1
+ """Optional local audit log for MCP tool operations.
2
+
3
+ Records: timestamp, tool, operation, repository, files accessed, duration.
4
+ File contents are NEVER recorded.
5
+ Log can be cleared via CLI. Disabled by default.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ import json
10
+ import sqlite3
11
+ import time
12
+ from collections.abc import Iterator
13
+ from contextlib import contextmanager
14
+ from pathlib import Path
15
+
16
+
17
+ class AuditLog:
18
+ """SQLite-backed local audit log. Contents never include source code."""
19
+
20
+ def __init__(self, db_path: Path, enabled: bool = False) -> None:
21
+ self.db_path = db_path
22
+ self.enabled = enabled
23
+ if enabled:
24
+ self._ensure_table()
25
+
26
+ def _connect(self) -> sqlite3.Connection:
27
+ con = sqlite3.connect(self.db_path)
28
+ con.row_factory = sqlite3.Row
29
+ return con
30
+
31
+ def _ensure_table(self) -> None:
32
+ with self._session() as con:
33
+ con.execute("""
34
+ CREATE TABLE IF NOT EXISTS audit_log (
35
+ id INTEGER PRIMARY KEY,
36
+ ts INTEGER NOT NULL,
37
+ tool TEXT NOT NULL,
38
+ operation TEXT NOT NULL,
39
+ repository TEXT NOT NULL,
40
+ files_accessed TEXT NOT NULL DEFAULT '[]',
41
+ duration_ms REAL NOT NULL DEFAULT 0
42
+ )
43
+ """)
44
+
45
+ @contextmanager
46
+ def _session(self) -> Iterator[sqlite3.Connection]:
47
+ con = self._connect()
48
+ try:
49
+ yield con
50
+ con.commit()
51
+ except BaseException:
52
+ con.rollback()
53
+ raise
54
+ finally:
55
+ con.close()
56
+
57
+ def record(
58
+ self,
59
+ tool: str,
60
+ operation: str,
61
+ repository: str,
62
+ files_accessed: list[str] | None = None,
63
+ duration_ms: float = 0.0,
64
+ ) -> None:
65
+ if not self.enabled:
66
+ return
67
+ with self._session() as con:
68
+ con.execute(
69
+ "INSERT INTO audit_log(ts, tool, operation, repository, files_accessed, duration_ms) "
70
+ "VALUES (?, ?, ?, ?, ?, ?)",
71
+ (
72
+ int(time.time()),
73
+ tool,
74
+ operation,
75
+ repository,
76
+ json.dumps(files_accessed or []),
77
+ duration_ms,
78
+ ),
79
+ )
80
+
81
+ def recent(self, limit: int = 100) -> list[dict[str, object]]:
82
+ if not self.enabled:
83
+ return []
84
+ with self._session() as con:
85
+ rows = con.execute(
86
+ "SELECT ts, tool, operation, repository, files_accessed, duration_ms "
87
+ "FROM audit_log ORDER BY id DESC LIMIT ?",
88
+ (limit,),
89
+ ).fetchall()
90
+ return [
91
+ {
92
+ "ts": r["ts"],
93
+ "tool": r["tool"],
94
+ "operation": r["operation"],
95
+ "repository": r["repository"],
96
+ "files_accessed": json.loads(r["files_accessed"]),
97
+ "duration_ms": r["duration_ms"],
98
+ }
99
+ for r in rows
100
+ ]
101
+
102
+ def clear(self) -> None:
103
+ if not self.enabled:
104
+ return
105
+ with self._session() as con:
106
+ con.execute("DELETE FROM audit_log")
codegraph/cache.py ADDED
@@ -0,0 +1,95 @@
1
+ """Deterministic Task Fingerprinting and Context Caching Engine.
2
+
3
+ Guarantees repository generation consistency and cache invalidation on:
4
+ - Source code or index generation change
5
+ - Parser version change
6
+ - Database schema change
7
+ - Configuration change
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import hashlib
12
+ import json
13
+ import sqlite3
14
+ import time
15
+ from typing import Any, cast
16
+
17
+ from codegraph.indexing.parser import PARSER_VERSION
18
+ from codegraph.task import TaskSpec
19
+
20
+ CACHE_SCHEMA_VERSION = 1
21
+
22
+
23
+ def compute_context_cache_key(
24
+ task_spec: TaskSpec,
25
+ repository_generation: int,
26
+ max_tokens: int,
27
+ parser_version: str = PARSER_VERSION,
28
+ extra_config: str = "",
29
+ ) -> str:
30
+ """Compute an immutable, deterministic cache key for a context compilation request."""
31
+ canonical_components = {
32
+ "intent": task_spec.intent,
33
+ "goal": task_spec.goal,
34
+ "targets": sorted(task_spec.targets),
35
+ "operations": sorted(task_spec.operations),
36
+ "constraints": sorted(task_spec.constraints),
37
+ "exclusions": sorted(task_spec.exclusions),
38
+ "scope_paths": sorted(task_spec.scope_paths),
39
+ "time_scope": task_spec.time_scope or "",
40
+ "repository_generation": repository_generation,
41
+ "max_tokens": max_tokens,
42
+ "parser_version": parser_version,
43
+ "cache_schema": CACHE_SCHEMA_VERSION,
44
+ "extra_config": extra_config,
45
+ }
46
+ raw = json.dumps(canonical_components, sort_keys=True, separators=(",", ":")).encode("utf-8")
47
+ return f"ctx:{hashlib.sha256(raw).hexdigest()}"
48
+
49
+
50
+ def get_cached_context_packet(
51
+ con: sqlite3.Connection,
52
+ cache_key: str,
53
+ ) -> dict[str, Any] | None:
54
+ """Retrieve cached context packet if present in SQLite context_cache."""
55
+ try:
56
+ row = con.execute(
57
+ "SELECT data FROM context_cache WHERE cache_key=?",
58
+ (cache_key,),
59
+ ).fetchone()
60
+ if row:
61
+ raw_data = row["data"] if isinstance(row, sqlite3.Row) else row[0]
62
+ parsed = json.loads(raw_data)
63
+ if isinstance(parsed, dict):
64
+ return cast(dict[str, Any], parsed)
65
+ return None
66
+ except Exception:
67
+ pass
68
+ return None
69
+
70
+
71
+ def store_cached_context_packet(
72
+ con: sqlite3.Connection,
73
+ cache_key: str,
74
+ packet_dict: dict[str, Any],
75
+ ) -> None:
76
+ """Store compiled ContextPacket in SQLite context_cache table."""
77
+ try:
78
+ now = int(time.time())
79
+ serialized = json.dumps(packet_dict, sort_keys=True)
80
+ con.execute(
81
+ "INSERT OR REPLACE INTO context_cache (cache_key, data, created_at) VALUES (?, ?, ?)",
82
+ (cache_key, serialized, now),
83
+ )
84
+ con.commit()
85
+ except Exception:
86
+ pass
87
+
88
+
89
+ def invalidate_context_cache(con: sqlite3.Connection) -> None:
90
+ """Explicitly wipe context cache upon repository mutation or index generation advancement."""
91
+ try:
92
+ con.execute("DELETE FROM context_cache")
93
+ con.commit()
94
+ except Exception:
95
+ pass