codegraph-engine 2.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codegraph/__init__.py +37 -0
- codegraph/agent.py +26 -0
- codegraph/architecture.py +328 -0
- codegraph/audit.py +106 -0
- codegraph/cache.py +95 -0
- codegraph/cli.py +854 -0
- codegraph/config.py +43 -0
- codegraph/constraints.py +238 -0
- codegraph/context.py +1228 -0
- codegraph/epistemic.py +90 -0
- codegraph/errors.py +275 -0
- codegraph/evidence/__init__.py +15 -0
- codegraph/evidence/citations.py +397 -0
- codegraph/frameworks.py +434 -0
- codegraph/freshness.py +295 -0
- codegraph/git.py +278 -0
- codegraph/graph/__init__.py +46 -0
- codegraph/graph/models.py +41 -0
- codegraph/graph/traversal.py +1291 -0
- codegraph/indexing/__init__.py +4 -0
- codegraph/indexing/classifier.py +274 -0
- codegraph/indexing/indexer.py +943 -0
- codegraph/indexing/models.py +338 -0
- codegraph/indexing/parser.py +1240 -0
- codegraph/indexing/scanner.py +200 -0
- codegraph/indexing/test_framework.py +116 -0
- codegraph/interrogation.py +1582 -0
- codegraph/llm/__init__.py +3 -0
- codegraph/llm/base.py +15 -0
- codegraph/llm/context.py +20 -0
- codegraph/mcp/__init__.py +3 -0
- codegraph/mcp/server.py +736 -0
- codegraph/memory/__init__.py +3 -0
- codegraph/memory/store.py +46 -0
- codegraph/models.py +289 -0
- codegraph/observability.py +151 -0
- codegraph/optimizer.py +372 -0
- codegraph/planner.py +417 -0
- codegraph/py.typed +1 -0
- codegraph/query_expansion.py +199 -0
- codegraph/ranking.py +363 -0
- codegraph/resolver.py +843 -0
- codegraph/resources/__init__.py +45 -0
- codegraph/resources/cache.py +117 -0
- codegraph/resources/coalescer.py +83 -0
- codegraph/resources/debouncer.py +98 -0
- codegraph/resources/governor.py +232 -0
- codegraph/resources/policy.py +123 -0
- codegraph/retrieval_policy.py +220 -0
- codegraph/search/__init__.py +23 -0
- codegraph/search/hybrid.py +301 -0
- codegraph/search/semantic.py +28 -0
- codegraph/security/__init__.py +3 -0
- codegraph/security/paths.py +35 -0
- codegraph/target_resolver.py +348 -0
- codegraph/task.py +637 -0
- codegraph_engine-2.1.1.dist-info/METADATA +334 -0
- codegraph_engine-2.1.1.dist-info/RECORD +62 -0
- codegraph_engine-2.1.1.dist-info/WHEEL +5 -0
- codegraph_engine-2.1.1.dist-info/entry_points.txt +2 -0
- codegraph_engine-2.1.1.dist-info/licenses/LICENSE +21 -0
- codegraph_engine-2.1.1.dist-info/top_level.txt +1 -0
codegraph/freshness.py
ADDED
|
@@ -0,0 +1,295 @@
|
|
|
1
|
+
"""Repository Freshness & Dependency-Aware Invalidation Engine.
|
|
2
|
+
|
|
3
|
+
Tracks:
|
|
4
|
+
current Git HEAD, indexed Git HEAD, file hashes, parser version,
|
|
5
|
+
schema version, index generation, timestamps, and parse_failed state.
|
|
6
|
+
|
|
7
|
+
Provides dependency-aware invalidation so localized file updates only
|
|
8
|
+
invalidate affected symbols, references, graph edges, evidence, and cached
|
|
9
|
+
context packets rather than rebuilding the entire repository.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import hashlib
|
|
14
|
+
import shlex
|
|
15
|
+
import sqlite3
|
|
16
|
+
import subprocess
|
|
17
|
+
from dataclasses import dataclass, field
|
|
18
|
+
from enum import StrEnum
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class FreshnessStatus(StrEnum):
|
|
23
|
+
FRESH = "FRESH"
|
|
24
|
+
STALE = "STALE"
|
|
25
|
+
PARTIALLY_STALE = "PARTIALLY_STALE"
|
|
26
|
+
UNKNOWN = "UNKNOWN"
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass
|
|
30
|
+
class FreshnessReport:
|
|
31
|
+
status: FreshnessStatus
|
|
32
|
+
indexed_commit: str | None
|
|
33
|
+
current_commit: str | None
|
|
34
|
+
index_timestamp: int | None
|
|
35
|
+
modified_files: list[str] = field(default_factory=list)
|
|
36
|
+
deleted_files: list[str] = field(default_factory=list)
|
|
37
|
+
parse_failed_files: list[str] = field(default_factory=list)
|
|
38
|
+
parser_version: str | None = None
|
|
39
|
+
index_generation: int = 0
|
|
40
|
+
detail: str = ""
|
|
41
|
+
|
|
42
|
+
def as_dict(self) -> dict[str, object]:
|
|
43
|
+
return {
|
|
44
|
+
"status": self.status.value,
|
|
45
|
+
"indexed_commit": self.indexed_commit,
|
|
46
|
+
"current_commit": self.current_commit,
|
|
47
|
+
"index_timestamp": self.index_timestamp,
|
|
48
|
+
"modified_files": self.modified_files,
|
|
49
|
+
"deleted_files": self.deleted_files,
|
|
50
|
+
"parse_failed_files": self.parse_failed_files,
|
|
51
|
+
"parser_version": self.parser_version,
|
|
52
|
+
"index_generation": self.index_generation,
|
|
53
|
+
"detail": self.detail,
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _run_git(repository: Path, *args: str, timeout: int = 10) -> str | None:
|
|
58
|
+
"""Run an allowlisted read-only git command inside repository."""
|
|
59
|
+
if not shlex.split("git")[0]: # pragma: no cover
|
|
60
|
+
return None
|
|
61
|
+
try:
|
|
62
|
+
result = subprocess.run(
|
|
63
|
+
["git", *args],
|
|
64
|
+
cwd=str(repository),
|
|
65
|
+
capture_output=True,
|
|
66
|
+
text=True,
|
|
67
|
+
timeout=timeout,
|
|
68
|
+
)
|
|
69
|
+
return result.stdout.strip() if result.returncode == 0 else None
|
|
70
|
+
except (FileNotFoundError, subprocess.TimeoutExpired, OSError):
|
|
71
|
+
return None
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def current_commit(repository: Path) -> str | None:
|
|
75
|
+
"""Return the current HEAD commit SHA, or None if not a git repository."""
|
|
76
|
+
return _run_git(repository, "rev-parse", "HEAD")
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def current_branch(repository: Path) -> str | None:
|
|
80
|
+
"""Return the current git branch name, or None."""
|
|
81
|
+
return _run_git(repository, "rev-parse", "--abbrev-ref", "HEAD")
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _get_meta(con: sqlite3.Connection, key: str) -> str | None:
|
|
85
|
+
for table in ("metadata", "codegraph_meta"):
|
|
86
|
+
try:
|
|
87
|
+
row = con.execute(
|
|
88
|
+
f"SELECT value FROM {table} WHERE key=?", (key,)
|
|
89
|
+
).fetchone()
|
|
90
|
+
if row and row[0] != "":
|
|
91
|
+
return str(row[0])
|
|
92
|
+
except sqlite3.OperationalError:
|
|
93
|
+
continue
|
|
94
|
+
return None
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def indexed_commit(con: sqlite3.Connection) -> str | None:
|
|
98
|
+
"""Read the commit SHA recorded when the index was last built."""
|
|
99
|
+
return _get_meta(con, "indexed_commit")
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def index_timestamp(con: sqlite3.Connection) -> int | None:
|
|
103
|
+
"""Return unix timestamp of the last indexing run."""
|
|
104
|
+
val = _get_meta(con, "index_timestamp")
|
|
105
|
+
try:
|
|
106
|
+
return int(val) if val is not None else None
|
|
107
|
+
except ValueError:
|
|
108
|
+
return None
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def index_generation(con: sqlite3.Connection) -> int:
|
|
112
|
+
"""Return the monotonic index generation counter."""
|
|
113
|
+
val = _get_meta(con, "index_generation")
|
|
114
|
+
try:
|
|
115
|
+
return int(val) if val is not None else 0
|
|
116
|
+
except ValueError:
|
|
117
|
+
return 0
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def save_commit(con: sqlite3.Connection, commit: str | None, timestamp: int) -> None:
|
|
121
|
+
"""Persist commit SHA and timestamp into metadata tables."""
|
|
122
|
+
_ensure_meta_table(con)
|
|
123
|
+
for table in ("codegraph_meta", "metadata"):
|
|
124
|
+
for key, val in [
|
|
125
|
+
("indexed_commit", commit or ""),
|
|
126
|
+
("index_timestamp", str(timestamp)),
|
|
127
|
+
]:
|
|
128
|
+
try:
|
|
129
|
+
con.execute(
|
|
130
|
+
f"INSERT INTO {table}(key, value) VALUES(?,?) "
|
|
131
|
+
"ON CONFLICT(key) DO UPDATE SET value=excluded.value",
|
|
132
|
+
(key, val),
|
|
133
|
+
)
|
|
134
|
+
except sqlite3.OperationalError:
|
|
135
|
+
pass
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def _ensure_meta_table(con: sqlite3.Connection) -> None:
|
|
139
|
+
con.execute(
|
|
140
|
+
"CREATE TABLE IF NOT EXISTS codegraph_meta "
|
|
141
|
+
"(key TEXT PRIMARY KEY, value TEXT NOT NULL)"
|
|
142
|
+
)
|
|
143
|
+
con.execute(
|
|
144
|
+
"CREATE TABLE IF NOT EXISTS metadata "
|
|
145
|
+
"(key TEXT PRIMARY KEY, value TEXT NOT NULL)"
|
|
146
|
+
)
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def compute_invalidated_dependents(
|
|
150
|
+
con: sqlite3.Connection, changed_paths: set[str]
|
|
151
|
+
) -> dict[str, list[str]]:
|
|
152
|
+
"""Given a set of modified/deleted file paths, compute dependent files and symbols that must be invalidated."""
|
|
153
|
+
from codegraph.indexing.models import normalize_module
|
|
154
|
+
|
|
155
|
+
if not changed_paths:
|
|
156
|
+
return {
|
|
157
|
+
"files": [],
|
|
158
|
+
"modules": [],
|
|
159
|
+
"symbols": [],
|
|
160
|
+
"dependent_files": [],
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
changed_modules = {normalize_module(p) for p in changed_paths}
|
|
164
|
+
invalidated_symbols: set[str] = set()
|
|
165
|
+
dependent_files: set[str] = set()
|
|
166
|
+
|
|
167
|
+
for p in changed_paths:
|
|
168
|
+
try:
|
|
169
|
+
rows = con.execute(
|
|
170
|
+
"SELECT canonical_id, qualified_name FROM symbols WHERE path=?", (p,)
|
|
171
|
+
).fetchall()
|
|
172
|
+
for r in rows:
|
|
173
|
+
invalidated_symbols.add(str(r["canonical_id"] or r["qualified_name"]))
|
|
174
|
+
except sqlite3.OperationalError:
|
|
175
|
+
pass
|
|
176
|
+
|
|
177
|
+
for mod in changed_modules:
|
|
178
|
+
try:
|
|
179
|
+
imp_rows = con.execute(
|
|
180
|
+
"SELECT DISTINCT source_path FROM imports "
|
|
181
|
+
"WHERE resolved_module=? OR imported_module=? OR module=? OR module LIKE ?",
|
|
182
|
+
(mod, mod, mod, f"%{mod.split('.')[-1]}"),
|
|
183
|
+
).fetchall()
|
|
184
|
+
for r in imp_rows:
|
|
185
|
+
if r["source_path"] not in changed_paths:
|
|
186
|
+
dependent_files.add(str(r["source_path"]))
|
|
187
|
+
except sqlite3.OperationalError:
|
|
188
|
+
pass
|
|
189
|
+
|
|
190
|
+
return {
|
|
191
|
+
"files": sorted(changed_paths),
|
|
192
|
+
"modules": sorted(changed_modules),
|
|
193
|
+
"symbols": sorted(invalidated_symbols),
|
|
194
|
+
"dependent_files": sorted(dependent_files),
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def check_freshness(
|
|
199
|
+
repository: Path, con: sqlite3.Connection
|
|
200
|
+
) -> FreshnessReport:
|
|
201
|
+
"""Compare index state against current filesystem state."""
|
|
202
|
+
_ensure_meta_table(con)
|
|
203
|
+
|
|
204
|
+
idx_commit = indexed_commit(con)
|
|
205
|
+
cur_commit = current_commit(repository)
|
|
206
|
+
idx_ts = index_timestamp(con)
|
|
207
|
+
parser_ver = _get_meta(con, "parser_version")
|
|
208
|
+
gen = index_generation(con)
|
|
209
|
+
|
|
210
|
+
try:
|
|
211
|
+
file_count = con.execute("SELECT count(*) FROM files").fetchone()[0]
|
|
212
|
+
except sqlite3.OperationalError:
|
|
213
|
+
return FreshnessReport(
|
|
214
|
+
status=FreshnessStatus.UNKNOWN,
|
|
215
|
+
indexed_commit=idx_commit,
|
|
216
|
+
current_commit=cur_commit,
|
|
217
|
+
index_timestamp=idx_ts,
|
|
218
|
+
parser_version=parser_ver,
|
|
219
|
+
index_generation=gen,
|
|
220
|
+
detail="No index found. Run `codegraph index` first.",
|
|
221
|
+
)
|
|
222
|
+
|
|
223
|
+
if file_count == 0:
|
|
224
|
+
return FreshnessReport(
|
|
225
|
+
status=FreshnessStatus.UNKNOWN,
|
|
226
|
+
indexed_commit=idx_commit,
|
|
227
|
+
current_commit=cur_commit,
|
|
228
|
+
index_timestamp=idx_ts,
|
|
229
|
+
parser_version=parser_ver,
|
|
230
|
+
index_generation=gen,
|
|
231
|
+
detail="Index is empty. Run `codegraph index` first.",
|
|
232
|
+
)
|
|
233
|
+
|
|
234
|
+
modified: list[str] = []
|
|
235
|
+
deleted: list[str] = []
|
|
236
|
+
parse_failed: list[str] = []
|
|
237
|
+
|
|
238
|
+
try:
|
|
239
|
+
rows = con.execute("SELECT path, hash, status FROM files").fetchall()
|
|
240
|
+
except sqlite3.OperationalError:
|
|
241
|
+
rows = con.execute("SELECT path, hash, 'ok' AS status FROM files").fetchall()
|
|
242
|
+
|
|
243
|
+
for row in rows:
|
|
244
|
+
rel = str(row["path"])
|
|
245
|
+
if row["status"] == "parse_failed":
|
|
246
|
+
parse_failed.append(rel)
|
|
247
|
+
if rel not in modified:
|
|
248
|
+
modified.append(rel)
|
|
249
|
+
abs_path = repository / rel
|
|
250
|
+
if not abs_path.exists():
|
|
251
|
+
deleted.append(rel)
|
|
252
|
+
else:
|
|
253
|
+
try:
|
|
254
|
+
digest = hashlib.sha256(
|
|
255
|
+
abs_path.read_text(encoding="utf-8", errors="replace").encode()
|
|
256
|
+
).hexdigest()
|
|
257
|
+
if digest != row["hash"] and rel not in modified:
|
|
258
|
+
modified.append(rel)
|
|
259
|
+
except OSError:
|
|
260
|
+
if rel not in modified:
|
|
261
|
+
modified.append(rel)
|
|
262
|
+
|
|
263
|
+
changed = len(modified) + len(deleted)
|
|
264
|
+
if changed == 0 and not parse_failed and (idx_commit == cur_commit or cur_commit is None):
|
|
265
|
+
status = FreshnessStatus.FRESH
|
|
266
|
+
detail = "Index matches current repository state."
|
|
267
|
+
elif parse_failed and changed == len(parse_failed):
|
|
268
|
+
status = FreshnessStatus.PARTIALLY_STALE
|
|
269
|
+
detail = f"{len(parse_failed)} file(s) had syntax errors; last-known-good index retained (stale)."
|
|
270
|
+
elif changed == 0:
|
|
271
|
+
status = FreshnessStatus.PARTIALLY_STALE
|
|
272
|
+
detail = f"Commit changed ({_short(idx_commit)} → {_short(cur_commit)}) but no file hashes differ."
|
|
273
|
+
elif changed < file_count:
|
|
274
|
+
status = FreshnessStatus.PARTIALLY_STALE
|
|
275
|
+
detail = f"{changed} file(s) changed since indexing."
|
|
276
|
+
else:
|
|
277
|
+
status = FreshnessStatus.STALE
|
|
278
|
+
detail = f"All {file_count} indexed files may have changed. Re-index recommended."
|
|
279
|
+
|
|
280
|
+
return FreshnessReport(
|
|
281
|
+
status=status,
|
|
282
|
+
indexed_commit=idx_commit,
|
|
283
|
+
current_commit=cur_commit,
|
|
284
|
+
index_timestamp=idx_ts,
|
|
285
|
+
modified_files=modified,
|
|
286
|
+
deleted_files=deleted,
|
|
287
|
+
parse_failed_files=parse_failed,
|
|
288
|
+
parser_version=parser_ver,
|
|
289
|
+
index_generation=gen,
|
|
290
|
+
detail=detail,
|
|
291
|
+
)
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
def _short(sha: str | None) -> str:
|
|
295
|
+
return sha[:8] if sha else "unknown"
|
codegraph/git.py
ADDED
|
@@ -0,0 +1,278 @@
|
|
|
1
|
+
"""Read-only Git intelligence.
|
|
2
|
+
|
|
3
|
+
All Git operations use an explicit allowlist of commands and validated
|
|
4
|
+
arguments. No arbitrary shell execution. No network access.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import sqlite3
|
|
9
|
+
import subprocess
|
|
10
|
+
from dataclasses import dataclass
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass
|
|
15
|
+
class CommitInfo:
|
|
16
|
+
sha: str
|
|
17
|
+
author: str
|
|
18
|
+
date: str
|
|
19
|
+
message: str
|
|
20
|
+
|
|
21
|
+
def as_dict(self) -> dict[str, str]:
|
|
22
|
+
return {"sha": self.sha, "author": self.author, "date": self.date, "message": self.message}
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass
|
|
26
|
+
class FileDiff:
|
|
27
|
+
path: str
|
|
28
|
+
status: str # M=modified, A=added, D=deleted, R=renamed
|
|
29
|
+
diff_snippet: str = ""
|
|
30
|
+
|
|
31
|
+
def as_dict(self) -> dict[str, str]:
|
|
32
|
+
return {"path": self.path, "status": self.status, "diff_snippet": self.diff_snippet}
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _git(repository: Path, *args: str, timeout: int = 15) -> str | None:
|
|
36
|
+
"""Run a git subprocess with validated arguments only."""
|
|
37
|
+
# Validate: no shell metacharacters, no absolute paths from user input
|
|
38
|
+
for arg in args:
|
|
39
|
+
if any(c in arg for c in (";", "|", "&", "`", "$", "\n", "\r")):
|
|
40
|
+
return None
|
|
41
|
+
try:
|
|
42
|
+
result = subprocess.run(
|
|
43
|
+
["git", *args],
|
|
44
|
+
cwd=str(repository),
|
|
45
|
+
capture_output=True,
|
|
46
|
+
text=True,
|
|
47
|
+
timeout=timeout,
|
|
48
|
+
)
|
|
49
|
+
return result.stdout if result.returncode == 0 else None
|
|
50
|
+
except (FileNotFoundError, subprocess.TimeoutExpired, OSError):
|
|
51
|
+
return None
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def is_git_repository(repository: Path) -> bool:
|
|
55
|
+
return _git(repository, "rev-parse", "--git-dir") is not None
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def current_commit(repository: Path) -> str | None:
|
|
59
|
+
"""Return the current HEAD commit SHA, or None if not a git repository."""
|
|
60
|
+
raw = _git(repository, "rev-parse", "HEAD")
|
|
61
|
+
return raw.strip() if raw else None
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def recent_commits(
|
|
65
|
+
repository: Path,
|
|
66
|
+
n: int = 10,
|
|
67
|
+
path: str | None = None,
|
|
68
|
+
) -> list[CommitInfo]:
|
|
69
|
+
"""Return the last N commits, optionally scoped to a file path."""
|
|
70
|
+
if not 1 <= n <= 100:
|
|
71
|
+
n = 10
|
|
72
|
+
args = ["log", f"-{n}", "--format=%H\x1f%an\x1f%ai\x1f%s"]
|
|
73
|
+
if path:
|
|
74
|
+
# Sanitize: path must not be absolute or contain traversal
|
|
75
|
+
if ".." in path or path.startswith("/"):
|
|
76
|
+
return []
|
|
77
|
+
args += ["--", path]
|
|
78
|
+
raw = _git(repository, *args)
|
|
79
|
+
if not raw:
|
|
80
|
+
return []
|
|
81
|
+
commits = []
|
|
82
|
+
for line in raw.strip().splitlines():
|
|
83
|
+
parts = line.split("\x1f", 3)
|
|
84
|
+
if len(parts) == 4:
|
|
85
|
+
commits.append(CommitInfo(*parts))
|
|
86
|
+
return commits
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def changed_files(
|
|
90
|
+
repository: Path,
|
|
91
|
+
since: str = "HEAD~10",
|
|
92
|
+
until: str = "HEAD",
|
|
93
|
+
) -> list[FileDiff]:
|
|
94
|
+
"""Return files changed between two commits."""
|
|
95
|
+
# Validate refs — allow only hex SHAs, HEAD, HEAD~N, branch names (alphanumeric + /-_.)
|
|
96
|
+
import re
|
|
97
|
+
_REF = re.compile(r"^[A-Za-z0-9_.~^/-]{1,100}$")
|
|
98
|
+
if not _REF.match(since) or not _REF.match(until):
|
|
99
|
+
return []
|
|
100
|
+
raw = _git(repository, "diff", "--name-status", since, until)
|
|
101
|
+
if not raw:
|
|
102
|
+
return []
|
|
103
|
+
diffs: list[FileDiff] = []
|
|
104
|
+
for line in raw.strip().splitlines():
|
|
105
|
+
parts = line.split("\t", 1)
|
|
106
|
+
if len(parts) == 2:
|
|
107
|
+
status, path = parts[0].strip()[0], parts[1].strip()
|
|
108
|
+
diffs.append(FileDiff(path=path, status=status))
|
|
109
|
+
return diffs
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def file_diff(
|
|
113
|
+
repository: Path,
|
|
114
|
+
path: str,
|
|
115
|
+
since: str = "HEAD~1",
|
|
116
|
+
until: str = "HEAD",
|
|
117
|
+
max_lines: int = 50,
|
|
118
|
+
) -> str:
|
|
119
|
+
"""Return a bounded unified diff for a single file."""
|
|
120
|
+
import re
|
|
121
|
+
_REF = re.compile(r"^[A-Za-z0-9_.~^/-]{1,100}$")
|
|
122
|
+
if not _REF.match(since) or not _REF.match(until):
|
|
123
|
+
return ""
|
|
124
|
+
if ".." in path or path.startswith("/"):
|
|
125
|
+
return ""
|
|
126
|
+
raw = _git(repository, "diff", since, until, "--", path)
|
|
127
|
+
if not raw:
|
|
128
|
+
return ""
|
|
129
|
+
lines = raw.splitlines()[:max_lines]
|
|
130
|
+
return "\n".join(lines)
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def blame_line(
|
|
134
|
+
repository: Path,
|
|
135
|
+
path: str,
|
|
136
|
+
line: int,
|
|
137
|
+
) -> dict[str, str] | None:
|
|
138
|
+
"""Return blame info for a single line."""
|
|
139
|
+
if ".." in path or path.startswith("/"):
|
|
140
|
+
return None
|
|
141
|
+
if not 1 <= line <= 100_000:
|
|
142
|
+
return None
|
|
143
|
+
raw = _git(
|
|
144
|
+
repository, "blame", f"-L{line},{line}", "--porcelain", "--", path
|
|
145
|
+
)
|
|
146
|
+
if not raw:
|
|
147
|
+
return None
|
|
148
|
+
info: dict[str, str] = {}
|
|
149
|
+
lines = raw.splitlines()
|
|
150
|
+
if lines:
|
|
151
|
+
info["sha"] = lines[0].split()[0] if lines else ""
|
|
152
|
+
for bl in lines[1:]:
|
|
153
|
+
if bl.startswith("author "):
|
|
154
|
+
info["author"] = bl[7:]
|
|
155
|
+
elif bl.startswith("author-time "):
|
|
156
|
+
info["timestamp"] = bl[12:]
|
|
157
|
+
elif bl.startswith("summary "):
|
|
158
|
+
info["summary"] = bl[8:]
|
|
159
|
+
return info or None
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def changed_symbols_since(
|
|
163
|
+
repository: Path,
|
|
164
|
+
since: str,
|
|
165
|
+
symbol_names: set[str],
|
|
166
|
+
) -> list[str]:
|
|
167
|
+
"""Return subset of symbol_names that appear in changed lines since `since`.
|
|
168
|
+
|
|
169
|
+
This is a heuristic: it checks whether the symbol name appears in the
|
|
170
|
+
diff text. Results are LOW confidence.
|
|
171
|
+
"""
|
|
172
|
+
import re
|
|
173
|
+
_REF = re.compile(r"^[A-Za-z0-9_.~^/-]{1,100}$")
|
|
174
|
+
if not _REF.match(since):
|
|
175
|
+
return []
|
|
176
|
+
raw = _git(repository, "diff", since, "HEAD")
|
|
177
|
+
if not raw:
|
|
178
|
+
return []
|
|
179
|
+
found = []
|
|
180
|
+
for name in symbol_names:
|
|
181
|
+
# Only match if the name appears in added/removed diff lines
|
|
182
|
+
if re.search(rf"^[+-].*\b{re.escape(name)}\b", raw, re.MULTILINE):
|
|
183
|
+
found.append(name)
|
|
184
|
+
return found
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def get_file_history(
|
|
188
|
+
repository: Path,
|
|
189
|
+
path: str,
|
|
190
|
+
n: int = 10,
|
|
191
|
+
) -> list[dict[str, str]]:
|
|
192
|
+
"""Return commit history for a single file path."""
|
|
193
|
+
commits = recent_commits(repository, n=n, path=path)
|
|
194
|
+
return [c.as_dict() for c in commits]
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def get_symbol_history(
|
|
198
|
+
repository: Path,
|
|
199
|
+
path: str,
|
|
200
|
+
symbol: str,
|
|
201
|
+
n: int = 10,
|
|
202
|
+
) -> list[dict[str, str]]:
|
|
203
|
+
"""Return commits where a specific symbol name was modified in a file."""
|
|
204
|
+
all_file_commits = recent_commits(repository, n=n * 2, path=path)
|
|
205
|
+
matching: list[dict[str, str]] = []
|
|
206
|
+
import re
|
|
207
|
+
for c in all_file_commits:
|
|
208
|
+
diff_text = file_diff(repository, path, since=f"{c.sha}~1", until=c.sha, max_lines=200)
|
|
209
|
+
if re.search(rf"\b{re.escape(symbol)}\b", diff_text):
|
|
210
|
+
matching.append(c.as_dict())
|
|
211
|
+
if len(matching) >= n:
|
|
212
|
+
break
|
|
213
|
+
return matching
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def get_change_context(
|
|
217
|
+
repository: Path,
|
|
218
|
+
since: str = "HEAD~1",
|
|
219
|
+
until: str = "HEAD",
|
|
220
|
+
) -> dict[str, object]:
|
|
221
|
+
"""Return summary of changed files and bounded snippets."""
|
|
222
|
+
diffs = changed_files(repository, since=since, until=until)
|
|
223
|
+
file_summaries: list[dict[str, str]] = []
|
|
224
|
+
for d in diffs[:20]:
|
|
225
|
+
snippet = file_diff(repository, d.path, since=since, until=until, max_lines=30)
|
|
226
|
+
file_summaries.append(
|
|
227
|
+
{
|
|
228
|
+
"path": d.path,
|
|
229
|
+
"status": d.status,
|
|
230
|
+
"diff_snippet": snippet,
|
|
231
|
+
}
|
|
232
|
+
)
|
|
233
|
+
return {
|
|
234
|
+
"since": since,
|
|
235
|
+
"until": until,
|
|
236
|
+
"total_changed": len(diffs),
|
|
237
|
+
"files": file_summaries,
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def analyze_change_impact(
|
|
242
|
+
con: sqlite3.Connection,
|
|
243
|
+
repository: Path,
|
|
244
|
+
since: str = "HEAD~1",
|
|
245
|
+
until: str = "HEAD",
|
|
246
|
+
) -> dict[str, object]:
|
|
247
|
+
"""Analyze downstream callers and tests affected by changes between since and until."""
|
|
248
|
+
diffs = changed_files(repository, since=since, until=until)
|
|
249
|
+
changed_paths = [d.path for d in diffs]
|
|
250
|
+
changed_symbols: list[str] = []
|
|
251
|
+
affected_callers: list[dict[str, object]] = []
|
|
252
|
+
affected_tests: list[dict[str, object]] = []
|
|
253
|
+
|
|
254
|
+
from codegraph.graph.traversal import find_callers, find_related_tests
|
|
255
|
+
|
|
256
|
+
for p in changed_paths:
|
|
257
|
+
rows = con.execute("SELECT name, qualified_name FROM symbols WHERE path=?", (p,)).fetchall()
|
|
258
|
+
sym_names = {r["name"] for r in rows}
|
|
259
|
+
matched = changed_symbols_since(repository, since, sym_names)
|
|
260
|
+
for s in matched:
|
|
261
|
+
changed_symbols.append(s)
|
|
262
|
+
callers = find_callers(con, s, max_results=5)
|
|
263
|
+
for c in callers:
|
|
264
|
+
affected_callers.append(dict(c))
|
|
265
|
+
tests = find_related_tests(con, s, max_results=5)
|
|
266
|
+
for t in tests:
|
|
267
|
+
if "result" not in t:
|
|
268
|
+
affected_tests.append(dict(t))
|
|
269
|
+
|
|
270
|
+
return {
|
|
271
|
+
"since": since,
|
|
272
|
+
"until": until,
|
|
273
|
+
"changed_files": [d.as_dict() for d in diffs],
|
|
274
|
+
"changed_symbols": changed_symbols,
|
|
275
|
+
"affected_callers": affected_callers,
|
|
276
|
+
"affected_tests": affected_tests,
|
|
277
|
+
}
|
|
278
|
+
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
from .models import GraphEdge, GraphSeedPolicy
|
|
2
|
+
from .traversal import (
|
|
3
|
+
analyze_impact,
|
|
4
|
+
definition_edges,
|
|
5
|
+
find_affected_tests,
|
|
6
|
+
find_callees,
|
|
7
|
+
find_callers,
|
|
8
|
+
find_implementations,
|
|
9
|
+
find_importers,
|
|
10
|
+
find_references,
|
|
11
|
+
find_related_tests,
|
|
12
|
+
find_symbol,
|
|
13
|
+
find_symbol_exact,
|
|
14
|
+
find_symbols,
|
|
15
|
+
get_call_graph,
|
|
16
|
+
get_callees,
|
|
17
|
+
get_callers,
|
|
18
|
+
get_dependency_graph,
|
|
19
|
+
get_symbol,
|
|
20
|
+
import_edges,
|
|
21
|
+
trace_call,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
__all__ = [
|
|
25
|
+
"GraphEdge",
|
|
26
|
+
"analyze_impact",
|
|
27
|
+
"definition_edges",
|
|
28
|
+
"find_affected_tests",
|
|
29
|
+
"find_callees",
|
|
30
|
+
"find_callers",
|
|
31
|
+
"find_implementations",
|
|
32
|
+
"find_importers",
|
|
33
|
+
"find_references",
|
|
34
|
+
"find_related_tests",
|
|
35
|
+
"find_symbol",
|
|
36
|
+
"find_symbol_exact",
|
|
37
|
+
"find_symbols",
|
|
38
|
+
"get_call_graph",
|
|
39
|
+
"get_callees",
|
|
40
|
+
"get_callers",
|
|
41
|
+
"get_dependency_graph",
|
|
42
|
+
"get_symbol",
|
|
43
|
+
"import_edges",
|
|
44
|
+
"trace_call",
|
|
45
|
+
"GraphSeedPolicy",
|
|
46
|
+
]
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import asdict, dataclass
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
@dataclass(frozen=True)
|
|
7
|
+
class GraphEdge:
|
|
8
|
+
source: str
|
|
9
|
+
target: str
|
|
10
|
+
relationship: str
|
|
11
|
+
confidence: str
|
|
12
|
+
file: str
|
|
13
|
+
start_line: int
|
|
14
|
+
end_line: int
|
|
15
|
+
evidence: str
|
|
16
|
+
|
|
17
|
+
def as_dict(self) -> dict[str, object]:
|
|
18
|
+
return asdict(self)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass(frozen=True)
|
|
22
|
+
class GraphSeedPolicy:
|
|
23
|
+
"""Policy for selecting grounded roots for graph traversal."""
|
|
24
|
+
|
|
25
|
+
source_type: str = "EXPLICIT_TARGET" # EXPLICIT_TARGET | ROUTE_HANDLER | DIRECT_MATCH | USER_REQUESTED
|
|
26
|
+
min_confidence: str = "HIGH" # HIGH | MEDIUM | LOW
|
|
27
|
+
max_seeds: int = 6
|
|
28
|
+
max_depth: int = 3
|
|
29
|
+
max_nodes_budget: int = 50
|
|
30
|
+
allowed_edges: frozenset[str] = frozenset({
|
|
31
|
+
"CALLS",
|
|
32
|
+
"CALLED_BY",
|
|
33
|
+
"HANDLED_BY",
|
|
34
|
+
"ROUTES_TO",
|
|
35
|
+
"DEPENDS_ON",
|
|
36
|
+
"IMPORTS",
|
|
37
|
+
"TESTS",
|
|
38
|
+
"DEFINES",
|
|
39
|
+
"CONTAINS",
|
|
40
|
+
})
|
|
41
|
+
|