codegraph-engine 2.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codegraph/__init__.py +37 -0
- codegraph/agent.py +26 -0
- codegraph/architecture.py +328 -0
- codegraph/audit.py +106 -0
- codegraph/cache.py +95 -0
- codegraph/cli.py +854 -0
- codegraph/config.py +43 -0
- codegraph/constraints.py +238 -0
- codegraph/context.py +1228 -0
- codegraph/epistemic.py +90 -0
- codegraph/errors.py +275 -0
- codegraph/evidence/__init__.py +15 -0
- codegraph/evidence/citations.py +397 -0
- codegraph/frameworks.py +434 -0
- codegraph/freshness.py +295 -0
- codegraph/git.py +278 -0
- codegraph/graph/__init__.py +46 -0
- codegraph/graph/models.py +41 -0
- codegraph/graph/traversal.py +1291 -0
- codegraph/indexing/__init__.py +4 -0
- codegraph/indexing/classifier.py +274 -0
- codegraph/indexing/indexer.py +943 -0
- codegraph/indexing/models.py +338 -0
- codegraph/indexing/parser.py +1240 -0
- codegraph/indexing/scanner.py +200 -0
- codegraph/indexing/test_framework.py +116 -0
- codegraph/interrogation.py +1582 -0
- codegraph/llm/__init__.py +3 -0
- codegraph/llm/base.py +15 -0
- codegraph/llm/context.py +20 -0
- codegraph/mcp/__init__.py +3 -0
- codegraph/mcp/server.py +736 -0
- codegraph/memory/__init__.py +3 -0
- codegraph/memory/store.py +46 -0
- codegraph/models.py +289 -0
- codegraph/observability.py +151 -0
- codegraph/optimizer.py +372 -0
- codegraph/planner.py +417 -0
- codegraph/py.typed +1 -0
- codegraph/query_expansion.py +199 -0
- codegraph/ranking.py +363 -0
- codegraph/resolver.py +843 -0
- codegraph/resources/__init__.py +45 -0
- codegraph/resources/cache.py +117 -0
- codegraph/resources/coalescer.py +83 -0
- codegraph/resources/debouncer.py +98 -0
- codegraph/resources/governor.py +232 -0
- codegraph/resources/policy.py +123 -0
- codegraph/retrieval_policy.py +220 -0
- codegraph/search/__init__.py +23 -0
- codegraph/search/hybrid.py +301 -0
- codegraph/search/semantic.py +28 -0
- codegraph/security/__init__.py +3 -0
- codegraph/security/paths.py +35 -0
- codegraph/target_resolver.py +348 -0
- codegraph/task.py +637 -0
- codegraph_engine-2.1.1.dist-info/METADATA +334 -0
- codegraph_engine-2.1.1.dist-info/RECORD +62 -0
- codegraph_engine-2.1.1.dist-info/WHEEL +5 -0
- codegraph_engine-2.1.1.dist-info/entry_points.txt +2 -0
- codegraph_engine-2.1.1.dist-info/licenses/LICENSE +21 -0
- codegraph_engine-2.1.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,220 @@
|
|
|
1
|
+
"""Intent-specific retrieval policies for CodeGraph MCP v2.1.
|
|
2
|
+
|
|
3
|
+
Each policy defines:
|
|
4
|
+
- Allowed relationship types (graph edges that may participate)
|
|
5
|
+
- Preferred coverage layers (ordered)
|
|
6
|
+
- Traversal flags (callers, callees, tests, git, framework, architecture)
|
|
7
|
+
- Maximum graph depth
|
|
8
|
+
|
|
9
|
+
Downstream retrieval MUST filter graph candidates to allowed_relationship_types.
|
|
10
|
+
This prevents unrelated relationships from polluting context merely because
|
|
11
|
+
they exist in the graph.
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from dataclasses import dataclass
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@dataclass(frozen=True)
|
|
19
|
+
class RetrievalPolicy:
|
|
20
|
+
intent: str
|
|
21
|
+
allowed_relationship_types: frozenset[str]
|
|
22
|
+
preferred_flow: tuple[str, ...]
|
|
23
|
+
include_callers: bool = True
|
|
24
|
+
include_callees: bool = True
|
|
25
|
+
include_tests: bool = True
|
|
26
|
+
include_git: bool = False
|
|
27
|
+
include_framework: bool = True
|
|
28
|
+
include_architecture: bool = False
|
|
29
|
+
max_graph_depth: int = 3
|
|
30
|
+
preferred_target_types: tuple[str, ...] = ()
|
|
31
|
+
|
|
32
|
+
def allows(self, relationship: str) -> bool:
|
|
33
|
+
"""Return True if the relationship type is permitted for this policy."""
|
|
34
|
+
if not self.allowed_relationship_types:
|
|
35
|
+
return True
|
|
36
|
+
return relationship.upper() in self.allowed_relationship_types
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
# ---------------------------------------------------------------------------
|
|
40
|
+
# Policy definitions
|
|
41
|
+
# ---------------------------------------------------------------------------
|
|
42
|
+
|
|
43
|
+
_POLICIES: dict[str, RetrievalPolicy] = {}
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _register(policy: RetrievalPolicy) -> None:
|
|
47
|
+
_POLICIES[policy.intent.upper()] = policy
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
# TRACE — follow the full call chain from endpoint to data layer
|
|
51
|
+
_register(RetrievalPolicy(
|
|
52
|
+
intent="TRACE",
|
|
53
|
+
allowed_relationship_types=frozenset({
|
|
54
|
+
"HANDLED_BY", "ROUTES_TO", "CALLS", "DEPENDS_ON", "IMPORTS", "TESTS",
|
|
55
|
+
"POSSIBLE_CALLS", "DEFINES", "CONTAINS", "PARALLEL_IMPLEMENTATION",
|
|
56
|
+
}),
|
|
57
|
+
preferred_flow=("ENTRYPOINT", "HANDLER", "SERVICE", "DATA", "TEST"),
|
|
58
|
+
include_callers=True,
|
|
59
|
+
include_callees=True,
|
|
60
|
+
include_tests=True,
|
|
61
|
+
include_framework=True,
|
|
62
|
+
max_graph_depth=4,
|
|
63
|
+
preferred_target_types=("API_ENDPOINT", "FUNCTION", "METHOD", "CLASS"),
|
|
64
|
+
))
|
|
65
|
+
|
|
66
|
+
# UNDERSTAND — explain structure, definitions, dependencies
|
|
67
|
+
_register(RetrievalPolicy(
|
|
68
|
+
intent="UNDERSTAND",
|
|
69
|
+
allowed_relationship_types=frozenset({
|
|
70
|
+
"DEFINES", "CONTAINS", "CALLS", "IMPORTS", "DEPENDS_ON",
|
|
71
|
+
"POSSIBLE_CALLS", "EXTENDS", "IMPLEMENTS", "PARALLEL_IMPLEMENTATION",
|
|
72
|
+
}),
|
|
73
|
+
preferred_flow=("TARGET", "DEFINITIONS", "DEPENDENCIES", "CALLERS_CALLEES"),
|
|
74
|
+
include_callers=True,
|
|
75
|
+
include_callees=True,
|
|
76
|
+
include_tests=False,
|
|
77
|
+
include_framework=True,
|
|
78
|
+
max_graph_depth=2,
|
|
79
|
+
preferred_target_types=("CLASS", "FUNCTION", "MODULE"),
|
|
80
|
+
))
|
|
81
|
+
|
|
82
|
+
# DEBUG — find callers, callees, error paths, tests
|
|
83
|
+
_register(RetrievalPolicy(
|
|
84
|
+
intent="DEBUG",
|
|
85
|
+
allowed_relationship_types=frozenset({
|
|
86
|
+
"CALLS", "CALLED_BY", "DEPENDS_ON", "TESTS", "IMPORTS",
|
|
87
|
+
"POSSIBLE_CALLS", "DEFINES", "PARALLEL_IMPLEMENTATION",
|
|
88
|
+
}),
|
|
89
|
+
preferred_flow=("TARGET", "EXECUTION_PATH", "ERROR_PATH", "TEST"),
|
|
90
|
+
include_callers=True,
|
|
91
|
+
include_callees=True,
|
|
92
|
+
include_tests=True,
|
|
93
|
+
include_framework=True,
|
|
94
|
+
max_graph_depth=3,
|
|
95
|
+
preferred_target_types=("FUNCTION", "METHOD", "CLASS"),
|
|
96
|
+
))
|
|
97
|
+
|
|
98
|
+
# IMPACT — what depends on, calls, or tests the target?
|
|
99
|
+
_register(RetrievalPolicy(
|
|
100
|
+
intent="IMPACT",
|
|
101
|
+
allowed_relationship_types=frozenset({
|
|
102
|
+
"CALLED_BY", "DEPENDS_ON", "IMPLEMENTED_BY", "TESTS", "CALLS",
|
|
103
|
+
"IMPORTS", "POSSIBLE_CALLS",
|
|
104
|
+
}),
|
|
105
|
+
preferred_flow=("TARGET", "CALLERS", "DEPENDENTS", "TEST"),
|
|
106
|
+
include_callers=True,
|
|
107
|
+
include_callees=False,
|
|
108
|
+
include_tests=True,
|
|
109
|
+
include_framework=False,
|
|
110
|
+
max_graph_depth=3,
|
|
111
|
+
preferred_target_types=("CLASS", "FUNCTION", "METHOD", "MODULE"),
|
|
112
|
+
))
|
|
113
|
+
|
|
114
|
+
# REVIEW — review changes; needs callers, callees, tests, and git context
|
|
115
|
+
_register(RetrievalPolicy(
|
|
116
|
+
intent="REVIEW",
|
|
117
|
+
allowed_relationship_types=frozenset({
|
|
118
|
+
"CALLS", "CALLED_BY", "DEPENDS_ON", "TESTS", "IMPORTS", "POSSIBLE_CALLS",
|
|
119
|
+
}),
|
|
120
|
+
preferred_flow=("TARGET", "HANDLER", "SERVICE", "TEST"),
|
|
121
|
+
include_callers=True,
|
|
122
|
+
include_callees=True,
|
|
123
|
+
include_tests=True,
|
|
124
|
+
include_git=True,
|
|
125
|
+
include_framework=True,
|
|
126
|
+
max_graph_depth=2,
|
|
127
|
+
preferred_target_types=("CLASS", "FUNCTION", "METHOD"),
|
|
128
|
+
))
|
|
129
|
+
|
|
130
|
+
# REFACTOR — similar to UNDERSTAND but emphasizes callers and tests
|
|
131
|
+
_register(RetrievalPolicy(
|
|
132
|
+
intent="REFACTOR",
|
|
133
|
+
allowed_relationship_types=frozenset({
|
|
134
|
+
"DEFINES", "CONTAINS", "CALLS", "IMPORTS", "DEPENDS_ON",
|
|
135
|
+
"POSSIBLE_CALLS", "EXTENDS", "IMPLEMENTS", "PARALLEL_IMPLEMENTATION",
|
|
136
|
+
}),
|
|
137
|
+
preferred_flow=("TARGET", "DEFINITIONS", "CALLERS_CALLEES", "TEST"),
|
|
138
|
+
include_callers=True,
|
|
139
|
+
include_callees=True,
|
|
140
|
+
include_tests=True,
|
|
141
|
+
include_framework=False,
|
|
142
|
+
max_graph_depth=2,
|
|
143
|
+
preferred_target_types=("CLASS", "FUNCTION", "METHOD"),
|
|
144
|
+
))
|
|
145
|
+
|
|
146
|
+
# CHANGE — modify existing behavior; needs definition + callers + tests
|
|
147
|
+
_register(RetrievalPolicy(
|
|
148
|
+
intent="CHANGE",
|
|
149
|
+
allowed_relationship_types=frozenset({
|
|
150
|
+
"DEFINES", "CALLS", "CALLED_BY", "IMPORTS", "DEPENDS_ON", "TESTS",
|
|
151
|
+
"POSSIBLE_CALLS", "PARALLEL_IMPLEMENTATION",
|
|
152
|
+
}),
|
|
153
|
+
preferred_flow=("TARGET", "HANDLER", "SERVICE", "DATA", "TEST"),
|
|
154
|
+
include_callers=True,
|
|
155
|
+
include_callees=True,
|
|
156
|
+
include_tests=True,
|
|
157
|
+
include_framework=True,
|
|
158
|
+
max_graph_depth=3,
|
|
159
|
+
preferred_target_types=("FUNCTION", "METHOD", "CLASS", "API_ENDPOINT"),
|
|
160
|
+
))
|
|
161
|
+
|
|
162
|
+
# TEST — locate and contextualize test coverage
|
|
163
|
+
_register(RetrievalPolicy(
|
|
164
|
+
intent="TEST",
|
|
165
|
+
allowed_relationship_types=frozenset({
|
|
166
|
+
"TESTS", "CALLS", "DEPENDS_ON", "DEFINES", "IMPORTS",
|
|
167
|
+
}),
|
|
168
|
+
preferred_flow=("TEST", "TARGET", "DEFINITIONS", "DEPENDENCIES"),
|
|
169
|
+
include_callers=False,
|
|
170
|
+
include_callees=True,
|
|
171
|
+
include_tests=True,
|
|
172
|
+
include_framework=False,
|
|
173
|
+
max_graph_depth=2,
|
|
174
|
+
preferred_target_types=("TEST", "FUNCTION", "METHOD"),
|
|
175
|
+
))
|
|
176
|
+
|
|
177
|
+
# ARCHITECTURE — top-level module and service relationships
|
|
178
|
+
_register(RetrievalPolicy(
|
|
179
|
+
intent="ARCHITECTURE",
|
|
180
|
+
allowed_relationship_types=frozenset({
|
|
181
|
+
"IMPORTS", "DEPENDS_ON", "EXTERNAL_SERVICE", "ROUTES_TO",
|
|
182
|
+
"DEFINES", "CONTAINS",
|
|
183
|
+
}),
|
|
184
|
+
preferred_flow=("ENTRYPOINT", "SERVICE", "DATA", "EXTERNAL"),
|
|
185
|
+
include_callers=False,
|
|
186
|
+
include_callees=False,
|
|
187
|
+
include_tests=False,
|
|
188
|
+
include_architecture=True,
|
|
189
|
+
include_framework=True,
|
|
190
|
+
max_graph_depth=1,
|
|
191
|
+
preferred_target_types=("MODULE", "API_ENDPOINT", "CLASS"),
|
|
192
|
+
))
|
|
193
|
+
|
|
194
|
+
# EXPLAIN — like UNDERSTAND but includes framework/route facts
|
|
195
|
+
_register(RetrievalPolicy(
|
|
196
|
+
intent="EXPLAIN",
|
|
197
|
+
allowed_relationship_types=frozenset({
|
|
198
|
+
"DEFINES", "CONTAINS", "CALLS", "IMPORTS", "DEPENDS_ON",
|
|
199
|
+
"HANDLED_BY", "ROUTES_TO", "POSSIBLE_CALLS", "EXTENDS", "IMPLEMENTS",
|
|
200
|
+
"PARALLEL_IMPLEMENTATION",
|
|
201
|
+
}),
|
|
202
|
+
preferred_flow=("TARGET", "DEFINITIONS", "DEPENDENCIES", "CALLERS_CALLEES"),
|
|
203
|
+
include_callers=True,
|
|
204
|
+
include_callees=True,
|
|
205
|
+
include_tests=False,
|
|
206
|
+
include_framework=True,
|
|
207
|
+
max_graph_depth=2,
|
|
208
|
+
preferred_target_types=("CLASS", "FUNCTION", "METHOD"),
|
|
209
|
+
))
|
|
210
|
+
|
|
211
|
+
# Default fallback (same as UNDERSTAND)
|
|
212
|
+
_DEFAULT_POLICY = _POLICIES["UNDERSTAND"]
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def get_retrieval_policy(intent: str) -> RetrievalPolicy:
|
|
216
|
+
"""Return the RetrievalPolicy for the given intent string.
|
|
217
|
+
|
|
218
|
+
Falls back to UNDERSTAND if the intent is not recognized.
|
|
219
|
+
"""
|
|
220
|
+
return _POLICIES.get(intent.upper().strip(), _DEFAULT_POLICY)
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
from .hybrid import (
|
|
2
|
+
SearchResult,
|
|
3
|
+
search,
|
|
4
|
+
search_exact_canonical_id,
|
|
5
|
+
search_exact_qualified,
|
|
6
|
+
search_exact_symbol,
|
|
7
|
+
search_lexical,
|
|
8
|
+
search_route,
|
|
9
|
+
)
|
|
10
|
+
from .semantic import EmbeddingProvider, SemanticSearchUnavailable, status
|
|
11
|
+
|
|
12
|
+
__all__ = [
|
|
13
|
+
"EmbeddingProvider",
|
|
14
|
+
"SearchResult",
|
|
15
|
+
"SemanticSearchUnavailable",
|
|
16
|
+
"search",
|
|
17
|
+
"search_exact_canonical_id",
|
|
18
|
+
"search_exact_qualified",
|
|
19
|
+
"search_exact_symbol",
|
|
20
|
+
"search_lexical",
|
|
21
|
+
"search_route",
|
|
22
|
+
"status",
|
|
23
|
+
]
|
|
@@ -0,0 +1,301 @@
|
|
|
1
|
+
"""Lexical search using SQLite FTS5.
|
|
2
|
+
|
|
3
|
+
Retrieval is lexical + path/symbol ranking. No semantic embeddings.
|
|
4
|
+
"""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import re
|
|
8
|
+
import sqlite3
|
|
9
|
+
from dataclasses import asdict, dataclass
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@dataclass(frozen=True)
|
|
13
|
+
class SearchResult:
|
|
14
|
+
file: str
|
|
15
|
+
symbol: str | None
|
|
16
|
+
start_line: int
|
|
17
|
+
end_line: int
|
|
18
|
+
score: float
|
|
19
|
+
reason: str
|
|
20
|
+
snippet: str
|
|
21
|
+
|
|
22
|
+
def as_dict(self) -> dict[str, object]:
|
|
23
|
+
return asdict(self)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
# FTS5 reserved words that cannot appear as bare terms
|
|
27
|
+
_FTS5_RESERVED = frozenset({
|
|
28
|
+
"AND", "OR", "NOT",
|
|
29
|
+
})
|
|
30
|
+
|
|
31
|
+
# FTS5 special characters to strip from terms
|
|
32
|
+
_FTS5_SPECIAL = re.compile(r"[^\w]") # keep only word chars inside terms
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _safe_fts5_terms(query: str) -> list[str]:
|
|
36
|
+
"""Extract word-like tokens safe for FTS5 MATCH expressions.
|
|
37
|
+
|
|
38
|
+
- Splits on non-alphanumeric characters (dots, dashes, etc.)
|
|
39
|
+
- Drops single-character tokens
|
|
40
|
+
- Drops FTS5 reserved words
|
|
41
|
+
- Returns deduplicated list preserving order
|
|
42
|
+
"""
|
|
43
|
+
# Split on anything that is not alphanumeric or underscore
|
|
44
|
+
raw_tokens = re.split(r"[^A-Za-z0-9_]+", query)
|
|
45
|
+
seen: set[str] = set()
|
|
46
|
+
result: list[str] = []
|
|
47
|
+
for tok in raw_tokens:
|
|
48
|
+
tok = tok.strip()
|
|
49
|
+
if len(tok) < 2:
|
|
50
|
+
continue
|
|
51
|
+
if tok.upper() in _FTS5_RESERVED:
|
|
52
|
+
continue
|
|
53
|
+
if tok not in seen:
|
|
54
|
+
seen.add(tok)
|
|
55
|
+
result.append(tok)
|
|
56
|
+
return result
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def search(
|
|
60
|
+
con: sqlite3.Connection,
|
|
61
|
+
query: str,
|
|
62
|
+
top_k: int = 10,
|
|
63
|
+
include_generated: bool = False,
|
|
64
|
+
) -> list[SearchResult]:
|
|
65
|
+
terms = _safe_fts5_terms(query)
|
|
66
|
+
if not terms:
|
|
67
|
+
return []
|
|
68
|
+
clauses = " OR ".join(terms)
|
|
69
|
+
try:
|
|
70
|
+
if include_generated:
|
|
71
|
+
rows = con.execute(
|
|
72
|
+
"SELECT c.path, c.symbol, c.start_line, c.end_line, c.content, "
|
|
73
|
+
"bm25(chunks_fts) AS rank "
|
|
74
|
+
"FROM chunks_fts "
|
|
75
|
+
"JOIN chunks c ON c.id = chunks_fts.rowid "
|
|
76
|
+
"WHERE chunks_fts MATCH ? ORDER BY rank LIMIT ?",
|
|
77
|
+
(clauses, top_k * 3),
|
|
78
|
+
).fetchall()
|
|
79
|
+
else:
|
|
80
|
+
rows = con.execute(
|
|
81
|
+
"SELECT c.path, c.symbol, c.start_line, c.end_line, c.content, "
|
|
82
|
+
"bm25(chunks_fts) AS rank "
|
|
83
|
+
"FROM chunks_fts "
|
|
84
|
+
"JOIN chunks c ON c.id = chunks_fts.rowid "
|
|
85
|
+
"LEFT JOIN files f ON f.path = c.path "
|
|
86
|
+
"WHERE chunks_fts MATCH ? AND (f.category IS NULL OR f.category != 'GENERATED') "
|
|
87
|
+
"ORDER BY rank LIMIT ?",
|
|
88
|
+
(clauses, top_k * 3),
|
|
89
|
+
).fetchall()
|
|
90
|
+
except sqlite3.OperationalError:
|
|
91
|
+
try:
|
|
92
|
+
rows = con.execute(
|
|
93
|
+
"SELECT c.path, c.symbol, c.start_line, c.end_line, c.content, "
|
|
94
|
+
"bm25(chunks_fts) AS rank "
|
|
95
|
+
"FROM chunks_fts "
|
|
96
|
+
"JOIN chunks c ON c.id = chunks_fts.rowid "
|
|
97
|
+
"WHERE chunks_fts MATCH ? ORDER BY rank LIMIT ?",
|
|
98
|
+
(clauses, top_k * 3),
|
|
99
|
+
).fetchall()
|
|
100
|
+
except sqlite3.OperationalError:
|
|
101
|
+
return []
|
|
102
|
+
|
|
103
|
+
results: list[SearchResult] = []
|
|
104
|
+
query.lower()
|
|
105
|
+
for row in rows:
|
|
106
|
+
path_bonus = 1.0 if any(t.lower() in row["path"].lower() for t in terms) else 0.0
|
|
107
|
+
symbol_bonus = (
|
|
108
|
+
1.0
|
|
109
|
+
if row["symbol"] and any(t.lower() in row["symbol"].lower() for t in terms)
|
|
110
|
+
else 0.0
|
|
111
|
+
)
|
|
112
|
+
# Exact symbol match (highest bonus)
|
|
113
|
+
exact_bonus = (
|
|
114
|
+
0.5
|
|
115
|
+
if row["symbol"] and any(
|
|
116
|
+
t.lower() == row["symbol"].lower()
|
|
117
|
+
or t.lower() == row["symbol"].split(".")[-1].lower()
|
|
118
|
+
for t in terms
|
|
119
|
+
)
|
|
120
|
+
else 0.0
|
|
121
|
+
)
|
|
122
|
+
score = round(min(1.0, 0.3 + path_bonus * 0.15 + symbol_bonus * 0.2 + exact_bonus * 0.35), 2)
|
|
123
|
+
reason = "lexical match"
|
|
124
|
+
if symbol_bonus:
|
|
125
|
+
reason += ", symbol match"
|
|
126
|
+
if path_bonus:
|
|
127
|
+
reason += ", path match"
|
|
128
|
+
snippet = row["content"][:900]
|
|
129
|
+
results.append(
|
|
130
|
+
SearchResult(
|
|
131
|
+
row["path"],
|
|
132
|
+
row["symbol"],
|
|
133
|
+
row["start_line"],
|
|
134
|
+
row["end_line"],
|
|
135
|
+
score,
|
|
136
|
+
reason,
|
|
137
|
+
snippet,
|
|
138
|
+
)
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
# Sort by score descending (bm25 is negative, lower = better match)
|
|
142
|
+
results.sort(key=lambda r: r.score, reverse=True)
|
|
143
|
+
return results[:top_k]
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def search_lexical(con: sqlite3.Connection, query: str, top_k: int = 10) -> list[SearchResult]:
|
|
147
|
+
"""Lexical FTS5 search (wrapper around search)."""
|
|
148
|
+
return search(con, query, top_k=top_k)
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def _fetch_snippet(
|
|
152
|
+
con: sqlite3.Connection, path: str, start: int, end: int, symbol: str | None = None
|
|
153
|
+
) -> str:
|
|
154
|
+
try:
|
|
155
|
+
row = con.execute(
|
|
156
|
+
"SELECT content FROM chunks WHERE path=? AND start_line<=? AND end_line>=? LIMIT 1",
|
|
157
|
+
(path, start, end),
|
|
158
|
+
).fetchone()
|
|
159
|
+
if row and row["content"]:
|
|
160
|
+
return str(row["content"])[:900]
|
|
161
|
+
if symbol:
|
|
162
|
+
row = con.execute(
|
|
163
|
+
"SELECT content FROM chunks WHERE symbol=? LIMIT 1", (symbol,)
|
|
164
|
+
).fetchone()
|
|
165
|
+
if row and row["content"]:
|
|
166
|
+
return str(row["content"])[:900]
|
|
167
|
+
except sqlite3.OperationalError:
|
|
168
|
+
pass
|
|
169
|
+
return ""
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def search_exact_canonical_id(
|
|
173
|
+
con: sqlite3.Connection, canonical_id: str
|
|
174
|
+
) -> list[SearchResult]:
|
|
175
|
+
"""Look up symbol by exact canonical ID."""
|
|
176
|
+
clean = canonical_id.strip()
|
|
177
|
+
if not clean:
|
|
178
|
+
return []
|
|
179
|
+
try:
|
|
180
|
+
rows = con.execute(
|
|
181
|
+
"SELECT canonical_id, qualified_name, path, start_line, end_line "
|
|
182
|
+
"FROM symbols WHERE canonical_id=? ORDER BY path, start_line",
|
|
183
|
+
(clean,),
|
|
184
|
+
).fetchall()
|
|
185
|
+
except sqlite3.OperationalError:
|
|
186
|
+
return []
|
|
187
|
+
|
|
188
|
+
results: list[SearchResult] = []
|
|
189
|
+
for r in rows:
|
|
190
|
+
snippet = _fetch_snippet(con, r["path"], r["start_line"], r["end_line"], r["qualified_name"])
|
|
191
|
+
results.append(
|
|
192
|
+
SearchResult(
|
|
193
|
+
file=r["path"],
|
|
194
|
+
symbol=r["qualified_name"] or r["canonical_id"],
|
|
195
|
+
start_line=r["start_line"],
|
|
196
|
+
end_line=r["end_line"],
|
|
197
|
+
score=1.0,
|
|
198
|
+
reason="exact canonical id match",
|
|
199
|
+
snippet=snippet,
|
|
200
|
+
)
|
|
201
|
+
)
|
|
202
|
+
return results
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def search_exact_qualified(
|
|
206
|
+
con: sqlite3.Connection, qualified_name: str
|
|
207
|
+
) -> list[SearchResult]:
|
|
208
|
+
"""Look up symbol by exact qualified name."""
|
|
209
|
+
clean = qualified_name.strip()
|
|
210
|
+
if not clean:
|
|
211
|
+
return []
|
|
212
|
+
try:
|
|
213
|
+
rows = con.execute(
|
|
214
|
+
"SELECT canonical_id, qualified_name, path, start_line, end_line "
|
|
215
|
+
"FROM symbols WHERE qualified_name=? OR canonical_id=? ORDER BY path, start_line",
|
|
216
|
+
(clean, clean),
|
|
217
|
+
).fetchall()
|
|
218
|
+
except sqlite3.OperationalError:
|
|
219
|
+
return []
|
|
220
|
+
|
|
221
|
+
results: list[SearchResult] = []
|
|
222
|
+
for r in rows:
|
|
223
|
+
snippet = _fetch_snippet(con, r["path"], r["start_line"], r["end_line"], r["qualified_name"])
|
|
224
|
+
results.append(
|
|
225
|
+
SearchResult(
|
|
226
|
+
file=r["path"],
|
|
227
|
+
symbol=r["qualified_name"],
|
|
228
|
+
start_line=r["start_line"],
|
|
229
|
+
end_line=r["end_line"],
|
|
230
|
+
score=0.95,
|
|
231
|
+
reason="exact qualified match",
|
|
232
|
+
snippet=snippet,
|
|
233
|
+
)
|
|
234
|
+
)
|
|
235
|
+
return results
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def search_exact_symbol(
|
|
239
|
+
con: sqlite3.Connection, name: str
|
|
240
|
+
) -> list[SearchResult]:
|
|
241
|
+
"""Look up symbol by short identifier name."""
|
|
242
|
+
clean = name.strip()
|
|
243
|
+
if not clean:
|
|
244
|
+
return []
|
|
245
|
+
try:
|
|
246
|
+
rows = con.execute(
|
|
247
|
+
"SELECT canonical_id, qualified_name, path, start_line, end_line "
|
|
248
|
+
"FROM symbols WHERE name=? ORDER BY path, start_line",
|
|
249
|
+
(clean,),
|
|
250
|
+
).fetchall()
|
|
251
|
+
except sqlite3.OperationalError:
|
|
252
|
+
return []
|
|
253
|
+
|
|
254
|
+
results: list[SearchResult] = []
|
|
255
|
+
for r in rows:
|
|
256
|
+
snippet = _fetch_snippet(con, r["path"], r["start_line"], r["end_line"], r["qualified_name"])
|
|
257
|
+
results.append(
|
|
258
|
+
SearchResult(
|
|
259
|
+
file=r["path"],
|
|
260
|
+
symbol=r["qualified_name"],
|
|
261
|
+
start_line=r["start_line"],
|
|
262
|
+
end_line=r["end_line"],
|
|
263
|
+
score=0.90,
|
|
264
|
+
reason="exact symbol match",
|
|
265
|
+
snippet=snippet,
|
|
266
|
+
)
|
|
267
|
+
)
|
|
268
|
+
return results
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
def search_route(
|
|
272
|
+
con: sqlite3.Connection, route_or_endpoint: str
|
|
273
|
+
) -> list[SearchResult]:
|
|
274
|
+
"""Look up framework route by path or endpoint ID."""
|
|
275
|
+
clean = route_or_endpoint.strip()
|
|
276
|
+
if not clean:
|
|
277
|
+
return []
|
|
278
|
+
try:
|
|
279
|
+
rows = con.execute(
|
|
280
|
+
"SELECT route_path, http_method, handler_name, file_path, line, endpoint_id, evidence "
|
|
281
|
+
"FROM framework_routes WHERE route_path=? OR endpoint_id=? ORDER BY file_path, line",
|
|
282
|
+
(clean, clean),
|
|
283
|
+
).fetchall()
|
|
284
|
+
except sqlite3.OperationalError:
|
|
285
|
+
return []
|
|
286
|
+
|
|
287
|
+
results: list[SearchResult] = []
|
|
288
|
+
for r in rows:
|
|
289
|
+
snippet = _fetch_snippet(con, r["file_path"], r["line"], r["line"], r["handler_name"])
|
|
290
|
+
results.append(
|
|
291
|
+
SearchResult(
|
|
292
|
+
file=r["file_path"],
|
|
293
|
+
symbol=r["handler_name"] or r["route_path"],
|
|
294
|
+
start_line=r["line"],
|
|
295
|
+
end_line=r["line"],
|
|
296
|
+
score=0.95,
|
|
297
|
+
reason=f"route match: {r['http_method']} {r['route_path']}",
|
|
298
|
+
snippet=snippet,
|
|
299
|
+
)
|
|
300
|
+
)
|
|
301
|
+
return results
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from abc import ABC, abstractmethod
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class SemanticSearchUnavailable(RuntimeError):
|
|
8
|
+
"""No embedding provider has been configured for semantic retrieval."""
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class EmbeddingProvider(ABC):
|
|
12
|
+
"""Provider-neutral embeddings interface; implementations must be deterministic in tests."""
|
|
13
|
+
|
|
14
|
+
@abstractmethod
|
|
15
|
+
def embed(self, texts: list[str]) -> list[list[float]]:
|
|
16
|
+
raise NotImplementedError
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass(frozen=True)
|
|
20
|
+
class SemanticStatus:
|
|
21
|
+
available: bool
|
|
22
|
+
detail: str
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def status(provider: EmbeddingProvider | None) -> SemanticStatus:
|
|
26
|
+
if provider is None:
|
|
27
|
+
return SemanticStatus(False, "No embedding provider configured; lexical and symbol retrieval remain active.")
|
|
28
|
+
return SemanticStatus(True, "Embedding provider configured.")
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import fnmatch
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from urllib.parse import unquote
|
|
6
|
+
|
|
7
|
+
from codegraph.errors import SecurityError
|
|
8
|
+
|
|
9
|
+
DEFAULT_SENSITIVE_PATTERNS = (
|
|
10
|
+
".env", ".env.*", "*.pem", "*.key", "id_rsa", "credentials.*", "secrets.*",
|
|
11
|
+
"service-account*.json", ".aws/*", ".ssh/*",
|
|
12
|
+
)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def is_sensitive(relative_path: Path, patterns: tuple[str, ...] = DEFAULT_SENSITIVE_PATTERNS) -> bool:
|
|
16
|
+
value = relative_path.as_posix()
|
|
17
|
+
name = relative_path.name
|
|
18
|
+
return any(fnmatch.fnmatch(value, p) or fnmatch.fnmatch(name, p) for p in patterns)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def safe_path(repository: Path, requested: str | Path) -> Path:
|
|
22
|
+
"""Resolve a repo-relative path and reject traversal and symlink escapes."""
|
|
23
|
+
root = repository.resolve(strict=True)
|
|
24
|
+
raw_value = unquote(str(requested))
|
|
25
|
+
if "\x00" in raw_value:
|
|
26
|
+
raise SecurityError("null bytes are not allowed in paths")
|
|
27
|
+
raw = Path(raw_value)
|
|
28
|
+
if raw.is_absolute():
|
|
29
|
+
raise SecurityError("absolute paths are not allowed")
|
|
30
|
+
candidate = (root / raw).resolve(strict=False)
|
|
31
|
+
try:
|
|
32
|
+
candidate.relative_to(root)
|
|
33
|
+
except ValueError as exc:
|
|
34
|
+
raise SecurityError("path escapes the configured repository") from exc
|
|
35
|
+
return candidate
|