codegraph-engine 2.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codegraph/__init__.py +37 -0
- codegraph/agent.py +26 -0
- codegraph/architecture.py +328 -0
- codegraph/audit.py +106 -0
- codegraph/cache.py +95 -0
- codegraph/cli.py +854 -0
- codegraph/config.py +43 -0
- codegraph/constraints.py +238 -0
- codegraph/context.py +1228 -0
- codegraph/epistemic.py +90 -0
- codegraph/errors.py +275 -0
- codegraph/evidence/__init__.py +15 -0
- codegraph/evidence/citations.py +397 -0
- codegraph/frameworks.py +434 -0
- codegraph/freshness.py +295 -0
- codegraph/git.py +278 -0
- codegraph/graph/__init__.py +46 -0
- codegraph/graph/models.py +41 -0
- codegraph/graph/traversal.py +1291 -0
- codegraph/indexing/__init__.py +4 -0
- codegraph/indexing/classifier.py +274 -0
- codegraph/indexing/indexer.py +943 -0
- codegraph/indexing/models.py +338 -0
- codegraph/indexing/parser.py +1240 -0
- codegraph/indexing/scanner.py +200 -0
- codegraph/indexing/test_framework.py +116 -0
- codegraph/interrogation.py +1582 -0
- codegraph/llm/__init__.py +3 -0
- codegraph/llm/base.py +15 -0
- codegraph/llm/context.py +20 -0
- codegraph/mcp/__init__.py +3 -0
- codegraph/mcp/server.py +736 -0
- codegraph/memory/__init__.py +3 -0
- codegraph/memory/store.py +46 -0
- codegraph/models.py +289 -0
- codegraph/observability.py +151 -0
- codegraph/optimizer.py +372 -0
- codegraph/planner.py +417 -0
- codegraph/py.typed +1 -0
- codegraph/query_expansion.py +199 -0
- codegraph/ranking.py +363 -0
- codegraph/resolver.py +843 -0
- codegraph/resources/__init__.py +45 -0
- codegraph/resources/cache.py +117 -0
- codegraph/resources/coalescer.py +83 -0
- codegraph/resources/debouncer.py +98 -0
- codegraph/resources/governor.py +232 -0
- codegraph/resources/policy.py +123 -0
- codegraph/retrieval_policy.py +220 -0
- codegraph/search/__init__.py +23 -0
- codegraph/search/hybrid.py +301 -0
- codegraph/search/semantic.py +28 -0
- codegraph/security/__init__.py +3 -0
- codegraph/security/paths.py +35 -0
- codegraph/target_resolver.py +348 -0
- codegraph/task.py +637 -0
- codegraph_engine-2.1.1.dist-info/METADATA +334 -0
- codegraph_engine-2.1.1.dist-info/RECORD +62 -0
- codegraph_engine-2.1.1.dist-info/WHEEL +5 -0
- codegraph_engine-2.1.1.dist-info/entry_points.txt +2 -0
- codegraph_engine-2.1.1.dist-info/licenses/LICENSE +21 -0
- codegraph_engine-2.1.1.dist-info/top_level.txt +1 -0
codegraph/planner.py
ADDED
|
@@ -0,0 +1,417 @@
|
|
|
1
|
+
"""Deterministic Retrieval Planner for CodeGraph MCP.
|
|
2
|
+
|
|
3
|
+
Translates a grounded TaskSpec into an intent-driven, deterministic RetrievalPlan.
|
|
4
|
+
Query groups distinguish between primary targets, callers, callees, dependencies,
|
|
5
|
+
tests, git changes, framework routes, and architecture slices without guessing.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import hashlib
|
|
10
|
+
import json
|
|
11
|
+
import sqlite3
|
|
12
|
+
from dataclasses import asdict, dataclass
|
|
13
|
+
|
|
14
|
+
from codegraph.task import AmbiguityStatus, TaskAmbiguity, TaskIntent, TaskSpec
|
|
15
|
+
|
|
16
|
+
PLAN_SCHEMA_VERSION = "1.0"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class QueryPurpose:
|
|
20
|
+
PRIMARY_TARGET = "PRIMARY_TARGET"
|
|
21
|
+
ENTRY_POINT = "ENTRY_POINT"
|
|
22
|
+
RELATED_SYMBOLS = "RELATED_SYMBOLS"
|
|
23
|
+
CALLERS = "CALLERS"
|
|
24
|
+
CALLEES = "CALLEES"
|
|
25
|
+
DEPENDENCIES = "DEPENDENCIES"
|
|
26
|
+
IMPLEMENTATIONS = "IMPLEMENTATIONS"
|
|
27
|
+
TESTS = "TESTS"
|
|
28
|
+
FRAMEWORK = "FRAMEWORK"
|
|
29
|
+
GIT = "GIT"
|
|
30
|
+
ARCHITECTURE = "ARCHITECTURE"
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass(frozen=True)
|
|
34
|
+
class QueryGroup:
|
|
35
|
+
purpose: str
|
|
36
|
+
queries: tuple[str, ...]
|
|
37
|
+
weight: float = 1.0
|
|
38
|
+
required: bool = True
|
|
39
|
+
|
|
40
|
+
def as_dict(self) -> dict[str, object]:
|
|
41
|
+
return asdict(self)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
@dataclass(frozen=True)
|
|
45
|
+
class RetrievalPlan:
|
|
46
|
+
schema_version: str = PLAN_SCHEMA_VERSION
|
|
47
|
+
task_spec_hash: str = ""
|
|
48
|
+
task_fingerprint: str = ""
|
|
49
|
+
intent: str = ""
|
|
50
|
+
|
|
51
|
+
query_groups: tuple[QueryGroup, ...] = ()
|
|
52
|
+
|
|
53
|
+
required_relationships: tuple[str, ...] = ()
|
|
54
|
+
optional_relationships: tuple[str, ...] = ()
|
|
55
|
+
|
|
56
|
+
include_tests: bool = True
|
|
57
|
+
include_git: bool = False
|
|
58
|
+
include_framework: bool = True
|
|
59
|
+
include_architecture: bool = False
|
|
60
|
+
|
|
61
|
+
entry_point_strategy: str = "DIRECT" # DIRECT | TRACE_FROM_ENDPOINT | TRACE_HIERARCHY | GIT_DIFF_ENTRY
|
|
62
|
+
|
|
63
|
+
max_depth: int = 3
|
|
64
|
+
max_nodes: int = 300
|
|
65
|
+
max_edges: int = 1200
|
|
66
|
+
token_budget: int = 20_000
|
|
67
|
+
|
|
68
|
+
freshness_policy: str = "REQUIRE_FRESH" # REQUIRE_FRESH | PERMIT_PARTIAL | BEST_EFFORT
|
|
69
|
+
evidence_policy: str = "VERIFY_SOURCE_HASH"
|
|
70
|
+
ambiguity_policy: str = "REPORT_OR_ASSUME" # REPORT_OR_ASSUME | STRICT_FAIL | BEST_EFFORT
|
|
71
|
+
|
|
72
|
+
def as_dict(self) -> dict[str, object]:
|
|
73
|
+
return {
|
|
74
|
+
"schema_version": self.schema_version,
|
|
75
|
+
"task_spec_hash": self.task_spec_hash or self.task_fingerprint,
|
|
76
|
+
"task_fingerprint": self.task_fingerprint or self.task_spec_hash,
|
|
77
|
+
"intent": self.intent,
|
|
78
|
+
"query_groups": [q.as_dict() for q in self.query_groups],
|
|
79
|
+
"required_relationships": list(self.required_relationships),
|
|
80
|
+
"optional_relationships": list(self.optional_relationships),
|
|
81
|
+
"include_tests": self.include_tests,
|
|
82
|
+
"include_git": self.include_git,
|
|
83
|
+
"include_framework": self.include_framework,
|
|
84
|
+
"include_architecture": self.include_architecture,
|
|
85
|
+
"entry_point_strategy": self.entry_point_strategy,
|
|
86
|
+
"max_depth": self.max_depth,
|
|
87
|
+
"max_nodes": self.max_nodes,
|
|
88
|
+
"max_edges": self.max_edges,
|
|
89
|
+
"token_budget": self.token_budget,
|
|
90
|
+
"freshness_policy": self.freshness_policy,
|
|
91
|
+
"evidence_policy": self.evidence_policy,
|
|
92
|
+
"ambiguity_policy": self.ambiguity_policy,
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def compute_task_spec_hash(task_spec: TaskSpec) -> str:
|
|
97
|
+
"""Compute a deterministic SHA-256 hash of a TaskSpec."""
|
|
98
|
+
canonical_payload = {
|
|
99
|
+
"intent": task_spec.intent,
|
|
100
|
+
"goal": task_spec.goal,
|
|
101
|
+
"targets": sorted(task_spec.targets),
|
|
102
|
+
"entities": sorted(task_spec.entities),
|
|
103
|
+
"operations": sorted(task_spec.operations),
|
|
104
|
+
"constraints": sorted(task_spec.constraints),
|
|
105
|
+
"exclusions": sorted(task_spec.exclusions),
|
|
106
|
+
"scope_paths": sorted(task_spec.scope_paths),
|
|
107
|
+
"scope_modules": sorted(task_spec.scope_modules),
|
|
108
|
+
"frameworks": sorted(task_spec.frameworks),
|
|
109
|
+
"time_scope": task_spec.time_scope or "",
|
|
110
|
+
}
|
|
111
|
+
raw_bytes = json.dumps(canonical_payload, sort_keys=True, separators=(",", ":")).encode("utf-8")
|
|
112
|
+
return hashlib.sha256(raw_bytes).hexdigest()[:16]
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def build_retrieval_plan(
|
|
116
|
+
task_spec: TaskSpec,
|
|
117
|
+
ambiguities: tuple[TaskAmbiguity, ...] = (),
|
|
118
|
+
con: sqlite3.Connection | None = None,
|
|
119
|
+
token_budget: int = 20_000,
|
|
120
|
+
resource_mode: str = "BALANCED",
|
|
121
|
+
) -> RetrievalPlan:
|
|
122
|
+
"""Construct an intent-specific, repository-aware RetrievalPlan for a TaskSpec."""
|
|
123
|
+
spec_hash = compute_task_spec_hash(task_spec)
|
|
124
|
+
intent = task_spec.intent.upper()
|
|
125
|
+
|
|
126
|
+
# Base query items from targets
|
|
127
|
+
primary_queries = tuple(task_spec.priority_targets or task_spec.targets or (task_spec.goal,))
|
|
128
|
+
related_queries: list[str] = []
|
|
129
|
+
|
|
130
|
+
if con and primary_queries:
|
|
131
|
+
from codegraph.query_expansion import expand_query_terms, get_search_queries
|
|
132
|
+
|
|
133
|
+
expansions = expand_query_terms(primary_queries[:5], con, max_expansions=8)
|
|
134
|
+
expanded_terms = get_search_queries(expansions)
|
|
135
|
+
for term in expanded_terms:
|
|
136
|
+
if term not in primary_queries and term not in related_queries:
|
|
137
|
+
related_queries.append(term)
|
|
138
|
+
|
|
139
|
+
query_groups: list[QueryGroup] = []
|
|
140
|
+
|
|
141
|
+
# Primary target group
|
|
142
|
+
query_groups.append(
|
|
143
|
+
QueryGroup(
|
|
144
|
+
purpose=QueryPurpose.PRIMARY_TARGET,
|
|
145
|
+
queries=primary_queries,
|
|
146
|
+
weight=1.0,
|
|
147
|
+
required=True,
|
|
148
|
+
)
|
|
149
|
+
)
|
|
150
|
+
|
|
151
|
+
if related_queries:
|
|
152
|
+
query_groups.append(
|
|
153
|
+
QueryGroup(
|
|
154
|
+
purpose=QueryPurpose.RELATED_SYMBOLS,
|
|
155
|
+
queries=tuple(related_queries[:8]),
|
|
156
|
+
weight=0.75,
|
|
157
|
+
required=False,
|
|
158
|
+
)
|
|
159
|
+
)
|
|
160
|
+
|
|
161
|
+
# Strategy configuration based on intent
|
|
162
|
+
include_tests = True
|
|
163
|
+
include_git = False
|
|
164
|
+
include_framework = True
|
|
165
|
+
include_architecture = False
|
|
166
|
+
entry_point_strategy = "DIRECT"
|
|
167
|
+
max_depth = 3
|
|
168
|
+
required_relationships: list[str] = []
|
|
169
|
+
optional_relationships: list[str] = ["CALLS", "IMPORTS"]
|
|
170
|
+
|
|
171
|
+
# Check explicit time_scope or operations in TaskSpec
|
|
172
|
+
if task_spec.time_scope or "INSPECT_RECENT_CHANGES" in task_spec.operations:
|
|
173
|
+
include_git = True
|
|
174
|
+
|
|
175
|
+
# Adaptive depth calculation based on query complexity and resource mode
|
|
176
|
+
goal_lower = (task_spec.goal or "").lower()
|
|
177
|
+
is_shallow = any(pattern in goal_lower for pattern in ("where is", "where's", "find definition", "show definition", "locate"))
|
|
178
|
+
mode_upper = (resource_mode or "BALANCED").upper()
|
|
179
|
+
|
|
180
|
+
if is_shallow:
|
|
181
|
+
max_depth = 1
|
|
182
|
+
elif mode_upper == "FAST":
|
|
183
|
+
max_depth = min(max_depth, 2)
|
|
184
|
+
elif mode_upper == "DEEP":
|
|
185
|
+
max_depth = max(max_depth, 5)
|
|
186
|
+
|
|
187
|
+
if mode_upper == "FAST":
|
|
188
|
+
max_nodes = 100
|
|
189
|
+
max_edges = 400
|
|
190
|
+
elif mode_upper == "DEEP":
|
|
191
|
+
max_nodes = 600
|
|
192
|
+
max_edges = 2400
|
|
193
|
+
else:
|
|
194
|
+
max_nodes = 300
|
|
195
|
+
max_edges = 1200
|
|
196
|
+
|
|
197
|
+
if intent in (TaskIntent.UNDERSTAND.value, TaskIntent.EXPLAIN.value):
|
|
198
|
+
include_architecture = True
|
|
199
|
+
required_relationships = ["HANDLED_BY", "CALLS"]
|
|
200
|
+
optional_relationships = ["IMPORTS", "EXTENDS"]
|
|
201
|
+
query_groups.append(
|
|
202
|
+
QueryGroup(
|
|
203
|
+
purpose=QueryPurpose.ARCHITECTURE,
|
|
204
|
+
queries=primary_queries[:3],
|
|
205
|
+
weight=0.6,
|
|
206
|
+
required=False,
|
|
207
|
+
)
|
|
208
|
+
)
|
|
209
|
+
query_groups.append(
|
|
210
|
+
QueryGroup(
|
|
211
|
+
purpose=QueryPurpose.TESTS,
|
|
212
|
+
queries=primary_queries[:3],
|
|
213
|
+
weight=0.5,
|
|
214
|
+
required=False,
|
|
215
|
+
)
|
|
216
|
+
)
|
|
217
|
+
|
|
218
|
+
elif intent == TaskIntent.DEBUG.value:
|
|
219
|
+
include_git = True
|
|
220
|
+
entry_point_strategy = "TRACE_FROM_ENDPOINT"
|
|
221
|
+
max_depth = 3
|
|
222
|
+
required_relationships = ["HANDLED_BY", "CALLS"]
|
|
223
|
+
optional_relationships = ["IMPORTS", "EXTENDS", "POSSIBLE_CALLS"]
|
|
224
|
+
query_groups.append(
|
|
225
|
+
QueryGroup(
|
|
226
|
+
purpose=QueryPurpose.CALLERS,
|
|
227
|
+
queries=primary_queries[:4],
|
|
228
|
+
weight=0.85,
|
|
229
|
+
required=True,
|
|
230
|
+
)
|
|
231
|
+
)
|
|
232
|
+
query_groups.append(
|
|
233
|
+
QueryGroup(
|
|
234
|
+
purpose=QueryPurpose.TESTS,
|
|
235
|
+
queries=primary_queries[:3],
|
|
236
|
+
weight=0.7,
|
|
237
|
+
required=True,
|
|
238
|
+
)
|
|
239
|
+
)
|
|
240
|
+
if include_git:
|
|
241
|
+
query_groups.append(
|
|
242
|
+
QueryGroup(
|
|
243
|
+
purpose=QueryPurpose.GIT,
|
|
244
|
+
queries=primary_queries[:3],
|
|
245
|
+
weight=0.8,
|
|
246
|
+
required=False,
|
|
247
|
+
)
|
|
248
|
+
)
|
|
249
|
+
|
|
250
|
+
elif intent in (TaskIntent.CHANGE.value, TaskIntent.REFACTOR.value):
|
|
251
|
+
entry_point_strategy = "DIRECT"
|
|
252
|
+
max_depth = 2
|
|
253
|
+
required_relationships = ["CALLS"]
|
|
254
|
+
optional_relationships = ["EXTENDS", "IMPLEMENTS", "IMPORTS"]
|
|
255
|
+
query_groups.append(
|
|
256
|
+
QueryGroup(
|
|
257
|
+
purpose=QueryPurpose.CALLERS,
|
|
258
|
+
queries=primary_queries[:4],
|
|
259
|
+
weight=0.9,
|
|
260
|
+
required=True,
|
|
261
|
+
)
|
|
262
|
+
)
|
|
263
|
+
query_groups.append(
|
|
264
|
+
QueryGroup(
|
|
265
|
+
purpose=QueryPurpose.CALLEES,
|
|
266
|
+
queries=primary_queries[:4],
|
|
267
|
+
weight=0.85,
|
|
268
|
+
required=True,
|
|
269
|
+
)
|
|
270
|
+
)
|
|
271
|
+
query_groups.append(
|
|
272
|
+
QueryGroup(
|
|
273
|
+
purpose=QueryPurpose.TESTS,
|
|
274
|
+
queries=primary_queries[:3],
|
|
275
|
+
weight=0.75,
|
|
276
|
+
required=True,
|
|
277
|
+
)
|
|
278
|
+
)
|
|
279
|
+
|
|
280
|
+
elif intent == TaskIntent.TRACE.value:
|
|
281
|
+
entry_point_strategy = "TRACE_HIERARCHY"
|
|
282
|
+
max_depth = 4
|
|
283
|
+
required_relationships = ["HANDLED_BY", "CALLS"]
|
|
284
|
+
optional_relationships = ["IMPORTS"]
|
|
285
|
+
query_groups.append(
|
|
286
|
+
QueryGroup(
|
|
287
|
+
purpose=QueryPurpose.FRAMEWORK,
|
|
288
|
+
queries=primary_queries[:4],
|
|
289
|
+
weight=0.95,
|
|
290
|
+
required=True,
|
|
291
|
+
)
|
|
292
|
+
)
|
|
293
|
+
query_groups.append(
|
|
294
|
+
QueryGroup(
|
|
295
|
+
purpose=QueryPurpose.CALLEES,
|
|
296
|
+
queries=primary_queries[:4],
|
|
297
|
+
weight=0.9,
|
|
298
|
+
required=True,
|
|
299
|
+
)
|
|
300
|
+
)
|
|
301
|
+
|
|
302
|
+
elif intent == TaskIntent.IMPACT.value:
|
|
303
|
+
entry_point_strategy = "DIRECT"
|
|
304
|
+
max_depth = 4
|
|
305
|
+
include_architecture = True
|
|
306
|
+
required_relationships = ["CALLS", "IMPORTS"]
|
|
307
|
+
optional_relationships = ["HANDLED_BY", "EXTENDS"]
|
|
308
|
+
query_groups.append(
|
|
309
|
+
QueryGroup(
|
|
310
|
+
purpose=QueryPurpose.CALLERS,
|
|
311
|
+
queries=primary_queries[:5],
|
|
312
|
+
weight=0.95,
|
|
313
|
+
required=True,
|
|
314
|
+
)
|
|
315
|
+
)
|
|
316
|
+
query_groups.append(
|
|
317
|
+
QueryGroup(
|
|
318
|
+
purpose=QueryPurpose.DEPENDENCIES,
|
|
319
|
+
queries=primary_queries[:5],
|
|
320
|
+
weight=0.8,
|
|
321
|
+
required=True,
|
|
322
|
+
)
|
|
323
|
+
)
|
|
324
|
+
query_groups.append(
|
|
325
|
+
QueryGroup(
|
|
326
|
+
purpose=QueryPurpose.TESTS,
|
|
327
|
+
queries=primary_queries[:3],
|
|
328
|
+
weight=0.75,
|
|
329
|
+
required=True,
|
|
330
|
+
)
|
|
331
|
+
)
|
|
332
|
+
|
|
333
|
+
elif intent == TaskIntent.REVIEW.value:
|
|
334
|
+
include_git = True
|
|
335
|
+
entry_point_strategy = "GIT_DIFF_ENTRY"
|
|
336
|
+
max_depth = 2
|
|
337
|
+
required_relationships = ["CALLS"]
|
|
338
|
+
optional_relationships = ["IMPORTS"]
|
|
339
|
+
query_groups.append(
|
|
340
|
+
QueryGroup(
|
|
341
|
+
purpose=QueryPurpose.GIT,
|
|
342
|
+
queries=primary_queries or ("HEAD~1..HEAD",),
|
|
343
|
+
weight=1.0,
|
|
344
|
+
required=True,
|
|
345
|
+
)
|
|
346
|
+
)
|
|
347
|
+
query_groups.append(
|
|
348
|
+
QueryGroup(
|
|
349
|
+
purpose=QueryPurpose.TESTS,
|
|
350
|
+
queries=primary_queries[:3],
|
|
351
|
+
weight=0.8,
|
|
352
|
+
required=True,
|
|
353
|
+
)
|
|
354
|
+
)
|
|
355
|
+
|
|
356
|
+
elif intent == TaskIntent.ARCHITECTURE.value:
|
|
357
|
+
include_architecture = True
|
|
358
|
+
include_tests = False
|
|
359
|
+
max_depth = 2
|
|
360
|
+
required_relationships = ["HANDLED_BY", "IMPORTS"]
|
|
361
|
+
optional_relationships = ["CALLS", "EXTENDS"]
|
|
362
|
+
query_groups.append(
|
|
363
|
+
QueryGroup(
|
|
364
|
+
purpose=QueryPurpose.ARCHITECTURE,
|
|
365
|
+
queries=primary_queries,
|
|
366
|
+
weight=1.0,
|
|
367
|
+
required=True,
|
|
368
|
+
)
|
|
369
|
+
)
|
|
370
|
+
query_groups.append(
|
|
371
|
+
QueryGroup(
|
|
372
|
+
purpose=QueryPurpose.FRAMEWORK,
|
|
373
|
+
queries=primary_queries,
|
|
374
|
+
weight=0.85,
|
|
375
|
+
required=False,
|
|
376
|
+
)
|
|
377
|
+
)
|
|
378
|
+
|
|
379
|
+
elif intent == TaskIntent.TEST.value:
|
|
380
|
+
include_tests = True
|
|
381
|
+
max_depth = 2
|
|
382
|
+
required_relationships = ["TESTED_BY", "CALLS"]
|
|
383
|
+
optional_relationships = ["IMPORTS"]
|
|
384
|
+
query_groups.append(
|
|
385
|
+
QueryGroup(
|
|
386
|
+
purpose=QueryPurpose.TESTS,
|
|
387
|
+
queries=primary_queries,
|
|
388
|
+
weight=1.0,
|
|
389
|
+
required=True,
|
|
390
|
+
)
|
|
391
|
+
)
|
|
392
|
+
|
|
393
|
+
# Ambiguity policy handling
|
|
394
|
+
has_ambiguity = any(a.status == AmbiguityStatus.AMBIGUOUS for a in ambiguities)
|
|
395
|
+
ambiguity_policy = "REPORT_OR_ASSUME" if has_ambiguity else "CLEAR"
|
|
396
|
+
|
|
397
|
+
return RetrievalPlan(
|
|
398
|
+
schema_version=PLAN_SCHEMA_VERSION,
|
|
399
|
+
task_spec_hash=spec_hash,
|
|
400
|
+
task_fingerprint=spec_hash,
|
|
401
|
+
intent=intent,
|
|
402
|
+
query_groups=tuple(query_groups),
|
|
403
|
+
required_relationships=tuple(dict.fromkeys(required_relationships)),
|
|
404
|
+
optional_relationships=tuple(dict.fromkeys(optional_relationships)),
|
|
405
|
+
include_tests=include_tests,
|
|
406
|
+
include_git=include_git,
|
|
407
|
+
include_framework=include_framework,
|
|
408
|
+
include_architecture=include_architecture,
|
|
409
|
+
entry_point_strategy=entry_point_strategy,
|
|
410
|
+
max_depth=max_depth,
|
|
411
|
+
max_nodes=max_nodes,
|
|
412
|
+
max_edges=max_edges,
|
|
413
|
+
token_budget=token_budget,
|
|
414
|
+
freshness_policy="REQUIRE_FRESH",
|
|
415
|
+
evidence_policy="VERIFY_SOURCE_HASH",
|
|
416
|
+
ambiguity_policy=ambiguity_policy,
|
|
417
|
+
)
|
codegraph/py.typed
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
# Marker file for PEP 561.
|
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
"""Qualified Query Expansion for CodeGraph MCP v2.1.
|
|
2
|
+
|
|
3
|
+
Converts raw target terms into precise, explainable search queries.
|
|
4
|
+
|
|
5
|
+
Key rules:
|
|
6
|
+
1. If a term resolves to a canonical_id, use canonical_id as the search term
|
|
7
|
+
2. If a term resolves to a qualified_name, use qualified_name
|
|
8
|
+
3. If a term matches a route, also expand to the handler symbol
|
|
9
|
+
4. Generic method names MUST be qualified when qualified context exists:
|
|
10
|
+
e.g. "get" → "AdminDashboardView.get" (never bare "get")
|
|
11
|
+
5. Every expansion is recorded as a QueryExpansion with full provenance
|
|
12
|
+
|
|
13
|
+
Every expansion is source-backed. No fabricated terms.
|
|
14
|
+
"""
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import sqlite3
|
|
18
|
+
from dataclasses import dataclass
|
|
19
|
+
|
|
20
|
+
from codegraph.target_resolver import TargetResolution, TargetType, resolve_target
|
|
21
|
+
|
|
22
|
+
# ---------------------------------------------------------------------------
|
|
23
|
+
# Generic name set — these must never be emitted as bare search terms
|
|
24
|
+
# ---------------------------------------------------------------------------
|
|
25
|
+
|
|
26
|
+
_GENERIC_NAMES: frozenset[str] = frozenset({
|
|
27
|
+
"get", "post", "put", "delete", "patch", "head", "options",
|
|
28
|
+
"run", "call", "start", "stop", "test", "init", "main",
|
|
29
|
+
"handler", "login", "logout", "register", "index", "home",
|
|
30
|
+
"create", "update", "destroy", "show", "list", "detail",
|
|
31
|
+
"load", "save", "close", "open", "reset", "check",
|
|
32
|
+
})
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
# ---------------------------------------------------------------------------
|
|
36
|
+
# Result type
|
|
37
|
+
# ---------------------------------------------------------------------------
|
|
38
|
+
|
|
39
|
+
@dataclass(frozen=True)
|
|
40
|
+
class QueryExpansion:
|
|
41
|
+
original_term: str
|
|
42
|
+
expanded_term: str
|
|
43
|
+
reason: str
|
|
44
|
+
source_symbol: str | None # canonical_id of the source, if any
|
|
45
|
+
qualification_level: str # CANONICAL | QUALIFIED | SHORT | LEXICAL | ROUTE
|
|
46
|
+
confidence: str # HIGH | MEDIUM | LOW
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
# ---------------------------------------------------------------------------
|
|
50
|
+
# Main expansion function
|
|
51
|
+
# ---------------------------------------------------------------------------
|
|
52
|
+
|
|
53
|
+
def expand_query_terms(
|
|
54
|
+
raw_terms: list[str] | tuple[str, ...],
|
|
55
|
+
con: sqlite3.Connection,
|
|
56
|
+
target_resolutions: dict[str, TargetResolution] | None = None,
|
|
57
|
+
max_expansions: int = 8,
|
|
58
|
+
) -> list[QueryExpansion]:
|
|
59
|
+
"""Expand raw target terms to precise, qualified search queries.
|
|
60
|
+
|
|
61
|
+
Args:
|
|
62
|
+
raw_terms: Terms from task_spec.priority_targets or task_spec.targets.
|
|
63
|
+
con: Active SQLite connection to the indexed repository.
|
|
64
|
+
target_resolutions: Pre-computed TargetResolution map (optional).
|
|
65
|
+
If provided, avoids duplicate resolution work.
|
|
66
|
+
max_expansions: Maximum total expansions to emit.
|
|
67
|
+
|
|
68
|
+
Returns:
|
|
69
|
+
List of QueryExpansion records, deduplicated by expanded_term.
|
|
70
|
+
"""
|
|
71
|
+
expansions: list[QueryExpansion] = []
|
|
72
|
+
seen_expanded: set[str] = set()
|
|
73
|
+
resolutions = target_resolutions or {}
|
|
74
|
+
|
|
75
|
+
def _emit(
|
|
76
|
+
original: str,
|
|
77
|
+
expanded: str,
|
|
78
|
+
reason: str,
|
|
79
|
+
source_symbol: str | None,
|
|
80
|
+
level: str,
|
|
81
|
+
confidence: str,
|
|
82
|
+
) -> None:
|
|
83
|
+
if expanded and expanded not in seen_expanded:
|
|
84
|
+
seen_expanded.add(expanded)
|
|
85
|
+
expansions.append(QueryExpansion(
|
|
86
|
+
original_term=original,
|
|
87
|
+
expanded_term=expanded,
|
|
88
|
+
reason=reason,
|
|
89
|
+
source_symbol=source_symbol,
|
|
90
|
+
qualification_level=level,
|
|
91
|
+
confidence=confidence,
|
|
92
|
+
))
|
|
93
|
+
|
|
94
|
+
for term in raw_terms:
|
|
95
|
+
if len(expansions) >= max_expansions:
|
|
96
|
+
break
|
|
97
|
+
|
|
98
|
+
clean = term.strip()
|
|
99
|
+
if not clean or len(clean) < 2:
|
|
100
|
+
continue
|
|
101
|
+
|
|
102
|
+
is_generic = clean.lower() in _GENERIC_NAMES
|
|
103
|
+
|
|
104
|
+
# Use pre-computed resolution if available, otherwise resolve fresh
|
|
105
|
+
resolution = resolutions.get(clean) or resolve_target(clean, con)
|
|
106
|
+
|
|
107
|
+
# ── 1. Canonical ID ───────────────────────────────────────────────
|
|
108
|
+
if resolution.canonical_id and resolution.confidence in ("HIGH", "MEDIUM"):
|
|
109
|
+
_emit(
|
|
110
|
+
clean, resolution.canonical_id,
|
|
111
|
+
f"Resolved canonical ID for '{clean}'",
|
|
112
|
+
resolution.canonical_id,
|
|
113
|
+
"CANONICAL", resolution.confidence,
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
# ── 2. Qualified name (if different from canonical_id) ─────────────
|
|
117
|
+
if (
|
|
118
|
+
resolution.qualified_name
|
|
119
|
+
and resolution.qualified_name != resolution.canonical_id
|
|
120
|
+
and resolution.confidence in ("HIGH", "MEDIUM")
|
|
121
|
+
):
|
|
122
|
+
_emit(
|
|
123
|
+
clean, resolution.qualified_name,
|
|
124
|
+
f"Qualified name resolution for '{clean}'",
|
|
125
|
+
resolution.canonical_id,
|
|
126
|
+
"QUALIFIED", resolution.confidence,
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
# ── 3. Route → handler expansion ──────────────────────────────────
|
|
130
|
+
if resolution.target_type == TargetType.API_ENDPOINT:
|
|
131
|
+
route_rows = con.execute(
|
|
132
|
+
"SELECT handler_name, file_path FROM framework_routes "
|
|
133
|
+
"WHERE route_path=? OR endpoint_id=? LIMIT 3",
|
|
134
|
+
(clean, resolution.canonical_id or clean),
|
|
135
|
+
).fetchall()
|
|
136
|
+
for rr in route_rows:
|
|
137
|
+
handler = rr["handler_name"]
|
|
138
|
+
if handler:
|
|
139
|
+
_emit(
|
|
140
|
+
clean, handler,
|
|
141
|
+
f"Route '{clean}' handler expansion",
|
|
142
|
+
handler, "ROUTE", "HIGH",
|
|
143
|
+
)
|
|
144
|
+
|
|
145
|
+
# ── 4. Generic names: qualify or suppress ──────────────────────────
|
|
146
|
+
if is_generic:
|
|
147
|
+
# We already emitted the qualified form via resolution above.
|
|
148
|
+
# If resolution didn't find a qualified form, search for one.
|
|
149
|
+
if not resolution.is_resolved():
|
|
150
|
+
# Try to find any symbol whose short name == clean
|
|
151
|
+
qrows = con.execute(
|
|
152
|
+
"SELECT canonical_id, qualified_name, name FROM symbols "
|
|
153
|
+
"WHERE name=? ORDER BY kind, path LIMIT 5",
|
|
154
|
+
(clean,),
|
|
155
|
+
).fetchall()
|
|
156
|
+
for qr in qrows:
|
|
157
|
+
qname = qr["qualified_name"]
|
|
158
|
+
cid = qr["canonical_id"]
|
|
159
|
+
if qname and "." in qname:
|
|
160
|
+
# Qualified form exists — emit it, not the bare name
|
|
161
|
+
_emit(
|
|
162
|
+
clean, qname,
|
|
163
|
+
f"Generic name '{clean}' qualified to '{qname}'",
|
|
164
|
+
cid, "QUALIFIED", "MEDIUM",
|
|
165
|
+
)
|
|
166
|
+
# Never emit the bare generic name if qualified form exists
|
|
167
|
+
# If still nothing: skip the bare generic name entirely
|
|
168
|
+
continue # do NOT fall through to SHORT expansion
|
|
169
|
+
|
|
170
|
+
# ── 5. Short name fallback (non-generic) ──────────────────────────
|
|
171
|
+
if not resolution.is_resolved():
|
|
172
|
+
# Use bare term only for non-generic identifiers with reasonable length
|
|
173
|
+
if len(clean) >= 3:
|
|
174
|
+
_emit(
|
|
175
|
+
clean, clean,
|
|
176
|
+
f"Lexical short name for unresolved term '{clean}'",
|
|
177
|
+
None, "LEXICAL", "LOW",
|
|
178
|
+
)
|
|
179
|
+
elif clean not in seen_expanded:
|
|
180
|
+
# Term already resolved but short name wasn't emitted separately
|
|
181
|
+
# — only emit if not generic
|
|
182
|
+
_emit(
|
|
183
|
+
clean, clean,
|
|
184
|
+
f"Short name '{clean}'",
|
|
185
|
+
resolution.canonical_id, "SHORT", "MEDIUM",
|
|
186
|
+
)
|
|
187
|
+
|
|
188
|
+
return expansions[:max_expansions]
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def get_search_queries(expansions: list[QueryExpansion]) -> list[str]:
|
|
192
|
+
"""Extract the deduplicated list of search query strings from expansions."""
|
|
193
|
+
seen: set[str] = set()
|
|
194
|
+
queries: list[str] = []
|
|
195
|
+
for e in expansions:
|
|
196
|
+
if e.expanded_term not in seen:
|
|
197
|
+
seen.add(e.expanded_term)
|
|
198
|
+
queries.append(e.expanded_term)
|
|
199
|
+
return queries
|