codegraph-engine 2.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codegraph/__init__.py +37 -0
- codegraph/agent.py +26 -0
- codegraph/architecture.py +328 -0
- codegraph/audit.py +106 -0
- codegraph/cache.py +95 -0
- codegraph/cli.py +854 -0
- codegraph/config.py +43 -0
- codegraph/constraints.py +238 -0
- codegraph/context.py +1228 -0
- codegraph/epistemic.py +90 -0
- codegraph/errors.py +275 -0
- codegraph/evidence/__init__.py +15 -0
- codegraph/evidence/citations.py +397 -0
- codegraph/frameworks.py +434 -0
- codegraph/freshness.py +295 -0
- codegraph/git.py +278 -0
- codegraph/graph/__init__.py +46 -0
- codegraph/graph/models.py +41 -0
- codegraph/graph/traversal.py +1291 -0
- codegraph/indexing/__init__.py +4 -0
- codegraph/indexing/classifier.py +274 -0
- codegraph/indexing/indexer.py +943 -0
- codegraph/indexing/models.py +338 -0
- codegraph/indexing/parser.py +1240 -0
- codegraph/indexing/scanner.py +200 -0
- codegraph/indexing/test_framework.py +116 -0
- codegraph/interrogation.py +1582 -0
- codegraph/llm/__init__.py +3 -0
- codegraph/llm/base.py +15 -0
- codegraph/llm/context.py +20 -0
- codegraph/mcp/__init__.py +3 -0
- codegraph/mcp/server.py +736 -0
- codegraph/memory/__init__.py +3 -0
- codegraph/memory/store.py +46 -0
- codegraph/models.py +289 -0
- codegraph/observability.py +151 -0
- codegraph/optimizer.py +372 -0
- codegraph/planner.py +417 -0
- codegraph/py.typed +1 -0
- codegraph/query_expansion.py +199 -0
- codegraph/ranking.py +363 -0
- codegraph/resolver.py +843 -0
- codegraph/resources/__init__.py +45 -0
- codegraph/resources/cache.py +117 -0
- codegraph/resources/coalescer.py +83 -0
- codegraph/resources/debouncer.py +98 -0
- codegraph/resources/governor.py +232 -0
- codegraph/resources/policy.py +123 -0
- codegraph/retrieval_policy.py +220 -0
- codegraph/search/__init__.py +23 -0
- codegraph/search/hybrid.py +301 -0
- codegraph/search/semantic.py +28 -0
- codegraph/security/__init__.py +3 -0
- codegraph/security/paths.py +35 -0
- codegraph/target_resolver.py +348 -0
- codegraph/task.py +637 -0
- codegraph_engine-2.1.1.dist-info/METADATA +334 -0
- codegraph_engine-2.1.1.dist-info/RECORD +62 -0
- codegraph_engine-2.1.1.dist-info/WHEEL +5 -0
- codegraph_engine-2.1.1.dist-info/entry_points.txt +2 -0
- codegraph_engine-2.1.1.dist-info/licenses/LICENSE +21 -0
- codegraph_engine-2.1.1.dist-info/top_level.txt +1 -0
codegraph/context.py
ADDED
|
@@ -0,0 +1,1228 @@
|
|
|
1
|
+
"""Context Compiler — flag-ship orchestration engine for CodeGraph MCP.
|
|
2
|
+
|
|
3
|
+
Translates TaskSpec and RetrievalPlan into a verified, bounded, coverage-aware,
|
|
4
|
+
and deduplicated ContextPacket with full provenance and zero LLM API dependency.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import concurrent.futures
|
|
9
|
+
import os
|
|
10
|
+
import sqlite3
|
|
11
|
+
import time
|
|
12
|
+
from dataclasses import field
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from typing import Any, Literal
|
|
15
|
+
|
|
16
|
+
from pydantic import BaseModel
|
|
17
|
+
|
|
18
|
+
from codegraph.architecture import get_architecture
|
|
19
|
+
from codegraph.cache import (
|
|
20
|
+
compute_context_cache_key,
|
|
21
|
+
get_cached_context_packet,
|
|
22
|
+
store_cached_context_packet,
|
|
23
|
+
)
|
|
24
|
+
from codegraph.constraints import ConstraintGuard
|
|
25
|
+
from codegraph.freshness import (
|
|
26
|
+
FreshnessStatus,
|
|
27
|
+
check_freshness,
|
|
28
|
+
index_generation,
|
|
29
|
+
)
|
|
30
|
+
from codegraph.git import changed_files, is_git_repository, recent_commits
|
|
31
|
+
from codegraph.graph import GraphSeedPolicy
|
|
32
|
+
from codegraph.graph.traversal import (
|
|
33
|
+
find_callees,
|
|
34
|
+
find_callers,
|
|
35
|
+
find_parallel_implementations,
|
|
36
|
+
find_related_tests,
|
|
37
|
+
)
|
|
38
|
+
from codegraph.observability import (
|
|
39
|
+
ExecutionMetadata,
|
|
40
|
+
ExecutionTiming,
|
|
41
|
+
OperationMetric,
|
|
42
|
+
Timer,
|
|
43
|
+
create_request_id,
|
|
44
|
+
get_global_metrics,
|
|
45
|
+
)
|
|
46
|
+
from codegraph.optimizer import (
|
|
47
|
+
CandidateContextItem,
|
|
48
|
+
optimize_context_budget,
|
|
49
|
+
)
|
|
50
|
+
from codegraph.planner import RetrievalPlan, build_retrieval_plan
|
|
51
|
+
from codegraph.query_expansion import QueryExpansion, expand_query_terms, get_search_queries
|
|
52
|
+
from codegraph.ranking import RankedItem, rank
|
|
53
|
+
from codegraph.resources import (
|
|
54
|
+
TaskPriority,
|
|
55
|
+
get_global_coalescer,
|
|
56
|
+
get_global_governor,
|
|
57
|
+
get_process_memory_mb,
|
|
58
|
+
)
|
|
59
|
+
from codegraph.retrieval_policy import RetrievalPolicy, get_retrieval_policy
|
|
60
|
+
from codegraph.search import search
|
|
61
|
+
from codegraph.target_resolver import TargetResolution, resolve_targets
|
|
62
|
+
from codegraph.task import (
|
|
63
|
+
TaskSpec,
|
|
64
|
+
canonicalize_intent,
|
|
65
|
+
normalize_task_spec,
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
SCHEMA_VERSION = "2.0"
|
|
69
|
+
|
|
70
|
+
Intent = Literal[
|
|
71
|
+
"explain", "debug", "modify", "review", "test", "trace", "impact", "architecture"
|
|
72
|
+
]
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
# ---------------------------------------------------------------------------
|
|
76
|
+
# ContextPacket Models
|
|
77
|
+
# ---------------------------------------------------------------------------
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class RankingReasonModel(BaseModel):
|
|
81
|
+
code: str
|
|
82
|
+
label: str
|
|
83
|
+
contribution: float
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
class SymbolRef(BaseModel):
|
|
87
|
+
symbol: str
|
|
88
|
+
file: str
|
|
89
|
+
kind: str
|
|
90
|
+
start_line: int
|
|
91
|
+
end_line: int
|
|
92
|
+
score: float
|
|
93
|
+
reasons: list[RankingReasonModel] = []
|
|
94
|
+
canonical_id: str | None = None
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
class Relationship(BaseModel):
|
|
98
|
+
source: str
|
|
99
|
+
target: str
|
|
100
|
+
relationship: str
|
|
101
|
+
confidence: str
|
|
102
|
+
file: str | None = None
|
|
103
|
+
start_line: int | None = None
|
|
104
|
+
evidence: str = ""
|
|
105
|
+
status: str = "FACT" # FACT | ASSUMPTION | INFERENCE | UNKNOWN | CONFLICT
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
class TestRef(BaseModel):
|
|
109
|
+
file: str
|
|
110
|
+
symbol: str | None
|
|
111
|
+
relationship: str
|
|
112
|
+
confidence: str
|
|
113
|
+
evidence: str
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
class FileRef(BaseModel):
|
|
117
|
+
file: str
|
|
118
|
+
score: float
|
|
119
|
+
reasons: list[RankingReasonModel] = []
|
|
120
|
+
snippet: str
|
|
121
|
+
token_estimate: int
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
class EvidenceRef(BaseModel):
|
|
125
|
+
file: str
|
|
126
|
+
start_line: int
|
|
127
|
+
end_line: int
|
|
128
|
+
symbol: str | None
|
|
129
|
+
snippet: str
|
|
130
|
+
confidence: str
|
|
131
|
+
source_hash: str | None = None
|
|
132
|
+
evidence_status: str = "current" # current | stale
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
class ContextPacket(BaseModel):
|
|
136
|
+
schema_version: str = SCHEMA_VERSION
|
|
137
|
+
task: str
|
|
138
|
+
intent: str | None
|
|
139
|
+
repository: str
|
|
140
|
+
indexed_commit: str | None
|
|
141
|
+
current_commit: str | None
|
|
142
|
+
freshness: str # FRESH | STALE | PARTIALLY_STALE | UNKNOWN
|
|
143
|
+
freshness_detail: str
|
|
144
|
+
|
|
145
|
+
# Production intelligence extensions
|
|
146
|
+
repository_generation: int = 0
|
|
147
|
+
task_spec: dict[str, object] | None = None
|
|
148
|
+
task_fingerprint: str = ""
|
|
149
|
+
budget: dict[str, object] | None = None
|
|
150
|
+
execution: dict[str, object] | None = None
|
|
151
|
+
mode: str = "BALANCED" # FAST | BALANCED | DEEP
|
|
152
|
+
entry_points: list[dict[str, object]] = []
|
|
153
|
+
framework_facts: list[dict[str, object]] = []
|
|
154
|
+
architecture: list[dict[str, object]] = []
|
|
155
|
+
git_changes: list[dict[str, object]] = []
|
|
156
|
+
git_facts: list[dict[str, object]] = []
|
|
157
|
+
assumptions: list[str] = []
|
|
158
|
+
unknowns: list[str] = []
|
|
159
|
+
conflicts: list[str] = []
|
|
160
|
+
|
|
161
|
+
# Knowledge
|
|
162
|
+
facts: list[dict[str, object]] = field(default_factory=list)
|
|
163
|
+
symbols: list[SymbolRef] = []
|
|
164
|
+
relationships: list[Relationship] = []
|
|
165
|
+
tests: list[TestRef] = []
|
|
166
|
+
files: list[FileRef] = []
|
|
167
|
+
evidence: list[EvidenceRef] = []
|
|
168
|
+
uncertainties: list[str] = []
|
|
169
|
+
|
|
170
|
+
# Token budget
|
|
171
|
+
candidate_token_estimate: int = 0
|
|
172
|
+
selected_token_estimate: int = 0
|
|
173
|
+
context_reduction_pct: float = 0.0
|
|
174
|
+
selected_files: list[str] = []
|
|
175
|
+
discarded_files: list[str] = []
|
|
176
|
+
|
|
177
|
+
model_config = {"arbitrary_types_allowed": True}
|
|
178
|
+
|
|
179
|
+
def as_dict(self) -> dict[str, object]:
|
|
180
|
+
return self.model_dump()
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
# ---------------------------------------------------------------------------
|
|
184
|
+
# Token estimation helper
|
|
185
|
+
# ---------------------------------------------------------------------------
|
|
186
|
+
|
|
187
|
+
_CHARS_PER_TOKEN = 4
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def _token_estimate(text: str) -> int:
|
|
191
|
+
return max(1, len(text) // _CHARS_PER_TOKEN)
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
# ---------------------------------------------------------------------------
|
|
195
|
+
# Main Context Compiler
|
|
196
|
+
# ---------------------------------------------------------------------------
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def get_context(
|
|
200
|
+
con: sqlite3.Connection,
|
|
201
|
+
repository: Path,
|
|
202
|
+
task: str | TaskSpec | dict[str, Any],
|
|
203
|
+
intent: Intent | str | None = None,
|
|
204
|
+
max_tokens: int = 20_000,
|
|
205
|
+
top_k: int = 15,
|
|
206
|
+
plan: RetrievalPlan | dict[str, Any] | None = None,
|
|
207
|
+
mode: str = "BALANCED", # FAST | BALANCED | DEEP
|
|
208
|
+
explain: bool = False,
|
|
209
|
+
) -> ContextPacket:
|
|
210
|
+
"""Compile a deterministic, verified ContextPacket for the given task and intent."""
|
|
211
|
+
governor = get_global_governor()
|
|
212
|
+
coalescer = get_global_coalescer()
|
|
213
|
+
flight_key = f"{repository}:{task}:{intent}:{max_tokens}:{top_k}:{mode}:{explain}"
|
|
214
|
+
|
|
215
|
+
def _execute() -> ContextPacket:
|
|
216
|
+
with governor.task_scope(TaskPriority.INTERACTIVE_HIGH):
|
|
217
|
+
return _get_context_impl(
|
|
218
|
+
con=con,
|
|
219
|
+
repository=repository,
|
|
220
|
+
task=task,
|
|
221
|
+
intent=intent,
|
|
222
|
+
max_tokens=max_tokens,
|
|
223
|
+
top_k=top_k,
|
|
224
|
+
plan=plan,
|
|
225
|
+
mode=mode,
|
|
226
|
+
explain=explain,
|
|
227
|
+
governor=governor,
|
|
228
|
+
)
|
|
229
|
+
|
|
230
|
+
return coalescer.coalesce(flight_key, _execute)
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def _get_context_impl(
|
|
234
|
+
con: sqlite3.Connection,
|
|
235
|
+
repository: Path,
|
|
236
|
+
task: str | TaskSpec | dict[str, Any],
|
|
237
|
+
intent: Intent | str | None = None,
|
|
238
|
+
max_tokens: int = 20_000,
|
|
239
|
+
top_k: int = 15,
|
|
240
|
+
plan: RetrievalPlan | dict[str, Any] | None = None,
|
|
241
|
+
mode: str = "BALANCED", # FAST | BALANCED | DEEP
|
|
242
|
+
explain: bool = False,
|
|
243
|
+
governor: Any = None,
|
|
244
|
+
) -> ContextPacket:
|
|
245
|
+
if governor is None:
|
|
246
|
+
governor = get_global_governor()
|
|
247
|
+
req_id = create_request_id()
|
|
248
|
+
timing = ExecutionTiming()
|
|
249
|
+
with Timer() as timer:
|
|
250
|
+
# 0. Latency Mode configuration
|
|
251
|
+
mode_upper = (mode or "BALANCED").upper()
|
|
252
|
+
if mode_upper not in ("FAST", "BALANCED", "DEEP"):
|
|
253
|
+
mode_upper = "BALANCED"
|
|
254
|
+
|
|
255
|
+
if mode_upper == "FAST":
|
|
256
|
+
effective_top_k = min(top_k, 5)
|
|
257
|
+
max_graph_syms = 2
|
|
258
|
+
max_graph_results = 5
|
|
259
|
+
max_test_syms = 2
|
|
260
|
+
max_test_results = 3
|
|
261
|
+
git_commits_n = 2
|
|
262
|
+
elif mode_upper == "DEEP":
|
|
263
|
+
effective_top_k = max(top_k, 25)
|
|
264
|
+
max_graph_syms = 10
|
|
265
|
+
max_graph_results = 15
|
|
266
|
+
max_test_syms = 8
|
|
267
|
+
max_test_results = 8
|
|
268
|
+
git_commits_n = 10
|
|
269
|
+
else: # BALANCED
|
|
270
|
+
effective_top_k = top_k
|
|
271
|
+
max_graph_syms = 6
|
|
272
|
+
max_graph_results = 10
|
|
273
|
+
max_test_syms = 4
|
|
274
|
+
max_test_results = 5
|
|
275
|
+
git_commits_n = 5
|
|
276
|
+
|
|
277
|
+
# 1. Planning stage
|
|
278
|
+
t_plan_start = time.perf_counter()
|
|
279
|
+
task_spec, ambiguities = normalize_task_spec(task, con)
|
|
280
|
+
|
|
281
|
+
caller_raw_intent = intent
|
|
282
|
+
if caller_raw_intent:
|
|
283
|
+
canonical_intent = canonicalize_intent(str(caller_raw_intent))
|
|
284
|
+
task_spec = TaskSpec(
|
|
285
|
+
schema_version=task_spec.schema_version,
|
|
286
|
+
raw_prompt=task_spec.raw_prompt,
|
|
287
|
+
intent=canonical_intent.value,
|
|
288
|
+
goal=task_spec.goal,
|
|
289
|
+
targets=task_spec.targets,
|
|
290
|
+
entities=task_spec.entities,
|
|
291
|
+
operations=task_spec.operations,
|
|
292
|
+
constraints=task_spec.constraints,
|
|
293
|
+
exclusions=task_spec.exclusions,
|
|
294
|
+
scope_paths=task_spec.scope_paths,
|
|
295
|
+
scope_modules=task_spec.scope_modules,
|
|
296
|
+
frameworks=task_spec.frameworks,
|
|
297
|
+
time_scope=task_spec.time_scope,
|
|
298
|
+
priority_targets=task_spec.priority_targets,
|
|
299
|
+
ambiguities=task_spec.ambiguities,
|
|
300
|
+
assumptions=task_spec.assumptions,
|
|
301
|
+
unknowns=task_spec.unknowns,
|
|
302
|
+
confidence=task_spec.confidence,
|
|
303
|
+
)
|
|
304
|
+
|
|
305
|
+
task_display_str = (
|
|
306
|
+
task if isinstance(task, str) else (task_spec.raw_prompt or task_spec.goal or "task")
|
|
307
|
+
)
|
|
308
|
+
|
|
309
|
+
gen = index_generation(con)
|
|
310
|
+
cache_key = compute_context_cache_key(
|
|
311
|
+
task_spec=task_spec,
|
|
312
|
+
repository_generation=gen,
|
|
313
|
+
max_tokens=max_tokens,
|
|
314
|
+
extra_config=f"{caller_raw_intent or ''}:{mode_upper}",
|
|
315
|
+
)
|
|
316
|
+
|
|
317
|
+
# 2. Check Cache
|
|
318
|
+
cached_data = get_cached_context_packet(con, cache_key)
|
|
319
|
+
if cached_data is not None:
|
|
320
|
+
try:
|
|
321
|
+
packet = ContextPacket.model_validate(cached_data)
|
|
322
|
+
timing.total_ms = timer.elapsed_ms
|
|
323
|
+
exec_meta = ExecutionMetadata(
|
|
324
|
+
latency_ms=timer.elapsed_ms,
|
|
325
|
+
cache_hit=True,
|
|
326
|
+
index_generation=gen,
|
|
327
|
+
candidate_count=packet.candidate_token_estimate,
|
|
328
|
+
selected_count=packet.selected_token_estimate,
|
|
329
|
+
mode=mode_upper,
|
|
330
|
+
timing=timing,
|
|
331
|
+
)
|
|
332
|
+
packet.execution = exec_meta.as_dict()
|
|
333
|
+
packet.execution["resource"] = {
|
|
334
|
+
"profile": governor.policy.profile.value,
|
|
335
|
+
"pressure": governor.get_pressure().value,
|
|
336
|
+
"activity": governor.get_activity_mode().value,
|
|
337
|
+
"estimated_memory_mb": get_process_memory_mb(),
|
|
338
|
+
}
|
|
339
|
+
packet.mode = mode_upper
|
|
340
|
+
get_global_metrics().record(
|
|
341
|
+
OperationMetric(
|
|
342
|
+
request_id=req_id,
|
|
343
|
+
tool_or_op="get_context",
|
|
344
|
+
latency_ms=timer.elapsed_ms,
|
|
345
|
+
index_generation=gen,
|
|
346
|
+
cache_hit=True,
|
|
347
|
+
candidate_count=packet.candidate_token_estimate,
|
|
348
|
+
selected_count=packet.selected_token_estimate,
|
|
349
|
+
estimated_tokens=packet.selected_token_estimate,
|
|
350
|
+
task_intent=task_spec.intent,
|
|
351
|
+
freshness=packet.freshness,
|
|
352
|
+
)
|
|
353
|
+
)
|
|
354
|
+
return packet
|
|
355
|
+
except Exception:
|
|
356
|
+
pass
|
|
357
|
+
|
|
358
|
+
if plan is None:
|
|
359
|
+
retrieval_plan = build_retrieval_plan(
|
|
360
|
+
task_spec=task_spec,
|
|
361
|
+
ambiguities=ambiguities,
|
|
362
|
+
con=con,
|
|
363
|
+
token_budget=max_tokens,
|
|
364
|
+
)
|
|
365
|
+
elif isinstance(plan, dict):
|
|
366
|
+
retrieval_plan = build_retrieval_plan(
|
|
367
|
+
task_spec=task_spec,
|
|
368
|
+
ambiguities=ambiguities,
|
|
369
|
+
con=con,
|
|
370
|
+
token_budget=int(plan.get("token_budget", max_tokens)),
|
|
371
|
+
)
|
|
372
|
+
else:
|
|
373
|
+
retrieval_plan = plan
|
|
374
|
+
timing.planning_ms = (time.perf_counter() - t_plan_start) * 1000
|
|
375
|
+
|
|
376
|
+
# ── v2.1: First-class target resolution + qualified query expansion ──
|
|
377
|
+
# Resolve every target in the spec against the indexed repository before
|
|
378
|
+
# any retrieval work. This drives qualified, precise search queries and
|
|
379
|
+
# prevents generic method names from polluting search results.
|
|
380
|
+
_raw_targets = list(task_spec.priority_targets or task_spec.targets)
|
|
381
|
+
_target_resolutions: dict[str, TargetResolution] = {}
|
|
382
|
+
_retrieval_policy: RetrievalPolicy = get_retrieval_policy(task_spec.intent)
|
|
383
|
+
if _raw_targets and con:
|
|
384
|
+
_target_resolutions = resolve_targets(_raw_targets, con)
|
|
385
|
+
_expanded_queries_obj: list[QueryExpansion] = expand_query_terms(
|
|
386
|
+
_raw_targets,
|
|
387
|
+
con,
|
|
388
|
+
target_resolutions=_target_resolutions,
|
|
389
|
+
max_expansions=10,
|
|
390
|
+
)
|
|
391
|
+
_v21_queries: list[str] = get_search_queries(_expanded_queries_obj)
|
|
392
|
+
# Merge v2.1 qualified queries with planner queries (prefer qualified)
|
|
393
|
+
_planner_queries: list[str] = [
|
|
394
|
+
q for group in retrieval_plan.query_groups for q in group.queries
|
|
395
|
+
]
|
|
396
|
+
# Deduplicated, qualified-first merge
|
|
397
|
+
_merged_queries: list[str] = []
|
|
398
|
+
_seen_q: set[str] = set()
|
|
399
|
+
for q in _v21_queries + _planner_queries:
|
|
400
|
+
if q and q not in _seen_q:
|
|
401
|
+
_seen_q.add(q)
|
|
402
|
+
_merged_queries.append(q)
|
|
403
|
+
# ────────────────────────────────────────────────────────────────────
|
|
404
|
+
|
|
405
|
+
# 3. Freshness check
|
|
406
|
+
freshness_report = check_freshness(repository, con)
|
|
407
|
+
stale_paths = set(freshness_report.modified_files + freshness_report.deleted_files)
|
|
408
|
+
|
|
409
|
+
# 4. Check whether database is file-backed for multi-threaded parallel read
|
|
410
|
+
db_file_path: str | None = None
|
|
411
|
+
try:
|
|
412
|
+
for r in con.execute("PRAGMA database_list").fetchall():
|
|
413
|
+
if r[1] == "main" and r[2]:
|
|
414
|
+
db_file_path = str(r[2])
|
|
415
|
+
break
|
|
416
|
+
except Exception:
|
|
417
|
+
pass
|
|
418
|
+
is_disk_db = db_file_path is not None and Path(db_file_path).is_file()
|
|
419
|
+
|
|
420
|
+
queries_to_run: list[str] = _merged_queries if _merged_queries else [
|
|
421
|
+
q for group in retrieval_plan.query_groups for q in group.queries
|
|
422
|
+
]
|
|
423
|
+
|
|
424
|
+
candidate_dicts: list[dict[str, object]] = []
|
|
425
|
+
target_symbols: set[str] = set()
|
|
426
|
+
for t in task_spec.targets:
|
|
427
|
+
target_symbols.add(t)
|
|
428
|
+
target_symbols.add(t.split(".")[-1])
|
|
429
|
+
# Also add canonical IDs from target resolutions as target symbols
|
|
430
|
+
for res in _target_resolutions.values():
|
|
431
|
+
if res.canonical_id:
|
|
432
|
+
target_symbols.add(res.canonical_id)
|
|
433
|
+
target_symbols.add(res.canonical_id.split(".")[-1])
|
|
434
|
+
if res.qualified_name:
|
|
435
|
+
target_symbols.add(res.qualified_name)
|
|
436
|
+
|
|
437
|
+
# Parallel / Sequential retrieval worker helpers
|
|
438
|
+
def _exec_search(path: str | None, queries: list[str]) -> list[dict[str, object]]:
|
|
439
|
+
items: list[dict[str, object]] = []
|
|
440
|
+
if path:
|
|
441
|
+
c = sqlite3.connect(f"file:{path}?mode=ro", uri=True)
|
|
442
|
+
c.row_factory = sqlite3.Row
|
|
443
|
+
try:
|
|
444
|
+
for query in queries:
|
|
445
|
+
for item in search(c, query, top_k=effective_top_k):
|
|
446
|
+
items.append(item.as_dict())
|
|
447
|
+
finally:
|
|
448
|
+
c.close()
|
|
449
|
+
else:
|
|
450
|
+
for query in queries:
|
|
451
|
+
for item in search(con, query, top_k=effective_top_k):
|
|
452
|
+
items.append(item.as_dict())
|
|
453
|
+
return items
|
|
454
|
+
|
|
455
|
+
def _exec_routes(path: str | None) -> list[Any]:
|
|
456
|
+
try:
|
|
457
|
+
if path:
|
|
458
|
+
c = sqlite3.connect(f"file:{path}?mode=ro", uri=True)
|
|
459
|
+
c.row_factory = sqlite3.Row
|
|
460
|
+
try:
|
|
461
|
+
return c.execute(
|
|
462
|
+
"SELECT route_path, http_method, handler_name, file_path, line, endpoint_id FROM framework_routes"
|
|
463
|
+
).fetchall()
|
|
464
|
+
finally:
|
|
465
|
+
c.close()
|
|
466
|
+
else:
|
|
467
|
+
return con.execute(
|
|
468
|
+
"SELECT route_path, http_method, handler_name, file_path, line, endpoint_id FROM framework_routes"
|
|
469
|
+
).fetchall()
|
|
470
|
+
except Exception:
|
|
471
|
+
return []
|
|
472
|
+
|
|
473
|
+
def _exec_git(repo: Path, include_git: bool, commit_limit: int) -> tuple[list[dict[str, Any]], set[str]]:
|
|
474
|
+
changes: list[dict[str, Any]] = []
|
|
475
|
+
recent: set[str] = set()
|
|
476
|
+
try:
|
|
477
|
+
if is_git_repository(repo):
|
|
478
|
+
diffs = changed_files(repo, since="HEAD~10", until="HEAD")
|
|
479
|
+
for d in diffs:
|
|
480
|
+
recent.add(d.path)
|
|
481
|
+
changes.append(d.as_dict())
|
|
482
|
+
if include_git:
|
|
483
|
+
for commit in recent_commits(repo, n=commit_limit):
|
|
484
|
+
changes.append(commit.as_dict())
|
|
485
|
+
except Exception:
|
|
486
|
+
pass
|
|
487
|
+
return changes, recent
|
|
488
|
+
|
|
489
|
+
include_arch = retrieval_plan.include_architecture or (mode_upper == "DEEP")
|
|
490
|
+
def _exec_arch(path: str | None, repo: Path) -> list[dict[str, object]]:
|
|
491
|
+
archs: list[dict[str, object]] = []
|
|
492
|
+
try:
|
|
493
|
+
if path:
|
|
494
|
+
c = sqlite3.connect(f"file:{path}?mode=ro", uri=True)
|
|
495
|
+
c.row_factory = sqlite3.Row
|
|
496
|
+
try:
|
|
497
|
+
archs.append(get_architecture(c, repo))
|
|
498
|
+
finally:
|
|
499
|
+
c.close()
|
|
500
|
+
else:
|
|
501
|
+
archs.append(get_architecture(con, repo))
|
|
502
|
+
except Exception:
|
|
503
|
+
pass
|
|
504
|
+
return archs
|
|
505
|
+
|
|
506
|
+
# Execute searches, framework routes, git, and architecture
|
|
507
|
+
t_stage2_start = time.perf_counter()
|
|
508
|
+
if is_disk_db:
|
|
509
|
+
with concurrent.futures.ThreadPoolExecutor(max_workers=min(4, os.cpu_count() or 4)) as executor:
|
|
510
|
+
f_search = executor.submit(_exec_search, db_file_path, queries_to_run)
|
|
511
|
+
f_routes = executor.submit(_exec_routes, db_file_path)
|
|
512
|
+
f_git = executor.submit(_exec_git, repository, retrieval_plan.include_git, git_commits_n)
|
|
513
|
+
f_arch = executor.submit(_exec_arch, db_file_path, repository) if include_arch else None
|
|
514
|
+
|
|
515
|
+
search_res = f_search.result()
|
|
516
|
+
routes_res = f_routes.result()
|
|
517
|
+
git_changes_out, recent_paths = f_git.result()
|
|
518
|
+
architecture_out = f_arch.result() if f_arch else []
|
|
519
|
+
else:
|
|
520
|
+
search_res = _exec_search(None, queries_to_run)
|
|
521
|
+
routes_res = _exec_routes(None)
|
|
522
|
+
git_changes_out, recent_paths = _exec_git(repository, retrieval_plan.include_git, git_commits_n)
|
|
523
|
+
architecture_out = _exec_arch(None, repository) if include_arch else []
|
|
524
|
+
|
|
525
|
+
stage2_duration = (time.perf_counter() - t_stage2_start) * 1000
|
|
526
|
+
timing.search_ms = stage2_duration * 0.4
|
|
527
|
+
timing.framework_ms = stage2_duration * 0.2
|
|
528
|
+
timing.git_ms = stage2_duration * 0.4
|
|
529
|
+
|
|
530
|
+
for r_dict in search_res:
|
|
531
|
+
candidate_dicts.append(r_dict)
|
|
532
|
+
sym = r_dict.get("symbol")
|
|
533
|
+
if sym and isinstance(sym, str):
|
|
534
|
+
target_symbols.add(sym)
|
|
535
|
+
target_symbols.add(sym.split(".")[-1])
|
|
536
|
+
|
|
537
|
+
stage1_candidate_count = len(candidate_dicts)
|
|
538
|
+
|
|
539
|
+
# Process framework routes
|
|
540
|
+
route_relationships: list[Relationship] = []
|
|
541
|
+
entry_points_out: list[dict[str, object]] = []
|
|
542
|
+
framework_facts_out: list[dict[str, object]] = []
|
|
543
|
+
|
|
544
|
+
for r in routes_res:
|
|
545
|
+
r_path = r["route_path"]
|
|
546
|
+
r_method = r["http_method"]
|
|
547
|
+
ep_id = r["endpoint_id"] or f"{r_method} {r_path}"
|
|
548
|
+
matches_task = (
|
|
549
|
+
r_path in task_display_str
|
|
550
|
+
or ep_id.lower() in task_display_str.lower()
|
|
551
|
+
or (r_method in task_display_str.upper() and r_path in task_display_str)
|
|
552
|
+
or any(
|
|
553
|
+
(len(tgt) >= 3 and (tgt.lower() == r["handler_name"].lower() or tgt.lower() in r_path.lower()))
|
|
554
|
+
for tgt in task_spec.targets
|
|
555
|
+
)
|
|
556
|
+
)
|
|
557
|
+
if matches_task:
|
|
558
|
+
target_symbols.add(r["handler_name"])
|
|
559
|
+
r_line = r["line"] or 1
|
|
560
|
+
candidate_dicts.append(
|
|
561
|
+
{
|
|
562
|
+
"file": r["file_path"],
|
|
563
|
+
"symbol": r["handler_name"],
|
|
564
|
+
"start_line": r_line,
|
|
565
|
+
"end_line": r_line,
|
|
566
|
+
"score": 0.95,
|
|
567
|
+
"snippet": f"Endpoint: {ep_id} handled by {r['handler_name']}",
|
|
568
|
+
}
|
|
569
|
+
)
|
|
570
|
+
route_rel = Relationship(
|
|
571
|
+
source=ep_id,
|
|
572
|
+
target=r["handler_name"],
|
|
573
|
+
relationship="HANDLED_BY",
|
|
574
|
+
confidence="HIGH",
|
|
575
|
+
file=r["file_path"],
|
|
576
|
+
start_line=r_line,
|
|
577
|
+
evidence=f"Route definition for {ep_id}",
|
|
578
|
+
status="FACT",
|
|
579
|
+
)
|
|
580
|
+
route_relationships.append(route_rel)
|
|
581
|
+
entry_points_out.append(
|
|
582
|
+
{
|
|
583
|
+
"endpoint_id": ep_id,
|
|
584
|
+
"route_path": r_path,
|
|
585
|
+
"http_method": r_method,
|
|
586
|
+
"handler": r["handler_name"],
|
|
587
|
+
"file": r["file_path"],
|
|
588
|
+
"line": r_line,
|
|
589
|
+
}
|
|
590
|
+
)
|
|
591
|
+
framework_facts_out.append(
|
|
592
|
+
{
|
|
593
|
+
"construct": "ROUTE",
|
|
594
|
+
"endpoint_id": ep_id,
|
|
595
|
+
"handler": r["handler_name"],
|
|
596
|
+
"file": r["file_path"],
|
|
597
|
+
"line": r_line,
|
|
598
|
+
}
|
|
599
|
+
)
|
|
600
|
+
try:
|
|
601
|
+
dec_rows = con.execute(
|
|
602
|
+
"SELECT decorators, canonical_id, qualified_name, start_line FROM symbols "
|
|
603
|
+
"WHERE (name=? OR qualified_name=? OR canonical_id=?) AND path=?",
|
|
604
|
+
(r["handler_name"], r["handler_name"], r["handler_name"], r["file_path"]),
|
|
605
|
+
).fetchall()
|
|
606
|
+
for d_row in dec_rows:
|
|
607
|
+
if d_row["decorators"]:
|
|
608
|
+
for dec in str(d_row["decorators"]).split(","):
|
|
609
|
+
clean_dec = dec.strip()
|
|
610
|
+
if clean_dec:
|
|
611
|
+
fmt_dec = f"@{clean_dec}" if not clean_dec.startswith("@") else clean_dec
|
|
612
|
+
framework_facts_out.append(
|
|
613
|
+
{
|
|
614
|
+
"framework": str(r["framework"] if "framework" in r.keys() else "framework"),
|
|
615
|
+
"decorator": fmt_dec,
|
|
616
|
+
"route": r_path,
|
|
617
|
+
"handler": r["handler_name"],
|
|
618
|
+
"canonical_id": str(d_row["canonical_id"]),
|
|
619
|
+
"file": r["file_path"],
|
|
620
|
+
"line": int(d_row["start_line"]),
|
|
621
|
+
"confidence": "HIGH",
|
|
622
|
+
"evidence": f"Decorator {fmt_dec} on {d_row['qualified_name']}",
|
|
623
|
+
}
|
|
624
|
+
)
|
|
625
|
+
except Exception:
|
|
626
|
+
pass
|
|
627
|
+
|
|
628
|
+
# Instantiate ConstraintGuard for hard safety invariants
|
|
629
|
+
_guard = ConstraintGuard.from_task_spec(task_spec)
|
|
630
|
+
|
|
631
|
+
# 5. Graph exploration (callers/callees) & Tests with deterministic seeds
|
|
632
|
+
t_graph_start = time.perf_counter()
|
|
633
|
+
|
|
634
|
+
# Build grounded, deterministic graph roots in priority order:
|
|
635
|
+
# 1. canonical targets from TargetResolver
|
|
636
|
+
# 2. exact qualified targets
|
|
637
|
+
# 3. route handlers
|
|
638
|
+
# 4. priority targets / task targets
|
|
639
|
+
ordered_seeds: list[str] = []
|
|
640
|
+
_seen_seeds: set[str] = set()
|
|
641
|
+
|
|
642
|
+
for res in _target_resolutions.values():
|
|
643
|
+
if res.canonical_id and _guard.allows_canonical_id(res.canonical_id) and res.canonical_id not in _seen_seeds:
|
|
644
|
+
_seen_seeds.add(res.canonical_id)
|
|
645
|
+
ordered_seeds.append(res.canonical_id)
|
|
646
|
+
if res.qualified_name and _guard.allows_symbol(res.qualified_name) and res.qualified_name not in _seen_seeds:
|
|
647
|
+
_seen_seeds.add(res.qualified_name)
|
|
648
|
+
ordered_seeds.append(res.qualified_name)
|
|
649
|
+
|
|
650
|
+
for ep in entry_points_out:
|
|
651
|
+
h = str(ep.get("handler", ""))
|
|
652
|
+
if h and _guard.allows_symbol(h) and h not in _seen_seeds:
|
|
653
|
+
_seen_seeds.add(h)
|
|
654
|
+
ordered_seeds.append(h)
|
|
655
|
+
|
|
656
|
+
for sym in list(task_spec.priority_targets or task_spec.targets):
|
|
657
|
+
if sym and _guard.allows_symbol(sym) and sym not in _seen_seeds:
|
|
658
|
+
_seen_seeds.add(sym)
|
|
659
|
+
ordered_seeds.append(sym)
|
|
660
|
+
|
|
661
|
+
# Include top search results as seeds when grounded symbols are found
|
|
662
|
+
for s_item in search_res[:3]:
|
|
663
|
+
s_sym = str(s_item.get("symbol") or "")
|
|
664
|
+
if s_sym and _guard.allows_symbol(s_sym) and s_sym not in _seen_seeds:
|
|
665
|
+
_seen_seeds.add(s_sym)
|
|
666
|
+
ordered_seeds.append(s_sym)
|
|
667
|
+
|
|
668
|
+
# Bounded seed policy
|
|
669
|
+
graph_seed_policy = GraphSeedPolicy(
|
|
670
|
+
source_type="EXPLICIT_TARGET",
|
|
671
|
+
min_confidence="HIGH",
|
|
672
|
+
max_seeds=max_graph_syms,
|
|
673
|
+
max_depth=_retrieval_policy.max_graph_depth,
|
|
674
|
+
allowed_edges=_retrieval_policy.allowed_relationship_types,
|
|
675
|
+
)
|
|
676
|
+
|
|
677
|
+
for sym in ordered_seeds[:graph_seed_policy.max_seeds]:
|
|
678
|
+
short = sym.split(".")[-1]
|
|
679
|
+
if _retrieval_policy.include_callers and (
|
|
680
|
+
retrieval_plan.entry_point_strategy in ("TRACE_FROM_ENDPOINT", "TRACE_HIERARCHY")
|
|
681
|
+
or task_spec.intent in ("DEBUG", "TRACE", "IMPACT", "CHANGE", "REVIEW")
|
|
682
|
+
):
|
|
683
|
+
callers = find_callers(con, short, max_results=max_graph_results)
|
|
684
|
+
for c in callers:
|
|
685
|
+
caller_sym = c.get("symbol")
|
|
686
|
+
rel = str(c.get("relationship", "CALLS"))
|
|
687
|
+
if not _retrieval_policy.allows(rel):
|
|
688
|
+
continue
|
|
689
|
+
if not _guard.allows_symbol(str(caller_sym) if caller_sym else None):
|
|
690
|
+
continue
|
|
691
|
+
if not _guard.allows_file(str(c.get("file", ""))):
|
|
692
|
+
continue
|
|
693
|
+
candidate_dicts.append(
|
|
694
|
+
{
|
|
695
|
+
"file": c.get("file"),
|
|
696
|
+
"symbol": caller_sym,
|
|
697
|
+
"start_line": c.get("start_line", 1),
|
|
698
|
+
"end_line": c.get("end_line", 1),
|
|
699
|
+
"score": 0.45,
|
|
700
|
+
"snippet": "",
|
|
701
|
+
"relationship": rel,
|
|
702
|
+
}
|
|
703
|
+
)
|
|
704
|
+
|
|
705
|
+
if _retrieval_policy.include_callees and task_spec.intent in ("EXPLAIN", "UNDERSTAND", "TRACE", "CHANGE", "REFACTOR", "DEBUG"):
|
|
706
|
+
callees = find_callees(con, short, max_results=max_graph_results)
|
|
707
|
+
for c in callees:
|
|
708
|
+
callee_sym = c.get("callee")
|
|
709
|
+
rel = str(c.get("relationship", "CALLS"))
|
|
710
|
+
if not _retrieval_policy.allows(rel):
|
|
711
|
+
continue
|
|
712
|
+
if not _guard.allows_symbol(str(callee_sym) if callee_sym else None):
|
|
713
|
+
continue
|
|
714
|
+
if not _guard.allows_file(str(c.get("file", ""))):
|
|
715
|
+
continue
|
|
716
|
+
candidate_dicts.append(
|
|
717
|
+
{
|
|
718
|
+
"file": c.get("file", ""),
|
|
719
|
+
"symbol": callee_sym,
|
|
720
|
+
"start_line": c.get("line", 1),
|
|
721
|
+
"end_line": c.get("line", 1),
|
|
722
|
+
"score": 0.50,
|
|
723
|
+
"snippet": "",
|
|
724
|
+
"relationship": rel,
|
|
725
|
+
}
|
|
726
|
+
)
|
|
727
|
+
# Depth 2 callee and parallel discovery
|
|
728
|
+
if _retrieval_policy.max_graph_depth >= 2 and callee_sym:
|
|
729
|
+
c2_list = find_callees(con, str(callee_sym), max_results=max_graph_results)
|
|
730
|
+
for c2 in c2_list:
|
|
731
|
+
c2_sym = c2.get("callee")
|
|
732
|
+
c2_rel = str(c2.get("relationship", "CALLS"))
|
|
733
|
+
if not _retrieval_policy.allows(c2_rel):
|
|
734
|
+
continue
|
|
735
|
+
if not _guard.allows_symbol(str(c2_sym) if c2_sym else None):
|
|
736
|
+
continue
|
|
737
|
+
if not _guard.allows_file(str(c2.get("file", ""))):
|
|
738
|
+
continue
|
|
739
|
+
candidate_dicts.append(
|
|
740
|
+
{
|
|
741
|
+
"file": c2.get("file", ""),
|
|
742
|
+
"symbol": c2_sym,
|
|
743
|
+
"start_line": c2.get("line", 1),
|
|
744
|
+
"end_line": c2.get("line", 1),
|
|
745
|
+
"score": 0.45,
|
|
746
|
+
"snippet": "",
|
|
747
|
+
"relationship": c2_rel,
|
|
748
|
+
}
|
|
749
|
+
)
|
|
750
|
+
if _retrieval_policy.allows("PARALLEL_IMPLEMENTATION"):
|
|
751
|
+
p2_list = find_parallel_implementations(con, str(callee_sym), max_results=2)
|
|
752
|
+
for p2 in p2_list:
|
|
753
|
+
p2_sym = p2.get("symbol")
|
|
754
|
+
if not _guard.allows_symbol(str(p2_sym) if p2_sym else None):
|
|
755
|
+
continue
|
|
756
|
+
if not _guard.allows_file(str(p2.get("file", ""))):
|
|
757
|
+
continue
|
|
758
|
+
candidate_dicts.append(
|
|
759
|
+
{
|
|
760
|
+
"file": p2.get("file", ""),
|
|
761
|
+
"symbol": p2_sym,
|
|
762
|
+
"canonical_id": p2.get("canonical_id"),
|
|
763
|
+
"start_line": p2.get("start_line", 1),
|
|
764
|
+
"end_line": p2.get("end_line", 1),
|
|
765
|
+
"score": 0.55,
|
|
766
|
+
"snippet": "",
|
|
767
|
+
"relationship": "PARALLEL_IMPLEMENTATION",
|
|
768
|
+
}
|
|
769
|
+
)
|
|
770
|
+
|
|
771
|
+
if _retrieval_policy.allows("PARALLEL_IMPLEMENTATION"):
|
|
772
|
+
parallels = find_parallel_implementations(con, short, max_results=3)
|
|
773
|
+
for p in parallels:
|
|
774
|
+
p_sym = p.get("symbol")
|
|
775
|
+
if not _guard.allows_symbol(str(p_sym) if p_sym else None):
|
|
776
|
+
continue
|
|
777
|
+
if not _guard.allows_file(str(p.get("file", ""))):
|
|
778
|
+
continue
|
|
779
|
+
candidate_dicts.append(
|
|
780
|
+
{
|
|
781
|
+
"file": p.get("file", ""),
|
|
782
|
+
"symbol": p_sym,
|
|
783
|
+
"canonical_id": p.get("canonical_id"),
|
|
784
|
+
"start_line": p.get("start_line", 1),
|
|
785
|
+
"end_line": p.get("end_line", 1),
|
|
786
|
+
"score": 0.65,
|
|
787
|
+
"snippet": "",
|
|
788
|
+
"relationship": "PARALLEL_IMPLEMENTATION",
|
|
789
|
+
}
|
|
790
|
+
)
|
|
791
|
+
|
|
792
|
+
# Also discover parallel implementations for top search results
|
|
793
|
+
if _retrieval_policy.allows("PARALLEL_IMPLEMENTATION"):
|
|
794
|
+
for s_item in search_res[:4]:
|
|
795
|
+
s_sym = str(s_item.get("symbol") or "")
|
|
796
|
+
if s_sym and s_sym not in _seen_seeds:
|
|
797
|
+
parallels = find_parallel_implementations(con, s_sym, max_results=2)
|
|
798
|
+
for p in parallels:
|
|
799
|
+
p_sym = p.get("symbol")
|
|
800
|
+
if not _guard.allows_symbol(str(p_sym) if p_sym else None):
|
|
801
|
+
continue
|
|
802
|
+
if not _guard.allows_file(str(p.get("file", ""))):
|
|
803
|
+
continue
|
|
804
|
+
candidate_dicts.append(
|
|
805
|
+
{
|
|
806
|
+
"file": p.get("file", ""),
|
|
807
|
+
"symbol": p_sym,
|
|
808
|
+
"canonical_id": p.get("canonical_id"),
|
|
809
|
+
"start_line": p.get("start_line", 1),
|
|
810
|
+
"end_line": p.get("end_line", 1),
|
|
811
|
+
"score": 0.60,
|
|
812
|
+
"snippet": "",
|
|
813
|
+
"relationship": "PARALLEL_IMPLEMENTATION",
|
|
814
|
+
}
|
|
815
|
+
)
|
|
816
|
+
|
|
817
|
+
timing.graph_ms = (time.perf_counter() - t_graph_start) * 1000
|
|
818
|
+
|
|
819
|
+
# Related tests
|
|
820
|
+
t_tests_start = time.perf_counter()
|
|
821
|
+
tests_out: list[TestRef] = []
|
|
822
|
+
if retrieval_plan.include_tests:
|
|
823
|
+
test_sym_targets = list(task_spec.priority_targets or task_spec.targets)
|
|
824
|
+
if not test_sym_targets:
|
|
825
|
+
test_sym_targets = list(target_symbols)
|
|
826
|
+
for sym in test_sym_targets[:max_test_syms]:
|
|
827
|
+
test_results = find_related_tests(con, sym, max_results=max_test_results)
|
|
828
|
+
for test_item in test_results:
|
|
829
|
+
if "result" in test_item:
|
|
830
|
+
continue
|
|
831
|
+
candidate_dicts.append(
|
|
832
|
+
{
|
|
833
|
+
"file": test_item.get("file"),
|
|
834
|
+
"symbol": test_item.get("symbol"),
|
|
835
|
+
"start_line": test_item.get("start_line", 1),
|
|
836
|
+
"end_line": test_item.get("start_line", 1),
|
|
837
|
+
"score": 0.55,
|
|
838
|
+
"snippet": "",
|
|
839
|
+
}
|
|
840
|
+
)
|
|
841
|
+
tests_out.append(
|
|
842
|
+
TestRef(
|
|
843
|
+
file=str(test_item.get("file", "")),
|
|
844
|
+
symbol=str(test_item["symbol"]) if test_item.get("symbol") is not None else None,
|
|
845
|
+
relationship=str(test_item.get("relationship", "")),
|
|
846
|
+
confidence=str(test_item.get("confidence", "LOW")),
|
|
847
|
+
evidence=str(test_item.get("evidence", "")),
|
|
848
|
+
)
|
|
849
|
+
)
|
|
850
|
+
timing.tests_ms = (time.perf_counter() - t_tests_start) * 1000
|
|
851
|
+
|
|
852
|
+
# 6. Rank candidates
|
|
853
|
+
t_rank_start = time.perf_counter()
|
|
854
|
+
legacy_intent_str = str(caller_raw_intent or task_spec.intent.lower())
|
|
855
|
+
ranked: list[RankedItem] = rank(
|
|
856
|
+
candidate_dicts,
|
|
857
|
+
query=task_display_str,
|
|
858
|
+
intent=legacy_intent_str,
|
|
859
|
+
target_symbols=target_symbols,
|
|
860
|
+
recent_paths=recent_paths,
|
|
861
|
+
max_results=effective_top_k * 4,
|
|
862
|
+
allowed_relationships=_retrieval_policy.allowed_relationship_types,
|
|
863
|
+
)
|
|
864
|
+
timing.ranking_ms = (time.perf_counter() - t_rank_start) * 1000
|
|
865
|
+
|
|
866
|
+
# 7. Snippets and Evidence
|
|
867
|
+
t_evidence_start = time.perf_counter()
|
|
868
|
+
for item in ranked:
|
|
869
|
+
if not item.snippet and item.symbol:
|
|
870
|
+
row = con.execute(
|
|
871
|
+
"SELECT content FROM chunks WHERE symbol=? LIMIT 1", (item.symbol,)
|
|
872
|
+
).fetchone()
|
|
873
|
+
if row:
|
|
874
|
+
snip = row["content"][:800]
|
|
875
|
+
object.__setattr__(item, "snippet", snip)
|
|
876
|
+
object.__setattr__(item, "token_estimate", _token_estimate(snip))
|
|
877
|
+
timing.evidence_ms = (time.perf_counter() - t_evidence_start) * 1000
|
|
878
|
+
|
|
879
|
+
# 8. Token Budget Optimization
|
|
880
|
+
t_compile_start = time.perf_counter()
|
|
881
|
+
candidate_items: list[CandidateContextItem] = []
|
|
882
|
+
for idx, item in enumerate(ranked):
|
|
883
|
+
layer = "GENERAL"
|
|
884
|
+
if any(ep["file"] == item.file for ep in entry_points_out):
|
|
885
|
+
layer = "ENTRYPOINT"
|
|
886
|
+
elif "test" in item.file.lower():
|
|
887
|
+
layer = "TEST"
|
|
888
|
+
elif item.file in recent_paths:
|
|
889
|
+
layer = "GIT"
|
|
890
|
+
elif "service" in item.file.lower() or "controller" in item.file.lower():
|
|
891
|
+
layer = "SERVICE"
|
|
892
|
+
elif "model" in item.file.lower() or "db" in item.file.lower() or "schema" in item.file.lower():
|
|
893
|
+
layer = "DATA"
|
|
894
|
+
|
|
895
|
+
canon = None
|
|
896
|
+
if item.symbol:
|
|
897
|
+
srow = con.execute(
|
|
898
|
+
"SELECT canonical_id FROM symbols WHERE canonical_id=? OR qualified_name=? OR name=? LIMIT 1",
|
|
899
|
+
(item.symbol, item.symbol, item.symbol),
|
|
900
|
+
).fetchone()
|
|
901
|
+
if srow:
|
|
902
|
+
canon = srow["canonical_id"]
|
|
903
|
+
|
|
904
|
+
candidate_items.append(
|
|
905
|
+
CandidateContextItem(
|
|
906
|
+
item_id=f"c_{idx}_{item.file}_{item.start_line}",
|
|
907
|
+
file_path=item.file,
|
|
908
|
+
start_line=item.start_line,
|
|
909
|
+
end_line=item.end_line,
|
|
910
|
+
canonical_id=canon,
|
|
911
|
+
estimated_tokens=item.token_estimate,
|
|
912
|
+
relevance_score=item.score,
|
|
913
|
+
evidence_quality=0.9 if item.file not in stale_paths else 0.4,
|
|
914
|
+
freshness="STALE" if item.file in stale_paths else "FRESH",
|
|
915
|
+
coverage_layer=layer,
|
|
916
|
+
source_type="symbol" if item.symbol else "chunk",
|
|
917
|
+
snippet=item.snippet,
|
|
918
|
+
confidence="HIGH" if item.file not in stale_paths else "LOW",
|
|
919
|
+
relationship_value=0.8 if layer in ("ENTRYPOINT", "SERVICE", "TEST") else 0.4,
|
|
920
|
+
epistemic_status="FACT" if item.file not in stale_paths else "UNKNOWN",
|
|
921
|
+
data={"reasons": item.reasons, "symbol": item.symbol},
|
|
922
|
+
)
|
|
923
|
+
)
|
|
924
|
+
|
|
925
|
+
selected_items, budget = optimize_context_budget(
|
|
926
|
+
candidates=candidate_items,
|
|
927
|
+
token_budget=max_tokens,
|
|
928
|
+
task_spec=task_spec,
|
|
929
|
+
)
|
|
930
|
+
timing.compilation_ms = (time.perf_counter() - t_compile_start) * 1000
|
|
931
|
+
|
|
932
|
+
# 9. Assembly & Serialization
|
|
933
|
+
t_serial_start = time.perf_counter()
|
|
934
|
+
selected_paths = {item.file_path for item in selected_items}
|
|
935
|
+
discarded_paths = {item.file_path for item in candidate_items if item.file_path not in selected_paths}
|
|
936
|
+
|
|
937
|
+
symbols_out: list[SymbolRef] = []
|
|
938
|
+
seen_sym_keys: set[tuple[str, str]] = set()
|
|
939
|
+
for it in selected_items:
|
|
940
|
+
sym_name = it.data.get("symbol")
|
|
941
|
+
if sym_name:
|
|
942
|
+
sym_key = (sym_name, it.file_path)
|
|
943
|
+
if sym_key in seen_sym_keys:
|
|
944
|
+
continue
|
|
945
|
+
seen_sym_keys.add(sym_key)
|
|
946
|
+
known_cid = it.canonical_id or it.data.get("canonical_id")
|
|
947
|
+
if known_cid:
|
|
948
|
+
row = con.execute(
|
|
949
|
+
"SELECT canonical_id, kind FROM symbols WHERE canonical_id=? LIMIT 1",
|
|
950
|
+
(known_cid,),
|
|
951
|
+
).fetchone()
|
|
952
|
+
else:
|
|
953
|
+
row = con.execute(
|
|
954
|
+
"SELECT canonical_id, kind FROM symbols WHERE (qualified_name=? OR name=?) AND path=? LIMIT 1",
|
|
955
|
+
(sym_name, sym_name, it.file_path),
|
|
956
|
+
).fetchone()
|
|
957
|
+
if not row:
|
|
958
|
+
row = con.execute(
|
|
959
|
+
"SELECT canonical_id, kind FROM symbols WHERE qualified_name=? OR name=? LIMIT 1",
|
|
960
|
+
(sym_name, sym_name),
|
|
961
|
+
).fetchone()
|
|
962
|
+
cid = known_cid or (row["canonical_id"] if row and "canonical_id" in row.keys() else None)
|
|
963
|
+
kind = row["kind"] if row else "unknown"
|
|
964
|
+
reasons = it.data.get("reasons", [])
|
|
965
|
+
symbols_out.append(
|
|
966
|
+
SymbolRef(
|
|
967
|
+
symbol=sym_name,
|
|
968
|
+
file=it.file_path,
|
|
969
|
+
kind=kind,
|
|
970
|
+
start_line=it.start_line,
|
|
971
|
+
end_line=it.end_line,
|
|
972
|
+
score=it.relevance_score,
|
|
973
|
+
reasons=[
|
|
974
|
+
RankingReasonModel(code=rs.code, label=rs.label, contribution=rs.contribution)
|
|
975
|
+
for rs in reasons
|
|
976
|
+
],
|
|
977
|
+
canonical_id=cid or it.canonical_id,
|
|
978
|
+
)
|
|
979
|
+
)
|
|
980
|
+
|
|
981
|
+
# ── Relationships extraction ────────────────────────────────────────
|
|
982
|
+
relationships_raw: list[Relationship] = list(route_relationships)
|
|
983
|
+
seen_rel: set[tuple[str, str, str]] = {
|
|
984
|
+
(r.source, r.target, r.relationship) for r in route_relationships
|
|
985
|
+
}
|
|
986
|
+
|
|
987
|
+
try:
|
|
988
|
+
all_syms = list(target_symbols) + [s.split(".")[-1] for s in target_symbols]
|
|
989
|
+
if all_syms:
|
|
990
|
+
placeholders = ",".join("?" for _ in all_syms)
|
|
991
|
+
edge_rows = con.execute(
|
|
992
|
+
f"""
|
|
993
|
+
SELECT source, target, relationship, confidence, file, start_line, evidence
|
|
994
|
+
FROM graph_edges
|
|
995
|
+
WHERE source IN ({placeholders}) OR target IN ({placeholders})
|
|
996
|
+
LIMIT 50
|
|
997
|
+
""",
|
|
998
|
+
all_syms + all_syms,
|
|
999
|
+
).fetchall()
|
|
1000
|
+
for er in edge_rows:
|
|
1001
|
+
key = (er["source"], er["target"], er["relationship"])
|
|
1002
|
+
if key not in seen_rel:
|
|
1003
|
+
seen_rel.add(key)
|
|
1004
|
+
relationships_raw.append(
|
|
1005
|
+
Relationship(
|
|
1006
|
+
source=er["source"],
|
|
1007
|
+
target=er["target"],
|
|
1008
|
+
relationship=er["relationship"],
|
|
1009
|
+
confidence=er["confidence"] or "HIGH",
|
|
1010
|
+
file=er["file"],
|
|
1011
|
+
start_line=er["start_line"],
|
|
1012
|
+
evidence=er["evidence"] or "",
|
|
1013
|
+
status="FACT" if (er["confidence"] or "HIGH").upper() == "HIGH" else "INFERENCE",
|
|
1014
|
+
)
|
|
1015
|
+
)
|
|
1016
|
+
except Exception:
|
|
1017
|
+
pass
|
|
1018
|
+
|
|
1019
|
+
# Backfill callers
|
|
1020
|
+
for sym in list(target_symbols)[:5]:
|
|
1021
|
+
short = sym.split(".")[-1]
|
|
1022
|
+
callers = find_callers(con, short, max_results=10)
|
|
1023
|
+
for c in callers:
|
|
1024
|
+
c_file = str(c.get("file", ""))
|
|
1025
|
+
if c_file in selected_paths:
|
|
1026
|
+
key = (c_file, sym, "POSSIBLE_CALLS")
|
|
1027
|
+
if key not in seen_rel:
|
|
1028
|
+
seen_rel.add(key)
|
|
1029
|
+
relationships_raw.append(
|
|
1030
|
+
Relationship(
|
|
1031
|
+
source=c_file,
|
|
1032
|
+
target=sym,
|
|
1033
|
+
relationship="POSSIBLE_CALLS",
|
|
1034
|
+
confidence=str(c.get("confidence", "LOW")),
|
|
1035
|
+
file=c_file,
|
|
1036
|
+
evidence=str(c.get("evidence", "")),
|
|
1037
|
+
status="INFERENCE",
|
|
1038
|
+
)
|
|
1039
|
+
)
|
|
1040
|
+
|
|
1041
|
+
# ── v2.1 HARD SAFETY & INTEGRITY INVARIANT ──────────────────────────
|
|
1042
|
+
# Every channel must strictly conform to ConstraintGuard and RetrievalPolicy:
|
|
1043
|
+
# 1. Symbols
|
|
1044
|
+
symbols_out = [
|
|
1045
|
+
s for s in symbols_out
|
|
1046
|
+
if _guard.allows_symbol(s.symbol) and _guard.allows_canonical_id(s.canonical_id) and _guard.allows_file(s.file)
|
|
1047
|
+
]
|
|
1048
|
+
|
|
1049
|
+
# 2. Relationships
|
|
1050
|
+
relationships_out: list[Relationship] = [
|
|
1051
|
+
r for r in relationships_raw
|
|
1052
|
+
if _guard.allows_relationship(r.source, r.relationship, r.target)
|
|
1053
|
+
and (_guard.allows_file(r.file) if r.file else True)
|
|
1054
|
+
and _retrieval_policy.allows(r.relationship)
|
|
1055
|
+
]
|
|
1056
|
+
|
|
1057
|
+
# 3. Tests
|
|
1058
|
+
tests_out = [
|
|
1059
|
+
t for t in tests_out
|
|
1060
|
+
if _guard.allows_test(t.file, t.symbol)
|
|
1061
|
+
]
|
|
1062
|
+
|
|
1063
|
+
# 4. Entry points & framework facts
|
|
1064
|
+
entry_points_out = [
|
|
1065
|
+
ep for ep in entry_points_out
|
|
1066
|
+
if _guard.allows_endpoint(
|
|
1067
|
+
str(ep.get("route_path")),
|
|
1068
|
+
str(ep.get("handler")),
|
|
1069
|
+
str(ep.get("file")),
|
|
1070
|
+
)
|
|
1071
|
+
]
|
|
1072
|
+
framework_facts_out = [
|
|
1073
|
+
ff for ff in framework_facts_out
|
|
1074
|
+
if _guard.allows_symbol(str(ff.get("handler"))) and _guard.allows_file(str(ff.get("file")))
|
|
1075
|
+
]
|
|
1076
|
+
|
|
1077
|
+
# 5. Evidence
|
|
1078
|
+
evidence_out: list[EvidenceRef] = []
|
|
1079
|
+
for it in selected_items[:12]:
|
|
1080
|
+
if it.snippet:
|
|
1081
|
+
if not _guard.allows_file(it.file_path) or not _guard.allows_symbol(it.data.get("symbol")):
|
|
1082
|
+
continue
|
|
1083
|
+
src_hash: str | None = None
|
|
1084
|
+
try:
|
|
1085
|
+
row = con.execute("SELECT content_hash FROM files WHERE path=?", (it.file_path,)).fetchone()
|
|
1086
|
+
if row:
|
|
1087
|
+
src_hash = row["content_hash"]
|
|
1088
|
+
except Exception:
|
|
1089
|
+
pass
|
|
1090
|
+
is_stale = it.file_path in stale_paths
|
|
1091
|
+
evidence_out.append(
|
|
1092
|
+
EvidenceRef(
|
|
1093
|
+
file=it.file_path,
|
|
1094
|
+
start_line=it.start_line,
|
|
1095
|
+
end_line=it.end_line,
|
|
1096
|
+
symbol=it.data.get("symbol"),
|
|
1097
|
+
snippet=it.snippet[:600],
|
|
1098
|
+
confidence="LOW" if is_stale else "HIGH",
|
|
1099
|
+
source_hash=src_hash,
|
|
1100
|
+
evidence_status="stale" if is_stale else "current",
|
|
1101
|
+
)
|
|
1102
|
+
)
|
|
1103
|
+
|
|
1104
|
+
# 6. Files
|
|
1105
|
+
files_out: list[FileRef] = []
|
|
1106
|
+
seen_files: set[str] = set()
|
|
1107
|
+
for it in selected_items:
|
|
1108
|
+
if not _guard.allows_file(it.file_path):
|
|
1109
|
+
continue
|
|
1110
|
+
if it.file_path in seen_files:
|
|
1111
|
+
continue
|
|
1112
|
+
seen_files.add(it.file_path)
|
|
1113
|
+
reasons = it.data.get("reasons", [])
|
|
1114
|
+
files_out.append(
|
|
1115
|
+
FileRef(
|
|
1116
|
+
file=it.file_path,
|
|
1117
|
+
score=it.relevance_score,
|
|
1118
|
+
reasons=[
|
|
1119
|
+
RankingReasonModel(code=rs.code, label=rs.label, contribution=rs.contribution)
|
|
1120
|
+
for rs in reasons
|
|
1121
|
+
],
|
|
1122
|
+
snippet=it.snippet[:400],
|
|
1123
|
+
token_estimate=it.estimated_tokens,
|
|
1124
|
+
)
|
|
1125
|
+
)
|
|
1126
|
+
|
|
1127
|
+
uncertainties: list[str] = []
|
|
1128
|
+
if freshness_report.status != FreshnessStatus.FRESH:
|
|
1129
|
+
uncertainties.append(f"Index is {freshness_report.status.value}: {freshness_report.detail}")
|
|
1130
|
+
if stale_paths & selected_paths:
|
|
1131
|
+
uncertainties.append(f"Evidence from {len(stale_paths & selected_paths)} file(s) may be stale.")
|
|
1132
|
+
if not candidate_dicts:
|
|
1133
|
+
uncertainties.append("No lexical or graph matches found. Results may be incomplete.")
|
|
1134
|
+
|
|
1135
|
+
assumptions_out = list(task_spec.assumptions)
|
|
1136
|
+
unknowns_out = list(task_spec.unknowns)
|
|
1137
|
+
|
|
1138
|
+
timing.serialization_ms = (time.perf_counter() - t_serial_start) * 1000
|
|
1139
|
+
timing.total_ms = timer.elapsed_ms
|
|
1140
|
+
|
|
1141
|
+
exec_meta = ExecutionMetadata(
|
|
1142
|
+
latency_ms=timer.elapsed_ms,
|
|
1143
|
+
cache_hit=False,
|
|
1144
|
+
index_generation=gen,
|
|
1145
|
+
candidate_count=budget.candidate_tokens,
|
|
1146
|
+
selected_count=budget.selected_tokens,
|
|
1147
|
+
mode=mode_upper,
|
|
1148
|
+
timing=timing,
|
|
1149
|
+
)
|
|
1150
|
+
exec_dict = exec_meta.as_dict()
|
|
1151
|
+
exec_dict["resource"] = {
|
|
1152
|
+
"profile": governor.policy.profile.value,
|
|
1153
|
+
"pressure": governor.get_pressure().value,
|
|
1154
|
+
"activity": governor.get_activity_mode().value,
|
|
1155
|
+
"estimated_memory_mb": get_process_memory_mb(),
|
|
1156
|
+
}
|
|
1157
|
+
if explain:
|
|
1158
|
+
exec_dict["rejections"] = list(budget.rejections)
|
|
1159
|
+
exec_dict["retrieval_plan"] = retrieval_plan.as_dict()
|
|
1160
|
+
exec_dict["explain"] = {
|
|
1161
|
+
"task_spec": task_spec.as_dict(),
|
|
1162
|
+
"retrieval_plan": retrieval_plan.as_dict(),
|
|
1163
|
+
"stage1_candidate_count": stage1_candidate_count,
|
|
1164
|
+
"stage2_candidate_count": len(candidate_dicts),
|
|
1165
|
+
"coverage_layers": {
|
|
1166
|
+
layer: sum(1 for it in selected_items if it.coverage_layer == layer)
|
|
1167
|
+
for layer in set(it.coverage_layer for it in selected_items)
|
|
1168
|
+
},
|
|
1169
|
+
"why_selected": {
|
|
1170
|
+
it.item_id: it.why_selected for it in selected_items
|
|
1171
|
+
},
|
|
1172
|
+
"rejections": list(budget.rejections),
|
|
1173
|
+
}
|
|
1174
|
+
|
|
1175
|
+
packet = ContextPacket(
|
|
1176
|
+
schema_version=SCHEMA_VERSION,
|
|
1177
|
+
task=task_display_str,
|
|
1178
|
+
intent=str(caller_raw_intent) if caller_raw_intent else task_spec.intent.lower(),
|
|
1179
|
+
repository=str(repository),
|
|
1180
|
+
indexed_commit=freshness_report.indexed_commit,
|
|
1181
|
+
current_commit=freshness_report.current_commit,
|
|
1182
|
+
freshness=freshness_report.status.value,
|
|
1183
|
+
freshness_detail=freshness_report.detail,
|
|
1184
|
+
repository_generation=gen,
|
|
1185
|
+
task_spec=task_spec.as_dict(),
|
|
1186
|
+
task_fingerprint=getattr(retrieval_plan, "task_fingerprint", "") or task_spec.fingerprint(),
|
|
1187
|
+
budget=budget.as_dict(),
|
|
1188
|
+
execution=exec_dict,
|
|
1189
|
+
mode=mode_upper,
|
|
1190
|
+
entry_points=entry_points_out,
|
|
1191
|
+
framework_facts=framework_facts_out,
|
|
1192
|
+
architecture=architecture_out,
|
|
1193
|
+
git_changes=git_changes_out,
|
|
1194
|
+
git_facts=git_changes_out,
|
|
1195
|
+
assumptions=assumptions_out,
|
|
1196
|
+
unknowns=unknowns_out,
|
|
1197
|
+
conflicts=list(task_spec.conflicts) if hasattr(task_spec, "conflicts") else [],
|
|
1198
|
+
symbols=symbols_out,
|
|
1199
|
+
relationships=relationships_out,
|
|
1200
|
+
tests=tests_out,
|
|
1201
|
+
files=files_out,
|
|
1202
|
+
evidence=evidence_out,
|
|
1203
|
+
uncertainties=uncertainties,
|
|
1204
|
+
candidate_token_estimate=budget.candidate_tokens,
|
|
1205
|
+
selected_token_estimate=budget.selected_tokens,
|
|
1206
|
+
context_reduction_pct=round(budget.reduction_ratio * 100, 1),
|
|
1207
|
+
selected_files=sorted(selected_paths),
|
|
1208
|
+
discarded_files=sorted(discarded_paths),
|
|
1209
|
+
)
|
|
1210
|
+
|
|
1211
|
+
store_cached_context_packet(con, cache_key, packet.as_dict())
|
|
1212
|
+
|
|
1213
|
+
get_global_metrics().record(
|
|
1214
|
+
OperationMetric(
|
|
1215
|
+
request_id=req_id,
|
|
1216
|
+
tool_or_op="get_context",
|
|
1217
|
+
latency_ms=timer.elapsed_ms,
|
|
1218
|
+
index_generation=gen,
|
|
1219
|
+
cache_hit=False,
|
|
1220
|
+
candidate_count=budget.candidate_tokens,
|
|
1221
|
+
selected_count=budget.selected_tokens,
|
|
1222
|
+
estimated_tokens=budget.selected_tokens,
|
|
1223
|
+
task_intent=task_spec.intent,
|
|
1224
|
+
freshness=freshness_report.status.value,
|
|
1225
|
+
)
|
|
1226
|
+
)
|
|
1227
|
+
|
|
1228
|
+
return packet
|