codegraph-engine 2.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codegraph/__init__.py +37 -0
- codegraph/agent.py +26 -0
- codegraph/architecture.py +328 -0
- codegraph/audit.py +106 -0
- codegraph/cache.py +95 -0
- codegraph/cli.py +854 -0
- codegraph/config.py +43 -0
- codegraph/constraints.py +238 -0
- codegraph/context.py +1228 -0
- codegraph/epistemic.py +90 -0
- codegraph/errors.py +275 -0
- codegraph/evidence/__init__.py +15 -0
- codegraph/evidence/citations.py +397 -0
- codegraph/frameworks.py +434 -0
- codegraph/freshness.py +295 -0
- codegraph/git.py +278 -0
- codegraph/graph/__init__.py +46 -0
- codegraph/graph/models.py +41 -0
- codegraph/graph/traversal.py +1291 -0
- codegraph/indexing/__init__.py +4 -0
- codegraph/indexing/classifier.py +274 -0
- codegraph/indexing/indexer.py +943 -0
- codegraph/indexing/models.py +338 -0
- codegraph/indexing/parser.py +1240 -0
- codegraph/indexing/scanner.py +200 -0
- codegraph/indexing/test_framework.py +116 -0
- codegraph/interrogation.py +1582 -0
- codegraph/llm/__init__.py +3 -0
- codegraph/llm/base.py +15 -0
- codegraph/llm/context.py +20 -0
- codegraph/mcp/__init__.py +3 -0
- codegraph/mcp/server.py +736 -0
- codegraph/memory/__init__.py +3 -0
- codegraph/memory/store.py +46 -0
- codegraph/models.py +289 -0
- codegraph/observability.py +151 -0
- codegraph/optimizer.py +372 -0
- codegraph/planner.py +417 -0
- codegraph/py.typed +1 -0
- codegraph/query_expansion.py +199 -0
- codegraph/ranking.py +363 -0
- codegraph/resolver.py +843 -0
- codegraph/resources/__init__.py +45 -0
- codegraph/resources/cache.py +117 -0
- codegraph/resources/coalescer.py +83 -0
- codegraph/resources/debouncer.py +98 -0
- codegraph/resources/governor.py +232 -0
- codegraph/resources/policy.py +123 -0
- codegraph/retrieval_policy.py +220 -0
- codegraph/search/__init__.py +23 -0
- codegraph/search/hybrid.py +301 -0
- codegraph/search/semantic.py +28 -0
- codegraph/security/__init__.py +3 -0
- codegraph/security/paths.py +35 -0
- codegraph/target_resolver.py +348 -0
- codegraph/task.py +637 -0
- codegraph_engine-2.1.1.dist-info/METADATA +334 -0
- codegraph_engine-2.1.1.dist-info/RECORD +62 -0
- codegraph_engine-2.1.1.dist-info/WHEEL +5 -0
- codegraph_engine-2.1.1.dist-info/entry_points.txt +2 -0
- codegraph_engine-2.1.1.dist-info/licenses/LICENSE +21 -0
- codegraph_engine-2.1.1.dist-info/top_level.txt +1 -0
codegraph/task.py
ADDED
|
@@ -0,0 +1,637 @@
|
|
|
1
|
+
"""Canonical TaskSpec, Ambiguity, and Task-Understanding Contract.
|
|
2
|
+
|
|
3
|
+
The user's AI model understands natural language and creates a structured TaskSpec.
|
|
4
|
+
CodeGraph MCP deterministically normalizes, validates, and grounds the TaskSpec against
|
|
5
|
+
actual repository facts, symbols, routes, and files with zero LLM API dependency.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import hashlib
|
|
10
|
+
import json
|
|
11
|
+
import re
|
|
12
|
+
import sqlite3
|
|
13
|
+
from dataclasses import asdict, dataclass
|
|
14
|
+
from enum import StrEnum
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
TASK_SPEC_SCHEMA_VERSION = "1.0"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class TaskIntent(StrEnum):
|
|
21
|
+
UNDERSTAND = "UNDERSTAND"
|
|
22
|
+
DEBUG = "DEBUG"
|
|
23
|
+
CHANGE = "CHANGE"
|
|
24
|
+
REFACTOR = "REFACTOR"
|
|
25
|
+
TRACE = "TRACE"
|
|
26
|
+
IMPACT = "IMPACT"
|
|
27
|
+
REVIEW = "REVIEW"
|
|
28
|
+
TEST = "TEST"
|
|
29
|
+
ARCHITECTURE = "ARCHITECTURE"
|
|
30
|
+
EXPLAIN = "EXPLAIN"
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
_INTENT_PRIORITY_ORDER: list[tuple[str, TaskIntent]] = [
|
|
34
|
+
("blast_radius", TaskIntent.IMPACT),
|
|
35
|
+
("impact", TaskIntent.IMPACT),
|
|
36
|
+
("trace", TaskIntent.TRACE),
|
|
37
|
+
("callgraph", TaskIntent.TRACE),
|
|
38
|
+
("architecture", TaskIntent.ARCHITECTURE),
|
|
39
|
+
("overview", TaskIntent.ARCHITECTURE),
|
|
40
|
+
("refactor", TaskIntent.REFACTOR),
|
|
41
|
+
("restructure", TaskIntent.REFACTOR),
|
|
42
|
+
("review", TaskIntent.REVIEW),
|
|
43
|
+
("diff", TaskIntent.REVIEW),
|
|
44
|
+
("debug", TaskIntent.DEBUG),
|
|
45
|
+
("troubleshoot", TaskIntent.DEBUG),
|
|
46
|
+
("investigate", TaskIntent.DEBUG),
|
|
47
|
+
("test", TaskIntent.TEST),
|
|
48
|
+
("tests", TaskIntent.TEST),
|
|
49
|
+
("change", TaskIntent.CHANGE),
|
|
50
|
+
("modify", TaskIntent.CHANGE),
|
|
51
|
+
("edit", TaskIntent.CHANGE),
|
|
52
|
+
("fix", TaskIntent.CHANGE),
|
|
53
|
+
("patch", TaskIntent.CHANGE),
|
|
54
|
+
("understand", TaskIntent.UNDERSTAND),
|
|
55
|
+
("explain", TaskIntent.EXPLAIN),
|
|
56
|
+
]
|
|
57
|
+
|
|
58
|
+
_INTENT_SYNONYMS: dict[str, TaskIntent] = dict(_INTENT_PRIORITY_ORDER)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class AmbiguityStatus(StrEnum):
|
|
62
|
+
CLEAR = "CLEAR"
|
|
63
|
+
ASSUMED = "ASSUMED"
|
|
64
|
+
AMBIGUOUS = "AMBIGUOUS"
|
|
65
|
+
UNKNOWN = "UNKNOWN"
|
|
66
|
+
CONFLICT = "CONFLICT"
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
@dataclass(frozen=True)
|
|
70
|
+
class AmbiguityCandidate:
|
|
71
|
+
canonical_id: str
|
|
72
|
+
name: str
|
|
73
|
+
path: str
|
|
74
|
+
kind: str
|
|
75
|
+
confidence: str = "HIGH" # HIGH | MEDIUM | LOW
|
|
76
|
+
reason: str = ""
|
|
77
|
+
|
|
78
|
+
def as_dict(self) -> dict[str, object]:
|
|
79
|
+
return asdict(self)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
@dataclass(frozen=True)
|
|
83
|
+
class TaskAmbiguity:
|
|
84
|
+
status: AmbiguityStatus
|
|
85
|
+
target: str
|
|
86
|
+
candidates: tuple[AmbiguityCandidate, ...] = ()
|
|
87
|
+
assumption: str | None = None
|
|
88
|
+
conflict_detail: str | None = None
|
|
89
|
+
|
|
90
|
+
def as_dict(self) -> dict[str, object]:
|
|
91
|
+
return {
|
|
92
|
+
"status": self.status.value,
|
|
93
|
+
"target": self.target,
|
|
94
|
+
"candidates": [c.as_dict() for c in self.candidates],
|
|
95
|
+
"assumption": self.assumption,
|
|
96
|
+
"conflict_detail": self.conflict_detail,
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
@dataclass(frozen=True)
|
|
101
|
+
class TaskSpec:
|
|
102
|
+
schema_version: str = TASK_SPEC_SCHEMA_VERSION
|
|
103
|
+
raw_prompt: str | None = None
|
|
104
|
+
|
|
105
|
+
intent: str = TaskIntent.UNDERSTAND.value
|
|
106
|
+
goal: str = ""
|
|
107
|
+
|
|
108
|
+
targets: tuple[str, ...] = ()
|
|
109
|
+
entities: tuple[str, ...] = ()
|
|
110
|
+
|
|
111
|
+
operations: tuple[str, ...] = ()
|
|
112
|
+
constraints: tuple[str, ...] = ()
|
|
113
|
+
exclusions: tuple[str, ...] = ()
|
|
114
|
+
|
|
115
|
+
scope_paths: tuple[str, ...] = ()
|
|
116
|
+
scope_modules: tuple[str, ...] = ()
|
|
117
|
+
frameworks: tuple[str, ...] = ()
|
|
118
|
+
|
|
119
|
+
time_scope: str | None = None
|
|
120
|
+
|
|
121
|
+
priority_targets: tuple[str, ...] = ()
|
|
122
|
+
|
|
123
|
+
ambiguities: tuple[str, ...] = ()
|
|
124
|
+
assumptions: tuple[str, ...] = ()
|
|
125
|
+
unknowns: tuple[str, ...] = ()
|
|
126
|
+
conflicts: tuple[str, ...] = ()
|
|
127
|
+
|
|
128
|
+
confidence: str = "HIGH"
|
|
129
|
+
|
|
130
|
+
def as_dict(self) -> dict[str, object]:
|
|
131
|
+
return asdict(self)
|
|
132
|
+
|
|
133
|
+
def fingerprint(self) -> str:
|
|
134
|
+
"""Compute deterministic SHA-256 fingerprint for this TaskSpec."""
|
|
135
|
+
canonical = json.dumps(
|
|
136
|
+
{
|
|
137
|
+
"intent": self.intent,
|
|
138
|
+
"goal": self.goal,
|
|
139
|
+
"targets": sorted(self.targets),
|
|
140
|
+
"entities": sorted(self.entities),
|
|
141
|
+
"operations": sorted(self.operations),
|
|
142
|
+
"constraints": sorted(self.constraints),
|
|
143
|
+
"exclusions": sorted(self.exclusions),
|
|
144
|
+
"scope_paths": sorted(self.scope_paths),
|
|
145
|
+
"scope_modules": sorted(self.scope_modules),
|
|
146
|
+
"frameworks": sorted(self.frameworks),
|
|
147
|
+
"time_scope": self.time_scope or "",
|
|
148
|
+
},
|
|
149
|
+
sort_keys=True,
|
|
150
|
+
)
|
|
151
|
+
return hashlib.sha256(canonical.encode("utf-8")).hexdigest()
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def canonicalize_intent(raw_intent: str | None) -> TaskIntent:
|
|
155
|
+
"""Deterministically map raw or case-insensitive intent strings to canonical TaskIntent."""
|
|
156
|
+
if not raw_intent:
|
|
157
|
+
return TaskIntent.UNDERSTAND
|
|
158
|
+
cleaned = raw_intent.strip().lower().replace("-", "_")
|
|
159
|
+
return _INTENT_SYNONYMS.get(cleaned, TaskIntent.UNDERSTAND)
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
class TargetExpressionType(StrEnum):
|
|
163
|
+
EXPLICIT_SYMBOL_TARGET = "EXPLICIT_SYMBOL_TARGET"
|
|
164
|
+
CONCEPT_TARGET = "CONCEPT_TARGET"
|
|
165
|
+
NATURAL_LANGUAGE_TERM = "NATURAL_LANGUAGE_TERM"
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
_COMMON_NL_WORDS = frozenset({
|
|
169
|
+
"the", "and", "for", "with", "this", "that", "from", "check", "verify", "trace",
|
|
170
|
+
"show", "what", "where", "how", "when", "does", "after", "before", "code", "file",
|
|
171
|
+
"route", "routes", "flow", "work", "works", "working", "database", "method", "class",
|
|
172
|
+
"function", "explain", "understand", "modify", "refactor", "debug", "impact", "review",
|
|
173
|
+
"find", "all", "case", "cases", "covering", "into", "over", "about", "such",
|
|
174
|
+
"input", "inputs", "output", "outputs", "final", "results", "result", "user", "users",
|
|
175
|
+
"data", "system", "systems", "handle", "handling", "start", "starting", "run", "running",
|
|
176
|
+
"call", "calling", "test", "tests", "view", "views", "item", "items", "order", "orders",
|
|
177
|
+
"list", "make", "need", "needs", "pass", "fail", "true", "false", "value", "values",
|
|
178
|
+
"key", "keys", "main", "step", "steps", "path", "paths", "helper", "helpers",
|
|
179
|
+
"service", "services", "component", "components", "feature", "features", "action", "actions",
|
|
180
|
+
"details", "detail", "overview", "summary", "create", "update", "delete",
|
|
181
|
+
})
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def classify_target_expression(expr: str) -> TargetExpressionType:
|
|
185
|
+
"""Classify a target expression into symbol, concept, or natural language term."""
|
|
186
|
+
clean = expr.strip()
|
|
187
|
+
if not clean:
|
|
188
|
+
return TargetExpressionType.NATURAL_LANGUAGE_TERM
|
|
189
|
+
|
|
190
|
+
# 1. Explicit syntax markers
|
|
191
|
+
if "::" in clean:
|
|
192
|
+
return TargetExpressionType.EXPLICIT_SYMBOL_TARGET
|
|
193
|
+
if "." in clean:
|
|
194
|
+
return TargetExpressionType.EXPLICIT_SYMBOL_TARGET
|
|
195
|
+
if clean.startswith("/") or re.match(r"^(?:GET|POST|PUT|DELETE|PATCH|OPTIONS|HEAD)\s+/", clean, re.IGNORECASE):
|
|
196
|
+
return TargetExpressionType.EXPLICIT_SYMBOL_TARGET
|
|
197
|
+
if "/" in clean and any(clean.endswith(ext) for ext in (".py", ".js", ".jsx", ".ts", ".tsx", ".json", ".html")):
|
|
198
|
+
return TargetExpressionType.EXPLICIT_SYMBOL_TARGET
|
|
199
|
+
|
|
200
|
+
# 2. Multi-word phrases
|
|
201
|
+
if " " in clean or "\t" in clean:
|
|
202
|
+
words = [w.lower() for w in clean.split()]
|
|
203
|
+
if all(w in _COMMON_NL_WORDS for w in words):
|
|
204
|
+
return TargetExpressionType.NATURAL_LANGUAGE_TERM
|
|
205
|
+
return TargetExpressionType.CONCEPT_TARGET
|
|
206
|
+
|
|
207
|
+
# 3. Code naming conventions
|
|
208
|
+
if "_" in clean:
|
|
209
|
+
return TargetExpressionType.EXPLICIT_SYMBOL_TARGET
|
|
210
|
+
|
|
211
|
+
# CamelCase (lowercase followed by uppercase, e.g. handleRequest)
|
|
212
|
+
if re.search(r"[a-z][A-Z]", clean):
|
|
213
|
+
return TargetExpressionType.EXPLICIT_SYMBOL_TARGET
|
|
214
|
+
|
|
215
|
+
# PascalCase (starts uppercase, contains lowercase, length >= 3, e.g. AuthService, FakePayService)
|
|
216
|
+
if clean[0].isupper() and any(c.islower() for c in clean[1:]) and len(clean) >= 3:
|
|
217
|
+
if clean.lower() not in _COMMON_NL_WORDS:
|
|
218
|
+
return TargetExpressionType.EXPLICIT_SYMBOL_TARGET
|
|
219
|
+
return TargetExpressionType.NATURAL_LANGUAGE_TERM
|
|
220
|
+
|
|
221
|
+
# 4. Check against common natural language words
|
|
222
|
+
if clean.lower() in _COMMON_NL_WORDS:
|
|
223
|
+
return TargetExpressionType.NATURAL_LANGUAGE_TERM
|
|
224
|
+
|
|
225
|
+
# 5. Plain single word not in stopwords (e.g. authenticate, login)
|
|
226
|
+
# Returns CONCEPT_TARGET: can match symbols in DB if present, but does not trigger UNKNOWN if absent
|
|
227
|
+
return TargetExpressionType.CONCEPT_TARGET
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
_CODE_TOKEN_RE = re.compile(r"\b([A-Za-z_][A-Za-z0-9_]{2,}(?:\.[A-Za-z_][A-Za-z0-9_]*)*)\b")
|
|
231
|
+
_ROUTE_TOKEN_RE = re.compile(r"\b(?:GET|POST|PUT|DELETE|PATCH|OPTIONS|HEAD)\s+(/[A-Za-z0-9_/{}\-.:*]*)", re.IGNORECASE)
|
|
232
|
+
_PATH_TOKEN_RE = re.compile(r"\b([A-Za-z0-9_.\-]+/[A-Za-z0-9_.\-/]+)\b")
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def extract_keywords_from_prompt(prompt: str) -> list[str]:
|
|
236
|
+
"""Deterministically extract candidate identifiers, endpoints, and terms from a prompt."""
|
|
237
|
+
candidates: list[str] = []
|
|
238
|
+
|
|
239
|
+
# 1. Quoted terms
|
|
240
|
+
for quoted in re.findall(r"['\"]([^'\"]+)['\"]", prompt):
|
|
241
|
+
q = quoted.strip()
|
|
242
|
+
if q and q not in candidates:
|
|
243
|
+
if classify_target_expression(q) != TargetExpressionType.NATURAL_LANGUAGE_TERM:
|
|
244
|
+
candidates.append(q)
|
|
245
|
+
elif len(q.split()) > 1:
|
|
246
|
+
candidates.append(q)
|
|
247
|
+
|
|
248
|
+
# 2. HTTP endpoints (e.g. POST /login, /api/v1/auth)
|
|
249
|
+
for m in _ROUTE_TOKEN_RE.finditer(prompt):
|
|
250
|
+
route_str = m.group(0).strip()
|
|
251
|
+
if route_str not in candidates:
|
|
252
|
+
candidates.append(route_str)
|
|
253
|
+
|
|
254
|
+
# 3. File paths
|
|
255
|
+
for p in _PATH_TOKEN_RE.findall(prompt):
|
|
256
|
+
if p not in candidates and not p.startswith("http"):
|
|
257
|
+
candidates.append(p)
|
|
258
|
+
|
|
259
|
+
# 4. Code identifiers (camelCase, snake_case, dotted symbols, potential functions)
|
|
260
|
+
for sym in _CODE_TOKEN_RE.findall(prompt):
|
|
261
|
+
if len(sym) >= 3:
|
|
262
|
+
expr_type = classify_target_expression(sym)
|
|
263
|
+
if expr_type in (TargetExpressionType.EXPLICIT_SYMBOL_TARGET, TargetExpressionType.CONCEPT_TARGET):
|
|
264
|
+
if sym not in candidates:
|
|
265
|
+
candidates.append(sym)
|
|
266
|
+
|
|
267
|
+
return candidates
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def decompose_prompt(prompt: str) -> list[str]:
|
|
271
|
+
"""Decompose complex, multi-concept prompts into structured conceptual search targets.
|
|
272
|
+
|
|
273
|
+
Splits multi-clause instructions, filters command boilerplate and instructions,
|
|
274
|
+
and returns deduplicated technical concepts and identifiers.
|
|
275
|
+
"""
|
|
276
|
+
cleaned = re.sub(
|
|
277
|
+
r"\b(?:investigate why|check|verify|trace|find|show me|look for|please|don't modify anything|do not modify|do not change|read only)\b",
|
|
278
|
+
"",
|
|
279
|
+
prompt,
|
|
280
|
+
flags=re.IGNORECASE,
|
|
281
|
+
).strip()
|
|
282
|
+
|
|
283
|
+
tokens = extract_keywords_from_prompt(prompt)
|
|
284
|
+
results: list[str] = list(tokens)
|
|
285
|
+
|
|
286
|
+
# Split into clause chunks by commas, semicolons, and clause boundaries
|
|
287
|
+
clauses = re.split(r"[,;\n\.\?!]+|\band\b", cleaned, flags=re.IGNORECASE)
|
|
288
|
+
for clause in clauses:
|
|
289
|
+
c = clause.strip()
|
|
290
|
+
c = re.sub(r"^(?:the|a|an|why|how|what|where|when|all|any)\s+", "", c, flags=re.IGNORECASE).strip()
|
|
291
|
+
if len(c) >= 3 and len(c.split()) <= 4:
|
|
292
|
+
expr_type = classify_target_expression(c)
|
|
293
|
+
if expr_type != TargetExpressionType.NATURAL_LANGUAGE_TERM:
|
|
294
|
+
if not any(c.lower() == r.lower() for r in results):
|
|
295
|
+
results.append(c)
|
|
296
|
+
|
|
297
|
+
deduped: list[str] = []
|
|
298
|
+
seen: set[str] = set()
|
|
299
|
+
for item in results:
|
|
300
|
+
norm = item.strip()
|
|
301
|
+
if norm and norm.lower() not in seen:
|
|
302
|
+
seen.add(norm.lower())
|
|
303
|
+
deduped.append(norm)
|
|
304
|
+
|
|
305
|
+
return deduped
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
def expand_terms_with_repository(
|
|
309
|
+
terms: list[str] | tuple[str, ...],
|
|
310
|
+
con: sqlite3.Connection | None,
|
|
311
|
+
max_expansions: int = 5,
|
|
312
|
+
) -> list[str]:
|
|
313
|
+
"""Expand user terminology with actual source-backed symbols and routes from the repository.
|
|
314
|
+
|
|
315
|
+
Canonical implementation delegating to `codegraph.query_expansion.expand_query_terms`.
|
|
316
|
+
Only introduces terms confirmed to exist in the repository index.
|
|
317
|
+
"""
|
|
318
|
+
if not con or not terms:
|
|
319
|
+
return list(terms)
|
|
320
|
+
|
|
321
|
+
from codegraph.query_expansion import expand_query_terms, get_search_queries
|
|
322
|
+
|
|
323
|
+
expanded_objs = expand_query_terms(terms, con, max_expansions=max_expansions)
|
|
324
|
+
queries = get_search_queries(expanded_objs)
|
|
325
|
+
|
|
326
|
+
expanded: list[str] = list(terms)
|
|
327
|
+
seen: set[str] = {t.lower() for t in terms}
|
|
328
|
+
for q in queries:
|
|
329
|
+
if q.lower() not in seen:
|
|
330
|
+
seen.add(q.lower())
|
|
331
|
+
expanded.append(q)
|
|
332
|
+
|
|
333
|
+
return expanded
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
def detect_target_ambiguity(
|
|
338
|
+
target: str,
|
|
339
|
+
con: sqlite3.Connection | None = None,
|
|
340
|
+
exclusions: tuple[str, ...] | None = None,
|
|
341
|
+
) -> TaskAmbiguity:
|
|
342
|
+
"""Ground a target against the repository symbol table and detect potential ambiguities."""
|
|
343
|
+
if not con or not target.strip():
|
|
344
|
+
return TaskAmbiguity(status=AmbiguityStatus.CLEAR, target=target)
|
|
345
|
+
|
|
346
|
+
clean_target = target.strip()
|
|
347
|
+
|
|
348
|
+
# Check for direct specification conflict
|
|
349
|
+
if exclusions and clean_target in exclusions:
|
|
350
|
+
return TaskAmbiguity(
|
|
351
|
+
status=AmbiguityStatus.CONFLICT,
|
|
352
|
+
target=clean_target,
|
|
353
|
+
conflict_detail=f"Target '{clean_target}' is in both targets and exclusions.",
|
|
354
|
+
)
|
|
355
|
+
|
|
356
|
+
# 1. Exact canonical ID check
|
|
357
|
+
exact_canon = con.execute(
|
|
358
|
+
"SELECT canonical_id, name, path, kind FROM symbols WHERE canonical_id=?",
|
|
359
|
+
(clean_target,),
|
|
360
|
+
).fetchone()
|
|
361
|
+
if exact_canon:
|
|
362
|
+
cand = AmbiguityCandidate(
|
|
363
|
+
canonical_id=exact_canon["canonical_id"],
|
|
364
|
+
name=exact_canon["name"],
|
|
365
|
+
path=exact_canon["path"],
|
|
366
|
+
kind=exact_canon["kind"],
|
|
367
|
+
confidence="HIGH",
|
|
368
|
+
reason="Exact canonical ID match",
|
|
369
|
+
)
|
|
370
|
+
return TaskAmbiguity(
|
|
371
|
+
status=AmbiguityStatus.CLEAR,
|
|
372
|
+
target=clean_target,
|
|
373
|
+
candidates=(cand,),
|
|
374
|
+
)
|
|
375
|
+
|
|
376
|
+
# 2. Framework endpoint check
|
|
377
|
+
if " " in clean_target or clean_target.startswith("/"):
|
|
378
|
+
route_row = con.execute(
|
|
379
|
+
"SELECT endpoint_id, handler_name, file_path FROM framework_routes WHERE route_path=? OR endpoint_id=?",
|
|
380
|
+
(clean_target, clean_target),
|
|
381
|
+
).fetchone()
|
|
382
|
+
if route_row:
|
|
383
|
+
cand = AmbiguityCandidate(
|
|
384
|
+
canonical_id=route_row["endpoint_id"] or route_row["handler_name"],
|
|
385
|
+
name=route_row["handler_name"],
|
|
386
|
+
path=route_row["file_path"],
|
|
387
|
+
kind="endpoint",
|
|
388
|
+
confidence="HIGH",
|
|
389
|
+
reason="Framework route match",
|
|
390
|
+
)
|
|
391
|
+
return TaskAmbiguity(
|
|
392
|
+
status=AmbiguityStatus.CLEAR,
|
|
393
|
+
target=clean_target,
|
|
394
|
+
candidates=(cand,),
|
|
395
|
+
)
|
|
396
|
+
|
|
397
|
+
# 3. Check symbols by name or qualified_name
|
|
398
|
+
rows = con.execute(
|
|
399
|
+
"SELECT canonical_id, name, path, kind FROM symbols WHERE name=? OR qualified_name=?",
|
|
400
|
+
(clean_target, clean_target),
|
|
401
|
+
).fetchall()
|
|
402
|
+
|
|
403
|
+
if len(rows) == 1:
|
|
404
|
+
r = rows[0]
|
|
405
|
+
cand = AmbiguityCandidate(
|
|
406
|
+
canonical_id=r["canonical_id"],
|
|
407
|
+
name=r["name"],
|
|
408
|
+
path=r["path"],
|
|
409
|
+
kind=r["kind"],
|
|
410
|
+
confidence="HIGH",
|
|
411
|
+
reason="Uniquely matched symbol in repository",
|
|
412
|
+
)
|
|
413
|
+
return TaskAmbiguity(
|
|
414
|
+
status=AmbiguityStatus.CLEAR,
|
|
415
|
+
target=clean_target,
|
|
416
|
+
candidates=(cand,),
|
|
417
|
+
)
|
|
418
|
+
|
|
419
|
+
if len(rows) > 1:
|
|
420
|
+
candidates = tuple(
|
|
421
|
+
AmbiguityCandidate(
|
|
422
|
+
canonical_id=r["canonical_id"],
|
|
423
|
+
name=r["name"],
|
|
424
|
+
path=r["path"],
|
|
425
|
+
kind=r["kind"],
|
|
426
|
+
confidence="MEDIUM",
|
|
427
|
+
reason=f"Defined in {r['path']}",
|
|
428
|
+
)
|
|
429
|
+
for r in rows
|
|
430
|
+
)
|
|
431
|
+
# Check if one candidate is in a primary/entrypoint module (e.g. routes.py, auth.py vs test_auth.py)
|
|
432
|
+
non_test = [c for c in candidates if "test" not in c.path.lower()]
|
|
433
|
+
kinds = {r["kind"] for r in rows}
|
|
434
|
+
if len(kinds) > 1 and len(non_test) > 1:
|
|
435
|
+
return TaskAmbiguity(
|
|
436
|
+
status=AmbiguityStatus.CONFLICT,
|
|
437
|
+
target=clean_target,
|
|
438
|
+
candidates=candidates,
|
|
439
|
+
conflict_detail=f"Target '{clean_target}' has conflicting symbol kinds ({', '.join(sorted(kinds))}) across multiple non-test definitions.",
|
|
440
|
+
)
|
|
441
|
+
|
|
442
|
+
if len(non_test) == 1:
|
|
443
|
+
best = non_test[0]
|
|
444
|
+
return TaskAmbiguity(
|
|
445
|
+
status=AmbiguityStatus.ASSUMED,
|
|
446
|
+
target=clean_target,
|
|
447
|
+
candidates=candidates,
|
|
448
|
+
assumption=f"Target '{clean_target}' assumed to resolve to {best.canonical_id} in {best.path}",
|
|
449
|
+
)
|
|
450
|
+
return TaskAmbiguity(
|
|
451
|
+
status=AmbiguityStatus.AMBIGUOUS,
|
|
452
|
+
target=clean_target,
|
|
453
|
+
candidates=candidates,
|
|
454
|
+
)
|
|
455
|
+
|
|
456
|
+
# Check file path match
|
|
457
|
+
file_row = con.execute(
|
|
458
|
+
"SELECT path FROM files WHERE path=? OR path LIKE ?",
|
|
459
|
+
(clean_target, f"%/{clean_target}"),
|
|
460
|
+
).fetchone()
|
|
461
|
+
if file_row:
|
|
462
|
+
cand = AmbiguityCandidate(
|
|
463
|
+
canonical_id=file_row["path"],
|
|
464
|
+
name=clean_target,
|
|
465
|
+
path=file_row["path"],
|
|
466
|
+
kind="file",
|
|
467
|
+
confidence="HIGH",
|
|
468
|
+
reason="Matched repository file path",
|
|
469
|
+
)
|
|
470
|
+
return TaskAmbiguity(
|
|
471
|
+
status=AmbiguityStatus.CLEAR,
|
|
472
|
+
target=clean_target,
|
|
473
|
+
candidates=(cand,),
|
|
474
|
+
)
|
|
475
|
+
|
|
476
|
+
expr_type = classify_target_expression(clean_target)
|
|
477
|
+
if expr_type == TargetExpressionType.EXPLICIT_SYMBOL_TARGET:
|
|
478
|
+
return TaskAmbiguity(
|
|
479
|
+
status=AmbiguityStatus.UNKNOWN,
|
|
480
|
+
target=clean_target,
|
|
481
|
+
candidates=(),
|
|
482
|
+
)
|
|
483
|
+
return TaskAmbiguity(
|
|
484
|
+
status=AmbiguityStatus.CLEAR,
|
|
485
|
+
target=clean_target,
|
|
486
|
+
candidates=(),
|
|
487
|
+
)
|
|
488
|
+
|
|
489
|
+
|
|
490
|
+
def normalize_task_spec(
|
|
491
|
+
task: TaskSpec | dict[str, Any] | str,
|
|
492
|
+
con: sqlite3.Connection | None = None,
|
|
493
|
+
) -> tuple[TaskSpec, tuple[TaskAmbiguity, ...]]:
|
|
494
|
+
"""Normalize user or agent input into a valid TaskSpec and detected ambiguities."""
|
|
495
|
+
ambiguity_list: list[TaskAmbiguity] = []
|
|
496
|
+
assumptions_list: list[str] = []
|
|
497
|
+
ambiguities_desc: list[str] = []
|
|
498
|
+
unknowns_list: list[str] = []
|
|
499
|
+
conflicts_list: list[str] = []
|
|
500
|
+
|
|
501
|
+
if isinstance(task, TaskSpec):
|
|
502
|
+
raw_spec = task
|
|
503
|
+
conflicts_list.extend(raw_spec.conflicts)
|
|
504
|
+
elif isinstance(task, dict):
|
|
505
|
+
raw_intent = task.get("intent")
|
|
506
|
+
canonical_intent = canonicalize_intent(str(raw_intent) if raw_intent else None)
|
|
507
|
+
targets_in = tuple(str(t).strip() for t in task.get("targets", []) if str(t).strip())
|
|
508
|
+
conflicts_in = tuple(str(cf).strip() for cf in task.get("conflicts", []) if str(cf).strip())
|
|
509
|
+
conflicts_list.extend(conflicts_in)
|
|
510
|
+
raw_spec = TaskSpec(
|
|
511
|
+
schema_version=str(task.get("schema_version", TASK_SPEC_SCHEMA_VERSION)),
|
|
512
|
+
raw_prompt=task.get("raw_prompt") or task.get("goal") or task.get("task"),
|
|
513
|
+
intent=canonical_intent.value,
|
|
514
|
+
goal=str(task.get("goal") or task.get("task") or ""),
|
|
515
|
+
targets=targets_in,
|
|
516
|
+
entities=tuple(str(e).strip() for e in task.get("entities", []) if str(e).strip()),
|
|
517
|
+
operations=tuple(str(o).strip().upper() for o in task.get("operations", []) if str(o).strip()),
|
|
518
|
+
constraints=tuple(str(c).strip().upper() for c in task.get("constraints", []) if str(c).strip()),
|
|
519
|
+
exclusions=tuple(str(x).strip() for x in task.get("exclusions", []) if str(x).strip()),
|
|
520
|
+
scope_paths=tuple(str(p).strip() for p in task.get("scope_paths", []) if str(p).strip()),
|
|
521
|
+
scope_modules=tuple(str(m).strip() for m in task.get("scope_modules", []) if str(m).strip()),
|
|
522
|
+
frameworks=tuple(str(f).strip().lower() for f in task.get("frameworks", []) if str(f).strip()),
|
|
523
|
+
time_scope=task.get("time_scope"),
|
|
524
|
+
priority_targets=tuple(str(pt).strip() for pt in task.get("priority_targets", []) if str(pt).strip()),
|
|
525
|
+
ambiguities=tuple(str(a).strip() for a in task.get("ambiguities", []) if str(a).strip()),
|
|
526
|
+
assumptions=tuple(str(asmp).strip() for asmp in task.get("assumptions", []) if str(asmp).strip()),
|
|
527
|
+
unknowns=tuple(str(u).strip() for u in task.get("unknowns", []) if str(u).strip()),
|
|
528
|
+
conflicts=conflicts_in,
|
|
529
|
+
confidence=str(task.get("confidence", "HIGH")),
|
|
530
|
+
)
|
|
531
|
+
else:
|
|
532
|
+
# Raw string input fallback
|
|
533
|
+
prompt_text = str(task).strip()
|
|
534
|
+
filtered_text = re.sub(
|
|
535
|
+
r"\b(?:do\s+not|don't|without)\s+(?:modify|edit|change)\b",
|
|
536
|
+
"",
|
|
537
|
+
prompt_text,
|
|
538
|
+
flags=re.IGNORECASE,
|
|
539
|
+
)
|
|
540
|
+
detected_intent = TaskIntent.UNDERSTAND
|
|
541
|
+
for term, intent_enum in _INTENT_PRIORITY_ORDER:
|
|
542
|
+
if re.search(rf"\b{re.escape(term)}\b", filtered_text, re.IGNORECASE):
|
|
543
|
+
detected_intent = intent_enum
|
|
544
|
+
break
|
|
545
|
+
|
|
546
|
+
extracted_targets = decompose_prompt(prompt_text)
|
|
547
|
+
operations: list[str] = []
|
|
548
|
+
if detected_intent == TaskIntent.TRACE:
|
|
549
|
+
operations.append("TRACE")
|
|
550
|
+
if "test" in prompt_text.lower():
|
|
551
|
+
operations.append("FIND_RELATED_TESTS")
|
|
552
|
+
if "recent" in prompt_text.lower() or "yesterday" in prompt_text.lower():
|
|
553
|
+
operations.append("INSPECT_RECENT_CHANGES")
|
|
554
|
+
|
|
555
|
+
constraints: list[str] = []
|
|
556
|
+
if "don't modify" in prompt_text.lower() or "read only" in prompt_text.lower() or "do not modify" in prompt_text.lower():
|
|
557
|
+
constraints.append("READ_ONLY")
|
|
558
|
+
|
|
559
|
+
time_scope = "RECENT_CHANGES" if ("recent" in prompt_text.lower() or "yesterday" in prompt_text.lower()) else None
|
|
560
|
+
|
|
561
|
+
raw_spec = TaskSpec(
|
|
562
|
+
raw_prompt=prompt_text,
|
|
563
|
+
intent=detected_intent.value,
|
|
564
|
+
goal=prompt_text,
|
|
565
|
+
targets=tuple(extracted_targets[:10]),
|
|
566
|
+
operations=tuple(operations),
|
|
567
|
+
constraints=tuple(constraints),
|
|
568
|
+
time_scope=time_scope,
|
|
569
|
+
)
|
|
570
|
+
|
|
571
|
+
# Check for direct exclusions conflict
|
|
572
|
+
for exc in raw_spec.exclusions:
|
|
573
|
+
if exc in raw_spec.targets or exc in raw_spec.priority_targets:
|
|
574
|
+
conflicts_list.append(f"Target '{exc}' is listed in both targets and exclusions.")
|
|
575
|
+
|
|
576
|
+
# Perform repository target grounding & ambiguity check if DB connection is available
|
|
577
|
+
grounded_targets: list[str] = []
|
|
578
|
+
for tgt in raw_spec.targets:
|
|
579
|
+
amb = detect_target_ambiguity(tgt, con, exclusions=raw_spec.exclusions)
|
|
580
|
+
ambiguity_list.append(amb)
|
|
581
|
+
if amb.status == AmbiguityStatus.CONFLICT:
|
|
582
|
+
if amb.conflict_detail:
|
|
583
|
+
conflicts_list.append(amb.conflict_detail)
|
|
584
|
+
grounded_targets.append(tgt)
|
|
585
|
+
elif amb.status == AmbiguityStatus.AMBIGUOUS:
|
|
586
|
+
ambiguities_desc.append(
|
|
587
|
+
f"Target '{tgt}' is ambiguous across {len(amb.candidates)} candidates: "
|
|
588
|
+
+ ", ".join(c.canonical_id for c in amb.candidates[:3])
|
|
589
|
+
)
|
|
590
|
+
grounded_targets.append(tgt)
|
|
591
|
+
elif amb.status == AmbiguityStatus.ASSUMED:
|
|
592
|
+
if amb.assumption:
|
|
593
|
+
assumptions_list.append(amb.assumption)
|
|
594
|
+
grounded_targets.append(tgt)
|
|
595
|
+
elif amb.status == AmbiguityStatus.UNKNOWN:
|
|
596
|
+
unknowns_list.append(f"Target '{tgt}' does not match any indexed repository symbol or route.")
|
|
597
|
+
grounded_targets.append(tgt)
|
|
598
|
+
else:
|
|
599
|
+
grounded_targets.append(tgt)
|
|
600
|
+
|
|
601
|
+
combined_assumptions = tuple(dict.fromkeys(list(raw_spec.assumptions) + assumptions_list))
|
|
602
|
+
combined_ambiguities = tuple(dict.fromkeys(list(raw_spec.ambiguities) + ambiguities_desc))
|
|
603
|
+
combined_unknowns = tuple(dict.fromkeys(list(raw_spec.unknowns) + unknowns_list))
|
|
604
|
+
combined_conflicts = tuple(dict.fromkeys(conflicts_list))
|
|
605
|
+
|
|
606
|
+
priority_grounded = [
|
|
607
|
+
tgt for tgt, amb in zip(raw_spec.targets, ambiguity_list, strict=False)
|
|
608
|
+
if amb.status != AmbiguityStatus.UNKNOWN
|
|
609
|
+
]
|
|
610
|
+
effective_priority_targets = (
|
|
611
|
+
raw_spec.priority_targets
|
|
612
|
+
or (tuple(priority_grounded[:3]) if priority_grounded else tuple(grounded_targets[:3]))
|
|
613
|
+
)
|
|
614
|
+
|
|
615
|
+
final_spec = TaskSpec(
|
|
616
|
+
schema_version=raw_spec.schema_version,
|
|
617
|
+
raw_prompt=raw_spec.raw_prompt,
|
|
618
|
+
intent=raw_spec.intent,
|
|
619
|
+
goal=raw_spec.goal,
|
|
620
|
+
targets=tuple(dict.fromkeys(grounded_targets)),
|
|
621
|
+
entities=raw_spec.entities,
|
|
622
|
+
operations=raw_spec.operations,
|
|
623
|
+
constraints=raw_spec.constraints,
|
|
624
|
+
exclusions=raw_spec.exclusions,
|
|
625
|
+
scope_paths=raw_spec.scope_paths,
|
|
626
|
+
scope_modules=raw_spec.scope_modules,
|
|
627
|
+
frameworks=raw_spec.frameworks,
|
|
628
|
+
time_scope=raw_spec.time_scope,
|
|
629
|
+
priority_targets=effective_priority_targets,
|
|
630
|
+
ambiguities=combined_ambiguities,
|
|
631
|
+
assumptions=combined_assumptions,
|
|
632
|
+
unknowns=combined_unknowns,
|
|
633
|
+
conflicts=combined_conflicts,
|
|
634
|
+
confidence="LOW" if (combined_ambiguities or combined_conflicts) else ("MEDIUM" if combined_assumptions else "HIGH"),
|
|
635
|
+
)
|
|
636
|
+
|
|
637
|
+
return final_spec, tuple(ambiguity_list)
|