codegraph-engine 2.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. codegraph/__init__.py +37 -0
  2. codegraph/agent.py +26 -0
  3. codegraph/architecture.py +328 -0
  4. codegraph/audit.py +106 -0
  5. codegraph/cache.py +95 -0
  6. codegraph/cli.py +854 -0
  7. codegraph/config.py +43 -0
  8. codegraph/constraints.py +238 -0
  9. codegraph/context.py +1228 -0
  10. codegraph/epistemic.py +90 -0
  11. codegraph/errors.py +275 -0
  12. codegraph/evidence/__init__.py +15 -0
  13. codegraph/evidence/citations.py +397 -0
  14. codegraph/frameworks.py +434 -0
  15. codegraph/freshness.py +295 -0
  16. codegraph/git.py +278 -0
  17. codegraph/graph/__init__.py +46 -0
  18. codegraph/graph/models.py +41 -0
  19. codegraph/graph/traversal.py +1291 -0
  20. codegraph/indexing/__init__.py +4 -0
  21. codegraph/indexing/classifier.py +274 -0
  22. codegraph/indexing/indexer.py +943 -0
  23. codegraph/indexing/models.py +338 -0
  24. codegraph/indexing/parser.py +1240 -0
  25. codegraph/indexing/scanner.py +200 -0
  26. codegraph/indexing/test_framework.py +116 -0
  27. codegraph/interrogation.py +1582 -0
  28. codegraph/llm/__init__.py +3 -0
  29. codegraph/llm/base.py +15 -0
  30. codegraph/llm/context.py +20 -0
  31. codegraph/mcp/__init__.py +3 -0
  32. codegraph/mcp/server.py +736 -0
  33. codegraph/memory/__init__.py +3 -0
  34. codegraph/memory/store.py +46 -0
  35. codegraph/models.py +289 -0
  36. codegraph/observability.py +151 -0
  37. codegraph/optimizer.py +372 -0
  38. codegraph/planner.py +417 -0
  39. codegraph/py.typed +1 -0
  40. codegraph/query_expansion.py +199 -0
  41. codegraph/ranking.py +363 -0
  42. codegraph/resolver.py +843 -0
  43. codegraph/resources/__init__.py +45 -0
  44. codegraph/resources/cache.py +117 -0
  45. codegraph/resources/coalescer.py +83 -0
  46. codegraph/resources/debouncer.py +98 -0
  47. codegraph/resources/governor.py +232 -0
  48. codegraph/resources/policy.py +123 -0
  49. codegraph/retrieval_policy.py +220 -0
  50. codegraph/search/__init__.py +23 -0
  51. codegraph/search/hybrid.py +301 -0
  52. codegraph/search/semantic.py +28 -0
  53. codegraph/security/__init__.py +3 -0
  54. codegraph/security/paths.py +35 -0
  55. codegraph/target_resolver.py +348 -0
  56. codegraph/task.py +637 -0
  57. codegraph_engine-2.1.1.dist-info/METADATA +334 -0
  58. codegraph_engine-2.1.1.dist-info/RECORD +62 -0
  59. codegraph_engine-2.1.1.dist-info/WHEEL +5 -0
  60. codegraph_engine-2.1.1.dist-info/entry_points.txt +2 -0
  61. codegraph_engine-2.1.1.dist-info/licenses/LICENSE +21 -0
  62. codegraph_engine-2.1.1.dist-info/top_level.txt +1 -0
codegraph/task.py ADDED
@@ -0,0 +1,637 @@
1
+ """Canonical TaskSpec, Ambiguity, and Task-Understanding Contract.
2
+
3
+ The user's AI model understands natural language and creates a structured TaskSpec.
4
+ CodeGraph MCP deterministically normalizes, validates, and grounds the TaskSpec against
5
+ actual repository facts, symbols, routes, and files with zero LLM API dependency.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ import hashlib
10
+ import json
11
+ import re
12
+ import sqlite3
13
+ from dataclasses import asdict, dataclass
14
+ from enum import StrEnum
15
+ from typing import Any
16
+
17
+ TASK_SPEC_SCHEMA_VERSION = "1.0"
18
+
19
+
20
+ class TaskIntent(StrEnum):
21
+ UNDERSTAND = "UNDERSTAND"
22
+ DEBUG = "DEBUG"
23
+ CHANGE = "CHANGE"
24
+ REFACTOR = "REFACTOR"
25
+ TRACE = "TRACE"
26
+ IMPACT = "IMPACT"
27
+ REVIEW = "REVIEW"
28
+ TEST = "TEST"
29
+ ARCHITECTURE = "ARCHITECTURE"
30
+ EXPLAIN = "EXPLAIN"
31
+
32
+
33
+ _INTENT_PRIORITY_ORDER: list[tuple[str, TaskIntent]] = [
34
+ ("blast_radius", TaskIntent.IMPACT),
35
+ ("impact", TaskIntent.IMPACT),
36
+ ("trace", TaskIntent.TRACE),
37
+ ("callgraph", TaskIntent.TRACE),
38
+ ("architecture", TaskIntent.ARCHITECTURE),
39
+ ("overview", TaskIntent.ARCHITECTURE),
40
+ ("refactor", TaskIntent.REFACTOR),
41
+ ("restructure", TaskIntent.REFACTOR),
42
+ ("review", TaskIntent.REVIEW),
43
+ ("diff", TaskIntent.REVIEW),
44
+ ("debug", TaskIntent.DEBUG),
45
+ ("troubleshoot", TaskIntent.DEBUG),
46
+ ("investigate", TaskIntent.DEBUG),
47
+ ("test", TaskIntent.TEST),
48
+ ("tests", TaskIntent.TEST),
49
+ ("change", TaskIntent.CHANGE),
50
+ ("modify", TaskIntent.CHANGE),
51
+ ("edit", TaskIntent.CHANGE),
52
+ ("fix", TaskIntent.CHANGE),
53
+ ("patch", TaskIntent.CHANGE),
54
+ ("understand", TaskIntent.UNDERSTAND),
55
+ ("explain", TaskIntent.EXPLAIN),
56
+ ]
57
+
58
+ _INTENT_SYNONYMS: dict[str, TaskIntent] = dict(_INTENT_PRIORITY_ORDER)
59
+
60
+
61
+ class AmbiguityStatus(StrEnum):
62
+ CLEAR = "CLEAR"
63
+ ASSUMED = "ASSUMED"
64
+ AMBIGUOUS = "AMBIGUOUS"
65
+ UNKNOWN = "UNKNOWN"
66
+ CONFLICT = "CONFLICT"
67
+
68
+
69
+ @dataclass(frozen=True)
70
+ class AmbiguityCandidate:
71
+ canonical_id: str
72
+ name: str
73
+ path: str
74
+ kind: str
75
+ confidence: str = "HIGH" # HIGH | MEDIUM | LOW
76
+ reason: str = ""
77
+
78
+ def as_dict(self) -> dict[str, object]:
79
+ return asdict(self)
80
+
81
+
82
+ @dataclass(frozen=True)
83
+ class TaskAmbiguity:
84
+ status: AmbiguityStatus
85
+ target: str
86
+ candidates: tuple[AmbiguityCandidate, ...] = ()
87
+ assumption: str | None = None
88
+ conflict_detail: str | None = None
89
+
90
+ def as_dict(self) -> dict[str, object]:
91
+ return {
92
+ "status": self.status.value,
93
+ "target": self.target,
94
+ "candidates": [c.as_dict() for c in self.candidates],
95
+ "assumption": self.assumption,
96
+ "conflict_detail": self.conflict_detail,
97
+ }
98
+
99
+
100
+ @dataclass(frozen=True)
101
+ class TaskSpec:
102
+ schema_version: str = TASK_SPEC_SCHEMA_VERSION
103
+ raw_prompt: str | None = None
104
+
105
+ intent: str = TaskIntent.UNDERSTAND.value
106
+ goal: str = ""
107
+
108
+ targets: tuple[str, ...] = ()
109
+ entities: tuple[str, ...] = ()
110
+
111
+ operations: tuple[str, ...] = ()
112
+ constraints: tuple[str, ...] = ()
113
+ exclusions: tuple[str, ...] = ()
114
+
115
+ scope_paths: tuple[str, ...] = ()
116
+ scope_modules: tuple[str, ...] = ()
117
+ frameworks: tuple[str, ...] = ()
118
+
119
+ time_scope: str | None = None
120
+
121
+ priority_targets: tuple[str, ...] = ()
122
+
123
+ ambiguities: tuple[str, ...] = ()
124
+ assumptions: tuple[str, ...] = ()
125
+ unknowns: tuple[str, ...] = ()
126
+ conflicts: tuple[str, ...] = ()
127
+
128
+ confidence: str = "HIGH"
129
+
130
+ def as_dict(self) -> dict[str, object]:
131
+ return asdict(self)
132
+
133
+ def fingerprint(self) -> str:
134
+ """Compute deterministic SHA-256 fingerprint for this TaskSpec."""
135
+ canonical = json.dumps(
136
+ {
137
+ "intent": self.intent,
138
+ "goal": self.goal,
139
+ "targets": sorted(self.targets),
140
+ "entities": sorted(self.entities),
141
+ "operations": sorted(self.operations),
142
+ "constraints": sorted(self.constraints),
143
+ "exclusions": sorted(self.exclusions),
144
+ "scope_paths": sorted(self.scope_paths),
145
+ "scope_modules": sorted(self.scope_modules),
146
+ "frameworks": sorted(self.frameworks),
147
+ "time_scope": self.time_scope or "",
148
+ },
149
+ sort_keys=True,
150
+ )
151
+ return hashlib.sha256(canonical.encode("utf-8")).hexdigest()
152
+
153
+
154
+ def canonicalize_intent(raw_intent: str | None) -> TaskIntent:
155
+ """Deterministically map raw or case-insensitive intent strings to canonical TaskIntent."""
156
+ if not raw_intent:
157
+ return TaskIntent.UNDERSTAND
158
+ cleaned = raw_intent.strip().lower().replace("-", "_")
159
+ return _INTENT_SYNONYMS.get(cleaned, TaskIntent.UNDERSTAND)
160
+
161
+
162
+ class TargetExpressionType(StrEnum):
163
+ EXPLICIT_SYMBOL_TARGET = "EXPLICIT_SYMBOL_TARGET"
164
+ CONCEPT_TARGET = "CONCEPT_TARGET"
165
+ NATURAL_LANGUAGE_TERM = "NATURAL_LANGUAGE_TERM"
166
+
167
+
168
+ _COMMON_NL_WORDS = frozenset({
169
+ "the", "and", "for", "with", "this", "that", "from", "check", "verify", "trace",
170
+ "show", "what", "where", "how", "when", "does", "after", "before", "code", "file",
171
+ "route", "routes", "flow", "work", "works", "working", "database", "method", "class",
172
+ "function", "explain", "understand", "modify", "refactor", "debug", "impact", "review",
173
+ "find", "all", "case", "cases", "covering", "into", "over", "about", "such",
174
+ "input", "inputs", "output", "outputs", "final", "results", "result", "user", "users",
175
+ "data", "system", "systems", "handle", "handling", "start", "starting", "run", "running",
176
+ "call", "calling", "test", "tests", "view", "views", "item", "items", "order", "orders",
177
+ "list", "make", "need", "needs", "pass", "fail", "true", "false", "value", "values",
178
+ "key", "keys", "main", "step", "steps", "path", "paths", "helper", "helpers",
179
+ "service", "services", "component", "components", "feature", "features", "action", "actions",
180
+ "details", "detail", "overview", "summary", "create", "update", "delete",
181
+ })
182
+
183
+
184
+ def classify_target_expression(expr: str) -> TargetExpressionType:
185
+ """Classify a target expression into symbol, concept, or natural language term."""
186
+ clean = expr.strip()
187
+ if not clean:
188
+ return TargetExpressionType.NATURAL_LANGUAGE_TERM
189
+
190
+ # 1. Explicit syntax markers
191
+ if "::" in clean:
192
+ return TargetExpressionType.EXPLICIT_SYMBOL_TARGET
193
+ if "." in clean:
194
+ return TargetExpressionType.EXPLICIT_SYMBOL_TARGET
195
+ if clean.startswith("/") or re.match(r"^(?:GET|POST|PUT|DELETE|PATCH|OPTIONS|HEAD)\s+/", clean, re.IGNORECASE):
196
+ return TargetExpressionType.EXPLICIT_SYMBOL_TARGET
197
+ if "/" in clean and any(clean.endswith(ext) for ext in (".py", ".js", ".jsx", ".ts", ".tsx", ".json", ".html")):
198
+ return TargetExpressionType.EXPLICIT_SYMBOL_TARGET
199
+
200
+ # 2. Multi-word phrases
201
+ if " " in clean or "\t" in clean:
202
+ words = [w.lower() for w in clean.split()]
203
+ if all(w in _COMMON_NL_WORDS for w in words):
204
+ return TargetExpressionType.NATURAL_LANGUAGE_TERM
205
+ return TargetExpressionType.CONCEPT_TARGET
206
+
207
+ # 3. Code naming conventions
208
+ if "_" in clean:
209
+ return TargetExpressionType.EXPLICIT_SYMBOL_TARGET
210
+
211
+ # CamelCase (lowercase followed by uppercase, e.g. handleRequest)
212
+ if re.search(r"[a-z][A-Z]", clean):
213
+ return TargetExpressionType.EXPLICIT_SYMBOL_TARGET
214
+
215
+ # PascalCase (starts uppercase, contains lowercase, length >= 3, e.g. AuthService, FakePayService)
216
+ if clean[0].isupper() and any(c.islower() for c in clean[1:]) and len(clean) >= 3:
217
+ if clean.lower() not in _COMMON_NL_WORDS:
218
+ return TargetExpressionType.EXPLICIT_SYMBOL_TARGET
219
+ return TargetExpressionType.NATURAL_LANGUAGE_TERM
220
+
221
+ # 4. Check against common natural language words
222
+ if clean.lower() in _COMMON_NL_WORDS:
223
+ return TargetExpressionType.NATURAL_LANGUAGE_TERM
224
+
225
+ # 5. Plain single word not in stopwords (e.g. authenticate, login)
226
+ # Returns CONCEPT_TARGET: can match symbols in DB if present, but does not trigger UNKNOWN if absent
227
+ return TargetExpressionType.CONCEPT_TARGET
228
+
229
+
230
+ _CODE_TOKEN_RE = re.compile(r"\b([A-Za-z_][A-Za-z0-9_]{2,}(?:\.[A-Za-z_][A-Za-z0-9_]*)*)\b")
231
+ _ROUTE_TOKEN_RE = re.compile(r"\b(?:GET|POST|PUT|DELETE|PATCH|OPTIONS|HEAD)\s+(/[A-Za-z0-9_/{}\-.:*]*)", re.IGNORECASE)
232
+ _PATH_TOKEN_RE = re.compile(r"\b([A-Za-z0-9_.\-]+/[A-Za-z0-9_.\-/]+)\b")
233
+
234
+
235
+ def extract_keywords_from_prompt(prompt: str) -> list[str]:
236
+ """Deterministically extract candidate identifiers, endpoints, and terms from a prompt."""
237
+ candidates: list[str] = []
238
+
239
+ # 1. Quoted terms
240
+ for quoted in re.findall(r"['\"]([^'\"]+)['\"]", prompt):
241
+ q = quoted.strip()
242
+ if q and q not in candidates:
243
+ if classify_target_expression(q) != TargetExpressionType.NATURAL_LANGUAGE_TERM:
244
+ candidates.append(q)
245
+ elif len(q.split()) > 1:
246
+ candidates.append(q)
247
+
248
+ # 2. HTTP endpoints (e.g. POST /login, /api/v1/auth)
249
+ for m in _ROUTE_TOKEN_RE.finditer(prompt):
250
+ route_str = m.group(0).strip()
251
+ if route_str not in candidates:
252
+ candidates.append(route_str)
253
+
254
+ # 3. File paths
255
+ for p in _PATH_TOKEN_RE.findall(prompt):
256
+ if p not in candidates and not p.startswith("http"):
257
+ candidates.append(p)
258
+
259
+ # 4. Code identifiers (camelCase, snake_case, dotted symbols, potential functions)
260
+ for sym in _CODE_TOKEN_RE.findall(prompt):
261
+ if len(sym) >= 3:
262
+ expr_type = classify_target_expression(sym)
263
+ if expr_type in (TargetExpressionType.EXPLICIT_SYMBOL_TARGET, TargetExpressionType.CONCEPT_TARGET):
264
+ if sym not in candidates:
265
+ candidates.append(sym)
266
+
267
+ return candidates
268
+
269
+
270
+ def decompose_prompt(prompt: str) -> list[str]:
271
+ """Decompose complex, multi-concept prompts into structured conceptual search targets.
272
+
273
+ Splits multi-clause instructions, filters command boilerplate and instructions,
274
+ and returns deduplicated technical concepts and identifiers.
275
+ """
276
+ cleaned = re.sub(
277
+ r"\b(?:investigate why|check|verify|trace|find|show me|look for|please|don't modify anything|do not modify|do not change|read only)\b",
278
+ "",
279
+ prompt,
280
+ flags=re.IGNORECASE,
281
+ ).strip()
282
+
283
+ tokens = extract_keywords_from_prompt(prompt)
284
+ results: list[str] = list(tokens)
285
+
286
+ # Split into clause chunks by commas, semicolons, and clause boundaries
287
+ clauses = re.split(r"[,;\n\.\?!]+|\band\b", cleaned, flags=re.IGNORECASE)
288
+ for clause in clauses:
289
+ c = clause.strip()
290
+ c = re.sub(r"^(?:the|a|an|why|how|what|where|when|all|any)\s+", "", c, flags=re.IGNORECASE).strip()
291
+ if len(c) >= 3 and len(c.split()) <= 4:
292
+ expr_type = classify_target_expression(c)
293
+ if expr_type != TargetExpressionType.NATURAL_LANGUAGE_TERM:
294
+ if not any(c.lower() == r.lower() for r in results):
295
+ results.append(c)
296
+
297
+ deduped: list[str] = []
298
+ seen: set[str] = set()
299
+ for item in results:
300
+ norm = item.strip()
301
+ if norm and norm.lower() not in seen:
302
+ seen.add(norm.lower())
303
+ deduped.append(norm)
304
+
305
+ return deduped
306
+
307
+
308
+ def expand_terms_with_repository(
309
+ terms: list[str] | tuple[str, ...],
310
+ con: sqlite3.Connection | None,
311
+ max_expansions: int = 5,
312
+ ) -> list[str]:
313
+ """Expand user terminology with actual source-backed symbols and routes from the repository.
314
+
315
+ Canonical implementation delegating to `codegraph.query_expansion.expand_query_terms`.
316
+ Only introduces terms confirmed to exist in the repository index.
317
+ """
318
+ if not con or not terms:
319
+ return list(terms)
320
+
321
+ from codegraph.query_expansion import expand_query_terms, get_search_queries
322
+
323
+ expanded_objs = expand_query_terms(terms, con, max_expansions=max_expansions)
324
+ queries = get_search_queries(expanded_objs)
325
+
326
+ expanded: list[str] = list(terms)
327
+ seen: set[str] = {t.lower() for t in terms}
328
+ for q in queries:
329
+ if q.lower() not in seen:
330
+ seen.add(q.lower())
331
+ expanded.append(q)
332
+
333
+ return expanded
334
+
335
+
336
+
337
+ def detect_target_ambiguity(
338
+ target: str,
339
+ con: sqlite3.Connection | None = None,
340
+ exclusions: tuple[str, ...] | None = None,
341
+ ) -> TaskAmbiguity:
342
+ """Ground a target against the repository symbol table and detect potential ambiguities."""
343
+ if not con or not target.strip():
344
+ return TaskAmbiguity(status=AmbiguityStatus.CLEAR, target=target)
345
+
346
+ clean_target = target.strip()
347
+
348
+ # Check for direct specification conflict
349
+ if exclusions and clean_target in exclusions:
350
+ return TaskAmbiguity(
351
+ status=AmbiguityStatus.CONFLICT,
352
+ target=clean_target,
353
+ conflict_detail=f"Target '{clean_target}' is in both targets and exclusions.",
354
+ )
355
+
356
+ # 1. Exact canonical ID check
357
+ exact_canon = con.execute(
358
+ "SELECT canonical_id, name, path, kind FROM symbols WHERE canonical_id=?",
359
+ (clean_target,),
360
+ ).fetchone()
361
+ if exact_canon:
362
+ cand = AmbiguityCandidate(
363
+ canonical_id=exact_canon["canonical_id"],
364
+ name=exact_canon["name"],
365
+ path=exact_canon["path"],
366
+ kind=exact_canon["kind"],
367
+ confidence="HIGH",
368
+ reason="Exact canonical ID match",
369
+ )
370
+ return TaskAmbiguity(
371
+ status=AmbiguityStatus.CLEAR,
372
+ target=clean_target,
373
+ candidates=(cand,),
374
+ )
375
+
376
+ # 2. Framework endpoint check
377
+ if " " in clean_target or clean_target.startswith("/"):
378
+ route_row = con.execute(
379
+ "SELECT endpoint_id, handler_name, file_path FROM framework_routes WHERE route_path=? OR endpoint_id=?",
380
+ (clean_target, clean_target),
381
+ ).fetchone()
382
+ if route_row:
383
+ cand = AmbiguityCandidate(
384
+ canonical_id=route_row["endpoint_id"] or route_row["handler_name"],
385
+ name=route_row["handler_name"],
386
+ path=route_row["file_path"],
387
+ kind="endpoint",
388
+ confidence="HIGH",
389
+ reason="Framework route match",
390
+ )
391
+ return TaskAmbiguity(
392
+ status=AmbiguityStatus.CLEAR,
393
+ target=clean_target,
394
+ candidates=(cand,),
395
+ )
396
+
397
+ # 3. Check symbols by name or qualified_name
398
+ rows = con.execute(
399
+ "SELECT canonical_id, name, path, kind FROM symbols WHERE name=? OR qualified_name=?",
400
+ (clean_target, clean_target),
401
+ ).fetchall()
402
+
403
+ if len(rows) == 1:
404
+ r = rows[0]
405
+ cand = AmbiguityCandidate(
406
+ canonical_id=r["canonical_id"],
407
+ name=r["name"],
408
+ path=r["path"],
409
+ kind=r["kind"],
410
+ confidence="HIGH",
411
+ reason="Uniquely matched symbol in repository",
412
+ )
413
+ return TaskAmbiguity(
414
+ status=AmbiguityStatus.CLEAR,
415
+ target=clean_target,
416
+ candidates=(cand,),
417
+ )
418
+
419
+ if len(rows) > 1:
420
+ candidates = tuple(
421
+ AmbiguityCandidate(
422
+ canonical_id=r["canonical_id"],
423
+ name=r["name"],
424
+ path=r["path"],
425
+ kind=r["kind"],
426
+ confidence="MEDIUM",
427
+ reason=f"Defined in {r['path']}",
428
+ )
429
+ for r in rows
430
+ )
431
+ # Check if one candidate is in a primary/entrypoint module (e.g. routes.py, auth.py vs test_auth.py)
432
+ non_test = [c for c in candidates if "test" not in c.path.lower()]
433
+ kinds = {r["kind"] for r in rows}
434
+ if len(kinds) > 1 and len(non_test) > 1:
435
+ return TaskAmbiguity(
436
+ status=AmbiguityStatus.CONFLICT,
437
+ target=clean_target,
438
+ candidates=candidates,
439
+ conflict_detail=f"Target '{clean_target}' has conflicting symbol kinds ({', '.join(sorted(kinds))}) across multiple non-test definitions.",
440
+ )
441
+
442
+ if len(non_test) == 1:
443
+ best = non_test[0]
444
+ return TaskAmbiguity(
445
+ status=AmbiguityStatus.ASSUMED,
446
+ target=clean_target,
447
+ candidates=candidates,
448
+ assumption=f"Target '{clean_target}' assumed to resolve to {best.canonical_id} in {best.path}",
449
+ )
450
+ return TaskAmbiguity(
451
+ status=AmbiguityStatus.AMBIGUOUS,
452
+ target=clean_target,
453
+ candidates=candidates,
454
+ )
455
+
456
+ # Check file path match
457
+ file_row = con.execute(
458
+ "SELECT path FROM files WHERE path=? OR path LIKE ?",
459
+ (clean_target, f"%/{clean_target}"),
460
+ ).fetchone()
461
+ if file_row:
462
+ cand = AmbiguityCandidate(
463
+ canonical_id=file_row["path"],
464
+ name=clean_target,
465
+ path=file_row["path"],
466
+ kind="file",
467
+ confidence="HIGH",
468
+ reason="Matched repository file path",
469
+ )
470
+ return TaskAmbiguity(
471
+ status=AmbiguityStatus.CLEAR,
472
+ target=clean_target,
473
+ candidates=(cand,),
474
+ )
475
+
476
+ expr_type = classify_target_expression(clean_target)
477
+ if expr_type == TargetExpressionType.EXPLICIT_SYMBOL_TARGET:
478
+ return TaskAmbiguity(
479
+ status=AmbiguityStatus.UNKNOWN,
480
+ target=clean_target,
481
+ candidates=(),
482
+ )
483
+ return TaskAmbiguity(
484
+ status=AmbiguityStatus.CLEAR,
485
+ target=clean_target,
486
+ candidates=(),
487
+ )
488
+
489
+
490
+ def normalize_task_spec(
491
+ task: TaskSpec | dict[str, Any] | str,
492
+ con: sqlite3.Connection | None = None,
493
+ ) -> tuple[TaskSpec, tuple[TaskAmbiguity, ...]]:
494
+ """Normalize user or agent input into a valid TaskSpec and detected ambiguities."""
495
+ ambiguity_list: list[TaskAmbiguity] = []
496
+ assumptions_list: list[str] = []
497
+ ambiguities_desc: list[str] = []
498
+ unknowns_list: list[str] = []
499
+ conflicts_list: list[str] = []
500
+
501
+ if isinstance(task, TaskSpec):
502
+ raw_spec = task
503
+ conflicts_list.extend(raw_spec.conflicts)
504
+ elif isinstance(task, dict):
505
+ raw_intent = task.get("intent")
506
+ canonical_intent = canonicalize_intent(str(raw_intent) if raw_intent else None)
507
+ targets_in = tuple(str(t).strip() for t in task.get("targets", []) if str(t).strip())
508
+ conflicts_in = tuple(str(cf).strip() for cf in task.get("conflicts", []) if str(cf).strip())
509
+ conflicts_list.extend(conflicts_in)
510
+ raw_spec = TaskSpec(
511
+ schema_version=str(task.get("schema_version", TASK_SPEC_SCHEMA_VERSION)),
512
+ raw_prompt=task.get("raw_prompt") or task.get("goal") or task.get("task"),
513
+ intent=canonical_intent.value,
514
+ goal=str(task.get("goal") or task.get("task") or ""),
515
+ targets=targets_in,
516
+ entities=tuple(str(e).strip() for e in task.get("entities", []) if str(e).strip()),
517
+ operations=tuple(str(o).strip().upper() for o in task.get("operations", []) if str(o).strip()),
518
+ constraints=tuple(str(c).strip().upper() for c in task.get("constraints", []) if str(c).strip()),
519
+ exclusions=tuple(str(x).strip() for x in task.get("exclusions", []) if str(x).strip()),
520
+ scope_paths=tuple(str(p).strip() for p in task.get("scope_paths", []) if str(p).strip()),
521
+ scope_modules=tuple(str(m).strip() for m in task.get("scope_modules", []) if str(m).strip()),
522
+ frameworks=tuple(str(f).strip().lower() for f in task.get("frameworks", []) if str(f).strip()),
523
+ time_scope=task.get("time_scope"),
524
+ priority_targets=tuple(str(pt).strip() for pt in task.get("priority_targets", []) if str(pt).strip()),
525
+ ambiguities=tuple(str(a).strip() for a in task.get("ambiguities", []) if str(a).strip()),
526
+ assumptions=tuple(str(asmp).strip() for asmp in task.get("assumptions", []) if str(asmp).strip()),
527
+ unknowns=tuple(str(u).strip() for u in task.get("unknowns", []) if str(u).strip()),
528
+ conflicts=conflicts_in,
529
+ confidence=str(task.get("confidence", "HIGH")),
530
+ )
531
+ else:
532
+ # Raw string input fallback
533
+ prompt_text = str(task).strip()
534
+ filtered_text = re.sub(
535
+ r"\b(?:do\s+not|don't|without)\s+(?:modify|edit|change)\b",
536
+ "",
537
+ prompt_text,
538
+ flags=re.IGNORECASE,
539
+ )
540
+ detected_intent = TaskIntent.UNDERSTAND
541
+ for term, intent_enum in _INTENT_PRIORITY_ORDER:
542
+ if re.search(rf"\b{re.escape(term)}\b", filtered_text, re.IGNORECASE):
543
+ detected_intent = intent_enum
544
+ break
545
+
546
+ extracted_targets = decompose_prompt(prompt_text)
547
+ operations: list[str] = []
548
+ if detected_intent == TaskIntent.TRACE:
549
+ operations.append("TRACE")
550
+ if "test" in prompt_text.lower():
551
+ operations.append("FIND_RELATED_TESTS")
552
+ if "recent" in prompt_text.lower() or "yesterday" in prompt_text.lower():
553
+ operations.append("INSPECT_RECENT_CHANGES")
554
+
555
+ constraints: list[str] = []
556
+ if "don't modify" in prompt_text.lower() or "read only" in prompt_text.lower() or "do not modify" in prompt_text.lower():
557
+ constraints.append("READ_ONLY")
558
+
559
+ time_scope = "RECENT_CHANGES" if ("recent" in prompt_text.lower() or "yesterday" in prompt_text.lower()) else None
560
+
561
+ raw_spec = TaskSpec(
562
+ raw_prompt=prompt_text,
563
+ intent=detected_intent.value,
564
+ goal=prompt_text,
565
+ targets=tuple(extracted_targets[:10]),
566
+ operations=tuple(operations),
567
+ constraints=tuple(constraints),
568
+ time_scope=time_scope,
569
+ )
570
+
571
+ # Check for direct exclusions conflict
572
+ for exc in raw_spec.exclusions:
573
+ if exc in raw_spec.targets or exc in raw_spec.priority_targets:
574
+ conflicts_list.append(f"Target '{exc}' is listed in both targets and exclusions.")
575
+
576
+ # Perform repository target grounding & ambiguity check if DB connection is available
577
+ grounded_targets: list[str] = []
578
+ for tgt in raw_spec.targets:
579
+ amb = detect_target_ambiguity(tgt, con, exclusions=raw_spec.exclusions)
580
+ ambiguity_list.append(amb)
581
+ if amb.status == AmbiguityStatus.CONFLICT:
582
+ if amb.conflict_detail:
583
+ conflicts_list.append(amb.conflict_detail)
584
+ grounded_targets.append(tgt)
585
+ elif amb.status == AmbiguityStatus.AMBIGUOUS:
586
+ ambiguities_desc.append(
587
+ f"Target '{tgt}' is ambiguous across {len(amb.candidates)} candidates: "
588
+ + ", ".join(c.canonical_id for c in amb.candidates[:3])
589
+ )
590
+ grounded_targets.append(tgt)
591
+ elif amb.status == AmbiguityStatus.ASSUMED:
592
+ if amb.assumption:
593
+ assumptions_list.append(amb.assumption)
594
+ grounded_targets.append(tgt)
595
+ elif amb.status == AmbiguityStatus.UNKNOWN:
596
+ unknowns_list.append(f"Target '{tgt}' does not match any indexed repository symbol or route.")
597
+ grounded_targets.append(tgt)
598
+ else:
599
+ grounded_targets.append(tgt)
600
+
601
+ combined_assumptions = tuple(dict.fromkeys(list(raw_spec.assumptions) + assumptions_list))
602
+ combined_ambiguities = tuple(dict.fromkeys(list(raw_spec.ambiguities) + ambiguities_desc))
603
+ combined_unknowns = tuple(dict.fromkeys(list(raw_spec.unknowns) + unknowns_list))
604
+ combined_conflicts = tuple(dict.fromkeys(conflicts_list))
605
+
606
+ priority_grounded = [
607
+ tgt for tgt, amb in zip(raw_spec.targets, ambiguity_list, strict=False)
608
+ if amb.status != AmbiguityStatus.UNKNOWN
609
+ ]
610
+ effective_priority_targets = (
611
+ raw_spec.priority_targets
612
+ or (tuple(priority_grounded[:3]) if priority_grounded else tuple(grounded_targets[:3]))
613
+ )
614
+
615
+ final_spec = TaskSpec(
616
+ schema_version=raw_spec.schema_version,
617
+ raw_prompt=raw_spec.raw_prompt,
618
+ intent=raw_spec.intent,
619
+ goal=raw_spec.goal,
620
+ targets=tuple(dict.fromkeys(grounded_targets)),
621
+ entities=raw_spec.entities,
622
+ operations=raw_spec.operations,
623
+ constraints=raw_spec.constraints,
624
+ exclusions=raw_spec.exclusions,
625
+ scope_paths=raw_spec.scope_paths,
626
+ scope_modules=raw_spec.scope_modules,
627
+ frameworks=raw_spec.frameworks,
628
+ time_scope=raw_spec.time_scope,
629
+ priority_targets=effective_priority_targets,
630
+ ambiguities=combined_ambiguities,
631
+ assumptions=combined_assumptions,
632
+ unknowns=combined_unknowns,
633
+ conflicts=combined_conflicts,
634
+ confidence="LOW" if (combined_ambiguities or combined_conflicts) else ("MEDIUM" if combined_assumptions else "HIGH"),
635
+ )
636
+
637
+ return final_spec, tuple(ambiguity_list)