codegraph-engine 2.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. codegraph/__init__.py +37 -0
  2. codegraph/agent.py +26 -0
  3. codegraph/architecture.py +328 -0
  4. codegraph/audit.py +106 -0
  5. codegraph/cache.py +95 -0
  6. codegraph/cli.py +854 -0
  7. codegraph/config.py +43 -0
  8. codegraph/constraints.py +238 -0
  9. codegraph/context.py +1228 -0
  10. codegraph/epistemic.py +90 -0
  11. codegraph/errors.py +275 -0
  12. codegraph/evidence/__init__.py +15 -0
  13. codegraph/evidence/citations.py +397 -0
  14. codegraph/frameworks.py +434 -0
  15. codegraph/freshness.py +295 -0
  16. codegraph/git.py +278 -0
  17. codegraph/graph/__init__.py +46 -0
  18. codegraph/graph/models.py +41 -0
  19. codegraph/graph/traversal.py +1291 -0
  20. codegraph/indexing/__init__.py +4 -0
  21. codegraph/indexing/classifier.py +274 -0
  22. codegraph/indexing/indexer.py +943 -0
  23. codegraph/indexing/models.py +338 -0
  24. codegraph/indexing/parser.py +1240 -0
  25. codegraph/indexing/scanner.py +200 -0
  26. codegraph/indexing/test_framework.py +116 -0
  27. codegraph/interrogation.py +1582 -0
  28. codegraph/llm/__init__.py +3 -0
  29. codegraph/llm/base.py +15 -0
  30. codegraph/llm/context.py +20 -0
  31. codegraph/mcp/__init__.py +3 -0
  32. codegraph/mcp/server.py +736 -0
  33. codegraph/memory/__init__.py +3 -0
  34. codegraph/memory/store.py +46 -0
  35. codegraph/models.py +289 -0
  36. codegraph/observability.py +151 -0
  37. codegraph/optimizer.py +372 -0
  38. codegraph/planner.py +417 -0
  39. codegraph/py.typed +1 -0
  40. codegraph/query_expansion.py +199 -0
  41. codegraph/ranking.py +363 -0
  42. codegraph/resolver.py +843 -0
  43. codegraph/resources/__init__.py +45 -0
  44. codegraph/resources/cache.py +117 -0
  45. codegraph/resources/coalescer.py +83 -0
  46. codegraph/resources/debouncer.py +98 -0
  47. codegraph/resources/governor.py +232 -0
  48. codegraph/resources/policy.py +123 -0
  49. codegraph/retrieval_policy.py +220 -0
  50. codegraph/search/__init__.py +23 -0
  51. codegraph/search/hybrid.py +301 -0
  52. codegraph/search/semantic.py +28 -0
  53. codegraph/security/__init__.py +3 -0
  54. codegraph/security/paths.py +35 -0
  55. codegraph/target_resolver.py +348 -0
  56. codegraph/task.py +637 -0
  57. codegraph_engine-2.1.1.dist-info/METADATA +334 -0
  58. codegraph_engine-2.1.1.dist-info/RECORD +62 -0
  59. codegraph_engine-2.1.1.dist-info/WHEEL +5 -0
  60. codegraph_engine-2.1.1.dist-info/entry_points.txt +2 -0
  61. codegraph_engine-2.1.1.dist-info/licenses/LICENSE +21 -0
  62. codegraph_engine-2.1.1.dist-info/top_level.txt +1 -0
codegraph/ranking.py ADDED
@@ -0,0 +1,363 @@
1
+ """Context ranking engine — rigorous, explainable, normalized codebase relevance.
2
+
3
+ Ranking Hierarchy (without double counting):
4
+ 1. EXACT_CANONICAL match > EXACT_QUALIFIED match > EXACT_SYMBOL match > NAME_MATCH > PATH_MATCH / TEXT_MATCH
5
+ 2. Verified relationships (CALLS with HIGH confidence) > POSSIBLE_CALLS (LOW confidence)
6
+ 3. Graph distance geometric decay
7
+ 4. Intent-specific weight policies (EXPLAIN, DEBUG, MODIFY, REVIEW, TEST, IMPACT, ARCHITECTURE)
8
+ 5. Freshness weighting (STALE and PARTIALLY_STALE penalties)
9
+ 6. Source-type classification (source > config > doc > generated / vendor)
10
+ 7. Duplicate context penalties
11
+ 8. Stable deterministic tie-breaking: (-score, -evidence_prio, -exact_prio, dist, len(file), file)
12
+ """
13
+ from __future__ import annotations
14
+
15
+ import re
16
+ from dataclasses import dataclass
17
+
18
+
19
+ @dataclass(frozen=True)
20
+ class RankingReason:
21
+ code: str
22
+ label: str
23
+ contribution: float
24
+
25
+
26
+ @dataclass
27
+ class RankedItem:
28
+ file: str
29
+ symbol: str | None
30
+ start_line: int
31
+ end_line: int
32
+ score: float
33
+ reasons: list[RankingReason]
34
+ snippet: str = ""
35
+ token_estimate: int = 0
36
+ canonical_id: str | None = None
37
+
38
+ def as_dict(self) -> dict[str, object]:
39
+ return {
40
+ "file": self.file,
41
+ "symbol": self.symbol,
42
+ "canonical_id": self.canonical_id,
43
+ "start_line": self.start_line,
44
+ "end_line": self.end_line,
45
+ "score": self.score,
46
+ "reasons": [
47
+ {"code": r.code, "label": r.label, "contribution": r.contribution}
48
+ for r in self.reasons
49
+ ],
50
+ "snippet": self.snippet,
51
+ "token_estimate": self.token_estimate,
52
+ }
53
+
54
+ def debug_format(self) -> str:
55
+ out = [f"{self.canonical_id or self.symbol or self.file}"]
56
+ out.append(f"score: {self.score:.2f}")
57
+ pos = [r for r in self.reasons if r.contribution > 0]
58
+ neg = [r for r in self.reasons if r.contribution < 0]
59
+ if pos:
60
+ out.append("\npositive:")
61
+ for r in pos:
62
+ out.append(f"+ {r.label:<24} +{r.contribution:.2f}")
63
+ if neg:
64
+ out.append("\nnegative:")
65
+ for r in neg:
66
+ out.append(f"- {r.label:<24} {r.contribution:.2f}")
67
+ return "\n".join(out)
68
+
69
+
70
+ @dataclass(frozen=True)
71
+ class IntentWeights:
72
+ target_bonus: float
73
+ caller_bonus: float
74
+ callee_bonus: float
75
+ test_bonus: float
76
+ entry_point_bonus: float
77
+ distance_decay: float
78
+ recent_bonus: float
79
+
80
+
81
+ _INTENT_WEIGHTS = {
82
+ "explain": IntentWeights(0.30, 0.10, 0.20, 0.05, 0.10, 0.70, 0.05),
83
+ "debug": IntentWeights(0.20, 0.20, 0.10, 0.15, 0.20, 0.80, 0.15),
84
+ "modify": IntentWeights(0.40, 0.15, 0.15, 0.20, 0.05, 0.60, 0.10),
85
+ "review": IntentWeights(0.20, 0.10, 0.10, 0.20, 0.05, 0.60, 0.20),
86
+ "test": IntentWeights(0.20, 0.10, 0.05, 0.40, 0.05, 0.50, 0.10),
87
+ "impact": IntentWeights(0.30, 0.30, 0.05, 0.10, 0.10, 0.90, 0.00),
88
+ "architecture": IntentWeights(0.15, 0.05, 0.05, 0.05, 0.30, 0.50, 0.00),
89
+ "default": IntentWeights(0.25, 0.10, 0.10, 0.10, 0.10, 0.70, 0.10),
90
+ }
91
+
92
+ _TEST_PATTERNS = re.compile(
93
+ r"(^|[_/])test[_s]?[_/]|[_/]spec[_/]|test[_s]?\.(py|js|ts|jsx|tsx)$|spec\.(py|js|ts|jsx|tsx)$",
94
+ re.IGNORECASE,
95
+ )
96
+
97
+ _ENTRY_PATTERNS = re.compile(
98
+ r"(^|[_/])(main|app|server|index|wsgi|asgi|manage|run)\.(py|js|ts)$",
99
+ re.IGNORECASE,
100
+ )
101
+
102
+ _VENDOR_PATTERNS = re.compile(
103
+ r"(^|[_/])(node_modules|vendor|\.venv|venv|dist|build|generated)[_/]",
104
+ re.IGNORECASE,
105
+ )
106
+
107
+
108
+ def _token_estimate(text: str) -> int:
109
+ return max(1, len(text) // 4)
110
+
111
+
112
+ def rank_candidates(
113
+ candidates: list[dict[str, object]],
114
+ query: str,
115
+ intent: str | None = None,
116
+ target_symbols: set[str] | None = None,
117
+ recent_paths: set[str] | None = None,
118
+ relationship_distance: dict[str, int] | None = None,
119
+ max_results: int = 20,
120
+ allowed_relationships: frozenset[str] | None = None,
121
+ ) -> list[RankedItem]:
122
+ """Deterministically rank candidates returning explainable, normalized scores between 0.0 and 1.0.
123
+
124
+ Args:
125
+ candidates: Raw candidate dicts from search and graph traversal.
126
+ query: The task display string used for lexical matching.
127
+ intent: Task intent string (e.g. "TRACE", "DEBUG", "UNDERSTAND").
128
+ target_symbols: Set of target symbol names/IDs boosted during ranking.
129
+ recent_paths: File paths recently modified (git-aware freshness boost).
130
+ relationship_distance: Pre-computed BFS distances from target symbols.
131
+ max_results: Maximum results to return.
132
+ allowed_relationships: If provided, candidates whose relationship type is
133
+ NOT in this set receive a penalty. From RetrievalPolicy.
134
+ """
135
+ weights = _INTENT_WEIGHTS.get(intent or "", _INTENT_WEIGHTS["default"])
136
+ target_symbols = target_symbols or set()
137
+ recent_paths = recent_paths or set()
138
+ relationship_distance = relationship_distance or {}
139
+
140
+ # Extract terms: dotted terms and word terms
141
+ dotted_terms = {t.lower() for t in re.findall(r"[A-Za-z_][\w.]*\.[\w.]+", query)}
142
+ word_terms = {t.lower() for t in re.findall(r"[A-Za-z_][\w]*", query) if len(t) > 1}
143
+ for ts in target_symbols:
144
+ word_terms.add(ts.lower().split(".")[-1])
145
+ if "." in ts:
146
+ dotted_terms.add(ts.lower())
147
+
148
+ scored_items: list[RankedItem] = []
149
+
150
+ for c in candidates:
151
+ file_ = str(c.get("file", ""))
152
+ canonical_id = str(c.get("canonical_id") or "") or None
153
+ symbol = c.get("symbol") or c.get("qualified_name")
154
+ symbol_str = str(symbol) if symbol else None
155
+ start = int(str(c.get("start_line", 1)))
156
+ end = int(str(c.get("end_line", start)))
157
+ snippet = str(c.get("snippet", "") or c.get("content", ""))[:1200]
158
+ confidence = str(c.get("confidence", "HIGH")).upper()
159
+ freshness = str(c.get("freshness", "FRESH")).upper()
160
+ relationship = str(c.get("relationship", ""))
161
+ base_search_score = float(str(c.get("score", 0.0)))
162
+
163
+ score = 0.0
164
+ reasons: list[RankingReason] = []
165
+
166
+ # 1. Base Symbol Matching (strictly hierarchical to avoid double counting)
167
+ matched_exact = False
168
+ canon_lower = canonical_id.lower() if canonical_id else ""
169
+ sym_lower = symbol_str.lower() if symbol_str else ""
170
+ short_name = sym_lower.split(".")[-1] if sym_lower else ""
171
+
172
+ # Exact canonical match
173
+ if canon_lower and any(dt == canon_lower or dt in canon_lower for dt in dotted_terms):
174
+ val = weights.target_bonus + 0.20
175
+ score += val
176
+ reasons.append(RankingReason("EXACT_CANONICAL", "Exact canonical symbol match", round(val, 3)))
177
+ matched_exact = True
178
+ # Exact qualified name match
179
+ elif sym_lower and any(dt == sym_lower or sym_lower.endswith(f".{dt}") for dt in dotted_terms):
180
+ val = weights.target_bonus + 0.15
181
+ score += val
182
+ reasons.append(RankingReason("EXACT_QUALIFIED", "Exact qualified symbol match", round(val, 3)))
183
+ matched_exact = True
184
+ # Exact symbol name match
185
+ elif short_name and short_name in word_terms:
186
+ val = weights.target_bonus + 0.10
187
+ score += val
188
+ reasons.append(RankingReason("EXACT_SYMBOL", "Exact target symbol", round(val, 3)))
189
+ matched_exact = True
190
+ # Name relevance
191
+ elif (short_name and any(wt in short_name for wt in word_terms)) or (
192
+ sym_lower and any(wt in sym_lower for wt in word_terms)
193
+ ):
194
+ val = min(0.25, max(0.12, base_search_score))
195
+ score += val
196
+ reasons.append(RankingReason("NAME_MATCH", "Symbol name relevance", round(val, 3)))
197
+ # Path relevance
198
+ elif any(wt in file_.lower() for wt in word_terms):
199
+ val = min(0.20, max(0.10, base_search_score))
200
+ score += val
201
+ reasons.append(RankingReason("PATH_MATCH", "File path relevance", round(val, 3)))
202
+ elif base_search_score > 0:
203
+ val = min(0.30, base_search_score)
204
+ score += val
205
+ reasons.append(RankingReason("TEXT_MATCH", "Lexical text match", round(val, 3)))
206
+
207
+ # 2. Graph Relationship Distance & Direction
208
+ sym_key = canonical_id or symbol_str
209
+ if sym_key and sym_key in relationship_distance:
210
+ dist = relationship_distance[sym_key]
211
+ decay = weights.distance_decay ** dist
212
+ val = 0.20 * decay
213
+ score += val
214
+ reasons.append(RankingReason("GRAPH_DISTANCE", f"Graph distance {dist}", round(val, 3)))
215
+ elif relationship == "CALLS":
216
+ val = weights.callee_bonus
217
+ score += val
218
+ reasons.append(RankingReason("DIRECT_CALLEE", "Direct callee", val))
219
+ elif relationship in ("POSSIBLE_CALLS", "CALLERS"):
220
+ multiplier = 1.0 if confidence in ("HIGH", "MEDIUM") else 0.5
221
+ val = round(weights.caller_bonus * multiplier, 3)
222
+ score += val
223
+ reasons.append(
224
+ RankingReason(
225
+ "DIRECT_CALLER" if multiplier == 1.0 else "POSSIBLE_CALLER",
226
+ "Direct caller" if multiplier == 1.0 else "Possible caller",
227
+ val,
228
+ )
229
+ )
230
+ elif relationship == "HANDLED_BY":
231
+ val = weights.target_bonus + 0.10
232
+ score += val
233
+ reasons.append(RankingReason("ENDPOINT_HANDLER", "Endpoint handler", round(val, 3)))
234
+ elif relationship in ("EXTENDS", "IMPLEMENTS"):
235
+ val = 0.10
236
+ score += val
237
+ reasons.append(RankingReason("INHERITANCE", f"Class {relationship.lower()}", val))
238
+
239
+ # 2b. Policy relationship allowlist — penalize relationships not permitted for this intent
240
+ if allowed_relationships and relationship and relationship.upper() not in allowed_relationships:
241
+ penalty = 0.15
242
+ score -= penalty
243
+ reasons.append(RankingReason(
244
+ "RELATIONSHIP_NOT_ALLOWED",
245
+ f"Relationship '{relationship}' not in policy allowlist",
246
+ -penalty,
247
+ ))
248
+
249
+ # 3. Evidence Quality
250
+ if confidence == "HIGH":
251
+ score += 0.10
252
+ reasons.append(RankingReason("EVIDENCE_HIGH", "Verified source evidence", 0.10))
253
+ elif confidence == "MEDIUM":
254
+ score += 0.05
255
+ reasons.append(RankingReason("EVIDENCE_MEDIUM", "Inferred relationship", 0.05))
256
+
257
+ # 4. Freshness
258
+ if freshness == "STALE":
259
+ score -= 0.30
260
+ reasons.append(RankingReason("STALE_SOURCE", "Stale evidence", -0.30))
261
+ elif freshness == "PARTIALLY_STALE":
262
+ score -= 0.10
263
+ reasons.append(RankingReason("PARTIALLY_STALE", "Partially stale evidence", -0.10))
264
+
265
+ # 5. Test Relationship
266
+ is_test = bool(_TEST_PATTERNS.search(file_))
267
+ if is_test and (relationship.startswith("TEST") or matched_exact):
268
+ val = weights.test_bonus
269
+ score += val
270
+ reasons.append(RankingReason("TEST_RELATION", "Related test file", val))
271
+
272
+ # 6. Entry Point & Framework Route
273
+ is_entry = bool(_ENTRY_PATTERNS.search(file_))
274
+ if is_entry:
275
+ val = weights.entry_point_bonus
276
+ score += val
277
+ reasons.append(RankingReason("ENTRY_POINT", "Entry point", val))
278
+
279
+ # 7. Vendor / Generated Penalty
280
+ if _VENDOR_PATTERNS.search(file_):
281
+ score -= 0.30
282
+ reasons.append(RankingReason("VENDOR_PENALTY", "Vendor/generated code penalty", -0.30))
283
+
284
+ # 8. Recency
285
+ if file_ in recent_paths:
286
+ val = weights.recent_bonus
287
+ score += val
288
+ reasons.append(RankingReason("RECENT_CHANGE", "Recently modified", val))
289
+
290
+ score = max(0.0, min(1.0, score))
291
+
292
+ scored_items.append(
293
+ RankedItem(
294
+ file=file_,
295
+ symbol=symbol_str,
296
+ canonical_id=canonical_id,
297
+ start_line=start,
298
+ end_line=end,
299
+ score=round(score, 3),
300
+ reasons=reasons,
301
+ snippet=snippet,
302
+ token_estimate=_token_estimate(snippet),
303
+ )
304
+ )
305
+
306
+ # 9. Duplicate Context Penalty (deterministic pre-sort ensures stable penalty application)
307
+ def pre_sort_key(item: RankedItem) -> tuple[float, str, str]:
308
+ return (-item.score, item.file, str(item.canonical_id or item.symbol or ""))
309
+
310
+ scored_items.sort(key=pre_sort_key)
311
+ seen_symbols: set[str] = set()
312
+ seen_files: set[str] = set()
313
+
314
+ for item in scored_items:
315
+ if item.score <= 0:
316
+ continue
317
+ penalty = 0.0
318
+ sym_key = item.canonical_id or item.symbol
319
+ if sym_key:
320
+ if sym_key in seen_symbols:
321
+ penalty += 0.20
322
+ seen_symbols.add(sym_key)
323
+ if item.file in seen_files:
324
+ penalty += 0.10
325
+ seen_files.add(item.file)
326
+
327
+ if penalty > 0:
328
+ item.score = max(0.0, round(item.score - penalty, 3))
329
+ item.reasons.append(
330
+ RankingReason("DUPLICATE_PENALTY", "Duplicate context penalty", -penalty)
331
+ )
332
+
333
+ # 10. Stable Tie-Breaking Sort
334
+ def sort_key(item: RankedItem) -> tuple[float, int, int, int, int, str]:
335
+ evidence_prio = 0
336
+ if any(r.code == "EVIDENCE_HIGH" for r in item.reasons):
337
+ evidence_prio = 2
338
+ elif any(r.code == "EVIDENCE_MEDIUM" for r in item.reasons):
339
+ evidence_prio = 1
340
+
341
+ exact_prio = 0
342
+ if any(r.code == "EXACT_CANONICAL" for r in item.reasons):
343
+ exact_prio = 3
344
+ elif any(r.code == "EXACT_QUALIFIED" for r in item.reasons):
345
+ exact_prio = 2
346
+ elif any(r.code == "EXACT_SYMBOL" for r in item.reasons):
347
+ exact_prio = 1
348
+
349
+ dist = 999
350
+ for r in item.reasons:
351
+ if r.code == "GRAPH_DISTANCE":
352
+ try:
353
+ dist = int(r.label.split()[-1])
354
+ except ValueError:
355
+ pass
356
+
357
+ return (-item.score, -evidence_prio, -exact_prio, dist, len(item.file), item.file)
358
+
359
+ scored_items.sort(key=sort_key)
360
+ return scored_items[:max_results]
361
+
362
+
363
+ rank = rank_candidates