codegraph-engine 2.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. codegraph/__init__.py +37 -0
  2. codegraph/agent.py +26 -0
  3. codegraph/architecture.py +328 -0
  4. codegraph/audit.py +106 -0
  5. codegraph/cache.py +95 -0
  6. codegraph/cli.py +854 -0
  7. codegraph/config.py +43 -0
  8. codegraph/constraints.py +238 -0
  9. codegraph/context.py +1228 -0
  10. codegraph/epistemic.py +90 -0
  11. codegraph/errors.py +275 -0
  12. codegraph/evidence/__init__.py +15 -0
  13. codegraph/evidence/citations.py +397 -0
  14. codegraph/frameworks.py +434 -0
  15. codegraph/freshness.py +295 -0
  16. codegraph/git.py +278 -0
  17. codegraph/graph/__init__.py +46 -0
  18. codegraph/graph/models.py +41 -0
  19. codegraph/graph/traversal.py +1291 -0
  20. codegraph/indexing/__init__.py +4 -0
  21. codegraph/indexing/classifier.py +274 -0
  22. codegraph/indexing/indexer.py +943 -0
  23. codegraph/indexing/models.py +338 -0
  24. codegraph/indexing/parser.py +1240 -0
  25. codegraph/indexing/scanner.py +200 -0
  26. codegraph/indexing/test_framework.py +116 -0
  27. codegraph/interrogation.py +1582 -0
  28. codegraph/llm/__init__.py +3 -0
  29. codegraph/llm/base.py +15 -0
  30. codegraph/llm/context.py +20 -0
  31. codegraph/mcp/__init__.py +3 -0
  32. codegraph/mcp/server.py +736 -0
  33. codegraph/memory/__init__.py +3 -0
  34. codegraph/memory/store.py +46 -0
  35. codegraph/models.py +289 -0
  36. codegraph/observability.py +151 -0
  37. codegraph/optimizer.py +372 -0
  38. codegraph/planner.py +417 -0
  39. codegraph/py.typed +1 -0
  40. codegraph/query_expansion.py +199 -0
  41. codegraph/ranking.py +363 -0
  42. codegraph/resolver.py +843 -0
  43. codegraph/resources/__init__.py +45 -0
  44. codegraph/resources/cache.py +117 -0
  45. codegraph/resources/coalescer.py +83 -0
  46. codegraph/resources/debouncer.py +98 -0
  47. codegraph/resources/governor.py +232 -0
  48. codegraph/resources/policy.py +123 -0
  49. codegraph/retrieval_policy.py +220 -0
  50. codegraph/search/__init__.py +23 -0
  51. codegraph/search/hybrid.py +301 -0
  52. codegraph/search/semantic.py +28 -0
  53. codegraph/security/__init__.py +3 -0
  54. codegraph/security/paths.py +35 -0
  55. codegraph/target_resolver.py +348 -0
  56. codegraph/task.py +637 -0
  57. codegraph_engine-2.1.1.dist-info/METADATA +334 -0
  58. codegraph_engine-2.1.1.dist-info/RECORD +62 -0
  59. codegraph_engine-2.1.1.dist-info/WHEEL +5 -0
  60. codegraph_engine-2.1.1.dist-info/entry_points.txt +2 -0
  61. codegraph_engine-2.1.1.dist-info/licenses/LICENSE +21 -0
  62. codegraph_engine-2.1.1.dist-info/top_level.txt +1 -0
codegraph/optimizer.py ADDED
@@ -0,0 +1,372 @@
1
+ """Global Token Budget Optimizer, Redundancy Control, and Coverage Engine.
2
+
3
+ Maximizes evidence quality, relationship coverage, freshness, and diversity of useful
4
+ context under a strict token budget constraint while eliminating redundant chunks.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ from dataclasses import dataclass, field
9
+ from typing import Any
10
+
11
+ from codegraph.task import TaskSpec
12
+
13
+
14
+ @dataclass(frozen=True)
15
+ class ContextBudget:
16
+ candidate_tokens: int
17
+ selected_tokens: int
18
+ reduction_ratio: float
19
+ budget_limit: int
20
+ rejections: tuple[dict[str, object], ...] = ()
21
+
22
+ def as_dict(self) -> dict[str, object]:
23
+ return {
24
+ "candidate_tokens": self.candidate_tokens,
25
+ "selected_tokens": self.selected_tokens,
26
+ "reduction_ratio": round(self.reduction_ratio, 4),
27
+ "budget_limit": self.budget_limit,
28
+ "rejections": list(self.rejections),
29
+ }
30
+
31
+
32
+ @dataclass(frozen=True)
33
+ class ContextCandidate:
34
+ candidate_id: str
35
+ canonical_id: str | None
36
+ source_type: str
37
+ estimated_tokens: int
38
+ relevance: float
39
+ evidence_quality: float
40
+ freshness: float
41
+ relationship_value: float
42
+ coverage_value: float
43
+ redundancy_group: str | None
44
+ status: str
45
+ confidence: str
46
+
47
+ def as_dict(self) -> dict[str, object]:
48
+ return {
49
+ "candidate_id": self.candidate_id,
50
+ "canonical_id": self.canonical_id,
51
+ "source_type": self.source_type,
52
+ "estimated_tokens": self.estimated_tokens,
53
+ "relevance": self.relevance,
54
+ "evidence_quality": self.evidence_quality,
55
+ "freshness": self.freshness,
56
+ "relationship_value": self.relationship_value,
57
+ "coverage_value": self.coverage_value,
58
+ "redundancy_group": self.redundancy_group,
59
+ "status": self.status,
60
+ "confidence": self.confidence,
61
+ }
62
+
63
+
64
+ @dataclass
65
+ class CandidateContextItem:
66
+ item_id: str
67
+ file_path: str
68
+ start_line: int
69
+ end_line: int
70
+ canonical_id: str | None
71
+ estimated_tokens: int
72
+ relevance_score: float # 0.0 - 1.0
73
+ evidence_quality: float # 0.0 - 1.0
74
+ freshness: str # FRESH | PARTIALLY_STALE | STALE | UNKNOWN
75
+ coverage_layer: str # ENTRYPOINT | HANDLER | SERVICE | DATA | TEST | GIT | ARCHITECTURE | GENERAL
76
+ source_type: str # chunk | symbol | edge | test | git | route
77
+ snippet: str
78
+ confidence: str = "HIGH"
79
+ relationship_value: float = 0.5
80
+ epistemic_status: str = "FACT" # FACT | ASSUMPTION | INFERENCE | UNKNOWN | CONFLICT
81
+ coverage_value: float = 0.5
82
+ data: dict[str, Any] = field(default_factory=dict)
83
+ why_selected: list[str] = field(default_factory=list)
84
+
85
+ @property
86
+ def relevance(self) -> float:
87
+ return self.relevance_score
88
+
89
+ @property
90
+ def status(self) -> str:
91
+ return self.epistemic_status
92
+
93
+ @property
94
+ def redundancy_group(self) -> str:
95
+ return self.duplicate_key
96
+
97
+ @property
98
+ def duplicate_key(self) -> str:
99
+ """Deterministic grouping key for redundancy detection."""
100
+ if self.canonical_id:
101
+ return f"sym:{self.canonical_id}"
102
+ return f"loc:{self.file_path}:{self.start_line}"
103
+
104
+ def to_context_candidate(self) -> ContextCandidate:
105
+ f_mult = _freshness_multiplier(self.freshness)
106
+ return ContextCandidate(
107
+ candidate_id=self.item_id,
108
+ canonical_id=self.canonical_id,
109
+ source_type=self.source_type,
110
+ estimated_tokens=self.estimated_tokens,
111
+ relevance=self.relevance_score,
112
+ evidence_quality=self.evidence_quality,
113
+ freshness=f_mult,
114
+ relationship_value=self.relationship_value,
115
+ coverage_value=self.coverage_value,
116
+ redundancy_group=self.duplicate_key,
117
+ status=self.epistemic_status,
118
+ confidence=self.confidence,
119
+ )
120
+
121
+
122
+ def _freshness_multiplier(freshness: str) -> float:
123
+ if freshness == "FRESH":
124
+ return 1.0
125
+ if freshness == "PARTIALLY_STALE":
126
+ return 0.7
127
+ if freshness == "STALE":
128
+ return 0.3
129
+ return 0.5
130
+
131
+
132
+ def _confidence_penalty(confidence: str) -> float:
133
+ if confidence == "HIGH":
134
+ return 0.0
135
+ if confidence == "MEDIUM":
136
+ return 0.1
137
+ return 0.25
138
+
139
+
140
+ def _epistemic_multiplier(status: str) -> float:
141
+ if status == "FACT":
142
+ return 1.0
143
+ if status == "ASSUMPTION":
144
+ return 0.85
145
+ if status == "INFERENCE":
146
+ return 0.70
147
+ if status == "UNKNOWN":
148
+ return 0.40
149
+ if status == "CONFLICT":
150
+ return 0.30
151
+ return 0.70
152
+
153
+
154
+ def compute_item_utility(
155
+ item: CandidateContextItem,
156
+ selected_layers: set[str],
157
+ selected_duplicate_keys: set[str],
158
+ ) -> float:
159
+ """Calculate deterministic utility score for a candidate context item."""
160
+ fresh_factor = _freshness_multiplier(item.freshness)
161
+ conf_penalty = _confidence_penalty(item.confidence)
162
+ epistemic_factor = _epistemic_multiplier(item.epistemic_status)
163
+
164
+ # Coverage reward: boost item if it introduces a required layer not yet represented
165
+ coverage_factor = 1.3 if item.coverage_layer not in selected_layers else 1.0
166
+
167
+ # Redundancy penalty: heavily discount items sharing duplicate keys with already selected items
168
+ redundancy_penalty = 0.5 if item.duplicate_key in selected_duplicate_keys else 0.0
169
+
170
+ utility = (
171
+ (item.relevance_score * 0.40)
172
+ + (item.evidence_quality * 0.20)
173
+ + (item.relationship_value * 0.15)
174
+ + (fresh_factor * 0.15)
175
+ + (epistemic_factor * 0.10)
176
+ ) * coverage_factor
177
+
178
+ final_score = max(0.01, utility - redundancy_penalty - conf_penalty)
179
+ return round(final_score, 4)
180
+
181
+
182
+ def _is_excluded(item: CandidateContextItem, task_spec: TaskSpec | None) -> bool:
183
+ """Return True if this candidate violates a hard exclusion constraint.
184
+
185
+ Hierarchical rules:
186
+ - "AuthService" excludes AuthService AND AuthService.login AND AuthService.*
187
+ - "src/admin_panel" excludes any file_path starting with that prefix
188
+ - Canonical-ID comparison preferred over raw string matching
189
+ """
190
+ if not task_spec or not task_spec.exclusions:
191
+ return False
192
+ item_sym = item.canonical_id or str(item.data.get("symbol") or "")
193
+ item_file = item.file_path or ""
194
+ for excl in task_spec.exclusions:
195
+ if not excl:
196
+ continue
197
+ # ── exact symbol match ───────────────────────────────────────────
198
+ if item_sym and (item_sym == excl):
199
+ return True
200
+ # ── hierarchical: AuthService excludes AuthService.anything ──────
201
+ if item_sym and (
202
+ item_sym.startswith(excl + ".") or item_sym.startswith(excl + ":")
203
+ ):
204
+ return True
205
+ # ── exact file match ─────────────────────────────────────────────
206
+ if item_file and item_file == excl:
207
+ return True
208
+ # ── file path prefix (module/directory exclusion) ────────────────
209
+ if item_file and (item_file.startswith(excl + "/") or item_file.startswith(excl + "\\")):
210
+ return True
211
+ # ── file contains the exclusion path segment ─────────────────────
212
+ if item_file and len(excl) >= 4 and excl in item_file:
213
+ return True
214
+ # ── substring match on symbol (kept for backward compat, len >= 4) ─
215
+ if item_sym and len(excl) >= 4 and excl in item_sym:
216
+ return True
217
+ return False
218
+
219
+
220
+ def optimize_context_budget(
221
+ candidates: list[CandidateContextItem],
222
+ token_budget: int,
223
+ task_spec: TaskSpec | None = None,
224
+ ) -> tuple[list[CandidateContextItem], ContextBudget]:
225
+ """Deterministically select the highest-utility, coverage-preserving subset of candidates."""
226
+ if not candidates or token_budget <= 0:
227
+ return [], ContextBudget(
228
+ candidate_tokens=0,
229
+ selected_tokens=0,
230
+ reduction_ratio=0.0,
231
+ budget_limit=token_budget,
232
+ )
233
+
234
+ candidate_tokens_total = sum(c.estimated_tokens for c in candidates)
235
+
236
+ # Initial deterministic sort of candidates:
237
+ # 1. Relevance score descending
238
+ # 2. Evidence quality descending
239
+ # 3. Canonical ID or file path ascending (stable tie-breaker)
240
+ # 4. Start line ascending
241
+ sorted_candidates = sorted(
242
+ candidates,
243
+ key=lambda x: (
244
+ -x.relevance_score,
245
+ -x.evidence_quality,
246
+ x.canonical_id or x.file_path,
247
+ x.start_line,
248
+ ),
249
+ )
250
+
251
+ selected: list[CandidateContextItem] = []
252
+ selected_layers: set[str] = set()
253
+ selected_duplicate_keys: set[str] = set()
254
+ selected_tokens = 0
255
+
256
+ # Pass 1: Coverage pass — ensure distinct layers (e.g. ENTRYPOINT, SERVICE, TEST, GIT)
257
+ # get represented if candidates exist and fit within budget
258
+ for item in sorted_candidates:
259
+ if _is_excluded(item, task_spec):
260
+ continue
261
+ if item.relevance_score < 0.2:
262
+ continue
263
+ if item.coverage_layer not in selected_layers and item.coverage_layer != "GENERAL":
264
+ if selected_tokens + item.estimated_tokens <= token_budget:
265
+ item.why_selected = [
266
+ f"Coverage layer guarantee: '{item.coverage_layer}'",
267
+ f"Relevance: {item.relevance_score:.2f}",
268
+ ]
269
+ selected.append(item)
270
+ selected_tokens += item.estimated_tokens
271
+ selected_layers.add(item.coverage_layer)
272
+ selected_duplicate_keys.add(item.duplicate_key)
273
+
274
+ # Pass 2: Utility-driven knapsack pass for remaining budget
275
+ remaining_candidates = [c for c in sorted_candidates if c not in selected]
276
+
277
+ # Re-score remaining candidates with dynamic utility taking into account chosen layers/duplicates
278
+ scored_pool: list[tuple[float, CandidateContextItem]] = []
279
+ for item in remaining_candidates:
280
+ u = compute_item_utility(item, selected_layers, selected_duplicate_keys)
281
+ scored_pool.append((u, item))
282
+
283
+ # Sort remaining pool deterministically by utility desc, item id asc
284
+ scored_pool.sort(
285
+ key=lambda pair: (
286
+ -pair[0],
287
+ -pair[1].relevance_score,
288
+ pair[1].canonical_id or pair[1].file_path,
289
+ pair[1].start_line,
290
+ )
291
+ )
292
+
293
+ for utility, item in scored_pool:
294
+ if _is_excluded(item, task_spec):
295
+ continue
296
+ # Check redundancy: if item is exact duplicate of something already selected, skip
297
+ if item.duplicate_key in selected_duplicate_keys and utility < 0.2:
298
+ continue
299
+ # Minimum relevance threshold to avoid packing irrelevant items
300
+ if item.relevance_score < 0.35 or utility < 0.25:
301
+ continue
302
+ if selected_tokens + item.estimated_tokens <= token_budget:
303
+ item.why_selected = [
304
+ f"Knapsack utility score: {utility:.2f}",
305
+ f"Relevance: {item.relevance_score:.2f}, Evidence: {item.evidence_quality:.2f}, Freshness: {item.freshness}",
306
+ ]
307
+ selected.append(item)
308
+ selected_tokens += item.estimated_tokens
309
+ selected_layers.add(item.coverage_layer)
310
+ selected_duplicate_keys.add(item.duplicate_key)
311
+
312
+ # Final deterministic ordering for presentation:
313
+ # Coverage layers first (ENTRYPOINT -> HANDLER -> SERVICE -> DATA -> TEST -> GIT -> ARCHITECTURE -> GENERAL),
314
+ # then file path and line number
315
+ _LAYER_ORDER = {
316
+ "ENTRYPOINT": 0,
317
+ "HANDLER": 1,
318
+ "SERVICE": 2,
319
+ "DATA": 3,
320
+ "TEST": 4,
321
+ "GIT": 5,
322
+ "ARCHITECTURE": 6,
323
+ "GENERAL": 7,
324
+ }
325
+
326
+ selected.sort(
327
+ key=lambda x: (
328
+ _LAYER_ORDER.get(x.coverage_layer, 9),
329
+ -x.relevance_score,
330
+ x.file_path,
331
+ x.start_line,
332
+ )
333
+ )
334
+
335
+ reduction = (
336
+ 1.0 - (selected_tokens / max(candidate_tokens_total, 1))
337
+ if candidate_tokens_total > 0
338
+ else 0.0
339
+ )
340
+
341
+ rejections_list: list[dict[str, object]] = []
342
+ selected_ids = {s.item_id for s in selected}
343
+ for item in candidates:
344
+ if item.item_id not in selected_ids:
345
+ if item.duplicate_key in selected_duplicate_keys:
346
+ reason = f"Duplicate of selected symbol/chunk in {item.file_path}"
347
+ elif item.estimated_tokens + selected_tokens > token_budget:
348
+ reason = f"Token budget limit reached ({token_budget} tokens)"
349
+ elif item.freshness == "STALE":
350
+ reason = "Stale source hash on disk"
351
+ else:
352
+ reason = f"Lower utility/relevance ({item.relevance_score:.2f}) compared to selected candidates"
353
+ rejections_list.append(
354
+ {
355
+ "item_id": item.item_id,
356
+ "file": item.file_path,
357
+ "start_line": item.start_line,
358
+ "symbol": item.canonical_id or item.data.get("symbol"),
359
+ "reason": reason,
360
+ "relevance": item.relevance_score,
361
+ }
362
+ )
363
+
364
+ budget = ContextBudget(
365
+ candidate_tokens=candidate_tokens_total,
366
+ selected_tokens=selected_tokens,
367
+ reduction_ratio=reduction,
368
+ budget_limit=token_budget,
369
+ rejections=tuple(rejections_list[:50]),
370
+ )
371
+
372
+ return selected, budget