codegraph-engine 2.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codegraph/__init__.py +37 -0
- codegraph/agent.py +26 -0
- codegraph/architecture.py +328 -0
- codegraph/audit.py +106 -0
- codegraph/cache.py +95 -0
- codegraph/cli.py +854 -0
- codegraph/config.py +43 -0
- codegraph/constraints.py +238 -0
- codegraph/context.py +1228 -0
- codegraph/epistemic.py +90 -0
- codegraph/errors.py +275 -0
- codegraph/evidence/__init__.py +15 -0
- codegraph/evidence/citations.py +397 -0
- codegraph/frameworks.py +434 -0
- codegraph/freshness.py +295 -0
- codegraph/git.py +278 -0
- codegraph/graph/__init__.py +46 -0
- codegraph/graph/models.py +41 -0
- codegraph/graph/traversal.py +1291 -0
- codegraph/indexing/__init__.py +4 -0
- codegraph/indexing/classifier.py +274 -0
- codegraph/indexing/indexer.py +943 -0
- codegraph/indexing/models.py +338 -0
- codegraph/indexing/parser.py +1240 -0
- codegraph/indexing/scanner.py +200 -0
- codegraph/indexing/test_framework.py +116 -0
- codegraph/interrogation.py +1582 -0
- codegraph/llm/__init__.py +3 -0
- codegraph/llm/base.py +15 -0
- codegraph/llm/context.py +20 -0
- codegraph/mcp/__init__.py +3 -0
- codegraph/mcp/server.py +736 -0
- codegraph/memory/__init__.py +3 -0
- codegraph/memory/store.py +46 -0
- codegraph/models.py +289 -0
- codegraph/observability.py +151 -0
- codegraph/optimizer.py +372 -0
- codegraph/planner.py +417 -0
- codegraph/py.typed +1 -0
- codegraph/query_expansion.py +199 -0
- codegraph/ranking.py +363 -0
- codegraph/resolver.py +843 -0
- codegraph/resources/__init__.py +45 -0
- codegraph/resources/cache.py +117 -0
- codegraph/resources/coalescer.py +83 -0
- codegraph/resources/debouncer.py +98 -0
- codegraph/resources/governor.py +232 -0
- codegraph/resources/policy.py +123 -0
- codegraph/retrieval_policy.py +220 -0
- codegraph/search/__init__.py +23 -0
- codegraph/search/hybrid.py +301 -0
- codegraph/search/semantic.py +28 -0
- codegraph/security/__init__.py +3 -0
- codegraph/security/paths.py +35 -0
- codegraph/target_resolver.py +348 -0
- codegraph/task.py +637 -0
- codegraph_engine-2.1.1.dist-info/METADATA +334 -0
- codegraph_engine-2.1.1.dist-info/RECORD +62 -0
- codegraph_engine-2.1.1.dist-info/WHEEL +5 -0
- codegraph_engine-2.1.1.dist-info/entry_points.txt +2 -0
- codegraph_engine-2.1.1.dist-info/licenses/LICENSE +21 -0
- codegraph_engine-2.1.1.dist-info/top_level.txt +1 -0
codegraph/optimizer.py
ADDED
|
@@ -0,0 +1,372 @@
|
|
|
1
|
+
"""Global Token Budget Optimizer, Redundancy Control, and Coverage Engine.
|
|
2
|
+
|
|
3
|
+
Maximizes evidence quality, relationship coverage, freshness, and diversity of useful
|
|
4
|
+
context under a strict token budget constraint while eliminating redundant chunks.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
from dataclasses import dataclass, field
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
from codegraph.task import TaskSpec
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass(frozen=True)
|
|
15
|
+
class ContextBudget:
|
|
16
|
+
candidate_tokens: int
|
|
17
|
+
selected_tokens: int
|
|
18
|
+
reduction_ratio: float
|
|
19
|
+
budget_limit: int
|
|
20
|
+
rejections: tuple[dict[str, object], ...] = ()
|
|
21
|
+
|
|
22
|
+
def as_dict(self) -> dict[str, object]:
|
|
23
|
+
return {
|
|
24
|
+
"candidate_tokens": self.candidate_tokens,
|
|
25
|
+
"selected_tokens": self.selected_tokens,
|
|
26
|
+
"reduction_ratio": round(self.reduction_ratio, 4),
|
|
27
|
+
"budget_limit": self.budget_limit,
|
|
28
|
+
"rejections": list(self.rejections),
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass(frozen=True)
|
|
33
|
+
class ContextCandidate:
|
|
34
|
+
candidate_id: str
|
|
35
|
+
canonical_id: str | None
|
|
36
|
+
source_type: str
|
|
37
|
+
estimated_tokens: int
|
|
38
|
+
relevance: float
|
|
39
|
+
evidence_quality: float
|
|
40
|
+
freshness: float
|
|
41
|
+
relationship_value: float
|
|
42
|
+
coverage_value: float
|
|
43
|
+
redundancy_group: str | None
|
|
44
|
+
status: str
|
|
45
|
+
confidence: str
|
|
46
|
+
|
|
47
|
+
def as_dict(self) -> dict[str, object]:
|
|
48
|
+
return {
|
|
49
|
+
"candidate_id": self.candidate_id,
|
|
50
|
+
"canonical_id": self.canonical_id,
|
|
51
|
+
"source_type": self.source_type,
|
|
52
|
+
"estimated_tokens": self.estimated_tokens,
|
|
53
|
+
"relevance": self.relevance,
|
|
54
|
+
"evidence_quality": self.evidence_quality,
|
|
55
|
+
"freshness": self.freshness,
|
|
56
|
+
"relationship_value": self.relationship_value,
|
|
57
|
+
"coverage_value": self.coverage_value,
|
|
58
|
+
"redundancy_group": self.redundancy_group,
|
|
59
|
+
"status": self.status,
|
|
60
|
+
"confidence": self.confidence,
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
@dataclass
|
|
65
|
+
class CandidateContextItem:
|
|
66
|
+
item_id: str
|
|
67
|
+
file_path: str
|
|
68
|
+
start_line: int
|
|
69
|
+
end_line: int
|
|
70
|
+
canonical_id: str | None
|
|
71
|
+
estimated_tokens: int
|
|
72
|
+
relevance_score: float # 0.0 - 1.0
|
|
73
|
+
evidence_quality: float # 0.0 - 1.0
|
|
74
|
+
freshness: str # FRESH | PARTIALLY_STALE | STALE | UNKNOWN
|
|
75
|
+
coverage_layer: str # ENTRYPOINT | HANDLER | SERVICE | DATA | TEST | GIT | ARCHITECTURE | GENERAL
|
|
76
|
+
source_type: str # chunk | symbol | edge | test | git | route
|
|
77
|
+
snippet: str
|
|
78
|
+
confidence: str = "HIGH"
|
|
79
|
+
relationship_value: float = 0.5
|
|
80
|
+
epistemic_status: str = "FACT" # FACT | ASSUMPTION | INFERENCE | UNKNOWN | CONFLICT
|
|
81
|
+
coverage_value: float = 0.5
|
|
82
|
+
data: dict[str, Any] = field(default_factory=dict)
|
|
83
|
+
why_selected: list[str] = field(default_factory=list)
|
|
84
|
+
|
|
85
|
+
@property
|
|
86
|
+
def relevance(self) -> float:
|
|
87
|
+
return self.relevance_score
|
|
88
|
+
|
|
89
|
+
@property
|
|
90
|
+
def status(self) -> str:
|
|
91
|
+
return self.epistemic_status
|
|
92
|
+
|
|
93
|
+
@property
|
|
94
|
+
def redundancy_group(self) -> str:
|
|
95
|
+
return self.duplicate_key
|
|
96
|
+
|
|
97
|
+
@property
|
|
98
|
+
def duplicate_key(self) -> str:
|
|
99
|
+
"""Deterministic grouping key for redundancy detection."""
|
|
100
|
+
if self.canonical_id:
|
|
101
|
+
return f"sym:{self.canonical_id}"
|
|
102
|
+
return f"loc:{self.file_path}:{self.start_line}"
|
|
103
|
+
|
|
104
|
+
def to_context_candidate(self) -> ContextCandidate:
|
|
105
|
+
f_mult = _freshness_multiplier(self.freshness)
|
|
106
|
+
return ContextCandidate(
|
|
107
|
+
candidate_id=self.item_id,
|
|
108
|
+
canonical_id=self.canonical_id,
|
|
109
|
+
source_type=self.source_type,
|
|
110
|
+
estimated_tokens=self.estimated_tokens,
|
|
111
|
+
relevance=self.relevance_score,
|
|
112
|
+
evidence_quality=self.evidence_quality,
|
|
113
|
+
freshness=f_mult,
|
|
114
|
+
relationship_value=self.relationship_value,
|
|
115
|
+
coverage_value=self.coverage_value,
|
|
116
|
+
redundancy_group=self.duplicate_key,
|
|
117
|
+
status=self.epistemic_status,
|
|
118
|
+
confidence=self.confidence,
|
|
119
|
+
)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _freshness_multiplier(freshness: str) -> float:
|
|
123
|
+
if freshness == "FRESH":
|
|
124
|
+
return 1.0
|
|
125
|
+
if freshness == "PARTIALLY_STALE":
|
|
126
|
+
return 0.7
|
|
127
|
+
if freshness == "STALE":
|
|
128
|
+
return 0.3
|
|
129
|
+
return 0.5
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def _confidence_penalty(confidence: str) -> float:
|
|
133
|
+
if confidence == "HIGH":
|
|
134
|
+
return 0.0
|
|
135
|
+
if confidence == "MEDIUM":
|
|
136
|
+
return 0.1
|
|
137
|
+
return 0.25
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _epistemic_multiplier(status: str) -> float:
|
|
141
|
+
if status == "FACT":
|
|
142
|
+
return 1.0
|
|
143
|
+
if status == "ASSUMPTION":
|
|
144
|
+
return 0.85
|
|
145
|
+
if status == "INFERENCE":
|
|
146
|
+
return 0.70
|
|
147
|
+
if status == "UNKNOWN":
|
|
148
|
+
return 0.40
|
|
149
|
+
if status == "CONFLICT":
|
|
150
|
+
return 0.30
|
|
151
|
+
return 0.70
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def compute_item_utility(
|
|
155
|
+
item: CandidateContextItem,
|
|
156
|
+
selected_layers: set[str],
|
|
157
|
+
selected_duplicate_keys: set[str],
|
|
158
|
+
) -> float:
|
|
159
|
+
"""Calculate deterministic utility score for a candidate context item."""
|
|
160
|
+
fresh_factor = _freshness_multiplier(item.freshness)
|
|
161
|
+
conf_penalty = _confidence_penalty(item.confidence)
|
|
162
|
+
epistemic_factor = _epistemic_multiplier(item.epistemic_status)
|
|
163
|
+
|
|
164
|
+
# Coverage reward: boost item if it introduces a required layer not yet represented
|
|
165
|
+
coverage_factor = 1.3 if item.coverage_layer not in selected_layers else 1.0
|
|
166
|
+
|
|
167
|
+
# Redundancy penalty: heavily discount items sharing duplicate keys with already selected items
|
|
168
|
+
redundancy_penalty = 0.5 if item.duplicate_key in selected_duplicate_keys else 0.0
|
|
169
|
+
|
|
170
|
+
utility = (
|
|
171
|
+
(item.relevance_score * 0.40)
|
|
172
|
+
+ (item.evidence_quality * 0.20)
|
|
173
|
+
+ (item.relationship_value * 0.15)
|
|
174
|
+
+ (fresh_factor * 0.15)
|
|
175
|
+
+ (epistemic_factor * 0.10)
|
|
176
|
+
) * coverage_factor
|
|
177
|
+
|
|
178
|
+
final_score = max(0.01, utility - redundancy_penalty - conf_penalty)
|
|
179
|
+
return round(final_score, 4)
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def _is_excluded(item: CandidateContextItem, task_spec: TaskSpec | None) -> bool:
|
|
183
|
+
"""Return True if this candidate violates a hard exclusion constraint.
|
|
184
|
+
|
|
185
|
+
Hierarchical rules:
|
|
186
|
+
- "AuthService" excludes AuthService AND AuthService.login AND AuthService.*
|
|
187
|
+
- "src/admin_panel" excludes any file_path starting with that prefix
|
|
188
|
+
- Canonical-ID comparison preferred over raw string matching
|
|
189
|
+
"""
|
|
190
|
+
if not task_spec or not task_spec.exclusions:
|
|
191
|
+
return False
|
|
192
|
+
item_sym = item.canonical_id or str(item.data.get("symbol") or "")
|
|
193
|
+
item_file = item.file_path or ""
|
|
194
|
+
for excl in task_spec.exclusions:
|
|
195
|
+
if not excl:
|
|
196
|
+
continue
|
|
197
|
+
# ── exact symbol match ───────────────────────────────────────────
|
|
198
|
+
if item_sym and (item_sym == excl):
|
|
199
|
+
return True
|
|
200
|
+
# ── hierarchical: AuthService excludes AuthService.anything ──────
|
|
201
|
+
if item_sym and (
|
|
202
|
+
item_sym.startswith(excl + ".") or item_sym.startswith(excl + ":")
|
|
203
|
+
):
|
|
204
|
+
return True
|
|
205
|
+
# ── exact file match ─────────────────────────────────────────────
|
|
206
|
+
if item_file and item_file == excl:
|
|
207
|
+
return True
|
|
208
|
+
# ── file path prefix (module/directory exclusion) ────────────────
|
|
209
|
+
if item_file and (item_file.startswith(excl + "/") or item_file.startswith(excl + "\\")):
|
|
210
|
+
return True
|
|
211
|
+
# ── file contains the exclusion path segment ─────────────────────
|
|
212
|
+
if item_file and len(excl) >= 4 and excl in item_file:
|
|
213
|
+
return True
|
|
214
|
+
# ── substring match on symbol (kept for backward compat, len >= 4) ─
|
|
215
|
+
if item_sym and len(excl) >= 4 and excl in item_sym:
|
|
216
|
+
return True
|
|
217
|
+
return False
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def optimize_context_budget(
|
|
221
|
+
candidates: list[CandidateContextItem],
|
|
222
|
+
token_budget: int,
|
|
223
|
+
task_spec: TaskSpec | None = None,
|
|
224
|
+
) -> tuple[list[CandidateContextItem], ContextBudget]:
|
|
225
|
+
"""Deterministically select the highest-utility, coverage-preserving subset of candidates."""
|
|
226
|
+
if not candidates or token_budget <= 0:
|
|
227
|
+
return [], ContextBudget(
|
|
228
|
+
candidate_tokens=0,
|
|
229
|
+
selected_tokens=0,
|
|
230
|
+
reduction_ratio=0.0,
|
|
231
|
+
budget_limit=token_budget,
|
|
232
|
+
)
|
|
233
|
+
|
|
234
|
+
candidate_tokens_total = sum(c.estimated_tokens for c in candidates)
|
|
235
|
+
|
|
236
|
+
# Initial deterministic sort of candidates:
|
|
237
|
+
# 1. Relevance score descending
|
|
238
|
+
# 2. Evidence quality descending
|
|
239
|
+
# 3. Canonical ID or file path ascending (stable tie-breaker)
|
|
240
|
+
# 4. Start line ascending
|
|
241
|
+
sorted_candidates = sorted(
|
|
242
|
+
candidates,
|
|
243
|
+
key=lambda x: (
|
|
244
|
+
-x.relevance_score,
|
|
245
|
+
-x.evidence_quality,
|
|
246
|
+
x.canonical_id or x.file_path,
|
|
247
|
+
x.start_line,
|
|
248
|
+
),
|
|
249
|
+
)
|
|
250
|
+
|
|
251
|
+
selected: list[CandidateContextItem] = []
|
|
252
|
+
selected_layers: set[str] = set()
|
|
253
|
+
selected_duplicate_keys: set[str] = set()
|
|
254
|
+
selected_tokens = 0
|
|
255
|
+
|
|
256
|
+
# Pass 1: Coverage pass — ensure distinct layers (e.g. ENTRYPOINT, SERVICE, TEST, GIT)
|
|
257
|
+
# get represented if candidates exist and fit within budget
|
|
258
|
+
for item in sorted_candidates:
|
|
259
|
+
if _is_excluded(item, task_spec):
|
|
260
|
+
continue
|
|
261
|
+
if item.relevance_score < 0.2:
|
|
262
|
+
continue
|
|
263
|
+
if item.coverage_layer not in selected_layers and item.coverage_layer != "GENERAL":
|
|
264
|
+
if selected_tokens + item.estimated_tokens <= token_budget:
|
|
265
|
+
item.why_selected = [
|
|
266
|
+
f"Coverage layer guarantee: '{item.coverage_layer}'",
|
|
267
|
+
f"Relevance: {item.relevance_score:.2f}",
|
|
268
|
+
]
|
|
269
|
+
selected.append(item)
|
|
270
|
+
selected_tokens += item.estimated_tokens
|
|
271
|
+
selected_layers.add(item.coverage_layer)
|
|
272
|
+
selected_duplicate_keys.add(item.duplicate_key)
|
|
273
|
+
|
|
274
|
+
# Pass 2: Utility-driven knapsack pass for remaining budget
|
|
275
|
+
remaining_candidates = [c for c in sorted_candidates if c not in selected]
|
|
276
|
+
|
|
277
|
+
# Re-score remaining candidates with dynamic utility taking into account chosen layers/duplicates
|
|
278
|
+
scored_pool: list[tuple[float, CandidateContextItem]] = []
|
|
279
|
+
for item in remaining_candidates:
|
|
280
|
+
u = compute_item_utility(item, selected_layers, selected_duplicate_keys)
|
|
281
|
+
scored_pool.append((u, item))
|
|
282
|
+
|
|
283
|
+
# Sort remaining pool deterministically by utility desc, item id asc
|
|
284
|
+
scored_pool.sort(
|
|
285
|
+
key=lambda pair: (
|
|
286
|
+
-pair[0],
|
|
287
|
+
-pair[1].relevance_score,
|
|
288
|
+
pair[1].canonical_id or pair[1].file_path,
|
|
289
|
+
pair[1].start_line,
|
|
290
|
+
)
|
|
291
|
+
)
|
|
292
|
+
|
|
293
|
+
for utility, item in scored_pool:
|
|
294
|
+
if _is_excluded(item, task_spec):
|
|
295
|
+
continue
|
|
296
|
+
# Check redundancy: if item is exact duplicate of something already selected, skip
|
|
297
|
+
if item.duplicate_key in selected_duplicate_keys and utility < 0.2:
|
|
298
|
+
continue
|
|
299
|
+
# Minimum relevance threshold to avoid packing irrelevant items
|
|
300
|
+
if item.relevance_score < 0.35 or utility < 0.25:
|
|
301
|
+
continue
|
|
302
|
+
if selected_tokens + item.estimated_tokens <= token_budget:
|
|
303
|
+
item.why_selected = [
|
|
304
|
+
f"Knapsack utility score: {utility:.2f}",
|
|
305
|
+
f"Relevance: {item.relevance_score:.2f}, Evidence: {item.evidence_quality:.2f}, Freshness: {item.freshness}",
|
|
306
|
+
]
|
|
307
|
+
selected.append(item)
|
|
308
|
+
selected_tokens += item.estimated_tokens
|
|
309
|
+
selected_layers.add(item.coverage_layer)
|
|
310
|
+
selected_duplicate_keys.add(item.duplicate_key)
|
|
311
|
+
|
|
312
|
+
# Final deterministic ordering for presentation:
|
|
313
|
+
# Coverage layers first (ENTRYPOINT -> HANDLER -> SERVICE -> DATA -> TEST -> GIT -> ARCHITECTURE -> GENERAL),
|
|
314
|
+
# then file path and line number
|
|
315
|
+
_LAYER_ORDER = {
|
|
316
|
+
"ENTRYPOINT": 0,
|
|
317
|
+
"HANDLER": 1,
|
|
318
|
+
"SERVICE": 2,
|
|
319
|
+
"DATA": 3,
|
|
320
|
+
"TEST": 4,
|
|
321
|
+
"GIT": 5,
|
|
322
|
+
"ARCHITECTURE": 6,
|
|
323
|
+
"GENERAL": 7,
|
|
324
|
+
}
|
|
325
|
+
|
|
326
|
+
selected.sort(
|
|
327
|
+
key=lambda x: (
|
|
328
|
+
_LAYER_ORDER.get(x.coverage_layer, 9),
|
|
329
|
+
-x.relevance_score,
|
|
330
|
+
x.file_path,
|
|
331
|
+
x.start_line,
|
|
332
|
+
)
|
|
333
|
+
)
|
|
334
|
+
|
|
335
|
+
reduction = (
|
|
336
|
+
1.0 - (selected_tokens / max(candidate_tokens_total, 1))
|
|
337
|
+
if candidate_tokens_total > 0
|
|
338
|
+
else 0.0
|
|
339
|
+
)
|
|
340
|
+
|
|
341
|
+
rejections_list: list[dict[str, object]] = []
|
|
342
|
+
selected_ids = {s.item_id for s in selected}
|
|
343
|
+
for item in candidates:
|
|
344
|
+
if item.item_id not in selected_ids:
|
|
345
|
+
if item.duplicate_key in selected_duplicate_keys:
|
|
346
|
+
reason = f"Duplicate of selected symbol/chunk in {item.file_path}"
|
|
347
|
+
elif item.estimated_tokens + selected_tokens > token_budget:
|
|
348
|
+
reason = f"Token budget limit reached ({token_budget} tokens)"
|
|
349
|
+
elif item.freshness == "STALE":
|
|
350
|
+
reason = "Stale source hash on disk"
|
|
351
|
+
else:
|
|
352
|
+
reason = f"Lower utility/relevance ({item.relevance_score:.2f}) compared to selected candidates"
|
|
353
|
+
rejections_list.append(
|
|
354
|
+
{
|
|
355
|
+
"item_id": item.item_id,
|
|
356
|
+
"file": item.file_path,
|
|
357
|
+
"start_line": item.start_line,
|
|
358
|
+
"symbol": item.canonical_id or item.data.get("symbol"),
|
|
359
|
+
"reason": reason,
|
|
360
|
+
"relevance": item.relevance_score,
|
|
361
|
+
}
|
|
362
|
+
)
|
|
363
|
+
|
|
364
|
+
budget = ContextBudget(
|
|
365
|
+
candidate_tokens=candidate_tokens_total,
|
|
366
|
+
selected_tokens=selected_tokens,
|
|
367
|
+
reduction_ratio=reduction,
|
|
368
|
+
budget_limit=token_budget,
|
|
369
|
+
rejections=tuple(rejections_list[:50]),
|
|
370
|
+
)
|
|
371
|
+
|
|
372
|
+
return selected, budget
|