codegraph-engine 2.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. codegraph/__init__.py +37 -0
  2. codegraph/agent.py +26 -0
  3. codegraph/architecture.py +328 -0
  4. codegraph/audit.py +106 -0
  5. codegraph/cache.py +95 -0
  6. codegraph/cli.py +854 -0
  7. codegraph/config.py +43 -0
  8. codegraph/constraints.py +238 -0
  9. codegraph/context.py +1228 -0
  10. codegraph/epistemic.py +90 -0
  11. codegraph/errors.py +275 -0
  12. codegraph/evidence/__init__.py +15 -0
  13. codegraph/evidence/citations.py +397 -0
  14. codegraph/frameworks.py +434 -0
  15. codegraph/freshness.py +295 -0
  16. codegraph/git.py +278 -0
  17. codegraph/graph/__init__.py +46 -0
  18. codegraph/graph/models.py +41 -0
  19. codegraph/graph/traversal.py +1291 -0
  20. codegraph/indexing/__init__.py +4 -0
  21. codegraph/indexing/classifier.py +274 -0
  22. codegraph/indexing/indexer.py +943 -0
  23. codegraph/indexing/models.py +338 -0
  24. codegraph/indexing/parser.py +1240 -0
  25. codegraph/indexing/scanner.py +200 -0
  26. codegraph/indexing/test_framework.py +116 -0
  27. codegraph/interrogation.py +1582 -0
  28. codegraph/llm/__init__.py +3 -0
  29. codegraph/llm/base.py +15 -0
  30. codegraph/llm/context.py +20 -0
  31. codegraph/mcp/__init__.py +3 -0
  32. codegraph/mcp/server.py +736 -0
  33. codegraph/memory/__init__.py +3 -0
  34. codegraph/memory/store.py +46 -0
  35. codegraph/models.py +289 -0
  36. codegraph/observability.py +151 -0
  37. codegraph/optimizer.py +372 -0
  38. codegraph/planner.py +417 -0
  39. codegraph/py.typed +1 -0
  40. codegraph/query_expansion.py +199 -0
  41. codegraph/ranking.py +363 -0
  42. codegraph/resolver.py +843 -0
  43. codegraph/resources/__init__.py +45 -0
  44. codegraph/resources/cache.py +117 -0
  45. codegraph/resources/coalescer.py +83 -0
  46. codegraph/resources/debouncer.py +98 -0
  47. codegraph/resources/governor.py +232 -0
  48. codegraph/resources/policy.py +123 -0
  49. codegraph/retrieval_policy.py +220 -0
  50. codegraph/search/__init__.py +23 -0
  51. codegraph/search/hybrid.py +301 -0
  52. codegraph/search/semantic.py +28 -0
  53. codegraph/security/__init__.py +3 -0
  54. codegraph/security/paths.py +35 -0
  55. codegraph/target_resolver.py +348 -0
  56. codegraph/task.py +637 -0
  57. codegraph_engine-2.1.1.dist-info/METADATA +334 -0
  58. codegraph_engine-2.1.1.dist-info/RECORD +62 -0
  59. codegraph_engine-2.1.1.dist-info/WHEEL +5 -0
  60. codegraph_engine-2.1.1.dist-info/entry_points.txt +2 -0
  61. codegraph_engine-2.1.1.dist-info/licenses/LICENSE +21 -0
  62. codegraph_engine-2.1.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,3 @@
1
+ from .store import MemoryStore
2
+
3
+ __all__ = ["MemoryStore"]
@@ -0,0 +1,46 @@
1
+ from __future__ import annotations
2
+
3
+ import builtins
4
+ import sqlite3
5
+ from collections.abc import Iterator
6
+ from contextlib import contextmanager
7
+ from pathlib import Path
8
+
9
+
10
+ class MemoryStore:
11
+ def __init__(self, db_path: Path) -> None:
12
+ self.db_path = db_path
13
+ with self.session() as con:
14
+ con.execute("CREATE TABLE IF NOT EXISTS memory (key TEXT PRIMARY KEY, value TEXT NOT NULL)")
15
+
16
+ def connect(self) -> sqlite3.Connection:
17
+ return sqlite3.connect(self.db_path)
18
+
19
+ @contextmanager
20
+ def session(self) -> Iterator[sqlite3.Connection]:
21
+ con = self.connect()
22
+ try:
23
+ yield con
24
+ con.commit()
25
+ except BaseException:
26
+ con.rollback()
27
+ raise
28
+ finally:
29
+ con.close()
30
+
31
+ def list(self) -> builtins.list[tuple[str, str]]:
32
+ with self.session() as con:
33
+ return list(con.execute("SELECT key, value FROM memory ORDER BY key"))
34
+
35
+ def search(self, query: str, limit: int = 20) -> builtins.list[tuple[str, str]]:
36
+ if not query.strip():
37
+ raise ValueError("memory query must be non-empty")
38
+ with self.session() as con:
39
+ return list(con.execute(
40
+ "SELECT key, value FROM memory WHERE key LIKE ? OR value LIKE ? ORDER BY key LIMIT ?",
41
+ (f"%{query}%", f"%{query}%", limit),
42
+ ))
43
+
44
+ def clear(self) -> None:
45
+ with self.session() as con:
46
+ con.execute("DELETE FROM memory")
codegraph/models.py ADDED
@@ -0,0 +1,289 @@
1
+ """Unified domain models for the CodeGraph intelligence engine."""
2
+ from __future__ import annotations
3
+
4
+ from dataclasses import asdict, dataclass, field
5
+ from typing import Literal
6
+
7
+ from pydantic import BaseModel
8
+
9
+ from codegraph.graph.models import GraphEdge
10
+ from codegraph.indexing.models import (
11
+ CallRef,
12
+ Chunk,
13
+ ImportRef,
14
+ InheritanceRef,
15
+ Reference,
16
+ Symbol,
17
+ build_canonical_id,
18
+ normalize_module,
19
+ )
20
+
21
+ Import = ImportRef
22
+ Call = CallRef
23
+
24
+ NodeType = Literal[
25
+ "FILE",
26
+ "MODULE",
27
+ "CLASS",
28
+ "FUNCTION",
29
+ "METHOD",
30
+ "API_ENDPOINT",
31
+ "API",
32
+ "TEST",
33
+ "TYPE",
34
+ "CONFIG",
35
+ "EXTERNAL_SERVICE",
36
+ ]
37
+
38
+ EdgeType = Literal[
39
+ "DEFINES",
40
+ "CONTAINS",
41
+ "IMPORTS",
42
+ "EXPORTS",
43
+ "REEXPORTS",
44
+ "CALLS",
45
+ "CALLED_BY",
46
+ "POSSIBLE_CALLS",
47
+ "REFERENCES",
48
+ "EXTENDS",
49
+ "IMPLEMENTS",
50
+ "USES",
51
+ "ROUTES_TO",
52
+ "TESTS",
53
+ "DEPENDS_ON",
54
+ "CONFIGURES",
55
+ ]
56
+
57
+ ErrorCategory = Literal[
58
+ "INVALID_INPUT",
59
+ "NOT_FOUND",
60
+ "SECURITY_DENIED",
61
+ "STALE_INDEX",
62
+ "INDEX_STALE",
63
+ "INDEX_UNAVAILABLE",
64
+ "PARSER_ERROR",
65
+ "UNSUPPORTED_LANGUAGE",
66
+ "STORAGE_ERROR",
67
+ "INTERNAL_ERROR",
68
+ ]
69
+
70
+
71
+ @dataclass(frozen=True)
72
+ class Repository:
73
+ path: str
74
+ name: str
75
+ head_commit: str | None = None
76
+ branch: str | None = None
77
+
78
+
79
+ @dataclass(frozen=True)
80
+ class File:
81
+ path: str
82
+ hash: str
83
+ language: str
84
+ indexed_at: int = 0
85
+ status: str = "ok" # ok | parse_failed
86
+ parse_error: str | None = None
87
+
88
+
89
+ @dataclass(frozen=True)
90
+ class SymbolLocation:
91
+ canonical_id: str
92
+ path: str
93
+ start_line: int
94
+ end_line: int
95
+ source_hash: str = ""
96
+
97
+
98
+ @dataclass(frozen=True)
99
+ class GraphNode:
100
+ node_id: str
101
+ node_type: str
102
+ name: str
103
+ file: str | None = None
104
+ start_line: int | None = None
105
+ end_line: int | None = None
106
+ metadata: dict[str, object] = field(default_factory=dict)
107
+
108
+ def as_dict(self) -> dict[str, object]:
109
+ return asdict(self)
110
+
111
+
112
+ @dataclass(frozen=True)
113
+ class RepositorySnapshot:
114
+ repository: str
115
+ indexed_commit: str | None
116
+ current_commit: str | None
117
+ parser_version: str
118
+ schema_version: int
119
+ index_generation: int
120
+ timestamp: int
121
+ file_count: int
122
+ symbol_count: int
123
+ reference_count: int
124
+
125
+
126
+ @dataclass(frozen=True)
127
+ class FreshnessState:
128
+ status: str # FRESH | PARTIALLY_STALE | STALE | UNKNOWN
129
+ indexed_commit: str | None
130
+ current_commit: str | None
131
+ modified_files: list[str]
132
+ deleted_files: list[str]
133
+ added_files: list[str]
134
+ parse_failed_files: list[str]
135
+ detail: str
136
+
137
+
138
+ @dataclass(frozen=True)
139
+ class ContextCandidate:
140
+ file: str
141
+ symbol: str | None
142
+ canonical_id: str | None
143
+ kind: str
144
+ start_line: int
145
+ end_line: int
146
+ snippet: str
147
+ confidence: str = "HIGH"
148
+ freshness: str = "FRESH"
149
+ relationship: str = ""
150
+ distance: int = 0
151
+ score: float = 0.0
152
+
153
+
154
+ class TestRelationship(BaseModel):
155
+ schema_version: str = "1.0"
156
+ file: str
157
+ symbol: str | None = None
158
+ target_symbol: str | None = None
159
+ classification: str = "VERIFIED_TEST" # VERIFIED_TEST | POSSIBLE_TEST
160
+ relationship: str
161
+ confidence: str
162
+ start_line: int = 1
163
+ evidence: str
164
+
165
+
166
+ class ImpactItem(BaseModel):
167
+ symbol: str | None = None
168
+ file: str
169
+ impact_type: str # direct | indirect | potential | test | api | config | doc
170
+ relationship: str = "CALLS"
171
+ confidence: str = "HIGH"
172
+ label: str = "verified" # verified | inferred | possible
173
+ distance: int = 1
174
+ line: int = 1
175
+ evidence: str = ""
176
+
177
+
178
+ class ImpactResult(BaseModel):
179
+ schema_version: str = "2.0"
180
+ subject: str
181
+ canonical_id: str | None = None
182
+ direct_callers: list[dict[str, object]] = []
183
+ transitive_callers: list[dict[str, object]] = []
184
+ directly_affected: list[dict[str, object]] = []
185
+ indirectly_affected: list[dict[str, object]] = []
186
+ potential: list[dict[str, object]] = []
187
+ dependencies: list[dict[str, object]] = []
188
+ dependent_modules: list[dict[str, object]] = []
189
+ related_apis: list[dict[str, object]] = []
190
+ related_tests: list[dict[str, object]] = []
191
+ configuration: list[dict[str, object]] = []
192
+ documentation: list[dict[str, object]] = []
193
+ recent_modifications: list[dict[str, object]] = []
194
+ evidence: list[dict[str, object]] = []
195
+ label_legend: dict[str, str] = {}
196
+ note: str = (
197
+ "Static analysis cannot confirm runtime behavior. "
198
+ "Do not assert failure solely from static dependency data."
199
+ )
200
+
201
+ def as_dict(self) -> dict[str, object]:
202
+ return self.model_dump()
203
+
204
+
205
+ class ArchitectureNode(BaseModel):
206
+ id: str
207
+ name: str
208
+ layer: str # entry_point | route | controller | service | repository | model | config | test | external
209
+ file: str
210
+ line: int = 1
211
+ confidence: str = "HIGH"
212
+
213
+
214
+ class ArchitectureEdge(BaseModel):
215
+ source: str
216
+ target: str
217
+ relationship: str
218
+ confidence: str = "HIGH"
219
+ evidence: str = ""
220
+
221
+
222
+ class MCPError(BaseModel):
223
+ schema_version: str = "1.0"
224
+ error_code: ErrorCategory
225
+ message: str
226
+ details: dict[str, object] = {}
227
+
228
+ def as_dict(self) -> dict[str, object]:
229
+ return self.model_dump()
230
+
231
+
232
+ @dataclass
233
+ class ObservabilityMetrics:
234
+ index_duration_ms: float = 0.0
235
+ files_scanned: int = 0
236
+ files_changed: int = 0
237
+ files_unchanged: int = 0
238
+ files_removed: int = 0
239
+ files_parse_failed: int = 0
240
+ symbols_count: int = 0
241
+ references_count: int = 0
242
+ graph_edges_count: int = 0
243
+ search_latency_ms: float = 0.0
244
+ context_compile_latency_ms: float = 0.0
245
+ database_size_bytes: int = 0
246
+ cache_hits: int = 0
247
+ cache_misses: int = 0
248
+
249
+ @property
250
+ def cache_hit_rate(self) -> float:
251
+ total = self.cache_hits + self.cache_misses
252
+ return round(self.cache_hits / total, 3) if total > 0 else 0.0
253
+
254
+ def as_dict(self) -> dict[str, object]:
255
+ d = asdict(self)
256
+ d["cache_hit_rate"] = self.cache_hit_rate
257
+ return d
258
+
259
+
260
+ __all__ = [
261
+ "ArchitectureEdge",
262
+ "ArchitectureNode",
263
+ "Call",
264
+ "CallRef",
265
+ "Chunk",
266
+ "ContextCandidate",
267
+ "EdgeType",
268
+ "ErrorCategory",
269
+ "File",
270
+ "FreshnessState",
271
+ "GraphEdge",
272
+ "GraphNode",
273
+ "ImpactItem",
274
+ "ImpactResult",
275
+ "Import",
276
+ "ImportRef",
277
+ "InheritanceRef",
278
+ "MCPError",
279
+ "NodeType",
280
+ "ObservabilityMetrics",
281
+ "Reference",
282
+ "Repository",
283
+ "RepositorySnapshot",
284
+ "Symbol",
285
+ "SymbolLocation",
286
+ "TestRelationship",
287
+ "build_canonical_id",
288
+ "normalize_module",
289
+ ]
@@ -0,0 +1,151 @@
1
+ """Production Observability and Structured Metrics.
2
+
3
+ Invariants:
4
+ - Never log raw repository source code.
5
+ - Never log secrets or credentials.
6
+ - Zero network dependencies or cloud uploads.
7
+ - Structured metrics buffer available locally for audits, benchmarks, and doctor diagnostics.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import time
12
+ import uuid
13
+ from dataclasses import asdict, dataclass, field
14
+ from typing import Any
15
+
16
+
17
+ @dataclass
18
+ class ExecutionTiming:
19
+ planning_ms: float = 0.0
20
+ search_ms: float = 0.0
21
+ graph_ms: float = 0.0
22
+ framework_ms: float = 0.0
23
+ tests_ms: float = 0.0
24
+ git_ms: float = 0.0
25
+ evidence_ms: float = 0.0
26
+ ranking_ms: float = 0.0
27
+ compilation_ms: float = 0.0
28
+ serialization_ms: float = 0.0
29
+ total_ms: float = 0.0
30
+
31
+ def as_dict(self) -> dict[str, float]:
32
+ return {
33
+ "planning_ms": round(self.planning_ms, 2),
34
+ "search_ms": round(self.search_ms, 2),
35
+ "graph_ms": round(self.graph_ms, 2),
36
+ "framework_ms": round(self.framework_ms, 2),
37
+ "tests_ms": round(self.tests_ms, 2),
38
+ "git_ms": round(self.git_ms, 2),
39
+ "evidence_ms": round(self.evidence_ms, 2),
40
+ "ranking_ms": round(self.ranking_ms, 2),
41
+ "compilation_ms": round(self.compilation_ms, 2),
42
+ "serialization_ms": round(self.serialization_ms, 2),
43
+ "total_ms": round(self.total_ms, 2),
44
+ }
45
+
46
+
47
+ @dataclass
48
+ class ExecutionMetadata:
49
+ latency_ms: float = 0.0
50
+ cache_hit: bool = False
51
+ index_generation: int = 0
52
+ candidate_count: int = 0
53
+ selected_count: int = 0
54
+ mode: str = "BALANCED" # FAST | BALANCED | DEEP
55
+ timing: ExecutionTiming | None = None
56
+
57
+ def as_dict(self) -> dict[str, object]:
58
+ res: dict[str, object] = {
59
+ "latency_ms": round(self.latency_ms, 2),
60
+ "cache_hit": self.cache_hit,
61
+ "index_generation": self.index_generation,
62
+ "candidate_count": self.candidate_count,
63
+ "selected_count": self.selected_count,
64
+ "mode": self.mode,
65
+ }
66
+ if self.timing:
67
+ res["timing"] = self.timing.as_dict()
68
+ return res
69
+
70
+
71
+ @dataclass
72
+ class OperationMetric:
73
+ request_id: str
74
+ tool_or_op: str
75
+ latency_ms: float
76
+ index_generation: int = 0
77
+ cache_hit: bool = False
78
+ candidate_count: int = 0
79
+ selected_count: int = 0
80
+ estimated_tokens: int = 0
81
+ task_intent: str = "UNDERSTAND"
82
+ ambiguity_state: str = "CLEAR"
83
+ freshness: str = "FRESH"
84
+ error_class: str | None = None
85
+ timestamp: float = field(default_factory=time.time)
86
+
87
+ def as_dict(self) -> dict[str, object]:
88
+ return asdict(self)
89
+
90
+
91
+ class MetricsRegistry:
92
+ """Thread-safe, bounded in-memory metrics tracker."""
93
+
94
+ def __init__(self, max_records: int = 500) -> None:
95
+ self.max_records = max_records
96
+ self._records: list[OperationMetric] = []
97
+
98
+ def record(self, metric: OperationMetric) -> None:
99
+ self._records.append(metric)
100
+ if len(self._records) > self.max_records:
101
+ self._records = self._records[-self.max_records :]
102
+
103
+ def get_summary(self) -> dict[str, object]:
104
+ total = len(self._records)
105
+ if not total:
106
+ return {
107
+ "total_operations": 0,
108
+ "avg_latency_ms": 0.0,
109
+ "cache_hit_rate": 0.0,
110
+ "errors": 0,
111
+ }
112
+ avg_lat = sum(r.latency_ms for r in self._records) / total
113
+ cache_hits = sum(1 for r in self._records if r.cache_hit)
114
+ errors = sum(1 for r in self._records if r.error_class is not None)
115
+ return {
116
+ "total_operations": total,
117
+ "avg_latency_ms": round(avg_lat, 2),
118
+ "cache_hit_rate": round(cache_hits / total, 3),
119
+ "errors": errors,
120
+ "recent_operations": [r.as_dict() for r in self._records[-10:]],
121
+ }
122
+
123
+ def clear(self) -> None:
124
+ self._records.clear()
125
+
126
+
127
+ # Global in-memory metrics registry for current process
128
+ _GLOBAL_METRICS = MetricsRegistry()
129
+
130
+
131
+ def get_global_metrics() -> MetricsRegistry:
132
+ return _GLOBAL_METRICS
133
+
134
+
135
+ class Timer:
136
+ """Context manager for measuring latency with high precision."""
137
+
138
+ def __init__(self) -> None:
139
+ self.start_time = 0.0
140
+ self.elapsed_ms = 0.0
141
+
142
+ def __enter__(self) -> Timer:
143
+ self.start_time = time.perf_counter()
144
+ return self
145
+
146
+ def __exit__(self, exc_type: Any, exc_val: Any, exc_tb: Any) -> None:
147
+ self.elapsed_ms = (time.perf_counter() - self.start_time) * 1000.0
148
+
149
+
150
+ def create_request_id() -> str:
151
+ return uuid.uuid4().hex[:12]