codegraph-engine 2.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. codegraph/__init__.py +37 -0
  2. codegraph/agent.py +26 -0
  3. codegraph/architecture.py +328 -0
  4. codegraph/audit.py +106 -0
  5. codegraph/cache.py +95 -0
  6. codegraph/cli.py +854 -0
  7. codegraph/config.py +43 -0
  8. codegraph/constraints.py +238 -0
  9. codegraph/context.py +1228 -0
  10. codegraph/epistemic.py +90 -0
  11. codegraph/errors.py +275 -0
  12. codegraph/evidence/__init__.py +15 -0
  13. codegraph/evidence/citations.py +397 -0
  14. codegraph/frameworks.py +434 -0
  15. codegraph/freshness.py +295 -0
  16. codegraph/git.py +278 -0
  17. codegraph/graph/__init__.py +46 -0
  18. codegraph/graph/models.py +41 -0
  19. codegraph/graph/traversal.py +1291 -0
  20. codegraph/indexing/__init__.py +4 -0
  21. codegraph/indexing/classifier.py +274 -0
  22. codegraph/indexing/indexer.py +943 -0
  23. codegraph/indexing/models.py +338 -0
  24. codegraph/indexing/parser.py +1240 -0
  25. codegraph/indexing/scanner.py +200 -0
  26. codegraph/indexing/test_framework.py +116 -0
  27. codegraph/interrogation.py +1582 -0
  28. codegraph/llm/__init__.py +3 -0
  29. codegraph/llm/base.py +15 -0
  30. codegraph/llm/context.py +20 -0
  31. codegraph/mcp/__init__.py +3 -0
  32. codegraph/mcp/server.py +736 -0
  33. codegraph/memory/__init__.py +3 -0
  34. codegraph/memory/store.py +46 -0
  35. codegraph/models.py +289 -0
  36. codegraph/observability.py +151 -0
  37. codegraph/optimizer.py +372 -0
  38. codegraph/planner.py +417 -0
  39. codegraph/py.typed +1 -0
  40. codegraph/query_expansion.py +199 -0
  41. codegraph/ranking.py +363 -0
  42. codegraph/resolver.py +843 -0
  43. codegraph/resources/__init__.py +45 -0
  44. codegraph/resources/cache.py +117 -0
  45. codegraph/resources/coalescer.py +83 -0
  46. codegraph/resources/debouncer.py +98 -0
  47. codegraph/resources/governor.py +232 -0
  48. codegraph/resources/policy.py +123 -0
  49. codegraph/retrieval_policy.py +220 -0
  50. codegraph/search/__init__.py +23 -0
  51. codegraph/search/hybrid.py +301 -0
  52. codegraph/search/semantic.py +28 -0
  53. codegraph/security/__init__.py +3 -0
  54. codegraph/security/paths.py +35 -0
  55. codegraph/target_resolver.py +348 -0
  56. codegraph/task.py +637 -0
  57. codegraph_engine-2.1.1.dist-info/METADATA +334 -0
  58. codegraph_engine-2.1.1.dist-info/RECORD +62 -0
  59. codegraph_engine-2.1.1.dist-info/WHEEL +5 -0
  60. codegraph_engine-2.1.1.dist-info/entry_points.txt +2 -0
  61. codegraph_engine-2.1.1.dist-info/licenses/LICENSE +21 -0
  62. codegraph_engine-2.1.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,338 @@
1
+ """Data models for the indexing, symbol identity, and reference resolution pipeline.
2
+
3
+ Normalization rules for canonical symbol IDs:
4
+ 1. Path separators ('\\' and '/') are normalized to '/'.
5
+ 2. Leading './' and '/' prefixes are stripped.
6
+ 3. Supported source extensions (.py, .pyi, .ts, .tsx, .js, .jsx, .mjs, .cjs) are stripped.
7
+ 4. Package index filenames ('/__init__' for Python, '/index' for JS/TS) are collapsed to
8
+ their parent package path (e.g. 'src/auth/__init__.py' -> 'src.auth',
9
+ 'src/auth/index.ts' -> 'src.auth'). Root-level '__init__.py' or 'index.ts' retain
10
+ '__init__' or 'index'.
11
+ 5. Path segments are joined with '.' to form the normalized module name.
12
+ 6. Canonical ID is constructed deterministically as:
13
+ - f"{module}.{scope}.{name}" when scope is non-empty
14
+ - f"{module}.{name}" when scope is empty
15
+ Line numbers are never part of the canonical identity.
16
+ """
17
+ from __future__ import annotations
18
+
19
+ import posixpath
20
+ from dataclasses import dataclass, field
21
+
22
+ _SOURCE_EXTENSIONS = (
23
+ ".d.ts",
24
+ ".pyi",
25
+ ".tsx",
26
+ ".jsx",
27
+ ".mjs",
28
+ ".cjs",
29
+ ".py",
30
+ ".ts",
31
+ ".js",
32
+ )
33
+
34
+
35
+ def normalize_module(file_path: str, language: str = "") -> str:
36
+ """Deterministically normalize a repository-relative file path into a dotted module ID."""
37
+ del language # normalization is uniform across Python/JS/TS after extension/index rules
38
+ cleaned = file_path.replace("\\", "/").strip()
39
+ while cleaned.startswith("./"):
40
+ cleaned = cleaned[2:]
41
+ cleaned = cleaned.lstrip("/")
42
+ if not cleaned:
43
+ return "root"
44
+
45
+ cleaned = posixpath.normpath(cleaned)
46
+ if cleaned in (".", ""):
47
+ return "root"
48
+
49
+ lower = cleaned.lower()
50
+ for ext in _SOURCE_EXTENSIONS:
51
+ if lower.endswith(ext):
52
+ cleaned = cleaned[: -len(ext)]
53
+ break
54
+
55
+ if cleaned.endswith("/__init__"):
56
+ cleaned = cleaned[: -len("/__init__")]
57
+ elif cleaned.endswith("/index"):
58
+ cleaned = cleaned[: -len("/index")]
59
+
60
+ parts = [p for p in cleaned.split("/") if p and p != "."]
61
+ return ".".join(parts) if parts else "root"
62
+
63
+
64
+ def build_canonical_id(module: str, scope: str, name: str) -> str:
65
+ """Build a deterministic canonical symbol ID from normalized module, scope, and name."""
66
+ mod = module.strip(".")
67
+ scp = scope.strip(".")
68
+ nm = name.strip(".") or "default"
69
+ if mod and scp:
70
+ return f"{mod}.{scp}.{nm}"
71
+ if mod:
72
+ return f"{mod}.{nm}"
73
+ if scp:
74
+ return f"{scp}.{nm}"
75
+ return nm
76
+
77
+
78
+ @dataclass(frozen=True)
79
+ class Symbol:
80
+ name: str
81
+ qualified_name: str
82
+ kind: str # function | method | class | interface | type | variable
83
+ start_line: int
84
+ end_line: int
85
+ file_path: str
86
+ decorators: list[str] = field(default_factory=list)
87
+ id: str = ""
88
+ canonical_id: str = ""
89
+ language: str = "python"
90
+ module: str = ""
91
+ path: str = ""
92
+ scope: str = ""
93
+ signature: str = ""
94
+ content_hash: str = ""
95
+ parent_symbol_id: str | None = None
96
+ visibility: str = ""
97
+ return_type: str | None = None
98
+ parameter_count: int | None = None
99
+ documentation: str | None = None
100
+
101
+ def __post_init__(self) -> None:
102
+ path_val = self.path or self.file_path
103
+ object.__setattr__(self, "path", path_val)
104
+
105
+ mod_val = self.module or normalize_module(path_val, self.language)
106
+ object.__setattr__(self, "module", mod_val)
107
+
108
+ scope_val = self.scope
109
+ if not scope_val and "." in self.qualified_name:
110
+ scope_val = self.qualified_name.rsplit(".", 1)[0]
111
+ object.__setattr__(self, "scope", scope_val)
112
+
113
+ canon_val = self.canonical_id or build_canonical_id(mod_val, scope_val, self.name)
114
+ object.__setattr__(self, "canonical_id", canon_val)
115
+
116
+ id_val = self.id or canon_val
117
+ object.__setattr__(self, "id", id_val)
118
+
119
+ if self.parent_symbol_id is None and scope_val:
120
+ parent_id = f"{mod_val}.{scope_val}" if mod_val else scope_val
121
+ object.__setattr__(self, "parent_symbol_id", parent_id)
122
+
123
+ if not self.visibility:
124
+ is_private = self.name.startswith("_") and not (
125
+ self.name.startswith("__") and self.name.endswith("__")
126
+ )
127
+ object.__setattr__(self, "visibility", "private" if is_private else "public")
128
+
129
+ def as_dict(self) -> dict[str, object]:
130
+ return {
131
+ "id": self.id,
132
+ "canonical_id": self.canonical_id,
133
+ "name": self.name,
134
+ "qualified_name": self.qualified_name,
135
+ "kind": self.kind,
136
+ "language": self.language,
137
+ "module": self.module,
138
+ "path": self.path,
139
+ "file": self.path,
140
+ "file_path": self.file_path,
141
+ "scope": self.scope,
142
+ "signature": self.signature,
143
+ "start_line": self.start_line,
144
+ "end_line": self.end_line,
145
+ "content_hash": self.content_hash,
146
+ "parent_symbol_id": self.parent_symbol_id,
147
+ "visibility": self.visibility,
148
+ "decorators": list(self.decorators),
149
+ "decorator": ",".join(self.decorators),
150
+ "return_type": self.return_type,
151
+ "parameter_count": self.parameter_count,
152
+ "documentation": self.documentation,
153
+ }
154
+
155
+ def __getitem__(self, key: str) -> object:
156
+ return self.as_dict()[key]
157
+
158
+ def get(self, key: str, default: object = None) -> object:
159
+ return self.as_dict().get(key, default)
160
+
161
+ def __contains__(self, key: str) -> bool:
162
+ return key in self.as_dict()
163
+
164
+
165
+ @dataclass(frozen=True)
166
+ class ImportRef:
167
+ module: str
168
+ source_file: str
169
+ name: str | None = None # imported name (from X import name)
170
+ alias: str | None = None # as Y
171
+ full: str | None = None # module.name combined
172
+ line: int = 1
173
+ source_module: str = ""
174
+ imported_module: str = ""
175
+ imported_name: str | None = None
176
+ local_name: str = ""
177
+ import_type: str = "" # named | alias | namespace | default | module | reexport
178
+ is_reexport: bool = False
179
+ exported_name: str | None = None
180
+
181
+ def __post_init__(self) -> None:
182
+ src_mod = self.source_module or normalize_module(self.source_file)
183
+ object.__setattr__(self, "source_module", src_mod)
184
+
185
+ imp_mod = self.imported_module or self.module
186
+ object.__setattr__(self, "imported_module", imp_mod)
187
+
188
+ imp_name = self.imported_name if self.imported_name is not None else self.name
189
+ object.__setattr__(self, "imported_name", imp_name)
190
+
191
+ if not self.local_name:
192
+ if self.alias:
193
+ loc = self.alias
194
+ elif imp_name and imp_name != "*":
195
+ loc = imp_name
196
+ else:
197
+ loc = imp_mod.lstrip(".").split("/")[0].split(".")[0]
198
+ object.__setattr__(self, "local_name", loc)
199
+
200
+ if not self.import_type:
201
+ if self.is_reexport:
202
+ itype = "reexport"
203
+ elif self.alias and imp_name == "*":
204
+ itype = "namespace"
205
+ elif self.alias and imp_name is None:
206
+ itype = "namespace"
207
+ elif self.alias:
208
+ itype = "alias"
209
+ elif imp_name == "default":
210
+ itype = "default"
211
+ elif imp_name:
212
+ itype = "named"
213
+ else:
214
+ itype = "module"
215
+ object.__setattr__(self, "import_type", itype)
216
+
217
+ if self.full is None:
218
+ full_val = f"{imp_mod}.{imp_name}" if (imp_mod and imp_name) else imp_mod
219
+ object.__setattr__(self, "full", full_val)
220
+
221
+ def as_dict(self) -> dict[str, object]:
222
+ return {
223
+ "source_file": self.source_file,
224
+ "source_module": self.source_module,
225
+ "module": self.module,
226
+ "imported_module": self.imported_module,
227
+ "name": self.name,
228
+ "imported_name": self.imported_name,
229
+ "alias": self.alias,
230
+ "local_name": self.local_name,
231
+ "import_type": self.import_type,
232
+ "is_reexport": self.is_reexport,
233
+ "exported_name": self.exported_name or self.alias or self.imported_name,
234
+ "full": self.full,
235
+ "line": self.line,
236
+ }
237
+
238
+ def __getitem__(self, key: str) -> object:
239
+ return self.as_dict()[key]
240
+
241
+ def get(self, key: str, default: object = None) -> object:
242
+ return self.as_dict().get(key, default)
243
+
244
+
245
+ @dataclass(frozen=True)
246
+ class CallRef:
247
+ callee: str
248
+ source_file: str
249
+ confidence: str = "LOW"
250
+ qualified_callee: str | None = None
251
+ line: int = 1
252
+ end_line: int = 0
253
+ caller_symbol: str | None = None # qualified_name within file, or None for module scope
254
+ caller_canonical_id: str | None = None
255
+ receiver: str | None = None
256
+
257
+ def __post_init__(self) -> None:
258
+ if self.end_line <= 0:
259
+ object.__setattr__(self, "end_line", self.line)
260
+ if self.caller_canonical_id is None:
261
+ mod = normalize_module(self.source_file)
262
+ if self.caller_symbol:
263
+ object.__setattr__(self, "caller_canonical_id", f"{mod}.{self.caller_symbol}")
264
+ else:
265
+ object.__setattr__(self, "caller_canonical_id", mod)
266
+
267
+
268
+ @dataclass(frozen=True)
269
+ class InheritanceRef:
270
+ source_symbol: str # qualified_name in source_file
271
+ base_name: str # raw base/interface identifier (e.g. Base or mod.Base)
272
+ relationship: str # EXTENDS | IMPLEMENTS
273
+ source_file: str
274
+ line: int = 1
275
+ source_canonical_id: str = ""
276
+
277
+ def __post_init__(self) -> None:
278
+ if not self.source_canonical_id:
279
+ mod = normalize_module(self.source_file)
280
+ object.__setattr__(self, "source_canonical_id", f"{mod}.{self.source_symbol}")
281
+
282
+
283
+ @dataclass(frozen=True)
284
+ class Reference:
285
+ source_symbol_id: str
286
+ target_symbol_id: str | None
287
+ relationship: str # CALLS | REFERENCES | IMPORTS | EXPORTS | REEXPORTS | EXTENDS | IMPLEMENTS | USES | UNRESOLVED_REFERENCE
288
+ confidence: str # HIGH | MEDIUM | LOW | UNKNOWN
289
+ path: str
290
+ start_line: int
291
+ end_line: int
292
+ evidence: str
293
+ source_hash: str = ""
294
+ indexed_commit: str | None = None
295
+ evidence_status: str = "current"
296
+
297
+ def as_dict(self) -> dict[str, object]:
298
+ return {
299
+ "source_symbol_id": self.source_symbol_id,
300
+ "target_symbol_id": self.target_symbol_id,
301
+ "source": self.source_symbol_id,
302
+ "target": self.target_symbol_id,
303
+ "symbol": self.source_symbol_id,
304
+ "callee": self.target_symbol_id.split(".")[-1] if self.target_symbol_id else None,
305
+ "qualified_callee": self.target_symbol_id,
306
+ "relationship": self.relationship,
307
+ "confidence": self.confidence,
308
+ "path": self.path,
309
+ "file": self.path,
310
+ "start_line": self.start_line,
311
+ "end_line": self.end_line,
312
+ "line": self.start_line,
313
+ "evidence": self.evidence,
314
+ "source_hash": self.source_hash,
315
+ "indexed_commit": self.indexed_commit,
316
+ "evidence_status": self.evidence_status,
317
+ }
318
+
319
+ def __getitem__(self, key: str) -> object:
320
+ return self.as_dict()[key]
321
+
322
+ def get(self, key: str, default: object = None) -> object:
323
+ return self.as_dict().get(key, default)
324
+
325
+ def __contains__(self, key: str) -> bool:
326
+ return key in self.as_dict()
327
+
328
+
329
+ @dataclass(frozen=True)
330
+ class Chunk:
331
+ file_path: str
332
+ language: str
333
+ symbol: str | None
334
+ symbol_type: str | None
335
+ start_line: int
336
+ end_line: int
337
+ content: str
338
+ content_hash: str