codegraph-engine 2.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codegraph/__init__.py +37 -0
- codegraph/agent.py +26 -0
- codegraph/architecture.py +328 -0
- codegraph/audit.py +106 -0
- codegraph/cache.py +95 -0
- codegraph/cli.py +854 -0
- codegraph/config.py +43 -0
- codegraph/constraints.py +238 -0
- codegraph/context.py +1228 -0
- codegraph/epistemic.py +90 -0
- codegraph/errors.py +275 -0
- codegraph/evidence/__init__.py +15 -0
- codegraph/evidence/citations.py +397 -0
- codegraph/frameworks.py +434 -0
- codegraph/freshness.py +295 -0
- codegraph/git.py +278 -0
- codegraph/graph/__init__.py +46 -0
- codegraph/graph/models.py +41 -0
- codegraph/graph/traversal.py +1291 -0
- codegraph/indexing/__init__.py +4 -0
- codegraph/indexing/classifier.py +274 -0
- codegraph/indexing/indexer.py +943 -0
- codegraph/indexing/models.py +338 -0
- codegraph/indexing/parser.py +1240 -0
- codegraph/indexing/scanner.py +200 -0
- codegraph/indexing/test_framework.py +116 -0
- codegraph/interrogation.py +1582 -0
- codegraph/llm/__init__.py +3 -0
- codegraph/llm/base.py +15 -0
- codegraph/llm/context.py +20 -0
- codegraph/mcp/__init__.py +3 -0
- codegraph/mcp/server.py +736 -0
- codegraph/memory/__init__.py +3 -0
- codegraph/memory/store.py +46 -0
- codegraph/models.py +289 -0
- codegraph/observability.py +151 -0
- codegraph/optimizer.py +372 -0
- codegraph/planner.py +417 -0
- codegraph/py.typed +1 -0
- codegraph/query_expansion.py +199 -0
- codegraph/ranking.py +363 -0
- codegraph/resolver.py +843 -0
- codegraph/resources/__init__.py +45 -0
- codegraph/resources/cache.py +117 -0
- codegraph/resources/coalescer.py +83 -0
- codegraph/resources/debouncer.py +98 -0
- codegraph/resources/governor.py +232 -0
- codegraph/resources/policy.py +123 -0
- codegraph/retrieval_policy.py +220 -0
- codegraph/search/__init__.py +23 -0
- codegraph/search/hybrid.py +301 -0
- codegraph/search/semantic.py +28 -0
- codegraph/security/__init__.py +3 -0
- codegraph/security/paths.py +35 -0
- codegraph/target_resolver.py +348 -0
- codegraph/task.py +637 -0
- codegraph_engine-2.1.1.dist-info/METADATA +334 -0
- codegraph_engine-2.1.1.dist-info/RECORD +62 -0
- codegraph_engine-2.1.1.dist-info/WHEEL +5 -0
- codegraph_engine-2.1.1.dist-info/entry_points.txt +2 -0
- codegraph_engine-2.1.1.dist-info/licenses/LICENSE +21 -0
- codegraph_engine-2.1.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1240 @@
|
|
|
1
|
+
"""Single-pass scope-aware source parsers for Python, JavaScript, and TypeScript.
|
|
2
|
+
|
|
3
|
+
Architectural Invariants:
|
|
4
|
+
1. Exactly ONE scoped traversal per source file.
|
|
5
|
+
2. Framework route analyzers consume AST node visit events during this primary traversal;
|
|
6
|
+
there is NEVER a second full-tree scan.
|
|
7
|
+
3. Lexical scope context (classes, functions, closures, methods) is maintained continuously
|
|
8
|
+
on an explicit traversal stack without line-number or string-based heuristics.
|
|
9
|
+
4. ParseResult is an immutable contract between parsing and indexing.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import ast
|
|
14
|
+
import hashlib
|
|
15
|
+
import posixpath
|
|
16
|
+
import re
|
|
17
|
+
from dataclasses import asdict, dataclass
|
|
18
|
+
|
|
19
|
+
from codegraph.frameworks import (
|
|
20
|
+
AnalysisContext,
|
|
21
|
+
FrameworkAnalyzer,
|
|
22
|
+
RouteDetection,
|
|
23
|
+
analyze_js_ts_line,
|
|
24
|
+
get_python_analyzers,
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
from .models import (
|
|
28
|
+
CallRef,
|
|
29
|
+
ImportRef,
|
|
30
|
+
InheritanceRef,
|
|
31
|
+
Symbol,
|
|
32
|
+
build_canonical_id,
|
|
33
|
+
normalize_module,
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
PARSER_VERSION = "3.0"
|
|
37
|
+
|
|
38
|
+
_PYTHON_BUILTINS = {
|
|
39
|
+
"abs",
|
|
40
|
+
"all",
|
|
41
|
+
"any",
|
|
42
|
+
"ascii",
|
|
43
|
+
"bin",
|
|
44
|
+
"bool",
|
|
45
|
+
"breakpoint",
|
|
46
|
+
"bytearray",
|
|
47
|
+
"bytes",
|
|
48
|
+
"callable",
|
|
49
|
+
"chr",
|
|
50
|
+
"classmethod",
|
|
51
|
+
"compile",
|
|
52
|
+
"complex",
|
|
53
|
+
"delattr",
|
|
54
|
+
"dict",
|
|
55
|
+
"dir",
|
|
56
|
+
"divmod",
|
|
57
|
+
"enumerate",
|
|
58
|
+
"eval",
|
|
59
|
+
"exec",
|
|
60
|
+
"filter",
|
|
61
|
+
"float",
|
|
62
|
+
"format",
|
|
63
|
+
"frozenset",
|
|
64
|
+
"getattr",
|
|
65
|
+
"globals",
|
|
66
|
+
"hasattr",
|
|
67
|
+
"hash",
|
|
68
|
+
"help",
|
|
69
|
+
"hex",
|
|
70
|
+
"id",
|
|
71
|
+
"input",
|
|
72
|
+
"int",
|
|
73
|
+
"isinstance",
|
|
74
|
+
"issubclass",
|
|
75
|
+
"iter",
|
|
76
|
+
"len",
|
|
77
|
+
"list",
|
|
78
|
+
"locals",
|
|
79
|
+
"map",
|
|
80
|
+
"max",
|
|
81
|
+
"memoryview",
|
|
82
|
+
"min",
|
|
83
|
+
"next",
|
|
84
|
+
"object",
|
|
85
|
+
"oct",
|
|
86
|
+
"open",
|
|
87
|
+
"ord",
|
|
88
|
+
"pow",
|
|
89
|
+
"print",
|
|
90
|
+
"property",
|
|
91
|
+
"range",
|
|
92
|
+
"repr",
|
|
93
|
+
"reversed",
|
|
94
|
+
"round",
|
|
95
|
+
"set",
|
|
96
|
+
"setattr",
|
|
97
|
+
"slice",
|
|
98
|
+
"sorted",
|
|
99
|
+
"staticmethod",
|
|
100
|
+
"str",
|
|
101
|
+
"sum",
|
|
102
|
+
"super",
|
|
103
|
+
"tuple",
|
|
104
|
+
"type",
|
|
105
|
+
"vars",
|
|
106
|
+
"zip",
|
|
107
|
+
"Exception",
|
|
108
|
+
"ValueError",
|
|
109
|
+
"TypeError",
|
|
110
|
+
"KeyError",
|
|
111
|
+
"IndexError",
|
|
112
|
+
"RuntimeError",
|
|
113
|
+
"AttributeError",
|
|
114
|
+
"NotImplementedError",
|
|
115
|
+
"OSError",
|
|
116
|
+
"FileNotFoundError",
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
_JS_KEYWORDS_AND_BUILTINS = {
|
|
120
|
+
"if",
|
|
121
|
+
"for",
|
|
122
|
+
"while",
|
|
123
|
+
"switch",
|
|
124
|
+
"catch",
|
|
125
|
+
"function",
|
|
126
|
+
"return",
|
|
127
|
+
"typeof",
|
|
128
|
+
"instanceof",
|
|
129
|
+
"new",
|
|
130
|
+
"delete",
|
|
131
|
+
"void",
|
|
132
|
+
"yield",
|
|
133
|
+
"await",
|
|
134
|
+
"super",
|
|
135
|
+
"this",
|
|
136
|
+
"import",
|
|
137
|
+
"require",
|
|
138
|
+
"console",
|
|
139
|
+
"Math",
|
|
140
|
+
"JSON",
|
|
141
|
+
"Object",
|
|
142
|
+
"Array",
|
|
143
|
+
"String",
|
|
144
|
+
"Number",
|
|
145
|
+
"Boolean",
|
|
146
|
+
"Promise",
|
|
147
|
+
"Set",
|
|
148
|
+
"Map",
|
|
149
|
+
"Error",
|
|
150
|
+
"TypeError",
|
|
151
|
+
"Date",
|
|
152
|
+
"RegExp",
|
|
153
|
+
"parseInt",
|
|
154
|
+
"parseFloat",
|
|
155
|
+
"setTimeout",
|
|
156
|
+
"clearTimeout",
|
|
157
|
+
"setInterval",
|
|
158
|
+
"clearInterval",
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
_PY_IMPORT = re.compile(
|
|
162
|
+
r"""^(?:from\s+([\w.]+)\s+import|import\s+([\w.,\s]+))""",
|
|
163
|
+
re.MULTILINE,
|
|
164
|
+
)
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
@dataclass(frozen=True)
|
|
168
|
+
class ParseResult:
|
|
169
|
+
path: str
|
|
170
|
+
language: str
|
|
171
|
+
parser_version: str = PARSER_VERSION
|
|
172
|
+
source_hash: str = ""
|
|
173
|
+
symbols: tuple[Symbol, ...] = ()
|
|
174
|
+
imports: tuple[ImportRef, ...] = ()
|
|
175
|
+
calls: tuple[CallRef, ...] = ()
|
|
176
|
+
inheritance: tuple[InheritanceRef, ...] = ()
|
|
177
|
+
routes: tuple[RouteDetection, ...] = ()
|
|
178
|
+
exports: tuple[str, ...] = ()
|
|
179
|
+
parse_failed: bool = False
|
|
180
|
+
parse_error: str | None = None
|
|
181
|
+
|
|
182
|
+
def as_dict(self) -> dict[str, object]:
|
|
183
|
+
return {
|
|
184
|
+
"path": self.path,
|
|
185
|
+
"language": self.language,
|
|
186
|
+
"parser_version": self.parser_version,
|
|
187
|
+
"source_hash": self.source_hash,
|
|
188
|
+
"symbols": [s.as_dict() for s in self.symbols],
|
|
189
|
+
"imports": [i.as_dict() for i in self.imports],
|
|
190
|
+
"calls": [c.qualified_callee or c.callee for c in self.calls],
|
|
191
|
+
"inheritance": [asdict(inh) for inh in self.inheritance],
|
|
192
|
+
"routes": [r.as_dict() for r in self.routes],
|
|
193
|
+
"exports": list(self.exports),
|
|
194
|
+
"parse_failed": self.parse_failed,
|
|
195
|
+
"parse_error": self.parse_error,
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
def __getitem__(self, key: str) -> object:
|
|
199
|
+
return getattr(self, key)
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def parse(content: str, language: str, file_path: str) -> ParseResult:
|
|
203
|
+
"""Parse a source file in a single scoped pass."""
|
|
204
|
+
if language == "python":
|
|
205
|
+
return _parse_python(content, file_path)
|
|
206
|
+
return _parse_js_ts(content, language, file_path)
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def parse_symbols(
|
|
210
|
+
content: str, language: str, file_path: str
|
|
211
|
+
) -> tuple[list[Symbol], list[str]]:
|
|
212
|
+
"""Backward-compatible helper returning symbols and unique imported module names."""
|
|
213
|
+
result = parse(content, language, file_path)
|
|
214
|
+
seen: list[str] = []
|
|
215
|
+
for imp in result.imports:
|
|
216
|
+
if imp.module and imp.module not in seen:
|
|
217
|
+
seen.append(imp.module)
|
|
218
|
+
return list(result.symbols), seen
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def _normalized_body_hash(lines: list[str], start_line: int, end_line: int) -> str:
|
|
222
|
+
"""Compute a content hash of a symbol body independent of absolute line numbers."""
|
|
223
|
+
snippet = "\n".join(line.rstrip() for line in lines[max(0, start_line - 1) : end_line]).strip()
|
|
224
|
+
return hashlib.sha256(snippet.encode("utf-8", errors="replace")).hexdigest()[:16]
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def _resolve_relative_py_module(file_path: str, level: int, module: str | None) -> str:
|
|
228
|
+
"""Resolve a Python relative import using source file directory depth."""
|
|
229
|
+
if level <= 0:
|
|
230
|
+
return module or ""
|
|
231
|
+
norm_dir = posixpath.dirname(file_path.replace("\\", "/").lstrip("./"))
|
|
232
|
+
parts = [p for p in norm_dir.split("/") if p and p != "."]
|
|
233
|
+
up = level - 1
|
|
234
|
+
if up > 0 and up <= len(parts):
|
|
235
|
+
parts = parts[: len(parts) - up]
|
|
236
|
+
elif up > len(parts):
|
|
237
|
+
parts = []
|
|
238
|
+
base = ".".join(parts)
|
|
239
|
+
if base and module:
|
|
240
|
+
return f"{base}.{module}"
|
|
241
|
+
if base:
|
|
242
|
+
return base
|
|
243
|
+
if module:
|
|
244
|
+
return module
|
|
245
|
+
return "." * level
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def _infer_function_return(node: ast.FunctionDef | ast.AsyncFunctionDef) -> str | None:
|
|
249
|
+
"""Conservatively infer return type from explicit annotation or direct constructor return."""
|
|
250
|
+
# 1. Explicit return annotation
|
|
251
|
+
if node.returns and hasattr(ast, "unparse"):
|
|
252
|
+
raw = ast.unparse(node.returns).strip()
|
|
253
|
+
m = re.match(r"^(?:Optional|typing\.Optional)\[\s*([A-Za-z_][\w.]*)\s*\]$", raw)
|
|
254
|
+
if m:
|
|
255
|
+
raw = m.group(1)
|
|
256
|
+
clean = raw.split("[")[0].strip()
|
|
257
|
+
if clean and clean not in _PYTHON_BUILTINS:
|
|
258
|
+
return clean
|
|
259
|
+
|
|
260
|
+
# 2. Direct constructor return in function body
|
|
261
|
+
ctor_candidates: set[str] = set()
|
|
262
|
+
for child in node.body:
|
|
263
|
+
if isinstance(child, ast.Return) and child.value is not None:
|
|
264
|
+
val = child.value
|
|
265
|
+
if isinstance(val, ast.Call):
|
|
266
|
+
if isinstance(val.func, ast.Name):
|
|
267
|
+
c_name = val.func.id
|
|
268
|
+
if c_name not in _PYTHON_BUILTINS and (
|
|
269
|
+
c_name[0].isupper()
|
|
270
|
+
or c_name.endswith("Service")
|
|
271
|
+
or c_name.endswith("Repository")
|
|
272
|
+
or c_name.endswith("Client")
|
|
273
|
+
or c_name.endswith("Store")
|
|
274
|
+
):
|
|
275
|
+
ctor_candidates.add(c_name)
|
|
276
|
+
elif isinstance(val.func, ast.Attribute) and hasattr(ast, "unparse"):
|
|
277
|
+
ctor_candidates.add(ast.unparse(val.func))
|
|
278
|
+
elif isinstance(child, ast.If):
|
|
279
|
+
for sub in child.body + child.orelse:
|
|
280
|
+
if isinstance(sub, ast.Return) and sub.value is not None and isinstance(sub.value, ast.Call):
|
|
281
|
+
val = sub.value
|
|
282
|
+
if isinstance(val.func, ast.Name):
|
|
283
|
+
c_name = val.func.id
|
|
284
|
+
if c_name not in _PYTHON_BUILTINS and (
|
|
285
|
+
c_name[0].isupper()
|
|
286
|
+
or c_name.endswith("Service")
|
|
287
|
+
or c_name.endswith("Repository")
|
|
288
|
+
or c_name.endswith("Client")
|
|
289
|
+
or c_name.endswith("Store")
|
|
290
|
+
):
|
|
291
|
+
ctor_candidates.add(c_name)
|
|
292
|
+
elif isinstance(val.func, ast.Attribute) and hasattr(ast, "unparse"):
|
|
293
|
+
ctor_candidates.add(ast.unparse(val.func))
|
|
294
|
+
|
|
295
|
+
if len(ctor_candidates) == 1:
|
|
296
|
+
return next(iter(ctor_candidates))
|
|
297
|
+
return None
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def _extract_python_file_declarations(tree: ast.AST) -> tuple[dict[str, str], set[str]]:
|
|
301
|
+
"""Extract known return types and class names in a single Python AST."""
|
|
302
|
+
fn_returns: dict[str, str] = {}
|
|
303
|
+
known_classes: set[str] = set()
|
|
304
|
+
for node in getattr(tree, "body", []):
|
|
305
|
+
if isinstance(node, ast.ClassDef):
|
|
306
|
+
known_classes.add(node.name)
|
|
307
|
+
for item in node.body:
|
|
308
|
+
if isinstance(item, (ast.FunctionDef, ast.AsyncFunctionDef)):
|
|
309
|
+
ret = _infer_function_return(item)
|
|
310
|
+
if ret:
|
|
311
|
+
fn_returns[f"{node.name}.{item.name}"] = ret
|
|
312
|
+
elif isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)):
|
|
313
|
+
ret = _infer_function_return(node)
|
|
314
|
+
if ret:
|
|
315
|
+
fn_returns[node.name] = ret
|
|
316
|
+
return fn_returns, known_classes
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
class _PythonScopeVisitor(ast.NodeVisitor):
|
|
320
|
+
"""Single-pass AST visitor maintaining scope context, bindings, and framework hooks."""
|
|
321
|
+
|
|
322
|
+
def __init__(
|
|
323
|
+
self,
|
|
324
|
+
content: str,
|
|
325
|
+
lines: list[str],
|
|
326
|
+
file_path: str,
|
|
327
|
+
analyzers: list[FrameworkAnalyzer],
|
|
328
|
+
fn_return_types: dict[str, str] | None = None,
|
|
329
|
+
known_classes: set[str] | None = None,
|
|
330
|
+
) -> None:
|
|
331
|
+
self.content = content
|
|
332
|
+
self.lines = lines
|
|
333
|
+
self.file_path = file_path
|
|
334
|
+
self.module = normalize_module(file_path, "python")
|
|
335
|
+
self.is_init = file_path.replace("\\", "/").endswith("__init__.py")
|
|
336
|
+
self.analyzers = analyzers
|
|
337
|
+
self.fn_return_types = fn_return_types or {}
|
|
338
|
+
self.known_classes = known_classes or set()
|
|
339
|
+
|
|
340
|
+
self.symbols: list[Symbol] = []
|
|
341
|
+
self.imports: list[ImportRef] = []
|
|
342
|
+
self.calls: list[CallRef] = []
|
|
343
|
+
self.inheritance: list[InheritanceRef] = []
|
|
344
|
+
self.routes: list[RouteDetection] = []
|
|
345
|
+
self.exports: list[str] = []
|
|
346
|
+
|
|
347
|
+
# Scope stack: (name, kind, qualified_name, canonical_id)
|
|
348
|
+
self.scope_stack: list[tuple[str, str, str, str]] = []
|
|
349
|
+
# Local variable -> class/alias bindings per scope
|
|
350
|
+
self.local_bindings_stack: list[dict[str, str]] = [{}]
|
|
351
|
+
self.import_modules_set: set[str] = set()
|
|
352
|
+
|
|
353
|
+
@property
|
|
354
|
+
def current_scope_qname(self) -> str:
|
|
355
|
+
return self.scope_stack[-1][2] if self.scope_stack else ""
|
|
356
|
+
|
|
357
|
+
@property
|
|
358
|
+
def current_scope_canonical_id(self) -> str:
|
|
359
|
+
return self.scope_stack[-1][3] if self.scope_stack else self.module
|
|
360
|
+
|
|
361
|
+
@property
|
|
362
|
+
def current_scope_kind(self) -> str:
|
|
363
|
+
return self.scope_stack[-1][1] if self.scope_stack else "module"
|
|
364
|
+
|
|
365
|
+
@property
|
|
366
|
+
def current_enclosing_class(self) -> str | None:
|
|
367
|
+
for _name, kind, qname, _canon in reversed(self.scope_stack):
|
|
368
|
+
if kind == "class":
|
|
369
|
+
return qname
|
|
370
|
+
return None
|
|
371
|
+
|
|
372
|
+
def _make_context(self) -> AnalysisContext:
|
|
373
|
+
return AnalysisContext(
|
|
374
|
+
file_path=self.file_path,
|
|
375
|
+
module=self.module,
|
|
376
|
+
imports_modules=self.import_modules_set,
|
|
377
|
+
scope_qname=self.current_scope_qname,
|
|
378
|
+
scope_canonical_id=self.current_scope_canonical_id,
|
|
379
|
+
scope_kind=self.current_scope_kind,
|
|
380
|
+
)
|
|
381
|
+
|
|
382
|
+
def _lookup_local_binding(self, var_name: str) -> str | None:
|
|
383
|
+
for env in reversed(self.local_bindings_stack):
|
|
384
|
+
if var_name in env:
|
|
385
|
+
return env[var_name]
|
|
386
|
+
return None
|
|
387
|
+
|
|
388
|
+
def _extract_decorators(
|
|
389
|
+
self, node: ast.FunctionDef | ast.AsyncFunctionDef | ast.ClassDef
|
|
390
|
+
) -> list[str]:
|
|
391
|
+
dec_names: list[str] = []
|
|
392
|
+
for d in node.decorator_list:
|
|
393
|
+
if isinstance(d, ast.Name):
|
|
394
|
+
dec_names.append(d.id)
|
|
395
|
+
elif isinstance(d, ast.Attribute):
|
|
396
|
+
dec_names.append(ast.unparse(d) if hasattr(ast, "unparse") else d.attr)
|
|
397
|
+
elif isinstance(d, ast.Call):
|
|
398
|
+
if isinstance(d.func, ast.Name):
|
|
399
|
+
dec_names.append(d.func.id)
|
|
400
|
+
elif isinstance(d.func, ast.Attribute):
|
|
401
|
+
dec_names.append(ast.unparse(d.func) if hasattr(ast, "unparse") else d.func.attr)
|
|
402
|
+
return dec_names
|
|
403
|
+
|
|
404
|
+
def _visit_symbol_node(
|
|
405
|
+
self, node: ast.FunctionDef | ast.AsyncFunctionDef | ast.ClassDef
|
|
406
|
+
) -> None:
|
|
407
|
+
parent_scope = self.current_scope_qname
|
|
408
|
+
parent_canon = self.scope_stack[-1][3] if self.scope_stack else None
|
|
409
|
+
parent_kind = self.scope_stack[-1][1] if self.scope_stack else None
|
|
410
|
+
|
|
411
|
+
if isinstance(node, ast.ClassDef):
|
|
412
|
+
kind = "class"
|
|
413
|
+
elif parent_kind == "class":
|
|
414
|
+
decorators = self._extract_decorators(node)
|
|
415
|
+
kind = "property" if "property" in decorators else "method"
|
|
416
|
+
else:
|
|
417
|
+
kind = "function"
|
|
418
|
+
|
|
419
|
+
qname = f"{parent_scope}.{node.name}" if parent_scope else node.name
|
|
420
|
+
canon_id = build_canonical_id(self.module, parent_scope, node.name)
|
|
421
|
+
decorators = self._extract_decorators(node)
|
|
422
|
+
end_line = node.end_lineno or node.lineno
|
|
423
|
+
|
|
424
|
+
sig = ""
|
|
425
|
+
ret_type: str | None = None
|
|
426
|
+
param_count: int | None = None
|
|
427
|
+
if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)):
|
|
428
|
+
args = [a.arg for a in node.args.args]
|
|
429
|
+
param_count = len(args)
|
|
430
|
+
if node.returns and hasattr(ast, "unparse"):
|
|
431
|
+
ret_type = ast.unparse(node.returns)
|
|
432
|
+
elif node.name in self.fn_return_types:
|
|
433
|
+
ret_type = self.fn_return_types[node.name]
|
|
434
|
+
elif qname in self.fn_return_types:
|
|
435
|
+
ret_type = self.fn_return_types[qname]
|
|
436
|
+
sig = f"({', '.join(args)})" + (f" -> {ret_type}" if ret_type else "")
|
|
437
|
+
elif isinstance(node, ast.ClassDef):
|
|
438
|
+
bases_str = [ast.unparse(b) for b in node.bases] if hasattr(ast, "unparse") else []
|
|
439
|
+
sig = f"({', '.join(bases_str)})" if bases_str else ""
|
|
440
|
+
for base in node.bases:
|
|
441
|
+
base_name = ast.unparse(base) if hasattr(ast, "unparse") else getattr(base, "id", "")
|
|
442
|
+
base_clean = base_name.split("[")[0].strip()
|
|
443
|
+
if base_clean:
|
|
444
|
+
self.inheritance.append(
|
|
445
|
+
InheritanceRef(
|
|
446
|
+
source_symbol=qname,
|
|
447
|
+
base_name=base_clean,
|
|
448
|
+
relationship="EXTENDS",
|
|
449
|
+
source_file=self.file_path,
|
|
450
|
+
line=node.lineno,
|
|
451
|
+
source_canonical_id=canon_id,
|
|
452
|
+
)
|
|
453
|
+
)
|
|
454
|
+
|
|
455
|
+
doc = ast.get_docstring(node)
|
|
456
|
+
c_hash = _normalized_body_hash(self.lines, node.lineno, end_line)
|
|
457
|
+
|
|
458
|
+
self.symbols.append(
|
|
459
|
+
Symbol(
|
|
460
|
+
id=canon_id,
|
|
461
|
+
canonical_id=canon_id,
|
|
462
|
+
name=node.name,
|
|
463
|
+
qualified_name=qname,
|
|
464
|
+
kind=kind,
|
|
465
|
+
language="python",
|
|
466
|
+
module=self.module,
|
|
467
|
+
path=self.file_path,
|
|
468
|
+
file_path=self.file_path,
|
|
469
|
+
scope=parent_scope,
|
|
470
|
+
signature=sig,
|
|
471
|
+
start_line=node.lineno,
|
|
472
|
+
end_line=end_line,
|
|
473
|
+
content_hash=c_hash,
|
|
474
|
+
parent_symbol_id=parent_canon,
|
|
475
|
+
decorators=decorators,
|
|
476
|
+
return_type=ret_type,
|
|
477
|
+
parameter_count=param_count,
|
|
478
|
+
documentation=doc,
|
|
479
|
+
)
|
|
480
|
+
)
|
|
481
|
+
|
|
482
|
+
# Framework analyzer hook: check function for route decorators during this traversal
|
|
483
|
+
ctx = self._make_context()
|
|
484
|
+
for analyzer in self.analyzers:
|
|
485
|
+
detected = analyzer.analyze_python_node(node, ctx)
|
|
486
|
+
if detected:
|
|
487
|
+
self.routes.extend(detected)
|
|
488
|
+
|
|
489
|
+
# Push scope and visit children in single pass
|
|
490
|
+
self.scope_stack.append((node.name, kind, qname, canon_id))
|
|
491
|
+
self.local_bindings_stack.append({})
|
|
492
|
+
for child in node.body:
|
|
493
|
+
self.visit(child)
|
|
494
|
+
self.local_bindings_stack.pop()
|
|
495
|
+
self.scope_stack.pop()
|
|
496
|
+
|
|
497
|
+
def visit_ClassDef(self, node: ast.ClassDef) -> None:
|
|
498
|
+
self._visit_symbol_node(node)
|
|
499
|
+
|
|
500
|
+
def visit_FunctionDef(self, node: ast.FunctionDef) -> None:
|
|
501
|
+
self._visit_symbol_node(node)
|
|
502
|
+
|
|
503
|
+
def visit_AsyncFunctionDef(self, node: ast.AsyncFunctionDef) -> None:
|
|
504
|
+
self._visit_symbol_node(node)
|
|
505
|
+
|
|
506
|
+
def visit_Import(self, node: ast.Import) -> None:
|
|
507
|
+
for alias in node.names:
|
|
508
|
+
self.import_modules_set.add(alias.name)
|
|
509
|
+
self.imports.append(
|
|
510
|
+
ImportRef(
|
|
511
|
+
module=alias.name,
|
|
512
|
+
imported_module=alias.name,
|
|
513
|
+
imported_name=None,
|
|
514
|
+
alias=alias.asname,
|
|
515
|
+
local_name=alias.asname or alias.name.split(".")[0],
|
|
516
|
+
import_type="namespace" if alias.asname else "module",
|
|
517
|
+
line=node.lineno,
|
|
518
|
+
source_file=self.file_path,
|
|
519
|
+
source_module=self.module,
|
|
520
|
+
)
|
|
521
|
+
)
|
|
522
|
+
|
|
523
|
+
def visit_ImportFrom(self, node: ast.ImportFrom) -> None:
|
|
524
|
+
raw_mod = node.module or ""
|
|
525
|
+
resolved_mod = (
|
|
526
|
+
_resolve_relative_py_module(self.file_path, node.level, node.module)
|
|
527
|
+
if node.level > 0
|
|
528
|
+
else raw_mod
|
|
529
|
+
)
|
|
530
|
+
display_mod = raw_mod or resolved_mod
|
|
531
|
+
if resolved_mod:
|
|
532
|
+
self.import_modules_set.add(resolved_mod)
|
|
533
|
+
if raw_mod:
|
|
534
|
+
self.import_modules_set.add(raw_mod)
|
|
535
|
+
|
|
536
|
+
for alias in node.names:
|
|
537
|
+
full = f"{resolved_mod}.{alias.name}" if resolved_mod else alias.name
|
|
538
|
+
is_reexp = self.is_init
|
|
539
|
+
self.imports.append(
|
|
540
|
+
ImportRef(
|
|
541
|
+
module=display_mod,
|
|
542
|
+
imported_module=resolved_mod or display_mod,
|
|
543
|
+
name=alias.name,
|
|
544
|
+
imported_name=alias.name,
|
|
545
|
+
alias=alias.asname,
|
|
546
|
+
local_name=alias.asname or alias.name,
|
|
547
|
+
import_type="reexport" if is_reexp else ("alias" if alias.asname else "named"),
|
|
548
|
+
is_reexport=is_reexp,
|
|
549
|
+
exported_name=alias.asname or alias.name if is_reexp else None,
|
|
550
|
+
line=node.lineno,
|
|
551
|
+
source_file=self.file_path,
|
|
552
|
+
source_module=self.module,
|
|
553
|
+
full=full,
|
|
554
|
+
)
|
|
555
|
+
)
|
|
556
|
+
|
|
557
|
+
def visit_Assign(self, node: ast.Assign) -> None:
|
|
558
|
+
for target in node.targets:
|
|
559
|
+
if isinstance(target, ast.Name) and target.id == "__all__":
|
|
560
|
+
if isinstance(node.value, (ast.List, ast.Tuple, ast.Set)):
|
|
561
|
+
for elt in node.value.elts:
|
|
562
|
+
if isinstance(elt, ast.Constant) and isinstance(elt.value, str):
|
|
563
|
+
self.exports.append(elt.value)
|
|
564
|
+
elif isinstance(target, ast.Name) and isinstance(node.value, ast.Call):
|
|
565
|
+
if isinstance(node.value.func, ast.Name):
|
|
566
|
+
callee_name = node.value.func.id
|
|
567
|
+
if callee_name not in _PYTHON_BUILTINS:
|
|
568
|
+
if callee_name in self.fn_return_types:
|
|
569
|
+
self.local_bindings_stack[-1][target.id] = self.fn_return_types[callee_name]
|
|
570
|
+
elif callee_name in self.known_classes or (callee_name[0].isupper() and "_" not in callee_name):
|
|
571
|
+
self.local_bindings_stack[-1][target.id] = callee_name
|
|
572
|
+
elif any(callee_name.endswith(sfx) for sfx in ("Service", "Client", "Repository", "Manager", "Store", "View")):
|
|
573
|
+
self.local_bindings_stack[-1][target.id] = callee_name
|
|
574
|
+
elif isinstance(node.value.func, ast.Attribute) and hasattr(ast, "unparse"):
|
|
575
|
+
attr_expr = ast.unparse(node.value.func)
|
|
576
|
+
if attr_expr in self.fn_return_types:
|
|
577
|
+
self.local_bindings_stack[-1][target.id] = self.fn_return_types[attr_expr]
|
|
578
|
+
elif attr_expr.split(".")[-1][0].isupper():
|
|
579
|
+
self.local_bindings_stack[-1][target.id] = attr_expr.split(".")[-1]
|
|
580
|
+
elif (
|
|
581
|
+
isinstance(target, ast.Attribute)
|
|
582
|
+
and isinstance(target.value, ast.Name)
|
|
583
|
+
and target.value.id == "self"
|
|
584
|
+
and isinstance(node.value, ast.Call)
|
|
585
|
+
and isinstance(node.value.func, ast.Name)
|
|
586
|
+
):
|
|
587
|
+
callee_name = node.value.func.id
|
|
588
|
+
if callee_name in self.fn_return_types:
|
|
589
|
+
self.local_bindings_stack[-1][f"self.{target.attr}"] = self.fn_return_types[callee_name]
|
|
590
|
+
elif callee_name in self.known_classes or (callee_name[0].isupper() and "_" not in callee_name):
|
|
591
|
+
self.local_bindings_stack[-1][f"self.{target.attr}"] = callee_name
|
|
592
|
+
|
|
593
|
+
self.generic_visit(node)
|
|
594
|
+
|
|
595
|
+
def visit_AnnAssign(self, node: ast.AnnAssign) -> None:
|
|
596
|
+
if isinstance(node.target, ast.Name) and hasattr(ast, "unparse"):
|
|
597
|
+
ann = ast.unparse(node.annotation).split("[")[0].strip()
|
|
598
|
+
if ann and ann not in _PYTHON_BUILTINS:
|
|
599
|
+
self.local_bindings_stack[-1][node.target.id] = ann
|
|
600
|
+
self.generic_visit(node)
|
|
601
|
+
|
|
602
|
+
def visit_Call(self, node: ast.Call) -> None:
|
|
603
|
+
# Framework analyzer hook: check calls (e.g. Django path("login/", handler))
|
|
604
|
+
ctx = self._make_context()
|
|
605
|
+
for analyzer in self.analyzers:
|
|
606
|
+
detected = analyzer.analyze_python_node(node, ctx)
|
|
607
|
+
if detected:
|
|
608
|
+
self.routes.extend(detected)
|
|
609
|
+
|
|
610
|
+
caller_qname = self.current_scope_qname or None
|
|
611
|
+
caller_canon = self.current_scope_canonical_id
|
|
612
|
+
end_ln = node.end_lineno or node.lineno
|
|
613
|
+
|
|
614
|
+
if isinstance(node.func, ast.Name):
|
|
615
|
+
callee_name = node.func.id
|
|
616
|
+
if callee_name not in _PYTHON_BUILTINS:
|
|
617
|
+
self.calls.append(
|
|
618
|
+
CallRef(
|
|
619
|
+
callee=callee_name,
|
|
620
|
+
qualified_callee=callee_name,
|
|
621
|
+
line=node.lineno,
|
|
622
|
+
end_line=end_ln,
|
|
623
|
+
source_file=self.file_path,
|
|
624
|
+
confidence="LOW",
|
|
625
|
+
caller_symbol=caller_qname,
|
|
626
|
+
caller_canonical_id=caller_canon,
|
|
627
|
+
receiver=None,
|
|
628
|
+
)
|
|
629
|
+
)
|
|
630
|
+
elif isinstance(node.func, ast.Attribute):
|
|
631
|
+
attr_name = node.func.attr
|
|
632
|
+
raw_recv = ast.unparse(node.func.value) if hasattr(ast, "unparse") else ""
|
|
633
|
+
recv_clean = raw_recv[:-2].strip() if raw_recv.endswith("()") else raw_recv.strip()
|
|
634
|
+
|
|
635
|
+
enclosing_cls = self.current_enclosing_class
|
|
636
|
+
if recv_clean in ("self", "cls") and enclosing_cls:
|
|
637
|
+
resolved_recv = enclosing_cls
|
|
638
|
+
elif recv_clean.startswith("self.") and self._lookup_local_binding(recv_clean):
|
|
639
|
+
resolved_recv = self._lookup_local_binding(recv_clean) or recv_clean
|
|
640
|
+
else:
|
|
641
|
+
bound = self._lookup_local_binding(recv_clean)
|
|
642
|
+
resolved_recv = bound if bound else recv_clean
|
|
643
|
+
|
|
644
|
+
q_callee = f"{resolved_recv}.{attr_name}" if resolved_recv else attr_name
|
|
645
|
+
self.calls.append(
|
|
646
|
+
CallRef(
|
|
647
|
+
callee=attr_name,
|
|
648
|
+
qualified_callee=q_callee,
|
|
649
|
+
line=node.lineno,
|
|
650
|
+
end_line=end_ln,
|
|
651
|
+
source_file=self.file_path,
|
|
652
|
+
confidence="LOW",
|
|
653
|
+
caller_symbol=caller_qname,
|
|
654
|
+
caller_canonical_id=caller_canon,
|
|
655
|
+
receiver=resolved_recv or None,
|
|
656
|
+
)
|
|
657
|
+
)
|
|
658
|
+
|
|
659
|
+
self.generic_visit(node)
|
|
660
|
+
|
|
661
|
+
|
|
662
|
+
def _parse_python(content: str, file_path: str) -> ParseResult:
|
|
663
|
+
source_hash = hashlib.sha256(content.encode("utf-8", errors="replace")).hexdigest()
|
|
664
|
+
try:
|
|
665
|
+
tree = ast.parse(content)
|
|
666
|
+
except SyntaxError as exc:
|
|
667
|
+
return ParseResult(
|
|
668
|
+
path=file_path,
|
|
669
|
+
language="python",
|
|
670
|
+
source_hash=source_hash,
|
|
671
|
+
imports=tuple(_py_imports_fallback(content, file_path)),
|
|
672
|
+
parse_failed=True,
|
|
673
|
+
parse_error=f"SyntaxError at line {exc.lineno}: {exc.msg}",
|
|
674
|
+
)
|
|
675
|
+
|
|
676
|
+
# First collect import module names cheaply from top-level imports to configure analyzers
|
|
677
|
+
import_mods: set[str] = set()
|
|
678
|
+
for node in tree.body:
|
|
679
|
+
if isinstance(node, ast.Import):
|
|
680
|
+
for alias in node.names:
|
|
681
|
+
import_mods.add(alias.name)
|
|
682
|
+
elif isinstance(node, ast.ImportFrom) and node.module:
|
|
683
|
+
import_mods.add(node.module)
|
|
684
|
+
|
|
685
|
+
analyzers = get_python_analyzers(import_mods)
|
|
686
|
+
lines = content.splitlines()
|
|
687
|
+
fn_returns, known_classes = _extract_python_file_declarations(tree)
|
|
688
|
+
|
|
689
|
+
# Single-pass scoped traversal
|
|
690
|
+
visitor = _PythonScopeVisitor(
|
|
691
|
+
content,
|
|
692
|
+
lines,
|
|
693
|
+
file_path,
|
|
694
|
+
analyzers,
|
|
695
|
+
fn_return_types=fn_returns,
|
|
696
|
+
known_classes=known_classes,
|
|
697
|
+
)
|
|
698
|
+
visitor.visit(tree)
|
|
699
|
+
|
|
700
|
+
# Mark __all__ re-exports
|
|
701
|
+
if visitor.exports:
|
|
702
|
+
exported_set = set(visitor.exports)
|
|
703
|
+
updated_imports: list[ImportRef] = []
|
|
704
|
+
for imp in visitor.imports:
|
|
705
|
+
if imp.local_name in exported_set and not imp.is_reexport:
|
|
706
|
+
updated_imports.append(
|
|
707
|
+
ImportRef(
|
|
708
|
+
module=imp.module,
|
|
709
|
+
imported_module=imp.imported_module,
|
|
710
|
+
name=imp.name,
|
|
711
|
+
imported_name=imp.imported_name,
|
|
712
|
+
alias=imp.alias,
|
|
713
|
+
local_name=imp.local_name,
|
|
714
|
+
import_type="reexport",
|
|
715
|
+
is_reexport=True,
|
|
716
|
+
exported_name=imp.local_name,
|
|
717
|
+
line=imp.line,
|
|
718
|
+
source_file=imp.source_file,
|
|
719
|
+
source_module=imp.source_module,
|
|
720
|
+
full=imp.full,
|
|
721
|
+
)
|
|
722
|
+
)
|
|
723
|
+
else:
|
|
724
|
+
updated_imports.append(imp)
|
|
725
|
+
visitor.imports = updated_imports
|
|
726
|
+
|
|
727
|
+
visitor.symbols.sort(key=lambda s: (s.start_line, s.canonical_id))
|
|
728
|
+
return ParseResult(
|
|
729
|
+
path=file_path,
|
|
730
|
+
language="python",
|
|
731
|
+
source_hash=source_hash,
|
|
732
|
+
symbols=tuple(visitor.symbols),
|
|
733
|
+
imports=tuple(visitor.imports),
|
|
734
|
+
calls=tuple(visitor.calls),
|
|
735
|
+
inheritance=tuple(visitor.inheritance),
|
|
736
|
+
routes=tuple(visitor.routes),
|
|
737
|
+
exports=tuple(visitor.exports),
|
|
738
|
+
parse_failed=False,
|
|
739
|
+
)
|
|
740
|
+
|
|
741
|
+
|
|
742
|
+
def _py_imports_fallback(content: str, file_path: str) -> list[ImportRef]:
|
|
743
|
+
imports: list[ImportRef] = []
|
|
744
|
+
for m in _PY_IMPORT.finditer(content):
|
|
745
|
+
module = (m.group(1) or "").strip() or (m.group(2) or "").strip().split(",")[0].strip()
|
|
746
|
+
if module:
|
|
747
|
+
imports.append(ImportRef(module=module, source_file=file_path))
|
|
748
|
+
return imports
|
|
749
|
+
|
|
750
|
+
|
|
751
|
+
# ---------------------------------------------------------------------------
|
|
752
|
+
# JavaScript / TypeScript single-pass scoped analyzer
|
|
753
|
+
# ---------------------------------------------------------------------------
|
|
754
|
+
|
|
755
|
+
_JS_CLASS_RE = re.compile(
|
|
756
|
+
r"\b(?:export\s+(?:default\s+)?)?(?:abstract\s+)?class\s+([A-Za-z_$][\w$]*)"
|
|
757
|
+
r"(?:\s+extends\s+([A-Za-z_$][\w$.]*))?"
|
|
758
|
+
r"(?:\s+implements\s+([A-Za-z_$][\w$.,\s]*))?\s*\{"
|
|
759
|
+
)
|
|
760
|
+
|
|
761
|
+
_TS_INTERFACE_RE = re.compile(
|
|
762
|
+
r"\b(?:export\s+(?:default\s+)?)?interface\s+([A-Za-z_$][\w$]*)"
|
|
763
|
+
r"(?:\s+extends\s+([A-Za-z_$][\w$.,\s]*))?"
|
|
764
|
+
)
|
|
765
|
+
|
|
766
|
+
_TS_TYPE_RE = re.compile(
|
|
767
|
+
r"\b(?:export\s+)?type\s+([A-Za-z_$][\w$]*)\s*="
|
|
768
|
+
)
|
|
769
|
+
|
|
770
|
+
_JS_FUNC_RE = re.compile(
|
|
771
|
+
r"(?:export\s+(?:default\s+)?)?(?:async\s+)?function\s*\*?\s*([A-Za-z_$][\w$]*)\s*\(([^)]*)\)"
|
|
772
|
+
r"|(?:export\s+)?(?:const|let|var)\s+([A-Za-z_$][\w$]*)\s*(?::\s*[^=]+)?=\s*(?:async\s*)?(?:\([^)]*\)|[A-Za-z_$][\w$]*)\s*=>"
|
|
773
|
+
)
|
|
774
|
+
|
|
775
|
+
_JS_METHOD_RE = re.compile(
|
|
776
|
+
r"^\s{2,}(?:public\s+|private\s+|protected\s+)?(?:async\s+)?(?:static\s+)?(?:get\s+|set\s+)?([A-Za-z_$][\w$]*)\s*\(([^)]*)\)\s*(?::\s*[^{]+)?\{"
|
|
777
|
+
)
|
|
778
|
+
|
|
779
|
+
_JS_IMPORT_NAMED = re.compile(
|
|
780
|
+
r"""import\s+(?:type\s+)?\{([^}]+)\}\s+from\s+['"]([^'"]+)['"]"""
|
|
781
|
+
)
|
|
782
|
+
_JS_IMPORT_NAMESPACE = re.compile(
|
|
783
|
+
r"""import\s+\*\s+as\s+([A-Za-z_$][\w$]*)\s+from\s+['"]([^'"]+)['"]"""
|
|
784
|
+
)
|
|
785
|
+
_JS_IMPORT_DEFAULT = re.compile(
|
|
786
|
+
r"""import\s+([A-Za-z_$][\w$]*)\s*(?:,\s*\{[^}]*\})?\s+from\s+['"]([^'"]+)['"]"""
|
|
787
|
+
)
|
|
788
|
+
_JS_IMPORT_SIDE_EFFECT = re.compile(
|
|
789
|
+
r"""import\s+['"]([^'"]+)['"]|require\(\s*['"]([^'"]+)['"]\s*\)"""
|
|
790
|
+
)
|
|
791
|
+
_JS_REEXPORT_NAMED = re.compile(
|
|
792
|
+
r"""export\s+(?:type\s+)?\{([^}]+)\}\s+from\s+['"]([^'"]+)['"]"""
|
|
793
|
+
)
|
|
794
|
+
_JS_REEXPORT_ALL = re.compile(
|
|
795
|
+
r"""export\s+\*\s*(?:as\s+([A-Za-z_$][\w$]*)\s+)?from\s+['"]([^'"]+)['"]"""
|
|
796
|
+
)
|
|
797
|
+
_JS_NEW_ASSIGN = re.compile(
|
|
798
|
+
r"""(?:const|let|var)\s+([A-Za-z_$][\w$]*)\s*=\s*new\s+([A-Za-z_$][\w$.]*)\s*\("""
|
|
799
|
+
)
|
|
800
|
+
_JS_CALL_SITE = re.compile(
|
|
801
|
+
r"""(?<!function\s)(?<!new\s)\b(?:([A-Za-z_$][\w$]*)\.)?([A-Za-z_$][\w$]*)\s*\("""
|
|
802
|
+
)
|
|
803
|
+
|
|
804
|
+
|
|
805
|
+
def _parse_js_ts(content: str, language: str, file_path: str) -> ParseResult:
|
|
806
|
+
source_hash = hashlib.sha256(content.encode("utf-8", errors="replace")).hexdigest()
|
|
807
|
+
module = normalize_module(file_path, language)
|
|
808
|
+
lines = content.splitlines()
|
|
809
|
+
|
|
810
|
+
symbols: list[Symbol] = []
|
|
811
|
+
imports: list[ImportRef] = []
|
|
812
|
+
calls: list[CallRef] = []
|
|
813
|
+
inheritance: list[InheritanceRef] = []
|
|
814
|
+
routes: list[RouteDetection] = []
|
|
815
|
+
exports: list[str] = []
|
|
816
|
+
|
|
817
|
+
def line_of(offset: int) -> int:
|
|
818
|
+
return content.count("\n", 0, offset) + 1
|
|
819
|
+
|
|
820
|
+
# 1. Imports and Re-exports
|
|
821
|
+
recorded_import_lines: set[tuple[str, str | None, str | None]] = set()
|
|
822
|
+
|
|
823
|
+
for m in _JS_REEXPORT_NAMED.finditer(content):
|
|
824
|
+
specifiers = m.group(1)
|
|
825
|
+
target_mod = m.group(2)
|
|
826
|
+
ln = line_of(m.start())
|
|
827
|
+
for raw_spec in specifiers.split(","):
|
|
828
|
+
spec = raw_spec.strip()
|
|
829
|
+
if not spec:
|
|
830
|
+
continue
|
|
831
|
+
if spec.startswith("type "):
|
|
832
|
+
spec = spec[5:].strip()
|
|
833
|
+
if " as " in spec:
|
|
834
|
+
orig, aliased = [p.strip() for p in spec.split(" as ", 1)]
|
|
835
|
+
else:
|
|
836
|
+
orig, aliased = spec, spec
|
|
837
|
+
exports.append(aliased)
|
|
838
|
+
imports.append(
|
|
839
|
+
ImportRef(
|
|
840
|
+
module=target_mod,
|
|
841
|
+
imported_module=target_mod,
|
|
842
|
+
name=orig,
|
|
843
|
+
imported_name=orig,
|
|
844
|
+
alias=aliased if aliased != orig else None,
|
|
845
|
+
local_name=aliased,
|
|
846
|
+
import_type="reexport",
|
|
847
|
+
is_reexport=True,
|
|
848
|
+
exported_name=aliased,
|
|
849
|
+
line=ln,
|
|
850
|
+
source_file=file_path,
|
|
851
|
+
source_module=module,
|
|
852
|
+
)
|
|
853
|
+
)
|
|
854
|
+
recorded_import_lines.add((target_mod, orig, aliased))
|
|
855
|
+
|
|
856
|
+
for m in _JS_REEXPORT_ALL.finditer(content):
|
|
857
|
+
ns_alias = m.group(1)
|
|
858
|
+
target_mod = m.group(2)
|
|
859
|
+
ln = line_of(m.start())
|
|
860
|
+
imports.append(
|
|
861
|
+
ImportRef(
|
|
862
|
+
module=target_mod,
|
|
863
|
+
imported_module=target_mod,
|
|
864
|
+
name="*",
|
|
865
|
+
imported_name="*",
|
|
866
|
+
alias=ns_alias,
|
|
867
|
+
local_name=ns_alias or "*",
|
|
868
|
+
import_type="reexport",
|
|
869
|
+
is_reexport=True,
|
|
870
|
+
exported_name=ns_alias or "*",
|
|
871
|
+
line=ln,
|
|
872
|
+
source_file=file_path,
|
|
873
|
+
source_module=module,
|
|
874
|
+
)
|
|
875
|
+
)
|
|
876
|
+
recorded_import_lines.add((target_mod, "*", ns_alias))
|
|
877
|
+
|
|
878
|
+
for m in _JS_IMPORT_NAMED.finditer(content):
|
|
879
|
+
specifiers = m.group(1)
|
|
880
|
+
target_mod = m.group(2)
|
|
881
|
+
ln = line_of(m.start())
|
|
882
|
+
for raw_spec in specifiers.split(","):
|
|
883
|
+
spec = raw_spec.strip()
|
|
884
|
+
if not spec:
|
|
885
|
+
continue
|
|
886
|
+
if spec.startswith("type "):
|
|
887
|
+
spec = spec[5:].strip()
|
|
888
|
+
if " as " in spec:
|
|
889
|
+
orig, aliased = [p.strip() for p in spec.split(" as ", 1)]
|
|
890
|
+
else:
|
|
891
|
+
orig, aliased = spec, None
|
|
892
|
+
imports.append(
|
|
893
|
+
ImportRef(
|
|
894
|
+
module=target_mod,
|
|
895
|
+
imported_module=target_mod,
|
|
896
|
+
name=orig,
|
|
897
|
+
imported_name=orig,
|
|
898
|
+
alias=aliased,
|
|
899
|
+
local_name=aliased or orig,
|
|
900
|
+
import_type="alias" if aliased else "named",
|
|
901
|
+
line=ln,
|
|
902
|
+
source_file=file_path,
|
|
903
|
+
source_module=module,
|
|
904
|
+
)
|
|
905
|
+
)
|
|
906
|
+
recorded_import_lines.add((target_mod, orig, aliased))
|
|
907
|
+
|
|
908
|
+
for m in _JS_IMPORT_NAMESPACE.finditer(content):
|
|
909
|
+
ns_alias = m.group(1)
|
|
910
|
+
target_mod = m.group(2)
|
|
911
|
+
ln = line_of(m.start())
|
|
912
|
+
imports.append(
|
|
913
|
+
ImportRef(
|
|
914
|
+
module=target_mod,
|
|
915
|
+
imported_module=target_mod,
|
|
916
|
+
name="*",
|
|
917
|
+
imported_name="*",
|
|
918
|
+
alias=ns_alias,
|
|
919
|
+
local_name=ns_alias,
|
|
920
|
+
import_type="namespace",
|
|
921
|
+
line=ln,
|
|
922
|
+
source_file=file_path,
|
|
923
|
+
source_module=module,
|
|
924
|
+
)
|
|
925
|
+
)
|
|
926
|
+
recorded_import_lines.add((target_mod, "*", ns_alias))
|
|
927
|
+
|
|
928
|
+
for m in _JS_IMPORT_DEFAULT.finditer(content):
|
|
929
|
+
def_name = m.group(1)
|
|
930
|
+
if def_name == "type":
|
|
931
|
+
continue
|
|
932
|
+
target_mod = m.group(2)
|
|
933
|
+
ln = line_of(m.start())
|
|
934
|
+
imports.append(
|
|
935
|
+
ImportRef(
|
|
936
|
+
module=target_mod,
|
|
937
|
+
imported_module=target_mod,
|
|
938
|
+
name="default",
|
|
939
|
+
imported_name="default",
|
|
940
|
+
alias=def_name,
|
|
941
|
+
local_name=def_name,
|
|
942
|
+
import_type="default",
|
|
943
|
+
line=ln,
|
|
944
|
+
source_file=file_path,
|
|
945
|
+
source_module=module,
|
|
946
|
+
)
|
|
947
|
+
)
|
|
948
|
+
recorded_import_lines.add((target_mod, "default", def_name))
|
|
949
|
+
|
|
950
|
+
for m in _JS_IMPORT_SIDE_EFFECT.finditer(content):
|
|
951
|
+
target_mod = m.group(1) or m.group(2)
|
|
952
|
+
if target_mod and not any(r[0] == target_mod for r in recorded_import_lines):
|
|
953
|
+
imports.append(
|
|
954
|
+
ImportRef(
|
|
955
|
+
module=target_mod,
|
|
956
|
+
imported_module=target_mod,
|
|
957
|
+
import_type="module",
|
|
958
|
+
line=line_of(m.start()),
|
|
959
|
+
source_file=file_path,
|
|
960
|
+
source_module=module,
|
|
961
|
+
)
|
|
962
|
+
)
|
|
963
|
+
|
|
964
|
+
local_new_bindings: dict[str, str] = {}
|
|
965
|
+
for m in _JS_NEW_ASSIGN.finditer(content):
|
|
966
|
+
local_new_bindings[m.group(1)] = m.group(2)
|
|
967
|
+
|
|
968
|
+
import_mods_set = {imp.imported_module for imp in imports}
|
|
969
|
+
|
|
970
|
+
# 2. Single-pass scoped line scan
|
|
971
|
+
active_class: tuple[str, str, int] | None = None
|
|
972
|
+
active_func: tuple[str, str, int] | None = None
|
|
973
|
+
seen_canonical: set[str] = set()
|
|
974
|
+
|
|
975
|
+
for idx, line in enumerate(lines):
|
|
976
|
+
lineno = idx + 1
|
|
977
|
+
if active_class and lineno > active_class[2]:
|
|
978
|
+
active_class = None
|
|
979
|
+
if active_func and lineno > active_func[2]:
|
|
980
|
+
active_func = None
|
|
981
|
+
|
|
982
|
+
# Route detection hook per line (Express and Next.js)
|
|
983
|
+
detected_routes = analyze_js_ts_line(line, lineno, file_path, module, import_mods_set)
|
|
984
|
+
if detected_routes:
|
|
985
|
+
routes.extend(detected_routes)
|
|
986
|
+
|
|
987
|
+
# Class declaration
|
|
988
|
+
cls_match = _JS_CLASS_RE.search(line)
|
|
989
|
+
if cls_match:
|
|
990
|
+
cls_name = cls_match.group(1)
|
|
991
|
+
extends_name = cls_match.group(2)
|
|
992
|
+
implements_raw = cls_match.group(3)
|
|
993
|
+
end_ln = _estimate_block_end(lines, idx)
|
|
994
|
+
canon_id = build_canonical_id(module, "", cls_name)
|
|
995
|
+
if canon_id not in seen_canonical:
|
|
996
|
+
seen_canonical.add(canon_id)
|
|
997
|
+
symbols.append(
|
|
998
|
+
Symbol(
|
|
999
|
+
id=canon_id,
|
|
1000
|
+
canonical_id=canon_id,
|
|
1001
|
+
name=cls_name,
|
|
1002
|
+
qualified_name=cls_name,
|
|
1003
|
+
kind="class",
|
|
1004
|
+
language=language,
|
|
1005
|
+
module=module,
|
|
1006
|
+
path=file_path,
|
|
1007
|
+
file_path=file_path,
|
|
1008
|
+
scope="",
|
|
1009
|
+
start_line=lineno,
|
|
1010
|
+
end_line=end_ln,
|
|
1011
|
+
content_hash=_normalized_body_hash(lines, lineno, end_ln),
|
|
1012
|
+
)
|
|
1013
|
+
)
|
|
1014
|
+
if extends_name:
|
|
1015
|
+
inheritance.append(
|
|
1016
|
+
InheritanceRef(
|
|
1017
|
+
source_symbol=cls_name,
|
|
1018
|
+
base_name=extends_name.strip(),
|
|
1019
|
+
relationship="EXTENDS",
|
|
1020
|
+
source_file=file_path,
|
|
1021
|
+
line=lineno,
|
|
1022
|
+
source_canonical_id=canon_id,
|
|
1023
|
+
)
|
|
1024
|
+
)
|
|
1025
|
+
if implements_raw:
|
|
1026
|
+
for iface in implements_raw.split(","):
|
|
1027
|
+
iface_clean = iface.split("<")[0].strip()
|
|
1028
|
+
if iface_clean:
|
|
1029
|
+
inheritance.append(
|
|
1030
|
+
InheritanceRef(
|
|
1031
|
+
source_symbol=cls_name,
|
|
1032
|
+
base_name=iface_clean,
|
|
1033
|
+
relationship="IMPLEMENTS",
|
|
1034
|
+
source_file=file_path,
|
|
1035
|
+
line=lineno,
|
|
1036
|
+
source_canonical_id=canon_id,
|
|
1037
|
+
)
|
|
1038
|
+
)
|
|
1039
|
+
active_class = (cls_name, canon_id, end_ln)
|
|
1040
|
+
|
|
1041
|
+
# TypeScript interface / type
|
|
1042
|
+
if language == "typescript":
|
|
1043
|
+
iface_match = _TS_INTERFACE_RE.search(line)
|
|
1044
|
+
if iface_match:
|
|
1045
|
+
iface_name = iface_match.group(1)
|
|
1046
|
+
extends_raw = iface_match.group(2)
|
|
1047
|
+
end_ln = _estimate_block_end(lines, idx)
|
|
1048
|
+
canon_id = build_canonical_id(module, "", iface_name)
|
|
1049
|
+
if canon_id not in seen_canonical:
|
|
1050
|
+
seen_canonical.add(canon_id)
|
|
1051
|
+
symbols.append(
|
|
1052
|
+
Symbol(
|
|
1053
|
+
id=canon_id,
|
|
1054
|
+
canonical_id=canon_id,
|
|
1055
|
+
name=iface_name,
|
|
1056
|
+
qualified_name=iface_name,
|
|
1057
|
+
kind="interface",
|
|
1058
|
+
language=language,
|
|
1059
|
+
module=module,
|
|
1060
|
+
path=file_path,
|
|
1061
|
+
file_path=file_path,
|
|
1062
|
+
scope="",
|
|
1063
|
+
start_line=lineno,
|
|
1064
|
+
end_line=end_ln,
|
|
1065
|
+
content_hash=_normalized_body_hash(lines, lineno, end_ln),
|
|
1066
|
+
)
|
|
1067
|
+
)
|
|
1068
|
+
if extends_raw:
|
|
1069
|
+
for base_if in extends_raw.split(","):
|
|
1070
|
+
base_clean = base_if.split("<")[0].strip()
|
|
1071
|
+
if base_clean:
|
|
1072
|
+
inheritance.append(
|
|
1073
|
+
InheritanceRef(
|
|
1074
|
+
source_symbol=iface_name,
|
|
1075
|
+
base_name=base_clean,
|
|
1076
|
+
relationship="EXTENDS",
|
|
1077
|
+
source_file=file_path,
|
|
1078
|
+
line=lineno,
|
|
1079
|
+
source_canonical_id=canon_id,
|
|
1080
|
+
)
|
|
1081
|
+
)
|
|
1082
|
+
|
|
1083
|
+
type_match = _TS_TYPE_RE.search(line)
|
|
1084
|
+
if type_match:
|
|
1085
|
+
t_name = type_match.group(1)
|
|
1086
|
+
canon_id = build_canonical_id(module, "", t_name)
|
|
1087
|
+
if canon_id not in seen_canonical:
|
|
1088
|
+
seen_canonical.add(canon_id)
|
|
1089
|
+
symbols.append(
|
|
1090
|
+
Symbol(
|
|
1091
|
+
id=canon_id,
|
|
1092
|
+
canonical_id=canon_id,
|
|
1093
|
+
name=t_name,
|
|
1094
|
+
qualified_name=t_name,
|
|
1095
|
+
kind="type",
|
|
1096
|
+
language=language,
|
|
1097
|
+
module=module,
|
|
1098
|
+
path=file_path,
|
|
1099
|
+
file_path=file_path,
|
|
1100
|
+
scope="",
|
|
1101
|
+
start_line=lineno,
|
|
1102
|
+
end_line=lineno,
|
|
1103
|
+
content_hash=_normalized_body_hash(lines, lineno, lineno),
|
|
1104
|
+
)
|
|
1105
|
+
)
|
|
1106
|
+
|
|
1107
|
+
# Method inside active class
|
|
1108
|
+
if active_class:
|
|
1109
|
+
m_match = _JS_METHOD_RE.match(line)
|
|
1110
|
+
if m_match:
|
|
1111
|
+
m_name = m_match.group(1)
|
|
1112
|
+
if m_name and m_name not in _JS_KEYWORDS_AND_BUILTINS:
|
|
1113
|
+
end_ln = _estimate_block_end(lines, idx)
|
|
1114
|
+
cls_name, cls_canon, _ = active_class
|
|
1115
|
+
qname = f"{cls_name}.{m_name}"
|
|
1116
|
+
canon_id = build_canonical_id(module, cls_name, m_name)
|
|
1117
|
+
if canon_id not in seen_canonical:
|
|
1118
|
+
seen_canonical.add(canon_id)
|
|
1119
|
+
symbols.append(
|
|
1120
|
+
Symbol(
|
|
1121
|
+
id=canon_id,
|
|
1122
|
+
canonical_id=canon_id,
|
|
1123
|
+
name=m_name,
|
|
1124
|
+
qualified_name=qname,
|
|
1125
|
+
kind="method",
|
|
1126
|
+
language=language,
|
|
1127
|
+
module=module,
|
|
1128
|
+
path=file_path,
|
|
1129
|
+
file_path=file_path,
|
|
1130
|
+
scope=cls_name,
|
|
1131
|
+
start_line=lineno,
|
|
1132
|
+
end_line=end_ln,
|
|
1133
|
+
content_hash=_normalized_body_hash(lines, lineno, end_ln),
|
|
1134
|
+
parent_symbol_id=cls_canon,
|
|
1135
|
+
)
|
|
1136
|
+
)
|
|
1137
|
+
active_func = (qname, canon_id, end_ln)
|
|
1138
|
+
|
|
1139
|
+
# Top-level or nested function / arrow function
|
|
1140
|
+
for fn_match in _JS_FUNC_RE.finditer(line):
|
|
1141
|
+
fn_name = fn_match.group(1) or fn_match.group(3)
|
|
1142
|
+
if fn_name and fn_name not in _JS_KEYWORDS_AND_BUILTINS:
|
|
1143
|
+
end_ln = _estimate_block_end(lines, idx)
|
|
1144
|
+
scope_str = active_class[0] if active_class else ""
|
|
1145
|
+
qname = f"{scope_str}.{fn_name}" if scope_str else fn_name
|
|
1146
|
+
canon_id = build_canonical_id(module, scope_str, fn_name)
|
|
1147
|
+
parent_id = active_class[1] if active_class else None
|
|
1148
|
+
kind = "method" if active_class else "function"
|
|
1149
|
+
if canon_id not in seen_canonical:
|
|
1150
|
+
seen_canonical.add(canon_id)
|
|
1151
|
+
symbols.append(
|
|
1152
|
+
Symbol(
|
|
1153
|
+
id=canon_id,
|
|
1154
|
+
canonical_id=canon_id,
|
|
1155
|
+
name=fn_name,
|
|
1156
|
+
qualified_name=qname,
|
|
1157
|
+
kind=kind,
|
|
1158
|
+
language=language,
|
|
1159
|
+
module=module,
|
|
1160
|
+
path=file_path,
|
|
1161
|
+
file_path=file_path,
|
|
1162
|
+
scope=scope_str,
|
|
1163
|
+
start_line=lineno,
|
|
1164
|
+
end_line=end_ln,
|
|
1165
|
+
content_hash=_normalized_body_hash(lines, lineno, end_ln),
|
|
1166
|
+
parent_symbol_id=parent_id,
|
|
1167
|
+
)
|
|
1168
|
+
)
|
|
1169
|
+
active_func = (qname, canon_id, end_ln)
|
|
1170
|
+
|
|
1171
|
+
# Call sites
|
|
1172
|
+
stripped = line.strip()
|
|
1173
|
+
if not stripped.startswith(("import ", "export {", "export *", "interface ", "type ", "//")):
|
|
1174
|
+
caller_qname = (
|
|
1175
|
+
active_func[0]
|
|
1176
|
+
if active_func
|
|
1177
|
+
else (active_class[0] if active_class else None)
|
|
1178
|
+
)
|
|
1179
|
+
caller_canon = (
|
|
1180
|
+
active_func[1]
|
|
1181
|
+
if active_func
|
|
1182
|
+
else (active_class[1] if active_class else module)
|
|
1183
|
+
)
|
|
1184
|
+
for cm in _JS_CALL_SITE.finditer(line):
|
|
1185
|
+
recv = cm.group(1)
|
|
1186
|
+
callee_nm = cm.group(2)
|
|
1187
|
+
if callee_nm in _JS_KEYWORDS_AND_BUILTINS:
|
|
1188
|
+
continue
|
|
1189
|
+
if recv in _JS_KEYWORDS_AND_BUILTINS and recv != "this":
|
|
1190
|
+
continue
|
|
1191
|
+
resolved_recv = recv
|
|
1192
|
+
if recv == "this" and active_class:
|
|
1193
|
+
resolved_recv = active_class[0]
|
|
1194
|
+
elif recv and recv in local_new_bindings:
|
|
1195
|
+
resolved_recv = local_new_bindings[recv]
|
|
1196
|
+
q_callee = f"{resolved_recv}.{callee_nm}" if resolved_recv else callee_nm
|
|
1197
|
+
calls.append(
|
|
1198
|
+
CallRef(
|
|
1199
|
+
callee=callee_nm,
|
|
1200
|
+
qualified_callee=q_callee,
|
|
1201
|
+
line=lineno,
|
|
1202
|
+
end_line=lineno,
|
|
1203
|
+
source_file=file_path,
|
|
1204
|
+
confidence="LOW",
|
|
1205
|
+
caller_symbol=caller_qname,
|
|
1206
|
+
caller_canonical_id=caller_canon,
|
|
1207
|
+
receiver=resolved_recv,
|
|
1208
|
+
)
|
|
1209
|
+
)
|
|
1210
|
+
|
|
1211
|
+
symbols.sort(key=lambda s: (s.start_line, s.canonical_id))
|
|
1212
|
+
return ParseResult(
|
|
1213
|
+
path=file_path,
|
|
1214
|
+
language=language,
|
|
1215
|
+
source_hash=source_hash,
|
|
1216
|
+
symbols=tuple(symbols),
|
|
1217
|
+
imports=tuple(imports),
|
|
1218
|
+
calls=tuple(calls),
|
|
1219
|
+
inheritance=tuple(inheritance),
|
|
1220
|
+
routes=tuple(routes),
|
|
1221
|
+
exports=tuple(exports),
|
|
1222
|
+
parse_failed=False,
|
|
1223
|
+
)
|
|
1224
|
+
|
|
1225
|
+
|
|
1226
|
+
def _estimate_block_end(lines: list[str], start_idx: int) -> int:
|
|
1227
|
+
"""Walk forward counting braces to estimate block end line."""
|
|
1228
|
+
depth = 0
|
|
1229
|
+
opened = False
|
|
1230
|
+
for i, line in enumerate(lines[start_idx:], start=start_idx):
|
|
1231
|
+
opens = line.count("{")
|
|
1232
|
+
closes = line.count("}")
|
|
1233
|
+
if opens > 0:
|
|
1234
|
+
opened = True
|
|
1235
|
+
depth += opens - closes
|
|
1236
|
+
if opened and depth <= 0:
|
|
1237
|
+
return i + 1
|
|
1238
|
+
if not opened and i == start_idx and ";" in line:
|
|
1239
|
+
return i + 1
|
|
1240
|
+
return len(lines)
|