devlensio 1.0.3 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/config/backcompat.test.d.ts +1 -0
- package/dist/config/backcompat.test.js +188 -0
- package/dist/config/index.d.ts +5 -3
- package/dist/config/index.js +37 -54
- package/dist/config/providers/file.d.ts +3 -1
- package/dist/config/providers/file.js +47 -53
- package/dist/config/providers/providers.default.js +0 -1
- package/dist/config/providers/request.js +1 -1
- package/dist/config/types.d.ts +0 -1
- package/dist/config/types.js +2 -18
- package/dist/graph/edges/callEdges.js +97 -5
- package/dist/graph/edges/callEdges.test.js +154 -0
- package/dist/index.d.ts +1 -1
- package/dist/index.js +1 -1
- package/dist/jobs/queue/memory.js +12 -1
- package/dist/parser/extractors/classes.js +23 -4
- package/dist/parser/extractors/functions.d.ts +10 -3
- package/dist/parser/extractors/functions.js +135 -65
- package/dist/parser/extractors/hooks.js +5 -29
- package/dist/parser/extractors/objectMethods.js +3 -4
- package/dist/parser/index.js +11 -6
- package/dist/parser/overloads.d.ts +2 -0
- package/dist/parser/overloads.js +96 -0
- package/dist/parser/overloads.test.d.ts +1 -0
- package/dist/parser/overloads.test.js +175 -0
- package/dist/parser/typeUtils.d.ts +2 -0
- package/dist/parser/typeUtils.js +2 -0
- package/dist/summarizer/providers/index.js +4 -4
- package/dist/summarizer/providers/openai.js +1 -1
- package/dist/types.d.ts +6 -0
- package/extractors/go/bin/darwin-amd64/devlens_go_extractor +0 -0
- package/extractors/go/bin/darwin-arm64/devlens_go_extractor +0 -0
- package/extractors/go/bin/linux-amd64/devlens_go_extractor +0 -0
- package/extractors/go/bin/linux-arm64/devlens_go_extractor +0 -0
- package/extractors/go/bin/windows-amd64/devlens_go_extractor.exe +0 -0
- package/extractors/go/calls.go +134 -2
- package/extractors/go/contract.go +1 -0
- package/extractors/go/exports.go +107 -0
- package/extractors/go/exports_test.go +64 -0
- package/extractors/go/extractor.go +1 -0
- package/extractors/go/lookup.go +63 -7
- package/extractors/go/nodes.go +28 -0
- package/extractors/go/parser.go +51 -8
- package/extractors/java/devlens_java_extractor.jar +0 -0
- package/extractors/java/src/devlens/extractor/ExportsMapBuilder.java +148 -0
- package/extractors/java/src/devlens/extractor/Extractor.java +34 -0
- package/extractors/java/src/devlens/extractor/LookupMaps.java +12 -0
- package/extractors/java/src/devlens/extractor/Overloads.java +290 -0
- package/extractors/java/src/devlens/extractor/Parser.java +39 -0
- package/extractors/java/src/devlens/extractor/TypeSolverFactory.java +9 -0
- package/extractors/java/src/devlens/extractor/edges/Calls.java +42 -36
- package/extractors/java/src/devlens/extractor/edges/Routes.java +10 -3
- package/extractors/python/devlens_extractors_python/edges/calls.py +127 -26
- package/extractors/python/devlens_extractors_python/exports_map.py +263 -0
- package/extractors/python/devlens_extractors_python/extractor.py +7 -1
- package/extractors/python/devlens_extractors_python/lookup.py +6 -0
- package/extractors/python/devlens_extractors_python/parser/__init__.py +6 -0
- package/extractors/python/devlens_extractors_python/parser/functions.py +103 -6
- package/extractors/python/devlens_extractors_python/parser/overloads.py +105 -0
- package/extractors/rust/bin/darwin-amd64/devlens_rust_extractor +0 -0
- package/extractors/rust/bin/darwin-arm64/devlens_rust_extractor +0 -0
- package/extractors/rust/bin/linux-amd64/devlens_rust_extractor +0 -0
- package/extractors/rust/bin/linux-arm64/devlens_rust_extractor +0 -0
- package/extractors/rust/bin/windows-amd64/devlens_rust_extractor.exe +0 -0
- package/extractors/rust/src/calls.rs +27 -16
- package/extractors/rust/src/contract.rs +2 -0
- package/extractors/rust/src/exports.rs +387 -0
- package/extractors/rust/src/exports_test.rs +105 -0
- package/extractors/rust/src/extractor.rs +2 -0
- package/extractors/rust/src/lookup.rs +60 -0
- package/extractors/rust/src/main.rs +4 -0
- package/extractors/rust/src/nodes.rs +19 -0
- package/extractors/rust/src/parser.rs +66 -1
- package/extractors/rust/src/routes.rs +2 -1
- package/package.json +10 -3
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
"""CALLS edge resolution — resolves metadata.
|
|
1
|
+
"""CALLS edge resolution — resolves metadata.callSites names to node ids.
|
|
2
2
|
|
|
3
3
|
Consumes the shared LookupMaps (lookup.py). Resolution ladder:
|
|
4
4
|
self./cls. → same-class method (exact dotted name)
|
|
@@ -8,6 +8,11 @@ Consumes the shared LookupMaps (lookup.py). Resolution ladder:
|
|
|
8
8
|
plain name → global nodes_by_name → import alias refinement
|
|
9
9
|
Unknown/builtin names resolve to nothing (no edge) — metadata.calls keeps
|
|
10
10
|
them for LLM context.
|
|
11
|
+
|
|
12
|
+
Overload-aware tiebreak (mirrors src/graph/edges/callEdges.ts): when a lookup
|
|
13
|
+
returns multiple same-name candidates, the call site's argument shape picks
|
|
14
|
+
the target — arity compatibility first, argTypes agreement second, then the
|
|
15
|
+
legacy closest-by-path heuristic. The edge records matchedBy.
|
|
11
16
|
"""
|
|
12
17
|
from __future__ import annotations
|
|
13
18
|
|
|
@@ -19,7 +24,97 @@ def _find_in_file(lookup: LookupMaps, rel: str, name: str) -> str | None:
|
|
|
19
24
|
return lookup.nodes_by_file.get(rel, {}).get(name)
|
|
20
25
|
|
|
21
26
|
|
|
22
|
-
def
|
|
27
|
+
def _find_all_in_file(lookup: LookupMaps, rel: str, name: str) -> list[tuple[str, str]]:
|
|
28
|
+
"""Every same-name node id in the file, in declaration order."""
|
|
29
|
+
ids = lookup.nodes_by_file_all.get(rel, {}).get(name, [])
|
|
30
|
+
return [(node_id, rel) for node_id in ids]
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _bound_params(node: dict) -> tuple[list, list]:
|
|
34
|
+
"""(effective_params, effective_parameters) for arity/type matching —
|
|
35
|
+
bound methods hide self/cls from the call site's argument count."""
|
|
36
|
+
metadata = node.get("metadata", {})
|
|
37
|
+
params = metadata.get("params") or []
|
|
38
|
+
parameters = metadata.get("parameters") or []
|
|
39
|
+
if node.get("type") == "METHOD" and params and params[0] in ("self", "cls"):
|
|
40
|
+
params = params[1:]
|
|
41
|
+
parameters = [p for p in parameters if p.get("name") not in ("self", "cls")]
|
|
42
|
+
return params, parameters
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _arity_matches(node: dict, arg_count: int) -> bool:
|
|
46
|
+
"""A candidate accepts the call when argCount covers every required
|
|
47
|
+
(non-optional, non-rest) param and does not exceed the declared count
|
|
48
|
+
unless a rest param (*args/**kwargs) absorbs the extras. Nodes without
|
|
49
|
+
parameter shape metadata are permissive."""
|
|
50
|
+
params, parameters = _bound_params(node)
|
|
51
|
+
if not params and not parameters:
|
|
52
|
+
return True
|
|
53
|
+
total = len(parameters) if parameters else len(params)
|
|
54
|
+
required = sum(1 for p in parameters if not p.get("isOptional") and not p.get("isRest")) \
|
|
55
|
+
if parameters else total
|
|
56
|
+
has_rest = any(p.get("isRest") for p in parameters) if parameters else False
|
|
57
|
+
return arg_count >= required and (has_rest or arg_count <= total)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _type_agrees(arg_type: str, param_type) -> bool:
|
|
61
|
+
if not arg_type or arg_type == "unknown" or not param_type:
|
|
62
|
+
return False
|
|
63
|
+
return param_type == arg_type or arg_type in param_type
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _type_score(node: dict, arg_types: list[str]) -> int:
|
|
67
|
+
_, parameters = _bound_params(node)
|
|
68
|
+
score = 0
|
|
69
|
+
for i, t in enumerate(arg_types):
|
|
70
|
+
if i >= len(parameters):
|
|
71
|
+
break
|
|
72
|
+
if _type_agrees(t, parameters[i].get("type")):
|
|
73
|
+
score += 1
|
|
74
|
+
return score
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _pick(candidates: list[tuple[str, str]], node: dict, call_site: dict | None,
|
|
78
|
+
lookup: LookupMaps, rel: str) -> tuple[str | None, str]:
|
|
79
|
+
"""Ladder: single hit → name · arity filter → type tiebreak → path.
|
|
80
|
+
Returns (target_id, matchedBy)."""
|
|
81
|
+
if not candidates:
|
|
82
|
+
return None, "name"
|
|
83
|
+
if len(candidates) == 1:
|
|
84
|
+
return candidates[0][0], "name"
|
|
85
|
+
|
|
86
|
+
usable_shape = (call_site is not None
|
|
87
|
+
and not call_site.get("hasSpread")
|
|
88
|
+
and isinstance(call_site.get("argCount"), int))
|
|
89
|
+
if not usable_shape:
|
|
90
|
+
return closest_by_path(candidates, rel), "name"
|
|
91
|
+
|
|
92
|
+
arg_count = call_site["argCount"]
|
|
93
|
+
pool = [c for c in candidates
|
|
94
|
+
if _arity_matches(lookup.node_by_id.get(c[0], {}), arg_count)]
|
|
95
|
+
if not pool:
|
|
96
|
+
pool = candidates # nothing accepts this arity — keep all
|
|
97
|
+
if len(pool) == 1:
|
|
98
|
+
return pool[0][0], "arity"
|
|
99
|
+
|
|
100
|
+
arg_types = call_site.get("argTypes") or []
|
|
101
|
+
if any(t and t != "unknown" for t in arg_types):
|
|
102
|
+
best = pool[0]
|
|
103
|
+
best_score = _type_score(lookup.node_by_id.get(best[0], {}), arg_types)
|
|
104
|
+
tie_at_best = 1
|
|
105
|
+
for cand in pool[1:]:
|
|
106
|
+
score = _type_score(lookup.node_by_id.get(cand[0], {}), arg_types)
|
|
107
|
+
if score > best_score:
|
|
108
|
+
best, best_score, tie_at_best = cand, score, 1
|
|
109
|
+
elif score == best_score:
|
|
110
|
+
tie_at_best += 1
|
|
111
|
+
if best_score > 0 and tie_at_best == 1:
|
|
112
|
+
return best[0], "signature"
|
|
113
|
+
|
|
114
|
+
return closest_by_path(pool, rel), "name"
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _walk_module(lookup: LookupMaps, alias_target: str, rest: str) -> list[tuple[str, str]]:
|
|
23
118
|
"""'user.User' with alias → file::models/__init__.py: consume submodule
|
|
24
119
|
segments via module_map, then look up the symbol in the final file."""
|
|
25
120
|
current_rel = alias_target[len("file::"):]
|
|
@@ -35,11 +130,11 @@ def _walk_module(lookup: LookupMaps, alias_target: str, rest: str) -> str | None
|
|
|
35
130
|
current_rel, current_module = nxt, candidate
|
|
36
131
|
consumed += 1
|
|
37
132
|
|
|
38
|
-
return
|
|
133
|
+
return _find_all_in_file(lookup, current_rel, ".".join(segments[consumed:]))
|
|
39
134
|
|
|
40
135
|
|
|
41
136
|
def _resolve(called: str, node: dict, lookup: LookupMaps,
|
|
42
|
-
registry: ThirdPartyRegistry) -> str | None:
|
|
137
|
+
registry: ThirdPartyRegistry, call_site: dict | None) -> tuple[str | None, str]:
|
|
43
138
|
rel = node["filePath"]
|
|
44
139
|
symbols = lookup.symbol_maps.get(rel, {})
|
|
45
140
|
|
|
@@ -47,9 +142,9 @@ def _resolve(called: str, node: dict, lookup: LookupMaps,
|
|
|
47
142
|
if called.startswith(("self.", "cls.")):
|
|
48
143
|
parent_class = node["metadata"].get("parentClass")
|
|
49
144
|
if parent_class:
|
|
50
|
-
return
|
|
51
|
-
|
|
52
|
-
return None
|
|
145
|
+
return _pick(_find_all_in_file(lookup, rel, f"{parent_class}.{called.split('.', 1)[1]}"),
|
|
146
|
+
node, call_site, lookup, rel)
|
|
147
|
+
return None, "name"
|
|
53
148
|
|
|
54
149
|
# ── 2. dotted name ──
|
|
55
150
|
if "." in called:
|
|
@@ -57,56 +152,62 @@ def _resolve(called: str, node: dict, lookup: LookupMaps,
|
|
|
57
152
|
|
|
58
153
|
# 2a. ClassName.method defined in the same file
|
|
59
154
|
if root in lookup.nodes_by_file.get(rel, {}):
|
|
60
|
-
return
|
|
155
|
+
return _pick(_find_all_in_file(lookup, rel, called),
|
|
156
|
+
node, call_site, lookup, rel)
|
|
61
157
|
|
|
62
158
|
# 2b. member access through an import alias
|
|
63
159
|
alias_target = symbols.get(root)
|
|
64
160
|
if not alias_target:
|
|
65
|
-
return None
|
|
161
|
+
return None, "name"
|
|
66
162
|
if alias_target.startswith("[pip]/"):
|
|
67
163
|
# third-party chain: requests.get → lazily create [pip]/requests::get
|
|
68
164
|
pkg = alias_target.split("/", 1)[1]
|
|
69
165
|
if "::" in pkg:
|
|
70
166
|
# chain root is already a named import — current_app.logger.info
|
|
71
167
|
# → [pip]/flask::current_app (the meaningful hop)
|
|
72
|
-
return alias_target
|
|
168
|
+
return alias_target, "name"
|
|
73
169
|
member = registry.method_node(pkg, rest.split(".")[0])
|
|
74
|
-
return member["id"] if member else None
|
|
170
|
+
return (member["id"] if member else None), "name"
|
|
75
171
|
if alias_target.startswith("file::"):
|
|
76
|
-
return _walk_module(lookup, alias_target, rest)
|
|
77
|
-
|
|
172
|
+
return _pick(_walk_module(lookup, alias_target, rest),
|
|
173
|
+
node, call_site, lookup, rel)
|
|
174
|
+
return None, "name"
|
|
78
175
|
|
|
79
|
-
# ── 3. plain name — global lookup,
|
|
176
|
+
# ── 3. plain name — global lookup, overload ladder on collision ──
|
|
80
177
|
candidates = lookup.nodes_by_name.get(called)
|
|
81
178
|
if candidates:
|
|
82
|
-
|
|
83
|
-
return candidates[0][0]
|
|
84
|
-
return closest_by_path(candidates, rel)
|
|
179
|
+
return _pick(candidates, node, call_site, lookup, rel)
|
|
85
180
|
|
|
86
181
|
# ── 4. import alias — refine file targets to the actual symbol ──
|
|
87
182
|
alias_target = symbols.get(called)
|
|
88
183
|
if alias_target:
|
|
89
184
|
if alias_target.startswith("file::"):
|
|
90
|
-
return
|
|
91
|
-
|
|
92
|
-
|
|
185
|
+
return _pick(_find_all_in_file(lookup, alias_target[len("file::"):], called),
|
|
186
|
+
node, call_site, lookup, rel)
|
|
187
|
+
return alias_target, "name" # "[pip]/..." or "path.py::Symbol"
|
|
188
|
+
return None, "name"
|
|
93
189
|
|
|
94
190
|
|
|
95
191
|
def resolve_calls(lookup: LookupMaps, registry: ThirdPartyRegistry) -> list[dict]:
|
|
96
|
-
"""Resolve every FUNCTION/METHOD node's
|
|
97
|
-
Also writes metadata.resolvedCalls back onto nodes (JS callEdges.ts mirror).
|
|
192
|
+
"""Resolve every FUNCTION/METHOD node's call sites → CALLS edges.
|
|
193
|
+
Also writes metadata.resolvedCalls back onto nodes (JS callEdges.ts mirror).
|
|
194
|
+
Each (name, arity, argTypes) call site resolves separately — a caller that
|
|
195
|
+
invokes two overload siblings of one name gets two edges."""
|
|
98
196
|
edges: list[dict] = []
|
|
99
197
|
|
|
100
198
|
for node in lookup.node_by_id.values():
|
|
101
199
|
if node["type"] not in ("FUNCTION", "METHOD"):
|
|
102
200
|
continue
|
|
201
|
+
metadata = node["metadata"]
|
|
202
|
+
sites = metadata.get("callSites") or [{"name": c} for c in metadata.get("calls", [])]
|
|
103
203
|
resolved: list[dict] = []
|
|
104
|
-
for
|
|
105
|
-
|
|
204
|
+
for site in sites:
|
|
205
|
+
called = site["name"]
|
|
206
|
+
target, matched_by = _resolve(called, node, lookup, registry, site)
|
|
106
207
|
if target and target != node["id"]:
|
|
107
208
|
edges.append({"from": node["id"], "to": target, "type": "CALLS",
|
|
108
|
-
"metadata": {"calledName": called}})
|
|
209
|
+
"metadata": {"calledName": called, "matchedBy": matched_by}})
|
|
109
210
|
resolved.append({"name": called, "nodeId": target})
|
|
110
|
-
|
|
211
|
+
metadata["resolvedCalls"] = resolved
|
|
111
212
|
|
|
112
213
|
return edges
|
|
@@ -0,0 +1,263 @@
|
|
|
1
|
+
"""Exports Map — the flattened public surface of a Python repo.
|
|
2
|
+
|
|
3
|
+
Mirrors the TypeScript reference (src/parser/exportsMap.ts) and the cross-repo
|
|
4
|
+
design (docs/EXPORTS_MAP.md in devlens-engine). Every value is the FINAL,
|
|
5
|
+
resolved nodeId as it appears in the nodes array — overload suffixes included.
|
|
6
|
+
Consumers of the map never walk import chains.
|
|
7
|
+
|
|
8
|
+
Python-specific concerns:
|
|
9
|
+
- The "door" is the package root __init__.py; the map key is the name an
|
|
10
|
+
importer sees (`from pkg import name`). Python has no package.json-style
|
|
11
|
+
subpaths, so single-package repos use "." and multi-package repos use
|
|
12
|
+
"./<pkgname>" per top-level package.
|
|
13
|
+
- __all__ (when present, plain list assignment) defines membership exactly;
|
|
14
|
+
otherwise public = non-underscore top-level bindings.
|
|
15
|
+
- `from .mod import name as alias` binds alias → the definition of `name`
|
|
16
|
+
(bindings tracked, never string-matched — name-swap safe).
|
|
17
|
+
- `from .mod import *` expands transitively with cycle guards; underscore
|
|
18
|
+
names stay excluded per Python semantics; explicit bindings always beat
|
|
19
|
+
star-collected ones, and a later star overrides an earlier star.
|
|
20
|
+
- Later bindings override earlier ones (Python semantics: last wins), so
|
|
21
|
+
duplicate imports resolve deterministically; ambiguousNames stays a
|
|
22
|
+
safety net that should remain empty for well-formed Python.
|
|
23
|
+
- Re-export chains through nested __init__.py files are followed
|
|
24
|
+
recursively; modules re-exported as values map to FILE nodes
|
|
25
|
+
(file::pkg/submod.py).
|
|
26
|
+
- Plain variables have no nodes → omitted (never fabricate an id).
|
|
27
|
+
|
|
28
|
+
Output shape (camelCase, same as the TS extractor):
|
|
29
|
+
{"exports": {subpath: {name: [nodeId, ...]}},
|
|
30
|
+
"ambiguousNames": {subpath: [name, ...]}}
|
|
31
|
+
"""
|
|
32
|
+
from __future__ import annotations
|
|
33
|
+
|
|
34
|
+
import ast
|
|
35
|
+
from pathlib import Path
|
|
36
|
+
|
|
37
|
+
_NON_PACKAGE_DIRS = {"tests", "docs", "examples", "scripts", "benchmarks", "node_modules", ".venv", "venv", "build", "dist"}
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def build_exports_map(repo_path: str, parsed_files: list, lookup) -> dict | None:
|
|
41
|
+
try:
|
|
42
|
+
packages = _discover_packages(repo_path)
|
|
43
|
+
if not packages:
|
|
44
|
+
return None
|
|
45
|
+
|
|
46
|
+
exports: dict[str, dict[str, list[str]]] = {}
|
|
47
|
+
ambiguous: dict[str, list[str]] = {}
|
|
48
|
+
parsed_by_rel = {pf.rel_path: pf for pf in parsed_files}
|
|
49
|
+
single = len(packages) == 1
|
|
50
|
+
|
|
51
|
+
for pkg_name, init_rel in packages:
|
|
52
|
+
subpath = "." if single else f"./{pkg_name}"
|
|
53
|
+
per_path, amb = _package_surface(pkg_name, init_rel, parsed_by_rel, lookup)
|
|
54
|
+
if per_path:
|
|
55
|
+
exports[subpath] = per_path
|
|
56
|
+
if amb:
|
|
57
|
+
ambiguous[subpath] = amb
|
|
58
|
+
|
|
59
|
+
if not exports:
|
|
60
|
+
return None
|
|
61
|
+
return {"exports": exports, "ambiguousNames": ambiguous}
|
|
62
|
+
except Exception:
|
|
63
|
+
return None
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _discover_packages(repo_path: str) -> list[tuple[str, str]]:
|
|
67
|
+
root = Path(repo_path)
|
|
68
|
+
found: list[tuple[str, str]] = []
|
|
69
|
+
for entry in sorted(root.iterdir()):
|
|
70
|
+
if not entry.is_dir() or entry.name.startswith(".") or entry.name in _NON_PACKAGE_DIRS:
|
|
71
|
+
continue
|
|
72
|
+
if (entry / "__init__.py").exists():
|
|
73
|
+
found.append((entry.name, f"{entry.name}/__init__.py"))
|
|
74
|
+
return found
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _package_surface(pkg_name: str, init_rel: str, parsed_by_rel: dict, lookup):
|
|
78
|
+
init_pf = parsed_by_rel.get(init_rel)
|
|
79
|
+
if init_pf is None:
|
|
80
|
+
return {}, []
|
|
81
|
+
|
|
82
|
+
bindings, star_modules = _collect_bindings(init_pf, pkg_name, lookup.module_map)
|
|
83
|
+
|
|
84
|
+
star_names: dict[str, list[tuple[str, str, str]]] = {}
|
|
85
|
+
for module in star_modules:
|
|
86
|
+
per_star: dict[str, list[tuple[str, str, str]]] = {}
|
|
87
|
+
_expand_star(module, per_star, lookup, parsed_by_rel, set())
|
|
88
|
+
for name, entries in per_star.items():
|
|
89
|
+
star_names[name] = entries
|
|
90
|
+
|
|
91
|
+
for name, entries in star_names.items():
|
|
92
|
+
if name not in bindings:
|
|
93
|
+
bindings[name] = entries
|
|
94
|
+
|
|
95
|
+
per_path: dict[str, list[str]] = {}
|
|
96
|
+
ambiguous: list[str] = []
|
|
97
|
+
|
|
98
|
+
public = _public_names(init_pf)
|
|
99
|
+
for name in sorted(set(list(bindings.keys()) + public)):
|
|
100
|
+
if name.startswith("_"):
|
|
101
|
+
continue
|
|
102
|
+
targets = bindings.get(name, [])
|
|
103
|
+
resolved_ids: list[str] = []
|
|
104
|
+
definition_files: set[str] = set()
|
|
105
|
+
for kind, target, original in targets:
|
|
106
|
+
ids = _resolve_binding(kind, target, original, parsed_by_rel, lookup, set())
|
|
107
|
+
if ids:
|
|
108
|
+
resolved_ids.extend(ids)
|
|
109
|
+
definition_files.add(_definition_file(ids[0]))
|
|
110
|
+
if not resolved_ids:
|
|
111
|
+
continue
|
|
112
|
+
unique_ids = sorted(set(resolved_ids))
|
|
113
|
+
if len(definition_files) > 1:
|
|
114
|
+
ambiguous.append(name)
|
|
115
|
+
per_path[name] = unique_ids
|
|
116
|
+
|
|
117
|
+
return per_path, ambiguous
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _collect_bindings(pf, pkg_name: str, module_map: dict[str, str]):
|
|
121
|
+
"""One pass over a module's top-level statements → name → binding targets."""
|
|
122
|
+
bindings: dict[str, list[tuple[str, str, str]]] = {}
|
|
123
|
+
star_modules: list[str] = []
|
|
124
|
+
|
|
125
|
+
for stmt in pf.tree.body:
|
|
126
|
+
if isinstance(stmt, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)):
|
|
127
|
+
bindings[stmt.name] = [("def", pf.rel_path, stmt.name)]
|
|
128
|
+
elif isinstance(stmt, ast.ImportFrom):
|
|
129
|
+
module = _absolute_module(stmt, pf, pkg_name)
|
|
130
|
+
if module is None:
|
|
131
|
+
continue
|
|
132
|
+
for alias in stmt.names:
|
|
133
|
+
if alias.name == "*":
|
|
134
|
+
if module not in star_modules:
|
|
135
|
+
star_modules.append(module)
|
|
136
|
+
continue
|
|
137
|
+
public_name = alias.asname or alias.name
|
|
138
|
+
bindings[public_name] = [("import", module, alias.name)]
|
|
139
|
+
elif isinstance(stmt, ast.Import):
|
|
140
|
+
for alias in stmt.names:
|
|
141
|
+
public_name = alias.asname or alias.name.split(".")[0]
|
|
142
|
+
module = alias.name if alias.asname else alias.name.split(".")[0]
|
|
143
|
+
leaf = module.split(".")[-1]
|
|
144
|
+
bindings[public_name] = [("import", module, leaf)]
|
|
145
|
+
|
|
146
|
+
return bindings, star_modules
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def _absolute_module(stmt: ast.ImportFrom, pf, pkg_name: str) -> str | None:
|
|
150
|
+
if stmt.level == 0:
|
|
151
|
+
return stmt.module or None
|
|
152
|
+
base_parts = pkg_name.split(".")
|
|
153
|
+
base_parts = base_parts[: max(0, len(base_parts) - (stmt.level - 1))]
|
|
154
|
+
base = ".".join(base_parts)
|
|
155
|
+
if not base:
|
|
156
|
+
return None
|
|
157
|
+
return f"{base}.{stmt.module}" if stmt.module else base
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def _expand_star(
|
|
161
|
+
module: str,
|
|
162
|
+
star_names: dict[str, list[tuple[str, str, str]]],
|
|
163
|
+
lookup,
|
|
164
|
+
parsed_by_rel: dict,
|
|
165
|
+
visited: set[str],
|
|
166
|
+
):
|
|
167
|
+
if module in visited:
|
|
168
|
+
return
|
|
169
|
+
visited.add(module)
|
|
170
|
+
|
|
171
|
+
rel = lookup.module_map.get(module)
|
|
172
|
+
pf = parsed_by_rel.get(rel) if rel else None
|
|
173
|
+
if pf is None:
|
|
174
|
+
return
|
|
175
|
+
|
|
176
|
+
bindings, sub_stars = _collect_bindings(pf, module, lookup.module_map)
|
|
177
|
+
for sub in sub_stars:
|
|
178
|
+
_expand_star(sub, star_names, lookup, parsed_by_rel, visited)
|
|
179
|
+
|
|
180
|
+
for name, entries in bindings.items():
|
|
181
|
+
if name.startswith("_") or name == "__all__":
|
|
182
|
+
continue
|
|
183
|
+
bucket = star_names.setdefault(name, [])
|
|
184
|
+
for entry in entries:
|
|
185
|
+
if entry not in bucket:
|
|
186
|
+
bucket.append(entry)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def _public_names(pf) -> list[str]:
|
|
190
|
+
for stmt in pf.tree.body:
|
|
191
|
+
if isinstance(stmt, ast.Assign):
|
|
192
|
+
for target in stmt.targets:
|
|
193
|
+
if isinstance(target, ast.Name) and target.id == "__all__":
|
|
194
|
+
try:
|
|
195
|
+
return [
|
|
196
|
+
elt.value
|
|
197
|
+
for elt in stmt.value.elts
|
|
198
|
+
if isinstance(elt, ast.Constant) and isinstance(elt.value, str)
|
|
199
|
+
]
|
|
200
|
+
except AttributeError:
|
|
201
|
+
return []
|
|
202
|
+
return [n for n in _top_level_names(pf) if not n.startswith("_")]
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def _top_level_names(pf) -> list[str]:
|
|
206
|
+
names: list[str] = []
|
|
207
|
+
for stmt in pf.tree.body:
|
|
208
|
+
if isinstance(stmt, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)):
|
|
209
|
+
names.append(stmt.name)
|
|
210
|
+
elif isinstance(stmt, ast.Assign):
|
|
211
|
+
for target in stmt.targets:
|
|
212
|
+
if isinstance(target, ast.Name):
|
|
213
|
+
names.append(target.id)
|
|
214
|
+
elif isinstance(stmt, ast.AnnAssign):
|
|
215
|
+
if isinstance(stmt.target, ast.Name):
|
|
216
|
+
names.append(stmt.target.id)
|
|
217
|
+
elif isinstance(stmt, (ast.ImportFrom, ast.Import)):
|
|
218
|
+
for alias in stmt.names:
|
|
219
|
+
if isinstance(stmt, ast.ImportFrom) and alias.name == "*":
|
|
220
|
+
continue
|
|
221
|
+
names.append(alias.asname or alias.name.split(".")[0])
|
|
222
|
+
return names
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def _resolve_binding(kind, target, original, parsed_by_rel, lookup, visited) -> list[str]:
|
|
226
|
+
if kind == "def":
|
|
227
|
+
ids = lookup.nodes_by_file_all.get(target, {}).get(original, [])
|
|
228
|
+
return list(ids)
|
|
229
|
+
|
|
230
|
+
key = (target, original)
|
|
231
|
+
if key in visited:
|
|
232
|
+
return []
|
|
233
|
+
visited.add(key)
|
|
234
|
+
|
|
235
|
+
submodule = lookup.module_map.get(f"{target}.{original}")
|
|
236
|
+
if submodule:
|
|
237
|
+
return [f"file::{submodule}"]
|
|
238
|
+
|
|
239
|
+
rel = lookup.module_map.get(target)
|
|
240
|
+
if not rel:
|
|
241
|
+
return []
|
|
242
|
+
|
|
243
|
+
ids = lookup.nodes_by_file_all.get(rel, {}).get(original, [])
|
|
244
|
+
if ids:
|
|
245
|
+
return list(ids)
|
|
246
|
+
|
|
247
|
+
if rel.endswith("__init__.py"):
|
|
248
|
+
pf = parsed_by_rel.get(rel)
|
|
249
|
+
if pf is not None:
|
|
250
|
+
pkg = target
|
|
251
|
+
bindings, _ = _collect_bindings(pf, pkg, lookup.module_map)
|
|
252
|
+
for b_kind, b_target, b_original in bindings.get(original, []):
|
|
253
|
+
deeper = _resolve_binding(b_kind, b_target, b_original, parsed_by_rel, lookup, visited)
|
|
254
|
+
if deeper:
|
|
255
|
+
return deeper
|
|
256
|
+
|
|
257
|
+
return [f"file::{rel}"]
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def _definition_file(node_id: str) -> str:
|
|
261
|
+
if node_id.startswith("file::"):
|
|
262
|
+
return node_id[len("file::"):]
|
|
263
|
+
return node_id.split("::")[0]
|
|
@@ -13,6 +13,7 @@ from __future__ import annotations
|
|
|
13
13
|
|
|
14
14
|
from . import edges, fingerprint, parser
|
|
15
15
|
from .contract import ExtractorError, Stats
|
|
16
|
+
from .exports_map import build_exports_map
|
|
16
17
|
from .lookup import build_lookup_maps
|
|
17
18
|
from .third_party import ThirdPartyRegistry
|
|
18
19
|
|
|
@@ -97,7 +98,9 @@ def extract(repo_path: str, options: dict) -> dict:
|
|
|
97
98
|
|
|
98
99
|
stats = Stats(totalFiles=total_files, totalNodes=len(nodes), skippedFiles=skipped)
|
|
99
100
|
|
|
100
|
-
|
|
101
|
+
exports_map = build_exports_map(repo_path, parsed_files, lookup)
|
|
102
|
+
|
|
103
|
+
result = {
|
|
101
104
|
"fingerprint": fp.to_dict(),
|
|
102
105
|
"nodes": nodes,
|
|
103
106
|
"edges": edges_out,
|
|
@@ -105,3 +108,6 @@ def extract(repo_path: str, options: dict) -> dict:
|
|
|
105
108
|
"stats": stats.to_dict(),
|
|
106
109
|
"errors": errors,
|
|
107
110
|
}
|
|
111
|
+
if exports_map is not None:
|
|
112
|
+
result["exports"] = exports_map
|
|
113
|
+
return result
|
|
@@ -20,6 +20,9 @@ class LookupMaps:
|
|
|
20
20
|
nodes_by_name: dict[str, list[tuple[str, str]]] = field(default_factory=dict)
|
|
21
21
|
# rel_path → {node name → node_id} — same-file / refinement lookups
|
|
22
22
|
nodes_by_file: dict[str, dict[str, str]] = field(default_factory=dict)
|
|
23
|
+
# rel_path → {node name → [node_id]} — ALL same-name ids in the file
|
|
24
|
+
# (overload-aware call resolution; single map above keeps first-hit)
|
|
25
|
+
nodes_by_file_all: dict[str, dict[str, list[str]]] = field(default_factory=dict)
|
|
23
26
|
# node_id → node dict — direct access to any parsed node
|
|
24
27
|
node_by_id: dict[str, dict] = field(default_factory=dict)
|
|
25
28
|
# rel_path → FILE/TEST node dict
|
|
@@ -47,11 +50,14 @@ def build_lookup_maps(parsed_files: list) -> LookupMaps:
|
|
|
47
50
|
lookup.node_by_id[pf.file_node["id"]] = pf.file_node
|
|
48
51
|
|
|
49
52
|
names: dict[str, str] = {}
|
|
53
|
+
names_all: dict[str, list[str]] = {}
|
|
50
54
|
for n in pf.nodes:
|
|
51
55
|
names.setdefault(n["name"], n["id"])
|
|
56
|
+
names_all.setdefault(n["name"], []).append(n["id"])
|
|
52
57
|
lookup.node_by_id[n["id"]] = n
|
|
53
58
|
lookup.nodes_by_name.setdefault(n["name"], []).append((n["id"], pf.rel_path))
|
|
54
59
|
lookup.nodes_by_file[pf.rel_path] = names
|
|
60
|
+
lookup.nodes_by_file_all[pf.rel_path] = names_all
|
|
55
61
|
|
|
56
62
|
return lookup
|
|
57
63
|
|
|
@@ -13,6 +13,7 @@ from pathlib import Path
|
|
|
13
13
|
from ..contract import code_hash, file_node
|
|
14
14
|
from .classes import extract_class
|
|
15
15
|
from .functions import extract_function
|
|
16
|
+
from .overloads import apply_overload_disambiguation
|
|
16
17
|
from .walker import is_test_file, walk_python_files
|
|
17
18
|
|
|
18
19
|
class ParseError(Exception):
|
|
@@ -57,6 +58,11 @@ def parse_file(repo_path: str, rel_path: str) -> ParsedFile:
|
|
|
57
58
|
end_line = len(source.splitlines()) or 1
|
|
58
59
|
file_node_ = file_node(rel_path, end_line, node_type="TEST" if is_test else "FILE")
|
|
59
60
|
|
|
61
|
+
# Same-name defs in one file (shadowing) get arity/signature id suffixes;
|
|
62
|
+
# signature-identical duplicates collapse (mirrors the JS parser).
|
|
63
|
+
apply_overload_disambiguation(children, rel_path)
|
|
64
|
+
children = [c for c in children if not c["metadata"].get("isDuplicateOverload")]
|
|
65
|
+
|
|
60
66
|
if is_test:
|
|
61
67
|
# Test files are leaf nodes in the graph — children become testCases
|
|
62
68
|
# metadata (mirrors the JS parser, which keeps test helpers out).
|