devlensio 0.6.2 → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +89 -31
- package/dist/extractors/detectLanguage.d.ts +2 -0
- package/dist/extractors/detectLanguage.js +33 -0
- package/dist/extractors/index.d.ts +6 -0
- package/dist/extractors/index.js +116 -0
- package/dist/extractors/runner.d.ts +7 -0
- package/dist/extractors/runner.js +194 -0
- package/dist/extractors/types.d.ts +37 -0
- package/dist/extractors/types.js +8 -0
- package/dist/graph/buildLookup.d.ts +1 -0
- package/dist/graph/buildLookup.js +2 -1
- package/dist/graph/edges/callEdges.js +19 -5
- package/dist/graph/edges/callEdges.test.d.ts +1 -0
- package/dist/graph/edges/callEdges.test.js +200 -0
- package/dist/graph/edges/importEdges.js +12 -0
- package/dist/graph/edges/inheritanceEdges.d.ts +3 -0
- package/dist/graph/edges/inheritanceEdges.js +73 -0
- package/dist/graph/edges/inheritanceEdges.test.d.ts +1 -0
- package/dist/graph/edges/inheritanceEdges.test.js +140 -0
- package/dist/graph/index.js +4 -0
- package/dist/parser/classes.test.d.ts +1 -0
- package/dist/parser/classes.test.js +360 -0
- package/dist/parser/extractors/classes.d.ts +5 -0
- package/dist/parser/extractors/classes.js +241 -0
- package/dist/parser/index.d.ts +2 -0
- package/dist/parser/index.js +7 -1
- package/dist/pipeline/index.d.ts +2 -1
- package/dist/pipeline/index.js +30 -42
- package/dist/scoring/index.js +8 -0
- package/dist/scoring/index.test.js +43 -0
- package/dist/scoring/nodeScorer.js +2 -0
- package/dist/scoring/pruneDisconnected.d.ts +8 -0
- package/dist/scoring/pruneDisconnected.js +66 -0
- package/dist/scoring/pruneDisconnected.test.d.ts +1 -0
- package/dist/scoring/pruneDisconnected.test.js +123 -0
- package/dist/summarizer/prompts.d.ts +1 -1
- package/dist/types.d.ts +5 -5
- package/extractors/go/bin/darwin-amd64/devlens_go_extractor +0 -0
- package/extractors/go/bin/darwin-arm64/devlens_go_extractor +0 -0
- package/extractors/go/bin/linux-amd64/devlens_go_extractor +0 -0
- package/extractors/go/bin/linux-arm64/devlens_go_extractor +0 -0
- package/extractors/go/bin/windows-amd64/devlens_go_extractor.exe +0 -0
- package/extractors/go/build.mjs +43 -0
- package/extractors/go/calls.go +289 -0
- package/extractors/go/contract.go +140 -0
- package/extractors/go/extractor.go +161 -0
- package/extractors/go/fingerprint.go +167 -0
- package/extractors/go/go.mod +3 -0
- package/extractors/go/imports.go +90 -0
- package/extractors/go/inheritance.go +138 -0
- package/extractors/go/lookup.go +106 -0
- package/extractors/go/main.go +57 -0
- package/extractors/go/nodes.go +177 -0
- package/extractors/go/orm_edges.go +277 -0
- package/extractors/go/parser.go +469 -0
- package/extractors/go/routes.go +521 -0
- package/extractors/go/tests.go +45 -0
- package/extractors/go/thirdparty.go +138 -0
- package/extractors/go/typeload.go +178 -0
- package/extractors/go/walker.go +67 -0
- package/extractors/java/build.mjs +90 -0
- package/extractors/java/devlens_java_extractor.jar +0 -0
- package/extractors/java/src/devlens/extractor/Contract.java +171 -0
- package/extractors/java/src/devlens/extractor/Extractor.java +262 -0
- package/extractors/java/src/devlens/extractor/ExtractorResult.java +12 -0
- package/extractors/java/src/devlens/extractor/Fingerprint.java +239 -0
- package/extractors/java/src/devlens/extractor/LookupMaps.java +123 -0
- package/extractors/java/src/devlens/extractor/Main.java +66 -0
- package/extractors/java/src/devlens/extractor/Parser.java +522 -0
- package/extractors/java/src/devlens/extractor/SourceWalker.java +82 -0
- package/extractors/java/src/devlens/extractor/ThirdParty.java +141 -0
- package/extractors/java/src/devlens/extractor/TypeSolverFactory.java +43 -0
- package/extractors/java/src/devlens/extractor/edges/Calls.java +266 -0
- package/extractors/java/src/devlens/extractor/edges/Enrich.java +54 -0
- package/extractors/java/src/devlens/extractor/edges/Imports.java +154 -0
- package/extractors/java/src/devlens/extractor/edges/Inheritance.java +79 -0
- package/extractors/java/src/devlens/extractor/edges/OrmEdges.java +209 -0
- package/extractors/java/src/devlens/extractor/edges/Routes.java +143 -0
- package/extractors/java/src/devlens/extractor/edges/Tests.java +53 -0
- package/extractors/python/devlens_extractors_python/__init__.py +3 -0
- package/extractors/python/devlens_extractors_python/__main__.py +33 -0
- package/extractors/python/devlens_extractors_python/contract.py +94 -0
- package/extractors/python/devlens_extractors_python/edges/__init__.py +28 -0
- package/extractors/python/devlens_extractors_python/edges/calls.py +112 -0
- package/extractors/python/devlens_extractors_python/edges/enrich.py +44 -0
- package/extractors/python/devlens_extractors_python/edges/imports.py +155 -0
- package/extractors/python/devlens_extractors_python/edges/inheritance.py +97 -0
- package/extractors/python/devlens_extractors_python/edges/orm_edges.py +225 -0
- package/extractors/python/devlens_extractors_python/edges/routes/__init__.py +28 -0
- package/extractors/python/devlens_extractors_python/edges/routes/common.py +79 -0
- package/extractors/python/devlens_extractors_python/edges/routes/decorators.py +212 -0
- package/extractors/python/devlens_extractors_python/edges/routes/django_urls.py +213 -0
- package/extractors/python/devlens_extractors_python/edges/routes/drf.py +128 -0
- package/extractors/python/devlens_extractors_python/edges/tests.py +53 -0
- package/extractors/python/devlens_extractors_python/extractor.py +107 -0
- package/extractors/python/devlens_extractors_python/fingerprint.py +201 -0
- package/extractors/python/devlens_extractors_python/lookup.py +72 -0
- package/extractors/python/devlens_extractors_python/parser/__init__.py +72 -0
- package/extractors/python/devlens_extractors_python/parser/classes.py +109 -0
- package/extractors/python/devlens_extractors_python/parser/functions.py +163 -0
- package/extractors/python/devlens_extractors_python/parser/walker.py +30 -0
- package/extractors/python/devlens_extractors_python/third_party.py +103 -0
- package/extractors/python/pyproject.toml +16 -0
- package/extractors/python/setup.mjs +54 -0
- package/extractors/rust/Cargo.toml +29 -0
- package/extractors/rust/bin/darwin-amd64/devlens_rust_extractor +0 -0
- package/extractors/rust/bin/darwin-arm64/devlens_rust_extractor +0 -0
- package/extractors/rust/bin/linux-amd64/devlens_rust_extractor +0 -0
- package/extractors/rust/bin/linux-arm64/devlens_rust_extractor +0 -0
- package/extractors/rust/bin/windows-amd64/devlens_rust_extractor.exe +0 -0
- package/extractors/rust/build.mjs +85 -0
- package/extractors/rust/src/calls.rs +295 -0
- package/extractors/rust/src/contract.rs +236 -0
- package/extractors/rust/src/enrich.rs +107 -0
- package/extractors/rust/src/extractor.rs +199 -0
- package/extractors/rust/src/fingerprint.rs +238 -0
- package/extractors/rust/src/imports.rs +101 -0
- package/extractors/rust/src/inheritance.rs +233 -0
- package/extractors/rust/src/lookup.rs +226 -0
- package/extractors/rust/src/main.rs +53 -0
- package/extractors/rust/src/module_map.rs +131 -0
- package/extractors/rust/src/nodes.rs +174 -0
- package/extractors/rust/src/orm_edges.rs +125 -0
- package/extractors/rust/src/parser.rs +859 -0
- package/extractors/rust/src/routes.rs +1265 -0
- package/extractors/rust/src/tests.rs +109 -0
- package/extractors/rust/src/thirdparty.rs +114 -0
- package/extractors/rust/src/walker.rs +82 -0
- package/package.json +22 -4
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
"""TESTS edges — test file → the production code it imports.
|
|
2
|
+
|
|
3
|
+
Mirror of the JS detectTestEdges (src/graph/edges/testEdges.ts):
|
|
4
|
+
- every TEST file node is scanned via its symbol map (import aliases)
|
|
5
|
+
- a local named import pointing at a FILE is refined to the symbol with
|
|
6
|
+
that name inside it (the calls.py refinement)
|
|
7
|
+
- third-party targets and test-importing-test targets are skipped
|
|
8
|
+
- edge: TEST file node → TESTS → production node (metadata.importPath)
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from ..lookup import LookupMaps, module_name
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def resolve_test_edges(lookup: LookupMaps) -> list[dict]:
|
|
16
|
+
edges: list[dict] = []
|
|
17
|
+
seen: set[tuple[str, str]] = set()
|
|
18
|
+
|
|
19
|
+
for rel, file_node in lookup.file_nodes_by_path.items():
|
|
20
|
+
if file_node["type"] != "TEST":
|
|
21
|
+
continue
|
|
22
|
+
symbols = lookup.symbol_maps.get(rel, {})
|
|
23
|
+
|
|
24
|
+
for alias, target in symbols.items():
|
|
25
|
+
# refine file targets to the symbol inside (calls.py refinement)
|
|
26
|
+
if target.startswith("file::"):
|
|
27
|
+
target_rel = target[len("file::"):]
|
|
28
|
+
target_id = lookup.nodes_by_file.get(target_rel, {}).get(alias)
|
|
29
|
+
if not target_id:
|
|
30
|
+
continue
|
|
31
|
+
elif target.startswith("[pip]/"):
|
|
32
|
+
continue # third-party — not production code under test
|
|
33
|
+
else:
|
|
34
|
+
target_id = target
|
|
35
|
+
|
|
36
|
+
target_node = lookup.node_by_id.get(target_id)
|
|
37
|
+
if not target_node or target_node["type"] in ("TEST", "FILE"):
|
|
38
|
+
continue # test importing test helpers / plain files
|
|
39
|
+
if target_node["filePath"] == rel:
|
|
40
|
+
continue # self-import
|
|
41
|
+
|
|
42
|
+
key = (file_node["id"], target_id)
|
|
43
|
+
if key in seen:
|
|
44
|
+
continue
|
|
45
|
+
seen.add(key)
|
|
46
|
+
|
|
47
|
+
edges.append({
|
|
48
|
+
"from": file_node["id"], "to": target_id, "type": "TESTS",
|
|
49
|
+
"metadata": {"importPath": module_name(target_node["filePath"]),
|
|
50
|
+
"testFileType": "TEST"},
|
|
51
|
+
})
|
|
52
|
+
|
|
53
|
+
return edges
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
"""Orchestration — walks the repo, parses files, resolves imports,
|
|
2
|
+
detects calls/routes/orm, and assembles the final ExtractorResult.
|
|
3
|
+
|
|
4
|
+
Pipeline:
|
|
5
|
+
1. walk + parse every .py file → nodes
|
|
6
|
+
2. build shared lookup maps (once) + resolve imports → IMPORTS edges + symbol maps
|
|
7
|
+
3. resolve calls → CALLS edges (lazily creating [pip]/pkg::method nodes)
|
|
8
|
+
4. routes → BACKEND_ROUTEs + ROUTE nodes + HANDLES edges
|
|
9
|
+
5. ORM data layer → model metadata + READS_FROM/WRITES_TO edges
|
|
10
|
+
6. collect third-party nodes (AFTER call resolution — lazy nodes)
|
|
11
|
+
"""
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from . import edges, fingerprint, parser
|
|
15
|
+
from .contract import ExtractorError, Stats
|
|
16
|
+
from .lookup import build_lookup_maps
|
|
17
|
+
from .third_party import ThirdPartyRegistry
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def extract(repo_path: str, options: dict) -> dict:
|
|
21
|
+
fp = fingerprint.detect(repo_path)
|
|
22
|
+
|
|
23
|
+
nodes: list[dict] = []
|
|
24
|
+
edges_out: list[dict] = []
|
|
25
|
+
errors: list[dict] = []
|
|
26
|
+
total_files = 0
|
|
27
|
+
skipped = 0
|
|
28
|
+
|
|
29
|
+
# 1. walk + parse every file (test files stay leaf nodes)
|
|
30
|
+
parsed_files: list[parser.ParsedFile] = []
|
|
31
|
+
rel_paths: list[str] = []
|
|
32
|
+
for rel in parser.walk_python_files(repo_path):
|
|
33
|
+
total_files += 1
|
|
34
|
+
rel_paths.append(rel)
|
|
35
|
+
try:
|
|
36
|
+
parsed = parser.parse_file(repo_path, rel)
|
|
37
|
+
except parser.ParseError as exc:
|
|
38
|
+
skipped += 1
|
|
39
|
+
errors.append(ExtractorError(file=rel, error=str(exc)).to_dict())
|
|
40
|
+
continue
|
|
41
|
+
parsed_files.append(parsed)
|
|
42
|
+
nodes.append(parsed.file_node)
|
|
43
|
+
nodes.extend(parsed.nodes)
|
|
44
|
+
|
|
45
|
+
# 2. shared lookup — built ONCE, consumed by every edge detector
|
|
46
|
+
lookup = build_lookup_maps(parsed_files)
|
|
47
|
+
lookup.module_map = edges.imports.build_module_map(rel_paths)
|
|
48
|
+
|
|
49
|
+
# third-party inclusion gate: only libs listed in options.includeThirdPartyLibs
|
|
50
|
+
# (engine contract name) / includedThirdPartyLibs (contract.html alias) get
|
|
51
|
+
# [pip]/ nodes + edges. Absent/empty → no third-party nodes at all.
|
|
52
|
+
libs = options.get("includeThirdPartyLibs") or options.get("includedThirdPartyLibs") or []
|
|
53
|
+
allowed = set(libs)
|
|
54
|
+
registry = ThirdPartyRegistry(fp.rawDependencies, allowed)
|
|
55
|
+
|
|
56
|
+
for parsed in parsed_files:
|
|
57
|
+
file_edges, imported = edges.resolve_file_imports(parsed, lookup, registry)
|
|
58
|
+
edges_out.extend(file_edges)
|
|
59
|
+
for n in parsed.nodes:
|
|
60
|
+
n["metadata"]["imports"] = imported
|
|
61
|
+
|
|
62
|
+
# 3. CALLS edges — same lookup, zero rebuilds
|
|
63
|
+
edges_out.extend(edges.resolve_calls(lookup, registry))
|
|
64
|
+
|
|
65
|
+
# 4. routes — BackendRouteNodes + ROUTE nodes + HANDLES edges
|
|
66
|
+
route_result = edges.detect_routes(parsed_files, lookup, fp)
|
|
67
|
+
edges_out.extend(route_result.handles_edges)
|
|
68
|
+
nodes.extend(route_result.route_nodes)
|
|
69
|
+
|
|
70
|
+
# 5. ORM data layer — model metadata + READS_FROM/WRITES_TO
|
|
71
|
+
models = edges.detect_models(lookup, parsed_files)
|
|
72
|
+
edges_out.extend(edges.resolve_orm_edges(lookup, models))
|
|
73
|
+
|
|
74
|
+
# 5c. inheritance — EXTENDS / IMPLEMENTS edges (class → base)
|
|
75
|
+
edges_out.extend(edges.resolve_inheritance(lookup, registry))
|
|
76
|
+
|
|
77
|
+
# 5d. TESTS edges — test file → production symbols it imports
|
|
78
|
+
edges_out.extend(edges.resolve_test_edges(lookup))
|
|
79
|
+
|
|
80
|
+
# 5b. metadata enrichment — pydantic isSchema, celery isTask
|
|
81
|
+
edges.enrich_metadata(lookup)
|
|
82
|
+
|
|
83
|
+
# 6. third-party nodes — AFTER call resolution (lazy method nodes)
|
|
84
|
+
nodes.extend(registry.nodes)
|
|
85
|
+
|
|
86
|
+
# 7. dedupe edges — same (from, to, type) can arise from repeated ormOps
|
|
87
|
+
# or overlapping patterns; the graph must not carry duplicates.
|
|
88
|
+
seen_edges: set[tuple[str, str, str]] = set()
|
|
89
|
+
unique_edges: list[dict] = []
|
|
90
|
+
for e in edges_out:
|
|
91
|
+
key = (e["from"], e["to"], e["type"])
|
|
92
|
+
if key in seen_edges:
|
|
93
|
+
continue
|
|
94
|
+
seen_edges.add(key)
|
|
95
|
+
unique_edges.append(e)
|
|
96
|
+
edges_out = unique_edges
|
|
97
|
+
|
|
98
|
+
stats = Stats(totalFiles=total_files, totalNodes=len(nodes), skippedFiles=skipped)
|
|
99
|
+
|
|
100
|
+
return {
|
|
101
|
+
"fingerprint": fp.to_dict(),
|
|
102
|
+
"nodes": nodes,
|
|
103
|
+
"edges": edges_out,
|
|
104
|
+
"routes": route_result.routes,
|
|
105
|
+
"stats": stats.to_dict(),
|
|
106
|
+
"errors": errors,
|
|
107
|
+
}
|
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
"""Fingerprint detection — reads dependency manifests and answers:
|
|
2
|
+
what framework, what databases, what raw dependencies?
|
|
3
|
+
|
|
4
|
+
Manifest priority when several exist: setup.py → pyproject.toml → requirements.txt
|
|
5
|
+
(later sources override earlier ones). requirements.txt is usually the
|
|
6
|
+
deployed truth, so it wins.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
import ast
|
|
11
|
+
import re
|
|
12
|
+
import tomllib
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from .contract import Fingerprint
|
|
15
|
+
|
|
16
|
+
#Framework detection order - first match wins
|
|
17
|
+
FRAMEWORK_ORDER = ["django", "fastapi", "flask"]
|
|
18
|
+
|
|
19
|
+
# Dependency package name → DatabaseLibrary value.
|
|
20
|
+
# ONLY union-safe values from src/types.ts (postgres, mongodb, mysql, sqlite).
|
|
21
|
+
# sqlalchemy/redis intentionally absent — they are ORM/cache layers, not a
|
|
22
|
+
# specific DB, and are not in the TS union. They still appear in rawDependencies.
|
|
23
|
+
DATABASE_PACKAGES = {
|
|
24
|
+
# postgres
|
|
25
|
+
"psycopg2": "postgres", "psycopg2-binary": "postgres",
|
|
26
|
+
"psycopg": "postgres", "asyncpg": "postgres", "pg8000": "postgres",
|
|
27
|
+
# mongodb
|
|
28
|
+
"pymongo": "mongodb", "motor": "mongodb",
|
|
29
|
+
# mysql
|
|
30
|
+
"pymysql": "mysql", "mysqlclient": "mysql",
|
|
31
|
+
# sqlite
|
|
32
|
+
"aiosqlite": "sqlite",
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
_VERSION_RE = re.compile(r"(\d+\.\d+(?:\.\d+)?(?:[a-z0-9.]*)?)")
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _extract_version(spec: str) -> str:
|
|
39
|
+
"""Best-effort version from any specifier: '==0.104.1', '>=1.0,<2.0',
|
|
40
|
+
'^1.0', '~=4.2' → first version-like token, else 'unknown'."""
|
|
41
|
+
match = _VERSION_RE.search(spec)
|
|
42
|
+
return match.group(1) if match else "unknown"
|
|
43
|
+
|
|
44
|
+
def _parse_pep508(dep: str) -> tuple[str, str]:
|
|
45
|
+
"""Parse one PEP 508 dependency string → (name, version).
|
|
46
|
+
|
|
47
|
+
Handles: 'fastapi==0.104.1', 'requests[security]>=2.31',
|
|
48
|
+
'pkg @ https://...', 'django~=4.2'.
|
|
49
|
+
"""
|
|
50
|
+
name = re.split(r"[<=>~!@\[;]", dep, maxsplit=1)[0].strip().lower()
|
|
51
|
+
return name, _extract_version(dep)
|
|
52
|
+
|
|
53
|
+
def _parse_requirements(path: Path, seen: set | None = None) -> dict[str, str]:
|
|
54
|
+
"""requirements.txt: one dep per line. Follows -r/--requirement and
|
|
55
|
+
-c/--constraint include files RECURSIVELY (paths relative to the
|
|
56
|
+
including file), e.g. `-r requirements/base.txt` — real projects
|
|
57
|
+
split requirements this way. Editable (-e) and other directives skip."""
|
|
58
|
+
deps: dict[str, str] = {}
|
|
59
|
+
seen = seen or set()
|
|
60
|
+
try:
|
|
61
|
+
text = path.read_text(encoding="utf-8", errors="replace")
|
|
62
|
+
except OSError:
|
|
63
|
+
return deps
|
|
64
|
+
|
|
65
|
+
for raw in text.splitlines():
|
|
66
|
+
line = raw.strip()
|
|
67
|
+
if not line or line.startswith("#"):
|
|
68
|
+
continue
|
|
69
|
+
|
|
70
|
+
# include directive: -r/-c with space or = form
|
|
71
|
+
inc = re.match(r"^(?:--requirement|--constraint|-r|-c)(?:=|\s+)(\S+)", line)
|
|
72
|
+
if inc:
|
|
73
|
+
inc_rel = re.split(r"\s+#", inc.group(1))[0].strip()
|
|
74
|
+
inc_path = (path.parent / inc_rel).resolve()
|
|
75
|
+
if inc_path.is_file() and inc_path not in seen:
|
|
76
|
+
seen.add(inc_path)
|
|
77
|
+
deps.update(_parse_requirements(inc_path, seen))
|
|
78
|
+
continue
|
|
79
|
+
|
|
80
|
+
if line.startswith("-"):
|
|
81
|
+
continue # -e editable, --index-url, --extra-index-url, etc.
|
|
82
|
+
|
|
83
|
+
line = re.split(r"\s+#", line)[0] # inline comment (must follow whitespace)
|
|
84
|
+
line = line.split(";")[0].strip() # env marker: django==4.2 ; python_version >= "3.8"
|
|
85
|
+
if not line:
|
|
86
|
+
continue
|
|
87
|
+
name, version = _parse_pep508(line)
|
|
88
|
+
deps[name] = version
|
|
89
|
+
return deps
|
|
90
|
+
|
|
91
|
+
def _parse_poetry_deps(data: dict) -> dict[str, str]:
|
|
92
|
+
"""Poetry-style deps under [tool.poetry.dependencies].
|
|
93
|
+
Values can be a version string, a dict {version, extras}, or a list of strings."""
|
|
94
|
+
deps: dict[str, str] = {}
|
|
95
|
+
poetry = data.get("tool", {}).get("poetry", {}).get("dependencies", {})
|
|
96
|
+
for name, spec in poetry.items():
|
|
97
|
+
if name == "python": # poetry requires a python constraint — not a dep
|
|
98
|
+
continue
|
|
99
|
+
if isinstance(spec, str):
|
|
100
|
+
deps[name] = _extract_version(spec)
|
|
101
|
+
elif isinstance(spec, dict) and isinstance(spec.get("version"), str):
|
|
102
|
+
deps[name] = _extract_version(spec["version"])
|
|
103
|
+
elif isinstance(spec, list):
|
|
104
|
+
deps[name] = _extract_version(spec[0]) if spec else "unknown"
|
|
105
|
+
return deps
|
|
106
|
+
|
|
107
|
+
def _parse_pyproject(path: Path) -> dict[str, str]:
|
|
108
|
+
"""pyproject.toml: PEP 621 [project] dependencies + optional-dependencies,
|
|
109
|
+
plus Poetry's [tool.poetry.dependencies]."""
|
|
110
|
+
try:
|
|
111
|
+
with open(path, "rb") as f: # tomllib requires binary mode
|
|
112
|
+
data = tomllib.load(f)
|
|
113
|
+
except (tomllib.TOMLDecodeError, OSError):
|
|
114
|
+
return {}
|
|
115
|
+
|
|
116
|
+
deps: dict[str, str] = {}
|
|
117
|
+
project = data.get("project", {}) or {}
|
|
118
|
+
|
|
119
|
+
for dep in project.get("dependencies", []) or []:
|
|
120
|
+
name, version = _parse_pep508(dep)
|
|
121
|
+
deps[name] = version
|
|
122
|
+
|
|
123
|
+
for group in (project.get("optional-dependencies", {}) or {}).values():
|
|
124
|
+
for dep in group or []:
|
|
125
|
+
name, version = _parse_pep508(dep)
|
|
126
|
+
deps[name] = version
|
|
127
|
+
|
|
128
|
+
deps.update(_parse_poetry_deps(data))
|
|
129
|
+
return deps
|
|
130
|
+
|
|
131
|
+
def _parse_setup_py(path: Path) -> dict[str, str]:
|
|
132
|
+
"""setup.py: parse (never execute!) and read install_requires/extras_require
|
|
133
|
+
string literals out of the AST. Executing repo code during analysis would
|
|
134
|
+
be a security hole."""
|
|
135
|
+
try:
|
|
136
|
+
tree = ast.parse(path.read_text(encoding="utf-8", errors="replace"))
|
|
137
|
+
except (SyntaxError, OSError):
|
|
138
|
+
return {}
|
|
139
|
+
|
|
140
|
+
deps: dict[str, str] = {}
|
|
141
|
+
|
|
142
|
+
for node in ast.walk(tree):
|
|
143
|
+
if not (isinstance(node, ast.Call) and isinstance(node.func, ast.Name)
|
|
144
|
+
and node.func.id == "setup"):
|
|
145
|
+
continue
|
|
146
|
+
for kw in node.keywords:
|
|
147
|
+
if kw.arg == "install_requires" and isinstance(kw.value, ast.List):
|
|
148
|
+
for elt in kw.value.elts:
|
|
149
|
+
if isinstance(elt, ast.Constant) and isinstance(elt.value, str):
|
|
150
|
+
name, version = _parse_pep508(elt.value)
|
|
151
|
+
deps[name] = version
|
|
152
|
+
elif kw.arg == "extras_require" and isinstance(kw.value, ast.Dict):
|
|
153
|
+
for value in kw.value.values:
|
|
154
|
+
if isinstance(value, ast.List):
|
|
155
|
+
for elt in value.elts:
|
|
156
|
+
if isinstance(elt, ast.Constant) and isinstance(elt.value, str):
|
|
157
|
+
name, version = _parse_pep508(elt.value)
|
|
158
|
+
deps[name] = version
|
|
159
|
+
return deps
|
|
160
|
+
|
|
161
|
+
def _collect_dependencies(repo_path: str) -> dict[str, str]:
|
|
162
|
+
"""Merge manifests. Later sources override earlier ones."""
|
|
163
|
+
root = Path(repo_path)
|
|
164
|
+
deps: dict[str, str] = {}
|
|
165
|
+
|
|
166
|
+
setup = root / "setup.py"
|
|
167
|
+
if setup.is_file():
|
|
168
|
+
deps.update(_parse_setup_py(setup))
|
|
169
|
+
|
|
170
|
+
pyproject = root / "pyproject.toml"
|
|
171
|
+
if pyproject.is_file():
|
|
172
|
+
deps.update(_parse_pyproject(pyproject))
|
|
173
|
+
|
|
174
|
+
requirements = root / "requirements.txt"
|
|
175
|
+
if requirements.is_file():
|
|
176
|
+
deps.update(_parse_requirements(requirements))
|
|
177
|
+
|
|
178
|
+
return deps
|
|
179
|
+
|
|
180
|
+
def detect(repo_path: str) -> Fingerprint:
|
|
181
|
+
"""Build the fingerprint from dependency manifests. Missing manifests →
|
|
182
|
+
unknown framework, empty deps. Never raises."""
|
|
183
|
+
deps = _collect_dependencies(repo_path)
|
|
184
|
+
|
|
185
|
+
fingerprint = Fingerprint()
|
|
186
|
+
fingerprint.rawDependencies = dict(sorted(deps.items()))
|
|
187
|
+
|
|
188
|
+
framework = next((fw for fw in FRAMEWORK_ORDER if fw in deps), None)
|
|
189
|
+
# DRF implies Django even if 'django' isn't declared directly
|
|
190
|
+
if framework is None and "djangorestframework" in deps:
|
|
191
|
+
framework = "django"
|
|
192
|
+
|
|
193
|
+
if framework:
|
|
194
|
+
fingerprint.framework = framework
|
|
195
|
+
fingerprint.projectType = "backend"
|
|
196
|
+
|
|
197
|
+
for pkg, db in DATABASE_PACKAGES.items():
|
|
198
|
+
if pkg in deps and db not in fingerprint.databases:
|
|
199
|
+
fingerprint.databases.append(db)
|
|
200
|
+
|
|
201
|
+
return fingerprint
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
"""Shared lookup maps — the analysis index.
|
|
2
|
+
|
|
3
|
+
Mirrors the JS buildLookupMaps() (src/graph/buildLookup.ts): built ONCE per
|
|
4
|
+
extraction and passed to every edge detector (calls, routes, orm, tests) so
|
|
5
|
+
nobody rebuilds indexes. All lookups are O(1) map hits.
|
|
6
|
+
|
|
7
|
+
Field owners:
|
|
8
|
+
nodes_by_name / nodes_by_file / node_by_id / file_nodes_by_path — build_lookup_maps()
|
|
9
|
+
module_map — extractor (imports.build_module_map)
|
|
10
|
+
symbol_maps — edges.imports.resolve_file_imports
|
|
11
|
+
"""
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from dataclasses import dataclass, field
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@dataclass
|
|
18
|
+
class LookupMaps:
|
|
19
|
+
# name → [(node_id, rel_path)] — global, closest-by-path on collision
|
|
20
|
+
nodes_by_name: dict[str, list[tuple[str, str]]] = field(default_factory=dict)
|
|
21
|
+
# rel_path → {node name → node_id} — same-file / refinement lookups
|
|
22
|
+
nodes_by_file: dict[str, dict[str, str]] = field(default_factory=dict)
|
|
23
|
+
# node_id → node dict — direct access to any parsed node
|
|
24
|
+
node_by_id: dict[str, dict] = field(default_factory=dict)
|
|
25
|
+
# rel_path → FILE/TEST node dict
|
|
26
|
+
file_nodes_by_path: dict[str, dict] = field(default_factory=dict)
|
|
27
|
+
# rel_path → {local alias → target id} — import precision (edges.imports writes)
|
|
28
|
+
symbol_maps: dict[str, dict[str, str]] = field(default_factory=dict)
|
|
29
|
+
# dotted module name → rel path (imports.build_module_map builds)
|
|
30
|
+
module_map: dict[str, str] = field(default_factory=dict)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def module_name(rel: str) -> str:
|
|
34
|
+
"""'models/user.py' → 'models.user' · 'models/__init__.py' → 'models'."""
|
|
35
|
+
parts = rel.split("/")
|
|
36
|
+
if parts[-1] == "__init__.py":
|
|
37
|
+
return ".".join(parts[:-1])
|
|
38
|
+
return ".".join(parts[:-1] + [parts[-1][:-3]])
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def build_lookup_maps(parsed_files: list) -> LookupMaps:
|
|
42
|
+
"""One pass over all parsed files → shared indexes."""
|
|
43
|
+
lookup = LookupMaps()
|
|
44
|
+
|
|
45
|
+
for pf in parsed_files:
|
|
46
|
+
lookup.file_nodes_by_path[pf.rel_path] = pf.file_node
|
|
47
|
+
lookup.node_by_id[pf.file_node["id"]] = pf.file_node
|
|
48
|
+
|
|
49
|
+
names: dict[str, str] = {}
|
|
50
|
+
for n in pf.nodes:
|
|
51
|
+
names.setdefault(n["name"], n["id"])
|
|
52
|
+
lookup.node_by_id[n["id"]] = n
|
|
53
|
+
lookup.nodes_by_name.setdefault(n["name"], []).append((n["id"], pf.rel_path))
|
|
54
|
+
lookup.nodes_by_file[pf.rel_path] = names
|
|
55
|
+
|
|
56
|
+
return lookup
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def closest_by_path(candidates: list[tuple[str, str]], rel: str) -> str:
|
|
60
|
+
"""Pick the candidate whose file shares the most leading path segments
|
|
61
|
+
with the caller (same heuristic as the JS closestByPath)."""
|
|
62
|
+
caller_parts = rel.split("/")
|
|
63
|
+
best_id, best_score = candidates[0][0], -1
|
|
64
|
+
for node_id, cand_rel in candidates:
|
|
65
|
+
score = 0
|
|
66
|
+
for a, b in zip(caller_parts, cand_rel.split("/")):
|
|
67
|
+
if a != b:
|
|
68
|
+
break
|
|
69
|
+
score += 1
|
|
70
|
+
if score > best_score:
|
|
71
|
+
best_id, best_score = node_id, score
|
|
72
|
+
return best_id
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
"""Parser package — walks the repo and turns .py files into DevLens nodes.
|
|
2
|
+
|
|
3
|
+
Public API (everything else is internal):
|
|
4
|
+
parse_file(repo_path, rel_path) -> ParsedFile
|
|
5
|
+
walk_python_files(repo_path) -> iterator of relative paths
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import ast
|
|
10
|
+
from dataclasses import dataclass
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
from ..contract import code_hash, file_node
|
|
14
|
+
from .classes import extract_class
|
|
15
|
+
from .functions import extract_function
|
|
16
|
+
from .walker import is_test_file, walk_python_files
|
|
17
|
+
|
|
18
|
+
class ParseError(Exception):
|
|
19
|
+
"""Raised when a file can't be read or parsed. Non-fatal — the caller
|
|
20
|
+
records it in errors[] and continues with the next file."""
|
|
21
|
+
|
|
22
|
+
@dataclass
|
|
23
|
+
class ParsedFile:
|
|
24
|
+
rel_path: str
|
|
25
|
+
file_node: dict
|
|
26
|
+
nodes: list[dict] # FUNCTION / CLASS / METHOD nodes (children)
|
|
27
|
+
source: str
|
|
28
|
+
tree: ast.Module
|
|
29
|
+
|
|
30
|
+
def parse_file(repo_path: str, rel_path: str) -> ParsedFile:
|
|
31
|
+
abs_path = Path(repo_path) / rel_path
|
|
32
|
+
try:
|
|
33
|
+
source = abs_path.read_text(encoding="utf-8", errors="replace")
|
|
34
|
+
except OSError as exc:
|
|
35
|
+
raise ParseError(f"unreadable file: {exc}") from exc
|
|
36
|
+
|
|
37
|
+
try:
|
|
38
|
+
tree = ast.parse(source, filename=rel_path)
|
|
39
|
+
except SyntaxError as exc:
|
|
40
|
+
line = exc.lineno or "?"
|
|
41
|
+
raise ParseError(f"SyntaxError at line {line}: {exc.msg}") from exc
|
|
42
|
+
|
|
43
|
+
is_test = is_test_file(rel_path)
|
|
44
|
+
children: list[dict] = []
|
|
45
|
+
test_cases: list[str] = []
|
|
46
|
+
|
|
47
|
+
for stmt in tree.body:
|
|
48
|
+
if isinstance(stmt, (ast.FunctionDef, ast.AsyncFunctionDef)):
|
|
49
|
+
children.append(extract_function(stmt, rel_path, source))
|
|
50
|
+
if is_test:
|
|
51
|
+
test_cases.append(stmt.name)
|
|
52
|
+
elif isinstance(stmt, ast.ClassDef):
|
|
53
|
+
children.extend(extract_class(stmt, rel_path, source))
|
|
54
|
+
if is_test:
|
|
55
|
+
test_cases.append(stmt.name)
|
|
56
|
+
|
|
57
|
+
end_line = len(source.splitlines()) or 1
|
|
58
|
+
file_node_ = file_node(rel_path, end_line, node_type="TEST" if is_test else "FILE")
|
|
59
|
+
|
|
60
|
+
if is_test:
|
|
61
|
+
# Test files are leaf nodes in the graph — children become testCases
|
|
62
|
+
# metadata (mirrors the JS parser, which keeps test helpers out).
|
|
63
|
+
file_node_["metadata"]["testCases"] = test_cases
|
|
64
|
+
else:
|
|
65
|
+
file_node_["metadata"]["nodeCount"] = len(children)
|
|
66
|
+
file_node_["metadata"]["childNodeIds"] = [c["id"] for c in children]
|
|
67
|
+
joined = "\n".join(c.get("rawCode", "") for c in children)
|
|
68
|
+
if joined.strip():
|
|
69
|
+
file_node_["codeHash"] = code_hash(joined)
|
|
70
|
+
|
|
71
|
+
return ParsedFile(rel_path=rel_path, file_node=file_node_, nodes=children if not is_test else [],
|
|
72
|
+
source=source, tree=tree)
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
"""Class extraction — CLASS node + METHOD children + model/schema metadata.
|
|
2
|
+
|
|
3
|
+
Collects at parse time (resolved later by edges/orm_edges.py):
|
|
4
|
+
fields[] — Django field / SQLAlchemy Column assignments
|
|
5
|
+
linkedModel — DRF serializer `class Meta: model = User`
|
|
6
|
+
ormOps — ORM calls in class-level statements (e.g. `queryset = User.objects.all()`)
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import ast
|
|
11
|
+
|
|
12
|
+
from ..contract import code_node
|
|
13
|
+
from .functions import NESTED_SCOPES, _match_orm_call, _walk_scope, extract_function
|
|
14
|
+
|
|
15
|
+
FIELD_TYPE_NAMES = {
|
|
16
|
+
# Django ORM fields
|
|
17
|
+
"CharField", "TextField", "IntegerField", "BigIntegerField",
|
|
18
|
+
"PositiveIntegerField", "PositiveBigIntegerField", "SmallIntegerField",
|
|
19
|
+
"FloatField", "DecimalField", "BooleanField", "DateField", "DateTimeField",
|
|
20
|
+
"TimeField", "DurationField", "EmailField", "URLField", "UUIDField",
|
|
21
|
+
"JSONField", "SlugField", "BinaryField", "FileField", "ImageField",
|
|
22
|
+
"FilePathField", "AutoField", "BigAutoField", "ForeignKey",
|
|
23
|
+
"OneToOneField", "ManyToManyField", "GenericForeignKey", "ArrayField",
|
|
24
|
+
# SQLAlchemy
|
|
25
|
+
"Column", "relationship", "mapped_column",
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _extract_fields(node: ast.ClassDef) -> list[str]:
|
|
30
|
+
"""Field names from class-body assignments: `name = CharField(...)`."""
|
|
31
|
+
fields: list[str] = []
|
|
32
|
+
for stmt in node.body:
|
|
33
|
+
if not isinstance(stmt, ast.Assign) or not isinstance(stmt.value, ast.Call):
|
|
34
|
+
continue
|
|
35
|
+
f = stmt.value.func
|
|
36
|
+
fname = f.id if isinstance(f, ast.Name) else (
|
|
37
|
+
f.attr if isinstance(f, ast.Attribute) else None)
|
|
38
|
+
if fname not in FIELD_TYPE_NAMES:
|
|
39
|
+
continue
|
|
40
|
+
for t in stmt.targets:
|
|
41
|
+
if isinstance(t, ast.Name):
|
|
42
|
+
fields.append(t.id)
|
|
43
|
+
return fields
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _extract_linked_model(node: ast.ClassDef) -> str | None:
|
|
47
|
+
"""DRF serializer: inner `class Meta: model = User` → 'User'."""
|
|
48
|
+
for stmt in node.body:
|
|
49
|
+
if isinstance(stmt, ast.ClassDef) and stmt.name == "Meta":
|
|
50
|
+
for inner in stmt.body:
|
|
51
|
+
if isinstance(inner, ast.Assign) and isinstance(inner.value, ast.Name):
|
|
52
|
+
for t in inner.targets:
|
|
53
|
+
if isinstance(t, ast.Name) and t.id == "model":
|
|
54
|
+
return inner.value.id
|
|
55
|
+
return None
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _extract_class_orm_ops(node: ast.ClassDef) -> list[dict]:
|
|
59
|
+
"""ORM calls in class-level statements (queryset = Model.objects.all())."""
|
|
60
|
+
stmts = [s for s in node.body
|
|
61
|
+
if not isinstance(s, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef))]
|
|
62
|
+
ops: list[dict] = []
|
|
63
|
+
|
|
64
|
+
def visit(n: ast.AST) -> None:
|
|
65
|
+
if isinstance(n, ast.Call):
|
|
66
|
+
op = _match_orm_call(n)
|
|
67
|
+
if op:
|
|
68
|
+
ops.append(op)
|
|
69
|
+
|
|
70
|
+
_walk_scope(stmts, visit)
|
|
71
|
+
return ops
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def extract_class(node: ast.ClassDef, rel_path: str, source: str,
|
|
75
|
+
prefix: str = "") -> list[dict]:
|
|
76
|
+
"""Returns [CLASS node, *METHOD nodes]. Nested classes get dotted names
|
|
77
|
+
(Outer.Inner) and their methods (Outer.Inner.method) — same convention
|
|
78
|
+
as methods."""
|
|
79
|
+
class_name = f"{prefix}.{node.name}" if prefix else node.name
|
|
80
|
+
raw = ast.get_source_segment(source, node) or ""
|
|
81
|
+
|
|
82
|
+
children: list[dict] = []
|
|
83
|
+
for stmt in node.body:
|
|
84
|
+
if isinstance(stmt, (ast.FunctionDef, ast.AsyncFunctionDef)):
|
|
85
|
+
children.append(extract_function(stmt, rel_path, source, parent_class=class_name))
|
|
86
|
+
elif isinstance(stmt, ast.ClassDef):
|
|
87
|
+
children.extend(extract_class(stmt, rel_path, source, prefix=class_name))
|
|
88
|
+
|
|
89
|
+
metadata = {
|
|
90
|
+
"bases": [ast.unparse(b) for b in node.bases],
|
|
91
|
+
"decorators": [ast.unparse(d) for d in node.decorator_list],
|
|
92
|
+
"methods": [c["name"] for c in children if c["type"] == "METHOD"],
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
fields = _extract_fields(node)
|
|
96
|
+
if fields:
|
|
97
|
+
metadata["fields"] = fields
|
|
98
|
+
|
|
99
|
+
linked = _extract_linked_model(node)
|
|
100
|
+
if linked:
|
|
101
|
+
metadata["linkedModel"] = linked
|
|
102
|
+
|
|
103
|
+
class_ops = _extract_class_orm_ops(node)
|
|
104
|
+
if class_ops:
|
|
105
|
+
metadata["ormOps"] = class_ops
|
|
106
|
+
|
|
107
|
+
class_node = code_node(rel_path, class_name, "CLASS", node.lineno,
|
|
108
|
+
node.end_lineno or node.lineno, raw, metadata)
|
|
109
|
+
return [class_node, *children]
|