ai-code-engineer 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ai_code_engineer/__init__.py +2 -0
- ai_code_engineer/catalog.py +143 -0
- ai_code_engineer/chat.py +181 -0
- ai_code_engineer/cli.py +384 -0
- ai_code_engineer/config.py +405 -0
- ai_code_engineer/engine.py +1282 -0
- ai_code_engineer/errors.py +27 -0
- ai_code_engineer/git_integration.py +443 -0
- ai_code_engineer/gui.py +2646 -0
- ai_code_engineer/host.py +81 -0
- ai_code_engineer/ignore.py +269 -0
- ai_code_engineer/intent.py +222 -0
- ai_code_engineer/labels.py +871 -0
- ai_code_engineer/memory.py +91 -0
- ai_code_engineer/modes.py +156 -0
- ai_code_engineer/overrides.py +540 -0
- ai_code_engineer/planbook.py +192 -0
- ai_code_engineer/providers.py +404 -0
- ai_code_engineer/redaction.py +54 -0
- ai_code_engineer/repair.py +564 -0
- ai_code_engineer/report.py +352 -0
- ai_code_engineer/runner.py +854 -0
- ai_code_engineer/setup.py +386 -0
- ai_code_engineer/symbols.py +1286 -0
- ai_code_engineer/verification.py +218 -0
- ai_code_engineer/webapp/__init__.py +1 -0
- ai_code_engineer/webapp/__main__.py +45 -0
- ai_code_engineer/webapp/contract.py +36 -0
- ai_code_engineer/webapp/controller.py +3556 -0
- ai_code_engineer/webapp/fake.py +1141 -0
- ai_code_engineer/webapp/launch.py +108 -0
- ai_code_engineer/webapp/server.py +349 -0
- ai_code_engineer/webapp/static/app.css +780 -0
- ai_code_engineer/webapp/static/app.js +2118 -0
- ai_code_engineer/webapp/static/boot.js +19 -0
- ai_code_engineer/webapp/static/index.html +89 -0
- ai_code_engineer/webapp/static/tokens.css +173 -0
- ai_code_engineer/workspace.py +385 -0
- ai_code_engineer-0.1.0.dist-info/METADATA +7 -0
- ai_code_engineer-0.1.0.dist-info/RECORD +44 -0
- ai_code_engineer-0.1.0.dist-info/WHEEL +5 -0
- ai_code_engineer-0.1.0.dist-info/entry_points.txt +2 -0
- ai_code_engineer-0.1.0.dist-info/licenses/LICENSE +21 -0
- ai_code_engineer-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1286 @@
|
|
|
1
|
+
"""What each source file declares, so a small model stops guessing from file names.
|
|
2
|
+
|
|
3
|
+
The repo map used to be one regex over every line, which invented "symbols" out of
|
|
4
|
+
comments and string literals and missed structure a parser sees for free. Python is
|
|
5
|
+
parsed with `ast`. Java, Kotlin, TypeScript, Go and Rust get a bounded scanner over
|
|
6
|
+
comment/string-blanked text: enough to answer "which type owns this method, and what
|
|
7
|
+
does this file import" without pretending to be a compiler.
|
|
8
|
+
|
|
9
|
+
Nothing here executes project code, and every list is capped, because the result is
|
|
10
|
+
pasted into a model prompt whose budget is the whole point.
|
|
11
|
+
"""
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import ast
|
|
15
|
+
import re
|
|
16
|
+
|
|
17
|
+
MAX_FILES = 300
|
|
18
|
+
MAX_TYPES = 40 # declared types per file
|
|
19
|
+
MAX_MEMBERS = 25 # methods per type
|
|
20
|
+
MAX_IMPORTS = 40 # imports per file
|
|
21
|
+
MAX_SIGNATURE = 60 # characters of one signature
|
|
22
|
+
SCRIPT_SUFFIXES = {".js", ".jsx", ".ts", ".tsx", ".mjs", ".cjs"}
|
|
23
|
+
INDEXABLE = {".py", ".java", ".kt", ".go", ".rs"} | SCRIPT_SUFFIXES
|
|
24
|
+
# Kinds whose declared types are importable by their simple name from another file.
|
|
25
|
+
NAMED_KINDS = {"jvm", "script", "go", "rust"}
|
|
26
|
+
|
|
27
|
+
# A name in this set means the line is a statement, not a declaration.
|
|
28
|
+
CONTROL = {"if", "for", "while", "switch", "catch", "return", "do", "else", "try",
|
|
29
|
+
"finally", "throw", "new", "synchronized", "case", "when", "init", "super",
|
|
30
|
+
"this", "assert", "yield", "await", "print", "lambda", "var", "val"}
|
|
31
|
+
LINE_STARTS = ("public ", "private ", "protected ", "static ", "final ", "abstract ",
|
|
32
|
+
"default ", "synchronized ", "native ", "open ", "override ", "suspend ",
|
|
33
|
+
"internal ", "inline ", "operator ", "fun ", "void ", "class ",
|
|
34
|
+
"interface ", "enum ", "record ", "object ", "annotation ", "sealed ",
|
|
35
|
+
"data ", "companion ", "external ", "tailrec ")
|
|
36
|
+
TYPE_RE = re.compile(r"\b(class|interface|enum|record|object|annotation)\s+([A-Za-z_$][\w$]*)")
|
|
37
|
+
PACKAGE_RE = re.compile(r"^\s*package\s+(?:`)?([\w.$]+)")
|
|
38
|
+
IMPORT_RE = re.compile(r"^\s*import\s+(?:static\s+)?(?:`)?([\w.$]+)")
|
|
39
|
+
JAVA_METHOD_RE = re.compile(
|
|
40
|
+
r"^\s*(?:(?:public|private|protected|static|final|abstract|default|synchronized|native"
|
|
41
|
+
r"|strictfp|open|transient|volatile)\s+)+"
|
|
42
|
+
r"(?:([\w$][\w$<>\[\],.?\s]*?)\s+)?([A-Za-z_$][\w$]*)\s*\(([^()]*)\)\s*"
|
|
43
|
+
r"(?:throws [\w$.,\s]+)?[{;]")
|
|
44
|
+
KOTLIN_FUN_RE = re.compile(
|
|
45
|
+
r"^\s*(?:(?:public|private|protected|internal|abstract|open|override|suspend|inline"
|
|
46
|
+
r"|operator|tailrec|external|annotation)\s+)*fun\s+(?:<[^<>]*>\s*)?"
|
|
47
|
+
r"(?:([\w$][\w$.<>,\s]*)\s*\.\s*)?([A-Za-z_$][\w$]*)\s*\(([^()]*)\)")
|
|
48
|
+
# Package-private Java methods carry no modifier, which is how JUnit tests are written.
|
|
49
|
+
JAVA_PLAIN_METHOD_RE = re.compile(
|
|
50
|
+
r"^\s*(?:@[\w.]+\s+)*(?:final\s+|static\s+|abstract\s+|default\s+)*"
|
|
51
|
+
r"([\w$][\w$<>\[\],.?\s]*?)\s+([A-Za-z_$][\w$]*)\s*\(([^()]*)\)\s*"
|
|
52
|
+
r"(?:throws [\w$.,\s]+)?[{;]")
|
|
53
|
+
|
|
54
|
+
# ---------------- JavaScript / TypeScript / JSX ----------------
|
|
55
|
+
# `export` is a modifier here, never a second declaration: the same name is recorded once.
|
|
56
|
+
JS_PREFIX = r"^\s*(?:export\s+)?(?:default\s+)?(?:export\s+)?(?:declare\s+)?"
|
|
57
|
+
# A type-parameter / type-argument list, optional as a whole. One nesting level is enough for the
|
|
58
|
+
# clauses real code writes, a bare `[^>]*` would stop at the inner `>` of `Map<String, T>`, and the
|
|
59
|
+
# fill excludes `{` so the tail of a declaration line is never swallowed by an empty match.
|
|
60
|
+
JS_GENERIC = r"(?:<(?:[^<{]|<[^<>]*>)*>)?"
|
|
61
|
+
JS_CLASS_RE = re.compile(JS_PREFIX + r"class\s+([A-Za-z_$][\w$]*)\s*" + JS_GENERIC + r"\s*"
|
|
62
|
+
r"(?:extends\s+([A-Za-z_$][\w$.]*)\s*" + JS_GENERIC + r"\s*)?"
|
|
63
|
+
r"(?:implements\s+[^{]*)?\{?")
|
|
64
|
+
JS_TYPE_RE = re.compile(JS_PREFIX + r"(interface|enum|type)\s+([A-Za-z_$][\w$]*)")
|
|
65
|
+
JS_FUNC_RE = re.compile(JS_PREFIX + r"(?:async\s+)?function\s*\*?\s*([A-Za-z_$][\w$]*)\s*"
|
|
66
|
+
+ JS_GENERIC + r"\s*\(([^()]*)\)")
|
|
67
|
+
JS_ARROW_RE = re.compile(JS_PREFIX + r"(?:const|let|var)\s+([A-Za-z_$][\w$]*)\s*=\s*"
|
|
68
|
+
r"(?:async\s*)?(?:\(([^()]*)\)|[A-Za-z_$][\w$]*)(?:\s*:[^=\n]+?)?\s*=>")
|
|
69
|
+
# A class member: `name(args) {`, with optional modifiers and an optional `: Type`.
|
|
70
|
+
JS_MEMBER_RE = re.compile(r"^\s*(?:(?:public|private|protected|static|readonly|abstract"
|
|
71
|
+
r"|override|async|declare|get|set|declare)\s+)*#?[A-Za-z_$][\w$]*"
|
|
72
|
+
r"\s*\(([^()]*)\)\s*(?::[^;{]+)?\{")
|
|
73
|
+
JS_MEMBER_NAME_RE = re.compile(r"^\s*(?:(?:public|private|protected|static|readonly|abstract"
|
|
74
|
+
r"|override|async|get|set|declare)\s+)*#?([A-Za-z_$][\w$]*)\s*\(")
|
|
75
|
+
JS_FROM_RE = re.compile(r"(?:^|;)\s*(?:import|export)\b[^;\n]*?\bfrom\s*['\"]([^'\"]+)['\"]")
|
|
76
|
+
JS_BARE_IMPORT_RE = re.compile(r"^\s*import\s*['\"]([^'\"]+)['\"]")
|
|
77
|
+
JS_REQUIRE_RE = re.compile(r"\brequire\(\s*['\"]([^'\"]+)['\"]\s*\)")
|
|
78
|
+
|
|
79
|
+
# ---------------- Go ----------------
|
|
80
|
+
GO_PACKAGE_RE = re.compile(r"^\s*package\s+([\w.]+)")
|
|
81
|
+
# func Name(...) and func (recv Type) Name(...) — the receiver decides the owning type.
|
|
82
|
+
GO_FUNC_RE = re.compile(r"^\s*func\s+(?:\(\s*(?:_\s+|\w+\s+)?\*?([\w\[\]]+)\s*\)\s*)?"
|
|
83
|
+
r"([A-Za-z_]\w*)\s*\(([^()]*)\)")
|
|
84
|
+
GO_TYPE_RE = re.compile(r"^\s*type\s+([A-Za-z_]\w*)\s+(struct|interface)\b")
|
|
85
|
+
GO_QUOTED_RE = re.compile(r'"([^"\n]+)"')
|
|
86
|
+
GO_IMPORT_RE = re.compile(r'^import\s+(?:[\w.]+\s+)?"([^"]+)"', re.M)
|
|
87
|
+
GO_IMPORT_BLOCK_RE = re.compile(r"^import\s*\(([^)]*)\)", re.M | re.S)
|
|
88
|
+
|
|
89
|
+
# ---------------- Rust ----------------
|
|
90
|
+
RS_TYPE_RE = re.compile(r"^\s*(?:pub(?:\([^)]*\))?\s+)?(struct|enum|trait|union|type|mod)\s+"
|
|
91
|
+
r"([A-Za-z_]\w*)")
|
|
92
|
+
RS_IMPL_RE = re.compile(r"^\s*(?:unsafe\s+)?impl\s*(?:<[^<>]*>\s*)?(?:([\w:]+)\s+for\s+)?"
|
|
93
|
+
r"([A-Za-z_][\w:]*)")
|
|
94
|
+
RS_FN_RE = re.compile(r"^\s*(?:(?:pub|crate)(?:\([^)]*\))?|default|const|unsafe|async"
|
|
95
|
+
r"|extern\s*(?:\([^)]*\))?|\s)*fn\s+([A-Za-z_]\w*)\s*(?:<[^<>]*>)?"
|
|
96
|
+
r"\s*\(([^()]*)\)")
|
|
97
|
+
RS_USE_RE = re.compile(r"^\s*(?:pub(?:\([^)]*\))?\s+)?use\s+([^;]+);")
|
|
98
|
+
RS_USE_BRACE_RE = re.compile(r"^([\w:]+)::\{([^}]*)\}$")
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _signature(name: str, params: list[str]) -> str:
|
|
102
|
+
text = re.sub(r"\s+", " ", name + "(" + ", ".join(p.strip() for p in params) + ")")
|
|
103
|
+
return text[:MAX_SIGNATURE] + "…" if len(text) > MAX_SIGNATURE else text
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def blank(source: str, backticks: bool = False) -> str:
|
|
107
|
+
"""Blank out comments and string contents, keeping every line where it was.
|
|
108
|
+
|
|
109
|
+
Without this, `// class Foo` becomes a class and a log message becomes a method.
|
|
110
|
+
`backticks` additionally blanks JavaScript template literals; a stray backtick in a
|
|
111
|
+
regex literal is not allowed to swallow the rest of the file, so an unterminated one
|
|
112
|
+
only reaches the end of its own line.
|
|
113
|
+
"""
|
|
114
|
+
out, i, n = [], 0, len(source)
|
|
115
|
+
while i < n:
|
|
116
|
+
ch = source[i]
|
|
117
|
+
pair = source[i:i + 2]
|
|
118
|
+
if pair == "//":
|
|
119
|
+
end = source.find("\n", i)
|
|
120
|
+
end = n if end < 0 else end
|
|
121
|
+
out.append(" " * (end - i))
|
|
122
|
+
i = end
|
|
123
|
+
elif pair == "/*":
|
|
124
|
+
end = source.find("*/", i + 2)
|
|
125
|
+
end = n if end < 0 else end + 2
|
|
126
|
+
out.append("".join(c if c == "\n" else " " for c in source[i:end]))
|
|
127
|
+
i = end
|
|
128
|
+
elif backticks and ch == "`":
|
|
129
|
+
end = source.find("`", i + 1)
|
|
130
|
+
if end < 0:
|
|
131
|
+
# A stray backtick — inside a regex literal, say — may not eat the file.
|
|
132
|
+
end = source.find("\n", i + 1)
|
|
133
|
+
end = n if end < 0 else end
|
|
134
|
+
out.append(" " * (end - i))
|
|
135
|
+
i = end
|
|
136
|
+
else:
|
|
137
|
+
out.append("".join(c if c == "\n" else " " for c in source[i:end + 1]))
|
|
138
|
+
i = end + 1
|
|
139
|
+
elif source.startswith('"""', i) or source.startswith("'''", i):
|
|
140
|
+
quote = source[i:i + 3]
|
|
141
|
+
end = source.find(quote, i + 3)
|
|
142
|
+
end = n if end < 0 else end + 3
|
|
143
|
+
out.append("".join(c if c == "\n" else " " for c in source[i:end]))
|
|
144
|
+
i = end
|
|
145
|
+
elif ch in "\"'":
|
|
146
|
+
end = i + 1
|
|
147
|
+
while end < n:
|
|
148
|
+
if source[end] == "\\":
|
|
149
|
+
end += 2
|
|
150
|
+
continue
|
|
151
|
+
if source[end] in (ch, "\n"):
|
|
152
|
+
break
|
|
153
|
+
end += 1
|
|
154
|
+
out.append(" " * (end - i))
|
|
155
|
+
# Step over the closing quote, or it reads as the opening one of a second
|
|
156
|
+
# string and swallows the rest of the line.
|
|
157
|
+
i = end + 1 if end < n and source[end] == ch else end
|
|
158
|
+
else:
|
|
159
|
+
out.append(ch)
|
|
160
|
+
i += 1
|
|
161
|
+
return "".join(out)
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def _jvm(path: str, source: str) -> dict:
|
|
165
|
+
"""Java/Kotlin declarations, found by a brace-depth scan of blanked text.
|
|
166
|
+
|
|
167
|
+
A method counts only at one level inside a type body, which is what separates
|
|
168
|
+
`public Token login(String p)` from `tokenService.login(password)` two lines later.
|
|
169
|
+
"""
|
|
170
|
+
row = {"path": path, "kind": "jvm", "package": "", "imports": [], "types": [],
|
|
171
|
+
"functions": []}
|
|
172
|
+
depth = 0
|
|
173
|
+
stack: list[tuple[str, int]] = [] # (type name, depth where it opened)
|
|
174
|
+
for number, line in enumerate(blank(source).splitlines(), 1):
|
|
175
|
+
head = line.strip()
|
|
176
|
+
opened, closed = line.count("{"), line.count("}")
|
|
177
|
+
if head.startswith("package") and not row["package"]:
|
|
178
|
+
match = PACKAGE_RE.match(line)
|
|
179
|
+
if match:
|
|
180
|
+
row["package"] = match.group(1)
|
|
181
|
+
elif head.startswith("import"):
|
|
182
|
+
match = IMPORT_RE.match(line)
|
|
183
|
+
if match and len(row["imports"]) < MAX_IMPORTS and match.group(1) not in row["imports"]:
|
|
184
|
+
row["imports"].append(match.group(1))
|
|
185
|
+
declared = TYPE_RE.findall(line)
|
|
186
|
+
if declared:
|
|
187
|
+
for kind, simple in declared:
|
|
188
|
+
outer = next((name for name, level in reversed(stack) if level == depth), "")
|
|
189
|
+
name = outer + "." + simple if outer else simple
|
|
190
|
+
_declare_type(row, simple, kind, number, outer)
|
|
191
|
+
# The frame opens even when the cap refused the type: without it a method
|
|
192
|
+
# inside has no owner at all and lands in the file's own function list.
|
|
193
|
+
stack.append((name, depth))
|
|
194
|
+
elif any(head.startswith(word) for word in LINE_STARTS):
|
|
195
|
+
method = None
|
|
196
|
+
if re.search(r"\bfun\s", line):
|
|
197
|
+
method = KOTLIN_FUN_RE.match(line)
|
|
198
|
+
method = method or JAVA_METHOD_RE.match(line) or JAVA_PLAIN_METHOD_RE.match(line)
|
|
199
|
+
if method and method.group(2) not in CONTROL:
|
|
200
|
+
params = [part for part in method.group(3).split(",") if part.strip()][:6]
|
|
201
|
+
if stack:
|
|
202
|
+
owner = next((item for item in reversed(row["types"])
|
|
203
|
+
if item["name"] == stack[-1][0]), None)
|
|
204
|
+
if owner and depth == stack[-1][1] + 1 and len(owner["members"]) < MAX_MEMBERS:
|
|
205
|
+
owner["members"].append(_signature(method.group(2), params))
|
|
206
|
+
elif len(row["functions"]) < MAX_MEMBERS:
|
|
207
|
+
# A Kotlin file-level function, which belongs to no type.
|
|
208
|
+
row["functions"].append(_signature(method.group(2), params))
|
|
209
|
+
depth += opened - closed
|
|
210
|
+
while stack and depth <= stack[-1][1]:
|
|
211
|
+
stack.pop()
|
|
212
|
+
return row
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def _row(path: str, kind: str) -> dict:
|
|
216
|
+
return {"path": path, "kind": kind, "package": "", "imports": [], "types": [],
|
|
217
|
+
"functions": []}
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def _add_import(row: dict, name: str) -> None:
|
|
221
|
+
name = name.strip().strip("\"'")
|
|
222
|
+
if name and name not in row["imports"] and len(row["imports"]) < MAX_IMPORTS:
|
|
223
|
+
row["imports"].append(name)
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def _declare_type(row: dict, name: str, kind: str, line: int, outer: str = "") -> dict | None:
|
|
227
|
+
"""The entry for a declared type, created once per name.
|
|
228
|
+
|
|
229
|
+
`export class X` and a later `export { X }`, or a Go type with its methods spread over
|
|
230
|
+
several files in the same folder, must not produce the same type twice.
|
|
231
|
+
"""
|
|
232
|
+
full = outer + "." + name if outer else name
|
|
233
|
+
for item in row["types"]:
|
|
234
|
+
if item["name"] == full:
|
|
235
|
+
return item
|
|
236
|
+
if len(row["types"]) >= MAX_TYPES:
|
|
237
|
+
return None
|
|
238
|
+
item = {"name": full, "kind": kind, "line": line, "members": []}
|
|
239
|
+
row["types"].append(item)
|
|
240
|
+
return item
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def _declare(row: dict, owner: dict | None, name: str, params: str) -> None:
|
|
244
|
+
text = _signature(name, [part for part in params.split(",") if part.strip()][:6])
|
|
245
|
+
bucket = owner["members"] if owner is not None else row["functions"]
|
|
246
|
+
if text not in bucket and len(bucket) < MAX_MEMBERS:
|
|
247
|
+
bucket.append(text)
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def _specifier(raw: str) -> str:
|
|
251
|
+
"""A JavaScript module specifier in the dotted form `dependencies()` compares against.
|
|
252
|
+
|
|
253
|
+
"./text" means this file's own folder, which is exactly what a single leading dot
|
|
254
|
+
means to the Python rules already in place, so one dot is dropped, not kept.
|
|
255
|
+
"""
|
|
256
|
+
text = raw.strip().strip("/")
|
|
257
|
+
if text.startswith("."):
|
|
258
|
+
return text.replace("/", ".")[1:]
|
|
259
|
+
return text.replace("/", ".")
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def _use_targets(clause: str) -> list[str]:
|
|
263
|
+
"""Rust `use` paths, without the crate root and with `::` read as `.`."""
|
|
264
|
+
def one(text: str) -> str:
|
|
265
|
+
path = text.strip().split(" as ")[0].replace("::", ".")
|
|
266
|
+
for root in ("crate.", "self.", "super."):
|
|
267
|
+
if path.startswith(root):
|
|
268
|
+
# `super` is one level up, exactly as a leading dot reads in Python.
|
|
269
|
+
return "." + path[len(root):] if root == "super." else path[len(root):]
|
|
270
|
+
return path
|
|
271
|
+
|
|
272
|
+
brace = RS_USE_BRACE_RE.match(clause.strip())
|
|
273
|
+
if brace:
|
|
274
|
+
return [one(brace.group(1) + "::" + item.strip())
|
|
275
|
+
for item in brace.group(2).split(",") if item.strip()][:MAX_IMPORTS]
|
|
276
|
+
return [one(clause)] if clause.strip() else []
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
def _go_imports(source: str) -> list[str]:
|
|
280
|
+
"""Every import path in a Go file, single-line and grouped.
|
|
281
|
+
|
|
282
|
+
Read from the original text because the blanker erases string contents, and that is
|
|
283
|
+
where the path lives; a line whose own text starts with `//` is a comment, not a path.
|
|
284
|
+
"""
|
|
285
|
+
found = GO_IMPORT_RE.findall(source)
|
|
286
|
+
for block in GO_IMPORT_BLOCK_RE.findall(source):
|
|
287
|
+
for line in block.splitlines():
|
|
288
|
+
text = line.strip()
|
|
289
|
+
if not text or text.startswith("//"):
|
|
290
|
+
continue
|
|
291
|
+
match = GO_QUOTED_RE.search(text)
|
|
292
|
+
if match:
|
|
293
|
+
found.append(match.group(1))
|
|
294
|
+
return found
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
def _script(path: str, source: str) -> dict:
|
|
298
|
+
"""TypeScript and JavaScript declarations.
|
|
299
|
+
|
|
300
|
+
Import specifiers live inside string literals, which the blanker erases, so every line
|
|
301
|
+
is read twice: the blanked copy decides whether a declaration is really there, the
|
|
302
|
+
original copy supplies the module path.
|
|
303
|
+
"""
|
|
304
|
+
row = _row(path, "script")
|
|
305
|
+
blanked = blank(source, backticks=True).splitlines()
|
|
306
|
+
original = source.splitlines()
|
|
307
|
+
depth = 0
|
|
308
|
+
stack: list[tuple[str, int]] = []
|
|
309
|
+
for number, (line, raw) in enumerate(zip(blanked, original), 1):
|
|
310
|
+
head = line.strip()
|
|
311
|
+
opened, closed = line.count("{"), line.count("}")
|
|
312
|
+
declared = None # a class body opened by this line
|
|
313
|
+
if head.startswith(("import", "export")) or "require(" in head:
|
|
314
|
+
found = JS_FROM_RE.findall(raw) + JS_BARE_IMPORT_RE.findall(raw)
|
|
315
|
+
if "require(" in line:
|
|
316
|
+
found += JS_REQUIRE_RE.findall(raw)
|
|
317
|
+
for specifier in found:
|
|
318
|
+
_add_import(row, _specifier(specifier))
|
|
319
|
+
match = JS_CLASS_RE.match(line)
|
|
320
|
+
if match and match.group(1) not in CONTROL:
|
|
321
|
+
outer = next((name for name, level in reversed(stack) if level == depth), "")
|
|
322
|
+
owner = _declare_type(row, match.group(1), "class", number, outer)
|
|
323
|
+
if owner is not None:
|
|
324
|
+
if match.group(2) and not owner.get("extends"):
|
|
325
|
+
owner["extends"] = [match.group(2)[:40]]
|
|
326
|
+
if opened > closed:
|
|
327
|
+
declared = owner["name"]
|
|
328
|
+
elif (match := JS_TYPE_RE.match(line)):
|
|
329
|
+
_declare_type(row, match.group(2), match.group(1), number)
|
|
330
|
+
elif (match := JS_FUNC_RE.match(line)) or (match := JS_ARROW_RE.match(line)):
|
|
331
|
+
_declare(row, _owner_of(row, stack, depth), match.group(1), match.group(2) or "")
|
|
332
|
+
elif "(" in line and line.rstrip().endswith("{"):
|
|
333
|
+
member = JS_MEMBER_RE.match(line)
|
|
334
|
+
owner = _owner_of(row, stack, depth)
|
|
335
|
+
if member and owner is not None:
|
|
336
|
+
_declare(row, owner, JS_MEMBER_NAME_RE.match(line).group(1), member.group(1))
|
|
337
|
+
depth += opened - closed
|
|
338
|
+
if declared:
|
|
339
|
+
stack.append((declared, depth - (opened - closed)))
|
|
340
|
+
while stack and depth <= stack[-1][1]:
|
|
341
|
+
stack.pop()
|
|
342
|
+
return row
|
|
343
|
+
|
|
344
|
+
|
|
345
|
+
def _owner_of(row: dict, stack: list[tuple[str, int]], depth: int) -> dict | None:
|
|
346
|
+
"""The type one level above this line, or None at file level.
|
|
347
|
+
|
|
348
|
+
A function at that level is a method; one deeper is inside a body and is not.
|
|
349
|
+
"""
|
|
350
|
+
if not stack or depth != stack[-1][1] + 1:
|
|
351
|
+
return None
|
|
352
|
+
return next((item for item in reversed(row["types"]) if item["name"] == stack[-1][0]), None)
|
|
353
|
+
|
|
354
|
+
|
|
355
|
+
def _go(path: str, source: str) -> dict:
|
|
356
|
+
"""Go declarations: a receiver picks the owning type, a package names the folder."""
|
|
357
|
+
row = _row(path, "go")
|
|
358
|
+
for imported in _go_imports(source)[:MAX_IMPORTS]:
|
|
359
|
+
_add_import(row, imported)
|
|
360
|
+
for number, line in enumerate(blank(source).splitlines(), 1):
|
|
361
|
+
head = line.strip()
|
|
362
|
+
if not head:
|
|
363
|
+
continue # a commented-out func is not a declaration
|
|
364
|
+
if head.startswith("package") and not row["package"]:
|
|
365
|
+
match = GO_PACKAGE_RE.match(line)
|
|
366
|
+
if match:
|
|
367
|
+
row["package"] = match.group(1)
|
|
368
|
+
elif head.startswith("type"):
|
|
369
|
+
match = GO_TYPE_RE.match(line)
|
|
370
|
+
if match:
|
|
371
|
+
_declare_type(row, match.group(1), match.group(2), number)
|
|
372
|
+
elif head.startswith("func"):
|
|
373
|
+
match = GO_FUNC_RE.match(line)
|
|
374
|
+
if match:
|
|
375
|
+
receiver, name, params = match.groups()
|
|
376
|
+
owner = (_declare_type(row, receiver.strip("*[]"), "type", number)
|
|
377
|
+
if receiver else None)
|
|
378
|
+
_declare(row, owner, name, params)
|
|
379
|
+
return row
|
|
380
|
+
|
|
381
|
+
|
|
382
|
+
def _rust(path: str, source: str) -> dict:
|
|
383
|
+
"""Rust declarations: `impl Trait for Type` and `trait` bodies own their functions."""
|
|
384
|
+
row = _row(path, "rust")
|
|
385
|
+
blanked = blank(source).splitlines()
|
|
386
|
+
original = source.splitlines()
|
|
387
|
+
depth = 0
|
|
388
|
+
owner: dict | None = None
|
|
389
|
+
owner_depth = 0
|
|
390
|
+
for number, (line, raw) in enumerate(zip(blanked, original), 1):
|
|
391
|
+
head = line.strip()
|
|
392
|
+
opened, closed = line.count("{"), line.count("}")
|
|
393
|
+
entered = None
|
|
394
|
+
if head.startswith("use "):
|
|
395
|
+
clause = RS_USE_RE.match(raw)
|
|
396
|
+
if clause:
|
|
397
|
+
for target in _use_targets(clause.group(1)):
|
|
398
|
+
_add_import(row, target)
|
|
399
|
+
elif head.startswith("impl"):
|
|
400
|
+
match = RS_IMPL_RE.match(line)
|
|
401
|
+
if match:
|
|
402
|
+
entered = _declare_type(row, match.group(2).split("::")[-1], "impl", number)
|
|
403
|
+
elif "fn " in line:
|
|
404
|
+
match = RS_FN_RE.match(line)
|
|
405
|
+
if match:
|
|
406
|
+
_declare(row, owner if depth >= owner_depth else None,
|
|
407
|
+
match.group(1), match.group(2))
|
|
408
|
+
else:
|
|
409
|
+
match = RS_TYPE_RE.match(line)
|
|
410
|
+
if match:
|
|
411
|
+
entered = _declare_type(row, match.group(2), match.group(1), number)
|
|
412
|
+
depth += opened - closed
|
|
413
|
+
if entered is not None and opened > closed:
|
|
414
|
+
owner, owner_depth = entered, depth
|
|
415
|
+
elif owner is not None and depth < owner_depth:
|
|
416
|
+
owner, owner_depth = None, 0
|
|
417
|
+
return row
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
def _params(node: ast.arguments) -> list[str]:
|
|
421
|
+
names = [arg.arg for arg in node.posonlyargs + node.args]
|
|
422
|
+
if names and names[0] in {"self", "cls"}:
|
|
423
|
+
names = names[1:]
|
|
424
|
+
names += [arg.arg for arg in node.kwonlyargs]
|
|
425
|
+
if node.vararg:
|
|
426
|
+
names.append("*" + node.vararg.arg)
|
|
427
|
+
if node.kwarg:
|
|
428
|
+
names.append("**" + node.kwarg.arg)
|
|
429
|
+
return names[:6]
|
|
430
|
+
|
|
431
|
+
|
|
432
|
+
def _python(path: str, source: str) -> dict | None:
|
|
433
|
+
try:
|
|
434
|
+
tree = ast.parse(source)
|
|
435
|
+
except (SyntaxError, ValueError, MemoryError, RecursionError):
|
|
436
|
+
return None
|
|
437
|
+
row = {"path": path, "kind": "python", "package": "", "imports": [], "types": [],
|
|
438
|
+
"functions": []}
|
|
439
|
+
|
|
440
|
+
def walk(body, prefix: str, owner: dict | None) -> None:
|
|
441
|
+
for node in body:
|
|
442
|
+
if isinstance(node, ast.ClassDef):
|
|
443
|
+
if len(row["types"]) >= MAX_TYPES:
|
|
444
|
+
return
|
|
445
|
+
item = {"name": prefix + node.name, "kind": "class", "line": node.lineno,
|
|
446
|
+
"extends": [ast.unparse(base)[:40] for base in node.bases][:4],
|
|
447
|
+
"members": []}
|
|
448
|
+
row["types"].append(item)
|
|
449
|
+
walk(node.body, prefix + node.name + ".", item)
|
|
450
|
+
elif isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)):
|
|
451
|
+
text = _signature(node.name, _params(node.args))
|
|
452
|
+
if owner is not None and len(owner["members"]) < MAX_MEMBERS:
|
|
453
|
+
owner["members"].append(text)
|
|
454
|
+
elif owner is None and len(row["functions"]) < MAX_MEMBERS:
|
|
455
|
+
row["functions"].append(text)
|
|
456
|
+
elif isinstance(node, ast.Import):
|
|
457
|
+
row["imports"].extend(alias.name for alias in node.names)
|
|
458
|
+
elif isinstance(node, ast.ImportFrom):
|
|
459
|
+
base = "." * node.level + (node.module or "")
|
|
460
|
+
row["imports"].extend(base + "." + alias.name for alias in node.names)
|
|
461
|
+
elif isinstance(node, (ast.If, ast.Try, ast.With, ast.For, ast.While)):
|
|
462
|
+
walk(node.body, prefix, owner)
|
|
463
|
+
|
|
464
|
+
walk(tree.body, "", None)
|
|
465
|
+
row["imports"] = list(dict.fromkeys(row["imports"]))[:MAX_IMPORTS]
|
|
466
|
+
return row
|
|
467
|
+
|
|
468
|
+
|
|
469
|
+
def fallback(path: str, source: str) -> dict:
|
|
470
|
+
"""The old lexical scan, kept for files a real parser refuses to read."""
|
|
471
|
+
names = re.findall(r"(?m)^\s*(?:(?:public|private|protected|abstract|final|static)\s+)*"
|
|
472
|
+
r"(?:class|interface|enum|record|def|fun)\s+(\w+)", source)
|
|
473
|
+
return {"path": path, "kind": "lexical", "package": "", "imports": [], "functions": [],
|
|
474
|
+
"types": [{"name": name, "kind": "lexical", "line": 0, "members": []}
|
|
475
|
+
for name in names[:MAX_TYPES]]}
|
|
476
|
+
|
|
477
|
+
|
|
478
|
+
def suffix_of(path: str) -> str:
|
|
479
|
+
parts = path.casefold().rsplit(".", 1)
|
|
480
|
+
return "." + parts[-1] if len(parts) == 2 else ""
|
|
481
|
+
|
|
482
|
+
|
|
483
|
+
# The files that say what a project *is* and how it runs. Not source, and not everything that is not
|
|
484
|
+
# source: a map that listed every YAML file would be a file listing again. This is the short set a
|
|
485
|
+
# model needs in order to know a reactor's modules, a service's port and a package's dependencies
|
|
486
|
+
# without spending three of its twelve turns reading them.
|
|
487
|
+
CONFIG_NAMES = {"pom.xml", "build.gradle", "build.gradle.kts", "settings.gradle",
|
|
488
|
+
"settings.gradle.kts", "package.json", "go.mod", "cargo.toml", "pyproject.toml"}
|
|
489
|
+
CONFIG_PREFIXES = ("application", "bootstrap")
|
|
490
|
+
CONFIG_SUFFIXES = {".yml", ".yaml", ".properties"}
|
|
491
|
+
MAX_FACTS = 6
|
|
492
|
+
MAX_FACT_CHARS = 160
|
|
493
|
+
# A key's *name* decides whether its value may be shown. `server.port` is what a starting order needs;
|
|
494
|
+
# `spring.datasource.password` is not, and a map that carried it would also carry it into
|
|
495
|
+
# `agent export-session`, which is the one artefact that leaves the machine.
|
|
496
|
+
# `user` is here because a leaf key can be allow-listed while the path above it is a credential store:
|
|
497
|
+
# `spring.security.user.name` and `spring.security.user.password` both sit under a login.
|
|
498
|
+
CREDENTIAL_KEY = re.compile(r"(?i)(pass(word|wd)?|secret|token|key|credential|cert(ificate)?|user)")
|
|
499
|
+
# Matched against the whole dotted path, so a key only means what its owner says it means. A bare
|
|
500
|
+
# `name` under `logging.file` is a log file, not the application, and a map that said otherwise would
|
|
501
|
+
# be a wrong fact the model has no way to doubt.
|
|
502
|
+
KEY_FACTS = {"port": "port", "server.port": "port", "management.server.port": "management port",
|
|
503
|
+
"name": "name", "application.name": "name", "spring.application.name": "name",
|
|
504
|
+
"artifactid": "artifact"}
|
|
505
|
+
# The three leaves that say the same thing whoever owns them, so the flat `spring.datasource.url=…`
|
|
506
|
+
# spelling of a properties file reads the same as the nested YAML one.
|
|
507
|
+
LEAF_FACTS = {"port": "port", "defaultzone": "registry", "url": "host"}
|
|
508
|
+
# A URL's authority, with the user and the password taken off the front of it. The leading group is
|
|
509
|
+
# repeatable because JDBC nests its own scheme -- `jdbc:postgresql://user:pw@host:5432/db` is one of the
|
|
510
|
+
# most common lines in a Spring project, and a single-scheme pattern read it as no host at all.
|
|
511
|
+
URL_HOST = re.compile(r"^(?:[\w+.\-]+:)*//(?:[^@/]+@)?([\w.\-]+(?::\d+)?)")
|
|
512
|
+
# The same authority with no scheme in front of it (`localhost:8761/eureka`). It cannot carry a user and
|
|
513
|
+
# a password, because those need a scheme to sit after.
|
|
514
|
+
BARE_HOST = re.compile(r"^([\w.\-]+(?::\d+)?)(?:[/?].*)?$")
|
|
515
|
+
|
|
516
|
+
|
|
517
|
+
def noteworthy(path: str) -> bool:
|
|
518
|
+
"""Whether this file's *facts* belong in the map, even though it is not code."""
|
|
519
|
+
name = str(path).rsplit("/", 1)[-1]
|
|
520
|
+
if name.casefold() in CONFIG_NAMES:
|
|
521
|
+
return True
|
|
522
|
+
folded = name.casefold()
|
|
523
|
+
suffix = "." + folded.rsplit(".", 1)[-1] if "." in folded else ""
|
|
524
|
+
return folded.startswith(CONFIG_PREFIXES) and suffix in CONFIG_SUFFIXES
|
|
525
|
+
|
|
526
|
+
|
|
527
|
+
def _shorten(label: str, value: str) -> tuple[str, str] | None:
|
|
528
|
+
value = re.sub(r"\s+", " ", str(value)).strip()
|
|
529
|
+
if not value or CREDENTIAL_KEY.search(label):
|
|
530
|
+
return None
|
|
531
|
+
if len(value) > MAX_FACT_CHARS:
|
|
532
|
+
# A dependency list cut mid-token leaves a name the repository does not contain, and the model
|
|
533
|
+
# goes looking for it. Prefer the last whole entry, and say that the list continues. The marker
|
|
534
|
+
# is inside the cap, not after it: `render()` prints these verbatim, and a fact that ran four
|
|
535
|
+
# characters long would be the one thing in the map to exceed its own limit.
|
|
536
|
+
room = MAX_FACT_CHARS - 5
|
|
537
|
+
if (cut := value.rfind(", ", 0, room)) > 0:
|
|
538
|
+
return (label, value[:cut] + ", ...")
|
|
539
|
+
return (label, value[:room].rstrip() + " ...")
|
|
540
|
+
return (label, value)
|
|
541
|
+
|
|
542
|
+
|
|
543
|
+
def _pom(source: str) -> list[tuple[str, str]]:
|
|
544
|
+
"""Maven's own answers: what this module is, what it builds, and what it needs.
|
|
545
|
+
|
|
546
|
+
A `<!DOCTYPE` is refused outright rather than parsed. Python's ElementTree resolves no external
|
|
547
|
+
entities, but it still expands an internal DTD, and this file comes from a repository the tool was
|
|
548
|
+
asked to read, not from a build system it trusts — a few kilobytes of nested entities would cost
|
|
549
|
+
the whole task. Real POMs carry no DOCTYPE, so nothing is lost by the rule; `engine` refuses the
|
|
550
|
+
same shape when it checks a POM it is about to write.
|
|
551
|
+
"""
|
|
552
|
+
import xml.etree.ElementTree as ET
|
|
553
|
+
|
|
554
|
+
if re.search(r"<!\s*DOCTYPE", source, re.I):
|
|
555
|
+
return []
|
|
556
|
+
try:
|
|
557
|
+
root = ET.fromstring(source)
|
|
558
|
+
except ET.ParseError:
|
|
559
|
+
return []
|
|
560
|
+
# Namespaces are removed from the parsed tree instead of from the text. Stripping the `xmlns…=`
|
|
561
|
+
# attributes with a regex was the first attempt and it failed on every real POM: a POM binds
|
|
562
|
+
# `xmlns:xsi` only to declare `xsi:schemaLocation`, so deleting the declaration leaves the
|
|
563
|
+
# attribute's prefix unbound and the parse dies before it starts.
|
|
564
|
+
for node in root.iter():
|
|
565
|
+
if isinstance(node.tag, str) and "}" in node.tag:
|
|
566
|
+
node.tag = node.tag.rsplit("}", 1)[-1]
|
|
567
|
+
|
|
568
|
+
def text(tag: str, node=None):
|
|
569
|
+
node = root if node is None else node
|
|
570
|
+
found = node.find(tag)
|
|
571
|
+
return found.text if found is not None and found.text else ""
|
|
572
|
+
|
|
573
|
+
facts = []
|
|
574
|
+
if (owner := root.find("parent")) is not None:
|
|
575
|
+
if named := _shorten("parent", text("artifactId", owner)):
|
|
576
|
+
facts.append(named)
|
|
577
|
+
if named := _shorten("artifact", text("artifactId")):
|
|
578
|
+
facts.append(named)
|
|
579
|
+
modules = [str(item.text).strip() for item in root.findall("modules/module")
|
|
580
|
+
if item.text and str(item.text).strip()]
|
|
581
|
+
if modules:
|
|
582
|
+
facts.append(("modules", ", ".join(modules[:12])))
|
|
583
|
+
deps = sorted({str(item.text).strip() for item in root.findall("dependencies/dependency/artifactId")
|
|
584
|
+
if item.text and str(item.text).strip()})
|
|
585
|
+
if deps:
|
|
586
|
+
facts.append(("needs", ", ".join(deps[:12])))
|
|
587
|
+
return [fact for fact in (_shorten(label, value) for label, value in facts) if fact][:MAX_FACTS]
|
|
588
|
+
|
|
589
|
+
|
|
590
|
+
GRADLE_DEP = re.compile(r"""(?:implementation|api|testImplementation|compileOnly|runtimeOnly|kapt|
|
|
591
|
+
annotationProcessor|classpath)\s*\(?\s*["']([\w.\-]+:[\w.\-]+)""", re.X)
|
|
592
|
+
GRADLE_INCLUDE = re.compile(r"""include\s+[\s,]*(['"]([\w.\-:]+)['"](?:\s*,\s*['"]([\w.\-:]+)['"])*)""")
|
|
593
|
+
# A module's own siblings. `implementation project(':common-lib')` is the internal edge the map exists
|
|
594
|
+
# to show, and it is spelled with no group and no version, so the coordinate pattern above cannot see it.
|
|
595
|
+
GRADLE_PROJECT = re.compile(r"""project\s*\(\s*['"]:?([\w.\-]+)['"]""")
|
|
596
|
+
PACKAGE_NAME = re.compile(r"""^\s*(?:(?:const|let|var)\s+)?rootProject\.name\s*=\s*['"]([\w.\-]+)""",
|
|
597
|
+
re.M)
|
|
598
|
+
|
|
599
|
+
|
|
600
|
+
def _gradle(source: str) -> list[tuple[str, str]]:
|
|
601
|
+
facts = []
|
|
602
|
+
if named := PACKAGE_NAME.search(source):
|
|
603
|
+
facts.append(("artifact", named.group(1)))
|
|
604
|
+
includes = set()
|
|
605
|
+
for match in GRADLE_INCLUDE.finditer(source):
|
|
606
|
+
for group in match.groups():
|
|
607
|
+
for piece in (group or "").split(","):
|
|
608
|
+
piece = piece.strip().strip("'\"").strip(":")
|
|
609
|
+
if piece:
|
|
610
|
+
includes.add(piece)
|
|
611
|
+
if includes:
|
|
612
|
+
facts.append(("modules", ", ".join(sorted(includes)[:12])))
|
|
613
|
+
deps = sorted({match.group(1).split(":")[-2] if match.group(1).count(":") >= 2
|
|
614
|
+
else match.group(1).split(":")[-1] for match in GRADLE_DEP.finditer(source)}
|
|
615
|
+
| {match.group(1) for match in GRADLE_PROJECT.finditer(source)})
|
|
616
|
+
if deps:
|
|
617
|
+
facts.append(("needs", ", ".join(deps[:12])))
|
|
618
|
+
return [fact for fact in (_shorten(label, value) for label, value in facts) if fact][:MAX_FACTS]
|
|
619
|
+
|
|
620
|
+
|
|
621
|
+
def _package_json(source: str) -> list[tuple[str, str]]:
|
|
622
|
+
import json
|
|
623
|
+
|
|
624
|
+
try:
|
|
625
|
+
data = json.loads(source)
|
|
626
|
+
except ValueError:
|
|
627
|
+
return []
|
|
628
|
+
if not isinstance(data, dict):
|
|
629
|
+
return []
|
|
630
|
+
facts = []
|
|
631
|
+
if isinstance(data.get("name"), str):
|
|
632
|
+
facts.append(("artifact", data["name"]))
|
|
633
|
+
scripts = [key for key in ("build", "test", "start", "lint") if key in (data.get("scripts") or {})]
|
|
634
|
+
if scripts:
|
|
635
|
+
facts.append(("scripts", ", ".join(scripts)))
|
|
636
|
+
for bucket in ("dependencies", "devDependencies"):
|
|
637
|
+
keys = data.get(bucket)
|
|
638
|
+
if isinstance(keys, dict) and keys:
|
|
639
|
+
facts.append((("needs" if bucket == "dependencies" else "dev-needs"),
|
|
640
|
+
", ".join(sorted(keys)[:12])))
|
|
641
|
+
return [fact for fact in (_shorten(label, value) for label, value in facts) if fact][:MAX_FACTS]
|
|
642
|
+
|
|
643
|
+
|
|
644
|
+
GO_MODULE = re.compile(r"^\s*module\s+(\S+)", re.M)
|
|
645
|
+
GO_REQUIRE = re.compile(r"^\s*(?:require|use)\s+(\S+)")
|
|
646
|
+
GO_BLOCK_ITEM = re.compile(r"^\s*(\S+)\s+\S")
|
|
647
|
+
|
|
648
|
+
|
|
649
|
+
def _go_mod(source: str) -> list[tuple[str, str]]:
|
|
650
|
+
"""The module this is and the modules it needs, in both of go.mod's two spellings.
|
|
651
|
+
|
|
652
|
+
A `require x v1` line and a `require ( x v1 )` block are the same statement written two ways, and
|
|
653
|
+
the block is the usual way for anything with more than a couple of dependencies -- reading only the
|
|
654
|
+
line form left a real project's manifest claiming it needed nothing.
|
|
655
|
+
"""
|
|
656
|
+
facts = []
|
|
657
|
+
if named := GO_MODULE.search(source):
|
|
658
|
+
facts.append(("artifact", named.group(1)))
|
|
659
|
+
needs, block = set(), False
|
|
660
|
+
for line in source.splitlines():
|
|
661
|
+
stripped = line.strip()
|
|
662
|
+
if not stripped or stripped.startswith("//"):
|
|
663
|
+
continue
|
|
664
|
+
if block:
|
|
665
|
+
if stripped.startswith(")"):
|
|
666
|
+
block = False
|
|
667
|
+
continue
|
|
668
|
+
if (item := GO_BLOCK_ITEM.match(stripped)):
|
|
669
|
+
needs.add(item.group(1))
|
|
670
|
+
continue
|
|
671
|
+
match = GO_REQUIRE.match(line)
|
|
672
|
+
if not match:
|
|
673
|
+
continue
|
|
674
|
+
if match.group(1) == "(":
|
|
675
|
+
block = True
|
|
676
|
+
continue
|
|
677
|
+
needs.add(match.group(1))
|
|
678
|
+
keep = sorted(item for item in needs if "." in item) # a host in front of the path, not stdlib
|
|
679
|
+
if keep:
|
|
680
|
+
facts.append(("needs", ", ".join(keep[:12])))
|
|
681
|
+
return [fact for fact in (_shorten(label, value) for label, value in facts) if fact][:MAX_FACTS]
|
|
682
|
+
|
|
683
|
+
|
|
684
|
+
def _toml(source: str) -> list[tuple[str, str]]:
|
|
685
|
+
"""Both TOML manifests this tool meets: PEP 621's `[project]` and Cargo's `[package]`.
|
|
686
|
+
|
|
687
|
+
They disagree in shape as well as in section name -- Python lists dependencies, Rust keys them --
|
|
688
|
+
so each half is read if it is there and the file simply has fewer facts if it is not.
|
|
689
|
+
"""
|
|
690
|
+
import tomllib
|
|
691
|
+
|
|
692
|
+
try:
|
|
693
|
+
data = tomllib.loads(source)
|
|
694
|
+
except (ValueError, tomllib.TOMLDecodeError):
|
|
695
|
+
return []
|
|
696
|
+
tables = [data.get(key) for key in ("project", "package")
|
|
697
|
+
if isinstance(data.get(key), dict)]
|
|
698
|
+
facts = []
|
|
699
|
+
named = next((str(table["name"]) for table in tables if isinstance(table.get("name"), str)), "")
|
|
700
|
+
if named:
|
|
701
|
+
facts.append(("artifact", named))
|
|
702
|
+
needs: list[str] = []
|
|
703
|
+
for table in tables:
|
|
704
|
+
found = table.get("dependencies")
|
|
705
|
+
if isinstance(found, dict):
|
|
706
|
+
needs.extend(found.keys())
|
|
707
|
+
elif isinstance(found, list):
|
|
708
|
+
needs.extend(str(item) for item in found)
|
|
709
|
+
if isinstance(data.get("dependencies"), dict):
|
|
710
|
+
needs.extend(data["dependencies"].keys())
|
|
711
|
+
if needs:
|
|
712
|
+
names = sorted({str(item).split(";")[0].strip().split("[")[0].split("=")[0].split(">")[0]
|
|
713
|
+
.split("<")[0].split("!")[0].strip() for item in needs if str(item).strip()})
|
|
714
|
+
names = [item for item in names if item]
|
|
715
|
+
if names:
|
|
716
|
+
facts.append(("needs", ", ".join(names[:12])))
|
|
717
|
+
return [fact for fact in (_shorten(label, value) for label, value in facts) if fact][:MAX_FACTS]
|
|
718
|
+
|
|
719
|
+
|
|
720
|
+
# The separator is required and the value is not, because a parent line (`application:`) carries
|
|
721
|
+
# nothing and still owns the keys under it -- without it in the stack every child resolves against the
|
|
722
|
+
# last *valued* key, and `spring.application.name` arrives as `port.name`. `=` is accepted too: a
|
|
723
|
+
# `.properties` file writes `server.port=8080`, its keys are already dotted and all sit at indent zero,
|
|
724
|
+
# so one scan reads both dialects.
|
|
725
|
+
YAML_PAIR = re.compile(r"^\s*([A-Za-z][\w.\-]*)\s*[=:](?:\s*(\S.*?))?\s*$")
|
|
726
|
+
|
|
727
|
+
|
|
728
|
+
def _runtime_yml(source: str) -> list[tuple[str, str]]:
|
|
729
|
+
"""A port, an application name, and the host a service registers with.
|
|
730
|
+
|
|
731
|
+
Line-oriented on purpose: YAML has no stdlib parser here, and the keys worth showing are the ones
|
|
732
|
+
that appear on their own line in every Spring Boot file this tool has been pointed at. A value is
|
|
733
|
+
only kept when its whole dotted key is on the allow-list, and a URL is reduced to its authority,
|
|
734
|
+
so the credential that fits inside the connection string never reaches the map.
|
|
735
|
+
"""
|
|
736
|
+
facts = {}
|
|
737
|
+
stack: list[tuple[int, str]] = []
|
|
738
|
+
for line in source.splitlines():
|
|
739
|
+
match = YAML_PAIR.match(line)
|
|
740
|
+
if not match:
|
|
741
|
+
continue
|
|
742
|
+
key = match.group(1)
|
|
743
|
+
indent = len(line) - len(line.lstrip(" "))
|
|
744
|
+
while stack and stack[-1][0] >= indent:
|
|
745
|
+
stack.pop()
|
|
746
|
+
stack.append((indent, key))
|
|
747
|
+
# An inline comment is not part of the value; YAML only starts one after a space, so a `#` inside
|
|
748
|
+
# a URL fragment survives.
|
|
749
|
+
value = re.sub(r"\s+#.*$", "", (match.group(2) or "")).strip().strip("'\"")
|
|
750
|
+
if not value:
|
|
751
|
+
continue
|
|
752
|
+
path = ".".join(part for _, part in stack).casefold()
|
|
753
|
+
label = KEY_FACTS.get(path) or LEAF_FACTS.get(path.rsplit(".", 1)[-1])
|
|
754
|
+
if not label or CREDENTIAL_KEY.search(path):
|
|
755
|
+
continue
|
|
756
|
+
if label in ("registry", "host"):
|
|
757
|
+
host, plain = URL_HOST.match(value), BARE_HOST.match(value)
|
|
758
|
+
kept = (host or plain).group(1) if (host or plain) else ""
|
|
759
|
+
if kept:
|
|
760
|
+
facts.setdefault(label, kept)
|
|
761
|
+
elif len(value) <= 60:
|
|
762
|
+
facts.setdefault(label, value)
|
|
763
|
+
return [(label, value) for label, value in facts.items()][:MAX_FACTS]
|
|
764
|
+
|
|
765
|
+
|
|
766
|
+
def config_facts(path: str, source: str) -> list[tuple[str, str]]:
|
|
767
|
+
"""What a configuration file says about the project, with nothing secret in it.
|
|
768
|
+
|
|
769
|
+
Failures are silent by design: an unparseable pom is a map without that line, not a task that
|
|
770
|
+
cannot start. The values that survive are names and numbers — module, artifact, port, dependency —
|
|
771
|
+
and any key whose *name* looks like a credential is dropped before it is considered, because the
|
|
772
|
+
repository's own redaction cannot see a value that was never supposed to be in the prompt.
|
|
773
|
+
"""
|
|
774
|
+
name = str(path).rsplit("/", 1)[-1].casefold()
|
|
775
|
+
if name == "pom.xml":
|
|
776
|
+
return _pom(source)
|
|
777
|
+
if name.startswith("build.gradle") or name.startswith("settings.gradle"):
|
|
778
|
+
return _gradle(source)
|
|
779
|
+
if name == "package.json":
|
|
780
|
+
return _package_json(source)
|
|
781
|
+
if name == "go.mod":
|
|
782
|
+
return _go_mod(source)
|
|
783
|
+
if name in ("cargo.toml", "pyproject.toml"):
|
|
784
|
+
return _toml(source)
|
|
785
|
+
return _runtime_yml(source)
|
|
786
|
+
|
|
787
|
+
|
|
788
|
+
def indexable(path: str) -> bool:
|
|
789
|
+
"""Whether this file is worth reading at all for the map — checked before the read."""
|
|
790
|
+
return suffix_of(path) in INDEXABLE
|
|
791
|
+
|
|
792
|
+
|
|
793
|
+
def parse(path: str, source: str) -> dict | None:
|
|
794
|
+
"""One file's declarations, or None when the suffix is not an indexable source."""
|
|
795
|
+
suffix = suffix_of(path)
|
|
796
|
+
if suffix not in INDEXABLE:
|
|
797
|
+
return None
|
|
798
|
+
if suffix == ".py":
|
|
799
|
+
# A model-written file that does not parse yet is exactly when the map matters most,
|
|
800
|
+
# so the old lexical scan covers for the parser instead of the file going blank.
|
|
801
|
+
return _python(path, source) or fallback(path, source)
|
|
802
|
+
if suffix in SCRIPT_SUFFIXES:
|
|
803
|
+
return _script(path, source)
|
|
804
|
+
if suffix == ".go":
|
|
805
|
+
return _go(path, source)
|
|
806
|
+
if suffix == ".rs":
|
|
807
|
+
return _rust(path, source)
|
|
808
|
+
return _jvm(path, source)
|
|
809
|
+
|
|
810
|
+
|
|
811
|
+
def module_name(path: str) -> str:
|
|
812
|
+
parts = [part for part in re.split(r"[/\\.]", path.rsplit(".", 1)[0]) if part]
|
|
813
|
+
if parts and parts[-1] == "__init__":
|
|
814
|
+
parts.pop()
|
|
815
|
+
return ".".join(parts)
|
|
816
|
+
|
|
817
|
+
|
|
818
|
+
def _keys(row: dict) -> dict[str, str]:
|
|
819
|
+
"""Every dotted name another file could import to reach this one."""
|
|
820
|
+
module = module_name(row["path"])
|
|
821
|
+
names = {module}
|
|
822
|
+
if row["kind"] == "go":
|
|
823
|
+
# Go is imported by folder, so the package directory addresses every file in it.
|
|
824
|
+
parts = module.split(".")
|
|
825
|
+
if len(parts) > 1:
|
|
826
|
+
names.add(".".join(parts[:-1]))
|
|
827
|
+
if row["kind"] in NAMED_KINDS:
|
|
828
|
+
for item in row["types"]:
|
|
829
|
+
simple = item["name"].split(".")[0]
|
|
830
|
+
names.add(simple)
|
|
831
|
+
if row["package"]:
|
|
832
|
+
names.add(row["package"] + "." + simple)
|
|
833
|
+
return {name: row["path"] for name in names}
|
|
834
|
+
|
|
835
|
+
|
|
836
|
+
def _linked(imported: str, key: str) -> bool:
|
|
837
|
+
"""True when either name is the whole tail of the other, or the other's prefix.
|
|
838
|
+
|
|
839
|
+
Both sides are split on `.` and `/`, because a Go or JavaScript import is a path while
|
|
840
|
+
the indexed module is a dotted name. A Java file lives under `src/main/java/...`, so the
|
|
841
|
+
import is shorter than the indexed module by exactly that prefix; a Python or Rust
|
|
842
|
+
`…::name` import is longer by the trailing symbol, which is the mirror case and only
|
|
843
|
+
matches as a prefix.
|
|
844
|
+
"""
|
|
845
|
+
left = [part for part in re.split(r"[./\\]", imported) if part]
|
|
846
|
+
right = [part for part in re.split(r"[./\\]", key) if part]
|
|
847
|
+
shared = min(len(left), len(right))
|
|
848
|
+
if shared and left[-shared:] == right[-shared:]:
|
|
849
|
+
return True
|
|
850
|
+
shorter, longer = (left, right) if len(left) <= len(right) else (right, left)
|
|
851
|
+
return bool(shorter) and len(shorter) < len(longer) and longer[:len(shorter)] == shorter
|
|
852
|
+
|
|
853
|
+
|
|
854
|
+
def dependencies(rows: list[dict]) -> dict[str, list[str]]:
|
|
855
|
+
"""Project-internal files each indexed file imports.
|
|
856
|
+
|
|
857
|
+
Only edges inside the repository are kept: a model cannot act on a JDK or stdlib
|
|
858
|
+
import, and listing those would crowd out the one that matters.
|
|
859
|
+
"""
|
|
860
|
+
owners: dict[str, str] = {}
|
|
861
|
+
for row in rows:
|
|
862
|
+
for name, path in _keys(row).items():
|
|
863
|
+
owners.setdefault(name, path)
|
|
864
|
+
edges: dict[str, list[str]] = {}
|
|
865
|
+
for row in rows:
|
|
866
|
+
targets = set()
|
|
867
|
+
for imported in row["imports"]:
|
|
868
|
+
dotted = imported.lstrip(".")
|
|
869
|
+
if imported.startswith("."):
|
|
870
|
+
level = len(imported) - len(dotted)
|
|
871
|
+
parts = module_name(row["path"]).split(".")[:-level]
|
|
872
|
+
dotted = ".".join(parts + ([dotted] if dotted else []))
|
|
873
|
+
for key, path in owners.items():
|
|
874
|
+
if path != row["path"] and _linked(dotted, key):
|
|
875
|
+
targets.add(path)
|
|
876
|
+
edges[row["path"]] = sorted(targets)[:12]
|
|
877
|
+
return edges
|
|
878
|
+
|
|
879
|
+
|
|
880
|
+
def module_of(relative: str) -> str:
|
|
881
|
+
"""The folder a file belongs to at the level a monorepo is divided — `auth-service` in
|
|
882
|
+
`auth-service/src/main/java/App.java`, and "." for a file loose at the root."""
|
|
883
|
+
parts = [part for part in str(relative).replace("\\", "/").split("/") if part not in ("", ".")]
|
|
884
|
+
return parts[0] if len(parts) > 1 else "."
|
|
885
|
+
|
|
886
|
+
|
|
887
|
+
def spread(files: list[str]) -> list[str]:
|
|
888
|
+
"""Round-robin the file list across modules, so a capped map shows every one of them.
|
|
889
|
+
|
|
890
|
+
Alphabetical order and a 12 000-char budget mean one thing in a nine-module reactor: the first
|
|
891
|
+
three modules are described in detail and the other six are not mentioned at all, and the model
|
|
892
|
+
is asked to plan a cross-module change from that. Within a module the order is unchanged.
|
|
893
|
+
"""
|
|
894
|
+
groups: dict[str, list[str]] = {}
|
|
895
|
+
for name in files:
|
|
896
|
+
groups.setdefault(module_of(name), []).append(name)
|
|
897
|
+
ordered: list[str] = []
|
|
898
|
+
depth = 0
|
|
899
|
+
while any(len(items) > depth for items in groups.values()):
|
|
900
|
+
for key in sorted(groups):
|
|
901
|
+
if len(groups[key]) > depth:
|
|
902
|
+
ordered.append(groups[key][depth])
|
|
903
|
+
depth += 1
|
|
904
|
+
return ordered
|
|
905
|
+
|
|
906
|
+
|
|
907
|
+
MAX_GRAPH_NODES = 24 # modules on screen at once
|
|
908
|
+
MAX_GRAPH_EDGES = 80 # lines between them
|
|
909
|
+
MAX_GRAPH_DEPTH = 8 # columns; a cycle is clamped here rather than followed forever
|
|
910
|
+
|
|
911
|
+
|
|
912
|
+
def graph(rows: list[dict]) -> dict:
|
|
913
|
+
"""The project as modules and the dependencies between them, in columns by build order.
|
|
914
|
+
|
|
915
|
+
Files are the wrong unit for a picture: a reactor of a thousand files drawn as a thousand nodes is
|
|
916
|
+
a picture of nothing, and `module_of()` already names the folders a person thinks in. Edges carry a
|
|
917
|
+
count so a thick line is visibly a heavier dependency, and a node's column is the longest chain that
|
|
918
|
+
must be built before it -- the same order the run command starts the projects in.
|
|
919
|
+
|
|
920
|
+
A cycle has no longest path. The layering peels what it can, breaks the remainder at one named node,
|
|
921
|
+
and says `cyclic` in the result: a diagram that quietly re-ordered itself would be a diagram of a
|
|
922
|
+
project that does not exist, and a real cycle between two Maven modules is exactly the thing a
|
|
923
|
+
reader looks at a graph to find.
|
|
924
|
+
"""
|
|
925
|
+
files: dict[str, int] = {}
|
|
926
|
+
for row in rows or []:
|
|
927
|
+
if row.get("kind") == "config":
|
|
928
|
+
continue # a pom is a fact about a module, not a dependency of one
|
|
929
|
+
name = module_of(row["path"])
|
|
930
|
+
files[name] = files.get(name, 0) + 1
|
|
931
|
+
|
|
932
|
+
counts: dict[tuple[str, str], int] = {}
|
|
933
|
+
for origin, targets in dependencies(rows or []).items():
|
|
934
|
+
source = module_of(origin)
|
|
935
|
+
for target in targets:
|
|
936
|
+
dest = module_of(target)
|
|
937
|
+
if dest != source:
|
|
938
|
+
counts[(source, dest)] = counts.get((source, dest), 0) + 1
|
|
939
|
+
|
|
940
|
+
# The heaviest modules are drawn; the rest are counted. A cap that dropped nodes in silence would
|
|
941
|
+
# turn a missing box into a claim that the module depends on nothing.
|
|
942
|
+
keep = sorted(files, key=lambda name: (-files[name], name))[:MAX_GRAPH_NODES]
|
|
943
|
+
live = set(keep)
|
|
944
|
+
edges = sorted(((pair, count) for pair, count in counts.items()
|
|
945
|
+
if pair[0] in live and pair[1] in live),
|
|
946
|
+
key=lambda item: (-item[1], item[0]))[:MAX_GRAPH_EDGES]
|
|
947
|
+
hidden = len(files) - len(live)
|
|
948
|
+
|
|
949
|
+
depends: dict[str, list[str]] = {}
|
|
950
|
+
names: set[str] = set(live)
|
|
951
|
+
for (source, dest), _count in edges:
|
|
952
|
+
depends.setdefault(source, []).append(dest)
|
|
953
|
+
names.update((source, dest))
|
|
954
|
+
|
|
955
|
+
# A node's column is one past the deepest thing it needs, so the columns read as a build order.
|
|
956
|
+
# Nodes are placed by peeling: everything with no unplaced dependency is at the front, and a graph
|
|
957
|
+
# where nothing is ready is a cycle -- broken at a fixed node, named in the result, because a
|
|
958
|
+
# diagram that silently re-ordered itself is a diagram of a project that does not exist.
|
|
959
|
+
placed: dict[str, int] = {}
|
|
960
|
+
remaining = set(names)
|
|
961
|
+
cyclic = False
|
|
962
|
+
while remaining:
|
|
963
|
+
ready = sorted(name for name in remaining
|
|
964
|
+
if not set(depends.get(name, [])) & remaining)
|
|
965
|
+
if not ready:
|
|
966
|
+
cyclic = True
|
|
967
|
+
ready = [sorted(remaining, key=lambda item: (
|
|
968
|
+
-len(set(depends.get(item, [])) - remaining), item))[0]]
|
|
969
|
+
for name in ready:
|
|
970
|
+
# Only dependencies already placed count. Peeling guarantees that in an acyclic graph, and it
|
|
971
|
+
# is what makes the layering a build order; in the node chosen to break a cycle the unplaced
|
|
972
|
+
# dependency is the loop itself, and counting it would shift every column right by one to
|
|
973
|
+
# draw an empty first one.
|
|
974
|
+
deepest = max((placed[dep] for dep in depends.get(name, []) if dep in placed), default=-1)
|
|
975
|
+
placed[name] = min(MAX_GRAPH_DEPTH, deepest + 1)
|
|
976
|
+
remaining.discard(name)
|
|
977
|
+
|
|
978
|
+
return {"nodes": [{"name": name, "files": files.get(name, 0), "column": placed.get(name, 0)}
|
|
979
|
+
for name in sorted(names, key=lambda item: (placed.get(item, 0), item))],
|
|
980
|
+
"edges": [{"from": source, "to": dest, "count": count}
|
|
981
|
+
for (source, dest), count in edges],
|
|
982
|
+
"columns": max(placed.values(), default=0) + 1,
|
|
983
|
+
"cyclic": cyclic,
|
|
984
|
+
"hidden": max(0, hidden)}
|
|
985
|
+
|
|
986
|
+
|
|
987
|
+
MAX_HITS = 40 # reference sites or symbol matches one query may return
|
|
988
|
+
MAX_SITE_TEXT = 200 # characters of the line itself
|
|
989
|
+
PER_FILE_LIMIT = 6 # sites from one file, so a 40-hit answer is not one file's grep
|
|
990
|
+
|
|
991
|
+
# The line kinds a reference can be. The distinction is the whole value of the verb: `search_code`
|
|
992
|
+
# already tells a model that a name appears 30 times, and what it cannot tell is which of those 30 is
|
|
993
|
+
# the declaration, which is an import, and which is somebody calling it.
|
|
994
|
+
IMPORT_LINE = re.compile(r"^\s*(?:from\s+[\w.]+\s+)?import\b|^\s*(?:use|using)\s|#include|^\s*require\s*\(",
|
|
995
|
+
re.I)
|
|
996
|
+
DECLARE_LINE = re.compile(
|
|
997
|
+
r"^\s*(?:@[\w.]+\s+)*(?:public|private|protected|internal|static|final|abstract|override|open|"
|
|
998
|
+
r"sealed|class|interface|enum|record|annotation|def|func|fn|type|impl|export|async|const|let|var|"
|
|
999
|
+
r"pub|local|friend|virtual|override)\b", re.I)
|
|
1000
|
+
|
|
1001
|
+
|
|
1002
|
+
def _member_name(signature: str) -> str:
|
|
1003
|
+
"""`login(String email)` out of a members list, which stores signatures not names."""
|
|
1004
|
+
return signature.split("(", 1)[0].strip()
|
|
1005
|
+
|
|
1006
|
+
|
|
1007
|
+
def find_symbol(rows: list[dict], name: str, limit: int = MAX_HITS) -> list[dict]:
|
|
1008
|
+
"""Every declaration in the index whose name is this one.
|
|
1009
|
+
|
|
1010
|
+
Types come with the line the index recorded; functions and members do not, because the parsers keep
|
|
1011
|
+
signatures and not one line number per member — a hit without a `line` is honest about that, and
|
|
1012
|
+
`read_file` on the path is the next step rather than a guessed one.
|
|
1013
|
+
"""
|
|
1014
|
+
needle = str(name or "").strip()
|
|
1015
|
+
if not needle:
|
|
1016
|
+
return []
|
|
1017
|
+
folded = needle.casefold()
|
|
1018
|
+
exact: list[dict] = []
|
|
1019
|
+
partial: list[dict] = []
|
|
1020
|
+
|
|
1021
|
+
def collect(entry: dict, whole: bool) -> None:
|
|
1022
|
+
(exact if whole else partial).append(entry)
|
|
1023
|
+
|
|
1024
|
+
for row in rows:
|
|
1025
|
+
for item in row["types"]:
|
|
1026
|
+
last = item["name"].rsplit(".", 1)[-1]
|
|
1027
|
+
if folded not in last.casefold():
|
|
1028
|
+
continue
|
|
1029
|
+
entry = {"name": item["name"], "kind": item["kind"], "path": row["path"],
|
|
1030
|
+
"package": row["package"], "line": item["line"]}
|
|
1031
|
+
if item.get("extends"):
|
|
1032
|
+
entry["extends"] = ", ".join(item["extends"])
|
|
1033
|
+
collect(entry, last.casefold() == folded)
|
|
1034
|
+
for signature in row["functions"]:
|
|
1035
|
+
member = _member_name(signature)
|
|
1036
|
+
if folded not in member.casefold():
|
|
1037
|
+
continue
|
|
1038
|
+
collect({"name": member, "kind": "function", "path": row["path"],
|
|
1039
|
+
"package": row["package"], "signature": signature}, member.casefold() == folded)
|
|
1040
|
+
for item in row["types"]:
|
|
1041
|
+
for signature in item["members"]:
|
|
1042
|
+
member = _member_name(signature)
|
|
1043
|
+
if folded not in member.casefold():
|
|
1044
|
+
continue
|
|
1045
|
+
collect({"name": member, "kind": "member", "path": row["path"],
|
|
1046
|
+
"package": row["package"], "in": item["name"], "signature": signature},
|
|
1047
|
+
member.casefold() == folded)
|
|
1048
|
+
hits = exact + partial
|
|
1049
|
+
return hits[:limit]
|
|
1050
|
+
|
|
1051
|
+
|
|
1052
|
+
def find_references(name: str, sources, rows: list[dict] | None = None,
|
|
1053
|
+
limit: int = MAX_HITS, per_file: int = PER_FILE_LIMIT) -> list[dict]:
|
|
1054
|
+
"""Where a name is used, and in what role — the reverse of the forward-only `dependencies()`.
|
|
1055
|
+
|
|
1056
|
+
`sources` is any iterable of `(path, text)` pairs, so the caller decides what to pay for: reading
|
|
1057
|
+
the workspace is the same cost as a search, and `symbols` stays free of a workspace import that
|
|
1058
|
+
would close a cycle. Text is matched in `blank()`ed source, which keeps every line where it was and
|
|
1059
|
+
removes comments and string contents — without it a log message naming a class is reported as a
|
|
1060
|
+
use of it, which is the exact mistake the parsers already had to be taught not to make.
|
|
1061
|
+
|
|
1062
|
+
The role is the answer, not the count: `declaration` (the index says this line declares it, or the
|
|
1063
|
+
line opens with a keyword that only a declaration uses), then `import`, then `call`, then
|
|
1064
|
+
`mention`. There is no type inference here and the verb does not claim otherwise — an overloaded
|
|
1065
|
+
method returns every site of that name, and `read_file` on the two that matter is still cheaper
|
|
1066
|
+
than a model guessing which file to open.
|
|
1067
|
+
"""
|
|
1068
|
+
needle = str(name or "").strip()
|
|
1069
|
+
if not needle:
|
|
1070
|
+
return []
|
|
1071
|
+
declared = {(hit["path"], hit.get("line")) for hit in find_symbol(rows or [], needle, limit=500)}
|
|
1072
|
+
site = re.compile(r"(?<!\w)" + re.escape(needle) + r"(?!\w)")
|
|
1073
|
+
call = re.compile(r"(?<!\w)" + re.escape(needle) + r"\s*\(")
|
|
1074
|
+
out: list[dict] = []
|
|
1075
|
+
for path, text in sources:
|
|
1076
|
+
if len(out) >= limit:
|
|
1077
|
+
break
|
|
1078
|
+
clean = blank(text, backticks=path.endswith((".js", ".jsx", ".ts", ".tsx", ".mjs", ".cjs")))
|
|
1079
|
+
counted = 0
|
|
1080
|
+
for number, (raw, bare) in enumerate(zip(text.splitlines(), clean.splitlines()), 1):
|
|
1081
|
+
if not site.search(bare):
|
|
1082
|
+
continue
|
|
1083
|
+
if (path, number) in declared:
|
|
1084
|
+
kind = "declaration"
|
|
1085
|
+
elif IMPORT_LINE.match(bare):
|
|
1086
|
+
kind = "import"
|
|
1087
|
+
elif DECLARE_LINE.match(bare) and site.search(bare):
|
|
1088
|
+
kind = "declaration"
|
|
1089
|
+
elif call.search(bare):
|
|
1090
|
+
kind = "call"
|
|
1091
|
+
else:
|
|
1092
|
+
kind = "mention"
|
|
1093
|
+
out.append({"path": path, "line": number, "kind": kind,
|
|
1094
|
+
"text": raw.strip()[:MAX_SITE_TEXT]})
|
|
1095
|
+
counted += 1
|
|
1096
|
+
if counted >= per_file or len(out) >= limit:
|
|
1097
|
+
break
|
|
1098
|
+
return out
|
|
1099
|
+
|
|
1100
|
+
|
|
1101
|
+
# Words a task says that name nothing. Deliberately tiny: this is not a search engine, and a list that
|
|
1102
|
+
# grows past the pronouns starts hiding the noun somebody meant. `add` was in here once and should not
|
|
1103
|
+
# have been -- it is one of the most common names in code (`add(a, b)`), so a task that says "fix add"
|
|
1104
|
+
# and means the function got nothing.
|
|
1105
|
+
STOP_WORDS = {"the", "and", "for", "with", "that", "this", "from", "into", "your", "please", "make",
|
|
1106
|
+
"change", "fix", "file", "files", "code", "class", "method", "function", "test",
|
|
1107
|
+
"tests", "when", "then", "them", "it", "to", "of", "in", "on", "is", "are", "do", "so",
|
|
1108
|
+
"java", "py", "xml", "yml", "yaml", "json", "sql", "md", "gradle", "pom"}
|
|
1109
|
+
WORD = re.compile(r"[A-Za-z_$][A-Za-z0-9_$]{2,}")
|
|
1110
|
+
CAMEL = re.compile(r"[A-Z]+(?![a-z])|[A-Z][a-z0-9]+|[a-z0-9]+")
|
|
1111
|
+
|
|
1112
|
+
|
|
1113
|
+
def task_words(task: str) -> set[str]:
|
|
1114
|
+
"""The names inside a sentence, including the parts of a camel-case or snake-case name.
|
|
1115
|
+
|
|
1116
|
+
"Wire the token provider into the login flow" has to reach `JwtTokenProvider.login`. The whole word
|
|
1117
|
+
and each of its parts are both offered because a person writes one and the source is spelled the
|
|
1118
|
+
other, and neither alone finds the pair.
|
|
1119
|
+
"""
|
|
1120
|
+
words: set[str] = set()
|
|
1121
|
+
for raw in WORD.findall(str(task or "")):
|
|
1122
|
+
folded = raw.casefold()
|
|
1123
|
+
if folded in STOP_WORDS:
|
|
1124
|
+
continue
|
|
1125
|
+
words.add(folded)
|
|
1126
|
+
for piece in CAMEL.findall(raw):
|
|
1127
|
+
piece = piece.casefold()
|
|
1128
|
+
if len(piece) > 2 and piece not in STOP_WORDS:
|
|
1129
|
+
words.add(piece)
|
|
1130
|
+
return words
|
|
1131
|
+
|
|
1132
|
+
|
|
1133
|
+
def _name_parts(name: str) -> list[str]:
|
|
1134
|
+
return [piece.casefold() for piece in CAMEL.findall(re.split(r"[(\s]", name)[0])]
|
|
1135
|
+
|
|
1136
|
+
|
|
1137
|
+
def rank(rows: list[dict], task: str, limit: int = 6) -> list[dict]:
|
|
1138
|
+
"""Which files the task is about, and the one-line reason each was picked.
|
|
1139
|
+
|
|
1140
|
+
The rule before this function was "keep a file whose exact name appears in the sentence", which
|
|
1141
|
+
works when a person types `JwtTokenProvider.java` and does nothing when they describe the bug. This
|
|
1142
|
+
scores the index instead: a declared name the task says, a part of such a name, the folder the task
|
|
1143
|
+
names, and one hop along an import from any of those. It is still not understanding — it is a ranked
|
|
1144
|
+
guess — which is why every entry carries the reason and the window prints it: an unexplained context
|
|
1145
|
+
block is a claim the tool cannot defend when the wrong file turns up in the diff.
|
|
1146
|
+
"""
|
|
1147
|
+
words = task_words(task)
|
|
1148
|
+
if not words:
|
|
1149
|
+
return []
|
|
1150
|
+
edges = dependencies(rows)
|
|
1151
|
+
reverse: dict[str, list[str]] = {}
|
|
1152
|
+
for importer, imported in edges.items():
|
|
1153
|
+
for target in imported:
|
|
1154
|
+
reverse.setdefault(target, []).append(importer)
|
|
1155
|
+
scores: dict[str, dict] = {}
|
|
1156
|
+
|
|
1157
|
+
def offer(path: str, score: int, why: str, symbol: str = "") -> None:
|
|
1158
|
+
best = scores.get(path)
|
|
1159
|
+
if best is None or score > best["score"]:
|
|
1160
|
+
scores[path] = {"path": path, "score": score, "why": why, "symbol": symbol}
|
|
1161
|
+
elif score == best["score"] and not best["symbol"] and symbol:
|
|
1162
|
+
best["symbol"] = symbol
|
|
1163
|
+
|
|
1164
|
+
for row in rows:
|
|
1165
|
+
path = row["path"]
|
|
1166
|
+
# Rows carry forward-slash relative paths, which is what `module_of` splits on too, so the
|
|
1167
|
+
# basename is a string operation here rather than a pathlib import this module never needed.
|
|
1168
|
+
basename = path.rsplit("/", 1)[-1]
|
|
1169
|
+
# The boundary test is the old rule kept verbatim: `app.py` in the sentence names `app.py`, and
|
|
1170
|
+
# one written inside a longer word does not. Weakening it here would spend context on a file
|
|
1171
|
+
# nobody asked for, which is the one thing this function is allowed to cost.
|
|
1172
|
+
if re.search(r"(?<![\w.])" + re.escape(basename) + r"(?![\w.])", str(task or ""), re.I):
|
|
1173
|
+
offer(path, 10, "names", basename)
|
|
1174
|
+
for item in row["types"]:
|
|
1175
|
+
name = item["name"].rsplit(".", 1)[-1]
|
|
1176
|
+
whole = name.casefold()
|
|
1177
|
+
parts = _name_parts(name)
|
|
1178
|
+
if whole in words:
|
|
1179
|
+
offer(path, 8, "declares", name)
|
|
1180
|
+
elif (hit := next((word for word in parts if word in words), "")):
|
|
1181
|
+
offer(path, 4, "declares", name)
|
|
1182
|
+
for signature in item["members"]:
|
|
1183
|
+
member = _member_name(signature)
|
|
1184
|
+
if member.casefold() in words:
|
|
1185
|
+
offer(path, 7, "declares", member)
|
|
1186
|
+
elif (hit := next((part for part in _name_parts(member) if part in words), "")):
|
|
1187
|
+
offer(path, 3, "defines", member)
|
|
1188
|
+
for signature in row["functions"]:
|
|
1189
|
+
member = _member_name(signature)
|
|
1190
|
+
if member.casefold() in words:
|
|
1191
|
+
offer(path, 7, "defines", member)
|
|
1192
|
+
elif (hit := next((part for part in _name_parts(member) if part in words), "")):
|
|
1193
|
+
offer(path, 3, "defines", member)
|
|
1194
|
+
folder = module_of(path)
|
|
1195
|
+
# A module folder is a name, spelled the way a class is: the task says "auth-service" and
|
|
1196
|
+
# `task_words` splits it in two exactly as it splits a camel-case class. Comparing only the
|
|
1197
|
+
# whole folder meant that every hyphenated module -- which is how a Spring reactor writes
|
|
1198
|
+
# itself -- ranked nothing at all.
|
|
1199
|
+
if folder != "." and (hit := next((word for word in [folder.casefold()] + _name_parts(folder)
|
|
1200
|
+
if word in words), "")):
|
|
1201
|
+
offer(path, 4, "module", hit)
|
|
1202
|
+
|
|
1203
|
+
for path, entry in list(scores.items()):
|
|
1204
|
+
if entry["score"] < 4:
|
|
1205
|
+
continue
|
|
1206
|
+
for other in reverse.get(path, []):
|
|
1207
|
+
offer(other, min(5, entry["score"] - 2), "imports",
|
|
1208
|
+
entry["symbol"] or path.rsplit("/", 1)[-1])
|
|
1209
|
+
|
|
1210
|
+
ordered = sorted(scores.values(), key=lambda entry: (-entry["score"], entry["path"]))
|
|
1211
|
+
return ordered[:limit]
|
|
1212
|
+
|
|
1213
|
+
|
|
1214
|
+
def config_row(path: str, source: str) -> dict | None:
|
|
1215
|
+
"""The map's entry for a configuration file: facts, not declarations.
|
|
1216
|
+
|
|
1217
|
+
It is a row so `render()` prints it in the same block position as a source file, and it is a row
|
|
1218
|
+
with no types and no functions so every name-based query in this module passes over it. A model
|
|
1219
|
+
that learns `modules: auth-service, product-service` from a pom must not then be told the pom
|
|
1220
|
+
"declares" auth-service — that is the confusion this split keeps out.
|
|
1221
|
+
"""
|
|
1222
|
+
facts = config_facts(path, source)
|
|
1223
|
+
if not facts:
|
|
1224
|
+
return None
|
|
1225
|
+
return {"path": path, "kind": "config", "package": "", "imports": [], "types": [],
|
|
1226
|
+
"functions": [], "facts": facts}
|
|
1227
|
+
|
|
1228
|
+
|
|
1229
|
+
def render(rows: list[dict], files: list[str], limit: int = 12000, spread_files: bool = False,
|
|
1230
|
+
note: str = "") -> str:
|
|
1231
|
+
"""The repository map: every visible file, with its declarations under it.
|
|
1232
|
+
|
|
1233
|
+
`note` goes on the first line, not the last: it says what the map does *not* contain, and a sentence
|
|
1234
|
+
about missing files that is itself truncated off the bottom would be a worse answer than none.
|
|
1235
|
+
"""
|
|
1236
|
+
indexed = {row["path"]: row for row in rows}
|
|
1237
|
+
edges = dependencies(rows)
|
|
1238
|
+
if spread_files:
|
|
1239
|
+
files = spread(files)
|
|
1240
|
+
out: list[str] = []
|
|
1241
|
+
if note:
|
|
1242
|
+
out.append(note)
|
|
1243
|
+
used = len(note) + 1 if note else 0
|
|
1244
|
+
shown = 0
|
|
1245
|
+
for name in files:
|
|
1246
|
+
row = indexed.get(name)
|
|
1247
|
+
block = [name]
|
|
1248
|
+
if row:
|
|
1249
|
+
# A configuration file's own lines: what it is, what it builds, what it needs. Printed under
|
|
1250
|
+
# the path like a package line, because that is the position the model already reads
|
|
1251
|
+
# structure from — a fact in prose elsewhere in the prompt is a fact it has to re-find.
|
|
1252
|
+
for label, value in row.get("facts") or []:
|
|
1253
|
+
block.append(" " + label + ": " + value)
|
|
1254
|
+
if row["package"]:
|
|
1255
|
+
block.append(" package " + row["package"])
|
|
1256
|
+
for item in row["types"]:
|
|
1257
|
+
text = " " + item["kind"] + " " + item["name"]
|
|
1258
|
+
if item.get("extends"):
|
|
1259
|
+
text += " extends " + ", ".join(item["extends"])
|
|
1260
|
+
if item["members"]:
|
|
1261
|
+
text += ": " + ", ".join(item["members"][:8])
|
|
1262
|
+
if len(item["members"]) > 8:
|
|
1263
|
+
text += " +" + str(len(item["members"]) - 8) + " more"
|
|
1264
|
+
block.append(text[:200])
|
|
1265
|
+
if row["functions"]:
|
|
1266
|
+
block.append(" functions: " + ", ".join(row["functions"][:8]) +
|
|
1267
|
+
(" …" if len(row["functions"]) > 8 else ""))
|
|
1268
|
+
if edges.get(name):
|
|
1269
|
+
block.append(" depends on: " + ", ".join(edges[name]))
|
|
1270
|
+
block.append("")
|
|
1271
|
+
text = "\n".join(block)
|
|
1272
|
+
if used + len(text) > limit:
|
|
1273
|
+
break
|
|
1274
|
+
out.append(text)
|
|
1275
|
+
used += len(text)
|
|
1276
|
+
shown += 1
|
|
1277
|
+
if shown < len(files):
|
|
1278
|
+
# Which modules were left out entirely, because "300 of 1 200 files" does not tell a reader
|
|
1279
|
+
# whether the missing ones are a detail or the half of the project they are asking about.
|
|
1280
|
+
missing = sorted({module_of(name) for name in files[shown:]}
|
|
1281
|
+
- {module_of(name) for name in files[:shown]})
|
|
1282
|
+
out.append("(index truncated: " + str(shown) + " of " + str(len(files)) + " files listed"
|
|
1283
|
+
+ ("; nothing shown from " + ", ".join(missing[:6])
|
|
1284
|
+
+ (" …" if len(missing) > 6 else "") if missing else "")
|
|
1285
|
+
+ "; read or search the rest on demand)")
|
|
1286
|
+
return "\n".join(out).strip()
|