ai-code-engineer 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. ai_code_engineer/__init__.py +2 -0
  2. ai_code_engineer/catalog.py +143 -0
  3. ai_code_engineer/chat.py +181 -0
  4. ai_code_engineer/cli.py +384 -0
  5. ai_code_engineer/config.py +405 -0
  6. ai_code_engineer/engine.py +1282 -0
  7. ai_code_engineer/errors.py +27 -0
  8. ai_code_engineer/git_integration.py +443 -0
  9. ai_code_engineer/gui.py +2646 -0
  10. ai_code_engineer/host.py +81 -0
  11. ai_code_engineer/ignore.py +269 -0
  12. ai_code_engineer/intent.py +222 -0
  13. ai_code_engineer/labels.py +871 -0
  14. ai_code_engineer/memory.py +91 -0
  15. ai_code_engineer/modes.py +156 -0
  16. ai_code_engineer/overrides.py +540 -0
  17. ai_code_engineer/planbook.py +192 -0
  18. ai_code_engineer/providers.py +404 -0
  19. ai_code_engineer/redaction.py +54 -0
  20. ai_code_engineer/repair.py +564 -0
  21. ai_code_engineer/report.py +352 -0
  22. ai_code_engineer/runner.py +854 -0
  23. ai_code_engineer/setup.py +386 -0
  24. ai_code_engineer/symbols.py +1286 -0
  25. ai_code_engineer/verification.py +218 -0
  26. ai_code_engineer/webapp/__init__.py +1 -0
  27. ai_code_engineer/webapp/__main__.py +45 -0
  28. ai_code_engineer/webapp/contract.py +36 -0
  29. ai_code_engineer/webapp/controller.py +3556 -0
  30. ai_code_engineer/webapp/fake.py +1141 -0
  31. ai_code_engineer/webapp/launch.py +108 -0
  32. ai_code_engineer/webapp/server.py +349 -0
  33. ai_code_engineer/webapp/static/app.css +780 -0
  34. ai_code_engineer/webapp/static/app.js +2118 -0
  35. ai_code_engineer/webapp/static/boot.js +19 -0
  36. ai_code_engineer/webapp/static/index.html +89 -0
  37. ai_code_engineer/webapp/static/tokens.css +173 -0
  38. ai_code_engineer/workspace.py +385 -0
  39. ai_code_engineer-0.1.0.dist-info/METADATA +7 -0
  40. ai_code_engineer-0.1.0.dist-info/RECORD +44 -0
  41. ai_code_engineer-0.1.0.dist-info/WHEEL +5 -0
  42. ai_code_engineer-0.1.0.dist-info/entry_points.txt +2 -0
  43. ai_code_engineer-0.1.0.dist-info/licenses/LICENSE +21 -0
  44. ai_code_engineer-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1286 @@
1
+ """What each source file declares, so a small model stops guessing from file names.
2
+
3
+ The repo map used to be one regex over every line, which invented "symbols" out of
4
+ comments and string literals and missed structure a parser sees for free. Python is
5
+ parsed with `ast`. Java, Kotlin, TypeScript, Go and Rust get a bounded scanner over
6
+ comment/string-blanked text: enough to answer "which type owns this method, and what
7
+ does this file import" without pretending to be a compiler.
8
+
9
+ Nothing here executes project code, and every list is capped, because the result is
10
+ pasted into a model prompt whose budget is the whole point.
11
+ """
12
+ from __future__ import annotations
13
+
14
+ import ast
15
+ import re
16
+
17
+ MAX_FILES = 300
18
+ MAX_TYPES = 40 # declared types per file
19
+ MAX_MEMBERS = 25 # methods per type
20
+ MAX_IMPORTS = 40 # imports per file
21
+ MAX_SIGNATURE = 60 # characters of one signature
22
+ SCRIPT_SUFFIXES = {".js", ".jsx", ".ts", ".tsx", ".mjs", ".cjs"}
23
+ INDEXABLE = {".py", ".java", ".kt", ".go", ".rs"} | SCRIPT_SUFFIXES
24
+ # Kinds whose declared types are importable by their simple name from another file.
25
+ NAMED_KINDS = {"jvm", "script", "go", "rust"}
26
+
27
+ # A name in this set means the line is a statement, not a declaration.
28
+ CONTROL = {"if", "for", "while", "switch", "catch", "return", "do", "else", "try",
29
+ "finally", "throw", "new", "synchronized", "case", "when", "init", "super",
30
+ "this", "assert", "yield", "await", "print", "lambda", "var", "val"}
31
+ LINE_STARTS = ("public ", "private ", "protected ", "static ", "final ", "abstract ",
32
+ "default ", "synchronized ", "native ", "open ", "override ", "suspend ",
33
+ "internal ", "inline ", "operator ", "fun ", "void ", "class ",
34
+ "interface ", "enum ", "record ", "object ", "annotation ", "sealed ",
35
+ "data ", "companion ", "external ", "tailrec ")
36
+ TYPE_RE = re.compile(r"\b(class|interface|enum|record|object|annotation)\s+([A-Za-z_$][\w$]*)")
37
+ PACKAGE_RE = re.compile(r"^\s*package\s+(?:`)?([\w.$]+)")
38
+ IMPORT_RE = re.compile(r"^\s*import\s+(?:static\s+)?(?:`)?([\w.$]+)")
39
+ JAVA_METHOD_RE = re.compile(
40
+ r"^\s*(?:(?:public|private|protected|static|final|abstract|default|synchronized|native"
41
+ r"|strictfp|open|transient|volatile)\s+)+"
42
+ r"(?:([\w$][\w$<>\[\],.?\s]*?)\s+)?([A-Za-z_$][\w$]*)\s*\(([^()]*)\)\s*"
43
+ r"(?:throws [\w$.,\s]+)?[{;]")
44
+ KOTLIN_FUN_RE = re.compile(
45
+ r"^\s*(?:(?:public|private|protected|internal|abstract|open|override|suspend|inline"
46
+ r"|operator|tailrec|external|annotation)\s+)*fun\s+(?:<[^<>]*>\s*)?"
47
+ r"(?:([\w$][\w$.<>,\s]*)\s*\.\s*)?([A-Za-z_$][\w$]*)\s*\(([^()]*)\)")
48
+ # Package-private Java methods carry no modifier, which is how JUnit tests are written.
49
+ JAVA_PLAIN_METHOD_RE = re.compile(
50
+ r"^\s*(?:@[\w.]+\s+)*(?:final\s+|static\s+|abstract\s+|default\s+)*"
51
+ r"([\w$][\w$<>\[\],.?\s]*?)\s+([A-Za-z_$][\w$]*)\s*\(([^()]*)\)\s*"
52
+ r"(?:throws [\w$.,\s]+)?[{;]")
53
+
54
+ # ---------------- JavaScript / TypeScript / JSX ----------------
55
+ # `export` is a modifier here, never a second declaration: the same name is recorded once.
56
+ JS_PREFIX = r"^\s*(?:export\s+)?(?:default\s+)?(?:export\s+)?(?:declare\s+)?"
57
+ # A type-parameter / type-argument list, optional as a whole. One nesting level is enough for the
58
+ # clauses real code writes, a bare `[^>]*` would stop at the inner `>` of `Map<String, T>`, and the
59
+ # fill excludes `{` so the tail of a declaration line is never swallowed by an empty match.
60
+ JS_GENERIC = r"(?:<(?:[^<{]|<[^<>]*>)*>)?"
61
+ JS_CLASS_RE = re.compile(JS_PREFIX + r"class\s+([A-Za-z_$][\w$]*)\s*" + JS_GENERIC + r"\s*"
62
+ r"(?:extends\s+([A-Za-z_$][\w$.]*)\s*" + JS_GENERIC + r"\s*)?"
63
+ r"(?:implements\s+[^{]*)?\{?")
64
+ JS_TYPE_RE = re.compile(JS_PREFIX + r"(interface|enum|type)\s+([A-Za-z_$][\w$]*)")
65
+ JS_FUNC_RE = re.compile(JS_PREFIX + r"(?:async\s+)?function\s*\*?\s*([A-Za-z_$][\w$]*)\s*"
66
+ + JS_GENERIC + r"\s*\(([^()]*)\)")
67
+ JS_ARROW_RE = re.compile(JS_PREFIX + r"(?:const|let|var)\s+([A-Za-z_$][\w$]*)\s*=\s*"
68
+ r"(?:async\s*)?(?:\(([^()]*)\)|[A-Za-z_$][\w$]*)(?:\s*:[^=\n]+?)?\s*=>")
69
+ # A class member: `name(args) {`, with optional modifiers and an optional `: Type`.
70
+ JS_MEMBER_RE = re.compile(r"^\s*(?:(?:public|private|protected|static|readonly|abstract"
71
+ r"|override|async|declare|get|set|declare)\s+)*#?[A-Za-z_$][\w$]*"
72
+ r"\s*\(([^()]*)\)\s*(?::[^;{]+)?\{")
73
+ JS_MEMBER_NAME_RE = re.compile(r"^\s*(?:(?:public|private|protected|static|readonly|abstract"
74
+ r"|override|async|get|set|declare)\s+)*#?([A-Za-z_$][\w$]*)\s*\(")
75
+ JS_FROM_RE = re.compile(r"(?:^|;)\s*(?:import|export)\b[^;\n]*?\bfrom\s*['\"]([^'\"]+)['\"]")
76
+ JS_BARE_IMPORT_RE = re.compile(r"^\s*import\s*['\"]([^'\"]+)['\"]")
77
+ JS_REQUIRE_RE = re.compile(r"\brequire\(\s*['\"]([^'\"]+)['\"]\s*\)")
78
+
79
+ # ---------------- Go ----------------
80
+ GO_PACKAGE_RE = re.compile(r"^\s*package\s+([\w.]+)")
81
+ # func Name(...) and func (recv Type) Name(...) — the receiver decides the owning type.
82
+ GO_FUNC_RE = re.compile(r"^\s*func\s+(?:\(\s*(?:_\s+|\w+\s+)?\*?([\w\[\]]+)\s*\)\s*)?"
83
+ r"([A-Za-z_]\w*)\s*\(([^()]*)\)")
84
+ GO_TYPE_RE = re.compile(r"^\s*type\s+([A-Za-z_]\w*)\s+(struct|interface)\b")
85
+ GO_QUOTED_RE = re.compile(r'"([^"\n]+)"')
86
+ GO_IMPORT_RE = re.compile(r'^import\s+(?:[\w.]+\s+)?"([^"]+)"', re.M)
87
+ GO_IMPORT_BLOCK_RE = re.compile(r"^import\s*\(([^)]*)\)", re.M | re.S)
88
+
89
+ # ---------------- Rust ----------------
90
+ RS_TYPE_RE = re.compile(r"^\s*(?:pub(?:\([^)]*\))?\s+)?(struct|enum|trait|union|type|mod)\s+"
91
+ r"([A-Za-z_]\w*)")
92
+ RS_IMPL_RE = re.compile(r"^\s*(?:unsafe\s+)?impl\s*(?:<[^<>]*>\s*)?(?:([\w:]+)\s+for\s+)?"
93
+ r"([A-Za-z_][\w:]*)")
94
+ RS_FN_RE = re.compile(r"^\s*(?:(?:pub|crate)(?:\([^)]*\))?|default|const|unsafe|async"
95
+ r"|extern\s*(?:\([^)]*\))?|\s)*fn\s+([A-Za-z_]\w*)\s*(?:<[^<>]*>)?"
96
+ r"\s*\(([^()]*)\)")
97
+ RS_USE_RE = re.compile(r"^\s*(?:pub(?:\([^)]*\))?\s+)?use\s+([^;]+);")
98
+ RS_USE_BRACE_RE = re.compile(r"^([\w:]+)::\{([^}]*)\}$")
99
+
100
+
101
+ def _signature(name: str, params: list[str]) -> str:
102
+ text = re.sub(r"\s+", " ", name + "(" + ", ".join(p.strip() for p in params) + ")")
103
+ return text[:MAX_SIGNATURE] + "…" if len(text) > MAX_SIGNATURE else text
104
+
105
+
106
+ def blank(source: str, backticks: bool = False) -> str:
107
+ """Blank out comments and string contents, keeping every line where it was.
108
+
109
+ Without this, `// class Foo` becomes a class and a log message becomes a method.
110
+ `backticks` additionally blanks JavaScript template literals; a stray backtick in a
111
+ regex literal is not allowed to swallow the rest of the file, so an unterminated one
112
+ only reaches the end of its own line.
113
+ """
114
+ out, i, n = [], 0, len(source)
115
+ while i < n:
116
+ ch = source[i]
117
+ pair = source[i:i + 2]
118
+ if pair == "//":
119
+ end = source.find("\n", i)
120
+ end = n if end < 0 else end
121
+ out.append(" " * (end - i))
122
+ i = end
123
+ elif pair == "/*":
124
+ end = source.find("*/", i + 2)
125
+ end = n if end < 0 else end + 2
126
+ out.append("".join(c if c == "\n" else " " for c in source[i:end]))
127
+ i = end
128
+ elif backticks and ch == "`":
129
+ end = source.find("`", i + 1)
130
+ if end < 0:
131
+ # A stray backtick — inside a regex literal, say — may not eat the file.
132
+ end = source.find("\n", i + 1)
133
+ end = n if end < 0 else end
134
+ out.append(" " * (end - i))
135
+ i = end
136
+ else:
137
+ out.append("".join(c if c == "\n" else " " for c in source[i:end + 1]))
138
+ i = end + 1
139
+ elif source.startswith('"""', i) or source.startswith("'''", i):
140
+ quote = source[i:i + 3]
141
+ end = source.find(quote, i + 3)
142
+ end = n if end < 0 else end + 3
143
+ out.append("".join(c if c == "\n" else " " for c in source[i:end]))
144
+ i = end
145
+ elif ch in "\"'":
146
+ end = i + 1
147
+ while end < n:
148
+ if source[end] == "\\":
149
+ end += 2
150
+ continue
151
+ if source[end] in (ch, "\n"):
152
+ break
153
+ end += 1
154
+ out.append(" " * (end - i))
155
+ # Step over the closing quote, or it reads as the opening one of a second
156
+ # string and swallows the rest of the line.
157
+ i = end + 1 if end < n and source[end] == ch else end
158
+ else:
159
+ out.append(ch)
160
+ i += 1
161
+ return "".join(out)
162
+
163
+
164
+ def _jvm(path: str, source: str) -> dict:
165
+ """Java/Kotlin declarations, found by a brace-depth scan of blanked text.
166
+
167
+ A method counts only at one level inside a type body, which is what separates
168
+ `public Token login(String p)` from `tokenService.login(password)` two lines later.
169
+ """
170
+ row = {"path": path, "kind": "jvm", "package": "", "imports": [], "types": [],
171
+ "functions": []}
172
+ depth = 0
173
+ stack: list[tuple[str, int]] = [] # (type name, depth where it opened)
174
+ for number, line in enumerate(blank(source).splitlines(), 1):
175
+ head = line.strip()
176
+ opened, closed = line.count("{"), line.count("}")
177
+ if head.startswith("package") and not row["package"]:
178
+ match = PACKAGE_RE.match(line)
179
+ if match:
180
+ row["package"] = match.group(1)
181
+ elif head.startswith("import"):
182
+ match = IMPORT_RE.match(line)
183
+ if match and len(row["imports"]) < MAX_IMPORTS and match.group(1) not in row["imports"]:
184
+ row["imports"].append(match.group(1))
185
+ declared = TYPE_RE.findall(line)
186
+ if declared:
187
+ for kind, simple in declared:
188
+ outer = next((name for name, level in reversed(stack) if level == depth), "")
189
+ name = outer + "." + simple if outer else simple
190
+ _declare_type(row, simple, kind, number, outer)
191
+ # The frame opens even when the cap refused the type: without it a method
192
+ # inside has no owner at all and lands in the file's own function list.
193
+ stack.append((name, depth))
194
+ elif any(head.startswith(word) for word in LINE_STARTS):
195
+ method = None
196
+ if re.search(r"\bfun\s", line):
197
+ method = KOTLIN_FUN_RE.match(line)
198
+ method = method or JAVA_METHOD_RE.match(line) or JAVA_PLAIN_METHOD_RE.match(line)
199
+ if method and method.group(2) not in CONTROL:
200
+ params = [part for part in method.group(3).split(",") if part.strip()][:6]
201
+ if stack:
202
+ owner = next((item for item in reversed(row["types"])
203
+ if item["name"] == stack[-1][0]), None)
204
+ if owner and depth == stack[-1][1] + 1 and len(owner["members"]) < MAX_MEMBERS:
205
+ owner["members"].append(_signature(method.group(2), params))
206
+ elif len(row["functions"]) < MAX_MEMBERS:
207
+ # A Kotlin file-level function, which belongs to no type.
208
+ row["functions"].append(_signature(method.group(2), params))
209
+ depth += opened - closed
210
+ while stack and depth <= stack[-1][1]:
211
+ stack.pop()
212
+ return row
213
+
214
+
215
+ def _row(path: str, kind: str) -> dict:
216
+ return {"path": path, "kind": kind, "package": "", "imports": [], "types": [],
217
+ "functions": []}
218
+
219
+
220
+ def _add_import(row: dict, name: str) -> None:
221
+ name = name.strip().strip("\"'")
222
+ if name and name not in row["imports"] and len(row["imports"]) < MAX_IMPORTS:
223
+ row["imports"].append(name)
224
+
225
+
226
+ def _declare_type(row: dict, name: str, kind: str, line: int, outer: str = "") -> dict | None:
227
+ """The entry for a declared type, created once per name.
228
+
229
+ `export class X` and a later `export { X }`, or a Go type with its methods spread over
230
+ several files in the same folder, must not produce the same type twice.
231
+ """
232
+ full = outer + "." + name if outer else name
233
+ for item in row["types"]:
234
+ if item["name"] == full:
235
+ return item
236
+ if len(row["types"]) >= MAX_TYPES:
237
+ return None
238
+ item = {"name": full, "kind": kind, "line": line, "members": []}
239
+ row["types"].append(item)
240
+ return item
241
+
242
+
243
+ def _declare(row: dict, owner: dict | None, name: str, params: str) -> None:
244
+ text = _signature(name, [part for part in params.split(",") if part.strip()][:6])
245
+ bucket = owner["members"] if owner is not None else row["functions"]
246
+ if text not in bucket and len(bucket) < MAX_MEMBERS:
247
+ bucket.append(text)
248
+
249
+
250
+ def _specifier(raw: str) -> str:
251
+ """A JavaScript module specifier in the dotted form `dependencies()` compares against.
252
+
253
+ "./text" means this file's own folder, which is exactly what a single leading dot
254
+ means to the Python rules already in place, so one dot is dropped, not kept.
255
+ """
256
+ text = raw.strip().strip("/")
257
+ if text.startswith("."):
258
+ return text.replace("/", ".")[1:]
259
+ return text.replace("/", ".")
260
+
261
+
262
+ def _use_targets(clause: str) -> list[str]:
263
+ """Rust `use` paths, without the crate root and with `::` read as `.`."""
264
+ def one(text: str) -> str:
265
+ path = text.strip().split(" as ")[0].replace("::", ".")
266
+ for root in ("crate.", "self.", "super."):
267
+ if path.startswith(root):
268
+ # `super` is one level up, exactly as a leading dot reads in Python.
269
+ return "." + path[len(root):] if root == "super." else path[len(root):]
270
+ return path
271
+
272
+ brace = RS_USE_BRACE_RE.match(clause.strip())
273
+ if brace:
274
+ return [one(brace.group(1) + "::" + item.strip())
275
+ for item in brace.group(2).split(",") if item.strip()][:MAX_IMPORTS]
276
+ return [one(clause)] if clause.strip() else []
277
+
278
+
279
+ def _go_imports(source: str) -> list[str]:
280
+ """Every import path in a Go file, single-line and grouped.
281
+
282
+ Read from the original text because the blanker erases string contents, and that is
283
+ where the path lives; a line whose own text starts with `//` is a comment, not a path.
284
+ """
285
+ found = GO_IMPORT_RE.findall(source)
286
+ for block in GO_IMPORT_BLOCK_RE.findall(source):
287
+ for line in block.splitlines():
288
+ text = line.strip()
289
+ if not text or text.startswith("//"):
290
+ continue
291
+ match = GO_QUOTED_RE.search(text)
292
+ if match:
293
+ found.append(match.group(1))
294
+ return found
295
+
296
+
297
+ def _script(path: str, source: str) -> dict:
298
+ """TypeScript and JavaScript declarations.
299
+
300
+ Import specifiers live inside string literals, which the blanker erases, so every line
301
+ is read twice: the blanked copy decides whether a declaration is really there, the
302
+ original copy supplies the module path.
303
+ """
304
+ row = _row(path, "script")
305
+ blanked = blank(source, backticks=True).splitlines()
306
+ original = source.splitlines()
307
+ depth = 0
308
+ stack: list[tuple[str, int]] = []
309
+ for number, (line, raw) in enumerate(zip(blanked, original), 1):
310
+ head = line.strip()
311
+ opened, closed = line.count("{"), line.count("}")
312
+ declared = None # a class body opened by this line
313
+ if head.startswith(("import", "export")) or "require(" in head:
314
+ found = JS_FROM_RE.findall(raw) + JS_BARE_IMPORT_RE.findall(raw)
315
+ if "require(" in line:
316
+ found += JS_REQUIRE_RE.findall(raw)
317
+ for specifier in found:
318
+ _add_import(row, _specifier(specifier))
319
+ match = JS_CLASS_RE.match(line)
320
+ if match and match.group(1) not in CONTROL:
321
+ outer = next((name for name, level in reversed(stack) if level == depth), "")
322
+ owner = _declare_type(row, match.group(1), "class", number, outer)
323
+ if owner is not None:
324
+ if match.group(2) and not owner.get("extends"):
325
+ owner["extends"] = [match.group(2)[:40]]
326
+ if opened > closed:
327
+ declared = owner["name"]
328
+ elif (match := JS_TYPE_RE.match(line)):
329
+ _declare_type(row, match.group(2), match.group(1), number)
330
+ elif (match := JS_FUNC_RE.match(line)) or (match := JS_ARROW_RE.match(line)):
331
+ _declare(row, _owner_of(row, stack, depth), match.group(1), match.group(2) or "")
332
+ elif "(" in line and line.rstrip().endswith("{"):
333
+ member = JS_MEMBER_RE.match(line)
334
+ owner = _owner_of(row, stack, depth)
335
+ if member and owner is not None:
336
+ _declare(row, owner, JS_MEMBER_NAME_RE.match(line).group(1), member.group(1))
337
+ depth += opened - closed
338
+ if declared:
339
+ stack.append((declared, depth - (opened - closed)))
340
+ while stack and depth <= stack[-1][1]:
341
+ stack.pop()
342
+ return row
343
+
344
+
345
+ def _owner_of(row: dict, stack: list[tuple[str, int]], depth: int) -> dict | None:
346
+ """The type one level above this line, or None at file level.
347
+
348
+ A function at that level is a method; one deeper is inside a body and is not.
349
+ """
350
+ if not stack or depth != stack[-1][1] + 1:
351
+ return None
352
+ return next((item for item in reversed(row["types"]) if item["name"] == stack[-1][0]), None)
353
+
354
+
355
+ def _go(path: str, source: str) -> dict:
356
+ """Go declarations: a receiver picks the owning type, a package names the folder."""
357
+ row = _row(path, "go")
358
+ for imported in _go_imports(source)[:MAX_IMPORTS]:
359
+ _add_import(row, imported)
360
+ for number, line in enumerate(blank(source).splitlines(), 1):
361
+ head = line.strip()
362
+ if not head:
363
+ continue # a commented-out func is not a declaration
364
+ if head.startswith("package") and not row["package"]:
365
+ match = GO_PACKAGE_RE.match(line)
366
+ if match:
367
+ row["package"] = match.group(1)
368
+ elif head.startswith("type"):
369
+ match = GO_TYPE_RE.match(line)
370
+ if match:
371
+ _declare_type(row, match.group(1), match.group(2), number)
372
+ elif head.startswith("func"):
373
+ match = GO_FUNC_RE.match(line)
374
+ if match:
375
+ receiver, name, params = match.groups()
376
+ owner = (_declare_type(row, receiver.strip("*[]"), "type", number)
377
+ if receiver else None)
378
+ _declare(row, owner, name, params)
379
+ return row
380
+
381
+
382
+ def _rust(path: str, source: str) -> dict:
383
+ """Rust declarations: `impl Trait for Type` and `trait` bodies own their functions."""
384
+ row = _row(path, "rust")
385
+ blanked = blank(source).splitlines()
386
+ original = source.splitlines()
387
+ depth = 0
388
+ owner: dict | None = None
389
+ owner_depth = 0
390
+ for number, (line, raw) in enumerate(zip(blanked, original), 1):
391
+ head = line.strip()
392
+ opened, closed = line.count("{"), line.count("}")
393
+ entered = None
394
+ if head.startswith("use "):
395
+ clause = RS_USE_RE.match(raw)
396
+ if clause:
397
+ for target in _use_targets(clause.group(1)):
398
+ _add_import(row, target)
399
+ elif head.startswith("impl"):
400
+ match = RS_IMPL_RE.match(line)
401
+ if match:
402
+ entered = _declare_type(row, match.group(2).split("::")[-1], "impl", number)
403
+ elif "fn " in line:
404
+ match = RS_FN_RE.match(line)
405
+ if match:
406
+ _declare(row, owner if depth >= owner_depth else None,
407
+ match.group(1), match.group(2))
408
+ else:
409
+ match = RS_TYPE_RE.match(line)
410
+ if match:
411
+ entered = _declare_type(row, match.group(2), match.group(1), number)
412
+ depth += opened - closed
413
+ if entered is not None and opened > closed:
414
+ owner, owner_depth = entered, depth
415
+ elif owner is not None and depth < owner_depth:
416
+ owner, owner_depth = None, 0
417
+ return row
418
+
419
+
420
+ def _params(node: ast.arguments) -> list[str]:
421
+ names = [arg.arg for arg in node.posonlyargs + node.args]
422
+ if names and names[0] in {"self", "cls"}:
423
+ names = names[1:]
424
+ names += [arg.arg for arg in node.kwonlyargs]
425
+ if node.vararg:
426
+ names.append("*" + node.vararg.arg)
427
+ if node.kwarg:
428
+ names.append("**" + node.kwarg.arg)
429
+ return names[:6]
430
+
431
+
432
+ def _python(path: str, source: str) -> dict | None:
433
+ try:
434
+ tree = ast.parse(source)
435
+ except (SyntaxError, ValueError, MemoryError, RecursionError):
436
+ return None
437
+ row = {"path": path, "kind": "python", "package": "", "imports": [], "types": [],
438
+ "functions": []}
439
+
440
+ def walk(body, prefix: str, owner: dict | None) -> None:
441
+ for node in body:
442
+ if isinstance(node, ast.ClassDef):
443
+ if len(row["types"]) >= MAX_TYPES:
444
+ return
445
+ item = {"name": prefix + node.name, "kind": "class", "line": node.lineno,
446
+ "extends": [ast.unparse(base)[:40] for base in node.bases][:4],
447
+ "members": []}
448
+ row["types"].append(item)
449
+ walk(node.body, prefix + node.name + ".", item)
450
+ elif isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)):
451
+ text = _signature(node.name, _params(node.args))
452
+ if owner is not None and len(owner["members"]) < MAX_MEMBERS:
453
+ owner["members"].append(text)
454
+ elif owner is None and len(row["functions"]) < MAX_MEMBERS:
455
+ row["functions"].append(text)
456
+ elif isinstance(node, ast.Import):
457
+ row["imports"].extend(alias.name for alias in node.names)
458
+ elif isinstance(node, ast.ImportFrom):
459
+ base = "." * node.level + (node.module or "")
460
+ row["imports"].extend(base + "." + alias.name for alias in node.names)
461
+ elif isinstance(node, (ast.If, ast.Try, ast.With, ast.For, ast.While)):
462
+ walk(node.body, prefix, owner)
463
+
464
+ walk(tree.body, "", None)
465
+ row["imports"] = list(dict.fromkeys(row["imports"]))[:MAX_IMPORTS]
466
+ return row
467
+
468
+
469
+ def fallback(path: str, source: str) -> dict:
470
+ """The old lexical scan, kept for files a real parser refuses to read."""
471
+ names = re.findall(r"(?m)^\s*(?:(?:public|private|protected|abstract|final|static)\s+)*"
472
+ r"(?:class|interface|enum|record|def|fun)\s+(\w+)", source)
473
+ return {"path": path, "kind": "lexical", "package": "", "imports": [], "functions": [],
474
+ "types": [{"name": name, "kind": "lexical", "line": 0, "members": []}
475
+ for name in names[:MAX_TYPES]]}
476
+
477
+
478
+ def suffix_of(path: str) -> str:
479
+ parts = path.casefold().rsplit(".", 1)
480
+ return "." + parts[-1] if len(parts) == 2 else ""
481
+
482
+
483
+ # The files that say what a project *is* and how it runs. Not source, and not everything that is not
484
+ # source: a map that listed every YAML file would be a file listing again. This is the short set a
485
+ # model needs in order to know a reactor's modules, a service's port and a package's dependencies
486
+ # without spending three of its twelve turns reading them.
487
+ CONFIG_NAMES = {"pom.xml", "build.gradle", "build.gradle.kts", "settings.gradle",
488
+ "settings.gradle.kts", "package.json", "go.mod", "cargo.toml", "pyproject.toml"}
489
+ CONFIG_PREFIXES = ("application", "bootstrap")
490
+ CONFIG_SUFFIXES = {".yml", ".yaml", ".properties"}
491
+ MAX_FACTS = 6
492
+ MAX_FACT_CHARS = 160
493
+ # A key's *name* decides whether its value may be shown. `server.port` is what a starting order needs;
494
+ # `spring.datasource.password` is not, and a map that carried it would also carry it into
495
+ # `agent export-session`, which is the one artefact that leaves the machine.
496
+ # `user` is here because a leaf key can be allow-listed while the path above it is a credential store:
497
+ # `spring.security.user.name` and `spring.security.user.password` both sit under a login.
498
+ CREDENTIAL_KEY = re.compile(r"(?i)(pass(word|wd)?|secret|token|key|credential|cert(ificate)?|user)")
499
+ # Matched against the whole dotted path, so a key only means what its owner says it means. A bare
500
+ # `name` under `logging.file` is a log file, not the application, and a map that said otherwise would
501
+ # be a wrong fact the model has no way to doubt.
502
+ KEY_FACTS = {"port": "port", "server.port": "port", "management.server.port": "management port",
503
+ "name": "name", "application.name": "name", "spring.application.name": "name",
504
+ "artifactid": "artifact"}
505
+ # The three leaves that say the same thing whoever owns them, so the flat `spring.datasource.url=…`
506
+ # spelling of a properties file reads the same as the nested YAML one.
507
+ LEAF_FACTS = {"port": "port", "defaultzone": "registry", "url": "host"}
508
+ # A URL's authority, with the user and the password taken off the front of it. The leading group is
509
+ # repeatable because JDBC nests its own scheme -- `jdbc:postgresql://user:pw@host:5432/db` is one of the
510
+ # most common lines in a Spring project, and a single-scheme pattern read it as no host at all.
511
+ URL_HOST = re.compile(r"^(?:[\w+.\-]+:)*//(?:[^@/]+@)?([\w.\-]+(?::\d+)?)")
512
+ # The same authority with no scheme in front of it (`localhost:8761/eureka`). It cannot carry a user and
513
+ # a password, because those need a scheme to sit after.
514
+ BARE_HOST = re.compile(r"^([\w.\-]+(?::\d+)?)(?:[/?].*)?$")
515
+
516
+
517
+ def noteworthy(path: str) -> bool:
518
+ """Whether this file's *facts* belong in the map, even though it is not code."""
519
+ name = str(path).rsplit("/", 1)[-1]
520
+ if name.casefold() in CONFIG_NAMES:
521
+ return True
522
+ folded = name.casefold()
523
+ suffix = "." + folded.rsplit(".", 1)[-1] if "." in folded else ""
524
+ return folded.startswith(CONFIG_PREFIXES) and suffix in CONFIG_SUFFIXES
525
+
526
+
527
+ def _shorten(label: str, value: str) -> tuple[str, str] | None:
528
+ value = re.sub(r"\s+", " ", str(value)).strip()
529
+ if not value or CREDENTIAL_KEY.search(label):
530
+ return None
531
+ if len(value) > MAX_FACT_CHARS:
532
+ # A dependency list cut mid-token leaves a name the repository does not contain, and the model
533
+ # goes looking for it. Prefer the last whole entry, and say that the list continues. The marker
534
+ # is inside the cap, not after it: `render()` prints these verbatim, and a fact that ran four
535
+ # characters long would be the one thing in the map to exceed its own limit.
536
+ room = MAX_FACT_CHARS - 5
537
+ if (cut := value.rfind(", ", 0, room)) > 0:
538
+ return (label, value[:cut] + ", ...")
539
+ return (label, value[:room].rstrip() + " ...")
540
+ return (label, value)
541
+
542
+
543
+ def _pom(source: str) -> list[tuple[str, str]]:
544
+ """Maven's own answers: what this module is, what it builds, and what it needs.
545
+
546
+ A `<!DOCTYPE` is refused outright rather than parsed. Python's ElementTree resolves no external
547
+ entities, but it still expands an internal DTD, and this file comes from a repository the tool was
548
+ asked to read, not from a build system it trusts — a few kilobytes of nested entities would cost
549
+ the whole task. Real POMs carry no DOCTYPE, so nothing is lost by the rule; `engine` refuses the
550
+ same shape when it checks a POM it is about to write.
551
+ """
552
+ import xml.etree.ElementTree as ET
553
+
554
+ if re.search(r"<!\s*DOCTYPE", source, re.I):
555
+ return []
556
+ try:
557
+ root = ET.fromstring(source)
558
+ except ET.ParseError:
559
+ return []
560
+ # Namespaces are removed from the parsed tree instead of from the text. Stripping the `xmlns…=`
561
+ # attributes with a regex was the first attempt and it failed on every real POM: a POM binds
562
+ # `xmlns:xsi` only to declare `xsi:schemaLocation`, so deleting the declaration leaves the
563
+ # attribute's prefix unbound and the parse dies before it starts.
564
+ for node in root.iter():
565
+ if isinstance(node.tag, str) and "}" in node.tag:
566
+ node.tag = node.tag.rsplit("}", 1)[-1]
567
+
568
+ def text(tag: str, node=None):
569
+ node = root if node is None else node
570
+ found = node.find(tag)
571
+ return found.text if found is not None and found.text else ""
572
+
573
+ facts = []
574
+ if (owner := root.find("parent")) is not None:
575
+ if named := _shorten("parent", text("artifactId", owner)):
576
+ facts.append(named)
577
+ if named := _shorten("artifact", text("artifactId")):
578
+ facts.append(named)
579
+ modules = [str(item.text).strip() for item in root.findall("modules/module")
580
+ if item.text and str(item.text).strip()]
581
+ if modules:
582
+ facts.append(("modules", ", ".join(modules[:12])))
583
+ deps = sorted({str(item.text).strip() for item in root.findall("dependencies/dependency/artifactId")
584
+ if item.text and str(item.text).strip()})
585
+ if deps:
586
+ facts.append(("needs", ", ".join(deps[:12])))
587
+ return [fact for fact in (_shorten(label, value) for label, value in facts) if fact][:MAX_FACTS]
588
+
589
+
590
+ GRADLE_DEP = re.compile(r"""(?:implementation|api|testImplementation|compileOnly|runtimeOnly|kapt|
591
+ annotationProcessor|classpath)\s*\(?\s*["']([\w.\-]+:[\w.\-]+)""", re.X)
592
+ GRADLE_INCLUDE = re.compile(r"""include\s+[\s,]*(['"]([\w.\-:]+)['"](?:\s*,\s*['"]([\w.\-:]+)['"])*)""")
593
+ # A module's own siblings. `implementation project(':common-lib')` is the internal edge the map exists
594
+ # to show, and it is spelled with no group and no version, so the coordinate pattern above cannot see it.
595
+ GRADLE_PROJECT = re.compile(r"""project\s*\(\s*['"]:?([\w.\-]+)['"]""")
596
+ PACKAGE_NAME = re.compile(r"""^\s*(?:(?:const|let|var)\s+)?rootProject\.name\s*=\s*['"]([\w.\-]+)""",
597
+ re.M)
598
+
599
+
600
+ def _gradle(source: str) -> list[tuple[str, str]]:
601
+ facts = []
602
+ if named := PACKAGE_NAME.search(source):
603
+ facts.append(("artifact", named.group(1)))
604
+ includes = set()
605
+ for match in GRADLE_INCLUDE.finditer(source):
606
+ for group in match.groups():
607
+ for piece in (group or "").split(","):
608
+ piece = piece.strip().strip("'\"").strip(":")
609
+ if piece:
610
+ includes.add(piece)
611
+ if includes:
612
+ facts.append(("modules", ", ".join(sorted(includes)[:12])))
613
+ deps = sorted({match.group(1).split(":")[-2] if match.group(1).count(":") >= 2
614
+ else match.group(1).split(":")[-1] for match in GRADLE_DEP.finditer(source)}
615
+ | {match.group(1) for match in GRADLE_PROJECT.finditer(source)})
616
+ if deps:
617
+ facts.append(("needs", ", ".join(deps[:12])))
618
+ return [fact for fact in (_shorten(label, value) for label, value in facts) if fact][:MAX_FACTS]
619
+
620
+
621
+ def _package_json(source: str) -> list[tuple[str, str]]:
622
+ import json
623
+
624
+ try:
625
+ data = json.loads(source)
626
+ except ValueError:
627
+ return []
628
+ if not isinstance(data, dict):
629
+ return []
630
+ facts = []
631
+ if isinstance(data.get("name"), str):
632
+ facts.append(("artifact", data["name"]))
633
+ scripts = [key for key in ("build", "test", "start", "lint") if key in (data.get("scripts") or {})]
634
+ if scripts:
635
+ facts.append(("scripts", ", ".join(scripts)))
636
+ for bucket in ("dependencies", "devDependencies"):
637
+ keys = data.get(bucket)
638
+ if isinstance(keys, dict) and keys:
639
+ facts.append((("needs" if bucket == "dependencies" else "dev-needs"),
640
+ ", ".join(sorted(keys)[:12])))
641
+ return [fact for fact in (_shorten(label, value) for label, value in facts) if fact][:MAX_FACTS]
642
+
643
+
644
+ GO_MODULE = re.compile(r"^\s*module\s+(\S+)", re.M)
645
+ GO_REQUIRE = re.compile(r"^\s*(?:require|use)\s+(\S+)")
646
+ GO_BLOCK_ITEM = re.compile(r"^\s*(\S+)\s+\S")
647
+
648
+
649
+ def _go_mod(source: str) -> list[tuple[str, str]]:
650
+ """The module this is and the modules it needs, in both of go.mod's two spellings.
651
+
652
+ A `require x v1` line and a `require ( x v1 )` block are the same statement written two ways, and
653
+ the block is the usual way for anything with more than a couple of dependencies -- reading only the
654
+ line form left a real project's manifest claiming it needed nothing.
655
+ """
656
+ facts = []
657
+ if named := GO_MODULE.search(source):
658
+ facts.append(("artifact", named.group(1)))
659
+ needs, block = set(), False
660
+ for line in source.splitlines():
661
+ stripped = line.strip()
662
+ if not stripped or stripped.startswith("//"):
663
+ continue
664
+ if block:
665
+ if stripped.startswith(")"):
666
+ block = False
667
+ continue
668
+ if (item := GO_BLOCK_ITEM.match(stripped)):
669
+ needs.add(item.group(1))
670
+ continue
671
+ match = GO_REQUIRE.match(line)
672
+ if not match:
673
+ continue
674
+ if match.group(1) == "(":
675
+ block = True
676
+ continue
677
+ needs.add(match.group(1))
678
+ keep = sorted(item for item in needs if "." in item) # a host in front of the path, not stdlib
679
+ if keep:
680
+ facts.append(("needs", ", ".join(keep[:12])))
681
+ return [fact for fact in (_shorten(label, value) for label, value in facts) if fact][:MAX_FACTS]
682
+
683
+
684
+ def _toml(source: str) -> list[tuple[str, str]]:
685
+ """Both TOML manifests this tool meets: PEP 621's `[project]` and Cargo's `[package]`.
686
+
687
+ They disagree in shape as well as in section name -- Python lists dependencies, Rust keys them --
688
+ so each half is read if it is there and the file simply has fewer facts if it is not.
689
+ """
690
+ import tomllib
691
+
692
+ try:
693
+ data = tomllib.loads(source)
694
+ except (ValueError, tomllib.TOMLDecodeError):
695
+ return []
696
+ tables = [data.get(key) for key in ("project", "package")
697
+ if isinstance(data.get(key), dict)]
698
+ facts = []
699
+ named = next((str(table["name"]) for table in tables if isinstance(table.get("name"), str)), "")
700
+ if named:
701
+ facts.append(("artifact", named))
702
+ needs: list[str] = []
703
+ for table in tables:
704
+ found = table.get("dependencies")
705
+ if isinstance(found, dict):
706
+ needs.extend(found.keys())
707
+ elif isinstance(found, list):
708
+ needs.extend(str(item) for item in found)
709
+ if isinstance(data.get("dependencies"), dict):
710
+ needs.extend(data["dependencies"].keys())
711
+ if needs:
712
+ names = sorted({str(item).split(";")[0].strip().split("[")[0].split("=")[0].split(">")[0]
713
+ .split("<")[0].split("!")[0].strip() for item in needs if str(item).strip()})
714
+ names = [item for item in names if item]
715
+ if names:
716
+ facts.append(("needs", ", ".join(names[:12])))
717
+ return [fact for fact in (_shorten(label, value) for label, value in facts) if fact][:MAX_FACTS]
718
+
719
+
720
+ # The separator is required and the value is not, because a parent line (`application:`) carries
721
+ # nothing and still owns the keys under it -- without it in the stack every child resolves against the
722
+ # last *valued* key, and `spring.application.name` arrives as `port.name`. `=` is accepted too: a
723
+ # `.properties` file writes `server.port=8080`, its keys are already dotted and all sit at indent zero,
724
+ # so one scan reads both dialects.
725
+ YAML_PAIR = re.compile(r"^\s*([A-Za-z][\w.\-]*)\s*[=:](?:\s*(\S.*?))?\s*$")
726
+
727
+
728
+ def _runtime_yml(source: str) -> list[tuple[str, str]]:
729
+ """A port, an application name, and the host a service registers with.
730
+
731
+ Line-oriented on purpose: YAML has no stdlib parser here, and the keys worth showing are the ones
732
+ that appear on their own line in every Spring Boot file this tool has been pointed at. A value is
733
+ only kept when its whole dotted key is on the allow-list, and a URL is reduced to its authority,
734
+ so the credential that fits inside the connection string never reaches the map.
735
+ """
736
+ facts = {}
737
+ stack: list[tuple[int, str]] = []
738
+ for line in source.splitlines():
739
+ match = YAML_PAIR.match(line)
740
+ if not match:
741
+ continue
742
+ key = match.group(1)
743
+ indent = len(line) - len(line.lstrip(" "))
744
+ while stack and stack[-1][0] >= indent:
745
+ stack.pop()
746
+ stack.append((indent, key))
747
+ # An inline comment is not part of the value; YAML only starts one after a space, so a `#` inside
748
+ # a URL fragment survives.
749
+ value = re.sub(r"\s+#.*$", "", (match.group(2) or "")).strip().strip("'\"")
750
+ if not value:
751
+ continue
752
+ path = ".".join(part for _, part in stack).casefold()
753
+ label = KEY_FACTS.get(path) or LEAF_FACTS.get(path.rsplit(".", 1)[-1])
754
+ if not label or CREDENTIAL_KEY.search(path):
755
+ continue
756
+ if label in ("registry", "host"):
757
+ host, plain = URL_HOST.match(value), BARE_HOST.match(value)
758
+ kept = (host or plain).group(1) if (host or plain) else ""
759
+ if kept:
760
+ facts.setdefault(label, kept)
761
+ elif len(value) <= 60:
762
+ facts.setdefault(label, value)
763
+ return [(label, value) for label, value in facts.items()][:MAX_FACTS]
764
+
765
+
766
+ def config_facts(path: str, source: str) -> list[tuple[str, str]]:
767
+ """What a configuration file says about the project, with nothing secret in it.
768
+
769
+ Failures are silent by design: an unparseable pom is a map without that line, not a task that
770
+ cannot start. The values that survive are names and numbers — module, artifact, port, dependency —
771
+ and any key whose *name* looks like a credential is dropped before it is considered, because the
772
+ repository's own redaction cannot see a value that was never supposed to be in the prompt.
773
+ """
774
+ name = str(path).rsplit("/", 1)[-1].casefold()
775
+ if name == "pom.xml":
776
+ return _pom(source)
777
+ if name.startswith("build.gradle") or name.startswith("settings.gradle"):
778
+ return _gradle(source)
779
+ if name == "package.json":
780
+ return _package_json(source)
781
+ if name == "go.mod":
782
+ return _go_mod(source)
783
+ if name in ("cargo.toml", "pyproject.toml"):
784
+ return _toml(source)
785
+ return _runtime_yml(source)
786
+
787
+
788
+ def indexable(path: str) -> bool:
789
+ """Whether this file is worth reading at all for the map — checked before the read."""
790
+ return suffix_of(path) in INDEXABLE
791
+
792
+
793
+ def parse(path: str, source: str) -> dict | None:
794
+ """One file's declarations, or None when the suffix is not an indexable source."""
795
+ suffix = suffix_of(path)
796
+ if suffix not in INDEXABLE:
797
+ return None
798
+ if suffix == ".py":
799
+ # A model-written file that does not parse yet is exactly when the map matters most,
800
+ # so the old lexical scan covers for the parser instead of the file going blank.
801
+ return _python(path, source) or fallback(path, source)
802
+ if suffix in SCRIPT_SUFFIXES:
803
+ return _script(path, source)
804
+ if suffix == ".go":
805
+ return _go(path, source)
806
+ if suffix == ".rs":
807
+ return _rust(path, source)
808
+ return _jvm(path, source)
809
+
810
+
811
+ def module_name(path: str) -> str:
812
+ parts = [part for part in re.split(r"[/\\.]", path.rsplit(".", 1)[0]) if part]
813
+ if parts and parts[-1] == "__init__":
814
+ parts.pop()
815
+ return ".".join(parts)
816
+
817
+
818
+ def _keys(row: dict) -> dict[str, str]:
819
+ """Every dotted name another file could import to reach this one."""
820
+ module = module_name(row["path"])
821
+ names = {module}
822
+ if row["kind"] == "go":
823
+ # Go is imported by folder, so the package directory addresses every file in it.
824
+ parts = module.split(".")
825
+ if len(parts) > 1:
826
+ names.add(".".join(parts[:-1]))
827
+ if row["kind"] in NAMED_KINDS:
828
+ for item in row["types"]:
829
+ simple = item["name"].split(".")[0]
830
+ names.add(simple)
831
+ if row["package"]:
832
+ names.add(row["package"] + "." + simple)
833
+ return {name: row["path"] for name in names}
834
+
835
+
836
+ def _linked(imported: str, key: str) -> bool:
837
+ """True when either name is the whole tail of the other, or the other's prefix.
838
+
839
+ Both sides are split on `.` and `/`, because a Go or JavaScript import is a path while
840
+ the indexed module is a dotted name. A Java file lives under `src/main/java/...`, so the
841
+ import is shorter than the indexed module by exactly that prefix; a Python or Rust
842
+ `…::name` import is longer by the trailing symbol, which is the mirror case and only
843
+ matches as a prefix.
844
+ """
845
+ left = [part for part in re.split(r"[./\\]", imported) if part]
846
+ right = [part for part in re.split(r"[./\\]", key) if part]
847
+ shared = min(len(left), len(right))
848
+ if shared and left[-shared:] == right[-shared:]:
849
+ return True
850
+ shorter, longer = (left, right) if len(left) <= len(right) else (right, left)
851
+ return bool(shorter) and len(shorter) < len(longer) and longer[:len(shorter)] == shorter
852
+
853
+
854
+ def dependencies(rows: list[dict]) -> dict[str, list[str]]:
855
+ """Project-internal files each indexed file imports.
856
+
857
+ Only edges inside the repository are kept: a model cannot act on a JDK or stdlib
858
+ import, and listing those would crowd out the one that matters.
859
+ """
860
+ owners: dict[str, str] = {}
861
+ for row in rows:
862
+ for name, path in _keys(row).items():
863
+ owners.setdefault(name, path)
864
+ edges: dict[str, list[str]] = {}
865
+ for row in rows:
866
+ targets = set()
867
+ for imported in row["imports"]:
868
+ dotted = imported.lstrip(".")
869
+ if imported.startswith("."):
870
+ level = len(imported) - len(dotted)
871
+ parts = module_name(row["path"]).split(".")[:-level]
872
+ dotted = ".".join(parts + ([dotted] if dotted else []))
873
+ for key, path in owners.items():
874
+ if path != row["path"] and _linked(dotted, key):
875
+ targets.add(path)
876
+ edges[row["path"]] = sorted(targets)[:12]
877
+ return edges
878
+
879
+
880
+ def module_of(relative: str) -> str:
881
+ """The folder a file belongs to at the level a monorepo is divided — `auth-service` in
882
+ `auth-service/src/main/java/App.java`, and "." for a file loose at the root."""
883
+ parts = [part for part in str(relative).replace("\\", "/").split("/") if part not in ("", ".")]
884
+ return parts[0] if len(parts) > 1 else "."
885
+
886
+
887
+ def spread(files: list[str]) -> list[str]:
888
+ """Round-robin the file list across modules, so a capped map shows every one of them.
889
+
890
+ Alphabetical order and a 12 000-char budget mean one thing in a nine-module reactor: the first
891
+ three modules are described in detail and the other six are not mentioned at all, and the model
892
+ is asked to plan a cross-module change from that. Within a module the order is unchanged.
893
+ """
894
+ groups: dict[str, list[str]] = {}
895
+ for name in files:
896
+ groups.setdefault(module_of(name), []).append(name)
897
+ ordered: list[str] = []
898
+ depth = 0
899
+ while any(len(items) > depth for items in groups.values()):
900
+ for key in sorted(groups):
901
+ if len(groups[key]) > depth:
902
+ ordered.append(groups[key][depth])
903
+ depth += 1
904
+ return ordered
905
+
906
+
907
+ MAX_GRAPH_NODES = 24 # modules on screen at once
908
+ MAX_GRAPH_EDGES = 80 # lines between them
909
+ MAX_GRAPH_DEPTH = 8 # columns; a cycle is clamped here rather than followed forever
910
+
911
+
912
+ def graph(rows: list[dict]) -> dict:
913
+ """The project as modules and the dependencies between them, in columns by build order.
914
+
915
+ Files are the wrong unit for a picture: a reactor of a thousand files drawn as a thousand nodes is
916
+ a picture of nothing, and `module_of()` already names the folders a person thinks in. Edges carry a
917
+ count so a thick line is visibly a heavier dependency, and a node's column is the longest chain that
918
+ must be built before it -- the same order the run command starts the projects in.
919
+
920
+ A cycle has no longest path. The layering peels what it can, breaks the remainder at one named node,
921
+ and says `cyclic` in the result: a diagram that quietly re-ordered itself would be a diagram of a
922
+ project that does not exist, and a real cycle between two Maven modules is exactly the thing a
923
+ reader looks at a graph to find.
924
+ """
925
+ files: dict[str, int] = {}
926
+ for row in rows or []:
927
+ if row.get("kind") == "config":
928
+ continue # a pom is a fact about a module, not a dependency of one
929
+ name = module_of(row["path"])
930
+ files[name] = files.get(name, 0) + 1
931
+
932
+ counts: dict[tuple[str, str], int] = {}
933
+ for origin, targets in dependencies(rows or []).items():
934
+ source = module_of(origin)
935
+ for target in targets:
936
+ dest = module_of(target)
937
+ if dest != source:
938
+ counts[(source, dest)] = counts.get((source, dest), 0) + 1
939
+
940
+ # The heaviest modules are drawn; the rest are counted. A cap that dropped nodes in silence would
941
+ # turn a missing box into a claim that the module depends on nothing.
942
+ keep = sorted(files, key=lambda name: (-files[name], name))[:MAX_GRAPH_NODES]
943
+ live = set(keep)
944
+ edges = sorted(((pair, count) for pair, count in counts.items()
945
+ if pair[0] in live and pair[1] in live),
946
+ key=lambda item: (-item[1], item[0]))[:MAX_GRAPH_EDGES]
947
+ hidden = len(files) - len(live)
948
+
949
+ depends: dict[str, list[str]] = {}
950
+ names: set[str] = set(live)
951
+ for (source, dest), _count in edges:
952
+ depends.setdefault(source, []).append(dest)
953
+ names.update((source, dest))
954
+
955
+ # A node's column is one past the deepest thing it needs, so the columns read as a build order.
956
+ # Nodes are placed by peeling: everything with no unplaced dependency is at the front, and a graph
957
+ # where nothing is ready is a cycle -- broken at a fixed node, named in the result, because a
958
+ # diagram that silently re-ordered itself is a diagram of a project that does not exist.
959
+ placed: dict[str, int] = {}
960
+ remaining = set(names)
961
+ cyclic = False
962
+ while remaining:
963
+ ready = sorted(name for name in remaining
964
+ if not set(depends.get(name, [])) & remaining)
965
+ if not ready:
966
+ cyclic = True
967
+ ready = [sorted(remaining, key=lambda item: (
968
+ -len(set(depends.get(item, [])) - remaining), item))[0]]
969
+ for name in ready:
970
+ # Only dependencies already placed count. Peeling guarantees that in an acyclic graph, and it
971
+ # is what makes the layering a build order; in the node chosen to break a cycle the unplaced
972
+ # dependency is the loop itself, and counting it would shift every column right by one to
973
+ # draw an empty first one.
974
+ deepest = max((placed[dep] for dep in depends.get(name, []) if dep in placed), default=-1)
975
+ placed[name] = min(MAX_GRAPH_DEPTH, deepest + 1)
976
+ remaining.discard(name)
977
+
978
+ return {"nodes": [{"name": name, "files": files.get(name, 0), "column": placed.get(name, 0)}
979
+ for name in sorted(names, key=lambda item: (placed.get(item, 0), item))],
980
+ "edges": [{"from": source, "to": dest, "count": count}
981
+ for (source, dest), count in edges],
982
+ "columns": max(placed.values(), default=0) + 1,
983
+ "cyclic": cyclic,
984
+ "hidden": max(0, hidden)}
985
+
986
+
987
+ MAX_HITS = 40 # reference sites or symbol matches one query may return
988
+ MAX_SITE_TEXT = 200 # characters of the line itself
989
+ PER_FILE_LIMIT = 6 # sites from one file, so a 40-hit answer is not one file's grep
990
+
991
+ # The line kinds a reference can be. The distinction is the whole value of the verb: `search_code`
992
+ # already tells a model that a name appears 30 times, and what it cannot tell is which of those 30 is
993
+ # the declaration, which is an import, and which is somebody calling it.
994
+ IMPORT_LINE = re.compile(r"^\s*(?:from\s+[\w.]+\s+)?import\b|^\s*(?:use|using)\s|#include|^\s*require\s*\(",
995
+ re.I)
996
+ DECLARE_LINE = re.compile(
997
+ r"^\s*(?:@[\w.]+\s+)*(?:public|private|protected|internal|static|final|abstract|override|open|"
998
+ r"sealed|class|interface|enum|record|annotation|def|func|fn|type|impl|export|async|const|let|var|"
999
+ r"pub|local|friend|virtual|override)\b", re.I)
1000
+
1001
+
1002
+ def _member_name(signature: str) -> str:
1003
+ """`login(String email)` out of a members list, which stores signatures not names."""
1004
+ return signature.split("(", 1)[0].strip()
1005
+
1006
+
1007
+ def find_symbol(rows: list[dict], name: str, limit: int = MAX_HITS) -> list[dict]:
1008
+ """Every declaration in the index whose name is this one.
1009
+
1010
+ Types come with the line the index recorded; functions and members do not, because the parsers keep
1011
+ signatures and not one line number per member — a hit without a `line` is honest about that, and
1012
+ `read_file` on the path is the next step rather than a guessed one.
1013
+ """
1014
+ needle = str(name or "").strip()
1015
+ if not needle:
1016
+ return []
1017
+ folded = needle.casefold()
1018
+ exact: list[dict] = []
1019
+ partial: list[dict] = []
1020
+
1021
+ def collect(entry: dict, whole: bool) -> None:
1022
+ (exact if whole else partial).append(entry)
1023
+
1024
+ for row in rows:
1025
+ for item in row["types"]:
1026
+ last = item["name"].rsplit(".", 1)[-1]
1027
+ if folded not in last.casefold():
1028
+ continue
1029
+ entry = {"name": item["name"], "kind": item["kind"], "path": row["path"],
1030
+ "package": row["package"], "line": item["line"]}
1031
+ if item.get("extends"):
1032
+ entry["extends"] = ", ".join(item["extends"])
1033
+ collect(entry, last.casefold() == folded)
1034
+ for signature in row["functions"]:
1035
+ member = _member_name(signature)
1036
+ if folded not in member.casefold():
1037
+ continue
1038
+ collect({"name": member, "kind": "function", "path": row["path"],
1039
+ "package": row["package"], "signature": signature}, member.casefold() == folded)
1040
+ for item in row["types"]:
1041
+ for signature in item["members"]:
1042
+ member = _member_name(signature)
1043
+ if folded not in member.casefold():
1044
+ continue
1045
+ collect({"name": member, "kind": "member", "path": row["path"],
1046
+ "package": row["package"], "in": item["name"], "signature": signature},
1047
+ member.casefold() == folded)
1048
+ hits = exact + partial
1049
+ return hits[:limit]
1050
+
1051
+
1052
+ def find_references(name: str, sources, rows: list[dict] | None = None,
1053
+ limit: int = MAX_HITS, per_file: int = PER_FILE_LIMIT) -> list[dict]:
1054
+ """Where a name is used, and in what role — the reverse of the forward-only `dependencies()`.
1055
+
1056
+ `sources` is any iterable of `(path, text)` pairs, so the caller decides what to pay for: reading
1057
+ the workspace is the same cost as a search, and `symbols` stays free of a workspace import that
1058
+ would close a cycle. Text is matched in `blank()`ed source, which keeps every line where it was and
1059
+ removes comments and string contents — without it a log message naming a class is reported as a
1060
+ use of it, which is the exact mistake the parsers already had to be taught not to make.
1061
+
1062
+ The role is the answer, not the count: `declaration` (the index says this line declares it, or the
1063
+ line opens with a keyword that only a declaration uses), then `import`, then `call`, then
1064
+ `mention`. There is no type inference here and the verb does not claim otherwise — an overloaded
1065
+ method returns every site of that name, and `read_file` on the two that matter is still cheaper
1066
+ than a model guessing which file to open.
1067
+ """
1068
+ needle = str(name or "").strip()
1069
+ if not needle:
1070
+ return []
1071
+ declared = {(hit["path"], hit.get("line")) for hit in find_symbol(rows or [], needle, limit=500)}
1072
+ site = re.compile(r"(?<!\w)" + re.escape(needle) + r"(?!\w)")
1073
+ call = re.compile(r"(?<!\w)" + re.escape(needle) + r"\s*\(")
1074
+ out: list[dict] = []
1075
+ for path, text in sources:
1076
+ if len(out) >= limit:
1077
+ break
1078
+ clean = blank(text, backticks=path.endswith((".js", ".jsx", ".ts", ".tsx", ".mjs", ".cjs")))
1079
+ counted = 0
1080
+ for number, (raw, bare) in enumerate(zip(text.splitlines(), clean.splitlines()), 1):
1081
+ if not site.search(bare):
1082
+ continue
1083
+ if (path, number) in declared:
1084
+ kind = "declaration"
1085
+ elif IMPORT_LINE.match(bare):
1086
+ kind = "import"
1087
+ elif DECLARE_LINE.match(bare) and site.search(bare):
1088
+ kind = "declaration"
1089
+ elif call.search(bare):
1090
+ kind = "call"
1091
+ else:
1092
+ kind = "mention"
1093
+ out.append({"path": path, "line": number, "kind": kind,
1094
+ "text": raw.strip()[:MAX_SITE_TEXT]})
1095
+ counted += 1
1096
+ if counted >= per_file or len(out) >= limit:
1097
+ break
1098
+ return out
1099
+
1100
+
1101
+ # Words a task says that name nothing. Deliberately tiny: this is not a search engine, and a list that
1102
+ # grows past the pronouns starts hiding the noun somebody meant. `add` was in here once and should not
1103
+ # have been -- it is one of the most common names in code (`add(a, b)`), so a task that says "fix add"
1104
+ # and means the function got nothing.
1105
+ STOP_WORDS = {"the", "and", "for", "with", "that", "this", "from", "into", "your", "please", "make",
1106
+ "change", "fix", "file", "files", "code", "class", "method", "function", "test",
1107
+ "tests", "when", "then", "them", "it", "to", "of", "in", "on", "is", "are", "do", "so",
1108
+ "java", "py", "xml", "yml", "yaml", "json", "sql", "md", "gradle", "pom"}
1109
+ WORD = re.compile(r"[A-Za-z_$][A-Za-z0-9_$]{2,}")
1110
+ CAMEL = re.compile(r"[A-Z]+(?![a-z])|[A-Z][a-z0-9]+|[a-z0-9]+")
1111
+
1112
+
1113
+ def task_words(task: str) -> set[str]:
1114
+ """The names inside a sentence, including the parts of a camel-case or snake-case name.
1115
+
1116
+ "Wire the token provider into the login flow" has to reach `JwtTokenProvider.login`. The whole word
1117
+ and each of its parts are both offered because a person writes one and the source is spelled the
1118
+ other, and neither alone finds the pair.
1119
+ """
1120
+ words: set[str] = set()
1121
+ for raw in WORD.findall(str(task or "")):
1122
+ folded = raw.casefold()
1123
+ if folded in STOP_WORDS:
1124
+ continue
1125
+ words.add(folded)
1126
+ for piece in CAMEL.findall(raw):
1127
+ piece = piece.casefold()
1128
+ if len(piece) > 2 and piece not in STOP_WORDS:
1129
+ words.add(piece)
1130
+ return words
1131
+
1132
+
1133
+ def _name_parts(name: str) -> list[str]:
1134
+ return [piece.casefold() for piece in CAMEL.findall(re.split(r"[(\s]", name)[0])]
1135
+
1136
+
1137
+ def rank(rows: list[dict], task: str, limit: int = 6) -> list[dict]:
1138
+ """Which files the task is about, and the one-line reason each was picked.
1139
+
1140
+ The rule before this function was "keep a file whose exact name appears in the sentence", which
1141
+ works when a person types `JwtTokenProvider.java` and does nothing when they describe the bug. This
1142
+ scores the index instead: a declared name the task says, a part of such a name, the folder the task
1143
+ names, and one hop along an import from any of those. It is still not understanding — it is a ranked
1144
+ guess — which is why every entry carries the reason and the window prints it: an unexplained context
1145
+ block is a claim the tool cannot defend when the wrong file turns up in the diff.
1146
+ """
1147
+ words = task_words(task)
1148
+ if not words:
1149
+ return []
1150
+ edges = dependencies(rows)
1151
+ reverse: dict[str, list[str]] = {}
1152
+ for importer, imported in edges.items():
1153
+ for target in imported:
1154
+ reverse.setdefault(target, []).append(importer)
1155
+ scores: dict[str, dict] = {}
1156
+
1157
+ def offer(path: str, score: int, why: str, symbol: str = "") -> None:
1158
+ best = scores.get(path)
1159
+ if best is None or score > best["score"]:
1160
+ scores[path] = {"path": path, "score": score, "why": why, "symbol": symbol}
1161
+ elif score == best["score"] and not best["symbol"] and symbol:
1162
+ best["symbol"] = symbol
1163
+
1164
+ for row in rows:
1165
+ path = row["path"]
1166
+ # Rows carry forward-slash relative paths, which is what `module_of` splits on too, so the
1167
+ # basename is a string operation here rather than a pathlib import this module never needed.
1168
+ basename = path.rsplit("/", 1)[-1]
1169
+ # The boundary test is the old rule kept verbatim: `app.py` in the sentence names `app.py`, and
1170
+ # one written inside a longer word does not. Weakening it here would spend context on a file
1171
+ # nobody asked for, which is the one thing this function is allowed to cost.
1172
+ if re.search(r"(?<![\w.])" + re.escape(basename) + r"(?![\w.])", str(task or ""), re.I):
1173
+ offer(path, 10, "names", basename)
1174
+ for item in row["types"]:
1175
+ name = item["name"].rsplit(".", 1)[-1]
1176
+ whole = name.casefold()
1177
+ parts = _name_parts(name)
1178
+ if whole in words:
1179
+ offer(path, 8, "declares", name)
1180
+ elif (hit := next((word for word in parts if word in words), "")):
1181
+ offer(path, 4, "declares", name)
1182
+ for signature in item["members"]:
1183
+ member = _member_name(signature)
1184
+ if member.casefold() in words:
1185
+ offer(path, 7, "declares", member)
1186
+ elif (hit := next((part for part in _name_parts(member) if part in words), "")):
1187
+ offer(path, 3, "defines", member)
1188
+ for signature in row["functions"]:
1189
+ member = _member_name(signature)
1190
+ if member.casefold() in words:
1191
+ offer(path, 7, "defines", member)
1192
+ elif (hit := next((part for part in _name_parts(member) if part in words), "")):
1193
+ offer(path, 3, "defines", member)
1194
+ folder = module_of(path)
1195
+ # A module folder is a name, spelled the way a class is: the task says "auth-service" and
1196
+ # `task_words` splits it in two exactly as it splits a camel-case class. Comparing only the
1197
+ # whole folder meant that every hyphenated module -- which is how a Spring reactor writes
1198
+ # itself -- ranked nothing at all.
1199
+ if folder != "." and (hit := next((word for word in [folder.casefold()] + _name_parts(folder)
1200
+ if word in words), "")):
1201
+ offer(path, 4, "module", hit)
1202
+
1203
+ for path, entry in list(scores.items()):
1204
+ if entry["score"] < 4:
1205
+ continue
1206
+ for other in reverse.get(path, []):
1207
+ offer(other, min(5, entry["score"] - 2), "imports",
1208
+ entry["symbol"] or path.rsplit("/", 1)[-1])
1209
+
1210
+ ordered = sorted(scores.values(), key=lambda entry: (-entry["score"], entry["path"]))
1211
+ return ordered[:limit]
1212
+
1213
+
1214
+ def config_row(path: str, source: str) -> dict | None:
1215
+ """The map's entry for a configuration file: facts, not declarations.
1216
+
1217
+ It is a row so `render()` prints it in the same block position as a source file, and it is a row
1218
+ with no types and no functions so every name-based query in this module passes over it. A model
1219
+ that learns `modules: auth-service, product-service` from a pom must not then be told the pom
1220
+ "declares" auth-service — that is the confusion this split keeps out.
1221
+ """
1222
+ facts = config_facts(path, source)
1223
+ if not facts:
1224
+ return None
1225
+ return {"path": path, "kind": "config", "package": "", "imports": [], "types": [],
1226
+ "functions": [], "facts": facts}
1227
+
1228
+
1229
+ def render(rows: list[dict], files: list[str], limit: int = 12000, spread_files: bool = False,
1230
+ note: str = "") -> str:
1231
+ """The repository map: every visible file, with its declarations under it.
1232
+
1233
+ `note` goes on the first line, not the last: it says what the map does *not* contain, and a sentence
1234
+ about missing files that is itself truncated off the bottom would be a worse answer than none.
1235
+ """
1236
+ indexed = {row["path"]: row for row in rows}
1237
+ edges = dependencies(rows)
1238
+ if spread_files:
1239
+ files = spread(files)
1240
+ out: list[str] = []
1241
+ if note:
1242
+ out.append(note)
1243
+ used = len(note) + 1 if note else 0
1244
+ shown = 0
1245
+ for name in files:
1246
+ row = indexed.get(name)
1247
+ block = [name]
1248
+ if row:
1249
+ # A configuration file's own lines: what it is, what it builds, what it needs. Printed under
1250
+ # the path like a package line, because that is the position the model already reads
1251
+ # structure from — a fact in prose elsewhere in the prompt is a fact it has to re-find.
1252
+ for label, value in row.get("facts") or []:
1253
+ block.append(" " + label + ": " + value)
1254
+ if row["package"]:
1255
+ block.append(" package " + row["package"])
1256
+ for item in row["types"]:
1257
+ text = " " + item["kind"] + " " + item["name"]
1258
+ if item.get("extends"):
1259
+ text += " extends " + ", ".join(item["extends"])
1260
+ if item["members"]:
1261
+ text += ": " + ", ".join(item["members"][:8])
1262
+ if len(item["members"]) > 8:
1263
+ text += " +" + str(len(item["members"]) - 8) + " more"
1264
+ block.append(text[:200])
1265
+ if row["functions"]:
1266
+ block.append(" functions: " + ", ".join(row["functions"][:8]) +
1267
+ (" …" if len(row["functions"]) > 8 else ""))
1268
+ if edges.get(name):
1269
+ block.append(" depends on: " + ", ".join(edges[name]))
1270
+ block.append("")
1271
+ text = "\n".join(block)
1272
+ if used + len(text) > limit:
1273
+ break
1274
+ out.append(text)
1275
+ used += len(text)
1276
+ shown += 1
1277
+ if shown < len(files):
1278
+ # Which modules were left out entirely, because "300 of 1 200 files" does not tell a reader
1279
+ # whether the missing ones are a detail or the half of the project they are asking about.
1280
+ missing = sorted({module_of(name) for name in files[shown:]}
1281
+ - {module_of(name) for name in files[:shown]})
1282
+ out.append("(index truncated: " + str(shown) + " of " + str(len(files)) + " files listed"
1283
+ + ("; nothing shown from " + ", ".join(missing[:6])
1284
+ + (" …" if len(missing) > 6 else "") if missing else "")
1285
+ + "; read or search the rest on demand)")
1286
+ return "\n".join(out).strip()