ai-code-engineer 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. ai_code_engineer/__init__.py +2 -0
  2. ai_code_engineer/catalog.py +143 -0
  3. ai_code_engineer/chat.py +181 -0
  4. ai_code_engineer/cli.py +384 -0
  5. ai_code_engineer/config.py +405 -0
  6. ai_code_engineer/engine.py +1282 -0
  7. ai_code_engineer/errors.py +27 -0
  8. ai_code_engineer/git_integration.py +443 -0
  9. ai_code_engineer/gui.py +2646 -0
  10. ai_code_engineer/host.py +81 -0
  11. ai_code_engineer/ignore.py +269 -0
  12. ai_code_engineer/intent.py +222 -0
  13. ai_code_engineer/labels.py +871 -0
  14. ai_code_engineer/memory.py +91 -0
  15. ai_code_engineer/modes.py +156 -0
  16. ai_code_engineer/overrides.py +540 -0
  17. ai_code_engineer/planbook.py +192 -0
  18. ai_code_engineer/providers.py +404 -0
  19. ai_code_engineer/redaction.py +54 -0
  20. ai_code_engineer/repair.py +564 -0
  21. ai_code_engineer/report.py +352 -0
  22. ai_code_engineer/runner.py +854 -0
  23. ai_code_engineer/setup.py +386 -0
  24. ai_code_engineer/symbols.py +1286 -0
  25. ai_code_engineer/verification.py +218 -0
  26. ai_code_engineer/webapp/__init__.py +1 -0
  27. ai_code_engineer/webapp/__main__.py +45 -0
  28. ai_code_engineer/webapp/contract.py +36 -0
  29. ai_code_engineer/webapp/controller.py +3556 -0
  30. ai_code_engineer/webapp/fake.py +1141 -0
  31. ai_code_engineer/webapp/launch.py +108 -0
  32. ai_code_engineer/webapp/server.py +349 -0
  33. ai_code_engineer/webapp/static/app.css +780 -0
  34. ai_code_engineer/webapp/static/app.js +2118 -0
  35. ai_code_engineer/webapp/static/boot.js +19 -0
  36. ai_code_engineer/webapp/static/index.html +89 -0
  37. ai_code_engineer/webapp/static/tokens.css +173 -0
  38. ai_code_engineer/workspace.py +385 -0
  39. ai_code_engineer-0.1.0.dist-info/METADATA +7 -0
  40. ai_code_engineer-0.1.0.dist-info/RECORD +44 -0
  41. ai_code_engineer-0.1.0.dist-info/WHEEL +5 -0
  42. ai_code_engineer-0.1.0.dist-info/entry_points.txt +2 -0
  43. ai_code_engineer-0.1.0.dist-info/licenses/LICENSE +21 -0
  44. ai_code_engineer-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,854 @@
1
+ """Run one allowlisted build/test recipe inside the approved project folder.
2
+
3
+ Commands are constant argv lists resolved with shutil.which; no shell, no model or
4
+ user text reaches the command line. This reduces blast radius but is not a sandbox:
5
+ the sandbox is here too (`sandbox_argv`, the `sandbox=` argument), and it is the only
6
+ place in the codebase that starts a container — `verification.docker_check` asks it for
7
+ its flags rather than keeping a second copy that can drift.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import io
12
+ import os
13
+ from pathlib import Path, PurePosixPath
14
+ import queue
15
+ import re
16
+ import shlex
17
+ import shutil
18
+ import subprocess
19
+ import sys
20
+ import tempfile
21
+ import threading
22
+ import time
23
+ import uuid
24
+ import xml.etree.ElementTree as ET
25
+ from collections import deque
26
+
27
+ from . import ignore
28
+ from .errors import PolicyError
29
+
30
+ MAX_OUTPUT_CHARS = 200_000
31
+ MODEL_OUTPUT_CHARS = 12_000
32
+ # A JUnit report is evidence, not an archive. Anything bigger is not read as proof.
33
+ MAX_REPORT_BYTES = 2_000_000
34
+ DEFAULT_TIMEOUT = 600
35
+ LONG_TIMEOUT = 1500
36
+ # A build can print a hundred thousand lines; a person reads a few hundred.
37
+ STREAM_BUDGET = 400
38
+ # After a tree kill the child still has buffered output to hand over.
39
+ POST_KILL_GRACE = 5.0
40
+
41
+ # Env entries a compiler toolchain needs. Anything else (tokens, cloud keys, proxies)
42
+ # stays out of the child process.
43
+ ENV_KEYS = ("PATH", "PATHEXT", "SYSTEMROOT", "WINDIR", "COMSPEC", "TEMP", "TMP",
44
+ "HOME", "USERPROFILE", "HOMEDRIVE", "HOMEPATH", "JAVA_HOME", "MAVEN_HOME",
45
+ "GRADLE_HOME", "M2_HOME", "NODE_PATH", "LANG", "LC_ALL", "PYTHONIOENCODING",
46
+ "GOPATH", "GOROOT", "CARGO_HOME", "RUSTUP_HOME")
47
+
48
+ RECIPES: dict[str, dict] = {
49
+ "maven-test": {
50
+ "command": ["mvn", "-B", "test"],
51
+ "markers": ["pom.xml"],
52
+ "label": "Maven test",
53
+ "test_counts": r"Tests run:\s*(\d+)",
54
+ "reports": ("target/surefire-reports/TEST-*.xml", "*/target/surefire-reports/TEST-*.xml"),
55
+ "proof_source": "surefire XML",
56
+ },
57
+ "maven-compile": {
58
+ "command": ["mvn", "-B", "-DskipTests", "compile"],
59
+ "markers": ["pom.xml"],
60
+ "label": "Maven compile",
61
+ "test_counts": None,
62
+ },
63
+ "gradle-test": {
64
+ "command": ["gradle", "--no-daemon", "test"],
65
+ "markers": ["build.gradle", "build.gradle.kts"],
66
+ "label": "Gradle test",
67
+ "test_counts": None,
68
+ "reports": ("build/test-results/**/TEST-*.xml", "*/build/test-results/**/TEST-*.xml"),
69
+ "proof_source": "Gradle JUnit XML",
70
+ },
71
+ "python-unittest": {
72
+ "command": [sys.executable, "-m", "unittest", "discover", "-s", "tests", "-v"],
73
+ "markers": [],
74
+ "test_folder": True,
75
+ "label": "Python unittest",
76
+ "test_counts": r"Ran (\d+) tests?",
77
+ },
78
+ "uv-pytest": {
79
+ "command": ["uv", "run", "pytest", "-q"],
80
+ "markers": ["pyproject.toml", "uv.lock"],
81
+ "label": "uv pytest",
82
+ "test_counts": r"(\d+) passed",
83
+ "junit_arg": "--junitxml",
84
+ "proof_source": "pytest JUnit XML",
85
+ },
86
+ "python-pytest": {
87
+ "command": [sys.executable, "-m", "pytest", "-q"],
88
+ "markers": ["pytest.ini", "pyproject.toml"],
89
+ "label": "pytest",
90
+ "test_counts": r"(\d+) passed",
91
+ "junit_arg": "--junitxml",
92
+ "proof_source": "pytest JUnit XML",
93
+ },
94
+ "node-test": {
95
+ "command": ["node", "--test"],
96
+ "markers": ["package.json"],
97
+ "label": "Node test runner",
98
+ "test_counts": r"pass\s+(\d+)",
99
+ },
100
+ "npm-test": {
101
+ "command": ["npm", "test"],
102
+ "markers": ["package.json"],
103
+ "label": "npm test",
104
+ "test_counts": r"pass\s+(\d+)",
105
+ },
106
+ "pnpm-test": {
107
+ "command": ["pnpm", "test"],
108
+ "markers": ["pnpm-lock.yaml"],
109
+ "label": "pnpm test",
110
+ # pnpm prints vitest's "Tests 12 passed" or the node runner's "pass 12".
111
+ "test_counts": r"(?i)(?:\bpass\b|\btests\b)\s+(\d+)",
112
+ },
113
+ "cargo-test": {
114
+ "command": ["cargo", "test"],
115
+ "markers": ["cargo.toml"],
116
+ "label": "Cargo test",
117
+ "test_counts": r"test result: \w+\.\s+(\d+) passed",
118
+ # Cargo writes no report file, so its own summary line is the only count there is. Every
119
+ # target prints one (`lib`, each bin, each integration test, the doc tests), and the run is
120
+ # the sum of them: reading only the first proved 20 of 24 tests and said nothing about the
121
+ # four that had not run.
122
+ "console_proof": {"line": r"(?m)^test result: \w+\.\s+(?P<tests>\d+) passed;\s+"
123
+ r"(?P<failures>\d+) failed;\s+(?P<skipped>\d+) ignored",
124
+ "source": "cargo's own summary"},
125
+ },
126
+ "go-test": {
127
+ "command": ["go", "test", "./..."],
128
+ "markers": ["go.mod"],
129
+ "label": "Go test",
130
+ # go test prints no total: one "ok <package>" line per package that ran.
131
+ "test_counts": None,
132
+ "test_ran": r"(?m)^ok\s+\S+",
133
+ },
134
+ }
135
+
136
+
137
+ def display_command(recipe: str) -> str:
138
+ """The recipe as one line a person can read before agreeing to run it.
139
+
140
+ The interpreter is named, not pathed: an absolute `python.exe` under a user profile is noise in a
141
+ confirmation, and the argv the child really gets is recorded in the log. Both windows ask the same
142
+ question about the same command, so both get it from here rather than spelling out the
143
+ substitution again.
144
+ """
145
+ return " ".join("python" if part == sys.executable else str(part)
146
+ for part in RECIPES[recipe]["command"])
147
+
148
+
149
+ def timeout_for(recipe: str) -> int:
150
+ """JVM builds need minutes even warm; script suites usually do not.
151
+
152
+ Cargo joins them: `cargo test` is a *compile* the first time a crate or one of its dependencies
153
+ changes, and a cold Rust build on this class of machine runs past the 600 seconds the script
154
+ recipes get. A killed compile reads as "the command timed out", which sends a fix round at a
155
+ problem that was only ever slow.
156
+ """
157
+ return LONG_TIMEOUT if recipe.startswith(("maven", "gradle", "cargo")) else DEFAULT_TIMEOUT
158
+
159
+
160
+ def available() -> list[str]:
161
+ """Recipes whose executable exists on this machine."""
162
+ found = []
163
+ for name, recipe in RECIPES.items():
164
+ program = recipe["command"][0]
165
+ if program == sys.executable or shutil.which(program):
166
+ found.append(name)
167
+ return found
168
+
169
+
170
+ def sandbox_available() -> bool:
171
+ """Whether this machine can start the container at all — the one question the switch asks.
172
+
173
+ Both windows grey the choice out from here rather than calling `which` themselves, so there is one
174
+ seam for a test to answer and one answer on a machine with and without a daemon.
175
+ """
176
+ return bool(shutil.which("docker"))
177
+
178
+
179
+ def sandbox_state(on, image: str, available: bool) -> str:
180
+ """Which of the four sentences a sandbox choice deserves.
181
+
182
+ A name, not a sentence: the words belong to `labels`, and the preview window has to be able to say
183
+ the same four things without a daemon to ask.
184
+ """
185
+ if not available:
186
+ return "sandbox_missing"
187
+ if not on:
188
+ return "sandbox_off"
189
+ return "sandbox_on" if SANDBOX_IMAGE.fullmatch(str(image or "").strip()) else "sandbox_unpinned"
190
+
191
+
192
+ def detect(repo: Path, installed: list[str] | None = None) -> list[str]:
193
+ """Recipes matching the project's build files, in RECIPES order.
194
+
195
+ `installed` lets a caller that scans many folders ask `available()` once: it resolves each
196
+ executable against PATH, and nine modules of one reactor cost half a second when every `detect()`
197
+ re-did it.
198
+ """
199
+ files = {path.name.lower() for path in repo.glob("*")}
200
+ folders = {path.name.lower() for path in repo.iterdir() if path.is_dir()}
201
+ result = []
202
+ for name in (available() if installed is None else installed):
203
+ recipe = RECIPES[name]
204
+ markers = set(recipe["markers"])
205
+ if recipe.get("test_folder") and "tests" in folders:
206
+ markers.add("tests")
207
+ if markers & (files | folders):
208
+ result.append(name)
209
+ return result
210
+
211
+
212
+ def child_env() -> dict:
213
+ return {key: value for key in ENV_KEYS
214
+ if (value := os.environ.get(key)) is not None}
215
+
216
+
217
+ # ---------------------------------------------------------------- the sandbox
218
+
219
+ SANDBOX_IMAGE = re.compile(r"[a-zA-Z0-9./:_-]+@sha256:[a-f0-9]{64}")
220
+ # Where the copied project lives inside the container. A POSIX path either way: the string is passed
221
+ # to Docker, not to this OS, and a Windows source is normalised to forward slashes below.
222
+ WORKDIR = "/work"
223
+ # The interpreter a recipe names once it is inside an image. `sys.executable` is this machine's
224
+ # `python.exe` under a user profile, and no image has that path.
225
+ IMAGE_PYTHON = "python3"
226
+ SANDBOX_FILES = 2000
227
+ SANDBOX_BYTES = 20_000_000
228
+ # Not part of what a clean build needs, and each one is bigger than the sources it sits beside. A
229
+ # project whose build genuinely depends on `node_modules` needs an image that carries it: the
230
+ # container has no network, so nothing can be installed into it.
231
+ SANDBOX_SKIP = (".git", ".svn", "node_modules", "__pycache__", ".mypy_cache", ".pytest_cache",
232
+ ".gradle", ".venv", "venv")
233
+
234
+
235
+ def copy_for_sandbox(root: Path, target: Path) -> dict:
236
+ """The project, on the host side of the mount, minus what a clean build does not need.
237
+
238
+ The copy is the isolation: the user's tree is never mounted, so nothing the build writes can land
239
+ in it, and the reports the build *does* write are readable here afterwards. Size is capped because
240
+ a container that cannot start in reasonable time is not a safer answer than a host run — it is just
241
+ a slower refusal, so it is refused as a policy instead.
242
+ """
243
+ files = total = 0
244
+ for path in root.rglob("*"):
245
+ if any(part in SANDBOX_SKIP for part in path.relative_to(root).parts[:-1]):
246
+ continue
247
+ if path.is_symlink() or not path.is_file():
248
+ continue
249
+ try:
250
+ size = path.stat().st_size
251
+ except OSError:
252
+ continue
253
+ files += 1
254
+ total += size
255
+ if files > SANDBOX_FILES:
256
+ raise PolicyError("The project is too large to copy into a sandbox (%d files)." % files)
257
+ if total > SANDBOX_BYTES:
258
+ raise PolicyError("The project is too large to copy into a sandbox (20 MB).")
259
+ shutil.copytree(root, target, ignore=shutil.ignore_patterns(*SANDBOX_SKIP), symlinks=False,
260
+ dirs_exist_ok=True)
261
+ # The container runs as an unprivileged uid; on a POSIX host a temp directory owned by this user
262
+ # is not writable by it. Windows maps its own permissions and ignores the mode.
263
+ os.chmod(target, 0o1777)
264
+ return {"files": files, "bytes": total}
265
+
266
+
267
+ def sandbox_argv(docker: str, source: Path, image: str, name: str, command: list[str],
268
+ workdir: str = WORKDIR) -> list[str]:
269
+ """The one container this tool knows how to start, and the only place its flags are written.
270
+
271
+ No network, no capabilities, no new privileges, an unprivileged uid, a read-only root filesystem
272
+ with the copied project as the single writable mount, and a pinned image — a tag can be retagged
273
+ at any time, a digest cannot. `--rm` cleans the container up; the caller's `docker rm -f` covers
274
+ the case where it never got to exit.
275
+
276
+ `command` is argv inside the container: no shell, and nothing here interpolates model or user text.
277
+ """
278
+ if not SANDBOX_IMAGE.fullmatch(image):
279
+ raise PolicyError("Choose a preloaded image pinned by sha256 digest.")
280
+ return [docker, "run", "--name", name, "--rm", "--pull=never", "--network=none",
281
+ "--read-only", "--cap-drop=ALL", "--security-opt=no-new-privileges",
282
+ "--user=65534:65534", "--pids-limit=128", "--memory=1g", "--cpus=1",
283
+ "--mount", "type=bind,source=%s,target=%s" % (source.as_posix(), workdir),
284
+ "--tmpfs", "/tmp:rw,nosuid,size=128m",
285
+ "--workdir", workdir, "--env", "HOME=/tmp", image, *command]
286
+
287
+
288
+ # Folders that hold a build of their own. This is the whole of what a "project" is here: the tool
289
+ # does not guess from the file tree, it reads the file a build tool would have to have. Lowercased,
290
+ # because the comparison is by folded name on a case-insensitive filesystem.
291
+ PROJECT_FILES = ("pom.xml", "build.gradle", "build.gradle.kts", "settings.gradle", "settings.gradle.kts",
292
+ "package.json", "pyproject.toml", "go.mod", "cargo.toml", "makefile")
293
+ # Walked past, never into: a vendored dependency and a build output both contain build files that
294
+ # belong to somebody else, and `mvn` would be happy to answer to any of them. The names are
295
+ # `ignore.project_dir` holds those names: the read gate's list plus the output folders that only
296
+ # disqualify a directory as a project root, never as a file somebody asked to read.
297
+ PROJECT_DEPTH = 3
298
+ PROJECT_LIMIT = 40
299
+
300
+
301
+ def projects(repo: Path, max_depth: int = PROJECT_DEPTH,
302
+ limit: int = PROJECT_LIMIT) -> list[str]:
303
+ """Every folder under `repo` that holds its own build file, the opened folder first.
304
+
305
+ A monorepo opened at the top is one folder to the model and five commands to run: the reactor
306
+ root, and a module each. `detect()` looks at one folder's own children, so `backend/pom.xml` was
307
+ simply invisible and the window said "no command was detected" about a project that builds
308
+ perfectly. Depth is capped and vendor folders skipped because a real Spring reactor nests
309
+ `node_modules` under a starter module, and every one of those has a `package.json` that would be
310
+ offered as this project's build. The opened folder is always first: it is what the user chose, and
311
+ a Python project with only a `tests/` folder answers to no marker file at all.
312
+ """
313
+ root = Path(repo)
314
+ if not root.is_dir():
315
+ return []
316
+ found = ["."]
317
+ level, depth = [root], 0
318
+ while level and len(found) < limit and depth < max_depth:
319
+ children: list[Path] = []
320
+ for folder in level:
321
+ try:
322
+ entries = sorted((child for child in folder.iterdir() if child.is_dir()),
323
+ key=lambda child: child.name.lower())
324
+ except OSError:
325
+ continue
326
+ for child in entries:
327
+ if ignore.project_dir(child.name):
328
+ continue
329
+ if _holds_build_file(child):
330
+ relative = child.relative_to(root).as_posix()
331
+ if relative not in found:
332
+ found.append(relative)
333
+ children.append(child)
334
+ level, depth = children, depth + 1
335
+ return found[:limit]
336
+
337
+
338
+ def _holds_build_file(folder: Path) -> bool:
339
+ """Whether this folder is a build of its own, read by folded name rather than by spelling."""
340
+ try:
341
+ names = {child.name.lower() for child in folder.iterdir()}
342
+ except OSError:
343
+ return False
344
+ return any(marker in names for marker in PROJECT_FILES)
345
+
346
+
347
+ def targets(repo: Path, max_depth: int = PROJECT_DEPTH, limit: int = PROJECT_LIMIT) -> list[dict]:
348
+ """Where a command can be run in this folder: each project that answers to an installed tool.
349
+
350
+ Both windows draw their picker from here, so the list of folders and the commands each one has are
351
+ decided once. A folder with a build file whose tool is not installed is left out rather than
352
+ offered as a button that can only answer "Maven is not on PATH" — `detect()` already filters by
353
+ `available()`, and an empty list means the window hides the whole Checks card.
354
+ """
355
+ root = Path(repo)
356
+ rows: list[dict] = []
357
+ installed = available()
358
+ for relative in projects(root, max_depth=max_depth, limit=limit):
359
+ try:
360
+ folder = project_folder(root, relative)
361
+ except (PolicyError, OSError):
362
+ continue
363
+ found = detect(folder, installed)
364
+ if not found:
365
+ continue
366
+ rows.append({"path": relative,
367
+ # The path is the label, because `api` in two places of one monorepo is two
368
+ # modules and the window cannot show one button that means both.
369
+ "label": root.name if relative == "." else relative,
370
+ "recipes": found})
371
+ if len(rows) > 1 and rows[0]["path"] == ".":
372
+ # The opened folder of a monorepo is also the reactor root, and the difference matters when
373
+ # the choice is between "everything" and "one module".
374
+ rows[0]["label"] = rows[0]["label"] + " (whole project)"
375
+ return rows
376
+
377
+
378
+ def project_folder(repo: Path, target: str = "") -> Path:
379
+ """The folder a command runs in: a chosen module of this project, never anything outside it.
380
+
381
+ `target` arrives from a window, which means from a click on a label — so it is resolved and then
382
+ checked against the root rather than trusted. A path that escapes the opened folder is refused
383
+ here, in the one place both windows go through, because the alternative is a build tool pointed
384
+ at a tree the user never approved.
385
+ """
386
+ root = Path(repo).resolve(strict=True)
387
+ wanted = str(target or ".").strip().replace("\\", "/")
388
+ if wanted in ("", ".", "./"):
389
+ return root
390
+ path = PurePosixPath(wanted)
391
+ parts = path.parts
392
+ if (path.is_absolute() or not parts or any(":" in part for part in parts)
393
+ or any(part in (".", "..") for part in parts)):
394
+ raise PolicyError("That folder is not inside the project: " + wanted[:120])
395
+ try:
396
+ candidate = (root / Path(*parts)).resolve(strict=True)
397
+ except OSError:
398
+ raise PolicyError("That folder is not in the project any more: " + wanted[:120]) from None
399
+ if root not in candidate.parents:
400
+ raise PolicyError("That folder is not inside the project: " + wanted[:120])
401
+ return candidate
402
+
403
+
404
+ def kill_tree(process: subprocess.Popen) -> None:
405
+ """Stop the child and its tree. This never raises: it runs on the way out of a timeout, and an
406
+ exception here would take a build that already finished with it."""
407
+ if os.name == "nt":
408
+ try:
409
+ flags = getattr(subprocess, "CREATE_NO_WINDOW", 0x08000000)
410
+ subprocess.run(["taskkill", "/F", "/T", "/PID", str(process.pid)],
411
+ capture_output=True, stdin=subprocess.DEVNULL, timeout=30,
412
+ creationflags=flags)
413
+ return
414
+ except (OSError, subprocess.SubprocessError):
415
+ pass # taskkill absent or wedged: fall back to the direct child
416
+ else:
417
+ import signal
418
+ try:
419
+ os.killpg(os.getpgid(process.pid), signal.SIGKILL)
420
+ return
421
+ except (ProcessLookupError, OSError):
422
+ pass
423
+ try:
424
+ process.kill()
425
+ except (ProcessLookupError, OSError):
426
+ pass # already gone; there is nothing left for us to stop
427
+
428
+
429
+ # Deliberately bounded alternatives: an open-ended "[a-z]+error" pattern backtracks
430
+ # catastrophically on the minified single-line output real build tools produce.
431
+ FAILURE_PATTERN = re.compile(
432
+ r"(?i)(\berrors?\b|\bfail(ed|ure|s)?\b|exception|traceback|cannot find symbol|unresolved|"
433
+ r"build failure|(syntax|name|type|value|key|import|attribute|index|assertion|runtime|null|"
434
+ r"conversion|parsing|resolution)error)")
435
+
436
+
437
+ # Surefire and friends print "Tests run: 8, Failures: 0, Errors: 0" on every green run.
438
+ # Those lines carry the failure words but are not failures, and feeding them to a model
439
+ # as evidence sends it looking for a problem that is not there.
440
+ CLEAN_COUNTS = re.compile(r"(?i)(?:failures?|errors?)\s*[:=]\s*0\b")
441
+ DIRTY_COUNTS = re.compile(r"(?i)(?:failures?|errors?)\s*[:=]\s*[1-9]")
442
+ # Cargo prints "test result: ok. 5 passed; 0 failed" and Go prints "ok <package>", where
443
+ # the package path can itself contain the word "error" — so those lines are read by prefix.
444
+ CLEAN_SUMMARY = re.compile(r"(?i)^(?:test result: ok\.|ok\s+\S)")
445
+
446
+
447
+ def _clean(line: str) -> bool:
448
+ if CLEAN_SUMMARY.match(line):
449
+ return True
450
+ if CLEAN_COUNTS.search(line) and not DIRTY_COUNTS.search(line):
451
+ return True
452
+ return line.endswith((": OK", "... ok", "...OK", "skipped"))
453
+
454
+
455
+ def _failures(text: str) -> list[str]:
456
+ rows = []
457
+ for line in text.splitlines():
458
+ stripped = line.strip()[:500]
459
+ if not stripped or _clean(stripped):
460
+ continue
461
+ if FAILURE_PATTERN.search(stripped):
462
+ rows.append(stripped)
463
+ return rows[:40]
464
+
465
+
466
+ # A console line is something the project can print; a report file is something the test
467
+ # framework wrote after running the tests. Where a machine-readable report exists it is
468
+ # the authority, and only a report fresh enough to belong to this run counts.
469
+ MTIME_GRACE_SECONDS = 2
470
+
471
+
472
+ def _suite_counts(suite: ET.Element) -> tuple[int, int, int, int]:
473
+ """(tests, failures, errors, skipped) for one <testsuite>, attributes or children."""
474
+ def number(name: str, default: int = 0) -> int:
475
+ raw = suite.get(name)
476
+ try:
477
+ return int(float(raw)) if raw not in (None, "") else default
478
+ except ValueError:
479
+ return default
480
+ cases = list(suite.iter("testcase"))
481
+ tests = number("tests", len(cases))
482
+ failures = number("failures", sum(1 for c in cases if c.find("failure") is not None))
483
+ errors = number("errors", sum(1 for c in cases if c.find("error") is not None))
484
+ skipped = number("skipped", sum(1 for c in cases if c.find("skipped") is not None))
485
+ return tests, failures, errors, skipped
486
+
487
+
488
+ def report_counts(files: list[Path]) -> dict | None:
489
+ """Aggregate JUnit-family report files; None when none of them is readable."""
490
+ tests = failures = errors = skipped = parsed = 0
491
+ for path in files:
492
+ try:
493
+ # A report is evidence, not an archive: a test that dumps its stack into a testcase
494
+ # element can produce hundreds of megabytes, and this reads it on the UI's thread.
495
+ if path.stat().st_size > MAX_REPORT_BYTES:
496
+ continue
497
+ root = ET.fromstring(path.read_bytes())
498
+ except (OSError, ET.ParseError):
499
+ continue
500
+ parsed += 1
501
+ suites = [root] if root.tag == "testsuite" else list(root.iter("testsuite"))
502
+ if not suites:
503
+ suites = [root]
504
+ for suite in suites:
505
+ count = _suite_counts(suite)
506
+ tests += count[0]
507
+ failures += count[1]
508
+ errors += count[2]
509
+ skipped += count[3]
510
+ if not parsed:
511
+ return None
512
+ return {"tests": tests, "failures": failures, "errors": errors, "skipped": skipped,
513
+ "reports": parsed}
514
+
515
+
516
+ def fresh_reports(root: Path, patterns: tuple, started_wall: float,
517
+ junit: Path | None) -> list[Path]:
518
+ """Report files this run can be held responsible for, not last build's leftovers."""
519
+ candidates = [junit] if junit is not None else [
520
+ found for pattern in patterns for found in root.glob(pattern)]
521
+ kept = []
522
+ for path in candidates:
523
+ try:
524
+ if path.is_file() and path.stat().st_mtime >= started_wall - MTIME_GRACE_SECONDS:
525
+ kept.append(path)
526
+ except OSError:
527
+ continue
528
+ return kept
529
+
530
+
531
+ def decide_status(timed_out: bool, exit_code: int, proof: dict | None,
532
+ console_tests: bool, expects_tests: bool) -> str:
533
+ """A green command that proved no test ran is not a pass."""
534
+ if timed_out:
535
+ return "timeout"
536
+ if proof is not None:
537
+ # The report outranks the exit code: a build that swallows failures still failed.
538
+ if proof["failures"] or proof["errors"]:
539
+ return "failed"
540
+ return "passed" if proof["tests"] - proof["skipped"] > 0 else "unverified"
541
+ if exit_code != 0:
542
+ return "failed"
543
+ if expects_tests and not console_tests:
544
+ return "unverified"
545
+ return "passed"
546
+
547
+
548
+ def tests_ran(recipe_entry: dict, output: str, counts: list) -> bool:
549
+ """Did a test actually execute? Some tools print a total, others only one line each."""
550
+ watcher = recipe_entry.get("test_ran")
551
+ if watcher:
552
+ return bool(re.search(watcher, output))
553
+ return any(int(value) > 0 for value in counts)
554
+
555
+
556
+ def console_proof(recipe_entry: dict, output: str) -> dict | None:
557
+ """A count the tool printed itself, for the runners that write no report file.
558
+
559
+ Only a recipe that declares one is read this way, and only its own summary lines: a proof built
560
+ from stdout is the runner's claim about its run, not this program guessing from the word "ok"
561
+ appearing somewhere. Returns None when the run printed no summary at all, which leaves the exit
562
+ code and `expects_proof` to decide — the honest answer then is "unverified", not zero.
563
+ """
564
+ spec = recipe_entry.get("console_proof")
565
+ if not spec:
566
+ return None
567
+ totals = {"tests": 0, "failures": 0, "errors": 0, "skipped": 0}
568
+ found = 0
569
+ for match in re.finditer(spec["line"], output):
570
+ found += 1
571
+ for name, value in match.groupdict().items():
572
+ if name in totals:
573
+ totals[name] += int(value or 0)
574
+ if not found:
575
+ return None
576
+ totals["source"] = spec["source"]
577
+ return totals
578
+
579
+
580
+ def expects_proof(recipe_entry: dict) -> bool:
581
+ """True for recipes whose whole purpose is running tests, so a silent green run is suspect."""
582
+ return bool(recipe_entry["test_counts"]) or bool(recipe_entry.get("test_ran"))
583
+
584
+
585
+ def _visible(chunk: str) -> str:
586
+ """One line as the terminal would show it: a spinner's carriage returns collapse.
587
+
588
+ `npm` and `cargo` redraw the same line dozens of a second, and the pipe is opened
589
+ without newline translation precisely so a `\r` stays inside its line — one entry per
590
+ redraw would bury the lines that matter. A trailing `\r` is a CRLF ending, not a redraw.
591
+ """
592
+ text = chunk.rstrip("\n").rstrip("\r")
593
+ return text.rsplit("\r", 1)[-1].rstrip()
594
+
595
+
596
+ def _collect(process, progress, deadline: float) -> tuple[str, bool, int]:
597
+ """Read the child until it closes its output, streaming each line as it lands.
598
+
599
+ Returns the kept text, whether the run was killed for taking too long, and how many
600
+ characters were dropped: the tail is what the model reads, so the front goes first.
601
+ """
602
+ chunks: queue.Queue = queue.Queue(maxsize=2000)
603
+ # Read as text, split only on "\n": `Popen` cannot be told this, and both the default
604
+ # universal-newlines mode and `newline=""` end a line at every carriage return, which
605
+ # would turn each spinner redraw into its own log entry.
606
+ stream = io.TextIOWrapper(process.stdout, encoding="utf-8", errors="replace", newline="\n")
607
+
608
+ def reader():
609
+ try:
610
+ for chunk in iter(stream.readline, ""):
611
+ chunks.put(chunk)
612
+ except (OSError, ValueError):
613
+ pass # the pipe closes when the process dies
614
+ finally:
615
+ chunks.put(None)
616
+
617
+ threading.Thread(target=reader, daemon=True, name="agent-output").start()
618
+ try:
619
+ kept: deque[str] = deque()
620
+ stored = dropped = streamed = hidden = 0
621
+ timed_out = False
622
+ while True:
623
+ remaining = deadline - time.monotonic()
624
+ if remaining <= 0:
625
+ if timed_out:
626
+ break
627
+ timed_out = True # kill it, then read whatever it flushes on the way
628
+ kill_tree(process)
629
+ deadline = time.monotonic() + POST_KILL_GRACE
630
+ try:
631
+ chunk = chunks.get(timeout=min(max(remaining, 0.05), 0.5))
632
+ except queue.Empty:
633
+ continue
634
+ if chunk is None:
635
+ break
636
+ pieces = chunk.split("\n")
637
+ if pieces[-1] == "":
638
+ pieces.pop() # that newline ended a line; it is not a new one
639
+ for part in pieces:
640
+ line = _visible(part)
641
+ kept.append(line + "\n")
642
+ stored += len(line) + 1
643
+ while stored > MAX_OUTPUT_CHARS and len(kept) > 1:
644
+ gone = kept.popleft()
645
+ stored -= len(gone)
646
+ dropped += len(gone)
647
+ if not line:
648
+ continue
649
+ if streamed < STREAM_BUDGET:
650
+ streamed += 1
651
+ # One minified bundle is a line too; the caller's channel is a status feed.
652
+ progress(line if len(line) <= 500 else line[:500] + "…")
653
+ else:
654
+ hidden += 1
655
+ if hidden:
656
+ progress(f"… {hidden} more lines; only the first {STREAM_BUDGET} are streamed.")
657
+ text = "".join(kept)
658
+ if len(text) > MAX_OUTPUT_CHARS: # one gigantic line is still only a tail
659
+ dropped += len(text) - MAX_OUTPUT_CHARS
660
+ text = text[-MAX_OUTPUT_CHARS:]
661
+ try:
662
+ process.wait(timeout=5.0 if timed_out else max(1.0, deadline - time.monotonic()))
663
+ except subprocess.TimeoutExpired:
664
+ kill_tree(process)
665
+ timed_out = True
666
+ try:
667
+ process.wait(timeout=30)
668
+ except subprocess.TimeoutExpired:
669
+ # A child the OS will not reap is not a reason to lose the output we already read.
670
+ pass
671
+ finally:
672
+ # A progress sink that raises escapes from the middle of the loop above with the child
673
+ # still running. Closing first would block here: the reader thread holds the buffer's lock
674
+ # until the pipe closes, so the child has to die before the stream can.
675
+ if process.returncode is None:
676
+ kill_tree(process)
677
+ try:
678
+ stream.close()
679
+ except OSError:
680
+ pass
681
+ return text, timed_out, dropped
682
+
683
+
684
+ def run(repo: Path, recipe: str, timeout: int = DEFAULT_TIMEOUT,
685
+ progress=lambda _text: None, target: str = "", sandbox: str = "") -> dict:
686
+ """Execute a fixed recipe in the project folder — or in one module of it, when a target is named.
687
+
688
+ `target` is a relative folder of `repo`, never a path the caller invented: `project_folder`
689
+ resolves and checks it, and the child's `cwd` is the result. A reactor build runs at the root and
690
+ Maven walks the modules itself; pointing one module at a time is for the monorepo where the
691
+ modules do not share a build file.
692
+
693
+ `sandbox` is the pinned image to run inside, or "" for this machine. With one named, the project is
694
+ copied to the host side of the only writable mount the container sees: the tree the user opened is
695
+ never mounted, so nothing the build writes can reach it, and the reports the build writes are still
696
+ readable afterwards — which is the whole reason the copy is a directory and not a tmpfs.
697
+ """
698
+ if recipe not in RECIPES:
699
+ # A PolicyError carries its reason through friendly_error; a bare ValueError would be
700
+ # replaced by "Could not complete the operation." in both windows.
701
+ raise PolicyError("Unknown recipe: " + str(recipe)[:80])
702
+ recipe_entry = RECIPES[recipe]
703
+ command = list(recipe_entry["command"])
704
+ sandbox = str(sandbox or "").strip()
705
+ if sandbox and not SANDBOX_IMAGE.fullmatch(sandbox):
706
+ # Refused before the copy: a tag can be retagged while the build is running, and a project
707
+ # tree has already been walked and written to a temp folder by the time argv is assembled.
708
+ raise PolicyError("Choose a preloaded image pinned by sha256 digest.")
709
+ docker = ""
710
+ if sandbox:
711
+ docker = shutil.which("docker") or ""
712
+ if not docker:
713
+ # The rule the request asked for twice: a missing sandbox is a refusal, not a hint to run
714
+ # the build on the machine the sandbox was chosen to keep the build away from.
715
+ return {"recipe": recipe, "label": recipe_entry["label"],
716
+ "command": display_command(recipe), "status": "blocked",
717
+ "reason": "Docker is not installed. The command did not run on the host either.",
718
+ "sandbox": {"image": sandbox, "container": ""}}
719
+ else:
720
+ program = shutil.which(command[0])
721
+ if program is None and command[0] != sys.executable:
722
+ return {"recipe": recipe, "command": shlex.join(command), "status": "unavailable",
723
+ "reason": command[0] + " is not installed or not on PATH."}
724
+ command[0] = program or command[0]
725
+ root = Path(repo).resolve(strict=True)
726
+ where = project_folder(root, target)
727
+ started = time.monotonic()
728
+ started_wall = time.time()
729
+ reports = (tempfile.TemporaryDirectory(prefix="agent-report-")
730
+ if recipe_entry.get("junit_arg") else None)
731
+ sandbox_dir = None
732
+ name = ""
733
+ process = None
734
+ try:
735
+ if sandbox:
736
+ sandbox_dir = tempfile.TemporaryDirectory(prefix="agent-sandbox-")
737
+ copy = Path(sandbox_dir.name)
738
+ copy_for_sandbox(root, copy)
739
+ relative = "" if where == root else where.relative_to(root).as_posix()
740
+ inside = WORKDIR + ("/" + relative if relative else "")
741
+ read_from = copy if not relative else copy / where.relative_to(root)
742
+ # The image resolves its own interpreter; this machine's absolute path means nothing in there.
743
+ argv = [IMAGE_PYTHON if part == sys.executable else str(part) for part in command]
744
+ junit = None
745
+ if reports is not None:
746
+ junit = read_from / "junit.xml"
747
+ argv += [recipe_entry["junit_arg"], inside + "/junit.xml"]
748
+ name = "ai-agent-" + uuid.uuid4().hex
749
+ argv = sandbox_argv(docker, copy, sandbox, name, argv, workdir=inside)
750
+ cwd = None
751
+ else:
752
+ argv = command
753
+ read_from, cwd = where, str(where)
754
+ junit = None
755
+ if reports is not None:
756
+ junit = Path(reports.name) / "junit.xml"
757
+ argv = command + [recipe_entry["junit_arg"], str(junit)]
758
+ progress("Running " + shlex.join(recipe_entry["command"])
759
+ + (" in Docker (" + sandbox.split("@")[0] + ")" if sandbox else ""))
760
+ options = {"cwd": cwd, "env": child_env(), "stdin": subprocess.DEVNULL,
761
+ "stdout": subprocess.PIPE, "stderr": subprocess.STDOUT}
762
+ if os.name == "nt":
763
+ no_window = getattr(subprocess, "CREATE_NO_WINDOW", 0x08000000)
764
+ new_group = getattr(subprocess, "CREATE_NEW_PROCESS_GROUP", 0x00000200)
765
+ options["creationflags"] = new_group | no_window
766
+ else:
767
+ options["start_new_session"] = True
768
+ process = subprocess.Popen(argv, **options)
769
+ output, timed_out, dropped = _collect(process, progress, started + timeout)
770
+ seconds = round(time.monotonic() - started, 1)
771
+ truncated = dropped > 0
772
+ exit_code = process.returncode
773
+ pattern = recipe_entry["test_counts"]
774
+ counts = re.findall(pattern, output) if pattern else []
775
+ ran = tests_ran(recipe_entry, output, counts)
776
+ proof = report_counts(fresh_reports(read_from, recipe_entry.get("reports", ()),
777
+ started_wall, junit))
778
+ if proof is not None:
779
+ proof["source"] = recipe_entry.get("proof_source", "JUnit XML")
780
+ else:
781
+ # A runner that writes no report file still gets read: cargo prints its own summary and
782
+ # nothing else, and "exit 0" alone hides how many tests that green actually was.
783
+ proof = console_proof(recipe_entry, output)
784
+ status = decide_status(timed_out, exit_code, proof, ran, expects_proof(recipe_entry))
785
+ finally:
786
+ if process is not None and process.returncode is None:
787
+ # Anything that raises between the spawn and the answer — a progress sink that throws,
788
+ # a cancelled job — would otherwise leave a build running with its pipe held open.
789
+ kill_tree(process)
790
+ try:
791
+ process.wait(timeout=POST_KILL_GRACE)
792
+ except subprocess.TimeoutExpired:
793
+ pass
794
+ if name:
795
+ # `--rm` removes a container that exits; one that was killed does not get the chance, and a
796
+ # leftover container holds its mount for the rest of the daemon's life. A failure here is
797
+ # not the result's business: the run it is cleaning up after is already answered.
798
+ try:
799
+ subprocess.run([docker, "rm", "-f", name], capture_output=True, timeout=20,
800
+ stdin=subprocess.DEVNULL, env=child_env(),
801
+ creationflags=getattr(subprocess, "CREATE_NO_WINDOW", 0)
802
+ if os.name == "nt" else 0)
803
+ except (OSError, subprocess.SubprocessError):
804
+ pass
805
+ if sandbox_dir is not None:
806
+ sandbox_dir.cleanup()
807
+ if reports is not None:
808
+ reports.cleanup()
809
+ return {"recipe": recipe, "label": recipe_entry["label"],
810
+ # Which folder of this project the command actually ran in, recorded rather than implied:
811
+ # a reactor's root and one module of it answer to the same recipe name, and a fix round
812
+ # sent to "the build failed" with no folder in it is a guess about which build failed.
813
+ "target": "." if where == root else where.relative_to(root).as_posix(),
814
+ # Where the command ran is part of what a green means: a passing Maven run inside
815
+ # `eclipse-temurin@sha256:…` is a claim about that image's JDK, not about this machine's.
816
+ "sandbox": {"image": sandbox, "container": name} if sandbox else None,
817
+ "command": shlex.join(recipe_entry["command"]), "status": status,
818
+ "exit_code": exit_code, "seconds": seconds,
819
+ "tests_observed": bool(proof and proof["tests"]) or ran,
820
+ "proof": proof, "truncated": truncated, "timed_out": timed_out,
821
+ "output": output, "tail": output[-MODEL_OUTPUT_CHARS:], "failures": _failures(output)}
822
+
823
+
824
+ def run_folder(run: dict) -> str:
825
+ """The module a run happened in, normalised: "" when it was the opened folder itself.
826
+
827
+ A target comes from a click, from a record written before this existed, or from a call that never
828
+ named one, and ".", "./" and "" all mean the same place — the root of the project.
829
+ """
830
+ parts = [part for part in str(run.get("target") or ".").replace("\\", "/").split("/")
831
+ if part not in ("", ".")]
832
+ return "/".join(parts)
833
+
834
+
835
+ def summarize(result: dict) -> str:
836
+ """Short human line for the status area.
837
+
838
+ The module is named when the command ran in one: in a nine-module reactor "Maven test: passed"
839
+ does not say whether the build that passed was the one the user was looking at.
840
+ """
841
+ folder = run_folder(result)
842
+ named = result["label"] + (" in " + PurePosixPath(folder).name if folder else "")
843
+ # Where a green happened is part of what it claims: a passing build inside an image says nothing
844
+ # about this machine's JDK, and a reader comparing two runs has to be able to tell them apart.
845
+ if result.get("sandbox"):
846
+ named += " in Docker"
847
+ if result["status"] in ("unavailable", "blocked"):
848
+ return f"{named}: {result['reason']}"
849
+ state = {"passed": "passed", "failed": "FAILED", "timeout": "timed out",
850
+ "unverified": "no tests ran"}[result["status"]]
851
+ proof = result.get("proof")
852
+ counted = (f"{proof['tests']} tests, {proof['failures']} failed, {proof['errors']} errors "
853
+ f"({proof['source']})" if proof else f"exit {result['exit_code']}")
854
+ return f"{named}: {state} ({counted}, {result['seconds']}s)"