ai-code-engineer 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. ai_code_engineer/__init__.py +2 -0
  2. ai_code_engineer/catalog.py +143 -0
  3. ai_code_engineer/chat.py +181 -0
  4. ai_code_engineer/cli.py +384 -0
  5. ai_code_engineer/config.py +405 -0
  6. ai_code_engineer/engine.py +1282 -0
  7. ai_code_engineer/errors.py +27 -0
  8. ai_code_engineer/git_integration.py +443 -0
  9. ai_code_engineer/gui.py +2646 -0
  10. ai_code_engineer/host.py +81 -0
  11. ai_code_engineer/ignore.py +269 -0
  12. ai_code_engineer/intent.py +222 -0
  13. ai_code_engineer/labels.py +871 -0
  14. ai_code_engineer/memory.py +91 -0
  15. ai_code_engineer/modes.py +156 -0
  16. ai_code_engineer/overrides.py +540 -0
  17. ai_code_engineer/planbook.py +192 -0
  18. ai_code_engineer/providers.py +404 -0
  19. ai_code_engineer/redaction.py +54 -0
  20. ai_code_engineer/repair.py +564 -0
  21. ai_code_engineer/report.py +352 -0
  22. ai_code_engineer/runner.py +854 -0
  23. ai_code_engineer/setup.py +386 -0
  24. ai_code_engineer/symbols.py +1286 -0
  25. ai_code_engineer/verification.py +218 -0
  26. ai_code_engineer/webapp/__init__.py +1 -0
  27. ai_code_engineer/webapp/__main__.py +45 -0
  28. ai_code_engineer/webapp/contract.py +36 -0
  29. ai_code_engineer/webapp/controller.py +3556 -0
  30. ai_code_engineer/webapp/fake.py +1141 -0
  31. ai_code_engineer/webapp/launch.py +108 -0
  32. ai_code_engineer/webapp/server.py +349 -0
  33. ai_code_engineer/webapp/static/app.css +780 -0
  34. ai_code_engineer/webapp/static/app.js +2118 -0
  35. ai_code_engineer/webapp/static/boot.js +19 -0
  36. ai_code_engineer/webapp/static/index.html +89 -0
  37. ai_code_engineer/webapp/static/tokens.css +173 -0
  38. ai_code_engineer/workspace.py +385 -0
  39. ai_code_engineer-0.1.0.dist-info/METADATA +7 -0
  40. ai_code_engineer-0.1.0.dist-info/RECORD +44 -0
  41. ai_code_engineer-0.1.0.dist-info/WHEEL +5 -0
  42. ai_code_engineer-0.1.0.dist-info/entry_points.txt +2 -0
  43. ai_code_engineer-0.1.0.dist-info/licenses/LICENSE +21 -0
  44. ai_code_engineer-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,218 @@
1
+ """Run fixed test recipes inside an explicit, preloaded Docker image."""
2
+ from __future__ import annotations
3
+
4
+ import ast
5
+ import json
6
+ import os
7
+ from pathlib import Path
8
+ import re
9
+ import shutil
10
+ import subprocess
11
+ import sys
12
+ import tempfile
13
+ import threading
14
+ import uuid
15
+
16
+ from .engine import atomic_json, event, load_session
17
+ from .errors import PolicyError
18
+ from . import runner
19
+ from .workspace import Workspace, digest
20
+
21
+ RECIPES = {
22
+ # The container has no network, so a build that would have downloaded a dependency on the host has
23
+ # to be told not to try. These are `runner.RECIPES` with that one flag added, and nothing else: a
24
+ # second table of hand-copied commands is how the two drifted apart on exactly that flag.
25
+ "python-unittest": ["python", "-m", "unittest", "discover", "-s", "tests", "-v"],
26
+ "maven-test": ["mvn", "-o", "-B", "test"],
27
+ "gradle-test": ["gradle", "--offline", "--no-daemon", "test"],
28
+ }
29
+
30
+ # Which flag each recipe needs to be honest about a machine with no network, keyed on the program so a
31
+ # recipe added to `runner.RECIPES` is not silently missing an offline mode here.
32
+ OFFLINE = {"mvn": "-o", "gradle": "--offline"}
33
+
34
+
35
+ def container_command(recipe: str) -> list[str]:
36
+ """The runner's own argv for a recipe, with the offline flag a `--network=none` build needs.
37
+
38
+ `sys.executable` becomes the image's interpreter: this machine's absolute path to a python under a
39
+ user profile means nothing inside a container, and a recipe that names it would fail to start.
40
+ """
41
+ command = [runner.IMAGE_PYTHON if part == sys.executable else str(part)
42
+ for part in runner.RECIPES[recipe]["command"]]
43
+ if command[0] in OFFLINE:
44
+ command.insert(1, OFFLINE[command[0]])
45
+ return command
46
+
47
+
48
+ def static_check(session: dict) -> list[dict]:
49
+ ws = Workspace(Path(session["root"]))
50
+ results = []
51
+ for change in session["changes"]:
52
+ if change.get("delete"):
53
+ # A removal's static check is that the file is gone: there is no content to parse, and
54
+ # reading it would raise for the one reason this row is supposed to accept.
55
+ if ws.path(change["path"]).exists():
56
+ raise PolicyError("Changed files no longer match the approved proposal.")
57
+ results.append({"path": change["path"], "status": "not_applicable"})
58
+ continue
59
+ current = ws.read(change["path"])
60
+ if current["sha256"] != change["after_hash"]:
61
+ raise PolicyError("Changed files no longer match the approved proposal.")
62
+ # The workspace admits a suffix in any case, so the gate has to compare the same way —
63
+ # otherwise APP.PY reaches disk unchecked, twice over.
64
+ suffix = Path(change["path"]).suffix.lower()
65
+ try:
66
+ if suffix == ".py":
67
+ ast.parse(current["content"], filename=change["path"])
68
+ elif suffix == ".json":
69
+ json.loads(current["content"])
70
+ else:
71
+ results.append({"path": change["path"], "status": "not_applicable"})
72
+ continue
73
+ results.append({"path": change["path"], "status": "passed"})
74
+ except (SyntaxError, ValueError):
75
+ results.append({"path": change["path"], "status": "failed"})
76
+ return results
77
+
78
+
79
+ def _readable(ws: Workspace, name: str) -> dict | None:
80
+ """A file's read record, or None when the read limits refuse it.
81
+
82
+ The 128 KiB and UTF-8 ceilings bound what may reach a model's context; they are not a reason
83
+ a repository with one generated lock file too big to quote cannot be verified at all. The
84
+ snapshot carries what it can, and the same rule decides both sides of the drift check.
85
+ """
86
+ try:
87
+ return ws.read(name)
88
+ except PolicyError:
89
+ return None
90
+
91
+
92
+ def _fingerprints(ws: Workspace) -> dict[str, str]:
93
+ out = {}
94
+ for name in ws.files(limit=2001):
95
+ item = _readable(ws, name)
96
+ if item is not None:
97
+ out[name] = item["sha256"]
98
+ return out
99
+
100
+
101
+ def snapshot(ws: Workspace, target: Path) -> dict:
102
+ hashes = {}
103
+ total = 0
104
+ names = ws.files(limit=2001)
105
+ if len(names) > 2000:
106
+ raise PolicyError("Verification snapshot exceeds 2000 files.")
107
+ for name in names:
108
+ item = _readable(ws, name)
109
+ if item is None:
110
+ continue
111
+ raw = item["content"].encode("utf-8")
112
+ total += len(raw)
113
+ if total > 20_000_000:
114
+ raise PolicyError("Verification snapshot exceeds 20 MB.")
115
+ out = target / name
116
+ out.parent.mkdir(parents=True, exist_ok=True)
117
+ out.write_bytes(raw)
118
+ hashes[name] = item["sha256"]
119
+ return hashes
120
+
121
+
122
+ def docker_check(source: Path, recipe: str, image: str, timeout: int = 180) -> dict:
123
+ """The session's own verification run: the snapshot copied to a temp folder, built inside a container.
124
+
125
+ `source` is that copy, never the user's tree, and it is mounted as the only writable filesystem the
126
+ build sees — which is also why the copy is a directory and not a tmpfs: the test reports have to be
127
+ readable afterwards, and a green with no proof is the answer this tool refuses to give elsewhere.
128
+ """
129
+ docker = shutil.which("docker")
130
+ if not docker:
131
+ return {"status": "blocked", "reason": "Docker is not installed. Host execution is disabled."}
132
+ if recipe not in RECIPES:
133
+ raise PolicyError("Choose a built-in recipe and a preloaded image pinned by sha256 digest.")
134
+ name = "ai-agent-" + uuid.uuid4().hex
135
+ # The flags live in one place now, with the runner's; this call used to carry its own copy of them
136
+ # and had already drifted from it on the one flag that matters without a network.
137
+ args = runner.sandbox_argv(docker, Path(source), image, name, container_command(recipe))
138
+ output = bytearray()
139
+ exceeded = threading.Event()
140
+ flags = getattr(subprocess, "CREATE_NO_WINDOW", 0x08000000) if os.name == "nt" else 0
141
+ process = subprocess.Popen(args, stdout=subprocess.PIPE, stderr=subprocess.STDOUT,
142
+ stdin=subprocess.DEVNULL, env=runner.child_env(),
143
+ creationflags=flags)
144
+
145
+ def read_output():
146
+ while chunk := process.stdout.read(4096):
147
+ remaining = 64000 - len(output)
148
+ output.extend(chunk[:max(0, remaining)])
149
+ if len(chunk) > remaining:
150
+ exceeded.set()
151
+ process.kill()
152
+ break
153
+
154
+ reader = threading.Thread(target=read_output, daemon=True)
155
+ reader.start()
156
+ timed_out = False
157
+ try:
158
+ process.wait(timeout=timeout)
159
+ except subprocess.TimeoutExpired:
160
+ timed_out = True
161
+ process.kill()
162
+ process.wait()
163
+ finally:
164
+ # The cleanup is a docker call like the run: scrubbed environment, no console handle to
165
+ # block on, and a failure here must not swallow the result it is trying to finish writing.
166
+ try:
167
+ subprocess.run([docker, "rm", "-f", name], capture_output=True, timeout=20,
168
+ stdin=subprocess.DEVNULL, env=runner.child_env(),
169
+ creationflags=flags)
170
+ except (OSError, subprocess.SubprocessError):
171
+ pass
172
+ reader.join(timeout=5)
173
+ process.stdout.close()
174
+ text = output.decode("utf-8", errors="replace")
175
+ # A green process with no tests must not count as passing.
176
+ if recipe == "python-unittest":
177
+ counts = re.findall(r"Ran (\d+) tests?", text)
178
+ elif recipe == "maven-test":
179
+ counts = re.findall(r"Tests run:\s*(\d+)", text)
180
+ else:
181
+ counts = [] # Gradle requires XML report parsing, not console heuristics.
182
+ proven_tests = any(int(n) > 0 for n in counts)
183
+ status = "passed" if process.returncode == 0 and proven_tests else "failed"
184
+ if process.returncode == 0 and not proven_tests:
185
+ status = "unverified"
186
+ if timed_out or exceeded.is_set():
187
+ status = "blocked"
188
+ return {"status": status, "exit_code": process.returncode, "recipe": recipe,
189
+ "image": image, "timeout": timed_out, "output_limited": exceeded.is_set(),
190
+ "output": text, "tests_observed": proven_tests}
191
+
192
+
193
+ def verify(path: Path, recipe: str | None = None, image: str | None = None) -> dict:
194
+ session = load_session(path)
195
+ if session["state"] not in {"APPLIED_UNVERIFIED", "CHECKS_PASSED", "VERIFICATION_FAILED", "VERIFICATION_BLOCKED"}:
196
+ raise PolicyError("Apply the reviewed proposal before verification.")
197
+ result = {"static": static_check(session)}
198
+ if any(r["status"] == "failed" for r in result["static"]):
199
+ result["status"] = "failed"
200
+ elif recipe is None:
201
+ result.update(status="unverified", reason="Static checks only; build/tests have not run.")
202
+ elif not image:
203
+ result.update(status="blocked", reason="An approved preloaded Docker image digest is required.")
204
+ else:
205
+ ws = Workspace(Path(session["root"]))
206
+ with tempfile.TemporaryDirectory(prefix="agent-verify-") as temp:
207
+ before = snapshot(ws, Path(temp))
208
+ result["sandbox"] = docker_check(Path(temp), recipe, image)
209
+ after = _fingerprints(ws)
210
+ result["snapshot_hash"] = digest(json.dumps(before, sort_keys=True).encode())
211
+ result["status"] = result["sandbox"]["status"] if before == after else "stale"
212
+ # Passing a selected recipe is not proof of all natural-language acceptance criteria.
213
+ session["state"] = {"passed": "CHECKS_PASSED", "failed": "VERIFICATION_FAILED"}.get(
214
+ result["status"], "VERIFICATION_BLOCKED")
215
+ session["verification"] = result
216
+ event(session, "verification", status=result["status"])
217
+ atomic_json(path, session)
218
+ return result
@@ -0,0 +1 @@
1
+ """Local web shell for the desktop app: stdlib HTTP server + a browser in app mode."""
@@ -0,0 +1,45 @@
1
+ """Run the UI server: python -m ai_code_engineer.webapp [--fake] [--no-browser] [--port N]"""
2
+ from __future__ import annotations
3
+
4
+ import argparse
5
+ import time
6
+ import webbrowser
7
+ from pathlib import Path
8
+
9
+ from .server import serve
10
+
11
+
12
+ def build(fake: bool):
13
+ if fake:
14
+ from .fake import FakeController
15
+ return FakeController()
16
+ from .controller import AgentController
17
+ return AgentController(Path(__file__).resolve().parents[3])
18
+
19
+
20
+ def main(argv=None) -> int:
21
+ parser = argparse.ArgumentParser(prog="python -m ai_code_engineer.webapp")
22
+ parser.add_argument("--fake", action="store_true", help="drive the UI with scripted data, no engine")
23
+ parser.add_argument("--no-browser", action="store_true")
24
+ parser.add_argument("--port", type=int, default=0)
25
+ args = parser.parse_args(argv)
26
+
27
+ controller = build(args.fake)
28
+ server, url, _token = serve(controller, port=args.port)
29
+ if not args.fake:
30
+ controller.check_setup()
31
+ print("AI Code Engineer UI -> " + url)
32
+ if not args.no_browser:
33
+ webbrowser.open(url)
34
+ try:
35
+ while True:
36
+ time.sleep(3600)
37
+ except KeyboardInterrupt:
38
+ pass
39
+ finally:
40
+ server.shutdown()
41
+ return 0
42
+
43
+
44
+ if __name__ == "__main__":
45
+ raise SystemExit(main())
@@ -0,0 +1,36 @@
1
+ """The seam between the UI server and whatever drives it.
2
+
3
+ The server knows nothing about planning, approvals or the file system: it only ever
4
+ calls the five methods below and forwards whatever the controller emits. That keeps the
5
+ front-end reviewable now (against ``fake.FakeController``) while the real controller is
6
+ still being extracted from the Tk window.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ from typing import Callable, Iterable, Protocol
11
+
12
+
13
+ Sink = Callable[[dict], None]
14
+ """Pushes one UI event to every connected browser."""
15
+
16
+
17
+ class Controller(Protocol):
18
+ def snapshot(self) -> dict:
19
+ """The whole UI state, serialisable, so a fresh client can paint in one round trip."""
20
+ ...
21
+
22
+ def action(self, type: str, payload: dict, emit: Sink) -> dict | None:
23
+ """Run one user intent. Long work goes to a thread and reports through ``emit``."""
24
+ ...
25
+
26
+ def project_info(self, key: str) -> dict:
27
+ """One granted folder, measured: its path, notes and size. Raises ``PolicyError`` for a
28
+ key that was never granted — the server turns that into a 400 the browser can show."""
29
+ ...
30
+
31
+ def list_dir(self, path: str, want_files: Iterable[str] | None = None) -> dict:
32
+ """One directory level, for the in-page folder browser (no native dialogs in a browser)."""
33
+ ...
34
+
35
+ def set_reply(self, request_id: str, reply: dict) -> None:
36
+ """Deliver the answer to a ``confirm`` or ``pick`` event the controller asked for."""