ai-code-engineer 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. ai_code_engineer/__init__.py +2 -0
  2. ai_code_engineer/catalog.py +143 -0
  3. ai_code_engineer/chat.py +181 -0
  4. ai_code_engineer/cli.py +384 -0
  5. ai_code_engineer/config.py +405 -0
  6. ai_code_engineer/engine.py +1282 -0
  7. ai_code_engineer/errors.py +27 -0
  8. ai_code_engineer/git_integration.py +443 -0
  9. ai_code_engineer/gui.py +2646 -0
  10. ai_code_engineer/host.py +81 -0
  11. ai_code_engineer/ignore.py +269 -0
  12. ai_code_engineer/intent.py +222 -0
  13. ai_code_engineer/labels.py +871 -0
  14. ai_code_engineer/memory.py +91 -0
  15. ai_code_engineer/modes.py +156 -0
  16. ai_code_engineer/overrides.py +540 -0
  17. ai_code_engineer/planbook.py +192 -0
  18. ai_code_engineer/providers.py +404 -0
  19. ai_code_engineer/redaction.py +54 -0
  20. ai_code_engineer/repair.py +564 -0
  21. ai_code_engineer/report.py +352 -0
  22. ai_code_engineer/runner.py +854 -0
  23. ai_code_engineer/setup.py +386 -0
  24. ai_code_engineer/symbols.py +1286 -0
  25. ai_code_engineer/verification.py +218 -0
  26. ai_code_engineer/webapp/__init__.py +1 -0
  27. ai_code_engineer/webapp/__main__.py +45 -0
  28. ai_code_engineer/webapp/contract.py +36 -0
  29. ai_code_engineer/webapp/controller.py +3556 -0
  30. ai_code_engineer/webapp/fake.py +1141 -0
  31. ai_code_engineer/webapp/launch.py +108 -0
  32. ai_code_engineer/webapp/server.py +349 -0
  33. ai_code_engineer/webapp/static/app.css +780 -0
  34. ai_code_engineer/webapp/static/app.js +2118 -0
  35. ai_code_engineer/webapp/static/boot.js +19 -0
  36. ai_code_engineer/webapp/static/index.html +89 -0
  37. ai_code_engineer/webapp/static/tokens.css +173 -0
  38. ai_code_engineer/workspace.py +385 -0
  39. ai_code_engineer-0.1.0.dist-info/METADATA +7 -0
  40. ai_code_engineer-0.1.0.dist-info/RECORD +44 -0
  41. ai_code_engineer-0.1.0.dist-info/WHEEL +5 -0
  42. ai_code_engineer-0.1.0.dist-info/entry_points.txt +2 -0
  43. ai_code_engineer-0.1.0.dist-info/licenses/LICENSE +21 -0
  44. ai_code_engineer-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,352 @@
1
+ """One session's record, handed over as something usable outside this tool.
2
+
3
+ The session file is already the complete ledger — `.agent-runs/<id>/session.json` keeps the task text
4
+ verbatim, the model that answered, every tool call with the digest of what was read, the proposal hash,
5
+ what was written, each command run with its exit code and test counts, and every refusal with its
6
+ reason. What it is not is *readable*: it is 200 KB of interleaved events, and the interesting question
7
+ ("what did this task actually do, and why did it stop") requires walking them in order.
8
+
9
+ Two rules the exporters keep:
10
+
11
+ - **Redact at the boundary.** The stored record is redacted where it was captured, and this re-redacts
12
+ anything that came from a command's output. A report is the one artefact that leaves the machine —
13
+ it gets pasted into an issue, mailed to a teammate, attached to a PR — so the guarantee cannot be
14
+ "the log happened to be clean".
15
+ - **Say nothing that was not recorded.** No estimated token counts, no inferred cause. Where the record
16
+ is silent — a run that produced no test evidence, a stage that never emitted a line — the report says
17
+ that, in the same words the windows use.
18
+ """
19
+ from __future__ import annotations
20
+
21
+ import json
22
+ from pathlib import Path
23
+
24
+ from .engine import load_session
25
+ from .errors import AgentError
26
+ from .repair import classify
27
+ from .runner import run_folder
28
+ from .labels import STATES, state_label
29
+ from .redaction import redact
30
+
31
+ # Events that mark a stage a person asks about later. Anything else is bookkeeping.
32
+ STAGE_KINDS = ("plan_attached", "memory_attached", "proposal", "approved", "written", "removed",
33
+ "run", "verification", "rolled_back", "rolled_back_file", "stopped",
34
+ "rejected_action", "blocked_retried", "file_not_found", "step", "tool")
35
+ # The kinds that answer "what did it read" — the model's own tool calls and the reads the loop did
36
+ # for it, which are different acts and stay listed apart.
37
+ READ_KINDS = ("tool", "auto_read")
38
+
39
+
40
+ def find_session(runs: Path, ident: str) -> Path:
41
+ """The session file for an id, a prefix, or a path to the folder or the file itself.
42
+
43
+ Ids are 32 hex characters, which nobody types. A unique prefix is accepted, an ambiguous one is
44
+ refused with the count, because guessing which task a person meant is the one thing an audit tool
45
+ must not do.
46
+ """
47
+ runs = Path(runs)
48
+ direct = Path(ident)
49
+ if direct.is_file():
50
+ return direct
51
+ if direct.is_dir() and (direct / "session.json").is_file():
52
+ return direct / "session.json"
53
+ wanted = ident.lower().removeprefix("runs/").removesuffix("/").removesuffix("/session.json")
54
+ if not wanted:
55
+ raise AgentError("Name a session by its id, its id prefix, or its folder.")
56
+ matches = sorted(path for path in runs.glob("*/session.json")
57
+ if path.parent.name.lower().startswith(wanted))
58
+ if not matches:
59
+ raise AgentError("No session starts with " + wanted + " in " + str(runs))
60
+ if len(matches) > 1:
61
+ raise AgentError(str(len(matches)) + " sessions start with " + wanted +
62
+ " — give more characters.")
63
+ return matches[0]
64
+
65
+
66
+ def _clock(session: dict) -> list[dict]:
67
+ """The event list, in recorded order, each with the seconds since the one before it."""
68
+ from datetime import datetime, timezone
69
+
70
+ previous = None
71
+ rows = []
72
+ for entry in session.get("events") or []:
73
+ at = entry.get("at") or ""
74
+ try:
75
+ stamp = datetime.fromisoformat(at)
76
+ if stamp.tzinfo is None:
77
+ stamp = stamp.replace(tzinfo=timezone.utc)
78
+ except ValueError:
79
+ stamp = None
80
+ gap = None
81
+ if stamp is not None:
82
+ if previous is not None:
83
+ gap = round((stamp - previous).total_seconds(), 2)
84
+ previous = stamp
85
+ # The detail is the event's own fields, and some of them are text a build tool or a model
86
+ # wrote: `rejected_action` carries the refusal reason, `stopped` carries an error string. A
87
+ # report is the artefact that leaves this machine, so it is redacted here as well as there.
88
+ detail = {key: (_clean(redact(str(value)))[:200] if isinstance(value, str) else value)
89
+ for key, value in entry.items() if key not in {"at", "kind"}}
90
+ rows.append({"at": at, "kind": entry.get("kind", ""), "seconds_since_previous": gap,
91
+ "detail": detail})
92
+ return rows
93
+
94
+
95
+ def reads(session: dict) -> dict:
96
+ """What was opened, with the digest that identifies which version of it."""
97
+ asked, automatic = [], []
98
+ for entry in session.get("events") or []:
99
+ if entry.get("kind") not in READ_KINDS:
100
+ continue
101
+ row = {"path": entry.get("path", ""), "sha256": entry.get("sha256", "")}
102
+ if entry.get("name") == "list_files" or "count" in entry:
103
+ row["listed"] = entry.get("count")
104
+ if "matches" in entry:
105
+ row["search_matches"] = entry.get("matches")
106
+ if row["path"]:
107
+ (automatic if entry["kind"] == "auto_read" else asked).append(row)
108
+ return {"by_model": asked, "by_tool": automatic,
109
+ "distinct_files": len({row["path"] for row in asked + automatic if row["path"]})}
110
+
111
+
112
+ def refusals(session: dict) -> list[dict]:
113
+ """Every action the tool declined, and the sentence that says why.
114
+
115
+ A report that lists only the successes is how a small model's flailing reads as progress. These are
116
+ the rows that make the difference between "three rounds" and "three rounds, two of them rejected".
117
+ """
118
+ out = []
119
+ for entry in session.get("events") or []:
120
+ kind = entry.get("kind")
121
+ if kind == "rejected_action":
122
+ out.append({"at": entry.get("at", ""), "what": "action", "why": entry.get("reason", "")})
123
+ elif kind == "blocked_retried":
124
+ out.append({"at": entry.get("at", ""), "what": "blocked answer",
125
+ "why": "retried once (attempt " + str(entry.get("attempt", "")) + ")"})
126
+ elif kind == "file_not_found":
127
+ out.append({"at": entry.get("at", ""), "what": "read " + str(entry.get("path", "")),
128
+ "why": "not found" + ("" if entry.get("can_create") else " (creation protected)")})
129
+ elif kind == "stopped":
130
+ out.append({"at": entry.get("at", ""), "what": "the task",
131
+ "why": entry.get("reason") or "the record does not say why it stopped"})
132
+ return [{"at": row["at"], "what": _clean(redact(row["what"]))[:200],
133
+ "why": _clean(redact(row["why"]))[:200]} for row in out]
134
+
135
+
136
+ def runs_of(session: dict) -> list[dict]:
137
+ """Each command, as stored: what ran, what it returned, and how much proof it produced."""
138
+ rows = []
139
+ for run in session.get("runs") or []:
140
+ proof = run.get("proof") or {}
141
+ rows.append({"recipe": run.get("recipe", ""), "label": run.get("label", ""),
142
+ # Which module of a multi-project folder ran, "" when it was the folder itself.
143
+ "folder": run_folder(run),
144
+ # And in what: a green inside a pinned image is a claim about that image, and an
145
+ # audit read a year later cannot tell the two apart without the name on the row.
146
+ "sandbox": (run.get("sandbox") or {}).get("image", ""),
147
+ "command": run.get("command", ""), "status": run.get("status", ""),
148
+ "exit_code": run.get("exit_code"), "seconds": run.get("seconds"),
149
+ "tests": proof.get("tests") or 0, "failures": proof.get("failures") or 0,
150
+ "errors": proof.get("errors") or 0, "proof_source": proof.get("source", ""),
151
+ # The same reading the loop acts on, so the report cannot tell a different story
152
+ # about a run the agent already classified.
153
+ "looks_like": classify(run),
154
+ "truncated": bool(run.get("truncated")), "timed_out": bool(run.get("timed_out")),
155
+ "reported_problems": [_clean(redact(str(line))[:200])
156
+ for line in (run.get("failures") or [])[:12]],
157
+ "end_of_output": _clean(redact(str(run.get("tail") or ""))[-1200:],
158
+ keep_newlines=True)})
159
+ return rows
160
+
161
+
162
+ def summary(session: dict) -> dict:
163
+ """The machine-readable export: the record plus what only a walk over it can tell you."""
164
+ state = session.get("state", "")
165
+ return {
166
+ "id": session.get("id", ""),
167
+ "project": session.get("root", ""),
168
+ "chat": session.get("chat_id", ""),
169
+ "created": session.get("created", ""),
170
+ "state": state,
171
+ "state_label": state_label(state) if state in STATES else state,
172
+ "model": session.get("model", ""),
173
+ "request": session.get("task", ""),
174
+ "plan": ({"file": session["plan_reference"]["path"], "step": session.get("plan_step")}
175
+ if session.get("plan_reference") else None),
176
+ "project_notes_characters": len(session.get("memory") or ""),
177
+ "proposal": {"hash": session.get("proposal_hash", ""),
178
+ "summary": redact(session.get("summary") or ""),
179
+ "files": [{"path": change.get("path", ""),
180
+ "delete": bool(change.get("delete")),
181
+ "bytes_before": len(change.get("before") or ""),
182
+ "bytes_after": len(change.get("after") or "")}
183
+ for change in session.get("changes") or []]},
184
+ "checks": list(session.get("checks") or []),
185
+ "written": [entry.get("path", "") for entry in session.get("events") or []
186
+ if entry.get("kind") == "written"],
187
+ "removed": [entry.get("path", "") for entry in session.get("events") or []
188
+ if entry.get("kind") == "removed"],
189
+ "rolled_back": [entry.get("path", "") for entry in session.get("events") or []
190
+ if entry.get("kind") == "rolled_back_file"],
191
+ "verification": session.get("verification") or None,
192
+ "error": _clean(redact(session.get("error", "")))[:300],
193
+ "runs": runs_of(session),
194
+ "reads": reads(session),
195
+ "refusals": refusals(session),
196
+ "timeline": _clock(session),
197
+ "fix_rounds": sum(1 for entry in session.get("events") or []
198
+ if entry.get("kind") == "run"),
199
+ }
200
+
201
+
202
+ def _clean(value: str, *, keep_newlines: bool = False) -> str:
203
+ """Strip the terminal's own vocabulary from text a build tool wrote.
204
+
205
+ A build log can contain `ESC[2J` or a bell, and a report is read in a browser, pasted into an
206
+ issue and opened in editors. `cli.safe_print` already does this to the screen; the file the user
207
+ asks for with `--out` never passes through the screen, so it is cleaned where it is assembled.
208
+ """
209
+ text = str(value)
210
+ allowed = "\n\t" if keep_newlines else ""
211
+ return "".join(char for char in text
212
+ if char in allowed or (ord(char) >= 32 and not 127 <= ord(char) <= 159))
213
+
214
+
215
+ def _line(value: str) -> str:
216
+ """One line of a markdown table: no pipes, no newlines, no control characters."""
217
+ flat = " ".join(_clean(value).split())
218
+ return redact(flat.replace("|", "\\|"))[:160]
219
+
220
+
221
+ def render_markdown(session: dict) -> str:
222
+ """The same facts as a page someone can read without this repository in front of them."""
223
+ data = summary(session)
224
+ parts = ["# Session " + data["id"][:12] + " — " + _line(Path(data["project"]).name or "?") + "\n\n"]
225
+ parts.append("| | |\n| --- | --- |\n")
226
+ for label, value in (("Project", data["project"]), ("State", data["state_label"]),
227
+ ("Model", data["model"]), ("Created", data["created"]),
228
+ ("Chat", data["chat"] or "—"),
229
+ ("Proposal hash", data["proposal"]["hash"] or "none"),
230
+ ("Files written", str(len(data["written"]))),
231
+ ("Files removed", str(len(data["removed"]))),
232
+ ("Command runs", str(len(data["runs"]))),
233
+ ("Refusals/retries", str(len(data["refusals"])))):
234
+ parts.append("| " + label + " | `" + _line(value) + "` |\n")
235
+
236
+ # The task text is what the window was given, character for character — that is the point of the
237
+ # section, and the dogfood ledger exists because paraphrasing a prompt loses the mistake. It is
238
+ # still cleaned of terminal control codes, because the person reading it did not write it.
239
+ parts.append("\n## The request, verbatim\n\n````text\n"
240
+ + _clean(data["request"].rstrip("\n"), keep_newlines=True) + "\n````\n")
241
+ if data["proposal"]["summary"]:
242
+ parts.append("\n*The model's own summary:* " + _line(data["proposal"]["summary"]) + "\n")
243
+ if data["plan"]:
244
+ parts.append("\n*Attached plan:* `" + _line(data["plan"]["file"]) + "`"
245
+ + (", step " + str(data["plan"]["step"]) if data["plan"]["step"] else "") + "\n")
246
+
247
+ if data["proposal"]["files"]:
248
+ parts.append("\n## Proposed changes\n\n")
249
+ parts.append("| file | action | before | after |\n| --- | --- | --- | --- |\n")
250
+ for change in data["proposal"]["files"]:
251
+ parts.append("| `{}` | {} | {} B | {} B |\n".format(
252
+ _line(change["path"]), "remove" if change["delete"] else "write",
253
+ change["bytes_before"], change["bytes_after"]))
254
+ if data["checks"]:
255
+ parts.append("\n*Checks the proposal asked for:* "
256
+ + "; ".join(_line(item) for item in data["checks"]) + "\n")
257
+
258
+ parts.append("\n## Files read\n\n")
259
+ parts.append("*By the model, " + str(len(data["reads"]["by_model"])) + " calls, "
260
+ + str(data["reads"]["distinct_files"]) + " distinct files:*\n")
261
+ for row in data["reads"]["by_model"][:40]:
262
+ digest = " — read as `" + row["sha256"][:8] + "`" if row["sha256"] else ""
263
+ parts.append("- `" + _line(row["path"]) + "`" + digest + "\n")
264
+ if data["reads"]["by_tool"]:
265
+ parts.append("\n*Read by the tool on the model's behalf:* "
266
+ + ", ".join("`" + _line(row["path"]) + "`" for row in data["reads"]["by_tool"][:20])
267
+ + "\n")
268
+
269
+ parts.append("\n## Command runs\n\n")
270
+ if not data["runs"]:
271
+ parts.append("No project command was run for this session. **Verification is incomplete** — "
272
+ "nothing here proves the change works beyond the static checks.\n")
273
+ for index, run in enumerate(data["runs"], 1):
274
+ counted = (str(run["tests"]) + " tests, " + str(run["failures"]) + " failed, "
275
+ + str(run["errors"]) + " errors"
276
+ if run["proof_source"] else "no test count was produced")
277
+ parts.append("### {} — `{}`{}{} · {}\n\n".format(
278
+ index, _line(run["label"] or run["recipe"]),
279
+ " in `" + run["folder"] + "`" if run["folder"] else "",
280
+ " inside `" + _line(run["sandbox"]) + "`" if run["sandbox"] else "", run["status"]))
281
+ parts.append("- Command: `{}`\n".format(_line(run["command"])))
282
+ parts.append("- Exit {} in {} s · {} · source: `{}`\n".format(
283
+ run["exit_code"], run["seconds"], counted, _line(run["proof_source"] or "-")))
284
+ if run["looks_like"]:
285
+ parts.append("- What the output looks like: {}\n".format(
286
+ "not a recognised kind of failure" if run["looks_like"] == "unknown"
287
+ else run["looks_like"]))
288
+ if run["timed_out"]:
289
+ parts.append("- **Timed out.** The state after a killed run is unknown until it is re-run.\n")
290
+ if run["truncated"]:
291
+ parts.append("- Output was cut off: what is shown is not all the command printed.\n")
292
+ if not run["tests"] and run["status"] == "passed":
293
+ parts.append("- It reported success with **no test evidence**. That is recorded as "
294
+ "`unverified` elsewhere in this tool, and it is not a pass.\n")
295
+ if run["reported_problems"]:
296
+ parts.append("\n```text\n" + "\n".join(run["reported_problems"]) + "\n```\n")
297
+ if run["end_of_output"]:
298
+ parts.append("\n<details><summary>End of output</summary>\n\n```text\n"
299
+ + run["end_of_output"] + "\n```\n\n</details>\n")
300
+
301
+ if data["verification"]:
302
+ verification = data["verification"]
303
+ parts.append("\n## Verification\n\n`" + _line(verification.get("status", "")) + "`")
304
+ if verification.get("reason"):
305
+ parts.append(" — " + _line(verification["reason"]))
306
+ parts.append("\n")
307
+ if verification.get("sandbox"):
308
+ sandbox = verification["sandbox"]
309
+ parts.append("- Sandbox: recipe `{}`, image `{}`, exit {}\n".format(
310
+ _line(sandbox.get("recipe", "")), _line(str(sandbox.get("image", ""))[:70]),
311
+ sandbox.get("exit_code")))
312
+ parts.append("- Container ran with no network, read-only project mount and no secrets"
313
+ " in its environment.\n")
314
+
315
+ if data["refusals"]:
316
+ parts.append("\n## Refusals, retries and dead ends\n\n")
317
+ parts.append("| when | what | why |\n| --- | --- | --- |\n")
318
+ for row in data["refusals"]:
319
+ parts.append("| {} | {} | {} |\n".format(_line(row["at"])[11:19], _line(row["what"]),
320
+ _line(row["why"])))
321
+
322
+ parts.append("\n## Timeline\n\n")
323
+ parts.append("| when | event | gap | detail |\n| --- | --- | --- | --- |\n")
324
+ for row in data["timeline"]:
325
+ if row["kind"] not in STAGE_KINDS:
326
+ continue
327
+ detail = ", ".join(_line(key) + "=" + _line(value)
328
+ for key, value in list(row["detail"].items())[:3])
329
+ parts.append("| {} | `{}` | {} | {} |\n".format(_line(row["at"])[11:19], _line(row["kind"]),
330
+ row["seconds_since_previous"], detail))
331
+
332
+ if data["error"]:
333
+ parts.append("\n## Why it ended\n\n" + _line(data["error"]) + "\n")
334
+ if data["rolled_back"]:
335
+ parts.append("\n*Rolled back:* " + ", ".join("`" + _line(path) + "`"
336
+ for path in data["rolled_back"]) + "\n")
337
+ parts.append("\n---\n\nGenerated from `.agent-runs/" + data["id"] + "/session.json` by "
338
+ "`agent export-session`. Command output in this file was redacted where it was "
339
+ "recorded and again here; it is still untrusted text from a project's build tool.\n")
340
+ return "".join(parts)
341
+
342
+
343
+ def export(session: dict, fmt: str = "markdown") -> str:
344
+ if fmt == "json":
345
+ return json.dumps(summary(session), ensure_ascii=False, indent=2)
346
+ if fmt == "markdown":
347
+ return render_markdown(session)
348
+ raise AgentError("Unknown export format: " + str(fmt) + " (json or markdown)")
349
+
350
+
351
+ def export_file(path: Path, fmt: str = "markdown") -> str:
352
+ return export(load_session(Path(path)), fmt)