ai-code-engineer 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ai_code_engineer/__init__.py +2 -0
- ai_code_engineer/catalog.py +143 -0
- ai_code_engineer/chat.py +181 -0
- ai_code_engineer/cli.py +384 -0
- ai_code_engineer/config.py +405 -0
- ai_code_engineer/engine.py +1282 -0
- ai_code_engineer/errors.py +27 -0
- ai_code_engineer/git_integration.py +443 -0
- ai_code_engineer/gui.py +2646 -0
- ai_code_engineer/host.py +81 -0
- ai_code_engineer/ignore.py +269 -0
- ai_code_engineer/intent.py +222 -0
- ai_code_engineer/labels.py +871 -0
- ai_code_engineer/memory.py +91 -0
- ai_code_engineer/modes.py +156 -0
- ai_code_engineer/overrides.py +540 -0
- ai_code_engineer/planbook.py +192 -0
- ai_code_engineer/providers.py +404 -0
- ai_code_engineer/redaction.py +54 -0
- ai_code_engineer/repair.py +564 -0
- ai_code_engineer/report.py +352 -0
- ai_code_engineer/runner.py +854 -0
- ai_code_engineer/setup.py +386 -0
- ai_code_engineer/symbols.py +1286 -0
- ai_code_engineer/verification.py +218 -0
- ai_code_engineer/webapp/__init__.py +1 -0
- ai_code_engineer/webapp/__main__.py +45 -0
- ai_code_engineer/webapp/contract.py +36 -0
- ai_code_engineer/webapp/controller.py +3556 -0
- ai_code_engineer/webapp/fake.py +1141 -0
- ai_code_engineer/webapp/launch.py +108 -0
- ai_code_engineer/webapp/server.py +349 -0
- ai_code_engineer/webapp/static/app.css +780 -0
- ai_code_engineer/webapp/static/app.js +2118 -0
- ai_code_engineer/webapp/static/boot.js +19 -0
- ai_code_engineer/webapp/static/index.html +89 -0
- ai_code_engineer/webapp/static/tokens.css +173 -0
- ai_code_engineer/workspace.py +385 -0
- ai_code_engineer-0.1.0.dist-info/METADATA +7 -0
- ai_code_engineer-0.1.0.dist-info/RECORD +44 -0
- ai_code_engineer-0.1.0.dist-info/WHEEL +5 -0
- ai_code_engineer-0.1.0.dist-info/entry_points.txt +2 -0
- ai_code_engineer-0.1.0.dist-info/licenses/LICENSE +21 -0
- ai_code_engineer-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,564 @@
|
|
|
1
|
+
"""Close the loop between a real command run and the next reviewable proposal.
|
|
2
|
+
|
|
3
|
+
Nothing here writes project files. A failed run becomes evidence for one more
|
|
4
|
+
plan() turn, which still produces a proposal the user approves before anything
|
|
5
|
+
is written; that keeps the review-first guarantee intact while letting the
|
|
6
|
+
agent iterate against a compiler or test suite.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
import hashlib
|
|
10
|
+
import re
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
from .engine import (atomic_json, chat_sessions, event, load_session, shrink_warning,
|
|
14
|
+
unexpected_notice)
|
|
15
|
+
from .errors import AgentError, PolicyError
|
|
16
|
+
from .labels import INTERRUPTED_STATES
|
|
17
|
+
from .redaction import redact
|
|
18
|
+
from . import runner
|
|
19
|
+
|
|
20
|
+
MAX_FIX_ROUNDS = 3
|
|
21
|
+
FAILURES_KEPT = 12
|
|
22
|
+
TAIL_KEPT = 2500
|
|
23
|
+
|
|
24
|
+
# The two windows ask the same question in the same words, so the wording lives beside the
|
|
25
|
+
# budget it names. Continuing is an answer, not a guess: the round is what the button says.
|
|
26
|
+
FIX_OFFER_TITLE = "Keep going?"
|
|
27
|
+
# The offer's third answer. It belongs to a batch, and a batch is the web window's queue: Tk has no
|
|
28
|
+
# queue to be the rest of, so the button is drawn by the browser alone and the two answers it already
|
|
29
|
+
# has keep meaning what they meant.
|
|
30
|
+
FIX_OFFER_ALT = "Don't ask again for this batch"
|
|
31
|
+
|
|
32
|
+
# What the loop says when it declines to spend another turn. Twice it said these in two places, worded
|
|
33
|
+
# differently in each, and the same stalled loop read as two different problems depending on which
|
|
34
|
+
# window was open. The advice names Checks because that is where both windows keep the full output.
|
|
35
|
+
FIX_STOP_HINT = "Try a narrower task, another model, or read the output in Checks."
|
|
36
|
+
REPEATED_PROPOSAL = ("This round proposed exactly the same files as an earlier one. Nothing new was "
|
|
37
|
+
"tried — the proposal is still there to review, but applying it again will not "
|
|
38
|
+
"change what the command says.")
|
|
39
|
+
REPEATED_ADVICE = ("The model repeated its previous proposal. Change the task, the model, or the "
|
|
40
|
+
"files it was shown — not the click.")
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def stop_line(reason: str) -> str:
|
|
44
|
+
return "🛑 " + reason.capitalize() + "."
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def stop_status(reason: str) -> str:
|
|
48
|
+
return reason.capitalize() + ". " + FIX_STOP_HINT
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def round_status(model: str, round_number: int, kind: str, limit: int = MAX_FIX_ROUNDS) -> str:
|
|
52
|
+
"""What the loop is doing right now, said the same way in both windows.
|
|
53
|
+
|
|
54
|
+
The kind is in the sentence because the odds are in it: "asking for the smallest change" means
|
|
55
|
+
something different when the machine is missing the tool than when an assertion is.
|
|
56
|
+
"""
|
|
57
|
+
return (f"Fix round {round_number}/{limit}: asking {model} for the smallest change that makes "
|
|
58
|
+
"the command pass"
|
|
59
|
+
+ (f" (looks like a {kind} failure)" if kind and kind != "unknown" else "") + "…")
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _approval_clause(auto_apply: bool) -> str:
|
|
63
|
+
"""What stands between a proposal and the file. Both offers quote this sentence, and the
|
|
64
|
+
folder's switch decides which of its two readings is true — so the switch is named in it."""
|
|
65
|
+
if auto_apply:
|
|
66
|
+
return ("This folder's Auto-Apply switch is on: what the model proposes writes itself and "
|
|
67
|
+
"the command runs again. It still refuses to empty a file you wrote, and a code "
|
|
68
|
+
"block you clicked still waits for Apply.")
|
|
69
|
+
return ("Nothing is written until you approve it — this folder's Auto-Apply switch is off.")
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def timeline(rows: list, limit: int = MAX_FIX_ROUNDS) -> str:
|
|
73
|
+
"""What the loop already tried, in the one wording both windows and the offer read from.
|
|
74
|
+
|
|
75
|
+
A round counter was the whole history a user could see, and a counter is a promise rather than a
|
|
76
|
+
record: three rounds of the same seven failures read exactly like three rounds of progress. These
|
|
77
|
+
are the runs themselves — the same rows the report exports, so the card, the chat and the file
|
|
78
|
+
cannot tell three different stories about one attempt.
|
|
79
|
+
"""
|
|
80
|
+
shown = rows[-limit:]
|
|
81
|
+
if not shown:
|
|
82
|
+
return ""
|
|
83
|
+
lines = ["Tried so far:"]
|
|
84
|
+
for index, row in enumerate(shown, start=len(rows) - len(shown) + 1):
|
|
85
|
+
detail = []
|
|
86
|
+
if row.get("tests"):
|
|
87
|
+
detail.append(str(row["tests"]) + " tests")
|
|
88
|
+
if row.get("failures"):
|
|
89
|
+
detail.append(str(row["failures"]) + " failing")
|
|
90
|
+
if row.get("files"):
|
|
91
|
+
detail.append(str(row["files"]) + " file(s) changed")
|
|
92
|
+
kind = str(row.get("category") or "")
|
|
93
|
+
if kind:
|
|
94
|
+
detail.append("looks like " + kind)
|
|
95
|
+
lines.append(str(index) + ". " + str(row.get("label") or "the command")
|
|
96
|
+
+ (" in " + row["folder"] if row.get("folder") else "") + ": "
|
|
97
|
+
+ str(row.get("status") or "ran") + " (" + ", ".join(detail) + ")")
|
|
98
|
+
return "\n".join(lines)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def fix_offer(model: str, next_round: int, auto_apply: bool, limit: int = MAX_FIX_ROUNDS,
|
|
102
|
+
history: list | None = None) -> str:
|
|
103
|
+
block = timeline(history or [])
|
|
104
|
+
return (f"Fix round {next_round} of {limit}: send this failure output back to {model} and ask "
|
|
105
|
+
"for the smallest change that makes the command pass. It proposes a diff. "
|
|
106
|
+
+ _approval_clause(auto_apply)
|
|
107
|
+
+ ("\n\n" + block if block else ""))
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def step_offer(model: str, done_id: int, nxt: dict, total: int, auto_apply: bool) -> str:
|
|
111
|
+
return (f"Step {done_id} is verified. The next one is step {nxt['id']} of {total}: "
|
|
112
|
+
f"“{nxt['title']}”. Starting it sends that task to {model} and proposes a diff. "
|
|
113
|
+
+ _approval_clause(auto_apply))
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def removal_notice(session: dict | None) -> str:
|
|
117
|
+
"""What a proposal says about itself when it empties a file, or "" when none does.
|
|
118
|
+
|
|
119
|
+
Both windows read this one copy: the Tk confirm text and the webapp's warning box quote it,
|
|
120
|
+
and the refusal below is built from it.
|
|
121
|
+
"""
|
|
122
|
+
if not session:
|
|
123
|
+
return ""
|
|
124
|
+
notes = [note for note in (shrink_warning(change) for change in session.get("changes", []))
|
|
125
|
+
if note]
|
|
126
|
+
if not notes:
|
|
127
|
+
return ""
|
|
128
|
+
return ("Removes most of an existing file:\n"
|
|
129
|
+
+ "\n".join("• " + note[:180] for note in notes[:3]) + "\n\n")
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def approval_advisories(session: dict | None) -> str:
|
|
133
|
+
"""Everything the operator should read before approving, in one string, from one owner.
|
|
134
|
+
|
|
135
|
+
Both windows call this instead of the two halves: the removal warning lived alone in the dialog until
|
|
136
|
+
the unexpected-file flag was added, and a sentence that reaches one window's confirm box is the drift
|
|
137
|
+
this project keeps recording. The order is the reading order — what a change destroys, then which
|
|
138
|
+
changes nobody asked for.
|
|
139
|
+
"""
|
|
140
|
+
return removal_notice(session) + unexpected_notice(session)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def must_ask(session: dict | None, prior: dict | None = None) -> str:
|
|
144
|
+
"""Why this proposal needs a click even in a folder whose Auto-Apply switch is on, or "".
|
|
145
|
+
|
|
146
|
+
Two situations: a proposal that empties a file the developer wrote, which is the mistake a
|
|
147
|
+
small model really does make; and a previous task that stopped mid-write, which means the
|
|
148
|
+
folder already matches nothing on this card, so no review the user saw covers this write.
|
|
149
|
+
A third: any deletion. The folder's switch takes away a click, and a file has no diff to
|
|
150
|
+
review after the click that removed it.
|
|
151
|
+
"""
|
|
152
|
+
deletes = [change.get("path", "") for change in (session or {}).get("changes", [])
|
|
153
|
+
if isinstance(change, dict) and change.get("delete")]
|
|
154
|
+
if deletes:
|
|
155
|
+
rest = f" and {len(deletes) - 3} more" if len(deletes) > 3 else ""
|
|
156
|
+
return ("Removes a file: " + ", ".join(deletes[:3]) + rest +
|
|
157
|
+
". Auto-Apply takes a click out of this, not a file out of the folder.")
|
|
158
|
+
notice = removal_notice(session)
|
|
159
|
+
if notice.strip():
|
|
160
|
+
return notice.strip()
|
|
161
|
+
if prior and prior.get("state") in INTERRUPTED_STATES:
|
|
162
|
+
return ("The last task in this folder stopped part way through writing, so what is on "
|
|
163
|
+
"disk is not what you reviewed.")
|
|
164
|
+
return ""
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
STATE_FOR_RUN = {"passed": "CHECKS_PASSED", "failed": "VERIFICATION_FAILED",
|
|
168
|
+
"timeout": "VERIFICATION_BLOCKED", "unverified": "VERIFICATION_BLOCKED",
|
|
169
|
+
"unavailable": "VERIFICATION_BLOCKED"}
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def record_run(session_path: Path, result: dict) -> dict:
|
|
173
|
+
"""Persist a slimmed run summary on the session; the full log stays out of the store."""
|
|
174
|
+
session = load_session(session_path)
|
|
175
|
+
if session["state"] not in {"APPLIED_UNVERIFIED", "CHECKS_PASSED", "VERIFICATION_FAILED",
|
|
176
|
+
"VERIFICATION_BLOCKED"}:
|
|
177
|
+
raise PolicyError("Apply the reviewed proposal before running commands.")
|
|
178
|
+
slim = {key: value for key, value in result.items() if key not in {"output", "tail", "failures"}}
|
|
179
|
+
slim["failures"] = [redact(str(row)) for row in result.get("failures", [])][:FAILURES_KEPT]
|
|
180
|
+
slim["tail"] = redact((result.get("tail") or "")[-TAIL_KEPT:])
|
|
181
|
+
session.setdefault("runs", []).append(slim)
|
|
182
|
+
session["state"] = STATE_FOR_RUN[result["status"]]
|
|
183
|
+
event(session, "run", **{key: slim[key] for key in ("recipe", "status", "exit_code", "seconds")})
|
|
184
|
+
atomic_json(session_path, session)
|
|
185
|
+
return session
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def round_history(runs_dir: Path, root: str, chat_id: str) -> list:
|
|
189
|
+
"""Every session in one chat and folder, oldest first — the rounds are sessions, not a counter.
|
|
190
|
+
|
|
191
|
+
Read fresh by both windows, because the point is to compare the run that just finished with the
|
|
192
|
+
ones before it, and a cached list is exactly as stale as the judgement it feeds. A folder whose
|
|
193
|
+
records cannot be read is a folder with no history, which is the same answer as an empty one.
|
|
194
|
+
"""
|
|
195
|
+
try:
|
|
196
|
+
return [item for _, item in chat_sessions(runs_dir, root, chat_id)]
|
|
197
|
+
except (AgentError, OSError):
|
|
198
|
+
return []
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def attempts_of(runs_dir: Path, root: str, chat_id: str, limit: int = MAX_FIX_ROUNDS) -> list:
|
|
202
|
+
"""The rounds the timeline and the offer draw from: this chat's attempts, newest window kept."""
|
|
203
|
+
return attempts(round_history(runs_dir, root, chat_id))[-limit:]
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def last_run(session: dict) -> dict | None:
|
|
207
|
+
runs = session.get("runs") or []
|
|
208
|
+
return runs[-1] if runs else None
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def fix_needed(session: dict) -> bool:
|
|
212
|
+
return bool(last_run(session)) and last_run(session)["status"] in {"failed", "timeout"}
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def fix_task(run: dict) -> str:
|
|
216
|
+
label = str(run.get("label") or run.get("recipe") or "the command")
|
|
217
|
+
return (f"The {label} command failed{where_it_ran(run)}. Read the affected files, then propose "
|
|
218
|
+
"the smallest change that makes it pass. Keep existing behavior and public APIs; do not "
|
|
219
|
+
"delete tests or weaken assertions to pass; do not add dependencies.")
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def where_it_ran(run: dict) -> str:
|
|
223
|
+
"""` in backend/auth-service`, or "" when the command ran at the root of the opened folder.
|
|
224
|
+
|
|
225
|
+
A multi-module project has to be told which module failed, or the model reads the wrong tree and
|
|
226
|
+
the round proposes a change to a file that was never involved. Relative, because the model's
|
|
227
|
+
workspace is already the root and an absolute path is noise that can carry a username.
|
|
228
|
+
"""
|
|
229
|
+
folder = runner.run_folder(run)
|
|
230
|
+
return "" if not folder else " in " + folder
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
# ---------------------------------------------------------------- the shape of a failure
|
|
234
|
+
#
|
|
235
|
+
# `evidence()` used to send the same 12 failure lines plus the last 2 500 characters whatever the
|
|
236
|
+
# command had said. For a Maven run that cannot resolve its parent POM those 2 500 characters are a
|
|
237
|
+
# stack trace and a help URL — and the model spends its context on a build system's complaint instead
|
|
238
|
+
# of on the dependency that is missing. Naming the kind of failure is also what lets the loop stop
|
|
239
|
+
# honestly: "3 rounds, still failing" and "3 rounds, still 7 failures of the same kind" are different
|
|
240
|
+
# sentences, and only the second one tells the user the loop is not going anywhere.
|
|
241
|
+
CATEGORIES = (
|
|
242
|
+
("environment", (r"command not found", r"is not recognized", r"no such file or directory",
|
|
243
|
+
r"cannot find program", r"JAVA_HOME", r"not installed", r"Permission denied",
|
|
244
|
+
r"connection refused", r"could not resolve host", r"no route to host")),
|
|
245
|
+
("timeout", (r"timed out", r"timeout", r"cancelled after")),
|
|
246
|
+
("dependency", (r"non-resolvable parent pom", r"could not resolve", r"failed to collect dependencies",
|
|
247
|
+
r"artifact.*(not found|could not be resolved)", r"unresolved import",
|
|
248
|
+
r"modulenotfounderror", r"no module named", r"package .* is not installed",
|
|
249
|
+
r"peer dependency", r"lock file", r"cannot find module", r"fetch.*registry")),
|
|
250
|
+
# Anchored where a compiler's own wording allows, because "AssertionError: expected 5 but was 4"
|
|
251
|
+
# contains both "error: expected" and an assertion, and it is the second one.
|
|
252
|
+
("syntax", (r"syntaxerror", r"syntax error", r"unterminated", r"invalid syntax",
|
|
253
|
+
r"parse error", r"unexpected token", r"^error: expected", r"\bmissing semicolon\b",
|
|
254
|
+
r"class, interface, enum, or record expected", r"illegal start of",
|
|
255
|
+
r"expected .{0,40}(at|on) line \d")),
|
|
256
|
+
("type", (r"typeerror", r"incompatible types", r"cannot be applied to", r"unresolved reference",
|
|
257
|
+
r"has no attribute", r"is not assignable", r"no overload matches", r"symbol not found",
|
|
258
|
+
r"cannot find symbol")),
|
|
259
|
+
("assertion", (r"assertionerror", r"assert\s*\(?", r"expected.*but was", r"want.*got",
|
|
260
|
+
r"comparison failure", r"expected:.*<", r"test failed", r"failed tests",
|
|
261
|
+
r"^\s*failed\b", r"^\s*\d+ failed", r"failures=\d", r"there were failing tests")),
|
|
262
|
+
)
|
|
263
|
+
|
|
264
|
+
# A build tool's own summary line, kept whatever the category is: it is the one part of the output
|
|
265
|
+
# that states how many tests ran, and "the run is not a proof" is decided from it elsewhere.
|
|
266
|
+
SUMMARY_PATTERNS = (r"^Tests run:", r"^test result:", r"^# tests", r"^ok\s", r"^FAIL", r"BUILD FAILURE",
|
|
267
|
+
r"BUILD SUCCESS", r"^ERROR: ", r"^\s*\d+ passed")
|
|
268
|
+
|
|
269
|
+
# Stack frames and separators: printed right below the line that explains the failure, and saying
|
|
270
|
+
# only which jar the tool was standing in when it noticed. `[ERROR] at org.apache.maven…` carries the
|
|
271
|
+
# error marker, so a keyword filter alone cannot tell a frame from a diagnosis.
|
|
272
|
+
FRAME_PATTERNS = (r"\bat\s+[\w.$]+\(", r'^file\s+"[^"]+",\s+line\s+\d+', r"^traceback \(most recent",
|
|
273
|
+
r"^\s*\.{3,}\s*$", r"^\s*~+\^~*\s*$", r"^\(\S+ exception:\s", r"^\s*at \S+\.\S+")
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
def is_frame(line: str) -> bool:
|
|
277
|
+
lowered = line.strip().lower()
|
|
278
|
+
return any(re.search(pattern, lowered) for pattern in FRAME_PATTERNS)
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
def relevant(run: dict, category: str, limit: int = 4000) -> str:
|
|
282
|
+
"""The lines of the tail that speak about this kind of failure, in the order they were printed.
|
|
283
|
+
|
|
284
|
+
The full output stays on the session and in Activity — this is the slice the model is sent, chosen
|
|
285
|
+
so a small context window is spent on the failure rather than on a stack trace below it.
|
|
286
|
+
"""
|
|
287
|
+
tail = str(run.get("tail") or "")
|
|
288
|
+
patterns = dict(CATEGORIES).get(category) or ()
|
|
289
|
+
rows, seen = [], set()
|
|
290
|
+
for line in tail.splitlines():
|
|
291
|
+
flat = line.strip()
|
|
292
|
+
if not flat or flat in seen or is_frame(flat):
|
|
293
|
+
continue
|
|
294
|
+
lowered = flat.lower()
|
|
295
|
+
if (any(re.search(pattern, lowered) for pattern in patterns)
|
|
296
|
+
or any(re.search(pattern, flat) for pattern in SUMMARY_PATTERNS)
|
|
297
|
+
or any(marker in lowered for marker in ("error", "fail", "cannot", "could not", "missing"))):
|
|
298
|
+
seen.add(flat)
|
|
299
|
+
rows.append(flat[:300])
|
|
300
|
+
if not rows:
|
|
301
|
+
rows = [line.strip()[:300] for line in tail.splitlines()
|
|
302
|
+
if line.strip() and not is_frame(line)][-12:]
|
|
303
|
+
return "\n".join(rows[-24:])[:limit]
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
def classify(run: dict) -> str:
|
|
307
|
+
"""What kind of failure this is, from the text the tool itself printed.
|
|
308
|
+
|
|
309
|
+
Deliberately first-match in a fixed order: an environment failure that also contains the word
|
|
310
|
+
"assert" is still the machine not having the tool, and the fix is not a code change. `unknown` is
|
|
311
|
+
an honest answer and the loop treats it as one — it does not pretend to a diagnosis it lacks.
|
|
312
|
+
|
|
313
|
+
A run that did not fail has no kind: labelling a passing build "unknown" would read as "the tool
|
|
314
|
+
could not tell what went wrong", and nothing went wrong.
|
|
315
|
+
"""
|
|
316
|
+
if run.get("status") == "passed":
|
|
317
|
+
return ""
|
|
318
|
+
text = "\n".join([str(run.get("tail") or ""), "\n".join(str(row) for row in (run.get("failures") or [])),
|
|
319
|
+
str(run.get("reason") or "")]).lower()
|
|
320
|
+
if run.get("status") == "timeout" or run.get("timed_out"):
|
|
321
|
+
return "timeout"
|
|
322
|
+
if run.get("status") == "unavailable":
|
|
323
|
+
return "environment"
|
|
324
|
+
for name, patterns in CATEGORIES:
|
|
325
|
+
if any(re.search(pattern, text, re.MULTILINE) for pattern in patterns):
|
|
326
|
+
return name
|
|
327
|
+
return "unknown"
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def evidence(run: dict) -> str:
|
|
331
|
+
"""Untrusted build/test output for the next turn, capped for small context budgets."""
|
|
332
|
+
category = classify(run)
|
|
333
|
+
folder = runner.run_folder(run)
|
|
334
|
+
lines = ["Command: " + str(run.get("command", ""))]
|
|
335
|
+
if folder:
|
|
336
|
+
lines.append("Folder: " + folder + " of the opened project")
|
|
337
|
+
lines += ["Exit code: " + str(run.get("exit_code")),
|
|
338
|
+
"Status: " + str(run.get("status", ""))]
|
|
339
|
+
if category:
|
|
340
|
+
lines.append("What this looks like: " + category)
|
|
341
|
+
if run.get("failures"):
|
|
342
|
+
lines.append("Reported problems:")
|
|
343
|
+
lines.extend(" " + str(row)[:300] for row in run["failures"][:FAILURES_KEPT])
|
|
344
|
+
picked = relevant(run, category)
|
|
345
|
+
if picked:
|
|
346
|
+
lines.append("The lines that say so:")
|
|
347
|
+
lines.append(picked)
|
|
348
|
+
return redact("\n".join(lines))[:12000]
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
# ---------------------------------------------------------------- the rounds already spent
|
|
352
|
+
|
|
353
|
+
def change_digest(changes: list) -> str:
|
|
354
|
+
"""A fingerprint of *what would be written*, independent of the task text around it.
|
|
355
|
+
|
|
356
|
+
`proposal_hash` covers the summary and the checks, which a re-worded second attempt changes; the
|
|
357
|
+
question the loop needs to answer is whether the files are the same, and that is this.
|
|
358
|
+
"""
|
|
359
|
+
digest = hashlib.sha256()
|
|
360
|
+
for change in sorted(changes or [], key=lambda item: str(item.get("path", ""))):
|
|
361
|
+
if not isinstance(change, dict):
|
|
362
|
+
continue
|
|
363
|
+
body = change.get("after")
|
|
364
|
+
digest.update(str(change.get("path", "")).encode("utf-8", "replace"))
|
|
365
|
+
digest.update(b"\0")
|
|
366
|
+
digest.update(b"DELETE" if change.get("delete") or body is None
|
|
367
|
+
else str(body).encode("utf-8", "replace"))
|
|
368
|
+
digest.update(b"\0")
|
|
369
|
+
return digest.hexdigest()
|
|
370
|
+
|
|
371
|
+
|
|
372
|
+
def attempts(sessions: list) -> list:
|
|
373
|
+
"""One row per attempt in a chat, oldest first: what ran, what it looked like, what it changed.
|
|
374
|
+
|
|
375
|
+
Rounds are separate sessions, not one session with a counter, so the history is the chat. This is
|
|
376
|
+
also what a report and the timeline in the window are both drawn from — one reading of the record,
|
|
377
|
+
three places that show it.
|
|
378
|
+
"""
|
|
379
|
+
rows = []
|
|
380
|
+
for item in sessions:
|
|
381
|
+
for run in (item.get("runs") or []):
|
|
382
|
+
folder = runner.run_folder(run)
|
|
383
|
+
rows.append({"session": item.get("id", ""), "at": run.get("at") or item.get("created", ""),
|
|
384
|
+
"state": item.get("state", ""), "label": run.get("label", ""),
|
|
385
|
+
"folder": Path(folder).name if folder else "",
|
|
386
|
+
"status": run.get("status", ""), "exit_code": run.get("exit_code"),
|
|
387
|
+
"seconds": run.get("seconds"),
|
|
388
|
+
"tests": (run.get("proof") or {}).get("tests") or 0,
|
|
389
|
+
"failures": (run.get("proof") or {}).get("failures") or 0,
|
|
390
|
+
"category": classify(run),
|
|
391
|
+
"files": len(item.get("changes") or [])})
|
|
392
|
+
return rows
|
|
393
|
+
|
|
394
|
+
|
|
395
|
+
def already_tried(sessions: list, changes: list) -> bool:
|
|
396
|
+
"""Is this proposal the same set of files with the same contents as one already attempted?
|
|
397
|
+
|
|
398
|
+
A model that is asked twice for the same fix will often answer with exactly the same diff. Burning
|
|
399
|
+
a round on it is not iteration, and the user watching the rounds counts it as progress.
|
|
400
|
+
"""
|
|
401
|
+
seen = set()
|
|
402
|
+
for item in sessions:
|
|
403
|
+
proposal = item.get("changes") or []
|
|
404
|
+
if proposal:
|
|
405
|
+
seen.add(change_digest(proposal))
|
|
406
|
+
return bool(changes) and change_digest(changes) in seen
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
def stalled(sessions: list) -> str:
|
|
410
|
+
"""Did the last attempt make the run any better? "" when it did, or when there is nothing to say.
|
|
411
|
+
|
|
412
|
+
Comparison is only honest between two runs of the same command — 7 failures in Maven and 4 in
|
|
413
|
+
pytest are not a trend. A run that never produced a test count has no numbers to compare, which
|
|
414
|
+
is exactly what a missing tool or an unresolvable artifact looks like, so its second reading is
|
|
415
|
+
by kind. Otherwise the loop keeps its budget and says nothing, because a false "no progress"
|
|
416
|
+
stop is worse than one more round.
|
|
417
|
+
"""
|
|
418
|
+
runs = [row for row in attempts(sessions) if row.get("seconds") is not None]
|
|
419
|
+
if len(runs) < 2:
|
|
420
|
+
return ""
|
|
421
|
+
previous, latest = runs[-2], runs[-1]
|
|
422
|
+
if (previous["label"], previous["folder"]) != (latest["label"], latest["folder"]):
|
|
423
|
+
# Two different commands are not a trend, and neither are the same command run in two modules
|
|
424
|
+
# of one reactor: auth-service's three failures say nothing about product-service's three.
|
|
425
|
+
return ""
|
|
426
|
+
if latest["status"] == "passed":
|
|
427
|
+
return ""
|
|
428
|
+
if (previous["category"] == latest["category"]
|
|
429
|
+
and latest["category"] in {"environment", "dependency"}):
|
|
430
|
+
return ("the same " + latest["category"] + " failure in the last two rounds: the last "
|
|
431
|
+
"change did not move it")
|
|
432
|
+
if previous["tests"] or latest["tests"]:
|
|
433
|
+
now, before = (latest["failures"], -latest["tests"]), (previous["failures"], -previous["tests"])
|
|
434
|
+
if now >= before:
|
|
435
|
+
word = "the same as" if now == before else "worse than"
|
|
436
|
+
return ("no progress: " + str(latest["failures"]) + " failing of "
|
|
437
|
+
+ str(latest["tests"]) + " tests, " + word + " the round before")
|
|
438
|
+
return ""
|
|
439
|
+
|
|
440
|
+
|
|
441
|
+
# What makes two recorded failures the same error. A stack line number, a test count, an absolute path
|
|
442
|
+
# and a `[ERROR]` prefix all move when nothing about the problem moved, so a fingerprint that kept them
|
|
443
|
+
# would call the same broken build a new one every round — and the whole point is to notice the repeat.
|
|
444
|
+
ERROR_PREFIX = re.compile(r"^(?:\s*(?:\[?\s*(?:ERROR|WARNING|INFO|SEVERE)\]?|ERROR:\s*|FAILED\b|FAIL\b)"
|
|
445
|
+
r"\s*[:\[\]]*\s*)+", re.I)
|
|
446
|
+
# The position a compiler hangs on a filename (`Foo.java:12`, `Foo.java:[45,12]`) is stripped with the
|
|
447
|
+
# file. What comes *after* a filename is kept: `::test_total` and `.method()` are the part of a pytest or
|
|
448
|
+
# Maven line that says which test broke, and a token that gobbled them would fingerprint a whole suite as
|
|
449
|
+
# one anonymous failure and hide which case is still open.
|
|
450
|
+
FILE_TOKEN = re.compile(r"(?:[A-Za-z]:[\\/])?[^\s|]*\.(?:java|kt|kts|py|go|rs|js|jsx|ts|tsx|xml|gradle|"
|
|
451
|
+
r"json|ya?ml|properties|mod|toml|c|cpp|h)\b(?:[:/]\[?\d+(?:,\s*\d+)?\]?)?",
|
|
452
|
+
re.I)
|
|
453
|
+
COORDINATES = re.compile(r"\b(?:at|line)\s+~?\d+\b|\b:\d+(?::\d+)?\b|\bL\d+:\d+\b", re.I)
|
|
454
|
+
COUNTERS = re.compile(r"\b\d+(?:\.\d+)*\b")
|
|
455
|
+
ERROR_LINES_KEPT = 6
|
|
456
|
+
ERROR_LINE_CHARS = 160
|
|
457
|
+
|
|
458
|
+
|
|
459
|
+
def error_identity(run: dict) -> str:
|
|
460
|
+
"""The failure's own words, with its coordinates taken out and its secrets taken out with them.
|
|
461
|
+
|
|
462
|
+
This text is written into a model prompt and into the thread, so redaction is not optional here:
|
|
463
|
+
a Maven or pytest failure line routinely quotes the connection string that broke, and the repository
|
|
464
|
+
redactor cannot help a value that never looked like a key.
|
|
465
|
+
"""
|
|
466
|
+
if not isinstance(run, dict) or run.get("status") == "passed":
|
|
467
|
+
return ""
|
|
468
|
+
lines = [str(row) for row in (run.get("failures") or []) if str(row).strip()]
|
|
469
|
+
if not lines:
|
|
470
|
+
# No parsed failures means the command failed before it reported any — a missing tool, an
|
|
471
|
+
# unresolvable artifact, a syntax error Maven never reached. The tail's last lines are where
|
|
472
|
+
# those say what happened.
|
|
473
|
+
lines = [line for line in str(run.get("tail", "")).splitlines() if line.strip()]
|
|
474
|
+
cleaned = []
|
|
475
|
+
for line in lines[-ERROR_LINES_KEPT:] if not run.get("failures") else lines[:ERROR_LINES_KEPT]:
|
|
476
|
+
text = ERROR_PREFIX.sub("", str(line))
|
|
477
|
+
text = FILE_TOKEN.sub("<file>", text)
|
|
478
|
+
text = COORDINATES.sub("", text)
|
|
479
|
+
text = COUNTERS.sub("N", " ".join(text.split()).casefold())
|
|
480
|
+
text = redact(text)[:ERROR_LINE_CHARS]
|
|
481
|
+
if text.strip(" .,:;-"):
|
|
482
|
+
cleaned.append(text)
|
|
483
|
+
return "\n".join(cleaned)
|
|
484
|
+
|
|
485
|
+
|
|
486
|
+
def error_fingerprint(run: dict) -> str:
|
|
487
|
+
"""A short, stable id for one error — the identity above, hashed so it is cheap to carry."""
|
|
488
|
+
identity = error_identity(run)
|
|
489
|
+
if not identity:
|
|
490
|
+
return ""
|
|
491
|
+
return hashlib.sha256(identity.encode("utf-8", "replace")).hexdigest()[:12]
|
|
492
|
+
|
|
493
|
+
|
|
494
|
+
def unresolved(sessions: list, exclude_chat: str = "") -> list[dict]:
|
|
495
|
+
"""Build errors this project has hit and never saw pass, whoever hit them.
|
|
496
|
+
|
|
497
|
+
`stalled()` answers "did this round help"; this answers the question a person asks when they start a
|
|
498
|
+
second chat for a build the first one could not fix — has this exact failure already been fought? An
|
|
499
|
+
error is settled when a run of the *same command in the same folder* passed after its last
|
|
500
|
+
occurrence; anything else is still open, and a task that walks into it is told before it spends a
|
|
501
|
+
round discovering it again.
|
|
502
|
+
"""
|
|
503
|
+
failures, passes = [], []
|
|
504
|
+
for item in sessions or []:
|
|
505
|
+
if not isinstance(item, dict):
|
|
506
|
+
continue
|
|
507
|
+
# A session with no chat identity is its own, the same way `chat_sessions` treats one, so the
|
|
508
|
+
# task being planned now cannot mistake a bare run for somebody else's conversation.
|
|
509
|
+
owner = str(item.get("chat_id", "") or item.get("id", ""))
|
|
510
|
+
if exclude_chat and owner == exclude_chat:
|
|
511
|
+
continue
|
|
512
|
+
created = str(item.get("created", ""))
|
|
513
|
+
for run in (item.get("runs") or []):
|
|
514
|
+
stamp = str(run.get("at") or created)
|
|
515
|
+
where = {"at": stamp, "label": str(run.get("label", "")),
|
|
516
|
+
"folder": runner.run_folder(run), "status": str(run.get("status", ""))}
|
|
517
|
+
if where["status"] == "passed":
|
|
518
|
+
passes.append(where)
|
|
519
|
+
continue
|
|
520
|
+
mark = error_fingerprint(run)
|
|
521
|
+
if mark:
|
|
522
|
+
failures.append({**where, "fingerprint": mark, "session": str(item.get("id", "")),
|
|
523
|
+
"sample": (error_identity(run).splitlines() or [""])[0]})
|
|
524
|
+
|
|
525
|
+
out = []
|
|
526
|
+
for mark in dict.fromkeys(row["fingerprint"] for row in failures):
|
|
527
|
+
seen = sorted((row for row in failures if row["fingerprint"] == mark), key=lambda row: row["at"])
|
|
528
|
+
last = seen[-1]
|
|
529
|
+
settled = any(row["status"] == "passed" and row["at"] > last["at"]
|
|
530
|
+
and (row["label"], row["folder"]) == (last["label"], last["folder"])
|
|
531
|
+
for row in passes)
|
|
532
|
+
if settled:
|
|
533
|
+
continue
|
|
534
|
+
out.append({"fingerprint": mark, "count": len(seen), "since": seen[0]["at"],
|
|
535
|
+
"last": last["at"], "label": last["label"], "folder": last["folder"],
|
|
536
|
+
"sample": last["sample"], "tasks": len({row["session"] for row in seen})})
|
|
537
|
+
return sorted(out, key=lambda row: (-row["count"], row["last"]))
|
|
538
|
+
|
|
539
|
+
|
|
540
|
+
def should_stop(sessions: list, round_number: int, limit: int = MAX_FIX_ROUNDS) -> tuple[bool, str]:
|
|
541
|
+
"""The one question both windows ask before spending a model turn: is there a reason not to?"""
|
|
542
|
+
rows = attempts(sessions)
|
|
543
|
+
if round_number >= limit:
|
|
544
|
+
# The extra clause is only said when the record shows a command that still fails: `ask_for_fix`
|
|
545
|
+
# can be reached with a budget spent and a run that passed, and then the honest sentence about
|
|
546
|
+
# why the loop is over is just the budget.
|
|
547
|
+
still = (" and the command still fails"
|
|
548
|
+
if rows and rows[-1]["status"] in {"failed", "timeout"} else "")
|
|
549
|
+
return True, "stopped after " + str(limit) + " fix rounds" + still
|
|
550
|
+
reason = stalled(sessions)
|
|
551
|
+
if reason:
|
|
552
|
+
return True, reason
|
|
553
|
+
return False, ""
|
|
554
|
+
|
|
555
|
+
|
|
556
|
+
def passes(session: dict) -> bool:
|
|
557
|
+
return bool(last_run(session)) and last_run(session)["status"] == "passed"
|
|
558
|
+
|
|
559
|
+
|
|
560
|
+
def choose_recipe(repo: Path, preferred: str | None = None) -> tuple[str | None, list[str]]:
|
|
561
|
+
options = runner.detect(repo)
|
|
562
|
+
if preferred and preferred in options:
|
|
563
|
+
return preferred, options
|
|
564
|
+
return (options[0] if options else None), options
|