py-harness-cli 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- finetune/__init__.py +1 -0
- finetune/agent_system.py +41 -0
- finetune/agent_traces.py +157 -0
- finetune/everyday.py +30 -0
- finetune/hf_ollama.py +158 -0
- finetune/huggingface_store.py +144 -0
- finetune/models.py +74 -0
- finetune/paths.py +11 -0
- finetune/python_vibe.py +788 -0
- finetune/splits.py +54 -0
- finetune/systems.py +9 -0
- harness/__init__.py +42 -0
- harness/__main__.py +8 -0
- harness/act/__init__.py +6 -0
- harness/act/autofix/__init__.py +110 -0
- harness/act/autofix/additions.py +217 -0
- harness/act/autofix/conflicts.py +124 -0
- harness/act/autofix/cover.py +419 -0
- harness/act/autofix/mechanical.py +151 -0
- harness/act/autofix/missing_imports.py +50 -0
- harness/act/autofix/moves.py +439 -0
- harness/act/autofix/names.py +339 -0
- harness/act/autofix/scaffold.py +224 -0
- harness/act/code.py +157 -0
- harness/act/gate.py +229 -0
- harness/act/parse.py +247 -0
- harness/act/patch_fix.py +138 -0
- harness/act/tools.py +244 -0
- harness/agent/__init__.py +11 -0
- harness/agent/dispatch.py +235 -0
- harness/agent/loop.py +699 -0
- harness/agent/options.py +144 -0
- harness/agent/policy.py +856 -0
- harness/agent/prompt.py +170 -0
- harness/cli.py +393 -0
- harness/editor_kit.py +265 -0
- harness/guard/__init__.py +6 -0
- harness/guard/fallbacks.py +6 -0
- harness/guard/loop_guard.py +57 -0
- harness/guard/python_vibe.py +68 -0
- harness/guard/run.py +41 -0
- harness/guard/types.py +19 -0
- harness/locate.py +767 -0
- harness/mcp_stdio.py +306 -0
- harness/memory/__init__.py +5 -0
- harness/memory/conversation.py +104 -0
- harness/model/__init__.py +6 -0
- harness/model/chat_backend.py +100 -0
- harness/model/engine.py +165 -0
- harness/model/ollama_generate.py +60 -0
- harness/model/openai_generate.py +156 -0
- harness/model/outbound.py +83 -0
- harness/model/route.py +90 -0
- harness/observe/__init__.py +6 -0
- harness/observe/eval_gate.py +80 -0
- harness/observe/eval_loop.py +185 -0
- harness/observe/eval_tasks.py +399 -0
- harness/observe/report_md.py +102 -0
- harness/observe/trace_record.py +79 -0
- harness/openai_api.py +81 -0
- harness/paths.py +88 -0
- harness/py.typed +0 -0
- harness/scan/__init__.py +6 -0
- harness/scan/app_spec.py +338 -0
- harness/scan/design.py +112 -0
- harness/scan/existing.py +131 -0
- harness/scan/layout.py +254 -0
- harness/scan/names.py +308 -0
- harness/scan/project_brief.py +287 -0
- harness/scan/project_docs.py +42 -0
- harness/scan/project_scan.py +49 -0
- harness/scan/repo_map.py +101 -0
- harness/secrets.py +39 -0
- harness/server.py +199 -0
- harness/ship/__init__.py +1 -0
- harness/ship/bot_pr.py +221 -0
- harness/ship/git_ship.py +262 -0
- harness/ship/identity.py +62 -0
- harness/ship/ticket.py +251 -0
- harness/skillkit/__init__.py +6 -0
- harness/skillkit/catalog.py +241 -0
- harness/skillkit/refuse_change.py +640 -0
- harness/skillkit/refuse_finish.py +295 -0
- harness/skillkit/target.py +238 -0
- harness/task.py +717 -0
- py_harness_cli-0.3.0.dist-info/METADATA +177 -0
- py_harness_cli-0.3.0.dist-info/RECORD +92 -0
- py_harness_cli-0.3.0.dist-info/WHEEL +5 -0
- py_harness_cli-0.3.0.dist-info/entry_points.txt +3 -0
- py_harness_cli-0.3.0.dist-info/licenses/LICENSE +202 -0
- py_harness_cli-0.3.0.dist-info/licenses/NOTICE +6 -0
- py_harness_cli-0.3.0.dist-info/top_level.txt +2 -0
harness/agent/loop.py
ADDED
|
@@ -0,0 +1,699 @@
|
|
|
1
|
+
"""Run one task from start to finish.
|
|
2
|
+
|
|
3
|
+
from harness import Agent, AgentOptions
|
|
4
|
+
|
|
5
|
+
result = Agent(AgentOptions(project=Path("~/app"))).run("fix the NameError")
|
|
6
|
+
|
|
7
|
+
This class is responsible for the order of steps and nothing else. It asks
|
|
8
|
+
`harness.agent.prompt` what to send to the model, `harness.agent.policy`
|
|
9
|
+
whether a proposed action is allowed, and `harness.agent.dispatch` to carry
|
|
10
|
+
an allowed action out.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import uuid
|
|
16
|
+
from dataclasses import dataclass, field
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from types import SimpleNamespace
|
|
19
|
+
|
|
20
|
+
from harness.act.autofix import (
|
|
21
|
+
apply_cli_mock_test,
|
|
22
|
+
apply_cover_test,
|
|
23
|
+
apply_person_bind,
|
|
24
|
+
unbound_typo,
|
|
25
|
+
)
|
|
26
|
+
from harness.act.parse import parse_turn_smart
|
|
27
|
+
from harness.act.tools import run_python
|
|
28
|
+
from harness.scan.names import undefined_in_file
|
|
29
|
+
from harness.agent.dispatch import ACTIONS, run_action
|
|
30
|
+
from harness.agent.options import AgentOptions, AgentResult, Step
|
|
31
|
+
from harness.agent.policy import (
|
|
32
|
+
LoopState,
|
|
33
|
+
done_without_proof,
|
|
34
|
+
next_prompt,
|
|
35
|
+
refuse_before,
|
|
36
|
+
refuse_done,
|
|
37
|
+
should_run_suite_after_write,
|
|
38
|
+
)
|
|
39
|
+
from harness.agent.prompt import Preamble, build_preamble
|
|
40
|
+
from harness.locate import named_file_review_summary
|
|
41
|
+
from harness.memory import Conversation
|
|
42
|
+
from harness.model.engine import make_generate
|
|
43
|
+
from harness.model.ollama_generate import CONTEXT_TOKENS
|
|
44
|
+
from harness.observe.trace_record import append_turn, default_trace_path
|
|
45
|
+
from harness.scan.design import render_design_review
|
|
46
|
+
from harness.task import (
|
|
47
|
+
looks_like_add_feature,
|
|
48
|
+
looks_like_app_loop,
|
|
49
|
+
looks_like_bugfix,
|
|
50
|
+
looks_like_design_loop,
|
|
51
|
+
looks_like_question,
|
|
52
|
+
looks_like_ship,
|
|
53
|
+
looks_unclear,
|
|
54
|
+
named_project_file,
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@dataclass(frozen=True)
|
|
59
|
+
class Question:
|
|
60
|
+
"""A question the agent needs answered before it can continue.
|
|
61
|
+
|
|
62
|
+
Fields:
|
|
63
|
+
text: the question.
|
|
64
|
+
options: answers the agent considers likely. May be empty.
|
|
65
|
+
"""
|
|
66
|
+
|
|
67
|
+
text: str
|
|
68
|
+
options: tuple[str, ...] = ()
|
|
69
|
+
|
|
70
|
+
def render(self) -> str:
|
|
71
|
+
if not self.options:
|
|
72
|
+
return self.text
|
|
73
|
+
listed = "\n".join(f" {i}. {opt}" for i, opt in enumerate(self.options, 1))
|
|
74
|
+
return f"{self.text}\n{listed}"
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def trace_path(options: AgentOptions) -> Path | None:
|
|
78
|
+
"""Where this run writes its turns, or None when it writes none.
|
|
79
|
+
|
|
80
|
+
Recording is on unless asked not to. A run that records nothing
|
|
81
|
+
leaves no way to measure it afterwards, and there is no getting the
|
|
82
|
+
trace back: this project reached sixty-five rows of training data
|
|
83
|
+
while doing a week of real work, because the flag was opt-in.
|
|
84
|
+
"""
|
|
85
|
+
if options.keep_no_record:
|
|
86
|
+
return None
|
|
87
|
+
if options.record is not None:
|
|
88
|
+
return options.record.expanduser()
|
|
89
|
+
return default_trace_path(options.resolved_project())
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def new_trace_id() -> str:
|
|
93
|
+
"""A short name for one run, so its turns can be found together.
|
|
94
|
+
|
|
95
|
+
A recorded turn carried no sign of which run it came from, so a turn
|
|
96
|
+
from a run that finished the job looked exactly like a turn from one
|
|
97
|
+
that spent twenty steps and wrote nothing. About a third of runs
|
|
98
|
+
fail, and training on both without being able to tell them apart
|
|
99
|
+
teaches the failure alongside the work.
|
|
100
|
+
"""
|
|
101
|
+
return uuid.uuid4().hex[:12]
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _trace_result(
|
|
105
|
+
options: AgentOptions, result: AgentResult, trace_id: str
|
|
106
|
+
) -> None:
|
|
107
|
+
"""Keep a row saying how the run ended, and under which id.
|
|
108
|
+
|
|
109
|
+
`trace_id` has no default on purpose. It had one, and a caller that
|
|
110
|
+
forgot it wrote a closing row signed with an empty string — which
|
|
111
|
+
reads as a run whose turns cannot be found, so its outcome could not
|
|
112
|
+
be used to filter anything. Three of the first thirty-five turns
|
|
113
|
+
collected after the change were exactly that.
|
|
114
|
+
"""
|
|
115
|
+
dest = trace_path(options)
|
|
116
|
+
if dest is None:
|
|
117
|
+
return
|
|
118
|
+
append_turn(
|
|
119
|
+
dest,
|
|
120
|
+
{
|
|
121
|
+
"run": trace_id,
|
|
122
|
+
"user": options.task,
|
|
123
|
+
"assistant": result.summary,
|
|
124
|
+
"action": result.stopped,
|
|
125
|
+
"stopped": result.stopped,
|
|
126
|
+
"ok": result.ok,
|
|
127
|
+
},
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _question_from(turn) -> Question:
|
|
132
|
+
text = (turn.query or turn.summary or "").strip() or "What should I do?"
|
|
133
|
+
raw = turn.append or turn.replace or ""
|
|
134
|
+
options = tuple(
|
|
135
|
+
line.strip(" -*\t")
|
|
136
|
+
for line in raw.splitlines()
|
|
137
|
+
if line.strip(" -*\t")
|
|
138
|
+
)
|
|
139
|
+
return Question(text, options[:4])
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
@dataclass
|
|
143
|
+
class RunState:
|
|
144
|
+
"""What one run carries while it decides whether to call the model.
|
|
145
|
+
|
|
146
|
+
`run()` used to hold all of this in local variables across two
|
|
147
|
+
hundred lines, which is why the steps could not be read or tested
|
|
148
|
+
apart from each other.
|
|
149
|
+
|
|
150
|
+
Fields:
|
|
151
|
+
options: the request, replaced when the user answers a question.
|
|
152
|
+
preamble: what the harness found before any model turn.
|
|
153
|
+
trace_id: a short name for this run, stamped on every turn it
|
|
154
|
+
records, so the turns of a run that worked can be told from
|
|
155
|
+
the turns of one that did not.
|
|
156
|
+
writes: project-relative paths this run has changed.
|
|
157
|
+
test_note: what the suite said after a mechanical fix, when that
|
|
158
|
+
fix was not the end of the job. It is put to the model so it
|
|
159
|
+
starts from the real failure instead of looking for one.
|
|
160
|
+
"""
|
|
161
|
+
|
|
162
|
+
options: AgentOptions
|
|
163
|
+
preamble: object
|
|
164
|
+
trace_id: str = field(default_factory=new_trace_id)
|
|
165
|
+
writes: list[str] = field(default_factory=list)
|
|
166
|
+
test_note: str = ""
|
|
167
|
+
|
|
168
|
+
def answer_was(self, answer: str, *, rebuild: bool = False) -> None:
|
|
169
|
+
"""Fold the user's answer into the task and say so in the trace."""
|
|
170
|
+
self.options = _with_task(self.options, f"{self.options.task} ({answer})")
|
|
171
|
+
if rebuild:
|
|
172
|
+
self.preamble = build_preamble(self.options)
|
|
173
|
+
self.options.emit("preamble", f"user answered: {answer}")
|
|
174
|
+
|
|
175
|
+
def mechanical_note(self, fallback: str) -> str:
|
|
176
|
+
"""The first line of what the mechanical pass reported."""
|
|
177
|
+
return next(
|
|
178
|
+
(
|
|
179
|
+
line[2:]
|
|
180
|
+
for line in (getattr(self.preamble, "autofix", "") or "").splitlines()
|
|
181
|
+
if line.startswith("- ")
|
|
182
|
+
),
|
|
183
|
+
fallback,
|
|
184
|
+
)
|
|
185
|
+
|
|
186
|
+
def first_prompt(self) -> str:
|
|
187
|
+
"""The prompt the model opens on."""
|
|
188
|
+
prompt = self.preamble.prompt
|
|
189
|
+
if not self.test_note:
|
|
190
|
+
return prompt
|
|
191
|
+
return (
|
|
192
|
+
f"{prompt}\n\nHarness ran tests after the mechanical fix:\n"
|
|
193
|
+
f"{self.test_note}\n"
|
|
194
|
+
"Action: patch the remaining failure, or Action: done if "
|
|
195
|
+
"the task is already met."
|
|
196
|
+
)
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
class Agent:
|
|
200
|
+
"""Runs one task against one project."""
|
|
201
|
+
|
|
202
|
+
def __init__(self, options: AgentOptions) -> None:
|
|
203
|
+
self.options = options
|
|
204
|
+
self.project = options.resolved_project()
|
|
205
|
+
|
|
206
|
+
def preamble(self, task: str | None = None) -> Preamble:
|
|
207
|
+
options = self.options if task is None else _with_task(self.options, task)
|
|
208
|
+
return build_preamble(options)
|
|
209
|
+
|
|
210
|
+
def run(self, task: str | None = None) -> AgentResult:
|
|
211
|
+
"""Answer the task, and say honestly how the run ended.
|
|
212
|
+
|
|
213
|
+
Read as four questions asked in order, before the model is
|
|
214
|
+
loaded at all: is the task clear enough to start from, is
|
|
215
|
+
reading the file the whole job, can the harness make the change
|
|
216
|
+
on its own, and is there a typo only a person can settle. Each
|
|
217
|
+
one either finishes the run or hands on to the next. Whatever
|
|
218
|
+
survives all four is what the model is actually needed for.
|
|
219
|
+
"""
|
|
220
|
+
options = self.options if task is None else _with_task(self.options, task)
|
|
221
|
+
if not options.task.strip():
|
|
222
|
+
raise ValueError("task required")
|
|
223
|
+
run = RunState(options=options, preamble=build_preamble(options))
|
|
224
|
+
run.options.emit("preamble", run.preamble.pre_text or "")
|
|
225
|
+
|
|
226
|
+
for decide in (
|
|
227
|
+
self._settle_an_unclear_task,
|
|
228
|
+
self._read_the_file_if_that_is_the_whole_job,
|
|
229
|
+
self._make_the_change_without_a_model,
|
|
230
|
+
self._settle_a_typo_only_a_person_can,
|
|
231
|
+
):
|
|
232
|
+
finished = decide(run)
|
|
233
|
+
if finished is not None:
|
|
234
|
+
_trace_result(run.options, finished, run.trace_id)
|
|
235
|
+
return finished
|
|
236
|
+
|
|
237
|
+
# Every run ends with a row saying how it ended. Without one,
|
|
238
|
+
# the turns of a run that spent its whole budget look exactly
|
|
239
|
+
# like the turns of a run that did the job.
|
|
240
|
+
result = self._work_with_the_model(run)
|
|
241
|
+
_trace_result(run.options, result, run.trace_id)
|
|
242
|
+
return result
|
|
243
|
+
|
|
244
|
+
# -- the four questions asked before the model is loaded ------------
|
|
245
|
+
|
|
246
|
+
def _settle_an_unclear_task(self, run: RunState) -> AgentResult | None:
|
|
247
|
+
"""A task naming no file and no symbol cannot be started from.
|
|
248
|
+
|
|
249
|
+
The harness asks rather than relying on the model to notice: a
|
|
250
|
+
small model reaches for `patch` long before it reaches for `ask`.
|
|
251
|
+
"""
|
|
252
|
+
question = opening_question(run.options.task, run.preamble)
|
|
253
|
+
if question is None:
|
|
254
|
+
return None
|
|
255
|
+
answer = self._ask(question, run.options)
|
|
256
|
+
if answer is None:
|
|
257
|
+
return AgentResult(ok=False, summary=question.render(), stopped="question")
|
|
258
|
+
run.answer_was(answer, rebuild=True)
|
|
259
|
+
return None
|
|
260
|
+
|
|
261
|
+
def _read_the_file_if_that_is_the_whole_job(
|
|
262
|
+
self, run: RunState
|
|
263
|
+
) -> AgentResult | None:
|
|
264
|
+
"""A review of a named file is reading, not editing."""
|
|
265
|
+
review = named_file_review_summary(self.project, run.options.task)
|
|
266
|
+
if not review:
|
|
267
|
+
return None
|
|
268
|
+
run.options.emit("result", review)
|
|
269
|
+
return AgentResult(ok=True, summary=review, stopped="done")
|
|
270
|
+
|
|
271
|
+
def _make_the_change_without_a_model(self, run: RunState) -> AgentResult | None:
|
|
272
|
+
"""Apply the mechanical repairs, and stop if they were enough.
|
|
273
|
+
|
|
274
|
+
These are the cases that cannot be got wrong: a misspelling with
|
|
275
|
+
exactly one candidate in scope, a missing import for a module
|
|
276
|
+
everyone knows, a test appended where one already exists. They
|
|
277
|
+
take a tenth of a second and give the same answer every time.
|
|
278
|
+
"""
|
|
279
|
+
if not run.preamble.autofix:
|
|
280
|
+
return None
|
|
281
|
+
run.writes.extend(_autofix_paths(run.preamble.autofix))
|
|
282
|
+
if not run.options.allow_writes:
|
|
283
|
+
return AgentResult(
|
|
284
|
+
ok=True,
|
|
285
|
+
summary=f"Read-only: would {run.mechanical_note('mechanical fix')}. "
|
|
286
|
+
"Nothing written.",
|
|
287
|
+
stopped="done",
|
|
288
|
+
writes=(),
|
|
289
|
+
)
|
|
290
|
+
still_undefined = self._names_left_undefined(run)
|
|
291
|
+
verdict, test_output = _verify_mechanical(self.project)
|
|
292
|
+
run.options.emit("result", test_output)
|
|
293
|
+
if still_undefined:
|
|
294
|
+
run.test_note = (
|
|
295
|
+
f"undefined name {still_undefined[0]} after the "
|
|
296
|
+
"mechanical fix. The suite is not enough."
|
|
297
|
+
)
|
|
298
|
+
run.options.emit("result", run.test_note)
|
|
299
|
+
return None
|
|
300
|
+
if verdict not in {"passed", "no suite"}:
|
|
301
|
+
run.test_note = test_output
|
|
302
|
+
return None
|
|
303
|
+
tail = (
|
|
304
|
+
"Tests passed."
|
|
305
|
+
if verdict == "passed"
|
|
306
|
+
else "This project has no tests to check it against."
|
|
307
|
+
)
|
|
308
|
+
note = run.mechanical_note("mechanical fix applied")
|
|
309
|
+
return AgentResult(
|
|
310
|
+
ok=True,
|
|
311
|
+
summary=f"{note}. {tail}",
|
|
312
|
+
stopped="done",
|
|
313
|
+
writes=tuple(run.writes),
|
|
314
|
+
)
|
|
315
|
+
|
|
316
|
+
def _names_left_undefined(self, run: RunState) -> list[str]:
|
|
317
|
+
"""Names still unbound in what the mechanical pass just wrote."""
|
|
318
|
+
if not looks_like_bugfix(run.options.task):
|
|
319
|
+
return []
|
|
320
|
+
found: list[str] = []
|
|
321
|
+
for rel in run.writes:
|
|
322
|
+
found.extend(undefined_in_file(self.project / rel))
|
|
323
|
+
return found
|
|
324
|
+
|
|
325
|
+
def _settle_a_typo_only_a_person_can(self, run: RunState) -> AgentResult | None:
|
|
326
|
+
"""A misspelling with no safe candidate is a question, not a guess."""
|
|
327
|
+
question = leftover_bind_question(run.options.task, self.project)
|
|
328
|
+
if question is None:
|
|
329
|
+
return None
|
|
330
|
+
answer = self._ask(question, run.options)
|
|
331
|
+
if answer is None:
|
|
332
|
+
return AgentResult(
|
|
333
|
+
ok=False,
|
|
334
|
+
summary=question.render(),
|
|
335
|
+
stopped="question",
|
|
336
|
+
writes=tuple(run.writes),
|
|
337
|
+
)
|
|
338
|
+
# Asking is not writing, so a read-only run may still ask. What it
|
|
339
|
+
# may not do is act on the answer: `ask` and `--dry-run` both
|
|
340
|
+
# promise the folder is left alone.
|
|
341
|
+
note = apply_person_bind(
|
|
342
|
+
self.project,
|
|
343
|
+
run.options.task,
|
|
344
|
+
answer,
|
|
345
|
+
write=run.options.allow_writes,
|
|
346
|
+
)
|
|
347
|
+
if not note:
|
|
348
|
+
return AgentResult(
|
|
349
|
+
ok=False,
|
|
350
|
+
summary=(
|
|
351
|
+
f"{question.render()} "
|
|
352
|
+
"That answer is still not something this method can return."
|
|
353
|
+
),
|
|
354
|
+
stopped="question",
|
|
355
|
+
writes=tuple(run.writes),
|
|
356
|
+
)
|
|
357
|
+
if not run.options.allow_writes:
|
|
358
|
+
return AgentResult(
|
|
359
|
+
ok=True,
|
|
360
|
+
summary=f"Read-only: would {note}. Nothing written.",
|
|
361
|
+
stopped="done",
|
|
362
|
+
writes=(),
|
|
363
|
+
)
|
|
364
|
+
named = named_project_file(run.options.task, self.project)
|
|
365
|
+
if named and named not in run.writes:
|
|
366
|
+
run.writes.append(named)
|
|
367
|
+
verdict, test_output = _verify_mechanical(self.project)
|
|
368
|
+
run.options.emit("result", test_output)
|
|
369
|
+
if verdict not in {"passed", "no suite"}:
|
|
370
|
+
# The name is bound, but the project is red. A red suite is
|
|
371
|
+
# never a finished run: the mechanical pass hands the real
|
|
372
|
+
# failure to the model, and so does this.
|
|
373
|
+
run.test_note = test_output
|
|
374
|
+
return None
|
|
375
|
+
tail = (
|
|
376
|
+
"Tests passed."
|
|
377
|
+
if verdict == "passed"
|
|
378
|
+
else "This project has no tests to check it against."
|
|
379
|
+
)
|
|
380
|
+
return AgentResult(
|
|
381
|
+
ok=True,
|
|
382
|
+
summary=f"{note}. {tail}",
|
|
383
|
+
stopped="done",
|
|
384
|
+
writes=tuple(run.writes),
|
|
385
|
+
)
|
|
386
|
+
|
|
387
|
+
# -- what is left is what the model is for --------------------------
|
|
388
|
+
|
|
389
|
+
def _work_with_the_model(self, run: RunState) -> AgentResult:
|
|
390
|
+
options = run.options
|
|
391
|
+
pre = run.preamble
|
|
392
|
+
# The run's memory belongs here, not in the model package: what
|
|
393
|
+
# is kept and what is let go is a harness decision.
|
|
394
|
+
memory = Conversation(
|
|
395
|
+
budget_tokens=CONTEXT_TOKENS, system=pre.system or options.system or ""
|
|
396
|
+
)
|
|
397
|
+
label, generate = make_generate(
|
|
398
|
+
options.engine,
|
|
399
|
+
options.max_tokens,
|
|
400
|
+
model=options.model,
|
|
401
|
+
system=pre.system or options.system,
|
|
402
|
+
memory=memory,
|
|
403
|
+
)
|
|
404
|
+
options.emit("engine", f"{label} project {self.project} mode {pre.brief.kind}")
|
|
405
|
+
state = self._starting_state(run)
|
|
406
|
+
prompt = run.first_prompt()
|
|
407
|
+
steps: list[Step] = []
|
|
408
|
+
|
|
409
|
+
for number in range(1, options.steps + 1):
|
|
410
|
+
draft = generate(prompt)
|
|
411
|
+
_remember(generate, prompt, draft)
|
|
412
|
+
options.emit("draft", f"--- step {number} ---\n{draft}")
|
|
413
|
+
turn = parse_turn_smart(
|
|
414
|
+
draft,
|
|
415
|
+
question=looks_like_question(options.task),
|
|
416
|
+
ship=looks_like_ship(options.task),
|
|
417
|
+
)
|
|
418
|
+
trace = trace_path(options)
|
|
419
|
+
if trace is not None:
|
|
420
|
+
append_turn(
|
|
421
|
+
trace,
|
|
422
|
+
{
|
|
423
|
+
"run": run.trace_id,
|
|
424
|
+
"user": prompt,
|
|
425
|
+
"assistant": draft,
|
|
426
|
+
"action": turn.action if turn else "",
|
|
427
|
+
},
|
|
428
|
+
)
|
|
429
|
+
if turn is None:
|
|
430
|
+
steps.append(Step(number, "", refused="unparsed", draft=draft))
|
|
431
|
+
prompt = f"Could not parse. One Action: {ACTIONS}"
|
|
432
|
+
continue
|
|
433
|
+
|
|
434
|
+
if turn.action == "done":
|
|
435
|
+
blocked = refuse_done(state, turn)
|
|
436
|
+
if blocked:
|
|
437
|
+
steps.append(Step(number, "done", refused=blocked, draft=draft))
|
|
438
|
+
options.emit("refused", blocked)
|
|
439
|
+
prompt = blocked
|
|
440
|
+
continue
|
|
441
|
+
steps.append(Step(number, "done", result=turn.summary, draft=draft))
|
|
442
|
+
# The model has run out of refusals but still has nothing
|
|
443
|
+
# to show. Let the run end; do not let it end as a win.
|
|
444
|
+
unproven = done_without_proof(state, turn)
|
|
445
|
+
return AgentResult(
|
|
446
|
+
ok=not unproven,
|
|
447
|
+
summary=unproven or turn.summary or "done",
|
|
448
|
+
stopped="done",
|
|
449
|
+
steps=tuple(steps),
|
|
450
|
+
writes=tuple(run.writes),
|
|
451
|
+
)
|
|
452
|
+
|
|
453
|
+
# Policy first, ask included: the cap on repeated questions
|
|
454
|
+
# lives there, so it has to run before the question is put.
|
|
455
|
+
blocked = refuse_before(state, turn)
|
|
456
|
+
if blocked:
|
|
457
|
+
steps.append(
|
|
458
|
+
Step(number, turn.action, turn.path, refused=blocked, draft=draft)
|
|
459
|
+
)
|
|
460
|
+
options.emit("refused", blocked)
|
|
461
|
+
prompt = blocked
|
|
462
|
+
continue
|
|
463
|
+
|
|
464
|
+
if turn.action == "ask":
|
|
465
|
+
question = _question_from(turn)
|
|
466
|
+
state.questions_asked += 1
|
|
467
|
+
answer = self._ask(question, options)
|
|
468
|
+
if answer is None:
|
|
469
|
+
steps.append(
|
|
470
|
+
Step(number, "ask", result=question.render(), draft=draft)
|
|
471
|
+
)
|
|
472
|
+
return AgentResult(
|
|
473
|
+
ok=False,
|
|
474
|
+
summary=question.render(),
|
|
475
|
+
stopped="question",
|
|
476
|
+
steps=tuple(steps),
|
|
477
|
+
writes=tuple(run.writes),
|
|
478
|
+
)
|
|
479
|
+
steps.append(Step(number, "ask", result=answer, draft=draft))
|
|
480
|
+
prompt = f"The user answered: {answer}\n\nNext Action:"
|
|
481
|
+
continue
|
|
482
|
+
|
|
483
|
+
result = self._carry_out(turn, state, run)
|
|
484
|
+
result, nudge = _nudge_after_action(
|
|
485
|
+
self.project, state, turn, result, pre.target
|
|
486
|
+
)
|
|
487
|
+
options.emit("result", result)
|
|
488
|
+
steps.append(
|
|
489
|
+
Step(number, turn.action, state.last_path, result=result, draft=draft)
|
|
490
|
+
)
|
|
491
|
+
prompt = (
|
|
492
|
+
f"Tool result:\n{result}\n\n{nudge}"
|
|
493
|
+
if nudge
|
|
494
|
+
else f"Tool result:\n{result}\n\nNext Action:"
|
|
495
|
+
)
|
|
496
|
+
|
|
497
|
+
return AgentResult(
|
|
498
|
+
ok=False,
|
|
499
|
+
summary=f"stopped after {options.steps} steps",
|
|
500
|
+
stopped="steps",
|
|
501
|
+
steps=tuple(steps),
|
|
502
|
+
writes=tuple(run.writes),
|
|
503
|
+
)
|
|
504
|
+
|
|
505
|
+
def _starting_state(self, run: RunState) -> LoopState:
|
|
506
|
+
pre, options = run.preamble, run.options
|
|
507
|
+
return LoopState(
|
|
508
|
+
task=options.task,
|
|
509
|
+
project=self.project,
|
|
510
|
+
located_path=pre.located_path,
|
|
511
|
+
located_signature=pre.located_signature,
|
|
512
|
+
prelude_ran=bool(pre.pre_text),
|
|
513
|
+
allow_writes=options.allow_writes,
|
|
514
|
+
last_path=pre.located_path,
|
|
515
|
+
instructions=_instruction_lines(pre),
|
|
516
|
+
scope=options.scope,
|
|
517
|
+
autofixed=bool(pre.autofix),
|
|
518
|
+
# Anything already written counts, not only the mechanical
|
|
519
|
+
# pass: a person's answer to an unbindable typo is written
|
|
520
|
+
# before the model starts, and leaving that out let the model
|
|
521
|
+
# say `done` over a suite nobody had run.
|
|
522
|
+
wrote_something=bool(pre.autofix or run.writes),
|
|
523
|
+
existing_paths=pre.existing_paths,
|
|
524
|
+
design_report=(
|
|
525
|
+
render_design_review(self.project, options.scope)
|
|
526
|
+
if looks_like_design_loop(options.task)
|
|
527
|
+
else ""
|
|
528
|
+
),
|
|
529
|
+
)
|
|
530
|
+
|
|
531
|
+
def _carry_out(self, turn, state: LoopState, run: RunState) -> str:
|
|
532
|
+
"""Run one action and record what it changed."""
|
|
533
|
+
try:
|
|
534
|
+
result, state.last_path = run_action(
|
|
535
|
+
self.project,
|
|
536
|
+
turn,
|
|
537
|
+
state.last_path,
|
|
538
|
+
run.options.scope,
|
|
539
|
+
run.preamble.target,
|
|
540
|
+
task=run.options.task,
|
|
541
|
+
)
|
|
542
|
+
except (ValueError, OSError) as exc:
|
|
543
|
+
return str(exc)
|
|
544
|
+
if turn.action == "read" and state.last_path:
|
|
545
|
+
state.files_seen.add(state.last_path)
|
|
546
|
+
if turn.action == "run" and result.startswith("exit 0"):
|
|
547
|
+
state.ran_tests = True
|
|
548
|
+
if result.startswith(("patched", "wrote")):
|
|
549
|
+
run.writes.append(turn.path or state.last_path)
|
|
550
|
+
state.wrote_something = True
|
|
551
|
+
cover = _cover_after_add(
|
|
552
|
+
self.project, run.options.task, turn.path or state.last_path
|
|
553
|
+
)
|
|
554
|
+
if cover:
|
|
555
|
+
for rel in _autofix_paths(f"- {cover}"):
|
|
556
|
+
if rel not in run.writes:
|
|
557
|
+
run.writes.append(rel)
|
|
558
|
+
result = f"{result}\n{cover}"
|
|
559
|
+
return result
|
|
560
|
+
|
|
561
|
+
def _ask(self, question: Question, options: AgentOptions) -> str | None:
|
|
562
|
+
"""None means nobody is there to answer — the caller decides."""
|
|
563
|
+
handler = getattr(options, "on_question", None)
|
|
564
|
+
if handler is None:
|
|
565
|
+
return None
|
|
566
|
+
return handler(question)
|
|
567
|
+
|
|
568
|
+
|
|
569
|
+
def _nudge_after_action(project, state: LoopState, turn, result: str, target):
|
|
570
|
+
"""After a write, run the suite when tests already cover the work."""
|
|
571
|
+
if not should_run_suite_after_write(state, result, state.last_path):
|
|
572
|
+
return result, next_prompt(state, turn, result, target)
|
|
573
|
+
suite = run_python(
|
|
574
|
+
project, ("-m", "unittest", "discover", "-s", "tests", "-q")
|
|
575
|
+
)
|
|
576
|
+
if suite.startswith("exit 0"):
|
|
577
|
+
state.ran_tests = True
|
|
578
|
+
run_turn = SimpleNamespace(action="run", path=getattr(turn, "path", "") or "")
|
|
579
|
+
return (
|
|
580
|
+
f"{result}\n{suite}",
|
|
581
|
+
next_prompt(state, run_turn, suite, target),
|
|
582
|
+
)
|
|
583
|
+
|
|
584
|
+
|
|
585
|
+
def _cover_after_add(project, task: str, path: str) -> str:
|
|
586
|
+
"""Add the AAA test once the new function exists. Empty if not this job."""
|
|
587
|
+
if looks_like_app_loop(task):
|
|
588
|
+
return apply_cli_mock_test(project, task, write=True)
|
|
589
|
+
if not looks_like_add_feature(task):
|
|
590
|
+
return ""
|
|
591
|
+
if "test" in (path or "").replace("\\", "/").lower():
|
|
592
|
+
return ""
|
|
593
|
+
return apply_cover_test(project, task, write=True)
|
|
594
|
+
|
|
595
|
+
|
|
596
|
+
def _autofix_paths(note: str) -> list[str]:
|
|
597
|
+
"""Paths named in mechanical-fix notes, in the order they were written."""
|
|
598
|
+
found: list[str] = []
|
|
599
|
+
for line in note.splitlines():
|
|
600
|
+
if not line.startswith("- ") or " in " not in line:
|
|
601
|
+
continue
|
|
602
|
+
tail = line.rsplit(" in ", 1)[-1].strip()
|
|
603
|
+
if tail.endswith(".py") and tail not in found:
|
|
604
|
+
found.append(tail)
|
|
605
|
+
return found
|
|
606
|
+
|
|
607
|
+
|
|
608
|
+
def _verify_mechanical(project) -> tuple[str, str]:
|
|
609
|
+
"""Run the project suite after a mechanical fix. No model.
|
|
610
|
+
|
|
611
|
+
Returns "passed", "failed", or "no suite". A project with no tests has
|
|
612
|
+
not failed anything, and saying so keeps the loop from asking the model
|
|
613
|
+
to repair a failure that does not exist.
|
|
614
|
+
"""
|
|
615
|
+
result = run_python(project, ("-m", "unittest", "discover", "-s", "tests", "-q"))
|
|
616
|
+
if result.startswith("exit 0"):
|
|
617
|
+
return "passed", result
|
|
618
|
+
if "no tests/ directory" in result:
|
|
619
|
+
return "no suite", result
|
|
620
|
+
return "failed", result
|
|
621
|
+
|
|
622
|
+
|
|
623
|
+
def leftover_bind_question(task: str, project) -> Question | None:
|
|
624
|
+
"""Ask when a named file holds a typo the harness must not guess at.
|
|
625
|
+
|
|
626
|
+
`stauts` inside `def status` reads as a misspelling of `status`, and
|
|
627
|
+
`status` is the method's own name, which is not in scope in its body.
|
|
628
|
+
Binding it writes `return status`, still a NameError, in a tenth of a
|
|
629
|
+
second and reports success. Sending it to the model instead spent
|
|
630
|
+
twenty steps and left `return stauts` untouched. A person has to say
|
|
631
|
+
what was meant. Their answer is written as a Constant or an in-scope
|
|
632
|
+
name. The method name is still refused.
|
|
633
|
+
|
|
634
|
+
Only a name that looks like a typo counts. Any other undefined name
|
|
635
|
+
is work the model can do: a missing import, or something the task is
|
|
636
|
+
asking to be written. Asking about those would stop a run that had
|
|
637
|
+
every chance of finishing.
|
|
638
|
+
"""
|
|
639
|
+
found = unbound_typo(task, project)
|
|
640
|
+
if found is None:
|
|
641
|
+
return None
|
|
642
|
+
shown = ", ".join(f"`{name}`" for name in found.near[:3])
|
|
643
|
+
return Question(
|
|
644
|
+
f"`{found.bad}` in {found.rel} looks like {shown}, but none of those "
|
|
645
|
+
"is in scope where it is used. What did you mean?",
|
|
646
|
+
)
|
|
647
|
+
|
|
648
|
+
|
|
649
|
+
def opening_question(task: str, pre) -> Question | None:
|
|
650
|
+
"""Return a question to put before the run starts, or None to proceed.
|
|
651
|
+
|
|
652
|
+
Only returned when the task names nothing the agent can search for and
|
|
653
|
+
the harness did not find a file on its own.
|
|
654
|
+
"""
|
|
655
|
+
if pre.located_path or not looks_unclear(task):
|
|
656
|
+
return None
|
|
657
|
+
options = tuple(
|
|
658
|
+
item for item in (pre.target.module, pre.target.test) if item
|
|
659
|
+
)
|
|
660
|
+
return Question(
|
|
661
|
+
f'"{task.strip()}" does not name a file or a function. '
|
|
662
|
+
"Which file should I work on?",
|
|
663
|
+
options,
|
|
664
|
+
)
|
|
665
|
+
|
|
666
|
+
|
|
667
|
+
def _instruction_lines(pre) -> tuple[str, ...]:
|
|
668
|
+
"""Every line the model was handed, so an echo of any of them is caught.
|
|
669
|
+
|
|
670
|
+
The skills were checked but the system prompt was not, and its examples
|
|
671
|
+
are handed over on every single turn. Asked what a function returns, an
|
|
672
|
+
8B answered with the system prompt's own example line: "one short
|
|
673
|
+
question, when the task could mean two different things".
|
|
674
|
+
"""
|
|
675
|
+
lines: list[str] = []
|
|
676
|
+
sources = [skill.body for skill in pre.skills]
|
|
677
|
+
if getattr(pre, "system", ""):
|
|
678
|
+
sources.append(pre.system)
|
|
679
|
+
for body in sources:
|
|
680
|
+
lines.extend(
|
|
681
|
+
line.strip()
|
|
682
|
+
for line in body.splitlines()
|
|
683
|
+
if len(line.strip()) >= 12 and not line.strip().startswith("Action:")
|
|
684
|
+
)
|
|
685
|
+
return tuple(lines)
|
|
686
|
+
|
|
687
|
+
|
|
688
|
+
def _with_task(options: AgentOptions, task: str) -> AgentOptions:
|
|
689
|
+
from dataclasses import replace
|
|
690
|
+
|
|
691
|
+
return replace(options, task=task)
|
|
692
|
+
|
|
693
|
+
|
|
694
|
+
def _remember(generate, prompt: str, draft: str) -> None:
|
|
695
|
+
"""Hand the exchange to the run's memory, if it keeps one."""
|
|
696
|
+
memory = getattr(generate, "memory", None)
|
|
697
|
+
if memory is None:
|
|
698
|
+
return
|
|
699
|
+
memory.remember(prompt, draft)
|