py-harness-cli 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. finetune/__init__.py +1 -0
  2. finetune/agent_system.py +41 -0
  3. finetune/agent_traces.py +157 -0
  4. finetune/everyday.py +30 -0
  5. finetune/hf_ollama.py +158 -0
  6. finetune/huggingface_store.py +144 -0
  7. finetune/models.py +74 -0
  8. finetune/paths.py +11 -0
  9. finetune/python_vibe.py +788 -0
  10. finetune/splits.py +54 -0
  11. finetune/systems.py +9 -0
  12. harness/__init__.py +42 -0
  13. harness/__main__.py +8 -0
  14. harness/act/__init__.py +6 -0
  15. harness/act/autofix/__init__.py +110 -0
  16. harness/act/autofix/additions.py +217 -0
  17. harness/act/autofix/conflicts.py +124 -0
  18. harness/act/autofix/cover.py +419 -0
  19. harness/act/autofix/mechanical.py +151 -0
  20. harness/act/autofix/missing_imports.py +50 -0
  21. harness/act/autofix/moves.py +439 -0
  22. harness/act/autofix/names.py +339 -0
  23. harness/act/autofix/scaffold.py +224 -0
  24. harness/act/code.py +157 -0
  25. harness/act/gate.py +229 -0
  26. harness/act/parse.py +247 -0
  27. harness/act/patch_fix.py +138 -0
  28. harness/act/tools.py +244 -0
  29. harness/agent/__init__.py +11 -0
  30. harness/agent/dispatch.py +235 -0
  31. harness/agent/loop.py +699 -0
  32. harness/agent/options.py +144 -0
  33. harness/agent/policy.py +856 -0
  34. harness/agent/prompt.py +170 -0
  35. harness/cli.py +393 -0
  36. harness/editor_kit.py +265 -0
  37. harness/guard/__init__.py +6 -0
  38. harness/guard/fallbacks.py +6 -0
  39. harness/guard/loop_guard.py +57 -0
  40. harness/guard/python_vibe.py +68 -0
  41. harness/guard/run.py +41 -0
  42. harness/guard/types.py +19 -0
  43. harness/locate.py +767 -0
  44. harness/mcp_stdio.py +306 -0
  45. harness/memory/__init__.py +5 -0
  46. harness/memory/conversation.py +104 -0
  47. harness/model/__init__.py +6 -0
  48. harness/model/chat_backend.py +100 -0
  49. harness/model/engine.py +165 -0
  50. harness/model/ollama_generate.py +60 -0
  51. harness/model/openai_generate.py +156 -0
  52. harness/model/outbound.py +83 -0
  53. harness/model/route.py +90 -0
  54. harness/observe/__init__.py +6 -0
  55. harness/observe/eval_gate.py +80 -0
  56. harness/observe/eval_loop.py +185 -0
  57. harness/observe/eval_tasks.py +399 -0
  58. harness/observe/report_md.py +102 -0
  59. harness/observe/trace_record.py +79 -0
  60. harness/openai_api.py +81 -0
  61. harness/paths.py +88 -0
  62. harness/py.typed +0 -0
  63. harness/scan/__init__.py +6 -0
  64. harness/scan/app_spec.py +338 -0
  65. harness/scan/design.py +112 -0
  66. harness/scan/existing.py +131 -0
  67. harness/scan/layout.py +254 -0
  68. harness/scan/names.py +308 -0
  69. harness/scan/project_brief.py +287 -0
  70. harness/scan/project_docs.py +42 -0
  71. harness/scan/project_scan.py +49 -0
  72. harness/scan/repo_map.py +101 -0
  73. harness/secrets.py +39 -0
  74. harness/server.py +199 -0
  75. harness/ship/__init__.py +1 -0
  76. harness/ship/bot_pr.py +221 -0
  77. harness/ship/git_ship.py +262 -0
  78. harness/ship/identity.py +62 -0
  79. harness/ship/ticket.py +251 -0
  80. harness/skillkit/__init__.py +6 -0
  81. harness/skillkit/catalog.py +241 -0
  82. harness/skillkit/refuse_change.py +640 -0
  83. harness/skillkit/refuse_finish.py +295 -0
  84. harness/skillkit/target.py +238 -0
  85. harness/task.py +717 -0
  86. py_harness_cli-0.3.0.dist-info/METADATA +177 -0
  87. py_harness_cli-0.3.0.dist-info/RECORD +92 -0
  88. py_harness_cli-0.3.0.dist-info/WHEEL +5 -0
  89. py_harness_cli-0.3.0.dist-info/entry_points.txt +3 -0
  90. py_harness_cli-0.3.0.dist-info/licenses/LICENSE +202 -0
  91. py_harness_cli-0.3.0.dist-info/licenses/NOTICE +6 -0
  92. py_harness_cli-0.3.0.dist-info/top_level.txt +2 -0
harness/agent/loop.py ADDED
@@ -0,0 +1,699 @@
1
+ """Run one task from start to finish.
2
+
3
+ from harness import Agent, AgentOptions
4
+
5
+ result = Agent(AgentOptions(project=Path("~/app"))).run("fix the NameError")
6
+
7
+ This class is responsible for the order of steps and nothing else. It asks
8
+ `harness.agent.prompt` what to send to the model, `harness.agent.policy`
9
+ whether a proposed action is allowed, and `harness.agent.dispatch` to carry
10
+ an allowed action out.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import uuid
16
+ from dataclasses import dataclass, field
17
+ from pathlib import Path
18
+ from types import SimpleNamespace
19
+
20
+ from harness.act.autofix import (
21
+ apply_cli_mock_test,
22
+ apply_cover_test,
23
+ apply_person_bind,
24
+ unbound_typo,
25
+ )
26
+ from harness.act.parse import parse_turn_smart
27
+ from harness.act.tools import run_python
28
+ from harness.scan.names import undefined_in_file
29
+ from harness.agent.dispatch import ACTIONS, run_action
30
+ from harness.agent.options import AgentOptions, AgentResult, Step
31
+ from harness.agent.policy import (
32
+ LoopState,
33
+ done_without_proof,
34
+ next_prompt,
35
+ refuse_before,
36
+ refuse_done,
37
+ should_run_suite_after_write,
38
+ )
39
+ from harness.agent.prompt import Preamble, build_preamble
40
+ from harness.locate import named_file_review_summary
41
+ from harness.memory import Conversation
42
+ from harness.model.engine import make_generate
43
+ from harness.model.ollama_generate import CONTEXT_TOKENS
44
+ from harness.observe.trace_record import append_turn, default_trace_path
45
+ from harness.scan.design import render_design_review
46
+ from harness.task import (
47
+ looks_like_add_feature,
48
+ looks_like_app_loop,
49
+ looks_like_bugfix,
50
+ looks_like_design_loop,
51
+ looks_like_question,
52
+ looks_like_ship,
53
+ looks_unclear,
54
+ named_project_file,
55
+ )
56
+
57
+
58
+ @dataclass(frozen=True)
59
+ class Question:
60
+ """A question the agent needs answered before it can continue.
61
+
62
+ Fields:
63
+ text: the question.
64
+ options: answers the agent considers likely. May be empty.
65
+ """
66
+
67
+ text: str
68
+ options: tuple[str, ...] = ()
69
+
70
+ def render(self) -> str:
71
+ if not self.options:
72
+ return self.text
73
+ listed = "\n".join(f" {i}. {opt}" for i, opt in enumerate(self.options, 1))
74
+ return f"{self.text}\n{listed}"
75
+
76
+
77
+ def trace_path(options: AgentOptions) -> Path | None:
78
+ """Where this run writes its turns, or None when it writes none.
79
+
80
+ Recording is on unless asked not to. A run that records nothing
81
+ leaves no way to measure it afterwards, and there is no getting the
82
+ trace back: this project reached sixty-five rows of training data
83
+ while doing a week of real work, because the flag was opt-in.
84
+ """
85
+ if options.keep_no_record:
86
+ return None
87
+ if options.record is not None:
88
+ return options.record.expanduser()
89
+ return default_trace_path(options.resolved_project())
90
+
91
+
92
+ def new_trace_id() -> str:
93
+ """A short name for one run, so its turns can be found together.
94
+
95
+ A recorded turn carried no sign of which run it came from, so a turn
96
+ from a run that finished the job looked exactly like a turn from one
97
+ that spent twenty steps and wrote nothing. About a third of runs
98
+ fail, and training on both without being able to tell them apart
99
+ teaches the failure alongside the work.
100
+ """
101
+ return uuid.uuid4().hex[:12]
102
+
103
+
104
+ def _trace_result(
105
+ options: AgentOptions, result: AgentResult, trace_id: str
106
+ ) -> None:
107
+ """Keep a row saying how the run ended, and under which id.
108
+
109
+ `trace_id` has no default on purpose. It had one, and a caller that
110
+ forgot it wrote a closing row signed with an empty string — which
111
+ reads as a run whose turns cannot be found, so its outcome could not
112
+ be used to filter anything. Three of the first thirty-five turns
113
+ collected after the change were exactly that.
114
+ """
115
+ dest = trace_path(options)
116
+ if dest is None:
117
+ return
118
+ append_turn(
119
+ dest,
120
+ {
121
+ "run": trace_id,
122
+ "user": options.task,
123
+ "assistant": result.summary,
124
+ "action": result.stopped,
125
+ "stopped": result.stopped,
126
+ "ok": result.ok,
127
+ },
128
+ )
129
+
130
+
131
+ def _question_from(turn) -> Question:
132
+ text = (turn.query or turn.summary or "").strip() or "What should I do?"
133
+ raw = turn.append or turn.replace or ""
134
+ options = tuple(
135
+ line.strip(" -*\t")
136
+ for line in raw.splitlines()
137
+ if line.strip(" -*\t")
138
+ )
139
+ return Question(text, options[:4])
140
+
141
+
142
+ @dataclass
143
+ class RunState:
144
+ """What one run carries while it decides whether to call the model.
145
+
146
+ `run()` used to hold all of this in local variables across two
147
+ hundred lines, which is why the steps could not be read or tested
148
+ apart from each other.
149
+
150
+ Fields:
151
+ options: the request, replaced when the user answers a question.
152
+ preamble: what the harness found before any model turn.
153
+ trace_id: a short name for this run, stamped on every turn it
154
+ records, so the turns of a run that worked can be told from
155
+ the turns of one that did not.
156
+ writes: project-relative paths this run has changed.
157
+ test_note: what the suite said after a mechanical fix, when that
158
+ fix was not the end of the job. It is put to the model so it
159
+ starts from the real failure instead of looking for one.
160
+ """
161
+
162
+ options: AgentOptions
163
+ preamble: object
164
+ trace_id: str = field(default_factory=new_trace_id)
165
+ writes: list[str] = field(default_factory=list)
166
+ test_note: str = ""
167
+
168
+ def answer_was(self, answer: str, *, rebuild: bool = False) -> None:
169
+ """Fold the user's answer into the task and say so in the trace."""
170
+ self.options = _with_task(self.options, f"{self.options.task} ({answer})")
171
+ if rebuild:
172
+ self.preamble = build_preamble(self.options)
173
+ self.options.emit("preamble", f"user answered: {answer}")
174
+
175
+ def mechanical_note(self, fallback: str) -> str:
176
+ """The first line of what the mechanical pass reported."""
177
+ return next(
178
+ (
179
+ line[2:]
180
+ for line in (getattr(self.preamble, "autofix", "") or "").splitlines()
181
+ if line.startswith("- ")
182
+ ),
183
+ fallback,
184
+ )
185
+
186
+ def first_prompt(self) -> str:
187
+ """The prompt the model opens on."""
188
+ prompt = self.preamble.prompt
189
+ if not self.test_note:
190
+ return prompt
191
+ return (
192
+ f"{prompt}\n\nHarness ran tests after the mechanical fix:\n"
193
+ f"{self.test_note}\n"
194
+ "Action: patch the remaining failure, or Action: done if "
195
+ "the task is already met."
196
+ )
197
+
198
+
199
+ class Agent:
200
+ """Runs one task against one project."""
201
+
202
+ def __init__(self, options: AgentOptions) -> None:
203
+ self.options = options
204
+ self.project = options.resolved_project()
205
+
206
+ def preamble(self, task: str | None = None) -> Preamble:
207
+ options = self.options if task is None else _with_task(self.options, task)
208
+ return build_preamble(options)
209
+
210
+ def run(self, task: str | None = None) -> AgentResult:
211
+ """Answer the task, and say honestly how the run ended.
212
+
213
+ Read as four questions asked in order, before the model is
214
+ loaded at all: is the task clear enough to start from, is
215
+ reading the file the whole job, can the harness make the change
216
+ on its own, and is there a typo only a person can settle. Each
217
+ one either finishes the run or hands on to the next. Whatever
218
+ survives all four is what the model is actually needed for.
219
+ """
220
+ options = self.options if task is None else _with_task(self.options, task)
221
+ if not options.task.strip():
222
+ raise ValueError("task required")
223
+ run = RunState(options=options, preamble=build_preamble(options))
224
+ run.options.emit("preamble", run.preamble.pre_text or "")
225
+
226
+ for decide in (
227
+ self._settle_an_unclear_task,
228
+ self._read_the_file_if_that_is_the_whole_job,
229
+ self._make_the_change_without_a_model,
230
+ self._settle_a_typo_only_a_person_can,
231
+ ):
232
+ finished = decide(run)
233
+ if finished is not None:
234
+ _trace_result(run.options, finished, run.trace_id)
235
+ return finished
236
+
237
+ # Every run ends with a row saying how it ended. Without one,
238
+ # the turns of a run that spent its whole budget look exactly
239
+ # like the turns of a run that did the job.
240
+ result = self._work_with_the_model(run)
241
+ _trace_result(run.options, result, run.trace_id)
242
+ return result
243
+
244
+ # -- the four questions asked before the model is loaded ------------
245
+
246
+ def _settle_an_unclear_task(self, run: RunState) -> AgentResult | None:
247
+ """A task naming no file and no symbol cannot be started from.
248
+
249
+ The harness asks rather than relying on the model to notice: a
250
+ small model reaches for `patch` long before it reaches for `ask`.
251
+ """
252
+ question = opening_question(run.options.task, run.preamble)
253
+ if question is None:
254
+ return None
255
+ answer = self._ask(question, run.options)
256
+ if answer is None:
257
+ return AgentResult(ok=False, summary=question.render(), stopped="question")
258
+ run.answer_was(answer, rebuild=True)
259
+ return None
260
+
261
+ def _read_the_file_if_that_is_the_whole_job(
262
+ self, run: RunState
263
+ ) -> AgentResult | None:
264
+ """A review of a named file is reading, not editing."""
265
+ review = named_file_review_summary(self.project, run.options.task)
266
+ if not review:
267
+ return None
268
+ run.options.emit("result", review)
269
+ return AgentResult(ok=True, summary=review, stopped="done")
270
+
271
+ def _make_the_change_without_a_model(self, run: RunState) -> AgentResult | None:
272
+ """Apply the mechanical repairs, and stop if they were enough.
273
+
274
+ These are the cases that cannot be got wrong: a misspelling with
275
+ exactly one candidate in scope, a missing import for a module
276
+ everyone knows, a test appended where one already exists. They
277
+ take a tenth of a second and give the same answer every time.
278
+ """
279
+ if not run.preamble.autofix:
280
+ return None
281
+ run.writes.extend(_autofix_paths(run.preamble.autofix))
282
+ if not run.options.allow_writes:
283
+ return AgentResult(
284
+ ok=True,
285
+ summary=f"Read-only: would {run.mechanical_note('mechanical fix')}. "
286
+ "Nothing written.",
287
+ stopped="done",
288
+ writes=(),
289
+ )
290
+ still_undefined = self._names_left_undefined(run)
291
+ verdict, test_output = _verify_mechanical(self.project)
292
+ run.options.emit("result", test_output)
293
+ if still_undefined:
294
+ run.test_note = (
295
+ f"undefined name {still_undefined[0]} after the "
296
+ "mechanical fix. The suite is not enough."
297
+ )
298
+ run.options.emit("result", run.test_note)
299
+ return None
300
+ if verdict not in {"passed", "no suite"}:
301
+ run.test_note = test_output
302
+ return None
303
+ tail = (
304
+ "Tests passed."
305
+ if verdict == "passed"
306
+ else "This project has no tests to check it against."
307
+ )
308
+ note = run.mechanical_note("mechanical fix applied")
309
+ return AgentResult(
310
+ ok=True,
311
+ summary=f"{note}. {tail}",
312
+ stopped="done",
313
+ writes=tuple(run.writes),
314
+ )
315
+
316
+ def _names_left_undefined(self, run: RunState) -> list[str]:
317
+ """Names still unbound in what the mechanical pass just wrote."""
318
+ if not looks_like_bugfix(run.options.task):
319
+ return []
320
+ found: list[str] = []
321
+ for rel in run.writes:
322
+ found.extend(undefined_in_file(self.project / rel))
323
+ return found
324
+
325
+ def _settle_a_typo_only_a_person_can(self, run: RunState) -> AgentResult | None:
326
+ """A misspelling with no safe candidate is a question, not a guess."""
327
+ question = leftover_bind_question(run.options.task, self.project)
328
+ if question is None:
329
+ return None
330
+ answer = self._ask(question, run.options)
331
+ if answer is None:
332
+ return AgentResult(
333
+ ok=False,
334
+ summary=question.render(),
335
+ stopped="question",
336
+ writes=tuple(run.writes),
337
+ )
338
+ # Asking is not writing, so a read-only run may still ask. What it
339
+ # may not do is act on the answer: `ask` and `--dry-run` both
340
+ # promise the folder is left alone.
341
+ note = apply_person_bind(
342
+ self.project,
343
+ run.options.task,
344
+ answer,
345
+ write=run.options.allow_writes,
346
+ )
347
+ if not note:
348
+ return AgentResult(
349
+ ok=False,
350
+ summary=(
351
+ f"{question.render()} "
352
+ "That answer is still not something this method can return."
353
+ ),
354
+ stopped="question",
355
+ writes=tuple(run.writes),
356
+ )
357
+ if not run.options.allow_writes:
358
+ return AgentResult(
359
+ ok=True,
360
+ summary=f"Read-only: would {note}. Nothing written.",
361
+ stopped="done",
362
+ writes=(),
363
+ )
364
+ named = named_project_file(run.options.task, self.project)
365
+ if named and named not in run.writes:
366
+ run.writes.append(named)
367
+ verdict, test_output = _verify_mechanical(self.project)
368
+ run.options.emit("result", test_output)
369
+ if verdict not in {"passed", "no suite"}:
370
+ # The name is bound, but the project is red. A red suite is
371
+ # never a finished run: the mechanical pass hands the real
372
+ # failure to the model, and so does this.
373
+ run.test_note = test_output
374
+ return None
375
+ tail = (
376
+ "Tests passed."
377
+ if verdict == "passed"
378
+ else "This project has no tests to check it against."
379
+ )
380
+ return AgentResult(
381
+ ok=True,
382
+ summary=f"{note}. {tail}",
383
+ stopped="done",
384
+ writes=tuple(run.writes),
385
+ )
386
+
387
+ # -- what is left is what the model is for --------------------------
388
+
389
+ def _work_with_the_model(self, run: RunState) -> AgentResult:
390
+ options = run.options
391
+ pre = run.preamble
392
+ # The run's memory belongs here, not in the model package: what
393
+ # is kept and what is let go is a harness decision.
394
+ memory = Conversation(
395
+ budget_tokens=CONTEXT_TOKENS, system=pre.system or options.system or ""
396
+ )
397
+ label, generate = make_generate(
398
+ options.engine,
399
+ options.max_tokens,
400
+ model=options.model,
401
+ system=pre.system or options.system,
402
+ memory=memory,
403
+ )
404
+ options.emit("engine", f"{label} project {self.project} mode {pre.brief.kind}")
405
+ state = self._starting_state(run)
406
+ prompt = run.first_prompt()
407
+ steps: list[Step] = []
408
+
409
+ for number in range(1, options.steps + 1):
410
+ draft = generate(prompt)
411
+ _remember(generate, prompt, draft)
412
+ options.emit("draft", f"--- step {number} ---\n{draft}")
413
+ turn = parse_turn_smart(
414
+ draft,
415
+ question=looks_like_question(options.task),
416
+ ship=looks_like_ship(options.task),
417
+ )
418
+ trace = trace_path(options)
419
+ if trace is not None:
420
+ append_turn(
421
+ trace,
422
+ {
423
+ "run": run.trace_id,
424
+ "user": prompt,
425
+ "assistant": draft,
426
+ "action": turn.action if turn else "",
427
+ },
428
+ )
429
+ if turn is None:
430
+ steps.append(Step(number, "", refused="unparsed", draft=draft))
431
+ prompt = f"Could not parse. One Action: {ACTIONS}"
432
+ continue
433
+
434
+ if turn.action == "done":
435
+ blocked = refuse_done(state, turn)
436
+ if blocked:
437
+ steps.append(Step(number, "done", refused=blocked, draft=draft))
438
+ options.emit("refused", blocked)
439
+ prompt = blocked
440
+ continue
441
+ steps.append(Step(number, "done", result=turn.summary, draft=draft))
442
+ # The model has run out of refusals but still has nothing
443
+ # to show. Let the run end; do not let it end as a win.
444
+ unproven = done_without_proof(state, turn)
445
+ return AgentResult(
446
+ ok=not unproven,
447
+ summary=unproven or turn.summary or "done",
448
+ stopped="done",
449
+ steps=tuple(steps),
450
+ writes=tuple(run.writes),
451
+ )
452
+
453
+ # Policy first, ask included: the cap on repeated questions
454
+ # lives there, so it has to run before the question is put.
455
+ blocked = refuse_before(state, turn)
456
+ if blocked:
457
+ steps.append(
458
+ Step(number, turn.action, turn.path, refused=blocked, draft=draft)
459
+ )
460
+ options.emit("refused", blocked)
461
+ prompt = blocked
462
+ continue
463
+
464
+ if turn.action == "ask":
465
+ question = _question_from(turn)
466
+ state.questions_asked += 1
467
+ answer = self._ask(question, options)
468
+ if answer is None:
469
+ steps.append(
470
+ Step(number, "ask", result=question.render(), draft=draft)
471
+ )
472
+ return AgentResult(
473
+ ok=False,
474
+ summary=question.render(),
475
+ stopped="question",
476
+ steps=tuple(steps),
477
+ writes=tuple(run.writes),
478
+ )
479
+ steps.append(Step(number, "ask", result=answer, draft=draft))
480
+ prompt = f"The user answered: {answer}\n\nNext Action:"
481
+ continue
482
+
483
+ result = self._carry_out(turn, state, run)
484
+ result, nudge = _nudge_after_action(
485
+ self.project, state, turn, result, pre.target
486
+ )
487
+ options.emit("result", result)
488
+ steps.append(
489
+ Step(number, turn.action, state.last_path, result=result, draft=draft)
490
+ )
491
+ prompt = (
492
+ f"Tool result:\n{result}\n\n{nudge}"
493
+ if nudge
494
+ else f"Tool result:\n{result}\n\nNext Action:"
495
+ )
496
+
497
+ return AgentResult(
498
+ ok=False,
499
+ summary=f"stopped after {options.steps} steps",
500
+ stopped="steps",
501
+ steps=tuple(steps),
502
+ writes=tuple(run.writes),
503
+ )
504
+
505
+ def _starting_state(self, run: RunState) -> LoopState:
506
+ pre, options = run.preamble, run.options
507
+ return LoopState(
508
+ task=options.task,
509
+ project=self.project,
510
+ located_path=pre.located_path,
511
+ located_signature=pre.located_signature,
512
+ prelude_ran=bool(pre.pre_text),
513
+ allow_writes=options.allow_writes,
514
+ last_path=pre.located_path,
515
+ instructions=_instruction_lines(pre),
516
+ scope=options.scope,
517
+ autofixed=bool(pre.autofix),
518
+ # Anything already written counts, not only the mechanical
519
+ # pass: a person's answer to an unbindable typo is written
520
+ # before the model starts, and leaving that out let the model
521
+ # say `done` over a suite nobody had run.
522
+ wrote_something=bool(pre.autofix or run.writes),
523
+ existing_paths=pre.existing_paths,
524
+ design_report=(
525
+ render_design_review(self.project, options.scope)
526
+ if looks_like_design_loop(options.task)
527
+ else ""
528
+ ),
529
+ )
530
+
531
+ def _carry_out(self, turn, state: LoopState, run: RunState) -> str:
532
+ """Run one action and record what it changed."""
533
+ try:
534
+ result, state.last_path = run_action(
535
+ self.project,
536
+ turn,
537
+ state.last_path,
538
+ run.options.scope,
539
+ run.preamble.target,
540
+ task=run.options.task,
541
+ )
542
+ except (ValueError, OSError) as exc:
543
+ return str(exc)
544
+ if turn.action == "read" and state.last_path:
545
+ state.files_seen.add(state.last_path)
546
+ if turn.action == "run" and result.startswith("exit 0"):
547
+ state.ran_tests = True
548
+ if result.startswith(("patched", "wrote")):
549
+ run.writes.append(turn.path or state.last_path)
550
+ state.wrote_something = True
551
+ cover = _cover_after_add(
552
+ self.project, run.options.task, turn.path or state.last_path
553
+ )
554
+ if cover:
555
+ for rel in _autofix_paths(f"- {cover}"):
556
+ if rel not in run.writes:
557
+ run.writes.append(rel)
558
+ result = f"{result}\n{cover}"
559
+ return result
560
+
561
+ def _ask(self, question: Question, options: AgentOptions) -> str | None:
562
+ """None means nobody is there to answer — the caller decides."""
563
+ handler = getattr(options, "on_question", None)
564
+ if handler is None:
565
+ return None
566
+ return handler(question)
567
+
568
+
569
+ def _nudge_after_action(project, state: LoopState, turn, result: str, target):
570
+ """After a write, run the suite when tests already cover the work."""
571
+ if not should_run_suite_after_write(state, result, state.last_path):
572
+ return result, next_prompt(state, turn, result, target)
573
+ suite = run_python(
574
+ project, ("-m", "unittest", "discover", "-s", "tests", "-q")
575
+ )
576
+ if suite.startswith("exit 0"):
577
+ state.ran_tests = True
578
+ run_turn = SimpleNamespace(action="run", path=getattr(turn, "path", "") or "")
579
+ return (
580
+ f"{result}\n{suite}",
581
+ next_prompt(state, run_turn, suite, target),
582
+ )
583
+
584
+
585
+ def _cover_after_add(project, task: str, path: str) -> str:
586
+ """Add the AAA test once the new function exists. Empty if not this job."""
587
+ if looks_like_app_loop(task):
588
+ return apply_cli_mock_test(project, task, write=True)
589
+ if not looks_like_add_feature(task):
590
+ return ""
591
+ if "test" in (path or "").replace("\\", "/").lower():
592
+ return ""
593
+ return apply_cover_test(project, task, write=True)
594
+
595
+
596
+ def _autofix_paths(note: str) -> list[str]:
597
+ """Paths named in mechanical-fix notes, in the order they were written."""
598
+ found: list[str] = []
599
+ for line in note.splitlines():
600
+ if not line.startswith("- ") or " in " not in line:
601
+ continue
602
+ tail = line.rsplit(" in ", 1)[-1].strip()
603
+ if tail.endswith(".py") and tail not in found:
604
+ found.append(tail)
605
+ return found
606
+
607
+
608
+ def _verify_mechanical(project) -> tuple[str, str]:
609
+ """Run the project suite after a mechanical fix. No model.
610
+
611
+ Returns "passed", "failed", or "no suite". A project with no tests has
612
+ not failed anything, and saying so keeps the loop from asking the model
613
+ to repair a failure that does not exist.
614
+ """
615
+ result = run_python(project, ("-m", "unittest", "discover", "-s", "tests", "-q"))
616
+ if result.startswith("exit 0"):
617
+ return "passed", result
618
+ if "no tests/ directory" in result:
619
+ return "no suite", result
620
+ return "failed", result
621
+
622
+
623
+ def leftover_bind_question(task: str, project) -> Question | None:
624
+ """Ask when a named file holds a typo the harness must not guess at.
625
+
626
+ `stauts` inside `def status` reads as a misspelling of `status`, and
627
+ `status` is the method's own name, which is not in scope in its body.
628
+ Binding it writes `return status`, still a NameError, in a tenth of a
629
+ second and reports success. Sending it to the model instead spent
630
+ twenty steps and left `return stauts` untouched. A person has to say
631
+ what was meant. Their answer is written as a Constant or an in-scope
632
+ name. The method name is still refused.
633
+
634
+ Only a name that looks like a typo counts. Any other undefined name
635
+ is work the model can do: a missing import, or something the task is
636
+ asking to be written. Asking about those would stop a run that had
637
+ every chance of finishing.
638
+ """
639
+ found = unbound_typo(task, project)
640
+ if found is None:
641
+ return None
642
+ shown = ", ".join(f"`{name}`" for name in found.near[:3])
643
+ return Question(
644
+ f"`{found.bad}` in {found.rel} looks like {shown}, but none of those "
645
+ "is in scope where it is used. What did you mean?",
646
+ )
647
+
648
+
649
+ def opening_question(task: str, pre) -> Question | None:
650
+ """Return a question to put before the run starts, or None to proceed.
651
+
652
+ Only returned when the task names nothing the agent can search for and
653
+ the harness did not find a file on its own.
654
+ """
655
+ if pre.located_path or not looks_unclear(task):
656
+ return None
657
+ options = tuple(
658
+ item for item in (pre.target.module, pre.target.test) if item
659
+ )
660
+ return Question(
661
+ f'"{task.strip()}" does not name a file or a function. '
662
+ "Which file should I work on?",
663
+ options,
664
+ )
665
+
666
+
667
+ def _instruction_lines(pre) -> tuple[str, ...]:
668
+ """Every line the model was handed, so an echo of any of them is caught.
669
+
670
+ The skills were checked but the system prompt was not, and its examples
671
+ are handed over on every single turn. Asked what a function returns, an
672
+ 8B answered with the system prompt's own example line: "one short
673
+ question, when the task could mean two different things".
674
+ """
675
+ lines: list[str] = []
676
+ sources = [skill.body for skill in pre.skills]
677
+ if getattr(pre, "system", ""):
678
+ sources.append(pre.system)
679
+ for body in sources:
680
+ lines.extend(
681
+ line.strip()
682
+ for line in body.splitlines()
683
+ if len(line.strip()) >= 12 and not line.strip().startswith("Action:")
684
+ )
685
+ return tuple(lines)
686
+
687
+
688
+ def _with_task(options: AgentOptions, task: str) -> AgentOptions:
689
+ from dataclasses import replace
690
+
691
+ return replace(options, task=task)
692
+
693
+
694
+ def _remember(generate, prompt: str, draft: str) -> None:
695
+ """Hand the exchange to the run's memory, if it keeps one."""
696
+ memory = getattr(generate, "memory", None)
697
+ if memory is None:
698
+ return
699
+ memory.remember(prompt, draft)