py-harness-cli 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. finetune/__init__.py +1 -0
  2. finetune/agent_system.py +41 -0
  3. finetune/agent_traces.py +157 -0
  4. finetune/everyday.py +30 -0
  5. finetune/hf_ollama.py +158 -0
  6. finetune/huggingface_store.py +144 -0
  7. finetune/models.py +74 -0
  8. finetune/paths.py +11 -0
  9. finetune/python_vibe.py +788 -0
  10. finetune/splits.py +54 -0
  11. finetune/systems.py +9 -0
  12. harness/__init__.py +42 -0
  13. harness/__main__.py +8 -0
  14. harness/act/__init__.py +6 -0
  15. harness/act/autofix/__init__.py +110 -0
  16. harness/act/autofix/additions.py +217 -0
  17. harness/act/autofix/conflicts.py +124 -0
  18. harness/act/autofix/cover.py +419 -0
  19. harness/act/autofix/mechanical.py +151 -0
  20. harness/act/autofix/missing_imports.py +50 -0
  21. harness/act/autofix/moves.py +439 -0
  22. harness/act/autofix/names.py +339 -0
  23. harness/act/autofix/scaffold.py +224 -0
  24. harness/act/code.py +157 -0
  25. harness/act/gate.py +229 -0
  26. harness/act/parse.py +247 -0
  27. harness/act/patch_fix.py +138 -0
  28. harness/act/tools.py +244 -0
  29. harness/agent/__init__.py +11 -0
  30. harness/agent/dispatch.py +235 -0
  31. harness/agent/loop.py +699 -0
  32. harness/agent/options.py +144 -0
  33. harness/agent/policy.py +856 -0
  34. harness/agent/prompt.py +170 -0
  35. harness/cli.py +393 -0
  36. harness/editor_kit.py +265 -0
  37. harness/guard/__init__.py +6 -0
  38. harness/guard/fallbacks.py +6 -0
  39. harness/guard/loop_guard.py +57 -0
  40. harness/guard/python_vibe.py +68 -0
  41. harness/guard/run.py +41 -0
  42. harness/guard/types.py +19 -0
  43. harness/locate.py +767 -0
  44. harness/mcp_stdio.py +306 -0
  45. harness/memory/__init__.py +5 -0
  46. harness/memory/conversation.py +104 -0
  47. harness/model/__init__.py +6 -0
  48. harness/model/chat_backend.py +100 -0
  49. harness/model/engine.py +165 -0
  50. harness/model/ollama_generate.py +60 -0
  51. harness/model/openai_generate.py +156 -0
  52. harness/model/outbound.py +83 -0
  53. harness/model/route.py +90 -0
  54. harness/observe/__init__.py +6 -0
  55. harness/observe/eval_gate.py +80 -0
  56. harness/observe/eval_loop.py +185 -0
  57. harness/observe/eval_tasks.py +399 -0
  58. harness/observe/report_md.py +102 -0
  59. harness/observe/trace_record.py +79 -0
  60. harness/openai_api.py +81 -0
  61. harness/paths.py +88 -0
  62. harness/py.typed +0 -0
  63. harness/scan/__init__.py +6 -0
  64. harness/scan/app_spec.py +338 -0
  65. harness/scan/design.py +112 -0
  66. harness/scan/existing.py +131 -0
  67. harness/scan/layout.py +254 -0
  68. harness/scan/names.py +308 -0
  69. harness/scan/project_brief.py +287 -0
  70. harness/scan/project_docs.py +42 -0
  71. harness/scan/project_scan.py +49 -0
  72. harness/scan/repo_map.py +101 -0
  73. harness/secrets.py +39 -0
  74. harness/server.py +199 -0
  75. harness/ship/__init__.py +1 -0
  76. harness/ship/bot_pr.py +221 -0
  77. harness/ship/git_ship.py +262 -0
  78. harness/ship/identity.py +62 -0
  79. harness/ship/ticket.py +251 -0
  80. harness/skillkit/__init__.py +6 -0
  81. harness/skillkit/catalog.py +241 -0
  82. harness/skillkit/refuse_change.py +640 -0
  83. harness/skillkit/refuse_finish.py +295 -0
  84. harness/skillkit/target.py +238 -0
  85. harness/task.py +717 -0
  86. py_harness_cli-0.3.0.dist-info/METADATA +177 -0
  87. py_harness_cli-0.3.0.dist-info/RECORD +92 -0
  88. py_harness_cli-0.3.0.dist-info/WHEEL +5 -0
  89. py_harness_cli-0.3.0.dist-info/entry_points.txt +3 -0
  90. py_harness_cli-0.3.0.dist-info/licenses/LICENSE +202 -0
  91. py_harness_cli-0.3.0.dist-info/licenses/NOTICE +6 -0
  92. py_harness_cli-0.3.0.dist-info/top_level.txt +2 -0
@@ -0,0 +1,856 @@
1
+ """Decide whether a proposed action is allowed, and what to do next.
2
+
3
+ The loop consults three functions, in this order:
4
+
5
+ * `refuse_before` runs before an action is carried out. It returns the
6
+ reason the action is not allowed, or an empty string if it is allowed.
7
+ * `refuse_done` runs when the model reports that the task is finished. It
8
+ returns the reason the work is not finished, or an empty string.
9
+ * `next_prompt` runs after an action has been carried out. It returns the
10
+ single next instruction to send, or an empty string to leave the choice
11
+ to the model.
12
+
13
+ Keeping these separate from the loop means a new rule is a new function
14
+ rather than another branch inside the loop.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import re
20
+ from dataclasses import dataclass, field
21
+ from pathlib import Path
22
+
23
+ from harness.act.tools import read_py
24
+ from harness.agent.dispatch import SHIP_ACTIONS, WRITE_ACTIONS
25
+ from harness.guard.loop_guard import LoopGuard
26
+ from harness.paths import as_project_rel
27
+ from harness.locate import (
28
+ refuse_app_ask,
29
+ refuse_app_overflow_explore,
30
+ refuse_app_tests_first,
31
+ refuse_app_wrong_path,
32
+ refuse_bugfix_explore,
33
+ refuse_bugfix_tests_first,
34
+ refuse_design_dirty,
35
+ refuse_early_done,
36
+ refuse_invented_review,
37
+ refuse_question_ask,
38
+ refuse_question_write,
39
+ refuse_redundant_explore,
40
+ refuse_redundant_locate,
41
+ refuse_shallow_done,
42
+ refuse_thin_review,
43
+ refuse_write_tests_ask,
44
+ )
45
+ from harness.skillkit.catalog import get_skill, render_skill
46
+ from harness.scan.names import undefined_in_file
47
+ from harness.skillkit.refuse_change import (
48
+ refuse_god_target,
49
+ refuse_smell_wrong_file,
50
+ )
51
+ from harness.skillkit.refuse_finish import (
52
+ refuse_app_done,
53
+ refuse_done_oracle,
54
+ refuse_unwired_addition,
55
+ refuse_write_done,
56
+ tests_call,
57
+ )
58
+
59
+ from harness.task import (
60
+ everyday_example_path,
61
+ looks_like_add_feature,
62
+ looks_like_ops,
63
+ looks_like_platform,
64
+ looks_like_bugfix,
65
+ looks_like_design_loop,
66
+ named_project_file,
67
+ looks_like_fix_smell,
68
+ looks_like_write_tests,
69
+ covered_symbol,
70
+ looks_like_merge,
71
+ looks_like_app_loop,
72
+ looks_like_app_overflow,
73
+ looks_like_new_package,
74
+ looks_like_question,
75
+ looks_like_ship,
76
+ looks_like_ticket,
77
+ question_symbol,
78
+ rename_target,
79
+ smell_symbol,
80
+ )
81
+
82
+ MAX_QUESTIONS = 2
83
+ # How often the loop may send a summary back for being too thin.
84
+ MAX_THIN_DONE = 2
85
+ # How often a failed Action: run is sent back with the traceback.
86
+ # One repair is the daily-work default; a second failure is reported.
87
+ MAX_REPAIRS = 1
88
+ # How often a change task may finish without changing anything before the
89
+ # run stops calling it done.
90
+ MAX_EMPTY_DONE = 2
91
+ # A quoted line short enough to appear by chance proves nothing. `return`
92
+ # and `import os` are in half the files in any project.
93
+ MIN_QUOTED_CHARS = 12
94
+ # A Summary this close to a line it was given is an echo, not an answer.
95
+ ECHO_RATIO = 0.75
96
+
97
+
98
+ @dataclass
99
+ class LoopState:
100
+ """Facts the loop has gathered, used to judge the next action.
101
+
102
+ Fields:
103
+ task: the user's request.
104
+ project: directory being worked in.
105
+ located_path: file the harness found before the model started.
106
+ located_signature: the definition line found in that file.
107
+ prelude_ran: whether the harness searched before the model started.
108
+ allow_writes: whether file changes are permitted in this run.
109
+ last_path: file the most recent action applied to.
110
+ ran_tests: whether the test suite has passed during this run.
111
+ design_report: last deterministic structure scan, if any.
112
+ autofixed: the harness already applied a rename or unique typo.
113
+ scope: optional subdirectory the run is limited to.
114
+ questions_asked: how many questions the agent has put to the user.
115
+ thin_done_refused: how often a summary was sent back as too thin.
116
+ instructions: skill lines the model was given, used to detect a
117
+ reply that repeats an instruction instead of answering.
118
+ guard: record of read-only actions already performed.
119
+ files_seen: files whose text this run has in front of it, either
120
+ because it read them or because the harness located them.
121
+ repairs: how many failed runs have already been sent back with
122
+ the traceback. Daily work gets one; a second failure stops
123
+ the nudge so the model reports rather than looping.
124
+ existing_paths: files the prelude already named as holding what
125
+ the task is about. A write to a different path is refused
126
+ until one of these has been read.
127
+ """
128
+
129
+ task: str
130
+ project: Path
131
+ located_path: str = ""
132
+ located_signature: str = ""
133
+ prelude_ran: bool = False
134
+ allow_writes: bool = True
135
+ last_path: str = ""
136
+ ran_tests: bool = False
137
+ design_report: str = ""
138
+ autofixed: bool = False
139
+ scope: str = ""
140
+ questions_asked: int = 0
141
+ wrote_something: bool = False
142
+ empty_done_refused: int = 0
143
+ thin_done_refused: int = 0
144
+ instructions: tuple[str, ...] = ()
145
+ guard: LoopGuard = field(default_factory=LoopGuard)
146
+ files_seen: set[str] = field(default_factory=set)
147
+ repairs: int = 0
148
+ existing_paths: tuple[str, ...] = ()
149
+
150
+
151
+ def refuse_patch_before_reading(state: LoopState, turn) -> str:
152
+ """Reject a Find: for a file this run has never looked at.
153
+
154
+ A `Find:` string has to match the file exactly. One written without
155
+ reading the file is written from memory, and memory is where
156
+ `result = run_case(case, model, steps)` came from — a line that is
157
+ not in the file at all, refused, and then sent again. Across every
158
+ failing run of one task the model went from `grep` straight to
159
+ `patch`, and spent the budget guessing at a line it could have read.
160
+
161
+ An append needs no Find, so it is not refused. Nor is a file the
162
+ harness located, because its text is already in the opening turn.
163
+ """
164
+ if turn.action != "patch" or not turn.find.strip():
165
+ return ""
166
+ rel = as_project_rel(turn.path or state.last_path)
167
+ if not rel:
168
+ return ""
169
+ seen = {as_project_rel(item) for item in state.files_seen}
170
+ if state.located_path:
171
+ seen.add(as_project_rel(state.located_path))
172
+ if rel in seen:
173
+ return ""
174
+ return (
175
+ f"Nothing has read {rel} in this run, so Find: is being written "
176
+ f"from memory. Action: read Path: {rel} first, then copy a whole "
177
+ "line from it."
178
+ )
179
+
180
+
181
+ def refuse_new_path_before_existing(state: LoopState, turn) -> str:
182
+ """Refuse a write to a new file while a prelude-named file is unread."""
183
+ if turn.action not in WRITE_ACTIONS or turn.action == "run":
184
+ return ""
185
+ wanted = tuple(as_project_rel(item) for item in state.existing_paths if item)
186
+ if not wanted:
187
+ return ""
188
+ seen = {as_project_rel(item) for item in state.files_seen}
189
+ if state.located_path:
190
+ seen.add(as_project_rel(state.located_path))
191
+ unread = [item for item in wanted if item not in seen]
192
+ if not unread:
193
+ return ""
194
+ dest = as_project_rel(turn.path or state.last_path)
195
+ if dest in wanted:
196
+ return ""
197
+ return (
198
+ f"This project already has {unread[0]} for this task. "
199
+ f"Action: read Path: {unread[0]} first. Do not write {dest or 'a new file'}."
200
+ )
201
+
202
+
203
+ def refuse_wrong_file(task: str, project: Path, action: str, path: str) -> str:
204
+ """Reject a write to a file other than the one the task named.
205
+
206
+ When a task names exactly one file that exists, that file is the whole
207
+ instruction. An 8B given `src/harness/model/engine.py` was observed
208
+ patching `src/harness/act/patch_fix.py` instead.
209
+ """
210
+ if action not in WRITE_ACTIONS or action == "run":
211
+ return ""
212
+ if looks_like_write_tests(task):
213
+ got = as_project_rel(path)
214
+ parts = got.split("/") if got else []
215
+ if got and "tests" not in parts and not parts[-1].startswith("test_"):
216
+ symbol = covered_symbol(task)
217
+ dest = (
218
+ f"tests/test_{symbol.split('.')[-1]}.py"
219
+ if symbol
220
+ else "tests/test_module.py"
221
+ )
222
+ return f"Tests go in {dest}. Do not change {got}."
223
+ # A file the task names outright beats any routing rule below. "in
224
+ # src/app.py ... fix it for Windows" was being sent to pkg/paths.py,
225
+ # which ignores the one instruction the person gave.
226
+ if named_project_file(task, project):
227
+ return _refuse_other_than_named(task, project, path)
228
+ if looks_like_platform(task) or looks_like_ops(task):
229
+ from harness.skillkit.target import pick_module
230
+
231
+ wanted = (
232
+ pick_module(project, "", task)
233
+ if looks_like_ops(task)
234
+ else everyday_example_path(task)
235
+ )
236
+ got = as_project_rel(path)
237
+ parts = got.split("/") if got else []
238
+ if (
239
+ got
240
+ and wanted
241
+ and got != wanted
242
+ and "tests" not in parts
243
+ and not parts[-1].startswith("test_")
244
+ ):
245
+ return (
246
+ f"This job writes {wanted}. Do not change {got}. "
247
+ f"Action: edit Path: {wanted}"
248
+ )
249
+ if looks_like_add_feature(task):
250
+ from harness.skillkit.target import pick_module
251
+
252
+ wanted = pick_module(project, "", task)
253
+ got = as_project_rel(path)
254
+ parts = got.split("/") if got else []
255
+ if (
256
+ got
257
+ and wanted
258
+ and got != wanted
259
+ and "tests" not in parts
260
+ and not parts[-1].startswith("test_")
261
+ ):
262
+ return (
263
+ f"The new function belongs in {wanted}. Do not change {got}. "
264
+ f"Action: patch Path: {wanted}"
265
+ )
266
+ named = named_project_file(task, project)
267
+ if not named:
268
+ return ""
269
+ wanted = as_project_rel(named)
270
+ got = as_project_rel(path)
271
+ if not got or got == wanted or wanted.endswith(got) or got.endswith(wanted):
272
+ return ""
273
+ # "write tests for apply_discount in src/orders.py" names the source
274
+ # file, but the test belongs beside it, not inside it. A test file is
275
+ # always an allowed destination.
276
+ parts = got.split("/")
277
+ if "tests" in parts or parts[-1].startswith("test_"):
278
+ return ""
279
+ return (
280
+ f"The task names {wanted}. Do not change {got}. "
281
+ f"Action: patch Path: {wanted}"
282
+ )
283
+
284
+
285
+ def _refuse_other_than_named(task: str, project: Path, path: str) -> str:
286
+ """Allow the named file and its tests; refuse anything else."""
287
+ named = named_project_file(task, project)
288
+ wanted = as_project_rel(named)
289
+ got = as_project_rel(path)
290
+ if not got or got == wanted or wanted.endswith(got) or got.endswith(wanted):
291
+ return ""
292
+ parts = got.split("/")
293
+ if "tests" in parts or parts[-1].startswith("test_"):
294
+ return ""
295
+ return (
296
+ f"The task names {wanted}. Do not change {got}. "
297
+ f"Action: patch Path: {wanted}"
298
+ )
299
+
300
+
301
+ def refuse_before(state: LoopState, turn) -> str:
302
+ """The turn is about to run a tool. Return a refusal, or ""."""
303
+ if state.autofixed and turn.action not in {"run", "done"}:
304
+ if turn.action in {"edit", "patch"} and "test" in (
305
+ turn.path or state.last_path or ""
306
+ ).lower():
307
+ return ""
308
+ return (
309
+ "Harness already applied the mechanical fix. "
310
+ "Action: run Argv: -m unittest discover -s tests -q"
311
+ )
312
+ if not state.allow_writes and turn.action in WRITE_ACTIONS:
313
+ return (
314
+ "This run is read-only. Do not patch, edit, or run. "
315
+ "Action: done Summary: say what you would change and why."
316
+ )
317
+ if turn.action == "ask" and state.wrote_something:
318
+ # A live run wrote a function and a test, then asked which of two
319
+ # readings was meant. The question was reasonable and far too
320
+ # late: the files were already on disk under one of them. Once
321
+ # something is written, the next move is to run or to report.
322
+ if state.ran_tests:
323
+ return (
324
+ "Tests already passed. Action: done Summary: say what you changed."
325
+ )
326
+ return (
327
+ "You have already changed files, so it is too late to ask. "
328
+ "Action: run Argv: -m unittest discover -s tests -q"
329
+ )
330
+ if turn.action == "ask" and state.questions_asked >= MAX_QUESTIONS:
331
+ return (
332
+ "You have already asked. Choose the most likely reading, say "
333
+ "which you chose, and continue."
334
+ )
335
+ return _tool_refusals(state, turn)
336
+
337
+
338
+ def _tool_refusals(state: LoopState, turn) -> str:
339
+ """The ordered refuse_before checks that are not the write/ask caps."""
340
+ path = turn.path or state.last_path
341
+ checks = (
342
+ lambda: refuse_write_tests_ask(state.task, turn.action),
343
+ lambda: refuse_app_overflow_explore(state.task, turn.action),
344
+ lambda: refuse_bugfix_explore(
345
+ state.task, state.project, turn.action, state.located_path
346
+ ),
347
+ lambda: refuse_app_ask(state.task, turn.action),
348
+ lambda: refuse_app_tests_first(state.task, state.project, turn.action, path),
349
+ lambda: refuse_bugfix_tests_first(state.task, state.project, turn.action, path),
350
+ lambda: refuse_app_wrong_path(state.task, turn.action, path),
351
+ lambda: refuse_patch_before_reading(state, turn),
352
+ lambda: refuse_new_path_before_existing(state, turn),
353
+ lambda: refuse_wrong_file(state.task, state.project, turn.action, path),
354
+ lambda: refuse_question_write(state.task, turn.action),
355
+ lambda: refuse_question_ask(state.task, turn.action, state.located_path),
356
+ lambda: refuse_redundant_explore(
357
+ state.task, turn.action, turn.path, state.located_path
358
+ ),
359
+ lambda: refuse_redundant_locate(
360
+ state.task, turn.action, state.prelude_ran, state.project
361
+ ),
362
+ lambda: refuse_god_target(state.task, state.project, turn.action, path),
363
+ lambda: refuse_smell_wrong_file(
364
+ state.task, turn.action, turn.path, state.located_path, _located_body(state)
365
+ ),
366
+ lambda: state.guard.check(turn),
367
+ lambda: _refuse_ship(state.task, turn.action)
368
+ if turn.action in SHIP_ACTIONS
369
+ else "",
370
+ )
371
+ for check in checks:
372
+ blocked = check()
373
+ if blocked:
374
+ return blocked
375
+ return ""
376
+
377
+
378
+ def _located_body(state: LoopState) -> str:
379
+ if not state.located_path:
380
+ return ""
381
+ try:
382
+ return read_py(state.project, state.located_path)
383
+ except (OSError, ValueError):
384
+ return ""
385
+
386
+
387
+ def _refuse_ship(task: str, action: str) -> str:
388
+ if action in {"issue", "pr"} and looks_like_ticket(task):
389
+ return ""
390
+ if not looks_like_ship(task):
391
+ return (
392
+ "Ship actions only when the task is about an issue, PR, commit, "
393
+ "or push."
394
+ )
395
+ if action == "merge" and not looks_like_merge(task):
396
+ return "merge only when the task says merge"
397
+ return ""
398
+
399
+
400
+ def refuse_echoed_summary(summary: str, instructions: tuple[str, ...]) -> str:
401
+ """Reject a closing summary that repeats an instruction the model was given.
402
+
403
+ A small model will sometimes copy a line out of its skill and present it
404
+ as the answer. For example, given the instruction "quote the -> type
405
+ from the def line (example: tuple[str, int])", it replies with that
406
+ exact sentence. A check that only looks for the return type finds "int"
407
+ inside the example and accepts it, so the text is compared against the
408
+ instructions as well.
409
+ """
410
+ said = _squash(summary)
411
+ if len(said) < 12:
412
+ return ""
413
+ for line in instructions:
414
+ want = _squash(line)
415
+ if len(want) < 12:
416
+ continue
417
+ if said in want or want in said:
418
+ return (
419
+ "That repeats an instruction you were given, it does not "
420
+ "answer. Action: done Summary: say it in your own words and "
421
+ "quote the code you read."
422
+ )
423
+ overlap = len(set(said.split()) & set(want.split()))
424
+ if overlap and overlap / max(1, len(set(said.split()))) >= ECHO_RATIO:
425
+ return (
426
+ "That repeats an instruction you were given, it does not "
427
+ "answer. Action: done Summary: quote the code you read."
428
+ )
429
+ return ""
430
+
431
+
432
+ def _squash(text: str) -> str:
433
+ return " ".join(text.lower().split())
434
+
435
+
436
+ def change_task_file(state: LoopState) -> str | None:
437
+ """The file a finish-without-changing would have to point at.
438
+
439
+ None when the task never asked for a change, so finishing without one
440
+ is a good answer and none of this applies.
441
+ """
442
+ task = state.task
443
+ if looks_like_question(task) or looks_like_ship(task):
444
+ return None
445
+ named = named_project_file(task, state.project)
446
+ wants_change = (
447
+ looks_like_add_feature(task)
448
+ or looks_like_fix_smell(task)
449
+ or looks_like_new_package(task)
450
+ or bool(named)
451
+ )
452
+ if not wants_change:
453
+ return None
454
+ return named or state.located_path
455
+
456
+
457
+ def quotes_a_line_from(summary: str, project: Path, rel: str) -> bool:
458
+ """True when the summary copies a line that really is in that file.
459
+
460
+ Copying a line is something only a reader can do. Saying a file is
461
+ already correct is something anyone can do, and a model handed the
462
+ words will hand them back: a run refused once for changing nothing
463
+ replied "The line is already correct." That names no line, and it
464
+ was accepted, so the run reported success having done nothing.
465
+ """
466
+ if not rel:
467
+ return False
468
+ try:
469
+ text = (project / rel).read_text(encoding="utf-8")
470
+ except OSError:
471
+ return False
472
+ flat = " ".join(summary.split())
473
+ if not flat:
474
+ return False
475
+ for line in text.splitlines():
476
+ quoted = " ".join(line.split())
477
+ if len(quoted) >= MIN_QUOTED_CHARS and quoted in flat:
478
+ return True
479
+ return False
480
+
481
+
482
+ # A dotted or underscored name in the task is something the task is
483
+ # about. Plain words are not: "add the field stopped" is prose, while
484
+ # `result.stopped` is a thing that either is in the file or is not.
485
+ _CODE_NAME = re.compile(r"\b[A-Za-z_][A-Za-z0-9_]*(?:\.[A-Za-z_][A-Za-z0-9_]*)+\b|\b[a-z][a-z0-9]*(?:_[a-z0-9]+)+\b")
486
+
487
+
488
+ def missing_from_file(task: str, project: Path, rel: str) -> str:
489
+ """A name the task is about that is nowhere in the file.
490
+
491
+ Quoting a line proves the model read the file. It does not prove it
492
+ read the right one: a run asked to add `result.stopped` cleared that
493
+ bar by quoting `if __name__ == "__main__":`, which is in every
494
+ script ever written. Nothing can be already correct about adding a
495
+ name that is not there, and that much is checkable.
496
+ """
497
+ if not rel:
498
+ return ""
499
+ try:
500
+ text = (project / rel).read_text(encoding="utf-8")
501
+ except OSError:
502
+ return ""
503
+ for name in _CODE_NAME.findall(task):
504
+ if name.endswith(".py") or name in rel:
505
+ continue
506
+ if name not in text:
507
+ return name
508
+ return ""
509
+
510
+
511
+ def _proved_already_correct(state: LoopState, turn) -> bool:
512
+ """The claim that nothing needed changing, backed by a line."""
513
+ rel = change_task_file(state)
514
+ if not rel:
515
+ return False
516
+ if missing_from_file(state.task, state.project, rel):
517
+ return False
518
+ summary = getattr(turn, "summary", "") or ""
519
+ return quotes_a_line_from(summary, state.project, rel)
520
+
521
+
522
+ def refuse_done_without_change(state: LoopState, turn) -> str:
523
+ """Reject a `done` on a change task that changed nothing.
524
+
525
+ An 8B told to fix a named file was seen finishing after four steps
526
+ with no write and a summary describing the project in general. A file
527
+ that is already correct is a real answer, but the model has to show
528
+ the line rather than assert it.
529
+ """
530
+ if state.wrote_something or change_task_file(state) is None:
531
+ return ""
532
+ if state.empty_done_refused and _proved_already_correct(state, turn):
533
+ return ""
534
+ if state.empty_done_refused >= MAX_EMPTY_DONE:
535
+ return ""
536
+ state.empty_done_refused += 1
537
+ rel = change_task_file(state) or ""
538
+ if state.empty_done_refused == 1:
539
+ where = f"Path: {rel}" if rel else "Path: the file you read"
540
+ return (
541
+ f"Nothing was changed. Action: patch {where} with a Find: line "
542
+ "copied whole from the file and a Replace:. If the file is "
543
+ "already correct, Action: done Summary: copy the line that "
544
+ "makes it correct, exactly as it appears."
545
+ )
546
+ if not rel:
547
+ return (
548
+ "Nothing was changed and the summary shows no line for it. "
549
+ "Action: patch the file you read, or Action: done Summary: "
550
+ "quote the line that already does the job."
551
+ )
552
+ return (
553
+ f"That summary copies no line from {rel}. Action: read Path: "
554
+ f"{rel}, then either patch it, or Action: done Summary: paste "
555
+ "the one line that already does the job."
556
+ )
557
+
558
+
559
+ def done_without_proof(state: LoopState, turn) -> str:
560
+ """What to report instead of a `done` that has nothing to show.
561
+
562
+ Returns the honest summary, or "" when the finish is a real one. The
563
+ run ends either way. What changes is whether it claims success, and
564
+ two of nine failures in a 45-run benchmark reported success having
565
+ written nothing, which no stop reason can catch.
566
+ """
567
+ if state.wrote_something or not state.empty_done_refused:
568
+ return ""
569
+ if change_task_file(state) is None or _proved_already_correct(state, turn):
570
+ return ""
571
+ rel = change_task_file(state) or ""
572
+ where = f" {rel} was" if rel else " the file was"
573
+ return (
574
+ f"Asked for a change,{where} left as it was, and the closing "
575
+ "summary points at no line that made the change unnecessary. "
576
+ "Reporting this as unfinished rather than done."
577
+ )
578
+
579
+
580
+ def refuse_done(state: LoopState, turn) -> str:
581
+ """The model says it is finished. Return a refusal, or ""."""
582
+ blocked = refuse_echoed_summary(turn.summary, state.instructions)
583
+ if not blocked:
584
+ blocked = refuse_early_done(state.task, state.last_path, state.located_path)
585
+ if not blocked:
586
+ # Asking for a fuller sentence is worth two turns, not the whole
587
+ # budget. A scripted or stubborn model that cannot produce one
588
+ # otherwise spends every remaining step being told the same thing
589
+ # and the run fails with the answer already in hand.
590
+ thin = refuse_shallow_done(
591
+ state.task, turn.summary, state.located_signature
592
+ )
593
+ if thin and state.thin_done_refused < MAX_THIN_DONE:
594
+ state.thin_done_refused += 1
595
+ blocked = thin
596
+ if not blocked:
597
+ blocked = refuse_unwired_addition(
598
+ state.project, state.last_path, state.task
599
+ )
600
+ if not blocked:
601
+ blocked = refuse_design_dirty(state.task, state.design_report)
602
+ if not blocked:
603
+ blocked = refuse_thin_review(state.task, turn.summary, state.design_report)
604
+ if not blocked:
605
+ blocked = refuse_invented_review(
606
+ state.task, turn.summary, _located_body(state)
607
+ )
608
+ if not blocked:
609
+ blocked = refuse_write_done(
610
+ state.task, state.ran_tests, wrote=state.wrote_something
611
+ )
612
+ if not blocked:
613
+ blocked = refuse_done_without_change(state, turn)
614
+ if not blocked:
615
+ blocked = refuse_done_oracle(state.task, state.project, state.last_path)
616
+ if not blocked:
617
+ blocked = refuse_app_done(state.task, state.project)
618
+ return blocked
619
+
620
+
621
+ RUN_SUITE = "Next Action must be run Argv: -m unittest discover -s tests -q\n"
622
+
623
+
624
+ def write_needs_a_test(state: LoopState, result: str, path: str) -> bool:
625
+ """True when the next step is still to write a test, not to run."""
626
+ if not (state.project / "tests").is_dir():
627
+ return False
628
+ if "AAA test" in result:
629
+ return False
630
+ if "test" in (path or "").replace("\\", "/").lower():
631
+ return False
632
+ symbol = covered_symbol(state.task) or question_symbol(state.task)
633
+ if looks_like_add_feature(state.task) or looks_like_write_tests(state.task):
634
+ return bool(symbol) and not tests_call(state.project, symbol)
635
+ if looks_like_bugfix(state.task) and symbol:
636
+ return not tests_call(state.project, symbol)
637
+ return False
638
+
639
+
640
+ def should_run_suite_after_write(state: LoopState, result: str, path: str) -> bool:
641
+ """Daily write: run the suite when tests already cover the work."""
642
+ if looks_like_design_loop(state.task) or looks_like_fix_smell(state.task):
643
+ return False
644
+ if looks_like_app_loop(state.task) or looks_like_app_overflow(state.task):
645
+ return _app_ready_to_run(state, result)
646
+ if not result.startswith(("patched", "wrote")):
647
+ return False
648
+ if not (state.project / "tests").is_dir():
649
+ return False
650
+ rel = path or state.last_path
651
+ if rel and looks_like_bugfix(state.task):
652
+ leftover = undefined_in_file(state.project / rel)
653
+ if leftover:
654
+ return False
655
+ return not write_needs_a_test(state, result, path)
656
+
657
+
658
+ def _app_ready_to_run(state: LoopState, result: str) -> bool:
659
+ """Once list/show/mocks exist, run the suite so done can fire."""
660
+ if not result.startswith(("patched", "wrote")):
661
+ return False
662
+ if not (state.project / "tests").is_dir():
663
+ return False
664
+ from harness.scan.app_spec import required_gaps
665
+
666
+ return not required_gaps(state.project, state.task)
667
+
668
+
669
+ def _named_bugfix_patch(state: LoopState) -> str:
670
+ """Point a named-file bugfix back at the impl, not at a rewritten test."""
671
+ if not looks_like_bugfix(state.task):
672
+ return ""
673
+ named = named_project_file(state.task, state.project)
674
+ if not named:
675
+ return ""
676
+ return (
677
+ f"Next Action must be patch Path: {named} with a Find: line "
678
+ "copied whole from the file. Do not edit tests.\n"
679
+ )
680
+
681
+
682
+ def repair_after_failed_run(state: LoopState, result: str) -> str:
683
+ """Send the traceback back once. A second failure is for the model to stop on."""
684
+ err = result.strip()
685
+ if len(err) > 1200:
686
+ err = err[-1200:]
687
+ rel = state.last_path or "the file you changed"
688
+ named = ""
689
+ if looks_like_bugfix(state.task):
690
+ named = named_project_file(state.task, state.project)
691
+ if named:
692
+ rel = named
693
+ if state.repairs >= MAX_REPAIRS:
694
+ return (
695
+ "The repair still fails. Action: done Summary: quote the "
696
+ "traceback line you could not fix.\n"
697
+ )
698
+ state.repairs += 1
699
+ return (
700
+ "The script failed when I ran it.\n"
701
+ f"```\n{err}\n```\n"
702
+ f"Action: patch Path: {rel} with a Find: line copied whole from "
703
+ "the file. Then Action: run Argv: -m unittest discover -s tests -q\n"
704
+ )
705
+
706
+
707
+ def next_prompt(state: LoopState, turn, result: str, target=None) -> str:
708
+ """A tool just ran. Name the one right next step, or "" to stay open."""
709
+ path = (turn.path or state.last_path).lower()
710
+ # Tests passing only means the work is finished if there was some work.
711
+ # An agent that runs the suite first, to see the starting state, would
712
+ # otherwise be told to finish before it had changed anything.
713
+ if turn.action in {"issue", "pr"}:
714
+ for line in result.splitlines():
715
+ if line.startswith("Next:"):
716
+ return line.split(":", 1)[1].strip() + "\n"
717
+ if turn.action == "run" and not result.startswith("exit 0"):
718
+ if result.startswith("refusing") or "no tests/ directory" in result:
719
+ return ""
720
+ if state.wrote_something:
721
+ return repair_after_failed_run(state, result)
722
+ return _named_bugfix_patch(state)
723
+ if (
724
+ turn.action == "run"
725
+ and result.startswith("exit 0")
726
+ and state.wrote_something
727
+ and not looks_like_design_loop(state.task)
728
+ ):
729
+ leftover = refuse_done_oracle(state.task, state.project, state.last_path)
730
+ if leftover:
731
+ return leftover + "\n"
732
+ if looks_like_app_overflow(state.task):
733
+ from harness.scan.app_spec import next_overflow_action
734
+
735
+ leftover = next_overflow_action(state.project, state.task)
736
+ if leftover:
737
+ return leftover
738
+ return "Tests passed. Action: done Summary: say what you changed.\n"
739
+ if looks_like_app_loop(state.task):
740
+ from harness.scan.app_spec import next_app_action, overflow_gaps
741
+
742
+ missing = next_app_action(state.project, state.task)
743
+ if missing:
744
+ return missing
745
+ extra = overflow_gaps(state.project, state.task)
746
+ if extra:
747
+ return (
748
+ "Tests passed for list and show. "
749
+ f"A later run can add {extra[0].key}. "
750
+ "Action: done Summary: say list and show work.\n"
751
+ )
752
+ return "Tests passed. Action: done Summary: say what you changed.\n"
753
+ wrote = result.startswith(("patched", "wrote"))
754
+ if wrote:
755
+ rel = turn.path or state.last_path
756
+ leftover = undefined_in_file(state.project / rel) if rel else []
757
+ if leftover and looks_like_bugfix(state.task):
758
+ return (
759
+ f"undefined name {leftover[0]} in {rel}. "
760
+ f"Next Action must be patch Path: {rel} Find: {leftover[0]} "
761
+ "Replace: the name you assigned.\n"
762
+ )
763
+ if looks_like_app_overflow(state.task) and wrote:
764
+ from harness.scan.app_spec import next_overflow_action
765
+
766
+ leftover = next_overflow_action(state.project, state.task)
767
+ return leftover or RUN_SUITE
768
+ if looks_like_app_loop(state.task) and wrote:
769
+ from harness.scan.app_spec import http_test_nudge, next_app_action
770
+
771
+ is_test = "test" in path
772
+ missing = next_app_action(state.project, state.task)
773
+ if missing and "mocked_tests" in missing and not is_test:
774
+ return http_test_nudge(state.task)
775
+ if missing:
776
+ return missing
777
+ return RUN_SUITE
778
+ if looks_like_design_loop(state.task) and wrote:
779
+ from harness.scan.design import design_is_clean, render_design_review
780
+
781
+ state.design_report = render_design_review(state.project, state.scope)
782
+ if not design_is_clean(state.design_report):
783
+ return (
784
+ f"{state.design_report}\n\n"
785
+ "Next Action must be edit Path: pkg/<new_concern>.py "
786
+ "with one function.\n"
787
+ )
788
+ return (
789
+ f"{state.design_report}\n\n"
790
+ "Next Action must be run Argv: -m unittest discover -s tests -q\n"
791
+ )
792
+ if not wrote:
793
+ return _named_bugfix_patch(state)
794
+ is_test = "test" in path
795
+ if looks_like_bugfix(state.task) and is_test:
796
+ leftover = _named_bugfix_patch(state)
797
+ if leftover:
798
+ return leftover
799
+ if (
800
+ looks_like_add_feature(state.task)
801
+ and turn.action in {"patch", "edit"}
802
+ and not is_test
803
+ and "AAA test" in result
804
+ ):
805
+ return RUN_SUITE
806
+ if (
807
+ (looks_like_add_feature(state.task) or looks_like_bugfix(state.task))
808
+ and turn.action in {"patch", "edit"}
809
+ and not is_test
810
+ ):
811
+ if not write_needs_a_test(state, result, path):
812
+ return RUN_SUITE
813
+ loaded = get_skill("write-tests", state.project)
814
+ if loaded is not None:
815
+ return (
816
+ f"{render_skill(loaded, target, state.project)}\n"
817
+ "Next Action must be this write-tests patch. "
818
+ "Do not Append after if __name__.\n"
819
+ )
820
+ if looks_like_new_package(state.task) and turn.action in {"edit", "patch"}:
821
+ from harness.task import mentions_cli, mentions_http, package_noun
822
+
823
+ noun = package_noun(state.task)
824
+ extra = ""
825
+ if mentions_cli(state.task):
826
+ extra += " argparse."
827
+ if mentions_http(state.task):
828
+ extra += " urllib. Token from the environment. No curl."
829
+ if "__init__" in path:
830
+ return (
831
+ f"Next Action must be edit Path: pkg/{noun}.py with one "
832
+ f"function def {noun}(...).{extra} snake_case. Not in __init__.py.\n"
833
+ )
834
+ if not is_test:
835
+ return (
836
+ f"Next Action must be edit Path: tests/test_{noun}.py as a "
837
+ f"unittest.TestCase. Name test_{noun}_<result>. "
838
+ f"AAA: got = {noun}(...); assert got. Then Action: run.\n"
839
+ )
840
+ return RUN_SUITE
841
+ if looks_like_fix_smell(state.task) and turn.action == "patch" and not is_test:
842
+ old, new = smell_symbol(state.task), rename_target(state.task)
843
+ if old and new:
844
+ return (
845
+ f"Next Action: patch tests to replace {old} with {new}, "
846
+ "then Action: run.\n"
847
+ )
848
+ return ""
849
+
850
+
851
+ def unclear(task: str) -> bool:
852
+ """A task with no verb and no symbol cannot be started from."""
853
+ text = task.strip()
854
+ if looks_like_question(text):
855
+ return False
856
+ return len(text.split()) < 3 and not question_symbol(text)