py-harness-cli 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. finetune/__init__.py +1 -0
  2. finetune/agent_system.py +41 -0
  3. finetune/agent_traces.py +157 -0
  4. finetune/everyday.py +30 -0
  5. finetune/hf_ollama.py +158 -0
  6. finetune/huggingface_store.py +144 -0
  7. finetune/models.py +74 -0
  8. finetune/paths.py +11 -0
  9. finetune/python_vibe.py +788 -0
  10. finetune/splits.py +54 -0
  11. finetune/systems.py +9 -0
  12. harness/__init__.py +42 -0
  13. harness/__main__.py +8 -0
  14. harness/act/__init__.py +6 -0
  15. harness/act/autofix/__init__.py +110 -0
  16. harness/act/autofix/additions.py +217 -0
  17. harness/act/autofix/conflicts.py +124 -0
  18. harness/act/autofix/cover.py +419 -0
  19. harness/act/autofix/mechanical.py +151 -0
  20. harness/act/autofix/missing_imports.py +50 -0
  21. harness/act/autofix/moves.py +439 -0
  22. harness/act/autofix/names.py +339 -0
  23. harness/act/autofix/scaffold.py +224 -0
  24. harness/act/code.py +157 -0
  25. harness/act/gate.py +229 -0
  26. harness/act/parse.py +247 -0
  27. harness/act/patch_fix.py +138 -0
  28. harness/act/tools.py +244 -0
  29. harness/agent/__init__.py +11 -0
  30. harness/agent/dispatch.py +235 -0
  31. harness/agent/loop.py +699 -0
  32. harness/agent/options.py +144 -0
  33. harness/agent/policy.py +856 -0
  34. harness/agent/prompt.py +170 -0
  35. harness/cli.py +393 -0
  36. harness/editor_kit.py +265 -0
  37. harness/guard/__init__.py +6 -0
  38. harness/guard/fallbacks.py +6 -0
  39. harness/guard/loop_guard.py +57 -0
  40. harness/guard/python_vibe.py +68 -0
  41. harness/guard/run.py +41 -0
  42. harness/guard/types.py +19 -0
  43. harness/locate.py +767 -0
  44. harness/mcp_stdio.py +306 -0
  45. harness/memory/__init__.py +5 -0
  46. harness/memory/conversation.py +104 -0
  47. harness/model/__init__.py +6 -0
  48. harness/model/chat_backend.py +100 -0
  49. harness/model/engine.py +165 -0
  50. harness/model/ollama_generate.py +60 -0
  51. harness/model/openai_generate.py +156 -0
  52. harness/model/outbound.py +83 -0
  53. harness/model/route.py +90 -0
  54. harness/observe/__init__.py +6 -0
  55. harness/observe/eval_gate.py +80 -0
  56. harness/observe/eval_loop.py +185 -0
  57. harness/observe/eval_tasks.py +399 -0
  58. harness/observe/report_md.py +102 -0
  59. harness/observe/trace_record.py +79 -0
  60. harness/openai_api.py +81 -0
  61. harness/paths.py +88 -0
  62. harness/py.typed +0 -0
  63. harness/scan/__init__.py +6 -0
  64. harness/scan/app_spec.py +338 -0
  65. harness/scan/design.py +112 -0
  66. harness/scan/existing.py +131 -0
  67. harness/scan/layout.py +254 -0
  68. harness/scan/names.py +308 -0
  69. harness/scan/project_brief.py +287 -0
  70. harness/scan/project_docs.py +42 -0
  71. harness/scan/project_scan.py +49 -0
  72. harness/scan/repo_map.py +101 -0
  73. harness/secrets.py +39 -0
  74. harness/server.py +199 -0
  75. harness/ship/__init__.py +1 -0
  76. harness/ship/bot_pr.py +221 -0
  77. harness/ship/git_ship.py +262 -0
  78. harness/ship/identity.py +62 -0
  79. harness/ship/ticket.py +251 -0
  80. harness/skillkit/__init__.py +6 -0
  81. harness/skillkit/catalog.py +241 -0
  82. harness/skillkit/refuse_change.py +640 -0
  83. harness/skillkit/refuse_finish.py +295 -0
  84. harness/skillkit/target.py +238 -0
  85. harness/task.py +717 -0
  86. py_harness_cli-0.3.0.dist-info/METADATA +177 -0
  87. py_harness_cli-0.3.0.dist-info/RECORD +92 -0
  88. py_harness_cli-0.3.0.dist-info/WHEEL +5 -0
  89. py_harness_cli-0.3.0.dist-info/entry_points.txt +3 -0
  90. py_harness_cli-0.3.0.dist-info/licenses/LICENSE +202 -0
  91. py_harness_cli-0.3.0.dist-info/licenses/NOTICE +6 -0
  92. py_harness_cli-0.3.0.dist-info/top_level.txt +2 -0
harness/act/code.py ADDED
@@ -0,0 +1,157 @@
1
+ """Pull a Python block out of a vibe draft and write or run it locally."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import ast
6
+ import re
7
+ import subprocess
8
+ import sys
9
+ from dataclasses import dataclass
10
+ from pathlib import Path
11
+
12
+ from harness.paths import SECRET_NAMES, TEXT_SUFFIXES, as_project_rel
13
+
14
+ _FENCE = re.compile(r"```(?:python|py)?\s*\n(.*?)```", re.DOTALL | re.IGNORECASE)
15
+ _SKIP_PARTS = {".git", ".venv", "node_modules", "adapters", "fused", "__pycache__"}
16
+ MAX_FILE_CHARS = 3500
17
+ # Small files are read whole so nearby constants (env, argv) stay in the quote.
18
+ WHOLE_FILE_CHARS = 12_000
19
+ # How much of the middle to keep when the task points at it.
20
+ MIDDLE_WINDOW_CHARS = 1200
21
+
22
+
23
+ def extract_python(text: str) -> str | None:
24
+ blocks = [m.group(1).strip() for m in _FENCE.finditer(text)]
25
+ if blocks:
26
+ return max(blocks, key=len)
27
+ stripped = text.strip()
28
+ if stripped.startswith(("import ", "from ", "def ", "class ", "#!/")):
29
+ return stripped
30
+ return None
31
+
32
+
33
+ @dataclass(frozen=True)
34
+ class RunResult:
35
+ code: int
36
+ stdout: str
37
+ stderr: str
38
+
39
+
40
+ def write_and_run(
41
+ source: str,
42
+ dest: Path,
43
+ argv: list[str] | None = None,
44
+ *,
45
+ cwd: Path | None = None,
46
+ timeout: float = 12,
47
+ stdin: str | None = None,
48
+ ) -> RunResult:
49
+ dest.parent.mkdir(parents=True, exist_ok=True)
50
+ dest.write_text(source.rstrip() + "\n", encoding="utf-8")
51
+ proc = subprocess.run(
52
+ [sys.executable, str(dest), *(argv or [])],
53
+ cwd=cwd or dest.parent,
54
+ capture_output=True,
55
+ text=True,
56
+ timeout=timeout,
57
+ check=False,
58
+ input=stdin,
59
+ )
60
+ return RunResult(proc.returncode, proc.stdout, proc.stderr)
61
+
62
+
63
+ def resolve_project_file(project: Path, rel: str) -> Path:
64
+ root = project.resolve()
65
+ rel = as_project_rel(rel)
66
+ path = (root / rel).resolve() if not Path(rel).is_absolute() else Path(rel).resolve()
67
+ try:
68
+ path.relative_to(root)
69
+ except ValueError as exc:
70
+ raise ValueError(f"{path} is outside {root}") from exc
71
+ if any(part in _SKIP_PARTS for part in path.parts):
72
+ raise ValueError(f"refusing {path}")
73
+ if path.name.lower() in {item.lower() for item in SECRET_NAMES}:
74
+ raise ValueError(f"refusing secret filename {path.name}")
75
+ if path.suffix.lower() not in TEXT_SUFFIXES:
76
+ raise ValueError(
77
+ "only project text files: " + ", ".join(sorted(TEXT_SUFFIXES))
78
+ )
79
+ return path
80
+
81
+
82
+ def _window_around(text: str, about: str, width: int) -> tuple[int, int] | None:
83
+ """Character range covering the first mention of `about`, or None."""
84
+ if not about:
85
+ return None
86
+ at = text.find(about)
87
+ if at < 0:
88
+ return None
89
+ half = width // 2
90
+ start = max(0, at - half)
91
+ return start, min(len(text), start + width)
92
+
93
+
94
+ def read_project_file(
95
+ path: Path, *, limit: int = MAX_FILE_CHARS, about: str = ""
96
+ ) -> str:
97
+ """The file, or as much of it as fits, keeping the part that matters.
98
+
99
+ A file too long to send whole used to be sent as its head and its
100
+ tail, with the middle dropped. Asked to add a field to a dict two
101
+ thirds of the way down a 13,476-character file, the model was handed
102
+ 3,500 characters from the top and 800 from the bottom — and the dict
103
+ was in neither. It then invented a `Find:` line that was not in the
104
+ file, was refused, and sent it again.
105
+
106
+ `about` names what the task is for. When the text contains it, a
107
+ window around it is kept as well, so the part being changed is one
108
+ of the parts that arrives.
109
+ """
110
+ text = path.read_text(encoding="utf-8")
111
+ cap = WHOLE_FILE_CHARS if limit == MAX_FILE_CHARS else limit
112
+ if len(text) <= cap:
113
+ return text
114
+ tail = min(800, max(0, len(text) - limit))
115
+ omitted = len(text) - limit - tail
116
+ if omitted <= 0:
117
+ return text
118
+ window = _window_around(text, about, MIDDLE_WINDOW_CHARS)
119
+ if window is None or window[0] < limit:
120
+ # Either nothing to centre on, or it is inside the head already.
121
+ return (
122
+ text[:limit]
123
+ + f"\n# … truncated {omitted} chars …\n"
124
+ + text[-tail:]
125
+ )
126
+ start, end = window
127
+ before = start - limit
128
+ after = max(0, len(text) - tail - end)
129
+ return (
130
+ text[:limit]
131
+ + f"\n# … truncated {before} chars …\n"
132
+ + text[start:end]
133
+ + f"\n# … truncated {after} chars …\n"
134
+ + text[-tail:]
135
+ )
136
+
137
+ def apply_source(path: Path, source: str, *, original: str) -> None:
138
+ if not source.strip():
139
+ raise ValueError("empty draft")
140
+ if original and len(source) < max(40, (len(original) * 2) // 3):
141
+ raise ValueError(
142
+ f"draft is too short ({len(source)} chars vs {len(original)}) — "
143
+ "use Action: patch for a small change"
144
+ )
145
+ if path.suffix in {".py", ".pyi"}:
146
+ try:
147
+ ast.parse(source)
148
+ except SyntaxError as exc:
149
+ raise ValueError(
150
+ f"syntax error: {exc} — file not written. "
151
+ "Use a full unique line for Find: (not a prefix of def …)"
152
+ ) from exc
153
+ bak = path.with_suffix(path.suffix + ".bak")
154
+ if path.is_file():
155
+ bak.write_text(path.read_text(encoding="utf-8"), encoding="utf-8")
156
+ path.parent.mkdir(parents=True, exist_ok=True)
157
+ path.write_text(source.rstrip() + "\n", encoding="utf-8")
harness/act/gate.py ADDED
@@ -0,0 +1,229 @@
1
+ """May this draft be written? One question, asked before any file changes.
2
+
3
+ The tools carry a change out. This decides whether it may happen: a
4
+ module that already exists under another name, an import with nothing
5
+ behind it, a style rule the project keeps, a definition already in the
6
+ file. Keeping it apart from `tools` means the answer to "where are the
7
+ tools" is one file, and so is the answer to "what stops a bad write".
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from collections.abc import Callable
13
+ from dataclasses import dataclass
14
+
15
+ import re
16
+ from pathlib import Path
17
+
18
+ from harness.paths import rel_posix
19
+ from harness.skillkit.refuse_change import (
20
+ refuse_add_opens_file,
21
+ refuse_layout,
22
+ refuse_opaque_module,
23
+ refuse_opaque_names,
24
+ refuse_ops_draft,
25
+ refuse_platform_draft,
26
+ refuse_rename_incomplete,
27
+ refuse_shell_fetch,
28
+ refuse_stdlib_shadow,
29
+ refuse_stub_body,
30
+ refuse_test_in_impl,
31
+ refuse_undefined_draft,
32
+ refuse_weak_test,
33
+ )
34
+
35
+
36
+
37
+ _TEST_METH = re.compile(r"def\s+(test_\w+)\s*\(")
38
+ _ASSERT_CALL = re.compile(r"assertEqual\s*\(\s*([A-Za-z_]\w+)\s*\(")
39
+ _IMPORT_LINE = re.compile(r"^(from\s+\S+\s+import\s+)(.+)$")
40
+
41
+
42
+ def _add_import_symbol(text: str, name: str) -> str:
43
+ if not name or name in {"self", "True", "False", "None"}:
44
+ return text
45
+ for line in text.splitlines():
46
+ match = _IMPORT_LINE.match(line)
47
+ if not match:
48
+ continue
49
+ imported = {part.strip() for part in match.group(2).split(",")}
50
+ if name in imported:
51
+ return text
52
+ if any(skip in line for skip in ("unittest", "pathlib", "typing")):
53
+ continue
54
+ return text.replace(line, f"{match.group(1)}{match.group(2).rstrip()}, {name}", 1)
55
+ return text
56
+
57
+
58
+ def _called_name(original: str, append: str) -> str:
59
+ """The function the new test needs imported.
60
+
61
+ It used to be read out of `assertEqual(multiply(...))`. The change rules
62
+ ask for the opposite shape — `got = multiply(...)`, then assert `got` —
63
+ so following them meant the import was never added and the suite broke.
64
+ Reading the names the new test leaves unbound covers both shapes.
65
+ """
66
+ from harness.scan.names import new_undefined
67
+
68
+ for name in new_undefined(original, original.rstrip() + "\n\n" + append):
69
+ return name
70
+ match = _ASSERT_CALL.search(append)
71
+ return match.group(1) if match else ""
72
+
73
+
74
+ def repair_unittest_append(original: str, append: str) -> str | None:
75
+ """8B Append: often lands after if __name__ and skips the import."""
76
+ if "def test_" not in append:
77
+ return None
78
+ if "TestCase" not in original and "unittest" not in original:
79
+ return None
80
+ meth = _TEST_METH.search(append)
81
+ if not meth or re.search(rf"def\s+{re.escape(meth.group(1))}\s*\(", original):
82
+ return None
83
+ lines = append.strip("\n").splitlines()
84
+ while lines and not lines[0].strip():
85
+ lines.pop(0)
86
+ if not lines:
87
+ return None
88
+ base = len(lines[0]) - len(lines[0].lstrip())
89
+ dedented = "\n".join(
90
+ line[base:] if len(line) >= base else line.lstrip() for line in lines
91
+ )
92
+ method = " " + dedented.replace("\n", "\n ")
93
+ text = _add_import_symbol(original, _called_name(original, append))
94
+ marker = "\nif __name__"
95
+ if marker in text:
96
+ return text.replace(marker, "\n" + method + "\n" + marker, 1)
97
+ return text.rstrip() + "\n\n" + method + "\n"
98
+
99
+
100
+ def refuse_duplicate_module(project: Path, rel: str, original: str) -> str:
101
+ """Refuse a new file that repeats a module this project already has.
102
+
103
+ Watched a run write the same function to `pkg/orders.py` and then
104
+ `src/orders.py`. Two modules with one name is worse than either: an
105
+ import finds whichever comes first on the path, and the other rots.
106
+ """
107
+ if original.strip():
108
+ return ""
109
+ from harness.paths import as_project_rel, rel_posix
110
+
111
+ wanted = Path(as_project_rel(rel))
112
+ if wanted.name.startswith("test_") or "tests" in wanted.parts:
113
+ return ""
114
+ root = Path(project).resolve()
115
+ for existing in sorted(root.rglob(f"{wanted.stem}.py")):
116
+ if any(part in {".git", ".venv", "__pycache__"} for part in existing.parts):
117
+ continue
118
+ found = rel_posix(existing, root)
119
+ if found != wanted.as_posix():
120
+ return (
121
+ f"{found} is already this project's {wanted.stem} module. "
122
+ f"Action: patch Path: {found} Append: the new function"
123
+ )
124
+ return ""
125
+
126
+
127
+ def refuse_missing_import_target(project: Path, rel: str, draft: str) -> str:
128
+ """Refuse a file importing a name this project has not defined yet.
129
+
130
+ Asked to create a module and a test for it, the model wrote only the
131
+ test, importing a function nobody had written. That reads as valid
132
+ Python — the import binds the name — and fails when the suite runs. The
133
+ function has to exist first.
134
+ """
135
+ from harness.scan.names import missing_import_targets
136
+
137
+ missing = missing_import_targets(project, draft)
138
+ if not missing:
139
+ return ""
140
+ module, name = missing[0]
141
+ return (
142
+ f"{module} does not define {name} yet. Write the function first: "
143
+ f"Action: patch Path: {module.replace('.', '/')}.py Append: def {name}(...)"
144
+ )
145
+
146
+
147
+ @dataclass(frozen=True)
148
+ class ProposedChange:
149
+ """One change the model has proposed, and what it is judged on.
150
+
151
+ Fields:
152
+ task: what the user asked for, in their own words.
153
+ rel: the file the change targets, project-relative.
154
+ original: the file as it stands, or "" when it is new.
155
+ draft: the file as this change would leave it.
156
+ fragment: only the part being added, when the change appends.
157
+ Judging a whole file as one test turns a single inline
158
+ assertion anywhere into a refusal for everything in it.
159
+ """
160
+
161
+ task: str
162
+ rel: str
163
+ original: str
164
+ draft: str
165
+ fragment: str = ""
166
+
167
+
168
+ # Every rule a proposed change is put through, in the order they run.
169
+ # This was thirty lines of `blocked = rule(...); if blocked: return
170
+ # blocked`, eleven times over, which hid both the order and the fact that
171
+ # a new rule has to be added to it. A rule written and not listed here
172
+ # does nothing, and a test below checks for exactly that.
173
+ CHANGE_RULES: tuple[tuple[str, Callable[[ProposedChange], str]], ...] = (
174
+ ("stdlib shadow", lambda c: refuse_stdlib_shadow(c.rel, c.original)),
175
+ ("layout", lambda c: refuse_layout(c.rel, c.original, c.draft)),
176
+ ("opens a file", lambda c: refuse_add_opens_file(c.task, c.rel, c.draft)),
177
+ ("shell fetch", lambda c: refuse_shell_fetch(c.rel, c.draft)),
178
+ ("platform draft", lambda c: refuse_platform_draft(c.rel, c.draft)),
179
+ ("operations draft", lambda c: refuse_ops_draft(c.rel, c.draft)),
180
+ ("test in implementation", lambda c: refuse_test_in_impl(c.rel, c.draft)),
181
+ ("stub body", lambda c: refuse_stub_body(c.task, c.rel, c.draft)),
182
+ (
183
+ "undefined name",
184
+ lambda c: refuse_undefined_draft(c.task, c.rel, c.original, c.draft),
185
+ ),
186
+ (
187
+ "half a rename",
188
+ lambda c: refuse_rename_incomplete(c.task, c.rel, c.draft),
189
+ ),
190
+ ("weak test", lambda c: refuse_weak_test(c.rel, c.fragment or c.draft)),
191
+ ("opaque names", lambda c: refuse_opaque_names(c.draft, c.task)),
192
+ ("opaque module", lambda c: refuse_opaque_module(c.rel, c.original)),
193
+ )
194
+
195
+
196
+ def first_refusal(
197
+ task: str, rel: str, original: str, draft: str, fragment: str = ""
198
+ ) -> str:
199
+ """The first refusal a proposed change earns, or "" if it earns none."""
200
+ change = ProposedChange(task, rel, original, draft, fragment)
201
+ for _name, rule in CHANGE_RULES:
202
+ refusal = rule(change)
203
+ if refusal:
204
+ return refusal
205
+ return ""
206
+
207
+
208
+ def already_defined(original: str, append: str, rel: str) -> str:
209
+ """Refuse appending a definition the file already has, or "".
210
+
211
+ Only test methods were checked, so an ordinary function could be
212
+ added twice: a live run appended `def slugify` to the same file on
213
+ two separate turns and left both in place.
214
+ """
215
+ meth = _TEST_METH.search(append)
216
+ if meth and re.search(rf"def\s+{re.escape(meth.group(1))}\s*\(", original):
217
+ return (
218
+ f"{meth.group(1)} already exists. Action: done Summary: "
219
+ "that function is already covered."
220
+ )
221
+ for name in re.findall(r"(?m)^def\s+(\w+)\s*\(", append):
222
+ if re.search(rf"(?m)^def\s+{re.escape(name)}\s*\(", original):
223
+ return (
224
+ f"{name} is already defined in {rel}. Action: done "
225
+ "Summary: say what it does, or patch the existing one."
226
+ )
227
+ return ""
228
+
229
+
harness/act/parse.py ADDED
@@ -0,0 +1,247 @@
1
+ """Parse one agent turn. Deterministic. No model."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from dataclasses import dataclass
7
+
8
+ from harness.act.code import extract_python
9
+
10
+ # A chat model decorates the line the harness is trying to read. It
11
+ # emboldens the label (`**Action:** patch`), numbers its steps
12
+ # (`1. Action: patch`), or explains the verb in the same breath
13
+ # (`Action: patch to add slugify`). Local weights write the bare line,
14
+ # so an anchored pattern was enough until the harness was pointed at a
15
+ # model it does not run itself — and then the turn parsed to nothing.
16
+ #
17
+ # Only `*`, `#`, `>` and list numbering are allowed as decoration, and
18
+ # never `_`, because a value may legitimately start with one and
19
+ # `Name: _helpers` must keep its underscore.
20
+ _DECOR = r"[ \t>*#-]{0,6}(?:\d{1,2}[.)][ \t]*)?[ \t>*]{0,4}"
21
+ _AFTER_KEY = r"[* \t]*"
22
+
23
+ # Hyphens are allowed so a skill name works as an action: every kit
24
+ # skill is hyphenated, and `\w+` could not match one. There is no `$`:
25
+ # anything after the verb is the model thinking aloud, and an unknown
26
+ # verb is still rejected by KNOWN_ACTIONS below.
27
+ _ACTION = re.compile(
28
+ rf"^{_DECOR}Action:{_AFTER_KEY}([\w-]+)", re.MULTILINE | re.IGNORECASE
29
+ )
30
+ # Every verb the loop can carry out. A model that writes `Action: find`
31
+ # has put a field name on the Action line; that block is skipped so the
32
+ # turn is not spent on an unknown verb.
33
+ KNOWN_ACTIONS = frozenset(
34
+ {
35
+ "glob", "grep", "read", "edit", "patch", "run", "map", "plan",
36
+ "skill", "locate", "layout", "ask", "done",
37
+ "issue", "branch", "commit", "push", "pr", "merge",
38
+ }
39
+ )
40
+ _FIELD = re.compile(
41
+ rf"^{_DECOR}(Path|File|Query|Pattern|Argv|Summary|Scope|Name|Number|Title):"
42
+ rf"{_AFTER_KEY}(.+?)[ \t*]*$",
43
+ re.MULTILINE,
44
+ )
45
+ _STOP = re.compile(
46
+ rf"^{_DECOR}(Action|Path|File|Query|Pattern|Argv|Summary|Scope|Name|Number"
47
+ rf"|Title|Body|Find|Replace|Append|Add):{_AFTER_KEY}",
48
+ re.IGNORECASE,
49
+ )
50
+
51
+
52
+ # A model that answers in chat wraps code in a fence. Local weights
53
+ # happen not to; every hosted one does, so this only bites the moment
54
+ # the harness is pointed at a model it does not run itself.
55
+ # A fence marker alone on the last line, with or without a language:
56
+ # a whole-reply fence closing, or a second block the model opened
57
+ # and never filled. Either way it is not Python.
58
+ _CLOSING = re.compile(r"^[ \t]*```[\w+-]*[ \t]*$")
59
+ _FENCED = re.compile(
60
+ r"^[^\S\n]*```[\w+-]*[^\S\n]*\n(.*?)\n[^\S\n]*```",
61
+ re.DOTALL,
62
+ )
63
+
64
+
65
+ def unfenced(body: str) -> str:
66
+ """The code inside a markdown fence, or the text unchanged.
67
+
68
+ An `Append:` body arriving as ```python … ``` used to reach the file
69
+ with the backticks still on it. What landed was a SyntaxError, and
70
+ by its third turn a hosted 32B was reporting an unterminated string
71
+ literal in a file it had broken itself. Nine of ten runs then spent
72
+ the whole budget writing nothing that would load.
73
+
74
+ Anything after the closing fence goes too. A model that signs off
75
+ with "That should do it." puts that sentence inside the fence's
76
+ file otherwise, which fails exactly the same way the backticks do.
77
+ """
78
+ found = _FENCED.match(body)
79
+ if found:
80
+ return found.group(1)
81
+ # The model fenced its whole reply rather than the code, so the
82
+ # block never opened with a fence and the closing one is left
83
+ # trailing on the end of it. That reaches the file and is the same
84
+ # SyntaxError, arriving by a different route.
85
+ lines = body.splitlines()
86
+ if lines and _CLOSING.match(lines[-1]):
87
+ return "\n".join(lines[:-1]).rstrip()
88
+ return body
89
+
90
+
91
+ def _block(text: str, key: str) -> str:
92
+ match = re.search(
93
+ rf"^{_DECOR}{key}:{_AFTER_KEY}(.*?)[ \t*]*$",
94
+ text,
95
+ re.MULTILINE | re.IGNORECASE,
96
+ )
97
+ if not match:
98
+ return ""
99
+ lines: list[str] = []
100
+ first = match.group(1).rstrip()
101
+ if first:
102
+ lines.append(first)
103
+ for line in text[match.end() :].lstrip("\n").splitlines():
104
+ if _STOP.match(line):
105
+ break
106
+ lines.append(line.rstrip())
107
+ return "\n".join(lines).rstrip()
108
+
109
+
110
+ @dataclass(frozen=True)
111
+ class AgentTurn:
112
+ action: str
113
+ path: str = ""
114
+ query: str = ""
115
+ pattern: str = ""
116
+ argv: tuple[str, ...] = ()
117
+ summary: str = ""
118
+ source: str | None = None
119
+ find: str = ""
120
+ replace: str = ""
121
+ scope: str = ""
122
+ name: str = ""
123
+ append: str = ""
124
+ number: str = ""
125
+ title: str = ""
126
+ body: str = ""
127
+
128
+
129
+ _PREFERRED_WRITE = ("patch", "edit", "run", "locate", "done")
130
+ _PREFERRED_QUESTION = ("done", "locate", "grep", "read")
131
+ _PREFERRED_SHIP = (
132
+ "issue",
133
+ "branch",
134
+ "commit",
135
+ "push",
136
+ "pr",
137
+ "merge",
138
+ "patch",
139
+ "done",
140
+ )
141
+
142
+
143
+ # A field name written on the Action line. The model means the action that
144
+ # field belongs to, and every such turn was spent on "unknown Action".
145
+ _FIELD_AS_ACTION = {
146
+ "append": "patch",
147
+ "add": "patch",
148
+ "find": "patch",
149
+ "replace": "patch",
150
+ "summary": "done",
151
+ }
152
+
153
+
154
+ def _body_after_action(text: str) -> str:
155
+ """The code below the Action line, with the field lines left out.
156
+
157
+ `Action: append` is followed by `Path:` and then the function. Stopping
158
+ at the first field line returned nothing, so the turn was spent for no
159
+ reason.
160
+ """
161
+ match = _ACTION.search(text)
162
+ if not match:
163
+ return ""
164
+ lines = [
165
+ line.rstrip()
166
+ for line in text[match.end():].splitlines()
167
+ if not _STOP.match(line)
168
+ ]
169
+ while lines and not lines[0].strip():
170
+ lines.pop(0)
171
+ return "\n".join(lines).strip("\n")
172
+
173
+
174
+ def parse_turn(text: str) -> AgentTurn | None:
175
+ match = _ACTION.search(text)
176
+ if not match:
177
+ return None
178
+ action = match.group(1).lower()
179
+ body_as_append = ""
180
+ if action in _FIELD_AS_ACTION and action not in KNOWN_ACTIONS:
181
+ if action in {"append", "add"}:
182
+ body_as_append = _body_after_action(text)
183
+ action = _FIELD_AS_ACTION[action]
184
+ fields = {m.group(1).lower(): m.group(2).strip() for m in _FIELD.finditer(text)}
185
+ argv = tuple(part for part in fields.get("argv", "").split() if part)
186
+ source = extract_python(text) if action == "edit" else None
187
+ if action == "edit" and not source:
188
+ extra = _block(text, "Append") or _block(text, "Add")
189
+ if extra:
190
+ source = extra
191
+ return AgentTurn(
192
+ action=action,
193
+ path=fields.get("path") or fields.get("file", ""),
194
+ query=fields.get("query", ""),
195
+ pattern=fields.get("pattern", ""),
196
+ argv=argv,
197
+ summary=fields.get("summary", ""),
198
+ source=source,
199
+ find=unfenced(_block(text, "Find")),
200
+ replace=unfenced(_block(text, "Replace")),
201
+ scope=fields.get("scope", ""),
202
+ name=fields.get("name", ""),
203
+ append=unfenced(
204
+ _block(text, "Append") or _block(text, "Add") or body_as_append
205
+ ),
206
+ number=fields.get("number", ""),
207
+ title=fields.get("title", ""),
208
+ body=_block(text, "Body"),
209
+ )
210
+
211
+
212
+ def _is_skill_name(verb: str) -> bool:
213
+ """`Action: write-tests` names a skill, which the loop loads."""
214
+ return "-" in verb or "_" in verb
215
+
216
+
217
+ def parse_turn_smart(
218
+ text: str, *, question: bool = False, ship: bool = False
219
+ ) -> AgentTurn | None:
220
+ """Small models paste the Action menu. Pick one block by task kind."""
221
+ matches = [
222
+ match
223
+ for match in _ACTION.finditer(text)
224
+ if match.group(1).lower() in KNOWN_ACTIONS or _is_skill_name(match.group(1))
225
+ ]
226
+ if not matches:
227
+ return parse_turn(text)
228
+ if len(matches) == 1:
229
+ return parse_turn(text[matches[0].start() :])
230
+ if question:
231
+ prefer = _PREFERRED_QUESTION
232
+ elif ship:
233
+ prefer = _PREFERRED_SHIP
234
+ else:
235
+ prefer = _PREFERRED_WRITE
236
+ chosen = matches[0]
237
+ for match in matches:
238
+ if match.group(1).lower() in prefer:
239
+ chosen = match
240
+ break
241
+ start = chosen.start()
242
+ end = len(text)
243
+ for match in matches:
244
+ if match.start() > start:
245
+ end = match.start()
246
+ break
247
+ return parse_turn(text[start:end])