py-harness-cli 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- finetune/__init__.py +1 -0
- finetune/agent_system.py +41 -0
- finetune/agent_traces.py +157 -0
- finetune/everyday.py +30 -0
- finetune/hf_ollama.py +158 -0
- finetune/huggingface_store.py +144 -0
- finetune/models.py +74 -0
- finetune/paths.py +11 -0
- finetune/python_vibe.py +788 -0
- finetune/splits.py +54 -0
- finetune/systems.py +9 -0
- harness/__init__.py +42 -0
- harness/__main__.py +8 -0
- harness/act/__init__.py +6 -0
- harness/act/autofix/__init__.py +110 -0
- harness/act/autofix/additions.py +217 -0
- harness/act/autofix/conflicts.py +124 -0
- harness/act/autofix/cover.py +419 -0
- harness/act/autofix/mechanical.py +151 -0
- harness/act/autofix/missing_imports.py +50 -0
- harness/act/autofix/moves.py +439 -0
- harness/act/autofix/names.py +339 -0
- harness/act/autofix/scaffold.py +224 -0
- harness/act/code.py +157 -0
- harness/act/gate.py +229 -0
- harness/act/parse.py +247 -0
- harness/act/patch_fix.py +138 -0
- harness/act/tools.py +244 -0
- harness/agent/__init__.py +11 -0
- harness/agent/dispatch.py +235 -0
- harness/agent/loop.py +699 -0
- harness/agent/options.py +144 -0
- harness/agent/policy.py +856 -0
- harness/agent/prompt.py +170 -0
- harness/cli.py +393 -0
- harness/editor_kit.py +265 -0
- harness/guard/__init__.py +6 -0
- harness/guard/fallbacks.py +6 -0
- harness/guard/loop_guard.py +57 -0
- harness/guard/python_vibe.py +68 -0
- harness/guard/run.py +41 -0
- harness/guard/types.py +19 -0
- harness/locate.py +767 -0
- harness/mcp_stdio.py +306 -0
- harness/memory/__init__.py +5 -0
- harness/memory/conversation.py +104 -0
- harness/model/__init__.py +6 -0
- harness/model/chat_backend.py +100 -0
- harness/model/engine.py +165 -0
- harness/model/ollama_generate.py +60 -0
- harness/model/openai_generate.py +156 -0
- harness/model/outbound.py +83 -0
- harness/model/route.py +90 -0
- harness/observe/__init__.py +6 -0
- harness/observe/eval_gate.py +80 -0
- harness/observe/eval_loop.py +185 -0
- harness/observe/eval_tasks.py +399 -0
- harness/observe/report_md.py +102 -0
- harness/observe/trace_record.py +79 -0
- harness/openai_api.py +81 -0
- harness/paths.py +88 -0
- harness/py.typed +0 -0
- harness/scan/__init__.py +6 -0
- harness/scan/app_spec.py +338 -0
- harness/scan/design.py +112 -0
- harness/scan/existing.py +131 -0
- harness/scan/layout.py +254 -0
- harness/scan/names.py +308 -0
- harness/scan/project_brief.py +287 -0
- harness/scan/project_docs.py +42 -0
- harness/scan/project_scan.py +49 -0
- harness/scan/repo_map.py +101 -0
- harness/secrets.py +39 -0
- harness/server.py +199 -0
- harness/ship/__init__.py +1 -0
- harness/ship/bot_pr.py +221 -0
- harness/ship/git_ship.py +262 -0
- harness/ship/identity.py +62 -0
- harness/ship/ticket.py +251 -0
- harness/skillkit/__init__.py +6 -0
- harness/skillkit/catalog.py +241 -0
- harness/skillkit/refuse_change.py +640 -0
- harness/skillkit/refuse_finish.py +295 -0
- harness/skillkit/target.py +238 -0
- harness/task.py +717 -0
- py_harness_cli-0.3.0.dist-info/METADATA +177 -0
- py_harness_cli-0.3.0.dist-info/RECORD +92 -0
- py_harness_cli-0.3.0.dist-info/WHEEL +5 -0
- py_harness_cli-0.3.0.dist-info/entry_points.txt +3 -0
- py_harness_cli-0.3.0.dist-info/licenses/LICENSE +202 -0
- py_harness_cli-0.3.0.dist-info/licenses/NOTICE +6 -0
- py_harness_cli-0.3.0.dist-info/top_level.txt +2 -0
harness/act/code.py
ADDED
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
"""Pull a Python block out of a vibe draft and write or run it locally."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import ast
|
|
6
|
+
import re
|
|
7
|
+
import subprocess
|
|
8
|
+
import sys
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
from harness.paths import SECRET_NAMES, TEXT_SUFFIXES, as_project_rel
|
|
13
|
+
|
|
14
|
+
_FENCE = re.compile(r"```(?:python|py)?\s*\n(.*?)```", re.DOTALL | re.IGNORECASE)
|
|
15
|
+
_SKIP_PARTS = {".git", ".venv", "node_modules", "adapters", "fused", "__pycache__"}
|
|
16
|
+
MAX_FILE_CHARS = 3500
|
|
17
|
+
# Small files are read whole so nearby constants (env, argv) stay in the quote.
|
|
18
|
+
WHOLE_FILE_CHARS = 12_000
|
|
19
|
+
# How much of the middle to keep when the task points at it.
|
|
20
|
+
MIDDLE_WINDOW_CHARS = 1200
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def extract_python(text: str) -> str | None:
|
|
24
|
+
blocks = [m.group(1).strip() for m in _FENCE.finditer(text)]
|
|
25
|
+
if blocks:
|
|
26
|
+
return max(blocks, key=len)
|
|
27
|
+
stripped = text.strip()
|
|
28
|
+
if stripped.startswith(("import ", "from ", "def ", "class ", "#!/")):
|
|
29
|
+
return stripped
|
|
30
|
+
return None
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass(frozen=True)
|
|
34
|
+
class RunResult:
|
|
35
|
+
code: int
|
|
36
|
+
stdout: str
|
|
37
|
+
stderr: str
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def write_and_run(
|
|
41
|
+
source: str,
|
|
42
|
+
dest: Path,
|
|
43
|
+
argv: list[str] | None = None,
|
|
44
|
+
*,
|
|
45
|
+
cwd: Path | None = None,
|
|
46
|
+
timeout: float = 12,
|
|
47
|
+
stdin: str | None = None,
|
|
48
|
+
) -> RunResult:
|
|
49
|
+
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
50
|
+
dest.write_text(source.rstrip() + "\n", encoding="utf-8")
|
|
51
|
+
proc = subprocess.run(
|
|
52
|
+
[sys.executable, str(dest), *(argv or [])],
|
|
53
|
+
cwd=cwd or dest.parent,
|
|
54
|
+
capture_output=True,
|
|
55
|
+
text=True,
|
|
56
|
+
timeout=timeout,
|
|
57
|
+
check=False,
|
|
58
|
+
input=stdin,
|
|
59
|
+
)
|
|
60
|
+
return RunResult(proc.returncode, proc.stdout, proc.stderr)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def resolve_project_file(project: Path, rel: str) -> Path:
|
|
64
|
+
root = project.resolve()
|
|
65
|
+
rel = as_project_rel(rel)
|
|
66
|
+
path = (root / rel).resolve() if not Path(rel).is_absolute() else Path(rel).resolve()
|
|
67
|
+
try:
|
|
68
|
+
path.relative_to(root)
|
|
69
|
+
except ValueError as exc:
|
|
70
|
+
raise ValueError(f"{path} is outside {root}") from exc
|
|
71
|
+
if any(part in _SKIP_PARTS for part in path.parts):
|
|
72
|
+
raise ValueError(f"refusing {path}")
|
|
73
|
+
if path.name.lower() in {item.lower() for item in SECRET_NAMES}:
|
|
74
|
+
raise ValueError(f"refusing secret filename {path.name}")
|
|
75
|
+
if path.suffix.lower() not in TEXT_SUFFIXES:
|
|
76
|
+
raise ValueError(
|
|
77
|
+
"only project text files: " + ", ".join(sorted(TEXT_SUFFIXES))
|
|
78
|
+
)
|
|
79
|
+
return path
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _window_around(text: str, about: str, width: int) -> tuple[int, int] | None:
|
|
83
|
+
"""Character range covering the first mention of `about`, or None."""
|
|
84
|
+
if not about:
|
|
85
|
+
return None
|
|
86
|
+
at = text.find(about)
|
|
87
|
+
if at < 0:
|
|
88
|
+
return None
|
|
89
|
+
half = width // 2
|
|
90
|
+
start = max(0, at - half)
|
|
91
|
+
return start, min(len(text), start + width)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def read_project_file(
|
|
95
|
+
path: Path, *, limit: int = MAX_FILE_CHARS, about: str = ""
|
|
96
|
+
) -> str:
|
|
97
|
+
"""The file, or as much of it as fits, keeping the part that matters.
|
|
98
|
+
|
|
99
|
+
A file too long to send whole used to be sent as its head and its
|
|
100
|
+
tail, with the middle dropped. Asked to add a field to a dict two
|
|
101
|
+
thirds of the way down a 13,476-character file, the model was handed
|
|
102
|
+
3,500 characters from the top and 800 from the bottom — and the dict
|
|
103
|
+
was in neither. It then invented a `Find:` line that was not in the
|
|
104
|
+
file, was refused, and sent it again.
|
|
105
|
+
|
|
106
|
+
`about` names what the task is for. When the text contains it, a
|
|
107
|
+
window around it is kept as well, so the part being changed is one
|
|
108
|
+
of the parts that arrives.
|
|
109
|
+
"""
|
|
110
|
+
text = path.read_text(encoding="utf-8")
|
|
111
|
+
cap = WHOLE_FILE_CHARS if limit == MAX_FILE_CHARS else limit
|
|
112
|
+
if len(text) <= cap:
|
|
113
|
+
return text
|
|
114
|
+
tail = min(800, max(0, len(text) - limit))
|
|
115
|
+
omitted = len(text) - limit - tail
|
|
116
|
+
if omitted <= 0:
|
|
117
|
+
return text
|
|
118
|
+
window = _window_around(text, about, MIDDLE_WINDOW_CHARS)
|
|
119
|
+
if window is None or window[0] < limit:
|
|
120
|
+
# Either nothing to centre on, or it is inside the head already.
|
|
121
|
+
return (
|
|
122
|
+
text[:limit]
|
|
123
|
+
+ f"\n# … truncated {omitted} chars …\n"
|
|
124
|
+
+ text[-tail:]
|
|
125
|
+
)
|
|
126
|
+
start, end = window
|
|
127
|
+
before = start - limit
|
|
128
|
+
after = max(0, len(text) - tail - end)
|
|
129
|
+
return (
|
|
130
|
+
text[:limit]
|
|
131
|
+
+ f"\n# … truncated {before} chars …\n"
|
|
132
|
+
+ text[start:end]
|
|
133
|
+
+ f"\n# … truncated {after} chars …\n"
|
|
134
|
+
+ text[-tail:]
|
|
135
|
+
)
|
|
136
|
+
|
|
137
|
+
def apply_source(path: Path, source: str, *, original: str) -> None:
|
|
138
|
+
if not source.strip():
|
|
139
|
+
raise ValueError("empty draft")
|
|
140
|
+
if original and len(source) < max(40, (len(original) * 2) // 3):
|
|
141
|
+
raise ValueError(
|
|
142
|
+
f"draft is too short ({len(source)} chars vs {len(original)}) — "
|
|
143
|
+
"use Action: patch for a small change"
|
|
144
|
+
)
|
|
145
|
+
if path.suffix in {".py", ".pyi"}:
|
|
146
|
+
try:
|
|
147
|
+
ast.parse(source)
|
|
148
|
+
except SyntaxError as exc:
|
|
149
|
+
raise ValueError(
|
|
150
|
+
f"syntax error: {exc} — file not written. "
|
|
151
|
+
"Use a full unique line for Find: (not a prefix of def …)"
|
|
152
|
+
) from exc
|
|
153
|
+
bak = path.with_suffix(path.suffix + ".bak")
|
|
154
|
+
if path.is_file():
|
|
155
|
+
bak.write_text(path.read_text(encoding="utf-8"), encoding="utf-8")
|
|
156
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
157
|
+
path.write_text(source.rstrip() + "\n", encoding="utf-8")
|
harness/act/gate.py
ADDED
|
@@ -0,0 +1,229 @@
|
|
|
1
|
+
"""May this draft be written? One question, asked before any file changes.
|
|
2
|
+
|
|
3
|
+
The tools carry a change out. This decides whether it may happen: a
|
|
4
|
+
module that already exists under another name, an import with nothing
|
|
5
|
+
behind it, a style rule the project keeps, a definition already in the
|
|
6
|
+
file. Keeping it apart from `tools` means the answer to "where are the
|
|
7
|
+
tools" is one file, and so is the answer to "what stops a bad write".
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from collections.abc import Callable
|
|
13
|
+
from dataclasses import dataclass
|
|
14
|
+
|
|
15
|
+
import re
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
from harness.paths import rel_posix
|
|
19
|
+
from harness.skillkit.refuse_change import (
|
|
20
|
+
refuse_add_opens_file,
|
|
21
|
+
refuse_layout,
|
|
22
|
+
refuse_opaque_module,
|
|
23
|
+
refuse_opaque_names,
|
|
24
|
+
refuse_ops_draft,
|
|
25
|
+
refuse_platform_draft,
|
|
26
|
+
refuse_rename_incomplete,
|
|
27
|
+
refuse_shell_fetch,
|
|
28
|
+
refuse_stdlib_shadow,
|
|
29
|
+
refuse_stub_body,
|
|
30
|
+
refuse_test_in_impl,
|
|
31
|
+
refuse_undefined_draft,
|
|
32
|
+
refuse_weak_test,
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
_TEST_METH = re.compile(r"def\s+(test_\w+)\s*\(")
|
|
38
|
+
_ASSERT_CALL = re.compile(r"assertEqual\s*\(\s*([A-Za-z_]\w+)\s*\(")
|
|
39
|
+
_IMPORT_LINE = re.compile(r"^(from\s+\S+\s+import\s+)(.+)$")
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _add_import_symbol(text: str, name: str) -> str:
|
|
43
|
+
if not name or name in {"self", "True", "False", "None"}:
|
|
44
|
+
return text
|
|
45
|
+
for line in text.splitlines():
|
|
46
|
+
match = _IMPORT_LINE.match(line)
|
|
47
|
+
if not match:
|
|
48
|
+
continue
|
|
49
|
+
imported = {part.strip() for part in match.group(2).split(",")}
|
|
50
|
+
if name in imported:
|
|
51
|
+
return text
|
|
52
|
+
if any(skip in line for skip in ("unittest", "pathlib", "typing")):
|
|
53
|
+
continue
|
|
54
|
+
return text.replace(line, f"{match.group(1)}{match.group(2).rstrip()}, {name}", 1)
|
|
55
|
+
return text
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _called_name(original: str, append: str) -> str:
|
|
59
|
+
"""The function the new test needs imported.
|
|
60
|
+
|
|
61
|
+
It used to be read out of `assertEqual(multiply(...))`. The change rules
|
|
62
|
+
ask for the opposite shape — `got = multiply(...)`, then assert `got` —
|
|
63
|
+
so following them meant the import was never added and the suite broke.
|
|
64
|
+
Reading the names the new test leaves unbound covers both shapes.
|
|
65
|
+
"""
|
|
66
|
+
from harness.scan.names import new_undefined
|
|
67
|
+
|
|
68
|
+
for name in new_undefined(original, original.rstrip() + "\n\n" + append):
|
|
69
|
+
return name
|
|
70
|
+
match = _ASSERT_CALL.search(append)
|
|
71
|
+
return match.group(1) if match else ""
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def repair_unittest_append(original: str, append: str) -> str | None:
|
|
75
|
+
"""8B Append: often lands after if __name__ and skips the import."""
|
|
76
|
+
if "def test_" not in append:
|
|
77
|
+
return None
|
|
78
|
+
if "TestCase" not in original and "unittest" not in original:
|
|
79
|
+
return None
|
|
80
|
+
meth = _TEST_METH.search(append)
|
|
81
|
+
if not meth or re.search(rf"def\s+{re.escape(meth.group(1))}\s*\(", original):
|
|
82
|
+
return None
|
|
83
|
+
lines = append.strip("\n").splitlines()
|
|
84
|
+
while lines and not lines[0].strip():
|
|
85
|
+
lines.pop(0)
|
|
86
|
+
if not lines:
|
|
87
|
+
return None
|
|
88
|
+
base = len(lines[0]) - len(lines[0].lstrip())
|
|
89
|
+
dedented = "\n".join(
|
|
90
|
+
line[base:] if len(line) >= base else line.lstrip() for line in lines
|
|
91
|
+
)
|
|
92
|
+
method = " " + dedented.replace("\n", "\n ")
|
|
93
|
+
text = _add_import_symbol(original, _called_name(original, append))
|
|
94
|
+
marker = "\nif __name__"
|
|
95
|
+
if marker in text:
|
|
96
|
+
return text.replace(marker, "\n" + method + "\n" + marker, 1)
|
|
97
|
+
return text.rstrip() + "\n\n" + method + "\n"
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def refuse_duplicate_module(project: Path, rel: str, original: str) -> str:
|
|
101
|
+
"""Refuse a new file that repeats a module this project already has.
|
|
102
|
+
|
|
103
|
+
Watched a run write the same function to `pkg/orders.py` and then
|
|
104
|
+
`src/orders.py`. Two modules with one name is worse than either: an
|
|
105
|
+
import finds whichever comes first on the path, and the other rots.
|
|
106
|
+
"""
|
|
107
|
+
if original.strip():
|
|
108
|
+
return ""
|
|
109
|
+
from harness.paths import as_project_rel, rel_posix
|
|
110
|
+
|
|
111
|
+
wanted = Path(as_project_rel(rel))
|
|
112
|
+
if wanted.name.startswith("test_") or "tests" in wanted.parts:
|
|
113
|
+
return ""
|
|
114
|
+
root = Path(project).resolve()
|
|
115
|
+
for existing in sorted(root.rglob(f"{wanted.stem}.py")):
|
|
116
|
+
if any(part in {".git", ".venv", "__pycache__"} for part in existing.parts):
|
|
117
|
+
continue
|
|
118
|
+
found = rel_posix(existing, root)
|
|
119
|
+
if found != wanted.as_posix():
|
|
120
|
+
return (
|
|
121
|
+
f"{found} is already this project's {wanted.stem} module. "
|
|
122
|
+
f"Action: patch Path: {found} Append: the new function"
|
|
123
|
+
)
|
|
124
|
+
return ""
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def refuse_missing_import_target(project: Path, rel: str, draft: str) -> str:
|
|
128
|
+
"""Refuse a file importing a name this project has not defined yet.
|
|
129
|
+
|
|
130
|
+
Asked to create a module and a test for it, the model wrote only the
|
|
131
|
+
test, importing a function nobody had written. That reads as valid
|
|
132
|
+
Python — the import binds the name — and fails when the suite runs. The
|
|
133
|
+
function has to exist first.
|
|
134
|
+
"""
|
|
135
|
+
from harness.scan.names import missing_import_targets
|
|
136
|
+
|
|
137
|
+
missing = missing_import_targets(project, draft)
|
|
138
|
+
if not missing:
|
|
139
|
+
return ""
|
|
140
|
+
module, name = missing[0]
|
|
141
|
+
return (
|
|
142
|
+
f"{module} does not define {name} yet. Write the function first: "
|
|
143
|
+
f"Action: patch Path: {module.replace('.', '/')}.py Append: def {name}(...)"
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
@dataclass(frozen=True)
|
|
148
|
+
class ProposedChange:
|
|
149
|
+
"""One change the model has proposed, and what it is judged on.
|
|
150
|
+
|
|
151
|
+
Fields:
|
|
152
|
+
task: what the user asked for, in their own words.
|
|
153
|
+
rel: the file the change targets, project-relative.
|
|
154
|
+
original: the file as it stands, or "" when it is new.
|
|
155
|
+
draft: the file as this change would leave it.
|
|
156
|
+
fragment: only the part being added, when the change appends.
|
|
157
|
+
Judging a whole file as one test turns a single inline
|
|
158
|
+
assertion anywhere into a refusal for everything in it.
|
|
159
|
+
"""
|
|
160
|
+
|
|
161
|
+
task: str
|
|
162
|
+
rel: str
|
|
163
|
+
original: str
|
|
164
|
+
draft: str
|
|
165
|
+
fragment: str = ""
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
# Every rule a proposed change is put through, in the order they run.
|
|
169
|
+
# This was thirty lines of `blocked = rule(...); if blocked: return
|
|
170
|
+
# blocked`, eleven times over, which hid both the order and the fact that
|
|
171
|
+
# a new rule has to be added to it. A rule written and not listed here
|
|
172
|
+
# does nothing, and a test below checks for exactly that.
|
|
173
|
+
CHANGE_RULES: tuple[tuple[str, Callable[[ProposedChange], str]], ...] = (
|
|
174
|
+
("stdlib shadow", lambda c: refuse_stdlib_shadow(c.rel, c.original)),
|
|
175
|
+
("layout", lambda c: refuse_layout(c.rel, c.original, c.draft)),
|
|
176
|
+
("opens a file", lambda c: refuse_add_opens_file(c.task, c.rel, c.draft)),
|
|
177
|
+
("shell fetch", lambda c: refuse_shell_fetch(c.rel, c.draft)),
|
|
178
|
+
("platform draft", lambda c: refuse_platform_draft(c.rel, c.draft)),
|
|
179
|
+
("operations draft", lambda c: refuse_ops_draft(c.rel, c.draft)),
|
|
180
|
+
("test in implementation", lambda c: refuse_test_in_impl(c.rel, c.draft)),
|
|
181
|
+
("stub body", lambda c: refuse_stub_body(c.task, c.rel, c.draft)),
|
|
182
|
+
(
|
|
183
|
+
"undefined name",
|
|
184
|
+
lambda c: refuse_undefined_draft(c.task, c.rel, c.original, c.draft),
|
|
185
|
+
),
|
|
186
|
+
(
|
|
187
|
+
"half a rename",
|
|
188
|
+
lambda c: refuse_rename_incomplete(c.task, c.rel, c.draft),
|
|
189
|
+
),
|
|
190
|
+
("weak test", lambda c: refuse_weak_test(c.rel, c.fragment or c.draft)),
|
|
191
|
+
("opaque names", lambda c: refuse_opaque_names(c.draft, c.task)),
|
|
192
|
+
("opaque module", lambda c: refuse_opaque_module(c.rel, c.original)),
|
|
193
|
+
)
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def first_refusal(
|
|
197
|
+
task: str, rel: str, original: str, draft: str, fragment: str = ""
|
|
198
|
+
) -> str:
|
|
199
|
+
"""The first refusal a proposed change earns, or "" if it earns none."""
|
|
200
|
+
change = ProposedChange(task, rel, original, draft, fragment)
|
|
201
|
+
for _name, rule in CHANGE_RULES:
|
|
202
|
+
refusal = rule(change)
|
|
203
|
+
if refusal:
|
|
204
|
+
return refusal
|
|
205
|
+
return ""
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def already_defined(original: str, append: str, rel: str) -> str:
|
|
209
|
+
"""Refuse appending a definition the file already has, or "".
|
|
210
|
+
|
|
211
|
+
Only test methods were checked, so an ordinary function could be
|
|
212
|
+
added twice: a live run appended `def slugify` to the same file on
|
|
213
|
+
two separate turns and left both in place.
|
|
214
|
+
"""
|
|
215
|
+
meth = _TEST_METH.search(append)
|
|
216
|
+
if meth and re.search(rf"def\s+{re.escape(meth.group(1))}\s*\(", original):
|
|
217
|
+
return (
|
|
218
|
+
f"{meth.group(1)} already exists. Action: done Summary: "
|
|
219
|
+
"that function is already covered."
|
|
220
|
+
)
|
|
221
|
+
for name in re.findall(r"(?m)^def\s+(\w+)\s*\(", append):
|
|
222
|
+
if re.search(rf"(?m)^def\s+{re.escape(name)}\s*\(", original):
|
|
223
|
+
return (
|
|
224
|
+
f"{name} is already defined in {rel}. Action: done "
|
|
225
|
+
"Summary: say what it does, or patch the existing one."
|
|
226
|
+
)
|
|
227
|
+
return ""
|
|
228
|
+
|
|
229
|
+
|
harness/act/parse.py
ADDED
|
@@ -0,0 +1,247 @@
|
|
|
1
|
+
"""Parse one agent turn. Deterministic. No model."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
|
|
8
|
+
from harness.act.code import extract_python
|
|
9
|
+
|
|
10
|
+
# A chat model decorates the line the harness is trying to read. It
|
|
11
|
+
# emboldens the label (`**Action:** patch`), numbers its steps
|
|
12
|
+
# (`1. Action: patch`), or explains the verb in the same breath
|
|
13
|
+
# (`Action: patch to add slugify`). Local weights write the bare line,
|
|
14
|
+
# so an anchored pattern was enough until the harness was pointed at a
|
|
15
|
+
# model it does not run itself — and then the turn parsed to nothing.
|
|
16
|
+
#
|
|
17
|
+
# Only `*`, `#`, `>` and list numbering are allowed as decoration, and
|
|
18
|
+
# never `_`, because a value may legitimately start with one and
|
|
19
|
+
# `Name: _helpers` must keep its underscore.
|
|
20
|
+
_DECOR = r"[ \t>*#-]{0,6}(?:\d{1,2}[.)][ \t]*)?[ \t>*]{0,4}"
|
|
21
|
+
_AFTER_KEY = r"[* \t]*"
|
|
22
|
+
|
|
23
|
+
# Hyphens are allowed so a skill name works as an action: every kit
|
|
24
|
+
# skill is hyphenated, and `\w+` could not match one. There is no `$`:
|
|
25
|
+
# anything after the verb is the model thinking aloud, and an unknown
|
|
26
|
+
# verb is still rejected by KNOWN_ACTIONS below.
|
|
27
|
+
_ACTION = re.compile(
|
|
28
|
+
rf"^{_DECOR}Action:{_AFTER_KEY}([\w-]+)", re.MULTILINE | re.IGNORECASE
|
|
29
|
+
)
|
|
30
|
+
# Every verb the loop can carry out. A model that writes `Action: find`
|
|
31
|
+
# has put a field name on the Action line; that block is skipped so the
|
|
32
|
+
# turn is not spent on an unknown verb.
|
|
33
|
+
KNOWN_ACTIONS = frozenset(
|
|
34
|
+
{
|
|
35
|
+
"glob", "grep", "read", "edit", "patch", "run", "map", "plan",
|
|
36
|
+
"skill", "locate", "layout", "ask", "done",
|
|
37
|
+
"issue", "branch", "commit", "push", "pr", "merge",
|
|
38
|
+
}
|
|
39
|
+
)
|
|
40
|
+
_FIELD = re.compile(
|
|
41
|
+
rf"^{_DECOR}(Path|File|Query|Pattern|Argv|Summary|Scope|Name|Number|Title):"
|
|
42
|
+
rf"{_AFTER_KEY}(.+?)[ \t*]*$",
|
|
43
|
+
re.MULTILINE,
|
|
44
|
+
)
|
|
45
|
+
_STOP = re.compile(
|
|
46
|
+
rf"^{_DECOR}(Action|Path|File|Query|Pattern|Argv|Summary|Scope|Name|Number"
|
|
47
|
+
rf"|Title|Body|Find|Replace|Append|Add):{_AFTER_KEY}",
|
|
48
|
+
re.IGNORECASE,
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
# A model that answers in chat wraps code in a fence. Local weights
|
|
53
|
+
# happen not to; every hosted one does, so this only bites the moment
|
|
54
|
+
# the harness is pointed at a model it does not run itself.
|
|
55
|
+
# A fence marker alone on the last line, with or without a language:
|
|
56
|
+
# a whole-reply fence closing, or a second block the model opened
|
|
57
|
+
# and never filled. Either way it is not Python.
|
|
58
|
+
_CLOSING = re.compile(r"^[ \t]*```[\w+-]*[ \t]*$")
|
|
59
|
+
_FENCED = re.compile(
|
|
60
|
+
r"^[^\S\n]*```[\w+-]*[^\S\n]*\n(.*?)\n[^\S\n]*```",
|
|
61
|
+
re.DOTALL,
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def unfenced(body: str) -> str:
|
|
66
|
+
"""The code inside a markdown fence, or the text unchanged.
|
|
67
|
+
|
|
68
|
+
An `Append:` body arriving as ```python … ``` used to reach the file
|
|
69
|
+
with the backticks still on it. What landed was a SyntaxError, and
|
|
70
|
+
by its third turn a hosted 32B was reporting an unterminated string
|
|
71
|
+
literal in a file it had broken itself. Nine of ten runs then spent
|
|
72
|
+
the whole budget writing nothing that would load.
|
|
73
|
+
|
|
74
|
+
Anything after the closing fence goes too. A model that signs off
|
|
75
|
+
with "That should do it." puts that sentence inside the fence's
|
|
76
|
+
file otherwise, which fails exactly the same way the backticks do.
|
|
77
|
+
"""
|
|
78
|
+
found = _FENCED.match(body)
|
|
79
|
+
if found:
|
|
80
|
+
return found.group(1)
|
|
81
|
+
# The model fenced its whole reply rather than the code, so the
|
|
82
|
+
# block never opened with a fence and the closing one is left
|
|
83
|
+
# trailing on the end of it. That reaches the file and is the same
|
|
84
|
+
# SyntaxError, arriving by a different route.
|
|
85
|
+
lines = body.splitlines()
|
|
86
|
+
if lines and _CLOSING.match(lines[-1]):
|
|
87
|
+
return "\n".join(lines[:-1]).rstrip()
|
|
88
|
+
return body
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _block(text: str, key: str) -> str:
|
|
92
|
+
match = re.search(
|
|
93
|
+
rf"^{_DECOR}{key}:{_AFTER_KEY}(.*?)[ \t*]*$",
|
|
94
|
+
text,
|
|
95
|
+
re.MULTILINE | re.IGNORECASE,
|
|
96
|
+
)
|
|
97
|
+
if not match:
|
|
98
|
+
return ""
|
|
99
|
+
lines: list[str] = []
|
|
100
|
+
first = match.group(1).rstrip()
|
|
101
|
+
if first:
|
|
102
|
+
lines.append(first)
|
|
103
|
+
for line in text[match.end() :].lstrip("\n").splitlines():
|
|
104
|
+
if _STOP.match(line):
|
|
105
|
+
break
|
|
106
|
+
lines.append(line.rstrip())
|
|
107
|
+
return "\n".join(lines).rstrip()
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
@dataclass(frozen=True)
|
|
111
|
+
class AgentTurn:
|
|
112
|
+
action: str
|
|
113
|
+
path: str = ""
|
|
114
|
+
query: str = ""
|
|
115
|
+
pattern: str = ""
|
|
116
|
+
argv: tuple[str, ...] = ()
|
|
117
|
+
summary: str = ""
|
|
118
|
+
source: str | None = None
|
|
119
|
+
find: str = ""
|
|
120
|
+
replace: str = ""
|
|
121
|
+
scope: str = ""
|
|
122
|
+
name: str = ""
|
|
123
|
+
append: str = ""
|
|
124
|
+
number: str = ""
|
|
125
|
+
title: str = ""
|
|
126
|
+
body: str = ""
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
_PREFERRED_WRITE = ("patch", "edit", "run", "locate", "done")
|
|
130
|
+
_PREFERRED_QUESTION = ("done", "locate", "grep", "read")
|
|
131
|
+
_PREFERRED_SHIP = (
|
|
132
|
+
"issue",
|
|
133
|
+
"branch",
|
|
134
|
+
"commit",
|
|
135
|
+
"push",
|
|
136
|
+
"pr",
|
|
137
|
+
"merge",
|
|
138
|
+
"patch",
|
|
139
|
+
"done",
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
# A field name written on the Action line. The model means the action that
|
|
144
|
+
# field belongs to, and every such turn was spent on "unknown Action".
|
|
145
|
+
_FIELD_AS_ACTION = {
|
|
146
|
+
"append": "patch",
|
|
147
|
+
"add": "patch",
|
|
148
|
+
"find": "patch",
|
|
149
|
+
"replace": "patch",
|
|
150
|
+
"summary": "done",
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _body_after_action(text: str) -> str:
|
|
155
|
+
"""The code below the Action line, with the field lines left out.
|
|
156
|
+
|
|
157
|
+
`Action: append` is followed by `Path:` and then the function. Stopping
|
|
158
|
+
at the first field line returned nothing, so the turn was spent for no
|
|
159
|
+
reason.
|
|
160
|
+
"""
|
|
161
|
+
match = _ACTION.search(text)
|
|
162
|
+
if not match:
|
|
163
|
+
return ""
|
|
164
|
+
lines = [
|
|
165
|
+
line.rstrip()
|
|
166
|
+
for line in text[match.end():].splitlines()
|
|
167
|
+
if not _STOP.match(line)
|
|
168
|
+
]
|
|
169
|
+
while lines and not lines[0].strip():
|
|
170
|
+
lines.pop(0)
|
|
171
|
+
return "\n".join(lines).strip("\n")
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def parse_turn(text: str) -> AgentTurn | None:
|
|
175
|
+
match = _ACTION.search(text)
|
|
176
|
+
if not match:
|
|
177
|
+
return None
|
|
178
|
+
action = match.group(1).lower()
|
|
179
|
+
body_as_append = ""
|
|
180
|
+
if action in _FIELD_AS_ACTION and action not in KNOWN_ACTIONS:
|
|
181
|
+
if action in {"append", "add"}:
|
|
182
|
+
body_as_append = _body_after_action(text)
|
|
183
|
+
action = _FIELD_AS_ACTION[action]
|
|
184
|
+
fields = {m.group(1).lower(): m.group(2).strip() for m in _FIELD.finditer(text)}
|
|
185
|
+
argv = tuple(part for part in fields.get("argv", "").split() if part)
|
|
186
|
+
source = extract_python(text) if action == "edit" else None
|
|
187
|
+
if action == "edit" and not source:
|
|
188
|
+
extra = _block(text, "Append") or _block(text, "Add")
|
|
189
|
+
if extra:
|
|
190
|
+
source = extra
|
|
191
|
+
return AgentTurn(
|
|
192
|
+
action=action,
|
|
193
|
+
path=fields.get("path") or fields.get("file", ""),
|
|
194
|
+
query=fields.get("query", ""),
|
|
195
|
+
pattern=fields.get("pattern", ""),
|
|
196
|
+
argv=argv,
|
|
197
|
+
summary=fields.get("summary", ""),
|
|
198
|
+
source=source,
|
|
199
|
+
find=unfenced(_block(text, "Find")),
|
|
200
|
+
replace=unfenced(_block(text, "Replace")),
|
|
201
|
+
scope=fields.get("scope", ""),
|
|
202
|
+
name=fields.get("name", ""),
|
|
203
|
+
append=unfenced(
|
|
204
|
+
_block(text, "Append") or _block(text, "Add") or body_as_append
|
|
205
|
+
),
|
|
206
|
+
number=fields.get("number", ""),
|
|
207
|
+
title=fields.get("title", ""),
|
|
208
|
+
body=_block(text, "Body"),
|
|
209
|
+
)
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _is_skill_name(verb: str) -> bool:
|
|
213
|
+
"""`Action: write-tests` names a skill, which the loop loads."""
|
|
214
|
+
return "-" in verb or "_" in verb
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def parse_turn_smart(
|
|
218
|
+
text: str, *, question: bool = False, ship: bool = False
|
|
219
|
+
) -> AgentTurn | None:
|
|
220
|
+
"""Small models paste the Action menu. Pick one block by task kind."""
|
|
221
|
+
matches = [
|
|
222
|
+
match
|
|
223
|
+
for match in _ACTION.finditer(text)
|
|
224
|
+
if match.group(1).lower() in KNOWN_ACTIONS or _is_skill_name(match.group(1))
|
|
225
|
+
]
|
|
226
|
+
if not matches:
|
|
227
|
+
return parse_turn(text)
|
|
228
|
+
if len(matches) == 1:
|
|
229
|
+
return parse_turn(text[matches[0].start() :])
|
|
230
|
+
if question:
|
|
231
|
+
prefer = _PREFERRED_QUESTION
|
|
232
|
+
elif ship:
|
|
233
|
+
prefer = _PREFERRED_SHIP
|
|
234
|
+
else:
|
|
235
|
+
prefer = _PREFERRED_WRITE
|
|
236
|
+
chosen = matches[0]
|
|
237
|
+
for match in matches:
|
|
238
|
+
if match.group(1).lower() in prefer:
|
|
239
|
+
chosen = match
|
|
240
|
+
break
|
|
241
|
+
start = chosen.start()
|
|
242
|
+
end = len(text)
|
|
243
|
+
for match in matches:
|
|
244
|
+
if match.start() > start:
|
|
245
|
+
end = match.start()
|
|
246
|
+
break
|
|
247
|
+
return parse_turn(text[start:end])
|