py-harness-cli 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- finetune/__init__.py +1 -0
- finetune/agent_system.py +41 -0
- finetune/agent_traces.py +157 -0
- finetune/everyday.py +30 -0
- finetune/hf_ollama.py +158 -0
- finetune/huggingface_store.py +144 -0
- finetune/models.py +74 -0
- finetune/paths.py +11 -0
- finetune/python_vibe.py +788 -0
- finetune/splits.py +54 -0
- finetune/systems.py +9 -0
- harness/__init__.py +42 -0
- harness/__main__.py +8 -0
- harness/act/__init__.py +6 -0
- harness/act/autofix/__init__.py +110 -0
- harness/act/autofix/additions.py +217 -0
- harness/act/autofix/conflicts.py +124 -0
- harness/act/autofix/cover.py +419 -0
- harness/act/autofix/mechanical.py +151 -0
- harness/act/autofix/missing_imports.py +50 -0
- harness/act/autofix/moves.py +439 -0
- harness/act/autofix/names.py +339 -0
- harness/act/autofix/scaffold.py +224 -0
- harness/act/code.py +157 -0
- harness/act/gate.py +229 -0
- harness/act/parse.py +247 -0
- harness/act/patch_fix.py +138 -0
- harness/act/tools.py +244 -0
- harness/agent/__init__.py +11 -0
- harness/agent/dispatch.py +235 -0
- harness/agent/loop.py +699 -0
- harness/agent/options.py +144 -0
- harness/agent/policy.py +856 -0
- harness/agent/prompt.py +170 -0
- harness/cli.py +393 -0
- harness/editor_kit.py +265 -0
- harness/guard/__init__.py +6 -0
- harness/guard/fallbacks.py +6 -0
- harness/guard/loop_guard.py +57 -0
- harness/guard/python_vibe.py +68 -0
- harness/guard/run.py +41 -0
- harness/guard/types.py +19 -0
- harness/locate.py +767 -0
- harness/mcp_stdio.py +306 -0
- harness/memory/__init__.py +5 -0
- harness/memory/conversation.py +104 -0
- harness/model/__init__.py +6 -0
- harness/model/chat_backend.py +100 -0
- harness/model/engine.py +165 -0
- harness/model/ollama_generate.py +60 -0
- harness/model/openai_generate.py +156 -0
- harness/model/outbound.py +83 -0
- harness/model/route.py +90 -0
- harness/observe/__init__.py +6 -0
- harness/observe/eval_gate.py +80 -0
- harness/observe/eval_loop.py +185 -0
- harness/observe/eval_tasks.py +399 -0
- harness/observe/report_md.py +102 -0
- harness/observe/trace_record.py +79 -0
- harness/openai_api.py +81 -0
- harness/paths.py +88 -0
- harness/py.typed +0 -0
- harness/scan/__init__.py +6 -0
- harness/scan/app_spec.py +338 -0
- harness/scan/design.py +112 -0
- harness/scan/existing.py +131 -0
- harness/scan/layout.py +254 -0
- harness/scan/names.py +308 -0
- harness/scan/project_brief.py +287 -0
- harness/scan/project_docs.py +42 -0
- harness/scan/project_scan.py +49 -0
- harness/scan/repo_map.py +101 -0
- harness/secrets.py +39 -0
- harness/server.py +199 -0
- harness/ship/__init__.py +1 -0
- harness/ship/bot_pr.py +221 -0
- harness/ship/git_ship.py +262 -0
- harness/ship/identity.py +62 -0
- harness/ship/ticket.py +251 -0
- harness/skillkit/__init__.py +6 -0
- harness/skillkit/catalog.py +241 -0
- harness/skillkit/refuse_change.py +640 -0
- harness/skillkit/refuse_finish.py +295 -0
- harness/skillkit/target.py +238 -0
- harness/task.py +717 -0
- py_harness_cli-0.3.0.dist-info/METADATA +177 -0
- py_harness_cli-0.3.0.dist-info/RECORD +92 -0
- py_harness_cli-0.3.0.dist-info/WHEEL +5 -0
- py_harness_cli-0.3.0.dist-info/entry_points.txt +3 -0
- py_harness_cli-0.3.0.dist-info/licenses/LICENSE +202 -0
- py_harness_cli-0.3.0.dist-info/licenses/NOTICE +6 -0
- py_harness_cli-0.3.0.dist-info/top_level.txt +2 -0
finetune/splits.py
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""Turn (user, assistant) pairs into mlx-lm chat JSONL splits."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import random
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def chat_row(system: str, user: str, assistant: str) -> dict:
|
|
11
|
+
return {
|
|
12
|
+
"messages": [
|
|
13
|
+
{"role": "system", "content": system},
|
|
14
|
+
{"role": "user", "content": user},
|
|
15
|
+
{"role": "assistant", "content": assistant},
|
|
16
|
+
]
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def write_splits(
|
|
21
|
+
pairs: list[tuple[str, str]],
|
|
22
|
+
system: str,
|
|
23
|
+
dest: Path,
|
|
24
|
+
*,
|
|
25
|
+
seed: int = 7,
|
|
26
|
+
valid_frac: float = 0.12,
|
|
27
|
+
test_frac: float = 0.12,
|
|
28
|
+
) -> dict[str, int]:
|
|
29
|
+
if len(pairs) < 12:
|
|
30
|
+
raise ValueError(f"need at least 12 pairs, got {len(pairs)}")
|
|
31
|
+
|
|
32
|
+
rng = random.Random(seed)
|
|
33
|
+
shuffled = list(pairs)
|
|
34
|
+
rng.shuffle(shuffled)
|
|
35
|
+
|
|
36
|
+
n = len(shuffled)
|
|
37
|
+
n_test = max(2, int(n * test_frac))
|
|
38
|
+
n_valid = max(2, int(n * valid_frac))
|
|
39
|
+
test = shuffled[:n_test]
|
|
40
|
+
valid = shuffled[n_test : n_test + n_valid]
|
|
41
|
+
train = shuffled[n_test + n_valid :]
|
|
42
|
+
if not train:
|
|
43
|
+
raise ValueError("train split is empty — add more examples")
|
|
44
|
+
|
|
45
|
+
dest.mkdir(parents=True, exist_ok=True)
|
|
46
|
+
counts = {}
|
|
47
|
+
for name, rows in (("train", train), ("valid", valid), ("test", test)):
|
|
48
|
+
path = dest / f"{name}.jsonl"
|
|
49
|
+
with path.open("w", encoding="utf-8") as fh:
|
|
50
|
+
for user, assistant in rows:
|
|
51
|
+
fh.write(json.dumps(chat_row(system, user, assistant), ensure_ascii=False))
|
|
52
|
+
fh.write("\n")
|
|
53
|
+
counts[name] = len(rows)
|
|
54
|
+
return counts
|
finetune/systems.py
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
"""System prompt baked into every training example."""
|
|
2
|
+
|
|
3
|
+
PYTHON_VIBE_SYSTEM = """\
|
|
4
|
+
You are a Python vibe-coding pair.
|
|
5
|
+
|
|
6
|
+
Ship working Python 3.12+ first. Prefer the standard library. Short note, \
|
|
7
|
+
then the code. No essays, no extra abstractions, no 'as an AI' preamble. \
|
|
8
|
+
Type-hint public functions. Skip comments that restate the next line.
|
|
9
|
+
"""
|
harness/__init__.py
ADDED
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
"""py-harness: a deterministic harness around a small local model.
|
|
2
|
+
|
|
3
|
+
The model drafts; the harness decides what ships.
|
|
4
|
+
|
|
5
|
+
from harness import Agent, AgentOptions
|
|
6
|
+
|
|
7
|
+
agent = Agent(AgentOptions(project=Path("~/app"), scope="src"))
|
|
8
|
+
result = agent.run("add multiply(a, b) and a unit test")
|
|
9
|
+
print(result.summary, result.writes)
|
|
10
|
+
|
|
11
|
+
Read-only is one flag:
|
|
12
|
+
|
|
13
|
+
AgentOptions(project=..., allow_writes=False)
|
|
14
|
+
|
|
15
|
+
Command line and HTTP are the same object:
|
|
16
|
+
|
|
17
|
+
python -m harness run ~/app "fix the NameError"
|
|
18
|
+
python -m harness serve --project ~/app
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from harness.agent import Agent, AgentOptions, AgentResult, Question, Step
|
|
22
|
+
|
|
23
|
+
# The one thing the command line needs from the model package. Going
|
|
24
|
+
# through here keeps that package's shape private: the CLI and the
|
|
25
|
+
# server do not import harness.model.* directly.
|
|
26
|
+
from harness.model.route import route_advice
|
|
27
|
+
from harness.guard.python_vibe import PythonVibeGuard
|
|
28
|
+
from harness.guard.run import complete
|
|
29
|
+
from harness.guard.types import Finding, Outcome
|
|
30
|
+
|
|
31
|
+
__all__ = [
|
|
32
|
+
"Agent",
|
|
33
|
+
"AgentOptions",
|
|
34
|
+
"route_advice",
|
|
35
|
+
"AgentResult",
|
|
36
|
+
"Question",
|
|
37
|
+
"Step",
|
|
38
|
+
"PythonVibeGuard",
|
|
39
|
+
"complete",
|
|
40
|
+
"Finding",
|
|
41
|
+
"Outcome",
|
|
42
|
+
]
|
harness/__main__.py
ADDED
harness/act/__init__.py
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
"""Read a proposed action and carry it out on the file system.
|
|
2
|
+
|
|
3
|
+
Parses one action from the model's reply and provides the operations it can
|
|
4
|
+
request: search, read, patch, replace and run. All file writes pass through
|
|
5
|
+
the path restriction and backup in `harness.act.code`.
|
|
6
|
+
"""
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
"""Repairs the harness can make on its own, before any model turn.
|
|
2
|
+
|
|
3
|
+
One file of nine hundred lines held six separate jobs: repairing a
|
|
4
|
+
name, resolving a conflict, writing a test, adding a function, adding
|
|
5
|
+
an import, and the pass that runs them in order. They are six modules
|
|
6
|
+
now, and this is the door on to them.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from harness.act.autofix.names import (
|
|
12
|
+
UnboundTypo,
|
|
13
|
+
_all_defined_names,
|
|
14
|
+
_bound_in_scope,
|
|
15
|
+
_class_body_ids,
|
|
16
|
+
_is_typo,
|
|
17
|
+
_rename_name_tokens,
|
|
18
|
+
apply_person_bind,
|
|
19
|
+
apply_typo_fixes,
|
|
20
|
+
apply_zero_return_sum,
|
|
21
|
+
levenshtein,
|
|
22
|
+
replacement_from_answer,
|
|
23
|
+
typo_pairs,
|
|
24
|
+
unbound_typo,
|
|
25
|
+
)
|
|
26
|
+
from harness.act.autofix.conflicts import (
|
|
27
|
+
CONFLICT_END,
|
|
28
|
+
CONFLICT_MID,
|
|
29
|
+
CONFLICT_START,
|
|
30
|
+
_resolve_conflict,
|
|
31
|
+
conflict_blocks,
|
|
32
|
+
looks_like_conflict,
|
|
33
|
+
resolve_keeping_both,
|
|
34
|
+
)
|
|
35
|
+
from harness.act.autofix.cover import (
|
|
36
|
+
MIN_SHARE_REACHED,
|
|
37
|
+
_add_import_symbol,
|
|
38
|
+
_append_class_method,
|
|
39
|
+
_body_lines,
|
|
40
|
+
_candidates,
|
|
41
|
+
_find_callable,
|
|
42
|
+
_imports,
|
|
43
|
+
_lines_reached,
|
|
44
|
+
_sample_values,
|
|
45
|
+
_test_file_for,
|
|
46
|
+
apply_cover_test,
|
|
47
|
+
)
|
|
48
|
+
from harness.act.autofix.additions import (
|
|
49
|
+
_COUNT_NAME,
|
|
50
|
+
_assign_names_for_module,
|
|
51
|
+
_impl_py,
|
|
52
|
+
_top_level_names,
|
|
53
|
+
append_instead_of_replacing,
|
|
54
|
+
apply_add_function,
|
|
55
|
+
apply_function_rename,
|
|
56
|
+
usual_first_arg,
|
|
57
|
+
)
|
|
58
|
+
from harness.act.autofix.missing_imports import (
|
|
59
|
+
apply_missing_imports,
|
|
60
|
+
)
|
|
61
|
+
from harness.act.autofix.moves import (
|
|
62
|
+
apply_file_move,
|
|
63
|
+
apply_function_move,
|
|
64
|
+
module_name,
|
|
65
|
+
move_targets,
|
|
66
|
+
)
|
|
67
|
+
from harness.act.autofix.mechanical import (
|
|
68
|
+
apply_mechanical,
|
|
69
|
+
)
|
|
70
|
+
from harness.act.autofix.scaffold import (
|
|
71
|
+
apply_cli_mock_test,
|
|
72
|
+
apply_home_config,
|
|
73
|
+
apply_list_page_query,
|
|
74
|
+
apply_package_scaffold,
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
__all__ = [
|
|
78
|
+
"apply_file_move",
|
|
79
|
+
"move_targets",
|
|
80
|
+
"CONFLICT_END",
|
|
81
|
+
"CONFLICT_MID",
|
|
82
|
+
"CONFLICT_START",
|
|
83
|
+
"MIN_SHARE_REACHED",
|
|
84
|
+
"UnboundTypo",
|
|
85
|
+
"_is_typo",
|
|
86
|
+
"_rename_name_tokens",
|
|
87
|
+
"_sample_values",
|
|
88
|
+
"_test_file_for",
|
|
89
|
+
"append_instead_of_replacing",
|
|
90
|
+
"apply_add_function",
|
|
91
|
+
"apply_cover_test",
|
|
92
|
+
"apply_function_rename",
|
|
93
|
+
"apply_cli_mock_test",
|
|
94
|
+
"apply_home_config",
|
|
95
|
+
"apply_list_page_query",
|
|
96
|
+
"apply_mechanical",
|
|
97
|
+
"apply_package_scaffold",
|
|
98
|
+
"apply_missing_imports",
|
|
99
|
+
"apply_person_bind",
|
|
100
|
+
"apply_typo_fixes",
|
|
101
|
+
"apply_zero_return_sum",
|
|
102
|
+
"conflict_blocks",
|
|
103
|
+
"levenshtein",
|
|
104
|
+
"looks_like_conflict",
|
|
105
|
+
"replacement_from_answer",
|
|
106
|
+
"resolve_keeping_both",
|
|
107
|
+
"typo_pairs",
|
|
108
|
+
"unbound_typo",
|
|
109
|
+
"usual_first_arg",
|
|
110
|
+
]
|
|
@@ -0,0 +1,217 @@
|
|
|
1
|
+
"""Adding a small function, and appending rather than replacing."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
"""Mechanical fixes the 8B fails to express. Deterministic. No model.
|
|
6
|
+
|
|
7
|
+
Live 8B (29 Aug 2026): left `subtotal` unbound after a NameError task, and
|
|
8
|
+
spent twelve `Find:` turns that never matched `def calc(x: int, ...)`.
|
|
9
|
+
Those are compiler jobs. The harness does them, then runs the suite,
|
|
10
|
+
before the first generate. A green suite ends the run without a model.
|
|
11
|
+
"""
|
|
12
|
+
import ast
|
|
13
|
+
import re
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from harness.act.code import apply_source
|
|
16
|
+
from harness.task import (
|
|
17
|
+
looks_like_add_feature,
|
|
18
|
+
named_project_file,
|
|
19
|
+
question_symbol,
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _top_level_names(tree: ast.Module) -> set[str]:
|
|
26
|
+
"""Names a module defines at its top level."""
|
|
27
|
+
found: set[str] = set()
|
|
28
|
+
for node in tree.body:
|
|
29
|
+
if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)):
|
|
30
|
+
found.add(node.name)
|
|
31
|
+
else:
|
|
32
|
+
found.update(_assign_names_for_module(node))
|
|
33
|
+
return found
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _assign_names_for_module(node: ast.AST) -> set[str]:
|
|
37
|
+
if isinstance(node, ast.Assign):
|
|
38
|
+
names: set[str] = set()
|
|
39
|
+
for target in node.targets:
|
|
40
|
+
if isinstance(target, ast.Name):
|
|
41
|
+
names.add(target.id)
|
|
42
|
+
return names
|
|
43
|
+
if isinstance(node, ast.AnnAssign) and isinstance(node.target, ast.Name):
|
|
44
|
+
return {node.target.id}
|
|
45
|
+
if isinstance(node, (ast.Import, ast.ImportFrom)):
|
|
46
|
+
return {alias.asname or alias.name.split(".")[0] for alias in node.names}
|
|
47
|
+
return set()
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def append_instead_of_replacing(original: str, draft: str) -> str:
|
|
51
|
+
"""Merge a short draft of new definitions onto the file, or "".
|
|
52
|
+
|
|
53
|
+
`edit` replaces a whole file, so a correct new function sent on its
|
|
54
|
+
own is shorter than what it would replace and is refused for that.
|
|
55
|
+
A live 8B wrote a working `slugify`, had it thrown away twice — once
|
|
56
|
+
for a missing fence, then for being 89 characters against 276 — and
|
|
57
|
+
spent the rest of its budget sending the same correct code back.
|
|
58
|
+
|
|
59
|
+
Appending is what it meant, and it is safe: nothing in the file is
|
|
60
|
+
removed. Only when every top-level name in the draft is new, so this
|
|
61
|
+
cannot quietly drop a rewrite of something that already exists.
|
|
62
|
+
"""
|
|
63
|
+
if not original.strip() or not draft.strip():
|
|
64
|
+
return ""
|
|
65
|
+
if len(draft) >= max(40, (len(original) * 2) // 3):
|
|
66
|
+
return "" # long enough to be a real rewrite; leave it alone
|
|
67
|
+
try:
|
|
68
|
+
first, second = ast.parse(original), ast.parse(draft)
|
|
69
|
+
except (SyntaxError, ValueError):
|
|
70
|
+
return ""
|
|
71
|
+
allowed = (
|
|
72
|
+
ast.FunctionDef,
|
|
73
|
+
ast.AsyncFunctionDef,
|
|
74
|
+
ast.ClassDef,
|
|
75
|
+
ast.Import,
|
|
76
|
+
ast.ImportFrom,
|
|
77
|
+
)
|
|
78
|
+
if not second.body or not all(isinstance(n, allowed) for n in second.body):
|
|
79
|
+
return ""
|
|
80
|
+
added = _top_level_names(second)
|
|
81
|
+
if not added or added & _top_level_names(first):
|
|
82
|
+
return ""
|
|
83
|
+
merged = original.rstrip() + "\n\n\n" + draft.strip() + "\n"
|
|
84
|
+
try:
|
|
85
|
+
ast.parse(merged)
|
|
86
|
+
except SyntaxError:
|
|
87
|
+
return ""
|
|
88
|
+
return merged
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def apply_function_rename(source: str, old: str, new: str) -> str:
|
|
92
|
+
"""Rename one `def old` and `old(` calls. Keep the rest of the signature.
|
|
93
|
+
|
|
94
|
+
Matching requires a call shape, so prose that merely mentions the name
|
|
95
|
+
is left alone. Text inside a string that looks like a call is rewritten
|
|
96
|
+
too; for a message naming the function that is usually wanted, and it
|
|
97
|
+
is the same on every supported Python version, which a token-based
|
|
98
|
+
rename would not be.
|
|
99
|
+
"""
|
|
100
|
+
if not old or not new or old == new:
|
|
101
|
+
return source
|
|
102
|
+
if not re.search(rf"^def {re.escape(old)}\b", source, re.MULTILINE):
|
|
103
|
+
return source
|
|
104
|
+
if re.search(rf"^def {re.escape(new)}\b", source, re.MULTILINE):
|
|
105
|
+
return source
|
|
106
|
+
text = re.sub(
|
|
107
|
+
rf"^def {re.escape(old)}\b",
|
|
108
|
+
f"def {new}",
|
|
109
|
+
source,
|
|
110
|
+
count=1,
|
|
111
|
+
flags=re.MULTILINE,
|
|
112
|
+
)
|
|
113
|
+
return re.sub(rf"\b{re.escape(old)}\s*\(", f"{new}(", text)
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _impl_py(project: Path) -> list[tuple[Path, str]]:
|
|
117
|
+
"""First-party Python files that are not tests, with project-relative paths."""
|
|
118
|
+
from harness.scan.project_brief import iter_text_files
|
|
119
|
+
|
|
120
|
+
root = Path(project).resolve()
|
|
121
|
+
found: list[tuple[Path, str]] = []
|
|
122
|
+
for path, _size in iter_text_files(root):
|
|
123
|
+
if path.suffix != ".py":
|
|
124
|
+
continue
|
|
125
|
+
rel = path.resolve().relative_to(root).as_posix()
|
|
126
|
+
if "test" in rel.lower():
|
|
127
|
+
continue
|
|
128
|
+
found.append((path, rel))
|
|
129
|
+
return found
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
_COUNT_NAME = re.compile(r"(lines?|count|^n_)", re.I)
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def usual_first_arg(source: str) -> tuple[str, str]:
|
|
136
|
+
"""Most common first parameter and its annotation, or empty strings."""
|
|
137
|
+
found: list[tuple[str, str]] = []
|
|
138
|
+
try:
|
|
139
|
+
tree = ast.parse(source)
|
|
140
|
+
except (SyntaxError, ValueError):
|
|
141
|
+
return "", ""
|
|
142
|
+
for node in tree.body:
|
|
143
|
+
if not isinstance(node, ast.FunctionDef):
|
|
144
|
+
continue
|
|
145
|
+
args = [a for a in node.args.args if a.arg not in {"self", "cls"}]
|
|
146
|
+
if not args:
|
|
147
|
+
continue
|
|
148
|
+
hint = ast.unparse(args[0].annotation) if args[0].annotation else ""
|
|
149
|
+
found.append((args[0].arg, hint))
|
|
150
|
+
if not found:
|
|
151
|
+
return "", ""
|
|
152
|
+
name = max(set(item[0] for item in found), key=lambda n: sum(1 for a, _h in found if a == n))
|
|
153
|
+
hints = [hint for arg, hint in found if arg == name and hint]
|
|
154
|
+
hint = max(set(hints), key=hints.count) if hints else "list[int]"
|
|
155
|
+
return name, hint
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _spells_another_argument(task: str, symbol: str, neighbor: str) -> bool:
|
|
159
|
+
"""True when the task wrote the signature and it is not this one.
|
|
160
|
+
|
|
161
|
+
`word_count(text)` says the argument is text. Guessing `prices` from
|
|
162
|
+
the neighbours then writes `return len(prices)`, a word counter that
|
|
163
|
+
counts list items — and this path writes the test too, so the suite
|
|
164
|
+
agrees with it and the run reports success having done the wrong
|
|
165
|
+
thing in zero model steps.
|
|
166
|
+
|
|
167
|
+
A task that spells its own arguments has already answered the
|
|
168
|
+
question this function guesses at, so when the two disagree there is
|
|
169
|
+
nothing to add mechanically and the model should do the work.
|
|
170
|
+
"""
|
|
171
|
+
spelled = re.search(rf"\b{re.escape(symbol)}\s*\(([^)]*)\)", task)
|
|
172
|
+
if not spelled:
|
|
173
|
+
return False
|
|
174
|
+
written = [a.strip().split(":")[0].strip() for a in spelled.group(1).split(",")]
|
|
175
|
+
written = [a for a in written if a]
|
|
176
|
+
return bool(written) and neighbor not in written
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def apply_add_function(project: Path, task: str, *, write: bool = True) -> str:
|
|
180
|
+
"""Add a count function that matches its neighbors. Empty if unsure.
|
|
181
|
+
|
|
182
|
+
Live 8B read `add a function total_lines` as a file-line counter and
|
|
183
|
+
opened a path. In an orders module the usual argument is `prices`.
|
|
184
|
+
"""
|
|
185
|
+
if not looks_like_add_feature(task):
|
|
186
|
+
return ""
|
|
187
|
+
symbol = question_symbol(task)
|
|
188
|
+
if not symbol or not _COUNT_NAME.search(symbol):
|
|
189
|
+
return ""
|
|
190
|
+
from harness.skillkit.target import pick_module
|
|
191
|
+
|
|
192
|
+
dest = named_project_file(task, project) or pick_module(project, "", task)
|
|
193
|
+
if not dest:
|
|
194
|
+
return ""
|
|
195
|
+
path = Path(project) / dest
|
|
196
|
+
if not path.is_file():
|
|
197
|
+
return ""
|
|
198
|
+
try:
|
|
199
|
+
body = path.read_text(encoding="utf-8")
|
|
200
|
+
except OSError:
|
|
201
|
+
return ""
|
|
202
|
+
if re.search(rf"^def {re.escape(symbol)}\b", body, re.MULTILINE):
|
|
203
|
+
return ""
|
|
204
|
+
name, hint = usual_first_arg(body)
|
|
205
|
+
if name != "prices" or "list" not in hint.lower():
|
|
206
|
+
return ""
|
|
207
|
+
if _spells_another_argument(task, symbol, name):
|
|
208
|
+
return ""
|
|
209
|
+
stub = f"\n\ndef {symbol}({name}: {hint}) -> int:\n return len({name})\n"
|
|
210
|
+
merged = body.rstrip() + stub
|
|
211
|
+
try:
|
|
212
|
+
ast.parse(merged)
|
|
213
|
+
except SyntaxError:
|
|
214
|
+
return ""
|
|
215
|
+
if write:
|
|
216
|
+
apply_source(path, merged, original=body)
|
|
217
|
+
return f"added def {symbol}({name}) in {dest}"
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
"""Resolving a merge conflict where keeping both sides is safe."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
"""Mechanical fixes the 8B fails to express. Deterministic. No model.
|
|
6
|
+
|
|
7
|
+
Live 8B (29 Aug 2026): left `subtotal` unbound after a NameError task, and
|
|
8
|
+
spent twelve `Find:` turns that never matched `def calc(x: int, ...)`.
|
|
9
|
+
Those are compiler jobs. The harness does them, then runs the suite,
|
|
10
|
+
before the first generate. A green suite ends the run without a model.
|
|
11
|
+
"""
|
|
12
|
+
import ast
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from harness.act.code import apply_source
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
CONFLICT_START = "<<<<<<< "
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
CONFLICT_MID = "======="
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
CONFLICT_END = ">>>>>>> "
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def conflict_blocks(source: str) -> list[tuple[list[str], list[str]]]:
|
|
29
|
+
"""The two sides of every merge conflict in `source`."""
|
|
30
|
+
blocks: list[tuple[list[str], list[str]]] = []
|
|
31
|
+
lines = source.splitlines(keepends=True)
|
|
32
|
+
index = 0
|
|
33
|
+
while index < len(lines):
|
|
34
|
+
if not lines[index].startswith(CONFLICT_START):
|
|
35
|
+
index += 1
|
|
36
|
+
continue
|
|
37
|
+
index += 1
|
|
38
|
+
ours: list[str] = []
|
|
39
|
+
while index < len(lines) and lines[index].rstrip("\n") != CONFLICT_MID:
|
|
40
|
+
ours.append(lines[index])
|
|
41
|
+
index += 1
|
|
42
|
+
index += 1
|
|
43
|
+
theirs: list[str] = []
|
|
44
|
+
while index < len(lines) and not lines[index].startswith(CONFLICT_END):
|
|
45
|
+
theirs.append(lines[index])
|
|
46
|
+
index += 1
|
|
47
|
+
index += 1
|
|
48
|
+
blocks.append((ours, theirs))
|
|
49
|
+
return blocks
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def resolve_keeping_both(source: str) -> str:
|
|
53
|
+
"""Merge every conflict by keeping both sides, or "" if that is unsafe.
|
|
54
|
+
|
|
55
|
+
Two branches that each added something to the same file is the common
|
|
56
|
+
case and the safe one: nothing is lost by keeping both. A conflict
|
|
57
|
+
where one side is empty is a deletion against an edit, and which one
|
|
58
|
+
is wanted is not something to guess at.
|
|
59
|
+
|
|
60
|
+
Asked to resolve a real conflict, a live 8B spent twenty steps and
|
|
61
|
+
left all three markers in place. Asked to resolve one in a file that
|
|
62
|
+
had none, it reported the conflict resolved. Neither needed a model.
|
|
63
|
+
"""
|
|
64
|
+
blocks = conflict_blocks(source)
|
|
65
|
+
if not blocks:
|
|
66
|
+
return ""
|
|
67
|
+
if any(not "".join(ours).strip() or not "".join(theirs).strip()
|
|
68
|
+
for ours, theirs in blocks):
|
|
69
|
+
return ""
|
|
70
|
+
out: list[str] = []
|
|
71
|
+
lines = source.splitlines(keepends=True)
|
|
72
|
+
index = 0
|
|
73
|
+
while index < len(lines):
|
|
74
|
+
if not lines[index].startswith(CONFLICT_START):
|
|
75
|
+
out.append(lines[index])
|
|
76
|
+
index += 1
|
|
77
|
+
continue
|
|
78
|
+
index += 1
|
|
79
|
+
ours = []
|
|
80
|
+
while lines[index].rstrip("\n") != CONFLICT_MID:
|
|
81
|
+
ours.append(lines[index]); index += 1
|
|
82
|
+
index += 1
|
|
83
|
+
theirs = []
|
|
84
|
+
while not lines[index].startswith(CONFLICT_END):
|
|
85
|
+
theirs.append(lines[index]); index += 1
|
|
86
|
+
index += 1
|
|
87
|
+
out.extend(ours)
|
|
88
|
+
# Two definitions need air between them; two import lines do not.
|
|
89
|
+
if theirs and theirs[0].lstrip().startswith(("def ", "class ", "@")):
|
|
90
|
+
out.append("\n\n")
|
|
91
|
+
out.extend(theirs)
|
|
92
|
+
merged = "".join(out)
|
|
93
|
+
try:
|
|
94
|
+
ast.parse(merged)
|
|
95
|
+
except SyntaxError:
|
|
96
|
+
return ""
|
|
97
|
+
return merged
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def looks_like_conflict(task: str) -> bool:
|
|
101
|
+
"""Whether the task is about a merge conflict."""
|
|
102
|
+
lowered = task.lower()
|
|
103
|
+
return "conflict" in lowered and any(
|
|
104
|
+
word in lowered for word in ("merge", "resolve", "rebase", "<<<<")
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _resolve_conflict(path: Path, rel: str, original: str, *, write: bool) -> str:
|
|
109
|
+
"""Keep both sides of every conflict in the file, or say what is there."""
|
|
110
|
+
blocks = conflict_blocks(original)
|
|
111
|
+
if not blocks:
|
|
112
|
+
# Saying so is the point. Asked to resolve a conflict in a file
|
|
113
|
+
# that had none, a live 8B reported the conflict resolved.
|
|
114
|
+
return f"{rel} has no merge conflict in it. Nothing to resolve"
|
|
115
|
+
merged = resolve_keeping_both(original)
|
|
116
|
+
if not merged:
|
|
117
|
+
return (
|
|
118
|
+
f"{rel} has {len(blocks)} conflict(s) where one side is empty. "
|
|
119
|
+
"That is a deletion against an edit, and which one you want is "
|
|
120
|
+
"not something to guess"
|
|
121
|
+
)
|
|
122
|
+
if write:
|
|
123
|
+
apply_source(path, merged, original=original)
|
|
124
|
+
return f"kept both sides of {len(blocks)} conflict(s) in {rel}"
|