codeplain 0.3.11.dev20__py3-none-any.whl → 0.3.11.dev22__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {codeplain-0.3.11.dev20.dist-info → codeplain-0.3.11.dev22.dist-info}/METADATA +1 -1
- {codeplain-0.3.11.dev20.dist-info → codeplain-0.3.11.dev22.dist-info}/RECORD +14 -17
- codeplain_REST_api.py +32 -26
- render_machine/actions/fix_unit_tests.py +47 -224
- render_machine/actions/run_unit_tests.py +0 -9
- render_machine/render_context.py +0 -9
- render_machine/render_types.py +0 -45
- tests/test_fix_unit_tests_conformance_context.py +25 -120
- tests/test_tui_components.py +60 -1
- tui/components.py +25 -2
- tui/plain2code_tui.py +9 -0
- render_machine/agent_tools.py +0 -270
- tests/test_agent_tools.py +0 -166
- tests/test_fix_unit_tests_action.py +0 -239
- {codeplain-0.3.11.dev20.dist-info → codeplain-0.3.11.dev22.dist-info}/WHEEL +0 -0
- {codeplain-0.3.11.dev20.dist-info → codeplain-0.3.11.dev22.dist-info}/entry_points.txt +0 -0
- {codeplain-0.3.11.dev20.dist-info → codeplain-0.3.11.dev22.dist-info}/licenses/LICENSE +0 -0
tests/test_agent_tools.py
DELETED
|
@@ -1,166 +0,0 @@
|
|
|
1
|
-
"""Tests for the client-side agent tool implementations."""
|
|
2
|
-
|
|
3
|
-
import os
|
|
4
|
-
from types import SimpleNamespace
|
|
5
|
-
|
|
6
|
-
import pytest
|
|
7
|
-
|
|
8
|
-
from render_machine import agent_tools
|
|
9
|
-
from render_machine.render_context import RenderContext
|
|
10
|
-
from render_machine.render_types import UnitTestsRunningContext
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
class FakeRenderContext(SimpleNamespace):
|
|
14
|
-
unit_tests_agent_session = RenderContext.unit_tests_agent_session
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
@pytest.fixture
|
|
18
|
-
def project(tmp_path, monkeypatch):
|
|
19
|
-
build = tmp_path / "plain_modules" / "m"
|
|
20
|
-
build.mkdir(parents=True)
|
|
21
|
-
(build / "app.py").write_text("def add(a, b):\n return a - b\n\n\ndef sub(a, b):\n return a - b\n")
|
|
22
|
-
(tmp_path / "outside.txt").write_text("outside\n")
|
|
23
|
-
monkeypatch.chdir(tmp_path)
|
|
24
|
-
render_context = FakeRenderContext(
|
|
25
|
-
build_folder=str(build),
|
|
26
|
-
unit_tests_running_context=UnitTestsRunningContext(fix_attempts=1),
|
|
27
|
-
conformance_tests_running_context=None,
|
|
28
|
-
unittests_script=None,
|
|
29
|
-
test_script_timeout=None,
|
|
30
|
-
stop_event=None,
|
|
31
|
-
)
|
|
32
|
-
return SimpleNamespace(root=tmp_path, build=build, rc=render_context)
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
def test_read_file_resolves_relative_to_build_folder_with_paging(project):
|
|
36
|
-
out = agent_tools.read_file({"file_path": "app.py", "offset": 2, "limit": 1}, project.rc)
|
|
37
|
-
assert out.startswith("2: return a - b")
|
|
38
|
-
assert "use offset=3 to continue" in out
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
def test_read_is_confined_to_build_folder(project):
|
|
42
|
-
outside = str(project.root / "outside.txt")
|
|
43
|
-
assert agent_tools.read_file({"file_path": outside}, project.rc).startswith("Error: read access denied")
|
|
44
|
-
assert agent_tools.read_file({"file_path": "../../outside.txt"}, project.rc).startswith("Error: read access denied")
|
|
45
|
-
assert agent_tools.grep({"pattern": "outside", "file_path": outside}, project.rc).startswith("Error: read access")
|
|
46
|
-
assert agent_tools.ls_files({"pattern": str(project.root)}, project.rc).startswith("Error: read access denied")
|
|
47
|
-
assert agent_tools.read_file({"file_path": "/etc/hosts"}, project.rc).startswith("Error: read access denied")
|
|
48
|
-
assert agent_tools.read_file({"file_path": "missing.py"}, project.rc).startswith("Error: file not found")
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
def test_edit_file_requires_unique_match_and_tracks_change(project):
|
|
52
|
-
ambiguous = agent_tools.edit_file({"file_path": "app.py", "search": "return a - b", "replace": "x"}, project.rc)
|
|
53
|
-
assert "found 2 times" in ambiguous
|
|
54
|
-
assert project.rc.unit_tests_running_context.changed_files == set()
|
|
55
|
-
|
|
56
|
-
ok = agent_tools.edit_file(
|
|
57
|
-
{
|
|
58
|
-
"file_path": "app.py",
|
|
59
|
-
"search": "def add(a, b):\n return a - b",
|
|
60
|
-
"replace": "def add(a, b):\n return a + b",
|
|
61
|
-
},
|
|
62
|
-
project.rc,
|
|
63
|
-
)
|
|
64
|
-
assert ok.startswith("Edited")
|
|
65
|
-
assert "return a + b" in (project.build / "app.py").read_text()
|
|
66
|
-
assert project.rc.unit_tests_running_context.changed_files == {"app.py"}
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
def test_write_and_delete_are_confined_to_build_folder(project):
|
|
70
|
-
denied = agent_tools.write_file({"file_path": str(project.root / "evil.py"), "content": "x"}, project.rc)
|
|
71
|
-
assert denied.startswith("Error: write access denied")
|
|
72
|
-
assert not (project.root / "evil.py").exists()
|
|
73
|
-
|
|
74
|
-
assert agent_tools.write_file({"file_path": "pkg/new.py", "content": "print(1)\n"}, project.rc).startswith("Wrote")
|
|
75
|
-
assert (project.build / "pkg" / "new.py").read_text() == "print(1)\n"
|
|
76
|
-
assert agent_tools.delete_file({"file_path": "pkg/new.py"}, project.rc).startswith("Deleted")
|
|
77
|
-
assert not (project.build / "pkg" / "new.py").exists()
|
|
78
|
-
assert project.rc.unit_tests_running_context.changed_files == {os.path.join("pkg", "new.py")}
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
def test_grep_and_ls(project):
|
|
82
|
-
hits = agent_tools.grep({"pattern": "def sub"}, project.rc)
|
|
83
|
-
assert hits == "app.py:5:def sub(a, b):"
|
|
84
|
-
assert agent_tools.grep({"pattern": "nope"}, project.rc).startswith("No matches")
|
|
85
|
-
assert agent_tools.grep({"pattern": ""}, project.rc).startswith("Error")
|
|
86
|
-
|
|
87
|
-
assert agent_tools.ls_files({}, project.rc).splitlines()[1:] == ["app.py"]
|
|
88
|
-
assert agent_tools.ls_files({"pattern": "**/*.py"}, project.rc).endswith("app.py")
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
def test_execute_calls_answers_every_call_and_captures_errors(project, monkeypatch):
|
|
92
|
-
def boom(_args, _rc):
|
|
93
|
-
raise RuntimeError("kaput")
|
|
94
|
-
|
|
95
|
-
monkeypatch.setitem(agent_tools.TOOLS, "boom", boom)
|
|
96
|
-
results = agent_tools.execute_calls(
|
|
97
|
-
[
|
|
98
|
-
{"id": "1", "name": "read_file", "args": {"file_path": "app.py", "limit": 1}},
|
|
99
|
-
{"id": "2", "name": "unknown_tool", "args": {}},
|
|
100
|
-
{"id": "3", "name": "boom"},
|
|
101
|
-
],
|
|
102
|
-
project.rc,
|
|
103
|
-
)
|
|
104
|
-
assert [r["call_id"] for r in results] == ["1", "2", "3"]
|
|
105
|
-
assert results[0]["output"].startswith("1: def add")
|
|
106
|
-
assert results[1]["output"] == "Error: unknown tool 'unknown_tool'."
|
|
107
|
-
assert results[2]["output"] == "Error: tool 'boom' failed: RuntimeError: kaput"
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
def test_run_unit_tests_reports_pass_and_failure(project, monkeypatch):
|
|
111
|
-
log = project.root.parent / "log.txt"
|
|
112
|
-
log.write_text("FAILED test_x\nCaused by: boom\n")
|
|
113
|
-
outcomes = iter([(0, "", "/tmp/pass.txt"), (1, "FAILED test_x", str(log))])
|
|
114
|
-
monkeypatch.setattr(agent_tools.render_utils, "execute_script", lambda *a, **k: next(outcomes))
|
|
115
|
-
project.rc.unittests_script = "run_tests.sh"
|
|
116
|
-
context = project.rc.unit_tests_running_context
|
|
117
|
-
|
|
118
|
-
assert agent_tools.run_unit_tests({}, project.rc) == {"output": "All unit tests passed."}
|
|
119
|
-
assert context.verified_passing and context.verified_passing_log_path == "/tmp/pass.txt"
|
|
120
|
-
|
|
121
|
-
failure = agent_tools.run_unit_tests({}, project.rc)
|
|
122
|
-
assert failure["output"].startswith(f"Unit tests failed (exit code 1). Full log: {log}")
|
|
123
|
-
# raw output goes to the server for condensing, not truncated here
|
|
124
|
-
assert failure["test_output"] == "FAILED test_x"
|
|
125
|
-
# the full log is outside the build folder but greppable
|
|
126
|
-
assert agent_tools.grep({"pattern": "Caused by", "file_path": str(log)}, project.rc).endswith("Caused by: boom")
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
def test_file_change_invalidates_verified_pass_and_read_cache(project):
|
|
130
|
-
context = project.rc.unit_tests_running_context
|
|
131
|
-
context.verified_passing = True
|
|
132
|
-
first = agent_tools.execute_calls([{"id": "1", "name": "read_file", "args": {"file_path": "app.py"}}], project.rc)
|
|
133
|
-
repeat = agent_tools.execute_calls([{"id": "2", "name": "read_file", "args": {"file_path": "app.py"}}], project.rc)
|
|
134
|
-
assert first[0]["output"].startswith("1: def add")
|
|
135
|
-
assert repeat[0]["output"].startswith("Same call as an earlier one")
|
|
136
|
-
|
|
137
|
-
agent_tools.write_file({"file_path": "other.py", "content": "y = 1\n"}, project.rc)
|
|
138
|
-
assert context.verified_passing is False
|
|
139
|
-
again = agent_tools.execute_calls([{"id": "3", "name": "read_file", "args": {"file_path": "app.py"}}], project.rc)
|
|
140
|
-
assert again[0]["output"].startswith("1: def add")
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
def test_edit_file_returns_the_edited_region(project):
|
|
144
|
-
out = agent_tools.edit_file(
|
|
145
|
-
{
|
|
146
|
-
"file_path": "app.py",
|
|
147
|
-
"search": "def sub(a, b):\n return a - b",
|
|
148
|
-
"replace": "def sub(a, b):\n return b",
|
|
149
|
-
},
|
|
150
|
-
project.rc,
|
|
151
|
-
)
|
|
152
|
-
assert out.startswith("Edited") and "5: def sub(a, b):\n6: return b" in out
|
|
153
|
-
assert "2: return a - b" in out and "1: def add" not in out
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
def test_grep_context_lines_and_include(project):
|
|
157
|
-
(project.build / "notes.txt").write_text("def sub is documented here\n")
|
|
158
|
-
hits = agent_tools.grep({"pattern": "def sub", "context_lines": 1, "include": "*.py"}, project.rc)
|
|
159
|
-
assert "notes.txt" not in hits
|
|
160
|
-
assert "app.py-4-" in hits and "app.py:5:def sub(a, b):" in hits and "app.py-6-" in hits
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
def test_bound_truncates_long_lines_and_large_output():
|
|
164
|
-
bounded = agent_tools._bound("a" * (agent_tools.MAX_LINE_CHARS + 5) + "\n" + "b\n" * 40_000)
|
|
165
|
-
assert "[line truncated]" in bounded and "[truncated" in bounded
|
|
166
|
-
assert len(bounded) < agent_tools.MAX_OUTPUT_CHARS + 200
|
|
@@ -1,239 +0,0 @@
|
|
|
1
|
-
"""Tests for the agentic FixUnitTests action driving a scripted fake API."""
|
|
2
|
-
|
|
3
|
-
from types import SimpleNamespace
|
|
4
|
-
|
|
5
|
-
import pytest
|
|
6
|
-
import requests
|
|
7
|
-
|
|
8
|
-
import plain_spec
|
|
9
|
-
from render_machine.actions.fix_unit_tests import MAX_AGENT_TURNS_PER_ATTEMPT, MAX_RELEVANT_FILES_CHARS, FixUnitTests
|
|
10
|
-
from render_machine.actions.run_unit_tests import RunUnitTests
|
|
11
|
-
from render_machine.render_context import RenderContext
|
|
12
|
-
from render_machine.render_types import ScriptExecutionHistory, UnitTestsRunningContext
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
class FakeRenderContext(SimpleNamespace):
|
|
16
|
-
"""A render context with just the attributes FixUnitTests uses, and the real session lookup."""
|
|
17
|
-
|
|
18
|
-
unit_tests_agent_session = RenderContext.unit_tests_agent_session
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
class FakeAPI:
|
|
22
|
-
def __init__(self, responses):
|
|
23
|
-
self.responses = list(responses)
|
|
24
|
-
self.calls = []
|
|
25
|
-
|
|
26
|
-
def agent_start(self, task_type, task_params, frid, module_name, run_state):
|
|
27
|
-
self.calls.append(("start", task_type, task_params, frid, module_name))
|
|
28
|
-
return self.responses.pop(0)
|
|
29
|
-
|
|
30
|
-
def agent_continue(self, session_id, tool_results, frid, module_name, run_state):
|
|
31
|
-
self.calls.append(("continue", session_id, tool_results, frid, module_name))
|
|
32
|
-
response = self.responses.pop(0)
|
|
33
|
-
if isinstance(response, Exception):
|
|
34
|
-
raise response
|
|
35
|
-
return response
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
def _tool_calls(*calls):
|
|
39
|
-
return {"session_id": "s1", "status": "tool_calls", "calls": list(calls)}
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
@pytest.fixture
|
|
43
|
-
def render_context(tmp_path, monkeypatch):
|
|
44
|
-
build = tmp_path / "build"
|
|
45
|
-
build.mkdir()
|
|
46
|
-
(build / "a.py").write_text("x = 1\n")
|
|
47
|
-
monkeypatch.chdir(tmp_path)
|
|
48
|
-
plain_source_tree = {"spec": True}
|
|
49
|
-
specifications = {
|
|
50
|
-
plain_spec.DEFINITIONS: ["- :Foo: is a thing."],
|
|
51
|
-
plain_spec.NON_FUNCTIONAL_REQUIREMENTS: ["- Python 3.11."],
|
|
52
|
-
plain_spec.FUNCTIONAL_REQUIREMENTS: ["- Old feature.", "- New feature."],
|
|
53
|
-
}
|
|
54
|
-
monkeypatch.setattr(plain_spec, "get_specifications_for_frid", lambda tree, frid: (specifications, None))
|
|
55
|
-
return FakeRenderContext(
|
|
56
|
-
codeplain_api=None,
|
|
57
|
-
build_folder=str(build),
|
|
58
|
-
module_name="m",
|
|
59
|
-
run_state=object(),
|
|
60
|
-
plain_source_tree=plain_source_tree,
|
|
61
|
-
unittests_script=None,
|
|
62
|
-
test_script_timeout=None,
|
|
63
|
-
stop_event=None,
|
|
64
|
-
frid_context=SimpleNamespace(frid="2", linked_resources={"schema.json": "{}"}, changed_files={"a.py"}),
|
|
65
|
-
unit_tests_running_context=UnitTestsRunningContext(fix_attempts=1),
|
|
66
|
-
conformance_tests_running_context=None,
|
|
67
|
-
script_execution_history=ScriptExecutionHistory(),
|
|
68
|
-
get_required_modules_functionalities=lambda: {"base": ["- Base feature."]},
|
|
69
|
-
)
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
def test_first_attempt_starts_session_runs_tools_and_stops_at_submit_fix(render_context):
|
|
73
|
-
api = FakeAPI(
|
|
74
|
-
[
|
|
75
|
-
_tool_calls({"id": "c1", "name": "read_file", "args": {"file_path": "a.py"}}),
|
|
76
|
-
_tool_calls(
|
|
77
|
-
{"id": "c2", "name": "edit_file", "args": {"file_path": "a.py", "search": "x = 1", "replace": "x = 2"}},
|
|
78
|
-
{"id": "c3", "name": "submit_fix", "args": {"root_cause": "off by one", "changes_made": "x = 2"}},
|
|
79
|
-
),
|
|
80
|
-
]
|
|
81
|
-
)
|
|
82
|
-
render_context.codeplain_api = api
|
|
83
|
-
|
|
84
|
-
outcome, payload = FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED test_a"})
|
|
85
|
-
|
|
86
|
-
assert (outcome, payload) == (FixUnitTests.SUCCESSFUL_OUTCOME, None)
|
|
87
|
-
kind, task_type, task_params, frid, module_name = api.calls[0]
|
|
88
|
-
assert (kind, task_type, frid, module_name) == ("start", "fix_unit_tests", "2", "m")
|
|
89
|
-
assert task_params["unittests_issue"] == "FAILED test_a"
|
|
90
|
-
assert task_params["definitions"] == "- :Foo: is a thing."
|
|
91
|
-
assert task_params["linked_resources"] == {"schema.json": "{}"}
|
|
92
|
-
assert (
|
|
93
|
-
"### Module: base (Already Implemented, for context)\n- Base feature." in task_params["functional_requirements"]
|
|
94
|
-
)
|
|
95
|
-
assert "### Module: m (Already Implemented, for context)\n- Old feature." in task_params["functional_requirements"]
|
|
96
|
-
assert task_params["functional_requirements"].endswith(
|
|
97
|
-
"### Module: m (Currently Being Implemented)\n- New feature."
|
|
98
|
-
)
|
|
99
|
-
|
|
100
|
-
# the read_file result went back to the server; the edit was applied locally
|
|
101
|
-
assert api.calls[1][0] == "continue" and api.calls[1][2][0]["call_id"] == "c1"
|
|
102
|
-
assert "1: x = 1" in api.calls[1][2][0]["output"]
|
|
103
|
-
assert open(render_context.build_folder + "/a.py").read() == "x = 2\n"
|
|
104
|
-
|
|
105
|
-
context = render_context.unit_tests_running_context
|
|
106
|
-
session = context.agent_session
|
|
107
|
-
assert session.session_id == "s1"
|
|
108
|
-
assert session.pending_submit_call_id == "c3"
|
|
109
|
-
assert [r["call_id"] for r in session.pending_tool_results] == ["c2"]
|
|
110
|
-
assert context.changed_files == {"a.py"}
|
|
111
|
-
assert "conformance_tests_fixes" not in task_params
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
def test_second_attempt_continues_session_answering_submit_fix(render_context):
|
|
115
|
-
context = render_context.unit_tests_running_context
|
|
116
|
-
context.agent_used_in_this_loop = True
|
|
117
|
-
session = context.agent_session
|
|
118
|
-
session.session_id = "s1"
|
|
119
|
-
session.pending_submit_call_id = "c3"
|
|
120
|
-
session.pending_tool_results = [{"call_id": "c2", "output": "Edited"}]
|
|
121
|
-
api = FakeAPI([_tool_calls({"id": "c4", "name": "submit_fix", "args": {"changes_made": "again"}})])
|
|
122
|
-
render_context.codeplain_api = api
|
|
123
|
-
|
|
124
|
-
outcome, _ = FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED test_b"})
|
|
125
|
-
|
|
126
|
-
assert outcome == FixUnitTests.SUCCESSFUL_OUTCOME
|
|
127
|
-
kind, session_id, tool_results, frid, module_name = api.calls[0]
|
|
128
|
-
assert (kind, session_id, frid, module_name) == ("continue", "s1", "2", "m")
|
|
129
|
-
assert tool_results[0] == {"call_id": "c2", "output": "Edited"}
|
|
130
|
-
assert tool_results[1]["call_id"] == "c3" and "still fail" in tool_results[1]["output"]
|
|
131
|
-
assert tool_results[1]["test_output"] == "FAILED test_b"
|
|
132
|
-
assert "conformance_tests_fixes" not in tool_results[1]
|
|
133
|
-
assert session.session_id == "s1" and session.pending_submit_call_id == "c4"
|
|
134
|
-
assert session.pending_tool_results == []
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
@pytest.mark.parametrize(
|
|
138
|
-
"final_response",
|
|
139
|
-
[
|
|
140
|
-
{"session_id": "s1", "status": "completed", "result": "done"},
|
|
141
|
-
{"session_id": "s1", "status": "failed", "error": "x"},
|
|
142
|
-
],
|
|
143
|
-
)
|
|
144
|
-
def test_session_ending_without_submission_resets_the_session(render_context, final_response):
|
|
145
|
-
render_context.codeplain_api = FakeAPI([final_response])
|
|
146
|
-
outcome, _ = FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED"})
|
|
147
|
-
assert outcome == FixUnitTests.SUCCESSFUL_OUTCOME
|
|
148
|
-
session = render_context.unit_tests_running_context.agent_session
|
|
149
|
-
assert session.session_id is None and session.pending_submit_call_id is None
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
def test_turn_cap_per_attempt_resets_the_session(render_context):
|
|
153
|
-
call = {"id": "c", "name": "ls_files", "args": {}}
|
|
154
|
-
render_context.codeplain_api = FakeAPI([_tool_calls(call)] * (MAX_AGENT_TURNS_PER_ATTEMPT + 1))
|
|
155
|
-
FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED"})
|
|
156
|
-
assert len(render_context.codeplain_api.calls) == MAX_AGENT_TURNS_PER_ATTEMPT + 1
|
|
157
|
-
assert render_context.unit_tests_running_context.agent_session.session_id is None
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
def test_missing_issue_is_an_internal_error(render_context):
|
|
161
|
-
from plain2code_exceptions import InternalClientError
|
|
162
|
-
|
|
163
|
-
with pytest.raises(InternalClientError):
|
|
164
|
-
FixUnitTests().execute(render_context, {})
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
def test_first_turn_is_seeded_with_file_tree_relevant_files_and_log_path(render_context, tmp_path):
|
|
168
|
-
(tmp_path / "build" / "tests").mkdir()
|
|
169
|
-
(tmp_path / "build" / "tests" / "test_a.py").write_text("assert True\n")
|
|
170
|
-
log = tmp_path / "unit.log"
|
|
171
|
-
log.write_text("full log\nCaused by: boom\n")
|
|
172
|
-
render_context.script_execution_history.latest_unit_test_output_path = str(log)
|
|
173
|
-
api = FakeAPI([{"session_id": "s1", "status": "completed", "result": "done"}])
|
|
174
|
-
render_context.codeplain_api = api
|
|
175
|
-
|
|
176
|
-
FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED"})
|
|
177
|
-
|
|
178
|
-
task_params = api.calls[0][2]
|
|
179
|
-
assert task_params["file_tree"].split("\n") == ["a.py", "tests/test_a.py"]
|
|
180
|
-
assert task_params["relevant_files"] == {"a.py": "x = 1\n"}
|
|
181
|
-
assert task_params["unittests_log_path"] == str(log)
|
|
182
|
-
assert "previous_session_id" not in task_params
|
|
183
|
-
# the agent may grep the full log although it is outside the build folder and project root
|
|
184
|
-
assert str(log) in render_context.unit_tests_running_context.agent_session.readable_log_paths
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
def test_relevant_files_stay_within_budget(tmp_path):
|
|
188
|
-
(tmp_path / "small.py").write_text("s")
|
|
189
|
-
(tmp_path / "big.py").write_text("b" * MAX_RELEVANT_FILES_CHARS)
|
|
190
|
-
assert FixUnitTests._relevant_files(str(tmp_path), {"small.py", "big.py", "deleted.py"}) == {"small.py": "s"}
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
def test_new_session_after_abandoned_one_references_it(render_context):
|
|
194
|
-
call = {"id": "c", "name": "ls_files", "args": {}}
|
|
195
|
-
render_context.codeplain_api = FakeAPI([_tool_calls(call)] * (MAX_AGENT_TURNS_PER_ATTEMPT + 1))
|
|
196
|
-
FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED"})
|
|
197
|
-
assert render_context.unit_tests_running_context.agent_session.previous_session_id == "s1"
|
|
198
|
-
|
|
199
|
-
api = FakeAPI([{"session_id": "s2", "status": "completed", "result": "done"}])
|
|
200
|
-
render_context.codeplain_api = api
|
|
201
|
-
FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED again"})
|
|
202
|
-
assert api.calls[0][0] == "start" and api.calls[0][2]["previous_session_id"] == "s1"
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
def test_run_unit_tests_action_skips_the_suite_after_a_verified_agent_run(render_context, monkeypatch):
|
|
206
|
-
import render_machine.render_utils as render_utils
|
|
207
|
-
|
|
208
|
-
context = render_context.unit_tests_running_context
|
|
209
|
-
context.verified_passing, context.verified_passing_log_path = True, "/logs/pass.log"
|
|
210
|
-
monkeypatch.setattr(render_utils, "execute_script", lambda *a, **k: pytest.fail("suite must not run"))
|
|
211
|
-
|
|
212
|
-
assert RunUnitTests().execute(render_context, None) == (RunUnitTests.SUCCESSFUL_OUTCOME, None)
|
|
213
|
-
assert context.verified_passing is False
|
|
214
|
-
assert render_context.script_execution_history.latest_unit_test_output_path == "/logs/pass.log"
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
def test_expired_session_is_replaced_by_a_new_one(render_context):
|
|
218
|
-
context = render_context.unit_tests_running_context
|
|
219
|
-
context.agent_used_in_this_loop = True
|
|
220
|
-
session = context.agent_session
|
|
221
|
-
session.session_id, session.pending_submit_call_id = "expired", "c3"
|
|
222
|
-
not_found = requests.exceptions.HTTPError(response=SimpleNamespace(status_code=404))
|
|
223
|
-
api = FakeAPI([not_found, {"session_id": "s2", "status": "completed", "result": "done"}])
|
|
224
|
-
render_context.codeplain_api = api
|
|
225
|
-
|
|
226
|
-
outcome, _ = FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED test_b"})
|
|
227
|
-
|
|
228
|
-
assert outcome == FixUnitTests.SUCCESSFUL_OUTCOME
|
|
229
|
-
assert [call[0] for call in api.calls] == ["continue", "start"]
|
|
230
|
-
assert api.calls[1][2]["unittests_issue"] == "FAILED test_b"
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
def test_other_http_errors_on_continue_are_not_swallowed(render_context):
|
|
234
|
-
context = render_context.unit_tests_running_context
|
|
235
|
-
context.agent_used_in_this_loop = True
|
|
236
|
-
context.agent_session.session_id, context.agent_session.pending_submit_call_id = "s1", "c3"
|
|
237
|
-
render_context.codeplain_api = FakeAPI([requests.exceptions.HTTPError(response=SimpleNamespace(status_code=500))])
|
|
238
|
-
with pytest.raises(requests.exceptions.HTTPError):
|
|
239
|
-
FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED"})
|
|
File without changes
|
|
File without changes
|
|
File without changes
|