codeplain 0.3.11.dev19__py3-none-any.whl → 0.3.11.dev21__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {codeplain-0.3.11.dev19.dist-info → codeplain-0.3.11.dev21.dist-info}/METADATA +1 -1
- {codeplain-0.3.11.dev19.dist-info → codeplain-0.3.11.dev21.dist-info}/RECORD +17 -14
- codeplain_REST_api.py +26 -32
- render_machine/actions/fix_unit_tests.py +224 -47
- render_machine/actions/run_unit_tests.py +9 -0
- render_machine/agent_tools.py +270 -0
- render_machine/render_context.py +9 -0
- render_machine/render_types.py +45 -0
- tests/test_agent_tools.py +166 -0
- tests/test_fix_unit_tests_action.py +239 -0
- tests/test_fix_unit_tests_conformance_context.py +120 -25
- tests/test_tui_components.py +60 -1
- tui/components.py +25 -2
- tui/plain2code_tui.py +9 -0
- {codeplain-0.3.11.dev19.dist-info → codeplain-0.3.11.dev21.dist-info}/WHEEL +0 -0
- {codeplain-0.3.11.dev19.dist-info → codeplain-0.3.11.dev21.dist-info}/entry_points.txt +0 -0
- {codeplain-0.3.11.dev19.dist-info → codeplain-0.3.11.dev21.dist-info}/licenses/LICENSE +0 -0
|
@@ -0,0 +1,239 @@
|
|
|
1
|
+
"""Tests for the agentic FixUnitTests action driving a scripted fake API."""
|
|
2
|
+
|
|
3
|
+
from types import SimpleNamespace
|
|
4
|
+
|
|
5
|
+
import pytest
|
|
6
|
+
import requests
|
|
7
|
+
|
|
8
|
+
import plain_spec
|
|
9
|
+
from render_machine.actions.fix_unit_tests import MAX_AGENT_TURNS_PER_ATTEMPT, MAX_RELEVANT_FILES_CHARS, FixUnitTests
|
|
10
|
+
from render_machine.actions.run_unit_tests import RunUnitTests
|
|
11
|
+
from render_machine.render_context import RenderContext
|
|
12
|
+
from render_machine.render_types import ScriptExecutionHistory, UnitTestsRunningContext
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class FakeRenderContext(SimpleNamespace):
|
|
16
|
+
"""A render context with just the attributes FixUnitTests uses, and the real session lookup."""
|
|
17
|
+
|
|
18
|
+
unit_tests_agent_session = RenderContext.unit_tests_agent_session
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class FakeAPI:
|
|
22
|
+
def __init__(self, responses):
|
|
23
|
+
self.responses = list(responses)
|
|
24
|
+
self.calls = []
|
|
25
|
+
|
|
26
|
+
def agent_start(self, task_type, task_params, frid, module_name, run_state):
|
|
27
|
+
self.calls.append(("start", task_type, task_params, frid, module_name))
|
|
28
|
+
return self.responses.pop(0)
|
|
29
|
+
|
|
30
|
+
def agent_continue(self, session_id, tool_results, frid, module_name, run_state):
|
|
31
|
+
self.calls.append(("continue", session_id, tool_results, frid, module_name))
|
|
32
|
+
response = self.responses.pop(0)
|
|
33
|
+
if isinstance(response, Exception):
|
|
34
|
+
raise response
|
|
35
|
+
return response
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _tool_calls(*calls):
|
|
39
|
+
return {"session_id": "s1", "status": "tool_calls", "calls": list(calls)}
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@pytest.fixture
|
|
43
|
+
def render_context(tmp_path, monkeypatch):
|
|
44
|
+
build = tmp_path / "build"
|
|
45
|
+
build.mkdir()
|
|
46
|
+
(build / "a.py").write_text("x = 1\n")
|
|
47
|
+
monkeypatch.chdir(tmp_path)
|
|
48
|
+
plain_source_tree = {"spec": True}
|
|
49
|
+
specifications = {
|
|
50
|
+
plain_spec.DEFINITIONS: ["- :Foo: is a thing."],
|
|
51
|
+
plain_spec.NON_FUNCTIONAL_REQUIREMENTS: ["- Python 3.11."],
|
|
52
|
+
plain_spec.FUNCTIONAL_REQUIREMENTS: ["- Old feature.", "- New feature."],
|
|
53
|
+
}
|
|
54
|
+
monkeypatch.setattr(plain_spec, "get_specifications_for_frid", lambda tree, frid: (specifications, None))
|
|
55
|
+
return FakeRenderContext(
|
|
56
|
+
codeplain_api=None,
|
|
57
|
+
build_folder=str(build),
|
|
58
|
+
module_name="m",
|
|
59
|
+
run_state=object(),
|
|
60
|
+
plain_source_tree=plain_source_tree,
|
|
61
|
+
unittests_script=None,
|
|
62
|
+
test_script_timeout=None,
|
|
63
|
+
stop_event=None,
|
|
64
|
+
frid_context=SimpleNamespace(frid="2", linked_resources={"schema.json": "{}"}, changed_files={"a.py"}),
|
|
65
|
+
unit_tests_running_context=UnitTestsRunningContext(fix_attempts=1),
|
|
66
|
+
conformance_tests_running_context=None,
|
|
67
|
+
script_execution_history=ScriptExecutionHistory(),
|
|
68
|
+
get_required_modules_functionalities=lambda: {"base": ["- Base feature."]},
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def test_first_attempt_starts_session_runs_tools_and_stops_at_submit_fix(render_context):
|
|
73
|
+
api = FakeAPI(
|
|
74
|
+
[
|
|
75
|
+
_tool_calls({"id": "c1", "name": "read_file", "args": {"file_path": "a.py"}}),
|
|
76
|
+
_tool_calls(
|
|
77
|
+
{"id": "c2", "name": "edit_file", "args": {"file_path": "a.py", "search": "x = 1", "replace": "x = 2"}},
|
|
78
|
+
{"id": "c3", "name": "submit_fix", "args": {"root_cause": "off by one", "changes_made": "x = 2"}},
|
|
79
|
+
),
|
|
80
|
+
]
|
|
81
|
+
)
|
|
82
|
+
render_context.codeplain_api = api
|
|
83
|
+
|
|
84
|
+
outcome, payload = FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED test_a"})
|
|
85
|
+
|
|
86
|
+
assert (outcome, payload) == (FixUnitTests.SUCCESSFUL_OUTCOME, None)
|
|
87
|
+
kind, task_type, task_params, frid, module_name = api.calls[0]
|
|
88
|
+
assert (kind, task_type, frid, module_name) == ("start", "fix_unit_tests", "2", "m")
|
|
89
|
+
assert task_params["unittests_issue"] == "FAILED test_a"
|
|
90
|
+
assert task_params["definitions"] == "- :Foo: is a thing."
|
|
91
|
+
assert task_params["linked_resources"] == {"schema.json": "{}"}
|
|
92
|
+
assert (
|
|
93
|
+
"### Module: base (Already Implemented, for context)\n- Base feature." in task_params["functional_requirements"]
|
|
94
|
+
)
|
|
95
|
+
assert "### Module: m (Already Implemented, for context)\n- Old feature." in task_params["functional_requirements"]
|
|
96
|
+
assert task_params["functional_requirements"].endswith(
|
|
97
|
+
"### Module: m (Currently Being Implemented)\n- New feature."
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
# the read_file result went back to the server; the edit was applied locally
|
|
101
|
+
assert api.calls[1][0] == "continue" and api.calls[1][2][0]["call_id"] == "c1"
|
|
102
|
+
assert "1: x = 1" in api.calls[1][2][0]["output"]
|
|
103
|
+
assert open(render_context.build_folder + "/a.py").read() == "x = 2\n"
|
|
104
|
+
|
|
105
|
+
context = render_context.unit_tests_running_context
|
|
106
|
+
session = context.agent_session
|
|
107
|
+
assert session.session_id == "s1"
|
|
108
|
+
assert session.pending_submit_call_id == "c3"
|
|
109
|
+
assert [r["call_id"] for r in session.pending_tool_results] == ["c2"]
|
|
110
|
+
assert context.changed_files == {"a.py"}
|
|
111
|
+
assert "conformance_tests_fixes" not in task_params
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def test_second_attempt_continues_session_answering_submit_fix(render_context):
|
|
115
|
+
context = render_context.unit_tests_running_context
|
|
116
|
+
context.agent_used_in_this_loop = True
|
|
117
|
+
session = context.agent_session
|
|
118
|
+
session.session_id = "s1"
|
|
119
|
+
session.pending_submit_call_id = "c3"
|
|
120
|
+
session.pending_tool_results = [{"call_id": "c2", "output": "Edited"}]
|
|
121
|
+
api = FakeAPI([_tool_calls({"id": "c4", "name": "submit_fix", "args": {"changes_made": "again"}})])
|
|
122
|
+
render_context.codeplain_api = api
|
|
123
|
+
|
|
124
|
+
outcome, _ = FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED test_b"})
|
|
125
|
+
|
|
126
|
+
assert outcome == FixUnitTests.SUCCESSFUL_OUTCOME
|
|
127
|
+
kind, session_id, tool_results, frid, module_name = api.calls[0]
|
|
128
|
+
assert (kind, session_id, frid, module_name) == ("continue", "s1", "2", "m")
|
|
129
|
+
assert tool_results[0] == {"call_id": "c2", "output": "Edited"}
|
|
130
|
+
assert tool_results[1]["call_id"] == "c3" and "still fail" in tool_results[1]["output"]
|
|
131
|
+
assert tool_results[1]["test_output"] == "FAILED test_b"
|
|
132
|
+
assert "conformance_tests_fixes" not in tool_results[1]
|
|
133
|
+
assert session.session_id == "s1" and session.pending_submit_call_id == "c4"
|
|
134
|
+
assert session.pending_tool_results == []
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
@pytest.mark.parametrize(
|
|
138
|
+
"final_response",
|
|
139
|
+
[
|
|
140
|
+
{"session_id": "s1", "status": "completed", "result": "done"},
|
|
141
|
+
{"session_id": "s1", "status": "failed", "error": "x"},
|
|
142
|
+
],
|
|
143
|
+
)
|
|
144
|
+
def test_session_ending_without_submission_resets_the_session(render_context, final_response):
|
|
145
|
+
render_context.codeplain_api = FakeAPI([final_response])
|
|
146
|
+
outcome, _ = FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED"})
|
|
147
|
+
assert outcome == FixUnitTests.SUCCESSFUL_OUTCOME
|
|
148
|
+
session = render_context.unit_tests_running_context.agent_session
|
|
149
|
+
assert session.session_id is None and session.pending_submit_call_id is None
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def test_turn_cap_per_attempt_resets_the_session(render_context):
|
|
153
|
+
call = {"id": "c", "name": "ls_files", "args": {}}
|
|
154
|
+
render_context.codeplain_api = FakeAPI([_tool_calls(call)] * (MAX_AGENT_TURNS_PER_ATTEMPT + 1))
|
|
155
|
+
FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED"})
|
|
156
|
+
assert len(render_context.codeplain_api.calls) == MAX_AGENT_TURNS_PER_ATTEMPT + 1
|
|
157
|
+
assert render_context.unit_tests_running_context.agent_session.session_id is None
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def test_missing_issue_is_an_internal_error(render_context):
|
|
161
|
+
from plain2code_exceptions import InternalClientError
|
|
162
|
+
|
|
163
|
+
with pytest.raises(InternalClientError):
|
|
164
|
+
FixUnitTests().execute(render_context, {})
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def test_first_turn_is_seeded_with_file_tree_relevant_files_and_log_path(render_context, tmp_path):
|
|
168
|
+
(tmp_path / "build" / "tests").mkdir()
|
|
169
|
+
(tmp_path / "build" / "tests" / "test_a.py").write_text("assert True\n")
|
|
170
|
+
log = tmp_path / "unit.log"
|
|
171
|
+
log.write_text("full log\nCaused by: boom\n")
|
|
172
|
+
render_context.script_execution_history.latest_unit_test_output_path = str(log)
|
|
173
|
+
api = FakeAPI([{"session_id": "s1", "status": "completed", "result": "done"}])
|
|
174
|
+
render_context.codeplain_api = api
|
|
175
|
+
|
|
176
|
+
FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED"})
|
|
177
|
+
|
|
178
|
+
task_params = api.calls[0][2]
|
|
179
|
+
assert task_params["file_tree"].split("\n") == ["a.py", "tests/test_a.py"]
|
|
180
|
+
assert task_params["relevant_files"] == {"a.py": "x = 1\n"}
|
|
181
|
+
assert task_params["unittests_log_path"] == str(log)
|
|
182
|
+
assert "previous_session_id" not in task_params
|
|
183
|
+
# the agent may grep the full log although it is outside the build folder and project root
|
|
184
|
+
assert str(log) in render_context.unit_tests_running_context.agent_session.readable_log_paths
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def test_relevant_files_stay_within_budget(tmp_path):
|
|
188
|
+
(tmp_path / "small.py").write_text("s")
|
|
189
|
+
(tmp_path / "big.py").write_text("b" * MAX_RELEVANT_FILES_CHARS)
|
|
190
|
+
assert FixUnitTests._relevant_files(str(tmp_path), {"small.py", "big.py", "deleted.py"}) == {"small.py": "s"}
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def test_new_session_after_abandoned_one_references_it(render_context):
|
|
194
|
+
call = {"id": "c", "name": "ls_files", "args": {}}
|
|
195
|
+
render_context.codeplain_api = FakeAPI([_tool_calls(call)] * (MAX_AGENT_TURNS_PER_ATTEMPT + 1))
|
|
196
|
+
FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED"})
|
|
197
|
+
assert render_context.unit_tests_running_context.agent_session.previous_session_id == "s1"
|
|
198
|
+
|
|
199
|
+
api = FakeAPI([{"session_id": "s2", "status": "completed", "result": "done"}])
|
|
200
|
+
render_context.codeplain_api = api
|
|
201
|
+
FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED again"})
|
|
202
|
+
assert api.calls[0][0] == "start" and api.calls[0][2]["previous_session_id"] == "s1"
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def test_run_unit_tests_action_skips_the_suite_after_a_verified_agent_run(render_context, monkeypatch):
|
|
206
|
+
import render_machine.render_utils as render_utils
|
|
207
|
+
|
|
208
|
+
context = render_context.unit_tests_running_context
|
|
209
|
+
context.verified_passing, context.verified_passing_log_path = True, "/logs/pass.log"
|
|
210
|
+
monkeypatch.setattr(render_utils, "execute_script", lambda *a, **k: pytest.fail("suite must not run"))
|
|
211
|
+
|
|
212
|
+
assert RunUnitTests().execute(render_context, None) == (RunUnitTests.SUCCESSFUL_OUTCOME, None)
|
|
213
|
+
assert context.verified_passing is False
|
|
214
|
+
assert render_context.script_execution_history.latest_unit_test_output_path == "/logs/pass.log"
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def test_expired_session_is_replaced_by_a_new_one(render_context):
|
|
218
|
+
context = render_context.unit_tests_running_context
|
|
219
|
+
context.agent_used_in_this_loop = True
|
|
220
|
+
session = context.agent_session
|
|
221
|
+
session.session_id, session.pending_submit_call_id = "expired", "c3"
|
|
222
|
+
not_found = requests.exceptions.HTTPError(response=SimpleNamespace(status_code=404))
|
|
223
|
+
api = FakeAPI([not_found, {"session_id": "s2", "status": "completed", "result": "done"}])
|
|
224
|
+
render_context.codeplain_api = api
|
|
225
|
+
|
|
226
|
+
outcome, _ = FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED test_b"})
|
|
227
|
+
|
|
228
|
+
assert outcome == FixUnitTests.SUCCESSFUL_OUTCOME
|
|
229
|
+
assert [call[0] for call in api.calls] == ["continue", "start"]
|
|
230
|
+
assert api.calls[1][2]["unittests_issue"] == "FAILED test_b"
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def test_other_http_errors_on_continue_are_not_swallowed(render_context):
|
|
234
|
+
context = render_context.unit_tests_running_context
|
|
235
|
+
context.agent_used_in_this_loop = True
|
|
236
|
+
context.agent_session.session_id, context.agent_session.pending_submit_call_id = "s1", "c3"
|
|
237
|
+
render_context.codeplain_api = FakeAPI([requests.exceptions.HTTPError(response=SimpleNamespace(status_code=500))])
|
|
238
|
+
with pytest.raises(requests.exceptions.HTTPError):
|
|
239
|
+
FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED"})
|
|
@@ -12,25 +12,44 @@ import pytest
|
|
|
12
12
|
|
|
13
13
|
import plain_spec
|
|
14
14
|
from memory_management import MemoryManager
|
|
15
|
-
from render_machine.actions import fix_unit_tests as fix_unit_tests_module
|
|
16
15
|
from render_machine.actions.fix_conformance_test import FixConformanceTest
|
|
17
16
|
from render_machine.actions.fix_unit_tests import FixUnitTests
|
|
18
17
|
from render_machine.implementation_code_helpers import ImplementationCodeHelpers
|
|
19
|
-
from render_machine.
|
|
18
|
+
from render_machine.render_context import RenderContext
|
|
19
|
+
from render_machine.render_types import ConformanceTestsRunningContext, ScriptExecutionHistory, UnitTestsRunningContext
|
|
20
20
|
|
|
21
21
|
|
|
22
22
|
class FakeCodeplainAPI:
|
|
23
|
-
|
|
23
|
+
"""Conformance fixes return a scripted response; every agent session submits a fix on its first turn."""
|
|
24
|
+
|
|
25
|
+
def __init__(self, conformance_fix_response=None):
|
|
24
26
|
self.conformance_fix_response = conformance_fix_response
|
|
25
|
-
self.
|
|
26
|
-
self.
|
|
27
|
+
self.agent_calls = []
|
|
28
|
+
self.sessions_started = 0
|
|
27
29
|
|
|
28
30
|
def fix_conformance_tests_issue(self, *args, **kwargs):
|
|
29
31
|
return self.conformance_fix_response
|
|
30
32
|
|
|
31
|
-
def
|
|
32
|
-
self.
|
|
33
|
-
|
|
33
|
+
def agent_start(self, task_type, task_params, frid, module_name, run_state):
|
|
34
|
+
self.sessions_started += 1
|
|
35
|
+
self.agent_calls.append(("start", task_params))
|
|
36
|
+
return self._submit(f"s{self.sessions_started}")
|
|
37
|
+
|
|
38
|
+
def agent_continue(self, session_id, tool_results, frid, module_name, run_state):
|
|
39
|
+
self.agent_calls.append(("continue", session_id, tool_results))
|
|
40
|
+
return self._submit(session_id)
|
|
41
|
+
|
|
42
|
+
def _submit(self, session_id):
|
|
43
|
+
call_id = f"submit-{len(self.agent_calls)}"
|
|
44
|
+
return {
|
|
45
|
+
"session_id": session_id,
|
|
46
|
+
"status": "tool_calls",
|
|
47
|
+
"calls": [{"id": call_id, "name": "submit_fix", "args": {"changes_made": "fixed"}}],
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class FakeRenderContext(SimpleNamespace):
|
|
52
|
+
unit_tests_agent_session = RenderContext.unit_tests_agent_session
|
|
34
53
|
|
|
35
54
|
|
|
36
55
|
class FakeConformanceTests:
|
|
@@ -59,7 +78,11 @@ def memory_folder():
|
|
|
59
78
|
def isolate_from_git_and_console(monkeypatch):
|
|
60
79
|
monkeypatch.setattr(ImplementationCodeHelpers, "get_code_diff", staticmethod(lambda *args: {}))
|
|
61
80
|
monkeypatch.setattr(plain_spec, "collect_linked_resources", lambda *args: None)
|
|
62
|
-
monkeypatch.setattr(
|
|
81
|
+
monkeypatch.setattr(
|
|
82
|
+
plain_spec,
|
|
83
|
+
"get_specifications_for_frid",
|
|
84
|
+
lambda tree, frid: ({plain_spec.FUNCTIONAL_REQUIREMENTS: ["- Add numbers."]}, None),
|
|
85
|
+
)
|
|
63
86
|
|
|
64
87
|
|
|
65
88
|
def make_conformance_context():
|
|
@@ -76,7 +99,7 @@ def make_conformance_context():
|
|
|
76
99
|
|
|
77
100
|
|
|
78
101
|
def make_render_context(api, build_folder, memory_folder, conformance_tests_running_context):
|
|
79
|
-
return
|
|
102
|
+
return FakeRenderContext(
|
|
80
103
|
codeplain_api=api,
|
|
81
104
|
build_folder=build_folder,
|
|
82
105
|
memory_manager=MemoryManager(api, memory_folder),
|
|
@@ -86,9 +109,11 @@ def make_render_context(api, build_folder, memory_folder, conformance_tests_runn
|
|
|
86
109
|
plain_source_tree={},
|
|
87
110
|
module_name="mod",
|
|
88
111
|
required_modules=None,
|
|
89
|
-
frid_context=SimpleNamespace(frid="1", linked_resources={}),
|
|
112
|
+
frid_context=SimpleNamespace(frid="1", linked_resources={}, changed_files=set()),
|
|
90
113
|
get_required_modules_functionalities=lambda: {},
|
|
91
114
|
run_state=SimpleNamespace(render_id="test-render-id", unittest_batch_id=1),
|
|
115
|
+
unittests_script=None,
|
|
116
|
+
script_execution_history=ScriptExecutionHistory(),
|
|
92
117
|
)
|
|
93
118
|
|
|
94
119
|
|
|
@@ -191,34 +216,104 @@ def test_implementation_fix_with_no_files_is_not_remembered(build_folder, memory
|
|
|
191
216
|
assert ctx.implementation_code_fixes == []
|
|
192
217
|
|
|
193
218
|
|
|
194
|
-
def run_unit_tests_fix(render_context):
|
|
195
|
-
return FixUnitTests().execute(render_context, {"previous_unittests_issue":
|
|
219
|
+
def run_unit_tests_fix(render_context, issue="1 failed"):
|
|
220
|
+
return FixUnitTests().execute(render_context, {"previous_unittests_issue": issue})
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def start_new_unit_test_loop(render_context):
|
|
224
|
+
"""What RenderContext.start_unittests_processing does when the unit tests are run again."""
|
|
225
|
+
render_context.unit_tests_running_context = UnitTestsRunningContext(fix_attempts=0)
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def conformance_fix(number):
|
|
229
|
+
return {
|
|
230
|
+
"hypothesis": f"hypothesis {number}",
|
|
231
|
+
"approach": f"approach {number}",
|
|
232
|
+
"code_diff": {"app.py": f"+{number}"},
|
|
233
|
+
}
|
|
196
234
|
|
|
197
235
|
|
|
198
|
-
def
|
|
199
|
-
api = FakeCodeplainAPI(
|
|
236
|
+
def test_first_unit_test_loop_of_conformance_phase_seeds_the_session_with_the_fixes(build_folder, memory_folder):
|
|
237
|
+
api = FakeCodeplainAPI()
|
|
200
238
|
ctx = make_conformance_context()
|
|
201
|
-
ctx.implementation_code_fixes.append(
|
|
202
|
-
{"hypothesis": "off by one", "approach": "add one", "code_diff": {"app.py": "+ return a + b + 1"}}
|
|
203
|
-
)
|
|
239
|
+
ctx.implementation_code_fixes.append(conformance_fix(1))
|
|
204
240
|
render_context = make_render_context(api, build_folder, memory_folder, ctx)
|
|
205
241
|
|
|
206
242
|
outcome, _ = run_unit_tests_fix(render_context)
|
|
207
243
|
|
|
208
244
|
assert outcome == FixUnitTests.SUCCESSFUL_OUTCOME
|
|
209
|
-
|
|
210
|
-
assert
|
|
245
|
+
kind, task_params = api.agent_calls[0]
|
|
246
|
+
assert kind == "start"
|
|
247
|
+
assert task_params["conformance_tests_fixes"] == ctx.implementation_code_fixes
|
|
211
248
|
# The forwarded list is a copy, so later conformance fixes do not mutate what was sent.
|
|
212
|
-
assert
|
|
249
|
+
assert task_params["conformance_tests_fixes"] is not ctx.implementation_code_fixes
|
|
250
|
+
# The file the conformance fix changed is seeded although the FRID did not change it.
|
|
251
|
+
assert "app.py" in task_params["relevant_files"]
|
|
252
|
+
assert ctx.unit_tests_agent_session.session_id == "s1"
|
|
213
253
|
|
|
214
254
|
|
|
215
|
-
def
|
|
255
|
+
def test_next_unit_test_loop_of_conformance_phase_continues_the_session_with_only_new_fixes(
|
|
256
|
+
build_folder, memory_folder
|
|
257
|
+
):
|
|
216
258
|
api = FakeCodeplainAPI()
|
|
217
|
-
|
|
259
|
+
ctx = make_conformance_context()
|
|
260
|
+
ctx.implementation_code_fixes.append(conformance_fix(1))
|
|
261
|
+
render_context = make_render_context(api, build_folder, memory_folder, ctx)
|
|
262
|
+
run_unit_tests_fix(render_context)
|
|
263
|
+
|
|
264
|
+
# The fix was accepted; the conformance tests fixer changes the code again and the unit tests fail again.
|
|
265
|
+
ctx.implementation_code_fixes.append(conformance_fix(2))
|
|
266
|
+
start_new_unit_test_loop(render_context)
|
|
267
|
+
run_unit_tests_fix(render_context, issue="2 failed")
|
|
268
|
+
|
|
269
|
+
assert api.sessions_started == 1
|
|
270
|
+
kind, session_id, tool_results = api.agent_calls[1]
|
|
271
|
+
assert (kind, session_id) == ("continue", "s1")
|
|
272
|
+
submit_answer = tool_results[-1]
|
|
273
|
+
assert submit_answer["call_id"] == "submit-1"
|
|
274
|
+
assert submit_answer["output"].startswith("Your fix was accepted: the unit tests passed.")
|
|
275
|
+
assert "Conformance Tests Fix below" in submit_answer["output"]
|
|
276
|
+
assert submit_answer["conformance_tests_fixes"] == [conformance_fix(2)]
|
|
277
|
+
assert submit_answer["test_output"] == "2 failed"
|
|
278
|
+
|
|
218
279
|
|
|
280
|
+
def test_retry_within_a_unit_test_loop_says_the_fix_did_not_work(build_folder, memory_folder):
|
|
281
|
+
api = FakeCodeplainAPI()
|
|
282
|
+
ctx = make_conformance_context()
|
|
283
|
+
ctx.implementation_code_fixes.append(conformance_fix(1))
|
|
284
|
+
render_context = make_render_context(api, build_folder, memory_folder, ctx)
|
|
285
|
+
run_unit_tests_fix(render_context)
|
|
286
|
+
|
|
287
|
+
run_unit_tests_fix(render_context, issue="still failing")
|
|
288
|
+
|
|
289
|
+
submit_answer = api.agent_calls[1][2][-1]
|
|
290
|
+
assert "still fail" in submit_answer["output"]
|
|
291
|
+
# Fix 1 was already shown to the session, so it is not sent again.
|
|
292
|
+
assert "conformance_tests_fixes" not in submit_answer
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
def test_new_conformance_phase_starts_a_new_session(build_folder, memory_folder):
|
|
296
|
+
api = FakeCodeplainAPI()
|
|
297
|
+
render_context = make_render_context(api, build_folder, memory_folder, make_conformance_context())
|
|
298
|
+
run_unit_tests_fix(render_context)
|
|
299
|
+
|
|
300
|
+
# E.g. the functionality is re-rendered from scratch: the conformance tests running context is recreated.
|
|
301
|
+
render_context.conformance_tests_running_context = make_conformance_context()
|
|
302
|
+
start_new_unit_test_loop(render_context)
|
|
303
|
+
run_unit_tests_fix(render_context)
|
|
304
|
+
|
|
305
|
+
assert [call[0] for call in api.agent_calls] == ["start", "start"]
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
def test_unit_tests_outside_conformance_phase_get_no_fixes_and_a_session_per_loop(build_folder, memory_folder):
|
|
309
|
+
api = FakeCodeplainAPI()
|
|
310
|
+
render_context = make_render_context(api, build_folder, memory_folder, None)
|
|
311
|
+
run_unit_tests_fix(render_context)
|
|
312
|
+
start_new_unit_test_loop(render_context)
|
|
219
313
|
run_unit_tests_fix(render_context)
|
|
220
314
|
|
|
221
|
-
assert
|
|
315
|
+
assert [call[0] for call in api.agent_calls] == ["start", "start"]
|
|
316
|
+
assert all("conformance_tests_fixes" not in call[1] for call in api.agent_calls)
|
|
222
317
|
|
|
223
318
|
|
|
224
319
|
def test_unit_tests_fix_in_conformance_phase_without_implementation_changes_sends_no_fixes(build_folder, memory_folder):
|
|
@@ -227,4 +322,4 @@ def test_unit_tests_fix_in_conformance_phase_without_implementation_changes_send
|
|
|
227
322
|
|
|
228
323
|
run_unit_tests_fix(render_context)
|
|
229
324
|
|
|
230
|
-
assert api.
|
|
325
|
+
assert "conformance_tests_fixes" not in api.agent_calls[0][1]
|
tests/test_tui_components.py
CHANGED
|
@@ -6,7 +6,7 @@ from textual.widgets import Static
|
|
|
6
6
|
|
|
7
7
|
from event_bus import EventBus
|
|
8
8
|
from plain2code_state import RunState
|
|
9
|
-
from tui.components import ProgressItem, SubstateLine, TUIComponents
|
|
9
|
+
from tui.components import FRIDProgress, ProgressItem, RenderingInfoBox, SubstateLine, TUIComponents
|
|
10
10
|
from tui.models import Substate
|
|
11
11
|
from tui.plain2code_tui import Plain2CodeTUI
|
|
12
12
|
from tui.widget_helpers import display_error_message, display_success_message, update_progress_item_substates
|
|
@@ -86,3 +86,62 @@ def test_status_messages_keep_brackets():
|
|
|
86
86
|
assert "[Errno 2] No such file: run_tests.sh [b" in str(status.content)
|
|
87
87
|
|
|
88
88
|
asyncio.run(scenario())
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
# A functionality quote as the render TUI receives it: several lines, square brackets included.
|
|
92
|
+
MULTILINE_FUNCTIONALITY_TEXT = (
|
|
93
|
+
"Functionality 3: :User: should be able to add a :Task: [optional]\n"
|
|
94
|
+
" - The :Task: must have non-empty content.\n"
|
|
95
|
+
" - The :Task: is appended to the end of the list."
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def test_functionality_text_collapses_and_expands_with_ctrl_o():
|
|
100
|
+
async def scenario():
|
|
101
|
+
event_bus = EventBus()
|
|
102
|
+
run_state = RunState(spec_filename="x.plain")
|
|
103
|
+
app = _make_app(run_state, event_bus)
|
|
104
|
+
async with app.run_test() as pilot:
|
|
105
|
+
frid_progress = app.query_one(f"#{TUIComponents.FRID_PROGRESS.value}", FRIDProgress)
|
|
106
|
+
info_box = frid_progress.query_one(RenderingInfoBox)
|
|
107
|
+
info_box.update_functionality(MULTILINE_FUNCTIONALITY_TEXT)
|
|
108
|
+
await pilot.pause()
|
|
109
|
+
|
|
110
|
+
widget = info_box.functionality_widget
|
|
111
|
+
assert widget is not None
|
|
112
|
+
|
|
113
|
+
# Collapsed by default: first line only, with the expand hint.
|
|
114
|
+
collapsed = str(widget.content)
|
|
115
|
+
assert "Functionality 3: :User: should be able to add a :Task: [optional]" in collapsed
|
|
116
|
+
assert "non-empty content" not in collapsed
|
|
117
|
+
assert RenderingInfoBox.EXPAND_HINT in collapsed
|
|
118
|
+
|
|
119
|
+
await pilot.press("ctrl+o")
|
|
120
|
+
await pilot.pause()
|
|
121
|
+
expanded = str(widget.content)
|
|
122
|
+
assert "non-empty content" in expanded
|
|
123
|
+
assert "appended to the end of the list" in expanded
|
|
124
|
+
assert RenderingInfoBox.COLLAPSE_HINT in expanded
|
|
125
|
+
|
|
126
|
+
await pilot.press("ctrl+o")
|
|
127
|
+
await pilot.pause()
|
|
128
|
+
assert str(widget.content) == collapsed
|
|
129
|
+
|
|
130
|
+
asyncio.run(scenario())
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def test_single_line_functionality_text_has_no_hint():
|
|
134
|
+
async def scenario():
|
|
135
|
+
event_bus = EventBus()
|
|
136
|
+
run_state = RunState(spec_filename="x.plain")
|
|
137
|
+
app = _make_app(run_state, event_bus)
|
|
138
|
+
async with app.run_test() as pilot:
|
|
139
|
+
info_box = app.query_one(f"#{TUIComponents.FRID_PROGRESS.value}", FRIDProgress).query_one(RenderingInfoBox)
|
|
140
|
+
info_box.update_functionality("Functionality 1: :User: should be able to add a :Task:")
|
|
141
|
+
await pilot.pause()
|
|
142
|
+
|
|
143
|
+
widget = info_box.functionality_widget
|
|
144
|
+
assert widget is not None
|
|
145
|
+
assert str(widget.content) == "Functionality 1: :User: should be able to add a :Task:"
|
|
146
|
+
|
|
147
|
+
asyncio.run(scenario())
|
tui/components.py
CHANGED
|
@@ -339,10 +339,14 @@ class ProgressItem(Vertical):
|
|
|
339
339
|
class RenderingInfoBox(Vertical):
|
|
340
340
|
"""Responsive container for module and functionality information."""
|
|
341
341
|
|
|
342
|
+
EXPAND_HINT = "(ctrl+o to expand)"
|
|
343
|
+
COLLAPSE_HINT = "(ctrl+o to collapse)"
|
|
344
|
+
|
|
342
345
|
def __init__(self, **kwargs):
|
|
343
346
|
super().__init__(**kwargs)
|
|
344
347
|
self.module_text = ""
|
|
345
348
|
self.functionality_text = ""
|
|
349
|
+
self.functionality_expanded = False
|
|
346
350
|
self.module_widget: Static | None = None
|
|
347
351
|
self.functionality_widget: Static | None = None
|
|
348
352
|
|
|
@@ -356,12 +360,31 @@ class RenderingInfoBox(Vertical):
|
|
|
356
360
|
self.functionality_text = text
|
|
357
361
|
self._refresh_content()
|
|
358
362
|
|
|
363
|
+
def toggle_functionality(self) -> None:
|
|
364
|
+
"""Expand or collapse the functionality text."""
|
|
365
|
+
self.functionality_expanded = not self.functionality_expanded
|
|
366
|
+
self._refresh_content()
|
|
367
|
+
|
|
368
|
+
def _format_functionality(self) -> Content:
|
|
369
|
+
"""Build the functionality line, collapsed to its first line unless expanded.
|
|
370
|
+
|
|
371
|
+
The functionality text comes from the spec and may contain square brackets, so it
|
|
372
|
+
is assembled as Content and never goes through the markup parser.
|
|
373
|
+
"""
|
|
374
|
+
text = self.functionality_text or ""
|
|
375
|
+
lines = text.splitlines()
|
|
376
|
+
if len(lines) <= 1:
|
|
377
|
+
return Content(text)
|
|
378
|
+
if self.functionality_expanded:
|
|
379
|
+
return Content.assemble(f"{text}\n", (self.COLLAPSE_HINT, "#888888"))
|
|
380
|
+
return Content.assemble(f"{lines[0]} \u2026 ", (self.EXPAND_HINT, "#888888"))
|
|
381
|
+
|
|
359
382
|
def _refresh_content(self) -> None:
|
|
360
383
|
"""Refresh text inside the box."""
|
|
361
384
|
if self.module_widget is not None:
|
|
362
385
|
self.module_widget.update(self.module_text or "")
|
|
363
386
|
if self.functionality_widget is not None:
|
|
364
|
-
self.functionality_widget.update(self.
|
|
387
|
+
self.functionality_widget.update(self._format_functionality())
|
|
365
388
|
|
|
366
389
|
def on_mount(self) -> None:
|
|
367
390
|
"""Initialize default labels on mount."""
|
|
@@ -372,7 +395,7 @@ class RenderingInfoBox(Vertical):
|
|
|
372
395
|
|
|
373
396
|
def compose(self):
|
|
374
397
|
self.module_widget = Static(self.module_text, classes="rendering-info-row", markup=False)
|
|
375
|
-
self.functionality_widget = Static(self.
|
|
398
|
+
self.functionality_widget = Static(self._format_functionality(), classes="rendering-info-row", markup=False)
|
|
376
399
|
yield Static("module status", classes="rendering-info-title")
|
|
377
400
|
with Vertical(classes="rendering-info-box"):
|
|
378
401
|
yield self.module_widget
|
tui/plain2code_tui.py
CHANGED
|
@@ -65,6 +65,7 @@ class Plain2CodeTUI(App):
|
|
|
65
65
|
Binding("ctrl+d", "quit", "Quit", show=False),
|
|
66
66
|
Binding("enter", "enter_exit", "Exit", show=False),
|
|
67
67
|
Binding("ctrl+p", "pause", "Pause", show=False, priority=True),
|
|
68
|
+
Binding("ctrl+o", "toggle_functionality", "Expand/Collapse", show=False),
|
|
68
69
|
("ctrl+l", "toggle_logs", "Toggle Logs"),
|
|
69
70
|
]
|
|
70
71
|
|
|
@@ -231,6 +232,14 @@ class Plain2CodeTUI(App):
|
|
|
231
232
|
self.run_state.render_time_accumulated,
|
|
232
233
|
)
|
|
233
234
|
|
|
235
|
+
def action_toggle_functionality(self) -> None:
|
|
236
|
+
"""Expand or collapse the functionality text in the rendering info box."""
|
|
237
|
+
try:
|
|
238
|
+
frid_progress = self.query_one(f"#{TUIComponents.FRID_PROGRESS.value}", FRIDProgress)
|
|
239
|
+
frid_progress.query_one(RenderingInfoBox).toggle_functionality()
|
|
240
|
+
except NoMatches:
|
|
241
|
+
pass
|
|
242
|
+
|
|
234
243
|
def action_toggle_logs(self) -> None:
|
|
235
244
|
"""Toggle between dashboard and log view."""
|
|
236
245
|
switcher = self.query_one(f"#{TUIComponents.CONTENT_SWITCHER.value}", ContentSwitcher)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|