codeplain 0.3.11.dev21__py3-none-any.whl → 0.3.11.dev22__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,239 +0,0 @@
1
- """Tests for the agentic FixUnitTests action driving a scripted fake API."""
2
-
3
- from types import SimpleNamespace
4
-
5
- import pytest
6
- import requests
7
-
8
- import plain_spec
9
- from render_machine.actions.fix_unit_tests import MAX_AGENT_TURNS_PER_ATTEMPT, MAX_RELEVANT_FILES_CHARS, FixUnitTests
10
- from render_machine.actions.run_unit_tests import RunUnitTests
11
- from render_machine.render_context import RenderContext
12
- from render_machine.render_types import ScriptExecutionHistory, UnitTestsRunningContext
13
-
14
-
15
- class FakeRenderContext(SimpleNamespace):
16
- """A render context with just the attributes FixUnitTests uses, and the real session lookup."""
17
-
18
- unit_tests_agent_session = RenderContext.unit_tests_agent_session
19
-
20
-
21
- class FakeAPI:
22
- def __init__(self, responses):
23
- self.responses = list(responses)
24
- self.calls = []
25
-
26
- def agent_start(self, task_type, task_params, frid, module_name, run_state):
27
- self.calls.append(("start", task_type, task_params, frid, module_name))
28
- return self.responses.pop(0)
29
-
30
- def agent_continue(self, session_id, tool_results, frid, module_name, run_state):
31
- self.calls.append(("continue", session_id, tool_results, frid, module_name))
32
- response = self.responses.pop(0)
33
- if isinstance(response, Exception):
34
- raise response
35
- return response
36
-
37
-
38
- def _tool_calls(*calls):
39
- return {"session_id": "s1", "status": "tool_calls", "calls": list(calls)}
40
-
41
-
42
- @pytest.fixture
43
- def render_context(tmp_path, monkeypatch):
44
- build = tmp_path / "build"
45
- build.mkdir()
46
- (build / "a.py").write_text("x = 1\n")
47
- monkeypatch.chdir(tmp_path)
48
- plain_source_tree = {"spec": True}
49
- specifications = {
50
- plain_spec.DEFINITIONS: ["- :Foo: is a thing."],
51
- plain_spec.NON_FUNCTIONAL_REQUIREMENTS: ["- Python 3.11."],
52
- plain_spec.FUNCTIONAL_REQUIREMENTS: ["- Old feature.", "- New feature."],
53
- }
54
- monkeypatch.setattr(plain_spec, "get_specifications_for_frid", lambda tree, frid: (specifications, None))
55
- return FakeRenderContext(
56
- codeplain_api=None,
57
- build_folder=str(build),
58
- module_name="m",
59
- run_state=object(),
60
- plain_source_tree=plain_source_tree,
61
- unittests_script=None,
62
- test_script_timeout=None,
63
- stop_event=None,
64
- frid_context=SimpleNamespace(frid="2", linked_resources={"schema.json": "{}"}, changed_files={"a.py"}),
65
- unit_tests_running_context=UnitTestsRunningContext(fix_attempts=1),
66
- conformance_tests_running_context=None,
67
- script_execution_history=ScriptExecutionHistory(),
68
- get_required_modules_functionalities=lambda: {"base": ["- Base feature."]},
69
- )
70
-
71
-
72
- def test_first_attempt_starts_session_runs_tools_and_stops_at_submit_fix(render_context):
73
- api = FakeAPI(
74
- [
75
- _tool_calls({"id": "c1", "name": "read_file", "args": {"file_path": "a.py"}}),
76
- _tool_calls(
77
- {"id": "c2", "name": "edit_file", "args": {"file_path": "a.py", "search": "x = 1", "replace": "x = 2"}},
78
- {"id": "c3", "name": "submit_fix", "args": {"root_cause": "off by one", "changes_made": "x = 2"}},
79
- ),
80
- ]
81
- )
82
- render_context.codeplain_api = api
83
-
84
- outcome, payload = FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED test_a"})
85
-
86
- assert (outcome, payload) == (FixUnitTests.SUCCESSFUL_OUTCOME, None)
87
- kind, task_type, task_params, frid, module_name = api.calls[0]
88
- assert (kind, task_type, frid, module_name) == ("start", "fix_unit_tests", "2", "m")
89
- assert task_params["unittests_issue"] == "FAILED test_a"
90
- assert task_params["definitions"] == "- :Foo: is a thing."
91
- assert task_params["linked_resources"] == {"schema.json": "{}"}
92
- assert (
93
- "### Module: base (Already Implemented, for context)\n- Base feature." in task_params["functional_requirements"]
94
- )
95
- assert "### Module: m (Already Implemented, for context)\n- Old feature." in task_params["functional_requirements"]
96
- assert task_params["functional_requirements"].endswith(
97
- "### Module: m (Currently Being Implemented)\n- New feature."
98
- )
99
-
100
- # the read_file result went back to the server; the edit was applied locally
101
- assert api.calls[1][0] == "continue" and api.calls[1][2][0]["call_id"] == "c1"
102
- assert "1: x = 1" in api.calls[1][2][0]["output"]
103
- assert open(render_context.build_folder + "/a.py").read() == "x = 2\n"
104
-
105
- context = render_context.unit_tests_running_context
106
- session = context.agent_session
107
- assert session.session_id == "s1"
108
- assert session.pending_submit_call_id == "c3"
109
- assert [r["call_id"] for r in session.pending_tool_results] == ["c2"]
110
- assert context.changed_files == {"a.py"}
111
- assert "conformance_tests_fixes" not in task_params
112
-
113
-
114
- def test_second_attempt_continues_session_answering_submit_fix(render_context):
115
- context = render_context.unit_tests_running_context
116
- context.agent_used_in_this_loop = True
117
- session = context.agent_session
118
- session.session_id = "s1"
119
- session.pending_submit_call_id = "c3"
120
- session.pending_tool_results = [{"call_id": "c2", "output": "Edited"}]
121
- api = FakeAPI([_tool_calls({"id": "c4", "name": "submit_fix", "args": {"changes_made": "again"}})])
122
- render_context.codeplain_api = api
123
-
124
- outcome, _ = FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED test_b"})
125
-
126
- assert outcome == FixUnitTests.SUCCESSFUL_OUTCOME
127
- kind, session_id, tool_results, frid, module_name = api.calls[0]
128
- assert (kind, session_id, frid, module_name) == ("continue", "s1", "2", "m")
129
- assert tool_results[0] == {"call_id": "c2", "output": "Edited"}
130
- assert tool_results[1]["call_id"] == "c3" and "still fail" in tool_results[1]["output"]
131
- assert tool_results[1]["test_output"] == "FAILED test_b"
132
- assert "conformance_tests_fixes" not in tool_results[1]
133
- assert session.session_id == "s1" and session.pending_submit_call_id == "c4"
134
- assert session.pending_tool_results == []
135
-
136
-
137
- @pytest.mark.parametrize(
138
- "final_response",
139
- [
140
- {"session_id": "s1", "status": "completed", "result": "done"},
141
- {"session_id": "s1", "status": "failed", "error": "x"},
142
- ],
143
- )
144
- def test_session_ending_without_submission_resets_the_session(render_context, final_response):
145
- render_context.codeplain_api = FakeAPI([final_response])
146
- outcome, _ = FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED"})
147
- assert outcome == FixUnitTests.SUCCESSFUL_OUTCOME
148
- session = render_context.unit_tests_running_context.agent_session
149
- assert session.session_id is None and session.pending_submit_call_id is None
150
-
151
-
152
- def test_turn_cap_per_attempt_resets_the_session(render_context):
153
- call = {"id": "c", "name": "ls_files", "args": {}}
154
- render_context.codeplain_api = FakeAPI([_tool_calls(call)] * (MAX_AGENT_TURNS_PER_ATTEMPT + 1))
155
- FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED"})
156
- assert len(render_context.codeplain_api.calls) == MAX_AGENT_TURNS_PER_ATTEMPT + 1
157
- assert render_context.unit_tests_running_context.agent_session.session_id is None
158
-
159
-
160
- def test_missing_issue_is_an_internal_error(render_context):
161
- from plain2code_exceptions import InternalClientError
162
-
163
- with pytest.raises(InternalClientError):
164
- FixUnitTests().execute(render_context, {})
165
-
166
-
167
- def test_first_turn_is_seeded_with_file_tree_relevant_files_and_log_path(render_context, tmp_path):
168
- (tmp_path / "build" / "tests").mkdir()
169
- (tmp_path / "build" / "tests" / "test_a.py").write_text("assert True\n")
170
- log = tmp_path / "unit.log"
171
- log.write_text("full log\nCaused by: boom\n")
172
- render_context.script_execution_history.latest_unit_test_output_path = str(log)
173
- api = FakeAPI([{"session_id": "s1", "status": "completed", "result": "done"}])
174
- render_context.codeplain_api = api
175
-
176
- FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED"})
177
-
178
- task_params = api.calls[0][2]
179
- assert task_params["file_tree"].split("\n") == ["a.py", "tests/test_a.py"]
180
- assert task_params["relevant_files"] == {"a.py": "x = 1\n"}
181
- assert task_params["unittests_log_path"] == str(log)
182
- assert "previous_session_id" not in task_params
183
- # the agent may grep the full log although it is outside the build folder and project root
184
- assert str(log) in render_context.unit_tests_running_context.agent_session.readable_log_paths
185
-
186
-
187
- def test_relevant_files_stay_within_budget(tmp_path):
188
- (tmp_path / "small.py").write_text("s")
189
- (tmp_path / "big.py").write_text("b" * MAX_RELEVANT_FILES_CHARS)
190
- assert FixUnitTests._relevant_files(str(tmp_path), {"small.py", "big.py", "deleted.py"}) == {"small.py": "s"}
191
-
192
-
193
- def test_new_session_after_abandoned_one_references_it(render_context):
194
- call = {"id": "c", "name": "ls_files", "args": {}}
195
- render_context.codeplain_api = FakeAPI([_tool_calls(call)] * (MAX_AGENT_TURNS_PER_ATTEMPT + 1))
196
- FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED"})
197
- assert render_context.unit_tests_running_context.agent_session.previous_session_id == "s1"
198
-
199
- api = FakeAPI([{"session_id": "s2", "status": "completed", "result": "done"}])
200
- render_context.codeplain_api = api
201
- FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED again"})
202
- assert api.calls[0][0] == "start" and api.calls[0][2]["previous_session_id"] == "s1"
203
-
204
-
205
- def test_run_unit_tests_action_skips_the_suite_after_a_verified_agent_run(render_context, monkeypatch):
206
- import render_machine.render_utils as render_utils
207
-
208
- context = render_context.unit_tests_running_context
209
- context.verified_passing, context.verified_passing_log_path = True, "/logs/pass.log"
210
- monkeypatch.setattr(render_utils, "execute_script", lambda *a, **k: pytest.fail("suite must not run"))
211
-
212
- assert RunUnitTests().execute(render_context, None) == (RunUnitTests.SUCCESSFUL_OUTCOME, None)
213
- assert context.verified_passing is False
214
- assert render_context.script_execution_history.latest_unit_test_output_path == "/logs/pass.log"
215
-
216
-
217
- def test_expired_session_is_replaced_by_a_new_one(render_context):
218
- context = render_context.unit_tests_running_context
219
- context.agent_used_in_this_loop = True
220
- session = context.agent_session
221
- session.session_id, session.pending_submit_call_id = "expired", "c3"
222
- not_found = requests.exceptions.HTTPError(response=SimpleNamespace(status_code=404))
223
- api = FakeAPI([not_found, {"session_id": "s2", "status": "completed", "result": "done"}])
224
- render_context.codeplain_api = api
225
-
226
- outcome, _ = FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED test_b"})
227
-
228
- assert outcome == FixUnitTests.SUCCESSFUL_OUTCOME
229
- assert [call[0] for call in api.calls] == ["continue", "start"]
230
- assert api.calls[1][2]["unittests_issue"] == "FAILED test_b"
231
-
232
-
233
- def test_other_http_errors_on_continue_are_not_swallowed(render_context):
234
- context = render_context.unit_tests_running_context
235
- context.agent_used_in_this_loop = True
236
- context.agent_session.session_id, context.agent_session.pending_submit_call_id = "s1", "c3"
237
- render_context.codeplain_api = FakeAPI([requests.exceptions.HTTPError(response=SimpleNamespace(status_code=500))])
238
- with pytest.raises(requests.exceptions.HTTPError):
239
- FixUnitTests().execute(render_context, {"previous_unittests_issue": "FAILED"})