codeplain 0.3.11.dev20__py3-none-any.whl → 0.3.11.dev22__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {codeplain-0.3.11.dev20.dist-info → codeplain-0.3.11.dev22.dist-info}/METADATA +1 -1
- {codeplain-0.3.11.dev20.dist-info → codeplain-0.3.11.dev22.dist-info}/RECORD +14 -17
- codeplain_REST_api.py +32 -26
- render_machine/actions/fix_unit_tests.py +47 -224
- render_machine/actions/run_unit_tests.py +0 -9
- render_machine/render_context.py +0 -9
- render_machine/render_types.py +0 -45
- tests/test_fix_unit_tests_conformance_context.py +25 -120
- tests/test_tui_components.py +60 -1
- tui/components.py +25 -2
- tui/plain2code_tui.py +9 -0
- render_machine/agent_tools.py +0 -270
- tests/test_agent_tools.py +0 -166
- tests/test_fix_unit_tests_action.py +0 -239
- {codeplain-0.3.11.dev20.dist-info → codeplain-0.3.11.dev22.dist-info}/WHEEL +0 -0
- {codeplain-0.3.11.dev20.dist-info → codeplain-0.3.11.dev22.dist-info}/entry_points.txt +0 -0
- {codeplain-0.3.11.dev20.dist-info → codeplain-0.3.11.dev22.dist-info}/licenses/LICENSE +0 -0
|
@@ -12,44 +12,25 @@ import pytest
|
|
|
12
12
|
|
|
13
13
|
import plain_spec
|
|
14
14
|
from memory_management import MemoryManager
|
|
15
|
+
from render_machine.actions import fix_unit_tests as fix_unit_tests_module
|
|
15
16
|
from render_machine.actions.fix_conformance_test import FixConformanceTest
|
|
16
17
|
from render_machine.actions.fix_unit_tests import FixUnitTests
|
|
17
18
|
from render_machine.implementation_code_helpers import ImplementationCodeHelpers
|
|
18
|
-
from render_machine.
|
|
19
|
-
from render_machine.render_types import ConformanceTestsRunningContext, ScriptExecutionHistory, UnitTestsRunningContext
|
|
19
|
+
from render_machine.render_types import ConformanceTestsRunningContext, UnitTestsRunningContext
|
|
20
20
|
|
|
21
21
|
|
|
22
22
|
class FakeCodeplainAPI:
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
def __init__(self, conformance_fix_response=None):
|
|
23
|
+
def __init__(self, conformance_fix_response=None, unittests_fix_response=None):
|
|
26
24
|
self.conformance_fix_response = conformance_fix_response
|
|
27
|
-
self.
|
|
28
|
-
self.
|
|
25
|
+
self.unittests_fix_response = unittests_fix_response or {}
|
|
26
|
+
self.unittests_fix_calls = []
|
|
29
27
|
|
|
30
28
|
def fix_conformance_tests_issue(self, *args, **kwargs):
|
|
31
29
|
return self.conformance_fix_response
|
|
32
30
|
|
|
33
|
-
def
|
|
34
|
-
self.
|
|
35
|
-
self.
|
|
36
|
-
return self._submit(f"s{self.sessions_started}")
|
|
37
|
-
|
|
38
|
-
def agent_continue(self, session_id, tool_results, frid, module_name, run_state):
|
|
39
|
-
self.agent_calls.append(("continue", session_id, tool_results))
|
|
40
|
-
return self._submit(session_id)
|
|
41
|
-
|
|
42
|
-
def _submit(self, session_id):
|
|
43
|
-
call_id = f"submit-{len(self.agent_calls)}"
|
|
44
|
-
return {
|
|
45
|
-
"session_id": session_id,
|
|
46
|
-
"status": "tool_calls",
|
|
47
|
-
"calls": [{"id": call_id, "name": "submit_fix", "args": {"changes_made": "fixed"}}],
|
|
48
|
-
}
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
class FakeRenderContext(SimpleNamespace):
|
|
52
|
-
unit_tests_agent_session = RenderContext.unit_tests_agent_session
|
|
31
|
+
def fix_unittests_issue(self, *args, **kwargs):
|
|
32
|
+
self.unittests_fix_calls.append(kwargs)
|
|
33
|
+
return self.unittests_fix_response
|
|
53
34
|
|
|
54
35
|
|
|
55
36
|
class FakeConformanceTests:
|
|
@@ -78,11 +59,7 @@ def memory_folder():
|
|
|
78
59
|
def isolate_from_git_and_console(monkeypatch):
|
|
79
60
|
monkeypatch.setattr(ImplementationCodeHelpers, "get_code_diff", staticmethod(lambda *args: {}))
|
|
80
61
|
monkeypatch.setattr(plain_spec, "collect_linked_resources", lambda *args: None)
|
|
81
|
-
monkeypatch.setattr(
|
|
82
|
-
plain_spec,
|
|
83
|
-
"get_specifications_for_frid",
|
|
84
|
-
lambda tree, frid: ({plain_spec.FUNCTIONAL_REQUIREMENTS: ["- Add numbers."]}, None),
|
|
85
|
-
)
|
|
62
|
+
monkeypatch.setattr(fix_unit_tests_module.render_utils, "print_inputs", lambda *args: None)
|
|
86
63
|
|
|
87
64
|
|
|
88
65
|
def make_conformance_context():
|
|
@@ -99,7 +76,7 @@ def make_conformance_context():
|
|
|
99
76
|
|
|
100
77
|
|
|
101
78
|
def make_render_context(api, build_folder, memory_folder, conformance_tests_running_context):
|
|
102
|
-
return
|
|
79
|
+
return SimpleNamespace(
|
|
103
80
|
codeplain_api=api,
|
|
104
81
|
build_folder=build_folder,
|
|
105
82
|
memory_manager=MemoryManager(api, memory_folder),
|
|
@@ -109,11 +86,9 @@ def make_render_context(api, build_folder, memory_folder, conformance_tests_runn
|
|
|
109
86
|
plain_source_tree={},
|
|
110
87
|
module_name="mod",
|
|
111
88
|
required_modules=None,
|
|
112
|
-
frid_context=SimpleNamespace(frid="1", linked_resources={}
|
|
89
|
+
frid_context=SimpleNamespace(frid="1", linked_resources={}),
|
|
113
90
|
get_required_modules_functionalities=lambda: {},
|
|
114
91
|
run_state=SimpleNamespace(render_id="test-render-id", unittest_batch_id=1),
|
|
115
|
-
unittests_script=None,
|
|
116
|
-
script_execution_history=ScriptExecutionHistory(),
|
|
117
92
|
)
|
|
118
93
|
|
|
119
94
|
|
|
@@ -216,104 +191,34 @@ def test_implementation_fix_with_no_files_is_not_remembered(build_folder, memory
|
|
|
216
191
|
assert ctx.implementation_code_fixes == []
|
|
217
192
|
|
|
218
193
|
|
|
219
|
-
def run_unit_tests_fix(render_context
|
|
220
|
-
return FixUnitTests().execute(render_context, {"previous_unittests_issue":
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
def start_new_unit_test_loop(render_context):
|
|
224
|
-
"""What RenderContext.start_unittests_processing does when the unit tests are run again."""
|
|
225
|
-
render_context.unit_tests_running_context = UnitTestsRunningContext(fix_attempts=0)
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
def conformance_fix(number):
|
|
229
|
-
return {
|
|
230
|
-
"hypothesis": f"hypothesis {number}",
|
|
231
|
-
"approach": f"approach {number}",
|
|
232
|
-
"code_diff": {"app.py": f"+{number}"},
|
|
233
|
-
}
|
|
194
|
+
def run_unit_tests_fix(render_context):
|
|
195
|
+
return FixUnitTests().execute(render_context, {"previous_unittests_issue": "1 failed"})
|
|
234
196
|
|
|
235
197
|
|
|
236
|
-
def
|
|
237
|
-
api = FakeCodeplainAPI()
|
|
198
|
+
def test_unit_tests_fix_forwards_conformance_fixes(build_folder, memory_folder):
|
|
199
|
+
api = FakeCodeplainAPI(unittests_fix_response={"test_app.py": "def test_add(): pass\n"})
|
|
238
200
|
ctx = make_conformance_context()
|
|
239
|
-
ctx.implementation_code_fixes.append(
|
|
201
|
+
ctx.implementation_code_fixes.append(
|
|
202
|
+
{"hypothesis": "off by one", "approach": "add one", "code_diff": {"app.py": "+ return a + b + 1"}}
|
|
203
|
+
)
|
|
240
204
|
render_context = make_render_context(api, build_folder, memory_folder, ctx)
|
|
241
205
|
|
|
242
206
|
outcome, _ = run_unit_tests_fix(render_context)
|
|
243
207
|
|
|
244
208
|
assert outcome == FixUnitTests.SUCCESSFUL_OUTCOME
|
|
245
|
-
|
|
246
|
-
assert
|
|
247
|
-
assert task_params["conformance_tests_fixes"] == ctx.implementation_code_fixes
|
|
209
|
+
assert len(api.unittests_fix_calls) == 1
|
|
210
|
+
assert api.unittests_fix_calls[0]["conformance_tests_fixes"] == ctx.implementation_code_fixes
|
|
248
211
|
# The forwarded list is a copy, so later conformance fixes do not mutate what was sent.
|
|
249
|
-
assert
|
|
250
|
-
# The file the conformance fix changed is seeded although the FRID did not change it.
|
|
251
|
-
assert "app.py" in task_params["relevant_files"]
|
|
252
|
-
assert ctx.unit_tests_agent_session.session_id == "s1"
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
def test_next_unit_test_loop_of_conformance_phase_continues_the_session_with_only_new_fixes(
|
|
256
|
-
build_folder, memory_folder
|
|
257
|
-
):
|
|
258
|
-
api = FakeCodeplainAPI()
|
|
259
|
-
ctx = make_conformance_context()
|
|
260
|
-
ctx.implementation_code_fixes.append(conformance_fix(1))
|
|
261
|
-
render_context = make_render_context(api, build_folder, memory_folder, ctx)
|
|
262
|
-
run_unit_tests_fix(render_context)
|
|
263
|
-
|
|
264
|
-
# The fix was accepted; the conformance tests fixer changes the code again and the unit tests fail again.
|
|
265
|
-
ctx.implementation_code_fixes.append(conformance_fix(2))
|
|
266
|
-
start_new_unit_test_loop(render_context)
|
|
267
|
-
run_unit_tests_fix(render_context, issue="2 failed")
|
|
212
|
+
assert api.unittests_fix_calls[0]["conformance_tests_fixes"] is not ctx.implementation_code_fixes
|
|
268
213
|
|
|
269
|
-
assert api.sessions_started == 1
|
|
270
|
-
kind, session_id, tool_results = api.agent_calls[1]
|
|
271
|
-
assert (kind, session_id) == ("continue", "s1")
|
|
272
|
-
submit_answer = tool_results[-1]
|
|
273
|
-
assert submit_answer["call_id"] == "submit-1"
|
|
274
|
-
assert submit_answer["output"].startswith("Your fix was accepted: the unit tests passed.")
|
|
275
|
-
assert "Conformance Tests Fix below" in submit_answer["output"]
|
|
276
|
-
assert submit_answer["conformance_tests_fixes"] == [conformance_fix(2)]
|
|
277
|
-
assert submit_answer["test_output"] == "2 failed"
|
|
278
214
|
|
|
279
|
-
|
|
280
|
-
def test_retry_within_a_unit_test_loop_says_the_fix_did_not_work(build_folder, memory_folder):
|
|
281
|
-
api = FakeCodeplainAPI()
|
|
282
|
-
ctx = make_conformance_context()
|
|
283
|
-
ctx.implementation_code_fixes.append(conformance_fix(1))
|
|
284
|
-
render_context = make_render_context(api, build_folder, memory_folder, ctx)
|
|
285
|
-
run_unit_tests_fix(render_context)
|
|
286
|
-
|
|
287
|
-
run_unit_tests_fix(render_context, issue="still failing")
|
|
288
|
-
|
|
289
|
-
submit_answer = api.agent_calls[1][2][-1]
|
|
290
|
-
assert "still fail" in submit_answer["output"]
|
|
291
|
-
# Fix 1 was already shown to the session, so it is not sent again.
|
|
292
|
-
assert "conformance_tests_fixes" not in submit_answer
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
def test_new_conformance_phase_starts_a_new_session(build_folder, memory_folder):
|
|
296
|
-
api = FakeCodeplainAPI()
|
|
297
|
-
render_context = make_render_context(api, build_folder, memory_folder, make_conformance_context())
|
|
298
|
-
run_unit_tests_fix(render_context)
|
|
299
|
-
|
|
300
|
-
# E.g. the functionality is re-rendered from scratch: the conformance tests running context is recreated.
|
|
301
|
-
render_context.conformance_tests_running_context = make_conformance_context()
|
|
302
|
-
start_new_unit_test_loop(render_context)
|
|
303
|
-
run_unit_tests_fix(render_context)
|
|
304
|
-
|
|
305
|
-
assert [call[0] for call in api.agent_calls] == ["start", "start"]
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
def test_unit_tests_outside_conformance_phase_get_no_fixes_and_a_session_per_loop(build_folder, memory_folder):
|
|
215
|
+
def test_unit_tests_fix_outside_conformance_phase_sends_no_fixes(build_folder, memory_folder):
|
|
309
216
|
api = FakeCodeplainAPI()
|
|
310
217
|
render_context = make_render_context(api, build_folder, memory_folder, None)
|
|
311
|
-
|
|
312
|
-
start_new_unit_test_loop(render_context)
|
|
218
|
+
|
|
313
219
|
run_unit_tests_fix(render_context)
|
|
314
220
|
|
|
315
|
-
assert [
|
|
316
|
-
assert all("conformance_tests_fixes" not in call[1] for call in api.agent_calls)
|
|
221
|
+
assert api.unittests_fix_calls[0]["conformance_tests_fixes"] is None
|
|
317
222
|
|
|
318
223
|
|
|
319
224
|
def test_unit_tests_fix_in_conformance_phase_without_implementation_changes_sends_no_fixes(build_folder, memory_folder):
|
|
@@ -322,4 +227,4 @@ def test_unit_tests_fix_in_conformance_phase_without_implementation_changes_send
|
|
|
322
227
|
|
|
323
228
|
run_unit_tests_fix(render_context)
|
|
324
229
|
|
|
325
|
-
assert
|
|
230
|
+
assert api.unittests_fix_calls[0]["conformance_tests_fixes"] is None
|
tests/test_tui_components.py
CHANGED
|
@@ -6,7 +6,7 @@ from textual.widgets import Static
|
|
|
6
6
|
|
|
7
7
|
from event_bus import EventBus
|
|
8
8
|
from plain2code_state import RunState
|
|
9
|
-
from tui.components import ProgressItem, SubstateLine, TUIComponents
|
|
9
|
+
from tui.components import FRIDProgress, ProgressItem, RenderingInfoBox, SubstateLine, TUIComponents
|
|
10
10
|
from tui.models import Substate
|
|
11
11
|
from tui.plain2code_tui import Plain2CodeTUI
|
|
12
12
|
from tui.widget_helpers import display_error_message, display_success_message, update_progress_item_substates
|
|
@@ -86,3 +86,62 @@ def test_status_messages_keep_brackets():
|
|
|
86
86
|
assert "[Errno 2] No such file: run_tests.sh [b" in str(status.content)
|
|
87
87
|
|
|
88
88
|
asyncio.run(scenario())
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
# A functionality quote as the render TUI receives it: several lines, square brackets included.
|
|
92
|
+
MULTILINE_FUNCTIONALITY_TEXT = (
|
|
93
|
+
"Functionality 3: :User: should be able to add a :Task: [optional]\n"
|
|
94
|
+
" - The :Task: must have non-empty content.\n"
|
|
95
|
+
" - The :Task: is appended to the end of the list."
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def test_functionality_text_collapses_and_expands_with_ctrl_o():
|
|
100
|
+
async def scenario():
|
|
101
|
+
event_bus = EventBus()
|
|
102
|
+
run_state = RunState(spec_filename="x.plain")
|
|
103
|
+
app = _make_app(run_state, event_bus)
|
|
104
|
+
async with app.run_test() as pilot:
|
|
105
|
+
frid_progress = app.query_one(f"#{TUIComponents.FRID_PROGRESS.value}", FRIDProgress)
|
|
106
|
+
info_box = frid_progress.query_one(RenderingInfoBox)
|
|
107
|
+
info_box.update_functionality(MULTILINE_FUNCTIONALITY_TEXT)
|
|
108
|
+
await pilot.pause()
|
|
109
|
+
|
|
110
|
+
widget = info_box.functionality_widget
|
|
111
|
+
assert widget is not None
|
|
112
|
+
|
|
113
|
+
# Collapsed by default: first line only, with the expand hint.
|
|
114
|
+
collapsed = str(widget.content)
|
|
115
|
+
assert "Functionality 3: :User: should be able to add a :Task: [optional]" in collapsed
|
|
116
|
+
assert "non-empty content" not in collapsed
|
|
117
|
+
assert RenderingInfoBox.EXPAND_HINT in collapsed
|
|
118
|
+
|
|
119
|
+
await pilot.press("ctrl+o")
|
|
120
|
+
await pilot.pause()
|
|
121
|
+
expanded = str(widget.content)
|
|
122
|
+
assert "non-empty content" in expanded
|
|
123
|
+
assert "appended to the end of the list" in expanded
|
|
124
|
+
assert RenderingInfoBox.COLLAPSE_HINT in expanded
|
|
125
|
+
|
|
126
|
+
await pilot.press("ctrl+o")
|
|
127
|
+
await pilot.pause()
|
|
128
|
+
assert str(widget.content) == collapsed
|
|
129
|
+
|
|
130
|
+
asyncio.run(scenario())
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def test_single_line_functionality_text_has_no_hint():
|
|
134
|
+
async def scenario():
|
|
135
|
+
event_bus = EventBus()
|
|
136
|
+
run_state = RunState(spec_filename="x.plain")
|
|
137
|
+
app = _make_app(run_state, event_bus)
|
|
138
|
+
async with app.run_test() as pilot:
|
|
139
|
+
info_box = app.query_one(f"#{TUIComponents.FRID_PROGRESS.value}", FRIDProgress).query_one(RenderingInfoBox)
|
|
140
|
+
info_box.update_functionality("Functionality 1: :User: should be able to add a :Task:")
|
|
141
|
+
await pilot.pause()
|
|
142
|
+
|
|
143
|
+
widget = info_box.functionality_widget
|
|
144
|
+
assert widget is not None
|
|
145
|
+
assert str(widget.content) == "Functionality 1: :User: should be able to add a :Task:"
|
|
146
|
+
|
|
147
|
+
asyncio.run(scenario())
|
tui/components.py
CHANGED
|
@@ -339,10 +339,14 @@ class ProgressItem(Vertical):
|
|
|
339
339
|
class RenderingInfoBox(Vertical):
|
|
340
340
|
"""Responsive container for module and functionality information."""
|
|
341
341
|
|
|
342
|
+
EXPAND_HINT = "(ctrl+o to expand)"
|
|
343
|
+
COLLAPSE_HINT = "(ctrl+o to collapse)"
|
|
344
|
+
|
|
342
345
|
def __init__(self, **kwargs):
|
|
343
346
|
super().__init__(**kwargs)
|
|
344
347
|
self.module_text = ""
|
|
345
348
|
self.functionality_text = ""
|
|
349
|
+
self.functionality_expanded = False
|
|
346
350
|
self.module_widget: Static | None = None
|
|
347
351
|
self.functionality_widget: Static | None = None
|
|
348
352
|
|
|
@@ -356,12 +360,31 @@ class RenderingInfoBox(Vertical):
|
|
|
356
360
|
self.functionality_text = text
|
|
357
361
|
self._refresh_content()
|
|
358
362
|
|
|
363
|
+
def toggle_functionality(self) -> None:
|
|
364
|
+
"""Expand or collapse the functionality text."""
|
|
365
|
+
self.functionality_expanded = not self.functionality_expanded
|
|
366
|
+
self._refresh_content()
|
|
367
|
+
|
|
368
|
+
def _format_functionality(self) -> Content:
|
|
369
|
+
"""Build the functionality line, collapsed to its first line unless expanded.
|
|
370
|
+
|
|
371
|
+
The functionality text comes from the spec and may contain square brackets, so it
|
|
372
|
+
is assembled as Content and never goes through the markup parser.
|
|
373
|
+
"""
|
|
374
|
+
text = self.functionality_text or ""
|
|
375
|
+
lines = text.splitlines()
|
|
376
|
+
if len(lines) <= 1:
|
|
377
|
+
return Content(text)
|
|
378
|
+
if self.functionality_expanded:
|
|
379
|
+
return Content.assemble(f"{text}\n", (self.COLLAPSE_HINT, "#888888"))
|
|
380
|
+
return Content.assemble(f"{lines[0]} \u2026 ", (self.EXPAND_HINT, "#888888"))
|
|
381
|
+
|
|
359
382
|
def _refresh_content(self) -> None:
|
|
360
383
|
"""Refresh text inside the box."""
|
|
361
384
|
if self.module_widget is not None:
|
|
362
385
|
self.module_widget.update(self.module_text or "")
|
|
363
386
|
if self.functionality_widget is not None:
|
|
364
|
-
self.functionality_widget.update(self.
|
|
387
|
+
self.functionality_widget.update(self._format_functionality())
|
|
365
388
|
|
|
366
389
|
def on_mount(self) -> None:
|
|
367
390
|
"""Initialize default labels on mount."""
|
|
@@ -372,7 +395,7 @@ class RenderingInfoBox(Vertical):
|
|
|
372
395
|
|
|
373
396
|
def compose(self):
|
|
374
397
|
self.module_widget = Static(self.module_text, classes="rendering-info-row", markup=False)
|
|
375
|
-
self.functionality_widget = Static(self.
|
|
398
|
+
self.functionality_widget = Static(self._format_functionality(), classes="rendering-info-row", markup=False)
|
|
376
399
|
yield Static("module status", classes="rendering-info-title")
|
|
377
400
|
with Vertical(classes="rendering-info-box"):
|
|
378
401
|
yield self.module_widget
|
tui/plain2code_tui.py
CHANGED
|
@@ -65,6 +65,7 @@ class Plain2CodeTUI(App):
|
|
|
65
65
|
Binding("ctrl+d", "quit", "Quit", show=False),
|
|
66
66
|
Binding("enter", "enter_exit", "Exit", show=False),
|
|
67
67
|
Binding("ctrl+p", "pause", "Pause", show=False, priority=True),
|
|
68
|
+
Binding("ctrl+o", "toggle_functionality", "Expand/Collapse", show=False),
|
|
68
69
|
("ctrl+l", "toggle_logs", "Toggle Logs"),
|
|
69
70
|
]
|
|
70
71
|
|
|
@@ -231,6 +232,14 @@ class Plain2CodeTUI(App):
|
|
|
231
232
|
self.run_state.render_time_accumulated,
|
|
232
233
|
)
|
|
233
234
|
|
|
235
|
+
def action_toggle_functionality(self) -> None:
|
|
236
|
+
"""Expand or collapse the functionality text in the rendering info box."""
|
|
237
|
+
try:
|
|
238
|
+
frid_progress = self.query_one(f"#{TUIComponents.FRID_PROGRESS.value}", FRIDProgress)
|
|
239
|
+
frid_progress.query_one(RenderingInfoBox).toggle_functionality()
|
|
240
|
+
except NoMatches:
|
|
241
|
+
pass
|
|
242
|
+
|
|
234
243
|
def action_toggle_logs(self) -> None:
|
|
235
244
|
"""Toggle between dashboard and log view."""
|
|
236
245
|
switcher = self.query_one(f"#{TUIComponents.CONTENT_SWITCHER.value}", ContentSwitcher)
|
render_machine/agent_tools.py
DELETED
|
@@ -1,270 +0,0 @@
|
|
|
1
|
-
"""Client-side implementations of the tools a server-side agent can call.
|
|
2
|
-
|
|
3
|
-
The server declares the tools to the LLM (codeplain-api: src/agent/tools.py) and forwards
|
|
4
|
-
the model's calls; this module executes them against the local build folder and returns
|
|
5
|
-
plain-text results. Relative paths resolve against the build folder. Reads are allowed only in the
|
|
6
|
-
build folder and the full test logs the agent was pointed to; writes only inside the build folder.
|
|
7
|
-
"""
|
|
8
|
-
|
|
9
|
-
import glob
|
|
10
|
-
import json
|
|
11
|
-
import os
|
|
12
|
-
import subprocess
|
|
13
|
-
import tempfile
|
|
14
|
-
from typing import Callable
|
|
15
|
-
|
|
16
|
-
from plain2code_console import console
|
|
17
|
-
from render_machine import render_utils
|
|
18
|
-
from render_machine.render_context import RenderContext
|
|
19
|
-
|
|
20
|
-
DEFAULT_READ_LIMIT = 200
|
|
21
|
-
MAX_LINE_CHARS = 10_000
|
|
22
|
-
MAX_OUTPUT_CHARS = 30_000
|
|
23
|
-
EDIT_SNIPPET_CONTEXT_LINES = 3
|
|
24
|
-
MAX_EDIT_SNIPPET_LINES = 60
|
|
25
|
-
MAX_GREP_CONTEXT_LINES = 20
|
|
26
|
-
# Tools without side effects; repeating one while no file changed returns a pointer instead.
|
|
27
|
-
READ_ONLY_TOOLS = ("read_file", "grep", "ls_files")
|
|
28
|
-
GREP_EXCLUDED_DIRS = (".git", "__pycache__", "node_modules", ".venv", "target", "dist", "build")
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
def _build_folder(render_context: RenderContext) -> str:
|
|
32
|
-
return os.path.normpath(os.path.abspath(render_context.build_folder))
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
def _resolve(file_path: str, render_context: RenderContext) -> str:
|
|
36
|
-
if os.path.isabs(file_path):
|
|
37
|
-
return os.path.normpath(file_path)
|
|
38
|
-
return os.path.normpath(os.path.join(_build_folder(render_context), file_path))
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
def _within(path: str, folder: str) -> bool:
|
|
42
|
-
return path == folder or path.startswith(folder + os.sep)
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
def _readable(path: str, render_context: RenderContext) -> bool:
|
|
46
|
-
return (
|
|
47
|
-
_within(path, _build_folder(render_context))
|
|
48
|
-
or path in render_context.unit_tests_agent_session.readable_log_paths
|
|
49
|
-
)
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
def register_log_path(log_path: str, render_context: RenderContext) -> None:
|
|
53
|
-
"""Allow read_file/grep on a full test log the agent is pointed to."""
|
|
54
|
-
render_context.unit_tests_agent_session.readable_log_paths.add(os.path.normpath(os.path.abspath(log_path)))
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
def _writable(path: str, render_context: RenderContext) -> bool:
|
|
58
|
-
return _within(path, _build_folder(render_context))
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
def _bound(text: str) -> str:
|
|
62
|
-
"""Cap very long lines and the total size so one tool result cannot flood the context."""
|
|
63
|
-
lines = [
|
|
64
|
-
line if len(line) <= MAX_LINE_CHARS else line[:MAX_LINE_CHARS] + "... [line truncated]"
|
|
65
|
-
for line in text.split("\n")
|
|
66
|
-
]
|
|
67
|
-
text = "\n".join(lines)
|
|
68
|
-
if len(text) > MAX_OUTPUT_CHARS:
|
|
69
|
-
head, tail = int(MAX_OUTPUT_CHARS * 0.6), int(MAX_OUTPUT_CHARS * 0.4)
|
|
70
|
-
text = text[:head] + f"\n\n... [truncated {len(text) - head - tail:,} chars] ...\n\n" + text[-tail:]
|
|
71
|
-
return text
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
def _track_change(full_path: str, render_context: RenderContext) -> None:
|
|
75
|
-
context = render_context.unit_tests_running_context
|
|
76
|
-
context.changed_files.add(os.path.relpath(full_path, _build_folder(render_context)))
|
|
77
|
-
context.verified_passing, context.verified_passing_log_path = False, None
|
|
78
|
-
context.tool_result_cache.clear()
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
def read_file(args: dict, render_context: RenderContext) -> str:
|
|
82
|
-
full_path = _resolve(args.get("file_path", ""), render_context)
|
|
83
|
-
if not _readable(full_path, render_context):
|
|
84
|
-
return f"Error: read access denied for '{full_path}' (readable: build folder and test logs)."
|
|
85
|
-
if not os.path.isfile(full_path):
|
|
86
|
-
return f"Error: file not found: '{full_path}'."
|
|
87
|
-
with open(full_path, "r", encoding="utf-8", errors="replace") as f:
|
|
88
|
-
lines = f.read().split("\n")
|
|
89
|
-
offset = max(int(args.get("offset") or 1), 1)
|
|
90
|
-
limit = int(args.get("limit") or DEFAULT_READ_LIMIT)
|
|
91
|
-
selected = lines[offset - 1 : offset - 1 + limit]
|
|
92
|
-
if not selected:
|
|
93
|
-
return f"Error: offset {offset} is past the end of the file ({len(lines)} lines)."
|
|
94
|
-
numbered = "\n".join(f"{offset + i}: {line}" for i, line in enumerate(selected))
|
|
95
|
-
last = offset - 1 + len(selected)
|
|
96
|
-
note = (
|
|
97
|
-
f"\n[showing lines {offset}-{last} of {len(lines)}; use offset={last + 1} to continue]"
|
|
98
|
-
if last < len(lines)
|
|
99
|
-
else ""
|
|
100
|
-
)
|
|
101
|
-
return _bound(numbered) + note
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
def grep(args: dict, render_context: RenderContext) -> str:
|
|
105
|
-
pattern = args.get("pattern", "")
|
|
106
|
-
if not pattern:
|
|
107
|
-
return "Error: pattern is required."
|
|
108
|
-
target = _resolve(args.get("file_path") or ".", render_context)
|
|
109
|
-
if not _readable(target, render_context):
|
|
110
|
-
return f"Error: read access denied for '{target}'."
|
|
111
|
-
if not os.path.exists(target):
|
|
112
|
-
return f"Error: path not found: '{target}'."
|
|
113
|
-
# Run from the build folder so matches inside it come back as build-relative paths, which
|
|
114
|
-
# is the form the other tools accept; a test log outside it is passed by absolute path.
|
|
115
|
-
cwd = _build_folder(render_context)
|
|
116
|
-
options = [f"--exclude-dir={d}" for d in GREP_EXCLUDED_DIRS]
|
|
117
|
-
context_lines = min(max(int(args.get("context_lines") or 0), 0), MAX_GREP_CONTEXT_LINES)
|
|
118
|
-
if context_lines:
|
|
119
|
-
options.append(f"-C{context_lines}")
|
|
120
|
-
if args.get("include"):
|
|
121
|
-
options.append(f"--include={args['include']}")
|
|
122
|
-
command = ["grep", "-rnI", *options, "-e", pattern, "--"]
|
|
123
|
-
command.append(os.path.relpath(target, cwd) if _within(target, cwd) else target)
|
|
124
|
-
result = subprocess.run(command, capture_output=True, text=True, cwd=cwd)
|
|
125
|
-
if result.returncode == 1:
|
|
126
|
-
return f"No matches for '{pattern}' in '{target}'."
|
|
127
|
-
if result.returncode != 0:
|
|
128
|
-
return f"Error: grep failed: {result.stderr.strip()}"
|
|
129
|
-
lines = [line[2:] if line.startswith("./") else line for line in result.stdout.rstrip("\n").split("\n")]
|
|
130
|
-
return _bound("\n".join(lines))
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
def ls_files(args: dict, render_context: RenderContext) -> str:
|
|
134
|
-
target = _resolve(args.get("pattern") or ".", render_context)
|
|
135
|
-
if not _readable(target, render_context):
|
|
136
|
-
return f"Error: read access denied for '{target}'."
|
|
137
|
-
if os.path.isdir(target):
|
|
138
|
-
entries = sorted(os.listdir(target))
|
|
139
|
-
listing = [entry + "/" if os.path.isdir(os.path.join(target, entry)) else entry for entry in entries]
|
|
140
|
-
return f"{target}:\n" + ("\n".join(listing) if listing else "(empty)")
|
|
141
|
-
matches = sorted(glob.glob(target, recursive=True))
|
|
142
|
-
return _bound("\n".join(matches)) if matches else f"No files match '{target}'."
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
def edit_file(args: dict, render_context: RenderContext) -> str:
|
|
146
|
-
full_path = _resolve(args.get("file_path", ""), render_context)
|
|
147
|
-
search, replace = args.get("search", ""), args.get("replace", "")
|
|
148
|
-
if not search:
|
|
149
|
-
return "Error: search is required."
|
|
150
|
-
if not _writable(full_path, render_context):
|
|
151
|
-
return f"Error: write access denied for '{full_path}' (writable: build folder only)."
|
|
152
|
-
if not os.path.isfile(full_path):
|
|
153
|
-
return f"Error: file not found: '{full_path}'. Use write_file to create new files."
|
|
154
|
-
with open(full_path, "r", encoding="utf-8") as f:
|
|
155
|
-
content = f.read()
|
|
156
|
-
occurrences = content.count(search)
|
|
157
|
-
if occurrences != 1:
|
|
158
|
-
return (
|
|
159
|
-
f"Error: search text found {occurrences} times in '{full_path}'; it must appear exactly once. "
|
|
160
|
-
"Read the file and use a larger, unique snippet."
|
|
161
|
-
)
|
|
162
|
-
with open(full_path, "w", encoding="utf-8") as f:
|
|
163
|
-
f.write(content.replace(search, replace, 1))
|
|
164
|
-
_track_change(full_path, render_context)
|
|
165
|
-
return f"Edited '{full_path}'. The edited region now reads:\n" + _edit_snippet(content, search, replace)
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
def _edit_snippet(content: str, search: str, replace: str) -> str:
|
|
169
|
-
"""Numbered lines of the replacement plus a little context, so the edit needs no re-read."""
|
|
170
|
-
lines = content.replace(search, replace, 1).split("\n")
|
|
171
|
-
first = content[: content.index(search)].count("\n")
|
|
172
|
-
last = first + replace.count("\n")
|
|
173
|
-
start = max(first - EDIT_SNIPPET_CONTEXT_LINES, 0)
|
|
174
|
-
end = min(last + EDIT_SNIPPET_CONTEXT_LINES + 1, len(lines), start + MAX_EDIT_SNIPPET_LINES)
|
|
175
|
-
snippet = "\n".join(f"{start + i + 1}: {line}" for i, line in enumerate(lines[start:end]))
|
|
176
|
-
return _bound(snippet)
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
def write_file(args: dict, render_context: RenderContext) -> str:
|
|
180
|
-
full_path = _resolve(args.get("file_path", ""), render_context)
|
|
181
|
-
if not _writable(full_path, render_context):
|
|
182
|
-
return f"Error: write access denied for '{full_path}' (writable: build folder only)."
|
|
183
|
-
os.makedirs(os.path.dirname(full_path), exist_ok=True)
|
|
184
|
-
with open(full_path, "w", encoding="utf-8") as f:
|
|
185
|
-
f.write(args.get("content", ""))
|
|
186
|
-
_track_change(full_path, render_context)
|
|
187
|
-
return f"Wrote '{full_path}'."
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
def delete_file(args: dict, render_context: RenderContext) -> str:
|
|
191
|
-
full_path = _resolve(args.get("file_path", ""), render_context)
|
|
192
|
-
if not _writable(full_path, render_context):
|
|
193
|
-
return f"Error: write access denied for '{full_path}' (writable: build folder only)."
|
|
194
|
-
if not os.path.isfile(full_path):
|
|
195
|
-
return f"Error: file not found: '{full_path}'."
|
|
196
|
-
os.remove(full_path)
|
|
197
|
-
_track_change(full_path, render_context)
|
|
198
|
-
return f"Deleted '{full_path}'."
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
def full_log_pointer(log_file_path: str | None) -> str:
|
|
202
|
-
return f" Full log: {log_file_path} (search it with grep, passing that path as file_path)." if log_file_path else ""
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
def run_unit_tests(_args: dict, render_context: RenderContext) -> dict:
|
|
206
|
-
"""Returns a result dict: a short header in `output` and the raw failure output in
|
|
207
|
-
`test_output`, which the server condenses (summarizing it when it is long)."""
|
|
208
|
-
exit_code, output, log_file_path = render_utils.execute_script(
|
|
209
|
-
os.path.normpath(render_context.unittests_script),
|
|
210
|
-
[render_context.build_folder],
|
|
211
|
-
"Unit Tests",
|
|
212
|
-
timeout=render_context.test_script_timeout,
|
|
213
|
-
stop_event=render_context.stop_event,
|
|
214
|
-
)
|
|
215
|
-
context = render_context.unit_tests_running_context
|
|
216
|
-
if exit_code == 0:
|
|
217
|
-
context.verified_passing, context.verified_passing_log_path = True, log_file_path
|
|
218
|
-
return {"output": "All unit tests passed."}
|
|
219
|
-
if not log_file_path and output:
|
|
220
|
-
with tempfile.NamedTemporaryFile("w", encoding="utf-8", delete=False, suffix=".unittest_output") as f:
|
|
221
|
-
f.write(output)
|
|
222
|
-
log_file_path = f.name
|
|
223
|
-
if log_file_path:
|
|
224
|
-
register_log_path(log_file_path, render_context)
|
|
225
|
-
return {
|
|
226
|
-
"output": f"Unit tests failed (exit code {exit_code}).{full_log_pointer(log_file_path)}",
|
|
227
|
-
"test_output": output,
|
|
228
|
-
}
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
TOOLS: dict[str, Callable[[dict, RenderContext], str | dict]] = {
|
|
232
|
-
"read_file": read_file,
|
|
233
|
-
"grep": grep,
|
|
234
|
-
"ls_files": ls_files,
|
|
235
|
-
"edit_file": edit_file,
|
|
236
|
-
"write_file": write_file,
|
|
237
|
-
"delete_file": delete_file,
|
|
238
|
-
"run_unit_tests": run_unit_tests,
|
|
239
|
-
}
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
def execute_calls(calls: list[dict], render_context: RenderContext) -> list[dict]:
|
|
243
|
-
"""Execute the agent's tool calls in order; every call gets a result, errors included.
|
|
244
|
-
|
|
245
|
-
A repeated read-only call (same tool and arguments, no file changed in between) is not
|
|
246
|
-
re-executed: its result is already in the conversation, so a short pointer is returned."""
|
|
247
|
-
cache = render_context.unit_tests_running_context.tool_result_cache
|
|
248
|
-
results = []
|
|
249
|
-
for call in calls:
|
|
250
|
-
tool = TOOLS.get(call["name"])
|
|
251
|
-
cache_key = json.dumps([call["name"], call.get("args") or {}], sort_keys=True)
|
|
252
|
-
result: dict
|
|
253
|
-
if tool is None:
|
|
254
|
-
result = {"output": f"Error: unknown tool '{call['name']}'."}
|
|
255
|
-
elif call["name"] in READ_ONLY_TOOLS and cache_key in cache:
|
|
256
|
-
result = {
|
|
257
|
-
"output": "Same call as an earlier one and no file has changed since; "
|
|
258
|
-
"its result is unchanged (see the earlier result above)."
|
|
259
|
-
}
|
|
260
|
-
else:
|
|
261
|
-
try:
|
|
262
|
-
output = tool(call.get("args") or {}, render_context)
|
|
263
|
-
result = output if isinstance(output, dict) else {"output": output}
|
|
264
|
-
except Exception as e:
|
|
265
|
-
result = {"output": f"Error: tool '{call['name']}' failed: {type(e).__name__}: {e}"}
|
|
266
|
-
if call["name"] in READ_ONLY_TOOLS and not result["output"].startswith("Error"):
|
|
267
|
-
cache[cache_key] = result["output"]
|
|
268
|
-
console.debug(f"Agent tool {call['name']}({call.get('args')}) -> {result['output'][:200]!r}")
|
|
269
|
-
results.append({"call_id": call["id"], **result})
|
|
270
|
-
return results
|