codeplain 0.3.11.dev19__py3-none-any.whl → 0.3.11.dev20__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {codeplain-0.3.11.dev19.dist-info → codeplain-0.3.11.dev20.dist-info}/METADATA +1 -1
- {codeplain-0.3.11.dev19.dist-info → codeplain-0.3.11.dev20.dist-info}/RECORD +14 -11
- codeplain_REST_api.py +26 -32
- render_machine/actions/fix_unit_tests.py +224 -47
- render_machine/actions/run_unit_tests.py +9 -0
- render_machine/agent_tools.py +270 -0
- render_machine/render_context.py +9 -0
- render_machine/render_types.py +45 -0
- tests/test_agent_tools.py +166 -0
- tests/test_fix_unit_tests_action.py +239 -0
- tests/test_fix_unit_tests_conformance_context.py +120 -25
- {codeplain-0.3.11.dev19.dist-info → codeplain-0.3.11.dev20.dist-info}/WHEEL +0 -0
- {codeplain-0.3.11.dev19.dist-info → codeplain-0.3.11.dev20.dist-info}/entry_points.txt +0 -0
- {codeplain-0.3.11.dev19.dist-info → codeplain-0.3.11.dev20.dist-info}/licenses/LICENSE +0 -0
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
change_detection.py,sha256=CazORZrezeZSw2I6VBvieN-c2SEjieZPo6pNrKwmMVA,6915
|
|
2
|
-
codeplain_REST_api.py,sha256=
|
|
2
|
+
codeplain_REST_api.py,sha256=EsKbrbwSxACoN62NNz8AukUgKZvg-YGOpv5aP8ZxF_E,22645
|
|
3
3
|
concept_utils.py,sha256=TpyiQ4DfMtO231j-lPj65i0suRTPr8t9j3QNf9KiM4A,8080
|
|
4
4
|
conformance_fix_journal.py,sha256=7wL3UOc3BWaUo5MCmiFbb3kIwOQ6iEpdQSyEmedo8Vk,9488
|
|
5
5
|
diff_utils.py,sha256=AjiQlqo5pRos_8hVXZo5yBurl5BzSrTMGrQv4dCtRCg,1198
|
|
@@ -37,11 +37,12 @@ config/system_config.yaml,sha256=of5R9vukSohU9AlkGJOcAVQA1UJgxYmDy6islU6iFuE,508
|
|
|
37
37
|
docs/generate_cli.py,sha256=0FHVhICbM8g7ahdInTNih3hNKpty1wbV5rdGECMJj48,656
|
|
38
38
|
examples/example_hello_world_python/harness_tests/hello_world_display/test_hello_world.py,sha256=dwTowrHiVKKbrDv21v8xJC30Q57AXZkQasdGOO5JsBE,470
|
|
39
39
|
render_machine/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
40
|
+
render_machine/agent_tools.py,sha256=Kp2g9GphV7oA9MbOHctKp0UJJJUOgcfCz2SeW9NL6jg,12305
|
|
40
41
|
render_machine/code_renderer.py,sha256=-vp_ltU7e3ssX9hbS8IY4ETSp-39JAuoJ3ZltKpdt1M,3959
|
|
41
42
|
render_machine/conformance_tests.py,sha256=PNTCnJNd_xNIhpTPPS8vlx8qws3cTgQSTLdADRab2YM,9617
|
|
42
43
|
render_machine/implementation_code_helpers.py,sha256=Fxm-IInoZjAIZn41_-WXWj7VIYczpW4yH44Gt8CYMAQ,2378
|
|
43
|
-
render_machine/render_context.py,sha256=
|
|
44
|
-
render_machine/render_types.py,sha256=
|
|
44
|
+
render_machine/render_context.py,sha256=Jo6eRYPYw5YxNLz2nhkS3-W_hYSPlhu5QEyIE-zJTaM,26812
|
|
45
|
+
render_machine/render_types.py,sha256=54A0O8IrKx9HGMAlBfiCnmVLcmacBbX0qo_qqXO3_YU,11572
|
|
45
46
|
render_machine/render_utils.py,sha256=yA3X15E_Lebk_-MxkSz9rM_qy-Pes8EQ9Ye-smkHCDw,9320
|
|
46
47
|
render_machine/state_machine_config.py,sha256=0pL4PyrZ5F2UumvHLMRfhyh64M4pDvf-9zEs-5OAEU8,27866
|
|
47
48
|
render_machine/states.py,sha256=Lu-7upbs9_bjJsy0P738qx5IiN0DzVcFdhF2-m6YjnQ,2067
|
|
@@ -55,14 +56,14 @@ render_machine/actions/distill_conformance_test_memory.py,sha256=ENjfhTWqRLePHeR
|
|
|
55
56
|
render_machine/actions/exit_with_error.py,sha256=W6TMULHJcHEaXE4WUnlELEu0_l7IZy9HWaPkXyqg7ck,1279
|
|
56
57
|
render_machine/actions/finish_functional_requirement.py,sha256=usQgPSg6PHpBOR1RhK7lPz9zjY-2VJJDqi3T0nlqdLw,841
|
|
57
58
|
render_machine/actions/fix_conformance_test.py,sha256=Yy6niLefPIyR3vSxyyX4NUi3NJ4dVc0aEAH9JBCBYX4,11572
|
|
58
|
-
render_machine/actions/fix_unit_tests.py,sha256=
|
|
59
|
+
render_machine/actions/fix_unit_tests.py,sha256=NgfoqCI4XANrA9muAT9bygEEuvqnwTTZGJS4nfsy0WM,13540
|
|
59
60
|
render_machine/actions/prepare_repositories.py,sha256=HyZ1R6E3H5V-G4WDSGfhxl_arskmW9FcaxNdQr6JjVs,3867
|
|
60
61
|
render_machine/actions/prepare_testing_environment.py,sha256=kIsXgV6-jOlqp-04cimPgTQoHPrVIu0zW-H_Jj-jowE,2106
|
|
61
62
|
render_machine/actions/refactor_code.py,sha256=gyPsnqgHseL2bkENu4WeGCRyV0R18_GH3JOrPdoxKvg,2609
|
|
62
63
|
render_machine/actions/render_conformance_tests.py,sha256=5gOQn-G1T9UWEI8sDfALcVdArk0ln9DgUprOPjccS-U,9268
|
|
63
64
|
render_machine/actions/render_functional_requirement.py,sha256=53KzMVlcefoag1x73roCEQvQ7FbmH8FzhXMQau-s8tg,4655
|
|
64
65
|
render_machine/actions/run_conformance_tests.py,sha256=yUG6buqp75EOmJYxdM6MmCNF5_G4TUxA-QSTKx7EzxA,3760
|
|
65
|
-
render_machine/actions/run_unit_tests.py,sha256=
|
|
66
|
+
render_machine/actions/run_unit_tests.py,sha256=j0qsHMLy49I_DX5QLlH0OdPT6bPMc0icmVg2N5MYm9U,2711
|
|
66
67
|
render_machine/actions/summarize_conformance_tests.py,sha256=Vy1kRK8vyazK_BbsKuz1MLDN_TQWVxxzkwEEm-50byc,1639
|
|
67
68
|
standard_template_library/golang-console-app-template.plain,sha256=JOusEjCj7jagnXUzaI-SdIr3hj1wTo2uShK017aNtVM,949
|
|
68
69
|
standard_template_library/python-console-app-template.plain,sha256=HW-REeF8Crk0r_w4-MxnleVHeW6PdOB2C8gO8FDG61g,931
|
|
@@ -70,6 +71,7 @@ standard_template_library/typescript-react-app-boilerplate.plain,sha256=6LFxhEOz
|
|
|
70
71
|
standard_template_library/typescript-react-app-template.plain,sha256=DN5wEJcT8ZqQTf0vFKi9Z1isi9cin2UVNOZb3r28d3M,715
|
|
71
72
|
tests/__init__.py,sha256=Wk73Io62J15BtlLVIzxmASDWaaJkQLevS4BLK5LDAQg,16
|
|
72
73
|
tests/conftest.py,sha256=QZcp08htUlJGgmHDlRWFgsWXZ8o8IBWTD5QqJaUMlU8,790
|
|
74
|
+
tests/test_agent_tools.py,sha256=1WQjKlDd1kna5D4fvlLJ1o5DxyjW0wy-9tx8tRPv6Sw,7840
|
|
73
75
|
tests/test_arg_provenance.py,sha256=d_J7aCeHMXBpctBWTt60WU-I8pEKN7XRXRTJZZGCdo0,3441
|
|
74
76
|
tests/test_change_detection.py,sha256=jH8SEhndX400DNqIKmcYLzxHqq9WFdgmuvY3XPq-ZFc,13534
|
|
75
77
|
tests/test_cli_output.py,sha256=9yTk0pD2w6vkiSdSkYwneUmEtNqX4x64F6PzYShUEcw,19331
|
|
@@ -78,7 +80,8 @@ tests/test_console_log_styles.py,sha256=sxlhynMAP-kmqhBn1-pKvZXdCnlKSyCGJqwYKrh6
|
|
|
78
80
|
tests/test_distill_conformance_test_memory.py,sha256=Tftw9l-gLVwIU6QttY03h2VedvuFxpIsCmZIzvWfxMA,5481
|
|
79
81
|
tests/test_dry_run_link_validation.py,sha256=ivvYXQPqUkQ8glbFyUrdCninCxF2mg-rAsDcfjcB7MM,2140
|
|
80
82
|
tests/test_file_utils.py,sha256=R_ZIhrTtwlhHxgmbBtA8PFEU5-QB4Fi5skJ7PS2KzdU,3614
|
|
81
|
-
tests/
|
|
83
|
+
tests/test_fix_unit_tests_action.py,sha256=0xY94BFbYa528LBXmO7_gfVNbQEqdzXsDHPjQV5qfy0,11410
|
|
84
|
+
tests/test_fix_unit_tests_conformance_context.py,sha256=G6-_YMhN5BY1sJ7cAx3BDnf_McgFcRpyP0iP6mwJ1eA,13154
|
|
82
85
|
tests/test_folder_path_resolution.py,sha256=K9v684xkPE0HOAsovg8uDixpZmaIk7ho-Wd3rTjKLF0,6560
|
|
83
86
|
tests/test_git_preflight.py,sha256=6aQqInDzpk03pS-6UfkGQWjErrIXTEdHs5oeq3ZYzTw,3186
|
|
84
87
|
tests/test_git_utils.py,sha256=h3rWhhs4XVRYqzVzdmj51siF_AYD-GkbZiefK2iGfWo,15614
|
|
@@ -117,8 +120,8 @@ tui/spinner.py,sha256=Ro6Gd9Przf-whuHqPRY6HwI0T57yJjyNPbhDbigZKZE,2471
|
|
|
117
120
|
tui/state_handlers.py,sha256=zbUS_D9eU8qIwVpaxPEp9zL9bpg8gXsWNoUaqg9TLrQ,16664
|
|
118
121
|
tui/styles.css,sha256=kiC7Og_2G1E7owoiZgV95-3_dSeQnsNj3qzaY04xkDg,6727
|
|
119
122
|
tui/widget_helpers.py,sha256=jitm2WHiKj-NVFL-hULxWzckPreRbUTl9OIZLN31-ek,6997
|
|
120
|
-
codeplain-0.3.11.
|
|
121
|
-
codeplain-0.3.11.
|
|
122
|
-
codeplain-0.3.11.
|
|
123
|
-
codeplain-0.3.11.
|
|
124
|
-
codeplain-0.3.11.
|
|
123
|
+
codeplain-0.3.11.dev20.dist-info/METADATA,sha256=O9c6YPWNtbYueXEKXIA4j3P_tzNcJQA8D0wlQmym5gQ,7257
|
|
124
|
+
codeplain-0.3.11.dev20.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
125
|
+
codeplain-0.3.11.dev20.dist-info/entry_points.txt,sha256=oDZkBqu9WhtZApb_K6ia8-fn9aojwmAsgnKELceX5T4,46
|
|
126
|
+
codeplain-0.3.11.dev20.dist-info/licenses/LICENSE,sha256=pCeKgQ1mXE5OmNUuKOfOZh1T1vqkeCUxghE7N8qgnzc,11345
|
|
127
|
+
codeplain-0.3.11.dev20.dist-info/RECORD,,
|
codeplain_REST_api.py
CHANGED
|
@@ -253,38 +253,6 @@ class CodeplainAPI:
|
|
|
253
253
|
|
|
254
254
|
return self.post_request(endpoint_url, headers, payload, run_state)
|
|
255
255
|
|
|
256
|
-
def fix_unittests_issue(
|
|
257
|
-
self,
|
|
258
|
-
frid,
|
|
259
|
-
plain_source_tree,
|
|
260
|
-
linked_resources,
|
|
261
|
-
existing_files_content,
|
|
262
|
-
module_name: str,
|
|
263
|
-
required_modules,
|
|
264
|
-
unittests_issue,
|
|
265
|
-
run_state: RunState,
|
|
266
|
-
conformance_tests_fixes: list[dict] | None = None,
|
|
267
|
-
):
|
|
268
|
-
endpoint_url = f"{self.api_url}/fix_unittests_issue"
|
|
269
|
-
headers = {"X-API-Key": self.api_key, "Content-Type": "application/json"}
|
|
270
|
-
|
|
271
|
-
payload = {
|
|
272
|
-
"frid": frid,
|
|
273
|
-
"plain_source_tree": plain_source_tree,
|
|
274
|
-
"linked_resources": linked_resources,
|
|
275
|
-
"existing_files_content": existing_files_content,
|
|
276
|
-
"module_name": module_name,
|
|
277
|
-
"required_modules": required_modules,
|
|
278
|
-
"unittests_issue": unittests_issue,
|
|
279
|
-
"unittest_batch_id": run_state.unittest_batch_id,
|
|
280
|
-
}
|
|
281
|
-
# Implementation code changes made by the conformance tests fixer right before this unit tests run;
|
|
282
|
-
# only sent when unit tests are processed inside the conformance tests phase.
|
|
283
|
-
if conformance_tests_fixes is not None:
|
|
284
|
-
payload["conformance_tests_fixes"] = conformance_tests_fixes
|
|
285
|
-
|
|
286
|
-
return self.post_request(endpoint_url, headers, payload, run_state)
|
|
287
|
-
|
|
288
256
|
def distill_conformance_test_memory(
|
|
289
257
|
self,
|
|
290
258
|
frid,
|
|
@@ -558,3 +526,29 @@ class CodeplainAPI:
|
|
|
558
526
|
}
|
|
559
527
|
|
|
560
528
|
return self.post_request(endpoint_url, headers, payload, run_state)
|
|
529
|
+
|
|
530
|
+
def agent_start(self, task_type: str, task_params: dict, frid: str, module_name: str, run_state: RunState):
|
|
531
|
+
"""Start a server-side agent session; returns the first turn (tool calls or completion)."""
|
|
532
|
+
endpoint_url = f"{self.api_url}/agent/start"
|
|
533
|
+
headers = {"X-API-Key": self.api_key, "Content-Type": "application/json"}
|
|
534
|
+
payload = {
|
|
535
|
+
"task_type": task_type,
|
|
536
|
+
"task_params": task_params,
|
|
537
|
+
"frid": frid,
|
|
538
|
+
"module_name": module_name,
|
|
539
|
+
}
|
|
540
|
+
return self.post_request(endpoint_url, headers, payload, run_state)
|
|
541
|
+
|
|
542
|
+
def agent_continue(
|
|
543
|
+
self, session_id: str, tool_results: list[dict], frid: str, module_name: str, run_state: RunState
|
|
544
|
+
):
|
|
545
|
+
"""Feed tool results into an agent session and run its next turn."""
|
|
546
|
+
endpoint_url = f"{self.api_url}/agent/continue"
|
|
547
|
+
headers = {"X-API-Key": self.api_key, "Content-Type": "application/json"}
|
|
548
|
+
payload = {
|
|
549
|
+
"session_id": session_id,
|
|
550
|
+
"tool_results": tool_results,
|
|
551
|
+
"frid": frid,
|
|
552
|
+
"module_name": module_name,
|
|
553
|
+
}
|
|
554
|
+
return self.post_request(endpoint_url, headers, payload, run_state)
|
|
@@ -1,79 +1,256 @@
|
|
|
1
|
+
import os
|
|
2
|
+
from http import HTTPStatus
|
|
1
3
|
from typing import Any
|
|
2
4
|
|
|
5
|
+
import requests
|
|
6
|
+
|
|
3
7
|
import file_utils
|
|
4
|
-
import
|
|
8
|
+
import plain_spec
|
|
5
9
|
from plain2code_console import console
|
|
6
10
|
from plain2code_exceptions import InternalClientError
|
|
11
|
+
from render_machine import agent_tools
|
|
7
12
|
from render_machine.actions.base_action import BaseAction
|
|
8
|
-
from render_machine.implementation_code_helpers import ImplementationCodeHelpers
|
|
9
13
|
from render_machine.render_context import RenderContext
|
|
10
14
|
|
|
11
|
-
|
|
15
|
+
TASK_TYPE = "fix_unit_tests"
|
|
16
|
+
SUBMIT_FIX_TOOL = "submit_fix"
|
|
17
|
+
# Upper bound on LLM turns spent on one fix attempt; the server bounds the whole session.
|
|
18
|
+
MAX_AGENT_TURNS_PER_ATTEMPT = 40
|
|
19
|
+
# Answer to a submit_fix whose fix was accepted in an earlier unit-test loop of the conformance phase.
|
|
20
|
+
ACCEPTED_THEN_CHANGED_MESSAGE = (
|
|
21
|
+
"Your fix was accepted: the unit tests passed. Afterwards the implementation code was changed to fix the "
|
|
22
|
+
"conformance tests{see_below}, and the unit tests now fail again. Files may have changed since you last read "
|
|
23
|
+
"them."
|
|
24
|
+
)
|
|
25
|
+
# Seeding the first turn: the build folder's file list and the files changed for the FRID.
|
|
26
|
+
MAX_FILE_TREE_ENTRIES = 500
|
|
27
|
+
MAX_RELEVANT_FILES_CHARS = 60_000
|
|
12
28
|
|
|
13
29
|
|
|
14
30
|
class FixUnitTests(BaseAction):
|
|
31
|
+
"""Fix failing unit tests with a server-side agent session that spans the fix attempts.
|
|
32
|
+
|
|
33
|
+
The first failure starts a session; the agent then drives read/grep/edit/run tool calls
|
|
34
|
+
(executed here) until it calls submit_fix. The state machine re-runs the unit tests and,
|
|
35
|
+
if they still fail, the next execution of this action answers that submit_fix call with
|
|
36
|
+
the new failure output inside the same session, so earlier attempts stay in context.
|
|
37
|
+
|
|
38
|
+
During the conformance phase the session also spans unit-test loops: after the conformance
|
|
39
|
+
tests fixer changes implementation code, the unit tests fail again and the same session is
|
|
40
|
+
continued, told what the conformance tests fixer changed (which it must preserve).
|
|
41
|
+
"""
|
|
42
|
+
|
|
15
43
|
SUCCESSFUL_OUTCOME = "unit_tests_fix_generated"
|
|
16
44
|
|
|
17
45
|
def execute(self, render_context: RenderContext, previous_action_payload: Any | None):
|
|
18
|
-
if not previous_action_payload.get("previous_unittests_issue"):
|
|
46
|
+
if not previous_action_payload or not previous_action_payload.get("previous_unittests_issue"):
|
|
19
47
|
raise InternalClientError(
|
|
20
48
|
"Internal client error: Previous action payload does not contain previous unit tests issue."
|
|
21
49
|
)
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
render_context
|
|
31
|
-
)
|
|
32
|
-
|
|
33
|
-
render_utils.print_inputs(render_context, existing_files_content, "Files sent as input to unit tests fixing:")
|
|
34
|
-
|
|
50
|
+
unittests_issue = previous_action_payload["previous_unittests_issue"]
|
|
51
|
+
context = render_context.unit_tests_running_context
|
|
52
|
+
session = render_context.unit_tests_agent_session
|
|
53
|
+
api = render_context.codeplain_api
|
|
54
|
+
frid, module_name = render_context.frid_context.frid, render_context.module_name
|
|
55
|
+
changed_files_before = set(context.changed_files)
|
|
56
|
+
log_path = render_context.script_execution_history.latest_unit_test_output_path
|
|
57
|
+
if log_path:
|
|
58
|
+
agent_tools.register_log_path(log_path, render_context)
|
|
35
59
|
conformance_tests_fixes = self._get_conformance_tests_fixes(render_context)
|
|
36
|
-
|
|
60
|
+
new_conformance_tests_fixes = conformance_tests_fixes[session.conformance_fixes_handed_off :]
|
|
61
|
+
if new_conformance_tests_fixes:
|
|
37
62
|
console.info(
|
|
38
|
-
f"Unit tests are fixed while preserving {len(
|
|
39
|
-
"made to fix the conformance tests."
|
|
63
|
+
f"Unit tests are fixed while preserving {len(new_conformance_tests_fixes)} implementation code "
|
|
64
|
+
"change(s) made to fix the conformance tests."
|
|
40
65
|
)
|
|
41
66
|
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
67
|
+
response = None
|
|
68
|
+
if session.session_id is not None:
|
|
69
|
+
if context.agent_used_in_this_loop:
|
|
70
|
+
console.info(f"Continuing agent session {session.session_id} with the new unit tests failure.")
|
|
71
|
+
output = "The fix was applied, but the unit tests still fail."
|
|
72
|
+
else:
|
|
73
|
+
# The session's last fix was accepted in an earlier unit-test loop of this conformance
|
|
74
|
+
# phase; since then the conformance tests fixer changed the code and the tests fail again.
|
|
75
|
+
console.info(
|
|
76
|
+
f"Continuing agent session {session.session_id}: the unit tests fail again after the "
|
|
77
|
+
"implementation was changed to fix the conformance tests."
|
|
78
|
+
)
|
|
79
|
+
output = ACCEPTED_THEN_CHANGED_MESSAGE.format(
|
|
80
|
+
see_below=" (see the Conformance Tests Fix below)" if new_conformance_tests_fixes else ""
|
|
81
|
+
)
|
|
82
|
+
# Files changed outside the session, so earlier read results are stale.
|
|
83
|
+
context.tool_result_cache.clear()
|
|
84
|
+
submit_result: dict = {
|
|
85
|
+
"call_id": session.pending_submit_call_id,
|
|
86
|
+
"output": output + agent_tools.full_log_pointer(log_path),
|
|
87
|
+
"test_output": unittests_issue,
|
|
88
|
+
}
|
|
89
|
+
if new_conformance_tests_fixes:
|
|
90
|
+
submit_result["conformance_tests_fixes"] = new_conformance_tests_fixes
|
|
91
|
+
tool_results = session.pending_tool_results + [submit_result]
|
|
92
|
+
session.pending_tool_results, session.pending_submit_call_id = [], None
|
|
93
|
+
response = self._continue_session(render_context, session.session_id, tool_results)
|
|
94
|
+
if response is None:
|
|
95
|
+
console.info("Starting an agent session to fix the unit tests.")
|
|
96
|
+
# Cached read results point at earlier turns, which a new session does not have.
|
|
97
|
+
context.tool_result_cache.clear()
|
|
98
|
+
response = api.agent_start(
|
|
99
|
+
TASK_TYPE,
|
|
100
|
+
self._build_task_params(render_context, unittests_issue, conformance_tests_fixes),
|
|
101
|
+
frid,
|
|
102
|
+
module_name,
|
|
103
|
+
render_context.run_state,
|
|
104
|
+
)
|
|
105
|
+
session.session_id = response["session_id"]
|
|
106
|
+
assert session.session_id is not None
|
|
107
|
+
session.conformance_fixes_handed_off = len(conformance_tests_fixes)
|
|
108
|
+
context.agent_used_in_this_loop = True
|
|
57
109
|
|
|
58
|
-
|
|
110
|
+
submitted = False
|
|
111
|
+
turns = 0
|
|
112
|
+
while response.get("status") == "tool_calls" and turns < MAX_AGENT_TURNS_PER_ATTEMPT:
|
|
113
|
+
turns += 1
|
|
114
|
+
calls = response["calls"]
|
|
115
|
+
submit_call = next((call for call in calls if call["name"] == SUBMIT_FIX_TOOL), None)
|
|
116
|
+
tool_results = agent_tools.execute_calls(
|
|
117
|
+
[call for call in calls if call is not submit_call], render_context
|
|
118
|
+
)
|
|
119
|
+
if submit_call is not None:
|
|
120
|
+
session.pending_submit_call_id = submit_call["id"]
|
|
121
|
+
session.pending_tool_results = tool_results
|
|
122
|
+
submitted = True
|
|
123
|
+
console.info(f"Agent submitted a fix: {submit_call['args'].get('changes_made', '')}")
|
|
124
|
+
break
|
|
125
|
+
response = api.agent_continue(session.session_id, tool_results, frid, module_name, render_context.run_state)
|
|
59
126
|
|
|
60
|
-
|
|
127
|
+
if not submitted:
|
|
128
|
+
# The session ended without a submission (finished in text, failed, or used up this
|
|
129
|
+
# attempt's turn budget) — a fresh session is started if the tests still fail.
|
|
130
|
+
status = response.get("status")
|
|
131
|
+
if status == "failed":
|
|
132
|
+
console.warning(f"Agent session failed: {response.get('error', 'unknown error')}")
|
|
133
|
+
elif status == "tool_calls":
|
|
134
|
+
console.warning(f"Agent used {MAX_AGENT_TURNS_PER_ATTEMPT} turns without submitting a fix.")
|
|
135
|
+
session.reset()
|
|
61
136
|
|
|
137
|
+
console.print_files(
|
|
138
|
+
"Files changed while fixing unit tests:",
|
|
139
|
+
render_context.build_folder,
|
|
140
|
+
{path: "" for path in sorted(context.changed_files - changed_files_before)},
|
|
141
|
+
style=console.OUTPUT_STYLE,
|
|
142
|
+
)
|
|
62
143
|
return self.SUCCESSFUL_OUTCOME, None
|
|
63
144
|
|
|
64
145
|
@staticmethod
|
|
65
|
-
def _get_conformance_tests_fixes(render_context: RenderContext) -> list[dict]
|
|
66
|
-
"""Implementation code changes the conformance tests fixer made
|
|
67
|
-
|
|
68
|
-
Only present when unit tests are processed inside the conformance tests phase - the implementation
|
|
69
|
-
and refactoring unit test passes have no conformance tests running context.
|
|
70
|
-
"""
|
|
146
|
+
def _get_conformance_tests_fixes(render_context: RenderContext) -> list[dict]:
|
|
147
|
+
"""Implementation code changes the conformance tests fixer made during this conformance phase, oldest
|
|
148
|
+
first. Empty outside the conformance phase (the implementation and refactoring unit-test loops)."""
|
|
71
149
|
conformance_tests_running_context = getattr(render_context, "conformance_tests_running_context", None)
|
|
72
150
|
if conformance_tests_running_context is None:
|
|
73
|
-
return
|
|
151
|
+
return []
|
|
152
|
+
return list(getattr(conformance_tests_running_context, "implementation_code_fixes", None) or [])
|
|
74
153
|
|
|
75
|
-
|
|
76
|
-
|
|
154
|
+
@staticmethod
|
|
155
|
+
def _continue_session(render_context: RenderContext, session_id: str, tool_results: list[dict]) -> dict | None:
|
|
156
|
+
"""Continue the session; None if the server no longer has it (expired), so a new one is started."""
|
|
157
|
+
try:
|
|
158
|
+
return render_context.codeplain_api.agent_continue(
|
|
159
|
+
session_id,
|
|
160
|
+
tool_results,
|
|
161
|
+
render_context.frid_context.frid,
|
|
162
|
+
render_context.module_name,
|
|
163
|
+
render_context.run_state,
|
|
164
|
+
)
|
|
165
|
+
except requests.exceptions.HTTPError as e:
|
|
166
|
+
if e.response is None or e.response.status_code != HTTPStatus.NOT_FOUND:
|
|
167
|
+
raise
|
|
168
|
+
console.warning(f"Agent session {session_id} has expired; starting a new one.")
|
|
169
|
+
render_context.unit_tests_agent_session.reset()
|
|
77
170
|
return None
|
|
78
171
|
|
|
79
|
-
|
|
172
|
+
@staticmethod
|
|
173
|
+
def _build_task_params(
|
|
174
|
+
render_context: RenderContext, unittests_issue: str, conformance_tests_fixes: list[dict]
|
|
175
|
+
) -> dict:
|
|
176
|
+
frid = render_context.frid_context.frid
|
|
177
|
+
specifications, _ = plain_spec.get_specifications_for_frid(render_context.plain_source_tree, frid)
|
|
178
|
+
context = render_context.unit_tests_running_context
|
|
179
|
+
# The files the conformance tests fixes changed are always seeded, so the agent sees what to preserve.
|
|
180
|
+
conformance_fix_files = {name for fix in conformance_tests_fixes for name in (fix or {}).get("code_diff") or {}}
|
|
181
|
+
task_params = {
|
|
182
|
+
"definitions": "\n".join(specifications.get(plain_spec.DEFINITIONS, [])),
|
|
183
|
+
"non_functional_requirements": "\n".join(specifications.get(plain_spec.NON_FUNCTIONAL_REQUIREMENTS, [])),
|
|
184
|
+
"functional_requirements": FixUnitTests._functional_requirements_section(render_context, specifications),
|
|
185
|
+
"linked_resources": render_context.frid_context.linked_resources,
|
|
186
|
+
"build_folder": render_context.build_folder,
|
|
187
|
+
"module_name": render_context.module_name,
|
|
188
|
+
"unittests_script_content": FixUnitTests._read_script(render_context.unittests_script),
|
|
189
|
+
"unittests_issue": unittests_issue,
|
|
190
|
+
"unittests_log_path": render_context.script_execution_history.latest_unit_test_output_path,
|
|
191
|
+
"file_tree": FixUnitTests._file_tree(render_context.build_folder),
|
|
192
|
+
"relevant_files": FixUnitTests._relevant_files(
|
|
193
|
+
render_context.build_folder,
|
|
194
|
+
render_context.frid_context.changed_files | context.changed_files | conformance_fix_files,
|
|
195
|
+
),
|
|
196
|
+
}
|
|
197
|
+
if conformance_tests_fixes:
|
|
198
|
+
task_params["conformance_tests_fixes"] = conformance_tests_fixes
|
|
199
|
+
session = render_context.unit_tests_agent_session
|
|
200
|
+
if session.previous_session_id:
|
|
201
|
+
task_params["previous_session_id"] = session.previous_session_id
|
|
202
|
+
return task_params
|
|
203
|
+
|
|
204
|
+
@staticmethod
|
|
205
|
+
def _file_tree(build_folder: str) -> str:
|
|
206
|
+
paths = []
|
|
207
|
+
for root, dirs, files in os.walk(build_folder):
|
|
208
|
+
dirs[:] = sorted(d for d in dirs if d not in agent_tools.GREP_EXCLUDED_DIRS and not d.startswith("."))
|
|
209
|
+
paths.extend(os.path.relpath(os.path.join(root, name), build_folder) for name in sorted(files))
|
|
210
|
+
if len(paths) > MAX_FILE_TREE_ENTRIES:
|
|
211
|
+
return "\n".join(paths[:MAX_FILE_TREE_ENTRIES]) + "\n... [more files not listed]"
|
|
212
|
+
return "\n".join(paths)
|
|
213
|
+
|
|
214
|
+
@staticmethod
|
|
215
|
+
def _relevant_files(build_folder: str, file_names: set[str]) -> dict[str, str]:
|
|
216
|
+
"""Contents of the given build-relative files, smallest first, within MAX_RELEVANT_FILES_CHARS."""
|
|
217
|
+
contents = {}
|
|
218
|
+
for name in file_names:
|
|
219
|
+
full_path = os.path.join(build_folder, name)
|
|
220
|
+
if os.path.isfile(full_path):
|
|
221
|
+
with open(full_path, "r", encoding="utf-8", errors="replace") as f:
|
|
222
|
+
contents[name] = f.read()
|
|
223
|
+
relevant, total = {}, 0
|
|
224
|
+
for name in sorted(contents, key=lambda n: (len(contents[n]), n)):
|
|
225
|
+
if total + len(contents[name]) > MAX_RELEVANT_FILES_CHARS:
|
|
226
|
+
break
|
|
227
|
+
relevant[name] = contents[name]
|
|
228
|
+
total += len(contents[name])
|
|
229
|
+
return dict(sorted(relevant.items()))
|
|
230
|
+
|
|
231
|
+
@staticmethod
|
|
232
|
+
def _functional_requirements_section(render_context: RenderContext, specifications: dict) -> str:
|
|
233
|
+
sections = []
|
|
234
|
+
for module_name, functionalities in render_context.get_required_modules_functionalities().items():
|
|
235
|
+
sections.append(
|
|
236
|
+
f"### Module: {module_name} (Already Implemented, for context)\n" + "\n".join(functionalities)
|
|
237
|
+
)
|
|
238
|
+
current = specifications.get(plain_spec.FUNCTIONAL_REQUIREMENTS, [])
|
|
239
|
+
if len(current) > 1:
|
|
240
|
+
sections.append(
|
|
241
|
+
f"### Module: {render_context.module_name} (Already Implemented, for context)\n"
|
|
242
|
+
+ "\n".join(current[:-1])
|
|
243
|
+
)
|
|
244
|
+
if current:
|
|
245
|
+
sections.append(f"### Module: {render_context.module_name} (Currently Being Implemented)\n{current[-1]}")
|
|
246
|
+
return "\n\n".join(sections)
|
|
247
|
+
|
|
248
|
+
@staticmethod
|
|
249
|
+
def _read_script(script: str | None) -> str:
|
|
250
|
+
if not script:
|
|
251
|
+
return ""
|
|
252
|
+
try:
|
|
253
|
+
with open(file_utils.add_current_path_if_no_path(script), "r", encoding="utf-8") as f:
|
|
254
|
+
return f.read()
|
|
255
|
+
except OSError:
|
|
256
|
+
return ""
|
|
@@ -16,6 +16,15 @@ class RunUnitTests(BaseAction):
|
|
|
16
16
|
UNRECOVERABLE_ERROR_OUTCOME = "unrecoverable_error_occurred"
|
|
17
17
|
|
|
18
18
|
def execute(self, render_context: RenderContext, _previous_action_payload: Any | None):
|
|
19
|
+
context = render_context.unit_tests_running_context
|
|
20
|
+
if context.verified_passing:
|
|
21
|
+
# The fixing agent's own run passed and no file changed since; running again is redundant.
|
|
22
|
+
context.verified_passing = False
|
|
23
|
+
console.info("Unit tests already passed in the fixing agent's last run; not running them again.")
|
|
24
|
+
render_context.script_execution_history.latest_unit_test_output_path = context.verified_passing_log_path
|
|
25
|
+
render_context.script_execution_history.should_update_script_outputs = True
|
|
26
|
+
return self.SUCCESSFUL_OUTCOME, None
|
|
27
|
+
|
|
19
28
|
unittests_script = os.path.normpath(render_context.unittests_script)
|
|
20
29
|
|
|
21
30
|
console.info(
|