codeplain 0.3.11.dev21__py3-none-any.whl → 0.3.11.dev22__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {codeplain-0.3.11.dev21.dist-info → codeplain-0.3.11.dev22.dist-info}/METADATA +1 -1
- {codeplain-0.3.11.dev21.dist-info → codeplain-0.3.11.dev22.dist-info}/RECORD +11 -14
- codeplain_REST_api.py +32 -26
- render_machine/actions/fix_unit_tests.py +47 -224
- render_machine/actions/run_unit_tests.py +0 -9
- render_machine/render_context.py +0 -9
- render_machine/render_types.py +0 -45
- tests/test_fix_unit_tests_conformance_context.py +25 -120
- render_machine/agent_tools.py +0 -270
- tests/test_agent_tools.py +0 -166
- tests/test_fix_unit_tests_action.py +0 -239
- {codeplain-0.3.11.dev21.dist-info → codeplain-0.3.11.dev22.dist-info}/WHEEL +0 -0
- {codeplain-0.3.11.dev21.dist-info → codeplain-0.3.11.dev22.dist-info}/entry_points.txt +0 -0
- {codeplain-0.3.11.dev21.dist-info → codeplain-0.3.11.dev22.dist-info}/licenses/LICENSE +0 -0
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
change_detection.py,sha256=CazORZrezeZSw2I6VBvieN-c2SEjieZPo6pNrKwmMVA,6915
|
|
2
|
-
codeplain_REST_api.py,sha256=
|
|
2
|
+
codeplain_REST_api.py,sha256=ptb2u0mC5FJqMNeUbcc-bNXNhy9NRJZa5vI80uSFSwQ,22715
|
|
3
3
|
concept_utils.py,sha256=TpyiQ4DfMtO231j-lPj65i0suRTPr8t9j3QNf9KiM4A,8080
|
|
4
4
|
conformance_fix_journal.py,sha256=7wL3UOc3BWaUo5MCmiFbb3kIwOQ6iEpdQSyEmedo8Vk,9488
|
|
5
5
|
diff_utils.py,sha256=AjiQlqo5pRos_8hVXZo5yBurl5BzSrTMGrQv4dCtRCg,1198
|
|
@@ -37,12 +37,11 @@ config/system_config.yaml,sha256=of5R9vukSohU9AlkGJOcAVQA1UJgxYmDy6islU6iFuE,508
|
|
|
37
37
|
docs/generate_cli.py,sha256=0FHVhICbM8g7ahdInTNih3hNKpty1wbV5rdGECMJj48,656
|
|
38
38
|
examples/example_hello_world_python/harness_tests/hello_world_display/test_hello_world.py,sha256=dwTowrHiVKKbrDv21v8xJC30Q57AXZkQasdGOO5JsBE,470
|
|
39
39
|
render_machine/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
40
|
-
render_machine/agent_tools.py,sha256=Kp2g9GphV7oA9MbOHctKp0UJJJUOgcfCz2SeW9NL6jg,12305
|
|
41
40
|
render_machine/code_renderer.py,sha256=-vp_ltU7e3ssX9hbS8IY4ETSp-39JAuoJ3ZltKpdt1M,3959
|
|
42
41
|
render_machine/conformance_tests.py,sha256=PNTCnJNd_xNIhpTPPS8vlx8qws3cTgQSTLdADRab2YM,9617
|
|
43
42
|
render_machine/implementation_code_helpers.py,sha256=Fxm-IInoZjAIZn41_-WXWj7VIYczpW4yH44Gt8CYMAQ,2378
|
|
44
|
-
render_machine/render_context.py,sha256=
|
|
45
|
-
render_machine/render_types.py,sha256=
|
|
43
|
+
render_machine/render_context.py,sha256=iaJWI_EMSn51LNF_b7KRvlir46ZgUkjbaEiVgSvMTCI,26314
|
|
44
|
+
render_machine/render_types.py,sha256=A5BHAkjmDHQ0pd5rR6THlRY5z6rzrdJdcgZFhN7yw9w,8860
|
|
46
45
|
render_machine/render_utils.py,sha256=yA3X15E_Lebk_-MxkSz9rM_qy-Pes8EQ9Ye-smkHCDw,9320
|
|
47
46
|
render_machine/state_machine_config.py,sha256=0pL4PyrZ5F2UumvHLMRfhyh64M4pDvf-9zEs-5OAEU8,27866
|
|
48
47
|
render_machine/states.py,sha256=Lu-7upbs9_bjJsy0P738qx5IiN0DzVcFdhF2-m6YjnQ,2067
|
|
@@ -56,14 +55,14 @@ render_machine/actions/distill_conformance_test_memory.py,sha256=ENjfhTWqRLePHeR
|
|
|
56
55
|
render_machine/actions/exit_with_error.py,sha256=W6TMULHJcHEaXE4WUnlELEu0_l7IZy9HWaPkXyqg7ck,1279
|
|
57
56
|
render_machine/actions/finish_functional_requirement.py,sha256=usQgPSg6PHpBOR1RhK7lPz9zjY-2VJJDqi3T0nlqdLw,841
|
|
58
57
|
render_machine/actions/fix_conformance_test.py,sha256=Yy6niLefPIyR3vSxyyX4NUi3NJ4dVc0aEAH9JBCBYX4,11572
|
|
59
|
-
render_machine/actions/fix_unit_tests.py,sha256=
|
|
58
|
+
render_machine/actions/fix_unit_tests.py,sha256=pqZzp7snSAe967o_kbh1yAhp9pZIMFY9uoJK1HGcCuc,3532
|
|
60
59
|
render_machine/actions/prepare_repositories.py,sha256=HyZ1R6E3H5V-G4WDSGfhxl_arskmW9FcaxNdQr6JjVs,3867
|
|
61
60
|
render_machine/actions/prepare_testing_environment.py,sha256=kIsXgV6-jOlqp-04cimPgTQoHPrVIu0zW-H_Jj-jowE,2106
|
|
62
61
|
render_machine/actions/refactor_code.py,sha256=gyPsnqgHseL2bkENu4WeGCRyV0R18_GH3JOrPdoxKvg,2609
|
|
63
62
|
render_machine/actions/render_conformance_tests.py,sha256=5gOQn-G1T9UWEI8sDfALcVdArk0ln9DgUprOPjccS-U,9268
|
|
64
63
|
render_machine/actions/render_functional_requirement.py,sha256=53KzMVlcefoag1x73roCEQvQ7FbmH8FzhXMQau-s8tg,4655
|
|
65
64
|
render_machine/actions/run_conformance_tests.py,sha256=yUG6buqp75EOmJYxdM6MmCNF5_G4TUxA-QSTKx7EzxA,3760
|
|
66
|
-
render_machine/actions/run_unit_tests.py,sha256=
|
|
65
|
+
render_machine/actions/run_unit_tests.py,sha256=A4WXtSzpCLDHYR8fQAbfxUf_E19GYflZSHZaSreoFPk,2101
|
|
67
66
|
render_machine/actions/summarize_conformance_tests.py,sha256=Vy1kRK8vyazK_BbsKuz1MLDN_TQWVxxzkwEEm-50byc,1639
|
|
68
67
|
standard_template_library/golang-console-app-template.plain,sha256=JOusEjCj7jagnXUzaI-SdIr3hj1wTo2uShK017aNtVM,949
|
|
69
68
|
standard_template_library/python-console-app-template.plain,sha256=HW-REeF8Crk0r_w4-MxnleVHeW6PdOB2C8gO8FDG61g,931
|
|
@@ -71,7 +70,6 @@ standard_template_library/typescript-react-app-boilerplate.plain,sha256=6LFxhEOz
|
|
|
71
70
|
standard_template_library/typescript-react-app-template.plain,sha256=DN5wEJcT8ZqQTf0vFKi9Z1isi9cin2UVNOZb3r28d3M,715
|
|
72
71
|
tests/__init__.py,sha256=Wk73Io62J15BtlLVIzxmASDWaaJkQLevS4BLK5LDAQg,16
|
|
73
72
|
tests/conftest.py,sha256=QZcp08htUlJGgmHDlRWFgsWXZ8o8IBWTD5QqJaUMlU8,790
|
|
74
|
-
tests/test_agent_tools.py,sha256=1WQjKlDd1kna5D4fvlLJ1o5DxyjW0wy-9tx8tRPv6Sw,7840
|
|
75
73
|
tests/test_arg_provenance.py,sha256=d_J7aCeHMXBpctBWTt60WU-I8pEKN7XRXRTJZZGCdo0,3441
|
|
76
74
|
tests/test_change_detection.py,sha256=jH8SEhndX400DNqIKmcYLzxHqq9WFdgmuvY3XPq-ZFc,13534
|
|
77
75
|
tests/test_cli_output.py,sha256=9yTk0pD2w6vkiSdSkYwneUmEtNqX4x64F6PzYShUEcw,19331
|
|
@@ -80,8 +78,7 @@ tests/test_console_log_styles.py,sha256=sxlhynMAP-kmqhBn1-pKvZXdCnlKSyCGJqwYKrh6
|
|
|
80
78
|
tests/test_distill_conformance_test_memory.py,sha256=Tftw9l-gLVwIU6QttY03h2VedvuFxpIsCmZIzvWfxMA,5481
|
|
81
79
|
tests/test_dry_run_link_validation.py,sha256=ivvYXQPqUkQ8glbFyUrdCninCxF2mg-rAsDcfjcB7MM,2140
|
|
82
80
|
tests/test_file_utils.py,sha256=R_ZIhrTtwlhHxgmbBtA8PFEU5-QB4Fi5skJ7PS2KzdU,3614
|
|
83
|
-
tests/
|
|
84
|
-
tests/test_fix_unit_tests_conformance_context.py,sha256=G6-_YMhN5BY1sJ7cAx3BDnf_McgFcRpyP0iP6mwJ1eA,13154
|
|
81
|
+
tests/test_fix_unit_tests_conformance_context.py,sha256=IEsXK0xMLRuHmoQGBwrY0ENWVng2n-db_YzSO2qBfqM,9136
|
|
85
82
|
tests/test_folder_path_resolution.py,sha256=K9v684xkPE0HOAsovg8uDixpZmaIk7ho-Wd3rTjKLF0,6560
|
|
86
83
|
tests/test_git_preflight.py,sha256=6aQqInDzpk03pS-6UfkGQWjErrIXTEdHs5oeq3ZYzTw,3186
|
|
87
84
|
tests/test_git_utils.py,sha256=h3rWhhs4XVRYqzVzdmj51siF_AYD-GkbZiefK2iGfWo,15614
|
|
@@ -120,8 +117,8 @@ tui/spinner.py,sha256=Ro6Gd9Przf-whuHqPRY6HwI0T57yJjyNPbhDbigZKZE,2471
|
|
|
120
117
|
tui/state_handlers.py,sha256=zbUS_D9eU8qIwVpaxPEp9zL9bpg8gXsWNoUaqg9TLrQ,16664
|
|
121
118
|
tui/styles.css,sha256=kiC7Og_2G1E7owoiZgV95-3_dSeQnsNj3qzaY04xkDg,6727
|
|
122
119
|
tui/widget_helpers.py,sha256=jitm2WHiKj-NVFL-hULxWzckPreRbUTl9OIZLN31-ek,6997
|
|
123
|
-
codeplain-0.3.11.
|
|
124
|
-
codeplain-0.3.11.
|
|
125
|
-
codeplain-0.3.11.
|
|
126
|
-
codeplain-0.3.11.
|
|
127
|
-
codeplain-0.3.11.
|
|
120
|
+
codeplain-0.3.11.dev22.dist-info/METADATA,sha256=FkoMgKky0bzyU1gl-SJ38cgvxGf8BxzxcHmRRjrWkBs,7257
|
|
121
|
+
codeplain-0.3.11.dev22.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
122
|
+
codeplain-0.3.11.dev22.dist-info/entry_points.txt,sha256=oDZkBqu9WhtZApb_K6ia8-fn9aojwmAsgnKELceX5T4,46
|
|
123
|
+
codeplain-0.3.11.dev22.dist-info/licenses/LICENSE,sha256=pCeKgQ1mXE5OmNUuKOfOZh1T1vqkeCUxghE7N8qgnzc,11345
|
|
124
|
+
codeplain-0.3.11.dev22.dist-info/RECORD,,
|
codeplain_REST_api.py
CHANGED
|
@@ -253,6 +253,38 @@ class CodeplainAPI:
|
|
|
253
253
|
|
|
254
254
|
return self.post_request(endpoint_url, headers, payload, run_state)
|
|
255
255
|
|
|
256
|
+
def fix_unittests_issue(
|
|
257
|
+
self,
|
|
258
|
+
frid,
|
|
259
|
+
plain_source_tree,
|
|
260
|
+
linked_resources,
|
|
261
|
+
existing_files_content,
|
|
262
|
+
module_name: str,
|
|
263
|
+
required_modules,
|
|
264
|
+
unittests_issue,
|
|
265
|
+
run_state: RunState,
|
|
266
|
+
conformance_tests_fixes: list[dict] | None = None,
|
|
267
|
+
):
|
|
268
|
+
endpoint_url = f"{self.api_url}/fix_unittests_issue"
|
|
269
|
+
headers = {"X-API-Key": self.api_key, "Content-Type": "application/json"}
|
|
270
|
+
|
|
271
|
+
payload = {
|
|
272
|
+
"frid": frid,
|
|
273
|
+
"plain_source_tree": plain_source_tree,
|
|
274
|
+
"linked_resources": linked_resources,
|
|
275
|
+
"existing_files_content": existing_files_content,
|
|
276
|
+
"module_name": module_name,
|
|
277
|
+
"required_modules": required_modules,
|
|
278
|
+
"unittests_issue": unittests_issue,
|
|
279
|
+
"unittest_batch_id": run_state.unittest_batch_id,
|
|
280
|
+
}
|
|
281
|
+
# Implementation code changes made by the conformance tests fixer right before this unit tests run;
|
|
282
|
+
# only sent when unit tests are processed inside the conformance tests phase.
|
|
283
|
+
if conformance_tests_fixes is not None:
|
|
284
|
+
payload["conformance_tests_fixes"] = conformance_tests_fixes
|
|
285
|
+
|
|
286
|
+
return self.post_request(endpoint_url, headers, payload, run_state)
|
|
287
|
+
|
|
256
288
|
def distill_conformance_test_memory(
|
|
257
289
|
self,
|
|
258
290
|
frid,
|
|
@@ -526,29 +558,3 @@ class CodeplainAPI:
|
|
|
526
558
|
}
|
|
527
559
|
|
|
528
560
|
return self.post_request(endpoint_url, headers, payload, run_state)
|
|
529
|
-
|
|
530
|
-
def agent_start(self, task_type: str, task_params: dict, frid: str, module_name: str, run_state: RunState):
|
|
531
|
-
"""Start a server-side agent session; returns the first turn (tool calls or completion)."""
|
|
532
|
-
endpoint_url = f"{self.api_url}/agent/start"
|
|
533
|
-
headers = {"X-API-Key": self.api_key, "Content-Type": "application/json"}
|
|
534
|
-
payload = {
|
|
535
|
-
"task_type": task_type,
|
|
536
|
-
"task_params": task_params,
|
|
537
|
-
"frid": frid,
|
|
538
|
-
"module_name": module_name,
|
|
539
|
-
}
|
|
540
|
-
return self.post_request(endpoint_url, headers, payload, run_state)
|
|
541
|
-
|
|
542
|
-
def agent_continue(
|
|
543
|
-
self, session_id: str, tool_results: list[dict], frid: str, module_name: str, run_state: RunState
|
|
544
|
-
):
|
|
545
|
-
"""Feed tool results into an agent session and run its next turn."""
|
|
546
|
-
endpoint_url = f"{self.api_url}/agent/continue"
|
|
547
|
-
headers = {"X-API-Key": self.api_key, "Content-Type": "application/json"}
|
|
548
|
-
payload = {
|
|
549
|
-
"session_id": session_id,
|
|
550
|
-
"tool_results": tool_results,
|
|
551
|
-
"frid": frid,
|
|
552
|
-
"module_name": module_name,
|
|
553
|
-
}
|
|
554
|
-
return self.post_request(endpoint_url, headers, payload, run_state)
|
|
@@ -1,256 +1,79 @@
|
|
|
1
|
-
import os
|
|
2
|
-
from http import HTTPStatus
|
|
3
1
|
from typing import Any
|
|
4
2
|
|
|
5
|
-
import requests
|
|
6
|
-
|
|
7
3
|
import file_utils
|
|
8
|
-
import
|
|
4
|
+
import render_machine.render_utils as render_utils
|
|
9
5
|
from plain2code_console import console
|
|
10
6
|
from plain2code_exceptions import InternalClientError
|
|
11
|
-
from render_machine import agent_tools
|
|
12
7
|
from render_machine.actions.base_action import BaseAction
|
|
8
|
+
from render_machine.implementation_code_helpers import ImplementationCodeHelpers
|
|
13
9
|
from render_machine.render_context import RenderContext
|
|
14
10
|
|
|
15
|
-
|
|
16
|
-
SUBMIT_FIX_TOOL = "submit_fix"
|
|
17
|
-
# Upper bound on LLM turns spent on one fix attempt; the server bounds the whole session.
|
|
18
|
-
MAX_AGENT_TURNS_PER_ATTEMPT = 40
|
|
19
|
-
# Answer to a submit_fix whose fix was accepted in an earlier unit-test loop of the conformance phase.
|
|
20
|
-
ACCEPTED_THEN_CHANGED_MESSAGE = (
|
|
21
|
-
"Your fix was accepted: the unit tests passed. Afterwards the implementation code was changed to fix the "
|
|
22
|
-
"conformance tests{see_below}, and the unit tests now fail again. Files may have changed since you last read "
|
|
23
|
-
"them."
|
|
24
|
-
)
|
|
25
|
-
# Seeding the first turn: the build folder's file list and the files changed for the FRID.
|
|
26
|
-
MAX_FILE_TREE_ENTRIES = 500
|
|
27
|
-
MAX_RELEVANT_FILES_CHARS = 60_000
|
|
11
|
+
MAX_ISSUE_LENGTH = 10000
|
|
28
12
|
|
|
29
13
|
|
|
30
14
|
class FixUnitTests(BaseAction):
|
|
31
|
-
"""Fix failing unit tests with a server-side agent session that spans the fix attempts.
|
|
32
|
-
|
|
33
|
-
The first failure starts a session; the agent then drives read/grep/edit/run tool calls
|
|
34
|
-
(executed here) until it calls submit_fix. The state machine re-runs the unit tests and,
|
|
35
|
-
if they still fail, the next execution of this action answers that submit_fix call with
|
|
36
|
-
the new failure output inside the same session, so earlier attempts stay in context.
|
|
37
|
-
|
|
38
|
-
During the conformance phase the session also spans unit-test loops: after the conformance
|
|
39
|
-
tests fixer changes implementation code, the unit tests fail again and the same session is
|
|
40
|
-
continued, told what the conformance tests fixer changed (which it must preserve).
|
|
41
|
-
"""
|
|
42
|
-
|
|
43
15
|
SUCCESSFUL_OUTCOME = "unit_tests_fix_generated"
|
|
44
16
|
|
|
45
17
|
def execute(self, render_context: RenderContext, previous_action_payload: Any | None):
|
|
46
|
-
if not previous_action_payload
|
|
18
|
+
if not previous_action_payload.get("previous_unittests_issue"):
|
|
47
19
|
raise InternalClientError(
|
|
48
20
|
"Internal client error: Previous action payload does not contain previous unit tests issue."
|
|
49
21
|
)
|
|
50
|
-
|
|
51
|
-
context = render_context.unit_tests_running_context
|
|
52
|
-
session = render_context.unit_tests_agent_session
|
|
53
|
-
api = render_context.codeplain_api
|
|
54
|
-
frid, module_name = render_context.frid_context.frid, render_context.module_name
|
|
55
|
-
changed_files_before = set(context.changed_files)
|
|
56
|
-
log_path = render_context.script_execution_history.latest_unit_test_output_path
|
|
57
|
-
if log_path:
|
|
58
|
-
agent_tools.register_log_path(log_path, render_context)
|
|
59
|
-
conformance_tests_fixes = self._get_conformance_tests_fixes(render_context)
|
|
60
|
-
new_conformance_tests_fixes = conformance_tests_fixes[session.conformance_fixes_handed_off :]
|
|
61
|
-
if new_conformance_tests_fixes:
|
|
62
|
-
console.info(
|
|
63
|
-
f"Unit tests are fixed while preserving {len(new_conformance_tests_fixes)} implementation code "
|
|
64
|
-
"change(s) made to fix the conformance tests."
|
|
65
|
-
)
|
|
22
|
+
previous_unittests_issue = previous_action_payload["previous_unittests_issue"]
|
|
66
23
|
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
console.info(f"Continuing agent session {session.session_id} with the new unit tests failure.")
|
|
71
|
-
output = "The fix was applied, but the unit tests still fail."
|
|
72
|
-
else:
|
|
73
|
-
# The session's last fix was accepted in an earlier unit-test loop of this conformance
|
|
74
|
-
# phase; since then the conformance tests fixer changed the code and the tests fail again.
|
|
75
|
-
console.info(
|
|
76
|
-
f"Continuing agent session {session.session_id}: the unit tests fail again after the "
|
|
77
|
-
"implementation was changed to fix the conformance tests."
|
|
78
|
-
)
|
|
79
|
-
output = ACCEPTED_THEN_CHANGED_MESSAGE.format(
|
|
80
|
-
see_below=" (see the Conformance Tests Fix below)" if new_conformance_tests_fixes else ""
|
|
81
|
-
)
|
|
82
|
-
# Files changed outside the session, so earlier read results are stale.
|
|
83
|
-
context.tool_result_cache.clear()
|
|
84
|
-
submit_result: dict = {
|
|
85
|
-
"call_id": session.pending_submit_call_id,
|
|
86
|
-
"output": output + agent_tools.full_log_pointer(log_path),
|
|
87
|
-
"test_output": unittests_issue,
|
|
88
|
-
}
|
|
89
|
-
if new_conformance_tests_fixes:
|
|
90
|
-
submit_result["conformance_tests_fixes"] = new_conformance_tests_fixes
|
|
91
|
-
tool_results = session.pending_tool_results + [submit_result]
|
|
92
|
-
session.pending_tool_results, session.pending_submit_call_id = [], None
|
|
93
|
-
response = self._continue_session(render_context, session.session_id, tool_results)
|
|
94
|
-
if response is None:
|
|
95
|
-
console.info("Starting an agent session to fix the unit tests.")
|
|
96
|
-
# Cached read results point at earlier turns, which a new session does not have.
|
|
97
|
-
context.tool_result_cache.clear()
|
|
98
|
-
response = api.agent_start(
|
|
99
|
-
TASK_TYPE,
|
|
100
|
-
self._build_task_params(render_context, unittests_issue, conformance_tests_fixes),
|
|
101
|
-
frid,
|
|
102
|
-
module_name,
|
|
103
|
-
render_context.run_state,
|
|
24
|
+
if previous_unittests_issue and len(previous_unittests_issue) > MAX_ISSUE_LENGTH:
|
|
25
|
+
console.debug(
|
|
26
|
+
f"Unit tests issue text is too long and will be smartly truncated to {MAX_ISSUE_LENGTH} characters."
|
|
104
27
|
)
|
|
105
|
-
session.session_id = response["session_id"]
|
|
106
|
-
assert session.session_id is not None
|
|
107
|
-
session.conformance_fixes_handed_off = len(conformance_tests_fixes)
|
|
108
|
-
context.agent_used_in_this_loop = True
|
|
109
28
|
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
29
|
+
existing_files, existing_files_content = ImplementationCodeHelpers.fetch_existing_files(
|
|
30
|
+
render_context.build_folder
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
render_utils.print_inputs(render_context, existing_files_content, "Files sent as input to unit tests fixing:")
|
|
34
|
+
|
|
35
|
+
conformance_tests_fixes = self._get_conformance_tests_fixes(render_context)
|
|
36
|
+
if conformance_tests_fixes:
|
|
37
|
+
console.info(
|
|
38
|
+
f"Unit tests are fixed while preserving {len(conformance_tests_fixes)} implementation code change(s) "
|
|
39
|
+
"made to fix the conformance tests."
|
|
118
40
|
)
|
|
119
|
-
if submit_call is not None:
|
|
120
|
-
session.pending_submit_call_id = submit_call["id"]
|
|
121
|
-
session.pending_tool_results = tool_results
|
|
122
|
-
submitted = True
|
|
123
|
-
console.info(f"Agent submitted a fix: {submit_call['args'].get('changes_made', '')}")
|
|
124
|
-
break
|
|
125
|
-
response = api.agent_continue(session.session_id, tool_results, frid, module_name, render_context.run_state)
|
|
126
41
|
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
42
|
+
response_files = render_context.codeplain_api.fix_unittests_issue(
|
|
43
|
+
render_context.frid_context.frid,
|
|
44
|
+
render_context.plain_source_tree,
|
|
45
|
+
render_context.frid_context.linked_resources,
|
|
46
|
+
existing_files_content,
|
|
47
|
+
render_context.module_name,
|
|
48
|
+
render_context.get_required_modules_functionalities(),
|
|
49
|
+
previous_unittests_issue,
|
|
50
|
+
run_state=render_context.run_state,
|
|
51
|
+
conformance_tests_fixes=conformance_tests_fixes,
|
|
52
|
+
)
|
|
136
53
|
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
render_context.build_folder,
|
|
140
|
-
{path: "" for path in sorted(context.changed_files - changed_files_before)},
|
|
141
|
-
style=console.OUTPUT_STYLE,
|
|
54
|
+
_, changed_files = file_utils.update_build_folder_with_rendered_files(
|
|
55
|
+
render_context.build_folder, existing_files, response_files
|
|
142
56
|
)
|
|
143
|
-
return self.SUCCESSFUL_OUTCOME, None
|
|
144
57
|
|
|
145
|
-
|
|
146
|
-
def _get_conformance_tests_fixes(render_context: RenderContext) -> list[dict]:
|
|
147
|
-
"""Implementation code changes the conformance tests fixer made during this conformance phase, oldest
|
|
148
|
-
first. Empty outside the conformance phase (the implementation and refactoring unit-test loops)."""
|
|
149
|
-
conformance_tests_running_context = getattr(render_context, "conformance_tests_running_context", None)
|
|
150
|
-
if conformance_tests_running_context is None:
|
|
151
|
-
return []
|
|
152
|
-
return list(getattr(conformance_tests_running_context, "implementation_code_fixes", None) or [])
|
|
58
|
+
render_context.unit_tests_running_context.changed_files.update(changed_files)
|
|
153
59
|
|
|
154
|
-
|
|
155
|
-
def _continue_session(render_context: RenderContext, session_id: str, tool_results: list[dict]) -> dict | None:
|
|
156
|
-
"""Continue the session; None if the server no longer has it (expired), so a new one is started."""
|
|
157
|
-
try:
|
|
158
|
-
return render_context.codeplain_api.agent_continue(
|
|
159
|
-
session_id,
|
|
160
|
-
tool_results,
|
|
161
|
-
render_context.frid_context.frid,
|
|
162
|
-
render_context.module_name,
|
|
163
|
-
render_context.run_state,
|
|
164
|
-
)
|
|
165
|
-
except requests.exceptions.HTTPError as e:
|
|
166
|
-
if e.response is None or e.response.status_code != HTTPStatus.NOT_FOUND:
|
|
167
|
-
raise
|
|
168
|
-
console.warning(f"Agent session {session_id} has expired; starting a new one.")
|
|
169
|
-
render_context.unit_tests_agent_session.reset()
|
|
170
|
-
return None
|
|
60
|
+
console.print_files("Files fixed:", render_context.build_folder, response_files, style=console.OUTPUT_STYLE)
|
|
171
61
|
|
|
172
|
-
|
|
173
|
-
def _build_task_params(
|
|
174
|
-
render_context: RenderContext, unittests_issue: str, conformance_tests_fixes: list[dict]
|
|
175
|
-
) -> dict:
|
|
176
|
-
frid = render_context.frid_context.frid
|
|
177
|
-
specifications, _ = plain_spec.get_specifications_for_frid(render_context.plain_source_tree, frid)
|
|
178
|
-
context = render_context.unit_tests_running_context
|
|
179
|
-
# The files the conformance tests fixes changed are always seeded, so the agent sees what to preserve.
|
|
180
|
-
conformance_fix_files = {name for fix in conformance_tests_fixes for name in (fix or {}).get("code_diff") or {}}
|
|
181
|
-
task_params = {
|
|
182
|
-
"definitions": "\n".join(specifications.get(plain_spec.DEFINITIONS, [])),
|
|
183
|
-
"non_functional_requirements": "\n".join(specifications.get(plain_spec.NON_FUNCTIONAL_REQUIREMENTS, [])),
|
|
184
|
-
"functional_requirements": FixUnitTests._functional_requirements_section(render_context, specifications),
|
|
185
|
-
"linked_resources": render_context.frid_context.linked_resources,
|
|
186
|
-
"build_folder": render_context.build_folder,
|
|
187
|
-
"module_name": render_context.module_name,
|
|
188
|
-
"unittests_script_content": FixUnitTests._read_script(render_context.unittests_script),
|
|
189
|
-
"unittests_issue": unittests_issue,
|
|
190
|
-
"unittests_log_path": render_context.script_execution_history.latest_unit_test_output_path,
|
|
191
|
-
"file_tree": FixUnitTests._file_tree(render_context.build_folder),
|
|
192
|
-
"relevant_files": FixUnitTests._relevant_files(
|
|
193
|
-
render_context.build_folder,
|
|
194
|
-
render_context.frid_context.changed_files | context.changed_files | conformance_fix_files,
|
|
195
|
-
),
|
|
196
|
-
}
|
|
197
|
-
if conformance_tests_fixes:
|
|
198
|
-
task_params["conformance_tests_fixes"] = conformance_tests_fixes
|
|
199
|
-
session = render_context.unit_tests_agent_session
|
|
200
|
-
if session.previous_session_id:
|
|
201
|
-
task_params["previous_session_id"] = session.previous_session_id
|
|
202
|
-
return task_params
|
|
62
|
+
return self.SUCCESSFUL_OUTCOME, None
|
|
203
63
|
|
|
204
64
|
@staticmethod
|
|
205
|
-
def
|
|
206
|
-
|
|
207
|
-
for root, dirs, files in os.walk(build_folder):
|
|
208
|
-
dirs[:] = sorted(d for d in dirs if d not in agent_tools.GREP_EXCLUDED_DIRS and not d.startswith("."))
|
|
209
|
-
paths.extend(os.path.relpath(os.path.join(root, name), build_folder) for name in sorted(files))
|
|
210
|
-
if len(paths) > MAX_FILE_TREE_ENTRIES:
|
|
211
|
-
return "\n".join(paths[:MAX_FILE_TREE_ENTRIES]) + "\n... [more files not listed]"
|
|
212
|
-
return "\n".join(paths)
|
|
65
|
+
def _get_conformance_tests_fixes(render_context: RenderContext) -> list[dict] | None:
|
|
66
|
+
"""Implementation code changes the conformance tests fixer made before these unit tests were run.
|
|
213
67
|
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
"""
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
if os.path.isfile(full_path):
|
|
221
|
-
with open(full_path, "r", encoding="utf-8", errors="replace") as f:
|
|
222
|
-
contents[name] = f.read()
|
|
223
|
-
relevant, total = {}, 0
|
|
224
|
-
for name in sorted(contents, key=lambda n: (len(contents[n]), n)):
|
|
225
|
-
if total + len(contents[name]) > MAX_RELEVANT_FILES_CHARS:
|
|
226
|
-
break
|
|
227
|
-
relevant[name] = contents[name]
|
|
228
|
-
total += len(contents[name])
|
|
229
|
-
return dict(sorted(relevant.items()))
|
|
68
|
+
Only present when unit tests are processed inside the conformance tests phase - the implementation
|
|
69
|
+
and refactoring unit test passes have no conformance tests running context.
|
|
70
|
+
"""
|
|
71
|
+
conformance_tests_running_context = getattr(render_context, "conformance_tests_running_context", None)
|
|
72
|
+
if conformance_tests_running_context is None:
|
|
73
|
+
return None
|
|
230
74
|
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
for module_name, functionalities in render_context.get_required_modules_functionalities().items():
|
|
235
|
-
sections.append(
|
|
236
|
-
f"### Module: {module_name} (Already Implemented, for context)\n" + "\n".join(functionalities)
|
|
237
|
-
)
|
|
238
|
-
current = specifications.get(plain_spec.FUNCTIONAL_REQUIREMENTS, [])
|
|
239
|
-
if len(current) > 1:
|
|
240
|
-
sections.append(
|
|
241
|
-
f"### Module: {render_context.module_name} (Already Implemented, for context)\n"
|
|
242
|
-
+ "\n".join(current[:-1])
|
|
243
|
-
)
|
|
244
|
-
if current:
|
|
245
|
-
sections.append(f"### Module: {render_context.module_name} (Currently Being Implemented)\n{current[-1]}")
|
|
246
|
-
return "\n\n".join(sections)
|
|
75
|
+
implementation_code_fixes = getattr(conformance_tests_running_context, "implementation_code_fixes", None)
|
|
76
|
+
if not implementation_code_fixes:
|
|
77
|
+
return None
|
|
247
78
|
|
|
248
|
-
|
|
249
|
-
def _read_script(script: str | None) -> str:
|
|
250
|
-
if not script:
|
|
251
|
-
return ""
|
|
252
|
-
try:
|
|
253
|
-
with open(file_utils.add_current_path_if_no_path(script), "r", encoding="utf-8") as f:
|
|
254
|
-
return f.read()
|
|
255
|
-
except OSError:
|
|
256
|
-
return ""
|
|
79
|
+
return list(implementation_code_fixes)
|
|
@@ -16,15 +16,6 @@ class RunUnitTests(BaseAction):
|
|
|
16
16
|
UNRECOVERABLE_ERROR_OUTCOME = "unrecoverable_error_occurred"
|
|
17
17
|
|
|
18
18
|
def execute(self, render_context: RenderContext, _previous_action_payload: Any | None):
|
|
19
|
-
context = render_context.unit_tests_running_context
|
|
20
|
-
if context.verified_passing:
|
|
21
|
-
# The fixing agent's own run passed and no file changed since; running again is redundant.
|
|
22
|
-
context.verified_passing = False
|
|
23
|
-
console.info("Unit tests already passed in the fixing agent's last run; not running them again.")
|
|
24
|
-
render_context.script_execution_history.latest_unit_test_output_path = context.verified_passing_log_path
|
|
25
|
-
render_context.script_execution_history.should_update_script_outputs = True
|
|
26
|
-
return self.SUCCESSFUL_OUTCOME, None
|
|
27
|
-
|
|
28
19
|
unittests_script = os.path.normpath(render_context.unittests_script)
|
|
29
20
|
|
|
30
21
|
console.info(
|
render_machine/render_context.py
CHANGED
|
@@ -19,7 +19,6 @@ from render_machine.render_types import (
|
|
|
19
19
|
FridContext,
|
|
20
20
|
ScriptExecutionHistory,
|
|
21
21
|
TestExecutionPhase,
|
|
22
|
-
UnitTestsAgentSession,
|
|
23
22
|
UnitTestsRunningContext,
|
|
24
23
|
)
|
|
25
24
|
|
|
@@ -189,14 +188,6 @@ class RenderContext:
|
|
|
189
188
|
def should_run_conformance_tests(self) -> bool:
|
|
190
189
|
return self.conformance_tests_script is not None
|
|
191
190
|
|
|
192
|
-
@property
|
|
193
|
-
def unit_tests_agent_session(self) -> UnitTestsAgentSession:
|
|
194
|
-
"""The agent session fixing the unit tests: per unit-test loop, except during the conformance
|
|
195
|
-
phase, where one session spans every loop (see UnitTestsAgentSession)."""
|
|
196
|
-
if self.conformance_tests_running_context is not None:
|
|
197
|
-
return self.conformance_tests_running_context.unit_tests_agent_session
|
|
198
|
-
return self.unit_tests_running_context.agent_session
|
|
199
|
-
|
|
200
191
|
def start_unittests_processing(self):
|
|
201
192
|
self.unit_tests_running_context = UnitTestsRunningContext(fix_attempts=0)
|
|
202
193
|
self.run_state.increment_unittest_batch_id()
|
render_machine/render_types.py
CHANGED
|
@@ -48,53 +48,10 @@ class FridContext:
|
|
|
48
48
|
refactoring_iteration: int = 0
|
|
49
49
|
|
|
50
50
|
|
|
51
|
-
@dataclass
|
|
52
|
-
class UnitTestsAgentSession:
|
|
53
|
-
"""Server-side agent session fixing a FRID's unit tests, and what it still has to be told.
|
|
54
|
-
|
|
55
|
-
Owned by the unit-tests running context (one session per unit-test loop) in the implementation
|
|
56
|
-
and refactoring phases, and by the conformance tests running context during the conformance
|
|
57
|
-
phase, so there one session spans every unit-test loop - the agent then sees that the
|
|
58
|
-
conformance tests fixer keeps changing the code it adjusts (see RenderContext.unit_tests_agent_session).
|
|
59
|
-
"""
|
|
60
|
-
|
|
61
|
-
session_id: Optional[str] = None
|
|
62
|
-
# The submit_fix call the agent ended its last attempt with, answered with the next test
|
|
63
|
-
# run's outcome, plus results of any tool calls made in the same turn as submit_fix.
|
|
64
|
-
pending_submit_call_id: Optional[str] = None
|
|
65
|
-
pending_tool_results: list[dict] = field(default_factory=list)
|
|
66
|
-
# Session abandoned without a submission (turn budget used up, LLM failure); the next session
|
|
67
|
-
# starts with a digest of what it tried.
|
|
68
|
-
previous_session_id: Optional[str] = None
|
|
69
|
-
# Full test logs the agent was pointed to; readable by read_file/grep although outside the
|
|
70
|
-
# build folder.
|
|
71
|
-
readable_log_paths: set[str] = field(default_factory=set)
|
|
72
|
-
# How many of the conformance tests fixes (ConformanceTestsRunningContext.implementation_code_fixes)
|
|
73
|
-
# the session has already been shown.
|
|
74
|
-
conformance_fixes_handed_off: int = 0
|
|
75
|
-
|
|
76
|
-
def reset(self) -> None:
|
|
77
|
-
"""Drop the session (keeping its id as the previous one) so the next failure starts a new one."""
|
|
78
|
-
self.previous_session_id = self.session_id
|
|
79
|
-
self.session_id, self.pending_submit_call_id, self.pending_tool_results = None, None, []
|
|
80
|
-
self.conformance_fixes_handed_off = 0
|
|
81
|
-
|
|
82
|
-
|
|
83
51
|
@dataclass
|
|
84
52
|
class UnitTestsRunningContext:
|
|
85
53
|
fix_attempts: int
|
|
86
54
|
changed_files: set[str] = field(default_factory=set)
|
|
87
|
-
# The agent session of this loop; used only outside the conformance phase.
|
|
88
|
-
agent_session: UnitTestsAgentSession = field(default_factory=UnitTestsAgentSession)
|
|
89
|
-
# Whether the agent already made a fix attempt in this loop. A session that is still open when
|
|
90
|
-
# a loop starts had its last fix accepted, which is what the agent is told.
|
|
91
|
-
agent_used_in_this_loop: bool = False
|
|
92
|
-
# Set when the agent's own run_unit_tests passed and no file changed since, so the harness
|
|
93
|
-
# can accept the fix without running the suite again.
|
|
94
|
-
verified_passing: bool = False
|
|
95
|
-
verified_passing_log_path: Optional[str] = None
|
|
96
|
-
# Results of read-only tool calls, keyed by call; cleared whenever a file changes.
|
|
97
|
-
tool_result_cache: dict[str, str] = field(default_factory=dict)
|
|
98
55
|
|
|
99
56
|
|
|
100
57
|
class ConformanceTestsRunningContext:
|
|
@@ -138,8 +95,6 @@ class ConformanceTestsRunningContext:
|
|
|
138
95
|
# order. Each entry is {"hypothesis": str | None, "approach": str | None, "code_diff": {file: diff}}.
|
|
139
96
|
# Handed to the unit tests fixer so it adjusts the unit tests instead of reverting these changes.
|
|
140
97
|
self.implementation_code_fixes: list[dict] = []
|
|
141
|
-
# The unit-test fixing agent session, shared by every unit-test loop of this conformance phase.
|
|
142
|
-
self.unit_tests_agent_session = UnitTestsAgentSession()
|
|
143
98
|
|
|
144
99
|
def get_conformance_tests_json(self, module_name: str) -> dict:
|
|
145
100
|
return self._conformance_tests_json[module_name]
|