ltcai 11.4.0 → 11.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. package/README.md +55 -44
  2. package/docs/CHANGELOG.md +48 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/ONBOARDING.md +1 -1
  6. package/docs/OPERATIONS.md +1 -1
  7. package/docs/TRUST_MODEL.md +1 -1
  8. package/docs/WHY_LATTICE.md +1 -1
  9. package/docs/kg-schema.md +1 -1
  10. package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +11 -6
  11. package/docs/v11.5.0_RUST_COMPLETE_PLAN.md +145 -0
  12. package/docs/v11.5.1_RUST_FULL_LOOP_PLAN.md +85 -0
  13. package/lattice_brain/__init__.py +1 -1
  14. package/lattice_brain/runtime/multi_agent.py +1 -1
  15. package/latticeai/__init__.py +1 -1
  16. package/latticeai/api/agent_worker_seam.py +389 -0
  17. package/latticeai/api/index_jobs.py +145 -0
  18. package/latticeai/core/legacy_compatibility.py +1 -1
  19. package/latticeai/core/marketplace.py +1 -1
  20. package/latticeai/core/messages.py +40 -0
  21. package/latticeai/core/workspace_os_constants.py +1 -1
  22. package/latticeai/runtime/build_phases/features.py +37 -2
  23. package/latticeai/services/architecture_readiness.py +1 -1
  24. package/latticeai/services/product_readiness.py +1 -1
  25. package/package.json +1 -1
  26. package/scripts/check_current_release_docs.mjs +1 -1
  27. package/scripts/check_server_i18n.mjs +2 -0
  28. package/scripts/chunking_parity_corpus.py +449 -0
  29. package/scripts/generate_agent_loop_fixtures.py +907 -0
  30. package/scripts/generate_agent_parity_fixtures.py +752 -0
  31. package/scripts/generate_chunking_parity_fixtures.py +259 -0
  32. package/scripts/generate_rust_parity_fixtures.py +541 -103
  33. package/scripts/parity_fixture_corpus_docgen.py +341 -0
  34. package/scripts/release_screen_claims.json +21 -0
  35. package/src-tauri/Cargo.lock +53 -4
  36. package/src-tauri/Cargo.toml +11 -4
  37. package/src-tauri/src/backend.rs +251 -140
  38. package/src-tauri/src/main.rs +16 -4
  39. package/src-tauri/src/topology.rs +356 -0
  40. package/src-tauri/tauri.conf.json +1 -1
  41. package/static/app/asset-manifest.json +41 -41
  42. package/static/app/assets/{Act-yYpYnn0v.js → Act-Drs_jd-O.js} +1 -1
  43. package/static/app/assets/{AdminConsole-DL3Cr5pL.js → AdminConsole-BqNx5oF6.js} +1 -1
  44. package/static/app/assets/{Brain-C1HBN0Wf.js → Brain-C7bdNmz-.js} +1 -1
  45. package/static/app/assets/{BrainHome-DoXRhUUC.js → BrainHome-DcTnINcI.js} +1 -1
  46. package/static/app/assets/{BrainSignals-6yR6ir5t.js → BrainSignals-Df1MRIPD.js} +1 -1
  47. package/static/app/assets/{Capture-CFIRsFNE.js → Capture-Dgsxvnbx.js} +1 -1
  48. package/static/app/assets/{Chronicle-BZbEgiwN.js → Chronicle-BrkTxJK1.js} +1 -1
  49. package/static/app/assets/{CommandPalette-D2pMxC2I.js → CommandPalette-DDziw7Oj.js} +1 -1
  50. package/static/app/assets/{Library-DwO3yZST.js → Library-CYxeuraA.js} +1 -1
  51. package/static/app/assets/{LivingBrain-Jn1GK0-S.js → LivingBrain-CNp6mvCm.js} +1 -1
  52. package/static/app/assets/{ProductFlow-B-w1R4Oo.js → ProductFlow-BVsowu3Z.js} +1 -1
  53. package/static/app/assets/{ReviewCard-6B27X8Vg.js → ReviewCard-D1LbOfJS.js} +1 -1
  54. package/static/app/assets/{System-DW8F-2xL.js → System-DcRAq7wj.js} +1 -1
  55. package/static/app/assets/arrow-left-BHYmOTWU.js +1 -0
  56. package/static/app/assets/{bot-IM_E_Y12.js → bot-BK2xQ9mN.js} +1 -1
  57. package/static/app/assets/{brain-Ci1CkWjM.js → brain-BZNztFBb.js} +1 -1
  58. package/static/app/assets/{button-COwyqfHM.js → button-CjueubVZ.js} +1 -1
  59. package/static/app/assets/circle-check-CLeWJr67.js +1 -0
  60. package/static/app/assets/{circle-pause-DEM4A1Y5.js → circle-pause-HRjSmpyY.js} +1 -1
  61. package/static/app/assets/{circle-play-C9djDuLd.js → circle-play-CMcKDVwu.js} +1 -1
  62. package/static/app/assets/{cpu-DFdo1gw-.js → cpu-B_MyTfYr.js} +1 -1
  63. package/static/app/assets/{download-SnJL6oqk.js → download-CTy0EByV.js} +1 -1
  64. package/static/app/assets/{folder-open-CqZeDkjE.js → folder-open-DF6bLhTg.js} +1 -1
  65. package/static/app/assets/{hard-drive-j1jJXYYf.js → hard-drive-CpWHG77C.js} +1 -1
  66. package/static/app/assets/{index-_u5iUHDr.js → index-Dd6abJHX.js} +3 -3
  67. package/static/app/assets/index-DxmOfNRi.css +2 -0
  68. package/static/app/assets/{input-B0lPdRQZ.js → input-bVgIRN3s.js} +1 -1
  69. package/static/app/assets/{link-2-CoFbooHS.js → link-2-bg6CG5Vk.js} +1 -1
  70. package/static/app/assets/{permissionCopy-BsyLxtao.js → permissionCopy-DFTf1HDZ.js} +1 -1
  71. package/static/app/assets/{primitives-DEbN-d6p.js → primitives-D7D-sQ3T.js} +1 -1
  72. package/static/app/assets/search-BgSVX6NG.js +1 -0
  73. package/static/app/assets/{share-2-CVtZ_ewX.js → share-2-BLzo7U4L.js} +1 -1
  74. package/static/app/assets/{shield-alert-CBi2GNWM.js → shield-alert-BmngqnsJ.js} +1 -1
  75. package/static/app/assets/{textarea-DNMpB5ih.js → textarea-DOSlAQQ3.js} +1 -1
  76. package/static/app/assets/{useFocusTrap-C83t3GXF.js → useFocusTrap-pZhgeee4.js} +1 -1
  77. package/static/app/assets/{useMutation-DtbJDoyz.js → useMutation-BMDwNk4I.js} +1 -1
  78. package/static/app/assets/{useQuery-Dcp1OChy.js → useQuery-CSjKttRo.js} +1 -1
  79. package/static/app/assets/{utils-BlZr7Pd4.js → utils-D3u_yv7B.js} +1 -1
  80. package/static/app/assets/{workspace-jJY4RuAV.js → workspace-H3bMjYXC.js} +1 -1
  81. package/static/app/index.html +4 -4
  82. package/static/sw.js +1 -1
  83. package/static/app/assets/arrow-left-DXvKg9U6.js +0 -1
  84. package/static/app/assets/circle-check-DfInj-qD.js +0 -1
  85. package/static/app/assets/index-BLPb5lmE.css +0 -2
  86. package/static/app/assets/search-BybIWPNd.js +0 -1
@@ -0,0 +1,907 @@
1
+ #!/usr/bin/env python3
2
+ """Build the committed Python↔Rust **agent loop** parity fixtures (v11.5.1).
3
+
4
+ ``rust/lattice-agent`` now owns the PLAN → EXECUTE → VERIFY → ROLLBACK
5
+ orchestration that ``latticeai.core.agent`` has always owned, with the Python
6
+ worker behind three seam endpoints. A port of a *state machine* is only worth
7
+ having if something keeps proving it still reaches the same states, by the same
8
+ route, with the same record — so this script is the Python half of that proof.
9
+
10
+ It runs the **real** functions, never a re-description of them:
11
+
12
+ * the deterministic helpers — ``extract_action_details``, ``normalize_plan``,
13
+ ``infer_file_target`` / ``infer_project_manifest``, ``requirement_coverage``,
14
+ ``artifact_checklist``, ``files_written``, ``compact_transcript``,
15
+ ``_truncate_strings``, ``filter_learnings``, ``PhaseBudgets`` /
16
+ ``TranscriptBudget``;
17
+ * the verification verdict mapping, by calling the real ``verify()`` over a
18
+ verdict × evidence × coverage × retry grid with a scripted critic;
19
+ * the run store's ``serialize_run_context`` / ``restore_run_context``;
20
+ * and **end-to-end trajectories**: the real :class:`SingleAgentRuntime`, driven
21
+ by a scripted LLM and the real tool registry inside a throwaway
22
+ ``AGENT_ROOT``, for seven scenarios that between them exercise every branch
23
+ the loop can take to a terminal state.
24
+
25
+ Two consumers read what it writes:
26
+
27
+ * ``tests/unit/test_agent_loop_parity_contract.py`` re-runs the Python loop over
28
+ the same scripts and asserts the committed goldens still hold — so a change to
29
+ a Python gate fails loudly instead of silently invalidating the contract the
30
+ Rust side is pinned to;
31
+ * ``rust/lattice-agent/tests/agent_loop.rs`` drives the native loop against a
32
+ fake worker replaying the recorded completions and tool results, and asserts
33
+ the same trajectories.
34
+
35
+ Determinism is the design constraint. Four normalisation rules make the record
36
+ machine-independent, and both sides apply them:
37
+
38
+ 1. the absolute workspace root becomes ``<AGENT_ROOT>``;
39
+ 2. ``at`` keys are dropped (trace timestamps);
40
+ 3. ``stderr`` keys are dropped (git's text is version- and locale-specific);
41
+ 4. a JSON decoder detail is collapsed to ``<decoder-detail>`` — CPython renamed
42
+ several of those messages in 3.14, and pinning a Python patch release is not
43
+ what this contract is for.
44
+
45
+ Usage::
46
+
47
+ .venv/bin/python scripts/generate_agent_loop_fixtures.py
48
+ """
49
+
50
+ from __future__ import annotations
51
+
52
+ import asyncio
53
+ import copy
54
+ import json
55
+ import os
56
+ import sys
57
+ import tempfile
58
+ from contextlib import contextmanager
59
+ from pathlib import Path
60
+ from typing import Any, Dict, Iterator, List, Optional
61
+
62
+ REPO_ROOT = Path(__file__).resolve().parents[1]
63
+ if str(REPO_ROOT) not in sys.path:
64
+ sys.path.insert(0, str(REPO_ROOT))
65
+
66
+ # The tool registry resolves AGENT_ROOT at import; point it somewhere harmless
67
+ # before anything imports it, then patch it per run through `use_workspace`.
68
+ os.environ.setdefault(
69
+ "LATTICEAI_AGENT_ROOT", str(Path(tempfile.gettempdir()) / "agent-loop-fixtures-import")
70
+ )
71
+
72
+ import latticeai.services.tool_dispatch as tool_dispatch # noqa: E402
73
+ import latticeai.tools as tools # noqa: E402
74
+ from latticeai.api.chat_contracts import AgentRequest # noqa: E402
75
+ from latticeai.core.agent import ( # noqa: E402
76
+ AgentRunContext,
77
+ AgentState,
78
+ SingleAgentRuntime,
79
+ )
80
+ from latticeai.core.agent.deps import AgentDeps # noqa: E402
81
+ from latticeai.core.agent_helpers import ( # noqa: E402
82
+ PhaseBudgets,
83
+ TranscriptBudget,
84
+ _truncate_strings,
85
+ artifact_checklist,
86
+ compact_transcript,
87
+ extract_action_details,
88
+ files_written,
89
+ filter_learnings,
90
+ normalize_plan,
91
+ requirement_coverage,
92
+ )
93
+ from latticeai.core.agent_trace import LoopTrace # noqa: E402
94
+ from latticeai.core.file_generation import ( # noqa: E402
95
+ infer_file_target,
96
+ infer_project_manifest,
97
+ sanitize_write_content,
98
+ )
99
+ from latticeai.core.run_store import ( # noqa: E402
100
+ restore_run_context,
101
+ serialize_run_context,
102
+ )
103
+ from latticeai.core.tool_registry import ( # noqa: E402
104
+ FILE_CREATE_ACTIONS,
105
+ LOCAL_WRITE_BLOCKED_PREFIXES,
106
+ SCOPED_KNOWLEDGE_TOOLS,
107
+ TOOL_GOVERNANCE,
108
+ TOOL_GOVERNANCE_DEFAULT,
109
+ )
110
+
111
+ FIXTURE_DIR = REPO_ROOT / "rust" / "fixtures" / "agent_loop"
112
+ GOLDEN_DIR = FIXTURE_DIR / "golden"
113
+ SCHEMA = "agent-loop-parity/v1"
114
+
115
+ DECODER_DETAIL_PREFIX = "Agent did not return valid JSON: "
116
+ ROOT_PLACEHOLDER = "<AGENT_ROOT>"
117
+
118
+
119
+ # ── normalisation ─────────────────────────────────────────────────────────────
120
+ def normalize(value: Any, root: Path) -> Any:
121
+ """Apply the four machine-independence rules, recursively."""
122
+ if isinstance(value, str):
123
+ text = value.replace(str(root), ROOT_PLACEHOLDER)
124
+ if text.startswith(DECODER_DETAIL_PREFIX):
125
+ return f"{DECODER_DETAIL_PREFIX}<decoder-detail>"
126
+ return text
127
+ if isinstance(value, dict):
128
+ return {
129
+ key: normalize(item, root)
130
+ for key, item in value.items()
131
+ if key not in ("at", "stderr")
132
+ }
133
+ if isinstance(value, list):
134
+ return [normalize(item, root) for item in value]
135
+ return value
136
+
137
+
138
+ @contextmanager
139
+ def use_workspace(root: Path) -> Iterator[Path]:
140
+ """Point the real tool registry at ``root`` for the duration.
141
+
142
+ Two modules hold the constant: ``latticeai.tools`` defines it and
143
+ ``latticeai.services.tool_dispatch`` imported the value, so patching one
144
+ would leave the snapshot/rollback ports reading the real workspace.
145
+ """
146
+ root = Path(root)
147
+ root.mkdir(parents=True, exist_ok=True)
148
+ resolved = root.resolve()
149
+ previous = (tools.AGENT_ROOT, tool_dispatch.AGENT_ROOT)
150
+ tools.AGENT_ROOT = resolved
151
+ tool_dispatch.AGENT_ROOT = resolved
152
+ try:
153
+ yield resolved
154
+ finally:
155
+ tools.AGENT_ROOT, tool_dispatch.AGENT_ROOT = previous
156
+
157
+
158
+ # ── the deterministic helper grids ────────────────────────────────────────────
159
+ #: Raw model outputs chosen for the rung of the tolerance chain each one reaches.
160
+ RAW_ACTIONS: Dict[str, str] = {
161
+ "clean": '{"action": "final", "message": "done"}',
162
+ "clean_with_args": '{"thoughts": "t", "action": "write_file", "args": {"path": "a.md"}}',
163
+ "fence_json": '```json\n{"action": "read_file", "args": {"path": "a.md"}}\n```',
164
+ "fence_bare": '```\n{"action": "final"}\n```',
165
+ "fence_with_prose": 'Sure!\n```json\n{"action": "final"}\n```\nHope that helps.',
166
+ "fence_multiline": '```json\n{\n "action": "final",\n "message": "ok"\n}\n```',
167
+ "think_then_json": '<think>hmm {"action": "wrong"}</think>\n{"action": "right"}',
168
+ "thinking_tag": '<thinking>plan</thinking>{"action": "final"}',
169
+ "reasoning_tag": '<reasoning>why</reasoning>\n{"action": "final"}',
170
+ "think_uppercase": '<THINK>x</THINK>{"action": "final"}',
171
+ "think_unclosed": '<think>never closed {"action": "final"}',
172
+ "think_mismatched": '<think>{"action": "a"}</reasoning>',
173
+ "slice_prefix": 'I will call: {"action": "write_file"}',
174
+ "slice_suffix": '{"action": "write_file"} — that is the call.',
175
+ "slice_both": 'Calling {"action": "final", "message": "완료"} now.',
176
+ "slice_nested": 'note {"action": "a", "args": {"b": {"c": 1}}} end',
177
+ "trailing_comma_object": '{"action": "final", "message": "x",}',
178
+ "trailing_comma_array": '{"action": "a", "args": {"items": [1, 2,]}}',
179
+ "trailing_comma_nested": '{"action": "a", "args": {"x": 1,},}',
180
+ "python_literal": "{'action': 'write_file', 'args': {'path': 'a.md'}}",
181
+ "python_literal_true": "{'action': 'a', 'ok': True, 'bad': False, 'none': None}",
182
+ "python_literal_trailing": "{'action': 'final', 'message': 'hi',}",
183
+ "python_literal_nested": "{'action': 'a', 'args': {'items': [1, 2.5, 'x']}}",
184
+ "python_literal_escapes": "{'action': 'a', 'note': 'line\\nbreak'}",
185
+ "python_literal_not_dict": "('a', 'b')",
186
+ "broken_prose": "I think we should start by reading the notes.",
187
+ "broken_empty": "",
188
+ "broken_whitespace": " \n ",
189
+ "broken_unclosed": '{"action": "final"',
190
+ "broken_missing_value": '{"action": }',
191
+ "broken_unquoted_key": "{action: 1}",
192
+ "no_action_key": '{"thoughts": "no action here"}',
193
+ "not_an_object": "[1, 2, 3]",
194
+ "bare_number": "42",
195
+ "bare_string": '"just text"',
196
+ "korean_prose_slice": '작업 계획: {"action": "final", "message": "완료"} 끝.',
197
+ "double_object": '{"action": "a"} {"action": "b"}',
198
+ "unicode_thoughts": '{"action": "final", "thoughts": "가나다라마바사"}',
199
+ }
200
+
201
+ #: Plans chosen for the seven normalisation rules and their interactions.
202
+ PLAN_CASES: Dict[str, Dict[str, Any]] = {
203
+ "complete": {
204
+ "plan": {"goal": "g", "steps": [{"action": "read_file", "args": {"path": "a.md"}}],
205
+ "estimated_steps": 1, "requires_approval": False, "rollback_strategy": "none"},
206
+ "message": "read a.md",
207
+ },
208
+ "not_an_object": {"plan": ["nope"], "message": "hi"},
209
+ "null_plan": {"plan": None, "message": "hi"},
210
+ "string_plan": {"plan": "a plan", "message": "hi"},
211
+ "blank_goal": {"plan": {"goal": " ", "steps": []}, "message": "do the thing"},
212
+ "missing_goal": {"plan": {"steps": []}, "message": "do the thing"},
213
+ "numeric_goal": {"plan": {"goal": 5, "steps": []}, "message": "hi"},
214
+ "junk_steps": {
215
+ "plan": {"goal": "g", "steps": ["x", {"no_action": 1}, {"action": ""},
216
+ {"action": "read_file"}]},
217
+ "message": "g",
218
+ },
219
+ "steps_not_a_list": {"plan": {"goal": "g", "steps": "read a file"}, "message": "g"},
220
+ "empty_steps": {"plan": {"goal": "g", "steps": []}, "message": "g"},
221
+ "manifest_empty_plan": {"plan": {"goal": "g", "steps": []},
222
+ "message": "todo 앱 html css js 만들어줘"},
223
+ "manifest_partial": {
224
+ "plan": {"goal": "g", "steps": [{"action": "write_file", "args": {"path": "index.html"}}]},
225
+ "message": "todo 앱 html css js 만들어줘",
226
+ },
227
+ "manifest_covered": {
228
+ "plan": {"goal": "g", "steps": [
229
+ {"action": "write_file", "args": {"path": "page.HTML"}},
230
+ {"action": "write_file", "args": {"path": "a.css"}},
231
+ {"action": "generate_file", "args": {"path": "b.js"}}]},
232
+ "message": "todo 앱 html css js 만들어줘",
233
+ },
234
+ "manifest_with_read": {
235
+ "plan": {"goal": "g", "steps": [{"action": "read_file", "args": {"path": "spec.md"}},
236
+ {"action": "write_file", "args": {"path": "index.html"}}]},
237
+ "message": "todo 앱 html css js 만들어줘",
238
+ },
239
+ "manifest_react": {"plan": {}, "message": "react 로 todo 앱 만들어줘"},
240
+ "manifest_python": {"plan": {}, "message": "mytool 패키지 파이썬으로 만들어줘"},
241
+ "heuristic_single_file": {"plan": {}, "message": "html 파일 만들어줘"},
242
+ "heuristic_long_message": {"plan": {}, "message": "html 파일 만들어줘 " + "가" * 200},
243
+ "estimated_string": {"plan": {"goal": "g", "estimated_steps": "4"}, "message": "g"},
244
+ "estimated_float": {"plan": {"goal": "g", "estimated_steps": 3.7}, "message": "g"},
245
+ "estimated_invalid": {"plan": {"goal": "g", "estimated_steps": "many"}, "message": "g"},
246
+ "estimated_list": {"plan": {"goal": "g", "estimated_steps": [3]}, "message": "g"},
247
+ "coerced_tail": {"plan": {"goal": "g", "requires_approval": "yes", "rollback_strategy": 7},
248
+ "message": "g"},
249
+ "rollback_kept": {"plan": {"goal": "g", "rollback_strategy": "git"}, "message": "g"},
250
+ }
251
+
252
+ #: Requests the two inference functions are asked about.
253
+ INFERENCE_MESSAGES: List[str] = [
254
+ "html 파일 만들어줘", "write me a python script", "csv 저장해줘", "html이 뭐야?",
255
+ "만들어줘", "", " ", "html과 css 만들어줘", "html and css 만들어줘",
256
+ "todo 앱 html+css+js로 만들어줘", "웹페이지 js로 만들어줘", "웹페이지 css로 만들어줘",
257
+ "react 로 todo 앱 만들어줘", "리액트로 만들어줘", "vite 앱 만들어줘",
258
+ "mytool 패키지 파이썬으로 만들어줘", "my-tool 패키지 파이썬으로 생성",
259
+ "파이썬 패키지 만들어줘", "index.html 이랑 style.css 만들어줘",
260
+ "웹사이트 css js 만들어줘", "마크다운 파일 작성해줘", "yaml 만들어줘",
261
+ ]
262
+
263
+ #: Transcripts the artifact/coverage helpers are asked about.
264
+ TRANSCRIPT_CASES: Dict[str, Dict[str, Any]] = {
265
+ "empty": {"message": "todo 앱 html css js 만들어줘", "transcript": []},
266
+ "one_write": {
267
+ "message": "todo 앱 html css js 만들어줘",
268
+ "transcript": [{"state": "EXECUTING", "action": "write_file",
269
+ "args": {"path": "index.html"},
270
+ "result": {"path": "index.html", "bytes": 10}}],
271
+ },
272
+ "all_written": {
273
+ "message": "todo 앱 html css js 만들어줘",
274
+ "transcript": [
275
+ {"state": "EXECUTING", "action": "write_file", "args": {"path": "index.html"},
276
+ "result": {"path": "index.html", "bytes": 10}},
277
+ {"state": "EXECUTING", "action": "write_file", "args": {"path": "sub/STYLE.CSS"},
278
+ "result": {"path": "sub/STYLE.CSS", "bytes": 3}},
279
+ {"state": "EXECUTING", "action": "write_file", "args": {"path": "app.js"},
280
+ "result": {"path": "app.js", "bytes": 3},
281
+ "content_sanitize": {"sanitized": True, "repaired": True}},
282
+ ],
283
+ },
284
+ "blocked_and_repeated": {
285
+ "message": "make a note",
286
+ "transcript": [
287
+ {"state": "EXECUTING", "action": "write_file", "args": {"path": "a.md"},
288
+ "error": "BLOCKED: nope"},
289
+ {"state": "EXECUTING", "action": "write_file", "args": {"path": "a.md"},
290
+ "result": {"path": "a.md", "bytes": 2}},
291
+ {"state": "EXECUTING", "action": "write_file", "args": {"path": "a.md"},
292
+ "result": {"path": "a.md", "bytes": 2}},
293
+ {"state": "VERIFYING", "action": "write_file", "result": {"path": "ignored.md"}},
294
+ {"state": "EXECUTING", "action": "read_file", "result": {"path": "skip.md"}},
295
+ ],
296
+ },
297
+ "requirements_listed": {
298
+ "message": "만들어줘:\n- 다크모드\n* dark mode\n1. 검색 기능\n2) 필터\nfree prose\n- ab",
299
+ "transcript": [],
300
+ },
301
+ "requirements_capped": {
302
+ "message": "\n".join(f"- item number {index}" for index in range(15)),
303
+ "transcript": [],
304
+ },
305
+ "proposed_step": {
306
+ "message": "make a note",
307
+ "transcript": [{"state": "EXECUTING", "action": "write_file", "args": {"path": "a.md"},
308
+ "result": {"proposed": True, "proposal_id": "p1"}}],
309
+ },
310
+ }
311
+
312
+ #: Values `_truncate_strings` is asked about, with the cap each one uses.
313
+ TRUNCATE_CASES: List[Dict[str, Any]] = [
314
+ {"key": "short", "limit": 700, "value": {"a": "hello"}},
315
+ {"key": "exact", "limit": 5, "value": "abcde"},
316
+ {"key": "over", "limit": 5, "value": "abcdefgh"},
317
+ {"key": "korean", "limit": 5, "value": "가" * 10},
318
+ {"key": "nested", "limit": 3, "value": {"a": ["abcdef", {"b": "xyz!"}], "n": 5, "t": True}},
319
+ {"key": "null_and_float", "limit": 2, "value": {"a": None, "b": 1.5, "c": []}},
320
+ ]
321
+
322
+ LEARNING_CASES: List[List[Any]] = [
323
+ ["short", "파일을 만들었습니다", "Successfully created the file",
324
+ "Vite needs the entry script tag before </body> or the app never mounts",
325
+ "VITE NEEDS THE ENTRY SCRIPT TAG BEFORE </BODY> OR THE APP NEVER MOUNTS", None],
326
+ ["작업을 완료했습니다", "task was completed", "file was created",
327
+ "Successfully created the file, but the CSS never loaded because the path was wrong"],
328
+ [],
329
+ [123456789012345, " padded learning that is long enough "],
330
+ ]
331
+
332
+
333
+ #: A root that matches nothing, for grids that never carry a path.
334
+ NO_ROOT = Path("/__no_workspace_in_this_grid__")
335
+
336
+
337
+ def helper_rows() -> Dict[str, Any]:
338
+ """Every deterministic helper, over its grid."""
339
+ actions = []
340
+ for key, raw in RAW_ACTIONS.items():
341
+ try:
342
+ action, repairs = extract_action_details(raw)
343
+ actions.append({"key": key, "raw": raw, "ok": True,
344
+ "action": action, "repairs": repairs})
345
+ except ValueError as exc:
346
+ actions.append({"key": key, "raw": raw, "ok": False, "error": str(exc)})
347
+
348
+ plans = []
349
+ for key, case in PLAN_CASES.items():
350
+ plan, fixes = normalize_plan(copy.deepcopy(case["plan"]), case["message"])
351
+ plans.append({"key": key, "plan": case["plan"], "message": case["message"],
352
+ "normalized": plan, "fixes": fixes})
353
+
354
+ inference = [
355
+ {"message": message,
356
+ "file_target": infer_file_target(message),
357
+ "manifest": infer_project_manifest(message)}
358
+ for message in INFERENCE_MESSAGES
359
+ ]
360
+
361
+ transcripts = []
362
+ for key, case in TRANSCRIPT_CASES.items():
363
+ transcripts.append({
364
+ "key": key,
365
+ "message": case["message"],
366
+ "transcript": case["transcript"],
367
+ "files_written": files_written(case["transcript"], FILE_CREATE_ACTIONS),
368
+ "artifact_checklist": artifact_checklist(case["transcript"], FILE_CREATE_ACTIONS),
369
+ "requirement_coverage": requirement_coverage(
370
+ case["message"], case["transcript"], FILE_CREATE_ACTIONS
371
+ ),
372
+ "compact_window_2": compact_transcript(case["transcript"], window=2, result_chars=40),
373
+ })
374
+
375
+ truncated = [
376
+ {"key": case["key"], "limit": case["limit"], "value": case["value"],
377
+ "truncated": _truncate_strings(case["value"], case["limit"])}
378
+ for case in TRUNCATE_CASES
379
+ ]
380
+
381
+ learnings = [
382
+ {"input": case, "kept": filter_learnings(case)} for case in LEARNING_CASES
383
+ ]
384
+
385
+ return normalize({
386
+ "schema": SCHEMA,
387
+ "extract_action_details": actions,
388
+ "normalize_plan": plans,
389
+ "inference": inference,
390
+ "transcript_helpers": transcripts,
391
+ "truncate_strings": truncated,
392
+ "filter_learnings": learnings,
393
+ "budgets": {
394
+ "phase": PhaseBudgets().__dict__,
395
+ "transcript": TranscriptBudget().__dict__,
396
+ },
397
+ }, NO_ROOT)
398
+
399
+
400
+ # ── the scripted ports ────────────────────────────────────────────────────────
401
+ class ScriptedLLM:
402
+ """`deps.generate_as`, answering a queue of recorded completions."""
403
+
404
+ def __init__(self, outputs: List[str]) -> None:
405
+ self.outputs = list(outputs)
406
+ self.calls: List[Dict[str, Any]] = []
407
+
408
+ async def generate_as(self, model_id: Optional[str] = None, **kwargs: Any) -> str:
409
+ text = self.outputs.pop(0) if self.outputs else ""
410
+ self.calls.append({
411
+ "model_id": model_id,
412
+ "message": kwargs.get("message"),
413
+ "temperature": kwargs.get("temperature"),
414
+ "max_tokens": kwargs.get("max_tokens"),
415
+ })
416
+ return text
417
+
418
+ async def generate(self, **kwargs: Any) -> str:
419
+ return await self.generate_as(None, **kwargs)
420
+
421
+
422
+ class ScriptedGovernor:
423
+ """`deps.change_governor`, answering one fixed verdict."""
424
+
425
+ governed_tools = frozenset({"write_file", "edit_file"})
426
+
427
+ def __init__(self, verdict: Optional[Dict[str, Any]]) -> None:
428
+ self.verdict = verdict
429
+
430
+ def review(self, name: str, args: Dict[str, Any], **kwargs: Any) -> Optional[Dict[str, Any]]:
431
+ return copy.deepcopy(self.verdict)
432
+
433
+
434
+ def build_deps(
435
+ root: Path,
436
+ llm: ScriptedLLM,
437
+ *,
438
+ governor: Optional[ScriptedGovernor],
439
+ tool_calls: List[Dict[str, Any]],
440
+ audit: List[Dict[str, Any]],
441
+ ) -> AgentDeps:
442
+ """The real ports, wired to the throwaway workspace."""
443
+ registry = tools.DEFAULT_TOOL_REGISTRY
444
+ service = tool_dispatch.DEFAULT_TOOL_DISPATCH_SERVICE
445
+
446
+ def execute_tool(name: str, args: Dict[str, Any]) -> Dict[str, Any]:
447
+ record: Dict[str, Any] = {"tool": name, "args": copy.deepcopy(args)}
448
+ try:
449
+ result = tools.execute_tool(name, args)
450
+ except Exception as exc: # noqa: BLE001 — recorded, then re-raised
451
+ record["error"] = str(exc)
452
+ tool_calls.append(record)
453
+ raise
454
+ record["result"] = result
455
+ tool_calls.append(record)
456
+ return result
457
+
458
+ def record_audit(event: str, **details: Any) -> None:
459
+ audit.append({"event": event, **details})
460
+
461
+ return AgentDeps(
462
+ generate_as=llm.generate_as,
463
+ generate=llm.generate,
464
+ execute_tool=execute_tool,
465
+ policy_for=registry.policy_for,
466
+ risk_level=registry.risk_level,
467
+ check_role=lambda name, user: None,
468
+ tool_governance=TOOL_GOVERNANCE,
469
+ file_create_actions=FILE_CREATE_ACTIONS,
470
+ recent_chat_context=lambda **kwargs: "",
471
+ clear_history=lambda keep_last: {"ok": True, "kept": keep_last},
472
+ knowledge_save=lambda *a, **k: None,
473
+ audit=record_audit,
474
+ planner_prompt="",
475
+ executor_prompt="",
476
+ critic_prompt="",
477
+ memory_updater_prompt="",
478
+ agent_root=root,
479
+ rollback_file=service.rollback_file,
480
+ snapshot_file=service.snapshot_file,
481
+ restore_snapshot=service.restore_snapshot,
482
+ hooks=None,
483
+ change_governor=governor,
484
+ phase_budgets=PhaseBudgets(),
485
+ transcript_budget=TranscriptBudget(),
486
+ )
487
+
488
+
489
+ # ── verification mapping grid ─────────────────────────────────────────────────
490
+ #: `(verdict, next_state)` pairs the mapping table distinguishes.
491
+ VERDICT_PAIRS = [
492
+ ("PASS", "DONE"), ("PASS", "COMPLETE"), ("PASS", "EXECUTING"), ("PASS", ""),
493
+ ("FAIL", "DONE"), ("FAIL", "EXECUTING"), ("FAIL", "RETRY"), ("FAIL", "ROLLBACK"),
494
+ ("FAIL", "FAILED"), ("FAIL", "SOMETHING_ELSE"), ("", "DONE"),
495
+ ]
496
+
497
+ EVIDENCE_STEP = {"state": "EXECUTING", "action": "write_file", "args": {"path": "index.html"},
498
+ "result": {"path": "index.html", "bytes": 4}}
499
+ NO_EVIDENCE_STEP = {"state": "EXECUTING", "action": "final", "thoughts": "t"}
500
+
501
+
502
+ async def verification_rows(root: Path) -> List[Dict[str, Any]]:
503
+ """The verdict mapping, from the real `verify()`."""
504
+ rows: List[Dict[str, Any]] = []
505
+ for verdict, next_state in VERDICT_PAIRS:
506
+ for evidence in (True, False):
507
+ for message in ("make a note", "todo 앱 html css js 만들어줘"):
508
+ for retry_count in (0, 3):
509
+ body = json.dumps(
510
+ {"action": "verdict", "verdict": verdict, "next_state": next_state,
511
+ "reason": "because", "corrections": ["be specific"], "confidence": 0.5},
512
+ ensure_ascii=False,
513
+ )
514
+ llm = ScriptedLLM([body])
515
+ ctx = AgentRunContext()
516
+ ctx.trace = LoopTrace()
517
+ ctx.retry_count = retry_count
518
+ ctx.transcript = [copy.deepcopy(
519
+ EVIDENCE_STEP if evidence else NO_EVIDENCE_STEP
520
+ )]
521
+ runtime = SingleAgentRuntime(build_deps(
522
+ root, llm, governor=None, tool_calls=[], audit=[]
523
+ ))
524
+ request = AgentRequest(message=message)
525
+ await runtime.verify(ctx, request, "Korean", "owner@example.com", max_retry=3)
526
+ rows.append({
527
+ "verdict": verdict, "next_state": next_state, "evidence": evidence,
528
+ "message": message, "retry_count": retry_count,
529
+ "final_state": ctx.state.value,
530
+ "final_message": ctx.final_message,
531
+ "retry_count_after": ctx.retry_count,
532
+ "transcript": normalize(ctx.transcript, root),
533
+ })
534
+ # The unparseable critic: one strict retry, then fail closed.
535
+ for outputs, key in (
536
+ (["prose", "still prose"], "never_parses"),
537
+ (["prose", '{"action": "v", "verdict": "PASS", "next_state": "DONE", "reason": "r"}'],
538
+ "strict_retry_recovers"),
539
+ ):
540
+ llm = ScriptedLLM(list(outputs))
541
+ ctx = AgentRunContext()
542
+ ctx.trace = LoopTrace()
543
+ ctx.transcript = [copy.deepcopy(EVIDENCE_STEP)]
544
+ runtime = SingleAgentRuntime(build_deps(root, llm, governor=None, tool_calls=[], audit=[]))
545
+ await runtime.verify(ctx, AgentRequest(message="make a note"), "Korean", "owner", max_retry=3)
546
+ rows.append({
547
+ "verdict": key, "next_state": "", "evidence": True, "message": "make a note",
548
+ "retry_count": 0, "final_state": ctx.state.value,
549
+ "final_message": ctx.final_message, "retry_count_after": ctx.retry_count,
550
+ "transcript": normalize(ctx.transcript, root),
551
+ "llm_calls": len(llm.calls),
552
+ "temperatures": [call["temperature"] for call in llm.calls],
553
+ })
554
+ return rows
555
+
556
+
557
+ # ── run-store round trips ─────────────────────────────────────────────────────
558
+ def run_store_rows() -> List[Dict[str, Any]]:
559
+ """`serialize_run_context` / `restore_run_context`, field for field."""
560
+ rows: List[Dict[str, Any]] = []
561
+
562
+ full = AgentRunContext()
563
+ full.state = AgentState.WAITING_APPROVAL
564
+ full.plan = {"goal": "g", "steps": [{"action": "write_file"}]}
565
+ full.transcript = [{"state": "PLANNING", "goal": "g"}]
566
+ full.retry_count = 2
567
+ full.state_history = ["PLANNING", "WAITING_APPROVAL"]
568
+ full.corrections = ["reply with JSON"]
569
+ full.final_message = "paused"
570
+ full.rollback_log = [{"path": "a.md", "existed": False}]
571
+ full.executing_model = "m-exec"
572
+ full.reviewing_model = "m-review"
573
+ full.approved_by_human = True
574
+ full.permission_mode = "trusted"
575
+ full.trace = LoopTrace(clock=lambda: "PINNED")
576
+ full.trace.llm_call("plan", model="m-exec")
577
+ full.trace.repair("plan", repairs=["fence"])
578
+ rows.append({"key": "full", "serialized": serialize_run_context(full)})
579
+
580
+ empty = AgentRunContext()
581
+ rows.append({"key": "empty", "serialized": serialize_run_context(empty)})
582
+
583
+ for key, payload in (
584
+ ("unknown_state", {"state": "SOMETHING_NEW"}),
585
+ ("no_state", {}),
586
+ ("null_state", {"state": None}),
587
+ ("blank_mode", {"state": "EXECUTING", "permission_mode": ""}),
588
+ ("kept_mode", {"state": "EXECUTING", "permission_mode": "bypass"}),
589
+ ("truthy_approval", {"approved_by_human": 1}),
590
+ ):
591
+ rows.append({"key": f"restore_{key}", "payload": payload,
592
+ "restored": serialize_run_context(restore_run_context(payload))})
593
+
594
+ for row in rows:
595
+ if "serialized" in row:
596
+ row["round_trip"] = serialize_run_context(restore_run_context(row["serialized"]))
597
+ return normalize(rows, NO_ROOT)
598
+
599
+
600
+ # ── end-to-end trajectories ───────────────────────────────────────────────────
601
+ def plan_json(goal: str, steps: List[Dict[str, Any]]) -> str:
602
+ return json.dumps(
603
+ {"action": "plan", "goal": goal, "steps": steps, "estimated_steps": max(1, len(steps)),
604
+ "requires_approval": False, "rollback_strategy": "none"},
605
+ ensure_ascii=False,
606
+ )
607
+
608
+
609
+ def action_json(**payload: Any) -> str:
610
+ return json.dumps(payload, ensure_ascii=False)
611
+
612
+
613
+ def verdict_json(verdict: str, next_state: str, reason: str = "checked") -> str:
614
+ return json.dumps(
615
+ {"action": "verdict", "verdict": verdict, "next_state": next_state,
616
+ "reason": reason, "corrections": []},
617
+ ensure_ascii=False,
618
+ )
619
+
620
+
621
+ WRITE_STEP = [{"action": "write_file", "args": {"path": "note.md"}, "description": "the note"}]
622
+
623
+ #: Seven trajectories: between them they reach every terminal state by every
624
+ #: route the loop has, under all three permission modes.
625
+ SCENARIOS: Dict[str, Dict[str, Any]] = {
626
+ "clean_done_trusted": {
627
+ "mode": "trusted", "message": "make a note", "seed": {},
628
+ "governor_verdict": None,
629
+ "script": [
630
+ plan_json("make a note", WRITE_STEP),
631
+ action_json(thoughts="writing", action="write_file",
632
+ args={"path": "note.md", "content": "# Note\n\nhello\n"}),
633
+ action_json(action="final", message="파일을 만들었습니다."),
634
+ verdict_json("PASS", "DONE", "the file exists"),
635
+ ],
636
+ },
637
+ "strict_proposal_pause": {
638
+ "mode": "strict", "message": "update the note", "seed": {"note.md": "original\n"},
639
+ "governor_verdict": {"decision": "proposed", "proposal": {"id": "prop-1"},
640
+ "classification": {"change_class": "mutation"}},
641
+ "script": [
642
+ plan_json("update the note", WRITE_STEP),
643
+ action_json(thoughts="rewriting", action="write_file",
644
+ args={"path": "note.md", "content": "rewritten\n"}),
645
+ action_json(action="final", message="제안으로 저장했습니다."),
646
+ verdict_json("PASS", "DONE", "staged for review"),
647
+ ],
648
+ },
649
+ "parse_budget_exhaustion": {
650
+ "mode": "trusted", "message": "make a note", "seed": {},
651
+ "governor_verdict": None,
652
+ "script": [
653
+ plan_json("make a note", WRITE_STEP),
654
+ "I will now write the file for you.",
655
+ "Writing it now, one moment.",
656
+ "All done, I think.",
657
+ verdict_json("PASS", "DONE", "looks fine"),
658
+ ],
659
+ },
660
+ "repeated_create_guard": {
661
+ "mode": "trusted", "message": "make a note", "seed": {},
662
+ "governor_verdict": None,
663
+ "script": [
664
+ plan_json("make a note", WRITE_STEP),
665
+ action_json(action="write_file", args={"path": "note.md", "content": "body\n"}),
666
+ action_json(action="write_file", args={"path": "note.md", "content": "body\n"}),
667
+ verdict_json("PASS", "DONE", "written once"),
668
+ ],
669
+ },
670
+ "verify_retry_then_failed": {
671
+ "mode": "trusted", "message": "make a note", "seed": {},
672
+ "governor_verdict": None,
673
+ "script": [
674
+ plan_json("make a note", WRITE_STEP),
675
+ action_json(action="write_file", args={"path": "note.md", "content": "v1\n"}),
676
+ action_json(action="final", message="done"),
677
+ verdict_json("FAIL", "EXECUTING", "not good enough"),
678
+ action_json(action="final", message="done"),
679
+ verdict_json("FAIL", "EXECUTING", "still not"),
680
+ action_json(action="final", message="done"),
681
+ verdict_json("FAIL", "EXECUTING", "no"),
682
+ action_json(action="final", message="done"),
683
+ verdict_json("FAIL", "EXECUTING", "give up"),
684
+ ],
685
+ },
686
+ "verify_pass_no_evidence": {
687
+ "mode": "trusted", "message": "tell me about the notes", "seed": {},
688
+ "governor_verdict": None,
689
+ "script": [
690
+ plan_json("answer the question", []),
691
+ action_json(action="final", message="여기 답변입니다."),
692
+ verdict_json("PASS", "DONE", "answered"),
693
+ ],
694
+ },
695
+ "rollback_path": {
696
+ "mode": "trusted", "message": "update the note", "seed": {"note.md": "original\n"},
697
+ "governor_verdict": None,
698
+ "script": [
699
+ plan_json("update the note", WRITE_STEP),
700
+ action_json(action="write_file", args={"path": "note.md", "content": "broken\n"}),
701
+ action_json(action="final", message="done"),
702
+ verdict_json("FAIL", "ROLLBACK", "the change is wrong"),
703
+ ],
704
+ },
705
+ "blocked_fail_closed_strict": {
706
+ "mode": "strict", "message": "delete the note", "seed": {"note.md": "original\n"},
707
+ "governor_verdict": None,
708
+ "script": [
709
+ # The plan itself is auto-approvable (a read); the *executor* then
710
+ # reaches for a destructive tool, which is where the gate fires.
711
+ plan_json("delete the note", [{"action": "read_file", "args": {"path": "note.md"},
712
+ "description": "look at it first"}]),
713
+ action_json(action="delete_file", args={"path": "note.md"}),
714
+ action_json(action="final", message="삭제하지 못했습니다."),
715
+ verdict_json("PASS", "DONE", "nothing was deleted"),
716
+ ],
717
+ },
718
+ "blocked_breaker_bypass": {
719
+ "mode": "bypass", "message": "fix the hosts file", "seed": {},
720
+ "governor_verdict": None,
721
+ "script": [
722
+ plan_json("fix the hosts file", [{"action": "read_file", "args": {"path": "note.md"},
723
+ "description": "look first"}]),
724
+ # The registry rewrites a write aimed at a blocked system prefix
725
+ # into a destructive policy, and the breaker refuses it — in
726
+ # `bypass`, which is the whole point of a mode-invariant gate.
727
+ action_json(action="write_file", args={"path": "/etc/hosts", "content": "x\n"}),
728
+ action_json(action="final", message="시스템 파일은 건드리지 않았습니다."),
729
+ verdict_json("FAIL", "FAILED", "nothing was changed"),
730
+ ],
731
+ },
732
+ "approval_pause_strict": {
733
+ "mode": "strict", "message": "run the tests", "seed": {},
734
+ "governor_verdict": None,
735
+ "pause_expected": True,
736
+ "script": [
737
+ plan_json("run the tests", [{"action": "run_command",
738
+ "args": {"command": "ls"}, "description": "list"}]),
739
+ ],
740
+ },
741
+ }
742
+
743
+
744
+ async def trajectory(key: str, scenario: Dict[str, Any], base: Path) -> Dict[str, Any]:
745
+ """Drive the real runtime through one scenario and record what happened."""
746
+ root = base / key / "agent_workspace"
747
+ with use_workspace(root) as resolved:
748
+ for name, body in scenario["seed"].items():
749
+ target = resolved / name
750
+ target.parent.mkdir(parents=True, exist_ok=True)
751
+ target.write_text(body, encoding="utf-8")
752
+ # A scripted write whose content the artifact pipeline would rewrite
753
+ # would make this trajectory untestable against the native loop, where
754
+ # sanitation is the worker's job. Prove it does not, at build time.
755
+ for output in scenario["script"]:
756
+ content = _scripted_write_content(output)
757
+ if content is not None:
758
+ _, meta = sanitize_write_content("note.md", content, user_request=scenario["message"])
759
+ if meta.get("sanitized"):
760
+ raise SystemExit(
761
+ f"scenario {key}: scripted content is rewritten by "
762
+ "sanitize_write_content; pick content the pipeline leaves alone"
763
+ )
764
+
765
+ tool_calls: List[Dict[str, Any]] = []
766
+ audit: List[Dict[str, Any]] = []
767
+ llm = ScriptedLLM(scenario["script"])
768
+ governor = (
769
+ ScriptedGovernor(scenario["governor_verdict"])
770
+ if scenario.get("governor_verdict") is not None or scenario["mode"] == "strict"
771
+ else ScriptedGovernor(None)
772
+ )
773
+ deps = build_deps(resolved, llm, governor=governor, tool_calls=tool_calls, audit=audit)
774
+ runtime = SingleAgentRuntime(deps)
775
+
776
+ request = AgentRequest(message=scenario["message"], user_email="owner@example.com")
777
+ ctx = AgentRunContext()
778
+ ctx.trace = LoopTrace()
779
+ ctx.permission_mode = scenario["mode"]
780
+ ctx.state = AgentState.PLANNING
781
+ ctx.state_history.append(ctx.state.value)
782
+ await runtime.plan(ctx, request, "Korean", "owner@example.com", model_id=None)
783
+ requirements = runtime.approval_requirements(ctx)
784
+ paused = bool(requirements["requires_approval"])
785
+ if paused:
786
+ ctx.state_history.append(AgentState.WAITING_APPROVAL.value)
787
+ else:
788
+ runtime.approve(ctx, "owner@example.com", approved_by_human=False)
789
+ await runtime.run_to_completion(
790
+ ctx, request, "Korean", "owner@example.com",
791
+ max(1, min(request.max_steps, 50)), 3,
792
+ )
793
+ if paused != bool(scenario.get("pause_expected")):
794
+ raise SystemExit(f"scenario {key}: pause={paused}, expected the opposite")
795
+
796
+ return {
797
+ "key": key,
798
+ "mode": scenario["mode"],
799
+ "message": scenario["message"],
800
+ "seed": scenario["seed"],
801
+ "scripted_llm": scenario["script"],
802
+ "governor_verdict": scenario["governor_verdict"],
803
+ "req": {"message": scenario["message"], "user_email": "owner@example.com",
804
+ "max_steps": request.max_steps, "temperature": request.temperature},
805
+ "paused": paused,
806
+ "approval_requirements": normalize(requirements, resolved),
807
+ "final_state": ctx.state.value,
808
+ "final_message": normalize(ctx.final_message, resolved),
809
+ "state_history": ctx.state_history,
810
+ "transcript": normalize(ctx.transcript, resolved),
811
+ "rollback_log": normalize(ctx.rollback_log, resolved),
812
+ "loop": ctx.trace.summary(),
813
+ "tool_calls": normalize(tool_calls, resolved),
814
+ "audit": normalize(audit, resolved),
815
+ "llm_calls": len(llm.calls),
816
+ "unused_script": len(llm.outputs),
817
+ }
818
+
819
+
820
+ def _scripted_write_content(output: str) -> Optional[str]:
821
+ """The `content` of a scripted `write_file` action, when it is one."""
822
+ try:
823
+ payload = json.loads(output)
824
+ except (json.JSONDecodeError, TypeError):
825
+ return None
826
+ if not isinstance(payload, dict) or payload.get("action") != "write_file":
827
+ return None
828
+ content = (payload.get("args") or {}).get("content")
829
+ return content if isinstance(content, str) else None
830
+
831
+
832
+ def policy_payload() -> Dict[str, Any]:
833
+ """The real registry, as the data the native loop takes as input."""
834
+ return {
835
+ "tools": {name: dict(policy) for name, policy in sorted(TOOL_GOVERNANCE.items())},
836
+ "default": dict(TOOL_GOVERNANCE_DEFAULT),
837
+ "blocked_write_prefixes": list(LOCAL_WRITE_BLOCKED_PREFIXES),
838
+ }
839
+
840
+
841
+ def manifest_payload() -> Dict[str, Any]:
842
+ return {
843
+ "schema": SCHEMA,
844
+ "scenarios": sorted(SCENARIOS),
845
+ "raw_actions": sorted(RAW_ACTIONS),
846
+ "plan_cases": sorted(PLAN_CASES),
847
+ "normalization": [
848
+ "the absolute workspace root becomes <AGENT_ROOT>",
849
+ "keys named `at` are dropped (trace timestamps)",
850
+ "keys named `stderr` are dropped (git text is version-specific)",
851
+ f"a string starting with {DECODER_DETAIL_PREFIX!r} keeps only that prefix",
852
+ ],
853
+ "constants": {
854
+ "file_create_actions": sorted(FILE_CREATE_ACTIONS),
855
+ "scoped_knowledge_tools": sorted(SCOPED_KNOWLEDGE_TOOLS),
856
+ "governed_tools": sorted(ScriptedGovernor.governed_tools),
857
+ "phase_budgets": PhaseBudgets().__dict__,
858
+ "transcript_budget": TranscriptBudget().__dict__,
859
+ "max_state_history": 200,
860
+ "max_retry": 3,
861
+ },
862
+ }
863
+
864
+
865
+ async def build_async(base: Path) -> Dict[str, Any]:
866
+ """Everything the goldens hold, as `{filename: payload}`."""
867
+ verify_root = base / "verify" / "agent_workspace"
868
+ with use_workspace(verify_root) as resolved:
869
+ verification = await verification_rows(resolved)
870
+ trajectories = [await trajectory(key, SCENARIOS[key], base) for key in sorted(SCENARIOS)]
871
+ return {
872
+ "manifest.json": manifest_payload(),
873
+ "policies.json": policy_payload(),
874
+ "helpers.json": helper_rows(),
875
+ "verification.json": {"schema": SCHEMA, "cases": verification},
876
+ "run_store.json": {"schema": SCHEMA, "cases": run_store_rows()},
877
+ "trajectories.json": {"schema": SCHEMA, "cases": trajectories},
878
+ }
879
+
880
+
881
+ def build(base: Optional[Path] = None) -> Dict[str, Any]:
882
+ """Synchronous entry point, for the generator and the contract test alike."""
883
+ if base is None:
884
+ base = Path(tempfile.mkdtemp(prefix="agent-loop-fixtures-"))
885
+ return asyncio.run(build_async(Path(base)))
886
+
887
+
888
+ def write(payloads: Dict[str, Any]) -> List[str]:
889
+ GOLDEN_DIR.mkdir(parents=True, exist_ok=True)
890
+ for name, payload in payloads.items():
891
+ (GOLDEN_DIR / name).write_text(
892
+ json.dumps(payload, ensure_ascii=False, indent=2, sort_keys=True) + "\n",
893
+ encoding="utf-8",
894
+ )
895
+ return sorted(payloads)
896
+
897
+
898
+ def main() -> int:
899
+ written = write(build())
900
+ for name in written:
901
+ path = GOLDEN_DIR / name
902
+ print(f"wrote {path.relative_to(REPO_ROOT)} ({path.stat().st_size:,} bytes)")
903
+ return 0
904
+
905
+
906
+ if __name__ == "__main__":
907
+ raise SystemExit(main())