ltcai 11.5.0 → 11.5.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (182) hide show
  1. package/README.md +86 -47
  2. package/docs/CHANGELOG.md +76 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/HYBRID_CLOUD_KG_STREAMING.md +8 -4
  6. package/docs/MULTI_AGENT_RUNTIME.md +1 -1
  7. package/docs/ONBOARDING.md +1 -1
  8. package/docs/OPERATIONS.md +1 -1
  9. package/docs/TRUST_MODEL.md +1 -1
  10. package/docs/WHY_LATTICE.md +1 -1
  11. package/docs/WORKFLOW_DESIGNER.md +1 -1
  12. package/docs/architecture.md +4 -4
  13. package/docs/kg-schema.md +1 -1
  14. package/docs/v11.5.1_RUST_FULL_LOOP_PLAN.md +85 -0
  15. package/docs/v11.5.2_TIGHT_SHIP_PLAN.md +144 -0
  16. package/lattice_brain/__init__.py +1 -1
  17. package/lattice_brain/graph/proactive.py +1 -1
  18. package/lattice_brain/graph/projection/v2_schema.py +0 -28
  19. package/lattice_brain/graph/vector_index/base.py +0 -3
  20. package/lattice_brain/graph/vector_index/brute_force.py +0 -4
  21. package/lattice_brain/graph/vector_index/hnsw.py +0 -3
  22. package/lattice_brain/graph/vector_index/quantized.py +0 -3
  23. package/lattice_brain/ingestion_jobs.py +0 -3
  24. package/lattice_brain/memory.py +0 -27
  25. package/lattice_brain/portability/sharing.py +0 -4
  26. package/lattice_brain/quality.py +0 -49
  27. package/lattice_brain/runtime/agent_runtime.py +1 -1
  28. package/lattice_brain/runtime/hooks.py +0 -13
  29. package/lattice_brain/runtime/multi_agent.py +1 -1
  30. package/lattice_brain/sealed_box.py +0 -4
  31. package/latticeai/__init__.py +1 -1
  32. package/latticeai/api/admin.py +12 -12
  33. package/latticeai/api/agent_worker_seam.py +389 -0
  34. package/latticeai/api/auth.py +25 -2
  35. package/latticeai/api/chat.py +3 -3
  36. package/latticeai/api/chat_agent_http.py +3 -12
  37. package/latticeai/api/chat_helpers.py +10 -13
  38. package/latticeai/api/computer_use.py +31 -42
  39. package/latticeai/api/health.py +21 -1
  40. package/latticeai/api/local_files.py +14 -0
  41. package/latticeai/api/permissions.py +38 -12
  42. package/latticeai/api/search.py +24 -0
  43. package/latticeai/api/static_routes.py +4 -1
  44. package/latticeai/cli/runtime.py +7 -3
  45. package/latticeai/core/config.py +47 -3
  46. package/latticeai/core/csrf.py +28 -2
  47. package/latticeai/core/embedding_providers/__init__.py +3 -5
  48. package/latticeai/core/embedding_providers/base.py +1 -1
  49. package/latticeai/core/embedding_providers/text.py +1 -1
  50. package/latticeai/core/enterprise.py +1 -4
  51. package/latticeai/core/http_origin.py +146 -0
  52. package/latticeai/core/invitations.py +3 -2
  53. package/latticeai/core/io_utils.py +2 -11
  54. package/latticeai/core/legacy_compatibility.py +1 -1
  55. package/latticeai/core/marketplace.py +1 -1
  56. package/latticeai/core/messages.py +40 -0
  57. package/latticeai/core/model_compat.py +2 -8
  58. package/latticeai/core/module_probe.py +37 -0
  59. package/latticeai/core/run_explain.py +5 -2
  60. package/latticeai/core/run_store.py +2 -2
  61. package/latticeai/core/security.py +13 -0
  62. package/latticeai/core/sessions.py +3 -2
  63. package/latticeai/core/sse.py +25 -0
  64. package/latticeai/core/users.py +0 -9
  65. package/latticeai/core/workspace_graph_trace.py +2 -8
  66. package/latticeai/core/workspace_os.py +0 -10
  67. package/latticeai/core/workspace_os_constants.py +1 -1
  68. package/latticeai/core/workspace_os_utils.py +34 -0
  69. package/latticeai/core/workspace_review_items.py +2 -8
  70. package/latticeai/core/workspace_runs.py +2 -8
  71. package/latticeai/core/workspace_skills.py +2 -7
  72. package/latticeai/models/router/documents.py +0 -35
  73. package/latticeai/models/router/registry.py +0 -4
  74. package/latticeai/runtime/access_runtime.py +16 -4
  75. package/latticeai/runtime/build_phases/features.py +25 -2
  76. package/latticeai/runtime/build_phases/web.py +2 -0
  77. package/latticeai/runtime/feature_toggle_wiring.py +12 -18
  78. package/latticeai/runtime/network_boundary_wiring.py +8 -19
  79. package/latticeai/runtime/permission_mode_wiring.py +6 -13
  80. package/latticeai/runtime/router_registration.py +4 -0
  81. package/latticeai/runtime/runtime_context.py +1 -13
  82. package/latticeai/runtime/service_singletons.py +55 -0
  83. package/latticeai/services/architecture_readiness.py +1 -1
  84. package/latticeai/services/brain_intelligence/proposals.py +0 -5
  85. package/latticeai/services/change_proposals.py +2 -36
  86. package/latticeai/services/chronicle.py +4 -6
  87. package/latticeai/services/command_center.py +4 -4
  88. package/latticeai/services/evidence_actions.py +5 -2
  89. package/latticeai/services/hybrid_chat.py +2 -2
  90. package/latticeai/services/hybrid_policy.py +8 -57
  91. package/latticeai/services/mode_store.py +132 -0
  92. package/latticeai/services/model_capability_registry.py +0 -15
  93. package/latticeai/services/model_catalog.py +0 -7
  94. package/latticeai/services/model_loading.py +3 -2
  95. package/latticeai/services/model_runtime/__init__.py +0 -57
  96. package/latticeai/services/model_runtime/engines.py +27 -128
  97. package/latticeai/services/model_runtime/loading.py +2 -2
  98. package/latticeai/services/network_boundary_service.py +7 -44
  99. package/latticeai/services/permission_mode_service.py +8 -56
  100. package/latticeai/services/product_readiness.py +1 -1
  101. package/latticeai/services/setup_detection.py +67 -1
  102. package/latticeai/services/tool_dispatch.py +0 -4
  103. package/latticeai/services/upload_service.py +7 -20
  104. package/latticeai/services/workspace_service.py +0 -6
  105. package/latticeai/setup/auto_setup.py +15 -26
  106. package/latticeai/setup/wizard/catalog.py +5 -2
  107. package/latticeai/setup/wizard/detect.py +6 -27
  108. package/latticeai/setup/wizard/paths.py +2 -5
  109. package/latticeai/setup/wizard/plans.py +2 -2
  110. package/package.json +2 -4
  111. package/scripts/bump_version.py +14 -1
  112. package/scripts/check_current_release_docs.mjs +1 -1
  113. package/scripts/check_legacy_debt.mjs +1 -1
  114. package/scripts/check_server_i18n.mjs +1 -0
  115. package/scripts/generate_agent_loop_fixtures.py +994 -0
  116. package/scripts/generate_rust_parity_fixtures.py +72 -161
  117. package/scripts/parity_fixture_corpus_context.py +162 -0
  118. package/scripts/parity_fixture_corpus_docgen.py +341 -0
  119. package/scripts/release_screen_claims.json +20 -0
  120. package/src-tauri/Cargo.lock +13 -7
  121. package/src-tauri/Cargo.toml +1 -1
  122. package/src-tauri/src/backend.rs +76 -2
  123. package/src-tauri/src/main.rs +13 -3
  124. package/src-tauri/tauri.conf.json +2 -2
  125. package/static/app/asset-manifest.json +41 -41
  126. package/static/app/assets/{Act-CWnxSCgN.js → Act-DcQizkl1.js} +1 -1
  127. package/static/app/assets/{AdminConsole-BEQYU6kF.js → AdminConsole-cf4npybT.js} +1 -1
  128. package/static/app/assets/{Brain-DWu1BhFg.js → Brain-3VCSHFcn.js} +1 -1
  129. package/static/app/assets/{BrainHome-95Hilr9R.js → BrainHome-Qm8eaztx.js} +1 -1
  130. package/static/app/assets/{BrainSignals-QdeqCpAF.js → BrainSignals-DS9BtKOW.js} +1 -1
  131. package/static/app/assets/{Capture-BHpCxnzb.js → Capture-DiQ219jW.js} +1 -1
  132. package/static/app/assets/{Chronicle-B4xYKoed.js → Chronicle-BGvuAchH.js} +1 -1
  133. package/static/app/assets/{CommandPalette-BVXnttSz.js → CommandPalette-Bqhm0Urn.js} +1 -1
  134. package/static/app/assets/{Library-DgYcHome.js → Library-BV6NnF0a.js} +1 -1
  135. package/static/app/assets/{LivingBrain-CrJLDbf7.js → LivingBrain-GzenJchP.js} +1 -1
  136. package/static/app/assets/{ProductFlow-DFlScKoJ.js → ProductFlow-DEP6-vML.js} +1 -1
  137. package/static/app/assets/{ReviewCard-Cy5f48Pj.js → ReviewCard-CNZ7XjWG.js} +1 -1
  138. package/static/app/assets/{System-NF8IfhTa.js → System-CieofHQa.js} +1 -1
  139. package/static/app/assets/arrow-left-kfsrk0mv.js +1 -0
  140. package/static/app/assets/{bot-CucuhLhm.js → bot-B_K1Tdmw.js} +1 -1
  141. package/static/app/assets/{brain-BBnSryW_.js → brain-DWyaV1L1.js} +1 -1
  142. package/static/app/assets/{button-C2GUj2Ai.js → button-aTn4s84A.js} +1 -1
  143. package/static/app/assets/circle-check-qqLug9nU.js +1 -0
  144. package/static/app/assets/{circle-pause-CbkWzBmG.js → circle-pause-xKgeGXkT.js} +1 -1
  145. package/static/app/assets/{circle-play-7lEaqHdJ.js → circle-play-DkT6tYPX.js} +1 -1
  146. package/static/app/assets/{cpu-DAlCXlIy.js → cpu-85xYObUC.js} +1 -1
  147. package/static/app/assets/{download-RNhuuJwh.js → download-B5Fm7YXo.js} +1 -1
  148. package/static/app/assets/{folder-open-CLW4odzM.js → folder-open-kk2Xa52u.js} +1 -1
  149. package/static/app/assets/{hard-drive-NKEiDIAJ.js → hard-drive-DkA3zBW_.js} +1 -1
  150. package/static/app/assets/{index-DMurvUuR.js → index-BMPdTmlY.js} +3 -3
  151. package/static/app/assets/index-DxmOfNRi.css +2 -0
  152. package/static/app/assets/{input-D2UhPC1X.js → input-B0nRf2jO.js} +1 -1
  153. package/static/app/assets/{link-2-6amKbP_P.js → link-2-Dwb4gnTc.js} +1 -1
  154. package/static/app/assets/{permissionCopy-Cu9TZtdR.js → permissionCopy-CQDUBrOZ.js} +1 -1
  155. package/static/app/assets/{primitives-gPsccucr.js → primitives-SNp0LRJz.js} +1 -1
  156. package/static/app/assets/search-BcHqkjoy.js +1 -0
  157. package/static/app/assets/{share-2-Bau7KkPq.js → share-2-BsrxFglO.js} +1 -1
  158. package/static/app/assets/{shield-alert-BufNYypi.js → shield-alert-5BStfp2_.js} +1 -1
  159. package/static/app/assets/{textarea-BQnVWhYs.js → textarea-Cg8IUA-k.js} +1 -1
  160. package/static/app/assets/{useFocusTrap-B3_w60si.js → useFocusTrap-CYKvE46M.js} +1 -1
  161. package/static/app/assets/{useMutation-BHhCflT6.js → useMutation-CSn9t1op.js} +1 -1
  162. package/static/app/assets/{useQuery-rBWfI-5t.js → useQuery-CY2OI2uy.js} +1 -1
  163. package/static/app/assets/{utils-V_5-wxr5.js → utils-Ddol2RWD.js} +1 -1
  164. package/static/app/assets/{workspace-K1zjYUHj.js → workspace-BqDwOz_p.js} +1 -1
  165. package/static/app/index.html +4 -4
  166. package/static/sw.js +1 -1
  167. package/desktop/electron/README.md +0 -9
  168. package/desktop/electron/main.cjs +0 -58
  169. package/desktop/electron/preload.cjs +0 -5
  170. package/latticeai/core/graph_curator.py +0 -11
  171. package/latticeai/core/hooks.py +0 -11
  172. package/latticeai/core/local_embeddings.py +0 -104
  173. package/latticeai/core/multi_agent.py +0 -11
  174. package/latticeai/core/workflow_engine.py +0 -11
  175. package/latticeai/services/ingestion.py +0 -11
  176. package/latticeai/services/kg_portability.py +0 -11
  177. package/latticeai/services/multimodal_streaming.py +0 -129
  178. package/scripts/measure_brain_home_fill.mjs +0 -141
  179. package/static/app/assets/arrow-left-DwkSYrjR.js +0 -1
  180. package/static/app/assets/circle-check-CxOVPwYq.js +0 -1
  181. package/static/app/assets/index-BLPb5lmE.css +0 -2
  182. package/static/app/assets/search-Cj_TKk_2.js +0 -1
@@ -0,0 +1,994 @@
1
+ #!/usr/bin/env python3
2
+ """Build the committed Python↔Rust **agent loop** parity fixtures (v11.5.1).
3
+
4
+ ``rust/lattice-agent`` now owns the PLAN → EXECUTE → VERIFY → ROLLBACK
5
+ orchestration that ``latticeai.core.agent`` has always owned, with the Python
6
+ worker behind three seam endpoints. A port of a *state machine* is only worth
7
+ having if something keeps proving it still reaches the same states, by the same
8
+ route, with the same record — so this script is the Python half of that proof.
9
+
10
+ It runs the **real** functions, never a re-description of them:
11
+
12
+ * the deterministic helpers — ``extract_action_details``, ``normalize_plan``,
13
+ ``infer_file_target`` / ``infer_project_manifest``, ``requirement_coverage``,
14
+ ``artifact_checklist``, ``files_written``, ``compact_transcript``,
15
+ ``_truncate_strings``, ``filter_learnings``, ``PhaseBudgets`` /
16
+ ``TranscriptBudget``;
17
+ * the verification verdict mapping, by calling the real ``verify()`` over a
18
+ verdict × evidence × coverage × retry grid with a scripted critic;
19
+ * the run store's ``serialize_run_context`` / ``restore_run_context``;
20
+ * and **end-to-end trajectories**: the real :class:`SingleAgentRuntime`, driven
21
+ by a scripted LLM and the real tool registry inside a throwaway
22
+ ``AGENT_ROOT``, for seven scenarios that between them exercise every branch
23
+ the loop can take to a terminal state.
24
+
25
+ Two consumers read what it writes:
26
+
27
+ * ``tests/unit/test_agent_loop_parity_contract.py`` re-runs the Python loop over
28
+ the same scripts and asserts the committed goldens still hold — so a change to
29
+ a Python gate fails loudly instead of silently invalidating the contract the
30
+ Rust side is pinned to;
31
+ * ``rust/lattice-agent/tests/agent_loop.rs`` drives the native loop against a
32
+ fake worker replaying the recorded completions and tool results, and asserts
33
+ the same trajectories.
34
+
35
+ Determinism is the design constraint. Four normalisation rules make the record
36
+ machine-independent, and both sides apply them:
37
+
38
+ 1. the absolute workspace root becomes ``<AGENT_ROOT>``;
39
+ 2. ``at`` keys are dropped (trace timestamps);
40
+ 3. ``stderr`` keys are dropped (git's text is version- and locale-specific);
41
+ 4. a JSON decoder detail is collapsed to ``<decoder-detail>`` — CPython renamed
42
+ several of those messages in 3.14, and pinning a Python patch release is not
43
+ what this contract is for.
44
+
45
+ Usage::
46
+
47
+ .venv/bin/python scripts/generate_agent_loop_fixtures.py
48
+ """
49
+
50
+ from __future__ import annotations
51
+
52
+ import asyncio
53
+ import copy
54
+ import json
55
+ import os
56
+ import sys
57
+ import tempfile
58
+ from contextlib import contextmanager
59
+ from pathlib import Path
60
+ from typing import Any, Dict, Iterator, List, Optional
61
+
62
+ REPO_ROOT = Path(__file__).resolve().parents[1]
63
+ if str(REPO_ROOT) not in sys.path:
64
+ sys.path.insert(0, str(REPO_ROOT))
65
+
66
+ # The tool registry resolves AGENT_ROOT at import; point it somewhere harmless
67
+ # before anything imports it, then patch it per run through `use_workspace`.
68
+ os.environ.setdefault(
69
+ "LATTICEAI_AGENT_ROOT", str(Path(tempfile.gettempdir()) / "agent-loop-fixtures-import")
70
+ )
71
+
72
+ import latticeai.services.tool_dispatch as tool_dispatch # noqa: E402
73
+ import latticeai.tools as tools # noqa: E402
74
+ from latticeai.api.chat_contracts import AgentRequest # noqa: E402
75
+ from latticeai.core.agent import ( # noqa: E402
76
+ AgentRunContext,
77
+ AgentState,
78
+ SingleAgentRuntime,
79
+ )
80
+ from latticeai.core.agent.deps import AgentDeps # noqa: E402
81
+ from latticeai.core.agent_helpers import ( # noqa: E402
82
+ PhaseBudgets,
83
+ TranscriptBudget,
84
+ _truncate_strings,
85
+ artifact_checklist,
86
+ compact_transcript,
87
+ extract_action_details,
88
+ files_written,
89
+ filter_learnings,
90
+ normalize_plan,
91
+ requirement_coverage,
92
+ )
93
+ from latticeai.core.agent_profiles import ( # noqa: E402
94
+ COMPACT_MAX_PARAMS_B,
95
+ model_size_b,
96
+ profile_for_model,
97
+ )
98
+ from latticeai.core.agent_trace import LoopTrace # noqa: E402
99
+ from latticeai.core.file_generation import ( # noqa: E402
100
+ infer_file_target,
101
+ infer_project_manifest,
102
+ sanitize_write_content,
103
+ )
104
+ from latticeai.core.run_store import ( # noqa: E402
105
+ restore_run_context,
106
+ serialize_run_context,
107
+ )
108
+ from latticeai.core.tool_registry import ( # noqa: E402
109
+ FILE_CREATE_ACTIONS,
110
+ LOCAL_WRITE_BLOCKED_PREFIXES,
111
+ SCOPED_KNOWLEDGE_TOOLS,
112
+ TOOL_GOVERNANCE,
113
+ TOOL_GOVERNANCE_DEFAULT,
114
+ )
115
+ from latticeai.tools.documents import document_output_target # noqa: E402
116
+
117
+ FIXTURE_DIR = REPO_ROOT / "rust" / "fixtures" / "agent_loop"
118
+ GOLDEN_DIR = FIXTURE_DIR / "golden"
119
+ SCHEMA = "agent-loop-parity/v1"
120
+
121
+ DECODER_DETAIL_PREFIX = "Agent did not return valid JSON: "
122
+ ROOT_PLACEHOLDER = "<AGENT_ROOT>"
123
+
124
+
125
+ # ── normalisation ─────────────────────────────────────────────────────────────
126
+ def normalize(value: Any, root: Path) -> Any:
127
+ """Apply the four machine-independence rules, recursively."""
128
+ if isinstance(value, str):
129
+ text = value.replace(str(root), ROOT_PLACEHOLDER)
130
+ if text.startswith(DECODER_DETAIL_PREFIX):
131
+ return f"{DECODER_DETAIL_PREFIX}<decoder-detail>"
132
+ return text
133
+ if isinstance(value, dict):
134
+ return {
135
+ key: normalize(item, root)
136
+ for key, item in value.items()
137
+ if key not in ("at", "stderr")
138
+ }
139
+ if isinstance(value, list):
140
+ return [normalize(item, root) for item in value]
141
+ return value
142
+
143
+
144
+ @contextmanager
145
+ def use_workspace(root: Path) -> Iterator[Path]:
146
+ """Point the real tool registry at ``root`` for the duration.
147
+
148
+ Two modules hold the constant: ``latticeai.tools`` defines it and
149
+ ``latticeai.services.tool_dispatch`` imported the value, so patching one
150
+ would leave the snapshot/rollback ports reading the real workspace.
151
+ """
152
+ root = Path(root)
153
+ root.mkdir(parents=True, exist_ok=True)
154
+ resolved = root.resolve()
155
+ previous = (tools.AGENT_ROOT, tool_dispatch.AGENT_ROOT)
156
+ tools.AGENT_ROOT = resolved
157
+ tool_dispatch.AGENT_ROOT = resolved
158
+ try:
159
+ yield resolved
160
+ finally:
161
+ tools.AGENT_ROOT, tool_dispatch.AGENT_ROOT = previous
162
+
163
+
164
+ # ── the deterministic helper grids ────────────────────────────────────────────
165
+ #: Raw model outputs chosen for the rung of the tolerance chain each one reaches.
166
+ RAW_ACTIONS: Dict[str, str] = {
167
+ "clean": '{"action": "final", "message": "done"}',
168
+ "clean_with_args": '{"thoughts": "t", "action": "write_file", "args": {"path": "a.md"}}',
169
+ "fence_json": '```json\n{"action": "read_file", "args": {"path": "a.md"}}\n```',
170
+ "fence_bare": '```\n{"action": "final"}\n```',
171
+ "fence_with_prose": 'Sure!\n```json\n{"action": "final"}\n```\nHope that helps.',
172
+ "fence_multiline": '```json\n{\n "action": "final",\n "message": "ok"\n}\n```',
173
+ "think_then_json": '<think>hmm {"action": "wrong"}</think>\n{"action": "right"}',
174
+ "thinking_tag": '<thinking>plan</thinking>{"action": "final"}',
175
+ "reasoning_tag": '<reasoning>why</reasoning>\n{"action": "final"}',
176
+ "think_uppercase": '<THINK>x</THINK>{"action": "final"}',
177
+ "think_unclosed": '<think>never closed {"action": "final"}',
178
+ "think_mismatched": '<think>{"action": "a"}</reasoning>',
179
+ "slice_prefix": 'I will call: {"action": "write_file"}',
180
+ "slice_suffix": '{"action": "write_file"} — that is the call.',
181
+ "slice_both": 'Calling {"action": "final", "message": "완료"} now.',
182
+ "slice_nested": 'note {"action": "a", "args": {"b": {"c": 1}}} end',
183
+ "trailing_comma_object": '{"action": "final", "message": "x",}',
184
+ "trailing_comma_array": '{"action": "a", "args": {"items": [1, 2,]}}',
185
+ "trailing_comma_nested": '{"action": "a", "args": {"x": 1,},}',
186
+ "python_literal": "{'action': 'write_file', 'args': {'path': 'a.md'}}",
187
+ "python_literal_true": "{'action': 'a', 'ok': True, 'bad': False, 'none': None}",
188
+ "python_literal_trailing": "{'action': 'final', 'message': 'hi',}",
189
+ "python_literal_nested": "{'action': 'a', 'args': {'items': [1, 2.5, 'x']}}",
190
+ "python_literal_escapes": "{'action': 'a', 'note': 'line\\nbreak'}",
191
+ "python_literal_not_dict": "('a', 'b')",
192
+ "broken_prose": "I think we should start by reading the notes.",
193
+ "broken_empty": "",
194
+ "broken_whitespace": " \n ",
195
+ "broken_unclosed": '{"action": "final"',
196
+ "broken_missing_value": '{"action": }',
197
+ "broken_unquoted_key": "{action: 1}",
198
+ "no_action_key": '{"thoughts": "no action here"}',
199
+ "not_an_object": "[1, 2, 3]",
200
+ "bare_number": "42",
201
+ "bare_string": '"just text"',
202
+ "korean_prose_slice": '작업 계획: {"action": "final", "message": "완료"} 끝.',
203
+ "double_object": '{"action": "a"} {"action": "b"}',
204
+ "unicode_thoughts": '{"action": "final", "thoughts": "가나다라마바사"}',
205
+ }
206
+
207
+ #: Plans chosen for the seven normalisation rules and their interactions.
208
+ PLAN_CASES: Dict[str, Dict[str, Any]] = {
209
+ "complete": {
210
+ "plan": {"goal": "g", "steps": [{"action": "read_file", "args": {"path": "a.md"}}],
211
+ "estimated_steps": 1, "requires_approval": False, "rollback_strategy": "none"},
212
+ "message": "read a.md",
213
+ },
214
+ "not_an_object": {"plan": ["nope"], "message": "hi"},
215
+ "null_plan": {"plan": None, "message": "hi"},
216
+ "string_plan": {"plan": "a plan", "message": "hi"},
217
+ "blank_goal": {"plan": {"goal": " ", "steps": []}, "message": "do the thing"},
218
+ "missing_goal": {"plan": {"steps": []}, "message": "do the thing"},
219
+ "numeric_goal": {"plan": {"goal": 5, "steps": []}, "message": "hi"},
220
+ "junk_steps": {
221
+ "plan": {"goal": "g", "steps": ["x", {"no_action": 1}, {"action": ""},
222
+ {"action": "read_file"}]},
223
+ "message": "g",
224
+ },
225
+ "steps_not_a_list": {"plan": {"goal": "g", "steps": "read a file"}, "message": "g"},
226
+ "empty_steps": {"plan": {"goal": "g", "steps": []}, "message": "g"},
227
+ "manifest_empty_plan": {"plan": {"goal": "g", "steps": []},
228
+ "message": "todo 앱 html css js 만들어줘"},
229
+ "manifest_partial": {
230
+ "plan": {"goal": "g", "steps": [{"action": "write_file", "args": {"path": "index.html"}}]},
231
+ "message": "todo 앱 html css js 만들어줘",
232
+ },
233
+ "manifest_covered": {
234
+ "plan": {"goal": "g", "steps": [
235
+ {"action": "write_file", "args": {"path": "page.HTML"}},
236
+ {"action": "write_file", "args": {"path": "a.css"}},
237
+ {"action": "generate_file", "args": {"path": "b.js"}}]},
238
+ "message": "todo 앱 html css js 만들어줘",
239
+ },
240
+ "manifest_with_read": {
241
+ "plan": {"goal": "g", "steps": [{"action": "read_file", "args": {"path": "spec.md"}},
242
+ {"action": "write_file", "args": {"path": "index.html"}}]},
243
+ "message": "todo 앱 html css js 만들어줘",
244
+ },
245
+ "manifest_react": {"plan": {}, "message": "react 로 todo 앱 만들어줘"},
246
+ "manifest_python": {"plan": {}, "message": "mytool 패키지 파이썬으로 만들어줘"},
247
+ "heuristic_single_file": {"plan": {}, "message": "html 파일 만들어줘"},
248
+ "heuristic_long_message": {"plan": {}, "message": "html 파일 만들어줘 " + "가" * 200},
249
+ "estimated_string": {"plan": {"goal": "g", "estimated_steps": "4"}, "message": "g"},
250
+ "estimated_float": {"plan": {"goal": "g", "estimated_steps": 3.7}, "message": "g"},
251
+ "estimated_invalid": {"plan": {"goal": "g", "estimated_steps": "many"}, "message": "g"},
252
+ "estimated_list": {"plan": {"goal": "g", "estimated_steps": [3]}, "message": "g"},
253
+ "coerced_tail": {"plan": {"goal": "g", "requires_approval": "yes", "rollback_strategy": 7},
254
+ "message": "g"},
255
+ "rollback_kept": {"plan": {"goal": "g", "rollback_strategy": "git"}, "message": "g"},
256
+ }
257
+
258
+ #: Requests the two inference functions are asked about.
259
+ INFERENCE_MESSAGES: List[str] = [
260
+ "html 파일 만들어줘", "write me a python script", "csv 저장해줘", "html이 뭐야?",
261
+ "만들어줘", "", " ", "html과 css 만들어줘", "html and css 만들어줘",
262
+ "todo 앱 html+css+js로 만들어줘", "웹페이지 js로 만들어줘", "웹페이지 css로 만들어줘",
263
+ "react 로 todo 앱 만들어줘", "리액트로 만들어줘", "vite 앱 만들어줘",
264
+ "mytool 패키지 파이썬으로 만들어줘", "my-tool 패키지 파이썬으로 생성",
265
+ "파이썬 패키지 만들어줘", "index.html 이랑 style.css 만들어줘",
266
+ "웹사이트 css js 만들어줘", "마크다운 파일 작성해줘", "yaml 만들어줘",
267
+ ]
268
+
269
+ #: ``(tool_name, filename)`` pairs for the document-target resolver. Chosen for
270
+ #: the branches they reach, not for realism: a tool that is not a document
271
+ #: creator, each of the four that are, a name that already carries its suffix,
272
+ #: one that carries the wrong one, a path that must be reduced to its basename,
273
+ #: characters the sanitizer replaces, and the empty names that fall back to
274
+ #: ``artifact<suffix>``. The Rust port takes the basename with its own
275
+ #: ``path_name`` and has its own empty-name fallback, so those two are exactly
276
+ #: where the two implementations could disagree unnoticed.
277
+ DOCUMENT_TARGET_CASES: List[Dict[str, str]] = [
278
+ {"tool": "write_file", "filename": "notes.md"},
279
+ {"tool": "create_docx", "filename": "report.docx"},
280
+ {"tool": "create_docx", "filename": "report"},
281
+ {"tool": "create_docx", "filename": "report.pdf"},
282
+ {"tool": "create_xlsx", "filename": "budget.xlsx"},
283
+ {"tool": "create_pptx", "filename": "deck"},
284
+ {"tool": "create_pdf", "filename": "invoice"},
285
+ {"tool": "create_pdf", "filename": "sub/dir/invoice.pdf"},
286
+ {"tool": "create_pdf", "filename": "../../escape.pdf"},
287
+ {"tool": "create_docx", "filename": "회의 기록.docx"},
288
+ {"tool": "create_docx", "filename": "a/b*c?d.docx"},
289
+ {"tool": "create_docx", "filename": ""},
290
+ {"tool": "create_pptx", "filename": " "},
291
+ {"tool": "create_xlsx", "filename": "REPORT.XLSX"},
292
+ {"tool": "create_pdf", "filename": ".pdf"},
293
+ {"tool": "unknown_tool", "filename": "x.docx"},
294
+ ]
295
+
296
+ #: Model ids for the profile dial. The interesting ones are the quantization
297
+ #: suffixes (``4bit`` is not a parameter count), the multi-size ids where the
298
+ #: *smallest* wins, and the boundary at ``COMPACT_MAX_PARAMS_B`` itself.
299
+ PROFILE_MODEL_IDS: List[str] = [
300
+ "mlx-community/gemma-4-12B-it-4bit",
301
+ "qwen2.5-1.5b",
302
+ "llama-3.2-3B",
303
+ "mlx-community/Llama-3.2-3B-Instruct-4bit",
304
+ "phi-4-mini-3.8b-8bit",
305
+ "some-model-4b",
306
+ "some-model-4.0b",
307
+ "some-model-4.1b",
308
+ "gpt-4o",
309
+ "claude-sonnet",
310
+ "",
311
+ " ",
312
+ "model-8bit",
313
+ "abc123b",
314
+ "7b-and-1.5b-mixed",
315
+ "Model-70B-Instruct",
316
+ ]
317
+
318
+ #: ``LATTICEAI_AGENT_PROFILE`` values, including the two that must fall through
319
+ #: to the size heuristic rather than failing the run.
320
+ PROFILE_OVERRIDES: List[str] = ["", "standard", "compact", "COMPACT", "nonsense"]
321
+
322
+ #: Transcripts the artifact/coverage helpers are asked about.
323
+ TRANSCRIPT_CASES: Dict[str, Dict[str, Any]] = {
324
+ "empty": {"message": "todo 앱 html css js 만들어줘", "transcript": []},
325
+ "one_write": {
326
+ "message": "todo 앱 html css js 만들어줘",
327
+ "transcript": [{"state": "EXECUTING", "action": "write_file",
328
+ "args": {"path": "index.html"},
329
+ "result": {"path": "index.html", "bytes": 10}}],
330
+ },
331
+ "all_written": {
332
+ "message": "todo 앱 html css js 만들어줘",
333
+ "transcript": [
334
+ {"state": "EXECUTING", "action": "write_file", "args": {"path": "index.html"},
335
+ "result": {"path": "index.html", "bytes": 10}},
336
+ {"state": "EXECUTING", "action": "write_file", "args": {"path": "sub/STYLE.CSS"},
337
+ "result": {"path": "sub/STYLE.CSS", "bytes": 3}},
338
+ {"state": "EXECUTING", "action": "write_file", "args": {"path": "app.js"},
339
+ "result": {"path": "app.js", "bytes": 3},
340
+ "content_sanitize": {"sanitized": True, "repaired": True}},
341
+ ],
342
+ },
343
+ "blocked_and_repeated": {
344
+ "message": "make a note",
345
+ "transcript": [
346
+ {"state": "EXECUTING", "action": "write_file", "args": {"path": "a.md"},
347
+ "error": "BLOCKED: nope"},
348
+ {"state": "EXECUTING", "action": "write_file", "args": {"path": "a.md"},
349
+ "result": {"path": "a.md", "bytes": 2}},
350
+ {"state": "EXECUTING", "action": "write_file", "args": {"path": "a.md"},
351
+ "result": {"path": "a.md", "bytes": 2}},
352
+ {"state": "VERIFYING", "action": "write_file", "result": {"path": "ignored.md"}},
353
+ {"state": "EXECUTING", "action": "read_file", "result": {"path": "skip.md"}},
354
+ ],
355
+ },
356
+ "requirements_listed": {
357
+ "message": "만들어줘:\n- 다크모드\n* dark mode\n1. 검색 기능\n2) 필터\nfree prose\n- ab",
358
+ "transcript": [],
359
+ },
360
+ "requirements_capped": {
361
+ "message": "\n".join(f"- item number {index}" for index in range(15)),
362
+ "transcript": [],
363
+ },
364
+ "proposed_step": {
365
+ "message": "make a note",
366
+ "transcript": [{"state": "EXECUTING", "action": "write_file", "args": {"path": "a.md"},
367
+ "result": {"proposed": True, "proposal_id": "p1"}}],
368
+ },
369
+ }
370
+
371
+ #: Values `_truncate_strings` is asked about, with the cap each one uses.
372
+ TRUNCATE_CASES: List[Dict[str, Any]] = [
373
+ {"key": "short", "limit": 700, "value": {"a": "hello"}},
374
+ {"key": "exact", "limit": 5, "value": "abcde"},
375
+ {"key": "over", "limit": 5, "value": "abcdefgh"},
376
+ {"key": "korean", "limit": 5, "value": "가" * 10},
377
+ {"key": "nested", "limit": 3, "value": {"a": ["abcdef", {"b": "xyz!"}], "n": 5, "t": True}},
378
+ {"key": "null_and_float", "limit": 2, "value": {"a": None, "b": 1.5, "c": []}},
379
+ ]
380
+
381
+ LEARNING_CASES: List[List[Any]] = [
382
+ ["short", "파일을 만들었습니다", "Successfully created the file",
383
+ "Vite needs the entry script tag before </body> or the app never mounts",
384
+ "VITE NEEDS THE ENTRY SCRIPT TAG BEFORE </BODY> OR THE APP NEVER MOUNTS", None],
385
+ ["작업을 완료했습니다", "task was completed", "file was created",
386
+ "Successfully created the file, but the CSS never loaded because the path was wrong"],
387
+ [],
388
+ [123456789012345, " padded learning that is long enough "],
389
+ ]
390
+
391
+
392
+ #: A root that matches nothing, for grids that never carry a path.
393
+ NO_ROOT = Path("/__no_workspace_in_this_grid__")
394
+
395
+
396
+ def helper_rows() -> Dict[str, Any]:
397
+ """Every deterministic helper, over its grid."""
398
+ actions = []
399
+ for key, raw in RAW_ACTIONS.items():
400
+ try:
401
+ action, repairs = extract_action_details(raw)
402
+ actions.append({"key": key, "raw": raw, "ok": True,
403
+ "action": action, "repairs": repairs})
404
+ except ValueError as exc:
405
+ actions.append({"key": key, "raw": raw, "ok": False, "error": str(exc)})
406
+
407
+ plans = []
408
+ for key, case in PLAN_CASES.items():
409
+ plan, fixes = normalize_plan(copy.deepcopy(case["plan"]), case["message"])
410
+ plans.append({"key": key, "plan": case["plan"], "message": case["message"],
411
+ "normalized": plan, "fixes": fixes})
412
+
413
+ inference = [
414
+ {"message": message,
415
+ "file_target": infer_file_target(message),
416
+ "manifest": infer_project_manifest(message)}
417
+ for message in INFERENCE_MESSAGES
418
+ ]
419
+
420
+ transcripts = []
421
+ for key, case in TRANSCRIPT_CASES.items():
422
+ transcripts.append({
423
+ "key": key,
424
+ "message": case["message"],
425
+ "transcript": case["transcript"],
426
+ "files_written": files_written(case["transcript"], FILE_CREATE_ACTIONS),
427
+ "artifact_checklist": artifact_checklist(case["transcript"], FILE_CREATE_ACTIONS),
428
+ "requirement_coverage": requirement_coverage(
429
+ case["message"], case["transcript"], FILE_CREATE_ACTIONS
430
+ ),
431
+ "compact_window_2": compact_transcript(case["transcript"], window=2, result_chars=40),
432
+ })
433
+
434
+ truncated = [
435
+ {"key": case["key"], "limit": case["limit"], "value": case["value"],
436
+ "truncated": _truncate_strings(case["value"], case["limit"])}
437
+ for case in TRUNCATE_CASES
438
+ ]
439
+
440
+ learnings = [
441
+ {"input": case, "kept": filter_learnings(case)} for case in LEARNING_CASES
442
+ ]
443
+
444
+ documents = [
445
+ {**case, "target": document_output_target(case["tool"], case["filename"])}
446
+ for case in DOCUMENT_TARGET_CASES
447
+ ]
448
+
449
+ # ``env`` is passed explicitly: the Rust twin reads the ambient process
450
+ # environment, and a golden that inherited this machine's would be a
451
+ # machine-specific value in a committed fixture.
452
+ profiles = []
453
+ for override in PROFILE_OVERRIDES:
454
+ env = {"LATTICEAI_AGENT_PROFILE": override} if override else {}
455
+ for model_id in PROFILE_MODEL_IDS:
456
+ profiles.append({
457
+ "override": override,
458
+ "model_id": model_id,
459
+ "size_b": model_size_b(model_id),
460
+ "profile": profile_for_model(model_id, env=env).__dict__,
461
+ })
462
+ profiles.append({
463
+ "override": "",
464
+ "model_id": None,
465
+ "size_b": model_size_b(""),
466
+ "profile": profile_for_model(None, env={}).__dict__,
467
+ })
468
+
469
+ return normalize({
470
+ "schema": SCHEMA,
471
+ "extract_action_details": actions,
472
+ "normalize_plan": plans,
473
+ "inference": inference,
474
+ "transcript_helpers": transcripts,
475
+ "truncate_strings": truncated,
476
+ "filter_learnings": learnings,
477
+ "document_targets": documents,
478
+ "agent_profiles": profiles,
479
+ "budgets": {
480
+ "phase": PhaseBudgets().__dict__,
481
+ "transcript": TranscriptBudget().__dict__,
482
+ },
483
+ }, NO_ROOT)
484
+
485
+
486
+ # ── the scripted ports ────────────────────────────────────────────────────────
487
+ class ScriptedLLM:
488
+ """`deps.generate_as`, answering a queue of recorded completions."""
489
+
490
+ def __init__(self, outputs: List[str]) -> None:
491
+ self.outputs = list(outputs)
492
+ self.calls: List[Dict[str, Any]] = []
493
+
494
+ async def generate_as(self, model_id: Optional[str] = None, **kwargs: Any) -> str:
495
+ text = self.outputs.pop(0) if self.outputs else ""
496
+ self.calls.append({
497
+ "model_id": model_id,
498
+ "message": kwargs.get("message"),
499
+ "temperature": kwargs.get("temperature"),
500
+ "max_tokens": kwargs.get("max_tokens"),
501
+ })
502
+ return text
503
+
504
+ async def generate(self, **kwargs: Any) -> str:
505
+ return await self.generate_as(None, **kwargs)
506
+
507
+
508
+ class ScriptedGovernor:
509
+ """`deps.change_governor`, answering one fixed verdict."""
510
+
511
+ governed_tools = frozenset({"write_file", "edit_file"})
512
+
513
+ def __init__(self, verdict: Optional[Dict[str, Any]]) -> None:
514
+ self.verdict = verdict
515
+
516
+ def review(self, name: str, args: Dict[str, Any], **kwargs: Any) -> Optional[Dict[str, Any]]:
517
+ return copy.deepcopy(self.verdict)
518
+
519
+
520
+ def build_deps(
521
+ root: Path,
522
+ llm: ScriptedLLM,
523
+ *,
524
+ governor: Optional[ScriptedGovernor],
525
+ tool_calls: List[Dict[str, Any]],
526
+ audit: List[Dict[str, Any]],
527
+ ) -> AgentDeps:
528
+ """The real ports, wired to the throwaway workspace."""
529
+ registry = tools.DEFAULT_TOOL_REGISTRY
530
+ service = tool_dispatch.DEFAULT_TOOL_DISPATCH_SERVICE
531
+
532
+ def execute_tool(name: str, args: Dict[str, Any]) -> Dict[str, Any]:
533
+ record: Dict[str, Any] = {"tool": name, "args": copy.deepcopy(args)}
534
+ try:
535
+ result = tools.execute_tool(name, args)
536
+ except Exception as exc: # noqa: BLE001 — recorded, then re-raised
537
+ record["error"] = str(exc)
538
+ tool_calls.append(record)
539
+ raise
540
+ record["result"] = result
541
+ tool_calls.append(record)
542
+ return result
543
+
544
+ def record_audit(event: str, **details: Any) -> None:
545
+ audit.append({"event": event, **details})
546
+
547
+ return AgentDeps(
548
+ generate_as=llm.generate_as,
549
+ generate=llm.generate,
550
+ execute_tool=execute_tool,
551
+ policy_for=registry.policy_for,
552
+ risk_level=registry.risk_level,
553
+ check_role=lambda name, user: None,
554
+ tool_governance=TOOL_GOVERNANCE,
555
+ file_create_actions=FILE_CREATE_ACTIONS,
556
+ recent_chat_context=lambda **kwargs: "",
557
+ clear_history=lambda keep_last: {"ok": True, "kept": keep_last},
558
+ knowledge_save=lambda *a, **k: None,
559
+ audit=record_audit,
560
+ planner_prompt="",
561
+ executor_prompt="",
562
+ critic_prompt="",
563
+ memory_updater_prompt="",
564
+ agent_root=root,
565
+ rollback_file=service.rollback_file,
566
+ snapshot_file=service.snapshot_file,
567
+ restore_snapshot=service.restore_snapshot,
568
+ hooks=None,
569
+ change_governor=governor,
570
+ phase_budgets=PhaseBudgets(),
571
+ transcript_budget=TranscriptBudget(),
572
+ )
573
+
574
+
575
+ # ── verification mapping grid ─────────────────────────────────────────────────
576
+ #: `(verdict, next_state)` pairs the mapping table distinguishes.
577
+ VERDICT_PAIRS = [
578
+ ("PASS", "DONE"), ("PASS", "COMPLETE"), ("PASS", "EXECUTING"), ("PASS", ""),
579
+ ("FAIL", "DONE"), ("FAIL", "EXECUTING"), ("FAIL", "RETRY"), ("FAIL", "ROLLBACK"),
580
+ ("FAIL", "FAILED"), ("FAIL", "SOMETHING_ELSE"), ("", "DONE"),
581
+ ]
582
+
583
+ EVIDENCE_STEP = {"state": "EXECUTING", "action": "write_file", "args": {"path": "index.html"},
584
+ "result": {"path": "index.html", "bytes": 4}}
585
+ NO_EVIDENCE_STEP = {"state": "EXECUTING", "action": "final", "thoughts": "t"}
586
+
587
+
588
+ async def verification_rows(root: Path) -> List[Dict[str, Any]]:
589
+ """The verdict mapping, from the real `verify()`."""
590
+ rows: List[Dict[str, Any]] = []
591
+ for verdict, next_state in VERDICT_PAIRS:
592
+ for evidence in (True, False):
593
+ for message in ("make a note", "todo 앱 html css js 만들어줘"):
594
+ for retry_count in (0, 3):
595
+ body = json.dumps(
596
+ {"action": "verdict", "verdict": verdict, "next_state": next_state,
597
+ "reason": "because", "corrections": ["be specific"], "confidence": 0.5},
598
+ ensure_ascii=False,
599
+ )
600
+ llm = ScriptedLLM([body])
601
+ ctx = AgentRunContext()
602
+ ctx.trace = LoopTrace()
603
+ ctx.retry_count = retry_count
604
+ ctx.transcript = [copy.deepcopy(
605
+ EVIDENCE_STEP if evidence else NO_EVIDENCE_STEP
606
+ )]
607
+ runtime = SingleAgentRuntime(build_deps(
608
+ root, llm, governor=None, tool_calls=[], audit=[]
609
+ ))
610
+ request = AgentRequest(message=message)
611
+ await runtime.verify(ctx, request, "Korean", "owner@example.com", max_retry=3)
612
+ rows.append({
613
+ "verdict": verdict, "next_state": next_state, "evidence": evidence,
614
+ "message": message, "retry_count": retry_count,
615
+ "final_state": ctx.state.value,
616
+ "final_message": ctx.final_message,
617
+ "retry_count_after": ctx.retry_count,
618
+ "transcript": normalize(ctx.transcript, root),
619
+ })
620
+ # The unparseable critic: one strict retry, then fail closed.
621
+ for outputs, key in (
622
+ (["prose", "still prose"], "never_parses"),
623
+ (["prose", '{"action": "v", "verdict": "PASS", "next_state": "DONE", "reason": "r"}'],
624
+ "strict_retry_recovers"),
625
+ ):
626
+ llm = ScriptedLLM(list(outputs))
627
+ ctx = AgentRunContext()
628
+ ctx.trace = LoopTrace()
629
+ ctx.transcript = [copy.deepcopy(EVIDENCE_STEP)]
630
+ runtime = SingleAgentRuntime(build_deps(root, llm, governor=None, tool_calls=[], audit=[]))
631
+ await runtime.verify(ctx, AgentRequest(message="make a note"), "Korean", "owner", max_retry=3)
632
+ rows.append({
633
+ "verdict": key, "next_state": "", "evidence": True, "message": "make a note",
634
+ "retry_count": 0, "final_state": ctx.state.value,
635
+ "final_message": ctx.final_message, "retry_count_after": ctx.retry_count,
636
+ "transcript": normalize(ctx.transcript, root),
637
+ "llm_calls": len(llm.calls),
638
+ "temperatures": [call["temperature"] for call in llm.calls],
639
+ })
640
+ return rows
641
+
642
+
643
+ # ── run-store round trips ─────────────────────────────────────────────────────
644
+ def run_store_rows() -> List[Dict[str, Any]]:
645
+ """`serialize_run_context` / `restore_run_context`, field for field."""
646
+ rows: List[Dict[str, Any]] = []
647
+
648
+ full = AgentRunContext()
649
+ full.state = AgentState.WAITING_APPROVAL
650
+ full.plan = {"goal": "g", "steps": [{"action": "write_file"}]}
651
+ full.transcript = [{"state": "PLANNING", "goal": "g"}]
652
+ full.retry_count = 2
653
+ full.state_history = ["PLANNING", "WAITING_APPROVAL"]
654
+ full.corrections = ["reply with JSON"]
655
+ full.final_message = "paused"
656
+ full.rollback_log = [{"path": "a.md", "existed": False}]
657
+ full.executing_model = "m-exec"
658
+ full.reviewing_model = "m-review"
659
+ full.approved_by_human = True
660
+ full.permission_mode = "trusted"
661
+ full.trace = LoopTrace(clock=lambda: "PINNED")
662
+ full.trace.llm_call("plan", model="m-exec")
663
+ full.trace.repair("plan", repairs=["fence"])
664
+ rows.append({"key": "full", "serialized": serialize_run_context(full)})
665
+
666
+ empty = AgentRunContext()
667
+ rows.append({"key": "empty", "serialized": serialize_run_context(empty)})
668
+
669
+ for key, payload in (
670
+ ("unknown_state", {"state": "SOMETHING_NEW"}),
671
+ ("no_state", {}),
672
+ ("null_state", {"state": None}),
673
+ ("blank_mode", {"state": "EXECUTING", "permission_mode": ""}),
674
+ ("kept_mode", {"state": "EXECUTING", "permission_mode": "bypass"}),
675
+ ("truthy_approval", {"approved_by_human": 1}),
676
+ ):
677
+ rows.append({"key": f"restore_{key}", "payload": payload,
678
+ "restored": serialize_run_context(restore_run_context(payload))})
679
+
680
+ for row in rows:
681
+ if "serialized" in row:
682
+ row["round_trip"] = serialize_run_context(restore_run_context(row["serialized"]))
683
+ return normalize(rows, NO_ROOT)
684
+
685
+
686
+ # ── end-to-end trajectories ───────────────────────────────────────────────────
687
+ def plan_json(goal: str, steps: List[Dict[str, Any]]) -> str:
688
+ return json.dumps(
689
+ {"action": "plan", "goal": goal, "steps": steps, "estimated_steps": max(1, len(steps)),
690
+ "requires_approval": False, "rollback_strategy": "none"},
691
+ ensure_ascii=False,
692
+ )
693
+
694
+
695
+ def action_json(**payload: Any) -> str:
696
+ return json.dumps(payload, ensure_ascii=False)
697
+
698
+
699
+ def verdict_json(verdict: str, next_state: str, reason: str = "checked") -> str:
700
+ return json.dumps(
701
+ {"action": "verdict", "verdict": verdict, "next_state": next_state,
702
+ "reason": reason, "corrections": []},
703
+ ensure_ascii=False,
704
+ )
705
+
706
+
707
+ WRITE_STEP = [{"action": "write_file", "args": {"path": "note.md"}, "description": "the note"}]
708
+
709
+ #: Seven trajectories: between them they reach every terminal state by every
710
+ #: route the loop has, under all three permission modes.
711
+ SCENARIOS: Dict[str, Dict[str, Any]] = {
712
+ "clean_done_trusted": {
713
+ "mode": "trusted", "message": "make a note", "seed": {},
714
+ "governor_verdict": None,
715
+ "script": [
716
+ plan_json("make a note", WRITE_STEP),
717
+ action_json(thoughts="writing", action="write_file",
718
+ args={"path": "note.md", "content": "# Note\n\nhello\n"}),
719
+ action_json(action="final", message="파일을 만들었습니다."),
720
+ verdict_json("PASS", "DONE", "the file exists"),
721
+ ],
722
+ },
723
+ "strict_proposal_pause": {
724
+ "mode": "strict", "message": "update the note", "seed": {"note.md": "original\n"},
725
+ "governor_verdict": {"decision": "proposed", "proposal": {"id": "prop-1"},
726
+ "classification": {"change_class": "mutation"}},
727
+ "script": [
728
+ plan_json("update the note", WRITE_STEP),
729
+ action_json(thoughts="rewriting", action="write_file",
730
+ args={"path": "note.md", "content": "rewritten\n"}),
731
+ action_json(action="final", message="제안으로 저장했습니다."),
732
+ verdict_json("PASS", "DONE", "staged for review"),
733
+ ],
734
+ },
735
+ "parse_budget_exhaustion": {
736
+ "mode": "trusted", "message": "make a note", "seed": {},
737
+ "governor_verdict": None,
738
+ "script": [
739
+ plan_json("make a note", WRITE_STEP),
740
+ "I will now write the file for you.",
741
+ "Writing it now, one moment.",
742
+ "All done, I think.",
743
+ verdict_json("PASS", "DONE", "looks fine"),
744
+ ],
745
+ },
746
+ "repeated_create_guard": {
747
+ "mode": "trusted", "message": "make a note", "seed": {},
748
+ "governor_verdict": None,
749
+ "script": [
750
+ plan_json("make a note", WRITE_STEP),
751
+ action_json(action="write_file", args={"path": "note.md", "content": "body\n"}),
752
+ action_json(action="write_file", args={"path": "note.md", "content": "body\n"}),
753
+ verdict_json("PASS", "DONE", "written once"),
754
+ ],
755
+ },
756
+ "verify_retry_then_failed": {
757
+ "mode": "trusted", "message": "make a note", "seed": {},
758
+ "governor_verdict": None,
759
+ "script": [
760
+ plan_json("make a note", WRITE_STEP),
761
+ action_json(action="write_file", args={"path": "note.md", "content": "v1\n"}),
762
+ action_json(action="final", message="done"),
763
+ verdict_json("FAIL", "EXECUTING", "not good enough"),
764
+ action_json(action="final", message="done"),
765
+ verdict_json("FAIL", "EXECUTING", "still not"),
766
+ action_json(action="final", message="done"),
767
+ verdict_json("FAIL", "EXECUTING", "no"),
768
+ action_json(action="final", message="done"),
769
+ verdict_json("FAIL", "EXECUTING", "give up"),
770
+ ],
771
+ },
772
+ "verify_pass_no_evidence": {
773
+ "mode": "trusted", "message": "tell me about the notes", "seed": {},
774
+ "governor_verdict": None,
775
+ "script": [
776
+ plan_json("answer the question", []),
777
+ action_json(action="final", message="여기 답변입니다."),
778
+ verdict_json("PASS", "DONE", "answered"),
779
+ ],
780
+ },
781
+ "rollback_path": {
782
+ "mode": "trusted", "message": "update the note", "seed": {"note.md": "original\n"},
783
+ "governor_verdict": None,
784
+ "script": [
785
+ plan_json("update the note", WRITE_STEP),
786
+ action_json(action="write_file", args={"path": "note.md", "content": "broken\n"}),
787
+ action_json(action="final", message="done"),
788
+ verdict_json("FAIL", "ROLLBACK", "the change is wrong"),
789
+ ],
790
+ },
791
+ "blocked_fail_closed_strict": {
792
+ "mode": "strict", "message": "delete the note", "seed": {"note.md": "original\n"},
793
+ "governor_verdict": None,
794
+ "script": [
795
+ # The plan itself is auto-approvable (a read); the *executor* then
796
+ # reaches for a destructive tool, which is where the gate fires.
797
+ plan_json("delete the note", [{"action": "read_file", "args": {"path": "note.md"},
798
+ "description": "look at it first"}]),
799
+ action_json(action="delete_file", args={"path": "note.md"}),
800
+ action_json(action="final", message="삭제하지 못했습니다."),
801
+ verdict_json("PASS", "DONE", "nothing was deleted"),
802
+ ],
803
+ },
804
+ "blocked_breaker_bypass": {
805
+ "mode": "bypass", "message": "fix the hosts file", "seed": {},
806
+ "governor_verdict": None,
807
+ "script": [
808
+ plan_json("fix the hosts file", [{"action": "read_file", "args": {"path": "note.md"},
809
+ "description": "look first"}]),
810
+ # The registry rewrites a write aimed at a blocked system prefix
811
+ # into a destructive policy, and the breaker refuses it — in
812
+ # `bypass`, which is the whole point of a mode-invariant gate.
813
+ action_json(action="write_file", args={"path": "/etc/hosts", "content": "x\n"}),
814
+ action_json(action="final", message="시스템 파일은 건드리지 않았습니다."),
815
+ verdict_json("FAIL", "FAILED", "nothing was changed"),
816
+ ],
817
+ },
818
+ "approval_pause_strict": {
819
+ "mode": "strict", "message": "run the tests", "seed": {},
820
+ "governor_verdict": None,
821
+ "pause_expected": True,
822
+ "script": [
823
+ plan_json("run the tests", [{"action": "run_command",
824
+ "args": {"command": "ls"}, "description": "list"}]),
825
+ ],
826
+ },
827
+ }
828
+
829
+
830
+ async def trajectory(key: str, scenario: Dict[str, Any], base: Path) -> Dict[str, Any]:
831
+ """Drive the real runtime through one scenario and record what happened."""
832
+ root = base / key / "agent_workspace"
833
+ with use_workspace(root) as resolved:
834
+ for name, body in scenario["seed"].items():
835
+ target = resolved / name
836
+ target.parent.mkdir(parents=True, exist_ok=True)
837
+ target.write_text(body, encoding="utf-8")
838
+ # A scripted write whose content the artifact pipeline would rewrite
839
+ # would make this trajectory untestable against the native loop, where
840
+ # sanitation is the worker's job. Prove it does not, at build time.
841
+ for output in scenario["script"]:
842
+ content = _scripted_write_content(output)
843
+ if content is not None:
844
+ _, meta = sanitize_write_content("note.md", content, user_request=scenario["message"])
845
+ if meta.get("sanitized"):
846
+ raise SystemExit(
847
+ f"scenario {key}: scripted content is rewritten by "
848
+ "sanitize_write_content; pick content the pipeline leaves alone"
849
+ )
850
+
851
+ tool_calls: List[Dict[str, Any]] = []
852
+ audit: List[Dict[str, Any]] = []
853
+ llm = ScriptedLLM(scenario["script"])
854
+ governor = (
855
+ ScriptedGovernor(scenario["governor_verdict"])
856
+ if scenario.get("governor_verdict") is not None or scenario["mode"] == "strict"
857
+ else ScriptedGovernor(None)
858
+ )
859
+ deps = build_deps(resolved, llm, governor=governor, tool_calls=tool_calls, audit=audit)
860
+ runtime = SingleAgentRuntime(deps)
861
+
862
+ request = AgentRequest(message=scenario["message"], user_email="owner@example.com")
863
+ ctx = AgentRunContext()
864
+ ctx.trace = LoopTrace()
865
+ ctx.permission_mode = scenario["mode"]
866
+ ctx.state = AgentState.PLANNING
867
+ ctx.state_history.append(ctx.state.value)
868
+ await runtime.plan(ctx, request, "Korean", "owner@example.com", model_id=None)
869
+ requirements = runtime.approval_requirements(ctx)
870
+ paused = bool(requirements["requires_approval"])
871
+ if paused:
872
+ ctx.state_history.append(AgentState.WAITING_APPROVAL.value)
873
+ else:
874
+ runtime.approve(ctx, "owner@example.com", approved_by_human=False)
875
+ await runtime.run_to_completion(
876
+ ctx, request, "Korean", "owner@example.com",
877
+ max(1, min(request.max_steps, 50)), 3,
878
+ )
879
+ if paused != bool(scenario.get("pause_expected")):
880
+ raise SystemExit(f"scenario {key}: pause={paused}, expected the opposite")
881
+
882
+ return {
883
+ "key": key,
884
+ "mode": scenario["mode"],
885
+ "message": scenario["message"],
886
+ "seed": scenario["seed"],
887
+ "scripted_llm": scenario["script"],
888
+ "governor_verdict": scenario["governor_verdict"],
889
+ "req": {"message": scenario["message"], "user_email": "owner@example.com",
890
+ "max_steps": request.max_steps, "temperature": request.temperature},
891
+ "paused": paused,
892
+ "approval_requirements": normalize(requirements, resolved),
893
+ "final_state": ctx.state.value,
894
+ "final_message": normalize(ctx.final_message, resolved),
895
+ "state_history": ctx.state_history,
896
+ "transcript": normalize(ctx.transcript, resolved),
897
+ "rollback_log": normalize(ctx.rollback_log, resolved),
898
+ "loop": ctx.trace.summary(),
899
+ "tool_calls": normalize(tool_calls, resolved),
900
+ "audit": normalize(audit, resolved),
901
+ "llm_calls": len(llm.calls),
902
+ "unused_script": len(llm.outputs),
903
+ }
904
+
905
+
906
+ def _scripted_write_content(output: str) -> Optional[str]:
907
+ """The `content` of a scripted `write_file` action, when it is one."""
908
+ try:
909
+ payload = json.loads(output)
910
+ except (json.JSONDecodeError, TypeError):
911
+ return None
912
+ if not isinstance(payload, dict) or payload.get("action") != "write_file":
913
+ return None
914
+ content = (payload.get("args") or {}).get("content")
915
+ return content if isinstance(content, str) else None
916
+
917
+
918
+ def policy_payload() -> Dict[str, Any]:
919
+ """The real registry, as the data the native loop takes as input."""
920
+ return {
921
+ "tools": {name: dict(policy) for name, policy in sorted(TOOL_GOVERNANCE.items())},
922
+ "default": dict(TOOL_GOVERNANCE_DEFAULT),
923
+ "blocked_write_prefixes": list(LOCAL_WRITE_BLOCKED_PREFIXES),
924
+ }
925
+
926
+
927
+ def manifest_payload() -> Dict[str, Any]:
928
+ return {
929
+ "schema": SCHEMA,
930
+ "scenarios": sorted(SCENARIOS),
931
+ "raw_actions": sorted(RAW_ACTIONS),
932
+ "plan_cases": sorted(PLAN_CASES),
933
+ "normalization": [
934
+ "the absolute workspace root becomes <AGENT_ROOT>",
935
+ "keys named `at` are dropped (trace timestamps)",
936
+ "keys named `stderr` are dropped (git text is version-specific)",
937
+ f"a string starting with {DECODER_DETAIL_PREFIX!r} keeps only that prefix",
938
+ ],
939
+ "constants": {
940
+ "file_create_actions": sorted(FILE_CREATE_ACTIONS),
941
+ "scoped_knowledge_tools": sorted(SCOPED_KNOWLEDGE_TOOLS),
942
+ "governed_tools": sorted(ScriptedGovernor.governed_tools),
943
+ "phase_budgets": PhaseBudgets().__dict__,
944
+ "transcript_budget": TranscriptBudget().__dict__,
945
+ "max_state_history": 200,
946
+ "max_retry": 3,
947
+ "compact_max_params_b": COMPACT_MAX_PARAMS_B,
948
+ },
949
+ }
950
+
951
+
952
+ async def build_async(base: Path) -> Dict[str, Any]:
953
+ """Everything the goldens hold, as `{filename: payload}`."""
954
+ verify_root = base / "verify" / "agent_workspace"
955
+ with use_workspace(verify_root) as resolved:
956
+ verification = await verification_rows(resolved)
957
+ trajectories = [await trajectory(key, SCENARIOS[key], base) for key in sorted(SCENARIOS)]
958
+ return {
959
+ "manifest.json": manifest_payload(),
960
+ "policies.json": policy_payload(),
961
+ "helpers.json": helper_rows(),
962
+ "verification.json": {"schema": SCHEMA, "cases": verification},
963
+ "run_store.json": {"schema": SCHEMA, "cases": run_store_rows()},
964
+ "trajectories.json": {"schema": SCHEMA, "cases": trajectories},
965
+ }
966
+
967
+
968
+ def build(base: Optional[Path] = None) -> Dict[str, Any]:
969
+ """Synchronous entry point, for the generator and the contract test alike."""
970
+ if base is None:
971
+ base = Path(tempfile.mkdtemp(prefix="agent-loop-fixtures-"))
972
+ return asyncio.run(build_async(Path(base)))
973
+
974
+
975
+ def write(payloads: Dict[str, Any]) -> List[str]:
976
+ GOLDEN_DIR.mkdir(parents=True, exist_ok=True)
977
+ for name, payload in payloads.items():
978
+ (GOLDEN_DIR / name).write_text(
979
+ json.dumps(payload, ensure_ascii=False, indent=2, sort_keys=True) + "\n",
980
+ encoding="utf-8",
981
+ )
982
+ return sorted(payloads)
983
+
984
+
985
+ def main() -> int:
986
+ written = write(build())
987
+ for name in written:
988
+ path = GOLDEN_DIR / name
989
+ print(f"wrote {path.relative_to(REPO_ROOT)} ({path.stat().st_size:,} bytes)")
990
+ return 0
991
+
992
+
993
+ if __name__ == "__main__":
994
+ raise SystemExit(main())