ltcai 11.2.0 → 11.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (247) hide show
  1. package/README.md +46 -53
  2. package/docs/CHANGELOG.md +61 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/MULTI_AGENT_RUNTIME.md +1 -1
  6. package/docs/ONBOARDING.md +1 -1
  7. package/docs/OPERATIONS.md +6 -2
  8. package/docs/PERMISSION_MODE.md +1 -1
  9. package/docs/TRUST_MODEL.md +1 -1
  10. package/docs/WHY_LATTICE.md +1 -1
  11. package/docs/kg-schema.md +2 -2
  12. package/docs/v11.3.0_PLAN.md +202 -0
  13. package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
  14. package/lattice_brain/__init__.py +1 -1
  15. package/lattice_brain/graph/_kg_common/__init__.py +287 -0
  16. package/lattice_brain/graph/_kg_common/extraction.py +516 -0
  17. package/lattice_brain/graph/_kg_common/relations.py +161 -0
  18. package/lattice_brain/graph/_kg_common/text.py +479 -0
  19. package/lattice_brain/graph/discovery_index/__init__.py +35 -0
  20. package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
  21. package/lattice_brain/graph/discovery_index/extract.py +137 -0
  22. package/lattice_brain/graph/discovery_index/scan.py +411 -0
  23. package/lattice_brain/graph/discovery_index/upsert.py +495 -0
  24. package/lattice_brain/graph/projection/__init__.py +42 -0
  25. package/lattice_brain/graph/projection/curation.py +500 -0
  26. package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
  27. package/lattice_brain/graph/retrieval/__init__.py +54 -0
  28. package/lattice_brain/graph/retrieval/context.py +197 -0
  29. package/lattice_brain/graph/retrieval/graph_view.py +319 -0
  30. package/lattice_brain/graph/retrieval/hybrid.py +488 -0
  31. package/lattice_brain/graph/retrieval/maintenance.py +121 -0
  32. package/lattice_brain/graph/retrieval/signals.py +95 -0
  33. package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
  34. package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
  35. package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
  36. package/lattice_brain/graph/retrieval_vector/search.py +560 -0
  37. package/lattice_brain/graph/retrieval_vector/status.py +374 -0
  38. package/lattice_brain/ingestion/__init__.py +130 -0
  39. package/lattice_brain/ingestion/_contract.py +90 -0
  40. package/lattice_brain/ingestion/constants.py +127 -0
  41. package/lattice_brain/ingestion/folder_scan.py +57 -0
  42. package/lattice_brain/ingestion/folders.py +258 -0
  43. package/lattice_brain/ingestion/hashing.py +26 -0
  44. package/lattice_brain/ingestion/jobs_api.py +107 -0
  45. package/lattice_brain/ingestion/models.py +80 -0
  46. package/lattice_brain/ingestion/pipeline.py +486 -0
  47. package/lattice_brain/ingestion/quality.py +209 -0
  48. package/lattice_brain/ingestion/routing.py +295 -0
  49. package/lattice_brain/multimodal/__init__.py +164 -0
  50. package/lattice_brain/multimodal/audio.py +77 -0
  51. package/lattice_brain/multimodal/common.py +118 -0
  52. package/lattice_brain/multimodal/images.py +498 -0
  53. package/lattice_brain/multimodal/ports.py +169 -0
  54. package/lattice_brain/multimodal/video.py +410 -0
  55. package/lattice_brain/portability/__init__.py +90 -0
  56. package/lattice_brain/portability/_contract.py +42 -0
  57. package/lattice_brain/portability/backups.py +338 -0
  58. package/lattice_brain/portability/bundles.py +136 -0
  59. package/lattice_brain/portability/constants.py +93 -0
  60. package/lattice_brain/portability/fsops.py +138 -0
  61. package/lattice_brain/portability/service.py +41 -0
  62. package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
  63. package/lattice_brain/runtime/__init__.py +1 -1
  64. package/lattice_brain/runtime/multi_agent.py +1 -1
  65. package/latticeai/__init__.py +1 -1
  66. package/latticeai/api/chronicle.py +63 -0
  67. package/latticeai/core/agent/__init__.py +93 -0
  68. package/latticeai/core/agent/_contract.py +79 -0
  69. package/latticeai/core/agent/context.py +57 -0
  70. package/latticeai/core/agent/deps.py +125 -0
  71. package/latticeai/core/agent/execution.py +622 -0
  72. package/latticeai/core/agent/planning.py +145 -0
  73. package/latticeai/core/agent/recovery.py +157 -0
  74. package/latticeai/core/agent/runtime.py +210 -0
  75. package/latticeai/core/agent/verification.py +231 -0
  76. package/latticeai/core/embedding_providers/__init__.py +151 -0
  77. package/latticeai/core/embedding_providers/base.py +199 -0
  78. package/latticeai/core/embedding_providers/captions.py +162 -0
  79. package/latticeai/core/embedding_providers/profiles.py +126 -0
  80. package/latticeai/core/embedding_providers/text.py +350 -0
  81. package/latticeai/core/embedding_providers/vision.py +352 -0
  82. package/latticeai/core/file_generation/__init__.py +115 -0
  83. package/latticeai/core/file_generation/bundles.py +76 -0
  84. package/latticeai/core/file_generation/extraction.py +154 -0
  85. package/latticeai/core/file_generation/inference.py +235 -0
  86. package/latticeai/core/file_generation/orchestration.py +152 -0
  87. package/latticeai/core/file_generation/prompting.py +117 -0
  88. package/latticeai/core/file_generation/repair.py +114 -0
  89. package/latticeai/core/file_generation/sanitize.py +61 -0
  90. package/latticeai/core/file_generation/validation.py +201 -0
  91. package/latticeai/core/legacy_compatibility.py +1 -1
  92. package/latticeai/core/marketplace.py +1 -1
  93. package/latticeai/core/messages.py +9 -0
  94. package/latticeai/core/workspace_os_constants.py +1 -1
  95. package/latticeai/integrations/telegram_bot/__init__.py +123 -0
  96. package/latticeai/integrations/telegram_bot/__main__.py +17 -0
  97. package/latticeai/integrations/telegram_bot/config.py +86 -0
  98. package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
  99. package/latticeai/integrations/telegram_bot/flows.py +478 -0
  100. package/latticeai/integrations/telegram_bot/helpers.py +322 -0
  101. package/latticeai/integrations/telegram_bot/screens.py +394 -0
  102. package/latticeai/models/router/__init__.py +88 -0
  103. package/latticeai/models/router/_contract.py +66 -0
  104. package/latticeai/models/router/branding.py +56 -0
  105. package/latticeai/models/router/catalog.py +69 -0
  106. package/latticeai/models/router/documents.py +199 -0
  107. package/latticeai/models/router/errors.py +37 -0
  108. package/latticeai/models/router/generation.py +258 -0
  109. package/latticeai/models/router/loading.py +291 -0
  110. package/latticeai/models/router/local_models.py +85 -0
  111. package/latticeai/models/router/registry.py +147 -0
  112. package/latticeai/runtime/build_phases/__init__.py +82 -0
  113. package/latticeai/runtime/build_phases/features.py +407 -0
  114. package/latticeai/runtime/build_phases/foundation.py +555 -0
  115. package/latticeai/runtime/build_phases/web.py +492 -0
  116. package/latticeai/runtime/runtime_context.py +1 -0
  117. package/latticeai/services/architecture_readiness.py +48 -19
  118. package/latticeai/services/brain_intelligence/__init__.py +58 -0
  119. package/latticeai/services/brain_intelligence/_contract.py +71 -0
  120. package/latticeai/services/brain_intelligence/consistency.py +193 -0
  121. package/latticeai/services/brain_intelligence/constants.py +47 -0
  122. package/latticeai/services/brain_intelligence/digest.py +258 -0
  123. package/latticeai/services/brain_intelligence/health.py +331 -0
  124. package/latticeai/services/brain_intelligence/proposals.py +264 -0
  125. package/latticeai/services/brain_intelligence/sampling.py +84 -0
  126. package/latticeai/services/brain_intelligence/service.py +48 -0
  127. package/latticeai/services/chronicle.py +557 -0
  128. package/latticeai/services/memory_service/__init__.py +52 -0
  129. package/latticeai/services/memory_service/_contract.py +100 -0
  130. package/latticeai/services/memory_service/brief.py +431 -0
  131. package/latticeai/services/memory_service/constants.py +57 -0
  132. package/latticeai/services/memory_service/maintenance.py +138 -0
  133. package/latticeai/services/memory_service/manager.py +186 -0
  134. package/latticeai/services/memory_service/proof.py +136 -0
  135. package/latticeai/services/memory_service/recall.py +225 -0
  136. package/latticeai/services/memory_service/service.py +48 -0
  137. package/latticeai/services/memory_service/stores.py +110 -0
  138. package/latticeai/services/model_runtime/__init__.py +322 -0
  139. package/latticeai/services/model_runtime/cloud.py +87 -0
  140. package/latticeai/services/model_runtime/download.py +282 -0
  141. package/latticeai/services/model_runtime/engines.py +341 -0
  142. package/latticeai/services/model_runtime/loading.py +178 -0
  143. package/latticeai/services/model_runtime/service.py +129 -0
  144. package/latticeai/services/model_runtime/state.py +131 -0
  145. package/latticeai/services/model_runtime/status.py +255 -0
  146. package/latticeai/services/product_readiness.py +15 -7
  147. package/latticeai/setup/wizard/__init__.py +126 -0
  148. package/latticeai/setup/wizard/catalog.py +172 -0
  149. package/latticeai/setup/wizard/detect.py +323 -0
  150. package/latticeai/setup/wizard/install.py +348 -0
  151. package/latticeai/setup/wizard/paths.py +168 -0
  152. package/latticeai/setup/wizard/plans.py +74 -0
  153. package/latticeai/setup/wizard/recommend.py +320 -0
  154. package/package.json +6 -2
  155. package/scripts/bump_version.py +14 -0
  156. package/scripts/capture_release_evidence.mjs +33 -21
  157. package/scripts/check_current_release_docs.mjs +1 -1
  158. package/scripts/check_i18n_namespace_coverage.mjs +41 -4
  159. package/scripts/check_max_file_lines.mjs +102 -0
  160. package/scripts/check_release_evidence_bound.mjs +30 -15
  161. package/scripts/check_screenshot_pixel_delta.py +34 -4
  162. package/scripts/check_server_i18n.mjs +1 -0
  163. package/scripts/generate_rust_parity_fixtures.py +562 -0
  164. package/scripts/lib/mock_server_fingerprint.mjs +94 -0
  165. package/scripts/release_screen_claims.json +31 -2
  166. package/src-tauri/Cargo.lock +361 -3
  167. package/src-tauri/Cargo.toml +6 -1
  168. package/src-tauri/src/backend.rs +349 -0
  169. package/src-tauri/src/folder.rs +33 -0
  170. package/src-tauri/src/main.rs +97 -399
  171. package/src-tauri/tauri.conf.json +1 -1
  172. package/static/app/asset-manifest.json +41 -37
  173. package/static/app/assets/Act-yYpYnn0v.js +1 -0
  174. package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
  175. package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
  176. package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
  177. package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
  178. package/static/app/assets/Capture-CFIRsFNE.js +1 -0
  179. package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
  180. package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
  181. package/static/app/assets/Library-DwO3yZST.js +1 -0
  182. package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
  183. package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
  184. package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
  185. package/static/app/assets/System-DW8F-2xL.js +1 -0
  186. package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
  187. package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
  188. package/static/app/assets/brain-Ci1CkWjM.js +1 -0
  189. package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
  190. package/static/app/assets/circle-check-DfInj-qD.js +1 -0
  191. package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
  192. package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
  193. package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
  194. package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
  195. package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
  196. package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
  197. package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
  198. package/static/app/assets/index-_u5iUHDr.js +10 -0
  199. package/static/app/assets/input-B0lPdRQZ.js +1 -0
  200. package/static/app/assets/link-2-CoFbooHS.js +1 -0
  201. package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
  202. package/static/app/assets/primitives-DEbN-d6p.js +1 -0
  203. package/static/app/assets/search-BybIWPNd.js +1 -0
  204. package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
  205. package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
  206. package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
  207. package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
  208. package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
  209. package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
  210. package/static/app/assets/utils-BlZr7Pd4.js +4 -0
  211. package/static/app/assets/workspace-jJY4RuAV.js +1 -0
  212. package/static/app/index.html +4 -4
  213. package/static/sw.js +1 -1
  214. package/lattice_brain/graph/_kg_common.py +0 -1331
  215. package/lattice_brain/graph/discovery_index.py +0 -1141
  216. package/lattice_brain/graph/retrieval.py +0 -1120
  217. package/lattice_brain/graph/retrieval_vector.py +0 -1293
  218. package/lattice_brain/ingestion.py +0 -1525
  219. package/lattice_brain/multimodal.py +0 -1258
  220. package/latticeai/core/agent.py +0 -1465
  221. package/latticeai/core/embedding_providers.py +0 -1196
  222. package/latticeai/core/file_generation.py +0 -1047
  223. package/latticeai/integrations/telegram_bot.py +0 -1390
  224. package/latticeai/models/router.py +0 -1007
  225. package/latticeai/runtime/build_phases.py +0 -1450
  226. package/latticeai/services/brain_intelligence.py +0 -1083
  227. package/latticeai/services/memory_service.py +0 -1177
  228. package/latticeai/services/model_runtime.py +0 -1281
  229. package/latticeai/setup/wizard.py +0 -1310
  230. package/static/app/assets/Act-AWf0SAKp.js +0 -1
  231. package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
  232. package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
  233. package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
  234. package/static/app/assets/Capture-CqOSzyPr.js +0 -1
  235. package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
  236. package/static/app/assets/Library-CX-bbhmK.js +0 -1
  237. package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
  238. package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
  239. package/static/app/assets/System-Bu2t5hn1.js +0 -1
  240. package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
  241. package/static/app/assets/brain-DJMoqrwx.js +0 -1
  242. package/static/app/assets/index-BpYkzcVm.js +0 -10
  243. package/static/app/assets/input-DSlJJxRs.js +0 -1
  244. package/static/app/assets/primitives-BCx6TvfG.js +0 -1
  245. package/static/app/assets/search-Cgy8cCFJ.js +0 -1
  246. package/static/app/assets/utils-zqPZJxdx.js +0 -4
  247. package/static/app/assets/workspace-DXTihhfU.js +0 -1
@@ -1,1465 +0,0 @@
1
- """Single-agent runtime — the Discover→Plan→Implement→Verify state machine.
2
-
3
- This module is the deep single-agent loop: a small interface (``AgentDeps`` ports +
4
- ``SingleAgentRuntime.run_to_completion``) over the whole role-phased state machine
5
- (planner → executor → critic → rollback → memory). It carries no FastAPI,
6
- no globals, and no I/O of its own — every collaborator is injected through
7
- ``AgentDeps``.
8
-
9
- Two adapters justify the seam:
10
-
11
- * production wires ``AgentDeps`` from ``latticeai.server_app``'s ``LLMRouter``, governance
12
- map, audit log, and prompts;
13
- * tests pass fake ports (an LLM that returns canned JSON, a recording tool
14
- executor) and drive a full PLAN→EXECUTE→VERIFY→DONE cycle without a server.
15
-
16
- HTTP concerns — request parsing, chat-history persistence, response shaping,
17
- scheduling the background memory update — stay in the app layer. This module
18
- only owns the state machine.
19
- """
20
-
21
- from __future__ import annotations
22
-
23
- import json
24
- import logging
25
- from dataclasses import dataclass
26
- from pathlib import Path
27
- from typing import (
28
- Any,
29
- Awaitable,
30
- Callable,
31
- Dict,
32
- FrozenSet,
33
- List,
34
- Mapping,
35
- Optional,
36
- Tuple,
37
- )
38
-
39
- from lattice_brain.runtime.contracts import (
40
- runtime_boundary_contract,
41
- single_agent_contract,
42
- )
43
- from lattice_brain.runtime.hooks import dispatch_tool
44
- from latticeai.core.agent_helpers import (
45
- PhaseBudgets,
46
- TranscriptBudget,
47
- _truncate_strings,
48
- artifact_checklist,
49
- compact_transcript,
50
- extract_action,
51
- extract_action_details,
52
- files_written,
53
- filter_learnings,
54
- format_artifact_checklist,
55
- format_requirement_coverage,
56
- normalize_plan,
57
- requirement_coverage,
58
- )
59
- from latticeai.core.agent_permission import (
60
- block_reason_for_tool,
61
- non_auto_plan_steps,
62
- resolve_deps_mode,
63
- )
64
- from latticeai.core.agent_profiles import AgentProfile, profile_for_model
65
- from latticeai.core.agent_prompts import executor_prompt_for
66
-
67
- # The state vocabulary and the pure helpers live in sibling modules so this one
68
- # holds only the loop. They are re-exported (see ``__all__``) because callers —
69
- # the HTTP layer, run_store, the eval harness, and the tests — have always
70
- # imported them from here, and that contract does not change.
71
- from latticeai.core.agent_state import AGENT_TERMINAL_STATES, AgentState
72
- from latticeai.core.agent_trace import LoopTrace
73
- from latticeai.core.file_generation import (
74
- generate_file_content,
75
- infer_file_target,
76
- sanitize_write_content,
77
- )
78
- from latticeai.core.permission_mode import (
79
- PermissionMode,
80
- is_circuit_breaker,
81
- plan_requires_approval,
82
- should_stage_proposal,
83
- )
84
- from latticeai.core.tool_governor import classify_tool_call
85
- from latticeai.core.tool_registry import SCOPED_KNOWLEDGE_TOOLS
86
- from latticeai.tools import ToolError, document_output_target
87
-
88
- __all__ = [
89
- # this module
90
- "AgentDeps",
91
- "AgentRunContext",
92
- "SingleAgentRuntime",
93
- # re-exported from agent_state
94
- "AGENT_TERMINAL_STATES",
95
- "AgentState",
96
- # re-exported from agent_helpers
97
- "PhaseBudgets",
98
- "TranscriptBudget",
99
- "artifact_checklist",
100
- "compact_transcript",
101
- "extract_action",
102
- "extract_action_details",
103
- "files_written",
104
- "filter_learnings",
105
- "format_artifact_checklist",
106
- "format_requirement_coverage",
107
- "normalize_plan",
108
- "requirement_coverage",
109
- ]
110
-
111
-
112
- class AgentRunContext:
113
- """Mutable state carrier passed through all agent phases."""
114
- __slots__ = ("state", "plan", "transcript", "retry_count",
115
- "state_history", "corrections", "final_message", "rollback_log",
116
- "executing_model", "reviewing_model", "approved_by_human", "trace",
117
- "on_step", "project_context", "permission_mode",
118
- "self_model_summary")
119
-
120
- def __init__(self) -> None:
121
- self.state: AgentState = AgentState.IDLE
122
- self.trace: LoopTrace = LoopTrace()
123
- self.plan: dict = {}
124
- self.transcript: list = []
125
- self.retry_count: int = 0
126
- self.state_history: list = []
127
- self.corrections: list = []
128
- self.final_message: str = ""
129
- self.rollback_log: list = []
130
- self.executing_model: Optional[str] = None
131
- self.reviewing_model: Optional[str] = None
132
- self.approved_by_human: bool = False
133
- # Per-run step observer (review Wave 1.1): the HTTP layer attaches a
134
- # callback here so live SSE clients see progress while EXECUTING.
135
- # Never serialized; a broken observer never breaks the loop.
136
- self.on_step: Optional[Callable[[Dict[str, Any]], None]] = None
137
- # Multi-turn project loop (v9.9.6): a prompt block describing where the
138
- # project stands — files already produced, open TODOs, the last honest
139
- # verification. Empty for a standalone run, which behaves exactly as
140
- # before. Set by the HTTP layer, read by plan/execute/verify.
141
- self.project_context: str = ""
142
- # Autonomy dial resolved once per run (v9.9.8). The HTTP layer stamps
143
- # the user/workspace-scoped mode here so the plan gate and every
144
- # per-tool gate in the same run agree; ``None`` falls back to the
145
- # process-wide resolver on ``deps``.
146
- self.permission_mode: Optional[str] = None
147
- # Self-Model summary resolved once per run (v11.2.0). The executor
148
- # prompt is rebuilt on every turn of the loop; reading the profile
149
- # once and reusing it keeps a graph query off that hot path — and
150
- # keeps every turn of one run describing the same person. ``None``
151
- # means "not resolved yet"; ``""`` is a resolved, empty profile.
152
- self.self_model_summary: Optional[str] = None
153
-
154
-
155
- @dataclass
156
- class AgentDeps:
157
- """The ports a :class:`SingleAgentRuntime` needs from the outside world.
158
-
159
- Everything the state machine touches is here, so the loop can be exercised
160
- against fakes. See module docstring for the two-adapter rationale.
161
- """
162
-
163
- # ── LLM port ─────────────────────────────────────────────────────
164
- # generate_as(model_id, message, context, max_tokens, temperature) -> str
165
- generate_as: Callable[..., Awaitable[Any]]
166
- # generate(message, context, max_tokens, temperature) -> str
167
- generate: Callable[..., Awaitable[Any]]
168
-
169
- # ── tool port ────────────────────────────────────────────────────
170
- execute_tool: Callable[[str, dict], dict]
171
- policy_for: Callable[[str, dict], Mapping[str, Any]] # name, args -> policy
172
- risk_level: Callable[[Any], str] # policy -> "low"|"medium"|"high"
173
- check_role: Callable[[str, str], None] # tool_name, user -> raises if not allowed
174
- tool_governance: Mapping[str, Mapping[str, Any]] # name -> policy (auto_approve set)
175
- file_create_actions: FrozenSet[str]
176
-
177
- # ── context / memory / audit ports ───────────────────────────────
178
- recent_chat_context: Callable[..., str] # (conversation_id=...) -> str
179
- clear_history: Callable[[int], dict]
180
- knowledge_save: Callable[..., Any]
181
- audit: Callable[..., None] # (event, **kw) -> None
182
-
183
- # ── prompts + config ─────────────────────────────────────────────
184
- planner_prompt: str
185
- executor_prompt: str
186
- critic_prompt: str
187
- memory_updater_prompt: str
188
- agent_root: Path
189
-
190
- # ── rollback port (optional) ─────────────────────────────────────
191
- # Production injects this from the tool dispatch service so this pure
192
- # state machine does not shell out directly. Tests can pass a recorder.
193
- rollback_file: Optional[Callable[[str], Dict[str, Any]]] = None
194
-
195
- # ── snapshot rollback ports (optional, review L7) ────────────────
196
- # git-only rollback left non-git workspaces and newly created files
197
- # unrecoverable. ``snapshot_file(path)`` captures pre-write state
198
- # ({"existed", "content", "too_large"}) before a file-create action;
199
- # ``restore_snapshot(path, content)`` restores it (content=None deletes
200
- # a file the run created). Both are production-wired with workspace
201
- # path safety; tests pass recorders.
202
- snapshot_file: Optional[Callable[[str], Dict[str, Any]]] = None
203
- restore_snapshot: Optional[Callable[[str, Optional[str]], Dict[str, Any]]] = None
204
-
205
- # ── lifecycle hooks port (optional) ──────────────────────────────
206
- # When present, every tool execution fires the shared pre_tool/post_tool
207
- # lifecycle, so the agent tool path no longer bypasses hooks.
208
- hooks: Any = None
209
-
210
- # ── brain memory port (optional) ─────────────────────────────────
211
- # When present, completed-run learnings become typed Experience records
212
- # through the unified ingestion pipeline (with provenance), replacing
213
- # the vault markdown dump.
214
- brain_memory: Any = None
215
-
216
- # ── change governor port (optional) ──────────────────────────────
217
- # When present, file writes are classified centrally: additive creates
218
- # run with minimal friction, while mutations/deletions of existing
219
- # content are staged as review proposals instead of applied. The port is
220
- # ``review(name, args, policy=..., user_email=..., workspace_id=...)``
221
- # returning None (fall through to the classic gates) or a verdict dict.
222
- change_governor: Any = None
223
-
224
- # ── phase budgets (optional) ─────────────────────────────────────
225
- # Per-phase token caps (plan/execute/verify/memory). None reads the
226
- # environment once at first use; tests inject a fixed PhaseBudgets.
227
- phase_budgets: Optional[PhaseBudgets] = None
228
-
229
- # ── transcript shaping (optional) ────────────────────────────────
230
- # Executor/critic prompt window caps. None reads the environment once;
231
- # tests inject a fixed TranscriptBudget.
232
- transcript_budget: Optional["TranscriptBudget"] = None
233
-
234
- # ── step observer port (optional) ────────────────────────────────
235
- # Default per-runtime observer for live step events; a per-run observer
236
- # can also be attached on AgentRunContext.on_step. Both are advisory.
237
- on_step: Optional[Callable[[Dict[str, Any]], None]] = None
238
-
239
- # ── agent profile (optional, v9.9.7) ─────────────────────────────
240
- # How hard the loop works to keep a weak model on contract. None selects
241
- # per-run from the executing model id (``profile_for_model``); tests
242
- # inject a fixed profile.
243
- agent_profile: Optional["AgentProfile"] = None
244
-
245
- # ── permission mode port (optional, v9.9.8) ──────────────────────
246
- # The autonomy dial. Either a static mode, or a resolver callable that
247
- # accepts ``user_email``/``workspace_id`` scope kwargs (preferred) or no
248
- # arguments at all — see ``call_mode_source``. ``None`` means strict,
249
- # which is exactly the pre-9.9.8 behaviour.
250
- permission_mode: Any = None
251
-
252
- # ── Self-Model port (optional, v11.2.0) ──────────────────────────
253
- # What the Brain has learned about its owner, injected into the executor
254
- # prompt. Either a fixed string or a resolver callable taking
255
- # ``user_email``/``workspace_id`` scope kwargs (or none at all). 11.1.0
256
- # built ``executor_prompt_for(self_model_summary=…)`` and then had nothing
257
- # to pass it, because the runtime held its executor prompt as a fixed
258
- # string; this is the port that was missing. ``None`` — and an empty
259
- # summary — produce exactly the prompt bytes the loop produced before it
260
- # existed.
261
- self_model_summary: Any = None
262
-
263
-
264
- class SingleAgentRuntime:
265
- """Drives the agent state machine over injected :class:`AgentDeps`."""
266
-
267
- def __init__(self, deps: AgentDeps) -> None:
268
- self.deps = deps
269
- self._env_phase_budgets: Optional[PhaseBudgets] = None
270
- self._env_transcript_budget: Optional[TranscriptBudget] = None
271
-
272
- @property
273
- def phase_budgets(self) -> PhaseBudgets:
274
- # getattr twice: partially-constructed runtimes/deps (tests build them
275
- # via __new__ or minimal fakes) still get working default budgets.
276
- injected = getattr(self.deps, "phase_budgets", None)
277
- if injected is not None:
278
- return injected
279
- cached = getattr(self, "_env_phase_budgets", None)
280
- if cached is None:
281
- cached = PhaseBudgets.from_env()
282
- self._env_phase_budgets = cached
283
- return cached
284
-
285
- @property
286
- def transcript_budget(self) -> TranscriptBudget:
287
- injected = getattr(self.deps, "transcript_budget", None)
288
- if injected is not None:
289
- return injected
290
- cached = getattr(self, "_env_transcript_budget", None)
291
- if cached is None:
292
- cached = TranscriptBudget.from_env()
293
- self._env_transcript_budget = cached
294
- return cached
295
-
296
- # ── permission mode (v9.9.8) ─────────────────────────────────────
297
- def resolve_permission_mode(
298
- self,
299
- ctx: Optional[AgentRunContext] = None,
300
- *,
301
- user_email: Optional[str] = None,
302
- workspace_id: Optional[str] = None,
303
- ) -> PermissionMode:
304
- """Autonomy dial for this run.
305
-
306
- A mode stamped on ``ctx`` wins, so the plan a user approved and every
307
- tool step in the same run are judged by one dial even if the stored
308
- preference changes mid-run. Otherwise the resolver on ``deps`` is
309
- consulted with the caller's scope — resolving unscoped would collapse
310
- every caller onto the process-wide default.
311
- """
312
- return resolve_deps_mode(
313
- self.deps, ctx, user_email=user_email, workspace_id=workspace_id,
314
- )
315
-
316
- def _governed_path_exists(self, name: str, path: str) -> bool:
317
- """Does this tool call's *real* target already exist?
318
-
319
- The document creators sanitize ``filename`` into their own output
320
- directory, so the raw argument is resolved through
321
- :func:`document_output_target` first — checking it verbatim would
322
- inspect a path nothing ever writes and the fail-closed overwrite guard
323
- would never fire. Workspace-relative paths resolve under
324
- ``deps.agent_root``; absolute paths (home-sandbox writes) are honored
325
- as-is. Never raises: governance must not be able to crash the loop, and
326
- an unresolvable path degrades to "new file", which the remaining gates
327
- still cover.
328
- """
329
- try:
330
- candidate = Path(document_output_target(name, path) or path)
331
- if not candidate.is_absolute():
332
- candidate = Path(self.deps.agent_root) / candidate
333
- return candidate.exists()
334
- except Exception: # noqa: BLE001 — classification is best-effort
335
- return False
336
-
337
- def _governed_tools(self) -> FrozenSet[str]:
338
- governor = getattr(self.deps, "change_governor", None)
339
- if governor is None:
340
- return frozenset()
341
- return frozenset(getattr(governor, "governed_tools", frozenset()))
342
-
343
- def profile_for(self, model_id: Optional[str]) -> AgentProfile:
344
- """Loop profile for the model actually executing this run (v9.9.7).
345
-
346
- An injected ``AgentDeps.agent_profile`` wins (tests, explicit config);
347
- otherwise the profile is derived from the model id, so a small local
348
- model gets the compact loop without any extra configuration.
349
- """
350
- injected = getattr(self.deps, "agent_profile", None)
351
- if injected is not None:
352
- return injected
353
- return profile_for_model(model_id)
354
-
355
- def _emit_step(self, ctx: AgentRunContext, phase: str, event: str, **details: Any) -> None:
356
- """Fire the per-run / deps step observers (review Wave 1.1).
357
-
358
- Observers power the live step timeline in the UI. They are pure
359
- telemetry: any observer failure is logged and swallowed — the loop
360
- itself must never notice.
361
- """
362
- payload: Dict[str, Any] = {"phase": phase, "event": event}
363
- for key, value in details.items():
364
- if value is not None:
365
- payload[key] = value
366
- for observer in (getattr(ctx, "on_step", None), getattr(self.deps, "on_step", None)):
367
- if observer is None:
368
- continue
369
- try:
370
- observer(dict(payload))
371
- except Exception as exc: # noqa: BLE001 — observers are advisory
372
- logging.warning("agent step observer failed: %s", exc)
373
-
374
- @staticmethod
375
- def _project_block(ctx: AgentRunContext) -> str:
376
- """Project-session context for prompts, or "" for a standalone run.
377
-
378
- Multi-turn project loop (v9.9.6): a later run must see the files the
379
- project already produced and what is still open, instead of planning
380
- from a blank workspace every time.
381
- """
382
- summary = str(getattr(ctx, "project_context", "") or "").strip()
383
- return f"\n\n[PROJECT SESSION]\n{summary}" if summary else ""
384
-
385
- def boundary(self) -> Dict[str, Any]:
386
- return runtime_boundary_contract(
387
- name="SingleAgentRuntime",
388
- runtime="single_agent",
389
- entrypoint="latticeai.core.agent.SingleAgentRuntime",
390
- surface="/agent",
391
- owns="single-agent PLAN / EXECUTE / VERIFY state machine over injected ports",
392
- compatibility_aliases=[],
393
- )
394
-
395
- def config(self) -> Dict[str, Any]:
396
- return {
397
- "boundary": self.boundary(),
398
- "states": [state.value for state in AgentState],
399
- "terminal_states": sorted(state.value for state in AGENT_TERMINAL_STATES),
400
- "execution_mode": "injected_ports",
401
- }
402
-
403
- def contract(self, ctx: AgentRunContext, req: Any, *, run_id: Optional[str] = None) -> Dict[str, Any]:
404
- """Expose the shared agent-run contract for the single-agent loop."""
405
- return single_agent_contract(ctx=ctx, goal=getattr(req, "message", ""), run_id=run_id)
406
-
407
- # ── PLAN ─────────────────────────────────────────────────────────
408
- async def plan(
409
- self, ctx: AgentRunContext, req: Any, lang_hint: str, current_user: str,
410
- model_id: Optional[str] = None,
411
- ) -> None:
412
- """PLAN: Planner role produces a structured plan JSON."""
413
- d = self.deps
414
- project_block = self._project_block(ctx)
415
- context = (
416
- f"{d.planner_prompt}\n\n"
417
- f"[LANGUAGE HINT: {lang_hint}]\n"
418
- f"Workspace root: {d.agent_root}{project_block}\n\n"
419
- f"User request: {req.message}"
420
- )
421
- raw = await d.generate_as(
422
- model_id,
423
- message="Produce a JSON execution plan for this request.",
424
- context=context, max_tokens=self.phase_budgets.plan_tokens, temperature=0.1,
425
- )
426
- ctx.trace.llm_call("plan", model=model_id)
427
- try:
428
- plan, plan_repairs = extract_action_details(str(raw))
429
- ctx.trace.repair("plan", repairs=plan_repairs)
430
- except ValueError as exc:
431
- ctx.trace.parse_error("plan", error=str(exc), recovered=True)
432
- plan = {
433
- "action": "plan", "state": "PLAN",
434
- "goal": req.message, "steps": [],
435
- "requires_approval": False, "rollback_strategy": "none", "estimated_steps": 1,
436
- }
437
- plan, plan_fixes = normalize_plan(plan, req.message)
438
- if plan_fixes:
439
- ctx.trace.repair("plan", repairs=plan_fixes)
440
- ctx.plan = plan
441
- ctx.transcript.append({
442
- "state": AgentState.PLANNING.value,
443
- "goal": plan.get("goal", req.message),
444
- "steps": plan.get("steps", []),
445
- "requires_approval": plan.get("requires_approval", False),
446
- "rollback_strategy": plan.get("rollback_strategy", "none"),
447
- "estimated_steps": plan.get("estimated_steps", 1),
448
- **({"plan_fixes": plan_fixes} if plan_fixes else {}),
449
- })
450
- self._emit_step(
451
- ctx, "plan", "planned",
452
- goal=str(plan.get("goal") or "")[:200],
453
- steps=len(plan.get("steps") or []),
454
- requires_approval=bool(plan.get("requires_approval", False)),
455
- )
456
- ctx.state = AgentState.WAITING_APPROVAL
457
-
458
- # ── APPROVAL ─────────────────────────────────────────────────────
459
- def approval_requirements(self, ctx: AgentRunContext) -> Dict[str, Any]:
460
- """Read-only preview of the approval gate for a planned run.
461
-
462
- Shares the exact predicate :meth:`approve` enforces, so the HTTP
463
- layer can pause a run as ``awaiting_approval`` (with a plan summary
464
- for the user) instead of letting it fail closed — without ever
465
- weakening the gate itself.
466
- """
467
- d = self.deps
468
- mode = self.resolve_permission_mode(ctx)
469
- # Governor-managed tools never hard-block the plan: each call is
470
- # classified at execution time — additive creates run, mutations and
471
- # deletions of existing content become review proposals.
472
- governed_tools = self._governed_tools()
473
- steps = ctx.plan.get("steps", [])
474
- non_auto = non_auto_plan_steps(
475
- mode, steps, d.tool_governance or {}, governed_tools=governed_tools,
476
- )
477
- requires = plan_requires_approval(
478
- mode,
479
- non_auto_steps=non_auto,
480
- plan_flag=bool(ctx.plan.get("requires_approval", False)),
481
- )
482
- lines = [
483
- f"{index}. {step.get('description') or step.get('action') or '?'}"
484
- for index, step in enumerate(steps, start=1)
485
- ]
486
- summary = str(ctx.plan.get("goal") or "").strip()
487
- if lines:
488
- summary = (summary + "\n" if summary else "") + "\n".join(lines)
489
- return {
490
- "requires_approval": requires,
491
- "non_auto_steps": non_auto,
492
- "permission_mode": mode.value,
493
- "plan_summary": summary,
494
- }
495
-
496
- def approve(self, ctx: AgentRunContext, current_user: str, *, approved_by_human: bool = False) -> None:
497
- """APPROVAL: Check governance, log decision, auto-approve (future: UI prompt)."""
498
- d = self.deps
499
- requirements = self.approval_requirements(ctx)
500
- non_auto = requirements["non_auto_steps"]
501
- requires = requirements["requires_approval"]
502
-
503
- ctx.transcript.append({
504
- "state": AgentState.WAITING_APPROVAL.value,
505
- "requires_approval": requires,
506
- "non_auto_approve_steps": non_auto,
507
- "decision": "human_approved" if requires and approved_by_human else ("blocked_pending_approval" if requires else "auto_approved"),
508
- })
509
- decision = "human_approved" if requires and approved_by_human else ("blocked_pending_approval" if requires else "auto_approved")
510
- ctx.trace.decision("approve", decision=decision, non_auto_steps=len(non_auto))
511
- self._emit_step(ctx, "approval", "decision", decision=decision)
512
- d.audit(
513
- "agent_approval", user_email=current_user,
514
- requires_approval=requires,
515
- non_auto_steps=non_auto,
516
- decision=decision,
517
- )
518
- if requires and not approved_by_human:
519
- ctx.final_message = (
520
- "이 작업에는 명시 승인이 필요한 도구가 포함되어 있어 자동 실행을 중단했습니다. "
521
- "human_in_loop 승인 흐름으로 다시 실행해 주세요."
522
- )
523
- ctx.state = AgentState.FAILED
524
- return
525
- ctx.approved_by_human = bool(approved_by_human)
526
- ctx.state = AgentState.EXECUTING
527
-
528
- # ── EXECUTE ──────────────────────────────────────────────────────
529
- async def execute(
530
- self, ctx: AgentRunContext, req: Any, lang_hint: str,
531
- current_user: str, max_steps: int, model_id: Optional[str] = None,
532
- ) -> None:
533
- """EXECUTE: Executor role calls tools one at a time until final or budget exhausted."""
534
- d = self.deps
535
- profile = self.profile_for(model_id)
536
- exec_count = sum(1 for s in ctx.transcript if s.get("state") == AgentState.EXECUTING.value)
537
- budget = max(1, max_steps - exec_count)
538
- parse_failures = 0
539
-
540
- for _ in range(budget):
541
- request_workspace = getattr(req, "workspace_id", None)
542
- context = self._executor_context(
543
- ctx, req, lang_hint, current_user, request_workspace, profile=profile
544
- )
545
- raw = await d.generate_as(
546
- model_id,
547
- message="Execute the next step.",
548
- context=context, max_tokens=self.phase_budgets.execute_tokens,
549
- temperature=req.temperature,
550
- )
551
- ctx.trace.llm_call("execute", model=model_id)
552
- try:
553
- action, exec_repairs = extract_action_details(str(raw))
554
- ctx.trace.repair("execute", repairs=exec_repairs)
555
- except ValueError as exc:
556
- parse_failures += 1
557
- if self._note_parse_failure(ctx, raw, exc, parse_failures, profile):
558
- # Direct-path fallback (v9.9.7): a small model that cannot
559
- # hold the tool-call protocol can still write a file. Run
560
- # the plan's own file steps without asking for any JSON.
561
- if profile.direct_path_fallback and await self._direct_file_path(
562
- ctx, req, current_user, model_id
563
- ):
564
- ctx.state = AgentState.VERIFYING
565
- return
566
- break
567
- continue
568
-
569
- name = str(action.get("action") or "")
570
- thoughts = str(action.get("thoughts") or "")[:600]
571
- args = action.get("args") or {}
572
-
573
- if name in SCOPED_KNOWLEDGE_TOOLS:
574
- # Scope is server-owned, never model-owned. Overwrite any
575
- # claimed values before policy evaluation, audit, and dispatch.
576
- args = dict(args)
577
- args["workspace_id"] = request_workspace or "personal"
578
- args["user_email"] = current_user or "local"
579
-
580
- if name == "final":
581
- ctx.final_message = action.get("message", "작업을 완료했습니다.")
582
- ctx.transcript.append({
583
- "state": AgentState.EXECUTING.value, "action": "final", "thoughts": thoughts,
584
- })
585
- ctx.trace.decision("execute", decision="final")
586
- self._emit_step(ctx, "execute", "final")
587
- ctx.state = AgentState.VERIFYING
588
- return
589
-
590
- # Loop guard
591
- if self._is_repeated_create(ctx, name, args):
592
- ctx.transcript.append({
593
- "state": AgentState.EXECUTING.value, "action": name,
594
- "error": "LOOP_DETECTED: identical action+args repeated — halted.",
595
- })
596
- ctx.trace.decision("execute", decision="loop_detected", tool=name)
597
- self._emit_step(ctx, "execute", "blocked", action=name, reason="loop_detected")
598
- break
599
-
600
- if name == "clear_history":
601
- result = d.clear_history(args.get("keep_last", 0))
602
- ctx.transcript.append({
603
- "state": AgentState.EXECUTING.value, "action": name,
604
- "thoughts": thoughts, "args": args, "result": result,
605
- })
606
- self._emit_step(ctx, "execute", "tool", action=name, ok=True)
607
- continue
608
-
609
- policy = d.policy_for(name, args)
610
- risk = d.risk_level(policy)
611
-
612
- proposed, governor_allows_additive = self._governor_review(
613
- ctx, name, thoughts, args, policy, risk, current_user, request_workspace,
614
- conversation_id=getattr(req, "conversation_id", None),
615
- )
616
- if proposed:
617
- continue
618
-
619
- if self._blocked_by_gates(
620
- ctx, req, name, thoughts, args, policy, risk,
621
- current_user, governor_allows_additive,
622
- ):
623
- continue
624
-
625
- self._dispatch_step(ctx, name, thoughts, args, policy, risk, current_user)
626
-
627
- ctx.state = AgentState.VERIFYING
628
-
629
- def _self_model_summary(
630
- self, ctx: AgentRunContext, current_user: str, request_workspace: Optional[str]
631
- ) -> str:
632
- """What this Brain knows about its owner, resolved once per run.
633
-
634
- The port may be a plain string or a resolver taking scope kwargs — the
635
- same two shapes ``permission_mode`` accepts, so there is one convention
636
- for "a value or a way to get one". Anything that goes wrong yields an
637
- empty summary: prompt assembly must not fail because a profile could
638
- not be read, and an empty summary is byte-identical to having no port
639
- at all.
640
- """
641
- if ctx.self_model_summary is not None:
642
- return ctx.self_model_summary
643
- source = self.deps.self_model_summary
644
- summary = ""
645
- if source is not None:
646
- try:
647
- if callable(source):
648
- try:
649
- summary = source(
650
- user_email=current_user or None,
651
- workspace_id=request_workspace,
652
- )
653
- except TypeError:
654
- # A resolver that takes no scope arguments is allowed.
655
- summary = source()
656
- else:
657
- summary = source
658
- except Exception: # noqa: BLE001 — a profile is never worth a failed run
659
- logging.debug("agent: self-model summary unavailable", exc_info=True)
660
- summary = ""
661
- ctx.self_model_summary = str(summary or "").strip()
662
- return ctx.self_model_summary
663
-
664
- def _executor_context(
665
- self, ctx: AgentRunContext, req: Any, lang_hint: str,
666
- current_user: str, request_workspace: Optional[str],
667
- profile: Optional[AgentProfile] = None,
668
- ) -> str:
669
- """Assemble one executor turn's prompt (plan, corrections, recent chat)."""
670
- d = self.deps
671
- # Only the latest corrections steer the next attempt — stale hints
672
- # from earlier retries dilute weak models (review Wave 0.3).
673
- active_corrections = ctx.corrections[-3:]
674
- corrections_hint = (
675
- "\n\nCritic corrections from previous attempt:\n"
676
- + "\n".join(f"- {c}" for c in active_corrections)
677
- ) if active_corrections else ""
678
-
679
- recent_kwargs = {
680
- "conversation_id": req.conversation_id,
681
- "user_email": current_user or None,
682
- }
683
- if request_workspace is not None:
684
- recent_kwargs["workspace_id"] = request_workspace
685
- recent_conversation = d.recent_chat_context(**recent_kwargs) or "(none)"
686
- budget = self.transcript_budget
687
- # A small model drowns in a long transcript far sooner than a large
688
- # one, so the profile may narrow the window (v9.9.7).
689
- window = min(budget.window, profile.transcript_window) if profile else budget.window
690
- bounded_transcript = compact_transcript(
691
- ctx.transcript,
692
- window=window,
693
- result_chars=budget.result_chars,
694
- )
695
- # Mid-run workspace awareness (review L5): later steps must see what
696
- # this run already produced instead of a stale workspace picture.
697
- written = files_written(ctx.transcript, d.file_create_actions)
698
- written_hint = (
699
- "\n\nFiles written by this run so far (they exist in the workspace now):\n"
700
- + "\n".join(f"- {path}" for path in written)
701
- ) if written else ""
702
- return (
703
- # v11.1.0: the executor prompt carries profile-aware file-writing
704
- # hints, because "wrote nothing at all" was the weak-model failure
705
- # mode the loop could not repair after the fact.
706
- f"{executor_prompt_for(d.executor_prompt, profile=profile, self_model_summary=self._self_model_summary(ctx, current_user, request_workspace))}\n\n"
707
- f"[LANGUAGE HINT: {lang_hint}]\n"
708
- f"Workspace root: {d.agent_root}{self._project_block(ctx)}\n\n"
709
- f"PLAN:\n{json.dumps(ctx.plan, ensure_ascii=False)}{written_hint}\n\n"
710
- f"Recent conversation:\n{recent_conversation}\n\n"
711
- f"User request: {req.message}{corrections_hint}\n\n"
712
- f"Execution transcript:\n{json.dumps(bounded_transcript, ensure_ascii=False, indent=2)}"
713
- )
714
-
715
- def _note_parse_failure(
716
- self, ctx: AgentRunContext, raw: Any, exc: ValueError, parse_failures: int,
717
- profile: Optional[AgentProfile] = None,
718
- ) -> bool:
719
- """Record one executor parse slip; True when the run should stop retrying."""
720
- profile = profile or self.profile_for(None)
721
- ctx.transcript.append({
722
- "state": AgentState.EXECUTING.value, "action": "parse_error",
723
- "raw": str(raw)[:400], "error": str(exc),
724
- })
725
- if parse_failures >= profile.parse_failure_budget:
726
- ctx.trace.parse_error("execute", error=str(exc), recovered=False)
727
- self._emit_step(ctx, "execute", "parse_error", recovered=False)
728
- return True
729
- ctx.trace.parse_error("execute", error=str(exc), recovered=True)
730
- self._emit_step(ctx, "execute", "parse_error", recovered=True)
731
- # Weak models often need one concrete reminder of the wire
732
- # format; feed it through the corrections channel and retry
733
- # instead of aborting the whole run on the first slip.
734
- hint = (
735
- 'Your last reply was not a single JSON action object. Reply with '
736
- 'EXACTLY one JSON object like {"thoughts": "...", "action": '
737
- '"tool_name", "args": {...}} and nothing else.'
738
- )
739
- if parse_failures >= profile.escalate_after:
740
- # Escalate: name the valid tools so the model stops
741
- # inventing action names or prose. The compact profile escalates
742
- # a slip earlier — a small model needs the list sooner.
743
- valid = ", ".join(sorted(self.deps.tool_governance.keys()))
744
- hint = (
745
- f"{hint} Valid action values are: {valid}, final. "
746
- 'Use {"action": "final", "message": "..."} to finish.'
747
- )
748
- if hint not in ctx.corrections:
749
- ctx.corrections.append(hint)
750
- ctx.trace.correction("execute", hint=hint)
751
- return False
752
-
753
- async def _direct_file_path(
754
- self, ctx: AgentRunContext, req: Any, current_user: str,
755
- model_id: Optional[str],
756
- ) -> bool:
757
- """Write the plan's file steps without asking the model for JSON (v9.9.7).
758
-
759
- The compact profile's escape hatch. A 1–4B local model that cannot hold
760
- the tool-call protocol can still write a file, so when JSON tool calls
761
- are exhausted the loop drops the protocol entirely: it takes the paths
762
- the *planner* already chose and asks only for file content in plain
763
- text, through the same validated
764
- :func:`~latticeai.core.file_generation.generate_file_content` pipeline
765
- the direct chat path uses.
766
-
767
- Returns True when at least one file was actually written. Honest
768
- failure modes: no planned paths, a governor that stages the write as a
769
- proposal, or a tool error all return False and leave the run to end as
770
- it would have — this never fabricates evidence.
771
- """
772
- d = self.deps
773
- planned: List[str] = []
774
- for step in ctx.plan.get("steps") or []:
775
- if not isinstance(step, dict) or step.get("action") not in d.file_create_actions:
776
- continue
777
- path = str((step.get("args") or {}).get("path") or "").strip()
778
- if path and path not in planned:
779
- planned.append(path)
780
- if not planned:
781
- inferred = infer_file_target(getattr(req, "message", "") or "")
782
- if inferred:
783
- planned = [inferred]
784
- if not planned:
785
- return False
786
-
787
- goal = str(ctx.plan.get("goal") or getattr(req, "message", "") or "")
788
-
789
- async def _generate(context: str) -> Any:
790
- return await d.generate_as(
791
- model_id,
792
- message="Write the file content.",
793
- context=context,
794
- max_tokens=self.phase_budgets.execute_tokens,
795
- temperature=0.2,
796
- )
797
-
798
- wrote = False
799
- for path in planned[:6]:
800
- try:
801
- content, meta = await generate_file_content(
802
- _generate,
803
- target_path=path,
804
- user_request=goal,
805
- bundle_files=planned if len(planned) > 1 else None,
806
- )
807
- except Exception as exc: # noqa: BLE001 — fallback must not raise
808
- logging.warning("direct file path generation failed for %s: %s", path, exc)
809
- continue
810
- ctx.trace.llm_call("execute", model=model_id)
811
- ctx.trace.repair("execute", repairs=["direct_path_fallback"])
812
- args = {"path": path, "content": content}
813
- policy = d.policy_for("write_file", args)
814
- risk = d.risk_level(policy)
815
- before = len(ctx.transcript)
816
- self._dispatch_step(ctx, "write_file", "direct path fallback", args, policy, risk, current_user)
817
- last = ctx.transcript[-1] if len(ctx.transcript) > before else {}
818
- if isinstance(last.get("result"), dict) and not last["result"].get("proposed"):
819
- wrote = True
820
- last["direct_path"] = True
821
- last["generation"] = {"repaired": bool(meta.get("repaired"))}
822
- if wrote:
823
- ctx.trace.decision("execute", decision="direct_path_fallback", files=len(planned))
824
- self._emit_step(ctx, "execute", "direct_path", files=len(planned))
825
- ctx.final_message = (
826
- "도구 호출 형식을 계속 벗어나서, 계획에 있던 파일을 직접 생성했습니다. "
827
- "내용을 확인해 주세요."
828
- )
829
- return wrote
830
-
831
- def _is_repeated_create(self, ctx: AgentRunContext, name: Any, args: dict) -> bool:
832
- """Loop guard: the same file-create action+args re-issued right after a result."""
833
- exec_steps = [s for s in ctx.transcript if s.get("state") == AgentState.EXECUTING.value]
834
- last = exec_steps[-1] if exec_steps else None
835
- return bool(
836
- name in self.deps.file_create_actions and last
837
- and last.get("action") == name
838
- and (last.get("args") or {}) == args
839
- and "result" in last
840
- )
841
-
842
- def _governor_review(
843
- self, ctx: AgentRunContext, name: str, thoughts: str, args: dict,
844
- policy: Mapping[str, Any], risk: str, current_user: str, request_workspace: Optional[str],
845
- conversation_id: Optional[str] = None,
846
- ) -> Tuple[bool, bool]:
847
- """Central change-class governance: create-new runs with minimal
848
- friction, change/delete-existing becomes a review proposal.
849
-
850
- Returns ``(proposed, governor_allows_additive)``: ``proposed`` means the
851
- step was staged as a proposal (skip execution); ``allows_additive`` lets
852
- an additive create pass the classic approval gate.
853
-
854
- Under a mode that does not stage proposals (``trusted`` / ``bypass``)
855
- the decision is made *before* the governor is consulted, because
856
- ``review`` persists a proposal as a side effect — reviewing first and
857
- discarding the verdict afterwards would apply the change *and* leave an
858
- orphan proposal pending in the Review Center.
859
- """
860
- d = self.deps
861
- if d.change_governor is None:
862
- return False, False
863
-
864
- mode = self.resolve_permission_mode(
865
- ctx, user_email=current_user, workspace_id=request_workspace,
866
- )
867
- if not should_stage_proposal(mode, proposal_required=True):
868
- if name not in self._governed_tools():
869
- return False, False
870
- if policy.get("destructive") or policy.get("risk") == "destructive":
871
- # Let the destructive gate downstream own the block + transcript.
872
- return False, False
873
- d.audit(
874
- "agent_change_auto_applied",
875
- user_email=current_user,
876
- workspace_id=request_workspace,
877
- action=name,
878
- path=str(args.get("path") or "") or None,
879
- permission_mode=mode.value,
880
- note="permission mode auto-applies mutation with audit",
881
- )
882
- return False, True
883
-
884
- verdict = d.change_governor.review(
885
- name, args, policy=dict(policy),
886
- user_email=current_user, workspace_id=request_workspace,
887
- conversation_id=conversation_id,
888
- )
889
- if verdict is not None and verdict.get("decision") == "proposed":
890
- proposal = verdict.get("proposal") or {}
891
- ctx.trace.tool("execute", name=name, outcome="proposed", risk=risk)
892
- self._emit_step(ctx, "execute", "proposed", action=name)
893
- ctx.transcript.append({
894
- "state": AgentState.EXECUTING.value, "action": name,
895
- "thoughts": thoughts, "args": {k: v for k, v in args.items() if k != "content"},
896
- "risk": risk, "governance": dict(policy),
897
- "result": {
898
- "proposed": True,
899
- "proposal_id": proposal.get("id"),
900
- "note": "기존 내용을 바꾸는 작업이라 변경 제안으로 저장했습니다. 검토함에서 승인하면 적용됩니다.",
901
- },
902
- })
903
- d.audit(
904
- "agent_change_proposed", user_email=current_user,
905
- action=name, proposal_id=proposal.get("id"),
906
- change_class=(verdict.get("classification") or {}).get("change_class"),
907
- )
908
- return True, False
909
- return False, (verdict is not None and verdict.get("decision") == "allow_additive")
910
-
911
- def _blocked_by_gates(
912
- self, ctx: AgentRunContext, req: Any, name: str, thoughts: str, args: dict,
913
- policy: Mapping[str, Any], risk: str, current_user: str, governor_allows_additive: bool,
914
- ) -> bool:
915
- """Destructive / circuit-breaker / fail-closed-overwrite / approval gates.
916
-
917
- Returns True when the step was blocked. The active permission mode can
918
- widen what runs without an extra approval prompt, but never widens a
919
- circuit breaker, the destructive gate, or the overwrite check.
920
- """
921
- d = self.deps
922
- mode = self.resolve_permission_mode(
923
- ctx,
924
- user_email=current_user,
925
- workspace_id=getattr(req, "workspace_id", None),
926
- )
927
- # Hard denials first — mode-invariant. A circuit breaker (root/home
928
- # paths, `rm -rf /` style commands) and a destructive policy are both
929
- # audited as ``blocked`` with the reason that actually fired, rather
930
- # than being flattened into the approval path.
931
- breaker = is_circuit_breaker(name, policy, args)
932
- hard_deny = breaker or (
933
- "destructive policy"
934
- if policy["risk"] == "destructive" or policy.get("destructive")
935
- else None
936
- )
937
- if hard_deny:
938
- error = (
939
- f"BLOCKED: destructive action '{name}' not permitted in agent mode."
940
- if hard_deny == "destructive policy"
941
- else f"BLOCKED: {hard_deny}"
942
- )
943
- ctx.trace.tool("execute", name=name, outcome="blocked_destructive", risk=risk)
944
- self._emit_step(ctx, "execute", "blocked", action=name, reason="destructive")
945
- ctx.transcript.append({
946
- "state": AgentState.EXECUTING.value, "action": name,
947
- "thoughts": thoughts, "args": args, "risk": risk,
948
- "governance": dict(policy),
949
- "permission_mode": mode.value,
950
- "error": error,
951
- })
952
- d.audit(
953
- "agent_blocked", user_email=current_user, source=getattr(req, "source", None) or "agent",
954
- action=name, reason="destructive", governance=dict(policy),
955
- )
956
- return True
957
-
958
- # Fail-closed overwrite guard — mode-invariant, like the two above.
959
- # A call that rewrites existing content but cannot be staged as a
960
- # reviewable proposal (binary document creators, home-sandbox writes)
961
- # has no safe apply path in ANY mode: trusted/bypass skip the approval
962
- # *prompt*, they never remove the existence check. Without this the
963
- # loop silently overwrote files that the HTTP surface refuses with 409
964
- # (``ToolDispatchService.enforce_policy``).
965
- overwrite = classify_tool_call(
966
- name, args, policy=dict(policy),
967
- path_exists=lambda candidate: self._governed_path_exists(name, candidate),
968
- )
969
- if overwrite.get("fail_closed"):
970
- target = str(args.get("path") or args.get("filename") or "")
971
- error = (
972
- f"NEEDS_REVIEW: '{name}' 은(는) 이미 있는 파일 '{target}' 을(를) 덮어씁니다. "
973
- "이 도구의 변경은 검토 가능한 제안으로 만들 수 없어 실행하지 않았습니다. "
974
- "새 파일 이름으로 만들거나 write_file/edit_file 로 수정하세요."
975
- )
976
- ctx.trace.tool("execute", name=name, outcome="blocked_overwrite", risk=risk)
977
- self._emit_step(ctx, "execute", "blocked", action=name, reason="overwrite")
978
- ctx.transcript.append({
979
- "state": AgentState.EXECUTING.value, "action": name,
980
- "thoughts": thoughts,
981
- # Same shape as a staged proposal: the payload is never worth
982
- # replaying into the transcript, only the decision is.
983
- "args": {k: v for k, v in args.items() if k != "content"},
984
- "risk": risk,
985
- "governance": dict(policy),
986
- "permission_mode": mode.value,
987
- "change_class": overwrite.get("change_class"),
988
- "error": error,
989
- })
990
- d.audit(
991
- "agent_blocked", user_email=current_user,
992
- source=getattr(req, "source", None) or "agent",
993
- action=name, reason="overwrite_fail_closed",
994
- path=target or None,
995
- change_class=overwrite.get("change_class"),
996
- permission_mode=mode.value,
997
- governance=dict(policy),
998
- )
999
- return True
1000
-
1001
- reason = block_reason_for_tool(
1002
- mode, name, policy, args,
1003
- approved_by_human=bool(ctx.approved_by_human),
1004
- governor_allows_additive=governor_allows_additive,
1005
- )
1006
- if reason is None:
1007
- return False
1008
-
1009
- d.audit(
1010
- "agent_exec", user_email=current_user, source=getattr(req, "source", None) or "agent",
1011
- state=AgentState.EXECUTING.value, action=name, risk=risk,
1012
- shell=policy["shell"], network=policy["network"],
1013
- destructive=policy["destructive"], sandbox=policy["sandbox"],
1014
- rollback=policy["rollback"],
1015
- permission_mode=mode.value,
1016
- args={k: v for k, v in args.items() if k != "content"},
1017
- )
1018
- ctx.trace.tool("execute", name=name, outcome="blocked_approval", risk=risk)
1019
- self._emit_step(ctx, "execute", "blocked", action=name, reason="approval")
1020
- ctx.transcript.append({
1021
- "state": AgentState.EXECUTING.value, "action": name,
1022
- "thoughts": thoughts, "args": args, "risk": risk,
1023
- "governance": dict(policy),
1024
- "permission_mode": mode.value,
1025
- "error": reason,
1026
- })
1027
- return True
1028
-
1029
- def _dispatch_step(
1030
- self, ctx: AgentRunContext, name: str, thoughts: str, args: dict,
1031
- policy: Mapping[str, Any], risk: str, current_user: str,
1032
- ) -> None:
1033
- """Role check + shared tool lifecycle, recorded on the transcript either way."""
1034
- d = self.deps
1035
- sanitize_meta: Optional[Dict[str, Any]] = None
1036
- if name == "write_file" and isinstance(args.get("content"), str):
1037
- # ArtifactWritePipeline: the executor's args.content is untrusted
1038
- # model output. The same extract→validate→repair guarantee as the
1039
- # direct chat path applies here, so a weak model driving the JSON
1040
- # loop can never persist fenced/chatty/truncated payloads.
1041
- cleaned, meta = sanitize_write_content(
1042
- str(args.get("path") or ""), args["content"],
1043
- user_request=str(ctx.plan.get("goal") or thoughts or name),
1044
- )
1045
- if meta.get("sanitized"):
1046
- args = dict(args)
1047
- args["content"] = cleaned
1048
- sanitize_meta = meta
1049
- ctx.trace.repair(
1050
- "execute",
1051
- repairs=[
1052
- "artifact_repair" if meta.get("repaired") else "artifact_sanitize"
1053
- ],
1054
- )
1055
- step_index = 1 + sum(
1056
- 1 for s in ctx.transcript
1057
- if s.get("state") == AgentState.EXECUTING.value
1058
- and s.get("action") not in (None, "final", "parse_error")
1059
- )
1060
- if (
1061
- name in d.file_create_actions
1062
- and d.snapshot_file is not None
1063
- and args.get("path")
1064
- ):
1065
- # Pre-write snapshot (review L7): the first capture per path is
1066
- # the true pre-run state — later writes to the same path must
1067
- # not overwrite it. Best-effort: a snapshot failure never
1068
- # blocks the write, it only narrows rollback options.
1069
- path_str = str(args["path"])
1070
- if not any(entry.get("path") == path_str for entry in ctx.rollback_log):
1071
- try:
1072
- pre = d.snapshot_file(path_str)
1073
- ctx.rollback_log.append({"path": path_str, **(pre or {})})
1074
- except Exception as exc: # noqa: BLE001
1075
- logging.warning("pre-write snapshot failed for %s: %s", path_str, exc)
1076
- try:
1077
- d.check_role(name, current_user)
1078
- # Shared tool lifecycle: pre_tool (may block) → execute → post_tool.
1079
- result = dispatch_tool(
1080
- d.hooks, name, args,
1081
- lambda: d.execute_tool(name, args),
1082
- user_email=current_user, source="agent",
1083
- )
1084
- ctx.trace.tool("execute", name=name, outcome="ok", risk=risk)
1085
- ctx.transcript.append({
1086
- "state": AgentState.EXECUTING.value, "action": name,
1087
- "thoughts": thoughts, "args": args,
1088
- "risk": risk, "governance": dict(policy), "result": result,
1089
- **({"content_sanitize": sanitize_meta} if sanitize_meta else {}),
1090
- })
1091
- self._emit_step(
1092
- ctx, "execute", "tool", action=name, ok=True, step=step_index,
1093
- path=str(args.get("path")) if args.get("path") else None,
1094
- )
1095
- except (ToolError, KeyError, TypeError, PermissionError) as exc:
1096
- ctx.trace.tool("execute", name=name, outcome="error", risk=risk)
1097
- ctx.transcript.append({
1098
- "state": AgentState.EXECUTING.value, "action": name,
1099
- "thoughts": thoughts, "args": args,
1100
- "risk": risk, "governance": dict(policy), "error": str(exc),
1101
- })
1102
- self._emit_step(
1103
- ctx, "execute", "tool", action=name, ok=False, step=step_index,
1104
- path=str(args.get("path")) if args.get("path") else None,
1105
- )
1106
-
1107
- # ── VERIFY ───────────────────────────────────────────────────────
1108
- def _has_execution_evidence(self, ctx: AgentRunContext) -> bool:
1109
- """Deterministic evidence check: at least one executing step actually
1110
- produced a result (tool ran, or a governed change was staged as a
1111
- proposal). ``final``/parse-error/blocked steps carry no result and do
1112
- not count — a critic PASS over an evidence-free transcript must not
1113
- become DONE."""
1114
- for step in ctx.transcript:
1115
- if step.get("state") != AgentState.EXECUTING.value:
1116
- continue
1117
- if step.get("action") in (None, "final", "parse_error"):
1118
- continue
1119
- if isinstance(step.get("result"), dict):
1120
- return True
1121
- return False
1122
-
1123
- async def verify(
1124
- self, ctx: AgentRunContext, req: Any, lang_hint: str, current_user: str,
1125
- max_retry: int = 3, model_id: Optional[str] = None,
1126
- ) -> None:
1127
- """VERIFYING: Critic role evaluates transcript → DONE / EXECUTING (retry) / ROLLBACK / NEEDS_REVIEW / FAILED.
1128
-
1129
- Fail-closed: a critic whose output cannot be parsed (after one strict
1130
- repair retry) never fabricates a PASS — the run terminates as
1131
- NEEDS_REVIEW so the user is told to check the result themselves.
1132
- """
1133
- d = self.deps
1134
- # The critic must see every step (evidence completeness), but not
1135
- # every byte of tool output — long bodies are capped per string so
1136
- # verification stays affordable on long runs (review Wave 0.3).
1137
- verify_transcript = _truncate_strings(
1138
- ctx.transcript, self.transcript_budget.verify_chars
1139
- )
1140
- # Deterministic artifact facts (review L4): the critic sees the
1141
- # sanitize/repair honesty flags per written file, not just prose.
1142
- checklist = artifact_checklist(ctx.transcript, d.file_create_actions)
1143
- checklist_hint = (
1144
- f"\n\n{format_artifact_checklist(checklist)}" if checklist else ""
1145
- )
1146
- # Requirement coverage (review 루프 §2): the critic previously judged
1147
- # "did this fulfill the request?" from prose alone. It now also sees
1148
- # which requested files actually exist and which requirements the user
1149
- # spelled out.
1150
- coverage = requirement_coverage(
1151
- req.message, ctx.transcript, d.file_create_actions
1152
- )
1153
- context = (
1154
- f"{d.critic_prompt}\n\n"
1155
- f"[LANGUAGE HINT: {lang_hint}]\n\n"
1156
- f"Original request: {req.message}\n"
1157
- f"Plan goal: {ctx.plan.get('goal', req.message)}{checklist_hint}"
1158
- f"{format_requirement_coverage(coverage)}\n\n"
1159
- f"Full transcript:\n{json.dumps(verify_transcript, ensure_ascii=False, indent=2)}"
1160
- )
1161
- raw = await d.generate_as(
1162
- model_id,
1163
- message="Review the execution transcript and return your verdict JSON.",
1164
- context=context, max_tokens=self.phase_budgets.verify_tokens, temperature=0.1,
1165
- )
1166
- ctx.trace.llm_call("verify", model=model_id)
1167
- verdict: Optional[Dict[str, Any]] = None
1168
- try:
1169
- verdict, verdict_repairs = extract_action_details(str(raw))
1170
- ctx.trace.repair("verify", repairs=verdict_repairs)
1171
- except ValueError as exc:
1172
- # One strict repair retry — re-ask the critic for the exact wire
1173
- # format instead of fabricating a verdict.
1174
- ctx.trace.parse_error("verify", error=str(exc), recovered=True)
1175
- strict_context = (
1176
- f"{context}\n\n"
1177
- "Your previous verdict was not parseable JSON. Reply with EXACTLY one "
1178
- 'JSON object like {"action": "verdict", "verdict": "PASS", '
1179
- '"next_state": "DONE", "reason": "...", "corrections": []} '
1180
- "and nothing else. verdict must be PASS or FAIL; next_state must be "
1181
- "one of DONE, EXECUTING, ROLLBACK, FAILED."
1182
- )
1183
- raw = await d.generate_as(
1184
- model_id,
1185
- message="Return your verdict as one strict JSON object.",
1186
- context=strict_context, max_tokens=self.phase_budgets.verify_tokens,
1187
- temperature=0.0,
1188
- )
1189
- ctx.trace.llm_call("verify", model=model_id)
1190
- try:
1191
- verdict, verdict_repairs = extract_action_details(str(raw))
1192
- ctx.trace.repair("verify", repairs=verdict_repairs)
1193
- except ValueError as retry_exc:
1194
- ctx.trace.parse_error("verify", error=str(retry_exc), recovered=False)
1195
- verdict = None
1196
-
1197
- has_evidence = self._has_execution_evidence(ctx)
1198
-
1199
- if verdict is None:
1200
- # Verifier unavailable — fail closed, never DONE.
1201
- ctx.transcript.append({
1202
- "state": AgentState.VERIFYING.value,
1203
- "verdict": "UNAVAILABLE",
1204
- "reason": "critic output unparseable after strict retry",
1205
- "verifier_available": False,
1206
- "verdict_valid": False,
1207
- "evidence": has_evidence,
1208
- })
1209
- ctx.trace.decision(
1210
- "verify", decision="verification_unavailable",
1211
- verifier_available=False, verdict_valid=False, evidence=has_evidence,
1212
- )
1213
- self._emit_step(ctx, "verify", "verdict", verdict="UNAVAILABLE")
1214
- ctx.final_message = (
1215
- "검증을 완료하지 못했습니다 — 검증 모델의 응답을 해석할 수 없었습니다. "
1216
- "실행 결과를 직접 확인해 주시고, 필요하면 다시 시도해 주세요."
1217
- )
1218
- ctx.state = AgentState.NEEDS_REVIEW
1219
- return
1220
-
1221
- ctx.corrections = verdict.get("corrections", [])
1222
- # Normalize legacy verdict next_state strings to current AgentState names
1223
- raw_next = verdict.get("next_state", "")
1224
- next_s = {"COMPLETE": "DONE", "RETRY": "EXECUTING"}.get(raw_next, raw_next)
1225
-
1226
- ctx.transcript.append({
1227
- "state": AgentState.VERIFYING.value,
1228
- "verdict": verdict.get("verdict", ""),
1229
- "reason": verdict.get("reason", ""),
1230
- "corrections": ctx.corrections,
1231
- "confidence": verdict.get("confidence", 0.9),
1232
- "next_state": next_s,
1233
- "verifier_available": True,
1234
- "verdict_valid": True,
1235
- "evidence": has_evidence,
1236
- })
1237
-
1238
- ctx.trace.decision(
1239
- "verify", decision=str(verdict.get("verdict", "")), next_state=next_s,
1240
- verifier_available=True, verdict_valid=True, evidence=has_evidence,
1241
- )
1242
- self._emit_step(
1243
- ctx, "verify", "verdict",
1244
- verdict=str(verdict.get("verdict", "")), next_state=next_s,
1245
- )
1246
- if verdict.get("verdict") == "PASS":
1247
- # DONE requires both: a validly parsed PASS verdict AND
1248
- # deterministic execution evidence in the transcript. A PASS over
1249
- # an evidence-free run is not a completion.
1250
- if not has_evidence:
1251
- ctx.trace.decision("verify", decision="needs_review_no_evidence")
1252
- ctx.final_message = (
1253
- "검증자는 통과를 보고했지만 실제 실행 근거(도구 실행 기록)가 없어 "
1254
- "완료로 처리하지 않았습니다. 결과를 직접 확인해 주세요."
1255
- )
1256
- ctx.state = AgentState.NEEDS_REVIEW
1257
- return
1258
- if not coverage["complete"]:
1259
- # A PASS that leaves a *requested file* unwritten is not a
1260
- # completion — this is a fact, not a judgement, so it is
1261
- # enforced rather than merely reported to the critic.
1262
- missing = ", ".join(coverage["missing_files"])
1263
- ctx.trace.decision(
1264
- "verify", decision="needs_review_missing_files",
1265
- missing=len(coverage["missing_files"]),
1266
- )
1267
- ctx.transcript.append({
1268
- "state": AgentState.VERIFYING.value,
1269
- "requirement_coverage": coverage,
1270
- })
1271
- ctx.final_message = (
1272
- f"요청한 파일 중 일부가 만들어지지 않아 완료로 처리하지 않았습니다: {missing}"
1273
- )
1274
- ctx.state = AgentState.NEEDS_REVIEW
1275
- return
1276
- if not ctx.final_message:
1277
- ctx.final_message = verdict.get("reason", "작업이 완료되었습니다.")
1278
- ctx.state = AgentState.DONE
1279
- elif next_s == "ROLLBACK":
1280
- ctx.state = AgentState.ROLLBACK
1281
- elif next_s == "EXECUTING":
1282
- if ctx.retry_count >= max_retry:
1283
- ctx.final_message = "처리 중 문제가 발생했습니다. 다시 시도해 주세요."
1284
- ctx.state = AgentState.FAILED
1285
- else:
1286
- ctx.retry_count += 1
1287
- ctx.trace.retry("verify", attempt=ctx.retry_count)
1288
- ctx.transcript.append({
1289
- "state": AgentState.EXECUTING.value,
1290
- "retry_attempt": ctx.retry_count,
1291
- "corrections": ctx.corrections,
1292
- })
1293
- ctx.state = AgentState.EXECUTING
1294
- elif next_s == "DONE":
1295
- # Contradictory verdict: the critic asked for DONE without a PASS.
1296
- # The loose "or next_state == DONE" success path is gone — this is
1297
- # a non-success that the user must review.
1298
- ctx.trace.decision("verify", decision="needs_review_inconsistent_verdict")
1299
- ctx.final_message = (
1300
- "검증 결과가 일관되지 않아 완료로 처리하지 않았습니다. "
1301
- "실행 결과를 직접 확인해 주세요."
1302
- )
1303
- ctx.state = AgentState.NEEDS_REVIEW
1304
- else:
1305
- ctx.final_message = verdict.get("reason", "검증자가 인식되지 않은 다음 상태를 반환했습니다.")
1306
- ctx.state = AgentState.FAILED
1307
-
1308
- # ── ROLLBACK ─────────────────────────────────────────────────────
1309
- def _snapshot_for(self, ctx: AgentRunContext, path: str) -> Optional[Dict[str, Any]]:
1310
- for entry in ctx.rollback_log:
1311
- if entry.get("path") == path:
1312
- return entry
1313
- return None
1314
-
1315
- def _rollback_one(self, ctx: AgentRunContext, path: str, gov: Dict[str, Any]) -> Dict[str, Any]:
1316
- """Recover one path: git when governed and available, else the
1317
- pre-write snapshot, else an honest ``mode="none"`` (review L7)."""
1318
- d = self.deps
1319
- if gov.get("rollback") == "git" and d.rollback_file is not None:
1320
- try:
1321
- result = dict(d.rollback_file(str(path)))
1322
- except Exception as exc: # noqa: BLE001
1323
- result = {"path": path, "ok": False, "error": str(exc)}
1324
- if result.get("ok"):
1325
- result["mode"] = "git"
1326
- return result
1327
- snapshot = self._snapshot_for(ctx, str(path))
1328
- if snapshot is not None and d.restore_snapshot is not None and not snapshot.get("too_large"):
1329
- content = snapshot.get("content") if snapshot.get("existed") else None
1330
- try:
1331
- restored = dict(d.restore_snapshot(str(path), content))
1332
- except Exception as exc: # noqa: BLE001
1333
- restored = {"path": path, "ok": False, "error": str(exc)}
1334
- restored.setdefault("path", path)
1335
- restored["mode"] = "snapshot"
1336
- return restored
1337
- return {
1338
- "path": path, "ok": False, "mode": "none",
1339
- "error": "no rollback available (git not applicable, no usable snapshot)",
1340
- }
1341
-
1342
- def rollback(self, ctx: AgentRunContext, current_user: str) -> None:
1343
- """ROLLBACK: recover written files (git → snapshot → none), then FAILED."""
1344
- d = self.deps
1345
- rolled: List[dict] = []
1346
- seen_paths: set = set()
1347
- for step in ctx.transcript:
1348
- if step.get("state") != AgentState.EXECUTING.value:
1349
- continue
1350
- if not isinstance(step.get("result"), dict):
1351
- continue
1352
- gov = step.get("governance", {}) or {}
1353
- path = step["result"].get("path") or (step.get("args") or {}).get("path", "")
1354
- if not path or str(path) in seen_paths:
1355
- continue
1356
- if gov.get("rollback") != "git" and step.get("action") not in d.file_create_actions:
1357
- continue
1358
- seen_paths.add(str(path))
1359
- rolled.append(self._rollback_one(ctx, str(path), gov))
1360
-
1361
- ctx.transcript.append({"state": AgentState.ROLLBACK.value, "rolled_back": rolled})
1362
- ctx.trace.decision(
1363
- "rollback", decision="rolled_back",
1364
- attempted=len(rolled), recovered=sum(1 for r in rolled if r.get("ok")),
1365
- )
1366
- recovered = [f"{r['path']} ({r.get('mode')})" for r in rolled if r.get("ok")]
1367
- ctx.final_message = (
1368
- f"실행 실패로 롤백했습니다. 복구 파일: {recovered}"
1369
- if recovered
1370
- else "롤백을 시도했으나 복구할 파일이 없거나 git/스냅샷 복구 수단이 없습니다."
1371
- )
1372
- d.audit("agent_rollback", user_email=current_user, rolled_back=rolled)
1373
- self._emit_step(ctx, "rollback", "rolled_back", recovered=len(recovered))
1374
- # Rollback is a recovery from a failed verification — terminal state is FAILED
1375
- ctx.state = AgentState.FAILED
1376
-
1377
- # ── MEMORY ───────────────────────────────────────────────────────
1378
- async def memory_update(self, ctx: AgentRunContext, req: Any, current_user: str) -> None:
1379
- """Background: Memory Updater role extracts learnings from a terminal run.
1380
-
1381
- Terminal-state learning policy (review §4.2 L6): DONE runs record what
1382
- worked; FAILED / NEEDS_REVIEW runs record what went wrong — failure is
1383
- exactly the experience worth remembering. The run status stored with
1384
- the experience is the *actual* terminal state, never a blanket "ok".
1385
- """
1386
- d = self.deps
1387
- terminal = ctx.state.value if ctx.state in AGENT_TERMINAL_STATES else "UNKNOWN"
1388
- outcome_hint = (
1389
- "The task completed successfully."
1390
- if ctx.state == AgentState.DONE
1391
- else (
1392
- f"The task ended as {terminal} — extract what went wrong and "
1393
- "what to do differently next time, not a success story."
1394
- )
1395
- )
1396
- context = (
1397
- f"{d.memory_updater_prompt}\n\n"
1398
- f"Task: {req.message}\n"
1399
- f"Terminal status: {terminal}. {outcome_hint}\n\n"
1400
- f"Last 5 transcript steps:\n{json.dumps(ctx.transcript[-5:], ensure_ascii=False)}"
1401
- )
1402
- try:
1403
- raw = await d.generate(
1404
- message="Extract learnings from this completed task.",
1405
- context=context, max_tokens=self.phase_budgets.memory_tokens, temperature=0.1,
1406
- )
1407
- mem = extract_action(str(raw))
1408
- kept_learnings = filter_learnings(mem.get("learnings") or [])
1409
- if mem.get("save_to_knowledge") and kept_learnings:
1410
- learnings = "\n".join(kept_learnings)
1411
- status_label = {
1412
- AgentState.DONE: "ok",
1413
- AgentState.NEEDS_REVIEW: "needs_review",
1414
- AgentState.FAILED: "failed",
1415
- }.get(ctx.state, "unknown")
1416
- if d.brain_memory is not None:
1417
- # This runtime is LLM-driven — its learnings are real
1418
- # experiences and enter the brain with provenance.
1419
- d.brain_memory.record_experience(
1420
- f"Agent: {req.message[:60]}",
1421
- learnings,
1422
- run={
1423
- "mode": "llm",
1424
- "status": status_label,
1425
- "agent_id": "agent:executor",
1426
- "steps": len(ctx.transcript),
1427
- },
1428
- user_email=current_user or None,
1429
- )
1430
- else:
1431
- d.knowledge_save(
1432
- learnings,
1433
- folder="30_Projects",
1434
- title=f"Agent: {req.message[:60]}",
1435
- )
1436
- except Exception as exc:
1437
- # Never crash a completed run, but never swallow silently either.
1438
- logging.warning("agent memory update failed: %s", exc)
1439
-
1440
- # ── DRIVE LOOP ───────────────────────────────────────────────────
1441
- async def run_to_completion(
1442
- self, ctx: AgentRunContext, req: Any, lang_hint: str,
1443
- current_user: str, max_steps: int, max_retry: int,
1444
- ) -> None:
1445
- """Run EXECUTING → VERIFYING → ROLLBACK loop until a terminal state."""
1446
- while ctx.state not in AGENT_TERMINAL_STATES:
1447
- ctx.state_history.append(ctx.state.value)
1448
- if len(ctx.state_history) > 200:
1449
- ctx.final_message = "에이전트 상태 머신이 최대 반복(200)에 도달해 중단했습니다."
1450
- ctx.state = AgentState.FAILED
1451
- break
1452
-
1453
- if ctx.state == AgentState.EXECUTING:
1454
- await self.execute(ctx, req, lang_hint, current_user, max_steps,
1455
- model_id=ctx.executing_model)
1456
- elif ctx.state == AgentState.VERIFYING:
1457
- await self.verify(ctx, req, lang_hint, current_user, max_retry,
1458
- model_id=ctx.reviewing_model)
1459
- elif ctx.state == AgentState.ROLLBACK:
1460
- self.rollback(ctx, current_user)
1461
- else:
1462
- ctx.state = AgentState.FAILED
1463
-
1464
- ctx.state_history.append(ctx.state.value)
1465
- self._emit_step(ctx, "terminal", "state", state=ctx.state.value)